diff --git a/.DS_Store b/.DS_Store new file mode 100644 index 00000000..5b63d8c5 Binary files /dev/null and b/.DS_Store differ diff --git a/.clang-format b/.clang-format new file mode 100644 index 00000000..61a52ea9 --- /dev/null +++ b/.clang-format @@ -0,0 +1,97 @@ +AccessModifierOffset: -2 +AlignAfterOpenBracket: DontAlign +AlignConsecutiveAssignments: false +AlignConsecutiveDeclarations: false +AlignEscapedNewlines: Left +AlignOperands: true +AlignTrailingComments: false +AllowAllParametersOfDeclarationOnNextLine: false +AllowShortBlocksOnASingleLine: true +AllowShortCaseLabelsOnASingleLine: false +AllowShortFunctionsOnASingleLine: All +AllowShortIfStatementsOnASingleLine: true +AllowShortLoopsOnASingleLine: true +AlwaysBreakAfterDefinitionReturnType: None +AlwaysBreakAfterReturnType: None +AlwaysBreakBeforeMultilineStrings: true +AlwaysBreakTemplateDeclarations: false +BinPackArguments: false +BinPackParameters: false +BraceWrapping: + AfterClass: true + AfterControlStatement: false + AfterEnum: false + AfterFunction: true + AfterNamespace: false + AfterObjCDeclaration: false + AfterStruct: true + AfterUnion: false + BeforeCatch: false + BeforeElse: false + IndentBraces: false + SplitEmptyFunction: false + SplitEmptyNamespace: true + SplitEmptyRecord: true +BreakAfterJavaFieldAnnotations: true +BreakBeforeBinaryOperators: NonAssignment +BreakBeforeBraces: Custom +BreakBeforeInheritanceComma: true +BreakBeforeTernaryOperators: true +BreakConstructorInitializers: BeforeColon +BreakConstructorInitializersBeforeComma: false +BreakStringLiterals: true +ColumnLimit: 120 +CommentPragmas: '^ IWYU pragma:' +CompactNamespaces: false +ConstructorInitializerAllOnOneLineOrOnePerLine: false +ConstructorInitializerIndentWidth: 2 +ContinuationIndentWidth: 2 +Cpp11BracedListStyle: false +DerivePointerAlignment: false +DisableFormat: false +ExperimentalAutoDetectBinPacking: true +FixNamespaceComments: true +ForEachMacros: +- foreach +- Q_FOREACH +- BOOST_FOREACH +IncludeCategories: +- Priority: 2 + Regex: ^"(llvm|llvm-c|clang|clang-c)/ +- Priority: 3 + Regex: ^(<|"(gtest|gmock|isl|json)/) +- Priority: 1 + Regex: .* +IncludeIsMainRegex: (Test)?$ +IndentCaseLabels: false +IndentWidth: 2 +IndentWrappedFunctionNames: true +JavaScriptQuotes: Leave +JavaScriptWrapImports: true +KeepEmptyLinesAtTheStartOfBlocks: true +Language: Cpp +MacroBlockBegin: '' +MacroBlockEnd: '' +MaxEmptyLinesToKeep: 2 +NamespaceIndentation: Inner +ObjCBlockIndentWidth: 7 +ObjCSpaceAfterProperty: true +ObjCSpaceBeforeProtocolList: false +PointerAlignment: Right +ReflowComments: true +SortIncludes: true +SortUsingDeclarations: false +SpaceAfterCStyleCast: false +SpaceAfterTemplateKeyword: false +SpaceBeforeAssignmentOperators: true +SpaceBeforeParens: ControlStatements +SpaceInEmptyParentheses: false +SpacesBeforeTrailingComments: 0 +SpacesInAngles: false +SpacesInCStyleCastParentheses: false +SpacesInContainerLiterals: true +SpacesInParentheses: false +SpacesInSquareBrackets: false +Standard: c++20 +TabWidth: 8 +UseTab: Never diff --git a/.clang-tidy b/.clang-tidy new file mode 100644 index 00000000..dadc3ede --- /dev/null +++ b/.clang-tidy @@ -0,0 +1,32 @@ +--- +Checks: "*, + -abseil-*, + -altera-*, + -android-*, + -fuchsia-*, + -google-*, + -llvm*, + -modernize-use-trailing-return-type, + -zircon-*, + -readability-else-after-return, + -readability-static-accessed-through-instance, + -readability-avoid-const-params-in-decls, + -cppcoreguidelines-non-private-member-variables-in-classes, + -misc-non-private-member-variables-in-classes, + -misc-no-recursion, + -misc-use-anonymous-namespace, + -misc-use-internal-linkage +" +WarningsAsErrors: '' +HeaderFilterRegex: '' +FormatStyle: none + +CheckOptions: + - key: readability-identifier-length.IgnoredVariableNames + value: 'x|y|z' + - key: readability-identifier-length.IgnoredParameterNames + value: 'x|y|z' + - key: readability-uppercase-literal-suffix.NewSuffixes + value: '' + - key: readability-uppercase-literal-suffix.IgnoreSuffixes + value: 'f' \ No newline at end of file diff --git a/.clangd b/.clangd new file mode 100644 index 00000000..fdffff8a --- /dev/null +++ b/.clangd @@ -0,0 +1,2 @@ +CompileFlags: + CompilationDatabase: build/relwithdebinfo diff --git a/.gitattributes b/.gitattributes new file mode 100644 index 00000000..be9586bc --- /dev/null +++ b/.gitattributes @@ -0,0 +1,2 @@ +tests/data/raw/*.NEF filter=lfs diff=lfs merge=lfs -text +tests/e2e/golden/golden_raw_pipeline.jpg filter=lfs diff=lfs merge=lfs -text diff --git a/.github/.DS_Store b/.github/.DS_Store new file mode 100644 index 00000000..3dd1d0d7 Binary files /dev/null and b/.github/.DS_Store differ diff --git a/.github/workflows/cmake.yml b/.github/workflows/cmake.yml index 6318b97e..e8a25c0b 100644 --- a/.github/workflows/cmake.yml +++ b/.github/workflows/cmake.yml @@ -14,6 +14,84 @@ env: FORCE_JAVASCRIPT_ACTIONS_TO_NODE24: true jobs: + static-analysis: + name: Static Analysis (Format & Tidy) + runs-on: ubuntu-latest + env: + BUILD_TYPE: Release + VCPKG_BINARY_SOURCES: "clear;nuget,https://nuget.pkg.github.com/${{ github.repository_owner }}/index.json,readwrite" + VCPKG_DEFAULT_TRIPLET: x64-linux + VCPKG_TARGET_TRIPLET: x64-linux + CMAKE_GENERATOR: "Ninja" + + steps: + - uses: actions/checkout@v4 + with: + submodules: recursive + + - name: Setup uv + uses: astral-sh/setup-uv@v5 + with: + python-version: "3.14.3" + enable-cache: true + + - name: Install dependencies + run: | + sudo sed -i 's/^Types: deb$/Types: deb deb-src/' /etc/apt/sources.list.d/ubuntu.sources + sudo apt-get update + sudo apt-get build-dep -y qtbase5-dev + sudo apt-get install -y \ + ninja-build \ + mold \ + autoconf \ + autoconf-archive \ + automake \ + libtool \ + libltdl-dev \ + mono-complete \ + clang-tidy \ + clang-format + + - name: Setup vcpkg (lukka/run-vcpkg) + uses: lukka/run-vcpkg@v11 + with: + vcpkgGitCommitId: b5d1a94fb7f88fd835e360fd23a45a09ceedbf48 + vcpkgJsonGlob: 'vcpkg.json' + + - name: Find NuGet + shell: bash + run: | + nuget_path=$(vcpkg fetch nuget | tail -n 1) + echo "NUGET_EXE=$nuget_path" >> "$GITHUB_ENV" + + - name: Setup NuGet Credentials + shell: bash + run: | + mono "$NUGET_EXE" sources remove -Name GitHubPackages -Source "https://nuget.pkg.github.com/${{ github.repository_owner }}/index.json" || true + mono "$NUGET_EXE" sources add \ + -Source "https://nuget.pkg.github.com/${{ github.repository_owner }}/index.json" \ + -StorePasswordInClearText \ + -Name GitHubPackages \ + -UserName "${{ github.repository_owner }}" \ + -Password "${{ secrets.GITHUB_TOKEN }}" + mono "$NUGET_EXE" setapikey "${{ secrets.GITHUB_TOKEN }}" \ + -Source "https://nuget.pkg.github.com/${{ github.repository_owner }}/index.json" + + - name: Setup CMake + uses: lukka/get-cmake@latest + + - name: Configure CMake + run: | + cmake -S . -B build/release -G Ninja -DCMAKE_BUILD_TYPE=Release -DVCPKG_TARGET_TRIPLET=x64-linux -DCMAKE_EXPORT_COMPILE_COMMANDS=ON + + - name: Check Formatting + run: | + cmake --build build/release --target format + git diff --exit-code + + - name: Run Clang-Tidy + run: cmake --build build/release --target tidy + build: name: ${{ matrix.os }}-build runs-on: ${{ matrix.os }} @@ -63,6 +141,12 @@ jobs: echo "Disk space after cleanup:" df -h / + - name: Setup uv + uses: astral-sh/setup-uv@v5 + with: + python-version: "3.14.3" + enable-cache: true + - name: Install Linux Dependencies if: runner.os == 'Linux' run: | @@ -79,7 +163,10 @@ jobs: automake \ libtool \ libltdl-dev \ - mono-complete + mono-complete \ + xvfb \ + clang-tidy \ + clang-format - name: Install macOS Dependencies if: runner.os == 'macOS' @@ -151,7 +238,7 @@ jobs: uses: lukka/get-cmake@latest - name: Configure CMake - run: cmake --preset ${{ matrix.preset }} -DVCPKG_TARGET_TRIPLET=${{ matrix.triplet }} + run: cmake --preset ${{ matrix.preset }} -DVCPKG_TARGET_TRIPLET=${{ matrix.triplet }} -DENABLE_SPIX_TESTING=ON - name: Print vcpkg ABI info if: always() @@ -177,7 +264,33 @@ jobs: fi - name: Build - run: cmake --build --preset ${{ matrix.preset }} + run: cmake --build build/${{ matrix.preset }} + + - name: Test + run: ctest --preset ${{ matrix.preset }} --output-on-failure + + - name: Run E2E Tests (Linux) + if: runner.os == 'Linux' + run: | + xvfb-run -a bash -c "./build/${{ matrix.preset }}/filmulator --test-mode & sleep 10 && uv run python tests/e2e/test_raw_pipeline.py" + + - name: Run E2E Tests (macOS) + if: runner.os == 'macOS' + run: | + ./build/${{ matrix.preset }}/filmulator.app/Contents/MacOS/filmulator --test-mode & + PID=$! + sleep 10 + uv run python tests/e2e/test_raw_pipeline.py + kill $PID || true + + - name: Run E2E Tests (Windows) + if: runner.os == 'Windows' + shell: pwsh + run: | + Start-Process -FilePath ".\build\${{ matrix.preset }}\filmulator.exe" -ArgumentList "--test-mode" -PassThru + Start-Sleep -Seconds 10 + uv run python tests/e2e/test_raw_pipeline.py + Stop-Process -Name "filmulator" -Force -ErrorAction SilentlyContinue - name: Install (Windows) if: runner.os == 'Windows' @@ -471,6 +584,14 @@ jobs: with: name: filmulator-windows-installer + - name: Checkout Source + uses: actions/checkout@v4 + + - name: Setup uv + uses: astral-sh/setup-uv@v5 + with: + enable-cache: true + - name: Test AppImage (Linux) if: runner.os == 'Linux' run: | @@ -492,24 +613,9 @@ jobs: # Also install system-level OpenGL graphics libraries (libgl1, libegl1) since headless runners lack them by default. sudo apt-get update && sudo apt-get install -y xvfb libfuse2 libgl1 libegl1 - echo "Running smoke test with xvfb..." + echo "Running E2E tests with xvfb..." # Use xvfb-run to provide virtual X11 display for xcb platform - xvfb-run -a ./"$APPIMAGE" &> smoke_test.log & - APP_PID=$! - - # Give it time to load and initialize - sleep 10 - - # Check if process is still running - if ! ps -p $APP_PID > /dev/null; then - echo "Application exited prematurely" - echo "--- Output Log ---" - cat smoke_test.log || echo "(no log captured)" - exit 1 - fi - - echo "Application started successfully. Smoke test passed!" - kill $APP_PID 2>/dev/null || true + xvfb-run -a uv run python tests/e2e/test_raw_pipeline.py --bin ./"$APPIMAGE" - name: Test DMG (macOS) if: runner.os == 'macOS' @@ -557,39 +663,11 @@ jobs: mkdir -p "$TEST_DIR" cp -R "$APP_BUNDLE" "$TEST_DIR/" - echo "Running smoke test..." + echo "Running E2E tests..." # macOS runners have a display server, so cocoa platform should work - LAUNCHER="$TEST_DIR/Filmulator.app/Contents/MacOS/filmulator-gui" - if [ ! -f "$LAUNCHER" ]; then - echo "Error: Launcher script not found at $LAUNCHER" - exit 1 - fi + LAUNCHER="$TEST_DIR/Filmulator.app/Contents/MacOS/filmulator" - # Make sure launcher is executable - chmod +x "$LAUNCHER" - - "$LAUNCHER" &> smoke_test.log & - APP_PID=$! - - # Give it time to load and initialize - sleep 10 - - # Check if process is still running - if ! ps -p $APP_PID > /dev/null; then - echo "Application exited prematurely or crashed" - echo "--- Smoke Test Log ---" - cat smoke_test.log || echo "(no log captured)" - - # Try to get more info from system log if it crashed - echo "--- Tail of system console log (last 50 lines) ---" - log show --predicate 'process == "filmulator"' --last 2m | tail -n 50 - - hdiutil detach "$VOLUME" || true - exit 1 - fi - - echo "Application started successfully. Smoke test passed!" - kill $APP_PID 2>/dev/null || true + uv run python tests/e2e/test_raw_pipeline.py --bin "$LAUNCHER" # Cleanup hdiutil detach "$VOLUME" @@ -639,35 +717,8 @@ jobs: } $env:PATH = $oldPath - Write-Host "Running smoke test on installed application..." - $outputLog = "${{ github.workspace }}\smoke_test_output.log" - $errorLog = "${{ github.workspace }}\smoke_test_error.log" - $startTime = Get-Date - - # Start the process and capture output - $process = Start-Process -FilePath $exePath -PassThru -RedirectStandardOutput $outputLog -RedirectStandardError $errorLog - - # Give it some time to load DLLs and fail if there's an issue - Start-Sleep -Seconds 10 - - if ($process.HasExited) { - $exitCode = $process.ExitCode - Write-Host "Application exited prematurely with code $exitCode." -ForegroundColor Red - - Write-Host "--- Standard Output ---" - if (Test-Path $outputLog) { Get-Content $outputLog } else { Write-Host "(no output captured)" } - Write-Host "--- Standard Error ---" - if (Test-Path $errorLog) { Get-Content $errorLog } else { Write-Host "(no error captured)" } - - Write-Host "--- Querying Windows Event Logs for Application Errors ---" - $events = Get-EventLog -LogName Application -After $startTime -EntryType Error -Newest 10 | Where-Object { $_.Source -match "Application Error" -or $_.Source -match "SideBySide" } - if ($events) { $events | Format-List } else { Write-Host "No relevant Event Log entries found." } - - exit 1 - } - - Write-Host "Application started successfully. Smoke test passed!" - Stop-Process -Id $process.Id -Force -ErrorAction SilentlyContinue + Write-Host "Running E2E tests on installed application..." + uv run python tests/e2e/test_raw_pipeline.py --bin $exePath # Tests will be added later when the test framework is set up # - name: Test diff --git a/.python-version b/.python-version new file mode 100644 index 00000000..da717732 --- /dev/null +++ b/.python-version @@ -0,0 +1 @@ +3.14.3 diff --git a/.vscode/launch.json b/.vscode/launch.json new file mode 100644 index 00000000..565fc93a --- /dev/null +++ b/.vscode/launch.json @@ -0,0 +1,63 @@ +{ + // Use IntelliSense to learn about possible attributes. + // Hover to view descriptions of existing attributes. + // For more information, visit: https://go.microsoft.com/fwlink/?linkid=830387 + "version": "0.2.0", + "configurations": [ + { + "name": "Python Debugger: Current File", + "type": "debugpy", + "request": "launch", + "program": "${file}", + "console": "integratedTerminal" + }, + { + "name": "Python: Debug E2E Pipeline", + "type": "debugpy", + "request": "launch", + "program": "${workspaceFolder}/tests/e2e/test_raw_pipeline.py", + "console": "integratedTerminal", + "cwd": "${workspaceFolder}", + "env": { + "FILMULATOR_BIN": "${workspaceFolder}/build/Filmulator.app/Contents/MacOS/filmulator", + "FILMULATOR_DB_DIR": "${workspaceFolder}/build/test_db" + } + }, + { + "name": "Python: Debug E2E (External App)", + "type": "debugpy", + "request": "launch", + "program": "${workspaceFolder}/tests/e2e/test_raw_pipeline.py", + "console": "integratedTerminal", + "cwd": "${workspaceFolder}", + "args": [ + "--no-start" + ], + "env": { + "FILMULATOR_DB_DIR": "${workspaceFolder}/build/test_db" + } + }, + { + "name": "C++: Debug Filmulator (MacOS)", + "type": "lldb", + "request": "launch", + "program": "${workspaceFolder}/build/Filmulator.app/Contents/MacOS/filmulator", + "args": [ + "--test-mode" + ], + "cwd": "${workspaceFolder}", + "env": { + "FILMULATOR_DB_DIR": "${workspaceFolder}/build/test_db" + } + } + ], + "compounds": [ + { + "name": "Compound: Debug Application + Test", + "configurations": [ + "C++: Debug Filmulator (MacOS)", + "Python: Debug E2E (External App)" + ] + } + ] +} \ No newline at end of file diff --git a/.vscode/settings.json b/.vscode/settings.json new file mode 100644 index 00000000..1c8aad35 --- /dev/null +++ b/.vscode/settings.json @@ -0,0 +1,5 @@ +{ + "editor.formatOnSave": true, + "C_Cpp.clang_format_path": "clang-format", + "C_Cpp.formatting": "clangFormat" +} \ No newline at end of file diff --git a/.vscode/tasks.json b/.vscode/tasks.json new file mode 100644 index 00000000..249d19a8 --- /dev/null +++ b/.vscode/tasks.json @@ -0,0 +1,59 @@ +{ + "version": "2.0.0", + "tasks": [ + { + "label": "Format", + "type": "shell", + "command": "cmake", + "args": [ + "--build", + "${command:cmake.buildDirectory}", + "--target", + "format" + ], + "group": "build", + "presentation": { + "reveal": "silent" + } + }, + { + "label": "Build", + "type": "shell", + "command": "cmake", + "args": [ + "--build", + "${command:cmake.buildDirectory}" + ], + "group": "build", + "problemMatcher": "$gcc" + }, + { + "label": "Test", + "type": "shell", + "command": "ctest", + "args": [ + "--test-dir", + "${command:cmake.buildDirectory}", + "--output-on-failure" + ], + "group": "test", + "presentation": { + "reveal": "always" + } + }, + { + "label": "Format, Build and Test", + "dependsOn": [ + "Format", + "Build", + "Test" + ], + "dependsOrder": "sequence", + "group": { + "kind": "build", + "isDefault": true + }, + "problemMatcher": [] + } + ] +} \ No newline at end of file diff --git a/CMakeLists.txt b/CMakeLists.txt index 0c7ba74f..986f1a0f 100644 --- a/CMakeLists.txt +++ b/CMakeLists.txt @@ -39,6 +39,7 @@ find_package(libraw CONFIG REQUIRED) find_package(JPEG REQUIRED) find_package(TIFF REQUIRED) find_package(OpenMP REQUIRED) + set(_lensfun_found FALSE) find_package(lensfun QUIET) if(lensfun_FOUND) @@ -61,6 +62,38 @@ endif() find_package(CURL CONFIG REQUIRED) find_package(LibArchive REQUIRED) find_package(ZLIB REQUIRED) +find_package(spdlog CONFIG REQUIRED) + +# Spix E2E testing support (optional) +option(ENABLE_SPIX_TESTING "Enable Spix-based E2E testing" OFF) + +if(NOT DEFINED ENABLE_NAN_TRAPPING) + if(CMAKE_BUILD_TYPE STREQUAL "Debug") + set(ENABLE_NAN_TRAPPING ON) + else() + set(ENABLE_NAN_TRAPPING OFF) + endif() +endif() +option(ENABLE_NAN_TRAPPING "Enable NaN trapping and debug breaks" ${ENABLE_NAN_TRAPPING}) + +if(ENABLE_SPIX_TESTING) + # AnyRPC from vcpkg (no CMake config, so set paths manually) + set(AnyRPC_INCLUDE_DIRS "${CMAKE_BINARY_DIR}/vcpkg_installed/${VCPKG_TARGET_TRIPLET}/include") + set(AnyRPC_LIBRARIES "${CMAKE_BINARY_DIR}/vcpkg_installed/${VCPKG_TARGET_TRIPLET}/lib/libanyrpc.a") + + include(FetchContent) + + # Spix - UI test automation library (not in vcpkg, so fetch it) + FetchContent_Declare( + spix + GIT_REPOSITORY https://github.com/faaxm/spix.git + GIT_TAG v0.9 + ) + set(SPIX_QT_MAJOR 5 CACHE STRING "" FORCE) + set(SPIX_BUILD_EXAMPLES OFF CACHE BOOL "" FORCE) + + FetchContent_MakeAvailable(spix) +endif() # SET( EIGEN3_INCLUDE_DIR "$ENV{EIGEN3_INCLUDE_DIR}" ) # IF( NOT EIGEN3_INCLUDE_DIR ) @@ -72,8 +105,7 @@ find_package(ZLIB REQUIRED) include(ConfigureChecks.cmake) configure_file(config.h.cmake ${CMAKE_CURRENT_BINARY_DIR}/config.h) -set(filmulator_SRCS - filmulator-gui/main.cpp +set(filmulator_CORE_SRCS filmulator-gui/core/agitate.cpp filmulator-gui/core/colorCurves.cpp filmulator-gui/core/colorSpaces.cpp @@ -83,6 +115,13 @@ set(filmulator_SRCS filmulator-gui/core/exposure.cpp filmulator-gui/core/filmulate.cpp filmulator-gui/core/imagePipeline.cpp + filmulator-gui/core/pipeline/LoadStage.cpp + filmulator-gui/core/pipeline/DemosaicStage.cpp + filmulator-gui/core/pipeline/PostDemosaicStage.cpp + filmulator-gui/core/pipeline/NlmeansNRStage.cpp + filmulator-gui/core/pipeline/ImpulseNRStage.cpp + filmulator-gui/core/pipeline/ChromaNRStage.cpp + filmulator-gui/core/pipeline/PrefilmulationStage.cpp filmulator-gui/core/imload.cpp filmulator-gui/core/imread.cpp filmulator-gui/core/imreadJpeg.cpp @@ -90,6 +129,7 @@ set(filmulator_SRCS filmulator-gui/core/imwriteJpeg.cpp filmulator-gui/core/imwriteTiff.cpp filmulator-gui/core/layerMix.cpp + filmulator-gui/core/logging.cpp filmulator-gui/core/mergeExps.cpp filmulator-gui/core/outputFile.cpp filmulator-gui/core/rotateImage.cpp @@ -143,51 +183,34 @@ qtquick_compiler_add_resources(filmulator_RSCS filmulator-gui/resources/pixmaps.qrc ) -if(APPLE) - add_executable(filmulator MACOSX_BUNDLE - ${filmulator_SRCS} - ${filmulator_RSCS} - ) -elseif(WIN32) - add_executable(filmulator # WIN32 # comment out WIN32 to get a terminal window - ${filmulator_SRCS} - ${filmulator_RSCS} - ) - target_link_directories(filmulator PRIVATE ${GLIB2_LIBRARY_DIRS}) -else() - add_executable(filmulator - ${filmulator_SRCS} - ${filmulator_RSCS} - ) -endif() -target_compile_definitions(filmulator PRIVATE) - +# Create the core library +add_library(filmulator-lib STATIC + ${filmulator_CORE_SRCS} + ${filmulator_RSCS} +) +target_compile_definitions(filmulator-lib + PUBLIC + $<$:ENABLE_NAN_TRAPPING> +) -target_compile_options(filmulator +# Compile options for the library +target_compile_options(filmulator-lib PRIVATE ${DEFAULT_CXX_COMPILER_FLAGS} -DHAVE_CONFIG_H $<$:/openmp:experimental> ) -set_target_properties(filmulator PROPERTIES - CXX_STANDARD 17 - CXX_STANDARD_REQUIRED ON -) - -# Note: Modern CMake imported targets handle include directories automatically -# when linked via target_link_libraries. Only local directories need explicit listing. -target_include_directories(filmulator - PRIVATE +# Include directories for the library +target_include_directories(filmulator-lib + PUBLIC filmulator-gui/core filmulator-gui/database filmulator-gui/ui filmulator-gui/qtquick2applicationviewer ${CMAKE_CURRENT_BINARY_DIR} ) -#glib2 needed for static lensfun on windows above and below -#libraw_include_dirs for pointing windows build at it if(TARGET PkgConfig::LENSFUN) set(LENSFUN_TARGET PkgConfig::LENSFUN) @@ -197,9 +220,9 @@ else() message(FATAL_ERROR "lensfun target not found") endif() -# Use modern CMake imported targets for proper include/link handling -target_link_libraries(filmulator - PRIVATE +# Link libraries for the library +target_link_libraries(filmulator-lib + PUBLIC Qt5::Core Qt5::Sql Qt5::Widgets @@ -211,13 +234,54 @@ target_link_libraries(filmulator TIFF::TIFF # TIFF modern target ${LENSFUN_TARGET} PkgConfig::GLIB - CURL::libcurl # CURL modern target + CURL::libcurl LibArchive::LibArchive - ZLIB::ZLIB # ZLIB modern target + ZLIB::ZLIB OpenMP::OpenMP_CXX rtprocess::rtprocess + spdlog::spdlog ) +# Define the executable +if(APPLE) + add_executable(filmulator MACOSX_BUNDLE + filmulator-gui/main.cpp + ) + +elseif(WIN32) + add_executable(filmulator + filmulator-gui/main.cpp + ) + target_link_directories(filmulator PRIVATE ${GLIB2_LIBRARY_DIRS}) +else() + add_executable(filmulator + filmulator-gui/main.cpp + ) +endif() +target_compile_definitions(filmulator PRIVATE) +set_target_properties(filmulator PROPERTIES + CXX_STANDARD 17 + CXX_STANDARD_REQUIRED ON +) + +# Link the executable to the library +target_link_libraries(filmulator + PRIVATE + filmulator-lib +) + +# Conditionally link Spix for E2E testing +if(ENABLE_SPIX_TESTING) + target_link_libraries(filmulator PRIVATE Spix ${AnyRPC_LIBRARIES}) + target_include_directories(filmulator PRIVATE + ${spix_SOURCE_DIR}/lib/include + ) + target_compile_definitions(filmulator PRIVATE ENABLE_SPIX_TESTING) +endif() + +enable_testing() +add_subdirectory(tests) + # let's see where all this stuff is... #message(STATUS "EXIV2_VERSION=${EXIV2_VERSION}") #message(STATUS "CURL_INCLUDE_DIRS=${CURL_INCLUDE_DIRS}") @@ -260,3 +324,36 @@ endif() qt5_import_qml_plugins(filmulator) add_subdirectory(filmulator-gui/qml) + +# Code formatting and linting +find_program(CLANG_FORMAT "clang-format") +if(CLANG_FORMAT) + file(GLOB_RECURSE ALL_S_FILES + "filmulator-gui/*.cpp" + "filmulator-gui/*.hpp" + "filmulator-gui/*.h" + "tests/*.cpp" + "tests/*.hpp" + "tests/*.h" + ) + # Exclude third-party cJSON + list(REMOVE_ITEM ALL_S_FILES "${CMAKE_SOURCE_DIR}/filmulator-gui/database/cJSON.c") + + add_custom_target( + format + COMMAND ${CLANG_FORMAT} -i -style=file ${ALL_S_FILES} + WORKING_DIRECTORY ${CMAKE_SOURCE_DIR} + COMMENT "Running clang-format on all source files..." + ) +endif() + +find_program(CLANG_TIDY "clang-tidy") +if(CLANG_TIDY) + set(CMAKE_EXPORT_COMPILE_COMMANDS ON CACHE BOOL "" FORCE) + add_custom_target( + tidy + COMMAND ${CLANG_TIDY} -p ${CMAKE_BINARY_DIR} ${filmulator_CORE_SRCS} filmulator-gui/main.cpp + WORKING_DIRECTORY ${CMAKE_SOURCE_DIR} + COMMENT "Running clang-tidy on core source files..." + ) +endif() diff --git a/CMakePresets.json b/CMakePresets.json index b424eb6c..50cf0a4e 100644 --- a/CMakePresets.json +++ b/CMakePresets.json @@ -53,15 +53,18 @@ "buildPresets": [ { "name": "debug", - "configurePreset": "debug" + "configurePreset": "debug", + "jobs": 0 }, { "name": "release", - "configurePreset": "release" + "configurePreset": "release", + "jobs": 0 }, { "name": "relwithdebinfo", - "configurePreset": "relwithdebinfo" + "configurePreset": "relwithdebinfo", + "jobs": 0 } ] } \ No newline at end of file diff --git a/GEMINI.md b/GEMINI.md new file mode 100644 index 00000000..cddbbf75 --- /dev/null +++ b/GEMINI.md @@ -0,0 +1,3 @@ +Remember to compile with all available cores +running `pkill -f filmulator` will crash the IDE! +Never increase a timeout without checking with the user first \ No newline at end of file diff --git a/README.md b/README.md index 2cf60272..f953c260 100644 --- a/README.md +++ b/README.md @@ -118,6 +118,38 @@ Tools and features of interest: If you want the UI to appear larger on a high-pixel density display, use the User Interface Scale slider in the `Settings` tab to set a desired scale, save the setting, and then restart the program. While it cannot automatically read your display pixel density, this setting enables otherwise full HiDPI support. +# Testing + +## E2E Tests + +The project uses Python-based E2E tests for the RAW pipeline. We use [**uv**](https://github.com/astral-sh/uv) to manage Python dependencies and the virtual environment. + +### Prerequisites + +Install `uv`: +```bash +curl -LsSf https://astral.ml/uv/install.sh | sh +``` + +### Running Tests + +To run the RAW pipeline E2E test: +```bash +uv run python tests/e2e/test_raw_pipeline.py +``` + +To preserve the output JPEG for analysis (e.g., if a test fails): +```bash +uv run python tests/e2e/test_raw_pipeline.py --keep-output +``` + +### Analyzing Differences + +If a test fails, you can use the diagnostic notebook to inspect pixel-wise differences: +```bash +uv run jupyter notebook tests/e2e/analyze_diff.ipynb +``` + # Status diff --git a/compile_commands.json b/compile_commands.json new file mode 120000 index 00000000..b1c5dedd --- /dev/null +++ b/compile_commands.json @@ -0,0 +1 @@ +build/relwithdebinfo/compile_commands.json \ No newline at end of file diff --git a/docs/pipeline_refactor_plan.md b/docs/pipeline_refactor_plan.md new file mode 100644 index 00000000..d42722ba --- /dev/null +++ b/docs/pipeline_refactor_plan.md @@ -0,0 +1,124 @@ +# Image Pipeline Refactor Plan + +## Goal +Refactor the monolithic `ImagePipeline` class into a modular system of generic **Pipeline Stages**. +This will improve maintainability, testability, and flexibility of the image processing chain. + +## User Review Required +> [!IMPORTANT] +> This is a pure refactor. No functionality changes are expected. +> The performance impact should be negligible (compile-time composition preferred). +> **Decision Point**: Should we use compile-time (templates/tuples) or runtime (polymorphism/variants) composition? +> *Recommendation*: **Compile-time** (std::tuple of stages) for type safety and performance, as the pipeline structure is static (always A -> B -> C, even if C is effectively a no-op). + +## Concept: Pipeline Stage +A **Stage** is a distinct step in processing. It is defined by: +- **Input Type**: Data it consumes (e.g., `RawImage`, `FloatImage`). +- **Output Type**: Data it produces. +- **Parameters**: Configuration struct (POD). +- **Function**: `Output process(Input, Params, Context)`. + +### Requirements +1. **Type Safety**: Stages must define their I/O types. +2. **Parameter Isolation**: Stages accept explicit parameter structs, not the full `ParameterManager`. +3. **Cancellation**: Stages must support checking for abort signals. +4. **Progress Reporting**: Stages report progress (0.0 to 1.0) which the pipeline scales. +5. **Caching**: The pipeline (or stage wrapper) caches the output based on parameter hashes/versioning. + +## Proposed Architecture + +### 1. The Stage Interface +We can use a CRTP or simple Template pattern. + +```cpp +template +class Stage { +public: + using Input = InputT; + using Output = OutputT; + using Params = ParamsT; + + virtual ~Stage() = default; + + // Returns std::nullopt if aborted + virtual std::optional process(const Input& input, + const Params& params, + PipelineContext& context) = 0; +}; +``` + +### 2. The Context +Holds shared state like `ExifData` that might be mutated or read by multiple stages, and the `Interface` for callback (histograms). Or handles cancellation. + +```cpp +struct PipelineContext { + std::atomic abortParams{false}; + Interface* interface; + // Helper to check abort status + bool isAborted(); + // Helper to report progress + void updateProgress(float p); +}; +``` + +### 3. The Pipeline Coordinator +Manages the sequence of stages and caching. + +```cpp +template +class Pipeline { + std::tuple stages; + std::tuple cachedOutputs; + std::tuple cachedParams; // To check if changed + + // Logic to run stages: + // For each stage I: + // If valid < Stage[I].validityLevel: + // Get Params + // Result = Stage[I].process(PreviousResult, Params, Context) + // Cache Result + // Update valid +}; +``` +*Note: Since the stages are heterogeneous (different types), the `previousResult` passing needs to be handled via template recursion or fold expressions.* + +## Proposed Stages (Mapping) + +| Existing Step | Input | Output | Params | +| I | DataFlow | DataFlow | `claim...Params` | +| --- | --- | --- | --- | +| **Loader** | `std::string` (Filename) | `RawImage` + `Exif` | `LoadParams` | +| **Demosaic** | `RawImage` | `FloatImage` (Crop) | `DemosaicParams` | +| **PostDemosaic** | `FloatImage` | `FloatImage` | `PostDemosaicParams` | +| **Denoise (NL)** | `FloatImage` | `FloatImage` (Lab) | `NlmeansNRParams` | +| **Denoise (Imp)**| `FloatImage` | `FloatImage` (Lab) | `ImpulseNRParams` | +| **Denoise (Chr)**| `FloatImage` | `FloatImage` (Lab) | `ChromaNRParams` | +| **PreFilm** | `FloatImage` | `FloatImage` | `PrefilmParams` | +| **Filmulation** | `FloatImage` | `FloatImage` (sRGB) | `FilmParams` | +| **Compositing** | `FloatImage` | `ShortImage` | `BlackWhiteParams` + ... | + +## Implementation Plan + +### Phase 1: Infrastructure +1. Define `PipelineContext` and base `Stage` templates in `core/pipeline/`. +2. Implement a simple `Pipeline` class that can chain 2 stages. + +### Phase 2: Migration (Incremental) +We can migrate stages one by one. +1. Create `LoadStage` class moving code from `processImage` case `Valid::load`. +2. Create `DemosaicStage` moving code from `Valid::demosaic`. +3. ... +For the transition, `ImagePipeline` will hold instances of these Stages and call them manually, replacing the inline code. + +### Phase 3: The New Pipeline +1. Once all logic is in Stage classes, replace the `ImagePipeline::processImage` giant switch with the generic `Pipeline` runner. +2. Wire up `ParameterManager` to feed the `Pipeline`. + +## Verification Plan +1. **Unit Tests**: + - Test `Pipeline` generic logic (caching, chaining). + - Test individual `Stage` classes in isolation (mock inputs). + - *New*: Create `tests/unit/test_pipeline_infrastructure.cpp`. +2. **Regression Tests**: + - Run `test_raw_pipeline.py` (End-to-End) to ensure the output image is identical pixel-for-pixel (or within float tolerance) to the current implementation. + - Since we have "Golden Data", this is perfect. diff --git a/filmulator-gui/.DS_Store b/filmulator-gui/.DS_Store new file mode 100644 index 00000000..93a78908 Binary files /dev/null and b/filmulator-gui/.DS_Store differ diff --git a/filmulator-gui/CMakePresets.json b/filmulator-gui/CMakePresets.json new file mode 100644 index 00000000..eaa77cfb --- /dev/null +++ b/filmulator-gui/CMakePresets.json @@ -0,0 +1,66 @@ +{ + "version": 6, + "cmakeMinimumRequired": { + "major": 3, + "minor": 21, + "patch": 0 + }, + "configurePresets": [ + { + "name": "vcpkg-base", + "hidden": true, + "cacheVariables": { + "CMAKE_TOOLCHAIN_FILE": { + "type": "FILEPATH", + "value": "$env{VCPKG_ROOT}/scripts/buildsystems/vcpkg.cmake" + }, + "VCPKG_OVERLAY_PORTS": { + "type": "PATH", + "value": "${sourceDir}/vcpkg-overlay" + }, + "VCPKG_INSTALL_OPTIONS": "--allow-unsupported" + } + }, + { + "name": "debug", + "displayName": "Debug with vcpkg", + "inherits": "vcpkg-base", + "binaryDir": "${sourceDir}/build/debug", + "cacheVariables": { + "CMAKE_BUILD_TYPE": "Debug" + } + }, + { + "name": "release", + "displayName": "Release with vcpkg", + "inherits": "vcpkg-base", + "binaryDir": "${sourceDir}/build/release", + "cacheVariables": { + "CMAKE_BUILD_TYPE": "Release" + } + }, + { + "name": "relwithdebinfo", + "displayName": "RelWithDebInfo with vcpkg (default)", + "inherits": "vcpkg-base", + "binaryDir": "${sourceDir}/build/relwithdebinfo", + "cacheVariables": { + "CMAKE_BUILD_TYPE": "RelWithDebInfo" + } + } + ], + "buildPresets": [ + { + "name": "debug", + "configurePreset": "debug" + }, + { + "name": "release", + "configurePreset": "release" + }, + { + "name": "relwithdebinfo", + "configurePreset": "relwithdebinfo" + } + ] +} \ No newline at end of file diff --git a/filmulator-gui/Halide/HSVtoRGB.cpp b/filmulator-gui/Halide/HSVtoRGB.cpp index c22cf2cc..3ba1f5a8 100644 --- a/filmulator-gui/Halide/HSVtoRGB.cpp +++ b/filmulator-gui/Halide/HSVtoRGB.cpp @@ -1,48 +1,31 @@ Halide::Func HSVtoRGB(Func in) { Func out; - Var x,y,c; - Func h,s,v; - h(x,y) = in(x,y,0); - s(x,y) = in(x,y,1); - v(x,y) = in(x,y,2); - Func r,g,b; - Func hd,i,f,p,q,t; - hd(x,y) = h(x,y)/60; - i(x,y) = Halide::floor(hd(x,y)); - f(x,y) = hd(x,y) - i(x,y); - p(x,y) = v(x,y)*(1-s(x,y)); - q(x,y) = v(x,y)*(1-(s(x,y)*f(x,y))); - t(x,y) = v(x,y)*(1-(s(x,y)*(1-f(x,y)))); + Var x, y, c; + Func h, s, v; + h(x, y) = in(x, y, 0); + s(x, y) = in(x, y, 1); + v(x, y) = in(x, y, 2); + Func r, g, b; + Func hd, i, f, p, q, t; + hd(x, y) = h(x, y) / 60; + i(x, y) = Halide::floor(hd(x, y)); + f(x, y) = hd(x, y) - i(x, y); + p(x, y) = v(x, y) * (1 - s(x, y)); + q(x, y) = v(x, y) * (1 - (s(x, y) * f(x, y))); + t(x, y) = v(x, y) * (1 - (s(x, y) * (1 - f(x, y)))); - r(x,y) = select(i(x,y) == 0 || i(x,y) == 5, - v(x,y), - select(i(x,y) == 1, - q(x,y), - select(i(x,y) == 2 || i(x,y) == 3, - p(x,y), - select(i(x,y) == 4, - t(x,y), - v(x,y)))));//default - g(x,y) = select(i(x,y) == 0, - t(x,y), - select(i(x,y) == 1 || i(x,y) == 2, - v(x,y), - select(i(x,y) == 3, - q(x,y), - p(x,y))));//4,5,default - b(x,y) = select(i(x,y) == 0 || i(x,y) == 1, - p(x,y), - select(i(x,y) == 2, - t(x,y), - select(i(x,y) == 3 || i(x,y) == 4, - v(x,y), - q(x,y))));//5,default - out(x,y,c) = select(s(x,y) == 0, - v(x,y), - select(c == 0, r(x,y), - c == 1, g(x,y), - b(x,y))); + r(x, y) = select(i(x, y) == 0 || i(x, y) == 5, + v(x, y), + select(i(x, y) == 1, + q(x, y), + select(i(x, y) == 2 || i(x, y) == 3, p(x, y), select(i(x, y) == 4, t(x, y), v(x, y)))));// default + g(x, y) = select(i(x, y) == 0, + t(x, y), + select(i(x, y) == 1 || i(x, y) == 2, v(x, y), select(i(x, y) == 3, q(x, y), p(x, y))));// 4,5,default + b(x, y) = select(i(x, y) == 0 || i(x, y) == 1, + p(x, y), + select(i(x, y) == 2, t(x, y), select(i(x, y) == 3 || i(x, y) == 4, v(x, y), q(x, y))));// 5,default + out(x, y, c) = select(s(x, y) == 0, v(x, y), select(c == 0, r(x, y), c == 1, g(x, y), b(x, y))); return out; } - diff --git a/filmulator-gui/Halide/RGBtoHSV.cpp b/filmulator-gui/Halide/RGBtoHSV.cpp index 881e7d09..50e5816c 100644 --- a/filmulator-gui/Halide/RGBtoHSV.cpp +++ b/filmulator-gui/Halide/RGBtoHSV.cpp @@ -1,24 +1,21 @@ Halide::Func RGBtoHSV(Func in) { Func out; - Var x,y,c; - Func minC,maxC,delta; - Func h,s,v; - minC(x,y) = min(in(x,y,0),min(in(x,y,1),in(x,y,2))); - maxC(x,y) = max(in(x,y,0),max(in(x,y,1),in(x,y,2))); - v(x,y) = maxC(x,y); - delta(x,y) = maxC(x,y) - minC(x,y); - s(x,y) = select(maxC(x,y) != 0, delta(x,y)/maxC(x,y), 0);//S - h(x,y) = select(maxC(x,y) == in(0,x,y), //R is highest - (in(1,x,y) - in(2,x,y))/delta(x,y), - select(maxC(x,y) == in(1,x,y), //G is highest - 2 + (in(2,x,y) - in(0,x,y))/delta(x,y), - 4 + (in(0,x,y) - in(1,x,y))/delta(x,y)));//B is highest + Var x, y, c; + Func minC, maxC, delta; + Func h, s, v; + minC(x, y) = min(in(x, y, 0), min(in(x, y, 1), in(x, y, 2))); + maxC(x, y) = max(in(x, y, 0), max(in(x, y, 1), in(x, y, 2))); + v(x, y) = maxC(x, y); + delta(x, y) = maxC(x, y) - minC(x, y); + s(x, y) = select(maxC(x, y) != 0, delta(x, y) / maxC(x, y), 0);// S + h(x, y) = select(maxC(x, y) == in(0, x, y),// R is highest + (in(1, x, y) - in(2, x, y)) / delta(x, y), + select(maxC(x, y) == in(1, x, y),// G is highest + 2 + (in(2, x, y) - in(0, x, y)) / delta(x, y), + 4 + (in(0, x, y) - in(1, x, y)) / delta(x, y)));// B is highest Func hNorm; - hNorm(x,y) = select(h(x,y) < 0, 60.0f*h(x,y) + 360.0f, 60.0f*h(x,y)); - out(x,y,c) = select(c == 0, hNorm(x,y), - c == 1, s(x,y), - v(x,y)); + hNorm(x, y) = select(h(x, y) < 0, 60.0f * h(x, y) + 360.0f, 60.0f * h(x, y)); + out(x, y, c) = select(c == 0, hNorm(x, y), c == 1, s(x, y), v(x, y)); return out; } - diff --git a/filmulator-gui/Halide/applyfilmlikecurve.cpp b/filmulator-gui/Halide/applyfilmlikecurve.cpp index b4907442..59e16d0f 100644 --- a/filmulator-gui/Halide/applyfilmlikecurve.cpp +++ b/filmulator-gui/Halide/applyfilmlikecurve.cpp @@ -1,43 +1,43 @@ #include -Func applyFilmlikeCurve(Func input, Func LUT){ +Func applyFilmlikeCurve(Func input, Func LUT) +{ - Var x,y,c,o; - Func order; - Expr rVal = input(x,y,0), gVal = input(x,y,1), bVal = input(x,y,1); - Expr r = 0, g = 1, b = 2; - order(x,y,o) = select(rVal >= gVal, - select(gVal >= bVal, - select(o == 0, b, o == 1, g, r), //Order is BGR - select(rVal >= bVal, - select(o == 0, g, o == 1, b, r), //GBR - select(o == 0, g, o == 1, r, b))), //GRB - select(rVal >= bVal, - select(o == 0, b, o == 1, r, g), //BRG - select(gVal >= bVal, - select(o == 0, r, o == 1, b, g), //RBG - select(o == 0, r, o == 1, g, b)))); //RGB - Func curved; - Expr maxVal = 2^16; - curved(x,y,o) = LUT(cast(UInt(16),maxVal*input(x,y,order(x,y,o)))); - - Expr lowOld = input(x,y,order(x,y,0)); - Expr midOld = input(x,y,order(x,y,1)); - Expr hiOld = input(x,y,order(x,y,2)); + Var x, y, c, o; + Func order; + Expr rVal = input(x, y, 0), gVal = input(x, y, 1), bVal = input(x, y, 1); + Expr r = 0, g = 1, b = 2; + order(x, y, o) = select(rVal >= gVal, + select(gVal >= bVal, + select(o == 0, b, o == 1, g, r),// Order is BGR + select(rVal >= bVal, + select(o == 0, g, o == 1, b, r),// GBR + select(o == 0, g, o == 1, r, b))),// GRB + select(rVal >= bVal, + select(o == 0, b, o == 1, r, g),// BRG + select(gVal >= bVal, + select(o == 0, r, o == 1, b, g),// RBG + select(o == 0, r, o == 1, g, b))));// RGB + Func curved; + Expr maxVal = 2 ^ 16; + curved(x, y, o) = LUT(cast(UInt(16), maxVal * input(x, y, order(x, y, o)))); - Expr lowNew = curved(x,y,0); - Expr hiNew = curved(x,y,2); + Expr lowOld = input(x, y, order(x, y, 0)); + Expr midOld = input(x, y, order(x, y, 1)); + Expr hiOld = input(x, y, order(x, y, 2)); - Expr epsilon = FLT_MIN; - Expr multFactor = (hiNew - lowNew + epsilon)/(hiOld - lowOld + epsilon); - Expr correction = (lowNew - multFactor*lowOld); - curved(x,y,1) = multFactor*midOld + correction; + Expr lowNew = curved(x, y, 0); + Expr hiNew = curved(x, y, 2); - Func output; - output(x,y,c) = undef(); - output(x,y,order(x,y,0)) = curved(x,y,0); - output(x,y,order(x,y,1)) = curved(x,y,1); - output(x,y,order(x,y,2)) = curved(x,y,2); + Expr epsilon = FLT_MIN; + Expr multFactor = (hiNew - lowNew + epsilon) / (hiOld - lowOld + epsilon); + Expr correction = (lowNew - multFactor * lowOld); + curved(x, y, 1) = multFactor * midOld + correction; - return output; + Func output; + output(x, y, c) = undef(); + output(x, y, order(x, y, 0)) = curved(x, y, 0); + output(x, y, order(x, y, 1)) = curved(x, y, 1); + output(x, y, order(x, y, 2)) = curved(x, y, 2); + + return output; } - diff --git a/filmulator-gui/Halide/calcLayerMix.cpp b/filmulator-gui/Halide/calcLayerMix.cpp index 5d9b9512..3c9dc41b 100644 --- a/filmulator-gui/Halide/calcLayerMix.cpp +++ b/filmulator-gui/Halide/calcLayerMix.cpp @@ -1,45 +1,49 @@ -#include #include +#include #include using namespace Halide; -//using namespace Halide::BoundaryConditions; +// using namespace Halide::BoundaryConditions; using namespace std; using Halide::Image; -#include "image_io.h" #include "halideFilmulate.h" +#include "image_io.h" Var x, y, c; -Func calcLayerMix(Func developer_concentration, Expr layer_mix_const, Expr timestep, - Expr layer_time_divisor, Expr reservoir_developer_concentration){ - Expr layer_mix = pow(layer_mix_const,timestep/layer_time_divisor); - - Expr reservoir_portion = (1 - layer_mix) * reservoir_developer_concentration; +Func calcLayerMix(Func developer_concentration, + Expr layer_mix_const, + Expr timestep, + Expr layer_time_divisor, + Expr reservoir_developer_concentration) +{ + Expr layer_mix = pow(layer_mix_const, timestep / layer_time_divisor); + + Expr reservoir_portion = (1 - layer_mix) * reservoir_developer_concentration; - Func output; - output(x,y) = developer_concentration(x,y) * (layer_mix - 1) + reservoir_portion; - return output; + Func output; + output(x, y) = developer_concentration(x, y) * (layer_mix - 1) + reservoir_portion; + return output; } -int main(int argc, char **argv) { - - Param reservoirConcentration; - Param stepTime; - Param layerMixConst; - Param layerTimeDivisor; - - Func sumDx; - Func layerMixed; - Func initialDeveloperMirrored; - ImageParam devConc(type_of(),2); - Func dDevelConc; - Func developerConcentration = lambda(x,y,devConc(x,y)); - dDevelConc = calcLayerMix(developerConcentration, layerMixConst, stepTime, - layerTimeDivisor, reservoirConcentration); - std::vector ddcArgs = dDevelConc.infer_arguments(); - dDevelConc.compile_to_file("calcLayerMix",ddcArgs); - - return 0; +int main(int argc, char **argv) +{ + + Param reservoirConcentration; + Param stepTime; + Param layerMixConst; + Param layerTimeDivisor; + + Func sumDx; + Func layerMixed; + Func initialDeveloperMirrored; + ImageParam devConc(type_of(), 2); + Func dDevelConc; + Func developerConcentration = lambda(x, y, devConc(x, y)); + dDevelConc = calcLayerMix(developerConcentration, layerMixConst, stepTime, layerTimeDivisor, reservoirConcentration); + std::vector ddcArgs = dDevelConc.infer_arguments(); + dDevelConc.compile_to_file("calcLayerMix", ddcArgs); + + return 0; } diff --git a/filmulator-gui/Halide/calcReservoirConcentration.cpp b/filmulator-gui/Halide/calcReservoirConcentration.cpp index a52190cb..f85cb210 100644 --- a/filmulator-gui/Halide/calcReservoirConcentration.cpp +++ b/filmulator-gui/Halide/calcReservoirConcentration.cpp @@ -1,36 +1,38 @@ -#include #include +#include #include using namespace Halide; -//using namespace Halide::BoundaryConditions; +// using namespace Halide::BoundaryConditions; using namespace std; using Halide::Image; -#include "image_io.h" #include "halideFilmulate.h" +#include "image_io.h" Var x, y, c; -int main(int argc, char **argv) { +int main(int argc, char **argv) +{ - Param reservoirConcentration; - Param reservoirThickness; - Param activeLayerThickness; - Param filmArea; + Param reservoirConcentration; + Param reservoirThickness; + Param activeLayerThickness; + Param filmArea; - ImageParam devMoved(type_of(),2); - Func developerMoved = lambda(x,y,devMoved(x,y)); - Expr pixelsPerMillimeter = sqrt(devMoved.width()*devMoved.height()/filmArea); - RDom r(0 ,devMoved.width(), 0, devMoved.height()); - Func sumD; - sumD(x) = 0.0f; - sumD(0) += developerMoved(r.x,r.y); - Func newResConc; - newResConc(x) = undef(); - newResConc(0) = reservoirConcentration - sumD(0)*activeLayerThickness/(pow(pixelsPerMillimeter,2)*reservoirThickness); - std::vector newResConcArgs = newResConc.infer_arguments(); - newResConc.compile_to_file("calcReservoirConcentration",newResConcArgs); + ImageParam devMoved(type_of(), 2); + Func developerMoved = lambda(x, y, devMoved(x, y)); + Expr pixelsPerMillimeter = sqrt(devMoved.width() * devMoved.height() / filmArea); + RDom r(0, devMoved.width(), 0, devMoved.height()); + Func sumD; + sumD(x) = 0.0f; + sumD(0) += developerMoved(r.x, r.y); + Func newResConc; + newResConc(x) = undef(); + newResConc(0) = + reservoirConcentration - sumD(0) * activeLayerThickness / (pow(pixelsPerMillimeter, 2) * reservoirThickness); + std::vector newResConcArgs = newResConc.infer_arguments(); + newResConc.compile_to_file("calcReservoirConcentration", newResConcArgs); - return 0; + return 0; } diff --git a/filmulator-gui/Halide/demosaic.cpp b/filmulator-gui/Halide/demosaic.cpp index 040788a6..0b5724bb 100644 --- a/filmulator-gui/Halide/demosaic.cpp +++ b/filmulator-gui/Halide/demosaic.cpp @@ -1,469 +1,462 @@ -//Compile with: -//g++ demosaic.cpp -g -I include/ -L bin/ -lHalide `libpng-config --cflags --ldflags` -lpthread -ldl -o demosaic -std=c++11 +// Compile with: +// g++ demosaic.cpp -g -I include/ -L bin/ -lHalide `libpng-config --cflags --ldflags` -lpthread -ldl -o demosaic +// -std=c++11 // -//Run with: -//LD_LIBRARY_PATH=bin ./demosaic -//For debug -//LD_LIBRARY_PATH=bin HL_DEBUG_CODEGEN=[#] ./demosaic +// Run with: +// LD_LIBRARY_PATH=bin ./demosaic +// For debug +// LD_LIBRARY_PATH=bin HL_DEBUG_CODEGEN=[#] ./demosaic // -//g++ demosaic.cpp -g -I include/ -L bin/ -lHalide `libpng-config --cflags --ldflags` -lpthread -ldl -o demosaic -std=c++11 && LD_LIBRARY_PATH=bin HL_DEBUG_CODEGEN=0 ./demosaic +// g++ demosaic.cpp -g -I include/ -L bin/ -lHalide `libpng-config --cflags --ldflags` -lpthread -ldl -o demosaic +// -std=c++11 && LD_LIBRARY_PATH=bin HL_DEBUG_CODEGEN=0 ./demosaic #include -#include #include +#include #include using Halide::Image; #include -//#include +// #include using namespace Halide; Halide::Func bayerize(Func in) { - Func out; - Var x,y,c; - Expr evenRow = (y%2 == 0); - Expr oddRow = (y%2 == 1); - Expr evenCol = (x%2 == 0); - Expr oddCol = (x%2 == 1); - out(x,y) = select( - evenRow, select( - evenCol, - in(x,y,1), - in(x,y,0)), - select( // odd row - evenCol, - in(x,y,2), - in(x,y,1))); - // G R G R - // B G B G - // G R G R - // B G B G - - return out; + Func out; + Var x, y, c; + Expr evenRow = (y % 2 == 0); + Expr oddRow = (y % 2 == 1); + Expr evenCol = (x % 2 == 0); + Expr oddCol = (x % 2 == 1); + out(x, y) = select(evenRow, + select(evenCol, in(x, y, 1), in(x, y, 0)), + select(// odd row + evenCol, + in(x, y, 2), + in(x, y, 1))); + // G R G R + // B G B G + // G R G R + // B G B G + + return out; } Halide::Func blurRatio_v(Func vert) { - Var x,y; - //Low pass filter (sigma=2 L=4) - Expr h0, h1, h2, h3, h4, hsum; - h0 = .203125f; - h1 = .1796875f; - h2 = .1171875f; - h3 = .0703125f; - h4 = .03125f; - - - Func out; - out(x,y) = h0 * vert(x,y) + - h1 * (vert(x,y-1) + vert(x,y+1)) + - h2 * (vert(x,y-2) + vert(x,y+2)) + - h3 * (vert(x,y-3) + vert(x,y+3)) + - h4 * (vert(x,y-4) + vert(x,y+4)); - return out; + Var x, y; + // Low pass filter (sigma=2 L=4) + Expr h0, h1, h2, h3, h4, hsum; + h0 = .203125f; + h1 = .1796875f; + h2 = .1171875f; + h3 = .0703125f; + h4 = .03125f; + + + Func out; + out(x, y) = h0 * vert(x, y) + h1 * (vert(x, y - 1) + vert(x, y + 1)) + h2 * (vert(x, y - 2) + vert(x, y + 2)) + + h3 * (vert(x, y - 3) + vert(x, y + 3)) + h4 * (vert(x, y - 4) + vert(x, y + 4)); + return out; } Halide::Func blurRatio_h(Func hor) { - Var x,y; - //Low pass filter (sigma=2 L=4) - Expr h0, h1, h2, h3, h4, hsum; - h0 = .203125f; - h1 = .1796875f; - h2 = .1171875f; - h3 = .0703125f; - h4 = .03125f; - - Func out; - out(x,y) = h0 * hor(x,y) + - h1 * (hor(x-1,y) + hor(x+1,y)) + - h2 * (hor(x-2,y) + hor(x+2,y)) + - h3 * (hor(x-3,y) + hor(x+3,y)) + - h4 * (hor(x-4,y) + hor(x+4,y)); - return out; + Var x, y; + // Low pass filter (sigma=2 L=4) + Expr h0, h1, h2, h3, h4, hsum; + h0 = .203125f; + h1 = .1796875f; + h2 = .1171875f; + h3 = .0703125f; + h4 = .03125f; + + Func out; + out(x, y) = h0 * hor(x, y) + h1 * (hor(x - 1, y) + hor(x + 1, y)) + h2 * (hor(x - 2, y) + hor(x + 2, y)) + + h3 * (hor(x - 3, y) + hor(x + 3, y)) + h4 * (hor(x - 4, y) + hor(x + 4, y)); + return out; } -Tuple swap2(Expr a, Expr b) { - return Tuple(min(a,b), max(a,b)); -} +Tuple swap2(Expr a, Expr b) { return Tuple(min(a, b), max(a, b)); } -Tuple sort3(Expr a, Expr b, Expr c) { - Tuple x = swap2(a,b); - Tuple y = swap2(x[1],c); - Tuple z = swap2(x[0],y[1]); - return Tuple(z[0], z[1], y[1]); +Tuple sort3(Expr a, Expr b, Expr c) +{ + Tuple x = swap2(a, b); + Tuple y = swap2(x[1], c); + Tuple z = swap2(x[0], y[1]); + return Tuple(z[0], z[1], y[1]); } Halide::Func demosaic(Func deinterleaved) { - Func output; - //A large part of the algorithm is spent processing vertical and horizontal separately. - //This means that we can duplicate the original data into vertical and horizontal - //And run the horizontal actually vertically in memory - //Then we can vectorize very easily. - - Var x, y, c; - Var xo, yo, xi, yi, tile_index; - //Optimization variables - - //Group the pixels into fours. - Func r_r, g_gr, g_gb, b_b; - - Func wb; wb(x) = 1; - g_gr(x, y) = deinterleaved(2*x ,2*y )*wb(0) + 0.01f; - r_r(x, y) = deinterleaved(2*x+1,2*y )*wb(1) + 0.01f; - b_b(x, y) = deinterleaved(2*x ,2*y+1)*wb(2) + 0.01f; - g_gb(x, y) = deinterleaved(2*x+1,2*y+1)*wb(3) + 0.01f; - - // G R G R - // B G B G - // G R G R - // B G B G - - //Initial demosaic: - //We need to make this bilinear, and sharpen the estimated colors at the end. - // - //The paper uses an implementation that sharpens the off colors first, but that - //violates their assumption of smooth color transitions and worsens performance. - - - //First calculate the green at the red and blue pixels. - - //Red pixels - Func gAtR_v, gAtR_h; - gAtR_h(x,y) = (g_gr(x,y) + g_gr(x+1,y))/2.0f; - gAtR_v(x,y) = (g_gb(x,y-1) + g_gb(x,y))/2.0f; - //Blue pixels - Func gAtB_v, gAtB_h; - gAtB_h(x,y) = (g_gb(x-1,y) + g_gb(x,y))/2.0f; - gAtB_v(x,y) = (g_gr(x,y) + g_gr(x,y+1))/2.0f; - - //Next, calculate the red and blue at the green pixels. - - //Red rows - Func rAtGR_h, bAtGR_v; - rAtGR_h(x,y) = (r_r(x-1,y) + r_r(x,y))/2.0f; - bAtGR_v(x,y) = (b_b(x,y-1) + b_b(x,y))/2.0f; - //Blue rows - Func bAtGB_h, rAtGB_v; - bAtGB_h(x,y) = (b_b(x,y) + b_b(x+1,y))/2.0f; - rAtGB_v(x,y) = (r_r(x,y) + r_r(x,y+1))/2.0f; - - - //Get the logs of the color ratios - - //On red pixels - Func grRatioAtR_h, grRatioAtR_v; - grRatioAtR_h(x,y) = gAtR_h(x,y)/r_r(x,y); - grRatioAtR_v(x,y) = gAtR_v(x,y)/r_r(x,y); - //On blue pixels - Func gbRatioAtB_h, gbRatioAtB_v; - gbRatioAtB_h(x,y) = gAtB_h(x,y)/b_b(x,y); - gbRatioAtB_v(x,y) = gAtB_v(x,y)/b_b(x,y); - //On green pixels in red rows - Func grRatioAtGR_h, gbRatioAtGR_v; - grRatioAtGR_h(x,y) = g_gr(x,y)/rAtGR_h(x,y); - gbRatioAtGR_v(x,y) = g_gr(x,y)/bAtGR_v(x,y); - //On green pixels in blue rows - Func gbRatioAtGB_h, grRatioAtGB_v; - gbRatioAtGB_h(x,y) = g_gb(x,y)/bAtGB_h(x,y); - grRatioAtGB_v(x,y) = g_gb(x,y)/rAtGB_v(x,y); - - - //Blur the color ratios, to estimate the actual color ratio. - //These are 1d blurs. - //Vertical is blurred vertically. - //Horizontal is blurred horizontally. - //It just happens to work out. - - //Combined color ratios - Func colorRatios_h, colorRatios_v; - colorRatios_h(x,y) = select(y%2 == 0,//Rows - select(x%2 == 0,//Red row - grRatioAtGR_h(x/2, y/2),//Green in red row - grRatioAtR_h(x/2, y/2)),//Red - select(x%2 == 0,//Blue row - gbRatioAtB_h(x/2, y/2),//Blue - gbRatioAtGB_h(x/2,y/2)));//Green in blue row - colorRatios_v(x,y) = select(y%2 == 0,//Rows - select(x%2 == 0,//Red row - gbRatioAtGR_v(x/2, y/2),//Green in red row - grRatioAtR_v(x/2, y/2)),//Red - select(x%2 == 0,//Blue row - gbRatioAtB_v(x/2, y/2),//Blue - grRatioAtGB_v(x/2,y/2)));//Green in blue row - - //Now we take the logs of them. - - //First is the log of the color ratios. - //This is the Observed Color Difference Y. - Func Y_h, Y_v; - Y_h(x,y) = log(colorRatios_h(x,y)); - Y_v(x,y) = log(colorRatios_v(x,y)); - - //Next is the blurred log of the color ratios. - //This is the Estimated True Color Difference Ys ~~= X - Func Ys_h, Ys_v; - Ys_h = blurRatio_h(Y_h); - Ys_v = blurRatio_v(Y_v); - - //To get a Linear Minimum Mean Square Error (LMMSE) estimate, - //we want - //xhat = E[x] + cov(x,y)/var(y) * (y - E[y]) - // - //Empirically, since we don't all of these, it becomes (locally - //xhat = mean(x) + var(x)/(var(x) + var(%nu)) * (y - mean(x)) - // - //x is approximated by Ys; it's low-pass. - //nu, the error, is approximated by (Ys - Y); it's high-pass. - Func X_h, X_v; - X_h(x,y) = Ys_h(x,y); - X_v(x,y) = Ys_v(x,y); - - //Neighborhood mean of X - Func MUx_h, MUx_v; - Func momh, momv; - RDom r1(-4, 9); + Func output; + // A large part of the algorithm is spent processing vertical and horizontal separately. + // This means that we can duplicate the original data into vertical and horizontal + // And run the horizontal actually vertically in memory + // Then we can vectorize very easily. + + Var x, y, c; + Var xo, yo, xi, yi, tile_index; + // Optimization variables + + // Group the pixels into fours. + Func r_r, g_gr, g_gb, b_b; + + Func wb; + wb(x) = 1; + g_gr(x, y) = deinterleaved(2 * x, 2 * y) * wb(0) + 0.01f; + r_r(x, y) = deinterleaved(2 * x + 1, 2 * y) * wb(1) + 0.01f; + b_b(x, y) = deinterleaved(2 * x, 2 * y + 1) * wb(2) + 0.01f; + g_gb(x, y) = deinterleaved(2 * x + 1, 2 * y + 1) * wb(3) + 0.01f; + + // G R G R + // B G B G + // G R G R + // B G B G + + // Initial demosaic: + // We need to make this bilinear, and sharpen the estimated colors at the end. + // + // The paper uses an implementation that sharpens the off colors first, but that + // violates their assumption of smooth color transitions and worsens performance. + + + // First calculate the green at the red and blue pixels. + + // Red pixels + Func gAtR_v, gAtR_h; + gAtR_h(x, y) = (g_gr(x, y) + g_gr(x + 1, y)) / 2.0f; + gAtR_v(x, y) = (g_gb(x, y - 1) + g_gb(x, y)) / 2.0f; + // Blue pixels + Func gAtB_v, gAtB_h; + gAtB_h(x, y) = (g_gb(x - 1, y) + g_gb(x, y)) / 2.0f; + gAtB_v(x, y) = (g_gr(x, y) + g_gr(x, y + 1)) / 2.0f; + + // Next, calculate the red and blue at the green pixels. + + // Red rows + Func rAtGR_h, bAtGR_v; + rAtGR_h(x, y) = (r_r(x - 1, y) + r_r(x, y)) / 2.0f; + bAtGR_v(x, y) = (b_b(x, y - 1) + b_b(x, y)) / 2.0f; + // Blue rows + Func bAtGB_h, rAtGB_v; + bAtGB_h(x, y) = (b_b(x, y) + b_b(x + 1, y)) / 2.0f; + rAtGB_v(x, y) = (r_r(x, y) + r_r(x, y + 1)) / 2.0f; + + + // Get the logs of the color ratios + + // On red pixels + Func grRatioAtR_h, grRatioAtR_v; + grRatioAtR_h(x, y) = gAtR_h(x, y) / r_r(x, y); + grRatioAtR_v(x, y) = gAtR_v(x, y) / r_r(x, y); + // On blue pixels + Func gbRatioAtB_h, gbRatioAtB_v; + gbRatioAtB_h(x, y) = gAtB_h(x, y) / b_b(x, y); + gbRatioAtB_v(x, y) = gAtB_v(x, y) / b_b(x, y); + // On green pixels in red rows + Func grRatioAtGR_h, gbRatioAtGR_v; + grRatioAtGR_h(x, y) = g_gr(x, y) / rAtGR_h(x, y); + gbRatioAtGR_v(x, y) = g_gr(x, y) / bAtGR_v(x, y); + // On green pixels in blue rows + Func gbRatioAtGB_h, grRatioAtGB_v; + gbRatioAtGB_h(x, y) = g_gb(x, y) / bAtGB_h(x, y); + grRatioAtGB_v(x, y) = g_gb(x, y) / rAtGB_v(x, y); + + + // Blur the color ratios, to estimate the actual color ratio. + // These are 1d blurs. + // Vertical is blurred vertically. + // Horizontal is blurred horizontally. + // It just happens to work out. + + // Combined color ratios + Func colorRatios_h, colorRatios_v; + colorRatios_h(x, y) = select(y % 2 == 0,// Rows + select(x % 2 == 0,// Red row + grRatioAtGR_h(x / 2, y / 2),// Green in red row + grRatioAtR_h(x / 2, y / 2)),// Red + select(x % 2 == 0,// Blue row + gbRatioAtB_h(x / 2, y / 2),// Blue + gbRatioAtGB_h(x / 2, y / 2)));// Green in blue row + colorRatios_v(x, y) = select(y % 2 == 0,// Rows + select(x % 2 == 0,// Red row + gbRatioAtGR_v(x / 2, y / 2),// Green in red row + grRatioAtR_v(x / 2, y / 2)),// Red + select(x % 2 == 0,// Blue row + gbRatioAtB_v(x / 2, y / 2),// Blue + grRatioAtGB_v(x / 2, y / 2)));// Green in blue row + + // Now we take the logs of them. + + // First is the log of the color ratios. + // This is the Observed Color Difference Y. + Func Y_h, Y_v; + Y_h(x, y) = log(colorRatios_h(x, y)); + Y_v(x, y) = log(colorRatios_v(x, y)); + + // Next is the blurred log of the color ratios. + // This is the Estimated True Color Difference Ys ~~= X + Func Ys_h, Ys_v; + Ys_h = blurRatio_h(Y_h); + Ys_v = blurRatio_v(Y_v); + + // To get a Linear Minimum Mean Square Error (LMMSE) estimate, + // we want + // xhat = E[x] + cov(x,y)/var(y) * (y - E[y]) + // + // Empirically, since we don't all of these, it becomes (locally + // xhat = mean(x) + var(x)/(var(x) + var(%nu)) * (y - mean(x)) + // + // x is approximated by Ys; it's low-pass. + // nu, the error, is approximated by (Ys - Y); it's high-pass. + Func X_h, X_v; + X_h(x, y) = Ys_h(x, y); + X_v(x, y) = Ys_v(x, y); + + // Neighborhood mean of X + Func MUx_h, MUx_v; + Func momh, momv; + RDom r1(-4, 9); #define DOMDIV 9.0f - momh(x,y) = 0.0f; - momh(x,y) += X_h(r1+x,y); - momv(x,y) = 0.0f; - momv(x,y) += X_v(x,r1+y); - - MUx_h(x,y) = momh(x,y)/DOMDIV; - MUx_v(x,y) = momv(x,y)/DOMDIV; - - //Confirmed to be exactly the same ========================================= - - //Neighborhood variance of X - Func SIGMAx_h, SIGMAx_v; - Func ph,pv;//sums of squares - RDom r2(-4,9); - ph(x,y) = 0.0f; - ph(x,y) += X_h(r2+x,y)*X_h(r2+x,y); - pv(x,y) = 0.0f; - pv(x,y) += X_v(x,r2+y)*X_v(x,r2+y); - SIGMAx_h(x,y) = ph(x,y) / 8.0f - momh(x,y)*momh(x,y)/(8.0f*9.0f); - SIGMAx_v(x,y) = pv(x,y) / 8.0f - momv(x,y)*momv(x,y)/(8.0f*9.0f); - //Confirmed error < 2e-9; absolute values peak at around 1e-3 - - //Neighborhood variance of nu - Func SIGMAnu_h, SIGMAnu_v; - RDom r3(-4,9); - SIGMAnu_h(x,y) = 0.0f; - SIGMAnu_h(x,y) += (X_h(r3+x,y) - Y_h(r3+x,y))*(X_h(r3+x,y) - Y_h(r3+x,y)) / DOMDIV; - SIGMAnu_v(x,y) = 0.0f; - SIGMAnu_v(x,y) += (X_v(x,r3+y) - Y_v(x,r3+y))*(X_v(x,r3+y) - Y_v(x,r3+y)) / DOMDIV; - - //LMMSE estimation in each direction - Func Xlmmse_h, Xlmmse_v; - Xlmmse_h(x,y) = MUx_h(x,y) + - (Y_h(x,y) - MUx_h(x,y)) * SIGMAx_h(x,y) / - (SIGMAx_h(x,y) + SIGMAnu_h(x,y) + 1e-7f); - Xlmmse_v(x,y) = MUx_v(x,y) + - (Y_v(x,y) - MUx_v(x,y)) * SIGMAx_v(x,y) / - (SIGMAx_v(x,y) + SIGMAnu_v(x,y) + 1e-7f); - - //Confirmed to be correct? ====================================================== - - //The expected estimation error is Xerror = X - Xlmmse - //We don't use it. - - //The variance of the estimation error, we do use for the weighting. - Func SIGMAer_h, SIGMAer_v; - SIGMAer_h(x,y) = SIGMAx_h(x,y) - SIGMAx_h(x,y)*SIGMAx_h(x,y)/(SIGMAx_h(x,y) + SIGMAnu_h(x,y) + 1e-7f); - SIGMAer_v(x,y) = SIGMAx_v(x,y) - SIGMAx_v(x,y)*SIGMAx_v(x,y)/(SIGMAx_v(x,y) + SIGMAnu_v(x,y) + 1e-7f); - - //Weight of estimate - Func W_h, W_v; - W_h(x,y) = SIGMAer_v(x,y) / (SIGMAer_h(x,y) + SIGMAer_v(x,y) + 1e-7f); - W_v(x,y) = 1.0f - W_h(x,y); - - //slightly different ============================================= - - //Combine to get the final log of the color ratio we'll use - Func X; - X(x,y) = W_h(x,y)*Xlmmse_h(x,y) + W_v(x,y)*Xlmmse_v(x,y); - - //Separate the green/color ratios back out. - //They're only valid on red and blue pixels. - //Reminder: these are logs, not actually the ratios yet. - Func grLogRatioAtR, gbLogRatioAtB; - grLogRatioAtR(x,y) = X(2*x+1, 2*y+0); - gbLogRatioAtB(x,y) = X(2*x+0, 2*y+1); - - //Compute the green/color ratios at the opposite colors. - //It's just the mean of the color ratios of the proper colors nearby. - //Once more, these are still logs. - Func gbLogRatioAtR, grLogRatioAtB; - gbLogRatioAtR(x,y) = (gbLogRatioAtB(x,y) + gbLogRatioAtB(x,y-1) + - gbLogRatioAtB(x+1,y-1) + gbLogRatioAtB(x+1,y)) / 4.0f; - grLogRatioAtB(x,y) = (grLogRatioAtR(x,y) + grLogRatioAtR(x,y+1) + - grLogRatioAtR(x-1,y+1) + grLogRatioAtR(x-1,y)) / 4.0f; - - Func logRatiosRB; - logRatiosRB(x,y) = Tuple(grLogRatioAtR(x,y),gbLogRatioAtB(x,y), - gbLogRatioAtR(x,y),grLogRatioAtB(x,y)); - Func grLogRatioAtRtup,grLogRatioAtBtup,gbLogRatioAtRtup,gbLogRatioAtBtup; - grLogRatioAtRtup(x,y) = logRatiosRB(x,y)[0]; - gbLogRatioAtBtup(x,y) = logRatiosRB(x,y)[1]; - gbLogRatioAtRtup(x,y) = logRatiosRB(x,y)[2]; - grLogRatioAtBtup(x,y) = logRatiosRB(x,y)[3]; - - //Compute the color ratios at green - //Again, still logs. - Func grLogRatioAtGR, gbLogRatioAtGR, grLogRatioAtGB, gbLogRatioAtGB; - grLogRatioAtGR(x,y) = (grLogRatioAtRtup(x-1,y) + grLogRatioAtRtup(x,y) + - grLogRatioAtBtup(x,y-1) + grLogRatioAtBtup(x,y)) / 4.0f; - grLogRatioAtGB(x,y) = (grLogRatioAtRtup(x,y) + grLogRatioAtRtup(x,y+1) + - grLogRatioAtBtup(x,y) + grLogRatioAtBtup(x+1,y)) / 4.0f; - gbLogRatioAtGR(x,y) = (gbLogRatioAtRtup(x-1,y) + gbLogRatioAtRtup(x,y) + - gbLogRatioAtBtup(x,y-1) + gbLogRatioAtBtup(x,y)) / 4.0f; - gbLogRatioAtGB(x,y) = (gbLogRatioAtRtup(x,y) + gbLogRatioAtRtup(x,y+1) + - gbLogRatioAtBtup(x,y) + gbLogRatioAtBtup(x+1,y)) / 4.0f; - - //The libraw lmmse does a median filter...why? - //Doesn't seem to do anything - - //First we must combine these ratios into one. - Func grLogRatio, gbLogRatio; - grLogRatio(x,y) = select(y%2 == 0, - select(x%2 == 0, - //Green pixel in red row - grLogRatioAtGR(x/2,y/2), - //Red pixel - grLogRatioAtR(x/2,y/2)), - select(x%2 == 0, - //Blue pixel - grLogRatioAtB(x/2,y/2), - //Green pixel in blue row - grLogRatioAtGB(x/2,y/2))); - gbLogRatio(x,y) = select(y%2 == 0, - select(x%2 == 0, - //Green pixel in red row - gbLogRatioAtGR(x/2,y/2), - //Red pixel - gbLogRatioAtR(x/2,y/2)), - select(x%2 == 0, - //Blue pixel - gbLogRatioAtB(x/2,y/2), - //Green pixel in blue row - gbLogRatioAtGB(x/2,y/2))); - - Func logRatios; - logRatios(x,y) = Tuple(grLogRatio(x,y),gbLogRatio(x,y)); - - //Turn the logs back into ratios, after the median filter - Func grRatio, gbRatio; - - //No median, but do turn the logs back into ratios. - grRatio(x,y) = exp(logRatios(x,y)[0]); - gbRatio(x,y) = exp(logRatios(x,y)[1]); - - //Output the values we want. - output(x,y,c) = select(c == 0, - //Red channel - select(y%2 == 0, - select(x%2 == 0, - //Green pixel in red row - g_gr(x/2,y/2) / grRatio(x,y), - //Red pixel - r_r(x/2,y/2)), - select(x%2 == 0, - //Blue pixel - b_b(x/2,y/2) * gbRatio(x,y) / grRatio(x,y), - //Green pixel in blue row - g_gb(x/2,y/2) / grRatio(x,y))), - select(c == 1, - //Green channel - select(y%2 == 0, - select(x%2 == 0, - //Green pixel in red row - g_gr(x/2,y/2), - //Red pixel - r_r(x/2,y/2) * grRatio(x,y)), - select(x%2 == 0, - //Blue pixel - b_b(x/2,y/2) * gbRatio(x,y), - //Green pixel in blue row - g_gb(x/2,y/2))), - //Blue channel - select(y%2 == 0, - select(x%2 == 0, - //Green pixel in red row - g_gr(x/2,y/2) / gbRatio(x,y), - //Red pixel - r_r(x/2,y/2) * grRatio(x,y) / gbRatio(x,y)), - select(x%2 == 0, - //Blue pixel - b_b(x/2,y/2), - //Green pixel in blue row - g_gb(x/2,y/2) / gbRatio(x,y))))) - 0.01f; - - - Y_h.compute_root().parallel(y); - Y_v.compute_root().split(x,xo,xi,16).parallel(xo).vectorize(xi,8); - X_h.compute_root().parallel(y); - X_v.compute_root().split(x,xo,xi,16).parallel(xo).vectorize(xi,8); - MUx_v.compute_root().split(x,xo,xi,16).parallel(xo).vectorize(xi,8); - MUx_h.compute_root().parallel(y); - SIGMAx_v.compute_root().split(x,xo,xi,16).parallel(xo).vectorize(xi,8); - SIGMAx_h.compute_root().parallel(y); - SIGMAnu_v.compute_root().split(x,xo,xi,16).parallel(xo).vectorize(xi,8); - SIGMAnu_h.compute_root().parallel(y); - //Xlmmse_v.split(x,xo,xi,16).parallel(xo).vectorize(xi,4); - //SIGMAer_v.split(x,xo,xi,16).parallel(xo).vectorize(xi,4); - //W_v.split(x,xo,xi,16).parallel(xo).vectorize(xi,4); - - //X.compute_root().parallel(y); - //grLogRatioAtR.store_at(output,tile_index).compute_at(grLogRatio,x); - //gbLogRatioAtB.store_at(output,tile_index).compute_at(gbLogRatio,x); - //gbLogRatioAtR.store_at(output,tile_index).compute_at(gbLogRatio,x); - //grLogRatioAtB.store_at(output,tile_index).compute_at(grLogRatio,x); - logRatiosRB.store_at(logRatios,xo).compute_at(logRatios,xi); - //grLogRatio.compute_at(output,tile_index); - //gbLogRatio.compute_at(output,tile_index); - logRatios.compute_at(output,tile_index).split(x,xo,xi,2).unroll(xi); - - output.tile(x,y,xo,yo,xi,yi,64,64) - .fuse(xo,yo,tile_index) - .parallel(tile_index) - .vectorize(xi,8) - .bound(c,0,3).unroll(c).reorder(c,xi,yi,tile_index).reorder_storage(c,x,y) - .compute_root();//.compile_to_lowered_stmt("output_unrolledC.html",HTML); - - return output; - - //Time beat: 2.25 seconds + momh(x, y) = 0.0f; + momh(x, y) += X_h(r1 + x, y); + momv(x, y) = 0.0f; + momv(x, y) += X_v(x, r1 + y); + + MUx_h(x, y) = momh(x, y) / DOMDIV; + MUx_v(x, y) = momv(x, y) / DOMDIV; + + // Confirmed to be exactly the same ========================================= + + // Neighborhood variance of X + Func SIGMAx_h, SIGMAx_v; + Func ph, pv;// sums of squares + RDom r2(-4, 9); + ph(x, y) = 0.0f; + ph(x, y) += X_h(r2 + x, y) * X_h(r2 + x, y); + pv(x, y) = 0.0f; + pv(x, y) += X_v(x, r2 + y) * X_v(x, r2 + y); + SIGMAx_h(x, y) = ph(x, y) / 8.0f - momh(x, y) * momh(x, y) / (8.0f * 9.0f); + SIGMAx_v(x, y) = pv(x, y) / 8.0f - momv(x, y) * momv(x, y) / (8.0f * 9.0f); + // Confirmed error < 2e-9; absolute values peak at around 1e-3 + + // Neighborhood variance of nu + Func SIGMAnu_h, SIGMAnu_v; + RDom r3(-4, 9); + SIGMAnu_h(x, y) = 0.0f; + SIGMAnu_h(x, y) += (X_h(r3 + x, y) - Y_h(r3 + x, y)) * (X_h(r3 + x, y) - Y_h(r3 + x, y)) / DOMDIV; + SIGMAnu_v(x, y) = 0.0f; + SIGMAnu_v(x, y) += (X_v(x, r3 + y) - Y_v(x, r3 + y)) * (X_v(x, r3 + y) - Y_v(x, r3 + y)) / DOMDIV; + + // LMMSE estimation in each direction + Func Xlmmse_h, Xlmmse_v; + Xlmmse_h(x, y) = + MUx_h(x, y) + (Y_h(x, y) - MUx_h(x, y)) * SIGMAx_h(x, y) / (SIGMAx_h(x, y) + SIGMAnu_h(x, y) + 1e-7f); + Xlmmse_v(x, y) = + MUx_v(x, y) + (Y_v(x, y) - MUx_v(x, y)) * SIGMAx_v(x, y) / (SIGMAx_v(x, y) + SIGMAnu_v(x, y) + 1e-7f); + + // Confirmed to be correct? ====================================================== + + // The expected estimation error is Xerror = X - Xlmmse + // We don't use it. + + // The variance of the estimation error, we do use for the weighting. + Func SIGMAer_h, SIGMAer_v; + SIGMAer_h(x, y) = SIGMAx_h(x, y) - SIGMAx_h(x, y) * SIGMAx_h(x, y) / (SIGMAx_h(x, y) + SIGMAnu_h(x, y) + 1e-7f); + SIGMAer_v(x, y) = SIGMAx_v(x, y) - SIGMAx_v(x, y) * SIGMAx_v(x, y) / (SIGMAx_v(x, y) + SIGMAnu_v(x, y) + 1e-7f); + + // Weight of estimate + Func W_h, W_v; + W_h(x, y) = SIGMAer_v(x, y) / (SIGMAer_h(x, y) + SIGMAer_v(x, y) + 1e-7f); + W_v(x, y) = 1.0f - W_h(x, y); + + // slightly different ============================================= + + // Combine to get the final log of the color ratio we'll use + Func X; + X(x, y) = W_h(x, y) * Xlmmse_h(x, y) + W_v(x, y) * Xlmmse_v(x, y); + + // Separate the green/color ratios back out. + // They're only valid on red and blue pixels. + // Reminder: these are logs, not actually the ratios yet. + Func grLogRatioAtR, gbLogRatioAtB; + grLogRatioAtR(x, y) = X(2 * x + 1, 2 * y + 0); + gbLogRatioAtB(x, y) = X(2 * x + 0, 2 * y + 1); + + // Compute the green/color ratios at the opposite colors. + // It's just the mean of the color ratios of the proper colors nearby. + // Once more, these are still logs. + Func gbLogRatioAtR, grLogRatioAtB; + gbLogRatioAtR(x, y) = + (gbLogRatioAtB(x, y) + gbLogRatioAtB(x, y - 1) + gbLogRatioAtB(x + 1, y - 1) + gbLogRatioAtB(x + 1, y)) / 4.0f; + grLogRatioAtB(x, y) = + (grLogRatioAtR(x, y) + grLogRatioAtR(x, y + 1) + grLogRatioAtR(x - 1, y + 1) + grLogRatioAtR(x - 1, y)) / 4.0f; + + Func logRatiosRB; + logRatiosRB(x, y) = Tuple(grLogRatioAtR(x, y), gbLogRatioAtB(x, y), gbLogRatioAtR(x, y), grLogRatioAtB(x, y)); + Func grLogRatioAtRtup, grLogRatioAtBtup, gbLogRatioAtRtup, gbLogRatioAtBtup; + grLogRatioAtRtup(x, y) = logRatiosRB(x, y)[0]; + gbLogRatioAtBtup(x, y) = logRatiosRB(x, y)[1]; + gbLogRatioAtRtup(x, y) = logRatiosRB(x, y)[2]; + grLogRatioAtBtup(x, y) = logRatiosRB(x, y)[3]; + + // Compute the color ratios at green + // Again, still logs. + Func grLogRatioAtGR, gbLogRatioAtGR, grLogRatioAtGB, gbLogRatioAtGB; + grLogRatioAtGR(x, y) = + (grLogRatioAtRtup(x - 1, y) + grLogRatioAtRtup(x, y) + grLogRatioAtBtup(x, y - 1) + grLogRatioAtBtup(x, y)) / 4.0f; + grLogRatioAtGB(x, y) = + (grLogRatioAtRtup(x, y) + grLogRatioAtRtup(x, y + 1) + grLogRatioAtBtup(x, y) + grLogRatioAtBtup(x + 1, y)) / 4.0f; + gbLogRatioAtGR(x, y) = + (gbLogRatioAtRtup(x - 1, y) + gbLogRatioAtRtup(x, y) + gbLogRatioAtBtup(x, y - 1) + gbLogRatioAtBtup(x, y)) / 4.0f; + gbLogRatioAtGB(x, y) = + (gbLogRatioAtRtup(x, y) + gbLogRatioAtRtup(x, y + 1) + gbLogRatioAtBtup(x, y) + gbLogRatioAtBtup(x + 1, y)) / 4.0f; + + // The libraw lmmse does a median filter...why? + // Doesn't seem to do anything + + // First we must combine these ratios into one. + Func grLogRatio, gbLogRatio; + grLogRatio(x, y) = select(y % 2 == 0, + select(x % 2 == 0, + // Green pixel in red row + grLogRatioAtGR(x / 2, y / 2), + // Red pixel + grLogRatioAtR(x / 2, y / 2)), + select(x % 2 == 0, + // Blue pixel + grLogRatioAtB(x / 2, y / 2), + // Green pixel in blue row + grLogRatioAtGB(x / 2, y / 2))); + gbLogRatio(x, y) = select(y % 2 == 0, + select(x % 2 == 0, + // Green pixel in red row + gbLogRatioAtGR(x / 2, y / 2), + // Red pixel + gbLogRatioAtR(x / 2, y / 2)), + select(x % 2 == 0, + // Blue pixel + gbLogRatioAtB(x / 2, y / 2), + // Green pixel in blue row + gbLogRatioAtGB(x / 2, y / 2))); + + Func logRatios; + logRatios(x, y) = Tuple(grLogRatio(x, y), gbLogRatio(x, y)); + + // Turn the logs back into ratios, after the median filter + Func grRatio, gbRatio; + + // No median, but do turn the logs back into ratios. + grRatio(x, y) = exp(logRatios(x, y)[0]); + gbRatio(x, y) = exp(logRatios(x, y)[1]); + + // Output the values we want. + output(x, y, c) = select(c == 0, + // Red channel + select(y % 2 == 0, + select(x % 2 == 0, + // Green pixel in red row + g_gr(x / 2, y / 2) / grRatio(x, y), + // Red pixel + r_r(x / 2, y / 2)), + select(x % 2 == 0, + // Blue pixel + b_b(x / 2, y / 2) * gbRatio(x, y) / grRatio(x, y), + // Green pixel in blue row + g_gb(x / 2, y / 2) / grRatio(x, y))), + select(c == 1, + // Green channel + select(y % 2 == 0, + select(x % 2 == 0, + // Green pixel in red row + g_gr(x / 2, y / 2), + // Red pixel + r_r(x / 2, y / 2) * grRatio(x, y)), + select(x % 2 == 0, + // Blue pixel + b_b(x / 2, y / 2) * gbRatio(x, y), + // Green pixel in blue row + g_gb(x / 2, y / 2))), + // Blue channel + select(y % 2 == 0, + select(x % 2 == 0, + // Green pixel in red row + g_gr(x / 2, y / 2) / gbRatio(x, y), + // Red pixel + r_r(x / 2, y / 2) * grRatio(x, y) / gbRatio(x, y)), + select(x % 2 == 0, + // Blue pixel + b_b(x / 2, y / 2), + // Green pixel in blue row + g_gb(x / 2, y / 2) / gbRatio(x, y))))) + - 0.01f; + + + Y_h.compute_root().parallel(y); + Y_v.compute_root().split(x, xo, xi, 16).parallel(xo).vectorize(xi, 8); + X_h.compute_root().parallel(y); + X_v.compute_root().split(x, xo, xi, 16).parallel(xo).vectorize(xi, 8); + MUx_v.compute_root().split(x, xo, xi, 16).parallel(xo).vectorize(xi, 8); + MUx_h.compute_root().parallel(y); + SIGMAx_v.compute_root().split(x, xo, xi, 16).parallel(xo).vectorize(xi, 8); + SIGMAx_h.compute_root().parallel(y); + SIGMAnu_v.compute_root().split(x, xo, xi, 16).parallel(xo).vectorize(xi, 8); + SIGMAnu_h.compute_root().parallel(y); + // Xlmmse_v.split(x,xo,xi,16).parallel(xo).vectorize(xi,4); + // SIGMAer_v.split(x,xo,xi,16).parallel(xo).vectorize(xi,4); + // W_v.split(x,xo,xi,16).parallel(xo).vectorize(xi,4); + + // X.compute_root().parallel(y); + // grLogRatioAtR.store_at(output,tile_index).compute_at(grLogRatio,x); + // gbLogRatioAtB.store_at(output,tile_index).compute_at(gbLogRatio,x); + // gbLogRatioAtR.store_at(output,tile_index).compute_at(gbLogRatio,x); + // grLogRatioAtB.store_at(output,tile_index).compute_at(grLogRatio,x); + logRatiosRB.store_at(logRatios, xo).compute_at(logRatios, xi); + // grLogRatio.compute_at(output,tile_index); + // gbLogRatio.compute_at(output,tile_index); + logRatios.compute_at(output, tile_index).split(x, xo, xi, 2).unroll(xi); + + output.tile(x, y, xo, yo, xi, yi, 64, 64) + .fuse(xo, yo, tile_index) + .parallel(tile_index) + .vectorize(xi, 8) + .bound(c, 0, 3) + .unroll(c) + .reorder(c, xi, yi, tile_index) + .reorder_storage(c, x, y) + .compute_root();//.compile_to_lowered_stmt("output_unrolledC.html",HTML); + + return output; + + // Time beat: 2.25 seconds } int main(int argc, char **argv) { - Var x, y, c; - //Halide::Image input = load("000734_levels.png"); - //Halide::Image input = load("porcupine2.png"); - Halide::Image input = load("P1040567.png"); - //Halide::Image input = load("teensy.png"); - Func toFloat, toBayer, toDemosaic, toInt, bayerFunc; - toFloat(x,y,c) = cast(input(x,y,c))/255.0f; - toBayer = bayerize(toFloat); - Halide::Image bayer = toBayer.realize(input.width(),input.height()); - //toBayer = bayerize(BoundaryConditions::constant_exterior(toFloat,0.0f,0,input.width(),0,input.height(),0,3)); - bayerFunc(x,y) = bayer(x,y); - toDemosaic = demosaic(BoundaryConditions::mirror_image(bayerFunc,0,input.width(),0,input.height())); - toInt(x,y,c) = cast(Halide::clamp(Halide::round(toDemosaic(x,y,c)*255.0f),0.0f,255.0f)); - toInt.compile_jit(); - timeval t1, t2; - std::cout << "Finished compilation, starting processing" << std::endl; - gettimeofday(&t1, NULL); - Halide::Image output = toInt.realize(input.width(),input.height(),3); - gettimeofday(&t2, NULL); - std::cout<< "Finished processing in " << - float(t2.tv_sec - t1.tv_sec) + float(t2.tv_usec - t1.tv_usec)/1000000.0f << - " seconds" << std::endl; - save(output,"000734_demosaiced7.png"); - //save(output,"porcupine_demosaiced2.png"); - //save(output,"P1040567-med-out.png"); - //save(output,"teensy_out.png"); - return 0; + Var x, y, c; + // Halide::Image input = load("000734_levels.png"); + // Halide::Image input = load("porcupine2.png"); + Halide::Image input = load("P1040567.png"); + // Halide::Image input = load("teensy.png"); + Func toFloat, toBayer, toDemosaic, toInt, bayerFunc; + toFloat(x, y, c) = cast(input(x, y, c)) / 255.0f; + toBayer = bayerize(toFloat); + Halide::Image bayer = toBayer.realize(input.width(), input.height()); + // toBayer = bayerize(BoundaryConditions::constant_exterior(toFloat,0.0f,0,input.width(),0,input.height(),0,3)); + bayerFunc(x, y) = bayer(x, y); + toDemosaic = demosaic(BoundaryConditions::mirror_image(bayerFunc, 0, input.width(), 0, input.height())); + toInt(x, y, c) = cast(Halide::clamp(Halide::round(toDemosaic(x, y, c) * 255.0f), 0.0f, 255.0f)); + toInt.compile_jit(); + timeval t1, t2; + std::cout << "Finished compilation, starting processing" << std::endl; + gettimeofday(&t1, NULL); + Halide::Image output = toInt.realize(input.width(), input.height(), 3); + gettimeofday(&t2, NULL); + std::cout << "Finished processing in " << float(t2.tv_sec - t1.tv_sec) + float(t2.tv_usec - t1.tv_usec) / 1000000.0f + << " seconds" << std::endl; + save(output, "000734_demosaiced7.png"); + // save(output,"porcupine_demosaiced2.png"); + // save(output,"P1040567-med-out.png"); + // save(output,"teensy_out.png"); + return 0; } diff --git a/filmulator-gui/Halide/develop.cpp b/filmulator-gui/Halide/develop.cpp index c3bcaa4a..b915385e 100644 --- a/filmulator-gui/Halide/develop.cpp +++ b/filmulator-gui/Halide/develop.cpp @@ -1,44 +1,50 @@ -Func develop(Func inputs, Expr crystalGrowthConst, - Expr activeLayerThickness, Expr developerConsumptionConst, - Expr silverSaltConsumptionConst, Expr timestep) { - - Expr cgc = crystalGrowthConst*timestep; - Expr dcc = 2.0f*developerConsumptionConst / ( activeLayerThickness*3.0f); +Func develop(Func inputs, + Expr crystalGrowthConst, + Expr activeLayerThickness, + Expr developerConsumptionConst, + Expr silverSaltConsumptionConst, + Expr timestep) +{ + + Expr cgc = crystalGrowthConst * timestep; + Expr dcc = 2.0f * developerConsumptionConst / (activeLayerThickness * 3.0f); Expr sscc = silverSaltConsumptionConst * 2.0f; Func dCrystalRad; // Silver Salt Density - dCrystalRad(x,y,c) = inputs(x,y,DEVEL_CONC) * inputs(x,y,c+6) * cgc; + dCrystalRad(x, y, c) = inputs(x, y, DEVEL_CONC) * inputs(x, y, c + 6) * cgc; Func dCrystalVol; // Crystal Radius Active Crystals - dCrystalVol(x,y,c) = dCrystalRad(x,y,c) * inputs(x,y,c) * inputs(x,y,c) * inputs(x,y,c+3); + dCrystalVol(x, y, c) = dCrystalRad(x, y, c) * inputs(x, y, c) * inputs(x, y, c) * inputs(x, y, c + 3); Func outputs; - outputs(x,y,c) = select( - c == CRYSTAL_RAD_R , inputs(x,y,c) + dCrystalRad(x,y,0), - select(c == CRYSTAL_RAD_G , inputs(x,y,c) + dCrystalRad(x,y,1), - select(c == CRYSTAL_RAD_B , inputs(x,y,c) + dCrystalRad(x,y,2), + outputs(x, y, c) = select(c == CRYSTAL_RAD_R, + inputs(x, y, c) + dCrystalRad(x, y, 0), + select(c == CRYSTAL_RAD_G, + inputs(x, y, c) + dCrystalRad(x, y, 1), + select(c == CRYSTAL_RAD_B, + inputs(x, y, c) + dCrystalRad(x, y, 2), - select(c == DEVEL_CONC , max(0,inputs(x,y,c) - - dcc*(dCrystalVol(x,y,0) + dCrystalVol(x,y,1) + dCrystalVol(x,y,2))), + select(c == DEVEL_CONC, + max(0, inputs(x, y, c) - dcc * (dCrystalVol(x, y, 0) + dCrystalVol(x, y, 1) + dCrystalVol(x, y, 2))), - select(c == SILVER_SALT_DEN_R, max(0,inputs(x,y,c) - sscc*dCrystalVol(x,y,0)), - select(c == SILVER_SALT_DEN_G, max(0,inputs(x,y,c) - sscc*dCrystalVol(x,y,1)), - select(c == SILVER_SALT_DEN_B, max(0,inputs(x,y,c) - sscc*dCrystalVol(x,y,2)), + select(c == SILVER_SALT_DEN_R, + max(0, inputs(x, y, c) - sscc * dCrystalVol(x, y, 0)), + select(c == SILVER_SALT_DEN_G, + max(0, inputs(x, y, c) - sscc * dCrystalVol(x, y, 1)), + select(c == SILVER_SALT_DEN_B, + max(0, inputs(x, y, c) - sscc * dCrystalVol(x, y, 2)), - //Otherwise (active crystals) output=input - inputs(x,y,c)))))))); + // Otherwise (active crystals) output=input + inputs(x, y, c)))))))); - Var x_outer, x_inner; - outputs.split(x,x_outer,x_inner,4).reorder(x_inner,c,x_outer,y) - .vectorize(x_inner).parallel(y); + Var x_outer, x_inner; + outputs.split(x, x_outer, x_inner, 4).reorder(x_inner, c, x_outer, y).vectorize(x_inner).parallel(y); - dCrystalVol.split(x,x_outer,x_inner,4).store_at(outputs,x_outer) - .compute_at(outputs,x_outer).vectorize(x_inner); + dCrystalVol.split(x, x_outer, x_inner, 4).store_at(outputs, x_outer).compute_at(outputs, x_outer).vectorize(x_inner); - dCrystalRad.split(x,x_outer,x_inner,4).store_at(outputs,x_outer) - .compute_at(outputs,x_outer).vectorize(x_inner); + dCrystalRad.split(x, x_outer, x_inner, 4).store_at(outputs, x_outer).compute_at(outputs, x_outer).vectorize(x_inner); return outputs; } diff --git a/filmulator-gui/Halide/diffuse.cpp b/filmulator-gui/Halide/diffuse.cpp index 0b237857..f0a99962 100644 --- a/filmulator-gui/Halide/diffuse.cpp +++ b/filmulator-gui/Halide/diffuse.cpp @@ -1,117 +1,106 @@ -Func performBlur(Func f, Func coeff, Expr size, Expr sigma) { - Func blurred; - blurred(x, y) = undef(); - - // Warm up - blurred(x, 0) = coeff(0) * f(x, 0); - blurred(x, 1) = (coeff(0) * f(x, 1) + - coeff(1) * blurred(x, 0)); - blurred(x, 2) = (coeff(0) * f(x, 2) + - coeff(1) * blurred(x, 1) + - coeff(2) * blurred(x, 0)); - - // Top to bottom - RDom fwd(3, size - 3); - blurred(x, fwd) = (coeff(0) * f(x, fwd) + - coeff(1) * blurred(x, fwd - 1) + - coeff(2) * blurred(x, fwd - 2) + - coeff(3) * blurred(x, fwd - 3)); - - // Tail end - Expr padding = cast(ceil(4*sigma) + 3); - RDom tail(size, padding); - blurred(x, tail) = (coeff(1) * blurred(x, tail - 1) + - coeff(2) * blurred(x, tail - 2) + - coeff(3) * blurred(x, tail - 3)); - - // Bottom to top - Expr last = size + padding - 1; - RDom backwards(0, last - 2); - Expr b = last - 3 - backwards; // runs from last - 3 down to zero - blurred(x, b) = (coeff(0) * blurred(x, b) + - coeff(1) * blurred(x, b + 1) + - coeff(2) * blurred(x, b + 2) + - coeff(3) * blurred(x, b + 3)); - return blurred; +Func performBlur(Func f, Func coeff, Expr size, Expr sigma) +{ + Func blurred; + blurred(x, y) = undef(); + + // Warm up + blurred(x, 0) = coeff(0) * f(x, 0); + blurred(x, 1) = (coeff(0) * f(x, 1) + coeff(1) * blurred(x, 0)); + blurred(x, 2) = (coeff(0) * f(x, 2) + coeff(1) * blurred(x, 1) + coeff(2) * blurred(x, 0)); + + // Top to bottom + RDom fwd(3, size - 3); + blurred(x, fwd) = (coeff(0) * f(x, fwd) + coeff(1) * blurred(x, fwd - 1) + coeff(2) * blurred(x, fwd - 2) + + coeff(3) * blurred(x, fwd - 3)); + + // Tail end + Expr padding = cast(ceil(4 * sigma) + 3); + RDom tail(size, padding); + blurred(x, tail) = + (coeff(1) * blurred(x, tail - 1) + coeff(2) * blurred(x, tail - 2) + coeff(3) * blurred(x, tail - 3)); + + // Bottom to top + Expr last = size + padding - 1; + RDom backwards(0, last - 2); + Expr b = last - 3 - backwards;// runs from last - 3 down to zero + blurred(x, b) = (coeff(0) * blurred(x, b) + coeff(1) * blurred(x, b + 1) + coeff(2) * blurred(x, b + 2) + + coeff(3) * blurred(x, b + 3)); + return blurred; } -Func blur_then_transpose(Func f, Func coeff, Expr size, Expr sigma) { - - Func blurred = performBlur(f, coeff, size, sigma); - - // Also compute attenuation due to zero boundary condition by - // blurring an image of ones in the same way. This gives a - // boundary condition equivalent to reweighting the Gaussian - // near the edge. (TODO: add a generator param to select - // different boundary conditions). - Func ones; - ones(x, y) = 1.0f; - Func attenuation = performBlur(ones, coeff, size, sigma); - - // Invert the attenuation so we can multiply by it. The - // attenuation is the same for every row/channel so we only - // need one column. - Func inverse_attenuation; - inverse_attenuation(y) = 1.0f / attenuation(0, y); - - // Transpose it - Func transposed; - transposed(x, y) = blurred(y, x); - - // Correct for attenuation - Func out; - out(x, y) = transposed(x, y) * inverse_attenuation(x); - - // Schedule it. - Var yi, xi, yii, xii; - - attenuation.compute_root(); - inverse_attenuation.compute_root().vectorize(y, 8); - out.compute_root() - .tile(x, y, xi, yi, 8, 32) - .tile(xi, yi, xii, yii, 8, 8) - .vectorize(xii).unroll(yii).parallel(y); - blurred.compute_at(out, y); - transposed.compute_at(out, xi).vectorize(y).unroll(x); - - for (int i = 0; i < blurred.num_update_definitions(); i++) { - RDom r = blurred.reduction_domain(i); - if (r.defined()) { - blurred.update(i).reorder(x, r); - } - blurred.update(i).vectorize(x, 8).unroll(x); - } - - return out; +Func blur_then_transpose(Func f, Func coeff, Expr size, Expr sigma) +{ + + Func blurred = performBlur(f, coeff, size, sigma); + + // Also compute attenuation due to zero boundary condition by + // blurring an image of ones in the same way. This gives a + // boundary condition equivalent to reweighting the Gaussian + // near the edge. (TODO: add a generator param to select + // different boundary conditions). + Func ones; + ones(x, y) = 1.0f; + Func attenuation = performBlur(ones, coeff, size, sigma); + + // Invert the attenuation so we can multiply by it. The + // attenuation is the same for every row/channel so we only + // need one column. + Func inverse_attenuation; + inverse_attenuation(y) = 1.0f / attenuation(0, y); + + // Transpose it + Func transposed; + transposed(x, y) = blurred(y, x); + + // Correct for attenuation + Func out; + out(x, y) = transposed(x, y) * inverse_attenuation(x); + + // Schedule it. + Var yi, xi, yii, xii; + + attenuation.compute_root(); + inverse_attenuation.compute_root().vectorize(y, 8); + out.compute_root().tile(x, y, xi, yi, 8, 32).tile(xi, yi, xii, yii, 8, 8).vectorize(xii).unroll(yii).parallel(y); + blurred.compute_at(out, y); + transposed.compute_at(out, xi).vectorize(y).unroll(x); + + for (int i = 0; i < blurred.num_update_definitions(); i++) { + RDom r = blurred.reduction_domain(i); + if (r.defined()) { blurred.update(i).reorder(x, r); } + blurred.update(i).vectorize(x, 8).unroll(x); + } + + return out; } -Func blur(Func input, Expr sigma, Expr width, Expr height) { - - // Compute IIR coefficients using the method of Young and Van Vliet. - Func coeff; - Expr q = select(sigma < 2.5f, - 3.97156f - 4.14554f*sqrt(1 - 0.26891f*sigma), - 0.98711f*sigma - 0.96330f); - Expr denom = 1.57825f + 2.44413f*q + 1.4281f*q*q + 0.422205f*q*q*q; - coeff(x) = undef(); - coeff(1) = (2.44413f*q + 2.85619f*q*q + 1.26661f*q*q*q)/denom; - coeff(2) = -(1.4281f*q*q + 1.26661f*q*q*q)/denom; - coeff(3) = (0.422205f*q*q*q)/denom; - coeff(0) = 1 - (coeff(1) + coeff(2) + coeff(3)); - coeff.compute_root(); - - Func blurY, blurX; - blurY = blur_then_transpose(input, coeff, height, sigma); - blurX = blur_then_transpose(blurY, coeff, width, sigma); - return blurX; +Func blur(Func input, Expr sigma, Expr width, Expr height) +{ + + // Compute IIR coefficients using the method of Young and Van Vliet. + Func coeff; + Expr q = select(sigma < 2.5f, 3.97156f - 4.14554f * sqrt(1 - 0.26891f * sigma), 0.98711f * sigma - 0.96330f); + Expr denom = 1.57825f + 2.44413f * q + 1.4281f * q * q + 0.422205f * q * q * q; + coeff(x) = undef(); + coeff(1) = (2.44413f * q + 2.85619f * q * q + 1.26661f * q * q * q) / denom; + coeff(2) = -(1.4281f * q * q + 1.26661f * q * q * q) / denom; + coeff(3) = (0.422205f * q * q * q) / denom; + coeff(0) = 1 - (coeff(1) + coeff(2) + coeff(3)); + coeff.compute_root(); + + Func blurY, blurX; + blurY = blur_then_transpose(input, coeff, height, sigma); + blurX = blur_then_transpose(blurY, coeff, width, sigma); + return blurX; } -Func diffuse(Func input, Expr sigma_const, Expr pixels_per_millimeter, Expr timestep, Expr width, Expr height){ +Func diffuse(Func input, Expr sigma_const, Expr pixels_per_millimeter, Expr timestep, Expr width, Expr height) +{ - Expr sigma = sqrt(timestep)*sigma_const*pixels_per_millimeter; -// The following causes bounds inference failure -// Expr sigma = sqrt(timestep*pow(sigma_const*pixels_per_millimeter,2)); - Func diffused; - diffused = blur(input,sigma,width,height); - return diffused; + Expr sigma = sqrt(timestep) * sigma_const * pixels_per_millimeter; + // The following causes bounds inference failure + // Expr sigma = sqrt(timestep*pow(sigma_const*pixels_per_millimeter,2)); + Func diffused; + diffused = blur(input, sigma, width, height); + return diffused; } diff --git a/filmulator-gui/Halide/downscaleDiffuse.cpp b/filmulator-gui/Halide/downscaleDiffuse.cpp index d87ecf20..ea85be99 100644 --- a/filmulator-gui/Halide/downscaleDiffuse.cpp +++ b/filmulator-gui/Halide/downscaleDiffuse.cpp @@ -1,5 +1,5 @@ -#include #include +#include #include using namespace Halide; @@ -8,166 +8,158 @@ using namespace Halide::BoundaryConditions; Var x, y; // Sum an image vertically using a box filter with a zero boundary condition. -Func downscale(Func input, Expr radius, Expr width, Expr height) { - - Func b,c,d; - // Accumulate as a uint16 - b(x, y) = cast(0); - c(x, y) = cast(0); - - // The first value is the sum over pixels [0, radius] - RDom r(-radius, radius); - b(x,y) += input(x + r - radius,y); - c(x,y) += b(x,y + r - radius); - - Expr denominator = pow(2*radius+1,2); - - d(x,y) = c(x, y)/denominator; - Var xo,xi; - //d.bound(x,0,ceil(width/denominator)).bound(y,0,ceil(height/denominator)); - b.compute_root().gpu_tile(x,y,16,16); - //d.reorder(y,x).gpu_tile(x,y,16,16); - return d; +Func downscale(Func input, Expr radius, Expr width, Expr height) +{ + + Func b, c, d; + // Accumulate as a uint16 + b(x, y) = cast(0); + c(x, y) = cast(0); + + // The first value is the sum over pixels [0, radius] + RDom r(-radius, radius); + b(x, y) += input(x + r - radius, y); + c(x, y) += b(x, y + r - radius); + + Expr denominator = pow(2 * radius + 1, 2); + + d(x, y) = c(x, y) / denominator; + Var xo, xi; + // d.bound(x,0,ceil(width/denominator)).bound(y,0,ceil(height/denominator)); + b.compute_root().gpu_tile(x, y, 16, 16); + // d.reorder(y,x).gpu_tile(x,y,16,16); + return d; } -Func upscale(Func input, Expr radius){ - Func output; - - Expr scaleFactor = (2*radius+1); - Expr xf = (x % scaleFactor)/cast(scaleFactor); - Expr yf = (y % scaleFactor)/cast(scaleFactor); - Expr x0 = cast(floor(x/scaleFactor)); - Expr y0 = cast(floor(y/scaleFactor)); - Expr x1 = cast( ceil(x/scaleFactor)); - Expr y1 = cast( ceil(x/scaleFactor)); - Expr Ix0y0 = input(x0,y0); - Expr Ix0y1 = input(x0,y1); - Expr Ix1y0 = input(x1,y0); - Expr Ix1y1 = input(x1,y1); - output(x,y) = lerp(lerp(Ix0y0,Ix1y0,xf), - lerp(Ix0y1,Ix1y1,xf),yf); - return output; +Func upscale(Func input, Expr radius) +{ + Func output; + + Expr scaleFactor = (2 * radius + 1); + Expr xf = (x % scaleFactor) / cast(scaleFactor); + Expr yf = (y % scaleFactor) / cast(scaleFactor); + Expr x0 = cast(floor(x / scaleFactor)); + Expr y0 = cast(floor(y / scaleFactor)); + Expr x1 = cast(ceil(x / scaleFactor)); + Expr y1 = cast(ceil(x / scaleFactor)); + Expr Ix0y0 = input(x0, y0); + Expr Ix0y1 = input(x0, y1); + Expr Ix1y0 = input(x1, y0); + Expr Ix1y1 = input(x1, y1); + output(x, y) = lerp(lerp(Ix0y0, Ix1y0, xf), lerp(Ix0y1, Ix1y1, xf), yf); + return output; } -Func gaussBlur(Func input){ - Func blurx,blury; - blurx(x, y) = (input(x-2, y) + - input(x-1, y)*4 + - input(x , y)*6 + - input(x+1, y)*4 + - input(x+2, y)); - blury(x, y) = (blurx(x, y-2) + - blurx(x, y-1)*4 + - blurx(x, y )*6 + - blurx(x, y+1)*4 + - blurx(x, y+2)); - return blury; +Func gaussBlur(Func input) +{ + Func blurx, blury; + blurx(x, y) = (input(x - 2, y) + input(x - 1, y) * 4 + input(x, y) * 6 + input(x + 1, y) * 4 + input(x + 2, y)); + blury(x, y) = (blurx(x, y - 2) + blurx(x, y - 1) * 4 + blurx(x, y) * 6 + blurx(x, y + 1) * 4 + blurx(x, y + 2)); + return blury; } -int main(int argc, char **argv) { - - ImageParam input(UInt(8), 2); - Param radius; - Expr denominator = cast(radius * 2 + 1); - - // radius can be at most 127 before the sums overflow - radius.set_range(0, 127); - - Func in = mirror_interior(input); - - Func downscaled; - downscaled = downscale(in,150,input.width(), input.height()); - - Func blurred_small; - blurred_small = gaussBlur(downscaled); - - Func blurred_float; - blurred_float = upscale(blurred_small,150); - - Func blurred; - blurred(x,y) = cast(blurred_float(x,y)); - - Target target = get_target_from_environment(); - std::cout << target.to_string() << std::endl; - if(target.has_gpu_feature()) - { - downscaled.compute_root().gpu_tile(x,y,16,16); - blurred_small.compute_root().gpu_tile(x,y,16,16); - blurred.gpu_tile(x,y,16,16); - - Target target = get_host_target(); - target.set_feature(Target::CUDA); - //target.set_feature(Target::GPUDebug); - blurred.compile_jit(target); - } - else - { - /*blur_in_x.compute_root().vectorize(x, 8).split(x, xo, xi, 2).reorder(xi, y, xo).parallel(xo).unroll(xi); - // It makes sense to explicitly compute the transpose so that the - // blur can be done using dense vector loads. - transpose_x.compute_at(blur_in_x, xo).vectorize(x, 8).unroll(x); - sum_in_x.compute_at(blur_in_x, xo); - for (int stage = 0; stage < 5; stage++) { - sum_in_x.update(stage).vectorize(x, 8).unroll(x); - if (stage > 0) { - RVar r = sum_in_x.reduction_domain(stage).x; - sum_in_x.update(stage).reorder(x, r); - } +int main(int argc, char **argv) +{ + + ImageParam input(UInt(8), 2); + Param radius; + Expr denominator = cast(radius * 2 + 1); + + // radius can be at most 127 before the sums overflow + radius.set_range(0, 127); + + Func in = mirror_interior(input); + + Func downscaled; + downscaled = downscale(in, 150, input.width(), input.height()); + + Func blurred_small; + blurred_small = gaussBlur(downscaled); + + Func blurred_float; + blurred_float = upscale(blurred_small, 150); + + Func blurred; + blurred(x, y) = cast(blurred_float(x, y)); + + Target target = get_target_from_environment(); + std::cout << target.to_string() << std::endl; + if (target.has_gpu_feature()) { + downscaled.compute_root().gpu_tile(x, y, 16, 16); + blurred_small.compute_root().gpu_tile(x, y, 16, 16); + blurred.gpu_tile(x, y, 16, 16); + + Target target = get_host_target(); + target.set_feature(Target::CUDA); + // target.set_feature(Target::GPUDebug); + blurred.compile_jit(target); + } else { + /*blur_in_x.compute_root().vectorize(x, 8).split(x, xo, xi, 2).reorder(xi, y, xo).parallel(xo).unroll(xi); + // It makes sense to explicitly compute the transpose so that the + // blur can be done using dense vector loads. + transpose_x.compute_at(blur_in_x, xo).vectorize(x, 8).unroll(x); + sum_in_x.compute_at(blur_in_x, xo); + for (int stage = 0; stage < 5; stage++) { + sum_in_x.update(stage).vectorize(x, 8).unroll(x); + if (stage > 0) { + RVar r = sum_in_x.reduction_domain(stage).x; + sum_in_x.update(stage).reorder(x, r); } - - - blur_in_y.compute_root().vectorize(x, 8).split(x, xo, xi, 2).reorder(xi, y, xo).parallel(xo).unroll(xi); - transpose_y.compute_at(blur_in_y, xo).vectorize(x, 16).unroll(x); - sum_in_y.compute_at(blur_in_y, xo); - for (int stage = 0; stage < 5; stage++) { - sum_in_y.update(stage).vectorize(x, 8).unroll(x); - if (stage > 0) { - RVar r = sum_in_y.reduction_domain(stage).x; - sum_in_y.update(stage).reorder(x, r); - } - }*/ } - // Dump the assembly for inspection. - //blur_in_y.compile_to_assembly("/dev/stdout", Internal::vec(input, radius), "box_blur"); - - // Save the output. Comment this out for benchmarking - it takes - // way more time than the actual algorithm. Should look like a - // blurry circle. - //blur_in_y.debug_to_file("output.tiff"); - - // Set some test parameters - radius.set(19); - // Make a test input image of a circle. - Image input_image(4000, 4000); - lambda(x, y, select(((x - 500)*(x - 500) + (y - 500)*(y - 500)) < 100*100, cast(255), cast(0))).realize(input_image); - input.set(input_image); - - Image out(4000, 4000); - // Realize it once to trigger compilation. + blur_in_y.compute_root().vectorize(x, 8).split(x, xo, xi, 2).reorder(xi, y, xo).parallel(xo).unroll(xi); + transpose_y.compute_at(blur_in_y, xo).vectorize(x, 16).unroll(x); + sum_in_y.compute_at(blur_in_y, xo); + for (int stage = 0; stage < 5; stage++) { + sum_in_y.update(stage).vectorize(x, 8).unroll(x); + if (stage > 0) { + RVar r = sum_in_y.reduction_domain(stage).x; + sum_in_y.update(stage).reorder(x, r); + } + }*/ + } + + // Dump the assembly for inspection. + // blur_in_y.compile_to_assembly("/dev/stdout", Internal::vec(input, radius), "box_blur"); + + // Save the output. Comment this out for benchmarking - it takes + // way more time than the actual algorithm. Should look like a + // blurry circle. + // blur_in_y.debug_to_file("output.tiff"); + + // Set some test parameters + radius.set(19); + + // Make a test input image of a circle. + Image input_image(4000, 4000); + lambda( + x, y, select(((x - 500) * (x - 500) + (y - 500) * (y - 500)) < 100 * 100, cast(255), cast(0))) + .realize(input_image); + input.set(input_image); + + Image out(4000, 4000); + // Realize it once to trigger compilation. + blurred.realize(out); + timeval t1, t2; + gettimeofday(&t1, NULL); + float numIter = 50; + for (size_t i = 0; i < numIter; i++) { + // Realize 100 more times for timing. blurred.realize(out); - timeval t1, t2; - gettimeofday(&t1, NULL); - float numIter = 50; - for (size_t i = 0; i < numIter; i++) { - // Realize 100 more times for timing. - blurred.realize(out); - } - gettimeofday(&t2, NULL); - - int64_t dt = t2.tv_sec - t1.tv_sec; - dt *= 1000; - dt += (t2.tv_usec - t1.tv_usec) / 1000; - printf("%0.4f ms\n", dt / numIter); - - /*std::vector arguments; - arguments.push_back(input_image); - arguments.push_back(radius); - arguments.push_back(radius); - blur_in_y.compile_to_c("gradient.cpp", arguments, "attachment");*/ - printf("Success!\n"); - return 0; + } + gettimeofday(&t2, NULL); + + int64_t dt = t2.tv_sec - t1.tv_sec; + dt *= 1000; + dt += (t2.tv_usec - t1.tv_usec) / 1000; + printf("%0.4f ms\n", dt / numIter); + + /*std::vector arguments; + arguments.push_back(input_image); + arguments.push_back(radius); + arguments.push_back(radius); + blur_in_y.compile_to_c("gradient.cpp", arguments, "attachment");*/ + printf("Success!\n"); + return 0; } - - diff --git a/filmulator-gui/Halide/filmulate.cpp b/filmulator-gui/Halide/filmulate.cpp index 437eb035..75c1c7cf 100644 --- a/filmulator-gui/Halide/filmulate.cpp +++ b/filmulator-gui/Halide/filmulate.cpp @@ -12,76 +12,74 @@ Var x, y, c; #include "develop.cpp" #include "diffuse.cpp" -class filmulateIterationGenerator : public Halide::Generator { - public: - - Param reservoirConcentration{"reservoirConcentration"}; - Param reservoirThickness{"reservoirThickness"}; - Param crystalGrowthConst{"crystalGrowthConst"}; - Param activeLayerThickness{"activeLayerThickness"}; - Param developerConsumptionConst{"developerConsumptionConst"}; - Param silverSaltConsumptionConst{"silverSaltConsumptionConst"}; - Param stepTime{"stepTime"}; - Param filmArea{"filmArea"}; - Param sigmaConst{"sigmaConst"}; - Param layerMixConst{"layerMixConst"}; - Param layerTimeDivisor{"layerTimeDivisor"}; - Param doDiffuse{"doDiffuse"}; - - ImageParam input{Float(32), 3,"input"}; - - Pipeline build() { - Func filmulationData = lambda(x,y,c,input(x,y,c)); - - Func developed; - developed = develop(filmulationData, crystalGrowthConst, activeLayerThickness, - developerConsumptionConst, silverSaltConsumptionConst, - stepTime); - developed.compute_root(); - - Func diffused; - Func initialDeveloper, initialDeveloperMirrored; - initialDeveloper(x,y) = developed(x,y,DEVEL_CONC); - initialDeveloperMirrored = BoundaryConditions::mirror_interior(initialDeveloper,0,input.width(),0,input.height()); - Expr pixelsPerMillimeter = sqrt(input.width()*input.height()/filmArea); - diffused = diffuse(initialDeveloper,sigmaConst,pixelsPerMillimeter, stepTime, - input.width(), input.height()); - diffused.compute_root(); - - Func developerFlux; //Developer moving from reservoir to active layer - Expr layerMixCoef = pow(layerMixConst,stepTime/layerTimeDivisor); - developerFlux(x,y) = (reservoirConcentration - diffused(x,y))*layerMixCoef; - developerFlux.compute_root(); - - Func layerMixed; - layerMixed(x,y) = diffused(x,y) + developerFlux(x,y); - layerMixed.compute_root(); - - Func fluxSum; // Total developer moved in units of density*pixelVolume^3 - RDom r(0, input.width(), 0, input.height()); - fluxSum(x) = 0.0f; - fluxSum(0) += developerFlux(r.x,r.y); - fluxSum.compute_root(); - - Func newReservoirConcentration; - // Total developer moved in units of density*mm^3 - Expr totalFluxMM = fluxSum(0)*activeLayerThickness * 1/pow(pixelsPerMillimeter,2); - Expr reservoirVolume = reservoirThickness*filmArea; - Expr reservoirTotalDeveloper = reservoirVolume*reservoirConcentration; - newReservoirConcentration(x) = (reservoirTotalDeveloper - totalFluxMM)/reservoirVolume; - - Func filmulationDataOut; - filmulationDataOut(x,y,c) = select(c == DEVEL_CONC && doDiffuse == 1, - layerMixed(x,y), - developed(x,y,c)); - Func reservoirConcentrationOut; - reservoirConcentrationOut(x) = select(doDiffuse == 1, - newReservoirConcentration(x), - reservoirConcentration); - return Pipeline({filmulationDataOut,reservoirConcentrationOut}); - }; +class filmulateIterationGenerator : public Halide::Generator +{ +public: + Param reservoirConcentration{ "reservoirConcentration" }; + Param reservoirThickness{ "reservoirThickness" }; + Param crystalGrowthConst{ "crystalGrowthConst" }; + Param activeLayerThickness{ "activeLayerThickness" }; + Param developerConsumptionConst{ "developerConsumptionConst" }; + Param silverSaltConsumptionConst{ "silverSaltConsumptionConst" }; + Param stepTime{ "stepTime" }; + Param filmArea{ "filmArea" }; + Param sigmaConst{ "sigmaConst" }; + Param layerMixConst{ "layerMixConst" }; + Param layerTimeDivisor{ "layerTimeDivisor" }; + Param doDiffuse{ "doDiffuse" }; + + ImageParam input{ Float(32), 3, "input" }; + + Pipeline build() + { + Func filmulationData = lambda(x, y, c, input(x, y, c)); + + Func developed; + developed = develop(filmulationData, + crystalGrowthConst, + activeLayerThickness, + developerConsumptionConst, + silverSaltConsumptionConst, + stepTime); + developed.compute_root(); + + Func diffused; + Func initialDeveloper, initialDeveloperMirrored; + initialDeveloper(x, y) = developed(x, y, DEVEL_CONC); + initialDeveloperMirrored = + BoundaryConditions::mirror_interior(initialDeveloper, 0, input.width(), 0, input.height()); + Expr pixelsPerMillimeter = sqrt(input.width() * input.height() / filmArea); + diffused = diffuse(initialDeveloper, sigmaConst, pixelsPerMillimeter, stepTime, input.width(), input.height()); + diffused.compute_root(); + + Func developerFlux;// Developer moving from reservoir to active layer + Expr layerMixCoef = pow(layerMixConst, stepTime / layerTimeDivisor); + developerFlux(x, y) = (reservoirConcentration - diffused(x, y)) * layerMixCoef; + developerFlux.compute_root(); + + Func layerMixed; + layerMixed(x, y) = diffused(x, y) + developerFlux(x, y); + layerMixed.compute_root(); + + Func fluxSum;// Total developer moved in units of density*pixelVolume^3 + RDom r(0, input.width(), 0, input.height()); + fluxSum(x) = 0.0f; + fluxSum(0) += developerFlux(r.x, r.y); + fluxSum.compute_root(); + + Func newReservoirConcentration; + // Total developer moved in units of density*mm^3 + Expr totalFluxMM = fluxSum(0) * activeLayerThickness * 1 / pow(pixelsPerMillimeter, 2); + Expr reservoirVolume = reservoirThickness * filmArea; + Expr reservoirTotalDeveloper = reservoirVolume * reservoirConcentration; + newReservoirConcentration(x) = (reservoirTotalDeveloper - totalFluxMM) / reservoirVolume; + + Func filmulationDataOut; + filmulationDataOut(x, y, c) = select(c == DEVEL_CONC && doDiffuse == 1, layerMixed(x, y), developed(x, y, c)); + Func reservoirConcentrationOut; + reservoirConcentrationOut(x) = select(doDiffuse == 1, newReservoirConcentration(x), reservoirConcentration); + return Pipeline({ filmulationDataOut, reservoirConcentrationOut }); + }; }; -RegisterGenerator filmulateIterationGenerator{"filmulateIterationGenerator"}; - - +RegisterGenerator filmulateIterationGenerator{ "filmulateIterationGenerator" }; diff --git a/filmulator-gui/Halide/generateFilmulatedImage.cpp b/filmulator-gui/Halide/generateFilmulatedImage.cpp index 97b00ae0..c96c4b09 100644 --- a/filmulator-gui/Halide/generateFilmulatedImage.cpp +++ b/filmulator-gui/Halide/generateFilmulatedImage.cpp @@ -1,25 +1,26 @@ -#include #include "halideFilmulate.h" +#include using namespace Halide; -Var x,y,c; +Var x, y, c; -int main(int argc, char **argv){ +int main(int argc, char **argv) +{ - ImageParam input(type_of(),3); - Func in = lambda(x,y,c,input(x,y,c)); - Func outputImage; - outputImage(x,y,c) = undef(); - outputImage(x,y,0) = cast(min((1000.0f*256.0f*pow(in(x,y,CRYSTAL_RAD_R),2)* - in(x,y,ACTIVE_CRYSTALS_R)),255.0f)); - outputImage(x,y,1) = cast(min((1000.0f*256.0f*pow(in(x,y,CRYSTAL_RAD_G),2)* - in(x,y,ACTIVE_CRYSTALS_G)),255.0f)); - outputImage(x,y,2) = cast(min((1000.0f*256.0f*pow(in(x,y,CRYSTAL_RAD_B),2)* - in(x,y,ACTIVE_CRYSTALS_B)),255.0f)); + ImageParam input(type_of(), 3); + Func in = lambda(x, y, c, input(x, y, c)); + Func outputImage; + outputImage(x, y, c) = undef(); + outputImage(x, y, 0) = + cast(min((1000.0f * 256.0f * pow(in(x, y, CRYSTAL_RAD_R), 2) * in(x, y, ACTIVE_CRYSTALS_R)), 255.0f)); + outputImage(x, y, 1) = + cast(min((1000.0f * 256.0f * pow(in(x, y, CRYSTAL_RAD_G), 2) * in(x, y, ACTIVE_CRYSTALS_G)), 255.0f)); + outputImage(x, y, 2) = + cast(min((1000.0f * 256.0f * pow(in(x, y, CRYSTAL_RAD_B), 2) * in(x, y, ACTIVE_CRYSTALS_B)), 255.0f)); - std::vector args(1); - args[0] = input; - outputImage.compile_to_file("generateFilmulatedImage",args); - return 0; + std::vector args(1); + args[0] = input; + outputImage.compile_to_file("generateFilmulatedImage", args); + return 0; } diff --git a/filmulator-gui/Halide/halideFilmulate.h b/filmulator-gui/Halide/halideFilmulate.h index 9c818117..794940de 100644 --- a/filmulator-gui/Halide/halideFilmulate.h +++ b/filmulator-gui/Halide/halideFilmulate.h @@ -9,4 +9,3 @@ #define SILVER_SALT_DEN_G 7 #define SILVER_SALT_DEN_B 8 #define DEVEL_CONC 9 - diff --git a/filmulator-gui/Halide/include/Halide.h b/filmulator-gui/Halide/include/Halide.h index 87765318..97133f9b 100644 --- a/filmulator-gui/Halide/include/Halide.h +++ b/filmulator-gui/Halide/include/Halide.h @@ -1,9 +1,9 @@ #ifndef HALIDE_INTROSPECTION_H #define HALIDE_INTROSPECTION_H -#include #include #include +#include // Always use assert, even if llvm-config defines NDEBUG #ifdef NDEBUG @@ -20,9 +20,9 @@ /** \file * Various utility functions used internally Halide. */ -#include -#include #include +#include +#include // by default, the symbol EXPORT does nothing. In windows dll builds we can define it to __declspec(dllexport) #if defined(_WIN32) && defined(Halide_SHARED) @@ -49,46 +49,46 @@ namespace Halide { namespace Internal { -/** Build small vectors of up to 10 elements. If we used C++11 and - * had vector initializers, this would not be necessary, but we - * don't want to rely on C++11 support. */ -//@{ -template -std::vector vec(T a) { + /** Build small vectors of up to 10 elements. If we used C++11 and + * had vector initializers, this would not be necessary, but we + * don't want to rely on C++11 support. */ + //@{ + template std::vector vec(T a) + { std::vector v(1); v[0] = a; return v; -} + } -template -std::vector vec(T a, T b) { + template std::vector vec(T a, T b) + { std::vector v(2); v[0] = a; v[1] = b; return v; -} + } -template -std::vector vec(T a, T b, T c) { + template std::vector vec(T a, T b, T c) + { std::vector v(3); v[0] = a; v[1] = b; v[2] = c; return v; -} + } -template -std::vector vec(T a, T b, T c, T d) { + template std::vector vec(T a, T b, T c, T d) + { std::vector v(4); v[0] = a; v[1] = b; v[2] = c; v[3] = d; return v; -} + } -template -std::vector vec(T a, T b, T c, T d, T e) { + template std::vector vec(T a, T b, T c, T d, T e) + { std::vector v(5); v[0] = a; v[1] = b; @@ -96,10 +96,10 @@ std::vector vec(T a, T b, T c, T d, T e) { v[3] = d; v[4] = e; return v; -} + } -template -std::vector vec(T a, T b, T c, T d, T e, T f) { + template std::vector vec(T a, T b, T c, T d, T e, T f) + { std::vector v(6); v[0] = a; v[1] = b; @@ -108,10 +108,10 @@ std::vector vec(T a, T b, T c, T d, T e, T f) { v[4] = e; v[5] = f; return v; -} + } -template -std::vector vec(T a, T b, T c, T d, T e, T f, T g) { + template std::vector vec(T a, T b, T c, T d, T e, T f, T g) + { std::vector v(7); v[0] = a; v[1] = b; @@ -121,10 +121,10 @@ std::vector vec(T a, T b, T c, T d, T e, T f, T g) { v[5] = f; v[6] = g; return v; -} + } -template -std::vector vec(T a, T b, T c, T d, T e, T f, T g, T h) { + template std::vector vec(T a, T b, T c, T d, T e, T f, T g, T h) + { std::vector v(8); v[0] = a; v[1] = b; @@ -135,10 +135,10 @@ std::vector vec(T a, T b, T c, T d, T e, T f, T g, T h) { v[6] = g; v[7] = h; return v; -} + } -template -std::vector vec(T a, T b, T c, T d, T e, T f, T g, T h, T i) { + template std::vector vec(T a, T b, T c, T d, T e, T f, T g, T h, T i) + { std::vector v(9); v[0] = a; v[1] = b; @@ -150,10 +150,10 @@ std::vector vec(T a, T b, T c, T d, T e, T f, T g, T h, T i) { v[7] = h; v[8] = i; return v; -} + } -template -std::vector vec(T a, T b, T c, T d, T e, T f, T g, T h, T i, T j) { + template std::vector vec(T a, T b, T c, T d, T e, T f, T g, T h, T i, T j) + { std::vector v(10); v[0] = a; v[1] = b; @@ -166,46 +166,46 @@ std::vector vec(T a, T b, T c, T d, T e, T f, T g, T h, T i, T j) { v[8] = i; v[9] = j; return v; -} -// @} + } + // @} -/** Convert an integer to a string. */ -EXPORT std::string int_to_string(int x); + /** Convert an integer to a string. */ + EXPORT std::string int_to_string(int x); -/** An aggressive form of reinterpret cast used for correct type-punning. */ -template -DstType reinterpret_bits(const SrcType &src) { + /** An aggressive form of reinterpret cast used for correct type-punning. */ + template DstType reinterpret_bits(const SrcType &src) + { assert(sizeof(SrcType) == sizeof(DstType)); DstType dst; memcpy(&dst, &src, sizeof(SrcType)); return dst; -} + } -/** Make a unique name for an object based on the name of the stack - * variable passed in. If introspection isn't working or there are no - * debug symbols, just uses unique_name with the given prefix. */ -EXPORT std::string make_entity_name(void *stack_ptr, const std::string &type, char prefix); + /** Make a unique name for an object based on the name of the stack + * variable passed in. If introspection isn't working or there are no + * debug symbols, just uses unique_name with the given prefix. */ + EXPORT std::string make_entity_name(void *stack_ptr, const std::string &type, char prefix); -/** Generate a unique name starting with the given character. It's - * unique relative to all other calls to unique_name done by this - * process. Not thread-safe. */ -EXPORT std::string unique_name(char prefix); + /** Generate a unique name starting with the given character. It's + * unique relative to all other calls to unique_name done by this + * process. Not thread-safe. */ + EXPORT std::string unique_name(char prefix); -/** Generate a unique name starting with the given string. Not - * thread-safe. */ -EXPORT std::string unique_name(const std::string &name, bool user = true); + /** Generate a unique name starting with the given string. Not + * thread-safe. */ + EXPORT std::string unique_name(const std::string &name, bool user = true); -/** Test if the first string starts with the second string */ -EXPORT bool starts_with(const std::string &str, const std::string &prefix); + /** Test if the first string starts with the second string */ + EXPORT bool starts_with(const std::string &str, const std::string &prefix); -/** Test if the first string ends with the second string */ -EXPORT bool ends_with(const std::string &str, const std::string &suffix); + /** Test if the first string ends with the second string */ + EXPORT bool ends_with(const std::string &str, const std::string &suffix); -/** Return the final token of the name string using the given delimiter. */ -EXPORT std::string base_name(const std::string &name, char delim = '.'); + /** Return the final token of the name string using the given delimiter. */ + EXPORT std::string base_name(const std::string &name, char delim = '.'); -} -} +}// namespace Internal +}// namespace Halide #endif @@ -219,23 +219,23 @@ EXPORT std::string base_name(const std::string &name, char delim = '.'); namespace Halide { namespace Internal { -/** Get the name of a stack variable from its address. The stack - * variable must be in a compilation unit compiled with -g to - * work. The expected type helps distinguish between variables at the - * same address, e.g a class instance vs its first member. */ -EXPORT std::string get_variable_name(const void *, const std::string &expected_type); + /** Get the name of a stack variable from its address. The stack + * variable must be in a compilation unit compiled with -g to + * work. The expected type helps distinguish between variables at the + * same address, e.g a class instance vs its first member. */ + EXPORT std::string get_variable_name(const void *, const std::string &expected_type); -/** Get the source location in the call stack, skipping over calls in - * the Halide namespace. */ -EXPORT std::string get_source_location(); + /** Get the source location in the call stack, skipping over calls in + * the Halide namespace. */ + EXPORT std::string get_source_location(); -// This gets called automatically by anyone who includes Halide.h by -// the code below. It tests if this functionality works for the given -// compilation unit, and disables it if not. -EXPORT void test_compilation_unit(bool (*test)(), void (*calib)()); + // This gets called automatically by anyone who includes Halide.h by + // the code below. It tests if this functionality works for the given + // compilation unit, and disables it if not. + EXPORT void test_compilation_unit(bool (*test)(), void (*calib)()); -} -} +}// namespace Internal +}// namespace Halide // This code verifies that introspection is working before relying on @@ -246,16 +246,19 @@ EXPORT void test_compilation_unit(bool (*test)(), void (*calib)()); namespace Halide { namespace Internal { -static bool check_introspection(const void *var, const std::string &type, - const std::string &correct_name, - const std::string &correct_file, int line) { + static bool check_introspection(const void *var, + const std::string &type, + const std::string &correct_name, + const std::string &correct_file, + int line) + { std::string correct_loc = correct_file + ":" + int_to_string(line); std::string loc = get_source_location(); std::string name = get_variable_name(var, type); return name == correct_name && loc == correct_loc; -} -} -} + } +}// namespace Internal +}// namespace Halide namespace HalideIntrospectionCanary { @@ -263,60 +266,62 @@ namespace HalideIntrospectionCanary { // comparing it to the program counter listed in the debugging info, // we can calibrate for any offset between the debugging info and the // actual memory layout where the code was loaded. -static void offset_marker() { - std::cerr << "You should not have called this function\n"; -} +static void offset_marker() { std::cerr << "You should not have called this function\n"; } -struct A { - int an_int; +struct A +{ + int an_int; - class B { - int private_member; - public: - float a_float; - A *parent; - B() : private_member(17) { - a_float = private_member * 2.0f; - } - }; + class B + { + int private_member; - B a_b; + public: + float a_float; + A *parent; + B() : private_member(17) { a_float = private_member * 2.0f; } + }; - A() { - a_b.parent = this; - } + B a_b; + + A() { a_b.parent = this; } - bool test(const std::string &my_name); + bool test(const std::string &my_name); }; -static bool test_a(const A &a, const std::string &my_name) { - bool success = true; - success &= Halide::Internal::check_introspection(&a.an_int, "int", my_name + ".an_int", __FILE__ , __LINE__); - success &= Halide::Internal::check_introspection(&a.a_b, "HalideIntrospectionCanary::A::B", my_name + ".a_b", __FILE__ , __LINE__); - success &= Halide::Internal::check_introspection(&a.a_b.parent, "HalideIntrospectionCanary::A *", my_name + ".a_b.parent", __FILE__ , __LINE__); - success &= Halide::Internal::check_introspection(&a.a_b.a_float, "float", my_name + ".a_b.a_float", __FILE__ , __LINE__); - success &= Halide::Internal::check_introspection(a.a_b.parent, "HalideIntrospectionCanary::A", my_name, __FILE__ , __LINE__); - return success; +static bool test_a(const A &a, const std::string &my_name) +{ + bool success = true; + success &= Halide::Internal::check_introspection(&a.an_int, "int", my_name + ".an_int", __FILE__, __LINE__); + success &= Halide::Internal::check_introspection( + &a.a_b, "HalideIntrospectionCanary::A::B", my_name + ".a_b", __FILE__, __LINE__); + success &= Halide::Internal::check_introspection( + &a.a_b.parent, "HalideIntrospectionCanary::A *", my_name + ".a_b.parent", __FILE__, __LINE__); + success &= + Halide::Internal::check_introspection(&a.a_b.a_float, "float", my_name + ".a_b.a_float", __FILE__, __LINE__); + success &= + Halide::Internal::check_introspection(a.a_b.parent, "HalideIntrospectionCanary::A", my_name, __FILE__, __LINE__); + return success; } -static bool test() { - A a1, a2; +static bool test() +{ + A a1, a2; - return test_a(a1, "a1") && test_a(a2, "a2"); + return test_a(a1, "a1") && test_a(a2, "a2"); } // Run the tests, and calibrate for the PC offset at static initialization time. namespace { -struct TestCompilationUnit { - TestCompilationUnit() { - Halide::Internal::test_compilation_unit(&test, &offset_marker); - } -}; -} + struct TestCompilationUnit + { + TestCompilationUnit() { Halide::Internal::test_compilation_unit(&test, &offset_marker); } + }; +}// namespace static TestCompilationUnit test_object; -} +}// namespace HalideIntrospectionCanary #endif @@ -339,197 +344,193 @@ struct Expr; * be vectors of the same (by setting the 'width' field to something * larger than one). Front-end code shouldn't use vector * types. Instead vectorize a function. */ -struct Type { - /** The basic type code: signed integer, unsigned integer, or floating point */ - enum TypeCode { - Int, //!< signed integers - UInt, //!< unsigned integers - Float, //!< floating point numbers - Handle //!< opaque pointer type (void *) - } code; - - /** The number of bits of precision of a single scalar value of this type. */ - int bits; +struct Type +{ + /** The basic type code: signed integer, unsigned integer, or floating point */ + enum TypeCode { + Int,//!< signed integers + UInt,//!< unsigned integers + Float,//!< floating point numbers + Handle//!< opaque pointer type (void *) + } code; - /** The number of bytes required to store a single scalar value of this type. Ignores vector width. */ - int bytes() const {return (bits + 7) / 8;} + /** The number of bits of precision of a single scalar value of this type. */ + int bits; - /** How many elements (if a vector type). Should be 1 for scalar types. */ - int width; + /** The number of bytes required to store a single scalar value of this type. Ignores vector width. */ + int bytes() const { return (bits + 7) / 8; } - /** Is this type boolean (represented as UInt(1))? */ - bool is_bool() const {return code == UInt && bits == 1;} + /** How many elements (if a vector type). Should be 1 for scalar types. */ + int width; - /** Is this type a vector type? (width > 1) */ - bool is_vector() const {return width > 1;} + /** Is this type boolean (represented as UInt(1))? */ + bool is_bool() const { return code == UInt && bits == 1; } - /** Is this type a scalar type? (width == 1) */ - bool is_scalar() const {return width == 1;} + /** Is this type a vector type? (width > 1) */ + bool is_vector() const { return width > 1; } - /** Is this type a floating point type (float or double). */ - bool is_float() const {return code == Float;} + /** Is this type a scalar type? (width == 1) */ + bool is_scalar() const { return width == 1; } - /** Is this type a signed integer type? */ - bool is_int() const {return code == Int;} + /** Is this type a floating point type (float or double). */ + bool is_float() const { return code == Float; } - /** Is this type an unsigned integer type? */ - bool is_uint() const {return code == UInt;} + /** Is this type a signed integer type? */ + bool is_int() const { return code == Int; } - /** Is this type an opaque handle type (void *) */ - bool is_handle() const {return code == Handle;} + /** Is this type an unsigned integer type? */ + bool is_uint() const { return code == UInt; } - /** Compare two types for equality */ - bool operator==(const Type &other) const { - return code == other.code && bits == other.bits && width == other.width; - } + /** Is this type an opaque handle type (void *) */ + bool is_handle() const { return code == Handle; } - /** Compare two types for inequality */ - bool operator!=(const Type &other) const { - return code != other.code || bits != other.bits || width != other.width; - } + /** Compare two types for equality */ + bool operator==(const Type &other) const { return code == other.code && bits == other.bits && width == other.width; } - /** Produce a vector of this type, with 'width' elements */ - Type vector_of(int w) const { - Type type = {code, bits, w}; - return type; - } + /** Compare two types for inequality */ + bool operator!=(const Type &other) const { return code != other.code || bits != other.bits || width != other.width; } - /** Produce the type of a single element of this vector type */ - Type element_of() const { - Type type = {code, bits, 1}; - return type; - } + /** Produce a vector of this type, with 'width' elements */ + Type vector_of(int w) const + { + Type type = { code, bits, w }; + return type; + } + + /** Produce the type of a single element of this vector type */ + Type element_of() const + { + Type type = { code, bits, 1 }; + return type; + } - /** Can this type represent all values of another type? */ - EXPORT bool can_represent(Type other) const; + /** Can this type represent all values of another type? */ + EXPORT bool can_represent(Type other) const; - /** Return an integer which is the maximum value of this type. */ - EXPORT int imax() const; + /** Return an integer which is the maximum value of this type. */ + EXPORT int imax() const; - /** Return an expression which is the maximum value of this type */ - EXPORT Expr max() const; + /** Return an expression which is the maximum value of this type */ + EXPORT Expr max() const; - /** Return an integer which is the minimum value of this type */ - EXPORT int imin() const; + /** Return an integer which is the minimum value of this type */ + EXPORT int imin() const; - /** Return an expression which is the minimum value of this type */ - EXPORT Expr min() const; + /** Return an expression which is the minimum value of this type */ + EXPORT Expr min() const; }; /** Constructing a signed integer type */ -inline Type Int(int bits, int width = 1) { - Type t; - t.code = Type::Int; - t.bits = bits; - t.width = width; - return t; +inline Type Int(int bits, int width = 1) +{ + Type t; + t.code = Type::Int; + t.bits = bits; + t.width = width; + return t; } /** Constructing an unsigned integer type */ -inline Type UInt(int bits, int width = 1) { - Type t; - t.code = Type::UInt; - t.bits = bits; - t.width = width; - return t; +inline Type UInt(int bits, int width = 1) +{ + Type t; + t.code = Type::UInt; + t.bits = bits; + t.width = width; + return t; } /** Construct a floating-point type */ -inline Type Float(int bits, int width = 1) { - Type t; - t.code = Type::Float; - t.bits = bits; - t.width = width; - return t; +inline Type Float(int bits, int width = 1) +{ + Type t; + t.code = Type::Float; + t.bits = bits; + t.width = width; + return t; } /** Construct a boolean type */ -inline Type Bool(int width = 1) { - return UInt(1, width); -} +inline Type Bool(int width = 1) { return UInt(1, width); } /** Construct a handle type */ -inline Type Handle(int width = 1) { - Type t; - t.code = Type::Handle; - t.bits = 64; // All handles are 64-bit for now - t.width = width; - return t; +inline Type Handle(int width = 1) +{ + Type t; + t.code = Type::Handle; + t.bits = 64;// All handles are 64-bit for now + t.width = width; + return t; } namespace { -template -struct type_of_helper; - -template -struct type_of_helper { - operator Type() { - return Handle(); - } -}; - -template<> -struct type_of_helper { - operator Type() {return Float(32);} -}; - -template<> -struct type_of_helper { - operator Type() {return Float(64);} -}; - -template<> -struct type_of_helper { - operator Type() {return UInt(8);} -}; - -template<> -struct type_of_helper { - operator Type() {return UInt(16);} -}; - -template<> -struct type_of_helper { - operator Type() {return UInt(32);} -}; - -template<> -struct type_of_helper { - operator Type() {return UInt(64);} -}; - -template<> -struct type_of_helper { - operator Type() {return Int(8);} -}; - -template<> -struct type_of_helper { - operator Type() {return Int(16);} -}; - -template<> -struct type_of_helper { - operator Type() {return Int(32);} -}; - -template<> -struct type_of_helper { - operator Type() {return Int(64);} -}; - -template<> -struct type_of_helper { - operator Type() {return Bool();} -}; -} + template struct type_of_helper; + + template struct type_of_helper + { + operator Type() { return Handle(); } + }; + + template<> struct type_of_helper + { + operator Type() { return Float(32); } + }; + + template<> struct type_of_helper + { + operator Type() { return Float(64); } + }; + + template<> struct type_of_helper + { + operator Type() { return UInt(8); } + }; + + template<> struct type_of_helper + { + operator Type() { return UInt(16); } + }; + + template<> struct type_of_helper + { + operator Type() { return UInt(32); } + }; + + template<> struct type_of_helper + { + operator Type() { return UInt(64); } + }; + + template<> struct type_of_helper + { + operator Type() { return Int(8); } + }; + + template<> struct type_of_helper + { + operator Type() { return Int(16); } + }; + + template<> struct type_of_helper + { + operator Type() { return Int(32); } + }; + + template<> struct type_of_helper + { + operator Type() { return Int(64); } + }; + + template<> struct type_of_helper + { + operator Type() { return Bool(); } + }; +}// namespace /** Construct the halide equivalent of a C type */ -template Type type_of() { - return Type(type_of_helper()); -} +template Type type_of() { return Type(type_of_helper()); } -} +}// namespace Halide #endif #ifndef HALIDE_ARGUMENT_H @@ -549,32 +550,33 @@ namespace Halide { * function. Used for specifying the function signature of * generated code. */ -struct Argument { - /** The name of the argument */ - std::string name; - - /** An argument is either a primitive type (for parameters), or a - * buffer pointer. If 'is_buffer' is true, then 'type' should be - * ignored. - */ - bool is_buffer; - - /** For buffers, these two variables can be used to specify whether the - * buffer is read or written. By default, we assume that the argument - * buffer is read-write and set both flags. */ - bool read; - bool write; - - /** If this is a scalar parameter, then this is its type */ - Type type; - - Argument() : is_buffer(false) {} - Argument(const std::string &_name, bool _is_buffer, Type _type) : - name(_name), is_buffer(_is_buffer), type(_type) { - read = write = is_buffer; - } +struct Argument +{ + /** The name of the argument */ + std::string name; + + /** An argument is either a primitive type (for parameters), or a + * buffer pointer. If 'is_buffer' is true, then 'type' should be + * ignored. + */ + bool is_buffer; + + /** For buffers, these two variables can be used to specify whether the + * buffer is read or written. By default, we assume that the argument + * buffer is read-write and set both flags. */ + bool read; + bool write; + + /** If this is a scalar parameter, then this is its type */ + Type type; + + Argument() : is_buffer(false) {} + Argument(const std::string &_name, bool _is_buffer, Type _type) : name(_name), is_buffer(_is_buffer), type(_type) + { + read = write = is_buffer; + } }; -} +}// namespace Halide #endif #ifndef HALIDE_BOUNDS_H @@ -598,8 +600,8 @@ struct Argument { */ #include -#include #include +#include namespace Halide { @@ -612,66 +614,68 @@ EXPORT std::ostream &operator<<(std::ostream &stream, const Type &); namespace Internal { -struct Stmt; -std::ostream &operator<<(std::ostream &stream, const Stmt &); - -/** For optional debugging during codegen, use the debug class as - * follows: - * - \code - debug(verbosity) << "The expression is " << expr << std::endl; - \endcode - * - * verbosity of 0 always prints, 1 should print after every major - * stage, 2 should be used for more detail, and 3 should be used for - * tracing everything that occurs. The verbosity with which to print - * is determined by the value of the environment variable - * HL_DEBUG_CODEGEN - */ - -struct debug { + struct Stmt; + std::ostream &operator<<(std::ostream &stream, const Stmt &); + + /** For optional debugging during codegen, use the debug class as + * follows: + * + \code + debug(verbosity) << "The expression is " << expr << std::endl; + \endcode + * + * verbosity of 0 always prints, 1 should print after every major + * stage, 2 should be used for more detail, and 3 should be used for + * tracing everything that occurs. The verbosity with which to print + * is determined by the value of the environment variable + * HL_DEBUG_CODEGEN + */ + + struct debug + { EXPORT static int debug_level; EXPORT static bool initialized; int verbosity; - debug(int v) : verbosity(v) { - if (!initialized) { - // Read the debug level from the environment - #ifdef _WIN32 - char lvl[32]; - size_t read = 0; - getenv_s(&read, lvl, "HL_DEBUG_CODEGEN"); - if (read) { - #else - if (char *lvl = getenv("HL_DEBUG_CODEGEN")) { - #endif - debug_level = atoi(lvl); - } else { - debug_level = 0; - } - - initialized = true; + debug(int v) : verbosity(v) + { + if (!initialized) { +// Read the debug level from the environment +#ifdef _WIN32 + char lvl[32]; + size_t read = 0; + getenv_s(&read, lvl, "HL_DEBUG_CODEGEN"); + if (read) { +#else + if (char *lvl = getenv("HL_DEBUG_CODEGEN")) { +#endif + debug_level = atoi(lvl); + } else { + debug_level = 0; } + + initialized = true; + } } - template - debug &operator<<(T x) { - if (verbosity > debug_level) return *this; - std::cerr << x; - return *this; + template debug &operator<<(T x) + { + if (verbosity > debug_level) return *this; + std::cerr << x; + return *this; } -}; + }; -} -} +}// namespace Internal +}// namespace Halide #endif #ifndef HALIDE_ERROR_H #define HALIDE_ERROR_H -#include #include #include +#include namespace Halide { @@ -680,35 +684,40 @@ namespace Halide { EXPORT bool exceptions_enabled(); /** A base class for Halide errors. */ -struct Error : public std::runtime_error { - // Give each class a non-inlined constructor so that the type - // doesn't get separately instantiated in each compilation unit. - EXPORT Error(const std::string &msg); +struct Error : public std::runtime_error +{ + // Give each class a non-inlined constructor so that the type + // doesn't get separately instantiated in each compilation unit. + EXPORT Error(const std::string &msg); }; /** An error that occurs while running a JIT-compiled Halide pipeline. */ -struct RuntimeError : public Error { - EXPORT RuntimeError(const std::string &msg); +struct RuntimeError : public Error +{ + EXPORT RuntimeError(const std::string &msg); }; /** An error that occurs while compiling a Halide pipeline that Halide * attributes to a user error. */ -struct CompileError : public Error { - EXPORT CompileError(const std::string &msg); +struct CompileError : public Error +{ + EXPORT CompileError(const std::string &msg); }; /** An error that occurs while compiling a Halide pipeline that Halide * attributes to an internal compiler bug, or to an invalid use of * Halide's internals. */ -struct InternalError : public Error { - EXPORT InternalError(const std::string &msg); +struct InternalError : public Error +{ + EXPORT InternalError(const std::string &msg); }; namespace Internal { -struct ErrorReport { + struct ErrorReport + { std::ostringstream *msg; const char *file; const char *condition_string; @@ -718,53 +727,50 @@ struct ErrorReport { bool warning; bool runtime; - ErrorReport(const char *f, int l, const char *cs, bool c, bool u, bool w, bool r) : - msg(NULL), file(f), condition_string(cs), line(l), condition(c), user(u), warning(w), runtime(r) { - if (condition) return; - msg = new std::ostringstream; - const std::string &source_loc = get_source_location(); - - if (user) { - // Only mention where inside of libHalide the error tripped if we have debug level > 0 - debug(1) << "User error triggered at " << f << ":" << l << "\n"; - if (condition_string) { - debug(1) << "Condition failed: " << condition_string << "\n"; - } - if (warning) { - (*msg) << "Warning"; - } else { - (*msg) << "Error"; - } - if (source_loc.empty()) { - (*msg) << ":\n"; - } else { - (*msg) << " at " << source_loc << ":\n"; - } + ErrorReport(const char *f, int l, const char *cs, bool c, bool u, bool w, bool r) + : msg(NULL), file(f), condition_string(cs), line(l), condition(c), user(u), warning(w), runtime(r) + { + if (condition) return; + msg = new std::ostringstream; + const std::string &source_loc = get_source_location(); + + if (user) { + // Only mention where inside of libHalide the error tripped if we have debug level > 0 + debug(1) << "User error triggered at " << f << ":" << l << "\n"; + if (condition_string) { debug(1) << "Condition failed: " << condition_string << "\n"; } + if (warning) { + (*msg) << "Warning"; + } else { + (*msg) << "Error"; + } + if (source_loc.empty()) { + (*msg) << ":\n"; + } else { + (*msg) << " at " << source_loc << ":\n"; + } + } else { + (*msg) << "Internal "; + if (warning) { + (*msg) << "warning"; + } else { + (*msg) << "error"; + } + (*msg) << " at " << f << ":" << l; + if (!source_loc.empty()) { + (*msg) << " triggered by user code at " << source_loc << ":\n"; } else { - (*msg) << "Internal "; - if (warning) { - (*msg) << "warning"; - } else { - (*msg) << "error"; - } - (*msg) << " at " << f << ":" << l; - if (!source_loc.empty()) { - (*msg) << " triggered by user code at " << source_loc << ":\n"; - } else { - (*msg) << "\n"; - } - if (condition_string) { - (*msg) << "Condition failed: " << condition_string << "\n"; - } + (*msg) << "\n"; } + if (condition_string) { (*msg) << "Condition failed: " << condition_string << "\n"; } + } } - template - ErrorReport &operator<<(T x) { - if (condition) return *this; - (*msg) << x; - return *this; + template ErrorReport &operator<<(T x) + { + if (condition) return *this; + (*msg) << x; + return *this; } /** When you're done using << on the object, and let it fall out of @@ -776,46 +782,48 @@ struct ErrorReport { * flight already. */ #if __cplusplus >= 201100 - ~ErrorReport() noexcept(false) { + ~ErrorReport() noexcept(false) + { #else - ~ErrorReport() { + ~ErrorReport() + { #endif - if (condition) return; - explode(); + if (condition) return; + explode(); } EXPORT void explode(); -}; + }; -#define internal_error Halide::Internal::ErrorReport(__FILE__, __LINE__, NULL, false, false, false, false) -#define internal_assert(c) Halide::Internal::ErrorReport(__FILE__, __LINE__, #c, c, false, false, false) -#define user_error Halide::Internal::ErrorReport(__FILE__, __LINE__, NULL, false, true, false, false) -#define user_assert(c) Halide::Internal::ErrorReport(__FILE__, __LINE__, #c, c, true, false, false) -#define user_warning Halide::Internal::ErrorReport(__FILE__, __LINE__, NULL, false, true, true, false) -#define halide_runtime_error Halide::Internal::ErrorReport(__FILE__, __LINE__, NULL, false, true, false, true) +#define internal_error Halide::Internal::ErrorReport(__FILE__, __LINE__, NULL, false, false, false, false) +#define internal_assert(c) Halide::Internal::ErrorReport(__FILE__, __LINE__, #c, c, false, false, false) +#define user_error Halide::Internal::ErrorReport(__FILE__, __LINE__, NULL, false, true, false, false) +#define user_assert(c) Halide::Internal::ErrorReport(__FILE__, __LINE__, #c, c, true, false, false) +#define user_warning Halide::Internal::ErrorReport(__FILE__, __LINE__, NULL, false, true, true, false) +#define halide_runtime_error Halide::Internal::ErrorReport(__FILE__, __LINE__, NULL, false, true, false, true) // The nicely named versions get cleaned up at the end of Halide.h, // but user code might want to do halide-style user_asserts (e.g. the // Extern macros introduce calls to user_assert), so for that purpose // we define an equivalent macro that can be used outside of Halide.h -#define _halide_user_assert(c) Halide::Internal::ErrorReport(__FILE__, __LINE__, #c, c, true, false, false) +#define _halide_user_assert(c) Halide::Internal::ErrorReport(__FILE__, __LINE__, #c, c, true, false, false) -// N.B. Any function that might throw a user_assert or user_error may -// not be inlined into the user's code, or the line number will be -// misattributed to Halide.h. Either make such functions internal to -// libHalide, or mark them as NO_INLINE. + // N.B. Any function that might throw a user_assert or user_error may + // not be inlined into the user's code, or the line number will be + // misattributed to Halide.h. Either make such functions internal to + // libHalide, or mark them as NO_INLINE. -} +}// namespace Internal -} +}// namespace Halide #endif #ifndef HALIDE_IR_VISITOR_H #define HALIDE_IR_VISITOR_H -#include #include +#include #include /** \file @@ -828,56 +836,57 @@ struct Expr; namespace Internal { -struct IRNode; -struct Stmt; -struct IntImm; -struct FloatImm; -struct StringImm; -struct Cast; -struct Variable; -struct Add; -struct Sub; -struct Mul; -struct Div; -struct Mod; -struct Min; -struct Max; -struct EQ; -struct NE; -struct LT; -struct LE; -struct GT; -struct GE; -struct And; -struct Or; -struct Not; -struct Select; -struct Load; -struct Ramp; -struct Broadcast; -struct Call; -struct Let; -struct LetStmt; -struct AssertStmt; -struct Pipeline; -struct For; -struct Store; -struct Provide; -struct Allocate; -struct Free; -struct Realize; -struct Block; -struct IfThenElse; -struct Evaluate; - -class Function; - -/** A base class for algorithms that need to recursively walk over the - * IR. The default implementations just recursively walk over the - * children. Override the ones you care about. - */ -class IRVisitor { -public: + struct IRNode; + struct Stmt; + struct IntImm; + struct FloatImm; + struct StringImm; + struct Cast; + struct Variable; + struct Add; + struct Sub; + struct Mul; + struct Div; + struct Mod; + struct Min; + struct Max; + struct EQ; + struct NE; + struct LT; + struct LE; + struct GT; + struct GE; + struct And; + struct Or; + struct Not; + struct Select; + struct Load; + struct Ramp; + struct Broadcast; + struct Call; + struct Let; + struct LetStmt; + struct AssertStmt; + struct Pipeline; + struct For; + struct Store; + struct Provide; + struct Allocate; + struct Free; + struct Realize; + struct Block; + struct IfThenElse; + struct Evaluate; + + class Function; + + /** A base class for algorithms that need to recursively walk over the + * IR. The default implementations just recursively walk over the + * children. Override the ones you care about. + */ + class IRVisitor + { + public: virtual ~IRVisitor(); virtual void visit(const IntImm *); virtual void visit(const FloatImm *); @@ -918,13 +927,14 @@ class IRVisitor { virtual void visit(const Block *); virtual void visit(const IfThenElse *); virtual void visit(const Evaluate *); -}; - -/** A base class for algorithms that walk recursively over the IR - * without visiting the same node twice. This is for passes that are - * capable of interpreting the IR as a DAG instead of a tree. */ -class IRGraphVisitor : public IRVisitor { -protected: + }; + + /** A base class for algorithms that walk recursively over the IR + * without visiting the same node twice. This is for passes that are + * capable of interpreting the IR as a DAG instead of a tree. */ + class IRGraphVisitor : public IRVisitor + { + protected: /** By default these methods add the node to the visited set, and * return whether or not it was already there. If it wasn't there, * it delegates to the appropriate visit method. You can override @@ -937,8 +947,7 @@ class IRGraphVisitor : public IRVisitor { /** The nodes visited so far */ std::set visited; -public: - + public: /** These methods should call 'include' on the children to only * visit them if they haven't been visited already. */ // @{ @@ -982,10 +991,10 @@ class IRGraphVisitor : public IRVisitor { virtual void visit(const IfThenElse *); virtual void visit(const Evaluate *); // @} -}; + }; -} -} +}// namespace Internal +}// namespace Halide #endif #ifndef HALIDE_BUFFER_H @@ -1018,43 +1027,44 @@ class IRGraphVisitor : public IRVisitor { * Halide code. It includes some stuff to track whether the image is * not actually in main memory, but instead on a device (like a * GPU). */ -typedef struct buffer_t { - /** A device-handle for e.g. GPU memory used to back this buffer. */ - uint64_t dev; - - /** A pointer to the start of the data in main memory. */ - uint8_t* host; - - /** The size of the buffer in each dimension. */ - int32_t extent[4]; - - /** Gives the spacing in memory between adjacent elements in the - * given dimension. The correct memory address for a load from - * this buffer at position x, y, z, w is: - * host + (x * stride[0] + y * stride[1] + z * stride[2] + w * stride[3]) * elem_size - * By manipulating the strides and extents you can lazily crop, - * transpose, and even flip buffers without modifying the data. - */ - int32_t stride[4]; - - /** Buffers often represent evaluation of a Func over some - * domain. The min field encodes the top left corner of the - * domain. */ - int32_t min[4]; - - /** How many bytes does each buffer element take. This may be - * replaced with a more general type code in the future. */ - int32_t elem_size; - - /** This should be true if there is an existing device allocation - * mirroring this buffer, and the data has been modified on the - * host side. */ - bool host_dirty; - - /** This should be true if there is an existing device allocation - mirroring this buffer, and the data has been modified on the - device side. */ - bool dev_dirty; +typedef struct buffer_t +{ + /** A device-handle for e.g. GPU memory used to back this buffer. */ + uint64_t dev; + + /** A pointer to the start of the data in main memory. */ + uint8_t *host; + + /** The size of the buffer in each dimension. */ + int32_t extent[4]; + + /** Gives the spacing in memory between adjacent elements in the + * given dimension. The correct memory address for a load from + * this buffer at position x, y, z, w is: + * host + (x * stride[0] + y * stride[1] + z * stride[2] + w * stride[3]) * elem_size + * By manipulating the strides and extents you can lazily crop, + * transpose, and even flip buffers without modifying the data. + */ + int32_t stride[4]; + + /** Buffers often represent evaluation of a Func over some + * domain. The min field encodes the top left corner of the + * domain. */ + int32_t min[4]; + + /** How many bytes does each buffer element take. This may be + * replaced with a more general type code in the future. */ + int32_t elem_size; + + /** This should be true if there is an existing device allocation + * mirroring this buffer, and the data has been modified on the + * host side. */ + bool host_dirty; + + /** This should be true if there is an existing device allocation + mirroring this buffer, and the data has been modified on the + device side. */ + bool dev_dirty; } buffer_t; #endif @@ -1070,118 +1080,106 @@ typedef struct buffer_t { */ -#include #include +#include namespace Halide { namespace Internal { -/** A class representing a reference count to be used with IntrusivePtr */ -class RefCount { + /** A class representing a reference count to be used with IntrusivePtr */ + class RefCount + { int count; -public: - RefCount() : count(0) {} - void increment() {count++;} - void decrement() {count--;} - bool is_zero() const {return count == 0;} -}; - -/** - * Because in this header we don't yet know how client classes store - * their RefCount (and we don't want to depend on the declarations of - * the client classes), any class that you want to hold onto via one - * of these must provide implementations of ref_count and destroy, - * which we forward-declare here. - * - * E.g. if you want to use IntrusivePtr, then you should - * define something like this in MyClass.cpp (assuming MyClass has - * a field: mutable RefCount ref_count): - * - * template<> RefCount &ref_count(const MyClass *c) {return c->ref_count;} - * template<> void destroy(const MyClass *c) {delete c;} - */ -// @{ -template EXPORT RefCount &ref_count(const T *); -template EXPORT void destroy(const T *); -// @} - -/** Intrusive shared pointers have a reference count (a - * RefCount object) stored in the class itself. This is perhaps more - * efficient than storing it externally, but more importantly, it - * means it's possible to recover a reference-counted handle from the - * raw pointer, and it's impossible to have two different reference - * counts attached to the same raw object. Seeing as we pass around - * raw pointers to concrete IRNodes and Expr's interchangeably, this - * is a useful property. - */ -template -struct IntrusivePtr { -private: - void incref(T *p) { - if (p) { - ref_count(p).increment(); - } + public: + RefCount() : count(0) {} + void increment() { count++; } + void decrement() { count--; } + bool is_zero() const { return count == 0; } + }; + + /** + * Because in this header we don't yet know how client classes store + * their RefCount (and we don't want to depend on the declarations of + * the client classes), any class that you want to hold onto via one + * of these must provide implementations of ref_count and destroy, + * which we forward-declare here. + * + * E.g. if you want to use IntrusivePtr, then you should + * define something like this in MyClass.cpp (assuming MyClass has + * a field: mutable RefCount ref_count): + * + * template<> RefCount &ref_count(const MyClass *c) {return c->ref_count;} + * template<> void destroy(const MyClass *c) {delete c;} + */ + // @{ + template EXPORT RefCount &ref_count(const T *); + template EXPORT void destroy(const T *); + // @} + + /** Intrusive shared pointers have a reference count (a + * RefCount object) stored in the class itself. This is perhaps more + * efficient than storing it externally, but more importantly, it + * means it's possible to recover a reference-counted handle from the + * raw pointer, and it's impossible to have two different reference + * counts attached to the same raw object. Seeing as we pass around + * raw pointers to concrete IRNodes and Expr's interchangeably, this + * is a useful property. + */ + template struct IntrusivePtr + { + private: + void incref(T *p) + { + if (p) { ref_count(p).increment(); } }; - void decref(T *p) { - if (p) { - // Note that if the refcount is already zero, then we're - // in a recursive destructor due to a self-reference (a - // cycle), where the ref_count has been adjusted to remove - // the counts due to the cycle. The next line then makes - // the ref_count negative, which prevents actually - // entering the destructor recursively. - ref_count(p).decrement(); - if (ref_count(p).is_zero()) { - destroy(p); - } - } - } - -public: + void decref(T *p) + { + if (p) { + // Note that if the refcount is already zero, then we're + // in a recursive destructor due to a self-reference (a + // cycle), where the ref_count has been adjusted to remove + // the counts due to the cycle. The next line then makes + // the ref_count negative, which prevents actually + // entering the destructor recursively. + ref_count(p).decrement(); + if (ref_count(p).is_zero()) { destroy(p); } + } + } + + public: T *ptr; - ~IntrusivePtr() { - decref(ptr); - } + ~IntrusivePtr() { decref(ptr); } - IntrusivePtr() : ptr(NULL) { - } + IntrusivePtr() : ptr(NULL) {} - IntrusivePtr(T *p) : ptr(p) { - incref(ptr); - } + IntrusivePtr(T *p) : ptr(p) { incref(ptr); } - IntrusivePtr(const IntrusivePtr &other) : ptr(other.ptr) { - incref(ptr); - } + IntrusivePtr(const IntrusivePtr &other) : ptr(other.ptr) { incref(ptr); } - IntrusivePtr &operator=(const IntrusivePtr &other) { - // Other can be inside of something owned by this, so we - // should be careful to incref other before we decref - // ourselves. - T *temp = other.ptr; - incref(temp); - decref(ptr); - ptr = temp; - return *this; + IntrusivePtr &operator=(const IntrusivePtr &other) + { + // Other can be inside of something owned by this, so we + // should be careful to incref other before we decref + // ourselves. + T *temp = other.ptr; + incref(temp); + decref(ptr); + ptr = temp; + return *this; } /* Handles can be null. This checks that. */ - bool defined() const { - return ptr != NULL; - } + bool defined() const { return ptr != NULL; } /* Check if two handles point to the same ptr. This is * equality of reference, not equality of value. */ - bool same_as(const IntrusivePtr &other) const { - return ptr == other.ptr; - } - -}; + bool same_as(const IntrusivePtr &other) const { return ptr == other.ptr; } + }; -} -} +}// namespace Internal +}// namespace Halide #endif @@ -1191,9 +1189,9 @@ struct IntrusivePtr { namespace Halide { namespace Internal { -struct BufferContents; -struct JITCompiledModule; -} + struct BufferContents; + struct JITCompiledModule; +}// namespace Internal /** The internal representation of an image, or other dense array * data. The Image type provides a typed view onto a buffer for the @@ -1204,122 +1202,126 @@ struct JITCompiledModule; * wrapper on a buffer_t, which is the C-style type Halide uses for * passing buffers around. */ -class Buffer { +class Buffer +{ private: - Internal::IntrusivePtr contents; + Internal::IntrusivePtr contents; public: - Buffer() : contents(NULL) {} + Buffer() : contents(NULL) {} + + EXPORT Buffer(Type t, + int x_size = 0, + int y_size = 0, + int z_size = 0, + int w_size = 0, + uint8_t *data = NULL, + const std::string &name = ""); + + EXPORT Buffer(Type t, const std::vector &sizes, uint8_t *data = NULL, const std::string &name = ""); + + EXPORT Buffer(Type t, const buffer_t *buf, const std::string &name = ""); + + /** Get a pointer to the host-side memory. */ + EXPORT void *host_ptr() const; - EXPORT Buffer(Type t, int x_size = 0, int y_size = 0, int z_size = 0, int w_size = 0, - uint8_t* data = NULL, const std::string &name = ""); + /** Get a pointer to the raw buffer_t struct that this class wraps. */ + EXPORT buffer_t *raw_buffer() const; - EXPORT Buffer(Type t, const std::vector &sizes, - uint8_t* data = NULL, const std::string &name = ""); + /** Get the device-side pointer/handle for this buffer. Will be + * zero if no device was involved in the creation of this + * buffer. */ + EXPORT uint64_t device_handle() const; - EXPORT Buffer(Type t, const buffer_t *buf, const std::string &name = ""); + /** Has this buffer been modified on the cpu since last copied to a + * device. Not meaningful unless there's a device involved. */ + EXPORT bool host_dirty() const; - /** Get a pointer to the host-side memory. */ - EXPORT void *host_ptr() const; + /** Let Halide know that the host-side memory backing this buffer + * has been externally modified. You shouldn't normally need to + * call this, because it is done for you when you cast a Buffer to + * an Image in order to modify it. */ + EXPORT void set_host_dirty(bool dirty = true); - /** Get a pointer to the raw buffer_t struct that this class wraps. */ - EXPORT buffer_t *raw_buffer() const; + /** Has this buffer been modified on device since last copied to + * the cpu. Not meaninful unless there's a device involved. */ + EXPORT bool device_dirty() const; - /** Get the device-side pointer/handle for this buffer. Will be - * zero if no device was involved in the creation of this - * buffer. */ - EXPORT uint64_t device_handle() const; + /** Let Halide know that the device-side memory backing this + * buffer has been externally modified, and so the cpu-side memory + * is invalid. A copy-back will occur the next time you cast this + * Buffer to an Image, or the next time this buffer is accessed on + * the host in a halide pipeline. */ + EXPORT void set_device_dirty(bool dirty = true); - /** Has this buffer been modified on the cpu since last copied to a - * device. Not meaningful unless there's a device involved. */ - EXPORT bool host_dirty() const; + /** Get the dimensionality of this buffer. Uses the convention + * that the extent field of a buffer_t should contain zero when + * the dimensions end. */ + EXPORT int dimensions() const; - /** Let Halide know that the host-side memory backing this buffer - * has been externally modified. You shouldn't normally need to - * call this, because it is done for you when you cast a Buffer to - * an Image in order to modify it. */ - EXPORT void set_host_dirty(bool dirty = true); + /** Get the extent of this buffer in the given dimension. */ + EXPORT int extent(int dim) const; - /** Has this buffer been modified on device since last copied to - * the cpu. Not meaninful unless there's a device involved. */ - EXPORT bool device_dirty() const; + /** Get the number of bytes between adjacent elements of this buffer along the given dimension. */ + EXPORT int stride(int dim) const; - /** Let Halide know that the device-side memory backing this - * buffer has been externally modified, and so the cpu-side memory - * is invalid. A copy-back will occur the next time you cast this - * Buffer to an Image, or the next time this buffer is accessed on - * the host in a halide pipeline. */ - EXPORT void set_device_dirty(bool dirty = true); + /** Get the coordinate in the function that this buffer represents + * that corresponds to the base address of the buffer. */ + EXPORT int min(int dim) const; - /** Get the dimensionality of this buffer. Uses the convention - * that the extent field of a buffer_t should contain zero when - * the dimensions end. */ - EXPORT int dimensions() const; + /** Set the coordinate in the function that this buffer represents + * that corresponds to the base address of the buffer. */ + EXPORT void set_min(int m0, int m1 = 0, int m2 = 0, int m3 = 0); - /** Get the extent of this buffer in the given dimension. */ - EXPORT int extent(int dim) const; + /** Get the Halide type of the contents of this buffer. */ + EXPORT Type type() const; - /** Get the number of bytes between adjacent elements of this buffer along the given dimension. */ - EXPORT int stride(int dim) const; + /** Compare two buffers for identity (not equality of data). */ + EXPORT bool same_as(const Buffer &other) const; - /** Get the coordinate in the function that this buffer represents - * that corresponds to the base address of the buffer. */ - EXPORT int min(int dim) const; + /** Check if this buffer handle actually points to data. */ + EXPORT bool defined() const; - /** Set the coordinate in the function that this buffer represents - * that corresponds to the base address of the buffer. */ - EXPORT void set_min(int m0, int m1 = 0, int m2 = 0, int m3 = 0); + /** Get the runtime name of this buffer used for debugging. */ + EXPORT const std::string &name() const; - /** Get the Halide type of the contents of this buffer. */ - EXPORT Type type() const; + /** Convert this buffer to an argument to a halide pipeline. */ + EXPORT operator Argument() const; - /** Compare two buffers for identity (not equality of data). */ - EXPORT bool same_as(const Buffer &other) const; + /** Declare that this buffer was created by the given jit-compiled + * module. Used internally for reference counting the module. */ + EXPORT void set_source_module(const Internal::JITCompiledModule &module); - /** Check if this buffer handle actually points to data. */ - EXPORT bool defined() const; + /** If this buffer was the output of a jit-compiled realization, + * retrieve the module it came from. Otherwise returns a module + * struct full of null pointers. */ + EXPORT const Internal::JITCompiledModule &source_module(); - /** Get the runtime name of this buffer used for debugging. */ - EXPORT const std::string &name() const; + /** If this buffer was created *on-device* by a jit-compiled + * realization, then copy it back to the cpu-side memory. This is + * usually achieved by casting the Buffer to an Image. */ + EXPORT int copy_to_host(); - /** Convert this buffer to an argument to a halide pipeline. */ - EXPORT operator Argument() const; - - /** Declare that this buffer was created by the given jit-compiled - * module. Used internally for reference counting the module. */ - EXPORT void set_source_module(const Internal::JITCompiledModule &module); - - /** If this buffer was the output of a jit-compiled realization, - * retrieve the module it came from. Otherwise returns a module - * struct full of null pointers. */ - EXPORT const Internal::JITCompiledModule &source_module(); - - /** If this buffer was created *on-device* by a jit-compiled - * realization, then copy it back to the cpu-side memory. This is - * usually achieved by casting the Buffer to an Image. */ - EXPORT int copy_to_host(); - - /** If this buffer was created by a jit-compiled realization on a - * device-aware target (e.g. PTX), then copy the cpu-side data to - * the device-side allocation. TODO: I believe this currently - * aborts messily if no device-side allocation exists. You might - * think you want to do this because you've modified the data - * manually on the host before calling another Halide pipeline, - * but what you actually want to do in that situation is set the - * host_dirty bit so that Halide can manage the copy lazily for - * you. Casting the Buffer to an Image sets the dirty bit for - * you. */ - EXPORT int copy_to_dev(); - - /** If this buffer was created by a jit-compiled realization on a - * device-aware target (e.g. PTX), then free the device-side - * allocation, if there is one. Done automatically when the last - * reference to this buffer dies. */ - EXPORT int free_dev_buffer(); + /** If this buffer was created by a jit-compiled realization on a + * device-aware target (e.g. PTX), then copy the cpu-side data to + * the device-side allocation. TODO: I believe this currently + * aborts messily if no device-side allocation exists. You might + * think you want to do this because you've modified the data + * manually on the host before calling another Halide pipeline, + * but what you actually want to do in that situation is set the + * host_dirty bit so that Halide can manage the copy lazily for + * you. Casting the Buffer to an Image sets the dirty bit for + * you. */ + EXPORT int copy_to_dev(); + /** If this buffer was created by a jit-compiled realization on a + * device-aware target (e.g. PTX), then free the device-side + * allocation, if there is one. Done automatically when the last + * reference to this buffer dies. */ + EXPORT int free_dev_buffer(); }; -} +}// namespace Halide #endif @@ -1327,12 +1329,15 @@ namespace Halide { namespace Internal { -/** A class representing a type of IR node (e.g. Add, or Mul, or - * For). We use it for rtti (without having to compile with rtti). */ -struct IRNodeType {}; + /** A class representing a type of IR node (e.g. Add, or Mul, or + * For). We use it for rtti (without having to compile with rtti). */ + struct IRNodeType + { + }; -/** The abstract base classes for a node in the Halide IR. */ -struct IRNode { + /** The abstract base classes for a node in the Halide IR. */ + struct IRNode + { /** We use the visitor pattern to traverse IR nodes throughout the * compiler, so we have a virtual accept method which accepts @@ -1355,68 +1360,63 @@ struct IRNode { * often breaks when linking external libraries compiled * without it), and we only want it for IR nodes. */ virtual const IRNodeType *type_info() const = 0; -}; + }; -template<> -EXPORT inline RefCount &ref_count(const IRNode *n) {return n->ref_count;} + template<> EXPORT inline RefCount &ref_count(const IRNode *n) { return n->ref_count; } -template<> -EXPORT inline void destroy(const IRNode *n) {delete n;} + template<> EXPORT inline void destroy(const IRNode *n) { delete n; } -/** IR nodes are split into expressions and statements. These are - similar to expressions and statements in C - expressions - represent some value and have some type (e.g. x + 3), and - statements are side-effecting pieces of code that do not - represent a value (e.g. assert(x > 3)) */ + /** IR nodes are split into expressions and statements. These are + similar to expressions and statements in C - expressions + represent some value and have some type (e.g. x + 3), and + statements are side-effecting pieces of code that do not + represent a value (e.g. assert(x > 3)) */ -/** A base class for statement nodes. They have no properties or - methods beyond base IR nodes for now */ -struct BaseStmtNode : public IRNode { -}; + /** A base class for statement nodes. They have no properties or + methods beyond base IR nodes for now */ + struct BaseStmtNode : public IRNode + { + }; -/** A base class for expression nodes. They all contain their types - * (e.g. Int(32), Float(32)) */ -struct BaseExprNode : public IRNode { + /** A base class for expression nodes. They all contain their types + * (e.g. Int(32), Float(32)) */ + struct BaseExprNode : public IRNode + { Type type; -}; - -/** We use the "curiously recurring template pattern" to avoid - duplicated code in the IR Nodes. These classes live between the - abstract base classes and the actual IR Nodes in the - inheritance hierarchy. It provides an implementation of the - accept function necessary for the visitor pattern to work, and - a concrete instantiation of a unique IRNodeType per class. */ -template -struct ExprNode : public BaseExprNode { - void accept(IRVisitor *v) const { - v->visit((const T *)this); - } - virtual IRNodeType *type_info() const {return &_type_info;} + }; + + /** We use the "curiously recurring template pattern" to avoid + duplicated code in the IR Nodes. These classes live between the + abstract base classes and the actual IR Nodes in the + inheritance hierarchy. It provides an implementation of the + accept function necessary for the visitor pattern to work, and + a concrete instantiation of a unique IRNodeType per class. */ + template struct ExprNode : public BaseExprNode + { + void accept(IRVisitor *v) const { v->visit((const T *)this); } + virtual IRNodeType *type_info() const { return &_type_info; } static EXPORT IRNodeType _type_info; -}; + }; -template -struct StmtNode : public BaseStmtNode { - void accept(IRVisitor *v) const { - v->visit((const T *)this); - } - virtual IRNodeType *type_info() const {return &_type_info;} + template struct StmtNode : public BaseStmtNode + { + void accept(IRVisitor *v) const { v->visit((const T *)this); } + virtual IRNodeType *type_info() const { return &_type_info; } static EXPORT IRNodeType _type_info; -}; + }; -/** IR nodes are passed around opaque handles to them. This is a - base class for those handles. It manages the reference count, - and dispatches visitors. */ -struct IRHandle : public IntrusivePtr { + /** IR nodes are passed around opaque handles to them. This is a + base class for those handles. It manages the reference count, + and dispatches visitors. */ + struct IRHandle : public IntrusivePtr + { IRHandle() : IntrusivePtr() {} IRHandle(const IRNode *p) : IntrusivePtr(p) {} /** Dispatch to the correct visitor method for this node. E.g. if * this node is actually an Add node, then this will call * IRVisitor::visit(const Add *) */ - void accept(IRVisitor *v) const { - ptr->accept(v); - } + void accept(IRVisitor *v) const { ptr->accept(v); } /** Downcast this ir node to its actual type (e.g. Add, or * Select). This returns NULL if the node is not of the requested @@ -1426,108 +1426,106 @@ struct IRHandle : public IntrusivePtr { * // This is an add node * } */ - template const T *as() const { - if (ptr->type_info() == &T::_type_info) { - return (const T *)ptr; - } - return NULL; + template const T *as() const + { + if (ptr->type_info() == &T::_type_info) { return (const T *)ptr; } + return NULL; } -}; + }; -/** Integer constants */ -struct IntImm : public ExprNode { + /** Integer constants */ + struct IntImm : public ExprNode + { int value; - static IntImm *make(int value) { - if (value >= -8 && value <= 8 && - !small_int_cache[value + 8].ref_count.is_zero()) { - return &small_int_cache[value + 8]; - } - IntImm *node = new IntImm; - node->type = Int(32); - node->value = value; - return node; + static IntImm *make(int value) + { + if (value >= -8 && value <= 8 && !small_int_cache[value + 8].ref_count.is_zero()) { + return &small_int_cache[value + 8]; + } + IntImm *node = new IntImm; + node->type = Int(32); + node->value = value; + return node; } -private: + private: /** ints from -8 to 8 */ static IntImm small_int_cache[17]; -}; + }; -/** Floating point constants */ -struct FloatImm : public ExprNode { + /** Floating point constants */ + struct FloatImm : public ExprNode + { float value; - static FloatImm *make(float value) { - FloatImm *node = new FloatImm; - node->type = Float(32); - node->value = value; - return node; + static FloatImm *make(float value) + { + FloatImm *node = new FloatImm; + node->type = Float(32); + node->value = value; + return node; } -}; + }; -/** String constants */ -struct StringImm : public ExprNode { + /** String constants */ + struct StringImm : public ExprNode + { std::string value; - static StringImm *make(const std::string &val) { - StringImm *node = new StringImm; - node->type = Handle(); - node->value = val; - return node; + static StringImm *make(const std::string &val) + { + StringImm *node = new StringImm; + node->type = Handle(); + node->value = val; + return node; } -}; + }; -} +}// namespace Internal /** A fragment of Halide syntax. It's implemented as reference-counted * handle to a concrete expression node, but it's immutable, so you * can treat it as a value type. */ -struct Expr : public Internal::IRHandle { - /** Make an undefined expression */ - Expr() : Internal::IRHandle() {} +struct Expr : public Internal::IRHandle +{ + /** Make an undefined expression */ + Expr() : Internal::IRHandle() {} - /** Make an expression from a concrete expression node pointer (e.g. Add) */ - Expr(const Internal::BaseExprNode *n) : IRHandle(n) {} + /** Make an expression from a concrete expression node pointer (e.g. Add) */ + Expr(const Internal::BaseExprNode *n) : IRHandle(n) {} - /** Make an expression representing a const 32-bit int (i.e. an IntImm) */ - EXPORT Expr(int x) : IRHandle(Internal::IntImm::make(x)) { - } + /** Make an expression representing a const 32-bit int (i.e. an IntImm) */ + EXPORT Expr(int x) : IRHandle(Internal::IntImm::make(x)) {} - /** Make an expression representing a const 32-bit float (i.e. a FloatImm) */ - EXPORT Expr(float x) : IRHandle(Internal::FloatImm::make(x)) { - } + /** Make an expression representing a const 32-bit float (i.e. a FloatImm) */ + EXPORT Expr(float x) : IRHandle(Internal::FloatImm::make(x)) {} - /** Make an expression representing a const 32-bit float, given a - * double. Also emits a warning due to truncation. */ - EXPORT Expr(double x) : IRHandle(Internal::FloatImm::make((float)x)) { - user_warning << "Halide cannot represent double constants. " - << "Converting " << x << " to float. " - << "If you wanted a double, use cast(" << x - << (x == (int64_t)(x) ? ".0f" : "f") - << ")\n"; - } + /** Make an expression representing a const 32-bit float, given a + * double. Also emits a warning due to truncation. */ + EXPORT Expr(double x) : IRHandle(Internal::FloatImm::make((float)x)) + { + user_warning << "Halide cannot represent double constants. " + << "Converting " << x << " to float. " + << "If you wanted a double, use cast(" << x << (x == (int64_t)(x) ? ".0f" : "f") << ")\n"; + } - /** Make an expression representing a const string (i.e. a StringImm) */ - EXPORT Expr(const std::string &s) : IRHandle(Internal::StringImm::make(s)) { - } + /** Make an expression representing a const string (i.e. a StringImm) */ + EXPORT Expr(const std::string &s) : IRHandle(Internal::StringImm::make(s)) {} - /** Get the type of this expression node */ - Type type() const { - return ((const Internal::BaseExprNode *)ptr)->type; - } + /** Get the type of this expression node */ + Type type() const { return ((const Internal::BaseExprNode *)ptr)->type; } }; /** This lets you use an Expr as a key in a map of the form * map */ -struct ExprCompare { - bool operator()(Expr a, Expr b) const { - return a.ptr < b.ptr; - } +struct ExprCompare +{ + bool operator()(Expr a, Expr b) const { return a.ptr < b.ptr; } }; -} +}// namespace Halide // Now that we've defined an Expr, we can include Parameter.h #ifndef HALIDE_PARAMETER_H @@ -1542,11 +1540,12 @@ struct ExprCompare { namespace Halide { namespace Internal { -struct ParameterContents; + struct ParameterContents; -/** A reference-counted handle to a parameter to a halide - * pipeline. May be a scalar parameter or a buffer */ -class Parameter { + /** A reference-counted handle to a parameter to a halide + * pipeline. May be a scalar parameter or a buffer */ + class Parameter + { IntrusivePtr contents; void check_defined() const; @@ -1554,7 +1553,7 @@ class Parameter { void check_is_scalar() const; void check_dim_ok(int dim) const; -public: + public: /** Construct a new undefined handle */ Parameter() : contents(NULL) {} @@ -1581,22 +1580,20 @@ class Parameter { /** If the parameter is a scalar parameter, get its currently * bound value. Only relevant when jitting */ - template - NO_INLINE T get_scalar() { - user_assert(type() == type_of()) - << "Can't get Param<" << type() - << "> as scalar of type " << type_of() << "\n"; - return *((T *)(get_scalar_address())); + template NO_INLINE T get_scalar() + { + user_assert(type() == type_of()) + << "Can't get Param<" << type() << "> as scalar of type " << type_of() << "\n"; + return *((T *)(get_scalar_address())); } /** If the parameter is a scalar parameter, set its current * value. Only relevant when jitting */ - template - NO_INLINE void set_scalar(T val) { - user_assert(type() == type_of()) - << "Can't set Param<" << type() - << "> to scalar of type " << type_of() << "\n"; - *((T *)(get_scalar_address())) = val; + template NO_INLINE void set_scalar(T val) + { + user_assert(type() == type_of()) + << "Can't set Param<" << type() << "> to scalar of type " << type_of() << "\n"; + *((T *)(get_scalar_address())) = val; } /** If the parameter is a buffer parameter, get its currently @@ -1637,174 +1634,193 @@ class Parameter { EXPORT void set_max_value(Expr e); EXPORT Expr get_max_value(); // @} -}; + }; -/** Validate arguments to a call to a func, image or imageparam. */ -void check_call_arg_types(const std::string &name, std::vector *args, int dims); + /** Validate arguments to a call to a func, image or imageparam. */ + void check_call_arg_types(const std::string &name, std::vector *args, int dims); -} -} +}// namespace Internal +}// namespace Halide #endif namespace Halide { namespace Internal { -/** A reference-counted handle to a statement node. */ -struct Stmt : public IRHandle { + /** A reference-counted handle to a statement node. */ + struct Stmt : public IRHandle + { Stmt() : IRHandle() {} Stmt(const BaseStmtNode *n) : IRHandle(n) {} /** This lets you use a Stmt as a key in a map of the form * map */ - struct Compare { - bool operator()(const Stmt &a, const Stmt &b) const { - return a.ptr < b.ptr; - } + struct Compare + { + bool operator()(const Stmt &a, const Stmt &b) const { return a.ptr < b.ptr; } }; -}; + }; -/** The actual IR nodes begin here. Remember that all the Expr - * nodes also have a public "type" property */ + /** The actual IR nodes begin here. Remember that all the Expr + * nodes also have a public "type" property */ -} +}// namespace Internal namespace Internal { -/** Cast a node from one type to another */ -struct Cast : public ExprNode { + /** Cast a node from one type to another */ + struct Cast : public ExprNode + { Expr value; EXPORT static Expr make(Type t, Expr v); -}; + }; -/** The sum of two expressions */ -struct Add : public ExprNode { + /** The sum of two expressions */ + struct Add : public ExprNode + { Expr a, b; EXPORT static Expr make(Expr a, Expr b); -}; + }; -/** The difference of two expressions */ -struct Sub : public ExprNode { + /** The difference of two expressions */ + struct Sub : public ExprNode + { Expr a, b; EXPORT static Expr make(Expr a, Expr b); -}; + }; -/** The product of two expressions */ -struct Mul : public ExprNode { + /** The product of two expressions */ + struct Mul : public ExprNode + { Expr a, b; EXPORT static Expr make(Expr a, Expr b); -}; + }; -/** The ratio of two expressions */ -struct Div : public ExprNode
{ + /** The ratio of two expressions */ + struct Div : public ExprNode
+ { Expr a, b; EXPORT static Expr make(Expr a, Expr b); -}; + }; -/** The remainder of a / b. Mostly equivalent to '%' in C, except that - * the result here is always positive. For floats, this is equivalent - * to calling fmod. */ -struct Mod : public ExprNode { + /** The remainder of a / b. Mostly equivalent to '%' in C, except that + * the result here is always positive. For floats, this is equivalent + * to calling fmod. */ + struct Mod : public ExprNode + { Expr a, b; EXPORT static Expr make(Expr a, Expr b); -}; + }; -/** The lesser of two values. */ -struct Min : public ExprNode { + /** The lesser of two values. */ + struct Min : public ExprNode + { Expr a, b; EXPORT static Expr make(Expr a, Expr b); -}; + }; -/** The greater of two values */ -struct Max : public ExprNode { + /** The greater of two values */ + struct Max : public ExprNode + { Expr a, b; EXPORT static Expr make(Expr a, Expr b); -}; + }; -/** Is the first expression equal to the second */ -struct EQ : public ExprNode { + /** Is the first expression equal to the second */ + struct EQ : public ExprNode + { Expr a, b; EXPORT static Expr make(Expr a, Expr b); -}; + }; -/** Is the first expression not equal to the second */ -struct NE : public ExprNode { + /** Is the first expression not equal to the second */ + struct NE : public ExprNode + { Expr a, b; EXPORT static Expr make(Expr a, Expr b); -}; + }; -/** Is the first expression less than the second. */ -struct LT : public ExprNode { + /** Is the first expression less than the second. */ + struct LT : public ExprNode + { Expr a, b; EXPORT static Expr make(Expr a, Expr b); -}; + }; -/** Is the first expression less than or equal to the second. */ -struct LE : public ExprNode { + /** Is the first expression less than or equal to the second. */ + struct LE : public ExprNode + { Expr a, b; EXPORT static Expr make(Expr a, Expr b); -}; + }; -/** Is the first expression greater than the second. */ -struct GT : public ExprNode { + /** Is the first expression greater than the second. */ + struct GT : public ExprNode + { Expr a, b; EXPORT static Expr make(Expr a, Expr b); -}; + }; -/** Is the first expression greater than or equal to the second. */ -struct GE : public ExprNode { + /** Is the first expression greater than or equal to the second. */ + struct GE : public ExprNode + { Expr a, b; EXPORT static Expr make(Expr a, Expr b); -}; + }; -/** Logical and - are both expressions true */ -struct And : public ExprNode { + /** Logical and - are both expressions true */ + struct And : public ExprNode + { Expr a, b; EXPORT static Expr make(Expr a, Expr b); -}; + }; -/** Logical or - is at least one of the expression true */ -struct Or : public ExprNode { + /** Logical or - is at least one of the expression true */ + struct Or : public ExprNode + { Expr a, b; EXPORT static Expr make(Expr a, Expr b); -}; + }; -/** Logical not - true if the expression false */ -struct Not : public ExprNode { + /** Logical not - true if the expression false */ + struct Not : public ExprNode + { Expr a; EXPORT static Expr make(Expr a); -}; + }; -/** A ternary operator. Evalutes 'true_value' and 'false_value', - * then selects between them based on 'condition'. Equivalent to - * the ternary operator in C. */ -struct Select : public ExprNode + { Expr condition, true_value, false_value; EXPORT static Expr make(Expr condition, Expr true_value, Expr false_value); -}; + }; -/** Load a value from a named buffer. The buffer is treated as an - * array of the 'type' of this Load node. That is, the buffer has - * no inherent type. */ -struct Load : public ExprNode { + /** Load a value from a named buffer. The buffer is treated as an + * array of the 'type' of this Load node. That is, the buffer has + * no inherent type. */ + struct Load : public ExprNode + { std::string name; Expr index; @@ -1817,53 +1833,58 @@ struct Load : public ExprNode { Parameter param; EXPORT static Expr make(Type type, std::string name, Expr index, Buffer image, Parameter param); -}; - -/** A linear ramp vector node. This is vector with 'width' elements, - * where element i is 'base' + i*'stride'. This is a convenient way to - * pass around vectors without busting them up into individual - * elements. E.g. a dense vector load from a buffer can use a ramp - * node with stride 1 as the index. */ -struct Ramp : public ExprNode { + }; + + /** A linear ramp vector node. This is vector with 'width' elements, + * where element i is 'base' + i*'stride'. This is a convenient way to + * pass around vectors without busting them up into individual + * elements. E.g. a dense vector load from a buffer can use a ramp + * node with stride 1 as the index. */ + struct Ramp : public ExprNode + { Expr base, stride; int width; EXPORT static Expr make(Expr base, Expr stride, int width); -}; + }; -/** A vector with 'width' elements, in which every element is - * 'value'. This is a special case of the ramp node above, in which - * the stride is zero. */ -struct Broadcast : public ExprNode { + /** A vector with 'width' elements, in which every element is + * 'value'. This is a special case of the ramp node above, in which + * the stride is zero. */ + struct Broadcast : public ExprNode + { Expr value; int width; EXPORT static Expr make(Expr value, int width); -}; + }; -/** A let expression, like you might find in a functional - * language. Within the expression \ref Let::body, instances of the Var - * node \ref Let::name refer to \ref Let::value. */ -struct Let : public ExprNode { + /** A let expression, like you might find in a functional + * language. Within the expression \ref Let::body, instances of the Var + * node \ref Let::name refer to \ref Let::value. */ + struct Let : public ExprNode + { std::string name; Expr value, body; EXPORT static Expr make(std::string name, Expr value, Expr body); -}; + }; -/** The statement form of a let node. Within the statement 'body', - * instances of the Var named 'name' refer to 'value' */ -struct LetStmt : public StmtNode { + /** The statement form of a let node. Within the statement 'body', + * instances of the Var named 'name' refer to 'value' */ + struct LetStmt : public StmtNode + { std::string name; Expr value; Stmt body; EXPORT static Stmt make(std::string name, Expr value, Stmt body); -}; + }; -/** If the 'condition' is false, then bail out printing the - * message to stderr */ -struct AssertStmt : public StmtNode { + /** If the 'condition' is false, then bail out printing the + * message to stderr */ + struct AssertStmt : public StmtNode + { // if condition then val else error out with message Expr condition; Expr message; @@ -1871,140 +1892,152 @@ struct AssertStmt : public StmtNode { EXPORT static Stmt make(Expr condition, const char *message); EXPORT static Stmt make(Expr condition, Expr message); EXPORT static Stmt make(Expr condition, const std::vector &message); -}; - -/** This node is a helpful annotation to do with permissions. The - * three child statements happen in order. In the 'produce' - * statement 'buffer' is write-only. In 'update' it is - * read-write. In 'consume' it is read-only. The 'update' node is - * often NULL. (check update.defined() to find out). None of this - * is actually enforced, the node is purely for informative - * purposes to help out our analysis during lowering. */ -struct Pipeline : public StmtNode { + }; + + /** This node is a helpful annotation to do with permissions. The + * three child statements happen in order. In the 'produce' + * statement 'buffer' is write-only. In 'update' it is + * read-write. In 'consume' it is read-only. The 'update' node is + * often NULL. (check update.defined() to find out). None of this + * is actually enforced, the node is purely for informative + * purposes to help out our analysis during lowering. */ + struct Pipeline : public StmtNode + { std::string name; Stmt produce, update, consume; EXPORT static Stmt make(std::string name, Stmt produce, Stmt update, Stmt consume); -}; - -/** A for loop. Execute the 'body' statement for all values of the - * variable 'name' from 'min' to 'min + extent'. There are four - * types of For nodes. A 'Serial' for loop is a conventional - * one. In a 'Parallel' for loop, each iteration of the loop - * happens in parallel or in some unspecified order. In a - * 'Vectorized' for loop, each iteration maps to one SIMD lane, - * and the whole loop is executed in one shot. For this case, - * 'extent' must be some small integer constant (probably 4, 8, or - * 16). An 'Unrolled' for loop compiles to a completely unrolled - * version of the loop. Each iteration becomes its own - * statement. Again in this case, 'extent' should be a small - * integer constant. */ -struct For : public StmtNode { + }; + + /** A for loop. Execute the 'body' statement for all values of the + * variable 'name' from 'min' to 'min + extent'. There are four + * types of For nodes. A 'Serial' for loop is a conventional + * one. In a 'Parallel' for loop, each iteration of the loop + * happens in parallel or in some unspecified order. In a + * 'Vectorized' for loop, each iteration maps to one SIMD lane, + * and the whole loop is executed in one shot. For this case, + * 'extent' must be some small integer constant (probably 4, 8, or + * 16). An 'Unrolled' for loop compiles to a completely unrolled + * version of the loop. Each iteration becomes its own + * statement. Again in this case, 'extent' should be a small + * integer constant. */ + struct For : public StmtNode + { std::string name; Expr min, extent; - typedef enum {Serial, Parallel, Vectorized, Unrolled} ForType; + typedef enum { Serial, Parallel, Vectorized, Unrolled } ForType; ForType for_type; Stmt body; EXPORT static Stmt make(std::string name, Expr min, Expr extent, ForType for_type, Stmt body); -}; + }; -/** Store a 'value' to the buffer called 'name' at a given - * 'index'. The buffer is interpreted as an array of the same type as - * 'value'. */ -struct Store : public StmtNode { + /** Store a 'value' to the buffer called 'name' at a given + * 'index'. The buffer is interpreted as an array of the same type as + * 'value'. */ + struct Store : public StmtNode + { std::string name; Expr value, index; EXPORT static Stmt make(std::string name, Expr value, Expr index); -}; - -/** This defines the value of a function at a multi-dimensional - * location. You should think of it as a store to a - * multi-dimensional array. It gets lowered to a conventional - * Store node. */ -struct Provide : public StmtNode { + }; + + /** This defines the value of a function at a multi-dimensional + * location. You should think of it as a store to a + * multi-dimensional array. It gets lowered to a conventional + * Store node. */ + struct Provide : public StmtNode + { std::string name; std::vector values; std::vector args; EXPORT static Stmt make(std::string name, const std::vector &values, const std::vector &args); -}; - -/** Allocate a scratch area called with the given name, type, and - * size. The buffer lives for at most the duration of the body - * statement, within which it is freed. It is an error for an allocate - * node not to contain a free node of the same buffer. */ -struct Allocate : public StmtNode { + }; + + /** Allocate a scratch area called with the given name, type, and + * size. The buffer lives for at most the duration of the body + * statement, within which it is freed. It is an error for an allocate + * node not to contain a free node of the same buffer. */ + struct Allocate : public StmtNode + { std::string name; Type type; std::vector extents; Expr condition; Stmt body; - EXPORT static Stmt make(std::string name, Type type, const std::vector &extents, - Expr condition, Stmt body); -}; + EXPORT static Stmt make(std::string name, Type type, const std::vector &extents, Expr condition, Stmt body); + }; -/** Free the resources associated with the given buffer. */ -struct Free : public StmtNode { + /** Free the resources associated with the given buffer. */ + struct Free : public StmtNode + { std::string name; static Stmt make(std::string name); -}; + }; -/** A single-dimensional span. Includes all numbers between min and - * (min + extent - 1) */ -struct Range { + /** A single-dimensional span. Includes all numbers between min and + * (min + extent - 1) */ + struct Range + { Expr min, extent; Range() {} - Range(Expr min, Expr extent) : min(min), extent(extent) { - internal_assert(min.type() == extent.type()) << "Region min and extent must have same type\n"; + Range(Expr min, Expr extent) : min(min), extent(extent) + { + internal_assert(min.type() == extent.type()) << "Region min and extent must have same type\n"; } -}; + }; -/** A multi-dimensional box. The outer product of the elements */ -typedef std::vector Region; + /** A multi-dimensional box. The outer product of the elements */ + typedef std::vector Region; -/** Allocate a multi-dimensional buffer of the given type and - * size. Create some scratch memory that will back the function 'name' - * over the range specified in 'bounds'. The bounds are a vector of - * (min, extent) pairs for each dimension. */ -struct Realize : public StmtNode { + /** Allocate a multi-dimensional buffer of the given type and + * size. Create some scratch memory that will back the function 'name' + * over the range specified in 'bounds'. The bounds are a vector of + * (min, extent) pairs for each dimension. */ + struct Realize : public StmtNode + { std::string name; std::vector types; Region bounds; Expr condition; Stmt body; - EXPORT static Stmt make(const std::string &name, const std::vector &types, const Region &bounds, Expr condition, Stmt body); -}; + EXPORT static Stmt + make(const std::string &name, const std::vector &types, const Region &bounds, Expr condition, Stmt body); + }; -/** A sequence of statements to be executed in-order. 'rest' may be - * NULL. Used rest.defined() to find out. */ -struct Block : public StmtNode { + /** A sequence of statements to be executed in-order. 'rest' may be + * NULL. Used rest.defined() to find out. */ + struct Block : public StmtNode + { Stmt first, rest; EXPORT static Stmt make(Stmt first, Stmt rest); -}; + }; -/** An if-then-else block. 'else' may be NULL. */ -struct IfThenElse : public StmtNode { + /** An if-then-else block. 'else' may be NULL. */ + struct IfThenElse : public StmtNode + { Expr condition; Stmt then_case, else_case; EXPORT static Stmt make(Expr condition, Stmt then_case, Stmt else_case = Stmt()); -}; + }; -/** Evaluate and discard an expression, presumably because it has some side-effect. */ -struct Evaluate : public StmtNode { + /** Evaluate and discard an expression, presumably because it has some side-effect. */ + struct Evaluate : public StmtNode + { Expr value; EXPORT static Stmt make(Expr v); -}; + }; -} -} +}// namespace Internal +}// namespace Halide // Now that we've defined an Expr and ForType, we can include the definition of a function #ifndef HALIDE_FUNCTION_H #define HALIDE_FUNCTION_H @@ -2026,13 +2059,14 @@ struct Evaluate : public StmtNode { namespace Halide { namespace Internal { -/** A reference to a site in a Halide statement at the top of the - * body of a particular for loop. Evaluating a region of a halide - * function is done by generating a loop nest that spans its - * dimensions. We schedule the inputs to that function by - * recursively injecting realizations for them at particular sites - * in this loop nest. A LoopLevel identifies such a site. */ -struct LoopLevel { + /** A reference to a site in a Halide statement at the top of the + * body of a particular for loop. Evaluating a region of a halide + * function is done by generating a loop nest that spans its + * dimensions. We schedule the inputs to that function by + * recursively injecting realizations for them at particular sites + * in this loop nest. A LoopLevel identifies such a site. */ + struct LoopLevel + { std::string func, var; /** Identify the loop nest corresponding to some dimension of some function */ @@ -2044,44 +2078,37 @@ struct LoopLevel { LoopLevel() {} /** Test if a loop level corresponds to inlining the function */ - bool is_inline() const {return var.empty();} + bool is_inline() const { return var.empty(); } /** root is a special LoopLevel value which represents the * location outside of all for loops */ - static LoopLevel root() { - return LoopLevel("", "__root"); - } + static LoopLevel root() { return LoopLevel("", "__root"); } /** Test if a loop level is 'root', which describes the site * outside of all for loops */ - bool is_root() const {return var == "__root";} + bool is_root() const { return var == "__root"; } /** Compare this loop level against the variable name of a for * loop, to see if this loop level refers to the site * immediately inside this loop. */ - bool match(const std::string &loop) const { - return starts_with(loop, func + ".") && ends_with(loop, "." + var); - } + bool match(const std::string &loop) const { return starts_with(loop, func + ".") && ends_with(loop, "." + var); } - bool match(const LoopLevel &other) const { - return (func == other.func && - (var == other.var || - ends_with(var, "." + other.var) || - ends_with(other.var, "." + var))); + bool match(const LoopLevel &other) const + { + return ( + func == other.func && (var == other.var || ends_with(var, "." + other.var) || ends_with(other.var, "." + var))); } /** Check if two loop levels are exactly the same. */ - bool operator==(const LoopLevel &other) const { - return func == other.func && var == other.var; - } + bool operator==(const LoopLevel &other) const { return func == other.func && var == other.var; } + }; -}; - -struct Split { + struct Split + { std::string old_var, outer, inner; Expr factor; - bool exact; // Is it required that the factor divides the extent of the old var. True for splits of RVars. + bool exact;// Is it required that the factor divides the extent of the old var. True for splits of RVars. - enum SplitType {SplitVar = 0, RenameVar, FuseVars}; + enum SplitType { SplitVar = 0, RenameVar, FuseVars }; // If split_type is Rename, then this is just a renaming of the // old_var to the outer and not a split. The inner var should @@ -2093,38 +2120,42 @@ struct Split { // split, it joins the outer and inner into the old_var. SplitType split_type; - bool is_rename() const {return split_type == RenameVar;} - bool is_split() const {return split_type == SplitVar;} - bool is_fuse() const {return split_type == FuseVars;} -}; + bool is_rename() const { return split_type == RenameVar; } + bool is_split() const { return split_type == SplitVar; } + bool is_fuse() const { return split_type == FuseVars; } + }; -struct Dim { + struct Dim + { std::string var; For::ForType for_type; bool pure; -}; + }; -struct Bound { + struct Bound + { std::string var; Expr min, extent; -}; + }; -struct ScheduleContents; + struct ScheduleContents; -struct Specialization { + struct Specialization + { Expr condition; IntrusivePtr schedule; -}; + }; -class ReductionDomain; + class ReductionDomain; -/** A schedule for a single stage of a Halide pipeline. Right now this - * interface is basically a struct, offering mutable access to its - * innards. In the future it may become more encapsulated. */ -class Schedule { + /** A schedule for a single stage of a Halide pipeline. Right now this + * interface is basically a struct, offering mutable access to its + * innards. In the future it may become more encapsulated. */ + class Schedule + { IntrusivePtr contents; -public: + public: Schedule(IntrusivePtr c) : contents(c) {} Schedule(const Schedule &other) : contents(other.contents) {} EXPORT Schedule(); @@ -2192,8 +2223,8 @@ class Schedule { // @{ const std::vector &specializations() const; const Specialization &add_specialization(Expr condition); - //std::vector &specializations(); - // @} + // std::vector &specializations(); + // @} /** At what sites should we inject the allocation and the * computation of this function? The store_level must be outside @@ -2212,11 +2243,10 @@ class Schedule { bool allow_race_conditions() const; bool &allow_race_conditions(); // @} + }; -}; - -} -} +}// namespace Internal +}// namespace Halide #endif #ifndef HALIDE_REDUCTION_H @@ -2232,22 +2262,26 @@ class Schedule { namespace Halide { namespace Internal { -/** A single named dimension of a reduction domain */ -struct ReductionVariable { + /** A single named dimension of a reduction domain */ + struct ReductionVariable + { std::string var; Expr min, extent; -}; + }; -struct ReductionDomainContents { + struct ReductionDomainContents + { mutable RefCount ref_count; std::vector domain; -}; + }; -/** A reference-counted handle on a reduction domain, which is just a - * vector of ReductionVariable. */ -class ReductionDomain { + /** A reference-counted handle on a reduction domain, which is just a + * vector of ReductionVariable. */ + class ReductionDomain + { IntrusivePtr contents; -public: + + public: /** Construct a new NULL reduction domain */ ReductionDomain() : contents(NULL) {} @@ -2255,82 +2289,80 @@ class ReductionDomain { * all values of the given ReductionVariable in scanline order, * with the start of the vector being innermost, and the end of * the vector being outermost. */ - ReductionDomain(const std::vector &domain) : - contents(new ReductionDomainContents) { - contents.ptr->domain = domain; + ReductionDomain(const std::vector &domain) : contents(new ReductionDomainContents) + { + contents.ptr->domain = domain; } /** Is this handle non-NULL */ - bool defined() const { - return contents.defined(); - } + bool defined() const { return contents.defined(); } /** Tests for equality of reference. Only one reduction domain is * allowed per reduction function, and this is used to verify * that */ - bool same_as(const ReductionDomain &other) const { - return contents.same_as(other.contents); - } + bool same_as(const ReductionDomain &other) const { return contents.same_as(other.contents); } /** Immutable access to the reduction variables. */ - const std::vector &domain() const { - return contents.ptr->domain; - } -}; + const std::vector &domain() const { return contents.ptr->domain; } + }; -} -} +}// namespace Internal +}// namespace Halide #endif +#include #include #include -#include namespace Halide { namespace Internal { -struct FunctionContents; + struct FunctionContents; } /** An argument to an extern-defined Func. May be a Function, Buffer, * ImageParam or Expr. */ -struct ExternFuncArgument { - enum ArgType {UndefinedArg = 0, FuncArg, BufferArg, ExprArg, ImageParamArg}; - ArgType arg_type; - Internal::IntrusivePtr func; - Buffer buffer; - Expr expr; - Internal::Parameter image_param; +struct ExternFuncArgument +{ + enum ArgType { UndefinedArg = 0, FuncArg, BufferArg, ExprArg, ImageParamArg }; + ArgType arg_type; + Internal::IntrusivePtr func; + Buffer buffer; + Expr expr; + Internal::Parameter image_param; - ExternFuncArgument(Internal::IntrusivePtr f): arg_type(FuncArg), func(f) {} + ExternFuncArgument(Internal::IntrusivePtr f) : arg_type(FuncArg), func(f) {} - ExternFuncArgument(Buffer b): arg_type(BufferArg), buffer(b) {} + ExternFuncArgument(Buffer b) : arg_type(BufferArg), buffer(b) {} - ExternFuncArgument(Expr e): arg_type(ExprArg), expr(e) {} + ExternFuncArgument(Expr e) : arg_type(ExprArg), expr(e) {} - ExternFuncArgument(Internal::Parameter p) : arg_type(ImageParamArg), image_param(p) { - // Scalar params come in via the Expr constructor. - internal_assert(p.is_buffer()); - } - ExternFuncArgument() : arg_type(UndefinedArg) {} + ExternFuncArgument(Internal::Parameter p) : arg_type(ImageParamArg), image_param(p) + { + // Scalar params come in via the Expr constructor. + internal_assert(p.is_buffer()); + } + ExternFuncArgument() : arg_type(UndefinedArg) {} - bool is_func() const {return arg_type == FuncArg;} - bool is_expr() const {return arg_type == ExprArg;} - bool is_buffer() const {return arg_type == BufferArg;} - bool is_image_param() const {return arg_type == ImageParamArg;} - bool defined() const {return arg_type != UndefinedArg;} + bool is_func() const { return arg_type == FuncArg; } + bool is_expr() const { return arg_type == ExprArg; } + bool is_buffer() const { return arg_type == BufferArg; } + bool is_image_param() const { return arg_type == ImageParamArg; } + bool defined() const { return arg_type != UndefinedArg; } }; namespace Internal { -struct UpdateDefinition { + struct UpdateDefinition + { std::vector values, args; Schedule schedule; ReductionDomain domain; -}; + }; -struct FunctionContents { + struct FunctionContents + { mutable RefCount ref_count; std::string name; std::vector args; @@ -2352,15 +2384,17 @@ struct FunctionContents { bool frozen; FunctionContents() : trace_loads(false), trace_stores(false), trace_realizations(false), frozen(false) {} -}; - -/** A reference-counted handle to Halide's internal representation of - * a function. Similar to a front-end Func object, but with no - * syntactic sugar to help with definitions. */ -class Function { -private: + }; + + /** A reference-counted handle to Halide's internal representation of + * a function. Similar to a front-end Func object, but with no + * syntactic sugar to help with definitions. */ + class Function + { + private: IntrusivePtr contents; -public: + + public: /** Construct a new function with no definitions and no name. This * constructor only exists so that you can make vectors of * functions, etc. @@ -2387,170 +2421,111 @@ class Function { void define_update(const std::vector &args, std::vector values); /** Construct a new function with the given name */ - Function(const std::string &n) : contents(new FunctionContents) { - for (size_t i = 0; i < n.size(); i++) { - user_assert(n[i] != '.') - << "Func name \"" << n << "\" is invalid. " - << "Func names may not contain the character '.', " - << "as it is used internally by Halide as a separator\n"; - } - contents.ptr->name = n; + Function(const std::string &n) : contents(new FunctionContents) + { + for (size_t i = 0; i < n.size(); i++) { + user_assert(n[i] != '.') << "Func name \"" << n << "\" is invalid. " + << "Func names may not contain the character '.', " + << "as it is used internally by Halide as a separator\n"; + } + contents.ptr->name = n; } /** Get the name of the function */ - const std::string &name() const { - return contents.ptr->name; - } + const std::string &name() const { return contents.ptr->name; } /** Get the pure arguments */ - const std::vector &args() const { - return contents.ptr->args; - } + const std::vector &args() const { return contents.ptr->args; } /** Get the dimensionality */ - int dimensions() const { - return (int)args().size(); - } + int dimensions() const { return (int)args().size(); } /** Get the number of outputs */ - int outputs() const { - return (int)output_types().size(); - } + int outputs() const { return (int)output_types().size(); } /** Get the types of the outputs */ - const std::vector &output_types() const { - return contents.ptr->output_types; - } + const std::vector &output_types() const { return contents.ptr->output_types; } /** Get the right-hand-side of the pure definition */ - const std::vector &values() const { - return contents.ptr->values; - } + const std::vector &values() const { return contents.ptr->values; } /** Does this function have a pure definition */ - bool has_pure_definition() const { - return !contents.ptr->values.empty(); - } + bool has_pure_definition() const { return !contents.ptr->values.empty(); } /** Does this function *only* have a pure definition */ - bool is_pure() const { - return (has_pure_definition() && - !has_update_definition() && - !has_extern_definition()); - } + bool is_pure() const { return (has_pure_definition() && !has_update_definition() && !has_extern_definition()); } /** Get a handle to the schedule for the purpose of modifying * it */ - Schedule &schedule() { - return contents.ptr->schedule; - } + Schedule &schedule() { return contents.ptr->schedule; } /** Get a const handle to the schedule for inspecting it */ - const Schedule &schedule() const { - return contents.ptr->schedule; - } + const Schedule &schedule() const { return contents.ptr->schedule; } /** Get a handle on the output buffer used for setting constraints * on it. */ - const std::vector &output_buffers() const { - return contents.ptr->output_buffers; - } + const std::vector &output_buffers() const { return contents.ptr->output_buffers; } /** Get a mutable handle to the schedule for the update * stage */ - Schedule &update_schedule(int idx = 0) { - return contents.ptr->updates[idx].schedule; - } + Schedule &update_schedule(int idx = 0) { return contents.ptr->updates[idx].schedule; } /** Get a const reference to this function's update definitions. */ - const std::vector &updates() const { - return contents.ptr->updates; - } + const std::vector &updates() const { return contents.ptr->updates; } /** Does this function have an update definition */ - bool has_update_definition() const { - return !contents.ptr->updates.empty(); - } + bool has_update_definition() const { return !contents.ptr->updates.empty(); } /** Check if the function has an extern definition */ - bool has_extern_definition() const { - return !contents.ptr->extern_function_name.empty(); - } + bool has_extern_definition() const { return !contents.ptr->extern_function_name.empty(); } /** Add an external definition of this Func */ void define_extern(const std::string &function_name, - const std::vector &args, - const std::vector &types, - int dimensionality); + const std::vector &args, + const std::vector &types, + int dimensionality); /** Retrive the arguments of the extern definition */ - const std::vector &extern_arguments() const { - return contents.ptr->extern_arguments; - } + const std::vector &extern_arguments() const { return contents.ptr->extern_arguments; } /** Get the name of the extern function called for an extern * definition. */ - const std::string &extern_function_name() const { - return contents.ptr->extern_function_name; - } + const std::string &extern_function_name() const { return contents.ptr->extern_function_name; } /** Equality of identity */ - bool same_as(const Function &other) const { - return contents.same_as(other.contents); - } + bool same_as(const Function &other) const { return contents.same_as(other.contents); } /** Get a const handle to the debug filename */ - const std::string &debug_file() const { - return contents.ptr->debug_file; - } + const std::string &debug_file() const { return contents.ptr->debug_file; } /** Get a handle to the debug filename */ - std::string &debug_file() { - return contents.ptr->debug_file; - } + std::string &debug_file() { return contents.ptr->debug_file; } /** Use an an extern argument to another function. */ - operator ExternFuncArgument() const { - return ExternFuncArgument(contents); - } + operator ExternFuncArgument() const { return ExternFuncArgument(contents); } /** Tracing calls and accessors, passed down from the Func * equivalents. */ // @{ - void trace_loads() { - contents.ptr->trace_loads = true; - } - void trace_stores() { - contents.ptr->trace_stores = true; - } - void trace_realizations() { - contents.ptr->trace_realizations = true; - } - bool is_tracing_loads() { - return contents.ptr->trace_loads; - } - bool is_tracing_stores() { - return contents.ptr->trace_stores; - } - bool is_tracing_realizations() { - return contents.ptr->trace_realizations; - } + void trace_loads() { contents.ptr->trace_loads = true; } + void trace_stores() { contents.ptr->trace_stores = true; } + void trace_realizations() { contents.ptr->trace_realizations = true; } + bool is_tracing_loads() { return contents.ptr->trace_loads; } + bool is_tracing_stores() { return contents.ptr->trace_stores; } + bool is_tracing_realizations() { return contents.ptr->trace_realizations; } // @} /** Mark function as frozen, which means it cannot accept new * definitions. */ - void freeze() { - contents.ptr->frozen = true; - } + void freeze() { contents.ptr->frozen = true; } /** Check if a function has been frozen. If so, it is an error to * add new definitions. */ - bool frozen() const { - return contents.ptr->frozen; - } -}; + bool frozen() const { return contents.ptr->frozen; } + }; -}} +}// namespace Internal +}// namespace Halide #endif @@ -2559,56 +2534,26 @@ class Function { namespace Halide { namespace Internal { -/** A function call. This can represent a call to some extern - * function (like sin), but it's also our multi-dimensional - * version of a Load, so it can be a load from an input image, or - * a call to another halide function. The latter two types of call - * nodes don't survive all the way down to code generation - the - * lowering process converts them to Load nodes. */ -struct Call : public ExprNode { + /** A function call. This can represent a call to some extern + * function (like sin), but it's also our multi-dimensional + * version of a Load, so it can be a load from an input image, or + * a call to another halide function. The latter two types of call + * nodes don't survive all the way down to code generation - the + * lowering process converts them to Load nodes. */ + struct Call : public ExprNode + { std::string name; std::vector args; - typedef enum {Image, Extern, Halide, Intrinsic} CallType; + typedef enum { Image, Extern, Halide, Intrinsic } CallType; CallType call_type; // Halide uses calls internally to represent certain operations // (instead of IR nodes). These are matched by name. - EXPORT static const std::string debug_to_file, - shuffle_vector, - interleave_vectors, - reinterpret, - bitwise_and, - bitwise_not, - bitwise_xor, - bitwise_or, - shift_left, - shift_right, - abs, - rewrite_buffer, - profiling_timer, - random, - lerp, - create_buffer_t, - extract_buffer_min, - extract_buffer_max, - set_host_dirty, - set_dev_dirty, - popcount, - count_leading_zeros, - count_trailing_zeros, - undef, - null_handle, - address_of, - return_second, - if_then_else, - trace, - trace_expr, - glsl_texture_load, - glsl_texture_store, - make_struct, - stringify, - memoize_expr, - copy_memory; + EXPORT static const std::string debug_to_file, shuffle_vector, interleave_vectors, reinterpret, bitwise_and, + bitwise_not, bitwise_xor, bitwise_or, shift_left, shift_right, abs, rewrite_buffer, profiling_timer, random, lerp, + create_buffer_t, extract_buffer_min, extract_buffer_max, set_host_dirty, set_dev_dirty, popcount, + count_leading_zeros, count_trailing_zeros, undef, null_handle, address_of, return_second, if_then_else, trace, + trace_expr, glsl_texture_load, glsl_texture_store, make_struct, stringify, memoize_expr, copy_memory; // If it's a call to another halide function, this call node // holds onto a pointer to that function. @@ -2626,36 +2571,42 @@ struct Call : public ExprNode { // pointer to that Parameter param; - EXPORT static Expr make(Type type, std::string name, const std::vector &args, CallType call_type, - Function func = Function(), int value_index = 0, - Buffer image = Buffer(), Parameter param = Parameter()); + EXPORT static Expr make(Type type, + std::string name, + const std::vector &args, + CallType call_type, + Function func = Function(), + int value_index = 0, + Buffer image = Buffer(), + Parameter param = Parameter()); /** Convenience constructor for calls to other halide functions */ - static Expr make(Function func, const std::vector &args, int idx = 0) { - internal_assert(idx >= 0 && - idx < func.outputs()) - << "Value index out of range in call to halide function\n"; - internal_assert(func.has_pure_definition() || func.has_extern_definition()) - << "Call to undefined halide function\n"; - return make(func.output_types()[idx], func.name(), args, Halide, func, idx, Buffer(), Parameter()); + static Expr make(Function func, const std::vector &args, int idx = 0) + { + internal_assert(idx >= 0 && idx < func.outputs()) << "Value index out of range in call to halide function\n"; + internal_assert(func.has_pure_definition() || func.has_extern_definition()) + << "Call to undefined halide function\n"; + return make(func.output_types()[idx], func.name(), args, Halide, func, idx, Buffer(), Parameter()); } /** Convenience constructor for loads from concrete images */ - static Expr make(Buffer image, const std::vector &args) { - return make(image.type(), image.name(), args, Image, Function(), 0, image, Parameter()); + static Expr make(Buffer image, const std::vector &args) + { + return make(image.type(), image.name(), args, Image, Function(), 0, image, Parameter()); } /** Convenience constructor for loads from images parameters */ - static Expr make(Parameter param, const std::vector &args) { - return make(param.type(), param.name(), args, Image, Function(), 0, Buffer(), param); + static Expr make(Parameter param, const std::vector &args) + { + return make(param.type(), param.name(), args, Image, Function(), 0, Buffer(), param); } + }; -}; - -/** A named variable. Might be a loop variable, function argument, - * parameter, reduction variable, or something defined by a Let or - * LetStmt node. */ -struct Variable : public ExprNode { + /** A named variable. Might be a loop variable, function argument, + * parameter, reduction variable, or something defined by a Let or + * LetStmt node. */ + struct Variable : public ExprNode + { std::string name; /** References to scalar parameters, or to the dimensions of buffer @@ -2668,27 +2619,29 @@ struct Variable : public ExprNode { /** Reduction variables hang onto their domains */ ReductionDomain reduction_domain; - static Expr make(Type type, std::string name) { - return make(type, name, Buffer(), Parameter(), ReductionDomain()); - } + static Expr make(Type type, std::string name) { return make(type, name, Buffer(), Parameter(), ReductionDomain()); } - static Expr make(Type type, std::string name, Parameter param) { - return make(type, name, Buffer(), param, ReductionDomain()); + static Expr make(Type type, std::string name, Parameter param) + { + return make(type, name, Buffer(), param, ReductionDomain()); } - static Expr make(Type type, std::string name, Buffer image) { - return make(type, name, image, Parameter(), ReductionDomain()); + static Expr make(Type type, std::string name, Buffer image) + { + return make(type, name, image, Parameter(), ReductionDomain()); } - static Expr make(Type type, std::string name, ReductionDomain reduction_domain) { - return make(type, name, Buffer(), Parameter(), reduction_domain); + static Expr make(Type type, std::string name, ReductionDomain reduction_domain) + { + return make(type, name, Buffer(), Parameter(), reduction_domain); } - EXPORT static Expr make(Type type, std::string name, Buffer image, Parameter param, ReductionDomain reduction_domain); -}; + EXPORT static Expr + make(Type type, std::string name, Buffer image, Parameter param, ReductionDomain reduction_domain); + }; -} -} +}// namespace Internal +}// namespace Halide #endif #ifndef HALIDE_IR_OPERATOR_H @@ -2704,197 +2657,198 @@ struct Variable : public ExprNode { namespace Halide { namespace Internal { -/** Is the expression either an IntImm, a FloatImm, or a Cast of the - * same, or a Ramp or Broadcast of the same. Doesn't do any constant - * folding. */ -EXPORT bool is_const(Expr e); - -/** Is the expression an IntImm, FloatImm of a particular value, or a - * Cast, or Broadcast of the same. */ -EXPORT bool is_const(Expr e, int v); - -/** If an expression is an IntImm, return a pointer to its - * value. Otherwise returns NULL. */ -EXPORT const int *as_const_int(Expr e); - -/** If an expression is a FloatImm, return a pointer to its - * value. Otherwise returns NULL. */ -EXPORT const float *as_const_float(Expr e); - -/** Is the expression a constant integer power of two. Also returns - * log base two of the expression if it is. */ -EXPORT bool is_const_power_of_two(Expr e, int *bits); - -/** Is the expression a const (as defined by is_const), and also - * strictly greater than zero (in all lanes, if a vector expression) */ -EXPORT bool is_positive_const(Expr e); - -/** Is the expression a const (as defined by is_const), and also - * strictly less than zero (in all lanes, if a vector expression) */ -EXPORT bool is_negative_const(Expr e); - -/** Is the expression a const (as defined by is_const), and also - * strictly less than zero (in all lanes, if a vector expression) and - * is its negative value representable. (This excludes the most - * negative value of the Expr's type from inclusion. Intended to be - * used when the value will be negated as part of simplification.) - */ -EXPORT bool is_negative_negatable_const(Expr e); - -/** Is the expression a const (as defined by is_const), and also equal - * to zero (in all lanes, if a vector expression) */ -EXPORT bool is_zero(Expr e); - -/** Is the expression a const (as defined by is_const), and also equal - * to one (in all lanes, if a vector expression) */ -EXPORT bool is_one(Expr e); - -/** Is the expression a const (as defined by is_const), and also equal - * to two (in all lanes, if a vector expression) */ -EXPORT bool is_two(Expr e); - -/** Given an integer value, cast it into a designated integer type - * and return the bits as int. Unsigned types are returned as bits in the int - * and should be cast to unsigned int for comparison. - * int_cast_constant implements bit manipulations to wrap val into the - * value range of the Type t. - * For example, int_cast_constant(UInt(16), -1) returns 65535 - * int_cast_constant(Int(8), 128) returns -128 - */ -EXPORT int int_cast_constant(Type t, int val); - -/** Construct a const of the given type */ -EXPORT Expr make_const(Type t, int val); - -/** Construct a boolean constant from a C++ boolean value. - * May also be a vector if width is given. - * It is not possible to coerce a C++ boolean to Expr because - * if we provide such a path then char objects can ambiguously - * be converted to Halide Expr or to std::string. The problem - * is that C++ does not have a real bool type - it is in fact - * close enough to char that C++ does not know how to distinguish them. - * make_bool is the explicit coercion. */ -EXPORT Expr make_bool(bool val, int width = 1); - -/** Construct the representation of zero in the given type */ -EXPORT Expr make_zero(Type t); - -/** Construct the representation of one in the given type */ -EXPORT Expr make_one(Type t); - -/** Construct the representation of two in the given type */ -EXPORT Expr make_two(Type t); - -/** Construct the constant boolean true. May also be a vector of - * trues, if a width argument is given. */ -EXPORT Expr const_true(int width = 1); - -/** Construct the constant boolean false. May also be a vector of - * falses, if a width argument is given. */ -EXPORT Expr const_false(int width = 1); - -/** Coerce the two expressions to have the same type, using C-style - * casting rules. For the purposes of casting, a boolean type is - * UInt(1). We use the following procedure: - * - * If the types already match, do nothing. - * - * Then, if one type is a vector and the other is a scalar, the scalar - * is broadcast to match the vector width, and we continue. - * - * Then, if one type is floating-point and the other is not, the - * non-float is cast to the floating-point type, and we're done. - * - * Then, if neither is a float but one of the two is a constant, the - * constant is cast to match the non-const type and we're done. For - * example, e has type UInt(8), then (e*32) also has type UInt(8), - * despite the overflow that may occur. Note that this also means that - * (e*(-1)) is positive, and is equivalent to (e*255) - i.e. the (-1) - * is cast to a UInt(8) before the multiplication. - * - * Then, if both types are unsigned ints, the one with fewer bits is - * cast to match the one with more bits and we're done. - * - * Then, if both types are signed ints, the one with fewer bits is - * cast to match the one with more bits and we're done. - * - * Finally, if one type is an unsigned int and the other type is a signed - * int, both are cast to a signed int with the greater of the two - * bit-widths. For example, matching an Int(8) with a UInt(16) results - * in an Int(16). - * - */ -EXPORT void match_types(Expr &a, Expr &b); - -/** Halide's vectorizable transcendentals. */ -// @{ -EXPORT Expr halide_log(Expr a); -EXPORT Expr halide_exp(Expr a); -EXPORT Expr halide_erf(Expr a); -// @} - -/** Raise an expression to an integer power by repeatedly multiplying - * it by itself. */ -EXPORT Expr raise_to_integer_power(Expr a, int b); - - -} + /** Is the expression either an IntImm, a FloatImm, or a Cast of the + * same, or a Ramp or Broadcast of the same. Doesn't do any constant + * folding. */ + EXPORT bool is_const(Expr e); + + /** Is the expression an IntImm, FloatImm of a particular value, or a + * Cast, or Broadcast of the same. */ + EXPORT bool is_const(Expr e, int v); + + /** If an expression is an IntImm, return a pointer to its + * value. Otherwise returns NULL. */ + EXPORT const int *as_const_int(Expr e); + + /** If an expression is a FloatImm, return a pointer to its + * value. Otherwise returns NULL. */ + EXPORT const float *as_const_float(Expr e); + + /** Is the expression a constant integer power of two. Also returns + * log base two of the expression if it is. */ + EXPORT bool is_const_power_of_two(Expr e, int *bits); + + /** Is the expression a const (as defined by is_const), and also + * strictly greater than zero (in all lanes, if a vector expression) */ + EXPORT bool is_positive_const(Expr e); + + /** Is the expression a const (as defined by is_const), and also + * strictly less than zero (in all lanes, if a vector expression) */ + EXPORT bool is_negative_const(Expr e); + + /** Is the expression a const (as defined by is_const), and also + * strictly less than zero (in all lanes, if a vector expression) and + * is its negative value representable. (This excludes the most + * negative value of the Expr's type from inclusion. Intended to be + * used when the value will be negated as part of simplification.) + */ + EXPORT bool is_negative_negatable_const(Expr e); + + /** Is the expression a const (as defined by is_const), and also equal + * to zero (in all lanes, if a vector expression) */ + EXPORT bool is_zero(Expr e); + + /** Is the expression a const (as defined by is_const), and also equal + * to one (in all lanes, if a vector expression) */ + EXPORT bool is_one(Expr e); + + /** Is the expression a const (as defined by is_const), and also equal + * to two (in all lanes, if a vector expression) */ + EXPORT bool is_two(Expr e); + + /** Given an integer value, cast it into a designated integer type + * and return the bits as int. Unsigned types are returned as bits in the int + * and should be cast to unsigned int for comparison. + * int_cast_constant implements bit manipulations to wrap val into the + * value range of the Type t. + * For example, int_cast_constant(UInt(16), -1) returns 65535 + * int_cast_constant(Int(8), 128) returns -128 + */ + EXPORT int int_cast_constant(Type t, int val); + + /** Construct a const of the given type */ + EXPORT Expr make_const(Type t, int val); + + /** Construct a boolean constant from a C++ boolean value. + * May also be a vector if width is given. + * It is not possible to coerce a C++ boolean to Expr because + * if we provide such a path then char objects can ambiguously + * be converted to Halide Expr or to std::string. The problem + * is that C++ does not have a real bool type - it is in fact + * close enough to char that C++ does not know how to distinguish them. + * make_bool is the explicit coercion. */ + EXPORT Expr make_bool(bool val, int width = 1); + + /** Construct the representation of zero in the given type */ + EXPORT Expr make_zero(Type t); + + /** Construct the representation of one in the given type */ + EXPORT Expr make_one(Type t); + + /** Construct the representation of two in the given type */ + EXPORT Expr make_two(Type t); + + /** Construct the constant boolean true. May also be a vector of + * trues, if a width argument is given. */ + EXPORT Expr const_true(int width = 1); + + /** Construct the constant boolean false. May also be a vector of + * falses, if a width argument is given. */ + EXPORT Expr const_false(int width = 1); + + /** Coerce the two expressions to have the same type, using C-style + * casting rules. For the purposes of casting, a boolean type is + * UInt(1). We use the following procedure: + * + * If the types already match, do nothing. + * + * Then, if one type is a vector and the other is a scalar, the scalar + * is broadcast to match the vector width, and we continue. + * + * Then, if one type is floating-point and the other is not, the + * non-float is cast to the floating-point type, and we're done. + * + * Then, if neither is a float but one of the two is a constant, the + * constant is cast to match the non-const type and we're done. For + * example, e has type UInt(8), then (e*32) also has type UInt(8), + * despite the overflow that may occur. Note that this also means that + * (e*(-1)) is positive, and is equivalent to (e*255) - i.e. the (-1) + * is cast to a UInt(8) before the multiplication. + * + * Then, if both types are unsigned ints, the one with fewer bits is + * cast to match the one with more bits and we're done. + * + * Then, if both types are signed ints, the one with fewer bits is + * cast to match the one with more bits and we're done. + * + * Finally, if one type is an unsigned int and the other type is a signed + * int, both are cast to a signed int with the greater of the two + * bit-widths. For example, matching an Int(8) with a UInt(16) results + * in an Int(16). + * + */ + EXPORT void match_types(Expr &a, Expr &b); + + /** Halide's vectorizable transcendentals. */ + // @{ + EXPORT Expr halide_log(Expr a); + EXPORT Expr halide_exp(Expr a); + EXPORT Expr halide_erf(Expr a); + // @} + + /** Raise an expression to an integer power by repeatedly multiplying + * it by itself. */ + EXPORT Expr raise_to_integer_power(Expr a, int b); + + +}// namespace Internal /** Cast an expression to the halide type corresponding to the C++ type T. */ -template -inline Expr cast(Expr a) { - return cast(type_of(), a); -} +template inline Expr cast(Expr a) { return cast(type_of(), a); } /** Cast an expression to a new type. */ -inline Expr cast(Type t, Expr a) { - user_assert(a.defined()) << "cast of undefined Expr\n"; - if (a.type() == t) return a; - - if (t.is_handle() && !a.type().is_handle()) { - user_error << "Can't cast \"" << a << "\" to a handle. " - << "The only legal cast from scalar types to a handle is: " - << "reinterpret(Handle(), cast(" << a << "));\n"; - } else if (a.type().is_handle() && !t.is_handle()) { - user_error << "Can't cast handle \"" << a << "\" to type " << t << ". " - << "The only legal cast from handles to scalar types is: " - << "reinterpret(UInt64(), " << a << ");\n"; - } +inline Expr cast(Type t, Expr a) +{ + user_assert(a.defined()) << "cast of undefined Expr\n"; + if (a.type() == t) return a; + + if (t.is_handle() && !a.type().is_handle()) { + user_error << "Can't cast \"" << a << "\" to a handle. " + << "The only legal cast from scalar types to a handle is: " + << "reinterpret(Handle(), cast(" << a << "));\n"; + } else if (a.type().is_handle() && !t.is_handle()) { + user_error << "Can't cast handle \"" << a << "\" to type " << t << ". " + << "The only legal cast from handles to scalar types is: " + << "reinterpret(UInt64(), " << a << ");\n"; + } - if (t.is_vector()) { - if (a.type().is_scalar()) { - return Internal::Broadcast::make(cast(t.element_of(), a), t.width); - } else if (const Internal::Broadcast *b = a.as()) { - internal_assert(b->width == t.width); - return Internal::Broadcast::make(cast(t.element_of(), b->value), t.width); - } + if (t.is_vector()) { + if (a.type().is_scalar()) { + return Internal::Broadcast::make(cast(t.element_of(), a), t.width); + } else if (const Internal::Broadcast *b = a.as()) { + internal_assert(b->width == t.width); + return Internal::Broadcast::make(cast(t.element_of(), b->value), t.width); } - return Internal::Cast::make(t, a); + } + return Internal::Cast::make(t, a); } /** Return the sum of two expressions, doing any necessary type * coercion using \ref Internal::match_types */ -inline Expr operator+(Expr a, Expr b) { - user_assert(a.defined() && b.defined()) << "operator+ of undefined Expr\n"; - Internal::match_types(a, b); - return Internal::Add::make(a, b); +inline Expr operator+(Expr a, Expr b) +{ + user_assert(a.defined() && b.defined()) << "operator+ of undefined Expr\n"; + Internal::match_types(a, b); + return Internal::Add::make(a, b); } /** Modify the first expression to be the sum of two expressions, * without changing its type. This casts the second argument to match * the type of the first. */ -inline Expr &operator+=(Expr &a, Expr b) { - user_assert(a.defined() && b.defined()) << "operator+= of undefined Expr\n"; - a = Internal::Add::make(a, cast(a.type(), b)); - return a; +inline Expr &operator+=(Expr &a, Expr b) +{ + user_assert(a.defined() && b.defined()) << "operator+= of undefined Expr\n"; + a = Internal::Add::make(a, cast(a.type(), b)); + return a; } /** Return the difference of two expressions, doing any necessary type * coercion using \ref Internal::match_types */ -inline Expr operator-(Expr a, Expr b) { - user_assert(a.defined() && b.defined()) << "operator- of undefined Expr\n"; - Internal::match_types(a, b); - return Internal::Sub::make(a, b); +inline Expr operator-(Expr a, Expr b) +{ + user_assert(a.defined() && b.defined()) << "operator- of undefined Expr\n"; + Internal::match_types(a, b); + return Internal::Sub::make(a, b); } /** Return the negative of the argument. Does no type casting, so more @@ -2902,205 +2856,208 @@ inline Expr operator-(Expr a, Expr b) { * yields zero of the same type. For unsigned integers the negative is * still an unsigned integer. E.g. in UInt(8), the negative of 56 is * 200, because 56 + 200 == 0 */ -inline Expr operator-(Expr a) { - user_assert(a.defined()) << "operator- of undefined Expr\n"; - return Internal::Sub::make(Internal::make_zero(a.type()), a); +inline Expr operator-(Expr a) +{ + user_assert(a.defined()) << "operator- of undefined Expr\n"; + return Internal::Sub::make(Internal::make_zero(a.type()), a); } /** Modify the first expression to be the difference of two expressions, * without changing its type. This casts the second argument to match * the type of the first. */ -inline Expr &operator-=(Expr &a, Expr b) { - user_assert(a.defined() && b.defined()) << "operator-= of undefined Expr\n"; - a = Internal::Sub::make(a, cast(a.type(), b)); - return a; +inline Expr &operator-=(Expr &a, Expr b) +{ + user_assert(a.defined() && b.defined()) << "operator-= of undefined Expr\n"; + a = Internal::Sub::make(a, cast(a.type(), b)); + return a; } /** Return the product of two expressions, doing any necessary type * coercion using \ref Internal::match_types */ -inline Expr operator*(Expr a, Expr b) { - user_assert(a.defined() && b.defined()) << "operator* of undefined Expr\n"; - Internal::match_types(a, b); - return Internal::Mul::make(a, b); +inline Expr operator*(Expr a, Expr b) +{ + user_assert(a.defined() && b.defined()) << "operator* of undefined Expr\n"; + Internal::match_types(a, b); + return Internal::Mul::make(a, b); } /** Modify the first expression to be the product of two expressions, * without changing its type. This casts the second argument to match * the type of the first. */ -inline Expr &operator*=(Expr &a, Expr b) { - user_assert(a.defined() && b.defined()) << "operator*= of undefined Expr\n"; - a = Internal::Mul::make(a, cast(a.type(), b)); - return a; +inline Expr &operator*=(Expr &a, Expr b) +{ + user_assert(a.defined() && b.defined()) << "operator*= of undefined Expr\n"; + a = Internal::Mul::make(a, cast(a.type(), b)); + return a; } /** Return the ratio of two expressions, doing any necessary type * coercion using \ref Internal::match_types */ -inline Expr operator/(Expr a, Expr b) { - user_assert(a.defined() && b.defined()) << "operator/ of undefined Expr\n"; - user_assert(!Internal::is_const(b, 0)) << "operator/ with constant 0 divisor\n"; - Internal::match_types(a, b); - return Internal::Div::make(a, b); +inline Expr operator/(Expr a, Expr b) +{ + user_assert(a.defined() && b.defined()) << "operator/ of undefined Expr\n"; + user_assert(!Internal::is_const(b, 0)) << "operator/ with constant 0 divisor\n"; + Internal::match_types(a, b); + return Internal::Div::make(a, b); } /** Modify the first expression to be the ratio of two expressions, * without changing its type. This casts the second argument to match * the type of the first. */ -inline Expr &operator/=(Expr &a, Expr b) { - user_assert(a.defined() && b.defined()) << "operator/= of undefined Expr\n"; - user_assert(!Internal::is_const(b, 0)) << "operator/= with constant 0 divisor\n"; - a = Internal::Div::make(a, cast(a.type(), b)); - return a; +inline Expr &operator/=(Expr &a, Expr b) +{ + user_assert(a.defined() && b.defined()) << "operator/= of undefined Expr\n"; + user_assert(!Internal::is_const(b, 0)) << "operator/= with constant 0 divisor\n"; + a = Internal::Div::make(a, cast(a.type(), b)); + return a; } /** Return the first argument reduced modulo the second, doing any * necessary type coercion using \ref Internal::match_types */ -inline Expr operator%(Expr a, Expr b) { - user_assert(a.defined() && b.defined()) << "operator% of undefined Expr\n"; - user_assert(!Internal::is_const(b, 0)) << "operator% with constant 0 modulus\n"; - Internal::match_types(a, b); - return Internal::Mod::make(a, b); +inline Expr operator%(Expr a, Expr b) +{ + user_assert(a.defined() && b.defined()) << "operator% of undefined Expr\n"; + user_assert(!Internal::is_const(b, 0)) << "operator% with constant 0 modulus\n"; + Internal::match_types(a, b); + return Internal::Mod::make(a, b); } /** Return a boolean expression that tests whether the first argument * is greater than the second, after doing any necessary type coercion * using \ref Internal::match_types */ -inline Expr operator>(Expr a, Expr b) { - user_assert(a.defined() && b.defined()) << "operator> of undefined Expr\n"; - Internal::match_types(a, b); - return Internal::GT::make(a, b); +inline Expr operator>(Expr a, Expr b) +{ + user_assert(a.defined() && b.defined()) << "operator> of undefined Expr\n"; + Internal::match_types(a, b); + return Internal::GT::make(a, b); } /** Return a boolean expression that tests whether the first argument * is less than the second, after doing any necessary type coercion * using \ref Internal::match_types */ -inline Expr operator<(Expr a, Expr b) { - user_assert(a.defined() && b.defined()) << "operator< of undefined Expr\n"; - Internal::match_types(a, b); - return Internal::LT::make(a, b); +inline Expr operator<(Expr a, Expr b) +{ + user_assert(a.defined() && b.defined()) << "operator< of undefined Expr\n"; + Internal::match_types(a, b); + return Internal::LT::make(a, b); } /** Return a boolean expression that tests whether the first argument * is less than or equal to the second, after doing any necessary type * coercion using \ref Internal::match_types */ -inline Expr operator<=(Expr a, Expr b) { - user_assert(a.defined() && b.defined()) << "operator<= of undefined Expr\n"; - Internal::match_types(a, b); - return Internal::LE::make(a, b); +inline Expr operator<=(Expr a, Expr b) +{ + user_assert(a.defined() && b.defined()) << "operator<= of undefined Expr\n"; + Internal::match_types(a, b); + return Internal::LE::make(a, b); } /** Return a boolean expression that tests whether the first argument * is greater than or equal to the second, after doing any necessary * type coercion using \ref Internal::match_types */ -inline Expr operator>=(Expr a, Expr b) { - user_assert(a.defined() && b.defined()) << "operator>= of undefined Expr\n"; - Internal::match_types(a, b); - return Internal::GE::make(a, b); +inline Expr operator>=(Expr a, Expr b) +{ + user_assert(a.defined() && b.defined()) << "operator>= of undefined Expr\n"; + Internal::match_types(a, b); + return Internal::GE::make(a, b); } /** Return a boolean expression that tests whether the first argument * is equal to the second, after doing any necessary type coercion * using \ref Internal::match_types */ -inline Expr operator==(Expr a, Expr b) { - user_assert(a.defined() && b.defined()) << "operator== of undefined Expr\n"; - Internal::match_types(a, b); - return Internal::EQ::make(a, b); +inline Expr operator==(Expr a, Expr b) +{ + user_assert(a.defined() && b.defined()) << "operator== of undefined Expr\n"; + Internal::match_types(a, b); + return Internal::EQ::make(a, b); } /** Return a boolean expression that tests whether the first argument * is not equal to the second, after doing any necessary type coercion * using \ref Internal::match_types */ -inline Expr operator!=(Expr a, Expr b) { - user_assert(a.defined() && b.defined()) << "operator!= of undefined Expr\n"; - Internal::match_types(a, b); - return Internal::NE::make(a, b); +inline Expr operator!=(Expr a, Expr b) +{ + user_assert(a.defined() && b.defined()) << "operator!= of undefined Expr\n"; + Internal::match_types(a, b); + return Internal::NE::make(a, b); } /** Returns the logical and of the two arguments */ -inline Expr operator&&(Expr a, Expr b) { - return Internal::And::make(a, b); -} +inline Expr operator&&(Expr a, Expr b) { return Internal::And::make(a, b); } /** Returns the logical or of the two arguments */ -inline Expr operator||(Expr a, Expr b) { - return Internal::Or::make(a, b); -} +inline Expr operator||(Expr a, Expr b) { return Internal::Or::make(a, b); } /** Returns the logical not the argument */ -inline Expr operator!(Expr a) { - return Internal::Not::make(a); -} +inline Expr operator!(Expr a) { return Internal::Not::make(a); } /** Returns an expression representing the greater of the two * arguments, after doing any necessary type coercion using * \ref Internal::match_types. Vectorizes cleanly on most platforms * (with the exception of integer types on x86 without SSE4). */ -inline Expr max(Expr a, Expr b) { - user_assert(a.defined() && b.defined()) - << "max of undefined Expr\n"; - Internal::match_types(a, b); - return Internal::Max::make(a, b); +inline Expr max(Expr a, Expr b) +{ + user_assert(a.defined() && b.defined()) << "max of undefined Expr\n"; + Internal::match_types(a, b); + return Internal::Max::make(a, b); } /** Returns an expression representing the lesser of the two * arguments, after doing any necessary type coercion using * \ref Internal::match_types. Vectorizes cleanly on most platforms * (with the exception of integer types on x86 without SSE4). */ -inline Expr min(Expr a, Expr b) { - user_assert(a.defined() && b.defined()) - << "min of undefined Expr\n"; - Internal::match_types(a, b); - return Internal::Min::make(a, b); +inline Expr min(Expr a, Expr b) +{ + user_assert(a.defined() && b.defined()) << "min of undefined Expr\n"; + Internal::match_types(a, b); + return Internal::Min::make(a, b); } /** Clamps an expression to lie within the given bounds. The bounds * are type-cast to match the expression. Vectorizes as well as min/max. */ -inline Expr clamp(Expr a, Expr min_val, Expr max_val) { - user_assert(a.defined() && min_val.defined() && max_val.defined()) - << "clamp of undefined Expr\n"; - min_val = cast(a.type(), min_val); - max_val = cast(a.type(), max_val); - return Internal::Max::make(Internal::Min::make(a, max_val), min_val); +inline Expr clamp(Expr a, Expr min_val, Expr max_val) +{ + user_assert(a.defined() && min_val.defined() && max_val.defined()) << "clamp of undefined Expr\n"; + min_val = cast(a.type(), min_val); + max_val = cast(a.type(), max_val); + return Internal::Max::make(Internal::Min::make(a, max_val), min_val); } /** Returns the absolute value of a signed integer or floating-point * expression. Vectorizes cleanly. Unlike in C, abs of a signed * integer returns an unsigned integer of the same bit width. This * means that abs of the most negative integer doesn't overflow. */ -inline Expr abs(Expr a) { - user_assert(a.defined()) - << "abs of undefined Expr\n"; - Type t = a.type(); - if (t.is_int()) { - t.code = Type::UInt; - } else if (t.is_uint()) { - user_warning << "Warning: abs of an unsigned type is a no-op\n"; - return a; - } - return Internal::Call::make(t, Internal::Call::abs, - vec(a), Internal::Call::Intrinsic); +inline Expr abs(Expr a) +{ + user_assert(a.defined()) << "abs of undefined Expr\n"; + Type t = a.type(); + if (t.is_int()) { + t.code = Type::UInt; + } else if (t.is_uint()) { + user_warning << "Warning: abs of an unsigned type is a no-op\n"; + return a; + } + return Internal::Call::make(t, Internal::Call::abs, vec(a), Internal::Call::Intrinsic); } /** Returns an expression similar to the ternary operator in C, except * that it always evaluates all arguments. If the first argument is * true, then return the second, else return the third. Typically * vectorizes cleanly, but benefits from SSE41 or newer on x86. */ -inline Expr select(Expr condition, Expr true_value, Expr false_value) { +inline Expr select(Expr condition, Expr true_value, Expr false_value) +{ - if (as_const_int(condition)) { - // Why are you doing this? We'll preserve the select node until constant folding for you. - condition = cast(Bool(), condition); - } + if (as_const_int(condition)) { + // Why are you doing this? We'll preserve the select node until constant folding for you. + condition = cast(Bool(), condition); + } - // Coerce int literals to the type of the other argument - if (as_const_int(true_value)) { - true_value = cast(false_value.type(), true_value); - } - if (as_const_int(false_value)) { - false_value = cast(true_value.type(), false_value); - } + // Coerce int literals to the type of the other argument + if (as_const_int(true_value)) { true_value = cast(false_value.type(), true_value); } + if (as_const_int(false_value)) { false_value = cast(true_value.type(), false_value); } - return Internal::Select::make(condition, true_value, false_value); + return Internal::Select::make(condition, true_value, false_value); } /** A multi-way variant of select similar to a switch statement in C, @@ -3108,312 +3065,315 @@ inline Expr select(Expr condition, Expr true_value, Expr false_value) { * to the first value for which the condition is true. Returns the * final value if all conditions are false. */ // @{ -inline Expr select(Expr c1, Expr v1, - Expr c2, Expr v2, - Expr default_val) { - return select(c1, v1, - select(c2, v2, default_val)); -} -inline Expr select(Expr c1, Expr v1, - Expr c2, Expr v2, - Expr c3, Expr v3, - Expr default_val) { - return select(c1, v1, - c2, v2, - select(c3, v3, default_val)); -} -inline Expr select(Expr c1, Expr v1, - Expr c2, Expr v2, - Expr c3, Expr v3, - Expr c4, Expr v4, - Expr default_val) { - return select(c1, v1, - c2, v2, - c3, v3, - select(c4, v4, default_val)); -} -inline Expr select(Expr c1, Expr v1, - Expr c2, Expr v2, - Expr c3, Expr v3, - Expr c4, Expr v4, - Expr c5, Expr v5, - Expr default_val) { - return select(c1, v1, - c2, v2, - c3, v3, - c4, v4, - select(c5, v5, default_val)); -} -inline Expr select(Expr c1, Expr v1, - Expr c2, Expr v2, - Expr c3, Expr v3, - Expr c4, Expr v4, - Expr c5, Expr v5, - Expr c6, Expr v6, - Expr default_val) { - return select(c1, v1, - c2, v2, - c3, v3, - c4, v4, - c5, v5, - select(c6, v6, default_val)); -} -inline Expr select(Expr c1, Expr v1, - Expr c2, Expr v2, - Expr c3, Expr v3, - Expr c4, Expr v4, - Expr c5, Expr v5, - Expr c6, Expr v6, - Expr c7, Expr v7, - Expr default_val) { - return select(c1, v1, - c2, v2, - c3, v3, - c4, v4, - c5, v5, - c6, v6, - select(c7, v7, default_val)); -} -inline Expr select(Expr c1, Expr v1, - Expr c2, Expr v2, - Expr c3, Expr v3, - Expr c4, Expr v4, - Expr c5, Expr v5, - Expr c6, Expr v6, - Expr c7, Expr v7, - Expr c8, Expr v8, - Expr default_val) { - return select(c1, v1, - c2, v2, - c3, v3, - c4, v4, - c5, v5, - c6, v6, - c7, v7, - select(c8, v8, default_val)); -} -inline Expr select(Expr c1, Expr v1, - Expr c2, Expr v2, - Expr c3, Expr v3, - Expr c4, Expr v4, - Expr c5, Expr v5, - Expr c6, Expr v6, - Expr c7, Expr v7, - Expr c8, Expr v8, - Expr c9, Expr v9, - Expr default_val) { - return select(c1, v1, - c2, v2, - c3, v3, - c4, v4, - c5, v5, - c6, v6, - c7, v7, - c8, v8, - select(c9, v9, default_val)); -} -inline Expr select(Expr c1, Expr v1, - Expr c2, Expr v2, - Expr c3, Expr v3, - Expr c4, Expr v4, - Expr c5, Expr v5, - Expr c6, Expr v6, - Expr c7, Expr v7, - Expr c8, Expr v8, - Expr c9, Expr v9, - Expr c10, Expr v10, - Expr default_val) { - return select(c1, v1, - c2, v2, - c3, v3, - c4, v4, - c5, v5, - c6, v6, - c7, v7, - c8, v8, - c9, v9, - select(c10, v10, default_val)); +inline Expr select(Expr c1, Expr v1, Expr c2, Expr v2, Expr default_val) +{ + return select(c1, v1, select(c2, v2, default_val)); +} +inline Expr select(Expr c1, Expr v1, Expr c2, Expr v2, Expr c3, Expr v3, Expr default_val) +{ + return select(c1, v1, c2, v2, select(c3, v3, default_val)); +} +inline Expr select(Expr c1, Expr v1, Expr c2, Expr v2, Expr c3, Expr v3, Expr c4, Expr v4, Expr default_val) +{ + return select(c1, v1, c2, v2, c3, v3, select(c4, v4, default_val)); +} +inline Expr + select(Expr c1, Expr v1, Expr c2, Expr v2, Expr c3, Expr v3, Expr c4, Expr v4, Expr c5, Expr v5, Expr default_val) +{ + return select(c1, v1, c2, v2, c3, v3, c4, v4, select(c5, v5, default_val)); +} +inline Expr select(Expr c1, + Expr v1, + Expr c2, + Expr v2, + Expr c3, + Expr v3, + Expr c4, + Expr v4, + Expr c5, + Expr v5, + Expr c6, + Expr v6, + Expr default_val) +{ + return select(c1, v1, c2, v2, c3, v3, c4, v4, c5, v5, select(c6, v6, default_val)); +} +inline Expr select(Expr c1, + Expr v1, + Expr c2, + Expr v2, + Expr c3, + Expr v3, + Expr c4, + Expr v4, + Expr c5, + Expr v5, + Expr c6, + Expr v6, + Expr c7, + Expr v7, + Expr default_val) +{ + return select(c1, v1, c2, v2, c3, v3, c4, v4, c5, v5, c6, v6, select(c7, v7, default_val)); +} +inline Expr select(Expr c1, + Expr v1, + Expr c2, + Expr v2, + Expr c3, + Expr v3, + Expr c4, + Expr v4, + Expr c5, + Expr v5, + Expr c6, + Expr v6, + Expr c7, + Expr v7, + Expr c8, + Expr v8, + Expr default_val) +{ + return select(c1, v1, c2, v2, c3, v3, c4, v4, c5, v5, c6, v6, c7, v7, select(c8, v8, default_val)); +} +inline Expr select(Expr c1, + Expr v1, + Expr c2, + Expr v2, + Expr c3, + Expr v3, + Expr c4, + Expr v4, + Expr c5, + Expr v5, + Expr c6, + Expr v6, + Expr c7, + Expr v7, + Expr c8, + Expr v8, + Expr c9, + Expr v9, + Expr default_val) +{ + return select(c1, v1, c2, v2, c3, v3, c4, v4, c5, v5, c6, v6, c7, v7, c8, v8, select(c9, v9, default_val)); +} +inline Expr select(Expr c1, + Expr v1, + Expr c2, + Expr v2, + Expr c3, + Expr v3, + Expr c4, + Expr v4, + Expr c5, + Expr v5, + Expr c6, + Expr v6, + Expr c7, + Expr v7, + Expr c8, + Expr v8, + Expr c9, + Expr v9, + Expr c10, + Expr v10, + Expr default_val) +{ + return select(c1, v1, c2, v2, c3, v3, c4, v4, c5, v5, c6, v6, c7, v7, c8, v8, c9, v9, select(c10, v10, default_val)); } // @} /** Return the sine of a floating-point expression. If the argument is * not floating-point, it is cast to Float(32). Does not vectorize * well. */ -inline Expr sin(Expr x) { - user_assert(x.defined()) << "sin of undefined Expr\n"; - if (x.type() == Float(64)) { - return Internal::Call::make(Float(64), "sin_f64", vec(x), Internal::Call::Extern); - } else { - return Internal::Call::make(Float(32), "sin_f32", vec(cast(x)), Internal::Call::Extern); - } +inline Expr sin(Expr x) +{ + user_assert(x.defined()) << "sin of undefined Expr\n"; + if (x.type() == Float(64)) { + return Internal::Call::make(Float(64), "sin_f64", vec(x), Internal::Call::Extern); + } else { + return Internal::Call::make(Float(32), "sin_f32", vec(cast(x)), Internal::Call::Extern); + } } /** Return the arcsine of a floating-point expression. If the argument * is not floating-point, it is cast to Float(32). Does not vectorize * well. */ -inline Expr asin(Expr x) { - user_assert(x.defined()) << "asin of undefined Expr\n"; - if (x.type() == Float(64)) { - return Internal::Call::make(Float(64), "asin_f64", vec(x), Internal::Call::Extern); - } else { - return Internal::Call::make(Float(32), "asin_f32", vec(cast(x)), Internal::Call::Extern); - } +inline Expr asin(Expr x) +{ + user_assert(x.defined()) << "asin of undefined Expr\n"; + if (x.type() == Float(64)) { + return Internal::Call::make(Float(64), "asin_f64", vec(x), Internal::Call::Extern); + } else { + return Internal::Call::make(Float(32), "asin_f32", vec(cast(x)), Internal::Call::Extern); + } } /** Return the cosine of a floating-point expression. If the argument * is not floating-point, it is cast to Float(32). Does not vectorize * well. */ -inline Expr cos(Expr x) { - user_assert(x.defined()) << "cos of undefined Expr\n"; - if (x.type() == Float(64)) { - return Internal::Call::make(Float(64), "cos_f64", vec(x), Internal::Call::Extern); - } else { - return Internal::Call::make(Float(32), "cos_f32", vec(cast(x)), Internal::Call::Extern); - } +inline Expr cos(Expr x) +{ + user_assert(x.defined()) << "cos of undefined Expr\n"; + if (x.type() == Float(64)) { + return Internal::Call::make(Float(64), "cos_f64", vec(x), Internal::Call::Extern); + } else { + return Internal::Call::make(Float(32), "cos_f32", vec(cast(x)), Internal::Call::Extern); + } } /** Return the arccosine of a floating-point expression. If the * argument is not floating-point, it is cast to Float(32). Does not * vectorize well. */ -inline Expr acos(Expr x) { - user_assert(x.defined()) << "acos of undefined Expr\n"; - if (x.type() == Float(64)) { - return Internal::Call::make(Float(64), "acos_f64", vec(x), Internal::Call::Extern); - } else { - return Internal::Call::make(Float(32), "acos_f32", vec(cast(x)), Internal::Call::Extern); - } +inline Expr acos(Expr x) +{ + user_assert(x.defined()) << "acos of undefined Expr\n"; + if (x.type() == Float(64)) { + return Internal::Call::make(Float(64), "acos_f64", vec(x), Internal::Call::Extern); + } else { + return Internal::Call::make(Float(32), "acos_f32", vec(cast(x)), Internal::Call::Extern); + } } /** Return the tangent of a floating-point expression. If the argument * is not floating-point, it is cast to Float(32). Does not vectorize * well. */ -inline Expr tan(Expr x) { - user_assert(x.defined()) << "tan of undefined Expr\n"; - if (x.type() == Float(64)) { - return Internal::Call::make(Float(64), "tan_f64", vec(x), Internal::Call::Extern); - } else { - return Internal::Call::make(Float(32), "tan_f32", vec(cast(x)), Internal::Call::Extern); - } +inline Expr tan(Expr x) +{ + user_assert(x.defined()) << "tan of undefined Expr\n"; + if (x.type() == Float(64)) { + return Internal::Call::make(Float(64), "tan_f64", vec(x), Internal::Call::Extern); + } else { + return Internal::Call::make(Float(32), "tan_f32", vec(cast(x)), Internal::Call::Extern); + } } /** Return the arctangent of a floating-point expression. If the * argument is not floating-point, it is cast to Float(32). Does not * vectorize well. */ -inline Expr atan(Expr x) { - user_assert(x.defined()) << "atan of undefined Expr\n"; - if (x.type() == Float(64)) { - return Internal::Call::make(Float(64), "atan_f64", vec(x), Internal::Call::Extern); - } else { - return Internal::Call::make(Float(32), "atan_f32", vec(cast(x)), Internal::Call::Extern); - } +inline Expr atan(Expr x) +{ + user_assert(x.defined()) << "atan of undefined Expr\n"; + if (x.type() == Float(64)) { + return Internal::Call::make(Float(64), "atan_f64", vec(x), Internal::Call::Extern); + } else { + return Internal::Call::make(Float(32), "atan_f32", vec(cast(x)), Internal::Call::Extern); + } } /** Return the angle of a floating-point gradient. If the argument is * not floating-point, it is cast to Float(32). Does not vectorize * well. */ -inline Expr atan2(Expr y, Expr x) { - user_assert(x.defined() && y.defined()) << "atan2 of undefined Expr\n"; - - if (y.type() == Float(64)) { - x = cast(x); - return Internal::Call::make(Float(64), "atan2_f64", vec(y, x), Internal::Call::Extern); - } else { - y = cast(y); - x = cast(x); - return Internal::Call::make(Float(32), "atan2_f32", vec(y, x), Internal::Call::Extern); - } +inline Expr atan2(Expr y, Expr x) +{ + user_assert(x.defined() && y.defined()) << "atan2 of undefined Expr\n"; + + if (y.type() == Float(64)) { + x = cast(x); + return Internal::Call::make(Float(64), "atan2_f64", vec(y, x), Internal::Call::Extern); + } else { + y = cast(y); + x = cast(x); + return Internal::Call::make(Float(32), "atan2_f32", vec(y, x), Internal::Call::Extern); + } } /** Return the hyperbolic sine of a floating-point expression. If the * argument is not floating-point, it is cast to Float(32). Does not * vectorize well. */ -inline Expr sinh(Expr x) { - user_assert(x.defined()) << "sinh of undefined Expr\n"; - if (x.type() == Float(64)) { - return Internal::Call::make(Float(64), "sinh_f64", vec(x), Internal::Call::Extern); - } else { - return Internal::Call::make(Float(32), "sinh_f32", vec(cast(x)), Internal::Call::Extern); - } +inline Expr sinh(Expr x) +{ + user_assert(x.defined()) << "sinh of undefined Expr\n"; + if (x.type() == Float(64)) { + return Internal::Call::make(Float(64), "sinh_f64", vec(x), Internal::Call::Extern); + } else { + return Internal::Call::make(Float(32), "sinh_f32", vec(cast(x)), Internal::Call::Extern); + } } /** Return the hyperbolic arcsinhe of a floating-point expression. If * the argument is not floating-point, it is cast to Float(32). Does * not vectorize well. */ -inline Expr asinh(Expr x) { - user_assert(x.defined()) << "asinh of undefined Expr\n"; - if (x.type() == Float(64)) { - return Internal::Call::make(Float(64), "asinh_f64", vec(x), Internal::Call::Extern); - } else { - return Internal::Call::make(Float(32), "asinh_f32", vec(cast(x)), Internal::Call::Extern); - } +inline Expr asinh(Expr x) +{ + user_assert(x.defined()) << "asinh of undefined Expr\n"; + if (x.type() == Float(64)) { + return Internal::Call::make(Float(64), "asinh_f64", vec(x), Internal::Call::Extern); + } else { + return Internal::Call::make(Float(32), "asinh_f32", vec(cast(x)), Internal::Call::Extern); + } } /** Return the hyperbolic cosine of a floating-point expression. If * the argument is not floating-point, it is cast to Float(32). Does * not vectorize well. */ -inline Expr cosh(Expr x) { - user_assert(x.defined()) << "cosh of undefined Expr\n"; - if (x.type() == Float(64)) { - return Internal::Call::make(Float(64), "cosh_f64", vec(x), Internal::Call::Extern); - } else { - return Internal::Call::make(Float(32), "cosh_f32", vec(cast(x)), Internal::Call::Extern); - } +inline Expr cosh(Expr x) +{ + user_assert(x.defined()) << "cosh of undefined Expr\n"; + if (x.type() == Float(64)) { + return Internal::Call::make(Float(64), "cosh_f64", vec(x), Internal::Call::Extern); + } else { + return Internal::Call::make(Float(32), "cosh_f32", vec(cast(x)), Internal::Call::Extern); + } } /** Return the hyperbolic arccosine of a floating-point expression. * If the argument is not floating-point, it is cast to * Float(32). Does not vectorize well. */ -inline Expr acosh(Expr x) { - user_assert(x.defined()) << "acosh of undefined Expr\n"; - if (x.type() == Float(64)) { - return Internal::Call::make(Float(64), "acosh_f64", vec(x), Internal::Call::Extern); - } else { - return Internal::Call::make(Float(32), "acosh_f32", vec(cast(x)), Internal::Call::Extern); - } +inline Expr acosh(Expr x) +{ + user_assert(x.defined()) << "acosh of undefined Expr\n"; + if (x.type() == Float(64)) { + return Internal::Call::make(Float(64), "acosh_f64", vec(x), Internal::Call::Extern); + } else { + return Internal::Call::make(Float(32), "acosh_f32", vec(cast(x)), Internal::Call::Extern); + } } /** Return the hyperbolic tangent of a floating-point expression. If * the argument is not floating-point, it is cast to Float(32). Does * not vectorize well. */ -inline Expr tanh(Expr x) { - user_assert(x.defined()) << "tanh of undefined Expr\n"; - if (x.type() == Float(64)) { - return Internal::Call::make(Float(64), "tanh_f64", vec(x), Internal::Call::Extern); - } else { - return Internal::Call::make(Float(32), "tanh_f32", vec(cast(x)), Internal::Call::Extern); - } +inline Expr tanh(Expr x) +{ + user_assert(x.defined()) << "tanh of undefined Expr\n"; + if (x.type() == Float(64)) { + return Internal::Call::make(Float(64), "tanh_f64", vec(x), Internal::Call::Extern); + } else { + return Internal::Call::make(Float(32), "tanh_f32", vec(cast(x)), Internal::Call::Extern); + } } /** Return the hyperbolic arctangent of a floating-point expression. * If the argument is not floating-point, it is cast to * Float(32). Does not vectorize well. */ -inline Expr atanh(Expr x) { - user_assert(x.defined()) << "atanh of undefined Expr\n"; - if (x.type() == Float(64)) { - return Internal::Call::make(Float(64), "atanh_f64", vec(x), Internal::Call::Extern); - } else { - return Internal::Call::make(Float(32), "atanh_f32", vec(cast(x)), Internal::Call::Extern); - } +inline Expr atanh(Expr x) +{ + user_assert(x.defined()) << "atanh of undefined Expr\n"; + if (x.type() == Float(64)) { + return Internal::Call::make(Float(64), "atanh_f64", vec(x), Internal::Call::Extern); + } else { + return Internal::Call::make(Float(32), "atanh_f32", vec(cast(x)), Internal::Call::Extern); + } } /** Return the square root of a floating-point expression. If the * argument is not floating-point, it is cast to Float(32). Typically * vectorizes cleanly. */ -inline Expr sqrt(Expr x) { - user_assert(x.defined()) << "sqrt of undefined Expr\n"; - if (x.type() == Float(64)) { - return Internal::Call::make(Float(64), "sqrt_f64", vec(x), Internal::Call::Extern); - } else { - return Internal::Call::make(Float(32), "sqrt_f32", vec(cast(x)), Internal::Call::Extern); - } +inline Expr sqrt(Expr x) +{ + user_assert(x.defined()) << "sqrt of undefined Expr\n"; + if (x.type() == Float(64)) { + return Internal::Call::make(Float(64), "sqrt_f64", vec(x), Internal::Call::Extern); + } else { + return Internal::Call::make(Float(32), "sqrt_f32", vec(cast(x)), Internal::Call::Extern); + } } /** Return the square root of the sum of the squares of two * floating-point expressions. If the argument is not floating-point, * it is cast to Float(32). Vectorizes cleanly. */ -inline Expr hypot(Expr x, Expr y) { - return sqrt(x*x + y*y); -} +inline Expr hypot(Expr x, Expr y) { return sqrt(x * x + y * y); } /** Return the exponential of a floating-point expression. If the * argument is not floating-point, it is cast to Float(32). For @@ -3422,13 +3382,14 @@ inline Expr hypot(Expr x, Expr y) { * vectorizable, does the right thing for extremely small or extremely * large inputs, and is accurate up to the last bit of the * mantissa. Vectorizes cleanly. */ -inline Expr exp(Expr x) { - user_assert(x.defined()) << "exp of undefined Expr\n"; - if (x.type() == Float(64)) { - return Internal::Call::make(Float(64), "exp_f64", vec(x), Internal::Call::Extern); - } else { - return Internal::Call::make(Float(32), "exp_f32", vec(cast(x)), Internal::Call::Extern); - } +inline Expr exp(Expr x) +{ + user_assert(x.defined()) << "exp of undefined Expr\n"; + if (x.type() == Float(64)) { + return Internal::Call::make(Float(64), "exp_f64", vec(x), Internal::Call::Extern); + } else { + return Internal::Call::make(Float(32), "exp_f32", vec(cast(x)), Internal::Call::Extern); + } } /** Return the logarithm of a floating-point expression. If the @@ -3438,13 +3399,14 @@ inline Expr exp(Expr x) { * vectorizable, does the right thing for inputs <= 0 (returns -inf or * nan), and is accurate up to the last bit of the * mantissa. Vectorizes cleanly. */ -inline Expr log(Expr x) { - user_assert(x.defined()) << "log of undefined Expr\n"; - if (x.type() == Float(64)) { - return Internal::Call::make(Float(64), "log_f64", vec(x), Internal::Call::Extern); - } else { - return Internal::Call::make(Float(32), "log_f32", vec(cast(x)), Internal::Call::Extern); - } +inline Expr log(Expr x) +{ + user_assert(x.defined()) << "log of undefined Expr\n"; + if (x.type() == Float(64)) { + return Internal::Call::make(Float(64), "log_f64", vec(x), Internal::Call::Extern); + } else { + return Internal::Call::make(Float(32), "log_f32", vec(cast(x)), Internal::Call::Extern); + } } /** Return one floating point expression raised to the power of @@ -3453,30 +3415,30 @@ inline Expr log(Expr x) { * cast to Float(32). For Float(32), cleanly vectorizable, and * accurate up to the last few bits of the mantissa. Gets worse when * approaching overflow. Vectorizes cleanly. */ -inline Expr pow(Expr x, Expr y) { - user_assert(x.defined() && y.defined()) << "pow of undefined Expr\n"; +inline Expr pow(Expr x, Expr y) +{ + user_assert(x.defined() && y.defined()) << "pow of undefined Expr\n"; - if (const int *i = as_const_int(y)) { - return raise_to_integer_power(x, *i); - } + if (const int *i = as_const_int(y)) { return raise_to_integer_power(x, *i); } - if (x.type() == Float(64)) { - y = cast(y); - return Internal::Call::make(Float(64), "pow_f64", vec(x, y), Internal::Call::Extern); - } else { - x = cast(x); - y = cast(y); - return Internal::Call::make(Float(32), "pow_f32", vec(x, y), Internal::Call::Extern); - } + if (x.type() == Float(64)) { + y = cast(y); + return Internal::Call::make(Float(64), "pow_f64", vec(x, y), Internal::Call::Extern); + } else { + x = cast(x); + y = cast(y); + return Internal::Call::make(Float(32), "pow_f32", vec(x, y), Internal::Call::Extern); + } } /** Evaluate the error function erf. Only available for * Float(32). Accurate up to the last three bits of the * mantissa. Vectorizes cleanly. */ -inline Expr erf(Expr x) { - user_assert(x.defined()) << "erf of undefined Expr\n"; - user_assert(x.type() == Float(32)) << "erf only takes float arguments\n"; - return Internal::halide_erf(x); +inline Expr erf(Expr x) +{ + user_assert(x.defined()) << "erf of undefined Expr\n"; + user_assert(x.type() == Float(32)) << "erf only takes float arguments\n"; + return Internal::halide_erf(x); } /** Fast approximate cleanly vectorizable log for Float(32). Returns @@ -3494,42 +3456,43 @@ EXPORT Expr fast_exp(Expr x); * nonsense for x < 0.0f. Accurate up to the last 5 bits of the * mantissa for typical exponents. Gets worse when approaching * overflow. Vectorizes cleanly. */ -inline Expr fast_pow(Expr x, Expr y) { - if (const int *i = as_const_int(y)) { - return raise_to_integer_power(x, *i); - } +inline Expr fast_pow(Expr x, Expr y) +{ + if (const int *i = as_const_int(y)) { return raise_to_integer_power(x, *i); } - x = cast(x); - y = cast(y); - return select(x == 0.0f, 0.0f, fast_exp(fast_log(x) * y)); + x = cast(x); + y = cast(y); + return select(x == 0.0f, 0.0f, fast_exp(fast_log(x) * y)); } /** Return the greatest whole number less than or equal to a * floating-point expression. If the argument is not floating-point, * it is cast to Float(32). The return value is still in floating * point, despite being a whole number. Vectorizes cleanly. */ -inline Expr floor(Expr x) { - user_assert(x.defined()) << "floor of undefined Expr\n"; - if (x.type().element_of() == Float(64)) { - return Internal::Call::make(x.type(), "floor_f64", vec(x), Internal::Call::Extern); - } else { - Type t = Float(32, x.type().width); - return Internal::Call::make(t, "floor_f32", vec(cast(t, x)), Internal::Call::Extern); - } +inline Expr floor(Expr x) +{ + user_assert(x.defined()) << "floor of undefined Expr\n"; + if (x.type().element_of() == Float(64)) { + return Internal::Call::make(x.type(), "floor_f64", vec(x), Internal::Call::Extern); + } else { + Type t = Float(32, x.type().width); + return Internal::Call::make(t, "floor_f32", vec(cast(t, x)), Internal::Call::Extern); + } } /** Return the least whole number greater than or equal to a * floating-point expression. If the argument is not floating-point, * it is cast to Float(32). The return value is still in floating * point, despite being a whole number. Vectorizes cleanly. */ -inline Expr ceil(Expr x) { - user_assert(x.defined()) << "ceil of undefined Expr\n"; - if (x.type().element_of() == Float(64)) { - return Internal::Call::make(x.type(), "ceil_f64", vec(x), Internal::Call::Extern); - } else { - Type t = Float(32, x.type().width); - return Internal::Call::make(t, "ceil_f32", vec(cast(t, x)), Internal::Call::Extern); - } +inline Expr ceil(Expr x) +{ + user_assert(x.defined()) << "ceil of undefined Expr\n"; + if (x.type().element_of() == Float(64)) { + return Internal::Call::make(x.type(), "ceil_f64", vec(x), Internal::Call::Extern); + } else { + Type t = Float(32, x.type().width); + return Internal::Call::make(t, "ceil_f32", vec(cast(t, x)), Internal::Call::Extern); + } } /** Return the whole number closest to a floating-point expression. If the @@ -3537,110 +3500,106 @@ inline Expr ceil(Expr x) { * is still in floating point, despite being a whole number. On ties, we * follow IEEE754 conventions and round to the nearest even number. Vectorizes * cleanly. */ -inline Expr round(Expr x) { - user_assert(x.defined()) << "round of undefined Expr\n"; - if (x.type().element_of() == Float(64)) { - return Internal::Call::make(Float(64), "round_f64", vec(x), Internal::Call::Extern); - } else { - Type t = Float(32, x.type().width); - return Internal::Call::make(t, "round_f32", vec(cast(t, x)), Internal::Call::Extern); - } +inline Expr round(Expr x) +{ + user_assert(x.defined()) << "round of undefined Expr\n"; + if (x.type().element_of() == Float(64)) { + return Internal::Call::make(Float(64), "round_f64", vec(x), Internal::Call::Extern); + } else { + Type t = Float(32, x.type().width); + return Internal::Call::make(t, "round_f32", vec(cast(t, x)), Internal::Call::Extern); + } } /** Return the integer part of a floating-point expression. If the argument is * not floating-point, it is cast to Float(32). The return value is still in * floating point, despite being a whole number. Vectorizes cleanly. */ -inline Expr trunc(Expr x) { - user_assert(x.defined()) << "trunc of undefined Expr\n"; - if (x.type().element_of() == Float(64)) { - return Internal::Call::make(Float(64), "trunc_f64", vec(x), Internal::Call::Extern); - } else { - Type t = Float(32, x.type().width); - return Internal::Call::make(t, "trunc_f32", vec(cast(t, x)), Internal::Call::Extern); - } +inline Expr trunc(Expr x) +{ + user_assert(x.defined()) << "trunc of undefined Expr\n"; + if (x.type().element_of() == Float(64)) { + return Internal::Call::make(Float(64), "trunc_f64", vec(x), Internal::Call::Extern); + } else { + Type t = Float(32, x.type().width); + return Internal::Call::make(t, "trunc_f32", vec(cast(t, x)), Internal::Call::Extern); + } } /** Return the fractional part of a floating-point expression. If the argument * is not floating-point, it is cast to Float(32). The return value has the * same sign as the original expression. Vectorizes cleanly. */ -inline Expr fract(Expr x) { - user_assert(x.defined()) << "fract of undefined Expr\n"; - return x - trunc(x); +inline Expr fract(Expr x) +{ + user_assert(x.defined()) << "fract of undefined Expr\n"; + return x - trunc(x); } /** Reinterpret the bits of one value as another type. */ -inline Expr reinterpret(Type t, Expr e) { - user_assert(e.defined()) << "reinterpret of undefined Expr\n"; - int from_bits = e.type().bits * e.type().width; - int to_bits = t.bits * t.width; - user_assert(from_bits == to_bits) - << "Reinterpret cast from type " << e.type() - << " which has " << from_bits - << " bits, to type " << t - << " which has " << to_bits << " bits\n"; - return Internal::Call::make(t, Internal::Call::reinterpret, vec(e), Internal::Call::Intrinsic); +inline Expr reinterpret(Type t, Expr e) +{ + user_assert(e.defined()) << "reinterpret of undefined Expr\n"; + int from_bits = e.type().bits * e.type().width; + int to_bits = t.bits * t.width; + user_assert(from_bits == to_bits) << "Reinterpret cast from type " << e.type() << " which has " << from_bits + << " bits, to type " << t << " which has " << to_bits << " bits\n"; + return Internal::Call::make(t, Internal::Call::reinterpret, vec(e), Internal::Call::Intrinsic); } -template -inline Expr reinterpret(Expr e) { - return reinterpret(type_of(), e); -} +template inline Expr reinterpret(Expr e) { return reinterpret(type_of(), e); } /** Return the bitwise and of two expressions (which need not have the * same type). The type of the result is the type of the first * argument. */ -inline Expr operator&(Expr x, Expr y) { - user_assert(x.defined() && y.defined()) << "bitwise and of undefined Expr\n"; - // First widen or narrow, then bitcast. - if (y.type().bits != x.type().bits) { - Type t = y.type(); - t.bits = x.type().bits; - y = cast(t, y); - } - if (y.type() != x.type()) { - y = reinterpret(x.type(), y); - } - return Internal::Call::make(x.type(), Internal::Call::bitwise_and, vec(x, y), Internal::Call::Intrinsic); +inline Expr operator&(Expr x, Expr y) +{ + user_assert(x.defined() && y.defined()) << "bitwise and of undefined Expr\n"; + // First widen or narrow, then bitcast. + if (y.type().bits != x.type().bits) { + Type t = y.type(); + t.bits = x.type().bits; + y = cast(t, y); + } + if (y.type() != x.type()) { y = reinterpret(x.type(), y); } + return Internal::Call::make(x.type(), Internal::Call::bitwise_and, vec(x, y), Internal::Call::Intrinsic); } /** Return the bitwise or of two expressions (which need not have the * same type). The type of the result is the type of the first * argument. */ -inline Expr operator|(Expr x, Expr y) { - user_assert(x.defined() && y.defined()) << "bitwise or of undefined Expr\n"; - // First widen or narrow, then bitcast. - if (y.type().bits != x.type().bits) { - Type t = y.type(); - t.bits = x.type().bits; - y = cast(t, y); - } - if (y.type() != x.type()) { - y = reinterpret(x.type(), y); - } - return Internal::Call::make(x.type(), Internal::Call::bitwise_or, vec(x, y), Internal::Call::Intrinsic); +inline Expr operator|(Expr x, Expr y) +{ + user_assert(x.defined() && y.defined()) << "bitwise or of undefined Expr\n"; + // First widen or narrow, then bitcast. + if (y.type().bits != x.type().bits) { + Type t = y.type(); + t.bits = x.type().bits; + y = cast(t, y); + } + if (y.type() != x.type()) { y = reinterpret(x.type(), y); } + return Internal::Call::make(x.type(), Internal::Call::bitwise_or, vec(x, y), Internal::Call::Intrinsic); } /** Return the bitwise exclusive or of two expressions (which need not * have the same type). The type of the result is the type of the * first argument. */ -inline Expr operator^(Expr x, Expr y) { - user_assert(x.defined() && y.defined()) << "bitwise or of undefined Expr\n"; - // First widen or narrow, then bitcast. - if (y.type().bits != x.type().bits) { - Type t = y.type(); - t.bits = x.type().bits; - y = cast(t, y); - } - if (y.type() != x.type()) { - y = reinterpret(x.type(), y); - } - return Internal::Call::make(x.type(), Internal::Call::bitwise_xor, vec(x, y), Internal::Call::Intrinsic); +inline Expr operator^(Expr x, Expr y) +{ + user_assert(x.defined() && y.defined()) << "bitwise or of undefined Expr\n"; + // First widen or narrow, then bitcast. + if (y.type().bits != x.type().bits) { + Type t = y.type(); + t.bits = x.type().bits; + y = cast(t, y); + } + if (y.type() != x.type()) { y = reinterpret(x.type(), y); } + return Internal::Call::make(x.type(), Internal::Call::bitwise_xor, vec(x, y), Internal::Call::Intrinsic); } /** Return the bitwise not of an expression. */ -inline Expr operator~(Expr x) { - user_assert(x.defined()) << "bitwise or of undefined Expr\n"; - return Internal::Call::make(x.type(), Internal::Call::bitwise_not, vec(x), Internal::Call::Intrinsic); +inline Expr operator~(Expr x) +{ + user_assert(x.defined()) << "bitwise or of undefined Expr\n"; + return Internal::Call::make(x.type(), Internal::Call::bitwise_not, vec(x), Internal::Call::Intrinsic); } /** Shift the bits of an integer value left. This is actually less @@ -3650,12 +3609,13 @@ inline Expr operator~(Expr x) { * shifting (e.g. because the exponent is a run-time parameter). The * type of the result is equal to the type of the first argument. Both * arguments must have integer type. */ -inline Expr operator<<(Expr x, Expr y) { - user_assert(x.defined() && y.defined()) << "shift left of undefined Expr\n"; - user_assert(!x.type().is_float()) << "First argument to shift left is a float: " << x << "\n"; - user_assert(!y.type().is_float()) << "Second argument to shift left is a float: " << y << "\n"; - Internal::match_types(x, y); - return Internal::Call::make(x.type(), Internal::Call::shift_left, vec(x, y), Internal::Call::Intrinsic); +inline Expr operator<<(Expr x, Expr y) +{ + user_assert(x.defined() && y.defined()) << "shift left of undefined Expr\n"; + user_assert(!x.type().is_float()) << "First argument to shift left is a float: " << x << "\n"; + user_assert(!y.type().is_float()) << "Second argument to shift left is a float: " << y << "\n"; + Internal::match_types(x, y); + return Internal::Call::make(x.type(), Internal::Call::shift_left, vec(x, y), Internal::Call::Intrinsic); } /** Shift the bits of an integer value right. Does sign extension for @@ -3666,12 +3626,13 @@ inline Expr operator<<(Expr x, Expr y) { * division and can work with it. The type of the result is equal to * the type of the first argument. Both arguments must have integer * type. */ -inline Expr operator>>(Expr x, Expr y) { - user_assert(x.defined() && y.defined()) << "shift right of undefined Expr\n"; - user_assert(!x.type().is_float()) << "First argument to shift right is a float: " << x << "\n"; - user_assert(!y.type().is_float()) << "Second argument to shift right is a float: " << y << "\n"; - Internal::match_types(x, y); - return Internal::Call::make(x.type(), Internal::Call::shift_right, vec(x, y), Internal::Call::Intrinsic); +inline Expr operator>>(Expr x, Expr y) +{ + user_assert(x.defined() && y.defined()) << "shift right of undefined Expr\n"; + user_assert(!x.type().is_float()) << "First argument to shift right is a float: " << x << "\n"; + user_assert(!y.type().is_float()) << "Second argument to shift right is a float: " << y << "\n"; + Internal::match_types(x, y); + return Internal::Call::make(x.type(), Internal::Call::shift_right, vec(x, y), Internal::Call::Intrinsic); } /** Linear interpolate between the two values according to a weight. @@ -3740,68 +3701,64 @@ inline Expr operator>>(Expr x, Expr y) { * * \endcode * */ -inline Expr lerp(Expr zero_val, Expr one_val, Expr weight) { - user_assert(zero_val.defined()) << "lerp with undefined zero value"; - user_assert(one_val.defined()) << "lerp with undefined one value"; - user_assert(weight.defined()) << "lerp with undefined weight"; - - // We allow integer constants through, so that you can say things - // like lerp(0, cast(x), alpha) and produce an 8-bit - // result. Note that lerp(0.0f, cast(x), alpha) will - // produce an error, as will lerp(0.0f, cast(x), - // alpha). lerp(0, cast(x), alpha) is also allowed and will - // produce a float result. - if (as_const_int(zero_val)) { - zero_val = cast(one_val.type(), zero_val); +inline Expr lerp(Expr zero_val, Expr one_val, Expr weight) +{ + user_assert(zero_val.defined()) << "lerp with undefined zero value"; + user_assert(one_val.defined()) << "lerp with undefined one value"; + user_assert(weight.defined()) << "lerp with undefined weight"; + + // We allow integer constants through, so that you can say things + // like lerp(0, cast(x), alpha) and produce an 8-bit + // result. Note that lerp(0.0f, cast(x), alpha) will + // produce an error, as will lerp(0.0f, cast(x), + // alpha). lerp(0, cast(x), alpha) is also allowed and will + // produce a float result. + if (as_const_int(zero_val)) { zero_val = cast(one_val.type(), zero_val); } + if (as_const_int(one_val)) { one_val = cast(zero_val.type(), one_val); } + + user_assert(zero_val.type() == one_val.type()) + << "Can't lerp between " << zero_val << " of type " << zero_val.type() << " and " << one_val + << " of different type " << one_val.type() << "\n"; + user_assert((weight.type().is_uint() || weight.type().is_float())) + << "A lerp weight must be an unsigned integer or a float, but " + << "lerp weight " << weight << " has type " << weight.type() << ".\n"; + user_assert((zero_val.type().is_float() || zero_val.type().width <= 32)) + << "Lerping between 64-bit integers is not supported\n"; + // Compilation error for constant weight that is out of range for integer use + // as this seems like an easy to catch gotcha. + if (!zero_val.type().is_float()) { + const float *const_weight = as_const_float(weight); + if (const_weight) { + user_assert(*const_weight >= 0.0f && *const_weight <= 1.0f) + << "Floating-point weight for lerp with integer arguments is " << *const_weight + << ", which is not in the range [0.0f, 1.0f].\n"; } - if (as_const_int(one_val)) { - one_val = cast(zero_val.type(), one_val); - } - - user_assert(zero_val.type() == one_val.type()) - << "Can't lerp between " << zero_val << " of type " << zero_val.type() - << " and " << one_val << " of different type " << one_val.type() << "\n"; - user_assert((weight.type().is_uint() || weight.type().is_float())) - << "A lerp weight must be an unsigned integer or a float, but " - << "lerp weight " << weight << " has type " << weight.type() << ".\n"; - user_assert((zero_val.type().is_float() || zero_val.type().width <= 32)) - << "Lerping between 64-bit integers is not supported\n"; - // Compilation error for constant weight that is out of range for integer use - // as this seems like an easy to catch gotcha. - if (!zero_val.type().is_float()) { - const float *const_weight = as_const_float(weight); - if (const_weight) { - user_assert(*const_weight >= 0.0f && *const_weight <= 1.0f) - << "Floating-point weight for lerp with integer arguments is " - << *const_weight << ", which is not in the range [0.0f, 1.0f].\n"; - } - } - return Internal::Call::make(zero_val.type(), Internal::Call::lerp, - vec(zero_val, one_val, weight), - Internal::Call::Intrinsic); + } + return Internal::Call::make( + zero_val.type(), Internal::Call::lerp, vec(zero_val, one_val, weight), Internal::Call::Intrinsic); } /** Count the number of set bits in an expression. */ -inline Expr popcount(Expr x) { - user_assert(x.defined()) << "popcount of undefined Expr\n"; - return Internal::Call::make(x.type(), Internal::Call::popcount, - vec(x), Internal::Call::Intrinsic); +inline Expr popcount(Expr x) +{ + user_assert(x.defined()) << "popcount of undefined Expr\n"; + return Internal::Call::make(x.type(), Internal::Call::popcount, vec(x), Internal::Call::Intrinsic); } /** Count the number of leading zero bits in an expression. The result is * undefined if the value of the expression is zero. */ -inline Expr count_leading_zeros(Expr x) { - user_assert(x.defined()) << "count leading zeros of undefined Expr\n"; - return Internal::Call::make(x.type(), Internal::Call::count_leading_zeros, - vec(x), Internal::Call::Intrinsic); +inline Expr count_leading_zeros(Expr x) +{ + user_assert(x.defined()) << "count leading zeros of undefined Expr\n"; + return Internal::Call::make(x.type(), Internal::Call::count_leading_zeros, vec(x), Internal::Call::Intrinsic); } /** Count the number of trailing zero bits in an expression. The result is * undefined if the value of the expression is zero. */ -inline Expr count_trailing_zeros(Expr x) { - user_assert(x.defined()) << "count trailing zeros of undefined Expr\n"; - return Internal::Call::make(x.type(), Internal::Call::count_trailing_zeros, - vec(x), Internal::Call::Intrinsic); +inline Expr count_trailing_zeros(Expr x) +{ + user_assert(x.defined()) << "count trailing zeros of undefined Expr\n"; + return Internal::Call::make(x.type(), Internal::Call::count_trailing_zeros, vec(x), Internal::Call::Intrinsic); } /** Return a random variable representing a uniformly distributed @@ -3833,51 +3790,50 @@ inline Expr count_trailing_zeros(Expr x) { * * This function vectorizes cleanly. */ -inline Expr random_float(Expr seed = Expr()) { - // Random floats get even IDs - static int counter = -2; - counter += 2; - - std::vector args; - if (seed.defined()) { - user_assert(seed.type() == Int(32)) - << "The seed passed to random_float must have type Int(32), but instead is " - << seed << " of type " << seed.type() << "\n"; - args.push_back(seed); - } - args.push_back(counter); +inline Expr random_float(Expr seed = Expr()) +{ + // Random floats get even IDs + static int counter = -2; + counter += 2; + + std::vector args; + if (seed.defined()) { + user_assert(seed.type() == Int(32)) << "The seed passed to random_float must have type Int(32), but instead is " + << seed << " of type " << seed.type() << "\n"; + args.push_back(seed); + } + args.push_back(counter); - return Internal::Call::make(Float(32), Internal::Call::random, - args, Internal::Call::Intrinsic); + return Internal::Call::make(Float(32), Internal::Call::random, args, Internal::Call::Intrinsic); } /** Return a random variable representing a uniformly distributed * 32-bit integer. See \ref random_float. Vectorizes cleanly. */ -inline Expr random_int(Expr seed = Expr()) { - // Random ints get odd IDs - static int counter = -1; - counter += 2; - - std::vector args; - if (seed.defined()) { - user_assert(seed.type() == Int(32)) - << "The seed passed to random_int must have type Int(32), but instead is " - << seed << " of type " << seed.type() << "\n"; - args.push_back(seed); - } - args.push_back(counter); +inline Expr random_int(Expr seed = Expr()) +{ + // Random ints get odd IDs + static int counter = -1; + counter += 2; + + std::vector args; + if (seed.defined()) { + user_assert(seed.type() == Int(32)) << "The seed passed to random_int must have type Int(32), but instead is " + << seed << " of type " << seed.type() << "\n"; + args.push_back(seed); + } + args.push_back(counter); - return Internal::Call::make(Int(32), Internal::Call::random, - args, Internal::Call::Intrinsic); + return Internal::Call::make(Int(32), Internal::Call::random, args, Internal::Call::Intrinsic); } // For the purposes of a call to print, const char * can convert // silently to an Expr -struct PrintArg { - Expr expr; - PrintArg(const char *str) : expr(std::string(str)) {} - template PrintArg(T e) : expr(e) {} - operator Expr() {return expr;} +struct PrintArg +{ + Expr expr; + PrintArg(const char *str) : expr(std::string(str)) {} + template PrintArg(T e) : expr(e) {} + operator Expr() { return expr; } }; /** Create an Expr that prints out its value whenever it is @@ -3885,29 +3841,25 @@ struct PrintArg { * list, separated by spaces. This can include string literals. */ // @{ EXPORT Expr print(const std::vector &values); -inline Expr print(PrintArg a) { - return print(Internal::vec(a)); -} -inline Expr print(PrintArg a, PrintArg b) { - return print(Internal::vec(a, b)); -} -inline Expr print(PrintArg a, PrintArg b, PrintArg c) { - return print(Internal::vec(a, b, c)); -} -inline Expr print(PrintArg a, PrintArg b, PrintArg c, PrintArg d) { - return print(Internal::vec(a, b, c, d)); -} -inline Expr print(PrintArg a, PrintArg b, PrintArg c, PrintArg d, PrintArg e) { - return print(Internal::vec(a, b, c, d, e)); -} -inline Expr print(PrintArg a, PrintArg b, PrintArg c, PrintArg d, PrintArg e, PrintArg f) { - return print(Internal::vec(a, b, c, d, e, f)); -} -inline Expr print(PrintArg a, PrintArg b, PrintArg c, PrintArg d, PrintArg e, PrintArg f, PrintArg g) { - return print(Internal::vec(a, b, c, d, e, f, g)); -} -inline Expr print(PrintArg a, PrintArg b, PrintArg c, PrintArg d, PrintArg e, PrintArg f, PrintArg g, PrintArg h) { - return print(Internal::vec(a, b, c, d, e, f, g, h)); +inline Expr print(PrintArg a) { return print(Internal::vec(a)); } +inline Expr print(PrintArg a, PrintArg b) { return print(Internal::vec(a, b)); } +inline Expr print(PrintArg a, PrintArg b, PrintArg c) { return print(Internal::vec(a, b, c)); } +inline Expr print(PrintArg a, PrintArg b, PrintArg c, PrintArg d) { return print(Internal::vec(a, b, c, d)); } +inline Expr print(PrintArg a, PrintArg b, PrintArg c, PrintArg d, PrintArg e) +{ + return print(Internal::vec(a, b, c, d, e)); +} +inline Expr print(PrintArg a, PrintArg b, PrintArg c, PrintArg d, PrintArg e, PrintArg f) +{ + return print(Internal::vec(a, b, c, d, e, f)); +} +inline Expr print(PrintArg a, PrintArg b, PrintArg c, PrintArg d, PrintArg e, PrintArg f, PrintArg g) +{ + return print(Internal::vec(a, b, c, d, e, f, g)); +} +inline Expr print(PrintArg a, PrintArg b, PrintArg c, PrintArg d, PrintArg e, PrintArg f, PrintArg g, PrintArg h) +{ + return print(Internal::vec(a, b, c, d, e, f, g, h)); } // @} @@ -3915,29 +3867,43 @@ inline Expr print(PrintArg a, PrintArg b, PrintArg c, PrintArg d, PrintArg e, Pr * the condition is true. */ // @{ EXPORT Expr print_when(Expr condition, const std::vector &values); -inline Expr print_when(Expr condition, PrintArg a) { - return print_when(condition, Internal::vec(a)); -} -inline Expr print_when(Expr condition, PrintArg a, PrintArg b) { - return print_when(condition, Internal::vec(a, b)); -} -inline Expr print_when(Expr condition, PrintArg a, PrintArg b, PrintArg c) { - return print_when(condition, Internal::vec(a, b, c)); -} -inline Expr print_when(Expr condition, PrintArg a, PrintArg b, PrintArg c, PrintArg d) { - return print_when(condition, Internal::vec(a, b, c, d)); -} -inline Expr print_when(Expr condition, PrintArg a, PrintArg b, PrintArg c, PrintArg d, PrintArg e) { - return print_when(condition, Internal::vec(a, b, c, d, e)); -} -inline Expr print_when(Expr condition, PrintArg a, PrintArg b, PrintArg c, PrintArg d, PrintArg e, PrintArg f) { - return print_when(condition, Internal::vec(a, b, c, d, e, f)); -} -inline Expr print_when(Expr condition, PrintArg a, PrintArg b, PrintArg c, PrintArg d, PrintArg e, PrintArg f, PrintArg g) { - return print_when(condition, Internal::vec(a, b, c, d, e, f, g)); -} -inline Expr print_when(Expr condition, PrintArg a, PrintArg b, PrintArg c, PrintArg d, PrintArg e, PrintArg f, PrintArg g, PrintArg h) { - return print_when(condition, Internal::vec(a, b, c, d, e, f, g, h)); +inline Expr print_when(Expr condition, PrintArg a) { return print_when(condition, Internal::vec(a)); } +inline Expr print_when(Expr condition, PrintArg a, PrintArg b) +{ + return print_when(condition, Internal::vec(a, b)); +} +inline Expr print_when(Expr condition, PrintArg a, PrintArg b, PrintArg c) +{ + return print_when(condition, Internal::vec(a, b, c)); +} +inline Expr print_when(Expr condition, PrintArg a, PrintArg b, PrintArg c, PrintArg d) +{ + return print_when(condition, Internal::vec(a, b, c, d)); +} +inline Expr print_when(Expr condition, PrintArg a, PrintArg b, PrintArg c, PrintArg d, PrintArg e) +{ + return print_when(condition, Internal::vec(a, b, c, d, e)); +} +inline Expr print_when(Expr condition, PrintArg a, PrintArg b, PrintArg c, PrintArg d, PrintArg e, PrintArg f) +{ + return print_when(condition, Internal::vec(a, b, c, d, e, f)); +} +inline Expr + print_when(Expr condition, PrintArg a, PrintArg b, PrintArg c, PrintArg d, PrintArg e, PrintArg f, PrintArg g) +{ + return print_when(condition, Internal::vec(a, b, c, d, e, f, g)); +} +inline Expr print_when(Expr condition, + PrintArg a, + PrintArg b, + PrintArg c, + PrintArg d, + PrintArg e, + PrintArg f, + PrintArg g, + PrintArg h) +{ + return print_when(condition, Internal::vec(a, b, c, d, e, f, g, h)); } // @} @@ -3960,16 +3926,12 @@ inline Expr print_when(Expr condition, PrintArg a, PrintArg b, PrintArg c, Print * Use this feature with great caution, as you can use it to load from * uninitialized memory. */ -inline Expr undef(Type t) { - return Internal::Call::make(t, Internal::Call::undef, - std::vector(), - Internal::Call::Intrinsic); +inline Expr undef(Type t) +{ + return Internal::Call::make(t, Internal::Call::undef, std::vector(), Internal::Call::Intrinsic); } -template -inline Expr undef() { - return undef(type_of()); -} +template inline Expr undef() { return undef(type_of()); } /** Control the values used in the memoization cache key for memoize. * Normally parameters and other external dependencies are @@ -3999,46 +3961,46 @@ inline Expr undef() { * on the digest. */ // @{ EXPORT Expr memoize_tag(Expr result, const std::vector &cache_key_values); -inline Expr memoize_tag(Expr result) { - return memoize_tag(result, std::vector()); -} -inline Expr memoize_tag(Expr result, Expr a) { - return memoize_tag(result, Internal::vec(a)); -} -inline Expr memoize_tag(Expr result, Expr a, Expr b) { - return memoize_tag(result, Internal::vec(a, b)); -} -inline Expr memoize_tag(Expr result, Expr a, Expr b, Expr c) { - return memoize_tag(result, Internal::vec(a, b, c)); +inline Expr memoize_tag(Expr result) { return memoize_tag(result, std::vector()); } +inline Expr memoize_tag(Expr result, Expr a) { return memoize_tag(result, Internal::vec(a)); } +inline Expr memoize_tag(Expr result, Expr a, Expr b) { return memoize_tag(result, Internal::vec(a, b)); } +inline Expr memoize_tag(Expr result, Expr a, Expr b, Expr c) +{ + return memoize_tag(result, Internal::vec(a, b, c)); } -inline Expr memoize_tag(Expr result, Expr a, Expr b, Expr c, Expr d) { - return memoize_tag(result, Internal::vec(a, b, c, d)); +inline Expr memoize_tag(Expr result, Expr a, Expr b, Expr c, Expr d) +{ + return memoize_tag(result, Internal::vec(a, b, c, d)); } -inline Expr memoize_tag(Expr result, Expr a, Expr b, Expr c, Expr d, Expr e) { - return memoize_tag(result, Internal::vec(a, b, c, d, e)); +inline Expr memoize_tag(Expr result, Expr a, Expr b, Expr c, Expr d, Expr e) +{ + return memoize_tag(result, Internal::vec(a, b, c, d, e)); } -inline Expr memoize_tag(Expr result, Expr a, Expr b, Expr c, Expr d, Expr e, Expr f) { - return memoize_tag(result, Internal::vec(a, b, c, d, e, f)); +inline Expr memoize_tag(Expr result, Expr a, Expr b, Expr c, Expr d, Expr e, Expr f) +{ + return memoize_tag(result, Internal::vec(a, b, c, d, e, f)); } -inline Expr memoize_tag(Expr result, Expr a, Expr b, Expr c, Expr d, Expr e, Expr f, Expr g) { - return memoize_tag(result, Internal::vec(a, b, c, d, e, f, g)); +inline Expr memoize_tag(Expr result, Expr a, Expr b, Expr c, Expr d, Expr e, Expr f, Expr g) +{ + return memoize_tag(result, Internal::vec(a, b, c, d, e, f, g)); } -inline Expr memoize_tag(Expr result, Expr a, Expr b, Expr c, Expr d, Expr e, Expr f, Expr g, Expr h) { - return memoize_tag(result, Internal::vec(a, b, c, d, e, f, g, h)); +inline Expr memoize_tag(Expr result, Expr a, Expr b, Expr c, Expr d, Expr e, Expr f, Expr g, Expr h) +{ + return memoize_tag(result, Internal::vec(a, b, c, d, e, f, g, h)); } // @} -} +}// namespace Halide #endif #ifndef HALIDE_SCOPE_H #define HALIDE_SCOPE_H -#include +#include #include #include +#include #include -#include /** \file @@ -4048,58 +4010,54 @@ inline Expr memoize_tag(Expr result, Expr a, Expr b, Expr c, Expr d, Expr e, Exp namespace Halide { namespace Internal { -/** A stack which can store one item very efficiently. Using this - * instead of std::stack speeds up Scope substantially. */ -template -class SmallStack { -private: + /** A stack which can store one item very efficiently. Using this + * instead of std::stack speeds up Scope substantially. */ + template class SmallStack + { + private: T _top; std::vector _rest; bool _empty; -public: + public: SmallStack() : _empty(true) {} - void pop() { - if (_rest.empty()) { - _empty = true; - _top = T(); - } else { - _top = _rest.back(); - _rest.pop_back(); - } + void pop() + { + if (_rest.empty()) { + _empty = true; + _top = T(); + } else { + _top = _rest.back(); + _rest.pop_back(); + } } - void push(const T &t) { - if (_empty) { - _empty = false; - } else { - _rest.push_back(_top); - } - _top = t; + void push(const T &t) + { + if (_empty) { + _empty = false; + } else { + _rest.push_back(_top); + } + _top = t; } - T top() const { - return _top; - } + T top() const { return _top; } - T &top_ref() { - return _top; - } + T &top_ref() { return _top; } - bool empty() const { - return _empty; - } -}; + bool empty() const { return _empty; } + }; -/** A common pattern when traversing Halide IR is that you need to - * keep track of stuff when you find a Let or a LetStmt, and that it - * should hide previous values with the same name until you leave the - * Let or LetStmt nodes This class helps with that. */ -template -class Scope { -private: - std::map > table; + /** A common pattern when traversing Halide IR is that you need to + * keep track of stuff when you find a Let or a LetStmt, and that it + * should hide previous values with the same name until you leave the + * Let or LetStmt nodes This class helps with that. */ + template class Scope + { + private: + std::map> table; // Copying a scope object copies a large table full of strings and // stacks. Bad idea. @@ -4109,174 +4067,142 @@ class Scope { const Scope *containing_scope; -public: + public: Scope() : containing_scope(NULL) {} /** Set the parent scope. If lookups fail in this scope, they * check the containing scope before returning an error. Caller is * responsible for managing the memory of the containing scope. */ - void set_containing_scope(const Scope *s) { - containing_scope = s; - } + void set_containing_scope(const Scope *s) { containing_scope = s; } /** A const ref to an empty scope. Useful for default function * arguments, which would otherwise require a copy constructor * (with llvm in c++98 mode) */ - static const Scope &empty_scope() { - static Scope _empty_scope; - return _empty_scope; + static const Scope &empty_scope() + { + static Scope _empty_scope; + return _empty_scope; } /** Retrieve the value referred to by a name */ - T get(const std::string &name) const { - typename std::map >::const_iterator iter = table.find(name); - if (iter == table.end() || iter->second.empty()) { - if (containing_scope) { - return containing_scope->get(name); - } else { - internal_error << "Symbol '" << name << "' not found\n"; - } + T get(const std::string &name) const + { + typename std::map>::const_iterator iter = table.find(name); + if (iter == table.end() || iter->second.empty()) { + if (containing_scope) { + return containing_scope->get(name); + } else { + internal_error << "Symbol '" << name << "' not found\n"; } - return iter->second.top(); + } + return iter->second.top(); } /** Return a reference to an entry. Does not consider the containing scope. */ - T &ref(const std::string &name) { - typename std::map >::iterator iter = table.find(name); - if (iter == table.end() || iter->second.empty()) { - internal_error << "Symbol '" << name << "' not found\n"; - } - return iter->second.top_ref(); + T &ref(const std::string &name) + { + typename std::map>::iterator iter = table.find(name); + if (iter == table.end() || iter->second.empty()) { internal_error << "Symbol '" << name << "' not found\n"; } + return iter->second.top_ref(); } /** Tests if a name is in scope */ - bool contains(const std::string &name) const { - typename std::map >::const_iterator iter = table.find(name); - if (iter == table.end() || iter->second.empty()) { - if (containing_scope) { - return containing_scope->contains(name); - } else { - return false; - } + bool contains(const std::string &name) const + { + typename std::map>::const_iterator iter = table.find(name); + if (iter == table.end() || iter->second.empty()) { + if (containing_scope) { + return containing_scope->contains(name); + } else { + return false; } - return true; + } + return true; } /** Add a new (name, value) pair to the current scope. Hide old * values that have this name until we pop this name. */ - void push(const std::string &name, const T &value) { - table[name].push(value); - } + void push(const std::string &name, const T &value) { table[name].push(value); } /** A name goes out of scope. Restore whatever its old value * was (or remove it entirely if there was nothing else of the * same name in an outer scope) */ - void pop(const std::string &name) { - typename std::map >::iterator iter = table.find(name); - internal_assert(iter != table.end()) << "Name not in symbol table: " << name << "\n"; - iter->second.pop(); - if (iter->second.empty()) { - table.erase(iter); - } + void pop(const std::string &name) + { + typename std::map>::iterator iter = table.find(name); + internal_assert(iter != table.end()) << "Name not in symbol table: " << name << "\n"; + iter->second.pop(); + if (iter->second.empty()) { table.erase(iter); } } /** Iterate through the scope. Does not capture any containing scope. */ - class const_iterator { - typename std::map >::const_iterator iter; + class const_iterator + { + typename std::map>::const_iterator iter; + public: - explicit const_iterator(const typename std::map >::const_iterator &i) : - iter(i) { - } + explicit const_iterator(const typename std::map>::const_iterator &i) : iter(i) {} - const_iterator() {} + const_iterator() {} - bool operator!=(const const_iterator &other) { - return iter != other.iter; - } + bool operator!=(const const_iterator &other) { return iter != other.iter; } - void operator++() { - ++iter; - } + void operator++() { ++iter; } - const std::string &name() { - return iter->first; - } + const std::string &name() { return iter->first; } - const SmallStack &stack() { - return iter->second; - } + const SmallStack &stack() { return iter->second; } - const T &value() { - return iter->second.top(); - } + const T &value() { return iter->second.top(); } }; - const_iterator cbegin() const { - return const_iterator(table.begin()); - } + const_iterator cbegin() const { return const_iterator(table.begin()); } - const_iterator cend() const { - return const_iterator(table.end()); - } + const_iterator cend() const { return const_iterator(table.end()); } + + class iterator + { + typename std::map>::iterator iter; - class iterator { - typename std::map >::iterator iter; public: - explicit iterator(typename std::map >::iterator i) : - iter(i) { - } + explicit iterator(typename std::map>::iterator i) : iter(i) {} - iterator() {} + iterator() {} - bool operator!=(const iterator &other) { - return iter != other.iter; - } + bool operator!=(const iterator &other) { return iter != other.iter; } - void operator++() { - ++iter; - } + void operator++() { ++iter; } - const std::string &name() { - return iter->first; - } + const std::string &name() { return iter->first; } - SmallStack &stack() { - return iter->second; - } + SmallStack &stack() { return iter->second; } - T &value() { - return iter->second.top_ref(); - } + T &value() { return iter->second.top_ref(); } }; - iterator begin() { - return iterator(table.begin()); - } + iterator begin() { return iterator(table.begin()); } - iterator end() { - return iterator(table.end()); - } + iterator end() { return iterator(table.end()); } - void swap(Scope &other) { - table.swap(other.table); - std::swap(containing_scope, other.containing_scope); + void swap(Scope &other) + { + table.swap(other.table); + std::swap(containing_scope, other.containing_scope); } -}; + }; -template -std::ostream &operator<<(std::ostream &stream, const Scope& s) { + template std::ostream &operator<<(std::ostream &stream, const Scope &s) + { stream << "{\n"; typename Scope::const_iterator iter; - for (iter = s.cbegin(); iter != s.cend(); ++iter) { - stream << " " << iter.name() << "\n"; - } + for (iter = s.cbegin(); iter != s.cend(); ++iter) { stream << " " << iter.name() << "\n"; } stream << "}"; return stream; -} + } -} -} +}// namespace Internal +}// namespace Halide #endif #include @@ -4289,31 +4215,33 @@ std::ostream &operator<<(std::ostream &stream, const Scope& s) { namespace Halide { namespace Internal { -struct Interval { + struct Interval + { Expr min, max; Interval() {} Interval(Expr min, Expr max) : min(min), max(max) {} -}; - -typedef std::map, Interval> FuncValueBounds; - -/** Given an expression in some variables, and a map from those - * variables to their bounds (in the form of (minimum possible value, - * maximum possible value)), compute two expressions that give the - * minimum possible value and the maximum possible value of this - * expression. Max or min may be undefined expressions if the value is - * not bounded above or below. - * - * This is for tasks such as deducing the region of a buffer - * loaded by a chunk of code. - */ -Interval bounds_of_expr_in_scope(Expr expr, - const Scope &scope, - const FuncValueBounds &func_bounds = FuncValueBounds()); - -/** Represents the bounds of a region of arbitrary dimension. Zero - * dimensions corresponds to a scalar region. */ -struct Box { + }; + + typedef std::map, Interval> FuncValueBounds; + + /** Given an expression in some variables, and a map from those + * variables to their bounds (in the form of (minimum possible value, + * maximum possible value)), compute two expressions that give the + * minimum possible value and the maximum possible value of this + * expression. Max or min may be undefined expressions if the value is + * not bounded above or below. + * + * This is for tasks such as deducing the region of a buffer + * loaded by a chunk of code. + */ + Interval bounds_of_expr_in_scope(Expr expr, + const Scope &scope, + const FuncValueBounds &func_bounds = FuncValueBounds()); + + /** Represents the bounds of a region of arbitrary dimension. Zero + * dimensions corresponds to a scalar region. */ + struct Box + { /** The conditions under which this region may be touched. */ Expr used; @@ -4324,92 +4252,98 @@ struct Box { Box(size_t sz) : bounds(sz) {} Box(const std::vector &b) : bounds(b) {} - size_t size() const {return bounds.size();} - bool empty() const {return bounds.empty();} - Interval &operator[](int i) {return bounds[i];} - const Interval &operator[](int i) const {return bounds[i];} - void resize(size_t sz) {bounds.resize(sz);} - void push_back(const Interval &i) {bounds.push_back(i);} + size_t size() const { return bounds.size(); } + bool empty() const { return bounds.empty(); } + Interval &operator[](int i) { return bounds[i]; } + const Interval &operator[](int i) const { return bounds[i]; } + void resize(size_t sz) { bounds.resize(sz); } + void push_back(const Interval &i) { bounds.push_back(i); } /** Check if the used condition is defined and not trivially true. */ - bool maybe_unused() const {return used.defined() && !is_one(used);} -}; - -// Expand box a to encompass box b -void merge_boxes(Box &a, const Box &b); -// Test if box a could possibly overlap box b. -bool boxes_overlap(const Box &a, const Box &b); - -/** Compute rectangular domains large enough to cover all the 'Call's - * to each function that occurs within a given statement or - * expression. This is useful for figuring out what regions of things - * to evaluate. */ -// @{ -std::map boxes_required(Expr e, - const Scope &scope = Scope::empty_scope(), - const FuncValueBounds &func_bounds = FuncValueBounds()); -std::map boxes_required(Stmt s, - const Scope &scope = Scope::empty_scope(), - const FuncValueBounds &func_bounds = FuncValueBounds()); -// @} - -/** Compute rectangular domains large enough to cover all the - * 'Provides's to each function that occurs within a given statement - * or expression. */ -// @{ -std::map boxes_provided(Expr e, - const Scope &scope = Scope::empty_scope(), - const FuncValueBounds &func_bounds = FuncValueBounds()); -std::map boxes_provided(Stmt s, - const Scope &scope = Scope::empty_scope(), - const FuncValueBounds &func_bounds = FuncValueBounds()); -// @} - -/** Compute rectangular domains large enough to cover all the 'Call's - * and 'Provides's to each function that occurs within a given - * statement or expression. */ -// @{ -std::map boxes_touched(Expr e, - const Scope &scope = Scope::empty_scope(), - const FuncValueBounds &func_bounds = FuncValueBounds()); -std::map boxes_touched(Stmt s, - const Scope &scope = Scope::empty_scope(), - const FuncValueBounds &func_bounds = FuncValueBounds()); -// @} - -/** Variants of the above that are only concerned with a single function. */ -// @{ -Box box_required(Expr e, std::string fn, - const Scope &scope = Scope::empty_scope(), - const FuncValueBounds &func_bounds = FuncValueBounds()); -Box box_required(Stmt s, std::string fn, - const Scope &scope = Scope::empty_scope(), - const FuncValueBounds &func_bounds = FuncValueBounds()); - -Box box_provided(Expr e, std::string fn, - const Scope &scope = Scope::empty_scope(), - const FuncValueBounds &func_bounds = FuncValueBounds()); -Box box_provided(Stmt s, std::string fn, - const Scope &scope = Scope::empty_scope(), - const FuncValueBounds &func_bounds = FuncValueBounds()); - -Box box_touched(Expr e, std::string fn, - const Scope &scope = Scope::empty_scope(), - const FuncValueBounds &func_bounds = FuncValueBounds()); -Box box_touched(Stmt s, std::string fn, - const Scope &scope = Scope::empty_scope(), - const FuncValueBounds &func_bounds = FuncValueBounds()); -// @} - -/** Compute the maximum and minimum possible value for each function - * in an environment. */ -FuncValueBounds compute_function_value_bounds(const std::vector &order, - const std::map &env); - -void bounds_test(); - -} -} + bool maybe_unused() const { return used.defined() && !is_one(used); } + }; + + // Expand box a to encompass box b + void merge_boxes(Box &a, const Box &b); + // Test if box a could possibly overlap box b. + bool boxes_overlap(const Box &a, const Box &b); + + /** Compute rectangular domains large enough to cover all the 'Call's + * to each function that occurs within a given statement or + * expression. This is useful for figuring out what regions of things + * to evaluate. */ + // @{ + std::map boxes_required(Expr e, + const Scope &scope = Scope::empty_scope(), + const FuncValueBounds &func_bounds = FuncValueBounds()); + std::map boxes_required(Stmt s, + const Scope &scope = Scope::empty_scope(), + const FuncValueBounds &func_bounds = FuncValueBounds()); + // @} + + /** Compute rectangular domains large enough to cover all the + * 'Provides's to each function that occurs within a given statement + * or expression. */ + // @{ + std::map boxes_provided(Expr e, + const Scope &scope = Scope::empty_scope(), + const FuncValueBounds &func_bounds = FuncValueBounds()); + std::map boxes_provided(Stmt s, + const Scope &scope = Scope::empty_scope(), + const FuncValueBounds &func_bounds = FuncValueBounds()); + // @} + + /** Compute rectangular domains large enough to cover all the 'Call's + * and 'Provides's to each function that occurs within a given + * statement or expression. */ + // @{ + std::map boxes_touched(Expr e, + const Scope &scope = Scope::empty_scope(), + const FuncValueBounds &func_bounds = FuncValueBounds()); + std::map boxes_touched(Stmt s, + const Scope &scope = Scope::empty_scope(), + const FuncValueBounds &func_bounds = FuncValueBounds()); + // @} + + /** Variants of the above that are only concerned with a single function. */ + // @{ + Box box_required(Expr e, + std::string fn, + const Scope &scope = Scope::empty_scope(), + const FuncValueBounds &func_bounds = FuncValueBounds()); + Box box_required(Stmt s, + std::string fn, + const Scope &scope = Scope::empty_scope(), + const FuncValueBounds &func_bounds = FuncValueBounds()); + + Box box_provided(Expr e, + std::string fn, + const Scope &scope = Scope::empty_scope(), + const FuncValueBounds &func_bounds = FuncValueBounds()); + Box box_provided(Stmt s, + std::string fn, + const Scope &scope = Scope::empty_scope(), + const FuncValueBounds &func_bounds = FuncValueBounds()); + + Box box_touched(Expr e, + std::string fn, + const Scope &scope = Scope::empty_scope(), + const FuncValueBounds &func_bounds = FuncValueBounds()); + Box box_touched(Stmt s, + std::string fn, + const Scope &scope = Scope::empty_scope(), + const FuncValueBounds &func_bounds = FuncValueBounds()); + // @} + + /** Compute the maximum and minimum possible value for each function + * in an environment. */ + FuncValueBounds compute_function_value_bounds(const std::vector &order, + const std::map &env); + + void bounds_test(); + +}// namespace Internal +}// namespace Halide #endif #ifndef HALIDE_BOUNDS_INFERENCE_H @@ -4425,17 +4359,17 @@ void bounds_test(); namespace Halide { namespace Internal { -/** Take a partially lowered statement that includes symbolic - * representations of the bounds over which things should be realized, - * and inject expressions defining those bounds. - */ -Stmt bounds_inference(Stmt, - const std::vector &realization_order, - const std::map &environment, - const std::map, Interval> &func_bounds); + /** Take a partially lowered statement that includes symbolic + * representations of the bounds over which things should be realized, + * and inject expressions defining those bounds. + */ + Stmt bounds_inference(Stmt, + const std::vector &realization_order, + const std::map &environment, + const std::map, Interval> &func_bounds); -} -} +}// namespace Internal +}// namespace Halide #endif #ifndef HALIDE_CODEGEN_C_H @@ -4446,10 +4380,10 @@ Stmt bounds_inference(Stmt, * Defines an IRPrinter that emits C++ code equivalent to a halide stmt */ +#include +#include #include #include -#include -#include #ifndef HALIDE_IR_PRINTER_H #define HALIDE_IR_PRINTER_H @@ -4482,20 +4416,21 @@ EXPORT std::ostream &operator<<(std::ostream &stream, const Type &); namespace Internal { -/** Emit a halide statement on an output stream (such as std::cout) in - * a human-readable form */ -std::ostream &operator<<(std::ostream &stream, const Stmt &); - -/** Emit a halide for loop type (vectorized, serial, etc) in a human - * readable form */ -std::ostream &operator<<(std::ostream &stream, const For::ForType &); - -/** An IRVisitor that emits IR to the given output stream in a human - * readable form. Can be subclassed if you want to modify the way in - * which it prints. - */ -class IRPrinter : public IRVisitor { -public: + /** Emit a halide statement on an output stream (such as std::cout) in + * a human-readable form */ + std::ostream &operator<<(std::ostream &stream, const Stmt &); + + /** Emit a halide for loop type (vectorized, serial, etc) in a human + * readable form */ + std::ostream &operator<<(std::ostream &stream, const For::ForType &); + + /** An IRVisitor that emits IR to the given output stream in a human + * readable form. Can be subclassed if you want to modify the way in + * which it prints. + */ + class IRPrinter : public IRVisitor + { + public: /** Construct an IRPrinter pointed at a given output stream * (e.g. std::cout, or a std::ofstream) */ IRPrinter(std::ostream &); @@ -4508,7 +4443,7 @@ class IRPrinter : public IRVisitor { static void test(); -protected: + protected: /** The stream we're outputting on */ std::ostream &stream; @@ -4558,10 +4493,9 @@ class IRPrinter : public IRVisitor { void visit(const Block *); void visit(const IfThenElse *); void visit(const Evaluate *); - -}; -} -} + }; +}// namespace Internal +}// namespace Halide #endif @@ -4571,22 +4505,24 @@ struct Argument; namespace Internal { -/** This class emits C++ code equivalent to a halide Stmt. It's - * mostly the same as an IRPrinter, but it's wrapped in a function - * definition, and some things are handled differently to be valid - * C++. - */ -class CodeGen_C : public IRPrinter { -public: + /** This class emits C++ code equivalent to a halide Stmt. It's + * mostly the same as an IRPrinter, but it's wrapped in a function + * definition, and some things are handled differently to be valid + * C++. + */ + class CodeGen_C : public IRPrinter + { + public: /** Initialize a C code generator pointing at a particular output * stream (e.g. a file, or std::cout) */ CodeGen_C(std::ostream &); /** Emit source code equivalent to the given statement, wrapped in * a function with the given type signature */ - void compile(Stmt stmt, std::string name, - const std::vector &args, - const std::vector &images_to_embed); + void compile(Stmt stmt, + std::string name, + const std::vector &args, + const std::vector &images_to_embed); /** Emit a header file defining a halide pipeline with the given * type signature */ @@ -4594,7 +4530,7 @@ class CodeGen_C : public IRPrinter { static void test(); -protected: + protected: /** An ID for the most recently generated ssa variable */ std::string id; @@ -4678,10 +4614,10 @@ class CodeGen_C : public IRPrinter { void visit(const Evaluate *); void visit_binop(Type t, Expr a, Expr b, const char *op); -}; + }; -} -} +}// namespace Internal +}// namespace Halide #endif #ifndef HALIDE_CODEGEN_H @@ -4710,7 +4646,7 @@ class AllocaInst; class Constant; class Triple; class MDNode; -} +}// namespace llvm #include #include @@ -4766,15 +4702,15 @@ typedef ptrdiff_t ssize_t; #define WEAK __attribute__((weak)) #ifdef BITS_64 -#define INT64_C(c) c ## L -#define UINT64_C(c) c ## UL +#define INT64_C(c) c##L +#define UINT64_C(c) c##UL typedef uint64_t uintptr_t; typedef int64_t intptr_t; #endif #ifdef BITS_32 -#define INT64_C(c) c ## LL -#define UINT64_C(c) c ## ULL +#define INT64_C(c) c##LL +#define UINT64_C(c) c##ULL typedef uint32_t uintptr_t; typedef int32_t intptr_t; #endif @@ -4793,12 +4729,12 @@ void free(void *); void *malloc(size_t); const char *strstr(const char *, const char *); int atoi(const char *); -int strcmp(const char* s, const char* t); -int strncmp(const char* s, const char* t, size_t n); -size_t strlen(const char* s); -char *strchr(const char* s, char c); -void* memcpy(void* s1, const void* s2, size_t n); -int memcmp(const void* s1, const void* s2, size_t n); +int strcmp(const char *s, const char *t); +int strncmp(const char *s, const char *t, size_t n); +size_t strlen(const char *s); +char *strchr(const char *s, char c); +void *memcpy(void *s1, const void *s2, size_t n); +int memcmp(const void *s1, const void *s2, size_t n); void *memset(void *s, int val, size_t n); int open(const char *filename, int opts, int mode); int close(int fd); @@ -4814,123 +4750,127 @@ WEAK char *halide_double_to_string(char *dst, char *end, double arg, int scienti WEAK char *halide_int64_to_string(char *dst, char *end, int64_t arg, int digits); WEAK char *halide_uint64_to_string(char *dst, char *end, uint64_t arg, int digits); WEAK char *halide_pointer_to_string(char *dst, char *end, const void *arg); - } // A convenient namespace for weak functions that are internal to the // halide runtime. -namespace Halide { namespace Runtime { namespace Internal { +namespace Halide { +namespace Runtime { + namespace Internal { -enum PrinterType {BasicPrinter = 0, - ErrorPrinter = 1, - StringStreamPrinter = 2}; + enum PrinterType { BasicPrinter = 0, ErrorPrinter = 1, StringStreamPrinter = 2 }; -// A class for constructing debug messages from the runtime. Dumps -// items into a stack array, then prints them when the object leaves -// scope using halide_print. Think of it as a stringstream that prints -// when it dies. Use it like this: + // A class for constructing debug messages from the runtime. Dumps + // items into a stack array, then prints them when the object leaves + // scope using halide_print. Think of it as a stringstream that prints + // when it dies. Use it like this: -// debug(user_context) << "A" << b << c << "\n"; + // debug(user_context) << "A" << b << c << "\n"; -// If you use it like this: + // If you use it like this: -// debug d(user_context); -// d << "A"; -// d << b; -// d << c << "\n"; + // debug d(user_context); + // d << "A"; + // d << b; + // d << c << "\n"; -// Then remember the print only happens when the debug object leaves -// scope, which may print at a confusing time. + // Then remember the print only happens when the debug object leaves + // scope, which may print at a confusing time. -template -class Printer { -public: - char buf[1024]; - char *dst, *end; - void *user_context; + template class Printer + { + public: + char buf[1024]; + char *dst, *end; + void *user_context; - Printer(void *ctx) : dst(buf), end(buf + 1023), user_context(ctx) { - *end = 0; - } + Printer(void *ctx) : dst(buf), end(buf + 1023), user_context(ctx) { *end = 0; } - Printer &operator<<(const char *arg) { + Printer &operator<<(const char *arg) + { dst = halide_string_to_string(dst, end, arg); return *this; - } + } - Printer &operator<<(int64_t arg) { + Printer &operator<<(int64_t arg) + { dst = halide_int64_to_string(dst, end, arg, 1); return *this; - } + } - Printer &operator<<(int32_t arg) { + Printer &operator<<(int32_t arg) + { dst = halide_int64_to_string(dst, end, arg, 1); return *this; - } + } - Printer &operator<<(uint64_t arg) { + Printer &operator<<(uint64_t arg) + { dst = halide_uint64_to_string(dst, end, arg, 1); return *this; - } + } - Printer &operator<<(uint32_t arg) { + Printer &operator<<(uint32_t arg) + { dst = halide_uint64_to_string(dst, end, arg, 1); return *this; - } + } - Printer &operator<<(double arg) { + Printer &operator<<(double arg) + { dst = halide_double_to_string(dst, end, arg, 1); return *this; - } + } - Printer &operator<<(float arg) { + Printer &operator<<(float arg) + { dst = halide_double_to_string(dst, end, arg, 0); return *this; - } + } - Printer &operator<<(const void *arg) { + Printer &operator<<(const void *arg) + { dst = halide_pointer_to_string(dst, end, arg); return *this; - } + } - // Use it like a stringstream. - const char *str() { - return buf; - } + // Use it like a stringstream. + const char *str() { return buf; } - ~Printer() { + ~Printer() + { if (type == ErrorPrinter) { - halide_error(user_context, buf); + halide_error(user_context, buf); } else if (type == BasicPrinter) { - halide_print(user_context, buf); + halide_print(user_context, buf); } else { - // It's a stringstream. Do nothing. + // It's a stringstream. Do nothing. } - } -}; + } + }; -// A class that supports << with all the same types as Printer, but -// does nothing and should compile to a no-op. -class SinkPrinter { -public: - SinkPrinter(void *user_context) {} -}; -template -SinkPrinter operator<<(const SinkPrinter &s, T) { - return s; -} + // A class that supports << with all the same types as Printer, but + // does nothing and should compile to a no-op. + class SinkPrinter + { + public: + SinkPrinter(void *user_context) {} + }; + template SinkPrinter operator<<(const SinkPrinter &s, T) { return s; } -typedef Printer print; -typedef Printer error; -typedef Printer stringstream; + typedef Printer print; + typedef Printer error; + typedef Printer stringstream; #ifdef DEBUG_RUNTIME -typedef Printer debug; + typedef Printer debug; #else -typedef SinkPrinter debug; + typedef SinkPrinter debug; #endif -}}} + }// namespace Internal +}// namespace Runtime +}// namespace Halide using namespace Halide::Runtime::Internal; @@ -4990,7 +4930,8 @@ extern void halide_print(void *user_context, const char *); extern void halide_error(void *user_context, const char *); /** A macro that calls halide_error if the supplied condition is false. */ -#define halide_assert(user_context, cond) if (!(cond)) halide_error(user_context, #cond); +#define halide_assert(user_context, cond) \ + if (!(cond)) halide_error(user_context, #cond); /** These are allocated statically inside the runtime, hence the fixed * size. They must be initialized with zero. The first time @@ -4999,8 +4940,9 @@ extern void halide_error(void *user_context, const char *); * mechanism, but makes the lock reliably easy to setup and use * without depending on e.g. C++ constructor logic. */ -struct halide_mutex { - unsigned char _private[64]; +struct halide_mutex +{ + unsigned char _private[64]; }; /** A basic set of mutex functions, which call platform specific code @@ -5023,9 +4965,8 @@ extern void halide_mutex_cleanup(struct halide_mutex *mutex_arg); * jobs otherwise. */ //@{ -extern int halide_do_par_for(void *user_context, - int (*f)(void *ctx, int, uint8_t *), - int min, int size, uint8_t *closure); +extern int + halide_do_par_for(void *user_context, int (*f)(void *ctx, int, uint8_t *), int min, int size, uint8_t *closure); extern void halide_shutdown_thread_pool(); //@} @@ -5050,32 +4991,40 @@ extern void halide_free(void *user_context, void *ptr); * * Cannot be replaced in JITted code at present. */ -extern int32_t halide_debug_to_file(void *user_context, const char *filename, - uint8_t *data, int32_t s0, int32_t s1, int32_t s2, - int32_t s3, int32_t type_code, - int32_t bytes_per_element); - - -enum halide_trace_event_code {halide_trace_load = 0, - halide_trace_store = 1, - halide_trace_begin_realization = 2, - halide_trace_end_realization = 3, - halide_trace_produce = 4, - halide_trace_update = 5, - halide_trace_consume = 6, - halide_trace_end_consume = 7}; - -struct halide_trace_event { - const char *func; - halide_trace_event_code event; - int32_t parent_id; - int32_t type_code; - int32_t bits; - int32_t vector_width; - int32_t value_index; - void *value; - int32_t dimensions; - int32_t *coordinates; +extern int32_t halide_debug_to_file(void *user_context, + const char *filename, + uint8_t *data, + int32_t s0, + int32_t s1, + int32_t s2, + int32_t s3, + int32_t type_code, + int32_t bytes_per_element); + + +enum halide_trace_event_code { + halide_trace_load = 0, + halide_trace_store = 1, + halide_trace_begin_realization = 2, + halide_trace_end_realization = 3, + halide_trace_produce = 4, + halide_trace_update = 5, + halide_trace_consume = 6, + halide_trace_end_consume = 7 +}; + +struct halide_trace_event +{ + const char *func; + halide_trace_event_code event; + int32_t parent_id; + int32_t type_code; + int32_t bits; + int32_t vector_width; + int32_t value_index; + void *value; + int32_t dimensions; + int32_t *coordinates; }; /** Called when Funcs are marked as trace_load, trace_store, or @@ -5157,16 +5106,19 @@ extern int halide_dev_free(void *user_context, struct buffer_t *buf); * signature across different Halide gpu backends. Do not call * them. */ // @{ -extern int halide_init_kernels(void *user_context, void **state_ptr, - const char *src, int size); +extern int halide_init_kernels(void *user_context, void **state_ptr, const char *src, int size); extern int halide_dev_run(void *user_context, - void *state_ptr, - const char *entry_name, - int blocksX, int blocksY, int blocksZ, - int threadsX, int threadsY, int threadsZ, - int shared_mem_bytes, - size_t arg_sizes[], - void *args[]); + void *state_ptr, + const char *entry_name, + int blocksX, + int blocksY, + int blocksZ, + int threadsX, + int threadsY, + int threadsZ, + int shared_mem_bytes, + size_t arg_sizes[], + void *args[]); // @} /** This function is called to populate the buffer_t.dev field with a constant @@ -5241,8 +5193,12 @@ extern void halide_memoization_cache_set_size(int64_t size); * return a Tuple, there will only be one buffer_t in the list. The * tuple_count parameters determines the length of the list. */ -extern bool halide_memoization_cache_lookup(void *user_context, const uint8_t *cache_key, int32_t size, - buffer_t *realized_bounds, int32_t tuple_count, buffer_t **tuple_buffers); +extern bool halide_memoization_cache_lookup(void *user_context, + const uint8_t *cache_key, + int32_t size, + buffer_t *realized_bounds, + int32_t tuple_count, + buffer_t **tuple_buffers); /** Given a cache key for a memoized result, currently constructed * from the Func name and top-level Func name plus the arguments of @@ -5255,8 +5211,12 @@ extern bool halide_memoization_cache_lookup(void *user_context, const uint8_t *c * only be one buffer_t in the list. The tuple_count parameters * determines the length of the list. */ -extern void halide_memoization_cache_store(void *user_context, const uint8_t *cache_key, int32_t size, - buffer_t *realized_bounds, int32_t tuple_count, buffer_t **tuple_buffers); +extern void halide_memoization_cache_store(void *user_context, + const uint8_t *cache_key, + int32_t size, + buffer_t *realized_bounds, + int32_t tuple_count, + buffer_t **tuple_buffers); /** Free all memory and resources associated with the memoization cache. @@ -5265,10 +5225,10 @@ extern void halide_memoization_cache_store(void *user_context, const uint8_t *ca extern void halide_memoization_cache_cleanup(); #ifdef __cplusplus -} // End extern "C" +}// End extern "C" #endif -#endif // HALIDE_HALIDERUNTIME_H +#endif// HALIDE_HALIDERUNTIME_H namespace llvm { class Module; @@ -5277,13 +5237,14 @@ class Module; namespace Halide { namespace Internal { -class JITModuleHolder; -class CodeGen; + class JITModuleHolder; + class CodeGen; -/** Function pointers into a compiled halide module. These function - * pointers are meaningless once the last copy of a JITCompiledModule - * is deleted, so don't cache them. */ -struct JITCompiledModule { + /** Function pointers into a compiled halide module. These function + * pointers are meaningless once the last copy of a JITCompiledModule + * is deleted, so don't cache them. */ + struct JITCompiledModule + { /** A pointer to the raw halide function. It's true type depends * on the Argument vector passed to CodeGen::compile. Image * parameters become (buffer_t *), and scalar parameters become @@ -5300,9 +5261,9 @@ struct JITCompiledModule { * objects. These pointers may be NULL if not compiling for a * gpu-like target. */ // @{ - int (*copy_to_host)(void *user_context, struct buffer_t*); - int (*copy_to_dev)(void *user_context, struct buffer_t*); - int (*free_dev_buffer)(void *user_context, struct buffer_t*); + int (*copy_to_host)(void *user_context, struct buffer_t *); + int (*copy_to_dev)(void *user_context, struct buffer_t *); + int (*free_dev_buffer)(void *user_context, struct buffer_t *); // @} /** The type of a halide runtime error handler function */ @@ -5313,19 +5274,17 @@ struct JITCompiledModule { /** Set a custom malloc and free for this module to use. See * \ref Func::set_custom_allocator */ - void (*set_custom_allocator)(void *(*malloc)(void *user_context, size_t), - void (*free)(void *user_context, void *ptr)); + void ( + *set_custom_allocator)(void *(*malloc)(void *user_context, size_t), void (*free)(void *user_context, void *ptr)); /** Set a custom parallel for loop launcher. See * \ref Func::set_custom_do_par_for */ typedef int (*HalideTask)(void *user_context, int, uint8_t *); - void (*set_custom_do_par_for)(int (*custom_do_par_for)(void *user_context, HalideTask, - int, int, uint8_t *)); + void (*set_custom_do_par_for)(int (*custom_do_par_for)(void *user_context, HalideTask, int, int, uint8_t *)); /** Set a custom do parallel task. See * \ref Func::set_custom_do_task */ - void (*set_custom_do_task)(int (*custom_do_task)(void *user_context, HalideTask, - int, uint8_t *)); + void (*set_custom_do_task)(int (*custom_do_task)(void *user_context, HalideTask, int, uint8_t *)); /** Set a custom trace function. See \ref Func::set_custom_trace. */ typedef int (*TraceFn)(void *, const halide_trace_event *); @@ -5346,37 +5305,29 @@ struct JITCompiledModule { // The JIT Module Allocator holds onto the memory storing the functions above. IntrusivePtr module; - JITCompiledModule() : - function(NULL), - wrapped_function(NULL), - copy_to_host(NULL), - copy_to_dev(NULL), - free_dev_buffer(NULL), - set_error_handler(NULL), - set_custom_allocator(NULL), - set_custom_do_par_for(NULL), - set_custom_do_task(NULL), - set_custom_trace(NULL), - set_custom_print(NULL), - shutdown_thread_pool(NULL), - memoization_cache_set_size(NULL) {} + JITCompiledModule() + : function(NULL), wrapped_function(NULL), copy_to_host(NULL), copy_to_dev(NULL), free_dev_buffer(NULL), + set_error_handler(NULL), set_custom_allocator(NULL), set_custom_do_par_for(NULL), set_custom_do_task(NULL), + set_custom_trace(NULL), set_custom_print(NULL), shutdown_thread_pool(NULL), memoization_cache_set_size(NULL) + {} /** Take an llvm module and compile it. Populates the function * pointer members above with the result. */ void compile_module(CodeGen *cg, llvm::Module *mod, const std::string &function_name); /** Holds a cleanup routine and context parameter. */ - struct CleanupRoutine { - void (*fn)(void *); - void *context; + struct CleanupRoutine + { + void (*fn)(void *); + void *context; - CleanupRoutine() : fn(NULL), context(NULL) {} - CleanupRoutine(void (*fn)(void *), void *context) : fn(fn), context(context) {} + CleanupRoutine() : fn(NULL), context(NULL) {} + CleanupRoutine(void (*fn)(void *), void *context) : fn(fn), context(context) {} }; -}; + }; -} -} +}// namespace Internal +}// namespace Halide #endif @@ -5391,46 +5342,47 @@ struct JITCompiledModule { namespace Halide { namespace Internal { -/** The result of modulus_remainder analysis */ -struct ModulusRemainder { + /** The result of modulus_remainder analysis */ + struct ModulusRemainder + { ModulusRemainder() : modulus(0), remainder(0) {} ModulusRemainder(int m, int r) : modulus(m), remainder(r) {} int modulus, remainder; -}; - -/** For things like alignment analysis, often it's helpful to know - * if an integer expression is some multiple of a constant plus - * some other constant. For example, it is straight-forward to - * deduce that ((10*x + 2)*(6*y - 3) - 1) is congruent to five - * modulo six. - * - * We get the most information when the modulus is large. E.g. if - * something is congruent to 208 modulo 384, then we also know it's - * congruent to 0 mod 8, and we can possibly use it as an index for an - * aligned load. If all else fails, we can just say that an integer is - * congruent to zero modulo one. - */ -EXPORT ModulusRemainder modulus_remainder(Expr e); - -/** If we have alignment information about external variables, we can - * let the analysis know about that using this version of - * modulus_remainder: */ -ModulusRemainder modulus_remainder(Expr e, const Scope &scope); - -/** Reduce an expression modulo some integer. Returns true and assigns - * to remainder if an answer could be found. */ -bool reduce_expr_modulo(Expr e, int modulus, int *remainder); - -void modulus_remainder_test(); - -/** The greatest common divisor of two integers */ -int gcd(int, int); - -/** The least common multiple of two integers */ -int lcm(int, int); - -} -} + }; + + /** For things like alignment analysis, often it's helpful to know + * if an integer expression is some multiple of a constant plus + * some other constant. For example, it is straight-forward to + * deduce that ((10*x + 2)*(6*y - 3) - 1) is congruent to five + * modulo six. + * + * We get the most information when the modulus is large. E.g. if + * something is congruent to 208 modulo 384, then we also know it's + * congruent to 0 mod 8, and we can possibly use it as an index for an + * aligned load. If all else fails, we can just say that an integer is + * congruent to zero modulo one. + */ + EXPORT ModulusRemainder modulus_remainder(Expr e); + + /** If we have alignment information about external variables, we can + * let the analysis know about that using this version of + * modulus_remainder: */ + ModulusRemainder modulus_remainder(Expr e, const Scope &scope); + + /** Reduce an expression modulo some integer. Returns true and assigns + * to remainder if an answer could be found. */ + bool reduce_expr_modulo(Expr e, int modulus, int *remainder); + + void modulus_remainder_test(); + + /** The greatest common divisor of two integers */ + int gcd(int, int); + + /** The least common multiple of two integers */ + int lcm(int, int); + +}// namespace Internal +}// namespace Halide #endif #ifndef HALIDE_TARGET_H @@ -5447,188 +5399,183 @@ int lcm(int, int); namespace llvm { class Module; class LLVMContext; -} +}// namespace llvm namespace Halide { /** A struct representing a target machine and os to generate code for. */ -struct Target { - /** The operating system used by the target. Determines which - * system calls to generate. */ - enum OS {OSUnknown = 0, Linux, Windows, OSX, Android, IOS, NaCl} os; - - /** The architecture used by the target. Determines the - * instruction set to use. For the PNaCl target, the "instruction - * set" is actually llvm bitcode. */ - enum Arch {ArchUnknown = 0, X86, ARM, PNaCl, MIPS} arch; - - /** The bit-width of the target machine. Must be 0 for unknown, or 32 or 64. */ - int bits; - - /** Optional features a target can have. */ - enum Feature { - JIT, ///< Generate code that will run immediately inside the calling process. - Debug, ///< Turn on debug info and output for runtime code. - NoAsserts, ///< Disable all runtime checks, for slightly tighter code. - NoBoundsQuery, ///< Disable the bounds querying functionality. - - SSE41, ///< Use SSE 4.1 and earlier instructions. Only relevant on x86. - AVX, ///< Use AVX 1 instructions. Only relevant on x86. - AVX2, ///< Use AVX 2 instructions. Only relevant on x86. - FMA, ///< Enable x86 FMA instruction - FMA4, ///< Enable x86 (AMD) FMA4 instruction set - F16C, ///< Enable x86 16-bit float support - - ARMv7s, ///< Generate code for ARMv7s. Only relevant for 32-bit ARM. - - CUDA, ///< Enable the CUDA runtime. Defaults to compute capability 2.0 (Fermi) - CUDACapability30, ///< Enable CUDA compute capability 3.0 (Kepler) - CUDACapability32, ///< Enable CUDA compute capability 3.2 (Tegra K1) - CUDACapability35, ///< Enable CUDA compute capability 3.5 (Kepler) - CUDACapability50, ///< Enable CUDA compute capability 5.0 (Maxwell) - - OpenCL, ///< Enable the OpenCL runtime. - CLDoubles, ///< Enable double support on OpenCL targets - - OpenGL, ///< Enable the OpenGL runtime. - - FeatureEnd - // NOTE: Changes to this enum must be reflected in the definition of - // to_string()! - }; - - Target() : os(OSUnknown), arch(ArchUnknown), bits(0) {} - Target(OS o, Arch a, int b, std::vector initial_features = std::vector()) - : os(o), arch(a), bits(b) { - for (size_t i = 0; i < initial_features.size(); i++) { - set_feature(initial_features[i]); - } - } - - void set_feature(Feature f, bool value = true) { - user_assert(f < FeatureEnd) << "Invalid Target feature.\n"; - features.set(f, value); - } - - void set_features(std::vector features_to_set, bool value = true) { - for (size_t i = 0; i < features_to_set.size(); i++) { - set_feature(features_to_set[i], value); - } - } +struct Target +{ + /** The operating system used by the target. Determines which + * system calls to generate. */ + enum OS { OSUnknown = 0, Linux, Windows, OSX, Android, IOS, NaCl } os; + + /** The architecture used by the target. Determines the + * instruction set to use. For the PNaCl target, the "instruction + * set" is actually llvm bitcode. */ + enum Arch { ArchUnknown = 0, X86, ARM, PNaCl, MIPS } arch; + + /** The bit-width of the target machine. Must be 0 for unknown, or 32 or 64. */ + int bits; + + /** Optional features a target can have. */ + enum Feature { + JIT,///< Generate code that will run immediately inside the calling process. + Debug,///< Turn on debug info and output for runtime code. + NoAsserts,///< Disable all runtime checks, for slightly tighter code. + NoBoundsQuery,///< Disable the bounds querying functionality. + + SSE41,///< Use SSE 4.1 and earlier instructions. Only relevant on x86. + AVX,///< Use AVX 1 instructions. Only relevant on x86. + AVX2,///< Use AVX 2 instructions. Only relevant on x86. + FMA,///< Enable x86 FMA instruction + FMA4,///< Enable x86 (AMD) FMA4 instruction set + F16C,///< Enable x86 16-bit float support + + ARMv7s,///< Generate code for ARMv7s. Only relevant for 32-bit ARM. + + CUDA,///< Enable the CUDA runtime. Defaults to compute capability 2.0 (Fermi) + CUDACapability30,///< Enable CUDA compute capability 3.0 (Kepler) + CUDACapability32,///< Enable CUDA compute capability 3.2 (Tegra K1) + CUDACapability35,///< Enable CUDA compute capability 3.5 (Kepler) + CUDACapability50,///< Enable CUDA compute capability 5.0 (Maxwell) + + OpenCL,///< Enable the OpenCL runtime. + CLDoubles,///< Enable double support on OpenCL targets + + OpenGL,///< Enable the OpenGL runtime. + + FeatureEnd + // NOTE: Changes to this enum must be reflected in the definition of + // to_string()! + }; + + Target() : os(OSUnknown), arch(ArchUnknown), bits(0) {} + Target(OS o, Arch a, int b, std::vector initial_features = std::vector()) : os(o), arch(a), bits(b) + { + for (size_t i = 0; i < initial_features.size(); i++) { set_feature(initial_features[i]); } + } - bool has_feature(Feature f) const { - user_assert(f < FeatureEnd) << "Invalid Target feature.\n"; - return features[f]; - } + void set_feature(Feature f, bool value = true) + { + user_assert(f < FeatureEnd) << "Invalid Target feature.\n"; + features.set(f, value); + } - bool features_any_of(std::vector test_features) const { - for (size_t i = 0; i < test_features.size(); i++) { - user_assert(test_features[i] < FeatureEnd) << "Invalid Target feature.\n"; + void set_features(std::vector features_to_set, bool value = true) + { + for (size_t i = 0; i < features_to_set.size(); i++) { set_feature(features_to_set[i], value); } + } - if (features[test_features[i]]) { - return true; - } - } - return false; - } + bool has_feature(Feature f) const + { + user_assert(f < FeatureEnd) << "Invalid Target feature.\n"; + return features[f]; + } - bool features_all_of(std::vector test_features) const { - for (size_t i = 0; i < test_features.size(); i++) { - user_assert(test_features[i] < FeatureEnd) << "Invalid Target feature.\n"; + bool features_any_of(std::vector test_features) const + { + for (size_t i = 0; i < test_features.size(); i++) { + user_assert(test_features[i] < FeatureEnd) << "Invalid Target feature.\n"; - if (!features[test_features[i]]) { - return false; - } - } - return true; + if (features[test_features[i]]) { return true; } } + return false; + } - /** Return a copy of the target with the given feature set. - * This is convenient when enabling certain features (e.g. NoBoundsQuery) - * in an initialization list, where the target to be mutated may be - * a const reference. */ - Target with_feature(Feature f) const { - Target copy = *this; - copy.set_feature(f); - return copy; - } + bool features_all_of(std::vector test_features) const + { + for (size_t i = 0; i < test_features.size(); i++) { + user_assert(test_features[i] < FeatureEnd) << "Invalid Target feature.\n"; - /** Return a copy of the target with the given feature cleared. - * This is convenient when disabling certain features (e.g. NoBoundsQuery) - * in an initialization list, where the target to be mutated may be - * a const reference. */ - Target without_feature(Feature f) const { - Target copy = *this; - copy.set_feature(f, false); - return copy; + if (!features[test_features[i]]) { return false; } } + return true; + } - /** Is OpenCL or CUDA enabled in this target? I.e. is - * Func::gpu_tile and similar going to work? We do not include - * OpenGL, because it is not capable of gpgpu, and is not - * scheduled via Func::gpu_tile. */ - bool has_gpu_feature() const { - return has_feature(CUDA) || has_feature(OpenCL); - } + /** Return a copy of the target with the given feature set. + * This is convenient when enabling certain features (e.g. NoBoundsQuery) + * in an initialization list, where the target to be mutated may be + * a const reference. */ + Target with_feature(Feature f) const + { + Target copy = *this; + copy.set_feature(f); + return copy; + } - bool operator==(const Target &other) const { - return os == other.os && - arch == other.arch && - bits == other.bits && - features == other.features; - } + /** Return a copy of the target with the given feature cleared. + * This is convenient when disabling certain features (e.g. NoBoundsQuery) + * in an initialization list, where the target to be mutated may be + * a const reference. */ + Target without_feature(Feature f) const + { + Target copy = *this; + copy.set_feature(f, false); + return copy; + } - bool operator!=(const Target &other) const { - return !(*this == other); - } + /** Is OpenCL or CUDA enabled in this target? I.e. is + * Func::gpu_tile and similar going to work? We do not include + * OpenGL, because it is not capable of gpgpu, and is not + * scheduled via Func::gpu_tile. */ + bool has_gpu_feature() const { return has_feature(CUDA) || has_feature(OpenCL); } - /** Convert the Target into a string form that can be reconstituted - * by merge_string(), which will always be of the form - * - * arch-bits-os-feature1-feature2...featureN. - * - * Note that is guaranteed that t2.from_string(t1.to_string()) == t1, - * but not that from_string(s).to_string() == s (since there can be - * multiple strings that parse to the same Target)... - * *unless* t1 contains 'unknown' fields (in which case you'll get a string - * that can't be parsed, which is intentional). - */ - EXPORT std::string to_string() const; - - /** - * Parse the contents of 'target' and merge into 'this', - * replacing only the parts that are specified. (e.g., if 'target' specifies - * only an arch, only the arch field of 'this' will be changed, leaving - * the other fields untouched). Any features specified in 'target' - * are added to 'this', whether or not originally present. - * - * If the string contains unknown tokens, or multiple tokens of the - * same category (e.g. multiple arch values), return false - * (possibly leaving 'this' munged). (Multiple feature specifications - * will not cause a failure.) - * - * If 'target' contains "host" as the first token, it replaces the entire - * contents of 'this' with get_host_target(), then proceeds to parse the - * remaining tokens (allowing for things like "host-opencl" to mean - * "host configuration, but with opencl added"). - * - * Note that unlike parse_from_string(), this will never print to cerr or - * assert in the event of a parse failure. Note also that an empty target - * string is essentially a no-op, leaving 'this' unaffected. - */ - EXPORT bool merge_string(const std::string &target); + bool operator==(const Target &other) const + { + return os == other.os && arch == other.arch && bits == other.bits && features == other.features; + } - /** - * Like merge_string(), but reset the contents of 'this' first. - */ - EXPORT bool from_string(const std::string &target) { - *this = Target(); - return merge_string(target); - } + bool operator!=(const Target &other) const { return !(*this == other); } + + /** Convert the Target into a string form that can be reconstituted + * by merge_string(), which will always be of the form + * + * arch-bits-os-feature1-feature2...featureN. + * + * Note that is guaranteed that t2.from_string(t1.to_string()) == t1, + * but not that from_string(s).to_string() == s (since there can be + * multiple strings that parse to the same Target)... + * *unless* t1 contains 'unknown' fields (in which case you'll get a string + * that can't be parsed, which is intentional). + */ + EXPORT std::string to_string() const; + + /** + * Parse the contents of 'target' and merge into 'this', + * replacing only the parts that are specified. (e.g., if 'target' specifies + * only an arch, only the arch field of 'this' will be changed, leaving + * the other fields untouched). Any features specified in 'target' + * are added to 'this', whether or not originally present. + * + * If the string contains unknown tokens, or multiple tokens of the + * same category (e.g. multiple arch values), return false + * (possibly leaving 'this' munged). (Multiple feature specifications + * will not cause a failure.) + * + * If 'target' contains "host" as the first token, it replaces the entire + * contents of 'this' with get_host_target(), then proceeds to parse the + * remaining tokens (allowing for things like "host-opencl" to mean + * "host configuration, but with opencl added"). + * + * Note that unlike parse_from_string(), this will never print to cerr or + * assert in the event of a parse failure. Note also that an empty target + * string is essentially a no-op, leaving 'this' unaffected. + */ + EXPORT bool merge_string(const std::string &target); + + /** + * Like merge_string(), but reset the contents of 'this' first. + */ + EXPORT bool from_string(const std::string &target) + { + *this = Target(); + return merge_string(target); + } private: - /** A bitmask that stores the active features. */ - std::bitset features; + /** A bitmask that stores the active features. */ + std::bitset features; }; /** Return the target corresponding to the host machine. */ @@ -5655,15 +5602,15 @@ EXPORT Target parse_target_string(const std::string &target); namespace Internal { -/** Create an llvm module containing the support code for a given target. */ -llvm::Module *get_initial_module_for_target(Target, llvm::LLVMContext *); + /** Create an llvm module containing the support code for a given target. */ + llvm::Module *get_initial_module_for_target(Target, llvm::LLVMContext *); -/** Create an llvm module containing the support code for ptx device. */ -llvm::Module *get_initial_module_for_ptx_device(Target, llvm::LLVMContext *c); + /** Create an llvm module containing the support code for ptx device. */ + llvm::Module *get_initial_module_for_ptx_device(Target, llvm::LLVMContext *c); -} +}// namespace Internal -} +}// namespace Halide #endif @@ -5671,14 +5618,15 @@ llvm::Module *get_initial_module_for_ptx_device(Target, llvm::LLVMContext *c); namespace Halide { namespace Internal { -/** A code generator abstract base class. Actual code generators - * (e.g. CodeGen_X86) inherit from this. This class is responsible - * for taking a Halide Stmt and producing llvm bitcode, machine - * code in an object file, or machine code accessible through a - * function pointer. - */ -class CodeGen : public IRVisitor { -public: + /** A code generator abstract base class. Actual code generators + * (e.g. CodeGen_X86) inherit from this. This class is responsible + * for taking a Halide Stmt and producing llvm bitcode, machine + * code in an object file, or machine code accessible through a + * function pointer. + */ + class CodeGen : public IRVisitor + { + public: mutable RefCount ref_count; CodeGen(Target t); @@ -5687,9 +5635,10 @@ class CodeGen : public IRVisitor { /** Take a halide statement and compiles it to an llvm module held * internally. Call this before calling compile_to_bitcode or * compile_to_native. */ - virtual void compile(Stmt stmt, std::string name, - const std::vector &args, - const std::vector &images_to_embed); + virtual void compile(Stmt stmt, + std::string name, + const std::vector &args, + const std::vector &images_to_embed); /** Emit a compiled halide statement as llvm bitcode. Call this * after calling compile. */ @@ -5723,8 +5672,8 @@ class CodeGen : public IRVisitor { * after it jits. Does nothing by default. The third argument * gives the target a chance to inject calls to target-specific * module cleanup routines. */ - virtual void jit_finalize(llvm::ExecutionEngine *, llvm::Module *, - std::vector *); + virtual void + jit_finalize(llvm::ExecutionEngine *, llvm::Module *, std::vector *); /** Initialize internal llvm state for the enabled targets. */ static void initialize_llvm(); @@ -5732,8 +5681,7 @@ class CodeGen : public IRVisitor { /** Which built-in functions require a user-context first argument? */ static bool function_takes_user_context(const std::string &name); -protected: - + protected: /** State needed by llvm for code generation, including the * current module, function, context, builder, and most recently * generated llvm value. */ @@ -5749,7 +5697,7 @@ class CodeGen : public IRVisitor { bool owns_module; llvm::Function *function; llvm::LLVMContext *context; - llvm::IRBuilder > *builder; + llvm::IRBuilder> *builder; llvm::Value *value; llvm::MDNode *very_likely_branch; //@} @@ -5777,8 +5725,7 @@ class CodeGen : public IRVisitor { /** Fetch an entry from the symbol table. If the symbol is not * found, it either errors out (if the second arg is true), or * returns NULL. */ - llvm::Value* sym_get(const std::string &name, - bool must_succeed = true) const; + llvm::Value *sym_get(const std::string &name, bool must_succeed = true) const; /** Test if an item exists in the symbol table. */ bool sym_exists(const std::string &name) const; @@ -5815,8 +5762,9 @@ class CodeGen : public IRVisitor { * vector of Expr arguments to print. */ // @{ void create_assertion(llvm::Value *condition, Expr message); - void create_assertion(llvm::Value *condition, const char *message) { - create_assertion(condition, StringImm::make(message)); + void create_assertion(llvm::Value *condition, const char *message) + { + create_assertion(condition, StringImm::make(message)); } // @} @@ -5945,10 +5893,9 @@ class CodeGen : public IRVisitor { /** Implementation of the intrinsic call to * interleave_vectors. This implementation allows for interleaving * an arbitrary number of vectors.*/ - llvm::Value *interleave_vectors(Type, const std::vector&); - -private: + llvm::Value *interleave_vectors(Type, const std::vector &); + private: /** All the values in scope at the current code location during * codegen. Use sym_push and sym_pop to access. */ Scope symbol_table; @@ -5959,9 +5906,10 @@ class CodeGen : public IRVisitor { /** String constants already emitted to the module. Tracked to * prevent emitting the same string many times. */ std::map string_constants; -}; + }; -}} +}// namespace Internal +}// namespace Halide #endif #ifndef HALIDE_CODEGEN_X86_H @@ -5982,16 +5930,15 @@ class CodeGen : public IRVisitor { namespace Halide { namespace Internal { -/** A code generator that emits posix code from a given Halide stmt. */ -class CodeGen_Posix : public CodeGen { -public: - + /** A code generator that emits posix code from a given Halide stmt. */ + class CodeGen_Posix : public CodeGen + { + public: /** Create an posix code generator. Processor features can be * enabled using the appropriate arguments */ CodeGen_Posix(Target t); -protected: - + protected: /** Some useful llvm types for subclasses */ // @{ llvm::Type *i8x8, *i8x16, *i8x32; @@ -6005,15 +5952,15 @@ class CodeGen_Posix : public CodeGen { /** Some wildcard variables used for peephole optimizations in * subclasses */ // @{ - Expr wild_i8x8, wild_i16x4, wild_i32x2; // 64-bit signed ints - Expr wild_u8x8, wild_u16x4, wild_u32x2; // 64-bit unsigned ints - Expr wild_i8x16, wild_i16x8, wild_i32x4, wild_i64x2; // 128-bit signed ints - Expr wild_u8x16, wild_u16x8, wild_u32x4, wild_u64x2; // 128-bit unsigned ints - Expr wild_i8x32, wild_i16x16, wild_i32x8, wild_i64x4; // 256-bit signed ints - Expr wild_u8x32, wild_u16x16, wild_u32x8, wild_u64x4; // 256-bit unsigned ints - Expr wild_f32x2; // 64-bit floats - Expr wild_f32x4, wild_f64x2; // 128-bit floats - Expr wild_f32x8, wild_f64x4; // 256-bit floats + Expr wild_i8x8, wild_i16x4, wild_i32x2;// 64-bit signed ints + Expr wild_u8x8, wild_u16x4, wild_u32x2;// 64-bit unsigned ints + Expr wild_i8x16, wild_i16x8, wild_i32x4, wild_i64x2;// 128-bit signed ints + Expr wild_u8x16, wild_u16x8, wild_u32x4, wild_u64x2;// 128-bit unsigned ints + Expr wild_i8x32, wild_i16x16, wild_i32x8, wild_i64x4;// 256-bit signed ints + Expr wild_u8x32, wild_u16x16, wild_u32x8, wild_u64x4;// 256-bit unsigned ints + Expr wild_f32x2;// 64-bit floats + Expr wild_f32x4, wild_f64x2;// 128-bit floats + Expr wild_f32x8, wild_f64x4;// 256-bit floats Expr min_i8, max_i8, max_u8; Expr min_i16, max_i16, max_u16; Expr min_i32, max_i32, max_u32; @@ -6032,16 +5979,17 @@ class CodeGen_Posix : public CodeGen { // @} /** A struct describing heap or stack allocations. */ - struct Allocation { - llvm::Value *ptr; + struct Allocation + { + llvm::Value *ptr; - /** How many bytes this allocation is, or 0 if not - * constant. */ - int constant_bytes; + /** How many bytes this allocation is, or 0 if not + * constant. */ + int constant_bytes; - /** How many bytes of stack space used. 0 implies it was a - * heap allocation. */ - int stack_bytes; + /** How many bytes of stack space used. 0 implies it was a + * heap allocation. */ + int stack_bytes; }; /** The allocations currently in scope. The stack gets pushed when @@ -6054,8 +6002,7 @@ class CodeGen_Posix : public CodeGen { /** Initialize the CodeGen internal state to compile a fresh module */ void init_module(); -private: - + private: /** Stack allocations that were freed, but haven't gone out of * scope yet. This allows us to re-use stack allocations when * they aren't being used. */ @@ -6080,18 +6027,17 @@ class CodeGen_Posix : public CodeGen { * * When the allocation can be freed call 'free_allocation', and * when it goes out of scope call 'destroy_allocation'. */ - Allocation create_allocation(const std::string &name, Type type, - const std::vector &extents, - Expr condition); + Allocation create_allocation(const std::string &name, Type type, const std::vector &extents, Expr condition); /** Free the memory backing an allocation and pop it from the * symbol table and the allocations map. For heap allocations it * calls halide_free in the runtime, for stack allocations it * marks the block as free so it can be reused. */ void free_allocation(const std::string &name); -}; + }; -}} +}// namespace Internal +}// namespace Halide #endif @@ -6102,9 +6048,10 @@ class JITEventListener; namespace Halide { namespace Internal { -/** A code generator that emits x86 code from a given Halide stmt. */ -class CodeGen_X86 : public CodeGen_Posix { -public: + /** A code generator that emits x86 code from a given Halide stmt. */ + class CodeGen_X86 : public CodeGen_Posix + { + public: /** Create an x86 code generator. Processor features can be * enabled using the appropriate flags in the target struct. */ CodeGen_X86(Target); @@ -6115,17 +6062,17 @@ class CodeGen_X86 : public CodeGen_Posix { * CodeGen::compile_to_file or * CodeGen::compile_to_function_pointer to get at the x86 machine * code. */ - void compile(Stmt stmt, std::string name, - const std::vector &args, - const std::vector &images_to_embed); + void compile(Stmt stmt, + std::string name, + const std::vector &args, + const std::vector &images_to_embed); static void test(); void jit_init(llvm::ExecutionEngine *, llvm::Module *); void jit_finalize(llvm::ExecutionEngine *, llvm::Module *, std::vector *); -protected: - + protected: llvm::Triple get_target_triple() const; /** Generate a call to an sse or avx intrinsic */ @@ -6150,11 +6097,12 @@ class CodeGen_X86 : public CodeGen_Posix { std::string mattrs() const; bool use_soft_float_abi() const; -private: - llvm::JITEventListener* jitEventListener; -}; + private: + llvm::JITEventListener *jitEventListener; + }; -}} +}// namespace Internal +}// namespace Halide #endif #ifndef HALIDE_CODEGEN_GPU_HOST_H @@ -6175,9 +6123,10 @@ class CodeGen_X86 : public CodeGen_Posix { namespace Halide { namespace Internal { -/** A code generator that emits ARM code from a given Halide stmt. */ -class CodeGen_ARM : public CodeGen_Posix { -public: + /** A code generator that emits ARM code from a given Halide stmt. */ + class CodeGen_ARM : public CodeGen_Posix + { + public: /** Create an ARM code generator for the given arm target. */ CodeGen_ARM(Target); @@ -6187,14 +6136,14 @@ class CodeGen_ARM : public CodeGen_Posix { * CodeGen::compile_to_file or * CodeGen::compile_to_function_pointer to get at the ARM machine * code. */ - void compile(Stmt stmt, std::string name, - const std::vector &args, - const std::vector &images_to_embed); + void compile(Stmt stmt, + std::string name, + const std::vector &args, + const std::vector &images_to_embed); static void test(); -protected: - + protected: llvm::Triple get_target_triple() const; /** Generate a call to a neon intrinsic */ @@ -6225,13 +6174,14 @@ class CodeGen_ARM : public CodeGen_Posix { // @} /** Various patterns to peephole match against */ - struct Pattern { - std::string intrin; - Expr pattern; - enum PatternType {Simple = 0, LeftShift, RightShift, NarrowArgs}; - PatternType type; - Pattern() {} - Pattern(std::string i, Expr p, PatternType t = Simple) : intrin(i), pattern(p), type(t) {} + struct Pattern + { + std::string intrin; + Expr pattern; + enum PatternType { Simple = 0, LeftShift, RightShift, NarrowArgs }; + PatternType type; + Pattern() {} + Pattern(std::string i, Expr p, PatternType t = Simple) : intrin(i), pattern(p), type(t) {} }; std::vector casts, left_shifts, averagings, negations; @@ -6239,9 +6189,10 @@ class CodeGen_ARM : public CodeGen_Posix { std::string mcpu() const; std::string mattrs() const; bool use_soft_float_abi() const; -}; + }; -}} +}// namespace Internal +}// namespace Halide #endif #ifndef HALIDE_CODEGEN_MIPS_H @@ -6255,9 +6206,10 @@ class CodeGen_ARM : public CodeGen_Posix { namespace Halide { namespace Internal { -/** A code generator that emits mips code from a given Halide stmt. */ -class CodeGen_MIPS : public CodeGen_Posix { -public: + /** A code generator that emits mips code from a given Halide stmt. */ + class CodeGen_MIPS : public CodeGen_Posix + { + public: /** Create a mips code generator. Processor features can be * enabled using the appropriate flags in the target struct. */ CodeGen_MIPS(Target); @@ -6268,14 +6220,14 @@ class CodeGen_MIPS : public CodeGen_Posix { * CodeGen::compile_to_file or * CodeGen::compile_to_function_pointer to get at the mips machine * code. */ - void compile(Stmt stmt, std::string name, - const std::vector &args, - const std::vector &images_to_embed); + void compile(Stmt stmt, + std::string name, + const std::vector &args, + const std::vector &images_to_embed); static void test(); -protected: - + protected: llvm::Triple get_target_triple() const; using CodeGen_Posix::visit; @@ -6283,9 +6235,10 @@ class CodeGen_MIPS : public CodeGen_Posix { std::string mcpu() const; std::string mattrs() const; bool use_soft_float_abi() const; -}; + }; -}} +}// namespace Internal +}// namespace Halide #endif #ifndef HALIDE_CODEGEN_PNACL_H @@ -6299,9 +6252,10 @@ class CodeGen_MIPS : public CodeGen_Posix { namespace Halide { namespace Internal { -/** A code generator that emits pnacl bitcode from a given Halide stmt. */ -class CodeGen_PNaCl : public CodeGen_Posix { -public: + /** A code generator that emits pnacl bitcode from a given Halide stmt. */ + class CodeGen_PNaCl : public CodeGen_Posix + { + public: /** Create a pnacl code generator. Processor features can be * enabled using the appropriate flags in the target struct. */ CodeGen_PNaCl(Target); @@ -6311,9 +6265,10 @@ class CodeGen_PNaCl : public CodeGen_Posix { * to the function produced. After calling this, call * CodeGen::compile_to_file or CodeGen::compile_to_bitcode to get * at the pnacl bitcode. */ - void compile(Stmt stmt, std::string name, - const std::vector &args, - const std::vector &images_to_embed); + void compile(Stmt stmt, + std::string name, + const std::vector &args, + const std::vector &images_to_embed); /** The PNaCl backend overrides compile_to_native to * compile_to_bitcode instead. It does *not* run the pnacl @@ -6322,35 +6277,35 @@ class CodeGen_PNaCl : public CodeGen_Posix { * everything as internal, including weak symbols that Halide * relies on being weak). The final linking stage (e.g. using * pnacl-clang++) handles the sandboxing. */ - void compile_to_native(const std::string &filename, bool /*assembly*/) { - // TODO: Emit .ll when assembly is true - compile_to_bitcode(filename); + void compile_to_native(const std::string &filename, bool /*assembly*/) + { + // TODO: Emit .ll when assembly is true + compile_to_bitcode(filename); } -protected: - + protected: using CodeGen_Posix::visit; std::string mcpu() const; std::string mattrs() const; bool use_soft_float_abi() const; -}; + }; -}} +}// namespace Internal +}// namespace Halide #endif namespace Halide { namespace Internal { -struct CodeGen_GPU_Dev; -struct GPU_Argument; - -/** A code generator that emits GPU code from a given Halide stmt. */ -template -class CodeGen_GPU_Host : public CodeGen_CPU { -public: + struct CodeGen_GPU_Dev; + struct GPU_Argument; + /** A code generator that emits GPU code from a given Halide stmt. */ + template class CodeGen_GPU_Host : public CodeGen_CPU + { + public: /** Create a GPU code generator. GPU target is selected via * CodeGen_GPU_Options. Processor features can be enabled using the * appropriate flags from Target */ @@ -6364,11 +6319,12 @@ class CodeGen_GPU_Host : public CodeGen_CPU { * CodeGen::compile_to_file or * CodeGen::compile_to_function_pointer to get at the generated machine * code. */ - void compile(Stmt stmt, std::string name, - const std::vector &args, - const std::vector &images_to_embed); + void compile(Stmt stmt, + std::string name, + const std::vector &args, + const std::vector &images_to_embed); -protected: + protected: /** Declare members of the base class that must exist to help the * compiler do name lookup. Annoying but necessary, because the * compiler doesn't know that CodeGen_CPU will in fact inherit @@ -6413,22 +6369,24 @@ class CodeGen_GPU_Host : public CodeGen_CPU { /** Reaches inside the module at sets it to use a single shared * cuda context */ - void jit_finalize(llvm::ExecutionEngine *ee, llvm::Module *mod, - std::vector *cleanup_routines); + void jit_finalize(llvm::ExecutionEngine *ee, + llvm::Module *mod, + std::vector *cleanup_routines); static bool lib_cuda_linked; - static CodeGen_GPU_Dev* make_dev(Target); + static CodeGen_GPU_Dev *make_dev(Target); llvm::Value *get_module_state(); -private: + private: /** Child code generator for device kernels. */ CodeGen_GPU_Dev *cgdev; -}; + }; -}} +}// namespace Internal +}// namespace Halide #endif #ifndef HALIDE_CODEGEN_PTX_DEV_H @@ -6449,29 +6407,28 @@ class CodeGen_GPU_Host : public CodeGen_CPU { namespace Halide { namespace Internal { -/** A bit more information attached to an argument useful for GPU backends. */ -struct GPU_Argument : public Argument { + /** A bit more information attached to an argument useful for GPU backends. */ + struct GPU_Argument : public Argument + { /** The static size of the argument if known, or zero otherwise. */ size_t size; GPU_Argument() : size(0) {} - GPU_Argument(const std::string &_name, bool _is_buffer, Type _type) : - Argument(_name, _is_buffer, _type), size(0) {} - GPU_Argument(const std::string &_name, bool _is_buffer, Type _type, - size_t _size) : - Argument(_name, _is_buffer, _type), size(_size) {} -}; - -/** A code generator that emits GPU code from a given Halide stmt. */ -struct CodeGen_GPU_Dev { + GPU_Argument(const std::string &_name, bool _is_buffer, Type _type) : Argument(_name, _is_buffer, _type), size(0) {} + GPU_Argument(const std::string &_name, bool _is_buffer, Type _type, size_t _size) + : Argument(_name, _is_buffer, _type), size(_size) + {} + }; + + /** A code generator that emits GPU code from a given Halide stmt. */ + struct CodeGen_GPU_Dev + { virtual ~CodeGen_GPU_Dev(); /** Compile a GPU kernel into the module. This may be called many times * with different kernels, which will all be accumulated into a single * source module shared by a given Halide pipeline. */ - virtual void add_kernel(Stmt stmt, - const std::string &name, - const std::vector &args) = 0; + virtual void add_kernel(Stmt stmt, const std::string &name, const std::vector &args) = 0; /** (Re)initialize the GPU kernel module. This is separate from compile, * since a GPU device module will often have many kernels compiled into it @@ -6495,9 +6452,10 @@ struct CodeGen_GPU_Dev { * GPUs (APIs) support a constant memory storage class that cannot be * written to and performs well for block uniform accesses. */ static bool is_buffer_constant(Stmt kernel, const std::string &buffer); -}; + }; -}} +}// namespace Internal +}// namespace Halide #endif @@ -6508,18 +6466,19 @@ class BasicBlock; namespace Halide { namespace Internal { -/** A code generator that emits GPU code from a given Halide stmt. */ -class CodeGen_PTX_Dev : public CodeGen, public CodeGen_GPU_Dev { -public: + /** A code generator that emits GPU code from a given Halide stmt. */ + class CodeGen_PTX_Dev + : public CodeGen + , public CodeGen_GPU_Dev + { + public: friend class CodeGen_GPU_Host; friend class CodeGen_GPU_Host; /** Create a PTX device code generator. */ CodeGen_PTX_Dev(Target host); - void add_kernel(Stmt stmt, - const std::string &name, - const std::vector &args); + void add_kernel(Stmt stmt, const std::string &name, const std::vector &args); /** (Re)initialize the PTX module. This is separate from compile, since * a PTX device module will often have many kernels compiled into it for @@ -6533,7 +6492,7 @@ class CodeGen_PTX_Dev : public CodeGen, public CodeGen_GPU_Dev { void dump(); -protected: + protected: using CodeGen::visit; /** We hold onto the basic block at the start of the device @@ -6555,9 +6514,10 @@ class CodeGen_PTX_Dev : public CodeGen, public CodeGen_GPU_Dev { /** Map from simt variable names (e.g. foo.__block_id_x) to the llvm * ptx intrinsic functions to call to get them. */ std::string simt_intrinsic(const std::string &name); -}; + }; -}} +}// namespace Internal +}// namespace Halide #endif #ifndef HALIDE_CODEGEN_OPENCL_DEV_H @@ -6573,16 +6533,15 @@ class CodeGen_PTX_Dev : public CodeGen, public CodeGen_GPU_Dev { namespace Halide { namespace Internal { -class CodeGen_OpenCL_Dev : public CodeGen_GPU_Dev { -public: + class CodeGen_OpenCL_Dev : public CodeGen_GPU_Dev + { + public: CodeGen_OpenCL_Dev(Target target); /** Compile a GPU kernel into the module. This may be called many times * with different kernels, which will all be accumulated into a single * source module shared by a given Halide pipeline. */ - void add_kernel(Stmt stmt, - const std::string &name, - const std::vector &args); + void add_kernel(Stmt stmt, const std::string &name, const std::vector &args); /** (Re)initialize the GPU kernel module. This is separate from compile, * since a GPU device module will often have many kernels compiled into it @@ -6595,41 +6554,40 @@ class CodeGen_OpenCL_Dev : public CodeGen_GPU_Dev { void dump(); -protected: - - class CodeGen_OpenCL_C : public CodeGen_C { + protected: + class CodeGen_OpenCL_C : public CodeGen_C + { public: - CodeGen_OpenCL_C(std::ostream &s) : CodeGen_C(s) {} - void add_kernel(Stmt stmt, - const std::string &name, - const std::vector &args); + CodeGen_OpenCL_C(std::ostream &s) : CodeGen_C(s) {} + void add_kernel(Stmt stmt, const std::string &name, const std::vector &args); protected: - using CodeGen_C::visit; - std::string print_type(Type type); - std::string print_reinterpret(Type type, Expr e); - - std::string get_memory_space(const std::string &); - - void visit(const Div *); - void visit(const Mod *); - void visit(const For *); - void visit(const Ramp *op); - void visit(const Broadcast *op); - void visit(const Load *op); - void visit(const Store *op); - void visit(const Cast *op); - void visit(const Allocate *op); - void visit(const Free *op); + using CodeGen_C::visit; + std::string print_type(Type type); + std::string print_reinterpret(Type type, Expr e); + + std::string get_memory_space(const std::string &); + + void visit(const Div *); + void visit(const Mod *); + void visit(const For *); + void visit(const Ramp *op); + void visit(const Broadcast *op); + void visit(const Load *op); + void visit(const Store *op); + void visit(const Cast *op); + void visit(const Allocate *op); + void visit(const Free *op); }; std::ostringstream src_stream; std::string cur_kernel_name; CodeGen_OpenCL_C clc; Target target; -}; + }; -}} +}// namespace Internal +}// namespace Halide #endif #ifndef DEINTERLEAVE_H @@ -6646,24 +6604,24 @@ class CodeGen_OpenCL_Dev : public CodeGen_GPU_Dev { namespace Halide { namespace Internal { -/** Extract the odd-numbered lanes in a vector */ -EXPORT Expr extract_odd_lanes(Expr a); + /** Extract the odd-numbered lanes in a vector */ + EXPORT Expr extract_odd_lanes(Expr a); -/** Extract the even-numbered lanes in a vector */ -EXPORT Expr extract_even_lanes(Expr a); + /** Extract the even-numbered lanes in a vector */ + EXPORT Expr extract_even_lanes(Expr a); -/** Extract the nth lane of a vector */ -EXPORT Expr extract_lane(Expr vec, int lane); + /** Extract the nth lane of a vector */ + EXPORT Expr extract_lane(Expr vec, int lane); -/** Look through a statement for expressions of the form select(ramp % - * 2 == 0, a, b) and replace them with calls to an interleave - * intrinsic */ -Stmt rewrite_interleavings(Stmt s); + /** Look through a statement for expressions of the form select(ramp % + * 2 == 0, a, b) and replace them with calls to an interleave + * intrinsic */ + Stmt rewrite_interleavings(Stmt s); -EXPORT void deinterleave_vector_test(); + EXPORT void deinterleave_vector_test(); -} -} +}// namespace Internal +}// namespace Halide #endif #ifndef HALIDE_DERIVATIVE_H @@ -6678,37 +6636,37 @@ EXPORT void deinterleave_vector_test(); namespace Halide { namespace Internal { -/** Compute the analytic derivative of the expression with respect to - * the variable. May returned an undefined Expr if it's - * non-differentiable. */ -//Expr derivative(Expr expr, const string &var); - -/** - * Compute the finite difference version of the derivative: - * expr(var+1) - expr(var). The reason to do this as a derivative, - * instead of just explicitly constructing expr(var+1) - - * expr(var), is so that we don't have to do so much - * simplification later. For example, the finite-difference - * derivative of 2*x is trivially 2, whereas 2*(x+1) - 2*x may or - * may not simplify down to 2, depending on the quality of our - * simplification routine. - * - * Most rules for the finite difference and the true derivative - * are the same. The quotient and product rules are not. - * - */ -Expr finite_difference(Expr expr, const std::string &var); - -/** - * Detect whether an expression is monotonic increasing in a variable, - * decreasing, or unknown. Returns -1, 0, or 1 for decreasing, - * unknown, and increasing. - */ -enum MonotonicResult {Constant, MonotonicIncreasing, MonotonicDecreasing, Unknown}; -MonotonicResult is_monotonic(Expr e, const std::string &var); - -} -} + /** Compute the analytic derivative of the expression with respect to + * the variable. May returned an undefined Expr if it's + * non-differentiable. */ + // Expr derivative(Expr expr, const string &var); + + /** + * Compute the finite difference version of the derivative: + * expr(var+1) - expr(var). The reason to do this as a derivative, + * instead of just explicitly constructing expr(var+1) - + * expr(var), is so that we don't have to do so much + * simplification later. For example, the finite-difference + * derivative of 2*x is trivially 2, whereas 2*(x+1) - 2*x may or + * may not simplify down to 2, depending on the quality of our + * simplification routine. + * + * Most rules for the finite difference and the true derivative + * are the same. The quotient and product rules are not. + * + */ + Expr finite_difference(Expr expr, const std::string &var); + + /** + * Detect whether an expression is monotonic increasing in a variable, + * decreasing, or unknown. Returns -1, 0, or 1 for decreasing, + * unknown, and increasing. + */ + enum MonotonicResult { Constant, MonotonicIncreasing, MonotonicDecreasing, Unknown }; + MonotonicResult is_monotonic(Expr e, const std::string &var); + +}// namespace Internal +}// namespace Halide #endif #ifndef HALIDE_ONE_TO_ONE_H @@ -6724,16 +6682,16 @@ MonotonicResult is_monotonic(Expr e, const std::string &var); namespace Halide { namespace Internal { -/** Conservatively determine whether an integer expression is - * one-to-one in its variables. For now this means it contains a - * single variable and its derivative is provably strictly positive or - * strictly negative. */ -bool is_one_to_one(Expr expr); + /** Conservatively determine whether an integer expression is + * one-to-one in its variables. For now this means it contains a + * single variable and its derivative is provably strictly positive or + * strictly negative. */ + bool is_one_to_one(Expr expr); -void is_one_to_one_test(); + void is_one_to_one_test(); -} -} +}// namespace Internal +}// namespace Halide #endif #ifndef HALIDE_EXTERN_H @@ -6748,47 +6706,57 @@ void is_one_to_one_test(); * example usage. */ -#define _halide_check_arg_type(t, name, e, n) \ - _halide_user_assert(e.type() == t) << "Type mismatch for argument " << n << " to extern function " << #name << ". Type expected is " << t << " but the argument " << e << " has type " << e.type() << ".\n"; +#define _halide_check_arg_type(t, name, e, n) \ + _halide_user_assert(e.type() == t) << "Type mismatch for argument " << n << " to extern function " << #name \ + << ". Type expected is " << t << " but the argument " << e << " has type " \ + << e.type() << ".\n"; -#define HalideExtern_1(rt, name, t1) \ - Halide::Expr name(Halide::Expr a1) { \ - _halide_check_arg_type(Halide::type_of(), name, a1, 1); \ - return Halide::Internal::Call::make(Halide::type_of(), #name, vec(a1), Halide::Internal::Call::Extern); \ - } +#define HalideExtern_1(rt, name, t1) \ + Halide::Expr name(Halide::Expr a1) \ + { \ + _halide_check_arg_type(Halide::type_of(), name, a1, 1); \ + return Halide::Internal::Call::make(Halide::type_of(), #name, vec(a1), Halide::Internal::Call::Extern); \ + } -#define HalideExtern_2(rt, name, t1, t2) \ - Halide::Expr name(Halide::Expr a1, Halide::Expr a2) { \ - _halide_check_arg_type(Halide::type_of(), name, a1, 1); \ - _halide_check_arg_type(Halide::type_of(), name, a2, 2); \ - return Halide::Internal::Call::make(Halide::type_of(), #name, vec(a1, a2), Halide::Internal::Call::Extern); \ - } +#define HalideExtern_2(rt, name, t1, t2) \ + Halide::Expr name(Halide::Expr a1, Halide::Expr a2) \ + { \ + _halide_check_arg_type(Halide::type_of(), name, a1, 1); \ + _halide_check_arg_type(Halide::type_of(), name, a2, 2); \ + return Halide::Internal::Call::make(Halide::type_of(), #name, vec(a1, a2), Halide::Internal::Call::Extern); \ + } -#define HalideExtern_3(rt, name, t1, t2, t3) \ - Halide::Expr name(Halide::Expr a1, Halide::Expr a2, Halide::Expr a3) { \ - _halide_check_arg_type(Halide::type_of(), name, a1, 1); \ - _halide_check_arg_type(Halide::type_of(), name, a2, 2); \ - _halide_check_arg_type(Halide::type_of(), name, a3, 3); \ - return Halide::Internal::Call::make(Halide::type_of(), #name, vec(a1, a2, a3), Halide::Internal::Call::Extern); \ - } +#define HalideExtern_3(rt, name, t1, t2, t3) \ + Halide::Expr name(Halide::Expr a1, Halide::Expr a2, Halide::Expr a3) \ + { \ + _halide_check_arg_type(Halide::type_of(), name, a1, 1); \ + _halide_check_arg_type(Halide::type_of(), name, a2, 2); \ + _halide_check_arg_type(Halide::type_of(), name, a3, 3); \ + return Halide::Internal::Call::make( \ + Halide::type_of(), #name, vec(a1, a2, a3), Halide::Internal::Call::Extern); \ + } -#define HalideExtern_4(rt, name, t1, t2, t3, t4) \ - Halide::Expr name(Halide::Expr a1, Halide::Expr a2, Halide::Expr a3, Halide::Expr a4) { \ - _halide_check_arg_type(Halide::type_of(), name, a1, 1); \ - _halide_check_arg_type(Halide::type_of(), name, a2, 2); \ - _halide_check_arg_type(Halide::type_of(), name, a3, 3); \ - _halide_check_arg_type(Halide::type_of(), name, a4, 4); \ - return Halide::Internal::Call::make(Halide::type_of(), #name, vec(a1, a2, a3, a4), Halide::Internal::Call::Extern); \ +#define HalideExtern_4(rt, name, t1, t2, t3, t4) \ + Halide::Expr name(Halide::Expr a1, Halide::Expr a2, Halide::Expr a3, Halide::Expr a4) \ + { \ + _halide_check_arg_type(Halide::type_of(), name, a1, 1); \ + _halide_check_arg_type(Halide::type_of(), name, a2, 2); \ + _halide_check_arg_type(Halide::type_of(), name, a3, 3); \ + _halide_check_arg_type(Halide::type_of(), name, a4, 4); \ + return Halide::Internal::Call::make( \ + Halide::type_of(), #name, vec(a1, a2, a3, a4), Halide::Internal::Call::Extern); \ } -#define HalideExtern_5(rt, name, t1, t2, t3, t4, t5) \ - Halide::Expr name(Halide::Expr a1, Halide::Expr a2, Halide::Expr a3, Halide::Expr a4, Halide::Expr a5) { \ - _halide_check_arg_type(Halide::type_of(), name, a1, 1); \ - _halide_check_arg_type(Halide::type_of(), name, a2, 2); \ - _halide_check_arg_type(Halide::type_of(), name, a3, 3); \ - _halide_check_arg_type(Halide::type_of(), name, a4, 4); \ - _halide_check_arg_type(Halide::type_of(), name, a5, 5); \ - return Halide::Internal::Call::make(Halide::type_of(), #name, vec(a1, a2, a3, a4, a5), Halide::Internal::Call::Extern); \ +#define HalideExtern_5(rt, name, t1, t2, t3, t4, t5) \ + Halide::Expr name(Halide::Expr a1, Halide::Expr a2, Halide::Expr a3, Halide::Expr a4, Halide::Expr a5) \ + { \ + _halide_check_arg_type(Halide::type_of(), name, a1, 1); \ + _halide_check_arg_type(Halide::type_of(), name, a2, 2); \ + _halide_check_arg_type(Halide::type_of(), name, a3, 3); \ + _halide_check_arg_type(Halide::type_of(), name, a4, 4); \ + _halide_check_arg_type(Halide::type_of(), name, a5, 5); \ + return Halide::Internal::Call::make( \ + Halide::type_of(), #name, vec(a1, a2, a3, a4, a5), Halide::Internal::Call::Extern); \ } #endif @@ -6807,8 +6775,8 @@ void is_one_to_one_test(); * Defines the Var - the front-end variable */ -#include #include +#include namespace Halide { @@ -6817,167 +6785,152 @@ namespace Halide { * occur. It can be used in the left-hand-side of a function * definition, or as an Expr. As an Expr, it always has type * Int(32). */ -class Var { - std::string _name; +class Var +{ + std::string _name; + public: - /** Construct a Var with the given name */ - Var(const std::string &n) : _name(n) { - // Make sure we don't get a unique name with the same name as - // this later: - Internal::unique_name(n, false); - } + /** Construct a Var with the given name */ + Var(const std::string &n) : _name(n) + { + // Make sure we don't get a unique name with the same name as + // this later: + Internal::unique_name(n, false); + } - /** Construct a Var with an automatically-generated unique name. */ - Var() : _name(Internal::make_entity_name(this, "Halide::Var", 'v')) {} + /** Construct a Var with an automatically-generated unique name. */ + Var() : _name(Internal::make_entity_name(this, "Halide::Var", 'v')) {} + + /** Get the name of a Var */ + const std::string &name() const { return _name; } + + /** Test if two Vars are the same. This simply compares the names. */ + bool same_as(const Var &other) const { return _name == other._name; } + + /** Implicit var constructor. Implicit variables are injected + * automatically into a function call if the number of arguments + * to the function are fewer than its dimensionality and a + * placeholder ("_") appears in its argument list. Defining a + * function to equal an expression containing implicit variables + * similarly appends those implicit variables, in the same order, + * to the left-hand-side of the definition where the placeholder + * ('_') appears. + * + * For example, consider the definition: + * + \code + Func f, g; + Var x, y; + f(x, y) = 3; + \endcode + * + * A call to f with the placeholder symbol \ref _ + * will have implicit arguments injected automatically, so f(2, \ref _) + * is equivalent to f(2, \ref _0), where \ref _0 = Var::implicit(0), and f(\ref _) + * (and indeed f when cast to an Expr) is equivalent to f(\ref _0, \ref _1). + * The following definitions are all equivalent, differing only in the + * variable names. + * + \code + g(_) = f*3; + g(_) = f(_)*3; + g(x, _) = f(x, _)*3; + g(x, y) = f(x, y)*3; + \endcode + * + * These are expanded internally as follows: + * + \code + g(_0, _1) = f(_0, _1)*3; + g(_0, _1) = f(_0, _1)*3; + g(x, _0) = f(x, _0)*3; + g(x, y) = f(x, y)*3; + \endcode + * + * The following, however, defines g as four dimensional: + \code + g(x, y, _) = f*3; + \endcode + * + * It is equivalent to: + * + \code + g(x, y, _0, _1) = f(_0, _1)*3; + \endcode + * + * Expressions requiring differing numbers of implicit variables + * can be combined. The left-hand-side of a definition injects + * enough implicit variables to cover all of them: + * + \code + Func h; + h(x) = x*3; + g(x) = h + (f + f(x)) * f(x, y); + \endcode + * + * expands to: + * + \code + Func h; + h(x) = x*3; + g(x, _0, _1) = h(_0) + (f(_0, _1) + f(x, _0)) * f(x, y); + \endcode + * + * The first ten implicits, _0 through _9, are predeclared in this + * header and can be used for scheduling. They should never be + * used as arguments in a declaration or used in a call. + * + * While it is possible to use Var::implicit or the predeclared + * implicits to create expressions that can be treated as small + * anonymous functions (e.g. Func(_0 + _1)) this is considered + * poor style. Instead use \ref lambda. + */ + static Var implicit(int n) + { + std::ostringstream str; + str << "_" << n; + return Var(str.str()); + } - /** Get the name of a Var */ - const std::string &name() const {return _name;} + /** Return whether a variable name is of the form for an implicit argument. + * TODO: This is almost guaranteed to incorrectly fire on user + * declared variables at some point. We should likely prevent + * user Var declarations from making names of this form. + */ + //{ + static bool is_implicit(const std::string &name) + { + return Internal::starts_with(name, "_") && name.find_first_not_of("0123456789", 1) == std::string::npos; + } + bool is_implicit() const { return is_implicit(name()); } + //} - /** Test if two Vars are the same. This simply compares the names. */ - bool same_as(const Var &other) const {return _name == other._name;} + /** Return the argument index for a placeholder argument given its + * name. Returns 0 for \ref _0, 1 for \ref _1, etc. Returns -1 if + * the variable is not of implicit form. + */ + //{ + static int implicit_index(const std::string &name) { return is_implicit(name) ? atoi(name.c_str() + 1) : -1; } + int implicit_index() const { return implicit_index(name()); } + //} - /** Implicit var constructor. Implicit variables are injected - * automatically into a function call if the number of arguments - * to the function are fewer than its dimensionality and a - * placeholder ("_") appears in its argument list. Defining a - * function to equal an expression containing implicit variables - * similarly appends those implicit variables, in the same order, - * to the left-hand-side of the definition where the placeholder - * ('_') appears. - * - * For example, consider the definition: - * - \code - Func f, g; - Var x, y; - f(x, y) = 3; - \endcode - * - * A call to f with the placeholder symbol \ref _ - * will have implicit arguments injected automatically, so f(2, \ref _) - * is equivalent to f(2, \ref _0), where \ref _0 = Var::implicit(0), and f(\ref _) - * (and indeed f when cast to an Expr) is equivalent to f(\ref _0, \ref _1). - * The following definitions are all equivalent, differing only in the - * variable names. - * - \code - g(_) = f*3; - g(_) = f(_)*3; - g(x, _) = f(x, _)*3; - g(x, y) = f(x, y)*3; - \endcode - * - * These are expanded internally as follows: - * - \code - g(_0, _1) = f(_0, _1)*3; - g(_0, _1) = f(_0, _1)*3; - g(x, _0) = f(x, _0)*3; - g(x, y) = f(x, y)*3; - \endcode - * - * The following, however, defines g as four dimensional: - \code - g(x, y, _) = f*3; - \endcode - * - * It is equivalent to: - * - \code - g(x, y, _0, _1) = f(_0, _1)*3; - \endcode - * - * Expressions requiring differing numbers of implicit variables - * can be combined. The left-hand-side of a definition injects - * enough implicit variables to cover all of them: - * - \code - Func h; - h(x) = x*3; - g(x) = h + (f + f(x)) * f(x, y); - \endcode - * - * expands to: - * - \code - Func h; - h(x) = x*3; - g(x, _0, _1) = h(_0) + (f(_0, _1) + f(x, _0)) * f(x, y); - \endcode - * - * The first ten implicits, _0 through _9, are predeclared in this - * header and can be used for scheduling. They should never be - * used as arguments in a declaration or used in a call. - * - * While it is possible to use Var::implicit or the predeclared - * implicits to create expressions that can be treated as small - * anonymous functions (e.g. Func(_0 + _1)) this is considered - * poor style. Instead use \ref lambda. - */ - static Var implicit(int n) { - std::ostringstream str; - str << "_" << n; - return Var(str.str()); - } + /** Test if a var is the placeholder variable \ref _ */ + //{ + static bool is_placeholder(const std::string &name) { return name == "_"; } + bool is_placeholder() const { return is_placeholder(name()); } + //} - /** Return whether a variable name is of the form for an implicit argument. - * TODO: This is almost guaranteed to incorrectly fire on user - * declared variables at some point. We should likely prevent - * user Var declarations from making names of this form. - */ - //{ - static bool is_implicit(const std::string &name) { - return Internal::starts_with(name, "_") && - name.find_first_not_of("0123456789", 1) == std::string::npos; - } - bool is_implicit() const { - return is_implicit(name()); - } - //} - - /** Return the argument index for a placeholder argument given its - * name. Returns 0 for \ref _0, 1 for \ref _1, etc. Returns -1 if - * the variable is not of implicit form. - */ - //{ - static int implicit_index(const std::string &name) { - return is_implicit(name) ? atoi(name.c_str() + 1) : -1; - } - int implicit_index() const { - return implicit_index(name()); - } - //} - - /** Test if a var is the placeholder variable \ref _ */ - //{ - static bool is_placeholder(const std::string &name) { - return name == "_"; - } - bool is_placeholder() const { - return is_placeholder(name()); - } - //} - - /** A Var can be treated as an Expr of type Int(32) */ - operator Expr() const { - return Internal::Variable::make(Int(32), name()); - } - - /** Vars to use for scheduling producer/consumer pairs on the gpu. */ - // @{ - static Var gpu_blocks() { - return Var("__block_id_x"); - } - static Var gpu_threads() { - return Var("__thread_id_x"); - } - // @} + /** A Var can be treated as an Expr of type Int(32) */ + operator Expr() const { return Internal::Variable::make(Int(32), name()); } - /** A Var that represents the location outside the outermost loop. */ - static Var outermost() { - return Var("__outermost"); - } + /** Vars to use for scheduling producer/consumer pairs on the gpu. */ + // @{ + static Var gpu_blocks() { return Var("__block_id_x"); } + static Var gpu_threads() { return Var("__thread_id_x"); } + // @} + /** A Var that represents the location outside the outermost loop. */ + static Var outermost() { return Var("__outermost"); } }; /** A placeholder variable for infered arguments. See \ref Var::implicit */ @@ -6988,7 +6941,7 @@ EXPORT extern Var _; EXPORT extern Var _0, _1, _2, _3, _4, _5, _6, _7, _8, _9; // @} -} +}// namespace Halide #endif #ifndef HALIDE_PARAM_H @@ -7009,94 +6962,73 @@ namespace Halide { * should be bound to an actual value of type T using the set method * before you realize the function uses this. If you're statically * compiling, this param should appear in the argument list. */ -template -class Param { - /** A reference-counted handle on the internal parameter object */ - Internal::Parameter param; +template class Param +{ + /** A reference-counted handle on the internal parameter object */ + Internal::Parameter param; public: - /** Construct a scalar parameter of type T with a unique - * auto-generated name */ - Param() : param(type_of(), false, Internal::make_entity_name(this, "Halide::Param(), false, Internal::make_entity_name(this, "Halide::Param(), false, n) {} + /** Construct a scalar parameter of type T with the given name */ + Param(const std::string &n) : param(type_of(), false, n) {} - /** Get the name of this parameter */ - const std::string &name() const { - return param.name(); - } + /** Get the name of this parameter */ + const std::string &name() const { return param.name(); } - /** Get the current value of this parameter. Only meaningful when jitting. */ - NO_INLINE T get() const { - return param.get_scalar(); - } + /** Get the current value of this parameter. Only meaningful when jitting. */ + NO_INLINE T get() const { return param.get_scalar(); } - /** Set the current value of this parameter. Only meaningful when jitting */ - NO_INLINE void set(T val) { - param.set_scalar(val); - } + /** Set the current value of this parameter. Only meaningful when jitting */ + NO_INLINE void set(T val) { param.set_scalar(val); } - /** Get a pointer to the location that stores the current value of - * this parameter. Only meaningful for jitting. */ - NO_INLINE T *get_address() const { - return (T *)(param.get_scalar_address()); - } + /** Get a pointer to the location that stores the current value of + * this parameter. Only meaningful for jitting. */ + NO_INLINE T *get_address() const { return (T *)(param.get_scalar_address()); } - /** Get the halide type of T */ - Type type() const { - return type_of(); - } + /** Get the halide type of T */ + Type type() const { return type_of(); } - /** Get or set the possible range of this parameter. Use undefined - * Exprs to mean unbounded. */ - // @{ - void set_range(Expr min, Expr max) { - set_min_value(min); - set_max_value(max); - } + /** Get or set the possible range of this parameter. Use undefined + * Exprs to mean unbounded. */ + // @{ + void set_range(Expr min, Expr max) + { + set_min_value(min); + set_max_value(max); + } - void set_min_value(Expr min) { - if (min.type() != type_of()) { - min = Internal::Cast::make(type_of(), min); - } - param.set_min_value(min); - } + void set_min_value(Expr min) + { + if (min.type() != type_of()) { min = Internal::Cast::make(type_of(), min); } + param.set_min_value(min); + } - void set_max_value(Expr max) { - if (max.type() != type_of()) { - max = Internal::Cast::make(type_of(), max); - } - param.set_max_value(max); - } + void set_max_value(Expr max) + { + if (max.type() != type_of()) { max = Internal::Cast::make(type_of(), max); } + param.set_max_value(max); + } - Expr get_min_value() { - return param.get_min_value(); - } + Expr get_min_value() { return param.get_min_value(); } - Expr get_max_value() { - return param.get_max_value(); - } - // @} + Expr get_max_value() { return param.get_max_value(); } + // @} - /** You can use this parameter as an expression in a halide - * function definition */ - operator Expr() const { - return Internal::Variable::make(type_of(), name(), param); - } + /** You can use this parameter as an expression in a halide + * function definition */ + operator Expr() const { return Internal::Variable::make(type_of(), name(), param); } - /** Using a param as the argument to an external stage treats it - * as an Expr */ - operator ExternFuncArgument() const { - return Expr(*this); - } + /** Using a param as the argument to an external stage treats it + * as an Expr */ + operator ExternFuncArgument() const { return Expr(*this); } - /** Construct the appropriate argument matching this parameter, - * for the purpose of generating the right type signature when - * statically compiling halide pipelines. */ - operator Argument() const { - return Argument(name(), false, type()); - } + /** Construct the appropriate argument matching this parameter, + * for the purpose of generating the right type signature when + * statically compiling halide pipelines. */ + operator Argument() const { return Argument(name(), false, type()); } }; /** Returns a Param corresponding to a pointer to a user context @@ -7104,191 +7036,187 @@ class Param { * calls a function from the Halide runtime (e.g. halide_printf()), it * passes the value of this pointer as the first argument to the * runtime function. */ -inline Param user_context_param() { - return Param("__user_context"); -} +inline Param user_context_param() { return Param("__user_context"); } /** A handle on the output buffer of a pipeline. Used to make static * promises about the output size and stride. */ -class OutputImageParam { +class OutputImageParam +{ protected: - /** A reference-counted handle on the internal parameter object */ - Internal::Parameter param; - - /** The dimensionality of this image. */ - int dims; - - void add_implicit_args_if_placeholder(std::vector &args, - Expr last_arg, - int total_args, - bool *placeholder_seen) const; -public: - - /** Construct a NULL image parameter handle. */ - OutputImageParam() : dims(0) {} - - /** Construct an OutputImageParam that wraps an Internal Parameter object. */ - EXPORT OutputImageParam(const Internal::Parameter &p, int d); - - /** Get the name of this Param */ - EXPORT const std::string &name() const; - - /** Get the type of the image data this Param refers to */ - EXPORT Type type() const; - - /** Is this parameter handle non-NULL */ - EXPORT bool defined(); - - /** Get an expression representing the minimum coordinates of this image - * parameter in the given dimension. */ - EXPORT Expr min(int x) const; + /** A reference-counted handle on the internal parameter object */ + Internal::Parameter param; - /** Get an expression representing the extent of this image - * parameter in the given dimension */ - EXPORT Expr extent(int x) const; + /** The dimensionality of this image. */ + int dims; - /** Get an expression representing the stride of this image in the - * given dimension */ - EXPORT Expr stride(int x) const; + void add_implicit_args_if_placeholder(std::vector &args, + Expr last_arg, + int total_args, + bool *placeholder_seen) const; - /** Set the extent in a given dimension to equal the given - * expression. Images passed in that fail this check will generate - * a runtime error. Returns a reference to the ImageParam so that - * these calls may be chained. - * - * This may help the compiler generate better - * code. E.g: - \code - im.set_extent(0, 100); - \endcode - * tells the compiler that dimension zero must be of extent 100, - * which may result in simplification of boundary checks. The - * value can be an arbitrary expression: - \code - im.set_extent(0, im.extent(1)); - \endcode - * declares that im is a square image (of unknown size), whereas: - \code - im.set_extent(0, (im.extent(0)/32)*32); - \endcode - * tells the compiler that the extent is a multiple of 32. */ - EXPORT OutputImageParam &set_extent(int dim, Expr extent); - - /** Set the min in a given dimension to equal the given - * expression. Setting the mins to zero may simplify some - * addressing math. */ - EXPORT OutputImageParam &set_min(int dim, Expr min); - - /** Set the stride in a given dimension to equal the given - * value. This is particularly helpful to set when - * vectorizing. Known strides for the vectorized dimension - * generate better code. */ - EXPORT OutputImageParam &set_stride(int dim, Expr stride); - - /** Set the min and extent in one call. */ - EXPORT OutputImageParam &set_bounds(int dim, Expr min, Expr extent); - - /** Get the dimensionality of this image parameter */ - EXPORT int dimensions() const; - - /** Get an expression giving the minimum coordinate in dimension 0, which - * by convention is the coordinate of the left edge of the image */ - EXPORT Expr left() const; - - /** Get an expression giving the maximum coordinate in dimension 0, which - * by convention is the coordinate of the right edge of the image */ - EXPORT Expr right() const; - - /** Get an expression giving the minimum coordinate in dimension 1, which - * by convention is the top of the image */ - EXPORT Expr top() const; - - /** Get an expression giving the maximum coordinate in dimension 1, which - * by convention is the bottom of the image */ - EXPORT Expr bottom() const; - - /** Get an expression giving the extent in dimension 0, which by - * convention is the width of the image */ - EXPORT Expr width() const; - - /** Get an expression giving the extent in dimension 1, which by - * convention is the height of the image */ - EXPORT Expr height() const; - - /** Get an expression giving the extent in dimension 2, which by - * convention is the channel-count of the image */ - EXPORT Expr channels() const; - - /** Get at the internal parameter object representing this ImageParam. */ - EXPORT Internal::Parameter parameter() const; - - /** Construct the appropriate argument matching this parameter, - * for the purpose of generating the right type signature when - * statically compiling halide pipelines. */ - EXPORT operator Argument() const; - - /** Using a param as the argument to an external stage treats it - * as an Expr */ - EXPORT operator ExternFuncArgument() const; +public: + /** Construct a NULL image parameter handle. */ + OutputImageParam() : dims(0) {} + + /** Construct an OutputImageParam that wraps an Internal Parameter object. */ + EXPORT OutputImageParam(const Internal::Parameter &p, int d); + + /** Get the name of this Param */ + EXPORT const std::string &name() const; + + /** Get the type of the image data this Param refers to */ + EXPORT Type type() const; + + /** Is this parameter handle non-NULL */ + EXPORT bool defined(); + + /** Get an expression representing the minimum coordinates of this image + * parameter in the given dimension. */ + EXPORT Expr min(int x) const; + + /** Get an expression representing the extent of this image + * parameter in the given dimension */ + EXPORT Expr extent(int x) const; + + /** Get an expression representing the stride of this image in the + * given dimension */ + EXPORT Expr stride(int x) const; + + /** Set the extent in a given dimension to equal the given + * expression. Images passed in that fail this check will generate + * a runtime error. Returns a reference to the ImageParam so that + * these calls may be chained. + * + * This may help the compiler generate better + * code. E.g: + \code + im.set_extent(0, 100); + \endcode + * tells the compiler that dimension zero must be of extent 100, + * which may result in simplification of boundary checks. The + * value can be an arbitrary expression: + \code + im.set_extent(0, im.extent(1)); + \endcode + * declares that im is a square image (of unknown size), whereas: + \code + im.set_extent(0, (im.extent(0)/32)*32); + \endcode + * tells the compiler that the extent is a multiple of 32. */ + EXPORT OutputImageParam &set_extent(int dim, Expr extent); + + /** Set the min in a given dimension to equal the given + * expression. Setting the mins to zero may simplify some + * addressing math. */ + EXPORT OutputImageParam &set_min(int dim, Expr min); + + /** Set the stride in a given dimension to equal the given + * value. This is particularly helpful to set when + * vectorizing. Known strides for the vectorized dimension + * generate better code. */ + EXPORT OutputImageParam &set_stride(int dim, Expr stride); + + /** Set the min and extent in one call. */ + EXPORT OutputImageParam &set_bounds(int dim, Expr min, Expr extent); + + /** Get the dimensionality of this image parameter */ + EXPORT int dimensions() const; + + /** Get an expression giving the minimum coordinate in dimension 0, which + * by convention is the coordinate of the left edge of the image */ + EXPORT Expr left() const; + + /** Get an expression giving the maximum coordinate in dimension 0, which + * by convention is the coordinate of the right edge of the image */ + EXPORT Expr right() const; + + /** Get an expression giving the minimum coordinate in dimension 1, which + * by convention is the top of the image */ + EXPORT Expr top() const; + + /** Get an expression giving the maximum coordinate in dimension 1, which + * by convention is the bottom of the image */ + EXPORT Expr bottom() const; + + /** Get an expression giving the extent in dimension 0, which by + * convention is the width of the image */ + EXPORT Expr width() const; + + /** Get an expression giving the extent in dimension 1, which by + * convention is the height of the image */ + EXPORT Expr height() const; + + /** Get an expression giving the extent in dimension 2, which by + * convention is the channel-count of the image */ + EXPORT Expr channels() const; + + /** Get at the internal parameter object representing this ImageParam. */ + EXPORT Internal::Parameter parameter() const; + + /** Construct the appropriate argument matching this parameter, + * for the purpose of generating the right type signature when + * statically compiling halide pipelines. */ + EXPORT operator Argument() const; + + /** Using a param as the argument to an external stage treats it + * as an Expr */ + EXPORT operator ExternFuncArgument() const; }; /** An Image parameter to a halide pipeline. E.g., the input image. */ -class ImageParam : public OutputImageParam { +class ImageParam : public OutputImageParam +{ public: - - /** Construct a NULL image parameter handle. */ - ImageParam() : OutputImageParam() {} - - /** Construct an image parameter of the given type and - * dimensionality, with an auto-generated unique name. */ - EXPORT ImageParam(Type t, int d); - - /** Construct an image parameter of the given type and - * dimensionality, with the given name */ - EXPORT ImageParam(Type t, int d, const std::string &n); - - /** Bind a buffer or image to this ImageParam. Only relevant for jitting */ - EXPORT void set(Buffer b); - - /** Get the buffer bound to this ImageParam. Only relevant for jitting */ - EXPORT Buffer get() const; - - /** Construct an expression which loads from this image - * parameter. The location is extended with enough implicit - * variables to match the dimensionality of the image - * (see \ref Var::implicit) - */ - // @{ - EXPORT Expr operator()() const; - EXPORT Expr operator()(Expr x) const; - EXPORT Expr operator()(Expr x, Expr y) const; - EXPORT Expr operator()(Expr x, Expr y, Expr z) const; - EXPORT Expr operator()(Expr x, Expr y, Expr z, Expr w) const; - EXPORT Expr operator()(std::vector) const; - EXPORT Expr operator()(std::vector) const; - // @} - - /** Treating the image parameter as an Expr is equivalent to call - * it with no arguments. For example, you can say: - * - \code - ImageParam im(UInt(8), 2); - Func f; - f = im*2; - \endcode - * - * This will define f as a two-dimensional function with value at - * position (x, y) equal to twice the value of the image parameter - * at the same location. - */ - operator Expr() const { - return (*this)(_); - } - -}; - -} + /** Construct a NULL image parameter handle. */ + ImageParam() : OutputImageParam() {} + + /** Construct an image parameter of the given type and + * dimensionality, with an auto-generated unique name. */ + EXPORT ImageParam(Type t, int d); + + /** Construct an image parameter of the given type and + * dimensionality, with the given name */ + EXPORT ImageParam(Type t, int d, const std::string &n); + + /** Bind a buffer or image to this ImageParam. Only relevant for jitting */ + EXPORT void set(Buffer b); + + /** Get the buffer bound to this ImageParam. Only relevant for jitting */ + EXPORT Buffer get() const; + + /** Construct an expression which loads from this image + * parameter. The location is extended with enough implicit + * variables to match the dimensionality of the image + * (see \ref Var::implicit) + */ + // @{ + EXPORT Expr operator()() const; + EXPORT Expr operator()(Expr x) const; + EXPORT Expr operator()(Expr x, Expr y) const; + EXPORT Expr operator()(Expr x, Expr y, Expr z) const; + EXPORT Expr operator()(Expr x, Expr y, Expr z, Expr w) const; + EXPORT Expr operator()(std::vector) const; + EXPORT Expr operator()(std::vector) const; + // @} + + /** Treating the image parameter as an Expr is equivalent to call + * it with no arguments. For example, you can say: + * + \code + ImageParam im(UInt(8), 2); + Func f; + f = im*2; + \endcode + * + * This will define f as a two-dimensional function with value at + * position (x, y) equal to twice the value of the image parameter + * at the same location. + */ + operator Expr() const { return (*this)(_); } +}; + +}// namespace Halide #endif #ifndef HALIDE_RDOM_H @@ -7307,47 +7235,45 @@ namespace Halide { * RDom, and use RDom::operator[] to get at the variables. For * single-dimensional reduction domains, you can just cast a * single-dimensional RDom to an RVar. */ -class RVar { - std::string _name; - Internal::ReductionDomain _domain; - int _index; +class RVar +{ + std::string _name; + Internal::ReductionDomain _domain; + int _index; - const Internal::ReductionVariable &_var() const { - return _domain.domain().at(_index); - } + const Internal::ReductionVariable &_var() const { return _domain.domain().at(_index); } public: - /** An empty reduction variable. */ - RVar() : _name(Internal::make_entity_name(this, "Halide::RVar", 'r')) {} - - /** Construct an RVar with the given name */ - explicit RVar(const std::string &n) : _name(n) { - // Make sure we don't get a unique name with the same name as - // this later: - Internal::unique_name(n, false); - } + /** An empty reduction variable. */ + RVar() : _name(Internal::make_entity_name(this, "Halide::RVar", 'r')) {} + + /** Construct an RVar with the given name */ + explicit RVar(const std::string &n) : _name(n) + { + // Make sure we don't get a unique name with the same name as + // this later: + Internal::unique_name(n, false); + } - /** Construct a reduction variable with the given name and - * bounds. Must be a member of the given reduction domain. */ - RVar(Internal::ReductionDomain domain, int index) : - _domain(domain), _index(index) { - } + /** Construct a reduction variable with the given name and + * bounds. Must be a member of the given reduction domain. */ + RVar(Internal::ReductionDomain domain, int index) : _domain(domain), _index(index) {} - /** The minimum value that this variable will take on */ - EXPORT Expr min() const; + /** The minimum value that this variable will take on */ + EXPORT Expr min() const; - /** The number that this variable will take on. The maximum value - * of this variable will be min() + extent() - 1 */ - EXPORT Expr extent() const; + /** The number that this variable will take on. The maximum value + * of this variable will be min() + extent() - 1 */ + EXPORT Expr extent() const; - /** The reduction domain this is associated with. */ - EXPORT Internal::ReductionDomain domain() const {return _domain;} + /** The reduction domain this is associated with. */ + EXPORT Internal::ReductionDomain domain() const { return _domain; } - /** The name of this reduction variable */ - EXPORT const std::string &name() const; + /** The name of this reduction variable */ + EXPORT const std::string &name() const; - /** Reduction variables can be used as expressions. */ - EXPORT operator Expr() const; + /** Reduction variables can be used as expressions. */ + EXPORT operator Expr() const; }; /** A multi-dimensional domain over which to iterate. Used when @@ -7467,79 +7393,134 @@ class RVar { * of the sum in y. This not only results in sum_x walking along the * rows, it also improves the locality of the entire pipeline. */ -class RDom { - Internal::ReductionDomain dom; +class RDom +{ + Internal::ReductionDomain dom; - void init_vars(std::string name); + void init_vars(std::string name); public: - /** Construct an undefined reduction domain. */ - EXPORT RDom() {} - - /** Construct a one-dimensional reduction domain with the given name. If the name - * is left blank, a unique one is auto-generated. */ - EXPORT RDom(Expr min, Expr extent, std::string name = ""); - - /** Construct a two-dimensional reduction domain with the given name. If the name - * is left blank, a unique one is auto-generated. */ - EXPORT RDom(Expr min0, Expr extent0, Expr min1, Expr extent1, std::string name = ""); - - /** Construct a multi-dimensional reduction domain with the given - * name. If the name is left blank, a unique one is - * auto-generated. */ - // @{ - EXPORT RDom(Expr min0, Expr extent0, Expr min1, Expr extent1, Expr min2, Expr extent2, std::string name = ""); - EXPORT RDom(Expr min0, Expr extent0, Expr min1, Expr extent1, Expr min2, Expr extent2, Expr min3, Expr extent3, - std::string name = ""); - EXPORT RDom(Expr min0, Expr extent0, Expr min1, Expr extent1, Expr min2, Expr extent2, Expr min3, Expr extent3, - Expr min4, Expr extent4, std::string name = ""); - EXPORT RDom(Expr min0, Expr extent0, Expr min1, Expr extent1, Expr min2, Expr extent2, Expr min3, Expr extent3, - Expr min4, Expr extent4, Expr min5, Expr extent5, std::string name = ""); - EXPORT RDom(Expr min0, Expr extent0, Expr min1, Expr extent1, Expr min2, Expr extent2, Expr min3, Expr extent3, - Expr min4, Expr extent4, Expr min5, Expr extent5, Expr min6, Expr extent6, std::string name = ""); - EXPORT RDom(Expr min0, Expr extent0, Expr min1, Expr extent1, Expr min2, Expr extent2, Expr min3, Expr extent3, - Expr min4, Expr extent4, Expr min5, Expr extent5, Expr min6, Expr extent6, Expr min7, Expr extent7, - std::string name = ""); - // @} - - /** Construct a reduction domain that iterates over all points in - * a given Buffer, Image, or ImageParam. Has the same - * dimensionality as the argument. */ - // @{ - EXPORT RDom(Buffer); - EXPORT RDom(ImageParam); - // @} - - /** Construct a reduction domain that wraps an Internal ReductionDomain object. */ - EXPORT RDom(Internal::ReductionDomain d); - - /** Get at the internal reduction domain object that this wraps. */ - Internal::ReductionDomain domain() const {return dom;} - - /** Check if this reduction domain is non-NULL */ - bool defined() const {return dom.defined();} - - /** Compare two reduction domains for equality of reference */ - bool same_as(const RDom &other) const {return dom.same_as(other.dom);} - - /** Get the dimensionality of a reduction domain */ - EXPORT int dimensions() const; - - /** Get at one of the dimensions of the reduction domain */ - EXPORT RVar operator[](int) const; - - /** Single-dimensional reduction domains can be used as RVars directly. */ - EXPORT operator RVar() const; - - /** Single-dimensional reduction domains can be also be used as Exprs directly. */ - EXPORT operator Expr() const; - - /** Direct access to the first four dimensions of the reduction - * domain. Some of these variables may be undefined if the - * reduction domain has fewer than four dimensions. */ - // @{ - RVar x, y, z, w; - // @} + /** Construct an undefined reduction domain. */ + EXPORT RDom() {} + + /** Construct a one-dimensional reduction domain with the given name. If the name + * is left blank, a unique one is auto-generated. */ + EXPORT RDom(Expr min, Expr extent, std::string name = ""); + + /** Construct a two-dimensional reduction domain with the given name. If the name + * is left blank, a unique one is auto-generated. */ + EXPORT RDom(Expr min0, Expr extent0, Expr min1, Expr extent1, std::string name = ""); + + /** Construct a multi-dimensional reduction domain with the given + * name. If the name is left blank, a unique one is + * auto-generated. */ + // @{ + EXPORT RDom(Expr min0, Expr extent0, Expr min1, Expr extent1, Expr min2, Expr extent2, std::string name = ""); + EXPORT RDom(Expr min0, + Expr extent0, + Expr min1, + Expr extent1, + Expr min2, + Expr extent2, + Expr min3, + Expr extent3, + std::string name = ""); + EXPORT RDom(Expr min0, + Expr extent0, + Expr min1, + Expr extent1, + Expr min2, + Expr extent2, + Expr min3, + Expr extent3, + Expr min4, + Expr extent4, + std::string name = ""); + EXPORT RDom(Expr min0, + Expr extent0, + Expr min1, + Expr extent1, + Expr min2, + Expr extent2, + Expr min3, + Expr extent3, + Expr min4, + Expr extent4, + Expr min5, + Expr extent5, + std::string name = ""); + EXPORT RDom(Expr min0, + Expr extent0, + Expr min1, + Expr extent1, + Expr min2, + Expr extent2, + Expr min3, + Expr extent3, + Expr min4, + Expr extent4, + Expr min5, + Expr extent5, + Expr min6, + Expr extent6, + std::string name = ""); + EXPORT RDom(Expr min0, + Expr extent0, + Expr min1, + Expr extent1, + Expr min2, + Expr extent2, + Expr min3, + Expr extent3, + Expr min4, + Expr extent4, + Expr min5, + Expr extent5, + Expr min6, + Expr extent6, + Expr min7, + Expr extent7, + std::string name = ""); + // @} + + /** Construct a reduction domain that iterates over all points in + * a given Buffer, Image, or ImageParam. Has the same + * dimensionality as the argument. */ + // @{ + EXPORT RDom(Buffer); + EXPORT RDom(ImageParam); + // @} + + /** Construct a reduction domain that wraps an Internal ReductionDomain object. */ + EXPORT RDom(Internal::ReductionDomain d); + + /** Get at the internal reduction domain object that this wraps. */ + Internal::ReductionDomain domain() const { return dom; } + + /** Check if this reduction domain is non-NULL */ + bool defined() const { return dom.defined(); } + + /** Compare two reduction domains for equality of reference */ + bool same_as(const RDom &other) const { return dom.same_as(other.dom); } + + /** Get the dimensionality of a reduction domain */ + EXPORT int dimensions() const; + + /** Get at one of the dimensions of the reduction domain */ + EXPORT RVar operator[](int) const; + + /** Single-dimensional reduction domains can be used as RVars directly. */ + EXPORT operator RVar() const; + + /** Single-dimensional reduction domains can be also be used as Exprs directly. */ + EXPORT operator Expr() const; + + /** Direct access to the first four dimensions of the reduction + * domain. Some of these variables may be undefined if the + * reduction domain has fewer than four dimensions. */ + // @{ + RVar x, y, z, w; + // @} }; /** Emit an RVar in a human-readable form */ @@ -7547,7 +7528,7 @@ std::ostream &operator<<(std::ostream &stream, RVar); /** Emit an RDom in a human-readable form. */ std::ostream &operator<<(std::ostream &stream, RDom); -} +}// namespace Halide #endif #ifndef HALIDE_IMAGE_H @@ -7574,141 +7555,133 @@ class FuncRefExpr; /** Create a small array of Exprs for defining and calling functions * with multiple outputs. */ -class Tuple { +class Tuple +{ private: - std::vector exprs; + std::vector exprs; + public: - /** The number of elements in the tuple. */ - size_t size() const { return exprs.size(); } + /** The number of elements in the tuple. */ + size_t size() const { return exprs.size(); } + + /** Get a reference to an element. */ + Expr &operator[](size_t x) + { + user_assert(x < exprs.size()) << "Tuple access out of bounds\n"; + return exprs[x]; + } - /** Get a reference to an element. */ - Expr &operator[](size_t x) { - user_assert(x < exprs.size()) << "Tuple access out of bounds\n"; - return exprs[x]; - } + /** Get a copy of an element. */ + Expr operator[](size_t x) const + { + user_assert(x < exprs.size()) << "Tuple access out of bounds\n"; + return exprs[x]; + } - /** Get a copy of an element. */ - Expr operator[](size_t x) const { - user_assert(x < exprs.size()) << "Tuple access out of bounds\n"; - return exprs[x]; - } + /** Construct a Tuple from some Exprs. */ + //@{ + Tuple(Expr a, Expr b) : exprs(Internal::vec(a, b)) {} - /** Construct a Tuple from some Exprs. */ - //@{ - Tuple(Expr a, Expr b) : - exprs(Internal::vec(a, b)) { - } + Tuple(Expr a, Expr b, Expr c) : exprs(Internal::vec(a, b, c)) {} - Tuple(Expr a, Expr b, Expr c) : - exprs(Internal::vec(a, b, c)) { - } + Tuple(Expr a, Expr b, Expr c, Expr d) : exprs(Internal::vec(a, b, c, d)) {} - Tuple(Expr a, Expr b, Expr c, Expr d) : - exprs(Internal::vec(a, b, c, d)) { - } + Tuple(Expr a, Expr b, Expr c, Expr d, Expr e) : exprs(Internal::vec(a, b, c, d, e)) {} - Tuple(Expr a, Expr b, Expr c, Expr d, Expr e) : - exprs(Internal::vec(a, b, c, d, e)) { - } + Tuple(Expr a, Expr b, Expr c, Expr d, Expr e, Expr f) : exprs(Internal::vec(a, b, c, d, e, f)) {} + //@} - Tuple(Expr a, Expr b, Expr c, Expr d, Expr e, Expr f) : - exprs(Internal::vec(a, b, c, d, e, f)) { - } - //@} - - /** Construct a Tuple from a vector of Exprs */ - explicit Tuple(const std::vector &e) : exprs(e) { - user_assert(e.size() > 0) << "Tuples must have at least one element\n"; - } + /** Construct a Tuple from a vector of Exprs */ + explicit Tuple(const std::vector &e) : exprs(e) + { + user_assert(e.size() > 0) << "Tuples must have at least one element\n"; + } - /** Construct a Tuple from a function reference. */ - // @{ - EXPORT Tuple(const FuncRefVar &); - EXPORT Tuple(const FuncRefExpr &); - // @} + /** Construct a Tuple from a function reference. */ + // @{ + EXPORT Tuple(const FuncRefVar &); + EXPORT Tuple(const FuncRefExpr &); + // @} - /** Treat the tuple as a vector of Exprs */ - const std::vector &as_vector() const { - return exprs; - } + /** Treat the tuple as a vector of Exprs */ + const std::vector &as_vector() const { return exprs; } }; /** Funcs with Tuple values return multiple buffers when you realize * them. Tuples are to Exprs as Realizations are to Buffers. */ -class Realization { +class Realization +{ private: - std::vector buffers; -public: - /** The number of buffers in the Realization. */ - size_t size() const { return buffers.size(); } + std::vector buffers; - /** Get a reference to one of the buffers. */ - Buffer &operator[](size_t x) { - user_assert(x < buffers.size()) << "Realization access out of bounds\n"; - return buffers[x]; - } +public: + /** The number of buffers in the Realization. */ + size_t size() const { return buffers.size(); } + + /** Get a reference to one of the buffers. */ + Buffer &operator[](size_t x) + { + user_assert(x < buffers.size()) << "Realization access out of bounds\n"; + return buffers[x]; + } - /** Get one of the buffers. */ - Buffer operator[](size_t x) const { - user_assert(x < buffers.size()) << "Realization access out of bounds\n"; - return buffers[x]; - } + /** Get one of the buffers. */ + Buffer operator[](size_t x) const + { + user_assert(x < buffers.size()) << "Realization access out of bounds\n"; + return buffers[x]; + } - /** Single-element realizations are implicitly castable to Buffers. */ - operator Buffer() const { - user_assert(buffers.size() == 1) << "Can only cast single-element realizations to buffers or images\n"; - return buffers[0]; - } + /** Single-element realizations are implicitly castable to Buffers. */ + operator Buffer() const + { + user_assert(buffers.size() == 1) << "Can only cast single-element realizations to buffers or images\n"; + return buffers[0]; + } - /** Construct a Realization from some Buffers. */ - //@{ - Realization(Buffer a, Buffer b) : - buffers(Internal::vec(a, b)) {} + /** Construct a Realization from some Buffers. */ + //@{ + Realization(Buffer a, Buffer b) : buffers(Internal::vec(a, b)) {} - Realization(Buffer a, Buffer b, Buffer c) : - buffers(Internal::vec(a, b, c)) {} + Realization(Buffer a, Buffer b, Buffer c) : buffers(Internal::vec(a, b, c)) {} - Realization(Buffer a, Buffer b, Buffer c, Buffer d) : - buffers(Internal::vec(a, b, c, d)) {} + Realization(Buffer a, Buffer b, Buffer c, Buffer d) : buffers(Internal::vec(a, b, c, d)) {} - Realization(Buffer a, Buffer b, Buffer c, Buffer d, Buffer e) : - buffers(Internal::vec(a, b, c, d, e)) {} + Realization(Buffer a, Buffer b, Buffer c, Buffer d, Buffer e) : buffers(Internal::vec(a, b, c, d, e)) {} - Realization(Buffer a, Buffer b, Buffer c, Buffer d, Buffer e, Buffer f) : - buffers(Internal::vec(a, b, c, d, e, f)) {} - //@} + Realization(Buffer a, Buffer b, Buffer c, Buffer d, Buffer e, Buffer f) + : buffers(Internal::vec(a, b, c, d, e, f)) + {} + //@} - /** Construct a Realization from a vector of Buffers */ - explicit Realization(const std::vector &e) : buffers(e) { - user_assert(e.size() > 0) << "Realizations must have at least one element\n"; - } + /** Construct a Realization from a vector of Buffers */ + explicit Realization(const std::vector &e) : buffers(e) + { + user_assert(e.size() > 0) << "Realizations must have at least one element\n"; + } - /** Treat the Realization as a vector of Buffers */ - const std::vector &as_vector() const { - return buffers; - } + /** Treat the Realization as a vector of Buffers */ + const std::vector &as_vector() const { return buffers; } }; /** Equivalents of some standard operators for tuples. */ // @{ -inline Tuple tuple_select(Tuple condition, const Tuple &true_value, const Tuple &false_value) { - Tuple result(std::vector(condition.size())); - for (size_t i = 0; i < result.size(); i++) { - result[i] = select(condition[i], true_value[i], false_value[i]); - } - return result; +inline Tuple tuple_select(Tuple condition, const Tuple &true_value, const Tuple &false_value) +{ + Tuple result(std::vector(condition.size())); + for (size_t i = 0; i < result.size(); i++) { result[i] = select(condition[i], true_value[i], false_value[i]); } + return result; } -inline Tuple tuple_select(Expr condition, const Tuple &true_value, const Tuple &false_value) { - Tuple result(std::vector(true_value.size())); - for (size_t i = 0; i < result.size(); i++) { - result[i] = select(condition, true_value[i], false_value[i]); - } - return result; +inline Tuple tuple_select(Expr condition, const Tuple &true_value, const Tuple &false_value) +{ + Tuple result(std::vector(true_value.size())); + for (size_t i = 0; i < result.size(); i++) { result[i] = select(condition, true_value[i], false_value[i]); } + return result; } // @} -} +}// namespace Halide #endif @@ -7719,136 +7692,138 @@ namespace Halide { * of Image private, so that they can safely throw errors without the * risk of being inlined (which in turns messes up reporting of line * numbers). */ -class ImageBase { +class ImageBase +{ protected: - /** The underlying memory object */ - Buffer buffer; - - /** These fields are also stored in the buffer, but they're cached - * here in the handle to make operator() fast. This is safe to do - * because the buffer is never modified - */ - // @{ - void *origin; - int stride_0, stride_1, stride_2, stride_3, dims; - // @} - - /** Prepare the buffer to be used as an image. Makes sure that the - * cached strides are correct, and that the image data is on the - * host. */ - void prepare_for_direct_pixel_access(); + /** The underlying memory object */ + Buffer buffer; + + /** These fields are also stored in the buffer, but they're cached + * here in the handle to make operator() fast. This is safe to do + * because the buffer is never modified + */ + // @{ + void *origin; + int stride_0, stride_1, stride_2, stride_3, dims; + // @} + + /** Prepare the buffer to be used as an image. Makes sure that the + * cached strides are correct, and that the image data is on the + * host. */ + void prepare_for_direct_pixel_access(); + + bool add_implicit_args_if_placeholder(std::vector &args, + Expr last_arg, + int total_args, + bool placeholder_seen) const; - bool add_implicit_args_if_placeholder(std::vector &args, - Expr last_arg, - int total_args, - bool placeholder_seen) const; public: - /** Construct an undefined image handle */ - ImageBase() : origin(NULL), stride_0(0), stride_1(0), stride_2(0), stride_3(0), dims(0) {} - - /** Allocate an image with the given dimensions. */ - EXPORT ImageBase(Type t, int x, int y = 0, int z = 0, int w = 0, const std::string &name = ""); - - /** Wrap a buffer in an Image object, so that we can directly - * access its pixels in a type-safe way. */ - EXPORT ImageBase(Type t, const Buffer &buf); - - /** Wrap a single-element realization in an Image object. */ - EXPORT ImageBase(Type t, const Realization &r); - - /** Wrap a buffer_t in an Image object, so that we can access its - * pixels. */ - EXPORT ImageBase(Type t, const buffer_t *b, const std::string &name = ""); - - /** Get the name of this image. */ - EXPORT const std::string &name(); - - /** Manually copy-back data to the host, if it's on a device. This - * is done for you if you construct an image from a buffer, but - * you might need to call this if you realize a gpu kernel into an - * existing image */ - EXPORT void copy_to_host(); - - /** Mark the buffer as dirty-on-host. is done for you if you - * construct an image from a buffer, but you might need to call - * this if you realize a gpu kernel into an existing image, or - * modify the data via some other back-door. */ - EXPORT void set_host_dirty(bool dirty = true); - - /** Check if this image handle points to actual data */ - EXPORT bool defined() const; - - /** Get the dimensionality of the data. Typically two for grayscale images, and three for color images. */ - EXPORT int dimensions() const; - - /** Get the size of a dimension */ - EXPORT int extent(int dim) const; - - /** Get the min coordinate of a dimension. The top left of the - * image represents this point in a function that was realized - * into this image. */ - EXPORT int min(int dim) const; - - /** Set the min coordinates of a dimension. */ - EXPORT void set_min(int m0, int m1 = 0, int m2 = 0, int m3 = 0); - - /** Get the number of elements in the buffer between two adjacent - * elements in the given dimension. For example, the stride in - * dimension 0 is usually 1, and the stride in dimension 1 is - * usually the extent of dimension 0. This is not necessarily true - * though. */ - EXPORT int stride(int dim) const; - - /** Get the extent of dimension 0, which by convention we use as - * the width of the image. Unlike extent(0), returns one if the - * buffer is zero-dimensional. */ - EXPORT int width() const; - - /** Get the extent of dimension 1, which by convention we use as - * the height of the image. Unlike extent(1), returns one if the - * buffer has fewer than two dimensions. */ - EXPORT int height() const; - - /** Get the extent of dimension 2, which by convention we use as - * the number of color channels (often 3). Unlike extent(2), - * returns one if the buffer has fewer than three dimensions. */ - EXPORT int channels() const; - - /** Get the minimum coordinate in dimension 0, which by convention - * is the coordinate of the left edge of the image. Returns zero - * for zero-dimensional images. */ - EXPORT int left() const; - - /** Get the maximum coordinate in dimension 0, which by convention - * is the coordinate of the right edge of the image. Returns zero - * for zero-dimensional images. */ - EXPORT int right() const; - - /** Get the minimum coordinate in dimension 1, which by convention - * is the top of the image. Returns zero for zero- or - * one-dimensional images. */ - EXPORT int top() const; - - /** Get the maximum coordinate in dimension 1, which by convention - * is the bottom of the image. Returns zero for zero- or - * one-dimensional images. */ - EXPORT int bottom() const; - - /** Construct an expression which loads from this image. The - * location is extended with enough implicit variables to match - * the dimensionality of the image (see \ref Var::implicit) */ - // @{ - EXPORT Expr operator()() const; - EXPORT Expr operator()(Expr x) const; - EXPORT Expr operator()(Expr x, Expr y) const; - EXPORT Expr operator()(Expr x, Expr y, Expr z) const; - EXPORT Expr operator()(Expr x, Expr y, Expr z, Expr w) const; - EXPORT Expr operator()(std::vector) const; - EXPORT Expr operator()(std::vector) const; - // @} - - /** Get a pointer to the raw buffer_t that this image holds */ - EXPORT buffer_t *raw_buffer() const; + /** Construct an undefined image handle */ + ImageBase() : origin(NULL), stride_0(0), stride_1(0), stride_2(0), stride_3(0), dims(0) {} + + /** Allocate an image with the given dimensions. */ + EXPORT ImageBase(Type t, int x, int y = 0, int z = 0, int w = 0, const std::string &name = ""); + + /** Wrap a buffer in an Image object, so that we can directly + * access its pixels in a type-safe way. */ + EXPORT ImageBase(Type t, const Buffer &buf); + + /** Wrap a single-element realization in an Image object. */ + EXPORT ImageBase(Type t, const Realization &r); + + /** Wrap a buffer_t in an Image object, so that we can access its + * pixels. */ + EXPORT ImageBase(Type t, const buffer_t *b, const std::string &name = ""); + + /** Get the name of this image. */ + EXPORT const std::string &name(); + + /** Manually copy-back data to the host, if it's on a device. This + * is done for you if you construct an image from a buffer, but + * you might need to call this if you realize a gpu kernel into an + * existing image */ + EXPORT void copy_to_host(); + + /** Mark the buffer as dirty-on-host. is done for you if you + * construct an image from a buffer, but you might need to call + * this if you realize a gpu kernel into an existing image, or + * modify the data via some other back-door. */ + EXPORT void set_host_dirty(bool dirty = true); + + /** Check if this image handle points to actual data */ + EXPORT bool defined() const; + + /** Get the dimensionality of the data. Typically two for grayscale images, and three for color images. */ + EXPORT int dimensions() const; + + /** Get the size of a dimension */ + EXPORT int extent(int dim) const; + + /** Get the min coordinate of a dimension. The top left of the + * image represents this point in a function that was realized + * into this image. */ + EXPORT int min(int dim) const; + + /** Set the min coordinates of a dimension. */ + EXPORT void set_min(int m0, int m1 = 0, int m2 = 0, int m3 = 0); + + /** Get the number of elements in the buffer between two adjacent + * elements in the given dimension. For example, the stride in + * dimension 0 is usually 1, and the stride in dimension 1 is + * usually the extent of dimension 0. This is not necessarily true + * though. */ + EXPORT int stride(int dim) const; + + /** Get the extent of dimension 0, which by convention we use as + * the width of the image. Unlike extent(0), returns one if the + * buffer is zero-dimensional. */ + EXPORT int width() const; + + /** Get the extent of dimension 1, which by convention we use as + * the height of the image. Unlike extent(1), returns one if the + * buffer has fewer than two dimensions. */ + EXPORT int height() const; + + /** Get the extent of dimension 2, which by convention we use as + * the number of color channels (often 3). Unlike extent(2), + * returns one if the buffer has fewer than three dimensions. */ + EXPORT int channels() const; + + /** Get the minimum coordinate in dimension 0, which by convention + * is the coordinate of the left edge of the image. Returns zero + * for zero-dimensional images. */ + EXPORT int left() const; + + /** Get the maximum coordinate in dimension 0, which by convention + * is the coordinate of the right edge of the image. Returns zero + * for zero-dimensional images. */ + EXPORT int right() const; + + /** Get the minimum coordinate in dimension 1, which by convention + * is the top of the image. Returns zero for zero- or + * one-dimensional images. */ + EXPORT int top() const; + + /** Get the maximum coordinate in dimension 1, which by convention + * is the bottom of the image. Returns zero for zero- or + * one-dimensional images. */ + EXPORT int bottom() const; + + /** Construct an expression which loads from this image. The + * location is extended with enough implicit variables to match + * the dimensionality of the image (see \ref Var::implicit) */ + // @{ + EXPORT Expr operator()() const; + EXPORT Expr operator()(Expr x) const; + EXPORT Expr operator()(Expr x, Expr y) const; + EXPORT Expr operator()(Expr x, Expr y, Expr z) const; + EXPORT Expr operator()(Expr x, Expr y, Expr z, Expr w) const; + EXPORT Expr operator()(std::vector) const; + EXPORT Expr operator()(std::vector) const; + // @} + + /** Get a pointer to the raw buffer_t that this image holds */ + EXPORT buffer_t *raw_buffer() const; }; /** A reference-counted handle on a dense multidimensional array @@ -7858,290 +7833,270 @@ class ImageBase { * the color channel. In general we store color images in * color-planes, as opposed to packed RGB, because this tends to * vectorize more cleanly. */ -template -class Image : public ImageBase { +template class Image : public ImageBase +{ public: - /** Construct an undefined image handle */ - Image() : ImageBase() {} + /** Construct an undefined image handle */ + Image() : ImageBase() {} - /** Allocate an image with the given dimensions. */ - // @{ - NO_INLINE Image(int x, int y = 0, int z = 0, int w = 0, const std::string &name = "") : - ImageBase(type_of(), x, y, z, w, name) {} + /** Allocate an image with the given dimensions. */ + // @{ + NO_INLINE Image(int x, int y = 0, int z = 0, int w = 0, const std::string &name = "") + : ImageBase(type_of(), x, y, z, w, name) + {} - NO_INLINE Image(int x, int y, int z, const std::string &name) : - ImageBase(type_of(), x, y, z, 0, name) {} + NO_INLINE Image(int x, int y, int z, const std::string &name) : ImageBase(type_of(), x, y, z, 0, name) {} - NO_INLINE Image(int x, int y, const std::string &name) : - ImageBase(type_of(), x, y, 0, 0, name) {} + NO_INLINE Image(int x, int y, const std::string &name) : ImageBase(type_of(), x, y, 0, 0, name) {} - NO_INLINE Image(int x, const std::string &name) : - ImageBase(type_of(), x, 0, 0, 0, name) {} - // @} + NO_INLINE Image(int x, const std::string &name) : ImageBase(type_of(), x, 0, 0, 0, name) {} + // @} - /** Wrap a buffer in an Image object, so that we can directly - * access its pixels in a type-safe way. */ - NO_INLINE Image(const Buffer &buf) : ImageBase(type_of(), buf) {} + /** Wrap a buffer in an Image object, so that we can directly + * access its pixels in a type-safe way. */ + NO_INLINE Image(const Buffer &buf) : ImageBase(type_of(), buf) {} - /** Wrap a single-element realization in an Image object. */ - NO_INLINE Image(const Realization &r) : ImageBase(type_of(), r) {} + /** Wrap a single-element realization in an Image object. */ + NO_INLINE Image(const Realization &r) : ImageBase(type_of(), r) {} - /** Wrap a buffer_t in an Image object, so that we can access its - * pixels. */ - NO_INLINE Image(const buffer_t *b, const std::string &name = "") : - ImageBase(type_of(), b, name) {} - - /** Get a pointer to the element at the min location. */ - NO_INLINE T *data() const { - user_assert(defined()) << "data of undefined Image\n"; - return (T *)buffer.host_ptr(); - } + /** Wrap a buffer_t in an Image object, so that we can access its + * pixels. */ + NO_INLINE Image(const buffer_t *b, const std::string &name = "") : ImageBase(type_of(), b, name) {} - using ImageBase::operator(); + /** Get a pointer to the element at the min location. */ + NO_INLINE T *data() const + { + user_assert(defined()) << "data of undefined Image\n"; + return (T *)buffer.host_ptr(); + } - /** Assuming this image is one-dimensional, get the value of the - * element at position x */ - const T &operator()(int x) const { - return ((T *)origin)[x*stride_0]; - } + using ImageBase::operator(); - /** Assuming this image is two-dimensional, get the value of the - * element at position (x, y) */ - const T &operator()(int x, int y) const { - return ((T *)origin)[x*stride_0 + y*stride_1]; - } + /** Assuming this image is one-dimensional, get the value of the + * element at position x */ + const T &operator()(int x) const { return ((T *)origin)[x * stride_0]; } - /** Assuming this image is three-dimensional, get the value of the - * element at position (x, y, z) */ - const T &operator()(int x, int y, int z) const { - return ((T *)origin)[x*stride_0 + y*stride_1 + z*stride_2]; - } + /** Assuming this image is two-dimensional, get the value of the + * element at position (x, y) */ + const T &operator()(int x, int y) const { return ((T *)origin)[x * stride_0 + y * stride_1]; } - /** Assuming this image is four-dimensional, get the value of the - * element at position (x, y, z, w) */ - const T &operator()(int x, int y, int z, int w) const { - return ((T *)origin)[x*stride_0 + y*stride_1 + z*stride_2 + w*stride_3]; - } + /** Assuming this image is three-dimensional, get the value of the + * element at position (x, y, z) */ + const T &operator()(int x, int y, int z) const { return ((T *)origin)[x * stride_0 + y * stride_1 + z * stride_2]; } - /** Assuming this image is one-dimensional, get a reference to the - * element at position x */ - T &operator()(int x) { - return ((T *)origin)[x*stride_0]; - } + /** Assuming this image is four-dimensional, get the value of the + * element at position (x, y, z, w) */ + const T &operator()(int x, int y, int z, int w) const + { + return ((T *)origin)[x * stride_0 + y * stride_1 + z * stride_2 + w * stride_3]; + } - /** Assuming this image is two-dimensional, get a reference to the - * element at position (x, y) */ - T &operator()(int x, int y) { - return ((T *)origin)[x*stride_0 + y*stride_1]; - } + /** Assuming this image is one-dimensional, get a reference to the + * element at position x */ + T &operator()(int x) { return ((T *)origin)[x * stride_0]; } - /** Assuming this image is three-dimensional, get a reference to the - * element at position (x, y, z) */ - T &operator()(int x, int y, int z) { - return ((T *)origin)[x*stride_0 + y*stride_1 + z*stride_2]; - } - - /** Assuming this image is four-dimensional, get a reference to the - * element at position (x, y, z, w) */ - T &operator()(int x, int y, int z, int w) { - return ((T *)origin)[x*stride_0 + y*stride_1 + z*stride_2 + w*stride_3]; - } + /** Assuming this image is two-dimensional, get a reference to the + * element at position (x, y) */ + T &operator()(int x, int y) { return ((T *)origin)[x * stride_0 + y * stride_1]; } - /** Get a handle on the Buffer that this image holds */ - operator Buffer() const { - return buffer; - } + /** Assuming this image is three-dimensional, get a reference to the + * element at position (x, y, z) */ + T &operator()(int x, int y, int z) { return ((T *)origin)[x * stride_0 + y * stride_1 + z * stride_2]; } - /** Convert this image to an argument to a halide pipeline. */ - operator Argument() const { - return Argument(buffer); - } + /** Assuming this image is four-dimensional, get a reference to the + * element at position (x, y, z, w) */ + T &operator()(int x, int y, int z, int w) + { + return ((T *)origin)[x * stride_0 + y * stride_1 + z * stride_2 + w * stride_3]; + } - /** Convert this image to an argument to an extern stage. */ - operator ExternFuncArgument() const { - return ExternFuncArgument(buffer); - } + /** Get a handle on the Buffer that this image holds */ + operator Buffer() const { return buffer; } - /** Treating the image as an Expr is equivalent to call it with no - * arguments. For example, you can say: - * - \code - Image im(10, 10); - Func f; - f = im*2; - \endcode - * - * This will define f as a two-dimensional function with value at - * position (x, y) equal to twice the value of the image at the - * same location. - */ - operator Expr() const { - return (*this)(_); - } + /** Convert this image to an argument to a halide pipeline. */ + operator Argument() const { return Argument(buffer); } + /** Convert this image to an argument to an extern stage. */ + operator ExternFuncArgument() const { return ExternFuncArgument(buffer); } + /** Treating the image as an Expr is equivalent to call it with no + * arguments. For example, you can say: + * + \code + Image im(10, 10); + Func f; + f = im*2; + \endcode + * + * This will define f as a two-dimensional function with value at + * position (x, y) equal to twice the value of the image at the + * same location. + */ + operator Expr() const { return (*this)(_); } }; -} +}// namespace Halide #endif namespace Halide { -enum GPUAPI { - GPU_Default, - GPU_CUDA, - GPU_OpenCL, - GPU_GLSL -}; +enum GPUAPI { GPU_Default, GPU_CUDA, GPU_OpenCL, GPU_GLSL }; /** A class that can represent Vars or RVars. Used for reorder calls * which can accept a mix of either. */ -struct VarOrRVar { - VarOrRVar(const std::string &n, bool r) : var(n), rvar(n), is_rvar(r) {} - VarOrRVar(const Var &v) : var(v), is_rvar(false) {} - VarOrRVar(const RVar &r) : rvar(r), is_rvar(true) {} - VarOrRVar(const RDom &r) : rvar(RVar(r)), is_rvar(true) {} - - const std::string &name() const { - if (is_rvar) return rvar.name(); - else return var.name(); - } +struct VarOrRVar +{ + VarOrRVar(const std::string &n, bool r) : var(n), rvar(n), is_rvar(r) {} + VarOrRVar(const Var &v) : var(v), is_rvar(false) {} + VarOrRVar(const RVar &r) : rvar(r), is_rvar(true) {} + VarOrRVar(const RDom &r) : rvar(RVar(r)), is_rvar(true) {} + + const std::string &name() const + { + if (is_rvar) + return rvar.name(); + else + return var.name(); + } - const Var var; - const RVar rvar; - const bool is_rvar; + const Var var; + const RVar rvar; + const bool is_rvar; }; /** A single definition of a Func. May be a pure or update definition. */ -class Stage { - Internal::Schedule schedule; - void set_dim_type(VarOrRVar var, Internal::For::ForType t); - void split(const std::string &old, const std::string &outer, const std::string &inner, Expr factor, bool exact); - std::string stage_name; -public: - Stage(Internal::Schedule s, const std::string &n) : - schedule(s), stage_name(n) {s.touched();} - - /** Return a string describing the current var list taking into - * account all the splits, reorders, and tiles. */ - EXPORT std::string dump_argument_list() const; - - /** Return the name of this stage, e.g. "f.update(2)" */ - EXPORT const std::string &name() const; - - /** Scheduling calls that control how the domain of this stage is - * traversed. See the documentation for Func for the meanings. */ - // @{ - - EXPORT Stage &split(VarOrRVar old, VarOrRVar outer, VarOrRVar inner, Expr factor); - EXPORT Stage &fuse(VarOrRVar inner, VarOrRVar outer, VarOrRVar fused); - EXPORT Stage &serial(VarOrRVar var); - EXPORT Stage ¶llel(VarOrRVar var); - EXPORT Stage &vectorize(VarOrRVar var); - EXPORT Stage &unroll(VarOrRVar var); - EXPORT Stage ¶llel(VarOrRVar var, Expr task_size); - EXPORT Stage &vectorize(VarOrRVar var, int factor); - EXPORT Stage &unroll(VarOrRVar var, int factor); - EXPORT Stage &tile(VarOrRVar x, VarOrRVar y, - VarOrRVar xo, VarOrRVar yo, - VarOrRVar xi, VarOrRVar yi, Expr - xfactor, Expr yfactor); - EXPORT Stage &tile(VarOrRVar x, VarOrRVar y, - VarOrRVar xi, VarOrRVar yi, - Expr xfactor, Expr yfactor); - EXPORT Stage &reorder(const std::vector &vars); - EXPORT Stage &reorder(VarOrRVar x, VarOrRVar y); - EXPORT Stage &reorder(VarOrRVar x, VarOrRVar y, VarOrRVar z); - EXPORT Stage &reorder(VarOrRVar x, VarOrRVar y, VarOrRVar z, - VarOrRVar w); - EXPORT Stage &reorder(VarOrRVar x, VarOrRVar y, VarOrRVar z, - VarOrRVar w, VarOrRVar t); - EXPORT Stage &reorder(VarOrRVar x, VarOrRVar y, VarOrRVar z, - VarOrRVar w, VarOrRVar t1, VarOrRVar t2); - EXPORT Stage &reorder(VarOrRVar x, VarOrRVar y, VarOrRVar z, - VarOrRVar w, VarOrRVar t1, VarOrRVar t2, - VarOrRVar t3); - EXPORT Stage &reorder(VarOrRVar x, VarOrRVar y, VarOrRVar z, - VarOrRVar w, VarOrRVar t1, VarOrRVar t2, - VarOrRVar t3, VarOrRVar t4); - EXPORT Stage &reorder(VarOrRVar x, VarOrRVar y, VarOrRVar z, - VarOrRVar w, VarOrRVar t1, VarOrRVar t2, - VarOrRVar t3, VarOrRVar t4, VarOrRVar t5); - EXPORT Stage &reorder(VarOrRVar x, VarOrRVar y, VarOrRVar z, - VarOrRVar w, VarOrRVar t1, VarOrRVar t2, - VarOrRVar t3, VarOrRVar t4, VarOrRVar t5, - VarOrRVar t6); - EXPORT Stage &rename(VarOrRVar old_name, VarOrRVar new_name); - EXPORT Stage specialize(Expr condition); - - EXPORT Stage &gpu_threads(VarOrRVar thread_x, GPUAPI gpu_api = GPU_Default); - EXPORT Stage &gpu_threads(VarOrRVar thread_x, VarOrRVar thread_y, GPUAPI gpu_api = GPU_Default); - EXPORT Stage &gpu_threads(VarOrRVar thread_x, VarOrRVar thread_y, VarOrRVar thread_z, GPUAPI gpu_api = GPU_Default); - EXPORT Stage &gpu_single_thread(GPUAPI gpu_api = GPU_Default); - - EXPORT Stage &gpu_blocks(VarOrRVar block_x, GPUAPI gpu_api = GPU_Default); - EXPORT Stage &gpu_blocks(VarOrRVar block_x, VarOrRVar block_y, GPUAPI gpu_api = GPU_Default); - EXPORT Stage &gpu_blocks(VarOrRVar block_x, VarOrRVar block_y, VarOrRVar block_z, GPUAPI gpu_api = GPU_Default); - - EXPORT Stage &gpu(VarOrRVar block_x, VarOrRVar thread_x, GPUAPI gpu_api = GPU_Default); - EXPORT Stage &gpu(VarOrRVar block_x, VarOrRVar block_y, - VarOrRVar thread_x, VarOrRVar thread_y, - GPUAPI gpu_api = GPU_Default); - EXPORT Stage &gpu(VarOrRVar block_x, VarOrRVar block_y, VarOrRVar block_z, - VarOrRVar thread_x, VarOrRVar thread_y, VarOrRVar thread_z, - GPUAPI gpu_api = GPU_Default); - EXPORT Stage &gpu_tile(VarOrRVar x, Expr x_size, GPUAPI gpu_api = GPU_Default); - EXPORT Stage &gpu_tile(VarOrRVar x, VarOrRVar y, Expr x_size, Expr y_size, - GPUAPI gpu_api = GPU_Default); - EXPORT Stage &gpu_tile(VarOrRVar x, VarOrRVar y, VarOrRVar z, - Expr x_size, Expr y_size, Expr z_size, GPUAPI gpu_api = GPU_Default); - - EXPORT Stage &allow_race_conditions(); - // @} +class Stage +{ + Internal::Schedule schedule; + void set_dim_type(VarOrRVar var, Internal::For::ForType t); + void split(const std::string &old, const std::string &outer, const std::string &inner, Expr factor, bool exact); + std::string stage_name; - // These calls are for legacy compatibility only. - EXPORT Stage &cuda_threads(VarOrRVar thread_x) { - return gpu_threads(thread_x); - } - EXPORT Stage &cuda_threads(VarOrRVar thread_x, VarOrRVar thread_y) { - return gpu_threads(thread_x, thread_y); - } - EXPORT Stage &cuda_threads(VarOrRVar thread_x, VarOrRVar thread_y, VarOrRVar thread_z) { - return gpu_threads(thread_x, thread_y, thread_z); - } +public: + Stage(Internal::Schedule s, const std::string &n) : schedule(s), stage_name(n) { s.touched(); } + + /** Return a string describing the current var list taking into + * account all the splits, reorders, and tiles. */ + EXPORT std::string dump_argument_list() const; + + /** Return the name of this stage, e.g. "f.update(2)" */ + EXPORT const std::string &name() const; + + /** Scheduling calls that control how the domain of this stage is + * traversed. See the documentation for Func for the meanings. */ + // @{ + + EXPORT Stage &split(VarOrRVar old, VarOrRVar outer, VarOrRVar inner, Expr factor); + EXPORT Stage &fuse(VarOrRVar inner, VarOrRVar outer, VarOrRVar fused); + EXPORT Stage &serial(VarOrRVar var); + EXPORT Stage ¶llel(VarOrRVar var); + EXPORT Stage &vectorize(VarOrRVar var); + EXPORT Stage &unroll(VarOrRVar var); + EXPORT Stage ¶llel(VarOrRVar var, Expr task_size); + EXPORT Stage &vectorize(VarOrRVar var, int factor); + EXPORT Stage &unroll(VarOrRVar var, int factor); + EXPORT Stage & + tile(VarOrRVar x, VarOrRVar y, VarOrRVar xo, VarOrRVar yo, VarOrRVar xi, VarOrRVar yi, Expr xfactor, Expr yfactor); + EXPORT Stage &tile(VarOrRVar x, VarOrRVar y, VarOrRVar xi, VarOrRVar yi, Expr xfactor, Expr yfactor); + EXPORT Stage &reorder(const std::vector &vars); + EXPORT Stage &reorder(VarOrRVar x, VarOrRVar y); + EXPORT Stage &reorder(VarOrRVar x, VarOrRVar y, VarOrRVar z); + EXPORT Stage &reorder(VarOrRVar x, VarOrRVar y, VarOrRVar z, VarOrRVar w); + EXPORT Stage &reorder(VarOrRVar x, VarOrRVar y, VarOrRVar z, VarOrRVar w, VarOrRVar t); + EXPORT Stage &reorder(VarOrRVar x, VarOrRVar y, VarOrRVar z, VarOrRVar w, VarOrRVar t1, VarOrRVar t2); + EXPORT Stage &reorder(VarOrRVar x, VarOrRVar y, VarOrRVar z, VarOrRVar w, VarOrRVar t1, VarOrRVar t2, VarOrRVar t3); + EXPORT Stage & + reorder(VarOrRVar x, VarOrRVar y, VarOrRVar z, VarOrRVar w, VarOrRVar t1, VarOrRVar t2, VarOrRVar t3, VarOrRVar t4); + EXPORT Stage &reorder(VarOrRVar x, + VarOrRVar y, + VarOrRVar z, + VarOrRVar w, + VarOrRVar t1, + VarOrRVar t2, + VarOrRVar t3, + VarOrRVar t4, + VarOrRVar t5); + EXPORT Stage &reorder(VarOrRVar x, + VarOrRVar y, + VarOrRVar z, + VarOrRVar w, + VarOrRVar t1, + VarOrRVar t2, + VarOrRVar t3, + VarOrRVar t4, + VarOrRVar t5, + VarOrRVar t6); + EXPORT Stage &rename(VarOrRVar old_name, VarOrRVar new_name); + EXPORT Stage specialize(Expr condition); + + EXPORT Stage &gpu_threads(VarOrRVar thread_x, GPUAPI gpu_api = GPU_Default); + EXPORT Stage &gpu_threads(VarOrRVar thread_x, VarOrRVar thread_y, GPUAPI gpu_api = GPU_Default); + EXPORT Stage &gpu_threads(VarOrRVar thread_x, VarOrRVar thread_y, VarOrRVar thread_z, GPUAPI gpu_api = GPU_Default); + EXPORT Stage &gpu_single_thread(GPUAPI gpu_api = GPU_Default); + + EXPORT Stage &gpu_blocks(VarOrRVar block_x, GPUAPI gpu_api = GPU_Default); + EXPORT Stage &gpu_blocks(VarOrRVar block_x, VarOrRVar block_y, GPUAPI gpu_api = GPU_Default); + EXPORT Stage &gpu_blocks(VarOrRVar block_x, VarOrRVar block_y, VarOrRVar block_z, GPUAPI gpu_api = GPU_Default); + + EXPORT Stage &gpu(VarOrRVar block_x, VarOrRVar thread_x, GPUAPI gpu_api = GPU_Default); + EXPORT Stage & + gpu(VarOrRVar block_x, VarOrRVar block_y, VarOrRVar thread_x, VarOrRVar thread_y, GPUAPI gpu_api = GPU_Default); + EXPORT Stage &gpu(VarOrRVar block_x, + VarOrRVar block_y, + VarOrRVar block_z, + VarOrRVar thread_x, + VarOrRVar thread_y, + VarOrRVar thread_z, + GPUAPI gpu_api = GPU_Default); + EXPORT Stage &gpu_tile(VarOrRVar x, Expr x_size, GPUAPI gpu_api = GPU_Default); + EXPORT Stage &gpu_tile(VarOrRVar x, VarOrRVar y, Expr x_size, Expr y_size, GPUAPI gpu_api = GPU_Default); + EXPORT Stage &gpu_tile(VarOrRVar x, + VarOrRVar y, + VarOrRVar z, + Expr x_size, + Expr y_size, + Expr z_size, + GPUAPI gpu_api = GPU_Default); + + EXPORT Stage &allow_race_conditions(); + // @} + + // These calls are for legacy compatibility only. + EXPORT Stage &cuda_threads(VarOrRVar thread_x) { return gpu_threads(thread_x); } + EXPORT Stage &cuda_threads(VarOrRVar thread_x, VarOrRVar thread_y) { return gpu_threads(thread_x, thread_y); } + EXPORT Stage &cuda_threads(VarOrRVar thread_x, VarOrRVar thread_y, VarOrRVar thread_z) + { + return gpu_threads(thread_x, thread_y, thread_z); + } - EXPORT Stage &cuda_blocks(VarOrRVar block_x) { - return gpu_blocks(block_x); - } - EXPORT Stage &cuda_blocks(VarOrRVar block_x, VarOrRVar block_y) { - return gpu_blocks(block_x, block_y); - } - EXPORT Stage &cuda_blocks(VarOrRVar block_x, VarOrRVar block_y, VarOrRVar block_z) { - return gpu_blocks(block_x, block_y, block_z); - } + EXPORT Stage &cuda_blocks(VarOrRVar block_x) { return gpu_blocks(block_x); } + EXPORT Stage &cuda_blocks(VarOrRVar block_x, VarOrRVar block_y) { return gpu_blocks(block_x, block_y); } + EXPORT Stage &cuda_blocks(VarOrRVar block_x, VarOrRVar block_y, VarOrRVar block_z) + { + return gpu_blocks(block_x, block_y, block_z); + } - EXPORT Stage &cuda(VarOrRVar block_x, VarOrRVar thread_x) { - return gpu(block_x, thread_x); - } - EXPORT Stage &cuda(VarOrRVar block_x, VarOrRVar block_y, - VarOrRVar thread_x, VarOrRVar thread_y) { - return gpu(block_x, thread_x, block_y, thread_y); - } - EXPORT Stage &cuda(VarOrRVar block_x, VarOrRVar block_y, VarOrRVar block_z, - VarOrRVar thread_x, VarOrRVar thread_y, VarOrRVar thread_z) { - return gpu(block_x, thread_x, block_y, thread_y, block_z, thread_z); - } - EXPORT Stage &cuda_tile(VarOrRVar x, int x_size) { - return gpu_tile(x, x_size); - } - EXPORT Stage &cuda_tile(VarOrRVar x, VarOrRVar y, int x_size, int y_size) { - return gpu_tile(x, y, x_size, y_size); - } - EXPORT Stage &cuda_tile(VarOrRVar x, VarOrRVar y, VarOrRVar z, - int x_size, int y_size, int z_size) { - return gpu_tile(x, y, z, x_size, y_size, z_size); - } + EXPORT Stage &cuda(VarOrRVar block_x, VarOrRVar thread_x) { return gpu(block_x, thread_x); } + EXPORT Stage &cuda(VarOrRVar block_x, VarOrRVar block_y, VarOrRVar thread_x, VarOrRVar thread_y) + { + return gpu(block_x, thread_x, block_y, thread_y); + } + EXPORT Stage &cuda(VarOrRVar block_x, + VarOrRVar block_y, + VarOrRVar block_z, + VarOrRVar thread_x, + VarOrRVar thread_y, + VarOrRVar thread_z) + { + return gpu(block_x, thread_x, block_y, thread_y, block_z, thread_z); + } + EXPORT Stage &cuda_tile(VarOrRVar x, int x_size) { return gpu_tile(x, x_size); } + EXPORT Stage &cuda_tile(VarOrRVar x, VarOrRVar y, int x_size, int y_size) { return gpu_tile(x, y, x_size, y_size); } + EXPORT Stage &cuda_tile(VarOrRVar x, VarOrRVar y, VarOrRVar z, int x_size, int y_size, int z_size) + { + return gpu_tile(x, y, z, x_size, y_size, z_size); + } }; // For backwards compatibility, keep the ScheduleHandle name. @@ -8154,1600 +8109,1602 @@ typedef Stage ScheduleHandle; */ class FuncRefExpr; -class FuncRefVar { - Internal::Function func; - int implicit_placeholder_pos; - std::vector args; - std::vector args_with_implicit_vars(const std::vector &e) const; -public: - FuncRefVar(Internal::Function, const std::vector &, int placeholder_pos = -1); - - /** Use this as the left-hand-side of a definition. */ - EXPORT Stage operator=(Expr); - - /** Use this as the left-hand-side of a definition for a Func with - * multiple outputs. */ - EXPORT Stage operator=(const Tuple &); - - /** Define this function as a sum reduction over the given - * expression. The expression should refer to some RDom to sum - * over. If the function does not already have a pure definition, - * this sets it to zero. - */ - EXPORT Stage operator+=(Expr); - - /** Define this function as a sum reduction over the negative of - * the given expression. The expression should refer to some RDom - * to sum over. If the function does not already have a pure - * definition, this sets it to zero. - */ - EXPORT Stage operator-=(Expr); - - /** Define this function as a product reduction. The expression - * should refer to some RDom to take the product over. If the - * function does not already have a pure definition, this sets it - * to 1. - */ - EXPORT Stage operator*=(Expr); - - /** Define this function as the product reduction over the inverse - * of the expression. The expression should refer to some RDom to - * take the product over. If the function does not already have a - * pure definition, this sets it to 1. - */ - EXPORT Stage operator/=(Expr); - - /** Override the usual assignment operator, so that - * f(x, y) = g(x, y) defines f. - */ - // @{ - EXPORT Stage operator=(const FuncRefVar &e); - EXPORT Stage operator=(const FuncRefExpr &e); - // @} - - /** Use this FuncRefVar as a call to the function, and not as the - * left-hand-side of a definition. Only works for single-output - * funcs. - */ - EXPORT operator Expr() const; - - /** When a FuncRefVar refers to a function that provides multiple - * outputs, you can access each output as an Expr using - * operator[] */ - EXPORT Expr operator[](int) const; - - /** How many outputs does the function this refers to produce. */ - EXPORT size_t size() const; - - /** What function is this calling? */ - EXPORT Internal::Function function() const {return func;} -}; +class FuncRefVar +{ + Internal::Function func; + int implicit_placeholder_pos; + std::vector args; + std::vector args_with_implicit_vars(const std::vector &e) const; -/** A fragment of front-end syntax of the form f(x, y, z), where x, y, - * z are Exprs. If could be the left hand side of an update - * definition, or it could be a call to a function. We don't know - * until we see how this object gets used. - */ -class FuncRefExpr { - Internal::Function func; - int implicit_placeholder_pos; - std::vector args; - std::vector args_with_implicit_vars(const std::vector &e) const; public: - FuncRefExpr(Internal::Function, const std::vector &, - int placeholder_pos = -1); - FuncRefExpr(Internal::Function, const std::vector &, - int placeholder_pos = -1); - - /** Use this as the left-hand-side of an update definition (see - * \ref RDom). The function must already have a pure definition. - */ - EXPORT Stage operator=(Expr); - - /** Use this as the left-hand-side of an update definition for a - * Func with multiple outputs. */ - EXPORT Stage operator=(const Tuple &); - - /** Define this function as a sum reduction over the negative of - * the given expression. The expression should refer to some RDom - * to sum over. If the function does not already have a pure - * definition, this sets it to zero. - */ - EXPORT Stage operator+=(Expr); - - /** Define this function as a sum reduction over the given - * expression. The expression should refer to some RDom to sum - * over. If the function does not already have a pure definition, - * this sets it to zero. - */ - EXPORT Stage operator-=(Expr); - - /** Define this function as a product reduction. The expression - * should refer to some RDom to take the product over. If the - * function does not already have a pure definition, this sets it - * to 1. - */ - EXPORT Stage operator*=(Expr); - - /** Define this function as the product reduction over the inverse - * of the expression. The expression should refer to some RDom to - * take the product over. If the function does not already have a - * pure definition, this sets it to 1. - */ - EXPORT Stage operator/=(Expr); - - /* Override the usual assignment operator, so that - * f(x, y) = g(x, y) defines f. - */ - // @{ - EXPORT Stage operator=(const FuncRefVar &); - EXPORT Stage operator=(const FuncRefExpr &); - // @} - - /** Use this as a call to the function, and not the left-hand-side - * of a definition. Only works for single-output Funcs. */ - EXPORT operator Expr() const; - - /** When a FuncRefExpr refers to a function that provides multiple - * outputs, you can access each output as an Expr using - * operator[]. - */ - EXPORT Expr operator[](int) const; - - /** How many outputs does the function this refers to produce. */ - EXPORT size_t size() const; - - /** What function is this calling? */ - EXPORT Internal::Function function() const {return func;} -}; - -/** - * Used to determine if the output printed to file should be as a normal string - * or as an HTML file which can be opened in a browerser and manipulated via JS and CSS.*/ -enum StmtOutputFormat { - Text, - HTML + FuncRefVar(Internal::Function, const std::vector &, int placeholder_pos = -1); + + /** Use this as the left-hand-side of a definition. */ + EXPORT Stage operator=(Expr); + + /** Use this as the left-hand-side of a definition for a Func with + * multiple outputs. */ + EXPORT Stage operator=(const Tuple &); + + /** Define this function as a sum reduction over the given + * expression. The expression should refer to some RDom to sum + * over. If the function does not already have a pure definition, + * this sets it to zero. + */ + EXPORT Stage operator+=(Expr); + + /** Define this function as a sum reduction over the negative of + * the given expression. The expression should refer to some RDom + * to sum over. If the function does not already have a pure + * definition, this sets it to zero. + */ + EXPORT Stage operator-=(Expr); + + /** Define this function as a product reduction. The expression + * should refer to some RDom to take the product over. If the + * function does not already have a pure definition, this sets it + * to 1. + */ + EXPORT Stage operator*=(Expr); + + /** Define this function as the product reduction over the inverse + * of the expression. The expression should refer to some RDom to + * take the product over. If the function does not already have a + * pure definition, this sets it to 1. + */ + EXPORT Stage operator/=(Expr); + + /** Override the usual assignment operator, so that + * f(x, y) = g(x, y) defines f. + */ + // @{ + EXPORT Stage operator=(const FuncRefVar &e); + EXPORT Stage operator=(const FuncRefExpr &e); + // @} + + /** Use this FuncRefVar as a call to the function, and not as the + * left-hand-side of a definition. Only works for single-output + * funcs. + */ + EXPORT operator Expr() const; + + /** When a FuncRefVar refers to a function that provides multiple + * outputs, you can access each output as an Expr using + * operator[] */ + EXPORT Expr operator[](int) const; + + /** How many outputs does the function this refers to produce. */ + EXPORT size_t size() const; + + /** What function is this calling? */ + EXPORT Internal::Function function() const { return func; } }; -/** A halide function. This class represents one stage in a Halide - * pipeline, and is the unit by which we schedule things. By default - * they are aggressively inlined, so you are encouraged to make lots - * of little functions, rather than storing things in Exprs. */ -class Func { - - /** A handle on the internal halide function that this - * represents */ - Internal::Function func; - - /** When you make a reference to this function with fewer - * arguments than it has dimensions, the argument list is bulked - * up with 'implicit' vars with canonical names. This lets you - * pass around partially applied Halide functions. */ - // @{ - int add_implicit_vars(std::vector &) const; - int add_implicit_vars(std::vector &) const; - // @} - - /** The lowered imperative form of this function and the target - * this was lowered for. Cached here so that recompilation doesn't - * necessarily require re-lowering */ - // @{ - Internal::Stmt lowered; - Target lowered_target; - // @} - - /** Lower the func if it hasn't been already. */ - void lower(const Target &t); - - /** A JIT-compiled version of this function that we save so that - * we don't have to rejit every time we want to evaluated it. */ - Internal::JITCompiledModule compiled_module; - - /** Invalidate the cached lowered stmt and compiled module. */ - void invalidate_cache(); - - /** The current error handler used for realizing this - * function. May be NULL. Only relevant when jitting. */ - void (*error_handler)(void *user_context, const char *); - - /** The current custom allocator used for realizing this - * function. May be NULL. Only relevant when jitting. */ - // @{ - void *(*custom_malloc)(void *user_context, size_t); - void (*custom_free)(void *user_context, void *ptr); - // @} - - /** The current custom parallel task launcher and handler for - * realizing this function. May be NULL. */ - // @{ - int (*custom_do_par_for)(void *user_context, - int (*)(void *, int, uint8_t *), - int, int, uint8_t *); - int (*custom_do_task)(void *user_context, int (*)(void *, int, uint8_t *), - int, uint8_t *); - // @} - - /** The current custom tracing function. May be NULL. */ - // @{ - int32_t (*custom_trace)(void *, const halide_trace_event *); - - // @} - - /** The current print function used for realizing this - * function. May be NULL. Only relevant when jitting. */ - void (*custom_print)(void *user_context, const char *); - - uint64_t cache_size; - - /** The random seed to use for realizations of this function. */ - uint32_t random_seed; - - /** Pointers to current values of the automatically inferred - * arguments (buffers and scalars) used to realize this - * function. Only relevant when jitting. We can hold these things - * with raw pointers instead of reference-counted handles, because - * func indirectly holds onto them with reference-counted handles - * via its value Expr. */ - std::vector arg_values; - - /** Some of the arg_values need to be rebound on every call if the - * image params change. The pointers for the scalar params will - * still be valid though. */ - std::vector > image_param_args; - - /** A context to use for JIT-realizations of this Func. */ - Param user_context; - - // Some infrastructure that helps Funcs catch and handle runtime errors in JIT-compiled code. - bool prepare_to_catch_runtime_errors(void *buf); - -public: - - EXPORT static void test(); - - /** Declare a new undefined function with the given name */ - EXPORT explicit Func(const std::string &name); - - /** Declare a new undefined function with an - * automatically-generated unique name */ - EXPORT Func(); - - /** Declare a new function with an automatically-generated unique - * name, and define it to return the given expression (which may - * not contain free variables). */ - EXPORT explicit Func(Expr e); - - /** Construct a new Func to wrap an existing, already-define - * Function object. */ - EXPORT explicit Func(Internal::Function f); - - /** Evaluate this function over some rectangular domain and return - * the resulting buffer or buffers. Performs compilation if the - * Func has not previously been realized and jit_compile has not - * been called. The returned Buffer should probably be instantly - * wrapped in an Image class of the appropriate type. That is, do - * this: - * - \code - f(x) = sin(x); - Image im = f.realize(...); - \endcode - * - * not this: - * - \code - f(x) = sin(x) - Buffer im = f.realize(...) - \endcode - * - * If your Func has multiple values, because you defined it using - * a Tuple, then casting the result of a realize call to a buffer - * or image will produce a run-time error. Instead you should do the - * following: - * - \code - f(x) = Tuple(x, sin(x)); - Realization r = f.realize(...); - Image im0 = r[0]; - Image im1 = r[1]; - \endcode - * - */ - // @{ - EXPORT Realization realize(std::vector sizes, const Target &target = get_jit_target_from_environment()); - EXPORT Realization realize(int x_size, int y_size, int z_size, int w_size, - const Target &target = get_jit_target_from_environment()); - EXPORT Realization realize(int x_size, int y_size, int z_size, - const Target &target = get_jit_target_from_environment()); - EXPORT Realization realize(int x_size, int y_size, - const Target &target = get_jit_target_from_environment()); - EXPORT Realization realize(int x_size = 0, - const Target &target = get_jit_target_from_environment()); - // @} - - /** Evaluate this function into an existing allocated buffer or - * buffers. If the buffer is also one of the arguments to the - * function, strange things may happen, as the pipeline isn't - * necessarily safe to run in-place. If you pass multiple buffers, - * they must have matching sizes. */ - // @{ - EXPORT void realize(Realization dst, const Target &target = get_jit_target_from_environment()); - EXPORT void realize(Buffer dst, const Target &target = get_jit_target_from_environment()); - - template - void realize(Image dst, const Target &target = get_jit_target_from_environment()) { - // Images are expected to exist on-host. - realize(Buffer(dst), target); - dst.copy_to_host(); - } - // @} - - /** For a given size of output, or a given output buffer, - * determine the bounds required of all unbound ImageParams - * referenced. Communicates the result by allocating new buffers - * of the appropriate size and binding them to the unbound - * ImageParams. */ - // @{ - EXPORT void infer_input_bounds(int x_size = 0, int y_size = 0, int z_size = 0, int w_size = 0); - EXPORT void infer_input_bounds(Realization dst); - EXPORT void infer_input_bounds(Buffer dst); - // @} - - /** Statically compile this function to llvm bitcode, with the - * given filename (which should probably end in .bc), type - * signature, and C function name (which defaults to the same name - * as this halide function */ - //@{ - EXPORT void compile_to_bitcode(const std::string &filename, std::vector, const std::string &fn_name, - const Target &target = get_target_from_environment()); - EXPORT void compile_to_bitcode(const std::string &filename, std::vector, - const Target &target = get_target_from_environment()); - // @} - - /** Statically compile this function to an object file, with the - * given filename (which should probably end in .o or .obj), type - * signature, and C function name (which defaults to the same name - * as this halide function. You probably don't want to use this - * directly; call compile_to_file instead. */ - //@{ - EXPORT void compile_to_object(const std::string &filename, std::vector, const std::string &fn_name, - const Target &target = get_target_from_environment()); - EXPORT void compile_to_object(const std::string &filename, std::vector, - const Target &target = get_target_from_environment()); - // @} - - /** Emit a header file with the given filename for this - * function. The header will define a function with the type - * signature given by the second argument, and a name given by the - * third. The name defaults to the same name as this halide - * function. You don't actually have to have defined this function - * yet to call this. You probably don't want to use this directly; - * call compile_to_file instead. */ - EXPORT void compile_to_header(const std::string &filename, std::vector, const std::string &fn_name = ""); - - /** Statically compile this function to text assembly equivalent - * to the object file generated by compile_to_object. This is - * useful for checking what Halide is producing without having to - * disassemble anything, or if you need to feed the assembly into - * some custom toolchain to produce an object file (e.g. iOS) */ - //@{ - EXPORT void compile_to_assembly(const std::string &filename, std::vector, const std::string &fn_name, - const Target &target = get_target_from_environment()); - EXPORT void compile_to_assembly(const std::string &filename, std::vector, - const Target &target = get_target_from_environment()); - // @} - /** Statically compile this function to C source code. This is - * useful for providing fallback code paths that will compile on - * many platforms. Vectorization will fail, and parallelization - * will produce serial code. */ - EXPORT void compile_to_c(const std::string &filename, - std::vector, - const std::string &fn_name = "", - const Target &target = get_target_from_environment()); - - /** Write out an internal representation of lowered code. Useful - * for analyzing and debugging scheduling. Can emit html or plain - * text. */ - EXPORT void compile_to_lowered_stmt(const std::string &filename, - StmtOutputFormat fmt = Text, - const Target &target = get_target_from_environment()); - - /** Write out an internal representation of lowered code as above - * but simplified using the provided realization bounds and other - * concrete parameter values. Can emit html or plain text. */ - //@{ - EXPORT void compile_to_simplified_lowered_stmt(const std::string &filename, - Realization dst, - const std::map &additional_replacements, - StmtOutputFormat fmt = Text, - const Target &t = get_target_from_environment()); - - EXPORT void compile_to_simplified_lowered_stmt(const std::string &filename, - Realization dst, - StmtOutputFormat fmt = Text, - const Target &t = get_target_from_environment()); - - EXPORT void compile_to_simplified_lowered_stmt(const std::string &filename, - Buffer dst, - const std::map &additional_replacements, - StmtOutputFormat fmt = Text, - const Target &target = get_target_from_environment()); - - EXPORT void compile_to_simplified_lowered_stmt(const std::string &filename, - Buffer dst, - StmtOutputFormat fmt = Text, - const Target &target = get_target_from_environment()); - - EXPORT void compile_to_simplified_lowered_stmt(const std::string &filename, - int x_size, int y_size, int z_size, int w_size, - const std::map &additional_replacements, - StmtOutputFormat fmt = Text, - const Target &t = get_target_from_environment()); - - EXPORT void compile_to_simplified_lowered_stmt(const std::string &filename, - int x_size, int y_size, int z_size, int w_size, - StmtOutputFormat fmt = Text, - const Target &t = get_target_from_environment()); - - EXPORT void compile_to_simplified_lowered_stmt(const std::string &filename, - int x_size, int y_size, int z_size, - const std::map &additional_replacements, - StmtOutputFormat fmt = Text, - const Target &t = get_target_from_environment()); - - EXPORT void compile_to_simplified_lowered_stmt(const std::string &filename, - int x_size, int y_size, int z_size, - StmtOutputFormat fmt = Text, - const Target &t = get_target_from_environment()); - - EXPORT void compile_to_simplified_lowered_stmt(const std::string &filename, - int x_size, int y_size, - const std::map &additional_replacements, - StmtOutputFormat fmt = Text, - const Target &t = get_target_from_environment()); - - EXPORT void compile_to_simplified_lowered_stmt(const std::string &filename, - int x_size, int y_size, - StmtOutputFormat fmt = Text, - const Target &t = get_target_from_environment()); - - EXPORT void compile_to_simplified_lowered_stmt(const std::string &filename, - int x_size, - const std::map &additional_replacements, - StmtOutputFormat fmt = Text, - const Target &t = get_target_from_environment()); - - EXPORT void compile_to_simplified_lowered_stmt(const std::string &filename, - int x_size, - StmtOutputFormat fmt = Text, - const Target &t = get_target_from_environment()); - - // @} - - /** Compile to object file and header pair, with the given - * arguments. Also names the C function to match the first - * argument. - */ - //@{ - EXPORT void compile_to_file(const std::string &filename_prefix, std::vector args, - const Target &target = get_target_from_environment()); - EXPORT void compile_to_file(const std::string &filename_prefix, - const Target &target = get_target_from_environment()); - EXPORT void compile_to_file(const std::string &filename_prefix, Argument a, - const Target &target = get_target_from_environment()); - EXPORT void compile_to_file(const std::string &filename_prefix, Argument a, Argument b, - const Target &target = get_target_from_environment()); - EXPORT void compile_to_file(const std::string &filename_prefix, Argument a, Argument b, Argument c, - const Target &target = get_target_from_environment()); - EXPORT void compile_to_file(const std::string &filename_prefix, Argument a, Argument b, Argument c, Argument d, - const Target &target = get_target_from_environment()); - EXPORT void compile_to_file(const std::string &filename_prefix, Argument a, Argument b, Argument c, Argument d, Argument e, - const Target &target = get_target_from_environment()); - // @} - - /** Eagerly jit compile the function to machine code. This - * normally happens on the first call to realize. If you're - * running your halide pipeline inside time-sensitive code and - * wish to avoid including the time taken to compile a pipeline, - * then you can call this ahead of time. Returns the raw function - * pointer to the compiled pipeline. Default is to use the Target - * returned from Halide::get_jit_target_from_environment() - */ - EXPORT void *compile_jit(const Target &target = get_jit_target_from_environment()); - - /** Set the error handler function that be called in the case of - * runtime errors during halide pipelines. If you are compiling - * statically, you can also just define your own function with - * signature - \code - extern "C" void halide_error(void *user_context, const char *); - \endcode - * This will clobber Halide's version. - */ - EXPORT void set_error_handler(void (*handler)(void *, const char *)); - - /** Set a custom malloc and free for halide to use. Malloc should - * return 32-byte aligned chunks of memory, and it should be safe - * for Halide to read slightly out of bounds (up to 8 bytes before - * the start or beyond the end). If compiling statically, routines - * with appropriate signatures can be provided directly - \code - extern "C" void *halide_malloc(void *, size_t) - extern "C" void halide_free(void *, void *) - \endcode - * These will clobber Halide's versions. See \file HalideRuntime.h - * for declarations. - */ - EXPORT void set_custom_allocator(void *(*malloc)(void *, size_t), - void (*free)(void *, void *)); - - /** Set a custom task handler to be called by the parallel for - * loop. It is useful to set this if you want to do some - * additional bookkeeping at the granularity of parallel - * tasks. The default implementation does this: - \code - extern "C" int halide_do_task(void *user_context, - int (*f)(void *, int, uint8_t *), - int idx, uint8_t *state) { - return f(user_context, idx, state); - } - \endcode - * If you are statically compiling, you can also just define your - * own version of the above function, and it will clobber Halide's - * version. - * - * If you're trying to use a custom parallel runtime, you probably - * don't want to call this. See instead \ref Func::set_custom_do_par_for . - */ - EXPORT void set_custom_do_task( - int (*custom_do_task)(void *, int (*)(void *, int, uint8_t *), - int, uint8_t *)); - - /** Set a custom parallel for loop launcher. Useful if your app - * already manages a thread pool. The default implementation is - * equivalent to this: - \code - extern "C" int halide_do_par_for(void *user_context, - int (*f)(void *, int, uint8_t *), - int min, int extent, uint8_t *state) { - int exit_status = 0; - parallel for (int idx = min; idx < min+extent; idx++) { - int job_status = halide_do_task(user_context, f, idx, state); - if (job_status) exit_status = job_status; - } - return exit_status; - } - \endcode - * - * However, notwithstanding the above example code, if one task - * fails, we may skip over other tasks, and if two tasks return - * different error codes, we may select one arbitrarily to return. - * - * If you are statically compiling, you can also just define your - * own version of the above function, and it will clobber Halide's - * version. - */ - EXPORT void set_custom_do_par_for( - int (*custom_do_par_for)(void *, int (*)(void *, int, uint8_t *), int, - int, uint8_t *)); - - /** Set custom routines to call when tracing is enabled. Call this - * on the output Func of your pipeline. This then sets custom - * routines for the entire pipeline, not just calls to this - * Func. - * - * If you are statically compiling, you can also just define your - * own versions of the tracing functions (see HalideRuntime.h), - * and they will clobber Halide's versions. */ - EXPORT void set_custom_trace(Internal::JITCompiledModule::TraceFn); - - /** Set the function called to print messages from the runtime. - * If you are compiling statically, you can also just define your - * own function with signature - \code - extern "C" void halide_print(void *user_context, const char *); - \endcode - * This will clobber Halide's version. - */ - EXPORT void set_custom_print(void (*handler)(void *, const char *)); - - /** Set the maximum number of bytes used by memoization caching. - * If you are compiling statically, you should include HalideRuntime.h - * and call halide_memoization_cache_set_size() instead. - */ - EXPORT void memoization_cache_set_size(uint64_t size); - - /** When this function is compiled, include code that dumps its - * values to a file after it is realized, for the purpose of - * debugging. - * - * If filename ends in ".tif" or ".tiff" (case insensitive) the file - * is in TIFF format and can be read by standard tools. Oherwise, the - * file format is as follows: - * - * All data is in the byte-order of the target platform. First, a - * 20 byte-header containing four 32-bit ints, giving the extents - * of the first four dimensions. Dimensions beyond four are - * folded into the fourth. Then, a fifth 32-bit int giving the - * data type of the function. The typecodes are given by: float = - * 0, double = 1, uint8_t = 2, int8_t = 3, uint16_t = 4, int16_t = - * 5, uint32_t = 6, int32_t = 7, uint64_t = 8, int64_t = 9. The - * data follows the header, as a densely packed array of the given - * size and the given type. If given the extension .tmp, this file - * format can be natively read by the program ImageStack. */ - EXPORT void debug_to_file(const std::string &filename); - - /** The name of this function, either given during construction, - * or automatically generated. */ - EXPORT const std::string &name() const; - - /** Get the pure arguments. */ - EXPORT std::vector args() const; - - /** The right-hand-side value of the pure definition of this - * function. Causes an error if there's no pure definition, or if - * the function is defined to return multiple values. */ - EXPORT Expr value() const; - - /** The values returned by this function. An error if the function - * has not been been defined. Returns a Tuple with one element for - * functions defined to return a single value. */ - EXPORT Tuple values() const; - - /** Does this function have at least a pure definition. */ - EXPORT bool defined() const; - - /** Get the left-hand-side of the update definition. An empty - * vector if there's no update definition. If there are - * multiple update definitions for this function, use the - * argument to select which one you want. */ - EXPORT const std::vector &update_args(int idx = 0) const; - - /** Get the right-hand-side of an update definition. An error if - * there's no update definition. If there are multiple - * update definitions for this function, use the argument to - * select which one you want. */ - EXPORT Expr update_value(int idx = 0) const; - - /** Get the right-hand-side of an update definition for - * functions that returns multiple values. An error if there's no - * update definition. Returns a Tuple with one element for - * functions that return a single value. */ - EXPORT Tuple update_values(int idx = 0) const; - - /** Get the reduction domain for an update definition, if there is - * one. */ - EXPORT RDom reduction_domain(int idx = 0) const; - - /** Does this function have at least one update definition? */ - EXPORT bool has_update_definition() const; - - /** How many update definitions does this function have? */ - EXPORT int num_update_definitions() const; - - /** Is this function an external stage? That is, was it defined - * using define_extern? */ - EXPORT bool is_extern() const; - - /** Add an extern definition for this Func. This lets you define a - * Func that represents an external pipeline stage. You can, for - * example, use it to wrap a call to an extern library such as - * fftw. */ - // @{ - EXPORT void define_extern(const std::string &function_name, - const std::vector ¶ms, - Type t, - int dimensionality) { - define_extern(function_name, params, Internal::vec(t), dimensionality); - } - - EXPORT void define_extern(const std::string &function_name, - const std::vector ¶ms, - const std::vector &types, - int dimensionality); - // @} +/** A fragment of front-end syntax of the form f(x, y, z), where x, y, + * z are Exprs. If could be the left hand side of an update + * definition, or it could be a call to a function. We don't know + * until we see how this object gets used. + */ +class FuncRefExpr +{ + Internal::Function func; + int implicit_placeholder_pos; + std::vector args; + std::vector args_with_implicit_vars(const std::vector &e) const; - /** Get the types of the outputs of this Func. */ - EXPORT const std::vector &output_types() const; +public: + FuncRefExpr(Internal::Function, const std::vector &, int placeholder_pos = -1); + FuncRefExpr(Internal::Function, const std::vector &, int placeholder_pos = -1); + + /** Use this as the left-hand-side of an update definition (see + * \ref RDom). The function must already have a pure definition. + */ + EXPORT Stage operator=(Expr); + + /** Use this as the left-hand-side of an update definition for a + * Func with multiple outputs. */ + EXPORT Stage operator=(const Tuple &); + + /** Define this function as a sum reduction over the negative of + * the given expression. The expression should refer to some RDom + * to sum over. If the function does not already have a pure + * definition, this sets it to zero. + */ + EXPORT Stage operator+=(Expr); + + /** Define this function as a sum reduction over the given + * expression. The expression should refer to some RDom to sum + * over. If the function does not already have a pure definition, + * this sets it to zero. + */ + EXPORT Stage operator-=(Expr); + + /** Define this function as a product reduction. The expression + * should refer to some RDom to take the product over. If the + * function does not already have a pure definition, this sets it + * to 1. + */ + EXPORT Stage operator*=(Expr); + + /** Define this function as the product reduction over the inverse + * of the expression. The expression should refer to some RDom to + * take the product over. If the function does not already have a + * pure definition, this sets it to 1. + */ + EXPORT Stage operator/=(Expr); + + /* Override the usual assignment operator, so that + * f(x, y) = g(x, y) defines f. + */ + // @{ + EXPORT Stage operator=(const FuncRefVar &); + EXPORT Stage operator=(const FuncRefExpr &); + // @} + + /** Use this as a call to the function, and not the left-hand-side + * of a definition. Only works for single-output Funcs. */ + EXPORT operator Expr() const; + + /** When a FuncRefExpr refers to a function that provides multiple + * outputs, you can access each output as an Expr using + * operator[]. + */ + EXPORT Expr operator[](int) const; + + /** How many outputs does the function this refers to produce. */ + EXPORT size_t size() const; + + /** What function is this calling? */ + EXPORT Internal::Function function() const { return func; } +}; - /** Get the number of outputs of this Func. Corresponds to the - * size of the Tuple this Func was defined to return. */ - EXPORT int outputs() const; +/** + * Used to determine if the output printed to file should be as a normal string + * or as an HTML file which can be opened in a browerser and manipulated via JS and CSS.*/ +enum StmtOutputFormat { Text, HTML }; - /** Get the name of the extern function called for an extern - * definition. */ - EXPORT const std::string &extern_function_name() const; - - /** The dimensionality (number of arguments) of this - * function. Zero if the function is not yet defined. */ - EXPORT int dimensions() const; - - /** Construct either the left-hand-side of a definition, or a call - * to a functions that happens to only contain vars as - * arguments. If the function has already been defined, and fewer - * arguments are given than the function has dimensions, then - * enough implicit vars are added to the end of the argument list - * to make up the difference (see \ref Var::implicit) */ - // @{ - EXPORT FuncRefVar operator()() const; - EXPORT FuncRefVar operator()(Var x) const; - EXPORT FuncRefVar operator()(Var x, Var y) const; - EXPORT FuncRefVar operator()(Var x, Var y, Var z) const; - EXPORT FuncRefVar operator()(Var x, Var y, Var z, Var w) const; - EXPORT FuncRefVar operator()(Var x, Var y, Var z, Var w, Var u) const; - EXPORT FuncRefVar operator()(Var x, Var y, Var z, Var w, Var u, Var v) const; - EXPORT FuncRefVar operator()(std::vector) const; - // @} +/** A halide function. This class represents one stage in a Halide + * pipeline, and is the unit by which we schedule things. By default + * they are aggressively inlined, so you are encouraged to make lots + * of little functions, rather than storing things in Exprs. */ +class Func +{ + + /** A handle on the internal halide function that this + * represents */ + Internal::Function func; + + /** When you make a reference to this function with fewer + * arguments than it has dimensions, the argument list is bulked + * up with 'implicit' vars with canonical names. This lets you + * pass around partially applied Halide functions. */ + // @{ + int add_implicit_vars(std::vector &) const; + int add_implicit_vars(std::vector &) const; + // @} + + /** The lowered imperative form of this function and the target + * this was lowered for. Cached here so that recompilation doesn't + * necessarily require re-lowering */ + // @{ + Internal::Stmt lowered; + Target lowered_target; + // @} + + /** Lower the func if it hasn't been already. */ + void lower(const Target &t); + + /** A JIT-compiled version of this function that we save so that + * we don't have to rejit every time we want to evaluated it. */ + Internal::JITCompiledModule compiled_module; + + /** Invalidate the cached lowered stmt and compiled module. */ + void invalidate_cache(); + + /** The current error handler used for realizing this + * function. May be NULL. Only relevant when jitting. */ + void (*error_handler)(void *user_context, const char *); + + /** The current custom allocator used for realizing this + * function. May be NULL. Only relevant when jitting. */ + // @{ + void *(*custom_malloc)(void *user_context, size_t); + void (*custom_free)(void *user_context, void *ptr); + // @} + + /** The current custom parallel task launcher and handler for + * realizing this function. May be NULL. */ + // @{ + int (*custom_do_par_for)(void *user_context, int (*)(void *, int, uint8_t *), int, int, uint8_t *); + int (*custom_do_task)(void *user_context, int (*)(void *, int, uint8_t *), int, uint8_t *); + // @} + + /** The current custom tracing function. May be NULL. */ + // @{ + int32_t (*custom_trace)(void *, const halide_trace_event *); + + // @} + + /** The current print function used for realizing this + * function. May be NULL. Only relevant when jitting. */ + void (*custom_print)(void *user_context, const char *); + + uint64_t cache_size; + + /** The random seed to use for realizations of this function. */ + uint32_t random_seed; + + /** Pointers to current values of the automatically inferred + * arguments (buffers and scalars) used to realize this + * function. Only relevant when jitting. We can hold these things + * with raw pointers instead of reference-counted handles, because + * func indirectly holds onto them with reference-counted handles + * via its value Expr. */ + std::vector arg_values; + + /** Some of the arg_values need to be rebound on every call if the + * image params change. The pointers for the scalar params will + * still be valid though. */ + std::vector> image_param_args; + + /** A context to use for JIT-realizations of this Func. */ + Param user_context; + + // Some infrastructure that helps Funcs catch and handle runtime errors in JIT-compiled code. + bool prepare_to_catch_runtime_errors(void *buf); - /** Either calls to the function, or the left-hand-side of a - * update definition (see \ref RDom). If the function has - * already been defined, and fewer arguments are given than the - * function has dimensions, then enough implicit vars are added to - * the end of the argument list to make up the difference. (see - * \ref Var::implicit)*/ - // @{ - EXPORT FuncRefExpr operator()(Expr x) const; - EXPORT FuncRefExpr operator()(Expr x, Expr y) const; - EXPORT FuncRefExpr operator()(Expr x, Expr y, Expr z) const; - EXPORT FuncRefExpr operator()(Expr x, Expr y, Expr z, Expr w) const; - EXPORT FuncRefExpr operator()(Expr x, Expr y, Expr z, Expr w, Expr u) const; - EXPORT FuncRefExpr operator()(Expr x, Expr y, Expr z, Expr w, Expr u, Expr v) const; - EXPORT FuncRefExpr operator()(std::vector) const; - // @} +public: + EXPORT static void test(); + + /** Declare a new undefined function with the given name */ + EXPORT explicit Func(const std::string &name); + + /** Declare a new undefined function with an + * automatically-generated unique name */ + EXPORT Func(); + + /** Declare a new function with an automatically-generated unique + * name, and define it to return the given expression (which may + * not contain free variables). */ + EXPORT explicit Func(Expr e); + + /** Construct a new Func to wrap an existing, already-define + * Function object. */ + EXPORT explicit Func(Internal::Function f); + + /** Evaluate this function over some rectangular domain and return + * the resulting buffer or buffers. Performs compilation if the + * Func has not previously been realized and jit_compile has not + * been called. The returned Buffer should probably be instantly + * wrapped in an Image class of the appropriate type. That is, do + * this: + * + \code + f(x) = sin(x); + Image im = f.realize(...); + \endcode + * + * not this: + * + \code + f(x) = sin(x) + Buffer im = f.realize(...) + \endcode + * + * If your Func has multiple values, because you defined it using + * a Tuple, then casting the result of a realize call to a buffer + * or image will produce a run-time error. Instead you should do the + * following: + * + \code + f(x) = Tuple(x, sin(x)); + Realization r = f.realize(...); + Image im0 = r[0]; + Image im1 = r[1]; + \endcode + * + */ + // @{ + EXPORT Realization realize(std::vector sizes, const Target &target = get_jit_target_from_environment()); + EXPORT Realization + realize(int x_size, int y_size, int z_size, int w_size, const Target &target = get_jit_target_from_environment()); + EXPORT Realization realize(int x_size, + int y_size, + int z_size, + const Target &target = get_jit_target_from_environment()); + EXPORT Realization realize(int x_size, int y_size, const Target &target = get_jit_target_from_environment()); + EXPORT Realization realize(int x_size = 0, const Target &target = get_jit_target_from_environment()); + // @} + + /** Evaluate this function into an existing allocated buffer or + * buffers. If the buffer is also one of the arguments to the + * function, strange things may happen, as the pipeline isn't + * necessarily safe to run in-place. If you pass multiple buffers, + * they must have matching sizes. */ + // @{ + EXPORT void realize(Realization dst, const Target &target = get_jit_target_from_environment()); + EXPORT void realize(Buffer dst, const Target &target = get_jit_target_from_environment()); + + template void realize(Image dst, const Target &target = get_jit_target_from_environment()) + { + // Images are expected to exist on-host. + realize(Buffer(dst), target); + dst.copy_to_host(); + } + // @} + + /** For a given size of output, or a given output buffer, + * determine the bounds required of all unbound ImageParams + * referenced. Communicates the result by allocating new buffers + * of the appropriate size and binding them to the unbound + * ImageParams. */ + // @{ + EXPORT void infer_input_bounds(int x_size = 0, int y_size = 0, int z_size = 0, int w_size = 0); + EXPORT void infer_input_bounds(Realization dst); + EXPORT void infer_input_bounds(Buffer dst); + // @} + + /** Statically compile this function to llvm bitcode, with the + * given filename (which should probably end in .bc), type + * signature, and C function name (which defaults to the same name + * as this halide function */ + //@{ + EXPORT void compile_to_bitcode(const std::string &filename, + std::vector, + const std::string &fn_name, + const Target &target = get_target_from_environment()); + EXPORT void compile_to_bitcode(const std::string &filename, + std::vector, + const Target &target = get_target_from_environment()); + // @} + + /** Statically compile this function to an object file, with the + * given filename (which should probably end in .o or .obj), type + * signature, and C function name (which defaults to the same name + * as this halide function. You probably don't want to use this + * directly; call compile_to_file instead. */ + //@{ + EXPORT void compile_to_object(const std::string &filename, + std::vector, + const std::string &fn_name, + const Target &target = get_target_from_environment()); + EXPORT void compile_to_object(const std::string &filename, + std::vector, + const Target &target = get_target_from_environment()); + // @} + + /** Emit a header file with the given filename for this + * function. The header will define a function with the type + * signature given by the second argument, and a name given by the + * third. The name defaults to the same name as this halide + * function. You don't actually have to have defined this function + * yet to call this. You probably don't want to use this directly; + * call compile_to_file instead. */ + EXPORT void compile_to_header(const std::string &filename, std::vector, const std::string &fn_name = ""); + + /** Statically compile this function to text assembly equivalent + * to the object file generated by compile_to_object. This is + * useful for checking what Halide is producing without having to + * disassemble anything, or if you need to feed the assembly into + * some custom toolchain to produce an object file (e.g. iOS) */ + //@{ + EXPORT void compile_to_assembly(const std::string &filename, + std::vector, + const std::string &fn_name, + const Target &target = get_target_from_environment()); + EXPORT void compile_to_assembly(const std::string &filename, + std::vector, + const Target &target = get_target_from_environment()); + // @} + /** Statically compile this function to C source code. This is + * useful for providing fallback code paths that will compile on + * many platforms. Vectorization will fail, and parallelization + * will produce serial code. */ + EXPORT void compile_to_c(const std::string &filename, + std::vector, + const std::string &fn_name = "", + const Target &target = get_target_from_environment()); + + /** Write out an internal representation of lowered code. Useful + * for analyzing and debugging scheduling. Can emit html or plain + * text. */ + EXPORT void compile_to_lowered_stmt(const std::string &filename, + StmtOutputFormat fmt = Text, + const Target &target = get_target_from_environment()); + + /** Write out an internal representation of lowered code as above + * but simplified using the provided realization bounds and other + * concrete parameter values. Can emit html or plain text. */ + //@{ + EXPORT void compile_to_simplified_lowered_stmt(const std::string &filename, + Realization dst, + const std::map &additional_replacements, + StmtOutputFormat fmt = Text, + const Target &t = get_target_from_environment()); + + EXPORT void compile_to_simplified_lowered_stmt(const std::string &filename, + Realization dst, + StmtOutputFormat fmt = Text, + const Target &t = get_target_from_environment()); + + EXPORT void compile_to_simplified_lowered_stmt(const std::string &filename, + Buffer dst, + const std::map &additional_replacements, + StmtOutputFormat fmt = Text, + const Target &target = get_target_from_environment()); + + EXPORT void compile_to_simplified_lowered_stmt(const std::string &filename, + Buffer dst, + StmtOutputFormat fmt = Text, + const Target &target = get_target_from_environment()); + + EXPORT void compile_to_simplified_lowered_stmt(const std::string &filename, + int x_size, + int y_size, + int z_size, + int w_size, + const std::map &additional_replacements, + StmtOutputFormat fmt = Text, + const Target &t = get_target_from_environment()); + + EXPORT void compile_to_simplified_lowered_stmt(const std::string &filename, + int x_size, + int y_size, + int z_size, + int w_size, + StmtOutputFormat fmt = Text, + const Target &t = get_target_from_environment()); + + EXPORT void compile_to_simplified_lowered_stmt(const std::string &filename, + int x_size, + int y_size, + int z_size, + const std::map &additional_replacements, + StmtOutputFormat fmt = Text, + const Target &t = get_target_from_environment()); + + EXPORT void compile_to_simplified_lowered_stmt(const std::string &filename, + int x_size, + int y_size, + int z_size, + StmtOutputFormat fmt = Text, + const Target &t = get_target_from_environment()); + + EXPORT void compile_to_simplified_lowered_stmt(const std::string &filename, + int x_size, + int y_size, + const std::map &additional_replacements, + StmtOutputFormat fmt = Text, + const Target &t = get_target_from_environment()); + + EXPORT void compile_to_simplified_lowered_stmt(const std::string &filename, + int x_size, + int y_size, + StmtOutputFormat fmt = Text, + const Target &t = get_target_from_environment()); + + EXPORT void compile_to_simplified_lowered_stmt(const std::string &filename, + int x_size, + const std::map &additional_replacements, + StmtOutputFormat fmt = Text, + const Target &t = get_target_from_environment()); + + EXPORT void compile_to_simplified_lowered_stmt(const std::string &filename, + int x_size, + StmtOutputFormat fmt = Text, + const Target &t = get_target_from_environment()); + + // @} + + /** Compile to object file and header pair, with the given + * arguments. Also names the C function to match the first + * argument. + */ + //@{ + EXPORT void compile_to_file(const std::string &filename_prefix, + std::vector args, + const Target &target = get_target_from_environment()); + EXPORT void compile_to_file(const std::string &filename_prefix, const Target &target = get_target_from_environment()); + EXPORT void compile_to_file(const std::string &filename_prefix, + Argument a, + const Target &target = get_target_from_environment()); + EXPORT void compile_to_file(const std::string &filename_prefix, + Argument a, + Argument b, + const Target &target = get_target_from_environment()); + EXPORT void compile_to_file(const std::string &filename_prefix, + Argument a, + Argument b, + Argument c, + const Target &target = get_target_from_environment()); + EXPORT void compile_to_file(const std::string &filename_prefix, + Argument a, + Argument b, + Argument c, + Argument d, + const Target &target = get_target_from_environment()); + EXPORT void compile_to_file(const std::string &filename_prefix, + Argument a, + Argument b, + Argument c, + Argument d, + Argument e, + const Target &target = get_target_from_environment()); + // @} + + /** Eagerly jit compile the function to machine code. This + * normally happens on the first call to realize. If you're + * running your halide pipeline inside time-sensitive code and + * wish to avoid including the time taken to compile a pipeline, + * then you can call this ahead of time. Returns the raw function + * pointer to the compiled pipeline. Default is to use the Target + * returned from Halide::get_jit_target_from_environment() + */ + EXPORT void *compile_jit(const Target &target = get_jit_target_from_environment()); + + /** Set the error handler function that be called in the case of + * runtime errors during halide pipelines. If you are compiling + * statically, you can also just define your own function with + * signature + \code + extern "C" void halide_error(void *user_context, const char *); + \endcode + * This will clobber Halide's version. + */ + EXPORT void set_error_handler(void (*handler)(void *, const char *)); + + /** Set a custom malloc and free for halide to use. Malloc should + * return 32-byte aligned chunks of memory, and it should be safe + * for Halide to read slightly out of bounds (up to 8 bytes before + * the start or beyond the end). If compiling statically, routines + * with appropriate signatures can be provided directly + \code + extern "C" void *halide_malloc(void *, size_t) + extern "C" void halide_free(void *, void *) + \endcode + * These will clobber Halide's versions. See \file HalideRuntime.h + * for declarations. + */ + EXPORT void set_custom_allocator(void *(*malloc)(void *, size_t), void (*free)(void *, void *)); + + /** Set a custom task handler to be called by the parallel for + * loop. It is useful to set this if you want to do some + * additional bookkeeping at the granularity of parallel + * tasks. The default implementation does this: + \code + extern "C" int halide_do_task(void *user_context, + int (*f)(void *, int, uint8_t *), + int idx, uint8_t *state) { + return f(user_context, idx, state); + } + \endcode + * If you are statically compiling, you can also just define your + * own version of the above function, and it will clobber Halide's + * version. + * + * If you're trying to use a custom parallel runtime, you probably + * don't want to call this. See instead \ref Func::set_custom_do_par_for . + */ + EXPORT void set_custom_do_task(int (*custom_do_task)(void *, int (*)(void *, int, uint8_t *), int, uint8_t *)); + + /** Set a custom parallel for loop launcher. Useful if your app + * already manages a thread pool. The default implementation is + * equivalent to this: + \code + extern "C" int halide_do_par_for(void *user_context, + int (*f)(void *, int, uint8_t *), + int min, int extent, uint8_t *state) { + int exit_status = 0; + parallel for (int idx = min; idx < min+extent; idx++) { + int job_status = halide_do_task(user_context, f, idx, state); + if (job_status) exit_status = job_status; + } + return exit_status; + } + \endcode + * + * However, notwithstanding the above example code, if one task + * fails, we may skip over other tasks, and if two tasks return + * different error codes, we may select one arbitrarily to return. + * + * If you are statically compiling, you can also just define your + * own version of the above function, and it will clobber Halide's + * version. + */ + EXPORT void set_custom_do_par_for( + int (*custom_do_par_for)(void *, int (*)(void *, int, uint8_t *), int, int, uint8_t *)); + + /** Set custom routines to call when tracing is enabled. Call this + * on the output Func of your pipeline. This then sets custom + * routines for the entire pipeline, not just calls to this + * Func. + * + * If you are statically compiling, you can also just define your + * own versions of the tracing functions (see HalideRuntime.h), + * and they will clobber Halide's versions. */ + EXPORT void set_custom_trace(Internal::JITCompiledModule::TraceFn); + + /** Set the function called to print messages from the runtime. + * If you are compiling statically, you can also just define your + * own function with signature + \code + extern "C" void halide_print(void *user_context, const char *); + \endcode + * This will clobber Halide's version. + */ + EXPORT void set_custom_print(void (*handler)(void *, const char *)); + + /** Set the maximum number of bytes used by memoization caching. + * If you are compiling statically, you should include HalideRuntime.h + * and call halide_memoization_cache_set_size() instead. + */ + EXPORT void memoization_cache_set_size(uint64_t size); + + /** When this function is compiled, include code that dumps its + * values to a file after it is realized, for the purpose of + * debugging. + * + * If filename ends in ".tif" or ".tiff" (case insensitive) the file + * is in TIFF format and can be read by standard tools. Oherwise, the + * file format is as follows: + * + * All data is in the byte-order of the target platform. First, a + * 20 byte-header containing four 32-bit ints, giving the extents + * of the first four dimensions. Dimensions beyond four are + * folded into the fourth. Then, a fifth 32-bit int giving the + * data type of the function. The typecodes are given by: float = + * 0, double = 1, uint8_t = 2, int8_t = 3, uint16_t = 4, int16_t = + * 5, uint32_t = 6, int32_t = 7, uint64_t = 8, int64_t = 9. The + * data follows the header, as a densely packed array of the given + * size and the given type. If given the extension .tmp, this file + * format can be natively read by the program ImageStack. */ + EXPORT void debug_to_file(const std::string &filename); + + /** The name of this function, either given during construction, + * or automatically generated. */ + EXPORT const std::string &name() const; + + /** Get the pure arguments. */ + EXPORT std::vector args() const; + + /** The right-hand-side value of the pure definition of this + * function. Causes an error if there's no pure definition, or if + * the function is defined to return multiple values. */ + EXPORT Expr value() const; + + /** The values returned by this function. An error if the function + * has not been been defined. Returns a Tuple with one element for + * functions defined to return a single value. */ + EXPORT Tuple values() const; + + /** Does this function have at least a pure definition. */ + EXPORT bool defined() const; + + /** Get the left-hand-side of the update definition. An empty + * vector if there's no update definition. If there are + * multiple update definitions for this function, use the + * argument to select which one you want. */ + EXPORT const std::vector &update_args(int idx = 0) const; + + /** Get the right-hand-side of an update definition. An error if + * there's no update definition. If there are multiple + * update definitions for this function, use the argument to + * select which one you want. */ + EXPORT Expr update_value(int idx = 0) const; + + /** Get the right-hand-side of an update definition for + * functions that returns multiple values. An error if there's no + * update definition. Returns a Tuple with one element for + * functions that return a single value. */ + EXPORT Tuple update_values(int idx = 0) const; + + /** Get the reduction domain for an update definition, if there is + * one. */ + EXPORT RDom reduction_domain(int idx = 0) const; + + /** Does this function have at least one update definition? */ + EXPORT bool has_update_definition() const; + + /** How many update definitions does this function have? */ + EXPORT int num_update_definitions() const; + + /** Is this function an external stage? That is, was it defined + * using define_extern? */ + EXPORT bool is_extern() const; + + /** Add an extern definition for this Func. This lets you define a + * Func that represents an external pipeline stage. You can, for + * example, use it to wrap a call to an extern library such as + * fftw. */ + // @{ + EXPORT void define_extern(const std::string &function_name, + const std::vector ¶ms, + Type t, + int dimensionality) + { + define_extern(function_name, params, Internal::vec(t), dimensionality); + } - /** Split a dimension into inner and outer subdimensions with the - * given names, where the inner dimension iterates from 0 to - * factor-1. The inner and outer subdimensions can then be dealt - * with using the other scheduling calls. It's ok to reuse the old - * variable name as either the inner or outer variable. */ - EXPORT Func &split(VarOrRVar old, VarOrRVar outer, VarOrRVar inner, Expr factor); - - /** Join two dimensions into a single fused dimenion. The fused - * dimension covers the product of the extents of the inner and - * outer dimensions given. */ - EXPORT Func &fuse(VarOrRVar inner, VarOrRVar outer, VarOrRVar fused); - - /** Mark a dimension to be traversed serially. This is the default. */ - EXPORT Func &serial(VarOrRVar var); - - /** Mark a dimension to be traversed in parallel */ - EXPORT Func ¶llel(VarOrRVar var); - - /** Split a dimension by the given task_size, and the parallelize the - * outer dimension. This creates parallel tasks that have size - * task_size. After this call, var refers to the outer dimension of - * the split. The inner dimension has a new anonymous name. If you - * wish to mutate it, or schedule with respect to it, do the split - * manually. */ - EXPORT Func ¶llel(VarOrRVar var, Expr task_size); - - /** Mark a dimension to be computed all-at-once as a single - * vector. The dimension should have constant extent - - * e.g. because it is the inner dimension following a split by a - * constant factor. For most uses of vectorize you want the two - * argument form. The variable to be vectorized should be the - * innermost one. */ - EXPORT Func &vectorize(VarOrRVar var); - - /** Mark a dimension to be completely unrolled. The dimension - * should have constant extent - e.g. because it is the inner - * dimension following a split by a constant factor. For most uses - * of unroll you want the two-argument form. */ - EXPORT Func &unroll(VarOrRVar var); - - /** Split a dimension by the given factor, then vectorize the - * inner dimension. This is how you vectorize a loop of unknown - * size. The variable to be vectorized should be the innermost - * one. After this call, var refers to the outer dimension of the - * split. */ - EXPORT Func &vectorize(VarOrRVar var, int factor); - - /** Split a dimension by the given factor, then unroll the inner - * dimension. This is how you unroll a loop of unknown size by - * some constant factor. After this call, var refers to the outer - * dimension of the split. */ - EXPORT Func &unroll(VarOrRVar var, int factor); - - /** Statically declare that the range over which a function should - * be evaluated is given by the second and third arguments. This - * can let Halide perform some optimizations. E.g. if you know - * there are going to be 4 color channels, you can completely - * vectorize the color channel dimension without the overhead of - * splitting it up. If bounds inference decides that it requires - * more of this function than the bounds you have stated, a - * runtime error will occur when you try to run your pipeline. */ - EXPORT Func &bound(Var var, Expr min, Expr extent); - - /** Split two dimensions at once by the given factors, and then - * reorder the resulting dimensions to be xi, yi, xo, yo from - * innermost outwards. This gives a tiled traversal. */ - EXPORT Func &tile(VarOrRVar x, VarOrRVar y, - VarOrRVar xo, VarOrRVar yo, - VarOrRVar xi, VarOrRVar yi, - Expr xfactor, Expr yfactor); - - /** A shorter form of tile, which reuses the old variable names as - * the new outer dimensions */ - EXPORT Func &tile(VarOrRVar x, VarOrRVar y, - VarOrRVar xi, VarOrRVar yi, - Expr xfactor, Expr yfactor); - - /** Reorder variables to have the given nesting order, from - * innermost out */ - EXPORT Func &reorder(const std::vector &vars); - - /** Reorder two dimensions so that x is traversed inside y. Does - * not affect the nesting order of other dimensions. E.g, if you - * say foo(x, y, z, w) = bar; foo.reorder(w, x); then foo will be - * traversed in the order (w, y, z, x), from innermost - * outwards. */ - EXPORT Func &reorder(VarOrRVar x, VarOrRVar y); - - /** Reorder three dimensions to have the given nesting order, from - * innermost out */ - EXPORT Func &reorder(VarOrRVar x, VarOrRVar y, VarOrRVar z); - - /** Reorder four dimensions to have the given nesting order, from - * innermost out */ - EXPORT Func &reorder(VarOrRVar x, VarOrRVar y, VarOrRVar z, - VarOrRVar w); - - /** Reorder five dimensions to have the given nesting order, from - * innermost out */ - EXPORT Func &reorder(VarOrRVar x, VarOrRVar y, VarOrRVar z, - VarOrRVar w, VarOrRVar t); - - /** Reorder six dimensions to have the given nesting order, from - * innermost out */ - EXPORT Func &reorder(VarOrRVar x, VarOrRVar y, VarOrRVar z, - VarOrRVar w, VarOrRVar t1, VarOrRVar t2); - - /** Reorder seven dimensions to have the given nesting order, from - * innermost out */ - EXPORT Func &reorder(VarOrRVar x, VarOrRVar y, VarOrRVar z, - VarOrRVar w, VarOrRVar t1, VarOrRVar t2, - VarOrRVar t3); - - /** Reorder eight dimensions to have the given nesting order, from - * innermost out */ - EXPORT Func &reorder(VarOrRVar x, VarOrRVar y, VarOrRVar z, - VarOrRVar w, VarOrRVar t1, VarOrRVar t2, - VarOrRVar t3, VarOrRVar t4); - - /** Reorder nine dimensions to have the given nesting order, from - * innermost out */ - EXPORT Func &reorder(VarOrRVar x, VarOrRVar y, VarOrRVar z, - VarOrRVar w, VarOrRVar t1, VarOrRVar t2, - VarOrRVar t3, VarOrRVar t4, VarOrRVar t5); - - /** Reorder ten dimensions to have the given nesting order, from - * innermost out */ - EXPORT Func &reorder(VarOrRVar x, VarOrRVar y, VarOrRVar z, - VarOrRVar w, VarOrRVar t1, VarOrRVar t2, - VarOrRVar t3, VarOrRVar t4, VarOrRVar t5, - VarOrRVar t6); - - /** Rename a dimension. Equivalent to split with a inner size of one. */ - EXPORT Func &rename(VarOrRVar old_name, VarOrRVar new_name); - - /** Specify that race conditions are permitted for this Func, - * which enables parallelizing over RVars even when Halide cannot - * prove that it is safe to do so. Use this with great caution, - * and only if you can prove to yourself that this is safe, as it - * may result in a non-deterministic routine that returns - * different values at different times or on different machines. */ - EXPORT Func &allow_race_conditions(); - - - /** Specialize a Func. This creates a special-case version of the - * Func where the given condition is true. The most effective - * conditions are those of the form param == value, and boolean - * Params. Consider a simple example: - \code - f(x) = x + select(cond, 0, 1); - f.compute_root(); - \endcode - * This is equivalent to: - \code + EXPORT void define_extern(const std::string &function_name, + const std::vector ¶ms, + const std::vector &types, + int dimensionality); + // @} + + /** Get the types of the outputs of this Func. */ + EXPORT const std::vector &output_types() const; + + /** Get the number of outputs of this Func. Corresponds to the + * size of the Tuple this Func was defined to return. */ + EXPORT int outputs() const; + + /** Get the name of the extern function called for an extern + * definition. */ + EXPORT const std::string &extern_function_name() const; + + /** The dimensionality (number of arguments) of this + * function. Zero if the function is not yet defined. */ + EXPORT int dimensions() const; + + /** Construct either the left-hand-side of a definition, or a call + * to a functions that happens to only contain vars as + * arguments. If the function has already been defined, and fewer + * arguments are given than the function has dimensions, then + * enough implicit vars are added to the end of the argument list + * to make up the difference (see \ref Var::implicit) */ + // @{ + EXPORT FuncRefVar operator()() const; + EXPORT FuncRefVar operator()(Var x) const; + EXPORT FuncRefVar operator()(Var x, Var y) const; + EXPORT FuncRefVar operator()(Var x, Var y, Var z) const; + EXPORT FuncRefVar operator()(Var x, Var y, Var z, Var w) const; + EXPORT FuncRefVar operator()(Var x, Var y, Var z, Var w, Var u) const; + EXPORT FuncRefVar operator()(Var x, Var y, Var z, Var w, Var u, Var v) const; + EXPORT FuncRefVar operator()(std::vector) const; + // @} + + /** Either calls to the function, or the left-hand-side of a + * update definition (see \ref RDom). If the function has + * already been defined, and fewer arguments are given than the + * function has dimensions, then enough implicit vars are added to + * the end of the argument list to make up the difference. (see + * \ref Var::implicit)*/ + // @{ + EXPORT FuncRefExpr operator()(Expr x) const; + EXPORT FuncRefExpr operator()(Expr x, Expr y) const; + EXPORT FuncRefExpr operator()(Expr x, Expr y, Expr z) const; + EXPORT FuncRefExpr operator()(Expr x, Expr y, Expr z, Expr w) const; + EXPORT FuncRefExpr operator()(Expr x, Expr y, Expr z, Expr w, Expr u) const; + EXPORT FuncRefExpr operator()(Expr x, Expr y, Expr z, Expr w, Expr u, Expr v) const; + EXPORT FuncRefExpr operator()(std::vector) const; + // @} + + /** Split a dimension into inner and outer subdimensions with the + * given names, where the inner dimension iterates from 0 to + * factor-1. The inner and outer subdimensions can then be dealt + * with using the other scheduling calls. It's ok to reuse the old + * variable name as either the inner or outer variable. */ + EXPORT Func &split(VarOrRVar old, VarOrRVar outer, VarOrRVar inner, Expr factor); + + /** Join two dimensions into a single fused dimenion. The fused + * dimension covers the product of the extents of the inner and + * outer dimensions given. */ + EXPORT Func &fuse(VarOrRVar inner, VarOrRVar outer, VarOrRVar fused); + + /** Mark a dimension to be traversed serially. This is the default. */ + EXPORT Func &serial(VarOrRVar var); + + /** Mark a dimension to be traversed in parallel */ + EXPORT Func ¶llel(VarOrRVar var); + + /** Split a dimension by the given task_size, and the parallelize the + * outer dimension. This creates parallel tasks that have size + * task_size. After this call, var refers to the outer dimension of + * the split. The inner dimension has a new anonymous name. If you + * wish to mutate it, or schedule with respect to it, do the split + * manually. */ + EXPORT Func ¶llel(VarOrRVar var, Expr task_size); + + /** Mark a dimension to be computed all-at-once as a single + * vector. The dimension should have constant extent - + * e.g. because it is the inner dimension following a split by a + * constant factor. For most uses of vectorize you want the two + * argument form. The variable to be vectorized should be the + * innermost one. */ + EXPORT Func &vectorize(VarOrRVar var); + + /** Mark a dimension to be completely unrolled. The dimension + * should have constant extent - e.g. because it is the inner + * dimension following a split by a constant factor. For most uses + * of unroll you want the two-argument form. */ + EXPORT Func &unroll(VarOrRVar var); + + /** Split a dimension by the given factor, then vectorize the + * inner dimension. This is how you vectorize a loop of unknown + * size. The variable to be vectorized should be the innermost + * one. After this call, var refers to the outer dimension of the + * split. */ + EXPORT Func &vectorize(VarOrRVar var, int factor); + + /** Split a dimension by the given factor, then unroll the inner + * dimension. This is how you unroll a loop of unknown size by + * some constant factor. After this call, var refers to the outer + * dimension of the split. */ + EXPORT Func &unroll(VarOrRVar var, int factor); + + /** Statically declare that the range over which a function should + * be evaluated is given by the second and third arguments. This + * can let Halide perform some optimizations. E.g. if you know + * there are going to be 4 color channels, you can completely + * vectorize the color channel dimension without the overhead of + * splitting it up. If bounds inference decides that it requires + * more of this function than the bounds you have stated, a + * runtime error will occur when you try to run your pipeline. */ + EXPORT Func &bound(Var var, Expr min, Expr extent); + + /** Split two dimensions at once by the given factors, and then + * reorder the resulting dimensions to be xi, yi, xo, yo from + * innermost outwards. This gives a tiled traversal. */ + EXPORT Func & + tile(VarOrRVar x, VarOrRVar y, VarOrRVar xo, VarOrRVar yo, VarOrRVar xi, VarOrRVar yi, Expr xfactor, Expr yfactor); + + /** A shorter form of tile, which reuses the old variable names as + * the new outer dimensions */ + EXPORT Func &tile(VarOrRVar x, VarOrRVar y, VarOrRVar xi, VarOrRVar yi, Expr xfactor, Expr yfactor); + + /** Reorder variables to have the given nesting order, from + * innermost out */ + EXPORT Func &reorder(const std::vector &vars); + + /** Reorder two dimensions so that x is traversed inside y. Does + * not affect the nesting order of other dimensions. E.g, if you + * say foo(x, y, z, w) = bar; foo.reorder(w, x); then foo will be + * traversed in the order (w, y, z, x), from innermost + * outwards. */ + EXPORT Func &reorder(VarOrRVar x, VarOrRVar y); + + /** Reorder three dimensions to have the given nesting order, from + * innermost out */ + EXPORT Func &reorder(VarOrRVar x, VarOrRVar y, VarOrRVar z); + + /** Reorder four dimensions to have the given nesting order, from + * innermost out */ + EXPORT Func &reorder(VarOrRVar x, VarOrRVar y, VarOrRVar z, VarOrRVar w); + + /** Reorder five dimensions to have the given nesting order, from + * innermost out */ + EXPORT Func &reorder(VarOrRVar x, VarOrRVar y, VarOrRVar z, VarOrRVar w, VarOrRVar t); + + /** Reorder six dimensions to have the given nesting order, from + * innermost out */ + EXPORT Func &reorder(VarOrRVar x, VarOrRVar y, VarOrRVar z, VarOrRVar w, VarOrRVar t1, VarOrRVar t2); + + /** Reorder seven dimensions to have the given nesting order, from + * innermost out */ + EXPORT Func &reorder(VarOrRVar x, VarOrRVar y, VarOrRVar z, VarOrRVar w, VarOrRVar t1, VarOrRVar t2, VarOrRVar t3); + + /** Reorder eight dimensions to have the given nesting order, from + * innermost out */ + EXPORT Func & + reorder(VarOrRVar x, VarOrRVar y, VarOrRVar z, VarOrRVar w, VarOrRVar t1, VarOrRVar t2, VarOrRVar t3, VarOrRVar t4); + + /** Reorder nine dimensions to have the given nesting order, from + * innermost out */ + EXPORT Func &reorder(VarOrRVar x, + VarOrRVar y, + VarOrRVar z, + VarOrRVar w, + VarOrRVar t1, + VarOrRVar t2, + VarOrRVar t3, + VarOrRVar t4, + VarOrRVar t5); + + /** Reorder ten dimensions to have the given nesting order, from + * innermost out */ + EXPORT Func &reorder(VarOrRVar x, + VarOrRVar y, + VarOrRVar z, + VarOrRVar w, + VarOrRVar t1, + VarOrRVar t2, + VarOrRVar t3, + VarOrRVar t4, + VarOrRVar t5, + VarOrRVar t6); + + /** Rename a dimension. Equivalent to split with a inner size of one. */ + EXPORT Func &rename(VarOrRVar old_name, VarOrRVar new_name); + + /** Specify that race conditions are permitted for this Func, + * which enables parallelizing over RVars even when Halide cannot + * prove that it is safe to do so. Use this with great caution, + * and only if you can prove to yourself that this is safe, as it + * may result in a non-deterministic routine that returns + * different values at different times or on different machines. */ + EXPORT Func &allow_race_conditions(); + + + /** Specialize a Func. This creates a special-case version of the + * Func where the given condition is true. The most effective + * conditions are those of the form param == value, and boolean + * Params. Consider a simple example: + \code + f(x) = x + select(cond, 0, 1); + f.compute_root(); + \endcode + * This is equivalent to: + \code + for (int x = 0; x < width; x++) { + f[x] = x + (cond ? 0 : 1); + } + \endcode + * Adding the scheduling directive: + \code + f.specialize(cond) + \endcode + * makes it equivalent to: + \code + if (cond) { + for (int x = 0; x < width; x++) { + f[x] = x; + } + } else { + for (int x = 0; x < width; x++) { + f[x] = x + 1; + } + } + \endcode + * Note that the inner loops have been simplified. In the first + * path Halide knows that cond is true, and in the second path + * Halide knows that it is false. + * + * The specialized version gets its own schedule, which inherits + * every directive made about the parent Func's schedule so far + * except for its specializations. This method returns a handle to + * the new schedule. If you wish to retrieve the specialized + * sub-schedule again later, you can call this method with the + * same condition. Consider the following example of scheduling + * the specialized version: + * + \code + f(x) = x; + f.compute_root(); + f.specialize(width > 1).unroll(x, 2); + \endcode + * Assuming for simplicity that width is even, this is equivalent to: + \code + if (width > 1) { + for (int x = 0; x < width/2; x++) { + f[2*x] = 2*x; + f[2*x + 1] = 2*x + 1; + } + } else { + for (int x = 0; x < width/2; x++) { + f[x] = x; + } + } + \endcode + * For this case, it may be better to schedule the un-specialized + * case instead: + \code + f(x) = x; + f.compute_root(); + f.specialize(width == 1); // Creates a copy of the schedule so far. + f.unroll(x, 2); // Only applies to the unspecialized case. + \endcode + * This is equivalent to: + \code + if (width == 1) { + f[0] = 0; + } else { + for (int x = 0; x < width/2; x++) { + f[2*x] = 2*x; + f[2*x + 1] = 2*x + 1; + } + } + \endcode + * This can be a good way to write a pipeline that splits, + * vectorizes, or tiles, but can still handle small inputs. + * + * If a Func has several specializations, the first matching one + * will be used, so the order in which you define specializations + * is significant. For example: + * + \code + f(x) = x + select(cond1, a, b) - select(cond2, c, d); + f.specialize(cond1); + f.specialize(cond2); + \endcode + * is equivalent to: + \code + if (cond1) { for (int x = 0; x < width; x++) { - f[x] = x + (cond ? 0 : 1); + f[x] = x + a - (cond2 ? c : d); } - \endcode - * Adding the scheduling directive: - \code - f.specialize(cond) - \endcode - * makes it equivalent to: - \code - if (cond) { + } else if (cond2) { + for (int x = 0; x < width; x++) { + f[x] = x + b - c; + } + } else { + for (int x = 0; x < width; x++) { + f[x] = x + b - d; + } + } + \endcode + * + * Specializations may in turn be specialized, which creates a + * nested if statement in the generated code. + * + \code + f(x) = x + select(cond1, a, b) - select(cond2, c, d); + f.specialize(cond1).specialize(cond2); + \endcode + * This is equivalent to: + \code + if (cond1) { + if (cond2) { for (int x = 0; x < width; x++) { - f[x] = x; + f[x] = x + a - c; } } else { for (int x = 0; x < width; x++) { - f[x] = x + 1; + f[x] = x + a - d; } } - \endcode - * Note that the inner loops have been simplified. In the first - * path Halide knows that cond is true, and in the second path - * Halide knows that it is false. - * - * The specialized version gets its own schedule, which inherits - * every directive made about the parent Func's schedule so far - * except for its specializations. This method returns a handle to - * the new schedule. If you wish to retrieve the specialized - * sub-schedule again later, you can call this method with the - * same condition. Consider the following example of scheduling - * the specialized version: - * - \code - f(x) = x; - f.compute_root(); - f.specialize(width > 1).unroll(x, 2); - \endcode - * Assuming for simplicity that width is even, this is equivalent to: - \code - if (width > 1) { - for (int x = 0; x < width/2; x++) { - f[2*x] = 2*x; - f[2*x + 1] = 2*x + 1; + } else { + for (int x = 0; x < width; x++) { + f[x] = x + b - (cond2 ? c : d); + } + } + \endcode + * To create a 4-way if statement that simplifies away all of the + * ternary operators above, you could say: + \code + f.specialize(cond1).specialize(cond2); + f.specialize(cond2); + \endcode + * or + \code + f.specialize(cond1 && cond2); + f.specialize(cond1); + f.specialize(cond2); + \endcode + * + * Any prior Func which is compute_at some variable of this Func + * gets separately included in all paths of the generated if + * statement. The Var in the compute_at call to must exist in all + * paths, but it may have been generated via a different path of + * splits, fuses, and renames. This can be used somewhat + * creatively. Consider the following code: + \code + g(x, y) = 8*x; + f(x, y) = g(x, y) + 1; + f.compute_root().specialize(cond); + Var g_loop; + f.specialize(cond).rename(y, g_loop); + f.rename(x, g_loop); + g.compute_at(f, g_loop); + \endcode + * When cond is true, this is equivalent to g.compute_at(f,y). + * When it is false, this is equivalent to g.compute_at(f,x). + */ + EXPORT Stage specialize(Expr condition); + + /** Tell Halide that the following dimensions correspond to GPU + * thread indices. This is useful if you compute a producer + * function within the block indices of a consumer function, and + * want to control how that function's dimensions map to GPU + * threads. If the selected target is not an appropriate GPU, this + * just marks those dimensions as parallel. */ + // @{ + EXPORT Func &gpu_threads(VarOrRVar thread_x, GPUAPI gpu_api = GPU_Default); + EXPORT Func &gpu_threads(VarOrRVar thread_x, VarOrRVar thread_y, GPUAPI gpu_api = GPU_Default); + EXPORT Func &gpu_threads(VarOrRVar thread_x, VarOrRVar thread_y, VarOrRVar thread_z, GPUAPI gpu_api = GPU_Default); + // @} + + /** Tell Halide to run this stage using a single gpu thread and + * block. This is not an efficient use of your GPU, but it can be + * useful to avoid copy-back for intermediate update stages that + * touch a very small part of your Func. */ + EXPORT Func &gpu_single_thread(GPUAPI gpu_api = GPU_Default); + + /** \deprecated Old name for #gpu_threads. */ + // @{ + EXPORT Func &cuda_threads(VarOrRVar thread_x) { return gpu_threads(thread_x); } + EXPORT Func &cuda_threads(VarOrRVar thread_x, VarOrRVar thread_y) { return gpu_threads(thread_x, thread_y); } + EXPORT Func &cuda_threads(VarOrRVar thread_x, VarOrRVar thread_y, VarOrRVar thread_z) + { + return gpu_threads(thread_x, thread_y, thread_z); + } + // @} + + /** Tell Halide that the following dimensions correspond to GPU + * block indices. This is useful for scheduling stages that will + * run serially within each GPU block. If the selected target is + * not ptx, this just marks those dimensions as parallel. */ + // @{ + EXPORT Func &gpu_blocks(VarOrRVar block_x, GPUAPI gpu_api = GPU_Default); + EXPORT Func &gpu_blocks(VarOrRVar block_x, VarOrRVar block_y, GPUAPI gpu_api = GPU_Default); + EXPORT Func &gpu_blocks(VarOrRVar block_x, VarOrRVar block_y, VarOrRVar block_z, GPUAPI gpu_api = GPU_Default); + // @} + + /** \deprecated Old name for #gpu_blocks. */ + // @{ + EXPORT Func &cuda_blocks(VarOrRVar block_x) { return gpu_blocks(block_x); } + EXPORT Func &cuda_blocks(VarOrRVar block_x, VarOrRVar block_y) { return gpu_blocks(block_x, block_y); } + EXPORT Func &cuda_blocks(VarOrRVar block_x, VarOrRVar block_y, VarOrRVar block_z) + { + return gpu_blocks(block_x, block_y, block_z); + } + // @} + + /** Tell Halide that the following dimensions correspond to GPU + * block indices and thread indices. If the selected target is not + * ptx, these just mark the given dimensions as parallel. The + * dimensions are consumed by this call, so do all other + * unrolling, reordering, etc first. */ + // @{ + EXPORT Func &gpu(VarOrRVar block_x, VarOrRVar thread_x, GPUAPI gpu_api = GPU_Default); + EXPORT Func & + gpu(VarOrRVar block_x, VarOrRVar block_y, VarOrRVar thread_x, VarOrRVar thread_y, GPUAPI gpu_api = GPU_Default); + EXPORT Func &gpu(VarOrRVar block_x, + VarOrRVar block_y, + VarOrRVar block_z, + VarOrRVar thread_x, + VarOrRVar thread_y, + VarOrRVar thread_z, + GPUAPI gpu_api = GPU_Default); + // @} + + /** \deprecated Old name for #gpu. */ + // @{ + EXPORT Func &cuda(VarOrRVar block_x, VarOrRVar thread_x) { return gpu(block_x, thread_x); } + EXPORT Func &cuda(VarOrRVar block_x, VarOrRVar block_y, VarOrRVar thread_x, VarOrRVar thread_y) + { + return gpu(block_x, thread_x, block_y, thread_y); + } + EXPORT Func &cuda(VarOrRVar block_x, + VarOrRVar block_y, + VarOrRVar block_z, + VarOrRVar thread_x, + VarOrRVar thread_y, + VarOrRVar thread_z) + { + return gpu(block_x, thread_x, block_y, thread_y, block_z, thread_z); + } + // @} + + /** Short-hand for tiling a domain and mapping the tile indices + * to GPU block indices and the coordinates within each tile to + * GPU thread indices. Consumes the variables given, so do all + * other scheduling first. */ + // @{ + EXPORT Func &gpu_tile(VarOrRVar x, int x_size, GPUAPI gpu_api = GPU_Default); + EXPORT Func &gpu_tile(VarOrRVar x, VarOrRVar y, int x_size, int y_size, GPUAPI gpu_api = GPU_Default); + EXPORT Func & + gpu_tile(VarOrRVar x, VarOrRVar y, VarOrRVar z, int x_size, int y_size, int z_size, GPUAPI gpu_api = GPU_Default); + // @} + + /** \deprecated Old name for #gpu_tile. */ + // @{ + EXPORT Func &cuda_tile(VarOrRVar x, int x_size) { return gpu_tile(x, x_size); } + EXPORT Func &cuda_tile(VarOrRVar x, VarOrRVar y, int x_size, int y_size) { return gpu_tile(x, y, x_size, y_size); } + EXPORT Func &cuda_tile(VarOrRVar x, VarOrRVar y, VarOrRVar z, int x_size, int y_size, int z_size) + { + return gpu_tile(x, y, z, x_size, y_size, z_size); + } + // @} + + /** Schedule for execution using GLSL. Conceptually, this is similar to + * parallelization over 'x' and 'y' (since GLSL shaders compute individual + * output pixels in parallel) and vectorization over 'c' (since GLSL + * implicitly vectorizes the color channel). */ + EXPORT Func &glsl(Var x, Var y, Var c); + + /** Specify how the storage for the function is laid out. These + * calls let you specify the nesting order of the dimensions. For + * example, foo.reorder_storage(y, x) tells Halide to use + * column-major storage for any realizations of foo, without + * changing how you refer to foo in the code. You may want to do + * this if you intend to vectorize across y. When representing + * color images, foo.reorder_storage(c, x, y) specifies packed + * storage (red, green, and blue values adjacent in memory), and + * foo.reorder_storage(x, y, c) specifies planar storage (entire + * red, green, and blue images one after the other in memory). + * + * If you leave out some dimensions, those remain in the same + * positions in the nesting order while the specified variables + * are reordered around them. */ + // @{ + EXPORT Func &reorder_storage(Var x, Var y); + EXPORT Func &reorder_storage(Var x, Var y, Var z); + EXPORT Func &reorder_storage(Var x, Var y, Var z, Var w); + EXPORT Func &reorder_storage(Var x, Var y, Var z, Var w, Var t); + // @} + + /** Compute this function as needed for each unique value of the + * given var for the given calling function f. + * + * For example, consider the simple pipeline: + \code + Func f, g; + Var x, y; + g(x, y) = x*y; + f(x, y) = g(x, y) + g(x, y+1) + g(x+1, y) + g(x+1, y+1); + \endcode + * + * If we schedule f like so: + * + \code + g.compute_at(f, x); + \endcode + * + * Then the C code equivalent to this pipeline will look like this + * + \code + + int f[height][width]; + for (int y = 0; y < height; y++) { + for (int x = 0; x < width; x++) { + int g[2][2]; + g[0][0] = x*y; + g[0][1] = (x+1)*y; + g[1][0] = x*(y+1); + g[1][1] = (x+1)*(y+1); + f[y][x] = g[0][0] + g[1][0] + g[0][1] + g[1][1]; } - } else { - for (int x = 0; x < width/2; x++) { - f[x] = x; + } + + \endcode + * + * The allocation and computation of g is within f's loop over x, + * and enough of g is computed to satisfy all that f will need for + * that iteration. This has excellent locality - values of g are + * used as soon as they are computed, but it does redundant + * work. Each value of g ends up getting computed four times. If + * we instead schedule f like so: + * + \code + g.compute_at(f, y); + \endcode + * + * The equivalent C code is: + * + \code + int f[height][width]; + for (int y = 0; y < height; y++) { + int g[2][width+1]; + for (int x = 0; x < width; x++) { + g[0][x] = x*y; + g[1][x] = x*(y+1); } - } - \endcode - * For this case, it may be better to schedule the un-specialized - * case instead: - \code - f(x) = x; - f.compute_root(); - f.specialize(width == 1); // Creates a copy of the schedule so far. - f.unroll(x, 2); // Only applies to the unspecialized case. - \endcode - * This is equivalent to: - \code - if (width == 1) { - f[0] = 0; - } else { - for (int x = 0; x < width/2; x++) { - f[2*x] = 2*x; - f[2*x + 1] = 2*x + 1; + for (int x = 0; x < width; x++) { + f[y][x] = g[0][x] + g[1][x] + g[0][x+1] + g[1][x+1]; } - } - \endcode - * This can be a good way to write a pipeline that splits, - * vectorizes, or tiles, but can still handle small inputs. - * - * If a Func has several specializations, the first matching one - * will be used, so the order in which you define specializations - * is significant. For example: - * - \code - f(x) = x + select(cond1, a, b) - select(cond2, c, d); - f.specialize(cond1); - f.specialize(cond2); - \endcode - * is equivalent to: - \code - if (cond1) { + } + \endcode + * + * The allocation and computation of g is within f's loop over y, + * and enough of g is computed to satisfy all that f will need for + * that iteration. This does less redundant work (each point in g + * ends up being evaluated twice), but the locality is not quite + * as good, and we have to allocate more temporary memory to store + * g. + */ + EXPORT Func &compute_at(Func f, Var var); + + /** Schedule a function to be computed within the iteration over + * some dimension of an update domain. Produces equivalent code + * to the version of compute_at that takes a Var. */ + EXPORT Func &compute_at(Func f, RVar var); + + /** Compute all of this function once ahead of time. Reusing + * the example in \ref Func::compute_at : + * + \code + Func f, g; + Var x, y; + g(x, y) = x*y; + f(x, y) = g(x, y) + g(x, y+1) + g(x+1, y) + g(x+1, y+1); + + g.compute_root(); + \endcode + * + * is equivalent to + * + \code + int f[height][width]; + int g[height+1][width+1]; + for (int y = 0; y < height+1; y++) { + for (int x = 0; x < width+1; x++) { + g[y][x] = x*y; + } + } + for (int y = 0; y < height; y++) { for (int x = 0; x < width; x++) { - f[x] = x + a - (cond2 ? c : d); + f[y][x] = g[y][x] + g[y+1][x] + g[y][x+1] + g[y+1][x+1]; } - } else if (cond2) { + } + \endcode + * + * g is computed once ahead of time, and enough is computed to + * satisfy all uses of it. This does no redundant work (each point + * in g is evaluated once), but has poor locality (values of g are + * probably not still in cache when they are used by f), and + * allocates lots of temporary memory to store g. + */ + EXPORT Func &compute_root(); + + /** Use the halide_memoization_cache_... interface to store a + * computed version of this function across invocations of the + * Func. + */ + EXPORT Func &memoize(); + + + /** Allocate storage for this function within f's loop over + * var. Scheduling storage is optional, and can be used to + * separate the loop level at which storage occurs from the loop + * level at which computation occurs to trade off between locality + * and redundant work. This can open the door for two types of + * optimization. + * + * Consider again the pipeline from \ref Func::compute_at : + \code + Func f, g; + Var x, y; + g(x, y) = x*y; + f(x, y) = g(x, y) + g(x+1, y) + g(x, y+1) + g(x+1, y+1); + \endcode + * + * If we schedule it like so: + * + \code + g.compute_at(f, x).store_at(f, y); + \endcode + * + * Then the computation of g takes place within the loop over x, + * but the storage takes place within the loop over y: + * + \code + int f[height][width]; + for (int y = 0; y < height; y++) { + int g[2][width+1]; for (int x = 0; x < width; x++) { - f[x] = x + b - c; + g[0][x] = x*y; + g[0][x+1] = (x+1)*y; + g[1][x] = x*(y+1); + g[1][x+1] = (x+1)*(y+1); + f[y][x] = g[0][x] + g[1][x] + g[0][x+1] + g[1][x+1]; } - } else { + } + \endcode + * + * Provided the for loop over x is serial, halide then + * automatically performs the following sliding window + * optimization: + * + \code + int f[height][width]; + for (int y = 0; y < height; y++) { + int g[2][width+1]; for (int x = 0; x < width; x++) { - f[x] = x + b - d; + if (x == 0) { + g[0][x] = x*y; + g[1][x] = x*(y+1); + } + g[0][x+1] = (x+1)*y; + g[1][x+1] = (x+1)*(y+1); + f[y][x] = g[0][x] + g[1][x] + g[0][x+1] + g[1][x+1]; } - } - \endcode - * - * Specializations may in turn be specialized, which creates a - * nested if statement in the generated code. - * - \code - f(x) = x + select(cond1, a, b) - select(cond2, c, d); - f.specialize(cond1).specialize(cond2); - \endcode - * This is equivalent to: - \code - if (cond1) { - if (cond2) { - for (int x = 0; x < width; x++) { - f[x] = x + a - c; - } - } else { - for (int x = 0; x < width; x++) { - f[x] = x + a - d; - } + } + \endcode + * + * Two of the assignments to g only need to be done when x is + * zero. The rest of the time, those sites have already been + * filled in by a previous iteration. This version has the + * locality of compute_at(f, x), but allocates more memory and + * does much less redundant work. + * + * Halide then further optimizes this pipeline like so: + * + \code + int f[height][width]; + for (int y = 0; y < height; y++) { + int g[2][2]; + for (int x = 0; x < width; x++) { + if (x == 0) { + g[0][0] = x*y; + g[1][0] = x*(y+1); + } + g[0][(x+1)%2] = (x+1)*y; + g[1][(x+1)%2] = (x+1)*(y+1); + f[y][x] = g[0][x%2] + g[1][x%2] + g[0][(x+1)%2] + g[1][(x+1)%2]; } - } else { + } + \endcode + * + * Halide has detected that it's possible to use a circular buffer + * to represent g, and has reduced all accesses to g modulo 2 in + * the x dimension. This optimization only triggers if the for + * loop over x is serial, and if halide can statically determine + * some power of two large enough to cover the range needed. For + * powers of two, the modulo operator compiles to more efficient + * bit-masking. This optimization reduces memory usage, and also + * improves locality by reusing recently-accessed memory instead + * of pulling new memory into cache. + * + */ + EXPORT Func &store_at(Func f, Var var); + + /** Equivalent to the version of store_at that takes a Var, but + * schedules storage within the loop over a dimension of a + * reduction domain */ + EXPORT Func &store_at(Func f, RVar var); + + /** Equivalent to \ref Func::store_at, but schedules storage + * outside the outermost loop. */ + EXPORT Func &store_root(); + + /** Aggressively inline all uses of this function. This is the + * default schedule, so you're unlikely to need to call this. For + * a Func with an update definition, that means it gets computed + * as close to the innermost loop as possible. + * + * Consider once more the pipeline from \ref Func::compute_at : + * + \code + Func f, g; + Var x, y; + g(x, y) = x*y; + f(x, y) = g(x, y) + g(x+1, y) + g(x, y+1) + g(x+1, y+1); + \endcode + * + * Leaving g as inline, this compiles to code equivalent to the following C: + * + \code + int f[height][width]; + for (int y = 0; y < height; y++) { for (int x = 0; x < width; x++) { - f[x] = x + b - (cond2 ? c : d); + f[y][x] = x*y + x*(y+1) + (x+1)*y + (x+1)*(y+1); } - } - \endcode - * To create a 4-way if statement that simplifies away all of the - * ternary operators above, you could say: - \code - f.specialize(cond1).specialize(cond2); - f.specialize(cond2); - \endcode - * or - \code - f.specialize(cond1 && cond2); - f.specialize(cond1); - f.specialize(cond2); - \endcode - * - * Any prior Func which is compute_at some variable of this Func - * gets separately included in all paths of the generated if - * statement. The Var in the compute_at call to must exist in all - * paths, but it may have been generated via a different path of - * splits, fuses, and renames. This can be used somewhat - * creatively. Consider the following code: - \code - g(x, y) = 8*x; - f(x, y) = g(x, y) + 1; - f.compute_root().specialize(cond); - Var g_loop; - f.specialize(cond).rename(y, g_loop); - f.rename(x, g_loop); - g.compute_at(f, g_loop); - \endcode - * When cond is true, this is equivalent to g.compute_at(f,y). - * When it is false, this is equivalent to g.compute_at(f,x). - */ - EXPORT Stage specialize(Expr condition); - - /** Tell Halide that the following dimensions correspond to GPU - * thread indices. This is useful if you compute a producer - * function within the block indices of a consumer function, and - * want to control how that function's dimensions map to GPU - * threads. If the selected target is not an appropriate GPU, this - * just marks those dimensions as parallel. */ - // @{ - EXPORT Func &gpu_threads(VarOrRVar thread_x, GPUAPI gpu_api = GPU_Default); - EXPORT Func &gpu_threads(VarOrRVar thread_x, VarOrRVar thread_y, GPUAPI gpu_api = GPU_Default); - EXPORT Func &gpu_threads(VarOrRVar thread_x, VarOrRVar thread_y, VarOrRVar thread_z, GPUAPI gpu_api = GPU_Default); - // @} - - /** Tell Halide to run this stage using a single gpu thread and - * block. This is not an efficient use of your GPU, but it can be - * useful to avoid copy-back for intermediate update stages that - * touch a very small part of your Func. */ - EXPORT Func &gpu_single_thread(GPUAPI gpu_api = GPU_Default); - - /** \deprecated Old name for #gpu_threads. */ - // @{ - EXPORT Func &cuda_threads(VarOrRVar thread_x) { - return gpu_threads(thread_x); - } - EXPORT Func &cuda_threads(VarOrRVar thread_x, VarOrRVar thread_y) { - return gpu_threads(thread_x, thread_y); - } - EXPORT Func &cuda_threads(VarOrRVar thread_x, VarOrRVar thread_y, VarOrRVar thread_z) { - return gpu_threads(thread_x, thread_y, thread_z); - } - // @} - - /** Tell Halide that the following dimensions correspond to GPU - * block indices. This is useful for scheduling stages that will - * run serially within each GPU block. If the selected target is - * not ptx, this just marks those dimensions as parallel. */ - // @{ - EXPORT Func &gpu_blocks(VarOrRVar block_x, GPUAPI gpu_api = GPU_Default); - EXPORT Func &gpu_blocks(VarOrRVar block_x, VarOrRVar block_y, GPUAPI gpu_api = GPU_Default); - EXPORT Func &gpu_blocks(VarOrRVar block_x, VarOrRVar block_y, VarOrRVar block_z, GPUAPI gpu_api = GPU_Default); - // @} - - /** \deprecated Old name for #gpu_blocks. */ - // @{ - EXPORT Func &cuda_blocks(VarOrRVar block_x) { - return gpu_blocks(block_x); - } - EXPORT Func &cuda_blocks(VarOrRVar block_x, VarOrRVar block_y) { - return gpu_blocks(block_x, block_y); - } - EXPORT Func &cuda_blocks(VarOrRVar block_x, VarOrRVar block_y, VarOrRVar block_z) { - return gpu_blocks(block_x, block_y, block_z); - } - // @} - - /** Tell Halide that the following dimensions correspond to GPU - * block indices and thread indices. If the selected target is not - * ptx, these just mark the given dimensions as parallel. The - * dimensions are consumed by this call, so do all other - * unrolling, reordering, etc first. */ - // @{ - EXPORT Func &gpu(VarOrRVar block_x, VarOrRVar thread_x, GPUAPI gpu_api = GPU_Default); - EXPORT Func &gpu(VarOrRVar block_x, VarOrRVar block_y, - VarOrRVar thread_x, VarOrRVar thread_y, GPUAPI gpu_api = GPU_Default); - EXPORT Func &gpu(VarOrRVar block_x, VarOrRVar block_y, VarOrRVar block_z, - VarOrRVar thread_x, VarOrRVar thread_y, VarOrRVar thread_z, GPUAPI gpu_api = GPU_Default); - // @} - - /** \deprecated Old name for #gpu. */ - // @{ - EXPORT Func &cuda(VarOrRVar block_x, VarOrRVar thread_x) { - return gpu(block_x, thread_x); - } - EXPORT Func &cuda(VarOrRVar block_x, VarOrRVar block_y, - VarOrRVar thread_x, VarOrRVar thread_y) { - return gpu(block_x, thread_x, block_y, thread_y); - } - EXPORT Func &cuda(VarOrRVar block_x, VarOrRVar block_y, VarOrRVar block_z, - VarOrRVar thread_x, VarOrRVar thread_y, VarOrRVar thread_z) { - return gpu(block_x, thread_x, block_y, thread_y, block_z, thread_z); - } - // @} - - /** Short-hand for tiling a domain and mapping the tile indices - * to GPU block indices and the coordinates within each tile to - * GPU thread indices. Consumes the variables given, so do all - * other scheduling first. */ - // @{ - EXPORT Func &gpu_tile(VarOrRVar x, int x_size, GPUAPI gpu_api = GPU_Default); - EXPORT Func &gpu_tile(VarOrRVar x, VarOrRVar y, int x_size, int y_size, GPUAPI gpu_api = GPU_Default); - EXPORT Func &gpu_tile(VarOrRVar x, VarOrRVar y, VarOrRVar z, - int x_size, int y_size, int z_size, GPUAPI gpu_api = GPU_Default); - // @} - - /** \deprecated Old name for #gpu_tile. */ - // @{ - EXPORT Func &cuda_tile(VarOrRVar x, int x_size) { - return gpu_tile(x, x_size); - } - EXPORT Func &cuda_tile(VarOrRVar x, VarOrRVar y, int x_size, int y_size) { - return gpu_tile(x, y, x_size, y_size); - } - EXPORT Func &cuda_tile(VarOrRVar x, VarOrRVar y, VarOrRVar z, - int x_size, int y_size, int z_size) { - return gpu_tile(x, y, z, x_size, y_size, z_size); - } - // @} - - /** Schedule for execution using GLSL. Conceptually, this is similar to - * parallelization over 'x' and 'y' (since GLSL shaders compute individual - * output pixels in parallel) and vectorization over 'c' (since GLSL - * implicitly vectorizes the color channel). */ - EXPORT Func &glsl(Var x, Var y, Var c); - - /** Specify how the storage for the function is laid out. These - * calls let you specify the nesting order of the dimensions. For - * example, foo.reorder_storage(y, x) tells Halide to use - * column-major storage for any realizations of foo, without - * changing how you refer to foo in the code. You may want to do - * this if you intend to vectorize across y. When representing - * color images, foo.reorder_storage(c, x, y) specifies packed - * storage (red, green, and blue values adjacent in memory), and - * foo.reorder_storage(x, y, c) specifies planar storage (entire - * red, green, and blue images one after the other in memory). - * - * If you leave out some dimensions, those remain in the same - * positions in the nesting order while the specified variables - * are reordered around them. */ - // @{ - EXPORT Func &reorder_storage(Var x, Var y); - EXPORT Func &reorder_storage(Var x, Var y, Var z); - EXPORT Func &reorder_storage(Var x, Var y, Var z, Var w); - EXPORT Func &reorder_storage(Var x, Var y, Var z, Var w, Var t); - // @} - - /** Compute this function as needed for each unique value of the - * given var for the given calling function f. - * - * For example, consider the simple pipeline: - \code - Func f, g; - Var x, y; - g(x, y) = x*y; - f(x, y) = g(x, y) + g(x, y+1) + g(x+1, y) + g(x+1, y+1); - \endcode - * - * If we schedule f like so: - * - \code - g.compute_at(f, x); - \endcode - * - * Then the C code equivalent to this pipeline will look like this - * - \code - - int f[height][width]; - for (int y = 0; y < height; y++) { - for (int x = 0; x < width; x++) { - int g[2][2]; - g[0][0] = x*y; - g[0][1] = (x+1)*y; - g[1][0] = x*(y+1); - g[1][1] = (x+1)*(y+1); - f[y][x] = g[0][0] + g[1][0] + g[0][1] + g[1][1]; - } - } - - \endcode - * - * The allocation and computation of g is within f's loop over x, - * and enough of g is computed to satisfy all that f will need for - * that iteration. This has excellent locality - values of g are - * used as soon as they are computed, but it does redundant - * work. Each value of g ends up getting computed four times. If - * we instead schedule f like so: - * - \code - g.compute_at(f, y); - \endcode - * - * The equivalent C code is: - * - \code - int f[height][width]; - for (int y = 0; y < height; y++) { - int g[2][width+1]; - for (int x = 0; x < width; x++) { - g[0][x] = x*y; - g[1][x] = x*(y+1); - } - for (int x = 0; x < width; x++) { - f[y][x] = g[0][x] + g[1][x] + g[0][x+1] + g[1][x+1]; - } - } - \endcode - * - * The allocation and computation of g is within f's loop over y, - * and enough of g is computed to satisfy all that f will need for - * that iteration. This does less redundant work (each point in g - * ends up being evaluated twice), but the locality is not quite - * as good, and we have to allocate more temporary memory to store - * g. - */ - EXPORT Func &compute_at(Func f, Var var); - - /** Schedule a function to be computed within the iteration over - * some dimension of an update domain. Produces equivalent code - * to the version of compute_at that takes a Var. */ - EXPORT Func &compute_at(Func f, RVar var); - - /** Compute all of this function once ahead of time. Reusing - * the example in \ref Func::compute_at : - * - \code - Func f, g; - Var x, y; - g(x, y) = x*y; - f(x, y) = g(x, y) + g(x, y+1) + g(x+1, y) + g(x+1, y+1); - - g.compute_root(); - \endcode - * - * is equivalent to - * - \code - int f[height][width]; - int g[height+1][width+1]; - for (int y = 0; y < height+1; y++) { - for (int x = 0; x < width+1; x++) { - g[y][x] = x*y; - } - } - for (int y = 0; y < height; y++) { - for (int x = 0; x < width; x++) { - f[y][x] = g[y][x] + g[y+1][x] + g[y][x+1] + g[y+1][x+1]; - } - } - \endcode - * - * g is computed once ahead of time, and enough is computed to - * satisfy all uses of it. This does no redundant work (each point - * in g is evaluated once), but has poor locality (values of g are - * probably not still in cache when they are used by f), and - * allocates lots of temporary memory to store g. - */ - EXPORT Func &compute_root(); - - /** Use the halide_memoization_cache_... interface to store a - * computed version of this function across invocations of the - * Func. - */ - EXPORT Func &memoize(); - - - /** Allocate storage for this function within f's loop over - * var. Scheduling storage is optional, and can be used to - * separate the loop level at which storage occurs from the loop - * level at which computation occurs to trade off between locality - * and redundant work. This can open the door for two types of - * optimization. - * - * Consider again the pipeline from \ref Func::compute_at : - \code - Func f, g; - Var x, y; - g(x, y) = x*y; - f(x, y) = g(x, y) + g(x+1, y) + g(x, y+1) + g(x+1, y+1); - \endcode - * - * If we schedule it like so: - * - \code - g.compute_at(f, x).store_at(f, y); - \endcode - * - * Then the computation of g takes place within the loop over x, - * but the storage takes place within the loop over y: - * - \code - int f[height][width]; - for (int y = 0; y < height; y++) { - int g[2][width+1]; - for (int x = 0; x < width; x++) { - g[0][x] = x*y; - g[0][x+1] = (x+1)*y; - g[1][x] = x*(y+1); - g[1][x+1] = (x+1)*(y+1); - f[y][x] = g[0][x] + g[1][x] + g[0][x+1] + g[1][x+1]; - } - } - \endcode - * - * Provided the for loop over x is serial, halide then - * automatically performs the following sliding window - * optimization: - * - \code - int f[height][width]; - for (int y = 0; y < height; y++) { - int g[2][width+1]; - for (int x = 0; x < width; x++) { - if (x == 0) { - g[0][x] = x*y; - g[1][x] = x*(y+1); - } - g[0][x+1] = (x+1)*y; - g[1][x+1] = (x+1)*(y+1); - f[y][x] = g[0][x] + g[1][x] + g[0][x+1] + g[1][x+1]; - } - } - \endcode - * - * Two of the assignments to g only need to be done when x is - * zero. The rest of the time, those sites have already been - * filled in by a previous iteration. This version has the - * locality of compute_at(f, x), but allocates more memory and - * does much less redundant work. - * - * Halide then further optimizes this pipeline like so: - * - \code - int f[height][width]; - for (int y = 0; y < height; y++) { - int g[2][2]; - for (int x = 0; x < width; x++) { - if (x == 0) { - g[0][0] = x*y; - g[1][0] = x*(y+1); - } - g[0][(x+1)%2] = (x+1)*y; - g[1][(x+1)%2] = (x+1)*(y+1); - f[y][x] = g[0][x%2] + g[1][x%2] + g[0][(x+1)%2] + g[1][(x+1)%2]; - } - } - \endcode - * - * Halide has detected that it's possible to use a circular buffer - * to represent g, and has reduced all accesses to g modulo 2 in - * the x dimension. This optimization only triggers if the for - * loop over x is serial, and if halide can statically determine - * some power of two large enough to cover the range needed. For - * powers of two, the modulo operator compiles to more efficient - * bit-masking. This optimization reduces memory usage, and also - * improves locality by reusing recently-accessed memory instead - * of pulling new memory into cache. - * - */ - EXPORT Func &store_at(Func f, Var var); - - /** Equivalent to the version of store_at that takes a Var, but - * schedules storage within the loop over a dimension of a - * reduction domain */ - EXPORT Func &store_at(Func f, RVar var); - - /** Equivalent to \ref Func::store_at, but schedules storage - * outside the outermost loop. */ - EXPORT Func &store_root(); - - /** Aggressively inline all uses of this function. This is the - * default schedule, so you're unlikely to need to call this. For - * a Func with an update definition, that means it gets computed - * as close to the innermost loop as possible. - * - * Consider once more the pipeline from \ref Func::compute_at : - * - \code - Func f, g; - Var x, y; - g(x, y) = x*y; - f(x, y) = g(x, y) + g(x+1, y) + g(x, y+1) + g(x+1, y+1); - \endcode - * - * Leaving g as inline, this compiles to code equivalent to the following C: - * - \code - int f[height][width]; - for (int y = 0; y < height; y++) { - for (int x = 0; x < width; x++) { - f[y][x] = x*y + x*(y+1) + (x+1)*y + (x+1)*(y+1); - } - } - \endcode - */ - EXPORT Func &compute_inline(); - - /** Get a handle on an update step for the purposes of scheduling - * it. */ - EXPORT Stage update(int idx = 0); - - /** Trace all loads from this Func by emitting calls to - * halide_trace. If the Func is inlined, this has no - * effect. */ - EXPORT Func &trace_loads(); - - /** Trace all stores to the buffer backing this Func by emitting - * calls to halide_trace. If the Func is inlined, this call - * has no effect. */ - EXPORT Func &trace_stores(); - - /** Trace all realizations of this Func by emitting calls to - * halide_trace. */ - EXPORT Func &trace_realizations(); - - /** Get a handle on the internal halide function that this Func - * represents. Useful if you want to do introspection on Halide - * functions */ - Internal::Function function() const { - return func; - } - - /** You can cast a Func to its pure stage for the purposes of - * scheduling it. */ - operator Stage() const; - - /** Get a handle on the output buffer for this Func. Only relevant - * if this is the output Func in a pipeline. Useful for making - * static promises about strides, mins, and extents. */ - // @{ - EXPORT OutputImageParam output_buffer() const; - EXPORT std::vector output_buffers() const; - // @} - - /** Casting a function to an expression is equivalent to calling - * the function with zero arguments. Implicit variables will be - * injected according to the function's dimensionality - * (see \ref Var::implicit). - * - * This lets you write things like: - * - \code - Func f, g; - Var x; - g(x) = ... - f(_) = g * 2; - \endcode - */ - operator Expr() const { - return (*this)(_); - } - - /** Use a Func as an argument to an external stage. */ - operator ExternFuncArgument() const { - return ExternFuncArgument(func); - } - - /** Infer the arguments to the Func, sorted into a canonical order: - * all buffers (sorted alphabetically by name), followed by all non-buffers - * (sorted alphabetically by name). - This lets you write things like: - \code - func.compile_to_assembly("/dev/stdout", func.infer_arguments()); - \endcode - */ - EXPORT std::vector infer_arguments() const; - + } + \endcode + */ + EXPORT Func &compute_inline(); + + /** Get a handle on an update step for the purposes of scheduling + * it. */ + EXPORT Stage update(int idx = 0); + + /** Trace all loads from this Func by emitting calls to + * halide_trace. If the Func is inlined, this has no + * effect. */ + EXPORT Func &trace_loads(); + + /** Trace all stores to the buffer backing this Func by emitting + * calls to halide_trace. If the Func is inlined, this call + * has no effect. */ + EXPORT Func &trace_stores(); + + /** Trace all realizations of this Func by emitting calls to + * halide_trace. */ + EXPORT Func &trace_realizations(); + + /** Get a handle on the internal halide function that this Func + * represents. Useful if you want to do introspection on Halide + * functions */ + Internal::Function function() const { return func; } + + /** You can cast a Func to its pure stage for the purposes of + * scheduling it. */ + operator Stage() const; + + /** Get a handle on the output buffer for this Func. Only relevant + * if this is the output Func in a pipeline. Useful for making + * static promises about strides, mins, and extents. */ + // @{ + EXPORT OutputImageParam output_buffer() const; + EXPORT std::vector output_buffers() const; + // @} + + /** Casting a function to an expression is equivalent to calling + * the function with zero arguments. Implicit variables will be + * injected according to the function's dimensionality + * (see \ref Var::implicit). + * + * This lets you write things like: + * + \code + Func f, g; + Var x; + g(x) = ... + f(_) = g * 2; + \endcode + */ + operator Expr() const { return (*this)(_); } + + /** Use a Func as an argument to an external stage. */ + operator ExternFuncArgument() const { return ExternFuncArgument(func); } + + /** Infer the arguments to the Func, sorted into a canonical order: + * all buffers (sorted alphabetically by name), followed by all non-buffers + * (sorted alphabetically by name). + This lets you write things like: + \code + func.compile_to_assembly("/dev/stdout", func.infer_arguments()); + \endcode + */ + EXPORT std::vector infer_arguments() const; }; - /** JIT-Compile and run enough code to evaluate a Halide - * expression. This can be thought of as a scalar version of - * \ref Func::realize */ -template -NO_INLINE T evaluate(Expr e) { - user_assert(e.type() == type_of()) - << "Can't evaluate expression " - << e << " of type " << e.type() - << " as a scalar of type " << type_of() << "\n"; - Func f; - f() = e; - Image im = f.realize(); - return im(0); +/** JIT-Compile and run enough code to evaluate a Halide + * expression. This can be thought of as a scalar version of + * \ref Func::realize */ +template NO_INLINE T evaluate(Expr e) +{ + user_assert(e.type() == type_of()) << "Can't evaluate expression " << e << " of type " << e.type() + << " as a scalar of type " << type_of() << "\n"; + Func f; + f() = e; + Image im = f.realize(); + return im(0); } /** JIT-compile and run enough code to evaluate a Halide Tuple. */ // @{ -template -NO_INLINE void evaluate(Tuple t, A *a, B *b) { - user_assert(t[0].type() == type_of()) - << "Can't evaluate expression " - << t[0] << " of type " << t[0].type() - << " as a scalar of type " << type_of() << "\n"; - user_assert(t[1].type() == type_of()) - << "Can't evaluate expression " - << t[1] << " of type " << t[1].type() - << " as a scalar of type " << type_of() << "\n"; - - Func f; - f() = t; - Realization r = f.realize(); - *a = Image(r[0])(0); - *b = Image(r[1])(0); -} - -template -NO_INLINE void evaluate(Tuple t, A *a, B *b, C *c) { - user_assert(t[0].type() == type_of()) - << "Can't evaluate expression " - << t[0] << " of type " << t[0].type() - << " as a scalar of type " << type_of() << "\n"; - user_assert(t[1].type() == type_of()) - << "Can't evaluate expression " - << t[1] << " of type " << t[1].type() - << " as a scalar of type " << type_of() << "\n"; - user_assert(t[2].type() == type_of()) - << "Can't evaluate expression " - << t[2] << " of type " << t[2].type() - << " as a scalar of type " << type_of() << "\n"; - - Func f; - f() = t; - Realization r = f.realize(); - *a = Image(r[0])(0); - *b = Image(r[1])(0); - *c = Image(r[2])(0); +template NO_INLINE void evaluate(Tuple t, A *a, B *b) +{ + user_assert(t[0].type() == type_of()) << "Can't evaluate expression " << t[0] << " of type " << t[0].type() + << " as a scalar of type " << type_of() << "\n"; + user_assert(t[1].type() == type_of()) << "Can't evaluate expression " << t[1] << " of type " << t[1].type() + << " as a scalar of type " << type_of() << "\n"; + + Func f; + f() = t; + Realization r = f.realize(); + *a = Image(r[0])(0); + *b = Image(r[1])(0); +} + +template NO_INLINE void evaluate(Tuple t, A *a, B *b, C *c) +{ + user_assert(t[0].type() == type_of()) << "Can't evaluate expression " << t[0] << " of type " << t[0].type() + << " as a scalar of type " << type_of() << "\n"; + user_assert(t[1].type() == type_of()) << "Can't evaluate expression " << t[1] << " of type " << t[1].type() + << " as a scalar of type " << type_of() << "\n"; + user_assert(t[2].type() == type_of()) << "Can't evaluate expression " << t[2] << " of type " << t[2].type() + << " as a scalar of type " << type_of() << "\n"; + + Func f; + f() = t; + Realization r = f.realize(); + *a = Image(r[0])(0); + *b = Image(r[1])(0); + *c = Image(r[2])(0); +} + +template NO_INLINE void evaluate(Tuple t, A *a, B *b, C *c, D *d) +{ + user_assert(t[0].type() == type_of()) << "Can't evaluate expression " << t[0] << " of type " << t[0].type() + << " as a scalar of type " << type_of() << "\n"; + user_assert(t[1].type() == type_of()) << "Can't evaluate expression " << t[1] << " of type " << t[1].type() + << " as a scalar of type " << type_of() << "\n"; + user_assert(t[2].type() == type_of()) << "Can't evaluate expression " << t[2] << " of type " << t[2].type() + << " as a scalar of type " << type_of() << "\n"; + user_assert(t[3].type() == type_of()) << "Can't evaluate expression " << t[3] << " of type " << t[3].type() + << " as a scalar of type " << type_of() << "\n"; + + Func f; + f() = t; + Realization r = f.realize(); + *a = Image(r[0])(0); + *b = Image(r[1])(0); + *c = Image(r[2])(0); + *d = Image(r[3])(0); } - -template -NO_INLINE void evaluate(Tuple t, A *a, B *b, C *c, D *d) { - user_assert(t[0].type() == type_of()) - << "Can't evaluate expression " - << t[0] << " of type " << t[0].type() - << " as a scalar of type " << type_of() << "\n"; - user_assert(t[1].type() == type_of()) - << "Can't evaluate expression " - << t[1] << " of type " << t[1].type() - << " as a scalar of type " << type_of() << "\n"; - user_assert(t[2].type() == type_of()) - << "Can't evaluate expression " - << t[2] << " of type " << t[2].type() - << " as a scalar of type " << type_of() << "\n"; - user_assert(t[3].type() == type_of()) - << "Can't evaluate expression " - << t[3] << " of type " << t[3].type() - << " as a scalar of type " << type_of() << "\n"; - - Func f; - f() = t; - Realization r = f.realize(); - *a = Image(r[0])(0); - *b = Image(r[1])(0); - *c = Image(r[2])(0); - *d = Image(r[3])(0); -} - // @} +// @} /** JIT-Compile and run enough code to evaluate a Halide @@ -9755,107 +9712,80 @@ NO_INLINE void evaluate(Tuple t, A *a, B *b, C *c, D *d) { * \ref Func::realize. Can use GPU if jit target from environment * specifies one. */ -template -NO_INLINE T evaluate_may_gpu(Expr e) { - user_assert(e.type() == type_of()) - << "Can't evaluate expression " - << e << " of type " << e.type() - << " as a scalar of type " << type_of() << "\n"; - bool has_gpu_feature = get_jit_target_from_environment().has_gpu_feature(); - Func f; - f() = e; - if (has_gpu_feature) { - f.gpu_single_thread(); - } - Image im = f.realize(); - return im(0); +template NO_INLINE T evaluate_may_gpu(Expr e) +{ + user_assert(e.type() == type_of()) << "Can't evaluate expression " << e << " of type " << e.type() + << " as a scalar of type " << type_of() << "\n"; + bool has_gpu_feature = get_jit_target_from_environment().has_gpu_feature(); + Func f; + f() = e; + if (has_gpu_feature) { f.gpu_single_thread(); } + Image im = f.realize(); + return im(0); } /** JIT-compile and run enough code to evaluate a Halide Tuple. Can * use GPU if jit target from environment specifies one. */ // @{ -template -NO_INLINE void evaluate_may_gpu(Tuple t, A *a, B *b) { - user_assert(t[0].type() == type_of()) - << "Can't evaluate expression " - << t[0] << " of type " << t[0].type() - << " as a scalar of type " << type_of() << "\n"; - user_assert(t[1].type() == type_of()) - << "Can't evaluate expression " - << t[1] << " of type " << t[1].type() - << " as a scalar of type " << type_of() << "\n"; - - bool has_gpu_feature = get_jit_target_from_environment().has_gpu_feature(); - Func f; - f() = t; - if (has_gpu_feature) { - f.gpu_single_thread(); - } - Realization r = f.realize(); - *a = Image(r[0])(0); - *b = Image(r[1])(0); -} - -template -NO_INLINE void evaluate_may_gpu(Tuple t, A *a, B *b, C *c) { - user_assert(t[0].type() == type_of()) - << "Can't evaluate expression " - << t[0] << " of type " << t[0].type() - << " as a scalar of type " << type_of() << "\n"; - user_assert(t[1].type() == type_of()) - << "Can't evaluate expression " - << t[1] << " of type " << t[1].type() - << " as a scalar of type " << type_of() << "\n"; - user_assert(t[2].type() == type_of()) - << "Can't evaluate expression " - << t[2] << " of type " << t[2].type() - << " as a scalar of type " << type_of() << "\n"; - bool has_gpu_feature = get_jit_target_from_environment().has_gpu_feature(); - Func f; - f() = t; - if (has_gpu_feature) { - f.gpu_single_thread(); - } - Realization r = f.realize(); - *a = Image(r[0])(0); - *b = Image(r[1])(0); - *c = Image(r[2])(0); +template NO_INLINE void evaluate_may_gpu(Tuple t, A *a, B *b) +{ + user_assert(t[0].type() == type_of()) << "Can't evaluate expression " << t[0] << " of type " << t[0].type() + << " as a scalar of type " << type_of() << "\n"; + user_assert(t[1].type() == type_of()) << "Can't evaluate expression " << t[1] << " of type " << t[1].type() + << " as a scalar of type " << type_of() << "\n"; + + bool has_gpu_feature = get_jit_target_from_environment().has_gpu_feature(); + Func f; + f() = t; + if (has_gpu_feature) { f.gpu_single_thread(); } + Realization r = f.realize(); + *a = Image(r[0])(0); + *b = Image(r[1])(0); +} + +template NO_INLINE void evaluate_may_gpu(Tuple t, A *a, B *b, C *c) +{ + user_assert(t[0].type() == type_of()) << "Can't evaluate expression " << t[0] << " of type " << t[0].type() + << " as a scalar of type " << type_of() << "\n"; + user_assert(t[1].type() == type_of()) << "Can't evaluate expression " << t[1] << " of type " << t[1].type() + << " as a scalar of type " << type_of() << "\n"; + user_assert(t[2].type() == type_of()) << "Can't evaluate expression " << t[2] << " of type " << t[2].type() + << " as a scalar of type " << type_of() << "\n"; + bool has_gpu_feature = get_jit_target_from_environment().has_gpu_feature(); + Func f; + f() = t; + if (has_gpu_feature) { f.gpu_single_thread(); } + Realization r = f.realize(); + *a = Image(r[0])(0); + *b = Image(r[1])(0); + *c = Image(r[2])(0); } template -NO_INLINE void evaluate_may_gpu(Tuple t, A *a, B *b, C *c, D *d) { - user_assert(t[0].type() == type_of()) - << "Can't evaluate expression " - << t[0] << " of type " << t[0].type() - << " as a scalar of type " << type_of() << "\n"; - user_assert(t[1].type() == type_of()) - << "Can't evaluate expression " - << t[1] << " of type " << t[1].type() - << " as a scalar of type " << type_of() << "\n"; - user_assert(t[2].type() == type_of()) - << "Can't evaluate expression " - << t[2] << " of type " << t[2].type() - << " as a scalar of type " << type_of() << "\n"; - user_assert(t[3].type() == type_of()) - << "Can't evaluate expression " - << t[3] << " of type " << t[3].type() - << " as a scalar of type " << type_of() << "\n"; - - bool has_gpu_feature = get_jit_target_from_environment().has_gpu_feature(); - Func f; - f() = t; - if (has_gpu_feature) { - f.gpu_single_thread(); - } - Realization r = f.realize(); - *a = Image(r[0])(0); - *b = Image(r[1])(0); - *c = Image(r[2])(0); - *d = Image(r[3])(0); +NO_INLINE void evaluate_may_gpu(Tuple t, A *a, B *b, C *c, D *d) +{ + user_assert(t[0].type() == type_of()) << "Can't evaluate expression " << t[0] << " of type " << t[0].type() + << " as a scalar of type " << type_of() << "\n"; + user_assert(t[1].type() == type_of()) << "Can't evaluate expression " << t[1] << " of type " << t[1].type() + << " as a scalar of type " << type_of() << "\n"; + user_assert(t[2].type() == type_of()) << "Can't evaluate expression " << t[2] << " of type " << t[2].type() + << " as a scalar of type " << type_of() << "\n"; + user_assert(t[3].type() == type_of()) << "Can't evaluate expression " << t[3] << " of type " << t[3].type() + << " as a scalar of type " << type_of() << "\n"; + + bool has_gpu_feature = get_jit_target_from_environment().has_gpu_feature(); + Func f; + f() = t; + if (has_gpu_feature) { f.gpu_single_thread(); } + Realization r = f.realize(); + *a = Image(r[0])(0); + *b = Image(r[1])(0); + *c = Image(r[2])(0); + *d = Image(r[3])(0); } // @} -} +}// namespace Halide #endif @@ -9925,7 +9855,7 @@ EXPORT Tuple argmax(RDom, Expr, const std::string &s = "argmax"); EXPORT Tuple argmin(RDom, Expr, const std::string &s = "argmin"); // @} -} +}// namespace Halide #endif #ifndef HALIDE_INTEGER_DIVISION_TABLE_H @@ -9937,25 +9867,25 @@ EXPORT Tuple argmin(RDom, Expr, const std::string &s = "argmin"); */ namespace Halide { namespace Internal { -namespace IntegerDivision { - -extern int64_t table_u8[256][4]; -extern int64_t table_s8[256][4]; -extern int64_t table_u16[256][4]; -extern int64_t table_s16[256][4]; -extern int64_t table_u32[256][4]; -extern int64_t table_s32[256][4]; - -extern int64_t table_runtime_u8[256][4]; -extern int64_t table_runtime_s8[256][4]; -extern int64_t table_runtime_u16[256][4]; -extern int64_t table_runtime_s16[256][4]; -extern int64_t table_runtime_u32[256][4]; -extern int64_t table_runtime_s32[256][4]; - -} -} -} + namespace IntegerDivision { + + extern int64_t table_u8[256][4]; + extern int64_t table_s8[256][4]; + extern int64_t table_u16[256][4]; + extern int64_t table_s16[256][4]; + extern int64_t table_u32[256][4]; + extern int64_t table_s32[256][4]; + + extern int64_t table_runtime_u8[256][4]; + extern int64_t table_runtime_s8[256][4]; + extern int64_t table_runtime_u16[256][4]; + extern int64_t table_runtime_s16[256][4]; + extern int64_t table_runtime_u32[256][4]; + extern int64_t table_runtime_s32[256][4]; + + }// namespace IntegerDivision +}// namespace Internal +}// namespace Halide #endif #ifndef HALIDE_IR_EQUALITY_H @@ -9969,87 +9899,93 @@ extern int64_t table_runtime_s32[256][4]; namespace Halide { namespace Internal { -/** A compare struct suitable for use in std::map and std::set that - * computes a lexical ordering on IR nodes. */ -struct IRDeepCompare { + /** A compare struct suitable for use in std::map and std::set that + * computes a lexical ordering on IR nodes. */ + struct IRDeepCompare + { EXPORT bool operator()(const Expr &a, const Expr &b) const; EXPORT bool operator()(const Stmt &a, const Stmt &b) const; -}; - -/** Lossily track known equal exprs with a cache. On collision, the - * old pair is evicted. Used below by ExprWithCompareCache. */ -class IRCompareCache { -private: - struct Entry { - Expr a, b; + }; + + /** Lossily track known equal exprs with a cache. On collision, the + * old pair is evicted. Used below by ExprWithCompareCache. */ + class IRCompareCache + { + private: + struct Entry + { + Expr a, b; }; int bits; - uint32_t hash(const Expr &a, const Expr &b) const { - // Note this hash is symmetric in a and b, so that a - // comparison in a and b hashes to the same bucket as - // a comparison on b and a. - uint64_t pa = (uint64_t)(a.ptr); - uint64_t pb = (uint64_t)(b.ptr); - pa ^= pb; - pa ^= pa >> bits; - pa ^= pa >> (bits*2); - return pa & ((1 << bits) - 1); + uint32_t hash(const Expr &a, const Expr &b) const + { + // Note this hash is symmetric in a and b, so that a + // comparison in a and b hashes to the same bucket as + // a comparison on b and a. + uint64_t pa = (uint64_t)(a.ptr); + uint64_t pb = (uint64_t)(b.ptr); + pa ^= pb; + pa ^= pa >> bits; + pa ^= pa >> (bits * 2); + return pa & ((1 << bits) - 1); } std::vector entries; -public: - void insert(const Expr &a, const Expr &b) { - uint32_t h = hash(a, b); - entries[h].a = a; - entries[h].b = b; + public: + void insert(const Expr &a, const Expr &b) + { + uint32_t h = hash(a, b); + entries[h].a = a; + entries[h].b = b; } - bool contains(const Expr &a, const Expr &b) const { - uint32_t h = hash(a, b); - const Entry &e = entries[h]; - return ((a.same_as(e.a) && b.same_as(e.b)) || - (a.same_as(e.b) && b.same_as(e.a))); + bool contains(const Expr &a, const Expr &b) const + { + uint32_t h = hash(a, b); + const Entry &e = entries[h]; + return ((a.same_as(e.a) && b.same_as(e.b)) || (a.same_as(e.b) && b.same_as(e.a))); } - void clear() { - for (size_t i = 0; i < entries.size(); i++) { - entries[i].a = Expr(); - entries[i].b = Expr(); - } + void clear() + { + for (size_t i = 0; i < entries.size(); i++) { + entries[i].a = Expr(); + entries[i].b = Expr(); + } } IRCompareCache() {} IRCompareCache(int b) : bits(b), entries(1 << bits) {} - -}; - -/** A wrapper about Exprs so that they can be deeply compared with a - * cache for known-equal subexpressions. Useful for unsanitized Exprs - * coming in from the front-end, which may be horrible graphs with - * sub-expressions that are equal by value but not by identity. This - * isn't a comparison object like IRDeepCompare above, because libc++ - * requires that comparison objects be stateless (and constructs a new - * one for each comparison!), so they can't have a cache associated - * with them. However, by sneakily making the cache a mutable member - * of the objects being compared, we can dodge this issue. - * - * Clunky example usage: - * -\code -Expr a, b, c, query; -std::set s; -IRCompareCache cache(8); -s.insert(ExprWithCompareCache(a, &cache)); -s.insert(ExprWithCompareCache(b, &cache)); -s.insert(ExprWithCompareCache(c, &cache)); -if (m.contains(ExprWithCompareCache(query, &cache))) {...} -\endcode - * - */ -struct ExprWithCompareCache { + }; + + /** A wrapper about Exprs so that they can be deeply compared with a + * cache for known-equal subexpressions. Useful for unsanitized Exprs + * coming in from the front-end, which may be horrible graphs with + * sub-expressions that are equal by value but not by identity. This + * isn't a comparison object like IRDeepCompare above, because libc++ + * requires that comparison objects be stateless (and constructs a new + * one for each comparison!), so they can't have a cache associated + * with them. However, by sneakily making the cache a mutable member + * of the objects being compared, we can dodge this issue. + * + * Clunky example usage: + * + \code + Expr a, b, c, query; + std::set s; + IRCompareCache cache(8); + s.insert(ExprWithCompareCache(a, &cache)); + s.insert(ExprWithCompareCache(b, &cache)); + s.insert(ExprWithCompareCache(c, &cache)); + if (m.contains(ExprWithCompareCache(query, &cache))) {...} + \endcode + * + */ + struct ExprWithCompareCache + { Expr expr; mutable IRCompareCache *cache; @@ -10058,19 +9994,19 @@ struct ExprWithCompareCache { /** The comparison uses (and updates) the cache */ EXPORT bool operator<(const ExprWithCompareCache &other) const; -}; + }; -/** Compare IR nodes for equality of value. Traverses entire IR - * tree. For equality of reference, use Expr::same_as */ -// @{ -EXPORT bool equal(Expr a, Expr b); -EXPORT bool equal(Stmt a, Stmt b); -// @} + /** Compare IR nodes for equality of value. Traverses entire IR + * tree. For equality of reference, use Expr::same_as */ + // @{ + EXPORT bool equal(Expr a, Expr b); + EXPORT bool equal(Stmt a, Stmt b); + // @} -EXPORT void ir_equality_test(); + EXPORT void ir_equality_test(); -} -} +}// namespace Internal +}// namespace Halide #endif #ifndef HALIDE_IR_MATCH_H @@ -10086,41 +10022,41 @@ EXPORT void ir_equality_test(); namespace Halide { namespace Internal { -/** Does the first expression have the same structure as the second? - * Variables in the first expression with the name * are interpreted - * as wildcards, and their matching equivalent in the second - * expression is placed in the vector give as the third argument. - * - * For example: - \code - Expr x = new Variable(Int(32), "*"); - match(x + x, 3 + (2*k), result) - \endcode - * should return true, and set result[0] to 3 and - * result[1] to 2*k. - */ - -bool expr_match(Expr pattern, Expr expr, std::vector &result); - -/** Does the first expression have the same structure as the second? - * Variables are matched consistently. The first time a variable is - * matched, it assumes the value of the matching part of the second - * expression. Subsequent matches must be equal to the first match. - * - * For example: - \code - Var x("x"), y("y"); - match(x*(x + 1), a*(a + b), result) - \endcode - * should return true, and set result["x"] = a, and result["y"] = b. - */ - -bool expr_match(Expr pattern, Expr expr, std::map &result); - -void expr_match_test(); - -} -} + /** Does the first expression have the same structure as the second? + * Variables in the first expression with the name * are interpreted + * as wildcards, and their matching equivalent in the second + * expression is placed in the vector give as the third argument. + * + * For example: + \code + Expr x = new Variable(Int(32), "*"); + match(x + x, 3 + (2*k), result) + \endcode + * should return true, and set result[0] to 3 and + * result[1] to 2*k. + */ + + bool expr_match(Expr pattern, Expr expr, std::vector &result); + + /** Does the first expression have the same structure as the second? + * Variables are matched consistently. The first time a variable is + * matched, it assumes the value of the matching part of the second + * expression. Subsequent matches must be equal to the first match. + * + * For example: + \code + Var x("x"), y("y"); + match(x*(x + 1), a*(a + b), result) + \endcode + * should return true, and set result["x"] = a, and result["y"] = b. + */ + + bool expr_match(Expr pattern, Expr expr, std::map &result); + + void expr_match_test(); + +}// namespace Internal +}// namespace Halide #endif #ifndef HALIDE_IR_MUTATOR_H @@ -10134,20 +10070,20 @@ void expr_match_test(); namespace Halide { namespace Internal { -/** A base class for passes over the IR which modify it - * (e.g. replacing a variable with a value (Substitute.h), or - * constant-folding). - * - * Your mutate should override the visit methods you care about. Return - * the new expression by assigning to expr or stmt. The default ones - * recursively mutate their children. To mutate sub-expressions and - * sub-statements you should the mutate method, which will dispatch to - * the appropriate visit method and then return the value of expr or - * stmt after the call to visit. - */ -class IRMutator : public IRVisitor { -public: - + /** A base class for passes over the IR which modify it + * (e.g. replacing a variable with a value (Substitute.h), or + * constant-folding). + * + * Your mutate should override the visit methods you care about. Return + * the new expression by assigning to expr or stmt. The default ones + * recursively mutate their children. To mutate sub-expressions and + * sub-statements you should the mutate method, which will dispatch to + * the appropriate visit method and then return the value of expr or + * stmt after the call to visit. + */ + class IRMutator : public IRVisitor + { + public: /** This is the main interface for using a mutator. Also call * these in your subclass to mutate sub-expressions and * sub-statements. @@ -10155,8 +10091,7 @@ class IRMutator : public IRVisitor { virtual Expr mutate(Expr expr); virtual Stmt mutate(Stmt stmt); -protected: - + protected: /** visit methods that take Exprs assign to this to return their * new value */ Expr expr; @@ -10204,10 +10139,10 @@ class IRMutator : public IRVisitor { virtual void visit(const Block *); virtual void visit(const IfThenElse *); virtual void visit(const Evaluate *); -}; + }; -} -} +}// namespace Internal +}// namespace Halide #endif #ifndef FIND_CALLS_H @@ -10218,29 +10153,29 @@ class IRMutator : public IRVisitor { * Defines analyses to extract the functions called a function. */ -#include #include +#include namespace Halide { namespace Internal { -/** Construct a map from name to Function definition object for all Halide - * functions called directly in the definition of the Function f, including - * in update definitions, update index expressions, and RDom extents. This map - * _does not_ include the Function f, unless it is called recursively by - * itself. - */ -std::map find_direct_calls(Function f); + /** Construct a map from name to Function definition object for all Halide + * functions called directly in the definition of the Function f, including + * in update definitions, update index expressions, and RDom extents. This map + * _does not_ include the Function f, unless it is called recursively by + * itself. + */ + std::map find_direct_calls(Function f); -/** Construct a map from name to Function definition object for all Halide - * functions called directly in the definition of the Function f, or - * indirectly in those functions' definitions, recursively. This map always - * _includes_ the Function f. - */ -std::map find_transitive_calls(Function f); + /** Construct a map from name to Function definition object for all Halide + * functions called directly in the definition of the Function f, or + * indirectly in those functions' definitions, recursively. This map always + * _includes_ the Function f. + */ + std::map find_transitive_calls(Function f); -} -} +}// namespace Internal +}// namespace Halide #endif #ifndef HALIDE_LAMBDA_H @@ -10256,65 +10191,71 @@ namespace Halide { /** Create a zero-dimensional halide function that returns the given * expression. The function may have more dimensions if the expression * contains implicit arguments. */ -inline Func lambda(Expr e) { - Func f("lambda" + Internal::unique_name('_')); - f(_) = e; - return f; +inline Func lambda(Expr e) +{ + Func f("lambda" + Internal::unique_name('_')); + f(_) = e; + return f; } /** Create a 1-D halide function in the first argument that returns * the second argument. The function may have more dimensions if the * expression contains implicit arguments and the list of Var * arguments contains a placeholder ("_"). */ -inline Func lambda(Var x, Expr e) { - Func f("lambda" + Internal::unique_name('_')); - f(x) = e; - return f; +inline Func lambda(Var x, Expr e) +{ + Func f("lambda" + Internal::unique_name('_')); + f(x) = e; + return f; } /** Create a 2-D halide function in the first two arguments that * returns the last argument. The function may have more dimensions if * the expression contains implicit arguments and the list of Var * arguments contains a placeholder ("_"). */ -inline Func lambda(Var x, Var y, Expr e) { - Func f("lambda" + Internal::unique_name('_')); - f(x, y) = e; - return f; +inline Func lambda(Var x, Var y, Expr e) +{ + Func f("lambda" + Internal::unique_name('_')); + f(x, y) = e; + return f; } /** Create a 3-D halide function in the first three arguments that * returns the last argument. The function may have more dimensions * if the expression contains implicit arguments and the list of Var * arguments contains a placeholder ("_"). */ -inline Func lambda(Var x, Var y, Var z, Expr e) { - Func f("lambda" + Internal::unique_name('_')); - f(x, y, z) = e; - return f; +inline Func lambda(Var x, Var y, Var z, Expr e) +{ + Func f("lambda" + Internal::unique_name('_')); + f(x, y, z) = e; + return f; } /** Create a 4-D halide function in the first four arguments that * returns the last argument. The function may have more dimensions if * the expression contains implicit arguments and the list of Var * arguments contains a placeholder ("_"). */ -inline Func lambda(Var x, Var y, Var z, Var w, Expr e) { - Func f("lambda" + Internal::unique_name('_')); - f(x, y, z, w) = e; - return f; +inline Func lambda(Var x, Var y, Var z, Var w, Expr e) +{ + Func f("lambda" + Internal::unique_name('_')); + f(x, y, z, w) = e; + return f; } /** Create a 5-D halide function in the first five arguments that * returns the last argument. The function may have more dimensions if * the expression contains implicit arguments and the list of Var * arguments contains a placeholder ("_"). */ -inline Func lambda(Var x, Var y, Var z, Var w, Var v, Expr e) { - Func f("lambda" + Internal::unique_name('_')); - f(x, y, z, w, v) = e; - return f; +inline Func lambda(Var x, Var y, Var z, Var w, Var v, Expr e) +{ + Func f("lambda" + Internal::unique_name('_')); + f(x, y, z, w, v) = e; + return f; } -} +}// namespace Halide -#endif //HALIDE_LAMBDA_H +#endif// HALIDE_LAMBDA_H #ifndef HALIDE_INTERNAL_LOWER_H #define HALIDE_INTERNAL_LOWER_H @@ -10328,15 +10269,15 @@ inline Func lambda(Var x, Var y, Var z, Var w, Var v, Expr e) { namespace Halide { namespace Internal { -/** Given a halide function with a schedule, create a statement that - * evaluates it. Automatically pulls in all the functions f depends - * on. Some stages of lowering may be target-specific. */ -Stmt lower(Function f, const Target &t); + /** Given a halide function with a schedule, create a statement that + * evaluates it. Automatically pulls in all the functions f depends + * on. Some stages of lowering may be target-specific. */ + Stmt lower(Function f, const Target &t); -void lower_test(); + void lower_test(); -} -} +}// namespace Internal +}// namespace Halide #endif /** \file @@ -10420,12 +10361,12 @@ void lower_test(); namespace Halide { namespace Internal { -/** Convert for loops of size 1 into LetStmt nodes, which allows for - * further simplification. Done during a late stage of lowering. */ -Stmt remove_trivial_for_loops(Stmt s); + /** Convert for loops of size 1 into LetStmt nodes, which allows for + * further simplification. Done during a late stage of lowering. */ + Stmt remove_trivial_for_loops(Stmt s); -} -} +}// namespace Internal +}// namespace Halide #endif #ifndef HALIDE_SIMPLIFY_H @@ -10438,90 +10379,90 @@ Stmt remove_trivial_for_loops(Stmt s); #include namespace Halide { -namespace Internal { - -/** Perform a a wide range of simplifications to expressions and - * statements, including constant folding, substituting in trivial - * values, arithmetic rearranging, etc. Simplifies across let - * statements, so must not be called on stmts with dangling or - * repeated variable names. - */ -// @{ -EXPORT Stmt simplify(Stmt, bool simplify_lets = true, - const Scope &bounds = Scope::empty_scope(), - const Scope &alignment = Scope::empty_scope()); -EXPORT Expr simplify(Expr, bool simplify_lets = true, - const Scope &bounds = Scope::empty_scope(), - const Scope &alignment = Scope::empty_scope()); -// @} - -/** Simplify expressions found in a statement, but don't simplify - * across different statements. This is safe to perform at an earlier - * stage in lowering than full simplification of a stmt. */ -EXPORT Stmt simplify_exprs(Stmt); +namespace Internal { -/** Implementations of division and mod that are specific to Halide. - * Use these implementations; do not use native C division or mod to - * simplify Halide expressions. Halide division and modulo satisify - * the Euclidean definition of division for integers a and b: - * - /code - (a/b)*b + a%b = a - 0 <= a%b < |b| - /endcode - * - */ -// @{ -template -inline T mod_imp(T a, T b) { + /** Perform a a wide range of simplifications to expressions and + * statements, including constant folding, substituting in trivial + * values, arithmetic rearranging, etc. Simplifies across let + * statements, so must not be called on stmts with dangling or + * repeated variable names. + */ + // @{ + EXPORT Stmt simplify(Stmt, + bool simplify_lets = true, + const Scope &bounds = Scope::empty_scope(), + const Scope &alignment = Scope::empty_scope()); + EXPORT Expr simplify(Expr, + bool simplify_lets = true, + const Scope &bounds = Scope::empty_scope(), + const Scope &alignment = Scope::empty_scope()); + // @} + + /** Simplify expressions found in a statement, but don't simplify + * across different statements. This is safe to perform at an earlier + * stage in lowering than full simplification of a stmt. */ + EXPORT Stmt simplify_exprs(Stmt); + + /** Implementations of division and mod that are specific to Halide. + * Use these implementations; do not use native C division or mod to + * simplify Halide expressions. Halide division and modulo satisify + * the Euclidean definition of division for integers a and b: + * + /code + (a/b)*b + a%b = a + 0 <= a%b < |b| + /endcode + * + */ + // @{ + template inline T mod_imp(T a, T b) + { Type t = type_of(); if (t.is_int()) { - T r = a % b; - r = r + (r < 0 ? (T)std::abs((int)b) : 0); - return r; + T r = a % b; + r = r + (r < 0 ? (T)std::abs((int)b) : 0); + return r; } else { - return a % b; + return a % b; } -} + } -template -inline T div_imp(T a, T b) { + template inline T div_imp(T a, T b) + { Type t = type_of(); if (t.is_int()) { - int q = a / b; - int r = a - q * b; - int bs = b >> (t.bits - 1); - int rs = r >> (t.bits - 1); - return q - (rs & bs) + (rs & ~bs); + int q = a / b; + int r = a - q * b; + int bs = b >> (t.bits - 1); + int rs = r >> (t.bits - 1); + return q - (rs & bs) + (rs & ~bs); } else { - return a / b; + return a / b; } -} -// @} + } + // @} -// Special cases for float, double. -template<> inline float mod_imp(float a, float b) { + // Special cases for float, double. + template<> inline float mod_imp(float a, float b) + { float f = a - b * (floorf(a / b)); // The remainder has the same sign as b. return f; -} -template<> inline double mod_imp(double a, double b) { + } + template<> inline double mod_imp(double a, double b) + { double f = a - b * (std::floor(a / b)); return f; -} + } -template<> inline float div_imp(float a, float b) { - return a/b; -} -template<> inline double div_imp(double a, double b) { - return a/b; -} + template<> inline float div_imp(float a, float b) { return a / b; } + template<> inline double div_imp(double a, double b) { return a / b; } -EXPORT void simplify_test(); + EXPORT void simplify_test(); -} -} +}// namespace Internal +}// namespace Halide #endif #ifndef HALIDE_SLIDING_WINDOW_H @@ -10538,14 +10479,14 @@ EXPORT void simplify_test(); namespace Halide { namespace Internal { -/** Perform sliding window optimizations on a halide - * statement. I.e. don't bother computing points in a function that - * have provably already been computed by a previous iteration. - */ -Stmt sliding_window(Stmt s, const std::map &env); + /** Perform sliding window optimizations on a halide + * statement. I.e. don't bother computing points in a function that + * have provably already been computed by a previous iteration. + */ + Stmt sliding_window(Stmt s, const std::map &env); -} -} +}// namespace Internal +}// namespace Halide #endif #ifndef HALIDE_STMT_COMPILER_H @@ -10564,14 +10505,15 @@ namespace Halide { namespace Internal { -/** A handle to a generic statement compiler. Can take Halide - * statements and turn them into assembly, bitcode, machine code, or a - * jit-compiled module. */ -class CodeGen; -class StmtCompiler { + /** A handle to a generic statement compiler. Can take Halide + * statements and turn them into assembly, bitcode, machine code, or a + * jit-compiled module. */ + class CodeGen; + class StmtCompiler + { IntrusivePtr contents; -public: + public: /** Build a code generator for the given target. */ StmtCompiler(Target target); @@ -10579,9 +10521,10 @@ class StmtCompiler { * the given toplevel arguments, and the given buffers embedded * inside it. The module is stored internally until one of the * later functions is called: */ - void compile(Stmt stmt, std::string name, - const std::vector &args, - const std::vector &images_to_embed); + void compile(Stmt stmt, + std::string name, + const std::vector &args, + const std::vector &images_to_embed); /** Write the module to an llvm bitcode file */ void compile_to_bitcode(const std::string &filename); @@ -10602,11 +10545,10 @@ class StmtCompiler { * fails. */ JITCompiledModule compile_to_function_pointers(); -}; - -} -} + }; +}// namespace Internal +}// namespace Halide #endif @@ -10624,14 +10566,13 @@ class StmtCompiler { namespace Halide { namespace Internal { -/** Take a statement with multi-dimensional Realize, Provide, and Call - * nodes, and turn it into a statement with single-dimensional - * Allocate, Store, and Load nodes respectively. */ -Stmt storage_flattening(Stmt s, const std::string &output, - const std::map &env); + /** Take a statement with multi-dimensional Realize, Provide, and Call + * nodes, and turn it into a statement with single-dimensional + * Allocate, Store, and Load nodes respectively. */ + Stmt storage_flattening(Stmt s, const std::string &output, const std::map &env); -} -} +}// namespace Internal +}// namespace Halide #endif #ifndef HALIDE_STORAGE_FOLDING_H @@ -10646,23 +10587,23 @@ Stmt storage_flattening(Stmt s, const std::string &output, namespace Halide { namespace Internal { -/** Fold storage of functions if possible. This means reducing one of - * the dimensions module something for the purpose of storage, if we - * can prove that this is safe to do. E.g consider: - * - \code - f(x) = ... - g(x) = f(x-1) + f(x) - f.store_root().compute_at(g, x); - \endcode - * - * We can store f as a circular buffer of size two, instead of - * allocating space for all of it. - */ -Stmt storage_folding(Stmt s); - -} -} + /** Fold storage of functions if possible. This means reducing one of + * the dimensions module something for the purpose of storage, if we + * can prove that this is safe to do. E.g consider: + * + \code + f(x) = ... + g(x) = f(x-1) + f(x) + f.store_root().compute_at(g, x); + \endcode + * + * We can store f as a circular buffer of size two, instead of + * allocating space for all of it. + */ + Stmt storage_folding(Stmt s); + +}// namespace Internal +}// namespace Halide #endif #ifndef HALIDE_SUBSTITUTE_H @@ -10679,32 +10620,32 @@ Stmt storage_folding(Stmt s); namespace Halide { namespace Internal { -/** Substitute variables with the given name with the replacement - * expression within expr. This is a dangerous thing to do if variable - * names have not been uniquified. While it won't traverse inside let - * statements with the same name as the first argument, moving a piece - * of syntax around can change its meaning, because it can cross lets - * that redefine variable names that it includes references to. */ -EXPORT Expr substitute(std::string name, Expr replacement, Expr expr); - -/** Substitute variables with the given name with the replacement - * expression within stmt. */ -EXPORT Stmt substitute(std::string name, Expr replacement, Stmt stmt); - -/** Substitute variables with names in the map. */ -// @{ -EXPORT Expr substitute(const std::map &replacements, Expr expr); -EXPORT Stmt substitute(const std::map &replacements, Stmt stmt); -// @} - -/** Substitute expressions for other expressions. */ -// @{ -EXPORT Expr substitute(Expr find, Expr replacement, Expr expr); -EXPORT Stmt substitute(Expr find, Expr replacement, Stmt stmt); -// @} - -} -} + /** Substitute variables with the given name with the replacement + * expression within expr. This is a dangerous thing to do if variable + * names have not been uniquified. While it won't traverse inside let + * statements with the same name as the first argument, moving a piece + * of syntax around can change its meaning, because it can cross lets + * that redefine variable names that it includes references to. */ + EXPORT Expr substitute(std::string name, Expr replacement, Expr expr); + + /** Substitute variables with the given name with the replacement + * expression within stmt. */ + EXPORT Stmt substitute(std::string name, Expr replacement, Stmt stmt); + + /** Substitute variables with names in the map. */ + // @{ + EXPORT Expr substitute(const std::map &replacements, Expr expr); + EXPORT Stmt substitute(const std::map &replacements, Stmt stmt); + // @} + + /** Substitute expressions for other expressions. */ + // @{ + EXPORT Expr substitute(Expr find, Expr replacement, Expr expr); + EXPORT Stmt substitute(Expr find, Expr replacement, Stmt stmt); + // @} + +}// namespace Internal +}// namespace Halide #endif #ifndef HALIDE_PROFILING_H @@ -10718,27 +10659,27 @@ EXPORT Stmt substitute(Expr find, Expr replacement, Stmt stmt); namespace Halide { namespace Internal { -/** Take a statement representing a halide pipeline, and (depending on - * the environment variable HL_PROFILE), insert high-resolution timing - * into the generated code; summaries of execution times and counts will - * be logged at the end. Should be done before storage flattening, but - * after all bounds inference. Use util/HalideProf to analyze the output. - * - * NOTE: this makes no effort to provide accurate or useful information - * when parallelization is scheduled; more work would need to be done - * to safely record data from multiple threads. - * - * NOTE: this makes no effort to account for overhead from the profiling - * instructions inserted; profile-enabled runtimes will be slower, - * and inner loops will be more profoundly affected. - */ -Stmt inject_profiling(Stmt, std::string); - -/** Gets the current profiling level (by reading HL_PROFILE) */ -int profiling_level(); - -} -} + /** Take a statement representing a halide pipeline, and (depending on + * the environment variable HL_PROFILE), insert high-resolution timing + * into the generated code; summaries of execution times and counts will + * be logged at the end. Should be done before storage flattening, but + * after all bounds inference. Use util/HalideProf to analyze the output. + * + * NOTE: this makes no effort to provide accurate or useful information + * when parallelization is scheduled; more work would need to be done + * to safely record data from multiple threads. + * + * NOTE: this makes no effort to account for overhead from the profiling + * instructions inserted; profile-enabled runtimes will be slower, + * and inner loops will be more profoundly affected. + */ + Stmt inject_profiling(Stmt, std::string); + + /** Gets the current profiling level (by reading HL_PROFILE) */ + int profiling_level(); + +}// namespace Internal +}// namespace Halide #endif #ifndef HALIDE_TRACING_H @@ -10753,14 +10694,14 @@ int profiling_level(); namespace Halide { namespace Internal { -/** Take a statement representing a halide pipeline, inject calls to - * tracing functions at interesting points, such as - * allocations. Should be done before storage flattening, but after - * all bounds inference. */ -Stmt inject_tracing(Stmt, const std::map &env, Function output); + /** Take a statement representing a halide pipeline, inject calls to + * tracing functions at interesting points, such as + * allocations. Should be done before storage flattening, but after + * all bounds inference. */ + Stmt inject_tracing(Stmt, const std::map &env, Function output); -} -} +}// namespace Internal +}// namespace Halide #endif #ifndef HALIDE_UNROLL_LOOPS_H @@ -10774,13 +10715,13 @@ Stmt inject_tracing(Stmt, const std::map &env, Function o namespace Halide { namespace Internal { -/** Take a statement with for loops marked for unrolling, and convert - * each into several copies of the innermost statement. I.e. unroll - * the loop. */ -Stmt unroll_loops(Stmt); + /** Take a statement with for loops marked for unrolling, and convert + * each into several copies of the innermost statement. I.e. unroll + * the loop. */ + Stmt unroll_loops(Stmt); -} -} +}// namespace Internal +}// namespace Halide #endif #ifndef HALIDE_VECTORIZE_LOOPS_H @@ -10794,14 +10735,14 @@ Stmt unroll_loops(Stmt); namespace Halide { namespace Internal { -/** Take a statement with for loops marked for vectorization, and turn - * them into single statements that operate on vectors. The loops in - * question must have constant extent. - */ -Stmt vectorize_loops(Stmt); + /** Take a statement with for loops marked for vectorization, and turn + * them into single statements that operate on vectors. The loops in + * question must have constant extent. + */ + Stmt vectorize_loops(Stmt); -} -} +}// namespace Internal +}// namespace Halide #endif #ifndef HALIDE_DEBUG_TO_FILE_H @@ -10816,14 +10757,14 @@ Stmt vectorize_loops(Stmt); namespace Halide { namespace Internal { -/** Takes a statement with Realize nodes still unlowered. If the - * corresponding functions have a debug_file set, then inject code - * that will dump the contents of those functions to a file after the - * realization. */ -Stmt debug_to_file(Stmt s, std::string output, const std::map &env); + /** Takes a statement with Realize nodes still unlowered. If the + * corresponding functions have a debug_file set, then inject code + * that will dump the contents of those functions to a file after the + * realization. */ + Stmt debug_to_file(Stmt s, std::string output, const std::map &env); -} -} +}// namespace Internal +}// namespace Halide #endif #ifndef HALIDE_EARLY_FREE_H @@ -10839,14 +10780,14 @@ Stmt debug_to_file(Stmt s, std::string output, const std::map &order); + /** Avoid computing certain stages if we can infer a runtime condition + * to check that tells us they won't be used. Does this by aanalyzing + * all reads of each buffer allocated, and inferring some condition + * that tells us if the reads occur. If the condition is non-trivial, + * inject ifs that guard the production. */ + Stmt skip_stages(Stmt s, const std::vector &order); -} -} +}// namespace Internal +}// namespace Halide #endif #ifndef HALIDE_SPECIALIZE_CLAMPED_RAMPS_H @@ -10953,13 +10894,13 @@ Stmt skip_stages(Stmt s, const std::vector &order); namespace Halide { namespace Internal { -/** Take a statement with multi-dimensional Realize, Provide, and Call - * nodes, and turn it into a statement with single-dimensional - * Allocate, Store, and Load nodes respectively. */ -Stmt specialize_clamped_ramps(Stmt s); + /** Take a statement with multi-dimensional Realize, Provide, and Call + * nodes, and turn it into a statement with single-dimensional + * Allocate, Store, and Load nodes respectively. */ + Stmt specialize_clamped_ramps(Stmt s); -} -} +}// namespace Internal +}// namespace Halide #endif #ifndef HALIDE_REMOVE_UNDEF @@ -10973,12 +10914,12 @@ Stmt specialize_clamped_ramps(Stmt s); namespace Halide { namespace Internal { -/** Removes stores that depend on undef values, and statements that - * only contain such stores. */ -Stmt remove_undef(Stmt s); + /** Removes stores that depend on undef values, and statements that + * only contain such stores. */ + Stmt remove_undef(Stmt s); -} -} +}// namespace Internal +}// namespace Halide #endif #ifndef HALIDE_FAST_INTEGER_DIVIDE_H @@ -10992,13 +10933,13 @@ namespace Halide { * in your object file. They are declared here in case you want to do * something non-default with them. */ namespace IntegerDivideTable { -EXPORT Image integer_divide_table_u8(); -EXPORT Image integer_divide_table_s8(); -EXPORT Image integer_divide_table_u16(); -EXPORT Image integer_divide_table_s16(); -EXPORT Image integer_divide_table_u32(); -EXPORT Image integer_divide_table_s32(); -} + EXPORT Image integer_divide_table_u8(); + EXPORT Image integer_divide_table_s8(); + EXPORT Image integer_divide_table_u16(); + EXPORT Image integer_divide_table_s16(); + EXPORT Image integer_divide_table_u32(); + EXPORT Image integer_divide_table_s32(); +}// namespace IntegerDivideTable /** Integer division by small values can be done exactly as multiplies @@ -11023,7 +10964,7 @@ EXPORT Image integer_divide_table_s32(); */ EXPORT Expr fast_integer_divide(Expr numerator, Expr denominator); -} +}// namespace Halide #endif #ifndef HALIDE_ALLOCATION_BOUNDS_INFERENCE_H @@ -11040,20 +10981,19 @@ EXPORT Expr fast_integer_divide(Expr numerator, Expr denominator); namespace Halide { namespace Internal { -/** Take a partially statement with Realize nodes in terms of - * variables, and define values for those variables. */ -Stmt allocation_bounds_inference(Stmt s, - const std::map &env, - const std::map, Interval> &func_bounds); -} -} + /** Take a partially statement with Realize nodes in terms of + * variables, and define values for those variables. */ + Stmt allocation_bounds_inference(Stmt s, + const std::map &env, + const std::map, Interval> &func_bounds); +}// namespace Internal +}// namespace Halide #endif #ifndef HALIDE_INLINE_H #define HALIDE_INLINE_H - /** \file * Methods for replacing calls to functions with their definitions. */ @@ -11061,14 +11001,14 @@ Stmt allocation_bounds_inference(Stmt s, namespace Halide { namespace Internal { -/** Inline a single named function, which must be pure. */ -// @{ -Stmt inline_function(Stmt, Function); -Expr inline_function(Expr, Function); -// @} + /** Inline a single named function, which must be pure. */ + // @{ + Stmt inline_function(Stmt, Function); + Expr inline_function(Expr, Function); + // @} -} -} +}// namespace Internal +}// namespace Halide #endif @@ -11084,11 +11024,11 @@ Expr inline_function(Expr, Function); namespace Halide { namespace Internal { -/** Prefix all variable names in the given expression with the prefix string. */ -Expr qualify(const std::string &prefix, Expr value); + /** Prefix all variable names in the given expression with the prefix string. */ + Expr qualify(const std::string &prefix, Expr value); -} -} +}// namespace Internal +}// namespace Halide #endif @@ -11103,12 +11043,12 @@ Expr qualify(const std::string &prefix, Expr value); namespace Halide { namespace Internal { -/** Find let statements that all define the same value, and make later - * ones just reuse the symbol names of the earlier ones. */ -Stmt unify_duplicate_lets(Stmt s); + /** Find let statements that all define the same value, and make later + * ones just reuse the symbol names of the earlier ones. */ + Stmt unify_duplicate_lets(Stmt s); -} -} +}// namespace Internal +}// namespace Halide #endif #ifndef HALIDE_EXPR_USES_VAR_H @@ -11122,42 +11062,42 @@ Stmt unify_duplicate_lets(Stmt s); namespace Halide { namespace Internal { -/** Test if an expression references the given variable. */ -bool expr_uses_var(Expr e, const std::string &v); + /** Test if an expression references the given variable. */ + bool expr_uses_var(Expr e, const std::string &v); -template -class ExprUsesVars : public IRVisitor { + template class ExprUsesVars : public IRVisitor + { using IRVisitor::visit; const Scope &scope; - void visit(const Variable *v) { - if (scope.contains(v->name)) { - result = true; - } + void visit(const Variable *v) + { + if (scope.contains(v->name)) { result = true; } } -public: + + public: ExprUsesVars(const Scope &s) : scope(s), result(false) {} bool result; -}; + }; -/** Test if an expression references any of the variables in a scope. */ -template -inline bool expr_uses_vars(Expr e, const Scope &s) { + /** Test if an expression references any of the variables in a scope. */ + template inline bool expr_uses_vars(Expr e, const Scope &s) + { ExprUsesVars uses(s); e.accept(&uses); return uses.result; -} + } -} -} +}// namespace Internal +}// namespace Halide #endif #ifndef HALIDE_RANDOM_H #define HALIDE_RANDOM_H -#include #include +#include /** \file * @@ -11168,22 +11108,22 @@ inline bool expr_uses_vars(Expr e, const Scope &s) { namespace Halide { namespace Internal { -/** Return a random floating-point number between zero and one that - * varies deterministically based on the input expressions. */ -Expr random_float(const std::vector &); + /** Return a random floating-point number between zero and one that + * varies deterministically based on the input expressions. */ + Expr random_float(const std::vector &); -/** Return a random unsigned integer between zero and 2^32-1 that - * varies deterministically based on the input expressions (which must - * be integers). */ -Expr random_int(const std::vector &); + /** Return a random unsigned integer between zero and 2^32-1 that + * varies deterministically based on the input expressions (which must + * be integers). */ + Expr random_int(const std::vector &); -/** Convert calls to random() to IR generated by random_float and - * random_int. Tags all calls with the variables in free_vars, and the - * integer given as the last argument. */ -Expr lower_random(Expr e, const std::vector &free_vars, int tag); + /** Convert calls to random() to IR generated by random_float and + * random_int. Tags all calls with the variables in free_vars, and the + * integer given as the last argument. */ + Expr lower_random(Expr e, const std::vector &free_vars, int tag); -} -} +}// namespace Internal +}// namespace Halide #endif #ifndef HALIDE_CODEGEN_OPENGL_DEV_H @@ -11194,48 +11134,46 @@ Expr lower_random(Expr e, const std::vector &free_vars, int tag); */ -#include #include +#include #include namespace Halide { namespace Internal { -class CodeGen_GLSL; + class CodeGen_GLSL; -class CodeGen_OpenGL_Dev : public CodeGen_GPU_Dev { -public: + class CodeGen_OpenGL_Dev : public CodeGen_GPU_Dev + { + public: CodeGen_OpenGL_Dev(const Target &target); ~CodeGen_OpenGL_Dev(); // CodeGen_GPU_Dev interface - void add_kernel(Stmt stmt, const std::string &name, - const std::vector &args); + void add_kernel(Stmt stmt, const std::string &name, const std::vector &args); void init_module(); std::vector compile_to_src(); std::string get_current_kernel_name(); void dump(); -private: + private: CodeGen_GLSL *glc; std::ostringstream src_stream; std::string cur_kernel_name; Target target; -}; + }; -/** Compile one statement into GLSL. */ -class CodeGen_GLSL : public CodeGen_C { -public: + /** Compile one statement into GLSL. */ + class CodeGen_GLSL : public CodeGen_C + { + public: CodeGen_GLSL(std::ostream &s); - void compile(Stmt stmt, - std::string name, - const std::vector &args, - const Target &target); + void compile(Stmt stmt, std::string name, const std::vector &args, const Target &target); EXPORT static void test(); -protected: + protected: using CodeGen_C::visit; std::string print_type(Type type); std::string print_name(const std::string &); @@ -11260,13 +11198,14 @@ class CodeGen_GLSL : public CodeGen_C { void visit(const Evaluate *); -private: + private: std::string get_vector_suffix(Expr e); std::map builtin; -}; + }; -}} +}// namespace Internal +}// namespace Halide #endif #ifndef HALIDE_INJECT_OPENGL_INTRINSICS_H @@ -11282,13 +11221,13 @@ class CodeGen_GLSL : public CodeGen_C { namespace Halide { namespace Internal { -/** Take a statement with for kernel for loops and turn loads and - * stores inside the loops into OpenGL texture load and store - * intrinsics. Should only be run when the OpenGL target is active. */ -Stmt inject_opengl_intrinsics(Stmt s); + /** Take a statement with for kernel for loops and turn loads and + * stores inside the loops into OpenGL texture load and store + * intrinsics. Should only be run when the OpenGL target is active. */ + Stmt inject_opengl_intrinsics(Stmt s); -} -} +}// namespace Internal +}// namespace Halide #endif #ifndef HALIDE_SYNCTHREADS_H @@ -11303,19 +11242,19 @@ Stmt inject_opengl_intrinsics(Stmt s); namespace Halide { namespace Internal { -/** Rewrite all GPU loops to have a min of zero. */ -Stmt zero_gpu_loop_mins(Stmt s); + /** Rewrite all GPU loops to have a min of zero. */ + Stmt zero_gpu_loop_mins(Stmt s); -/** Converts Halide's GPGPU IR to the OpenCL/CUDA model. Within every - * loop over gpu block indices, fuse the inner loops over thread - * indices into a single loop (with predication to turn off - * threads). Also injects synchronization points as needed, and hoists - * allocations at the block level out into a single shared memory - * array. */ -Stmt fuse_gpu_thread_loops(Stmt s); + /** Converts Halide's GPGPU IR to the OpenCL/CUDA model. Within every + * loop over gpu block indices, fuse the inner loops over thread + * indices into a single loop (with predication to turn off + * threads). Also injects synchronization points as needed, and hoists + * allocations at the block level out into a single shared memory + * array. */ + Stmt fuse_gpu_thread_loops(Stmt s); -} -} +}// namespace Internal +}// namespace Halide #endif #ifndef HALIDE_HOST_GPU_BUFFER_COPIES_H @@ -11329,15 +11268,15 @@ Stmt fuse_gpu_thread_loops(Stmt s); namespace Halide { namespace Internal { -/** Inject calls to halide_dev_malloc, halide_copy_to_dev, and - * halide_copy_to_host as needed. */ -Stmt inject_host_dev_buffer_copies(Stmt s); + /** Inject calls to halide_dev_malloc, halide_copy_to_dev, and + * halide_copy_to_host as needed. */ + Stmt inject_host_dev_buffer_copies(Stmt s); -/** Inject calls to halide_dev_free as needed. */ -Stmt inject_dev_frees(Stmt s); + /** Inject calls to halide_dev_free as needed. */ + Stmt inject_dev_frees(Stmt s); -} -} +}// namespace Internal +}// namespace Halide #endif #ifndef HALIDE_PARALLEL_RVAR_H @@ -11354,17 +11293,15 @@ Stmt inject_dev_frees(Stmt s); namespace Halide { namespace Internal { -/** Returns whether or not Halide can prove that it is safe to - * parallelize an update definition across a specific variable. If - * this returns true, it's definitely safe. If this returns false, it - * may still be safe, but Halide couldn't prove it. - */ -bool can_parallelize_rvar(const std::string &rvar, - const std::string &func, - const UpdateDefinition &r); + /** Returns whether or not Halide can prove that it is safe to + * parallelize an update definition across a specific variable. If + * this returns true, it's definitely safe. If this returns false, it + * may still be safe, but Halide couldn't prove it. + */ + bool can_parallelize_rvar(const std::string &rvar, const std::string &func, const UpdateDefinition &r); -} -} +}// namespace Internal +}// namespace Halide #endif #ifndef HALIDE_BOUNDARY_CONDITIONS_H @@ -11409,145 +11346,165 @@ namespace Halide { */ namespace BoundaryConditions { -namespace Internal { + namespace Internal { -#if __cplusplus > 199711L // C++11 arbitrary number of args support +#if __cplusplus > 199711L// C++11 arbitrary number of args support -inline NO_INLINE void collect_bounds(std::vector > &collected_bounds, - Expr min, Expr extent) { - collected_bounds.push_back(std::make_pair(min, extent)); -} + inline NO_INLINE void collect_bounds(std::vector> &collected_bounds, Expr min, Expr extent) + { + collected_bounds.push_back(std::make_pair(min, extent)); + } -template -inline NO_INLINE void collect_bounds(std::vector > &collected_bounds, - Expr min, Expr extent, Bounds... bounds) { - collected_bounds.push_back(std::make_pair(min, extent)); - collect_bounds(collected_bounds, bounds...); -} + template + inline NO_INLINE void + collect_bounds(std::vector> &collected_bounds, Expr min, Expr extent, Bounds... bounds) + { + collected_bounds.push_back(std::make_pair(min, extent)); + collect_bounds(collected_bounds, bounds...); + } -#endif // C++11 support. +#endif// C++11 support. -inline const Func &func_like_to_func(const Func &func) { - return func; -} + inline const Func &func_like_to_func(const Func &func) { return func; } -template -inline NO_INLINE Func func_like_to_func(T func_like) { - std::vector args; - for (int i = 0; i < func_like.dimensions(); i++) { - args.push_back(Var::implicit(i)); + template inline NO_INLINE Func func_like_to_func(T func_like) + { + std::vector args; + for (int i = 0; i < func_like.dimensions(); i++) { args.push_back(Var::implicit(i)); } + Func func; + func(args) = func_like(args); + return func; } - Func func; - func(args) = func_like(args); - return func; -} -} + }// namespace Internal -/** Impose a boundary condition such that a given expression is returned - * everywhere outside the boundary. Generally the expression will be a - * constant, though the code currently allows accessing the arguments - * of source. - * - * An ImageParam, Image, or similar can be passed instead of a Func. If this - * is done and no bounds are given, the boundaries will be taken from the - * min and extent methods of the passed object. - * - * (This is similar to setting GL_TEXTURE_WRAP_* to GL_CLAMP_TO_BORDER - * and putting value in the border of the texture.) - */ -// @{ -EXPORT Func constant_exterior(const Func &source, Expr value, - const std::vector > &bounds); + /** Impose a boundary condition such that a given expression is returned + * everywhere outside the boundary. Generally the expression will be a + * constant, though the code currently allows accessing the arguments + * of source. + * + * An ImageParam, Image, or similar can be passed instead of a Func. If this + * is done and no bounds are given, the boundaries will be taken from the + * min and extent methods of the passed object. + * + * (This is similar to setting GL_TEXTURE_WRAP_* to GL_CLAMP_TO_BORDER + * and putting value in the border of the texture.) + */ + // @{ + EXPORT Func constant_exterior(const Func &source, Expr value, const std::vector> &bounds); -template -inline NO_INLINE Func constant_exterior(T func_like, Expr value) { - std::vector > object_bounds; + template inline NO_INLINE Func constant_exterior(T func_like, Expr value) + { + std::vector> object_bounds; for (int i = 0; i < func_like.dimensions(); i++) { - object_bounds.push_back(std::make_pair(Expr(func_like.min(i)), Expr(func_like.extent(i)))); + object_bounds.push_back(std::make_pair(Expr(func_like.min(i)), Expr(func_like.extent(i)))); } return constant_exterior(Internal::func_like_to_func(func_like), value, object_bounds); -} + } -#if __cplusplus > 199711L // C++11 arbitrary number of args support -template -inline NO_INLINE Func constant_exterior(T func_like, Expr value, - Bounds... bounds) { - std::vector > collected_bounds; +#if __cplusplus > 199711L// C++11 arbitrary number of args support + template + inline NO_INLINE Func constant_exterior(T func_like, Expr value, Bounds... bounds) + { + std::vector> collected_bounds; Internal::collect_bounds(collected_bounds, bounds...); return constant_exterior(Internal::func_like_to_func(func_like), value, collected_bounds); -} + } #else -template -inline NO_INLINE Func constant_exterior(T func_like, Expr value, - Expr min0, Expr extent0) { - std::vector > bounds; + template inline NO_INLINE Func constant_exterior(T func_like, Expr value, Expr min0, Expr extent0) + { + std::vector> bounds; bounds.push_back(std::make_pair(min0, extent0)); return constant_exterior(Internal::func_like_to_func(func_like), value, bounds); -} + } -template -inline NO_INLINE Func constant_exterior(T func_like, Expr value, - Expr min0, Expr extent0, - Expr min1, Expr extent1) { - std::vector > bounds; + template + inline NO_INLINE Func constant_exterior(T func_like, Expr value, Expr min0, Expr extent0, Expr min1, Expr extent1) + { + std::vector> bounds; bounds.push_back(std::make_pair(min0, extent0)); bounds.push_back(std::make_pair(min1, extent1)); return constant_exterior(Internal::func_like_to_func(func_like), value, bounds); -} + } -template -inline NO_INLINE Func constant_exterior(T func_like, Expr value, - Expr min0, Expr extent0, - Expr min1, Expr extent1, - Expr min2, Expr extent2) { - std::vector > bounds; + template + inline NO_INLINE Func constant_exterior(T func_like, + Expr value, + Expr min0, + Expr extent0, + Expr min1, + Expr extent1, + Expr min2, + Expr extent2) + { + std::vector> bounds; bounds.push_back(std::make_pair(min0, extent0)); bounds.push_back(std::make_pair(min1, extent1)); bounds.push_back(std::make_pair(min2, extent2)); return constant_exterior(Internal::func_like_to_func(func_like), value, bounds); -} + } -template -inline NO_INLINE Func constant_exterior(T func_like, Expr value, - Expr min0, Expr extent0, - Expr min1, Expr extent1, - Expr min2, Expr extent2, - Expr min3, Expr extent3) { - std::vector > bounds; + template + inline NO_INLINE Func constant_exterior(T func_like, + Expr value, + Expr min0, + Expr extent0, + Expr min1, + Expr extent1, + Expr min2, + Expr extent2, + Expr min3, + Expr extent3) + { + std::vector> bounds; bounds.push_back(std::make_pair(min0, extent0)); bounds.push_back(std::make_pair(min1, extent1)); bounds.push_back(std::make_pair(min2, extent2)); bounds.push_back(std::make_pair(min3, extent3)); return constant_exterior(Internal::func_like_to_func(func_like), value, bounds); -} + } -template -inline NO_INLINE Func constant_exterior(T func_like, Expr value, - Expr min0, Expr extent0, - Expr min1, Expr extent1, - Expr min2, Expr extent2, - Expr min3, Expr extent3, - Expr min4, Expr extent4) { - std::vector > bounds; + template + inline NO_INLINE Func constant_exterior(T func_like, + Expr value, + Expr min0, + Expr extent0, + Expr min1, + Expr extent1, + Expr min2, + Expr extent2, + Expr min3, + Expr extent3, + Expr min4, + Expr extent4) + { + std::vector> bounds; bounds.push_back(std::make_pair(min0, extent0)); bounds.push_back(std::make_pair(min1, extent1)); bounds.push_back(std::make_pair(min2, extent2)); bounds.push_back(std::make_pair(min3, extent3)); bounds.push_back(std::make_pair(min4, extent4)); return constant_exterior(Internal::func_like_to_func(func_like), value, bounds); -} + } -template -inline NO_INLINE Func constant_exterior(T func_like, Expr value, - Expr min0, Expr extent0, - Expr min1, Expr extent1, - Expr min2, Expr extent2, - Expr min3, Expr extent3, - Expr min4, Expr extent4, - Expr min5, Expr extent5) { - std::vector > bounds; + template + inline NO_INLINE Func constant_exterior(T func_like, + Expr value, + Expr min0, + Expr extent0, + Expr min1, + Expr extent1, + Expr min2, + Expr extent2, + Expr min3, + Expr extent3, + Expr min4, + Expr extent4, + Expr min5, + Expr extent5) + { + std::vector> bounds; bounds.push_back(std::make_pair(min0, extent0)); bounds.push_back(std::make_pair(min1, extent1)); bounds.push_back(std::make_pair(min2, extent2)); @@ -11555,111 +11512,124 @@ inline NO_INLINE Func constant_exterior(T func_like, Expr value, bounds.push_back(std::make_pair(min4, extent4)); bounds.push_back(std::make_pair(min5, extent5)); return constant_exterior(Internal::func_like_to_func(func_like), value, bounds); -} + } #endif -// @} - -/** Impose a boundary condition such that the nearest edge sample is returned - * everywhere outside the given region. - * - * An ImageParam, Image, or similar can be passed instead of a Func. If this - * is done and no bounds are given, the boundaries will be taken from the - * min and extent methods of the passed object. - * - * (This is similar to setting GL_TEXTURE_WRAP_* to GL_CLAMP_TO_EDGE.) - */ -// @{ -EXPORT Func repeat_edge(const Func &source, - const std::vector > &bounds); - -template -inline NO_INLINE Func repeat_edge(T func_like) { - std::vector > object_bounds; + // @} + + /** Impose a boundary condition such that the nearest edge sample is returned + * everywhere outside the given region. + * + * An ImageParam, Image, or similar can be passed instead of a Func. If this + * is done and no bounds are given, the boundaries will be taken from the + * min and extent methods of the passed object. + * + * (This is similar to setting GL_TEXTURE_WRAP_* to GL_CLAMP_TO_EDGE.) + */ + // @{ + EXPORT Func repeat_edge(const Func &source, const std::vector> &bounds); + + template inline NO_INLINE Func repeat_edge(T func_like) + { + std::vector> object_bounds; for (int i = 0; i < func_like.dimensions(); i++) { - object_bounds.push_back(std::make_pair(Expr(func_like.min(i)), Expr(func_like.extent(i)))); + object_bounds.push_back(std::make_pair(Expr(func_like.min(i)), Expr(func_like.extent(i)))); } return repeat_edge(Internal::func_like_to_func(func_like), object_bounds); -} + } -#if __cplusplus > 199711L // C++11 arbitrary number of args support -template -inline NO_INLINE Func repeat_edge(T func_like, Bounds... bounds) { - std::vector > collected_bounds; +#if __cplusplus > 199711L// C++11 arbitrary number of args support + template inline NO_INLINE Func repeat_edge(T func_like, Bounds... bounds) + { + std::vector> collected_bounds; Internal::collect_bounds(collected_bounds, bounds...); return repeat_edge(Internal::func_like_to_func(func_like), collected_bounds); -} + } #else -template -inline NO_INLINE Func repeat_edge(T func_like, - Expr min0, Expr extent0) { - std::vector > bounds; + template inline NO_INLINE Func repeat_edge(T func_like, Expr min0, Expr extent0) + { + std::vector> bounds; bounds.push_back(std::make_pair(min0, extent0)); return repeat_edge(Internal::func_like_to_func(func_like), bounds); -} + } -template -inline NO_INLINE Func repeat_edge(T func_like, - Expr min0, Expr extent0, - Expr min1, Expr extent1) { - std::vector > bounds; + template inline NO_INLINE Func repeat_edge(T func_like, Expr min0, Expr extent0, Expr min1, Expr extent1) + { + std::vector> bounds; bounds.push_back(std::make_pair(min0, extent0)); bounds.push_back(std::make_pair(min1, extent1)); return repeat_edge(Internal::func_like_to_func(func_like), bounds); -} + } -template -inline NO_INLINE Func repeat_edge(T func_like, - Expr min0, Expr extent0, - Expr min1, Expr extent1, - Expr min2, Expr extent2) { - std::vector > bounds; + template + inline NO_INLINE Func + repeat_edge(T func_like, Expr min0, Expr extent0, Expr min1, Expr extent1, Expr min2, Expr extent2) + { + std::vector> bounds; bounds.push_back(std::make_pair(min0, extent0)); bounds.push_back(std::make_pair(min1, extent1)); bounds.push_back(std::make_pair(min2, extent2)); return repeat_edge(Internal::func_like_to_func(func_like), bounds); -} + } -template -inline NO_INLINE Func repeat_edge(T func_like, - Expr min0, Expr extent0, - Expr min1, Expr extent1, - Expr min2, Expr extent2, - Expr min3, Expr extent3) { - std::vector > bounds; + template + inline NO_INLINE Func repeat_edge(T func_like, + Expr min0, + Expr extent0, + Expr min1, + Expr extent1, + Expr min2, + Expr extent2, + Expr min3, + Expr extent3) + { + std::vector> bounds; bounds.push_back(std::make_pair(min0, extent0)); bounds.push_back(std::make_pair(min1, extent1)); bounds.push_back(std::make_pair(min2, extent2)); bounds.push_back(std::make_pair(min3, extent3)); return repeat_edge(Internal::func_like_to_func(func_like), bounds); -} + } -template -inline NO_INLINE Func repeat_edge(T func_like, - Expr min0, Expr extent0, - Expr min1, Expr extent1, - Expr min2, Expr extent2, - Expr min3, Expr extent3, - Expr min4, Expr extent4) { - std::vector > bounds; + template + inline NO_INLINE Func repeat_edge(T func_like, + Expr min0, + Expr extent0, + Expr min1, + Expr extent1, + Expr min2, + Expr extent2, + Expr min3, + Expr extent3, + Expr min4, + Expr extent4) + { + std::vector> bounds; bounds.push_back(std::make_pair(min0, extent0)); bounds.push_back(std::make_pair(min1, extent1)); bounds.push_back(std::make_pair(min2, extent2)); bounds.push_back(std::make_pair(min3, extent3)); bounds.push_back(std::make_pair(min4, extent4)); return repeat_edge(Internal::func_like_to_func(func_like), bounds); -} + } -template -inline NO_INLINE Func repeat_edge(T func_like, - Expr min0, Expr extent0, - Expr min1, Expr extent1, - Expr min2, Expr extent2, - Expr min3, Expr extent3, - Expr min4, Expr extent4, - Expr min5, Expr extent5) { - std::vector > bounds; + template + inline NO_INLINE Func repeat_edge(T func_like, + Expr min0, + Expr extent0, + Expr min1, + Expr extent1, + Expr min2, + Expr extent2, + Expr min3, + Expr extent3, + Expr min4, + Expr extent4, + Expr min5, + Expr extent5) + { + std::vector> bounds; bounds.push_back(std::make_pair(min0, extent0)); bounds.push_back(std::make_pair(min1, extent1)); bounds.push_back(std::make_pair(min2, extent2)); @@ -11667,110 +11637,123 @@ inline NO_INLINE Func repeat_edge(T func_like, bounds.push_back(std::make_pair(min4, extent4)); bounds.push_back(std::make_pair(min5, extent5)); return repeat_edge(Internal::func_like_to_func(func_like), bounds); -} + } #endif -// @} - -/** Impose a boundary condition such that the entire coordinate space is - * tiled with copies of the image abutted against each other. - * - * An ImageParam, Image, or similar can be passed instead of a Func. If this - * is done and no bounds are given, the boundaries will be taken from the - * min and extent methods of the passed object. - * - * (This is similar to setting GL_TEXTURE_WRAP_* to GL_REPEAT.) - */ -// @{ -EXPORT Func repeat_image(const Func &source, - const std::vector > &bounds); - -template -inline NO_INLINE Func repeat_image(T func_like) { - std::vector > object_bounds; + // @} + + /** Impose a boundary condition such that the entire coordinate space is + * tiled with copies of the image abutted against each other. + * + * An ImageParam, Image, or similar can be passed instead of a Func. If this + * is done and no bounds are given, the boundaries will be taken from the + * min and extent methods of the passed object. + * + * (This is similar to setting GL_TEXTURE_WRAP_* to GL_REPEAT.) + */ + // @{ + EXPORT Func repeat_image(const Func &source, const std::vector> &bounds); + + template inline NO_INLINE Func repeat_image(T func_like) + { + std::vector> object_bounds; for (int i = 0; i < func_like.dimensions(); i++) { - object_bounds.push_back(std::make_pair(Expr(func_like.min(i)), Expr(func_like.extent(i)))); + object_bounds.push_back(std::make_pair(Expr(func_like.min(i)), Expr(func_like.extent(i)))); } return repeat_image(Internal::func_like_to_func(func_like), object_bounds); -} + } -#if __cplusplus > 199711L // C++11 arbitrary number of args support -template -inline NO_INLINE Func repeat_image(T func_like, Bounds... bounds) { - std::vector > collected_bounds; +#if __cplusplus > 199711L// C++11 arbitrary number of args support + template inline NO_INLINE Func repeat_image(T func_like, Bounds... bounds) + { + std::vector> collected_bounds; Internal::collect_bounds(collected_bounds, bounds...); return repeat_image(Internal::func_like_to_func(func_like), collected_bounds); -} + } #else -template -inline NO_INLINE Func repeat_image(T func_like, - Expr min0, Expr extent0) { - std::vector > bounds; + template inline NO_INLINE Func repeat_image(T func_like, Expr min0, Expr extent0) + { + std::vector> bounds; bounds.push_back(std::make_pair(min0, extent0)); return repeat_image(Internal::func_like_to_func(func_like), bounds); -} + } -template -inline NO_INLINE Func repeat_image(T func_like, - Expr min0, Expr extent0, - Expr min1, Expr extent1) { - std::vector > bounds; + template inline NO_INLINE Func repeat_image(T func_like, Expr min0, Expr extent0, Expr min1, Expr extent1) + { + std::vector> bounds; bounds.push_back(std::make_pair(min0, extent0)); bounds.push_back(std::make_pair(min1, extent1)); return repeat_image(Internal::func_like_to_func(func_like), bounds); -} + } -template -inline NO_INLINE Func repeat_image(T func_like, - Expr min0, Expr extent0, - Expr min1, Expr extent1, - Expr min2, Expr extent2) { - std::vector > bounds; + template + inline NO_INLINE Func + repeat_image(T func_like, Expr min0, Expr extent0, Expr min1, Expr extent1, Expr min2, Expr extent2) + { + std::vector> bounds; bounds.push_back(std::make_pair(min0, extent0)); bounds.push_back(std::make_pair(min1, extent1)); bounds.push_back(std::make_pair(min2, extent2)); return repeat_image(Internal::func_like_to_func(func_like), bounds); -} + } -template -inline NO_INLINE Func repeat_image(T func_like, - Expr min0, Expr extent0, - Expr min1, Expr extent1, - Expr min2, Expr extent2, - Expr min3, Expr extent3) { - std::vector > bounds; + template + inline NO_INLINE Func repeat_image(T func_like, + Expr min0, + Expr extent0, + Expr min1, + Expr extent1, + Expr min2, + Expr extent2, + Expr min3, + Expr extent3) + { + std::vector> bounds; bounds.push_back(std::make_pair(min0, extent0)); bounds.push_back(std::make_pair(min1, extent1)); bounds.push_back(std::make_pair(min2, extent2)); bounds.push_back(std::make_pair(min3, extent3)); return repeat_image(Internal::func_like_to_func(func_like), bounds); -} + } -template -inline NO_INLINE Func repeat_image(T func_like, - Expr min0, Expr extent0, - Expr min1, Expr extent1, - Expr min2, Expr extent2, - Expr min3, Expr extent3, - Expr min4, Expr extent4) { - std::vector > bounds; + template + inline NO_INLINE Func repeat_image(T func_like, + Expr min0, + Expr extent0, + Expr min1, + Expr extent1, + Expr min2, + Expr extent2, + Expr min3, + Expr extent3, + Expr min4, + Expr extent4) + { + std::vector> bounds; bounds.push_back(std::make_pair(min0, extent0)); bounds.push_back(std::make_pair(min1, extent1)); bounds.push_back(std::make_pair(min2, extent2)); bounds.push_back(std::make_pair(min3, extent3)); bounds.push_back(std::make_pair(min4, extent4)); return repeat_image(Internal::func_like_to_func(func_like), bounds); -} + } -template -inline NO_INLINE Func repeat_image(T func_like, - Expr min0, Expr extent0, - Expr min1, Expr extent1, - Expr min2, Expr extent2, - Expr min3, Expr extent3, - Expr min4, Expr extent4, - Expr min5, Expr extent5) { - std::vector > bounds; + template + inline NO_INLINE Func repeat_image(T func_like, + Expr min0, + Expr extent0, + Expr min1, + Expr extent1, + Expr min2, + Expr extent2, + Expr min3, + Expr extent3, + Expr min4, + Expr extent4, + Expr min5, + Expr extent5) + { + std::vector> bounds; bounds.push_back(std::make_pair(min0, extent0)); bounds.push_back(std::make_pair(min1, extent1)); bounds.push_back(std::make_pair(min2, extent2)); @@ -11778,110 +11761,123 @@ inline NO_INLINE Func repeat_image(T func_like, bounds.push_back(std::make_pair(min4, extent4)); bounds.push_back(std::make_pair(min5, extent5)); return repeat_image(Internal::func_like_to_func(func_like), bounds); -} + } #endif -/** Impose a boundary condition such that the entire coordinate space is - * tiled with copies of the image abutted against each other, but mirror - * them such that adjacent edges are the same. - * - * An ImageParam, Image, or similar can be passed instead of a Func. If this - * is done and no bounds are given, the boundaries will be taken from the - * min and extent methods of the passed object. - * - * (This is similar to setting GL_TEXTURE_WRAP_* to GL_MIRRORED_REPEAT.) - */ -// @{ -EXPORT Func mirror_image(const Func &source, - const std::vector > &bounds); - -template -inline NO_INLINE Func mirror_image(T func_like) { - std::vector > object_bounds; + /** Impose a boundary condition such that the entire coordinate space is + * tiled with copies of the image abutted against each other, but mirror + * them such that adjacent edges are the same. + * + * An ImageParam, Image, or similar can be passed instead of a Func. If this + * is done and no bounds are given, the boundaries will be taken from the + * min and extent methods of the passed object. + * + * (This is similar to setting GL_TEXTURE_WRAP_* to GL_MIRRORED_REPEAT.) + */ + // @{ + EXPORT Func mirror_image(const Func &source, const std::vector> &bounds); + + template inline NO_INLINE Func mirror_image(T func_like) + { + std::vector> object_bounds; for (int i = 0; i < func_like.dimensions(); i++) { - object_bounds.push_back(std::make_pair(Expr(func_like.min(i)), Expr(func_like.extent(i)))); + object_bounds.push_back(std::make_pair(Expr(func_like.min(i)), Expr(func_like.extent(i)))); } return mirror_image(Internal::func_like_to_func(func_like), object_bounds); -} + } -#if __cplusplus > 199711L // C++11 arbitrary number of args support -template -inline NO_INLINE Func mirror_image(T func_like, Bounds... bounds) { - std::vector > collected_bounds; +#if __cplusplus > 199711L// C++11 arbitrary number of args support + template inline NO_INLINE Func mirror_image(T func_like, Bounds... bounds) + { + std::vector> collected_bounds; Internal::collect_bounds(collected_bounds, bounds...); return mirror_image(Internal::func_like_to_func(func_like), collected_bounds); -} + } #else -template -inline NO_INLINE Func mirror_image(T func_like, - Expr min0, Expr extent0) { - std::vector > bounds; + template inline NO_INLINE Func mirror_image(T func_like, Expr min0, Expr extent0) + { + std::vector> bounds; bounds.push_back(std::make_pair(min0, extent0)); return mirror_image(Internal::func_like_to_func(func_like), bounds); -} + } -template -inline NO_INLINE Func mirror_image(T func_like, - Expr min0, Expr extent0, - Expr min1, Expr extent1) { - std::vector > bounds; + template inline NO_INLINE Func mirror_image(T func_like, Expr min0, Expr extent0, Expr min1, Expr extent1) + { + std::vector> bounds; bounds.push_back(std::make_pair(min0, extent0)); bounds.push_back(std::make_pair(min1, extent1)); return mirror_image(Internal::func_like_to_func(func_like), bounds); -} + } -template -inline NO_INLINE Func mirror_image(T func_like, - Expr min0, Expr extent0, - Expr min1, Expr extent1, - Expr min2, Expr extent2) { - std::vector > bounds; + template + inline NO_INLINE Func + mirror_image(T func_like, Expr min0, Expr extent0, Expr min1, Expr extent1, Expr min2, Expr extent2) + { + std::vector> bounds; bounds.push_back(std::make_pair(min0, extent0)); bounds.push_back(std::make_pair(min1, extent1)); bounds.push_back(std::make_pair(min2, extent2)); return mirror_image(Internal::func_like_to_func(func_like), bounds); -} + } -template -inline NO_INLINE Func mirror_image(T func_like, - Expr min0, Expr extent0, - Expr min1, Expr extent1, - Expr min2, Expr extent2, - Expr min3, Expr extent3) { - std::vector > bounds; + template + inline NO_INLINE Func mirror_image(T func_like, + Expr min0, + Expr extent0, + Expr min1, + Expr extent1, + Expr min2, + Expr extent2, + Expr min3, + Expr extent3) + { + std::vector> bounds; bounds.push_back(std::make_pair(min0, extent0)); bounds.push_back(std::make_pair(min1, extent1)); bounds.push_back(std::make_pair(min2, extent2)); bounds.push_back(std::make_pair(min3, extent3)); return mirror_image(Internal::func_like_to_func(func_like), bounds); -} + } -template -inline NO_INLINE Func mirror_image(T func_like, - Expr min0, Expr extent0, - Expr min1, Expr extent1, - Expr min2, Expr extent2, - Expr min3, Expr extent3, - Expr min4, Expr extent4) { - std::vector > bounds; + template + inline NO_INLINE Func mirror_image(T func_like, + Expr min0, + Expr extent0, + Expr min1, + Expr extent1, + Expr min2, + Expr extent2, + Expr min3, + Expr extent3, + Expr min4, + Expr extent4) + { + std::vector> bounds; bounds.push_back(std::make_pair(min0, extent0)); bounds.push_back(std::make_pair(min1, extent1)); bounds.push_back(std::make_pair(min2, extent2)); bounds.push_back(std::make_pair(min3, extent3)); bounds.push_back(std::make_pair(min4, extent4)); return mirror_image(Internal::func_like_to_func(func_like), bounds); -} + } -template -inline NO_INLINE Func mirror_image(T func_like, - Expr min0, Expr extent0, - Expr min1, Expr extent1, - Expr min2, Expr extent2, - Expr min3, Expr extent3, - Expr min4, Expr extent4, - Expr min5, Expr extent5) { - std::vector > bounds; + template + inline NO_INLINE Func mirror_image(T func_like, + Expr min0, + Expr extent0, + Expr min1, + Expr extent1, + Expr min2, + Expr extent2, + Expr min3, + Expr extent3, + Expr min4, + Expr extent4, + Expr min5, + Expr extent5) + { + std::vector> bounds; bounds.push_back(std::make_pair(min0, extent0)); bounds.push_back(std::make_pair(min1, extent1)); bounds.push_back(std::make_pair(min2, extent2)); @@ -11889,113 +11885,127 @@ inline NO_INLINE Func mirror_image(T func_like, bounds.push_back(std::make_pair(min4, extent4)); bounds.push_back(std::make_pair(min5, extent5)); return mirror_image(Internal::func_like_to_func(func_like), bounds); -} + } #endif -// @} - -/** Impose a boundary condition such that the entire coordinate space is - * tiled with copies of the image abutted against each other, but mirror - * them such that adjacent edges are the same and then overlap the edges. - * - * This produces an error if any extent is 1 or less. (TODO: check this.) - * - * An ImageParam, Image, or similar can be passed instead of a Func. If this - * is done and no bounds are given, the boundaries will be taken from the - * min and extent methods of the passed object. - * - * (I do not believe there is a direct GL_TEXTURE_WRAP_* equivalent for this.) - */ -// @{ -EXPORT Func mirror_interior(const Func &source, - const std::vector > &bounds); - -template -inline NO_INLINE Func mirror_interior(T func_like) { - std::vector > object_bounds; + // @} + + /** Impose a boundary condition such that the entire coordinate space is + * tiled with copies of the image abutted against each other, but mirror + * them such that adjacent edges are the same and then overlap the edges. + * + * This produces an error if any extent is 1 or less. (TODO: check this.) + * + * An ImageParam, Image, or similar can be passed instead of a Func. If this + * is done and no bounds are given, the boundaries will be taken from the + * min and extent methods of the passed object. + * + * (I do not believe there is a direct GL_TEXTURE_WRAP_* equivalent for this.) + */ + // @{ + EXPORT Func mirror_interior(const Func &source, const std::vector> &bounds); + + template inline NO_INLINE Func mirror_interior(T func_like) + { + std::vector> object_bounds; for (int i = 0; i < func_like.dimensions(); i++) { - object_bounds.push_back(std::make_pair(Expr(func_like.min(i)), Expr(func_like.extent(i)))); + object_bounds.push_back(std::make_pair(Expr(func_like.min(i)), Expr(func_like.extent(i)))); } return mirror_interior(Internal::func_like_to_func(func_like), object_bounds); -} + } -#if __cplusplus > 199711L // C++11 arbitrary number of args support -template -inline NO_INLINE Func mirror_interior(T func_like, Bounds... bounds) { - std::vector > collected_bounds; +#if __cplusplus > 199711L// C++11 arbitrary number of args support + template inline NO_INLINE Func mirror_interior(T func_like, Bounds... bounds) + { + std::vector> collected_bounds; Internal::collect_bounds(collected_bounds, bounds...); return mirror_interior(Internal::func_like_to_func(func_like), collected_bounds); -} + } #else -template -inline NO_INLINE Func mirror_interior(T func_like, - Expr min0, Expr extent0) { - std::vector > bounds; + template inline NO_INLINE Func mirror_interior(T func_like, Expr min0, Expr extent0) + { + std::vector> bounds; bounds.push_back(std::make_pair(min0, extent0)); return mirror_interior(Internal::func_like_to_func(func_like), bounds); -} + } -template -inline NO_INLINE Func mirror_interior(T func_like, - Expr min0, Expr extent0, - Expr min1, Expr extent1) { - std::vector > bounds; + template + inline NO_INLINE Func mirror_interior(T func_like, Expr min0, Expr extent0, Expr min1, Expr extent1) + { + std::vector> bounds; bounds.push_back(std::make_pair(min0, extent0)); bounds.push_back(std::make_pair(min1, extent1)); return mirror_interior(Internal::func_like_to_func(func_like), bounds); -} + } -template -inline NO_INLINE Func mirror_interior(T func_like, - Expr min0, Expr extent0, - Expr min1, Expr extent1, - Expr min2, Expr extent2) { - std::vector > bounds; + template + inline NO_INLINE Func + mirror_interior(T func_like, Expr min0, Expr extent0, Expr min1, Expr extent1, Expr min2, Expr extent2) + { + std::vector> bounds; bounds.push_back(std::make_pair(min0, extent0)); bounds.push_back(std::make_pair(min1, extent1)); bounds.push_back(std::make_pair(min2, extent2)); return mirror_interior(Internal::func_like_to_func(func_like), bounds); -} + } -template -inline NO_INLINE Func mirror_interior(T func_like, - Expr min0, Expr extent0, - Expr min1, Expr extent1, - Expr min2, Expr extent2, - Expr min3, Expr extent3) { - std::vector > bounds; + template + inline NO_INLINE Func mirror_interior(T func_like, + Expr min0, + Expr extent0, + Expr min1, + Expr extent1, + Expr min2, + Expr extent2, + Expr min3, + Expr extent3) + { + std::vector> bounds; bounds.push_back(std::make_pair(min0, extent0)); bounds.push_back(std::make_pair(min1, extent1)); bounds.push_back(std::make_pair(min2, extent2)); bounds.push_back(std::make_pair(min3, extent3)); return mirror_interior(Internal::func_like_to_func(func_like), bounds); -} + } -template -inline NO_INLINE Func mirror_interior(T func_like, - Expr min0, Expr extent0, - Expr min1, Expr extent1, - Expr min2, Expr extent2, - Expr min3, Expr extent3, - Expr min4, Expr extent4) { - std::vector > bounds; + template + inline NO_INLINE Func mirror_interior(T func_like, + Expr min0, + Expr extent0, + Expr min1, + Expr extent1, + Expr min2, + Expr extent2, + Expr min3, + Expr extent3, + Expr min4, + Expr extent4) + { + std::vector> bounds; bounds.push_back(std::make_pair(min0, extent0)); bounds.push_back(std::make_pair(min1, extent1)); bounds.push_back(std::make_pair(min2, extent2)); bounds.push_back(std::make_pair(min3, extent3)); bounds.push_back(std::make_pair(min4, extent4)); return mirror_interior(Internal::func_like_to_func(func_like), bounds); -} + } -template -inline NO_INLINE Func mirror_interior(T func_like, - Expr min0, Expr extent0, - Expr min1, Expr extent1, - Expr min2, Expr extent2, - Expr min3, Expr extent3, - Expr min4, Expr extent4, - Expr min5, Expr extent5) { - std::vector > bounds; + template + inline NO_INLINE Func mirror_interior(T func_like, + Expr min0, + Expr extent0, + Expr min1, + Expr extent1, + Expr min2, + Expr extent2, + Expr min3, + Expr extent3, + Expr min4, + Expr extent4, + Expr min5, + Expr extent5) + { + std::vector> bounds; bounds.push_back(std::make_pair(min0, extent0)); bounds.push_back(std::make_pair(min1, extent1)); bounds.push_back(std::make_pair(min2, extent2)); @@ -12003,13 +12013,13 @@ inline NO_INLINE Func mirror_interior(T func_like, bounds.push_back(std::make_pair(min4, extent4)); bounds.push_back(std::make_pair(min5, extent5)); return mirror_interior(Internal::func_like_to_func(func_like), bounds); -} + } #endif -// @} + // @} -} +}// namespace BoundaryConditions -} +}// namespace Halide #endif #ifndef HALIDE_INTERNAL_CACHING_H @@ -12026,41 +12036,40 @@ inline NO_INLINE Func mirror_interior(T func_like, namespace Halide { namespace Internal { -/** Transform pipeline calls for Funcs scheduled with memoize to do a - * lookup call to the runtime cache implementation, and if there is a - * miss, compute the results and call the runtime to store it back to - * the cache. - * Should leave non-memoized Funcs unchanged. - */ -Stmt inject_memoization(Stmt s, const std::map &env, - const std::string &name); + /** Transform pipeline calls for Funcs scheduled with memoize to do a + * lookup call to the runtime cache implementation, and if there is a + * miss, compute the results and call the runtime to store it back to + * the cache. + * Should leave non-memoized Funcs unchanged. + */ + Stmt inject_memoization(Stmt s, const std::map &env, const std::string &name); -} -} +}// namespace Internal +}// namespace Halide #endif #ifndef HALIDE_HUMAN_READABLE_STMT #define HALIDE_HUMAN_READABLE_STMT /** \file -* Defines methods for simplifying a stmt into a human-readable form. -*/ + * Defines methods for simplifying a stmt into a human-readable form. + */ namespace Halide { namespace Internal { -/** - * Returns a Stmt simplified using a concrete size of the output, and - * other optional values for parameters. - */ -// @{ -EXPORT Stmt human_readable_stmt(Function f, Stmt s, Buffer buf); -EXPORT Stmt human_readable_stmt(Function f, Stmt s, Buffer buf, - std::map additional_replacements); -// @} + /** + * Returns a Stmt simplified using a concrete size of the output, and + * other optional values for parameters. + */ + // @{ + EXPORT Stmt human_readable_stmt(Function f, Stmt s, Buffer buf); + EXPORT Stmt human_readable_stmt(Function f, Stmt s, Buffer buf, std::map additional_replacements); + // @} -}} +}// namespace Internal +}// namespace Halide #endif #ifndef HALIDE_STMT_TO_HTML @@ -12074,12 +12083,13 @@ EXPORT Stmt human_readable_stmt(Function f, Stmt s, Buffer buf, namespace Halide { namespace Internal { -/** - * Dump an HTML-formatted print of a Stmt to filename. - */ -EXPORT void print_to_html(std::string filename, Stmt s); + /** + * Dump an HTML-formatted print of a Stmt to filename. + */ + EXPORT void print_to_html(std::string filename, Stmt s); -}} +}// namespace Internal +}// namespace Halide #endif // This file gets included at the end of Halide.h diff --git a/filmulator-gui/Halide/include/HalideRuntime.h b/filmulator-gui/Halide/include/HalideRuntime.h index fd6a6495..d932e539 100644 --- a/filmulator-gui/Halide/include/HalideRuntime.h +++ b/filmulator-gui/Halide/include/HalideRuntime.h @@ -60,7 +60,8 @@ extern void halide_print(void *user_context, const char *); extern void halide_error(void *user_context, const char *); /** A macro that calls halide_error if the supplied condition is false. */ -#define halide_assert(user_context, cond) if (!(cond)) halide_error(user_context, #cond); +#define halide_assert(user_context, cond) \ + if (!(cond)) halide_error(user_context, #cond); /** These are allocated statically inside the runtime, hence the fixed * size. They must be initialized with zero. The first time @@ -69,8 +70,9 @@ extern void halide_error(void *user_context, const char *); * mechanism, but makes the lock reliably easy to setup and use * without depending on e.g. C++ constructor logic. */ -struct halide_mutex { - unsigned char _private[64]; +struct halide_mutex +{ + unsigned char _private[64]; }; /** A basic set of mutex functions, which call platform specific code @@ -93,9 +95,8 @@ extern void halide_mutex_cleanup(struct halide_mutex *mutex_arg); * jobs otherwise. */ //@{ -extern int halide_do_par_for(void *user_context, - int (*f)(void *ctx, int, uint8_t *), - int min, int size, uint8_t *closure); +extern int + halide_do_par_for(void *user_context, int (*f)(void *ctx, int, uint8_t *), int min, int size, uint8_t *closure); extern void halide_shutdown_thread_pool(); //@} @@ -120,32 +121,40 @@ extern void halide_free(void *user_context, void *ptr); * * Cannot be replaced in JITted code at present. */ -extern int32_t halide_debug_to_file(void *user_context, const char *filename, - uint8_t *data, int32_t s0, int32_t s1, int32_t s2, - int32_t s3, int32_t type_code, - int32_t bytes_per_element); - - -enum halide_trace_event_code {halide_trace_load = 0, - halide_trace_store = 1, - halide_trace_begin_realization = 2, - halide_trace_end_realization = 3, - halide_trace_produce = 4, - halide_trace_update = 5, - halide_trace_consume = 6, - halide_trace_end_consume = 7}; - -struct halide_trace_event { - const char *func; - halide_trace_event_code event; - int32_t parent_id; - int32_t type_code; - int32_t bits; - int32_t vector_width; - int32_t value_index; - void *value; - int32_t dimensions; - int32_t *coordinates; +extern int32_t halide_debug_to_file(void *user_context, + const char *filename, + uint8_t *data, + int32_t s0, + int32_t s1, + int32_t s2, + int32_t s3, + int32_t type_code, + int32_t bytes_per_element); + + +enum halide_trace_event_code { + halide_trace_load = 0, + halide_trace_store = 1, + halide_trace_begin_realization = 2, + halide_trace_end_realization = 3, + halide_trace_produce = 4, + halide_trace_update = 5, + halide_trace_consume = 6, + halide_trace_end_consume = 7 +}; + +struct halide_trace_event +{ + const char *func; + halide_trace_event_code event; + int32_t parent_id; + int32_t type_code; + int32_t bits; + int32_t vector_width; + int32_t value_index; + void *value; + int32_t dimensions; + int32_t *coordinates; }; /** Called when Funcs are marked as trace_load, trace_store, or @@ -227,16 +236,19 @@ extern int halide_dev_free(void *user_context, struct buffer_t *buf); * signature across different Halide gpu backends. Do not call * them. */ // @{ -extern int halide_init_kernels(void *user_context, void **state_ptr, - const char *src, int size); +extern int halide_init_kernels(void *user_context, void **state_ptr, const char *src, int size); extern int halide_dev_run(void *user_context, - void *state_ptr, - const char *entry_name, - int blocksX, int blocksY, int blocksZ, - int threadsX, int threadsY, int threadsZ, - int shared_mem_bytes, - size_t arg_sizes[], - void *args[]); + void *state_ptr, + const char *entry_name, + int blocksX, + int blocksY, + int blocksZ, + int threadsX, + int threadsY, + int threadsZ, + int shared_mem_bytes, + size_t arg_sizes[], + void *args[]); // @} /** This function is called to populate the buffer_t.dev field with a constant @@ -311,8 +323,12 @@ extern void halide_memoization_cache_set_size(int64_t size); * return a Tuple, there will only be one buffer_t in the list. The * tuple_count parameters determines the length of the list. */ -extern bool halide_memoization_cache_lookup(void *user_context, const uint8_t *cache_key, int32_t size, - buffer_t *realized_bounds, int32_t tuple_count, buffer_t **tuple_buffers); +extern bool halide_memoization_cache_lookup(void *user_context, + const uint8_t *cache_key, + int32_t size, + buffer_t *realized_bounds, + int32_t tuple_count, + buffer_t **tuple_buffers); /** Given a cache key for a memoized result, currently constructed * from the Func name and top-level Func name plus the arguments of @@ -325,8 +341,12 @@ extern bool halide_memoization_cache_lookup(void *user_context, const uint8_t *c * only be one buffer_t in the list. The tuple_count parameters * determines the length of the list. */ -extern void halide_memoization_cache_store(void *user_context, const uint8_t *cache_key, int32_t size, - buffer_t *realized_bounds, int32_t tuple_count, buffer_t **tuple_buffers); +extern void halide_memoization_cache_store(void *user_context, + const uint8_t *cache_key, + int32_t size, + buffer_t *realized_bounds, + int32_t tuple_count, + buffer_t **tuple_buffers); /** Free all memory and resources associated with the memoization cache. @@ -335,7 +355,7 @@ extern void halide_memoization_cache_store(void *user_context, const uint8_t *ca extern void halide_memoization_cache_cleanup(); #ifdef __cplusplus -} // End extern "C" +}// End extern "C" #endif -#endif // HALIDE_HALIDERUNTIME_H +#endif// HALIDE_HALIDERUNTIME_H diff --git a/filmulator-gui/Halide/include/clock.h b/filmulator-gui/Halide/include/clock.h index 9ddf0370..3eb5f067 100644 --- a/filmulator-gui/Halide/include/clock.h +++ b/filmulator-gui/Halide/include/clock.h @@ -4,26 +4,27 @@ #ifdef _WIN32 extern "C" bool QueryPerformanceCounter(uint64_t *); extern "C" bool QueryPerformanceFrequency(uint64_t *); -double current_time() { - uint64_t t, freq; - QueryPerformanceCounter(&t); - QueryPerformanceFrequency(&freq); - return (t * 1000.0) / freq; +double current_time() +{ + uint64_t t, freq; + QueryPerformanceCounter(&t); + QueryPerformanceFrequency(&freq); + return (t * 1000.0) / freq; } #else #include -double current_time() { - static bool first_call = true; - static timeval reference_time; - if (first_call) { - first_call = false; - gettimeofday(&reference_time, NULL); - return 0.0; - } else { - timeval t; - gettimeofday(&t, NULL); - return ((t.tv_sec - reference_time.tv_sec)*1000.0 + - (t.tv_usec - reference_time.tv_usec)/1000.0); - } +double current_time() +{ + static bool first_call = true; + static timeval reference_time; + if (first_call) { + first_call = false; + gettimeofday(&reference_time, NULL); + return 0.0; + } else { + timeval t; + gettimeofday(&t, NULL); + return ((t.tv_sec - reference_time.tv_sec) * 1000.0 + (t.tv_usec - reference_time.tv_usec) / 1000.0); + } } #endif diff --git a/filmulator-gui/Halide/include/image_io.h b/filmulator-gui/Halide/include/image_io.h index 15964b37..3d7cec72 100644 --- a/filmulator-gui/Halide/include/image_io.h +++ b/filmulator-gui/Halide/include/image_io.h @@ -7,401 +7,410 @@ #ifndef STATIC_IMAGE_LOADER_H #define STATIC_IMAGE_LOADER_H +#include #include -#include #include -#include #include +#include -//#include +// #include -#define _assert(condition, ...) if (!(condition)) {fprintf(stderr, __VA_ARGS__); exit(-1);} +#define _assert(condition, ...) \ + if (!(condition)) { \ + fprintf(stderr, __VA_ARGS__); \ + exit(-1); \ + } // Convert to u8 -inline void convert(uint8_t in, uint8_t &out) {out = in;} -inline void convert(uint16_t in, uint8_t &out) {out = in >> 8;} -inline void convert(uint32_t in, uint8_t &out) {out = in >> 24;} -inline void convert(int8_t in, uint8_t &out) {out = in;} -inline void convert(int16_t in, uint8_t &out) {out = in >> 8;} -inline void convert(int32_t in, uint8_t &out) {out = in >> 24;} -inline void convert(float in, uint8_t &out) {out = (uint8_t)(in*255.0f);} -inline void convert(double in, uint8_t &out) {out = (uint8_t)(in*255.0f);} +inline void convert(uint8_t in, uint8_t &out) { out = in; } +inline void convert(uint16_t in, uint8_t &out) { out = in >> 8; } +inline void convert(uint32_t in, uint8_t &out) { out = in >> 24; } +inline void convert(int8_t in, uint8_t &out) { out = in; } +inline void convert(int16_t in, uint8_t &out) { out = in >> 8; } +inline void convert(int32_t in, uint8_t &out) { out = in >> 24; } +inline void convert(float in, uint8_t &out) { out = (uint8_t)(in * 255.0f); } +inline void convert(double in, uint8_t &out) { out = (uint8_t)(in * 255.0f); } // Convert to u16 -inline void convert(uint8_t in, uint16_t &out) {out = in << 8;} -inline void convert(uint16_t in, uint16_t &out) {out = in;} -inline void convert(uint32_t in, uint16_t &out) {out = in >> 16;} -inline void convert(int8_t in, uint16_t &out) {out = in << 8;} -inline void convert(int16_t in, uint16_t &out) {out = in;} -inline void convert(int32_t in, uint16_t &out) {out = in >> 16;} -inline void convert(float in, uint16_t &out) {out = (uint16_t)(in*65535.0f);} -inline void convert(double in, uint16_t &out) {out = (uint16_t)(in*65535.0f);} +inline void convert(uint8_t in, uint16_t &out) { out = in << 8; } +inline void convert(uint16_t in, uint16_t &out) { out = in; } +inline void convert(uint32_t in, uint16_t &out) { out = in >> 16; } +inline void convert(int8_t in, uint16_t &out) { out = in << 8; } +inline void convert(int16_t in, uint16_t &out) { out = in; } +inline void convert(int32_t in, uint16_t &out) { out = in >> 16; } +inline void convert(float in, uint16_t &out) { out = (uint16_t)(in * 65535.0f); } +inline void convert(double in, uint16_t &out) { out = (uint16_t)(in * 65535.0f); } // Convert from u8 -inline void convert(uint8_t in, uint32_t &out) {out = in << 24;} -inline void convert(uint8_t in, int8_t &out) {out = in;} -inline void convert(uint8_t in, int16_t &out) {out = in << 8;} -inline void convert(uint8_t in, int32_t &out) {out = in << 24;} -inline void convert(uint8_t in, float &out) {out = in/255.0f;} -inline void convert(uint8_t in, double &out) {out = in/255.0f;} +inline void convert(uint8_t in, uint32_t &out) { out = in << 24; } +inline void convert(uint8_t in, int8_t &out) { out = in; } +inline void convert(uint8_t in, int16_t &out) { out = in << 8; } +inline void convert(uint8_t in, int32_t &out) { out = in << 24; } +inline void convert(uint8_t in, float &out) { out = in / 255.0f; } +inline void convert(uint8_t in, double &out) { out = in / 255.0f; } // Convert from u16 -inline void convert(uint16_t in, uint32_t &out) {out = in << 16;} -inline void convert(uint16_t in, int8_t &out) {out = in >> 8;} -inline void convert(uint16_t in, int16_t &out) {out = in;} -inline void convert(uint16_t in, int32_t &out) {out = in << 16;} -inline void convert(uint16_t in, float &out) {out = in/65535.0f;} -inline void convert(uint16_t in, double &out) {out = in/65535.0f;} - - -inline bool ends_with_ignore_case(std::string a, std::string b) { - if (a.length() < b.length()) { return false; } - std::transform(a.begin(), a.end(), a.begin(), ::tolower); - std::transform(b.begin(), b.end(), b.begin(), ::tolower); - return a.compare(a.length()-b.length(), b.length(), b) == 0; +inline void convert(uint16_t in, uint32_t &out) { out = in << 16; } +inline void convert(uint16_t in, int8_t &out) { out = in >> 8; } +inline void convert(uint16_t in, int16_t &out) { out = in; } +inline void convert(uint16_t in, int32_t &out) { out = in << 16; } +inline void convert(uint16_t in, float &out) { out = in / 65535.0f; } +inline void convert(uint16_t in, double &out) { out = in / 65535.0f; } + + +inline bool ends_with_ignore_case(std::string a, std::string b) +{ + if (a.length() < b.length()) { return false; } + std::transform(a.begin(), a.end(), a.begin(), ::tolower); + std::transform(b.begin(), b.end(), b.begin(), ::tolower); + return a.compare(a.length() - b.length(), b.length(), b) == 0; } -template -Image load_png(std::string filename) { - png_byte header[8]; - png_structp png_ptr; - png_infop info_ptr; - png_bytep *row_pointers; +template Image load_png(std::string filename) +{ + png_byte header[8]; + png_structp png_ptr; + png_infop info_ptr; + png_bytep *row_pointers; - /* open file and test for it being a png */ - FILE *f = fopen(filename.c_str(), "rb"); - _assert(f, "File %s could not be opened for reading\n", filename.c_str()); - _assert(fread(header, 1, 8, f) == 8, "File ended before end of header\n"); - _assert(!png_sig_cmp(header, 0, 8), "File %s is not recognized as a PNG file\n", filename.c_str()); + /* open file and test for it being a png */ + FILE *f = fopen(filename.c_str(), "rb"); + _assert(f, "File %s could not be opened for reading\n", filename.c_str()); + _assert(fread(header, 1, 8, f) == 8, "File ended before end of header\n"); + _assert(!png_sig_cmp(header, 0, 8), "File %s is not recognized as a PNG file\n", filename.c_str()); - /* initialize stuff */ - png_ptr = png_create_read_struct(PNG_LIBPNG_VER_STRING, NULL, NULL, NULL); + /* initialize stuff */ + png_ptr = png_create_read_struct(PNG_LIBPNG_VER_STRING, NULL, NULL, NULL); - _assert(png_ptr, "png_create_read_struct failed\n"); + _assert(png_ptr, "png_create_read_struct failed\n"); - info_ptr = png_create_info_struct(png_ptr); - _assert(info_ptr, "png_create_info_struct failed\n"); + info_ptr = png_create_info_struct(png_ptr); + _assert(info_ptr, "png_create_info_struct failed\n"); - _assert(!setjmp(png_jmpbuf(png_ptr)), "Error during init_io\n"); + _assert(!setjmp(png_jmpbuf(png_ptr)), "Error during init_io\n"); - png_init_io(png_ptr, f); - png_set_sig_bytes(png_ptr, 8); + png_init_io(png_ptr, f); + png_set_sig_bytes(png_ptr, 8); - png_read_info(png_ptr, info_ptr); + png_read_info(png_ptr, info_ptr); - int width = png_get_image_width(png_ptr, info_ptr); - int height = png_get_image_height(png_ptr, info_ptr); - int channels = png_get_channels(png_ptr, info_ptr); - int bit_depth = png_get_bit_depth(png_ptr, info_ptr); + int width = png_get_image_width(png_ptr, info_ptr); + int height = png_get_image_height(png_ptr, info_ptr); + int channels = png_get_channels(png_ptr, info_ptr); + int bit_depth = png_get_bit_depth(png_ptr, info_ptr); - // Expand low-bpp images to have only 1 pixel per byte (As opposed to tight packing) - if (bit_depth < 8) { - png_set_packing(png_ptr); - } + // Expand low-bpp images to have only 1 pixel per byte (As opposed to tight packing) + if (bit_depth < 8) { png_set_packing(png_ptr); } - Image im(1); - if (channels != 1) { - im = Image(width, height, channels); - } else { - im = Image(width, height); - } + Image im(1); + if (channels != 1) { + im = Image(width, height, channels); + } else { + im = Image(width, height); + } - png_set_interlace_handling(png_ptr); - png_read_update_info(png_ptr, info_ptr); + png_set_interlace_handling(png_ptr); + png_read_update_info(png_ptr, info_ptr); - // read the file - _assert(!setjmp(png_jmpbuf(png_ptr)), "Error during read_image\n"); + // read the file + _assert(!setjmp(png_jmpbuf(png_ptr)), "Error during read_image\n"); - row_pointers = new png_bytep[im.height()]; - for (int y = 0; y < im.height(); y++) { - row_pointers[y] = new png_byte[png_get_rowbytes(png_ptr, info_ptr)]; - } + row_pointers = new png_bytep[im.height()]; + for (int y = 0; y < im.height(); y++) { row_pointers[y] = new png_byte[png_get_rowbytes(png_ptr, info_ptr)]; } - png_read_image(png_ptr, row_pointers); + png_read_image(png_ptr, row_pointers); - fclose(f); + fclose(f); - _assert((bit_depth == 8) || (bit_depth == 16), "Can only handle 8-bit or 16-bit pngs\n"); - - // convert the data to T - - int c_stride = (im.channels() == 1) ? 0 : im.stride(2); - T *ptr = (T*)im.data(); - if (bit_depth == 8) { - for (int y = 0; y < im.height(); y++) { - uint8_t *srcPtr = (uint8_t *)(row_pointers[y]); - for (int x = 0; x < im.width(); x++) { - for (int c = 0; c < im.channels(); c++) { - convert(*srcPtr++, ptr[c*c_stride]); - } - ptr++; - } - } - } else if (bit_depth == 16) { - for (int y = 0; y < im.height(); y++) { - uint8_t *srcPtr = (uint8_t *)(row_pointers[y]); - for (int x = 0; x < im.width(); x++) { - for (int c = 0; c < im.channels(); c++) { - uint16_t hi = (*srcPtr++) << 8; - uint16_t lo = hi | (*srcPtr++); - convert(lo, ptr[c*c_stride]); - } - ptr++; - } - } - } + _assert((bit_depth == 8) || (bit_depth == 16), "Can only handle 8-bit or 16-bit pngs\n"); + + // convert the data to T - // clean up + int c_stride = (im.channels() == 1) ? 0 : im.stride(2); + T *ptr = (T *)im.data(); + if (bit_depth == 8) { + for (int y = 0; y < im.height(); y++) { + uint8_t *srcPtr = (uint8_t *)(row_pointers[y]); + for (int x = 0; x < im.width(); x++) { + for (int c = 0; c < im.channels(); c++) { convert(*srcPtr++, ptr[c * c_stride]); } + ptr++; + } + } + } else if (bit_depth == 16) { for (int y = 0; y < im.height(); y++) { - delete[] row_pointers[y]; + uint8_t *srcPtr = (uint8_t *)(row_pointers[y]); + for (int x = 0; x < im.width(); x++) { + for (int c = 0; c < im.channels(); c++) { + uint16_t hi = (*srcPtr++) << 8; + uint16_t lo = hi | (*srcPtr++); + convert(lo, ptr[c * c_stride]); + } + ptr++; + } } - delete[] row_pointers; + } + + // clean up + for (int y = 0; y < im.height(); y++) { delete[] row_pointers[y]; } + delete[] row_pointers; - png_destroy_read_struct(&png_ptr, &info_ptr, NULL); + png_destroy_read_struct(&png_ptr, &info_ptr, NULL); - im.set_host_dirty(); - return im; + im.set_host_dirty(); + return im; } -template -void save_png(Image im, std::string filename) { - png_structp png_ptr; - png_infop info_ptr; - png_bytep *row_pointers; - png_byte color_type; +template void save_png(Image im, std::string filename) +{ + png_structp png_ptr; + png_infop info_ptr; + png_bytep *row_pointers; + png_byte color_type; - im.copy_to_host(); + im.copy_to_host(); - _assert(im.channels() > 0 && im.channels() < 5, - "Can't write PNG files that have other than 1, 2, 3, or 4 channels\n"); + _assert( + im.channels() > 0 && im.channels() < 5, "Can't write PNG files that have other than 1, 2, 3, or 4 channels\n"); - png_byte color_types[4] = {PNG_COLOR_TYPE_GRAY, PNG_COLOR_TYPE_GRAY_ALPHA, - PNG_COLOR_TYPE_RGB, PNG_COLOR_TYPE_RGB_ALPHA - }; - color_type = color_types[im.channels() - 1]; + png_byte color_types[4] = { + PNG_COLOR_TYPE_GRAY, PNG_COLOR_TYPE_GRAY_ALPHA, PNG_COLOR_TYPE_RGB, PNG_COLOR_TYPE_RGB_ALPHA + }; + color_type = color_types[im.channels() - 1]; - // open file - FILE *f = fopen(filename.c_str(), "wb"); - _assert(f, "[write_png_file] File %s could not be opened for writing\n", filename.c_str()); + // open file + FILE *f = fopen(filename.c_str(), "wb"); + _assert(f, "[write_png_file] File %s could not be opened for writing\n", filename.c_str()); - // initialize stuff - png_ptr = png_create_write_struct(PNG_LIBPNG_VER_STRING, NULL, NULL, NULL); - _assert(png_ptr, "[write_png_file] png_create_write_struct failed\n"); + // initialize stuff + png_ptr = png_create_write_struct(PNG_LIBPNG_VER_STRING, NULL, NULL, NULL); + _assert(png_ptr, "[write_png_file] png_create_write_struct failed\n"); - info_ptr = png_create_info_struct(png_ptr); - _assert(info_ptr, "[write_png_file] png_create_info_struct failed\n"); + info_ptr = png_create_info_struct(png_ptr); + _assert(info_ptr, "[write_png_file] png_create_info_struct failed\n"); - _assert(!setjmp(png_jmpbuf(png_ptr)), "[write_png_file] Error during init_io\n"); + _assert(!setjmp(png_jmpbuf(png_ptr)), "[write_png_file] Error during init_io\n"); - png_init_io(png_ptr, f); + png_init_io(png_ptr, f); - unsigned int bit_depth = 16; - if (sizeof(T) == 1) { - bit_depth = 8; - } + unsigned int bit_depth = 16; + if (sizeof(T) == 1) { bit_depth = 8; } - // write header - _assert(!setjmp(png_jmpbuf(png_ptr)), "[write_png_file] Error during writing header\n"); + // write header + _assert(!setjmp(png_jmpbuf(png_ptr)), "[write_png_file] Error during writing header\n"); - png_set_IHDR(png_ptr, info_ptr, im.width(), im.height(), - bit_depth, color_type, PNG_INTERLACE_NONE, - PNG_COMPRESSION_TYPE_BASE, PNG_FILTER_TYPE_BASE); + png_set_IHDR(png_ptr, + info_ptr, + im.width(), + im.height(), + bit_depth, + color_type, + PNG_INTERLACE_NONE, + PNG_COMPRESSION_TYPE_BASE, + PNG_FILTER_TYPE_BASE); - png_write_info(png_ptr, info_ptr); + png_write_info(png_ptr, info_ptr); - row_pointers = new png_bytep[im.height()]; + row_pointers = new png_bytep[im.height()]; - // im.copyToHost(); // in case the image is on the gpu + // im.copyToHost(); // in case the image is on the gpu - int c_stride = (im.channels() == 1) ? 0 : im.stride(2); - T *srcPtr = (T*)im.data(); + int c_stride = (im.channels() == 1) ? 0 : im.stride(2); + T *srcPtr = (T *)im.data(); - for (int y = 0; y < im.height(); y++) { - row_pointers[y] = new png_byte[png_get_rowbytes(png_ptr, info_ptr)]; - uint8_t *dstPtr = (uint8_t *)(row_pointers[y]); - if (bit_depth == 16) { - // convert to uint16_t - for (int x = 0; x < im.width(); x++) { - for (int c = 0; c < im.channels(); c++) { - uint16_t out; - convert(srcPtr[c*c_stride], out); - *dstPtr++ = out >> 8; - *dstPtr++ = out & 0xff; - } - srcPtr++; - } - } else if (bit_depth == 8) { - // convert to uint8_t - for (int x = 0; x < im.width(); x++) { - for (int c = 0; c < im.channels(); c++) { - uint8_t out; - convert(srcPtr[c*c_stride], out); - *dstPtr++ = out; - } - srcPtr++; - } - } else { - _assert(bit_depth == 8 || bit_depth == 16, "We only support saving 8- and 16-bit images."); + for (int y = 0; y < im.height(); y++) { + row_pointers[y] = new png_byte[png_get_rowbytes(png_ptr, info_ptr)]; + uint8_t *dstPtr = (uint8_t *)(row_pointers[y]); + if (bit_depth == 16) { + // convert to uint16_t + for (int x = 0; x < im.width(); x++) { + for (int c = 0; c < im.channels(); c++) { + uint16_t out; + convert(srcPtr[c * c_stride], out); + *dstPtr++ = out >> 8; + *dstPtr++ = out & 0xff; + } + srcPtr++; + } + } else if (bit_depth == 8) { + // convert to uint8_t + for (int x = 0; x < im.width(); x++) { + for (int c = 0; c < im.channels(); c++) { + uint8_t out; + convert(srcPtr[c * c_stride], out); + *dstPtr++ = out; } + srcPtr++; + } + } else { + _assert(bit_depth == 8 || bit_depth == 16, "We only support saving 8- and 16-bit images."); } + } - // write data - _assert(!setjmp(png_jmpbuf(png_ptr)), "[write_png_file] Error during writing bytes"); + // write data + _assert(!setjmp(png_jmpbuf(png_ptr)), "[write_png_file] Error during writing bytes"); - png_write_image(png_ptr, row_pointers); + png_write_image(png_ptr, row_pointers); - // finish write - _assert(!setjmp(png_jmpbuf(png_ptr)), "[write_png_file] Error during end of write"); + // finish write + _assert(!setjmp(png_jmpbuf(png_ptr)), "[write_png_file] Error during end of write"); - png_write_end(png_ptr, NULL); + png_write_end(png_ptr, NULL); - // clean up - for (int y = 0; y < im.height(); y++) { - delete[] row_pointers[y]; - } - delete[] row_pointers; + // clean up + for (int y = 0; y < im.height(); y++) { delete[] row_pointers[y]; } + delete[] row_pointers; - fclose(f); + fclose(f); - png_destroy_write_struct(&png_ptr, &info_ptr); + png_destroy_write_struct(&png_ptr, &info_ptr); } - -inline int is_little_endian() { - int value = 1; - return ((char *) &value)[0] == 1; +inline int is_little_endian() +{ + int value = 1; + return ((char *)&value)[0] == 1; } -#define SWAP_ENDIAN16(little_endian, value) if (little_endian) { (value) = (((value) & 0xff)<<8)|(((value) & 0xff00)>>8); } - -template -Image load_ppm(std::string filename) { - - /* open file and test for it being a ppm */ - FILE *f = fopen(filename.c_str(), "rb"); - _assert(f, "File %s could not be opened for reading\n", filename.c_str()); - - int width, height, maxval; - char header[256]; - _assert(fscanf(f, "%255s", header) == 1, "Could not read PPM header\n"); - _assert(fscanf(f, "%d %d\n", &width, &height) == 2, "Could not read PPM width and height\n"); - _assert(fscanf(f, "%d", &maxval) == 1, "Could not read PPM max value\n"); - _assert(fgetc(f) != EOF, "Could not read char from PPM\n"); - - int bit_depth = 0; - if (maxval == 255) { bit_depth = 8; } - else if (maxval == 65535) { bit_depth = 16; } - else { _assert(false, "Invalid bit depth in PPM\n"); } - - _assert(strcmp(header, "P6") == 0 || strcmp(header, "p6") == 0, "Input is not binary PPM\n"); - - int channels = 3; - Image im(width, height, channels); - - // convert the data to T - if (bit_depth == 8) { - uint8_t *data = new uint8_t[width*height*3]; - _assert(fread((void *) data, - sizeof(uint8_t), width*height*3, f) == (size_t) (width*height*3), - "Could not read PPM 8-bit data\n"); - fclose(f); - - T *im_data = (T*) im.data(); - for (int y = 0; y < im.height(); y++) { - uint8_t *row = (uint8_t *)(&data[(y*width)*3]); - for (int x = 0; x < im.width(); x++) { - convert(*row++, im_data[(0*height+y)*width+x]); - convert(*row++, im_data[(1*height+y)*width+x]); - convert(*row++, im_data[(2*height+y)*width+x]); - } - } - delete[] data; - } else if (bit_depth == 16) { - int little_endian = is_little_endian(); - uint16_t *data = new uint16_t[width*height*3]; - _assert(fread((void *) data, sizeof(uint16_t), width*height*3, f) == (size_t) (width*height*3), "Could not read PPM 16-bit data\n"); - fclose(f); - T *im_data = (T*) im.data(); - for (int y = 0; y < im.height(); y++) { - uint16_t *row = (uint16_t *) (&data[(y*width)*3]); - for (int x = 0; x < im.width(); x++) { - uint16_t value; - value = *row++; SWAP_ENDIAN16(little_endian, value); convert(value, im_data[(0*height+y)*width+x]); - value = *row++; SWAP_ENDIAN16(little_endian, value); convert(value, im_data[(1*height+y)*width+x]); - value = *row++; SWAP_ENDIAN16(little_endian, value); convert(value, im_data[(2*height+y)*width+x]); - } - } - delete[] data; +#define SWAP_ENDIAN16(little_endian, value) \ + if (little_endian) { (value) = (((value) & 0xff) << 8) | (((value) & 0xff00) >> 8); } + +template Image load_ppm(std::string filename) +{ + + /* open file and test for it being a ppm */ + FILE *f = fopen(filename.c_str(), "rb"); + _assert(f, "File %s could not be opened for reading\n", filename.c_str()); + + int width, height, maxval; + char header[256]; + _assert(fscanf(f, "%255s", header) == 1, "Could not read PPM header\n"); + _assert(fscanf(f, "%d %d\n", &width, &height) == 2, "Could not read PPM width and height\n"); + _assert(fscanf(f, "%d", &maxval) == 1, "Could not read PPM max value\n"); + _assert(fgetc(f) != EOF, "Could not read char from PPM\n"); + + int bit_depth = 0; + if (maxval == 255) { + bit_depth = 8; + } else if (maxval == 65535) { + bit_depth = 16; + } else { + _assert(false, "Invalid bit depth in PPM\n"); + } + + _assert(strcmp(header, "P6") == 0 || strcmp(header, "p6") == 0, "Input is not binary PPM\n"); + + int channels = 3; + Image im(width, height, channels); + + // convert the data to T + if (bit_depth == 8) { + uint8_t *data = new uint8_t[width * height * 3]; + _assert(fread((void *)data, sizeof(uint8_t), width * height * 3, f) == (size_t)(width * height * 3), + "Could not read PPM 8-bit data\n"); + fclose(f); + + T *im_data = (T *)im.data(); + for (int y = 0; y < im.height(); y++) { + uint8_t *row = (uint8_t *)(&data[(y * width) * 3]); + for (int x = 0; x < im.width(); x++) { + convert(*row++, im_data[(0 * height + y) * width + x]); + convert(*row++, im_data[(1 * height + y) * width + x]); + convert(*row++, im_data[(2 * height + y) * width + x]); + } + } + delete[] data; + } else if (bit_depth == 16) { + int little_endian = is_little_endian(); + uint16_t *data = new uint16_t[width * height * 3]; + _assert(fread((void *)data, sizeof(uint16_t), width * height * 3, f) == (size_t)(width * height * 3), + "Could not read PPM 16-bit data\n"); + fclose(f); + T *im_data = (T *)im.data(); + for (int y = 0; y < im.height(); y++) { + uint16_t *row = (uint16_t *)(&data[(y * width) * 3]); + for (int x = 0; x < im.width(); x++) { + uint16_t value; + value = *row++; + SWAP_ENDIAN16(little_endian, value); + convert(value, im_data[(0 * height + y) * width + x]); + value = *row++; + SWAP_ENDIAN16(little_endian, value); + convert(value, im_data[(1 * height + y) * width + x]); + value = *row++; + SWAP_ENDIAN16(little_endian, value); + convert(value, im_data[(2 * height + y) * width + x]); + } } - im(0,0,0) = im(0,0,0); /* Mark dirty inside read/write functions. */ + delete[] data; + } + im(0, 0, 0) = im(0, 0, 0); /* Mark dirty inside read/write functions. */ - return im; + return im; } -template -void save_ppm(Image im, std::string filename) { - unsigned int bit_depth = sizeof(T) == 1 ? 8: 16; - - FILE *f = fopen(filename.c_str(), "wb"); - _assert(f, "File %s could not be opened for writing\n", filename.c_str()); - fprintf(f, "P6\n%d %d\n%d\n", im.width(), im.height(), (1< void save_ppm(Image im, std::string filename) +{ + unsigned int bit_depth = sizeof(T) == 1 ? 8 : 16; + + FILE *f = fopen(filename.c_str(), "wb"); + _assert(f, "File %s could not be opened for writing\n", filename.c_str()); + fprintf(f, "P6\n%d %d\n%d\n", im.width(), im.height(), (1 << bit_depth) - 1); + int width = im.width(), height = im.height(); + + if (bit_depth == 8) { + uint8_t *data = new uint8_t[width * height * 3]; + for (int y = 0; y < im.height(); y++) { + for (int x = 0; x < im.width(); x++) { + uint8_t *p = (uint8_t *)(&data[(y * width + x) * 3]); + for (int c = 0; c < im.channels(); c++) { convert(im(x, y, c), p[c]); } + } + } + _assert(fwrite((void *)data, sizeof(uint8_t), width * height * 3, f) == (size_t)(width * height * 3), + "Could not write PPM 8-bit data\n"); + delete[] data; + } else if (bit_depth == 16) { + int little_endian = is_little_endian(); + uint16_t *data = new uint16_t[width * height * 3]; + for (int y = 0; y < im.height(); y++) { + for (int x = 0; x < im.width(); x++) { + uint16_t *p = (uint16_t *)(&data[(y * width + x) * 3]); + for (int c = 0; c < im.channels(); c++) { + uint16_t value; + convert(im(x, y, c), value); + SWAP_ENDIAN16(little_endian, value); + p[c] = value; } - _assert(fwrite((void *) data, sizeof(uint16_t), width*height*3, f) == (size_t) (width*height*3), "Could not write PPM 16-bit data\n"); - delete[] data; + } } - fclose(f); + _assert(fwrite((void *)data, sizeof(uint16_t), width * height * 3, f) == (size_t)(width * height * 3), + "Could not write PPM 16-bit data\n"); + delete[] data; + } + fclose(f); } -template -Image load(std::string filename) { - if (ends_with_ignore_case(filename, ".png")) { - return load_png(filename); - } else if (ends_with_ignore_case(filename, ".ppm")) { - return load_ppm(filename); - } else { - _assert(false, "[load] unsupported file extension (png|ppm supported)"); - } +template Image load(std::string filename) +{ + if (ends_with_ignore_case(filename, ".png")) { + return load_png(filename); + } else if (ends_with_ignore_case(filename, ".ppm")) { + return load_ppm(filename); + } else { + _assert(false, "[load] unsupported file extension (png|ppm supported)"); + } } -template -void save(Image im, std::string filename) { - if (ends_with_ignore_case(filename, ".png")) { - save_png(im, filename); - } else if (ends_with_ignore_case(filename, ".ppm")) { - save_ppm(im, filename); - } else { - _assert(false, "[save] unsupported file extension (png|ppm supported)"); - } +template void save(Image im, std::string filename) +{ + if (ends_with_ignore_case(filename, ".png")) { + save_png(im, filename); + } else if (ends_with_ignore_case(filename, ".ppm")) { + save_ppm(im, filename); + } else { + _assert(false, "[save] unsupported file extension (png|ppm supported)"); + } } - #endif diff --git a/filmulator-gui/Halide/performLayerMix.cpp b/filmulator-gui/Halide/performLayerMix.cpp index 41302f83..c527f618 100644 --- a/filmulator-gui/Halide/performLayerMix.cpp +++ b/filmulator-gui/Halide/performLayerMix.cpp @@ -1,33 +1,34 @@ -#include #include +#include #include using namespace Halide; -//using namespace Halide::BoundaryConditions; +// using namespace Halide::BoundaryConditions; using namespace std; using Halide::Image; -#include "image_io.h" #include "halideFilmulate.h" +#include "image_io.h" Var x, y, c; -int main(int argc, char **argv) { +int main(int argc, char **argv) +{ - ImageParam devConc(type_of(),2); - Func developerConcentration = lambda(x,y,devConc(x,y)); + ImageParam devConc(type_of(), 2); + Func developerConcentration = lambda(x, y, devConc(x, y)); - ImageParam devMoved(type_of(),2); - Func developerMoved = lambda(x,y,devMoved(x,y)); + ImageParam devMoved(type_of(), 2); + Func developerMoved = lambda(x, y, devMoved(x, y)); - ImageParam filmData(type_of(), 3); - Func filmulationData = lambda(x,y,c,filmData(x,y,c)); + ImageParam filmData(type_of(), 3); + Func filmulationData = lambda(x, y, c, filmData(x, y, c)); - Func filmulationDataOut; - filmulationDataOut(x,y,c) = undef(); //filmulationData(x,y,c); - filmulationDataOut(x,y,DEVEL_CONC) = developerMoved(x,y) + developerConcentration(x,y); - std::vector combArgs = filmulationDataOut.infer_arguments(); - filmulationDataOut.compile_to_file("performLayerMix",combArgs); + Func filmulationDataOut; + filmulationDataOut(x, y, c) = undef();// filmulationData(x,y,c); + filmulationDataOut(x, y, DEVEL_CONC) = developerMoved(x, y) + developerConcentration(x, y); + std::vector combArgs = filmulationDataOut.infer_arguments(); + filmulationDataOut.compile_to_file("performLayerMix", combArgs); - return 0; + return 0; } diff --git a/filmulator-gui/Halide/postFilmulatorPipeline.cpp b/filmulator-gui/Halide/postFilmulatorPipeline.cpp index ffb90680..6d3cbbda 100644 --- a/filmulator-gui/Halide/postFilmulatorPipeline.cpp +++ b/filmulator-gui/Halide/postFilmulatorPipeline.cpp @@ -2,100 +2,101 @@ using namespace Halide; -#include "RGBtoHSV.cpp" #include "HSVtoRGB.cpp" +#include "RGBtoHSV.cpp" #include "applyfilmlikecurve.cpp" -struct BlackWhiteParams { - float blackpoint; - float whitepoint; +struct BlackWhiteParams +{ + float blackpoint; + float whitepoint; }; -struct FilmlikeCurvesParams { - float shadowsX; - float shadowsY; - float highlightsX; - float highlightsY; - float vibrance; - float saturation; +struct FilmlikeCurvesParams +{ + float shadowsX; + float shadowsY; + float highlightsX; + float highlightsY; + float vibrance; + float saturation; }; -struct OrientationParams { - int rotation; +struct OrientationParams +{ + int rotation; }; -extern "C" float shadows_highlights(float x,float shadowsX,float shadowsY, - float highlightsX,float highlightsY){ +extern "C" float shadows_highlights(float x, float shadowsX, float shadowsY, float highlightsX, float highlightsY) +{ return x; } -HalideExtern_5(float,shadows_highlights,float,float,float,float,float); - -class postFilmulationGenerator : public Halide::Generator { - public: - Param blackPoint{"blackPoint"}; - Param whitePoint{"whitePoint"}; - - Param shadowsX{"shadowsX"}; - Param shadowsY{"shadowsY"}; - Param highlightsX{"highlightsX"}; - Param highlightsY{"highlightsY"}; - - Param saturation{"saturation"}; - Param vibrance{"vibrance"}; - - Param rotation{"rotation"}; - - ImageParam postFilmulationImage{Float(32), 3, "postFilmulationImage"}; - - Var x, y, c; - - Func build() { - - Func blackWhited; - blackWhited(x,y,c) = clamp(whitePoint*(postFilmulationImage(x,y,c)-blackPoint),0,1); - - Func LUT; - Expr maxval = 2^16 - 1; - LUT(x) = shadows_highlights(cast(x)/(maxval), - shadowsX, - shadowsY, - highlightsX, - highlightsY); - LUT.bound(x,0,maxval+1); - - Func curved; - curved = applyFilmlikeCurve(blackWhited,LUT); - - Func HSVed; - HSVed = RGBtoHSV(curved); - - Func vibranceSaturated; - Expr sat = pow(2, saturation); - Expr gamma = pow(2,-vibrance); - vibranceSaturated(x,y,c) = select(c == 1, // Saturation - clamp(sat*fast_pow(HSVed(x,y,c),gamma),0,1), - HSVed(x,y,c)); - - Func RGBed; - RGBed = HSVtoRGB(vibranceSaturated); - - Func intcast; - intcast(x,y,c) = cast(UInt(16),RGBed(x,y,c)); - - Func rotated; - int rotation = rotation; - Expr width = postFilmulationImage.width(); - Expr height = postFilmulationImage.height(); - rotated(x,y,c) = select(rotation == 2, // Upside down - intcast(width-x,height-y,c), - select(rotation == 3, // Right side down - intcast(y,width-x,c), - select(rotation == 1, // Left side down - intcast(height-y,x,c), - // Default: no rotation - intcast(x,y,c)))); - return rotated; - } +HalideExtern_5(float, shadows_highlights, float, float, float, float, float); + +class postFilmulationGenerator : public Halide::Generator +{ +public: + Param blackPoint{ "blackPoint" }; + Param whitePoint{ "whitePoint" }; + + Param shadowsX{ "shadowsX" }; + Param shadowsY{ "shadowsY" }; + Param highlightsX{ "highlightsX" }; + Param highlightsY{ "highlightsY" }; + + Param saturation{ "saturation" }; + Param vibrance{ "vibrance" }; + + Param rotation{ "rotation" }; + + ImageParam postFilmulationImage{ Float(32), 3, "postFilmulationImage" }; + + Var x, y, c; + + Func build() + { + + Func blackWhited; + blackWhited(x, y, c) = clamp(whitePoint * (postFilmulationImage(x, y, c) - blackPoint), 0, 1); + + Func LUT; + Expr maxval = 2 ^ 16 - 1; + LUT(x) = shadows_highlights(cast(x) / (maxval), shadowsX, shadowsY, highlightsX, highlightsY); + LUT.bound(x, 0, maxval + 1); + + Func curved; + curved = applyFilmlikeCurve(blackWhited, LUT); + + Func HSVed; + HSVed = RGBtoHSV(curved); + + Func vibranceSaturated; + Expr sat = pow(2, saturation); + Expr gamma = pow(2, -vibrance); + vibranceSaturated(x, y, c) = select(c == 1,// Saturation + clamp(sat * fast_pow(HSVed(x, y, c), gamma), 0, 1), + HSVed(x, y, c)); + + Func RGBed; + RGBed = HSVtoRGB(vibranceSaturated); + + Func intcast; + intcast(x, y, c) = cast(UInt(16), RGBed(x, y, c)); + + Func rotated; + int rotation = rotation; + Expr width = postFilmulationImage.width(); + Expr height = postFilmulationImage.height(); + rotated(x, y, c) = select(rotation == 2,// Upside down + intcast(width - x, height - y, c), + select(rotation == 3,// Right side down + intcast(y, width - x, c), + select(rotation == 1,// Left side down + intcast(height - y, x, c), + // Default: no rotation + intcast(x, y, c)))); + return rotated; + } }; -RegisterGenerator postFilmulationGenerator{"postFilmulationGenerator"}; +RegisterGenerator postFilmulationGenerator{ "postFilmulationGenerator" }; diff --git a/filmulator-gui/Halide/runFilmulator.cpp b/filmulator-gui/Halide/runFilmulator.cpp index 6fbdcab5..f874f7de 100644 --- a/filmulator-gui/Halide/runFilmulator.cpp +++ b/filmulator-gui/Halide/runFilmulator.cpp @@ -1,175 +1,174 @@ -#include #include -//#include "readPNG.h" -//#include "savePNG.h" -#include "halideFilmulate.h" -#include "develop.h" -#include "diffuse.h" +#include +// #include "readPNG.h" +// #include "savePNG.h" #include "calcLayerMix.h" #include "calcReservoirConcentration.h" +#include "clock.h" +#include "develop.h" +#include "diffuse.h" +#include "generateFilmulatedImage.h" +#include "halideFilmulate.h" #include "performLayerMix.h" #include "setupFilmulator.h" -#include "generateFilmulatedImage.h" -#include "clock.h" using namespace std; using Halide::Image; #include "image_io.h" -int main(){ - - int error; - /*buffer_t inputImage = {0}; - error = readPNG("imageName.png",&inputImage); - if (error) - cout << "png error" << endl; - */ - Halide::Image input = load("P1040567.png"); - Halide::Buffer inputBufferClass = (Halide::Buffer) input; - buffer_t* inputImage = inputBufferClass.raw_buffer(); - int width = input.width(); - int height = input.height(); - - float reservoirConcentration = 1; - float reservoirThickness = 1000; - float activeLayerThickness = 0.1; - float crystalsPerPixel = 500; - float initialCrystalRadius = 0.00001; - float initialSilverSaltDensity = 1; - float developerConsumptionConst = 2000000.0; - float crystalGrowthConst = 0.00001; - float silverSaltConsumptionConst = 2000000.0; - float totalDevelTime = 100; - int agitateCount = 0; - int developmentSteps = 12; - float filmArea = 864; - float sigmaConst = 0.2; - float layerMixConst = 0.2; - float layerTimeDivisor = 20; - float rolloffBoundary = 51275; - - buffer_t filmulationData = {0}; - float* filmulationDataMemory = new float[10*width*height]; - filmulationData.host = (uint8_t*)filmulationDataMemory; - filmulationData.stride[0] = 1; - filmulationData.stride[1] = width; - filmulationData.stride[2] = width*height; - filmulationData.extent[0] = width; - filmulationData.extent[1] = height; - filmulationData.extent[2] = 10; - filmulationData.elem_size = 4; - error = setupFilmulator(inputImage,crystalsPerPixel,initialCrystalRadius, - initialSilverSaltDensity, reservoirConcentration, - rolloffBoundary,&filmulationData); - if (error) - cout << "setup error" << endl; - - buffer_t devConc = {0}; - float* devConcMemory = new float[width*height]; - devConc.host = (uint8_t*)devConcMemory; - devConc.stride[0] = 1; - devConc.stride[1] = width; - devConc.extent[0] = width; - devConc.extent[1] = height; - devConc.elem_size = 4; - - buffer_t devMoved = {0}; - float* devMovedMemory = new float[width*height]; - devMoved.host = (uint8_t*)devMovedMemory; - devMoved.stride[0] = 1; - devMoved.stride[1] = width; - devMoved.extent[0] = width; - devMoved.extent[1] = height; - devMoved.elem_size = 4; - - buffer_t resBuffer = {0}; - float resMemory; - resBuffer.host = (uint8_t*)&resMemory; - resBuffer.stride[0] = 1; - resBuffer.stride[1] = 1; - resBuffer.extent[0] = 1; - resBuffer.extent[1] = 1; - resBuffer.elem_size = 4; - - double totalDevelopTime = 0; - double totalDiffuseTime = 0; - double calcLayer = 0; - double calcRes = 0; - double perfLayer = 0; - current_time(); - for (int i = 0; i < 1*(developmentSteps-0); i++) - { - float timeStep = totalDevelTime/float(developmentSteps); - - //* - double beforeDevelopTime = current_time(); - error = develop(&filmulationData,activeLayerThickness,crystalGrowthConst, - developerConsumptionConst,silverSaltConsumptionConst, - timeStep,&filmulationData); - if (error) - cout << "development error on iteration " << i << endl; - totalDevelopTime += current_time() - beforeDevelopTime; - - //*/ - //* - double beforeDiffuseTime = current_time(); - error = diffuse(&filmulationData,filmArea,sigmaConst,timeStep,&devConc); - if (error) - cout << "diffuse error on iteration " << i << endl; - totalDiffuseTime += current_time() - beforeDiffuseTime; - - //* - double beforeCalcLayer = current_time(); - error = calcLayerMix(&devConc,layerMixConst,layerTimeDivisor, - reservoirConcentration,timeStep,&devMoved); - if (error) - cout << "calcLayer error on iteration " << i << endl; - calcLayer += current_time() - beforeCalcLayer; - - double beforeCalcRes = current_time(); - error = calcReservoirConcentration(&devMoved,activeLayerThickness, - filmArea,reservoirConcentration, - reservoirThickness,&resBuffer); - if (error) - cout << "calcRes error on iteration " << i << endl; - calcRes += current_time() - beforeCalcRes; - - reservoirConcentration = resMemory; - - double beforePerfLayer = current_time(); - error = performLayerMix(&devMoved,&devConc,&filmulationData); - if (error) - cout << "perfLayer error on iteration " << i << endl; - perfLayer += current_time() - beforePerfLayer; - //*/ - } - cout << "filmulation time: " << current_time() << "ms" << endl; - cout << "Develop time : " << totalDevelopTime << "ms" << endl; - cout << "Diffuse time : " << totalDiffuseTime << "ms" << endl; - cout << "Calc Layer time : " << calcLayer << "ms" << endl; - cout << "Calc Res time : " << calcRes << "ms" << endl; - cout << "Perf Layer time : " << perfLayer << "ms" << endl; - buffer_t outputImage = {0}; - uint8_t* outputImageMemory= new uint8_t[3*width*height]; - outputImage.host = outputImageMemory; - outputImage.stride[0] = 1; - outputImage.stride[1] = width; - outputImage.stride[2] = width*height; - outputImage.extent[0] = width; - outputImage.extent[1] = height; - outputImage.extent[2] = 3; - outputImage.elem_size = 1; - error = generateFilmulatedImage(&filmulationData,&outputImage); - if (error) - cout << "output error" << endl; - - /*error = savePNG(&outputImage); - if (error) - cout << "save error" << endl; - */ - Image output(&outputImage); - save(output,"filmulationOutput.png"); - - return 0; +int main() +{ + + int error; + /*buffer_t inputImage = {0}; + error = readPNG("imageName.png",&inputImage); + if (error) + cout << "png error" << endl; + */ + Halide::Image input = load("P1040567.png"); + Halide::Buffer inputBufferClass = (Halide::Buffer)input; + buffer_t *inputImage = inputBufferClass.raw_buffer(); + int width = input.width(); + int height = input.height(); + + float reservoirConcentration = 1; + float reservoirThickness = 1000; + float activeLayerThickness = 0.1; + float crystalsPerPixel = 500; + float initialCrystalRadius = 0.00001; + float initialSilverSaltDensity = 1; + float developerConsumptionConst = 2000000.0; + float crystalGrowthConst = 0.00001; + float silverSaltConsumptionConst = 2000000.0; + float totalDevelTime = 100; + int agitateCount = 0; + int developmentSteps = 12; + float filmArea = 864; + float sigmaConst = 0.2; + float layerMixConst = 0.2; + float layerTimeDivisor = 20; + float rolloffBoundary = 51275; + + buffer_t filmulationData = { 0 }; + float *filmulationDataMemory = new float[10 * width * height]; + filmulationData.host = (uint8_t *)filmulationDataMemory; + filmulationData.stride[0] = 1; + filmulationData.stride[1] = width; + filmulationData.stride[2] = width * height; + filmulationData.extent[0] = width; + filmulationData.extent[1] = height; + filmulationData.extent[2] = 10; + filmulationData.elem_size = 4; + error = setupFilmulator(inputImage, + crystalsPerPixel, + initialCrystalRadius, + initialSilverSaltDensity, + reservoirConcentration, + rolloffBoundary, + &filmulationData); + if (error) cout << "setup error" << endl; + + buffer_t devConc = { 0 }; + float *devConcMemory = new float[width * height]; + devConc.host = (uint8_t *)devConcMemory; + devConc.stride[0] = 1; + devConc.stride[1] = width; + devConc.extent[0] = width; + devConc.extent[1] = height; + devConc.elem_size = 4; + + buffer_t devMoved = { 0 }; + float *devMovedMemory = new float[width * height]; + devMoved.host = (uint8_t *)devMovedMemory; + devMoved.stride[0] = 1; + devMoved.stride[1] = width; + devMoved.extent[0] = width; + devMoved.extent[1] = height; + devMoved.elem_size = 4; + + buffer_t resBuffer = { 0 }; + float resMemory; + resBuffer.host = (uint8_t *)&resMemory; + resBuffer.stride[0] = 1; + resBuffer.stride[1] = 1; + resBuffer.extent[0] = 1; + resBuffer.extent[1] = 1; + resBuffer.elem_size = 4; + + double totalDevelopTime = 0; + double totalDiffuseTime = 0; + double calcLayer = 0; + double calcRes = 0; + double perfLayer = 0; + current_time(); + for (int i = 0; i < 1 * (developmentSteps - 0); i++) { + float timeStep = totalDevelTime / float(developmentSteps); + + //* + double beforeDevelopTime = current_time(); + error = develop(&filmulationData, + activeLayerThickness, + crystalGrowthConst, + developerConsumptionConst, + silverSaltConsumptionConst, + timeStep, + &filmulationData); + if (error) cout << "development error on iteration " << i << endl; + totalDevelopTime += current_time() - beforeDevelopTime; + + //*/ + //* + double beforeDiffuseTime = current_time(); + error = diffuse(&filmulationData, filmArea, sigmaConst, timeStep, &devConc); + if (error) cout << "diffuse error on iteration " << i << endl; + totalDiffuseTime += current_time() - beforeDiffuseTime; + + //* + double beforeCalcLayer = current_time(); + error = calcLayerMix(&devConc, layerMixConst, layerTimeDivisor, reservoirConcentration, timeStep, &devMoved); + if (error) cout << "calcLayer error on iteration " << i << endl; + calcLayer += current_time() - beforeCalcLayer; + + double beforeCalcRes = current_time(); + error = calcReservoirConcentration( + &devMoved, activeLayerThickness, filmArea, reservoirConcentration, reservoirThickness, &resBuffer); + if (error) cout << "calcRes error on iteration " << i << endl; + calcRes += current_time() - beforeCalcRes; + + reservoirConcentration = resMemory; + + double beforePerfLayer = current_time(); + error = performLayerMix(&devMoved, &devConc, &filmulationData); + if (error) cout << "perfLayer error on iteration " << i << endl; + perfLayer += current_time() - beforePerfLayer; + //*/ + } + cout << "filmulation time: " << current_time() << "ms" << endl; + cout << "Develop time : " << totalDevelopTime << "ms" << endl; + cout << "Diffuse time : " << totalDiffuseTime << "ms" << endl; + cout << "Calc Layer time : " << calcLayer << "ms" << endl; + cout << "Calc Res time : " << calcRes << "ms" << endl; + cout << "Perf Layer time : " << perfLayer << "ms" << endl; + buffer_t outputImage = { 0 }; + uint8_t *outputImageMemory = new uint8_t[3 * width * height]; + outputImage.host = outputImageMemory; + outputImage.stride[0] = 1; + outputImage.stride[1] = width; + outputImage.stride[2] = width * height; + outputImage.extent[0] = width; + outputImage.extent[1] = height; + outputImage.extent[2] = 3; + outputImage.elem_size = 1; + error = generateFilmulatedImage(&filmulationData, &outputImage); + if (error) cout << "output error" << endl; + + /*error = savePNG(&outputImage); + if (error) + cout << "save error" << endl; + */ + Image output(&outputImage); + save(output, "filmulationOutput.png"); + + return 0; } diff --git a/filmulator-gui/Halide/setupFilmulator.cpp b/filmulator-gui/Halide/setupFilmulator.cpp index a0a0d91f..659ea074 100644 --- a/filmulator-gui/Halide/setupFilmulator.cpp +++ b/filmulator-gui/Halide/setupFilmulator.cpp @@ -1,51 +1,53 @@ -#include #include "halideFilmulate.h" +#include using namespace Halide; -Var x,y,c; +Var x, y, c; -Func exposure(Func input, Expr crystals_per_pixel, Expr rolloff_boundary){ +Func exposure(Func input, Expr crystals_per_pixel, Expr rolloff_boundary) +{ Expr crystal_headroom = 65535.0f - rolloff_boundary; Func rolloff; - rolloff(x,y,c) = select(input(x,y,c) > rolloff_boundary, - 65535.0f - crystal_headroom*crystal_headroom/ - (input(x,y,c) + crystal_headroom-rolloff_boundary), - input(x,y,c)); + rolloff(x, y, c) = select(input(x, y, c) > rolloff_boundary, + 65535.0f - crystal_headroom * crystal_headroom / (input(x, y, c) + crystal_headroom - rolloff_boundary), + input(x, y, c)); Func output; - output(x,y,c) = rolloff(x,y,c)*crystals_per_pixel*0.00015387105f; + output(x, y, c) = rolloff(x, y, c) * crystals_per_pixel * 0.00015387105f; return output; } -int main(int argc, char **argv){ - - Param crystalsPerPixel; - Param rolloffBoundary; - Param initialCrystalRadius; - Param initialSilverSaltDensity; - Param reservoirConcentration; - - ImageParam input(type_of(), 3); - Func in = lambda(x, y, c, 65535.0f*input(x, y, c));; - - Func activeCrystalsPerPixel; - activeCrystalsPerPixel = exposure(in, crystalsPerPixel, rolloffBoundary); - - Func filmulationData; - filmulationData(x,y,c) = cast(0); - filmulationData(x,y,CRYSTAL_RAD_R) = initialCrystalRadius; - filmulationData(x,y,CRYSTAL_RAD_G) = initialCrystalRadius; - filmulationData(x,y,CRYSTAL_RAD_B) = initialCrystalRadius; - filmulationData(x,y,ACTIVE_CRYSTALS_R) = activeCrystalsPerPixel(x,y,0); - filmulationData(x,y,ACTIVE_CRYSTALS_G) = activeCrystalsPerPixel(x,y,1); - filmulationData(x,y,ACTIVE_CRYSTALS_B) = activeCrystalsPerPixel(x,y,2); - filmulationData(x,y,SILVER_SALT_DEN_R) = initialSilverSaltDensity; - filmulationData(x,y,SILVER_SALT_DEN_G) = initialSilverSaltDensity; - filmulationData(x,y,SILVER_SALT_DEN_B) = initialSilverSaltDensity; - filmulationData(x,y,DEVEL_CONC) = reservoirConcentration; - - std::vector args = filmulationData.infer_arguments(); - filmulationData.compile_to_file("setupFilmulator",args); - return 0; +int main(int argc, char **argv) +{ + + Param crystalsPerPixel; + Param rolloffBoundary; + Param initialCrystalRadius; + Param initialSilverSaltDensity; + Param reservoirConcentration; + + ImageParam input(type_of(), 3); + Func in = lambda(x, y, c, 65535.0f * input(x, y, c)); + ; + + Func activeCrystalsPerPixel; + activeCrystalsPerPixel = exposure(in, crystalsPerPixel, rolloffBoundary); + + Func filmulationData; + filmulationData(x, y, c) = cast(0); + filmulationData(x, y, CRYSTAL_RAD_R) = initialCrystalRadius; + filmulationData(x, y, CRYSTAL_RAD_G) = initialCrystalRadius; + filmulationData(x, y, CRYSTAL_RAD_B) = initialCrystalRadius; + filmulationData(x, y, ACTIVE_CRYSTALS_R) = activeCrystalsPerPixel(x, y, 0); + filmulationData(x, y, ACTIVE_CRYSTALS_G) = activeCrystalsPerPixel(x, y, 1); + filmulationData(x, y, ACTIVE_CRYSTALS_B) = activeCrystalsPerPixel(x, y, 2); + filmulationData(x, y, SILVER_SALT_DEN_R) = initialSilverSaltDensity; + filmulationData(x, y, SILVER_SALT_DEN_G) = initialSilverSaltDensity; + filmulationData(x, y, SILVER_SALT_DEN_B) = initialSilverSaltDensity; + filmulationData(x, y, DEVEL_CONC) = reservoirConcentration; + + std::vector args = filmulationData.infer_arguments(); + filmulationData.compile_to_file("setupFilmulator", args); + return 0; } diff --git a/filmulator-gui/Halide/vibranceSaturation.cpp b/filmulator-gui/Halide/vibranceSaturation.cpp index 880446cd..f9085bcd 100644 --- a/filmulator-gui/Halide/vibranceSaturation.cpp +++ b/filmulator-gui/Halide/vibranceSaturation.cpp @@ -1,11 +1,10 @@ -//Compile with: -//g++ vibranceSaturation.cpp -g -I include/ -L bin/ -lHalide `libpng-config --cflags --ldflags` -lpthread -ldl -o vibranceSaturation -//Run With: -//LD_LIBRARY_PATH=bin ./vibranceSaturation -#include -#include -#include -#include +// Compile with: +// g++ vibranceSaturation.cpp -g -I include/ -L bin/ -lHalide `libpng-config --cflags --ldflags` -lpthread -ldl -o +// vibranceSaturation Run With: LD_LIBRARY_PATH=bin ./vibranceSaturation +#include +#include +#include +#include using Halide::Image; #include using namespace Halide; @@ -13,68 +12,54 @@ using namespace Halide; Halide::Func hsv(Func in) { Func out; - Var x,y,c; - Func minC,maxC,delta; - minC(x,y) = min(in(0,x,y),min(in(1,x,y),in(2,x,y))); - maxC(x,y) = max(in(0,x,y),max(in(1,x,y),in(2,x,y))); - out(c,x,y) = 0.0f; - out(2,x,y) = maxC(x,y);//V - delta(x,y) = maxC(x,y) - minC(x,y); - out(1,x,y) = select(maxC(x,y) != 0, delta(x,y)/maxC(x,y), 0);//S + Var x, y, c; + Func minC, maxC, delta; + minC(x, y) = min(in(0, x, y), min(in(1, x, y), in(2, x, y))); + maxC(x, y) = max(in(0, x, y), max(in(1, x, y), in(2, x, y))); + out(c, x, y) = 0.0f; + out(2, x, y) = maxC(x, y);// V + delta(x, y) = maxC(x, y) - minC(x, y); + out(1, x, y) = select(maxC(x, y) != 0, delta(x, y) / maxC(x, y), 0);// S Func h; - h(x,y) = select(maxC(x,y) == in(0,x,y), //R is highest - (in(1,x,y) - in(2,x,y))/delta(x,y), - select(maxC(x,y) == in(1,x,y), //G is highest - 2 + (in(2,x,y) - in(0,x,y))/delta(x,y), - 4 + (in(0,x,y) - in(1,x,y))/delta(x,y)));//B is highest - out(0,x,y) = select(h(x,y) < 0, 60.0f*h(x,y) + 360.0f, 60.0f*h(x,y)); + h(x, y) = select(maxC(x, y) == in(0, x, y),// R is highest + (in(1, x, y) - in(2, x, y)) / delta(x, y), + select(maxC(x, y) == in(1, x, y),// G is highest + 2 + (in(2, x, y) - in(0, x, y)) / delta(x, y), + 4 + (in(0, x, y) - in(1, x, y)) / delta(x, y)));// B is highest + out(0, x, y) = select(h(x, y) < 0, 60.0f * h(x, y) + 360.0f, 60.0f * h(x, y)); return out; } Halide::Func rgb(Func in) { Func out; - Var x,y,c; - Func h,s,v; - h(x,y) = in(0,x,y); - s(x,y) = in(1,x,y); - v(x,y) = in(2,x,y); - Func r,g,b; - Func hd,i,f,p,q,t; - hd(x,y) = h(x,y)/60; - i(x,y) = Halide::floor(hd(x,y)); - f(x,y) = hd(x,y) - i(x,y); - p(x,y) = v(x,y)*(1-s(x,y)); - q(x,y) = v(x,y)*(1-(s(x,y)*f(x,y))); - t(x,y) = v(x,y)*(1-(s(x,y)*(1-f(x,y)))); + Var x, y, c; + Func h, s, v; + h(x, y) = in(0, x, y); + s(x, y) = in(1, x, y); + v(x, y) = in(2, x, y); + Func r, g, b; + Func hd, i, f, p, q, t; + hd(x, y) = h(x, y) / 60; + i(x, y) = Halide::floor(hd(x, y)); + f(x, y) = hd(x, y) - i(x, y); + p(x, y) = v(x, y) * (1 - s(x, y)); + q(x, y) = v(x, y) * (1 - (s(x, y) * f(x, y))); + t(x, y) = v(x, y) * (1 - (s(x, y) * (1 - f(x, y)))); - r(x,y) = select(i(x,y) == 0 || i(x,y) == 5, - v(x,y), - select(i(x,y) == 1, - q(x,y), - select(i(x,y) == 2 || i(x,y) == 3, - p(x,y), - select(i(x,y) == 4, - t(x,y), - v(x,y)))));//default - g(x,y) = select(i(x,y) == 0, - t(x,y), - select(i(x,y) == 1 || i(x,y) == 2, - v(x,y), - select(i(x,y) == 3, - q(x,y), - p(x,y))));//4,5,default - b(x,y) = select(i(x,y) == 0 || i(x,y) == 1, - p(x,y), - select(i(x,y) == 2, - t(x,y), - select(i(x,y) == 3 || i(x,y) == 4, - v(x,y), - q(x,y))));//5,default - out(c,x,y) = select(s(x,y) != 0, - select(c == 0, r(x,y), select(c == 1, g(x,y), b(x,y))), - v(x,y)); - //out(x,y,c) = h(x,y)/360; + r(x, y) = select(i(x, y) == 0 || i(x, y) == 5, + v(x, y), + select(i(x, y) == 1, + q(x, y), + select(i(x, y) == 2 || i(x, y) == 3, p(x, y), select(i(x, y) == 4, t(x, y), v(x, y)))));// default + g(x, y) = select(i(x, y) == 0, + t(x, y), + select(i(x, y) == 1 || i(x, y) == 2, v(x, y), select(i(x, y) == 3, q(x, y), p(x, y))));// 4,5,default + b(x, y) = select(i(x, y) == 0 || i(x, y) == 1, + p(x, y), + select(i(x, y) == 2, t(x, y), select(i(x, y) == 3 || i(x, y) == 4, v(x, y), q(x, y))));// 5,default + out(c, x, y) = select(s(x, y) != 0, select(c == 0, r(x, y), select(c == 1, g(x, y), b(x, y))), v(x, y)); + // out(x,y,c) = h(x,y)/360; return out; } @@ -84,28 +69,24 @@ int main(int argc, char **argv) timeval t1, t2; gettimeofday(&t1, NULL); - Var x,y,c; + Var x, y, c; Func toFloat; - toFloat(c,x,y) = cast(input(x,y,c))/255.0; + toFloat(c, x, y) = cast(input(x, y, c)) / 255.0; Func toHSV; toHSV = hsv(toFloat); Func saturated; - saturated(c,x,y) = select(c != 1, - toHSV(c,x,y), - clamp(1*fast_pow(toHSV(c,x,y),0.5), - 0,1)); - Func toRGB,toInt; + saturated(c, x, y) = select(c != 1, toHSV(c, x, y), clamp(1 * fast_pow(toHSV(c, x, y), 0.5), 0, 1)); + Func toRGB, toInt; toRGB = rgb(saturated); - toInt(x,y,c) = cast(toRGB(c,x,y)*255.0); - Var y_outer,y_inner; - toInt.reorder(c,x,y); - toInt.split(y,y_outer, y_inner, 256); + toInt(x, y, c) = cast(toRGB(c, x, y) * 255.0); + Var y_outer, y_inner; + toInt.reorder(c, x, y); + toInt.split(y, y_outer, y_inner, 256); toInt.parallel(y_outer); - toHSV.compute_at(toInt,x); - Halide::Image output = toInt.realize(input.width(),input.height(),input.channels()); + toHSV.compute_at(toInt, x); + Halide::Image output = toInt.realize(input.width(), input.height(), input.channels()); gettimeofday(&t2, NULL); - save(output,"vibSat.png"); - std::cout< &developerConcentration, - float activeLayerThickness, - float &reservoirDeveloperConcentration, - float reservoirThickness, - float pixelsPerMillimeter ) +void agitate(matrix &developerConcentration, + float activeLayerThickness, + float &reservoirDeveloperConcentration, + float reservoirThickness, + float pixelsPerMillimeter) { - int npixels = developerConcentration.nc()* - developerConcentration.nr(); - float totalDeveloper = sum( developerConcentration ) * - activeLayerThickness / pow( pixelsPerMillimeter, 2 ) + - reservoirDeveloperConcentration * reservoirThickness; - float contactLayerSize = npixels * activeLayerThickness / - pow( pixelsPerMillimeter, 2 ); - reservoirDeveloperConcentration = totalDeveloper / ( reservoirThickness + - contactLayerSize ); - developerConcentration = reservoirDeveloperConcentration; - return; + int npixels = developerConcentration.nc() * developerConcentration.nr(); + float totalDeveloper = sum(developerConcentration) * activeLayerThickness / pow(pixelsPerMillimeter, 2) + + reservoirDeveloperConcentration * reservoirThickness; + float contactLayerSize = npixels * activeLayerThickness / pow(pixelsPerMillimeter, 2); + reservoirDeveloperConcentration = totalDeveloper / (reservoirThickness + contactLayerSize); + developerConcentration = reservoirDeveloperConcentration; + return; } diff --git a/filmulator-gui/core/colorCurves.cpp b/filmulator-gui/core/colorCurves.cpp index 571cfbb5..745e90a6 100644 --- a/filmulator-gui/core/colorCurves.cpp +++ b/filmulator-gui/core/colorCurves.cpp @@ -1,30 +1,29 @@ #include "filmSim.hpp" -void colorCurves(matrix &input, matrix &output, - LUT &lutR, LUT &lutG, LUT &lutB) +void colorCurves(matrix &input, + matrix &output, + LUT &lutR, + LUT &lutG, + LUT &lutB) { - //Check for null inputs - if(lutR.isUnity() && lutG.isUnity() && lutB.isUnity()) - { - output = input; - } - else - { - int nrows = input.nr(); - int ncols = input.nc(); - output.set_size(nrows, ncols); + // Check for null inputs + if (lutR.isUnity() && lutG.isUnity() && lutB.isUnity()) { + output = input; + } else { + int nrows = input.nr(); + int ncols = input.nc(); + output.set_size(nrows, ncols); #pragma omp parallel shared(output, input) firstprivate(nrows, ncols) - { + { #pragma omp for schedule(dynamic) nowait - for (int i = 0; i < nrows; i++) - for (int j = 0; j < ncols; j = j + 3) - { - output(i, j ) = lutR[input(i, j )]; - output(i, j+1) = lutG[input(i, j+1)]; - output(i, j+2) = lutB[input(i, j+2)]; - } + for (int i = 0; i < nrows; i++) + for (int j = 0; j < ncols; j = j + 3) { + output(i, j) = lutR[input(i, j)]; + output(i, j + 1) = lutG[input(i, j + 1)]; + output(i, j + 2) = lutB[input(i, j + 2)]; } } - return; + } + return; } diff --git a/filmulator-gui/core/colorSpaces.cpp b/filmulator-gui/core/colorSpaces.cpp index 9ebdb7ac..fcd8362d 100644 --- a/filmulator-gui/core/colorSpaces.cpp +++ b/filmulator-gui/core/colorSpaces.cpp @@ -1,534 +1,466 @@ -//#include "filmSim.hpp" +// #include "filmSim.hpp" #include "lut.hpp" -//Constants for conversion to and from L*a*b* -#define LAB_EPSILON (216.0/24389.0) -#define LAB_KAPPA (24389.0/27.0) +// Constants for conversion to and from L*a*b* +#define LAB_EPSILON (216.0 / 24389.0) +#define LAB_KAPPA (24389.0 / 27.0) -//D50 reference white point in XYZ for conversion to and from L*a*b* -//#define LAB_XR 0.9642 //This is supposedly what Adobe uses. -#define LAB_XR 0.96422 //This is the sum of the X row of sRGB_to_XYZ +// D50 reference white point in XYZ for conversion to and from L*a*b* +// #define LAB_XR 0.9642 //This is supposedly what Adobe uses. +#define LAB_XR 0.96422// This is the sum of the X row of sRGB_to_XYZ #define LAB_YR 1.00000 -//#define LAB_ZR 0.8249 //This is supposedly what Adobe uses. -#define LAB_ZR 0.82511 //This is the sum of the Z row of sRGB_to_XYZ +// #define LAB_ZR 0.8249 //This is supposedly what Adobe uses. +#define LAB_ZR 0.82511// This is the sum of the Z row of sRGB_to_XYZ -//Converts sRGB with D50 illuminant to XYZ with D50 illuminant. -void sRGB_to_XYZ(float R, float G, float B, - float &X, float &Y, float &Z) +// Converts sRGB with D50 illuminant to XYZ with D50 illuminant. +void sRGB_to_XYZ(float R, float G, float B, float &X, float &Y, float &Z) { - X = 0.4360747*R + 0.3850649*G + 0.1430804*B; - Y = 0.2225045*R + 0.7168786*G + 0.0606169*B; - Z = 0.0139322*R + 0.0971045*G + 0.7141733*B; + X = 0.4360747 * R + 0.3850649 * G + 0.1430804 * B; + Y = 0.2225045 * R + 0.7168786 * G + 0.0606169 * B; + Z = 0.0139322 * R + 0.0971045 * G + 0.7141733 * B; } -//Converts XYZ with D50 illuminant to sRGB with D50 illuminant. -void XYZ_to_sRGB(float X, float Y, float Z, - float &R, float &G, float &B) +// Converts XYZ with D50 illuminant to sRGB with D50 illuminant. +void XYZ_to_sRGB(float X, float Y, float Z, float &R, float &G, float &B) { - R = 3.1338561*X - 1.6168667*Y - 0.4906146*Z; - G = -0.9787684*X + 1.9161415*Y + 0.0334540*Z; - B = 0.0719453*X - 0.2289914*Y + 1.4052427*Z; + R = 3.1338561 * X - 1.6168667 * Y - 0.4906146 * Z; + G = -0.9787684 * X + 1.9161415 * Y + 0.0334540 * Z; + B = 0.0719453 * X - 0.2289914 * Y + 1.4052427 * Z; } -//Matrix inverse for 3x3 matrices +// Matrix inverse for 3x3 matrices void inverse(const float in[3][3], float (&out)[3][3]) { - float det = in[0][0] * (in[1][1]*in[2][2] - in[2][1]*in[1][2]) - - in[0][1] * (in[1][0]*in[2][2] - in[1][2]*in[2][0]) + - in[0][2] * (in[1][0]*in[2][1] - in[1][1]*in[2][0]); - float invdet = 1 / det; - - out[0][0] = (in[1][1]*in[2][2] - in[2][1]*in[1][2]) * invdet; - out[0][1] = (in[0][2]*in[2][1] - in[0][1]*in[2][2]) * invdet; - out[0][2] = (in[0][1]*in[1][2] - in[0][2]*in[1][1]) * invdet; - out[1][0] = (in[1][2]*in[2][0] - in[1][0]*in[2][2]) * invdet; - out[1][1] = (in[0][0]*in[2][2] - in[0][2]*in[2][0]) * invdet; - out[1][2] = (in[1][0]*in[0][2] - in[0][0]*in[1][2]) * invdet; - out[2][0] = (in[1][0]*in[2][1] - in[2][0]*in[1][1]) * invdet; - out[2][1] = (in[2][0]*in[0][1] - in[0][0]*in[2][1]) * invdet; - out[2][2] = (in[0][0]*in[1][1] - in[1][0]*in[0][1]) * invdet; + float det = in[0][0] * (in[1][1] * in[2][2] - in[2][1] * in[1][2]) + - in[0][1] * (in[1][0] * in[2][2] - in[1][2] * in[2][0]) + + in[0][2] * (in[1][0] * in[2][1] - in[1][1] * in[2][0]); + float invdet = 1 / det; + + out[0][0] = (in[1][1] * in[2][2] - in[2][1] * in[1][2]) * invdet; + out[0][1] = (in[0][2] * in[2][1] - in[0][1] * in[2][2]) * invdet; + out[0][2] = (in[0][1] * in[1][2] - in[0][2] * in[1][1]) * invdet; + out[1][0] = (in[1][2] * in[2][0] - in[1][0] * in[2][2]) * invdet; + out[1][1] = (in[0][0] * in[2][2] - in[0][2] * in[2][0]) * invdet; + out[1][2] = (in[1][0] * in[0][2] - in[0][0] * in[1][2]) * invdet; + out[2][0] = (in[1][0] * in[2][1] - in[2][0] * in[1][1]) * invdet; + out[2][1] = (in[2][0] * in[0][1] - in[0][0] * in[2][1]) * invdet; + out[2][2] = (in[0][0] * in[1][1] - in[1][0] * in[0][1]) * invdet; } -//Linearizes gamma-curved sRGB. -//Reference: http://www.brucelindbloom.com/index.html?Eqn_RGB_to_XYZ.html -//http://stackoverflow.com/questions/6475373/optimizations-for-pow-with-const-non-integer-exponent +// Linearizes gamma-curved sRGB. +// Reference: http://www.brucelindbloom.com/index.html?Eqn_RGB_to_XYZ.html +// http://stackoverflow.com/questions/6475373/optimizations-for-pow-with-const-non-integer-exponent float sRGB_inverse_gamma(float c) { - if (c < 0) - { - return 0; - } - else if (c <= 0.04045) - { - return c / 12.92; - } - else if (c < 1) - { - return pow((c+0.055)/1.055,2.4); - } - else - { - return 1; - } + if (c < 0) { + return 0; + } else if (c <= 0.04045) { + return c / 12.92; + } else if (c < 1) { + return pow((c + 0.055) / 1.055, 2.4); + } else { + return 1; + } } -//Gamma-compresses linear into sRGB. -//http://stackoverflow.com/questions/6475373/optimizations-for-pow-with-const-non-integer-exponent +// Gamma-compresses linear into sRGB. +// http://stackoverflow.com/questions/6475373/optimizations-for-pow-with-const-non-integer-exponent float sRGB_forward_gamma(float c) { - if (c < 0) - { - return 0; - } - else if (c <= 0.0031308) - { - return c * 12.92; - } - else if (c < 1) - { - return 1.055*pow(c,1/2.4) - 0.055; - } - else - { - return 1; - } + if (c < 0) { + return 0; + } else if (c <= 0.0031308) { + return c * 12.92; + } else if (c < 1) { + return 1.055 * pow(c, 1 / 2.4) - 0.055; + } else { + return 1; + } } -//Linearizes gamma-curved sRGB, but unbounded. -//Reference: http://www.brucelindbloom.com/index.html?Eqn_RGB_to_XYZ.html -//http://stackoverflow.com/questions/6475373/optimizations-for-pow-with-const-non-integer-exponent +// Linearizes gamma-curved sRGB, but unbounded. +// Reference: http://www.brucelindbloom.com/index.html?Eqn_RGB_to_XYZ.html +// http://stackoverflow.com/questions/6475373/optimizations-for-pow-with-const-non-integer-exponent float sRGB_inverse_gamma_unclipped(float c) { - if (c <= 0.04045) - { - return c / 12.92; - } - else - { - return pow((c+0.055)/1.055,2.4); - } + if (c <= 0.04045) { + return c / 12.92; + } else { + return pow((c + 0.055) / 1.055, 2.4); + } } -//Gamma-compresses linear into sRGB, but unbounded. -//http://stackoverflow.com/questions/6475373/optimizations-for-pow-with-const-non-integer-exponent +// Gamma-compresses linear into sRGB, but unbounded. +// http://stackoverflow.com/questions/6475373/optimizations-for-pow-with-const-non-integer-exponent float sRGB_forward_gamma_unclipped(float c) { - if (c <= 0.0031308) - { - return c * 12.92; - } - else - { - return 1.055*pow(c,1/2.4) - 0.055; - } + if (c <= 0.0031308) { + return c * 12.92; + } else { + return 1.055 * pow(c, 1 / 2.4) - 0.055; + } } -//Linearize L* curved XYZ (coming from L*a*b*) -//Reference: http://www.brucelindbloom.com/index.html?Eqn_Lab_to_XYZ.html +// Linearize L* curved XYZ (coming from L*a*b*) +// Reference: http://www.brucelindbloom.com/index.html?Eqn_Lab_to_XYZ.html float Lab_inverse_gamma(float c) { - if(c < 0) - { - return 0; - } - else if (c <= 0.08) - { - return c / LAB_KAPPA; - } - else if (c < 1) - { - return pow((c+0.16)/1.16,3); - } - else - { - return 1; - } + if (c < 0) { + return 0; + } else if (c <= 0.08) { + return c / LAB_KAPPA; + } else if (c < 1) { + return pow((c + 0.16) / 1.16, 3); + } else { + return 1; + } } -//L* curve linear XYZ data in preparation to going to L*a*b* -//Reference: http://www.brucelindbloom.com/index.html?Eqn_XYZ_to_Lab.html +// L* curve linear XYZ data in preparation to going to L*a*b* +// Reference: http://www.brucelindbloom.com/index.html?Eqn_XYZ_to_Lab.html float Lab_forward_gamma(float c) { - if(c < 0) - { - return 0; - } - else if (c <= LAB_EPSILON) - { - return (LAB_KAPPA*c+16)/116; - } - else if (c < 1) - { - return pow(c,1.0/3.0); - } - else - { - return 1; - } + if (c < 0) { + return 0; + } else if (c <= LAB_EPSILON) { + return (LAB_KAPPA * c + 16) / 116; + } else if (c < 1) { + return pow(c, 1.0 / 3.0); + } else { + return 1; + } } -//Arithmetic operations from L* curved XYZ to L*a*b* -//Reference: http://www.brucelindbloom.com/index.html?Eqn_XYZ_to_Lab.html -void XYZ_to_Lab(float fx, float fy, float fz, - float &L, float &a, float &b) +// Arithmetic operations from L* curved XYZ to L*a*b* +// Reference: http://www.brucelindbloom.com/index.html?Eqn_XYZ_to_Lab.html +void XYZ_to_Lab(float fx, float fy, float fz, float &L, float &a, float &b) { - L = max( 0.0f, min(1.0f, 116*fy - 16 )); - a = max(-1.0f, min(1.0f, 500*(fx - fy))); - b = max(-1.0f, min(1.0f, 200*(fy - fz))); + L = max(0.0f, min(1.0f, 116 * fy - 16)); + a = max(-1.0f, min(1.0f, 500 * (fx - fy))); + b = max(-1.0f, min(1.0f, 200 * (fy - fz))); } -//Arithmetic operations from L*a*b* to L* curved XYZ -void Lab_to_XYZ(float L, float a, float b, - float &fx, float &fy, float &fz) +// Arithmetic operations from L*a*b* to L* curved XYZ +void Lab_to_XYZ(float L, float a, float b, float &fx, float &fy, float &fz) { - fy = (L + 16)/116; - fx = a/500 + fy; - fz = fy - b/200; + fy = (L + 16) / 116; + fx = a / 500 + fy; + fz = fy - b / 200; } -//Converts gamma-curved sRGB D50 to L*a*b*, all unsigned shorts from 0 to 65535. -//The L* is 0 = 0, 1 = 65535. -//a* and b* are -1 = 1, 0 = 32768, +1 = 65535. -//Reference: http://www.brucelindbloom.com/index.html?Eqn_RGB_to_XYZ.html -void sRGB_to_Lab_s(matrix &in, - matrix &Lab) +// Converts gamma-curved sRGB D50 to L*a*b*, all unsigned shorts from 0 to 65535. +// The L* is 0 = 0, 1 = 65535. +// a* and b* are -1 = 1, 0 = 32768, +1 = 65535. +// Reference: http://www.brucelindbloom.com/index.html?Eqn_RGB_to_XYZ.html +void sRGB_to_Lab_s(matrix &in, matrix &Lab) { - int nRows = in.nr(); - int nCols = in.nc(); + int nRows = in.nr(); + int nCols = in.nc(); - Lab.set_size(nRows, nCols); + Lab.set_size(nRows, nCols); #pragma omp parallel shared(in, Lab) firstprivate(nRows, nCols) - { + { #pragma omp for schedule(dynamic) nowait - for (int i = 0; i < nRows; i++) - { - for (int j = 0; j < nCols; j += 3) - { - //First, linearize the sRGB. - float r = sRGB_inverse_gamma(float(in(i, j ))/65535.0); - float g = sRGB_inverse_gamma(float(in(i, j+1))/65535.0); - float b = sRGB_inverse_gamma(float(in(i, j+2))/65535.0); - - //Next, convert to XYZ. - float x, y, z; - sRGB_to_XYZ(r, g, b, x, y, z); - - //Next, convert to L*a*b* - //First we must scale by the white point of D50. - float xr = x/LAB_XR; - float yr = y/LAB_YR; - float zr = z/LAB_ZR; - //Using that, we apply the L* curve. - float fx = Lab_forward_gamma(xr); - float fy = Lab_forward_gamma(yr); - float fz = Lab_forward_gamma(zr); - //Next is the not-quite-matrix operations. - float L, a;// b is already declared; - XYZ_to_Lab(fx, fy, fz, L, a, b); - Lab(i, j ) = (unsigned short)(65535*L); - Lab(i, j+1) = (unsigned short)(32767*a + 32768); - Lab(i, j+2) = (unsigned short)(32767*b + 32768); - } - } + for (int i = 0; i < nRows; i++) { + for (int j = 0; j < nCols; j += 3) { + // First, linearize the sRGB. + float r = sRGB_inverse_gamma(float(in(i, j)) / 65535.0); + float g = sRGB_inverse_gamma(float(in(i, j + 1)) / 65535.0); + float b = sRGB_inverse_gamma(float(in(i, j + 2)) / 65535.0); + + // Next, convert to XYZ. + float x, y, z; + sRGB_to_XYZ(r, g, b, x, y, z); + + // Next, convert to L*a*b* + // First we must scale by the white point of D50. + float xr = x / LAB_XR; + float yr = y / LAB_YR; + float zr = z / LAB_ZR; + // Using that, we apply the L* curve. + float fx = Lab_forward_gamma(xr); + float fy = Lab_forward_gamma(yr); + float fz = Lab_forward_gamma(zr); + // Next is the not-quite-matrix operations. + float L, a;// b is already declared; + XYZ_to_Lab(fx, fy, fz, L, a, b); + Lab(i, j) = (unsigned short)(65535 * L); + Lab(i, j + 1) = (unsigned short)(32767 * a + 32768); + Lab(i, j + 2) = (unsigned short)(32767 * b + 32768); + } } + } } -//Converts gamma-curved sRGB D50 to linear, short int to float. -void sRGB_linearize(matrix &in, - matrix &out) +// Converts gamma-curved sRGB D50 to linear, short int to float. +void sRGB_linearize(matrix &in, matrix &out) { - int nRows = in.nr(); - int nCols = in.nc(); - - // build lookup table - float invgamma[65536]; - for (int i = 0; i < 65536; i++) - { - invgamma[i] = sRGB_inverse_gamma(i / 65535.0f); - } + int nRows = in.nr(); + int nCols = in.nc(); + + // build lookup table + float invgamma[65536]; + for (int i = 0; i < 65536; i++) { invgamma[i] = sRGB_inverse_gamma(i / 65535.0f); } - out.set_size(nRows, nCols); + out.set_size(nRows, nCols); #pragma omp parallel shared(in, out) firstprivate(nRows, nCols) - { + { #pragma omp for schedule(dynamic) nowait - for (int i = 0; i < nRows; i++) - { - for (int j = 0; j < nCols; j += 3) - { - out(i, j ) = invgamma[in(i, j )]; - out(i, j+1) = invgamma[in(i, j+1)]; - out(i, j+2) = invgamma[in(i, j+2)]; - } - } + for (int i = 0; i < nRows; i++) { + for (int j = 0; j < nCols; j += 3) { + out(i, j) = invgamma[in(i, j)]; + out(i, j + 1) = invgamma[in(i, j + 1)]; + out(i, j + 2) = invgamma[in(i, j + 2)]; + } } + } } -//Converts linear float sRGB D50 to gamma-curved, float to short int. -//Reference: http://www.brucelindbloom.com/index.html?Eqn_RGB_to_XYZ.html -void sRGB_gammacurve(matrix &in, - matrix &out) +// Converts linear float sRGB D50 to gamma-curved, float to short int. +// Reference: http://www.brucelindbloom.com/index.html?Eqn_RGB_to_XYZ.html +void sRGB_gammacurve(matrix &in, matrix &out) { - int nRows = in.nr(); - int nCols = in.nc(); + int nRows = in.nr(); + int nCols = in.nc(); - out.set_size(nRows,nCols); + out.set_size(nRows, nCols); #pragma omp parallel shared(in, out) firstprivate(nRows, nCols) - { + { #pragma omp for schedule(dynamic) nowait - for (int i = 0; i < nRows; i++) - { - for (int j = 0; j < nCols; j += 3) - { - //First, linearize the sRGB. - out(i, j ) = (unsigned short)(65535*sRGB_forward_gamma(in(i, j ))); - out(i, j+1) = (unsigned short)(65535*sRGB_forward_gamma(in(i, j+1))); - out(i, j+2) = (unsigned short)(65535*sRGB_forward_gamma(in(i, j+2))); - } - } + for (int i = 0; i < nRows; i++) { + for (int j = 0; j < nCols; j += 3) { + // First, linearize the sRGB. + out(i, j) = (unsigned short)(65535 * sRGB_forward_gamma(in(i, j))); + out(i, j + 1) = (unsigned short)(65535 * sRGB_forward_gamma(in(i, j + 1))); + out(i, j + 2) = (unsigned short)(65535 * sRGB_forward_gamma(in(i, j + 2))); + } } + } } -//Convert linear float sRGB with a range of roughly 65535 to 200*oklab -//Reference: https://bottosson.github.io/posts/oklab/ -void sRGB_to_oklab(matrix &in, - matrix &out) +// Convert linear float sRGB with a range of roughly 65535 to 200*oklab +// Reference: https://bottosson.github.io/posts/oklab/ +void sRGB_to_oklab(matrix &in, matrix &out) { - int nRows = in.nr(); - int nCols = in.nc(); + int nRows = in.nr(); + int nCols = in.nc(); - out.set_size(nRows,nCols); + out.set_size(nRows, nCols); #pragma omp parallel shared(in, out) firstprivate(nRows, nCols) - { + { #pragma omp for schedule(dynamic) nowait - for (int i = 0; i < nRows; i++) - { - for (int j = 0; j < nCols; j += 3) - { - const float r = in(i, j+0) / 65535.0f; - const float g = in(i, j+1) / 65535.0f; - const float b = in(i, j+2) / 65535.0f; - - const float l = 0.4122214708f * r + 0.5363325363f * g + 0.0514459929f * b; - const float m = 0.2119034982f * r + 0.6806995451f * g + 0.1073969566f * b; - const float s = 0.0883024619f * r + 0.2817188376f * g + 0.6299787005f * b; - - const float l_ = cbrtf(l); - const float m_ = cbrtf(m); - const float s_ = cbrtf(s); - - out(i, j+0) = (0.2104542553f*l_ + 0.7936177850f*m_ - 0.0040720468f*s_) * 200; - out(i, j+1) = (1.9779984951f*l_ - 2.4285922050f*m_ + 0.4505937099f*s_) * 200; - out(i, j+2) = (0.0259040371f*l_ + 0.7827717662f*m_ - 0.8086757660f*s_) * 200; - } - } + for (int i = 0; i < nRows; i++) { + for (int j = 0; j < nCols; j += 3) { + const float r = in(i, j + 0) / 65535.0f; + const float g = in(i, j + 1) / 65535.0f; + const float b = in(i, j + 2) / 65535.0f; + + const float l = 0.4122214708f * r + 0.5363325363f * g + 0.0514459929f * b; + const float m = 0.2119034982f * r + 0.6806995451f * g + 0.1073969566f * b; + const float s = 0.0883024619f * r + 0.2817188376f * g + 0.6299787005f * b; + + const float l_ = cbrtf(l); + const float m_ = cbrtf(m); + const float s_ = cbrtf(s); + + out(i, j + 0) = (0.2104542553f * l_ + 0.7936177850f * m_ - 0.0040720468f * s_) * 200; + out(i, j + 1) = (1.9779984951f * l_ - 2.4285922050f * m_ + 0.4505937099f * s_) * 200; + out(i, j + 2) = (0.0259040371f * l_ + 0.7827717662f * m_ - 0.8086757660f * s_) * 200; + } } + } } -//Convert float 200*oklab to linear float sRGB up to roughly 65535 +// Convert float 200*oklab to linear float sRGB up to roughly 65535 -//Reference: https://bottosson.github.io/posts/oklab/ -void oklab_to_sRGB(matrix &in, - matrix &out) +// Reference: https://bottosson.github.io/posts/oklab/ +void oklab_to_sRGB(matrix &in, matrix &out) { - int nRows = in.nr(); - int nCols = in.nc(); + int nRows = in.nr(); + int nCols = in.nc(); - out.set_size(nRows,nCols); + out.set_size(nRows, nCols); #pragma omp parallel shared(in, out) firstprivate(nRows, nCols) - { + { #pragma omp for schedule(dynamic) nowait - for (int i = 0; i < nRows; i++) - { - for (int j = 0; j < nCols; j += 3) - { - const float L = in(i, j+0)/200; - const float a = in(i, j+1)/200; - const float b = in(i, j+2)/200; - - const float l_ = L + 0.3963377774f * a + 0.2158037573f * b; - const float m_ = L - 0.1055613458f * a - 0.0638541728f * b; - const float s_ = L - 0.0894841775f * a - 1.2914855480f * b; - - const float l = l_*l_*l_; - const float m = m_*m_*m_; - const float s = s_*s_*s_; - - out(i, j+0) = (+4.0767416621f * l - 3.3077115913f * m + 0.2309699292f * s) * 65535.0f; - out(i, j+1) = (-1.2684380046f * l + 2.6097574011f * m - 0.3413193965f * s) * 65535.0f; - out(i, j+2) = (-0.0041960863f * l - 0.7034186147f * m + 1.7076147010f * s) * 65535.0f; - } - } + for (int i = 0; i < nRows; i++) { + for (int j = 0; j < nCols; j += 3) { + const float L = in(i, j + 0) / 200; + const float a = in(i, j + 1) / 200; + const float b = in(i, j + 2) / 200; + + const float l_ = L + 0.3963377774f * a + 0.2158037573f * b; + const float m_ = L - 0.1055613458f * a - 0.0638541728f * b; + const float s_ = L - 0.0894841775f * a - 1.2914855480f * b; + + const float l = l_ * l_ * l_; + const float m = m_ * m_ * m_; + const float s = s_ * s_ * s_; + + out(i, j + 0) = (+4.0767416621f * l - 3.3077115913f * m + 0.2309699292f * s) * 65535.0f; + out(i, j + 1) = (-1.2684380046f * l + 2.6097574011f * m - 0.3413193965f * s) * 65535.0f; + out(i, j + 2) = (-0.0041960863f * l - 0.7034186147f * m + 1.7076147010f * s) * 65535.0f; + } } + } } -//Convert raw color to sRGB, don't clip negatives. -void raw_to_sRGB(matrix &input, - matrix &output, - const float cam2rgb[3][3]) +// Convert raw color to sRGB, don't clip negatives. +void raw_to_sRGB(matrix &input, matrix &output, const float cam2rgb[3][3]) { - int nRows = input.nr(); - int nCols = input.nc(); + int nRows = input.nr(); + int nCols = input.nc(); - output.set_size(nRows, nCols); + output.set_size(nRows, nCols); #pragma omp parallel shared(output, input) firstprivate(nRows, nCols) - { + { #pragma omp for schedule(dynamic) nowait - for (int i = 0; i < nRows; i++) - { - for (int j = 0; j < nCols; j += 3) - { - output(i, j ) = cam2rgb[0][0]*input(i, j) + cam2rgb[0][1]*input(i, j+1) + cam2rgb[0][2]*input(i, j+2); - output(i, j+1) = cam2rgb[1][0]*input(i, j) + cam2rgb[1][1]*input(i, j+1) + cam2rgb[1][2]*input(i, j+2); - output(i, j+2) = cam2rgb[2][0]*input(i, j) + cam2rgb[2][1]*input(i, j+1) + cam2rgb[2][2]*input(i, j+2); - - } - } + for (int i = 0; i < nRows; i++) { + for (int j = 0; j < nCols; j += 3) { + output(i, j) = cam2rgb[0][0] * input(i, j) + cam2rgb[0][1] * input(i, j + 1) + cam2rgb[0][2] * input(i, j + 2); + output(i, j + 1) = + cam2rgb[1][0] * input(i, j) + cam2rgb[1][1] * input(i, j + 1) + cam2rgb[1][2] * input(i, j + 2); + output(i, j + 2) = + cam2rgb[2][0] * input(i, j) + cam2rgb[2][1] * input(i, j + 1) + cam2rgb[2][2] * input(i, j + 2); + } } + } } -//Convert raw color to sRGB, don't clip negatives. -void sRGB_to_raw(matrix &input, - matrix &output, - const float cam2rgb[3][3]) +// Convert raw color to sRGB, don't clip negatives. +void sRGB_to_raw(matrix &input, matrix &output, const float cam2rgb[3][3]) { - int nRows = input.nr(); - int nCols = input.nc(); + int nRows = input.nr(); + int nCols = input.nc(); - float rgb2cam[3][3]; - inverse(cam2rgb, rgb2cam); + float rgb2cam[3][3]; + inverse(cam2rgb, rgb2cam); - output.set_size(nRows, nCols); + output.set_size(nRows, nCols); #pragma omp parallel shared(output, input) firstprivate(nRows, nCols) - { + { #pragma omp for schedule(dynamic) nowait - for (int i = 0; i < nRows; i++) - { - for (int j = 0; j < nCols; j += 3) - { - output(i, j ) = rgb2cam[0][0]*input(i, j) + rgb2cam[0][1]*input(i, j+1) + rgb2cam[0][2]*input(i, j+2); - output(i, j+1) = rgb2cam[1][0]*input(i, j) + rgb2cam[1][1]*input(i, j+1) + rgb2cam[1][2]*input(i, j+2); - output(i, j+2) = rgb2cam[2][0]*input(i, j) + rgb2cam[2][1]*input(i, j+1) + rgb2cam[2][2]*input(i, j+2); - - } - } + for (int i = 0; i < nRows; i++) { + for (int j = 0; j < nCols; j += 3) { + output(i, j) = rgb2cam[0][0] * input(i, j) + rgb2cam[0][1] * input(i, j + 1) + rgb2cam[0][2] * input(i, j + 2); + output(i, j + 1) = + rgb2cam[1][0] * input(i, j) + rgb2cam[1][1] * input(i, j + 1) + rgb2cam[1][2] * input(i, j + 2); + output(i, j + 2) = + rgb2cam[2][0] * input(i, j) + rgb2cam[2][1] * input(i, j + 1) + rgb2cam[2][2] * input(i, j + 2); + } } + } } -//Convert raw color (range of roughly 65535) to 200*oklab, don't clip negatives. -//Reference: https://bottosson.github.io/posts/oklab/ -void raw_to_oklab(matrix &input, - matrix &output, - const float cam2rgb[3][3]) +// Convert raw color (range of roughly 65535) to 200*oklab, don't clip negatives. +// Reference: https://bottosson.github.io/posts/oklab/ +void raw_to_oklab(matrix &input, matrix &output, const float cam2rgb[3][3]) { - int nRows = input.nr(); - int nCols = input.nc(); + int nRows = input.nr(); + int nCols = input.nc(); - output.set_size(nRows, nCols); + output.set_size(nRows, nCols); #pragma omp parallel shared(output, input) firstprivate(nRows, nCols) - { + { #pragma omp for schedule(dynamic) nowait - for (int i = 0; i < nRows; i++) - { - for (int j = 0; j < nCols; j += 3) - { - //raw to sRGB 0-1 - const float r = (cam2rgb[0][0]*input(i, j) + cam2rgb[0][1]*input(i, j+1) + cam2rgb[0][2]*input(i, j+2))/65535.0f; - const float g = (cam2rgb[1][0]*input(i, j) + cam2rgb[1][1]*input(i, j+1) + cam2rgb[1][2]*input(i, j+2))/65535.0f; - const float b = (cam2rgb[2][0]*input(i, j) + cam2rgb[2][1]*input(i, j+1) + cam2rgb[2][2]*input(i, j+2))/65535.0f; - - //sRGB to oklab - const float l = 0.4122214708f * r + 0.5363325363f * g + 0.0514459929f * b; - const float m = 0.2119034982f * r + 0.6806995451f * g + 0.1073969566f * b; - const float s = 0.0883024619f * r + 0.2817188376f * g + 0.6299787005f * b; - - const float l_ = cbrtf(l); - const float m_ = cbrtf(m); - const float s_ = cbrtf(s); - - output(i, j+0) = (0.2104542553f*l_ + 0.7936177850f*m_ - 0.0040720468f*s_) * 200; - output(i, j+1) = (1.9779984951f*l_ - 2.4285922050f*m_ + 0.4505937099f*s_) * 200; - output(i, j+2) = (0.0259040371f*l_ + 0.7827717662f*m_ - 0.8086757660f*s_) * 200; - } - } + for (int i = 0; i < nRows; i++) { + for (int j = 0; j < nCols; j += 3) { + // raw to sRGB 0-1 + const float r = + (cam2rgb[0][0] * input(i, j) + cam2rgb[0][1] * input(i, j + 1) + cam2rgb[0][2] * input(i, j + 2)) / 65535.0f; + const float g = + (cam2rgb[1][0] * input(i, j) + cam2rgb[1][1] * input(i, j + 1) + cam2rgb[1][2] * input(i, j + 2)) / 65535.0f; + const float b = + (cam2rgb[2][0] * input(i, j) + cam2rgb[2][1] * input(i, j + 1) + cam2rgb[2][2] * input(i, j + 2)) / 65535.0f; + + // sRGB to oklab + const float l = 0.4122214708f * r + 0.5363325363f * g + 0.0514459929f * b; + const float m = 0.2119034982f * r + 0.6806995451f * g + 0.1073969566f * b; + const float s = 0.0883024619f * r + 0.2817188376f * g + 0.6299787005f * b; + + const float l_ = cbrtf(l); + const float m_ = cbrtf(m); + const float s_ = cbrtf(s); + + output(i, j + 0) = (0.2104542553f * l_ + 0.7936177850f * m_ - 0.0040720468f * s_) * 200; + output(i, j + 1) = (1.9779984951f * l_ - 2.4285922050f * m_ + 0.4505937099f * s_) * 200; + output(i, j + 2) = (0.0259040371f * l_ + 0.7827717662f * m_ - 0.8086757660f * s_) * 200; + } } + } } -//Convert 200*oklab to xyz -//Reference: https://bottosson.github.io/posts/oklab/ -void oklab_to_xyz(float L, float a, float b, - float &x, float &y, float &z) +// Convert 200*oklab to xyz +// Reference: https://bottosson.github.io/posts/oklab/ +void oklab_to_xyz(float L, float a, float b, float &x, float &y, float &z) { - L /= 200; - a /= 200; - b /= 200; - - float lp, mp, sp; - - //m2 inverse - lp = L + 0.3963377921737679*a + 0.2158037580607588*b; - mp = L - 0.1055613423236563*a - 0.06385417477170588*b; - sp = L - 0.08948418209496574*a - 1.291485537864092*b; - - float l, m, s; - l = lp*lp*lp; - m = mp*mp*mp; - s = sp*sp*sp; - - //m1 inverse - x = 1.227013851103521*l - 0.5577999806518222*m + 0.2812561489664678*s; - y = -0.0405801784232806*l + 1.11225686961683*m - 0.07167667866560121*s; - z = -0.07638128450570689*l - 0.4214819784180127*m + 1.586163220440795*s; + L /= 200; + a /= 200; + b /= 200; + + float lp, mp, sp; + + // m2 inverse + lp = L + 0.3963377921737679 * a + 0.2158037580607588 * b; + mp = L - 0.1055613423236563 * a - 0.06385417477170588 * b; + sp = L - 0.08948418209496574 * a - 1.291485537864092 * b; + + float l, m, s; + l = lp * lp * lp; + m = mp * mp * mp; + s = sp * sp * sp; + + // m1 inverse + x = 1.227013851103521 * l - 0.5577999806518222 * m + 0.2812561489664678 * s; + y = -0.0405801784232806 * l + 1.11225686961683 * m - 0.07167667866560121 * s; + z = -0.07638128450570689 * l - 0.4214819784180127 * m + 1.586163220440795 * s; } -//Convert 200*oklab to raw color (roughly 65535 max) -void oklab_to_raw(matrix &input, - matrix &output, - const float cam2rgb[3][3]) +// Convert 200*oklab to raw color (roughly 65535 max) +void oklab_to_raw(matrix &input, matrix &output, const float cam2rgb[3][3]) { - int nRows = input.nr(); - int nCols = input.nc(); + int nRows = input.nr(); + int nCols = input.nc(); - float rgb2cam[3][3]; - inverse(cam2rgb, rgb2cam); + float rgb2cam[3][3]; + inverse(cam2rgb, rgb2cam); - output.set_size(nRows, nCols); + output.set_size(nRows, nCols); #pragma omp parallel shared(output, input) firstprivate(nRows, nCols) - { + { #pragma omp for schedule(dynamic) nowait - for (int i = 0; i < nRows; i++) - { - for (int j = 0; j < nCols; j += 3) - { - //copied straight from oklab_to_sRGB - const float L = input(i, j+0)/200; - const float a = input(i, j+1)/200; - const float b = input(i, j+2)/200; - - const float l_ = L + 0.3963377774f * a + 0.2158037573f * b; - const float m_ = L - 0.1055613458f * a - 0.0638541728f * b; - const float s_ = L - 0.0894841775f * a - 1.2914855480f * b; - - const float l = l_*l_*l_; - const float m = m_*m_*m_; - const float s = s_*s_*s_; - - const float sRGBr = (+4.0767416621f * l - 3.3077115913f * m + 0.2309699292f * s) * 65535.0f; - const float sRGBg = (-1.2684380046f * l + 2.6097574011f * m - 0.3413193965f * s) * 65535.0f; - const float sRGBb = (-0.0041960863f * l - 0.7034186147f * m + 1.7076147010f * s) * 65535.0f; - - //Use eigen to invert cam2rgb, then use that to go from sRGB to raw color - output(i, j+0) = rgb2cam[0][0] * sRGBr + rgb2cam[0][1] * sRGBg + rgb2cam[0][2] * sRGBb; - output(i, j+1) = rgb2cam[1][0] * sRGBr + rgb2cam[1][1] * sRGBg + rgb2cam[1][2] * sRGBb; - output(i, j+2) = rgb2cam[2][0] * sRGBr + rgb2cam[2][1] * sRGBg + rgb2cam[2][2] * sRGBb; - } - } + for (int i = 0; i < nRows; i++) { + for (int j = 0; j < nCols; j += 3) { + // copied straight from oklab_to_sRGB + const float L = input(i, j + 0) / 200; + const float a = input(i, j + 1) / 200; + const float b = input(i, j + 2) / 200; + + const float l_ = L + 0.3963377774f * a + 0.2158037573f * b; + const float m_ = L - 0.1055613458f * a - 0.0638541728f * b; + const float s_ = L - 0.0894841775f * a - 1.2914855480f * b; + + const float l = l_ * l_ * l_; + const float m = m_ * m_ * m_; + const float s = s_ * s_ * s_; + + const float sRGBr = (+4.0767416621f * l - 3.3077115913f * m + 0.2309699292f * s) * 65535.0f; + const float sRGBg = (-1.2684380046f * l + 2.6097574011f * m - 0.3413193965f * s) * 65535.0f; + const float sRGBb = (-0.0041960863f * l - 0.7034186147f * m + 1.7076147010f * s) * 65535.0f; + + // Use eigen to invert cam2rgb, then use that to go from sRGB to raw color + output(i, j + 0) = rgb2cam[0][0] * sRGBr + rgb2cam[0][1] * sRGBg + rgb2cam[0][2] * sRGBb; + output(i, j + 1) = rgb2cam[1][0] * sRGBr + rgb2cam[1][1] * sRGBg + rgb2cam[1][2] * sRGBb; + output(i, j + 2) = rgb2cam[2][0] * sRGBr + rgb2cam[2][1] * sRGBg + rgb2cam[2][2] * sRGBb; + } } + } } diff --git a/filmulator-gui/core/curves.cpp b/filmulator-gui/core/curves.cpp index 2fa199fd..a09e6930 100644 --- a/filmulator-gui/core/curves.cpp +++ b/filmulator-gui/core/curves.cpp @@ -1,4 +1,4 @@ -/* +/* * This file is part of Filmulator. * * Copyright 2013 Omer Mano and Carlo Vaccari @@ -17,182 +17,175 @@ * along with Filmulator. If not, see */ #include "filmSim.hpp" -#include #include +#include float default_tonecurve(float input) { - //These are the coordinates for a quadratic bezier curve's - //control points. They should be in ascending order. - double p0x = 0; - double p0y = 0; - double p1x = 0.2; - double p1y = 1; - double p2x = 1; - double p2y = 1; - - //Math for a quadratic bezier curve - //a, b, c are as in a standard quadratic formula - //x or y are the two axes of computation. - double ax = p0x - 2*p1x + p2x; - double ay = p0y - 2*p1y + p2y; - double bx = -2*p0x + 2*p1x; - double by = -2*p0y + 2*p1y; - double cx = p0x - double(input); - double cy = p0y; - - //The bezier curves are defined parametrically, with respect to t. - //We need to find with respect to x, so we need to find what t value - //corresponds to the x. - double t_value = (-bx + sqrt(bx*bx - 4*ax*cx)) / (2*ax); - - double y_out = (ay*t_value*t_value + by*t_value + cy); - float output = float(y_out); - - return output; + // These are the coordinates for a quadratic bezier curve's + // control points. They should be in ascending order. + double p0x = 0; + double p0y = 0; + double p1x = 0.2; + double p1y = 1; + double p2x = 1; + double p2y = 1; + + // Math for a quadratic bezier curve + // a, b, c are as in a standard quadratic formula + // x or y are the two axes of computation. + double ax = p0x - 2 * p1x + p2x; + double ay = p0y - 2 * p1y + p2y; + double bx = -2 * p0x + 2 * p1x; + double by = -2 * p0y + 2 * p1y; + double cx = p0x - double(input); + double cy = p0y; + + // The bezier curves are defined parametrically, with respect to t. + // We need to find with respect to x, so we need to find what t value + // corresponds to the x. + double t_value = (-bx + sqrt(bx * bx - 4 * ax * cx)) / (2 * ax); + + double y_out = (ay * t_value * t_value + by * t_value + cy); + float output = float(y_out); + + return output; } -float shadows_highlights (float input, - float shadowsX, - float shadowsY, - float highlightsX, - float highlightsY) +float shadows_highlights(float input, float shadowsX, float shadowsY, float highlightsX, float highlightsY) { - float x = input; - float a = shadowsX; - float b = shadowsY; - float c = highlightsX; - float d = highlightsY; - - float y0a = 0.00; // initial y - float x0a = 0.00; // initial x - float y1a = b; // 1st influence y - float x1a = a; // 1st influence x - float y2a = d; // 2nd influence y - float x2a = c; // 2nd influence x - float y3a = 1.00; // final y - float x3a = 1.00; // final x - - float A = x3a - 3*x2a + 3*x1a - x0a; - float B = 3*x2a - 6*x1a + 3*x0a; - float C = 3*x1a - 3*x0a; - float D = x0a; - - float E = y3a - 3*y2a + 3*y1a - y0a; - float F = 3*y2a - 6*y1a + 3*y0a; - float G = 3*y1a - 3*y0a; - float H = y0a; - - // Solve for t given x (using Newton-Raphson), then solve for y given t. - // Assume for the first guess that t = x. - float currentt = x; - int nRefinementIterations = 5; - for (int i = 0; i < nRefinementIterations; i++) - { - float currentx = xFromT(currentt, A, B, C, D); - float currentslope = slopeFromT(currentt, A, B, C); - currentt -= (currentx - x) * (currentslope); - currentt = min(max(currentt, float(0)), float(1)); - } - - float y = yFromT (currentt, E, F, G, H); - return y; + float x = input; + float a = shadowsX; + float b = shadowsY; + float c = highlightsX; + float d = highlightsY; + + float y0a = 0.00;// initial y + float x0a = 0.00;// initial x + float y1a = b;// 1st influence y + float x1a = a;// 1st influence x + float y2a = d;// 2nd influence y + float x2a = c;// 2nd influence x + float y3a = 1.00;// final y + float x3a = 1.00;// final x + + float A = x3a - 3 * x2a + 3 * x1a - x0a; + float B = 3 * x2a - 6 * x1a + 3 * x0a; + float C = 3 * x1a - 3 * x0a; + float D = x0a; + + float E = y3a - 3 * y2a + 3 * y1a - y0a; + float F = 3 * y2a - 6 * y1a + 3 * y0a; + float G = 3 * y1a - 3 * y0a; + float H = y0a; + + // Solve for t given x (using Newton-Raphson), then solve for y given t. + // Assume for the first guess that t = x. + float currentt = x; + int nRefinementIterations = 5; + for (int i = 0; i < nRefinementIterations; i++) { + float currentx = xFromT(currentt, A, B, C, D); + float currentslope = slopeFromT(currentt, A, B, C); + currentt -= (currentx - x) * (currentslope); + currentt = min(max(currentt, float(0)), float(1)); + } + + float y = yFromT(currentt, E, F, G, H); + return y; } // Helper functions: float slopeFromT(float t, float A, float B, float C) { - float dtdx = 1.0 / (3.0*A*t*t + 2.0*B*t + C); - return dtdx; + float dtdx = 1.0 / (3.0 * A * t * t + 2.0 * B * t + C); + return dtdx; } float xFromT(float t, float A, float B, float C, float D) { - float x = A*(t*t*t) + B*(t*t) + C*t + D; - return x; + float x = A * (t * t * t) + B * (t * t) + C * t + D; + return x; } float yFromT(float t, float E, float F, float G, float H) { - float y = E*(t*t*t) + F*(t*t) + G*t + H; - return y; + float y = E * (t * t * t) + F * (t * t) + G * t + H; + return y; } -//This code was derived from the RawTherapee project, which says it was taken -//from Adobe's reference implementation for tone curves. +// This code was derived from the RawTherapee project, which says it was taken +// from Adobe's reference implementation for tone curves. // -//I couldn't find the original source, though... +// I couldn't find the original source, though... // -//An explanation of the algorithm: -//The algorithm applies the designated tone curve on the highest and the lowest -// values. -//On the middle value, instead of using the tone curve, which would induce hue -// shifts, it simply maintains the relative spacing between the color components. -//It makes no difference for linear tone "curves". -void film_like_curve(matrix &input, - matrix &output, - LUT &lookup) +// An explanation of the algorithm: +// The algorithm applies the designated tone curve on the highest and the lowest +// values. +// On the middle value, instead of using the tone curve, which would induce hue +// shifts, it simply maintains the relative spacing between the color components. +// It makes no difference for linear tone "curves". +void film_like_curve(matrix &input, matrix &output, LUT &lookup) { - int xsize = input.nc(); - int ysize = input.nr(); - output.set_size(ysize, xsize); + int xsize = input.nc(); + int ysize = input.nr(); + output.set_size(ysize, xsize); #pragma omp parallel shared(lookup, input, output, xsize, ysize) - { + { #pragma omp for schedule(dynamic) nowait - for (int i = 0; i < ysize; i++) - { - for (int j = 0; j < xsize; j = j + 3) - { - unsigned short r = input(i, j ); - unsigned short g = input(i, j+1); - unsigned short b = input(i, j+2); - - if (r >= g) - { - if (g > b) midValueShift (r, g, b, lookup); // Case1: r>= g> b - else if (b > r) midValueShift (b, r, g, lookup); // Case2: b> r>= g - else if (b > g) midValueShift (r, b, g, lookup); // Case3: r>= b> g - else // Case4: r>= g== b - { - //RGBTone fails if the first and last arguments are the same. - //So in this case, since that might happen, don't call it. - r = lookup[ r ]; - g = lookup[ g ]; - b = g; - } - } - else - { - if (r >= b) midValueShift (g, r, b, lookup); // Case5: g> r>= b - else if (b > g) midValueShift (b, g, r, lookup); // Case6: b> g> r - else midValueShift (g, b, r, lookup); // Case7: g>= b> r - } - output(i, j ) = r; - output(i, j+1) = g; - output(i, j+2) = b; + for (int i = 0; i < ysize; i++) { + for (int j = 0; j < xsize; j = j + 3) { + unsigned short r = input(i, j); + unsigned short g = input(i, j + 1); + unsigned short b = input(i, j + 2); + + if (r >= g) { + if (g > b) + midValueShift(r, g, b, lookup);// Case1: r>= g> b + else if (b > r) + midValueShift(b, r, g, lookup);// Case2: b> r>= g + else if (b > g) + midValueShift(r, b, g, lookup);// Case3: r>= b> g + else// Case4: r>= g== b + { + // RGBTone fails if the first and last arguments are the same. + // So in this case, since that might happen, don't call it. + r = lookup[r]; + g = lookup[g]; + b = g; + } + } else { + if (r >= b) + midValueShift(g, r, b, lookup);// Case5: g> r>= b + else if (b > g) + midValueShift(b, g, r, lookup);// Case6: b> g> r + else + midValueShift(g, b, r, lookup);// Case7: g>= b> r } + output(i, j) = r; + output(i, j + 1) = g; + output(i, j + 2) = b; + } } - } + } } -//This is what does the actual computation of the middle value. -//This was called RGBTone in RawTherapee. -//It assumes that r and b are the extreme values, and that they are different. -void midValueShift(unsigned short& hi, unsigned short& mid, unsigned short& lo, - LUT &lookup) +// This is what does the actual computation of the middle value. +// This was called RGBTone in RawTherapee. +// It assumes that r and b are the extreme values, and that they are different. +void midValueShift(unsigned short &hi, unsigned short &mid, unsigned short &lo, LUT &lookup) { - unsigned short oldHi = hi, oldMid = mid, oldLo = lo; - - hi = lookup[ oldHi ]; - lo = lookup[ oldLo ]; - float hi_f = hi; - float lo_f = lo; - float oldHi_f = oldHi; - float oldMid_f = oldMid; - float oldLo_f = oldLo; - mid = lo_f + ((hi_f - lo_f) * (oldMid_f - oldLo_f) / (oldHi_f - oldLo_f)); + unsigned short oldHi = hi, oldMid = mid, oldLo = lo; + + hi = lookup[oldHi]; + lo = lookup[oldLo]; + float hi_f = hi; + float lo_f = lo; + float oldHi_f = oldHi; + float oldMid_f = oldMid; + float oldLo_f = oldLo; + mid = lo_f + ((hi_f - lo_f) * (oldMid_f - oldLo_f) / (oldHi_f - oldLo_f)); } diff --git a/filmulator-gui/core/debug_utils.h b/filmulator-gui/core/debug_utils.h new file mode 100644 index 00000000..34019cd8 --- /dev/null +++ b/filmulator-gui/core/debug_utils.h @@ -0,0 +1,59 @@ +#ifndef DEBUG_UTILS_H +#define DEBUG_UTILS_H + +#include "logging.h" +#include + +#ifdef ENABLE_NAN_TRAPPING +#ifdef __APPLE__ +#include +#if TARGET_OS_MAC +#include +#define BREAK_ON_NAN(v) \ + if (std::isnan(v)) { \ + FILM_ERROR("NaN detected! Breaking..."); \ + __builtin_debugtrap(); \ + } +#else +#define BREAK_ON_NAN(v) \ + if (std::isnan(v)) { \ + FILM_ERROR("NaN detected! Aborting..."); \ + abort(); \ + } +#endif +#elif defined(_WIN32) +#include +#define BREAK_ON_NAN(v) \ + if (std::isnan(v)) { \ + FILM_ERROR("NaN detected! Breaking..."); \ + __debugbreak(); \ + } +#else +#include +#define BREAK_ON_NAN(v) \ + if (std::isnan(v)) { \ + FILM_ERROR("NaN detected! Aborting..."); \ + abort(); \ + } +#endif + +#define SCAN_MATRIX_FOR_NAN(m, name) \ + do { \ + const float *data = static_cast(m); \ + if (data) { \ + size_t total = static_cast((m).nr()) * (m).nc(); \ + for (size_t i = 0; i < total; ++i) { \ + if (std::isnan(data[i])) { \ + FILM_ERROR("NaN found in matrix {} at index {}", name, i); \ + BREAK_ON_NAN(data[i]); \ + break; \ + } \ + } \ + } \ + } while (0) +#else +#define BREAK_ON_NAN(v) ((void)0) +#define SCAN_MATRIX_FOR_NAN(m, name) ((void)0) +#endif + +#endif// DEBUG_UTILS_H diff --git a/filmulator-gui/core/develop.cpp b/filmulator-gui/core/develop.cpp index f92009dc..8cc4b9f6 100644 --- a/filmulator-gui/core/develop.cpp +++ b/filmulator-gui/core/develop.cpp @@ -1,4 +1,4 @@ -/* +/* * This file is part of Filmulator. * * Copyright 2013 Omer Mano and Carlo Vaccari @@ -18,94 +18,84 @@ */ #include "filmSim.hpp" -void develop( matrix &crystalRad, - float crystalGrowthConst, - const matrix &activeCrystalsPerPixel, - matrix &silverSaltDensity, - matrix &develConcentration, - float activeLayerThickness, - float developerConsumptionConst, - float silverSaltConsumptionConst, - float timestep) +void develop(matrix &crystalRad, + float crystalGrowthConst, + const matrix &activeCrystalsPerPixel, + matrix &silverSaltDensity, + matrix &develConcentration, + float activeLayerThickness, + float developerConsumptionConst, + float silverSaltConsumptionConst, + float timestep) { - //Setting up dimensions and boundaries. - int height = develConcentration.nr(); - int width = develConcentration.nc(); - //We still count columns of pixels, because we must process them - // whole, so to ensure this we runthree adjacent elements at a time. - - //Here we pre-compute some repeatedly used values. - float cgc = crystalGrowthConst*timestep; - float dcc = 2.0*developerConsumptionConst / ( activeLayerThickness*3.0 ); - float sscc = silverSaltConsumptionConst*2.0; + // Setting up dimensions and boundaries. + int height = develConcentration.nr(); + int width = develConcentration.nc(); + // We still count columns of pixels, because we must process them + // whole, so to ensure this we runthree adjacent elements at a time. - //These are only used once per loop, so they don't have to be matrices. - float dCrystalRadR; - float dCrystalRadG; - float dCrystalRadB; - float dCrystalVolR; - float dCrystalVolG; - float dCrystalVolB; + // Here we pre-compute some repeatedly used values. + float cgc = crystalGrowthConst * timestep; + float dcc = 2.0 * developerConsumptionConst / (activeLayerThickness * 3.0); + float sscc = silverSaltConsumptionConst * 2.0; - //These are the column indices for red, green, and blue. - int row, col, colr, colg, colb; + // These are only used once per loop, so they don't have to be matrices. + float dCrystalRadR; + float dCrystalRadG; + float dCrystalRadB; + float dCrystalVolR; + float dCrystalVolG; + float dCrystalVolB; -#pragma omp parallel shared( develConcentration, silverSaltDensity,\ - crystalRad, activeCrystalsPerPixel, cgc, dcc, sscc )\ - private( row, col,\ - colr, colg, colb,\ - dCrystalRadR, dCrystalRadG, dCrystalRadB,\ - dCrystalVolR, dCrystalVolG, dCrystalVolB ) - { + // These are the column indices for red, green, and blue. + int row, col, colr, colg, colb; -#pragma omp for schedule( dynamic ) nowait - for ( row = 0; row < height; row++ ) - { - for ( col = 0; col < width; col++ ) - { - colr = col * 3; - colg = colr + 1; - colb = colr + 2; - //This is the rate of thickness accumulating on the crystals. - dCrystalRadR = develConcentration( row, col ) * silverSaltDensity( row, colr ) * cgc; - dCrystalRadG = develConcentration( row, col ) * silverSaltDensity( row, colg ) * cgc; - dCrystalRadB = develConcentration( row, col ) * silverSaltDensity( row, colb ) * cgc; - - //The volume change is proportional to 4*pi*r^2*dr. - //We kinda shuffled around the constants, so ignore the lack of - //the 4 and the pi. - //However, there are varying numbers of crystals, so we also - //multiply by the number of crystals per pixel. - dCrystalVolR = dCrystalRadR * crystalRad( row, colr ) * crystalRad( row, colr ) * - activeCrystalsPerPixel( row, colr ); - dCrystalVolG = dCrystalRadG * crystalRad( row, colg ) * crystalRad( row, colg ) * - activeCrystalsPerPixel( row, colg ); - dCrystalVolB = dCrystalRadB * crystalRad( row, colb ) * crystalRad( row, colb ) * - activeCrystalsPerPixel( row, colb ); - - //Now we apply the new crystal radius. - crystalRad( row, colr ) += dCrystalRadR; - crystalRad( row, colg ) += dCrystalRadG; - crystalRad( row, colb ) += dCrystalRadB; - - //Here is where we consume developer. The 3 layers of film, - //(one per color) share the same developer. - develConcentration( row ,col ) -= dcc * ( dCrystalVolR + - dCrystalVolG + - dCrystalVolB ); +#pragma omp parallel \ + shared(develConcentration, silverSaltDensity, crystalRad, activeCrystalsPerPixel, cgc, dcc, sscc) private( \ + row, col, colr, colg, colb, dCrystalRadR, dCrystalRadG, dCrystalRadB, dCrystalVolR, dCrystalVolG, dCrystalVolB) + { - //Prevent developer concentration from going negative. - develConcentration(row, col) = std::max(develConcentration(row, col), 0.0f); +#pragma omp for schedule(dynamic) nowait + for (row = 0; row < height; row++) { + for (col = 0; col < width; col++) { + colr = col * 3; + colg = colr + 1; + colb = colr + 2; + // This is the rate of thickness accumulating on the crystals. + dCrystalRadR = develConcentration(row, col) * silverSaltDensity(row, colr) * cgc; + dCrystalRadG = develConcentration(row, col) * silverSaltDensity(row, colg) * cgc; + dCrystalRadB = develConcentration(row, col) * silverSaltDensity(row, colb) * cgc; - //Here, silver salts are consumed in proportion to how much - //silver was deposited on the crystals. Unlike the developer, - //each color layer has its own separate amount in this sim. - silverSaltDensity( row, colr ) -= sscc * dCrystalVolR; - silverSaltDensity( row, colg ) -= sscc * dCrystalVolG; - silverSaltDensity( row, colb ) -= sscc * dCrystalVolB; - } - } + // The volume change is proportional to 4*pi*r^2*dr. + // We kinda shuffled around the constants, so ignore the lack of + // the 4 and the pi. + // However, there are varying numbers of crystals, so we also + // multiply by the number of crystals per pixel. + dCrystalVolR = dCrystalRadR * crystalRad(row, colr) * crystalRad(row, colr) * activeCrystalsPerPixel(row, colr); + dCrystalVolG = dCrystalRadG * crystalRad(row, colg) * crystalRad(row, colg) * activeCrystalsPerPixel(row, colg); + dCrystalVolB = dCrystalRadB * crystalRad(row, colb) * crystalRad(row, colb) * activeCrystalsPerPixel(row, colb); + + // Now we apply the new crystal radius. + crystalRad(row, colr) += dCrystalRadR; + crystalRad(row, colg) += dCrystalRadG; + crystalRad(row, colb) += dCrystalRadB; + + // Here is where we consume developer. The 3 layers of film, + //(one per color) share the same developer. + develConcentration(row, col) -= dcc * (dCrystalVolR + dCrystalVolG + dCrystalVolB); + + // Prevent developer concentration from going negative. + develConcentration(row, col) = std::max(develConcentration(row, col), 0.0f); + + // Here, silver salts are consumed in proportion to how much + // silver was deposited on the crystals. Unlike the developer, + // each color layer has its own separate amount in this sim. + silverSaltDensity(row, colr) -= sscc * dCrystalVolR; + silverSaltDensity(row, colg) -= sscc * dCrystalVolG; + silverSaltDensity(row, colb) -= sscc * dCrystalVolB; + } } - return; + } + return; } diff --git a/filmulator-gui/core/diffuse.cpp b/filmulator-gui/core/diffuse.cpp index b57aae76..96f2f3f7 100644 --- a/filmulator-gui/core/diffuse.cpp +++ b/filmulator-gui/core/diffuse.cpp @@ -1,4 +1,4 @@ -/* +/* * This file is part of Filmulator. * * Copyright 2013 Omer Mano and Carlo Vaccari @@ -17,589 +17,470 @@ * along with Filmulator. If not, see */ #include "filmSim.hpp" +#include "logging.h" +#include "math.h" #include -#include #include -#include "math.h" -#define ORDER 7 //This is the number of box blur iterations. +#include +#define ORDER 7// This is the number of box blur iterations. using namespace std; -//In real film, developer diffuses in a 2D gaussian on the film surface. -//Since 2D gaussians are separable, we perform two 1D gaussians in x and y. -//Since approximating a gaussian with a convolution is fairly computationally -//expensive, we approximate it using repeated box blurs. +// In real film, developer diffuses in a 2D gaussian on the film surface. +// Since 2D gaussians are separable, we perform two 1D gaussians in x and y. +// Since approximating a gaussian with a convolution is fairly computationally +// expensive, we approximate it using repeated box blurs. -//Helper function to diffuse in x direction +// Helper function to diffuse in x direction void diffuse_x(matrix &developer_concentration, - int convlength, - int convrad, int pad, int paddedwidth, int order, - float swell_factor); + int convlength, + int convrad, + int pad, + int paddedwidth, + int order, + float swell_factor); void diffuse_y(matrix &developer_concentration, - int convlength, - int convrad, int pad, int paddedlength, int order, - float swell_factor); - -void diffuse(matrix &developer_concentration, - float sigma_const, - float pixels_per_millimeter, - float timestep) + int convlength, + int convrad, + int pad, + int paddedlength, + int order, + float swell_factor); + +void diffuse(matrix &developer_concentration, float sigma_const, float pixels_per_millimeter, float timestep) { - //This is the standard deviation we want for the blur in pixels. - float sigma = sqrt(timestep*pow(sigma_const*pixels_per_millimeter,2)); - int order = ORDER; - - int length = developer_concentration.nr(); - int width = developer_concentration.nc(); - - //Length is the total size of the blur box determined by the desired - //gaussian size and the number of iterations. - int convlength = floor(sqrt(pow(sigma,2)*(12/ORDER)+1)); - if (convlength % 2 == 0) - { - convlength += 1; - } - - //convrad is the radius of the convolution box. It's useful later on. - int convrad = (convlength-1)/2; - - //We will be doing lots of averaging, but holding off on dividing by the - //number of values averaged. The number of values averaged is always - //^ - float swell_factor = 1.0/pow(convlength,order); - - //Here we allocate a matrix sized the same as a row, plus room for padding - //mirrored from the internal content. - int pad = order*convrad; - int paddedwidth = 2*pad + width + 1; - int paddedheight = 2*pad + length + 1; - - diffuse_x(developer_concentration,convlength,convrad,pad,paddedwidth, - order, swell_factor); - diffuse_y(developer_concentration,convlength,convrad,pad, - paddedheight, order, swell_factor); - return; + // This is the standard deviation we want for the blur in pixels. + float sigma = sqrt(timestep * pow(sigma_const * pixels_per_millimeter, 2)); + int order = ORDER; + + int length = developer_concentration.nr(); + int width = developer_concentration.nc(); + + // Length is the total size of the blur box determined by the desired + // gaussian size and the number of iterations. + int convlength = floor(sqrt(pow(sigma, 2) * (12 / ORDER) + 1)); + if (convlength % 2 == 0) { convlength += 1; } + + // convrad is the radius of the convolution box. It's useful later on. + int convrad = (convlength - 1) / 2; + + // We will be doing lots of averaging, but holding off on dividing by the + // number of values averaged. The number of values averaged is always + //^ + float swell_factor = 1.0 / pow(convlength, order); + + // Here we allocate a matrix sized the same as a row, plus room for padding + // mirrored from the internal content. + int pad = order * convrad; + int paddedwidth = 2 * pad + width + 1; + int paddedheight = 2 * pad + length + 1; + + diffuse_x(developer_concentration, convlength, convrad, pad, paddedwidth, order, swell_factor); + diffuse_y(developer_concentration, convlength, convrad, pad, paddedheight, order, swell_factor); + return; } -void diffuse_x(matrix &developer_concentration, int convlength, - int convrad, int pad, int paddedwidth, int order, - float swell_factor) +void diffuse_x(matrix &developer_concentration, + int convlength, + int convrad, + int pad, + int paddedwidth, + int order, + float swell_factor) { - const int length = developer_concentration.nr(); - const int width = developer_concentration.nc(); + const int length = developer_concentration.nr(); + const int width = developer_concentration.nc(); -#pragma omp parallel shared(developer_concentration,convlength,convrad,order,\ - paddedwidth,pad,swell_factor) - { - vector hpadded(paddedwidth); //stores one padded line - vector htemp(paddedwidth); // stores result of box blur +#pragma omp parallel shared(developer_concentration, convlength, convrad, order, paddedwidth, pad, swell_factor) + { + vector hpadded(paddedwidth);// stores one padded line + vector htemp(paddedwidth);// stores result of box blur #pragma omp for schedule(dynamic) nowait - for (int row = 0; row &developer_concentration, int convlength, - int convrad, int pad, int paddedlength, int order, - float swell_factor) +void diffuse_y(matrix &developer_concentration, + int convlength, + int convrad, + int pad, + int paddedlength, + int order, + float swell_factor) { - const int length = developer_concentration.nr(); - const int width = developer_concentration.nc(); + const int length = developer_concentration.nr(); + const int width = developer_concentration.nc(); + +#pragma omp parallel shared(developer_concentration, convlength, convrad, order, paddedlength, pad, swell_factor) + { + constexpr int numcols = 8;// process numcols columns at once for better usage of L1 cpu cache + vector> hpadded(paddedlength);// stores one padded line + vector> htemp(paddedlength);// stores result of box blur +#pragma omp for nowait + for (int col = 0; col < width - (numcols - 1); col += numcols) { + // Mirror the start of padded from the row. + for (int row = 0; row < pad; row++) { + for (int c = 0; c < numcols; ++c) hpadded[row][c] = developer_concentration(pad - row, col + c); + } + // Fill the center of padded with a row. + for (int row = 0; row < length; row++) { + for (int c = 0; c < numcols; ++c) hpadded[row + pad][c] = developer_concentration(row, col + c); + } + // Mirror the row onto the end of padded. + for (int row = 0; row < pad + 1; row++) { + for (int c = 0; c < numcols; ++c) + hpadded[row + pad + length][c] = developer_concentration(length - 2 - row, col + c); + }// This is the running sum used to compute the box blur. + + for (int pass = 0; pass < order; pass++) { + // Perform a box blur, but hold off on the divisions in the + // averaging calculations + + // Start the running sum going at the beginning of the pad. + float running_sum[numcols] = {}; + for (int row = pass * convrad; row < (pass * convrad) + convlength; row++) { + for (int c = 0; c < numcols; ++c) running_sum[c] += hpadded[row][c]; + } -#pragma omp parallel shared(developer_concentration,convlength,convrad,order,\ - paddedlength,pad,swell_factor) - { - constexpr int numcols = 8; // process numcols columns at once for better usage of L1 cpu cache - vector> hpadded(paddedlength); //stores one padded line - vector> htemp(paddedlength); // stores result of box blur - #pragma omp for nowait - for (int col = 0; col < width - (numcols - 1); col += numcols) - { - //Mirror the start of padded from the row. - for (int row = 0; row < pad; row++) - { - for (int c = 0; c < numcols; ++c) - hpadded[row][c] = developer_concentration(pad - row, col + c); - } - //Fill the center of padded with a row. - for (int row = 0; row < length; row++) - { - for (int c = 0; c < numcols; ++c) - hpadded[row + pad][c] = developer_concentration(row, col + c); - } - //Mirror the row onto the end of padded. - for (int row = 0; row < pad + 1; row++) - { - for (int c = 0; c < numcols; ++c) - hpadded[row + pad + length][c] = developer_concentration(length - 2 - row, col + c); - } //This is the running sum used to compute the box blur. - - for (int pass = 0; pass < order; pass++) - { - //Perform a box blur, but hold off on the divisions in the - //averaging calculations - - //Start the running sum going at the beginning of the pad. - float running_sum[numcols] = {}; - for (int row = pass * convrad; row < (pass * convrad) + convlength; row++) - { - for (int c = 0; c < numcols; ++c) - running_sum[c] += hpadded[row][c]; - } - - //Start moving down the row. - for (int row = (pass + 1) * convrad; row < paddedlength - 1 - (pass + 1) * convrad; row++) - { - for (int c = 0; c < numcols; ++c) { - htemp[row][c] = running_sum[c]; - running_sum[c] += hpadded[row + convrad + 1][c] - hpadded[row - convrad][c]; - } - } - - //Copy what was in htemp to hpadded for the next iteration. - //Swap just swaps pointers, so it is O(1) - swap(htemp,hpadded); - } - //Now we're done with the convolution of one row, and we can copy - //it back into developer_concentration. But we should also do the - //divisions that we never did during our previous averaging. - for (int row = 0; row < length; row++) - { - for (int c = 0; c < numcols; ++c) - developer_concentration(row, col + c) = hpadded[row + pad][c] * swell_factor; - } + // Start moving down the row. + for (int row = (pass + 1) * convrad; row < paddedlength - 1 - (pass + 1) * convrad; row++) { + for (int c = 0; c < numcols; ++c) { + htemp[row][c] = running_sum[c]; + running_sum[c] += hpadded[row + convrad + 1][c] - hpadded[row - convrad][c]; + } } + + // Copy what was in htemp to hpadded for the next iteration. + // Swap just swaps pointers, so it is O(1) + swap(htemp, hpadded); + } + // Now we're done with the convolution of one row, and we can copy + // it back into developer_concentration. But we should also do the + // divisions that we never did during our previous averaging. + for (int row = 0; row < length; row++) { + for (int c = 0; c < numcols; ++c) developer_concentration(row, col + c) = hpadded[row + pad][c] * swell_factor; + } + } // remaining columns #pragma omp single - for (int col = width - (width % numcols); col < width; col++) - { - //Mirror the start of padded from the row. - for (int row = 0; row < pad; row++) - { - hpadded[row][0] = developer_concentration(pad - row, col); - } - //Fill the center of padded with a row. - for (int row = 0; row < length; row++) - { - hpadded[row + pad][0] = developer_concentration(row, col); - } - //Mirror the row onto the end of padded. - for (int row = 0; row < pad + 1; row++) - { - hpadded[row + pad + length][0] = developer_concentration(length - 2 - row, col); - } //This is the running sum used to compute the box blur. - - for (int pass = 0; pass < order; pass++) - { - //Perform a box blur, but hold off on the divisions in the - //averaging calculations - - //Start the running sum going at the beginning of the pad. - float running_sum = 0; - for (int row = pass * convrad; row < (pass * convrad) + convlength; row++) - { - running_sum += hpadded[row][0]; - } - - //Start moving down the row. - for (int row = (pass + 1) * convrad; row < paddedlength - 1 - (pass + 1) * convrad; row++) - { - htemp[row][0] = running_sum; - running_sum += hpadded[row + convrad + 1][0] - hpadded[row - convrad][0]; - } - - //Copy what was in htemp to hpadded for the next iteration. - //Swap just swaps pointers, so it is O(1) - swap(htemp,hpadded); - } - //Now we're done with the convolution of one row, and we can copy - //it back into developer_concentration. But we should also do the - //divisions that we never did during our previous averaging. - for (int row = 0; row < length; row++) - { - developer_concentration(row, col) = hpadded[row + pad][0] * swell_factor; - } - } + for (int col = width - (width % numcols); col < width; col++) { + // Mirror the start of padded from the row. + for (int row = 0; row < pad; row++) { hpadded[row][0] = developer_concentration(pad - row, col); } + // Fill the center of padded with a row. + for (int row = 0; row < length; row++) { hpadded[row + pad][0] = developer_concentration(row, col); } + // Mirror the row onto the end of padded. + for (int row = 0; row < pad + 1; row++) { + hpadded[row + pad + length][0] = developer_concentration(length - 2 - row, col); + }// This is the running sum used to compute the box blur. + + for (int pass = 0; pass < order; pass++) { + // Perform a box blur, but hold off on the divisions in the + // averaging calculations + + // Start the running sum going at the beginning of the pad. + float running_sum = 0; + for (int row = pass * convrad; row < (pass * convrad) + convlength; row++) { running_sum += hpadded[row][0]; } + + // Start moving down the row. + for (int row = (pass + 1) * convrad; row < paddedlength - 1 - (pass + 1) * convrad; row++) { + htemp[row][0] = running_sum; + running_sum += hpadded[row + convrad + 1][0] - hpadded[row - convrad][0]; + } + + // Copy what was in htemp to hpadded for the next iteration. + // Swap just swaps pointers, so it is O(1) + swap(htemp, hpadded); + } + // Now we're done with the convolution of one row, and we can copy + // it back into developer_concentration. But we should also do the + // divisions that we never did during our previous averaging. + for (int row = 0; row < length; row++) { + developer_concentration(row, col) = hpadded[row + pad][0] * swell_factor; + } } + } } -//This uses a convolution forward and backward with a particular -// 4-length, 1-dimensional kernel to mimic a gaussian. -//In the first pass, it starts at 0, then goes out 4 standard deviations -// onto 0-clamped padding, then convolves back to the start. -//Naturally this attenuates the edges, so it does the same to all ones, -// and divides the image by that. +// This uses a convolution forward and backward with a particular +// 4-length, 1-dimensional kernel to mimic a gaussian. +// In the first pass, it starts at 0, then goes out 4 standard deviations +// onto 0-clamped padding, then convolves back to the start. +// Naturally this attenuates the edges, so it does the same to all ones, +// and divides the image by that. -//Based on the paper "Recursive implementation of the Gaussian filter" -// in Signal Processing 44 (1995) 139-151 -//Referencing code from here: -//https://github.com/halide/Halide/blob/e23f83b9bde63ed64f4d9a2fbe1ed29b9cfbf2e6/test/generator/gaussian_blur_generator.cpp +// Based on the paper "Recursive implementation of the Gaussian filter" +// in Signal Processing 44 (1995) 139-151 +// Referencing code from here: +// https://github.com/halide/Halide/blob/e23f83b9bde63ed64f4d9a2fbe1ed29b9cfbf2e6/test/generator/gaussian_blur_generator.cpp -//Don't use this for radii > 70!!! +// Don't use this for radii > 70!!! void diffuse_short_convolution(matrix &developer_concentration, - const float sigma_const, - const float pixels_per_millimeter, - const float timestep) + const float sigma_const, + const float pixels_per_millimeter, + const float timestep) { - const int height = developer_concentration.nr(); - const int width = developer_concentration.nc(); - - //Compute the standard deviation of the blur we want, in pixels. - const double sigma = sqrt(timestep*pow(sigma_const*pixels_per_millimeter,2)); - - //We set the padding to be 4 standard deviations so as to catch as much as possible. - const int paddedWidth = width + 4*sigma + 3; - const int paddedHeight = height + 4*sigma + 3; - - double q;//constant for computing coefficients - if (sigma < 2.5) - { - q = 3.97156 - 4.14554*sqrt(1 - 0.26891*sigma); - } - else - { - q = 0.98711*sigma - 0.96330; - } - - double denom = 1.57825 + 2.44413*q + 1.4281*q*q + 0.422205*q*q*q; - - const double coeff1 = (2.44413*q + 2.85619*q*q + 1.26661*q*q*q)/denom; - const double coeff2 = (-1.4281*q*q - 1.26661*q*q*q)/denom; - const double coeff3 = (0.422205*q*q*q)/denom; - const double coeff0 = 1 - (coeff1 + coeff2 + coeff3); - - //We blur ones in order to cancel the edge attenuation. - - //First we do horizontally. - vector attenuationX(paddedWidth); - //Set up the boundary - attenuationX[0] = coeff0; //times 1 - attenuationX[1] = coeff0 + coeff1*attenuationX[0]; - attenuationX[2] = coeff0 + coeff1*attenuationX[1] + coeff2*attenuationX[0]; - //Go over the image width - for (int i = 3; i < width; i++) - { - attenuationX[i] = coeff0 + //times 1 - coeff1 * attenuationX[i-1] + - coeff2 * attenuationX[i-2] + - coeff3 * attenuationX[i-3]; - } - //Fill in the padding (which is all zeros) - for (int i = width; i < paddedWidth; i++) - { - //All zeros, so no coeff0*1 here. - attenuationX[i] = coeff1 * attenuationX[i-1] + - coeff2 * attenuationX[i-2] + - coeff3 * attenuationX[i-3]; - } - //And go back. - for (int i = paddedWidth - 3 - 1; i >= 0; i--) - { - attenuationX[i] = coeff0 * attenuationX[i] + - coeff1 * attenuationX[i+1] + - coeff2 * attenuationX[i+2] + - coeff3 * attenuationX[i+3]; - } + const int height = developer_concentration.nr(); + const int width = developer_concentration.nc(); + + // Compute the standard deviation of the blur we want, in pixels. + const double sigma = sqrt(timestep * pow(sigma_const * pixels_per_millimeter, 2)); + + // We set the padding to be 4 standard deviations so as to catch as much as possible. + const int paddedWidth = width + 4 * sigma + 3; + const int paddedHeight = height + 4 * sigma + 3; + + double q;// constant for computing coefficients + if (sigma < 2.5) { + q = 3.97156 - 4.14554 * sqrt(1 - 0.26891 * sigma); + } else { + q = 0.98711 * sigma - 0.96330; + } + + double denom = 1.57825 + 2.44413 * q + 1.4281 * q * q + 0.422205 * q * q * q; + + const double coeff1 = (2.44413 * q + 2.85619 * q * q + 1.26661 * q * q * q) / denom; + const double coeff2 = (-1.4281 * q * q - 1.26661 * q * q * q) / denom; + const double coeff3 = (0.422205 * q * q * q) / denom; + const double coeff0 = 1 - (coeff1 + coeff2 + coeff3); + + // We blur ones in order to cancel the edge attenuation. + + // First we do horizontally. + vector attenuationX(paddedWidth); + // Set up the boundary + attenuationX[0] = coeff0;// times 1 + attenuationX[1] = coeff0 + coeff1 * attenuationX[0]; + attenuationX[2] = coeff0 + coeff1 * attenuationX[1] + coeff2 * attenuationX[0]; + // Go over the image width + for (int i = 3; i < width; i++) { + attenuationX[i] = coeff0 +// times 1 + coeff1 * attenuationX[i - 1] + coeff2 * attenuationX[i - 2] + coeff3 * attenuationX[i - 3]; + } + // Fill in the padding (which is all zeros) + for (int i = width; i < paddedWidth; i++) { + // All zeros, so no coeff0*1 here. + attenuationX[i] = coeff1 * attenuationX[i - 1] + coeff2 * attenuationX[i - 2] + coeff3 * attenuationX[i - 3]; + } + // And go back. + for (int i = paddedWidth - 3 - 1; i >= 0; i--) { + attenuationX[i] = coeff0 * attenuationX[i] + coeff1 * attenuationX[i + 1] + coeff2 * attenuationX[i + 2] + + coeff3 * attenuationX[i + 3]; + } #pragma omp parallel for - for (int i = 0; i < width; i++) + for (int i = 0; i < width; i++) { + if (attenuationX[i] <= 0) { + FILM_WARN("gonna blow X"); + } else// we can invert this { - if (attenuationX[i] <= 0) - { - std::cout << "gonna blow X" << std::endl; - } - else //we can invert this - { - attenuationX[i] = 1.0/attenuationX[i]; - } - } - - //And now vertically. - vector attenuationY(paddedHeight); - //Set up the boundary - attenuationY[0] = coeff0; //times 1 - attenuationY[1] = coeff0 + coeff1*attenuationY[0]; - attenuationY[2] = coeff0 + coeff1*attenuationY[1] + coeff2*attenuationY[0]; - //Go over the image height - for (int i = 3; i < height; i++) - { - attenuationY[i] = coeff0 + //times 1 - coeff1 * attenuationY[i-1] + - coeff2 * attenuationY[i-2] + - coeff3 * attenuationY[i-3]; - } - //Fill in the padding (which is all zeros) - for (int i = height; i < paddedHeight; i++) - { - //All zeros, so no coeff0*1 here. - attenuationY[i] = coeff1 * attenuationY[i-1] + - coeff2 * attenuationY[i-2] + - coeff3 * attenuationY[i-3]; - } - //And go back. - for (int i = paddedHeight - 3 - 1; i >= 0; i--) - { - attenuationY[i] = coeff0 * attenuationY[i ] + - coeff1 * attenuationY[i+1] + - coeff2 * attenuationY[i+2] + - coeff3 * attenuationY[i+3]; + attenuationX[i] = 1.0 / attenuationX[i]; } + } + + // And now vertically. + vector attenuationY(paddedHeight); + // Set up the boundary + attenuationY[0] = coeff0;// times 1 + attenuationY[1] = coeff0 + coeff1 * attenuationY[0]; + attenuationY[2] = coeff0 + coeff1 * attenuationY[1] + coeff2 * attenuationY[0]; + // Go over the image height + for (int i = 3; i < height; i++) { + attenuationY[i] = coeff0 +// times 1 + coeff1 * attenuationY[i - 1] + coeff2 * attenuationY[i - 2] + coeff3 * attenuationY[i - 3]; + } + // Fill in the padding (which is all zeros) + for (int i = height; i < paddedHeight; i++) { + // All zeros, so no coeff0*1 here. + attenuationY[i] = coeff1 * attenuationY[i - 1] + coeff2 * attenuationY[i - 2] + coeff3 * attenuationY[i - 3]; + } + // And go back. + for (int i = paddedHeight - 3 - 1; i >= 0; i--) { + attenuationY[i] = coeff0 * attenuationY[i] + coeff1 * attenuationY[i + 1] + coeff2 * attenuationY[i + 2] + + coeff3 * attenuationY[i + 3]; + } #pragma omp parallel for - for (int i = 0; i < height; i++) - { - if (attenuationY[i] <= 0) - { - std::cout << "gonna blow Y" << std::endl; - } - else - { - attenuationY[i] = 1.0/attenuationY[i]; - } + for (int i = 0; i < height; i++) { + if (attenuationY[i] <= 0) { + FILM_WARN("gonna blow Y"); + } else { + attenuationY[i] = 1.0 / attenuationY[i]; } + } - //X direction blurring. - //We slice by individual rows. + // X direction blurring. + // We slice by individual rows. #pragma omp parallel shared(developer_concentration, attenuationX) - { - vector devel_concX(paddedWidth); + { + vector devel_concX(paddedWidth); #pragma omp for schedule(dynamic) - for (int row = 0; row < height; row++) - { - //Copy data into the temp. - for (int col = 0; col < width; col++) - { - devel_concX[col] = double(developer_concentration(row,col)); - } - //Set up the boundary - devel_concX[0] = coeff0 * devel_concX[0]; - devel_concX[1] = coeff0 * devel_concX[1] + - coeff1 * devel_concX[0]; - devel_concX[2] = coeff0 * devel_concX[2] + - coeff1 * devel_concX[1] + - coeff2 * devel_concX[0]; - //Iterate over the main part of the image, except for the setup - for (int col = 3; col < width; col++) - { - devel_concX[col] = coeff0 * devel_concX[col ] + - coeff1 * devel_concX[col-1] + - coeff2 * devel_concX[col-2] + - coeff3 * devel_concX[col-3]; - } - //Iterate over the zeroed tail - for (int col = width; col < paddedWidth; col++) - { - devel_concX[col] = coeff1 * devel_concX[col-1] + - coeff2 * devel_concX[col-2] + - coeff3 * devel_concX[col-3]; - } - //And go back - for (int col = paddedWidth - 3 - 1; col >= 0; col--) - { - devel_concX[col] = coeff0 * devel_concX[col ] + - coeff1 * devel_concX[col+1] + - coeff2 * devel_concX[col+2] + - coeff3 * devel_concX[col+3]; - } - //And undo the attenuation, copying back from the temp. + for (int row = 0; row < height; row++) { + // Copy data into the temp. + for (int col = 0; col < width; col++) { devel_concX[col] = double(developer_concentration(row, col)); } + // Set up the boundary + devel_concX[0] = coeff0 * devel_concX[0]; + devel_concX[1] = coeff0 * devel_concX[1] + coeff1 * devel_concX[0]; + devel_concX[2] = coeff0 * devel_concX[2] + coeff1 * devel_concX[1] + coeff2 * devel_concX[0]; + // Iterate over the main part of the image, except for the setup + for (int col = 3; col < width; col++) { + devel_concX[col] = coeff0 * devel_concX[col] + coeff1 * devel_concX[col - 1] + coeff2 * devel_concX[col - 2] + + coeff3 * devel_concX[col - 3]; + } + // Iterate over the zeroed tail + for (int col = width; col < paddedWidth; col++) { + devel_concX[col] = + coeff1 * devel_concX[col - 1] + coeff2 * devel_concX[col - 2] + coeff3 * devel_concX[col - 3]; + } + // And go back + for (int col = paddedWidth - 3 - 1; col >= 0; col--) { + devel_concX[col] = coeff0 * devel_concX[col] + coeff1 * devel_concX[col + 1] + coeff2 * devel_concX[col + 2] + + coeff3 * devel_concX[col + 3]; + } + // And undo the attenuation, copying back from the temp. #pragma omp simd - for (int col = 0; col < width; col++) - { - developer_concentration(row,col) = devel_concX[col]*attenuationX[col];//not going SIMD - } - } + for (int col = 0; col < width; col++) { + developer_concentration(row, col) = devel_concX[col] * attenuationX[col];// not going SIMD + } } + } - //Y direction blurring. We slice into columns a whole number of cache lines wide. - //Each cache line is 8 doubles wide. + // Y direction blurring. We slice into columns a whole number of cache lines wide. + // Each cache line is 8 doubles wide. #pragma omp parallel shared(developer_concentration, attenuationY) - { - matrix devel_concY; - int thickness = 8; //of the slice - devel_concY.set_size(paddedHeight, thickness); - int slices = ceil(width / float(thickness)); + { + matrix devel_concY; + int thickness = 8;// of the slice + devel_concY.set_size(paddedHeight, thickness); + int slices = ceil(width / float(thickness)); #pragma omp for schedule(dynamic) - for (int slice = 0; slice < slices; slice++) - { - int offset = slice*thickness; - int iter = thickness;//number of columns to loop through - if (offset + thickness > width) //If it's the last row, - { - iter = width - offset;//Don't go beyond the bounds - } - - //Copy data into the temp. - if (iter < 8) //we can't SIMD this nicely - { - for (int row = 0; row < height; row++) - { - for (int col = 0; col < iter; col++) - { - devel_concY(row,col) = developer_concentration(row,col+offset); - } - } - } - else //we can simd this - { - for (int row = 0; row < height; row++) - { + for (int slice = 0; slice < slices; slice++) { + int offset = slice * thickness; + int iter = thickness;// number of columns to loop through + if (offset + thickness > width)// If it's the last row, + { + iter = width - offset;// Don't go beyond the bounds + } + + // Copy data into the temp. + if (iter < 8)// we can't SIMD this nicely + { + for (int row = 0; row < height; row++) { + for (int col = 0; col < iter; col++) { devel_concY(row, col) = developer_concentration(row, col + offset); } + } + } else// we can simd this + { + for (int row = 0; row < height; row++) { #pragma omp simd - for (int col = 0; col < 8; col++) - { - devel_concY(row,col) = developer_concentration(row,col+offset);//not going SIMD - } - } - } - - //Set up the boundary. + for (int col = 0; col < 8; col++) { + devel_concY(row, col) = developer_concentration(row, col + offset);// not going SIMD + } + } + } + + // Set up the boundary. #pragma omp simd - for (int col = 0; col < 8; col++) - { - devel_concY(0,col) = coeff0 * devel_concY(0,col); - devel_concY(1,col) = coeff0 * devel_concY(1,col) + - coeff1 * devel_concY(0,col); - devel_concY(2,col) = coeff0 * devel_concY(2,col) + - coeff1 * devel_concY(1,col) + - coeff2 * devel_concY(0,col); - } - //Iterate over the main part of the image, except for the setup. - for (int row = 3; row < height; row++) - { + for (int col = 0; col < 8; col++) { + devel_concY(0, col) = coeff0 * devel_concY(0, col); + devel_concY(1, col) = coeff0 * devel_concY(1, col) + coeff1 * devel_concY(0, col); + devel_concY(2, col) = + coeff0 * devel_concY(2, col) + coeff1 * devel_concY(1, col) + coeff2 * devel_concY(0, col); + } + // Iterate over the main part of the image, except for the setup. + for (int row = 3; row < height; row++) { #pragma omp simd - for (int col = 0; col < 8; col++) - { - devel_concY(row,col) = coeff0 * devel_concY(row ,col) + - coeff1 * devel_concY(row-1,col) + - coeff2 * devel_concY(row-2,col) + - coeff3 * devel_concY(row-3,col); - } - } - //Iterate over the zeroed tail - for (int row = height; row < paddedHeight; row++) - { + for (int col = 0; col < 8; col++) { + devel_concY(row, col) = coeff0 * devel_concY(row, col) + coeff1 * devel_concY(row - 1, col) + + coeff2 * devel_concY(row - 2, col) + coeff3 * devel_concY(row - 3, col); + } + } + // Iterate over the zeroed tail + for (int row = height; row < paddedHeight; row++) { #pragma omp simd - for (int col = 0; col < 8; col++) - { - devel_concY(row,col) = coeff1 * devel_concY(row-1,col) + - coeff2 * devel_concY(row-2,col) + - coeff3 * devel_concY(row-3,col); - } - } - //And go back - for (int row = paddedHeight - 3 - 1; row >= 0; row--) - { + for (int col = 0; col < 8; col++) { + devel_concY(row, col) = coeff1 * devel_concY(row - 1, col) + coeff2 * devel_concY(row - 2, col) + + coeff3 * devel_concY(row - 3, col); + } + } + // And go back + for (int row = paddedHeight - 3 - 1; row >= 0; row--) { #pragma omp simd - for (int col = 0; col < 8; col++) - { - devel_concY(row,col) = coeff0 * devel_concY(row ,col) + - coeff1 * devel_concY(row+1,col) + - coeff2 * devel_concY(row+2,col) + - coeff3 * devel_concY(row+3,col); - } - } - //And undo the attenuation, copying back from the temp. - if (iter < 8) //we can't SIMD this nicely - { - for (int row = 0; row < height; row++) - { - for (int col = 0; col < iter; col++) - { - developer_concentration(row,col+offset) = devel_concY(row,col)*attenuationY[row]; - } - } - } - else - { - for (int row = 0; row < height; row++) - { + for (int col = 0; col < 8; col++) { + devel_concY(row, col) = coeff0 * devel_concY(row, col) + coeff1 * devel_concY(row + 1, col) + + coeff2 * devel_concY(row + 2, col) + coeff3 * devel_concY(row + 3, col); + } + } + // And undo the attenuation, copying back from the temp. + if (iter < 8)// we can't SIMD this nicely + { + for (int row = 0; row < height; row++) { + for (int col = 0; col < iter; col++) { + developer_concentration(row, col + offset) = devel_concY(row, col) * attenuationY[row]; + } + } + } else { + for (int row = 0; row < height; row++) { #pragma omp simd - for (int col = 0; col < 8; col++) - { - developer_concentration(row,col+offset) = devel_concY(row,col)*attenuationY[row];//not going SIMD - } - } - } + for (int col = 0; col < 8; col++) { + developer_concentration(row, col + offset) = devel_concY(row, col) * attenuationY[row];// not going SIMD + } } + } } + } } -//Since the aforementioned infinite impulse response doesn't work nicely with large radii, -//this will downsample it so that the radius ends up at about 30. -//Then, it'll apply the van Vliet IIR filter. -//If the radius was already less than 70, then it won't downsample at all. +// Since the aforementioned infinite impulse response doesn't work nicely with large radii, +// this will downsample it so that the radius ends up at about 30. +// Then, it'll apply the van Vliet IIR filter. +// If the radius was already less than 70, then it won't downsample at all. void diffuse_resize_iir(matrix &developer_concentration, - const float sigma_const, - const float pixels_per_millimeter, - const float timestep) + const float sigma_const, + const float pixels_per_millimeter, + const float timestep) { - //set up test sigma - const double sigma = sqrt(timestep*pow(sigma_const*pixels_per_millimeter,2)); + // set up test sigma + const double sigma = sqrt(timestep * pow(sigma_const * pixels_per_millimeter, 2)); - std::cout << "sigma: " << sigma << "=======================================================" << std::endl; + FILM_INFO("sigma: {} =======================================================", sigma); - //If it's small enough, we're not going to resize at all. - if (sigma < 70) - { - diffuse_short_convolution(developer_concentration, - sigma_const, - pixels_per_millimeter, - timestep); - } - else - { - diffuse(developer_concentration, - sigma_const, - pixels_per_millimeter, - timestep); - } + // If it's small enough, we're not going to resize at all. + if (sigma < 70) { + diffuse_short_convolution(developer_concentration, sigma_const, pixels_per_millimeter, timestep); + } else { + diffuse(developer_concentration, sigma_const, pixels_per_millimeter, timestep); + } } diff --git a/filmulator-gui/core/exposure.cpp b/filmulator-gui/core/exposure.cpp index 6b6453f0..94fe9e72 100644 --- a/filmulator-gui/core/exposure.cpp +++ b/filmulator-gui/core/exposure.cpp @@ -1,4 +1,4 @@ -/* +/* * This file is part of Filmulator. * * Copyright 2013 Omer Mano and Carlo Vaccari @@ -18,53 +18,75 @@ */ #include "filmSim.hpp" -void exposure(matrix &input_image, float crystals_per_pixel, - float rolloff_boundary, float toe_boundary, float highlight_crosstalk) +void exposure(matrix &input_image, + float crystals_per_pixel, + float rolloff_boundary, + float toe_boundary, + float highlight_crosstalk) { - rolloff_boundary = std::max(std::min(rolloff_boundary, 65534.f), 1.f); - toe_boundary = std::max(std::min(toe_boundary, rolloff_boundary/2),0.f);//bound this to lower than half the rolloff boundary - rolloff_boundary = std::min(65535.f, rolloff_boundary - toe_boundary);//we mustn't let rolloff boundary exceed 65535 - const int nrows = input_image.nr(); - const int ncols = input_image.nc()/3; - const float max_crystals = 65535.f - toe_boundary; - const float crystal_headroom = max_crystals - rolloff_boundary; - float crosstalkToe = rolloff_boundary/2; - //Magic number mostly for historical reasons - crystals_per_pixel *= 0.00015387105f; + rolloff_boundary = std::max(std::min(rolloff_boundary, 65534.f), 1.f); + toe_boundary = + std::max(std::min(toe_boundary, rolloff_boundary / 2), 0.f);// bound this to lower than half the rolloff boundary + rolloff_boundary = std::min(65535.f, rolloff_boundary - toe_boundary);// we mustn't let rolloff boundary exceed 65535 + const int nrows = input_image.nr(); + const int ncols = input_image.nc() / 3; + const float max_crystals = 65535.f - toe_boundary; + const float crystal_headroom = max_crystals - rolloff_boundary; + float crosstalkToe = rolloff_boundary / 2; + // Magic number mostly for historical reasons + crystals_per_pixel *= 0.00015387105f; #pragma omp parallel - { - #pragma omp for schedule(dynamic) nowait - for(int row = 0; row < nrows; row++) { - for(int col = 0; col rolloff_boundary ? 65535.f - ((crystal_headroom * crystal_headroom) / (rOutput + crystal_headroom - rolloff_boundary)) : rOutput; - input_image(row,colr) = rOutput * crystals_per_pixel; - float gOutput = max(0.0f, crosstalkG - toe_boundary + (toe_boundary*toe_boundary)/(crosstalkG + toe_boundary+1/65535.0f)); - gOutput = gOutput > rolloff_boundary ? 65535.f - ((crystal_headroom * crystal_headroom) / (gOutput + crystal_headroom - rolloff_boundary)) : gOutput; - input_image(row,colg) = gOutput * crystals_per_pixel; - float bOutput = max(0.0f, crosstalkB - toe_boundary + (toe_boundary*toe_boundary)/(crosstalkB + toe_boundary+1/65535.0f)); - bOutput = bOutput > rolloff_boundary ? 65535.f - ((crystal_headroom * crystal_headroom) / (bOutput + crystal_headroom - rolloff_boundary)) : bOutput; - input_image(row,colb) = bOutput * crystals_per_pixel; - } - } + { +#pragma omp for schedule(dynamic) nowait + for (int row = 0; row < nrows; row++) { + for (int col = 0; col < ncols; col++) { + const int colr = col * 3; + const int colg = colr + 1; + const int colb = colr + 2; + const float inputR = max(0.0f, input_image(row, colr)); + const float inputG = max(0.0f, input_image(row, colg)); + const float inputB = max(0.0f, input_image(row, colb)); + // crosstalk + const float rsurp = max(0.0f, inputR - rolloff_boundary); + const float gsurp = max(0.0f, inputG - rolloff_boundary); + const float bsurp = max(0.0f, inputB - rolloff_boundary); + const float rSurplus = + rsurp - crosstalkToe + (crosstalkToe * crosstalkToe) / (rsurp + crosstalkToe + 1 / 65535.0f); + const float gSurplus = + gsurp - crosstalkToe + (crosstalkToe * crosstalkToe) / (gsurp + crosstalkToe + 1 / 65535.0f); + const float bSurplus = + bsurp - crosstalkToe + (crosstalkToe * crosstalkToe) / (bsurp + crosstalkToe + 1 / 65535.0f); + const float maxSurplus = max(max(max(rSurplus, gSurplus), bSurplus), 0.000001f); + const float sumSurplus = rSurplus + gSurplus + bSurplus; + const float crosstalkR = + inputR + sumSurplus * highlight_crosstalk - (rSurplus * rSurplus / maxSurplus) * highlight_crosstalk; + const float crosstalkG = + inputG + sumSurplus * highlight_crosstalk - (gSurplus * gSurplus / maxSurplus) * highlight_crosstalk; + const float crosstalkB = + inputB + sumSurplus * highlight_crosstalk - (bSurplus * bSurplus / maxSurplus) * highlight_crosstalk; + // rolloff + float rOutput = max( + 0.0f, crosstalkR - toe_boundary + (toe_boundary * toe_boundary) / (crosstalkR + toe_boundary + 1 / 65535.0f)); + rOutput = + rOutput > rolloff_boundary + ? 65535.f - ((crystal_headroom * crystal_headroom) / (rOutput + crystal_headroom - rolloff_boundary)) + : rOutput; + input_image(row, colr) = rOutput * crystals_per_pixel; + float gOutput = max( + 0.0f, crosstalkG - toe_boundary + (toe_boundary * toe_boundary) / (crosstalkG + toe_boundary + 1 / 65535.0f)); + gOutput = + gOutput > rolloff_boundary + ? 65535.f - ((crystal_headroom * crystal_headroom) / (gOutput + crystal_headroom - rolloff_boundary)) + : gOutput; + input_image(row, colg) = gOutput * crystals_per_pixel; + float bOutput = max( + 0.0f, crosstalkB - toe_boundary + (toe_boundary * toe_boundary) / (crosstalkB + toe_boundary + 1 / 65535.0f)); + bOutput = + bOutput > rolloff_boundary + ? 65535.f - ((crystal_headroom * crystal_headroom) / (bOutput + crystal_headroom - rolloff_boundary)) + : bOutput; + input_image(row, colb) = bOutput * crystals_per_pixel; + } } + } } diff --git a/filmulator-gui/core/filmSim.hpp b/filmulator-gui/core/filmSim.hpp index 17012078..6f593870 100644 --- a/filmulator-gui/core/filmSim.hpp +++ b/filmulator-gui/core/filmSim.hpp @@ -33,132 +33,138 @@ #include #include #include -//#include +// #include #include "myLibraw.h" #include -#ifdef DOUT -#define dout cout -#else -#define dout 0 && cout #ifndef NDEBUG #define NDEBUG #endif -#endif -#include "assert.h" //Included later so NDEBUG has an effect - -#ifdef TOUT -#define tout cout -#else -#define tout 0 && cout -#endif +#include "assert.h"//Included later so NDEBUG has an effect -struct filmulateParams { // TODO: adjust variable names. +struct filmulateParams +{// TODO: adjust variable names. float initialDeveloperConcentration; - float reservoirThickness; // once reservoir_size - float activeLayerThickness; // once developer_thickness + float reservoirThickness;// once reservoir_size + float activeLayerThickness;// once developer_thickness float crystalsPerPixel; float initialCrystalRadius; float initialSilverSaltDensity; float developerConsumptionConst; float crystalGrowthConst; float silverSaltConsumptionConst; - float totalDevelTime; // once was int + float totalDevelTime;// once was int int agitateCount; - int developmentSteps; // once was float; development_resolution + int developmentSteps;// once was float; development_resolution float filmArea; float sigmaConst; float layerMixConst; float layerTimeDivisor; float rolloffBoundary; + float highlightCrosstalk; }; -void exposure(matrix &input_image, float crystals_per_pixel, - float rolloff_boundary, float toe_boundary, float highlight_crosstalk); - -//Equalizes the concentration of developer across the reservoir and all pixels. -void agitate( matrix &developerConcentration, float activeLayerThickness, - float &reservoirDeveloperConcentration, float reservoirThickness, - float pixelsPerMillimeter ); - -//This simulates one step of the development reaction. -void develop( matrix &crystalRad, - float crystalGrowthConst, - const matrix &activeCrystalsPerPixel, - matrix &silverSaltDensity, - matrix &develConcentration, - float activeLayerThickness, - float developerConsumptionConst, - float silverSaltConsumptionConst, - float timestep); - -void diffuse(matrix &developer_concentration, - float sigma_const, - float pixels_per_millimeter, - float timestep); +void exposure(matrix &input_image, + float crystals_per_pixel, + float rolloff_boundary, + float toe_boundary, + float highlight_crosstalk); + +void develop(matrix &crystalRad, + float crystalGrowthConst, + const matrix &activeCrystalsPerPixel, + matrix &silverSaltDensity, + matrix &develConcentration, + float activeLayerThickness, + float developerConsumptionConst, + float silverSaltConsumptionConst, + float timestep); + +void diffuse(matrix &developer_concentration, float sigma_const, float pixels_per_millimeter, float timestep); + +void agitate(matrix &developerConcentration, + float activeLayerThickness, + float &reservoirDeveloperConcentration, + float reservoirThickness, + float pixelsPerMillimeter); + void diffuse_short_convolution(matrix &developer_concentration, - const float sigma_const, - const float pixels_per_millimeter, - const float timestep); + const float sigma_const, + const float pixels_per_millimeter, + const float timestep); void diffuse_resize_iir(matrix &developer_concentration, - const float sigma_const, - const float pixels_per_millimeter, - const float timestep); + const float sigma_const, + const float pixels_per_millimeter, + const float timestep); // Reading raws with libraw // TODO: remove // PROBABLY NOT NECESSARY ANYMORE // The code is included in ImagePipeline now. -bool imread(std::string input_image_filename, matrix &returnmatrix, - Exiv2::ExifData &exifData, int highlights, bool caEnabled, - bool lowQuality); +bool imread(std::string input_image_filename, + matrix &returnmatrix, + Exiv2::ExifData &exifData, + int highlights, + bool caEnabled, + bool lowQuality); // Reading tiff files -bool imread_tiff(string input_image_filename, matrix &returnmatrix, - Exiv2::ExifData &exifData); +bool imread_tiff(string input_image_filename, matrix &returnmatrix, Exiv2::ExifData &exifData); // Reading jpeg files -bool imread_jpeg(string input_image_filename, matrix &returnmatrix, - Exiv2::ExifData &exifData); +bool imread_jpeg(string input_image_filename, matrix &returnmatrix, Exiv2::ExifData &exifData); // TODO: remove // PROBABLY NOT NECESSARY ANYMORE // The code is included in ImagePipeline now. -bool imload(std::string filename, matrix &input_image, bool tiff, - bool jpeg_in, Exiv2::ExifData &exifData, int highlights, - bool caEnabled, bool lowQuality); +bool imload(std::string filename, + matrix &input_image, + bool tiff, + bool jpeg_in, + Exiv2::ExifData &exifData, + int highlights, + bool caEnabled, + bool lowQuality); void layer_mix(matrix &developer_concentration, - float active_layer_thickness, - float &reservoir_developer_concentration, - float reservoir_thickness, float layer_mix_const, - float layer_time_divisor, float pixels_per_millimeter, - float timestep); + float active_layer_thickness, + float &reservoir_developer_concentration, + float reservoir_thickness, + float layer_mix_const, + float layer_time_divisor, + float pixels_per_millimeter, + float timestep); bool ppm_read_header(ifstream &input, int &xsize, int &ysize); -bool ppm_read_data(ifstream &input, int xsize, int ysize, - matrix &returnmatrix); +bool ppm_read_data(ifstream &input, int xsize, int ysize, matrix &returnmatrix); -void imwrite(matrix &densityr, matrix &densityg, - matrix &densityb, string outputfilename, bool sixteen_bit); +void imwrite(matrix &densityr, + matrix &densityg, + matrix &densityb, + string outputfilename, + bool sixteen_bit); -bool merge_exps(matrix &input_image, const matrix &temp_image, - float &exposure_weight, float initial_exposure_comp, - float &last_exposure_factor, string filename, - float input_exposure_comp); +bool merge_exps(matrix &input_image, + const matrix &temp_image, + float &exposure_weight, + float initial_exposure_comp, + float &last_exposure_factor, + string filename, + float input_exposure_comp); -string convert_from_raw(char *raw_filename, int i, string tempdir, - int highlights); +string convert_from_raw(char *raw_filename, int i, string tempdir, int highlights); -bool imwrite_tiff(const matrix &output, string outputfilename, - Exiv2::ExifData exifData); +bool imwrite_tiff(const matrix &output, string outputfilename, Exiv2::ExifData exifData); -bool imwrite_jpeg(matrix &output, string outputfilename, - Exiv2::ExifData exifData, int quality, string thumbPath, - bool writeExif = true); +bool imwrite_jpeg(matrix &output, + string outputfilename, + Exiv2::ExifData exifData, + int quality, + string thumbPath, + bool writeExif = true); void cleanExif(Exiv2::ExifData &exifData); @@ -167,8 +173,7 @@ float default_tonecurve(float input); // Applies the effective tonecurve specified by the two control points to the // image. -float shadows_highlights(float input, float shadowsX, float shadowsY, - float highlightsX, float highlightsY); +float shadows_highlights(float input, float shadowsX, float shadowsY, float highlightsX, float highlightsY); // Computes the slope of the cubic polynomial at time t. float slopeFromT(float t, float A, float B, float C); @@ -182,36 +187,43 @@ float yFromT(float t, float E, float F, float G, float H); // Applies the LUT to the extreme values while maintaining the relative position // of the middle value. -void film_like_curve(matrix &input, - matrix &output, - LUT &lookup); +void film_like_curve(matrix &input, matrix &output, LUT &lookup); // Applies the LUT to the first and last values, interpolating the middle value. -void midValueShift(unsigned short &hi, unsigned short &mid, unsigned short &lo, - LUT &lookup); +void midValueShift(unsigned short &hi, unsigned short &mid, unsigned short &lo, LUT &lookup); JSAMPLE dither_round(int full_int); double timeDiff(std::chrono::steady_clock::time_point start); -int read_args(int argc, char *argv[], string &input_configuration, - std::vector &input_filename_list, - std::vector &input_exposure_compensation, int &hdr_count, - bool &hdr, bool &tiff, bool &jpeg_in, bool &set_whitepoint, - float &whitepoint, bool &jpeg_out, bool &tonecurve_out, - int &highlights); +int read_args(int argc, + char *argv[], + string &input_configuration, + std::vector &input_filename_list, + std::vector &input_exposure_compensation, + int &hdr_count, + bool &hdr, + bool &tiff, + bool &jpeg_in, + bool &set_whitepoint, + float &whitepoint, + bool &jpeg_out, + bool &tonecurve_out, + int &highlights); void output_file(matrix &output, - vector input_filename_list, bool jpeg_out, - Exiv2::ExifData exifData); + vector input_filename_list, + bool jpeg_out, + Exiv2::ExifData exifData); -void whitepoint_blackpoint(matrix &input, matrix &output, - float whitepoint, float blackpoint); +void whitepoint_blackpoint(matrix &input, matrix &output, float whitepoint, float blackpoint); // Applies LUTs individually to each color. -void colorCurves(matrix &input, matrix &output, - LUT &lutR, LUT &lutG, - LUT &lutB); +void colorCurves(matrix &input, + matrix &output, + LUT &lutR, + LUT &lutG, + LUT &lutB); void rotate_image(matrix &input, matrix &output, int rotation); @@ -219,46 +231,77 @@ void rotate_image(matrix &input, matrix &output, int rotation); // multipliers. If all the optional arguments are positive, use those instead of // the raw file's // camera WB multipliers. -void optimizeWBMults(std::string inputFilename, float &temperature, float &tint, - const float rMul = -1, const float gMul = -1, - const float bMul = -1); +void optimizeWBMults(std::string inputFilename, + float &temperature, + float &tint, + const float rMul = -1, + const float gMul = -1, + const float bMul = -1); // Applies the desired temperature and tint adjustments to the image. // It also converts raw color into sRGB, and applies exposure comp. -void whiteBalance(matrix &input, matrix &output, - float temperature, float tint, float cam2rgb[3][3], - float rCamMul, float gCamMul, float bCamMul, float rPreMul, - float gPreMul, float bPreMul, float expCompMult = 1.f); +void whiteBalance(matrix &input, + matrix &output, + float temperature, + float tint, + float cam2rgb[3][3], + float rCamMul, + float gCamMul, + float bCamMul, + float rPreMul, + float gPreMul, + float bPreMul, + float expCompMult = 1.f); // This should be applied between demosaic and highlight recovery. // It undoes the camera WB and applies the user's WB, and reports the user WB // multipliers. -void rawWhiteBalance(const matrix &input, matrix &output, - const float temperature, const float tint, - const float xyz2cam[3][3], float rCamMul, float gCamMul, - float bCamMul, float &rUserMul, float &gUserMul, - float &bUserMul); +void rawWhiteBalance(const matrix &input, + matrix &output, + const float temperature, + const float tint, + const float xyz2cam[3][3], + float rCamMul, + float gCamMul, + float bCamMul, + float &rUserMul, + float &gUserMul, + float &bUserMul); // This one only applies temperature and tint and exposure comp // to images that are already sRGB. -void sRGBwhiteBalance(matrix &input, matrix &output, - float temperature, float tint, float cam2rgb[3][3], - float rCamMul, float gCamMul, float bCamMul, - float rPreMul, float gPreMul, float bPreMul, - float expCompMult = 1.f); +void sRGBwhiteBalance(matrix &input, + matrix &output, + float temperature, + float tint, + float cam2rgb[3][3], + float rCamMul, + float gCamMul, + float bCamMul, + float rPreMul, + float gPreMul, + float bPreMul, + float expCompMult = 1.f); void vibrance_saturation(const matrix &input, - matrix &output, float vibrance, - float saturation); + matrix &output, + float vibrance, + float saturation); void monochrome_convert(const matrix &input, - matrix &output, float rmult, - float gmult, float bmult); - -void downscale_and_crop(const matrix &input, matrix &output, - const int inputStartX, const int inputStartY, - const int inputEndX, const int inputEndY, - const int outputXSizeLimit, const int outputYSizeLimit); + matrix &output, + float rmult, + float gmult, + float bmult); + +void downscale_and_crop(const matrix &input, + matrix &output, + const int inputStartX, + const int inputStartY, + const int inputEndX, + const int inputEndY, + const int outputXSizeLimit, + const int outputYSizeLimit); // Converts sRGB with D50 illuminant to XYZ with D50 illuminant. void sRGB_to_XYZ(float r, float g, float b, float &x, float &y, float &z); @@ -307,20 +350,16 @@ void sRGB_to_oklab(matrix &in, matrix &out); void oklab_to_sRGB(matrix &in, matrix &out); // Converts raw to linear sRGB -void raw_to_sRGB(matrix &in, matrix &out, - const float cam2rgb[3][3]); +void raw_to_sRGB(matrix &in, matrix &out, const float cam2rgb[3][3]); // Converts linear sRGB to raw -void sRGB_to_raw(matrix &in, matrix &out, - const float cam2rgb[3][3]); +void sRGB_to_raw(matrix &in, matrix &out, const float cam2rgb[3][3]); // Converts raw to oklab -void raw_to_oklab(matrix &in, matrix &out, - const float cam2rgb[3][3]); +void raw_to_oklab(matrix &in, matrix &out, const float cam2rgb[3][3]); // Converts oklab back to raw -void oklab_to_raw(matrix &in, matrix &out, - const float cam2rgb[3][3]); +void oklab_to_raw(matrix &in, matrix &out, const float cam2rgb[3][3]); // Matrix inverse for 3x3 matrices void inverse(const float in[3][3], float (&out)[3][3]); @@ -328,4 +367,4 @@ void inverse(const float in[3][3], float (&out)[3][3]); // For picking which illuminant profile we want. int daylightScore(const int illuminant); -#endif // FILMSIM_H +#endif// FILMSIM_H diff --git a/filmulator-gui/core/filmulate.cpp b/filmulator-gui/core/filmulate.cpp index 134fc2ce..f887a7fa 100644 --- a/filmulator-gui/core/filmulate.cpp +++ b/filmulator-gui/core/filmulate.cpp @@ -17,200 +17,187 @@ * along with Filmulator. If not, see */ #include "imagePipeline.h" +#include "logging.h" #include -#include +#include // Function------------------------------------------------------------------------- bool ImagePipeline::filmulate(matrix &input_image, - matrix &output_density, - ParameterManager * paramManager, - ImagePipeline * pipeline) + matrix &output_density, + ParameterManager *paramManager, + ImagePipeline *pipeline) { - FilmParams filmParam; - AbortStatus abort; - Valid valid; - std::tie(valid, abort, filmParam) = paramManager->claimFilmParams(); - if(abort == AbortStatus::restart) - { - return true; - } - - //Extract parameters from struct - float initial_developer_concentration = filmParam.initialDeveloperConcentration; - float reservoir_thickness = filmParam.reservoirThickness; - float active_layer_thickness = filmParam.activeLayerThickness; - float crystals_per_pixel = filmParam.crystalsPerPixel; - float initial_crystal_radius = filmParam.initialCrystalRadius; - float initial_silver_salt_density = filmParam.initialSilverSaltDensity; - float developer_consumption_const = filmParam.developerConsumptionConst; - float crystal_growth_const = filmParam.crystalGrowthConst; - float silver_salt_consumption_const = filmParam.silverSaltConsumptionConst; - float total_development_time = filmParam.totalDevelopmentTime; - int agitate_count = filmParam.agitateCount; - int development_steps = filmParam.developmentSteps; - float film_area = filmParam.filmArea; - float sigma_const = filmParam.sigmaConst; - float layer_mix_const = filmParam.layerMixConst; - float layer_time_divisor = filmParam.layerTimeDivisor; - float rolloff_boundary = filmParam.rolloffBoundary; - float toe_boundary = filmParam.toeBoundary; - float highlight_crosstalk = filmParam.highlightCrosstalk; - - //Set up timers - std::chrono::steady_clock::time_point initialize_start, development_start, develop_start, - diffuse_start, agitate_start, layer_mix_start; - double develop_dif = 0, diffuse_dif = 0, agitate_dif = 0, layer_mix_dif= 0; - initialize_start = std::chrono::steady_clock::now(); - - int nrows = (int) input_image.nr(); - int ncols = (int) input_image.nc()/3; - int npix = nrows*ncols; - - //Now we activate some of the crystals on the film. This is literally - //akin to exposing film to light. - matrix active_crystals_per_pixel = input_image; - exposure(active_crystals_per_pixel, - crystals_per_pixel, - rolloff_boundary, - toe_boundary, - highlight_crosstalk); - //We set the crystal radius to a small seed value for each color. - matrix& crystal_radius = output_density; - crystal_radius.set_size(nrows,ncols*3); - crystal_radius = initial_crystal_radius; - - //All layers share developer, so we only make it the original image size. - matrix developer_concentration; - developer_concentration.set_size(nrows,ncols); - developer_concentration = initial_developer_concentration; - - //Each layer gets its own silver salt which will feed crystal growth. - matrix silver_salt_density; - silver_salt_density.set_size(nrows,ncols*3); - silver_salt_density = initial_silver_salt_density; - - //Now, we set up the reservoir. - //Because we don't want the film area to influence the brightness, we - // increase the reservoir size in proportion. -#define FILMSIZE 864;//36x24mm - reservoir_thickness *= film_area/FILMSIZE; - float reservoir_developer_concentration = initial_developer_concentration; - - //This is a value used in diffuse to set the length scale. - float pixels_per_millimeter = sqrt(npix/film_area); - - //Here we do some math for the control logic for the differential - //equation approximation computations. - float timestep = total_development_time/development_steps; - int agitate_period; - if(agitate_count > 0) - { - agitate_period = floor(development_steps/agitate_count); - } - else - { - agitate_period = 3*development_steps; - } - int half_agitate_period = floor(agitate_period/2); - - tout << "Initialization time: " << timeDiff(initialize_start) - << " seconds" << endl; - development_start = std::chrono::steady_clock::now(); - - //Now we begin the main development/diffusion loop, which approximates the - //differential equation of film development. - for(int i = 0; i <= development_steps; i++) - { - //Check for cancellation - abort = paramManager->claimFilmAbort(); - if(abort == AbortStatus::restart) - { - return true; - } - - //Updating for starting the development simulation. Valid is one too high here. - pipeline->updateProgress(Valid::partfilmulation, float(i)/float(development_steps)); - - develop_start = std::chrono::steady_clock::now(); - - //This is where we perform the chemical reaction part. - //The crystals grow. - //The developer in the active layer is consumed. - //So is the silver salt in the film. - // The amount consumed increases as the crystals grow larger. - //Because the developer and silver salts are consumed in bright regions, - // this reduces the rate at which they grow. This gives us global - // contrast reduction. - develop(crystal_radius,crystal_growth_const,active_crystals_per_pixel, - silver_salt_density,developer_concentration, - active_layer_thickness,developer_consumption_const, - silver_salt_consumption_const,timestep); - - develop_dif += timeDiff(develop_start); - diffuse_start = std::chrono::steady_clock::now(); - - //Check for cancellation - abort = paramManager->claimFilmAbort(); - if(abort == AbortStatus::restart) - { - return true; - } - - //Updating for starting the diffusion simulation. Valid is one too high here. - pipeline->updateProgress(Valid::partfilmulation, float(i)/float(development_steps)); - - //Now, we are going to perform the diffusion part. - //Here we mix the layer among itself, which grants us the - // local contrast increases. - diffuse(developer_concentration, - sigma_const, - pixels_per_millimeter, - timestep); -// diffuse_short_convolution(developer_concentration, -// sigma_const, -// pixels_per_millimeter, -// timestep); -// diffuse_resize_iir(developer_concentration, -// sigma_const, -// pixels_per_millimeter, -// timestep); - - diffuse_dif += timeDiff(diffuse_start); - - layer_mix_start = std::chrono::steady_clock::now(); - //This performs mixing between the active layer adjacent to the film - // and the reservoir. - //This keeps the effects from getting too crazy. - layer_mix(developer_concentration, - active_layer_thickness, - reservoir_developer_concentration, - reservoir_thickness, - layer_mix_const, - layer_time_divisor, - pixels_per_millimeter, - timestep); - - layer_mix_dif += timeDiff(layer_mix_start); - - agitate_start = std::chrono::steady_clock::now(); - - //I want agitation to only occur in the middle of development, not - //at the very beginning or the ends. So, I add half the agitate - //period to the current cycle count. - if((i+half_agitate_period) % agitate_period ==0) - agitate(developer_concentration, active_layer_thickness, - reservoir_developer_concentration, reservoir_thickness, - pixels_per_millimeter); - - agitate_dif += timeDiff(agitate_start); - } - - tout << "Development time: " << timeDiff(development_start) << " seconds" - << endl; - tout << "Develop time: " << develop_dif << " seconds" << endl; - tout << "Diffuse time: " << diffuse_dif << " seconds" << endl; - tout << "Layer mix time: " << layer_mix_dif << " seconds" << endl; - tout << "Agitate time: " << agitate_dif << " seconds" << endl; + FilmParams filmParam; + AbortStatus abort; + Valid valid; + std::tie(valid, abort, filmParam) = paramManager->claimFilmParams(); + if (abort == AbortStatus::restart) { return true; } + + // Extract parameters from struct + float initial_developer_concentration = filmParam.initialDeveloperConcentration; + float reservoir_thickness = filmParam.reservoirThickness; + float active_layer_thickness = filmParam.activeLayerThickness; + float crystals_per_pixel = filmParam.crystalsPerPixel; + float initial_crystal_radius = filmParam.initialCrystalRadius; + float initial_silver_salt_density = filmParam.initialSilverSaltDensity; + float developer_consumption_const = filmParam.developerConsumptionConst; + float crystal_growth_const = filmParam.crystalGrowthConst; + float silver_salt_consumption_const = filmParam.silverSaltConsumptionConst; + float total_development_time = filmParam.totalDevelopmentTime; + int agitate_count = filmParam.agitateCount; + int development_steps = filmParam.developmentSteps; + float film_area = filmParam.filmArea; + float sigma_const = filmParam.sigmaConst; + float layer_mix_const = filmParam.layerMixConst; + float layer_time_divisor = filmParam.layerTimeDivisor; + float rolloff_boundary = filmParam.rolloffBoundary; + float toe_boundary = filmParam.toeBoundary; + float highlight_crosstalk = filmParam.highlightCrosstalk; + + // Set up timers + std::chrono::steady_clock::time_point initialize_start, development_start, develop_start, diffuse_start, + agitate_start, layer_mix_start; + double develop_dif = 0, diffuse_dif = 0, agitate_dif = 0, layer_mix_dif = 0; + initialize_start = std::chrono::steady_clock::now(); + + int nrows = (int)input_image.nr(); + int ncols = (int)input_image.nc() / 3; + int npix = nrows * ncols; + + // Now we activate some of the crystals on the film. This is literally + // akin to exposing film to light. + matrix active_crystals_per_pixel = input_image; + exposure(active_crystals_per_pixel, crystals_per_pixel, rolloff_boundary, toe_boundary, highlight_crosstalk); + // We set the crystal radius to a small seed value for each color. + matrix &crystal_radius = output_density; + crystal_radius.set_size(nrows, ncols * 3); + crystal_radius = initial_crystal_radius; + + // All layers share developer, so we only make it the original image size. + matrix developer_concentration; + developer_concentration.set_size(nrows, ncols); + developer_concentration = initial_developer_concentration; + + // Each layer gets its own silver salt which will feed crystal growth. + matrix silver_salt_density; + silver_salt_density.set_size(nrows, ncols * 3); + silver_salt_density = initial_silver_salt_density; + + // Now, we set up the reservoir. + // Because we don't want the film area to influence the brightness, we + // increase the reservoir size in proportion. +#define FILMSIZE 864;// 36x24mm + reservoir_thickness *= film_area / FILMSIZE; + float reservoir_developer_concentration = initial_developer_concentration; + + // This is a value used in diffuse to set the length scale. + float pixels_per_millimeter = sqrt(npix / film_area); + + // Here we do some math for the control logic for the differential + // equation approximation computations. + float timestep = total_development_time / development_steps; + int agitate_period; + if (agitate_count > 0) { + agitate_period = floor(development_steps / agitate_count); + } else { + agitate_period = 3 * development_steps; + } + int half_agitate_period = floor(agitate_period / 2); + + FILM_DEBUG("Initialization time: {} seconds", timeDiff(initialize_start)); + development_start = std::chrono::steady_clock::now(); + + // Now we begin the main development/diffusion loop, which approximates the + // differential equation of film development. + for (int i = 0; i <= development_steps; i++) { + // Check for cancellation + abort = paramManager->claimFilmAbort(); + if (abort == AbortStatus::restart) { return true; } + + // Updating for starting the development simulation. Valid is one too high + // here. + pipeline->updateProgress(Valid::partfilmulation, float(i) / float(development_steps)); + + develop_start = std::chrono::steady_clock::now(); + + // This is where we perform the chemical reaction part. + // The crystals grow. + // The developer in the active layer is consumed. + // So is the silver salt in the film. + // The amount consumed increases as the crystals grow larger. + // Because the developer and silver salts are consumed in bright regions, + // this reduces the rate at which they grow. This gives us global + // contrast reduction. + develop(crystal_radius, + crystal_growth_const, + active_crystals_per_pixel, + silver_salt_density, + developer_concentration, + active_layer_thickness, + developer_consumption_const, + silver_salt_consumption_const, + timestep); + + develop_dif += timeDiff(develop_start); + diffuse_start = std::chrono::steady_clock::now(); + + // Check for cancellation + abort = paramManager->claimFilmAbort(); + if (abort == AbortStatus::restart) { return true; } + + // Updating for starting the diffusion simulation. Valid is one too high + // here. + pipeline->updateProgress(Valid::partfilmulation, float(i) / float(development_steps)); + + // Now, we are going to perform the diffusion part. + // Here we mix the layer among itself, which grants us the + // local contrast increases. + diffuse(developer_concentration, sigma_const, pixels_per_millimeter, timestep); + // diffuse_short_convolution(developer_concentration, + // sigma_const, + // pixels_per_millimeter, + // timestep); + // diffuse_resize_iir(developer_concentration, + // sigma_const, + // pixels_per_millimeter, + // timestep); + + diffuse_dif += timeDiff(diffuse_start); + + layer_mix_start = std::chrono::steady_clock::now(); + // This performs mixing between the active layer adjacent to the film + // and the reservoir. + // This keeps the effects from getting too crazy. + layer_mix(developer_concentration, + active_layer_thickness, + reservoir_developer_concentration, + reservoir_thickness, + layer_mix_const, + layer_time_divisor, + pixels_per_millimeter, + timestep); + + layer_mix_dif += timeDiff(layer_mix_start); + + agitate_start = std::chrono::steady_clock::now(); + + // I want agitation to only occur in the middle of development, not + // at the very beginning or the ends. So, I add half the agitate + // period to the current cycle count. + if ((i + half_agitate_period) % agitate_period == 0) + agitate(developer_concentration, + active_layer_thickness, + reservoir_developer_concentration, + reservoir_thickness, + pixels_per_millimeter); + + agitate_dif += timeDiff(agitate_start); + } + FILM_DEBUG("Development time: {} seconds", timeDiff(development_start)); + FILM_DEBUG("Develop time: {} seconds", develop_dif); + FILM_DEBUG("Diffuse time: {} seconds", diffuse_dif); + FILM_DEBUG("Layer mix time: {} seconds", layer_mix_dif); + FILM_DEBUG("Agitate time: {} seconds", agitate_dif); // Now we compute the density (opacity) of the film. // We assume that overlapping crystals or dye clouds are @@ -220,9 +207,7 @@ bool ImagePipeline::filmulate(matrix &input_image, mult_start = std::chrono::steady_clock::now(); abort = paramManager->claimFilmAbort(); - if (abort == AbortStatus::restart) { - return true; - } + if (abort == AbortStatus::restart) { return true; } const int numRows = crystal_radius.nr(); const int numCols = crystal_radius.nc(); @@ -230,11 +215,10 @@ bool ImagePipeline::filmulate(matrix &input_image, #pragma omp parallel for for (int i = 0; i < numRows; ++i) { for (int j = 0; j < numCols; ++j) { - output_density(i, j) = crystal_radius(i, j) * crystal_radius(i, j) * - active_crystals_per_pixel(i, j); + output_density(i, j) = crystal_radius(i, j) * crystal_radius(i, j) * active_crystals_per_pixel(i, j); } } - tout << "Output density time: " << timeDiff(mult_start) << endl; + FILM_DEBUG("Output density time: {}", timeDiff(mult_start)); #ifdef DOUT debug_out.close(); #endif diff --git a/filmulator-gui/core/filmulator.cpp b/filmulator-gui/core/filmulator.cpp index 5195e35d..1872f8a2 100644 --- a/filmulator-gui/core/filmulator.cpp +++ b/filmulator-gui/core/filmulator.cpp @@ -1,4 +1,4 @@ -/* +/* * This file is part of Filmulator. * * Copyright 2013 Omer Mano and Carlo Vaccari @@ -17,141 +17,135 @@ * along with Filmulator. If not, see */ #include "filmSim.hpp" +#include "logging.h" #include "string.h" #include -int main(int argc, char* argv[]) +int main(int argc, char *argv[]) { - //Setup for timing - struct timeval total_start; - gettimeofday(&total_start,NULL); - - string input_configuration; - std::vector input_filename_list; - std::vector input_exposure_compensation; - int hdr_count = 0; - bool hdr = false; - bool tiff = false; - bool jpeg_in = false; - bool slide_reversal = false; - bool set_whitepoint = false; - float whitepoint; - bool jpeg_out = false; - bool tonecurve_out = false; - int highlights = 0; //dcraw highlight recovery parameter - - - if(read_args(argc, - argv, - input_configuration, - input_filename_list, - input_exposure_compensation, - hdr_count, - hdr, - tiff, - jpeg_in, - set_whitepoint, - whitepoint, - jpeg_out, - tonecurve_out, - highlights)) - { - return 1; - } - - float initial_developer_concentration; - float reservoir_size; - float developer_thickness; - float crystals_per_pixel; - float initial_crystal_radius; - float initial_silver_salt_density; - float developer_consumption_const; - float crystal_growth_const; - float silver_salt_consumption_const; - int total_development_time; - int agitate_count; - float development_resolution; - float film_area; - float sigma_const; - float layer_mix_const; - float layer_time_divisor; - float std_cutoff; - string tempdir; - int rolloff_boundary; - - initialize(input_configuration, - initial_developer_concentration, - reservoir_size, - developer_thickness, - crystals_per_pixel, - initial_crystal_radius, - initial_silver_salt_density, - developer_consumption_const, - crystal_growth_const, - silver_salt_consumption_const, - total_development_time, - agitate_count, - development_resolution, - film_area, - sigma_const, - layer_mix_const, - layer_time_divisor, - std_cutoff, - rolloff_boundary); - - //Set up for timing - struct timeval read_start; - gettimeofday(&read_start,NULL); - - //Load image - matrix input_image; - Exiv2::ExifData exifData; - if(imload(input_filename_list, input_exposure_compensation, - input_image, tiff, jpeg_in, exifData, highlights)) - { - cout << "Error loading images" << endl; - return 1; - } - exifData["Exif.Image.ProcessingSoftware"] = "Filmulator"; - - tout << "Read time: " << time_diff(read_start) << " seconds" << endl; - cout << "Image loaded." << endl; - - matrix output_density = filmulate(input_image, - initial_developer_concentration, - reservoir_size, - developer_thickness, - crystals_per_pixel, - initial_crystal_radius, - initial_silver_salt_density, - developer_consumption_const, - crystal_growth_const, - silver_salt_consumption_const, - total_development_time, - agitate_count, - development_resolution, - film_area, - sigma_const, - layer_mix_const, - layer_time_divisor, - rolloff_boundary); - - - //Postprocessing: normallize and apply a tone curve - int nrows = output_density.nr(); - int ncols = output_density.nc()/3; - matrix output_r(nrows,ncols); - matrix output_g(nrows,ncols); - matrix output_b(nrows,ncols); - - postprocess(output_density, set_whitepoint, whitepoint, tonecurve_out, - std_cutoff, output_r,output_g,output_b); - - //Output the file - output_file(output_r, output_g, output_b, input_filename_list, jpeg_out, - exifData); - - tout << "Total time: " << time_diff(total_start) << " seconds" << endl; - - return 0; + // Setup for timing + auto total_start = std::chrono::steady_clock::now(); + + string input_configuration; + std::vector input_filename_list; + std::vector input_exposure_compensation; + int hdr_count = 0; + bool hdr = false; + bool tiff = false; + bool jpeg_in = false; + bool slide_reversal = false; + bool set_whitepoint = false; + float whitepoint; + bool jpeg_out = false; + bool tonecurve_out = false; + int highlights = 0;// dcraw highlight recovery parameter + + + if (read_args(argc, + argv, + input_configuration, + input_filename_list, + input_exposure_compensation, + hdr_count, + hdr, + tiff, + jpeg_in, + set_whitepoint, + whitepoint, + jpeg_out, + tonecurve_out, + highlights)) { + return 1; + } + + float initial_developer_concentration; + float reservoir_size; + float developer_thickness; + float crystals_per_pixel; + float initial_crystal_radius; + float initial_silver_salt_density; + float developer_consumption_const; + float crystal_growth_const; + float silver_salt_consumption_const; + int total_development_time; + int agitate_count; + float development_resolution; + float film_area; + float sigma_const; + float layer_mix_const; + float layer_time_divisor; + float std_cutoff; + string tempdir; + int rolloff_boundary; + + initialize(input_configuration, + initial_developer_concentration, + reservoir_size, + developer_thickness, + crystals_per_pixel, + initial_crystal_radius, + initial_silver_salt_density, + developer_consumption_const, + crystal_growth_const, + silver_salt_consumption_const, + total_development_time, + agitate_count, + development_resolution, + film_area, + sigma_const, + layer_mix_const, + layer_time_divisor, + std_cutoff, + rolloff_boundary); + + // Set up for timing + auto read_start = std::chrono::steady_clock::now(); + + // Load image + matrix input_image; + Exiv2::ExifData exifData; + if (imload(input_filename_list, input_exposure_compensation, input_image, tiff, jpeg_in, exifData, highlights)) { + FILM_ERROR("Error loading images"); + return 1; + } + exifData["Exif.Image.ProcessingSoftware"] = "Filmulator"; + + FILM_DEBUG("Read time: {} seconds", timeDiff(read_start)); + FILM_INFO("Image loaded."); + + matrix output_density = filmulate(input_image, + initial_developer_concentration, + reservoir_size, + developer_thickness, + crystals_per_pixel, + initial_crystal_radius, + initial_silver_salt_density, + developer_consumption_const, + crystal_growth_const, + silver_salt_consumption_const, + total_development_time, + agitate_count, + development_resolution, + film_area, + sigma_const, + layer_mix_const, + layer_time_divisor, + rolloff_boundary); + + + // Postprocessing: normallize and apply a tone curve + int nrows = output_density.nr(); + int ncols = output_density.nc() / 3; + matrix output_r(nrows, ncols); + matrix output_g(nrows, ncols); + matrix output_b(nrows, ncols); + + postprocess(output_density, set_whitepoint, whitepoint, tonecurve_out, std_cutoff, output_r, output_g, output_b); + + // Output the file + output_file(output_r, output_g, output_b, input_filename_list, jpeg_out, exifData); + + FILM_DEBUG("Total time: {} seconds", timeDiff(total_start)); + + return 0; } diff --git a/filmulator-gui/core/imagePipeline.cpp b/filmulator-gui/core/imagePipeline.cpp index c501b609..ef60bdf1 100644 --- a/filmulator-gui/core/imagePipeline.cpp +++ b/filmulator-gui/core/imagePipeline.cpp @@ -2,57 +2,59 @@ #include "../database/camconst.h" #include "../database/exifFunctions.h" #include "filmSim.hpp" +#include "logging.h" #include "nlmeans/nlmeans.hpp" #include "rawtherapee/rt_routines.h" #include #include -ImagePipeline::ImagePipeline(Cache cacheIn, Histo histoIn, - QuickQuality qualityIn) { +#include "debug_utils.h" +#include "pipeline/ChromaNRStage.h" +#include "pipeline/DemosaicStage.h" +#include "pipeline/ImpulseNRStage.h" +#include "pipeline/LoadStage.h" +#include "pipeline/NlmeansNRStage.h" +#include "pipeline/PostDemosaicStage.h" +#include "pipeline/PrefilmulationStage.h" + +ImagePipeline::ImagePipeline(Cache cacheIn, Histo histoIn, QuickQuality qualityIn) +{ cache = cacheIn; histo = histoIn; quality = qualityIn; valid = Valid::none; filename = ""; - //initialize lensfun db + // initialize lensfun db QString dirstr = QStandardPaths::writableLocation(QStandardPaths::GenericDataLocation); dirstr.append("/filmulator/version_1/"); - //cout << "ImagePipeline lensfun dirstring: " << dirstr.toStdString() << endl; + // cout << "ImagePipeline lensfun dirstring: " << dirstr.toStdString() << endl; QDir dir(dirstr); QStringList filters; filters << "*.xml"; QFileInfoList fileList = dir.entryInfoList(filters, QDir::Files | QDir::NoDotAndDotDot); - //cout << "ImagePipeline initializing lensfun db" << endl; + // cout << "ImagePipeline initializing lensfun db" << endl; ldb = lf_db_new(); - if (!ldb) - { - cout << "ImagePipeline lensfun failed to create database!" << endl; - } + if (!ldb) { cout << "ImagePipeline lensfun failed to create database!" << endl; } foreach (const QFileInfo &fileInfo, fileList) { - const QString filename = fileInfo.absoluteFilePath(); - const std::string stdstring = filename.toStdString(); - //cout << "ImagePipeline lensfun loading file " << stdstring << endl; - lfError loadError = ldb->Load(stdstring.c_str()); - if (loadError == LF_WRONG_FORMAT) - { - cout << "ImagePipeline lensfun loading file " << stdstring << endl; - cout << "ImagePipeline lensfun database wrong format!" << endl; - } - else if (loadError == LF_NO_DATABASE) - { - cout << "ImagePipeline lensfun loading file " << stdstring << endl; - cout << "ImagePipeline lensfun no database found!" << endl; - } - else if (loadError == LF_NO_ERROR) - { - //cout << "ImagePipeline lensfun database loaded" << endl; - } else { - cout << "ImagePipeline lensfun loading file " << stdstring << endl; - cout << "ImagePipeline lensfun what happened? " << loadError << endl; - } + const QString filename = fileInfo.absoluteFilePath(); + const std::string stdstring = filename.toStdString(); + // cout << "ImagePipeline lensfun loading file " << stdstring << endl; + lfError loadError = ldb->Load(stdstring.c_str()); + if (loadError == LF_WRONG_FORMAT) { + cout << "ImagePipeline lensfun loading file " << stdstring << endl; + cout << "ImagePipeline lensfun database wrong format!" << endl; + } else if (loadError == LF_NO_DATABASE) { + cout << "ImagePipeline lensfun loading file " << stdstring << endl; + cout << "ImagePipeline lensfun no database found!" << endl; + } else if (loadError == LF_NO_ERROR) { + // cout << "ImagePipeline lensfun database loaded" << endl; + } else { + cout << "ImagePipeline lensfun loading file " << stdstring << endl; + cout << "ImagePipeline lensfun what happened? " << loadError << endl; + } } completionTimes.resize(Valid::count); @@ -67,18 +69,27 @@ ImagePipeline::ImagePipeline(Cache cacheIn, Histo histoIn, completionTimes[Valid::filmulation] = 50; completionTimes[Valid::blackwhite] = 10; completionTimes[Valid::colorcurve] = 10; - // completionTimes[Valid::filmlikecurve] = 10; + rCamMul = gCamMul = bCamMul = 1.0f; + rPreMul = gPreMul = bPreMul = 1.0f; + rUserMul = gUserMul = bUserMul = 1.0f; + maxValue = 65535.0f; + for (int i = 0; i < 3; ++i) colorMaxValue[i] = 65535.0f; + isSraw = isNikonSraw = isMonochrome = isCR3 = false; + resolution = 0; + progress = 0; + cropHeight = cropAspect = cropHoffset = cropVoffset = 0; + rotation = 0; } -ImagePipeline::~ImagePipeline() { - if (ldb != NULL) { - lf_db_destroy(ldb); - } +ImagePipeline::~ImagePipeline() +{ + if (ldb != NULL) { lf_db_destroy(ldb); } } // int ImagePipeline::libraw_callback(void *data, LibRaw_progress p, int // iteration, int expected) but we only need data. -int ImagePipeline::libraw_callback(void *data, LibRaw_progress, int, int) { +int ImagePipeline::libraw_callback(void *data, LibRaw_progress, int, int) +{ AbortStatus abort; // Recover the param_manager from the data @@ -86,18 +97,18 @@ int ImagePipeline::libraw_callback(void *data, LibRaw_progress, int, int) { // See whether to abort or not. abort = pManager->claimDemosaicAbort(); if (abort == AbortStatus::restart) { - return 1; // cancel processing + return 1;// cancel processing } else { return 0; } } -matrix &ImagePipeline::processImage( - ParameterManager *paramManager, Interface *interface_in, - Exiv2::ExifData &exifOutput, - const QString fileHash, // make this empty string if you don't want to mess - // around with validity - ImagePipeline *stealVictim) // defaults to nullptr +matrix &ImagePipeline::processImage(ParameterManager *paramManager, + Interface *interface_in, + Exiv2::ExifData &exifOutput, + const QString fileHash,// make this empty string if you don't want to mess + // around with validity + ImagePipeline *stealVictim)// defaults to nullptr { // Say that we've started processing to prevent cache status from changing.. hasStartedProcessing = true; @@ -113,12 +124,9 @@ matrix &ImagePipeline::processImage( QString paramIndex = paramManager->getImageIndex(); paramIndex.truncate(32); if (fileHash != paramIndex) { - cout << "processImage shuffle mismatch: Requested Index: " - << fileHash.toStdString() << endl; - cout << "processImage shuffle mismatch: Parameter Index: " - << paramIndex.toStdString() << endl; - cout << "processImage shuffle mismatch: full pipeline?: " - << (quality == HighQuality) << endl; + FILM_WARN("processImage shuffle mismatch: Requested Index: {}", fileHash.toStdString()); + FILM_WARN("processImage shuffle mismatch: Parameter Index: {}", paramIndex.toStdString()); + FILM_WARN("processImage shuffle mismatch: full pipeline?: {}", (quality == HighQuality)); valid = none; } fileID = paramIndex; @@ -130,15 +138,13 @@ matrix &ImagePipeline::processImage( valid = paramManager->getValid(); if (NoCache == cache || true == cacheEmpty) { - valid = none; // we need to start fresh if nothing is going to be cached. + valid = none;// we need to start fresh if nothing is going to be cached. } // If we are a high-res pipeline that's going to steal data, skip to // filmulation if (stealData) { - if (stealVictim == nullptr) { - cout << "stealVictim should not be null!" << endl; - } + if (stealVictim == nullptr) { FILM_ERROR("stealVictim should not be null!"); } valid = max(valid, prefilmulation); paramManager->setValid(valid); } @@ -149,1454 +155,395 @@ matrix &ImagePipeline::processImage( // if something has been processed before, and we think it's valid // it had better be the same filename. if (paramManager->getFullFilename() != filename.toStdString()) { - cout << "processImage paramManager filename doesn't match pipeline " - "filename" - << endl; - cout << "processImage paramManager filename: " - << paramManager->getFullFilename() << endl; - cout << "processImage pipeline filename: " << filename.toStdString() - << endl; - cout << "processImage setting validity to none due to filename" << endl; + FILM_WARN("processImage paramManager filename doesn't match pipeline filename"); + FILM_WARN("processImage paramManager filename: {}", paramManager->getFullFilename()); + FILM_WARN("processImage pipeline filename: {}", filename.toStdString()); + FILM_WARN("processImage setting validity to none due to filename"); valid = none; } } - cout << "ImagePipeline::processImage valid: " << valid << endl; + FILM_INFO("ImagePipeline::processImage valid: {}", (int)valid); updateProgress(valid, 0.0f); + PipelineContext context; + context.interface = interface_in; + context.paramManager = paramManager; + context.cache = cache; + context.histo = histo; + context.quality = quality; + context.resolution = resolution; + switch (valid) { case partload: [[fallthrough]]; - case none: // Load image into buffer + case none:// Load image into buffer { LoadParams loadParam; AbortStatus abort; - // See whether to abort or not, while grabbing the latest parameters. std::tie(valid, abort, loadParam) = paramManager->claimLoadParams(); - if (abort == AbortStatus::restart) { - cout << "ImagePipeline::processImage: aborted at the start" << endl; - return emptyMatrix(); - } - - filename = QString::fromStdString(loadParam.fullFilename); - - isCR3 = false; - - isCR3 = QString::fromStdString(loadParam.fullFilename) - .endsWith(".cr3", Qt::CaseInsensitive); - const bool isDNG = QString::fromStdString(loadParam.fullFilename) - .endsWith(".dng", Qt::CaseInsensitive); - if (isCR3) { - cout << "processImage this is a CR3!" << endl; - } + if (abort == AbortStatus::restart) return emptyMatrix(); - if (!loadParam.tiffIn && !loadParam.jpegIn) { - std::unique_ptr libraw = unique_ptr(new MyLibRaw()); - - // Open the file. - int libraw_error; -#if (defined(_WIN32) || defined(__WIN32__)) - const QString tempFilename = - QString::fromStdString(loadParam.fullFilename); - std::wstring wstr = tempFilename.toStdWString(); - libraw_error = libraw->open_file(wstr.c_str()); -#else - const char *cstr = loadParam.fullFilename.c_str(); - libraw_error = libraw->open_file(cstr); -#endif - if (libraw_error) { - cout << "processImage: Could not read input file!" << endl; - cout << "libraw error text: " << libraw_strerror(libraw_error) << endl; - return emptyMatrix(); - } + LoadStage stage; + auto result = stage.process(loadParam.fullFilename, loadParam, context); - // Make abbreviations for brevity in accessing data. -#define RSIZE libraw->imgdata.sizes -#define PARAM libraw->imgdata.params -#define IMAGE libraw->imgdata.image -#define RAW libraw->imgdata.rawdata.raw_image -#define RAW3 libraw->imgdata.rawdata.color3_image -#define RAW4 libraw->imgdata.rawdata.color4_image -#define RAWF libraw->imgdata.rawdata.float_image -#define RAWF3 libraw->imgdata.rawdata.float3_image -#define RAWF4 libraw->imgdata.rawdata.float4_image -#define IDATA libraw->imgdata.idata -#define LENS libraw->imgdata.lens -#define MAKER libraw->imgdata.lens.makernotes -#define OTHER libraw->imgdata.other -#define SIZES libraw->imgdata.sizes -#define OPTIONS libraw->imgdata.rawparams.options - -#ifndef WIN32 - if (libraw->is_floating_point()) { - // tell libraw to not convert to int when unpacking. - OPTIONS = OPTIONS & ~LIBRAW_RAWOPTIONS_CONVERTFLOAT_TO_INT; - } -#endif // WIN32 - // This makes IMAGE contains the sensel value and 3 blank values at every - // location. - libraw_error = libraw->unpack(); - if (libraw_error) { - cout << "processImage: Could not read input file, or was canceled" - << endl; - cout << "libraw error text: " << libraw_strerror(libraw_error) << endl; - return emptyMatrix(); - } - - bool needs_phase_one_free = false; - libraw_error = libraw->phaseone_fix(needs_phase_one_free); - if (libraw_error) { - cout << "mylibraw phaseone_fix failed?" << endl; - cout << "mylibraw error text: " << libraw_strerror(libraw_error) << endl; - return emptyMatrix(); - } + if (!result) return emptyMatrix(); - bool isFloat = libraw->have_fpdata(); + RawImage &output = *result; - // get dimensions - raw_width = RSIZE.width; - raw_height = RSIZE.height; - cout << "raw width: " << raw_width << endl; - cout << "raw height: " << raw_height << endl; - - int topmargin = RSIZE.top_margin; - int leftmargin = RSIZE.left_margin; - int full_width = RSIZE.raw_width; - // int full_height = RSIZE.raw_height; - - // get color matrix - cout << "processImage filename (matrix): " << loadParam.fullFilename - << endl; - for (int i = 0; i < 3; i++) { - cout << "processImage camToRGB matrix: "; - for (int j = 0; j < 3; j++) { - camToRGB[i][j] = libraw->imgdata.color.rgb_cam[i][j]; - cout << camToRGB[i][j] << " "; - } - cout << endl; - } - if (!isDNG) { - for (int i = 0; i < 3; i++) { - cout << "processImage xyzToCam matrix: "; - for (int j = 0; j < 3; j++) { - xyzToCam[i][j] = libraw->imgdata.color.cam_xyz[i][j]; - cout << xyzToCam[i][j] << " "; - } - cout << endl; - } - } else { // For Sigma fp and fp L cameras LibRaw doesn't report cam_xyz - cout << "processImage dng color matrix illuminant: " - << libraw->imgdata.color.dng_color[0].illuminant << endl; - cout << "processImage dng color matrix illuminant: " - << libraw->imgdata.color.dng_color[1].illuminant << endl; - int dngProfile = 1; - if (daylightScore(libraw->imgdata.color.dng_color[0].illuminant) < - daylightScore(libraw->imgdata.color.dng_color[1].illuminant)) { - dngProfile = 0; - } - cout << "processImage Using dng color matrix number " << dngProfile - << endl; - for (int i = 0; i < 3; i++) { - cout << "processImage xyzToCam matrix: "; - for (int j = 0; j < 3; j++) { - xyzToCam[i][j] = - libraw->imgdata.color.dng_color[dngProfile].colormatrix[i][j]; - cout << xyzToCam[i][j] << " "; - } - cout << endl; - } - } - // LibRaw doesn't give a cam_xyz matrix from the Sigma fp's dng - // We must reconstruct cam_xyz from rgb_cam and the srgb-to-xyz d65 matrix - for (int i = 0; i < 3; i++) { - // cout << "camToRGB4: "; - for (int j = 0; j < 4; j++) { - camToRGB4[i][j] = libraw->imgdata.color.rgb_cam[i][j]; - if (i == j) { - camToRGB4[i][j] = 1; - } else { - camToRGB4[i][j] = 0; - } - if (j == 3) { - camToRGB4[i][j] = camToRGB4[i][1]; - } - // cout << camToRGB4[i][j] << " "; - } - // cout << endl; - } - rCamMul = libraw->imgdata.color.cam_mul[0]; - gCamMul = libraw->imgdata.color.cam_mul[1]; - bCamMul = libraw->imgdata.color.cam_mul[2]; - float minMult = min(min(rCamMul, gCamMul), bCamMul); - rCamMul /= minMult; - gCamMul /= minMult; - bCamMul /= minMult; - rPreMul = libraw->imgdata.color.pre_mul[0]; - gPreMul = libraw->imgdata.color.pre_mul[1]; - bPreMul = libraw->imgdata.color.pre_mul[2]; - minMult = min(min(rPreMul, gPreMul), bPreMul); - rPreMul /= minMult; - gPreMul /= minMult; - bPreMul /= minMult; - - // get black subtraction values - // for everything - float blackpoint = libraw->imgdata.color.black; - // some cameras have individual color channel subtraction. - // this seems to be an offset from the overall blackpoint - float rBlack = libraw->imgdata.color.cblack[0]; - float gBlack = libraw->imgdata.color.cblack[1]; - float bBlack = libraw->imgdata.color.cblack[2]; - float g2Black = libraw->imgdata.color.cblack[3]; - float maxChanBlack = max(rBlack, max(gBlack, max(bBlack, g2Black))); - // Still others have a matrix to subtract. - int blackRow = int(libraw->imgdata.color.cblack[4]); - int blackCol = int(libraw->imgdata.color.cblack[5]); - - cout << "BLACKPOINT: "; - cout << blackpoint << endl; - cout << "color channel blackpoints" << endl; - cout << "R blackpoint: " << rBlack << endl; - cout << "G blackpoint: " << gBlack << endl; - cout << "B blackpoint: " << bBlack << endl; - cout << "G2 blackpoint: " << g2Black << endl; - cout << "block-based blackpoint dimensions:" << endl; - cout << "blackpoint dim 1: " << libraw->imgdata.color.cblack[4] << endl; - cout << "blackpoint dim 2: " << libraw->imgdata.color.cblack[5] << endl; - double sumBlockBlackpoint = 0; - int count = 0; - if (blackRow > 0 && blackCol > 0) { - cout << "block-based blackpoint: " << endl; - for (int i = 0; i < blackRow; i++) { - for (int j = 0; j < blackCol; j++) { - sumBlockBlackpoint += - libraw->imgdata.color.cblack[6 + i * blackCol + j]; - count++; - cout << libraw->imgdata.color.cblack[6 + i * blackCol + j] << " "; - } - cout << endl; - } - } - double meanBlockBlackpoint = 0; - if (count > 0) { - meanBlockBlackpoint = sumBlockBlackpoint / count; - } - cout << "Mean of block-based blackpoint: " << meanBlockBlackpoint << endl; - - // get white saturation values - cout << "WHITE SATURATION ===================================" << endl; - cout << "data_maximum: " << libraw->imgdata.color.data_maximum << endl; - cout << "maximum: " << libraw->imgdata.color.maximum << endl; - - // Calculate the white point based on the camera settings. - // This needs the black point subtracted, and a fudge factor to ensure - // clipping is hard and fast. - double camconstWhite[4]; - - // Some cameras have a black offset, as well, even if the black level is - // already specified. - double camconstBlack[4]; - - QString makeModel = IDATA.make; - makeModel.append(" "); - makeModel.append(IDATA.model); - bool camconstSuccess = - CAMCONST_READ_OK == camconst_read(makeModel, OTHER.iso_speed, - OTHER.aperture, camconstWhite, - camconstBlack); - - cout << "is the file dng?: " << isDNG << endl; - - // we only process with 3 channels later, so we merge the two green - // channels here... - camconstWhite[1] = min(camconstWhite[1], camconstWhite[3]); - - double camconstWhiteMax = - max(max(max(camconstWhite[0], camconstWhite[1]), camconstWhite[2]), - camconstWhite[3]); - double camconstWhiteAvg = (camconstWhite[0] + camconstWhite[1] + - camconstWhite[2] + camconstWhite[3]) / - 4; - double camconstBlackAvg = (camconstBlack[0] + camconstBlack[1] + - camconstBlack[2] + camconstBlack[3]) / - 4; - - // If the black levels are significantly different, we'll add them. - if (camconstBlackAvg != blackpoint && - !isDNG) // dngs provide their own correct black level and we should - // trust it - { - cout << "Black level discrepancy" << endl; - cout << "CamConst black: " << camconstBlackAvg << endl; - cout << "LibRaw black: " << blackpoint << endl; - cout << "block black: " << meanBlockBlackpoint << endl; - if (blackpoint != 0 && camconstSuccess) { - if (abs((camconstBlackAvg / blackpoint) - 1) < 0.5) { - // if they're within 50%, we want to replace the libraw one - blackpoint = camconstBlack[0]; - rBlack = 0; - gBlack = camconstBlack[1] - camconstBlack[0]; - bBlack = camconstBlack[2] - camconstBlack[0]; - maxChanBlack = max(rBlack, max(gBlack, bBlack)); - } else { - // Ignore if they're very different, this only applies to Panasonics - // and there's a better way - } - } else { - // if the libraw blackpoint is 0 then we replace it, unless there was - // a block-based blackpoint - if (meanBlockBlackpoint == 0 && camconstSuccess) { - blackpoint = camconstBlack[0]; - rBlack = 0; - gBlack = camconstBlack[1] - camconstBlack[0]; - bBlack = camconstBlack[2] - camconstBlack[0]; - g2Black = camconstBlack[3] - camconstBlack[0]; - maxChanBlack = max(rBlack, max(gBlack, max(bBlack, g2Black))); - } - } - cout << "new black: " << blackpoint << endl; - } - - // Modern Nikons have camconst.json white levels specified as if they were - // 14-bit - // even if the raw files are 12-bit-only, like the entry level cams - // So we need to detect if it's 12-bit and if the camconst specifies as - // 14-bit. - if ((QString(IDATA.make) == "Nikon") && - (libraw->imgdata.color.maximum < 4096) && - (camconstWhiteAvg >= 4096)) { - camconstWhite[0] = camconstWhite[0] * 4095 / 16383; - camconstWhite[1] = camconstWhite[1] * 4095 / 16383; - camconstWhite[2] = camconstWhite[2] * 4095 / 16383; - cout << "Nikon 12-bit camconst white clipping point: " - << camconstWhite[0] << endl; - } - - if (camconstSuccess && camconstWhiteAvg > 0 && - !isDNG) // dngs provide their own correct whitepoint and we should - // trust it over camconst - { - maxValue = - camconstWhiteMax - blackpoint - maxChanBlack - meanBlockBlackpoint; - cout << "camconst r white clipping point: " << camconstWhite[0] << endl; - cout << "camconst g white clipping point: " << camconstWhite[1] << endl; - cout << "camconst b white clipping point: " << camconstWhite[2] << endl; - colorMaxValue[0] = - camconstWhite[0] - blackpoint - maxChanBlack - meanBlockBlackpoint; - colorMaxValue[1] = - camconstWhite[1] - blackpoint - maxChanBlack - meanBlockBlackpoint; - colorMaxValue[2] = - camconstWhite[2] - blackpoint - maxChanBlack - meanBlockBlackpoint; - } else { - maxValue = libraw->imgdata.color.maximum - blackpoint - maxChanBlack - - meanBlockBlackpoint; - cout << "libraw fallback or dng white clipping point: " - << libraw->imgdata.color.maximum << endl; - colorMaxValue[0] = maxValue; - colorMaxValue[1] = maxValue; - colorMaxValue[2] = maxValue; - } - cout << "black-subtracted maximum: " << maxValue << endl; - cout << "fmaximum: " << libraw->imgdata.color.fmaximum << endl; - cout << "fnorm: " << libraw->imgdata.color.fnorm << endl; - - // get color filter array - // if all the libraw.imgdata.idata.xtrans values are 0, it's bayer. - // bayer only for now - for (unsigned int i = 0; i < 2; i++) { - // cout << "bayer: "; - for (unsigned int j = 0; j < 2; j++) { - cfa[i][j] = unsigned(libraw->COLOR(int(i), int(j))); - if (cfa[i][j] == 3) // Auto CA correct doesn't like 0123 for RGBG; we - // change it to 0121. - { - cfa[i][j] = 1; - } - // cout << cfa[i][j]; - } - // cout << endl; + // Unpack Logic + filename = QString::fromStdString(loadParam.fullFilename); + isCR3 = output.isCR3; + isSraw = output.isSraw; + isNikonSraw = output.isNikonSraw; + isMonochrome = output.isMonochrome; + bool isFloat = output.isFloat; + raw_image.swap(output.data); + + if ((isSraw && !isMonochrome) || (output.isFloat && output.isSraw)) {// Use flags from output + raw_width = raw_image.nc() / 3; + } else { + raw_width = raw_image.nc(); + } + raw_height = raw_image.nr(); + + for (int i = 0; i < 2; ++i) + for (int j = 0; j < 2; ++j) cfa[i][j] = output.cfa[i][j]; + for (int i = 0; i < 6; ++i) + for (int j = 0; j < 6; ++j) xtrans[i][j] = output.xtrans[i][j]; + maxXtrans = output.maxXtrans; + + for (int i = 0; i < 3; ++i) + for (int j = 0; j < 3; ++j) { + camToRGB[i][j] = output.camToRGB[i][j]; + xyzToCam[i][j] = output.xyzToCam[i][j]; } - - // get xtrans color filter array - maxXtrans = 0; - for (int i = 0; i < 6; i++) { - // cout << "xtrans: "; - for (int j = 0; j < 6; j++) { - xtrans[i][j] = uint(libraw->imgdata.idata.xtrans[i][j]); - maxXtrans = max(maxXtrans, int(libraw->imgdata.idata.xtrans[i][j])); - // cout << xtrans[i][j]; - } - // cout << endl; + // Reconstruct camToRGB4 + for (int i = 0; i < 3; i++) { + for (int j = 0; j < 4; j++) { + camToRGB4[i][j] = camToRGB[i][j]; + if (i == j) + camToRGB4[i][j] = 1; + else + camToRGB4[i][j] = 0; + if (j == 3) camToRGB4[i][j] = camToRGB4[i][1]; } + } - if (!isCR3) // we can't use exiv2 on CR3 yet - { - cout << "processImage exiv filename: " << loadParam.fullFilename - << endl; - auto image = Exiv2::ImageFactory::open(loadParam.fullFilename); - assert(image.get() != 0); - image->readMetadata(); - exifData = image->exifData(); - } else { - // We need to fabricate fresh exif data from what libraw gives us - Exiv2::ExifData basicExifData; - - basicExifData["Exif.Image.Orientation"] = uint16_t(1); - basicExifData["Exif.Image.ImageWidth"] = - vibrance_saturation_image.nc() / 3; - basicExifData["Exif.Image.ImageLength"] = - vibrance_saturation_image.nr(); - basicExifData["Exif.Image.Make"] = IDATA.make; - basicExifData["Exif.Image.Model"] = IDATA.model; - basicExifData["Exif.Image.DateTime"] = - exifDateTimeString(OTHER.timestamp); - basicExifData["Exif.Photo.DateTimeOriginal"] = - exifDateTimeString(OTHER.timestamp); - basicExifData["Exif.Photo.DateTimeDigitized"] = - exifDateTimeString(OTHER.timestamp); - basicExifData["Exif.Photo.ExposureTime"] = rationalTv(OTHER.shutter); - basicExifData["Exif.Photo.FNumber"] = rationalAvFL(OTHER.aperture); - basicExifData["Exif.Photo.ISOSpeed"] = int(round(OTHER.iso_speed)); - basicExifData["Exif.Photo.FocalLength"] = rationalAvFL(OTHER.focal_len); - - exifData = basicExifData; - } + exifData = output.exif; - raw_image.set_size(raw_height, raw_width); - - // copy raw data - float rawMin = std::numeric_limits::max(); - float rawMax = std::numeric_limits::min(); - float rawRMax = std::numeric_limits::min(); - float rawGMax = std::numeric_limits::min(); - float rawBMax = std::numeric_limits::min(); - - isSraw = libraw->is_sraw(); - - // Iridient X-Transformer creates full-color files that aren't sraw - // They have 6666 as the cfa and all 0 for xtrans - // However, Leica M Monochrom files are exactly the same! - // So we have to check if the white balance tag exists. - bool isWeird = (cfa[0][0] == 6 && cfa[0][1] == 6 && cfa[1][0] == 6 && - cfa[1][1] == 6); - cout << "raw identification is weird: " << isWeird << endl; - - bool noWB = false; - if (!isCR3) { // we can't use exiv2 on CR3 yet and no CR3 cameras are mono - if (isDNG) { - //DNGs don't have an exif tag for white balance - //check cam_xyz - cout << "raw identification cam_xyz rr: " << xyzToCam[0][0] << endl; - noWB = abs(xyzToCam[0][0]) < 0.01f; - cout << "raw identification is dng" << endl; - } else { - noWB = exifData["Exif.Photo.WhiteBalance"].toString().length() == 0; - } - } - cout << "raw identification no white balance: " << noWB << endl; - isMonochrome = isWeird && noWB; - cout << "raw identification is monochrome: " << isMonochrome << endl; - isSraw = isSraw || (isWeird && !isMonochrome); - cout << "raw identification is full color raw: " << isSraw << endl; - - isNikonSraw = libraw->is_nikon_sraw(); - if (isFloat && isSraw) { // floating point full-color-per-pixel raws - raw_image.set_size(raw_height, raw_width * 3); -#pragma omp parallel for reduction(min : rawMin) reduction(max : rawMax) \ - reduction(max : rawRMax) reduction(max : rawGMax) reduction(max : rawBMax) - for (int row = 0; row < raw_height; row++) { - // IMAGE is an (width*height) by 4 array, not width by height by 4. - int rowoffset = (row + topmargin) * full_width; - for (int col = 0; col < raw_width; col++) { - float tempBlackpoint = blackpoint; - if (blackRow > 0 && blackCol > 0) { - tempBlackpoint = - tempBlackpoint + - libraw->imgdata.color - .cblack[6 + (row % blackRow) * blackCol + col % blackCol]; - } - // sraw comes from raw4 but only uses 3 channels - raw_image[row][col * 3] = - min(RAWF4[rowoffset + col + leftmargin][0] - tempBlackpoint, - colorMaxValue[0]); - rawMin = std::min(rawMin, raw_image[row][col * 3]); - rawMax = std::max(rawMax, raw_image[row][col * 3]); - rawRMax = std::max(rawRMax, raw_image[row][col * 3]); - raw_image[row][col * 3 + 1] = - min(RAWF4[rowoffset + col + leftmargin][1] - tempBlackpoint, - colorMaxValue[1]); - rawMin = std::min(rawMin, raw_image[row][col * 3 + 1]); - rawMax = std::max(rawMax, raw_image[row][col * 3 + 1]); - rawGMax = std::max(rawGMax, raw_image[row][col * 3 + 1]); - raw_image[row][col * 3 + 2] = - min(RAWF4[rowoffset + col + leftmargin][2] - tempBlackpoint, - colorMaxValue[2]); - rawMin = std::min(rawMin, raw_image[row][col * 3 + 2]); - rawMax = std::max(rawMax, raw_image[row][col * 3 + 2]); - rawBMax = std::max(rawBMax, raw_image[row][col * 3 + 2]); - } - } - } else if (isSraw) { // full-color-per-pixel integer raws - raw_image.set_size(raw_height, raw_width * 3); -#pragma omp parallel for reduction(min : rawMin) reduction(max : rawMax) \ - reduction(max : rawRMax) reduction(max : rawGMax) reduction(max : rawBMax) - - for (int row = 0; row < raw_height; row++) { - // IMAGE is an (width*height) by 4 array, not width by height by 4. - int rowoffset = (row + topmargin) * full_width; - for (int col = 0; col < raw_width; col++) { - float tempBlackpoint = blackpoint; - if (blackRow > 0 && blackCol > 0) { - tempBlackpoint = - tempBlackpoint + - libraw->imgdata.color - .cblack[6 + (row % blackRow) * blackCol + col % blackCol]; - } - // sraw comes from raw4 but only uses 3 channels - raw_image[row][col * 3] = - min(RAW4[rowoffset + col + leftmargin][0] - tempBlackpoint, - colorMaxValue[0]); - rawMin = std::min(rawMin, raw_image[row][col * 3]); - rawMax = std::max(rawMax, raw_image[row][col * 3]); - rawRMax = std::max(rawRMax, raw_image[row][col * 3]); - raw_image[row][col * 3 + 1] = - min(RAW4[rowoffset + col + leftmargin][1] - tempBlackpoint, - colorMaxValue[1]); - rawMin = std::min(rawMin, raw_image[row][col * 3 + 1]); - rawMax = std::max(rawMax, raw_image[row][col * 3 + 1]); - rawGMax = std::max(rawGMax, raw_image[row][col * 3 + 1]); - raw_image[row][col * 3 + 2] = - min(RAW4[rowoffset + col + leftmargin][2] - tempBlackpoint, - colorMaxValue[2]); - rawMin = std::min(rawMin, raw_image[row][col * 3 + 2]); - rawMax = std::max(rawMax, raw_image[row][col * 3 + 2]); - rawBMax = std::max(rawBMax, raw_image[row][col * 3 + 2]); - } - } - } else if (isFloat) { // floating point one-color-per-pixel raws -#pragma omp parallel for reduction(min : rawMin) reduction(max : rawMax) \ - reduction(max : rawRMax) reduction(max : rawGMax) reduction(max : rawBMax) - - for (int row = 0; row < raw_height; row++) { - // IMAGE is an (width*height) by 4 array, not width by height by 4. - int rowoffset = (row + topmargin) * full_width; - for (int col = 0; col < raw_width; col++) { - float tempBlackpoint = blackpoint; - float tempWhitepoint = maxValue; - int color = cfa[row % 2][col % 2]; - if (color == 0) { - tempBlackpoint += rBlack; - tempWhitepoint = colorMaxValue[0]; - } - if (color == 1) { - tempBlackpoint += gBlack; - tempWhitepoint = colorMaxValue[1]; - } - if (color == 2) { - tempBlackpoint += bBlack; - tempWhitepoint = colorMaxValue[2]; - } - if (color == 3) { - tempBlackpoint += g2Black; - tempWhitepoint = colorMaxValue[1]; - } - if (blackRow > 0 && blackCol > 0) { - tempBlackpoint = min( - tempBlackpoint + libraw->imgdata.color - .cblack[6 + (row % blackRow) * blackCol + - col % blackCol], - tempWhitepoint); - } - raw_image[row][col] = - RAWF[rowoffset + col + leftmargin] - tempBlackpoint; - rawMin = std::min(rawMin, raw_image[row][col]); - rawMax = std::max(rawMax, raw_image[row][col]); - if (color == 0) { - rawRMax = std::max(rawRMax, raw_image[row][col]); - } - if (color == 1) { - rawGMax = std::max(rawGMax, raw_image[row][col]); - } - if (color == 2) { - rawBMax = std::max(rawBMax, raw_image[row][col]); - } - if (color == 3) { - rawGMax = std::max(rawGMax, raw_image[row][col]); - } - } - } - } else { // normal one-color-per-pixel integer raws -#pragma omp parallel for reduction(min : rawMin) reduction(max : rawMax) \ - reduction(max : rawRMax) reduction(max : rawGMax) reduction(max : rawBMax) - - for (int row = 0; row < raw_height; row++) { - // IMAGE is an (width*height) by 4 array, not width by height by 4. - int rowoffset = (row + topmargin) * full_width; - for (int col = 0; col < raw_width; col++) { - float tempBlackpoint = blackpoint; - float tempWhitepoint = maxValue; - int color = cfa[row % 2][col % 2]; - if (color == 0) { - tempBlackpoint += rBlack; - tempWhitepoint = colorMaxValue[0]; - } - if (color == 1) { - tempBlackpoint += gBlack; - tempWhitepoint = colorMaxValue[1]; - } - if (color == 2) { - tempBlackpoint += bBlack; - tempWhitepoint = colorMaxValue[2]; - } - if (color == 3) { - tempBlackpoint += g2Black; - tempWhitepoint = colorMaxValue[1]; - } - if (blackRow > 0 && blackCol > 0) { - tempBlackpoint = min( - tempBlackpoint + libraw->imgdata.color - .cblack[6 + (row % blackRow) * blackCol + - col % blackCol], - tempWhitepoint); - } - raw_image[row][col] = - RAW[rowoffset + col + leftmargin] - tempBlackpoint; - rawMin = std::min(rawMin, raw_image[row][col]); - rawMax = std::max(rawMax, raw_image[row][col]); - if (color == 0) { - rawRMax = std::max(rawRMax, raw_image[row][col]); - } - if (color == 1) { - rawGMax = std::max(rawGMax, raw_image[row][col]); - } - if (color == 2) { - rawBMax = std::max(rawBMax, raw_image[row][col]); - } - if (color == 3) { - rawGMax = std::max(rawGMax, raw_image[row][col]); - } - } - } - } + rPreMul = output.rPreMul; + gPreMul = output.gPreMul; + bPreMul = output.bPreMul; + rCamMul = output.rCamMul; + gCamMul = output.gCamMul; + bCamMul = output.bCamMul; - // generate raw histogram - if (WithHisto == histo) { - histoInterface->updateHistRaw(raw_image, colorMaxValue, cfa, xtrans, - maxXtrans, isSraw, isMonochrome); - } + rUserMul = rCamMul; + gUserMul = gCamMul; + bUserMul = bCamMul; - cout << "max of raw_image: " << rawMax << endl; - cout << "min of raw_image: " << rawMin << endl; - cout << "max of raw red: " << rawRMax << endl; - cout << "max of raw green: " << rawGMax << endl; - cout << "max of raw blue: " << rawBMax << endl; + maxValue = output.maxValue; + for (int i = 0; i < 3; ++i) colorMaxValue[i] = output.colorMaxValue[i]; - if(needs_phase_one_free) { - libraw->my_phase_one_free_tempbuffer(); - } - } valid = paramManager->markLoadComplete(); updateProgress(valid, 0.0f); [[fallthrough]]; } case partdemosaic: [[fallthrough]]; - case load: // Do demosaic, or load non-raw images + case load:// Do demosaic, or load non-raw images { LoadParams loadParam; DemosaicParams demosaicParam; AbortStatus abort; - std::tie(valid, abort, loadParam, demosaicParam) = - paramManager->claimDemosaicParams(); + std::tie(valid, abort, loadParam, demosaicParam) = paramManager->claimDemosaicParams(); if (abort == AbortStatus::restart) { - cout << "imagePipeline.cpp: aborted at demosaic" << endl; + FILM_WARN("imagePipeline.cpp: aborted at demosaic"); return emptyMatrix(); } - cout << "imagePipeline.cpp: Opening " << loadParam.fullFilename << endl; + // Construct Input + RawImage input; + input.data.swap(raw_image);// Move logic + + // Copy Metadata + input.isSraw = isSraw; + input.isNikonSraw = isNikonSraw; + input.isMonochrome = isMonochrome; + input.isCR3 = isCR3; + input.maxXtrans = maxXtrans; + for (int i = 0; i < 2; ++i) + for (int j = 0; j < 2; ++j) input.cfa[i][j] = cfa[i][j]; + for (int i = 0; i < 6; ++i) + for (int j = 0; j < 6; ++j) input.xtrans[i][j] = xtrans[i][j]; + + input.rPreMul = rPreMul; + input.gPreMul = gPreMul; + input.bPreMul = bPreMul; + input.rCamMul = rCamMul; + input.gCamMul = gCamMul; + input.bCamMul = bCamMul; + input.maxValue = maxValue; + for (int i = 0; i < 3; ++i) input.colorMaxValue[i] = colorMaxValue[i]; + + for (int i = 0; i < 3; ++i) + for (int j = 0; j < 3; ++j) { + input.camToRGB[i][j] = camToRGB[i][j]; + input.xyzToCam[i][j] = xyzToCam[i][j]; + } + // DemosaicStage doesn't seem to need camToRGB4 per my check. - // Reads in the photo. - cout << "load start:" << timeDiff(timeRequested) << endl; - auto imload_time = std::chrono::steady_clock::now(); + DemosaicStage stage; + auto result = stage.process(input, demosaicParam, context); - if (loadParam.tiffIn) { - if (imread_tiff(loadParam.fullFilename, demosaiced_image, exifData)) { - cerr << "Could not open image " << loadParam.fullFilename - << "; Exiting..." << endl; - return emptyMatrix(); - } - } else if (loadParam.jpegIn) { - if (imread_jpeg(loadParam.fullFilename, demosaiced_image, exifData)) { - cerr << "Could not open image " << loadParam.fullFilename - << "; Exiting..." << endl; - return emptyMatrix(); - } - } else if (isSraw) // already demosaiced - { - // We just need to scale to 65535, and apply camera WB - float inputscale = maxValue; - float outputscale = 65535.0; - float scaleFactor = outputscale / inputscale; - demosaiced_image.set_size(raw_height, raw_width * 3); - if (isNikonSraw) { -#pragma omp parallel for - for (int row = 0; row < raw_height; row++) { - for (int col = 0; col < raw_width * 3; col++) { - demosaiced_image(row, col) = raw_image(row, col) * scaleFactor; - } - } - } else { -#pragma omp parallel for - for (int row = 0; row < raw_height; row++) { - for (int col = 0; col < raw_width * 3; col++) { - int color = col % 3; - demosaiced_image(row, col) = raw_image(row, col) * scaleFactor * - ((color == 0) ? rPreMul - : (color == 1) ? gPreMul - : bPreMul); - } - } - } - } else // raw - { - matrix red(raw_height, raw_width); - matrix green(raw_height, raw_width); - matrix blue(raw_height, raw_width); - - double initialGain = 1.0; - float inputscale = maxValue; - float outputscale = 65535.0; - const int border = 4; // used for amaze - std::function setProg = [](double) -> bool { - return false; - }; - - cout << "raw width: " << raw_width << endl; - cout << "raw height: " << raw_height << endl; - - // before demosaic, you want to apply raw white balance - //====================================================================== - // TODO: If the camera white balance disagrees with some sort of AWB by a - // *lot*, use an awb instead - //====================================================================== - matrix premultiplied(raw_height, raw_width); - - cout << "demosaic start" << timeDiff(timeRequested) << endl; - auto demosaic_time = std::chrono::steady_clock::now(); - - if (maxXtrans > 0) { -#pragma omp parallel for - for (int row = 0; row < raw_height; row++) { - for (int col = 0; col < raw_width; col++) { - uint color = xtrans[uint(row) % 6][uint(col) % 6]; - premultiplied(row, col) = - raw_image(row, col) * ((color == 0) ? rPreMul - : (color == 1) ? gPreMul - : bPreMul); - } - } - if (demosaicParam.demosaicMethod == 0) { - markesteijn_demosaic(raw_width, raw_height, premultiplied, red, green, - blue, xtrans, camToRGB4, setProg, 3, true); - } else { // if it's 1, use xtransfast - xtransfast_demosaic(raw_width, raw_height, premultiplied, red, green, - blue, xtrans, setProg); - } - // there's no inputscale for markesteijn so we need to scale - float scaleFactor = outputscale / inputscale; -#pragma omp parallel for - for (int row = 0; row < red.nr(); row++) { - for (int col = 0; col < red.nc(); col++) { - red(row, col) = red(row, col) * scaleFactor; - green(row, col) = green(row, col) * scaleFactor; - blue(row, col) = blue(row, col) * scaleFactor; - } - } - } else if (isMonochrome) { - float scaleFactor = outputscale / inputscale; - for (int row = 0; row < raw_height; row++) { - for (int col = 0; col < raw_width; col++) { - red(row, col) = raw_image(row, col) * scaleFactor; - green(row, col) = raw_image(row, col) * scaleFactor; - blue(row, col) = raw_image(row, col) * scaleFactor; - } - } - } else { -#pragma omp parallel for - for (int row = 0; row < raw_height; row++) { - for (int col = 0; col < raw_width; col++) { - uint color = cfa[uint(row) & 1][uint(col) & 1]; - premultiplied(row, col) = - raw_image(row, col) * ((color == 0) ? rPreMul - : (color == 1) ? gPreMul - : bPreMul); - } - } - if (demosaicParam.caEnabled > 0) { - // we need to apply white balance and then remove it for Auto CA - // Correct to work properly - double fitparams[2][2][16]; - CA_correct(0, 0, raw_width, raw_height, true, demosaicParam.caEnabled, - 0.0, 0.0, true, premultiplied, premultiplied, cfa, setProg, - fitparams, false); - } + input.data.swap(raw_image);// Restore raw_image - if (demosaicParam.demosaicMethod == 0) { - amaze_demosaic(raw_width, raw_height, 0, 0, raw_width, raw_height, - premultiplied, red, green, blue, cfa, setProg, - initialGain, border, inputscale, outputscale); - } else { // if it's 1, use LMMSE - premultiplied.mult_this(1 / inputscale); - lmmse_demosaic(raw_width, raw_height, premultiplied, red, green, blue, - cfa, setProg, 3); // doesn't like inputs > 1 - // igv_demosaic(raw_width, raw_height, premultiplied, red, green, - // blue, cfa, setProg);//doesn't like inputs > 1 - red.mult_this(outputscale); - green.mult_this(outputscale); - blue.mult_this(outputscale); - } - } - premultiplied.set_size(0, 0); - cout << "demosaic end: " << timeDiff(demosaic_time) << endl; - - demosaiced_image.set_size(raw_height, raw_width * 3); -#pragma omp parallel for - for (int row = 0; row < raw_height; row++) { - for (int col = 0; col < raw_width; col++) { - demosaiced_image(row, col * 3) = red(row, col); - demosaiced_image(row, col * 3 + 1) = green(row, col); - demosaiced_image(row, col * 3 + 2) = blue(row, col); - } - } - } - cout << "load time: " << timeDiff(imload_time) << endl; + if (!result) return emptyMatrix(); + + RawImage &output = *result; + demosaiced_image.swap(output.data); - cout << "ImagePipeline::processImage: Demosaic complete." << endl; + // Mark complete valid = paramManager->markDemosaicComplete(); updateProgress(valid, 0.0f); [[fallthrough]]; } case partpostdemosaic: [[fallthrough]]; - case demosaic: // Do postdemosaic work + case demosaic:// Do postdemosaic work { PostDemosaicParams postDemosaicParam; AbortStatus abort; - std::tie(valid, abort, postDemosaicParam) = - paramManager->claimPostDemosaicParams(); - if (abort == AbortStatus::restart) { - cout << "imagePipeline.cpp: aborted at demosaic" << endl; - return emptyMatrix(); - } - - // First thing after demosaic is to apply the user's white balance. - if (!isMonochrome) { - rawWhiteBalance(demosaiced_image, post_demosaic_image, - postDemosaicParam.temperature, postDemosaicParam.tint, - xyzToCam, rPreMul, gPreMul, bPreMul, // undoes these - rUserMul, gUserMul, - bUserMul); // used later for highlight recovery - - cout << "WB pre multiplier R: " << rPreMul << endl; - cout << "WB pre multiplier G: " << gPreMul << endl; - cout << "WB pre multiplier B: " << bPreMul << endl; - cout << "WB cam multiplier R: " << rCamMul << endl; - cout << "WB cam multiplier G: " << gCamMul << endl; - cout << "WB cam multiplier B: " << bCamMul << endl; - cout << "WB user multiplier R: " << rUserMul << endl; - cout << "WB user multiplier G: " << gUserMul << endl; - cout << "WB user multiplier B: " << bUserMul << endl; - } else { - post_demosaic_image = demosaiced_image; - } - - // Recover highlights now - cout << "hlrecovery start:" << timeDiff(timeRequested) << endl; - std::chrono::steady_clock::time_point hlrecovery_time; - hlrecovery_time = std::chrono::steady_clock::now(); - - int height = post_demosaic_image.nr(); - int width = post_demosaic_image.nc() / 3; - - // Now, recover highlights. - std::function setProg = [](double) -> bool { return false; }; - // And return it back to a single layer - if (postDemosaicParam.highlights >= 2 && !isMonochrome) { - // For highlight recovery, we need to split up the image into three - // separate layers. - matrix rChannel(height, width), gChannel(height, width), - bChannel(height, width); - -#pragma omp parallel for - for (int row = 0; row < height; row++) { - for (int col = 0; col < width; col++) { - rChannel(row, col) = post_demosaic_image(row, col * 3); - gChannel(row, col) = post_demosaic_image(row, col * 3 + 1); - bChannel(row, col) = post_demosaic_image(row, col * 3 + 2); - } - } - - // We applied the camMul camera multipliers before applying white balance. - // Now we need to calculate the channel max and the raw clip levels. - // Channel max: - const float chmax[3] = {rChannel.max(), gChannel.max(), bChannel.max()}; - // Max clip point: - const float clmax[3] = {65535.0f * rUserMul * colorMaxValue[0] / maxValue, - 65535.0f * gUserMul * colorMaxValue[1] / maxValue, - 65535.0f * bUserMul * colorMaxValue[2] / - maxValue}; - - HLRecovery_inpaint(width, height, rChannel, gChannel, bChannel, chmax, - clmax, setProg); -#pragma omp parallel for - for (int row = 0; row < height; row++) { - for (int col = 0; col < width; col++) { - post_demosaic_image(row, col * 3) = rChannel(row, col); - post_demosaic_image(row, col * 3 + 1) = gChannel(row, col); - post_demosaic_image(row, col * 3 + 2) = bChannel(row, col); - } - } - } else if (postDemosaicParam.highlights == 0 && !isMonochrome) { -#pragma omp parallel for - for (int row = 0; row < height; row++) { - for (int col = 0; col < width; col++) { - post_demosaic_image(row, col * 3) = - min(post_demosaic_image(row, col * 3), 65535.0f); - post_demosaic_image(row, col * 3 + 1) = - min(post_demosaic_image(row, col * 3 + 1), 65535.0f); - post_demosaic_image(row, col * 3 + 2) = - min(post_demosaic_image(row, col * 3 + 2), 65535.0f); - } - } - } else { // params = 1, or isMonochrome - // do nothing - } - cout << "hlrecovery duration: " << timeDiff(hlrecovery_time) << endl; - - // Apply exposure compensation. - // This is ideally done before noise reduction so that you don't need - // dramatically different - // thresholds for underexposed images. - - const float expCompMult = pow(2, postDemosaicParam.exposureComp); - -#pragma omp parallel for - for (int row = 0; row < height; row++) { - for (int col = 0; col < width * 3; col++) { - post_demosaic_image(row, col) = - post_demosaic_image(row, col) * expCompMult; - } - } + std::tie(valid, abort, postDemosaicParam) = paramManager->claimPostDemosaicParams(); + if (abort == AbortStatus::restart) return emptyMatrix(); + + // Construct Input + RawImage input; + input.data.swap(demosaiced_image); + + input.rPreMul = rPreMul; + input.gPreMul = gPreMul; + input.bPreMul = bPreMul; + input.rUserMul = rUserMul; + input.gUserMul = gUserMul; + input.bUserMul = bUserMul; + input.rCamMul = rCamMul; + input.gCamMul = gCamMul; + input.bCamMul = bCamMul; + input.maxValue = maxValue; + for (int i = 0; i < 3; ++i) input.colorMaxValue[i] = colorMaxValue[i]; + for (int i = 0; i < 3; ++i) + for (int j = 0; j < 3; ++j) input.xyzToCam[i][j] = xyzToCam[i][j]; + input.isMonochrome = isMonochrome; + + PostDemosaicStage stage; + auto result = stage.process(input, postDemosaicParam, context); + + input.data.swap(demosaiced_image);// Restore + + if (!result) return emptyMatrix(); + + RawImage &output = *result; + post_demosaic_image.swap(output.data); + + rUserMul = output.rUserMul; + gUserMul = output.gUserMul; + bUserMul = output.bUserMul; + + if (rUserMul == 0) rUserMul = 1; + if (gUserMul == 0) gUserMul = 1; + if (bUserMul == 0) bUserMul = 1; valid = paramManager->markPostDemosaicComplete(); +#ifdef ENABLE_NAN_TRAPPING + SCAN_MATRIX_FOR_NAN(post_demosaic_image, "post_demosaic_image"); +#endif updateProgress(valid, 0.0f); [[fallthrough]]; } case partnrnlmeans: [[fallthrough]]; - case postdemosaic: // Do nonlocal means (luma+chroma) noise reduction + case postdemosaic:// Do nonlocal means (luma+chroma) noise reduction { NlmeansNRParams nrParam; AbortStatus abort; std::tie(valid, abort, nrParam) = paramManager->claimNlmeansNRParams(); - if (abort == AbortStatus::restart) { - cout << "imagePipeline aborted at nlmeans noise reduction" << endl; - return emptyMatrix(); - } + if (abort == AbortStatus::restart) return emptyMatrix(); - if (nrParam.nrEnabled && nrParam.nlStrength > 0) { - cout << "Luma NR preprocessing start: " << timeDiff(timeRequested) - << endl; - matrix denoised(post_demosaic_image.nr(), - post_demosaic_image.nc()); - matrix preconditioned = post_demosaic_image; + // Construct Input + RawImage input; + input.data.swap(post_demosaic_image); - if (cache == NoCache) { - post_demosaic_image.set_size(0, 0); - cacheEmpty = true; - } else { - cacheEmpty = false; - } + input.rUserMul = rUserMul; + input.gUserMul = gUserMul; + input.bUserMul = bUserMul; + input.isMonochrome = isMonochrome; + for (int i = 0; i < 3; ++i) + for (int j = 0; j < 3; ++j) input.camToRGB[i][j] = camToRGB[i][j]; - // we don't want to apply nonexistent WB multipliers to monochrome images - const float rMulTemp = isMonochrome ? 1.0f : rUserMul; - const float gMulTemp = isMonochrome ? 1.0f : gUserMul; - const float bMulTemp = isMonochrome ? 1.0f : bUserMul; - -#pragma omp parallel for - for (int row = 0; row < preconditioned.nr(); row++) { - for (int col = 0; col < preconditioned.nc(); col += 3) { - preconditioned(row, col + 0) = sRGB_forward_gamma_unclipped( - preconditioned(row, col + 0) / (rMulTemp * 65535.0f)); - preconditioned(row, col + 1) = sRGB_forward_gamma_unclipped( - preconditioned(row, col + 1) / (gMulTemp * 65535.0f)); - preconditioned(row, col + 2) = sRGB_forward_gamma_unclipped( - preconditioned(row, col + 2) / (bMulTemp * 65535.0f)); - } - } - float offset = std::max(-preconditioned.min() + 0.001f, 0.001f); - float scale = std::max(preconditioned.max() + offset, 1.0f); -#pragma omp parallel for - for (int row = 0; row < preconditioned.nr(); row++) { - for (int col = 0; col < preconditioned.nc(); col++) { - preconditioned(row, col) = - (preconditioned(row, col) + offset) / scale; - if (isnan(preconditioned(row, col))) { - preconditioned(row, col) = 0.0f; - } - } - } + NlmeansNRStage stage; + auto result = stage.process(input, nrParam, context); - const int numClusters = nrParam.nlClusters; - const float clusterThreshold = nrParam.nlThresh; - const float strength = nrParam.nlStrength; + input.data.swap(post_demosaic_image);// Restore - cout << "Luma NR processing start: " << timeDiff(timeRequested) << endl; - auto nrTime = std::chrono::steady_clock::now(); + if (!result) return emptyMatrix(); + RawImage &output = *result; - if (kMeansNLMApprox(preconditioned, numClusters, clusterThreshold, - strength, preconditioned.nr(), - preconditioned.nc() / 3, denoised, paramManager)) { - cout << "imagePipeline aborted at nlmeans noise reduction" << endl; - return emptyMatrix(); - } - cout << "Nlmeans NR duration: " << timeDiff(nrTime) << endl; - - // Undo the preconditioning -#pragma omp parallel for - for (int row = 0; row < denoised.nr(); row++) { - for (int col = 0; col < denoised.nc(); col += 3) { - denoised(row, col + 0) = - sRGB_inverse_gamma_unclipped(scale * denoised(row, col + 0) - - offset) * - rMulTemp * 65535.0f; - denoised(row, col + 1) = - sRGB_inverse_gamma_unclipped(scale * denoised(row, col + 1) - - offset) * - gMulTemp * 65535.0f; - denoised(row, col + 2) = - sRGB_inverse_gamma_unclipped(scale * denoised(row, col + 2) - - offset) * - bMulTemp * 65535.0f; - } - } - - raw_to_oklab(denoised, nlmeans_nr_image, camToRGB); - denoised.set_size(0, 0); - } else if (nrParam.nrEnabled) { - raw_to_oklab(post_demosaic_image, nlmeans_nr_image, camToRGB); - - if (cache == NoCache) { - post_demosaic_image.set_size(0, 0); - cacheEmpty = true; - } else { - cacheEmpty = false; - } - } - // If noise reduction is not enabled, we'll completely skip copying so that - // the non-NR path is as fast as possible If we're not caching, we do not - // want to erase the input data unless it actually got used already. + nlmeans_nr_image.swap(output.data); valid = paramManager->markNlmeansNRComplete(); +#ifdef ENABLE_NAN_TRAPPING + SCAN_MATRIX_FOR_NAN(nlmeans_nr_image, "nlmeans_nr_image"); +#endif updateProgress(valid, 0.0f); [[fallthrough]]; } case partnrimpulse: [[fallthrough]]; - case nrnlmeans: // do impulse denoise + case nrnlmeans:// do impulse denoise { ImpulseNRParams nrParam; AbortStatus abort; std::tie(valid, abort, nrParam) = paramManager->claimImpulseNRParams(); - if (abort == AbortStatus::restart) { - cout << "imagePipeline aborted at impulse noise reduction" << endl; - return emptyMatrix(); - } + if (abort == AbortStatus::restart) return emptyMatrix(); - if (nrParam.nrEnabled && nrParam.impulseThresh > 0) { - cout << "Impulse NR processing start: " << timeDiff(timeRequested) - << endl; - auto nrTime = std::chrono::steady_clock::now(); + RawImage input; + input.data.swap(nlmeans_nr_image); + input.isOklab = nrParam.nrEnabled; - const bool eraseNRInput = (cache == NoCache); - impulse_nr(nlmeans_nr_image, impulse_nr_image, nrParam.impulseThresh, 1.0, - eraseNRInput); + ImpulseNRStage stage; + auto result = stage.process(input, nrParam, context); - cout << "Impulse NR duration: " << timeDiff(nrTime) << endl; - } else if (nrParam.nrEnabled) { - impulse_nr_image = nlmeans_nr_image; - } + input.data.swap(nlmeans_nr_image);// Restore - if (cache == NoCache) { - nlmeans_nr_image.set_size(0, 0); - cacheEmpty = true; - } else { - cacheEmpty = false; - } + if (!result) return emptyMatrix(); + RawImage &output = *result; + + impulse_nr_image.swap(output.data); valid = paramManager->markImpulseNRComplete(); +#ifdef ENABLE_NAN_TRAPPING + SCAN_MATRIX_FOR_NAN(impulse_nr_image, "impulse_nr_image"); +#endif updateProgress(valid, 0.0f); [[fallthrough]]; } case partnrchroma: [[fallthrough]]; - case nrimpulse: // do chroma denoise + case nrimpulse:// do chroma denoise { ChromaNRParams nrParam; AbortStatus abort; std::tie(valid, abort, nrParam) = paramManager->claimChromaNRParams(); - if (abort == AbortStatus::restart) { - cout << "imagePipeline aborted at chroma noise reduction" << endl; - return emptyMatrix(); - } + if (abort == AbortStatus::restart) return emptyMatrix(); - if (nrParam.nrEnabled && nrParam.chromaStrength > 0) { - cout << "Chroma NR processing start: " << timeDiff(timeRequested) << endl; - auto nrTime = std::chrono::steady_clock::now(); + RawImage input; + input.data.swap(impulse_nr_image); + input.isOklab = nrParam.nrEnabled; + for (int i = 0; i < 3; ++i) + for (int j = 0; j < 3; ++j) input.camToRGB[i][j] = camToRGB[i][j]; - const bool eraseNRInput = (cache == NoCache); - RGB_denoise(0, impulse_nr_image, chroma_nr_image, nrParam.chromaStrength, - 0.0f, 0.0f, paramManager, eraseNRInput); + ChromaNRStage stage; + auto result = stage.process(input, nrParam, context); - cout << "Chroma NR duration: " << timeDiff(nrTime) << endl; - } else if (nrParam.nrEnabled) { - chroma_nr_image = impulse_nr_image; - } - paramManager->markChromaNRComplete(); + input.data.swap(impulse_nr_image);// Restore + + if (!result) return emptyMatrix(); + RawImage &output = *result; + + chroma_nr_image.swap(output.data); - if (cache == NoCache) { - impulse_nr_image.set_size(0, 0); - cacheEmpty = true; - } else { - cacheEmpty = false; - } valid = paramManager->markChromaNRComplete(); +#ifdef ENABLE_NAN_TRAPPING + SCAN_MATRIX_FOR_NAN(chroma_nr_image, "chroma_nr_image"); +#endif updateProgress(valid, 0.0f); [[fallthrough]]; } case partprefilmulation: [[fallthrough]]; - case nrchroma: // Do pre-filmulation work. + case nrchroma:// Do pre-filmulation work. { PrefilmParams prefilmParam; - cout << "imagePipeline beginning pre-filmulation" << endl; + FILM_DEBUG("imagePipeline beginning pre-filmulation"); AbortStatus abort; std::tie(valid, abort, prefilmParam) = paramManager->claimPrefilmParams(); - if (abort == AbortStatus::restart) { - cout << "imagePipeline aborted at pre-filmulation" << endl; - return emptyMatrix(); + if (abort == AbortStatus::restart) return emptyMatrix(); + + RawImage input; + if (prefilmParam.nrEnabled) { + input.data.swap(chroma_nr_image); + input.isOklab = true; + } else { + input.data.swap(post_demosaic_image); + input.isOklab = false; } + for (int i = 0; i < 3; ++i) + for (int j = 0; j < 3; ++j) input.camToRGB[i][j] = camToRGB[i][j]; - // Copy from the correct location if no NR - // If NR, then convert from oklab back to raw colors - // Then apply lensfun to raw colors + PrefilmulationStage stage; + auto result = stage.process(input, prefilmParam, context); - matrix prefilm_input_image; if (prefilmParam.nrEnabled) { - oklab_to_raw(chroma_nr_image, prefilm_input_image, camToRGB); - if (cache == NoCache) { - chroma_nr_image.set_size(0, 0); - } + input.data.swap(chroma_nr_image); } else { - prefilm_input_image = post_demosaic_image; - } - int height = prefilm_input_image.nr(); - int width = prefilm_input_image.nc() / 3; - - // Lensfun processing - cout << "lensfun start" << endl; - std::string camName = prefilmParam.cameraName.toStdString(); - const lfCamera *camera = NULL; - const lfCamera **cameraList = ldb->FindCamerasExt(NULL, camName.c_str()); - - // Set up stuff for rotation. - // We expect rotation to be from -45 to +45 - // But -50 will be the signal from the UI to disable it. - float rotationAngle = prefilmParam.rotationAngle * 3.1415926535 / - 180; // convert degrees to radians - if (prefilmParam.rotationAngle <= -49) { - rotationAngle = 0; + input.data.swap(post_demosaic_image); } - cout << "cos rotationangle: " << cos(rotationAngle) << endl; - cout << "sin rotationangle: " << sin(rotationAngle) << endl; - bool lensfunGeometryCorrectionApplied = false; - - if (cameraList) { - const float cropFactor = cameraList[0]->CropFactor; - - QString tempLensName = prefilmParam.lensName; - if (tempLensName.length() > 0) { - if (tempLensName.front() == "\\") { - // if the lens name starts with a backslash, don't filter by camera - tempLensName.remove(0, 1); - } else { - // if it doesn't start with a backslash, filter by camera - camera = cameraList[0]; - } - } - std::string lensName = tempLensName.toStdString(); - const lfLens *lens = NULL; - const lfLens **lensList = NULL; - lensList = ldb->FindLenses(camera, NULL, lensName.c_str()); - if (lensList) { - lens = lensList[0]; - - // Now we set up the modifier itself with the lens and processing flags - lfModifier *mod = new lfModifier(lens, cropFactor, width, height); - - int flags = 0; - if (prefilmParam.lensfunCA && !isMonochrome) { - flags |= LF_MODIFY_TCA; - } - if (prefilmParam.lensfunVignetting) { - flags |= LF_MODIFY_VIGNETTING; - } - if (prefilmParam.lensfunDistortion) { - flags |= LF_MODIFY_DISTORTION | LF_MODIFY_SCALE; - cout << "Auto scale factor: " << mod->GetAutoScale(false) << endl; - } - float scale = - (flags & LF_MODIFY_SCALE) ? mod->GetAutoScale(false) : 1.0f; - mod->Initialize(lens, LF_PF_F32, prefilmParam.focalLength, - prefilmParam.fnumber, 1000.0f, scale, LF_RECTILINEAR, - flags, false); - - // Now we actually perform the required processing. - // First is vignetting. - if (prefilmParam.lensfunVignetting) { - bool success = true; -#pragma omp parallel for - for (int row = 0; row < height; row++) { - success = mod->ApplyColorModification( - prefilm_input_image[row], 0.0f, row, width, 1, - LF_CR_3(RED, GREEN, BLUE), width); - } - } + if (!result) return emptyMatrix(); + RawImage &output = *result; - // Next is CA, or distortion, or both. - if (prefilmParam.lensfunCA || prefilmParam.lensfunDistortion) { - // ApplySubpixelGeometryDistortion - lensfunGeometryCorrectionApplied = true; - bool success = true; - int listWidth = width * 2 * 3; - - // Check how far out of bounds we go - float maxOvershootDistance = 1.0f; - float semiwidth = (width - 1) / 2.0f; - float semiheight = (height - 1) / 2.0f; -#pragma omp parallel for reduction(max : maxOvershootDistance) - for (int row = 0; row < height; row++) { - float positionList[listWidth]; - success = mod->ApplySubpixelGeometryDistortion(0.0f, row, width, 1, - positionList); - if (success) { - for (int col = 0; col < width; col++) { - int listIndex = col * 2 * 3; // list index - for (int c = 0; c < 3; c++) { - float coordX = positionList[listIndex + 2 * c] - semiwidth; - float coordY = - positionList[listIndex + 2 * c + 1] - semiheight; - float rotatedX = - coordX * cos(rotationAngle) - coordY * sin(rotationAngle); - float rotatedY = - coordX * sin(rotationAngle) + coordY * cos(rotationAngle); - - float overshoot = 1.0f; - - if (abs(rotatedX) > semiwidth) { - overshoot = max(abs(rotatedX) / semiwidth, overshoot); - } - if (abs(rotatedY) > semiheight) { - overshoot = max(abs(rotatedY) / semiheight, overshoot); - } - - if (overshoot > maxOvershootDistance) { - maxOvershootDistance = overshoot; - } - } - } - } - } - - pre_film_image.set_size(height, width * 3); -#pragma omp parallel for - for (int row = 0; row < height; row++) { - float positionList[listWidth]; - success = mod->ApplySubpixelGeometryDistortion(0.0f, row, width, 1, - positionList); - if (success) { - for (int col = 0; col < width; col++) { - int listIndex = col * 2 * 3; // list index - for (int c = 0; c < 3; c++) { - float coordX = positionList[listIndex + 2 * c] - semiwidth; - float coordY = - positionList[listIndex + 2 * c + 1] - semiheight; - float rotatedX = (coordX * cos(rotationAngle) - - coordY * sin(rotationAngle)) / - maxOvershootDistance + - semiwidth; - float rotatedY = (coordX * sin(rotationAngle) + - coordY * cos(rotationAngle)) / - maxOvershootDistance + - semiheight; - int sX = max(0, min(width - 1, int(floor(rotatedX)))) * 3 + - c; // startX - int eX = max(0, min(width - 1, int(ceil(rotatedX)))) * 3 + - c; // endX - int sY = - max(0, min(height - 1, int(floor(rotatedY)))); // startY - int eY = max(0, min(height - 1, int(ceil(rotatedY)))); // endY - float notUsed; - float eWX = modf(rotatedX, ¬Used); // end weight X - float eWY = modf(rotatedY, ¬Used); // end weight Y; - float sWX = 1 - eWX; // start weight X - float sWY = 1 - eWY; // start weight Y; - pre_film_image(row, col * 3 + c) = - prefilm_input_image(sY, sX) * sWY * sWX + - prefilm_input_image(eY, sX) * eWY * sWX + - prefilm_input_image(sY, eX) * sWY * eWX + - prefilm_input_image(eY, eX) * eWY * eWX; - } - } - } - } - } // else { - // demosaiced image isn't populated - // if geometry wasn't changed, then we'll move stuff over later. - //} - - if (mod != NULL) { - delete mod; - } - } - lf_free(lensList); - } - lf_free(cameraList); - - cout << "after lensfun " << endl; - - // also do rotations on non-corrected images - if (!lensfunGeometryCorrectionApplied) { - if (rotationAngle != 0.0f) { - float maxOvershootDistance = 1.0f; - float semiwidth = (width - 1) / 2.0f; - float semiheight = (height - 1) / 2.0f; - - // check the four corners - for (int row = 0; row < height; row += height - 1) { - for (int col = 0; col < width; col += width - 1) { - float coordX = col - semiwidth; - float coordY = row - semiheight; - float rotatedX = - coordX * cos(rotationAngle) - coordY * sin(rotationAngle); - float rotatedY = - coordX * sin(rotationAngle) + coordY * cos(rotationAngle); - - float overshoot = 1.0f; - - if (abs(rotatedX) > semiwidth) { - overshoot = max(abs(rotatedX) / semiwidth, overshoot); - } - if (abs(rotatedY) > semiheight) { - overshoot = max(abs(rotatedY) / semiheight, overshoot); - } - - if (overshoot > maxOvershootDistance) { - maxOvershootDistance = overshoot; - } - } - } + // Always save the full size image, so we can steal it later if needed. + pre_film_image.swap(output.data); - // Apply the rotation - pre_film_image.set_size(height, width * 3); - - for (int row = 0; row < height; row++) { - for (int col = 0; col < width; col++) { - float coordX = col - semiwidth; - float coordY = row - semiheight; - float rotatedX = - (coordX * cos(rotationAngle) - coordY * sin(rotationAngle)) / - maxOvershootDistance + - semiwidth; - float rotatedY = - (coordX * sin(rotationAngle) + coordY * cos(rotationAngle)) / - maxOvershootDistance + - semiheight; - int sX = max(0, min(width - 1, int(floor(rotatedX)))) * 3; // startX - int eX = max(0, min(width - 1, int(ceil(rotatedX)))) * 3; // endX - int sY = max(0, min(height - 1, int(floor(rotatedY)))); // startY - int eY = max(0, min(height - 1, int(ceil(rotatedY)))); // endY - float notUsed; - float eWX = modf(rotatedX, ¬Used); // end weight X - float eWY = modf(rotatedY, ¬Used); // end weight Y; - float sWX = 1 - eWX; // start weight X - float sWY = 1 - eWY; // start weight Y; - for (int c = 0; c < 3; c++) { - pre_film_image(row, col * 3 + c) = - prefilm_input_image(sY, sX + c) * sWY * sWX + - prefilm_input_image(eY, sX + c) * eWY * sWX + - prefilm_input_image(sY, eX + c) * sWY * eWX + - prefilm_input_image(eY, eX + c) * eWY * eWX; - } - } - } - } else { - // if we never rotate and never use lensfun geometry correction, we need - // to move the image over - pre_film_image.swap(prefilm_input_image); +#ifdef ENABLE_NAN_TRAPPING + { + float *pdata = pre_film_image; + if (pdata) { + float pmin = *std::min_element(pdata, pdata + pre_film_image.nr() * pre_film_image.nc()); + float pmax = *std::max_element(pdata, pdata + pre_film_image.nr() * pre_film_image.nc()); + FILM_TRACE("QuickPipe: pre_film_image min: {} max: {}", pmin, pmax); } } +#endif - // resize into a small image if (quality == LowQuality) { - cout << "thumbnail scale start:" << timeDiff(timeRequested) << endl; - auto downscale_time = std::chrono::steady_clock::now(); - downscale_and_crop(pre_film_image, pre_film_image_small, 0, 0, - (pre_film_image.nc() / 3) - 1, pre_film_image.nr() - 1, - 600, 600); - cout << "thumbnail scale end: " << timeDiff(downscale_time) << endl; + downscale_and_crop( + pre_film_image, pre_film_image_small, 0, 0, (pre_film_image.nc() / 3) - 1, pre_film_image.nr() - 1, 600, 600); } else if (quality == PreviewQuality) { - cout << "preview scale start:" << timeDiff(timeRequested) << endl; - auto downscale_time = std::chrono::steady_clock::now(); - // Make previews have same even/oddness as the source image - int paritywidth = - resolution + resolution % 2 + (pre_film_image.nc() / 3) % 2; - int parityheight = - resolution + resolution % 2 + (pre_film_image.nr()) % 2; - downscale_and_crop(pre_film_image, pre_film_image_small, 0, 0, - (pre_film_image.nc() / 3) - 1, pre_film_image.nr() - 1, - paritywidth, parityheight); - cout << "preview scale end: " << timeDiff(downscale_time) << endl; - } - - if (WithHisto == histo) { - // grab crop and rotation parameters - CropParams cropParam = paramManager->claimCropParams(); - cropHeight = cropParam.cropHeight; - cropAspect = cropParam.cropAspect; - cropHoffset = cropParam.cropHoffset; - cropVoffset = cropParam.cropVoffset; - rotation = cropParam.rotation; - histoInterface->updateHistPreFilm(pre_film_image, 65535, rotation, - cropHeight, cropAspect, cropHoffset, - cropVoffset); + int paritywidth = resolution + resolution % 2 + (pre_film_image.nc() / 3) % 2; + int parityheight = resolution + resolution % 2 + (pre_film_image.nr()) % 2; + downscale_and_crop(pre_film_image, + pre_film_image_small, + 0, + 0, + (pre_film_image.nc() / 3) - 1, + pre_film_image.nr() - 1, + paritywidth, + parityheight); + } else { + // High quality, we don't need the small image necessarily, but we can clear it or leave it. + // Let's clear it to save memory if we are doing full processing and don't need preview? + // Actually, existing logic cleared it. Use: + if (cache == NoCache) pre_film_image_small.set_size(0, 0); } - cout << "ImagePipeline::processImage: Prefilmulation complete." << endl; - valid = paramManager->markPrefilmComplete(); +#ifdef ENABLE_NAN_TRAPPING + SCAN_MATRIX_FOR_NAN(pre_film_image, "pre_film_image"); +#endif updateProgress(valid, 0.0f); [[fallthrough]]; } case partfilmulation: [[fallthrough]]; - case prefilmulation: // Do filmulation + case prefilmulation:// Do filmulation { matrix film_input_image; if (stealData) { - cout << "imagePipeline stealing data" << endl; + FILM_DEBUG("imagePipeline stealing data"); exifData = stealVictim->exifData; rCamMul = stealVictim->rCamMul; gCamMul = stealVictim->gCamMul; @@ -1623,17 +570,40 @@ matrix &ImagePipeline::processImage( } } for (int i = 0; i < 3; i++) { - for (int j = 0; j < 4; j++) { - camToRGB4[i][j] = stealVictim->camToRGB4[i][j]; - } + for (int j = 0; j < 4; j++) { camToRGB4[i][j] = stealVictim->camToRGB4[i][j]; } + } + matrix *sourceImage = &stealVictim->pre_film_image; + if (sourceImage->nr() == 0 && stealVictim->pre_film_image_small.nr() > 0) { + sourceImage = &stealVictim->pre_film_image_small; + FILM_DEBUG("Stealing from small image"); } + if (!isMonochrome) { - raw_to_sRGB(stealVictim->pre_film_image, film_input_image, camToRGB); + FILM_DEBUG("Stealing: sourceImage sizes: {}x{}", sourceImage->nr(), sourceImage->nc()); + float *sdata = *sourceImage; + if (sdata) { + float smin = *std::min_element(sdata, sdata + sourceImage->nr() * sourceImage->nc()); + float smax = *std::max_element(sdata, sdata + sourceImage->nr() * sourceImage->nc()); + FILM_DEBUG("Stealing: sourceImage min: {} max: {}", smin, smax); + } else { + FILM_ERROR("Stealing: sourceImage data is NULL"); + } + + FILM_DEBUG("Stealing: camToRGB[0][0]: {}", camToRGB[0][0]); + + raw_to_sRGB(*sourceImage, film_input_image, camToRGB); + + float *fdata = film_input_image; + if (fdata) { + float fmin = *std::min_element(fdata, fdata + film_input_image.nr() * film_input_image.nc()); + float fmax = *std::max_element(fdata, fdata + film_input_image.nr() * film_input_image.nc()); + FILM_DEBUG("Stealing: film_input_image min: {} max: {}", fmin, fmax); + } } else { - film_input_image = stealVictim->pre_film_image; + film_input_image = *sourceImage; } } else { - cout << "imagePipeline not stealing data" << endl; + FILM_DEBUG("imagePipeline not stealing data"); if (quality == LowQuality || quality == PreviewQuality) { // grab shrunken image if (!isMonochrome) { @@ -1658,9 +628,9 @@ matrix &ImagePipeline::processImage( } } - cout << "imagePipeline beginning filmulation" << endl; - cout << "imagePipeline image width: " << film_input_image.nc() / 3 << endl; - cout << "imagePipeline image height: " << film_input_image.nr() << endl; + FILM_DEBUG("imagePipeline beginning filmulation"); + FILM_DEBUG("imagePipeline image width: {}", film_input_image.nc() / 3); + FILM_DEBUG("imagePipeline image height: {}", film_input_image.nr()); // We don't need to check abort status out here, because // the filmulate function will do so inside its loop. @@ -1669,7 +639,7 @@ matrix &ImagePipeline::processImage( // Here we do the film simulation on the image... // If filmulate detects an abort, it returns true. if (filmulate(film_input_image, filmulated_image, paramManager, this)) { - cout << "imagePipeline aborted at filmulation" << endl; + FILM_WARN("imagePipeline aborted at filmulation"); return emptyMatrix(); } @@ -1677,11 +647,9 @@ matrix &ImagePipeline::processImage( // grab crop and rotation parameters CropParams cropParam = paramManager->claimCropParams(); bool updatePreFilm = false; - if (cropHeight != cropParam.cropHeight || - cropAspect != cropParam.cropAspect || - cropHoffset != cropParam.cropHoffset || - cropVoffset != cropParam.cropVoffset || - rotation != cropParam.rotation) { + if (cropHeight != cropParam.cropHeight || cropAspect != cropParam.cropAspect + || cropHoffset != cropParam.cropHoffset || cropVoffset != cropParam.cropVoffset + || rotation != cropParam.rotation) { updatePreFilm = true; } cropHeight = cropParam.cropHeight; @@ -1691,21 +659,23 @@ matrix &ImagePipeline::processImage( rotation = cropParam.rotation; if (updatePreFilm) { if (!stealData) { - histoInterface->updateHistPreFilm(pre_film_image, 65535, rotation, - cropHeight, cropAspect, cropHoffset, - cropVoffset); + histoInterface->updateHistPreFilm( + pre_film_image, 65535, rotation, cropHeight, cropAspect, cropHoffset, cropVoffset); } else { - histoInterface->updateHistPreFilm(stealVictim->pre_film_image, 65535, - rotation, cropHeight, cropAspect, - cropHoffset, cropVoffset); + histoInterface->updateHistPreFilm( + stealVictim->pre_film_image, 65535, rotation, cropHeight, cropAspect, cropHoffset, cropVoffset); } } - histoInterface->updateHistPostFilm( - filmulated_image, .0025f, // TODO connect this magic number to the qml - rotation, cropHeight, cropAspect, cropHoffset, cropVoffset); + histoInterface->updateHistPostFilm(filmulated_image, + .0025f,// TODO connect this magic number to the qml + rotation, + cropHeight, + cropAspect, + cropHoffset, + cropVoffset); } - cout << "ImagePipeline::processImage: Filmulation complete." << endl; + FILM_INFO("ImagePipeline::processImage: Filmulation complete."); valid = paramManager->markFilmComplete(); updateProgress(valid, 0.0f); @@ -1713,18 +683,17 @@ matrix &ImagePipeline::processImage( } case partblackwhite: [[fallthrough]]; - case filmulation: // Do whitepoint_blackpoint + case filmulation:// Do whitepoint_blackpoint { - cout << "imagePipeline beginning whitepoint blackpoint" << endl; - cout << "imagePipeline image width: " << filmulated_image.nc() / 3 << endl; - cout << "imagePipeline image height: " << filmulated_image.nr() << endl; + FILM_DEBUG("imagePipeline beginning whitepoint blackpoint"); + FILM_DEBUG("imagePipeline image width: {}", filmulated_image.nc() / 3); + FILM_DEBUG("imagePipeline image height: {}", filmulated_image.nr()); BlackWhiteParams blackWhiteParam; AbortStatus abort; - std::tie(valid, abort, blackWhiteParam) = - paramManager->claimBlackWhiteParams(); + std::tie(valid, abort, blackWhiteParam) = paramManager->claimBlackWhiteParams(); if (abort == AbortStatus::restart) { - cout << "imagePipeline aborted at whitepoint blackpoint" << endl; + FILM_WARN("imagePipeline aborted at whitepoint blackpoint"); return emptyMatrix(); } @@ -1732,11 +701,9 @@ matrix &ImagePipeline::processImage( if (WithHisto == histo) { // grab crop and rotation parameters bool updatePrePostFilm = false; - if (cropHeight != blackWhiteParam.cropHeight || - cropAspect != blackWhiteParam.cropAspect || - cropHoffset != blackWhiteParam.cropHoffset || - cropVoffset != blackWhiteParam.cropVoffset || - rotation != blackWhiteParam.rotation) { + if (cropHeight != blackWhiteParam.cropHeight || cropAspect != blackWhiteParam.cropAspect + || cropHoffset != blackWhiteParam.cropHoffset || cropVoffset != blackWhiteParam.cropVoffset + || rotation != blackWhiteParam.rotation) { updatePrePostFilm = true; } cropHeight = blackWhiteParam.cropHeight; @@ -1746,26 +713,27 @@ matrix &ImagePipeline::processImage( rotation = blackWhiteParam.rotation; if (updatePrePostFilm) { if (!stealData) { - histoInterface->updateHistPreFilm(pre_film_image, 65535, rotation, - cropHeight, cropAspect, cropHoffset, - cropVoffset); + histoInterface->updateHistPreFilm( + pre_film_image, 65535, rotation, cropHeight, cropAspect, cropHoffset, cropVoffset); } else { - histoInterface->updateHistPreFilm(stealVictim->pre_film_image, 65535, - rotation, cropHeight, cropAspect, - cropHoffset, cropVoffset); + histoInterface->updateHistPreFilm( + stealVictim->pre_film_image, 65535, rotation, cropHeight, cropAspect, cropHoffset, cropVoffset); } - histoInterface->updateHistPostFilm( - filmulated_image, - .0025f, // TODO connect this magic number to the qml - rotation, cropHeight, cropAspect, cropHoffset, cropVoffset); + histoInterface->updateHistPostFilm(filmulated_image, + .0025f,// TODO connect this magic number to the qml + rotation, + cropHeight, + cropAspect, + cropHoffset, + cropVoffset); } } matrix rotated_image; rotate_image(filmulated_image, rotated_image, blackWhiteParam.rotation); - if (NoCache == cache) // clean up ram that's not needed anymore in order to - // reduce peak consumption + if (NoCache == cache)// clean up ram that's not needed anymore in order to + // reduce peak consumption { filmulated_image.set_size(0, 0); cacheEmpty = true; @@ -1776,39 +744,28 @@ matrix &ImagePipeline::processImage( const int imWidth = rotated_image.nc() / 3; const int imHeight = rotated_image.nr(); - const float tempHeight = - imHeight * max(min(1.0f, blackWhiteParam.cropHeight), - 0.0f); // restrict domain to 0:1 + const float tempHeight = imHeight * max(min(1.0f, blackWhiteParam.cropHeight), + 0.0f);// restrict domain to 0:1 const float tempAspect = max(min(10000.0f, blackWhiteParam.cropAspect), - 0.0001f); // restrict aspect ratio + 0.0001f);// restrict aspect ratio int width = int(round(min(tempHeight * tempAspect, float(imWidth)))); int height = int(round(min(tempHeight, imWidth / tempAspect))); const float maxHoffset = (1.0f - (float(width) / float(imWidth))) / 2.0f; const float maxVoffset = (1.0f - (float(height) / float(imHeight))) / 2.0f; const float oddH = - (!(int(round((imWidth - width) / 2.0)) * 2 == (imWidth - width))) * - 0.5f; // it's 0.5 if it's odd, 0 otherwise + (!(int(round((imWidth - width) / 2.0)) * 2 == (imWidth - width))) * 0.5f;// it's 0.5 if it's odd, 0 otherwise const float oddV = - (!(int(round((imHeight - height) / 2.0)) * 2 == (imHeight - height))) * - 0.5f; // it's 0.5 if it's odd, 0 otherwise + (!(int(round((imHeight - height) / 2.0)) * 2 == (imHeight - height))) * 0.5f;// it's 0.5 if it's odd, 0 otherwise const float hoffset = - (round(max(min(blackWhiteParam.cropHoffset, maxHoffset), -maxHoffset) * - imWidth + - oddH) - - oddH) / - imWidth; + (round(max(min(blackWhiteParam.cropHoffset, maxHoffset), -maxHoffset) * imWidth + oddH) - oddH) / imWidth; const float voffset = - (round(max(min(blackWhiteParam.cropVoffset, maxVoffset), -maxVoffset) * - imHeight + - oddV) - - oddV) / - imHeight; + (round(max(min(blackWhiteParam.cropVoffset, maxVoffset), -maxVoffset) * imHeight + oddV) - oddV) / imHeight; int startX = int(round(0.5f * (imWidth - width) + hoffset * imWidth)); int startY = int(round(0.5f * (imHeight - height) + voffset * imHeight)); int endX = startX + width - 1; int endY = startY + height - 1; - if (blackWhiteParam.cropHeight <= 0) // it shall be turned off + if (blackWhiteParam.cropHeight <= 0)// it shall be turned off { startX = 0; startY = 0; @@ -1819,20 +776,20 @@ matrix &ImagePipeline::processImage( } matrix cropped_image; - cout << "crop start:" << timeDiff(timeRequested) << endl; + FILM_DEBUG("crop start: {}", timeDiff(timeRequested)); std::chrono::steady_clock::time_point crop_time; crop_time = std::chrono::steady_clock::now(); - downscale_and_crop(rotated_image, cropped_image, startX, startY, endX, endY, - width, height); + downscale_and_crop(rotated_image, cropped_image, startX, startY, endX, endY, width, height); - cout << "crop end: " << timeDiff(crop_time) << endl; + FILM_DEBUG("crop end: {}", timeDiff(crop_time)); - rotated_image.set_size(0, 0); // clean up ram that's not needed anymore + rotated_image.set_size(0, 0);// clean up ram that's not needed anymore - whitepoint_blackpoint(cropped_image, // filmulated_image, - contrast_image, blackWhiteParam.whitepoint, - blackWhiteParam.blackpoint); + whitepoint_blackpoint(cropped_image,// filmulated_image, + contrast_image, + blackWhiteParam.whitepoint, + blackWhiteParam.blackpoint); valid = paramManager->markBlackWhiteComplete(); updateProgress(valid, 0.0f); @@ -1840,9 +797,9 @@ matrix &ImagePipeline::processImage( } case partcolorcurve: [[fallthrough]]; - case blackwhite: // Do color_curve + case blackwhite:// Do color_curve { - cout << "imagePipeline beginning dummy color curve" << endl; + FILM_DEBUG("imagePipeline beginning dummy color curve"); // It's not gonna abort because we have no color curves yet.. // Prepare LUT's for individual color processin.g lutR.setUnity(); @@ -1863,23 +820,24 @@ matrix &ImagePipeline::processImage( } case partfilmlikecurve: [[fallthrough]]; - case colorcurve: // Do film-like curve + case colorcurve:// Do film-like curve { - cout << "imagePipeline beginning film like curve" << endl; + FILM_DEBUG("imagePipeline beginning film like curve"); FilmlikeCurvesParams curvesParam; AbortStatus abort; - std::tie(valid, abort, curvesParam) = - paramManager->claimFilmlikeCurvesParams(); + std::tie(valid, abort, curvesParam) = paramManager->claimFilmlikeCurvesParams(); if (abort == AbortStatus::restart) { - cout << "imagePipeline aborted at color curve" << endl; + FILM_WARN("imagePipeline aborted at color curve"); return emptyMatrix(); } filmLikeLUT.fill([=](unsigned short in) -> unsigned short { - float shResult = shadows_highlights( - float(in) / 65535.0f, curvesParam.shadowsX, curvesParam.shadowsY, - curvesParam.highlightsX, curvesParam.highlightsY); + float shResult = shadows_highlights(float(in) / 65535.0f, + curvesParam.shadowsX, + curvesParam.shadowsY, + curvesParam.highlightsX, + curvesParam.highlightsY); return ushort(65535 * default_tonecurve(shResult)); }); matrix &film_curve_image = vibrance_saturation_image; @@ -1894,38 +852,34 @@ matrix &ImagePipeline::processImage( } if (!curvesParam.monochrome) { - vibrance_saturation(film_curve_image, vibrance_saturation_image, - curvesParam.vibrance, curvesParam.saturation); + vibrance_saturation(film_curve_image, vibrance_saturation_image, curvesParam.vibrance, curvesParam.saturation); } else { - monochrome_convert(film_curve_image, vibrance_saturation_image, - curvesParam.bwRmult, curvesParam.bwGmult, - curvesParam.bwBmult); + monochrome_convert( + film_curve_image, vibrance_saturation_image, curvesParam.bwRmult, curvesParam.bwGmult, curvesParam.bwBmult); } updateProgress(valid, 0.0f); [[fallthrough]]; } - default: // output + default:// output { - cout << "imagePipeline beginning output" << endl; + FILM_DEBUG("imagePipeline beginning output"); if (NoCache == cache) { // vibrance_saturation_image.set_size(0, 0); cacheEmpty = true; } else { cacheEmpty = false; } - if (WithHisto == histo) { - histoInterface->updateHistFinal(vibrance_saturation_image); - } + if (WithHisto == histo) { histoInterface->updateHistFinal(vibrance_saturation_image); } valid = paramManager->markFilmLikeCurvesComplete(); updateProgress(valid, 0.0f); exifOutput = exifData; return vibrance_saturation_image; } - } // End task switch + }// End task switch - cout << "imagePipeline aborted at end" << endl; + FILM_WARN("imagePipeline aborted at end"); return emptyMatrix(); } @@ -1944,16 +898,15 @@ matrix &ImagePipeline::processImage( // } // } -void ImagePipeline::updateProgress(Valid valid, float stepProgress) { +void ImagePipeline::updateProgress(Valid valid, float stepProgress) +{ double totalTime = numeric_limits::epsilon(); double totalCompletedTime = 0; for (ulong i = 0; i < completionTimes.size(); i++) { totalTime += completionTimes[i]; float fractionCompleted = 0; - if (i <= valid) - fractionCompleted = 1; - if (i == valid + 1) - fractionCompleted = stepProgress; + if (i <= valid) fractionCompleted = 1; + if (i == valid + 1) fractionCompleted = stepProgress; // if greater -> 0 totalCompletedTime += completionTimes[i] * double(fractionCompleted); } @@ -1961,15 +914,15 @@ void ImagePipeline::updateProgress(Valid valid, float stepProgress) { } // Do not call this on something that's already been used! -void ImagePipeline::setCache(Cache cacheIn) { - if (false == hasStartedProcessing) { - cache = cacheIn; - } +void ImagePipeline::setCache(Cache cacheIn) +{ + if (false == hasStartedProcessing) { cache = cacheIn; } } // This swaps the data between pipelines. // The intended use is for preloading. -void ImagePipeline::swapPipeline(ImagePipeline *swapTarget) { +void ImagePipeline::swapPipeline(ImagePipeline *swapTarget) +{ std::swap(valid, swapTarget->valid); std::swap(progress, swapTarget->progress); @@ -2029,25 +982,21 @@ void ImagePipeline::swapPipeline(ImagePipeline *swapTarget) { } // This is used to update the histograms once data is copied on an image change -void ImagePipeline::rerunHistograms() { +void ImagePipeline::rerunHistograms() +{ if (WithHisto == histo) { if (valid >= Valid::load) { - histoInterface->updateHistRaw(raw_image, colorMaxValue, cfa, xtrans, - maxXtrans, isSraw, isMonochrome); + histoInterface->updateHistRaw(raw_image, colorMaxValue, cfa, xtrans, maxXtrans, isSraw, isMonochrome); } if (valid >= Valid::prefilmulation) { - histoInterface->updateHistPreFilm(pre_film_image, 65535, rotation, - cropHeight, cropAspect, cropHoffset, - cropVoffset); + histoInterface->updateHistPreFilm( + pre_film_image, 65535, rotation, cropHeight, cropAspect, cropHoffset, cropVoffset); } if (valid >= Valid::filmulation) { - histoInterface->updateHistPostFilm(filmulated_image, .0025f, rotation, - cropHeight, cropAspect, cropHoffset, - cropVoffset); - } - if (valid >= Valid::filmlikecurve) { - histoInterface->updateHistFinal(vibrance_saturation_image); + histoInterface->updateHistPostFilm( + filmulated_image, .0025f, rotation, cropHeight, cropAspect, cropHoffset, cropVoffset); } + if (valid >= Valid::filmlikecurve) { histoInterface->updateHistFinal(vibrance_saturation_image); } } } @@ -2055,11 +1004,17 @@ void ImagePipeline::rerunHistograms() { // square. // The square is positioned relative to the image dimensions of the cropped // image. -void ImagePipeline::sampleWB(const float xPos, const float yPos, - const int rotation, const float cropHeight, - const float cropAspect, const float cropVoffset, - const float cropHoffset, float &red, float &green, - float &blue) { +void ImagePipeline::sampleWB(const float xPos, + const float yPos, + const int rotation, + const float cropHeight, + const float cropAspect, + const float cropVoffset, + const float cropHoffset, + float &red, + float &green, + float &blue) +{ if (xPos < 0 || xPos > 1 || yPos < 0 || yPos > 1) { red = -1; green = -1; @@ -2080,34 +1035,24 @@ void ImagePipeline::sampleWB(const float xPos, const float yPos, const int imWidth = rotated_image.nc() / 3; const int imHeight = rotated_image.nr(); - const float tempHeight = - imHeight * max(min(1.0f, cropHeight), 0.0f); // restrict domain to 0:1 - const float tempAspect = - max(min(10000.0f, cropAspect), 0.0001f); // restrict aspect ratio + const float tempHeight = imHeight * max(min(1.0f, cropHeight), 0.0f);// restrict domain to 0:1 + const float tempAspect = max(min(10000.0f, cropAspect), 0.0001f);// restrict aspect ratio int width = int(round(min(tempHeight * tempAspect, float(imWidth)))); int height = int(round(min(tempHeight, imWidth / tempAspect))); const float maxHoffset = (1.0f - (float(width) / float(imWidth))) / 2.0f; const float maxVoffset = (1.0f - (float(height) / float(imHeight))) / 2.0f; const float oddH = - (!(int(round((imWidth - width) / 2.0)) * 2 == (imWidth - width))) * - 0.5f; // it's 0.5 if it's odd, 0 otherwise + (!(int(round((imWidth - width) / 2.0)) * 2 == (imWidth - width))) * 0.5f;// it's 0.5 if it's odd, 0 otherwise const float oddV = - (!(int(round((imHeight - height) / 2.0)) * 2 == (imHeight - height))) * - 0.5f; // it's 0.5 if it's odd, 0 otherwise - const float hoffset = - (round(max(min(cropHoffset, maxHoffset), -maxHoffset) * imWidth + oddH) - - oddH) / - imWidth; - const float voffset = - (round(max(min(cropVoffset, maxVoffset), -maxVoffset) * imHeight + oddV) - - oddV) / - imHeight; + (!(int(round((imHeight - height) / 2.0)) * 2 == (imHeight - height))) * 0.5f;// it's 0.5 if it's odd, 0 otherwise + const float hoffset = (round(max(min(cropHoffset, maxHoffset), -maxHoffset) * imWidth + oddH) - oddH) / imWidth; + const float voffset = (round(max(min(cropVoffset, maxVoffset), -maxVoffset) * imHeight + oddV) - oddV) / imHeight; int startX = int(round(0.5f * (imWidth - width) + hoffset * imWidth)); int startY = int(round(0.5f * (imHeight - height) + voffset * imHeight)); int endX = startX + width - 1; int endY = startY + height - 1; - if (cropHeight <= 0) // it shall be turned off + if (cropHeight <= 0)// it shall be turned off { startX = 0; startY = 0; @@ -2119,8 +1064,7 @@ void ImagePipeline::sampleWB(const float xPos, const float yPos, matrix cropped_image; - downscale_and_crop(rotated_image, cropped_image, startX, startY, endX, endY, - width, height); + downscale_and_crop(rotated_image, cropped_image, startX, startY, endX, endY, width, height); rotated_image.set_size(0, 0); @@ -2144,7 +1088,7 @@ void ImagePipeline::sampleWB(const float xPos, const float yPos, count++; } } - if (count < 1) // some sort of error occurs + if (count < 1)// some sort of error occurs { red = -1; green = -1; @@ -2156,44 +1100,25 @@ void ImagePipeline::sampleWB(const float xPos, const float yPos, red = rSum / (rUserMul * count); green = gSum / (gUserMul * count); blue = bSum / (bUserMul * count); - cout << "custom WB sampled r: " << red << endl; - cout << "custom WB sampled g: " << green << endl; - cout << "custom WB sampled b: " << blue << endl; + FILM_DEBUG("custom WB sampled r: {}", red); + FILM_DEBUG("custom WB sampled g: {}", green); + FILM_DEBUG("custom WB sampled b: {}", blue); } -void ImagePipeline::clearInvalid(Valid validIn) { - if (validIn < load) { - raw_image.set_size(0, 0); - } - if (validIn < demosaic) { - demosaiced_image.set_size(0, 0); - } - if (validIn < postdemosaic) { - post_demosaic_image.set_size(0, 0); - } - if (validIn < nrnlmeans) { - nlmeans_nr_image.set_size(0, 0); - } - if (validIn < nrimpulse) { - impulse_nr_image.set_size(0, 0); - } - if (validIn < nrchroma) { - chroma_nr_image.set_size(0, 0); - } +void ImagePipeline::clearInvalid(Valid validIn) +{ + if (validIn < load) { raw_image.set_size(0, 0); } + if (validIn < demosaic) { demosaiced_image.set_size(0, 0); } + if (validIn < postdemosaic) { post_demosaic_image.set_size(0, 0); } + if (validIn < nrnlmeans) { nlmeans_nr_image.set_size(0, 0); } + if (validIn < nrimpulse) { impulse_nr_image.set_size(0, 0); } + if (validIn < nrchroma) { chroma_nr_image.set_size(0, 0); } if (validIn < prefilmulation) { pre_film_image.set_size(0, 0); pre_film_image_small.set_size(0, 0); } - if (validIn < filmulation) { - filmulated_image.set_size(0, 0); - } - if (validIn < blackwhite) { - contrast_image.set_size(0, 0); - } - if (validIn < colorcurve) { - color_curve_image.set_size(0, 0); - } - if (validIn < filmlikecurve) { - vibrance_saturation_image.set_size(0, 0); - } + if (validIn < filmulation) { filmulated_image.set_size(0, 0); } + if (validIn < blackwhite) { contrast_image.set_size(0, 0); } + if (validIn < colorcurve) { color_curve_image.set_size(0, 0); } + if (validIn < filmlikecurve) { vibrance_saturation_image.set_size(0, 0); } } diff --git a/filmulator-gui/core/imagePipeline.h b/filmulator-gui/core/imagePipeline.h index ad4368f4..ad6c3307 100644 --- a/filmulator-gui/core/imagePipeline.h +++ b/filmulator-gui/core/imagePipeline.h @@ -6,11 +6,12 @@ #include #include -enum Cache { HighCache, NoCache }; -enum Histo { WithHisto, NoHisto }; -enum QuickQuality { LowQuality, PreviewQuality, HighQuality }; +#include "pipeline/PipelineContext.h" -class ImagePipeline { +using namespace Pipeline; + +class ImagePipeline +{ public: ImagePipeline(Cache, Histo, QuickQuality); ~ImagePipeline(); @@ -18,10 +19,10 @@ class ImagePipeline { // Loads and processes an image according to the 'params' structure, // monitoring 'aborted' for cancellation. matrix &processImage(ParameterManager *paramManager, - Interface *histoInterface, - Exiv2::ExifData &exifOutput, - const QString fileHash, - ImagePipeline *stealVictim = nullptr); + Interface *histoInterface, + Exiv2::ExifData &exifOutput, + const QString fileHash, + ImagePipeline *stealVictim = nullptr); // Returns the progress of the pipeline from 0, incomplete, to 1, complete. float getProgress() { return progress; } @@ -52,10 +53,15 @@ class ImagePipeline { // Sample the image and return the average level of each channel void sampleWB(const float xPos, - const float yPos, // relative to the rotated and cropped image - const int rotation, const float cropHeight, - const float cropAspect, const float cropVoffset, - const float cropHoffset, float &red, float &green, float &blue); + const float yPos,// relative to the rotated and cropped image + const int rotation, + const float cropHeight, + const float cropAspect, + const float cropVoffset, + const float cropHoffset, + float &red, + float &green, + float &blue); // The resolution of a quick preview int resolution; @@ -91,30 +97,30 @@ class ImagePipeline { unsigned xtrans[6][6]; int maxXtrans; int raw_width, raw_height; - float camToRGB[3][3]; // rgb_cam from libraw - float xyzToCam[3][3]; // cam_xyz from libraw + float camToRGB[3][3];// rgb_cam from libraw + float xyzToCam[3][3];// cam_xyz from libraw float camToRGB4[3][4]; - float rCamMul, gCamMul, bCamMul; // wb used on the image by the camera - float rPreMul, gPreMul, bPreMul; //"daylight" wb according to libraw - float rUserMul, gUserMul, bUserMul; // wb actually applied + float rCamMul, gCamMul, bCamMul;// wb used on the image by the camera + float rPreMul, gPreMul, bPreMul;//"daylight" wb according to libraw + float rUserMul, gUserMul, bUserMul;// wb actually applied float maxValue; float colorMaxValue[3]; - bool isSraw; // Actually we should set this for all full-color raws (including - // X-Transformer) + bool isSraw;// Actually we should set this for all full-color raws (including + // X-Transformer) bool isNikonSraw; bool isMonochrome; bool isCR3; - matrix demosaiced_image; // raw - matrix post_demosaic_image; // raw - matrix nlmeans_nr_image; // lab - matrix impulse_nr_image; // lab - matrix chroma_nr_image; // lab - matrix pre_film_image; // back to raw - matrix pre_film_image_small; // + matrix demosaiced_image;// raw + matrix post_demosaic_image;// raw + matrix nlmeans_nr_image;// lab + matrix impulse_nr_image;// lab + matrix chroma_nr_image;// lab + matrix pre_film_image;// back to raw + matrix pre_film_image_small;// Exiv2::ExifData exifData; - Exiv2::ExifData basicExifData; // for tiff writing - matrix filmulated_image; // sRGB + Exiv2::ExifData basicExifData;// for tiff writing + matrix filmulated_image;// sRGB matrix contrast_image; matrix color_curve_image; matrix vibrance_saturation_image; @@ -132,12 +138,13 @@ class ImagePipeline { // The core filmulation. It needs to access ProcessingParameters, so it's // here. - bool filmulate(matrix &scaled_image, matrix &output_density, - ParameterManager *paramManager, ImagePipeline *pipeline); + bool filmulate(matrix &scaled_image, + matrix &output_density, + ParameterManager *paramManager, + ImagePipeline *pipeline); // Callback for LibRaw cancellation - static int libraw_callback(void *data, enum LibRaw_progress p, int iteration, - int expected); + static int libraw_callback(void *data, enum LibRaw_progress p, int iteration, int expected); }; -#endif // IMAGEPIPELINE_H +#endif// IMAGEPIPELINE_H diff --git a/filmulator-gui/core/imload.cpp b/filmulator-gui/core/imload.cpp index eb3872a8..e224863d 100644 --- a/filmulator-gui/core/imload.cpp +++ b/filmulator-gui/core/imload.cpp @@ -1,4 +1,4 @@ -/* +/* * This file is part of Filmulator. * * Copyright 2013 Omer Mano and Carlo Vaccari @@ -20,47 +20,36 @@ // This file contains the function imload, which calls an image retrieval // function and loads the data into a matrix. -//TODO: remove this file -//PROBABLY NOT NECESSARY ANYMORE -//We've included this code into ImagePipeline now. +// TODO: remove this file +// PROBABLY NOT NECESSARY ANYMORE +// We've included this code into ImagePipeline now. #include "filmSim.hpp" bool imload(std::string filename, - matrix &input_image, - bool tiff, - bool jpeg_in, - Exiv2::ExifData &exifData, - int highlights, - bool caEnabled, - bool lowQuality ) + matrix &input_image, + bool tiff, + bool jpeg_in, + Exiv2::ExifData &exifData, + int highlights, + bool caEnabled, + bool lowQuality) { - if(tiff) - { - if(imread_tiff(filename, input_image, exifData)) - { - cerr << "Could not open image " << filename << - "; Exiting..." << endl; - return true; - } + if (tiff) { + if (imread_tiff(filename, input_image, exifData)) { + cerr << "Could not open image " << filename << "; Exiting..." << endl; + return true; } - else if(jpeg_in) - { - if(imread_jpeg(filename, input_image, exifData)) - { - cerr << "Could not open image " << filename << - "; Exiting..." << endl; - return true; - } + } else if (jpeg_in) { + if (imread_jpeg(filename, input_image, exifData)) { + cerr << "Could not open image " << filename << "; Exiting..." << endl; + return true; } - else//raw - { - if( imread(filename, input_image, exifData, highlights, - caEnabled, lowQuality)) - { - cerr << "Could not open image " << filename << - "; Exiting..." << endl; - return true; - } + } else// raw + { + if (imread(filename, input_image, exifData, highlights, caEnabled, lowQuality)) { + cerr << "Could not open image " << filename << "; Exiting..." << endl; + return true; } - return false; + } + return false; } diff --git a/filmulator-gui/core/imread.cpp b/filmulator-gui/core/imread.cpp index 7a97cb6f..b05497f4 100644 --- a/filmulator-gui/core/imread.cpp +++ b/filmulator-gui/core/imread.cpp @@ -1,4 +1,4 @@ -/* +/* * This file is part of Filmulator. * * Copyright 2013 Omer Mano and Carlo Vaccari @@ -17,79 +17,79 @@ * along with Filmulator. If not, see */ -//imread.cpp uses libraw to load raw files. -//TODO: remove this file -//PROBABLY NOT NECESSARY ANYMORE -//We've included the loader into ImagePipeline now. +// imread.cpp uses libraw to load raw files. +// TODO: remove this file +// PROBABLY NOT NECESSARY ANYMORE +// We've included the loader into ImagePipeline now. #include "filmSim.hpp" -bool imread(std::string input_image_filename, matrix &returnmatrix, - Exiv2::ExifData &exifData, int highlights, bool /*caEnabled*/, bool lowQuality) +bool imread(std::string input_image_filename, + matrix &returnmatrix, + Exiv2::ExifData &exifData, + int highlights, + bool /*caEnabled*/, + bool lowQuality) { - //Create image processor for reading raws. - LibRaw image_processor; + // Create image processor for reading raws. + LibRaw image_processor; - //Open the file. - const char *cstr = input_image_filename.c_str(); - if (0 != image_processor.open_file(cstr)) - { - cerr << "imread: Could not read input file!" << endl; - return true; - } - //Make abbreviations for brevity in accessing data. + // Open the file. + const char *cstr = input_image_filename.c_str(); + if (0 != image_processor.open_file(cstr)) { + cerr << "imread: Could not read input file!" << endl; + return true; + } + // Make abbreviations for brevity in accessing data. #define SIZES image_processor.imgdata.sizes #define PARAM image_processor.imgdata.params #define IMAGE image_processor.imgdata.image #define COLOR image_processor.imgdata.color - //Now we'll set demosaic and other processing settings. - PARAM.no_auto_bright = 1;//Don't autoadjust brightness (-W) - PARAM.output_bps = 16;//16 bits per channel (-6) - PARAM.gamm[0] = 1; - PARAM.gamm[1] = 1;//Linear gamma (-g 1 1) - PARAM.output_color = 1;//1: Use sRGB regardless. - PARAM.use_camera_wb = 1;//1: Use camera WB setting (-w) - PARAM.highlight = highlights;//Set highlight recovery (-H #) - PARAM.med_passes = 1;//median filter + // Now we'll set demosaic and other processing settings. + PARAM.no_auto_bright = 1;// Don't autoadjust brightness (-W) + PARAM.output_bps = 16;// 16 bits per channel (-6) + PARAM.gamm[0] = 1; + PARAM.gamm[1] = 1;// Linear gamma (-g 1 1) + PARAM.output_color = 1;// 1: Use sRGB regardless. + PARAM.use_camera_wb = 1;// 1: Use camera WB setting (-w) + PARAM.highlight = highlights;// Set highlight recovery (-H #) + PARAM.med_passes = 1;// median filter - if (lowQuality) - { - //PARAM.half_size = 1;//half-size output, should dummy down demosaic. - /* The above sometimes read out a dng thumbnail instead of the image itself. */ - PARAM.user_qual = 0;//nearest-neighbor demosaic - } + if (lowQuality) { + // PARAM.half_size = 1;//half-size output, should dummy down demosaic. + /* The above sometimes read out a dng thumbnail instead of the image itself. */ + PARAM.user_qual = 0;// nearest-neighbor demosaic + } - //This makes IMAGE contains the sensel value and 3 blank values at every - //location. - image_processor.unpack(); + // This makes IMAGE contains the sensel value and 3 blank values at every + // location. + image_processor.unpack(); - //This calls the dcraw processing on the raw sensel data. - //Now, it contains 3 color values and one blank value at every location. - //We will ignore the last blank value. - image_processor.dcraw_process(); + // This calls the dcraw processing on the raw sensel data. + // Now, it contains 3 color values and one blank value at every location. + // We will ignore the last blank value. + image_processor.dcraw_process(); - long rSum = 0, gSum = 0, bSum = 0; - returnmatrix.set_size(SIZES.iheight, SIZES.iwidth*3); - for (int row = 0; row < SIZES.iheight; row++) - { - //IMAGE is an (width*height) by 4 array, not width by height by 4. - int rowoffset = row*SIZES.iwidth; - for (int col = 0; col < SIZES.iwidth; col++) - { - returnmatrix(row, col*3 ) = IMAGE[rowoffset + col][0];//R - returnmatrix(row, col*3 + 1) = IMAGE[rowoffset + col][1];//G - returnmatrix(row, col*3 + 2) = IMAGE[rowoffset + col][2];//B - rSum += IMAGE[rowoffset + col][0]; - gSum += IMAGE[rowoffset + col][1]; - bSum += IMAGE[rowoffset + col][2]; - } + long rSum = 0, gSum = 0, bSum = 0; + returnmatrix.set_size(SIZES.iheight, SIZES.iwidth * 3); + for (int row = 0; row < SIZES.iheight; row++) { + // IMAGE is an (width*height) by 4 array, not width by height by 4. + int rowoffset = row * SIZES.iwidth; + for (int col = 0; col < SIZES.iwidth; col++) { + returnmatrix(row, col * 3) = IMAGE[rowoffset + col][0];// R + returnmatrix(row, col * 3 + 1) = IMAGE[rowoffset + col][1];// G + returnmatrix(row, col * 3 + 2) = IMAGE[rowoffset + col][2];// B + rSum += IMAGE[rowoffset + col][0]; + gSum += IMAGE[rowoffset + col][1]; + bSum += IMAGE[rowoffset + col][2]; } - image_processor.recycle(); - cout << "imread exiv filename: " << input_image_filename << endl; - auto image = Exiv2::ImageFactory::open(input_image_filename); - assert(image.get() != 0); - image->readMetadata(); - exifData = image->exifData(); + } + image_processor.recycle(); + cout << "imread exiv filename: " << input_image_filename << endl; + auto image = Exiv2::ImageFactory::open(input_image_filename); + assert(image.get() != 0); + image->readMetadata(); + exifData = image->exifData(); - return false; + return false; } diff --git a/filmulator-gui/core/imreadJpeg.cpp b/filmulator-gui/core/imreadJpeg.cpp index 6e14baab..27bdab66 100644 --- a/filmulator-gui/core/imreadJpeg.cpp +++ b/filmulator-gui/core/imreadJpeg.cpp @@ -1,4 +1,4 @@ -/* +/* * This file is part of Filmulator. * * Copyright 2013 Omer Mano and Carlo Vaccari @@ -17,68 +17,58 @@ * along with Filmulator. If not, see */ #include "filmSim.hpp" +#include "logging.h" -bool imread_jpeg(string input_image_filename, matrix &returnmatrix, - Exiv2::ExifData &exifData) +bool imread_jpeg(string input_image_filename, matrix &returnmatrix, Exiv2::ExifData &exifData) { - struct jpeg_decompress_struct cinfo; - struct jpeg_error_mgr jerr; - JSAMPROW row_pointer[1]; - FILE* jpeg = fopen(input_image_filename.c_str(), "r"); - if (!jpeg) - { - cerr << "imread_jpeg: Could not read input file!" << endl; - return true; - } - /* here we set up the standard libjpeg error handler */ - cinfo.err = jpeg_std_error( &jerr ); - /* setup decompression process and source, then read JPEG header */ - jpeg_create_decompress( &cinfo ); - /* this makes the library read from infile */ - jpeg_stdio_src( &cinfo, jpeg ); - /* reading the image header which contains image information */ - jpeg_read_header( &cinfo, TRUE ); - int imagelength = cinfo.image_height; - int imagewidth = cinfo.image_width; - int num_chan = 3; - /* Start decompression jpeg here */ - jpeg_start_decompress( &cinfo ); - /* Allocate resulting image */ - returnmatrix.set_size(imagelength,imagewidth*num_chan); - /* now actually read the jpeg into the raw buffer */ - row_pointer[0] = - (unsigned char *)malloc( cinfo.output_width*cinfo.num_components ); - float read_in; //Contains a 0 to 1 value from the jpeg - /* read one scan line at a time */ - for (int row = 0; row < imagelength; row++) - { - jpeg_read_scanlines( &cinfo, row_pointer, 1 ); - for(int col = 0; col < imagewidth; col++) - { - for(int channel = 0; channel < 3; channel++) - { - read_in = float(row_pointer[0][col*num_chan + channel])/255; - /*Apply reverse sRGB gamma transformation: - http://en.wikipedia.org/wiki/SRGB#The_reverse_transformation*/ - if( read_in < 0.04045) - { - returnmatrix(row,col*num_chan + channel) = - read_in*5072.36842105; //(2^16 -1)/12.92 - } - else - { - returnmatrix(row,col*num_chan + channel) = - 65535*pow((read_in+0.055)/1.055, 2.4); - } - } - } - } + struct jpeg_decompress_struct cinfo; + struct jpeg_error_mgr jerr; + JSAMPROW row_pointer[1]; + FILE *jpeg = fopen(input_image_filename.c_str(), "r"); + if (!jpeg) { + FILM_ERROR("imread_jpeg: Could not read input file!"); + return true; + } + /* here we set up the standard libjpeg error handler */ + cinfo.err = jpeg_std_error(&jerr); + /* setup decompression process and source, then read JPEG header */ + jpeg_create_decompress(&cinfo); + /* this makes the library read from infile */ + jpeg_stdio_src(&cinfo, jpeg); + /* reading the image header which contains image information */ + jpeg_read_header(&cinfo, TRUE); + int imagelength = cinfo.image_height; + int imagewidth = cinfo.image_width; + int num_chan = 3; + /* Start decompression jpeg here */ + jpeg_start_decompress(&cinfo); + /* Allocate resulting image */ + returnmatrix.set_size(imagelength, imagewidth * num_chan); + /* now actually read the jpeg into the raw buffer */ + row_pointer[0] = (unsigned char *)malloc(cinfo.output_width * cinfo.num_components); + float read_in;// Contains a 0 to 1 value from the jpeg + /* read one scan line at a time */ + for (int row = 0; row < imagelength; row++) { + jpeg_read_scanlines(&cinfo, row_pointer, 1); + for (int col = 0; col < imagewidth; col++) { + for (int channel = 0; channel < 3; channel++) { + read_in = float(row_pointer[0][col * num_chan + channel]) / 255; + /*Apply reverse sRGB gamma transformation: + http://en.wikipedia.org/wiki/SRGB#The_reverse_transformation*/ + if (read_in < 0.04045) { + returnmatrix(row, col * num_chan + channel) = read_in * 5072.36842105;//(2^16 -1)/12.92 + } else { + returnmatrix(row, col * num_chan + channel) = 65535 * pow((read_in + 0.055) / 1.055, 2.4); + } + } + } + } - cout << "imread_jpeg exiv filename: " << input_image_filename << endl; - auto image = Exiv2::ImageFactory::open(input_image_filename); - assert(image.get() != 0); - image->readMetadata(); - exifData = image->exifData(); + FILM_TRACE("imread_jpeg exiv filename: {}", input_image_filename); + auto image = Exiv2::ImageFactory::open(input_image_filename); + assert(image.get() != 0); + image->readMetadata(); + exifData = image->exifData(); - return false; + return false; } diff --git a/filmulator-gui/core/imreadTiff.cpp b/filmulator-gui/core/imreadTiff.cpp index 544d1e73..e240f66b 100644 --- a/filmulator-gui/core/imreadTiff.cpp +++ b/filmulator-gui/core/imreadTiff.cpp @@ -1,4 +1,4 @@ -/* +/* * This file is part of Filmulator. * * Copyright 2013 Omer Mano and Carlo Vaccari @@ -18,71 +18,62 @@ */ #include "filmSim.hpp" -bool imread_tiff(string input_image_filename, matrix &returnmatrix, - Exiv2::ExifData &exifData) +bool imread_tiff(string input_image_filename, matrix &returnmatrix, Exiv2::ExifData &exifData) { - TIFFSetWarningHandler(NULL); - TIFF* tif = TIFFOpen(input_image_filename.c_str(), "r"); - if (!tif) - { - cerr << "imread_tiff: Could not read input file!" << endl; - return true; - } - uint32 imagelength; - uint32 imagewidth; - uint16 num_chan;//number of color channels - unsigned short * buf16; - unsigned char * buf8; - uint16 bits_per_sample; + TIFFSetWarningHandler(NULL); + TIFF *tif = TIFFOpen(input_image_filename.c_str(), "r"); + if (!tif) { + cerr << "imread_tiff: Could not read input file!" << endl; + return true; + } + uint32 imagelength; + uint32 imagewidth; + uint16 num_chan;// number of color channels + unsigned short *buf16; + unsigned char *buf8; + uint16 bits_per_sample; - TIFFGetField(tif, TIFFTAG_IMAGELENGTH, &imagelength); - TIFFGetField(tif, TIFFTAG_IMAGEWIDTH, &imagewidth); - TIFFGetField(tif, TIFFTAG_SAMPLESPERPIXEL, &num_chan); - TIFFGetField(tif, TIFFTAG_BITSPERSAMPLE, &bits_per_sample); + TIFFGetField(tif, TIFFTAG_IMAGELENGTH, &imagelength); + TIFFGetField(tif, TIFFTAG_IMAGEWIDTH, &imagewidth); + TIFFGetField(tif, TIFFTAG_SAMPLESPERPIXEL, &num_chan); + TIFFGetField(tif, TIFFTAG_BITSPERSAMPLE, &bits_per_sample); - returnmatrix.set_size(imagelength,imagewidth*3); + returnmatrix.set_size(imagelength, imagewidth * 3); - //The matrix is 3x wider than the image, interleaving the channels. - if(bits_per_sample == 16) - { - buf16 = (unsigned short *)_TIFFmalloc(TIFFScanlineSize(tif)); - for ( unsigned int row = 0; row < imagelength; row++) - { - TIFFReadScanline(tif, buf16, row); - for( unsigned int col = 0; col < imagewidth; col++) - { - returnmatrix(row,col*3 ) = buf16[col*num_chan ]; - returnmatrix(row,col*3 + 1) = buf16[col*num_chan + 1]; - returnmatrix(row,col*3 + 2) = buf16[col*num_chan + 2]; - } - } - _TIFFfree(buf16); - } - else - { - buf8 = (unsigned char *)_TIFFmalloc(TIFFScanlineSize(tif)); - for ( unsigned int row = 0; row < imagelength; row++) - { - TIFFReadScanline(tif, buf8, row); - for( unsigned int col = 0; col < imagewidth; col++) - { - returnmatrix(row,col*3 ) = buf8[col*num_chan ]; - returnmatrix(row,col*3 ) *= 257; - returnmatrix(row,col*3 + 1) = buf8[col*num_chan + 1]; - returnmatrix(row,col*3 + 1) *= 257; - returnmatrix(row,col*3 + 2) = buf8[col*num_chan + 2]; - returnmatrix(row,col*3 + 2) *= 257; - } - } - _TIFFfree(buf8 ); - } - TIFFClose(tif); + // The matrix is 3x wider than the image, interleaving the channels. + if (bits_per_sample == 16) { + buf16 = (unsigned short *)_TIFFmalloc(TIFFScanlineSize(tif)); + for (unsigned int row = 0; row < imagelength; row++) { + TIFFReadScanline(tif, buf16, row); + for (unsigned int col = 0; col < imagewidth; col++) { + returnmatrix(row, col * 3) = buf16[col * num_chan]; + returnmatrix(row, col * 3 + 1) = buf16[col * num_chan + 1]; + returnmatrix(row, col * 3 + 2) = buf16[col * num_chan + 2]; + } + } + _TIFFfree(buf16); + } else { + buf8 = (unsigned char *)_TIFFmalloc(TIFFScanlineSize(tif)); + for (unsigned int row = 0; row < imagelength; row++) { + TIFFReadScanline(tif, buf8, row); + for (unsigned int col = 0; col < imagewidth; col++) { + returnmatrix(row, col * 3) = buf8[col * num_chan]; + returnmatrix(row, col * 3) *= 257; + returnmatrix(row, col * 3 + 1) = buf8[col * num_chan + 1]; + returnmatrix(row, col * 3 + 1) *= 257; + returnmatrix(row, col * 3 + 2) = buf8[col * num_chan + 2]; + returnmatrix(row, col * 3 + 2) *= 257; + } + } + _TIFFfree(buf8); + } + TIFFClose(tif); - cout << "imread_tiff exiv filename: " << input_image_filename << endl; - auto image = Exiv2::ImageFactory::open(input_image_filename); - assert(image.get() != 0); - image->readMetadata(); - exifData = image->exifData(); + cout << "imread_tiff exiv filename: " << input_image_filename << endl; + auto image = Exiv2::ImageFactory::open(input_image_filename); + assert(image.get() != 0); + image->readMetadata(); + exifData = image->exifData(); - return false; + return false; } diff --git a/filmulator-gui/core/imwrite.cpp b/filmulator-gui/core/imwrite.cpp index ea0d7156..9f792898 100644 --- a/filmulator-gui/core/imwrite.cpp +++ b/filmulator-gui/core/imwrite.cpp @@ -1,4 +1,4 @@ -/* +/* * This file is part of Filmulator. * * Copyright 2013 Omer Mano and Carlo Vaccari @@ -17,10 +17,15 @@ * along with Filmulator. If not, see */ #include "filmSim.hpp" +#include "logging.h" -bool ppmb_write_data ( ofstream &output, int xsize, int ysize, - matrix &densityr, matrix &densityg, matrix &densityb, - bool sixteen_bit ) +bool ppmb_write_data(ofstream &output, + int xsize, + int ysize, + matrix &densityr, + matrix &densityg, + matrix &densityb, + bool sixteen_bit) //****************************************************************************80 // @@ -53,83 +58,77 @@ bool ppmb_write_data ( ofstream &output, int xsize, int ysize, // Output, bool PPMB_WRITE_DATA, is true if an error occurred. // { - int i; - int j; - unsigned short out; - unsigned char out2; - - if(sixteen_bit) - { - for ( j = 0; j < ysize; j++ ) - { - for ( i = 0; i < xsize; i++ ) - { - if(densityr(j,i) > 65535) - out = 65535; //clip to white - else if(densityr(j,i) < 0) - out = 0; //clip to black - else - out = (unsigned short) densityr(j,i); //normal values - out = ((out & 0x00ff)<<8)|((out & 0xff00)>>8); - output.write(reinterpret_cast(&out), sizeof(unsigned short)); - - if(densityg(j,i) > 65535) - out = 65535; //clip to white - else if(densityg(j,i) < 0) - out = 0; //clip to black - else - out = (unsigned short) densityg(j,i); //normal values - out = ((out & 0x00ff)<<8)|((out & 0xff00)>>8); - output.write(reinterpret_cast(&out), sizeof(unsigned short)); - - if(densityb(j,i) > 65535) - out = 65535; //clip to white - else if(densityb(j,i) < 0) - out = 0; //clip to black - else - out = (unsigned short) densityb(j,i); //normal values - out = ((out & 0x00ff)<<8)|((out & 0xff00)>>8); - output.write(reinterpret_cast(&out), sizeof(unsigned short)); - } + int i; + int j; + unsigned short out; + unsigned char out2; + + if (sixteen_bit) { + for (j = 0; j < ysize; j++) { + for (i = 0; i < xsize; i++) { + if (densityr(j, i) > 65535) + out = 65535;// clip to white + else if (densityr(j, i) < 0) + out = 0;// clip to black + else + out = (unsigned short)densityr(j, i);// normal values + out = ((out & 0x00ff) << 8) | ((out & 0xff00) >> 8); + output.write(reinterpret_cast(&out), sizeof(unsigned short)); + + if (densityg(j, i) > 65535) + out = 65535;// clip to white + else if (densityg(j, i) < 0) + out = 0;// clip to black + else + out = (unsigned short)densityg(j, i);// normal values + out = ((out & 0x00ff) << 8) | ((out & 0xff00) >> 8); + output.write(reinterpret_cast(&out), sizeof(unsigned short)); + + if (densityb(j, i) > 65535) + out = 65535;// clip to white + else if (densityb(j, i) < 0) + out = 0;// clip to black + else + out = (unsigned short)densityb(j, i);// normal values + out = ((out & 0x00ff) << 8) | ((out & 0xff00) >> 8); + output.write(reinterpret_cast(&out), sizeof(unsigned short)); } } - else //8 bit - { - for ( j = 0; j < ysize; j++ ) - { - for ( i = 0; i < xsize; i++ ) - { - if(densityr(j,i) > 255) - out2 = 255; //clip to white - else if(densityr(j,i) < 0) - out2 = 0; //clip to black - else - out2 = (unsigned char) densityr(j,i); //normal values - output.write(reinterpret_cast(&out2), sizeof(unsigned char)); - - if(densityg(j,i) > 255) - out2 = 255; //clip to white - else if(densityg(j,i) < 0) - out2 = 0; //clip to black - else - out2 = (unsigned char) densityg(j,i); //normal values - output.write(reinterpret_cast(&out2), sizeof(unsigned char)); - - if(densityb(j,i) > 255) - out2 = 255; //clip to white - else if(densityb(j,i) < 0) - out2 = 0; //clip to black - else - out2 = (unsigned char) densityb(j,i); //normal values - output.write(reinterpret_cast(&out2), sizeof(unsigned char)); - } + } else// 8 bit + { + for (j = 0; j < ysize; j++) { + for (i = 0; i < xsize; i++) { + if (densityr(j, i) > 255) + out2 = 255;// clip to white + else if (densityr(j, i) < 0) + out2 = 0;// clip to black + else + out2 = (unsigned char)densityr(j, i);// normal values + output.write(reinterpret_cast(&out2), sizeof(unsigned char)); + + if (densityg(j, i) > 255) + out2 = 255;// clip to white + else if (densityg(j, i) < 0) + out2 = 0;// clip to black + else + out2 = (unsigned char)densityg(j, i);// normal values + output.write(reinterpret_cast(&out2), sizeof(unsigned char)); + + if (densityb(j, i) > 255) + out2 = 255;// clip to white + else if (densityb(j, i) < 0) + out2 = 0;// clip to black + else + out2 = (unsigned char)densityb(j, i);// normal values + output.write(reinterpret_cast(&out2), sizeof(unsigned char)); } } - return false; + } + return false; } //****************************************************************************80 -bool ppmb_write_header ( ofstream &output, int xsize, int ysize, int maxrgb ) +bool ppmb_write_header(ofstream &output, int xsize, int ysize, int maxrgb) //****************************************************************************80 // @@ -161,71 +160,61 @@ bool ppmb_write_header ( ofstream &output, int xsize, int ysize, int maxrgb ) // Output, bool PPMB_WRITE_HEADER, is true if an error occurred. // { - output << "P6" << " " - << xsize << " " - << ysize << " " - << maxrgb << "\n"; + output << "P6" << " " << xsize << " " << ysize << " " << maxrgb << "\n"; return false; } -void imwrite(matrix &densityr, matrix &densityg, - matrix &densityb, string output_name, bool sixteen_bit ) +void imwrite(matrix &densityr, + matrix &densityg, + matrix &densityb, + string output_name, + bool sixteen_bit) { bool error; ofstream output; int maxrgb; int xsize = densityr.nc(); int ysize = densityr.nr(); -// -// Open the output file. -// - output.open ( output_name.c_str( ), ios::binary ); + // + // Open the output file. + // + output.open(output_name.c_str(), ios::binary); - if ( !output ) - { - cout << "\n"; - cout << "PPMB_WRITE: Fatal error!\n"; - cout << " Cannot open the output file " << output_name << ".\n"; + if (!output) { + FILM_ERROR("PPMB_WRITE: Fatal error! Cannot open the output file {}", output_name); return; } -// -// Compute the maximum. -// - if ( sixteen_bit) + // + // Compute the maximum. + // + if (sixteen_bit) maxrgb = 65535; else maxrgb = 255; -// -// Write the header. -// - error = ppmb_write_header ( output, xsize, ysize, maxrgb ); + // + // Write the header. + // + error = ppmb_write_header(output, xsize, ysize, maxrgb); - if ( error ) - { - cout << "\n"; - cout << "PPMB_WRITE: Fatal error!\n"; - cout << " PPMB_WRITE_HEADER failed.\n"; + if (error) { + FILM_ERROR("PPMB_WRITE: Fatal error! PPMB_WRITE_HEADER failed."); return; } -// -// Write the data. -// - error = ppmb_write_data ( output, xsize, ysize, densityr, densityg, densityb, - sixteen_bit); + // + // Write the data. + // + error = ppmb_write_data(output, xsize, ysize, densityr, densityg, densityb, sixteen_bit); - if ( error ) - { - cout << "\n"; - cout << "PPMB_WRITE: Fatal error!\n"; - cout << " PPMB_WRITE_DATA failed.\n"; + if (error) { + FILM_ERROR("PPMB_WRITE: Fatal error! PPMB_WRITE_DATA failed."); return; } -// -// Close the file. -// - output.close ( ); + // + // Close the file. + // + output.close(); return; } diff --git a/filmulator-gui/core/imwriteJpeg.cpp b/filmulator-gui/core/imwriteJpeg.cpp index 94e356b8..20889a48 100644 --- a/filmulator-gui/core/imwriteJpeg.cpp +++ b/filmulator-gui/core/imwriteJpeg.cpp @@ -1,4 +1,4 @@ -/* +/* * This file is part of Filmulator. * * Copyright 2013 Omer Mano and Carlo Vaccari @@ -17,355 +17,337 @@ * along with Filmulator. If not, see */ #include "filmSim.hpp" - -bool imwrite_jpeg(matrix &output, string outputfilename, - Exiv2::ExifData exifData, int quality, string thumbPath, bool writeExif) +#include "logging.h" + +bool imwrite_jpeg(matrix &output, + string outputfilename, + Exiv2::ExifData exifData, + int quality, + string thumbPath, + bool writeExif) { - int xsize = output.nc()/3; - int ysize = output.nr(); - if (quality > 100) - { - quality = 100; - } - else if (quality < 10) - { - quality = 10; + int xsize = output.nc() / 3; + int ysize = output.nr(); + if (quality > 100) { + quality = 100; + } else if (quality < 10) { + quality = 10; + } + + outputfilename = outputfilename + ".jpg"; + // From example.c of libjpeg-turbo + /* This struct contains the JPEG compression parameters and pointers to + * working space (which is allocated as needed by the JPEG library). + * It is possible to have several such structures, representing multiple + * compression/decompression processes, in existence at once. We refer + * to any one struct (and its associated working data) as a "JPEG object". + */ + struct jpeg_compress_struct cinfo; + /* This struct represents a JPEG error handler. It is declared separately + * because applications often want to supply a specialized error handler + * (see the second half of this file for an example). But here we just + * take the easy way out and use the standard error handler, which will + * print a message on stderr and call exit() if compression fails. + * Note that this struct must live as long as the main JPEG parameter + * struct, to avoid dangling-pointer problems. + */ + struct jpeg_error_mgr jerr; + /* More stuff */ + FILE *outfile; /* target file */ + + /* Step 1: allocate and initialize JPEG compression object */ + + /* We have to set up the error handler first, in case the initialization + * step fails. (Unlikely, but it could happen if you are out of memory.) + * This routine fills in the contents of struct jerr, and returns jerr's + * address which we place into the link field in cinfo. + */ + cinfo.err = jpeg_std_error(&jerr); + /* Now we can initialize the JPEG compression object. */ + jpeg_create_compress(&cinfo); + + /* Step 2: specify data destination (eg, a file) */ + /* Note: steps 2 and 3 can be done in either order. */ + + /* Here we use the library-supplied code to send compressed data to a + * stdio stream. You can also write your own code to do something else. + * VERY IMPORTANT: use "b" option to fopen() if you are on a machine that + * requires it in order to write binary files. + */ + if ((outfile = fopen(outputfilename.c_str(), "wb")) == NULL) { + FILM_ERROR("can't open {} for writing", outputfilename); + exit(1); + } + jpeg_stdio_dest(&cinfo, outfile); + + /* Step 3: set parameters for compression */ + + /* First we supply a description of the input image. + * Four fields of the cinfo struct must be filled in: + */ + cinfo.image_width = xsize; /* image width and height, in pixels */ + cinfo.image_height = ysize; + cinfo.input_components = 3; /* # of color components per pixel */ + cinfo.in_color_space = JCS_RGB; /* colorspace of input image */ + /* Now use the library's routine to set default compression parameters. + * (You must set at least cinfo.in_color_space before calling this, + * since the defaults depend on the source color space.) + */ + jpeg_set_defaults(&cinfo); + /* Now you can set any non-default parameters you wish to. + * Here we just illustrate the use of quality (quantization table) scaling: + */ + jpeg_set_quality(&cinfo, quality, TRUE /* limit to baseline-JPEG values */); + + /* Step 4: Start compressor */ + + /* TRUE ensures that we will write a complete interchange-JPEG file. + * Pass TRUE unless you are very sure of what you're doing. + */ + jpeg_start_compress(&cinfo, TRUE); + + /* Step 5: while (scan lines remain to be written) */ + /* jpeg_write_scanlines(...); */ + + /* Here we use the library's state variable cinfo.next_scanline as the + * loop counter, so that we don't have to keep track ourselves. + * To keep things simple, we pass one scanline per call; you can pass + * more if you wish, though. + */ + + tsize_t linebytes = xsize * 3; /* JSAMPLEs per row in image */ + JSAMPLE *buf = NULL; + buf = (JSAMPLE *)malloc(linebytes); + for (int j = 0; j < ysize; j++) { + for (int i = 0; i < xsize; i++) { + buf[i * 3] = dither_round(output(j, i * 3)); + buf[i * 3 + 1] = dither_round(output(j, i * 3 + 1)); + buf[i * 3 + 2] = dither_round(output(j, i * 3 + 2)); } - - outputfilename = outputfilename + ".jpg"; - //From example.c of libjpeg-turbo - /* This struct contains the JPEG compression parameters and pointers to - * working space (which is allocated as needed by the JPEG library). - * It is possible to have several such structures, representing multiple - * compression/decompression processes, in existence at once. We refer - * to any one struct (and its associated working data) as a "JPEG object". - */ - struct jpeg_compress_struct cinfo; - /* This struct represents a JPEG error handler. It is declared separately - * because applications often want to supply a specialized error handler - * (see the second half of this file for an example). But here we just - * take the easy way out and use the standard error handler, which will - * print a message on stderr and call exit() if compression fails. - * Note that this struct must live as long as the main JPEG parameter - * struct, to avoid dangling-pointer problems. - */ - struct jpeg_error_mgr jerr; - /* More stuff */ - FILE * outfile; /* target file */ - - /* Step 1: allocate and initialize JPEG compression object */ - - /* We have to set up the error handler first, in case the initialization - * step fails. (Unlikely, but it could happen if you are out of memory.) - * This routine fills in the contents of struct jerr, and returns jerr's - * address which we place into the link field in cinfo. - */ - cinfo.err = jpeg_std_error(&jerr); - /* Now we can initialize the JPEG compression object. */ - jpeg_create_compress(&cinfo); - - /* Step 2: specify data destination (eg, a file) */ - /* Note: steps 2 and 3 can be done in either order. */ - - /* Here we use the library-supplied code to send compressed data to a - * stdio stream. You can also write your own code to do something else. - * VERY IMPORTANT: use "b" option to fopen() if you are on a machine that - * requires it in order to write binary files. - */ - if ((outfile = fopen(outputfilename.c_str(), "wb")) == NULL) { - fprintf(stderr, "can't open %s for writing\n", outputfilename.c_str()); - exit(1); - } - jpeg_stdio_dest(&cinfo, outfile); - - /* Step 3: set parameters for compression */ - - /* First we supply a description of the input image. - * Four fields of the cinfo struct must be filled in: - */ - cinfo.image_width = xsize; /* image width and height, in pixels */ - cinfo.image_height = ysize; - cinfo.input_components = 3; /* # of color components per pixel */ - cinfo.in_color_space = JCS_RGB; /* colorspace of input image */ - /* Now use the library's routine to set default compression parameters. - * (You must set at least cinfo.in_color_space before calling this, - * since the defaults depend on the source color space.) - */ - jpeg_set_defaults(&cinfo); - /* Now you can set any non-default parameters you wish to. - * Here we just illustrate the use of quality (quantization table) scaling: - */ - jpeg_set_quality(&cinfo, quality, TRUE /* limit to baseline-JPEG values */); - - /* Step 4: Start compressor */ - - /* TRUE ensures that we will write a complete interchange-JPEG file. - * Pass TRUE unless you are very sure of what you're doing. - */ - jpeg_start_compress(&cinfo, TRUE); - - /* Step 5: while (scan lines remain to be written) */ - /* jpeg_write_scanlines(...); */ - - /* Here we use the library's state variable cinfo.next_scanline as the - * loop counter, so that we don't have to keep track ourselves. - * To keep things simple, we pass one scanline per call; you can pass - * more if you wish, though. - */ - - tsize_t linebytes = xsize*3; /* JSAMPLEs per row in image */ - JSAMPLE *buf = NULL; - buf =(JSAMPLE *)malloc(linebytes); - for (int j = 0; j < ysize; j++) - { - for(int i = 0; i < xsize; i ++) - { - buf[i*3 ] = dither_round(output(j,i*3 )); - buf[i*3+1] = dither_round(output(j,i*3+1)); - buf[i*3+2] = dither_round(output(j,i*3+2)); - } - (void) jpeg_write_scanlines(&cinfo,&buf,1); - } - - /* Step 6: Finish compression */ - - jpeg_finish_compress(&cinfo); - /* After finish_compress, we can close the output file. */ - fclose(outfile); - - /* Step 7: release JPEG compression object */ - - /* This is an important step since it will release a good deal of memory. */ - jpeg_destroy_compress(&cinfo); - - if (writeExif) - { - cleanExif(exifData); - - exifData["Exif.Image.Orientation"] = uint16_t(1);//set all images to unrotated - exifData["Exif.Image.ImageWidth"] = output.nc()/3; - exifData["Exif.Image.ImageLength"] = output.nr(); - exifData["Exif.Photo.ColorSpace"] = 1; - exifData["Exif.Image.ProcessingSoftware"] = "Filmulator"; - - //Fix the thumbnails - auto thumb = Exiv2::ExifThumb(exifData); - thumb.erase(); - //Find the thumbnail for this image - //thumb.setJpegThumbnail(thumbPath); - - cout << "imwrite_jpeg exiv filename: " << outputfilename << endl; - auto image = Exiv2::ImageFactory::open(outputfilename); - assert(image.get() != 0); - - image->setExifData(exifData); - image->writeMetadata(); - } - - return 0; + (void)jpeg_write_scanlines(&cinfo, &buf, 1); + } + + /* Step 6: Finish compression */ + + jpeg_finish_compress(&cinfo); + /* After finish_compress, we can close the output file. */ + fclose(outfile); + + /* Step 7: release JPEG compression object */ + + /* This is an important step since it will release a good deal of memory. */ + jpeg_destroy_compress(&cinfo); + + if (writeExif) { + cleanExif(exifData); + + exifData["Exif.Image.Orientation"] = uint16_t(1);// set all images to unrotated + exifData["Exif.Image.ImageWidth"] = output.nc() / 3; + exifData["Exif.Image.ImageLength"] = output.nr(); + exifData["Exif.Photo.ColorSpace"] = 1; + exifData["Exif.Image.ProcessingSoftware"] = "Filmulator"; + + // Fix the thumbnails + auto thumb = Exiv2::ExifThumb(exifData); + thumb.erase(); + // Find the thumbnail for this image + // thumb.setJpegThumbnail(thumbPath); + + FILM_TRACE("imwrite_jpeg exiv filename: {}", outputfilename); + auto image = Exiv2::ImageFactory::open(outputfilename); + assert(image.get() != 0); + + image->setExifData(exifData); + image->writeMetadata(); + } + + return 0; } JSAMPLE dither_round(int full_int) { - float intermediate = full_int; - intermediate = ((intermediate + 1)/256) -1; // converted to output range - float dither = rand(); - dither = dither/RAND_MAX; //produces 0<=dither<=1 - dither = 0.9999*dither; //dither now cannot cause rounding one integer to the next - return (JSAMPLE)(intermediate + dither); + float intermediate = full_int; + intermediate = ((intermediate + 1) / 256) - 1;// converted to output range + float dither = rand(); + dither = dither / RAND_MAX;// produces 0<=dither<=1 + dither = 0.9999 * dither;// dither now cannot cause rounding one integer to the next + return (JSAMPLE)(intermediate + dither); } -//exif stripping based on darktable's exif.cc +// exif stripping based on darktable's exif.cc static void remove_exif_keys(Exiv2::ExifData &exifData, const char *keys[], unsigned int n_keys) { - for (unsigned int i=0; i < n_keys; i++) - { - try { - Exiv2::ExifData::iterator pos; - while ((pos = exifData.findKey(Exiv2::ExifKey(keys[i]))) != exifData.end()) - { - exifData.erase(pos); - } - } catch (Exiv2::Error &e) { - //catch invalid tag - } + for (unsigned int i = 0; i < n_keys; i++) { + try { + Exiv2::ExifData::iterator pos; + while ((pos = exifData.findKey(Exiv2::ExifKey(keys[i]))) != exifData.end()) { exifData.erase(pos); } + } catch (Exiv2::Error &e) { + // catch invalid tag } + } } void cleanExif(Exiv2::ExifData &exifData) { - // Remove thumbnail - auto thumb = Exiv2::ExifThumb(exifData); - thumb.erase(); - { - static const char *keys[] = { - "Exif.Thumbnail.Compression", - "Exif.Thumbnail.XResolution", - "Exif.Thumbnail.YResolution", - "Exif.Thumbnail.ResolutionUnit", - "Exif.Thumbnail.JPEGInterchangeFormat", - "Exif.Thumbnail.JPEGInterchangeFormatLength" - }; - static const int n_keys = 6; - remove_exif_keys(exifData, keys, n_keys); - } - - // only compressed images may set these - { - static const char *keys[] = { - "Exif.Photo.PixelXDimension", - "Exif.Photo.PixelYDimension" - }; - static const int n_keys = 2; - remove_exif_keys(exifData, keys, n_keys); - } - - { - static const char *keys[] = { - "Exif.Image.ImageWidth", - "Exif.Image.ImageLength", - "Exif.Image.BitsPerSample", - "Exif.Image.Compression", - "Exif.Image.PhotometricInterpretation", - "Exif.Image.FillOrder", - "Exif.Image.SamplesPerPixel", - "Exif.Image.StripOffsets", - "Exif.Image.RowsPerStrip", - "Exif.Image.StripByteCounts", - "Exif.Image.PlanarConfiguration", - "Exif.Image.DNGVersion", - "Exif.Image.DNGBackwardVersion" - }; - static const int n_keys = 13; - remove_exif_keys(exifData, keys, n_keys); - } - - //remove subimages - for(Exiv2::ExifData::iterator i = exifData.begin(); i != exifData.end();) - { - static const std::string needle = "Exif.SubImage"; - if(i->key().compare(0, needle.length(), needle) == 0) - { - i = exifData.erase(i); - } else { - ++i; - } - } - - { - static const char *keys[] = { - // Canon color space info - "Exif.Canon.ColorSpace", - "Exif.Canon.ColorData", - - // Nikon thumbnail data - "Exif.Nikon3.Preview", - "Exif.NikonPreview.JPEGInterchangeFormat", - - // DNG stuff that is irrelevant or misleading - "Exif.Image.DNGPrivateData", - "Exif.Image.DefaultBlackRender", - "Exif.Image.DefaultCropOrigin", - "Exif.Image.DefaultCropSize", - "Exif.Image.RawDataUniqueID", - "Exif.Image.OriginalRawFileName", - "Exif.Image.OriginalRawFileData", - "Exif.Image.ActiveArea", - "Exif.Image.MaskedAreas", - "Exif.Image.AsShotICCProfile", - "Exif.Image.OpcodeList1", - "Exif.Image.OpcodeList2", - "Exif.Image.OpcodeList3", - "Exif.Photo.MakerNote", - - // Pentax thumbnail data - "Exif.Pentax.PreviewResolution", - "Exif.Pentax.PreviewLength", - "Exif.Pentax.PreviewOffset", - "Exif.PentaxDng.PreviewResolution", - "Exif.PentaxDng.PreviewLength", - "Exif.PentaxDng.PreviewOffset", - // Pentax color info - "Exif.PentaxDng.ColorInfo", - - // Minolta thumbnail data - "Exif.Minolta.Thumbnail", - "Exif.Minolta.ThumbnailOffset", - "Exif.Minolta.ThumbnailLength", - - // Sony thumbnail data - "Exif.SonyMinolta.ThumbnailOffset", - "Exif.SonyMinolta.ThumbnailLength", - - // Olympus thumbnail data - "Exif.Olympus.Thumbnail", - "Exif.Olympus.ThumbnailOffset", - "Exif.Olympus.ThumbnailLength", - - "Exif.Image.BaselineExposureOffset" - }; - static const int n_keys = 34; - remove_exif_keys(exifData, keys, n_keys); + // Remove thumbnail + auto thumb = Exiv2::ExifThumb(exifData); + thumb.erase(); + { + static const char *keys[] = { "Exif.Thumbnail.Compression", + "Exif.Thumbnail.XResolution", + "Exif.Thumbnail.YResolution", + "Exif.Thumbnail.ResolutionUnit", + "Exif.Thumbnail.JPEGInterchangeFormat", + "Exif.Thumbnail.JPEGInterchangeFormatLength" }; + static const int n_keys = 6; + remove_exif_keys(exifData, keys, n_keys); + } + + // only compressed images may set these + { + static const char *keys[] = { "Exif.Photo.PixelXDimension", "Exif.Photo.PixelYDimension" }; + static const int n_keys = 2; + remove_exif_keys(exifData, keys, n_keys); + } + + { + static const char *keys[] = { "Exif.Image.ImageWidth", + "Exif.Image.ImageLength", + "Exif.Image.BitsPerSample", + "Exif.Image.Compression", + "Exif.Image.PhotometricInterpretation", + "Exif.Image.FillOrder", + "Exif.Image.SamplesPerPixel", + "Exif.Image.StripOffsets", + "Exif.Image.RowsPerStrip", + "Exif.Image.StripByteCounts", + "Exif.Image.PlanarConfiguration", + "Exif.Image.DNGVersion", + "Exif.Image.DNGBackwardVersion" }; + static const int n_keys = 13; + remove_exif_keys(exifData, keys, n_keys); + } + + // remove subimages + for (Exiv2::ExifData::iterator i = exifData.begin(); i != exifData.end();) { + static const std::string needle = "Exif.SubImage"; + if (i->key().compare(0, needle.length(), needle) == 0) { + i = exifData.erase(i); + } else { + ++i; } + } + + { + static const char *keys[] = { // Canon color space info + "Exif.Canon.ColorSpace", + "Exif.Canon.ColorData", + + // Nikon thumbnail data + "Exif.Nikon3.Preview", + "Exif.NikonPreview.JPEGInterchangeFormat", + + // DNG stuff that is irrelevant or misleading + "Exif.Image.DNGPrivateData", + "Exif.Image.DefaultBlackRender", + "Exif.Image.DefaultCropOrigin", + "Exif.Image.DefaultCropSize", + "Exif.Image.RawDataUniqueID", + "Exif.Image.OriginalRawFileName", + "Exif.Image.OriginalRawFileData", + "Exif.Image.ActiveArea", + "Exif.Image.MaskedAreas", + "Exif.Image.AsShotICCProfile", + "Exif.Image.OpcodeList1", + "Exif.Image.OpcodeList2", + "Exif.Image.OpcodeList3", + "Exif.Photo.MakerNote", + + // Pentax thumbnail data + "Exif.Pentax.PreviewResolution", + "Exif.Pentax.PreviewLength", + "Exif.Pentax.PreviewOffset", + "Exif.PentaxDng.PreviewResolution", + "Exif.PentaxDng.PreviewLength", + "Exif.PentaxDng.PreviewOffset", + // Pentax color info + "Exif.PentaxDng.ColorInfo", + + // Minolta thumbnail data + "Exif.Minolta.Thumbnail", + "Exif.Minolta.ThumbnailOffset", + "Exif.Minolta.ThumbnailLength", + + // Sony thumbnail data + "Exif.SonyMinolta.ThumbnailOffset", + "Exif.SonyMinolta.ThumbnailLength", + + // Olympus thumbnail data + "Exif.Olympus.Thumbnail", + "Exif.Olympus.ThumbnailOffset", + "Exif.Olympus.ThumbnailLength", + + "Exif.Image.BaselineExposureOffset" + }; + static const int n_keys = 34; + remove_exif_keys(exifData, keys, n_keys); + } #if EXIV2_MINOR_VERSION >= 23 - { - //older exiv2 will drop all exif if this is executed - //Samsung makernote cleanup - static const char *keys[] = { - "Exif.Samsung2.SensorAreas", - "Exif.Samsung2.ColorSpace", - "Exif.Samsung2.EncryptionKey", - "Exif.Samsung2.WB_RGGBLevelsUncorrected", - "Exif.Samsung2.WB_RGGBLevelsAuto", - "Exif.Samsung2.WB_RGGBLevelsIlluminator1", - "Exif.Samsung2.WB_RGGBLevelsIlluminator2", - "Exif.Samsung2.WB_RGGBLevelsBlack", - "Exif.Samsung2.ColorMatrix", - "Exif.Samsung2.ColorMatrixSRGB", - "Exif.Samsung2.ColorMatrixAdobeRGB", - "Exif.Samsung2.ToneCurve1", - "Exif.Samsung2.ToneCurve2", - "Exif.Samsung2.ToneCurve3", - "Exif.Samsung2.ToneCurve4" - }; - static const int n_keys = 15; - remove_exif_keys(exifData, keys, n_keys); - } + { + // older exiv2 will drop all exif if this is executed + // Samsung makernote cleanup + static const char *keys[] = { "Exif.Samsung2.SensorAreas", + "Exif.Samsung2.ColorSpace", + "Exif.Samsung2.EncryptionKey", + "Exif.Samsung2.WB_RGGBLevelsUncorrected", + "Exif.Samsung2.WB_RGGBLevelsAuto", + "Exif.Samsung2.WB_RGGBLevelsIlluminator1", + "Exif.Samsung2.WB_RGGBLevelsIlluminator2", + "Exif.Samsung2.WB_RGGBLevelsBlack", + "Exif.Samsung2.ColorMatrix", + "Exif.Samsung2.ColorMatrixSRGB", + "Exif.Samsung2.ColorMatrixAdobeRGB", + "Exif.Samsung2.ToneCurve1", + "Exif.Samsung2.ToneCurve2", + "Exif.Samsung2.ToneCurve3", + "Exif.Samsung2.ToneCurve4" }; + static const int n_keys = 15; + remove_exif_keys(exifData, keys, n_keys); + } #endif - { - static const char *keys[] = { - // Embedded color profile info - "Exif.Image.CalibrationIlluminant1", - "Exif.Image.CalibrationIlluminant2", - "Exif.Image.ColorMatrix1", - "Exif.Image.ColorMatrix2", - "Exif.Image.ForwardMatrix1", - "Exif.Image.ForwardMatrix2", - "Exif.Image.ProfileCalibrationSignature", - "Exif.Image.ProfileCopyright", - "Exif.Image.ProfileEmbedPolicy", - "Exif.Image.ProfileHueSatMapData1", - "Exif.Image.ProfileHueSatMapData2", - "Exif.Image.ProfileHueSatMapDims", - "Exif.Image.ProfileHueSatMapEncoding", - "Exif.Image.ProfileLookTableData", - "Exif.Image.ProfileLookTableDims", - "Exif.Image.ProfileLookTableEncoding", - "Exif.Image.ProfileName", - "Exif.Image.ProfileToneCurve", - "Exif.Image.ReductionMatrix1", - "Exif.Image.ReductionMatrix2" - }; - static const int n_keys = 20; - remove_exif_keys(exifData, keys, n_keys); - } - - /*{ - static const char *keys[] = { - }; - static const int n_keys = 0; - remove_exif_keys(exifData, keys, n_keys); - }*/ + { + static const char *keys[] = { // Embedded color profile info + "Exif.Image.CalibrationIlluminant1", + "Exif.Image.CalibrationIlluminant2", + "Exif.Image.ColorMatrix1", + "Exif.Image.ColorMatrix2", + "Exif.Image.ForwardMatrix1", + "Exif.Image.ForwardMatrix2", + "Exif.Image.ProfileCalibrationSignature", + "Exif.Image.ProfileCopyright", + "Exif.Image.ProfileEmbedPolicy", + "Exif.Image.ProfileHueSatMapData1", + "Exif.Image.ProfileHueSatMapData2", + "Exif.Image.ProfileHueSatMapDims", + "Exif.Image.ProfileHueSatMapEncoding", + "Exif.Image.ProfileLookTableData", + "Exif.Image.ProfileLookTableDims", + "Exif.Image.ProfileLookTableEncoding", + "Exif.Image.ProfileName", + "Exif.Image.ProfileToneCurve", + "Exif.Image.ReductionMatrix1", + "Exif.Image.ReductionMatrix2" + }; + static const int n_keys = 20; + remove_exif_keys(exifData, keys, n_keys); + } + + /*{ + static const char *keys[] = { + }; + static const int n_keys = 0; + remove_exif_keys(exifData, keys, n_keys); + }*/ } diff --git a/filmulator-gui/core/imwriteTiff.cpp b/filmulator-gui/core/imwriteTiff.cpp index 6cf6fd91..0279afc0 100644 --- a/filmulator-gui/core/imwriteTiff.cpp +++ b/filmulator-gui/core/imwriteTiff.cpp @@ -1,4 +1,4 @@ -/* +/* * This file is part of Filmulator. * * Copyright 2013 Omer Mano and Carlo Vaccari @@ -18,82 +18,75 @@ */ #include "filmSim.hpp" -bool imwrite_tiff(const matrix& output, string outputfilename, - Exiv2::ExifData exifData) +bool imwrite_tiff(const matrix &output, string outputfilename, Exiv2::ExifData exifData) { - int xsize, ysize; - xsize = output.nc()/3; - ysize = output.nr(); + int xsize, ysize; + xsize = output.nc() / 3; + ysize = output.nr(); + outputfilename = outputfilename + ".tif"; + TIFF *out = TIFFOpen(outputfilename.c_str(), "w"); + if (!out) { + cerr << "Can't open file for writing" << endl; + return 1; + } + TIFFSetField(out, TIFFTAG_IMAGEWIDTH, xsize); + TIFFSetField(out, TIFFTAG_IMAGELENGTH, ysize); + TIFFSetField(out, TIFFTAG_SAMPLESPERPIXEL, 3);// RGB + TIFFSetField(out, TIFFTAG_BITSPERSAMPLE, 16); + TIFFSetField(out, TIFFTAG_ORIENTATION, ORIENTATION_TOPLEFT);// set the origin of the image. + // Magic below + TIFFSetField(out, TIFFTAG_PLANARCONFIG, PLANARCONFIG_CONTIG); + TIFFSetField(out, TIFFTAG_PHOTOMETRIC, PHOTOMETRIC_RGB); + if (!TIFFSetField(out, TIFFTAG_COMPRESSION, COMPRESSION_NONE)) { cout << "couldn't set tiff compression" << endl; } + // End Magic - outputfilename = outputfilename + ".tif"; - TIFF *out = TIFFOpen(outputfilename.c_str(),"w"); - if (!out) - { - cerr << "Can't open file for writing" << endl; - return 1; - } - TIFFSetField(out, TIFFTAG_IMAGEWIDTH, xsize); - TIFFSetField(out, TIFFTAG_IMAGELENGTH, ysize); - TIFFSetField(out, TIFFTAG_SAMPLESPERPIXEL, 3); //RGB - TIFFSetField(out, TIFFTAG_BITSPERSAMPLE, 16); - TIFFSetField(out, TIFFTAG_ORIENTATION, ORIENTATION_TOPLEFT); // set the origin of the image. - //Magic below - TIFFSetField(out, TIFFTAG_PLANARCONFIG, PLANARCONFIG_CONTIG); - TIFFSetField(out, TIFFTAG_PHOTOMETRIC, PHOTOMETRIC_RGB); - if(!TIFFSetField(out, TIFFTAG_COMPRESSION, COMPRESSION_NONE)) {cout << "couldn't set tiff compression" << endl;} - //End Magic + std::string make = exifData["Exif.Image.Make"].toString(); + TIFFSetField(out, TIFFTAG_MAKE, make.c_str()); + std::string model = exifData["Exif.Image.Model"].toString(); + TIFFSetField(out, TIFFTAG_MODEL, model.c_str()); + TIFFSetField(out, TIFFTAG_SOFTWARE, "Filmulator"); + std::string copyright = exifData["Exif.Image.Copyright"].toString(); + TIFFSetField(out, TIFFTAG_COPYRIGHT, copyright.c_str()); + std::string lensinfo = exifData["Exif.Image.LensInfo"].toString(); + TIFFSetField(out, TIFFTAG_LENSINFO, lensinfo.c_str()); + std::string datetime = exifData["Exif.Image.DateTime"].toString(); + TIFFSetField(out, TIFFTAG_DATETIME, datetime.c_str()); - std::string make = exifData["Exif.Image.Make"].toString(); - TIFFSetField(out, TIFFTAG_MAKE, make.c_str()); - std::string model = exifData["Exif.Image.Model"].toString(); - TIFFSetField(out, TIFFTAG_MODEL, model.c_str()); - TIFFSetField(out, TIFFTAG_SOFTWARE, "Filmulator"); - std::string copyright = exifData["Exif.Image.Copyright"].toString(); - TIFFSetField(out, TIFFTAG_COPYRIGHT, copyright.c_str()); - std::string lensinfo = exifData["Exif.Image.LensInfo"].toString(); - TIFFSetField(out, TIFFTAG_LENSINFO, lensinfo.c_str()); - std::string datetime = exifData["Exif.Image.DateTime"].toString(); - TIFFSetField(out, TIFFTAG_DATETIME, datetime.c_str()); - - tsize_t linebytes = 3 * xsize * 2;//Size in bytes of a line - unsigned short *buf = NULL; - buf =(unsigned short *)_TIFFmalloc(linebytes); - for (int j = 0; j < ysize; j++) - { - for(int i = 0; i < xsize; i ++) - { - buf[i*3 ] = output(j,i*3 ); - buf[i*3+1] = output(j,i*3+1); - buf[i*3+2] = output(j,i*3+2); - } - if (TIFFWriteScanline(out, buf, j, 0) < 0) - break; + tsize_t linebytes = 3 * xsize * 2;// Size in bytes of a line + unsigned short *buf = NULL; + buf = (unsigned short *)_TIFFmalloc(linebytes); + for (int j = 0; j < ysize; j++) { + for (int i = 0; i < xsize; i++) { + buf[i * 3] = output(j, i * 3); + buf[i * 3 + 1] = output(j, i * 3 + 1); + buf[i * 3 + 2] = output(j, i * 3 + 2); } + if (TIFFWriteScanline(out, buf, j, 0) < 0) break; + } - TIFFWriteDirectory(out); + TIFFWriteDirectory(out); - (void) TIFFClose(out); + (void)TIFFClose(out); - if (buf) - _TIFFfree(buf); + if (buf) _TIFFfree(buf); - //Programs reading tiffs really freak out if you write the full exif data - //So we write only the basics here. - Exiv2::ExifData minimumData; - minimumData["Exif.Photo.ISOSpeedRatings"] = exifData["Exif.Photo.ISOSpeedRatings"]; - minimumData["Exif.Photo.ExposureTime"] = exifData["Exif.Photo.ExposureTime"]; - minimumData["Exif.Photo.FNumber"] = exifData["Exif.Photo.FNumber"]; - minimumData["Exif.Photo.FocalLength"] = exifData["Exif.Photo.FocalLength"]; + // Programs reading tiffs really freak out if you write the full exif data + // So we write only the basics here. + Exiv2::ExifData minimumData; + minimumData["Exif.Photo.ISOSpeedRatings"] = exifData["Exif.Photo.ISOSpeedRatings"]; + minimumData["Exif.Photo.ExposureTime"] = exifData["Exif.Photo.ExposureTime"]; + minimumData["Exif.Photo.FNumber"] = exifData["Exif.Photo.FNumber"]; + minimumData["Exif.Photo.FocalLength"] = exifData["Exif.Photo.FocalLength"]; - minimumData.sortByTag(); //darktable's does this so maybe we should + minimumData.sortByTag();// darktable's does this so maybe we should - auto image = Exiv2::ImageFactory::open(outputfilename); - assert(image.get() != 0); + auto image = Exiv2::ImageFactory::open(outputfilename); + assert(image.get() != 0); - image->setExifData(minimumData); - image->writeMetadata(); + image->setExifData(minimumData); + image->writeMetadata(); - return 0; + return 0; } diff --git a/filmulator-gui/core/interface.h b/filmulator-gui/core/interface.h index 1bca5135..08dc3a7c 100644 --- a/filmulator-gui/core/interface.h +++ b/filmulator-gui/core/interface.h @@ -2,38 +2,52 @@ #define INTERFACE_H #include "matrix.hpp" -enum LogY {no, yes}; +enum LogY { no, yes }; -struct Histogram { - long long lHist[128]; - long long rHist[128]; - long long gHist[128]; - long long bHist[128]; +struct Histogram +{ + long long lHist[128]; + long long rHist[128]; + long long gHist[128]; + long long bHist[128]; - float lHistMax; - float rHistMax; - float gHistMax; - float bHistMax; + float lHistMax; + float rHistMax; + float gHistMax; + float bHistMax; - bool empty = true; + bool empty = true; }; class Interface { public: - virtual void setProgress(float){} - virtual void updateHistRaw(const matrix& /*image*/, const float /*maximum*/[3], unsigned /*cfa*/[2][2], unsigned /*xtrans*/[6][6], int /*maxXtrans*/, bool /*isRGB*/, bool /*isMonochrome*/){} - virtual void updateHistPreFilm(const matrix& /*image*/, const float /*maximum*/, - const int /*rotation*/, - const float /*cropHeight*/, const float /*cropAspect*/, - const float /*cropHoffset*/, const float /*cropVoffset*/){} - virtual void updateHistPostFilm(const matrix& /*image*/, const float /*maximum*/, - const int /*rotation*/, - const float /*cropHeight*/, const float /*cropAspect*/, - const float /*cropHoffset*/, const float /*cropVoffset*/){} - virtual void updateHistFinal(const matrix& /*image*/){} - - + virtual void setProgress(float) {} + virtual void updateHistRaw(const matrix & /*image*/, + const float /*maximum*/[3], + unsigned /*cfa*/[2][2], + unsigned /*xtrans*/[6][6], + int /*maxXtrans*/, + bool /*isRGB*/, + bool /*isMonochrome*/) + {} + virtual void updateHistPreFilm(const matrix & /*image*/, + const float /*maximum*/, + const int /*rotation*/, + const float /*cropHeight*/, + const float /*cropAspect*/, + const float /*cropHoffset*/, + const float /*cropVoffset*/) + {} + virtual void updateHistPostFilm(const matrix & /*image*/, + const float /*maximum*/, + const int /*rotation*/, + const float /*cropHeight*/, + const float /*cropAspect*/, + const float /*cropHoffset*/, + const float /*cropVoffset*/) + {} + virtual void updateHistFinal(const matrix & /*image*/) {} }; -#endif // INTERFACE_H +#endif// INTERFACE_H diff --git a/filmulator-gui/core/layerMix.cpp b/filmulator-gui/core/layerMix.cpp index 56535a27..82251d5e 100644 --- a/filmulator-gui/core/layerMix.cpp +++ b/filmulator-gui/core/layerMix.cpp @@ -1,4 +1,4 @@ -/* +/* * This file is part of Filmulator. * * Copyright 2013 Omer Mano and Carlo Vaccari @@ -20,63 +20,58 @@ #include #include -//This function implements diffusion between the active developer layer -// adjacent to the film and the reservoir of inactive developer. +// This function implements diffusion between the active developer layer +// adjacent to the film and the reservoir of inactive developer. void layer_mix(matrix &developer_concentration, - float active_layer_thickness, - float &reservoir_developer_concentration, - float reservoir_thickness, - float layer_mix_const, - float layer_time_divisor, - float pixels_per_millimeter, - float timestep) + float active_layer_thickness, + float &reservoir_developer_concentration, + float reservoir_thickness, + float layer_mix_const, + float layer_time_divisor, + float pixels_per_millimeter, + float timestep) { - int length = developer_concentration.nr(); - int width = developer_concentration.nc(); - - //layer_time_divisor adjusts the ratio between the timestep used to compute - //the diffuse within the layer and this diffuse. - float layer_mix = pow(layer_mix_const,timestep/layer_time_divisor); + int length = developer_concentration.nr(); + int width = developer_concentration.nc(); - //layer_mix is the proportion of developer that stays in the layer. + // layer_time_divisor adjusts the ratio between the timestep used to compute + // the diffuse within the layer and this diffuse. + float layer_mix = pow(layer_mix_const, timestep / layer_time_divisor); - //This gives us the amount of developer that comes from the reservoir. - float reservoir_portion = (1-layer_mix) * reservoir_developer_concentration; + // layer_mix is the proportion of developer that stays in the layer. - //This lets us count how much developer got added to the layer in total. - double sum = 0; + // This gives us the amount of developer that comes from the reservoir. + float reservoir_portion = (1 - layer_mix) * reservoir_developer_concentration; - //Here we add developer to the layer. -#pragma omp parallel shared(developer_concentration) \ - firstprivate(layer_mix, reservoir_portion) reduction(+:sum) - { + // This lets us count how much developer got added to the layer in total. + double sum = 0; + + // Here we add developer to the layer. +#pragma omp parallel shared(developer_concentration) firstprivate(layer_mix, reservoir_portion) reduction(+ : sum) + { #pragma omp for schedule(dynamic) nowait - for(int row=0; row +#include +#include +#include +#include +#include + +namespace Filmulator { + +void init_logging() +{ + auto console_sink = std::make_shared(); + console_sink->set_level(spdlog::level::trace); + // Format: [time] [level] [thread] message + console_sink->set_pattern("%^[%T] [%l] [t %t] %v%$"); + + // File sink + QString log_dir_path = QStandardPaths::writableLocation(QStandardPaths::AppDataLocation); + QDir().mkpath(log_dir_path); + QString log_file_path = log_dir_path + "/filmulator.log"; + + const size_t max_file_size = 5ULL * 1024ULL * 1024ULL;// 5 MB + const size_t max_files = 3; + auto file_sink = + std::make_shared(log_file_path.toStdString(), max_file_size, max_files); + file_sink->set_level(spdlog::level::info); + file_sink->set_pattern("[%Y-%m-%d %T] [%l] [t %t] %v"); + + std::vector sinks{ console_sink, file_sink }; + auto logger = std::make_shared("filmulator", sinks.begin(), sinks.end()); + + // Set global logger + spdlog::set_default_logger(logger); + spdlog::set_level(spdlog::level::trace); + spdlog::flush_on(spdlog::level::info); + + FILM_INFO("Logging initialized. Log file: {}", log_file_path.toStdString()); +} + +}// namespace Filmulator diff --git a/filmulator-gui/core/logging.h b/filmulator-gui/core/logging.h new file mode 100644 index 00000000..58773b34 --- /dev/null +++ b/filmulator-gui/core/logging.h @@ -0,0 +1,20 @@ +#ifndef FILMULATOR_LOGGING_H +#define FILMULATOR_LOGGING_H + +#include +#include + +namespace Filmulator { + +void init_logging(); + +}// namespace Filmulator + +#define FILM_TRACE(...) spdlog::trace(__VA_ARGS__) +#define FILM_DEBUG(...) spdlog::debug(__VA_ARGS__) +#define FILM_INFO(...) spdlog::info(__VA_ARGS__) +#define FILM_WARN(...) spdlog::warn(__VA_ARGS__) +#define FILM_ERROR(...) spdlog::error(__VA_ARGS__) +#define FILM_CRITICAL(...) spdlog::critical(__VA_ARGS__) + +#endif// FILMULATOR_LOGGING_H diff --git a/filmulator-gui/core/lut.hpp b/filmulator-gui/core/lut.hpp index 0898a981..1b4e7f16 100644 --- a/filmulator-gui/core/lut.hpp +++ b/filmulator-gui/core/lut.hpp @@ -1,4 +1,4 @@ -/* +/* * This file is part of Filmulator. * * Copyright 2013 Omer Mano and Carlo Vaccari @@ -20,62 +20,55 @@ #define LUT_H #define MAXVAL 65536 +#include "interface.h" #include #include -#include "interface.h" using namespace std; -template -class LUT +template class LUT { private: - numberType table[MAXVAL]; - bool unity; - bool linear; - float slope; - float y_intercept; - float brightest; - float darkest; + numberType table[MAXVAL]; + bool unity; + bool linear; + float slope; + float y_intercept; + float brightest; + float darkest; + public: - void setLinear(float slope_in, float y_intercept_in, - float brightest_in, float darkest_in) - { - linear = true; - unity = false; - slope = slope_in; - y_intercept = y_intercept_in; - brightest = brightest_in; - darkest = darkest_in; - } + void setLinear(float slope_in, float y_intercept_in, float brightest_in, float darkest_in) + { + linear = true; + unity = false; + slope = slope_in; + y_intercept = y_intercept_in; + brightest = brightest_in; + darkest = darkest_in; + } + + void setUnity() + { + linear = false; + unity = true; + } - void setUnity() - { - linear = false; - unity = true; - } + bool isUnity() { return unity; } - bool isUnity() - { - return unity; - } + void fill(std::function func) + { + linear = false; + unity = false; - void fill(std::function func) - { - linear = false; - unity = false; + for (int i = 0; i < MAXVAL; i++) table[i] = func(i); + } - for(int i = 0; i < MAXVAL; i++) - table[i] = func(i); - } - - numberType operator[](unsigned short index) - { - if (unity) - return index; - if (linear) - return min(max((index*slope)+y_intercept,darkest),brightest); - return table[index]; - } + numberType operator[](unsigned short index) + { + if (unity) return index; + if (linear) return min(max((index * slope) + y_intercept, darkest), brightest); + return table[index]; + } }; -#endif //LUT_H +#endif// LUT_H diff --git a/filmulator-gui/core/matrix.hpp b/filmulator-gui/core/matrix.hpp index ca0675c8..e85c012e 100644 --- a/filmulator-gui/core/matrix.hpp +++ b/filmulator-gui/core/matrix.hpp @@ -1,4 +1,4 @@ -/* +/* * This file is part of Filmulator. * * Copyright 2013 Omer Mano and Carlo Vaccari @@ -18,9 +18,9 @@ */ #ifndef MATRIX_H #define MATRIX_H -#include -#include #include "math.h" +#include +#include #ifdef __SSE2__ #include #endif @@ -28,728 +28,561 @@ #include #include -#ifdef DOUT -#define dout cout -#else -#define dout 0 && cout #ifndef NDEBUG #define NDEBUG #endif -#endif -#include "assert.h" //Included later so NDEBUG has an effect +#include "assert.h"//Included later so NDEBUG has an effect -template -class matrix +template class matrix { - private: - T** ptr; - T* data; - T* ua_data;//unaligned - int num_rows; - int num_cols; - inline void slow_transpose_to(const matrix &target) const; +private: + T **ptr; + T *data; + T *ua_data;// unaligned + int num_rows; + int num_cols; + inline void slow_transpose_to(const matrix &target) const; #ifdef __SSE2__ - inline void fast_transpose_to(const matrix &target) const; - inline void transpose4x4_SSE(float *A, float *B, const int lda, - const int ldb) const; - inline void transpose_block_SSE4x4(float *A, float *B, const int n, - const int m, const int lda, - const int ldb, - const int block_size) const; + inline void fast_transpose_to(const matrix &target) const; + inline void transpose4x4_SSE(float *A, float *B, const int lda, const int ldb) const; + inline void transpose_block_SSE4x4(float *A, + float *B, + const int n, + const int m, + const int lda, + const int ldb, + const int block_size) const; #endif - inline void transpose_scalar_block(float *A, float *B, const int lda, - const int ldb, const int block_size) const; - inline void transpose_block(float *A, float *B, const int n, - const int m, const int lda, - const int ldb, - const int block_size) const; - public: - matrix(const int nrows = 0, const int ncols = 0); - matrix(const matrix &toCopy); - ~matrix(); - void set_size(const int nrows, const int ncols); - void free(); - int nr() const; - int nc() const; - T& operator()(const int row, const int col) const; - - void swap(matrix &swapTarget); - - //for use with librtprocess - T* operator[](const int index) const // use with indices - { - return ptr[index]; - } - operator T**()// pointer to T** - { - return ptr; - } - operator T*()// pointer to data - { - return data; - } - - //template //Never gets called if use matrix - matrix& operator=(const matrix &toCopy); - matrix& operator=(matrix &&toMove); - template - matrix& operator=(const U value); - template - const matrix add (const matrix &rhs) const; - template - const matrix add (const U value) const ; - template - const matrix& add_this(const U value); - template - const matrix subtract (const matrix &rhs) const ; - template - const matrix subtract (const U value) const; - template - const matrix pointmult (const matrix &rhs) const; - template - const matrix mult (const U value) const; - template - const matrix& mult_this(const U value); - template - const matrix divide (const U value) const; - inline void transpose_to(const matrix &target) const; - double sum(); - T max(); - T min(); - double mean(); - double variance(); + inline void transpose_scalar_block(float *A, float *B, const int lda, const int ldb, const int block_size) const; + inline void transpose_block(float *A, + float *B, + const int n, + const int m, + const int lda, + const int ldb, + const int block_size) const; + +public: + matrix(const int nrows = 0, const int ncols = 0); + matrix(const matrix &toCopy); + ~matrix(); + void set_size(const int nrows, const int ncols); + void free(); + int nr() const; + int nc() const; + T &operator()(const int row, const int col) const; + + void swap(matrix &swapTarget); + + // for use with librtprocess + T *operator[](const int index) const// use with indices + { + return ptr[index]; + } + operator T **()// pointer to T** + { + return ptr; + } + operator T *()// pointer to data + { + return data; + } + + // template //Never gets called if use matrix + matrix &operator=(const matrix &toCopy); + matrix &operator=(matrix &&toMove); + template matrix &operator=(const U value); + template const matrix add(const matrix &rhs) const; + template const matrix add(const U value) const; + template const matrix &add_this(const U value); + template const matrix subtract(const matrix &rhs) const; + template const matrix subtract(const U value) const; + template const matrix pointmult(const matrix &rhs) const; + template const matrix mult(const U value) const; + template const matrix &mult_this(const U value); + template const matrix divide(const U value) const; + inline void transpose_to(const matrix &target) const; + double sum(); + T max(); + T min(); + double mean(); + double variance(); }; -template -double sum(matrix &mat); +template double sum(matrix &mat); -template -T max(matrix &mat); +template T max(matrix &mat); -template -T min(matrix &mat); +template T min(matrix &mat); -template -double mean(matrix &mat); +template double mean(matrix &mat); -template -double variance(matrix &mat); +template double variance(matrix &mat); -template -const matrix operator+(const matrix &mat1, const matrix &mat2); +template const matrix operator+(const matrix &mat1, const matrix &mat2); -template -const matrix operator+(const U value, const matrix &mat); +template const matrix operator+(const U value, const matrix &mat); -template -const matrix operator+(const matrix &mat, const U value); +template const matrix operator+(const matrix &mat, const U value); -template -const matrix operator+=(matrix &mat, const U value); +template const matrix operator+=(matrix &mat, const U value); -template -const matrix operator-(const matrix &mat1, const matrix &mat2); +template const matrix operator-(const matrix &mat1, const matrix &mat2); -template -const matrix operator-(const matrix &mat, const U value); +template const matrix operator-(const matrix &mat, const U value); -template -const matrix operator%(const matrix &mat1, const matrix &mat2); +template const matrix operator%(const matrix &mat1, const matrix &mat2); -template -const matrix operator*(const U value, const matrix &mat); +template const matrix operator*(const U value, const matrix &mat); -template -const matrix operator*(const matrix &mat, const U value); +template const matrix operator*(const matrix &mat, const U value); -template -const matrix operator*=(matrix &mat, const U value); +template const matrix operator*=(matrix &mat, const U value); -template -const matrix operator/(const matrix &mat1, const U value); +template const matrix operator/(const matrix &mat1, const U value); // IMPLEMENTATION: -template -matrix::matrix(const int nrows, const int ncols) -{ - assert(nrows >= 0 && ncols >= 0); - num_rows = nrows; - num_cols = ncols; - if (nrows == 0 || ncols == 0) - { - ptr = nullptr; - data = nullptr; - ua_data = nullptr; - } - else - { - std::size_t ua_num_elements = num_rows*num_cols + 16; - ua_data = new T[ua_num_elements]; - void * buffer = ua_data; - void * aligned_ptr = std::align(16, sizeof(T), buffer, ua_num_elements); - data = static_cast(aligned_ptr); - //data = new T[nrows*ncols]; - ptr = new T*[nrows]; - for(int row = 0; row < nrows; row++) - ptr[row] = data + row*num_cols; - } +template matrix::matrix(const int nrows, const int ncols) +{ + assert(nrows >= 0 && ncols >= 0); + num_rows = nrows; + num_cols = ncols; + if (nrows == 0 || ncols == 0) { + ptr = nullptr; + data = nullptr; + ua_data = nullptr; + } else { + std::size_t ua_num_elements = num_rows * num_cols + 16; + ua_data = new T[ua_num_elements]; + void *buffer = ua_data; + void *aligned_ptr = std::align(16, sizeof(T), buffer, ua_num_elements); + data = static_cast(aligned_ptr); + // data = new T[nrows*ncols]; + ptr = new T *[nrows]; + for (int row = 0; row < nrows; row++) ptr[row] = data + row * num_cols; + } } -template -matrix::matrix(const matrix &toCopy) +template matrix::matrix(const matrix &toCopy) { - if(this == &toCopy) - return; + if (this == &toCopy) return; - num_rows = toCopy.num_rows; - num_cols = toCopy.num_cols; + num_rows = toCopy.num_rows; + num_cols = toCopy.num_cols; - std::size_t ua_num_elements = num_rows*num_cols + 16; - ua_data = new T[ua_num_elements]; - void * buffer = ua_data; - void * aligned_ptr = std::align(16, sizeof(T), buffer, ua_num_elements); - data = static_cast(aligned_ptr); - //data = new T[num_rows*num_cols]; + std::size_t ua_num_elements = num_rows * num_cols + 16; + ua_data = new T[ua_num_elements]; + void *buffer = ua_data; + void *aligned_ptr = std::align(16, sizeof(T), buffer, ua_num_elements); + data = static_cast(aligned_ptr); + // data = new T[num_rows*num_cols]; - ptr = new T*[num_rows]; - for(int row = 0; row < num_rows; row++) - ptr[row] = data + row*num_cols; + ptr = new T *[num_rows]; + for (int row = 0; row < num_rows; row++) ptr[row] = data + row * num_cols; #pragma omp parallel for - for(int row = 0; row < num_rows; row++) - for(int col = 0; col < num_cols; col++) - data[row*num_cols + col] = - toCopy.data[row*num_cols + col]; + for (int row = 0; row < num_rows; row++) + for (int col = 0; col < num_cols; col++) data[row * num_cols + col] = toCopy.data[row * num_cols + col]; } -template -matrix::~matrix() +template matrix::~matrix() { - if(ua_data) - delete [] ua_data; - if(ptr) - delete [] ptr; + if (ua_data) delete[] ua_data; + if (ptr) delete[] ptr; } -template -void matrix::set_size(const int nrows, const int ncols) +template void matrix::set_size(const int nrows, const int ncols) { - assert(nrows >= 0 && ncols >= 0); - if (num_rows == nrows && num_cols == ncols) { - return; - } + assert(nrows >= 0 && ncols >= 0); + if (num_rows == nrows && num_cols == ncols) { return; } - num_rows = nrows; - num_cols = ncols; - if(ua_data) - delete [] ua_data; - if(ptr) - delete [] ptr; - std::size_t ua_num_elements = num_rows*num_cols + 16; - ua_data = new (std::nothrow) T[ua_num_elements]; - if (ua_data == nullptr) - std::cout << "matrix::set_size memory could not be alloc'd" << std::endl; - void * buffer = ua_data; - void * aligned_ptr = std::align(16, sizeof(T), buffer, ua_num_elements); - data = static_cast(aligned_ptr); - //data = new (std::nothrow) T[nrows*ncols]; - //if (data == nullptr) - // std::cout << "matrix::set_size memory could not be alloc'd" << std::endl; + num_rows = nrows; + num_cols = ncols; + if (ua_data) delete[] ua_data; + if (ptr) delete[] ptr; + std::size_t ua_num_elements = num_rows * num_cols + 16; + ua_data = new (std::nothrow) T[ua_num_elements]; + if (ua_data == nullptr) std::cout << "matrix::set_size memory could not be alloc'd" << std::endl; + void *buffer = ua_data; + void *aligned_ptr = std::align(16, sizeof(T), buffer, ua_num_elements); + data = static_cast(aligned_ptr); + // data = new (std::nothrow) T[nrows*ncols]; + // if (data == nullptr) + // std::cout << "matrix::set_size memory could not be alloc'd" << std::endl; - ptr = new (std::nothrow) T*[nrows]; - for(int row = 0; row < nrows; row++) - ptr[row] = data + row*num_cols; + ptr = new (std::nothrow) T *[nrows]; + for (int row = 0; row < nrows; row++) ptr[row] = data + row * num_cols; } -template -void matrix::free() -{ - set_size(0,0); -} +template void matrix::free() { set_size(0, 0); } -template -int matrix::nr() const -{ - return num_rows; -} +template int matrix::nr() const { return num_rows; } -template -int matrix::nc() const -{ - return num_cols; -} +template int matrix::nc() const { return num_cols; } -template -T& matrix::operator()(const int row, const int col) const +template T &matrix::operator()(const int row, const int col) const { - assert(row < num_rows && col < num_cols); - return data[row*num_cols + col]; + assert(row < num_rows && col < num_cols); + return data[row * num_cols + col]; } -template //template -matrix& matrix::operator=(const matrix &toCopy) +template// template +matrix &matrix::operator=(const matrix &toCopy) { - if(this == &toCopy) - return *this; + if (this == &toCopy) return *this; - set_size(toCopy.nr(),toCopy.nc()); + set_size(toCopy.nr(), toCopy.nc()); #pragma omp parallel for shared(toCopy) - for(int row = 0; row < num_rows; row++) - for(int col = 0; col < num_cols; col++) - data[row*num_cols + col] = - toCopy.data[row*num_cols + col]; - return *this; -} - -template //template -matrix& matrix::operator=(matrix &&toMove) -{ - if(this != &toMove) { - delete [] ua_data; - ua_data = toMove.ua_data; - toMove.ua_data = nullptr; - data = toMove.data; - toMove.data = nullptr; - delete [] ptr; - ptr = toMove.ptr; - toMove.ptr = nullptr; - num_rows = toMove.num_rows; - toMove.num_rows = 0; - num_cols = toMove.num_cols; - toMove.num_cols = 0; - } - return *this; -} - -template -void matrix::swap(matrix &swapTarget) -{ - if(this != &swapTarget) { - auto temp_ua = ua_data; - ua_data = swapTarget.ua_data; - swapTarget.ua_data = temp_ua; - auto temp_data = data; - data = swapTarget.data; - swapTarget.data = temp_data; - auto temp_ptr = ptr; - ptr = swapTarget.ptr; - swapTarget.ptr = temp_ptr; - auto temp_nr = num_rows; - num_rows = swapTarget.num_rows; - swapTarget.num_rows = temp_nr; - auto temp_nc = num_cols; - num_cols = swapTarget.num_cols; - swapTarget.num_cols = temp_nc; - } -} - -template template -matrix& matrix::operator=(const U value) + for (int row = 0; row < num_rows; row++) + for (int col = 0; col < num_cols; col++) data[row * num_cols + col] = toCopy.data[row * num_cols + col]; + return *this; +} + +template// template +matrix &matrix::operator=(matrix &&toMove) +{ + if (this != &toMove) { + delete[] ua_data; + ua_data = toMove.ua_data; + toMove.ua_data = nullptr; + data = toMove.data; + toMove.data = nullptr; + delete[] ptr; + ptr = toMove.ptr; + toMove.ptr = nullptr; + num_rows = toMove.num_rows; + toMove.num_rows = 0; + num_cols = toMove.num_cols; + toMove.num_cols = 0; + } + return *this; +} + +template void matrix::swap(matrix &swapTarget) +{ + if (this != &swapTarget) { + auto temp_ua = ua_data; + ua_data = swapTarget.ua_data; + swapTarget.ua_data = temp_ua; + auto temp_data = data; + data = swapTarget.data; + swapTarget.data = temp_data; + auto temp_ptr = ptr; + ptr = swapTarget.ptr; + swapTarget.ptr = temp_ptr; + auto temp_nr = num_rows; + num_rows = swapTarget.num_rows; + swapTarget.num_rows = temp_nr; + auto temp_nc = num_cols; + num_cols = swapTarget.num_cols; + swapTarget.num_cols = temp_nc; + } +} + +template template matrix &matrix::operator=(const U value) { #pragma omp parallel for - for(int row = 0; row < num_rows; row++) - for(int col = 0; col < num_cols; col++) - data[row*num_cols + col] = value; - return *this; + for (int row = 0; row < num_rows; row++) + for (int col = 0; col < num_cols; col++) data[row * num_cols + col] = value; + return *this; } -template template -const matrix matrix::add(const matrix &rhs) const +template template const matrix matrix::add(const matrix &rhs) const { - assert(num_rows == rhs.num_rows && num_cols == rhs.num_cols); - matrix result(num_rows,num_cols); + assert(num_rows == rhs.num_rows && num_cols == rhs.num_cols); + matrix result(num_rows, num_cols); - T* pdata = data; - int pnum_cols = num_cols; -#pragma omp parallel for shared(pdata,pnum_cols,result,rhs) - for(int row = 0; row < num_rows; row++) - for(int col = 0; col < num_cols; col++) - result.data[row*num_cols + col] = - data[row*num_cols + col] + - rhs.data[row*num_cols + col]; - return result; + T *pdata = data; + int pnum_cols = num_cols; +#pragma omp parallel for shared(pdata, pnum_cols, result, rhs) + for (int row = 0; row < num_rows; row++) + for (int col = 0; col < num_cols; col++) + result.data[row * num_cols + col] = data[row * num_cols + col] + rhs.data[row * num_cols + col]; + return result; } -template template -const matrix matrix::add(const U value) const +template template const matrix matrix::add(const U value) const { - matrix result(num_rows,num_cols); + matrix result(num_rows, num_cols); - T* pdata = data; - int pnum_cols = num_cols; -#pragma omp parallel for shared(pdata,pnum_cols) - for(int row = 0; row < num_rows; row++) - for(int col = 0; col < num_cols; col++) - result.data[row*num_cols + col] = - data[row*num_cols + col] + - value; - return result; + T *pdata = data; + int pnum_cols = num_cols; +#pragma omp parallel for shared(pdata, pnum_cols) + for (int row = 0; row < num_rows; row++) + for (int col = 0; col < num_cols; col++) result.data[row * num_cols + col] = data[row * num_cols + col] + value; + return result; } -template template -const matrix& matrix::add_this(const U value) +template template const matrix &matrix::add_this(const U value) { - T* pdata = data; - int pnum_cols = num_cols; -#pragma omp parallel for shared(pdata,pnum_cols) - for(int row = 0; row < num_rows; row++) - for(int col = 0; col < num_cols; col++) - data[row*num_cols + col] += value; - return *this; + T *pdata = data; + int pnum_cols = num_cols; +#pragma omp parallel for shared(pdata, pnum_cols) + for (int row = 0; row < num_rows; row++) + for (int col = 0; col < num_cols; col++) data[row * num_cols + col] += value; + return *this; } -template template -const matrix matrix::subtract(const matrix &rhs) const +template template const matrix matrix::subtract(const matrix &rhs) const { - assert(num_rows == rhs.num_rows && num_cols == rhs.num_cols); - matrix result(num_rows,num_cols); + assert(num_rows == rhs.num_rows && num_cols == rhs.num_cols); + matrix result(num_rows, num_cols); - T* pdata = data; - int pnum_cols = num_cols; -#pragma omp parallel for shared(pdata,pnum_cols,result,rhs) - for(int row = 0; row < num_rows; row++) - for(int col = 0; col < num_cols; col++) - result.data[row*num_cols + col] = - data[row*num_cols + col] - - rhs.data[row*num_cols + col]; - return result; + T *pdata = data; + int pnum_cols = num_cols; +#pragma omp parallel for shared(pdata, pnum_cols, result, rhs) + for (int row = 0; row < num_rows; row++) + for (int col = 0; col < num_cols; col++) + result.data[row * num_cols + col] = data[row * num_cols + col] - rhs.data[row * num_cols + col]; + return result; } -template template -const matrix matrix::subtract(const U value) const +template template const matrix matrix::subtract(const U value) const { - matrix result(num_rows,num_cols); + matrix result(num_rows, num_cols); - T* pdata = data; - int pnum_cols = num_cols; -#pragma omp parallel for shared(pdata,pnum_cols,result) - for(int row = 0; row < num_rows; row++) - for(int col = 0; col < num_cols; col++) - result.data[row*num_cols + col] = - data[row*num_cols + col] + - value; - return result; + T *pdata = data; + int pnum_cols = num_cols; +#pragma omp parallel for shared(pdata, pnum_cols, result) + for (int row = 0; row < num_rows; row++) + for (int col = 0; col < num_cols; col++) result.data[row * num_cols + col] = data[row * num_cols + col] + value; + return result; } -template template -const matrix matrix::pointmult(const matrix &rhs) const +template template const matrix matrix::pointmult(const matrix &rhs) const { - matrix result(num_rows,num_cols); + matrix result(num_rows, num_cols); -#pragma omp parallel for shared(result,rhs) - for(int row = 0; row < num_rows; row++) - for(int col = 0; col < num_cols; col++) - result.data[row*num_cols + col] = - data[row*num_cols + col] * - rhs.data[row*num_cols + col]; - return result; +#pragma omp parallel for shared(result, rhs) + for (int row = 0; row < num_rows; row++) + for (int col = 0; col < num_cols; col++) + result.data[row * num_cols + col] = data[row * num_cols + col] * rhs.data[row * num_cols + col]; + return result; } -template template -const matrix matrix::mult(const U value) const +template template const matrix matrix::mult(const U value) const { - matrix result(num_rows,num_cols); + matrix result(num_rows, num_cols); #pragma omp parallel for shared(result) - for(int row = 0; row < num_rows; row++) - for(int col = 0; col < num_cols; col++) - result.data[row*num_cols + col] = - data[row*num_cols + col] * - value; - return result; + for (int row = 0; row < num_rows; row++) + for (int col = 0; col < num_cols; col++) result.data[row * num_cols + col] = data[row * num_cols + col] * value; + return result; } -template template -const matrix& matrix::mult_this(const U value) +template template const matrix &matrix::mult_this(const U value) { #pragma omp parallel for - for(int row = 0; row < num_rows; row++) - for(int col = 0; col < num_cols; col++) - data[row*num_cols + col] *= value; - return *this; + for (int row = 0; row < num_rows; row++) + for (int col = 0; col < num_cols; col++) data[row * num_cols + col] *= value; + return *this; } -template template -const matrix matrix::divide(const U value) const +template template const matrix matrix::divide(const U value) const { - matrix result(num_rows,num_cols); + matrix result(num_rows, num_cols); #pragma omp parallel for shared(result) - for(int row = 0; row < num_rows; row++) - for(int col = 0; col < num_cols; col++) - result.data[row*num_cols + col] = - data[row*num_cols + col] / - value; - return result; + for (int row = 0; row < num_rows; row++) + for (int col = 0; col < num_cols; col++) result.data[row * num_cols + col] = data[row * num_cols + col] / value; + return result; } -template -inline void matrix::slow_transpose_to (const matrix &target) const +template inline void matrix::slow_transpose_to(const matrix &target) const { - assert(target.num_rows == num_cols && target.num_cols == num_rows); + assert(target.num_rows == num_cols && target.num_cols == num_rows); #pragma omp parallel for shared(target) - for(int row = 0; row < num_rows; row++) - for(int col = 0; col < num_cols; col++) - target.data[col*num_rows + row] = - data[row*num_cols + col]; + for (int row = 0; row < num_rows; row++) + for (int col = 0; col < num_cols; col++) target.data[col * num_rows + row] = data[row * num_cols + col]; } #ifdef __SSE2__ -template<> -inline void matrix::fast_transpose_to (const matrix &target) const +template<> inline void matrix::fast_transpose_to(const matrix &target) const { - assert(target.num_rows == num_cols && target.num_cols == num_rows); + assert(target.num_rows == num_cols && target.num_cols == num_rows); - transpose_block_SSE4x4(data,target.data,num_rows,num_cols, - num_cols,num_rows, 16); + transpose_block_SSE4x4(data, target.data, num_rows, num_cols, num_cols, num_rows, 16); } -//There is no fast transpose in the general case -template -inline void matrix::fast_transpose_to (const matrix &target) const -{ - slow_transpose_to(target); -} +// There is no fast transpose in the general case +template inline void matrix::fast_transpose_to(const matrix &target) const { slow_transpose_to(target); } #endif -template -inline void matrix::transpose_to (const matrix &target) const -{ - slow_transpose_to(target); -} +template inline void matrix::transpose_to(const matrix &target) const { slow_transpose_to(target); } -template<> -inline void matrix::transpose_to (const matrix &target) const +template<> inline void matrix::transpose_to(const matrix &target) const { #ifdef __SSE2__ - //Fast transpose only work with matricies with dimensions of multiples of 16 - if((num_rows%16 != 0) || (num_cols%16 !=0)) + // Fast transpose only work with matricies with dimensions of multiples of 16 + if ((num_rows % 16 != 0) || (num_cols % 16 != 0)) #endif - slow_transpose_to(target); + slow_transpose_to(target); #ifdef __SSE2__ - else - fast_transpose_to(target); + else + fast_transpose_to(target); #endif } #ifdef __SSE2__ -template -inline void matrix::transpose4x4_SSE(float *A, float *B, const int lda, - const int ldb) const +template inline void matrix::transpose4x4_SSE(float *A, float *B, const int lda, const int ldb) const { - __m128 row1 = _mm_load_ps(&A[0*lda]); - __m128 row2 = _mm_load_ps(&A[1*lda]); - __m128 row3 = _mm_load_ps(&A[2*lda]); - __m128 row4 = _mm_load_ps(&A[3*lda]); - _MM_TRANSPOSE4_PS(row1, row2, row3, row4); - _mm_store_ps(&B[0*ldb], row1); - _mm_store_ps(&B[1*ldb], row2); - _mm_store_ps(&B[2*ldb], row3); - _mm_store_ps(&B[3*ldb], row4); + __m128 row1 = _mm_load_ps(&A[0 * lda]); + __m128 row2 = _mm_load_ps(&A[1 * lda]); + __m128 row3 = _mm_load_ps(&A[2 * lda]); + __m128 row4 = _mm_load_ps(&A[3 * lda]); + _MM_TRANSPOSE4_PS(row1, row2, row3, row4); + _mm_store_ps(&B[0 * ldb], row1); + _mm_store_ps(&B[1 * ldb], row2); + _mm_store_ps(&B[2 * ldb], row3); + _mm_store_ps(&B[3 * ldb], row4); } -//block_size = 16 works best +// block_size = 16 works best template -inline void matrix::transpose_block_SSE4x4(float *A, float *B, const int n, - const int m, const int lda, const int ldb, - const int block_size) const -{ - #pragma omp parallel for - for(int i=0; i::transpose_block_SSE4x4(float *A, + float *B, + const int n, + const int m, + const int lda, + const int ldb, + const int block_size) const +{ +#pragma omp parallel for + for (int i = 0; i < n; i += block_size) + for (int j = 0; j < m; j += block_size) { + int max_i2 = i + block_size < n ? i + block_size : n; + int max_j2 = j + block_size < m ? j + block_size : m; + for (int i2 = i; i2 < max_i2; i2 += 4) + for (int j2 = j; j2 < max_j2; j2 += 4) transpose4x4_SSE(&A[i2 * lda + j2], &B[j2 * ldb + i2], lda, ldb); + } } #endif template -inline void matrix::transpose_scalar_block(float *A, float *B, const int lda, const int ldb, const int block_size) const { - #pragma omp parallel for - for(int i=0; i::transpose_scalar_block(float *A, float *B, const int lda, const int ldb, const int block_size) const +{ +#pragma omp parallel for + for (int i = 0; i < block_size; i++) { + for (int j = 0; j < block_size; j++) { B[j * ldb + i] = A[i * lda + j]; } + } } template -inline void matrix::transpose_block(float *A, float *B, const int n, const int m, const int lda, const int ldb, const int block_size) const { - #pragma omp parallel for - for(int i=0; i::transpose_block(float *A, + float *B, + const int n, + const int m, + const int lda, + const int ldb, + const int block_size) const +{ +#pragma omp parallel for + for (int i = 0; i < n; i += block_size) { + for (int j = 0; j < m; j += block_size) { + transpose_scalar_block(&A[i * lda + j], &B[j * ldb + i], lda, ldb, block_size); } + } } -template -double matrix::sum() +template double matrix::sum() { - double sum = 0; + double sum = 0; -#pragma omp parallel for reduction(+:sum) - for(int row = 0; row < num_rows; row++) - for(int col = 0; col < num_cols; col++) - sum += data[row*num_cols + col]; - return sum; +#pragma omp parallel for reduction(+ : sum) + for (int row = 0; row < num_rows; row++) + for (int col = 0; col < num_cols; col++) sum += data[row * num_cols + col]; + return sum; } -template -T matrix::max() +template T matrix::max() { - T max = std::numeric_limits::min(); + T max = std::numeric_limits::min(); - #pragma omp parallel for reduction(max:max) schedule(dynamic,16) - for(int row = 0; row < num_rows; row++) - for(int col = 0; col < num_cols; col++) - max = std::max(data[row*num_cols + col],max); - return max; +#pragma omp parallel for reduction(max : max) schedule(dynamic, 16) + for (int row = 0; row < num_rows; row++) + for (int col = 0; col < num_cols; col++) max = std::max(data[row * num_cols + col], max); + return max; } -template -T matrix::min() +template T matrix::min() { - T min = std::numeric_limits::max(); + T min = std::numeric_limits::max(); - #pragma omp parallel for reduction(min:min) schedule(dynamic,16) - for(int row = 0; row < num_rows; row++) - for(int col = 0; col < num_cols; col++) - min = std::min(data[row*num_cols + col],min); - return min; +#pragma omp parallel for reduction(min : min) schedule(dynamic, 16) + for (int row = 0; row < num_rows; row++) + for (int col = 0; col < num_cols; col++) min = std::min(data[row * num_cols + col], min); + return min; } -template -double matrix::mean() +template double matrix::mean() { - assert(num_rows > 0 && num_cols > 0); - double size = num_rows*num_cols; - return sum()/size; + assert(num_rows > 0 && num_cols > 0); + double size = num_rows * num_cols; + return sum() / size; } -template -double matrix::variance() +template double matrix::variance() { - double m = mean(); - double size = num_rows*num_cols; - double variance = 0; + double m = mean(); + double size = num_rows * num_cols; + double variance = 0; -#pragma omp parallel for reduction(+:variance) - for(int row = 0; row < num_rows; row++) - for(int col = 0; col < num_cols; col++) - variance += pow(data[row*num_cols+col]-m,2); - return variance/size; +#pragma omp parallel for reduction(+ : variance) + for (int row = 0; row < num_rows; row++) + for (int col = 0; col < num_cols; col++) variance += pow(data[row * num_cols + col] - m, 2); + return variance / size; } -//Non object functions +// Non object functions -template -double sum(matrix &mat) -{ - return mat.sum(); -} +template double sum(matrix &mat) { return mat.sum(); } -template -T max(matrix &mat) -{ - return mat.max(); -} +template T max(matrix &mat) { return mat.max(); } -template -T min(matrix &mat) -{ - return mat.min(); -} +template T min(matrix &mat) { return mat.min(); } -template -double mean(matrix &mat) -{ - return mat.mean(); -} +template double mean(matrix &mat) { return mat.mean(); } -template -double variance(matrix &mat) -{ - return mat.variance(); -} +template double variance(matrix &mat) { return mat.variance(); } -template -const matrix operator+(const matrix &mat1, const matrix &mat2) +template const matrix operator+(const matrix &mat1, const matrix &mat2) { - return mat1.add(mat2); + return mat1.add(mat2); } -template -const matrix operator+(const U value, const matrix &mat) -{ - return mat.add(value); -} +template const matrix operator+(const U value, const matrix &mat) { return mat.add(value); } -template -const matrix operator+(const matrix &mat, const U value) -{ - return mat.add(value); -} +template const matrix operator+(const matrix &mat, const U value) { return mat.add(value); } -template -const matrix operator+=(matrix &mat, const U value) -{ - return mat.add_this(value); -} +template const matrix operator+=(matrix &mat, const U value) { return mat.add_this(value); } -template -const matrix operator-(const matrix &mat1, const matrix &mat2) +template const matrix operator-(const matrix &mat1, const matrix &mat2) { - return mat1.subtract(mat2); + return mat1.subtract(mat2); } -template -const matrix operator-(const matrix &mat, const U value) -{ - return mat.subtact(value); -} +template const matrix operator-(const matrix &mat, const U value) { return mat.subtact(value); } -template -const matrix operator%(const matrix &mat1, const matrix &mat2) +template const matrix operator%(const matrix &mat1, const matrix &mat2) { - return mat1.pointmult(mat2); + return mat1.pointmult(mat2); } -template -const matrix operator*(const U value, const matrix &mat) -{ - return mat.mult(value); -} +template const matrix operator*(const U value, const matrix &mat) { return mat.mult(value); } -template -const matrix operator*(const matrix &mat, const U value) -{ - return mat.mult(value); -} +template const matrix operator*(const matrix &mat, const U value) { return mat.mult(value); } -template -const matrix operator*=(matrix &mat, const U value) -{ - return mat.mult_this(value); -} +template const matrix operator*=(matrix &mat, const U value) { return mat.mult_this(value); } -template -const matrix operator/(const matrix &mat, const U value) -{ - return mat.divide(value); -} +template const matrix operator/(const matrix &mat, const U value) { return mat.divide(value); } -#endif //MATRIX_H +#endif// MATRIX_H diff --git a/filmulator-gui/core/mergeExps.cpp b/filmulator-gui/core/mergeExps.cpp index de5956d8..65830a71 100644 --- a/filmulator-gui/core/mergeExps.cpp +++ b/filmulator-gui/core/mergeExps.cpp @@ -1,4 +1,4 @@ -/* +/* * This file is part of Filmulator. * * Copyright 2013 Omer Mano and Carlo Vaccari @@ -25,95 +25,85 @@ #include #include -bool merge_exps(matrix &input_image, const matrix &temp_image, - float &exposure_weight, float initial_exposure_comp, - float &last_exposure_factor, string filename, - float input_exposure_comp) +bool merge_exps(matrix &input_image, + const matrix &temp_image, + float &exposure_weight, + float initial_exposure_comp, + float &last_exposure_factor, + string filename, + float input_exposure_comp) { -//First order of business is to determine the relative brightness of the two -//images (we don't trust the exposure compensation value given by the user to -//be correct or even if it is, perfectly accurate). - int numrows = input_image.nr(); - int numcols = input_image.nc(); - double init_exp_factor = pow(2,-initial_exposure_comp); - double input_exp_factor = pow(2,-input_exposure_comp); - double input_sum = init_exp_factor;//only initially - double exp_factor; - double temp_sum = 0; - int i = 0; - srand(time(NULL)); - int xcoord; - int ycoord; - //This samples points that are valid in both images - while (i < max(max(numrows, numcols/3),300)) - { - xcoord = int(fmod(double(rand())*double(rand()), double(numcols/3))); - ycoord = int(fmod(double(rand())*double(rand()), double(numrows))); - if(input_image(ycoord,xcoord*3)/init_exp_factor < 60000 && - input_image(ycoord,xcoord*3+1)/init_exp_factor < 60000 && - input_image(ycoord,xcoord*3+2)/init_exp_factor < 60000 && - temp_image(ycoord,xcoord*3) < 60000 && - temp_image(ycoord,xcoord*3+1) < 60000 && - temp_image(ycoord,xcoord*3+2) < 60000) - { - input_sum += double(input_image(ycoord,xcoord*3)) - + double(input_image(ycoord,xcoord*3+1)) - + double(input_image(ycoord,xcoord*3+2)); - temp_sum += double(temp_image(ycoord,xcoord*3)) - + double(temp_image(ycoord,xcoord*3+1)) - + double(temp_image(ycoord,xcoord*3+2)); - i++; - } + // First order of business is to determine the relative brightness of the two + // images (we don't trust the exposure compensation value given by the user to + // be correct or even if it is, perfectly accurate). + int numrows = input_image.nr(); + int numcols = input_image.nc(); + double init_exp_factor = pow(2, -initial_exposure_comp); + double input_exp_factor = pow(2, -input_exposure_comp); + double input_sum = init_exp_factor;// only initially + double exp_factor; + double temp_sum = 0; + int i = 0; + srand(time(NULL)); + int xcoord; + int ycoord; + // This samples points that are valid in both images + while (i < max(max(numrows, numcols / 3), 300)) { + xcoord = int(fmod(double(rand()) * double(rand()), double(numcols / 3))); + ycoord = int(fmod(double(rand()) * double(rand()), double(numrows))); + if (input_image(ycoord, xcoord * 3) / init_exp_factor < 60000 + && input_image(ycoord, xcoord * 3 + 1) / init_exp_factor < 60000 + && input_image(ycoord, xcoord * 3 + 2) / init_exp_factor < 60000 && temp_image(ycoord, xcoord * 3) < 60000 + && temp_image(ycoord, xcoord * 3 + 1) < 60000 && temp_image(ycoord, xcoord * 3 + 2) < 60000) { + input_sum += double(input_image(ycoord, xcoord * 3)) + double(input_image(ycoord, xcoord * 3 + 1)) + + double(input_image(ycoord, xcoord * 3 + 2)); + temp_sum += double(temp_image(ycoord, xcoord * 3)) + double(temp_image(ycoord, xcoord * 3 + 1)) + + double(temp_image(ycoord, xcoord * 3 + 2)); + i++; } - exp_factor = input_sum/temp_sum; - //This checks that the image is brighter than the previous one, within 5% - //error. - if(exp_factor > (last_exposure_factor*1.05)) - { - cerr << "Image " << filename << " is darker than the previous image. Exiting..." << endl; - return true; - } - //Now we merge the exposures. - //The weighting of the new image is the ratio of the first exposure factor - //to the current exposure factor. - //The first exposure factor should be the biggest, so the greater the - //ratio the more the weighting should be. - float new_exp_weight = init_exp_factor/exp_factor; - int temp_max_channel; - int clip_thresh = 61000;//If any channel exceeds this value, we start - // to roll off its weight. - float exp_weight_factor;//This will store the weighting for the rolloff. - int colr, colg, colb;//hold indices for red, green, and blue columns - for (int col = 0; col < numcols/3; col++) - { - colr = col*3; - colg = col*3+1; - colb = col*3+2; - for (int row = 0; row < numrows; row++) - { - temp_max_channel = max(max(temp_image(row,colr), - temp_image(row,colg)),temp_image(row,colb)); - // Here we compute the weighting - exp_weight_factor = min(1,max(clip_thresh-temp_max_channel,0)/1000); + } + exp_factor = input_sum / temp_sum; + // This checks that the image is brighter than the previous one, within 5% + // error. + if (exp_factor > (last_exposure_factor * 1.05)) { + cerr << "Image " << filename << " is darker than the previous image. Exiting..." << endl; + return true; + } + // Now we merge the exposures. + // The weighting of the new image is the ratio of the first exposure factor + // to the current exposure factor. + // The first exposure factor should be the biggest, so the greater the + // ratio the more the weighting should be. + float new_exp_weight = init_exp_factor / exp_factor; + int temp_max_channel; + int clip_thresh = 61000;// If any channel exceeds this value, we start + // to roll off its weight. + float exp_weight_factor;// This will store the weighting for the rolloff. + int colr, colg, colb;// hold indices for red, green, and blue columns + for (int col = 0; col < numcols / 3; col++) { + colr = col * 3; + colg = col * 3 + 1; + colb = col * 3 + 2; + for (int row = 0; row < numrows; row++) { + temp_max_channel = max(max(temp_image(row, colr), temp_image(row, colg)), temp_image(row, colb)); + // Here we compute the weighting + exp_weight_factor = min(1, max(clip_thresh - temp_max_channel, 0) / 1000); - //Here we do a weighted average of the cumulative exposure - //and the corrected (multiplied by exp_factor) exposure from the - //more brightly exposed image. - input_image(row,colr) = (input_image(row,colr)* - exposure_weight + temp_image(row,colr)*exp_factor* - new_exp_weight*exp_weight_factor) / - (exposure_weight + new_exp_weight*exp_weight_factor); - input_image(row,colg) = (input_image(row,colg)* - exposure_weight + temp_image(row,colg)*exp_factor* - new_exp_weight*exp_weight_factor) / - (exposure_weight + new_exp_weight*exp_weight_factor); - input_image(row,colb) = (input_image(row,colb)* - exposure_weight + temp_image(row,colb)*exp_factor* - new_exp_weight*exp_weight_factor) / - (exposure_weight + new_exp_weight*exp_weight_factor); - } + // Here we do a weighted average of the cumulative exposure + // and the corrected (multiplied by exp_factor) exposure from the + // more brightly exposed image. + input_image(row, colr) = (input_image(row, colr) * exposure_weight + + temp_image(row, colr) * exp_factor * new_exp_weight * exp_weight_factor) + / (exposure_weight + new_exp_weight * exp_weight_factor); + input_image(row, colg) = (input_image(row, colg) * exposure_weight + + temp_image(row, colg) * exp_factor * new_exp_weight * exp_weight_factor) + / (exposure_weight + new_exp_weight * exp_weight_factor); + input_image(row, colb) = (input_image(row, colb) * exposure_weight + + temp_image(row, colb) * exp_factor * new_exp_weight * exp_weight_factor) + / (exposure_weight + new_exp_weight * exp_weight_factor); } - last_exposure_factor = input_exp_factor;//TODO: Questionable if correct. - exposure_weight = exposure_weight + new_exp_weight; - return false; + } + last_exposure_factor = input_exp_factor;// TODO: Questionable if correct. + exposure_weight = exposure_weight + new_exp_weight; + return false; } diff --git a/filmulator-gui/core/myLibraw.h b/filmulator-gui/core/myLibraw.h index 6736f59e..a54b6279 100644 --- a/filmulator-gui/core/myLibraw.h +++ b/filmulator-gui/core/myLibraw.h @@ -1,31 +1,32 @@ #ifndef MYLIBRAW_H #define MYLIBRAW_H -#include #include +#include -class MyLibRaw : public LibRaw { +class MyLibRaw : public LibRaw +{ public: - void my_phase_one_free_tempbuffer() {phase_one_free_tempbuffer();} - int phaseone_fix(bool &needs_phase_one_free) { - std::cout << "MyLibRaw is phaseone compressed: " << is_phaseone_compressed() << std::endl; - std::cout << "MyLibRaw split_col: " << imgdata.color.phase_one_data.split_col << std::endl; - std::cout << "MyLibRaw split_row: " << imgdata.color.phase_one_data.split_row << std::endl; - if(is_phaseone_compressed() && imgdata.rawdata.raw_alloc) { - phase_one_allocate_tempbuffer(); - needs_phase_one_free = true; - int rc = phase_one_subtract_black((ushort *) imgdata.rawdata.raw_alloc, - imgdata.rawdata.raw_image); - if(rc == 0 && imgdata.params.use_p1_correction) { - std::cout << "using p1 correction" << std::endl; - rc = phase_one_correct(); - std::cout << "after using p1 correction" << std::endl; - return rc; - } - return rc; - } - return 0; + void my_phase_one_free_tempbuffer() { phase_one_free_tempbuffer(); } + int phaseone_fix(bool &needs_phase_one_free) + { + std::cout << "MyLibRaw is phaseone compressed: " << is_phaseone_compressed() << std::endl; + std::cout << "MyLibRaw split_col: " << imgdata.color.phase_one_data.split_col << std::endl; + std::cout << "MyLibRaw split_row: " << imgdata.color.phase_one_data.split_row << std::endl; + if (is_phaseone_compressed() && imgdata.rawdata.raw_alloc) { + phase_one_allocate_tempbuffer(); + needs_phase_one_free = true; + int rc = phase_one_subtract_black((ushort *)imgdata.rawdata.raw_alloc, imgdata.rawdata.raw_image); + if (rc == 0 && imgdata.params.use_p1_correction) { + std::cout << "using p1 correction" << std::endl; + rc = phase_one_correct(); + std::cout << "after using p1 correction" << std::endl; + return rc; + } + return rc; } + return 0; + } }; -#endif // MYLIBRAW_H +#endif// MYLIBRAW_H diff --git a/filmulator-gui/core/nlmeans/.DS_Store b/filmulator-gui/core/nlmeans/.DS_Store new file mode 100644 index 00000000..48beeccd Binary files /dev/null and b/filmulator-gui/core/nlmeans/.DS_Store differ diff --git a/filmulator-gui/core/nlmeans/bisecting_kmeans.cpp b/filmulator-gui/core/nlmeans/bisecting_kmeans.cpp index da475f94..1449144e 100644 --- a/filmulator-gui/core/nlmeans/bisecting_kmeans.cpp +++ b/filmulator-gui/core/nlmeans/bisecting_kmeans.cpp @@ -5,7 +5,8 @@ #include #include -struct clusterInfo { +struct clusterInfo +{ std::vector center; std::vector members; double summedSquareDistances; @@ -14,9 +15,8 @@ struct clusterInfo { // X is in points major, dimensions minor order // Returns center locations (clusters major). Will always return at least two // clusters -std::vector bisecting_kmeans(float *__restrict const X, - const int maxNumClusters, - const float threshold) { +std::vector bisecting_kmeans(float *__restrict const X, const int maxNumClusters, const float threshold) +{ const int numDimensions = patchSize * numChannels; std::vector currentClusters; @@ -26,31 +26,27 @@ std::vector bisecting_kmeans(float *__restrict const X, clusterInfo allPointsClustInfo; allPointsClustInfo.members = std::vector(numPoints); allPointsClustInfo.summedSquareDistances = std::numeric_limits::max(); - std::iota(allPointsClustInfo.members.begin(), - allPointsClustInfo.members.end(), 0); + std::iota(allPointsClustInfo.members.begin(), allPointsClustInfo.members.end(), 0); currentClusters.push_back(allPointsClustInfo); double totalSSD = std::numeric_limits::max(); - while ((currentClusters.size() < maxNumClusters) & - (totalSSD > threshold * numPoints)) { + while ((currentClusters.size() < maxNumClusters) & (totalSSD > threshold * numPoints)) { // Split the cluster with the highest summed squared distance int nextClusterToSplitIdx = -1; float highestSSD = 0; for (int clustIdx = 0; clustIdx < currentClusters.size(); clustIdx++) { auto &thisCluster = currentClusters[clustIdx]; - if ((thisCluster.summedSquareDistances > highestSSD) & - (thisCluster.members.size() > - 2)) { // Don't split a cluster of only 2 points- it just has two - // outliers + if ((thisCluster.summedSquareDistances > highestSSD) + & (thisCluster.members.size() > 2)) {// Don't split a cluster of only 2 points- it just has two + // outliers nextClusterToSplitIdx = clustIdx; highestSSD = thisCluster.summedSquareDistances; } } - if (nextClusterToSplitIdx == - -1) { // all clusters have only two points, so we are done. + if (nextClusterToSplitIdx == -1) {// all clusters have only two points, so we are done. break; } @@ -60,25 +56,21 @@ std::vector bisecting_kmeans(float *__restrict const X, for (int dimIdx = 0; dimIdx < numDimensions; dimIdx++) { for (int pointIdx = 0; pointIdx < numPointsToSplit; pointIdx++) { int sourcePointIdx = clusterToSplit.members[pointIdx]; - clusterToSplitX[pointIdx + dimIdx * numPointsToSplit] = - X[sourcePointIdx + dimIdx * numPoints]; + clusterToSplitX[pointIdx + dimIdx * numPointsToSplit] = X[sourcePointIdx + dimIdx * numPoints]; } } - auto [twoCenters, isInSecondCluster, SSDs, numMembers] = - splitCluster(clusterToSplitX.data(), numPointsToSplit); + auto [twoCenters, isInSecondCluster, SSDs, numMembers] = splitCluster(clusterToSplitX.data(), numPointsToSplit); // If either cluster only has one member, discard that cluster - int discardCluster = std::distance( - numMembers.begin(), std::find(numMembers.begin(), numMembers.end(), - 1)); // takes on values 0,1 or 2 for none + int discardCluster = std::distance(numMembers.begin(), + std::find(numMembers.begin(), + numMembers.end(), + 1));// takes on values 0,1 or 2 for none for (int clustIdx = 0, numInsertedClusters = 0; clustIdx < 2; clustIdx++) { - if (clustIdx == discardCluster) { - continue; - } - auto centerRangeToCopyStart = - twoCenters.begin() + clustIdx * numDimensions; + if (clustIdx == discardCluster) { continue; } + auto centerRangeToCopyStart = twoCenters.begin() + clustIdx * numDimensions; auto centerRangeToCopyEnd = centerRangeToCopyStart + numDimensions; std::vector clusterMembers; @@ -88,8 +80,7 @@ std::vector bisecting_kmeans(float *__restrict const X, } } clusterInfo clustInfo; - clustInfo.center = - std::vector(centerRangeToCopyStart, centerRangeToCopyEnd); + clustInfo.center = std::vector(centerRangeToCopyStart, centerRangeToCopyEnd); clustInfo.members = clusterMembers; clustInfo.summedSquareDistances = SSDs[clustIdx]; @@ -111,8 +102,7 @@ std::vector bisecting_kmeans(float *__restrict const X, std::vector clusterCenters(numClusters * numDimensions); for (int dimIdx = 0; dimIdx < numDimensions; dimIdx++) { for (int clustIdx = 0; clustIdx < numClusters; clustIdx++) { - clusterCenters[clustIdx + dimIdx * numClusters] = - currentClusters[clustIdx].center[dimIdx]; + clusterCenters[clustIdx + dimIdx * numClusters] = currentClusters[clustIdx].center[dimIdx]; } } diff --git a/filmulator-gui/core/nlmeans/calcC1chanT.cpp b/filmulator-gui/core/nlmeans/calcC1chanT.cpp index 2d72e6d7..cf83ea11 100644 --- a/filmulator-gui/core/nlmeans/calcC1chanT.cpp +++ b/filmulator-gui/core/nlmeans/calcC1chanT.cpp @@ -1,46 +1,45 @@ -#include -#include #include "eigen/Eigen/Dense" #include "nlmeans.hpp" +#include +#include -template -Eigen::Matrix -pseudoinverse(const MatT& mat, typename MatT::Scalar tolerance = typename MatT::Scalar{1e-4}) // choose appropriately +template +Eigen::Matrix pseudoinverse(const MatT &mat, + typename MatT::Scalar tolerance = typename MatT::Scalar{ 1e-4 })// choose appropriately { - typedef typename MatT::Scalar Scalar; - auto svd = mat.jacobiSvd(Eigen::ComputeFullU | Eigen::ComputeFullV); - const auto& singularValues = svd.singularValues(); - Eigen::Matrix singularValuesInv(mat.cols(), mat.rows()); - singularValuesInv.setZero(); - for (unsigned int i = 0; i < singularValues.size(); ++i) { - if (singularValues(i) > tolerance) - { - singularValuesInv(i, i) = Scalar{ 1 } / singularValues(i); - } - else - { - singularValuesInv(i, i) = Scalar{ 0 }; - } + typedef typename MatT::Scalar Scalar; + auto svd = mat.jacobiSvd(Eigen::ComputeFullU | Eigen::ComputeFullV); + const auto &singularValues = svd.singularValues(); + Eigen::Matrix singularValuesInv(mat.cols(), mat.rows()); + singularValuesInv.setZero(); + for (unsigned int i = 0; i < singularValues.size(); ++i) { + if (singularValues(i) > tolerance) { + singularValuesInv(i, i) = Scalar{ 1 } / singularValues(i); + } else { + singularValuesInv(i, i) = Scalar{ 0 }; } - return svd.matrixV() * singularValuesInv * svd.matrixU().adjoint(); + } + return svd.matrixV() * singularValuesInv * svd.matrixU().adjoint(); } -//centers is in cluster major order -std::vector calcC1ChanT(std::vector centers, int numClusters, const float h) { +// centers is in cluster major order +std::vector calcC1ChanT(std::vector centers, int numClusters, const float h) +{ - int numDimensions = centers.size() / numClusters; - Eigen::MatrixXf C1(numClusters, numClusters); - for (int row = 0; row < numClusters; row++) { - for (int col = 0; col < numClusters; col++) { - float squaredDistance = 0; - for (int dimIdx = 0; dimIdx < numDimensions; dimIdx++) { - float dist = centers[row + dimIdx * numClusters] - centers[col + dimIdx * numClusters]; - squaredDistance += dist * dist; - } - C1(col, row) = std::exp(-squaredDistance / (2 * h * h)); //We're assigning to (col,row) in order to do the transpose - } + int numDimensions = centers.size() / numClusters; + Eigen::MatrixXf C1(numClusters, numClusters); + for (int row = 0; row < numClusters; row++) { + for (int col = 0; col < numClusters; col++) { + float squaredDistance = 0; + for (int dimIdx = 0; dimIdx < numDimensions; dimIdx++) { + float dist = centers[row + dimIdx * numClusters] - centers[col + dimIdx * numClusters]; + squaredDistance += dist * dist; + } + C1(col, row) = + std::exp(-squaredDistance / (2 * h * h));// We're assigning to (col,row) in order to do the transpose } - auto C1Chan = pseudoinverse(C1); - float* matPtr = C1Chan.data(); - return std::vector(matPtr, matPtr + C1Chan.size()); + } + auto C1Chan = pseudoinverse(C1); + float *matPtr = C1Chan.data(); + return std::vector(matPtr, matPtr + C1Chan.size()); } diff --git a/filmulator-gui/core/nlmeans/calcW_generic.cpp b/filmulator-gui/core/nlmeans/calcW_generic.cpp index f77ab418..dfc272cd 100644 --- a/filmulator-gui/core/nlmeans/calcW_generic.cpp +++ b/filmulator-gui/core/nlmeans/calcW_generic.cpp @@ -1,32 +1,37 @@ -#include +#include "nlmeans.hpp" #include #include -#include "nlmeans.hpp" +#include -//Aguide is called p in the paper, W is called c_k, centers is called mu_k -//requires W be set to 0 -void calcW(float* __restrict const Aguide_ptr, float* __restrict const centers_ptr, - ptrdiff_t rangeDims, ptrdiff_t numClusters, ptrdiff_t expandedBlockSize, - float const h, float* __restrict W_ptr){ +// Aguide is called p in the paper, W is called c_k, centers is called mu_k +// requires W be set to 0 +void calcW(float *__restrict const Aguide_ptr, + float *__restrict const centers_ptr, + ptrdiff_t rangeDims, + ptrdiff_t numClusters, + ptrdiff_t expandedBlockSize, + float const h, + float *__restrict W_ptr) +{ - for (ptrdiff_t x = 0; x < expandedBlockSize; x++){ - for (ptrdiff_t c = 0; c < numClusters; c++) { - for (ptrdiff_t pIdx = 0; pIdx < rangeDims; pIdx++) { - for (ptrdiff_t y = 0; y < expandedBlockSize; y++){ - - ptrdiff_t A_idx = y + x*expandedBlockSize + pIdx*expandedBlockSize*expandedBlockSize; - ptrdiff_t centers_idx = c + pIdx*numClusters; - ptrdiff_t W_idx = y + x*expandedBlockSize + c*expandedBlockSize*expandedBlockSize; + for (ptrdiff_t x = 0; x < expandedBlockSize; x++) { + for (ptrdiff_t c = 0; c < numClusters; c++) { + for (ptrdiff_t pIdx = 0; pIdx < rangeDims; pIdx++) { + for (ptrdiff_t y = 0; y < expandedBlockSize; y++) { - float centerDist = Aguide_ptr[A_idx] - centers_ptr[centers_idx]; - W_ptr[W_idx] += centerDist*centerDist; - } - } + ptrdiff_t A_idx = y + x * expandedBlockSize + pIdx * expandedBlockSize * expandedBlockSize; + ptrdiff_t centers_idx = c + pIdx * numClusters; + ptrdiff_t W_idx = y + x * expandedBlockSize + c * expandedBlockSize * expandedBlockSize; - for (ptrdiff_t y = 0; y < expandedBlockSize; y++) { - ptrdiff_t W_idx = y + x*expandedBlockSize + c*expandedBlockSize*expandedBlockSize; - W_ptr[W_idx] = std::exp(-W_ptr[W_idx] / (2 * h * h)); - } + float centerDist = Aguide_ptr[A_idx] - centers_ptr[centers_idx]; + W_ptr[W_idx] += centerDist * centerDist; } + } + + for (ptrdiff_t y = 0; y < expandedBlockSize; y++) { + ptrdiff_t W_idx = y + x * expandedBlockSize + c * expandedBlockSize * expandedBlockSize; + W_ptr[W_idx] = std::exp(-W_ptr[W_idx] / (2 * h * h)); + } } + } } diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Cholesky/LDLT.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Cholesky/LDLT.h index 15ccf24f..c6150cf5 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Cholesky/LDLT.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Cholesky/LDLT.h @@ -20,481 +20,463 @@ namespace internal { // PositiveSemiDef means positive semi-definite and non-zero; same for NegativeSemiDef enum SignMatrix { PositiveSemiDef, NegativeSemiDef, ZeroSign, Indefinite }; -} +}// namespace internal /** \ingroup Cholesky_Module - * - * \class LDLT - * - * \brief Robust Cholesky decomposition of a matrix with pivoting - * - * \tparam _MatrixType the type of the matrix of which to compute the LDL^T Cholesky decomposition - * \tparam _UpLo the triangular part that will be used for the decompositon: Lower (default) or Upper. - * The other triangular part won't be read. - * - * Perform a robust Cholesky decomposition of a positive semidefinite or negative semidefinite - * matrix \f$ A \f$ such that \f$ A = P^TLDL^*P \f$, where P is a permutation matrix, L - * is lower triangular with a unit diagonal and D is a diagonal matrix. - * - * The decomposition uses pivoting to ensure stability, so that L will have - * zeros in the bottom right rank(A) - n submatrix. Avoiding the square root - * on D also stabilizes the computation. - * - * Remember that Cholesky decompositions are not rank-revealing. Also, do not use a Cholesky - * decomposition to determine whether a system of equations has a solution. - * - * This class supports the \link InplaceDecomposition inplace decomposition \endlink mechanism. - * - * \sa MatrixBase::ldlt(), SelfAdjointView::ldlt(), class LLT - */ + * + * \class LDLT + * + * \brief Robust Cholesky decomposition of a matrix with pivoting + * + * \tparam _MatrixType the type of the matrix of which to compute the LDL^T Cholesky decomposition + * \tparam _UpLo the triangular part that will be used for the decompositon: Lower (default) or Upper. + * The other triangular part won't be read. + * + * Perform a robust Cholesky decomposition of a positive semidefinite or negative semidefinite + * matrix \f$ A \f$ such that \f$ A = P^TLDL^*P \f$, where P is a permutation matrix, L + * is lower triangular with a unit diagonal and D is a diagonal matrix. + * + * The decomposition uses pivoting to ensure stability, so that L will have + * zeros in the bottom right rank(A) - n submatrix. Avoiding the square root + * on D also stabilizes the computation. + * + * Remember that Cholesky decompositions are not rank-revealing. Also, do not use a Cholesky + * decomposition to determine whether a system of equations has a solution. + * + * This class supports the \link InplaceDecomposition inplace decomposition \endlink mechanism. + * + * \sa MatrixBase::ldlt(), SelfAdjointView::ldlt(), class LLT + */ template class LDLT { - public: - typedef _MatrixType MatrixType; - enum { - RowsAtCompileTime = MatrixType::RowsAtCompileTime, - ColsAtCompileTime = MatrixType::ColsAtCompileTime, - MaxRowsAtCompileTime = MatrixType::MaxRowsAtCompileTime, - MaxColsAtCompileTime = MatrixType::MaxColsAtCompileTime, - UpLo = _UpLo - }; - typedef typename MatrixType::Scalar Scalar; - typedef typename NumTraits::Real RealScalar; - typedef Eigen::Index Index; ///< \deprecated since Eigen 3.3 - typedef typename MatrixType::StorageIndex StorageIndex; - typedef Matrix TmpMatrixType; - - typedef Transpositions TranspositionType; - typedef PermutationMatrix PermutationType; - - typedef internal::LDLT_Traits Traits; - - /** \brief Default Constructor. - * - * The default constructor is useful in cases in which the user intends to - * perform decompositions via LDLT::compute(const MatrixType&). - */ - LDLT() - : m_matrix(), - m_transpositions(), - m_sign(internal::ZeroSign), - m_isInitialized(false) - {} - - /** \brief Default Constructor with memory preallocation - * - * Like the default constructor but with preallocation of the internal data - * according to the specified problem \a size. - * \sa LDLT() - */ - explicit LDLT(Index size) - : m_matrix(size, size), - m_transpositions(size), - m_temporary(size), - m_sign(internal::ZeroSign), - m_isInitialized(false) - {} - - /** \brief Constructor with decomposition - * - * This calculates the decomposition for the input \a matrix. - * - * \sa LDLT(Index size) - */ - template - explicit LDLT(const EigenBase& matrix) - : m_matrix(matrix.rows(), matrix.cols()), - m_transpositions(matrix.rows()), - m_temporary(matrix.rows()), - m_sign(internal::ZeroSign), - m_isInitialized(false) - { - compute(matrix.derived()); - } - - /** \brief Constructs a LDLT factorization from a given matrix - * - * This overloaded constructor is provided for \link InplaceDecomposition inplace decomposition \endlink when \c MatrixType is a Eigen::Ref. - * - * \sa LDLT(const EigenBase&) - */ - template - explicit LDLT(EigenBase& matrix) - : m_matrix(matrix.derived()), - m_transpositions(matrix.rows()), - m_temporary(matrix.rows()), - m_sign(internal::ZeroSign), - m_isInitialized(false) - { - compute(matrix.derived()); - } - - /** Clear any existing decomposition - * \sa rankUpdate(w,sigma) - */ - void setZero() - { - m_isInitialized = false; - } - - /** \returns a view of the upper triangular matrix U */ - inline typename Traits::MatrixU matrixU() const - { - eigen_assert(m_isInitialized && "LDLT is not initialized."); - return Traits::getU(m_matrix); - } - - /** \returns a view of the lower triangular matrix L */ - inline typename Traits::MatrixL matrixL() const - { - eigen_assert(m_isInitialized && "LDLT is not initialized."); - return Traits::getL(m_matrix); - } - - /** \returns the permutation matrix P as a transposition sequence. - */ - inline const TranspositionType& transpositionsP() const - { - eigen_assert(m_isInitialized && "LDLT is not initialized."); - return m_transpositions; - } +public: + typedef _MatrixType MatrixType; + enum { + RowsAtCompileTime = MatrixType::RowsAtCompileTime, + ColsAtCompileTime = MatrixType::ColsAtCompileTime, + MaxRowsAtCompileTime = MatrixType::MaxRowsAtCompileTime, + MaxColsAtCompileTime = MatrixType::MaxColsAtCompileTime, + UpLo = _UpLo + }; + typedef typename MatrixType::Scalar Scalar; + typedef typename NumTraits::Real RealScalar; + typedef Eigen::Index Index;///< \deprecated since Eigen 3.3 + typedef typename MatrixType::StorageIndex StorageIndex; + typedef Matrix TmpMatrixType; + + typedef Transpositions TranspositionType; + typedef PermutationMatrix PermutationType; + + typedef internal::LDLT_Traits Traits; + + /** \brief Default Constructor. + * + * The default constructor is useful in cases in which the user intends to + * perform decompositions via LDLT::compute(const MatrixType&). + */ + LDLT() : m_matrix(), m_transpositions(), m_sign(internal::ZeroSign), m_isInitialized(false) {} + + /** \brief Default Constructor with memory preallocation + * + * Like the default constructor but with preallocation of the internal data + * according to the specified problem \a size. + * \sa LDLT() + */ + explicit LDLT(Index size) + : m_matrix(size, size), m_transpositions(size), m_temporary(size), m_sign(internal::ZeroSign), + m_isInitialized(false) + {} + + /** \brief Constructor with decomposition + * + * This calculates the decomposition for the input \a matrix. + * + * \sa LDLT(Index size) + */ + template + explicit LDLT(const EigenBase &matrix) + : m_matrix(matrix.rows(), matrix.cols()), m_transpositions(matrix.rows()), m_temporary(matrix.rows()), + m_sign(internal::ZeroSign), m_isInitialized(false) + { + compute(matrix.derived()); + } - /** \returns the coefficients of the diagonal matrix D */ - inline Diagonal vectorD() const - { - eigen_assert(m_isInitialized && "LDLT is not initialized."); - return m_matrix.diagonal(); - } + /** \brief Constructs a LDLT factorization from a given matrix + * + * This overloaded constructor is provided for \link InplaceDecomposition inplace decomposition \endlink when \c + * MatrixType is a Eigen::Ref. + * + * \sa LDLT(const EigenBase&) + */ + template + explicit LDLT(EigenBase &matrix) + : m_matrix(matrix.derived()), m_transpositions(matrix.rows()), m_temporary(matrix.rows()), + m_sign(internal::ZeroSign), m_isInitialized(false) + { + compute(matrix.derived()); + } - /** \returns true if the matrix is positive (semidefinite) */ - inline bool isPositive() const - { - eigen_assert(m_isInitialized && "LDLT is not initialized."); - return m_sign == internal::PositiveSemiDef || m_sign == internal::ZeroSign; - } + /** Clear any existing decomposition + * \sa rankUpdate(w,sigma) + */ + void setZero() { m_isInitialized = false; } - /** \returns true if the matrix is negative (semidefinite) */ - inline bool isNegative(void) const - { - eigen_assert(m_isInitialized && "LDLT is not initialized."); - return m_sign == internal::NegativeSemiDef || m_sign == internal::ZeroSign; - } + /** \returns a view of the upper triangular matrix U */ + inline typename Traits::MatrixU matrixU() const + { + eigen_assert(m_isInitialized && "LDLT is not initialized."); + return Traits::getU(m_matrix); + } - /** \returns a solution x of \f$ A x = b \f$ using the current decomposition of A. - * - * This function also supports in-place solves using the syntax x = decompositionObject.solve(x) . - * - * \note_about_checking_solutions - * - * More precisely, this method solves \f$ A x = b \f$ using the decomposition \f$ A = P^T L D L^* P \f$ - * by solving the systems \f$ P^T y_1 = b \f$, \f$ L y_2 = y_1 \f$, \f$ D y_3 = y_2 \f$, - * \f$ L^* y_4 = y_3 \f$ and \f$ P x = y_4 \f$ in succession. If the matrix \f$ A \f$ is singular, then - * \f$ D \f$ will also be singular (all the other matrices are invertible). In that case, the - * least-square solution of \f$ D y_3 = y_2 \f$ is computed. This does not mean that this function - * computes the least-square solution of \f$ A x = b \f$ is \f$ A \f$ is singular. - * - * \sa MatrixBase::ldlt(), SelfAdjointView::ldlt() - */ - template - inline const Solve - solve(const MatrixBase& b) const - { - eigen_assert(m_isInitialized && "LDLT is not initialized."); - eigen_assert(m_matrix.rows()==b.rows() - && "LDLT::solve(): invalid number of rows of the right hand side matrix b"); - return Solve(*this, b.derived()); - } + /** \returns a view of the lower triangular matrix L */ + inline typename Traits::MatrixL matrixL() const + { + eigen_assert(m_isInitialized && "LDLT is not initialized."); + return Traits::getL(m_matrix); + } - template - bool solveInPlace(MatrixBase &bAndX) const; + /** \returns the permutation matrix P as a transposition sequence. + */ + inline const TranspositionType &transpositionsP() const + { + eigen_assert(m_isInitialized && "LDLT is not initialized."); + return m_transpositions; + } - template - LDLT& compute(const EigenBase& matrix); + /** \returns the coefficients of the diagonal matrix D */ + inline Diagonal vectorD() const + { + eigen_assert(m_isInitialized && "LDLT is not initialized."); + return m_matrix.diagonal(); + } - /** \returns an estimate of the reciprocal condition number of the matrix of - * which \c *this is the LDLT decomposition. - */ - RealScalar rcond() const - { - eigen_assert(m_isInitialized && "LDLT is not initialized."); - return internal::rcond_estimate_helper(m_l1_norm, *this); - } + /** \returns true if the matrix is positive (semidefinite) */ + inline bool isPositive() const + { + eigen_assert(m_isInitialized && "LDLT is not initialized."); + return m_sign == internal::PositiveSemiDef || m_sign == internal::ZeroSign; + } - template - LDLT& rankUpdate(const MatrixBase& w, const RealScalar& alpha=1); + /** \returns true if the matrix is negative (semidefinite) */ + inline bool isNegative(void) const + { + eigen_assert(m_isInitialized && "LDLT is not initialized."); + return m_sign == internal::NegativeSemiDef || m_sign == internal::ZeroSign; + } - /** \returns the internal LDLT decomposition matrix - * - * TODO: document the storage layout - */ - inline const MatrixType& matrixLDLT() const - { - eigen_assert(m_isInitialized && "LDLT is not initialized."); - return m_matrix; - } + /** \returns a solution x of \f$ A x = b \f$ using the current decomposition of A. + * + * This function also supports in-place solves using the syntax x = decompositionObject.solve(x) . + * + * \note_about_checking_solutions + * + * More precisely, this method solves \f$ A x = b \f$ using the decomposition \f$ A = P^T L D L^* P \f$ + * by solving the systems \f$ P^T y_1 = b \f$, \f$ L y_2 = y_1 \f$, \f$ D y_3 = y_2 \f$, + * \f$ L^* y_4 = y_3 \f$ and \f$ P x = y_4 \f$ in succession. If the matrix \f$ A \f$ is singular, then + * \f$ D \f$ will also be singular (all the other matrices are invertible). In that case, the + * least-square solution of \f$ D y_3 = y_2 \f$ is computed. This does not mean that this function + * computes the least-square solution of \f$ A x = b \f$ is \f$ A \f$ is singular. + * + * \sa MatrixBase::ldlt(), SelfAdjointView::ldlt() + */ + template inline const Solve solve(const MatrixBase &b) const + { + eigen_assert(m_isInitialized && "LDLT is not initialized."); + eigen_assert( + m_matrix.rows() == b.rows() && "LDLT::solve(): invalid number of rows of the right hand side matrix b"); + return Solve(*this, b.derived()); + } - MatrixType reconstructedMatrix() const; + template bool solveInPlace(MatrixBase &bAndX) const; - /** \returns the adjoint of \c *this, that is, a const reference to the decomposition itself as the underlying matrix is self-adjoint. - * - * This method is provided for compatibility with other matrix decompositions, thus enabling generic code such as: - * \code x = decomposition.adjoint().solve(b) \endcode - */ - const LDLT& adjoint() const { return *this; }; + template LDLT &compute(const EigenBase &matrix); - inline Index rows() const { return m_matrix.rows(); } - inline Index cols() const { return m_matrix.cols(); } + /** \returns an estimate of the reciprocal condition number of the matrix of + * which \c *this is the LDLT decomposition. + */ + RealScalar rcond() const + { + eigen_assert(m_isInitialized && "LDLT is not initialized."); + return internal::rcond_estimate_helper(m_l1_norm, *this); + } - /** \brief Reports whether previous computation was successful. - * - * \returns \c Success if computation was succesful, - * \c NumericalIssue if the factorization failed because of a zero pivot. - */ - ComputationInfo info() const - { - eigen_assert(m_isInitialized && "LDLT is not initialized."); - return m_info; - } + template LDLT &rankUpdate(const MatrixBase &w, const RealScalar &alpha = 1); - #ifndef EIGEN_PARSED_BY_DOXYGEN - template - EIGEN_DEVICE_FUNC - void _solve_impl(const RhsType &rhs, DstType &dst) const; - #endif + /** \returns the internal LDLT decomposition matrix + * + * TODO: document the storage layout + */ + inline const MatrixType &matrixLDLT() const + { + eigen_assert(m_isInitialized && "LDLT is not initialized."); + return m_matrix; + } - protected: + MatrixType reconstructedMatrix() const; + + /** \returns the adjoint of \c *this, that is, a const reference to the decomposition itself as the underlying matrix + * is self-adjoint. + * + * This method is provided for compatibility with other matrix decompositions, thus enabling generic code such as: + * \code x = decomposition.adjoint().solve(b) \endcode + */ + const LDLT &adjoint() const { return *this; }; + + inline Index rows() const { return m_matrix.rows(); } + inline Index cols() const { return m_matrix.cols(); } + + /** \brief Reports whether previous computation was successful. + * + * \returns \c Success if computation was succesful, + * \c NumericalIssue if the factorization failed because of a zero pivot. + */ + ComputationInfo info() const + { + eigen_assert(m_isInitialized && "LDLT is not initialized."); + return m_info; + } - static void check_template_parameters() - { - EIGEN_STATIC_ASSERT_NON_INTEGER(Scalar); - } +#ifndef EIGEN_PARSED_BY_DOXYGEN + template + EIGEN_DEVICE_FUNC void _solve_impl(const RhsType &rhs, DstType &dst) const; +#endif - /** \internal - * Used to compute and store the Cholesky decomposition A = L D L^* = U^* D U. - * The strict upper part is used during the decomposition, the strict lower - * part correspond to the coefficients of L (its diagonal is equal to 1 and - * is not stored), and the diagonal entries correspond to D. - */ - MatrixType m_matrix; - RealScalar m_l1_norm; - TranspositionType m_transpositions; - TmpMatrixType m_temporary; - internal::SignMatrix m_sign; - bool m_isInitialized; - ComputationInfo m_info; +protected: + static void check_template_parameters() { EIGEN_STATIC_ASSERT_NON_INTEGER(Scalar); } + + /** \internal + * Used to compute and store the Cholesky decomposition A = L D L^* = U^* D U. + * The strict upper part is used during the decomposition, the strict lower + * part correspond to the coefficients of L (its diagonal is equal to 1 and + * is not stored), and the diagonal entries correspond to D. + */ + MatrixType m_matrix; + RealScalar m_l1_norm; + TranspositionType m_transpositions; + TmpMatrixType m_temporary; + internal::SignMatrix m_sign; + bool m_isInitialized; + ComputationInfo m_info; }; namespace internal { -template struct ldlt_inplace; + template struct ldlt_inplace; -template<> struct ldlt_inplace -{ - template - static bool unblocked(MatrixType& mat, TranspositionType& transpositions, Workspace& temp, SignMatrix& sign) + template<> struct ldlt_inplace { - using std::abs; - typedef typename MatrixType::Scalar Scalar; - typedef typename MatrixType::RealScalar RealScalar; - typedef typename TranspositionType::StorageIndex IndexType; - eigen_assert(mat.rows()==mat.cols()); - const Index size = mat.rows(); - bool found_zero_pivot = false; - bool ret = true; - - if (size <= 1) + template + static bool unblocked(MatrixType &mat, TranspositionType &transpositions, Workspace &temp, SignMatrix &sign) { - transpositions.setIdentity(); - if(size==0) sign = ZeroSign; - else if (numext::real(mat.coeff(0,0)) > static_cast(0) ) sign = PositiveSemiDef; - else if (numext::real(mat.coeff(0,0)) < static_cast(0)) sign = NegativeSemiDef; - else sign = ZeroSign; - return true; - } + using std::abs; + typedef typename MatrixType::Scalar Scalar; + typedef typename MatrixType::RealScalar RealScalar; + typedef typename TranspositionType::StorageIndex IndexType; + eigen_assert(mat.rows() == mat.cols()); + const Index size = mat.rows(); + bool found_zero_pivot = false; + bool ret = true; + + if (size <= 1) { + transpositions.setIdentity(); + if (size == 0) + sign = ZeroSign; + else if (numext::real(mat.coeff(0, 0)) > static_cast(0)) + sign = PositiveSemiDef; + else if (numext::real(mat.coeff(0, 0)) < static_cast(0)) + sign = NegativeSemiDef; + else + sign = ZeroSign; + return true; + } - for (Index k = 0; k < size; ++k) - { - // Find largest diagonal element - Index index_of_biggest_in_corner; - mat.diagonal().tail(size-k).cwiseAbs().maxCoeff(&index_of_biggest_in_corner); - index_of_biggest_in_corner += k; - - transpositions.coeffRef(k) = IndexType(index_of_biggest_in_corner); - if(k != index_of_biggest_in_corner) - { - // apply the transposition while taking care to consider only - // the lower triangular part - Index s = size-index_of_biggest_in_corner-1; // trailing size after the biggest element - mat.row(k).head(k).swap(mat.row(index_of_biggest_in_corner).head(k)); - mat.col(k).tail(s).swap(mat.col(index_of_biggest_in_corner).tail(s)); - std::swap(mat.coeffRef(k,k),mat.coeffRef(index_of_biggest_in_corner,index_of_biggest_in_corner)); - for(Index i=k+1;i::IsComplex) + mat.coeffRef(index_of_biggest_in_corner, k) = numext::conj(mat.coeff(index_of_biggest_in_corner, k)); } - if(NumTraits::IsComplex) - mat.coeffRef(index_of_biggest_in_corner,k) = numext::conj(mat.coeff(index_of_biggest_in_corner,k)); - } - // partition the matrix: - // A00 | - | - - // lu = A10 | A11 | - - // A20 | A21 | A22 - Index rs = size - k - 1; - Block A21(mat,k+1,k,rs,1); - Block A10(mat,k,0,1,k); - Block A20(mat,k+1,0,rs,k); - - if(k>0) - { - temp.head(k) = mat.diagonal().real().head(k).asDiagonal() * A10.adjoint(); - mat.coeffRef(k,k) -= (A10 * temp.head(k)).value(); - if(rs>0) - A21.noalias() -= A20 * temp.head(k); - } + // partition the matrix: + // A00 | - | - + // lu = A10 | A11 | - + // A20 | A21 | A22 + Index rs = size - k - 1; + Block A21(mat, k + 1, k, rs, 1); + Block A10(mat, k, 0, 1, k); + Block A20(mat, k + 1, 0, rs, k); + + if (k > 0) { + temp.head(k) = mat.diagonal().real().head(k).asDiagonal() * A10.adjoint(); + mat.coeffRef(k, k) -= (A10 * temp.head(k)).value(); + if (rs > 0) A21.noalias() -= A20 * temp.head(k); + } - // In some previous versions of Eigen (e.g., 3.2.1), the scaling was omitted if the pivot - // was smaller than the cutoff value. However, since LDLT is not rank-revealing - // we should only make sure that we do not introduce INF or NaN values. - // Remark that LAPACK also uses 0 as the cutoff value. - RealScalar realAkk = numext::real(mat.coeffRef(k,k)); - bool pivot_is_valid = (abs(realAkk) > RealScalar(0)); - - if(k==0 && !pivot_is_valid) - { - // The entire diagonal is zero, there is nothing more to do - // except filling the transpositions, and checking whether the matrix is zero. - sign = ZeroSign; - for(Index j = 0; j RealScalar(0)); + + if (k == 0 && !pivot_is_valid) { + // The entire diagonal is zero, there is nothing more to do + // except filling the transpositions, and checking whether the matrix is zero. + sign = ZeroSign; + for (Index j = 0; j < size; ++j) { + transpositions.coeffRef(j) = IndexType(j); + ret = ret && (mat.col(j).tail(size - j - 1).array() == Scalar(0)).all(); + } + return ret; } - return ret; - } - if((rs>0) && pivot_is_valid) - A21 /= realAkk; - else if(rs>0) - ret = ret && (A21.array()==Scalar(0)).all(); - - if(found_zero_pivot && pivot_is_valid) ret = false; // factorization failed - else if(!pivot_is_valid) found_zero_pivot = true; - - if (sign == PositiveSemiDef) { - if (realAkk < static_cast(0)) sign = Indefinite; - } else if (sign == NegativeSemiDef) { - if (realAkk > static_cast(0)) sign = Indefinite; - } else if (sign == ZeroSign) { - if (realAkk > static_cast(0)) sign = PositiveSemiDef; - else if (realAkk < static_cast(0)) sign = NegativeSemiDef; + if ((rs > 0) && pivot_is_valid) + A21 /= realAkk; + else if (rs > 0) + ret = ret && (A21.array() == Scalar(0)).all(); + + if (found_zero_pivot && pivot_is_valid) + ret = false;// factorization failed + else if (!pivot_is_valid) + found_zero_pivot = true; + + if (sign == PositiveSemiDef) { + if (realAkk < static_cast(0)) sign = Indefinite; + } else if (sign == NegativeSemiDef) { + if (realAkk > static_cast(0)) sign = Indefinite; + } else if (sign == ZeroSign) { + if (realAkk > static_cast(0)) + sign = PositiveSemiDef; + else if (realAkk < static_cast(0)) + sign = NegativeSemiDef; + } } + + return ret; } - return ret; - } + // Reference for the algorithm: Davis and Hager, "Multiple Rank + // Modifications of a Sparse Cholesky Factorization" (Algorithm 1) + // Trivial rearrangements of their computations (Timothy E. Holy) + // allow their algorithm to work for rank-1 updates even if the + // original matrix is not of full rank. + // Here only rank-1 updates are implemented, to reduce the + // requirement for intermediate storage and improve accuracy + template + static bool + updateInPlace(MatrixType &mat, MatrixBase &w, const typename MatrixType::RealScalar &sigma = 1) + { + using numext::isfinite; + typedef typename MatrixType::Scalar Scalar; + typedef typename MatrixType::RealScalar RealScalar; - // Reference for the algorithm: Davis and Hager, "Multiple Rank - // Modifications of a Sparse Cholesky Factorization" (Algorithm 1) - // Trivial rearrangements of their computations (Timothy E. Holy) - // allow their algorithm to work for rank-1 updates even if the - // original matrix is not of full rank. - // Here only rank-1 updates are implemented, to reduce the - // requirement for intermediate storage and improve accuracy - template - static bool updateInPlace(MatrixType& mat, MatrixBase& w, const typename MatrixType::RealScalar& sigma=1) - { - using numext::isfinite; - typedef typename MatrixType::Scalar Scalar; - typedef typename MatrixType::RealScalar RealScalar; + const Index size = mat.rows(); + eigen_assert(mat.cols() == size && w.size() == size); - const Index size = mat.rows(); - eigen_assert(mat.cols() == size && w.size()==size); + RealScalar alpha = 1; - RealScalar alpha = 1; + // Apply the update + for (Index j = 0; j < size; j++) { + // Check for termination due to an original decomposition of low-rank + if (!(isfinite)(alpha)) break; - // Apply the update - for (Index j = 0; j < size; j++) - { - // Check for termination due to an original decomposition of low-rank - if (!(isfinite)(alpha)) - break; + // Update the diagonal terms + RealScalar dj = numext::real(mat.coeff(j, j)); + Scalar wj = w.coeff(j); + RealScalar swj2 = sigma * numext::abs2(wj); + RealScalar gamma = dj * alpha + swj2; - // Update the diagonal terms - RealScalar dj = numext::real(mat.coeff(j,j)); - Scalar wj = w.coeff(j); - RealScalar swj2 = sigma*numext::abs2(wj); - RealScalar gamma = dj*alpha + swj2; + mat.coeffRef(j, j) += swj2 / alpha; + alpha += swj2 / dj; - mat.coeffRef(j,j) += swj2/alpha; - alpha += swj2/dj; + // Update the terms of L + Index rs = size - j - 1; + w.tail(rs) -= wj * mat.col(j).tail(rs); + if (gamma != 0) mat.col(j).tail(rs) += (sigma * numext::conj(wj) / gamma) * w.tail(rs); + } + return true; + } - // Update the terms of L - Index rs = size-j-1; - w.tail(rs) -= wj * mat.col(j).tail(rs); - if(gamma != 0) - mat.col(j).tail(rs) += (sigma*numext::conj(wj)/gamma)*w.tail(rs); + template + static bool update(MatrixType &mat, + const TranspositionType &transpositions, + Workspace &tmp, + const WType &w, + const typename MatrixType::RealScalar &sigma = 1) + { + // Apply the permutation to the input w + tmp = transpositions * w; + + return ldlt_inplace::updateInPlace(mat, tmp, sigma); } - return true; - } + }; - template - static bool update(MatrixType& mat, const TranspositionType& transpositions, Workspace& tmp, const WType& w, const typename MatrixType::RealScalar& sigma=1) + template<> struct ldlt_inplace { - // Apply the permutation to the input w - tmp = transpositions * w; + template + static EIGEN_STRONG_INLINE bool + unblocked(MatrixType &mat, TranspositionType &transpositions, Workspace &temp, SignMatrix &sign) + { + Transpose matt(mat); + return ldlt_inplace::unblocked(matt, transpositions, temp, sign); + } - return ldlt_inplace::updateInPlace(mat,tmp,sigma); - } -}; + template + static EIGEN_STRONG_INLINE bool update(MatrixType &mat, + TranspositionType &transpositions, + Workspace &tmp, + WType &w, + const typename MatrixType::RealScalar &sigma = 1) + { + Transpose matt(mat); + return ldlt_inplace::update(matt, transpositions, tmp, w.conjugate(), sigma); + } + }; -template<> struct ldlt_inplace -{ - template - static EIGEN_STRONG_INLINE bool unblocked(MatrixType& mat, TranspositionType& transpositions, Workspace& temp, SignMatrix& sign) + template struct LDLT_Traits { - Transpose matt(mat); - return ldlt_inplace::unblocked(matt, transpositions, temp, sign); - } + typedef const TriangularView MatrixL; + typedef const TriangularView MatrixU; + static inline MatrixL getL(const MatrixType &m) { return MatrixL(m); } + static inline MatrixU getU(const MatrixType &m) { return MatrixU(m.adjoint()); } + }; - template - static EIGEN_STRONG_INLINE bool update(MatrixType& mat, TranspositionType& transpositions, Workspace& tmp, WType& w, const typename MatrixType::RealScalar& sigma=1) + template struct LDLT_Traits { - Transpose matt(mat); - return ldlt_inplace::update(matt, transpositions, tmp, w.conjugate(), sigma); - } -}; + typedef const TriangularView MatrixL; + typedef const TriangularView MatrixU; + static inline MatrixL getL(const MatrixType &m) { return MatrixL(m.adjoint()); } + static inline MatrixU getU(const MatrixType &m) { return MatrixU(m); } + }; -template struct LDLT_Traits -{ - typedef const TriangularView MatrixL; - typedef const TriangularView MatrixU; - static inline MatrixL getL(const MatrixType& m) { return MatrixL(m); } - static inline MatrixU getU(const MatrixType& m) { return MatrixU(m.adjoint()); } -}; - -template struct LDLT_Traits -{ - typedef const TriangularView MatrixL; - typedef const TriangularView MatrixU; - static inline MatrixL getL(const MatrixType& m) { return MatrixL(m.adjoint()); } - static inline MatrixU getU(const MatrixType& m) { return MatrixU(m); } -}; - -} // end namespace internal +}// end namespace internal /** Compute / recompute the LDLT decomposition A = L D L^* = U^* D U of \a matrix - */ + */ template template -LDLT& LDLT::compute(const EigenBase& a) +LDLT &LDLT::compute(const EigenBase &a) { check_template_parameters(); - eigen_assert(a.rows()==a.cols()); + eigen_assert(a.rows() == a.cols()); const Index size = a.rows(); m_matrix = a.derived(); @@ -505,11 +487,12 @@ LDLT& LDLT::compute(const EigenBase() + m_matrix.row(col).head(col).template lpNorm<1>(); + abs_col_sum = + m_matrix.col(col).tail(size - col).template lpNorm<1>() + m_matrix.row(col).head(col).template lpNorm<1>(); else - abs_col_sum = m_matrix.col(col).head(col).template lpNorm<1>() + m_matrix.row(col).tail(size - col).template lpNorm<1>(); - if (abs_col_sum > m_l1_norm) - m_l1_norm = abs_col_sum; + abs_col_sum = + m_matrix.col(col).head(col).template lpNorm<1>() + m_matrix.row(col).tail(size - col).template lpNorm<1>(); + if (abs_col_sum > m_l1_norm) m_l1_norm = abs_col_sum; } m_transpositions.resize(size); @@ -517,7 +500,8 @@ LDLT& LDLT::compute(const EigenBase::unblocked(m_matrix, m_transpositions, m_temporary, m_sign) ? Success : NumericalIssue; + m_info = + internal::ldlt_inplace::unblocked(m_matrix, m_transpositions, m_temporary, m_sign) ? Success : NumericalIssue; m_isInitialized = true; return *this; @@ -525,28 +509,26 @@ LDLT& LDLT::compute(const EigenBase template -LDLT& LDLT::rankUpdate(const MatrixBase& w, const typename LDLT::RealScalar& sigma) +LDLT &LDLT::rankUpdate(const MatrixBase &w, + const typename LDLT::RealScalar &sigma) { typedef typename TranspositionType::StorageIndex IndexType; const Index size = w.rows(); - if (m_isInitialized) - { - eigen_assert(m_matrix.rows()==size); - } - else - { - m_matrix.resize(size,size); + if (m_isInitialized) { + eigen_assert(m_matrix.rows() == size); + } else { + m_matrix.resize(size, size); m_matrix.setZero(); m_transpositions.resize(size); - for (Index i = 0; i < size; i++) - m_transpositions.coeffRef(i) = IndexType(i); + for (Index i = 0; i < size; i++) m_transpositions.coeffRef(i) = IndexType(i); m_temporary.resize(size); - m_sign = sigma>=0 ? internal::PositiveSemiDef : internal::NegativeSemiDef; + m_sign = sigma >= 0 ? internal::PositiveSemiDef : internal::NegativeSemiDef; m_isInitialized = true; } @@ -558,7 +540,7 @@ LDLT& LDLT::rankUpdate(const MatrixBase template -void LDLT<_MatrixType,_UpLo>::_solve_impl(const RhsType &rhs, DstType &dst) const +void LDLT<_MatrixType, _UpLo>::_solve_impl(const RhsType &rhs, DstType &dst) const { eigen_assert(rhs.rows() == rows()); // dst = P b @@ -573,16 +555,14 @@ void LDLT<_MatrixType,_UpLo>::_solve_impl(const RhsType &rhs, DstType &dst) cons const typename Diagonal::RealReturnType vecD(vectorD()); // In some previous versions, tolerance was set to the max of 1/highest (or rather numeric_limits::min()) // and the maximal diagonal entry * epsilon as motivated by LAPACK's xGELSS: - // RealScalar tolerance = numext::maxi(vecD.array().abs().maxCoeff() * NumTraits::epsilon(),RealScalar(1) / NumTraits::highest()); - // However, LDLT is not rank revealing, and so adjusting the tolerance wrt to the highest - // diagonal element is not well justified and leads to numerical issues in some cases. - // Moreover, Lapack's xSYTRS routines use 0 for the tolerance. - // Using numeric_limits::min() gives us more robustness to denormals. + // RealScalar tolerance = numext::maxi(vecD.array().abs().maxCoeff() * NumTraits::epsilon(),RealScalar(1) + // / NumTraits::highest()); However, LDLT is not rank revealing, and so adjusting the tolerance wrt to the + // highest diagonal element is not well justified and leads to numerical issues in some cases. Moreover, Lapack's + // xSYTRS routines use 0 for the tolerance. Using numeric_limits::min() gives us more robustness to denormals. RealScalar tolerance = (std::numeric_limits::min)(); - for (Index i = 0; i < vecD.size(); ++i) - { - if(abs(vecD(i)) > tolerance) + for (Index i = 0; i < vecD.size(); ++i) { + if (abs(vecD(i)) > tolerance) dst.row(i) /= vecD(i); else dst.row(i).setZero(); @@ -597,21 +577,21 @@ void LDLT<_MatrixType,_UpLo>::_solve_impl(const RhsType &rhs, DstType &dst) cons #endif /** \internal use x = ldlt_object.solve(x); - * - * This is the \em in-place version of solve(). - * - * \param bAndX represents both the right-hand side matrix b and result x. - * - * \returns true always! If you need to check for existence of solutions, use another decomposition like LU, QR, or SVD. - * - * This version avoids a copy when the right hand side matrix b is not - * needed anymore. - * - * \sa LDLT::solve(), MatrixBase::ldlt() - */ -template + * + * This is the \em in-place version of solve(). + * + * \param bAndX represents both the right-hand side matrix b and result x. + * + * \returns true always! If you need to check for existence of solutions, use another decomposition like LU, QR, or SVD. + * + * This version avoids a copy when the right hand side matrix b is not + * needed anymore. + * + * \sa LDLT::solve(), MatrixBase::ldlt() + */ +template template -bool LDLT::solveInPlace(MatrixBase &bAndX) const +bool LDLT::solveInPlace(MatrixBase &bAndX) const { eigen_assert(m_isInitialized && "LDLT is not initialized."); eigen_assert(m_matrix.rows() == bAndX.rows()); @@ -624,12 +604,11 @@ bool LDLT::solveInPlace(MatrixBase &bAndX) const /** \returns the matrix represented by the decomposition, * i.e., it returns the product: P^T L D L^* P. * This function is provided for debug purpose. */ -template -MatrixType LDLT::reconstructedMatrix() const +template MatrixType LDLT::reconstructedMatrix() const { eigen_assert(m_isInitialized && "LDLT is not initialized."); const Index size = m_matrix.rows(); - MatrixType res(size,size); + MatrixType res(size, size); // P res.setIdentity(); @@ -647,27 +626,26 @@ MatrixType LDLT::reconstructedMatrix() const } /** \cholesky_module - * \returns the Cholesky decomposition with full pivoting without square root of \c *this - * \sa MatrixBase::ldlt() - */ + * \returns the Cholesky decomposition with full pivoting without square root of \c *this + * \sa MatrixBase::ldlt() + */ template inline const LDLT::PlainObject, UpLo> -SelfAdjointView::ldlt() const + SelfAdjointView::ldlt() const { - return LDLT(m_matrix); + return LDLT(m_matrix); } /** \cholesky_module - * \returns the Cholesky decomposition with full pivoting without square root of \c *this - * \sa SelfAdjointView::ldlt() - */ + * \returns the Cholesky decomposition with full pivoting without square root of \c *this + * \sa SelfAdjointView::ldlt() + */ template -inline const LDLT::PlainObject> -MatrixBase::ldlt() const +inline const LDLT::PlainObject> MatrixBase::ldlt() const { return LDLT(derived()); } -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_LDLT_H +#endif// EIGEN_LDLT_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Cholesky/LLT.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Cholesky/LLT.h index e1624d21..bfb60ed4 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Cholesky/LLT.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Cholesky/LLT.h @@ -12,425 +12,397 @@ namespace Eigen { -namespace internal{ -template struct LLT_Traits; +namespace internal { + template struct LLT_Traits; } /** \ingroup Cholesky_Module - * - * \class LLT - * - * \brief Standard Cholesky decomposition (LL^T) of a matrix and associated features - * - * \tparam _MatrixType the type of the matrix of which we are computing the LL^T Cholesky decomposition - * \tparam _UpLo the triangular part that will be used for the decompositon: Lower (default) or Upper. - * The other triangular part won't be read. - * - * This class performs a LL^T Cholesky decomposition of a symmetric, positive definite - * matrix A such that A = LL^* = U^*U, where L is lower triangular. - * - * While the Cholesky decomposition is particularly useful to solve selfadjoint problems like D^*D x = b, - * for that purpose, we recommend the Cholesky decomposition without square root which is more stable - * and even faster. Nevertheless, this standard Cholesky decomposition remains useful in many other - * situations like generalised eigen problems with hermitian matrices. - * - * Remember that Cholesky decompositions are not rank-revealing. This LLT decomposition is only stable on positive definite matrices, - * use LDLT instead for the semidefinite case. Also, do not use a Cholesky decomposition to determine whether a system of equations - * has a solution. - * - * Example: \include LLT_example.cpp - * Output: \verbinclude LLT_example.out - * - * \b Performance: for best performance, it is recommended to use a column-major storage format - * with the Lower triangular part (the default), or, equivalently, a row-major storage format - * with the Upper triangular part. Otherwise, you might get a 20% slowdown for the full factorization - * step, and rank-updates can be up to 3 times slower. - * - * This class supports the \link InplaceDecomposition inplace decomposition \endlink mechanism. - * - * Note that during the decomposition, only the lower (or upper, as defined by _UpLo) triangular part of A is considered. - * Therefore, the strict lower part does not have to store correct values. - * - * \sa MatrixBase::llt(), SelfAdjointView::llt(), class LDLT - */ + * + * \class LLT + * + * \brief Standard Cholesky decomposition (LL^T) of a matrix and associated features + * + * \tparam _MatrixType the type of the matrix of which we are computing the LL^T Cholesky decomposition + * \tparam _UpLo the triangular part that will be used for the decompositon: Lower (default) or Upper. + * The other triangular part won't be read. + * + * This class performs a LL^T Cholesky decomposition of a symmetric, positive definite + * matrix A such that A = LL^* = U^*U, where L is lower triangular. + * + * While the Cholesky decomposition is particularly useful to solve selfadjoint problems like D^*D x = b, + * for that purpose, we recommend the Cholesky decomposition without square root which is more stable + * and even faster. Nevertheless, this standard Cholesky decomposition remains useful in many other + * situations like generalised eigen problems with hermitian matrices. + * + * Remember that Cholesky decompositions are not rank-revealing. This LLT decomposition is only stable on positive + * definite matrices, use LDLT instead for the semidefinite case. Also, do not use a Cholesky decomposition to determine + * whether a system of equations has a solution. + * + * Example: \include LLT_example.cpp + * Output: \verbinclude LLT_example.out + * + * \b Performance: for best performance, it is recommended to use a column-major storage format + * with the Lower triangular part (the default), or, equivalently, a row-major storage format + * with the Upper triangular part. Otherwise, you might get a 20% slowdown for the full factorization + * step, and rank-updates can be up to 3 times slower. + * + * This class supports the \link InplaceDecomposition inplace decomposition \endlink mechanism. + * + * Note that during the decomposition, only the lower (or upper, as defined by _UpLo) triangular part of A is + * considered. Therefore, the strict lower part does not have to store correct values. + * + * \sa MatrixBase::llt(), SelfAdjointView::llt(), class LDLT + */ template class LLT { - public: - typedef _MatrixType MatrixType; - enum { - RowsAtCompileTime = MatrixType::RowsAtCompileTime, - ColsAtCompileTime = MatrixType::ColsAtCompileTime, - MaxColsAtCompileTime = MatrixType::MaxColsAtCompileTime - }; - typedef typename MatrixType::Scalar Scalar; - typedef typename NumTraits::Real RealScalar; - typedef Eigen::Index Index; ///< \deprecated since Eigen 3.3 - typedef typename MatrixType::StorageIndex StorageIndex; - - enum { - PacketSize = internal::packet_traits::size, - AlignmentMask = int(PacketSize)-1, - UpLo = _UpLo - }; - - typedef internal::LLT_Traits Traits; - - /** - * \brief Default Constructor. - * - * The default constructor is useful in cases in which the user intends to - * perform decompositions via LLT::compute(const MatrixType&). - */ - LLT() : m_matrix(), m_isInitialized(false) {} - - /** \brief Default Constructor with memory preallocation - * - * Like the default constructor but with preallocation of the internal data - * according to the specified problem \a size. - * \sa LLT() - */ - explicit LLT(Index size) : m_matrix(size, size), - m_isInitialized(false) {} - - template - explicit LLT(const EigenBase& matrix) - : m_matrix(matrix.rows(), matrix.cols()), - m_isInitialized(false) - { - compute(matrix.derived()); - } +public: + typedef _MatrixType MatrixType; + enum { + RowsAtCompileTime = MatrixType::RowsAtCompileTime, + ColsAtCompileTime = MatrixType::ColsAtCompileTime, + MaxColsAtCompileTime = MatrixType::MaxColsAtCompileTime + }; + typedef typename MatrixType::Scalar Scalar; + typedef typename NumTraits::Real RealScalar; + typedef Eigen::Index Index;///< \deprecated since Eigen 3.3 + typedef typename MatrixType::StorageIndex StorageIndex; + + enum { PacketSize = internal::packet_traits::size, AlignmentMask = int(PacketSize) - 1, UpLo = _UpLo }; + + typedef internal::LLT_Traits Traits; + + /** + * \brief Default Constructor. + * + * The default constructor is useful in cases in which the user intends to + * perform decompositions via LLT::compute(const MatrixType&). + */ + LLT() : m_matrix(), m_isInitialized(false) {} + + /** \brief Default Constructor with memory preallocation + * + * Like the default constructor but with preallocation of the internal data + * according to the specified problem \a size. + * \sa LLT() + */ + explicit LLT(Index size) : m_matrix(size, size), m_isInitialized(false) {} + + template + explicit LLT(const EigenBase &matrix) : m_matrix(matrix.rows(), matrix.cols()), m_isInitialized(false) + { + compute(matrix.derived()); + } - /** \brief Constructs a LDLT factorization from a given matrix - * - * This overloaded constructor is provided for \link InplaceDecomposition inplace decomposition \endlink when - * \c MatrixType is a Eigen::Ref. - * - * \sa LLT(const EigenBase&) - */ - template - explicit LLT(EigenBase& matrix) - : m_matrix(matrix.derived()), - m_isInitialized(false) - { - compute(matrix.derived()); - } + /** \brief Constructs a LDLT factorization from a given matrix + * + * This overloaded constructor is provided for \link InplaceDecomposition inplace decomposition \endlink when + * \c MatrixType is a Eigen::Ref. + * + * \sa LLT(const EigenBase&) + */ + template + explicit LLT(EigenBase &matrix) : m_matrix(matrix.derived()), m_isInitialized(false) + { + compute(matrix.derived()); + } - /** \returns a view of the upper triangular matrix U */ - inline typename Traits::MatrixU matrixU() const - { - eigen_assert(m_isInitialized && "LLT is not initialized."); - return Traits::getU(m_matrix); - } + /** \returns a view of the upper triangular matrix U */ + inline typename Traits::MatrixU matrixU() const + { + eigen_assert(m_isInitialized && "LLT is not initialized."); + return Traits::getU(m_matrix); + } - /** \returns a view of the lower triangular matrix L */ - inline typename Traits::MatrixL matrixL() const - { - eigen_assert(m_isInitialized && "LLT is not initialized."); - return Traits::getL(m_matrix); - } + /** \returns a view of the lower triangular matrix L */ + inline typename Traits::MatrixL matrixL() const + { + eigen_assert(m_isInitialized && "LLT is not initialized."); + return Traits::getL(m_matrix); + } - /** \returns the solution x of \f$ A x = b \f$ using the current decomposition of A. - * - * Since this LLT class assumes anyway that the matrix A is invertible, the solution - * theoretically exists and is unique regardless of b. - * - * Example: \include LLT_solve.cpp - * Output: \verbinclude LLT_solve.out - * - * \sa solveInPlace(), MatrixBase::llt(), SelfAdjointView::llt() - */ - template - inline const Solve - solve(const MatrixBase& b) const - { - eigen_assert(m_isInitialized && "LLT is not initialized."); - eigen_assert(m_matrix.rows()==b.rows() - && "LLT::solve(): invalid number of rows of the right hand side matrix b"); - return Solve(*this, b.derived()); - } + /** \returns the solution x of \f$ A x = b \f$ using the current decomposition of A. + * + * Since this LLT class assumes anyway that the matrix A is invertible, the solution + * theoretically exists and is unique regardless of b. + * + * Example: \include LLT_solve.cpp + * Output: \verbinclude LLT_solve.out + * + * \sa solveInPlace(), MatrixBase::llt(), SelfAdjointView::llt() + */ + template inline const Solve solve(const MatrixBase &b) const + { + eigen_assert(m_isInitialized && "LLT is not initialized."); + eigen_assert(m_matrix.rows() == b.rows() && "LLT::solve(): invalid number of rows of the right hand side matrix b"); + return Solve(*this, b.derived()); + } - template - void solveInPlace(const MatrixBase &bAndX) const; + template void solveInPlace(const MatrixBase &bAndX) const; - template - LLT& compute(const EigenBase& matrix); + template LLT &compute(const EigenBase &matrix); - /** \returns an estimate of the reciprocal condition number of the matrix of - * which \c *this is the Cholesky decomposition. - */ - RealScalar rcond() const - { - eigen_assert(m_isInitialized && "LLT is not initialized."); - eigen_assert(m_info == Success && "LLT failed because matrix appears to be negative"); - return internal::rcond_estimate_helper(m_l1_norm, *this); - } - - /** \returns the LLT decomposition matrix - * - * TODO: document the storage layout - */ - inline const MatrixType& matrixLLT() const - { - eigen_assert(m_isInitialized && "LLT is not initialized."); - return m_matrix; - } - - MatrixType reconstructedMatrix() const; + /** \returns an estimate of the reciprocal condition number of the matrix of + * which \c *this is the Cholesky decomposition. + */ + RealScalar rcond() const + { + eigen_assert(m_isInitialized && "LLT is not initialized."); + eigen_assert(m_info == Success && "LLT failed because matrix appears to be negative"); + return internal::rcond_estimate_helper(m_l1_norm, *this); + } + /** \returns the LLT decomposition matrix + * + * TODO: document the storage layout + */ + inline const MatrixType &matrixLLT() const + { + eigen_assert(m_isInitialized && "LLT is not initialized."); + return m_matrix; + } - /** \brief Reports whether previous computation was successful. - * - * \returns \c Success if computation was succesful, - * \c NumericalIssue if the matrix.appears not to be positive definite. - */ - ComputationInfo info() const - { - eigen_assert(m_isInitialized && "LLT is not initialized."); - return m_info; - } + MatrixType reconstructedMatrix() const; - /** \returns the adjoint of \c *this, that is, a const reference to the decomposition itself as the underlying matrix is self-adjoint. - * - * This method is provided for compatibility with other matrix decompositions, thus enabling generic code such as: - * \code x = decomposition.adjoint().solve(b) \endcode - */ - const LLT& adjoint() const { return *this; }; - inline Index rows() const { return m_matrix.rows(); } - inline Index cols() const { return m_matrix.cols(); } + /** \brief Reports whether previous computation was successful. + * + * \returns \c Success if computation was succesful, + * \c NumericalIssue if the matrix.appears not to be positive definite. + */ + ComputationInfo info() const + { + eigen_assert(m_isInitialized && "LLT is not initialized."); + return m_info; + } - template - LLT rankUpdate(const VectorType& vec, const RealScalar& sigma = 1); + /** \returns the adjoint of \c *this, that is, a const reference to the decomposition itself as the underlying matrix + * is self-adjoint. + * + * This method is provided for compatibility with other matrix decompositions, thus enabling generic code such as: + * \code x = decomposition.adjoint().solve(b) \endcode + */ + const LLT &adjoint() const { return *this; }; - #ifndef EIGEN_PARSED_BY_DOXYGEN - template - EIGEN_DEVICE_FUNC - void _solve_impl(const RhsType &rhs, DstType &dst) const; - #endif + inline Index rows() const { return m_matrix.rows(); } + inline Index cols() const { return m_matrix.cols(); } - protected: + template LLT rankUpdate(const VectorType &vec, const RealScalar &sigma = 1); - static void check_template_parameters() - { - EIGEN_STATIC_ASSERT_NON_INTEGER(Scalar); - } +#ifndef EIGEN_PARSED_BY_DOXYGEN + template + EIGEN_DEVICE_FUNC void _solve_impl(const RhsType &rhs, DstType &dst) const; +#endif - /** \internal - * Used to compute and store L - * The strict upper part is not used and even not initialized. - */ - MatrixType m_matrix; - RealScalar m_l1_norm; - bool m_isInitialized; - ComputationInfo m_info; +protected: + static void check_template_parameters() { EIGEN_STATIC_ASSERT_NON_INTEGER(Scalar); } + + /** \internal + * Used to compute and store L + * The strict upper part is not used and even not initialized. + */ + MatrixType m_matrix; + RealScalar m_l1_norm; + bool m_isInitialized; + ComputationInfo m_info; }; namespace internal { -template struct llt_inplace; - -template -static Index llt_rank_update_lower(MatrixType& mat, const VectorType& vec, const typename MatrixType::RealScalar& sigma) -{ - using std::sqrt; - typedef typename MatrixType::Scalar Scalar; - typedef typename MatrixType::RealScalar RealScalar; - typedef typename MatrixType::ColXpr ColXpr; - typedef typename internal::remove_all::type ColXprCleaned; - typedef typename ColXprCleaned::SegmentReturnType ColXprSegment; - typedef Matrix TempVectorType; - typedef typename TempVectorType::SegmentReturnType TempVecSegment; - - Index n = mat.cols(); - eigen_assert(mat.rows()==n && vec.size()==n); - - TempVectorType temp; + template struct llt_inplace; - if(sigma>0) + template + static Index + llt_rank_update_lower(MatrixType &mat, const VectorType &vec, const typename MatrixType::RealScalar &sigma) { - // This version is based on Givens rotations. - // It is faster than the other one below, but only works for updates, - // i.e., for sigma > 0 - temp = sqrt(sigma) * vec; - - for(Index i=0; i g; - g.makeGivens(mat(i,i), -temp(i), &mat(i,i)); - - Index rs = n-i-1; - if(rs>0) - { - ColXprSegment x(mat.col(i).tail(rs)); - TempVecSegment y(temp.tail(rs)); - apply_rotation_in_the_plane(x, y, g); + using std::sqrt; + typedef typename MatrixType::Scalar Scalar; + typedef typename MatrixType::RealScalar RealScalar; + typedef typename MatrixType::ColXpr ColXpr; + typedef typename internal::remove_all::type ColXprCleaned; + typedef typename ColXprCleaned::SegmentReturnType ColXprSegment; + typedef Matrix TempVectorType; + typedef typename TempVectorType::SegmentReturnType TempVecSegment; + + Index n = mat.cols(); + eigen_assert(mat.rows() == n && vec.size() == n); + + TempVectorType temp; + + if (sigma > 0) { + // This version is based on Givens rotations. + // It is faster than the other one below, but only works for updates, + // i.e., for sigma > 0 + temp = sqrt(sigma) * vec; + + for (Index i = 0; i < n; ++i) { + JacobiRotation g; + g.makeGivens(mat(i, i), -temp(i), &mat(i, i)); + + Index rs = n - i - 1; + if (rs > 0) { + ColXprSegment x(mat.col(i).tail(rs)); + TempVecSegment y(temp.tail(rs)); + apply_rotation_in_the_plane(x, y, g); + } + } + } else { + temp = vec; + RealScalar beta = 1; + for (Index j = 0; j < n; ++j) { + RealScalar Ljj = numext::real(mat.coeff(j, j)); + RealScalar dj = numext::abs2(Ljj); + Scalar wj = temp.coeff(j); + RealScalar swj2 = sigma * numext::abs2(wj); + RealScalar gamma = dj * beta + swj2; + + RealScalar x = dj + swj2 / beta; + if (x <= RealScalar(0)) return j; + RealScalar nLjj = sqrt(x); + mat.coeffRef(j, j) = nLjj; + beta += swj2 / dj; + + // Update the terms of L + Index rs = n - j - 1; + if (rs) { + temp.tail(rs) -= (wj / Ljj) * mat.col(j).tail(rs); + if (gamma != 0) + mat.col(j).tail(rs) = + (nLjj / Ljj) * mat.col(j).tail(rs) + (nLjj * sigma * numext::conj(wj) / gamma) * temp.tail(rs); + } } } + return -1; } - else + + template struct llt_inplace { - temp = vec; - RealScalar beta = 1; - for(Index j=0; j::Real RealScalar; + template static Index unblocked(MatrixType &mat) { - RealScalar Ljj = numext::real(mat.coeff(j,j)); - RealScalar dj = numext::abs2(Ljj); - Scalar wj = temp.coeff(j); - RealScalar swj2 = sigma*numext::abs2(wj); - RealScalar gamma = dj*beta + swj2; - - RealScalar x = dj + swj2/beta; - if (x<=RealScalar(0)) - return j; - RealScalar nLjj = sqrt(x); - mat.coeffRef(j,j) = nLjj; - beta += swj2/dj; - - // Update the terms of L - Index rs = n-j-1; - if(rs) - { - temp.tail(rs) -= (wj/Ljj) * mat.col(j).tail(rs); - if(gamma != 0) - mat.col(j).tail(rs) = (nLjj/Ljj) * mat.col(j).tail(rs) + (nLjj * sigma*numext::conj(wj)/gamma)*temp.tail(rs); + using std::sqrt; + + eigen_assert(mat.rows() == mat.cols()); + const Index size = mat.rows(); + for (Index k = 0; k < size; ++k) { + Index rs = size - k - 1;// remaining size + + Block A21(mat, k + 1, k, rs, 1); + Block A10(mat, k, 0, 1, k); + Block A20(mat, k + 1, 0, rs, k); + + RealScalar x = numext::real(mat.coeff(k, k)); + if (k > 0) x -= A10.squaredNorm(); + if (x <= RealScalar(0)) return k; + mat.coeffRef(k, k) = x = sqrt(x); + if (k > 0 && rs > 0) A21.noalias() -= A20 * A10.adjoint(); + if (rs > 0) A21 /= x; } + return -1; } - } - return -1; -} -template struct llt_inplace -{ - typedef typename NumTraits::Real RealScalar; - template - static Index unblocked(MatrixType& mat) - { - using std::sqrt; + template static Index blocked(MatrixType &m) + { + eigen_assert(m.rows() == m.cols()); + Index size = m.rows(); + if (size < 32) return unblocked(m); + + Index blockSize = size / 8; + blockSize = (blockSize / 16) * 16; + blockSize = (std::min)((std::max)(blockSize, Index(8)), Index(128)); + + for (Index k = 0; k < size; k += blockSize) { + // partition the matrix: + // A00 | - | - + // lu = A10 | A11 | - + // A20 | A21 | A22 + Index bs = (std::min)(blockSize, size - k); + Index rs = size - k - bs; + Block A11(m, k, k, bs, bs); + Block A21(m, k + bs, k, rs, bs); + Block A22(m, k + bs, k + bs, rs, rs); + + Index ret; + if ((ret = unblocked(A11)) >= 0) return k + ret; + if (rs > 0) A11.adjoint().template triangularView().template solveInPlace(A21); + if (rs > 0) + A22.template selfadjointView().rankUpdate( + A21, typename NumTraits::Literal(-1));// bottleneck + } + return -1; + } - eigen_assert(mat.rows()==mat.cols()); - const Index size = mat.rows(); - for(Index k = 0; k < size; ++k) + template + static Index rankUpdate(MatrixType &mat, const VectorType &vec, const RealScalar &sigma) { - Index rs = size-k-1; // remaining size - - Block A21(mat,k+1,k,rs,1); - Block A10(mat,k,0,1,k); - Block A20(mat,k+1,0,rs,k); - - RealScalar x = numext::real(mat.coeff(k,k)); - if (k>0) x -= A10.squaredNorm(); - if (x<=RealScalar(0)) - return k; - mat.coeffRef(k,k) = x = sqrt(x); - if (k>0 && rs>0) A21.noalias() -= A20 * A10.adjoint(); - if (rs>0) A21 /= x; + return Eigen::internal::llt_rank_update_lower(mat, vec, sigma); } - return -1; - } + }; - template - static Index blocked(MatrixType& m) + template struct llt_inplace { - eigen_assert(m.rows()==m.cols()); - Index size = m.rows(); - if(size<32) - return unblocked(m); - - Index blockSize = size/8; - blockSize = (blockSize/16)*16; - blockSize = (std::min)((std::max)(blockSize,Index(8)), Index(128)); + typedef typename NumTraits::Real RealScalar; - for (Index k=0; k static EIGEN_STRONG_INLINE Index unblocked(MatrixType &mat) { - // partition the matrix: - // A00 | - | - - // lu = A10 | A11 | - - // A20 | A21 | A22 - Index bs = (std::min)(blockSize, size-k); - Index rs = size - k - bs; - Block A11(m,k, k, bs,bs); - Block A21(m,k+bs,k, rs,bs); - Block A22(m,k+bs,k+bs,rs,rs); - - Index ret; - if((ret=unblocked(A11))>=0) return k+ret; - if(rs>0) A11.adjoint().template triangularView().template solveInPlace(A21); - if(rs>0) A22.template selfadjointView().rankUpdate(A21,typename NumTraits::Literal(-1)); // bottleneck + Transpose matt(mat); + return llt_inplace::unblocked(matt); } - return -1; - } + template static EIGEN_STRONG_INLINE Index blocked(MatrixType &mat) + { + Transpose matt(mat); + return llt_inplace::blocked(matt); + } + template + static Index rankUpdate(MatrixType &mat, const VectorType &vec, const RealScalar &sigma) + { + Transpose matt(mat); + return llt_inplace::rankUpdate(matt, vec.conjugate(), sigma); + } + }; - template - static Index rankUpdate(MatrixType& mat, const VectorType& vec, const RealScalar& sigma) + template struct LLT_Traits { - return Eigen::internal::llt_rank_update_lower(mat, vec, sigma); - } -}; - -template struct llt_inplace -{ - typedef typename NumTraits::Real RealScalar; + typedef const TriangularView MatrixL; + typedef const TriangularView MatrixU; + static inline MatrixL getL(const MatrixType &m) { return MatrixL(m); } + static inline MatrixU getU(const MatrixType &m) { return MatrixU(m.adjoint()); } + static bool inplace_decomposition(MatrixType &m) + { + return llt_inplace::blocked(m) == -1; + } + }; - template - static EIGEN_STRONG_INLINE Index unblocked(MatrixType& mat) + template struct LLT_Traits { - Transpose matt(mat); - return llt_inplace::unblocked(matt); - } - template - static EIGEN_STRONG_INLINE Index blocked(MatrixType& mat) - { - Transpose matt(mat); - return llt_inplace::blocked(matt); - } - template - static Index rankUpdate(MatrixType& mat, const VectorType& vec, const RealScalar& sigma) - { - Transpose matt(mat); - return llt_inplace::rankUpdate(matt, vec.conjugate(), sigma); - } -}; - -template struct LLT_Traits -{ - typedef const TriangularView MatrixL; - typedef const TriangularView MatrixU; - static inline MatrixL getL(const MatrixType& m) { return MatrixL(m); } - static inline MatrixU getU(const MatrixType& m) { return MatrixU(m.adjoint()); } - static bool inplace_decomposition(MatrixType& m) - { return llt_inplace::blocked(m)==-1; } -}; - -template struct LLT_Traits -{ - typedef const TriangularView MatrixL; - typedef const TriangularView MatrixU; - static inline MatrixL getL(const MatrixType& m) { return MatrixL(m.adjoint()); } - static inline MatrixU getU(const MatrixType& m) { return MatrixU(m); } - static bool inplace_decomposition(MatrixType& m) - { return llt_inplace::blocked(m)==-1; } -}; + typedef const TriangularView MatrixL; + typedef const TriangularView MatrixU; + static inline MatrixL getL(const MatrixType &m) { return MatrixL(m.adjoint()); } + static inline MatrixU getU(const MatrixType &m) { return MatrixU(m); } + static bool inplace_decomposition(MatrixType &m) + { + return llt_inplace::blocked(m) == -1; + } + }; -} // end namespace internal +}// end namespace internal /** Computes / recomputes the Cholesky decomposition A = LL^* = U^*U of \a matrix - * - * \returns a reference to *this - * - * Example: \include TutorialLinAlgComputeTwice.cpp - * Output: \verbinclude TutorialLinAlgComputeTwice.out - */ + * + * \returns a reference to *this + * + * Example: \include TutorialLinAlgComputeTwice.cpp + * Output: \verbinclude TutorialLinAlgComputeTwice.out + */ template template -LLT& LLT::compute(const EigenBase& a) +LLT &LLT::compute(const EigenBase &a) { check_template_parameters(); - eigen_assert(a.rows()==a.cols()); + eigen_assert(a.rows() == a.cols()); const Index size = a.rows(); m_matrix.resize(size, size); - if (!internal::is_same_dense(m_matrix, a.derived())) - m_matrix = a.derived(); + if (!internal::is_same_dense(m_matrix, a.derived())) m_matrix = a.derived(); // Compute matrix L1 norm = max abs column sum. m_l1_norm = RealScalar(0); @@ -438,11 +410,12 @@ LLT& LLT::compute(const EigenBase for (Index col = 0; col < size; ++col) { RealScalar abs_col_sum; if (_UpLo == Lower) - abs_col_sum = m_matrix.col(col).tail(size - col).template lpNorm<1>() + m_matrix.row(col).head(col).template lpNorm<1>(); + abs_col_sum = + m_matrix.col(col).tail(size - col).template lpNorm<1>() + m_matrix.row(col).head(col).template lpNorm<1>(); else - abs_col_sum = m_matrix.col(col).head(col).template lpNorm<1>() + m_matrix.row(col).tail(size - col).template lpNorm<1>(); - if (abs_col_sum > m_l1_norm) - m_l1_norm = abs_col_sum; + abs_col_sum = + m_matrix.col(col).head(col).template lpNorm<1>() + m_matrix.row(col).tail(size - col).template lpNorm<1>(); + if (abs_col_sum > m_l1_norm) m_l1_norm = abs_col_sum; } m_isInitialized = true; @@ -453,18 +426,18 @@ LLT& LLT::compute(const EigenBase } /** Performs a rank one update (or dowdate) of the current decomposition. - * If A = LL^* before the rank one update, - * then after it we have LL^* = A + sigma * v v^* where \a v must be a vector - * of same dimension. - */ + * If A = LL^* before the rank one update, + * then after it we have LL^* = A + sigma * v v^* where \a v must be a vector + * of same dimension. + */ template template -LLT<_MatrixType,_UpLo> LLT<_MatrixType,_UpLo>::rankUpdate(const VectorType& v, const RealScalar& sigma) +LLT<_MatrixType, _UpLo> LLT<_MatrixType, _UpLo>::rankUpdate(const VectorType &v, const RealScalar &sigma) { EIGEN_STATIC_ASSERT_VECTOR_ONLY(VectorType); - eigen_assert(v.size()==m_matrix.cols()); + eigen_assert(v.size() == m_matrix.cols()); eigen_assert(m_isInitialized); - if(internal::llt_inplace::rankUpdate(m_matrix,v,sigma)>=0) + if (internal::llt_inplace::rankUpdate(m_matrix, v, sigma) >= 0) m_info = NumericalIssue; else m_info = Success; @@ -473,9 +446,9 @@ LLT<_MatrixType,_UpLo> LLT<_MatrixType,_UpLo>::rankUpdate(const VectorType& v, c } #ifndef EIGEN_PARSED_BY_DOXYGEN -template +template template -void LLT<_MatrixType,_UpLo>::_solve_impl(const RhsType &rhs, DstType &dst) const +void LLT<_MatrixType, _UpLo>::_solve_impl(const RhsType &rhs, DstType &dst) const { dst = rhs; solveInPlace(dst); @@ -483,24 +456,24 @@ void LLT<_MatrixType,_UpLo>::_solve_impl(const RhsType &rhs, DstType &dst) const #endif /** \internal use x = llt_object.solve(x); - * - * This is the \em in-place version of solve(). - * - * \param bAndX represents both the right-hand side matrix b and result x. - * - * This version avoids a copy when the right hand side matrix b is not needed anymore. - * - * \warning The parameter is only marked 'const' to make the C++ compiler accept a temporary expression here. - * This function will const_cast it, so constness isn't honored here. - * - * \sa LLT::solve(), MatrixBase::llt() - */ + * + * This is the \em in-place version of solve(). + * + * \param bAndX represents both the right-hand side matrix b and result x. + * + * This version avoids a copy when the right hand side matrix b is not needed anymore. + * + * \warning The parameter is only marked 'const' to make the C++ compiler accept a temporary expression here. + * This function will const_cast it, so constness isn't honored here. + * + * \sa LLT::solve(), MatrixBase::llt() + */ template template -void LLT::solveInPlace(const MatrixBase &bAndX) const +void LLT::solveInPlace(const MatrixBase &bAndX) const { eigen_assert(m_isInitialized && "LLT is not initialized."); - eigen_assert(m_matrix.rows()==bAndX.rows()); + eigen_assert(m_matrix.rows() == bAndX.rows()); matrixL().solveInPlace(bAndX); matrixU().solveInPlace(bAndX); } @@ -508,35 +481,32 @@ void LLT::solveInPlace(const MatrixBase &bAndX) const /** \returns the matrix represented by the decomposition, * i.e., it returns the product: L L^*. * This function is provided for debug purpose. */ -template -MatrixType LLT::reconstructedMatrix() const +template MatrixType LLT::reconstructedMatrix() const { eigen_assert(m_isInitialized && "LLT is not initialized."); return matrixL() * matrixL().adjoint().toDenseMatrix(); } /** \cholesky_module - * \returns the LLT decomposition of \c *this - * \sa SelfAdjointView::llt() - */ -template -inline const LLT::PlainObject> -MatrixBase::llt() const + * \returns the LLT decomposition of \c *this + * \sa SelfAdjointView::llt() + */ +template inline const LLT::PlainObject> MatrixBase::llt() const { return LLT(derived()); } /** \cholesky_module - * \returns the LLT decomposition of \c *this - * \sa SelfAdjointView::llt() - */ + * \returns the LLT decomposition of \c *this + * \sa SelfAdjointView::llt() + */ template inline const LLT::PlainObject, UpLo> -SelfAdjointView::llt() const + SelfAdjointView::llt() const { - return LLT(m_matrix); + return LLT(m_matrix); } -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_LLT_H +#endif// EIGEN_LLT_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Cholesky/LLT_LAPACKE.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Cholesky/LLT_LAPACKE.h index bc6489e6..6788af8b 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Cholesky/LLT_LAPACKE.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Cholesky/LLT_LAPACKE.h @@ -33,67 +33,60 @@ #ifndef EIGEN_LLT_LAPACKE_H #define EIGEN_LLT_LAPACKE_H -namespace Eigen { +namespace Eigen { namespace internal { -template struct lapacke_llt; + template struct lapacke_llt; -#define EIGEN_LAPACKE_LLT(EIGTYPE, BLASTYPE, LAPACKE_PREFIX) \ -template<> struct lapacke_llt \ -{ \ - template \ - static inline Index potrf(MatrixType& m, char uplo) \ - { \ - lapack_int matrix_order; \ - lapack_int size, lda, info, StorageOrder; \ - EIGTYPE* a; \ - eigen_assert(m.rows()==m.cols()); \ - /* Set up parameters for ?potrf */ \ - size = convert_index(m.rows()); \ - StorageOrder = MatrixType::Flags&RowMajorBit?RowMajor:ColMajor; \ - matrix_order = StorageOrder==RowMajor ? LAPACK_ROW_MAJOR : LAPACK_COL_MAJOR; \ - a = &(m.coeffRef(0,0)); \ - lda = convert_index(m.outerStride()); \ -\ - info = LAPACKE_##LAPACKE_PREFIX##potrf( matrix_order, uplo, size, (BLASTYPE*)a, lda ); \ - info = (info==0) ? -1 : info>0 ? info-1 : size; \ - return info; \ - } \ -}; \ -template<> struct llt_inplace \ -{ \ - template \ - static Index blocked(MatrixType& m) \ - { \ - return lapacke_llt::potrf(m, 'L'); \ - } \ - template \ - static Index rankUpdate(MatrixType& mat, const VectorType& vec, const typename MatrixType::RealScalar& sigma) \ - { return Eigen::internal::llt_rank_update_lower(mat, vec, sigma); } \ -}; \ -template<> struct llt_inplace \ -{ \ - template \ - static Index blocked(MatrixType& m) \ - { \ - return lapacke_llt::potrf(m, 'U'); \ - } \ - template \ - static Index rankUpdate(MatrixType& mat, const VectorType& vec, const typename MatrixType::RealScalar& sigma) \ - { \ - Transpose matt(mat); \ - return llt_inplace::rankUpdate(matt, vec.conjugate(), sigma); \ - } \ -}; +#define EIGEN_LAPACKE_LLT(EIGTYPE, BLASTYPE, LAPACKE_PREFIX) \ + template<> struct lapacke_llt \ + { \ + template static inline Index potrf(MatrixType &m, char uplo) \ + { \ + lapack_int matrix_order; \ + lapack_int size, lda, info, StorageOrder; \ + EIGTYPE *a; \ + eigen_assert(m.rows() == m.cols()); \ + /* Set up parameters for ?potrf */ \ + size = convert_index(m.rows()); \ + StorageOrder = MatrixType::Flags & RowMajorBit ? RowMajor : ColMajor; \ + matrix_order = StorageOrder == RowMajor ? LAPACK_ROW_MAJOR : LAPACK_COL_MAJOR; \ + a = &(m.coeffRef(0, 0)); \ + lda = convert_index(m.outerStride()); \ + \ + info = LAPACKE_##LAPACKE_PREFIX##potrf(matrix_order, uplo, size, (BLASTYPE *)a, lda); \ + info = (info == 0) ? -1 : info > 0 ? info - 1 : size; \ + return info; \ + } \ + }; \ + template<> struct llt_inplace \ + { \ + template static Index blocked(MatrixType &m) { return lapacke_llt::potrf(m, 'L'); } \ + template \ + static Index rankUpdate(MatrixType &mat, const VectorType &vec, const typename MatrixType::RealScalar &sigma) \ + { \ + return Eigen::internal::llt_rank_update_lower(mat, vec, sigma); \ + } \ + }; \ + template<> struct llt_inplace \ + { \ + template static Index blocked(MatrixType &m) { return lapacke_llt::potrf(m, 'U'); } \ + template \ + static Index rankUpdate(MatrixType &mat, const VectorType &vec, const typename MatrixType::RealScalar &sigma) \ + { \ + Transpose matt(mat); \ + return llt_inplace::rankUpdate(matt, vec.conjugate(), sigma); \ + } \ + }; -EIGEN_LAPACKE_LLT(double, double, d) -EIGEN_LAPACKE_LLT(float, float, s) -EIGEN_LAPACKE_LLT(dcomplex, lapack_complex_double, z) -EIGEN_LAPACKE_LLT(scomplex, lapack_complex_float, c) + EIGEN_LAPACKE_LLT(double, double, d) + EIGEN_LAPACKE_LLT(float, float, s) + EIGEN_LAPACKE_LLT(dcomplex, lapack_complex_double, z) + EIGEN_LAPACKE_LLT(scomplex, lapack_complex_float, c) -} // end namespace internal +}// end namespace internal -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_LLT_LAPACKE_H +#endif// EIGEN_LLT_LAPACKE_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/CholmodSupport/CholmodSupport.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/CholmodSupport/CholmodSupport.h index 57197202..03916010 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/CholmodSupport/CholmodSupport.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/CholmodSupport/CholmodSupport.h @@ -10,139 +10,133 @@ #ifndef EIGEN_CHOLMODSUPPORT_H #define EIGEN_CHOLMODSUPPORT_H -namespace Eigen { +namespace Eigen { namespace internal { -template struct cholmod_configure_matrix; + template struct cholmod_configure_matrix; -template<> struct cholmod_configure_matrix { - template - static void run(CholmodType& mat) { - mat.xtype = CHOLMOD_REAL; - mat.dtype = CHOLMOD_DOUBLE; - } -}; - -template<> struct cholmod_configure_matrix > { - template - static void run(CholmodType& mat) { - mat.xtype = CHOLMOD_COMPLEX; - mat.dtype = CHOLMOD_DOUBLE; - } -}; - -// Other scalar types are not yet suppotred by Cholmod -// template<> struct cholmod_configure_matrix { -// template -// static void run(CholmodType& mat) { -// mat.xtype = CHOLMOD_REAL; -// mat.dtype = CHOLMOD_SINGLE; -// } -// }; -// -// template<> struct cholmod_configure_matrix > { -// template -// static void run(CholmodType& mat) { -// mat.xtype = CHOLMOD_COMPLEX; -// mat.dtype = CHOLMOD_SINGLE; -// } -// }; + template<> struct cholmod_configure_matrix + { + template static void run(CholmodType &mat) + { + mat.xtype = CHOLMOD_REAL; + mat.dtype = CHOLMOD_DOUBLE; + } + }; -} // namespace internal + template<> struct cholmod_configure_matrix> + { + template static void run(CholmodType &mat) + { + mat.xtype = CHOLMOD_COMPLEX; + mat.dtype = CHOLMOD_DOUBLE; + } + }; + + // Other scalar types are not yet suppotred by Cholmod + // template<> struct cholmod_configure_matrix { + // template + // static void run(CholmodType& mat) { + // mat.xtype = CHOLMOD_REAL; + // mat.dtype = CHOLMOD_SINGLE; + // } + // }; + // + // template<> struct cholmod_configure_matrix > { + // template + // static void run(CholmodType& mat) { + // mat.xtype = CHOLMOD_COMPLEX; + // mat.dtype = CHOLMOD_SINGLE; + // } + // }; + +}// namespace internal /** Wraps the Eigen sparse matrix \a mat into a Cholmod sparse matrix object. - * Note that the data are shared. - */ + * Note that the data are shared. + */ template -cholmod_sparse viewAsCholmod(Ref > mat) +cholmod_sparse viewAsCholmod(Ref> mat) { cholmod_sparse res; - res.nzmax = mat.nonZeros(); - res.nrow = mat.rows(); - res.ncol = mat.cols(); - res.p = mat.outerIndexPtr(); - res.i = mat.innerIndexPtr(); - res.x = mat.valuePtr(); - res.z = 0; - res.sorted = 1; - if(mat.isCompressed()) - { - res.packed = 1; + res.nzmax = mat.nonZeros(); + res.nrow = mat.rows(); + res.ncol = mat.cols(); + res.p = mat.outerIndexPtr(); + res.i = mat.innerIndexPtr(); + res.x = mat.valuePtr(); + res.z = 0; + res.sorted = 1; + if (mat.isCompressed()) { + res.packed = 1; res.nz = 0; - } - else - { - res.packed = 0; + } else { + res.packed = 0; res.nz = mat.innerNonZeroPtr(); } - res.dtype = 0; - res.stype = -1; - - if (internal::is_same<_StorageIndex,int>::value) - { + res.dtype = 0; + res.stype = -1; + + if (internal::is_same<_StorageIndex, int>::value) { res.itype = CHOLMOD_INT; - } - else if (internal::is_same<_StorageIndex,long>::value) - { + } else if (internal::is_same<_StorageIndex, long>::value) { res.itype = CHOLMOD_LONG; - } - else - { + } else { eigen_assert(false && "Index type not supported yet"); } // setup res.xtype internal::cholmod_configure_matrix<_Scalar>::run(res); - + res.stype = 0; - + return res; } template -const cholmod_sparse viewAsCholmod(const SparseMatrix<_Scalar,_Options,_Index>& mat) +const cholmod_sparse viewAsCholmod(const SparseMatrix<_Scalar, _Options, _Index> &mat) { - cholmod_sparse res = viewAsCholmod(Ref >(mat.const_cast_derived())); + cholmod_sparse res = viewAsCholmod(Ref>(mat.const_cast_derived())); return res; } template -const cholmod_sparse viewAsCholmod(const SparseVector<_Scalar,_Options,_Index>& mat) +const cholmod_sparse viewAsCholmod(const SparseVector<_Scalar, _Options, _Index> &mat) { - cholmod_sparse res = viewAsCholmod(Ref >(mat.const_cast_derived())); + cholmod_sparse res = viewAsCholmod(Ref>(mat.const_cast_derived())); return res; } /** Returns a view of the Eigen sparse matrix \a mat as Cholmod sparse matrix. - * The data are not copied but shared. */ + * The data are not copied but shared. */ template -cholmod_sparse viewAsCholmod(const SparseSelfAdjointView, UpLo>& mat) +cholmod_sparse viewAsCholmod(const SparseSelfAdjointView, UpLo> &mat) { - cholmod_sparse res = viewAsCholmod(Ref >(mat.matrix().const_cast_derived())); - - if(UpLo==Upper) res.stype = 1; - if(UpLo==Lower) res.stype = -1; + cholmod_sparse res = viewAsCholmod(Ref>(mat.matrix().const_cast_derived())); + + if (UpLo == Upper) res.stype = 1; + if (UpLo == Lower) res.stype = -1; return res; } /** Returns a view of the Eigen \b dense matrix \a mat as Cholmod dense matrix. - * The data are not copied but shared. */ -template -cholmod_dense viewAsCholmod(MatrixBase& mat) + * The data are not copied but shared. */ +template cholmod_dense viewAsCholmod(MatrixBase &mat) { - EIGEN_STATIC_ASSERT((internal::traits::Flags&RowMajorBit)==0,THIS_METHOD_IS_ONLY_FOR_COLUMN_MAJOR_MATRICES); + EIGEN_STATIC_ASSERT( + (internal::traits::Flags & RowMajorBit) == 0, THIS_METHOD_IS_ONLY_FOR_COLUMN_MAJOR_MATRICES); typedef typename Derived::Scalar Scalar; cholmod_dense res; - res.nrow = mat.rows(); - res.ncol = mat.cols(); - res.nzmax = res.nrow * res.ncol; - res.d = Derived::IsVectorAtCompileTime ? mat.derived().size() : mat.derived().outerStride(); - res.x = (void*)(mat.derived().data()); - res.z = 0; + res.nrow = mat.rows(); + res.ncol = mat.cols(); + res.nzmax = res.nrow * res.ncol; + res.d = Derived::IsVectorAtCompileTime ? mat.derived().size() : mat.derived().outerStride(); + res.x = (void *)(mat.derived().data()); + res.z = 0; internal::cholmod_configure_matrix::run(res); @@ -150,490 +144,479 @@ cholmod_dense viewAsCholmod(MatrixBase& mat) } /** Returns a view of the Cholmod sparse matrix \a cm as an Eigen sparse matrix. - * The data are not copied but shared. */ + * The data are not copied but shared. */ template -MappedSparseMatrix viewAsEigen(cholmod_sparse& cm) +MappedSparseMatrix viewAsEigen(cholmod_sparse &cm) { - return MappedSparseMatrix - (cm.nrow, cm.ncol, static_cast(cm.p)[cm.ncol], - static_cast(cm.p), static_cast(cm.i),static_cast(cm.x) ); + return MappedSparseMatrix(cm.nrow, + cm.ncol, + static_cast(cm.p)[cm.ncol], + static_cast(cm.p), + static_cast(cm.i), + static_cast(cm.x)); } -enum CholmodMode { - CholmodAuto, CholmodSimplicialLLt, CholmodSupernodalLLt, CholmodLDLt -}; +enum CholmodMode { CholmodAuto, CholmodSimplicialLLt, CholmodSupernodalLLt, CholmodLDLt }; /** \ingroup CholmodSupport_Module - * \class CholmodBase - * \brief The base class for the direct Cholesky factorization of Cholmod - * \sa class CholmodSupernodalLLT, class CholmodSimplicialLDLT, class CholmodSimplicialLLT - */ -template -class CholmodBase : public SparseSolverBase + * \class CholmodBase + * \brief The base class for the direct Cholesky factorization of Cholmod + * \sa class CholmodSupernodalLLT, class CholmodSimplicialLDLT, class CholmodSimplicialLLT + */ +template class CholmodBase : public SparseSolverBase { - protected: - typedef SparseSolverBase Base; - using Base::derived; - using Base::m_isInitialized; - public: - typedef _MatrixType MatrixType; - enum { UpLo = _UpLo }; - typedef typename MatrixType::Scalar Scalar; - typedef typename MatrixType::RealScalar RealScalar; - typedef MatrixType CholMatrixType; - typedef typename MatrixType::StorageIndex StorageIndex; - enum { - ColsAtCompileTime = MatrixType::ColsAtCompileTime, - MaxColsAtCompileTime = MatrixType::MaxColsAtCompileTime - }; - - public: - - CholmodBase() - : m_cholmodFactor(0), m_info(Success), m_factorizationIsOk(false), m_analysisIsOk(false) - { - EIGEN_STATIC_ASSERT((internal::is_same::value), CHOLMOD_SUPPORTS_DOUBLE_PRECISION_ONLY); - m_shiftOffset[0] = m_shiftOffset[1] = 0.0; - cholmod_start(&m_cholmod); - } +protected: + typedef SparseSolverBase Base; + using Base::derived; + using Base::m_isInitialized; + +public: + typedef _MatrixType MatrixType; + enum { UpLo = _UpLo }; + typedef typename MatrixType::Scalar Scalar; + typedef typename MatrixType::RealScalar RealScalar; + typedef MatrixType CholMatrixType; + typedef typename MatrixType::StorageIndex StorageIndex; + enum { ColsAtCompileTime = MatrixType::ColsAtCompileTime, MaxColsAtCompileTime = MatrixType::MaxColsAtCompileTime }; + +public: + CholmodBase() : m_cholmodFactor(0), m_info(Success), m_factorizationIsOk(false), m_analysisIsOk(false) + { + EIGEN_STATIC_ASSERT((internal::is_same::value), CHOLMOD_SUPPORTS_DOUBLE_PRECISION_ONLY); + m_shiftOffset[0] = m_shiftOffset[1] = 0.0; + cholmod_start(&m_cholmod); + } - explicit CholmodBase(const MatrixType& matrix) - : m_cholmodFactor(0), m_info(Success), m_factorizationIsOk(false), m_analysisIsOk(false) - { - EIGEN_STATIC_ASSERT((internal::is_same::value), CHOLMOD_SUPPORTS_DOUBLE_PRECISION_ONLY); - m_shiftOffset[0] = m_shiftOffset[1] = 0.0; - cholmod_start(&m_cholmod); - compute(matrix); - } + explicit CholmodBase(const MatrixType &matrix) + : m_cholmodFactor(0), m_info(Success), m_factorizationIsOk(false), m_analysisIsOk(false) + { + EIGEN_STATIC_ASSERT((internal::is_same::value), CHOLMOD_SUPPORTS_DOUBLE_PRECISION_ONLY); + m_shiftOffset[0] = m_shiftOffset[1] = 0.0; + cholmod_start(&m_cholmod); + compute(matrix); + } - ~CholmodBase() - { - if(m_cholmodFactor) - cholmod_free_factor(&m_cholmodFactor, &m_cholmod); - cholmod_finish(&m_cholmod); - } - - inline StorageIndex cols() const { return internal::convert_index(m_cholmodFactor->n); } - inline StorageIndex rows() const { return internal::convert_index(m_cholmodFactor->n); } - - /** \brief Reports whether previous computation was successful. - * - * \returns \c Success if computation was succesful, - * \c NumericalIssue if the matrix.appears to be negative. - */ - ComputationInfo info() const - { - eigen_assert(m_isInitialized && "Decomposition is not initialized."); - return m_info; - } + ~CholmodBase() + { + if (m_cholmodFactor) cholmod_free_factor(&m_cholmodFactor, &m_cholmod); + cholmod_finish(&m_cholmod); + } - /** Computes the sparse Cholesky decomposition of \a matrix */ - Derived& compute(const MatrixType& matrix) - { - analyzePattern(matrix); - factorize(matrix); - return derived(); - } - - /** Performs a symbolic decomposition on the sparsity pattern of \a matrix. - * - * This function is particularly useful when solving for several problems having the same structure. - * - * \sa factorize() - */ - void analyzePattern(const MatrixType& matrix) - { - if(m_cholmodFactor) - { - cholmod_free_factor(&m_cholmodFactor, &m_cholmod); - m_cholmodFactor = 0; - } - cholmod_sparse A = viewAsCholmod(matrix.template selfadjointView()); - m_cholmodFactor = cholmod_analyze(&A, &m_cholmod); - - this->m_isInitialized = true; - this->m_info = Success; - m_analysisIsOk = true; - m_factorizationIsOk = false; - } - - /** Performs a numeric decomposition of \a matrix - * - * The given matrix must have the same sparsity pattern as the matrix on which the symbolic decomposition has been performed. - * - * \sa analyzePattern() - */ - void factorize(const MatrixType& matrix) - { - eigen_assert(m_analysisIsOk && "You must first call analyzePattern()"); - cholmod_sparse A = viewAsCholmod(matrix.template selfadjointView()); - cholmod_factorize_p(&A, m_shiftOffset, 0, 0, m_cholmodFactor, &m_cholmod); + inline StorageIndex cols() const { return internal::convert_index(m_cholmodFactor->n); } + inline StorageIndex rows() const { return internal::convert_index(m_cholmodFactor->n); } - // If the factorization failed, minor is the column at which it did. On success minor == n. - this->m_info = (m_cholmodFactor->minor == m_cholmodFactor->n ? Success : NumericalIssue); - m_factorizationIsOk = true; - } - - /** Returns a reference to the Cholmod's configuration structure to get a full control over the performed operations. - * See the Cholmod user guide for details. */ - cholmod_common& cholmod() { return m_cholmod; } - - #ifndef EIGEN_PARSED_BY_DOXYGEN - /** \internal */ - template - void _solve_impl(const MatrixBase &b, MatrixBase &dest) const - { - eigen_assert(m_factorizationIsOk && "The decomposition is not in a valid state for solving, you must first call either compute() or symbolic()/numeric()"); - const Index size = m_cholmodFactor->n; - EIGEN_UNUSED_VARIABLE(size); - eigen_assert(size==b.rows()); - - // Cholmod needs column-major stoarge without inner-stride, which corresponds to the default behavior of Ref. - Ref > b_ref(b.derived()); - - cholmod_dense b_cd = viewAsCholmod(b_ref); - cholmod_dense* x_cd = cholmod_solve(CHOLMOD_A, m_cholmodFactor, &b_cd, &m_cholmod); - if(!x_cd) - { - this->m_info = NumericalIssue; - return; - } - // TODO optimize this copy by swapping when possible (be careful with alignment, etc.) - dest = Matrix::Map(reinterpret_cast(x_cd->x),b.rows(),b.cols()); - cholmod_free_dense(&x_cd, &m_cholmod); - } - - /** \internal */ - template - void _solve_impl(const SparseMatrixBase &b, SparseMatrixBase &dest) const - { - eigen_assert(m_factorizationIsOk && "The decomposition is not in a valid state for solving, you must first call either compute() or symbolic()/numeric()"); - const Index size = m_cholmodFactor->n; - EIGEN_UNUSED_VARIABLE(size); - eigen_assert(size==b.rows()); - - // note: cs stands for Cholmod Sparse - Ref > b_ref(b.const_cast_derived()); - cholmod_sparse b_cs = viewAsCholmod(b_ref); - cholmod_sparse* x_cs = cholmod_spsolve(CHOLMOD_A, m_cholmodFactor, &b_cs, &m_cholmod); - if(!x_cs) - { - this->m_info = NumericalIssue; - return; - } - // TODO optimize this copy by swapping when possible (be careful with alignment, etc.) - dest.derived() = viewAsEigen(*x_cs); - cholmod_free_sparse(&x_cs, &m_cholmod); + /** \brief Reports whether previous computation was successful. + * + * \returns \c Success if computation was succesful, + * \c NumericalIssue if the matrix.appears to be negative. + */ + ComputationInfo info() const + { + eigen_assert(m_isInitialized && "Decomposition is not initialized."); + return m_info; + } + + /** Computes the sparse Cholesky decomposition of \a matrix */ + Derived &compute(const MatrixType &matrix) + { + analyzePattern(matrix); + factorize(matrix); + return derived(); + } + + /** Performs a symbolic decomposition on the sparsity pattern of \a matrix. + * + * This function is particularly useful when solving for several problems having the same structure. + * + * \sa factorize() + */ + void analyzePattern(const MatrixType &matrix) + { + if (m_cholmodFactor) { + cholmod_free_factor(&m_cholmodFactor, &m_cholmod); + m_cholmodFactor = 0; } - #endif // EIGEN_PARSED_BY_DOXYGEN - - - /** Sets the shift parameter that will be used to adjust the diagonal coefficients during the numerical factorization. - * - * During the numerical factorization, an offset term is added to the diagonal coefficients:\n - * \c d_ii = \a offset + \c d_ii - * - * The default is \a offset=0. - * - * \returns a reference to \c *this. - */ - Derived& setShift(const RealScalar& offset) - { - m_shiftOffset[0] = double(offset); - return derived(); + cholmod_sparse A = viewAsCholmod(matrix.template selfadjointView()); + m_cholmodFactor = cholmod_analyze(&A, &m_cholmod); + + this->m_isInitialized = true; + this->m_info = Success; + m_analysisIsOk = true; + m_factorizationIsOk = false; + } + + /** Performs a numeric decomposition of \a matrix + * + * The given matrix must have the same sparsity pattern as the matrix on which the symbolic decomposition has been + * performed. + * + * \sa analyzePattern() + */ + void factorize(const MatrixType &matrix) + { + eigen_assert(m_analysisIsOk && "You must first call analyzePattern()"); + cholmod_sparse A = viewAsCholmod(matrix.template selfadjointView()); + cholmod_factorize_p(&A, m_shiftOffset, 0, 0, m_cholmodFactor, &m_cholmod); + + // If the factorization failed, minor is the column at which it did. On success minor == n. + this->m_info = (m_cholmodFactor->minor == m_cholmodFactor->n ? Success : NumericalIssue); + m_factorizationIsOk = true; + } + + /** Returns a reference to the Cholmod's configuration structure to get a full control over the performed operations. + * See the Cholmod user guide for details. */ + cholmod_common &cholmod() { return m_cholmod; } + +#ifndef EIGEN_PARSED_BY_DOXYGEN + /** \internal */ + template void _solve_impl(const MatrixBase &b, MatrixBase &dest) const + { + eigen_assert(m_factorizationIsOk && "The decomposition is not in a valid state for solving, you must first call either compute() or symbolic()/numeric()"); + const Index size = m_cholmodFactor->n; + EIGEN_UNUSED_VARIABLE(size); + eigen_assert(size == b.rows()); + + // Cholmod needs column-major stoarge without inner-stride, which corresponds to the default behavior of Ref. + Ref> b_ref(b.derived()); + + cholmod_dense b_cd = viewAsCholmod(b_ref); + cholmod_dense *x_cd = cholmod_solve(CHOLMOD_A, m_cholmodFactor, &b_cd, &m_cholmod); + if (!x_cd) { + this->m_info = NumericalIssue; + return; } - - /** \returns the determinant of the underlying matrix from the current factorization */ - Scalar determinant() const - { - using std::exp; - return exp(logDeterminant()); + // TODO optimize this copy by swapping when possible (be careful with alignment, etc.) + dest = Matrix::Map( + reinterpret_cast(x_cd->x), b.rows(), b.cols()); + cholmod_free_dense(&x_cd, &m_cholmod); + } + + /** \internal */ + template + void _solve_impl(const SparseMatrixBase &b, SparseMatrixBase &dest) const + { + eigen_assert(m_factorizationIsOk && "The decomposition is not in a valid state for solving, you must first call either compute() or symbolic()/numeric()"); + const Index size = m_cholmodFactor->n; + EIGEN_UNUSED_VARIABLE(size); + eigen_assert(size == b.rows()); + + // note: cs stands for Cholmod Sparse + Ref> b_ref( + b.const_cast_derived()); + cholmod_sparse b_cs = viewAsCholmod(b_ref); + cholmod_sparse *x_cs = cholmod_spsolve(CHOLMOD_A, m_cholmodFactor, &b_cs, &m_cholmod); + if (!x_cs) { + this->m_info = NumericalIssue; + return; } + // TODO optimize this copy by swapping when possible (be careful with alignment, etc.) + dest.derived() = viewAsEigen(*x_cs); + cholmod_free_sparse(&x_cs, &m_cholmod); + } +#endif// EIGEN_PARSED_BY_DOXYGEN + + + /** Sets the shift parameter that will be used to adjust the diagonal coefficients during the numerical factorization. + * + * During the numerical factorization, an offset term is added to the diagonal coefficients:\n + * \c d_ii = \a offset + \c d_ii + * + * The default is \a offset=0. + * + * \returns a reference to \c *this. + */ + Derived &setShift(const RealScalar &offset) + { + m_shiftOffset[0] = double(offset); + return derived(); + } - /** \returns the log determinant of the underlying matrix from the current factorization */ - Scalar logDeterminant() const - { - using std::log; - using numext::real; - eigen_assert(m_factorizationIsOk && "The decomposition is not in a valid state for solving, you must first call either compute() or symbolic()/numeric()"); - - RealScalar logDet = 0; - Scalar *x = static_cast(m_cholmodFactor->x); - if (m_cholmodFactor->is_super) - { - // Supernodal factorization stored as a packed list of dense column-major blocs, - // as described by the following structure: - - // super[k] == index of the first column of the j-th super node - StorageIndex *super = static_cast(m_cholmodFactor->super); - // pi[k] == offset to the description of row indices - StorageIndex *pi = static_cast(m_cholmodFactor->pi); - // px[k] == offset to the respective dense block - StorageIndex *px = static_cast(m_cholmodFactor->px); - - Index nb_super_nodes = m_cholmodFactor->nsuper; - for (Index k=0; k < nb_super_nodes; ++k) - { - StorageIndex ncols = super[k + 1] - super[k]; - StorageIndex nrows = pi[k + 1] - pi[k]; - - Map, 0, InnerStride<> > sk(x + px[k], ncols, InnerStride<>(nrows+1)); - logDet += sk.real().log().sum(); - } - } - else - { - // Simplicial factorization stored as standard CSC matrix. - StorageIndex *p = static_cast(m_cholmodFactor->p); - Index size = m_cholmodFactor->n; - for (Index k=0; k(m_cholmodFactor->x); + if (m_cholmodFactor->is_super) { + // Supernodal factorization stored as a packed list of dense column-major blocs, + // as described by the following structure: + + // super[k] == index of the first column of the j-th super node + StorageIndex *super = static_cast(m_cholmodFactor->super); + // pi[k] == offset to the description of row indices + StorageIndex *pi = static_cast(m_cholmodFactor->pi); + // px[k] == offset to the respective dense block + StorageIndex *px = static_cast(m_cholmodFactor->px); + + Index nb_super_nodes = m_cholmodFactor->nsuper; + for (Index k = 0; k < nb_super_nodes; ++k) { + StorageIndex ncols = super[k + 1] - super[k]; + StorageIndex nrows = pi[k + 1] - pi[k]; + + Map, 0, InnerStride<>> sk(x + px[k], ncols, InnerStride<>(nrows + 1)); + logDet += sk.real().log().sum(); } - if (m_cholmodFactor->is_ll) - logDet *= 2.0; - return logDet; - }; - - template - void dumpMemory(Stream& /*s*/) - {} - - protected: - mutable cholmod_common m_cholmod; - cholmod_factor* m_cholmodFactor; - double m_shiftOffset[2]; - mutable ComputationInfo m_info; - int m_factorizationIsOk; - int m_analysisIsOk; + } else { + // Simplicial factorization stored as standard CSC matrix. + StorageIndex *p = static_cast(m_cholmodFactor->p); + Index size = m_cholmodFactor->n; + for (Index k = 0; k < size; ++k) logDet += log(real(x[p[k]])); + } + if (m_cholmodFactor->is_ll) logDet *= 2.0; + return logDet; + }; + + template void dumpMemory(Stream & /*s*/) {} + +protected: + mutable cholmod_common m_cholmod; + cholmod_factor *m_cholmodFactor; + double m_shiftOffset[2]; + mutable ComputationInfo m_info; + int m_factorizationIsOk; + int m_analysisIsOk; }; /** \ingroup CholmodSupport_Module - * \class CholmodSimplicialLLT - * \brief A simplicial direct Cholesky (LLT) factorization and solver based on Cholmod - * - * This class allows to solve for A.X = B sparse linear problems via a simplicial LL^T Cholesky factorization - * using the Cholmod library. - * This simplicial variant is equivalent to Eigen's built-in SimplicialLLT class. Therefore, it has little practical interest. - * The sparse matrix A must be selfadjoint and positive definite. The vectors or matrices - * X and B can be either dense or sparse. - * - * \tparam _MatrixType the type of the sparse matrix A, it must be a SparseMatrix<> - * \tparam _UpLo the triangular part that will be used for the computations. It can be Lower - * or Upper. Default is Lower. - * - * \implsparsesolverconcept - * - * This class supports all kind of SparseMatrix<>: row or column major; upper, lower, or both; compressed or non compressed. - * - * \warning Only double precision real and complex scalar types are supported by Cholmod. - * - * \sa \ref TutorialSparseSolverConcept, class CholmodSupernodalLLT, class SimplicialLLT - */ + * \class CholmodSimplicialLLT + * \brief A simplicial direct Cholesky (LLT) factorization and solver based on Cholmod + * + * This class allows to solve for A.X = B sparse linear problems via a simplicial LL^T Cholesky factorization + * using the Cholmod library. + * This simplicial variant is equivalent to Eigen's built-in SimplicialLLT class. Therefore, it has little practical + * interest. The sparse matrix A must be selfadjoint and positive definite. The vectors or matrices X and B can be + * either dense or sparse. + * + * \tparam _MatrixType the type of the sparse matrix A, it must be a SparseMatrix<> + * \tparam _UpLo the triangular part that will be used for the computations. It can be Lower + * or Upper. Default is Lower. + * + * \implsparsesolverconcept + * + * This class supports all kind of SparseMatrix<>: row or column major; upper, lower, or both; compressed or non + * compressed. + * + * \warning Only double precision real and complex scalar types are supported by Cholmod. + * + * \sa \ref TutorialSparseSolverConcept, class CholmodSupernodalLLT, class SimplicialLLT + */ template -class CholmodSimplicialLLT : public CholmodBase<_MatrixType, _UpLo, CholmodSimplicialLLT<_MatrixType, _UpLo> > +class CholmodSimplicialLLT : public CholmodBase<_MatrixType, _UpLo, CholmodSimplicialLLT<_MatrixType, _UpLo>> { - typedef CholmodBase<_MatrixType, _UpLo, CholmodSimplicialLLT> Base; - using Base::m_cholmod; - - public: - - typedef _MatrixType MatrixType; - - CholmodSimplicialLLT() : Base() { init(); } - - CholmodSimplicialLLT(const MatrixType& matrix) : Base() - { - init(); - this->compute(matrix); - } + typedef CholmodBase<_MatrixType, _UpLo, CholmodSimplicialLLT> Base; + using Base::m_cholmod; - ~CholmodSimplicialLLT() {} - protected: - void init() - { - m_cholmod.final_asis = 0; - m_cholmod.supernodal = CHOLMOD_SIMPLICIAL; - m_cholmod.final_ll = 1; - } +public: + typedef _MatrixType MatrixType; + + CholmodSimplicialLLT() : Base() { init(); } + + CholmodSimplicialLLT(const MatrixType &matrix) : Base() + { + init(); + this->compute(matrix); + } + + ~CholmodSimplicialLLT() {} + +protected: + void init() + { + m_cholmod.final_asis = 0; + m_cholmod.supernodal = CHOLMOD_SIMPLICIAL; + m_cholmod.final_ll = 1; + } }; /** \ingroup CholmodSupport_Module - * \class CholmodSimplicialLDLT - * \brief A simplicial direct Cholesky (LDLT) factorization and solver based on Cholmod - * - * This class allows to solve for A.X = B sparse linear problems via a simplicial LDL^T Cholesky factorization - * using the Cholmod library. - * This simplicial variant is equivalent to Eigen's built-in SimplicialLDLT class. Therefore, it has little practical interest. - * The sparse matrix A must be selfadjoint and positive definite. The vectors or matrices - * X and B can be either dense or sparse. - * - * \tparam _MatrixType the type of the sparse matrix A, it must be a SparseMatrix<> - * \tparam _UpLo the triangular part that will be used for the computations. It can be Lower - * or Upper. Default is Lower. - * - * \implsparsesolverconcept - * - * This class supports all kind of SparseMatrix<>: row or column major; upper, lower, or both; compressed or non compressed. - * - * \warning Only double precision real and complex scalar types are supported by Cholmod. - * - * \sa \ref TutorialSparseSolverConcept, class CholmodSupernodalLLT, class SimplicialLDLT - */ + * \class CholmodSimplicialLDLT + * \brief A simplicial direct Cholesky (LDLT) factorization and solver based on Cholmod + * + * This class allows to solve for A.X = B sparse linear problems via a simplicial LDL^T Cholesky factorization + * using the Cholmod library. + * This simplicial variant is equivalent to Eigen's built-in SimplicialLDLT class. Therefore, it has little practical + * interest. The sparse matrix A must be selfadjoint and positive definite. The vectors or matrices X and B can be + * either dense or sparse. + * + * \tparam _MatrixType the type of the sparse matrix A, it must be a SparseMatrix<> + * \tparam _UpLo the triangular part that will be used for the computations. It can be Lower + * or Upper. Default is Lower. + * + * \implsparsesolverconcept + * + * This class supports all kind of SparseMatrix<>: row or column major; upper, lower, or both; compressed or non + * compressed. + * + * \warning Only double precision real and complex scalar types are supported by Cholmod. + * + * \sa \ref TutorialSparseSolverConcept, class CholmodSupernodalLLT, class SimplicialLDLT + */ template -class CholmodSimplicialLDLT : public CholmodBase<_MatrixType, _UpLo, CholmodSimplicialLDLT<_MatrixType, _UpLo> > +class CholmodSimplicialLDLT : public CholmodBase<_MatrixType, _UpLo, CholmodSimplicialLDLT<_MatrixType, _UpLo>> { - typedef CholmodBase<_MatrixType, _UpLo, CholmodSimplicialLDLT> Base; - using Base::m_cholmod; - - public: - - typedef _MatrixType MatrixType; - - CholmodSimplicialLDLT() : Base() { init(); } - - CholmodSimplicialLDLT(const MatrixType& matrix) : Base() - { - init(); - this->compute(matrix); - } + typedef CholmodBase<_MatrixType, _UpLo, CholmodSimplicialLDLT> Base; + using Base::m_cholmod; - ~CholmodSimplicialLDLT() {} - protected: - void init() - { - m_cholmod.final_asis = 1; - m_cholmod.supernodal = CHOLMOD_SIMPLICIAL; - } +public: + typedef _MatrixType MatrixType; + + CholmodSimplicialLDLT() : Base() { init(); } + + CholmodSimplicialLDLT(const MatrixType &matrix) : Base() + { + init(); + this->compute(matrix); + } + + ~CholmodSimplicialLDLT() {} + +protected: + void init() + { + m_cholmod.final_asis = 1; + m_cholmod.supernodal = CHOLMOD_SIMPLICIAL; + } }; /** \ingroup CholmodSupport_Module - * \class CholmodSupernodalLLT - * \brief A supernodal Cholesky (LLT) factorization and solver based on Cholmod - * - * This class allows to solve for A.X = B sparse linear problems via a supernodal LL^T Cholesky factorization - * using the Cholmod library. - * This supernodal variant performs best on dense enough problems, e.g., 3D FEM, or very high order 2D FEM. - * The sparse matrix A must be selfadjoint and positive definite. The vectors or matrices - * X and B can be either dense or sparse. - * - * \tparam _MatrixType the type of the sparse matrix A, it must be a SparseMatrix<> - * \tparam _UpLo the triangular part that will be used for the computations. It can be Lower - * or Upper. Default is Lower. - * - * \implsparsesolverconcept - * - * This class supports all kind of SparseMatrix<>: row or column major; upper, lower, or both; compressed or non compressed. - * - * \warning Only double precision real and complex scalar types are supported by Cholmod. - * - * \sa \ref TutorialSparseSolverConcept - */ + * \class CholmodSupernodalLLT + * \brief A supernodal Cholesky (LLT) factorization and solver based on Cholmod + * + * This class allows to solve for A.X = B sparse linear problems via a supernodal LL^T Cholesky factorization + * using the Cholmod library. + * This supernodal variant performs best on dense enough problems, e.g., 3D FEM, or very high order 2D FEM. + * The sparse matrix A must be selfadjoint and positive definite. The vectors or matrices + * X and B can be either dense or sparse. + * + * \tparam _MatrixType the type of the sparse matrix A, it must be a SparseMatrix<> + * \tparam _UpLo the triangular part that will be used for the computations. It can be Lower + * or Upper. Default is Lower. + * + * \implsparsesolverconcept + * + * This class supports all kind of SparseMatrix<>: row or column major; upper, lower, or both; compressed or non + * compressed. + * + * \warning Only double precision real and complex scalar types are supported by Cholmod. + * + * \sa \ref TutorialSparseSolverConcept + */ template -class CholmodSupernodalLLT : public CholmodBase<_MatrixType, _UpLo, CholmodSupernodalLLT<_MatrixType, _UpLo> > +class CholmodSupernodalLLT : public CholmodBase<_MatrixType, _UpLo, CholmodSupernodalLLT<_MatrixType, _UpLo>> { - typedef CholmodBase<_MatrixType, _UpLo, CholmodSupernodalLLT> Base; - using Base::m_cholmod; - - public: - - typedef _MatrixType MatrixType; - - CholmodSupernodalLLT() : Base() { init(); } - - CholmodSupernodalLLT(const MatrixType& matrix) : Base() - { - init(); - this->compute(matrix); - } + typedef CholmodBase<_MatrixType, _UpLo, CholmodSupernodalLLT> Base; + using Base::m_cholmod; - ~CholmodSupernodalLLT() {} - protected: - void init() - { - m_cholmod.final_asis = 1; - m_cholmod.supernodal = CHOLMOD_SUPERNODAL; - } +public: + typedef _MatrixType MatrixType; + + CholmodSupernodalLLT() : Base() { init(); } + + CholmodSupernodalLLT(const MatrixType &matrix) : Base() + { + init(); + this->compute(matrix); + } + + ~CholmodSupernodalLLT() {} + +protected: + void init() + { + m_cholmod.final_asis = 1; + m_cholmod.supernodal = CHOLMOD_SUPERNODAL; + } }; /** \ingroup CholmodSupport_Module - * \class CholmodDecomposition - * \brief A general Cholesky factorization and solver based on Cholmod - * - * This class allows to solve for A.X = B sparse linear problems via a LL^T or LDL^T Cholesky factorization - * using the Cholmod library. The sparse matrix A must be selfadjoint and positive definite. The vectors or matrices - * X and B can be either dense or sparse. - * - * This variant permits to change the underlying Cholesky method at runtime. - * On the other hand, it does not provide access to the result of the factorization. - * The default is to let Cholmod automatically choose between a simplicial and supernodal factorization. - * - * \tparam _MatrixType the type of the sparse matrix A, it must be a SparseMatrix<> - * \tparam _UpLo the triangular part that will be used for the computations. It can be Lower - * or Upper. Default is Lower. - * - * \implsparsesolverconcept - * - * This class supports all kind of SparseMatrix<>: row or column major; upper, lower, or both; compressed or non compressed. - * - * \warning Only double precision real and complex scalar types are supported by Cholmod. - * - * \sa \ref TutorialSparseSolverConcept - */ + * \class CholmodDecomposition + * \brief A general Cholesky factorization and solver based on Cholmod + * + * This class allows to solve for A.X = B sparse linear problems via a LL^T or LDL^T Cholesky factorization + * using the Cholmod library. The sparse matrix A must be selfadjoint and positive definite. The vectors or matrices + * X and B can be either dense or sparse. + * + * This variant permits to change the underlying Cholesky method at runtime. + * On the other hand, it does not provide access to the result of the factorization. + * The default is to let Cholmod automatically choose between a simplicial and supernodal factorization. + * + * \tparam _MatrixType the type of the sparse matrix A, it must be a SparseMatrix<> + * \tparam _UpLo the triangular part that will be used for the computations. It can be Lower + * or Upper. Default is Lower. + * + * \implsparsesolverconcept + * + * This class supports all kind of SparseMatrix<>: row or column major; upper, lower, or both; compressed or non + * compressed. + * + * \warning Only double precision real and complex scalar types are supported by Cholmod. + * + * \sa \ref TutorialSparseSolverConcept + */ template -class CholmodDecomposition : public CholmodBase<_MatrixType, _UpLo, CholmodDecomposition<_MatrixType, _UpLo> > +class CholmodDecomposition : public CholmodBase<_MatrixType, _UpLo, CholmodDecomposition<_MatrixType, _UpLo>> { - typedef CholmodBase<_MatrixType, _UpLo, CholmodDecomposition> Base; - using Base::m_cholmod; - - public: - - typedef _MatrixType MatrixType; - - CholmodDecomposition() : Base() { init(); } - - CholmodDecomposition(const MatrixType& matrix) : Base() - { - init(); - this->compute(matrix); - } + typedef CholmodBase<_MatrixType, _UpLo, CholmodDecomposition> Base; + using Base::m_cholmod; - ~CholmodDecomposition() {} - - void setMode(CholmodMode mode) - { - switch(mode) - { - case CholmodAuto: - m_cholmod.final_asis = 1; - m_cholmod.supernodal = CHOLMOD_AUTO; - break; - case CholmodSimplicialLLt: - m_cholmod.final_asis = 0; - m_cholmod.supernodal = CHOLMOD_SIMPLICIAL; - m_cholmod.final_ll = 1; - break; - case CholmodSupernodalLLt: - m_cholmod.final_asis = 1; - m_cholmod.supernodal = CHOLMOD_SUPERNODAL; - break; - case CholmodLDLt: - m_cholmod.final_asis = 1; - m_cholmod.supernodal = CHOLMOD_SIMPLICIAL; - break; - default: - break; - } - } - protected: - void init() - { +public: + typedef _MatrixType MatrixType; + + CholmodDecomposition() : Base() { init(); } + + CholmodDecomposition(const MatrixType &matrix) : Base() + { + init(); + this->compute(matrix); + } + + ~CholmodDecomposition() {} + + void setMode(CholmodMode mode) + { + switch (mode) { + case CholmodAuto: m_cholmod.final_asis = 1; m_cholmod.supernodal = CHOLMOD_AUTO; + break; + case CholmodSimplicialLLt: + m_cholmod.final_asis = 0; + m_cholmod.supernodal = CHOLMOD_SIMPLICIAL; + m_cholmod.final_ll = 1; + break; + case CholmodSupernodalLLt: + m_cholmod.final_asis = 1; + m_cholmod.supernodal = CHOLMOD_SUPERNODAL; + break; + case CholmodLDLt: + m_cholmod.final_asis = 1; + m_cholmod.supernodal = CHOLMOD_SIMPLICIAL; + break; + default: + break; } + } + +protected: + void init() + { + m_cholmod.final_asis = 1; + m_cholmod.supernodal = CHOLMOD_AUTO; + } }; -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_CHOLMODSUPPORT_H +#endif// EIGEN_CHOLMODSUPPORT_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/Array.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/Array.h index 16770fc7..b7ef18f1 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/Array.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/Array.h @@ -13,292 +13,278 @@ namespace Eigen { namespace internal { -template -struct traits > : traits > -{ - typedef ArrayXpr XprKind; - typedef ArrayBase > XprBase; -}; -} + template + struct traits> + : traits> + { + typedef ArrayXpr XprKind; + typedef ArrayBase> XprBase; + }; +}// namespace internal /** \class Array - * \ingroup Core_Module - * - * \brief General-purpose arrays with easy API for coefficient-wise operations - * - * The %Array class is very similar to the Matrix class. It provides - * general-purpose one- and two-dimensional arrays. The difference between the - * %Array and the %Matrix class is primarily in the API: the API for the - * %Array class provides easy access to coefficient-wise operations, while the - * API for the %Matrix class provides easy access to linear-algebra - * operations. - * - * See documentation of class Matrix for detailed information on the template parameters - * storage layout. - * - * This class can be extended with the help of the plugin mechanism described on the page - * \ref TopicCustomizing_Plugins by defining the preprocessor symbol \c EIGEN_ARRAY_PLUGIN. - * - * \sa \blank \ref TutorialArrayClass, \ref TopicClassHierarchy - */ + * \ingroup Core_Module + * + * \brief General-purpose arrays with easy API for coefficient-wise operations + * + * The %Array class is very similar to the Matrix class. It provides + * general-purpose one- and two-dimensional arrays. The difference between the + * %Array and the %Matrix class is primarily in the API: the API for the + * %Array class provides easy access to coefficient-wise operations, while the + * API for the %Matrix class provides easy access to linear-algebra + * operations. + * + * See documentation of class Matrix for detailed information on the template parameters + * storage layout. + * + * This class can be extended with the help of the plugin mechanism described on the page + * \ref TopicCustomizing_Plugins by defining the preprocessor symbol \c EIGEN_ARRAY_PLUGIN. + * + * \sa \blank \ref TutorialArrayClass, \ref TopicClassHierarchy + */ template -class Array - : public PlainObjectBase > +class Array : public PlainObjectBase> { - public: - - typedef PlainObjectBase Base; - EIGEN_DENSE_PUBLIC_INTERFACE(Array) - - enum { Options = _Options }; - typedef typename Base::PlainObject PlainObject; - - protected: - template - friend struct internal::conservative_resize_like_impl; - - using Base::m_storage; - - public: - - using Base::base; - using Base::coeff; - using Base::coeffRef; - - /** - * The usage of - * using Base::operator=; - * fails on MSVC. Since the code below is working with GCC and MSVC, we skipped - * the usage of 'using'. This should be done only for operator=. - */ - template - EIGEN_DEVICE_FUNC - EIGEN_STRONG_INLINE Array& operator=(const EigenBase &other) - { - return Base::operator=(other); - } - - /** Set all the entries to \a value. - * \sa DenseBase::setConstant(), DenseBase::fill() - */ - /* This overload is needed because the usage of - * using Base::operator=; - * fails on MSVC. Since the code below is working with GCC and MSVC, we skipped - * the usage of 'using'. This should be done only for operator=. - */ - EIGEN_DEVICE_FUNC - EIGEN_STRONG_INLINE Array& operator=(const Scalar &value) - { - Base::setConstant(value); - return *this; - } - - /** Copies the value of the expression \a other into \c *this with automatic resizing. - * - * *this might be resized to match the dimensions of \a other. If *this was a null matrix (not already initialized), - * it will be initialized. - * - * Note that copying a row-vector into a vector (and conversely) is allowed. - * The resizing, if any, is then done in the appropriate way so that row-vectors - * remain row-vectors and vectors remain vectors. - */ - template - EIGEN_DEVICE_FUNC - EIGEN_STRONG_INLINE Array& operator=(const DenseBase& other) - { - return Base::_set(other); - } - - /** This is a special case of the templated operator=. Its purpose is to - * prevent a default operator= from hiding the templated operator=. - */ - EIGEN_DEVICE_FUNC - EIGEN_STRONG_INLINE Array& operator=(const Array& other) - { - return Base::_set(other); - } - - /** Default constructor. - * - * For fixed-size matrices, does nothing. - * - * For dynamic-size matrices, creates an empty matrix of size 0. Does not allocate any array. Such a matrix - * is called a null matrix. This constructor is the unique way to create null matrices: resizing - * a matrix to 0 is not supported. - * - * \sa resize(Index,Index) - */ - EIGEN_DEVICE_FUNC - EIGEN_STRONG_INLINE Array() : Base() - { - Base::_check_template_params(); - EIGEN_INITIALIZE_COEFFS_IF_THAT_OPTION_IS_ENABLED - } +public: + typedef PlainObjectBase Base; + EIGEN_DENSE_PUBLIC_INTERFACE(Array) + + enum { Options = _Options }; + typedef typename Base::PlainObject PlainObject; + +protected: + template + friend struct internal::conservative_resize_like_impl; + + using Base::m_storage; + +public: + using Base::base; + using Base::coeff; + using Base::coeffRef; + + /** + * The usage of + * using Base::operator=; + * fails on MSVC. Since the code below is working with GCC and MSVC, we skipped + * the usage of 'using'. This should be done only for operator=. + */ + template + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Array &operator=(const EigenBase &other) + { + return Base::operator=(other); + } + + /** Set all the entries to \a value. + * \sa DenseBase::setConstant(), DenseBase::fill() + */ + /* This overload is needed because the usage of + * using Base::operator=; + * fails on MSVC. Since the code below is working with GCC and MSVC, we skipped + * the usage of 'using'. This should be done only for operator=. + */ + EIGEN_DEVICE_FUNC + EIGEN_STRONG_INLINE Array &operator=(const Scalar &value) + { + Base::setConstant(value); + return *this; + } + + /** Copies the value of the expression \a other into \c *this with automatic resizing. + * + * *this might be resized to match the dimensions of \a other. If *this was a null matrix (not already initialized), + * it will be initialized. + * + * Note that copying a row-vector into a vector (and conversely) is allowed. + * The resizing, if any, is then done in the appropriate way so that row-vectors + * remain row-vectors and vectors remain vectors. + */ + template + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Array &operator=(const DenseBase &other) + { + return Base::_set(other); + } + + /** This is a special case of the templated operator=. Its purpose is to + * prevent a default operator= from hiding the templated operator=. + */ + EIGEN_DEVICE_FUNC + EIGEN_STRONG_INLINE Array &operator=(const Array &other) { return Base::_set(other); } + + /** Default constructor. + * + * For fixed-size matrices, does nothing. + * + * For dynamic-size matrices, creates an empty matrix of size 0. Does not allocate any array. Such a matrix + * is called a null matrix. This constructor is the unique way to create null matrices: resizing + * a matrix to 0 is not supported. + * + * \sa resize(Index,Index) + */ + EIGEN_DEVICE_FUNC + EIGEN_STRONG_INLINE Array() : Base() + { + Base::_check_template_params(); + EIGEN_INITIALIZE_COEFFS_IF_THAT_OPTION_IS_ENABLED + } #ifndef EIGEN_PARSED_BY_DOXYGEN - // FIXME is it still needed ?? - /** \internal */ - EIGEN_DEVICE_FUNC - Array(internal::constructor_without_unaligned_array_assert) - : Base(internal::constructor_without_unaligned_array_assert()) - { - Base::_check_template_params(); - EIGEN_INITIALIZE_COEFFS_IF_THAT_OPTION_IS_ENABLED - } + // FIXME is it still needed ?? + /** \internal */ + EIGEN_DEVICE_FUNC + Array(internal::constructor_without_unaligned_array_assert) + : Base(internal::constructor_without_unaligned_array_assert()) + { + Base::_check_template_params(); + EIGEN_INITIALIZE_COEFFS_IF_THAT_OPTION_IS_ENABLED + } #endif #if EIGEN_HAS_RVALUE_REFERENCES - EIGEN_DEVICE_FUNC - Array(Array&& other) EIGEN_NOEXCEPT_IF(std::is_nothrow_move_constructible::value) - : Base(std::move(other)) - { - Base::_check_template_params(); - } - EIGEN_DEVICE_FUNC - Array& operator=(Array&& other) EIGEN_NOEXCEPT_IF(std::is_nothrow_move_assignable::value) - { - other.swap(*this); - return *this; - } + EIGEN_DEVICE_FUNC + Array(Array &&other) EIGEN_NOEXCEPT_IF(std::is_nothrow_move_constructible::value) : Base(std::move(other)) + { + Base::_check_template_params(); + } + EIGEN_DEVICE_FUNC + Array &operator=(Array &&other) EIGEN_NOEXCEPT_IF(std::is_nothrow_move_assignable::value) + { + other.swap(*this); + return *this; + } +#endif + +#ifndef EIGEN_PARSED_BY_DOXYGEN + template EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE explicit Array(const T &x) + { + Base::_check_template_params(); + Base::template _init1(x); + } + + template EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Array(const T0 &val0, const T1 &val1) + { + Base::_check_template_params(); + this->template _init2(val0, val1); + } +#else + /** \brief Constructs a fixed-sized array initialized with coefficients starting at \a data */ + EIGEN_DEVICE_FUNC explicit Array(const Scalar *data); + /** Constructs a vector or row-vector with given dimension. \only_for_vectors + * + * Note that this is only useful for dynamic-size vectors. For fixed-size vectors, + * it is redundant to pass the dimension here, so it makes more sense to use the default + * constructor Array() instead. + */ + EIGEN_DEVICE_FUNC + EIGEN_STRONG_INLINE explicit Array(Index dim); + /** constructs an initialized 1x1 Array with the given coefficient */ + Array(const Scalar &value); + /** constructs an uninitialized array with \a rows rows and \a cols columns. + * + * This is useful for dynamic-size arrays. For fixed-size arrays, + * it is redundant to pass these parameters, so one should use the default constructor + * Array() instead. */ + Array(Index rows, Index cols); + /** constructs an initialized 2D vector with given coefficients */ + Array(const Scalar &val0, const Scalar &val1); +#endif + + /** constructs an initialized 3D vector with given coefficients */ + EIGEN_DEVICE_FUNC + EIGEN_STRONG_INLINE Array(const Scalar &val0, const Scalar &val1, const Scalar &val2) + { + Base::_check_template_params(); + EIGEN_STATIC_ASSERT_VECTOR_SPECIFIC_SIZE(Array, 3) + m_storage.data()[0] = val0; + m_storage.data()[1] = val1; + m_storage.data()[2] = val2; + } + /** constructs an initialized 4D vector with given coefficients */ + EIGEN_DEVICE_FUNC + EIGEN_STRONG_INLINE Array(const Scalar &val0, const Scalar &val1, const Scalar &val2, const Scalar &val3) + { + Base::_check_template_params(); + EIGEN_STATIC_ASSERT_VECTOR_SPECIFIC_SIZE(Array, 4) + m_storage.data()[0] = val0; + m_storage.data()[1] = val1; + m_storage.data()[2] = val2; + m_storage.data()[3] = val3; + } + + /** Copy constructor */ + EIGEN_DEVICE_FUNC + EIGEN_STRONG_INLINE Array(const Array &other) : Base(other) {} + +private: + struct PrivateType + { + }; + +public: + /** \sa MatrixBase::operator=(const EigenBase&) */ + template + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Array(const EigenBase &other, + typename internal::enable_if::value, + PrivateType>::type = PrivateType()) + : Base(other.derived()) + {} + + EIGEN_DEVICE_FUNC inline Index innerStride() const { return 1; } + EIGEN_DEVICE_FUNC inline Index outerStride() const { return this->innerSize(); } + +#ifdef EIGEN_ARRAY_PLUGIN +#include EIGEN_ARRAY_PLUGIN #endif - #ifndef EIGEN_PARSED_BY_DOXYGEN - template - EIGEN_DEVICE_FUNC - EIGEN_STRONG_INLINE explicit Array(const T& x) - { - Base::_check_template_params(); - Base::template _init1(x); - } - - template - EIGEN_DEVICE_FUNC - EIGEN_STRONG_INLINE Array(const T0& val0, const T1& val1) - { - Base::_check_template_params(); - this->template _init2(val0, val1); - } - #else - /** \brief Constructs a fixed-sized array initialized with coefficients starting at \a data */ - EIGEN_DEVICE_FUNC explicit Array(const Scalar *data); - /** Constructs a vector or row-vector with given dimension. \only_for_vectors - * - * Note that this is only useful for dynamic-size vectors. For fixed-size vectors, - * it is redundant to pass the dimension here, so it makes more sense to use the default - * constructor Array() instead. - */ - EIGEN_DEVICE_FUNC - EIGEN_STRONG_INLINE explicit Array(Index dim); - /** constructs an initialized 1x1 Array with the given coefficient */ - Array(const Scalar& value); - /** constructs an uninitialized array with \a rows rows and \a cols columns. - * - * This is useful for dynamic-size arrays. For fixed-size arrays, - * it is redundant to pass these parameters, so one should use the default constructor - * Array() instead. */ - Array(Index rows, Index cols); - /** constructs an initialized 2D vector with given coefficients */ - Array(const Scalar& val0, const Scalar& val1); - #endif - - /** constructs an initialized 3D vector with given coefficients */ - EIGEN_DEVICE_FUNC - EIGEN_STRONG_INLINE Array(const Scalar& val0, const Scalar& val1, const Scalar& val2) - { - Base::_check_template_params(); - EIGEN_STATIC_ASSERT_VECTOR_SPECIFIC_SIZE(Array, 3) - m_storage.data()[0] = val0; - m_storage.data()[1] = val1; - m_storage.data()[2] = val2; - } - /** constructs an initialized 4D vector with given coefficients */ - EIGEN_DEVICE_FUNC - EIGEN_STRONG_INLINE Array(const Scalar& val0, const Scalar& val1, const Scalar& val2, const Scalar& val3) - { - Base::_check_template_params(); - EIGEN_STATIC_ASSERT_VECTOR_SPECIFIC_SIZE(Array, 4) - m_storage.data()[0] = val0; - m_storage.data()[1] = val1; - m_storage.data()[2] = val2; - m_storage.data()[3] = val3; - } - - /** Copy constructor */ - EIGEN_DEVICE_FUNC - EIGEN_STRONG_INLINE Array(const Array& other) - : Base(other) - { } - - private: - struct PrivateType {}; - public: - - /** \sa MatrixBase::operator=(const EigenBase&) */ - template - EIGEN_DEVICE_FUNC - EIGEN_STRONG_INLINE Array(const EigenBase &other, - typename internal::enable_if::value, - PrivateType>::type = PrivateType()) - : Base(other.derived()) - { } - - EIGEN_DEVICE_FUNC inline Index innerStride() const { return 1; } - EIGEN_DEVICE_FUNC inline Index outerStride() const { return this->innerSize(); } - - #ifdef EIGEN_ARRAY_PLUGIN - #include EIGEN_ARRAY_PLUGIN - #endif - - private: - - template - friend struct internal::matrix_swap_impl; +private: + template friend struct internal::matrix_swap_impl; }; /** \defgroup arraytypedefs Global array typedefs - * \ingroup Core_Module - * - * Eigen defines several typedef shortcuts for most common 1D and 2D array types. - * - * The general patterns are the following: - * - * \c ArrayRowsColsType where \c Rows and \c Cols can be \c 2,\c 3,\c 4 for fixed size square matrices or \c X for dynamic size, - * and where \c Type can be \c i for integer, \c f for float, \c d for double, \c cf for complex float, \c cd - * for complex double. - * - * For example, \c Array33d is a fixed-size 3x3 array type of doubles, and \c ArrayXXf is a dynamic-size matrix of floats. - * - * There are also \c ArraySizeType which are self-explanatory. For example, \c Array4cf is - * a fixed-size 1D array of 4 complex floats. - * - * \sa class Array - */ - -#define EIGEN_MAKE_ARRAY_TYPEDEFS(Type, TypeSuffix, Size, SizeSuffix) \ -/** \ingroup arraytypedefs */ \ -typedef Array Array##SizeSuffix##SizeSuffix##TypeSuffix; \ -/** \ingroup arraytypedefs */ \ -typedef Array Array##SizeSuffix##TypeSuffix; - -#define EIGEN_MAKE_ARRAY_FIXED_TYPEDEFS(Type, TypeSuffix, Size) \ -/** \ingroup arraytypedefs */ \ -typedef Array Array##Size##X##TypeSuffix; \ -/** \ingroup arraytypedefs */ \ -typedef Array Array##X##Size##TypeSuffix; + * \ingroup Core_Module + * + * Eigen defines several typedef shortcuts for most common 1D and 2D array types. + * + * The general patterns are the following: + * + * \c ArrayRowsColsType where \c Rows and \c Cols can be \c 2,\c 3,\c 4 for fixed size square matrices or \c X for + * dynamic size, and where \c Type can be \c i for integer, \c f for float, \c d for double, \c cf for complex float, \c + * cd for complex double. + * + * For example, \c Array33d is a fixed-size 3x3 array type of doubles, and \c ArrayXXf is a dynamic-size matrix of + * floats. + * + * There are also \c ArraySizeType which are self-explanatory. For example, \c Array4cf is + * a fixed-size 1D array of 4 complex floats. + * + * \sa class Array + */ + +#define EIGEN_MAKE_ARRAY_TYPEDEFS(Type, TypeSuffix, Size, SizeSuffix) \ + /** \ingroup arraytypedefs */ \ + typedef Array Array##SizeSuffix##SizeSuffix##TypeSuffix; \ + /** \ingroup arraytypedefs */ \ + typedef Array Array##SizeSuffix##TypeSuffix; + +#define EIGEN_MAKE_ARRAY_FIXED_TYPEDEFS(Type, TypeSuffix, Size) \ + /** \ingroup arraytypedefs */ \ + typedef Array Array##Size##X##TypeSuffix; \ + /** \ingroup arraytypedefs */ \ + typedef Array Array##X##Size##TypeSuffix; #define EIGEN_MAKE_ARRAY_TYPEDEFS_ALL_SIZES(Type, TypeSuffix) \ -EIGEN_MAKE_ARRAY_TYPEDEFS(Type, TypeSuffix, 2, 2) \ -EIGEN_MAKE_ARRAY_TYPEDEFS(Type, TypeSuffix, 3, 3) \ -EIGEN_MAKE_ARRAY_TYPEDEFS(Type, TypeSuffix, 4, 4) \ -EIGEN_MAKE_ARRAY_TYPEDEFS(Type, TypeSuffix, Dynamic, X) \ -EIGEN_MAKE_ARRAY_FIXED_TYPEDEFS(Type, TypeSuffix, 2) \ -EIGEN_MAKE_ARRAY_FIXED_TYPEDEFS(Type, TypeSuffix, 3) \ -EIGEN_MAKE_ARRAY_FIXED_TYPEDEFS(Type, TypeSuffix, 4) - -EIGEN_MAKE_ARRAY_TYPEDEFS_ALL_SIZES(int, i) -EIGEN_MAKE_ARRAY_TYPEDEFS_ALL_SIZES(float, f) -EIGEN_MAKE_ARRAY_TYPEDEFS_ALL_SIZES(double, d) -EIGEN_MAKE_ARRAY_TYPEDEFS_ALL_SIZES(std::complex, cf) + EIGEN_MAKE_ARRAY_TYPEDEFS(Type, TypeSuffix, 2, 2) \ + EIGEN_MAKE_ARRAY_TYPEDEFS(Type, TypeSuffix, 3, 3) \ + EIGEN_MAKE_ARRAY_TYPEDEFS(Type, TypeSuffix, 4, 4) \ + EIGEN_MAKE_ARRAY_TYPEDEFS(Type, TypeSuffix, Dynamic, X) \ + EIGEN_MAKE_ARRAY_FIXED_TYPEDEFS(Type, TypeSuffix, 2) \ + EIGEN_MAKE_ARRAY_FIXED_TYPEDEFS(Type, TypeSuffix, 3) \ + EIGEN_MAKE_ARRAY_FIXED_TYPEDEFS(Type, TypeSuffix, 4) + +EIGEN_MAKE_ARRAY_TYPEDEFS_ALL_SIZES(int, i) +EIGEN_MAKE_ARRAY_TYPEDEFS_ALL_SIZES(float, f) +EIGEN_MAKE_ARRAY_TYPEDEFS_ALL_SIZES(double, d) +EIGEN_MAKE_ARRAY_TYPEDEFS_ALL_SIZES(std::complex, cf) EIGEN_MAKE_ARRAY_TYPEDEFS_ALL_SIZES(std::complex, cd) #undef EIGEN_MAKE_ARRAY_TYPEDEFS_ALL_SIZES @@ -307,23 +293,23 @@ EIGEN_MAKE_ARRAY_TYPEDEFS_ALL_SIZES(std::complex, cd) #undef EIGEN_MAKE_ARRAY_TYPEDEFS_LARGE #define EIGEN_USING_ARRAY_TYPEDEFS_FOR_TYPE_AND_SIZE(TypeSuffix, SizeSuffix) \ -using Eigen::Matrix##SizeSuffix##TypeSuffix; \ -using Eigen::Vector##SizeSuffix##TypeSuffix; \ -using Eigen::RowVector##SizeSuffix##TypeSuffix; - -#define EIGEN_USING_ARRAY_TYPEDEFS_FOR_TYPE(TypeSuffix) \ -EIGEN_USING_ARRAY_TYPEDEFS_FOR_TYPE_AND_SIZE(TypeSuffix, 2) \ -EIGEN_USING_ARRAY_TYPEDEFS_FOR_TYPE_AND_SIZE(TypeSuffix, 3) \ -EIGEN_USING_ARRAY_TYPEDEFS_FOR_TYPE_AND_SIZE(TypeSuffix, 4) \ -EIGEN_USING_ARRAY_TYPEDEFS_FOR_TYPE_AND_SIZE(TypeSuffix, X) \ - -#define EIGEN_USING_ARRAY_TYPEDEFS \ -EIGEN_USING_ARRAY_TYPEDEFS_FOR_TYPE(i) \ -EIGEN_USING_ARRAY_TYPEDEFS_FOR_TYPE(f) \ -EIGEN_USING_ARRAY_TYPEDEFS_FOR_TYPE(d) \ -EIGEN_USING_ARRAY_TYPEDEFS_FOR_TYPE(cf) \ -EIGEN_USING_ARRAY_TYPEDEFS_FOR_TYPE(cd) - -} // end namespace Eigen - -#endif // EIGEN_ARRAY_H + using Eigen::Matrix##SizeSuffix##TypeSuffix; \ + using Eigen::Vector##SizeSuffix##TypeSuffix; \ + using Eigen::RowVector##SizeSuffix##TypeSuffix; + +#define EIGEN_USING_ARRAY_TYPEDEFS_FOR_TYPE(TypeSuffix) \ + EIGEN_USING_ARRAY_TYPEDEFS_FOR_TYPE_AND_SIZE(TypeSuffix, 2) \ + EIGEN_USING_ARRAY_TYPEDEFS_FOR_TYPE_AND_SIZE(TypeSuffix, 3) \ + EIGEN_USING_ARRAY_TYPEDEFS_FOR_TYPE_AND_SIZE(TypeSuffix, 4) \ + EIGEN_USING_ARRAY_TYPEDEFS_FOR_TYPE_AND_SIZE(TypeSuffix, X) + +#define EIGEN_USING_ARRAY_TYPEDEFS \ + EIGEN_USING_ARRAY_TYPEDEFS_FOR_TYPE(i) \ + EIGEN_USING_ARRAY_TYPEDEFS_FOR_TYPE(f) \ + EIGEN_USING_ARRAY_TYPEDEFS_FOR_TYPE(d) \ + EIGEN_USING_ARRAY_TYPEDEFS_FOR_TYPE(cf) \ + EIGEN_USING_ARRAY_TYPEDEFS_FOR_TYPE(cd) + +}// end namespace Eigen + +#endif// EIGEN_ARRAY_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/ArrayBase.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/ArrayBase.h index 3dbc7084..43c1baea 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/ArrayBase.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/ArrayBase.h @@ -10,217 +10,216 @@ #ifndef EIGEN_ARRAYBASE_H #define EIGEN_ARRAYBASE_H -namespace Eigen { +namespace Eigen { template class MatrixWrapper; /** \class ArrayBase - * \ingroup Core_Module - * - * \brief Base class for all 1D and 2D array, and related expressions - * - * An array is similar to a dense vector or matrix. While matrices are mathematical - * objects with well defined linear algebra operators, an array is just a collection - * of scalar values arranged in a one or two dimensionnal fashion. As the main consequence, - * all operations applied to an array are performed coefficient wise. Furthermore, - * arrays support scalar math functions of the c++ standard library (e.g., std::sin(x)), and convenient - * constructors allowing to easily write generic code working for both scalar values - * and arrays. - * - * This class is the base that is inherited by all array expression types. - * - * \tparam Derived is the derived type, e.g., an array or an expression type. - * - * This class can be extended with the help of the plugin mechanism described on the page - * \ref TopicCustomizing_Plugins by defining the preprocessor symbol \c EIGEN_ARRAYBASE_PLUGIN. - * - * \sa class MatrixBase, \ref TopicClassHierarchy - */ -template class ArrayBase - : public DenseBase + * \ingroup Core_Module + * + * \brief Base class for all 1D and 2D array, and related expressions + * + * An array is similar to a dense vector or matrix. While matrices are mathematical + * objects with well defined linear algebra operators, an array is just a collection + * of scalar values arranged in a one or two dimensionnal fashion. As the main consequence, + * all operations applied to an array are performed coefficient wise. Furthermore, + * arrays support scalar math functions of the c++ standard library (e.g., std::sin(x)), and convenient + * constructors allowing to easily write generic code working for both scalar values + * and arrays. + * + * This class is the base that is inherited by all array expression types. + * + * \tparam Derived is the derived type, e.g., an array or an expression type. + * + * This class can be extended with the help of the plugin mechanism described on the page + * \ref TopicCustomizing_Plugins by defining the preprocessor symbol \c EIGEN_ARRAYBASE_PLUGIN. + * + * \sa class MatrixBase, \ref TopicClassHierarchy + */ +template class ArrayBase : public DenseBase { - public: +public: #ifndef EIGEN_PARSED_BY_DOXYGEN - /** The base class for a given storage type. */ - typedef ArrayBase StorageBaseType; - - typedef ArrayBase Eigen_BaseClassForSpecializationOfGlobalMathFuncImpl; - - typedef typename internal::traits::StorageKind StorageKind; - typedef typename internal::traits::Scalar Scalar; - typedef typename internal::packet_traits::type PacketScalar; - typedef typename NumTraits::Real RealScalar; - - typedef DenseBase Base; - using Base::RowsAtCompileTime; - using Base::ColsAtCompileTime; - using Base::SizeAtCompileTime; - using Base::MaxRowsAtCompileTime; - using Base::MaxColsAtCompileTime; - using Base::MaxSizeAtCompileTime; - using Base::IsVectorAtCompileTime; - using Base::Flags; - - using Base::derived; - using Base::const_cast_derived; - using Base::rows; - using Base::cols; - using Base::size; - using Base::coeff; - using Base::coeffRef; - using Base::lazyAssign; - using Base::operator=; - using Base::operator+=; - using Base::operator-=; - using Base::operator*=; - using Base::operator/=; - - typedef typename Base::CoeffReturnType CoeffReturnType; - -#endif // not EIGEN_PARSED_BY_DOXYGEN + /** The base class for a given storage type. */ + typedef ArrayBase StorageBaseType; + + typedef ArrayBase Eigen_BaseClassForSpecializationOfGlobalMathFuncImpl; + + typedef typename internal::traits::StorageKind StorageKind; + typedef typename internal::traits::Scalar Scalar; + typedef typename internal::packet_traits::type PacketScalar; + typedef typename NumTraits::Real RealScalar; + + typedef DenseBase Base; + using Base::RowsAtCompileTime; + using Base::ColsAtCompileTime; + using Base::SizeAtCompileTime; + using Base::MaxRowsAtCompileTime; + using Base::MaxColsAtCompileTime; + using Base::MaxSizeAtCompileTime; + using Base::IsVectorAtCompileTime; + using Base::Flags; + + using Base::derived; + using Base::const_cast_derived; + using Base::rows; + using Base::cols; + using Base::size; + using Base::coeff; + using Base::coeffRef; + using Base::lazyAssign; + using Base::operator=; + using Base::operator+=; + using Base::operator-=; + using Base::operator*=; + using Base::operator/=; + + typedef typename Base::CoeffReturnType CoeffReturnType; + +#endif// not EIGEN_PARSED_BY_DOXYGEN #ifndef EIGEN_PARSED_BY_DOXYGEN - typedef typename Base::PlainObject PlainObject; + typedef typename Base::PlainObject PlainObject; - /** \internal Represents a matrix with all coefficients equal to one another*/ - typedef CwiseNullaryOp,PlainObject> ConstantReturnType; -#endif // not EIGEN_PARSED_BY_DOXYGEN + /** \internal Represents a matrix with all coefficients equal to one another*/ + typedef CwiseNullaryOp, PlainObject> ConstantReturnType; +#endif// not EIGEN_PARSED_BY_DOXYGEN #define EIGEN_CURRENT_STORAGE_BASE_CLASS Eigen::ArrayBase -#define EIGEN_DOC_UNARY_ADDONS(X,Y) -# include "../plugins/CommonCwiseUnaryOps.h" -# include "../plugins/MatrixCwiseUnaryOps.h" -# include "../plugins/ArrayCwiseUnaryOps.h" -# include "../plugins/CommonCwiseBinaryOps.h" -# include "../plugins/MatrixCwiseBinaryOps.h" -# include "../plugins/ArrayCwiseBinaryOps.h" -# ifdef EIGEN_ARRAYBASE_PLUGIN -# include EIGEN_ARRAYBASE_PLUGIN -# endif +#define EIGEN_DOC_UNARY_ADDONS(X, Y) +#include "../plugins/ArrayCwiseBinaryOps.h" +#include "../plugins/ArrayCwiseUnaryOps.h" +#include "../plugins/CommonCwiseBinaryOps.h" +#include "../plugins/CommonCwiseUnaryOps.h" +#include "../plugins/MatrixCwiseBinaryOps.h" +#include "../plugins/MatrixCwiseUnaryOps.h" +#ifdef EIGEN_ARRAYBASE_PLUGIN +#include EIGEN_ARRAYBASE_PLUGIN +#endif #undef EIGEN_CURRENT_STORAGE_BASE_CLASS #undef EIGEN_DOC_UNARY_ADDONS - /** Special case of the template operator=, in order to prevent the compiler - * from generating a default operator= (issue hit with g++ 4.1) - */ - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE - Derived& operator=(const ArrayBase& other) - { - internal::call_assignment(derived(), other.derived()); - return derived(); - } - - /** Set all the entries to \a value. - * \sa DenseBase::setConstant(), DenseBase::fill() */ - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE - Derived& operator=(const Scalar &value) - { Base::setConstant(value); return derived(); } - - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE - Derived& operator+=(const Scalar& scalar); - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE - Derived& operator-=(const Scalar& scalar); - - template - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE - Derived& operator+=(const ArrayBase& other); - template - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE - Derived& operator-=(const ArrayBase& other); - - template - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE - Derived& operator*=(const ArrayBase& other); - - template - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE - Derived& operator/=(const ArrayBase& other); - - public: - EIGEN_DEVICE_FUNC - ArrayBase& array() { return *this; } - EIGEN_DEVICE_FUNC - const ArrayBase& array() const { return *this; } - - /** \returns an \link Eigen::MatrixBase Matrix \endlink expression of this array - * \sa MatrixBase::array() */ - EIGEN_DEVICE_FUNC - MatrixWrapper matrix() { return MatrixWrapper(derived()); } - EIGEN_DEVICE_FUNC - const MatrixWrapper matrix() const { return MatrixWrapper(derived()); } - -// template -// inline void evalTo(Dest& dst) const { dst = matrix(); } - - protected: - EIGEN_DEVICE_FUNC - ArrayBase() : Base() {} - - private: - explicit ArrayBase(Index); - ArrayBase(Index,Index); - template explicit ArrayBase(const ArrayBase&); - protected: - // mixing arrays and matrices is not legal - template Derived& operator+=(const MatrixBase& ) - {EIGEN_STATIC_ASSERT(std::ptrdiff_t(sizeof(typename OtherDerived::Scalar))==-1,YOU_CANNOT_MIX_ARRAYS_AND_MATRICES); return *this;} - // mixing arrays and matrices is not legal - template Derived& operator-=(const MatrixBase& ) - {EIGEN_STATIC_ASSERT(std::ptrdiff_t(sizeof(typename OtherDerived::Scalar))==-1,YOU_CANNOT_MIX_ARRAYS_AND_MATRICES); return *this;} + /** Special case of the template operator=, in order to prevent the compiler + * from generating a default operator= (issue hit with g++ 4.1) + */ + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Derived &operator=(const ArrayBase &other) + { + internal::call_assignment(derived(), other.derived()); + return derived(); + } + + /** Set all the entries to \a value. + * \sa DenseBase::setConstant(), DenseBase::fill() */ + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Derived &operator=(const Scalar &value) + { + Base::setConstant(value); + return derived(); + } + + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Derived &operator+=(const Scalar &scalar); + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Derived &operator-=(const Scalar &scalar); + + template + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Derived &operator+=(const ArrayBase &other); + template + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Derived &operator-=(const ArrayBase &other); + + template + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Derived &operator*=(const ArrayBase &other); + + template + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Derived &operator/=(const ArrayBase &other); + +public: + EIGEN_DEVICE_FUNC + ArrayBase &array() { return *this; } + EIGEN_DEVICE_FUNC + const ArrayBase &array() const { return *this; } + + /** \returns an \link Eigen::MatrixBase Matrix \endlink expression of this array + * \sa MatrixBase::array() */ + EIGEN_DEVICE_FUNC + MatrixWrapper matrix() { return MatrixWrapper(derived()); } + EIGEN_DEVICE_FUNC + const MatrixWrapper matrix() const { return MatrixWrapper(derived()); } + + // template + // inline void evalTo(Dest& dst) const { dst = matrix(); } + +protected: + EIGEN_DEVICE_FUNC + ArrayBase() : Base() {} + +private: + explicit ArrayBase(Index); + ArrayBase(Index, Index); + template explicit ArrayBase(const ArrayBase &); + +protected: + // mixing arrays and matrices is not legal + template Derived &operator+=(const MatrixBase &) + { + EIGEN_STATIC_ASSERT( + std::ptrdiff_t(sizeof(typename OtherDerived::Scalar)) == -1, YOU_CANNOT_MIX_ARRAYS_AND_MATRICES); + return *this; + } + // mixing arrays and matrices is not legal + template Derived &operator-=(const MatrixBase &) + { + EIGEN_STATIC_ASSERT( + std::ptrdiff_t(sizeof(typename OtherDerived::Scalar)) == -1, YOU_CANNOT_MIX_ARRAYS_AND_MATRICES); + return *this; + } }; /** replaces \c *this by \c *this - \a other. - * - * \returns a reference to \c *this - */ + * + * \returns a reference to \c *this + */ template template -EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Derived & -ArrayBase::operator-=(const ArrayBase &other) +EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Derived &ArrayBase::operator-=(const ArrayBase &other) { - call_assignment(derived(), other.derived(), internal::sub_assign_op()); + call_assignment(derived(), other.derived(), internal::sub_assign_op()); return derived(); } /** replaces \c *this by \c *this + \a other. - * - * \returns a reference to \c *this - */ + * + * \returns a reference to \c *this + */ template template -EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Derived & -ArrayBase::operator+=(const ArrayBase& other) +EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Derived &ArrayBase::operator+=(const ArrayBase &other) { - call_assignment(derived(), other.derived(), internal::add_assign_op()); + call_assignment(derived(), other.derived(), internal::add_assign_op()); return derived(); } /** replaces \c *this by \c *this * \a other coefficient wise. - * - * \returns a reference to \c *this - */ + * + * \returns a reference to \c *this + */ template template -EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Derived & -ArrayBase::operator*=(const ArrayBase& other) +EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Derived &ArrayBase::operator*=(const ArrayBase &other) { - call_assignment(derived(), other.derived(), internal::mul_assign_op()); + call_assignment(derived(), other.derived(), internal::mul_assign_op()); return derived(); } /** replaces \c *this by \c *this / \a other coefficient wise. - * - * \returns a reference to \c *this - */ + * + * \returns a reference to \c *this + */ template template -EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Derived & -ArrayBase::operator/=(const ArrayBase& other) +EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Derived &ArrayBase::operator/=(const ArrayBase &other) { - call_assignment(derived(), other.derived(), internal::div_assign_op()); + call_assignment(derived(), other.derived(), internal::div_assign_op()); return derived(); } -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_ARRAYBASE_H +#endif// EIGEN_ARRAYBASE_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/ArrayWrapper.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/ArrayWrapper.h index 688aadd6..a515a7ac 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/ArrayWrapper.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/ArrayWrapper.h @@ -10,200 +10,175 @@ #ifndef EIGEN_ARRAYWRAPPER_H #define EIGEN_ARRAYWRAPPER_H -namespace Eigen { +namespace Eigen { /** \class ArrayWrapper - * \ingroup Core_Module - * - * \brief Expression of a mathematical vector or matrix as an array object - * - * This class is the return type of MatrixBase::array(), and most of the time - * this is the only way it is use. - * - * \sa MatrixBase::array(), class MatrixWrapper - */ + * \ingroup Core_Module + * + * \brief Expression of a mathematical vector or matrix as an array object + * + * This class is the return type of MatrixBase::array(), and most of the time + * this is the only way it is use. + * + * \sa MatrixBase::array(), class MatrixWrapper + */ namespace internal { -template -struct traits > - : public traits::type > -{ - typedef ArrayXpr XprKind; - // Let's remove NestByRefBit - enum { - Flags0 = traits::type >::Flags, - LvalueBitFlag = is_lvalue::value ? LvalueBit : 0, - Flags = (Flags0 & ~(NestByRefBit | LvalueBit)) | LvalueBitFlag + template + struct traits> + : public traits::type> + { + typedef ArrayXpr XprKind; + // Let's remove NestByRefBit + enum { + Flags0 = traits::type>::Flags, + LvalueBitFlag = is_lvalue::value ? LvalueBit : 0, + Flags = (Flags0 & ~(NestByRefBit | LvalueBit)) | LvalueBitFlag + }; }; -}; -} +}// namespace internal -template -class ArrayWrapper : public ArrayBase > +template class ArrayWrapper : public ArrayBase> { - public: - typedef ArrayBase Base; - EIGEN_DENSE_PUBLIC_INTERFACE(ArrayWrapper) - EIGEN_INHERIT_ASSIGNMENT_OPERATORS(ArrayWrapper) - typedef typename internal::remove_all::type NestedExpression; - - typedef typename internal::conditional< - internal::is_lvalue::value, - Scalar, - const Scalar - >::type ScalarWithConstIfNotLvalue; - - typedef typename internal::ref_selector::non_const_type NestedExpressionType; - - using Base::coeffRef; - - EIGEN_DEVICE_FUNC - explicit EIGEN_STRONG_INLINE ArrayWrapper(ExpressionType& matrix) : m_expression(matrix) {} - - EIGEN_DEVICE_FUNC - inline Index rows() const { return m_expression.rows(); } - EIGEN_DEVICE_FUNC - inline Index cols() const { return m_expression.cols(); } - EIGEN_DEVICE_FUNC - inline Index outerStride() const { return m_expression.outerStride(); } - EIGEN_DEVICE_FUNC - inline Index innerStride() const { return m_expression.innerStride(); } - - EIGEN_DEVICE_FUNC - inline ScalarWithConstIfNotLvalue* data() { return m_expression.data(); } - EIGEN_DEVICE_FUNC - inline const Scalar* data() const { return m_expression.data(); } - - EIGEN_DEVICE_FUNC - inline const Scalar& coeffRef(Index rowId, Index colId) const - { - return m_expression.coeffRef(rowId, colId); - } - - EIGEN_DEVICE_FUNC - inline const Scalar& coeffRef(Index index) const - { - return m_expression.coeffRef(index); - } - - template - EIGEN_DEVICE_FUNC - inline void evalTo(Dest& dst) const { dst = m_expression; } - - const typename internal::remove_all::type& - EIGEN_DEVICE_FUNC - nestedExpression() const - { - return m_expression; - } - - /** Forwards the resizing request to the nested expression - * \sa DenseBase::resize(Index) */ - EIGEN_DEVICE_FUNC - void resize(Index newSize) { m_expression.resize(newSize); } - /** Forwards the resizing request to the nested expression - * \sa DenseBase::resize(Index,Index)*/ - EIGEN_DEVICE_FUNC - void resize(Index rows, Index cols) { m_expression.resize(rows,cols); } - - protected: - NestedExpressionType m_expression; +public: + typedef ArrayBase Base; + EIGEN_DENSE_PUBLIC_INTERFACE(ArrayWrapper) + EIGEN_INHERIT_ASSIGNMENT_OPERATORS(ArrayWrapper) + typedef typename internal::remove_all::type NestedExpression; + + typedef typename internal::conditional::value, Scalar, const Scalar>::type + ScalarWithConstIfNotLvalue; + + typedef typename internal::ref_selector::non_const_type NestedExpressionType; + + using Base::coeffRef; + + EIGEN_DEVICE_FUNC + explicit EIGEN_STRONG_INLINE ArrayWrapper(ExpressionType &matrix) : m_expression(matrix) {} + + EIGEN_DEVICE_FUNC + inline Index rows() const { return m_expression.rows(); } + EIGEN_DEVICE_FUNC + inline Index cols() const { return m_expression.cols(); } + EIGEN_DEVICE_FUNC + inline Index outerStride() const { return m_expression.outerStride(); } + EIGEN_DEVICE_FUNC + inline Index innerStride() const { return m_expression.innerStride(); } + + EIGEN_DEVICE_FUNC + inline ScalarWithConstIfNotLvalue *data() { return m_expression.data(); } + EIGEN_DEVICE_FUNC + inline const Scalar *data() const { return m_expression.data(); } + + EIGEN_DEVICE_FUNC + inline const Scalar &coeffRef(Index rowId, Index colId) const { return m_expression.coeffRef(rowId, colId); } + + EIGEN_DEVICE_FUNC + inline const Scalar &coeffRef(Index index) const { return m_expression.coeffRef(index); } + + template EIGEN_DEVICE_FUNC inline void evalTo(Dest &dst) const { dst = m_expression; } + + const typename internal::remove_all::type &EIGEN_DEVICE_FUNC nestedExpression() const + { + return m_expression; + } + + /** Forwards the resizing request to the nested expression + * \sa DenseBase::resize(Index) */ + EIGEN_DEVICE_FUNC + void resize(Index newSize) { m_expression.resize(newSize); } + /** Forwards the resizing request to the nested expression + * \sa DenseBase::resize(Index,Index)*/ + EIGEN_DEVICE_FUNC + void resize(Index rows, Index cols) { m_expression.resize(rows, cols); } + +protected: + NestedExpressionType m_expression; }; /** \class MatrixWrapper - * \ingroup Core_Module - * - * \brief Expression of an array as a mathematical vector or matrix - * - * This class is the return type of ArrayBase::matrix(), and most of the time - * this is the only way it is use. - * - * \sa MatrixBase::matrix(), class ArrayWrapper - */ + * \ingroup Core_Module + * + * \brief Expression of an array as a mathematical vector or matrix + * + * This class is the return type of ArrayBase::matrix(), and most of the time + * this is the only way it is use. + * + * \sa MatrixBase::matrix(), class ArrayWrapper + */ namespace internal { -template -struct traits > - : public traits::type > -{ - typedef MatrixXpr XprKind; - // Let's remove NestByRefBit - enum { - Flags0 = traits::type >::Flags, - LvalueBitFlag = is_lvalue::value ? LvalueBit : 0, - Flags = (Flags0 & ~(NestByRefBit | LvalueBit)) | LvalueBitFlag + template + struct traits> + : public traits::type> + { + typedef MatrixXpr XprKind; + // Let's remove NestByRefBit + enum { + Flags0 = traits::type>::Flags, + LvalueBitFlag = is_lvalue::value ? LvalueBit : 0, + Flags = (Flags0 & ~(NestByRefBit | LvalueBit)) | LvalueBitFlag + }; }; -}; -} +}// namespace internal -template -class MatrixWrapper : public MatrixBase > +template class MatrixWrapper : public MatrixBase> { - public: - typedef MatrixBase > Base; - EIGEN_DENSE_PUBLIC_INTERFACE(MatrixWrapper) - EIGEN_INHERIT_ASSIGNMENT_OPERATORS(MatrixWrapper) - typedef typename internal::remove_all::type NestedExpression; - - typedef typename internal::conditional< - internal::is_lvalue::value, - Scalar, - const Scalar - >::type ScalarWithConstIfNotLvalue; - - typedef typename internal::ref_selector::non_const_type NestedExpressionType; - - using Base::coeffRef; - - EIGEN_DEVICE_FUNC - explicit inline MatrixWrapper(ExpressionType& matrix) : m_expression(matrix) {} - - EIGEN_DEVICE_FUNC - inline Index rows() const { return m_expression.rows(); } - EIGEN_DEVICE_FUNC - inline Index cols() const { return m_expression.cols(); } - EIGEN_DEVICE_FUNC - inline Index outerStride() const { return m_expression.outerStride(); } - EIGEN_DEVICE_FUNC - inline Index innerStride() const { return m_expression.innerStride(); } - - EIGEN_DEVICE_FUNC - inline ScalarWithConstIfNotLvalue* data() { return m_expression.data(); } - EIGEN_DEVICE_FUNC - inline const Scalar* data() const { return m_expression.data(); } - - EIGEN_DEVICE_FUNC - inline const Scalar& coeffRef(Index rowId, Index colId) const - { - return m_expression.derived().coeffRef(rowId, colId); - } - - EIGEN_DEVICE_FUNC - inline const Scalar& coeffRef(Index index) const - { - return m_expression.coeffRef(index); - } - - EIGEN_DEVICE_FUNC - const typename internal::remove_all::type& - nestedExpression() const - { - return m_expression; - } - - /** Forwards the resizing request to the nested expression - * \sa DenseBase::resize(Index) */ - EIGEN_DEVICE_FUNC - void resize(Index newSize) { m_expression.resize(newSize); } - /** Forwards the resizing request to the nested expression - * \sa DenseBase::resize(Index,Index)*/ - EIGEN_DEVICE_FUNC - void resize(Index rows, Index cols) { m_expression.resize(rows,cols); } - - protected: - NestedExpressionType m_expression; +public: + typedef MatrixBase> Base; + EIGEN_DENSE_PUBLIC_INTERFACE(MatrixWrapper) + EIGEN_INHERIT_ASSIGNMENT_OPERATORS(MatrixWrapper) + typedef typename internal::remove_all::type NestedExpression; + + typedef typename internal::conditional::value, Scalar, const Scalar>::type + ScalarWithConstIfNotLvalue; + + typedef typename internal::ref_selector::non_const_type NestedExpressionType; + + using Base::coeffRef; + + EIGEN_DEVICE_FUNC + explicit inline MatrixWrapper(ExpressionType &matrix) : m_expression(matrix) {} + + EIGEN_DEVICE_FUNC + inline Index rows() const { return m_expression.rows(); } + EIGEN_DEVICE_FUNC + inline Index cols() const { return m_expression.cols(); } + EIGEN_DEVICE_FUNC + inline Index outerStride() const { return m_expression.outerStride(); } + EIGEN_DEVICE_FUNC + inline Index innerStride() const { return m_expression.innerStride(); } + + EIGEN_DEVICE_FUNC + inline ScalarWithConstIfNotLvalue *data() { return m_expression.data(); } + EIGEN_DEVICE_FUNC + inline const Scalar *data() const { return m_expression.data(); } + + EIGEN_DEVICE_FUNC + inline const Scalar &coeffRef(Index rowId, Index colId) const + { + return m_expression.derived().coeffRef(rowId, colId); + } + + EIGEN_DEVICE_FUNC + inline const Scalar &coeffRef(Index index) const { return m_expression.coeffRef(index); } + + EIGEN_DEVICE_FUNC + const typename internal::remove_all::type &nestedExpression() const { return m_expression; } + + /** Forwards the resizing request to the nested expression + * \sa DenseBase::resize(Index) */ + EIGEN_DEVICE_FUNC + void resize(Index newSize) { m_expression.resize(newSize); } + /** Forwards the resizing request to the nested expression + * \sa DenseBase::resize(Index,Index)*/ + EIGEN_DEVICE_FUNC + void resize(Index rows, Index cols) { m_expression.resize(rows, cols); } + +protected: + NestedExpressionType m_expression; }; -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_ARRAYWRAPPER_H +#endif// EIGEN_ARRAYWRAPPER_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/Assign.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/Assign.h index 53806ba3..0de44a55 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/Assign.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/Assign.h @@ -16,61 +16,54 @@ namespace Eigen { template template -EIGEN_STRONG_INLINE Derived& DenseBase - ::lazyAssign(const DenseBase& other) +EIGEN_STRONG_INLINE Derived &DenseBase::lazyAssign(const DenseBase &other) { - enum{ - SameType = internal::is_same::value - }; + enum { SameType = internal::is_same::value }; EIGEN_STATIC_ASSERT_LVALUE(Derived) - EIGEN_STATIC_ASSERT_SAME_MATRIX_SIZE(Derived,OtherDerived) - EIGEN_STATIC_ASSERT(SameType,YOU_MIXED_DIFFERENT_NUMERIC_TYPES__YOU_NEED_TO_USE_THE_CAST_METHOD_OF_MATRIXBASE_TO_CAST_NUMERIC_TYPES_EXPLICITLY) + EIGEN_STATIC_ASSERT_SAME_MATRIX_SIZE(Derived, OtherDerived) + EIGEN_STATIC_ASSERT(SameType, + YOU_MIXED_DIFFERENT_NUMERIC_TYPES__YOU_NEED_TO_USE_THE_CAST_METHOD_OF_MATRIXBASE_TO_CAST_NUMERIC_TYPES_EXPLICITLY) eigen_assert(rows() == other.rows() && cols() == other.cols()); - internal::call_assignment_no_alias(derived(),other.derived()); - + internal::call_assignment_no_alias(derived(), other.derived()); + return derived(); } template template -EIGEN_DEVICE_FUNC -EIGEN_STRONG_INLINE Derived& DenseBase::operator=(const DenseBase& other) +EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Derived &DenseBase::operator=(const DenseBase &other) { internal::call_assignment(derived(), other.derived()); return derived(); } template -EIGEN_DEVICE_FUNC -EIGEN_STRONG_INLINE Derived& DenseBase::operator=(const DenseBase& other) +EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Derived &DenseBase::operator=(const DenseBase &other) { internal::call_assignment(derived(), other.derived()); return derived(); } template -EIGEN_DEVICE_FUNC -EIGEN_STRONG_INLINE Derived& MatrixBase::operator=(const MatrixBase& other) +EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Derived &MatrixBase::operator=(const MatrixBase &other) { internal::call_assignment(derived(), other.derived()); return derived(); } template -template -EIGEN_DEVICE_FUNC -EIGEN_STRONG_INLINE Derived& MatrixBase::operator=(const DenseBase& other) +template +EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Derived &MatrixBase::operator=(const DenseBase &other) { internal::call_assignment(derived(), other.derived()); return derived(); } template -template -EIGEN_DEVICE_FUNC -EIGEN_STRONG_INLINE Derived& MatrixBase::operator=(const EigenBase& other) +template +EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Derived &MatrixBase::operator=(const EigenBase &other) { internal::call_assignment(derived(), other.derived()); return derived(); @@ -78,13 +71,12 @@ EIGEN_STRONG_INLINE Derived& MatrixBase::operator=(const EigenBase template -EIGEN_DEVICE_FUNC -EIGEN_STRONG_INLINE Derived& MatrixBase::operator=(const ReturnByValue& other) +EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Derived &MatrixBase::operator=(const ReturnByValue &other) { other.derived().evalTo(derived()); return derived(); } -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_ASSIGN_H +#endif// EIGEN_ASSIGN_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/AssignEvaluator.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/AssignEvaluator.h index dbe435d8..8006d2c2 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/AssignEvaluator.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/AssignEvaluator.h @@ -17,919 +17,903 @@ namespace Eigen { // This implementation is based on Assign.h namespace internal { - -/*************************************************************************** -* Part 1 : the logic deciding a strategy for traversal and unrolling * -***************************************************************************/ - -// copy_using_evaluator_traits is based on assign_traits - -template -struct copy_using_evaluator_traits -{ - typedef typename DstEvaluator::XprType Dst; - typedef typename Dst::Scalar DstScalar; - - enum { - DstFlags = DstEvaluator::Flags, - SrcFlags = SrcEvaluator::Flags - }; - -public: - enum { - DstAlignment = DstEvaluator::Alignment, - SrcAlignment = SrcEvaluator::Alignment, - DstHasDirectAccess = (DstFlags & DirectAccessBit) == DirectAccessBit, - JointAlignment = EIGEN_PLAIN_ENUM_MIN(DstAlignment,SrcAlignment) - }; - -private: - enum { - InnerSize = int(Dst::IsVectorAtCompileTime) ? int(Dst::SizeAtCompileTime) - : int(DstFlags)&RowMajorBit ? int(Dst::ColsAtCompileTime) - : int(Dst::RowsAtCompileTime), - InnerMaxSize = int(Dst::IsVectorAtCompileTime) ? int(Dst::MaxSizeAtCompileTime) - : int(DstFlags)&RowMajorBit ? int(Dst::MaxColsAtCompileTime) - : int(Dst::MaxRowsAtCompileTime), - OuterStride = int(outer_stride_at_compile_time::ret), - MaxSizeAtCompileTime = Dst::SizeAtCompileTime - }; - - // TODO distinguish between linear traversal and inner-traversals - typedef typename find_best_packet::type LinearPacketType; - typedef typename find_best_packet::type InnerPacketType; - - enum { - LinearPacketSize = unpacket_traits::size, - InnerPacketSize = unpacket_traits::size - }; - -public: - enum { - LinearRequiredAlignment = unpacket_traits::alignment, - InnerRequiredAlignment = unpacket_traits::alignment - }; - -private: - enum { - DstIsRowMajor = DstFlags&RowMajorBit, - SrcIsRowMajor = SrcFlags&RowMajorBit, - StorageOrdersAgree = (int(DstIsRowMajor) == int(SrcIsRowMajor)), - MightVectorize = bool(StorageOrdersAgree) - && (int(DstFlags) & int(SrcFlags) & ActualPacketAccessBit) - && bool(functor_traits::PacketAccess), - MayInnerVectorize = MightVectorize - && int(InnerSize)!=Dynamic && int(InnerSize)%int(InnerPacketSize)==0 - && int(OuterStride)!=Dynamic && int(OuterStride)%int(InnerPacketSize)==0 - && (EIGEN_UNALIGNED_VECTORIZE || int(JointAlignment)>=int(InnerRequiredAlignment)), - MayLinearize = bool(StorageOrdersAgree) && (int(DstFlags) & int(SrcFlags) & LinearAccessBit), - MayLinearVectorize = bool(MightVectorize) && bool(MayLinearize) && bool(DstHasDirectAccess) - && (EIGEN_UNALIGNED_VECTORIZE || (int(DstAlignment)>=int(LinearRequiredAlignment)) || MaxSizeAtCompileTime == Dynamic), + + /*************************************************************************** + * Part 1 : the logic deciding a strategy for traversal and unrolling * + ***************************************************************************/ + + // copy_using_evaluator_traits is based on assign_traits + + template struct copy_using_evaluator_traits + { + typedef typename DstEvaluator::XprType Dst; + typedef typename Dst::Scalar DstScalar; + + enum { DstFlags = DstEvaluator::Flags, SrcFlags = SrcEvaluator::Flags }; + + public: + enum { + DstAlignment = DstEvaluator::Alignment, + SrcAlignment = SrcEvaluator::Alignment, + DstHasDirectAccess = (DstFlags & DirectAccessBit) == DirectAccessBit, + JointAlignment = EIGEN_PLAIN_ENUM_MIN(DstAlignment, SrcAlignment) + }; + + private: + enum { + InnerSize = int(Dst::IsVectorAtCompileTime) ? int(Dst::SizeAtCompileTime) + : int(DstFlags) & RowMajorBit ? int(Dst::ColsAtCompileTime) + : int(Dst::RowsAtCompileTime), + InnerMaxSize = int(Dst::IsVectorAtCompileTime) ? int(Dst::MaxSizeAtCompileTime) + : int(DstFlags) & RowMajorBit ? int(Dst::MaxColsAtCompileTime) + : int(Dst::MaxRowsAtCompileTime), + OuterStride = int(outer_stride_at_compile_time::ret), + MaxSizeAtCompileTime = Dst::SizeAtCompileTime + }; + + // TODO distinguish between linear traversal and inner-traversals + typedef typename find_best_packet::type LinearPacketType; + typedef typename find_best_packet::type InnerPacketType; + + enum { + LinearPacketSize = unpacket_traits::size, + InnerPacketSize = unpacket_traits::size + }; + + public: + enum { + LinearRequiredAlignment = unpacket_traits::alignment, + InnerRequiredAlignment = unpacket_traits::alignment + }; + + private: + enum { + DstIsRowMajor = DstFlags & RowMajorBit, + SrcIsRowMajor = SrcFlags & RowMajorBit, + StorageOrdersAgree = (int(DstIsRowMajor) == int(SrcIsRowMajor)), + MightVectorize = bool(StorageOrdersAgree) && (int(DstFlags) & int(SrcFlags) & ActualPacketAccessBit) + && bool(functor_traits::PacketAccess), + MayInnerVectorize = MightVectorize && int(InnerSize) != Dynamic && int(InnerSize) % int(InnerPacketSize) == 0 + && int(OuterStride) != Dynamic && int(OuterStride) % int(InnerPacketSize) == 0 + && (EIGEN_UNALIGNED_VECTORIZE || int(JointAlignment) >= int(InnerRequiredAlignment)), + MayLinearize = bool(StorageOrdersAgree) && (int(DstFlags) & int(SrcFlags) & LinearAccessBit), + MayLinearVectorize = bool(MightVectorize) && bool(MayLinearize) && bool(DstHasDirectAccess) + && (EIGEN_UNALIGNED_VECTORIZE || (int(DstAlignment) >= int(LinearRequiredAlignment)) + || MaxSizeAtCompileTime == Dynamic), /* If the destination isn't aligned, we have to do runtime checks and we don't unroll, so it's only good for large enough sizes. */ - MaySliceVectorize = bool(MightVectorize) && bool(DstHasDirectAccess) - && (int(InnerMaxSize)==Dynamic || int(InnerMaxSize)>=(EIGEN_UNALIGNED_VECTORIZE?InnerPacketSize:(3*InnerPacketSize))) + MaySliceVectorize = + bool(MightVectorize) && bool(DstHasDirectAccess) + && (int(InnerMaxSize) == Dynamic + || int(InnerMaxSize) >= (EIGEN_UNALIGNED_VECTORIZE ? InnerPacketSize : (3 * InnerPacketSize))) /* slice vectorization can be slow, so we only want it if the slices are big, which is indicated by InnerMaxSize rather than InnerSize, think of the case of a dynamic block in a fixed-size matrix However, with EIGEN_UNALIGNED_VECTORIZE and unrolling, slice vectorization is still worth it */ - }; + }; + + public: + enum { + Traversal = int(MayLinearVectorize) && (LinearPacketSize > InnerPacketSize) ? int(LinearVectorizedTraversal) + : int(MayInnerVectorize) ? int(InnerVectorizedTraversal) + : int(MayLinearVectorize) ? int(LinearVectorizedTraversal) + : int(MaySliceVectorize) ? int(SliceVectorizedTraversal) + : int(MayLinearize) ? int(LinearTraversal) + : int(DefaultTraversal), + Vectorized = int(Traversal) == InnerVectorizedTraversal || int(Traversal) == LinearVectorizedTraversal + || int(Traversal) == SliceVectorizedTraversal + }; + + typedef typename conditional::type + PacketType; + + private: + enum { + ActualPacketSize = int(Traversal) == LinearVectorizedTraversal ? LinearPacketSize + : Vectorized ? InnerPacketSize + : 1, + UnrollingLimit = EIGEN_UNROLLING_LIMIT * ActualPacketSize, + MayUnrollCompletely = + int(Dst::SizeAtCompileTime) != Dynamic + && int(Dst::SizeAtCompileTime) * (int(DstEvaluator::CoeffReadCost) + int(SrcEvaluator::CoeffReadCost)) + <= int(UnrollingLimit), + MayUnrollInner = + int(InnerSize) != Dynamic + && int(InnerSize) * (int(DstEvaluator::CoeffReadCost) + int(SrcEvaluator::CoeffReadCost)) <= int(UnrollingLimit) + }; -public: - enum { - Traversal = int(MayLinearVectorize) && (LinearPacketSize>InnerPacketSize) ? int(LinearVectorizedTraversal) - : int(MayInnerVectorize) ? int(InnerVectorizedTraversal) - : int(MayLinearVectorize) ? int(LinearVectorizedTraversal) - : int(MaySliceVectorize) ? int(SliceVectorizedTraversal) - : int(MayLinearize) ? int(LinearTraversal) - : int(DefaultTraversal), - Vectorized = int(Traversal) == InnerVectorizedTraversal - || int(Traversal) == LinearVectorizedTraversal - || int(Traversal) == SliceVectorizedTraversal - }; - - typedef typename conditional::type PacketType; - -private: - enum { - ActualPacketSize = int(Traversal)==LinearVectorizedTraversal ? LinearPacketSize - : Vectorized ? InnerPacketSize - : 1, - UnrollingLimit = EIGEN_UNROLLING_LIMIT * ActualPacketSize, - MayUnrollCompletely = int(Dst::SizeAtCompileTime) != Dynamic - && int(Dst::SizeAtCompileTime) * (int(DstEvaluator::CoeffReadCost)+int(SrcEvaluator::CoeffReadCost)) <= int(UnrollingLimit), - MayUnrollInner = int(InnerSize) != Dynamic - && int(InnerSize) * (int(DstEvaluator::CoeffReadCost)+int(SrcEvaluator::CoeffReadCost)) <= int(UnrollingLimit) - }; - -public: - enum { - Unrolling = (int(Traversal) == int(InnerVectorizedTraversal) || int(Traversal) == int(DefaultTraversal)) - ? ( - int(MayUnrollCompletely) ? int(CompleteUnrolling) - : int(MayUnrollInner) ? int(InnerUnrolling) - : int(NoUnrolling) - ) - : int(Traversal) == int(LinearVectorizedTraversal) - ? ( bool(MayUnrollCompletely) && ( EIGEN_UNALIGNED_VECTORIZE || (int(DstAlignment)>=int(LinearRequiredAlignment))) + public: + enum { + Unrolling = (int(Traversal) == int(InnerVectorizedTraversal) || int(Traversal) == int(DefaultTraversal)) + ? (int(MayUnrollCompletely) ? int(CompleteUnrolling) + : int(MayUnrollInner) ? int(InnerUnrolling) + : int(NoUnrolling)) + : int(Traversal) == int(LinearVectorizedTraversal) + ? (bool(MayUnrollCompletely) + && (EIGEN_UNALIGNED_VECTORIZE || (int(DstAlignment) >= int(LinearRequiredAlignment))) ? int(CompleteUnrolling) - : int(NoUnrolling) ) - : int(Traversal) == int(LinearTraversal) - ? ( bool(MayUnrollCompletely) ? int(CompleteUnrolling) - : int(NoUnrolling) ) + : int(NoUnrolling)) + : int(Traversal) == int(LinearTraversal) + ? (bool(MayUnrollCompletely) ? int(CompleteUnrolling) : int(NoUnrolling)) #if EIGEN_UNALIGNED_VECTORIZE - : int(Traversal) == int(SliceVectorizedTraversal) - ? ( bool(MayUnrollInner) ? int(InnerUnrolling) - : int(NoUnrolling) ) + : int(Traversal) == int(SliceVectorizedTraversal) + ? (bool(MayUnrollInner) ? int(InnerUnrolling) : int(NoUnrolling)) #endif - : int(NoUnrolling) - }; + : int(NoUnrolling) + }; #ifdef EIGEN_DEBUG_ASSIGN - static void debug() - { - std::cerr << "DstXpr: " << typeid(typename DstEvaluator::XprType).name() << std::endl; - std::cerr << "SrcXpr: " << typeid(typename SrcEvaluator::XprType).name() << std::endl; - std::cerr.setf(std::ios::hex, std::ios::basefield); - std::cerr << "DstFlags" << " = " << DstFlags << " (" << demangle_flags(DstFlags) << " )" << std::endl; - std::cerr << "SrcFlags" << " = " << SrcFlags << " (" << demangle_flags(SrcFlags) << " )" << std::endl; - std::cerr.unsetf(std::ios::hex); - EIGEN_DEBUG_VAR(DstAlignment) - EIGEN_DEBUG_VAR(SrcAlignment) - EIGEN_DEBUG_VAR(LinearRequiredAlignment) - EIGEN_DEBUG_VAR(InnerRequiredAlignment) - EIGEN_DEBUG_VAR(JointAlignment) - EIGEN_DEBUG_VAR(InnerSize) - EIGEN_DEBUG_VAR(InnerMaxSize) - EIGEN_DEBUG_VAR(LinearPacketSize) - EIGEN_DEBUG_VAR(InnerPacketSize) - EIGEN_DEBUG_VAR(ActualPacketSize) - EIGEN_DEBUG_VAR(StorageOrdersAgree) - EIGEN_DEBUG_VAR(MightVectorize) - EIGEN_DEBUG_VAR(MayLinearize) - EIGEN_DEBUG_VAR(MayInnerVectorize) - EIGEN_DEBUG_VAR(MayLinearVectorize) - EIGEN_DEBUG_VAR(MaySliceVectorize) - std::cerr << "Traversal" << " = " << Traversal << " (" << demangle_traversal(Traversal) << ")" << std::endl; - EIGEN_DEBUG_VAR(SrcEvaluator::CoeffReadCost) - EIGEN_DEBUG_VAR(UnrollingLimit) - EIGEN_DEBUG_VAR(MayUnrollCompletely) - EIGEN_DEBUG_VAR(MayUnrollInner) - std::cerr << "Unrolling" << " = " << Unrolling << " (" << demangle_unrolling(Unrolling) << ")" << std::endl; - std::cerr << std::endl; - } + static void debug() + { + std::cerr << "DstXpr: " << typeid(typename DstEvaluator::XprType).name() << std::endl; + std::cerr << "SrcXpr: " << typeid(typename SrcEvaluator::XprType).name() << std::endl; + std::cerr.setf(std::ios::hex, std::ios::basefield); + std::cerr << "DstFlags" << " = " << DstFlags << " (" << demangle_flags(DstFlags) << " )" << std::endl; + std::cerr << "SrcFlags" << " = " << SrcFlags << " (" << demangle_flags(SrcFlags) << " )" << std::endl; + std::cerr.unsetf(std::ios::hex); + EIGEN_DEBUG_VAR(DstAlignment) + EIGEN_DEBUG_VAR(SrcAlignment) + EIGEN_DEBUG_VAR(LinearRequiredAlignment) + EIGEN_DEBUG_VAR(InnerRequiredAlignment) + EIGEN_DEBUG_VAR(JointAlignment) + EIGEN_DEBUG_VAR(InnerSize) + EIGEN_DEBUG_VAR(InnerMaxSize) + EIGEN_DEBUG_VAR(LinearPacketSize) + EIGEN_DEBUG_VAR(InnerPacketSize) + EIGEN_DEBUG_VAR(ActualPacketSize) + EIGEN_DEBUG_VAR(StorageOrdersAgree) + EIGEN_DEBUG_VAR(MightVectorize) + EIGEN_DEBUG_VAR(MayLinearize) + EIGEN_DEBUG_VAR(MayInnerVectorize) + EIGEN_DEBUG_VAR(MayLinearVectorize) + EIGEN_DEBUG_VAR(MaySliceVectorize) + std::cerr << "Traversal" << " = " << Traversal << " (" << demangle_traversal(Traversal) << ")" << std::endl; + EIGEN_DEBUG_VAR(SrcEvaluator::CoeffReadCost) + EIGEN_DEBUG_VAR(UnrollingLimit) + EIGEN_DEBUG_VAR(MayUnrollCompletely) + EIGEN_DEBUG_VAR(MayUnrollInner) + std::cerr << "Unrolling" << " = " << Unrolling << " (" << demangle_unrolling(Unrolling) << ")" << std::endl; + std::cerr << std::endl; + } #endif -}; + }; -/*************************************************************************** -* Part 2 : meta-unrollers -***************************************************************************/ + /*************************************************************************** + * Part 2 : meta-unrollers + ***************************************************************************/ -/************************ -*** Default traversal *** -************************/ + /************************ + *** Default traversal *** + ************************/ -template -struct copy_using_evaluator_DefaultTraversal_CompleteUnrolling -{ - // FIXME: this is not very clean, perhaps this information should be provided by the kernel? - typedef typename Kernel::DstEvaluatorType DstEvaluatorType; - typedef typename DstEvaluatorType::XprType DstXprType; - - enum { - outer = Index / DstXprType::InnerSizeAtCompileTime, - inner = Index % DstXprType::InnerSizeAtCompileTime + template struct copy_using_evaluator_DefaultTraversal_CompleteUnrolling + { + // FIXME: this is not very clean, perhaps this information should be provided by the kernel? + typedef typename Kernel::DstEvaluatorType DstEvaluatorType; + typedef typename DstEvaluatorType::XprType DstXprType; + + enum { outer = Index / DstXprType::InnerSizeAtCompileTime, inner = Index % DstXprType::InnerSizeAtCompileTime }; + + EIGEN_DEVICE_FUNC static EIGEN_STRONG_INLINE void run(Kernel &kernel) + { + kernel.assignCoeffByOuterInner(outer, inner); + copy_using_evaluator_DefaultTraversal_CompleteUnrolling::run(kernel); + } }; - EIGEN_DEVICE_FUNC static EIGEN_STRONG_INLINE void run(Kernel &kernel) + template struct copy_using_evaluator_DefaultTraversal_CompleteUnrolling { - kernel.assignCoeffByOuterInner(outer, inner); - copy_using_evaluator_DefaultTraversal_CompleteUnrolling::run(kernel); - } -}; + EIGEN_DEVICE_FUNC static EIGEN_STRONG_INLINE void run(Kernel &) {} + }; -template -struct copy_using_evaluator_DefaultTraversal_CompleteUnrolling -{ - EIGEN_DEVICE_FUNC static EIGEN_STRONG_INLINE void run(Kernel&) { } -}; + template struct copy_using_evaluator_DefaultTraversal_InnerUnrolling + { + EIGEN_DEVICE_FUNC static EIGEN_STRONG_INLINE void run(Kernel &kernel, Index outer) + { + kernel.assignCoeffByOuterInner(outer, Index_); + copy_using_evaluator_DefaultTraversal_InnerUnrolling::run(kernel, outer); + } + }; -template -struct copy_using_evaluator_DefaultTraversal_InnerUnrolling -{ - EIGEN_DEVICE_FUNC static EIGEN_STRONG_INLINE void run(Kernel &kernel, Index outer) + template struct copy_using_evaluator_DefaultTraversal_InnerUnrolling { - kernel.assignCoeffByOuterInner(outer, Index_); - copy_using_evaluator_DefaultTraversal_InnerUnrolling::run(kernel, outer); - } -}; + EIGEN_DEVICE_FUNC static EIGEN_STRONG_INLINE void run(Kernel &, Index) {} + }; -template -struct copy_using_evaluator_DefaultTraversal_InnerUnrolling -{ - EIGEN_DEVICE_FUNC static EIGEN_STRONG_INLINE void run(Kernel&, Index) { } -}; + /*********************** + *** Linear traversal *** + ***********************/ -/*********************** -*** Linear traversal *** -***********************/ + template struct copy_using_evaluator_LinearTraversal_CompleteUnrolling + { + EIGEN_DEVICE_FUNC static EIGEN_STRONG_INLINE void run(Kernel &kernel) + { + kernel.assignCoeff(Index); + copy_using_evaluator_LinearTraversal_CompleteUnrolling::run(kernel); + } + }; -template -struct copy_using_evaluator_LinearTraversal_CompleteUnrolling -{ - EIGEN_DEVICE_FUNC static EIGEN_STRONG_INLINE void run(Kernel& kernel) + template struct copy_using_evaluator_LinearTraversal_CompleteUnrolling { - kernel.assignCoeff(Index); - copy_using_evaluator_LinearTraversal_CompleteUnrolling::run(kernel); - } -}; - -template -struct copy_using_evaluator_LinearTraversal_CompleteUnrolling -{ - EIGEN_DEVICE_FUNC static EIGEN_STRONG_INLINE void run(Kernel&) { } -}; - -/************************** -*** Inner vectorization *** -**************************/ - -template -struct copy_using_evaluator_innervec_CompleteUnrolling -{ - // FIXME: this is not very clean, perhaps this information should be provided by the kernel? - typedef typename Kernel::DstEvaluatorType DstEvaluatorType; - typedef typename DstEvaluatorType::XprType DstXprType; - typedef typename Kernel::PacketType PacketType; - - enum { - outer = Index / DstXprType::InnerSizeAtCompileTime, - inner = Index % DstXprType::InnerSizeAtCompileTime, - SrcAlignment = Kernel::AssignmentTraits::SrcAlignment, - DstAlignment = Kernel::AssignmentTraits::DstAlignment - }; - - EIGEN_DEVICE_FUNC static EIGEN_STRONG_INLINE void run(Kernel &kernel) - { - kernel.template assignPacketByOuterInner(outer, inner); - enum { NextIndex = Index + unpacket_traits::size }; - copy_using_evaluator_innervec_CompleteUnrolling::run(kernel); - } -}; - -template -struct copy_using_evaluator_innervec_CompleteUnrolling -{ - EIGEN_DEVICE_FUNC static EIGEN_STRONG_INLINE void run(Kernel&) { } -}; - -template -struct copy_using_evaluator_innervec_InnerUnrolling -{ - typedef typename Kernel::PacketType PacketType; - EIGEN_DEVICE_FUNC static EIGEN_STRONG_INLINE void run(Kernel &kernel, Index outer) - { - kernel.template assignPacketByOuterInner(outer, Index_); - enum { NextIndex = Index_ + unpacket_traits::size }; - copy_using_evaluator_innervec_InnerUnrolling::run(kernel, outer); - } -}; + EIGEN_DEVICE_FUNC static EIGEN_STRONG_INLINE void run(Kernel &) {} + }; -template -struct copy_using_evaluator_innervec_InnerUnrolling -{ - EIGEN_DEVICE_FUNC static EIGEN_STRONG_INLINE void run(Kernel &, Index) { } -}; + /************************** + *** Inner vectorization *** + **************************/ -/*************************************************************************** -* Part 3 : implementation of all cases -***************************************************************************/ + template struct copy_using_evaluator_innervec_CompleteUnrolling + { + // FIXME: this is not very clean, perhaps this information should be provided by the kernel? + typedef typename Kernel::DstEvaluatorType DstEvaluatorType; + typedef typename DstEvaluatorType::XprType DstXprType; + typedef typename Kernel::PacketType PacketType; -// dense_assignment_loop is based on assign_impl + enum { + outer = Index / DstXprType::InnerSizeAtCompileTime, + inner = Index % DstXprType::InnerSizeAtCompileTime, + SrcAlignment = Kernel::AssignmentTraits::SrcAlignment, + DstAlignment = Kernel::AssignmentTraits::DstAlignment + }; -template -struct dense_assignment_loop; + EIGEN_DEVICE_FUNC static EIGEN_STRONG_INLINE void run(Kernel &kernel) + { + kernel.template assignPacketByOuterInner(outer, inner); + enum { NextIndex = Index + unpacket_traits::size }; + copy_using_evaluator_innervec_CompleteUnrolling::run(kernel); + } + }; -/************************ -*** Default traversal *** -************************/ + template struct copy_using_evaluator_innervec_CompleteUnrolling + { + EIGEN_DEVICE_FUNC static EIGEN_STRONG_INLINE void run(Kernel &) {} + }; + + template + struct copy_using_evaluator_innervec_InnerUnrolling + { + typedef typename Kernel::PacketType PacketType; + EIGEN_DEVICE_FUNC static EIGEN_STRONG_INLINE void run(Kernel &kernel, Index outer) + { + kernel.template assignPacketByOuterInner(outer, Index_); + enum { NextIndex = Index_ + unpacket_traits::size }; + copy_using_evaluator_innervec_InnerUnrolling::run( + kernel, outer); + } + }; -template -struct dense_assignment_loop -{ - EIGEN_DEVICE_FUNC static void EIGEN_STRONG_INLINE run(Kernel &kernel) + template + struct copy_using_evaluator_innervec_InnerUnrolling { - for(Index outer = 0; outer < kernel.outerSize(); ++outer) { - for(Index inner = 0; inner < kernel.innerSize(); ++inner) { - kernel.assignCoeffByOuterInner(outer, inner); + EIGEN_DEVICE_FUNC static EIGEN_STRONG_INLINE void run(Kernel &, Index) {} + }; + + /*************************************************************************** + * Part 3 : implementation of all cases + ***************************************************************************/ + + // dense_assignment_loop is based on assign_impl + + template + struct dense_assignment_loop; + + /************************ + *** Default traversal *** + ************************/ + + template struct dense_assignment_loop + { + EIGEN_DEVICE_FUNC static void EIGEN_STRONG_INLINE run(Kernel &kernel) + { + for (Index outer = 0; outer < kernel.outerSize(); ++outer) { + for (Index inner = 0; inner < kernel.innerSize(); ++inner) { kernel.assignCoeffByOuterInner(outer, inner); } } } - } -}; + }; -template -struct dense_assignment_loop -{ - EIGEN_DEVICE_FUNC static EIGEN_STRONG_INLINE void run(Kernel &kernel) + template struct dense_assignment_loop { - typedef typename Kernel::DstEvaluatorType::XprType DstXprType; - copy_using_evaluator_DefaultTraversal_CompleteUnrolling::run(kernel); - } -}; + EIGEN_DEVICE_FUNC static EIGEN_STRONG_INLINE void run(Kernel &kernel) + { + typedef typename Kernel::DstEvaluatorType::XprType DstXprType; + copy_using_evaluator_DefaultTraversal_CompleteUnrolling::run(kernel); + } + }; -template -struct dense_assignment_loop -{ - EIGEN_DEVICE_FUNC static EIGEN_STRONG_INLINE void run(Kernel &kernel) + template struct dense_assignment_loop { - typedef typename Kernel::DstEvaluatorType::XprType DstXprType; + EIGEN_DEVICE_FUNC static EIGEN_STRONG_INLINE void run(Kernel &kernel) + { + typedef typename Kernel::DstEvaluatorType::XprType DstXprType; - const Index outerSize = kernel.outerSize(); - for(Index outer = 0; outer < outerSize; ++outer) - copy_using_evaluator_DefaultTraversal_InnerUnrolling::run(kernel, outer); - } -}; - -/*************************** -*** Linear vectorization *** -***************************/ - - -// The goal of unaligned_dense_assignment_loop is simply to factorize the handling -// of the non vectorizable beginning and ending parts - -template -struct unaligned_dense_assignment_loop -{ - // if IsAligned = true, then do nothing - template - EIGEN_DEVICE_FUNC static EIGEN_STRONG_INLINE void run(Kernel&, Index, Index) {} -}; - -template <> -struct unaligned_dense_assignment_loop -{ - // MSVC must not inline this functions. If it does, it fails to optimize the - // packet access path. - // FIXME check which version exhibits this issue + const Index outerSize = kernel.outerSize(); + for (Index outer = 0; outer < outerSize; ++outer) + copy_using_evaluator_DefaultTraversal_InnerUnrolling::run( + kernel, outer); + } + }; + + /*************************** + *** Linear vectorization *** + ***************************/ + + + // The goal of unaligned_dense_assignment_loop is simply to factorize the handling + // of the non vectorizable beginning and ending parts + + template struct unaligned_dense_assignment_loop + { + // if IsAligned = true, then do nothing + template EIGEN_DEVICE_FUNC static EIGEN_STRONG_INLINE void run(Kernel &, Index, Index) {} + }; + + template<> struct unaligned_dense_assignment_loop + { + // MSVC must not inline this functions. If it does, it fails to optimize the + // packet access path. + // FIXME check which version exhibits this issue #if EIGEN_COMP_MSVC - template - static EIGEN_DONT_INLINE void run(Kernel &kernel, - Index start, - Index end) + template static EIGEN_DONT_INLINE void run(Kernel &kernel, Index start, Index end) #else - template - EIGEN_DEVICE_FUNC static EIGEN_STRONG_INLINE void run(Kernel &kernel, - Index start, - Index end) + template + EIGEN_DEVICE_FUNC static EIGEN_STRONG_INLINE void run(Kernel &kernel, Index start, Index end) #endif + { + for (Index index = start; index < end; ++index) kernel.assignCoeff(index); + } + }; + + template struct dense_assignment_loop { - for (Index index = start; index < end; ++index) - kernel.assignCoeff(index); - } -}; + EIGEN_DEVICE_FUNC static EIGEN_STRONG_INLINE void run(Kernel &kernel) + { + const Index size = kernel.size(); + typedef typename Kernel::Scalar Scalar; + typedef typename Kernel::PacketType PacketType; + enum { + requestedAlignment = Kernel::AssignmentTraits::LinearRequiredAlignment, + packetSize = unpacket_traits::size, + dstIsAligned = int(Kernel::AssignmentTraits::DstAlignment) >= int(requestedAlignment), + dstAlignment = packet_traits::AlignedOnScalar ? int(requestedAlignment) + : int(Kernel::AssignmentTraits::DstAlignment), + srcAlignment = Kernel::AssignmentTraits::JointAlignment + }; + const Index alignedStart = + dstIsAligned ? 0 : internal::first_aligned(kernel.dstDataPtr(), size); + const Index alignedEnd = alignedStart + ((size - alignedStart) / packetSize) * packetSize; + + unaligned_dense_assignment_loop::run(kernel, 0, alignedStart); + + for (Index index = alignedStart; index < alignedEnd; index += packetSize) + kernel.template assignPacket(index); + + unaligned_dense_assignment_loop<>::run(kernel, alignedEnd, size); + } + }; -template -struct dense_assignment_loop -{ - EIGEN_DEVICE_FUNC static EIGEN_STRONG_INLINE void run(Kernel &kernel) + template struct dense_assignment_loop { - const Index size = kernel.size(); - typedef typename Kernel::Scalar Scalar; - typedef typename Kernel::PacketType PacketType; - enum { - requestedAlignment = Kernel::AssignmentTraits::LinearRequiredAlignment, - packetSize = unpacket_traits::size, - dstIsAligned = int(Kernel::AssignmentTraits::DstAlignment)>=int(requestedAlignment), - dstAlignment = packet_traits::AlignedOnScalar ? int(requestedAlignment) - : int(Kernel::AssignmentTraits::DstAlignment), - srcAlignment = Kernel::AssignmentTraits::JointAlignment - }; - const Index alignedStart = dstIsAligned ? 0 : internal::first_aligned(kernel.dstDataPtr(), size); - const Index alignedEnd = alignedStart + ((size-alignedStart)/packetSize)*packetSize; + EIGEN_DEVICE_FUNC static EIGEN_STRONG_INLINE void run(Kernel &kernel) + { + typedef typename Kernel::DstEvaluatorType::XprType DstXprType; + typedef typename Kernel::PacketType PacketType; - unaligned_dense_assignment_loop::run(kernel, 0, alignedStart); + enum { + size = DstXprType::SizeAtCompileTime, + packetSize = unpacket_traits::size, + alignedSize = (size / packetSize) * packetSize + }; - for(Index index = alignedStart; index < alignedEnd; index += packetSize) - kernel.template assignPacket(index); + copy_using_evaluator_innervec_CompleteUnrolling::run(kernel); + copy_using_evaluator_DefaultTraversal_CompleteUnrolling::run(kernel); + } + }; - unaligned_dense_assignment_loop<>::run(kernel, alignedEnd, size); - } -}; + /************************** + *** Inner vectorization *** + **************************/ -template -struct dense_assignment_loop -{ - EIGEN_DEVICE_FUNC static EIGEN_STRONG_INLINE void run(Kernel &kernel) + template struct dense_assignment_loop { - typedef typename Kernel::DstEvaluatorType::XprType DstXprType; typedef typename Kernel::PacketType PacketType; - - enum { size = DstXprType::SizeAtCompileTime, - packetSize =unpacket_traits::size, - alignedSize = (size/packetSize)*packetSize }; + enum { + SrcAlignment = Kernel::AssignmentTraits::SrcAlignment, + DstAlignment = Kernel::AssignmentTraits::DstAlignment + }; + EIGEN_DEVICE_FUNC static EIGEN_STRONG_INLINE void run(Kernel &kernel) + { + const Index innerSize = kernel.innerSize(); + const Index outerSize = kernel.outerSize(); + const Index packetSize = unpacket_traits::size; + for (Index outer = 0; outer < outerSize; ++outer) + for (Index inner = 0; inner < innerSize; inner += packetSize) + kernel.template assignPacketByOuterInner(outer, inner); + } + }; - copy_using_evaluator_innervec_CompleteUnrolling::run(kernel); - copy_using_evaluator_DefaultTraversal_CompleteUnrolling::run(kernel); - } -}; - -/************************** -*** Inner vectorization *** -**************************/ - -template -struct dense_assignment_loop -{ - typedef typename Kernel::PacketType PacketType; - enum { - SrcAlignment = Kernel::AssignmentTraits::SrcAlignment, - DstAlignment = Kernel::AssignmentTraits::DstAlignment - }; - EIGEN_DEVICE_FUNC static EIGEN_STRONG_INLINE void run(Kernel &kernel) - { - const Index innerSize = kernel.innerSize(); - const Index outerSize = kernel.outerSize(); - const Index packetSize = unpacket_traits::size; - for(Index outer = 0; outer < outerSize; ++outer) - for(Index inner = 0; inner < innerSize; inner+=packetSize) - kernel.template assignPacketByOuterInner(outer, inner); - } -}; + template struct dense_assignment_loop + { + EIGEN_DEVICE_FUNC static EIGEN_STRONG_INLINE void run(Kernel &kernel) + { + typedef typename Kernel::DstEvaluatorType::XprType DstXprType; + copy_using_evaluator_innervec_CompleteUnrolling::run(kernel); + } + }; -template -struct dense_assignment_loop -{ - EIGEN_DEVICE_FUNC static EIGEN_STRONG_INLINE void run(Kernel &kernel) + template struct dense_assignment_loop { - typedef typename Kernel::DstEvaluatorType::XprType DstXprType; - copy_using_evaluator_innervec_CompleteUnrolling::run(kernel); - } -}; - -template -struct dense_assignment_loop -{ - EIGEN_DEVICE_FUNC static EIGEN_STRONG_INLINE void run(Kernel &kernel) - { - typedef typename Kernel::DstEvaluatorType::XprType DstXprType; - typedef typename Kernel::AssignmentTraits Traits; - const Index outerSize = kernel.outerSize(); - for(Index outer = 0; outer < outerSize; ++outer) - copy_using_evaluator_innervec_InnerUnrolling::run(kernel, outer); - } -}; + EIGEN_DEVICE_FUNC static EIGEN_STRONG_INLINE void run(Kernel &kernel) + { + typedef typename Kernel::DstEvaluatorType::XprType DstXprType; + typedef typename Kernel::AssignmentTraits Traits; + const Index outerSize = kernel.outerSize(); + for (Index outer = 0; outer < outerSize; ++outer) + copy_using_evaluator_innervec_InnerUnrolling::run(kernel, outer); + } + }; -/*********************** -*** Linear traversal *** -***********************/ + /*********************** + *** Linear traversal *** + ***********************/ -template -struct dense_assignment_loop -{ - EIGEN_DEVICE_FUNC static EIGEN_STRONG_INLINE void run(Kernel &kernel) + template struct dense_assignment_loop { - const Index size = kernel.size(); - for(Index i = 0; i < size; ++i) - kernel.assignCoeff(i); - } -}; + EIGEN_DEVICE_FUNC static EIGEN_STRONG_INLINE void run(Kernel &kernel) + { + const Index size = kernel.size(); + for (Index i = 0; i < size; ++i) kernel.assignCoeff(i); + } + }; -template -struct dense_assignment_loop -{ - EIGEN_DEVICE_FUNC static EIGEN_STRONG_INLINE void run(Kernel &kernel) + template struct dense_assignment_loop { - typedef typename Kernel::DstEvaluatorType::XprType DstXprType; - copy_using_evaluator_LinearTraversal_CompleteUnrolling::run(kernel); - } -}; + EIGEN_DEVICE_FUNC static EIGEN_STRONG_INLINE void run(Kernel &kernel) + { + typedef typename Kernel::DstEvaluatorType::XprType DstXprType; + copy_using_evaluator_LinearTraversal_CompleteUnrolling::run(kernel); + } + }; -/************************** -*** Slice vectorization *** -***************************/ + /************************** + *** Slice vectorization *** + ***************************/ -template -struct dense_assignment_loop -{ - EIGEN_DEVICE_FUNC static EIGEN_STRONG_INLINE void run(Kernel &kernel) + template struct dense_assignment_loop { - typedef typename Kernel::Scalar Scalar; - typedef typename Kernel::PacketType PacketType; - enum { - packetSize = unpacket_traits::size, - requestedAlignment = int(Kernel::AssignmentTraits::InnerRequiredAlignment), - alignable = packet_traits::AlignedOnScalar || int(Kernel::AssignmentTraits::DstAlignment)>=sizeof(Scalar), - dstIsAligned = int(Kernel::AssignmentTraits::DstAlignment)>=int(requestedAlignment), - dstAlignment = alignable ? int(requestedAlignment) - : int(Kernel::AssignmentTraits::DstAlignment) - }; - const Scalar *dst_ptr = kernel.dstDataPtr(); - if((!bool(dstIsAligned)) && (UIntPtr(dst_ptr) % sizeof(Scalar))>0) + EIGEN_DEVICE_FUNC static EIGEN_STRONG_INLINE void run(Kernel &kernel) + { + typedef typename Kernel::Scalar Scalar; + typedef typename Kernel::PacketType PacketType; + enum { + packetSize = unpacket_traits::size, + requestedAlignment = int(Kernel::AssignmentTraits::InnerRequiredAlignment), + alignable = + packet_traits::AlignedOnScalar || int(Kernel::AssignmentTraits::DstAlignment) >= sizeof(Scalar), + dstIsAligned = int(Kernel::AssignmentTraits::DstAlignment) >= int(requestedAlignment), + dstAlignment = alignable ? int(requestedAlignment) : int(Kernel::AssignmentTraits::DstAlignment) + }; + const Scalar *dst_ptr = kernel.dstDataPtr(); + if ((!bool(dstIsAligned)) && (UIntPtr(dst_ptr) % sizeof(Scalar)) > 0) { + // the pointer is not aligend-on scalar, so alignment is not possible + return dense_assignment_loop::run(kernel); + } + const Index packetAlignedMask = packetSize - 1; + const Index innerSize = kernel.innerSize(); + const Index outerSize = kernel.outerSize(); + const Index alignedStep = alignable ? (packetSize - kernel.outerStride() % packetSize) & packetAlignedMask : 0; + Index alignedStart = + ((!alignable) || bool(dstIsAligned)) ? 0 : internal::first_aligned(dst_ptr, innerSize); + + for (Index outer = 0; outer < outerSize; ++outer) { + const Index alignedEnd = alignedStart + ((innerSize - alignedStart) & ~packetAlignedMask); + // do the non-vectorizable part of the assignment + for (Index inner = 0; inner < alignedStart; ++inner) kernel.assignCoeffByOuterInner(outer, inner); + + // do the vectorizable part of the assignment + for (Index inner = alignedStart; inner < alignedEnd; inner += packetSize) + kernel.template assignPacketByOuterInner(outer, inner); + + // do the non-vectorizable part of the assignment + for (Index inner = alignedEnd; inner < innerSize; ++inner) kernel.assignCoeffByOuterInner(outer, inner); + + alignedStart = numext::mini((alignedStart + alignedStep) % packetSize, innerSize); + } + } + }; + +#if EIGEN_UNALIGNED_VECTORIZE + template struct dense_assignment_loop + { + EIGEN_DEVICE_FUNC static EIGEN_STRONG_INLINE void run(Kernel &kernel) { - // the pointer is not aligend-on scalar, so alignment is not possible - return dense_assignment_loop::run(kernel); + typedef typename Kernel::DstEvaluatorType::XprType DstXprType; + typedef typename Kernel::PacketType PacketType; + + enum { + size = DstXprType::InnerSizeAtCompileTime, + packetSize = unpacket_traits::size, + vectorizableSize = (size / packetSize) * packetSize + }; + + for (Index outer = 0; outer < kernel.outerSize(); ++outer) { + copy_using_evaluator_innervec_InnerUnrolling::run(kernel, outer); + copy_using_evaluator_DefaultTraversal_InnerUnrolling::run(kernel, outer); + } } - const Index packetAlignedMask = packetSize - 1; - const Index innerSize = kernel.innerSize(); - const Index outerSize = kernel.outerSize(); - const Index alignedStep = alignable ? (packetSize - kernel.outerStride() % packetSize) & packetAlignedMask : 0; - Index alignedStart = ((!alignable) || bool(dstIsAligned)) ? 0 : internal::first_aligned(dst_ptr, innerSize); + }; +#endif - for(Index outer = 0; outer < outerSize; ++outer) + + /*************************************************************************** + * Part 4 : Generic dense assignment kernel + ***************************************************************************/ + + // This class generalize the assignment of a coefficient (or packet) from one dense evaluator + // to another dense writable evaluator. + // It is parametrized by the two evaluators, and the actual assignment functor. + // This abstraction level permits to keep the evaluation loops as simple and as generic as possible. + // One can customize the assignment using this generic dense_assignment_kernel with different + // functors, or by completely overloading it, by-passing a functor. + template + class generic_dense_assignment_kernel + { + protected: + typedef typename DstEvaluatorTypeT::XprType DstXprType; + typedef typename SrcEvaluatorTypeT::XprType SrcXprType; + + public: + typedef DstEvaluatorTypeT DstEvaluatorType; + typedef SrcEvaluatorTypeT SrcEvaluatorType; + typedef typename DstEvaluatorType::Scalar Scalar; + typedef copy_using_evaluator_traits AssignmentTraits; + typedef typename AssignmentTraits::PacketType PacketType; + + + EIGEN_DEVICE_FUNC generic_dense_assignment_kernel(DstEvaluatorType &dst, + const SrcEvaluatorType &src, + const Functor &func, + DstXprType &dstExpr) + : m_dst(dst), m_src(src), m_functor(func), m_dstExpr(dstExpr) { - const Index alignedEnd = alignedStart + ((innerSize-alignedStart) & ~packetAlignedMask); - // do the non-vectorizable part of the assignment - for(Index inner = 0; inner(outer, inner); + EIGEN_DEVICE_FUNC Index size() const { return m_dstExpr.size(); } + EIGEN_DEVICE_FUNC Index innerSize() const { return m_dstExpr.innerSize(); } + EIGEN_DEVICE_FUNC Index outerSize() const { return m_dstExpr.outerSize(); } + EIGEN_DEVICE_FUNC Index rows() const { return m_dstExpr.rows(); } + EIGEN_DEVICE_FUNC Index cols() const { return m_dstExpr.cols(); } + EIGEN_DEVICE_FUNC Index outerStride() const { return m_dstExpr.outerStride(); } - // do the non-vectorizable part of the assignment - for(Index inner = alignedEnd; inner -struct dense_assignment_loop -{ - EIGEN_DEVICE_FUNC static EIGEN_STRONG_INLINE void run(Kernel &kernel) - { - typedef typename Kernel::DstEvaluatorType::XprType DstXprType; - typedef typename Kernel::PacketType PacketType; + /// \sa assignCoeff(Index,Index) + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void assignCoeff(Index index) + { + m_functor.assignCoeff(m_dst.coeffRef(index), m_src.coeff(index)); + } + + /// \sa assignCoeff(Index,Index) + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void assignCoeffByOuterInner(Index outer, Index inner) + { + Index row = rowIndexByOuterInner(outer, inner); + Index col = colIndexByOuterInner(outer, inner); + assignCoeff(row, col); + } - enum { size = DstXprType::InnerSizeAtCompileTime, - packetSize =unpacket_traits::size, - vectorizableSize = (size/packetSize)*packetSize }; - for(Index outer = 0; outer < kernel.outerSize(); ++outer) + template + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void assignPacket(Index row, Index col) { - copy_using_evaluator_innervec_InnerUnrolling::run(kernel, outer); - copy_using_evaluator_DefaultTraversal_InnerUnrolling::run(kernel, outer); + m_functor.template assignPacket( + &m_dst.coeffRef(row, col), m_src.template packet(row, col)); } - } -}; -#endif + template + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void assignPacket(Index index) + { + m_functor.template assignPacket( + &m_dst.coeffRef(index), m_src.template packet(index)); + } -/*************************************************************************** -* Part 4 : Generic dense assignment kernel -***************************************************************************/ - -// This class generalize the assignment of a coefficient (or packet) from one dense evaluator -// to another dense writable evaluator. -// It is parametrized by the two evaluators, and the actual assignment functor. -// This abstraction level permits to keep the evaluation loops as simple and as generic as possible. -// One can customize the assignment using this generic dense_assignment_kernel with different -// functors, or by completely overloading it, by-passing a functor. -template -class generic_dense_assignment_kernel -{ -protected: - typedef typename DstEvaluatorTypeT::XprType DstXprType; - typedef typename SrcEvaluatorTypeT::XprType SrcXprType; -public: - - typedef DstEvaluatorTypeT DstEvaluatorType; - typedef SrcEvaluatorTypeT SrcEvaluatorType; - typedef typename DstEvaluatorType::Scalar Scalar; - typedef copy_using_evaluator_traits AssignmentTraits; - typedef typename AssignmentTraits::PacketType PacketType; - - - EIGEN_DEVICE_FUNC generic_dense_assignment_kernel(DstEvaluatorType &dst, const SrcEvaluatorType &src, const Functor &func, DstXprType& dstExpr) - : m_dst(dst), m_src(src), m_functor(func), m_dstExpr(dstExpr) - { - #ifdef EIGEN_DEBUG_ASSIGN - AssignmentTraits::debug(); - #endif - } - - EIGEN_DEVICE_FUNC Index size() const { return m_dstExpr.size(); } - EIGEN_DEVICE_FUNC Index innerSize() const { return m_dstExpr.innerSize(); } - EIGEN_DEVICE_FUNC Index outerSize() const { return m_dstExpr.outerSize(); } - EIGEN_DEVICE_FUNC Index rows() const { return m_dstExpr.rows(); } - EIGEN_DEVICE_FUNC Index cols() const { return m_dstExpr.cols(); } - EIGEN_DEVICE_FUNC Index outerStride() const { return m_dstExpr.outerStride(); } - - EIGEN_DEVICE_FUNC DstEvaluatorType& dstEvaluator() { return m_dst; } - EIGEN_DEVICE_FUNC const SrcEvaluatorType& srcEvaluator() const { return m_src; } - - /// Assign src(row,col) to dst(row,col) through the assignment functor. - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void assignCoeff(Index row, Index col) - { - m_functor.assignCoeff(m_dst.coeffRef(row,col), m_src.coeff(row,col)); - } - - /// \sa assignCoeff(Index,Index) - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void assignCoeff(Index index) + template + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void assignPacketByOuterInner(Index outer, Index inner) + { + Index row = rowIndexByOuterInner(outer, inner); + Index col = colIndexByOuterInner(outer, inner); + assignPacket(row, col); + } + + EIGEN_DEVICE_FUNC static EIGEN_STRONG_INLINE Index rowIndexByOuterInner(Index outer, Index inner) + { + typedef typename DstEvaluatorType::ExpressionTraits Traits; + return int(Traits::RowsAtCompileTime) == 1 ? 0 + : int(Traits::ColsAtCompileTime) == 1 ? inner + : int(DstEvaluatorType::Flags) & RowMajorBit ? outer + : inner; + } + + EIGEN_DEVICE_FUNC static EIGEN_STRONG_INLINE Index colIndexByOuterInner(Index outer, Index inner) + { + typedef typename DstEvaluatorType::ExpressionTraits Traits; + return int(Traits::ColsAtCompileTime) == 1 ? 0 + : int(Traits::RowsAtCompileTime) == 1 ? inner + : int(DstEvaluatorType::Flags) & RowMajorBit ? inner + : outer; + } + + EIGEN_DEVICE_FUNC const Scalar *dstDataPtr() const { return m_dstExpr.data(); } + + protected: + DstEvaluatorType &m_dst; + const SrcEvaluatorType &m_src; + const Functor &m_functor; + // TODO find a way to avoid the needs of the original expression + DstXprType &m_dstExpr; + }; + + /*************************************************************************** + * Part 5 : Entry point for dense rectangular assignment + ***************************************************************************/ + + template + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void + resize_if_allowed(DstXprType &dst, const SrcXprType &src, const Functor & /*func*/) { - m_functor.assignCoeff(m_dst.coeffRef(index), m_src.coeff(index)); + EIGEN_ONLY_USED_FOR_DEBUG(dst); + EIGEN_ONLY_USED_FOR_DEBUG(src); + eigen_assert(dst.rows() == src.rows() && dst.cols() == src.cols()); } - - /// \sa assignCoeff(Index,Index) - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void assignCoeffByOuterInner(Index outer, Index inner) + + template + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void + resize_if_allowed(DstXprType &dst, const SrcXprType &src, const internal::assign_op & /*func*/) { - Index row = rowIndexByOuterInner(outer, inner); - Index col = colIndexByOuterInner(outer, inner); - assignCoeff(row, col); + Index dstRows = src.rows(); + Index dstCols = src.cols(); + if (((dst.rows() != dstRows) || (dst.cols() != dstCols))) dst.resize(dstRows, dstCols); + eigen_assert(dst.rows() == dstRows && dst.cols() == dstCols); } - - - template - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void assignPacket(Index row, Index col) + + template + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void + call_dense_assignment_loop(DstXprType &dst, const SrcXprType &src, const Functor &func) { - m_functor.template assignPacket(&m_dst.coeffRef(row,col), m_src.template packet(row,col)); + typedef evaluator DstEvaluatorType; + typedef evaluator SrcEvaluatorType; + + SrcEvaluatorType srcEvaluator(src); + + // NOTE To properly handle A = (A*A.transpose())/s with A rectangular, + // we need to resize the destination after the source evaluator has been created. + resize_if_allowed(dst, src, func); + + DstEvaluatorType dstEvaluator(dst); + + typedef generic_dense_assignment_kernel Kernel; + Kernel kernel(dstEvaluator, srcEvaluator, func, dst.const_cast_derived()); + + dense_assignment_loop::run(kernel); } - - template - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void assignPacket(Index index) + + template + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void call_dense_assignment_loop(DstXprType &dst, const SrcXprType &src) { - m_functor.template assignPacket(&m_dst.coeffRef(index), m_src.template packet(index)); + call_dense_assignment_loop( + dst, src, internal::assign_op()); } - - template - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void assignPacketByOuterInner(Index outer, Index inner) + + /*************************************************************************** + * Part 6 : Generic assignment + ***************************************************************************/ + + // Based on the respective shapes of the destination and source, + // the class AssignmentKind determine the kind of assignment mechanism. + // AssignmentKind must define a Kind typedef. + template struct AssignmentKind; + + // Assignement kind defined in this file: + struct Dense2Dense + { + }; + struct EigenBase2EigenBase + { + }; + + template struct AssignmentKind + { + typedef EigenBase2EigenBase Kind; + }; + template<> struct AssignmentKind + { + typedef Dense2Dense Kind; + }; + + // This is the main assignment class + template::Shape, + typename evaluator_traits::Shape>::Kind, + typename EnableIf = void> + struct Assignment; + + + // The only purpose of this call_assignment() function is to deal with noalias() / "assume-aliasing" and automatic + // transposition. Indeed, I (Gael) think that this concept of "assume-aliasing" was a mistake, and it makes thing + // quite complicated. So this intermediate function removes everything related to "assume-aliasing" such that + // Assignment does not has to bother about these annoying details. + + template + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void call_assignment(Dst &dst, const Src &src) { - Index row = rowIndexByOuterInner(outer, inner); - Index col = colIndexByOuterInner(outer, inner); - assignPacket(row, col); + call_assignment(dst, src, internal::assign_op()); } - - EIGEN_DEVICE_FUNC static EIGEN_STRONG_INLINE Index rowIndexByOuterInner(Index outer, Index inner) - { - typedef typename DstEvaluatorType::ExpressionTraits Traits; - return int(Traits::RowsAtCompileTime) == 1 ? 0 - : int(Traits::ColsAtCompileTime) == 1 ? inner - : int(DstEvaluatorType::Flags)&RowMajorBit ? outer - : inner; + template + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void call_assignment(const Dst &dst, const Src &src) + { + call_assignment(dst, src, internal::assign_op()); } - EIGEN_DEVICE_FUNC static EIGEN_STRONG_INLINE Index colIndexByOuterInner(Index outer, Index inner) + // Deal with "assume-aliasing" + template + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void call_assignment(Dst &dst, + const Src &src, + const Func &func, + typename enable_if::value, void *>::type = 0) { - typedef typename DstEvaluatorType::ExpressionTraits Traits; - return int(Traits::ColsAtCompileTime) == 1 ? 0 - : int(Traits::RowsAtCompileTime) == 1 ? inner - : int(DstEvaluatorType::Flags)&RowMajorBit ? inner - : outer; + typename plain_matrix_type::type tmp(src); + call_assignment_no_alias(dst, tmp, func); } - EIGEN_DEVICE_FUNC const Scalar* dstDataPtr() const + template + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void call_assignment(Dst &dst, + const Src &src, + const Func &func, + typename enable_if::value, void *>::type = 0) { - return m_dstExpr.data(); + call_assignment_no_alias(dst, src, func); } - -protected: - DstEvaluatorType& m_dst; - const SrcEvaluatorType& m_src; - const Functor &m_functor; - // TODO find a way to avoid the needs of the original expression - DstXprType& m_dstExpr; -}; - -/*************************************************************************** -* Part 5 : Entry point for dense rectangular assignment -***************************************************************************/ - -template -EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE -void resize_if_allowed(DstXprType &dst, const SrcXprType& src, const Functor &/*func*/) -{ - EIGEN_ONLY_USED_FOR_DEBUG(dst); - EIGEN_ONLY_USED_FOR_DEBUG(src); - eigen_assert(dst.rows() == src.rows() && dst.cols() == src.cols()); -} - -template -EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE -void resize_if_allowed(DstXprType &dst, const SrcXprType& src, const internal::assign_op &/*func*/) -{ - Index dstRows = src.rows(); - Index dstCols = src.cols(); - if(((dst.rows()!=dstRows) || (dst.cols()!=dstCols))) - dst.resize(dstRows, dstCols); - eigen_assert(dst.rows() == dstRows && dst.cols() == dstCols); -} - -template -EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void call_dense_assignment_loop(DstXprType& dst, const SrcXprType& src, const Functor &func) -{ - typedef evaluator DstEvaluatorType; - typedef evaluator SrcEvaluatorType; - - SrcEvaluatorType srcEvaluator(src); - - // NOTE To properly handle A = (A*A.transpose())/s with A rectangular, - // we need to resize the destination after the source evaluator has been created. - resize_if_allowed(dst, src, func); - - DstEvaluatorType dstEvaluator(dst); - - typedef generic_dense_assignment_kernel Kernel; - Kernel kernel(dstEvaluator, srcEvaluator, func, dst.const_cast_derived()); - - dense_assignment_loop::run(kernel); -} - -template -EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void call_dense_assignment_loop(DstXprType& dst, const SrcXprType& src) -{ - call_dense_assignment_loop(dst, src, internal::assign_op()); -} - -/*************************************************************************** -* Part 6 : Generic assignment -***************************************************************************/ - -// Based on the respective shapes of the destination and source, -// the class AssignmentKind determine the kind of assignment mechanism. -// AssignmentKind must define a Kind typedef. -template struct AssignmentKind; - -// Assignement kind defined in this file: -struct Dense2Dense {}; -struct EigenBase2EigenBase {}; - -template struct AssignmentKind { typedef EigenBase2EigenBase Kind; }; -template<> struct AssignmentKind { typedef Dense2Dense Kind; }; - -// This is the main assignment class -template< typename DstXprType, typename SrcXprType, typename Functor, - typename Kind = typename AssignmentKind< typename evaluator_traits::Shape , typename evaluator_traits::Shape >::Kind, - typename EnableIf = void> -struct Assignment; - - -// The only purpose of this call_assignment() function is to deal with noalias() / "assume-aliasing" and automatic transposition. -// Indeed, I (Gael) think that this concept of "assume-aliasing" was a mistake, and it makes thing quite complicated. -// So this intermediate function removes everything related to "assume-aliasing" such that Assignment -// does not has to bother about these annoying details. - -template -EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE -void call_assignment(Dst& dst, const Src& src) -{ - call_assignment(dst, src, internal::assign_op()); -} -template -EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE -void call_assignment(const Dst& dst, const Src& src) -{ - call_assignment(dst, src, internal::assign_op()); -} - -// Deal with "assume-aliasing" -template -EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE -void call_assignment(Dst& dst, const Src& src, const Func& func, typename enable_if< evaluator_assume_aliasing::value, void*>::type = 0) -{ - typename plain_matrix_type::type tmp(src); - call_assignment_no_alias(dst, tmp, func); -} - -template -EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE -void call_assignment(Dst& dst, const Src& src, const Func& func, typename enable_if::value, void*>::type = 0) -{ - call_assignment_no_alias(dst, src, func); -} - -// by-pass "assume-aliasing" -// When there is no aliasing, we require that 'dst' has been properly resized -template class StorageBase, typename Src, typename Func> -EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE -void call_assignment(NoAlias& dst, const Src& src, const Func& func) -{ - call_assignment_no_alias(dst.expression(), src, func); -} - - -template -EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE -void call_assignment_no_alias(Dst& dst, const Src& src, const Func& func) -{ - enum { - NeedToTranspose = ( (int(Dst::RowsAtCompileTime) == 1 && int(Src::ColsAtCompileTime) == 1) - || (int(Dst::ColsAtCompileTime) == 1 && int(Src::RowsAtCompileTime) == 1) - ) && int(Dst::SizeAtCompileTime) != 1 - }; - - typedef typename internal::conditional, Dst>::type ActualDstTypeCleaned; - typedef typename internal::conditional, Dst&>::type ActualDstType; - ActualDstType actualDst(dst); - - // TODO check whether this is the right place to perform these checks: - EIGEN_STATIC_ASSERT_LVALUE(Dst) - EIGEN_STATIC_ASSERT_SAME_MATRIX_SIZE(ActualDstTypeCleaned,Src) - EIGEN_CHECK_BINARY_COMPATIBILIY(Func,typename ActualDstTypeCleaned::Scalar,typename Src::Scalar); - - Assignment::run(actualDst, src, func); -} -template -EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE -void call_assignment_no_alias(Dst& dst, const Src& src) -{ - call_assignment_no_alias(dst, src, internal::assign_op()); -} - -template -EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE -void call_assignment_no_alias_no_transpose(Dst& dst, const Src& src, const Func& func) -{ - // TODO check whether this is the right place to perform these checks: - EIGEN_STATIC_ASSERT_LVALUE(Dst) - EIGEN_STATIC_ASSERT_SAME_MATRIX_SIZE(Dst,Src) - EIGEN_CHECK_BINARY_COMPATIBILIY(Func,typename Dst::Scalar,typename Src::Scalar); - - Assignment::run(dst, src, func); -} -template -EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE -void call_assignment_no_alias_no_transpose(Dst& dst, const Src& src) -{ - call_assignment_no_alias_no_transpose(dst, src, internal::assign_op()); -} - -// forward declaration -template void check_for_aliasing(const Dst &dst, const Src &src); - -// Generic Dense to Dense assignment -// Note that the last template argument "Weak" is needed to make it possible to perform -// both partial specialization+SFINAE without ambiguous specialization -template< typename DstXprType, typename SrcXprType, typename Functor, typename Weak> -struct Assignment -{ - EIGEN_DEVICE_FUNC - static EIGEN_STRONG_INLINE void run(DstXprType &dst, const SrcXprType &src, const Functor &func) + + // by-pass "assume-aliasing" + // When there is no aliasing, we require that 'dst' has been properly resized + template class StorageBase, typename Src, typename Func> + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void + call_assignment(NoAlias &dst, const Src &src, const Func &func) { -#ifndef EIGEN_NO_DEBUG - internal::check_for_aliasing(dst, src); -#endif - - call_dense_assignment_loop(dst, src, func); + call_assignment_no_alias(dst.expression(), src, func); } -}; - -// Generic assignment through evalTo. -// TODO: not sure we have to keep that one, but it helps porting current code to new evaluator mechanism. -// Note that the last template argument "Weak" is needed to make it possible to perform -// both partial specialization+SFINAE without ambiguous specialization -template< typename DstXprType, typename SrcXprType, typename Functor, typename Weak> -struct Assignment -{ - EIGEN_DEVICE_FUNC - static EIGEN_STRONG_INLINE void run(DstXprType &dst, const SrcXprType &src, const internal::assign_op &/*func*/) + + + template + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void call_assignment_no_alias(Dst &dst, const Src &src, const Func &func) { - Index dstRows = src.rows(); - Index dstCols = src.cols(); - if((dst.rows()!=dstRows) || (dst.cols()!=dstCols)) - dst.resize(dstRows, dstCols); + enum { + NeedToTranspose = ((int(Dst::RowsAtCompileTime) == 1 && int(Src::ColsAtCompileTime) == 1) + || (int(Dst::ColsAtCompileTime) == 1 && int(Src::RowsAtCompileTime) == 1)) + && int(Dst::SizeAtCompileTime) != 1 + }; - eigen_assert(dst.rows() == src.rows() && dst.cols() == src.cols()); - src.evalTo(dst); + typedef typename internal::conditional, Dst>::type ActualDstTypeCleaned; + typedef typename internal::conditional, Dst &>::type ActualDstType; + ActualDstType actualDst(dst); + + // TODO check whether this is the right place to perform these checks: + EIGEN_STATIC_ASSERT_LVALUE(Dst) + EIGEN_STATIC_ASSERT_SAME_MATRIX_SIZE(ActualDstTypeCleaned, Src) + EIGEN_CHECK_BINARY_COMPATIBILIY(Func, typename ActualDstTypeCleaned::Scalar, typename Src::Scalar); + + Assignment::run(actualDst, src, func); + } + template + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void call_assignment_no_alias(Dst &dst, const Src &src) + { + call_assignment_no_alias(dst, src, internal::assign_op()); } - // NOTE The following two functions are templated to avoid their instanciation if not needed - // This is needed because some expressions supports evalTo only and/or have 'void' as scalar type. - template - EIGEN_DEVICE_FUNC - static EIGEN_STRONG_INLINE void run(DstXprType &dst, const SrcXprType &src, const internal::add_assign_op &/*func*/) + template + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void + call_assignment_no_alias_no_transpose(Dst &dst, const Src &src, const Func &func) { - Index dstRows = src.rows(); - Index dstCols = src.cols(); - if((dst.rows()!=dstRows) || (dst.cols()!=dstCols)) - dst.resize(dstRows, dstCols); + // TODO check whether this is the right place to perform these checks: + EIGEN_STATIC_ASSERT_LVALUE(Dst) + EIGEN_STATIC_ASSERT_SAME_MATRIX_SIZE(Dst, Src) + EIGEN_CHECK_BINARY_COMPATIBILIY(Func, typename Dst::Scalar, typename Src::Scalar); - eigen_assert(dst.rows() == src.rows() && dst.cols() == src.cols()); - src.addTo(dst); + Assignment::run(dst, src, func); + } + template + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void call_assignment_no_alias_no_transpose(Dst &dst, const Src &src) + { + call_assignment_no_alias_no_transpose(dst, src, internal::assign_op()); } - template - EIGEN_DEVICE_FUNC - static EIGEN_STRONG_INLINE void run(DstXprType &dst, const SrcXprType &src, const internal::sub_assign_op &/*func*/) + // forward declaration + template void check_for_aliasing(const Dst &dst, const Src &src); + + // Generic Dense to Dense assignment + // Note that the last template argument "Weak" is needed to make it possible to perform + // both partial specialization+SFINAE without ambiguous specialization + template + struct Assignment { - Index dstRows = src.rows(); - Index dstCols = src.cols(); - if((dst.rows()!=dstRows) || (dst.cols()!=dstCols)) - dst.resize(dstRows, dstCols); + EIGEN_DEVICE_FUNC + static EIGEN_STRONG_INLINE void run(DstXprType &dst, const SrcXprType &src, const Functor &func) + { +#ifndef EIGEN_NO_DEBUG + internal::check_for_aliasing(dst, src); +#endif - eigen_assert(dst.rows() == src.rows() && dst.cols() == src.cols()); - src.subTo(dst); - } -}; + call_dense_assignment_loop(dst, src, func); + } + }; + + // Generic assignment through evalTo. + // TODO: not sure we have to keep that one, but it helps porting current code to new evaluator mechanism. + // Note that the last template argument "Weak" is needed to make it possible to perform + // both partial specialization+SFINAE without ambiguous specialization + template + struct Assignment + { + EIGEN_DEVICE_FUNC + static EIGEN_STRONG_INLINE void run(DstXprType &dst, + const SrcXprType &src, + const internal::assign_op & /*func*/) + { + Index dstRows = src.rows(); + Index dstCols = src.cols(); + if ((dst.rows() != dstRows) || (dst.cols() != dstCols)) dst.resize(dstRows, dstCols); + + eigen_assert(dst.rows() == src.rows() && dst.cols() == src.cols()); + src.evalTo(dst); + } + + // NOTE The following two functions are templated to avoid their instanciation if not needed + // This is needed because some expressions supports evalTo only and/or have 'void' as scalar type. + template + EIGEN_DEVICE_FUNC static EIGEN_STRONG_INLINE void run(DstXprType &dst, + const SrcXprType &src, + const internal::add_assign_op & /*func*/) + { + Index dstRows = src.rows(); + Index dstCols = src.cols(); + if ((dst.rows() != dstRows) || (dst.cols() != dstCols)) dst.resize(dstRows, dstCols); + + eigen_assert(dst.rows() == src.rows() && dst.cols() == src.cols()); + src.addTo(dst); + } + + template + EIGEN_DEVICE_FUNC static EIGEN_STRONG_INLINE void run(DstXprType &dst, + const SrcXprType &src, + const internal::sub_assign_op & /*func*/) + { + Index dstRows = src.rows(); + Index dstCols = src.cols(); + if ((dst.rows() != dstRows) || (dst.cols() != dstCols)) dst.resize(dstRows, dstCols); + + eigen_assert(dst.rows() == src.rows() && dst.cols() == src.cols()); + src.subTo(dst); + } + }; -} // namespace internal +}// namespace internal -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_ASSIGN_EVALUATOR_H +#endif// EIGEN_ASSIGN_EVALUATOR_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/Assign_MKL.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/Assign_MKL.h old mode 100755 new mode 100644 index 6866095b..28e083b1 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/Assign_MKL.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/Assign_MKL.h @@ -1,7 +1,7 @@ /* Copyright (c) 2011, Intel Corporation. All rights reserved. Copyright (C) 2015 Gael Guennebaud - + Redistribution and use in source and binary forms, with or without modification, are permitted provided that the following conditions are met: @@ -34,145 +34,163 @@ #ifndef EIGEN_ASSIGN_VML_H #define EIGEN_ASSIGN_VML_H -namespace Eigen { +namespace Eigen { namespace internal { -template -class vml_assign_traits -{ + template class vml_assign_traits + { private: enum { DstHasDirectAccess = Dst::Flags & DirectAccessBit, SrcHasDirectAccess = Src::Flags & DirectAccessBit, StorageOrdersAgree = (int(Dst::IsRowMajor) == int(Src::IsRowMajor)), InnerSize = int(Dst::IsVectorAtCompileTime) ? int(Dst::SizeAtCompileTime) - : int(Dst::Flags)&RowMajorBit ? int(Dst::ColsAtCompileTime) - : int(Dst::RowsAtCompileTime), - InnerMaxSize = int(Dst::IsVectorAtCompileTime) ? int(Dst::MaxSizeAtCompileTime) - : int(Dst::Flags)&RowMajorBit ? int(Dst::MaxColsAtCompileTime) - : int(Dst::MaxRowsAtCompileTime), + : int(Dst::Flags) & RowMajorBit ? int(Dst::ColsAtCompileTime) + : int(Dst::RowsAtCompileTime), + InnerMaxSize = int(Dst::IsVectorAtCompileTime) ? int(Dst::MaxSizeAtCompileTime) + : int(Dst::Flags) & RowMajorBit ? int(Dst::MaxColsAtCompileTime) + : int(Dst::MaxRowsAtCompileTime), MaxSizeAtCompileTime = Dst::SizeAtCompileTime, - MightEnableVml = StorageOrdersAgree && DstHasDirectAccess && SrcHasDirectAccess && Src::InnerStrideAtCompileTime==1 && Dst::InnerStrideAtCompileTime==1, + MightEnableVml = StorageOrdersAgree && DstHasDirectAccess && SrcHasDirectAccess + && Src::InnerStrideAtCompileTime == 1 && Dst::InnerStrideAtCompileTime == 1, MightLinearize = MightEnableVml && (int(Dst::Flags) & int(Src::Flags) & LinearAccessBit), VmlSize = MightLinearize ? MaxSizeAtCompileTime : InnerMaxSize, - LargeEnough = VmlSize==Dynamic || VmlSize>=EIGEN_MKL_VML_THRESHOLD + LargeEnough = VmlSize == Dynamic || VmlSize >= EIGEN_MKL_VML_THRESHOLD }; + public: - enum { - EnableVml = MightEnableVml && LargeEnough, - Traversal = MightLinearize ? LinearTraversal : DefaultTraversal - }; -}; + enum { EnableVml = MightEnableVml && LargeEnough, Traversal = MightLinearize ? LinearTraversal : DefaultTraversal }; + }; #define EIGEN_PP_EXPAND(ARG) ARG -#if !defined (EIGEN_FAST_MATH) || (EIGEN_FAST_MATH != 1) +#if !defined(EIGEN_FAST_MATH) || (EIGEN_FAST_MATH != 1) #define EIGEN_VMLMODE_EXPAND_LA , VML_HA #else #define EIGEN_VMLMODE_EXPAND_LA , VML_LA #endif -#define EIGEN_VMLMODE_EXPAND__ +#define EIGEN_VMLMODE_EXPAND__ #define EIGEN_VMLMODE_PREFIX_LA vm -#define EIGEN_VMLMODE_PREFIX__ v -#define EIGEN_VMLMODE_PREFIX(VMLMODE) EIGEN_CAT(EIGEN_VMLMODE_PREFIX_,VMLMODE) - -#define EIGEN_MKL_VML_DECLARE_UNARY_CALL(EIGENOP, VMLOP, EIGENTYPE, VMLTYPE, VMLMODE) \ - template< typename DstXprType, typename SrcXprNested> \ - struct Assignment, SrcXprNested>, assign_op, \ - Dense2Dense, typename enable_if::EnableVml>::type> { \ - typedef CwiseUnaryOp, SrcXprNested> SrcXprType; \ - static void run(DstXprType &dst, const SrcXprType &src, const assign_op &func) { \ - resize_if_allowed(dst, src, func); \ - eigen_assert(dst.rows() == src.rows() && dst.cols() == src.cols()); \ - if(vml_assign_traits::Traversal==LinearTraversal) { \ - VMLOP(dst.size(), (const VMLTYPE*)src.nestedExpression().data(), \ - (VMLTYPE*)dst.data() EIGEN_PP_EXPAND(EIGEN_VMLMODE_EXPAND_##VMLMODE) ); \ - } else { \ - const Index outerSize = dst.outerSize(); \ - for(Index outer = 0; outer < outerSize; ++outer) { \ - const EIGENTYPE *src_ptr = src.IsRowMajor ? &(src.nestedExpression().coeffRef(outer,0)) : \ - &(src.nestedExpression().coeffRef(0, outer)); \ - EIGENTYPE *dst_ptr = dst.IsRowMajor ? &(dst.coeffRef(outer,0)) : &(dst.coeffRef(0, outer)); \ - VMLOP( dst.innerSize(), (const VMLTYPE*)src_ptr, \ - (VMLTYPE*)dst_ptr EIGEN_PP_EXPAND(EIGEN_VMLMODE_EXPAND_##VMLMODE)); \ - } \ - } \ - } \ - }; \ - - -#define EIGEN_MKL_VML_DECLARE_UNARY_CALLS_REAL(EIGENOP, VMLOP, VMLMODE) \ - EIGEN_MKL_VML_DECLARE_UNARY_CALL(EIGENOP, EIGEN_CAT(EIGEN_VMLMODE_PREFIX(VMLMODE),s##VMLOP), float, float, VMLMODE) \ - EIGEN_MKL_VML_DECLARE_UNARY_CALL(EIGENOP, EIGEN_CAT(EIGEN_VMLMODE_PREFIX(VMLMODE),d##VMLOP), double, double, VMLMODE) - -#define EIGEN_MKL_VML_DECLARE_UNARY_CALLS_CPLX(EIGENOP, VMLOP, VMLMODE) \ - EIGEN_MKL_VML_DECLARE_UNARY_CALL(EIGENOP, EIGEN_CAT(EIGEN_VMLMODE_PREFIX(VMLMODE),c##VMLOP), scomplex, MKL_Complex8, VMLMODE) \ - EIGEN_MKL_VML_DECLARE_UNARY_CALL(EIGENOP, EIGEN_CAT(EIGEN_VMLMODE_PREFIX(VMLMODE),z##VMLOP), dcomplex, MKL_Complex16, VMLMODE) - -#define EIGEN_MKL_VML_DECLARE_UNARY_CALLS(EIGENOP, VMLOP, VMLMODE) \ - EIGEN_MKL_VML_DECLARE_UNARY_CALLS_REAL(EIGENOP, VMLOP, VMLMODE) \ +#define EIGEN_VMLMODE_PREFIX__ v +#define EIGEN_VMLMODE_PREFIX(VMLMODE) EIGEN_CAT(EIGEN_VMLMODE_PREFIX_, VMLMODE) + +#define EIGEN_MKL_VML_DECLARE_UNARY_CALL(EIGENOP, VMLOP, EIGENTYPE, VMLTYPE, VMLMODE) \ + template \ + struct Assignment, SrcXprNested>, \ + assign_op, \ + Dense2Dense, \ + typename enable_if::EnableVml>::type> \ + { \ + typedef CwiseUnaryOp, SrcXprNested> SrcXprType; \ + static void run(DstXprType &dst, const SrcXprType &src, const assign_op &func) \ + { \ + resize_if_allowed(dst, src, func); \ + eigen_assert(dst.rows() == src.rows() && dst.cols() == src.cols()); \ + if (vml_assign_traits::Traversal == LinearTraversal) { \ + VMLOP(dst.size(), \ + (const VMLTYPE *)src.nestedExpression().data(), \ + (VMLTYPE *)dst.data() EIGEN_PP_EXPAND(EIGEN_VMLMODE_EXPAND_##VMLMODE)); \ + } else { \ + const Index outerSize = dst.outerSize(); \ + for (Index outer = 0; outer < outerSize; ++outer) { \ + const EIGENTYPE *src_ptr = src.IsRowMajor ? &(src.nestedExpression().coeffRef(outer, 0)) \ + : &(src.nestedExpression().coeffRef(0, outer)); \ + EIGENTYPE *dst_ptr = dst.IsRowMajor ? &(dst.coeffRef(outer, 0)) : &(dst.coeffRef(0, outer)); \ + VMLOP(dst.innerSize(), \ + (const VMLTYPE *)src_ptr, \ + (VMLTYPE *)dst_ptr EIGEN_PP_EXPAND(EIGEN_VMLMODE_EXPAND_##VMLMODE)); \ + } \ + } \ + } \ + }; + + +#define EIGEN_MKL_VML_DECLARE_UNARY_CALLS_REAL(EIGENOP, VMLOP, VMLMODE) \ + EIGEN_MKL_VML_DECLARE_UNARY_CALL(EIGENOP, EIGEN_CAT(EIGEN_VMLMODE_PREFIX(VMLMODE), s##VMLOP), float, float, VMLMODE) \ + EIGEN_MKL_VML_DECLARE_UNARY_CALL(EIGENOP, EIGEN_CAT(EIGEN_VMLMODE_PREFIX(VMLMODE), d##VMLOP), double, double, VMLMODE) + +#define EIGEN_MKL_VML_DECLARE_UNARY_CALLS_CPLX(EIGENOP, VMLOP, VMLMODE) \ + EIGEN_MKL_VML_DECLARE_UNARY_CALL( \ + EIGENOP, EIGEN_CAT(EIGEN_VMLMODE_PREFIX(VMLMODE), c##VMLOP), scomplex, MKL_Complex8, VMLMODE) \ + EIGEN_MKL_VML_DECLARE_UNARY_CALL( \ + EIGENOP, EIGEN_CAT(EIGEN_VMLMODE_PREFIX(VMLMODE), z##VMLOP), dcomplex, MKL_Complex16, VMLMODE) + +#define EIGEN_MKL_VML_DECLARE_UNARY_CALLS(EIGENOP, VMLOP, VMLMODE) \ + EIGEN_MKL_VML_DECLARE_UNARY_CALLS_REAL(EIGENOP, VMLOP, VMLMODE) \ EIGEN_MKL_VML_DECLARE_UNARY_CALLS_CPLX(EIGENOP, VMLOP, VMLMODE) - -EIGEN_MKL_VML_DECLARE_UNARY_CALLS(sin, Sin, LA) -EIGEN_MKL_VML_DECLARE_UNARY_CALLS(asin, Asin, LA) -EIGEN_MKL_VML_DECLARE_UNARY_CALLS(sinh, Sinh, LA) -EIGEN_MKL_VML_DECLARE_UNARY_CALLS(cos, Cos, LA) -EIGEN_MKL_VML_DECLARE_UNARY_CALLS(acos, Acos, LA) -EIGEN_MKL_VML_DECLARE_UNARY_CALLS(cosh, Cosh, LA) -EIGEN_MKL_VML_DECLARE_UNARY_CALLS(tan, Tan, LA) -EIGEN_MKL_VML_DECLARE_UNARY_CALLS(atan, Atan, LA) -EIGEN_MKL_VML_DECLARE_UNARY_CALLS(tanh, Tanh, LA) -// EIGEN_MKL_VML_DECLARE_UNARY_CALLS(abs, Abs, _) -EIGEN_MKL_VML_DECLARE_UNARY_CALLS(exp, Exp, LA) -EIGEN_MKL_VML_DECLARE_UNARY_CALLS(log, Ln, LA) -EIGEN_MKL_VML_DECLARE_UNARY_CALLS(log10, Log10, LA) -EIGEN_MKL_VML_DECLARE_UNARY_CALLS(sqrt, Sqrt, _) - -EIGEN_MKL_VML_DECLARE_UNARY_CALLS_REAL(square, Sqr, _) -EIGEN_MKL_VML_DECLARE_UNARY_CALLS_CPLX(arg, Arg, _) -EIGEN_MKL_VML_DECLARE_UNARY_CALLS_REAL(round, Round, _) -EIGEN_MKL_VML_DECLARE_UNARY_CALLS_REAL(floor, Floor, _) -EIGEN_MKL_VML_DECLARE_UNARY_CALLS_REAL(ceil, Ceil, _) - -#define EIGEN_MKL_VML_DECLARE_POW_CALL(EIGENOP, VMLOP, EIGENTYPE, VMLTYPE, VMLMODE) \ - template< typename DstXprType, typename SrcXprNested, typename Plain> \ - struct Assignment, SrcXprNested, \ - const CwiseNullaryOp,Plain> >, assign_op, \ - Dense2Dense, typename enable_if::EnableVml>::type> { \ - typedef CwiseBinaryOp, SrcXprNested, \ - const CwiseNullaryOp,Plain> > SrcXprType; \ - static void run(DstXprType &dst, const SrcXprType &src, const assign_op &func) { \ - resize_if_allowed(dst, src, func); \ - eigen_assert(dst.rows() == src.rows() && dst.cols() == src.cols()); \ - VMLTYPE exponent = reinterpret_cast(src.rhs().functor().m_other); \ - if(vml_assign_traits::Traversal==LinearTraversal) \ - { \ - VMLOP( dst.size(), (const VMLTYPE*)src.lhs().data(), exponent, \ - (VMLTYPE*)dst.data() EIGEN_PP_EXPAND(EIGEN_VMLMODE_EXPAND_##VMLMODE) ); \ - } else { \ - const Index outerSize = dst.outerSize(); \ - for(Index outer = 0; outer < outerSize; ++outer) { \ - const EIGENTYPE *src_ptr = src.IsRowMajor ? &(src.lhs().coeffRef(outer,0)) : \ - &(src.lhs().coeffRef(0, outer)); \ - EIGENTYPE *dst_ptr = dst.IsRowMajor ? &(dst.coeffRef(outer,0)) : &(dst.coeffRef(0, outer)); \ - VMLOP( dst.innerSize(), (const VMLTYPE*)src_ptr, exponent, \ - (VMLTYPE*)dst_ptr EIGEN_PP_EXPAND(EIGEN_VMLMODE_EXPAND_##VMLMODE)); \ - } \ - } \ - } \ + + EIGEN_MKL_VML_DECLARE_UNARY_CALLS(sin, Sin, LA) + EIGEN_MKL_VML_DECLARE_UNARY_CALLS(asin, Asin, LA) + EIGEN_MKL_VML_DECLARE_UNARY_CALLS(sinh, Sinh, LA) + EIGEN_MKL_VML_DECLARE_UNARY_CALLS(cos, Cos, LA) + EIGEN_MKL_VML_DECLARE_UNARY_CALLS(acos, Acos, LA) + EIGEN_MKL_VML_DECLARE_UNARY_CALLS(cosh, Cosh, LA) + EIGEN_MKL_VML_DECLARE_UNARY_CALLS(tan, Tan, LA) + EIGEN_MKL_VML_DECLARE_UNARY_CALLS(atan, Atan, LA) + EIGEN_MKL_VML_DECLARE_UNARY_CALLS(tanh, Tanh, LA) + // EIGEN_MKL_VML_DECLARE_UNARY_CALLS(abs, Abs, _) + EIGEN_MKL_VML_DECLARE_UNARY_CALLS(exp, Exp, LA) + EIGEN_MKL_VML_DECLARE_UNARY_CALLS(log, Ln, LA) + EIGEN_MKL_VML_DECLARE_UNARY_CALLS(log10, Log10, LA) + EIGEN_MKL_VML_DECLARE_UNARY_CALLS(sqrt, Sqrt, _) + + EIGEN_MKL_VML_DECLARE_UNARY_CALLS_REAL(square, Sqr, _) + EIGEN_MKL_VML_DECLARE_UNARY_CALLS_CPLX(arg, Arg, _) + EIGEN_MKL_VML_DECLARE_UNARY_CALLS_REAL(round, Round, _) + EIGEN_MKL_VML_DECLARE_UNARY_CALLS_REAL(floor, Floor, _) + EIGEN_MKL_VML_DECLARE_UNARY_CALLS_REAL(ceil, Ceil, _) + +#define EIGEN_MKL_VML_DECLARE_POW_CALL(EIGENOP, VMLOP, EIGENTYPE, VMLTYPE, VMLMODE) \ + template \ + struct Assignment, \ + SrcXprNested, \ + const CwiseNullaryOp, Plain>>, \ + assign_op, \ + Dense2Dense, \ + typename enable_if::EnableVml>::type> \ + { \ + typedef CwiseBinaryOp, \ + SrcXprNested, \ + const CwiseNullaryOp, Plain>> \ + SrcXprType; \ + static void run(DstXprType &dst, const SrcXprType &src, const assign_op &func) \ + { \ + resize_if_allowed(dst, src, func); \ + eigen_assert(dst.rows() == src.rows() && dst.cols() == src.cols()); \ + VMLTYPE exponent = reinterpret_cast(src.rhs().functor().m_other); \ + if (vml_assign_traits::Traversal == LinearTraversal) { \ + VMLOP(dst.size(), \ + (const VMLTYPE *)src.lhs().data(), \ + exponent, \ + (VMLTYPE *)dst.data() EIGEN_PP_EXPAND(EIGEN_VMLMODE_EXPAND_##VMLMODE)); \ + } else { \ + const Index outerSize = dst.outerSize(); \ + for (Index outer = 0; outer < outerSize; ++outer) { \ + const EIGENTYPE *src_ptr = \ + src.IsRowMajor ? &(src.lhs().coeffRef(outer, 0)) : &(src.lhs().coeffRef(0, outer)); \ + EIGENTYPE *dst_ptr = dst.IsRowMajor ? &(dst.coeffRef(outer, 0)) : &(dst.coeffRef(0, outer)); \ + VMLOP(dst.innerSize(), \ + (const VMLTYPE *)src_ptr, \ + exponent, \ + (VMLTYPE *)dst_ptr EIGEN_PP_EXPAND(EIGEN_VMLMODE_EXPAND_##VMLMODE)); \ + } \ + } \ + } \ }; - -EIGEN_MKL_VML_DECLARE_POW_CALL(pow, vmsPowx, float, float, LA) -EIGEN_MKL_VML_DECLARE_POW_CALL(pow, vmdPowx, double, double, LA) -EIGEN_MKL_VML_DECLARE_POW_CALL(pow, vmcPowx, scomplex, MKL_Complex8, LA) -EIGEN_MKL_VML_DECLARE_POW_CALL(pow, vmzPowx, dcomplex, MKL_Complex16, LA) -} // end namespace internal + EIGEN_MKL_VML_DECLARE_POW_CALL(pow, vmsPowx, float, float, LA) + EIGEN_MKL_VML_DECLARE_POW_CALL(pow, vmdPowx, double, double, LA) + EIGEN_MKL_VML_DECLARE_POW_CALL(pow, vmcPowx, scomplex, MKL_Complex8, LA) + EIGEN_MKL_VML_DECLARE_POW_CALL(pow, vmzPowx, dcomplex, MKL_Complex16, LA) + +}// end namespace internal -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_ASSIGN_VML_H +#endif// EIGEN_ASSIGN_VML_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/BandMatrix.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/BandMatrix.h index 4978c914..7e9b51eb 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/BandMatrix.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/BandMatrix.h @@ -10,15 +10,13 @@ #ifndef EIGEN_BANDMATRIX_H #define EIGEN_BANDMATRIX_H -namespace Eigen { +namespace Eigen { namespace internal { -template -class BandMatrixBase : public EigenBase -{ + template class BandMatrixBase : public EigenBase + { public: - enum { Flags = internal::traits::Flags, CoeffReadCost = internal::traits::CoeffReadCost, @@ -27,25 +25,22 @@ class BandMatrixBase : public EigenBase MaxRowsAtCompileTime = internal::traits::MaxRowsAtCompileTime, MaxColsAtCompileTime = internal::traits::MaxColsAtCompileTime, Supers = internal::traits::Supers, - Subs = internal::traits::Subs, + Subs = internal::traits::Subs, Options = internal::traits::Options }; typedef typename internal::traits::Scalar Scalar; - typedef Matrix DenseMatrixType; + typedef Matrix DenseMatrixType; typedef typename DenseMatrixType::StorageIndex StorageIndex; typedef typename internal::traits::CoefficientsType CoefficientsType; typedef EigenBase Base; protected: enum { - DataRowsAtCompileTime = ((Supers!=Dynamic) && (Subs!=Dynamic)) - ? 1 + Supers + Subs - : Dynamic, - SizeAtCompileTime = EIGEN_SIZE_MIN_PREFER_DYNAMIC(RowsAtCompileTime,ColsAtCompileTime) + DataRowsAtCompileTime = ((Supers != Dynamic) && (Subs != Dynamic)) ? 1 + Supers + Subs : Dynamic, + SizeAtCompileTime = EIGEN_SIZE_MIN_PREFER_DYNAMIC(RowsAtCompileTime, ColsAtCompileTime) }; public: - using Base::derived; using Base::rows; using Base::cols; @@ -55,160 +50,162 @@ class BandMatrixBase : public EigenBase /** \returns the number of sub diagonals */ inline Index subs() const { return derived().subs(); } - + /** \returns an expression of the underlying coefficient matrix */ - inline const CoefficientsType& coeffs() const { return derived().coeffs(); } - + inline const CoefficientsType &coeffs() const { return derived().coeffs(); } + /** \returns an expression of the underlying coefficient matrix */ - inline CoefficientsType& coeffs() { return derived().coeffs(); } + inline CoefficientsType &coeffs() { return derived().coeffs(); } /** \returns a vector expression of the \a i -th column, - * only the meaningful part is returned. - * \warning the internal storage must be column major. */ - inline Block col(Index i) + * only the meaningful part is returned. + * \warning the internal storage must be column major. */ + inline Block col(Index i) { - EIGEN_STATIC_ASSERT((Options&RowMajor)==0,THIS_METHOD_IS_ONLY_FOR_COLUMN_MAJOR_MATRICES); + EIGEN_STATIC_ASSERT((Options & RowMajor) == 0, THIS_METHOD_IS_ONLY_FOR_COLUMN_MAJOR_MATRICES); Index start = 0; Index len = coeffs().rows(); - if (i<=supers()) - { - start = supers()-i; - len = (std::min)(rows(),std::max(0,coeffs().rows() - (supers()-i))); - } - else if (i>=rows()-subs()) - len = std::max(0,coeffs().rows() - (i + 1 - rows() + subs())); - return Block(coeffs(), start, i, len, 1); + if (i <= supers()) { + start = supers() - i; + len = (std::min)(rows(), std::max(0, coeffs().rows() - (supers() - i))); + } else if (i >= rows() - subs()) + len = std::max(0, coeffs().rows() - (i + 1 - rows() + subs())); + return Block(coeffs(), start, i, len, 1); } /** \returns a vector expression of the main diagonal */ - inline Block diagonal() - { return Block(coeffs(),supers(),0,1,(std::min)(rows(),cols())); } + inline Block diagonal() + { + return Block(coeffs(), supers(), 0, 1, (std::min)(rows(), cols())); + } /** \returns a vector expression of the main diagonal (const version) */ - inline const Block diagonal() const - { return Block(coeffs(),supers(),0,1,(std::min)(rows(),cols())); } + inline const Block diagonal() const + { + return Block(coeffs(), supers(), 0, 1, (std::min)(rows(), cols())); + } - template struct DiagonalIntReturnType { + template struct DiagonalIntReturnType + { enum { - ReturnOpposite = (Options&SelfAdjoint) && (((Index)>0 && Supers==0) || ((Index)<0 && Subs==0)), + ReturnOpposite = (Options & SelfAdjoint) && (((Index) > 0 && Supers == 0) || ((Index) < 0 && Subs == 0)), Conjugate = ReturnOpposite && NumTraits::IsComplex, ActualIndex = ReturnOpposite ? -Index : Index, - DiagonalSize = (RowsAtCompileTime==Dynamic || ColsAtCompileTime==Dynamic) - ? Dynamic - : (ActualIndex<0 - ? EIGEN_SIZE_MIN_PREFER_DYNAMIC(ColsAtCompileTime, RowsAtCompileTime + ActualIndex) - : EIGEN_SIZE_MIN_PREFER_DYNAMIC(RowsAtCompileTime, ColsAtCompileTime - ActualIndex)) + DiagonalSize = + (RowsAtCompileTime == Dynamic || ColsAtCompileTime == Dynamic) + ? Dynamic + : (ActualIndex < 0 ? EIGEN_SIZE_MIN_PREFER_DYNAMIC(ColsAtCompileTime, RowsAtCompileTime + ActualIndex) + : EIGEN_SIZE_MIN_PREFER_DYNAMIC(RowsAtCompileTime, ColsAtCompileTime - ActualIndex)) }; - typedef Block BuildType; - typedef typename internal::conditional,BuildType >, - BuildType>::type Type; + typedef Block BuildType; + typedef typename internal:: + conditional, BuildType>, BuildType>::type Type; }; /** \returns a vector expression of the \a N -th sub or super diagonal */ template inline typename DiagonalIntReturnType::Type diagonal() { - return typename DiagonalIntReturnType::BuildType(coeffs(), supers()-N, (std::max)(0,N), 1, diagonalLength(N)); + return + typename DiagonalIntReturnType::BuildType(coeffs(), supers() - N, (std::max)(0, N), 1, diagonalLength(N)); } /** \returns a vector expression of the \a N -th sub or super diagonal */ template inline const typename DiagonalIntReturnType::Type diagonal() const { - return typename DiagonalIntReturnType::BuildType(coeffs(), supers()-N, (std::max)(0,N), 1, diagonalLength(N)); + return + typename DiagonalIntReturnType::BuildType(coeffs(), supers() - N, (std::max)(0, N), 1, diagonalLength(N)); } /** \returns a vector expression of the \a i -th sub or super diagonal */ - inline Block diagonal(Index i) + inline Block diagonal(Index i) { - eigen_assert((i<0 && -i<=subs()) || (i>=0 && i<=supers())); - return Block(coeffs(), supers()-i, std::max(0,i), 1, diagonalLength(i)); + eigen_assert((i < 0 && -i <= subs()) || (i >= 0 && i <= supers())); + return Block(coeffs(), supers() - i, std::max(0, i), 1, diagonalLength(i)); } /** \returns a vector expression of the \a i -th sub or super diagonal */ - inline const Block diagonal(Index i) const + inline const Block diagonal(Index i) const { - eigen_assert((i<0 && -i<=subs()) || (i>=0 && i<=supers())); - return Block(coeffs(), supers()-i, std::max(0,i), 1, diagonalLength(i)); + eigen_assert((i < 0 && -i <= subs()) || (i >= 0 && i <= supers())); + return Block( + coeffs(), supers() - i, std::max(0, i), 1, diagonalLength(i)); } - - template inline void evalTo(Dest& dst) const + + template inline void evalTo(Dest &dst) const { - dst.resize(rows(),cols()); + dst.resize(rows(), cols()); dst.setZero(); dst.diagonal() = diagonal(); - for (Index i=1; i<=supers();++i) - dst.diagonal(i) = diagonal(i); - for (Index i=1; i<=subs();++i) - dst.diagonal(-i) = diagonal(-i); + for (Index i = 1; i <= supers(); ++i) dst.diagonal(i) = diagonal(i); + for (Index i = 1; i <= subs(); ++i) dst.diagonal(-i) = diagonal(-i); } DenseMatrixType toDenseMatrix() const { - DenseMatrixType res(rows(),cols()); + DenseMatrixType res(rows(), cols()); evalTo(res); return res; } protected: - inline Index diagonalLength(Index i) const - { return i<0 ? (std::min)(cols(),rows()+i) : (std::min)(rows(),cols()-i); } -}; - -/** - * \class BandMatrix - * \ingroup Core_Module - * - * \brief Represents a rectangular matrix with a banded storage - * - * \tparam _Scalar Numeric type, i.e. float, double, int - * \tparam _Rows Number of rows, or \b Dynamic - * \tparam _Cols Number of columns, or \b Dynamic - * \tparam _Supers Number of super diagonal - * \tparam _Subs Number of sub diagonal - * \tparam _Options A combination of either \b #RowMajor or \b #ColMajor, and of \b #SelfAdjoint - * The former controls \ref TopicStorageOrders "storage order", and defaults to - * column-major. The latter controls whether the matrix represents a selfadjoint - * matrix in which case either Supers of Subs have to be null. - * - * \sa class TridiagonalMatrix - */ - -template -struct traits > -{ - typedef _Scalar Scalar; - typedef Dense StorageKind; - typedef Eigen::Index StorageIndex; - enum { - CoeffReadCost = NumTraits::ReadCost, - RowsAtCompileTime = _Rows, - ColsAtCompileTime = _Cols, - MaxRowsAtCompileTime = _Rows, - MaxColsAtCompileTime = _Cols, - Flags = LvalueBit, - Supers = _Supers, - Subs = _Subs, - Options = _Options, - DataRowsAtCompileTime = ((Supers!=Dynamic) && (Subs!=Dynamic)) ? 1 + Supers + Subs : Dynamic + { + return i < 0 ? (std::min)(cols(), rows() + i) : (std::min)(rows(), cols() - i); + } }; - typedef Matrix CoefficientsType; -}; -template -class BandMatrix : public BandMatrixBase > -{ - public: + /** + * \class BandMatrix + * \ingroup Core_Module + * + * \brief Represents a rectangular matrix with a banded storage + * + * \tparam _Scalar Numeric type, i.e. float, double, int + * \tparam _Rows Number of rows, or \b Dynamic + * \tparam _Cols Number of columns, or \b Dynamic + * \tparam _Supers Number of super diagonal + * \tparam _Subs Number of sub diagonal + * \tparam _Options A combination of either \b #RowMajor or \b #ColMajor, and of \b #SelfAdjoint + * The former controls \ref TopicStorageOrders "storage order", and defaults to + * column-major. The latter controls whether the matrix represents a selfadjoint + * matrix in which case either Supers of Subs have to be null. + * + * \sa class TridiagonalMatrix + */ + + template + struct traits> + { + typedef _Scalar Scalar; + typedef Dense StorageKind; + typedef Eigen::Index StorageIndex; + enum { + CoeffReadCost = NumTraits::ReadCost, + RowsAtCompileTime = _Rows, + ColsAtCompileTime = _Cols, + MaxRowsAtCompileTime = _Rows, + MaxColsAtCompileTime = _Cols, + Flags = LvalueBit, + Supers = _Supers, + Subs = _Subs, + Options = _Options, + DataRowsAtCompileTime = ((Supers != Dynamic) && (Subs != Dynamic)) ? 1 + Supers + Subs : Dynamic + }; + typedef Matrix + CoefficientsType; + }; + template + class BandMatrix : public BandMatrixBase> + { + public: typedef typename internal::traits::Scalar Scalar; typedef typename internal::traits::StorageIndex StorageIndex; typedef typename internal::traits::CoefficientsType CoefficientsType; - explicit inline BandMatrix(Index rows=Rows, Index cols=Cols, Index supers=Supers, Index subs=Subs) - : m_coeffs(1+supers+subs,cols), - m_rows(rows), m_supers(supers), m_subs(subs) - { - } + explicit inline BandMatrix(Index rows = Rows, Index cols = Cols, Index supers = Supers, Index subs = Subs) + : m_coeffs(1 + supers + subs, cols), m_rows(rows), m_supers(supers), m_subs(subs) + {} /** \returns the number of columns */ inline Index rows() const { return m_rows.value(); } @@ -222,56 +219,58 @@ class BandMatrix : public BandMatrixBase m_rows; + internal::variable_if_dynamic m_rows; internal::variable_if_dynamic m_supers; - internal::variable_if_dynamic m_subs; -}; - -template -class BandMatrixWrapper; - -template -struct traits > -{ - typedef typename _CoefficientsType::Scalar Scalar; - typedef typename _CoefficientsType::StorageKind StorageKind; - typedef typename _CoefficientsType::StorageIndex StorageIndex; - enum { - CoeffReadCost = internal::traits<_CoefficientsType>::CoeffReadCost, - RowsAtCompileTime = _Rows, - ColsAtCompileTime = _Cols, - MaxRowsAtCompileTime = _Rows, - MaxColsAtCompileTime = _Cols, - Flags = LvalueBit, - Supers = _Supers, - Subs = _Subs, - Options = _Options, - DataRowsAtCompileTime = ((Supers!=Dynamic) && (Subs!=Dynamic)) ? 1 + Supers + Subs : Dynamic + internal::variable_if_dynamic m_subs; }; - typedef _CoefficientsType CoefficientsType; -}; -template -class BandMatrixWrapper : public BandMatrixBase > -{ - public: + template + class BandMatrixWrapper; + + template + struct traits> + { + typedef typename _CoefficientsType::Scalar Scalar; + typedef typename _CoefficientsType::StorageKind StorageKind; + typedef typename _CoefficientsType::StorageIndex StorageIndex; + enum { + CoeffReadCost = internal::traits<_CoefficientsType>::CoeffReadCost, + RowsAtCompileTime = _Rows, + ColsAtCompileTime = _Cols, + MaxRowsAtCompileTime = _Rows, + MaxColsAtCompileTime = _Cols, + Flags = LvalueBit, + Supers = _Supers, + Subs = _Subs, + Options = _Options, + DataRowsAtCompileTime = ((Supers != Dynamic) && (Subs != Dynamic)) ? 1 + Supers + Subs : Dynamic + }; + typedef _CoefficientsType CoefficientsType; + }; + template + class BandMatrixWrapper + : public BandMatrixBase> + { + public: typedef typename internal::traits::Scalar Scalar; typedef typename internal::traits::CoefficientsType CoefficientsType; typedef typename internal::traits::StorageIndex StorageIndex; - explicit inline BandMatrixWrapper(const CoefficientsType& coeffs, Index rows=_Rows, Index cols=_Cols, Index supers=_Supers, Index subs=_Subs) - : m_coeffs(coeffs), - m_rows(rows), m_supers(supers), m_subs(subs) + explicit inline BandMatrixWrapper(const CoefficientsType &coeffs, + Index rows = _Rows, + Index cols = _Cols, + Index supers = _Supers, + Index subs = _Subs) + : m_coeffs(coeffs), m_rows(rows), m_supers(supers), m_subs(subs) { EIGEN_UNUSED_VARIABLE(cols); - //internal::assert(coeffs.cols()==cols() && (supers()+subs()+1)==coeffs.rows()); + // internal::assert(coeffs.cols()==cols() && (supers()+subs()+1)==coeffs.rows()); } /** \returns the number of columns */ @@ -286,68 +285,76 @@ class BandMatrixWrapper : public BandMatrixBase m_rows; + const CoefficientsType &m_coeffs; + internal::variable_if_dynamic m_rows; internal::variable_if_dynamic m_supers; - internal::variable_if_dynamic m_subs; -}; - -/** - * \class TridiagonalMatrix - * \ingroup Core_Module - * - * \brief Represents a tridiagonal matrix with a compact banded storage - * - * \tparam Scalar Numeric type, i.e. float, double, int - * \tparam Size Number of rows and cols, or \b Dynamic - * \tparam Options Can be 0 or \b SelfAdjoint - * - * \sa class BandMatrix - */ -template -class TridiagonalMatrix : public BandMatrix -{ - typedef BandMatrix Base; + internal::variable_if_dynamic m_subs; + }; + + /** + * \class TridiagonalMatrix + * \ingroup Core_Module + * + * \brief Represents a tridiagonal matrix with a compact banded storage + * + * \tparam Scalar Numeric type, i.e. float, double, int + * \tparam Size Number of rows and cols, or \b Dynamic + * \tparam Options Can be 0 or \b SelfAdjoint + * + * \sa class BandMatrix + */ + template + class TridiagonalMatrix : public BandMatrix + { + typedef BandMatrix Base; typedef typename Base::StorageIndex StorageIndex; + public: - explicit TridiagonalMatrix(Index size = Size) : Base(size,size,Options&SelfAdjoint?0:1,1) {} + explicit TridiagonalMatrix(Index size = Size) : Base(size, size, Options & SelfAdjoint ? 0 : 1, 1) {} - inline typename Base::template DiagonalIntReturnType<1>::Type super() - { return Base::template diagonal<1>(); } + inline typename Base::template DiagonalIntReturnType<1>::Type super() { return Base::template diagonal<1>(); } inline const typename Base::template DiagonalIntReturnType<1>::Type super() const - { return Base::template diagonal<1>(); } - inline typename Base::template DiagonalIntReturnType<-1>::Type sub() - { return Base::template diagonal<-1>(); } + { + return Base::template diagonal<1>(); + } + inline typename Base::template DiagonalIntReturnType<-1>::Type sub() { return Base::template diagonal<-1>(); } inline const typename Base::template DiagonalIntReturnType<-1>::Type sub() const - { return Base::template diagonal<-1>(); } + { + return Base::template diagonal<-1>(); + } + protected: -}; + }; -struct BandShape {}; + struct BandShape + { + }; -template -struct evaluator_traits > - : public evaluator_traits_base > -{ - typedef BandShape Shape; -}; + template + struct evaluator_traits> + : public evaluator_traits_base> + { + typedef BandShape Shape; + }; -template -struct evaluator_traits > - : public evaluator_traits_base > -{ - typedef BandShape Shape; -}; + template + struct evaluator_traits> + : public evaluator_traits_base> + { + typedef BandShape Shape; + }; -template<> struct AssignmentKind { typedef EigenBase2EigenBase Kind; }; + template<> struct AssignmentKind + { + typedef EigenBase2EigenBase Kind; + }; -} // end namespace internal +}// end namespace internal -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_BANDMATRIX_H +#endif// EIGEN_BANDMATRIX_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/Block.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/Block.h index 11de45c2..cfe915ee 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/Block.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/Block.h @@ -11,171 +11,177 @@ #ifndef EIGEN_BLOCK_H #define EIGEN_BLOCK_H -namespace Eigen { +namespace Eigen { namespace internal { -template -struct traits > : traits -{ - typedef typename traits::Scalar Scalar; - typedef typename traits::StorageKind StorageKind; - typedef typename traits::XprKind XprKind; - typedef typename ref_selector::type XprTypeNested; - typedef typename remove_reference::type _XprTypeNested; - enum{ - MatrixRows = traits::RowsAtCompileTime, - MatrixCols = traits::ColsAtCompileTime, - RowsAtCompileTime = MatrixRows == 0 ? 0 : BlockRows, - ColsAtCompileTime = MatrixCols == 0 ? 0 : BlockCols, - MaxRowsAtCompileTime = BlockRows==0 ? 0 - : RowsAtCompileTime != Dynamic ? int(RowsAtCompileTime) - : int(traits::MaxRowsAtCompileTime), - MaxColsAtCompileTime = BlockCols==0 ? 0 - : ColsAtCompileTime != Dynamic ? int(ColsAtCompileTime) - : int(traits::MaxColsAtCompileTime), - - XprTypeIsRowMajor = (int(traits::Flags)&RowMajorBit) != 0, - IsRowMajor = (MaxRowsAtCompileTime==1&&MaxColsAtCompileTime!=1) ? 1 - : (MaxColsAtCompileTime==1&&MaxRowsAtCompileTime!=1) ? 0 - : XprTypeIsRowMajor, - HasSameStorageOrderAsXprType = (IsRowMajor == XprTypeIsRowMajor), - InnerSize = IsRowMajor ? int(ColsAtCompileTime) : int(RowsAtCompileTime), - InnerStrideAtCompileTime = HasSameStorageOrderAsXprType - ? int(inner_stride_at_compile_time::ret) - : int(outer_stride_at_compile_time::ret), - OuterStrideAtCompileTime = HasSameStorageOrderAsXprType - ? int(outer_stride_at_compile_time::ret) - : int(inner_stride_at_compile_time::ret), - - // FIXME, this traits is rather specialized for dense object and it needs to be cleaned further - FlagsLvalueBit = is_lvalue::value ? LvalueBit : 0, - FlagsRowMajorBit = IsRowMajor ? RowMajorBit : 0, - Flags = (traits::Flags & (DirectAccessBit | (InnerPanel?CompressedAccessBit:0))) | FlagsLvalueBit | FlagsRowMajorBit, - // FIXME DirectAccessBit should not be handled by expressions - // - // Alignment is needed by MapBase's assertions - // We can sefely set it to false here. Internal alignment errors will be detected by an eigen_internal_assert in the respective evaluator - Alignment = 0 + template + struct traits> : traits + { + typedef typename traits::Scalar Scalar; + typedef typename traits::StorageKind StorageKind; + typedef typename traits::XprKind XprKind; + typedef typename ref_selector::type XprTypeNested; + typedef typename remove_reference::type _XprTypeNested; + enum { + MatrixRows = traits::RowsAtCompileTime, + MatrixCols = traits::ColsAtCompileTime, + RowsAtCompileTime = MatrixRows == 0 ? 0 : BlockRows, + ColsAtCompileTime = MatrixCols == 0 ? 0 : BlockCols, + MaxRowsAtCompileTime = BlockRows == 0 ? 0 + : RowsAtCompileTime != Dynamic ? int(RowsAtCompileTime) + : int(traits::MaxRowsAtCompileTime), + MaxColsAtCompileTime = BlockCols == 0 ? 0 + : ColsAtCompileTime != Dynamic ? int(ColsAtCompileTime) + : int(traits::MaxColsAtCompileTime), + + XprTypeIsRowMajor = (int(traits::Flags) & RowMajorBit) != 0, + IsRowMajor = (MaxRowsAtCompileTime == 1 && MaxColsAtCompileTime != 1) ? 1 + : (MaxColsAtCompileTime == 1 && MaxRowsAtCompileTime != 1) ? 0 + : XprTypeIsRowMajor, + HasSameStorageOrderAsXprType = (IsRowMajor == XprTypeIsRowMajor), + InnerSize = IsRowMajor ? int(ColsAtCompileTime) : int(RowsAtCompileTime), + InnerStrideAtCompileTime = HasSameStorageOrderAsXprType ? int(inner_stride_at_compile_time::ret) + : int(outer_stride_at_compile_time::ret), + OuterStrideAtCompileTime = HasSameStorageOrderAsXprType ? int(outer_stride_at_compile_time::ret) + : int(inner_stride_at_compile_time::ret), + + // FIXME, this traits is rather specialized for dense object and it needs to be cleaned further + FlagsLvalueBit = is_lvalue::value ? LvalueBit : 0, + FlagsRowMajorBit = IsRowMajor ? RowMajorBit : 0, + Flags = (traits::Flags & (DirectAccessBit | (InnerPanel ? CompressedAccessBit : 0))) | FlagsLvalueBit + | FlagsRowMajorBit, + // FIXME DirectAccessBit should not be handled by expressions + // + // Alignment is needed by MapBase's assertions + // We can sefely set it to false here. Internal alignment errors will be detected by an eigen_internal_assert in + // the respective evaluator + Alignment = 0 + }; }; -}; -template::ret> class BlockImpl_dense; - -} // end namespace internal + template::ret> + class BlockImpl_dense; + +}// end namespace internal template class BlockImpl; /** \class Block - * \ingroup Core_Module - * - * \brief Expression of a fixed-size or dynamic-size block - * - * \tparam XprType the type of the expression in which we are taking a block - * \tparam BlockRows the number of rows of the block we are taking at compile time (optional) - * \tparam BlockCols the number of columns of the block we are taking at compile time (optional) - * \tparam InnerPanel is true, if the block maps to a set of rows of a row major matrix or - * to set of columns of a column major matrix (optional). The parameter allows to determine - * at compile time whether aligned access is possible on the block expression. - * - * This class represents an expression of either a fixed-size or dynamic-size block. It is the return - * type of DenseBase::block(Index,Index,Index,Index) and DenseBase::block(Index,Index) and - * most of the time this is the only way it is used. - * - * However, if you want to directly maniputate block expressions, - * for instance if you want to write a function returning such an expression, you - * will need to use this class. - * - * Here is an example illustrating the dynamic case: - * \include class_Block.cpp - * Output: \verbinclude class_Block.out - * - * \note Even though this expression has dynamic size, in the case where \a XprType - * has fixed size, this expression inherits a fixed maximal size which means that evaluating - * it does not cause a dynamic memory allocation. - * - * Here is an example illustrating the fixed-size case: - * \include class_FixedBlock.cpp - * Output: \verbinclude class_FixedBlock.out - * - * \sa DenseBase::block(Index,Index,Index,Index), DenseBase::block(Index,Index), class VectorBlock - */ -template class Block + * \ingroup Core_Module + * + * \brief Expression of a fixed-size or dynamic-size block + * + * \tparam XprType the type of the expression in which we are taking a block + * \tparam BlockRows the number of rows of the block we are taking at compile time (optional) + * \tparam BlockCols the number of columns of the block we are taking at compile time (optional) + * \tparam InnerPanel is true, if the block maps to a set of rows of a row major matrix or + * to set of columns of a column major matrix (optional). The parameter allows to determine + * at compile time whether aligned access is possible on the block expression. + * + * This class represents an expression of either a fixed-size or dynamic-size block. It is the return + * type of DenseBase::block(Index,Index,Index,Index) and DenseBase::block(Index,Index) and + * most of the time this is the only way it is used. + * + * However, if you want to directly maniputate block expressions, + * for instance if you want to write a function returning such an expression, you + * will need to use this class. + * + * Here is an example illustrating the dynamic case: + * \include class_Block.cpp + * Output: \verbinclude class_Block.out + * + * \note Even though this expression has dynamic size, in the case where \a XprType + * has fixed size, this expression inherits a fixed maximal size which means that evaluating + * it does not cause a dynamic memory allocation. + * + * Here is an example illustrating the fixed-size case: + * \include class_FixedBlock.cpp + * Output: \verbinclude class_FixedBlock.out + * + * \sa DenseBase::block(Index,Index,Index,Index), DenseBase::block(Index,Index), class VectorBlock + */ +template +class Block : public BlockImpl::StorageKind> { - typedef BlockImpl::StorageKind> Impl; - public: - //typedef typename Impl::Base Base; - typedef Impl Base; - EIGEN_GENERIC_PUBLIC_INTERFACE(Block) - EIGEN_INHERIT_ASSIGNMENT_OPERATORS(Block) - - typedef typename internal::remove_all::type NestedExpression; - - /** Column or Row constructor - */ - EIGEN_DEVICE_FUNC - inline Block(XprType& xpr, Index i) : Impl(xpr,i) - { - eigen_assert( (i>=0) && ( - ((BlockRows==1) && (BlockCols==XprType::ColsAtCompileTime) && i= 0 && BlockRows >= 0 && startRow + BlockRows <= xpr.rows() - && startCol >= 0 && BlockCols >= 0 && startCol + BlockCols <= xpr.cols()); - } - - /** Dynamic-size constructor - */ - EIGEN_DEVICE_FUNC - inline Block(XprType& xpr, - Index startRow, Index startCol, - Index blockRows, Index blockCols) - : Impl(xpr, startRow, startCol, blockRows, blockCols) - { - eigen_assert((RowsAtCompileTime==Dynamic || RowsAtCompileTime==blockRows) - && (ColsAtCompileTime==Dynamic || ColsAtCompileTime==blockCols)); - eigen_assert(startRow >= 0 && blockRows >= 0 && startRow <= xpr.rows() - blockRows - && startCol >= 0 && blockCols >= 0 && startCol <= xpr.cols() - blockCols); - } + typedef BlockImpl::StorageKind> Impl; + +public: + // typedef typename Impl::Base Base; + typedef Impl Base; + EIGEN_GENERIC_PUBLIC_INTERFACE(Block) + EIGEN_INHERIT_ASSIGNMENT_OPERATORS(Block) + + typedef typename internal::remove_all::type NestedExpression; + + /** Column or Row constructor + */ + EIGEN_DEVICE_FUNC + inline Block(XprType &xpr, Index i) : Impl(xpr, i) + { + eigen_assert((i >= 0) + && (((BlockRows == 1) && (BlockCols == XprType::ColsAtCompileTime) && i < xpr.rows()) + || ((BlockRows == XprType::RowsAtCompileTime) && (BlockCols == 1) && i < xpr.cols()))); + } + + /** Fixed-size constructor + */ + EIGEN_DEVICE_FUNC + inline Block(XprType &xpr, Index startRow, Index startCol) : Impl(xpr, startRow, startCol) + { + EIGEN_STATIC_ASSERT( + RowsAtCompileTime != Dynamic && ColsAtCompileTime != Dynamic, THIS_METHOD_IS_ONLY_FOR_FIXED_SIZE) + eigen_assert(startRow >= 0 && BlockRows >= 0 && startRow + BlockRows <= xpr.rows() && startCol >= 0 + && BlockCols >= 0 && startCol + BlockCols <= xpr.cols()); + } + + /** Dynamic-size constructor + */ + EIGEN_DEVICE_FUNC + inline Block(XprType &xpr, Index startRow, Index startCol, Index blockRows, Index blockCols) + : Impl(xpr, startRow, startCol, blockRows, blockCols) + { + eigen_assert((RowsAtCompileTime == Dynamic || RowsAtCompileTime == blockRows) + && (ColsAtCompileTime == Dynamic || ColsAtCompileTime == blockCols)); + eigen_assert(startRow >= 0 && blockRows >= 0 && startRow <= xpr.rows() - blockRows && startCol >= 0 + && blockCols >= 0 && startCol <= xpr.cols() - blockCols); + } }; - + // The generic default implementation for dense block simplu forward to the internal::BlockImpl_dense // that must be specialized for direct and non-direct access... template class BlockImpl : public internal::BlockImpl_dense { - typedef internal::BlockImpl_dense Impl; - typedef typename XprType::StorageIndex StorageIndex; - public: - typedef Impl Base; - EIGEN_INHERIT_ASSIGNMENT_OPERATORS(BlockImpl) - EIGEN_DEVICE_FUNC inline BlockImpl(XprType& xpr, Index i) : Impl(xpr,i) {} - EIGEN_DEVICE_FUNC inline BlockImpl(XprType& xpr, Index startRow, Index startCol) : Impl(xpr, startRow, startCol) {} - EIGEN_DEVICE_FUNC - inline BlockImpl(XprType& xpr, Index startRow, Index startCol, Index blockRows, Index blockCols) - : Impl(xpr, startRow, startCol, blockRows, blockCols) {} + typedef internal::BlockImpl_dense Impl; + typedef typename XprType::StorageIndex StorageIndex; + +public: + typedef Impl Base; + EIGEN_INHERIT_ASSIGNMENT_OPERATORS(BlockImpl) + EIGEN_DEVICE_FUNC inline BlockImpl(XprType &xpr, Index i) : Impl(xpr, i) {} + EIGEN_DEVICE_FUNC inline BlockImpl(XprType &xpr, Index startRow, Index startCol) : Impl(xpr, startRow, startCol) {} + EIGEN_DEVICE_FUNC + inline BlockImpl(XprType &xpr, Index startRow, Index startCol, Index blockRows, Index blockCols) + : Impl(xpr, startRow, startCol, blockRows, blockCols) + {} }; namespace internal { -/** \internal Internal implementation of dense Blocks in the general case. */ -template class BlockImpl_dense - : public internal::dense_xpr_base >::type -{ + /** \internal Internal implementation of dense Blocks in the general case. */ + template + class BlockImpl_dense : public internal::dense_xpr_base>::type + { typedef Block BlockType; typedef typename internal::ref_selector::non_const_type XprTypeNested; - public: + public: typedef typename internal::dense_xpr_base::type Base; EIGEN_DENSE_PUBLIC_INTERFACE(BlockType) EIGEN_INHERIT_ASSIGNMENT_OPERATORS(BlockImpl_dense) @@ -183,50 +189,45 @@ template - inline PacketScalar packet(Index rowId, Index colId) const + template inline PacketScalar packet(Index rowId, Index colId) const { return m_xpr.template packet(rowId + m_startRow.value(), colId + m_startCol.value()); } - template - inline void writePacket(Index rowId, Index colId, const PacketScalar& val) + template inline void writePacket(Index rowId, Index colId, const PacketScalar &val) { m_xpr.template writePacket(rowId + m_startRow.value(), colId + m_startCol.value(), val); } - template - inline PacketScalar packet(Index index) const + template inline PacketScalar packet(Index index) const { - return m_xpr.template packet - (m_startRow.value() + (RowsAtCompileTime == 1 ? 0 : index), - m_startCol.value() + (RowsAtCompileTime == 1 ? index : 0)); + return m_xpr.template packet(m_startRow.value() + (RowsAtCompileTime == 1 ? 0 : index), + m_startCol.value() + (RowsAtCompileTime == 1 ? index : 0)); } - template - inline void writePacket(Index index, const PacketScalar& val) + template inline void writePacket(Index index, const PacketScalar &val) { - m_xpr.template writePacket - (m_startRow.value() + (RowsAtCompileTime == 1 ? 0 : index), - m_startCol.value() + (RowsAtCompileTime == 1 ? index : 0), val); + m_xpr.template writePacket(m_startRow.value() + (RowsAtCompileTime == 1 ? 0 : index), + m_startCol.value() + (RowsAtCompileTime == 1 ? index : 0), + val); } - #ifdef EIGEN_PARSED_BY_DOXYGEN +#ifdef EIGEN_PARSED_BY_DOXYGEN /** \sa MapBase::data() */ - EIGEN_DEVICE_FUNC inline const Scalar* data() const; + EIGEN_DEVICE_FUNC inline const Scalar *data() const; EIGEN_DEVICE_FUNC inline Index innerStride() const; EIGEN_DEVICE_FUNC inline Index outerStride() const; - #endif +#endif EIGEN_DEVICE_FUNC - const typename internal::remove_all::type& nestedExpression() const - { - return m_xpr; - } + const typename internal::remove_all::type &nestedExpression() const { return m_xpr; } EIGEN_DEVICE_FUNC - XprType& nestedExpression() { return m_xpr; } - + XprType &nestedExpression() { return m_xpr; } + EIGEN_DEVICE_FUNC - StorageIndex startRow() const - { - return m_startRow.value(); - } - + StorageIndex startRow() const { return m_startRow.value(); } + EIGEN_DEVICE_FUNC - StorageIndex startCol() const - { - return m_startCol.value(); - } + StorageIndex startCol() const { return m_startCol.value(); } protected: - XprTypeNested m_xpr; - const internal::variable_if_dynamic m_startRow; - const internal::variable_if_dynamic m_startCol; + const internal::variable_if_dynamic + m_startRow; + const internal::variable_if_dynamic + m_startCol; const internal::variable_if_dynamic m_blockRows; const internal::variable_if_dynamic m_blockCols; -}; + }; -/** \internal Internal implementation of dense Blocks in the direct access case.*/ -template -class BlockImpl_dense - : public MapBase > -{ + /** \internal Internal implementation of dense Blocks in the direct access case.*/ + template + class BlockImpl_dense + : public MapBase> + { typedef Block BlockType; typedef typename internal::ref_selector::non_const_type XprTypeNested; - enum { - XprTypeIsRowMajor = (int(traits::Flags)&RowMajorBit) != 0 - }; - public: + enum { XprTypeIsRowMajor = (int(traits::Flags) & RowMajorBit) != 0 }; + public: typedef MapBase Base; EIGEN_DENSE_PUBLIC_INTERFACE(BlockType) EIGEN_INHERIT_ASSIGNMENT_OPERATORS(BlockImpl_dense) /** Column or Row constructor - */ + */ EIGEN_DEVICE_FUNC - inline BlockImpl_dense(XprType& xpr, Index i) - : Base(xpr.data() + i * ( ((BlockRows==1) && (BlockCols==XprType::ColsAtCompileTime) && (!XprTypeIsRowMajor)) - || ((BlockRows==XprType::RowsAtCompileTime) && (BlockCols==1) && ( XprTypeIsRowMajor)) ? xpr.innerStride() : xpr.outerStride()), - BlockRows==1 ? 1 : xpr.rows(), - BlockCols==1 ? 1 : xpr.cols()), - m_xpr(xpr), - m_startRow( (BlockRows==1) && (BlockCols==XprType::ColsAtCompileTime) ? i : 0), - m_startCol( (BlockRows==XprType::RowsAtCompileTime) && (BlockCols==1) ? i : 0) + inline BlockImpl_dense(XprType &xpr, Index i) + : Base(xpr.data() + + i + * (((BlockRows == 1) && (BlockCols == XprType::ColsAtCompileTime) && (!XprTypeIsRowMajor)) + || ((BlockRows == XprType::RowsAtCompileTime) && (BlockCols == 1) && (XprTypeIsRowMajor)) + ? xpr.innerStride() + : xpr.outerStride()), + BlockRows == 1 ? 1 : xpr.rows(), + BlockCols == 1 ? 1 : xpr.cols()), + m_xpr(xpr), m_startRow((BlockRows == 1) && (BlockCols == XprType::ColsAtCompileTime) ? i : 0), + m_startCol((BlockRows == XprType::RowsAtCompileTime) && (BlockCols == 1) ? i : 0) { init(); } /** Fixed-size constructor - */ + */ EIGEN_DEVICE_FUNC - inline BlockImpl_dense(XprType& xpr, Index startRow, Index startCol) - : Base(xpr.data()+xpr.innerStride()*(XprTypeIsRowMajor?startCol:startRow) + xpr.outerStride()*(XprTypeIsRowMajor?startRow:startCol)), + inline BlockImpl_dense(XprType &xpr, Index startRow, Index startCol) + : Base(xpr.data() + xpr.innerStride() * (XprTypeIsRowMajor ? startCol : startRow) + + xpr.outerStride() * (XprTypeIsRowMajor ? startRow : startCol)), m_xpr(xpr), m_startRow(startRow), m_startCol(startCol) { init(); } /** Dynamic-size constructor - */ + */ EIGEN_DEVICE_FUNC - inline BlockImpl_dense(XprType& xpr, - Index startRow, Index startCol, - Index blockRows, Index blockCols) - : Base(xpr.data()+xpr.innerStride()*(XprTypeIsRowMajor?startCol:startRow) + xpr.outerStride()*(XprTypeIsRowMajor?startRow:startCol), blockRows, blockCols), + inline BlockImpl_dense(XprType &xpr, Index startRow, Index startCol, Index blockRows, Index blockCols) + : Base(xpr.data() + xpr.innerStride() * (XprTypeIsRowMajor ? startCol : startRow) + + xpr.outerStride() * (XprTypeIsRowMajor ? startRow : startCol), + blockRows, + blockCols), m_xpr(xpr), m_startRow(startRow), m_startCol(startCol) { init(); } EIGEN_DEVICE_FUNC - const typename internal::remove_all::type& nestedExpression() const - { - return m_xpr; - } + const typename internal::remove_all::type &nestedExpression() const { return m_xpr; } EIGEN_DEVICE_FUNC - XprType& nestedExpression() { return m_xpr; } - + XprType &nestedExpression() { return m_xpr; } + /** \sa MapBase::innerStride() */ EIGEN_DEVICE_FUNC inline Index innerStride() const { - return internal::traits::HasSameStorageOrderAsXprType - ? m_xpr.innerStride() - : m_xpr.outerStride(); + return internal::traits::HasSameStorageOrderAsXprType ? m_xpr.innerStride() : m_xpr.outerStride(); } /** \sa MapBase::outerStride() */ EIGEN_DEVICE_FUNC - inline Index outerStride() const - { - return m_outerStride; - } + inline Index outerStride() const { return m_outerStride; } EIGEN_DEVICE_FUNC - StorageIndex startRow() const - { - return m_startRow.value(); - } + StorageIndex startRow() const { return m_startRow.value(); } EIGEN_DEVICE_FUNC - StorageIndex startCol() const - { - return m_startCol.value(); - } + StorageIndex startCol() const { return m_startCol.value(); } - #ifndef __SUNPRO_CC - // FIXME sunstudio is not friendly with the above friend... - // META-FIXME there is no 'friend' keyword around here. Is this obsolete? +#ifndef __SUNPRO_CC + // FIXME sunstudio is not friendly with the above friend... + // META-FIXME there is no 'friend' keyword around here. Is this obsolete? protected: - #endif +#endif - #ifndef EIGEN_PARSED_BY_DOXYGEN +#ifndef EIGEN_PARSED_BY_DOXYGEN /** \internal used by allowAligned() */ EIGEN_DEVICE_FUNC - inline BlockImpl_dense(XprType& xpr, const Scalar* data, Index blockRows, Index blockCols) + inline BlockImpl_dense(XprType &xpr, const Scalar *data, Index blockRows, Index blockCols) : Base(data, blockRows, blockCols), m_xpr(xpr) { init(); } - #endif +#endif protected: EIGEN_DEVICE_FUNC void init() { - m_outerStride = internal::traits::HasSameStorageOrderAsXprType - ? m_xpr.outerStride() - : m_xpr.innerStride(); + m_outerStride = + internal::traits::HasSameStorageOrderAsXprType ? m_xpr.outerStride() : m_xpr.innerStride(); } XprTypeNested m_xpr; - const internal::variable_if_dynamic m_startRow; - const internal::variable_if_dynamic m_startCol; + const internal::variable_if_dynamic + m_startRow; + const internal::variable_if_dynamic + m_startCol; Index m_outerStride; -}; + }; -} // end namespace internal +}// end namespace internal -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_BLOCK_H +#endif// EIGEN_BLOCK_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/BooleanRedux.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/BooleanRedux.h index 8409d874..4dd6aab3 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/BooleanRedux.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/BooleanRedux.h @@ -10,155 +10,136 @@ #ifndef EIGEN_ALLANDANY_H #define EIGEN_ALLANDANY_H -namespace Eigen { +namespace Eigen { namespace internal { -template -struct all_unroller -{ - typedef typename Derived::ExpressionTraits Traits; - enum { - col = (UnrollCount-1) / Traits::RowsAtCompileTime, - row = (UnrollCount-1) % Traits::RowsAtCompileTime + template struct all_unroller + { + typedef typename Derived::ExpressionTraits Traits; + enum { col = (UnrollCount - 1) / Traits::RowsAtCompileTime, row = (UnrollCount - 1) % Traits::RowsAtCompileTime }; + + static inline bool run(const Derived &mat) + { + return all_unroller::run(mat) && mat.coeff(row, col); + } }; - static inline bool run(const Derived &mat) + template struct all_unroller { - return all_unroller::run(mat) && mat.coeff(row, col); - } -}; + static inline bool run(const Derived & /*mat*/) { return true; } + }; -template -struct all_unroller -{ - static inline bool run(const Derived &/*mat*/) { return true; } -}; + template struct all_unroller + { + static inline bool run(const Derived &) { return false; } + }; -template -struct all_unroller -{ - static inline bool run(const Derived &) { return false; } -}; + template struct any_unroller + { + typedef typename Derived::ExpressionTraits Traits; + enum { col = (UnrollCount - 1) / Traits::RowsAtCompileTime, row = (UnrollCount - 1) % Traits::RowsAtCompileTime }; -template -struct any_unroller -{ - typedef typename Derived::ExpressionTraits Traits; - enum { - col = (UnrollCount-1) / Traits::RowsAtCompileTime, - row = (UnrollCount-1) % Traits::RowsAtCompileTime + static inline bool run(const Derived &mat) + { + return any_unroller::run(mat) || mat.coeff(row, col); + } }; - - static inline bool run(const Derived &mat) - { - return any_unroller::run(mat) || mat.coeff(row, col); - } -}; -template -struct any_unroller -{ - static inline bool run(const Derived & /*mat*/) { return false; } -}; + template struct any_unroller + { + static inline bool run(const Derived & /*mat*/) { return false; } + }; -template -struct any_unroller -{ - static inline bool run(const Derived &) { return false; } -}; + template struct any_unroller + { + static inline bool run(const Derived &) { return false; } + }; -} // end namespace internal +}// end namespace internal /** \returns true if all coefficients are true - * - * Example: \include MatrixBase_all.cpp - * Output: \verbinclude MatrixBase_all.out - * - * \sa any(), Cwise::operator<() - */ -template -inline bool DenseBase::all() const + * + * Example: \include MatrixBase_all.cpp + * Output: \verbinclude MatrixBase_all.out + * + * \sa any(), Cwise::operator<() + */ +template inline bool DenseBase::all() const { typedef internal::evaluator Evaluator; enum { unroll = SizeAtCompileTime != Dynamic - && SizeAtCompileTime * (Evaluator::CoeffReadCost + NumTraits::AddCost) <= EIGEN_UNROLLING_LIMIT + && SizeAtCompileTime * (Evaluator::CoeffReadCost + NumTraits::AddCost) <= EIGEN_UNROLLING_LIMIT }; Evaluator evaluator(derived()); - if(unroll) - return internal::all_unroller::run(evaluator); - else - { - for(Index j = 0; j < cols(); ++j) - for(Index i = 0; i < rows(); ++i) + if (unroll) + return internal::all_unroller < Evaluator, unroll ? int(SizeAtCompileTime) : Dynamic > ::run(evaluator); + else { + for (Index j = 0; j < cols(); ++j) + for (Index i = 0; i < rows(); ++i) if (!evaluator.coeff(i, j)) return false; return true; } } /** \returns true if at least one coefficient is true - * - * \sa all() - */ -template -inline bool DenseBase::any() const + * + * \sa all() + */ +template inline bool DenseBase::any() const { typedef internal::evaluator Evaluator; enum { unroll = SizeAtCompileTime != Dynamic - && SizeAtCompileTime * (Evaluator::CoeffReadCost + NumTraits::AddCost) <= EIGEN_UNROLLING_LIMIT + && SizeAtCompileTime * (Evaluator::CoeffReadCost + NumTraits::AddCost) <= EIGEN_UNROLLING_LIMIT }; Evaluator evaluator(derived()); - if(unroll) - return internal::any_unroller::run(evaluator); - else - { - for(Index j = 0; j < cols(); ++j) - for(Index i = 0; i < rows(); ++i) + if (unroll) + return internal::any_unroller < Evaluator, unroll ? int(SizeAtCompileTime) : Dynamic > ::run(evaluator); + else { + for (Index j = 0; j < cols(); ++j) + for (Index i = 0; i < rows(); ++i) if (evaluator.coeff(i, j)) return true; return false; } } /** \returns the number of coefficients which evaluate to true - * - * \sa all(), any() - */ -template -inline Eigen::Index DenseBase::count() const + * + * \sa all(), any() + */ +template inline Eigen::Index DenseBase::count() const { return derived().template cast().template cast().sum(); } /** \returns true is \c *this contains at least one Not A Number (NaN). - * - * \sa allFinite() - */ -template -inline bool DenseBase::hasNaN() const + * + * \sa allFinite() + */ +template inline bool DenseBase::hasNaN() const { #if EIGEN_COMP_MSVC || (defined __FAST_MATH__) return derived().array().isNaN().any(); #else - return !((derived().array()==derived().array()).all()); + return !((derived().array() == derived().array()).all()); #endif } /** \returns true if \c *this contains only finite numbers, i.e., no NaN and no +/-INF values. - * - * \sa hasNaN() - */ -template -inline bool DenseBase::allFinite() const + * + * \sa hasNaN() + */ +template inline bool DenseBase::allFinite() const { #if EIGEN_COMP_MSVC || (defined __FAST_MATH__) return derived().array().isFinite().all(); #else - return !((derived()-derived()).hasNaN()); + return !((derived() - derived()).hasNaN()); #endif } - -} // end namespace Eigen -#endif // EIGEN_ALLANDANY_H +}// end namespace Eigen + +#endif// EIGEN_ALLANDANY_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/CommaInitializer.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/CommaInitializer.h index d218e981..1e6351ae 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/CommaInitializer.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/CommaInitializer.h @@ -11,88 +11,80 @@ #ifndef EIGEN_COMMAINITIALIZER_H #define EIGEN_COMMAINITIALIZER_H -namespace Eigen { +namespace Eigen { /** \class CommaInitializer - * \ingroup Core_Module - * - * \brief Helper class used by the comma initializer operator - * - * This class is internally used to implement the comma initializer feature. It is - * the return type of MatrixBase::operator<<, and most of the time this is the only - * way it is used. - * - * \sa \blank \ref MatrixBaseCommaInitRef "MatrixBase::operator<<", CommaInitializer::finished() - */ -template -struct CommaInitializer + * \ingroup Core_Module + * + * \brief Helper class used by the comma initializer operator + * + * This class is internally used to implement the comma initializer feature. It is + * the return type of MatrixBase::operator<<, and most of the time this is the only + * way it is used. + * + * \sa \blank \ref MatrixBaseCommaInitRef "MatrixBase::operator<<", CommaInitializer::finished() + */ +template struct CommaInitializer { typedef typename XprType::Scalar Scalar; EIGEN_DEVICE_FUNC - inline CommaInitializer(XprType& xpr, const Scalar& s) - : m_xpr(xpr), m_row(0), m_col(1), m_currentBlockRows(1) + inline CommaInitializer(XprType &xpr, const Scalar &s) : m_xpr(xpr), m_row(0), m_col(1), m_currentBlockRows(1) { - m_xpr.coeffRef(0,0) = s; + m_xpr.coeffRef(0, 0) = s; } template - EIGEN_DEVICE_FUNC - inline CommaInitializer(XprType& xpr, const DenseBase& other) + EIGEN_DEVICE_FUNC inline CommaInitializer(XprType &xpr, const DenseBase &other) : m_xpr(xpr), m_row(0), m_col(other.cols()), m_currentBlockRows(other.rows()) { m_xpr.block(0, 0, other.rows(), other.cols()) = other; } - /* Copy/Move constructor which transfers ownership. This is crucial in + /* Copy/Move constructor which transfers ownership. This is crucial in * absence of return value optimization to avoid assertions during destruction. */ // FIXME in C++11 mode this could be replaced by a proper RValue constructor EIGEN_DEVICE_FUNC - inline CommaInitializer(const CommaInitializer& o) - : m_xpr(o.m_xpr), m_row(o.m_row), m_col(o.m_col), m_currentBlockRows(o.m_currentBlockRows) { + inline CommaInitializer(const CommaInitializer &o) + : m_xpr(o.m_xpr), m_row(o.m_row), m_col(o.m_col), m_currentBlockRows(o.m_currentBlockRows) + { // Mark original object as finished. In absence of R-value references we need to const_cast: - const_cast(o).m_row = m_xpr.rows(); - const_cast(o).m_col = m_xpr.cols(); - const_cast(o).m_currentBlockRows = 0; + const_cast(o).m_row = m_xpr.rows(); + const_cast(o).m_col = m_xpr.cols(); + const_cast(o).m_currentBlockRows = 0; } /* inserts a scalar value in the target matrix */ EIGEN_DEVICE_FUNC - CommaInitializer& operator,(const Scalar& s) + CommaInitializer &operator,(const Scalar &s) { - if (m_col==m_xpr.cols()) - { - m_row+=m_currentBlockRows; + if (m_col == m_xpr.cols()) { + m_row += m_currentBlockRows; m_col = 0; m_currentBlockRows = 1; - eigen_assert(m_row - EIGEN_DEVICE_FUNC - CommaInitializer& operator,(const DenseBase& other) + template EIGEN_DEVICE_FUNC CommaInitializer &operator,(const DenseBase &other) { - if (m_col==m_xpr.cols() && (other.cols()!=0 || other.rows()!=m_currentBlockRows)) - { - m_row+=m_currentBlockRows; + if (m_col == m_xpr.cols() && (other.cols() != 0 || other.rows() != m_currentBlockRows)) { + m_row += m_currentBlockRows; m_col = 0; m_currentBlockRows = other.rows(); - eigen_assert(m_row+m_currentBlockRows<=m_xpr.rows() - && "Too many rows passed to comma initializer (operator<<)"); + eigen_assert( + m_row + m_currentBlockRows <= m_xpr.rows() && "Too many rows passed to comma initializer (operator<<)"); } - eigen_assert((m_col + other.cols() <= m_xpr.cols()) - && "Too many coefficients passed to comma initializer (operator<<)"); - eigen_assert(m_currentBlockRows==other.rows()); - m_xpr.template block - (m_row, m_col, other.rows(), other.cols()) = other; + eigen_assert( + (m_col + other.cols() <= m_xpr.cols()) && "Too many coefficients passed to comma initializer (operator<<)"); + eigen_assert(m_currentBlockRows == other.rows()); + m_xpr.template block( + m_row, m_col, other.rows(), other.cols()) = other; m_col += other.cols(); return *this; } @@ -100,61 +92,60 @@ struct CommaInitializer EIGEN_DEVICE_FUNC inline ~CommaInitializer() #if defined VERIFY_RAISES_ASSERT && (!defined EIGEN_NO_ASSERTION_CHECKING) && defined EIGEN_EXCEPTIONS - EIGEN_EXCEPTION_SPEC(Eigen::eigen_assert_exception) + EIGEN_EXCEPTION_SPEC(Eigen::eigen_assert_exception) #endif { - finished(); + finished(); } /** \returns the built matrix once all its coefficients have been set. - * Calling finished is 100% optional. Its purpose is to write expressions - * like this: - * \code - * quaternion.fromRotationMatrix((Matrix3f() << axis0, axis1, axis2).finished()); - * \endcode - */ + * Calling finished is 100% optional. Its purpose is to write expressions + * like this: + * \code + * quaternion.fromRotationMatrix((Matrix3f() << axis0, axis1, axis2).finished()); + * \endcode + */ EIGEN_DEVICE_FUNC - inline XprType& finished() { - eigen_assert(((m_row+m_currentBlockRows) == m_xpr.rows() || m_xpr.cols() == 0) - && m_col == m_xpr.cols() - && "Too few coefficients passed to comma initializer (operator<<)"); - return m_xpr; + inline XprType &finished() + { + eigen_assert(((m_row + m_currentBlockRows) == m_xpr.rows() || m_xpr.cols() == 0) && m_col == m_xpr.cols() + && "Too few coefficients passed to comma initializer (operator<<)"); + return m_xpr; } - XprType& m_xpr; // target expression - Index m_row; // current row id - Index m_col; // current col id - Index m_currentBlockRows; // current block height + XprType &m_xpr;// target expression + Index m_row;// current row id + Index m_col;// current col id + Index m_currentBlockRows;// current block height }; /** \anchor MatrixBaseCommaInitRef - * Convenient operator to set the coefficients of a matrix. - * - * The coefficients must be provided in a row major order and exactly match - * the size of the matrix. Otherwise an assertion is raised. - * - * Example: \include MatrixBase_set.cpp - * Output: \verbinclude MatrixBase_set.out - * - * \note According the c++ standard, the argument expressions of this comma initializer are evaluated in arbitrary order. - * - * \sa CommaInitializer::finished(), class CommaInitializer - */ -template -inline CommaInitializer DenseBase::operator<< (const Scalar& s) + * Convenient operator to set the coefficients of a matrix. + * + * The coefficients must be provided in a row major order and exactly match + * the size of the matrix. Otherwise an assertion is raised. + * + * Example: \include MatrixBase_set.cpp + * Output: \verbinclude MatrixBase_set.out + * + * \note According the c++ standard, the argument expressions of this comma initializer are evaluated in arbitrary + * order. + * + * \sa CommaInitializer::finished(), class CommaInitializer + */ +template inline CommaInitializer DenseBase::operator<<(const Scalar &s) { - return CommaInitializer(*static_cast(this), s); + return CommaInitializer(*static_cast(this), s); } /** \sa operator<<(const Scalar&) */ template template -inline CommaInitializer -DenseBase::operator<<(const DenseBase& other) +inline CommaInitializer DenseBase::operator<<(const DenseBase &other) { return CommaInitializer(*static_cast(this), other); } -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_COMMAINITIALIZER_H +#endif// EIGEN_COMMAINITIALIZER_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/ConditionEstimator.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/ConditionEstimator.h index 51a2e5f1..6fc5ea59 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/ConditionEstimator.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/ConditionEstimator.h @@ -14,162 +14,158 @@ namespace Eigen { namespace internal { -template -struct rcond_compute_sign { - static inline Vector run(const Vector& v) { - const RealVector v_abs = v.cwiseAbs(); - return (v_abs.array() == static_cast(0)) - .select(Vector::Ones(v.size()), v.cwiseQuotient(v_abs)); - } -}; + template struct rcond_compute_sign + { + static inline Vector run(const Vector &v) + { + const RealVector v_abs = v.cwiseAbs(); + return (v_abs.array() == static_cast(0)) + .select(Vector::Ones(v.size()), v.cwiseQuotient(v_abs)); + } + }; -// Partial specialization to avoid elementwise division for real vectors. -template -struct rcond_compute_sign { - static inline Vector run(const Vector& v) { - return (v.array() < static_cast(0)) - .select(-Vector::Ones(v.size()), Vector::Ones(v.size())); - } -}; + // Partial specialization to avoid elementwise division for real vectors. + template struct rcond_compute_sign + { + static inline Vector run(const Vector &v) + { + return (v.array() < static_cast(0)) + .select(-Vector::Ones(v.size()), Vector::Ones(v.size())); + } + }; -/** - * \returns an estimate of ||inv(matrix)||_1 given a decomposition of - * \a matrix that implements .solve() and .adjoint().solve() methods. - * - * This function implements Algorithms 4.1 and 5.1 from - * http://www.maths.manchester.ac.uk/~higham/narep/narep135.pdf - * which also forms the basis for the condition number estimators in - * LAPACK. Since at most 10 calls to the solve method of dec are - * performed, the total cost is O(dims^2), as opposed to O(dims^3) - * needed to compute the inverse matrix explicitly. - * - * The most common usage is in estimating the condition number - * ||matrix||_1 * ||inv(matrix)||_1. The first term ||matrix||_1 can be - * computed directly in O(n^2) operations. - * - * Supports the following decompositions: FullPivLU, PartialPivLU, LDLT, and - * LLT. - * - * \sa FullPivLU, PartialPivLU, LDLT, LLT. - */ -template -typename Decomposition::RealScalar rcond_invmatrix_L1_norm_estimate(const Decomposition& dec) -{ - typedef typename Decomposition::MatrixType MatrixType; - typedef typename Decomposition::Scalar Scalar; - typedef typename Decomposition::RealScalar RealScalar; - typedef typename internal::plain_col_type::type Vector; - typedef typename internal::plain_col_type::type RealVector; - const bool is_complex = (NumTraits::IsComplex != 0); + /** + * \returns an estimate of ||inv(matrix)||_1 given a decomposition of + * \a matrix that implements .solve() and .adjoint().solve() methods. + * + * This function implements Algorithms 4.1 and 5.1 from + * http://www.maths.manchester.ac.uk/~higham/narep/narep135.pdf + * which also forms the basis for the condition number estimators in + * LAPACK. Since at most 10 calls to the solve method of dec are + * performed, the total cost is O(dims^2), as opposed to O(dims^3) + * needed to compute the inverse matrix explicitly. + * + * The most common usage is in estimating the condition number + * ||matrix||_1 * ||inv(matrix)||_1. The first term ||matrix||_1 can be + * computed directly in O(n^2) operations. + * + * Supports the following decompositions: FullPivLU, PartialPivLU, LDLT, and + * LLT. + * + * \sa FullPivLU, PartialPivLU, LDLT, LLT. + */ + template + typename Decomposition::RealScalar rcond_invmatrix_L1_norm_estimate(const Decomposition &dec) + { + typedef typename Decomposition::MatrixType MatrixType; + typedef typename Decomposition::Scalar Scalar; + typedef typename Decomposition::RealScalar RealScalar; + typedef typename internal::plain_col_type::type Vector; + typedef typename internal::plain_col_type::type RealVector; + const bool is_complex = (NumTraits::IsComplex != 0); - eigen_assert(dec.rows() == dec.cols()); - const Index n = dec.rows(); - if (n == 0) - return 0; + eigen_assert(dec.rows() == dec.cols()); + const Index n = dec.rows(); + if (n == 0) return 0; - // Disable Index to float conversion warning + // Disable Index to float conversion warning #ifdef __INTEL_COMPILER - #pragma warning push - #pragma warning ( disable : 2259 ) +#pragma warning push +#pragma warning(disable : 2259) #endif - Vector v = dec.solve(Vector::Ones(n) / Scalar(n)); + Vector v = dec.solve(Vector::Ones(n) / Scalar(n)); #ifdef __INTEL_COMPILER - #pragma warning pop +#pragma warning pop #endif - // lower_bound is a lower bound on - // ||inv(matrix)||_1 = sup_v ||inv(matrix) v||_1 / ||v||_1 - // and is the objective maximized by the ("super-") gradient ascent - // algorithm below. - RealScalar lower_bound = v.template lpNorm<1>(); - if (n == 1) - return lower_bound; + // lower_bound is a lower bound on + // ||inv(matrix)||_1 = sup_v ||inv(matrix) v||_1 / ||v||_1 + // and is the objective maximized by the ("super-") gradient ascent + // algorithm below. + RealScalar lower_bound = v.template lpNorm<1>(); + if (n == 1) return lower_bound; - // Gradient ascent algorithm follows: We know that the optimum is achieved at - // one of the simplices v = e_i, so in each iteration we follow a - // super-gradient to move towards the optimal one. - RealScalar old_lower_bound = lower_bound; - Vector sign_vector(n); - Vector old_sign_vector; - Index v_max_abs_index = -1; - Index old_v_max_abs_index = v_max_abs_index; - for (int k = 0; k < 4; ++k) - { - sign_vector = internal::rcond_compute_sign::run(v); - if (k > 0 && !is_complex && sign_vector == old_sign_vector) { - // Break if the solution stagnated. - break; - } - // v_max_abs_index = argmax |real( inv(matrix)^T * sign_vector )| - v = dec.adjoint().solve(sign_vector); - v.real().cwiseAbs().maxCoeff(&v_max_abs_index); - if (v_max_abs_index == old_v_max_abs_index) { - // Break if the solution stagnated. - break; - } - // Move to the new simplex e_j, where j = v_max_abs_index. - v = dec.solve(Vector::Unit(n, v_max_abs_index)); // v = inv(matrix) * e_j. - lower_bound = v.template lpNorm<1>(); - if (lower_bound <= old_lower_bound) { - // Break if the gradient step did not increase the lower_bound. - break; + // Gradient ascent algorithm follows: We know that the optimum is achieved at + // one of the simplices v = e_i, so in each iteration we follow a + // super-gradient to move towards the optimal one. + RealScalar old_lower_bound = lower_bound; + Vector sign_vector(n); + Vector old_sign_vector; + Index v_max_abs_index = -1; + Index old_v_max_abs_index = v_max_abs_index; + for (int k = 0; k < 4; ++k) { + sign_vector = internal::rcond_compute_sign::run(v); + if (k > 0 && !is_complex && sign_vector == old_sign_vector) { + // Break if the solution stagnated. + break; + } + // v_max_abs_index = argmax |real( inv(matrix)^T * sign_vector )| + v = dec.adjoint().solve(sign_vector); + v.real().cwiseAbs().maxCoeff(&v_max_abs_index); + if (v_max_abs_index == old_v_max_abs_index) { + // Break if the solution stagnated. + break; + } + // Move to the new simplex e_j, where j = v_max_abs_index. + v = dec.solve(Vector::Unit(n, v_max_abs_index));// v = inv(matrix) * e_j. + lower_bound = v.template lpNorm<1>(); + if (lower_bound <= old_lower_bound) { + // Break if the gradient step did not increase the lower_bound. + break; + } + if (!is_complex) { old_sign_vector = sign_vector; } + old_v_max_abs_index = v_max_abs_index; + old_lower_bound = lower_bound; } - if (!is_complex) { - old_sign_vector = sign_vector; + // The following calculates an independent estimate of ||matrix||_1 by + // multiplying matrix by a vector with entries of slowly increasing + // magnitude and alternating sign: + // v_i = (-1)^{i} (1 + (i / (dim-1))), i = 0,...,dim-1. + // This improvement to Hager's algorithm above is due to Higham. It was + // added to make the algorithm more robust in certain corner cases where + // large elements in the matrix might otherwise escape detection due to + // exact cancellation (especially when op and op_adjoint correspond to a + // sequence of backsubstitutions and permutations), which could cause + // Hager's algorithm to vastly underestimate ||matrix||_1. + Scalar alternating_sign(RealScalar(1)); + for (Index i = 0; i < n; ++i) { + // The static_cast is needed when Scalar is a complex and RealScalar implements expression templates + v[i] = alternating_sign * static_cast(RealScalar(1) + (RealScalar(i) / (RealScalar(n - 1)))); + alternating_sign = -alternating_sign; } - old_v_max_abs_index = v_max_abs_index; - old_lower_bound = lower_bound; + v = dec.solve(v); + const RealScalar alternate_lower_bound = (2 * v.template lpNorm<1>()) / (3 * RealScalar(n)); + return numext::maxi(lower_bound, alternate_lower_bound); } - // The following calculates an independent estimate of ||matrix||_1 by - // multiplying matrix by a vector with entries of slowly increasing - // magnitude and alternating sign: - // v_i = (-1)^{i} (1 + (i / (dim-1))), i = 0,...,dim-1. - // This improvement to Hager's algorithm above is due to Higham. It was - // added to make the algorithm more robust in certain corner cases where - // large elements in the matrix might otherwise escape detection due to - // exact cancellation (especially when op and op_adjoint correspond to a - // sequence of backsubstitutions and permutations), which could cause - // Hager's algorithm to vastly underestimate ||matrix||_1. - Scalar alternating_sign(RealScalar(1)); - for (Index i = 0; i < n; ++i) { - // The static_cast is needed when Scalar is a complex and RealScalar implements expression templates - v[i] = alternating_sign * static_cast(RealScalar(1) + (RealScalar(i) / (RealScalar(n - 1)))); - alternating_sign = -alternating_sign; - } - v = dec.solve(v); - const RealScalar alternate_lower_bound = (2 * v.template lpNorm<1>()) / (3 * RealScalar(n)); - return numext::maxi(lower_bound, alternate_lower_bound); -} -/** \brief Reciprocal condition number estimator. - * - * Computing a decomposition of a dense matrix takes O(n^3) operations, while - * this method estimates the condition number quickly and reliably in O(n^2) - * operations. - * - * \returns an estimate of the reciprocal condition number - * (1 / (||matrix||_1 * ||inv(matrix)||_1)) of matrix, given ||matrix||_1 and - * its decomposition. Supports the following decompositions: FullPivLU, - * PartialPivLU, LDLT, and LLT. - * - * \sa FullPivLU, PartialPivLU, LDLT, LLT. - */ -template -typename Decomposition::RealScalar -rcond_estimate_helper(typename Decomposition::RealScalar matrix_norm, const Decomposition& dec) -{ - typedef typename Decomposition::RealScalar RealScalar; - eigen_assert(dec.rows() == dec.cols()); - if (dec.rows() == 0) return NumTraits::infinity(); - if (matrix_norm == RealScalar(0)) return RealScalar(0); - if (dec.rows() == 1) return RealScalar(1); - const RealScalar inverse_matrix_norm = rcond_invmatrix_L1_norm_estimate(dec); - return (inverse_matrix_norm == RealScalar(0) ? RealScalar(0) - : (RealScalar(1) / inverse_matrix_norm) / matrix_norm); -} + /** \brief Reciprocal condition number estimator. + * + * Computing a decomposition of a dense matrix takes O(n^3) operations, while + * this method estimates the condition number quickly and reliably in O(n^2) + * operations. + * + * \returns an estimate of the reciprocal condition number + * (1 / (||matrix||_1 * ||inv(matrix)||_1)) of matrix, given ||matrix||_1 and + * its decomposition. Supports the following decompositions: FullPivLU, + * PartialPivLU, LDLT, and LLT. + * + * \sa FullPivLU, PartialPivLU, LDLT, LLT. + */ + template + typename Decomposition::RealScalar rcond_estimate_helper(typename Decomposition::RealScalar matrix_norm, + const Decomposition &dec) + { + typedef typename Decomposition::RealScalar RealScalar; + eigen_assert(dec.rows() == dec.cols()); + if (dec.rows() == 0) return NumTraits::infinity(); + if (matrix_norm == RealScalar(0)) return RealScalar(0); + if (dec.rows() == 1) return RealScalar(1); + const RealScalar inverse_matrix_norm = rcond_invmatrix_L1_norm_estimate(dec); + return (inverse_matrix_norm == RealScalar(0) ? RealScalar(0) : (RealScalar(1) / inverse_matrix_norm) / matrix_norm); + } -} // namespace internal +}// namespace internal -} // namespace Eigen +}// namespace Eigen #endif diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/CoreEvaluators.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/CoreEvaluators.h index 910889ef..0291bdaf 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/CoreEvaluators.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/CoreEvaluators.h @@ -14,386 +14,394 @@ #define EIGEN_COREEVALUATORS_H namespace Eigen { - -namespace internal { - -// This class returns the evaluator kind from the expression storage kind. -// Default assumes index based accessors -template -struct storage_kind_to_evaluator_kind { - typedef IndexBased Kind; -}; - -// This class returns the evaluator shape from the expression storage kind. -// It can be Dense, Sparse, Triangular, Diagonal, SelfAdjoint, Band, etc. -template struct storage_kind_to_shape; - -template<> struct storage_kind_to_shape { typedef DenseShape Shape; }; -template<> struct storage_kind_to_shape { typedef SolverShape Shape; }; -template<> struct storage_kind_to_shape { typedef PermutationShape Shape; }; -template<> struct storage_kind_to_shape { typedef TranspositionsShape Shape; }; - -// Evaluators have to be specialized with respect to various criteria such as: -// - storage/structure/shape -// - scalar type -// - etc. -// Therefore, we need specialization of evaluator providing additional template arguments for each kind of evaluators. -// We currently distinguish the following kind of evaluators: -// - unary_evaluator for expressions taking only one arguments (CwiseUnaryOp, CwiseUnaryView, Transpose, MatrixWrapper, ArrayWrapper, Reverse, Replicate) -// - binary_evaluator for expression taking two arguments (CwiseBinaryOp) -// - ternary_evaluator for expression taking three arguments (CwiseTernaryOp) -// - product_evaluator for linear algebra products (Product); special case of binary_evaluator because it requires additional tags for dispatching. -// - mapbase_evaluator for Map, Block, Ref -// - block_evaluator for Block (special dispatching to a mapbase_evaluator or unary_evaluator) - -template< typename T, - typename Arg1Kind = typename evaluator_traits::Kind, - typename Arg2Kind = typename evaluator_traits::Kind, - typename Arg3Kind = typename evaluator_traits::Kind, - typename Arg1Scalar = typename traits::Scalar, - typename Arg2Scalar = typename traits::Scalar, - typename Arg3Scalar = typename traits::Scalar> struct ternary_evaluator; - -template< typename T, - typename LhsKind = typename evaluator_traits::Kind, - typename RhsKind = typename evaluator_traits::Kind, - typename LhsScalar = typename traits::Scalar, - typename RhsScalar = typename traits::Scalar> struct binary_evaluator; - -template< typename T, - typename Kind = typename evaluator_traits::Kind, - typename Scalar = typename T::Scalar> struct unary_evaluator; - -// evaluator_traits contains traits for evaluator - -template -struct evaluator_traits_base -{ - // by default, get evaluator kind and shape from storage - typedef typename storage_kind_to_evaluator_kind::StorageKind>::Kind Kind; - typedef typename storage_kind_to_shape::StorageKind>::Shape Shape; -}; - -// Default evaluator traits -template -struct evaluator_traits : public evaluator_traits_base -{ -}; - -template::Shape > -struct evaluator_assume_aliasing { - static const bool value = false; -}; - -// By default, we assume a unary expression: -template -struct evaluator : public unary_evaluator -{ - typedef unary_evaluator Base; - EIGEN_DEVICE_FUNC explicit evaluator(const T& xpr) : Base(xpr) {} -}; +namespace internal { -// TODO: Think about const-correctness -template -struct evaluator - : evaluator -{ - EIGEN_DEVICE_FUNC - explicit evaluator(const T& xpr) : evaluator(xpr) {} -}; + // This class returns the evaluator kind from the expression storage kind. + // Default assumes index based accessors + template struct storage_kind_to_evaluator_kind + { + typedef IndexBased Kind; + }; -// ---------- base class for all evaluators ---------- + // This class returns the evaluator shape from the expression storage kind. + // It can be Dense, Sparse, Triangular, Diagonal, SelfAdjoint, Band, etc. + template struct storage_kind_to_shape; -template -struct evaluator_base : public noncopyable -{ - // TODO that's not very nice to have to propagate all these traits. They are currently only needed to handle outer,inner indices. - typedef traits ExpressionTraits; - - enum { - Alignment = 0 + template<> struct storage_kind_to_shape + { + typedef DenseShape Shape; }; -}; - -// -------------------- Matrix and Array -------------------- -// -// evaluator is a common base class for the -// Matrix and Array evaluators. -// Here we directly specialize evaluator. This is not really a unary expression, and it is, by definition, dense, -// so no need for more sophisticated dispatching. - -template -struct evaluator > - : evaluator_base -{ - typedef PlainObjectBase PlainObjectType; - typedef typename PlainObjectType::Scalar Scalar; - typedef typename PlainObjectType::CoeffReturnType CoeffReturnType; - - enum { - IsRowMajor = PlainObjectType::IsRowMajor, - IsVectorAtCompileTime = PlainObjectType::IsVectorAtCompileTime, - RowsAtCompileTime = PlainObjectType::RowsAtCompileTime, - ColsAtCompileTime = PlainObjectType::ColsAtCompileTime, - - CoeffReadCost = NumTraits::ReadCost, - Flags = traits::EvaluatorFlags, - Alignment = traits::Alignment + template<> struct storage_kind_to_shape + { + typedef SolverShape Shape; }; - - EIGEN_DEVICE_FUNC evaluator() - : m_data(0), - m_outerStride(IsVectorAtCompileTime ? 0 - : int(IsRowMajor) ? ColsAtCompileTime - : RowsAtCompileTime) + template<> struct storage_kind_to_shape { - EIGEN_INTERNAL_CHECK_COST_VALUE(CoeffReadCost); - } - - EIGEN_DEVICE_FUNC explicit evaluator(const PlainObjectType& m) - : m_data(m.data()), m_outerStride(IsVectorAtCompileTime ? 0 : m.outerStride()) + typedef PermutationShape Shape; + }; + template<> struct storage_kind_to_shape { - EIGEN_INTERNAL_CHECK_COST_VALUE(CoeffReadCost); - } + typedef TranspositionsShape Shape; + }; - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE - CoeffReturnType coeff(Index row, Index col) const - { - if (IsRowMajor) - return m_data[row * m_outerStride.value() + col]; - else - return m_data[row + col * m_outerStride.value()]; - } + // Evaluators have to be specialized with respect to various criteria such as: + // - storage/structure/shape + // - scalar type + // - etc. + // Therefore, we need specialization of evaluator providing additional template arguments for each kind of evaluators. + // We currently distinguish the following kind of evaluators: + // - unary_evaluator for expressions taking only one arguments (CwiseUnaryOp, CwiseUnaryView, Transpose, + // MatrixWrapper, ArrayWrapper, Reverse, Replicate) + // - binary_evaluator for expression taking two arguments (CwiseBinaryOp) + // - ternary_evaluator for expression taking three arguments (CwiseTernaryOp) + // - product_evaluator for linear algebra products (Product); special case of binary_evaluator because it requires + // additional tags for dispatching. + // - mapbase_evaluator for Map, Block, Ref + // - block_evaluator for Block (special dispatching to a mapbase_evaluator or unary_evaluator) + + template::Kind, + typename Arg2Kind = typename evaluator_traits::Kind, + typename Arg3Kind = typename evaluator_traits::Kind, + typename Arg1Scalar = typename traits::Scalar, + typename Arg2Scalar = typename traits::Scalar, + typename Arg3Scalar = typename traits::Scalar> + struct ternary_evaluator; + + template::Kind, + typename RhsKind = typename evaluator_traits::Kind, + typename LhsScalar = typename traits::Scalar, + typename RhsScalar = typename traits::Scalar> + struct binary_evaluator; + + template::Kind, + typename Scalar = typename T::Scalar> + struct unary_evaluator; + + // evaluator_traits contains traits for evaluator + + template struct evaluator_traits_base + { + // by default, get evaluator kind and shape from storage + typedef typename storage_kind_to_evaluator_kind::StorageKind>::Kind Kind; + typedef typename storage_kind_to_shape::StorageKind>::Shape Shape; + }; - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE - CoeffReturnType coeff(Index index) const + // Default evaluator traits + template struct evaluator_traits : public evaluator_traits_base { - return m_data[index]; - } + }; - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE - Scalar& coeffRef(Index row, Index col) + template::Shape> struct evaluator_assume_aliasing { - if (IsRowMajor) - return const_cast(m_data)[row * m_outerStride.value() + col]; - else - return const_cast(m_data)[row + col * m_outerStride.value()]; - } + static const bool value = false; + }; - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE - Scalar& coeffRef(Index index) + // By default, we assume a unary expression: + template struct evaluator : public unary_evaluator { - return const_cast(m_data)[index]; - } + typedef unary_evaluator Base; + EIGEN_DEVICE_FUNC explicit evaluator(const T &xpr) : Base(xpr) {} + }; - template - EIGEN_STRONG_INLINE - PacketType packet(Index row, Index col) const - { - if (IsRowMajor) - return ploadt(m_data + row * m_outerStride.value() + col); - else - return ploadt(m_data + row + col * m_outerStride.value()); - } - template - EIGEN_STRONG_INLINE - PacketType packet(Index index) const + // TODO: Think about const-correctness + template struct evaluator : evaluator { - return ploadt(m_data + index); - } + EIGEN_DEVICE_FUNC + explicit evaluator(const T &xpr) : evaluator(xpr) {} + }; - template - EIGEN_STRONG_INLINE - void writePacket(Index row, Index col, const PacketType& x) - { - if (IsRowMajor) - return pstoret - (const_cast(m_data) + row * m_outerStride.value() + col, x); - else - return pstoret - (const_cast(m_data) + row + col * m_outerStride.value(), x); - } + // ---------- base class for all evaluators ---------- - template - EIGEN_STRONG_INLINE - void writePacket(Index index, const PacketType& x) + template struct evaluator_base : public noncopyable { - return pstoret(const_cast(m_data) + index, x); - } + // TODO that's not very nice to have to propagate all these traits. They are currently only needed to handle + // outer,inner indices. + typedef traits ExpressionTraits; -protected: - const Scalar *m_data; + enum { Alignment = 0 }; + }; - // We do not need to know the outer stride for vectors - variable_if_dynamic m_outerStride; -}; + // -------------------- Matrix and Array -------------------- + // + // evaluator is a common base class for the + // Matrix and Array evaluators. + // Here we directly specialize evaluator. This is not really a unary expression, and it is, by definition, dense, + // so no need for more sophisticated dispatching. -template -struct evaluator > - : evaluator > > -{ - typedef Matrix XprType; - - EIGEN_DEVICE_FUNC evaluator() {} + template struct evaluator> : evaluator_base + { + typedef PlainObjectBase PlainObjectType; + typedef typename PlainObjectType::Scalar Scalar; + typedef typename PlainObjectType::CoeffReturnType CoeffReturnType; - EIGEN_DEVICE_FUNC explicit evaluator(const XprType& m) - : evaluator >(m) - { } -}; + enum { + IsRowMajor = PlainObjectType::IsRowMajor, + IsVectorAtCompileTime = PlainObjectType::IsVectorAtCompileTime, + RowsAtCompileTime = PlainObjectType::RowsAtCompileTime, + ColsAtCompileTime = PlainObjectType::ColsAtCompileTime, + + CoeffReadCost = NumTraits::ReadCost, + Flags = traits::EvaluatorFlags, + Alignment = traits::Alignment + }; -template -struct evaluator > - : evaluator > > -{ - typedef Array XprType; + EIGEN_DEVICE_FUNC evaluator() + : m_data(0), m_outerStride(IsVectorAtCompileTime ? 0 + : int(IsRowMajor) ? ColsAtCompileTime + : RowsAtCompileTime) + { + EIGEN_INTERNAL_CHECK_COST_VALUE(CoeffReadCost); + } + + EIGEN_DEVICE_FUNC explicit evaluator(const PlainObjectType &m) + : m_data(m.data()), m_outerStride(IsVectorAtCompileTime ? 0 : m.outerStride()) + { + EIGEN_INTERNAL_CHECK_COST_VALUE(CoeffReadCost); + } + + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE CoeffReturnType coeff(Index row, Index col) const + { + if (IsRowMajor) + return m_data[row * m_outerStride.value() + col]; + else + return m_data[row + col * m_outerStride.value()]; + } + + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE CoeffReturnType coeff(Index index) const { return m_data[index]; } + + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Scalar &coeffRef(Index row, Index col) + { + if (IsRowMajor) + return const_cast(m_data)[row * m_outerStride.value() + col]; + else + return const_cast(m_data)[row + col * m_outerStride.value()]; + } + + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Scalar &coeffRef(Index index) { return const_cast(m_data)[index]; } + + template EIGEN_STRONG_INLINE PacketType packet(Index row, Index col) const + { + if (IsRowMajor) + return ploadt(m_data + row * m_outerStride.value() + col); + else + return ploadt(m_data + row + col * m_outerStride.value()); + } + + template EIGEN_STRONG_INLINE PacketType packet(Index index) const + { + return ploadt(m_data + index); + } + + template + EIGEN_STRONG_INLINE void writePacket(Index row, Index col, const PacketType &x) + { + if (IsRowMajor) + return pstoret( + const_cast(m_data) + row * m_outerStride.value() + col, x); + else + return pstoret( + const_cast(m_data) + row + col * m_outerStride.value(), x); + } + + template EIGEN_STRONG_INLINE void writePacket(Index index, const PacketType &x) + { + return pstoret(const_cast(m_data) + index, x); + } + + protected: + const Scalar *m_data; + + // We do not need to know the outer stride for vectors + variable_if_dynamic + m_outerStride; + }; - EIGEN_DEVICE_FUNC evaluator() {} - - EIGEN_DEVICE_FUNC explicit evaluator(const XprType& m) - : evaluator >(m) - { } -}; + template + struct evaluator> + : evaluator>> + { + typedef Matrix XprType; -// -------------------- Transpose -------------------- + EIGEN_DEVICE_FUNC evaluator() {} -template -struct unary_evaluator, IndexBased> - : evaluator_base > -{ - typedef Transpose XprType; - - enum { - CoeffReadCost = evaluator::CoeffReadCost, - Flags = evaluator::Flags ^ RowMajorBit, - Alignment = evaluator::Alignment + EIGEN_DEVICE_FUNC explicit evaluator(const XprType &m) : evaluator>(m) {} }; - EIGEN_DEVICE_FUNC explicit unary_evaluator(const XprType& t) : m_argImpl(t.nestedExpression()) {} + template + struct evaluator> + : evaluator>> + { + typedef Array XprType; - typedef typename XprType::Scalar Scalar; - typedef typename XprType::CoeffReturnType CoeffReturnType; + EIGEN_DEVICE_FUNC evaluator() {} - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE - CoeffReturnType coeff(Index row, Index col) const - { - return m_argImpl.coeff(col, row); - } + EIGEN_DEVICE_FUNC explicit evaluator(const XprType &m) : evaluator>(m) {} + }; - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE - CoeffReturnType coeff(Index index) const - { - return m_argImpl.coeff(index); - } + // -------------------- Transpose -------------------- - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE - Scalar& coeffRef(Index row, Index col) + template struct unary_evaluator, IndexBased> : evaluator_base> { - return m_argImpl.coeffRef(col, row); - } + typedef Transpose XprType; - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE - typename XprType::Scalar& coeffRef(Index index) - { - return m_argImpl.coeffRef(index); - } + enum { + CoeffReadCost = evaluator::CoeffReadCost, + Flags = evaluator::Flags ^ RowMajorBit, + Alignment = evaluator::Alignment + }; - template - EIGEN_STRONG_INLINE - PacketType packet(Index row, Index col) const - { - return m_argImpl.template packet(col, row); - } + EIGEN_DEVICE_FUNC explicit unary_evaluator(const XprType &t) : m_argImpl(t.nestedExpression()) {} - template - EIGEN_STRONG_INLINE - PacketType packet(Index index) const - { - return m_argImpl.template packet(index); - } + typedef typename XprType::Scalar Scalar; + typedef typename XprType::CoeffReturnType CoeffReturnType; - template - EIGEN_STRONG_INLINE - void writePacket(Index row, Index col, const PacketType& x) - { - m_argImpl.template writePacket(col, row, x); - } + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE CoeffReturnType coeff(Index row, Index col) const + { + return m_argImpl.coeff(col, row); + } - template - EIGEN_STRONG_INLINE - void writePacket(Index index, const PacketType& x) - { - m_argImpl.template writePacket(index, x); - } + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE CoeffReturnType coeff(Index index) const { return m_argImpl.coeff(index); } -protected: - evaluator m_argImpl; -}; + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Scalar &coeffRef(Index row, Index col) + { + return m_argImpl.coeffRef(col, row); + } -// -------------------- CwiseNullaryOp -------------------- -// Like Matrix and Array, this is not really a unary expression, so we directly specialize evaluator. -// Likewise, there is not need to more sophisticated dispatching here. + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE typename XprType::Scalar &coeffRef(Index index) + { + return m_argImpl.coeffRef(index); + } -template::value, - bool has_unary = has_unary_operator::value, - bool has_binary = has_binary_operator::value> -struct nullary_wrapper -{ - template - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Scalar operator()(const NullaryOp& op, IndexType i, IndexType j) const { return op(i,j); } - template - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Scalar operator()(const NullaryOp& op, IndexType i) const { return op(i); } + template EIGEN_STRONG_INLINE PacketType packet(Index row, Index col) const + { + return m_argImpl.template packet(col, row); + } - template EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE T packetOp(const NullaryOp& op, IndexType i, IndexType j) const { return op.template packetOp(i,j); } - template EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE T packetOp(const NullaryOp& op, IndexType i) const { return op.template packetOp(i); } -}; + template EIGEN_STRONG_INLINE PacketType packet(Index index) const + { + return m_argImpl.template packet(index); + } -template -struct nullary_wrapper -{ - template - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Scalar operator()(const NullaryOp& op, IndexType=0, IndexType=0) const { return op(); } - template EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE T packetOp(const NullaryOp& op, IndexType=0, IndexType=0) const { return op.template packetOp(); } -}; + template + EIGEN_STRONG_INLINE void writePacket(Index row, Index col, const PacketType &x) + { + m_argImpl.template writePacket(col, row, x); + } -template -struct nullary_wrapper -{ - template - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Scalar operator()(const NullaryOp& op, IndexType i, IndexType j=0) const { return op(i,j); } - template EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE T packetOp(const NullaryOp& op, IndexType i, IndexType j=0) const { return op.template packetOp(i,j); } -}; + template EIGEN_STRONG_INLINE void writePacket(Index index, const PacketType &x) + { + m_argImpl.template writePacket(index, x); + } -// We need the following specialization for vector-only functors assigned to a runtime vector, -// for instance, using linspace and assigning a RowVectorXd to a MatrixXd or even a row of a MatrixXd. -// In this case, i==0 and j is used for the actual iteration. -template -struct nullary_wrapper -{ - template - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Scalar operator()(const NullaryOp& op, IndexType i, IndexType j) const { - eigen_assert(i==0 || j==0); - return op(i+j); - } - template EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE T packetOp(const NullaryOp& op, IndexType i, IndexType j) const { - eigen_assert(i==0 || j==0); - return op.template packetOp(i+j); - } + protected: + evaluator m_argImpl; + }; - template - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Scalar operator()(const NullaryOp& op, IndexType i) const { return op(i); } - template - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE T packetOp(const NullaryOp& op, IndexType i) const { return op.template packetOp(i); } -}; + // -------------------- CwiseNullaryOp -------------------- + // Like Matrix and Array, this is not really a unary expression, so we directly specialize evaluator. + // Likewise, there is not need to more sophisticated dispatching here. + + template::value, + bool has_unary = has_unary_operator::value, + bool has_binary = has_binary_operator::value> + struct nullary_wrapper + { + template + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Scalar operator()(const NullaryOp &op, IndexType i, IndexType j) const + { + return op(i, j); + } + template + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Scalar operator()(const NullaryOp &op, IndexType i) const + { + return op(i); + } + + template + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE T packetOp(const NullaryOp &op, IndexType i, IndexType j) const + { + return op.template packetOp(i, j); + } + template + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE T packetOp(const NullaryOp &op, IndexType i) const + { + return op.template packetOp(i); + } + }; -template -struct nullary_wrapper {}; + template struct nullary_wrapper + { + template + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Scalar operator()(const NullaryOp &op, IndexType = 0, IndexType = 0) const + { + return op(); + } + template + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE T packetOp(const NullaryOp &op, IndexType = 0, IndexType = 0) const + { + return op.template packetOp(); + } + }; + + template struct nullary_wrapper + { + template + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Scalar operator()(const NullaryOp &op, IndexType i, IndexType j = 0) const + { + return op(i, j); + } + template + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE T packetOp(const NullaryOp &op, IndexType i, IndexType j = 0) const + { + return op.template packetOp(i, j); + } + }; + + // We need the following specialization for vector-only functors assigned to a runtime vector, + // for instance, using linspace and assigning a RowVectorXd to a MatrixXd or even a row of a MatrixXd. + // In this case, i==0 and j is used for the actual iteration. + template struct nullary_wrapper + { + template + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Scalar operator()(const NullaryOp &op, IndexType i, IndexType j) const + { + eigen_assert(i == 0 || j == 0); + return op(i + j); + } + template + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE T packetOp(const NullaryOp &op, IndexType i, IndexType j) const + { + eigen_assert(i == 0 || j == 0); + return op.template packetOp(i + j); + } + + template + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Scalar operator()(const NullaryOp &op, IndexType i) const + { + return op(i); + } + template + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE T packetOp(const NullaryOp &op, IndexType i) const + { + return op.template packetOp(i); + } + }; + + template struct nullary_wrapper + { + }; -#if 0 && EIGEN_COMP_MSVC>0 +#if 0 && EIGEN_COMP_MSVC > 0 // Disable this ugly workaround. This is now handled in traits::match, // but this piece of code might still become handly if some other weird compilation // erros pop up again. @@ -449,1240 +457,1097 @@ struct nullary_wrapper has_binary_operator >::value>().template packetOp(op,i); } }; -#endif // MSVC workaround - -template -struct evaluator > - : evaluator_base > -{ - typedef CwiseNullaryOp XprType; - typedef typename internal::remove_all::type PlainObjectTypeCleaned; - - enum { - CoeffReadCost = internal::functor_traits::Cost, - - Flags = (evaluator::Flags - & ( HereditaryBits - | (functor_has_linear_access::ret ? LinearAccessBit : 0) - | (functor_traits::PacketAccess ? PacketAccessBit : 0))) - | (functor_traits::IsRepeatable ? 0 : EvalBeforeNestingBit), - Alignment = AlignedMax - }; +#endif// MSVC workaround - EIGEN_DEVICE_FUNC explicit evaluator(const XprType& n) - : m_functor(n.functor()), m_wrapper() + template + struct evaluator> + : evaluator_base> { - EIGEN_INTERNAL_CHECK_COST_VALUE(CoeffReadCost); - } + typedef CwiseNullaryOp XprType; + typedef typename internal::remove_all::type PlainObjectTypeCleaned; - typedef typename XprType::CoeffReturnType CoeffReturnType; + enum { + CoeffReadCost = internal::functor_traits::Cost, - template - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE - CoeffReturnType coeff(IndexType row, IndexType col) const - { - return m_wrapper(m_functor, row, col); - } + Flags = (evaluator::Flags + & (HereditaryBits | (functor_has_linear_access::ret ? LinearAccessBit : 0) + | (functor_traits::PacketAccess ? PacketAccessBit : 0))) + | (functor_traits::IsRepeatable ? 0 : EvalBeforeNestingBit), + Alignment = AlignedMax + }; - template - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE - CoeffReturnType coeff(IndexType index) const - { - return m_wrapper(m_functor,index); - } + EIGEN_DEVICE_FUNC explicit evaluator(const XprType &n) : m_functor(n.functor()), m_wrapper() + { + EIGEN_INTERNAL_CHECK_COST_VALUE(CoeffReadCost); + } + + typedef typename XprType::CoeffReturnType CoeffReturnType; + + template + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE CoeffReturnType coeff(IndexType row, IndexType col) const + { + return m_wrapper(m_functor, row, col); + } + + template EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE CoeffReturnType coeff(IndexType index) const + { + return m_wrapper(m_functor, index); + } + + template + EIGEN_STRONG_INLINE PacketType packet(IndexType row, IndexType col) const + { + return m_wrapper.template packetOp(m_functor, row, col); + } + + template + EIGEN_STRONG_INLINE PacketType packet(IndexType index) const + { + return m_wrapper.template packetOp(m_functor, index); + } + + protected: + const NullaryOp m_functor; + const internal::nullary_wrapper m_wrapper; + }; - template - EIGEN_STRONG_INLINE - PacketType packet(IndexType row, IndexType col) const - { - return m_wrapper.template packetOp(m_functor, row, col); - } + // -------------------- CwiseUnaryOp -------------------- - template - EIGEN_STRONG_INLINE - PacketType packet(IndexType index) const + template + struct unary_evaluator, IndexBased> : evaluator_base> { - return m_wrapper.template packetOp(m_functor, index); - } + typedef CwiseUnaryOp XprType; -protected: - const NullaryOp m_functor; - const internal::nullary_wrapper m_wrapper; -}; + enum { + CoeffReadCost = evaluator::CoeffReadCost + functor_traits::Cost, -// -------------------- CwiseUnaryOp -------------------- + Flags = evaluator::Flags + & (HereditaryBits | LinearAccessBit | (functor_traits::PacketAccess ? PacketAccessBit : 0)), + Alignment = evaluator::Alignment + }; -template -struct unary_evaluator, IndexBased > - : evaluator_base > -{ - typedef CwiseUnaryOp XprType; - - enum { - CoeffReadCost = evaluator::CoeffReadCost + functor_traits::Cost, - - Flags = evaluator::Flags - & (HereditaryBits | LinearAccessBit | (functor_traits::PacketAccess ? PacketAccessBit : 0)), - Alignment = evaluator::Alignment + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE explicit unary_evaluator(const XprType &op) + : m_functor(op.functor()), m_argImpl(op.nestedExpression()) + { + EIGEN_INTERNAL_CHECK_COST_VALUE(functor_traits::Cost); + EIGEN_INTERNAL_CHECK_COST_VALUE(CoeffReadCost); + } + + typedef typename XprType::CoeffReturnType CoeffReturnType; + + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE CoeffReturnType coeff(Index row, Index col) const + { + return m_functor(m_argImpl.coeff(row, col)); + } + + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE CoeffReturnType coeff(Index index) const + { + return m_functor(m_argImpl.coeff(index)); + } + + template EIGEN_STRONG_INLINE PacketType packet(Index row, Index col) const + { + return m_functor.packetOp(m_argImpl.template packet(row, col)); + } + + template EIGEN_STRONG_INLINE PacketType packet(Index index) const + { + return m_functor.packetOp(m_argImpl.template packet(index)); + } + + protected: + const UnaryOp m_functor; + evaluator m_argImpl; }; - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE - explicit unary_evaluator(const XprType& op) - : m_functor(op.functor()), - m_argImpl(op.nestedExpression()) - { - EIGEN_INTERNAL_CHECK_COST_VALUE(functor_traits::Cost); - EIGEN_INTERNAL_CHECK_COST_VALUE(CoeffReadCost); - } - - typedef typename XprType::CoeffReturnType CoeffReturnType; + // -------------------- CwiseTernaryOp -------------------- - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE - CoeffReturnType coeff(Index row, Index col) const + // this is a ternary expression + template + struct evaluator> + : public ternary_evaluator> { - return m_functor(m_argImpl.coeff(row, col)); - } + typedef CwiseTernaryOp XprType; + typedef ternary_evaluator> Base; - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE - CoeffReturnType coeff(Index index) const - { - return m_functor(m_argImpl.coeff(index)); - } - - template - EIGEN_STRONG_INLINE - PacketType packet(Index row, Index col) const - { - return m_functor.packetOp(m_argImpl.template packet(row, col)); - } + EIGEN_DEVICE_FUNC explicit evaluator(const XprType &xpr) : Base(xpr) {} + }; - template - EIGEN_STRONG_INLINE - PacketType packet(Index index) const + template + struct ternary_evaluator, IndexBased, IndexBased> + : evaluator_base> { - return m_functor.packetOp(m_argImpl.template packet(index)); - } + typedef CwiseTernaryOp XprType; -protected: - const UnaryOp m_functor; - evaluator m_argImpl; -}; - -// -------------------- CwiseTernaryOp -------------------- - -// this is a ternary expression -template -struct evaluator > - : public ternary_evaluator > -{ - typedef CwiseTernaryOp XprType; - typedef ternary_evaluator > Base; - - EIGEN_DEVICE_FUNC explicit evaluator(const XprType& xpr) : Base(xpr) {} -}; - -template -struct ternary_evaluator, IndexBased, IndexBased> - : evaluator_base > -{ - typedef CwiseTernaryOp XprType; - - enum { - CoeffReadCost = evaluator::CoeffReadCost + evaluator::CoeffReadCost + evaluator::CoeffReadCost + functor_traits::Cost, - - Arg1Flags = evaluator::Flags, - Arg2Flags = evaluator::Flags, - Arg3Flags = evaluator::Flags, - SameType = is_same::value && is_same::value, - StorageOrdersAgree = (int(Arg1Flags)&RowMajorBit)==(int(Arg2Flags)&RowMajorBit) && (int(Arg1Flags)&RowMajorBit)==(int(Arg3Flags)&RowMajorBit), - Flags0 = (int(Arg1Flags) | int(Arg2Flags) | int(Arg3Flags)) & ( - HereditaryBits - | (int(Arg1Flags) & int(Arg2Flags) & int(Arg3Flags) & - ( (StorageOrdersAgree ? LinearAccessBit : 0) - | (functor_traits::PacketAccess && StorageOrdersAgree && SameType ? PacketAccessBit : 0) - ) - ) - ), - Flags = (Flags0 & ~RowMajorBit) | (Arg1Flags & RowMajorBit), - Alignment = EIGEN_PLAIN_ENUM_MIN( - EIGEN_PLAIN_ENUM_MIN(evaluator::Alignment, evaluator::Alignment), + enum { + CoeffReadCost = evaluator::CoeffReadCost + evaluator::CoeffReadCost + evaluator::CoeffReadCost + + functor_traits::Cost, + + Arg1Flags = evaluator::Flags, + Arg2Flags = evaluator::Flags, + Arg3Flags = evaluator::Flags, + SameType = is_same::value + && is_same::value, + StorageOrdersAgree = (int(Arg1Flags) & RowMajorBit) == (int(Arg2Flags) & RowMajorBit) + && (int(Arg1Flags) & RowMajorBit) == (int(Arg3Flags) & RowMajorBit), + Flags0 = + (int(Arg1Flags) | int(Arg2Flags) | int(Arg3Flags)) + & (HereditaryBits + | (int(Arg1Flags) & int(Arg2Flags) & int(Arg3Flags) + & ((StorageOrdersAgree ? LinearAccessBit : 0) + | (functor_traits::PacketAccess && StorageOrdersAgree && SameType ? PacketAccessBit : 0)))), + Flags = (Flags0 & ~RowMajorBit) | (Arg1Flags & RowMajorBit), + Alignment = EIGEN_PLAIN_ENUM_MIN(EIGEN_PLAIN_ENUM_MIN(evaluator::Alignment, evaluator::Alignment), evaluator::Alignment) - }; + }; - EIGEN_DEVICE_FUNC explicit ternary_evaluator(const XprType& xpr) - : m_functor(xpr.functor()), - m_arg1Impl(xpr.arg1()), - m_arg2Impl(xpr.arg2()), - m_arg3Impl(xpr.arg3()) - { - EIGEN_INTERNAL_CHECK_COST_VALUE(functor_traits::Cost); - EIGEN_INTERNAL_CHECK_COST_VALUE(CoeffReadCost); - } + EIGEN_DEVICE_FUNC explicit ternary_evaluator(const XprType &xpr) + : m_functor(xpr.functor()), m_arg1Impl(xpr.arg1()), m_arg2Impl(xpr.arg2()), m_arg3Impl(xpr.arg3()) + { + EIGEN_INTERNAL_CHECK_COST_VALUE(functor_traits::Cost); + EIGEN_INTERNAL_CHECK_COST_VALUE(CoeffReadCost); + } + + typedef typename XprType::CoeffReturnType CoeffReturnType; + + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE CoeffReturnType coeff(Index row, Index col) const + { + return m_functor(m_arg1Impl.coeff(row, col), m_arg2Impl.coeff(row, col), m_arg3Impl.coeff(row, col)); + } + + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE CoeffReturnType coeff(Index index) const + { + return m_functor(m_arg1Impl.coeff(index), m_arg2Impl.coeff(index), m_arg3Impl.coeff(index)); + } + + template EIGEN_STRONG_INLINE PacketType packet(Index row, Index col) const + { + return m_functor.packetOp(m_arg1Impl.template packet(row, col), + m_arg2Impl.template packet(row, col), + m_arg3Impl.template packet(row, col)); + } + + template EIGEN_STRONG_INLINE PacketType packet(Index index) const + { + return m_functor.packetOp(m_arg1Impl.template packet(index), + m_arg2Impl.template packet(index), + m_arg3Impl.template packet(index)); + } + + protected: + const TernaryOp m_functor; + evaluator m_arg1Impl; + evaluator m_arg2Impl; + evaluator m_arg3Impl; + }; - typedef typename XprType::CoeffReturnType CoeffReturnType; + // -------------------- CwiseBinaryOp -------------------- - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE - CoeffReturnType coeff(Index row, Index col) const + // this is a binary expression + template + struct evaluator> : public binary_evaluator> { - return m_functor(m_arg1Impl.coeff(row, col), m_arg2Impl.coeff(row, col), m_arg3Impl.coeff(row, col)); - } + typedef CwiseBinaryOp XprType; + typedef binary_evaluator> Base; - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE - CoeffReturnType coeff(Index index) const - { - return m_functor(m_arg1Impl.coeff(index), m_arg2Impl.coeff(index), m_arg3Impl.coeff(index)); - } + EIGEN_DEVICE_FUNC explicit evaluator(const XprType &xpr) : Base(xpr) {} + }; - template - EIGEN_STRONG_INLINE - PacketType packet(Index row, Index col) const + template + struct binary_evaluator, IndexBased, IndexBased> + : evaluator_base> { - return m_functor.packetOp(m_arg1Impl.template packet(row, col), - m_arg2Impl.template packet(row, col), - m_arg3Impl.template packet(row, col)); - } + typedef CwiseBinaryOp XprType; - template - EIGEN_STRONG_INLINE - PacketType packet(Index index) const - { - return m_functor.packetOp(m_arg1Impl.template packet(index), - m_arg2Impl.template packet(index), - m_arg3Impl.template packet(index)); - } + enum { + CoeffReadCost = evaluator::CoeffReadCost + evaluator::CoeffReadCost + functor_traits::Cost, + + LhsFlags = evaluator::Flags, + RhsFlags = evaluator::Flags, + SameType = is_same::value, + StorageOrdersAgree = (int(LhsFlags) & RowMajorBit) == (int(RhsFlags) & RowMajorBit), + Flags0 = + (int(LhsFlags) | int(RhsFlags)) + & (HereditaryBits + | (int(LhsFlags) & int(RhsFlags) + & ((StorageOrdersAgree ? LinearAccessBit : 0) + | (functor_traits::PacketAccess && StorageOrdersAgree && SameType ? PacketAccessBit : 0)))), + Flags = (Flags0 & ~RowMajorBit) | (LhsFlags & RowMajorBit), + Alignment = EIGEN_PLAIN_ENUM_MIN(evaluator::Alignment, evaluator::Alignment) + }; -protected: - const TernaryOp m_functor; - evaluator m_arg1Impl; - evaluator m_arg2Impl; - evaluator m_arg3Impl; -}; + EIGEN_DEVICE_FUNC explicit binary_evaluator(const XprType &xpr) + : m_functor(xpr.functor()), m_lhsImpl(xpr.lhs()), m_rhsImpl(xpr.rhs()) + { + EIGEN_INTERNAL_CHECK_COST_VALUE(functor_traits::Cost); + EIGEN_INTERNAL_CHECK_COST_VALUE(CoeffReadCost); + } + + typedef typename XprType::CoeffReturnType CoeffReturnType; + + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE CoeffReturnType coeff(Index row, Index col) const + { + return m_functor(m_lhsImpl.coeff(row, col), m_rhsImpl.coeff(row, col)); + } + + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE CoeffReturnType coeff(Index index) const + { + return m_functor(m_lhsImpl.coeff(index), m_rhsImpl.coeff(index)); + } + + template EIGEN_STRONG_INLINE PacketType packet(Index row, Index col) const + { + return m_functor.packetOp(m_lhsImpl.template packet(row, col), + m_rhsImpl.template packet(row, col)); + } + + template EIGEN_STRONG_INLINE PacketType packet(Index index) const + { + return m_functor.packetOp( + m_lhsImpl.template packet(index), m_rhsImpl.template packet(index)); + } + + protected: + const BinaryOp m_functor; + evaluator m_lhsImpl; + evaluator m_rhsImpl; + }; -// -------------------- CwiseBinaryOp -------------------- + // -------------------- CwiseUnaryView -------------------- -// this is a binary expression -template -struct evaluator > - : public binary_evaluator > -{ - typedef CwiseBinaryOp XprType; - typedef binary_evaluator > Base; - - EIGEN_DEVICE_FUNC explicit evaluator(const XprType& xpr) : Base(xpr) {} -}; + template + struct unary_evaluator, IndexBased> + : evaluator_base> + { + typedef CwiseUnaryView XprType; -template -struct binary_evaluator, IndexBased, IndexBased> - : evaluator_base > -{ - typedef CwiseBinaryOp XprType; - - enum { - CoeffReadCost = evaluator::CoeffReadCost + evaluator::CoeffReadCost + functor_traits::Cost, - - LhsFlags = evaluator::Flags, - RhsFlags = evaluator::Flags, - SameType = is_same::value, - StorageOrdersAgree = (int(LhsFlags)&RowMajorBit)==(int(RhsFlags)&RowMajorBit), - Flags0 = (int(LhsFlags) | int(RhsFlags)) & ( - HereditaryBits - | (int(LhsFlags) & int(RhsFlags) & - ( (StorageOrdersAgree ? LinearAccessBit : 0) - | (functor_traits::PacketAccess && StorageOrdersAgree && SameType ? PacketAccessBit : 0) - ) - ) - ), - Flags = (Flags0 & ~RowMajorBit) | (LhsFlags & RowMajorBit), - Alignment = EIGEN_PLAIN_ENUM_MIN(evaluator::Alignment,evaluator::Alignment) - }; + enum { + CoeffReadCost = evaluator::CoeffReadCost + functor_traits::Cost, - EIGEN_DEVICE_FUNC explicit binary_evaluator(const XprType& xpr) - : m_functor(xpr.functor()), - m_lhsImpl(xpr.lhs()), - m_rhsImpl(xpr.rhs()) - { - EIGEN_INTERNAL_CHECK_COST_VALUE(functor_traits::Cost); - EIGEN_INTERNAL_CHECK_COST_VALUE(CoeffReadCost); - } + Flags = (evaluator::Flags & (HereditaryBits | LinearAccessBit | DirectAccessBit)), - typedef typename XprType::CoeffReturnType CoeffReturnType; + Alignment = 0// FIXME it is not very clear why alignment is necessarily lost... + }; - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE - CoeffReturnType coeff(Index row, Index col) const - { - return m_functor(m_lhsImpl.coeff(row, col), m_rhsImpl.coeff(row, col)); - } + EIGEN_DEVICE_FUNC explicit unary_evaluator(const XprType &op) + : m_unaryOp(op.functor()), m_argImpl(op.nestedExpression()) + { + EIGEN_INTERNAL_CHECK_COST_VALUE(functor_traits::Cost); + EIGEN_INTERNAL_CHECK_COST_VALUE(CoeffReadCost); + } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE - CoeffReturnType coeff(Index index) const - { - return m_functor(m_lhsImpl.coeff(index), m_rhsImpl.coeff(index)); - } + typedef typename XprType::Scalar Scalar; + typedef typename XprType::CoeffReturnType CoeffReturnType; - template - EIGEN_STRONG_INLINE - PacketType packet(Index row, Index col) const - { - return m_functor.packetOp(m_lhsImpl.template packet(row, col), - m_rhsImpl.template packet(row, col)); - } + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE CoeffReturnType coeff(Index row, Index col) const + { + return m_unaryOp(m_argImpl.coeff(row, col)); + } - template - EIGEN_STRONG_INLINE - PacketType packet(Index index) const - { - return m_functor.packetOp(m_lhsImpl.template packet(index), - m_rhsImpl.template packet(index)); - } + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE CoeffReturnType coeff(Index index) const + { + return m_unaryOp(m_argImpl.coeff(index)); + } -protected: - const BinaryOp m_functor; - evaluator m_lhsImpl; - evaluator m_rhsImpl; -}; + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Scalar &coeffRef(Index row, Index col) + { + return m_unaryOp(m_argImpl.coeffRef(row, col)); + } -// -------------------- CwiseUnaryView -------------------- + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Scalar &coeffRef(Index index) { return m_unaryOp(m_argImpl.coeffRef(index)); } -template -struct unary_evaluator, IndexBased> - : evaluator_base > -{ - typedef CwiseUnaryView XprType; - - enum { - CoeffReadCost = evaluator::CoeffReadCost + functor_traits::Cost, - - Flags = (evaluator::Flags & (HereditaryBits | LinearAccessBit | DirectAccessBit)), - - Alignment = 0 // FIXME it is not very clear why alignment is necessarily lost... + protected: + const UnaryOp m_unaryOp; + evaluator m_argImpl; }; - EIGEN_DEVICE_FUNC explicit unary_evaluator(const XprType& op) - : m_unaryOp(op.functor()), - m_argImpl(op.nestedExpression()) - { - EIGEN_INTERNAL_CHECK_COST_VALUE(functor_traits::Cost); - EIGEN_INTERNAL_CHECK_COST_VALUE(CoeffReadCost); - } + // -------------------- Map -------------------- - typedef typename XprType::Scalar Scalar; - typedef typename XprType::CoeffReturnType CoeffReturnType; + // FIXME perhaps the PlainObjectType could be provided by Derived::PlainObject ? + // but that might complicate template specialization + template struct mapbase_evaluator; - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE - CoeffReturnType coeff(Index row, Index col) const + template struct mapbase_evaluator : evaluator_base { - return m_unaryOp(m_argImpl.coeff(row, col)); - } + typedef Derived XprType; + typedef typename XprType::PointerType PointerType; + typedef typename XprType::Scalar Scalar; + typedef typename XprType::CoeffReturnType CoeffReturnType; - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE - CoeffReturnType coeff(Index index) const - { - return m_unaryOp(m_argImpl.coeff(index)); - } + enum { + IsRowMajor = XprType::RowsAtCompileTime, + ColsAtCompileTime = XprType::ColsAtCompileTime, + CoeffReadCost = NumTraits::ReadCost + }; - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE - Scalar& coeffRef(Index row, Index col) - { - return m_unaryOp(m_argImpl.coeffRef(row, col)); - } + EIGEN_DEVICE_FUNC explicit mapbase_evaluator(const XprType &map) + : m_data(const_cast(map.data())), m_innerStride(map.innerStride()), m_outerStride(map.outerStride()) + { + EIGEN_STATIC_ASSERT(EIGEN_IMPLIES(evaluator::Flags & PacketAccessBit, + internal::inner_stride_at_compile_time::ret == 1), + PACKET_ACCESS_REQUIRES_TO_HAVE_INNER_STRIDE_FIXED_TO_1); + EIGEN_INTERNAL_CHECK_COST_VALUE(CoeffReadCost); + } + + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE CoeffReturnType coeff(Index row, Index col) const + { + return m_data[col * colStride() + row * rowStride()]; + } + + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE CoeffReturnType coeff(Index index) const + { + return m_data[index * m_innerStride.value()]; + } + + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Scalar &coeffRef(Index row, Index col) + { + return m_data[col * colStride() + row * rowStride()]; + } + + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Scalar &coeffRef(Index index) + { + return m_data[index * m_innerStride.value()]; + } + + template EIGEN_STRONG_INLINE PacketType packet(Index row, Index col) const + { + PointerType ptr = m_data + row * rowStride() + col * colStride(); + return internal::ploadt(ptr); + } + + template EIGEN_STRONG_INLINE PacketType packet(Index index) const + { + return internal::ploadt(m_data + index * m_innerStride.value()); + } + + template + EIGEN_STRONG_INLINE void writePacket(Index row, Index col, const PacketType &x) + { + PointerType ptr = m_data + row * rowStride() + col * colStride(); + return internal::pstoret(ptr, x); + } + + template EIGEN_STRONG_INLINE void writePacket(Index index, const PacketType &x) + { + internal::pstoret(m_data + index * m_innerStride.value(), x); + } + + protected: + EIGEN_DEVICE_FUNC + inline Index rowStride() const { return XprType::IsRowMajor ? m_outerStride.value() : m_innerStride.value(); } + EIGEN_DEVICE_FUNC + inline Index colStride() const { return XprType::IsRowMajor ? m_innerStride.value() : m_outerStride.value(); } + + PointerType m_data; + const internal::variable_if_dynamic m_innerStride; + const internal::variable_if_dynamic m_outerStride; + }; - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE - Scalar& coeffRef(Index index) + template + struct evaluator> + : public mapbase_evaluator, PlainObjectType> { - return m_unaryOp(m_argImpl.coeffRef(index)); - } - -protected: - const UnaryOp m_unaryOp; - evaluator m_argImpl; -}; + typedef Map XprType; + typedef typename XprType::Scalar Scalar; + // TODO: should check for smaller packet types once we can handle multi-sized packet types + typedef typename packet_traits::type PacketScalar; -// -------------------- Map -------------------- - -// FIXME perhaps the PlainObjectType could be provided by Derived::PlainObject ? -// but that might complicate template specialization -template -struct mapbase_evaluator; + enum { + InnerStrideAtCompileTime = StrideType::InnerStrideAtCompileTime == 0 + ? int(PlainObjectType::InnerStrideAtCompileTime) + : int(StrideType::InnerStrideAtCompileTime), + OuterStrideAtCompileTime = StrideType::OuterStrideAtCompileTime == 0 + ? int(PlainObjectType::OuterStrideAtCompileTime) + : int(StrideType::OuterStrideAtCompileTime), + HasNoInnerStride = InnerStrideAtCompileTime == 1, + HasNoOuterStride = StrideType::OuterStrideAtCompileTime == 0, + HasNoStride = HasNoInnerStride && HasNoOuterStride, + IsDynamicSize = PlainObjectType::SizeAtCompileTime == Dynamic, + + PacketAccessMask = bool(HasNoInnerStride) ? ~int(0) : ~int(PacketAccessBit), + LinearAccessMask = + bool(HasNoStride) || bool(PlainObjectType::IsVectorAtCompileTime) ? ~int(0) : ~int(LinearAccessBit), + Flags = int(evaluator::Flags) & (LinearAccessMask & PacketAccessMask), + + Alignment = int(MapOptions) & int(AlignedMask) + }; -template -struct mapbase_evaluator : evaluator_base -{ - typedef Derived XprType; - typedef typename XprType::PointerType PointerType; - typedef typename XprType::Scalar Scalar; - typedef typename XprType::CoeffReturnType CoeffReturnType; - - enum { - IsRowMajor = XprType::RowsAtCompileTime, - ColsAtCompileTime = XprType::ColsAtCompileTime, - CoeffReadCost = NumTraits::ReadCost + EIGEN_DEVICE_FUNC explicit evaluator(const XprType &map) : mapbase_evaluator(map) {} }; - EIGEN_DEVICE_FUNC explicit mapbase_evaluator(const XprType& map) - : m_data(const_cast(map.data())), - m_innerStride(map.innerStride()), - m_outerStride(map.outerStride()) - { - EIGEN_STATIC_ASSERT(EIGEN_IMPLIES(evaluator::Flags&PacketAccessBit, internal::inner_stride_at_compile_time::ret==1), - PACKET_ACCESS_REQUIRES_TO_HAVE_INNER_STRIDE_FIXED_TO_1); - EIGEN_INTERNAL_CHECK_COST_VALUE(CoeffReadCost); - } + // -------------------- Ref -------------------- - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE - CoeffReturnType coeff(Index row, Index col) const + template + struct evaluator> + : public mapbase_evaluator, PlainObjectType> { - return m_data[col * colStride() + row * rowStride()]; - } + typedef Ref XprType; - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE - CoeffReturnType coeff(Index index) const - { - return m_data[index * m_innerStride.value()]; - } + enum { + Flags = evaluator>::Flags, + Alignment = evaluator>::Alignment + }; - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE - Scalar& coeffRef(Index row, Index col) - { - return m_data[col * colStride() + row * rowStride()]; - } + EIGEN_DEVICE_FUNC explicit evaluator(const XprType &ref) : mapbase_evaluator(ref) {} + }; - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE - Scalar& coeffRef(Index index) - { - return m_data[index * m_innerStride.value()]; - } + // -------------------- Block -------------------- - template - EIGEN_STRONG_INLINE - PacketType packet(Index row, Index col) const - { - PointerType ptr = m_data + row * rowStride() + col * colStride(); - return internal::ploadt(ptr); - } + template::ret> + struct block_evaluator; - template - EIGEN_STRONG_INLINE - PacketType packet(Index index) const + template + struct evaluator> + : block_evaluator { - return internal::ploadt(m_data + index * m_innerStride.value()); - } + typedef Block XprType; + typedef typename XprType::Scalar Scalar; + // TODO: should check for smaller packet types once we can handle multi-sized packet types + typedef typename packet_traits::type PacketScalar; - template - EIGEN_STRONG_INLINE - void writePacket(Index row, Index col, const PacketType& x) - { - PointerType ptr = m_data + row * rowStride() + col * colStride(); - return internal::pstoret(ptr, x); - } + enum { + CoeffReadCost = evaluator::CoeffReadCost, + + RowsAtCompileTime = traits::RowsAtCompileTime, + ColsAtCompileTime = traits::ColsAtCompileTime, + MaxRowsAtCompileTime = traits::MaxRowsAtCompileTime, + MaxColsAtCompileTime = traits::MaxColsAtCompileTime, + + ArgTypeIsRowMajor = (int(evaluator::Flags) & RowMajorBit) != 0, + IsRowMajor = (MaxRowsAtCompileTime == 1 && MaxColsAtCompileTime != 1) ? 1 + : (MaxColsAtCompileTime == 1 && MaxRowsAtCompileTime != 1) ? 0 + : ArgTypeIsRowMajor, + HasSameStorageOrderAsArgType = (IsRowMajor == ArgTypeIsRowMajor), + InnerSize = IsRowMajor ? int(ColsAtCompileTime) : int(RowsAtCompileTime), + InnerStrideAtCompileTime = HasSameStorageOrderAsArgType ? int(inner_stride_at_compile_time::ret) + : int(outer_stride_at_compile_time::ret), + OuterStrideAtCompileTime = HasSameStorageOrderAsArgType ? int(outer_stride_at_compile_time::ret) + : int(inner_stride_at_compile_time::ret), + MaskPacketAccessBit = (InnerStrideAtCompileTime == 1 || HasSameStorageOrderAsArgType) ? PacketAccessBit : 0, + + FlagsLinearAccessBit = (RowsAtCompileTime == 1 || ColsAtCompileTime == 1 + || (InnerPanel && (evaluator::Flags & LinearAccessBit))) + ? LinearAccessBit + : 0, + FlagsRowMajorBit = XprType::Flags & RowMajorBit, + Flags0 = evaluator::Flags & ((HereditaryBits & ~RowMajorBit) | DirectAccessBit | MaskPacketAccessBit), + Flags = Flags0 | FlagsLinearAccessBit | FlagsRowMajorBit, + + PacketAlignment = unpacket_traits::alignment, + Alignment0 = (InnerPanel && (OuterStrideAtCompileTime != Dynamic) && (OuterStrideAtCompileTime != 0) + && (((OuterStrideAtCompileTime * int(sizeof(Scalar))) % int(PacketAlignment)) == 0)) + ? int(PacketAlignment) + : 0, + Alignment = EIGEN_PLAIN_ENUM_MIN(evaluator::Alignment, Alignment0) + }; + typedef block_evaluator block_evaluator_type; + EIGEN_DEVICE_FUNC explicit evaluator(const XprType &block) : block_evaluator_type(block) + { + EIGEN_INTERNAL_CHECK_COST_VALUE(CoeffReadCost); + } + }; - template - EIGEN_STRONG_INLINE - void writePacket(Index index, const PacketType& x) + // no direct-access => dispatch to a unary evaluator + template + struct block_evaluator + : unary_evaluator> { - internal::pstoret(m_data + index * m_innerStride.value(), x); - } -protected: - EIGEN_DEVICE_FUNC - inline Index rowStride() const { return XprType::IsRowMajor ? m_outerStride.value() : m_innerStride.value(); } - EIGEN_DEVICE_FUNC - inline Index colStride() const { return XprType::IsRowMajor ? m_innerStride.value() : m_outerStride.value(); } - - PointerType m_data; - const internal::variable_if_dynamic m_innerStride; - const internal::variable_if_dynamic m_outerStride; -}; + typedef Block XprType; -template -struct evaluator > - : public mapbase_evaluator, PlainObjectType> -{ - typedef Map XprType; - typedef typename XprType::Scalar Scalar; - // TODO: should check for smaller packet types once we can handle multi-sized packet types - typedef typename packet_traits::type PacketScalar; - - enum { - InnerStrideAtCompileTime = StrideType::InnerStrideAtCompileTime == 0 - ? int(PlainObjectType::InnerStrideAtCompileTime) - : int(StrideType::InnerStrideAtCompileTime), - OuterStrideAtCompileTime = StrideType::OuterStrideAtCompileTime == 0 - ? int(PlainObjectType::OuterStrideAtCompileTime) - : int(StrideType::OuterStrideAtCompileTime), - HasNoInnerStride = InnerStrideAtCompileTime == 1, - HasNoOuterStride = StrideType::OuterStrideAtCompileTime == 0, - HasNoStride = HasNoInnerStride && HasNoOuterStride, - IsDynamicSize = PlainObjectType::SizeAtCompileTime==Dynamic, - - PacketAccessMask = bool(HasNoInnerStride) ? ~int(0) : ~int(PacketAccessBit), - LinearAccessMask = bool(HasNoStride) || bool(PlainObjectType::IsVectorAtCompileTime) ? ~int(0) : ~int(LinearAccessBit), - Flags = int( evaluator::Flags) & (LinearAccessMask&PacketAccessMask), - - Alignment = int(MapOptions)&int(AlignedMask) + EIGEN_DEVICE_FUNC explicit block_evaluator(const XprType &block) : unary_evaluator(block) {} }; - EIGEN_DEVICE_FUNC explicit evaluator(const XprType& map) - : mapbase_evaluator(map) - { } -}; - -// -------------------- Ref -------------------- + template + struct unary_evaluator, IndexBased> + : evaluator_base> + { + typedef Block XprType; -template -struct evaluator > - : public mapbase_evaluator, PlainObjectType> -{ - typedef Ref XprType; - - enum { - Flags = evaluator >::Flags, - Alignment = evaluator >::Alignment - }; + EIGEN_DEVICE_FUNC explicit unary_evaluator(const XprType &block) + : m_argImpl(block.nestedExpression()), m_startRow(block.startRow()), m_startCol(block.startCol()), + m_linear_offset( + InnerPanel ? (XprType::IsRowMajor ? block.startRow() * block.cols() : block.startCol() * block.rows()) : 0) + {} - EIGEN_DEVICE_FUNC explicit evaluator(const XprType& ref) - : mapbase_evaluator(ref) - { } -}; + typedef typename XprType::Scalar Scalar; + typedef typename XprType::CoeffReturnType CoeffReturnType; -// -------------------- Block -------------------- + enum { + RowsAtCompileTime = XprType::RowsAtCompileTime, + ForwardLinearAccess = InnerPanel && bool(evaluator::Flags & LinearAccessBit) + }; -template::ret> struct block_evaluator; - -template -struct evaluator > - : block_evaluator -{ - typedef Block XprType; - typedef typename XprType::Scalar Scalar; - // TODO: should check for smaller packet types once we can handle multi-sized packet types - typedef typename packet_traits::type PacketScalar; - - enum { - CoeffReadCost = evaluator::CoeffReadCost, - - RowsAtCompileTime = traits::RowsAtCompileTime, - ColsAtCompileTime = traits::ColsAtCompileTime, - MaxRowsAtCompileTime = traits::MaxRowsAtCompileTime, - MaxColsAtCompileTime = traits::MaxColsAtCompileTime, - - ArgTypeIsRowMajor = (int(evaluator::Flags)&RowMajorBit) != 0, - IsRowMajor = (MaxRowsAtCompileTime==1 && MaxColsAtCompileTime!=1) ? 1 - : (MaxColsAtCompileTime==1 && MaxRowsAtCompileTime!=1) ? 0 - : ArgTypeIsRowMajor, - HasSameStorageOrderAsArgType = (IsRowMajor == ArgTypeIsRowMajor), - InnerSize = IsRowMajor ? int(ColsAtCompileTime) : int(RowsAtCompileTime), - InnerStrideAtCompileTime = HasSameStorageOrderAsArgType - ? int(inner_stride_at_compile_time::ret) - : int(outer_stride_at_compile_time::ret), - OuterStrideAtCompileTime = HasSameStorageOrderAsArgType - ? int(outer_stride_at_compile_time::ret) - : int(inner_stride_at_compile_time::ret), - MaskPacketAccessBit = (InnerStrideAtCompileTime == 1 || HasSameStorageOrderAsArgType) ? PacketAccessBit : 0, - - FlagsLinearAccessBit = (RowsAtCompileTime == 1 || ColsAtCompileTime == 1 || (InnerPanel && (evaluator::Flags&LinearAccessBit))) ? LinearAccessBit : 0, - FlagsRowMajorBit = XprType::Flags&RowMajorBit, - Flags0 = evaluator::Flags & ( (HereditaryBits & ~RowMajorBit) | - DirectAccessBit | - MaskPacketAccessBit), - Flags = Flags0 | FlagsLinearAccessBit | FlagsRowMajorBit, - - PacketAlignment = unpacket_traits::alignment, - Alignment0 = (InnerPanel && (OuterStrideAtCompileTime!=Dynamic) - && (OuterStrideAtCompileTime!=0) - && (((OuterStrideAtCompileTime * int(sizeof(Scalar))) % int(PacketAlignment)) == 0)) ? int(PacketAlignment) : 0, - Alignment = EIGEN_PLAIN_ENUM_MIN(evaluator::Alignment, Alignment0) + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE CoeffReturnType coeff(Index row, Index col) const + { + return m_argImpl.coeff(m_startRow.value() + row, m_startCol.value() + col); + } + + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE CoeffReturnType coeff(Index index) const + { + if (ForwardLinearAccess) + return m_argImpl.coeff(m_linear_offset.value() + index); + else + return coeff(RowsAtCompileTime == 1 ? 0 : index, RowsAtCompileTime == 1 ? index : 0); + } + + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Scalar &coeffRef(Index row, Index col) + { + return m_argImpl.coeffRef(m_startRow.value() + row, m_startCol.value() + col); + } + + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Scalar &coeffRef(Index index) + { + if (ForwardLinearAccess) + return m_argImpl.coeffRef(m_linear_offset.value() + index); + else + return coeffRef(RowsAtCompileTime == 1 ? 0 : index, RowsAtCompileTime == 1 ? index : 0); + } + + template EIGEN_STRONG_INLINE PacketType packet(Index row, Index col) const + { + return m_argImpl.template packet(m_startRow.value() + row, m_startCol.value() + col); + } + + template EIGEN_STRONG_INLINE PacketType packet(Index index) const + { + if (ForwardLinearAccess) + return m_argImpl.template packet(m_linear_offset.value() + index); + else + return packet(RowsAtCompileTime == 1 ? 0 : index, RowsAtCompileTime == 1 ? index : 0); + } + + template + EIGEN_STRONG_INLINE void writePacket(Index row, Index col, const PacketType &x) + { + return m_argImpl.template writePacket( + m_startRow.value() + row, m_startCol.value() + col, x); + } + + template EIGEN_STRONG_INLINE void writePacket(Index index, const PacketType &x) + { + if (ForwardLinearAccess) + return m_argImpl.template writePacket(m_linear_offset.value() + index, x); + else + return writePacket( + RowsAtCompileTime == 1 ? 0 : index, RowsAtCompileTime == 1 ? index : 0, x); + } + + protected: + evaluator m_argImpl; + const variable_if_dynamic m_startRow; + const variable_if_dynamic m_startCol; + const variable_if_dynamic m_linear_offset; }; - typedef block_evaluator block_evaluator_type; - EIGEN_DEVICE_FUNC explicit evaluator(const XprType& block) : block_evaluator_type(block) - { - EIGEN_INTERNAL_CHECK_COST_VALUE(CoeffReadCost); - } -}; - -// no direct-access => dispatch to a unary evaluator -template -struct block_evaluator - : unary_evaluator > -{ - typedef Block XprType; - EIGEN_DEVICE_FUNC explicit block_evaluator(const XprType& block) - : unary_evaluator(block) - {} -}; - -template -struct unary_evaluator, IndexBased> - : evaluator_base > -{ - typedef Block XprType; - - EIGEN_DEVICE_FUNC explicit unary_evaluator(const XprType& block) - : m_argImpl(block.nestedExpression()), - m_startRow(block.startRow()), - m_startCol(block.startCol()), - m_linear_offset(InnerPanel?(XprType::IsRowMajor ? block.startRow()*block.cols() : block.startCol()*block.rows()):0) - { } - - typedef typename XprType::Scalar Scalar; - typedef typename XprType::CoeffReturnType CoeffReturnType; - - enum { - RowsAtCompileTime = XprType::RowsAtCompileTime, - ForwardLinearAccess = InnerPanel && bool(evaluator::Flags&LinearAccessBit) + // TODO: This evaluator does not actually use the child evaluator; + // all action is via the data() as returned by the Block expression. + + template + struct block_evaluator + : mapbase_evaluator, + typename Block::PlainObject> + { + typedef Block XprType; + typedef typename XprType::Scalar Scalar; + + EIGEN_DEVICE_FUNC explicit block_evaluator(const XprType &block) + : mapbase_evaluator(block) + { + // TODO: for the 3.3 release, this should be turned to an internal assertion, but let's keep it as is for the beta + // lifetime + eigen_assert(((internal::UIntPtr(block.data()) % EIGEN_PLAIN_ENUM_MAX(1, evaluator::Alignment)) == 0) + && "data is not aligned"); + } }; - - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE - CoeffReturnType coeff(Index row, Index col) const - { - return m_argImpl.coeff(m_startRow.value() + row, m_startCol.value() + col); - } - - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE - CoeffReturnType coeff(Index index) const - { - if (ForwardLinearAccess) - return m_argImpl.coeff(m_linear_offset.value() + index); - else - return coeff(RowsAtCompileTime == 1 ? 0 : index, RowsAtCompileTime == 1 ? index : 0); - } - - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE - Scalar& coeffRef(Index row, Index col) - { - return m_argImpl.coeffRef(m_startRow.value() + row, m_startCol.value() + col); - } - - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE - Scalar& coeffRef(Index index) - { - if (ForwardLinearAccess) - return m_argImpl.coeffRef(m_linear_offset.value() + index); - else - return coeffRef(RowsAtCompileTime == 1 ? 0 : index, RowsAtCompileTime == 1 ? index : 0); - } - - template - EIGEN_STRONG_INLINE - PacketType packet(Index row, Index col) const - { - return m_argImpl.template packet(m_startRow.value() + row, m_startCol.value() + col); - } - template - EIGEN_STRONG_INLINE - PacketType packet(Index index) const - { - if (ForwardLinearAccess) - return m_argImpl.template packet(m_linear_offset.value() + index); - else - return packet(RowsAtCompileTime == 1 ? 0 : index, - RowsAtCompileTime == 1 ? index : 0); - } - - template - EIGEN_STRONG_INLINE - void writePacket(Index row, Index col, const PacketType& x) - { - return m_argImpl.template writePacket(m_startRow.value() + row, m_startCol.value() + col, x); - } - - template - EIGEN_STRONG_INLINE - void writePacket(Index index, const PacketType& x) - { - if (ForwardLinearAccess) - return m_argImpl.template writePacket(m_linear_offset.value() + index, x); - else - return writePacket(RowsAtCompileTime == 1 ? 0 : index, - RowsAtCompileTime == 1 ? index : 0, - x); - } - -protected: - evaluator m_argImpl; - const variable_if_dynamic m_startRow; - const variable_if_dynamic m_startCol; - const variable_if_dynamic m_linear_offset; -}; -// TODO: This evaluator does not actually use the child evaluator; -// all action is via the data() as returned by the Block expression. + // -------------------- Select -------------------- + // NOTE shall we introduce a ternary_evaluator? -template -struct block_evaluator - : mapbase_evaluator, - typename Block::PlainObject> -{ - typedef Block XprType; - typedef typename XprType::Scalar Scalar; - - EIGEN_DEVICE_FUNC explicit block_evaluator(const XprType& block) - : mapbase_evaluator(block) + // TODO enable vectorization for Select + template + struct evaluator> + : evaluator_base> { - // TODO: for the 3.3 release, this should be turned to an internal assertion, but let's keep it as is for the beta lifetime - eigen_assert(((internal::UIntPtr(block.data()) % EIGEN_PLAIN_ENUM_MAX(1,evaluator::Alignment)) == 0) && "data is not aligned"); - } -}; + typedef Select XprType; + enum { + CoeffReadCost = + evaluator::CoeffReadCost + + EIGEN_PLAIN_ENUM_MAX(evaluator::CoeffReadCost, evaluator::CoeffReadCost), + Flags = (unsigned int)evaluator::Flags & evaluator::Flags & HereditaryBits, -// -------------------- Select -------------------- -// NOTE shall we introduce a ternary_evaluator? + Alignment = EIGEN_PLAIN_ENUM_MIN(evaluator::Alignment, evaluator::Alignment) + }; -// TODO enable vectorization for Select -template -struct evaluator > - : evaluator_base > -{ - typedef Select XprType; - enum { - CoeffReadCost = evaluator::CoeffReadCost - + EIGEN_PLAIN_ENUM_MAX(evaluator::CoeffReadCost, - evaluator::CoeffReadCost), - - Flags = (unsigned int)evaluator::Flags & evaluator::Flags & HereditaryBits, - - Alignment = EIGEN_PLAIN_ENUM_MIN(evaluator::Alignment, evaluator::Alignment) + EIGEN_DEVICE_FUNC explicit evaluator(const XprType &select) + : m_conditionImpl(select.conditionMatrix()), m_thenImpl(select.thenMatrix()), m_elseImpl(select.elseMatrix()) + { + EIGEN_INTERNAL_CHECK_COST_VALUE(CoeffReadCost); + } + + typedef typename XprType::CoeffReturnType CoeffReturnType; + + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE CoeffReturnType coeff(Index row, Index col) const + { + if (m_conditionImpl.coeff(row, col)) + return m_thenImpl.coeff(row, col); + else + return m_elseImpl.coeff(row, col); + } + + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE CoeffReturnType coeff(Index index) const + { + if (m_conditionImpl.coeff(index)) + return m_thenImpl.coeff(index); + else + return m_elseImpl.coeff(index); + } + + protected: + evaluator m_conditionImpl; + evaluator m_thenImpl; + evaluator m_elseImpl; }; - EIGEN_DEVICE_FUNC explicit evaluator(const XprType& select) - : m_conditionImpl(select.conditionMatrix()), - m_thenImpl(select.thenMatrix()), - m_elseImpl(select.elseMatrix()) - { - EIGEN_INTERNAL_CHECK_COST_VALUE(CoeffReadCost); - } - - typedef typename XprType::CoeffReturnType CoeffReturnType; - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE - CoeffReturnType coeff(Index row, Index col) const - { - if (m_conditionImpl.coeff(row, col)) - return m_thenImpl.coeff(row, col); - else - return m_elseImpl.coeff(row, col); - } + // -------------------- Replicate -------------------- - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE - CoeffReturnType coeff(Index index) const + template + struct unary_evaluator> + : evaluator_base> { - if (m_conditionImpl.coeff(index)) - return m_thenImpl.coeff(index); - else - return m_elseImpl.coeff(index); - } - -protected: - evaluator m_conditionImpl; - evaluator m_thenImpl; - evaluator m_elseImpl; -}; + typedef Replicate XprType; + typedef typename XprType::CoeffReturnType CoeffReturnType; + enum { Factor = (RowFactor == Dynamic || ColFactor == Dynamic) ? Dynamic : RowFactor * ColFactor }; + typedef typename internal::nested_eval::type ArgTypeNested; + typedef typename internal::remove_all::type ArgTypeNestedCleaned; + enum { + CoeffReadCost = evaluator::CoeffReadCost, + LinearAccessMask = XprType::IsVectorAtCompileTime ? LinearAccessBit : 0, + Flags = (evaluator::Flags & (HereditaryBits | LinearAccessMask) & ~RowMajorBit) + | (traits::Flags & RowMajorBit), -// -------------------- Replicate -------------------- + Alignment = evaluator::Alignment + }; -template -struct unary_evaluator > - : evaluator_base > -{ - typedef Replicate XprType; - typedef typename XprType::CoeffReturnType CoeffReturnType; - enum { - Factor = (RowFactor==Dynamic || ColFactor==Dynamic) ? Dynamic : RowFactor*ColFactor - }; - typedef typename internal::nested_eval::type ArgTypeNested; - typedef typename internal::remove_all::type ArgTypeNestedCleaned; - - enum { - CoeffReadCost = evaluator::CoeffReadCost, - LinearAccessMask = XprType::IsVectorAtCompileTime ? LinearAccessBit : 0, - Flags = (evaluator::Flags & (HereditaryBits|LinearAccessMask) & ~RowMajorBit) | (traits::Flags & RowMajorBit), - - Alignment = evaluator::Alignment + EIGEN_DEVICE_FUNC explicit unary_evaluator(const XprType &replicate) + : m_arg(replicate.nestedExpression()), m_argImpl(m_arg), m_rows(replicate.nestedExpression().rows()), + m_cols(replicate.nestedExpression().cols()) + {} + + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE CoeffReturnType coeff(Index row, Index col) const + { + // try to avoid using modulo; this is a pure optimization strategy + const Index actual_row = internal::traits::RowsAtCompileTime == 1 ? 0 + : RowFactor == 1 ? row + : row % m_rows.value(); + const Index actual_col = internal::traits::ColsAtCompileTime == 1 ? 0 + : ColFactor == 1 ? col + : col % m_cols.value(); + + return m_argImpl.coeff(actual_row, actual_col); + } + + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE CoeffReturnType coeff(Index index) const + { + // try to avoid using modulo; this is a pure optimization strategy + const Index actual_index = internal::traits::RowsAtCompileTime == 1 + ? (ColFactor == 1 ? index : index % m_cols.value()) + : (RowFactor == 1 ? index : index % m_rows.value()); + + return m_argImpl.coeff(actual_index); + } + + template EIGEN_STRONG_INLINE PacketType packet(Index row, Index col) const + { + const Index actual_row = internal::traits::RowsAtCompileTime == 1 ? 0 + : RowFactor == 1 ? row + : row % m_rows.value(); + const Index actual_col = internal::traits::ColsAtCompileTime == 1 ? 0 + : ColFactor == 1 ? col + : col % m_cols.value(); + + return m_argImpl.template packet(actual_row, actual_col); + } + + template EIGEN_STRONG_INLINE PacketType packet(Index index) const + { + const Index actual_index = internal::traits::RowsAtCompileTime == 1 + ? (ColFactor == 1 ? index : index % m_cols.value()) + : (RowFactor == 1 ? index : index % m_rows.value()); + + return m_argImpl.template packet(actual_index); + } + + protected: + const ArgTypeNested m_arg; + evaluator m_argImpl; + const variable_if_dynamic m_rows; + const variable_if_dynamic m_cols; }; - EIGEN_DEVICE_FUNC explicit unary_evaluator(const XprType& replicate) - : m_arg(replicate.nestedExpression()), - m_argImpl(m_arg), - m_rows(replicate.nestedExpression().rows()), - m_cols(replicate.nestedExpression().cols()) - {} - - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE - CoeffReturnType coeff(Index row, Index col) const - { - // try to avoid using modulo; this is a pure optimization strategy - const Index actual_row = internal::traits::RowsAtCompileTime==1 ? 0 - : RowFactor==1 ? row - : row % m_rows.value(); - const Index actual_col = internal::traits::ColsAtCompileTime==1 ? 0 - : ColFactor==1 ? col - : col % m_cols.value(); - - return m_argImpl.coeff(actual_row, actual_col); - } - - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE - CoeffReturnType coeff(Index index) const - { - // try to avoid using modulo; this is a pure optimization strategy - const Index actual_index = internal::traits::RowsAtCompileTime==1 - ? (ColFactor==1 ? index : index%m_cols.value()) - : (RowFactor==1 ? index : index%m_rows.value()); - - return m_argImpl.coeff(actual_index); - } - template - EIGEN_STRONG_INLINE - PacketType packet(Index row, Index col) const - { - const Index actual_row = internal::traits::RowsAtCompileTime==1 ? 0 - : RowFactor==1 ? row - : row % m_rows.value(); - const Index actual_col = internal::traits::ColsAtCompileTime==1 ? 0 - : ColFactor==1 ? col - : col % m_cols.value(); - - return m_argImpl.template packet(actual_row, actual_col); - } - - template - EIGEN_STRONG_INLINE - PacketType packet(Index index) const - { - const Index actual_index = internal::traits::RowsAtCompileTime==1 - ? (ColFactor==1 ? index : index%m_cols.value()) - : (RowFactor==1 ? index : index%m_rows.value()); + // -------------------- PartialReduxExpr -------------------- - return m_argImpl.template packet(actual_index); - } - -protected: - const ArgTypeNested m_arg; - evaluator m_argImpl; - const variable_if_dynamic m_rows; - const variable_if_dynamic m_cols; -}; + template + struct evaluator> + : evaluator_base> + { + typedef PartialReduxExpr XprType; + typedef typename internal::nested_eval::type ArgTypeNested; + typedef typename internal::remove_all::type ArgTypeNestedCleaned; + typedef typename ArgType::Scalar InputScalar; + typedef typename XprType::Scalar Scalar; + enum { + TraversalSize = Direction == int(Vertical) ? int(ArgType::RowsAtCompileTime) : int(ArgType::ColsAtCompileTime) + }; + typedef typename MemberOp::template Cost CostOpType; + enum { + CoeffReadCost = TraversalSize == Dynamic + ? HugeCost + : TraversalSize * evaluator::CoeffReadCost + int(CostOpType::value), + Flags = (traits::Flags & RowMajorBit) | (evaluator::Flags & (HereditaryBits & (~RowMajorBit))) + | LinearAccessBit, -// -------------------- PartialReduxExpr -------------------- + Alignment = 0// FIXME this will need to be improved once PartialReduxExpr is vectorized + }; -template< typename ArgType, typename MemberOp, int Direction> -struct evaluator > - : evaluator_base > -{ - typedef PartialReduxExpr XprType; - typedef typename internal::nested_eval::type ArgTypeNested; - typedef typename internal::remove_all::type ArgTypeNestedCleaned; - typedef typename ArgType::Scalar InputScalar; - typedef typename XprType::Scalar Scalar; - enum { - TraversalSize = Direction==int(Vertical) ? int(ArgType::RowsAtCompileTime) : int(ArgType::ColsAtCompileTime) + EIGEN_DEVICE_FUNC explicit evaluator(const XprType xpr) : m_arg(xpr.nestedExpression()), m_functor(xpr.functor()) + { + EIGEN_INTERNAL_CHECK_COST_VALUE(TraversalSize == Dynamic ? HugeCost : int(CostOpType::value)); + EIGEN_INTERNAL_CHECK_COST_VALUE(CoeffReadCost); + } + + typedef typename XprType::CoeffReturnType CoeffReturnType; + + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const Scalar coeff(Index i, Index j) const + { + if (Direction == Vertical) + return m_functor(m_arg.col(j)); + else + return m_functor(m_arg.row(i)); + } + + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const Scalar coeff(Index index) const + { + if (Direction == Vertical) + return m_functor(m_arg.col(index)); + else + return m_functor(m_arg.row(index)); + } + + protected: + typename internal::add_const_on_value_type::type m_arg; + const MemberOp m_functor; }; - typedef typename MemberOp::template Cost CostOpType; - enum { - CoeffReadCost = TraversalSize==Dynamic ? HugeCost - : TraversalSize * evaluator::CoeffReadCost + int(CostOpType::value), - - Flags = (traits::Flags&RowMajorBit) | (evaluator::Flags&(HereditaryBits&(~RowMajorBit))) | LinearAccessBit, - - Alignment = 0 // FIXME this will need to be improved once PartialReduxExpr is vectorized - }; - - EIGEN_DEVICE_FUNC explicit evaluator(const XprType xpr) - : m_arg(xpr.nestedExpression()), m_functor(xpr.functor()) - { - EIGEN_INTERNAL_CHECK_COST_VALUE(TraversalSize==Dynamic ? HugeCost : int(CostOpType::value)); - EIGEN_INTERNAL_CHECK_COST_VALUE(CoeffReadCost); - } - typedef typename XprType::CoeffReturnType CoeffReturnType; - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE - const Scalar coeff(Index i, Index j) const - { - if (Direction==Vertical) - return m_functor(m_arg.col(j)); - else - return m_functor(m_arg.row(i)); - } + // -------------------- MatrixWrapper and ArrayWrapper -------------------- + // + // evaluator_wrapper_base is a common base class for the + // MatrixWrapper and ArrayWrapper evaluators. - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE - const Scalar coeff(Index index) const + template struct evaluator_wrapper_base : evaluator_base { - if (Direction==Vertical) - return m_functor(m_arg.col(index)); - else - return m_functor(m_arg.row(index)); - } + typedef typename remove_all::type ArgType; + enum { + CoeffReadCost = evaluator::CoeffReadCost, + Flags = evaluator::Flags, + Alignment = evaluator::Alignment + }; -protected: - typename internal::add_const_on_value_type::type m_arg; - const MemberOp m_functor; -}; + EIGEN_DEVICE_FUNC explicit evaluator_wrapper_base(const ArgType &arg) : m_argImpl(arg) {} + typedef typename ArgType::Scalar Scalar; + typedef typename ArgType::CoeffReturnType CoeffReturnType; -// -------------------- MatrixWrapper and ArrayWrapper -------------------- -// -// evaluator_wrapper_base is a common base class for the -// MatrixWrapper and ArrayWrapper evaluators. + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE CoeffReturnType coeff(Index row, Index col) const + { + return m_argImpl.coeff(row, col); + } -template -struct evaluator_wrapper_base - : evaluator_base -{ - typedef typename remove_all::type ArgType; - enum { - CoeffReadCost = evaluator::CoeffReadCost, - Flags = evaluator::Flags, - Alignment = evaluator::Alignment - }; + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE CoeffReturnType coeff(Index index) const { return m_argImpl.coeff(index); } - EIGEN_DEVICE_FUNC explicit evaluator_wrapper_base(const ArgType& arg) : m_argImpl(arg) {} + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Scalar &coeffRef(Index row, Index col) + { + return m_argImpl.coeffRef(row, col); + } - typedef typename ArgType::Scalar Scalar; - typedef typename ArgType::CoeffReturnType CoeffReturnType; + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Scalar &coeffRef(Index index) { return m_argImpl.coeffRef(index); } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE - CoeffReturnType coeff(Index row, Index col) const - { - return m_argImpl.coeff(row, col); - } + template EIGEN_STRONG_INLINE PacketType packet(Index row, Index col) const + { + return m_argImpl.template packet(row, col); + } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE - CoeffReturnType coeff(Index index) const - { - return m_argImpl.coeff(index); - } + template EIGEN_STRONG_INLINE PacketType packet(Index index) const + { + return m_argImpl.template packet(index); + } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE - Scalar& coeffRef(Index row, Index col) - { - return m_argImpl.coeffRef(row, col); - } + template + EIGEN_STRONG_INLINE void writePacket(Index row, Index col, const PacketType &x) + { + m_argImpl.template writePacket(row, col, x); + } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE - Scalar& coeffRef(Index index) - { - return m_argImpl.coeffRef(index); - } + template EIGEN_STRONG_INLINE void writePacket(Index index, const PacketType &x) + { + m_argImpl.template writePacket(index, x); + } - template - EIGEN_STRONG_INLINE - PacketType packet(Index row, Index col) const - { - return m_argImpl.template packet(row, col); - } + protected: + evaluator m_argImpl; + }; - template - EIGEN_STRONG_INLINE - PacketType packet(Index index) const + template + struct unary_evaluator> : evaluator_wrapper_base> { - return m_argImpl.template packet(index); - } + typedef MatrixWrapper XprType; - template - EIGEN_STRONG_INLINE - void writePacket(Index row, Index col, const PacketType& x) - { - m_argImpl.template writePacket(row, col, x); - } + EIGEN_DEVICE_FUNC explicit unary_evaluator(const XprType &wrapper) + : evaluator_wrapper_base>(wrapper.nestedExpression()) + {} + }; - template - EIGEN_STRONG_INLINE - void writePacket(Index index, const PacketType& x) + template + struct unary_evaluator> : evaluator_wrapper_base> { - m_argImpl.template writePacket(index, x); - } - -protected: - evaluator m_argImpl; -}; - -template -struct unary_evaluator > - : evaluator_wrapper_base > -{ - typedef MatrixWrapper XprType; + typedef ArrayWrapper XprType; - EIGEN_DEVICE_FUNC explicit unary_evaluator(const XprType& wrapper) - : evaluator_wrapper_base >(wrapper.nestedExpression()) - { } -}; + EIGEN_DEVICE_FUNC explicit unary_evaluator(const XprType &wrapper) + : evaluator_wrapper_base>(wrapper.nestedExpression()) + {} + }; -template -struct unary_evaluator > - : evaluator_wrapper_base > -{ - typedef ArrayWrapper XprType; - EIGEN_DEVICE_FUNC explicit unary_evaluator(const XprType& wrapper) - : evaluator_wrapper_base >(wrapper.nestedExpression()) - { } -}; + // -------------------- Reverse -------------------- + // defined in Reverse.h: + template struct reverse_packet_cond; -// -------------------- Reverse -------------------- + template + struct unary_evaluator> : evaluator_base> + { + typedef Reverse XprType; + typedef typename XprType::Scalar Scalar; + typedef typename XprType::CoeffReturnType CoeffReturnType; -// defined in Reverse.h: -template struct reverse_packet_cond; + enum { + IsRowMajor = XprType::IsRowMajor, + IsColMajor = !IsRowMajor, + ReverseRow = (Direction == Vertical) || (Direction == BothDirections), + ReverseCol = (Direction == Horizontal) || (Direction == BothDirections), + ReversePacket = (Direction == BothDirections) || ((Direction == Vertical) && IsColMajor) + || ((Direction == Horizontal) && IsRowMajor), + + CoeffReadCost = evaluator::CoeffReadCost, + + // let's enable LinearAccess only with vectorization because of the product overhead + // FIXME enable DirectAccess with negative strides? + Flags0 = evaluator::Flags, + LinearAccess = + ((Direction == BothDirections) && (int(Flags0) & PacketAccessBit)) + || ((ReverseRow && XprType::ColsAtCompileTime == 1) || (ReverseCol && XprType::RowsAtCompileTime == 1)) + ? LinearAccessBit + : 0, + + Flags = int(Flags0) & (HereditaryBits | PacketAccessBit | LinearAccess), + + Alignment = 0// FIXME in some rare cases, Alignment could be preserved, like a Vector4f. + }; -template -struct unary_evaluator > - : evaluator_base > -{ - typedef Reverse XprType; - typedef typename XprType::Scalar Scalar; - typedef typename XprType::CoeffReturnType CoeffReturnType; - - enum { - IsRowMajor = XprType::IsRowMajor, - IsColMajor = !IsRowMajor, - ReverseRow = (Direction == Vertical) || (Direction == BothDirections), - ReverseCol = (Direction == Horizontal) || (Direction == BothDirections), - ReversePacket = (Direction == BothDirections) - || ((Direction == Vertical) && IsColMajor) - || ((Direction == Horizontal) && IsRowMajor), - - CoeffReadCost = evaluator::CoeffReadCost, - - // let's enable LinearAccess only with vectorization because of the product overhead - // FIXME enable DirectAccess with negative strides? - Flags0 = evaluator::Flags, - LinearAccess = ( (Direction==BothDirections) && (int(Flags0)&PacketAccessBit) ) - || ((ReverseRow && XprType::ColsAtCompileTime==1) || (ReverseCol && XprType::RowsAtCompileTime==1)) - ? LinearAccessBit : 0, - - Flags = int(Flags0) & (HereditaryBits | PacketAccessBit | LinearAccess), - - Alignment = 0 // FIXME in some rare cases, Alignment could be preserved, like a Vector4f. + EIGEN_DEVICE_FUNC explicit unary_evaluator(const XprType &reverse) + : m_argImpl(reverse.nestedExpression()), m_rows(ReverseRow ? reverse.nestedExpression().rows() : 1), + m_cols(ReverseCol ? reverse.nestedExpression().cols() : 1) + {} + + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE CoeffReturnType coeff(Index row, Index col) const + { + return m_argImpl.coeff(ReverseRow ? m_rows.value() - row - 1 : row, ReverseCol ? m_cols.value() - col - 1 : col); + } + + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE CoeffReturnType coeff(Index index) const + { + return m_argImpl.coeff(m_rows.value() * m_cols.value() - index - 1); + } + + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Scalar &coeffRef(Index row, Index col) + { + return m_argImpl.coeffRef( + ReverseRow ? m_rows.value() - row - 1 : row, ReverseCol ? m_cols.value() - col - 1 : col); + } + + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Scalar &coeffRef(Index index) + { + return m_argImpl.coeffRef(m_rows.value() * m_cols.value() - index - 1); + } + + template EIGEN_STRONG_INLINE PacketType packet(Index row, Index col) const + { + enum { + PacketSize = unpacket_traits::size, + OffsetRow = ReverseRow && IsColMajor ? PacketSize : 1, + OffsetCol = ReverseCol && IsRowMajor ? PacketSize : 1 + }; + typedef internal::reverse_packet_cond reverse_packet; + return reverse_packet::run(m_argImpl.template packet( + ReverseRow ? m_rows.value() - row - OffsetRow : row, ReverseCol ? m_cols.value() - col - OffsetCol : col)); + } + + template EIGEN_STRONG_INLINE PacketType packet(Index index) const + { + enum { PacketSize = unpacket_traits::size }; + return preverse( + m_argImpl.template packet(m_rows.value() * m_cols.value() - index - PacketSize)); + } + + template + EIGEN_STRONG_INLINE void writePacket(Index row, Index col, const PacketType &x) + { + // FIXME we could factorize some code with packet(i,j) + enum { + PacketSize = unpacket_traits::size, + OffsetRow = ReverseRow && IsColMajor ? PacketSize : 1, + OffsetCol = ReverseCol && IsRowMajor ? PacketSize : 1 + }; + typedef internal::reverse_packet_cond reverse_packet; + m_argImpl.template writePacket(ReverseRow ? m_rows.value() - row - OffsetRow : row, + ReverseCol ? m_cols.value() - col - OffsetCol : col, + reverse_packet::run(x)); + } + + template EIGEN_STRONG_INLINE void writePacket(Index index, const PacketType &x) + { + enum { PacketSize = unpacket_traits::size }; + m_argImpl.template writePacket(m_rows.value() * m_cols.value() - index - PacketSize, preverse(x)); + } + + protected: + evaluator m_argImpl; + + // If we do not reverse rows, then we do not need to know the number of rows; same for columns + // Nonetheless, in this case it is important to set to 1 such that the coeff(index) method works fine for vectors. + const variable_if_dynamic m_rows; + const variable_if_dynamic m_cols; }; - EIGEN_DEVICE_FUNC explicit unary_evaluator(const XprType& reverse) - : m_argImpl(reverse.nestedExpression()), - m_rows(ReverseRow ? reverse.nestedExpression().rows() : 1), - m_cols(ReverseCol ? reverse.nestedExpression().cols() : 1) - { } - - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE - CoeffReturnType coeff(Index row, Index col) const - { - return m_argImpl.coeff(ReverseRow ? m_rows.value() - row - 1 : row, - ReverseCol ? m_cols.value() - col - 1 : col); - } - - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE - CoeffReturnType coeff(Index index) const - { - return m_argImpl.coeff(m_rows.value() * m_cols.value() - index - 1); - } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE - Scalar& coeffRef(Index row, Index col) - { - return m_argImpl.coeffRef(ReverseRow ? m_rows.value() - row - 1 : row, - ReverseCol ? m_cols.value() - col - 1 : col); - } + // -------------------- Diagonal -------------------- - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE - Scalar& coeffRef(Index index) + template + struct evaluator> : evaluator_base> { - return m_argImpl.coeffRef(m_rows.value() * m_cols.value() - index - 1); - } + typedef Diagonal XprType; - template - EIGEN_STRONG_INLINE - PacketType packet(Index row, Index col) const - { enum { - PacketSize = unpacket_traits::size, - OffsetRow = ReverseRow && IsColMajor ? PacketSize : 1, - OffsetCol = ReverseCol && IsRowMajor ? PacketSize : 1 - }; - typedef internal::reverse_packet_cond reverse_packet; - return reverse_packet::run(m_argImpl.template packet( - ReverseRow ? m_rows.value() - row - OffsetRow : row, - ReverseCol ? m_cols.value() - col - OffsetCol : col)); - } + CoeffReadCost = evaluator::CoeffReadCost, - template - EIGEN_STRONG_INLINE - PacketType packet(Index index) const - { - enum { PacketSize = unpacket_traits::size }; - return preverse(m_argImpl.template packet(m_rows.value() * m_cols.value() - index - PacketSize)); - } + Flags = + (unsigned int)(evaluator::Flags & (HereditaryBits | DirectAccessBit) & ~RowMajorBit) | LinearAccessBit, - template - EIGEN_STRONG_INLINE - void writePacket(Index row, Index col, const PacketType& x) - { - // FIXME we could factorize some code with packet(i,j) - enum { - PacketSize = unpacket_traits::size, - OffsetRow = ReverseRow && IsColMajor ? PacketSize : 1, - OffsetCol = ReverseCol && IsRowMajor ? PacketSize : 1 + Alignment = 0 }; - typedef internal::reverse_packet_cond reverse_packet; - m_argImpl.template writePacket( - ReverseRow ? m_rows.value() - row - OffsetRow : row, - ReverseCol ? m_cols.value() - col - OffsetCol : col, - reverse_packet::run(x)); - } - template - EIGEN_STRONG_INLINE - void writePacket(Index index, const PacketType& x) - { - enum { PacketSize = unpacket_traits::size }; - m_argImpl.template writePacket - (m_rows.value() * m_cols.value() - index - PacketSize, preverse(x)); - } - -protected: - evaluator m_argImpl; - - // If we do not reverse rows, then we do not need to know the number of rows; same for columns - // Nonetheless, in this case it is important to set to 1 such that the coeff(index) method works fine for vectors. - const variable_if_dynamic m_rows; - const variable_if_dynamic m_cols; -}; + EIGEN_DEVICE_FUNC explicit evaluator(const XprType &diagonal) + : m_argImpl(diagonal.nestedExpression()), m_index(diagonal.index()) + {} + typedef typename XprType::Scalar Scalar; + typedef typename XprType::CoeffReturnType CoeffReturnType; -// -------------------- Diagonal -------------------- + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE CoeffReturnType coeff(Index row, Index) const + { + return m_argImpl.coeff(row + rowOffset(), row + colOffset()); + } -template -struct evaluator > - : evaluator_base > -{ - typedef Diagonal XprType; - - enum { - CoeffReadCost = evaluator::CoeffReadCost, - - Flags = (unsigned int)(evaluator::Flags & (HereditaryBits | DirectAccessBit) & ~RowMajorBit) | LinearAccessBit, - - Alignment = 0 - }; + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE CoeffReturnType coeff(Index index) const + { + return m_argImpl.coeff(index + rowOffset(), index + colOffset()); + } - EIGEN_DEVICE_FUNC explicit evaluator(const XprType& diagonal) - : m_argImpl(diagonal.nestedExpression()), - m_index(diagonal.index()) - { } - - typedef typename XprType::Scalar Scalar; - typedef typename XprType::CoeffReturnType CoeffReturnType; + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Scalar &coeffRef(Index row, Index) + { + return m_argImpl.coeffRef(row + rowOffset(), row + colOffset()); + } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE - CoeffReturnType coeff(Index row, Index) const - { - return m_argImpl.coeff(row + rowOffset(), row + colOffset()); - } + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Scalar &coeffRef(Index index) + { + return m_argImpl.coeffRef(index + rowOffset(), index + colOffset()); + } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE - CoeffReturnType coeff(Index index) const - { - return m_argImpl.coeff(index + rowOffset(), index + colOffset()); - } + protected: + evaluator m_argImpl; + const internal::variable_if_dynamicindex m_index; - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE - Scalar& coeffRef(Index row, Index) - { - return m_argImpl.coeffRef(row + rowOffset(), row + colOffset()); - } + private: + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Index rowOffset() const { return m_index.value() > 0 ? 0 : -m_index.value(); } + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Index colOffset() const { return m_index.value() > 0 ? m_index.value() : 0; } + }; - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE - Scalar& coeffRef(Index index) - { - return m_argImpl.coeffRef(index + rowOffset(), index + colOffset()); - } -protected: - evaluator m_argImpl; - const internal::variable_if_dynamicindex m_index; + //---------------------------------------------------------------------- + // deprecated code + //---------------------------------------------------------------------- -private: - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Index rowOffset() const { return m_index.value() > 0 ? 0 : -m_index.value(); } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Index colOffset() const { return m_index.value() > 0 ? m_index.value() : 0; } -}; + // -------------------- EvalToTemp -------------------- + // expression class for evaluating nested expression to a temporary -//---------------------------------------------------------------------- -// deprecated code -//---------------------------------------------------------------------- + template class EvalToTemp; -// -------------------- EvalToTemp -------------------- + template struct traits> : public traits + { + }; -// expression class for evaluating nested expression to a temporary + template class EvalToTemp : public dense_xpr_base>::type + { + public: + typedef typename dense_xpr_base::type Base; + EIGEN_GENERIC_PUBLIC_INTERFACE(EvalToTemp) -template class EvalToTemp; + explicit EvalToTemp(const ArgType &arg) : m_arg(arg) {} -template -struct traits > - : public traits -{ }; + const ArgType &arg() const { return m_arg; } -template -class EvalToTemp - : public dense_xpr_base >::type -{ - public: - - typedef typename dense_xpr_base::type Base; - EIGEN_GENERIC_PUBLIC_INTERFACE(EvalToTemp) - - explicit EvalToTemp(const ArgType& arg) - : m_arg(arg) - { } - - const ArgType& arg() const - { - return m_arg; - } + Index rows() const { return m_arg.rows(); } - Index rows() const - { - return m_arg.rows(); - } + Index cols() const { return m_arg.cols(); } - Index cols() const - { - return m_arg.cols(); - } + private: + const ArgType &m_arg; + }; - private: - const ArgType& m_arg; -}; - -template -struct evaluator > - : public evaluator -{ - typedef EvalToTemp XprType; - typedef typename ArgType::PlainObject PlainObject; - typedef evaluator Base; - - EIGEN_DEVICE_FUNC explicit evaluator(const XprType& xpr) - : m_result(xpr.arg()) + template struct evaluator> : public evaluator { - ::new (static_cast(this)) Base(m_result); - } + typedef EvalToTemp XprType; + typedef typename ArgType::PlainObject PlainObject; + typedef evaluator Base; - // This constructor is used when nesting an EvalTo evaluator in another evaluator - EIGEN_DEVICE_FUNC evaluator(const ArgType& arg) - : m_result(arg) - { - ::new (static_cast(this)) Base(m_result); - } + EIGEN_DEVICE_FUNC explicit evaluator(const XprType &xpr) : m_result(xpr.arg()) + { + ::new (static_cast(this)) Base(m_result); + } -protected: - PlainObject m_result; -}; + // This constructor is used when nesting an EvalTo evaluator in another evaluator + EIGEN_DEVICE_FUNC evaluator(const ArgType &arg) : m_result(arg) + { + ::new (static_cast(this)) Base(m_result); + } + + protected: + PlainObject m_result; + }; -} // namespace internal +}// namespace internal -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_COREEVALUATORS_H +#endif// EIGEN_COREEVALUATORS_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/CoreIterators.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/CoreIterators.h index 4eb42b93..04949bc3 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/CoreIterators.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/CoreIterators.h @@ -10,118 +10,123 @@ #ifndef EIGEN_COREITERATORS_H #define EIGEN_COREITERATORS_H -namespace Eigen { +namespace Eigen { /* This file contains the respective InnerIterator definition of the expressions defined in Eigen/Core */ namespace internal { -template -class inner_iterator_selector; + template class inner_iterator_selector; } /** \class InnerIterator - * \brief An InnerIterator allows to loop over the element of any matrix expression. - * - * \warning To be used with care because an evaluator is constructed every time an InnerIterator iterator is constructed. - * - * TODO: add a usage example - */ -template -class InnerIterator + * \brief An InnerIterator allows to loop over the element of any matrix expression. + * + * \warning To be used with care because an evaluator is constructed every time an InnerIterator iterator is + * constructed. + * + * TODO: add a usage example + */ +template class InnerIterator { protected: typedef internal::inner_iterator_selector::Kind> IteratorType; typedef internal::evaluator EvaluatorType; typedef typename internal::traits::Scalar Scalar; + public: /** Construct an iterator over the \a outerId -th row or column of \a xpr */ - InnerIterator(const XprType &xpr, const Index &outerId) - : m_eval(xpr), m_iter(m_eval, outerId, xpr.innerSize()) - {} - + InnerIterator(const XprType &xpr, const Index &outerId) : m_eval(xpr), m_iter(m_eval, outerId, xpr.innerSize()) {} + /// \returns the value of the current coefficient. - EIGEN_STRONG_INLINE Scalar value() const { return m_iter.value(); } + EIGEN_STRONG_INLINE Scalar value() const { return m_iter.value(); } /** Increment the iterator \c *this to the next non-zero coefficient. - * Explicit zeros are not skipped over. To skip explicit zeros, see class SparseView - */ - EIGEN_STRONG_INLINE InnerIterator& operator++() { m_iter.operator++(); return *this; } + * Explicit zeros are not skipped over. To skip explicit zeros, see class SparseView + */ + EIGEN_STRONG_INLINE InnerIterator &operator++() + { + m_iter.operator++(); + return *this; + } /// \returns the column or row index of the current coefficient. - EIGEN_STRONG_INLINE Index index() const { return m_iter.index(); } + EIGEN_STRONG_INLINE Index index() const { return m_iter.index(); } /// \returns the row index of the current coefficient. - EIGEN_STRONG_INLINE Index row() const { return m_iter.row(); } + EIGEN_STRONG_INLINE Index row() const { return m_iter.row(); } /// \returns the column index of the current coefficient. - EIGEN_STRONG_INLINE Index col() const { return m_iter.col(); } + EIGEN_STRONG_INLINE Index col() const { return m_iter.col(); } /// \returns \c true if the iterator \c *this still references a valid coefficient. - EIGEN_STRONG_INLINE operator bool() const { return m_iter; } - + EIGEN_STRONG_INLINE operator bool() const { return m_iter; } + protected: EvaluatorType m_eval; IteratorType m_iter; + private: // If you get here, then you're not using the right InnerIterator type, e.g.: // SparseMatrix A; // SparseMatrix::InnerIterator it(A,0); - template InnerIterator(const EigenBase&,Index outer); + template InnerIterator(const EigenBase &, Index outer); }; namespace internal { -// Generic inner iterator implementation for dense objects -template -class inner_iterator_selector -{ -protected: - typedef evaluator EvaluatorType; - typedef typename traits::Scalar Scalar; - enum { IsRowMajor = (XprType::Flags&RowMajorBit)==RowMajorBit }; - -public: - EIGEN_STRONG_INLINE inner_iterator_selector(const EvaluatorType &eval, const Index &outerId, const Index &innerSize) - : m_eval(eval), m_inner(0), m_outer(outerId), m_end(innerSize) - {} - - EIGEN_STRONG_INLINE Scalar value() const + // Generic inner iterator implementation for dense objects + template class inner_iterator_selector { - return (IsRowMajor) ? m_eval.coeff(m_outer, m_inner) - : m_eval.coeff(m_inner, m_outer); - } - - EIGEN_STRONG_INLINE inner_iterator_selector& operator++() { m_inner++; return *this; } - - EIGEN_STRONG_INLINE Index index() const { return m_inner; } - inline Index row() const { return IsRowMajor ? m_outer : index(); } - inline Index col() const { return IsRowMajor ? index() : m_outer; } - - EIGEN_STRONG_INLINE operator bool() const { return m_inner < m_end && m_inner>=0; } - -protected: - const EvaluatorType& m_eval; - Index m_inner; - const Index m_outer; - const Index m_end; -}; + protected: + typedef evaluator EvaluatorType; + typedef typename traits::Scalar Scalar; + enum { IsRowMajor = (XprType::Flags & RowMajorBit) == RowMajorBit }; + + public: + EIGEN_STRONG_INLINE inner_iterator_selector(const EvaluatorType &eval, const Index &outerId, const Index &innerSize) + : m_eval(eval), m_inner(0), m_outer(outerId), m_end(innerSize) + {} + + EIGEN_STRONG_INLINE Scalar value() const + { + return (IsRowMajor) ? m_eval.coeff(m_outer, m_inner) : m_eval.coeff(m_inner, m_outer); + } + + EIGEN_STRONG_INLINE inner_iterator_selector &operator++() + { + m_inner++; + return *this; + } + + EIGEN_STRONG_INLINE Index index() const { return m_inner; } + inline Index row() const { return IsRowMajor ? m_outer : index(); } + inline Index col() const { return IsRowMajor ? index() : m_outer; } + + EIGEN_STRONG_INLINE operator bool() const { return m_inner < m_end && m_inner >= 0; } + + protected: + const EvaluatorType &m_eval; + Index m_inner; + const Index m_outer; + const Index m_end; + }; + + // For iterator-based evaluator, inner-iterator is already implemented as + // evaluator<>::InnerIterator + template + class inner_iterator_selector : public evaluator::InnerIterator + { + protected: + typedef typename evaluator::InnerIterator Base; + typedef evaluator EvaluatorType; -// For iterator-based evaluator, inner-iterator is already implemented as -// evaluator<>::InnerIterator -template -class inner_iterator_selector - : public evaluator::InnerIterator -{ -protected: - typedef typename evaluator::InnerIterator Base; - typedef evaluator EvaluatorType; - -public: - EIGEN_STRONG_INLINE inner_iterator_selector(const EvaluatorType &eval, const Index &outerId, const Index &/*innerSize*/) - : Base(eval, outerId) - {} -}; + public: + EIGEN_STRONG_INLINE + inner_iterator_selector(const EvaluatorType &eval, const Index &outerId, const Index & /*innerSize*/) + : Base(eval, outerId) + {} + }; -} // end namespace internal +}// end namespace internal -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_COREITERATORS_H +#endif// EIGEN_COREITERATORS_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/CwiseBinaryOp.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/CwiseBinaryOp.h index a36765e3..1abbe344 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/CwiseBinaryOp.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/CwiseBinaryOp.h @@ -14,171 +14,167 @@ namespace Eigen { namespace internal { -template -struct traits > -{ - // we must not inherit from traits since it has - // the potential to cause problems with MSVC - typedef typename remove_all::type Ancestor; - typedef typename traits::XprKind XprKind; - enum { - RowsAtCompileTime = traits::RowsAtCompileTime, - ColsAtCompileTime = traits::ColsAtCompileTime, - MaxRowsAtCompileTime = traits::MaxRowsAtCompileTime, - MaxColsAtCompileTime = traits::MaxColsAtCompileTime + template struct traits> + { + // we must not inherit from traits since it has + // the potential to cause problems with MSVC + typedef typename remove_all::type Ancestor; + typedef typename traits::XprKind XprKind; + enum { + RowsAtCompileTime = traits::RowsAtCompileTime, + ColsAtCompileTime = traits::ColsAtCompileTime, + MaxRowsAtCompileTime = traits::MaxRowsAtCompileTime, + MaxColsAtCompileTime = traits::MaxColsAtCompileTime + }; + + // even though we require Lhs and Rhs to have the same scalar type (see CwiseBinaryOp constructor), + // we still want to handle the case when the result type is different. + typedef typename result_of::type Scalar; + typedef typename cwise_promote_storage_type::StorageKind, + typename traits::StorageKind, + BinaryOp>::ret StorageKind; + typedef typename promote_index_type::StorageIndex, typename traits::StorageIndex>::type + StorageIndex; + typedef typename Lhs::Nested LhsNested; + typedef typename Rhs::Nested RhsNested; + typedef typename remove_reference::type _LhsNested; + typedef typename remove_reference::type _RhsNested; + enum { + Flags = cwise_promote_storage_order::StorageKind, + typename traits::StorageKind, + _LhsNested::Flags & RowMajorBit, + _RhsNested::Flags & RowMajorBit>::value + }; }; +}// end namespace internal - // even though we require Lhs and Rhs to have the same scalar type (see CwiseBinaryOp constructor), - // we still want to handle the case when the result type is different. - typedef typename result_of< - BinaryOp( - const typename Lhs::Scalar&, - const typename Rhs::Scalar& - ) - >::type Scalar; - typedef typename cwise_promote_storage_type::StorageKind, - typename traits::StorageKind, - BinaryOp>::ret StorageKind; - typedef typename promote_index_type::StorageIndex, - typename traits::StorageIndex>::type StorageIndex; - typedef typename Lhs::Nested LhsNested; - typedef typename Rhs::Nested RhsNested; - typedef typename remove_reference::type _LhsNested; - typedef typename remove_reference::type _RhsNested; - enum { - Flags = cwise_promote_storage_order::StorageKind,typename traits::StorageKind,_LhsNested::Flags & RowMajorBit,_RhsNested::Flags & RowMajorBit>::value - }; -}; -} // end namespace internal - -template -class CwiseBinaryOpImpl; +template class CwiseBinaryOpImpl; /** \class CwiseBinaryOp - * \ingroup Core_Module - * - * \brief Generic expression where a coefficient-wise binary operator is applied to two expressions - * - * \tparam BinaryOp template functor implementing the operator - * \tparam LhsType the type of the left-hand side - * \tparam RhsType the type of the right-hand side - * - * This class represents an expression where a coefficient-wise binary operator is applied to two expressions. - * It is the return type of binary operators, by which we mean only those binary operators where - * both the left-hand side and the right-hand side are Eigen expressions. - * For example, the return type of matrix1+matrix2 is a CwiseBinaryOp. - * - * Most of the time, this is the only way that it is used, so you typically don't have to name - * CwiseBinaryOp types explicitly. - * - * \sa MatrixBase::binaryExpr(const MatrixBase &,const CustomBinaryOp &) const, class CwiseUnaryOp, class CwiseNullaryOp - */ + * \ingroup Core_Module + * + * \brief Generic expression where a coefficient-wise binary operator is applied to two expressions + * + * \tparam BinaryOp template functor implementing the operator + * \tparam LhsType the type of the left-hand side + * \tparam RhsType the type of the right-hand side + * + * This class represents an expression where a coefficient-wise binary operator is applied to two expressions. + * It is the return type of binary operators, by which we mean only those binary operators where + * both the left-hand side and the right-hand side are Eigen expressions. + * For example, the return type of matrix1+matrix2 is a CwiseBinaryOp. + * + * Most of the time, this is the only way that it is used, so you typically don't have to name + * CwiseBinaryOp types explicitly. + * + * \sa MatrixBase::binaryExpr(const MatrixBase &,const CustomBinaryOp &) const, class CwiseUnaryOp, class + * CwiseNullaryOp + */ template -class CwiseBinaryOp : - public CwiseBinaryOpImpl< - BinaryOp, LhsType, RhsType, - typename internal::cwise_promote_storage_type::StorageKind, - typename internal::traits::StorageKind, - BinaryOp>::ret>, - internal::no_assignment_operator +class CwiseBinaryOp + : public CwiseBinaryOpImpl::StorageKind, + typename internal::traits::StorageKind, + BinaryOp>::ret> + , internal::no_assignment_operator { - public: - - typedef typename internal::remove_all::type Functor; - typedef typename internal::remove_all::type Lhs; - typedef typename internal::remove_all::type Rhs; - - typedef typename CwiseBinaryOpImpl< - BinaryOp, LhsType, RhsType, - typename internal::cwise_promote_storage_type::StorageKind, - typename internal::traits::StorageKind, - BinaryOp>::ret>::Base Base; - EIGEN_GENERIC_PUBLIC_INTERFACE(CwiseBinaryOp) - - typedef typename internal::ref_selector::type LhsNested; - typedef typename internal::ref_selector::type RhsNested; - typedef typename internal::remove_reference::type _LhsNested; - typedef typename internal::remove_reference::type _RhsNested; - - EIGEN_DEVICE_FUNC - EIGEN_STRONG_INLINE CwiseBinaryOp(const Lhs& aLhs, const Rhs& aRhs, const BinaryOp& func = BinaryOp()) - : m_lhs(aLhs), m_rhs(aRhs), m_functor(func) - { - EIGEN_CHECK_BINARY_COMPATIBILIY(BinaryOp,typename Lhs::Scalar,typename Rhs::Scalar); - // require the sizes to match - EIGEN_STATIC_ASSERT_SAME_MATRIX_SIZE(Lhs, Rhs) - eigen_assert(aLhs.rows() == aRhs.rows() && aLhs.cols() == aRhs.cols()); - } - - EIGEN_DEVICE_FUNC - EIGEN_STRONG_INLINE Index rows() const { - // return the fixed size type if available to enable compile time optimizations - if (internal::traits::type>::RowsAtCompileTime==Dynamic) - return m_rhs.rows(); - else - return m_lhs.rows(); - } - EIGEN_DEVICE_FUNC - EIGEN_STRONG_INLINE Index cols() const { - // return the fixed size type if available to enable compile time optimizations - if (internal::traits::type>::ColsAtCompileTime==Dynamic) - return m_rhs.cols(); - else - return m_lhs.cols(); - } - - /** \returns the left hand side nested expression */ - EIGEN_DEVICE_FUNC - const _LhsNested& lhs() const { return m_lhs; } - /** \returns the right hand side nested expression */ - EIGEN_DEVICE_FUNC - const _RhsNested& rhs() const { return m_rhs; } - /** \returns the functor representing the binary operation */ - EIGEN_DEVICE_FUNC - const BinaryOp& functor() const { return m_functor; } - - protected: - LhsNested m_lhs; - RhsNested m_rhs; - const BinaryOp m_functor; +public: + typedef typename internal::remove_all::type Functor; + typedef typename internal::remove_all::type Lhs; + typedef typename internal::remove_all::type Rhs; + + typedef typename CwiseBinaryOpImpl::StorageKind, + typename internal::traits::StorageKind, + BinaryOp>::ret>::Base Base; + EIGEN_GENERIC_PUBLIC_INTERFACE(CwiseBinaryOp) + + typedef typename internal::ref_selector::type LhsNested; + typedef typename internal::ref_selector::type RhsNested; + typedef typename internal::remove_reference::type _LhsNested; + typedef typename internal::remove_reference::type _RhsNested; + + EIGEN_DEVICE_FUNC + EIGEN_STRONG_INLINE CwiseBinaryOp(const Lhs &aLhs, const Rhs &aRhs, const BinaryOp &func = BinaryOp()) + : m_lhs(aLhs), m_rhs(aRhs), m_functor(func) + { + EIGEN_CHECK_BINARY_COMPATIBILIY(BinaryOp, typename Lhs::Scalar, typename Rhs::Scalar); + // require the sizes to match + EIGEN_STATIC_ASSERT_SAME_MATRIX_SIZE(Lhs, Rhs) + eigen_assert(aLhs.rows() == aRhs.rows() && aLhs.cols() == aRhs.cols()); + } + + EIGEN_DEVICE_FUNC + EIGEN_STRONG_INLINE Index rows() const + { + // return the fixed size type if available to enable compile time optimizations + if (internal::traits::type>::RowsAtCompileTime == Dynamic) + return m_rhs.rows(); + else + return m_lhs.rows(); + } + EIGEN_DEVICE_FUNC + EIGEN_STRONG_INLINE Index cols() const + { + // return the fixed size type if available to enable compile time optimizations + if (internal::traits::type>::ColsAtCompileTime == Dynamic) + return m_rhs.cols(); + else + return m_lhs.cols(); + } + + /** \returns the left hand side nested expression */ + EIGEN_DEVICE_FUNC + const _LhsNested &lhs() const { return m_lhs; } + /** \returns the right hand side nested expression */ + EIGEN_DEVICE_FUNC + const _RhsNested &rhs() const { return m_rhs; } + /** \returns the functor representing the binary operation */ + EIGEN_DEVICE_FUNC + const BinaryOp &functor() const { return m_functor; } + +protected: + LhsNested m_lhs; + RhsNested m_rhs; + const BinaryOp m_functor; }; // Generic API dispatcher template -class CwiseBinaryOpImpl - : public internal::generic_xpr_base >::type +class CwiseBinaryOpImpl : public internal::generic_xpr_base>::type { public: - typedef typename internal::generic_xpr_base >::type Base; + typedef typename internal::generic_xpr_base>::type Base; }; /** replaces \c *this by \c *this - \a other. - * - * \returns a reference to \c *this - */ + * + * \returns a reference to \c *this + */ template template -EIGEN_STRONG_INLINE Derived & -MatrixBase::operator-=(const MatrixBase &other) +EIGEN_STRONG_INLINE Derived &MatrixBase::operator-=(const MatrixBase &other) { - call_assignment(derived(), other.derived(), internal::sub_assign_op()); + call_assignment(derived(), other.derived(), internal::sub_assign_op()); return derived(); } /** replaces \c *this by \c *this + \a other. - * - * \returns a reference to \c *this - */ + * + * \returns a reference to \c *this + */ template template -EIGEN_STRONG_INLINE Derived & -MatrixBase::operator+=(const MatrixBase& other) +EIGEN_STRONG_INLINE Derived &MatrixBase::operator+=(const MatrixBase &other) { - call_assignment(derived(), other.derived(), internal::add_assign_op()); + call_assignment(derived(), other.derived(), internal::add_assign_op()); return derived(); } -} // end namespace Eigen - -#endif // EIGEN_CWISE_BINARY_OP_H +}// end namespace Eigen +#endif// EIGEN_CWISE_BINARY_OP_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/CwiseNullaryOp.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/CwiseNullaryOp.h index ddd607e3..649990e1 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/CwiseNullaryOp.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/CwiseNullaryOp.h @@ -13,15 +13,13 @@ namespace Eigen { namespace internal { -template -struct traits > : traits -{ - enum { - Flags = traits::Flags & RowMajorBit + template + struct traits> : traits + { + enum { Flags = traits::Flags & RowMajorBit }; }; -}; -} // namespace internal +}// namespace internal /** \class CwiseNullaryOp * \ingroup Core_Module @@ -40,11 +38,14 @@ struct traits > : traits - \c operator()() if the procedural generation does not depend on the coefficient entries (e.g., random numbers) - \c operator()(Index i)if the procedural generation makes sense for vectors only and that it depends on the coefficient index \c i (e.g., linspace) - \c operator()(Index i,Index j)if the procedural generation depends on the matrix coordinates \c i, \c j (e.g., to generate a checkerboard with 0 and 1) + \c operator()() if the procedural generation does not depend on the coefficient entries + (e.g., random numbers) \c operator()(Index i)if the procedural generation makes + sense for vectors only and that it depends on the coefficient index \c i (e.g., linspace) \c + operator()(Index i,Index j)if the procedural generation depends on the matrix coordinates \c i, \c j (e.g., + to generate a checkerboard with 0 and 1) - * It is also possible to expose the last two operators if the generation makes sense for matrices but can be optimized for vectors. + * It is also possible to expose the last two operators if the generation makes sense for matrices but can be optimized + for vectors. * * See DenseBase::NullaryExpr(Index,const CustomNullaryOp&) for an example binding * C++11 random number generators. @@ -57,350 +58,353 @@ struct traits > : traits -class CwiseNullaryOp : public internal::dense_xpr_base< CwiseNullaryOp >::type, internal::no_assignment_operator +class CwiseNullaryOp + : public internal::dense_xpr_base>::type + , internal::no_assignment_operator { - public: +public: + typedef typename internal::dense_xpr_base::type Base; + EIGEN_DENSE_PUBLIC_INTERFACE(CwiseNullaryOp) - typedef typename internal::dense_xpr_base::type Base; - EIGEN_DENSE_PUBLIC_INTERFACE(CwiseNullaryOp) - - EIGEN_DEVICE_FUNC - CwiseNullaryOp(Index rows, Index cols, const NullaryOp& func = NullaryOp()) - : m_rows(rows), m_cols(cols), m_functor(func) - { - eigen_assert(rows >= 0 - && (RowsAtCompileTime == Dynamic || RowsAtCompileTime == rows) - && cols >= 0 - && (ColsAtCompileTime == Dynamic || ColsAtCompileTime == cols)); - } + EIGEN_DEVICE_FUNC + CwiseNullaryOp(Index rows, Index cols, const NullaryOp &func = NullaryOp()) + : m_rows(rows), m_cols(cols), m_functor(func) + { + eigen_assert(rows >= 0 && (RowsAtCompileTime == Dynamic || RowsAtCompileTime == rows) && cols >= 0 + && (ColsAtCompileTime == Dynamic || ColsAtCompileTime == cols)); + } - EIGEN_DEVICE_FUNC - EIGEN_STRONG_INLINE Index rows() const { return m_rows.value(); } - EIGEN_DEVICE_FUNC - EIGEN_STRONG_INLINE Index cols() const { return m_cols.value(); } + EIGEN_DEVICE_FUNC + EIGEN_STRONG_INLINE Index rows() const { return m_rows.value(); } + EIGEN_DEVICE_FUNC + EIGEN_STRONG_INLINE Index cols() const { return m_cols.value(); } - /** \returns the functor representing the nullary operation */ - EIGEN_DEVICE_FUNC - const NullaryOp& functor() const { return m_functor; } + /** \returns the functor representing the nullary operation */ + EIGEN_DEVICE_FUNC + const NullaryOp &functor() const { return m_functor; } - protected: - const internal::variable_if_dynamic m_rows; - const internal::variable_if_dynamic m_cols; - const NullaryOp m_functor; +protected: + const internal::variable_if_dynamic m_rows; + const internal::variable_if_dynamic m_cols; + const NullaryOp m_functor; }; /** \returns an expression of a matrix defined by a custom functor \a func - * - * The parameters \a rows and \a cols are the number of rows and of columns of - * the returned matrix. Must be compatible with this MatrixBase type. - * - * This variant is meant to be used for dynamic-size matrix types. For fixed-size types, - * it is redundant to pass \a rows and \a cols as arguments, so Zero() should be used - * instead. - * - * The template parameter \a CustomNullaryOp is the type of the functor. - * - * \sa class CwiseNullaryOp - */ + * + * The parameters \a rows and \a cols are the number of rows and of columns of + * the returned matrix. Must be compatible with this MatrixBase type. + * + * This variant is meant to be used for dynamic-size matrix types. For fixed-size types, + * it is redundant to pass \a rows and \a cols as arguments, so Zero() should be used + * instead. + * + * The template parameter \a CustomNullaryOp is the type of the functor. + * + * \sa class CwiseNullaryOp + */ template template EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const CwiseNullaryOp::PlainObject> -DenseBase::NullaryExpr(Index rows, Index cols, const CustomNullaryOp& func) + DenseBase::NullaryExpr(Index rows, Index cols, const CustomNullaryOp &func) { return CwiseNullaryOp(rows, cols, func); } /** \returns an expression of a matrix defined by a custom functor \a func - * - * The parameter \a size is the size of the returned vector. - * Must be compatible with this MatrixBase type. - * - * \only_for_vectors - * - * This variant is meant to be used for dynamic-size vector types. For fixed-size types, - * it is redundant to pass \a size as argument, so Zero() should be used - * instead. - * - * The template parameter \a CustomNullaryOp is the type of the functor. - * - * Here is an example with C++11 random generators: \include random_cpp11.cpp - * Output: \verbinclude random_cpp11.out - * - * \sa class CwiseNullaryOp - */ + * + * The parameter \a size is the size of the returned vector. + * Must be compatible with this MatrixBase type. + * + * \only_for_vectors + * + * This variant is meant to be used for dynamic-size vector types. For fixed-size types, + * it is redundant to pass \a size as argument, so Zero() should be used + * instead. + * + * The template parameter \a CustomNullaryOp is the type of the functor. + * + * Here is an example with C++11 random generators: \include random_cpp11.cpp + * Output: \verbinclude random_cpp11.out + * + * \sa class CwiseNullaryOp + */ template template EIGEN_STRONG_INLINE const CwiseNullaryOp::PlainObject> -DenseBase::NullaryExpr(Index size, const CustomNullaryOp& func) + DenseBase::NullaryExpr(Index size, const CustomNullaryOp &func) { EIGEN_STATIC_ASSERT_VECTOR_ONLY(Derived) - if(RowsAtCompileTime == 1) return CwiseNullaryOp(1, size, func); - else return CwiseNullaryOp(size, 1, func); + if (RowsAtCompileTime == 1) + return CwiseNullaryOp(1, size, func); + else + return CwiseNullaryOp(size, 1, func); } /** \returns an expression of a matrix defined by a custom functor \a func - * - * This variant is only for fixed-size DenseBase types. For dynamic-size types, you - * need to use the variants taking size arguments. - * - * The template parameter \a CustomNullaryOp is the type of the functor. - * - * \sa class CwiseNullaryOp - */ + * + * This variant is only for fixed-size DenseBase types. For dynamic-size types, you + * need to use the variants taking size arguments. + * + * The template parameter \a CustomNullaryOp is the type of the functor. + * + * \sa class CwiseNullaryOp + */ template template EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const CwiseNullaryOp::PlainObject> -DenseBase::NullaryExpr(const CustomNullaryOp& func) + DenseBase::NullaryExpr(const CustomNullaryOp &func) { return CwiseNullaryOp(RowsAtCompileTime, ColsAtCompileTime, func); } /** \returns an expression of a constant matrix of value \a value - * - * The parameters \a rows and \a cols are the number of rows and of columns of - * the returned matrix. Must be compatible with this DenseBase type. - * - * This variant is meant to be used for dynamic-size matrix types. For fixed-size types, - * it is redundant to pass \a rows and \a cols as arguments, so Zero() should be used - * instead. - * - * The template parameter \a CustomNullaryOp is the type of the functor. - * - * \sa class CwiseNullaryOp - */ + * + * The parameters \a rows and \a cols are the number of rows and of columns of + * the returned matrix. Must be compatible with this DenseBase type. + * + * This variant is meant to be used for dynamic-size matrix types. For fixed-size types, + * it is redundant to pass \a rows and \a cols as arguments, so Zero() should be used + * instead. + * + * The template parameter \a CustomNullaryOp is the type of the functor. + * + * \sa class CwiseNullaryOp + */ template EIGEN_STRONG_INLINE const typename DenseBase::ConstantReturnType -DenseBase::Constant(Index rows, Index cols, const Scalar& value) + DenseBase::Constant(Index rows, Index cols, const Scalar &value) { return DenseBase::NullaryExpr(rows, cols, internal::scalar_constant_op(value)); } /** \returns an expression of a constant matrix of value \a value - * - * The parameter \a size is the size of the returned vector. - * Must be compatible with this DenseBase type. - * - * \only_for_vectors - * - * This variant is meant to be used for dynamic-size vector types. For fixed-size types, - * it is redundant to pass \a size as argument, so Zero() should be used - * instead. - * - * The template parameter \a CustomNullaryOp is the type of the functor. - * - * \sa class CwiseNullaryOp - */ + * + * The parameter \a size is the size of the returned vector. + * Must be compatible with this DenseBase type. + * + * \only_for_vectors + * + * This variant is meant to be used for dynamic-size vector types. For fixed-size types, + * it is redundant to pass \a size as argument, so Zero() should be used + * instead. + * + * The template parameter \a CustomNullaryOp is the type of the functor. + * + * \sa class CwiseNullaryOp + */ template EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const typename DenseBase::ConstantReturnType -DenseBase::Constant(Index size, const Scalar& value) + DenseBase::Constant(Index size, const Scalar &value) { return DenseBase::NullaryExpr(size, internal::scalar_constant_op(value)); } /** \returns an expression of a constant matrix of value \a value - * - * This variant is only for fixed-size DenseBase types. For dynamic-size types, you - * need to use the variants taking size arguments. - * - * The template parameter \a CustomNullaryOp is the type of the functor. - * - * \sa class CwiseNullaryOp - */ + * + * This variant is only for fixed-size DenseBase types. For dynamic-size types, you + * need to use the variants taking size arguments. + * + * The template parameter \a CustomNullaryOp is the type of the functor. + * + * \sa class CwiseNullaryOp + */ template EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const typename DenseBase::ConstantReturnType -DenseBase::Constant(const Scalar& value) + DenseBase::Constant(const Scalar &value) { EIGEN_STATIC_ASSERT_FIXED_SIZE(Derived) - return DenseBase::NullaryExpr(RowsAtCompileTime, ColsAtCompileTime, internal::scalar_constant_op(value)); + return DenseBase::NullaryExpr( + RowsAtCompileTime, ColsAtCompileTime, internal::scalar_constant_op(value)); } /** \deprecated because of accuracy loss. In Eigen 3.3, it is an alias for LinSpaced(Index,const Scalar&,const Scalar&) - * - * \sa LinSpaced(Index,Scalar,Scalar), setLinSpaced(Index,const Scalar&,const Scalar&) - */ + * + * \sa LinSpaced(Index,Scalar,Scalar), setLinSpaced(Index,const Scalar&,const Scalar&) + */ template EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const typename DenseBase::RandomAccessLinSpacedReturnType -DenseBase::LinSpaced(Sequential_t, Index size, const Scalar& low, const Scalar& high) + DenseBase::LinSpaced(Sequential_t, Index size, const Scalar &low, const Scalar &high) { EIGEN_STATIC_ASSERT_VECTOR_ONLY(Derived) - return DenseBase::NullaryExpr(size, internal::linspaced_op(low,high,size)); + return DenseBase::NullaryExpr(size, internal::linspaced_op(low, high, size)); } /** \deprecated because of accuracy loss. In Eigen 3.3, it is an alias for LinSpaced(const Scalar&,const Scalar&) - * - * \sa LinSpaced(Scalar,Scalar) - */ + * + * \sa LinSpaced(Scalar,Scalar) + */ template EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const typename DenseBase::RandomAccessLinSpacedReturnType -DenseBase::LinSpaced(Sequential_t, const Scalar& low, const Scalar& high) + DenseBase::LinSpaced(Sequential_t, const Scalar &low, const Scalar &high) { EIGEN_STATIC_ASSERT_VECTOR_ONLY(Derived) EIGEN_STATIC_ASSERT_FIXED_SIZE(Derived) - return DenseBase::NullaryExpr(Derived::SizeAtCompileTime, internal::linspaced_op(low,high,Derived::SizeAtCompileTime)); + return DenseBase::NullaryExpr( + Derived::SizeAtCompileTime, internal::linspaced_op(low, high, Derived::SizeAtCompileTime)); } /** - * \brief Sets a linearly spaced vector. - * - * The function generates 'size' equally spaced values in the closed interval [low,high]. - * When size is set to 1, a vector of length 1 containing 'high' is returned. - * - * \only_for_vectors - * - * Example: \include DenseBase_LinSpaced.cpp - * Output: \verbinclude DenseBase_LinSpaced.out - * - * For integer scalar types, an even spacing is possible if and only if the length of the range, - * i.e., \c high-low is a scalar multiple of \c size-1, or if \c size is a scalar multiple of the - * number of values \c high-low+1 (meaning each value can be repeated the same number of time). - * If one of these two considions is not satisfied, then \c high is lowered to the largest value - * satisfying one of this constraint. - * Here are some examples: - * - * Example: \include DenseBase_LinSpacedInt.cpp - * Output: \verbinclude DenseBase_LinSpacedInt.out - * - * \sa setLinSpaced(Index,const Scalar&,const Scalar&), CwiseNullaryOp - */ + * \brief Sets a linearly spaced vector. + * + * The function generates 'size' equally spaced values in the closed interval [low,high]. + * When size is set to 1, a vector of length 1 containing 'high' is returned. + * + * \only_for_vectors + * + * Example: \include DenseBase_LinSpaced.cpp + * Output: \verbinclude DenseBase_LinSpaced.out + * + * For integer scalar types, an even spacing is possible if and only if the length of the range, + * i.e., \c high-low is a scalar multiple of \c size-1, or if \c size is a scalar multiple of the + * number of values \c high-low+1 (meaning each value can be repeated the same number of time). + * If one of these two considions is not satisfied, then \c high is lowered to the largest value + * satisfying one of this constraint. + * Here are some examples: + * + * Example: \include DenseBase_LinSpacedInt.cpp + * Output: \verbinclude DenseBase_LinSpacedInt.out + * + * \sa setLinSpaced(Index,const Scalar&,const Scalar&), CwiseNullaryOp + */ template EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const typename DenseBase::RandomAccessLinSpacedReturnType -DenseBase::LinSpaced(Index size, const Scalar& low, const Scalar& high) + DenseBase::LinSpaced(Index size, const Scalar &low, const Scalar &high) { EIGEN_STATIC_ASSERT_VECTOR_ONLY(Derived) - return DenseBase::NullaryExpr(size, internal::linspaced_op(low,high,size)); + return DenseBase::NullaryExpr(size, internal::linspaced_op(low, high, size)); } /** - * \copydoc DenseBase::LinSpaced(Index, const Scalar&, const Scalar&) - * Special version for fixed size types which does not require the size parameter. - */ + * \copydoc DenseBase::LinSpaced(Index, const Scalar&, const Scalar&) + * Special version for fixed size types which does not require the size parameter. + */ template EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const typename DenseBase::RandomAccessLinSpacedReturnType -DenseBase::LinSpaced(const Scalar& low, const Scalar& high) + DenseBase::LinSpaced(const Scalar &low, const Scalar &high) { EIGEN_STATIC_ASSERT_VECTOR_ONLY(Derived) EIGEN_STATIC_ASSERT_FIXED_SIZE(Derived) - return DenseBase::NullaryExpr(Derived::SizeAtCompileTime, internal::linspaced_op(low,high,Derived::SizeAtCompileTime)); + return DenseBase::NullaryExpr( + Derived::SizeAtCompileTime, internal::linspaced_op(low, high, Derived::SizeAtCompileTime)); } /** \returns true if all coefficients in this matrix are approximately equal to \a val, to within precision \a prec */ template -EIGEN_DEVICE_FUNC bool DenseBase::isApproxToConstant -(const Scalar& val, const RealScalar& prec) const +EIGEN_DEVICE_FUNC bool DenseBase::isApproxToConstant(const Scalar &val, const RealScalar &prec) const { - typename internal::nested_eval::type self(derived()); - for(Index j = 0; j < cols(); ++j) - for(Index i = 0; i < rows(); ++i) - if(!internal::isApprox(self.coeff(i, j), val, prec)) - return false; + typename internal::nested_eval::type self(derived()); + for (Index j = 0; j < cols(); ++j) + for (Index i = 0; i < rows(); ++i) + if (!internal::isApprox(self.coeff(i, j), val, prec)) return false; return true; } /** This is just an alias for isApproxToConstant(). - * - * \returns true if all coefficients in this matrix are approximately equal to \a value, to within precision \a prec */ + * + * \returns true if all coefficients in this matrix are approximately equal to \a value, to within precision \a prec */ template -EIGEN_DEVICE_FUNC bool DenseBase::isConstant -(const Scalar& val, const RealScalar& prec) const +EIGEN_DEVICE_FUNC bool DenseBase::isConstant(const Scalar &val, const RealScalar &prec) const { return isApproxToConstant(val, prec); } /** Alias for setConstant(): sets all coefficients in this expression to \a val. - * - * \sa setConstant(), Constant(), class CwiseNullaryOp - */ -template -EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void DenseBase::fill(const Scalar& val) + * + * \sa setConstant(), Constant(), class CwiseNullaryOp + */ +template EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void DenseBase::fill(const Scalar &val) { setConstant(val); } /** Sets all coefficients in this expression to value \a val. - * - * \sa fill(), setConstant(Index,const Scalar&), setConstant(Index,Index,const Scalar&), setZero(), setOnes(), Constant(), class CwiseNullaryOp, setZero(), setOnes() - */ + * + * \sa fill(), setConstant(Index,const Scalar&), setConstant(Index,Index,const Scalar&), setZero(), setOnes(), + * Constant(), class CwiseNullaryOp, setZero(), setOnes() + */ template -EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Derived& DenseBase::setConstant(const Scalar& val) +EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Derived &DenseBase::setConstant(const Scalar &val) { return derived() = Constant(rows(), cols(), val); } /** Resizes to the given \a size, and sets all coefficients in this expression to the given value \a val. - * - * \only_for_vectors - * - * Example: \include Matrix_setConstant_int.cpp - * Output: \verbinclude Matrix_setConstant_int.out - * - * \sa MatrixBase::setConstant(const Scalar&), setConstant(Index,Index,const Scalar&), class CwiseNullaryOp, MatrixBase::Constant(const Scalar&) - */ -template -EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Derived& -PlainObjectBase::setConstant(Index size, const Scalar& val) + * + * \only_for_vectors + * + * Example: \include Matrix_setConstant_int.cpp + * Output: \verbinclude Matrix_setConstant_int.out + * + * \sa MatrixBase::setConstant(const Scalar&), setConstant(Index,Index,const Scalar&), class CwiseNullaryOp, + * MatrixBase::Constant(const Scalar&) + */ +template +EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Derived &PlainObjectBase::setConstant(Index size, const Scalar &val) { resize(size); return setConstant(val); } /** Resizes to the given size, and sets all coefficients in this expression to the given value \a val. - * - * \param rows the new number of rows - * \param cols the new number of columns - * \param val the value to which all coefficients are set - * - * Example: \include Matrix_setConstant_int_int.cpp - * Output: \verbinclude Matrix_setConstant_int_int.out - * - * \sa MatrixBase::setConstant(const Scalar&), setConstant(Index,const Scalar&), class CwiseNullaryOp, MatrixBase::Constant(const Scalar&) - */ -template -EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Derived& -PlainObjectBase::setConstant(Index rows, Index cols, const Scalar& val) + * + * \param rows the new number of rows + * \param cols the new number of columns + * \param val the value to which all coefficients are set + * + * Example: \include Matrix_setConstant_int_int.cpp + * Output: \verbinclude Matrix_setConstant_int_int.out + * + * \sa MatrixBase::setConstant(const Scalar&), setConstant(Index,const Scalar&), class CwiseNullaryOp, + * MatrixBase::Constant(const Scalar&) + */ +template +EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Derived & + PlainObjectBase::setConstant(Index rows, Index cols, const Scalar &val) { resize(rows, cols); return setConstant(val); } /** - * \brief Sets a linearly spaced vector. - * - * The function generates 'size' equally spaced values in the closed interval [low,high]. - * When size is set to 1, a vector of length 1 containing 'high' is returned. - * - * \only_for_vectors - * - * Example: \include DenseBase_setLinSpaced.cpp - * Output: \verbinclude DenseBase_setLinSpaced.out - * - * For integer scalar types, do not miss the explanations on the definition - * of \link LinSpaced(Index,const Scalar&,const Scalar&) even spacing \endlink. - * - * \sa LinSpaced(Index,const Scalar&,const Scalar&), CwiseNullaryOp - */ -template -EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Derived& DenseBase::setLinSpaced(Index newSize, const Scalar& low, const Scalar& high) + * \brief Sets a linearly spaced vector. + * + * The function generates 'size' equally spaced values in the closed interval [low,high]. + * When size is set to 1, a vector of length 1 containing 'high' is returned. + * + * \only_for_vectors + * + * Example: \include DenseBase_setLinSpaced.cpp + * Output: \verbinclude DenseBase_setLinSpaced.out + * + * For integer scalar types, do not miss the explanations on the definition + * of \link LinSpaced(Index,const Scalar&,const Scalar&) even spacing \endlink. + * + * \sa LinSpaced(Index,const Scalar&,const Scalar&), CwiseNullaryOp + */ +template +EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Derived & + DenseBase::setLinSpaced(Index newSize, const Scalar &low, const Scalar &high) { EIGEN_STATIC_ASSERT_VECTOR_ONLY(Derived) - return derived() = Derived::NullaryExpr(newSize, internal::linspaced_op(low,high,newSize)); + return derived() = Derived::NullaryExpr(newSize, internal::linspaced_op(low, high, newSize)); } /** - * \brief Sets a linearly spaced vector. - * - * The function fills \c *this with equally spaced values in the closed interval [low,high]. - * When size is set to 1, a vector of length 1 containing 'high' is returned. - * - * \only_for_vectors - * - * For integer scalar types, do not miss the explanations on the definition - * of \link LinSpaced(Index,const Scalar&,const Scalar&) even spacing \endlink. - * - * \sa LinSpaced(Index,const Scalar&,const Scalar&), setLinSpaced(Index, const Scalar&, const Scalar&), CwiseNullaryOp - */ -template -EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Derived& DenseBase::setLinSpaced(const Scalar& low, const Scalar& high) + * \brief Sets a linearly spaced vector. + * + * The function fills \c *this with equally spaced values in the closed interval [low,high]. + * When size is set to 1, a vector of length 1 containing 'high' is returned. + * + * \only_for_vectors + * + * For integer scalar types, do not miss the explanations on the definition + * of \link LinSpaced(Index,const Scalar&,const Scalar&) even spacing \endlink. + * + * \sa LinSpaced(Index,const Scalar&,const Scalar&), setLinSpaced(Index, const Scalar&, const Scalar&), CwiseNullaryOp + */ +template +EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Derived &DenseBase::setLinSpaced(const Scalar &low, const Scalar &high) { EIGEN_STATIC_ASSERT_VECTOR_ONLY(Derived) return setLinSpaced(size(), low, high); @@ -409,128 +413,122 @@ EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Derived& DenseBase::setLinSpaced( // zero: /** \returns an expression of a zero matrix. - * - * The parameters \a rows and \a cols are the number of rows and of columns of - * the returned matrix. Must be compatible with this MatrixBase type. - * - * This variant is meant to be used for dynamic-size matrix types. For fixed-size types, - * it is redundant to pass \a rows and \a cols as arguments, so Zero() should be used - * instead. - * - * Example: \include MatrixBase_zero_int_int.cpp - * Output: \verbinclude MatrixBase_zero_int_int.out - * - * \sa Zero(), Zero(Index) - */ + * + * The parameters \a rows and \a cols are the number of rows and of columns of + * the returned matrix. Must be compatible with this MatrixBase type. + * + * This variant is meant to be used for dynamic-size matrix types. For fixed-size types, + * it is redundant to pass \a rows and \a cols as arguments, so Zero() should be used + * instead. + * + * Example: \include MatrixBase_zero_int_int.cpp + * Output: \verbinclude MatrixBase_zero_int_int.out + * + * \sa Zero(), Zero(Index) + */ template EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const typename DenseBase::ConstantReturnType -DenseBase::Zero(Index rows, Index cols) + DenseBase::Zero(Index rows, Index cols) { return Constant(rows, cols, Scalar(0)); } /** \returns an expression of a zero vector. - * - * The parameter \a size is the size of the returned vector. - * Must be compatible with this MatrixBase type. - * - * \only_for_vectors - * - * This variant is meant to be used for dynamic-size vector types. For fixed-size types, - * it is redundant to pass \a size as argument, so Zero() should be used - * instead. - * - * Example: \include MatrixBase_zero_int.cpp - * Output: \verbinclude MatrixBase_zero_int.out - * - * \sa Zero(), Zero(Index,Index) - */ -template -EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const typename DenseBase::ConstantReturnType -DenseBase::Zero(Index size) + * + * The parameter \a size is the size of the returned vector. + * Must be compatible with this MatrixBase type. + * + * \only_for_vectors + * + * This variant is meant to be used for dynamic-size vector types. For fixed-size types, + * it is redundant to pass \a size as argument, so Zero() should be used + * instead. + * + * Example: \include MatrixBase_zero_int.cpp + * Output: \verbinclude MatrixBase_zero_int.out + * + * \sa Zero(), Zero(Index,Index) + */ +template +EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const typename DenseBase::ConstantReturnType DenseBase::Zero( + Index size) { return Constant(size, Scalar(0)); } /** \returns an expression of a fixed-size zero matrix or vector. - * - * This variant is only for fixed-size MatrixBase types. For dynamic-size types, you - * need to use the variants taking size arguments. - * - * Example: \include MatrixBase_zero.cpp - * Output: \verbinclude MatrixBase_zero.out - * - * \sa Zero(Index), Zero(Index,Index) - */ -template -EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const typename DenseBase::ConstantReturnType -DenseBase::Zero() + * + * This variant is only for fixed-size MatrixBase types. For dynamic-size types, you + * need to use the variants taking size arguments. + * + * Example: \include MatrixBase_zero.cpp + * Output: \verbinclude MatrixBase_zero.out + * + * \sa Zero(Index), Zero(Index,Index) + */ +template +EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const typename DenseBase::ConstantReturnType DenseBase::Zero() { return Constant(Scalar(0)); } /** \returns true if *this is approximately equal to the zero matrix, - * within the precision given by \a prec. - * - * Example: \include MatrixBase_isZero.cpp - * Output: \verbinclude MatrixBase_isZero.out - * - * \sa class CwiseNullaryOp, Zero() - */ -template -EIGEN_DEVICE_FUNC bool DenseBase::isZero(const RealScalar& prec) const -{ - typename internal::nested_eval::type self(derived()); - for(Index j = 0; j < cols(); ++j) - for(Index i = 0; i < rows(); ++i) - if(!internal::isMuchSmallerThan(self.coeff(i, j), static_cast(1), prec)) - return false; + * within the precision given by \a prec. + * + * Example: \include MatrixBase_isZero.cpp + * Output: \verbinclude MatrixBase_isZero.out + * + * \sa class CwiseNullaryOp, Zero() + */ +template EIGEN_DEVICE_FUNC bool DenseBase::isZero(const RealScalar &prec) const +{ + typename internal::nested_eval::type self(derived()); + for (Index j = 0; j < cols(); ++j) + for (Index i = 0; i < rows(); ++i) + if (!internal::isMuchSmallerThan(self.coeff(i, j), static_cast(1), prec)) return false; return true; } /** Sets all coefficients in this expression to zero. - * - * Example: \include MatrixBase_setZero.cpp - * Output: \verbinclude MatrixBase_setZero.out - * - * \sa class CwiseNullaryOp, Zero() - */ -template -EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Derived& DenseBase::setZero() + * + * Example: \include MatrixBase_setZero.cpp + * Output: \verbinclude MatrixBase_setZero.out + * + * \sa class CwiseNullaryOp, Zero() + */ +template EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Derived &DenseBase::setZero() { return setConstant(Scalar(0)); } /** Resizes to the given \a size, and sets all coefficients in this expression to zero. - * - * \only_for_vectors - * - * Example: \include Matrix_setZero_int.cpp - * Output: \verbinclude Matrix_setZero_int.out - * - * \sa DenseBase::setZero(), setZero(Index,Index), class CwiseNullaryOp, DenseBase::Zero() - */ + * + * \only_for_vectors + * + * Example: \include Matrix_setZero_int.cpp + * Output: \verbinclude Matrix_setZero_int.out + * + * \sa DenseBase::setZero(), setZero(Index,Index), class CwiseNullaryOp, DenseBase::Zero() + */ template -EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Derived& -PlainObjectBase::setZero(Index newSize) +EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Derived &PlainObjectBase::setZero(Index newSize) { resize(newSize); return setConstant(Scalar(0)); } /** Resizes to the given size, and sets all coefficients in this expression to zero. - * - * \param rows the new number of rows - * \param cols the new number of columns - * - * Example: \include Matrix_setZero_int_int.cpp - * Output: \verbinclude Matrix_setZero_int_int.out - * - * \sa DenseBase::setZero(), setZero(Index), class CwiseNullaryOp, DenseBase::Zero() - */ -template -EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Derived& -PlainObjectBase::setZero(Index rows, Index cols) + * + * \param rows the new number of rows + * \param cols the new number of columns + * + * Example: \include Matrix_setZero_int_int.cpp + * Output: \verbinclude Matrix_setZero_int_int.out + * + * \sa DenseBase::setZero(), setZero(Index), class CwiseNullaryOp, DenseBase::Zero() + */ +template +EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Derived &PlainObjectBase::setZero(Index rows, Index cols) { resize(rows, cols); return setConstant(Scalar(0)); @@ -539,124 +537,118 @@ PlainObjectBase::setZero(Index rows, Index cols) // ones: /** \returns an expression of a matrix where all coefficients equal one. - * - * The parameters \a rows and \a cols are the number of rows and of columns of - * the returned matrix. Must be compatible with this MatrixBase type. - * - * This variant is meant to be used for dynamic-size matrix types. For fixed-size types, - * it is redundant to pass \a rows and \a cols as arguments, so Ones() should be used - * instead. - * - * Example: \include MatrixBase_ones_int_int.cpp - * Output: \verbinclude MatrixBase_ones_int_int.out - * - * \sa Ones(), Ones(Index), isOnes(), class Ones - */ + * + * The parameters \a rows and \a cols are the number of rows and of columns of + * the returned matrix. Must be compatible with this MatrixBase type. + * + * This variant is meant to be used for dynamic-size matrix types. For fixed-size types, + * it is redundant to pass \a rows and \a cols as arguments, so Ones() should be used + * instead. + * + * Example: \include MatrixBase_ones_int_int.cpp + * Output: \verbinclude MatrixBase_ones_int_int.out + * + * \sa Ones(), Ones(Index), isOnes(), class Ones + */ template EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const typename DenseBase::ConstantReturnType -DenseBase::Ones(Index rows, Index cols) + DenseBase::Ones(Index rows, Index cols) { return Constant(rows, cols, Scalar(1)); } /** \returns an expression of a vector where all coefficients equal one. - * - * The parameter \a newSize is the size of the returned vector. - * Must be compatible with this MatrixBase type. - * - * \only_for_vectors - * - * This variant is meant to be used for dynamic-size vector types. For fixed-size types, - * it is redundant to pass \a size as argument, so Ones() should be used - * instead. - * - * Example: \include MatrixBase_ones_int.cpp - * Output: \verbinclude MatrixBase_ones_int.out - * - * \sa Ones(), Ones(Index,Index), isOnes(), class Ones - */ -template -EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const typename DenseBase::ConstantReturnType -DenseBase::Ones(Index newSize) + * + * The parameter \a newSize is the size of the returned vector. + * Must be compatible with this MatrixBase type. + * + * \only_for_vectors + * + * This variant is meant to be used for dynamic-size vector types. For fixed-size types, + * it is redundant to pass \a size as argument, so Ones() should be used + * instead. + * + * Example: \include MatrixBase_ones_int.cpp + * Output: \verbinclude MatrixBase_ones_int.out + * + * \sa Ones(), Ones(Index,Index), isOnes(), class Ones + */ +template +EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const typename DenseBase::ConstantReturnType DenseBase::Ones( + Index newSize) { return Constant(newSize, Scalar(1)); } /** \returns an expression of a fixed-size matrix or vector where all coefficients equal one. - * - * This variant is only for fixed-size MatrixBase types. For dynamic-size types, you - * need to use the variants taking size arguments. - * - * Example: \include MatrixBase_ones.cpp - * Output: \verbinclude MatrixBase_ones.out - * - * \sa Ones(Index), Ones(Index,Index), isOnes(), class Ones - */ -template -EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const typename DenseBase::ConstantReturnType -DenseBase::Ones() + * + * This variant is only for fixed-size MatrixBase types. For dynamic-size types, you + * need to use the variants taking size arguments. + * + * Example: \include MatrixBase_ones.cpp + * Output: \verbinclude MatrixBase_ones.out + * + * \sa Ones(Index), Ones(Index,Index), isOnes(), class Ones + */ +template +EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const typename DenseBase::ConstantReturnType DenseBase::Ones() { return Constant(Scalar(1)); } /** \returns true if *this is approximately equal to the matrix where all coefficients - * are equal to 1, within the precision given by \a prec. - * - * Example: \include MatrixBase_isOnes.cpp - * Output: \verbinclude MatrixBase_isOnes.out - * - * \sa class CwiseNullaryOp, Ones() - */ -template -EIGEN_DEVICE_FUNC bool DenseBase::isOnes -(const RealScalar& prec) const + * are equal to 1, within the precision given by \a prec. + * + * Example: \include MatrixBase_isOnes.cpp + * Output: \verbinclude MatrixBase_isOnes.out + * + * \sa class CwiseNullaryOp, Ones() + */ +template EIGEN_DEVICE_FUNC bool DenseBase::isOnes(const RealScalar &prec) const { return isApproxToConstant(Scalar(1), prec); } /** Sets all coefficients in this expression to one. - * - * Example: \include MatrixBase_setOnes.cpp - * Output: \verbinclude MatrixBase_setOnes.out - * - * \sa class CwiseNullaryOp, Ones() - */ -template -EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Derived& DenseBase::setOnes() + * + * Example: \include MatrixBase_setOnes.cpp + * Output: \verbinclude MatrixBase_setOnes.out + * + * \sa class CwiseNullaryOp, Ones() + */ +template EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Derived &DenseBase::setOnes() { return setConstant(Scalar(1)); } /** Resizes to the given \a newSize, and sets all coefficients in this expression to one. - * - * \only_for_vectors - * - * Example: \include Matrix_setOnes_int.cpp - * Output: \verbinclude Matrix_setOnes_int.out - * - * \sa MatrixBase::setOnes(), setOnes(Index,Index), class CwiseNullaryOp, MatrixBase::Ones() - */ + * + * \only_for_vectors + * + * Example: \include Matrix_setOnes_int.cpp + * Output: \verbinclude Matrix_setOnes_int.out + * + * \sa MatrixBase::setOnes(), setOnes(Index,Index), class CwiseNullaryOp, MatrixBase::Ones() + */ template -EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Derived& -PlainObjectBase::setOnes(Index newSize) +EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Derived &PlainObjectBase::setOnes(Index newSize) { resize(newSize); return setConstant(Scalar(1)); } /** Resizes to the given size, and sets all coefficients in this expression to one. - * - * \param rows the new number of rows - * \param cols the new number of columns - * - * Example: \include Matrix_setOnes_int_int.cpp - * Output: \verbinclude Matrix_setOnes_int_int.out - * - * \sa MatrixBase::setOnes(), setOnes(Index), class CwiseNullaryOp, MatrixBase::Ones() - */ -template -EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Derived& -PlainObjectBase::setOnes(Index rows, Index cols) + * + * \param rows the new number of rows + * \param cols the new number of columns + * + * Example: \include Matrix_setOnes_int_int.cpp + * Output: \verbinclude Matrix_setOnes_int_int.out + * + * \sa MatrixBase::setOnes(), setOnes(Index), class CwiseNullaryOp, MatrixBase::Ones() + */ +template +EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Derived &PlainObjectBase::setOnes(Index rows, Index cols) { resize(rows, cols); return setConstant(Scalar(1)); @@ -665,71 +657,62 @@ PlainObjectBase::setOnes(Index rows, Index cols) // Identity: /** \returns an expression of the identity matrix (not necessarily square). - * - * The parameters \a rows and \a cols are the number of rows and of columns of - * the returned matrix. Must be compatible with this MatrixBase type. - * - * This variant is meant to be used for dynamic-size matrix types. For fixed-size types, - * it is redundant to pass \a rows and \a cols as arguments, so Identity() should be used - * instead. - * - * Example: \include MatrixBase_identity_int_int.cpp - * Output: \verbinclude MatrixBase_identity_int_int.out - * - * \sa Identity(), setIdentity(), isIdentity() - */ + * + * The parameters \a rows and \a cols are the number of rows and of columns of + * the returned matrix. Must be compatible with this MatrixBase type. + * + * This variant is meant to be used for dynamic-size matrix types. For fixed-size types, + * it is redundant to pass \a rows and \a cols as arguments, so Identity() should be used + * instead. + * + * Example: \include MatrixBase_identity_int_int.cpp + * Output: \verbinclude MatrixBase_identity_int_int.out + * + * \sa Identity(), setIdentity(), isIdentity() + */ template EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const typename MatrixBase::IdentityReturnType -MatrixBase::Identity(Index rows, Index cols) + MatrixBase::Identity(Index rows, Index cols) { return DenseBase::NullaryExpr(rows, cols, internal::scalar_identity_op()); } /** \returns an expression of the identity matrix (not necessarily square). - * - * This variant is only for fixed-size MatrixBase types. For dynamic-size types, you - * need to use the variant taking size arguments. - * - * Example: \include MatrixBase_identity.cpp - * Output: \verbinclude MatrixBase_identity.out - * - * \sa Identity(Index,Index), setIdentity(), isIdentity() - */ + * + * This variant is only for fixed-size MatrixBase types. For dynamic-size types, you + * need to use the variant taking size arguments. + * + * Example: \include MatrixBase_identity.cpp + * Output: \verbinclude MatrixBase_identity.out + * + * \sa Identity(Index,Index), setIdentity(), isIdentity() + */ template EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const typename MatrixBase::IdentityReturnType -MatrixBase::Identity() + MatrixBase::Identity() { EIGEN_STATIC_ASSERT_FIXED_SIZE(Derived) return MatrixBase::NullaryExpr(RowsAtCompileTime, ColsAtCompileTime, internal::scalar_identity_op()); } /** \returns true if *this is approximately equal to the identity matrix - * (not necessarily square), - * within the precision given by \a prec. - * - * Example: \include MatrixBase_isIdentity.cpp - * Output: \verbinclude MatrixBase_isIdentity.out - * - * \sa class CwiseNullaryOp, Identity(), Identity(Index,Index), setIdentity() - */ -template -bool MatrixBase::isIdentity -(const RealScalar& prec) const -{ - typename internal::nested_eval::type self(derived()); - for(Index j = 0; j < cols(); ++j) - { - for(Index i = 0; i < rows(); ++i) - { - if(i == j) - { - if(!internal::isApprox(self.coeff(i, j), static_cast(1), prec)) - return false; - } - else - { - if(!internal::isMuchSmallerThan(self.coeff(i, j), static_cast(1), prec)) - return false; + * (not necessarily square), + * within the precision given by \a prec. + * + * Example: \include MatrixBase_isIdentity.cpp + * Output: \verbinclude MatrixBase_isIdentity.out + * + * \sa class CwiseNullaryOp, Identity(), Identity(Index,Index), setIdentity() + */ +template bool MatrixBase::isIdentity(const RealScalar &prec) const +{ + typename internal::nested_eval::type self(derived()); + for (Index j = 0; j < cols(); ++j) { + for (Index i = 0; i < rows(); ++i) { + if (i == j) { + if (!internal::isApprox(self.coeff(i, j), static_cast(1), prec)) return false; + } else { + if (!internal::isMuchSmallerThan(self.coeff(i, j), static_cast(1), prec)) return false; } } } @@ -738,129 +721,137 @@ bool MatrixBase::isIdentity namespace internal { -template=16)> -struct setIdentity_impl -{ - EIGEN_DEVICE_FUNC - static EIGEN_STRONG_INLINE Derived& run(Derived& m) + template= 16)> struct setIdentity_impl { - return m = Derived::Identity(m.rows(), m.cols()); - } -}; + EIGEN_DEVICE_FUNC + static EIGEN_STRONG_INLINE Derived &run(Derived &m) { return m = Derived::Identity(m.rows(), m.cols()); } + }; -template -struct setIdentity_impl -{ - EIGEN_DEVICE_FUNC - static EIGEN_STRONG_INLINE Derived& run(Derived& m) + template struct setIdentity_impl { - m.setZero(); - const Index size = numext::mini(m.rows(), m.cols()); - for(Index i = 0; i < size; ++i) m.coeffRef(i,i) = typename Derived::Scalar(1); - return m; - } -}; + EIGEN_DEVICE_FUNC + static EIGEN_STRONG_INLINE Derived &run(Derived &m) + { + m.setZero(); + const Index size = numext::mini(m.rows(), m.cols()); + for (Index i = 0; i < size; ++i) m.coeffRef(i, i) = typename Derived::Scalar(1); + return m; + } + }; -} // end namespace internal +}// end namespace internal /** Writes the identity expression (not necessarily square) into *this. - * - * Example: \include MatrixBase_setIdentity.cpp - * Output: \verbinclude MatrixBase_setIdentity.out - * - * \sa class CwiseNullaryOp, Identity(), Identity(Index,Index), isIdentity() - */ -template -EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Derived& MatrixBase::setIdentity() + * + * Example: \include MatrixBase_setIdentity.cpp + * Output: \verbinclude MatrixBase_setIdentity.out + * + * \sa class CwiseNullaryOp, Identity(), Identity(Index,Index), isIdentity() + */ +template EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Derived &MatrixBase::setIdentity() { return internal::setIdentity_impl::run(derived()); } /** \brief Resizes to the given size, and writes the identity expression (not necessarily square) into *this. - * - * \param rows the new number of rows - * \param cols the new number of columns - * - * Example: \include Matrix_setIdentity_int_int.cpp - * Output: \verbinclude Matrix_setIdentity_int_int.out - * - * \sa MatrixBase::setIdentity(), class CwiseNullaryOp, MatrixBase::Identity() - */ -template -EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Derived& MatrixBase::setIdentity(Index rows, Index cols) + * + * \param rows the new number of rows + * \param cols the new number of columns + * + * Example: \include Matrix_setIdentity_int_int.cpp + * Output: \verbinclude Matrix_setIdentity_int_int.out + * + * \sa MatrixBase::setIdentity(), class CwiseNullaryOp, MatrixBase::Identity() + */ +template +EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Derived &MatrixBase::setIdentity(Index rows, Index cols) { derived().resize(rows, cols); return setIdentity(); } /** \returns an expression of the i-th unit (basis) vector. - * - * \only_for_vectors - * - * \sa MatrixBase::Unit(Index), MatrixBase::UnitX(), MatrixBase::UnitY(), MatrixBase::UnitZ(), MatrixBase::UnitW() - */ + * + * \only_for_vectors + * + * \sa MatrixBase::Unit(Index), MatrixBase::UnitX(), MatrixBase::UnitY(), MatrixBase::UnitZ(), MatrixBase::UnitW() + */ template -EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const typename MatrixBase::BasisReturnType MatrixBase::Unit(Index newSize, Index i) +EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const typename MatrixBase::BasisReturnType + MatrixBase::Unit(Index newSize, Index i) { EIGEN_STATIC_ASSERT_VECTOR_ONLY(Derived) - return BasisReturnType(SquareMatrixType::Identity(newSize,newSize), i); + return BasisReturnType(SquareMatrixType::Identity(newSize, newSize), i); } /** \returns an expression of the i-th unit (basis) vector. - * - * \only_for_vectors - * - * This variant is for fixed-size vector only. - * - * \sa MatrixBase::Unit(Index,Index), MatrixBase::UnitX(), MatrixBase::UnitY(), MatrixBase::UnitZ(), MatrixBase::UnitW() - */ + * + * \only_for_vectors + * + * This variant is for fixed-size vector only. + * + * \sa MatrixBase::Unit(Index,Index), MatrixBase::UnitX(), MatrixBase::UnitY(), MatrixBase::UnitZ(), MatrixBase::UnitW() + */ template -EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const typename MatrixBase::BasisReturnType MatrixBase::Unit(Index i) +EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const typename MatrixBase::BasisReturnType MatrixBase::Unit( + Index i) { EIGEN_STATIC_ASSERT_VECTOR_ONLY(Derived) - return BasisReturnType(SquareMatrixType::Identity(),i); + return BasisReturnType(SquareMatrixType::Identity(), i); } /** \returns an expression of the X axis unit vector (1{,0}^*) - * - * \only_for_vectors - * - * \sa MatrixBase::Unit(Index,Index), MatrixBase::Unit(Index), MatrixBase::UnitY(), MatrixBase::UnitZ(), MatrixBase::UnitW() - */ + * + * \only_for_vectors + * + * \sa MatrixBase::Unit(Index,Index), MatrixBase::Unit(Index), MatrixBase::UnitY(), MatrixBase::UnitZ(), + * MatrixBase::UnitW() + */ template EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const typename MatrixBase::BasisReturnType MatrixBase::UnitX() -{ return Derived::Unit(0); } +{ + return Derived::Unit(0); +} /** \returns an expression of the Y axis unit vector (0,1{,0}^*) - * - * \only_for_vectors - * - * \sa MatrixBase::Unit(Index,Index), MatrixBase::Unit(Index), MatrixBase::UnitY(), MatrixBase::UnitZ(), MatrixBase::UnitW() - */ + * + * \only_for_vectors + * + * \sa MatrixBase::Unit(Index,Index), MatrixBase::Unit(Index), MatrixBase::UnitY(), MatrixBase::UnitZ(), + * MatrixBase::UnitW() + */ template EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const typename MatrixBase::BasisReturnType MatrixBase::UnitY() -{ return Derived::Unit(1); } +{ + return Derived::Unit(1); +} /** \returns an expression of the Z axis unit vector (0,0,1{,0}^*) - * - * \only_for_vectors - * - * \sa MatrixBase::Unit(Index,Index), MatrixBase::Unit(Index), MatrixBase::UnitY(), MatrixBase::UnitZ(), MatrixBase::UnitW() - */ + * + * \only_for_vectors + * + * \sa MatrixBase::Unit(Index,Index), MatrixBase::Unit(Index), MatrixBase::UnitY(), MatrixBase::UnitZ(), + * MatrixBase::UnitW() + */ template EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const typename MatrixBase::BasisReturnType MatrixBase::UnitZ() -{ return Derived::Unit(2); } +{ + return Derived::Unit(2); +} /** \returns an expression of the W axis unit vector (0,0,0,1) - * - * \only_for_vectors - * - * \sa MatrixBase::Unit(Index,Index), MatrixBase::Unit(Index), MatrixBase::UnitY(), MatrixBase::UnitZ(), MatrixBase::UnitW() - */ + * + * \only_for_vectors + * + * \sa MatrixBase::Unit(Index,Index), MatrixBase::Unit(Index), MatrixBase::UnitY(), MatrixBase::UnitZ(), + * MatrixBase::UnitW() + */ template EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const typename MatrixBase::BasisReturnType MatrixBase::UnitW() -{ return Derived::Unit(3); } +{ + return Derived::Unit(3); +} -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_CWISE_NULLARY_OP_H +#endif// EIGEN_CWISE_NULLARY_OP_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/CwiseTernaryOp.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/CwiseTernaryOp.h index 9f3576fe..63d00e1b 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/CwiseTernaryOp.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/CwiseTernaryOp.h @@ -15,85 +15,85 @@ namespace Eigen { namespace internal { -template -struct traits > { - // we must not inherit from traits since it has - // the potential to cause problems with MSVC - typedef typename remove_all::type Ancestor; - typedef typename traits::XprKind XprKind; - enum { - RowsAtCompileTime = traits::RowsAtCompileTime, - ColsAtCompileTime = traits::ColsAtCompileTime, - MaxRowsAtCompileTime = traits::MaxRowsAtCompileTime, - MaxColsAtCompileTime = traits::MaxColsAtCompileTime + template + struct traits> + { + // we must not inherit from traits since it has + // the potential to cause problems with MSVC + typedef typename remove_all::type Ancestor; + typedef typename traits::XprKind XprKind; + enum { + RowsAtCompileTime = traits::RowsAtCompileTime, + ColsAtCompileTime = traits::ColsAtCompileTime, + MaxRowsAtCompileTime = traits::MaxRowsAtCompileTime, + MaxColsAtCompileTime = traits::MaxColsAtCompileTime + }; + + // even though we require Arg1, Arg2, and Arg3 to have the same scalar type + // (see CwiseTernaryOp constructor), + // we still want to handle the case when the result type is different. + typedef typename result_of::type Scalar; + + typedef typename internal::traits::StorageKind StorageKind; + typedef typename internal::traits::StorageIndex StorageIndex; + + typedef typename Arg1::Nested Arg1Nested; + typedef typename Arg2::Nested Arg2Nested; + typedef typename Arg3::Nested Arg3Nested; + typedef typename remove_reference::type _Arg1Nested; + typedef typename remove_reference::type _Arg2Nested; + typedef typename remove_reference::type _Arg3Nested; + enum { Flags = _Arg1Nested::Flags & RowMajorBit }; }; +}// end namespace internal - // even though we require Arg1, Arg2, and Arg3 to have the same scalar type - // (see CwiseTernaryOp constructor), - // we still want to handle the case when the result type is different. - typedef typename result_of::type Scalar; - - typedef typename internal::traits::StorageKind StorageKind; - typedef typename internal::traits::StorageIndex StorageIndex; - - typedef typename Arg1::Nested Arg1Nested; - typedef typename Arg2::Nested Arg2Nested; - typedef typename Arg3::Nested Arg3Nested; - typedef typename remove_reference::type _Arg1Nested; - typedef typename remove_reference::type _Arg2Nested; - typedef typename remove_reference::type _Arg3Nested; - enum { Flags = _Arg1Nested::Flags & RowMajorBit }; -}; -} // end namespace internal - -template +template class CwiseTernaryOpImpl; /** \class CwiseTernaryOp - * \ingroup Core_Module - * - * \brief Generic expression where a coefficient-wise ternary operator is + * \ingroup Core_Module + * + * \brief Generic expression where a coefficient-wise ternary operator is * applied to two expressions - * - * \tparam TernaryOp template functor implementing the operator - * \tparam Arg1Type the type of the first argument - * \tparam Arg2Type the type of the second argument - * \tparam Arg3Type the type of the third argument - * - * This class represents an expression where a coefficient-wise ternary + * + * \tparam TernaryOp template functor implementing the operator + * \tparam Arg1Type the type of the first argument + * \tparam Arg2Type the type of the second argument + * \tparam Arg3Type the type of the third argument + * + * This class represents an expression where a coefficient-wise ternary * operator is applied to three expressions. - * It is the return type of ternary operators, by which we mean only those + * It is the return type of ternary operators, by which we mean only those * ternary operators where - * all three arguments are Eigen expressions. - * For example, the return type of betainc(matrix1, matrix2, matrix3) is a + * all three arguments are Eigen expressions. + * For example, the return type of betainc(matrix1, matrix2, matrix3) is a * CwiseTernaryOp. - * - * Most of the time, this is the only way that it is used, so you typically + * + * Most of the time, this is the only way that it is used, so you typically * don't have to name - * CwiseTernaryOp types explicitly. - * - * \sa MatrixBase::ternaryExpr(const MatrixBase &, const + * CwiseTernaryOp types explicitly. + * + * \sa MatrixBase::ternaryExpr(const MatrixBase &, const * MatrixBase &, const CustomTernaryOp &) const, class CwiseBinaryOp, * class CwiseUnaryOp, class CwiseNullaryOp - */ -template -class CwiseTernaryOp : public CwiseTernaryOpImpl< - TernaryOp, Arg1Type, Arg2Type, Arg3Type, - typename internal::traits::StorageKind>, - internal::no_assignment_operator + */ +template +class CwiseTernaryOp + : public CwiseTernaryOpImpl::StorageKind> + , internal::no_assignment_operator { - public: +public: typedef typename internal::remove_all::type Arg1; typedef typename internal::remove_all::type Arg2; typedef typename internal::remove_all::type Arg3; - typedef typename CwiseTernaryOpImpl< - TernaryOp, Arg1Type, Arg2Type, Arg3Type, - typename internal::traits::StorageKind>::Base Base; + typedef typename CwiseTernaryOpImpl::StorageKind>::Base Base; EIGEN_GENERIC_PUBLIC_INTERFACE(CwiseTernaryOp) typedef typename internal::ref_selector::type Arg1Nested; @@ -104,58 +104,49 @@ class CwiseTernaryOp : public CwiseTernaryOpImpl< typedef typename internal::remove_reference::type _Arg3Nested; EIGEN_DEVICE_FUNC - EIGEN_STRONG_INLINE CwiseTernaryOp(const Arg1& a1, const Arg2& a2, - const Arg3& a3, - const TernaryOp& func = TernaryOp()) - : m_arg1(a1), m_arg2(a2), m_arg3(a3), m_functor(func) { + EIGEN_STRONG_INLINE + CwiseTernaryOp(const Arg1 &a1, const Arg2 &a2, const Arg3 &a3, const TernaryOp &func = TernaryOp()) + : m_arg1(a1), m_arg2(a2), m_arg3(a3), m_functor(func) + { // require the sizes to match EIGEN_STATIC_ASSERT_SAME_MATRIX_SIZE(Arg1, Arg2) EIGEN_STATIC_ASSERT_SAME_MATRIX_SIZE(Arg1, Arg3) // The index types should match - EIGEN_STATIC_ASSERT((internal::is_same< - typename internal::traits::StorageKind, - typename internal::traits::StorageKind>::value), - STORAGE_KIND_MUST_MATCH) - EIGEN_STATIC_ASSERT((internal::is_same< - typename internal::traits::StorageKind, - typename internal::traits::StorageKind>::value), - STORAGE_KIND_MUST_MATCH) - - eigen_assert(a1.rows() == a2.rows() && a1.cols() == a2.cols() && - a1.rows() == a3.rows() && a1.cols() == a3.cols()); + EIGEN_STATIC_ASSERT((internal::is_same::StorageKind, + typename internal::traits::StorageKind>::value), + STORAGE_KIND_MUST_MATCH) + EIGEN_STATIC_ASSERT((internal::is_same::StorageKind, + typename internal::traits::StorageKind>::value), + STORAGE_KIND_MUST_MATCH) + + eigen_assert(a1.rows() == a2.rows() && a1.cols() == a2.cols() && a1.rows() == a3.rows() && a1.cols() == a3.cols()); } EIGEN_DEVICE_FUNC - EIGEN_STRONG_INLINE Index rows() const { + EIGEN_STRONG_INLINE Index rows() const + { // return the fixed size type if available to enable compile time // optimizations - if (internal::traits::type>:: - RowsAtCompileTime == Dynamic && - internal::traits::type>:: - RowsAtCompileTime == Dynamic) + if (internal::traits::type>::RowsAtCompileTime == Dynamic + && internal::traits::type>::RowsAtCompileTime == Dynamic) return m_arg3.rows(); - else if (internal::traits::type>:: - RowsAtCompileTime == Dynamic && - internal::traits::type>:: - RowsAtCompileTime == Dynamic) + else if (internal::traits::type>::RowsAtCompileTime == Dynamic + && internal::traits::type>::RowsAtCompileTime == Dynamic) return m_arg2.rows(); else return m_arg1.rows(); } EIGEN_DEVICE_FUNC - EIGEN_STRONG_INLINE Index cols() const { + EIGEN_STRONG_INLINE Index cols() const + { // return the fixed size type if available to enable compile time // optimizations - if (internal::traits::type>:: - ColsAtCompileTime == Dynamic && - internal::traits::type>:: - ColsAtCompileTime == Dynamic) + if (internal::traits::type>::ColsAtCompileTime == Dynamic + && internal::traits::type>::ColsAtCompileTime == Dynamic) return m_arg3.cols(); - else if (internal::traits::type>:: - ColsAtCompileTime == Dynamic && - internal::traits::type>:: - ColsAtCompileTime == Dynamic) + else if (internal::traits::type>::ColsAtCompileTime == Dynamic + && internal::traits::type>::ColsAtCompileTime == Dynamic) return m_arg2.cols(); else return m_arg1.cols(); @@ -163,18 +154,18 @@ class CwiseTernaryOp : public CwiseTernaryOpImpl< /** \returns the first argument nested expression */ EIGEN_DEVICE_FUNC - const _Arg1Nested& arg1() const { return m_arg1; } + const _Arg1Nested &arg1() const { return m_arg1; } /** \returns the first argument nested expression */ EIGEN_DEVICE_FUNC - const _Arg2Nested& arg2() const { return m_arg2; } + const _Arg2Nested &arg2() const { return m_arg2; } /** \returns the third argument nested expression */ EIGEN_DEVICE_FUNC - const _Arg3Nested& arg3() const { return m_arg3; } + const _Arg3Nested &arg3() const { return m_arg3; } /** \returns the functor representing the ternary operation */ EIGEN_DEVICE_FUNC - const TernaryOp& functor() const { return m_functor; } + const TernaryOp &functor() const { return m_functor; } - protected: +protected: Arg1Nested m_arg1; Arg2Nested m_arg2; Arg3Nested m_arg3; @@ -182,16 +173,13 @@ class CwiseTernaryOp : public CwiseTernaryOpImpl< }; // Generic API dispatcher -template -class CwiseTernaryOpImpl - : public internal::generic_xpr_base< - CwiseTernaryOp >::type { - public: - typedef typename internal::generic_xpr_base< - CwiseTernaryOp >::type Base; +template +class CwiseTernaryOpImpl : public internal::generic_xpr_base>::type +{ +public: + typedef typename internal::generic_xpr_base>::type Base; }; -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_CWISE_TERNARY_OP_H +#endif// EIGEN_CWISE_TERNARY_OP_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/CwiseUnaryOp.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/CwiseUnaryOp.h index 1d2dd19f..04003009 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/CwiseUnaryOp.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/CwiseUnaryOp.h @@ -11,93 +11,86 @@ #ifndef EIGEN_CWISE_UNARY_OP_H #define EIGEN_CWISE_UNARY_OP_H -namespace Eigen { +namespace Eigen { namespace internal { -template -struct traits > - : traits -{ - typedef typename result_of< - UnaryOp(const typename XprType::Scalar&) - >::type Scalar; - typedef typename XprType::Nested XprTypeNested; - typedef typename remove_reference::type _XprTypeNested; - enum { - Flags = _XprTypeNested::Flags & RowMajorBit + template struct traits> : traits + { + typedef typename result_of::type Scalar; + typedef typename XprType::Nested XprTypeNested; + typedef typename remove_reference::type _XprTypeNested; + enum { Flags = _XprTypeNested::Flags & RowMajorBit }; }; -}; -} +}// namespace internal -template -class CwiseUnaryOpImpl; +template class CwiseUnaryOpImpl; /** \class CwiseUnaryOp - * \ingroup Core_Module - * - * \brief Generic expression where a coefficient-wise unary operator is applied to an expression - * - * \tparam UnaryOp template functor implementing the operator - * \tparam XprType the type of the expression to which we are applying the unary operator - * - * This class represents an expression where a unary operator is applied to an expression. - * It is the return type of all operations taking exactly 1 input expression, regardless of the - * presence of other inputs such as scalars. For example, the operator* in the expression 3*matrix - * is considered unary, because only the right-hand side is an expression, and its - * return type is a specialization of CwiseUnaryOp. - * - * Most of the time, this is the only way that it is used, so you typically don't have to name - * CwiseUnaryOp types explicitly. - * - * \sa MatrixBase::unaryExpr(const CustomUnaryOp &) const, class CwiseBinaryOp, class CwiseNullaryOp - */ + * \ingroup Core_Module + * + * \brief Generic expression where a coefficient-wise unary operator is applied to an expression + * + * \tparam UnaryOp template functor implementing the operator + * \tparam XprType the type of the expression to which we are applying the unary operator + * + * This class represents an expression where a unary operator is applied to an expression. + * It is the return type of all operations taking exactly 1 input expression, regardless of the + * presence of other inputs such as scalars. For example, the operator* in the expression 3*matrix + * is considered unary, because only the right-hand side is an expression, and its + * return type is a specialization of CwiseUnaryOp. + * + * Most of the time, this is the only way that it is used, so you typically don't have to name + * CwiseUnaryOp types explicitly. + * + * \sa MatrixBase::unaryExpr(const CustomUnaryOp &) const, class CwiseBinaryOp, class CwiseNullaryOp + */ template -class CwiseUnaryOp : public CwiseUnaryOpImpl::StorageKind>, internal::no_assignment_operator +class CwiseUnaryOp + : public CwiseUnaryOpImpl::StorageKind> + , internal::no_assignment_operator { - public: - - typedef typename CwiseUnaryOpImpl::StorageKind>::Base Base; - EIGEN_GENERIC_PUBLIC_INTERFACE(CwiseUnaryOp) - typedef typename internal::ref_selector::type XprTypeNested; - typedef typename internal::remove_all::type NestedExpression; +public: + typedef typename CwiseUnaryOpImpl::StorageKind>::Base Base; + EIGEN_GENERIC_PUBLIC_INTERFACE(CwiseUnaryOp) + typedef typename internal::ref_selector::type XprTypeNested; + typedef typename internal::remove_all::type NestedExpression; - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE - explicit CwiseUnaryOp(const XprType& xpr, const UnaryOp& func = UnaryOp()) - : m_xpr(xpr), m_functor(func) {} + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE explicit CwiseUnaryOp(const XprType &xpr, const UnaryOp &func = UnaryOp()) + : m_xpr(xpr), m_functor(func) + {} - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE - Index rows() const { return m_xpr.rows(); } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE - Index cols() const { return m_xpr.cols(); } + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Index rows() const { return m_xpr.rows(); } + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Index cols() const { return m_xpr.cols(); } - /** \returns the functor representing the unary operation */ - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE - const UnaryOp& functor() const { return m_functor; } + /** \returns the functor representing the unary operation */ + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const UnaryOp &functor() const { return m_functor; } - /** \returns the nested expression */ - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE - const typename internal::remove_all::type& - nestedExpression() const { return m_xpr; } + /** \returns the nested expression */ + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const typename internal::remove_all::type & + nestedExpression() const + { + return m_xpr; + } - /** \returns the nested expression */ - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE - typename internal::remove_all::type& - nestedExpression() { return m_xpr; } + /** \returns the nested expression */ + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE typename internal::remove_all::type &nestedExpression() + { + return m_xpr; + } - protected: - XprTypeNested m_xpr; - const UnaryOp m_functor; +protected: + XprTypeNested m_xpr; + const UnaryOp m_functor; }; // Generic API dispatcher template -class CwiseUnaryOpImpl - : public internal::generic_xpr_base >::type +class CwiseUnaryOpImpl : public internal::generic_xpr_base>::type { public: - typedef typename internal::generic_xpr_base >::type Base; + typedef typename internal::generic_xpr_base>::type Base; }; -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_CWISE_UNARY_OP_H +#endif// EIGEN_CWISE_UNARY_OP_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/CwiseUnaryView.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/CwiseUnaryView.h index 27103305..ce602f68 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/CwiseUnaryView.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/CwiseUnaryView.h @@ -13,116 +13,115 @@ namespace Eigen { namespace internal { -template -struct traits > - : traits -{ - typedef typename result_of< - ViewOp(const typename traits::Scalar&) - >::type Scalar; - typedef typename MatrixType::Nested MatrixTypeNested; - typedef typename remove_all::type _MatrixTypeNested; - enum { - FlagsLvalueBit = is_lvalue::value ? LvalueBit : 0, - Flags = traits<_MatrixTypeNested>::Flags & (RowMajorBit | FlagsLvalueBit | DirectAccessBit), // FIXME DirectAccessBit should not be handled by expressions - MatrixTypeInnerStride = inner_stride_at_compile_time::ret, - // need to cast the sizeof's from size_t to int explicitly, otherwise: - // "error: no integral type can represent all of the enumerator values - InnerStrideAtCompileTime = MatrixTypeInnerStride == Dynamic - ? int(Dynamic) - : int(MatrixTypeInnerStride) * int(sizeof(typename traits::Scalar) / sizeof(Scalar)), - OuterStrideAtCompileTime = outer_stride_at_compile_time::ret == Dynamic - ? int(Dynamic) - : outer_stride_at_compile_time::ret * int(sizeof(typename traits::Scalar) / sizeof(Scalar)) + template struct traits> : traits + { + typedef typename result_of::Scalar &)>::type Scalar; + typedef typename MatrixType::Nested MatrixTypeNested; + typedef typename remove_all::type _MatrixTypeNested; + enum { + FlagsLvalueBit = is_lvalue::value ? LvalueBit : 0, + Flags = + traits<_MatrixTypeNested>::Flags + & (RowMajorBit | FlagsLvalueBit | DirectAccessBit),// FIXME DirectAccessBit should not be handled by expressions + MatrixTypeInnerStride = inner_stride_at_compile_time::ret, + // need to cast the sizeof's from size_t to int explicitly, otherwise: + // "error: no integral type can represent all of the enumerator values + InnerStrideAtCompileTime = + MatrixTypeInnerStride == Dynamic + ? int(Dynamic) + : int(MatrixTypeInnerStride) * int(sizeof(typename traits::Scalar) / sizeof(Scalar)), + OuterStrideAtCompileTime = outer_stride_at_compile_time::ret == Dynamic + ? int(Dynamic) + : outer_stride_at_compile_time::ret + * int(sizeof(typename traits::Scalar) / sizeof(Scalar)) + }; }; -}; -} +}// namespace internal -template -class CwiseUnaryViewImpl; +template class CwiseUnaryViewImpl; /** \class CwiseUnaryView - * \ingroup Core_Module - * - * \brief Generic lvalue expression of a coefficient-wise unary operator of a matrix or a vector - * - * \tparam ViewOp template functor implementing the view - * \tparam MatrixType the type of the matrix we are applying the unary operator - * - * This class represents a lvalue expression of a generic unary view operator of a matrix or a vector. - * It is the return type of real() and imag(), and most of the time this is the only way it is used. - * - * \sa MatrixBase::unaryViewExpr(const CustomUnaryOp &) const, class CwiseUnaryOp - */ + * \ingroup Core_Module + * + * \brief Generic lvalue expression of a coefficient-wise unary operator of a matrix or a vector + * + * \tparam ViewOp template functor implementing the view + * \tparam MatrixType the type of the matrix we are applying the unary operator + * + * This class represents a lvalue expression of a generic unary view operator of a matrix or a vector. + * It is the return type of real() and imag(), and most of the time this is the only way it is used. + * + * \sa MatrixBase::unaryViewExpr(const CustomUnaryOp &) const, class CwiseUnaryOp + */ template class CwiseUnaryView : public CwiseUnaryViewImpl::StorageKind> { - public: - - typedef typename CwiseUnaryViewImpl::StorageKind>::Base Base; - EIGEN_GENERIC_PUBLIC_INTERFACE(CwiseUnaryView) - typedef typename internal::ref_selector::non_const_type MatrixTypeNested; - typedef typename internal::remove_all::type NestedExpression; +public: + typedef + typename CwiseUnaryViewImpl::StorageKind>::Base Base; + EIGEN_GENERIC_PUBLIC_INTERFACE(CwiseUnaryView) + typedef typename internal::ref_selector::non_const_type MatrixTypeNested; + typedef typename internal::remove_all::type NestedExpression; - explicit inline CwiseUnaryView(MatrixType& mat, const ViewOp& func = ViewOp()) - : m_matrix(mat), m_functor(func) {} + explicit inline CwiseUnaryView(MatrixType &mat, const ViewOp &func = ViewOp()) : m_matrix(mat), m_functor(func) {} - EIGEN_INHERIT_ASSIGNMENT_OPERATORS(CwiseUnaryView) + EIGEN_INHERIT_ASSIGNMENT_OPERATORS(CwiseUnaryView) - EIGEN_STRONG_INLINE Index rows() const { return m_matrix.rows(); } - EIGEN_STRONG_INLINE Index cols() const { return m_matrix.cols(); } + EIGEN_STRONG_INLINE Index rows() const { return m_matrix.rows(); } + EIGEN_STRONG_INLINE Index cols() const { return m_matrix.cols(); } - /** \returns the functor representing unary operation */ - const ViewOp& functor() const { return m_functor; } + /** \returns the functor representing unary operation */ + const ViewOp &functor() const { return m_functor; } - /** \returns the nested expression */ - const typename internal::remove_all::type& - nestedExpression() const { return m_matrix; } + /** \returns the nested expression */ + const typename internal::remove_all::type &nestedExpression() const { return m_matrix; } - /** \returns the nested expression */ - typename internal::remove_reference::type& - nestedExpression() { return m_matrix.const_cast_derived(); } + /** \returns the nested expression */ + typename internal::remove_reference::type &nestedExpression() + { + return m_matrix.const_cast_derived(); + } - protected: - MatrixTypeNested m_matrix; - ViewOp m_functor; +protected: + MatrixTypeNested m_matrix; + ViewOp m_functor; }; // Generic API dispatcher template -class CwiseUnaryViewImpl - : public internal::generic_xpr_base >::type +class CwiseUnaryViewImpl : public internal::generic_xpr_base>::type { public: - typedef typename internal::generic_xpr_base >::type Base; + typedef typename internal::generic_xpr_base>::type Base; }; template -class CwiseUnaryViewImpl - : public internal::dense_xpr_base< CwiseUnaryView >::type +class CwiseUnaryViewImpl + : public internal::dense_xpr_base>::type { - public: - - typedef CwiseUnaryView Derived; - typedef typename internal::dense_xpr_base< CwiseUnaryView >::type Base; - - EIGEN_DENSE_PUBLIC_INTERFACE(Derived) - EIGEN_INHERIT_ASSIGNMENT_OPERATORS(CwiseUnaryViewImpl) - - EIGEN_DEVICE_FUNC inline Scalar* data() { return &(this->coeffRef(0)); } - EIGEN_DEVICE_FUNC inline const Scalar* data() const { return &(this->coeff(0)); } - - EIGEN_DEVICE_FUNC inline Index innerStride() const - { - return derived().nestedExpression().innerStride() * sizeof(typename internal::traits::Scalar) / sizeof(Scalar); - } - - EIGEN_DEVICE_FUNC inline Index outerStride() const - { - return derived().nestedExpression().outerStride() * sizeof(typename internal::traits::Scalar) / sizeof(Scalar); - } +public: + typedef CwiseUnaryView Derived; + typedef typename internal::dense_xpr_base>::type Base; + + EIGEN_DENSE_PUBLIC_INTERFACE(Derived) + EIGEN_INHERIT_ASSIGNMENT_OPERATORS(CwiseUnaryViewImpl) + + EIGEN_DEVICE_FUNC inline Scalar *data() { return &(this->coeffRef(0)); } + EIGEN_DEVICE_FUNC inline const Scalar *data() const { return &(this->coeff(0)); } + + EIGEN_DEVICE_FUNC inline Index innerStride() const + { + return derived().nestedExpression().innerStride() * sizeof(typename internal::traits::Scalar) + / sizeof(Scalar); + } + + EIGEN_DEVICE_FUNC inline Index outerStride() const + { + return derived().nestedExpression().outerStride() * sizeof(typename internal::traits::Scalar) + / sizeof(Scalar); + } }; -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_CWISE_UNARY_VIEW_H +#endif// EIGEN_CWISE_UNARY_VIEW_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/DenseBase.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/DenseBase.h index 90066ae7..8cee86cb 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/DenseBase.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/DenseBase.h @@ -14,598 +14,561 @@ namespace Eigen { namespace internal { - -// The index type defined by EIGEN_DEFAULT_DENSE_INDEX_TYPE must be a signed type. -// This dummy function simply aims at checking that at compile time. -static inline void check_DenseIndex_is_signed() { - EIGEN_STATIC_ASSERT(NumTraits::IsSigned,THE_INDEX_TYPE_MUST_BE_A_SIGNED_TYPE); -} - -} // end namespace internal - + + // The index type defined by EIGEN_DEFAULT_DENSE_INDEX_TYPE must be a signed type. + // This dummy function simply aims at checking that at compile time. + static inline void check_DenseIndex_is_signed() + { + EIGEN_STATIC_ASSERT(NumTraits::IsSigned, THE_INDEX_TYPE_MUST_BE_A_SIGNED_TYPE); + } + +}// end namespace internal + /** \class DenseBase - * \ingroup Core_Module - * - * \brief Base class for all dense matrices, vectors, and arrays - * - * This class is the base that is inherited by all dense objects (matrix, vector, arrays, - * and related expression types). The common Eigen API for dense objects is contained in this class. - * - * \tparam Derived is the derived type, e.g., a matrix type or an expression. - * - * This class can be extended with the help of the plugin mechanism described on the page - * \ref TopicCustomizing_Plugins by defining the preprocessor symbol \c EIGEN_DENSEBASE_PLUGIN. - * - * \sa \blank \ref TopicClassHierarchy - */ -template class DenseBase + * \ingroup Core_Module + * + * \brief Base class for all dense matrices, vectors, and arrays + * + * This class is the base that is inherited by all dense objects (matrix, vector, arrays, + * and related expression types). The common Eigen API for dense objects is contained in this class. + * + * \tparam Derived is the derived type, e.g., a matrix type or an expression. + * + * This class can be extended with the help of the plugin mechanism described on the page + * \ref TopicCustomizing_Plugins by defining the preprocessor symbol \c EIGEN_DENSEBASE_PLUGIN. + * + * \sa \blank \ref TopicClassHierarchy + */ +template +class DenseBase #ifndef EIGEN_PARSED_BY_DOXYGEN : public DenseCoeffsBase #else - : public DenseCoeffsBase -#endif // not EIGEN_PARSED_BY_DOXYGEN + : public DenseCoeffsBase +#endif// not EIGEN_PARSED_BY_DOXYGEN { - public: +public: + /** Inner iterator type to iterate over the coefficients of a row or column. + * \sa class InnerIterator + */ + typedef Eigen::InnerIterator InnerIterator; + + typedef typename internal::traits::StorageKind StorageKind; + + /** + * \brief The type used to store indices + * \details This typedef is relevant for types that store multiple indices such as + * PermutationMatrix or Transpositions, otherwise it defaults to Eigen::Index + * \sa \blank \ref TopicPreprocessorDirectives, Eigen::Index, SparseMatrixBase. + */ + typedef typename internal::traits::StorageIndex StorageIndex; + + /** The numeric type of the expression' coefficients, e.g. float, double, int or std::complex, etc. */ + typedef typename internal::traits::Scalar Scalar; + + /** The numeric type of the expression' coefficients, e.g. float, double, int or std::complex, etc. + * + * It is an alias for the Scalar type */ + typedef Scalar value_type; + + typedef typename NumTraits::Real RealScalar; + typedef DenseCoeffsBase Base; + + using Base::derived; + using Base::const_cast_derived; + using Base::rows; + using Base::cols; + using Base::size; + using Base::rowIndexByOuterInner; + using Base::colIndexByOuterInner; + using Base::coeff; + using Base::coeffByOuterInner; + using Base::operator(); + using Base::operator[]; + using Base::x; + using Base::y; + using Base::z; + using Base::w; + using Base::stride; + using Base::innerStride; + using Base::outerStride; + using Base::rowStride; + using Base::colStride; + typedef typename Base::CoeffReturnType CoeffReturnType; + + enum { + + RowsAtCompileTime = internal::traits::RowsAtCompileTime, + /**< The number of rows at compile-time. This is just a copy of the value provided + * by the \a Derived type. If a value is not known at compile-time, + * it is set to the \a Dynamic constant. + * \sa MatrixBase::rows(), MatrixBase::cols(), ColsAtCompileTime, SizeAtCompileTime */ + + ColsAtCompileTime = internal::traits::ColsAtCompileTime, + /**< The number of columns at compile-time. This is just a copy of the value provided + * by the \a Derived type. If a value is not known at compile-time, + * it is set to the \a Dynamic constant. + * \sa MatrixBase::rows(), MatrixBase::cols(), RowsAtCompileTime, SizeAtCompileTime */ + + + SizeAtCompileTime = (internal::size_at_compile_time::RowsAtCompileTime, + internal::traits::ColsAtCompileTime>::ret), + /**< This is equal to the number of coefficients, i.e. the number of + * rows times the number of columns, or to \a Dynamic if this is not + * known at compile-time. \sa RowsAtCompileTime, ColsAtCompileTime */ + + MaxRowsAtCompileTime = internal::traits::MaxRowsAtCompileTime, + /**< This value is equal to the maximum possible number of rows that this expression + * might have. If this expression might have an arbitrarily high number of rows, + * this value is set to \a Dynamic. + * + * This value is useful to know when evaluating an expression, in order to determine + * whether it is possible to avoid doing a dynamic memory allocation. + * + * \sa RowsAtCompileTime, MaxColsAtCompileTime, MaxSizeAtCompileTime + */ + + MaxColsAtCompileTime = internal::traits::MaxColsAtCompileTime, + /**< This value is equal to the maximum possible number of columns that this expression + * might have. If this expression might have an arbitrarily high number of columns, + * this value is set to \a Dynamic. + * + * This value is useful to know when evaluating an expression, in order to determine + * whether it is possible to avoid doing a dynamic memory allocation. + * + * \sa ColsAtCompileTime, MaxRowsAtCompileTime, MaxSizeAtCompileTime + */ - /** Inner iterator type to iterate over the coefficients of a row or column. - * \sa class InnerIterator - */ - typedef Eigen::InnerIterator InnerIterator; + MaxSizeAtCompileTime = (internal::size_at_compile_time::MaxRowsAtCompileTime, + internal::traits::MaxColsAtCompileTime>::ret), + /**< This value is equal to the maximum possible number of coefficients that this expression + * might have. If this expression might have an arbitrarily high number of coefficients, + * this value is set to \a Dynamic. + * + * This value is useful to know when evaluating an expression, in order to determine + * whether it is possible to avoid doing a dynamic memory allocation. + * + * \sa SizeAtCompileTime, MaxRowsAtCompileTime, MaxColsAtCompileTime + */ - typedef typename internal::traits::StorageKind StorageKind; + IsVectorAtCompileTime = + internal::traits::MaxRowsAtCompileTime == 1 || internal::traits::MaxColsAtCompileTime == 1, + /**< This is set to true if either the number of rows or the number of + * columns is known at compile-time to be equal to 1. Indeed, in that case, + * we are dealing with a column-vector (if there is only one column) or with + * a row-vector (if there is only one row). */ - /** - * \brief The type used to store indices - * \details This typedef is relevant for types that store multiple indices such as - * PermutationMatrix or Transpositions, otherwise it defaults to Eigen::Index - * \sa \blank \ref TopicPreprocessorDirectives, Eigen::Index, SparseMatrixBase. + Flags = internal::traits::Flags, + /**< This stores expression \ref flags flags which may or may not be inherited by new expressions + * constructed from this one. See the \ref flags "list of flags". */ - typedef typename internal::traits::StorageIndex StorageIndex; - - /** The numeric type of the expression' coefficients, e.g. float, double, int or std::complex, etc. */ - typedef typename internal::traits::Scalar Scalar; - - /** The numeric type of the expression' coefficients, e.g. float, double, int or std::complex, etc. - * - * It is an alias for the Scalar type */ - typedef Scalar value_type; - - typedef typename NumTraits::Real RealScalar; - typedef DenseCoeffsBase Base; - - using Base::derived; - using Base::const_cast_derived; - using Base::rows; - using Base::cols; - using Base::size; - using Base::rowIndexByOuterInner; - using Base::colIndexByOuterInner; - using Base::coeff; - using Base::coeffByOuterInner; - using Base::operator(); - using Base::operator[]; - using Base::x; - using Base::y; - using Base::z; - using Base::w; - using Base::stride; - using Base::innerStride; - using Base::outerStride; - using Base::rowStride; - using Base::colStride; - typedef typename Base::CoeffReturnType CoeffReturnType; - - enum { - - RowsAtCompileTime = internal::traits::RowsAtCompileTime, - /**< The number of rows at compile-time. This is just a copy of the value provided - * by the \a Derived type. If a value is not known at compile-time, - * it is set to the \a Dynamic constant. - * \sa MatrixBase::rows(), MatrixBase::cols(), ColsAtCompileTime, SizeAtCompileTime */ - - ColsAtCompileTime = internal::traits::ColsAtCompileTime, - /**< The number of columns at compile-time. This is just a copy of the value provided - * by the \a Derived type. If a value is not known at compile-time, - * it is set to the \a Dynamic constant. - * \sa MatrixBase::rows(), MatrixBase::cols(), RowsAtCompileTime, SizeAtCompileTime */ - - - SizeAtCompileTime = (internal::size_at_compile_time::RowsAtCompileTime, - internal::traits::ColsAtCompileTime>::ret), - /**< This is equal to the number of coefficients, i.e. the number of - * rows times the number of columns, or to \a Dynamic if this is not - * known at compile-time. \sa RowsAtCompileTime, ColsAtCompileTime */ - - MaxRowsAtCompileTime = internal::traits::MaxRowsAtCompileTime, - /**< This value is equal to the maximum possible number of rows that this expression - * might have. If this expression might have an arbitrarily high number of rows, - * this value is set to \a Dynamic. - * - * This value is useful to know when evaluating an expression, in order to determine - * whether it is possible to avoid doing a dynamic memory allocation. - * - * \sa RowsAtCompileTime, MaxColsAtCompileTime, MaxSizeAtCompileTime - */ - - MaxColsAtCompileTime = internal::traits::MaxColsAtCompileTime, - /**< This value is equal to the maximum possible number of columns that this expression - * might have. If this expression might have an arbitrarily high number of columns, - * this value is set to \a Dynamic. - * - * This value is useful to know when evaluating an expression, in order to determine - * whether it is possible to avoid doing a dynamic memory allocation. - * - * \sa ColsAtCompileTime, MaxRowsAtCompileTime, MaxSizeAtCompileTime - */ - - MaxSizeAtCompileTime = (internal::size_at_compile_time::MaxRowsAtCompileTime, - internal::traits::MaxColsAtCompileTime>::ret), - /**< This value is equal to the maximum possible number of coefficients that this expression - * might have. If this expression might have an arbitrarily high number of coefficients, - * this value is set to \a Dynamic. - * - * This value is useful to know when evaluating an expression, in order to determine - * whether it is possible to avoid doing a dynamic memory allocation. - * - * \sa SizeAtCompileTime, MaxRowsAtCompileTime, MaxColsAtCompileTime - */ - - IsVectorAtCompileTime = internal::traits::MaxRowsAtCompileTime == 1 - || internal::traits::MaxColsAtCompileTime == 1, - /**< This is set to true if either the number of rows or the number of - * columns is known at compile-time to be equal to 1. Indeed, in that case, - * we are dealing with a column-vector (if there is only one column) or with - * a row-vector (if there is only one row). */ - - Flags = internal::traits::Flags, - /**< This stores expression \ref flags flags which may or may not be inherited by new expressions - * constructed from this one. See the \ref flags "list of flags". - */ - - IsRowMajor = int(Flags) & RowMajorBit, /**< True if this expression has row-major storage order. */ - - InnerSizeAtCompileTime = int(IsVectorAtCompileTime) ? int(SizeAtCompileTime) - : int(IsRowMajor) ? int(ColsAtCompileTime) : int(RowsAtCompileTime), - - InnerStrideAtCompileTime = internal::inner_stride_at_compile_time::ret, - OuterStrideAtCompileTime = internal::outer_stride_at_compile_time::ret - }; - - typedef typename internal::find_best_packet::type PacketScalar; - - enum { IsPlainObjectBase = 0 }; - - /** The plain matrix type corresponding to this expression. - * \sa PlainObject */ - typedef Matrix::Scalar, - internal::traits::RowsAtCompileTime, - internal::traits::ColsAtCompileTime, - AutoAlign | (internal::traits::Flags&RowMajorBit ? RowMajor : ColMajor), - internal::traits::MaxRowsAtCompileTime, - internal::traits::MaxColsAtCompileTime - > PlainMatrix; - - /** The plain array type corresponding to this expression. - * \sa PlainObject */ - typedef Array::Scalar, - internal::traits::RowsAtCompileTime, - internal::traits::ColsAtCompileTime, - AutoAlign | (internal::traits::Flags&RowMajorBit ? RowMajor : ColMajor), - internal::traits::MaxRowsAtCompileTime, - internal::traits::MaxColsAtCompileTime - > PlainArray; - - /** \brief The plain matrix or array type corresponding to this expression. - * - * This is not necessarily exactly the return type of eval(). In the case of plain matrices, - * the return type of eval() is a const reference to a matrix, not a matrix! It is however guaranteed - * that the return type of eval() is either PlainObject or const PlainObject&. - */ - typedef typename internal::conditional::XprKind,MatrixXpr >::value, - PlainMatrix, PlainArray>::type PlainObject; - - /** \returns the number of nonzero coefficients which is in practice the number - * of stored coefficients. */ - EIGEN_DEVICE_FUNC - inline Index nonZeros() const { return size(); } - - /** \returns the outer size. - * - * \note For a vector, this returns just 1. For a matrix (non-vector), this is the major dimension - * with respect to the \ref TopicStorageOrders "storage order", i.e., the number of columns for a - * column-major matrix, and the number of rows for a row-major matrix. */ - EIGEN_DEVICE_FUNC - Index outerSize() const - { - return IsVectorAtCompileTime ? 1 - : int(IsRowMajor) ? this->rows() : this->cols(); - } - - /** \returns the inner size. - * - * \note For a vector, this is just the size. For a matrix (non-vector), this is the minor dimension - * with respect to the \ref TopicStorageOrders "storage order", i.e., the number of rows for a - * column-major matrix, and the number of columns for a row-major matrix. */ - EIGEN_DEVICE_FUNC - Index innerSize() const - { - return IsVectorAtCompileTime ? this->size() - : int(IsRowMajor) ? this->cols() : this->rows(); - } - - /** Only plain matrices/arrays, not expressions, may be resized; therefore the only useful resize methods are - * Matrix::resize() and Array::resize(). The present method only asserts that the new size equals the old size, and does - * nothing else. - */ - EIGEN_DEVICE_FUNC - void resize(Index newSize) - { - EIGEN_ONLY_USED_FOR_DEBUG(newSize); - eigen_assert(newSize == this->size() - && "DenseBase::resize() does not actually allow to resize."); - } - /** Only plain matrices/arrays, not expressions, may be resized; therefore the only useful resize methods are - * Matrix::resize() and Array::resize(). The present method only asserts that the new size equals the old size, and does - * nothing else. - */ - EIGEN_DEVICE_FUNC - void resize(Index rows, Index cols) - { - EIGEN_ONLY_USED_FOR_DEBUG(rows); - EIGEN_ONLY_USED_FOR_DEBUG(cols); - eigen_assert(rows == this->rows() && cols == this->cols() - && "DenseBase::resize() does not actually allow to resize."); - } + + IsRowMajor = int(Flags) & RowMajorBit, /**< True if this expression has row-major storage order. */ + + InnerSizeAtCompileTime = int(IsVectorAtCompileTime) ? int(SizeAtCompileTime) + : int(IsRowMajor) ? int(ColsAtCompileTime) + : int(RowsAtCompileTime), + + InnerStrideAtCompileTime = internal::inner_stride_at_compile_time::ret, + OuterStrideAtCompileTime = internal::outer_stride_at_compile_time::ret + }; + + typedef typename internal::find_best_packet::type PacketScalar; + + enum { IsPlainObjectBase = 0 }; + + /** The plain matrix type corresponding to this expression. + * \sa PlainObject */ + typedef Matrix::Scalar, + internal::traits::RowsAtCompileTime, + internal::traits::ColsAtCompileTime, + AutoAlign | (internal::traits::Flags & RowMajorBit ? RowMajor : ColMajor), + internal::traits::MaxRowsAtCompileTime, + internal::traits::MaxColsAtCompileTime> + PlainMatrix; + + /** The plain array type corresponding to this expression. + * \sa PlainObject */ + typedef Array::Scalar, + internal::traits::RowsAtCompileTime, + internal::traits::ColsAtCompileTime, + AutoAlign | (internal::traits::Flags & RowMajorBit ? RowMajor : ColMajor), + internal::traits::MaxRowsAtCompileTime, + internal::traits::MaxColsAtCompileTime> + PlainArray; + + /** \brief The plain matrix or array type corresponding to this expression. + * + * This is not necessarily exactly the return type of eval(). In the case of plain matrices, + * the return type of eval() is a const reference to a matrix, not a matrix! It is however guaranteed + * that the return type of eval() is either PlainObject or const PlainObject&. + */ + typedef + typename internal::conditional::XprKind, MatrixXpr>::value, + PlainMatrix, + PlainArray>::type PlainObject; + + /** \returns the number of nonzero coefficients which is in practice the number + * of stored coefficients. */ + EIGEN_DEVICE_FUNC + inline Index nonZeros() const { return size(); } + + /** \returns the outer size. + * + * \note For a vector, this returns just 1. For a matrix (non-vector), this is the major dimension + * with respect to the \ref TopicStorageOrders "storage order", i.e., the number of columns for a + * column-major matrix, and the number of rows for a row-major matrix. */ + EIGEN_DEVICE_FUNC + Index outerSize() const { return IsVectorAtCompileTime ? 1 : int(IsRowMajor) ? this->rows() : this->cols(); } + + /** \returns the inner size. + * + * \note For a vector, this is just the size. For a matrix (non-vector), this is the minor dimension + * with respect to the \ref TopicStorageOrders "storage order", i.e., the number of rows for a + * column-major matrix, and the number of columns for a row-major matrix. */ + EIGEN_DEVICE_FUNC + Index innerSize() const + { + return IsVectorAtCompileTime ? this->size() : int(IsRowMajor) ? this->cols() : this->rows(); + } + + /** Only plain matrices/arrays, not expressions, may be resized; therefore the only useful resize methods are + * Matrix::resize() and Array::resize(). The present method only asserts that the new size equals the old size, and + * does nothing else. + */ + EIGEN_DEVICE_FUNC + void resize(Index newSize) + { + EIGEN_ONLY_USED_FOR_DEBUG(newSize); + eigen_assert(newSize == this->size() && "DenseBase::resize() does not actually allow to resize."); + } + /** Only plain matrices/arrays, not expressions, may be resized; therefore the only useful resize methods are + * Matrix::resize() and Array::resize(). The present method only asserts that the new size equals the old size, and + * does nothing else. + */ + EIGEN_DEVICE_FUNC + void resize(Index rows, Index cols) + { + EIGEN_ONLY_USED_FOR_DEBUG(rows); + EIGEN_ONLY_USED_FOR_DEBUG(cols); + eigen_assert( + rows == this->rows() && cols == this->cols() && "DenseBase::resize() does not actually allow to resize."); + } #ifndef EIGEN_PARSED_BY_DOXYGEN - /** \internal Represents a matrix with all coefficients equal to one another*/ - typedef CwiseNullaryOp,PlainObject> ConstantReturnType; - /** \internal \deprecated Represents a vector with linearly spaced coefficients that allows sequential access only. */ - typedef CwiseNullaryOp,PlainObject> SequentialLinSpacedReturnType; - /** \internal Represents a vector with linearly spaced coefficients that allows random access. */ - typedef CwiseNullaryOp,PlainObject> RandomAccessLinSpacedReturnType; - /** \internal the return type of MatrixBase::eigenvalues() */ - typedef Matrix::Scalar>::Real, internal::traits::ColsAtCompileTime, 1> EigenvaluesReturnType; - -#endif // not EIGEN_PARSED_BY_DOXYGEN - - /** Copies \a other into *this. \returns a reference to *this. */ - template - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE - Derived& operator=(const DenseBase& other); - - /** Special case of the template operator=, in order to prevent the compiler - * from generating a default operator= (issue hit with g++ 4.1) - */ - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE - Derived& operator=(const DenseBase& other); - - template - EIGEN_DEVICE_FUNC - Derived& operator=(const EigenBase &other); - - template - EIGEN_DEVICE_FUNC - Derived& operator+=(const EigenBase &other); - - template - EIGEN_DEVICE_FUNC - Derived& operator-=(const EigenBase &other); - - template - EIGEN_DEVICE_FUNC - Derived& operator=(const ReturnByValue& func); - - /** \internal - * Copies \a other into *this without evaluating other. \returns a reference to *this. - * \deprecated */ - template - EIGEN_DEVICE_FUNC - Derived& lazyAssign(const DenseBase& other); - - EIGEN_DEVICE_FUNC - CommaInitializer operator<< (const Scalar& s); - - /** \deprecated it now returns \c *this */ - template - EIGEN_DEPRECATED - const Derived& flagged() const - { return derived(); } - - template - EIGEN_DEVICE_FUNC - CommaInitializer operator<< (const DenseBase& other); - - typedef Transpose TransposeReturnType; - EIGEN_DEVICE_FUNC - TransposeReturnType transpose(); - typedef typename internal::add_const >::type ConstTransposeReturnType; - EIGEN_DEVICE_FUNC - ConstTransposeReturnType transpose() const; - EIGEN_DEVICE_FUNC - void transposeInPlace(); - - EIGEN_DEVICE_FUNC static const ConstantReturnType - Constant(Index rows, Index cols, const Scalar& value); - EIGEN_DEVICE_FUNC static const ConstantReturnType - Constant(Index size, const Scalar& value); - EIGEN_DEVICE_FUNC static const ConstantReturnType - Constant(const Scalar& value); - - EIGEN_DEVICE_FUNC static const SequentialLinSpacedReturnType - LinSpaced(Sequential_t, Index size, const Scalar& low, const Scalar& high); - EIGEN_DEVICE_FUNC static const RandomAccessLinSpacedReturnType - LinSpaced(Index size, const Scalar& low, const Scalar& high); - EIGEN_DEVICE_FUNC static const SequentialLinSpacedReturnType - LinSpaced(Sequential_t, const Scalar& low, const Scalar& high); - EIGEN_DEVICE_FUNC static const RandomAccessLinSpacedReturnType - LinSpaced(const Scalar& low, const Scalar& high); - - template EIGEN_DEVICE_FUNC - static const CwiseNullaryOp - NullaryExpr(Index rows, Index cols, const CustomNullaryOp& func); - template EIGEN_DEVICE_FUNC - static const CwiseNullaryOp - NullaryExpr(Index size, const CustomNullaryOp& func); - template EIGEN_DEVICE_FUNC - static const CwiseNullaryOp - NullaryExpr(const CustomNullaryOp& func); - - EIGEN_DEVICE_FUNC static const ConstantReturnType Zero(Index rows, Index cols); - EIGEN_DEVICE_FUNC static const ConstantReturnType Zero(Index size); - EIGEN_DEVICE_FUNC static const ConstantReturnType Zero(); - EIGEN_DEVICE_FUNC static const ConstantReturnType Ones(Index rows, Index cols); - EIGEN_DEVICE_FUNC static const ConstantReturnType Ones(Index size); - EIGEN_DEVICE_FUNC static const ConstantReturnType Ones(); - - EIGEN_DEVICE_FUNC void fill(const Scalar& value); - EIGEN_DEVICE_FUNC Derived& setConstant(const Scalar& value); - EIGEN_DEVICE_FUNC Derived& setLinSpaced(Index size, const Scalar& low, const Scalar& high); - EIGEN_DEVICE_FUNC Derived& setLinSpaced(const Scalar& low, const Scalar& high); - EIGEN_DEVICE_FUNC Derived& setZero(); - EIGEN_DEVICE_FUNC Derived& setOnes(); - EIGEN_DEVICE_FUNC Derived& setRandom(); - - template EIGEN_DEVICE_FUNC - bool isApprox(const DenseBase& other, - const RealScalar& prec = NumTraits::dummy_precision()) const; - EIGEN_DEVICE_FUNC - bool isMuchSmallerThan(const RealScalar& other, - const RealScalar& prec = NumTraits::dummy_precision()) const; - template EIGEN_DEVICE_FUNC - bool isMuchSmallerThan(const DenseBase& other, - const RealScalar& prec = NumTraits::dummy_precision()) const; - - EIGEN_DEVICE_FUNC bool isApproxToConstant(const Scalar& value, const RealScalar& prec = NumTraits::dummy_precision()) const; - EIGEN_DEVICE_FUNC bool isConstant(const Scalar& value, const RealScalar& prec = NumTraits::dummy_precision()) const; - EIGEN_DEVICE_FUNC bool isZero(const RealScalar& prec = NumTraits::dummy_precision()) const; - EIGEN_DEVICE_FUNC bool isOnes(const RealScalar& prec = NumTraits::dummy_precision()) const; - - inline bool hasNaN() const; - inline bool allFinite() const; - - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE - Derived& operator*=(const Scalar& other); - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE - Derived& operator/=(const Scalar& other); - - typedef typename internal::add_const_on_value_type::type>::type EvalReturnType; - /** \returns the matrix or vector obtained by evaluating this expression. - * - * Notice that in the case of a plain matrix or vector (not an expression) this function just returns - * a const reference, in order to avoid a useless copy. - * - * \warning Be carefull with eval() and the auto C++ keyword, as detailed in this \link TopicPitfalls_auto_keyword page \endlink. - */ - EIGEN_DEVICE_FUNC - EIGEN_STRONG_INLINE EvalReturnType eval() const - { - // Even though MSVC does not honor strong inlining when the return type - // is a dynamic matrix, we desperately need strong inlining for fixed - // size types on MSVC. - return typename internal::eval::type(derived()); - } - - /** swaps *this with the expression \a other. - * - */ - template - EIGEN_DEVICE_FUNC - void swap(const DenseBase& other) - { - EIGEN_STATIC_ASSERT(!OtherDerived::IsPlainObjectBase,THIS_EXPRESSION_IS_NOT_A_LVALUE__IT_IS_READ_ONLY); - eigen_assert(rows()==other.rows() && cols()==other.cols()); - call_assignment(derived(), other.const_cast_derived(), internal::swap_assign_op()); - } - - /** swaps *this with the matrix or array \a other. - * - */ - template - EIGEN_DEVICE_FUNC - void swap(PlainObjectBase& other) - { - eigen_assert(rows()==other.rows() && cols()==other.cols()); - call_assignment(derived(), other.derived(), internal::swap_assign_op()); - } - - EIGEN_DEVICE_FUNC inline const NestByValue nestByValue() const; - EIGEN_DEVICE_FUNC inline const ForceAlignedAccess forceAlignedAccess() const; - EIGEN_DEVICE_FUNC inline ForceAlignedAccess forceAlignedAccess(); - template EIGEN_DEVICE_FUNC - inline const typename internal::conditional,Derived&>::type forceAlignedAccessIf() const; - template EIGEN_DEVICE_FUNC - inline typename internal::conditional,Derived&>::type forceAlignedAccessIf(); - - EIGEN_DEVICE_FUNC Scalar sum() const; - EIGEN_DEVICE_FUNC Scalar mean() const; - EIGEN_DEVICE_FUNC Scalar trace() const; - - EIGEN_DEVICE_FUNC Scalar prod() const; - - EIGEN_DEVICE_FUNC typename internal::traits::Scalar minCoeff() const; - EIGEN_DEVICE_FUNC typename internal::traits::Scalar maxCoeff() const; - - template EIGEN_DEVICE_FUNC - typename internal::traits::Scalar minCoeff(IndexType* row, IndexType* col) const; - template EIGEN_DEVICE_FUNC - typename internal::traits::Scalar maxCoeff(IndexType* row, IndexType* col) const; - template EIGEN_DEVICE_FUNC - typename internal::traits::Scalar minCoeff(IndexType* index) const; - template EIGEN_DEVICE_FUNC - typename internal::traits::Scalar maxCoeff(IndexType* index) const; - - template - EIGEN_DEVICE_FUNC - Scalar redux(const BinaryOp& func) const; - - template - EIGEN_DEVICE_FUNC - void visit(Visitor& func) const; - - /** \returns a WithFormat proxy object allowing to print a matrix the with given - * format \a fmt. - * - * See class IOFormat for some examples. - * - * \sa class IOFormat, class WithFormat - */ - inline const WithFormat format(const IOFormat& fmt) const - { - return WithFormat(derived(), fmt); - } - - /** \returns the unique coefficient of a 1x1 expression */ - EIGEN_DEVICE_FUNC - CoeffReturnType value() const - { - EIGEN_STATIC_ASSERT_SIZE_1x1(Derived) - eigen_assert(this->rows() == 1 && this->cols() == 1); - return derived().coeff(0,0); - } - - EIGEN_DEVICE_FUNC bool all() const; - EIGEN_DEVICE_FUNC bool any() const; - EIGEN_DEVICE_FUNC Index count() const; - - typedef VectorwiseOp RowwiseReturnType; - typedef const VectorwiseOp ConstRowwiseReturnType; - typedef VectorwiseOp ColwiseReturnType; - typedef const VectorwiseOp ConstColwiseReturnType; - - /** \returns a VectorwiseOp wrapper of *this providing additional partial reduction operations - * - * Example: \include MatrixBase_rowwise.cpp - * Output: \verbinclude MatrixBase_rowwise.out - * - * \sa colwise(), class VectorwiseOp, \ref TutorialReductionsVisitorsBroadcasting - */ - //Code moved here due to a CUDA compiler bug - EIGEN_DEVICE_FUNC inline ConstRowwiseReturnType rowwise() const { - return ConstRowwiseReturnType(derived()); - } - EIGEN_DEVICE_FUNC RowwiseReturnType rowwise(); - - /** \returns a VectorwiseOp wrapper of *this providing additional partial reduction operations - * - * Example: \include MatrixBase_colwise.cpp - * Output: \verbinclude MatrixBase_colwise.out - * - * \sa rowwise(), class VectorwiseOp, \ref TutorialReductionsVisitorsBroadcasting - */ - EIGEN_DEVICE_FUNC inline ConstColwiseReturnType colwise() const { - return ConstColwiseReturnType(derived()); - } - EIGEN_DEVICE_FUNC ColwiseReturnType colwise(); - - typedef CwiseNullaryOp,PlainObject> RandomReturnType; - static const RandomReturnType Random(Index rows, Index cols); - static const RandomReturnType Random(Index size); - static const RandomReturnType Random(); - - template - const Select - select(const DenseBase& thenMatrix, - const DenseBase& elseMatrix) const; - - template - inline const Select - select(const DenseBase& thenMatrix, const typename ThenDerived::Scalar& elseScalar) const; - - template - inline const Select - select(const typename ElseDerived::Scalar& thenScalar, const DenseBase& elseMatrix) const; - - template RealScalar lpNorm() const; - - template - EIGEN_DEVICE_FUNC - const Replicate replicate() const; - /** - * \return an expression of the replication of \c *this - * - * Example: \include MatrixBase_replicate_int_int.cpp - * Output: \verbinclude MatrixBase_replicate_int_int.out - * - * \sa VectorwiseOp::replicate(), DenseBase::replicate(), class Replicate - */ - //Code moved here due to a CUDA compiler bug - EIGEN_DEVICE_FUNC - const Replicate replicate(Index rowFactor, Index colFactor) const - { - return Replicate(derived(), rowFactor, colFactor); - } - - typedef Reverse ReverseReturnType; - typedef const Reverse ConstReverseReturnType; - EIGEN_DEVICE_FUNC ReverseReturnType reverse(); - /** This is the const version of reverse(). */ - //Code moved here due to a CUDA compiler bug - EIGEN_DEVICE_FUNC ConstReverseReturnType reverse() const - { - return ConstReverseReturnType(derived()); - } - EIGEN_DEVICE_FUNC void reverseInPlace(); + /** \internal Represents a matrix with all coefficients equal to one another*/ + typedef CwiseNullaryOp, PlainObject> ConstantReturnType; + /** \internal \deprecated Represents a vector with linearly spaced coefficients that allows sequential access only. */ + typedef CwiseNullaryOp, PlainObject> SequentialLinSpacedReturnType; + /** \internal Represents a vector with linearly spaced coefficients that allows random access. */ + typedef CwiseNullaryOp, PlainObject> RandomAccessLinSpacedReturnType; + /** \internal the return type of MatrixBase::eigenvalues() */ + typedef Matrix::Scalar>::Real, + internal::traits::ColsAtCompileTime, + 1> + EigenvaluesReturnType; + +#endif// not EIGEN_PARSED_BY_DOXYGEN + + /** Copies \a other into *this. \returns a reference to *this. */ + template + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Derived &operator=(const DenseBase &other); + + /** Special case of the template operator=, in order to prevent the compiler + * from generating a default operator= (issue hit with g++ 4.1) + */ + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Derived &operator=(const DenseBase &other); + + template EIGEN_DEVICE_FUNC Derived &operator=(const EigenBase &other); + + template EIGEN_DEVICE_FUNC Derived &operator+=(const EigenBase &other); + + template EIGEN_DEVICE_FUNC Derived &operator-=(const EigenBase &other); + + template EIGEN_DEVICE_FUNC Derived &operator=(const ReturnByValue &func); + + /** \internal + * Copies \a other into *this without evaluating other. \returns a reference to *this. + * \deprecated */ + template EIGEN_DEVICE_FUNC Derived &lazyAssign(const DenseBase &other); + + EIGEN_DEVICE_FUNC + CommaInitializer operator<<(const Scalar &s); + + /** \deprecated it now returns \c *this */ + template EIGEN_DEPRECATED const Derived &flagged() const + { + return derived(); + } + + template + EIGEN_DEVICE_FUNC CommaInitializer operator<<(const DenseBase &other); + + typedef Transpose TransposeReturnType; + EIGEN_DEVICE_FUNC + TransposeReturnType transpose(); + typedef typename internal::add_const>::type ConstTransposeReturnType; + EIGEN_DEVICE_FUNC + ConstTransposeReturnType transpose() const; + EIGEN_DEVICE_FUNC + void transposeInPlace(); + + EIGEN_DEVICE_FUNC static const ConstantReturnType Constant(Index rows, Index cols, const Scalar &value); + EIGEN_DEVICE_FUNC static const ConstantReturnType Constant(Index size, const Scalar &value); + EIGEN_DEVICE_FUNC static const ConstantReturnType Constant(const Scalar &value); + + EIGEN_DEVICE_FUNC static const SequentialLinSpacedReturnType + LinSpaced(Sequential_t, Index size, const Scalar &low, const Scalar &high); + EIGEN_DEVICE_FUNC static const RandomAccessLinSpacedReturnType + LinSpaced(Index size, const Scalar &low, const Scalar &high); + EIGEN_DEVICE_FUNC static const SequentialLinSpacedReturnType + LinSpaced(Sequential_t, const Scalar &low, const Scalar &high); + EIGEN_DEVICE_FUNC static const RandomAccessLinSpacedReturnType LinSpaced(const Scalar &low, const Scalar &high); + + template + EIGEN_DEVICE_FUNC static const CwiseNullaryOp + NullaryExpr(Index rows, Index cols, const CustomNullaryOp &func); + template + EIGEN_DEVICE_FUNC static const CwiseNullaryOp NullaryExpr(Index size, + const CustomNullaryOp &func); + template + EIGEN_DEVICE_FUNC static const CwiseNullaryOp NullaryExpr(const CustomNullaryOp &func); + + EIGEN_DEVICE_FUNC static const ConstantReturnType Zero(Index rows, Index cols); + EIGEN_DEVICE_FUNC static const ConstantReturnType Zero(Index size); + EIGEN_DEVICE_FUNC static const ConstantReturnType Zero(); + EIGEN_DEVICE_FUNC static const ConstantReturnType Ones(Index rows, Index cols); + EIGEN_DEVICE_FUNC static const ConstantReturnType Ones(Index size); + EIGEN_DEVICE_FUNC static const ConstantReturnType Ones(); + + EIGEN_DEVICE_FUNC void fill(const Scalar &value); + EIGEN_DEVICE_FUNC Derived &setConstant(const Scalar &value); + EIGEN_DEVICE_FUNC Derived &setLinSpaced(Index size, const Scalar &low, const Scalar &high); + EIGEN_DEVICE_FUNC Derived &setLinSpaced(const Scalar &low, const Scalar &high); + EIGEN_DEVICE_FUNC Derived &setZero(); + EIGEN_DEVICE_FUNC Derived &setOnes(); + EIGEN_DEVICE_FUNC Derived &setRandom(); + + template + EIGEN_DEVICE_FUNC bool isApprox(const DenseBase &other, + const RealScalar &prec = NumTraits::dummy_precision()) const; + EIGEN_DEVICE_FUNC + bool isMuchSmallerThan(const RealScalar &other, const RealScalar &prec = NumTraits::dummy_precision()) const; + template + EIGEN_DEVICE_FUNC bool isMuchSmallerThan(const DenseBase &other, + const RealScalar &prec = NumTraits::dummy_precision()) const; + + EIGEN_DEVICE_FUNC bool isApproxToConstant(const Scalar &value, + const RealScalar &prec = NumTraits::dummy_precision()) const; + EIGEN_DEVICE_FUNC bool isConstant(const Scalar &value, + const RealScalar &prec = NumTraits::dummy_precision()) const; + EIGEN_DEVICE_FUNC bool isZero(const RealScalar &prec = NumTraits::dummy_precision()) const; + EIGEN_DEVICE_FUNC bool isOnes(const RealScalar &prec = NumTraits::dummy_precision()) const; + + inline bool hasNaN() const; + inline bool allFinite() const; + + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Derived &operator*=(const Scalar &other); + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Derived &operator/=(const Scalar &other); + + typedef typename internal::add_const_on_value_type::type>::type EvalReturnType; + /** \returns the matrix or vector obtained by evaluating this expression. + * + * Notice that in the case of a plain matrix or vector (not an expression) this function just returns + * a const reference, in order to avoid a useless copy. + * + * \warning Be carefull with eval() and the auto C++ keyword, as detailed in this \link TopicPitfalls_auto_keyword + * page \endlink. + */ + EIGEN_DEVICE_FUNC + EIGEN_STRONG_INLINE EvalReturnType eval() const + { + // Even though MSVC does not honor strong inlining when the return type + // is a dynamic matrix, we desperately need strong inlining for fixed + // size types on MSVC. + return typename internal::eval::type(derived()); + } + + /** swaps *this with the expression \a other. + * + */ + template EIGEN_DEVICE_FUNC void swap(const DenseBase &other) + { + EIGEN_STATIC_ASSERT(!OtherDerived::IsPlainObjectBase, THIS_EXPRESSION_IS_NOT_A_LVALUE__IT_IS_READ_ONLY); + eigen_assert(rows() == other.rows() && cols() == other.cols()); + call_assignment(derived(), other.const_cast_derived(), internal::swap_assign_op()); + } + + /** swaps *this with the matrix or array \a other. + * + */ + template EIGEN_DEVICE_FUNC void swap(PlainObjectBase &other) + { + eigen_assert(rows() == other.rows() && cols() == other.cols()); + call_assignment(derived(), other.derived(), internal::swap_assign_op()); + } + + EIGEN_DEVICE_FUNC inline const NestByValue nestByValue() const; + EIGEN_DEVICE_FUNC inline const ForceAlignedAccess forceAlignedAccess() const; + EIGEN_DEVICE_FUNC inline ForceAlignedAccess forceAlignedAccess(); + template + EIGEN_DEVICE_FUNC inline const typename internal::conditional, Derived &>::type + forceAlignedAccessIf() const; + template + EIGEN_DEVICE_FUNC inline typename internal::conditional, Derived &>::type + forceAlignedAccessIf(); + + EIGEN_DEVICE_FUNC Scalar sum() const; + EIGEN_DEVICE_FUNC Scalar mean() const; + EIGEN_DEVICE_FUNC Scalar trace() const; + + EIGEN_DEVICE_FUNC Scalar prod() const; + + EIGEN_DEVICE_FUNC typename internal::traits::Scalar minCoeff() const; + EIGEN_DEVICE_FUNC typename internal::traits::Scalar maxCoeff() const; + + template + EIGEN_DEVICE_FUNC typename internal::traits::Scalar minCoeff(IndexType *row, IndexType *col) const; + template + EIGEN_DEVICE_FUNC typename internal::traits::Scalar maxCoeff(IndexType *row, IndexType *col) const; + template + EIGEN_DEVICE_FUNC typename internal::traits::Scalar minCoeff(IndexType *index) const; + template + EIGEN_DEVICE_FUNC typename internal::traits::Scalar maxCoeff(IndexType *index) const; + + template EIGEN_DEVICE_FUNC Scalar redux(const BinaryOp &func) const; + + template EIGEN_DEVICE_FUNC void visit(Visitor &func) const; + + /** \returns a WithFormat proxy object allowing to print a matrix the with given + * format \a fmt. + * + * See class IOFormat for some examples. + * + * \sa class IOFormat, class WithFormat + */ + inline const WithFormat format(const IOFormat &fmt) const { return WithFormat(derived(), fmt); } + + /** \returns the unique coefficient of a 1x1 expression */ + EIGEN_DEVICE_FUNC + CoeffReturnType value() const + { + EIGEN_STATIC_ASSERT_SIZE_1x1(Derived) eigen_assert(this->rows() == 1 && this->cols() == 1); + return derived().coeff(0, 0); + } + + EIGEN_DEVICE_FUNC bool all() const; + EIGEN_DEVICE_FUNC bool any() const; + EIGEN_DEVICE_FUNC Index count() const; + + typedef VectorwiseOp RowwiseReturnType; + typedef const VectorwiseOp ConstRowwiseReturnType; + typedef VectorwiseOp ColwiseReturnType; + typedef const VectorwiseOp ConstColwiseReturnType; + + /** \returns a VectorwiseOp wrapper of *this providing additional partial reduction operations + * + * Example: \include MatrixBase_rowwise.cpp + * Output: \verbinclude MatrixBase_rowwise.out + * + * \sa colwise(), class VectorwiseOp, \ref TutorialReductionsVisitorsBroadcasting + */ + // Code moved here due to a CUDA compiler bug + EIGEN_DEVICE_FUNC inline ConstRowwiseReturnType rowwise() const { return ConstRowwiseReturnType(derived()); } + EIGEN_DEVICE_FUNC RowwiseReturnType rowwise(); + + /** \returns a VectorwiseOp wrapper of *this providing additional partial reduction operations + * + * Example: \include MatrixBase_colwise.cpp + * Output: \verbinclude MatrixBase_colwise.out + * + * \sa rowwise(), class VectorwiseOp, \ref TutorialReductionsVisitorsBroadcasting + */ + EIGEN_DEVICE_FUNC inline ConstColwiseReturnType colwise() const { return ConstColwiseReturnType(derived()); } + EIGEN_DEVICE_FUNC ColwiseReturnType colwise(); + + typedef CwiseNullaryOp, PlainObject> RandomReturnType; + static const RandomReturnType Random(Index rows, Index cols); + static const RandomReturnType Random(Index size); + static const RandomReturnType Random(); + + template + const Select select(const DenseBase &thenMatrix, + const DenseBase &elseMatrix) const; + + template + inline const Select + select(const DenseBase &thenMatrix, const typename ThenDerived::Scalar &elseScalar) const; + + template + inline const Select + select(const typename ElseDerived::Scalar &thenScalar, const DenseBase &elseMatrix) const; + + template RealScalar lpNorm() const; + + template + EIGEN_DEVICE_FUNC const Replicate replicate() const; + /** + * \return an expression of the replication of \c *this + * + * Example: \include MatrixBase_replicate_int_int.cpp + * Output: \verbinclude MatrixBase_replicate_int_int.out + * + * \sa VectorwiseOp::replicate(), DenseBase::replicate(), class Replicate + */ + // Code moved here due to a CUDA compiler bug + EIGEN_DEVICE_FUNC + const Replicate replicate(Index rowFactor, Index colFactor) const + { + return Replicate(derived(), rowFactor, colFactor); + } + + typedef Reverse ReverseReturnType; + typedef const Reverse ConstReverseReturnType; + EIGEN_DEVICE_FUNC ReverseReturnType reverse(); + /** This is the const version of reverse(). */ + // Code moved here due to a CUDA compiler bug + EIGEN_DEVICE_FUNC ConstReverseReturnType reverse() const { return ConstReverseReturnType(derived()); } + EIGEN_DEVICE_FUNC void reverseInPlace(); #define EIGEN_CURRENT_STORAGE_BASE_CLASS Eigen::DenseBase #define EIGEN_DOC_BLOCK_ADDONS_NOT_INNER_PANEL #define EIGEN_DOC_BLOCK_ADDONS_INNER_PANEL_IF(COND) -# include "../plugins/BlockMethods.h" -# ifdef EIGEN_DENSEBASE_PLUGIN -# include EIGEN_DENSEBASE_PLUGIN -# endif +#include "../plugins/BlockMethods.h" +#ifdef EIGEN_DENSEBASE_PLUGIN +#include EIGEN_DENSEBASE_PLUGIN +#endif #undef EIGEN_CURRENT_STORAGE_BASE_CLASS #undef EIGEN_DOC_BLOCK_ADDONS_NOT_INNER_PANEL #undef EIGEN_DOC_BLOCK_ADDONS_INNER_PANEL_IF - // disable the use of evalTo for dense objects with a nice compilation error - template - EIGEN_DEVICE_FUNC - inline void evalTo(Dest& ) const - { - EIGEN_STATIC_ASSERT((internal::is_same::value),THE_EVAL_EVALTO_FUNCTION_SHOULD_NEVER_BE_CALLED_FOR_DENSE_OBJECTS); - } - - protected: - /** Default constructor. Do nothing. */ - EIGEN_DEVICE_FUNC DenseBase() - { - /* Just checks for self-consistency of the flags. - * Only do it when debugging Eigen, as this borders on paranoiac and could slow compilation down - */ + // disable the use of evalTo for dense objects with a nice compilation error + template EIGEN_DEVICE_FUNC inline void evalTo(Dest &) const + { + EIGEN_STATIC_ASSERT( + (internal::is_same::value), THE_EVAL_EVALTO_FUNCTION_SHOULD_NEVER_BE_CALLED_FOR_DENSE_OBJECTS); + } + +protected: + /** Default constructor. Do nothing. */ + EIGEN_DEVICE_FUNC DenseBase() + { + /* Just checks for self-consistency of the flags. + * Only do it when debugging Eigen, as this borders on paranoiac and could slow compilation down + */ #ifdef EIGEN_INTERNAL_DEBUGGING - EIGEN_STATIC_ASSERT((EIGEN_IMPLIES(MaxRowsAtCompileTime==1 && MaxColsAtCompileTime!=1, int(IsRowMajor)) - && EIGEN_IMPLIES(MaxColsAtCompileTime==1 && MaxRowsAtCompileTime!=1, int(!IsRowMajor))), - INVALID_STORAGE_ORDER_FOR_THIS_VECTOR_EXPRESSION) + EIGEN_STATIC_ASSERT((EIGEN_IMPLIES(MaxRowsAtCompileTime == 1 && MaxColsAtCompileTime != 1, int(IsRowMajor)) + && EIGEN_IMPLIES(MaxColsAtCompileTime == 1 && MaxRowsAtCompileTime != 1, int(!IsRowMajor))), + INVALID_STORAGE_ORDER_FOR_THIS_VECTOR_EXPRESSION) #endif - } + } - private: - EIGEN_DEVICE_FUNC explicit DenseBase(int); - EIGEN_DEVICE_FUNC DenseBase(int,int); - template EIGEN_DEVICE_FUNC explicit DenseBase(const DenseBase&); +private: + EIGEN_DEVICE_FUNC explicit DenseBase(int); + EIGEN_DEVICE_FUNC DenseBase(int, int); + template EIGEN_DEVICE_FUNC explicit DenseBase(const DenseBase &); }; -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_DENSEBASE_H +#endif// EIGEN_DENSEBASE_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/DenseCoeffsBase.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/DenseCoeffsBase.h index c4af48ab..8f1f37b6 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/DenseCoeffsBase.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/DenseCoeffsBase.h @@ -13,669 +13,596 @@ namespace Eigen { namespace internal { -template struct add_const_on_value_type_if_arithmetic -{ - typedef typename conditional::value, T, typename add_const_on_value_type::type>::type type; -}; -} + template struct add_const_on_value_type_if_arithmetic + { + typedef typename conditional::value, T, typename add_const_on_value_type::type>::type type; + }; +}// namespace internal /** \brief Base class providing read-only coefficient access to matrices and arrays. - * \ingroup Core_Module - * \tparam Derived Type of the derived class - * \tparam #ReadOnlyAccessors Constant indicating read-only access - * - * This class defines the \c operator() \c const function and friends, which can be used to read specific - * entries of a matrix or array. - * - * \sa DenseCoeffsBase, DenseCoeffsBase, - * \ref TopicClassHierarchy - */ -template -class DenseCoeffsBase : public EigenBase + * \ingroup Core_Module + * \tparam Derived Type of the derived class + * \tparam #ReadOnlyAccessors Constant indicating read-only access + * + * This class defines the \c operator() \c const function and friends, which can be used to read specific + * entries of a matrix or array. + * + * \sa DenseCoeffsBase, DenseCoeffsBase, + * \ref TopicClassHierarchy + */ +template class DenseCoeffsBase : public EigenBase { - public: - typedef typename internal::traits::StorageKind StorageKind; - typedef typename internal::traits::Scalar Scalar; - typedef typename internal::packet_traits::type PacketScalar; - - // Explanation for this CoeffReturnType typedef. - // - This is the return type of the coeff() method. - // - The LvalueBit means exactly that we can offer a coeffRef() method, which means exactly that we can get references - // to coeffs, which means exactly that we can have coeff() return a const reference (as opposed to returning a value). - // - The is_artihmetic check is required since "const int", "const double", etc. will cause warnings on some systems - // while the declaration of "const T", where T is a non arithmetic type does not. Always returning "const Scalar&" is - // not possible, since the underlying expressions might not offer a valid address the reference could be referring to. - typedef typename internal::conditional::Flags&LvalueBit), - const Scalar&, - typename internal::conditional::value, Scalar, const Scalar>::type - >::type CoeffReturnType; - - typedef typename internal::add_const_on_value_type_if_arithmetic< - typename internal::packet_traits::type - >::type PacketReturnType; - - typedef EigenBase Base; - using Base::rows; - using Base::cols; - using Base::size; - using Base::derived; - - EIGEN_DEVICE_FUNC - EIGEN_STRONG_INLINE Index rowIndexByOuterInner(Index outer, Index inner) const - { - return int(Derived::RowsAtCompileTime) == 1 ? 0 - : int(Derived::ColsAtCompileTime) == 1 ? inner - : int(Derived::Flags)&RowMajorBit ? outer - : inner; - } - - EIGEN_DEVICE_FUNC - EIGEN_STRONG_INLINE Index colIndexByOuterInner(Index outer, Index inner) const - { - return int(Derived::ColsAtCompileTime) == 1 ? 0 - : int(Derived::RowsAtCompileTime) == 1 ? inner - : int(Derived::Flags)&RowMajorBit ? inner - : outer; - } - - /** Short version: don't use this function, use - * \link operator()(Index,Index) const \endlink instead. - * - * Long version: this function is similar to - * \link operator()(Index,Index) const \endlink, but without the assertion. - * Use this for limiting the performance cost of debugging code when doing - * repeated coefficient access. Only use this when it is guaranteed that the - * parameters \a row and \a col are in range. - * - * If EIGEN_INTERNAL_DEBUGGING is defined, an assertion will be made, making this - * function equivalent to \link operator()(Index,Index) const \endlink. - * - * \sa operator()(Index,Index) const, coeffRef(Index,Index), coeff(Index) const - */ - EIGEN_DEVICE_FUNC - EIGEN_STRONG_INLINE CoeffReturnType coeff(Index row, Index col) const - { - eigen_internal_assert(row >= 0 && row < rows() - && col >= 0 && col < cols()); - return internal::evaluator(derived()).coeff(row,col); - } - - EIGEN_DEVICE_FUNC - EIGEN_STRONG_INLINE CoeffReturnType coeffByOuterInner(Index outer, Index inner) const - { - return coeff(rowIndexByOuterInner(outer, inner), - colIndexByOuterInner(outer, inner)); - } - - /** \returns the coefficient at given the given row and column. - * - * \sa operator()(Index,Index), operator[](Index) - */ - EIGEN_DEVICE_FUNC - EIGEN_STRONG_INLINE CoeffReturnType operator()(Index row, Index col) const - { - eigen_assert(row >= 0 && row < rows() - && col >= 0 && col < cols()); - return coeff(row, col); - } - - /** Short version: don't use this function, use - * \link operator[](Index) const \endlink instead. - * - * Long version: this function is similar to - * \link operator[](Index) const \endlink, but without the assertion. - * Use this for limiting the performance cost of debugging code when doing - * repeated coefficient access. Only use this when it is guaranteed that the - * parameter \a index is in range. - * - * If EIGEN_INTERNAL_DEBUGGING is defined, an assertion will be made, making this - * function equivalent to \link operator[](Index) const \endlink. - * - * \sa operator[](Index) const, coeffRef(Index), coeff(Index,Index) const - */ - - EIGEN_DEVICE_FUNC - EIGEN_STRONG_INLINE CoeffReturnType - coeff(Index index) const - { - EIGEN_STATIC_ASSERT(internal::evaluator::Flags & LinearAccessBit, - THIS_COEFFICIENT_ACCESSOR_TAKING_ONE_ACCESS_IS_ONLY_FOR_EXPRESSIONS_ALLOWING_LINEAR_ACCESS) - eigen_internal_assert(index >= 0 && index < size()); - return internal::evaluator(derived()).coeff(index); - } - - - /** \returns the coefficient at given index. - * - * This method is allowed only for vector expressions, and for matrix expressions having the LinearAccessBit. - * - * \sa operator[](Index), operator()(Index,Index) const, x() const, y() const, - * z() const, w() const - */ - - EIGEN_DEVICE_FUNC - EIGEN_STRONG_INLINE CoeffReturnType - operator[](Index index) const - { - EIGEN_STATIC_ASSERT(Derived::IsVectorAtCompileTime, - THE_BRACKET_OPERATOR_IS_ONLY_FOR_VECTORS__USE_THE_PARENTHESIS_OPERATOR_INSTEAD) - eigen_assert(index >= 0 && index < size()); - return coeff(index); - } - - /** \returns the coefficient at given index. - * - * This is synonymous to operator[](Index) const. - * - * This method is allowed only for vector expressions, and for matrix expressions having the LinearAccessBit. - * - * \sa operator[](Index), operator()(Index,Index) const, x() const, y() const, - * z() const, w() const - */ - - EIGEN_DEVICE_FUNC - EIGEN_STRONG_INLINE CoeffReturnType - operator()(Index index) const - { - eigen_assert(index >= 0 && index < size()); - return coeff(index); - } - - /** equivalent to operator[](0). */ - - EIGEN_DEVICE_FUNC - EIGEN_STRONG_INLINE CoeffReturnType - x() const { return (*this)[0]; } - - /** equivalent to operator[](1). */ - - EIGEN_DEVICE_FUNC - EIGEN_STRONG_INLINE CoeffReturnType - y() const - { - EIGEN_STATIC_ASSERT(Derived::SizeAtCompileTime==-1 || Derived::SizeAtCompileTime>=2, OUT_OF_RANGE_ACCESS); - return (*this)[1]; - } - - /** equivalent to operator[](2). */ - - EIGEN_DEVICE_FUNC - EIGEN_STRONG_INLINE CoeffReturnType - z() const - { - EIGEN_STATIC_ASSERT(Derived::SizeAtCompileTime==-1 || Derived::SizeAtCompileTime>=3, OUT_OF_RANGE_ACCESS); - return (*this)[2]; - } - - /** equivalent to operator[](3). */ - - EIGEN_DEVICE_FUNC - EIGEN_STRONG_INLINE CoeffReturnType - w() const - { - EIGEN_STATIC_ASSERT(Derived::SizeAtCompileTime==-1 || Derived::SizeAtCompileTime>=4, OUT_OF_RANGE_ACCESS); - return (*this)[3]; - } - - /** \internal - * \returns the packet of coefficients starting at the given row and column. It is your responsibility - * to ensure that a packet really starts there. This method is only available on expressions having the - * PacketAccessBit. - * - * The \a LoadMode parameter may have the value \a #Aligned or \a #Unaligned. Its effect is to select - * the appropriate vectorization instruction. Aligned access is faster, but is only possible for packets - * starting at an address which is a multiple of the packet size. - */ - - template - EIGEN_STRONG_INLINE PacketReturnType packet(Index row, Index col) const - { - typedef typename internal::packet_traits::type DefaultPacketType; - eigen_internal_assert(row >= 0 && row < rows() && col >= 0 && col < cols()); - return internal::evaluator(derived()).template packet(row,col); - } - - - /** \internal */ - template - EIGEN_STRONG_INLINE PacketReturnType packetByOuterInner(Index outer, Index inner) const - { - return packet(rowIndexByOuterInner(outer, inner), - colIndexByOuterInner(outer, inner)); - } - - /** \internal - * \returns the packet of coefficients starting at the given index. It is your responsibility - * to ensure that a packet really starts there. This method is only available on expressions having the - * PacketAccessBit and the LinearAccessBit. - * - * The \a LoadMode parameter may have the value \a #Aligned or \a #Unaligned. Its effect is to select - * the appropriate vectorization instruction. Aligned access is faster, but is only possible for packets - * starting at an address which is a multiple of the packet size. - */ - - template - EIGEN_STRONG_INLINE PacketReturnType packet(Index index) const - { - EIGEN_STATIC_ASSERT(internal::evaluator::Flags & LinearAccessBit, - THIS_COEFFICIENT_ACCESSOR_TAKING_ONE_ACCESS_IS_ONLY_FOR_EXPRESSIONS_ALLOWING_LINEAR_ACCESS) - typedef typename internal::packet_traits::type DefaultPacketType; - eigen_internal_assert(index >= 0 && index < size()); - return internal::evaluator(derived()).template packet(index); - } - - protected: - // explanation: DenseBase is doing "using ..." on the methods from DenseCoeffsBase. - // But some methods are only available in the DirectAccess case. - // So we add dummy methods here with these names, so that "using... " doesn't fail. - // It's not private so that the child class DenseBase can access them, and it's not public - // either since it's an implementation detail, so has to be protected. - void coeffRef(); - void coeffRefByOuterInner(); - void writePacket(); - void writePacketByOuterInner(); - void copyCoeff(); - void copyCoeffByOuterInner(); - void copyPacket(); - void copyPacketByOuterInner(); - void stride(); - void innerStride(); - void outerStride(); - void rowStride(); - void colStride(); +public: + typedef typename internal::traits::StorageKind StorageKind; + typedef typename internal::traits::Scalar Scalar; + typedef typename internal::packet_traits::type PacketScalar; + + // Explanation for this CoeffReturnType typedef. + // - This is the return type of the coeff() method. + // - The LvalueBit means exactly that we can offer a coeffRef() method, which means exactly that we can get references + // to coeffs, which means exactly that we can have coeff() return a const reference (as opposed to returning a value). + // - The is_artihmetic check is required since "const int", "const double", etc. will cause warnings on some systems + // while the declaration of "const T", where T is a non arithmetic type does not. Always returning "const Scalar&" is + // not possible, since the underlying expressions might not offer a valid address the reference could be referring to. + typedef typename internal::conditional::Flags &LvalueBit), + const Scalar &, + typename internal::conditional::value, Scalar, const Scalar>::type>::type + CoeffReturnType; + + typedef typename internal::add_const_on_value_type_if_arithmetic::type>::type + PacketReturnType; + + typedef EigenBase Base; + using Base::rows; + using Base::cols; + using Base::size; + using Base::derived; + + EIGEN_DEVICE_FUNC + EIGEN_STRONG_INLINE Index rowIndexByOuterInner(Index outer, Index inner) const + { + return int(Derived::RowsAtCompileTime) == 1 ? 0 + : int(Derived::ColsAtCompileTime) == 1 ? inner + : int(Derived::Flags) & RowMajorBit ? outer + : inner; + } + + EIGEN_DEVICE_FUNC + EIGEN_STRONG_INLINE Index colIndexByOuterInner(Index outer, Index inner) const + { + return int(Derived::ColsAtCompileTime) == 1 ? 0 + : int(Derived::RowsAtCompileTime) == 1 ? inner + : int(Derived::Flags) & RowMajorBit ? inner + : outer; + } + + /** Short version: don't use this function, use + * \link operator()(Index,Index) const \endlink instead. + * + * Long version: this function is similar to + * \link operator()(Index,Index) const \endlink, but without the assertion. + * Use this for limiting the performance cost of debugging code when doing + * repeated coefficient access. Only use this when it is guaranteed that the + * parameters \a row and \a col are in range. + * + * If EIGEN_INTERNAL_DEBUGGING is defined, an assertion will be made, making this + * function equivalent to \link operator()(Index,Index) const \endlink. + * + * \sa operator()(Index,Index) const, coeffRef(Index,Index), coeff(Index) const + */ + EIGEN_DEVICE_FUNC + EIGEN_STRONG_INLINE CoeffReturnType coeff(Index row, Index col) const + { + eigen_internal_assert(row >= 0 && row < rows() && col >= 0 && col < cols()); + return internal::evaluator(derived()).coeff(row, col); + } + + EIGEN_DEVICE_FUNC + EIGEN_STRONG_INLINE CoeffReturnType coeffByOuterInner(Index outer, Index inner) const + { + return coeff(rowIndexByOuterInner(outer, inner), colIndexByOuterInner(outer, inner)); + } + + /** \returns the coefficient at given the given row and column. + * + * \sa operator()(Index,Index), operator[](Index) + */ + EIGEN_DEVICE_FUNC + EIGEN_STRONG_INLINE CoeffReturnType operator()(Index row, Index col) const + { + eigen_assert(row >= 0 && row < rows() && col >= 0 && col < cols()); + return coeff(row, col); + } + + /** Short version: don't use this function, use + * \link operator[](Index) const \endlink instead. + * + * Long version: this function is similar to + * \link operator[](Index) const \endlink, but without the assertion. + * Use this for limiting the performance cost of debugging code when doing + * repeated coefficient access. Only use this when it is guaranteed that the + * parameter \a index is in range. + * + * If EIGEN_INTERNAL_DEBUGGING is defined, an assertion will be made, making this + * function equivalent to \link operator[](Index) const \endlink. + * + * \sa operator[](Index) const, coeffRef(Index), coeff(Index,Index) const + */ + + EIGEN_DEVICE_FUNC + EIGEN_STRONG_INLINE CoeffReturnType coeff(Index index) const + { + EIGEN_STATIC_ASSERT(internal::evaluator::Flags & LinearAccessBit, + THIS_COEFFICIENT_ACCESSOR_TAKING_ONE_ACCESS_IS_ONLY_FOR_EXPRESSIONS_ALLOWING_LINEAR_ACCESS) + eigen_internal_assert(index >= 0 && index < size()); + return internal::evaluator(derived()).coeff(index); + } + + + /** \returns the coefficient at given index. + * + * This method is allowed only for vector expressions, and for matrix expressions having the LinearAccessBit. + * + * \sa operator[](Index), operator()(Index,Index) const, x() const, y() const, + * z() const, w() const + */ + + EIGEN_DEVICE_FUNC + EIGEN_STRONG_INLINE CoeffReturnType operator[](Index index) const + { + EIGEN_STATIC_ASSERT( + Derived::IsVectorAtCompileTime, THE_BRACKET_OPERATOR_IS_ONLY_FOR_VECTORS__USE_THE_PARENTHESIS_OPERATOR_INSTEAD) + eigen_assert(index >= 0 && index < size()); + return coeff(index); + } + + /** \returns the coefficient at given index. + * + * This is synonymous to operator[](Index) const. + * + * This method is allowed only for vector expressions, and for matrix expressions having the LinearAccessBit. + * + * \sa operator[](Index), operator()(Index,Index) const, x() const, y() const, + * z() const, w() const + */ + + EIGEN_DEVICE_FUNC + EIGEN_STRONG_INLINE CoeffReturnType operator()(Index index) const + { + eigen_assert(index >= 0 && index < size()); + return coeff(index); + } + + /** equivalent to operator[](0). */ + + EIGEN_DEVICE_FUNC + EIGEN_STRONG_INLINE CoeffReturnType x() const { return (*this)[0]; } + + /** equivalent to operator[](1). */ + + EIGEN_DEVICE_FUNC + EIGEN_STRONG_INLINE CoeffReturnType y() const + { + EIGEN_STATIC_ASSERT(Derived::SizeAtCompileTime == -1 || Derived::SizeAtCompileTime >= 2, OUT_OF_RANGE_ACCESS); + return (*this)[1]; + } + + /** equivalent to operator[](2). */ + + EIGEN_DEVICE_FUNC + EIGEN_STRONG_INLINE CoeffReturnType z() const + { + EIGEN_STATIC_ASSERT(Derived::SizeAtCompileTime == -1 || Derived::SizeAtCompileTime >= 3, OUT_OF_RANGE_ACCESS); + return (*this)[2]; + } + + /** equivalent to operator[](3). */ + + EIGEN_DEVICE_FUNC + EIGEN_STRONG_INLINE CoeffReturnType w() const + { + EIGEN_STATIC_ASSERT(Derived::SizeAtCompileTime == -1 || Derived::SizeAtCompileTime >= 4, OUT_OF_RANGE_ACCESS); + return (*this)[3]; + } + + /** \internal + * \returns the packet of coefficients starting at the given row and column. It is your responsibility + * to ensure that a packet really starts there. This method is only available on expressions having the + * PacketAccessBit. + * + * The \a LoadMode parameter may have the value \a #Aligned or \a #Unaligned. Its effect is to select + * the appropriate vectorization instruction. Aligned access is faster, but is only possible for packets + * starting at an address which is a multiple of the packet size. + */ + + template EIGEN_STRONG_INLINE PacketReturnType packet(Index row, Index col) const + { + typedef typename internal::packet_traits::type DefaultPacketType; + eigen_internal_assert(row >= 0 && row < rows() && col >= 0 && col < cols()); + return internal::evaluator(derived()).template packet(row, col); + } + + + /** \internal */ + template EIGEN_STRONG_INLINE PacketReturnType packetByOuterInner(Index outer, Index inner) const + { + return packet(rowIndexByOuterInner(outer, inner), colIndexByOuterInner(outer, inner)); + } + + /** \internal + * \returns the packet of coefficients starting at the given index. It is your responsibility + * to ensure that a packet really starts there. This method is only available on expressions having the + * PacketAccessBit and the LinearAccessBit. + * + * The \a LoadMode parameter may have the value \a #Aligned or \a #Unaligned. Its effect is to select + * the appropriate vectorization instruction. Aligned access is faster, but is only possible for packets + * starting at an address which is a multiple of the packet size. + */ + + template EIGEN_STRONG_INLINE PacketReturnType packet(Index index) const + { + EIGEN_STATIC_ASSERT(internal::evaluator::Flags & LinearAccessBit, + THIS_COEFFICIENT_ACCESSOR_TAKING_ONE_ACCESS_IS_ONLY_FOR_EXPRESSIONS_ALLOWING_LINEAR_ACCESS) + typedef typename internal::packet_traits::type DefaultPacketType; + eigen_internal_assert(index >= 0 && index < size()); + return internal::evaluator(derived()).template packet(index); + } + +protected: + // explanation: DenseBase is doing "using ..." on the methods from DenseCoeffsBase. + // But some methods are only available in the DirectAccess case. + // So we add dummy methods here with these names, so that "using... " doesn't fail. + // It's not private so that the child class DenseBase can access them, and it's not public + // either since it's an implementation detail, so has to be protected. + void coeffRef(); + void coeffRefByOuterInner(); + void writePacket(); + void writePacketByOuterInner(); + void copyCoeff(); + void copyCoeffByOuterInner(); + void copyPacket(); + void copyPacketByOuterInner(); + void stride(); + void innerStride(); + void outerStride(); + void rowStride(); + void colStride(); }; /** \brief Base class providing read/write coefficient access to matrices and arrays. - * \ingroup Core_Module - * \tparam Derived Type of the derived class - * \tparam #WriteAccessors Constant indicating read/write access - * - * This class defines the non-const \c operator() function and friends, which can be used to write specific - * entries of a matrix or array. This class inherits DenseCoeffsBase which - * defines the const variant for reading specific entries. - * - * \sa DenseCoeffsBase, \ref TopicClassHierarchy - */ + * \ingroup Core_Module + * \tparam Derived Type of the derived class + * \tparam #WriteAccessors Constant indicating read/write access + * + * This class defines the non-const \c operator() function and friends, which can be used to write specific + * entries of a matrix or array. This class inherits DenseCoeffsBase which + * defines the const variant for reading specific entries. + * + * \sa DenseCoeffsBase, \ref TopicClassHierarchy + */ template class DenseCoeffsBase : public DenseCoeffsBase { - public: - - typedef DenseCoeffsBase Base; - - typedef typename internal::traits::StorageKind StorageKind; - typedef typename internal::traits::Scalar Scalar; - typedef typename internal::packet_traits::type PacketScalar; - typedef typename NumTraits::Real RealScalar; - - using Base::coeff; - using Base::rows; - using Base::cols; - using Base::size; - using Base::derived; - using Base::rowIndexByOuterInner; - using Base::colIndexByOuterInner; - using Base::operator[]; - using Base::operator(); - using Base::x; - using Base::y; - using Base::z; - using Base::w; - - /** Short version: don't use this function, use - * \link operator()(Index,Index) \endlink instead. - * - * Long version: this function is similar to - * \link operator()(Index,Index) \endlink, but without the assertion. - * Use this for limiting the performance cost of debugging code when doing - * repeated coefficient access. Only use this when it is guaranteed that the - * parameters \a row and \a col are in range. - * - * If EIGEN_INTERNAL_DEBUGGING is defined, an assertion will be made, making this - * function equivalent to \link operator()(Index,Index) \endlink. - * - * \sa operator()(Index,Index), coeff(Index, Index) const, coeffRef(Index) - */ - EIGEN_DEVICE_FUNC - EIGEN_STRONG_INLINE Scalar& coeffRef(Index row, Index col) - { - eigen_internal_assert(row >= 0 && row < rows() - && col >= 0 && col < cols()); - return internal::evaluator(derived()).coeffRef(row,col); - } - - EIGEN_DEVICE_FUNC - EIGEN_STRONG_INLINE Scalar& - coeffRefByOuterInner(Index outer, Index inner) - { - return coeffRef(rowIndexByOuterInner(outer, inner), - colIndexByOuterInner(outer, inner)); - } - - /** \returns a reference to the coefficient at given the given row and column. - * - * \sa operator[](Index) - */ - - EIGEN_DEVICE_FUNC - EIGEN_STRONG_INLINE Scalar& - operator()(Index row, Index col) - { - eigen_assert(row >= 0 && row < rows() - && col >= 0 && col < cols()); - return coeffRef(row, col); - } - - - /** Short version: don't use this function, use - * \link operator[](Index) \endlink instead. - * - * Long version: this function is similar to - * \link operator[](Index) \endlink, but without the assertion. - * Use this for limiting the performance cost of debugging code when doing - * repeated coefficient access. Only use this when it is guaranteed that the - * parameters \a row and \a col are in range. - * - * If EIGEN_INTERNAL_DEBUGGING is defined, an assertion will be made, making this - * function equivalent to \link operator[](Index) \endlink. - * - * \sa operator[](Index), coeff(Index) const, coeffRef(Index,Index) - */ - - EIGEN_DEVICE_FUNC - EIGEN_STRONG_INLINE Scalar& - coeffRef(Index index) - { - EIGEN_STATIC_ASSERT(internal::evaluator::Flags & LinearAccessBit, - THIS_COEFFICIENT_ACCESSOR_TAKING_ONE_ACCESS_IS_ONLY_FOR_EXPRESSIONS_ALLOWING_LINEAR_ACCESS) - eigen_internal_assert(index >= 0 && index < size()); - return internal::evaluator(derived()).coeffRef(index); - } - - /** \returns a reference to the coefficient at given index. - * - * This method is allowed only for vector expressions, and for matrix expressions having the LinearAccessBit. - * - * \sa operator[](Index) const, operator()(Index,Index), x(), y(), z(), w() - */ - - EIGEN_DEVICE_FUNC - EIGEN_STRONG_INLINE Scalar& - operator[](Index index) - { - EIGEN_STATIC_ASSERT(Derived::IsVectorAtCompileTime, - THE_BRACKET_OPERATOR_IS_ONLY_FOR_VECTORS__USE_THE_PARENTHESIS_OPERATOR_INSTEAD) - eigen_assert(index >= 0 && index < size()); - return coeffRef(index); - } - - /** \returns a reference to the coefficient at given index. - * - * This is synonymous to operator[](Index). - * - * This method is allowed only for vector expressions, and for matrix expressions having the LinearAccessBit. - * - * \sa operator[](Index) const, operator()(Index,Index), x(), y(), z(), w() - */ - - EIGEN_DEVICE_FUNC - EIGEN_STRONG_INLINE Scalar& - operator()(Index index) - { - eigen_assert(index >= 0 && index < size()); - return coeffRef(index); - } - - /** equivalent to operator[](0). */ - - EIGEN_DEVICE_FUNC - EIGEN_STRONG_INLINE Scalar& - x() { return (*this)[0]; } - - /** equivalent to operator[](1). */ - - EIGEN_DEVICE_FUNC - EIGEN_STRONG_INLINE Scalar& - y() - { - EIGEN_STATIC_ASSERT(Derived::SizeAtCompileTime==-1 || Derived::SizeAtCompileTime>=2, OUT_OF_RANGE_ACCESS); - return (*this)[1]; - } - - /** equivalent to operator[](2). */ - - EIGEN_DEVICE_FUNC - EIGEN_STRONG_INLINE Scalar& - z() - { - EIGEN_STATIC_ASSERT(Derived::SizeAtCompileTime==-1 || Derived::SizeAtCompileTime>=3, OUT_OF_RANGE_ACCESS); - return (*this)[2]; - } - - /** equivalent to operator[](3). */ - - EIGEN_DEVICE_FUNC - EIGEN_STRONG_INLINE Scalar& - w() - { - EIGEN_STATIC_ASSERT(Derived::SizeAtCompileTime==-1 || Derived::SizeAtCompileTime>=4, OUT_OF_RANGE_ACCESS); - return (*this)[3]; - } +public: + typedef DenseCoeffsBase Base; + + typedef typename internal::traits::StorageKind StorageKind; + typedef typename internal::traits::Scalar Scalar; + typedef typename internal::packet_traits::type PacketScalar; + typedef typename NumTraits::Real RealScalar; + + using Base::coeff; + using Base::rows; + using Base::cols; + using Base::size; + using Base::derived; + using Base::rowIndexByOuterInner; + using Base::colIndexByOuterInner; + using Base::operator[]; + using Base::operator(); + using Base::x; + using Base::y; + using Base::z; + using Base::w; + + /** Short version: don't use this function, use + * \link operator()(Index,Index) \endlink instead. + * + * Long version: this function is similar to + * \link operator()(Index,Index) \endlink, but without the assertion. + * Use this for limiting the performance cost of debugging code when doing + * repeated coefficient access. Only use this when it is guaranteed that the + * parameters \a row and \a col are in range. + * + * If EIGEN_INTERNAL_DEBUGGING is defined, an assertion will be made, making this + * function equivalent to \link operator()(Index,Index) \endlink. + * + * \sa operator()(Index,Index), coeff(Index, Index) const, coeffRef(Index) + */ + EIGEN_DEVICE_FUNC + EIGEN_STRONG_INLINE Scalar &coeffRef(Index row, Index col) + { + eigen_internal_assert(row >= 0 && row < rows() && col >= 0 && col < cols()); + return internal::evaluator(derived()).coeffRef(row, col); + } + + EIGEN_DEVICE_FUNC + EIGEN_STRONG_INLINE Scalar &coeffRefByOuterInner(Index outer, Index inner) + { + return coeffRef(rowIndexByOuterInner(outer, inner), colIndexByOuterInner(outer, inner)); + } + + /** \returns a reference to the coefficient at given the given row and column. + * + * \sa operator[](Index) + */ + + EIGEN_DEVICE_FUNC + EIGEN_STRONG_INLINE Scalar &operator()(Index row, Index col) + { + eigen_assert(row >= 0 && row < rows() && col >= 0 && col < cols()); + return coeffRef(row, col); + } + + + /** Short version: don't use this function, use + * \link operator[](Index) \endlink instead. + * + * Long version: this function is similar to + * \link operator[](Index) \endlink, but without the assertion. + * Use this for limiting the performance cost of debugging code when doing + * repeated coefficient access. Only use this when it is guaranteed that the + * parameters \a row and \a col are in range. + * + * If EIGEN_INTERNAL_DEBUGGING is defined, an assertion will be made, making this + * function equivalent to \link operator[](Index) \endlink. + * + * \sa operator[](Index), coeff(Index) const, coeffRef(Index,Index) + */ + + EIGEN_DEVICE_FUNC + EIGEN_STRONG_INLINE Scalar &coeffRef(Index index) + { + EIGEN_STATIC_ASSERT(internal::evaluator::Flags & LinearAccessBit, + THIS_COEFFICIENT_ACCESSOR_TAKING_ONE_ACCESS_IS_ONLY_FOR_EXPRESSIONS_ALLOWING_LINEAR_ACCESS) + eigen_internal_assert(index >= 0 && index < size()); + return internal::evaluator(derived()).coeffRef(index); + } + + /** \returns a reference to the coefficient at given index. + * + * This method is allowed only for vector expressions, and for matrix expressions having the LinearAccessBit. + * + * \sa operator[](Index) const, operator()(Index,Index), x(), y(), z(), w() + */ + + EIGEN_DEVICE_FUNC + EIGEN_STRONG_INLINE Scalar &operator[](Index index) + { + EIGEN_STATIC_ASSERT( + Derived::IsVectorAtCompileTime, THE_BRACKET_OPERATOR_IS_ONLY_FOR_VECTORS__USE_THE_PARENTHESIS_OPERATOR_INSTEAD) + eigen_assert(index >= 0 && index < size()); + return coeffRef(index); + } + + /** \returns a reference to the coefficient at given index. + * + * This is synonymous to operator[](Index). + * + * This method is allowed only for vector expressions, and for matrix expressions having the LinearAccessBit. + * + * \sa operator[](Index) const, operator()(Index,Index), x(), y(), z(), w() + */ + + EIGEN_DEVICE_FUNC + EIGEN_STRONG_INLINE Scalar &operator()(Index index) + { + eigen_assert(index >= 0 && index < size()); + return coeffRef(index); + } + + /** equivalent to operator[](0). */ + + EIGEN_DEVICE_FUNC + EIGEN_STRONG_INLINE Scalar &x() { return (*this)[0]; } + + /** equivalent to operator[](1). */ + + EIGEN_DEVICE_FUNC + EIGEN_STRONG_INLINE Scalar &y() + { + EIGEN_STATIC_ASSERT(Derived::SizeAtCompileTime == -1 || Derived::SizeAtCompileTime >= 2, OUT_OF_RANGE_ACCESS); + return (*this)[1]; + } + + /** equivalent to operator[](2). */ + + EIGEN_DEVICE_FUNC + EIGEN_STRONG_INLINE Scalar &z() + { + EIGEN_STATIC_ASSERT(Derived::SizeAtCompileTime == -1 || Derived::SizeAtCompileTime >= 3, OUT_OF_RANGE_ACCESS); + return (*this)[2]; + } + + /** equivalent to operator[](3). */ + + EIGEN_DEVICE_FUNC + EIGEN_STRONG_INLINE Scalar &w() + { + EIGEN_STATIC_ASSERT(Derived::SizeAtCompileTime == -1 || Derived::SizeAtCompileTime >= 4, OUT_OF_RANGE_ACCESS); + return (*this)[3]; + } }; /** \brief Base class providing direct read-only coefficient access to matrices and arrays. - * \ingroup Core_Module - * \tparam Derived Type of the derived class - * \tparam #DirectAccessors Constant indicating direct access - * - * This class defines functions to work with strides which can be used to access entries directly. This class - * inherits DenseCoeffsBase which defines functions to access entries read-only using - * \c operator() . - * - * \sa \blank \ref TopicClassHierarchy - */ + * \ingroup Core_Module + * \tparam Derived Type of the derived class + * \tparam #DirectAccessors Constant indicating direct access + * + * This class defines functions to work with strides which can be used to access entries directly. This class + * inherits DenseCoeffsBase which defines functions to access entries read-only using + * \c operator() . + * + * \sa \blank \ref TopicClassHierarchy + */ template class DenseCoeffsBase : public DenseCoeffsBase { - public: - - typedef DenseCoeffsBase Base; - typedef typename internal::traits::Scalar Scalar; - typedef typename NumTraits::Real RealScalar; - - using Base::rows; - using Base::cols; - using Base::size; - using Base::derived; - - /** \returns the pointer increment between two consecutive elements within a slice in the inner direction. - * - * \sa outerStride(), rowStride(), colStride() - */ - EIGEN_DEVICE_FUNC - inline Index innerStride() const - { - return derived().innerStride(); - } - - /** \returns the pointer increment between two consecutive inner slices (for example, between two consecutive columns - * in a column-major matrix). - * - * \sa innerStride(), rowStride(), colStride() - */ - EIGEN_DEVICE_FUNC - inline Index outerStride() const - { - return derived().outerStride(); - } - - // FIXME shall we remove it ? - inline Index stride() const - { - return Derived::IsVectorAtCompileTime ? innerStride() : outerStride(); - } - - /** \returns the pointer increment between two consecutive rows. - * - * \sa innerStride(), outerStride(), colStride() - */ - EIGEN_DEVICE_FUNC - inline Index rowStride() const - { - return Derived::IsRowMajor ? outerStride() : innerStride(); - } - - /** \returns the pointer increment between two consecutive columns. - * - * \sa innerStride(), outerStride(), rowStride() - */ - EIGEN_DEVICE_FUNC - inline Index colStride() const - { - return Derived::IsRowMajor ? innerStride() : outerStride(); - } +public: + typedef DenseCoeffsBase Base; + typedef typename internal::traits::Scalar Scalar; + typedef typename NumTraits::Real RealScalar; + + using Base::rows; + using Base::cols; + using Base::size; + using Base::derived; + + /** \returns the pointer increment between two consecutive elements within a slice in the inner direction. + * + * \sa outerStride(), rowStride(), colStride() + */ + EIGEN_DEVICE_FUNC + inline Index innerStride() const { return derived().innerStride(); } + + /** \returns the pointer increment between two consecutive inner slices (for example, between two consecutive columns + * in a column-major matrix). + * + * \sa innerStride(), rowStride(), colStride() + */ + EIGEN_DEVICE_FUNC + inline Index outerStride() const { return derived().outerStride(); } + + // FIXME shall we remove it ? + inline Index stride() const { return Derived::IsVectorAtCompileTime ? innerStride() : outerStride(); } + + /** \returns the pointer increment between two consecutive rows. + * + * \sa innerStride(), outerStride(), colStride() + */ + EIGEN_DEVICE_FUNC + inline Index rowStride() const { return Derived::IsRowMajor ? outerStride() : innerStride(); } + + /** \returns the pointer increment between two consecutive columns. + * + * \sa innerStride(), outerStride(), rowStride() + */ + EIGEN_DEVICE_FUNC + inline Index colStride() const { return Derived::IsRowMajor ? innerStride() : outerStride(); } }; /** \brief Base class providing direct read/write coefficient access to matrices and arrays. - * \ingroup Core_Module - * \tparam Derived Type of the derived class - * \tparam #DirectWriteAccessors Constant indicating direct access - * - * This class defines functions to work with strides which can be used to access entries directly. This class - * inherits DenseCoeffsBase which defines functions to access entries read/write using - * \c operator(). - * - * \sa \blank \ref TopicClassHierarchy - */ + * \ingroup Core_Module + * \tparam Derived Type of the derived class + * \tparam #DirectWriteAccessors Constant indicating direct access + * + * This class defines functions to work with strides which can be used to access entries directly. This class + * inherits DenseCoeffsBase which defines functions to access entries read/write using + * \c operator(). + * + * \sa \blank \ref TopicClassHierarchy + */ template -class DenseCoeffsBase - : public DenseCoeffsBase +class DenseCoeffsBase : public DenseCoeffsBase { - public: - - typedef DenseCoeffsBase Base; - typedef typename internal::traits::Scalar Scalar; - typedef typename NumTraits::Real RealScalar; - - using Base::rows; - using Base::cols; - using Base::size; - using Base::derived; - - /** \returns the pointer increment between two consecutive elements within a slice in the inner direction. - * - * \sa outerStride(), rowStride(), colStride() - */ - EIGEN_DEVICE_FUNC - inline Index innerStride() const - { - return derived().innerStride(); - } - - /** \returns the pointer increment between two consecutive inner slices (for example, between two consecutive columns - * in a column-major matrix). - * - * \sa innerStride(), rowStride(), colStride() - */ - EIGEN_DEVICE_FUNC - inline Index outerStride() const - { - return derived().outerStride(); - } - - // FIXME shall we remove it ? - inline Index stride() const - { - return Derived::IsVectorAtCompileTime ? innerStride() : outerStride(); - } - - /** \returns the pointer increment between two consecutive rows. - * - * \sa innerStride(), outerStride(), colStride() - */ - EIGEN_DEVICE_FUNC - inline Index rowStride() const - { - return Derived::IsRowMajor ? outerStride() : innerStride(); - } - - /** \returns the pointer increment between two consecutive columns. - * - * \sa innerStride(), outerStride(), rowStride() - */ - EIGEN_DEVICE_FUNC - inline Index colStride() const - { - return Derived::IsRowMajor ? innerStride() : outerStride(); - } +public: + typedef DenseCoeffsBase Base; + typedef typename internal::traits::Scalar Scalar; + typedef typename NumTraits::Real RealScalar; + + using Base::rows; + using Base::cols; + using Base::size; + using Base::derived; + + /** \returns the pointer increment between two consecutive elements within a slice in the inner direction. + * + * \sa outerStride(), rowStride(), colStride() + */ + EIGEN_DEVICE_FUNC + inline Index innerStride() const { return derived().innerStride(); } + + /** \returns the pointer increment between two consecutive inner slices (for example, between two consecutive columns + * in a column-major matrix). + * + * \sa innerStride(), rowStride(), colStride() + */ + EIGEN_DEVICE_FUNC + inline Index outerStride() const { return derived().outerStride(); } + + // FIXME shall we remove it ? + inline Index stride() const { return Derived::IsVectorAtCompileTime ? innerStride() : outerStride(); } + + /** \returns the pointer increment between two consecutive rows. + * + * \sa innerStride(), outerStride(), colStride() + */ + EIGEN_DEVICE_FUNC + inline Index rowStride() const { return Derived::IsRowMajor ? outerStride() : innerStride(); } + + /** \returns the pointer increment between two consecutive columns. + * + * \sa innerStride(), outerStride(), rowStride() + */ + EIGEN_DEVICE_FUNC + inline Index colStride() const { return Derived::IsRowMajor ? innerStride() : outerStride(); } }; namespace internal { -template -struct first_aligned_impl -{ - static inline Index run(const Derived&) - { return 0; } -}; + template struct first_aligned_impl + { + static inline Index run(const Derived &) { return 0; } + }; -template -struct first_aligned_impl -{ - static inline Index run(const Derived& m) + template struct first_aligned_impl + { + static inline Index run(const Derived &m) { return internal::first_aligned(m.data(), m.size()); } + }; + + /** \internal \returns the index of the first element of the array stored by \a m that is properly aligned with + * respect to \a Alignment for vectorization. + * + * \tparam Alignment requested alignment in Bytes. + * + * There is also the variant first_aligned(const Scalar*, Integer) defined in Memory.h. See it for more + * documentation. + */ + template static inline Index first_aligned(const DenseBase &m) { - return internal::first_aligned(m.data(), m.size()); + enum { ReturnZero = (int(evaluator::Alignment) >= Alignment) || !(Derived::Flags & DirectAccessBit) }; + return first_aligned_impl::run(m.derived()); } -}; - -/** \internal \returns the index of the first element of the array stored by \a m that is properly aligned with respect to \a Alignment for vectorization. - * - * \tparam Alignment requested alignment in Bytes. - * - * There is also the variant first_aligned(const Scalar*, Integer) defined in Memory.h. See it for more - * documentation. - */ -template -static inline Index first_aligned(const DenseBase& m) -{ - enum { ReturnZero = (int(evaluator::Alignment) >= Alignment) || !(Derived::Flags & DirectAccessBit) }; - return first_aligned_impl::run(m.derived()); -} -template -static inline Index first_default_aligned(const DenseBase& m) -{ - typedef typename Derived::Scalar Scalar; - typedef typename packet_traits::type DefaultPacketType; - return internal::first_aligned::alignment),Derived>(m); -} + template static inline Index first_default_aligned(const DenseBase &m) + { + typedef typename Derived::Scalar Scalar; + typedef typename packet_traits::type DefaultPacketType; + return internal::first_aligned::alignment), Derived>(m); + } -template::ret> -struct inner_stride_at_compile_time -{ - enum { ret = traits::InnerStrideAtCompileTime }; -}; + template::ret> struct inner_stride_at_compile_time + { + enum { ret = traits::InnerStrideAtCompileTime }; + }; -template -struct inner_stride_at_compile_time -{ - enum { ret = 0 }; -}; + template struct inner_stride_at_compile_time + { + enum { ret = 0 }; + }; -template::ret> -struct outer_stride_at_compile_time -{ - enum { ret = traits::OuterStrideAtCompileTime }; -}; + template::ret> struct outer_stride_at_compile_time + { + enum { ret = traits::OuterStrideAtCompileTime }; + }; -template -struct outer_stride_at_compile_time -{ - enum { ret = 0 }; -}; + template struct outer_stride_at_compile_time + { + enum { ret = 0 }; + }; -} // end namespace internal +}// end namespace internal -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_DENSECOEFFSBASE_H +#endif// EIGEN_DENSECOEFFSBASE_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/DenseStorage.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/DenseStorage.h index 7958feeb..2a0bb7c7 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/DenseStorage.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/DenseStorage.h @@ -13,558 +13,593 @@ #define EIGEN_MATRIXSTORAGE_H #ifdef EIGEN_DENSE_STORAGE_CTOR_PLUGIN - #define EIGEN_INTERNAL_DENSE_STORAGE_CTOR_PLUGIN(X) X; EIGEN_DENSE_STORAGE_CTOR_PLUGIN; +#define EIGEN_INTERNAL_DENSE_STORAGE_CTOR_PLUGIN(X) \ + X; \ + EIGEN_DENSE_STORAGE_CTOR_PLUGIN; #else - #define EIGEN_INTERNAL_DENSE_STORAGE_CTOR_PLUGIN(X) +#define EIGEN_INTERNAL_DENSE_STORAGE_CTOR_PLUGIN(X) #endif namespace Eigen { namespace internal { -struct constructor_without_unaligned_array_assert {}; + struct constructor_without_unaligned_array_assert + { + }; -template -EIGEN_DEVICE_FUNC -void check_static_allocation_size() -{ - // if EIGEN_STACK_ALLOCATION_LIMIT is defined to 0, then no limit - #if EIGEN_STACK_ALLOCATION_LIMIT - EIGEN_STATIC_ASSERT(Size * sizeof(T) <= EIGEN_STACK_ALLOCATION_LIMIT, OBJECT_ALLOCATED_ON_STACK_IS_TOO_BIG); - #endif -} + template EIGEN_DEVICE_FUNC void check_static_allocation_size() + { +// if EIGEN_STACK_ALLOCATION_LIMIT is defined to 0, then no limit +#if EIGEN_STACK_ALLOCATION_LIMIT + EIGEN_STATIC_ASSERT(Size * sizeof(T) <= EIGEN_STACK_ALLOCATION_LIMIT, OBJECT_ALLOCATED_ON_STACK_IS_TOO_BIG); +#endif + } -/** \internal - * Static array. If the MatrixOrArrayOptions require auto-alignment, the array will be automatically aligned: - * to 16 bytes boundary if the total size is a multiple of 16 bytes. - */ -template ::value > -struct plain_array -{ - T array[Size]; + /** \internal + * Static array. If the MatrixOrArrayOptions require auto-alignment, the array will be automatically aligned: + * to 16 bytes boundary if the total size is a multiple of 16 bytes. + */ + template::value> + struct plain_array + { + T array[Size]; - EIGEN_DEVICE_FUNC - plain_array() - { - check_static_allocation_size(); - } + EIGEN_DEVICE_FUNC + plain_array() { check_static_allocation_size(); } - EIGEN_DEVICE_FUNC - plain_array(constructor_without_unaligned_array_assert) - { - check_static_allocation_size(); - } -}; + EIGEN_DEVICE_FUNC + plain_array(constructor_without_unaligned_array_assert) { check_static_allocation_size(); } + }; #if defined(EIGEN_DISABLE_UNALIGNED_ARRAY_ASSERT) - #define EIGEN_MAKE_UNALIGNED_ARRAY_ASSERT(sizemask) -#elif EIGEN_GNUC_AT_LEAST(4,7) - // GCC 4.7 is too aggressive in its optimizations and remove the alignement test based on the fact the array is declared to be aligned. - // See this bug report: http://gcc.gnu.org/bugzilla/show_bug.cgi?id=53900 - // Hiding the origin of the array pointer behind a function argument seems to do the trick even if the function is inlined: - template - EIGEN_ALWAYS_INLINE PtrType eigen_unaligned_array_assert_workaround_gcc47(PtrType array) { return array; } - #define EIGEN_MAKE_UNALIGNED_ARRAY_ASSERT(sizemask) \ - eigen_assert((internal::UIntPtr(eigen_unaligned_array_assert_workaround_gcc47(array)) & (sizemask)) == 0 \ +#define EIGEN_MAKE_UNALIGNED_ARRAY_ASSERT(sizemask) +#elif EIGEN_GNUC_AT_LEAST(4, 7) + // GCC 4.7 is too aggressive in its optimizations and remove the alignement test based on the fact the array is + // declared to be aligned. See this bug report: http://gcc.gnu.org/bugzilla/show_bug.cgi?id=53900 Hiding the origin of + // the array pointer behind a function argument seems to do the trick even if the function is inlined: + template EIGEN_ALWAYS_INLINE PtrType eigen_unaligned_array_assert_workaround_gcc47(PtrType array) + { + return array; + } +#define EIGEN_MAKE_UNALIGNED_ARRAY_ASSERT(sizemask) \ + eigen_assert((internal::UIntPtr(eigen_unaligned_array_assert_workaround_gcc47(array)) & (sizemask)) == 0 \ && "this assertion is explained here: " \ "http://eigen.tuxfamily.org/dox-devel/group__TopicUnalignedArrayAssert.html" \ " **** READ THIS WEB PAGE !!! ****"); #else - #define EIGEN_MAKE_UNALIGNED_ARRAY_ASSERT(sizemask) \ - eigen_assert((internal::UIntPtr(array) & (sizemask)) == 0 \ +#define EIGEN_MAKE_UNALIGNED_ARRAY_ASSERT(sizemask) \ + eigen_assert((internal::UIntPtr(array) & (sizemask)) == 0 \ && "this assertion is explained here: " \ "http://eigen.tuxfamily.org/dox-devel/group__TopicUnalignedArrayAssert.html" \ " **** READ THIS WEB PAGE !!! ****"); #endif -template -struct plain_array -{ - EIGEN_ALIGN_TO_BOUNDARY(8) T array[Size]; - - EIGEN_DEVICE_FUNC - plain_array() + template struct plain_array { - EIGEN_MAKE_UNALIGNED_ARRAY_ASSERT(7); - check_static_allocation_size(); - } + EIGEN_ALIGN_TO_BOUNDARY(8) T array[Size]; - EIGEN_DEVICE_FUNC - plain_array(constructor_without_unaligned_array_assert) - { - check_static_allocation_size(); - } -}; + EIGEN_DEVICE_FUNC + plain_array() + { + EIGEN_MAKE_UNALIGNED_ARRAY_ASSERT(7); + check_static_allocation_size(); + } -template -struct plain_array -{ - EIGEN_ALIGN_TO_BOUNDARY(16) T array[Size]; + EIGEN_DEVICE_FUNC + plain_array(constructor_without_unaligned_array_assert) { check_static_allocation_size(); } + }; - EIGEN_DEVICE_FUNC - plain_array() - { - EIGEN_MAKE_UNALIGNED_ARRAY_ASSERT(15); - check_static_allocation_size(); - } + template struct plain_array + { + EIGEN_ALIGN_TO_BOUNDARY(16) T array[Size]; - EIGEN_DEVICE_FUNC - plain_array(constructor_without_unaligned_array_assert) - { - check_static_allocation_size(); - } -}; + EIGEN_DEVICE_FUNC + plain_array() + { + EIGEN_MAKE_UNALIGNED_ARRAY_ASSERT(15); + check_static_allocation_size(); + } -template -struct plain_array -{ - EIGEN_ALIGN_TO_BOUNDARY(32) T array[Size]; + EIGEN_DEVICE_FUNC + plain_array(constructor_without_unaligned_array_assert) { check_static_allocation_size(); } + }; - EIGEN_DEVICE_FUNC - plain_array() + template struct plain_array { - EIGEN_MAKE_UNALIGNED_ARRAY_ASSERT(31); - check_static_allocation_size(); - } + EIGEN_ALIGN_TO_BOUNDARY(32) T array[Size]; - EIGEN_DEVICE_FUNC - plain_array(constructor_without_unaligned_array_assert) - { - check_static_allocation_size(); - } -}; + EIGEN_DEVICE_FUNC + plain_array() + { + EIGEN_MAKE_UNALIGNED_ARRAY_ASSERT(31); + check_static_allocation_size(); + } -template -struct plain_array -{ - EIGEN_ALIGN_TO_BOUNDARY(64) T array[Size]; + EIGEN_DEVICE_FUNC + plain_array(constructor_without_unaligned_array_assert) { check_static_allocation_size(); } + }; - EIGEN_DEVICE_FUNC - plain_array() - { - EIGEN_MAKE_UNALIGNED_ARRAY_ASSERT(63); - check_static_allocation_size(); - } + template struct plain_array + { + EIGEN_ALIGN_TO_BOUNDARY(64) T array[Size]; - EIGEN_DEVICE_FUNC - plain_array(constructor_without_unaligned_array_assert) - { - check_static_allocation_size(); - } -}; + EIGEN_DEVICE_FUNC + plain_array() + { + EIGEN_MAKE_UNALIGNED_ARRAY_ASSERT(63); + check_static_allocation_size(); + } -template -struct plain_array -{ - T array[1]; - EIGEN_DEVICE_FUNC plain_array() {} - EIGEN_DEVICE_FUNC plain_array(constructor_without_unaligned_array_assert) {} -}; + EIGEN_DEVICE_FUNC + plain_array(constructor_without_unaligned_array_assert) { check_static_allocation_size(); } + }; -} // end namespace internal + template + struct plain_array + { + T array[1]; + EIGEN_DEVICE_FUNC plain_array() {} + EIGEN_DEVICE_FUNC plain_array(constructor_without_unaligned_array_assert) {} + }; + +}// end namespace internal /** \internal - * - * \class DenseStorage - * \ingroup Core_Module - * - * \brief Stores the data of a matrix - * - * This class stores the data of fixed-size, dynamic-size or mixed matrices - * in a way as compact as possible. - * - * \sa Matrix - */ + * + * \class DenseStorage + * \ingroup Core_Module + * + * \brief Stores the data of a matrix + * + * This class stores the data of fixed-size, dynamic-size or mixed matrices + * in a way as compact as possible. + * + * \sa Matrix + */ template class DenseStorage; // purely fixed-size matrix template class DenseStorage { - internal::plain_array m_data; - public: - EIGEN_DEVICE_FUNC DenseStorage() { - EIGEN_INTERNAL_DENSE_STORAGE_CTOR_PLUGIN(Index size = Size) - } - EIGEN_DEVICE_FUNC + internal::plain_array m_data; + +public: + EIGEN_DEVICE_FUNC DenseStorage(){ EIGEN_INTERNAL_DENSE_STORAGE_CTOR_PLUGIN(Index size = Size) } EIGEN_DEVICE_FUNC explicit DenseStorage(internal::constructor_without_unaligned_array_assert) - : m_data(internal::constructor_without_unaligned_array_assert()) {} - EIGEN_DEVICE_FUNC - DenseStorage(const DenseStorage& other) : m_data(other.m_data) { - EIGEN_INTERNAL_DENSE_STORAGE_CTOR_PLUGIN(Index size = Size) - } - EIGEN_DEVICE_FUNC - DenseStorage& operator=(const DenseStorage& other) - { - if (this != &other) m_data = other.m_data; - return *this; - } - EIGEN_DEVICE_FUNC DenseStorage(Index size, Index rows, Index cols) { - EIGEN_INTERNAL_DENSE_STORAGE_CTOR_PLUGIN({}) - eigen_internal_assert(size==rows*cols && rows==_Rows && cols==_Cols); - EIGEN_UNUSED_VARIABLE(size); - EIGEN_UNUSED_VARIABLE(rows); - EIGEN_UNUSED_VARIABLE(cols); - } - EIGEN_DEVICE_FUNC void swap(DenseStorage& other) { std::swap(m_data,other.m_data); } - EIGEN_DEVICE_FUNC static Index rows(void) {return _Rows;} - EIGEN_DEVICE_FUNC static Index cols(void) {return _Cols;} - EIGEN_DEVICE_FUNC void conservativeResize(Index,Index,Index) {} - EIGEN_DEVICE_FUNC void resize(Index,Index,Index) {} - EIGEN_DEVICE_FUNC const T *data() const { return m_data.array; } - EIGEN_DEVICE_FUNC T *data() { return m_data.array; } + : m_data(internal::constructor_without_unaligned_array_assert()) + {} + EIGEN_DEVICE_FUNC + DenseStorage(const DenseStorage &other) + : m_data(other.m_data){ EIGEN_INTERNAL_DENSE_STORAGE_CTOR_PLUGIN(Index size = Size) } EIGEN_DEVICE_FUNC DenseStorage + & operator=(const DenseStorage & other) + { + if (this != &other) m_data = other.m_data; + return *this; + } + EIGEN_DEVICE_FUNC DenseStorage(Index size, Index rows, Index cols) + { + EIGEN_INTERNAL_DENSE_STORAGE_CTOR_PLUGIN({}) + eigen_internal_assert(size == rows * cols && rows == _Rows && cols == _Cols); + EIGEN_UNUSED_VARIABLE(size); + EIGEN_UNUSED_VARIABLE(rows); + EIGEN_UNUSED_VARIABLE(cols); + } + EIGEN_DEVICE_FUNC void swap(DenseStorage &other) { std::swap(m_data, other.m_data); } + EIGEN_DEVICE_FUNC static Index rows(void) { return _Rows; } + EIGEN_DEVICE_FUNC static Index cols(void) { return _Cols; } + EIGEN_DEVICE_FUNC void conservativeResize(Index, Index, Index) {} + EIGEN_DEVICE_FUNC void resize(Index, Index, Index) {} + EIGEN_DEVICE_FUNC const T *data() const { return m_data.array; } + EIGEN_DEVICE_FUNC T *data() { return m_data.array; } }; // null matrix template class DenseStorage { - public: - EIGEN_DEVICE_FUNC DenseStorage() {} - EIGEN_DEVICE_FUNC explicit DenseStorage(internal::constructor_without_unaligned_array_assert) {} - EIGEN_DEVICE_FUNC DenseStorage(const DenseStorage&) {} - EIGEN_DEVICE_FUNC DenseStorage& operator=(const DenseStorage&) { return *this; } - EIGEN_DEVICE_FUNC DenseStorage(Index,Index,Index) {} - EIGEN_DEVICE_FUNC void swap(DenseStorage& ) {} - EIGEN_DEVICE_FUNC static Index rows(void) {return _Rows;} - EIGEN_DEVICE_FUNC static Index cols(void) {return _Cols;} - EIGEN_DEVICE_FUNC void conservativeResize(Index,Index,Index) {} - EIGEN_DEVICE_FUNC void resize(Index,Index,Index) {} - EIGEN_DEVICE_FUNC const T *data() const { return 0; } - EIGEN_DEVICE_FUNC T *data() { return 0; } +public: + EIGEN_DEVICE_FUNC DenseStorage() {} + EIGEN_DEVICE_FUNC explicit DenseStorage(internal::constructor_without_unaligned_array_assert) {} + EIGEN_DEVICE_FUNC DenseStorage(const DenseStorage &) {} + EIGEN_DEVICE_FUNC DenseStorage &operator=(const DenseStorage &) { return *this; } + EIGEN_DEVICE_FUNC DenseStorage(Index, Index, Index) {} + EIGEN_DEVICE_FUNC void swap(DenseStorage &) {} + EIGEN_DEVICE_FUNC static Index rows(void) { return _Rows; } + EIGEN_DEVICE_FUNC static Index cols(void) { return _Cols; } + EIGEN_DEVICE_FUNC void conservativeResize(Index, Index, Index) {} + EIGEN_DEVICE_FUNC void resize(Index, Index, Index) {} + EIGEN_DEVICE_FUNC const T *data() const { return 0; } + EIGEN_DEVICE_FUNC T *data() { return 0; } }; // more specializations for null matrices; these are necessary to resolve ambiguities -template class DenseStorage -: public DenseStorage { }; +template +class DenseStorage : public DenseStorage +{ +}; -template class DenseStorage -: public DenseStorage { }; +template +class DenseStorage : public DenseStorage +{ +}; -template class DenseStorage -: public DenseStorage { }; +template +class DenseStorage : public DenseStorage +{ +}; // dynamic-size matrix with fixed-size storage template class DenseStorage { - internal::plain_array m_data; - Index m_rows; - Index m_cols; - public: - EIGEN_DEVICE_FUNC DenseStorage() : m_rows(0), m_cols(0) {} - EIGEN_DEVICE_FUNC explicit DenseStorage(internal::constructor_without_unaligned_array_assert) - : m_data(internal::constructor_without_unaligned_array_assert()), m_rows(0), m_cols(0) {} - EIGEN_DEVICE_FUNC DenseStorage(const DenseStorage& other) : m_data(other.m_data), m_rows(other.m_rows), m_cols(other.m_cols) {} - EIGEN_DEVICE_FUNC DenseStorage& operator=(const DenseStorage& other) - { - if (this != &other) - { - m_data = other.m_data; - m_rows = other.m_rows; - m_cols = other.m_cols; - } - return *this; + internal::plain_array m_data; + Index m_rows; + Index m_cols; + +public: + EIGEN_DEVICE_FUNC DenseStorage() : m_rows(0), m_cols(0) {} + EIGEN_DEVICE_FUNC explicit DenseStorage(internal::constructor_without_unaligned_array_assert) + : m_data(internal::constructor_without_unaligned_array_assert()), m_rows(0), m_cols(0) + {} + EIGEN_DEVICE_FUNC DenseStorage(const DenseStorage &other) + : m_data(other.m_data), m_rows(other.m_rows), m_cols(other.m_cols) + {} + EIGEN_DEVICE_FUNC DenseStorage &operator=(const DenseStorage &other) + { + if (this != &other) { + m_data = other.m_data; + m_rows = other.m_rows; + m_cols = other.m_cols; } - EIGEN_DEVICE_FUNC DenseStorage(Index, Index rows, Index cols) : m_rows(rows), m_cols(cols) {} - EIGEN_DEVICE_FUNC void swap(DenseStorage& other) - { std::swap(m_data,other.m_data); std::swap(m_rows,other.m_rows); std::swap(m_cols,other.m_cols); } - EIGEN_DEVICE_FUNC Index rows() const {return m_rows;} - EIGEN_DEVICE_FUNC Index cols() const {return m_cols;} - EIGEN_DEVICE_FUNC void conservativeResize(Index, Index rows, Index cols) { m_rows = rows; m_cols = cols; } - EIGEN_DEVICE_FUNC void resize(Index, Index rows, Index cols) { m_rows = rows; m_cols = cols; } - EIGEN_DEVICE_FUNC const T *data() const { return m_data.array; } - EIGEN_DEVICE_FUNC T *data() { return m_data.array; } + return *this; + } + EIGEN_DEVICE_FUNC DenseStorage(Index, Index rows, Index cols) : m_rows(rows), m_cols(cols) {} + EIGEN_DEVICE_FUNC void swap(DenseStorage &other) + { + std::swap(m_data, other.m_data); + std::swap(m_rows, other.m_rows); + std::swap(m_cols, other.m_cols); + } + EIGEN_DEVICE_FUNC Index rows() const { return m_rows; } + EIGEN_DEVICE_FUNC Index cols() const { return m_cols; } + EIGEN_DEVICE_FUNC void conservativeResize(Index, Index rows, Index cols) + { + m_rows = rows; + m_cols = cols; + } + EIGEN_DEVICE_FUNC void resize(Index, Index rows, Index cols) + { + m_rows = rows; + m_cols = cols; + } + EIGEN_DEVICE_FUNC const T *data() const { return m_data.array; } + EIGEN_DEVICE_FUNC T *data() { return m_data.array; } }; // dynamic-size matrix with fixed-size storage and fixed width template class DenseStorage { - internal::plain_array m_data; - Index m_rows; - public: - EIGEN_DEVICE_FUNC DenseStorage() : m_rows(0) {} - EIGEN_DEVICE_FUNC explicit DenseStorage(internal::constructor_without_unaligned_array_assert) - : m_data(internal::constructor_without_unaligned_array_assert()), m_rows(0) {} - EIGEN_DEVICE_FUNC DenseStorage(const DenseStorage& other) : m_data(other.m_data), m_rows(other.m_rows) {} - EIGEN_DEVICE_FUNC DenseStorage& operator=(const DenseStorage& other) - { - if (this != &other) - { - m_data = other.m_data; - m_rows = other.m_rows; - } - return *this; + internal::plain_array m_data; + Index m_rows; + +public: + EIGEN_DEVICE_FUNC DenseStorage() : m_rows(0) {} + EIGEN_DEVICE_FUNC explicit DenseStorage(internal::constructor_without_unaligned_array_assert) + : m_data(internal::constructor_without_unaligned_array_assert()), m_rows(0) + {} + EIGEN_DEVICE_FUNC DenseStorage(const DenseStorage &other) : m_data(other.m_data), m_rows(other.m_rows) {} + EIGEN_DEVICE_FUNC DenseStorage &operator=(const DenseStorage &other) + { + if (this != &other) { + m_data = other.m_data; + m_rows = other.m_rows; } - EIGEN_DEVICE_FUNC DenseStorage(Index, Index rows, Index) : m_rows(rows) {} - EIGEN_DEVICE_FUNC void swap(DenseStorage& other) { std::swap(m_data,other.m_data); std::swap(m_rows,other.m_rows); } - EIGEN_DEVICE_FUNC Index rows(void) const {return m_rows;} - EIGEN_DEVICE_FUNC Index cols(void) const {return _Cols;} - EIGEN_DEVICE_FUNC void conservativeResize(Index, Index rows, Index) { m_rows = rows; } - EIGEN_DEVICE_FUNC void resize(Index, Index rows, Index) { m_rows = rows; } - EIGEN_DEVICE_FUNC const T *data() const { return m_data.array; } - EIGEN_DEVICE_FUNC T *data() { return m_data.array; } + return *this; + } + EIGEN_DEVICE_FUNC DenseStorage(Index, Index rows, Index) : m_rows(rows) {} + EIGEN_DEVICE_FUNC void swap(DenseStorage &other) + { + std::swap(m_data, other.m_data); + std::swap(m_rows, other.m_rows); + } + EIGEN_DEVICE_FUNC Index rows(void) const { return m_rows; } + EIGEN_DEVICE_FUNC Index cols(void) const { return _Cols; } + EIGEN_DEVICE_FUNC void conservativeResize(Index, Index rows, Index) { m_rows = rows; } + EIGEN_DEVICE_FUNC void resize(Index, Index rows, Index) { m_rows = rows; } + EIGEN_DEVICE_FUNC const T *data() const { return m_data.array; } + EIGEN_DEVICE_FUNC T *data() { return m_data.array; } }; // dynamic-size matrix with fixed-size storage and fixed height template class DenseStorage { - internal::plain_array m_data; - Index m_cols; - public: - EIGEN_DEVICE_FUNC DenseStorage() : m_cols(0) {} - EIGEN_DEVICE_FUNC explicit DenseStorage(internal::constructor_without_unaligned_array_assert) - : m_data(internal::constructor_without_unaligned_array_assert()), m_cols(0) {} - EIGEN_DEVICE_FUNC DenseStorage(const DenseStorage& other) : m_data(other.m_data), m_cols(other.m_cols) {} - EIGEN_DEVICE_FUNC DenseStorage& operator=(const DenseStorage& other) - { - if (this != &other) - { - m_data = other.m_data; - m_cols = other.m_cols; - } - return *this; + internal::plain_array m_data; + Index m_cols; + +public: + EIGEN_DEVICE_FUNC DenseStorage() : m_cols(0) {} + EIGEN_DEVICE_FUNC explicit DenseStorage(internal::constructor_without_unaligned_array_assert) + : m_data(internal::constructor_without_unaligned_array_assert()), m_cols(0) + {} + EIGEN_DEVICE_FUNC DenseStorage(const DenseStorage &other) : m_data(other.m_data), m_cols(other.m_cols) {} + EIGEN_DEVICE_FUNC DenseStorage &operator=(const DenseStorage &other) + { + if (this != &other) { + m_data = other.m_data; + m_cols = other.m_cols; } - EIGEN_DEVICE_FUNC DenseStorage(Index, Index, Index cols) : m_cols(cols) {} - EIGEN_DEVICE_FUNC void swap(DenseStorage& other) { std::swap(m_data,other.m_data); std::swap(m_cols,other.m_cols); } - EIGEN_DEVICE_FUNC Index rows(void) const {return _Rows;} - EIGEN_DEVICE_FUNC Index cols(void) const {return m_cols;} - void conservativeResize(Index, Index, Index cols) { m_cols = cols; } - void resize(Index, Index, Index cols) { m_cols = cols; } - EIGEN_DEVICE_FUNC const T *data() const { return m_data.array; } - EIGEN_DEVICE_FUNC T *data() { return m_data.array; } + return *this; + } + EIGEN_DEVICE_FUNC DenseStorage(Index, Index, Index cols) : m_cols(cols) {} + EIGEN_DEVICE_FUNC void swap(DenseStorage &other) + { + std::swap(m_data, other.m_data); + std::swap(m_cols, other.m_cols); + } + EIGEN_DEVICE_FUNC Index rows(void) const { return _Rows; } + EIGEN_DEVICE_FUNC Index cols(void) const { return m_cols; } + void conservativeResize(Index, Index, Index cols) { m_cols = cols; } + void resize(Index, Index, Index cols) { m_cols = cols; } + EIGEN_DEVICE_FUNC const T *data() const { return m_data.array; } + EIGEN_DEVICE_FUNC T *data() { return m_data.array; } }; // purely dynamic matrix. template class DenseStorage { - T *m_data; - Index m_rows; - Index m_cols; - public: - EIGEN_DEVICE_FUNC DenseStorage() : m_data(0), m_rows(0), m_cols(0) {} - EIGEN_DEVICE_FUNC explicit DenseStorage(internal::constructor_without_unaligned_array_assert) - : m_data(0), m_rows(0), m_cols(0) {} - EIGEN_DEVICE_FUNC DenseStorage(Index size, Index rows, Index cols) - : m_data(internal::conditional_aligned_new_auto(size)), m_rows(rows), m_cols(cols) - { - EIGEN_INTERNAL_DENSE_STORAGE_CTOR_PLUGIN({}) - eigen_internal_assert(size==rows*cols && rows>=0 && cols >=0); - } - EIGEN_DEVICE_FUNC DenseStorage(const DenseStorage& other) - : m_data(internal::conditional_aligned_new_auto(other.m_rows*other.m_cols)) - , m_rows(other.m_rows) - , m_cols(other.m_cols) - { - EIGEN_INTERNAL_DENSE_STORAGE_CTOR_PLUGIN(Index size = m_rows*m_cols) - internal::smart_copy(other.m_data, other.m_data+other.m_rows*other.m_cols, m_data); - } - EIGEN_DEVICE_FUNC DenseStorage& operator=(const DenseStorage& other) - { - if (this != &other) - { - DenseStorage tmp(other); - this->swap(tmp); - } - return *this; + T *m_data; + Index m_rows; + Index m_cols; + +public: + EIGEN_DEVICE_FUNC DenseStorage() : m_data(0), m_rows(0), m_cols(0) {} + EIGEN_DEVICE_FUNC explicit DenseStorage(internal::constructor_without_unaligned_array_assert) + : m_data(0), m_rows(0), m_cols(0) + {} + EIGEN_DEVICE_FUNC DenseStorage(Index size, Index rows, Index cols) + : m_data(internal::conditional_aligned_new_auto(size)), m_rows(rows), m_cols(cols) + { + EIGEN_INTERNAL_DENSE_STORAGE_CTOR_PLUGIN({}) + eigen_internal_assert(size == rows * cols && rows >= 0 && cols >= 0); + } + EIGEN_DEVICE_FUNC DenseStorage(const DenseStorage &other) + : m_data(internal::conditional_aligned_new_auto(other.m_rows * other.m_cols)), + m_rows(other.m_rows), m_cols(other.m_cols) + { + EIGEN_INTERNAL_DENSE_STORAGE_CTOR_PLUGIN(Index size = m_rows * m_cols) + internal::smart_copy(other.m_data, other.m_data + other.m_rows * other.m_cols, m_data); + } + EIGEN_DEVICE_FUNC DenseStorage &operator=(const DenseStorage &other) + { + if (this != &other) { + DenseStorage tmp(other); + this->swap(tmp); } + return *this; + } #if EIGEN_HAS_RVALUE_REFERENCES - EIGEN_DEVICE_FUNC - DenseStorage(DenseStorage&& other) EIGEN_NOEXCEPT - : m_data(std::move(other.m_data)) - , m_rows(std::move(other.m_rows)) - , m_cols(std::move(other.m_cols)) - { - other.m_data = nullptr; - other.m_rows = 0; - other.m_cols = 0; - } - EIGEN_DEVICE_FUNC - DenseStorage& operator=(DenseStorage&& other) EIGEN_NOEXCEPT - { - using std::swap; - swap(m_data, other.m_data); - swap(m_rows, other.m_rows); - swap(m_cols, other.m_cols); - return *this; - } + EIGEN_DEVICE_FUNC + DenseStorage(DenseStorage &&other) EIGEN_NOEXCEPT + : m_data(std::move(other.m_data)) + , m_rows(std::move(other.m_rows)) + , m_cols(std::move(other.m_cols)) + { + other.m_data = nullptr; + other.m_rows = 0; + other.m_cols = 0; + } + EIGEN_DEVICE_FUNC + DenseStorage &operator=(DenseStorage &&other) EIGEN_NOEXCEPT + { + using std::swap; + swap(m_data, other.m_data); + swap(m_rows, other.m_rows); + swap(m_cols, other.m_cols); + return *this; + } #endif - EIGEN_DEVICE_FUNC ~DenseStorage() { internal::conditional_aligned_delete_auto(m_data, m_rows*m_cols); } - EIGEN_DEVICE_FUNC void swap(DenseStorage& other) - { std::swap(m_data,other.m_data); std::swap(m_rows,other.m_rows); std::swap(m_cols,other.m_cols); } - EIGEN_DEVICE_FUNC Index rows(void) const {return m_rows;} - EIGEN_DEVICE_FUNC Index cols(void) const {return m_cols;} - void conservativeResize(Index size, Index rows, Index cols) - { - m_data = internal::conditional_aligned_realloc_new_auto(m_data, size, m_rows*m_cols); - m_rows = rows; - m_cols = cols; - } - EIGEN_DEVICE_FUNC void resize(Index size, Index rows, Index cols) - { - if(size != m_rows*m_cols) - { - internal::conditional_aligned_delete_auto(m_data, m_rows*m_cols); - if (size) - m_data = internal::conditional_aligned_new_auto(size); - else - m_data = 0; - EIGEN_INTERNAL_DENSE_STORAGE_CTOR_PLUGIN({}) - } - m_rows = rows; - m_cols = cols; + EIGEN_DEVICE_FUNC ~DenseStorage() + { + internal::conditional_aligned_delete_auto(m_data, m_rows * m_cols); + } + EIGEN_DEVICE_FUNC void swap(DenseStorage &other) + { + std::swap(m_data, other.m_data); + std::swap(m_rows, other.m_rows); + std::swap(m_cols, other.m_cols); + } + EIGEN_DEVICE_FUNC Index rows(void) const { return m_rows; } + EIGEN_DEVICE_FUNC Index cols(void) const { return m_cols; } + void conservativeResize(Index size, Index rows, Index cols) + { + m_data = + internal::conditional_aligned_realloc_new_auto(m_data, size, m_rows * m_cols); + m_rows = rows; + m_cols = cols; + } + EIGEN_DEVICE_FUNC void resize(Index size, Index rows, Index cols) + { + if (size != m_rows * m_cols) { + internal::conditional_aligned_delete_auto(m_data, m_rows * m_cols); + if (size) + m_data = internal::conditional_aligned_new_auto(size); + else + m_data = 0; + EIGEN_INTERNAL_DENSE_STORAGE_CTOR_PLUGIN({}) } - EIGEN_DEVICE_FUNC const T *data() const { return m_data; } - EIGEN_DEVICE_FUNC T *data() { return m_data; } + m_rows = rows; + m_cols = cols; + } + EIGEN_DEVICE_FUNC const T *data() const { return m_data; } + EIGEN_DEVICE_FUNC T *data() { return m_data; } }; // matrix with dynamic width and fixed height (so that matrix has dynamic size). template class DenseStorage { - T *m_data; - Index m_cols; - public: - EIGEN_DEVICE_FUNC DenseStorage() : m_data(0), m_cols(0) {} - explicit DenseStorage(internal::constructor_without_unaligned_array_assert) : m_data(0), m_cols(0) {} - EIGEN_DEVICE_FUNC DenseStorage(Index size, Index rows, Index cols) : m_data(internal::conditional_aligned_new_auto(size)), m_cols(cols) - { - EIGEN_INTERNAL_DENSE_STORAGE_CTOR_PLUGIN({}) - eigen_internal_assert(size==rows*cols && rows==_Rows && cols >=0); - EIGEN_UNUSED_VARIABLE(rows); - } - EIGEN_DEVICE_FUNC DenseStorage(const DenseStorage& other) - : m_data(internal::conditional_aligned_new_auto(_Rows*other.m_cols)) - , m_cols(other.m_cols) - { - EIGEN_INTERNAL_DENSE_STORAGE_CTOR_PLUGIN(Index size = m_cols*_Rows) - internal::smart_copy(other.m_data, other.m_data+_Rows*m_cols, m_data); + T *m_data; + Index m_cols; + +public: + EIGEN_DEVICE_FUNC DenseStorage() : m_data(0), m_cols(0) {} + explicit DenseStorage(internal::constructor_without_unaligned_array_assert) : m_data(0), m_cols(0) {} + EIGEN_DEVICE_FUNC DenseStorage(Index size, Index rows, Index cols) + : m_data(internal::conditional_aligned_new_auto(size)), m_cols(cols) + { + EIGEN_INTERNAL_DENSE_STORAGE_CTOR_PLUGIN({}) + eigen_internal_assert(size == rows * cols && rows == _Rows && cols >= 0); + EIGEN_UNUSED_VARIABLE(rows); + } + EIGEN_DEVICE_FUNC DenseStorage(const DenseStorage &other) + : m_data(internal::conditional_aligned_new_auto(_Rows * other.m_cols)), + m_cols(other.m_cols) + { + EIGEN_INTERNAL_DENSE_STORAGE_CTOR_PLUGIN(Index size = m_cols * _Rows) + internal::smart_copy(other.m_data, other.m_data + _Rows * m_cols, m_data); + } + EIGEN_DEVICE_FUNC DenseStorage &operator=(const DenseStorage &other) + { + if (this != &other) { + DenseStorage tmp(other); + this->swap(tmp); } - EIGEN_DEVICE_FUNC DenseStorage& operator=(const DenseStorage& other) - { - if (this != &other) - { - DenseStorage tmp(other); - this->swap(tmp); - } - return *this; - } + return *this; + } #if EIGEN_HAS_RVALUE_REFERENCES - EIGEN_DEVICE_FUNC - DenseStorage(DenseStorage&& other) EIGEN_NOEXCEPT - : m_data(std::move(other.m_data)) - , m_cols(std::move(other.m_cols)) - { - other.m_data = nullptr; - other.m_cols = 0; - } - EIGEN_DEVICE_FUNC - DenseStorage& operator=(DenseStorage&& other) EIGEN_NOEXCEPT - { - using std::swap; - swap(m_data, other.m_data); - swap(m_cols, other.m_cols); - return *this; - } + EIGEN_DEVICE_FUNC + DenseStorage(DenseStorage &&other) EIGEN_NOEXCEPT + : m_data(std::move(other.m_data)) + , m_cols(std::move(other.m_cols)) + { + other.m_data = nullptr; + other.m_cols = 0; + } + EIGEN_DEVICE_FUNC + DenseStorage &operator=(DenseStorage &&other) EIGEN_NOEXCEPT + { + using std::swap; + swap(m_data, other.m_data); + swap(m_cols, other.m_cols); + return *this; + } #endif - EIGEN_DEVICE_FUNC ~DenseStorage() { internal::conditional_aligned_delete_auto(m_data, _Rows*m_cols); } - EIGEN_DEVICE_FUNC void swap(DenseStorage& other) { std::swap(m_data,other.m_data); std::swap(m_cols,other.m_cols); } - EIGEN_DEVICE_FUNC static Index rows(void) {return _Rows;} - EIGEN_DEVICE_FUNC Index cols(void) const {return m_cols;} - EIGEN_DEVICE_FUNC void conservativeResize(Index size, Index, Index cols) - { - m_data = internal::conditional_aligned_realloc_new_auto(m_data, size, _Rows*m_cols); - m_cols = cols; - } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void resize(Index size, Index, Index cols) - { - if(size != _Rows*m_cols) - { - internal::conditional_aligned_delete_auto(m_data, _Rows*m_cols); - if (size) - m_data = internal::conditional_aligned_new_auto(size); - else - m_data = 0; - EIGEN_INTERNAL_DENSE_STORAGE_CTOR_PLUGIN({}) - } - m_cols = cols; + EIGEN_DEVICE_FUNC ~DenseStorage() + { + internal::conditional_aligned_delete_auto(m_data, _Rows * m_cols); + } + EIGEN_DEVICE_FUNC void swap(DenseStorage &other) + { + std::swap(m_data, other.m_data); + std::swap(m_cols, other.m_cols); + } + EIGEN_DEVICE_FUNC static Index rows(void) { return _Rows; } + EIGEN_DEVICE_FUNC Index cols(void) const { return m_cols; } + EIGEN_DEVICE_FUNC void conservativeResize(Index size, Index, Index cols) + { + m_data = + internal::conditional_aligned_realloc_new_auto(m_data, size, _Rows * m_cols); + m_cols = cols; + } + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void resize(Index size, Index, Index cols) + { + if (size != _Rows * m_cols) { + internal::conditional_aligned_delete_auto(m_data, _Rows * m_cols); + if (size) + m_data = internal::conditional_aligned_new_auto(size); + else + m_data = 0; + EIGEN_INTERNAL_DENSE_STORAGE_CTOR_PLUGIN({}) } - EIGEN_DEVICE_FUNC const T *data() const { return m_data; } - EIGEN_DEVICE_FUNC T *data() { return m_data; } + m_cols = cols; + } + EIGEN_DEVICE_FUNC const T *data() const { return m_data; } + EIGEN_DEVICE_FUNC T *data() { return m_data; } }; // matrix with dynamic height and fixed width (so that matrix has dynamic size). template class DenseStorage { - T *m_data; - Index m_rows; - public: - EIGEN_DEVICE_FUNC DenseStorage() : m_data(0), m_rows(0) {} - explicit DenseStorage(internal::constructor_without_unaligned_array_assert) : m_data(0), m_rows(0) {} - EIGEN_DEVICE_FUNC DenseStorage(Index size, Index rows, Index cols) : m_data(internal::conditional_aligned_new_auto(size)), m_rows(rows) - { - EIGEN_INTERNAL_DENSE_STORAGE_CTOR_PLUGIN({}) - eigen_internal_assert(size==rows*cols && rows>=0 && cols == _Cols); - EIGEN_UNUSED_VARIABLE(cols); - } - EIGEN_DEVICE_FUNC DenseStorage(const DenseStorage& other) - : m_data(internal::conditional_aligned_new_auto(other.m_rows*_Cols)) - , m_rows(other.m_rows) - { - EIGEN_INTERNAL_DENSE_STORAGE_CTOR_PLUGIN(Index size = m_rows*_Cols) - internal::smart_copy(other.m_data, other.m_data+other.m_rows*_Cols, m_data); + T *m_data; + Index m_rows; + +public: + EIGEN_DEVICE_FUNC DenseStorage() : m_data(0), m_rows(0) {} + explicit DenseStorage(internal::constructor_without_unaligned_array_assert) : m_data(0), m_rows(0) {} + EIGEN_DEVICE_FUNC DenseStorage(Index size, Index rows, Index cols) + : m_data(internal::conditional_aligned_new_auto(size)), m_rows(rows) + { + EIGEN_INTERNAL_DENSE_STORAGE_CTOR_PLUGIN({}) + eigen_internal_assert(size == rows * cols && rows >= 0 && cols == _Cols); + EIGEN_UNUSED_VARIABLE(cols); + } + EIGEN_DEVICE_FUNC DenseStorage(const DenseStorage &other) + : m_data(internal::conditional_aligned_new_auto(other.m_rows * _Cols)), + m_rows(other.m_rows) + { + EIGEN_INTERNAL_DENSE_STORAGE_CTOR_PLUGIN(Index size = m_rows * _Cols) + internal::smart_copy(other.m_data, other.m_data + other.m_rows * _Cols, m_data); + } + EIGEN_DEVICE_FUNC DenseStorage &operator=(const DenseStorage &other) + { + if (this != &other) { + DenseStorage tmp(other); + this->swap(tmp); } - EIGEN_DEVICE_FUNC DenseStorage& operator=(const DenseStorage& other) - { - if (this != &other) - { - DenseStorage tmp(other); - this->swap(tmp); - } - return *this; - } + return *this; + } #if EIGEN_HAS_RVALUE_REFERENCES - EIGEN_DEVICE_FUNC - DenseStorage(DenseStorage&& other) EIGEN_NOEXCEPT - : m_data(std::move(other.m_data)) - , m_rows(std::move(other.m_rows)) - { - other.m_data = nullptr; - other.m_rows = 0; - } - EIGEN_DEVICE_FUNC - DenseStorage& operator=(DenseStorage&& other) EIGEN_NOEXCEPT - { - using std::swap; - swap(m_data, other.m_data); - swap(m_rows, other.m_rows); - return *this; - } + EIGEN_DEVICE_FUNC + DenseStorage(DenseStorage &&other) EIGEN_NOEXCEPT + : m_data(std::move(other.m_data)) + , m_rows(std::move(other.m_rows)) + { + other.m_data = nullptr; + other.m_rows = 0; + } + EIGEN_DEVICE_FUNC + DenseStorage &operator=(DenseStorage &&other) EIGEN_NOEXCEPT + { + using std::swap; + swap(m_data, other.m_data); + swap(m_rows, other.m_rows); + return *this; + } #endif - EIGEN_DEVICE_FUNC ~DenseStorage() { internal::conditional_aligned_delete_auto(m_data, _Cols*m_rows); } - EIGEN_DEVICE_FUNC void swap(DenseStorage& other) { std::swap(m_data,other.m_data); std::swap(m_rows,other.m_rows); } - EIGEN_DEVICE_FUNC Index rows(void) const {return m_rows;} - EIGEN_DEVICE_FUNC static Index cols(void) {return _Cols;} - void conservativeResize(Index size, Index rows, Index) - { - m_data = internal::conditional_aligned_realloc_new_auto(m_data, size, m_rows*_Cols); - m_rows = rows; - } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void resize(Index size, Index rows, Index) - { - if(size != m_rows*_Cols) - { - internal::conditional_aligned_delete_auto(m_data, _Cols*m_rows); - if (size) - m_data = internal::conditional_aligned_new_auto(size); - else - m_data = 0; - EIGEN_INTERNAL_DENSE_STORAGE_CTOR_PLUGIN({}) - } - m_rows = rows; + EIGEN_DEVICE_FUNC ~DenseStorage() + { + internal::conditional_aligned_delete_auto(m_data, _Cols * m_rows); + } + EIGEN_DEVICE_FUNC void swap(DenseStorage &other) + { + std::swap(m_data, other.m_data); + std::swap(m_rows, other.m_rows); + } + EIGEN_DEVICE_FUNC Index rows(void) const { return m_rows; } + EIGEN_DEVICE_FUNC static Index cols(void) { return _Cols; } + void conservativeResize(Index size, Index rows, Index) + { + m_data = + internal::conditional_aligned_realloc_new_auto(m_data, size, m_rows * _Cols); + m_rows = rows; + } + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void resize(Index size, Index rows, Index) + { + if (size != m_rows * _Cols) { + internal::conditional_aligned_delete_auto(m_data, _Cols * m_rows); + if (size) + m_data = internal::conditional_aligned_new_auto(size); + else + m_data = 0; + EIGEN_INTERNAL_DENSE_STORAGE_CTOR_PLUGIN({}) } - EIGEN_DEVICE_FUNC const T *data() const { return m_data; } - EIGEN_DEVICE_FUNC T *data() { return m_data; } + m_rows = rows; + } + EIGEN_DEVICE_FUNC const T *data() const { return m_data; } + EIGEN_DEVICE_FUNC T *data() { return m_data; } }; -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_MATRIX_H +#endif// EIGEN_MATRIX_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/Diagonal.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/Diagonal.h index afcaf357..7592fc30 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/Diagonal.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/Diagonal.h @@ -11,237 +11,210 @@ #ifndef EIGEN_DIAGONAL_H #define EIGEN_DIAGONAL_H -namespace Eigen { +namespace Eigen { /** \class Diagonal - * \ingroup Core_Module - * - * \brief Expression of a diagonal/subdiagonal/superdiagonal in a matrix - * - * \param MatrixType the type of the object in which we are taking a sub/main/super diagonal - * \param DiagIndex the index of the sub/super diagonal. The default is 0 and it means the main diagonal. - * A positive value means a superdiagonal, a negative value means a subdiagonal. - * You can also use DynamicIndex so the index can be set at runtime. - * - * The matrix is not required to be square. - * - * This class represents an expression of the main diagonal, or any sub/super diagonal - * of a square matrix. It is the return type of MatrixBase::diagonal() and MatrixBase::diagonal(Index) and most of the - * time this is the only way it is used. - * - * \sa MatrixBase::diagonal(), MatrixBase::diagonal(Index) - */ + * \ingroup Core_Module + * + * \brief Expression of a diagonal/subdiagonal/superdiagonal in a matrix + * + * \param MatrixType the type of the object in which we are taking a sub/main/super diagonal + * \param DiagIndex the index of the sub/super diagonal. The default is 0 and it means the main diagonal. + * A positive value means a superdiagonal, a negative value means a subdiagonal. + * You can also use DynamicIndex so the index can be set at runtime. + * + * The matrix is not required to be square. + * + * This class represents an expression of the main diagonal, or any sub/super diagonal + * of a square matrix. It is the return type of MatrixBase::diagonal() and MatrixBase::diagonal(Index) and most of the + * time this is the only way it is used. + * + * \sa MatrixBase::diagonal(), MatrixBase::diagonal(Index) + */ namespace internal { -template -struct traits > - : traits -{ - typedef typename ref_selector::type MatrixTypeNested; - typedef typename remove_reference::type _MatrixTypeNested; - typedef typename MatrixType::StorageKind StorageKind; - enum { - RowsAtCompileTime = (int(DiagIndex) == DynamicIndex || int(MatrixType::SizeAtCompileTime) == Dynamic) ? Dynamic - : (EIGEN_PLAIN_ENUM_MIN(MatrixType::RowsAtCompileTime - EIGEN_PLAIN_ENUM_MAX(-DiagIndex, 0), - MatrixType::ColsAtCompileTime - EIGEN_PLAIN_ENUM_MAX( DiagIndex, 0))), - ColsAtCompileTime = 1, - MaxRowsAtCompileTime = int(MatrixType::MaxSizeAtCompileTime) == Dynamic ? Dynamic - : DiagIndex == DynamicIndex ? EIGEN_SIZE_MIN_PREFER_FIXED(MatrixType::MaxRowsAtCompileTime, - MatrixType::MaxColsAtCompileTime) - : (EIGEN_PLAIN_ENUM_MIN(MatrixType::MaxRowsAtCompileTime - EIGEN_PLAIN_ENUM_MAX(-DiagIndex, 0), - MatrixType::MaxColsAtCompileTime - EIGEN_PLAIN_ENUM_MAX( DiagIndex, 0))), - MaxColsAtCompileTime = 1, - MaskLvalueBit = is_lvalue::value ? LvalueBit : 0, - Flags = (unsigned int)_MatrixTypeNested::Flags & (RowMajorBit | MaskLvalueBit | DirectAccessBit) & ~RowMajorBit, // FIXME DirectAccessBit should not be handled by expressions - MatrixTypeOuterStride = outer_stride_at_compile_time::ret, - InnerStrideAtCompileTime = MatrixTypeOuterStride == Dynamic ? Dynamic : MatrixTypeOuterStride+1, - OuterStrideAtCompileTime = 0 + template struct traits> : traits + { + typedef typename ref_selector::type MatrixTypeNested; + typedef typename remove_reference::type _MatrixTypeNested; + typedef typename MatrixType::StorageKind StorageKind; + enum { + RowsAtCompileTime = (int(DiagIndex) == DynamicIndex || int(MatrixType::SizeAtCompileTime) == Dynamic) + ? Dynamic + : (EIGEN_PLAIN_ENUM_MIN(MatrixType::RowsAtCompileTime - EIGEN_PLAIN_ENUM_MAX(-DiagIndex, 0), + MatrixType::ColsAtCompileTime - EIGEN_PLAIN_ENUM_MAX(DiagIndex, 0))), + ColsAtCompileTime = 1, + MaxRowsAtCompileTime = + int(MatrixType::MaxSizeAtCompileTime) == Dynamic ? Dynamic + : DiagIndex == DynamicIndex + ? EIGEN_SIZE_MIN_PREFER_FIXED(MatrixType::MaxRowsAtCompileTime, MatrixType::MaxColsAtCompileTime) + : (EIGEN_PLAIN_ENUM_MIN(MatrixType::MaxRowsAtCompileTime - EIGEN_PLAIN_ENUM_MAX(-DiagIndex, 0), + MatrixType::MaxColsAtCompileTime - EIGEN_PLAIN_ENUM_MAX(DiagIndex, 0))), + MaxColsAtCompileTime = 1, + MaskLvalueBit = is_lvalue::value ? LvalueBit : 0, + Flags = (unsigned int)_MatrixTypeNested::Flags & (RowMajorBit | MaskLvalueBit | DirectAccessBit) + & ~RowMajorBit,// FIXME DirectAccessBit should not be handled by expressions + MatrixTypeOuterStride = outer_stride_at_compile_time::ret, + InnerStrideAtCompileTime = MatrixTypeOuterStride == Dynamic ? Dynamic : MatrixTypeOuterStride + 1, + OuterStrideAtCompileTime = 0 + }; }; -}; -} +}// namespace internal -template class Diagonal - : public internal::dense_xpr_base< Diagonal >::type +template +class Diagonal : public internal::dense_xpr_base>::type { - public: - - enum { DiagIndex = _DiagIndex }; - typedef typename internal::dense_xpr_base::type Base; - EIGEN_DENSE_PUBLIC_INTERFACE(Diagonal) - - EIGEN_DEVICE_FUNC - explicit inline Diagonal(MatrixType& matrix, Index a_index = DiagIndex) : m_matrix(matrix), m_index(a_index) - { - eigen_assert( a_index <= m_matrix.cols() && -a_index <= m_matrix.rows() ); - } - - EIGEN_INHERIT_ASSIGNMENT_OPERATORS(Diagonal) - - EIGEN_DEVICE_FUNC - inline Index rows() const - { - return m_index.value()<0 ? numext::mini(m_matrix.cols(),m_matrix.rows()+m_index.value()) - : numext::mini(m_matrix.rows(),m_matrix.cols()-m_index.value()); - } - - EIGEN_DEVICE_FUNC - inline Index cols() const { return 1; } - - EIGEN_DEVICE_FUNC - inline Index innerStride() const - { - return m_matrix.outerStride() + 1; - } - - EIGEN_DEVICE_FUNC - inline Index outerStride() const - { - return 0; - } - - typedef typename internal::conditional< - internal::is_lvalue::value, - Scalar, - const Scalar - >::type ScalarWithConstIfNotLvalue; - - EIGEN_DEVICE_FUNC - inline ScalarWithConstIfNotLvalue* data() { return &(m_matrix.coeffRef(rowOffset(), colOffset())); } - EIGEN_DEVICE_FUNC - inline const Scalar* data() const { return &(m_matrix.coeffRef(rowOffset(), colOffset())); } - - EIGEN_DEVICE_FUNC - inline Scalar& coeffRef(Index row, Index) - { - EIGEN_STATIC_ASSERT_LVALUE(MatrixType) - return m_matrix.coeffRef(row+rowOffset(), row+colOffset()); - } - - EIGEN_DEVICE_FUNC - inline const Scalar& coeffRef(Index row, Index) const - { - return m_matrix.coeffRef(row+rowOffset(), row+colOffset()); - } - - EIGEN_DEVICE_FUNC - inline CoeffReturnType coeff(Index row, Index) const - { - return m_matrix.coeff(row+rowOffset(), row+colOffset()); - } - - EIGEN_DEVICE_FUNC - inline Scalar& coeffRef(Index idx) - { - EIGEN_STATIC_ASSERT_LVALUE(MatrixType) - return m_matrix.coeffRef(idx+rowOffset(), idx+colOffset()); - } - - EIGEN_DEVICE_FUNC - inline const Scalar& coeffRef(Index idx) const - { - return m_matrix.coeffRef(idx+rowOffset(), idx+colOffset()); - } - - EIGEN_DEVICE_FUNC - inline CoeffReturnType coeff(Index idx) const - { - return m_matrix.coeff(idx+rowOffset(), idx+colOffset()); - } - - EIGEN_DEVICE_FUNC - inline const typename internal::remove_all::type& - nestedExpression() const - { - return m_matrix; - } - - EIGEN_DEVICE_FUNC - inline Index index() const - { - return m_index.value(); - } - - protected: - typename internal::ref_selector::non_const_type m_matrix; - const internal::variable_if_dynamicindex m_index; - - private: - // some compilers may fail to optimize std::max etc in case of compile-time constants... - EIGEN_DEVICE_FUNC - EIGEN_STRONG_INLINE Index absDiagIndex() const { return m_index.value()>0 ? m_index.value() : -m_index.value(); } - EIGEN_DEVICE_FUNC - EIGEN_STRONG_INLINE Index rowOffset() const { return m_index.value()>0 ? 0 : -m_index.value(); } - EIGEN_DEVICE_FUNC - EIGEN_STRONG_INLINE Index colOffset() const { return m_index.value()>0 ? m_index.value() : 0; } - // trigger a compile-time error if someone try to call packet - template typename MatrixType::PacketReturnType packet(Index) const; - template typename MatrixType::PacketReturnType packet(Index,Index) const; +public: + enum { DiagIndex = _DiagIndex }; + typedef typename internal::dense_xpr_base::type Base; + EIGEN_DENSE_PUBLIC_INTERFACE(Diagonal) + + EIGEN_DEVICE_FUNC + explicit inline Diagonal(MatrixType &matrix, Index a_index = DiagIndex) : m_matrix(matrix), m_index(a_index) + { + eigen_assert(a_index <= m_matrix.cols() && -a_index <= m_matrix.rows()); + } + + EIGEN_INHERIT_ASSIGNMENT_OPERATORS(Diagonal) + + EIGEN_DEVICE_FUNC + inline Index rows() const + { + return m_index.value() < 0 ? numext::mini(m_matrix.cols(), m_matrix.rows() + m_index.value()) + : numext::mini(m_matrix.rows(), m_matrix.cols() - m_index.value()); + } + + EIGEN_DEVICE_FUNC + inline Index cols() const { return 1; } + + EIGEN_DEVICE_FUNC + inline Index innerStride() const { return m_matrix.outerStride() + 1; } + + EIGEN_DEVICE_FUNC + inline Index outerStride() const { return 0; } + + typedef typename internal::conditional::value, Scalar, const Scalar>::type + ScalarWithConstIfNotLvalue; + + EIGEN_DEVICE_FUNC + inline ScalarWithConstIfNotLvalue *data() { return &(m_matrix.coeffRef(rowOffset(), colOffset())); } + EIGEN_DEVICE_FUNC + inline const Scalar *data() const { return &(m_matrix.coeffRef(rowOffset(), colOffset())); } + + EIGEN_DEVICE_FUNC + inline Scalar &coeffRef(Index row, Index) + { + EIGEN_STATIC_ASSERT_LVALUE(MatrixType) + return m_matrix.coeffRef(row + rowOffset(), row + colOffset()); + } + + EIGEN_DEVICE_FUNC + inline const Scalar &coeffRef(Index row, Index) const + { + return m_matrix.coeffRef(row + rowOffset(), row + colOffset()); + } + + EIGEN_DEVICE_FUNC + inline CoeffReturnType coeff(Index row, Index) const { return m_matrix.coeff(row + rowOffset(), row + colOffset()); } + + EIGEN_DEVICE_FUNC + inline Scalar &coeffRef(Index idx) + { + EIGEN_STATIC_ASSERT_LVALUE(MatrixType) + return m_matrix.coeffRef(idx + rowOffset(), idx + colOffset()); + } + + EIGEN_DEVICE_FUNC + inline const Scalar &coeffRef(Index idx) const { return m_matrix.coeffRef(idx + rowOffset(), idx + colOffset()); } + + EIGEN_DEVICE_FUNC + inline CoeffReturnType coeff(Index idx) const { return m_matrix.coeff(idx + rowOffset(), idx + colOffset()); } + + EIGEN_DEVICE_FUNC + inline const typename internal::remove_all::type &nestedExpression() const + { + return m_matrix; + } + + EIGEN_DEVICE_FUNC + inline Index index() const { return m_index.value(); } + +protected: + typename internal::ref_selector::non_const_type m_matrix; + const internal::variable_if_dynamicindex m_index; + +private: + // some compilers may fail to optimize std::max etc in case of compile-time constants... + EIGEN_DEVICE_FUNC + EIGEN_STRONG_INLINE Index absDiagIndex() const { return m_index.value() > 0 ? m_index.value() : -m_index.value(); } + EIGEN_DEVICE_FUNC + EIGEN_STRONG_INLINE Index rowOffset() const { return m_index.value() > 0 ? 0 : -m_index.value(); } + EIGEN_DEVICE_FUNC + EIGEN_STRONG_INLINE Index colOffset() const { return m_index.value() > 0 ? m_index.value() : 0; } + // trigger a compile-time error if someone try to call packet + template typename MatrixType::PacketReturnType packet(Index) const; + template typename MatrixType::PacketReturnType packet(Index, Index) const; }; /** \returns an expression of the main diagonal of the matrix \c *this - * - * \c *this is not required to be square. - * - * Example: \include MatrixBase_diagonal.cpp - * Output: \verbinclude MatrixBase_diagonal.out - * - * \sa class Diagonal */ -template -inline typename MatrixBase::DiagonalReturnType -MatrixBase::diagonal() + * + * \c *this is not required to be square. + * + * Example: \include MatrixBase_diagonal.cpp + * Output: \verbinclude MatrixBase_diagonal.out + * + * \sa class Diagonal */ +template inline typename MatrixBase::DiagonalReturnType MatrixBase::diagonal() { return DiagonalReturnType(derived()); } /** This is the const version of diagonal(). */ template -inline typename MatrixBase::ConstDiagonalReturnType -MatrixBase::diagonal() const +inline typename MatrixBase::ConstDiagonalReturnType MatrixBase::diagonal() const { return ConstDiagonalReturnType(derived()); } /** \returns an expression of the \a DiagIndex-th sub or super diagonal of the matrix \c *this - * - * \c *this is not required to be square. - * - * The template parameter \a DiagIndex represent a super diagonal if \a DiagIndex > 0 - * and a sub diagonal otherwise. \a DiagIndex == 0 is equivalent to the main diagonal. - * - * Example: \include MatrixBase_diagonal_int.cpp - * Output: \verbinclude MatrixBase_diagonal_int.out - * - * \sa MatrixBase::diagonal(), class Diagonal */ + * + * \c *this is not required to be square. + * + * The template parameter \a DiagIndex represent a super diagonal if \a DiagIndex > 0 + * and a sub diagonal otherwise. \a DiagIndex == 0 is equivalent to the main diagonal. + * + * Example: \include MatrixBase_diagonal_int.cpp + * Output: \verbinclude MatrixBase_diagonal_int.out + * + * \sa MatrixBase::diagonal(), class Diagonal */ template -inline typename MatrixBase::DiagonalDynamicIndexReturnType -MatrixBase::diagonal(Index index) +inline typename MatrixBase::DiagonalDynamicIndexReturnType MatrixBase::diagonal(Index index) { return DiagonalDynamicIndexReturnType(derived(), index); } /** This is the const version of diagonal(Index). */ template -inline typename MatrixBase::ConstDiagonalDynamicIndexReturnType -MatrixBase::diagonal(Index index) const +inline typename MatrixBase::ConstDiagonalDynamicIndexReturnType MatrixBase::diagonal( + Index index) const { return ConstDiagonalDynamicIndexReturnType(derived(), index); } /** \returns an expression of the \a DiagIndex-th sub or super diagonal of the matrix \c *this - * - * \c *this is not required to be square. - * - * The template parameter \a DiagIndex represent a super diagonal if \a DiagIndex > 0 - * and a sub diagonal otherwise. \a DiagIndex == 0 is equivalent to the main diagonal. - * - * Example: \include MatrixBase_diagonal_template_int.cpp - * Output: \verbinclude MatrixBase_diagonal_template_int.out - * - * \sa MatrixBase::diagonal(), class Diagonal */ + * + * \c *this is not required to be square. + * + * The template parameter \a DiagIndex represent a super diagonal if \a DiagIndex > 0 + * and a sub diagonal otherwise. \a DiagIndex == 0 is equivalent to the main diagonal. + * + * Example: \include MatrixBase_diagonal_template_int.cpp + * Output: \verbinclude MatrixBase_diagonal_template_int.out + * + * \sa MatrixBase::diagonal(), class Diagonal */ template template -inline typename MatrixBase::template DiagonalIndexReturnType::Type -MatrixBase::diagonal() +inline typename MatrixBase::template DiagonalIndexReturnType::Type MatrixBase::diagonal() { return typename DiagonalIndexReturnType::Type(derived()); } @@ -250,11 +223,11 @@ MatrixBase::diagonal() template template inline typename MatrixBase::template ConstDiagonalIndexReturnType::Type -MatrixBase::diagonal() const + MatrixBase::diagonal() const { return typename ConstDiagonalIndexReturnType::Type(derived()); } -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_DIAGONAL_H +#endif// EIGEN_DIAGONAL_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/DiagonalMatrix.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/DiagonalMatrix.h index ecfdce8e..127504d6 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/DiagonalMatrix.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/DiagonalMatrix.h @@ -11,333 +11,336 @@ #ifndef EIGEN_DIAGONALMATRIX_H #define EIGEN_DIAGONALMATRIX_H -namespace Eigen { +namespace Eigen { #ifndef EIGEN_PARSED_BY_DOXYGEN -template -class DiagonalBase : public EigenBase +template class DiagonalBase : public EigenBase { - public: - typedef typename internal::traits::DiagonalVectorType DiagonalVectorType; - typedef typename DiagonalVectorType::Scalar Scalar; - typedef typename DiagonalVectorType::RealScalar RealScalar; - typedef typename internal::traits::StorageKind StorageKind; - typedef typename internal::traits::StorageIndex StorageIndex; +public: + typedef typename internal::traits::DiagonalVectorType DiagonalVectorType; + typedef typename DiagonalVectorType::Scalar Scalar; + typedef typename DiagonalVectorType::RealScalar RealScalar; + typedef typename internal::traits::StorageKind StorageKind; + typedef typename internal::traits::StorageIndex StorageIndex; - enum { - RowsAtCompileTime = DiagonalVectorType::SizeAtCompileTime, - ColsAtCompileTime = DiagonalVectorType::SizeAtCompileTime, - MaxRowsAtCompileTime = DiagonalVectorType::MaxSizeAtCompileTime, - MaxColsAtCompileTime = DiagonalVectorType::MaxSizeAtCompileTime, - IsVectorAtCompileTime = 0, - Flags = NoPreferredStorageOrderBit - }; + enum { + RowsAtCompileTime = DiagonalVectorType::SizeAtCompileTime, + ColsAtCompileTime = DiagonalVectorType::SizeAtCompileTime, + MaxRowsAtCompileTime = DiagonalVectorType::MaxSizeAtCompileTime, + MaxColsAtCompileTime = DiagonalVectorType::MaxSizeAtCompileTime, + IsVectorAtCompileTime = 0, + Flags = NoPreferredStorageOrderBit + }; - typedef Matrix DenseMatrixType; - typedef DenseMatrixType DenseType; - typedef DiagonalMatrix PlainObject; - - EIGEN_DEVICE_FUNC - inline const Derived& derived() const { return *static_cast(this); } - EIGEN_DEVICE_FUNC - inline Derived& derived() { return *static_cast(this); } - - EIGEN_DEVICE_FUNC - DenseMatrixType toDenseMatrix() const { return derived(); } - - EIGEN_DEVICE_FUNC - inline const DiagonalVectorType& diagonal() const { return derived().diagonal(); } - EIGEN_DEVICE_FUNC - inline DiagonalVectorType& diagonal() { return derived().diagonal(); } - - EIGEN_DEVICE_FUNC - inline Index rows() const { return diagonal().size(); } - EIGEN_DEVICE_FUNC - inline Index cols() const { return diagonal().size(); } - - template - EIGEN_DEVICE_FUNC - const Product - operator*(const MatrixBase &matrix) const - { - return Product(derived(),matrix.derived()); - } + typedef Matrix + DenseMatrixType; + typedef DenseMatrixType DenseType; + typedef DiagonalMatrix + PlainObject; + + EIGEN_DEVICE_FUNC + inline const Derived &derived() const { return *static_cast(this); } + EIGEN_DEVICE_FUNC + inline Derived &derived() { return *static_cast(this); } + + EIGEN_DEVICE_FUNC + DenseMatrixType toDenseMatrix() const { return derived(); } + + EIGEN_DEVICE_FUNC + inline const DiagonalVectorType &diagonal() const { return derived().diagonal(); } + EIGEN_DEVICE_FUNC + inline DiagonalVectorType &diagonal() { return derived().diagonal(); } + + EIGEN_DEVICE_FUNC + inline Index rows() const { return diagonal().size(); } + EIGEN_DEVICE_FUNC + inline Index cols() const { return diagonal().size(); } + + template + EIGEN_DEVICE_FUNC const Product operator*( + const MatrixBase &matrix) const + { + return Product(derived(), matrix.derived()); + } - typedef DiagonalWrapper, const DiagonalVectorType> > InverseReturnType; - EIGEN_DEVICE_FUNC - inline const InverseReturnType - inverse() const - { - return InverseReturnType(diagonal().cwiseInverse()); - } - - EIGEN_DEVICE_FUNC - inline const DiagonalWrapper - operator*(const Scalar& scalar) const - { - return DiagonalWrapper(diagonal() * scalar); - } - EIGEN_DEVICE_FUNC - friend inline const DiagonalWrapper - operator*(const Scalar& scalar, const DiagonalBase& other) - { - return DiagonalWrapper(scalar * other.diagonal()); - } + typedef DiagonalWrapper, const DiagonalVectorType>> + InverseReturnType; + EIGEN_DEVICE_FUNC + inline const InverseReturnType inverse() const { return InverseReturnType(diagonal().cwiseInverse()); } + + EIGEN_DEVICE_FUNC + inline const DiagonalWrapper + operator*(const Scalar &scalar) const + { + return DiagonalWrapper( + diagonal() * scalar); + } + EIGEN_DEVICE_FUNC + friend inline const DiagonalWrapper + operator*(const Scalar &scalar, const DiagonalBase &other) + { + return DiagonalWrapper( + scalar * other.diagonal()); + } }; #endif /** \class DiagonalMatrix - * \ingroup Core_Module - * - * \brief Represents a diagonal matrix with its storage - * - * \param _Scalar the type of coefficients - * \param SizeAtCompileTime the dimension of the matrix, or Dynamic - * \param MaxSizeAtCompileTime the dimension of the matrix, or Dynamic. This parameter is optional and defaults - * to SizeAtCompileTime. Most of the time, you do not need to specify it. - * - * \sa class DiagonalWrapper - */ + * \ingroup Core_Module + * + * \brief Represents a diagonal matrix with its storage + * + * \param _Scalar the type of coefficients + * \param SizeAtCompileTime the dimension of the matrix, or Dynamic + * \param MaxSizeAtCompileTime the dimension of the matrix, or Dynamic. This parameter is optional and defaults + * to SizeAtCompileTime. Most of the time, you do not need to specify it. + * + * \sa class DiagonalWrapper + */ namespace internal { -template -struct traits > - : traits > -{ - typedef Matrix<_Scalar,SizeAtCompileTime,1,0,MaxSizeAtCompileTime,1> DiagonalVectorType; - typedef DiagonalShape StorageKind; - enum { - Flags = LvalueBit | NoPreferredStorageOrderBit + template + struct traits> + : traits> + { + typedef Matrix<_Scalar, SizeAtCompileTime, 1, 0, MaxSizeAtCompileTime, 1> DiagonalVectorType; + typedef DiagonalShape StorageKind; + enum { Flags = LvalueBit | NoPreferredStorageOrderBit }; }; -}; -} +}// namespace internal template -class DiagonalMatrix - : public DiagonalBase > +class DiagonalMatrix : public DiagonalBase> { - public: - #ifndef EIGEN_PARSED_BY_DOXYGEN - typedef typename internal::traits::DiagonalVectorType DiagonalVectorType; - typedef const DiagonalMatrix& Nested; - typedef _Scalar Scalar; - typedef typename internal::traits::StorageKind StorageKind; - typedef typename internal::traits::StorageIndex StorageIndex; - #endif - - protected: - - DiagonalVectorType m_diagonal; - - public: - - /** const version of diagonal(). */ - EIGEN_DEVICE_FUNC - inline const DiagonalVectorType& diagonal() const { return m_diagonal; } - /** \returns a reference to the stored vector of diagonal coefficients. */ - EIGEN_DEVICE_FUNC - inline DiagonalVectorType& diagonal() { return m_diagonal; } - - /** Default constructor without initialization */ - EIGEN_DEVICE_FUNC - inline DiagonalMatrix() {} - - /** Constructs a diagonal matrix with given dimension */ - EIGEN_DEVICE_FUNC - explicit inline DiagonalMatrix(Index dim) : m_diagonal(dim) {} - - /** 2D constructor. */ - EIGEN_DEVICE_FUNC - inline DiagonalMatrix(const Scalar& x, const Scalar& y) : m_diagonal(x,y) {} - - /** 3D constructor. */ - EIGEN_DEVICE_FUNC - inline DiagonalMatrix(const Scalar& x, const Scalar& y, const Scalar& z) : m_diagonal(x,y,z) {} - - /** Copy constructor. */ - template - EIGEN_DEVICE_FUNC - inline DiagonalMatrix(const DiagonalBase& other) : m_diagonal(other.diagonal()) {} - - #ifndef EIGEN_PARSED_BY_DOXYGEN - /** copy constructor. prevent a default copy constructor from hiding the other templated constructor */ - inline DiagonalMatrix(const DiagonalMatrix& other) : m_diagonal(other.diagonal()) {} - #endif - - /** generic constructor from expression of the diagonal coefficients */ - template - EIGEN_DEVICE_FUNC - explicit inline DiagonalMatrix(const MatrixBase& other) : m_diagonal(other) - {} - - /** Copy operator. */ - template - EIGEN_DEVICE_FUNC - DiagonalMatrix& operator=(const DiagonalBase& other) - { - m_diagonal = other.diagonal(); - return *this; - } +public: +#ifndef EIGEN_PARSED_BY_DOXYGEN + typedef typename internal::traits::DiagonalVectorType DiagonalVectorType; + typedef const DiagonalMatrix &Nested; + typedef _Scalar Scalar; + typedef typename internal::traits::StorageKind StorageKind; + typedef typename internal::traits::StorageIndex StorageIndex; +#endif - #ifndef EIGEN_PARSED_BY_DOXYGEN - /** This is a special case of the templated operator=. Its purpose is to - * prevent a default operator= from hiding the templated operator=. - */ - EIGEN_DEVICE_FUNC - DiagonalMatrix& operator=(const DiagonalMatrix& other) - { - m_diagonal = other.diagonal(); - return *this; - } - #endif - - /** Resizes to given size. */ - EIGEN_DEVICE_FUNC - inline void resize(Index size) { m_diagonal.resize(size); } - /** Sets all coefficients to zero. */ - EIGEN_DEVICE_FUNC - inline void setZero() { m_diagonal.setZero(); } - /** Resizes and sets all coefficients to zero. */ - EIGEN_DEVICE_FUNC - inline void setZero(Index size) { m_diagonal.setZero(size); } - /** Sets this matrix to be the identity matrix of the current size. */ - EIGEN_DEVICE_FUNC - inline void setIdentity() { m_diagonal.setOnes(); } - /** Sets this matrix to be the identity matrix of the given size. */ - EIGEN_DEVICE_FUNC - inline void setIdentity(Index size) { m_diagonal.setOnes(size); } +protected: + DiagonalVectorType m_diagonal; + +public: + /** const version of diagonal(). */ + EIGEN_DEVICE_FUNC + inline const DiagonalVectorType &diagonal() const { return m_diagonal; } + /** \returns a reference to the stored vector of diagonal coefficients. */ + EIGEN_DEVICE_FUNC + inline DiagonalVectorType &diagonal() { return m_diagonal; } + + /** Default constructor without initialization */ + EIGEN_DEVICE_FUNC + inline DiagonalMatrix() {} + + /** Constructs a diagonal matrix with given dimension */ + EIGEN_DEVICE_FUNC + explicit inline DiagonalMatrix(Index dim) : m_diagonal(dim) {} + + /** 2D constructor. */ + EIGEN_DEVICE_FUNC + inline DiagonalMatrix(const Scalar &x, const Scalar &y) : m_diagonal(x, y) {} + + /** 3D constructor. */ + EIGEN_DEVICE_FUNC + inline DiagonalMatrix(const Scalar &x, const Scalar &y, const Scalar &z) : m_diagonal(x, y, z) {} + + /** Copy constructor. */ + template + EIGEN_DEVICE_FUNC inline DiagonalMatrix(const DiagonalBase &other) : m_diagonal(other.diagonal()) + {} + +#ifndef EIGEN_PARSED_BY_DOXYGEN + /** copy constructor. prevent a default copy constructor from hiding the other templated constructor */ + inline DiagonalMatrix(const DiagonalMatrix &other) : m_diagonal(other.diagonal()) {} +#endif + + /** generic constructor from expression of the diagonal coefficients */ + template + EIGEN_DEVICE_FUNC explicit inline DiagonalMatrix(const MatrixBase &other) : m_diagonal(other) + {} + + /** Copy operator. */ + template EIGEN_DEVICE_FUNC DiagonalMatrix &operator=(const DiagonalBase &other) + { + m_diagonal = other.diagonal(); + return *this; + } + +#ifndef EIGEN_PARSED_BY_DOXYGEN + /** This is a special case of the templated operator=. Its purpose is to + * prevent a default operator= from hiding the templated operator=. + */ + EIGEN_DEVICE_FUNC + DiagonalMatrix &operator=(const DiagonalMatrix &other) + { + m_diagonal = other.diagonal(); + return *this; + } +#endif + + /** Resizes to given size. */ + EIGEN_DEVICE_FUNC + inline void resize(Index size) { m_diagonal.resize(size); } + /** Sets all coefficients to zero. */ + EIGEN_DEVICE_FUNC + inline void setZero() { m_diagonal.setZero(); } + /** Resizes and sets all coefficients to zero. */ + EIGEN_DEVICE_FUNC + inline void setZero(Index size) { m_diagonal.setZero(size); } + /** Sets this matrix to be the identity matrix of the current size. */ + EIGEN_DEVICE_FUNC + inline void setIdentity() { m_diagonal.setOnes(); } + /** Sets this matrix to be the identity matrix of the given size. */ + EIGEN_DEVICE_FUNC + inline void setIdentity(Index size) { m_diagonal.setOnes(size); } }; /** \class DiagonalWrapper - * \ingroup Core_Module - * - * \brief Expression of a diagonal matrix - * - * \param _DiagonalVectorType the type of the vector of diagonal coefficients - * - * This class is an expression of a diagonal matrix, but not storing its own vector of diagonal coefficients, - * instead wrapping an existing vector expression. It is the return type of MatrixBase::asDiagonal() - * and most of the time this is the only way that it is used. - * - * \sa class DiagonalMatrix, class DiagonalBase, MatrixBase::asDiagonal() - */ + * \ingroup Core_Module + * + * \brief Expression of a diagonal matrix + * + * \param _DiagonalVectorType the type of the vector of diagonal coefficients + * + * This class is an expression of a diagonal matrix, but not storing its own vector of diagonal coefficients, + * instead wrapping an existing vector expression. It is the return type of MatrixBase::asDiagonal() + * and most of the time this is the only way that it is used. + * + * \sa class DiagonalMatrix, class DiagonalBase, MatrixBase::asDiagonal() + */ namespace internal { -template -struct traits > -{ - typedef _DiagonalVectorType DiagonalVectorType; - typedef typename DiagonalVectorType::Scalar Scalar; - typedef typename DiagonalVectorType::StorageIndex StorageIndex; - typedef DiagonalShape StorageKind; - typedef typename traits::XprKind XprKind; - enum { - RowsAtCompileTime = DiagonalVectorType::SizeAtCompileTime, - ColsAtCompileTime = DiagonalVectorType::SizeAtCompileTime, - MaxRowsAtCompileTime = DiagonalVectorType::MaxSizeAtCompileTime, - MaxColsAtCompileTime = DiagonalVectorType::MaxSizeAtCompileTime, - Flags = (traits::Flags & LvalueBit) | NoPreferredStorageOrderBit + template struct traits> + { + typedef _DiagonalVectorType DiagonalVectorType; + typedef typename DiagonalVectorType::Scalar Scalar; + typedef typename DiagonalVectorType::StorageIndex StorageIndex; + typedef DiagonalShape StorageKind; + typedef typename traits::XprKind XprKind; + enum { + RowsAtCompileTime = DiagonalVectorType::SizeAtCompileTime, + ColsAtCompileTime = DiagonalVectorType::SizeAtCompileTime, + MaxRowsAtCompileTime = DiagonalVectorType::MaxSizeAtCompileTime, + MaxColsAtCompileTime = DiagonalVectorType::MaxSizeAtCompileTime, + Flags = (traits::Flags & LvalueBit) | NoPreferredStorageOrderBit + }; }; -}; -} +}// namespace internal template class DiagonalWrapper - : public DiagonalBase >, internal::no_assignment_operator + : public DiagonalBase> + , internal::no_assignment_operator { - public: - #ifndef EIGEN_PARSED_BY_DOXYGEN - typedef _DiagonalVectorType DiagonalVectorType; - typedef DiagonalWrapper Nested; - #endif +public: +#ifndef EIGEN_PARSED_BY_DOXYGEN + typedef _DiagonalVectorType DiagonalVectorType; + typedef DiagonalWrapper Nested; +#endif - /** Constructor from expression of diagonal coefficients to wrap. */ - EIGEN_DEVICE_FUNC - explicit inline DiagonalWrapper(DiagonalVectorType& a_diagonal) : m_diagonal(a_diagonal) {} + /** Constructor from expression of diagonal coefficients to wrap. */ + EIGEN_DEVICE_FUNC + explicit inline DiagonalWrapper(DiagonalVectorType &a_diagonal) : m_diagonal(a_diagonal) {} - /** \returns a const reference to the wrapped expression of diagonal coefficients. */ - EIGEN_DEVICE_FUNC - const DiagonalVectorType& diagonal() const { return m_diagonal; } + /** \returns a const reference to the wrapped expression of diagonal coefficients. */ + EIGEN_DEVICE_FUNC + const DiagonalVectorType &diagonal() const { return m_diagonal; } - protected: - typename DiagonalVectorType::Nested m_diagonal; +protected: + typename DiagonalVectorType::Nested m_diagonal; }; /** \returns a pseudo-expression of a diagonal matrix with *this as vector of diagonal coefficients - * - * \only_for_vectors - * - * Example: \include MatrixBase_asDiagonal.cpp - * Output: \verbinclude MatrixBase_asDiagonal.out - * - * \sa class DiagonalWrapper, class DiagonalMatrix, diagonal(), isDiagonal() - **/ -template -inline const DiagonalWrapper -MatrixBase::asDiagonal() const + * + * \only_for_vectors + * + * Example: \include MatrixBase_asDiagonal.cpp + * Output: \verbinclude MatrixBase_asDiagonal.out + * + * \sa class DiagonalWrapper, class DiagonalMatrix, diagonal(), isDiagonal() + **/ +template inline const DiagonalWrapper MatrixBase::asDiagonal() const { return DiagonalWrapper(derived()); } /** \returns true if *this is approximately equal to a diagonal matrix, - * within the precision given by \a prec. - * - * Example: \include MatrixBase_isDiagonal.cpp - * Output: \verbinclude MatrixBase_isDiagonal.out - * - * \sa asDiagonal() - */ -template -bool MatrixBase::isDiagonal(const RealScalar& prec) const + * within the precision given by \a prec. + * + * Example: \include MatrixBase_isDiagonal.cpp + * Output: \verbinclude MatrixBase_isDiagonal.out + * + * \sa asDiagonal() + */ +template bool MatrixBase::isDiagonal(const RealScalar &prec) const { - if(cols() != rows()) return false; + if (cols() != rows()) return false; RealScalar maxAbsOnDiagonal = static_cast(-1); - for(Index j = 0; j < cols(); ++j) - { - RealScalar absOnDiagonal = numext::abs(coeff(j,j)); - if(absOnDiagonal > maxAbsOnDiagonal) maxAbsOnDiagonal = absOnDiagonal; + for (Index j = 0; j < cols(); ++j) { + RealScalar absOnDiagonal = numext::abs(coeff(j, j)); + if (absOnDiagonal > maxAbsOnDiagonal) maxAbsOnDiagonal = absOnDiagonal; } - for(Index j = 0; j < cols(); ++j) - for(Index i = 0; i < j; ++i) - { - if(!internal::isMuchSmallerThan(coeff(i, j), maxAbsOnDiagonal, prec)) return false; - if(!internal::isMuchSmallerThan(coeff(j, i), maxAbsOnDiagonal, prec)) return false; + for (Index j = 0; j < cols(); ++j) + for (Index i = 0; i < j; ++i) { + if (!internal::isMuchSmallerThan(coeff(i, j), maxAbsOnDiagonal, prec)) return false; + if (!internal::isMuchSmallerThan(coeff(j, i), maxAbsOnDiagonal, prec)) return false; } return true; } namespace internal { -template<> struct storage_kind_to_shape { typedef DiagonalShape Shape; }; + template<> struct storage_kind_to_shape + { + typedef DiagonalShape Shape; + }; -struct Diagonal2Dense {}; + struct Diagonal2Dense + { + }; -template<> struct AssignmentKind { typedef Diagonal2Dense Kind; }; + template<> struct AssignmentKind + { + typedef Diagonal2Dense Kind; + }; -// Diagonal matrix to Dense assignment -template< typename DstXprType, typename SrcXprType, typename Functor> -struct Assignment -{ - static void run(DstXprType &dst, const SrcXprType &src, const internal::assign_op &/*func*/) + // Diagonal matrix to Dense assignment + template + struct Assignment { - Index dstRows = src.rows(); - Index dstCols = src.cols(); - if((dst.rows()!=dstRows) || (dst.cols()!=dstCols)) - dst.resize(dstRows, dstCols); - - dst.setZero(); - dst.diagonal() = src.diagonal(); - } - - static void run(DstXprType &dst, const SrcXprType &src, const internal::add_assign_op &/*func*/) - { dst.diagonal() += src.diagonal(); } - - static void run(DstXprType &dst, const SrcXprType &src, const internal::sub_assign_op &/*func*/) - { dst.diagonal() -= src.diagonal(); } -}; + static void run(DstXprType &dst, + const SrcXprType &src, + const internal::assign_op & /*func*/) + { + Index dstRows = src.rows(); + Index dstCols = src.cols(); + if ((dst.rows() != dstRows) || (dst.cols() != dstCols)) dst.resize(dstRows, dstCols); + + dst.setZero(); + dst.diagonal() = src.diagonal(); + } + + static void run(DstXprType &dst, + const SrcXprType &src, + const internal::add_assign_op & /*func*/) + { + dst.diagonal() += src.diagonal(); + } + + static void run(DstXprType &dst, + const SrcXprType &src, + const internal::sub_assign_op & /*func*/) + { + dst.diagonal() -= src.diagonal(); + } + }; -} // namespace internal +}// namespace internal -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_DIAGONALMATRIX_H +#endif// EIGEN_DIAGONALMATRIX_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/DiagonalProduct.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/DiagonalProduct.h index d372b938..50d1dff8 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/DiagonalProduct.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/DiagonalProduct.h @@ -11,18 +11,18 @@ #ifndef EIGEN_DIAGONALPRODUCT_H #define EIGEN_DIAGONALPRODUCT_H -namespace Eigen { +namespace Eigen { /** \returns the diagonal matrix product of \c *this by the diagonal matrix \a diagonal. - */ + */ template template -inline const Product -MatrixBase::operator*(const DiagonalBase &a_diagonal) const +inline const Product MatrixBase::operator*( + const DiagonalBase &a_diagonal) const { - return Product(derived(),a_diagonal.derived()); + return Product(derived(), a_diagonal.derived()); } -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_DIAGONALPRODUCT_H +#endif// EIGEN_DIAGONALPRODUCT_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/Dot.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/Dot.h index 1fe7a84a..42e84a57 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/Dot.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/Dot.h @@ -10,253 +10,246 @@ #ifndef EIGEN_DOT_H #define EIGEN_DOT_H -namespace Eigen { +namespace Eigen { namespace internal { -// helper function for dot(). The problem is that if we put that in the body of dot(), then upon calling dot -// with mismatched types, the compiler emits errors about failing to instantiate cwiseProduct BEFORE -// looking at the static assertions. Thus this is a trick to get better compile errors. -template -struct dot_nocheck -{ - typedef scalar_conj_product_op::Scalar,typename traits::Scalar> conj_prod; - typedef typename conj_prod::result_type ResScalar; - EIGEN_DEVICE_FUNC - EIGEN_STRONG_INLINE - static ResScalar run(const MatrixBase& a, const MatrixBase& b) + // helper function for dot(). The problem is that if we put that in the body of dot(), then upon calling dot + // with mismatched types, the compiler emits errors about failing to instantiate cwiseProduct BEFORE + // looking at the static assertions. Thus this is a trick to get better compile errors. + template + struct dot_nocheck { - return a.template binaryExpr(b).sum(); - } -}; + typedef scalar_conj_product_op::Scalar, typename traits::Scalar> conj_prod; + typedef typename conj_prod::result_type ResScalar; + EIGEN_DEVICE_FUNC + EIGEN_STRONG_INLINE + static ResScalar run(const MatrixBase &a, const MatrixBase &b) + { + return a.template binaryExpr(b).sum(); + } + }; -template -struct dot_nocheck -{ - typedef scalar_conj_product_op::Scalar,typename traits::Scalar> conj_prod; - typedef typename conj_prod::result_type ResScalar; - EIGEN_DEVICE_FUNC - EIGEN_STRONG_INLINE - static ResScalar run(const MatrixBase& a, const MatrixBase& b) + template struct dot_nocheck { - return a.transpose().template binaryExpr(b).sum(); - } -}; + typedef scalar_conj_product_op::Scalar, typename traits::Scalar> conj_prod; + typedef typename conj_prod::result_type ResScalar; + EIGEN_DEVICE_FUNC + EIGEN_STRONG_INLINE + static ResScalar run(const MatrixBase &a, const MatrixBase &b) + { + return a.transpose().template binaryExpr(b).sum(); + } + }; -} // end namespace internal +}// end namespace internal /** \fn MatrixBase::dot - * \returns the dot product of *this with other. - * - * \only_for_vectors - * - * \note If the scalar type is complex numbers, then this function returns the hermitian - * (sesquilinear) dot product, conjugate-linear in the first variable and linear in the - * second variable. - * - * \sa squaredNorm(), norm() - */ + * \returns the dot product of *this with other. + * + * \only_for_vectors + * + * \note If the scalar type is complex numbers, then this function returns the hermitian + * (sesquilinear) dot product, conjugate-linear in the first variable and linear in the + * second variable. + * + * \sa squaredNorm(), norm() + */ template template -EIGEN_DEVICE_FUNC -EIGEN_STRONG_INLINE -typename ScalarBinaryOpTraits::Scalar,typename internal::traits::Scalar>::ReturnType -MatrixBase::dot(const MatrixBase& other) const +EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE typename ScalarBinaryOpTraits::Scalar, + typename internal::traits::Scalar>::ReturnType + MatrixBase::dot(const MatrixBase &other) const { EIGEN_STATIC_ASSERT_VECTOR_ONLY(Derived) EIGEN_STATIC_ASSERT_VECTOR_ONLY(OtherDerived) - EIGEN_STATIC_ASSERT_SAME_VECTOR_SIZE(Derived,OtherDerived) + EIGEN_STATIC_ASSERT_SAME_VECTOR_SIZE(Derived, OtherDerived) #if !(defined(EIGEN_NO_STATIC_ASSERT) && defined(EIGEN_NO_DEBUG)) - typedef internal::scalar_conj_product_op func; - EIGEN_CHECK_BINARY_COMPATIBILIY(func,Scalar,typename OtherDerived::Scalar); + typedef internal::scalar_conj_product_op func; + EIGEN_CHECK_BINARY_COMPATIBILIY(func, Scalar, typename OtherDerived::Scalar); #endif - + eigen_assert(size() == other.size()); - return internal::dot_nocheck::run(*this, other); + return internal::dot_nocheck::run(*this, other); } //---------- implementation of L2 norm and related functions ---------- /** \returns, for vectors, the squared \em l2 norm of \c *this, and for matrices the Frobenius norm. - * In both cases, it consists in the sum of the square of all the matrix entries. - * For vectors, this is also equals to the dot product of \c *this with itself. - * - * \sa dot(), norm(), lpNorm() - */ + * In both cases, it consists in the sum of the square of all the matrix entries. + * For vectors, this is also equals to the dot product of \c *this with itself. + * + * \sa dot(), norm(), lpNorm() + */ template -EIGEN_STRONG_INLINE typename NumTraits::Scalar>::Real MatrixBase::squaredNorm() const +EIGEN_STRONG_INLINE typename NumTraits::Scalar>::Real + MatrixBase::squaredNorm() const { return numext::real((*this).cwiseAbs2().sum()); } /** \returns, for vectors, the \em l2 norm of \c *this, and for matrices the Frobenius norm. - * In both cases, it consists in the square root of the sum of the square of all the matrix entries. - * For vectors, this is also equals to the square root of the dot product of \c *this with itself. - * - * \sa lpNorm(), dot(), squaredNorm() - */ + * In both cases, it consists in the square root of the sum of the square of all the matrix entries. + * For vectors, this is also equals to the square root of the dot product of \c *this with itself. + * + * \sa lpNorm(), dot(), squaredNorm() + */ template -EIGEN_STRONG_INLINE typename NumTraits::Scalar>::Real MatrixBase::norm() const +EIGEN_STRONG_INLINE typename NumTraits::Scalar>::Real + MatrixBase::norm() const { return numext::sqrt(squaredNorm()); } /** \returns an expression of the quotient of \c *this by its own norm. - * - * \warning If the input vector is too small (i.e., this->norm()==0), - * then this function returns a copy of the input. - * - * \only_for_vectors - * - * \sa norm(), normalize() - */ + * + * \warning If the input vector is too small (i.e., this->norm()==0), + * then this function returns a copy of the input. + * + * \only_for_vectors + * + * \sa norm(), normalize() + */ template -EIGEN_STRONG_INLINE const typename MatrixBase::PlainObject -MatrixBase::normalized() const +EIGEN_STRONG_INLINE const typename MatrixBase::PlainObject MatrixBase::normalized() const { - typedef typename internal::nested_eval::type _Nested; + typedef typename internal::nested_eval::type _Nested; _Nested n(derived()); RealScalar z = n.squaredNorm(); // NOTE: after extensive benchmarking, this conditional does not impact performance, at least on recent x86 CPU - if(z>RealScalar(0)) + if (z > RealScalar(0)) return n / numext::sqrt(z); else return n; } /** Normalizes the vector, i.e. divides it by its own norm. - * - * \only_for_vectors - * - * \warning If the input vector is too small (i.e., this->norm()==0), then \c *this is left unchanged. - * - * \sa norm(), normalized() - */ -template -EIGEN_STRONG_INLINE void MatrixBase::normalize() + * + * \only_for_vectors + * + * \warning If the input vector is too small (i.e., this->norm()==0), then \c *this is left unchanged. + * + * \sa norm(), normalized() + */ +template EIGEN_STRONG_INLINE void MatrixBase::normalize() { RealScalar z = squaredNorm(); // NOTE: after extensive benchmarking, this conditional does not impact performance, at least on recent x86 CPU - if(z>RealScalar(0)) - derived() /= numext::sqrt(z); + if (z > RealScalar(0)) derived() /= numext::sqrt(z); } /** \returns an expression of the quotient of \c *this by its own norm while avoiding underflow and overflow. - * - * \only_for_vectors - * - * This method is analogue to the normalized() method, but it reduces the risk of - * underflow and overflow when computing the norm. - * - * \warning If the input vector is too small (i.e., this->norm()==0), - * then this function returns a copy of the input. - * - * \sa stableNorm(), stableNormalize(), normalized() - */ + * + * \only_for_vectors + * + * This method is analogue to the normalized() method, but it reduces the risk of + * underflow and overflow when computing the norm. + * + * \warning If the input vector is too small (i.e., this->norm()==0), + * then this function returns a copy of the input. + * + * \sa stableNorm(), stableNormalize(), normalized() + */ template -EIGEN_STRONG_INLINE const typename MatrixBase::PlainObject -MatrixBase::stableNormalized() const +EIGEN_STRONG_INLINE const typename MatrixBase::PlainObject MatrixBase::stableNormalized() const { - typedef typename internal::nested_eval::type _Nested; + typedef typename internal::nested_eval::type _Nested; _Nested n(derived()); RealScalar w = n.cwiseAbs().maxCoeff(); - RealScalar z = (n/w).squaredNorm(); - if(z>RealScalar(0)) - return n / (numext::sqrt(z)*w); + RealScalar z = (n / w).squaredNorm(); + if (z > RealScalar(0)) + return n / (numext::sqrt(z) * w); else return n; } /** Normalizes the vector while avoid underflow and overflow - * - * \only_for_vectors - * - * This method is analogue to the normalize() method, but it reduces the risk of - * underflow and overflow when computing the norm. - * - * \warning If the input vector is too small (i.e., this->norm()==0), then \c *this is left unchanged. - * - * \sa stableNorm(), stableNormalized(), normalize() - */ -template -EIGEN_STRONG_INLINE void MatrixBase::stableNormalize() + * + * \only_for_vectors + * + * This method is analogue to the normalize() method, but it reduces the risk of + * underflow and overflow when computing the norm. + * + * \warning If the input vector is too small (i.e., this->norm()==0), then \c *this is left unchanged. + * + * \sa stableNorm(), stableNormalized(), normalize() + */ +template EIGEN_STRONG_INLINE void MatrixBase::stableNormalize() { RealScalar w = cwiseAbs().maxCoeff(); - RealScalar z = (derived()/w).squaredNorm(); - if(z>RealScalar(0)) - derived() /= numext::sqrt(z)*w; + RealScalar z = (derived() / w).squaredNorm(); + if (z > RealScalar(0)) derived() /= numext::sqrt(z) * w; } //---------- implementation of other norms ---------- namespace internal { -template -struct lpNorm_selector -{ - typedef typename NumTraits::Scalar>::Real RealScalar; - EIGEN_DEVICE_FUNC - static inline RealScalar run(const MatrixBase& m) + template struct lpNorm_selector { - EIGEN_USING_STD_MATH(pow) - return pow(m.cwiseAbs().array().pow(p).sum(), RealScalar(1)/p); - } -}; + typedef typename NumTraits::Scalar>::Real RealScalar; + EIGEN_DEVICE_FUNC + static inline RealScalar run(const MatrixBase &m) + { + EIGEN_USING_STD_MATH(pow) + return pow(m.cwiseAbs().array().pow(p).sum(), RealScalar(1) / p); + } + }; -template -struct lpNorm_selector -{ - EIGEN_DEVICE_FUNC - static inline typename NumTraits::Scalar>::Real run(const MatrixBase& m) + template struct lpNorm_selector { - return m.cwiseAbs().sum(); - } -}; + EIGEN_DEVICE_FUNC + static inline typename NumTraits::Scalar>::Real run(const MatrixBase &m) + { + return m.cwiseAbs().sum(); + } + }; -template -struct lpNorm_selector -{ - EIGEN_DEVICE_FUNC - static inline typename NumTraits::Scalar>::Real run(const MatrixBase& m) + template struct lpNorm_selector { - return m.norm(); - } -}; + EIGEN_DEVICE_FUNC + static inline typename NumTraits::Scalar>::Real run(const MatrixBase &m) + { + return m.norm(); + } + }; -template -struct lpNorm_selector -{ - typedef typename NumTraits::Scalar>::Real RealScalar; - EIGEN_DEVICE_FUNC - static inline RealScalar run(const MatrixBase& m) + template struct lpNorm_selector { - if(Derived::SizeAtCompileTime==0 || (Derived::SizeAtCompileTime==Dynamic && m.size()==0)) - return RealScalar(0); - return m.cwiseAbs().maxCoeff(); - } -}; + typedef typename NumTraits::Scalar>::Real RealScalar; + EIGEN_DEVICE_FUNC + static inline RealScalar run(const MatrixBase &m) + { + if (Derived::SizeAtCompileTime == 0 || (Derived::SizeAtCompileTime == Dynamic && m.size() == 0)) + return RealScalar(0); + return m.cwiseAbs().maxCoeff(); + } + }; -} // end namespace internal +}// end namespace internal -/** \returns the \b coefficient-wise \f$ \ell^p \f$ norm of \c *this, that is, returns the p-th root of the sum of the p-th powers of the absolute values - * of the coefficients of \c *this. If \a p is the special value \a Eigen::Infinity, this function returns the \f$ \ell^\infty \f$ - * norm, that is the maximum of the absolute values of the coefficients of \c *this. - * - * In all cases, if \c *this is empty, then the value 0 is returned. - * - * \note For matrices, this function does not compute the operator-norm. That is, if \c *this is a matrix, then its coefficients are interpreted as a 1D vector. Nonetheless, you can easily compute the 1-norm and \f$\infty\f$-norm matrix operator norms using \link TutorialReductionsVisitorsBroadcastingReductionsNorm partial reductions \endlink. - * - * \sa norm() - */ +/** \returns the \b coefficient-wise \f$ \ell^p \f$ norm of \c *this, that is, returns the p-th root of the sum of the + * p-th powers of the absolute values of the coefficients of \c *this. If \a p is the special value \a Eigen::Infinity, + * this function returns the \f$ \ell^\infty \f$ norm, that is the maximum of the absolute values of the coefficients of + * \c *this. + * + * In all cases, if \c *this is empty, then the value 0 is returned. + * + * \note For matrices, this function does not compute the operator-norm. That is, if \c *this is a matrix, then its + * coefficients are interpreted as a 1D vector. Nonetheless, you can easily compute the 1-norm and \f$\infty\f$-norm + * matrix operator norms using \link TutorialReductionsVisitorsBroadcastingReductionsNorm partial reductions \endlink. + * + * \sa norm() + */ template template #ifndef EIGEN_PARSED_BY_DOXYGEN @@ -264,7 +257,7 @@ inline typename NumTraits::Scalar>::Real #else MatrixBase::RealScalar #endif -MatrixBase::lpNorm() const + MatrixBase::lpNorm() const { return internal::lpNorm_selector::run(*this); } @@ -272,47 +265,42 @@ MatrixBase::lpNorm() const //---------- implementation of isOrthogonal / isUnitary ---------- /** \returns true if *this is approximately orthogonal to \a other, - * within the precision given by \a prec. - * - * Example: \include MatrixBase_isOrthogonal.cpp - * Output: \verbinclude MatrixBase_isOrthogonal.out - */ + * within the precision given by \a prec. + * + * Example: \include MatrixBase_isOrthogonal.cpp + * Output: \verbinclude MatrixBase_isOrthogonal.out + */ template template -bool MatrixBase::isOrthogonal -(const MatrixBase& other, const RealScalar& prec) const +bool MatrixBase::isOrthogonal(const MatrixBase &other, const RealScalar &prec) const { - typename internal::nested_eval::type nested(derived()); - typename internal::nested_eval::type otherNested(other.derived()); + typename internal::nested_eval::type nested(derived()); + typename internal::nested_eval::type otherNested(other.derived()); return numext::abs2(nested.dot(otherNested)) <= prec * prec * nested.squaredNorm() * otherNested.squaredNorm(); } /** \returns true if *this is approximately an unitary matrix, - * within the precision given by \a prec. In the case where the \a Scalar - * type is real numbers, a unitary matrix is an orthogonal matrix, whence the name. - * - * \note This can be used to check whether a family of vectors forms an orthonormal basis. - * Indeed, \c m.isUnitary() returns true if and only if the columns (equivalently, the rows) of m form an - * orthonormal basis. - * - * Example: \include MatrixBase_isUnitary.cpp - * Output: \verbinclude MatrixBase_isUnitary.out - */ -template -bool MatrixBase::isUnitary(const RealScalar& prec) const + * within the precision given by \a prec. In the case where the \a Scalar + * type is real numbers, a unitary matrix is an orthogonal matrix, whence the name. + * + * \note This can be used to check whether a family of vectors forms an orthonormal basis. + * Indeed, \c m.isUnitary() returns true if and only if the columns (equivalently, the rows) of m form an + * orthonormal basis. + * + * Example: \include MatrixBase_isUnitary.cpp + * Output: \verbinclude MatrixBase_isUnitary.out + */ +template bool MatrixBase::isUnitary(const RealScalar &prec) const { - typename internal::nested_eval::type self(derived()); - for(Index i = 0; i < cols(); ++i) - { - if(!internal::isApprox(self.col(i).squaredNorm(), static_cast(1), prec)) - return false; - for(Index j = 0; j < i; ++j) - if(!internal::isMuchSmallerThan(self.col(i).dot(self.col(j)), static_cast(1), prec)) - return false; + typename internal::nested_eval::type self(derived()); + for (Index i = 0; i < cols(); ++i) { + if (!internal::isApprox(self.col(i).squaredNorm(), static_cast(1), prec)) return false; + for (Index j = 0; j < i; ++j) + if (!internal::isMuchSmallerThan(self.col(i).dot(self.col(j)), static_cast(1), prec)) return false; } return true; } -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_DOT_H +#endif// EIGEN_DOT_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/EigenBase.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/EigenBase.h index b195506a..3e1c61f6 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/EigenBase.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/EigenBase.h @@ -14,27 +14,27 @@ namespace Eigen { /** \class EigenBase - * \ingroup Core_Module - * - * Common base class for all classes T such that MatrixBase has an operator=(T) and a constructor MatrixBase(T). - * - * In other words, an EigenBase object is an object that can be copied into a MatrixBase. - * - * Besides MatrixBase-derived classes, this also includes special matrix classes such as diagonal matrices, etc. - * - * Notice that this class is trivial, it is only used to disambiguate overloaded functions. - * - * \sa \blank \ref TopicClassHierarchy - */ + * \ingroup Core_Module + * + * Common base class for all classes T such that MatrixBase has an operator=(T) and a constructor MatrixBase(T). + * + * In other words, an EigenBase object is an object that can be copied into a MatrixBase. + * + * Besides MatrixBase-derived classes, this also includes special matrix classes such as diagonal matrices, etc. + * + * Notice that this class is trivial, it is only used to disambiguate overloaded functions. + * + * \sa \blank \ref TopicClassHierarchy + */ template struct EigenBase { -// typedef typename internal::plain_matrix_type::type PlainObject; - + // typedef typename internal::plain_matrix_type::type PlainObject; + /** \brief The interface type of indices - * \details To change this, \c \#define the preprocessor symbol \c EIGEN_DEFAULT_DENSE_INDEX_TYPE. - * \deprecated Since Eigen 3.3, its usage is deprecated. Use Eigen::Index instead. - * \sa StorageIndex, \ref TopicPreprocessorDirectives. - */ + * \details To change this, \c \#define the preprocessor symbol \c EIGEN_DEFAULT_DENSE_INDEX_TYPE. + * \deprecated Since Eigen 3.3, its usage is deprecated. Use Eigen::Index instead. + * \sa StorageIndex, \ref TopicPreprocessorDirectives. + */ typedef Eigen::Index Index; // FIXME is it needed? @@ -42,17 +42,15 @@ template struct EigenBase /** \returns a reference to the derived object */ EIGEN_DEVICE_FUNC - Derived& derived() { return *static_cast(this); } + Derived &derived() { return *static_cast(this); } /** \returns a const reference to the derived object */ EIGEN_DEVICE_FUNC - const Derived& derived() const { return *static_cast(this); } + const Derived &derived() const { return *static_cast(this); } EIGEN_DEVICE_FUNC - inline Derived& const_cast_derived() const - { return *static_cast(const_cast(this)); } + inline Derived &const_cast_derived() const { return *static_cast(const_cast(this)); } EIGEN_DEVICE_FUNC - inline const Derived& const_derived() const - { return *static_cast(this); } + inline const Derived &const_derived() const { return *static_cast(this); } /** \returns the number of rows. \sa cols(), RowsAtCompileTime */ EIGEN_DEVICE_FUNC @@ -61,43 +59,35 @@ template struct EigenBase EIGEN_DEVICE_FUNC inline Index cols() const { return derived().cols(); } /** \returns the number of coefficients, which is rows()*cols(). - * \sa rows(), cols(), SizeAtCompileTime. */ + * \sa rows(), cols(), SizeAtCompileTime. */ EIGEN_DEVICE_FUNC inline Index size() const { return rows() * cols(); } /** \internal Don't use it, but do the equivalent: \code dst = *this; \endcode */ - template - EIGEN_DEVICE_FUNC - inline void evalTo(Dest& dst) const - { derived().evalTo(dst); } + template EIGEN_DEVICE_FUNC inline void evalTo(Dest &dst) const { derived().evalTo(dst); } /** \internal Don't use it, but do the equivalent: \code dst += *this; \endcode */ - template - EIGEN_DEVICE_FUNC - inline void addTo(Dest& dst) const + template EIGEN_DEVICE_FUNC inline void addTo(Dest &dst) const { // This is the default implementation, // derived class can reimplement it in a more optimized way. - typename Dest::PlainObject res(rows(),cols()); + typename Dest::PlainObject res(rows(), cols()); evalTo(res); dst += res; } /** \internal Don't use it, but do the equivalent: \code dst -= *this; \endcode */ - template - EIGEN_DEVICE_FUNC - inline void subTo(Dest& dst) const + template EIGEN_DEVICE_FUNC inline void subTo(Dest &dst) const { // This is the default implementation, // derived class can reimplement it in a more optimized way. - typename Dest::PlainObject res(rows(),cols()); + typename Dest::PlainObject res(rows(), cols()); evalTo(res); dst -= res; } /** \internal Don't use it, but do the equivalent: \code dst.applyOnTheRight(*this); \endcode */ - template - EIGEN_DEVICE_FUNC inline void applyThisOnTheRight(Dest& dst) const + template EIGEN_DEVICE_FUNC inline void applyThisOnTheRight(Dest &dst) const { // This is the default implementation, // derived class can reimplement it in a more optimized way. @@ -105,32 +95,29 @@ template struct EigenBase } /** \internal Don't use it, but do the equivalent: \code dst.applyOnTheLeft(*this); \endcode */ - template - EIGEN_DEVICE_FUNC inline void applyThisOnTheLeft(Dest& dst) const + template EIGEN_DEVICE_FUNC inline void applyThisOnTheLeft(Dest &dst) const { // This is the default implementation, // derived class can reimplement it in a more optimized way. dst = this->derived() * dst; } - }; /*************************************************************************** -* Implementation of matrix base methods -***************************************************************************/ + * Implementation of matrix base methods + ***************************************************************************/ /** \brief Copies the generic expression \a other into *this. - * - * \details The expression must provide a (templated) evalTo(Derived& dst) const - * function which does the actual job. In practice, this allows any user to write - * its own special matrix without having to modify MatrixBase - * - * \returns a reference to *this. - */ + * + * \details The expression must provide a (templated) evalTo(Derived& dst) const + * function which does the actual job. In practice, this allows any user to write + * its own special matrix without having to modify MatrixBase + * + * \returns a reference to *this. + */ template template -EIGEN_DEVICE_FUNC -Derived& DenseBase::operator=(const EigenBase &other) +EIGEN_DEVICE_FUNC Derived &DenseBase::operator=(const EigenBase &other) { call_assignment(derived(), other.derived()); return derived(); @@ -138,22 +125,20 @@ Derived& DenseBase::operator=(const EigenBase &other) template template -EIGEN_DEVICE_FUNC -Derived& DenseBase::operator+=(const EigenBase &other) +EIGEN_DEVICE_FUNC Derived &DenseBase::operator+=(const EigenBase &other) { - call_assignment(derived(), other.derived(), internal::add_assign_op()); + call_assignment(derived(), other.derived(), internal::add_assign_op()); return derived(); } template template -EIGEN_DEVICE_FUNC -Derived& DenseBase::operator-=(const EigenBase &other) +EIGEN_DEVICE_FUNC Derived &DenseBase::operator-=(const EigenBase &other) { - call_assignment(derived(), other.derived(), internal::sub_assign_op()); + call_assignment(derived(), other.derived(), internal::sub_assign_op()); return derived(); } -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_EIGENBASE_H +#endif// EIGEN_EIGENBASE_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/ForceAlignedAccess.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/ForceAlignedAccess.h index 7b08b45e..6b130c91 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/ForceAlignedAccess.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/ForceAlignedAccess.h @@ -13,134 +13,120 @@ namespace Eigen { /** \class ForceAlignedAccess - * \ingroup Core_Module - * - * \brief Enforce aligned packet loads and stores regardless of what is requested - * - * \param ExpressionType the type of the object of which we are forcing aligned packet access - * - * This class is the return type of MatrixBase::forceAlignedAccess() - * and most of the time this is the only way it is used. - * - * \sa MatrixBase::forceAlignedAccess() - */ + * \ingroup Core_Module + * + * \brief Enforce aligned packet loads and stores regardless of what is requested + * + * \param ExpressionType the type of the object of which we are forcing aligned packet access + * + * This class is the return type of MatrixBase::forceAlignedAccess() + * and most of the time this is the only way it is used. + * + * \sa MatrixBase::forceAlignedAccess() + */ namespace internal { -template -struct traits > : public traits -{}; -} + template struct traits> : public traits + { + }; +}// namespace internal -template class ForceAlignedAccess - : public internal::dense_xpr_base< ForceAlignedAccess >::type +template +class ForceAlignedAccess : public internal::dense_xpr_base>::type { - public: - - typedef typename internal::dense_xpr_base::type Base; - EIGEN_DENSE_PUBLIC_INTERFACE(ForceAlignedAccess) - - EIGEN_DEVICE_FUNC explicit inline ForceAlignedAccess(const ExpressionType& matrix) : m_expression(matrix) {} - - EIGEN_DEVICE_FUNC inline Index rows() const { return m_expression.rows(); } - EIGEN_DEVICE_FUNC inline Index cols() const { return m_expression.cols(); } - EIGEN_DEVICE_FUNC inline Index outerStride() const { return m_expression.outerStride(); } - EIGEN_DEVICE_FUNC inline Index innerStride() const { return m_expression.innerStride(); } - - EIGEN_DEVICE_FUNC inline const CoeffReturnType coeff(Index row, Index col) const - { - return m_expression.coeff(row, col); - } - - EIGEN_DEVICE_FUNC inline Scalar& coeffRef(Index row, Index col) - { - return m_expression.const_cast_derived().coeffRef(row, col); - } - - EIGEN_DEVICE_FUNC inline const CoeffReturnType coeff(Index index) const - { - return m_expression.coeff(index); - } - - EIGEN_DEVICE_FUNC inline Scalar& coeffRef(Index index) - { - return m_expression.const_cast_derived().coeffRef(index); - } - - template - inline const PacketScalar packet(Index row, Index col) const - { - return m_expression.template packet(row, col); - } - - template - inline void writePacket(Index row, Index col, const PacketScalar& x) - { - m_expression.const_cast_derived().template writePacket(row, col, x); - } - - template - inline const PacketScalar packet(Index index) const - { - return m_expression.template packet(index); - } - - template - inline void writePacket(Index index, const PacketScalar& x) - { - m_expression.const_cast_derived().template writePacket(index, x); - } - - EIGEN_DEVICE_FUNC operator const ExpressionType&() const { return m_expression; } - - protected: - const ExpressionType& m_expression; - - private: - ForceAlignedAccess& operator=(const ForceAlignedAccess&); +public: + typedef typename internal::dense_xpr_base::type Base; + EIGEN_DENSE_PUBLIC_INTERFACE(ForceAlignedAccess) + + EIGEN_DEVICE_FUNC explicit inline ForceAlignedAccess(const ExpressionType &matrix) : m_expression(matrix) {} + + EIGEN_DEVICE_FUNC inline Index rows() const { return m_expression.rows(); } + EIGEN_DEVICE_FUNC inline Index cols() const { return m_expression.cols(); } + EIGEN_DEVICE_FUNC inline Index outerStride() const { return m_expression.outerStride(); } + EIGEN_DEVICE_FUNC inline Index innerStride() const { return m_expression.innerStride(); } + + EIGEN_DEVICE_FUNC inline const CoeffReturnType coeff(Index row, Index col) const + { + return m_expression.coeff(row, col); + } + + EIGEN_DEVICE_FUNC inline Scalar &coeffRef(Index row, Index col) + { + return m_expression.const_cast_derived().coeffRef(row, col); + } + + EIGEN_DEVICE_FUNC inline const CoeffReturnType coeff(Index index) const { return m_expression.coeff(index); } + + EIGEN_DEVICE_FUNC inline Scalar &coeffRef(Index index) { return m_expression.const_cast_derived().coeffRef(index); } + + template inline const PacketScalar packet(Index row, Index col) const + { + return m_expression.template packet(row, col); + } + + template inline void writePacket(Index row, Index col, const PacketScalar &x) + { + m_expression.const_cast_derived().template writePacket(row, col, x); + } + + template inline const PacketScalar packet(Index index) const + { + return m_expression.template packet(index); + } + + template inline void writePacket(Index index, const PacketScalar &x) + { + m_expression.const_cast_derived().template writePacket(index, x); + } + + EIGEN_DEVICE_FUNC operator const ExpressionType &() const { return m_expression; } + +protected: + const ExpressionType &m_expression; + +private: + ForceAlignedAccess &operator=(const ForceAlignedAccess &); }; /** \returns an expression of *this with forced aligned access - * \sa forceAlignedAccessIf(),class ForceAlignedAccess - */ -template -inline const ForceAlignedAccess -MatrixBase::forceAlignedAccess() const + * \sa forceAlignedAccessIf(),class ForceAlignedAccess + */ +template inline const ForceAlignedAccess MatrixBase::forceAlignedAccess() const { return ForceAlignedAccess(derived()); } /** \returns an expression of *this with forced aligned access - * \sa forceAlignedAccessIf(), class ForceAlignedAccess - */ -template -inline ForceAlignedAccess -MatrixBase::forceAlignedAccess() + * \sa forceAlignedAccessIf(), class ForceAlignedAccess + */ +template inline ForceAlignedAccess MatrixBase::forceAlignedAccess() { return ForceAlignedAccess(derived()); } /** \returns an expression of *this with forced aligned access if \a Enable is true. - * \sa forceAlignedAccess(), class ForceAlignedAccess - */ + * \sa forceAlignedAccess(), class ForceAlignedAccess + */ template template -inline typename internal::add_const_on_value_type,Derived&>::type>::type -MatrixBase::forceAlignedAccessIf() const +inline typename internal::add_const_on_value_type< + typename internal::conditional, Derived &>::type>::type + MatrixBase::forceAlignedAccessIf() const { - return derived(); // FIXME This should not work but apparently is never used + return derived();// FIXME This should not work but apparently is never used } /** \returns an expression of *this with forced aligned access if \a Enable is true. - * \sa forceAlignedAccess(), class ForceAlignedAccess - */ + * \sa forceAlignedAccess(), class ForceAlignedAccess + */ template template -inline typename internal::conditional,Derived&>::type -MatrixBase::forceAlignedAccessIf() +inline typename internal::conditional, Derived &>::type + MatrixBase::forceAlignedAccessIf() { - return derived(); // FIXME This should not work but apparently is never used + return derived();// FIXME This should not work but apparently is never used } -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_FORCEALIGNEDACCESS_H +#endif// EIGEN_FORCEALIGNEDACCESS_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/Fuzzy.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/Fuzzy.h index 3e403a09..81c4a66b 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/Fuzzy.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/Fuzzy.h @@ -11,145 +11,134 @@ #ifndef EIGEN_FUZZY_H #define EIGEN_FUZZY_H -namespace Eigen { +namespace Eigen { -namespace internal -{ +namespace internal { -template::IsInteger> -struct isApprox_selector -{ - EIGEN_DEVICE_FUNC - static bool run(const Derived& x, const OtherDerived& y, const typename Derived::RealScalar& prec) + template::IsInteger> + struct isApprox_selector { - typename internal::nested_eval::type nested(x); - typename internal::nested_eval::type otherNested(y); - return (nested - otherNested).cwiseAbs2().sum() <= prec * prec * numext::mini(nested.cwiseAbs2().sum(), otherNested.cwiseAbs2().sum()); - } -}; - -template -struct isApprox_selector -{ - EIGEN_DEVICE_FUNC - static bool run(const Derived& x, const OtherDerived& y, const typename Derived::RealScalar&) + EIGEN_DEVICE_FUNC + static bool run(const Derived &x, const OtherDerived &y, const typename Derived::RealScalar &prec) + { + typename internal::nested_eval::type nested(x); + typename internal::nested_eval::type otherNested(y); + return (nested - otherNested).cwiseAbs2().sum() + <= prec * prec * numext::mini(nested.cwiseAbs2().sum(), otherNested.cwiseAbs2().sum()); + } + }; + + template struct isApprox_selector { - return x.matrix() == y.matrix(); - } -}; - -template::IsInteger> -struct isMuchSmallerThan_object_selector -{ - EIGEN_DEVICE_FUNC - static bool run(const Derived& x, const OtherDerived& y, const typename Derived::RealScalar& prec) + EIGEN_DEVICE_FUNC + static bool run(const Derived &x, const OtherDerived &y, const typename Derived::RealScalar &) + { + return x.matrix() == y.matrix(); + } + }; + + template::IsInteger> + struct isMuchSmallerThan_object_selector { - return x.cwiseAbs2().sum() <= numext::abs2(prec) * y.cwiseAbs2().sum(); - } -}; - -template -struct isMuchSmallerThan_object_selector -{ - EIGEN_DEVICE_FUNC - static bool run(const Derived& x, const OtherDerived&, const typename Derived::RealScalar&) + EIGEN_DEVICE_FUNC + static bool run(const Derived &x, const OtherDerived &y, const typename Derived::RealScalar &prec) + { + return x.cwiseAbs2().sum() <= numext::abs2(prec) * y.cwiseAbs2().sum(); + } + }; + + template + struct isMuchSmallerThan_object_selector { - return x.matrix() == Derived::Zero(x.rows(), x.cols()).matrix(); - } -}; - -template::IsInteger> -struct isMuchSmallerThan_scalar_selector -{ - EIGEN_DEVICE_FUNC - static bool run(const Derived& x, const typename Derived::RealScalar& y, const typename Derived::RealScalar& prec) + EIGEN_DEVICE_FUNC + static bool run(const Derived &x, const OtherDerived &, const typename Derived::RealScalar &) + { + return x.matrix() == Derived::Zero(x.rows(), x.cols()).matrix(); + } + }; + + template::IsInteger> + struct isMuchSmallerThan_scalar_selector { - return x.cwiseAbs2().sum() <= numext::abs2(prec * y); - } -}; - -template -struct isMuchSmallerThan_scalar_selector -{ - EIGEN_DEVICE_FUNC - static bool run(const Derived& x, const typename Derived::RealScalar&, const typename Derived::RealScalar&) + EIGEN_DEVICE_FUNC + static bool run(const Derived &x, const typename Derived::RealScalar &y, const typename Derived::RealScalar &prec) + { + return x.cwiseAbs2().sum() <= numext::abs2(prec * y); + } + }; + + template struct isMuchSmallerThan_scalar_selector { - return x.matrix() == Derived::Zero(x.rows(), x.cols()).matrix(); - } -}; + EIGEN_DEVICE_FUNC + static bool run(const Derived &x, const typename Derived::RealScalar &, const typename Derived::RealScalar &) + { + return x.matrix() == Derived::Zero(x.rows(), x.cols()).matrix(); + } + }; -} // end namespace internal +}// end namespace internal /** \returns \c true if \c *this is approximately equal to \a other, within the precision - * determined by \a prec. - * - * \note The fuzzy compares are done multiplicatively. Two vectors \f$ v \f$ and \f$ w \f$ - * are considered to be approximately equal within precision \f$ p \f$ if - * \f[ \Vert v - w \Vert \leqslant p\,\min(\Vert v\Vert, \Vert w\Vert). \f] - * For matrices, the comparison is done using the Hilbert-Schmidt norm (aka Frobenius norm - * L2 norm). - * - * \note Because of the multiplicativeness of this comparison, one can't use this function - * to check whether \c *this is approximately equal to the zero matrix or vector. - * Indeed, \c isApprox(zero) returns false unless \c *this itself is exactly the zero matrix - * or vector. If you want to test whether \c *this is zero, use internal::isMuchSmallerThan(const - * RealScalar&, RealScalar) instead. - * - * \sa internal::isMuchSmallerThan(const RealScalar&, RealScalar) const - */ + * determined by \a prec. + * + * \note The fuzzy compares are done multiplicatively. Two vectors \f$ v \f$ and \f$ w \f$ + * are considered to be approximately equal within precision \f$ p \f$ if + * \f[ \Vert v - w \Vert \leqslant p\,\min(\Vert v\Vert, \Vert w\Vert). \f] + * For matrices, the comparison is done using the Hilbert-Schmidt norm (aka Frobenius norm + * L2 norm). + * + * \note Because of the multiplicativeness of this comparison, one can't use this function + * to check whether \c *this is approximately equal to the zero matrix or vector. + * Indeed, \c isApprox(zero) returns false unless \c *this itself is exactly the zero matrix + * or vector. If you want to test whether \c *this is zero, use internal::isMuchSmallerThan(const + * RealScalar&, RealScalar) instead. + * + * \sa internal::isMuchSmallerThan(const RealScalar&, RealScalar) const + */ template template -bool DenseBase::isApprox( - const DenseBase& other, - const RealScalar& prec -) const +bool DenseBase::isApprox(const DenseBase &other, const RealScalar &prec) const { return internal::isApprox_selector::run(derived(), other.derived(), prec); } /** \returns \c true if the norm of \c *this is much smaller than \a other, - * within the precision determined by \a prec. - * - * \note The fuzzy compares are done multiplicatively. A vector \f$ v \f$ is - * considered to be much smaller than \f$ x \f$ within precision \f$ p \f$ if - * \f[ \Vert v \Vert \leqslant p\,\vert x\vert. \f] - * - * For matrices, the comparison is done using the Hilbert-Schmidt norm. For this reason, - * the value of the reference scalar \a other should come from the Hilbert-Schmidt norm - * of a reference matrix of same dimensions. - * - * \sa isApprox(), isMuchSmallerThan(const DenseBase&, RealScalar) const - */ + * within the precision determined by \a prec. + * + * \note The fuzzy compares are done multiplicatively. A vector \f$ v \f$ is + * considered to be much smaller than \f$ x \f$ within precision \f$ p \f$ if + * \f[ \Vert v \Vert \leqslant p\,\vert x\vert. \f] + * + * For matrices, the comparison is done using the Hilbert-Schmidt norm. For this reason, + * the value of the reference scalar \a other should come from the Hilbert-Schmidt norm + * of a reference matrix of same dimensions. + * + * \sa isApprox(), isMuchSmallerThan(const DenseBase&, RealScalar) const + */ template -bool DenseBase::isMuchSmallerThan( - const typename NumTraits::Real& other, - const RealScalar& prec -) const +bool DenseBase::isMuchSmallerThan(const typename NumTraits::Real &other, const RealScalar &prec) const { return internal::isMuchSmallerThan_scalar_selector::run(derived(), other, prec); } /** \returns \c true if the norm of \c *this is much smaller than the norm of \a other, - * within the precision determined by \a prec. - * - * \note The fuzzy compares are done multiplicatively. A vector \f$ v \f$ is - * considered to be much smaller than a vector \f$ w \f$ within precision \f$ p \f$ if - * \f[ \Vert v \Vert \leqslant p\,\Vert w\Vert. \f] - * For matrices, the comparison is done using the Hilbert-Schmidt norm. - * - * \sa isApprox(), isMuchSmallerThan(const RealScalar&, RealScalar) const - */ + * within the precision determined by \a prec. + * + * \note The fuzzy compares are done multiplicatively. A vector \f$ v \f$ is + * considered to be much smaller than a vector \f$ w \f$ within precision \f$ p \f$ if + * \f[ \Vert v \Vert \leqslant p\,\Vert w\Vert. \f] + * For matrices, the comparison is done using the Hilbert-Schmidt norm. + * + * \sa isApprox(), isMuchSmallerThan(const RealScalar&, RealScalar) const + */ template template -bool DenseBase::isMuchSmallerThan( - const DenseBase& other, - const RealScalar& prec -) const +bool DenseBase::isMuchSmallerThan(const DenseBase &other, const RealScalar &prec) const { return internal::isMuchSmallerThan_object_selector::run(derived(), other.derived(), prec); } -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_FUZZY_H +#endif// EIGEN_FUZZY_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/GeneralProduct.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/GeneralProduct.h index 6f0cc80e..93d57b82 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/GeneralProduct.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/GeneralProduct.h @@ -13,64 +13,55 @@ namespace Eigen { -enum { - Large = 2, - Small = 3 -}; +enum { Large = 2, Small = 3 }; namespace internal { -template struct product_type_selector; + template struct product_type_selector; -template struct product_size_category -{ - enum { - #ifndef EIGEN_CUDA_ARCH - is_large = MaxSize == Dynamic || - Size >= EIGEN_CACHEFRIENDLY_PRODUCT_THRESHOLD || - (Size==Dynamic && MaxSize>=EIGEN_CACHEFRIENDLY_PRODUCT_THRESHOLD), - #else - is_large = 0, - #endif - value = is_large ? Large - : Size == 1 ? 1 - : Small - }; -}; - -template struct product_type -{ - typedef typename remove_all::type _Lhs; - typedef typename remove_all::type _Rhs; - enum { - MaxRows = traits<_Lhs>::MaxRowsAtCompileTime, - Rows = traits<_Lhs>::RowsAtCompileTime, - MaxCols = traits<_Rhs>::MaxColsAtCompileTime, - Cols = traits<_Rhs>::ColsAtCompileTime, - MaxDepth = EIGEN_SIZE_MIN_PREFER_FIXED(traits<_Lhs>::MaxColsAtCompileTime, - traits<_Rhs>::MaxRowsAtCompileTime), - Depth = EIGEN_SIZE_MIN_PREFER_FIXED(traits<_Lhs>::ColsAtCompileTime, - traits<_Rhs>::RowsAtCompileTime) - }; - - // the splitting into different lines of code here, introducing the _select enums and the typedef below, - // is to work around an internal compiler error with gcc 4.1 and 4.2. -private: - enum { - rows_select = product_size_category::value, - cols_select = product_size_category::value, - depth_select = product_size_category::value + template struct product_size_category + { + enum { +#ifndef EIGEN_CUDA_ARCH + is_large = MaxSize == Dynamic || Size >= EIGEN_CACHEFRIENDLY_PRODUCT_THRESHOLD + || (Size == Dynamic && MaxSize >= EIGEN_CACHEFRIENDLY_PRODUCT_THRESHOLD), +#else + is_large = 0, +#endif + value = is_large ? Large + : Size == 1 ? 1 + : Small + }; }; - typedef product_type_selector selector; -public: - enum { - value = selector::ret, - ret = selector::ret - }; -#ifdef EIGEN_DEBUG_PRODUCT - static void debug() + template struct product_type { + typedef typename remove_all::type _Lhs; + typedef typename remove_all::type _Rhs; + enum { + MaxRows = traits<_Lhs>::MaxRowsAtCompileTime, + Rows = traits<_Lhs>::RowsAtCompileTime, + MaxCols = traits<_Rhs>::MaxColsAtCompileTime, + Cols = traits<_Rhs>::ColsAtCompileTime, + MaxDepth = EIGEN_SIZE_MIN_PREFER_FIXED(traits<_Lhs>::MaxColsAtCompileTime, traits<_Rhs>::MaxRowsAtCompileTime), + Depth = EIGEN_SIZE_MIN_PREFER_FIXED(traits<_Lhs>::ColsAtCompileTime, traits<_Rhs>::RowsAtCompileTime) + }; + + // the splitting into different lines of code here, introducing the _select enums and the typedef below, + // is to work around an internal compiler error with gcc 4.1 and 4.2. + private: + enum { + rows_select = product_size_category::value, + cols_select = product_size_category::value, + depth_select = product_size_category::value + }; + typedef product_type_selector selector; + + public: + enum { value = selector::ret, ret = selector::ret }; +#ifdef EIGEN_DEBUG_PRODUCT + static void debug() + { EIGEN_DEBUG_VAR(Rows); EIGEN_DEBUG_VAR(Cols); EIGEN_DEBUG_VAR(Depth); @@ -78,44 +69,116 @@ template struct product_type EIGEN_DEBUG_VAR(cols_select); EIGEN_DEBUG_VAR(depth_select); EIGEN_DEBUG_VAR(value); - } + } #endif -}; - -/* The following allows to select the kind of product at compile time - * based on the three dimensions of the product. - * This is a compile time mapping from {1,Small,Large}^3 -> {product types} */ -// FIXME I'm not sure the current mapping is the ideal one. -template struct product_type_selector { enum { ret = OuterProduct }; }; -template struct product_type_selector { enum { ret = LazyCoeffBasedProductMode }; }; -template struct product_type_selector<1, N, 1> { enum { ret = LazyCoeffBasedProductMode }; }; -template struct product_type_selector<1, 1, Depth> { enum { ret = InnerProduct }; }; -template<> struct product_type_selector<1, 1, 1> { enum { ret = InnerProduct }; }; -template<> struct product_type_selector { enum { ret = CoeffBasedProductMode }; }; -template<> struct product_type_selector<1, Small,Small> { enum { ret = CoeffBasedProductMode }; }; -template<> struct product_type_selector { enum { ret = CoeffBasedProductMode }; }; -template<> struct product_type_selector { enum { ret = LazyCoeffBasedProductMode }; }; -template<> struct product_type_selector { enum { ret = LazyCoeffBasedProductMode }; }; -template<> struct product_type_selector { enum { ret = LazyCoeffBasedProductMode }; }; -template<> struct product_type_selector<1, Large,Small> { enum { ret = CoeffBasedProductMode }; }; -template<> struct product_type_selector<1, Large,Large> { enum { ret = GemvProduct }; }; -template<> struct product_type_selector<1, Small,Large> { enum { ret = CoeffBasedProductMode }; }; -template<> struct product_type_selector { enum { ret = CoeffBasedProductMode }; }; -template<> struct product_type_selector { enum { ret = GemvProduct }; }; -template<> struct product_type_selector { enum { ret = CoeffBasedProductMode }; }; -template<> struct product_type_selector { enum { ret = GemmProduct }; }; -template<> struct product_type_selector { enum { ret = GemmProduct }; }; -template<> struct product_type_selector { enum { ret = GemmProduct }; }; -template<> struct product_type_selector { enum { ret = GemmProduct }; }; -template<> struct product_type_selector { enum { ret = CoeffBasedProductMode }; }; -template<> struct product_type_selector { enum { ret = CoeffBasedProductMode }; }; -template<> struct product_type_selector { enum { ret = GemmProduct }; }; - -} // end namespace internal + }; + + /* The following allows to select the kind of product at compile time + * based on the three dimensions of the product. + * This is a compile time mapping from {1,Small,Large}^3 -> {product types} */ + // FIXME I'm not sure the current mapping is the ideal one. + template struct product_type_selector + { + enum { ret = OuterProduct }; + }; + template struct product_type_selector + { + enum { ret = LazyCoeffBasedProductMode }; + }; + template struct product_type_selector<1, N, 1> + { + enum { ret = LazyCoeffBasedProductMode }; + }; + template struct product_type_selector<1, 1, Depth> + { + enum { ret = InnerProduct }; + }; + template<> struct product_type_selector<1, 1, 1> + { + enum { ret = InnerProduct }; + }; + template<> struct product_type_selector + { + enum { ret = CoeffBasedProductMode }; + }; + template<> struct product_type_selector<1, Small, Small> + { + enum { ret = CoeffBasedProductMode }; + }; + template<> struct product_type_selector + { + enum { ret = CoeffBasedProductMode }; + }; + template<> struct product_type_selector + { + enum { ret = LazyCoeffBasedProductMode }; + }; + template<> struct product_type_selector + { + enum { ret = LazyCoeffBasedProductMode }; + }; + template<> struct product_type_selector + { + enum { ret = LazyCoeffBasedProductMode }; + }; + template<> struct product_type_selector<1, Large, Small> + { + enum { ret = CoeffBasedProductMode }; + }; + template<> struct product_type_selector<1, Large, Large> + { + enum { ret = GemvProduct }; + }; + template<> struct product_type_selector<1, Small, Large> + { + enum { ret = CoeffBasedProductMode }; + }; + template<> struct product_type_selector + { + enum { ret = CoeffBasedProductMode }; + }; + template<> struct product_type_selector + { + enum { ret = GemvProduct }; + }; + template<> struct product_type_selector + { + enum { ret = CoeffBasedProductMode }; + }; + template<> struct product_type_selector + { + enum { ret = GemmProduct }; + }; + template<> struct product_type_selector + { + enum { ret = GemmProduct }; + }; + template<> struct product_type_selector + { + enum { ret = GemmProduct }; + }; + template<> struct product_type_selector + { + enum { ret = GemmProduct }; + }; + template<> struct product_type_selector + { + enum { ret = CoeffBasedProductMode }; + }; + template<> struct product_type_selector + { + enum { ret = CoeffBasedProductMode }; + }; + template<> struct product_type_selector + { + enum { ret = GemmProduct }; + }; + +}// end namespace internal /*********************************************************************** -* Implementation of Inner Vector Vector Product -***********************************************************************/ + * Implementation of Inner Vector Vector Product + ***********************************************************************/ // FIXME : maybe the "inner product" could return a Scalar // instead of a 1x1 matrix ?? @@ -125,12 +188,12 @@ template<> struct product_type_selector { enum // case, we could have a specialization for Block with: operator=(Scalar x); /*********************************************************************** -* Implementation of Outer Vector Vector Product -***********************************************************************/ + * Implementation of Outer Vector Vector Product + ***********************************************************************/ /*********************************************************************** -* Implementation of General Matrix Vector Product -***********************************************************************/ + * Implementation of General Matrix Vector Product + ***********************************************************************/ /* According to the shape/flags of the matrix we have to distinghish 3 different cases: * 1 - the matrix is col-major, BLAS compatible and M is large => call fast BLAS-like colmajor routine @@ -141,264 +204,293 @@ template<> struct product_type_selector { enum */ namespace internal { -template -struct gemv_dense_selector; + template struct gemv_dense_selector; -} // end namespace internal +}// end namespace internal namespace internal { -template struct gemv_static_vector_if; - -template -struct gemv_static_vector_if -{ - EIGEN_STRONG_INLINE Scalar* data() { eigen_internal_assert(false && "should never be called"); return 0; } -}; + template struct gemv_static_vector_if; -template -struct gemv_static_vector_if -{ - EIGEN_STRONG_INLINE Scalar* data() { return 0; } -}; - -template -struct gemv_static_vector_if -{ - enum { - ForceAlignment = internal::packet_traits::Vectorizable, - PacketSize = internal::packet_traits::size - }; - #if EIGEN_MAX_STATIC_ALIGN_BYTES!=0 - internal::plain_array m_data; - EIGEN_STRONG_INLINE Scalar* data() { return m_data.array; } - #else - // Some architectures cannot align on the stack, - // => let's manually enforce alignment by allocating more data and return the address of the first aligned element. - internal::plain_array m_data; - EIGEN_STRONG_INLINE Scalar* data() { - return ForceAlignment - ? reinterpret_cast((internal::UIntPtr(m_data.array) & ~(std::size_t(EIGEN_MAX_ALIGN_BYTES-1))) + EIGEN_MAX_ALIGN_BYTES) - : m_data.array; - } - #endif -}; - -// The vector is on the left => transposition -template -struct gemv_dense_selector -{ - template - static void run(const Lhs &lhs, const Rhs &rhs, Dest& dest, const typename Dest::Scalar& alpha) + template struct gemv_static_vector_if { - Transpose destT(dest); - enum { OtherStorageOrder = StorageOrder == RowMajor ? ColMajor : RowMajor }; - gemv_dense_selector - ::run(rhs.transpose(), lhs.transpose(), destT, alpha); - } -}; + EIGEN_STRONG_INLINE Scalar *data() + { + eigen_internal_assert(false && "should never be called"); + return 0; + } + }; -template<> struct gemv_dense_selector -{ - template - static inline void run(const Lhs &lhs, const Rhs &rhs, Dest& dest, const typename Dest::Scalar& alpha) - { - typedef typename Lhs::Scalar LhsScalar; - typedef typename Rhs::Scalar RhsScalar; - typedef typename Dest::Scalar ResScalar; - typedef typename Dest::RealScalar RealScalar; - - typedef internal::blas_traits LhsBlasTraits; - typedef typename LhsBlasTraits::DirectLinearAccessType ActualLhsType; - typedef internal::blas_traits RhsBlasTraits; - typedef typename RhsBlasTraits::DirectLinearAccessType ActualRhsType; - - typedef Map, EIGEN_PLAIN_ENUM_MIN(AlignedMax,internal::packet_traits::size)> MappedDest; - - ActualLhsType actualLhs = LhsBlasTraits::extract(lhs); - ActualRhsType actualRhs = RhsBlasTraits::extract(rhs); - - ResScalar actualAlpha = alpha * LhsBlasTraits::extractScalarFactor(lhs) - * RhsBlasTraits::extractScalarFactor(rhs); - - // make sure Dest is a compile-time vector type (bug 1166) - typedef typename conditional::type ActualDest; + template struct gemv_static_vector_if + { + EIGEN_STRONG_INLINE Scalar *data() { return 0; } + }; + template struct gemv_static_vector_if + { enum { - // FIXME find a way to allow an inner stride on the result if packet_traits::size==1 - // on, the other hand it is good for the cache to pack the vector anyways... - EvalToDestAtCompileTime = (ActualDest::InnerStrideAtCompileTime==1), - ComplexByReal = (NumTraits::IsComplex) && (!NumTraits::IsComplex), - MightCannotUseDest = (!EvalToDestAtCompileTime) || ComplexByReal + ForceAlignment = internal::packet_traits::Vectorizable, + PacketSize = internal::packet_traits::size }; +#if EIGEN_MAX_STATIC_ALIGN_BYTES != 0 + internal:: + plain_array + m_data; + EIGEN_STRONG_INLINE Scalar *data() { return m_data.array; } +#else + // Some architectures cannot align on the stack, + // => let's manually enforce alignment by allocating more data and return the address of the first aligned element. + internal:: + plain_array + m_data; + EIGEN_STRONG_INLINE Scalar *data() + { + return ForceAlignment ? reinterpret_cast( + (internal::UIntPtr(m_data.array) & ~(std::size_t(EIGEN_MAX_ALIGN_BYTES - 1))) + + EIGEN_MAX_ALIGN_BYTES) + : m_data.array; + } +#endif + }; - typedef const_blas_data_mapper LhsMapper; - typedef const_blas_data_mapper RhsMapper; - RhsScalar compatibleAlpha = get_factor::run(actualAlpha); + // The vector is on the left => transposition + template struct gemv_dense_selector + { + template + static void run(const Lhs &lhs, const Rhs &rhs, Dest &dest, const typename Dest::Scalar &alpha) + { + Transpose destT(dest); + enum { OtherStorageOrder = StorageOrder == RowMajor ? ColMajor : RowMajor }; + gemv_dense_selector::run( + rhs.transpose(), lhs.transpose(), destT, alpha); + } + }; - if(!MightCannotUseDest) + template<> struct gemv_dense_selector + { + template + static inline void run(const Lhs &lhs, const Rhs &rhs, Dest &dest, const typename Dest::Scalar &alpha) { - // shortcut if we are sure to be able to use dest directly, - // this ease the compiler to generate cleaner and more optimzized code for most common cases - general_matrix_vector_product - ::run( - actualLhs.rows(), actualLhs.cols(), + typedef typename Lhs::Scalar LhsScalar; + typedef typename Rhs::Scalar RhsScalar; + typedef typename Dest::Scalar ResScalar; + typedef typename Dest::RealScalar RealScalar; + + typedef internal::blas_traits LhsBlasTraits; + typedef typename LhsBlasTraits::DirectLinearAccessType ActualLhsType; + typedef internal::blas_traits RhsBlasTraits; + typedef typename RhsBlasTraits::DirectLinearAccessType ActualRhsType; + + typedef Map, + EIGEN_PLAIN_ENUM_MIN(AlignedMax, internal::packet_traits::size)> + MappedDest; + + ActualLhsType actualLhs = LhsBlasTraits::extract(lhs); + ActualRhsType actualRhs = RhsBlasTraits::extract(rhs); + + ResScalar actualAlpha = alpha * LhsBlasTraits::extractScalarFactor(lhs) * RhsBlasTraits::extractScalarFactor(rhs); + + // make sure Dest is a compile-time vector type (bug 1166) + typedef typename conditional::type ActualDest; + + enum { + // FIXME find a way to allow an inner stride on the result if packet_traits::size==1 + // on, the other hand it is good for the cache to pack the vector anyways... + EvalToDestAtCompileTime = (ActualDest::InnerStrideAtCompileTime == 1), + ComplexByReal = (NumTraits::IsComplex) && (!NumTraits::IsComplex), + MightCannotUseDest = (!EvalToDestAtCompileTime) || ComplexByReal + }; + + typedef const_blas_data_mapper LhsMapper; + typedef const_blas_data_mapper RhsMapper; + RhsScalar compatibleAlpha = get_factor::run(actualAlpha); + + if (!MightCannotUseDest) { + // shortcut if we are sure to be able to use dest directly, + // this ease the compiler to generate cleaner and more optimzized code for most common cases + general_matrix_vector_product::run(actualLhs.rows(), + actualLhs.cols(), LhsMapper(actualLhs.data(), actualLhs.outerStride()), RhsMapper(actualRhs.data(), actualRhs.innerStride()), - dest.data(), 1, + dest.data(), + 1, compatibleAlpha); - } - else - { - gemv_static_vector_if static_dest; - - const bool alphaIsCompatible = (!ComplexByReal) || (numext::imag(actualAlpha)==RealScalar(0)); - const bool evalToDest = EvalToDestAtCompileTime && alphaIsCompatible; - - ei_declare_aligned_stack_constructed_variable(ResScalar,actualDestPtr,dest.size(), - evalToDest ? dest.data() : static_dest.data()); - - if(!evalToDest) - { - #ifdef EIGEN_DENSE_STORAGE_CTOR_PLUGIN - Index size = dest.size(); - EIGEN_DENSE_STORAGE_CTOR_PLUGIN - #endif - if(!alphaIsCompatible) - { - MappedDest(actualDestPtr, dest.size()).setZero(); - compatibleAlpha = RhsScalar(1); + } else { + gemv_static_vector_if + static_dest; + + const bool alphaIsCompatible = (!ComplexByReal) || (numext::imag(actualAlpha) == RealScalar(0)); + const bool evalToDest = EvalToDestAtCompileTime && alphaIsCompatible; + + ei_declare_aligned_stack_constructed_variable( + ResScalar, actualDestPtr, dest.size(), evalToDest ? dest.data() : static_dest.data()); + + if (!evalToDest) { +#ifdef EIGEN_DENSE_STORAGE_CTOR_PLUGIN + Index size = dest.size(); + EIGEN_DENSE_STORAGE_CTOR_PLUGIN +#endif + if (!alphaIsCompatible) { + MappedDest(actualDestPtr, dest.size()).setZero(); + compatibleAlpha = RhsScalar(1); + } else + MappedDest(actualDestPtr, dest.size()) = dest; } - else - MappedDest(actualDestPtr, dest.size()) = dest; - } - general_matrix_vector_product - ::run( - actualLhs.rows(), actualLhs.cols(), + general_matrix_vector_product::run(actualLhs.rows(), + actualLhs.cols(), LhsMapper(actualLhs.data(), actualLhs.outerStride()), RhsMapper(actualRhs.data(), actualRhs.innerStride()), - actualDestPtr, 1, + actualDestPtr, + 1, compatibleAlpha); - if (!evalToDest) - { - if(!alphaIsCompatible) - dest.matrix() += actualAlpha * MappedDest(actualDestPtr, dest.size()); - else - dest = MappedDest(actualDestPtr, dest.size()); + if (!evalToDest) { + if (!alphaIsCompatible) + dest.matrix() += actualAlpha * MappedDest(actualDestPtr, dest.size()); + else + dest = MappedDest(actualDestPtr, dest.size()); + } } } - } -}; + }; -template<> struct gemv_dense_selector -{ - template - static void run(const Lhs &lhs, const Rhs &rhs, Dest& dest, const typename Dest::Scalar& alpha) + template<> struct gemv_dense_selector { - typedef typename Lhs::Scalar LhsScalar; - typedef typename Rhs::Scalar RhsScalar; - typedef typename Dest::Scalar ResScalar; - - typedef internal::blas_traits LhsBlasTraits; - typedef typename LhsBlasTraits::DirectLinearAccessType ActualLhsType; - typedef internal::blas_traits RhsBlasTraits; - typedef typename RhsBlasTraits::DirectLinearAccessType ActualRhsType; - typedef typename internal::remove_all::type ActualRhsTypeCleaned; - - typename add_const::type actualLhs = LhsBlasTraits::extract(lhs); - typename add_const::type actualRhs = RhsBlasTraits::extract(rhs); - - ResScalar actualAlpha = alpha * LhsBlasTraits::extractScalarFactor(lhs) - * RhsBlasTraits::extractScalarFactor(rhs); - - enum { - // FIXME find a way to allow an inner stride on the result if packet_traits::size==1 - // on, the other hand it is good for the cache to pack the vector anyways... - DirectlyUseRhs = ActualRhsTypeCleaned::InnerStrideAtCompileTime==1 - }; - - gemv_static_vector_if static_rhs; - - ei_declare_aligned_stack_constructed_variable(RhsScalar,actualRhsPtr,actualRhs.size(), - DirectlyUseRhs ? const_cast(actualRhs.data()) : static_rhs.data()); - - if(!DirectlyUseRhs) + template + static void run(const Lhs &lhs, const Rhs &rhs, Dest &dest, const typename Dest::Scalar &alpha) { - #ifdef EIGEN_DENSE_STORAGE_CTOR_PLUGIN - Index size = actualRhs.size(); - EIGEN_DENSE_STORAGE_CTOR_PLUGIN - #endif - Map(actualRhsPtr, actualRhs.size()) = actualRhs; - } + typedef typename Lhs::Scalar LhsScalar; + typedef typename Rhs::Scalar RhsScalar; + typedef typename Dest::Scalar ResScalar; + + typedef internal::blas_traits LhsBlasTraits; + typedef typename LhsBlasTraits::DirectLinearAccessType ActualLhsType; + typedef internal::blas_traits RhsBlasTraits; + typedef typename RhsBlasTraits::DirectLinearAccessType ActualRhsType; + typedef typename internal::remove_all::type ActualRhsTypeCleaned; + + typename add_const::type actualLhs = LhsBlasTraits::extract(lhs); + typename add_const::type actualRhs = RhsBlasTraits::extract(rhs); + + ResScalar actualAlpha = alpha * LhsBlasTraits::extractScalarFactor(lhs) * RhsBlasTraits::extractScalarFactor(rhs); + + enum { + // FIXME find a way to allow an inner stride on the result if packet_traits::size==1 + // on, the other hand it is good for the cache to pack the vector anyways... + DirectlyUseRhs = ActualRhsTypeCleaned::InnerStrideAtCompileTime == 1 + }; + + gemv_static_vector_if + static_rhs; + + ei_declare_aligned_stack_constructed_variable(RhsScalar, + actualRhsPtr, + actualRhs.size(), + DirectlyUseRhs ? const_cast(actualRhs.data()) : static_rhs.data()); + + if (!DirectlyUseRhs) { +#ifdef EIGEN_DENSE_STORAGE_CTOR_PLUGIN + Index size = actualRhs.size(); + EIGEN_DENSE_STORAGE_CTOR_PLUGIN +#endif + Map(actualRhsPtr, actualRhs.size()) = actualRhs; + } - typedef const_blas_data_mapper LhsMapper; - typedef const_blas_data_mapper RhsMapper; - general_matrix_vector_product - ::run( - actualLhs.rows(), actualLhs.cols(), + typedef const_blas_data_mapper LhsMapper; + typedef const_blas_data_mapper RhsMapper; + general_matrix_vector_product::run(actualLhs.rows(), + actualLhs.cols(), LhsMapper(actualLhs.data(), actualLhs.outerStride()), RhsMapper(actualRhsPtr, 1), - dest.data(), dest.col(0).innerStride(), //NOTE if dest is not a vector at compile-time, then dest.innerStride() might be wrong. (bug 1166) + dest.data(), + dest.col(0).innerStride(),// NOTE if dest is not a vector at compile-time, then dest.innerStride() might be + // wrong. (bug 1166) actualAlpha); - } -}; + } + }; -template<> struct gemv_dense_selector -{ - template - static void run(const Lhs &lhs, const Rhs &rhs, Dest& dest, const typename Dest::Scalar& alpha) - { - EIGEN_STATIC_ASSERT((!nested_eval::Evaluate),EIGEN_INTERNAL_COMPILATION_ERROR_OR_YOU_MADE_A_PROGRAMMING_MISTAKE); - // TODO if rhs is large enough it might be beneficial to make sure that dest is sequentially stored in memory, otherwise use a temp - typename nested_eval::type actual_rhs(rhs); - const Index size = rhs.rows(); - for(Index k=0; k struct gemv_dense_selector -{ - template - static void run(const Lhs &lhs, const Rhs &rhs, Dest& dest, const typename Dest::Scalar& alpha) + template<> struct gemv_dense_selector + { + template + static void run(const Lhs &lhs, const Rhs &rhs, Dest &dest, const typename Dest::Scalar &alpha) + { + EIGEN_STATIC_ASSERT( + (!nested_eval::Evaluate), EIGEN_INTERNAL_COMPILATION_ERROR_OR_YOU_MADE_A_PROGRAMMING_MISTAKE); + // TODO if rhs is large enough it might be beneficial to make sure that dest is sequentially stored in memory, + // otherwise use a temp + typename nested_eval::type actual_rhs(rhs); + const Index size = rhs.rows(); + for (Index k = 0; k < size; ++k) dest += (alpha * actual_rhs.coeff(k)) * lhs.col(k); + } + }; + + template<> struct gemv_dense_selector { - EIGEN_STATIC_ASSERT((!nested_eval::Evaluate),EIGEN_INTERNAL_COMPILATION_ERROR_OR_YOU_MADE_A_PROGRAMMING_MISTAKE); - typename nested_eval::type actual_rhs(rhs); - const Index rows = dest.rows(); - for(Index i=0; i + static void run(const Lhs &lhs, const Rhs &rhs, Dest &dest, const typename Dest::Scalar &alpha) + { + EIGEN_STATIC_ASSERT( + (!nested_eval::Evaluate), EIGEN_INTERNAL_COMPILATION_ERROR_OR_YOU_MADE_A_PROGRAMMING_MISTAKE); + typename nested_eval::type actual_rhs(rhs); + const Index rows = dest.rows(); + for (Index i = 0; i < rows; ++i) + dest.coeffRef(i) += alpha * (lhs.row(i).cwiseProduct(actual_rhs.transpose())).sum(); + } + }; -} // end namespace internal +}// end namespace internal /*************************************************************************** -* Implementation of matrix base methods -***************************************************************************/ + * Implementation of matrix base methods + ***************************************************************************/ /** \returns the matrix product of \c *this and \a other. - * - * \note If instead of the matrix product you want the coefficient-wise product, see Cwise::operator*(). - * - * \sa lazyProduct(), operator*=(const MatrixBase&), Cwise::operator*() - */ + * + * \note If instead of the matrix product you want the coefficient-wise product, see Cwise::operator*(). + * + * \sa lazyProduct(), operator*=(const MatrixBase&), Cwise::operator*() + */ template template -inline const Product -MatrixBase::operator*(const MatrixBase &other) const +inline const Product MatrixBase::operator*(const MatrixBase &other) const { // A note regarding the function declaration: In MSVC, this function will sometimes // not be inlined since DenseStorage is an unwindable object for dynamic // matrices and product types are holding a member to store the result. // Thus it does not help tagging this function with EIGEN_STRONG_INLINE. enum { - ProductIsValid = Derived::ColsAtCompileTime==Dynamic - || OtherDerived::RowsAtCompileTime==Dynamic - || int(Derived::ColsAtCompileTime)==int(OtherDerived::RowsAtCompileTime), + ProductIsValid = Derived::ColsAtCompileTime == Dynamic || OtherDerived::RowsAtCompileTime == Dynamic + || int(Derived::ColsAtCompileTime) == int(OtherDerived::RowsAtCompileTime), AreVectors = Derived::IsVectorAtCompileTime && OtherDerived::IsVectorAtCompileTime, - SameSizes = EIGEN_PREDICATE_SAME_MATRIX_SIZE(Derived,OtherDerived) + SameSizes = EIGEN_PREDICATE_SAME_MATRIX_SIZE(Derived, OtherDerived) }; // note to the lost user: // * for a dot product use: v1.dot(v2) @@ -409,34 +501,33 @@ MatrixBase::operator*(const MatrixBase &other) const INVALID_MATRIX_PRODUCT__IF_YOU_WANTED_A_COEFF_WISE_PRODUCT_YOU_MUST_USE_THE_EXPLICIT_FUNCTION) EIGEN_STATIC_ASSERT(ProductIsValid || SameSizes, INVALID_MATRIX_PRODUCT) #ifdef EIGEN_DEBUG_PRODUCT - internal::product_type::debug(); + internal::product_type::debug(); #endif return Product(derived(), other.derived()); } /** \returns an expression of the matrix product of \c *this and \a other without implicit evaluation. - * - * The returned product will behave like any other expressions: the coefficients of the product will be - * computed once at a time as requested. This might be useful in some extremely rare cases when only - * a small and no coherent fraction of the result's coefficients have to be computed. - * - * \warning This version of the matrix product can be much much slower. So use it only if you know - * what you are doing and that you measured a true speed improvement. - * - * \sa operator*(const MatrixBase&) - */ + * + * The returned product will behave like any other expressions: the coefficients of the product will be + * computed once at a time as requested. This might be useful in some extremely rare cases when only + * a small and no coherent fraction of the result's coefficients have to be computed. + * + * \warning This version of the matrix product can be much much slower. So use it only if you know + * what you are doing and that you measured a true speed improvement. + * + * \sa operator*(const MatrixBase&) + */ template template -const Product -MatrixBase::lazyProduct(const MatrixBase &other) const +const Product MatrixBase::lazyProduct( + const MatrixBase &other) const { enum { - ProductIsValid = Derived::ColsAtCompileTime==Dynamic - || OtherDerived::RowsAtCompileTime==Dynamic - || int(Derived::ColsAtCompileTime)==int(OtherDerived::RowsAtCompileTime), + ProductIsValid = Derived::ColsAtCompileTime == Dynamic || OtherDerived::RowsAtCompileTime == Dynamic + || int(Derived::ColsAtCompileTime) == int(OtherDerived::RowsAtCompileTime), AreVectors = Derived::IsVectorAtCompileTime && OtherDerived::IsVectorAtCompileTime, - SameSizes = EIGEN_PREDICATE_SAME_MATRIX_SIZE(Derived,OtherDerived) + SameSizes = EIGEN_PREDICATE_SAME_MATRIX_SIZE(Derived, OtherDerived) }; // note to the lost user: // * for a dot product use: v1.dot(v2) @@ -447,9 +538,9 @@ MatrixBase::lazyProduct(const MatrixBase &other) const INVALID_MATRIX_PRODUCT__IF_YOU_WANTED_A_COEFF_WISE_PRODUCT_YOU_MUST_USE_THE_EXPLICIT_FUNCTION) EIGEN_STATIC_ASSERT(ProductIsValid || SameSizes, INVALID_MATRIX_PRODUCT) - return Product(derived(), other.derived()); + return Product(derived(), other.derived()); } -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_PRODUCT_H +#endif// EIGEN_PRODUCT_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/GenericPacketMath.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/GenericPacketMath.h index 029f8ac3..410f1e60 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/GenericPacketMath.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/GenericPacketMath.h @@ -15,13 +15,13 @@ namespace Eigen { namespace internal { -/** \internal - * \file GenericPacketMath.h - * - * Default implementation for types not supported by the vectorization. - * In practice these functions are provided to make easier the writing - * of generic vectorized code. - */ + /** \internal + * \file GenericPacketMath.h + * + * Default implementation for types not supported by the vectorization. + * In practice these functions are provided to make easier the writing + * of generic vectorized code. + */ #ifndef EIGEN_DEBUG_ALIGNED_LOAD #define EIGEN_DEBUG_ALIGNED_LOAD @@ -39,555 +39,637 @@ namespace internal { #define EIGEN_DEBUG_UNALIGNED_STORE #endif -struct default_packet_traits -{ - enum { - HasHalfPacket = 0, - - HasAdd = 1, - HasSub = 1, - HasMul = 1, - HasNegate = 1, - HasAbs = 1, - HasArg = 0, - HasAbs2 = 1, - HasMin = 1, - HasMax = 1, - HasConj = 1, - HasSetLinear = 1, - HasBlend = 0, - - HasDiv = 0, - HasSqrt = 0, - HasRsqrt = 0, - HasExp = 0, - HasLog = 0, - HasLog1p = 0, - HasLog10 = 0, - HasPow = 0, - - HasSin = 0, - HasCos = 0, - HasTan = 0, - HasASin = 0, - HasACos = 0, - HasATan = 0, - HasSinh = 0, - HasCosh = 0, - HasTanh = 0, - HasLGamma = 0, - HasDiGamma = 0, - HasZeta = 0, - HasPolygamma = 0, - HasErf = 0, - HasErfc = 0, - HasIGamma = 0, - HasIGammac = 0, - HasBetaInc = 0, - - HasRound = 0, - HasFloor = 0, - HasCeil = 0, - - HasSign = 0 + struct default_packet_traits + { + enum { + HasHalfPacket = 0, + + HasAdd = 1, + HasSub = 1, + HasMul = 1, + HasNegate = 1, + HasAbs = 1, + HasArg = 0, + HasAbs2 = 1, + HasMin = 1, + HasMax = 1, + HasConj = 1, + HasSetLinear = 1, + HasBlend = 0, + + HasDiv = 0, + HasSqrt = 0, + HasRsqrt = 0, + HasExp = 0, + HasLog = 0, + HasLog1p = 0, + HasLog10 = 0, + HasPow = 0, + + HasSin = 0, + HasCos = 0, + HasTan = 0, + HasASin = 0, + HasACos = 0, + HasATan = 0, + HasSinh = 0, + HasCosh = 0, + HasTanh = 0, + HasLGamma = 0, + HasDiGamma = 0, + HasZeta = 0, + HasPolygamma = 0, + HasErf = 0, + HasErfc = 0, + HasIGamma = 0, + HasIGammac = 0, + HasBetaInc = 0, + + HasRound = 0, + HasFloor = 0, + HasCeil = 0, + + HasSign = 0 + }; }; -}; - -template struct packet_traits : default_packet_traits -{ - typedef T type; - typedef T half; - enum { - Vectorizable = 0, - size = 1, - AlignedOnScalar = 0, - HasHalfPacket = 0 - }; - enum { - HasAdd = 0, - HasSub = 0, - HasMul = 0, - HasNegate = 0, - HasAbs = 0, - HasAbs2 = 0, - HasMin = 0, - HasMax = 0, - HasConj = 0, - HasSetLinear = 0 + + template struct packet_traits : default_packet_traits + { + typedef T type; + typedef T half; + enum { Vectorizable = 0, size = 1, AlignedOnScalar = 0, HasHalfPacket = 0 }; + enum { + HasAdd = 0, + HasSub = 0, + HasMul = 0, + HasNegate = 0, + HasAbs = 0, + HasAbs2 = 0, + HasMin = 0, + HasMax = 0, + HasConj = 0, + HasSetLinear = 0 + }; }; -}; -template struct packet_traits : packet_traits { }; + template struct packet_traits : packet_traits + { + }; -template struct type_casting_traits { - enum { - VectorizedCast = 0, - SrcCoeffRatio = 1, - TgtCoeffRatio = 1 + template struct type_casting_traits + { + enum { VectorizedCast = 0, SrcCoeffRatio = 1, TgtCoeffRatio = 1 }; }; -}; - - -/** \internal \returns static_cast(a) (coeff-wise) */ -template -EIGEN_DEVICE_FUNC inline TgtPacket -pcast(const SrcPacket& a) { - return static_cast(a); -} -template -EIGEN_DEVICE_FUNC inline TgtPacket -pcast(const SrcPacket& a, const SrcPacket& /*b*/) { - return static_cast(a); -} - -template -EIGEN_DEVICE_FUNC inline TgtPacket -pcast(const SrcPacket& a, const SrcPacket& /*b*/, const SrcPacket& /*c*/, const SrcPacket& /*d*/) { - return static_cast(a); -} - -/** \internal \returns a + b (coeff-wise) */ -template EIGEN_DEVICE_FUNC inline Packet -padd(const Packet& a, - const Packet& b) { return a+b; } - -/** \internal \returns a - b (coeff-wise) */ -template EIGEN_DEVICE_FUNC inline Packet -psub(const Packet& a, - const Packet& b) { return a-b; } - -/** \internal \returns -a (coeff-wise) */ -template EIGEN_DEVICE_FUNC inline Packet -pnegate(const Packet& a) { return -a; } - -/** \internal \returns conj(a) (coeff-wise) */ - -template EIGEN_DEVICE_FUNC inline Packet -pconj(const Packet& a) { return numext::conj(a); } - -/** \internal \returns a * b (coeff-wise) */ -template EIGEN_DEVICE_FUNC inline Packet -pmul(const Packet& a, - const Packet& b) { return a*b; } - -/** \internal \returns a / b (coeff-wise) */ -template EIGEN_DEVICE_FUNC inline Packet -pdiv(const Packet& a, - const Packet& b) { return a/b; } - -/** \internal \returns the min of \a a and \a b (coeff-wise) */ -template EIGEN_DEVICE_FUNC inline Packet -pmin(const Packet& a, - const Packet& b) { return numext::mini(a, b); } - -/** \internal \returns the max of \a a and \a b (coeff-wise) */ -template EIGEN_DEVICE_FUNC inline Packet -pmax(const Packet& a, - const Packet& b) { return numext::maxi(a, b); } - -/** \internal \returns the absolute value of \a a */ -template EIGEN_DEVICE_FUNC inline Packet -pabs(const Packet& a) { using std::abs; return abs(a); } - -/** \internal \returns the phase angle of \a a */ -template EIGEN_DEVICE_FUNC inline Packet -parg(const Packet& a) { using numext::arg; return arg(a); } - -/** \internal \returns the bitwise and of \a a and \a b */ -template EIGEN_DEVICE_FUNC inline Packet -pand(const Packet& a, const Packet& b) { return a & b; } - -/** \internal \returns the bitwise or of \a a and \a b */ -template EIGEN_DEVICE_FUNC inline Packet -por(const Packet& a, const Packet& b) { return a | b; } - -/** \internal \returns the bitwise xor of \a a and \a b */ -template EIGEN_DEVICE_FUNC inline Packet -pxor(const Packet& a, const Packet& b) { return a ^ b; } - -/** \internal \returns the bitwise andnot of \a a and \a b */ -template EIGEN_DEVICE_FUNC inline Packet -pandnot(const Packet& a, const Packet& b) { return a & (!b); } - -/** \internal \returns a packet version of \a *from, from must be 16 bytes aligned */ -template EIGEN_DEVICE_FUNC inline Packet -pload(const typename unpacket_traits::type* from) { return *from; } - -/** \internal \returns a packet version of \a *from, (un-aligned load) */ -template EIGEN_DEVICE_FUNC inline Packet -ploadu(const typename unpacket_traits::type* from) { return *from; } - -/** \internal \returns a packet with constant coefficients \a a, e.g.: (a,a,a,a) */ -template EIGEN_DEVICE_FUNC inline Packet -pset1(const typename unpacket_traits::type& a) { return a; } - -/** \internal \returns a packet with constant coefficients \a a[0], e.g.: (a[0],a[0],a[0],a[0]) */ -template EIGEN_DEVICE_FUNC inline Packet -pload1(const typename unpacket_traits::type *a) { return pset1(*a); } - -/** \internal \returns a packet with elements of \a *from duplicated. - * For instance, for a packet of 8 elements, 4 scalars will be read from \a *from and - * duplicated to form: {from[0],from[0],from[1],from[1],from[2],from[2],from[3],from[3]} - * Currently, this function is only used for scalar * complex products. - */ -template EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Packet -ploaddup(const typename unpacket_traits::type* from) { return *from; } - -/** \internal \returns a packet with elements of \a *from quadrupled. - * For instance, for a packet of 8 elements, 2 scalars will be read from \a *from and - * replicated to form: {from[0],from[0],from[0],from[0],from[1],from[1],from[1],from[1]} - * Currently, this function is only used in matrix products. - * For packet-size smaller or equal to 4, this function is equivalent to pload1 - */ -template EIGEN_DEVICE_FUNC inline Packet -ploadquad(const typename unpacket_traits::type* from) -{ return pload1(from); } - -/** \internal equivalent to - * \code - * a0 = pload1(a+0); - * a1 = pload1(a+1); - * a2 = pload1(a+2); - * a3 = pload1(a+3); - * \endcode - * \sa pset1, pload1, ploaddup, pbroadcast2 - */ -template EIGEN_DEVICE_FUNC -inline void pbroadcast4(const typename unpacket_traits::type *a, - Packet& a0, Packet& a1, Packet& a2, Packet& a3) -{ - a0 = pload1(a+0); - a1 = pload1(a+1); - a2 = pload1(a+2); - a3 = pload1(a+3); -} - -/** \internal equivalent to - * \code - * a0 = pload1(a+0); - * a1 = pload1(a+1); - * \endcode - * \sa pset1, pload1, ploaddup, pbroadcast4 - */ -template EIGEN_DEVICE_FUNC -inline void pbroadcast2(const typename unpacket_traits::type *a, - Packet& a0, Packet& a1) -{ - a0 = pload1(a+0); - a1 = pload1(a+1); -} - -/** \internal \brief Returns a packet with coefficients (a,a+1,...,a+packet_size-1). */ -template EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Packet -plset(const typename unpacket_traits::type& a) { return a; } - -/** \internal copy the packet \a from to \a *to, \a to must be 16 bytes aligned */ -template EIGEN_DEVICE_FUNC inline void pstore(Scalar* to, const Packet& from) -{ (*to) = from; } - -/** \internal copy the packet \a from to \a *to, (un-aligned store) */ -template EIGEN_DEVICE_FUNC inline void pstoreu(Scalar* to, const Packet& from) -{ (*to) = from; } - - template EIGEN_DEVICE_FUNC inline Packet pgather(const Scalar* from, Index /*stride*/) - { return ploadu(from); } - - template EIGEN_DEVICE_FUNC inline void pscatter(Scalar* to, const Packet& from, Index /*stride*/) - { pstore(to, from); } - -/** \internal tries to do cache prefetching of \a addr */ -template EIGEN_DEVICE_FUNC inline void prefetch(const Scalar* addr) -{ + + + /** \internal \returns static_cast(a) (coeff-wise) */ + template EIGEN_DEVICE_FUNC inline TgtPacket pcast(const SrcPacket &a) + { + return static_cast(a); + } + template + EIGEN_DEVICE_FUNC inline TgtPacket pcast(const SrcPacket &a, const SrcPacket & /*b*/) + { + return static_cast(a); + } + + template + EIGEN_DEVICE_FUNC inline TgtPacket + pcast(const SrcPacket &a, const SrcPacket & /*b*/, const SrcPacket & /*c*/, const SrcPacket & /*d*/) + { + return static_cast(a); + } + + /** \internal \returns a + b (coeff-wise) */ + template EIGEN_DEVICE_FUNC inline Packet padd(const Packet &a, const Packet &b) { return a + b; } + + /** \internal \returns a - b (coeff-wise) */ + template EIGEN_DEVICE_FUNC inline Packet psub(const Packet &a, const Packet &b) { return a - b; } + + /** \internal \returns -a (coeff-wise) */ + template EIGEN_DEVICE_FUNC inline Packet pnegate(const Packet &a) { return -a; } + + /** \internal \returns conj(a) (coeff-wise) */ + + template EIGEN_DEVICE_FUNC inline Packet pconj(const Packet &a) { return numext::conj(a); } + + /** \internal \returns a * b (coeff-wise) */ + template EIGEN_DEVICE_FUNC inline Packet pmul(const Packet &a, const Packet &b) { return a * b; } + + /** \internal \returns a / b (coeff-wise) */ + template EIGEN_DEVICE_FUNC inline Packet pdiv(const Packet &a, const Packet &b) { return a / b; } + + /** \internal \returns the min of \a a and \a b (coeff-wise) */ + template EIGEN_DEVICE_FUNC inline Packet pmin(const Packet &a, const Packet &b) + { + return numext::mini(a, b); + } + + /** \internal \returns the max of \a a and \a b (coeff-wise) */ + template EIGEN_DEVICE_FUNC inline Packet pmax(const Packet &a, const Packet &b) + { + return numext::maxi(a, b); + } + + /** \internal \returns the absolute value of \a a */ + template EIGEN_DEVICE_FUNC inline Packet pabs(const Packet &a) + { + using std::abs; + return abs(a); + } + + /** \internal \returns the phase angle of \a a */ + template EIGEN_DEVICE_FUNC inline Packet parg(const Packet &a) + { + using numext::arg; + return arg(a); + } + + /** \internal \returns the bitwise and of \a a and \a b */ + template EIGEN_DEVICE_FUNC inline Packet pand(const Packet &a, const Packet &b) { return a & b; } + + /** \internal \returns the bitwise or of \a a and \a b */ + template EIGEN_DEVICE_FUNC inline Packet por(const Packet &a, const Packet &b) { return a | b; } + + /** \internal \returns the bitwise xor of \a a and \a b */ + template EIGEN_DEVICE_FUNC inline Packet pxor(const Packet &a, const Packet &b) { return a ^ b; } + + /** \internal \returns the bitwise andnot of \a a and \a b */ + template EIGEN_DEVICE_FUNC inline Packet pandnot(const Packet &a, const Packet &b) + { + return a & (!b); + } + + /** \internal \returns a packet version of \a *from, from must be 16 bytes aligned */ + template EIGEN_DEVICE_FUNC inline Packet pload(const typename unpacket_traits::type *from) + { + return *from; + } + + /** \internal \returns a packet version of \a *from, (un-aligned load) */ + template EIGEN_DEVICE_FUNC inline Packet ploadu(const typename unpacket_traits::type *from) + { + return *from; + } + + /** \internal \returns a packet with constant coefficients \a a, e.g.: (a,a,a,a) */ + template EIGEN_DEVICE_FUNC inline Packet pset1(const typename unpacket_traits::type &a) + { + return a; + } + + /** \internal \returns a packet with constant coefficients \a a[0], e.g.: (a[0],a[0],a[0],a[0]) */ + template EIGEN_DEVICE_FUNC inline Packet pload1(const typename unpacket_traits::type *a) + { + return pset1(*a); + } + + /** \internal \returns a packet with elements of \a *from duplicated. + * For instance, for a packet of 8 elements, 4 scalars will be read from \a *from and + * duplicated to form: {from[0],from[0],from[1],from[1],from[2],from[2],from[3],from[3]} + * Currently, this function is only used for scalar * complex products. + */ + template + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Packet ploaddup(const typename unpacket_traits::type *from) + { + return *from; + } + + /** \internal \returns a packet with elements of \a *from quadrupled. + * For instance, for a packet of 8 elements, 2 scalars will be read from \a *from and + * replicated to form: {from[0],from[0],from[0],from[0],from[1],from[1],from[1],from[1]} + * Currently, this function is only used in matrix products. + * For packet-size smaller or equal to 4, this function is equivalent to pload1 + */ + template + EIGEN_DEVICE_FUNC inline Packet ploadquad(const typename unpacket_traits::type *from) + { + return pload1(from); + } + + /** \internal equivalent to + * \code + * a0 = pload1(a+0); + * a1 = pload1(a+1); + * a2 = pload1(a+2); + * a3 = pload1(a+3); + * \endcode + * \sa pset1, pload1, ploaddup, pbroadcast2 + */ + template + EIGEN_DEVICE_FUNC inline void + pbroadcast4(const typename unpacket_traits::type *a, Packet &a0, Packet &a1, Packet &a2, Packet &a3) + { + a0 = pload1(a + 0); + a1 = pload1(a + 1); + a2 = pload1(a + 2); + a3 = pload1(a + 3); + } + + /** \internal equivalent to + * \code + * a0 = pload1(a+0); + * a1 = pload1(a+1); + * \endcode + * \sa pset1, pload1, ploaddup, pbroadcast4 + */ + template + EIGEN_DEVICE_FUNC inline void pbroadcast2(const typename unpacket_traits::type *a, Packet &a0, Packet &a1) + { + a0 = pload1(a + 0); + a1 = pload1(a + 1); + } + + /** \internal \brief Returns a packet with coefficients (a,a+1,...,a+packet_size-1). */ + template + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Packet plset(const typename unpacket_traits::type &a) + { + return a; + } + + /** \internal copy the packet \a from to \a *to, \a to must be 16 bytes aligned */ + template EIGEN_DEVICE_FUNC inline void pstore(Scalar *to, const Packet &from) + { + (*to) = from; + } + + /** \internal copy the packet \a from to \a *to, (un-aligned store) */ + template EIGEN_DEVICE_FUNC inline void pstoreu(Scalar *to, const Packet &from) + { + (*to) = from; + } + + template + EIGEN_DEVICE_FUNC inline Packet pgather(const Scalar *from, Index /*stride*/) + { + return ploadu(from); + } + + template + EIGEN_DEVICE_FUNC inline void pscatter(Scalar *to, const Packet &from, Index /*stride*/) + { + pstore(to, from); + } + + /** \internal tries to do cache prefetching of \a addr */ + template EIGEN_DEVICE_FUNC inline void prefetch(const Scalar *addr) + { #ifdef __CUDA_ARCH__ #if defined(__LP64__) - // 64-bit pointer operand constraint for inlined asm - asm(" prefetch.L1 [ %1 ];" : "=l"(addr) : "l"(addr)); + // 64-bit pointer operand constraint for inlined asm + asm(" prefetch.L1 [ %1 ];" : "=l"(addr) : "l"(addr)); #else - // 32-bit pointer operand constraint for inlined asm - asm(" prefetch.L1 [ %1 ];" : "=r"(addr) : "r"(addr)); + // 32-bit pointer operand constraint for inlined asm + asm(" prefetch.L1 [ %1 ];" : "=r"(addr) : "r"(addr)); #endif #elif (!EIGEN_COMP_MSVC) && (EIGEN_COMP_GNUC || EIGEN_COMP_CLANG || EIGEN_COMP_ICC) - __builtin_prefetch(addr); + __builtin_prefetch(addr); #endif -} - -/** \internal \returns the first element of a packet */ -template EIGEN_DEVICE_FUNC inline typename unpacket_traits::type pfirst(const Packet& a) -{ return a; } - -/** \internal \returns a packet where the element i contains the sum of the packet of \a vec[i] */ -template EIGEN_DEVICE_FUNC inline Packet -preduxp(const Packet* vecs) { return vecs[0]; } - -/** \internal \returns the sum of the elements of \a a*/ -template EIGEN_DEVICE_FUNC inline typename unpacket_traits::type predux(const Packet& a) -{ return a; } - -/** \internal \returns the sum of the elements of \a a by block of 4 elements. - * For a packet {a0, a1, a2, a3, a4, a5, a6, a7}, it returns a half packet {a0+a4, a1+a5, a2+a6, a3+a7} - * For packet-size smaller or equal to 4, this boils down to a noop. - */ -template EIGEN_DEVICE_FUNC inline -typename conditional<(unpacket_traits::size%8)==0,typename unpacket_traits::half,Packet>::type -predux_downto4(const Packet& a) -{ return a; } - -/** \internal \returns the product of the elements of \a a*/ -template EIGEN_DEVICE_FUNC inline typename unpacket_traits::type predux_mul(const Packet& a) -{ return a; } - -/** \internal \returns the min of the elements of \a a*/ -template EIGEN_DEVICE_FUNC inline typename unpacket_traits::type predux_min(const Packet& a) -{ return a; } - -/** \internal \returns the max of the elements of \a a*/ -template EIGEN_DEVICE_FUNC inline typename unpacket_traits::type predux_max(const Packet& a) -{ return a; } - -/** \internal \returns the reversed elements of \a a*/ -template EIGEN_DEVICE_FUNC inline Packet preverse(const Packet& a) -{ return a; } - -/** \internal \returns \a a with real and imaginary part flipped (for complex type only) */ -template EIGEN_DEVICE_FUNC inline Packet pcplxflip(const Packet& a) -{ - // FIXME: uncomment the following in case we drop the internal imag and real functions. -// using std::imag; -// using std::real; - return Packet(imag(a),real(a)); -} - -/************************** -* Special math functions -***************************/ - -/** \internal \returns the sine of \a a (coeff-wise) */ -template EIGEN_DECLARE_FUNCTION_ALLOWING_MULTIPLE_DEFINITIONS -Packet psin(const Packet& a) { using std::sin; return sin(a); } - -/** \internal \returns the cosine of \a a (coeff-wise) */ -template EIGEN_DECLARE_FUNCTION_ALLOWING_MULTIPLE_DEFINITIONS -Packet pcos(const Packet& a) { using std::cos; return cos(a); } - -/** \internal \returns the tan of \a a (coeff-wise) */ -template EIGEN_DECLARE_FUNCTION_ALLOWING_MULTIPLE_DEFINITIONS -Packet ptan(const Packet& a) { using std::tan; return tan(a); } - -/** \internal \returns the arc sine of \a a (coeff-wise) */ -template EIGEN_DECLARE_FUNCTION_ALLOWING_MULTIPLE_DEFINITIONS -Packet pasin(const Packet& a) { using std::asin; return asin(a); } - -/** \internal \returns the arc cosine of \a a (coeff-wise) */ -template EIGEN_DECLARE_FUNCTION_ALLOWING_MULTIPLE_DEFINITIONS -Packet pacos(const Packet& a) { using std::acos; return acos(a); } - -/** \internal \returns the arc tangent of \a a (coeff-wise) */ -template EIGEN_DECLARE_FUNCTION_ALLOWING_MULTIPLE_DEFINITIONS -Packet patan(const Packet& a) { using std::atan; return atan(a); } - -/** \internal \returns the hyperbolic sine of \a a (coeff-wise) */ -template EIGEN_DECLARE_FUNCTION_ALLOWING_MULTIPLE_DEFINITIONS -Packet psinh(const Packet& a) { using std::sinh; return sinh(a); } - -/** \internal \returns the hyperbolic cosine of \a a (coeff-wise) */ -template EIGEN_DECLARE_FUNCTION_ALLOWING_MULTIPLE_DEFINITIONS -Packet pcosh(const Packet& a) { using std::cosh; return cosh(a); } - -/** \internal \returns the hyperbolic tan of \a a (coeff-wise) */ -template EIGEN_DECLARE_FUNCTION_ALLOWING_MULTIPLE_DEFINITIONS -Packet ptanh(const Packet& a) { using std::tanh; return tanh(a); } - -/** \internal \returns the exp of \a a (coeff-wise) */ -template EIGEN_DECLARE_FUNCTION_ALLOWING_MULTIPLE_DEFINITIONS -Packet pexp(const Packet& a) { using std::exp; return exp(a); } - -/** \internal \returns the log of \a a (coeff-wise) */ -template EIGEN_DECLARE_FUNCTION_ALLOWING_MULTIPLE_DEFINITIONS -Packet plog(const Packet& a) { using std::log; return log(a); } - -/** \internal \returns the log1p of \a a (coeff-wise) */ -template EIGEN_DECLARE_FUNCTION_ALLOWING_MULTIPLE_DEFINITIONS -Packet plog1p(const Packet& a) { return numext::log1p(a); } - -/** \internal \returns the log10 of \a a (coeff-wise) */ -template EIGEN_DECLARE_FUNCTION_ALLOWING_MULTIPLE_DEFINITIONS -Packet plog10(const Packet& a) { using std::log10; return log10(a); } - -/** \internal \returns the square-root of \a a (coeff-wise) */ -template EIGEN_DECLARE_FUNCTION_ALLOWING_MULTIPLE_DEFINITIONS -Packet psqrt(const Packet& a) { using std::sqrt; return sqrt(a); } - -/** \internal \returns the reciprocal square-root of \a a (coeff-wise) */ -template EIGEN_DECLARE_FUNCTION_ALLOWING_MULTIPLE_DEFINITIONS -Packet prsqrt(const Packet& a) { - return pdiv(pset1(1), psqrt(a)); -} - -/** \internal \returns the rounded value of \a a (coeff-wise) */ -template EIGEN_DECLARE_FUNCTION_ALLOWING_MULTIPLE_DEFINITIONS -Packet pround(const Packet& a) { using numext::round; return round(a); } - -/** \internal \returns the floor of \a a (coeff-wise) */ -template EIGEN_DECLARE_FUNCTION_ALLOWING_MULTIPLE_DEFINITIONS -Packet pfloor(const Packet& a) { using numext::floor; return floor(a); } - -/** \internal \returns the ceil of \a a (coeff-wise) */ -template EIGEN_DECLARE_FUNCTION_ALLOWING_MULTIPLE_DEFINITIONS -Packet pceil(const Packet& a) { using numext::ceil; return ceil(a); } + } + + /** \internal \returns the first element of a packet */ + template EIGEN_DEVICE_FUNC inline typename unpacket_traits::type pfirst(const Packet &a) + { + return a; + } + + /** \internal \returns a packet where the element i contains the sum of the packet of \a vec[i] */ + template EIGEN_DEVICE_FUNC inline Packet preduxp(const Packet *vecs) { return vecs[0]; } + + /** \internal \returns the sum of the elements of \a a*/ + template EIGEN_DEVICE_FUNC inline typename unpacket_traits::type predux(const Packet &a) + { + return a; + } + + /** \internal \returns the sum of the elements of \a a by block of 4 elements. + * For a packet {a0, a1, a2, a3, a4, a5, a6, a7}, it returns a half packet {a0+a4, a1+a5, a2+a6, a3+a7} + * For packet-size smaller or equal to 4, this boils down to a noop. + */ + template + EIGEN_DEVICE_FUNC inline + typename conditional<(unpacket_traits::size % 8) == 0, typename unpacket_traits::half, Packet>::type + predux_downto4(const Packet &a) + { + return a; + } + + /** \internal \returns the product of the elements of \a a*/ + template EIGEN_DEVICE_FUNC inline typename unpacket_traits::type predux_mul(const Packet &a) + { + return a; + } + + /** \internal \returns the min of the elements of \a a*/ + template EIGEN_DEVICE_FUNC inline typename unpacket_traits::type predux_min(const Packet &a) + { + return a; + } + + /** \internal \returns the max of the elements of \a a*/ + template EIGEN_DEVICE_FUNC inline typename unpacket_traits::type predux_max(const Packet &a) + { + return a; + } + + /** \internal \returns the reversed elements of \a a*/ + template EIGEN_DEVICE_FUNC inline Packet preverse(const Packet &a) { return a; } + + /** \internal \returns \a a with real and imaginary part flipped (for complex type only) */ + template EIGEN_DEVICE_FUNC inline Packet pcplxflip(const Packet &a) + { + // FIXME: uncomment the following in case we drop the internal imag and real functions. + // using std::imag; + // using std::real; + return Packet(imag(a), real(a)); + } + + /************************** + * Special math functions + ***************************/ + + /** \internal \returns the sine of \a a (coeff-wise) */ + template EIGEN_DECLARE_FUNCTION_ALLOWING_MULTIPLE_DEFINITIONS Packet psin(const Packet &a) + { + using std::sin; + return sin(a); + } + + /** \internal \returns the cosine of \a a (coeff-wise) */ + template EIGEN_DECLARE_FUNCTION_ALLOWING_MULTIPLE_DEFINITIONS Packet pcos(const Packet &a) + { + using std::cos; + return cos(a); + } + + /** \internal \returns the tan of \a a (coeff-wise) */ + template EIGEN_DECLARE_FUNCTION_ALLOWING_MULTIPLE_DEFINITIONS Packet ptan(const Packet &a) + { + using std::tan; + return tan(a); + } + + /** \internal \returns the arc sine of \a a (coeff-wise) */ + template EIGEN_DECLARE_FUNCTION_ALLOWING_MULTIPLE_DEFINITIONS Packet pasin(const Packet &a) + { + using std::asin; + return asin(a); + } + + /** \internal \returns the arc cosine of \a a (coeff-wise) */ + template EIGEN_DECLARE_FUNCTION_ALLOWING_MULTIPLE_DEFINITIONS Packet pacos(const Packet &a) + { + using std::acos; + return acos(a); + } + + /** \internal \returns the arc tangent of \a a (coeff-wise) */ + template EIGEN_DECLARE_FUNCTION_ALLOWING_MULTIPLE_DEFINITIONS Packet patan(const Packet &a) + { + using std::atan; + return atan(a); + } + + /** \internal \returns the hyperbolic sine of \a a (coeff-wise) */ + template EIGEN_DECLARE_FUNCTION_ALLOWING_MULTIPLE_DEFINITIONS Packet psinh(const Packet &a) + { + using std::sinh; + return sinh(a); + } + + /** \internal \returns the hyperbolic cosine of \a a (coeff-wise) */ + template EIGEN_DECLARE_FUNCTION_ALLOWING_MULTIPLE_DEFINITIONS Packet pcosh(const Packet &a) + { + using std::cosh; + return cosh(a); + } + + /** \internal \returns the hyperbolic tan of \a a (coeff-wise) */ + template EIGEN_DECLARE_FUNCTION_ALLOWING_MULTIPLE_DEFINITIONS Packet ptanh(const Packet &a) + { + using std::tanh; + return tanh(a); + } + + /** \internal \returns the exp of \a a (coeff-wise) */ + template EIGEN_DECLARE_FUNCTION_ALLOWING_MULTIPLE_DEFINITIONS Packet pexp(const Packet &a) + { + using std::exp; + return exp(a); + } + + /** \internal \returns the log of \a a (coeff-wise) */ + template EIGEN_DECLARE_FUNCTION_ALLOWING_MULTIPLE_DEFINITIONS Packet plog(const Packet &a) + { + using std::log; + return log(a); + } + + /** \internal \returns the log1p of \a a (coeff-wise) */ + template EIGEN_DECLARE_FUNCTION_ALLOWING_MULTIPLE_DEFINITIONS Packet plog1p(const Packet &a) + { + return numext::log1p(a); + } + + /** \internal \returns the log10 of \a a (coeff-wise) */ + template EIGEN_DECLARE_FUNCTION_ALLOWING_MULTIPLE_DEFINITIONS Packet plog10(const Packet &a) + { + using std::log10; + return log10(a); + } + + /** \internal \returns the square-root of \a a (coeff-wise) */ + template EIGEN_DECLARE_FUNCTION_ALLOWING_MULTIPLE_DEFINITIONS Packet psqrt(const Packet &a) + { + using std::sqrt; + return sqrt(a); + } + + /** \internal \returns the reciprocal square-root of \a a (coeff-wise) */ + template EIGEN_DECLARE_FUNCTION_ALLOWING_MULTIPLE_DEFINITIONS Packet prsqrt(const Packet &a) + { + return pdiv(pset1(1), psqrt(a)); + } + + /** \internal \returns the rounded value of \a a (coeff-wise) */ + template EIGEN_DECLARE_FUNCTION_ALLOWING_MULTIPLE_DEFINITIONS Packet pround(const Packet &a) + { + using numext::round; + return round(a); + } + + /** \internal \returns the floor of \a a (coeff-wise) */ + template EIGEN_DECLARE_FUNCTION_ALLOWING_MULTIPLE_DEFINITIONS Packet pfloor(const Packet &a) + { + using numext::floor; + return floor(a); + } + + /** \internal \returns the ceil of \a a (coeff-wise) */ + template EIGEN_DECLARE_FUNCTION_ALLOWING_MULTIPLE_DEFINITIONS Packet pceil(const Packet &a) + { + using numext::ceil; + return ceil(a); + } + + /*************************************************************************** + * The following functions might not have to be overwritten for vectorized types + ***************************************************************************/ + + /** \internal copy a packet with constant coeficient \a a (e.g., [a,a,a,a]) to \a *to. \a to must be 16 bytes aligned + */ + // NOTE: this function must really be templated on the packet type (think about different packet types for the same + // scalar type) + template + inline void pstore1(typename unpacket_traits::type *to, const typename unpacket_traits::type &a) + { + pstore(to, pset1(a)); + } + + /** \internal \returns a * b + c (coeff-wise) */ + template EIGEN_DEVICE_FUNC inline Packet pmadd(const Packet &a, const Packet &b, const Packet &c) + { + return padd(pmul(a, b), c); + } + + /** \internal \returns a packet version of \a *from. + * The pointer \a from must be aligned on a \a Alignment bytes boundary. */ + template + EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE Packet ploadt(const typename unpacket_traits::type *from) + { + if (Alignment >= unpacket_traits::alignment) + return pload(from); + else + return ploadu(from); + } + + /** \internal copy the packet \a from to \a *to. + * The pointer \a from must be aligned on a \a Alignment bytes boundary. */ + template + EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE void pstoret(Scalar *to, const Packet &from) + { + if (Alignment >= unpacket_traits::alignment) + pstore(to, from); + else + pstoreu(to, from); + } + + /** \internal \returns a packet version of \a *from. + * Unlike ploadt, ploadt_ro takes advantage of the read-only memory path on the + * hardware if available to speedup the loading of data that won't be modified + * by the current computation. + */ + template + EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE Packet ploadt_ro(const typename unpacket_traits::type *from) + { + return ploadt(from); + } + + /** \internal default implementation of palign() allowing partial specialization */ + template struct palign_impl + { + // by default data are aligned, so there is nothing to be done :) + static inline void run(PacketType &, const PacketType &) {} + }; -/*************************************************************************** -* The following functions might not have to be overwritten for vectorized types -***************************************************************************/ - -/** \internal copy a packet with constant coeficient \a a (e.g., [a,a,a,a]) to \a *to. \a to must be 16 bytes aligned */ -// NOTE: this function must really be templated on the packet type (think about different packet types for the same scalar type) -template -inline void pstore1(typename unpacket_traits::type* to, const typename unpacket_traits::type& a) -{ - pstore(to, pset1(a)); -} - -/** \internal \returns a * b + c (coeff-wise) */ -template EIGEN_DEVICE_FUNC inline Packet -pmadd(const Packet& a, - const Packet& b, - const Packet& c) -{ return padd(pmul(a, b),c); } - -/** \internal \returns a packet version of \a *from. - * The pointer \a from must be aligned on a \a Alignment bytes boundary. */ -template -EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE Packet ploadt(const typename unpacket_traits::type* from) -{ - if(Alignment >= unpacket_traits::alignment) - return pload(from); - else - return ploadu(from); -} - -/** \internal copy the packet \a from to \a *to. - * The pointer \a from must be aligned on a \a Alignment bytes boundary. */ -template -EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE void pstoret(Scalar* to, const Packet& from) -{ - if(Alignment >= unpacket_traits::alignment) - pstore(to, from); - else - pstoreu(to, from); -} - -/** \internal \returns a packet version of \a *from. - * Unlike ploadt, ploadt_ro takes advantage of the read-only memory path on the - * hardware if available to speedup the loading of data that won't be modified - * by the current computation. - */ -template -EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE Packet ploadt_ro(const typename unpacket_traits::type* from) -{ - return ploadt(from); -} - -/** \internal default implementation of palign() allowing partial specialization */ -template -struct palign_impl -{ - // by default data are aligned, so there is nothing to be done :) - static inline void run(PacketType&, const PacketType&) {} -}; - -/** \internal update \a first using the concatenation of the packet_size minus \a Offset last elements - * of \a first and \a Offset first elements of \a second. - * - * This function is currently only used to optimize matrix-vector products on unligned matrices. - * It takes 2 packets that represent a contiguous memory array, and returns a packet starting - * at the position \a Offset. For instance, for packets of 4 elements, we have: - * Input: - * - first = {f0,f1,f2,f3} - * - second = {s0,s1,s2,s3} - * Output: - * - if Offset==0 then {f0,f1,f2,f3} - * - if Offset==1 then {f1,f2,f3,s0} - * - if Offset==2 then {f2,f3,s0,s1} - * - if Offset==3 then {f3,s0,s1,s3} - */ -template -inline void palign(PacketType& first, const PacketType& second) -{ - palign_impl::run(first,second); -} + /** \internal update \a first using the concatenation of the packet_size minus \a Offset last elements + * of \a first and \a Offset first elements of \a second. + * + * This function is currently only used to optimize matrix-vector products on unligned matrices. + * It takes 2 packets that represent a contiguous memory array, and returns a packet starting + * at the position \a Offset. For instance, for packets of 4 elements, we have: + * Input: + * - first = {f0,f1,f2,f3} + * - second = {s0,s1,s2,s3} + * Output: + * - if Offset==0 then {f0,f1,f2,f3} + * - if Offset==1 then {f1,f2,f3,s0} + * - if Offset==2 then {f2,f3,s0,s1} + * - if Offset==3 then {f3,s0,s1,s3} + */ + template inline void palign(PacketType &first, const PacketType &second) + { + palign_impl::run(first, second); + } /*************************************************************************** -* Fast complex products (GCC generates a function call which is very slow) -***************************************************************************/ + * Fast complex products (GCC generates a function call which is very slow) + ***************************************************************************/ // Eigen+CUDA does not support complexes. #ifndef __CUDACC__ -template<> inline std::complex pmul(const std::complex& a, const std::complex& b) -{ return std::complex(real(a)*real(b) - imag(a)*imag(b), imag(a)*real(b) + real(a)*imag(b)); } + template<> inline std::complex pmul(const std::complex &a, const std::complex &b) + { + return std::complex(real(a) * real(b) - imag(a) * imag(b), imag(a) * real(b) + real(a) * imag(b)); + } -template<> inline std::complex pmul(const std::complex& a, const std::complex& b) -{ return std::complex(real(a)*real(b) - imag(a)*imag(b), imag(a)*real(b) + real(a)*imag(b)); } + template<> inline std::complex pmul(const std::complex &a, const std::complex &b) + { + return std::complex(real(a) * real(b) - imag(a) * imag(b), imag(a) * real(b) + real(a) * imag(b)); + } #endif -/*************************************************************************** - * PacketBlock, that is a collection of N packets where the number of words - * in the packet is a multiple of N. -***************************************************************************/ -template ::size> struct PacketBlock { - Packet packet[N]; -}; - -template EIGEN_DEVICE_FUNC inline void -ptranspose(PacketBlock& /*kernel*/) { - // Nothing to do in the scalar case, i.e. a 1x1 matrix. -} + /*************************************************************************** + * PacketBlock, that is a collection of N packets where the number of words + * in the packet is a multiple of N. + ***************************************************************************/ + template::size> struct PacketBlock + { + Packet packet[N]; + }; -/*************************************************************************** - * Selector, i.e. vector of N boolean values used to select (i.e. blend) - * words from 2 packets. -***************************************************************************/ -template struct Selector { - bool select[N]; -}; - -template EIGEN_DEVICE_FUNC inline Packet -pblend(const Selector::size>& ifPacket, const Packet& thenPacket, const Packet& elsePacket) { - return ifPacket.select[0] ? thenPacket : elsePacket; -} - -/** \internal \returns \a a with the first coefficient replaced by the scalar b */ -template EIGEN_DEVICE_FUNC inline Packet -pinsertfirst(const Packet& a, typename unpacket_traits::type b) -{ - // Default implementation based on pblend. - // It must be specialized for higher performance. - Selector::size> mask; - mask.select[0] = true; - // This for loop should be optimized away by the compiler. - for(Index i=1; i::size; ++i) - mask.select[i] = false; - return pblend(mask, pset1(b), a); -} - -/** \internal \returns \a a with the last coefficient replaced by the scalar b */ -template EIGEN_DEVICE_FUNC inline Packet -pinsertlast(const Packet& a, typename unpacket_traits::type b) -{ - // Default implementation based on pblend. - // It must be specialized for higher performance. - Selector::size> mask; - // This for loop should be optimized away by the compiler. - for(Index i=0; i::size-1; ++i) - mask.select[i] = false; - mask.select[unpacket_traits::size-1] = true; - return pblend(mask, pset1(b), a); -} - -} // end namespace internal - -} // end namespace Eigen - -#endif // EIGEN_GENERIC_PACKET_MATH_H + template EIGEN_DEVICE_FUNC inline void ptranspose(PacketBlock & /*kernel*/) + { + // Nothing to do in the scalar case, i.e. a 1x1 matrix. + } + + /*************************************************************************** + * Selector, i.e. vector of N boolean values used to select (i.e. blend) + * words from 2 packets. + ***************************************************************************/ + template struct Selector + { + bool select[N]; + }; + + template + EIGEN_DEVICE_FUNC inline Packet + pblend(const Selector::size> &ifPacket, const Packet &thenPacket, const Packet &elsePacket) + { + return ifPacket.select[0] ? thenPacket : elsePacket; + } + + /** \internal \returns \a a with the first coefficient replaced by the scalar b */ + template + EIGEN_DEVICE_FUNC inline Packet pinsertfirst(const Packet &a, typename unpacket_traits::type b) + { + // Default implementation based on pblend. + // It must be specialized for higher performance. + Selector::size> mask; + mask.select[0] = true; + // This for loop should be optimized away by the compiler. + for (Index i = 1; i < unpacket_traits::size; ++i) mask.select[i] = false; + return pblend(mask, pset1(b), a); + } + + /** \internal \returns \a a with the last coefficient replaced by the scalar b */ + template + EIGEN_DEVICE_FUNC inline Packet pinsertlast(const Packet &a, typename unpacket_traits::type b) + { + // Default implementation based on pblend. + // It must be specialized for higher performance. + Selector::size> mask; + // This for loop should be optimized away by the compiler. + for (Index i = 0; i < unpacket_traits::size - 1; ++i) mask.select[i] = false; + mask.select[unpacket_traits::size - 1] = true; + return pblend(mask, pset1(b), a); + } + +}// end namespace internal + +}// end namespace Eigen + +#endif// EIGEN_GENERIC_PACKET_MATH_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/GlobalFunctions.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/GlobalFunctions.h index 769dc255..0f5beb20 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/GlobalFunctions.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/GlobalFunctions.h @@ -13,175 +13,185 @@ #ifdef EIGEN_PARSED_BY_DOXYGEN -#define EIGEN_ARRAY_DECLARE_GLOBAL_UNARY(NAME,FUNCTOR,DOC_OP,DOC_DETAILS) \ - /** \returns an expression of the coefficient-wise DOC_OP of \a x - - DOC_DETAILS - - \sa Math functions, class CwiseUnaryOp - */ \ - template \ - inline const Eigen::CwiseUnaryOp, const Derived> \ - NAME(const Eigen::ArrayBase& x); +#define EIGEN_ARRAY_DECLARE_GLOBAL_UNARY(NAME, FUNCTOR, DOC_OP, DOC_DETAILS) \ + /** \returns an expression of the coefficient-wise DOC_OP of \a x \ + \ + DOC_DETAILS \ + \ + \sa Math functions, class CwiseUnaryOp \ + */ \ + template \ + inline const Eigen::CwiseUnaryOp, const Derived> NAME( \ + const Eigen::ArrayBase &x); #else -#define EIGEN_ARRAY_DECLARE_GLOBAL_UNARY(NAME,FUNCTOR,DOC_OP,DOC_DETAILS) \ - template \ - inline const Eigen::CwiseUnaryOp, const Derived> \ - (NAME)(const Eigen::ArrayBase& x) { \ +#define EIGEN_ARRAY_DECLARE_GLOBAL_UNARY(NAME, FUNCTOR, DOC_OP, DOC_DETAILS) \ + template \ + inline const Eigen::CwiseUnaryOp, const Derived>(NAME)( \ + const Eigen::ArrayBase &x) \ + { \ return Eigen::CwiseUnaryOp, const Derived>(x.derived()); \ } -#endif // EIGEN_PARSED_BY_DOXYGEN +#endif// EIGEN_PARSED_BY_DOXYGEN -#define EIGEN_ARRAY_DECLARE_GLOBAL_EIGEN_UNARY(NAME,FUNCTOR) \ - \ - template \ - struct NAME##_retval > \ - { \ +#define EIGEN_ARRAY_DECLARE_GLOBAL_EIGEN_UNARY(NAME, FUNCTOR) \ + \ + template struct NAME##_retval> \ + { \ typedef const Eigen::CwiseUnaryOp, const Derived> type; \ - }; \ - template \ - struct NAME##_impl > \ - { \ - static inline typename NAME##_retval >::type run(const Eigen::ArrayBase& x) \ - { \ - return typename NAME##_retval >::type(x.derived()); \ - } \ + }; \ + template struct NAME##_impl> \ + { \ + static inline typename NAME##_retval>::type run(const Eigen::ArrayBase &x) \ + { \ + return typename NAME##_retval>::type(x.derived()); \ + } \ }; -namespace Eigen -{ - EIGEN_ARRAY_DECLARE_GLOBAL_UNARY(real,scalar_real_op,real part,\sa ArrayBase::real) - EIGEN_ARRAY_DECLARE_GLOBAL_UNARY(imag,scalar_imag_op,imaginary part,\sa ArrayBase::imag) - EIGEN_ARRAY_DECLARE_GLOBAL_UNARY(conj,scalar_conjugate_op,complex conjugate,\sa ArrayBase::conjugate) - EIGEN_ARRAY_DECLARE_GLOBAL_UNARY(inverse,scalar_inverse_op,inverse,\sa ArrayBase::inverse) - EIGEN_ARRAY_DECLARE_GLOBAL_UNARY(sin,scalar_sin_op,sine,\sa ArrayBase::sin) - EIGEN_ARRAY_DECLARE_GLOBAL_UNARY(cos,scalar_cos_op,cosine,\sa ArrayBase::cos) - EIGEN_ARRAY_DECLARE_GLOBAL_UNARY(tan,scalar_tan_op,tangent,\sa ArrayBase::tan) - EIGEN_ARRAY_DECLARE_GLOBAL_UNARY(atan,scalar_atan_op,arc-tangent,\sa ArrayBase::atan) - EIGEN_ARRAY_DECLARE_GLOBAL_UNARY(asin,scalar_asin_op,arc-sine,\sa ArrayBase::asin) - EIGEN_ARRAY_DECLARE_GLOBAL_UNARY(acos,scalar_acos_op,arc-consine,\sa ArrayBase::acos) - EIGEN_ARRAY_DECLARE_GLOBAL_UNARY(sinh,scalar_sinh_op,hyperbolic sine,\sa ArrayBase::sinh) - EIGEN_ARRAY_DECLARE_GLOBAL_UNARY(cosh,scalar_cosh_op,hyperbolic cosine,\sa ArrayBase::cosh) - EIGEN_ARRAY_DECLARE_GLOBAL_UNARY(tanh,scalar_tanh_op,hyperbolic tangent,\sa ArrayBase::tanh) - EIGEN_ARRAY_DECLARE_GLOBAL_UNARY(lgamma,scalar_lgamma_op,natural logarithm of the gamma function,\sa ArrayBase::lgamma) - EIGEN_ARRAY_DECLARE_GLOBAL_UNARY(digamma,scalar_digamma_op,derivative of lgamma,\sa ArrayBase::digamma) - EIGEN_ARRAY_DECLARE_GLOBAL_UNARY(erf,scalar_erf_op,error function,\sa ArrayBase::erf) - EIGEN_ARRAY_DECLARE_GLOBAL_UNARY(erfc,scalar_erfc_op,complement error function,\sa ArrayBase::erfc) - EIGEN_ARRAY_DECLARE_GLOBAL_UNARY(exp,scalar_exp_op,exponential,\sa ArrayBase::exp) - EIGEN_ARRAY_DECLARE_GLOBAL_UNARY(log,scalar_log_op,natural logarithm,\sa Eigen::log10 DOXCOMMA ArrayBase::log) - EIGEN_ARRAY_DECLARE_GLOBAL_UNARY(log1p,scalar_log1p_op,natural logarithm of 1 plus the value,\sa ArrayBase::log1p) - EIGEN_ARRAY_DECLARE_GLOBAL_UNARY(log10,scalar_log10_op,base 10 logarithm,\sa Eigen::log DOXCOMMA ArrayBase::log) - EIGEN_ARRAY_DECLARE_GLOBAL_UNARY(abs,scalar_abs_op,absolute value,\sa ArrayBase::abs DOXCOMMA MatrixBase::cwiseAbs) - EIGEN_ARRAY_DECLARE_GLOBAL_UNARY(abs2,scalar_abs2_op,squared absolute value,\sa ArrayBase::abs2 DOXCOMMA MatrixBase::cwiseAbs2) - EIGEN_ARRAY_DECLARE_GLOBAL_UNARY(arg,scalar_arg_op,complex argument,\sa ArrayBase::arg) - EIGEN_ARRAY_DECLARE_GLOBAL_UNARY(sqrt,scalar_sqrt_op,square root,\sa ArrayBase::sqrt DOXCOMMA MatrixBase::cwiseSqrt) - EIGEN_ARRAY_DECLARE_GLOBAL_UNARY(rsqrt,scalar_rsqrt_op,reciprocal square root,\sa ArrayBase::rsqrt) - EIGEN_ARRAY_DECLARE_GLOBAL_UNARY(square,scalar_square_op,square (power 2),\sa Eigen::abs2 DOXCOMMA Eigen::pow DOXCOMMA ArrayBase::square) - EIGEN_ARRAY_DECLARE_GLOBAL_UNARY(cube,scalar_cube_op,cube (power 3),\sa Eigen::pow DOXCOMMA ArrayBase::cube) - EIGEN_ARRAY_DECLARE_GLOBAL_UNARY(round,scalar_round_op,nearest integer,\sa Eigen::floor DOXCOMMA Eigen::ceil DOXCOMMA ArrayBase::round) - EIGEN_ARRAY_DECLARE_GLOBAL_UNARY(floor,scalar_floor_op,nearest integer not greater than the giben value,\sa Eigen::ceil DOXCOMMA ArrayBase::floor) - EIGEN_ARRAY_DECLARE_GLOBAL_UNARY(ceil,scalar_ceil_op,nearest integer not less than the giben value,\sa Eigen::floor DOXCOMMA ArrayBase::ceil) - EIGEN_ARRAY_DECLARE_GLOBAL_UNARY(isnan,scalar_isnan_op,not-a-number test,\sa Eigen::isinf DOXCOMMA Eigen::isfinite DOXCOMMA ArrayBase::isnan) - EIGEN_ARRAY_DECLARE_GLOBAL_UNARY(isinf,scalar_isinf_op,infinite value test,\sa Eigen::isnan DOXCOMMA Eigen::isfinite DOXCOMMA ArrayBase::isinf) - EIGEN_ARRAY_DECLARE_GLOBAL_UNARY(isfinite,scalar_isfinite_op,finite value test,\sa Eigen::isinf DOXCOMMA Eigen::isnan DOXCOMMA ArrayBase::isfinite) - EIGEN_ARRAY_DECLARE_GLOBAL_UNARY(sign,scalar_sign_op,sign (or 0),\sa ArrayBase::sign) - - /** \returns an expression of the coefficient-wise power of \a x to the given constant \a exponent. - * - * \tparam ScalarExponent is the scalar type of \a exponent. It must be compatible with the scalar type of the given expression (\c Derived::Scalar). - * - * \sa ArrayBase::pow() - * - * \relates ArrayBase - */ +namespace Eigen { +EIGEN_ARRAY_DECLARE_GLOBAL_UNARY(real, scalar_real_op, real part,\sa ArrayBase::real) +EIGEN_ARRAY_DECLARE_GLOBAL_UNARY(imag, scalar_imag_op, imaginary part,\sa ArrayBase::imag) +EIGEN_ARRAY_DECLARE_GLOBAL_UNARY(conj, scalar_conjugate_op, complex conjugate,\sa ArrayBase::conjugate) +EIGEN_ARRAY_DECLARE_GLOBAL_UNARY(inverse, scalar_inverse_op, inverse,\sa ArrayBase::inverse) +EIGEN_ARRAY_DECLARE_GLOBAL_UNARY(sin, scalar_sin_op, sine,\sa ArrayBase::sin) +EIGEN_ARRAY_DECLARE_GLOBAL_UNARY(cos, scalar_cos_op, cosine,\sa ArrayBase::cos) +EIGEN_ARRAY_DECLARE_GLOBAL_UNARY(tan, scalar_tan_op, tangent,\sa ArrayBase::tan) +EIGEN_ARRAY_DECLARE_GLOBAL_UNARY(atan, scalar_atan_op, arc - tangent,\sa ArrayBase::atan) +EIGEN_ARRAY_DECLARE_GLOBAL_UNARY(asin, scalar_asin_op, arc - sine,\sa ArrayBase::asin) +EIGEN_ARRAY_DECLARE_GLOBAL_UNARY(acos, scalar_acos_op, arc - consine,\sa ArrayBase::acos) +EIGEN_ARRAY_DECLARE_GLOBAL_UNARY(sinh, scalar_sinh_op, hyperbolic sine,\sa ArrayBase::sinh) +EIGEN_ARRAY_DECLARE_GLOBAL_UNARY(cosh, scalar_cosh_op, hyperbolic cosine,\sa ArrayBase::cosh) +EIGEN_ARRAY_DECLARE_GLOBAL_UNARY(tanh, scalar_tanh_op, hyperbolic tangent,\sa ArrayBase::tanh) +EIGEN_ARRAY_DECLARE_GLOBAL_UNARY(lgamma, scalar_lgamma_op, natural logarithm of the gamma function,\sa ArrayBase::lgamma) +EIGEN_ARRAY_DECLARE_GLOBAL_UNARY(digamma, scalar_digamma_op, derivative of lgamma,\sa ArrayBase::digamma) +EIGEN_ARRAY_DECLARE_GLOBAL_UNARY(erf, scalar_erf_op, error function,\sa ArrayBase::erf) +EIGEN_ARRAY_DECLARE_GLOBAL_UNARY(erfc, scalar_erfc_op, complement error function,\sa ArrayBase::erfc) +EIGEN_ARRAY_DECLARE_GLOBAL_UNARY(exp, scalar_exp_op, exponential,\sa ArrayBase::exp) +EIGEN_ARRAY_DECLARE_GLOBAL_UNARY(log, scalar_log_op, natural logarithm,\sa Eigen::log10 DOXCOMMA ArrayBase::log) +EIGEN_ARRAY_DECLARE_GLOBAL_UNARY(log1p, scalar_log1p_op, natural logarithm of 1 plus the value,\sa ArrayBase::log1p) +EIGEN_ARRAY_DECLARE_GLOBAL_UNARY(log10, scalar_log10_op, base 10 logarithm,\sa Eigen::log DOXCOMMA ArrayBase::log) +EIGEN_ARRAY_DECLARE_GLOBAL_UNARY(abs, scalar_abs_op, absolute value,\sa ArrayBase::abs DOXCOMMA MatrixBase::cwiseAbs) +EIGEN_ARRAY_DECLARE_GLOBAL_UNARY(abs2, scalar_abs2_op, squared absolute value,\sa ArrayBase::abs2 DOXCOMMA MatrixBase::cwiseAbs2) +EIGEN_ARRAY_DECLARE_GLOBAL_UNARY(arg, scalar_arg_op, complex argument,\sa ArrayBase::arg) +EIGEN_ARRAY_DECLARE_GLOBAL_UNARY(sqrt, scalar_sqrt_op, square root,\sa ArrayBase::sqrt DOXCOMMA MatrixBase::cwiseSqrt) +EIGEN_ARRAY_DECLARE_GLOBAL_UNARY(rsqrt, scalar_rsqrt_op, reciprocal square root,\sa ArrayBase::rsqrt) +EIGEN_ARRAY_DECLARE_GLOBAL_UNARY(square, scalar_square_op, square(power 2),\sa Eigen::abs2 DOXCOMMA Eigen::pow DOXCOMMA ArrayBase::square) +EIGEN_ARRAY_DECLARE_GLOBAL_UNARY(cube, scalar_cube_op, cube(power 3),\sa Eigen::pow DOXCOMMA ArrayBase::cube) +EIGEN_ARRAY_DECLARE_GLOBAL_UNARY(round, scalar_round_op, nearest integer,\sa Eigen::floor DOXCOMMA Eigen::ceil DOXCOMMA ArrayBase::round) +EIGEN_ARRAY_DECLARE_GLOBAL_UNARY(floor, scalar_floor_op, nearest integer not greater than the giben value,\sa Eigen::ceil DOXCOMMA ArrayBase::floor) +EIGEN_ARRAY_DECLARE_GLOBAL_UNARY(ceil, scalar_ceil_op, nearest integer not less than the giben value,\sa Eigen::floor DOXCOMMA ArrayBase::ceil) +EIGEN_ARRAY_DECLARE_GLOBAL_UNARY(isnan, scalar_isnan_op, not -a - number test,\sa Eigen::isinf DOXCOMMA Eigen::isfinite DOXCOMMA ArrayBase::isnan) +EIGEN_ARRAY_DECLARE_GLOBAL_UNARY(isinf, scalar_isinf_op, infinite value test,\sa Eigen::isnan DOXCOMMA Eigen::isfinite DOXCOMMA ArrayBase::isinf) +EIGEN_ARRAY_DECLARE_GLOBAL_UNARY(isfinite, scalar_isfinite_op, finite value test,\sa Eigen::isinf DOXCOMMA Eigen::isnan DOXCOMMA ArrayBase::isfinite) +EIGEN_ARRAY_DECLARE_GLOBAL_UNARY(sign, scalar_sign_op, sign(or 0),\sa ArrayBase::sign) + +/** \returns an expression of the coefficient-wise power of \a x to the given constant \a exponent. + * + * \tparam ScalarExponent is the scalar type of \a exponent. It must be compatible with the scalar type of the given + * expression (\c Derived::Scalar). + * + * \sa ArrayBase::pow() + * + * \relates ArrayBase + */ #ifdef EIGEN_PARSED_BY_DOXYGEN - template - inline const CwiseBinaryOp,Derived,Constant > - pow(const Eigen::ArrayBase& x, const ScalarExponent& exponent); +template +inline const CwiseBinaryOp, Derived, Constant> + pow(const Eigen::ArrayBase &x, const ScalarExponent &exponent); #else - template - inline typename internal::enable_if< !(internal::is_same::value) && EIGEN_SCALAR_BINARY_SUPPORTED(pow,typename Derived::Scalar,ScalarExponent), - const EIGEN_EXPR_BINARYOP_SCALAR_RETURN_TYPE(Derived,ScalarExponent,pow) >::type - pow(const Eigen::ArrayBase& x, const ScalarExponent& exponent) { - return x.derived().pow(exponent); - } +template +inline typename internal::enable_if::value) + && EIGEN_SCALAR_BINARY_SUPPORTED(pow, typename Derived::Scalar, ScalarExponent), + const EIGEN_EXPR_BINARYOP_SCALAR_RETURN_TYPE(Derived, ScalarExponent, pow)>::type + pow(const Eigen::ArrayBase &x, const ScalarExponent &exponent) +{ + return x.derived().pow(exponent); +} - template - inline const EIGEN_EXPR_BINARYOP_SCALAR_RETURN_TYPE(Derived,typename Derived::Scalar,pow) - pow(const Eigen::ArrayBase& x, const typename Derived::Scalar& exponent) { - return x.derived().pow(exponent); - } +template +inline const EIGEN_EXPR_BINARYOP_SCALAR_RETURN_TYPE(Derived, typename Derived::Scalar, pow) + pow(const Eigen::ArrayBase &x, const typename Derived::Scalar &exponent) +{ + return x.derived().pow(exponent); +} #endif - /** \returns an expression of the coefficient-wise power of \a x to the given array of \a exponents. - * - * This function computes the coefficient-wise power. - * - * Example: \include Cwise_array_power_array.cpp - * Output: \verbinclude Cwise_array_power_array.out - * - * \sa ArrayBase::pow() - * - * \relates ArrayBase - */ - template - inline const Eigen::CwiseBinaryOp, const Derived, const ExponentDerived> - pow(const Eigen::ArrayBase& x, const Eigen::ArrayBase& exponents) - { - return Eigen::CwiseBinaryOp, const Derived, const ExponentDerived>( - x.derived(), - exponents.derived() - ); - } - - /** \returns an expression of the coefficient-wise power of the scalar \a x to the given array of \a exponents. - * - * This function computes the coefficient-wise power between a scalar and an array of exponents. - * - * \tparam Scalar is the scalar type of \a x. It must be compatible with the scalar type of the given array expression (\c Derived::Scalar). - * - * Example: \include Cwise_scalar_power_array.cpp - * Output: \verbinclude Cwise_scalar_power_array.out - * - * \sa ArrayBase::pow() - * - * \relates ArrayBase - */ +/** \returns an expression of the coefficient-wise power of \a x to the given array of \a exponents. + * + * This function computes the coefficient-wise power. + * + * Example: \include Cwise_array_power_array.cpp + * Output: \verbinclude Cwise_array_power_array.out + * + * \sa ArrayBase::pow() + * + * \relates ArrayBase + */ +template +inline const Eigen::CwiseBinaryOp< + Eigen::internal::scalar_pow_op, + const Derived, + const ExponentDerived> + pow(const Eigen::ArrayBase &x, const Eigen::ArrayBase &exponents) +{ + return Eigen::CwiseBinaryOp< + Eigen::internal::scalar_pow_op, + const Derived, + const ExponentDerived>(x.derived(), exponents.derived()); +} + +/** \returns an expression of the coefficient-wise power of the scalar \a x to the given array of \a exponents. + * + * This function computes the coefficient-wise power between a scalar and an array of exponents. + * + * \tparam Scalar is the scalar type of \a x. It must be compatible with the scalar type of the given array expression + * (\c Derived::Scalar). + * + * Example: \include Cwise_scalar_power_array.cpp + * Output: \verbinclude Cwise_scalar_power_array.out + * + * \sa ArrayBase::pow() + * + * \relates ArrayBase + */ #ifdef EIGEN_PARSED_BY_DOXYGEN - template - inline const CwiseBinaryOp,Constant,Derived> - pow(const Scalar& x,const Eigen::ArrayBase& x); +template +inline const CwiseBinaryOp, Constant, Derived> + pow(const Scalar &x, const Eigen::ArrayBase &x); #else - template - inline typename internal::enable_if< !(internal::is_same::value) && EIGEN_SCALAR_BINARY_SUPPORTED(pow,Scalar,typename Derived::Scalar), - const EIGEN_SCALAR_BINARYOP_EXPR_RETURN_TYPE(Scalar,Derived,pow) >::type - pow(const Scalar& x, const Eigen::ArrayBase& exponents) - { - return EIGEN_SCALAR_BINARYOP_EXPR_RETURN_TYPE(Scalar,Derived,pow)( - typename internal::plain_constant_type::type(exponents.rows(), exponents.cols(), x), exponents.derived() ); - } +template +inline typename internal::enable_if::value) + && EIGEN_SCALAR_BINARY_SUPPORTED(pow, Scalar, typename Derived::Scalar), + const EIGEN_SCALAR_BINARYOP_EXPR_RETURN_TYPE(Scalar, Derived, pow)>::type pow(const Scalar &x, + const Eigen::ArrayBase &exponents) +{ + return EIGEN_SCALAR_BINARYOP_EXPR_RETURN_TYPE(Scalar, Derived, pow)( + typename internal::plain_constant_type::type(exponents.rows(), exponents.cols(), x), + exponents.derived()); +} - template - inline const EIGEN_SCALAR_BINARYOP_EXPR_RETURN_TYPE(typename Derived::Scalar,Derived,pow) - pow(const typename Derived::Scalar& x, const Eigen::ArrayBase& exponents) - { - return EIGEN_SCALAR_BINARYOP_EXPR_RETURN_TYPE(typename Derived::Scalar,Derived,pow)( - typename internal::plain_constant_type::type(exponents.rows(), exponents.cols(), x), exponents.derived() ); - } +template +inline const EIGEN_SCALAR_BINARYOP_EXPR_RETURN_TYPE(typename Derived::Scalar, Derived, pow) + pow(const typename Derived::Scalar &x, const Eigen::ArrayBase &exponents) +{ + return EIGEN_SCALAR_BINARYOP_EXPR_RETURN_TYPE(typename Derived::Scalar, Derived, pow)( + typename internal::plain_constant_type::type( + exponents.rows(), exponents.cols(), x), + exponents.derived()); +} #endif - namespace internal - { - EIGEN_ARRAY_DECLARE_GLOBAL_EIGEN_UNARY(real,scalar_real_op) - EIGEN_ARRAY_DECLARE_GLOBAL_EIGEN_UNARY(imag,scalar_imag_op) - EIGEN_ARRAY_DECLARE_GLOBAL_EIGEN_UNARY(abs2,scalar_abs2_op) - } -} +namespace internal { + EIGEN_ARRAY_DECLARE_GLOBAL_EIGEN_UNARY(real, scalar_real_op) + EIGEN_ARRAY_DECLARE_GLOBAL_EIGEN_UNARY(imag, scalar_imag_op) + EIGEN_ARRAY_DECLARE_GLOBAL_EIGEN_UNARY(abs2, scalar_abs2_op) +}// namespace internal +}// namespace Eigen -// TODO: cleanly disable those functions that are not supported on Array (numext::real_ref, internal::random, internal::isApprox...) +// TODO: cleanly disable those functions that are not supported on Array (numext::real_ref, internal::random, +// internal::isApprox...) -#endif // EIGEN_GLOBAL_FUNCTIONS_H +#endif// EIGEN_GLOBAL_FUNCTIONS_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/IO.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/IO.h index da7fd6cc..06440a05 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/IO.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/IO.h @@ -11,59 +11,59 @@ #ifndef EIGEN_IO_H #define EIGEN_IO_H -namespace Eigen { +namespace Eigen { enum { DontAlignCols = 1 }; -enum { StreamPrecision = -1, - FullPrecision = -2 }; +enum { StreamPrecision = -1, FullPrecision = -2 }; namespace internal { -template -std::ostream & print_matrix(std::ostream & s, const Derived& _m, const IOFormat& fmt); + template std::ostream &print_matrix(std::ostream &s, const Derived &_m, const IOFormat &fmt); } /** \class IOFormat - * \ingroup Core_Module - * - * \brief Stores a set of parameters controlling the way matrices are printed - * - * List of available parameters: - * - \b precision number of digits for floating point values, or one of the special constants \c StreamPrecision and \c FullPrecision. - * The default is the special value \c StreamPrecision which means to use the - * stream's own precision setting, as set for instance using \c cout.precision(3). The other special value - * \c FullPrecision means that the number of digits will be computed to match the full precision of each floating-point - * type. - * - \b flags an OR-ed combination of flags, the default value is 0, the only currently available flag is \c DontAlignCols which - * allows to disable the alignment of columns, resulting in faster code. - * - \b coeffSeparator string printed between two coefficients of the same row - * - \b rowSeparator string printed between two rows - * - \b rowPrefix string printed at the beginning of each row - * - \b rowSuffix string printed at the end of each row - * - \b matPrefix string printed at the beginning of the matrix - * - \b matSuffix string printed at the end of the matrix - * - * Example: \include IOFormat.cpp - * Output: \verbinclude IOFormat.out - * - * \sa DenseBase::format(), class WithFormat - */ + * \ingroup Core_Module + * + * \brief Stores a set of parameters controlling the way matrices are printed + * + * List of available parameters: + * - \b precision number of digits for floating point values, or one of the special constants \c StreamPrecision and \c + * FullPrecision. The default is the special value \c StreamPrecision which means to use the stream's own precision + * setting, as set for instance using \c cout.precision(3). The other special value + * \c FullPrecision means that the number of digits will be computed to match the full precision of each + * floating-point type. + * - \b flags an OR-ed combination of flags, the default value is 0, the only currently available flag is \c + * DontAlignCols which allows to disable the alignment of columns, resulting in faster code. + * - \b coeffSeparator string printed between two coefficients of the same row + * - \b rowSeparator string printed between two rows + * - \b rowPrefix string printed at the beginning of each row + * - \b rowSuffix string printed at the end of each row + * - \b matPrefix string printed at the beginning of the matrix + * - \b matSuffix string printed at the end of the matrix + * + * Example: \include IOFormat.cpp + * Output: \verbinclude IOFormat.out + * + * \sa DenseBase::format(), class WithFormat + */ struct IOFormat { /** Default constructor, see class IOFormat for the meaning of the parameters */ - IOFormat(int _precision = StreamPrecision, int _flags = 0, - const std::string& _coeffSeparator = " ", - const std::string& _rowSeparator = "\n", const std::string& _rowPrefix="", const std::string& _rowSuffix="", - const std::string& _matPrefix="", const std::string& _matSuffix="") - : matPrefix(_matPrefix), matSuffix(_matSuffix), rowPrefix(_rowPrefix), rowSuffix(_rowSuffix), rowSeparator(_rowSeparator), - rowSpacer(""), coeffSeparator(_coeffSeparator), precision(_precision), flags(_flags) + IOFormat(int _precision = StreamPrecision, + int _flags = 0, + const std::string &_coeffSeparator = " ", + const std::string &_rowSeparator = "\n", + const std::string &_rowPrefix = "", + const std::string &_rowSuffix = "", + const std::string &_matPrefix = "", + const std::string &_matSuffix = "") + : matPrefix(_matPrefix), matSuffix(_matSuffix), rowPrefix(_rowPrefix), rowSuffix(_rowSuffix), + rowSeparator(_rowSeparator), rowSpacer(""), coeffSeparator(_coeffSeparator), precision(_precision), flags(_flags) { // TODO check if rowPrefix, rowSuffix or rowSeparator contains a newline // don't add rowSpacer if columns are not to be aligned - if((flags & DontAlignCols)) - return; - int i = int(matSuffix.length())-1; - while (i>=0 && matSuffix[i]!='\n') - { + if ((flags & DontAlignCols)) return; + int i = int(matSuffix.length()) - 1; + while (i >= 0 && matSuffix[i] != '\n') { rowSpacer += ' '; i--; } @@ -76,150 +76,124 @@ struct IOFormat }; /** \class WithFormat - * \ingroup Core_Module - * - * \brief Pseudo expression providing matrix output with given format - * - * \tparam ExpressionType the type of the object on which IO stream operations are performed - * - * This class represents an expression with stream operators controlled by a given IOFormat. - * It is the return type of DenseBase::format() - * and most of the time this is the only way it is used. - * - * See class IOFormat for some examples. - * - * \sa DenseBase::format(), class IOFormat - */ -template -class WithFormat + * \ingroup Core_Module + * + * \brief Pseudo expression providing matrix output with given format + * + * \tparam ExpressionType the type of the object on which IO stream operations are performed + * + * This class represents an expression with stream operators controlled by a given IOFormat. + * It is the return type of DenseBase::format() + * and most of the time this is the only way it is used. + * + * See class IOFormat for some examples. + * + * \sa DenseBase::format(), class IOFormat + */ +template class WithFormat { - public: +public: + WithFormat(const ExpressionType &matrix, const IOFormat &format) : m_matrix(matrix), m_format(format) {} - WithFormat(const ExpressionType& matrix, const IOFormat& format) - : m_matrix(matrix), m_format(format) - {} - - friend std::ostream & operator << (std::ostream & s, const WithFormat& wf) - { - return internal::print_matrix(s, wf.m_matrix.eval(), wf.m_format); - } + friend std::ostream &operator<<(std::ostream &s, const WithFormat &wf) + { + return internal::print_matrix(s, wf.m_matrix.eval(), wf.m_format); + } - protected: - typename ExpressionType::Nested m_matrix; - IOFormat m_format; +protected: + typename ExpressionType::Nested m_matrix; + IOFormat m_format; }; namespace internal { -// NOTE: This helper is kept for backward compatibility with previous code specializing -// this internal::significant_decimals_impl structure. In the future we should directly -// call digits10() which has been introduced in July 2016 in 3.3. -template -struct significant_decimals_impl -{ - static inline int run() + // NOTE: This helper is kept for backward compatibility with previous code specializing + // this internal::significant_decimals_impl structure. In the future we should directly + // call digits10() which has been introduced in July 2016 in 3.3. + template struct significant_decimals_impl { - return NumTraits::digits10(); - } -}; + static inline int run() { return NumTraits::digits10(); } + }; -/** \internal - * print the matrix \a _m to the output stream \a s using the output format \a fmt */ -template -std::ostream & print_matrix(std::ostream & s, const Derived& _m, const IOFormat& fmt) -{ - if(_m.size() == 0) + /** \internal + * print the matrix \a _m to the output stream \a s using the output format \a fmt */ + template std::ostream &print_matrix(std::ostream &s, const Derived &_m, const IOFormat &fmt) { - s << fmt.matPrefix << fmt.matSuffix; - return s; - } - - typename Derived::Nested m = _m; - typedef typename Derived::Scalar Scalar; + if (_m.size() == 0) { + s << fmt.matPrefix << fmt.matSuffix; + return s; + } - Index width = 0; + typename Derived::Nested m = _m; + typedef typename Derived::Scalar Scalar; - std::streamsize explicit_precision; - if(fmt.precision == StreamPrecision) - { - explicit_precision = 0; - } - else if(fmt.precision == FullPrecision) - { - if (NumTraits::IsInteger) - { + Index width = 0; + + std::streamsize explicit_precision; + if (fmt.precision == StreamPrecision) { explicit_precision = 0; + } else if (fmt.precision == FullPrecision) { + if (NumTraits::IsInteger) { + explicit_precision = 0; + } else { + explicit_precision = significant_decimals_impl::run(); + } + } else { + explicit_precision = fmt.precision; } - else - { - explicit_precision = significant_decimals_impl::run(); - } - } - else - { - explicit_precision = fmt.precision; - } - - std::streamsize old_precision = 0; - if(explicit_precision) old_precision = s.precision(explicit_precision); - bool align_cols = !(fmt.flags & DontAlignCols); - if(align_cols) - { - // compute the largest width - for(Index j = 0; j < m.cols(); ++j) - for(Index i = 0; i < m.rows(); ++i) - { - std::stringstream sstr; - sstr.copyfmt(s); - sstr << m.coeff(i,j); - width = std::max(width, Index(sstr.str().length())); - } - } - s << fmt.matPrefix; - for(Index i = 0; i < m.rows(); ++i) - { - if (i) - s << fmt.rowSpacer; - s << fmt.rowPrefix; - if(width) s.width(width); - s << m.coeff(i, 0); - for(Index j = 1; j < m.cols(); ++j) - { - s << fmt.coeffSeparator; + std::streamsize old_precision = 0; + if (explicit_precision) old_precision = s.precision(explicit_precision); + + bool align_cols = !(fmt.flags & DontAlignCols); + if (align_cols) { + // compute the largest width + for (Index j = 0; j < m.cols(); ++j) + for (Index i = 0; i < m.rows(); ++i) { + std::stringstream sstr; + sstr.copyfmt(s); + sstr << m.coeff(i, j); + width = std::max(width, Index(sstr.str().length())); + } + } + s << fmt.matPrefix; + for (Index i = 0; i < m.rows(); ++i) { + if (i) s << fmt.rowSpacer; + s << fmt.rowPrefix; if (width) s.width(width); - s << m.coeff(i, j); + s << m.coeff(i, 0); + for (Index j = 1; j < m.cols(); ++j) { + s << fmt.coeffSeparator; + if (width) s.width(width); + s << m.coeff(i, j); + } + s << fmt.rowSuffix; + if (i < m.rows() - 1) s << fmt.rowSeparator; } - s << fmt.rowSuffix; - if( i < m.rows() - 1) - s << fmt.rowSeparator; + s << fmt.matSuffix; + if (explicit_precision) s.precision(old_precision); + return s; } - s << fmt.matSuffix; - if(explicit_precision) s.precision(old_precision); - return s; -} -} // end namespace internal +}// end namespace internal /** \relates DenseBase - * - * Outputs the matrix, to the given stream. - * - * If you wish to print the matrix with a format different than the default, use DenseBase::format(). - * - * It is also possible to change the default format by defining EIGEN_DEFAULT_IO_FORMAT before including Eigen headers. - * If not defined, this will automatically be defined to Eigen::IOFormat(), that is the Eigen::IOFormat with default parameters. - * - * \sa DenseBase::format() - */ -template -std::ostream & operator << -(std::ostream & s, - const DenseBase & m) + * + * Outputs the matrix, to the given stream. + * + * If you wish to print the matrix with a format different than the default, use DenseBase::format(). + * + * It is also possible to change the default format by defining EIGEN_DEFAULT_IO_FORMAT before including Eigen headers. + * If not defined, this will automatically be defined to Eigen::IOFormat(), that is the Eigen::IOFormat with default + * parameters. + * + * \sa DenseBase::format() + */ +template std::ostream &operator<<(std::ostream &s, const DenseBase &m) { return internal::print_matrix(s, m.eval(), EIGEN_DEFAULT_IO_FORMAT); } -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_IO_H +#endif// EIGEN_IO_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/Inverse.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/Inverse.h index b76f0439..ea42d218 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/Inverse.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/Inverse.h @@ -10,55 +10,48 @@ #ifndef EIGEN_INVERSE_H #define EIGEN_INVERSE_H -namespace Eigen { +namespace Eigen { -template class InverseImpl; +template class InverseImpl; namespace internal { -template -struct traits > - : traits -{ - typedef typename XprType::PlainObject PlainObject; - typedef traits BaseTraits; - enum { - Flags = BaseTraits::Flags & RowMajorBit + template struct traits> : traits + { + typedef typename XprType::PlainObject PlainObject; + typedef traits BaseTraits; + enum { Flags = BaseTraits::Flags & RowMajorBit }; }; -}; -} // end namespace internal +}// end namespace internal /** \class Inverse - * - * \brief Expression of the inverse of another expression - * - * \tparam XprType the type of the expression we are taking the inverse - * - * This class represents an abstract expression of A.inverse() - * and most of the time this is the only way it is used. - * - */ -template -class Inverse : public InverseImpl::StorageKind> + * + * \brief Expression of the inverse of another expression + * + * \tparam XprType the type of the expression we are taking the inverse + * + * This class represents an abstract expression of A.inverse() + * and most of the time this is the only way it is used. + * + */ +template class Inverse : public InverseImpl::StorageKind> { public: typedef typename XprType::StorageIndex StorageIndex; - typedef typename XprType::PlainObject PlainObject; - typedef typename XprType::Scalar Scalar; - typedef typename internal::ref_selector::type XprTypeNested; - typedef typename internal::remove_all::type XprTypeNestedCleaned; + typedef typename XprType::PlainObject PlainObject; + typedef typename XprType::Scalar Scalar; + typedef typename internal::ref_selector::type XprTypeNested; + typedef typename internal::remove_all::type XprTypeNestedCleaned; typedef typename internal::ref_selector::type Nested; typedef typename internal::remove_all::type NestedExpression; - - explicit EIGEN_DEVICE_FUNC Inverse(const XprType &xpr) - : m_xpr(xpr) - {} + + explicit EIGEN_DEVICE_FUNC Inverse(const XprType &xpr) : m_xpr(xpr) {} EIGEN_DEVICE_FUNC Index rows() const { return m_xpr.rows(); } EIGEN_DEVICE_FUNC Index cols() const { return m_xpr.cols(); } - EIGEN_DEVICE_FUNC const XprTypeNestedCleaned& nestedExpression() const { return m_xpr; } + EIGEN_DEVICE_FUNC const XprTypeNestedCleaned &nestedExpression() const { return m_xpr; } protected: XprTypeNested m_xpr; @@ -66,53 +59,50 @@ class Inverse : public InverseImpl::S // Generic API dispatcher template -class InverseImpl - : public internal::generic_xpr_base >::type +class InverseImpl : public internal::generic_xpr_base>::type { public: - typedef typename internal::generic_xpr_base >::type Base; + typedef typename internal::generic_xpr_base>::type Base; typedef typename XprType::Scalar Scalar; -private: +private: Scalar coeff(Index row, Index col) const; Scalar coeff(Index i) const; }; namespace internal { -/** \internal - * \brief Default evaluator for Inverse expression. - * - * This default evaluator for Inverse expression simply evaluate the inverse into a temporary - * by a call to internal::call_assignment_no_alias. - * Therefore, inverse implementers only have to specialize Assignment, ...> for - * there own nested expression. - * - * \sa class Inverse - */ -template -struct unary_evaluator > - : public evaluator::PlainObject> -{ - typedef Inverse InverseType; - typedef typename InverseType::PlainObject PlainObject; - typedef evaluator Base; - - enum { Flags = Base::Flags | EvalBeforeNestingBit }; - - unary_evaluator(const InverseType& inv_xpr) - : m_result(inv_xpr.rows(), inv_xpr.cols()) + /** \internal + * \brief Default evaluator for Inverse expression. + * + * This default evaluator for Inverse expression simply evaluate the inverse into a temporary + * by a call to internal::call_assignment_no_alias. + * Therefore, inverse implementers only have to specialize Assignment, ...> for + * there own nested expression. + * + * \sa class Inverse + */ + template + struct unary_evaluator> : public evaluator::PlainObject> { - ::new (static_cast(this)) Base(m_result); - internal::call_assignment_no_alias(m_result, inv_xpr); - } - -protected: - PlainObject m_result; -}; - -} // end namespace internal + typedef Inverse InverseType; + typedef typename InverseType::PlainObject PlainObject; + typedef evaluator Base; + + enum { Flags = Base::Flags | EvalBeforeNestingBit }; + + unary_evaluator(const InverseType &inv_xpr) : m_result(inv_xpr.rows(), inv_xpr.cols()) + { + ::new (static_cast(this)) Base(m_result); + internal::call_assignment_no_alias(m_result, inv_xpr); + } + + protected: + PlainObject m_result; + }; + +}// end namespace internal -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_INVERSE_H +#endif// EIGEN_INVERSE_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/Map.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/Map.h index 548bf9a2..e309e8c7 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/Map.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/Map.h @@ -11,161 +11,158 @@ #ifndef EIGEN_MAP_H #define EIGEN_MAP_H -namespace Eigen { +namespace Eigen { namespace internal { -template -struct traits > - : public traits -{ - typedef traits TraitsBase; - enum { - PlainObjectTypeInnerSize = ((traits::Flags&RowMajorBit)==RowMajorBit) - ? PlainObjectType::ColsAtCompileTime - : PlainObjectType::RowsAtCompileTime, - - InnerStrideAtCompileTime = StrideType::InnerStrideAtCompileTime == 0 - ? int(PlainObjectType::InnerStrideAtCompileTime) - : int(StrideType::InnerStrideAtCompileTime), - OuterStrideAtCompileTime = StrideType::OuterStrideAtCompileTime == 0 - ? (InnerStrideAtCompileTime==Dynamic || PlainObjectTypeInnerSize==Dynamic - ? Dynamic - : int(InnerStrideAtCompileTime) * int(PlainObjectTypeInnerSize)) - : int(StrideType::OuterStrideAtCompileTime), - Alignment = int(MapOptions)&int(AlignedMask), - Flags0 = TraitsBase::Flags & (~NestByRefBit), - Flags = is_lvalue::value ? int(Flags0) : (int(Flags0) & ~LvalueBit) + template + struct traits> : public traits + { + typedef traits TraitsBase; + enum { + PlainObjectTypeInnerSize = ((traits::Flags & RowMajorBit) == RowMajorBit) + ? PlainObjectType::ColsAtCompileTime + : PlainObjectType::RowsAtCompileTime, + + InnerStrideAtCompileTime = StrideType::InnerStrideAtCompileTime == 0 + ? int(PlainObjectType::InnerStrideAtCompileTime) + : int(StrideType::InnerStrideAtCompileTime), + OuterStrideAtCompileTime = StrideType::OuterStrideAtCompileTime == 0 + ? (InnerStrideAtCompileTime == Dynamic || PlainObjectTypeInnerSize == Dynamic + ? Dynamic + : int(InnerStrideAtCompileTime) * int(PlainObjectTypeInnerSize)) + : int(StrideType::OuterStrideAtCompileTime), + Alignment = int(MapOptions) & int(AlignedMask), + Flags0 = TraitsBase::Flags & (~NestByRefBit), + Flags = is_lvalue::value ? int(Flags0) : (int(Flags0) & ~LvalueBit) + }; + + private: + enum { Options };// Expressions don't have Options }; -private: - enum { Options }; // Expressions don't have Options -}; -} +}// namespace internal /** \class Map - * \ingroup Core_Module - * - * \brief A matrix or vector expression mapping an existing array of data. - * - * \tparam PlainObjectType the equivalent matrix type of the mapped data - * \tparam MapOptions specifies the pointer alignment in bytes. It can be: \c #Aligned128, , \c #Aligned64, \c #Aligned32, \c #Aligned16, \c #Aligned8 or \c #Unaligned. - * The default is \c #Unaligned. - * \tparam StrideType optionally specifies strides. By default, Map assumes the memory layout - * of an ordinary, contiguous array. This can be overridden by specifying strides. - * The type passed here must be a specialization of the Stride template, see examples below. - * - * This class represents a matrix or vector expression mapping an existing array of data. - * It can be used to let Eigen interface without any overhead with non-Eigen data structures, - * such as plain C arrays or structures from other libraries. By default, it assumes that the - * data is laid out contiguously in memory. You can however override this by explicitly specifying - * inner and outer strides. - * - * Here's an example of simply mapping a contiguous array as a \ref TopicStorageOrders "column-major" matrix: - * \include Map_simple.cpp - * Output: \verbinclude Map_simple.out - * - * If you need to map non-contiguous arrays, you can do so by specifying strides: - * - * Here's an example of mapping an array as a vector, specifying an inner stride, that is, the pointer - * increment between two consecutive coefficients. Here, we're specifying the inner stride as a compile-time - * fixed value. - * \include Map_inner_stride.cpp - * Output: \verbinclude Map_inner_stride.out - * - * Here's an example of mapping an array while specifying an outer stride. Here, since we're mapping - * as a column-major matrix, 'outer stride' means the pointer increment between two consecutive columns. - * Here, we're specifying the outer stride as a runtime parameter. Note that here \c OuterStride<> is - * a short version of \c OuterStride because the default template parameter of OuterStride - * is \c Dynamic - * \include Map_outer_stride.cpp - * Output: \verbinclude Map_outer_stride.out - * - * For more details and for an example of specifying both an inner and an outer stride, see class Stride. - * - * \b Tip: to change the array of data mapped by a Map object, you can use the C++ - * placement new syntax: - * - * Example: \include Map_placement_new.cpp - * Output: \verbinclude Map_placement_new.out - * - * This class is the return type of PlainObjectBase::Map() but can also be used directly. - * - * \sa PlainObjectBase::Map(), \ref TopicStorageOrders - */ -template class Map - : public MapBase > + * \ingroup Core_Module + * + * \brief A matrix or vector expression mapping an existing array of data. + * + * \tparam PlainObjectType the equivalent matrix type of the mapped data + * \tparam MapOptions specifies the pointer alignment in bytes. It can be: \c #Aligned128, , \c #Aligned64, \c + * #Aligned32, \c #Aligned16, \c #Aligned8 or \c #Unaligned. The default is \c #Unaligned. + * \tparam StrideType optionally specifies strides. By default, Map assumes the memory layout + * of an ordinary, contiguous array. This can be overridden by specifying strides. + * The type passed here must be a specialization of the Stride template, see examples below. + * + * This class represents a matrix or vector expression mapping an existing array of data. + * It can be used to let Eigen interface without any overhead with non-Eigen data structures, + * such as plain C arrays or structures from other libraries. By default, it assumes that the + * data is laid out contiguously in memory. You can however override this by explicitly specifying + * inner and outer strides. + * + * Here's an example of simply mapping a contiguous array as a \ref TopicStorageOrders "column-major" matrix: + * \include Map_simple.cpp + * Output: \verbinclude Map_simple.out + * + * If you need to map non-contiguous arrays, you can do so by specifying strides: + * + * Here's an example of mapping an array as a vector, specifying an inner stride, that is, the pointer + * increment between two consecutive coefficients. Here, we're specifying the inner stride as a compile-time + * fixed value. + * \include Map_inner_stride.cpp + * Output: \verbinclude Map_inner_stride.out + * + * Here's an example of mapping an array while specifying an outer stride. Here, since we're mapping + * as a column-major matrix, 'outer stride' means the pointer increment between two consecutive columns. + * Here, we're specifying the outer stride as a runtime parameter. Note that here \c OuterStride<> is + * a short version of \c OuterStride because the default template parameter of OuterStride + * is \c Dynamic + * \include Map_outer_stride.cpp + * Output: \verbinclude Map_outer_stride.out + * + * For more details and for an example of specifying both an inner and an outer stride, see class Stride. + * + * \b Tip: to change the array of data mapped by a Map object, you can use the C++ + * placement new syntax: + * + * Example: \include Map_placement_new.cpp + * Output: \verbinclude Map_placement_new.out + * + * This class is the return type of PlainObjectBase::Map() but can also be used directly. + * + * \sa PlainObjectBase::Map(), \ref TopicStorageOrders + */ +template +class Map : public MapBase> { - public: - - typedef MapBase Base; - EIGEN_DENSE_PUBLIC_INTERFACE(Map) - - typedef typename Base::PointerType PointerType; - typedef PointerType PointerArgType; - EIGEN_DEVICE_FUNC - inline PointerType cast_to_pointer_type(PointerArgType ptr) { return ptr; } - - EIGEN_DEVICE_FUNC - inline Index innerStride() const - { - return StrideType::InnerStrideAtCompileTime != 0 ? m_stride.inner() : 1; - } - - EIGEN_DEVICE_FUNC - inline Index outerStride() const - { - return int(StrideType::OuterStrideAtCompileTime) != 0 ? m_stride.outer() - : int(internal::traits::OuterStrideAtCompileTime) != Dynamic ? Index(internal::traits::OuterStrideAtCompileTime) - : IsVectorAtCompileTime ? (this->size() * innerStride()) - : (int(Flags)&RowMajorBit) ? (this->cols() * innerStride()) - : (this->rows() * innerStride()); - } - - /** Constructor in the fixed-size case. - * - * \param dataPtr pointer to the array to map - * \param stride optional Stride object, passing the strides. - */ - EIGEN_DEVICE_FUNC - explicit inline Map(PointerArgType dataPtr, const StrideType& stride = StrideType()) - : Base(cast_to_pointer_type(dataPtr)), m_stride(stride) - { - PlainObjectType::Base::_check_template_params(); - } - - /** Constructor in the dynamic-size vector case. - * - * \param dataPtr pointer to the array to map - * \param size the size of the vector expression - * \param stride optional Stride object, passing the strides. - */ - EIGEN_DEVICE_FUNC - inline Map(PointerArgType dataPtr, Index size, const StrideType& stride = StrideType()) - : Base(cast_to_pointer_type(dataPtr), size), m_stride(stride) - { - PlainObjectType::Base::_check_template_params(); - } - - /** Constructor in the dynamic-size matrix case. - * - * \param dataPtr pointer to the array to map - * \param rows the number of rows of the matrix expression - * \param cols the number of columns of the matrix expression - * \param stride optional Stride object, passing the strides. - */ - EIGEN_DEVICE_FUNC - inline Map(PointerArgType dataPtr, Index rows, Index cols, const StrideType& stride = StrideType()) - : Base(cast_to_pointer_type(dataPtr), rows, cols), m_stride(stride) - { - PlainObjectType::Base::_check_template_params(); - } - - EIGEN_INHERIT_ASSIGNMENT_OPERATORS(Map) - - protected: - StrideType m_stride; +public: + typedef MapBase Base; + EIGEN_DENSE_PUBLIC_INTERFACE(Map) + + typedef typename Base::PointerType PointerType; + typedef PointerType PointerArgType; + EIGEN_DEVICE_FUNC + inline PointerType cast_to_pointer_type(PointerArgType ptr) { return ptr; } + + EIGEN_DEVICE_FUNC + inline Index innerStride() const { return StrideType::InnerStrideAtCompileTime != 0 ? m_stride.inner() : 1; } + + EIGEN_DEVICE_FUNC + inline Index outerStride() const + { + return int(StrideType::OuterStrideAtCompileTime) != 0 ? m_stride.outer() + : int(internal::traits::OuterStrideAtCompileTime) != Dynamic + ? Index(internal::traits::OuterStrideAtCompileTime) + : IsVectorAtCompileTime ? (this->size() * innerStride()) + : (int(Flags) & RowMajorBit) ? (this->cols() * innerStride()) + : (this->rows() * innerStride()); + } + + /** Constructor in the fixed-size case. + * + * \param dataPtr pointer to the array to map + * \param stride optional Stride object, passing the strides. + */ + EIGEN_DEVICE_FUNC + explicit inline Map(PointerArgType dataPtr, const StrideType &stride = StrideType()) + : Base(cast_to_pointer_type(dataPtr)), m_stride(stride) + { + PlainObjectType::Base::_check_template_params(); + } + + /** Constructor in the dynamic-size vector case. + * + * \param dataPtr pointer to the array to map + * \param size the size of the vector expression + * \param stride optional Stride object, passing the strides. + */ + EIGEN_DEVICE_FUNC + inline Map(PointerArgType dataPtr, Index size, const StrideType &stride = StrideType()) + : Base(cast_to_pointer_type(dataPtr), size), m_stride(stride) + { + PlainObjectType::Base::_check_template_params(); + } + + /** Constructor in the dynamic-size matrix case. + * + * \param dataPtr pointer to the array to map + * \param rows the number of rows of the matrix expression + * \param cols the number of columns of the matrix expression + * \param stride optional Stride object, passing the strides. + */ + EIGEN_DEVICE_FUNC + inline Map(PointerArgType dataPtr, Index rows, Index cols, const StrideType &stride = StrideType()) + : Base(cast_to_pointer_type(dataPtr), rows, cols), m_stride(stride) + { + PlainObjectType::Base::_check_template_params(); + } + + EIGEN_INHERIT_ASSIGNMENT_OPERATORS(Map) + +protected: + StrideType m_stride; }; -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_MAP_H +#endif// EIGEN_MAP_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/MapBase.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/MapBase.h index 668922ff..a5d82e4b 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/MapBase.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/MapBase.h @@ -11,293 +11,279 @@ #ifndef EIGEN_MAPBASE_H #define EIGEN_MAPBASE_H -#define EIGEN_STATIC_ASSERT_INDEX_BASED_ACCESS(Derived) \ - EIGEN_STATIC_ASSERT((int(internal::evaluator::Flags) & LinearAccessBit) || Derived::IsVectorAtCompileTime, \ - YOU_ARE_TRYING_TO_USE_AN_INDEX_BASED_ACCESSOR_ON_AN_EXPRESSION_THAT_DOES_NOT_SUPPORT_THAT) +#define EIGEN_STATIC_ASSERT_INDEX_BASED_ACCESS(Derived) \ + EIGEN_STATIC_ASSERT((int(internal::evaluator::Flags) & LinearAccessBit) || Derived::IsVectorAtCompileTime, \ + YOU_ARE_TRYING_TO_USE_AN_INDEX_BASED_ACCESSOR_ON_AN_EXPRESSION_THAT_DOES_NOT_SUPPORT_THAT) -namespace Eigen { +namespace Eigen { /** \ingroup Core_Module - * - * \brief Base class for dense Map and Block expression with direct access - * - * This base class provides the const low-level accessors (e.g. coeff, coeffRef) of dense - * Map and Block objects with direct access. - * Typical users do not have to directly deal with this class. - * - * This class can be extended by through the macro plugin \c EIGEN_MAPBASE_PLUGIN. - * See \link TopicCustomizing_Plugins customizing Eigen \endlink for details. - * - * The \c Derived class has to provide the following two methods describing the memory layout: - * \code Index innerStride() const; \endcode - * \code Index outerStride() const; \endcode - * - * \sa class Map, class Block - */ -template class MapBase - : public internal::dense_xpr_base::type + * + * \brief Base class for dense Map and Block expression with direct access + * + * This base class provides the const low-level accessors (e.g. coeff, coeffRef) of dense + * Map and Block objects with direct access. + * Typical users do not have to directly deal with this class. + * + * This class can be extended by through the macro plugin \c EIGEN_MAPBASE_PLUGIN. + * See \link TopicCustomizing_Plugins customizing Eigen \endlink for details. + * + * The \c Derived class has to provide the following two methods describing the memory layout: + * \code Index innerStride() const; \endcode + * \code Index outerStride() const; \endcode + * + * \sa class Map, class Block + */ +template class MapBase : public internal::dense_xpr_base::type { - public: - - typedef typename internal::dense_xpr_base::type Base; - enum { - RowsAtCompileTime = internal::traits::RowsAtCompileTime, - ColsAtCompileTime = internal::traits::ColsAtCompileTime, - InnerStrideAtCompileTime = internal::traits::InnerStrideAtCompileTime, - SizeAtCompileTime = Base::SizeAtCompileTime - }; - - typedef typename internal::traits::StorageKind StorageKind; - typedef typename internal::traits::Scalar Scalar; - typedef typename internal::packet_traits::type PacketScalar; - typedef typename NumTraits::Real RealScalar; - typedef typename internal::conditional< - bool(internal::is_lvalue::value), - Scalar *, - const Scalar *>::type - PointerType; - - using Base::derived; -// using Base::RowsAtCompileTime; -// using Base::ColsAtCompileTime; -// using Base::SizeAtCompileTime; - using Base::MaxRowsAtCompileTime; - using Base::MaxColsAtCompileTime; - using Base::MaxSizeAtCompileTime; - using Base::IsVectorAtCompileTime; - using Base::Flags; - using Base::IsRowMajor; - - using Base::rows; - using Base::cols; - using Base::size; - using Base::coeff; - using Base::coeffRef; - using Base::lazyAssign; - using Base::eval; - - using Base::innerStride; - using Base::outerStride; - using Base::rowStride; - using Base::colStride; - - // bug 217 - compile error on ICC 11.1 - using Base::operator=; - - typedef typename Base::CoeffReturnType CoeffReturnType; - - /** \copydoc DenseBase::rows() */ - EIGEN_DEVICE_FUNC inline Index rows() const { return m_rows.value(); } - /** \copydoc DenseBase::cols() */ - EIGEN_DEVICE_FUNC inline Index cols() const { return m_cols.value(); } - - /** Returns a pointer to the first coefficient of the matrix or vector. - * - * \note When addressing this data, make sure to honor the strides returned by innerStride() and outerStride(). - * - * \sa innerStride(), outerStride() - */ - EIGEN_DEVICE_FUNC inline const Scalar* data() const { return m_data; } - - /** \copydoc PlainObjectBase::coeff(Index,Index) const */ - EIGEN_DEVICE_FUNC - inline const Scalar& coeff(Index rowId, Index colId) const - { - return m_data[colId * colStride() + rowId * rowStride()]; - } - - /** \copydoc PlainObjectBase::coeff(Index) const */ - EIGEN_DEVICE_FUNC - inline const Scalar& coeff(Index index) const - { - EIGEN_STATIC_ASSERT_INDEX_BASED_ACCESS(Derived) - return m_data[index * innerStride()]; - } - - /** \copydoc PlainObjectBase::coeffRef(Index,Index) const */ - EIGEN_DEVICE_FUNC - inline const Scalar& coeffRef(Index rowId, Index colId) const - { - return this->m_data[colId * colStride() + rowId * rowStride()]; - } - - /** \copydoc PlainObjectBase::coeffRef(Index) const */ - EIGEN_DEVICE_FUNC - inline const Scalar& coeffRef(Index index) const - { - EIGEN_STATIC_ASSERT_INDEX_BASED_ACCESS(Derived) - return this->m_data[index * innerStride()]; - } - - /** \internal */ - template - inline PacketScalar packet(Index rowId, Index colId) const - { - return internal::ploadt - (m_data + (colId * colStride() + rowId * rowStride())); - } - - /** \internal */ - template - inline PacketScalar packet(Index index) const - { - EIGEN_STATIC_ASSERT_INDEX_BASED_ACCESS(Derived) - return internal::ploadt(m_data + index * innerStride()); - } - - /** \internal Constructor for fixed size matrices or vectors */ - EIGEN_DEVICE_FUNC - explicit inline MapBase(PointerType dataPtr) : m_data(dataPtr), m_rows(RowsAtCompileTime), m_cols(ColsAtCompileTime) - { - EIGEN_STATIC_ASSERT_FIXED_SIZE(Derived) - checkSanity(); - } - - /** \internal Constructor for dynamically sized vectors */ - EIGEN_DEVICE_FUNC - inline MapBase(PointerType dataPtr, Index vecSize) - : m_data(dataPtr), - m_rows(RowsAtCompileTime == Dynamic ? vecSize : Index(RowsAtCompileTime)), - m_cols(ColsAtCompileTime == Dynamic ? vecSize : Index(ColsAtCompileTime)) - { - EIGEN_STATIC_ASSERT_VECTOR_ONLY(Derived) - eigen_assert(vecSize >= 0); - eigen_assert(dataPtr == 0 || SizeAtCompileTime == Dynamic || SizeAtCompileTime == vecSize); - checkSanity(); - } - - /** \internal Constructor for dynamically sized matrices */ - EIGEN_DEVICE_FUNC - inline MapBase(PointerType dataPtr, Index rows, Index cols) - : m_data(dataPtr), m_rows(rows), m_cols(cols) - { - eigen_assert( (dataPtr == 0) - || ( rows >= 0 && (RowsAtCompileTime == Dynamic || RowsAtCompileTime == rows) - && cols >= 0 && (ColsAtCompileTime == Dynamic || ColsAtCompileTime == cols))); - checkSanity(); - } - - #ifdef EIGEN_MAPBASE_PLUGIN - #include EIGEN_MAPBASE_PLUGIN - #endif - - protected: - - template - EIGEN_DEVICE_FUNC - void checkSanity(typename internal::enable_if<(internal::traits::Alignment>0),void*>::type = 0) const - { -#if EIGEN_MAX_ALIGN_BYTES>0 - // innerStride() is not set yet when this function is called, so we optimistically assume the lowest plausible value: - const Index minInnerStride = InnerStrideAtCompileTime == Dynamic ? 1 : Index(InnerStrideAtCompileTime); - EIGEN_ONLY_USED_FOR_DEBUG(minInnerStride); - eigen_assert(( ((internal::UIntPtr(m_data) % internal::traits::Alignment) == 0) - || (cols() * rows() * minInnerStride * sizeof(Scalar)) < internal::traits::Alignment ) && "data is not aligned"); +public: + typedef typename internal::dense_xpr_base::type Base; + enum { + RowsAtCompileTime = internal::traits::RowsAtCompileTime, + ColsAtCompileTime = internal::traits::ColsAtCompileTime, + InnerStrideAtCompileTime = internal::traits::InnerStrideAtCompileTime, + SizeAtCompileTime = Base::SizeAtCompileTime + }; + + typedef typename internal::traits::StorageKind StorageKind; + typedef typename internal::traits::Scalar Scalar; + typedef typename internal::packet_traits::type PacketScalar; + typedef typename NumTraits::Real RealScalar; + typedef typename internal::conditional::value), Scalar *, const Scalar *>::type + PointerType; + + using Base::derived; + // using Base::RowsAtCompileTime; + // using Base::ColsAtCompileTime; + // using Base::SizeAtCompileTime; + using Base::MaxRowsAtCompileTime; + using Base::MaxColsAtCompileTime; + using Base::MaxSizeAtCompileTime; + using Base::IsVectorAtCompileTime; + using Base::Flags; + using Base::IsRowMajor; + + using Base::rows; + using Base::cols; + using Base::size; + using Base::coeff; + using Base::coeffRef; + using Base::lazyAssign; + using Base::eval; + + using Base::innerStride; + using Base::outerStride; + using Base::rowStride; + using Base::colStride; + + // bug 217 - compile error on ICC 11.1 + using Base::operator=; + + typedef typename Base::CoeffReturnType CoeffReturnType; + + /** \copydoc DenseBase::rows() */ + EIGEN_DEVICE_FUNC inline Index rows() const { return m_rows.value(); } + /** \copydoc DenseBase::cols() */ + EIGEN_DEVICE_FUNC inline Index cols() const { return m_cols.value(); } + + /** Returns a pointer to the first coefficient of the matrix or vector. + * + * \note When addressing this data, make sure to honor the strides returned by innerStride() and outerStride(). + * + * \sa innerStride(), outerStride() + */ + EIGEN_DEVICE_FUNC inline const Scalar *data() const { return m_data; } + + /** \copydoc PlainObjectBase::coeff(Index,Index) const */ + EIGEN_DEVICE_FUNC + inline const Scalar &coeff(Index rowId, Index colId) const + { + return m_data[colId * colStride() + rowId * rowStride()]; + } + + /** \copydoc PlainObjectBase::coeff(Index) const */ + EIGEN_DEVICE_FUNC + inline const Scalar &coeff(Index index) const + { + EIGEN_STATIC_ASSERT_INDEX_BASED_ACCESS(Derived) + return m_data[index * innerStride()]; + } + + /** \copydoc PlainObjectBase::coeffRef(Index,Index) const */ + EIGEN_DEVICE_FUNC + inline const Scalar &coeffRef(Index rowId, Index colId) const + { + return this->m_data[colId * colStride() + rowId * rowStride()]; + } + + /** \copydoc PlainObjectBase::coeffRef(Index) const */ + EIGEN_DEVICE_FUNC + inline const Scalar &coeffRef(Index index) const + { + EIGEN_STATIC_ASSERT_INDEX_BASED_ACCESS(Derived) + return this->m_data[index * innerStride()]; + } + + /** \internal */ + template inline PacketScalar packet(Index rowId, Index colId) const + { + return internal::ploadt(m_data + (colId * colStride() + rowId * rowStride())); + } + + /** \internal */ + template inline PacketScalar packet(Index index) const + { + EIGEN_STATIC_ASSERT_INDEX_BASED_ACCESS(Derived) + return internal::ploadt(m_data + index * innerStride()); + } + + /** \internal Constructor for fixed size matrices or vectors */ + EIGEN_DEVICE_FUNC + explicit inline MapBase(PointerType dataPtr) : m_data(dataPtr), m_rows(RowsAtCompileTime), m_cols(ColsAtCompileTime) + { + EIGEN_STATIC_ASSERT_FIXED_SIZE(Derived) + checkSanity(); + } + + /** \internal Constructor for dynamically sized vectors */ + EIGEN_DEVICE_FUNC + inline MapBase(PointerType dataPtr, Index vecSize) + : m_data(dataPtr), m_rows(RowsAtCompileTime == Dynamic ? vecSize : Index(RowsAtCompileTime)), + m_cols(ColsAtCompileTime == Dynamic ? vecSize : Index(ColsAtCompileTime)) + { + EIGEN_STATIC_ASSERT_VECTOR_ONLY(Derived) + eigen_assert(vecSize >= 0); + eigen_assert(dataPtr == 0 || SizeAtCompileTime == Dynamic || SizeAtCompileTime == vecSize); + checkSanity(); + } + + /** \internal Constructor for dynamically sized matrices */ + EIGEN_DEVICE_FUNC + inline MapBase(PointerType dataPtr, Index rows, Index cols) : m_data(dataPtr), m_rows(rows), m_cols(cols) + { + eigen_assert((dataPtr == 0) + || (rows >= 0 && (RowsAtCompileTime == Dynamic || RowsAtCompileTime == rows) && cols >= 0 + && (ColsAtCompileTime == Dynamic || ColsAtCompileTime == cols))); + checkSanity(); + } + +#ifdef EIGEN_MAPBASE_PLUGIN +#include EIGEN_MAPBASE_PLUGIN #endif - } - template - EIGEN_DEVICE_FUNC - void checkSanity(typename internal::enable_if::Alignment==0,void*>::type = 0) const - {} +protected: + template + EIGEN_DEVICE_FUNC void checkSanity( + typename internal::enable_if<(internal::traits::Alignment > 0), void *>::type = 0) const + { +#if EIGEN_MAX_ALIGN_BYTES > 0 + // innerStride() is not set yet when this function is called, so we optimistically assume the lowest plausible + // value: + const Index minInnerStride = InnerStrideAtCompileTime == Dynamic ? 1 : Index(InnerStrideAtCompileTime); + EIGEN_ONLY_USED_FOR_DEBUG(minInnerStride); + eigen_assert((((internal::UIntPtr(m_data) % internal::traits::Alignment) == 0) + || (cols() * rows() * minInnerStride * sizeof(Scalar)) < internal::traits::Alignment) + && "data is not aligned"); +#endif + } + + template + EIGEN_DEVICE_FUNC void checkSanity( + typename internal::enable_if::Alignment == 0, void *>::type = 0) const + {} - PointerType m_data; - const internal::variable_if_dynamic m_rows; - const internal::variable_if_dynamic m_cols; + PointerType m_data; + const internal::variable_if_dynamic m_rows; + const internal::variable_if_dynamic m_cols; }; /** \ingroup Core_Module - * - * \brief Base class for non-const dense Map and Block expression with direct access - * - * This base class provides the non-const low-level accessors (e.g. coeff and coeffRef) of - * dense Map and Block objects with direct access. - * It inherits MapBase which defines the const variant for reading specific entries. - * - * \sa class Map, class Block - */ -template class MapBase - : public MapBase + * + * \brief Base class for non-const dense Map and Block expression with direct access + * + * This base class provides the non-const low-level accessors (e.g. coeff and coeffRef) of + * dense Map and Block objects with direct access. + * It inherits MapBase which defines the const variant for reading specific entries. + * + * \sa class Map, class Block + */ +template class MapBase : public MapBase { - typedef MapBase ReadOnlyMapBase; - public: - - typedef MapBase Base; - - typedef typename Base::Scalar Scalar; - typedef typename Base::PacketScalar PacketScalar; - typedef typename Base::StorageIndex StorageIndex; - typedef typename Base::PointerType PointerType; - - using Base::derived; - using Base::rows; - using Base::cols; - using Base::size; - using Base::coeff; - using Base::coeffRef; - - using Base::innerStride; - using Base::outerStride; - using Base::rowStride; - using Base::colStride; - - typedef typename internal::conditional< - internal::is_lvalue::value, - Scalar, - const Scalar - >::type ScalarWithConstIfNotLvalue; - - EIGEN_DEVICE_FUNC - inline const Scalar* data() const { return this->m_data; } - EIGEN_DEVICE_FUNC - inline ScalarWithConstIfNotLvalue* data() { return this->m_data; } // no const-cast here so non-const-correct code will give a compile error - - EIGEN_DEVICE_FUNC - inline ScalarWithConstIfNotLvalue& coeffRef(Index row, Index col) - { - return this->m_data[col * colStride() + row * rowStride()]; - } - - EIGEN_DEVICE_FUNC - inline ScalarWithConstIfNotLvalue& coeffRef(Index index) - { - EIGEN_STATIC_ASSERT_INDEX_BASED_ACCESS(Derived) - return this->m_data[index * innerStride()]; - } - - template - inline void writePacket(Index row, Index col, const PacketScalar& val) - { - internal::pstoret - (this->m_data + (col * colStride() + row * rowStride()), val); - } - - template - inline void writePacket(Index index, const PacketScalar& val) - { - EIGEN_STATIC_ASSERT_INDEX_BASED_ACCESS(Derived) - internal::pstoret - (this->m_data + index * innerStride(), val); - } - - EIGEN_DEVICE_FUNC explicit inline MapBase(PointerType dataPtr) : Base(dataPtr) {} - EIGEN_DEVICE_FUNC inline MapBase(PointerType dataPtr, Index vecSize) : Base(dataPtr, vecSize) {} - EIGEN_DEVICE_FUNC inline MapBase(PointerType dataPtr, Index rows, Index cols) : Base(dataPtr, rows, cols) {} - - EIGEN_DEVICE_FUNC - Derived& operator=(const MapBase& other) - { - ReadOnlyMapBase::Base::operator=(other); - return derived(); - } - - // In theory we could simply refer to Base:Base::operator=, but MSVC does not like Base::Base, - // see bugs 821 and 920. - using ReadOnlyMapBase::Base::operator=; + typedef MapBase ReadOnlyMapBase; + +public: + typedef MapBase Base; + + typedef typename Base::Scalar Scalar; + typedef typename Base::PacketScalar PacketScalar; + typedef typename Base::StorageIndex StorageIndex; + typedef typename Base::PointerType PointerType; + + using Base::derived; + using Base::rows; + using Base::cols; + using Base::size; + using Base::coeff; + using Base::coeffRef; + + using Base::innerStride; + using Base::outerStride; + using Base::rowStride; + using Base::colStride; + + typedef typename internal::conditional::value, Scalar, const Scalar>::type + ScalarWithConstIfNotLvalue; + + EIGEN_DEVICE_FUNC + inline const Scalar *data() const { return this->m_data; } + EIGEN_DEVICE_FUNC + inline ScalarWithConstIfNotLvalue *data() + { + return this->m_data; + }// no const-cast here so non-const-correct code will give a compile error + + EIGEN_DEVICE_FUNC + inline ScalarWithConstIfNotLvalue &coeffRef(Index row, Index col) + { + return this->m_data[col * colStride() + row * rowStride()]; + } + + EIGEN_DEVICE_FUNC + inline ScalarWithConstIfNotLvalue &coeffRef(Index index) + { + EIGEN_STATIC_ASSERT_INDEX_BASED_ACCESS(Derived) + return this->m_data[index * innerStride()]; + } + + template inline void writePacket(Index row, Index col, const PacketScalar &val) + { + internal::pstoret(this->m_data + (col * colStride() + row * rowStride()), val); + } + + template inline void writePacket(Index index, const PacketScalar &val) + { + EIGEN_STATIC_ASSERT_INDEX_BASED_ACCESS(Derived) + internal::pstoret(this->m_data + index * innerStride(), val); + } + + EIGEN_DEVICE_FUNC explicit inline MapBase(PointerType dataPtr) : Base(dataPtr) {} + EIGEN_DEVICE_FUNC inline MapBase(PointerType dataPtr, Index vecSize) : Base(dataPtr, vecSize) {} + EIGEN_DEVICE_FUNC inline MapBase(PointerType dataPtr, Index rows, Index cols) : Base(dataPtr, rows, cols) {} + + EIGEN_DEVICE_FUNC + Derived &operator=(const MapBase &other) + { + ReadOnlyMapBase::Base::operator=(other); + return derived(); + } + + // In theory we could simply refer to Base:Base::operator=, but MSVC does not like Base::Base, + // see bugs 821 and 920. + using ReadOnlyMapBase::Base::operator=; }; #undef EIGEN_STATIC_ASSERT_INDEX_BASED_ACCESS -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_MAPBASE_H +#endif// EIGEN_MAPBASE_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/MathFunctions.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/MathFunctions.h index b249ce0c..dc644154 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/MathFunctions.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/MathFunctions.h @@ -19,374 +19,315 @@ namespace Eigen { // On WINCE, std::abs is defined for int only, so let's defined our own overloads: // This issue has been confirmed with MSVC 2008 only, but the issue might exist for more recent versions too. -#if EIGEN_OS_WINCE && EIGEN_COMP_MSVC && EIGEN_COMP_MSVC<=1500 -long abs(long x) { return (labs(x)); } -double abs(double x) { return (fabs(x)); } -float abs(float x) { return (fabsf(x)); } +#if EIGEN_OS_WINCE && EIGEN_COMP_MSVC && EIGEN_COMP_MSVC <= 1500 +long abs(long x) { return (labs(x)); } +double abs(double x) { return (fabs(x)); } +float abs(float x) { return (fabsf(x)); } long double abs(long double x) { return (fabsl(x)); } #endif namespace internal { -/** \internal \class global_math_functions_filtering_base - * - * What it does: - * Defines a typedef 'type' as follows: - * - if type T has a member typedef Eigen_BaseClassForSpecializationOfGlobalMathFuncImpl, then - * global_math_functions_filtering_base::type is a typedef for it. - * - otherwise, global_math_functions_filtering_base::type is a typedef for T. - * - * How it's used: - * To allow to defined the global math functions (like sin...) in certain cases, like the Array expressions. - * When you do sin(array1+array2), the object array1+array2 has a complicated expression type, all what you want to know - * is that it inherits ArrayBase. So we implement a partial specialization of sin_impl for ArrayBase. - * So we must make sure to use sin_impl > and not sin_impl, otherwise our partial specialization - * won't be used. How does sin know that? That's exactly what global_math_functions_filtering_base tells it. - * - * How it's implemented: - * SFINAE in the style of enable_if. Highly susceptible of breaking compilers. With GCC, it sure does work, but if you replace - * the typename dummy by an integer template parameter, it doesn't work anymore! - */ - -template -struct global_math_functions_filtering_base -{ - typedef T type; -}; - -template struct always_void { typedef void type; }; - -template -struct global_math_functions_filtering_base - ::type - > -{ - typedef typename T::Eigen_BaseClassForSpecializationOfGlobalMathFuncImpl type; -}; - -#define EIGEN_MATHFUNC_IMPL(func, scalar) Eigen::internal::func##_impl::type> -#define EIGEN_MATHFUNC_RETVAL(func, scalar) typename Eigen::internal::func##_retval::type>::type + /** \internal \class global_math_functions_filtering_base + * + * What it does: + * Defines a typedef 'type' as follows: + * - if type T has a member typedef Eigen_BaseClassForSpecializationOfGlobalMathFuncImpl, then + * global_math_functions_filtering_base::type is a typedef for it. + * - otherwise, global_math_functions_filtering_base::type is a typedef for T. + * + * How it's used: + * To allow to defined the global math functions (like sin...) in certain cases, like the Array expressions. + * When you do sin(array1+array2), the object array1+array2 has a complicated expression type, all what you want to + * know is that it inherits ArrayBase. So we implement a partial specialization of sin_impl for ArrayBase. So + * we must make sure to use sin_impl > and not sin_impl, otherwise our partial + * specialization won't be used. How does sin know that? That's exactly what global_math_functions_filtering_base + * tells it. + * + * How it's implemented: + * SFINAE in the style of enable_if. Highly susceptible of breaking compilers. With GCC, it sure does work, but if you + * replace the typename dummy by an integer template parameter, it doesn't work anymore! + */ + + template struct global_math_functions_filtering_base + { + typedef T type; + }; -/**************************************************************************** -* Implementation of real * -****************************************************************************/ + template struct always_void + { + typedef void type; + }; -template::IsComplex> -struct real_default_impl -{ - typedef typename NumTraits::Real RealScalar; - EIGEN_DEVICE_FUNC - static inline RealScalar run(const Scalar& x) + template + struct global_math_functions_filtering_base::type> { - return x; - } -}; + typedef typename T::Eigen_BaseClassForSpecializationOfGlobalMathFuncImpl type; + }; -template -struct real_default_impl -{ - typedef typename NumTraits::Real RealScalar; - EIGEN_DEVICE_FUNC - static inline RealScalar run(const Scalar& x) +#define EIGEN_MATHFUNC_IMPL(func, scalar) \ + Eigen::internal::func##_impl::type> +#define EIGEN_MATHFUNC_RETVAL(func, scalar) \ + typename Eigen::internal::func##_retval< \ + typename Eigen::internal::global_math_functions_filtering_base::type>::type + + /**************************************************************************** + * Implementation of real * + ****************************************************************************/ + + template::IsComplex> struct real_default_impl { - using std::real; - return real(x); - } -}; + typedef typename NumTraits::Real RealScalar; + EIGEN_DEVICE_FUNC + static inline RealScalar run(const Scalar &x) { return x; } + }; + + template struct real_default_impl + { + typedef typename NumTraits::Real RealScalar; + EIGEN_DEVICE_FUNC + static inline RealScalar run(const Scalar &x) + { + using std::real; + return real(x); + } + }; -template struct real_impl : real_default_impl {}; + template struct real_impl : real_default_impl + { + }; #ifdef __CUDA_ARCH__ -template -struct real_impl > -{ - typedef T RealScalar; - EIGEN_DEVICE_FUNC - static inline T run(const std::complex& x) + template struct real_impl> { - return x.real(); - } -}; + typedef T RealScalar; + EIGEN_DEVICE_FUNC + static inline T run(const std::complex &x) { return x.real(); } + }; #endif -template -struct real_retval -{ - typedef typename NumTraits::Real type; -}; + template struct real_retval + { + typedef typename NumTraits::Real type; + }; -/**************************************************************************** -* Implementation of imag * -****************************************************************************/ + /**************************************************************************** + * Implementation of imag * + ****************************************************************************/ -template::IsComplex> -struct imag_default_impl -{ - typedef typename NumTraits::Real RealScalar; - EIGEN_DEVICE_FUNC - static inline RealScalar run(const Scalar&) + template::IsComplex> struct imag_default_impl { - return RealScalar(0); - } -}; + typedef typename NumTraits::Real RealScalar; + EIGEN_DEVICE_FUNC + static inline RealScalar run(const Scalar &) { return RealScalar(0); } + }; -template -struct imag_default_impl -{ - typedef typename NumTraits::Real RealScalar; - EIGEN_DEVICE_FUNC - static inline RealScalar run(const Scalar& x) + template struct imag_default_impl { - using std::imag; - return imag(x); - } -}; + typedef typename NumTraits::Real RealScalar; + EIGEN_DEVICE_FUNC + static inline RealScalar run(const Scalar &x) + { + using std::imag; + return imag(x); + } + }; -template struct imag_impl : imag_default_impl {}; + template struct imag_impl : imag_default_impl + { + }; #ifdef __CUDA_ARCH__ -template -struct imag_impl > -{ - typedef T RealScalar; - EIGEN_DEVICE_FUNC - static inline T run(const std::complex& x) + template struct imag_impl> { - return x.imag(); - } -}; + typedef T RealScalar; + EIGEN_DEVICE_FUNC + static inline T run(const std::complex &x) { return x.imag(); } + }; #endif -template -struct imag_retval -{ - typedef typename NumTraits::Real type; -}; + template struct imag_retval + { + typedef typename NumTraits::Real type; + }; -/**************************************************************************** -* Implementation of real_ref * -****************************************************************************/ + /**************************************************************************** + * Implementation of real_ref * + ****************************************************************************/ -template -struct real_ref_impl -{ - typedef typename NumTraits::Real RealScalar; - EIGEN_DEVICE_FUNC - static inline RealScalar& run(Scalar& x) + template struct real_ref_impl { - return reinterpret_cast(&x)[0]; - } - EIGEN_DEVICE_FUNC - static inline const RealScalar& run(const Scalar& x) - { - return reinterpret_cast(&x)[0]; - } -}; + typedef typename NumTraits::Real RealScalar; + EIGEN_DEVICE_FUNC + static inline RealScalar &run(Scalar &x) { return reinterpret_cast(&x)[0]; } + EIGEN_DEVICE_FUNC + static inline const RealScalar &run(const Scalar &x) { return reinterpret_cast(&x)[0]; } + }; -template -struct real_ref_retval -{ - typedef typename NumTraits::Real & type; -}; + template struct real_ref_retval + { + typedef typename NumTraits::Real &type; + }; -/**************************************************************************** -* Implementation of imag_ref * -****************************************************************************/ + /**************************************************************************** + * Implementation of imag_ref * + ****************************************************************************/ -template -struct imag_ref_default_impl -{ - typedef typename NumTraits::Real RealScalar; - EIGEN_DEVICE_FUNC - static inline RealScalar& run(Scalar& x) + template struct imag_ref_default_impl { - return reinterpret_cast(&x)[1]; - } - EIGEN_DEVICE_FUNC - static inline const RealScalar& run(const Scalar& x) - { - return reinterpret_cast(&x)[1]; - } -}; + typedef typename NumTraits::Real RealScalar; + EIGEN_DEVICE_FUNC + static inline RealScalar &run(Scalar &x) { return reinterpret_cast(&x)[1]; } + EIGEN_DEVICE_FUNC + static inline const RealScalar &run(const Scalar &x) { return reinterpret_cast(&x)[1]; } + }; -template -struct imag_ref_default_impl -{ - EIGEN_DEVICE_FUNC - static inline Scalar run(Scalar&) - { - return Scalar(0); - } - EIGEN_DEVICE_FUNC - static inline const Scalar run(const Scalar&) + template struct imag_ref_default_impl { - return Scalar(0); - } -}; + EIGEN_DEVICE_FUNC + static inline Scalar run(Scalar &) { return Scalar(0); } + EIGEN_DEVICE_FUNC + static inline const Scalar run(const Scalar &) { return Scalar(0); } + }; -template -struct imag_ref_impl : imag_ref_default_impl::IsComplex> {}; + template struct imag_ref_impl : imag_ref_default_impl::IsComplex> + { + }; -template -struct imag_ref_retval -{ - typedef typename NumTraits::Real & type; -}; + template struct imag_ref_retval + { + typedef typename NumTraits::Real &type; + }; -/**************************************************************************** -* Implementation of conj * -****************************************************************************/ + /**************************************************************************** + * Implementation of conj * + ****************************************************************************/ -template::IsComplex> -struct conj_impl -{ - EIGEN_DEVICE_FUNC - static inline Scalar run(const Scalar& x) + template::IsComplex> struct conj_impl { - return x; - } -}; + EIGEN_DEVICE_FUNC + static inline Scalar run(const Scalar &x) { return x; } + }; -template -struct conj_impl -{ - EIGEN_DEVICE_FUNC - static inline Scalar run(const Scalar& x) + template struct conj_impl { - using std::conj; - return conj(x); - } -}; + EIGEN_DEVICE_FUNC + static inline Scalar run(const Scalar &x) + { + using std::conj; + return conj(x); + } + }; -template -struct conj_retval -{ - typedef Scalar type; -}; + template struct conj_retval + { + typedef Scalar type; + }; -/**************************************************************************** -* Implementation of abs2 * -****************************************************************************/ + /**************************************************************************** + * Implementation of abs2 * + ****************************************************************************/ -template -struct abs2_impl_default -{ - typedef typename NumTraits::Real RealScalar; - EIGEN_DEVICE_FUNC - static inline RealScalar run(const Scalar& x) + template struct abs2_impl_default { - return x*x; - } -}; + typedef typename NumTraits::Real RealScalar; + EIGEN_DEVICE_FUNC + static inline RealScalar run(const Scalar &x) { return x * x; } + }; -template -struct abs2_impl_default // IsComplex -{ - typedef typename NumTraits::Real RealScalar; - EIGEN_DEVICE_FUNC - static inline RealScalar run(const Scalar& x) + template struct abs2_impl_default// IsComplex { - return real(x)*real(x) + imag(x)*imag(x); - } -}; + typedef typename NumTraits::Real RealScalar; + EIGEN_DEVICE_FUNC + static inline RealScalar run(const Scalar &x) { return real(x) * real(x) + imag(x) * imag(x); } + }; -template -struct abs2_impl -{ - typedef typename NumTraits::Real RealScalar; - EIGEN_DEVICE_FUNC - static inline RealScalar run(const Scalar& x) + template struct abs2_impl { - return abs2_impl_default::IsComplex>::run(x); - } -}; + typedef typename NumTraits::Real RealScalar; + EIGEN_DEVICE_FUNC + static inline RealScalar run(const Scalar &x) + { + return abs2_impl_default::IsComplex>::run(x); + } + }; -template -struct abs2_retval -{ - typedef typename NumTraits::Real type; -}; + template struct abs2_retval + { + typedef typename NumTraits::Real type; + }; -/**************************************************************************** -* Implementation of norm1 * -****************************************************************************/ + /**************************************************************************** + * Implementation of norm1 * + ****************************************************************************/ -template -struct norm1_default_impl -{ - typedef typename NumTraits::Real RealScalar; - EIGEN_DEVICE_FUNC - static inline RealScalar run(const Scalar& x) + template struct norm1_default_impl { - EIGEN_USING_STD_MATH(abs); - return abs(real(x)) + abs(imag(x)); - } -}; + typedef typename NumTraits::Real RealScalar; + EIGEN_DEVICE_FUNC + static inline RealScalar run(const Scalar &x) + { + EIGEN_USING_STD_MATH(abs); + return abs(real(x)) + abs(imag(x)); + } + }; -template -struct norm1_default_impl -{ - EIGEN_DEVICE_FUNC - static inline Scalar run(const Scalar& x) + template struct norm1_default_impl { - EIGEN_USING_STD_MATH(abs); - return abs(x); - } -}; + EIGEN_DEVICE_FUNC + static inline Scalar run(const Scalar &x) + { + EIGEN_USING_STD_MATH(abs); + return abs(x); + } + }; -template -struct norm1_impl : norm1_default_impl::IsComplex> {}; + template struct norm1_impl : norm1_default_impl::IsComplex> + { + }; -template -struct norm1_retval -{ - typedef typename NumTraits::Real type; -}; + template struct norm1_retval + { + typedef typename NumTraits::Real type; + }; -/**************************************************************************** -* Implementation of hypot * -****************************************************************************/ + /**************************************************************************** + * Implementation of hypot * + ****************************************************************************/ -template struct hypot_impl; + template struct hypot_impl; -template -struct hypot_retval -{ - typedef typename NumTraits::Real type; -}; + template struct hypot_retval + { + typedef typename NumTraits::Real type; + }; -/**************************************************************************** -* Implementation of cast * -****************************************************************************/ + /**************************************************************************** + * Implementation of cast * + ****************************************************************************/ -template -struct cast_impl -{ - EIGEN_DEVICE_FUNC - static inline NewType run(const OldType& x) + template struct cast_impl { - return static_cast(x); - } -}; + EIGEN_DEVICE_FUNC + static inline NewType run(const OldType &x) { return static_cast(x); } + }; -// here, for once, we're plainly returning NewType: we don't want cast to do weird things. + // here, for once, we're plainly returning NewType: we don't want cast to do weird things. -template -EIGEN_DEVICE_FUNC -inline NewType cast(const OldType& x) -{ - return cast_impl::run(x); -} + template EIGEN_DEVICE_FUNC inline NewType cast(const OldType &x) + { + return cast_impl::run(x); + } -/**************************************************************************** -* Implementation of round * -****************************************************************************/ + /**************************************************************************** + * Implementation of round * + ****************************************************************************/ #if EIGEN_HAS_CXX11_MATH - template - struct round_impl { - static inline Scalar run(const Scalar& x) + template struct round_impl + { + static inline Scalar run(const Scalar &x) { EIGEN_STATIC_ASSERT((!NumTraits::IsComplex), NUMERIC_TYPE_MUST_BE_REAL) using std::round; @@ -394,10 +335,9 @@ inline NewType cast(const OldType& x) } }; #else - template - struct round_impl + template struct round_impl { - static inline Scalar run(const Scalar& x) + static inline Scalar run(const Scalar &x) { EIGEN_STATIC_ASSERT((!NumTraits::IsComplex), NUMERIC_TYPE_MUST_BE_REAL) EIGEN_USING_STD_MATH(floor); @@ -407,386 +347,374 @@ inline NewType cast(const OldType& x) }; #endif -template -struct round_retval -{ - typedef Scalar type; -}; + template struct round_retval + { + typedef Scalar type; + }; -/**************************************************************************** -* Implementation of arg * -****************************************************************************/ + /**************************************************************************** + * Implementation of arg * + ****************************************************************************/ #if EIGEN_HAS_CXX11_MATH - template - struct arg_impl { - static inline Scalar run(const Scalar& x) + template struct arg_impl + { + static inline Scalar run(const Scalar &x) { EIGEN_USING_STD_MATH(arg); return arg(x); } }; #else - template::IsComplex> - struct arg_default_impl + template::IsComplex> struct arg_default_impl { typedef typename NumTraits::Real RealScalar; EIGEN_DEVICE_FUNC - static inline RealScalar run(const Scalar& x) - { - return (x < Scalar(0)) ? Scalar(EIGEN_PI) : Scalar(0); } + static inline RealScalar run(const Scalar &x) { return (x < Scalar(0)) ? Scalar(EIGEN_PI) : Scalar(0); } }; - template - struct arg_default_impl + template struct arg_default_impl { typedef typename NumTraits::Real RealScalar; EIGEN_DEVICE_FUNC - static inline RealScalar run(const Scalar& x) + static inline RealScalar run(const Scalar &x) { EIGEN_USING_STD_MATH(arg); return arg(x); } }; - template struct arg_impl : arg_default_impl {}; + template struct arg_impl : arg_default_impl + { + }; #endif -template -struct arg_retval -{ - typedef typename NumTraits::Real type; -}; + template struct arg_retval + { + typedef typename NumTraits::Real type; + }; -/**************************************************************************** -* Implementation of log1p * -****************************************************************************/ + /**************************************************************************** + * Implementation of log1p * + ****************************************************************************/ -namespace std_fallback { - // fallback log1p implementation in case there is no log1p(Scalar) function in namespace of Scalar, - // or that there is no suitable std::log1p function available - template - EIGEN_DEVICE_FUNC inline Scalar log1p(const Scalar& x) { - EIGEN_STATIC_ASSERT_NON_INTEGER(Scalar) - typedef typename NumTraits::Real RealScalar; - EIGEN_USING_STD_MATH(log); - Scalar x1p = RealScalar(1) + x; - return numext::equal_strict(x1p, Scalar(1)) ? x : x * ( log(x1p) / (x1p - RealScalar(1)) ); - } -} - -template -struct log1p_impl { - static inline Scalar run(const Scalar& x) - { - EIGEN_STATIC_ASSERT_NON_INTEGER(Scalar) - #if EIGEN_HAS_CXX11_MATH - using std::log1p; - #endif - using std_fallback::log1p; - return log1p(x); - } -}; + namespace std_fallback { + // fallback log1p implementation in case there is no log1p(Scalar) function in namespace of Scalar, + // or that there is no suitable std::log1p function available + template EIGEN_DEVICE_FUNC inline Scalar log1p(const Scalar &x) + { + EIGEN_STATIC_ASSERT_NON_INTEGER(Scalar) + typedef typename NumTraits::Real RealScalar; + EIGEN_USING_STD_MATH(log); + Scalar x1p = RealScalar(1) + x; + return numext::equal_strict(x1p, Scalar(1)) ? x : x * (log(x1p) / (x1p - RealScalar(1))); + } + }// namespace std_fallback + template struct log1p_impl + { + static inline Scalar run(const Scalar &x) + { + EIGEN_STATIC_ASSERT_NON_INTEGER(Scalar) +#if EIGEN_HAS_CXX11_MATH + using std::log1p; +#endif + using std_fallback::log1p; + return log1p(x); + } + }; -template -struct log1p_retval -{ - typedef Scalar type; -}; -/**************************************************************************** -* Implementation of pow * -****************************************************************************/ - -template::IsInteger&&NumTraits::IsInteger> -struct pow_impl -{ - //typedef Scalar retval; - typedef typename ScalarBinaryOpTraits >::ReturnType result_type; - static EIGEN_DEVICE_FUNC inline result_type run(const ScalarX& x, const ScalarY& y) - { - EIGEN_USING_STD_MATH(pow); - return pow(x, y); - } -}; - -template -struct pow_impl -{ - typedef ScalarX result_type; - static EIGEN_DEVICE_FUNC inline ScalarX run(ScalarX x, ScalarY y) - { - ScalarX res(1); - eigen_assert(!NumTraits::IsSigned || y >= 0); - if(y & 1) res *= x; - y >>= 1; - while(y) + template struct log1p_retval + { + typedef Scalar type; + }; + + /**************************************************************************** + * Implementation of pow * + ****************************************************************************/ + + template::IsInteger && NumTraits::IsInteger> + struct pow_impl + { + // typedef Scalar retval; + typedef typename ScalarBinaryOpTraits>::ReturnType + result_type; + static EIGEN_DEVICE_FUNC inline result_type run(const ScalarX &x, const ScalarY &y) + { + EIGEN_USING_STD_MATH(pow); + return pow(x, y); + } + }; + + template struct pow_impl + { + typedef ScalarX result_type; + static EIGEN_DEVICE_FUNC inline ScalarX run(ScalarX x, ScalarY y) { - x *= x; - if(y&1) res *= x; + ScalarX res(1); + eigen_assert(!NumTraits::IsSigned || y >= 0); + if (y & 1) res *= x; y >>= 1; + while (y) { + x *= x; + if (y & 1) res *= x; + y >>= 1; + } + return res; } - return res; - } -}; + }; -/**************************************************************************** -* Implementation of random * -****************************************************************************/ + /**************************************************************************** + * Implementation of random * + ****************************************************************************/ -template -struct random_default_impl {}; + template struct random_default_impl + { + }; -template -struct random_impl : random_default_impl::IsComplex, NumTraits::IsInteger> {}; + template + struct random_impl : random_default_impl::IsComplex, NumTraits::IsInteger> + { + }; -template -struct random_retval -{ - typedef Scalar type; -}; + template struct random_retval + { + typedef Scalar type; + }; -template inline EIGEN_MATHFUNC_RETVAL(random, Scalar) random(const Scalar& x, const Scalar& y); -template inline EIGEN_MATHFUNC_RETVAL(random, Scalar) random(); + template inline EIGEN_MATHFUNC_RETVAL(random, Scalar) random(const Scalar &x, const Scalar &y); + template inline EIGEN_MATHFUNC_RETVAL(random, Scalar) random(); -template -struct random_default_impl -{ - static inline Scalar run(const Scalar& x, const Scalar& y) + template struct random_default_impl { - return x + (y-x) * Scalar(std::rand()) / Scalar(RAND_MAX); - } - static inline Scalar run() + static inline Scalar run(const Scalar &x, const Scalar &y) + { + return x + (y - x) * Scalar(std::rand()) / Scalar(RAND_MAX); + } + static inline Scalar run() { return run(Scalar(NumTraits::IsSigned ? -1 : 0), Scalar(1)); } + }; + + enum { meta_floor_log2_terminate, meta_floor_log2_move_up, meta_floor_log2_move_down, meta_floor_log2_bogus }; + + template struct meta_floor_log2_selector { - return run(Scalar(NumTraits::IsSigned ? -1 : 0), Scalar(1)); - } -}; - -enum { - meta_floor_log2_terminate, - meta_floor_log2_move_up, - meta_floor_log2_move_down, - meta_floor_log2_bogus -}; - -template struct meta_floor_log2_selector -{ - enum { middle = (lower + upper) / 2, - value = (upper <= lower + 1) ? int(meta_floor_log2_terminate) - : (n < (1 << middle)) ? int(meta_floor_log2_move_down) - : (n==0) ? int(meta_floor_log2_bogus) - : int(meta_floor_log2_move_up) + enum { + middle = (lower + upper) / 2, + value = (upper <= lower + 1) ? int(meta_floor_log2_terminate) + : (n < (1 << middle)) ? int(meta_floor_log2_move_down) + : (n == 0) ? int(meta_floor_log2_bogus) + : int(meta_floor_log2_move_up) + }; + }; + + template::value> + struct meta_floor_log2 + { + }; + + template struct meta_floor_log2 + { + enum { value = meta_floor_log2::middle>::value }; + }; + + template struct meta_floor_log2 + { + enum { value = meta_floor_log2::middle, upper>::value }; + }; + + template struct meta_floor_log2 + { + enum { value = (n >= ((unsigned int)(1) << (lower + 1))) ? lower + 1 : lower }; + }; + + template struct meta_floor_log2 + { + // no value, error at compile time }; -}; - -template::value> -struct meta_floor_log2 {}; - -template -struct meta_floor_log2 -{ - enum { value = meta_floor_log2::middle>::value }; -}; - -template -struct meta_floor_log2 -{ - enum { value = meta_floor_log2::middle, upper>::value }; -}; - -template -struct meta_floor_log2 -{ - enum { value = (n >= ((unsigned int)(1) << (lower+1))) ? lower+1 : lower }; -}; - -template -struct meta_floor_log2 -{ - // no value, error at compile time -}; - -template -struct random_default_impl -{ - static inline Scalar run(const Scalar& x, const Scalar& y) - { - if (y <= x) - return x; - // ScalarU is the unsigned counterpart of Scalar, possibly Scalar itself. - typedef typename make_unsigned::type ScalarU; - // ScalarX is the widest of ScalarU and unsigned int. - // We'll deal only with ScalarX and unsigned int below thus avoiding signed - // types and arithmetic and signed overflows (which are undefined behavior). - typedef typename conditional<(ScalarU(-1) > unsigned(-1)), ScalarU, unsigned>::type ScalarX; - // The following difference doesn't overflow, provided our integer types are two's - // complement and have the same number of padding bits in signed and unsigned variants. - // This is the case in most modern implementations of C++. - ScalarX range = ScalarX(y) - ScalarX(x); - ScalarX offset = 0; - ScalarX divisor = 1; - ScalarX multiplier = 1; - const unsigned rand_max = RAND_MAX; - if (range <= rand_max) divisor = (rand_max + 1) / (range + 1); - else multiplier = 1 + range / (rand_max + 1); - // Rejection sampling. - do { - offset = (unsigned(std::rand()) * multiplier) / divisor; - } while (offset > range); - return Scalar(ScalarX(x) + offset); - } - static inline Scalar run() + template struct random_default_impl { + static inline Scalar run(const Scalar &x, const Scalar &y) + { + if (y <= x) return x; + // ScalarU is the unsigned counterpart of Scalar, possibly Scalar itself. + typedef typename make_unsigned::type ScalarU; + // ScalarX is the widest of ScalarU and unsigned int. + // We'll deal only with ScalarX and unsigned int below thus avoiding signed + // types and arithmetic and signed overflows (which are undefined behavior). + typedef typename conditional<(ScalarU(-1) > unsigned(-1)), ScalarU, unsigned>::type ScalarX; + // The following difference doesn't overflow, provided our integer types are two's + // complement and have the same number of padding bits in signed and unsigned variants. + // This is the case in most modern implementations of C++. + ScalarX range = ScalarX(y) - ScalarX(x); + ScalarX offset = 0; + ScalarX divisor = 1; + ScalarX multiplier = 1; + const unsigned rand_max = RAND_MAX; + if (range <= rand_max) + divisor = (rand_max + 1) / (range + 1); + else + multiplier = 1 + range / (rand_max + 1); + // Rejection sampling. + do { + offset = (unsigned(std::rand()) * multiplier) / divisor; + } while (offset > range); + return Scalar(ScalarX(x) + offset); + } + + static inline Scalar run() + { #ifdef EIGEN_MAKING_DOCS - return run(Scalar(NumTraits::IsSigned ? -10 : 0), Scalar(10)); + return run(Scalar(NumTraits::IsSigned ? -10 : 0), Scalar(10)); #else - enum { rand_bits = meta_floor_log2<(unsigned int)(RAND_MAX)+1>::value, - scalar_bits = sizeof(Scalar) * CHAR_BIT, - shift = EIGEN_PLAIN_ENUM_MAX(0, int(rand_bits) - int(scalar_bits)), - offset = NumTraits::IsSigned ? (1 << (EIGEN_PLAIN_ENUM_MIN(rand_bits,scalar_bits)-1)) : 0 - }; - return Scalar((std::rand() >> shift) - offset); + enum { + rand_bits = meta_floor_log2<(unsigned int)(RAND_MAX) + 1>::value, + scalar_bits = sizeof(Scalar) * CHAR_BIT, + shift = EIGEN_PLAIN_ENUM_MAX(0, int(rand_bits) - int(scalar_bits)), + offset = NumTraits::IsSigned ? (1 << (EIGEN_PLAIN_ENUM_MIN(rand_bits, scalar_bits) - 1)) : 0 + }; + return Scalar((std::rand() >> shift) - offset); #endif - } -}; + } + }; -template -struct random_default_impl -{ - static inline Scalar run(const Scalar& x, const Scalar& y) + template struct random_default_impl { - return Scalar(random(real(x), real(y)), - random(imag(x), imag(y))); - } - static inline Scalar run() + static inline Scalar run(const Scalar &x, const Scalar &y) + { + return Scalar(random(real(x), real(y)), random(imag(x), imag(y))); + } + static inline Scalar run() + { + typedef typename NumTraits::Real RealScalar; + return Scalar(random(), random()); + } + }; + + template inline EIGEN_MATHFUNC_RETVAL(random, Scalar) random(const Scalar &x, const Scalar &y) { - typedef typename NumTraits::Real RealScalar; - return Scalar(random(), random()); + return EIGEN_MATHFUNC_IMPL(random, Scalar)::run(x, y); } -}; - -template -inline EIGEN_MATHFUNC_RETVAL(random, Scalar) random(const Scalar& x, const Scalar& y) -{ - return EIGEN_MATHFUNC_IMPL(random, Scalar)::run(x, y); -} -template -inline EIGEN_MATHFUNC_RETVAL(random, Scalar) random() -{ - return EIGEN_MATHFUNC_IMPL(random, Scalar)::run(); -} + template inline EIGEN_MATHFUNC_RETVAL(random, Scalar) random() + { + return EIGEN_MATHFUNC_IMPL(random, Scalar)::run(); + } // Implementatin of is* functions // std::is* do not work with fast-math and gcc, std::is* are available on MSVC 2013 and newer, as well as in clang. -#if (EIGEN_HAS_CXX11_MATH && !(EIGEN_COMP_GNUC_STRICT && __FINITE_MATH_ONLY__)) || (EIGEN_COMP_MSVC>=1800) || (EIGEN_COMP_CLANG) +#if (EIGEN_HAS_CXX11_MATH && !(EIGEN_COMP_GNUC_STRICT && __FINITE_MATH_ONLY__)) || (EIGEN_COMP_MSVC >= 1800) \ + || (EIGEN_COMP_CLANG) #define EIGEN_USE_STD_FPCLASSIFY 1 #else #define EIGEN_USE_STD_FPCLASSIFY 0 #endif -template -EIGEN_DEVICE_FUNC -typename internal::enable_if::value,bool>::type -isnan_impl(const T&) { return false; } - -template -EIGEN_DEVICE_FUNC -typename internal::enable_if::value,bool>::type -isinf_impl(const T&) { return false; } - -template -EIGEN_DEVICE_FUNC -typename internal::enable_if::value,bool>::type -isfinite_impl(const T&) { return true; } - -template -EIGEN_DEVICE_FUNC -typename internal::enable_if<(!internal::is_integral::value)&&(!NumTraits::IsComplex),bool>::type -isfinite_impl(const T& x) -{ - #ifdef __CUDA_ARCH__ + template + EIGEN_DEVICE_FUNC typename internal::enable_if::value, bool>::type isnan_impl(const T &) + { + return false; + } + + template + EIGEN_DEVICE_FUNC typename internal::enable_if::value, bool>::type isinf_impl(const T &) + { + return false; + } + + template + EIGEN_DEVICE_FUNC typename internal::enable_if::value, bool>::type isfinite_impl(const T &) + { + return true; + } + + template + EIGEN_DEVICE_FUNC + typename internal::enable_if<(!internal::is_integral::value) && (!NumTraits::IsComplex), bool>::type + isfinite_impl(const T &x) + { +#ifdef __CUDA_ARCH__ return (::isfinite)(x); - #elif EIGEN_USE_STD_FPCLASSIFY +#elif EIGEN_USE_STD_FPCLASSIFY using std::isfinite; - return isfinite EIGEN_NOT_A_MACRO (x); - #else - return x<=NumTraits::highest() && x>=NumTraits::lowest(); - #endif -} - -template -EIGEN_DEVICE_FUNC -typename internal::enable_if<(!internal::is_integral::value)&&(!NumTraits::IsComplex),bool>::type -isinf_impl(const T& x) -{ - #ifdef __CUDA_ARCH__ + return isfinite EIGEN_NOT_A_MACRO(x); +#else + return x <= NumTraits::highest() && x >= NumTraits::lowest(); +#endif + } + + template + EIGEN_DEVICE_FUNC + typename internal::enable_if<(!internal::is_integral::value) && (!NumTraits::IsComplex), bool>::type + isinf_impl(const T &x) + { +#ifdef __CUDA_ARCH__ return (::isinf)(x); - #elif EIGEN_USE_STD_FPCLASSIFY +#elif EIGEN_USE_STD_FPCLASSIFY using std::isinf; - return isinf EIGEN_NOT_A_MACRO (x); - #else - return x>NumTraits::highest() || x::lowest(); - #endif -} - -template -EIGEN_DEVICE_FUNC -typename internal::enable_if<(!internal::is_integral::value)&&(!NumTraits::IsComplex),bool>::type -isnan_impl(const T& x) -{ - #ifdef __CUDA_ARCH__ + return isinf EIGEN_NOT_A_MACRO(x); +#else + return x > NumTraits::highest() || x < NumTraits::lowest(); +#endif + } + + template + EIGEN_DEVICE_FUNC + typename internal::enable_if<(!internal::is_integral::value) && (!NumTraits::IsComplex), bool>::type + isnan_impl(const T &x) + { +#ifdef __CUDA_ARCH__ return (::isnan)(x); - #elif EIGEN_USE_STD_FPCLASSIFY +#elif EIGEN_USE_STD_FPCLASSIFY using std::isnan; - return isnan EIGEN_NOT_A_MACRO (x); - #else + return isnan EIGEN_NOT_A_MACRO(x); +#else return x != x; - #endif -} +#endif + } #if (!EIGEN_USE_STD_FPCLASSIFY) #if EIGEN_COMP_MSVC -template EIGEN_DEVICE_FUNC bool isinf_msvc_helper(T x) -{ - return _fpclass(x)==_FPCLASS_NINF || _fpclass(x)==_FPCLASS_PINF; -} + template EIGEN_DEVICE_FUNC bool isinf_msvc_helper(T x) + { + return _fpclass(x) == _FPCLASS_NINF || _fpclass(x) == _FPCLASS_PINF; + } -//MSVC defines a _isnan builtin function, but for double only -EIGEN_DEVICE_FUNC inline bool isnan_impl(const long double& x) { return _isnan(x)!=0; } -EIGEN_DEVICE_FUNC inline bool isnan_impl(const double& x) { return _isnan(x)!=0; } -EIGEN_DEVICE_FUNC inline bool isnan_impl(const float& x) { return _isnan(x)!=0; } + // MSVC defines a _isnan builtin function, but for double only + EIGEN_DEVICE_FUNC inline bool isnan_impl(const long double &x) { return _isnan(x) != 0; } + EIGEN_DEVICE_FUNC inline bool isnan_impl(const double &x) { return _isnan(x) != 0; } + EIGEN_DEVICE_FUNC inline bool isnan_impl(const float &x) { return _isnan(x) != 0; } -EIGEN_DEVICE_FUNC inline bool isinf_impl(const long double& x) { return isinf_msvc_helper(x); } -EIGEN_DEVICE_FUNC inline bool isinf_impl(const double& x) { return isinf_msvc_helper(x); } -EIGEN_DEVICE_FUNC inline bool isinf_impl(const float& x) { return isinf_msvc_helper(x); } + EIGEN_DEVICE_FUNC inline bool isinf_impl(const long double &x) { return isinf_msvc_helper(x); } + EIGEN_DEVICE_FUNC inline bool isinf_impl(const double &x) { return isinf_msvc_helper(x); } + EIGEN_DEVICE_FUNC inline bool isinf_impl(const float &x) { return isinf_msvc_helper(x); } #elif (defined __FINITE_MATH_ONLY__ && __FINITE_MATH_ONLY__ && EIGEN_COMP_GNUC) -#if EIGEN_GNUC_AT_LEAST(5,0) - #define EIGEN_TMP_NOOPT_ATTRIB EIGEN_DEVICE_FUNC inline __attribute__((optimize("no-finite-math-only"))) +#if EIGEN_GNUC_AT_LEAST(5, 0) +#define EIGEN_TMP_NOOPT_ATTRIB EIGEN_DEVICE_FUNC inline __attribute__((optimize("no-finite-math-only"))) #else - // NOTE the inline qualifier and noinline attribute are both needed: the former is to avoid linking issue (duplicate symbol), - // while the second prevent too aggressive optimizations in fast-math mode: - #define EIGEN_TMP_NOOPT_ATTRIB EIGEN_DEVICE_FUNC inline __attribute__((noinline,optimize("no-finite-math-only"))) +// NOTE the inline qualifier and noinline attribute are both needed: the former is to avoid linking issue (duplicate +// symbol), +// while the second prevent too aggressive optimizations in fast-math mode: +#define EIGEN_TMP_NOOPT_ATTRIB EIGEN_DEVICE_FUNC inline __attribute__((noinline, optimize("no-finite-math-only"))) #endif -template<> EIGEN_TMP_NOOPT_ATTRIB bool isnan_impl(const long double& x) { return __builtin_isnan(x); } -template<> EIGEN_TMP_NOOPT_ATTRIB bool isnan_impl(const double& x) { return __builtin_isnan(x); } -template<> EIGEN_TMP_NOOPT_ATTRIB bool isnan_impl(const float& x) { return __builtin_isnan(x); } -template<> EIGEN_TMP_NOOPT_ATTRIB bool isinf_impl(const double& x) { return __builtin_isinf(x); } -template<> EIGEN_TMP_NOOPT_ATTRIB bool isinf_impl(const float& x) { return __builtin_isinf(x); } -template<> EIGEN_TMP_NOOPT_ATTRIB bool isinf_impl(const long double& x) { return __builtin_isinf(x); } + template<> EIGEN_TMP_NOOPT_ATTRIB bool isnan_impl(const long double &x) { return __builtin_isnan(x); } + template<> EIGEN_TMP_NOOPT_ATTRIB bool isnan_impl(const double &x) { return __builtin_isnan(x); } + template<> EIGEN_TMP_NOOPT_ATTRIB bool isnan_impl(const float &x) { return __builtin_isnan(x); } + template<> EIGEN_TMP_NOOPT_ATTRIB bool isinf_impl(const double &x) { return __builtin_isinf(x); } + template<> EIGEN_TMP_NOOPT_ATTRIB bool isinf_impl(const float &x) { return __builtin_isinf(x); } + template<> EIGEN_TMP_NOOPT_ATTRIB bool isinf_impl(const long double &x) { return __builtin_isinf(x); } #undef EIGEN_TMP_NOOPT_ATTRIB @@ -794,622 +722,542 @@ template<> EIGEN_TMP_NOOPT_ATTRIB bool isinf_impl(const long double& x) { return #endif -// The following overload are defined at the end of this file -template EIGEN_DEVICE_FUNC bool isfinite_impl(const std::complex& x); -template EIGEN_DEVICE_FUNC bool isnan_impl(const std::complex& x); -template EIGEN_DEVICE_FUNC bool isinf_impl(const std::complex& x); + // The following overload are defined at the end of this file + template EIGEN_DEVICE_FUNC bool isfinite_impl(const std::complex &x); + template EIGEN_DEVICE_FUNC bool isnan_impl(const std::complex &x); + template EIGEN_DEVICE_FUNC bool isinf_impl(const std::complex &x); -template T generic_fast_tanh_float(const T& a_x); + template T generic_fast_tanh_float(const T &a_x); -} // end namespace internal +}// end namespace internal /**************************************************************************** -* Generic math functions * -****************************************************************************/ + * Generic math functions * + ****************************************************************************/ namespace numext { #ifndef __CUDA_ARCH__ -template -EIGEN_DEVICE_FUNC -EIGEN_ALWAYS_INLINE T mini(const T& x, const T& y) -{ - EIGEN_USING_STD_MATH(min); - return min EIGEN_NOT_A_MACRO (x,y); -} - -template -EIGEN_DEVICE_FUNC -EIGEN_ALWAYS_INLINE T maxi(const T& x, const T& y) -{ - EIGEN_USING_STD_MATH(max); - return max EIGEN_NOT_A_MACRO (x,y); -} + template EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE T mini(const T &x, const T &y) + { + EIGEN_USING_STD_MATH(min); + return min EIGEN_NOT_A_MACRO(x, y); + } + + template EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE T maxi(const T &x, const T &y) + { + EIGEN_USING_STD_MATH(max); + return max EIGEN_NOT_A_MACRO(x, y); + } #else -template -EIGEN_DEVICE_FUNC -EIGEN_ALWAYS_INLINE T mini(const T& x, const T& y) -{ - return y < x ? y : x; -} -template<> -EIGEN_DEVICE_FUNC -EIGEN_ALWAYS_INLINE float mini(const float& x, const float& y) -{ - return fminf(x, y); -} -template -EIGEN_DEVICE_FUNC -EIGEN_ALWAYS_INLINE T maxi(const T& x, const T& y) -{ - return x < y ? y : x; -} -template<> -EIGEN_DEVICE_FUNC -EIGEN_ALWAYS_INLINE float maxi(const float& x, const float& y) -{ - return fmaxf(x, y); -} + template EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE T mini(const T &x, const T &y) { return y < x ? y : x; } + template<> EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE float mini(const float &x, const float &y) { return fminf(x, y); } + template EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE T maxi(const T &x, const T &y) { return x < y ? y : x; } + template<> EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE float maxi(const float &x, const float &y) { return fmaxf(x, y); } #endif -template -EIGEN_DEVICE_FUNC -inline EIGEN_MATHFUNC_RETVAL(real, Scalar) real(const Scalar& x) -{ - return EIGEN_MATHFUNC_IMPL(real, Scalar)::run(x); -} - -template -EIGEN_DEVICE_FUNC -inline typename internal::add_const_on_value_type< EIGEN_MATHFUNC_RETVAL(real_ref, Scalar) >::type real_ref(const Scalar& x) -{ - return internal::real_ref_impl::run(x); -} - -template -EIGEN_DEVICE_FUNC -inline EIGEN_MATHFUNC_RETVAL(real_ref, Scalar) real_ref(Scalar& x) -{ - return EIGEN_MATHFUNC_IMPL(real_ref, Scalar)::run(x); -} - -template -EIGEN_DEVICE_FUNC -inline EIGEN_MATHFUNC_RETVAL(imag, Scalar) imag(const Scalar& x) -{ - return EIGEN_MATHFUNC_IMPL(imag, Scalar)::run(x); -} - -template -EIGEN_DEVICE_FUNC -inline EIGEN_MATHFUNC_RETVAL(arg, Scalar) arg(const Scalar& x) -{ - return EIGEN_MATHFUNC_IMPL(arg, Scalar)::run(x); -} - -template -EIGEN_DEVICE_FUNC -inline typename internal::add_const_on_value_type< EIGEN_MATHFUNC_RETVAL(imag_ref, Scalar) >::type imag_ref(const Scalar& x) -{ - return internal::imag_ref_impl::run(x); -} - -template -EIGEN_DEVICE_FUNC -inline EIGEN_MATHFUNC_RETVAL(imag_ref, Scalar) imag_ref(Scalar& x) -{ - return EIGEN_MATHFUNC_IMPL(imag_ref, Scalar)::run(x); -} - -template -EIGEN_DEVICE_FUNC -inline EIGEN_MATHFUNC_RETVAL(conj, Scalar) conj(const Scalar& x) -{ - return EIGEN_MATHFUNC_IMPL(conj, Scalar)::run(x); -} - -template -EIGEN_DEVICE_FUNC -inline EIGEN_MATHFUNC_RETVAL(abs2, Scalar) abs2(const Scalar& x) -{ - return EIGEN_MATHFUNC_IMPL(abs2, Scalar)::run(x); -} - -template -EIGEN_DEVICE_FUNC -inline EIGEN_MATHFUNC_RETVAL(norm1, Scalar) norm1(const Scalar& x) -{ - return EIGEN_MATHFUNC_IMPL(norm1, Scalar)::run(x); -} - -template -EIGEN_DEVICE_FUNC -inline EIGEN_MATHFUNC_RETVAL(hypot, Scalar) hypot(const Scalar& x, const Scalar& y) -{ - return EIGEN_MATHFUNC_IMPL(hypot, Scalar)::run(x, y); -} - -template -EIGEN_DEVICE_FUNC -inline EIGEN_MATHFUNC_RETVAL(log1p, Scalar) log1p(const Scalar& x) -{ - return EIGEN_MATHFUNC_IMPL(log1p, Scalar)::run(x); -} + template EIGEN_DEVICE_FUNC inline EIGEN_MATHFUNC_RETVAL(real, Scalar) real(const Scalar &x) + { + return EIGEN_MATHFUNC_IMPL(real, Scalar)::run(x); + } + + template + EIGEN_DEVICE_FUNC inline typename internal::add_const_on_value_type::type + real_ref(const Scalar &x) + { + return internal::real_ref_impl::run(x); + } + + template EIGEN_DEVICE_FUNC inline EIGEN_MATHFUNC_RETVAL(real_ref, Scalar) real_ref(Scalar &x) + { + return EIGEN_MATHFUNC_IMPL(real_ref, Scalar)::run(x); + } + + template EIGEN_DEVICE_FUNC inline EIGEN_MATHFUNC_RETVAL(imag, Scalar) imag(const Scalar &x) + { + return EIGEN_MATHFUNC_IMPL(imag, Scalar)::run(x); + } + + template EIGEN_DEVICE_FUNC inline EIGEN_MATHFUNC_RETVAL(arg, Scalar) arg(const Scalar &x) + { + return EIGEN_MATHFUNC_IMPL(arg, Scalar)::run(x); + } + + template + EIGEN_DEVICE_FUNC inline typename internal::add_const_on_value_type::type + imag_ref(const Scalar &x) + { + return internal::imag_ref_impl::run(x); + } + + template EIGEN_DEVICE_FUNC inline EIGEN_MATHFUNC_RETVAL(imag_ref, Scalar) imag_ref(Scalar &x) + { + return EIGEN_MATHFUNC_IMPL(imag_ref, Scalar)::run(x); + } + + template EIGEN_DEVICE_FUNC inline EIGEN_MATHFUNC_RETVAL(conj, Scalar) conj(const Scalar &x) + { + return EIGEN_MATHFUNC_IMPL(conj, Scalar)::run(x); + } + + template EIGEN_DEVICE_FUNC inline EIGEN_MATHFUNC_RETVAL(abs2, Scalar) abs2(const Scalar &x) + { + return EIGEN_MATHFUNC_IMPL(abs2, Scalar)::run(x); + } + + template EIGEN_DEVICE_FUNC inline EIGEN_MATHFUNC_RETVAL(norm1, Scalar) norm1(const Scalar &x) + { + return EIGEN_MATHFUNC_IMPL(norm1, Scalar)::run(x); + } + + template + EIGEN_DEVICE_FUNC inline EIGEN_MATHFUNC_RETVAL(hypot, Scalar) hypot(const Scalar &x, const Scalar &y) + { + return EIGEN_MATHFUNC_IMPL(hypot, Scalar)::run(x, y); + } + + template EIGEN_DEVICE_FUNC inline EIGEN_MATHFUNC_RETVAL(log1p, Scalar) log1p(const Scalar &x) + { + return EIGEN_MATHFUNC_IMPL(log1p, Scalar)::run(x); + } #ifdef __CUDACC__ -template<> EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE -float log1p(const float &x) { return ::log1pf(x); } + template<> EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE float log1p(const float &x) { return ::log1pf(x); } -template<> EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE -double log1p(const double &x) { return ::log1p(x); } + template<> EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE double log1p(const double &x) { return ::log1p(x); } #endif -template -EIGEN_DEVICE_FUNC -inline typename internal::pow_impl::result_type pow(const ScalarX& x, const ScalarY& y) -{ - return internal::pow_impl::run(x, y); -} - -template EIGEN_DEVICE_FUNC bool (isnan) (const T &x) { return internal::isnan_impl(x); } -template EIGEN_DEVICE_FUNC bool (isinf) (const T &x) { return internal::isinf_impl(x); } -template EIGEN_DEVICE_FUNC bool (isfinite)(const T &x) { return internal::isfinite_impl(x); } - -template -EIGEN_DEVICE_FUNC -inline EIGEN_MATHFUNC_RETVAL(round, Scalar) round(const Scalar& x) -{ - return EIGEN_MATHFUNC_IMPL(round, Scalar)::run(x); -} - -template -EIGEN_DEVICE_FUNC -T (floor)(const T& x) -{ - EIGEN_USING_STD_MATH(floor); - return floor(x); -} + template + EIGEN_DEVICE_FUNC inline + typename internal::pow_impl::result_type pow(const ScalarX &x, const ScalarY &y) + { + return internal::pow_impl::run(x, y); + } + + template EIGEN_DEVICE_FUNC bool(isnan)(const T &x) { return internal::isnan_impl(x); } + template EIGEN_DEVICE_FUNC bool(isinf)(const T &x) { return internal::isinf_impl(x); } + template EIGEN_DEVICE_FUNC bool(isfinite)(const T &x) { return internal::isfinite_impl(x); } + + template EIGEN_DEVICE_FUNC inline EIGEN_MATHFUNC_RETVAL(round, Scalar) round(const Scalar &x) + { + return EIGEN_MATHFUNC_IMPL(round, Scalar)::run(x); + } + + template EIGEN_DEVICE_FUNC T(floor)(const T &x) + { + EIGEN_USING_STD_MATH(floor); + return floor(x); + } #ifdef __CUDACC__ -template<> EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE -float floor(const float &x) { return ::floorf(x); } + template<> EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE float floor(const float &x) { return ::floorf(x); } -template<> EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE -double floor(const double &x) { return ::floor(x); } + template<> EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE double floor(const double &x) { return ::floor(x); } #endif -template -EIGEN_DEVICE_FUNC -T (ceil)(const T& x) -{ - EIGEN_USING_STD_MATH(ceil); - return ceil(x); -} + template EIGEN_DEVICE_FUNC T(ceil)(const T &x) + { + EIGEN_USING_STD_MATH(ceil); + return ceil(x); + } #ifdef __CUDACC__ -template<> EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE -float ceil(const float &x) { return ::ceilf(x); } + template<> EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE float ceil(const float &x) { return ::ceilf(x); } -template<> EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE -double ceil(const double &x) { return ::ceil(x); } + template<> EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE double ceil(const double &x) { return ::ceil(x); } #endif -/** Log base 2 for 32 bits positive integers. - * Conveniently returns 0 for x==0. */ -inline int log2(int x) -{ - eigen_assert(x>=0); - unsigned int v(x); - static const int table[32] = { 0, 9, 1, 10, 13, 21, 2, 29, 11, 14, 16, 18, 22, 25, 3, 30, 8, 12, 20, 28, 15, 17, 24, 7, 19, 27, 23, 6, 26, 5, 4, 31 }; - v |= v >> 1; - v |= v >> 2; - v |= v >> 4; - v |= v >> 8; - v |= v >> 16; - return table[(v * 0x07C4ACDDU) >> 27]; -} - -/** \returns the square root of \a x. - * - * It is essentially equivalent to - * \code using std::sqrt; return sqrt(x); \endcode - * but slightly faster for float/double and some compilers (e.g., gcc), thanks to - * specializations when SSE is enabled. - * - * It's usage is justified in performance critical functions, like norm/normalize. - */ -template -EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE -T sqrt(const T &x) -{ - EIGEN_USING_STD_MATH(sqrt); - return sqrt(x); -} - -template -EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE -T log(const T &x) { - EIGEN_USING_STD_MATH(log); - return log(x); -} + /** Log base 2 for 32 bits positive integers. + * Conveniently returns 0 for x==0. */ + inline int log2(int x) + { + eigen_assert(x >= 0); + unsigned int v(x); + static const int table[32] = { 0, + 9, + 1, + 10, + 13, + 21, + 2, + 29, + 11, + 14, + 16, + 18, + 22, + 25, + 3, + 30, + 8, + 12, + 20, + 28, + 15, + 17, + 24, + 7, + 19, + 27, + 23, + 6, + 26, + 5, + 4, + 31 }; + v |= v >> 1; + v |= v >> 2; + v |= v >> 4; + v |= v >> 8; + v |= v >> 16; + return table[(v * 0x07C4ACDDU) >> 27]; + } + + /** \returns the square root of \a x. + * + * It is essentially equivalent to + * \code using std::sqrt; return sqrt(x); \endcode + * but slightly faster for float/double and some compilers (e.g., gcc), thanks to + * specializations when SSE is enabled. + * + * It's usage is justified in performance critical functions, like norm/normalize. + */ + template EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE T sqrt(const T &x) + { + EIGEN_USING_STD_MATH(sqrt); + return sqrt(x); + } + + template EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE T log(const T &x) + { + EIGEN_USING_STD_MATH(log); + return log(x); + } #ifdef __CUDACC__ -template<> EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE -float log(const float &x) { return ::logf(x); } + template<> EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE float log(const float &x) { return ::logf(x); } -template<> EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE -double log(const double &x) { return ::log(x); } + template<> EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE double log(const double &x) { return ::log(x); } #endif -template -EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE -typename internal::enable_if::IsSigned || NumTraits::IsComplex,typename NumTraits::Real>::type -abs(const T &x) { - EIGEN_USING_STD_MATH(abs); - return abs(x); -} - -template -EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE -typename internal::enable_if::IsSigned || NumTraits::IsComplex),typename NumTraits::Real>::type -abs(const T &x) { - return x; -} + template + EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE + typename internal::enable_if::IsSigned || NumTraits::IsComplex, typename NumTraits::Real>::type + abs(const T &x) + { + EIGEN_USING_STD_MATH(abs); + return abs(x); + } + + template + EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE + typename internal::enable_if::IsSigned || NumTraits::IsComplex), + typename NumTraits::Real>::type + abs(const T &x) + { + return x; + } #if defined(__SYCL_DEVICE_ONLY__) -EIGEN_ALWAYS_INLINE float abs(float x) { return cl::sycl::fabs(x); } -EIGEN_ALWAYS_INLINE double abs(double x) { return cl::sycl::fabs(x); } -#endif // defined(__SYCL_DEVICE_ONLY__) + EIGEN_ALWAYS_INLINE float abs(float x) { return cl::sycl::fabs(x); } + EIGEN_ALWAYS_INLINE double abs(double x) { return cl::sycl::fabs(x); } +#endif// defined(__SYCL_DEVICE_ONLY__) #ifdef __CUDACC__ -template<> EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE -float abs(const float &x) { return ::fabsf(x); } + template<> EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE float abs(const float &x) { return ::fabsf(x); } -template<> EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE -double abs(const double &x) { return ::fabs(x); } + template<> EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE double abs(const double &x) { return ::fabs(x); } -template <> EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE -float abs(const std::complex& x) { - return ::hypotf(x.real(), x.imag()); -} + template<> EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE float abs(const std::complex &x) + { + return ::hypotf(x.real(), x.imag()); + } -template <> EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE -double abs(const std::complex& x) { - return ::hypot(x.real(), x.imag()); -} + template<> EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE double abs(const std::complex &x) + { + return ::hypot(x.real(), x.imag()); + } #endif -template -EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE -T exp(const T &x) { - EIGEN_USING_STD_MATH(exp); - return exp(x); -} + template EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE T exp(const T &x) + { + EIGEN_USING_STD_MATH(exp); + return exp(x); + } #ifdef __CUDACC__ -template<> EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE -float exp(const float &x) { return ::expf(x); } + template<> EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE float exp(const float &x) { return ::expf(x); } -template<> EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE -double exp(const double &x) { return ::exp(x); } + template<> EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE double exp(const double &x) { return ::exp(x); } #endif -template -EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE -T cos(const T &x) { - EIGEN_USING_STD_MATH(cos); - return cos(x); -} + template EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE T cos(const T &x) + { + EIGEN_USING_STD_MATH(cos); + return cos(x); + } #ifdef __CUDACC__ -template<> EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE -float cos(const float &x) { return ::cosf(x); } + template<> EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE float cos(const float &x) { return ::cosf(x); } -template<> EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE -double cos(const double &x) { return ::cos(x); } + template<> EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE double cos(const double &x) { return ::cos(x); } #endif -template -EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE -T sin(const T &x) { - EIGEN_USING_STD_MATH(sin); - return sin(x); -} + template EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE T sin(const T &x) + { + EIGEN_USING_STD_MATH(sin); + return sin(x); + } #ifdef __CUDACC__ -template<> EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE -float sin(const float &x) { return ::sinf(x); } + template<> EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE float sin(const float &x) { return ::sinf(x); } -template<> EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE -double sin(const double &x) { return ::sin(x); } + template<> EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE double sin(const double &x) { return ::sin(x); } #endif -template -EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE -T tan(const T &x) { - EIGEN_USING_STD_MATH(tan); - return tan(x); -} + template EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE T tan(const T &x) + { + EIGEN_USING_STD_MATH(tan); + return tan(x); + } #ifdef __CUDACC__ -template<> EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE -float tan(const float &x) { return ::tanf(x); } + template<> EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE float tan(const float &x) { return ::tanf(x); } -template<> EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE -double tan(const double &x) { return ::tan(x); } + template<> EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE double tan(const double &x) { return ::tan(x); } #endif -template -EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE -T acos(const T &x) { - EIGEN_USING_STD_MATH(acos); - return acos(x); -} + template EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE T acos(const T &x) + { + EIGEN_USING_STD_MATH(acos); + return acos(x); + } #ifdef __CUDACC__ -template<> EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE -float acos(const float &x) { return ::acosf(x); } + template<> EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE float acos(const float &x) { return ::acosf(x); } -template<> EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE -double acos(const double &x) { return ::acos(x); } + template<> EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE double acos(const double &x) { return ::acos(x); } #endif -template -EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE -T asin(const T &x) { - EIGEN_USING_STD_MATH(asin); - return asin(x); -} + template EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE T asin(const T &x) + { + EIGEN_USING_STD_MATH(asin); + return asin(x); + } #ifdef __CUDACC__ -template<> EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE -float asin(const float &x) { return ::asinf(x); } + template<> EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE float asin(const float &x) { return ::asinf(x); } -template<> EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE -double asin(const double &x) { return ::asin(x); } + template<> EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE double asin(const double &x) { return ::asin(x); } #endif -template -EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE -T atan(const T &x) { - EIGEN_USING_STD_MATH(atan); - return atan(x); -} + template EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE T atan(const T &x) + { + EIGEN_USING_STD_MATH(atan); + return atan(x); + } #ifdef __CUDACC__ -template<> EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE -float atan(const float &x) { return ::atanf(x); } + template<> EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE float atan(const float &x) { return ::atanf(x); } -template<> EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE -double atan(const double &x) { return ::atan(x); } + template<> EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE double atan(const double &x) { return ::atan(x); } #endif -template -EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE -T cosh(const T &x) { - EIGEN_USING_STD_MATH(cosh); - return cosh(x); -} + template EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE T cosh(const T &x) + { + EIGEN_USING_STD_MATH(cosh); + return cosh(x); + } #ifdef __CUDACC__ -template<> EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE -float cosh(const float &x) { return ::coshf(x); } + template<> EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE float cosh(const float &x) { return ::coshf(x); } -template<> EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE -double cosh(const double &x) { return ::cosh(x); } + template<> EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE double cosh(const double &x) { return ::cosh(x); } #endif -template -EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE -T sinh(const T &x) { - EIGEN_USING_STD_MATH(sinh); - return sinh(x); -} + template EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE T sinh(const T &x) + { + EIGEN_USING_STD_MATH(sinh); + return sinh(x); + } #ifdef __CUDACC__ -template<> EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE -float sinh(const float &x) { return ::sinhf(x); } + template<> EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE float sinh(const float &x) { return ::sinhf(x); } -template<> EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE -double sinh(const double &x) { return ::sinh(x); } + template<> EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE double sinh(const double &x) { return ::sinh(x); } #endif -template -EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE -T tanh(const T &x) { - EIGEN_USING_STD_MATH(tanh); - return tanh(x); -} + template EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE T tanh(const T &x) + { + EIGEN_USING_STD_MATH(tanh); + return tanh(x); + } #if (!defined(__CUDACC__)) && EIGEN_FAST_MATH -EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE -float tanh(float x) { return internal::generic_fast_tanh_float(x); } + EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE float tanh(float x) { return internal::generic_fast_tanh_float(x); } #endif #ifdef __CUDACC__ -template<> EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE -float tanh(const float &x) { return ::tanhf(x); } + template<> EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE float tanh(const float &x) { return ::tanhf(x); } -template<> EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE -double tanh(const double &x) { return ::tanh(x); } + template<> EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE double tanh(const double &x) { return ::tanh(x); } #endif -template -EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE -T fmod(const T& a, const T& b) { - EIGEN_USING_STD_MATH(fmod); - return fmod(a, b); -} + template EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE T fmod(const T &a, const T &b) + { + EIGEN_USING_STD_MATH(fmod); + return fmod(a, b); + } #ifdef __CUDACC__ -template <> -EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE -float fmod(const float& a, const float& b) { - return ::fmodf(a, b); -} - -template <> -EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE -double fmod(const double& a, const double& b) { - return ::fmod(a, b); -} + template<> EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE float fmod(const float &a, const float &b) { return ::fmodf(a, b); } + + template<> EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE double fmod(const double &a, const double &b) + { + return ::fmod(a, b); + } #endif -} // end namespace numext +}// end namespace numext namespace internal { -template -EIGEN_DEVICE_FUNC bool isfinite_impl(const std::complex& x) -{ - return (numext::isfinite)(numext::real(x)) && (numext::isfinite)(numext::imag(x)); -} - -template -EIGEN_DEVICE_FUNC bool isnan_impl(const std::complex& x) -{ - return (numext::isnan)(numext::real(x)) || (numext::isnan)(numext::imag(x)); -} - -template -EIGEN_DEVICE_FUNC bool isinf_impl(const std::complex& x) -{ - return ((numext::isinf)(numext::real(x)) || (numext::isinf)(numext::imag(x))) && (!(numext::isnan)(x)); -} - -/**************************************************************************** -* Implementation of fuzzy comparisons * -****************************************************************************/ - -template -struct scalar_fuzzy_default_impl {}; - -template -struct scalar_fuzzy_default_impl -{ - typedef typename NumTraits::Real RealScalar; - template EIGEN_DEVICE_FUNC - static inline bool isMuchSmallerThan(const Scalar& x, const OtherScalar& y, const RealScalar& prec) - { - return numext::abs(x) <= numext::abs(y) * prec; - } - EIGEN_DEVICE_FUNC - static inline bool isApprox(const Scalar& x, const Scalar& y, const RealScalar& prec) + template EIGEN_DEVICE_FUNC bool isfinite_impl(const std::complex &x) { - return numext::abs(x - y) <= numext::mini(numext::abs(x), numext::abs(y)) * prec; + return (numext::isfinite)(numext::real(x)) && (numext::isfinite)(numext::imag(x)); } - EIGEN_DEVICE_FUNC - static inline bool isApproxOrLessThan(const Scalar& x, const Scalar& y, const RealScalar& prec) + + template EIGEN_DEVICE_FUNC bool isnan_impl(const std::complex &x) { - return x <= y || isApprox(x, y, prec); + return (numext::isnan)(numext::real(x)) || (numext::isnan)(numext::imag(x)); } -}; -template -struct scalar_fuzzy_default_impl -{ - typedef typename NumTraits::Real RealScalar; - template EIGEN_DEVICE_FUNC - static inline bool isMuchSmallerThan(const Scalar& x, const Scalar&, const RealScalar&) + template EIGEN_DEVICE_FUNC bool isinf_impl(const std::complex &x) { - return x == Scalar(0); + return ((numext::isinf)(numext::real(x)) || (numext::isinf)(numext::imag(x))) && (!(numext::isnan)(x)); } - EIGEN_DEVICE_FUNC - static inline bool isApprox(const Scalar& x, const Scalar& y, const RealScalar&) + + /**************************************************************************** + * Implementation of fuzzy comparisons * + ****************************************************************************/ + + template struct scalar_fuzzy_default_impl { - return x == y; - } - EIGEN_DEVICE_FUNC - static inline bool isApproxOrLessThan(const Scalar& x, const Scalar& y, const RealScalar&) + }; + + template struct scalar_fuzzy_default_impl { - return x <= y; - } -}; + typedef typename NumTraits::Real RealScalar; + template + EIGEN_DEVICE_FUNC static inline bool + isMuchSmallerThan(const Scalar &x, const OtherScalar &y, const RealScalar &prec) + { + return numext::abs(x) <= numext::abs(y) * prec; + } + EIGEN_DEVICE_FUNC + static inline bool isApprox(const Scalar &x, const Scalar &y, const RealScalar &prec) + { + return numext::abs(x - y) <= numext::mini(numext::abs(x), numext::abs(y)) * prec; + } + EIGEN_DEVICE_FUNC + static inline bool isApproxOrLessThan(const Scalar &x, const Scalar &y, const RealScalar &prec) + { + return x <= y || isApprox(x, y, prec); + } + }; -template -struct scalar_fuzzy_default_impl -{ - typedef typename NumTraits::Real RealScalar; - template EIGEN_DEVICE_FUNC - static inline bool isMuchSmallerThan(const Scalar& x, const OtherScalar& y, const RealScalar& prec) + template struct scalar_fuzzy_default_impl { - return numext::abs2(x) <= numext::abs2(y) * prec * prec; - } - EIGEN_DEVICE_FUNC - static inline bool isApprox(const Scalar& x, const Scalar& y, const RealScalar& prec) + typedef typename NumTraits::Real RealScalar; + template + EIGEN_DEVICE_FUNC static inline bool isMuchSmallerThan(const Scalar &x, const Scalar &, const RealScalar &) + { + return x == Scalar(0); + } + EIGEN_DEVICE_FUNC + static inline bool isApprox(const Scalar &x, const Scalar &y, const RealScalar &) { return x == y; } + EIGEN_DEVICE_FUNC + static inline bool isApproxOrLessThan(const Scalar &x, const Scalar &y, const RealScalar &) { return x <= y; } + }; + + template struct scalar_fuzzy_default_impl { - return numext::abs2(x - y) <= numext::mini(numext::abs2(x), numext::abs2(y)) * prec * prec; - } -}; - -template -struct scalar_fuzzy_impl : scalar_fuzzy_default_impl::IsComplex, NumTraits::IsInteger> {}; - -template EIGEN_DEVICE_FUNC -inline bool isMuchSmallerThan(const Scalar& x, const OtherScalar& y, - const typename NumTraits::Real &precision = NumTraits::dummy_precision()) -{ - return scalar_fuzzy_impl::template isMuchSmallerThan(x, y, precision); -} - -template EIGEN_DEVICE_FUNC -inline bool isApprox(const Scalar& x, const Scalar& y, - const typename NumTraits::Real &precision = NumTraits::dummy_precision()) -{ - return scalar_fuzzy_impl::isApprox(x, y, precision); -} - -template EIGEN_DEVICE_FUNC -inline bool isApproxOrLessThan(const Scalar& x, const Scalar& y, - const typename NumTraits::Real &precision = NumTraits::dummy_precision()) -{ - return scalar_fuzzy_impl::isApproxOrLessThan(x, y, precision); -} - -/****************************************** -*** The special case of the bool type *** -******************************************/ - -template<> struct random_impl -{ - static inline bool run() - { - return random(0,1)==0 ? false : true; - } -}; + typedef typename NumTraits::Real RealScalar; + template + EIGEN_DEVICE_FUNC static inline bool + isMuchSmallerThan(const Scalar &x, const OtherScalar &y, const RealScalar &prec) + { + return numext::abs2(x) <= numext::abs2(y) * prec * prec; + } + EIGEN_DEVICE_FUNC + static inline bool isApprox(const Scalar &x, const Scalar &y, const RealScalar &prec) + { + return numext::abs2(x - y) <= numext::mini(numext::abs2(x), numext::abs2(y)) * prec * prec; + } + }; + + template + struct scalar_fuzzy_impl + : scalar_fuzzy_default_impl::IsComplex, NumTraits::IsInteger> + { + }; -template<> struct scalar_fuzzy_impl -{ - typedef bool RealScalar; - - template EIGEN_DEVICE_FUNC - static inline bool isMuchSmallerThan(const bool& x, const bool&, const bool&) + template + EIGEN_DEVICE_FUNC inline bool isMuchSmallerThan(const Scalar &x, + const OtherScalar &y, + const typename NumTraits::Real &precision = NumTraits::dummy_precision()) { - return !x; + return scalar_fuzzy_impl::template isMuchSmallerThan(x, y, precision); } - - EIGEN_DEVICE_FUNC - static inline bool isApprox(bool x, bool y, bool) + + template + EIGEN_DEVICE_FUNC inline bool isApprox(const Scalar &x, + const Scalar &y, + const typename NumTraits::Real &precision = NumTraits::dummy_precision()) { - return x == y; + return scalar_fuzzy_impl::isApprox(x, y, precision); } - EIGEN_DEVICE_FUNC - static inline bool isApproxOrLessThan(const bool& x, const bool& y, const bool&) + template + EIGEN_DEVICE_FUNC inline bool isApproxOrLessThan(const Scalar &x, + const Scalar &y, + const typename NumTraits::Real &precision = NumTraits::dummy_precision()) { - return (!x) || y; + return scalar_fuzzy_impl::isApproxOrLessThan(x, y, precision); } - -}; - -} // end namespace internal + /****************************************** + *** The special case of the bool type *** + ******************************************/ + + template<> struct random_impl + { + static inline bool run() { return random(0, 1) == 0 ? false : true; } + }; + + template<> struct scalar_fuzzy_impl + { + typedef bool RealScalar; + + template + EIGEN_DEVICE_FUNC static inline bool isMuchSmallerThan(const bool &x, const bool &, const bool &) + { + return !x; + } + + EIGEN_DEVICE_FUNC + static inline bool isApprox(bool x, bool y, bool) { return x == y; } + + EIGEN_DEVICE_FUNC + static inline bool isApproxOrLessThan(const bool &x, const bool &y, const bool &) { return (!x) || y; } + }; + + +}// end namespace internal -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_MATHFUNCTIONS_H +#endif// EIGEN_MATHFUNCTIONS_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/MathFunctionsImpl.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/MathFunctionsImpl.h index 9c1ceb0e..872b77a0 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/MathFunctionsImpl.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/MathFunctionsImpl.h @@ -15,87 +15,84 @@ namespace Eigen { namespace internal { -/** \internal \returns the hyperbolic tan of \a a (coeff-wise) - Doesn't do anything fancy, just a 13/6-degree rational interpolant which - is accurate up to a couple of ulp in the range [-9, 9], outside of which - the tanh(x) = +/-1. + /** \internal \returns the hyperbolic tan of \a a (coeff-wise) + Doesn't do anything fancy, just a 13/6-degree rational interpolant which + is accurate up to a couple of ulp in the range [-9, 9], outside of which + the tanh(x) = +/-1. - This implementation works on both scalars and packets. -*/ -template -T generic_fast_tanh_float(const T& a_x) -{ - // Clamp the inputs to the range [-9, 9] since anything outside - // this range is +/-1.0f in single-precision. - const T plus_9 = pset1(9.f); - const T minus_9 = pset1(-9.f); - // NOTE GCC prior to 6.3 might improperly optimize this max/min - // step such that if a_x is nan, x will be either 9 or -9, - // and tanh will return 1 or -1 instead of nan. - // This is supposed to be fixed in gcc6.3, - // see: https://gcc.gnu.org/bugzilla/show_bug.cgi?id=72867 - const T x = pmax(minus_9,pmin(plus_9,a_x)); - // The monomial coefficients of the numerator polynomial (odd). - const T alpha_1 = pset1(4.89352455891786e-03f); - const T alpha_3 = pset1(6.37261928875436e-04f); - const T alpha_5 = pset1(1.48572235717979e-05f); - const T alpha_7 = pset1(5.12229709037114e-08f); - const T alpha_9 = pset1(-8.60467152213735e-11f); - const T alpha_11 = pset1(2.00018790482477e-13f); - const T alpha_13 = pset1(-2.76076847742355e-16f); - - // The monomial coefficients of the denominator polynomial (even). - const T beta_0 = pset1(4.89352518554385e-03f); - const T beta_2 = pset1(2.26843463243900e-03f); - const T beta_4 = pset1(1.18534705686654e-04f); - const T beta_6 = pset1(1.19825839466702e-06f); + This implementation works on both scalars and packets. + */ + template T generic_fast_tanh_float(const T &a_x) + { + // Clamp the inputs to the range [-9, 9] since anything outside + // this range is +/-1.0f in single-precision. + const T plus_9 = pset1(9.f); + const T minus_9 = pset1(-9.f); + // NOTE GCC prior to 6.3 might improperly optimize this max/min + // step such that if a_x is nan, x will be either 9 or -9, + // and tanh will return 1 or -1 instead of nan. + // This is supposed to be fixed in gcc6.3, + // see: https://gcc.gnu.org/bugzilla/show_bug.cgi?id=72867 + const T x = pmax(minus_9, pmin(plus_9, a_x)); + // The monomial coefficients of the numerator polynomial (odd). + const T alpha_1 = pset1(4.89352455891786e-03f); + const T alpha_3 = pset1(6.37261928875436e-04f); + const T alpha_5 = pset1(1.48572235717979e-05f); + const T alpha_7 = pset1(5.12229709037114e-08f); + const T alpha_9 = pset1(-8.60467152213735e-11f); + const T alpha_11 = pset1(2.00018790482477e-13f); + const T alpha_13 = pset1(-2.76076847742355e-16f); - // Since the polynomials are odd/even, we need x^2. - const T x2 = pmul(x, x); + // The monomial coefficients of the denominator polynomial (even). + const T beta_0 = pset1(4.89352518554385e-03f); + const T beta_2 = pset1(2.26843463243900e-03f); + const T beta_4 = pset1(1.18534705686654e-04f); + const T beta_6 = pset1(1.19825839466702e-06f); - // Evaluate the numerator polynomial p. - T p = pmadd(x2, alpha_13, alpha_11); - p = pmadd(x2, p, alpha_9); - p = pmadd(x2, p, alpha_7); - p = pmadd(x2, p, alpha_5); - p = pmadd(x2, p, alpha_3); - p = pmadd(x2, p, alpha_1); - p = pmul(x, p); + // Since the polynomials are odd/even, we need x^2. + const T x2 = pmul(x, x); - // Evaluate the denominator polynomial p. - T q = pmadd(x2, beta_6, beta_4); - q = pmadd(x2, q, beta_2); - q = pmadd(x2, q, beta_0); + // Evaluate the numerator polynomial p. + T p = pmadd(x2, alpha_13, alpha_11); + p = pmadd(x2, p, alpha_9); + p = pmadd(x2, p, alpha_7); + p = pmadd(x2, p, alpha_5); + p = pmadd(x2, p, alpha_3); + p = pmadd(x2, p, alpha_1); + p = pmul(x, p); - // Divide the numerator by the denominator. - return pdiv(p, q); -} + // Evaluate the denominator polynomial p. + T q = pmadd(x2, beta_6, beta_4); + q = pmadd(x2, q, beta_2); + q = pmadd(x2, q, beta_0); -template -EIGEN_STRONG_INLINE -RealScalar positive_real_hypot(const RealScalar& x, const RealScalar& y) -{ - EIGEN_USING_STD_MATH(sqrt); - RealScalar p, qp; - p = numext::maxi(x,y); - if(p==RealScalar(0)) return RealScalar(0); - qp = numext::mini(y,x) / p; - return p * sqrt(RealScalar(1) + qp*qp); -} + // Divide the numerator by the denominator. + return pdiv(p, q); + } -template -struct hypot_impl -{ - typedef typename NumTraits::Real RealScalar; - static inline RealScalar run(const Scalar& x, const Scalar& y) + template + EIGEN_STRONG_INLINE RealScalar positive_real_hypot(const RealScalar &x, const RealScalar &y) { - EIGEN_USING_STD_MATH(abs); - return positive_real_hypot(abs(x), abs(y)); + EIGEN_USING_STD_MATH(sqrt); + RealScalar p, qp; + p = numext::maxi(x, y); + if (p == RealScalar(0)) return RealScalar(0); + qp = numext::mini(y, x) / p; + return p * sqrt(RealScalar(1) + qp * qp); } -}; -} // end namespace internal + template struct hypot_impl + { + typedef typename NumTraits::Real RealScalar; + static inline RealScalar run(const Scalar &x, const Scalar &y) + { + EIGEN_USING_STD_MATH(abs); + return positive_real_hypot(abs(x), abs(y)); + } + }; + +}// end namespace internal -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_MATHFUNCTIONSIMPL_H +#endif// EIGEN_MATHFUNCTIONSIMPL_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/Matrix.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/Matrix.h index 7f4a7af9..885393aa 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/Matrix.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/Matrix.h @@ -14,446 +14,439 @@ namespace Eigen { namespace internal { -template -struct traits > -{ -private: - enum { size = internal::size_at_compile_time<_Rows,_Cols>::ret }; - typedef typename find_best_packet<_Scalar,size>::type PacketScalar; - enum { - row_major_bit = _Options&RowMajor ? RowMajorBit : 0, - is_dynamic_size_storage = _MaxRows==Dynamic || _MaxCols==Dynamic, - max_size = is_dynamic_size_storage ? Dynamic : _MaxRows*_MaxCols, - default_alignment = compute_default_alignment<_Scalar,max_size>::value, - actual_alignment = ((_Options&DontAlign)==0) ? default_alignment : 0, + template + struct traits> + { + private: + enum { size = internal::size_at_compile_time<_Rows, _Cols>::ret }; + typedef typename find_best_packet<_Scalar, size>::type PacketScalar; + enum { + row_major_bit = _Options & RowMajor ? RowMajorBit : 0, + is_dynamic_size_storage = _MaxRows == Dynamic || _MaxCols == Dynamic, + max_size = is_dynamic_size_storage ? Dynamic : _MaxRows * _MaxCols, + default_alignment = compute_default_alignment<_Scalar, max_size>::value, + actual_alignment = ((_Options & DontAlign) == 0) ? default_alignment : 0, required_alignment = unpacket_traits::alignment, - packet_access_bit = (packet_traits<_Scalar>::Vectorizable && (EIGEN_UNALIGNED_VECTORIZE || (actual_alignment>=required_alignment))) ? PacketAccessBit : 0 + packet_access_bit = (packet_traits<_Scalar>::Vectorizable + && (EIGEN_UNALIGNED_VECTORIZE || (actual_alignment >= required_alignment))) + ? PacketAccessBit + : 0 + }; + + public: + typedef _Scalar Scalar; + typedef Dense StorageKind; + typedef Eigen::Index StorageIndex; + typedef MatrixXpr XprKind; + enum { + RowsAtCompileTime = _Rows, + ColsAtCompileTime = _Cols, + MaxRowsAtCompileTime = _MaxRows, + MaxColsAtCompileTime = _MaxCols, + Flags = compute_matrix_flags<_Scalar, _Rows, _Cols, _Options, _MaxRows, _MaxCols>::ret, + Options = _Options, + InnerStrideAtCompileTime = 1, + OuterStrideAtCompileTime = (Options & RowMajor) ? ColsAtCompileTime : RowsAtCompileTime, + + // FIXME, the following flag in only used to define NeedsToAlign in PlainObjectBase + EvaluatorFlags = LinearAccessBit | DirectAccessBit | packet_access_bit | row_major_bit, + Alignment = actual_alignment }; - -public: - typedef _Scalar Scalar; - typedef Dense StorageKind; - typedef Eigen::Index StorageIndex; - typedef MatrixXpr XprKind; - enum { - RowsAtCompileTime = _Rows, - ColsAtCompileTime = _Cols, - MaxRowsAtCompileTime = _MaxRows, - MaxColsAtCompileTime = _MaxCols, - Flags = compute_matrix_flags<_Scalar, _Rows, _Cols, _Options, _MaxRows, _MaxCols>::ret, - Options = _Options, - InnerStrideAtCompileTime = 1, - OuterStrideAtCompileTime = (Options&RowMajor) ? ColsAtCompileTime : RowsAtCompileTime, - - // FIXME, the following flag in only used to define NeedsToAlign in PlainObjectBase - EvaluatorFlags = LinearAccessBit | DirectAccessBit | packet_access_bit | row_major_bit, - Alignment = actual_alignment }; -}; -} +}// namespace internal /** \class Matrix - * \ingroup Core_Module - * - * \brief The matrix class, also used for vectors and row-vectors - * - * The %Matrix class is the work-horse for all \em dense (\ref dense "note") matrices and vectors within Eigen. - * Vectors are matrices with one column, and row-vectors are matrices with one row. - * - * The %Matrix class encompasses \em both fixed-size and dynamic-size objects (\ref fixedsize "note"). - * - * The first three template parameters are required: - * \tparam _Scalar Numeric type, e.g. float, double, int or std::complex. - * User defined scalar types are supported as well (see \ref user_defined_scalars "here"). - * \tparam _Rows Number of rows, or \b Dynamic - * \tparam _Cols Number of columns, or \b Dynamic - * - * The remaining template parameters are optional -- in most cases you don't have to worry about them. - * \tparam _Options A combination of either \b #RowMajor or \b #ColMajor, and of either - * \b #AutoAlign or \b #DontAlign. - * The former controls \ref TopicStorageOrders "storage order", and defaults to column-major. The latter controls alignment, which is required - * for vectorization. It defaults to aligning matrices except for fixed sizes that aren't a multiple of the packet size. - * \tparam _MaxRows Maximum number of rows. Defaults to \a _Rows (\ref maxrows "note"). - * \tparam _MaxCols Maximum number of columns. Defaults to \a _Cols (\ref maxrows "note"). - * - * Eigen provides a number of typedefs covering the usual cases. Here are some examples: - * - * \li \c Matrix2d is a 2x2 square matrix of doubles (\c Matrix) - * \li \c Vector4f is a vector of 4 floats (\c Matrix) - * \li \c RowVector3i is a row-vector of 3 ints (\c Matrix) - * - * \li \c MatrixXf is a dynamic-size matrix of floats (\c Matrix) - * \li \c VectorXf is a dynamic-size vector of floats (\c Matrix) - * - * \li \c Matrix2Xf is a partially fixed-size (dynamic-size) matrix of floats (\c Matrix) - * \li \c MatrixX3d is a partially dynamic-size (fixed-size) matrix of double (\c Matrix) - * - * See \link matrixtypedefs this page \endlink for a complete list of predefined \em %Matrix and \em Vector typedefs. - * - * You can access elements of vectors and matrices using normal subscripting: - * - * \code - * Eigen::VectorXd v(10); - * v[0] = 0.1; - * v[1] = 0.2; - * v(0) = 0.3; - * v(1) = 0.4; - * - * Eigen::MatrixXi m(10, 10); - * m(0, 1) = 1; - * m(0, 2) = 2; - * m(0, 3) = 3; - * \endcode - * - * This class can be extended with the help of the plugin mechanism described on the page - * \ref TopicCustomizing_Plugins by defining the preprocessor symbol \c EIGEN_MATRIX_PLUGIN. - * - * Some notes: - * - *
- *
\anchor dense Dense versus sparse:
- *
This %Matrix class handles dense, not sparse matrices and vectors. For sparse matrices and vectors, see the Sparse module. - * - * Dense matrices and vectors are plain usual arrays of coefficients. All the coefficients are stored, in an ordinary contiguous array. - * This is unlike Sparse matrices and vectors where the coefficients are stored as a list of nonzero coefficients.
- * - *
\anchor fixedsize Fixed-size versus dynamic-size:
- *
Fixed-size means that the numbers of rows and columns are known are compile-time. In this case, Eigen allocates the array - * of coefficients as a fixed-size array, as a class member. This makes sense for very small matrices, typically up to 4x4, sometimes up - * to 16x16. Larger matrices should be declared as dynamic-size even if one happens to know their size at compile-time. - * - * Dynamic-size means that the numbers of rows or columns are not necessarily known at compile-time. In this case they are runtime - * variables, and the array of coefficients is allocated dynamically on the heap. - * - * Note that \em dense matrices, be they Fixed-size or Dynamic-size, do not expand dynamically in the sense of a std::map. - * If you want this behavior, see the Sparse module.
- * - *
\anchor maxrows _MaxRows and _MaxCols:
- *
In most cases, one just leaves these parameters to the default values. - * These parameters mean the maximum size of rows and columns that the matrix may have. They are useful in cases - * when the exact numbers of rows and columns are not known are compile-time, but it is known at compile-time that they cannot - * exceed a certain value. This happens when taking dynamic-size blocks inside fixed-size matrices: in this case _MaxRows and _MaxCols - * are the dimensions of the original matrix, while _Rows and _Cols are Dynamic.
- *
- * - * ABI and storage layout - * - * The table below summarizes the ABI of some possible Matrix instances which is fixed thorough the lifetime of Eigen 3. - * - * - * - * - * - * - *
Matrix typeEquivalent C structure
\code Matrix \endcode\code - * struct { - * T *data; // with (size_t(data)%EIGEN_MAX_ALIGN_BYTES)==0 - * Eigen::Index rows, cols; - * }; - * \endcode
\code - * Matrix - * Matrix \endcode\code - * struct { - * T *data; // with (size_t(data)%EIGEN_MAX_ALIGN_BYTES)==0 - * Eigen::Index size; - * }; - * \endcode
\code Matrix \endcode\code - * struct { - * T data[Rows*Cols]; // with (size_t(data)%A(Rows*Cols*sizeof(T)))==0 - * }; - * \endcode
\code Matrix \endcode\code - * struct { - * T data[MaxRows*MaxCols]; // with (size_t(data)%A(MaxRows*MaxCols*sizeof(T)))==0 - * Eigen::Index rows, cols; - * }; - * \endcode
- * Note that in this table Rows, Cols, MaxRows and MaxCols are all positive integers. A(S) is defined to the largest possible power-of-two - * smaller to EIGEN_MAX_STATIC_ALIGN_BYTES. - * - * \see MatrixBase for the majority of the API methods for matrices, \ref TopicClassHierarchy, - * \ref TopicStorageOrders - */ + * \ingroup Core_Module + * + * \brief The matrix class, also used for vectors and row-vectors + * + * The %Matrix class is the work-horse for all \em dense (\ref dense "note") matrices and vectors within Eigen. + * Vectors are matrices with one column, and row-vectors are matrices with one row. + * + * The %Matrix class encompasses \em both fixed-size and dynamic-size objects (\ref fixedsize "note"). + * + * The first three template parameters are required: + * \tparam _Scalar Numeric type, e.g. float, double, int or std::complex. + * User defined scalar types are supported as well (see \ref user_defined_scalars "here"). + * \tparam _Rows Number of rows, or \b Dynamic + * \tparam _Cols Number of columns, or \b Dynamic + * + * The remaining template parameters are optional -- in most cases you don't have to worry about them. + * \tparam _Options A combination of either \b #RowMajor or \b #ColMajor, and of either + * \b #AutoAlign or \b #DontAlign. + * The former controls \ref TopicStorageOrders "storage order", and defaults to column-major. The latter + * controls alignment, which is required for vectorization. It defaults to aligning matrices except for fixed sizes that + * aren't a multiple of the packet size. + * \tparam _MaxRows Maximum number of rows. Defaults to \a _Rows (\ref maxrows "note"). + * \tparam _MaxCols Maximum number of columns. Defaults to \a _Cols (\ref maxrows "note"). + * + * Eigen provides a number of typedefs covering the usual cases. Here are some examples: + * + * \li \c Matrix2d is a 2x2 square matrix of doubles (\c Matrix) + * \li \c Vector4f is a vector of 4 floats (\c Matrix) + * \li \c RowVector3i is a row-vector of 3 ints (\c Matrix) + * + * \li \c MatrixXf is a dynamic-size matrix of floats (\c Matrix) + * \li \c VectorXf is a dynamic-size vector of floats (\c Matrix) + * + * \li \c Matrix2Xf is a partially fixed-size (dynamic-size) matrix of floats (\c Matrix) + * \li \c MatrixX3d is a partially dynamic-size (fixed-size) matrix of double (\c Matrix) + * + * See \link matrixtypedefs this page \endlink for a complete list of predefined \em %Matrix and \em Vector typedefs. + * + * You can access elements of vectors and matrices using normal subscripting: + * + * \code + * Eigen::VectorXd v(10); + * v[0] = 0.1; + * v[1] = 0.2; + * v(0) = 0.3; + * v(1) = 0.4; + * + * Eigen::MatrixXi m(10, 10); + * m(0, 1) = 1; + * m(0, 2) = 2; + * m(0, 3) = 3; + * \endcode + * + * This class can be extended with the help of the plugin mechanism described on the page + * \ref TopicCustomizing_Plugins by defining the preprocessor symbol \c EIGEN_MATRIX_PLUGIN. + * + * Some notes: + * + *
+ *
\anchor dense Dense versus sparse:
+ *
This %Matrix class handles dense, not sparse matrices and vectors. For sparse matrices and vectors, see the + * Sparse module. + * + * Dense matrices and vectors are plain usual arrays of coefficients. All the coefficients are stored, in an ordinary + * contiguous array. This is unlike Sparse matrices and vectors where the coefficients are stored as a list of nonzero + * coefficients.
+ * + *
\anchor fixedsize Fixed-size versus dynamic-size:
+ *
Fixed-size means that the numbers of rows and columns are known are compile-time. In this case, Eigen allocates + * the array of coefficients as a fixed-size array, as a class member. This makes sense for very small matrices, + * typically up to 4x4, sometimes up to 16x16. Larger matrices should be declared as dynamic-size even if one happens to + * know their size at compile-time. + * + * Dynamic-size means that the numbers of rows or columns are not necessarily known at compile-time. In this case they + * are runtime variables, and the array of coefficients is allocated dynamically on the heap. + * + * Note that \em dense matrices, be they Fixed-size or Dynamic-size, do not expand dynamically in the sense of + * a std::map. If you want this behavior, see the Sparse module.
+ * + *
\anchor maxrows _MaxRows and _MaxCols:
+ *
In most cases, one just leaves these parameters to the default values. + * These parameters mean the maximum size of rows and columns that the matrix may have. They are useful in cases + * when the exact numbers of rows and columns are not known are compile-time, but it is known at compile-time that they + * cannot exceed a certain value. This happens when taking dynamic-size blocks inside fixed-size matrices: in this case + * _MaxRows and _MaxCols are the dimensions of the original matrix, while _Rows and _Cols are Dynamic.
+ *
+ * + * ABI and storage layout + * + * The table below summarizes the ABI of some possible Matrix instances which is fixed thorough the lifetime of Eigen 3. + * + * + * + * + * + * + *
Matrix typeEquivalent C structure
\code Matrix \endcode\code + * struct { + * T *data; // with (size_t(data)%EIGEN_MAX_ALIGN_BYTES)==0 + * Eigen::Index rows, cols; + * }; + * \endcode
\code + * Matrix + * Matrix \endcode\code + * struct { + * T *data; // with (size_t(data)%EIGEN_MAX_ALIGN_BYTES)==0 + * Eigen::Index size; + * }; + * \endcode
\code Matrix \endcode\code + * struct { + * T data[Rows*Cols]; // with (size_t(data)%A(Rows*Cols*sizeof(T)))==0 + * }; + * \endcode
\code Matrix \endcode\code + * struct { + * T data[MaxRows*MaxCols]; // with (size_t(data)%A(MaxRows*MaxCols*sizeof(T)))==0 + * Eigen::Index rows, cols; + * }; + * \endcode
+ * Note that in this table Rows, Cols, MaxRows and MaxCols are all positive integers. A(S) is defined to the largest + * possible power-of-two smaller to EIGEN_MAX_STATIC_ALIGN_BYTES. + * + * \see MatrixBase for the majority of the API methods for matrices, \ref TopicClassHierarchy, + * \ref TopicStorageOrders + */ template -class Matrix - : public PlainObjectBase > +class Matrix : public PlainObjectBase> { - public: - - /** \brief Base class typedef. - * \sa PlainObjectBase - */ - typedef PlainObjectBase Base; - - enum { Options = _Options }; - - EIGEN_DENSE_PUBLIC_INTERFACE(Matrix) - - typedef typename Base::PlainObject PlainObject; - - using Base::base; - using Base::coeffRef; - - /** - * \brief Assigns matrices to each other. - * - * \note This is a special case of the templated operator=. Its purpose is - * to prevent a default operator= from hiding the templated operator=. - * - * \callgraph - */ - EIGEN_DEVICE_FUNC - EIGEN_STRONG_INLINE Matrix& operator=(const Matrix& other) - { - return Base::_set(other); - } - - /** \internal - * \brief Copies the value of the expression \a other into \c *this with automatic resizing. - * - * *this might be resized to match the dimensions of \a other. If *this was a null matrix (not already initialized), - * it will be initialized. - * - * Note that copying a row-vector into a vector (and conversely) is allowed. - * The resizing, if any, is then done in the appropriate way so that row-vectors - * remain row-vectors and vectors remain vectors. - */ - template - EIGEN_DEVICE_FUNC - EIGEN_STRONG_INLINE Matrix& operator=(const DenseBase& other) - { - return Base::_set(other); - } - - /* Here, doxygen failed to copy the brief information when using \copydoc */ - - /** - * \brief Copies the generic expression \a other into *this. - * \copydetails DenseBase::operator=(const EigenBase &other) - */ - template - EIGEN_DEVICE_FUNC - EIGEN_STRONG_INLINE Matrix& operator=(const EigenBase &other) - { - return Base::operator=(other); - } - - template - EIGEN_DEVICE_FUNC - EIGEN_STRONG_INLINE Matrix& operator=(const ReturnByValue& func) - { - return Base::operator=(func); - } - - /** \brief Default constructor. - * - * For fixed-size matrices, does nothing. - * - * For dynamic-size matrices, creates an empty matrix of size 0. Does not allocate any array. Such a matrix - * is called a null matrix. This constructor is the unique way to create null matrices: resizing - * a matrix to 0 is not supported. - * - * \sa resize(Index,Index) - */ - EIGEN_DEVICE_FUNC - EIGEN_STRONG_INLINE Matrix() : Base() - { - Base::_check_template_params(); - EIGEN_INITIALIZE_COEFFS_IF_THAT_OPTION_IS_ENABLED - } - - // FIXME is it still needed - EIGEN_DEVICE_FUNC - explicit Matrix(internal::constructor_without_unaligned_array_assert) - : Base(internal::constructor_without_unaligned_array_assert()) - { Base::_check_template_params(); EIGEN_INITIALIZE_COEFFS_IF_THAT_OPTION_IS_ENABLED } +public: + /** \brief Base class typedef. + * \sa PlainObjectBase + */ + typedef PlainObjectBase Base; + + enum { Options = _Options }; + + EIGEN_DENSE_PUBLIC_INTERFACE(Matrix) + + typedef typename Base::PlainObject PlainObject; + + using Base::base; + using Base::coeffRef; + + /** + * \brief Assigns matrices to each other. + * + * \note This is a special case of the templated operator=. Its purpose is + * to prevent a default operator= from hiding the templated operator=. + * + * \callgraph + */ + EIGEN_DEVICE_FUNC + EIGEN_STRONG_INLINE Matrix &operator=(const Matrix &other) { return Base::_set(other); } + + /** \internal + * \brief Copies the value of the expression \a other into \c *this with automatic resizing. + * + * *this might be resized to match the dimensions of \a other. If *this was a null matrix (not already initialized), + * it will be initialized. + * + * Note that copying a row-vector into a vector (and conversely) is allowed. + * The resizing, if any, is then done in the appropriate way so that row-vectors + * remain row-vectors and vectors remain vectors. + */ + template + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Matrix &operator=(const DenseBase &other) + { + return Base::_set(other); + } + + /* Here, doxygen failed to copy the brief information when using \copydoc */ + + /** + * \brief Copies the generic expression \a other into *this. + * \copydetails DenseBase::operator=(const EigenBase &other) + */ + template + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Matrix &operator=(const EigenBase &other) + { + return Base::operator=(other); + } + + template + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Matrix &operator=(const ReturnByValue &func) + { + return Base::operator=(func); + } + + /** \brief Default constructor. + * + * For fixed-size matrices, does nothing. + * + * For dynamic-size matrices, creates an empty matrix of size 0. Does not allocate any array. Such a matrix + * is called a null matrix. This constructor is the unique way to create null matrices: resizing + * a matrix to 0 is not supported. + * + * \sa resize(Index,Index) + */ + EIGEN_DEVICE_FUNC + EIGEN_STRONG_INLINE Matrix() : Base() + { + Base::_check_template_params(); + EIGEN_INITIALIZE_COEFFS_IF_THAT_OPTION_IS_ENABLED + } + + // FIXME is it still needed + EIGEN_DEVICE_FUNC + explicit Matrix(internal::constructor_without_unaligned_array_assert) + : Base(internal::constructor_without_unaligned_array_assert()) + { + Base::_check_template_params(); + EIGEN_INITIALIZE_COEFFS_IF_THAT_OPTION_IS_ENABLED + } #if EIGEN_HAS_RVALUE_REFERENCES - EIGEN_DEVICE_FUNC - Matrix(Matrix&& other) EIGEN_NOEXCEPT_IF(std::is_nothrow_move_constructible::value) - : Base(std::move(other)) - { - Base::_check_template_params(); - } - EIGEN_DEVICE_FUNC - Matrix& operator=(Matrix&& other) EIGEN_NOEXCEPT_IF(std::is_nothrow_move_assignable::value) - { - other.swap(*this); - return *this; - } + EIGEN_DEVICE_FUNC + Matrix(Matrix &&other) EIGEN_NOEXCEPT_IF(std::is_nothrow_move_constructible::value) : Base(std::move(other)) + { + Base::_check_template_params(); + } + EIGEN_DEVICE_FUNC + Matrix &operator=(Matrix &&other) EIGEN_NOEXCEPT_IF(std::is_nothrow_move_assignable::value) + { + other.swap(*this); + return *this; + } +#endif + +#ifndef EIGEN_PARSED_BY_DOXYGEN + + // This constructor is for both 1x1 matrices and dynamic vectors + template EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE explicit Matrix(const T &x) + { + Base::_check_template_params(); + Base::template _init1(x); + } + + template EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Matrix(const T0 &x, const T1 &y) + { + Base::_check_template_params(); + Base::template _init2(x, y); + } +#else + /** \brief Constructs a fixed-sized matrix initialized with coefficients starting at \a data */ + EIGEN_DEVICE_FUNC + explicit Matrix(const Scalar *data); + + /** \brief Constructs a vector or row-vector with given dimension. \only_for_vectors + * + * This is useful for dynamic-size vectors. For fixed-size vectors, + * it is redundant to pass these parameters, so one should use the default constructor + * Matrix() instead. + * + * \warning This constructor is disabled for fixed-size \c 1x1 matrices. For instance, + * calling Matrix(1) will call the initialization constructor: Matrix(const Scalar&). + * For fixed-size \c 1x1 matrices it is therefore recommended to use the default + * constructor Matrix() instead, especially when using one of the non standard + * \c EIGEN_INITIALIZE_MATRICES_BY_{ZERO,\c NAN} macros (see \ref TopicPreprocessorDirectives). + */ + EIGEN_STRONG_INLINE explicit Matrix(Index dim); + /** \brief Constructs an initialized 1x1 matrix with the given coefficient */ + Matrix(const Scalar &x); + /** \brief Constructs an uninitialized matrix with \a rows rows and \a cols columns. + * + * This is useful for dynamic-size matrices. For fixed-size matrices, + * it is redundant to pass these parameters, so one should use the default constructor + * Matrix() instead. + * + * \warning This constructor is disabled for fixed-size \c 1x2 and \c 2x1 vectors. For instance, + * calling Matrix2f(2,1) will call the initialization constructor: Matrix(const Scalar& x, const Scalar& y). + * For fixed-size \c 1x2 or \c 2x1 vectors it is therefore recommended to use the default + * constructor Matrix() instead, especially when using one of the non standard + * \c EIGEN_INITIALIZE_MATRICES_BY_{ZERO,\c NAN} macros (see \ref TopicPreprocessorDirectives). + */ + EIGEN_DEVICE_FUNC + Matrix(Index rows, Index cols); + + /** \brief Constructs an initialized 2D vector with given coefficients */ + Matrix(const Scalar &x, const Scalar &y); #endif - #ifndef EIGEN_PARSED_BY_DOXYGEN - - // This constructor is for both 1x1 matrices and dynamic vectors - template - EIGEN_DEVICE_FUNC - EIGEN_STRONG_INLINE explicit Matrix(const T& x) - { - Base::_check_template_params(); - Base::template _init1(x); - } - - template - EIGEN_DEVICE_FUNC - EIGEN_STRONG_INLINE Matrix(const T0& x, const T1& y) - { - Base::_check_template_params(); - Base::template _init2(x, y); - } - #else - /** \brief Constructs a fixed-sized matrix initialized with coefficients starting at \a data */ - EIGEN_DEVICE_FUNC - explicit Matrix(const Scalar *data); - - /** \brief Constructs a vector or row-vector with given dimension. \only_for_vectors - * - * This is useful for dynamic-size vectors. For fixed-size vectors, - * it is redundant to pass these parameters, so one should use the default constructor - * Matrix() instead. - * - * \warning This constructor is disabled for fixed-size \c 1x1 matrices. For instance, - * calling Matrix(1) will call the initialization constructor: Matrix(const Scalar&). - * For fixed-size \c 1x1 matrices it is therefore recommended to use the default - * constructor Matrix() instead, especially when using one of the non standard - * \c EIGEN_INITIALIZE_MATRICES_BY_{ZERO,\c NAN} macros (see \ref TopicPreprocessorDirectives). - */ - EIGEN_STRONG_INLINE explicit Matrix(Index dim); - /** \brief Constructs an initialized 1x1 matrix with the given coefficient */ - Matrix(const Scalar& x); - /** \brief Constructs an uninitialized matrix with \a rows rows and \a cols columns. - * - * This is useful for dynamic-size matrices. For fixed-size matrices, - * it is redundant to pass these parameters, so one should use the default constructor - * Matrix() instead. - * - * \warning This constructor is disabled for fixed-size \c 1x2 and \c 2x1 vectors. For instance, - * calling Matrix2f(2,1) will call the initialization constructor: Matrix(const Scalar& x, const Scalar& y). - * For fixed-size \c 1x2 or \c 2x1 vectors it is therefore recommended to use the default - * constructor Matrix() instead, especially when using one of the non standard - * \c EIGEN_INITIALIZE_MATRICES_BY_{ZERO,\c NAN} macros (see \ref TopicPreprocessorDirectives). - */ - EIGEN_DEVICE_FUNC - Matrix(Index rows, Index cols); - - /** \brief Constructs an initialized 2D vector with given coefficients */ - Matrix(const Scalar& x, const Scalar& y); - #endif - - /** \brief Constructs an initialized 3D vector with given coefficients */ - EIGEN_DEVICE_FUNC - EIGEN_STRONG_INLINE Matrix(const Scalar& x, const Scalar& y, const Scalar& z) - { - Base::_check_template_params(); - EIGEN_STATIC_ASSERT_VECTOR_SPECIFIC_SIZE(Matrix, 3) - m_storage.data()[0] = x; - m_storage.data()[1] = y; - m_storage.data()[2] = z; - } - /** \brief Constructs an initialized 4D vector with given coefficients */ - EIGEN_DEVICE_FUNC - EIGEN_STRONG_INLINE Matrix(const Scalar& x, const Scalar& y, const Scalar& z, const Scalar& w) - { - Base::_check_template_params(); - EIGEN_STATIC_ASSERT_VECTOR_SPECIFIC_SIZE(Matrix, 4) - m_storage.data()[0] = x; - m_storage.data()[1] = y; - m_storage.data()[2] = z; - m_storage.data()[3] = w; - } - - - /** \brief Copy constructor */ - EIGEN_DEVICE_FUNC - EIGEN_STRONG_INLINE Matrix(const Matrix& other) : Base(other) - { } - - /** \brief Copy constructor for generic expressions. - * \sa MatrixBase::operator=(const EigenBase&) - */ - template - EIGEN_DEVICE_FUNC - EIGEN_STRONG_INLINE Matrix(const EigenBase &other) - : Base(other.derived()) - { } - - EIGEN_DEVICE_FUNC inline Index innerStride() const { return 1; } - EIGEN_DEVICE_FUNC inline Index outerStride() const { return this->innerSize(); } - - /////////// Geometry module /////////// - - template - EIGEN_DEVICE_FUNC - explicit Matrix(const RotationBase& r); - template - EIGEN_DEVICE_FUNC - Matrix& operator=(const RotationBase& r); - - // allow to extend Matrix outside Eigen - #ifdef EIGEN_MATRIX_PLUGIN - #include EIGEN_MATRIX_PLUGIN - #endif - - protected: - template - friend struct internal::conservative_resize_like_impl; - - using Base::m_storage; + /** \brief Constructs an initialized 3D vector with given coefficients */ + EIGEN_DEVICE_FUNC + EIGEN_STRONG_INLINE Matrix(const Scalar &x, const Scalar &y, const Scalar &z) + { + Base::_check_template_params(); + EIGEN_STATIC_ASSERT_VECTOR_SPECIFIC_SIZE(Matrix, 3) + m_storage.data()[0] = x; + m_storage.data()[1] = y; + m_storage.data()[2] = z; + } + /** \brief Constructs an initialized 4D vector with given coefficients */ + EIGEN_DEVICE_FUNC + EIGEN_STRONG_INLINE Matrix(const Scalar &x, const Scalar &y, const Scalar &z, const Scalar &w) + { + Base::_check_template_params(); + EIGEN_STATIC_ASSERT_VECTOR_SPECIFIC_SIZE(Matrix, 4) + m_storage.data()[0] = x; + m_storage.data()[1] = y; + m_storage.data()[2] = z; + m_storage.data()[3] = w; + } + + + /** \brief Copy constructor */ + EIGEN_DEVICE_FUNC + EIGEN_STRONG_INLINE Matrix(const Matrix &other) : Base(other) {} + + /** \brief Copy constructor for generic expressions. + * \sa MatrixBase::operator=(const EigenBase&) + */ + template + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Matrix(const EigenBase &other) : Base(other.derived()) + {} + + EIGEN_DEVICE_FUNC inline Index innerStride() const { return 1; } + EIGEN_DEVICE_FUNC inline Index outerStride() const { return this->innerSize(); } + + /////////// Geometry module /////////// + + template + EIGEN_DEVICE_FUNC explicit Matrix(const RotationBase &r); + template + EIGEN_DEVICE_FUNC Matrix &operator=(const RotationBase &r); + +// allow to extend Matrix outside Eigen +#ifdef EIGEN_MATRIX_PLUGIN +#include EIGEN_MATRIX_PLUGIN +#endif + +protected: + template + friend struct internal::conservative_resize_like_impl; + + using Base::m_storage; }; /** \defgroup matrixtypedefs Global matrix typedefs - * - * \ingroup Core_Module - * - * Eigen defines several typedef shortcuts for most common matrix and vector types. - * - * The general patterns are the following: - * - * \c MatrixSizeType where \c Size can be \c 2,\c 3,\c 4 for fixed size square matrices or \c X for dynamic size, - * and where \c Type can be \c i for integer, \c f for float, \c d for double, \c cf for complex float, \c cd - * for complex double. - * - * For example, \c Matrix3d is a fixed-size 3x3 matrix type of doubles, and \c MatrixXf is a dynamic-size matrix of floats. - * - * There are also \c VectorSizeType and \c RowVectorSizeType which are self-explanatory. For example, \c Vector4cf is - * a fixed-size vector of 4 complex floats. - * - * \sa class Matrix - */ - -#define EIGEN_MAKE_TYPEDEFS(Type, TypeSuffix, Size, SizeSuffix) \ -/** \ingroup matrixtypedefs */ \ -typedef Matrix Matrix##SizeSuffix##TypeSuffix; \ -/** \ingroup matrixtypedefs */ \ -typedef Matrix Vector##SizeSuffix##TypeSuffix; \ -/** \ingroup matrixtypedefs */ \ -typedef Matrix RowVector##SizeSuffix##TypeSuffix; - -#define EIGEN_MAKE_FIXED_TYPEDEFS(Type, TypeSuffix, Size) \ -/** \ingroup matrixtypedefs */ \ -typedef Matrix Matrix##Size##X##TypeSuffix; \ -/** \ingroup matrixtypedefs */ \ -typedef Matrix Matrix##X##Size##TypeSuffix; + * + * \ingroup Core_Module + * + * Eigen defines several typedef shortcuts for most common matrix and vector types. + * + * The general patterns are the following: + * + * \c MatrixSizeType where \c Size can be \c 2,\c 3,\c 4 for fixed size square matrices or \c X for dynamic size, + * and where \c Type can be \c i for integer, \c f for float, \c d for double, \c cf for complex float, \c cd + * for complex double. + * + * For example, \c Matrix3d is a fixed-size 3x3 matrix type of doubles, and \c MatrixXf is a dynamic-size matrix of + * floats. + * + * There are also \c VectorSizeType and \c RowVectorSizeType which are self-explanatory. For example, \c Vector4cf is + * a fixed-size vector of 4 complex floats. + * + * \sa class Matrix + */ + +#define EIGEN_MAKE_TYPEDEFS(Type, TypeSuffix, Size, SizeSuffix) \ + /** \ingroup matrixtypedefs */ \ + typedef Matrix Matrix##SizeSuffix##TypeSuffix; \ + /** \ingroup matrixtypedefs */ \ + typedef Matrix Vector##SizeSuffix##TypeSuffix; \ + /** \ingroup matrixtypedefs */ \ + typedef Matrix RowVector##SizeSuffix##TypeSuffix; + +#define EIGEN_MAKE_FIXED_TYPEDEFS(Type, TypeSuffix, Size) \ + /** \ingroup matrixtypedefs */ \ + typedef Matrix Matrix##Size##X##TypeSuffix; \ + /** \ingroup matrixtypedefs */ \ + typedef Matrix Matrix##X##Size##TypeSuffix; #define EIGEN_MAKE_TYPEDEFS_ALL_SIZES(Type, TypeSuffix) \ -EIGEN_MAKE_TYPEDEFS(Type, TypeSuffix, 2, 2) \ -EIGEN_MAKE_TYPEDEFS(Type, TypeSuffix, 3, 3) \ -EIGEN_MAKE_TYPEDEFS(Type, TypeSuffix, 4, 4) \ -EIGEN_MAKE_TYPEDEFS(Type, TypeSuffix, Dynamic, X) \ -EIGEN_MAKE_FIXED_TYPEDEFS(Type, TypeSuffix, 2) \ -EIGEN_MAKE_FIXED_TYPEDEFS(Type, TypeSuffix, 3) \ -EIGEN_MAKE_FIXED_TYPEDEFS(Type, TypeSuffix, 4) - -EIGEN_MAKE_TYPEDEFS_ALL_SIZES(int, i) -EIGEN_MAKE_TYPEDEFS_ALL_SIZES(float, f) -EIGEN_MAKE_TYPEDEFS_ALL_SIZES(double, d) -EIGEN_MAKE_TYPEDEFS_ALL_SIZES(std::complex, cf) + EIGEN_MAKE_TYPEDEFS(Type, TypeSuffix, 2, 2) \ + EIGEN_MAKE_TYPEDEFS(Type, TypeSuffix, 3, 3) \ + EIGEN_MAKE_TYPEDEFS(Type, TypeSuffix, 4, 4) \ + EIGEN_MAKE_TYPEDEFS(Type, TypeSuffix, Dynamic, X) \ + EIGEN_MAKE_FIXED_TYPEDEFS(Type, TypeSuffix, 2) \ + EIGEN_MAKE_FIXED_TYPEDEFS(Type, TypeSuffix, 3) \ + EIGEN_MAKE_FIXED_TYPEDEFS(Type, TypeSuffix, 4) + +EIGEN_MAKE_TYPEDEFS_ALL_SIZES(int, i) +EIGEN_MAKE_TYPEDEFS_ALL_SIZES(float, f) +EIGEN_MAKE_TYPEDEFS_ALL_SIZES(double, d) +EIGEN_MAKE_TYPEDEFS_ALL_SIZES(std::complex, cf) EIGEN_MAKE_TYPEDEFS_ALL_SIZES(std::complex, cd) #undef EIGEN_MAKE_TYPEDEFS_ALL_SIZES #undef EIGEN_MAKE_TYPEDEFS #undef EIGEN_MAKE_FIXED_TYPEDEFS -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_MATRIX_H +#endif// EIGEN_MATRIX_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/MatrixBase.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/MatrixBase.h index e6c35907..d1d5be5f 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/MatrixBase.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/MatrixBase.h @@ -45,466 +45,471 @@ namespace Eigen { * * \sa \blank \ref TopicClassHierarchy */ -template class MatrixBase - : public DenseBase +template class MatrixBase : public DenseBase { - public: +public: #ifndef EIGEN_PARSED_BY_DOXYGEN - typedef MatrixBase StorageBaseType; - typedef typename internal::traits::StorageKind StorageKind; - typedef typename internal::traits::StorageIndex StorageIndex; - typedef typename internal::traits::Scalar Scalar; - typedef typename internal::packet_traits::type PacketScalar; - typedef typename NumTraits::Real RealScalar; - - typedef DenseBase Base; - using Base::RowsAtCompileTime; - using Base::ColsAtCompileTime; - using Base::SizeAtCompileTime; - using Base::MaxRowsAtCompileTime; - using Base::MaxColsAtCompileTime; - using Base::MaxSizeAtCompileTime; - using Base::IsVectorAtCompileTime; - using Base::Flags; - - using Base::derived; - using Base::const_cast_derived; - using Base::rows; - using Base::cols; - using Base::size; - using Base::coeff; - using Base::coeffRef; - using Base::lazyAssign; - using Base::eval; - using Base::operator+=; - using Base::operator-=; - using Base::operator*=; - using Base::operator/=; - - typedef typename Base::CoeffReturnType CoeffReturnType; - typedef typename Base::ConstTransposeReturnType ConstTransposeReturnType; - typedef typename Base::RowXpr RowXpr; - typedef typename Base::ColXpr ColXpr; -#endif // not EIGEN_PARSED_BY_DOXYGEN - + typedef MatrixBase StorageBaseType; + typedef typename internal::traits::StorageKind StorageKind; + typedef typename internal::traits::StorageIndex StorageIndex; + typedef typename internal::traits::Scalar Scalar; + typedef typename internal::packet_traits::type PacketScalar; + typedef typename NumTraits::Real RealScalar; + + typedef DenseBase Base; + using Base::RowsAtCompileTime; + using Base::ColsAtCompileTime; + using Base::SizeAtCompileTime; + using Base::MaxRowsAtCompileTime; + using Base::MaxColsAtCompileTime; + using Base::MaxSizeAtCompileTime; + using Base::IsVectorAtCompileTime; + using Base::Flags; + + using Base::derived; + using Base::const_cast_derived; + using Base::rows; + using Base::cols; + using Base::size; + using Base::coeff; + using Base::coeffRef; + using Base::lazyAssign; + using Base::eval; + using Base::operator+=; + using Base::operator-=; + using Base::operator*=; + using Base::operator/=; + + typedef typename Base::CoeffReturnType CoeffReturnType; + typedef typename Base::ConstTransposeReturnType ConstTransposeReturnType; + typedef typename Base::RowXpr RowXpr; + typedef typename Base::ColXpr ColXpr; +#endif// not EIGEN_PARSED_BY_DOXYGEN #ifndef EIGEN_PARSED_BY_DOXYGEN - /** type of the equivalent square matrix */ - typedef Matrix SquareMatrixType; -#endif // not EIGEN_PARSED_BY_DOXYGEN + /** type of the equivalent square matrix */ + typedef Matrix + SquareMatrixType; +#endif// not EIGEN_PARSED_BY_DOXYGEN - /** \returns the size of the main diagonal, which is min(rows(),cols()). - * \sa rows(), cols(), SizeAtCompileTime. */ - EIGEN_DEVICE_FUNC - inline Index diagonalSize() const { return (numext::mini)(rows(),cols()); } + /** \returns the size of the main diagonal, which is min(rows(),cols()). + * \sa rows(), cols(), SizeAtCompileTime. */ + EIGEN_DEVICE_FUNC + inline Index diagonalSize() const { return (numext::mini)(rows(), cols()); } - typedef typename Base::PlainObject PlainObject; + typedef typename Base::PlainObject PlainObject; #ifndef EIGEN_PARSED_BY_DOXYGEN - /** \internal Represents a matrix with all coefficients equal to one another*/ - typedef CwiseNullaryOp,PlainObject> ConstantReturnType; - /** \internal the return type of MatrixBase::adjoint() */ - typedef typename internal::conditional::IsComplex, - CwiseUnaryOp, ConstTransposeReturnType>, - ConstTransposeReturnType - >::type AdjointReturnType; - /** \internal Return type of eigenvalues() */ - typedef Matrix, internal::traits::ColsAtCompileTime, 1, ColMajor> EigenvaluesReturnType; - /** \internal the return type of identity */ - typedef CwiseNullaryOp,PlainObject> IdentityReturnType; - /** \internal the return type of unit vectors */ - typedef Block, SquareMatrixType>, - internal::traits::RowsAtCompileTime, - internal::traits::ColsAtCompileTime> BasisReturnType; -#endif // not EIGEN_PARSED_BY_DOXYGEN + /** \internal Represents a matrix with all coefficients equal to one another*/ + typedef CwiseNullaryOp, PlainObject> ConstantReturnType; + /** \internal the return type of MatrixBase::adjoint() */ + typedef typename internal::conditional::IsComplex, + CwiseUnaryOp, ConstTransposeReturnType>, + ConstTransposeReturnType>::type AdjointReturnType; + /** \internal Return type of eigenvalues() */ + typedef Matrix, internal::traits::ColsAtCompileTime, 1, ColMajor> + EigenvaluesReturnType; + /** \internal the return type of identity */ + typedef CwiseNullaryOp, PlainObject> IdentityReturnType; + /** \internal the return type of unit vectors */ + typedef Block, SquareMatrixType>, + internal::traits::RowsAtCompileTime, + internal::traits::ColsAtCompileTime> + BasisReturnType; +#endif// not EIGEN_PARSED_BY_DOXYGEN #define EIGEN_CURRENT_STORAGE_BASE_CLASS Eigen::MatrixBase -#define EIGEN_DOC_UNARY_ADDONS(X,Y) -# include "../plugins/CommonCwiseUnaryOps.h" -# include "../plugins/CommonCwiseBinaryOps.h" -# include "../plugins/MatrixCwiseUnaryOps.h" -# include "../plugins/MatrixCwiseBinaryOps.h" -# ifdef EIGEN_MATRIXBASE_PLUGIN -# include EIGEN_MATRIXBASE_PLUGIN -# endif +#define EIGEN_DOC_UNARY_ADDONS(X, Y) +#include "../plugins/CommonCwiseBinaryOps.h" +#include "../plugins/CommonCwiseUnaryOps.h" +#include "../plugins/MatrixCwiseBinaryOps.h" +#include "../plugins/MatrixCwiseUnaryOps.h" +#ifdef EIGEN_MATRIXBASE_PLUGIN +#include EIGEN_MATRIXBASE_PLUGIN +#endif #undef EIGEN_CURRENT_STORAGE_BASE_CLASS #undef EIGEN_DOC_UNARY_ADDONS - /** Special case of the template operator=, in order to prevent the compiler - * from generating a default operator= (issue hit with g++ 4.1) - */ - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE - Derived& operator=(const MatrixBase& other); - - // We cannot inherit here via Base::operator= since it is causing - // trouble with MSVC. - - template - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE - Derived& operator=(const DenseBase& other); - - template - EIGEN_DEVICE_FUNC - Derived& operator=(const EigenBase& other); - - template - EIGEN_DEVICE_FUNC - Derived& operator=(const ReturnByValue& other); - - template - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE - Derived& operator+=(const MatrixBase& other); - template - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE - Derived& operator-=(const MatrixBase& other); - - template - EIGEN_DEVICE_FUNC - const Product - operator*(const MatrixBase &other) const; - - template - EIGEN_DEVICE_FUNC - const Product - lazyProduct(const MatrixBase &other) const; - - template - Derived& operator*=(const EigenBase& other); - - template - void applyOnTheLeft(const EigenBase& other); - - template - void applyOnTheRight(const EigenBase& other); - - template - EIGEN_DEVICE_FUNC - const Product - operator*(const DiagonalBase &diagonal) const; - - template - EIGEN_DEVICE_FUNC - typename ScalarBinaryOpTraits::Scalar,typename internal::traits::Scalar>::ReturnType - dot(const MatrixBase& other) const; - - EIGEN_DEVICE_FUNC RealScalar squaredNorm() const; - EIGEN_DEVICE_FUNC RealScalar norm() const; - RealScalar stableNorm() const; - RealScalar blueNorm() const; - RealScalar hypotNorm() const; - EIGEN_DEVICE_FUNC const PlainObject normalized() const; - EIGEN_DEVICE_FUNC const PlainObject stableNormalized() const; - EIGEN_DEVICE_FUNC void normalize(); - EIGEN_DEVICE_FUNC void stableNormalize(); - - EIGEN_DEVICE_FUNC const AdjointReturnType adjoint() const; - EIGEN_DEVICE_FUNC void adjointInPlace(); - - typedef Diagonal DiagonalReturnType; - EIGEN_DEVICE_FUNC - DiagonalReturnType diagonal(); - - typedef typename internal::add_const >::type ConstDiagonalReturnType; - EIGEN_DEVICE_FUNC - ConstDiagonalReturnType diagonal() const; - - template struct DiagonalIndexReturnType { typedef Diagonal Type; }; - template struct ConstDiagonalIndexReturnType { typedef const Diagonal Type; }; - - template - EIGEN_DEVICE_FUNC - typename DiagonalIndexReturnType::Type diagonal(); - - template - EIGEN_DEVICE_FUNC - typename ConstDiagonalIndexReturnType::Type diagonal() const; - - typedef Diagonal DiagonalDynamicIndexReturnType; - typedef typename internal::add_const >::type ConstDiagonalDynamicIndexReturnType; - - EIGEN_DEVICE_FUNC - DiagonalDynamicIndexReturnType diagonal(Index index); - EIGEN_DEVICE_FUNC - ConstDiagonalDynamicIndexReturnType diagonal(Index index) const; - - template struct TriangularViewReturnType { typedef TriangularView Type; }; - template struct ConstTriangularViewReturnType { typedef const TriangularView Type; }; - - template - EIGEN_DEVICE_FUNC - typename TriangularViewReturnType::Type triangularView(); - template - EIGEN_DEVICE_FUNC - typename ConstTriangularViewReturnType::Type triangularView() const; - - template struct SelfAdjointViewReturnType { typedef SelfAdjointView Type; }; - template struct ConstSelfAdjointViewReturnType { typedef const SelfAdjointView Type; }; - - template - EIGEN_DEVICE_FUNC - typename SelfAdjointViewReturnType::Type selfadjointView(); - template - EIGEN_DEVICE_FUNC - typename ConstSelfAdjointViewReturnType::Type selfadjointView() const; - - const SparseView sparseView(const Scalar& m_reference = Scalar(0), - const typename NumTraits::Real& m_epsilon = NumTraits::dummy_precision()) const; - EIGEN_DEVICE_FUNC static const IdentityReturnType Identity(); - EIGEN_DEVICE_FUNC static const IdentityReturnType Identity(Index rows, Index cols); - EIGEN_DEVICE_FUNC static const BasisReturnType Unit(Index size, Index i); - EIGEN_DEVICE_FUNC static const BasisReturnType Unit(Index i); - EIGEN_DEVICE_FUNC static const BasisReturnType UnitX(); - EIGEN_DEVICE_FUNC static const BasisReturnType UnitY(); - EIGEN_DEVICE_FUNC static const BasisReturnType UnitZ(); - EIGEN_DEVICE_FUNC static const BasisReturnType UnitW(); - - EIGEN_DEVICE_FUNC - const DiagonalWrapper asDiagonal() const; - const PermutationWrapper asPermutation() const; - - EIGEN_DEVICE_FUNC - Derived& setIdentity(); - EIGEN_DEVICE_FUNC - Derived& setIdentity(Index rows, Index cols); - - bool isIdentity(const RealScalar& prec = NumTraits::dummy_precision()) const; - bool isDiagonal(const RealScalar& prec = NumTraits::dummy_precision()) const; - - bool isUpperTriangular(const RealScalar& prec = NumTraits::dummy_precision()) const; - bool isLowerTriangular(const RealScalar& prec = NumTraits::dummy_precision()) const; - - template - bool isOrthogonal(const MatrixBase& other, - const RealScalar& prec = NumTraits::dummy_precision()) const; - bool isUnitary(const RealScalar& prec = NumTraits::dummy_precision()) const; - - /** \returns true if each coefficients of \c *this and \a other are all exactly equal. - * \warning When using floating point scalar values you probably should rather use a - * fuzzy comparison such as isApprox() - * \sa isApprox(), operator!= */ - template - EIGEN_DEVICE_FUNC inline bool operator==(const MatrixBase& other) const - { return cwiseEqual(other).all(); } - - /** \returns true if at least one pair of coefficients of \c *this and \a other are not exactly equal to each other. - * \warning When using floating point scalar values you probably should rather use a - * fuzzy comparison such as isApprox() - * \sa isApprox(), operator== */ - template - EIGEN_DEVICE_FUNC inline bool operator!=(const MatrixBase& other) const - { return cwiseNotEqual(other).any(); } - - NoAlias noalias(); - - // TODO forceAlignedAccess is temporarily disabled - // Need to find a nicer workaround. - inline const Derived& forceAlignedAccess() const { return derived(); } - inline Derived& forceAlignedAccess() { return derived(); } - template inline const Derived& forceAlignedAccessIf() const { return derived(); } - template inline Derived& forceAlignedAccessIf() { return derived(); } - - EIGEN_DEVICE_FUNC Scalar trace() const; - - template EIGEN_DEVICE_FUNC RealScalar lpNorm() const; - - EIGEN_DEVICE_FUNC MatrixBase& matrix() { return *this; } - EIGEN_DEVICE_FUNC const MatrixBase& matrix() const { return *this; } - - /** \returns an \link Eigen::ArrayBase Array \endlink expression of this matrix - * \sa ArrayBase::matrix() */ - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE ArrayWrapper array() { return ArrayWrapper(derived()); } - /** \returns a const \link Eigen::ArrayBase Array \endlink expression of this matrix - * \sa ArrayBase::matrix() */ - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const ArrayWrapper array() const { return ArrayWrapper(derived()); } - -/////////// LU module /////////// - - inline const FullPivLU fullPivLu() const; - inline const PartialPivLU partialPivLu() const; - - inline const PartialPivLU lu() const; - - inline const Inverse inverse() const; - - template - inline void computeInverseAndDetWithCheck( - ResultType& inverse, - typename ResultType::Scalar& determinant, - bool& invertible, - const RealScalar& absDeterminantThreshold = NumTraits::dummy_precision() - ) const; - template - inline void computeInverseWithCheck( - ResultType& inverse, - bool& invertible, - const RealScalar& absDeterminantThreshold = NumTraits::dummy_precision() - ) const; - Scalar determinant() const; - -/////////// Cholesky module /////////// - - inline const LLT llt() const; - inline const LDLT ldlt() const; - -/////////// QR module /////////// - - inline const HouseholderQR householderQr() const; - inline const ColPivHouseholderQR colPivHouseholderQr() const; - inline const FullPivHouseholderQR fullPivHouseholderQr() const; - inline const CompleteOrthogonalDecomposition completeOrthogonalDecomposition() const; - -/////////// Eigenvalues module /////////// - - inline EigenvaluesReturnType eigenvalues() const; - inline RealScalar operatorNorm() const; - -/////////// SVD module /////////// - - inline JacobiSVD jacobiSvd(unsigned int computationOptions = 0) const; - inline BDCSVD bdcSvd(unsigned int computationOptions = 0) const; - -/////////// Geometry module /////////// - - #ifndef EIGEN_PARSED_BY_DOXYGEN - /// \internal helper struct to form the return type of the cross product - template struct cross_product_return_type { - typedef typename ScalarBinaryOpTraits::Scalar,typename internal::traits::Scalar>::ReturnType Scalar; - typedef Matrix type; - }; - #endif // EIGEN_PARSED_BY_DOXYGEN - template - EIGEN_DEVICE_FUNC + /** Special case of the template operator=, in order to prevent the compiler + * from generating a default operator= (issue hit with g++ 4.1) + */ + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Derived &operator=(const MatrixBase &other); + + // We cannot inherit here via Base::operator= since it is causing + // trouble with MSVC. + + template + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Derived &operator=(const DenseBase &other); + + template EIGEN_DEVICE_FUNC Derived &operator=(const EigenBase &other); + + template EIGEN_DEVICE_FUNC Derived &operator=(const ReturnByValue &other); + + template + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Derived &operator+=(const MatrixBase &other); + template + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Derived &operator-=(const MatrixBase &other); + + template + EIGEN_DEVICE_FUNC const Product operator*(const MatrixBase &other) const; + + template + EIGEN_DEVICE_FUNC const Product lazyProduct( + const MatrixBase &other) const; + + template Derived &operator*=(const EigenBase &other); + + template void applyOnTheLeft(const EigenBase &other); + + template void applyOnTheRight(const EigenBase &other); + + template + EIGEN_DEVICE_FUNC const Product operator*( + const DiagonalBase &diagonal) const; + + template + EIGEN_DEVICE_FUNC typename ScalarBinaryOpTraits::Scalar, + typename internal::traits::Scalar>::ReturnType + dot(const MatrixBase &other) const; + + EIGEN_DEVICE_FUNC RealScalar squaredNorm() const; + EIGEN_DEVICE_FUNC RealScalar norm() const; + RealScalar stableNorm() const; + RealScalar blueNorm() const; + RealScalar hypotNorm() const; + EIGEN_DEVICE_FUNC const PlainObject normalized() const; + EIGEN_DEVICE_FUNC const PlainObject stableNormalized() const; + EIGEN_DEVICE_FUNC void normalize(); + EIGEN_DEVICE_FUNC void stableNormalize(); + + EIGEN_DEVICE_FUNC const AdjointReturnType adjoint() const; + EIGEN_DEVICE_FUNC void adjointInPlace(); + + typedef Diagonal DiagonalReturnType; + EIGEN_DEVICE_FUNC + DiagonalReturnType diagonal(); + + typedef typename internal::add_const>::type ConstDiagonalReturnType; + EIGEN_DEVICE_FUNC + ConstDiagonalReturnType diagonal() const; + + template struct DiagonalIndexReturnType + { + typedef Diagonal Type; + }; + template struct ConstDiagonalIndexReturnType + { + typedef const Diagonal Type; + }; + + template EIGEN_DEVICE_FUNC typename DiagonalIndexReturnType::Type diagonal(); + + template EIGEN_DEVICE_FUNC typename ConstDiagonalIndexReturnType::Type diagonal() const; + + typedef Diagonal DiagonalDynamicIndexReturnType; + typedef typename internal::add_const>::type ConstDiagonalDynamicIndexReturnType; + + EIGEN_DEVICE_FUNC + DiagonalDynamicIndexReturnType diagonal(Index index); + EIGEN_DEVICE_FUNC + ConstDiagonalDynamicIndexReturnType diagonal(Index index) const; + + template struct TriangularViewReturnType + { + typedef TriangularView Type; + }; + template struct ConstTriangularViewReturnType + { + typedef const TriangularView Type; + }; + + template EIGEN_DEVICE_FUNC typename TriangularViewReturnType::Type triangularView(); + template + EIGEN_DEVICE_FUNC typename ConstTriangularViewReturnType::Type triangularView() const; + + template struct SelfAdjointViewReturnType + { + typedef SelfAdjointView Type; + }; + template struct ConstSelfAdjointViewReturnType + { + typedef const SelfAdjointView Type; + }; + + template EIGEN_DEVICE_FUNC typename SelfAdjointViewReturnType::Type selfadjointView(); + template + EIGEN_DEVICE_FUNC typename ConstSelfAdjointViewReturnType::Type selfadjointView() const; + + const SparseView sparseView(const Scalar &m_reference = Scalar(0), + const typename NumTraits::Real &m_epsilon = NumTraits::dummy_precision()) const; + EIGEN_DEVICE_FUNC static const IdentityReturnType Identity(); + EIGEN_DEVICE_FUNC static const IdentityReturnType Identity(Index rows, Index cols); + EIGEN_DEVICE_FUNC static const BasisReturnType Unit(Index size, Index i); + EIGEN_DEVICE_FUNC static const BasisReturnType Unit(Index i); + EIGEN_DEVICE_FUNC static const BasisReturnType UnitX(); + EIGEN_DEVICE_FUNC static const BasisReturnType UnitY(); + EIGEN_DEVICE_FUNC static const BasisReturnType UnitZ(); + EIGEN_DEVICE_FUNC static const BasisReturnType UnitW(); + + EIGEN_DEVICE_FUNC + const DiagonalWrapper asDiagonal() const; + const PermutationWrapper asPermutation() const; + + EIGEN_DEVICE_FUNC + Derived &setIdentity(); + EIGEN_DEVICE_FUNC + Derived &setIdentity(Index rows, Index cols); + + bool isIdentity(const RealScalar &prec = NumTraits::dummy_precision()) const; + bool isDiagonal(const RealScalar &prec = NumTraits::dummy_precision()) const; + + bool isUpperTriangular(const RealScalar &prec = NumTraits::dummy_precision()) const; + bool isLowerTriangular(const RealScalar &prec = NumTraits::dummy_precision()) const; + + template + bool isOrthogonal(const MatrixBase &other, + const RealScalar &prec = NumTraits::dummy_precision()) const; + bool isUnitary(const RealScalar &prec = NumTraits::dummy_precision()) const; + + /** \returns true if each coefficients of \c *this and \a other are all exactly equal. + * \warning When using floating point scalar values you probably should rather use a + * fuzzy comparison such as isApprox() + * \sa isApprox(), operator!= */ + template EIGEN_DEVICE_FUNC inline bool operator==(const MatrixBase &other) const + { + return cwiseEqual(other).all(); + } + + /** \returns true if at least one pair of coefficients of \c *this and \a other are not exactly equal to each other. + * \warning When using floating point scalar values you probably should rather use a + * fuzzy comparison such as isApprox() + * \sa isApprox(), operator== */ + template EIGEN_DEVICE_FUNC inline bool operator!=(const MatrixBase &other) const + { + return cwiseNotEqual(other).any(); + } + + NoAlias noalias(); + + // TODO forceAlignedAccess is temporarily disabled + // Need to find a nicer workaround. + inline const Derived &forceAlignedAccess() const { return derived(); } + inline Derived &forceAlignedAccess() { return derived(); } + template inline const Derived &forceAlignedAccessIf() const { return derived(); } + template inline Derived &forceAlignedAccessIf() { return derived(); } + + EIGEN_DEVICE_FUNC Scalar trace() const; + + template EIGEN_DEVICE_FUNC RealScalar lpNorm() const; + + EIGEN_DEVICE_FUNC MatrixBase &matrix() { return *this; } + EIGEN_DEVICE_FUNC const MatrixBase &matrix() const { return *this; } + + /** \returns an \link Eigen::ArrayBase Array \endlink expression of this matrix + * \sa ArrayBase::matrix() */ + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE ArrayWrapper array() { return ArrayWrapper(derived()); } + /** \returns a const \link Eigen::ArrayBase Array \endlink expression of this matrix + * \sa ArrayBase::matrix() */ + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const ArrayWrapper array() const + { + return ArrayWrapper(derived()); + } + + /////////// LU module /////////// + + inline const FullPivLU fullPivLu() const; + inline const PartialPivLU partialPivLu() const; + + inline const PartialPivLU lu() const; + + inline const Inverse inverse() const; + + template + inline void computeInverseAndDetWithCheck(ResultType &inverse, + typename ResultType::Scalar &determinant, + bool &invertible, + const RealScalar &absDeterminantThreshold = NumTraits::dummy_precision()) const; + template + inline void computeInverseWithCheck(ResultType &inverse, + bool &invertible, + const RealScalar &absDeterminantThreshold = NumTraits::dummy_precision()) const; + Scalar determinant() const; + + /////////// Cholesky module /////////// + + inline const LLT llt() const; + inline const LDLT ldlt() const; + + /////////// QR module /////////// + + inline const HouseholderQR householderQr() const; + inline const ColPivHouseholderQR colPivHouseholderQr() const; + inline const FullPivHouseholderQR fullPivHouseholderQr() const; + inline const CompleteOrthogonalDecomposition completeOrthogonalDecomposition() const; + + /////////// Eigenvalues module /////////// + + inline EigenvaluesReturnType eigenvalues() const; + inline RealScalar operatorNorm() const; + + /////////// SVD module /////////// + + inline JacobiSVD jacobiSvd(unsigned int computationOptions = 0) const; + inline BDCSVD bdcSvd(unsigned int computationOptions = 0) const; + + /////////// Geometry module /////////// + +#ifndef EIGEN_PARSED_BY_DOXYGEN + /// \internal helper struct to form the return type of the cross product + template struct cross_product_return_type + { + typedef typename ScalarBinaryOpTraits::Scalar, + typename internal::traits::Scalar>::ReturnType Scalar; + typedef Matrix type; + }; +#endif// EIGEN_PARSED_BY_DOXYGEN + template + EIGEN_DEVICE_FUNC #ifndef EIGEN_PARSED_BY_DOXYGEN inline typename cross_product_return_type::type #else inline PlainObject #endif - cross(const MatrixBase& other) const; - - template - EIGEN_DEVICE_FUNC - inline PlainObject cross3(const MatrixBase& other) const; - - EIGEN_DEVICE_FUNC - inline PlainObject unitOrthogonal(void) const; - - EIGEN_DEVICE_FUNC - inline Matrix eulerAngles(Index a0, Index a1, Index a2) const; - - // put this as separate enum value to work around possible GCC 4.3 bug (?) - enum { HomogeneousReturnTypeDirection = ColsAtCompileTime==1&&RowsAtCompileTime==1 ? ((internal::traits::Flags&RowMajorBit)==RowMajorBit ? Horizontal : Vertical) - : ColsAtCompileTime==1 ? Vertical : Horizontal }; - typedef Homogeneous HomogeneousReturnType; - EIGEN_DEVICE_FUNC - inline HomogeneousReturnType homogeneous() const; - - enum { - SizeMinusOne = SizeAtCompileTime==Dynamic ? Dynamic : SizeAtCompileTime-1 - }; - typedef Block::ColsAtCompileTime==1 ? SizeMinusOne : 1, - internal::traits::ColsAtCompileTime==1 ? 1 : SizeMinusOne> ConstStartMinusOne; - typedef EIGEN_EXPR_BINARYOP_SCALAR_RETURN_TYPE(ConstStartMinusOne,Scalar,quotient) HNormalizedReturnType; - EIGEN_DEVICE_FUNC - inline const HNormalizedReturnType hnormalized() const; - -////////// Householder module /////////// - - void makeHouseholderInPlace(Scalar& tau, RealScalar& beta); - template - void makeHouseholder(EssentialPart& essential, - Scalar& tau, RealScalar& beta) const; - template - void applyHouseholderOnTheLeft(const EssentialPart& essential, - const Scalar& tau, - Scalar* workspace); - template - void applyHouseholderOnTheRight(const EssentialPart& essential, - const Scalar& tau, - Scalar* workspace); - -///////// Jacobi module ///////// - - template - void applyOnTheLeft(Index p, Index q, const JacobiRotation& j); - template - void applyOnTheRight(Index p, Index q, const JacobiRotation& j); - -///////// SparseCore module ///////// - - template - EIGEN_STRONG_INLINE const typename SparseMatrixBase::template CwiseProductDenseReturnType::Type + cross(const MatrixBase &other) const; + + template + EIGEN_DEVICE_FUNC inline PlainObject cross3(const MatrixBase &other) const; + + EIGEN_DEVICE_FUNC + inline PlainObject unitOrthogonal(void) const; + + EIGEN_DEVICE_FUNC + inline Matrix eulerAngles(Index a0, Index a1, Index a2) const; + + // put this as separate enum value to work around possible GCC 4.3 bug (?) + enum { + HomogeneousReturnTypeDirection = + ColsAtCompileTime == 1 && RowsAtCompileTime == 1 + ? ((internal::traits::Flags & RowMajorBit) == RowMajorBit ? Horizontal : Vertical) + : ColsAtCompileTime == 1 ? Vertical + : Horizontal + }; + typedef Homogeneous HomogeneousReturnType; + EIGEN_DEVICE_FUNC + inline HomogeneousReturnType homogeneous() const; + + enum { SizeMinusOne = SizeAtCompileTime == Dynamic ? Dynamic : SizeAtCompileTime - 1 }; + typedef Block::ColsAtCompileTime == 1 ? SizeMinusOne : 1, + internal::traits::ColsAtCompileTime == 1 ? 1 : SizeMinusOne> + ConstStartMinusOne; + typedef EIGEN_EXPR_BINARYOP_SCALAR_RETURN_TYPE(ConstStartMinusOne, Scalar, quotient) HNormalizedReturnType; + EIGEN_DEVICE_FUNC + inline const HNormalizedReturnType hnormalized() const; + + ////////// Householder module /////////// + + void makeHouseholderInPlace(Scalar &tau, RealScalar &beta); + template void makeHouseholder(EssentialPart &essential, Scalar &tau, RealScalar &beta) const; + template + void applyHouseholderOnTheLeft(const EssentialPart &essential, const Scalar &tau, Scalar *workspace); + template + void applyHouseholderOnTheRight(const EssentialPart &essential, const Scalar &tau, Scalar *workspace); + + ///////// Jacobi module ///////// + + template void applyOnTheLeft(Index p, Index q, const JacobiRotation &j); + template void applyOnTheRight(Index p, Index q, const JacobiRotation &j); + + ///////// SparseCore module ///////// + + template + EIGEN_STRONG_INLINE const typename SparseMatrixBase::template CwiseProductDenseReturnType::Type cwiseProduct(const SparseMatrixBase &other) const - { - return other.cwiseProduct(derived()); - } - -///////// MatrixFunctions module ///////// - - typedef typename internal::stem_function::type StemFunction; -#define EIGEN_MATRIX_FUNCTION(ReturnType, Name, Description) \ - /** \returns an expression of the matrix Description of \c *this. \brief This function requires the unsupported MatrixFunctions module. To compute the coefficient-wise Description use ArrayBase::##Name . */ \ - const ReturnType Name() const; -#define EIGEN_MATRIX_FUNCTION_1(ReturnType, Name, Description, Argument) \ - /** \returns an expression of the matrix Description of \c *this. \brief This function requires the unsupported MatrixFunctions module. To compute the coefficient-wise Description use ArrayBase::##Name . */ \ - const ReturnType Name(Argument) const; - - EIGEN_MATRIX_FUNCTION(MatrixExponentialReturnValue, exp, exponential) - /** \brief Helper function for the unsupported MatrixFunctions module.*/ - const MatrixFunctionReturnValue matrixFunction(StemFunction f) const; - EIGEN_MATRIX_FUNCTION(MatrixFunctionReturnValue, cosh, hyperbolic cosine) - EIGEN_MATRIX_FUNCTION(MatrixFunctionReturnValue, sinh, hyperbolic sine) - EIGEN_MATRIX_FUNCTION(MatrixFunctionReturnValue, cos, cosine) - EIGEN_MATRIX_FUNCTION(MatrixFunctionReturnValue, sin, sine) - EIGEN_MATRIX_FUNCTION(MatrixSquareRootReturnValue, sqrt, square root) - EIGEN_MATRIX_FUNCTION(MatrixLogarithmReturnValue, log, logarithm) - EIGEN_MATRIX_FUNCTION_1(MatrixPowerReturnValue, pow, power to \c p, const RealScalar& p) - EIGEN_MATRIX_FUNCTION_1(MatrixComplexPowerReturnValue, pow, power to \c p, const std::complex& p) - - protected: - EIGEN_DEVICE_FUNC MatrixBase() : Base() {} - - private: - EIGEN_DEVICE_FUNC explicit MatrixBase(int); - EIGEN_DEVICE_FUNC MatrixBase(int,int); - template EIGEN_DEVICE_FUNC explicit MatrixBase(const MatrixBase&); - protected: - // mixing arrays and matrices is not legal - template Derived& operator+=(const ArrayBase& ) - {EIGEN_STATIC_ASSERT(std::ptrdiff_t(sizeof(typename OtherDerived::Scalar))==-1,YOU_CANNOT_MIX_ARRAYS_AND_MATRICES); return *this;} - // mixing arrays and matrices is not legal - template Derived& operator-=(const ArrayBase& ) - {EIGEN_STATIC_ASSERT(std::ptrdiff_t(sizeof(typename OtherDerived::Scalar))==-1,YOU_CANNOT_MIX_ARRAYS_AND_MATRICES); return *this;} + { + return other.cwiseProduct(derived()); + } + + ///////// MatrixFunctions module ///////// + + typedef typename internal::stem_function::type StemFunction; +#define EIGEN_MATRIX_FUNCTION(ReturnType, Name, Description) \ + /** \returns an expression of the matrix Description of \c *this. \brief This function requires the unsupported MatrixFunctions module. To compute the \ + * coefficient-wise Description use ArrayBase::##Name . */ \ + const ReturnType Name() const; +#define EIGEN_MATRIX_FUNCTION_1(ReturnType, Name, Description, Argument) \ + /** \returns an expression of the matrix Description of \c *this. \brief This function requires the unsupported MatrixFunctions module. To compute the \ + * coefficient-wise Description use ArrayBase::##Name . */ \ + const ReturnType Name(Argument) const; + + EIGEN_MATRIX_FUNCTION(MatrixExponentialReturnValue, exp, exponential) + /** \brief Helper function for the unsupported + * MatrixFunctions module.*/ + const MatrixFunctionReturnValue matrixFunction(StemFunction f) const; + EIGEN_MATRIX_FUNCTION(MatrixFunctionReturnValue, cosh, hyperbolic cosine) + EIGEN_MATRIX_FUNCTION(MatrixFunctionReturnValue, sinh, hyperbolic sine) + EIGEN_MATRIX_FUNCTION(MatrixFunctionReturnValue, cos, cosine) + EIGEN_MATRIX_FUNCTION(MatrixFunctionReturnValue, sin, sine) + EIGEN_MATRIX_FUNCTION(MatrixSquareRootReturnValue, sqrt, square root) + EIGEN_MATRIX_FUNCTION(MatrixLogarithmReturnValue, log, logarithm) + EIGEN_MATRIX_FUNCTION_1(MatrixPowerReturnValue, pow, power to \c p, const RealScalar &p) + EIGEN_MATRIX_FUNCTION_1(MatrixComplexPowerReturnValue, pow, power to \c p, const std::complex &p) + +protected: + EIGEN_DEVICE_FUNC MatrixBase() : Base() {} + +private: + EIGEN_DEVICE_FUNC explicit MatrixBase(int); + EIGEN_DEVICE_FUNC MatrixBase(int, int); + template EIGEN_DEVICE_FUNC explicit MatrixBase(const MatrixBase &); + +protected: + // mixing arrays and matrices is not legal + template Derived &operator+=(const ArrayBase &) + { + EIGEN_STATIC_ASSERT( + std::ptrdiff_t(sizeof(typename OtherDerived::Scalar)) == -1, YOU_CANNOT_MIX_ARRAYS_AND_MATRICES); + return *this; + } + // mixing arrays and matrices is not legal + template Derived &operator-=(const ArrayBase &) + { + EIGEN_STATIC_ASSERT( + std::ptrdiff_t(sizeof(typename OtherDerived::Scalar)) == -1, YOU_CANNOT_MIX_ARRAYS_AND_MATRICES); + return *this; + } }; /*************************************************************************** -* Implementation of matrix base methods -***************************************************************************/ + * Implementation of matrix base methods + ***************************************************************************/ /** replaces \c *this by \c *this * \a other. - * - * \returns a reference to \c *this - * - * Example: \include MatrixBase_applyOnTheRight.cpp - * Output: \verbinclude MatrixBase_applyOnTheRight.out - */ + * + * \returns a reference to \c *this + * + * Example: \include MatrixBase_applyOnTheRight.cpp + * Output: \verbinclude MatrixBase_applyOnTheRight.out + */ template template -inline Derived& -MatrixBase::operator*=(const EigenBase &other) +inline Derived &MatrixBase::operator*=(const EigenBase &other) { other.derived().applyThisOnTheRight(derived()); return derived(); } /** replaces \c *this by \c *this * \a other. It is equivalent to MatrixBase::operator*=(). - * - * Example: \include MatrixBase_applyOnTheRight.cpp - * Output: \verbinclude MatrixBase_applyOnTheRight.out - */ + * + * Example: \include MatrixBase_applyOnTheRight.cpp + * Output: \verbinclude MatrixBase_applyOnTheRight.out + */ template template inline void MatrixBase::applyOnTheRight(const EigenBase &other) @@ -513,10 +518,10 @@ inline void MatrixBase::applyOnTheRight(const EigenBase & } /** replaces \c *this by \a other * \c *this. - * - * Example: \include MatrixBase_applyOnTheLeft.cpp - * Output: \verbinclude MatrixBase_applyOnTheLeft.out - */ + * + * Example: \include MatrixBase_applyOnTheLeft.cpp + * Output: \verbinclude MatrixBase_applyOnTheLeft.out + */ template template inline void MatrixBase::applyOnTheLeft(const EigenBase &other) @@ -524,6 +529,6 @@ inline void MatrixBase::applyOnTheLeft(const EigenBase &o other.derived().applyThisOnTheLeft(derived()); } -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_MATRIXBASE_H +#endif// EIGEN_MATRIXBASE_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/NestByValue.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/NestByValue.h index 13adf070..8a810dd4 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/NestByValue.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/NestByValue.h @@ -14,97 +14,83 @@ namespace Eigen { namespace internal { -template -struct traits > : public traits -{}; -} + template struct traits> : public traits + { + }; +}// namespace internal /** \class NestByValue - * \ingroup Core_Module - * - * \brief Expression which must be nested by value - * - * \tparam ExpressionType the type of the object of which we are requiring nesting-by-value - * - * This class is the return type of MatrixBase::nestByValue() - * and most of the time this is the only way it is used. - * - * \sa MatrixBase::nestByValue() - */ -template class NestByValue - : public internal::dense_xpr_base< NestByValue >::type + * \ingroup Core_Module + * + * \brief Expression which must be nested by value + * + * \tparam ExpressionType the type of the object of which we are requiring nesting-by-value + * + * This class is the return type of MatrixBase::nestByValue() + * and most of the time this is the only way it is used. + * + * \sa MatrixBase::nestByValue() + */ +template class NestByValue : public internal::dense_xpr_base>::type { - public: - - typedef typename internal::dense_xpr_base::type Base; - EIGEN_DENSE_PUBLIC_INTERFACE(NestByValue) - - EIGEN_DEVICE_FUNC explicit inline NestByValue(const ExpressionType& matrix) : m_expression(matrix) {} - - EIGEN_DEVICE_FUNC inline Index rows() const { return m_expression.rows(); } - EIGEN_DEVICE_FUNC inline Index cols() const { return m_expression.cols(); } - EIGEN_DEVICE_FUNC inline Index outerStride() const { return m_expression.outerStride(); } - EIGEN_DEVICE_FUNC inline Index innerStride() const { return m_expression.innerStride(); } - - EIGEN_DEVICE_FUNC inline const CoeffReturnType coeff(Index row, Index col) const - { - return m_expression.coeff(row, col); - } - - EIGEN_DEVICE_FUNC inline Scalar& coeffRef(Index row, Index col) - { - return m_expression.const_cast_derived().coeffRef(row, col); - } - - EIGEN_DEVICE_FUNC inline const CoeffReturnType coeff(Index index) const - { - return m_expression.coeff(index); - } - - EIGEN_DEVICE_FUNC inline Scalar& coeffRef(Index index) - { - return m_expression.const_cast_derived().coeffRef(index); - } - - template - inline const PacketScalar packet(Index row, Index col) const - { - return m_expression.template packet(row, col); - } - - template - inline void writePacket(Index row, Index col, const PacketScalar& x) - { - m_expression.const_cast_derived().template writePacket(row, col, x); - } - - template - inline const PacketScalar packet(Index index) const - { - return m_expression.template packet(index); - } - - template - inline void writePacket(Index index, const PacketScalar& x) - { - m_expression.const_cast_derived().template writePacket(index, x); - } - - EIGEN_DEVICE_FUNC operator const ExpressionType&() const { return m_expression; } - - protected: - const ExpressionType m_expression; +public: + typedef typename internal::dense_xpr_base::type Base; + EIGEN_DENSE_PUBLIC_INTERFACE(NestByValue) + + EIGEN_DEVICE_FUNC explicit inline NestByValue(const ExpressionType &matrix) : m_expression(matrix) {} + + EIGEN_DEVICE_FUNC inline Index rows() const { return m_expression.rows(); } + EIGEN_DEVICE_FUNC inline Index cols() const { return m_expression.cols(); } + EIGEN_DEVICE_FUNC inline Index outerStride() const { return m_expression.outerStride(); } + EIGEN_DEVICE_FUNC inline Index innerStride() const { return m_expression.innerStride(); } + + EIGEN_DEVICE_FUNC inline const CoeffReturnType coeff(Index row, Index col) const + { + return m_expression.coeff(row, col); + } + + EIGEN_DEVICE_FUNC inline Scalar &coeffRef(Index row, Index col) + { + return m_expression.const_cast_derived().coeffRef(row, col); + } + + EIGEN_DEVICE_FUNC inline const CoeffReturnType coeff(Index index) const { return m_expression.coeff(index); } + + EIGEN_DEVICE_FUNC inline Scalar &coeffRef(Index index) { return m_expression.const_cast_derived().coeffRef(index); } + + template inline const PacketScalar packet(Index row, Index col) const + { + return m_expression.template packet(row, col); + } + + template inline void writePacket(Index row, Index col, const PacketScalar &x) + { + m_expression.const_cast_derived().template writePacket(row, col, x); + } + + template inline const PacketScalar packet(Index index) const + { + return m_expression.template packet(index); + } + + template inline void writePacket(Index index, const PacketScalar &x) + { + m_expression.const_cast_derived().template writePacket(index, x); + } + + EIGEN_DEVICE_FUNC operator const ExpressionType &() const { return m_expression; } + +protected: + const ExpressionType m_expression; }; /** \returns an expression of the temporary version of *this. - */ -template -inline const NestByValue -DenseBase::nestByValue() const + */ +template inline const NestByValue DenseBase::nestByValue() const { return NestByValue(derived()); } -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_NESTBYVALUE_H +#endif// EIGEN_NESTBYVALUE_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/NoAlias.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/NoAlias.h index 33908010..3d25f41f 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/NoAlias.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/NoAlias.h @@ -13,96 +13,91 @@ namespace Eigen { /** \class NoAlias - * \ingroup Core_Module - * - * \brief Pseudo expression providing an operator = assuming no aliasing - * - * \tparam ExpressionType the type of the object on which to do the lazy assignment - * - * This class represents an expression with special assignment operators - * assuming no aliasing between the target expression and the source expression. - * More precisely it alloas to bypass the EvalBeforeAssignBit flag of the source expression. - * It is the return type of MatrixBase::noalias() - * and most of the time this is the only way it is used. - * - * \sa MatrixBase::noalias() - */ -template class StorageBase> -class NoAlias + * \ingroup Core_Module + * + * \brief Pseudo expression providing an operator = assuming no aliasing + * + * \tparam ExpressionType the type of the object on which to do the lazy assignment + * + * This class represents an expression with special assignment operators + * assuming no aliasing between the target expression and the source expression. + * More precisely it alloas to bypass the EvalBeforeAssignBit flag of the source expression. + * It is the return type of MatrixBase::noalias() + * and most of the time this is the only way it is used. + * + * \sa MatrixBase::noalias() + */ +template class StorageBase> class NoAlias { - public: - typedef typename ExpressionType::Scalar Scalar; - - explicit NoAlias(ExpressionType& expression) : m_expression(expression) {} - - template - EIGEN_DEVICE_FUNC - EIGEN_STRONG_INLINE ExpressionType& operator=(const StorageBase& other) - { - call_assignment_no_alias(m_expression, other.derived(), internal::assign_op()); - return m_expression; - } - - template - EIGEN_DEVICE_FUNC - EIGEN_STRONG_INLINE ExpressionType& operator+=(const StorageBase& other) - { - call_assignment_no_alias(m_expression, other.derived(), internal::add_assign_op()); - return m_expression; - } - - template - EIGEN_DEVICE_FUNC - EIGEN_STRONG_INLINE ExpressionType& operator-=(const StorageBase& other) - { - call_assignment_no_alias(m_expression, other.derived(), internal::sub_assign_op()); - return m_expression; - } +public: + typedef typename ExpressionType::Scalar Scalar; - EIGEN_DEVICE_FUNC - ExpressionType& expression() const - { - return m_expression; - } + explicit NoAlias(ExpressionType &expression) : m_expression(expression) {} - protected: - ExpressionType& m_expression; + template + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE ExpressionType &operator=(const StorageBase &other) + { + call_assignment_no_alias( + m_expression, other.derived(), internal::assign_op()); + return m_expression; + } + + template + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE ExpressionType &operator+=(const StorageBase &other) + { + call_assignment_no_alias( + m_expression, other.derived(), internal::add_assign_op()); + return m_expression; + } + + template + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE ExpressionType &operator-=(const StorageBase &other) + { + call_assignment_no_alias( + m_expression, other.derived(), internal::sub_assign_op()); + return m_expression; + } + + EIGEN_DEVICE_FUNC + ExpressionType &expression() const { return m_expression; } + +protected: + ExpressionType &m_expression; }; /** \returns a pseudo expression of \c *this with an operator= assuming - * no aliasing between \c *this and the source expression. - * - * More precisely, noalias() allows to bypass the EvalBeforeAssignBit flag. - * Currently, even though several expressions may alias, only product - * expressions have this flag. Therefore, noalias() is only usefull when - * the source expression contains a matrix product. - * - * Here are some examples where noalias is usefull: - * \code - * D.noalias() = A * B; - * D.noalias() += A.transpose() * B; - * D.noalias() -= 2 * A * B.adjoint(); - * \endcode - * - * On the other hand the following example will lead to a \b wrong result: - * \code - * A.noalias() = A * B; - * \endcode - * because the result matrix A is also an operand of the matrix product. Therefore, - * there is no alternative than evaluating A * B in a temporary, that is the default - * behavior when you write: - * \code - * A = A * B; - * \endcode - * - * \sa class NoAlias - */ -template -NoAlias MatrixBase::noalias() + * no aliasing between \c *this and the source expression. + * + * More precisely, noalias() allows to bypass the EvalBeforeAssignBit flag. + * Currently, even though several expressions may alias, only product + * expressions have this flag. Therefore, noalias() is only usefull when + * the source expression contains a matrix product. + * + * Here are some examples where noalias is usefull: + * \code + * D.noalias() = A * B; + * D.noalias() += A.transpose() * B; + * D.noalias() -= 2 * A * B.adjoint(); + * \endcode + * + * On the other hand the following example will lead to a \b wrong result: + * \code + * A.noalias() = A * B; + * \endcode + * because the result matrix A is also an operand of the matrix product. Therefore, + * there is no alternative than evaluating A * B in a temporary, that is the default + * behavior when you write: + * \code + * A = A * B; + * \endcode + * + * \sa class NoAlias + */ +template NoAlias MatrixBase::noalias() { - return NoAlias(derived()); + return NoAlias(derived()); } -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_NOALIAS_H +#endif// EIGEN_NOALIAS_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/NumTraits.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/NumTraits.h index daf48987..d55282f1 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/NumTraits.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/NumTraits.h @@ -14,76 +14,78 @@ namespace Eigen { namespace internal { -// default implementation of digits10(), based on numeric_limits if specialized, -// 0 for integer types, and log10(epsilon()) otherwise. -template< typename T, - bool use_numeric_limits = std::numeric_limits::is_specialized, - bool is_integer = NumTraits::IsInteger> -struct default_digits10_impl -{ - static int run() { return std::numeric_limits::digits10; } -}; + // default implementation of digits10(), based on numeric_limits if specialized, + // 0 for integer types, and log10(epsilon()) otherwise. + template::is_specialized, + bool is_integer = NumTraits::IsInteger> + struct default_digits10_impl + { + static int run() { return std::numeric_limits::digits10; } + }; -template -struct default_digits10_impl // Floating point -{ - static int run() { - using std::log10; - using std::ceil; - typedef typename NumTraits::Real Real; - return int(ceil(-log10(NumTraits::epsilon()))); - } -}; + template struct default_digits10_impl// Floating point + { + static int run() + { + using std::log10; + using std::ceil; + typedef typename NumTraits::Real Real; + return int(ceil(-log10(NumTraits::epsilon()))); + } + }; -template -struct default_digits10_impl // Integer -{ - static int run() { return 0; } -}; + template struct default_digits10_impl// Integer + { + static int run() { return 0; } + }; -} // end namespace internal +}// end namespace internal /** \class NumTraits - * \ingroup Core_Module - * - * \brief Holds information about the various numeric (i.e. scalar) types allowed by Eigen. - * - * \tparam T the numeric type at hand - * - * This class stores enums, typedefs and static methods giving information about a numeric type. - * - * The provided data consists of: - * \li A typedef \c Real, giving the "real part" type of \a T. If \a T is already real, - * then \c Real is just a typedef to \a T. If \a T is \c std::complex then \c Real - * is a typedef to \a U. - * \li A typedef \c NonInteger, giving the type that should be used for operations producing non-integral values, - * such as quotients, square roots, etc. If \a T is a floating-point type, then this typedef just gives - * \a T again. Note however that many Eigen functions such as internal::sqrt simply refuse to - * take integers. Outside of a few cases, Eigen doesn't do automatic type promotion. Thus, this typedef is - * only intended as a helper for code that needs to explicitly promote types. - * \li A typedef \c Literal giving the type to use for numeric literals such as "2" or "0.5". For instance, for \c std::complex, Literal is defined as \c U. - * Of course, this type must be fully compatible with \a T. In doubt, just use \a T here. - * \li A typedef \a Nested giving the type to use to nest a value inside of the expression tree. If you don't know what - * this means, just use \a T here. - * \li An enum value \a IsComplex. It is equal to 1 if \a T is a \c std::complex - * type, and to 0 otherwise. - * \li An enum value \a IsInteger. It is equal to \c 1 if \a T is an integer type such as \c int, - * and to \c 0 otherwise. - * \li Enum values ReadCost, AddCost and MulCost representing a rough estimate of the number of CPU cycles needed - * to by move / add / mul instructions respectively, assuming the data is already stored in CPU registers. - * Stay vague here. No need to do architecture-specific stuff. - * \li An enum value \a IsSigned. It is equal to \c 1 if \a T is a signed type and to 0 if \a T is unsigned. - * \li An enum value \a RequireInitialization. It is equal to \c 1 if the constructor of the numeric type \a T must - * be called, and to 0 if it is safe not to call it. Default is 0 if \a T is an arithmetic type, and 1 otherwise. - * \li An epsilon() function which, unlike std::numeric_limits::epsilon(), - * it returns a \a Real instead of a \a T. - * \li A dummy_precision() function returning a weak epsilon value. It is mainly used as a default - * value by the fuzzy comparison operators. - * \li highest() and lowest() functions returning the highest and lowest possible values respectively. - * \li digits10() function returning the number of decimal digits that can be represented without change. This is - * the analogue of std::numeric_limits::digits10 - * which is used as the default implementation if specialized. - */ + * \ingroup Core_Module + * + * \brief Holds information about the various numeric (i.e. scalar) types allowed by Eigen. + * + * \tparam T the numeric type at hand + * + * This class stores enums, typedefs and static methods giving information about a numeric type. + * + * The provided data consists of: + * \li A typedef \c Real, giving the "real part" type of \a T. If \a T is already real, + * then \c Real is just a typedef to \a T. If \a T is \c std::complex then \c Real + * is a typedef to \a U. + * \li A typedef \c NonInteger, giving the type that should be used for operations producing non-integral values, + * such as quotients, square roots, etc. If \a T is a floating-point type, then this typedef just gives + * \a T again. Note however that many Eigen functions such as internal::sqrt simply refuse to + * take integers. Outside of a few cases, Eigen doesn't do automatic type promotion. Thus, this typedef is + * only intended as a helper for code that needs to explicitly promote types. + * \li A typedef \c Literal giving the type to use for numeric literals such as "2" or "0.5". For instance, for \c + * std::complex, Literal is defined as \c U. Of course, this type must be fully compatible with \a T. In doubt, just + * use \a T here. + * \li A typedef \a Nested giving the type to use to nest a value inside of the expression tree. If you don't know what + * this means, just use \a T here. + * \li An enum value \a IsComplex. It is equal to 1 if \a T is a \c std::complex + * type, and to 0 otherwise. + * \li An enum value \a IsInteger. It is equal to \c 1 if \a T is an integer type such as \c int, + * and to \c 0 otherwise. + * \li Enum values ReadCost, AddCost and MulCost representing a rough estimate of the number of CPU cycles needed + * to by move / add / mul instructions respectively, assuming the data is already stored in CPU registers. + * Stay vague here. No need to do architecture-specific stuff. + * \li An enum value \a IsSigned. It is equal to \c 1 if \a T is a signed type and to 0 if \a T is unsigned. + * \li An enum value \a RequireInitialization. It is equal to \c 1 if the constructor of the numeric type \a T must + * be called, and to 0 if it is safe not to call it. Default is 0 if \a T is an arithmetic type, and 1 otherwise. + * \li An epsilon() function which, unlike std::numeric_limits::epsilon(), it returns a + * \a Real instead of a \a T. + * \li A dummy_precision() function returning a weak epsilon value. It is mainly used as a default + * value by the fuzzy comparison operators. + * \li highest() and lowest() functions returning the highest and lowest possible values respectively. + * \li digits10() function returning the number of decimal digits that can be represented without change. This is + * the analogue of std::numeric_limits::digits10 which is + * used as the default implementation if specialized. + */ template struct GenericNumTraits { @@ -98,25 +100,16 @@ template struct GenericNumTraits }; typedef T Real; - typedef typename internal::conditional< - IsInteger, - typename internal::conditional::type, - T - >::type NonInteger; + typedef typename internal:: + conditional::type, T>::type NonInteger; typedef T Nested; typedef T Literal; EIGEN_DEVICE_FUNC - static inline Real epsilon() - { - return numext::numeric_limits::epsilon(); - } + static inline Real epsilon() { return numext::numeric_limits::epsilon(); } EIGEN_DEVICE_FUNC - static inline int digits10() - { - return internal::default_digits10_impl::run(); - } + static inline int digits10() { return internal::default_digits10_impl::run(); } EIGEN_DEVICE_FUNC static inline Real dummy_precision() @@ -127,31 +120,26 @@ template struct GenericNumTraits EIGEN_DEVICE_FUNC - static inline T highest() { - return (numext::numeric_limits::max)(); - } + static inline T highest() { return (numext::numeric_limits::max)(); } EIGEN_DEVICE_FUNC - static inline T lowest() { + static inline T lowest() + { return IsInteger ? (numext::numeric_limits::min)() : (-(numext::numeric_limits::max)()); } EIGEN_DEVICE_FUNC - static inline T infinity() { - return numext::numeric_limits::infinity(); - } + static inline T infinity() { return numext::numeric_limits::infinity(); } EIGEN_DEVICE_FUNC - static inline T quiet_NaN() { - return numext::numeric_limits::quiet_NaN(); - } + static inline T quiet_NaN() { return numext::numeric_limits::quiet_NaN(); } }; template struct NumTraits : GenericNumTraits -{}; +{ +}; -template<> struct NumTraits - : GenericNumTraits +template<> struct NumTraits : GenericNumTraits { EIGEN_DEVICE_FUNC static inline float dummy_precision() { return 1e-5f; } @@ -163,14 +151,12 @@ template<> struct NumTraits : GenericNumTraits static inline double dummy_precision() { return 1e-12; } }; -template<> struct NumTraits - : GenericNumTraits +template<> struct NumTraits : GenericNumTraits { static inline long double dummy_precision() { return 1e-15l; } }; -template struct NumTraits > - : GenericNumTraits > +template struct NumTraits> : GenericNumTraits> { typedef _Real Real; typedef typename NumTraits<_Real>::Literal Literal; @@ -191,24 +177,27 @@ template struct NumTraits > }; template -struct NumTraits > +struct NumTraits> { typedef Array ArrayType; typedef typename NumTraits::Real RealScalar; typedef Array Real; typedef typename NumTraits::NonInteger NonIntegerScalar; typedef Array NonInteger; - typedef ArrayType & Nested; + typedef ArrayType &Nested; typedef typename NumTraits::Literal Literal; enum { IsComplex = NumTraits::IsComplex, IsInteger = NumTraits::IsInteger, - IsSigned = NumTraits::IsSigned, + IsSigned = NumTraits::IsSigned, RequireInitialization = 1, - ReadCost = ArrayType::SizeAtCompileTime==Dynamic ? HugeCost : ArrayType::SizeAtCompileTime * NumTraits::ReadCost, - AddCost = ArrayType::SizeAtCompileTime==Dynamic ? HugeCost : ArrayType::SizeAtCompileTime * NumTraits::AddCost, - MulCost = ArrayType::SizeAtCompileTime==Dynamic ? HugeCost : ArrayType::SizeAtCompileTime * NumTraits::MulCost + ReadCost = + ArrayType::SizeAtCompileTime == Dynamic ? HugeCost : ArrayType::SizeAtCompileTime * NumTraits::ReadCost, + AddCost = + ArrayType::SizeAtCompileTime == Dynamic ? HugeCost : ArrayType::SizeAtCompileTime * NumTraits::AddCost, + MulCost = + ArrayType::SizeAtCompileTime == Dynamic ? HugeCost : ArrayType::SizeAtCompileTime * NumTraits::MulCost }; EIGEN_DEVICE_FUNC @@ -219,15 +208,9 @@ struct NumTraits > static inline int digits10() { return NumTraits::digits10(); } }; -template<> struct NumTraits - : GenericNumTraits +template<> struct NumTraits : GenericNumTraits { - enum { - RequireInitialization = 1, - ReadCost = HugeCost, - AddCost = HugeCost, - MulCost = HugeCost - }; + enum { RequireInitialization = 1, ReadCost = HugeCost, AddCost = HugeCost, MulCost = HugeCost }; static inline int digits10() { return 0; } @@ -241,8 +224,10 @@ template<> struct NumTraits }; // Empty specialization for void to allow template specialization based on NumTraits::Real with T==void and SFINAE. -template<> struct NumTraits {}; +template<> struct NumTraits +{ +}; -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_NUMTRAITS_H +#endif// EIGEN_NUMTRAITS_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/PermutationMatrix.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/PermutationMatrix.h index b1fb455b..111844cd 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/PermutationMatrix.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/PermutationMatrix.h @@ -11,623 +11,587 @@ #ifndef EIGEN_PERMUTATIONMATRIX_H #define EIGEN_PERMUTATIONMATRIX_H -namespace Eigen { +namespace Eigen { namespace internal { -enum PermPermProduct_t {PermPermProduct}; + enum PermPermProduct_t { PermPermProduct }; -} // end namespace internal +}// end namespace internal /** \class PermutationBase - * \ingroup Core_Module - * - * \brief Base class for permutations - * - * \tparam Derived the derived class - * - * This class is the base class for all expressions representing a permutation matrix, - * internally stored as a vector of integers. - * The convention followed here is that if \f$ \sigma \f$ is a permutation, the corresponding permutation matrix - * \f$ P_\sigma \f$ is such that if \f$ (e_1,\ldots,e_p) \f$ is the canonical basis, we have: - * \f[ P_\sigma(e_i) = e_{\sigma(i)}. \f] - * This convention ensures that for any two permutations \f$ \sigma, \tau \f$, we have: - * \f[ P_{\sigma\circ\tau} = P_\sigma P_\tau. \f] - * - * Permutation matrices are square and invertible. - * - * Notice that in addition to the member functions and operators listed here, there also are non-member - * operator* to multiply any kind of permutation object with any kind of matrix expression (MatrixBase) - * on either side. - * - * \sa class PermutationMatrix, class PermutationWrapper - */ -template -class PermutationBase : public EigenBase + * \ingroup Core_Module + * + * \brief Base class for permutations + * + * \tparam Derived the derived class + * + * This class is the base class for all expressions representing a permutation matrix, + * internally stored as a vector of integers. + * The convention followed here is that if \f$ \sigma \f$ is a permutation, the corresponding permutation matrix + * \f$ P_\sigma \f$ is such that if \f$ (e_1,\ldots,e_p) \f$ is the canonical basis, we have: + * \f[ P_\sigma(e_i) = e_{\sigma(i)}. \f] + * This convention ensures that for any two permutations \f$ \sigma, \tau \f$, we have: + * \f[ P_{\sigma\circ\tau} = P_\sigma P_\tau. \f] + * + * Permutation matrices are square and invertible. + * + * Notice that in addition to the member functions and operators listed here, there also are non-member + * operator* to multiply any kind of permutation object with any kind of matrix expression (MatrixBase) + * on either side. + * + * \sa class PermutationMatrix, class PermutationWrapper + */ +template class PermutationBase : public EigenBase { - typedef internal::traits Traits; - typedef EigenBase Base; - public: + typedef internal::traits Traits; + typedef EigenBase Base; - #ifndef EIGEN_PARSED_BY_DOXYGEN - typedef typename Traits::IndicesType IndicesType; - enum { - Flags = Traits::Flags, - RowsAtCompileTime = Traits::RowsAtCompileTime, - ColsAtCompileTime = Traits::ColsAtCompileTime, - MaxRowsAtCompileTime = Traits::MaxRowsAtCompileTime, - MaxColsAtCompileTime = Traits::MaxColsAtCompileTime - }; - typedef typename Traits::StorageIndex StorageIndex; - typedef Matrix - DenseMatrixType; - typedef PermutationMatrix - PlainPermutationType; - typedef PlainPermutationType PlainObject; - using Base::derived; - typedef Inverse InverseReturnType; - typedef void Scalar; - #endif - - /** Copies the other permutation into *this */ - template - Derived& operator=(const PermutationBase& other) - { - indices() = other.indices(); - return derived(); - } - - /** Assignment from the Transpositions \a tr */ - template - Derived& operator=(const TranspositionsBase& tr) - { - setIdentity(tr.size()); - for(Index k=size()-1; k>=0; --k) - applyTranspositionOnTheRight(k,tr.coeff(k)); - return derived(); - } - - #ifndef EIGEN_PARSED_BY_DOXYGEN - /** This is a special case of the templated operator=. Its purpose is to - * prevent a default operator= from hiding the templated operator=. - */ - Derived& operator=(const PermutationBase& other) - { - indices() = other.indices(); - return derived(); - } - #endif - - /** \returns the number of rows */ - inline Index rows() const { return Index(indices().size()); } - - /** \returns the number of columns */ - inline Index cols() const { return Index(indices().size()); } +public: +#ifndef EIGEN_PARSED_BY_DOXYGEN + typedef typename Traits::IndicesType IndicesType; + enum { + Flags = Traits::Flags, + RowsAtCompileTime = Traits::RowsAtCompileTime, + ColsAtCompileTime = Traits::ColsAtCompileTime, + MaxRowsAtCompileTime = Traits::MaxRowsAtCompileTime, + MaxColsAtCompileTime = Traits::MaxColsAtCompileTime + }; + typedef typename Traits::StorageIndex StorageIndex; + typedef Matrix + DenseMatrixType; + typedef PermutationMatrix + PlainPermutationType; + typedef PlainPermutationType PlainObject; + using Base::derived; + typedef Inverse InverseReturnType; + typedef void Scalar; +#endif - /** \returns the size of a side of the respective square matrix, i.e., the number of indices */ - inline Index size() const { return Index(indices().size()); } + /** Copies the other permutation into *this */ + template Derived &operator=(const PermutationBase &other) + { + indices() = other.indices(); + return derived(); + } + + /** Assignment from the Transpositions \a tr */ + template Derived &operator=(const TranspositionsBase &tr) + { + setIdentity(tr.size()); + for (Index k = size() - 1; k >= 0; --k) applyTranspositionOnTheRight(k, tr.coeff(k)); + return derived(); + } - #ifndef EIGEN_PARSED_BY_DOXYGEN - template - void evalTo(MatrixBase& other) const - { - other.setZero(); - for (Index i=0; i void evalTo(MatrixBase &other) const + { + other.setZero(); + for (Index i = 0; i < rows(); ++i) other.coeffRef(indices().coeff(i), i) = typename DenseDerived::Scalar(1); + } +#endif - /** Multiplies *this by the transposition \f$(ij)\f$ on the left. - * - * \returns a reference to *this. - * - * \warning This is much slower than applyTranspositionOnTheRight(Index,Index): - * this has linear complexity and requires a lot of branching. - * - * \sa applyTranspositionOnTheRight(Index,Index) - */ - Derived& applyTranspositionOnTheLeft(Index i, Index j) - { - eigen_assert(i>=0 && j>=0 && i= 0 && j >= 0 && i < size() && j < size()); + for (Index k = 0; k < size(); ++k) { + if (indices().coeff(k) == i) + indices().coeffRef(k) = StorageIndex(j); + else if (indices().coeff(k) == j) + indices().coeffRef(k) = StorageIndex(i); } + return derived(); + } + + /** Multiplies *this by the transposition \f$(ij)\f$ on the right. + * + * \returns a reference to *this. + * + * This is a fast operation, it only consists in swapping two indices. + * + * \sa applyTranspositionOnTheLeft(Index,Index) + */ + Derived &applyTranspositionOnTheRight(Index i, Index j) + { + eigen_assert(i >= 0 && j >= 0 && i < size() && j < size()); + std::swap(indices().coeffRef(i), indices().coeffRef(j)); + return derived(); + } + + /** \returns the inverse permutation matrix. + * + * \note \blank \note_try_to_help_rvo + */ + inline InverseReturnType inverse() const { return InverseReturnType(derived()); } + /** \returns the tranpose permutation matrix. + * + * \note \blank \note_try_to_help_rvo + */ + inline InverseReturnType transpose() const { return InverseReturnType(derived()); } + + /**** multiplication helpers to hopefully get RVO ****/ - /** Multiplies *this by the transposition \f$(ij)\f$ on the right. - * - * \returns a reference to *this. - * - * This is a fast operation, it only consists in swapping two indices. - * - * \sa applyTranspositionOnTheLeft(Index,Index) - */ - Derived& applyTranspositionOnTheRight(Index i, Index j) - { - eigen_assert(i>=0 && j>=0 && i - void assignTranspose(const PermutationBase& other) - { - for (Index i=0; i - void assignProduct(const Lhs& lhs, const Rhs& rhs) - { - eigen_assert(lhs.cols() == rhs.rows()); - for (Index i=0; i void assignTranspose(const PermutationBase &other) + { + for (Index i = 0; i < rows(); ++i) indices().coeffRef(other.indices().coeff(i)) = i; + } + template void assignProduct(const Lhs &lhs, const Rhs &rhs) + { + eigen_assert(lhs.cols() == rhs.rows()); + for (Index i = 0; i < rows(); ++i) indices().coeffRef(i) = lhs.indices().coeff(rhs.indices().coeff(i)); + } #endif - public: - - /** \returns the product permutation matrix. - * - * \note \blank \note_try_to_help_rvo - */ - template - inline PlainPermutationType operator*(const PermutationBase& other) const - { return PlainPermutationType(internal::PermPermProduct, derived(), other.derived()); } - - /** \returns the product of a permutation with another inverse permutation. - * - * \note \blank \note_try_to_help_rvo - */ - template - inline PlainPermutationType operator*(const InverseImpl& other) const - { return PlainPermutationType(internal::PermPermProduct, *this, other.eval()); } - - /** \returns the product of an inverse permutation with another permutation. - * - * \note \blank \note_try_to_help_rvo - */ - template friend - inline PlainPermutationType operator*(const InverseImpl& other, const PermutationBase& perm) - { return PlainPermutationType(internal::PermPermProduct, other.eval(), perm); } - - /** \returns the determinant of the permutation matrix, which is either 1 or -1 depending on the parity of the permutation. - * - * This function is O(\c n) procedure allocating a buffer of \c n booleans. - */ - Index determinant() const - { - Index res = 1; - Index n = size(); - Matrix mask(n); - mask.fill(false); - Index r = 0; - while(r < n) - { - // search for the next seed - while(r=n) - break; - // we got one, let's follow it until we are back to the seed - Index k0 = r++; - mask.coeffRef(k0) = true; - for(Index k=indices().coeff(k0); k!=k0; k=indices().coeff(k)) - { - mask.coeffRef(k) = true; - res = -res; - } +public: + /** \returns the product permutation matrix. + * + * \note \blank \note_try_to_help_rvo + */ + template inline PlainPermutationType operator*(const PermutationBase &other) const + { + return PlainPermutationType(internal::PermPermProduct, derived(), other.derived()); + } + + /** \returns the product of a permutation with another inverse permutation. + * + * \note \blank \note_try_to_help_rvo + */ + template + inline PlainPermutationType operator*(const InverseImpl &other) const + { + return PlainPermutationType(internal::PermPermProduct, *this, other.eval()); + } + + /** \returns the product of an inverse permutation with another permutation. + * + * \note \blank \note_try_to_help_rvo + */ + template + friend inline PlainPermutationType operator*(const InverseImpl &other, + const PermutationBase &perm) + { + return PlainPermutationType(internal::PermPermProduct, other.eval(), perm); + } + + /** \returns the determinant of the permutation matrix, which is either 1 or -1 depending on the parity of the + * permutation. + * + * This function is O(\c n) procedure allocating a buffer of \c n booleans. + */ + Index determinant() const + { + Index res = 1; + Index n = size(); + Matrix mask(n); + mask.fill(false); + Index r = 0; + while (r < n) { + // search for the next seed + while (r < n && mask[r]) r++; + if (r >= n) break; + // we got one, let's follow it until we are back to the seed + Index k0 = r++; + mask.coeffRef(k0) = true; + for (Index k = indices().coeff(k0); k != k0; k = indices().coeff(k)) { + mask.coeffRef(k) = true; + res = -res; } - return res; } + return res; + } - protected: - +protected: }; namespace internal { -template -struct traits > - : traits > -{ - typedef PermutationStorage StorageKind; - typedef Matrix<_StorageIndex, SizeAtCompileTime, 1, 0, MaxSizeAtCompileTime, 1> IndicesType; - typedef _StorageIndex StorageIndex; - typedef void Scalar; -}; -} + template + struct traits> + : traits> + { + typedef PermutationStorage StorageKind; + typedef Matrix<_StorageIndex, SizeAtCompileTime, 1, 0, MaxSizeAtCompileTime, 1> IndicesType; + typedef _StorageIndex StorageIndex; + typedef void Scalar; + }; +}// namespace internal /** \class PermutationMatrix - * \ingroup Core_Module - * - * \brief Permutation matrix - * - * \tparam SizeAtCompileTime the number of rows/cols, or Dynamic - * \tparam MaxSizeAtCompileTime the maximum number of rows/cols, or Dynamic. This optional parameter defaults to SizeAtCompileTime. Most of the time, you should not have to specify it. - * \tparam _StorageIndex the integer type of the indices - * - * This class represents a permutation matrix, internally stored as a vector of integers. - * - * \sa class PermutationBase, class PermutationWrapper, class DiagonalMatrix - */ + * \ingroup Core_Module + * + * \brief Permutation matrix + * + * \tparam SizeAtCompileTime the number of rows/cols, or Dynamic + * \tparam MaxSizeAtCompileTime the maximum number of rows/cols, or Dynamic. This optional parameter defaults to + * SizeAtCompileTime. Most of the time, you should not have to specify it. + * \tparam _StorageIndex the integer type of the indices + * + * This class represents a permutation matrix, internally stored as a vector of integers. + * + * \sa class PermutationBase, class PermutationWrapper, class DiagonalMatrix + */ template -class PermutationMatrix : public PermutationBase > +class PermutationMatrix + : public PermutationBase> { - typedef PermutationBase Base; - typedef internal::traits Traits; - public: + typedef PermutationBase Base; + typedef internal::traits Traits; - typedef const PermutationMatrix& Nested; +public: + typedef const PermutationMatrix &Nested; - #ifndef EIGEN_PARSED_BY_DOXYGEN - typedef typename Traits::IndicesType IndicesType; - typedef typename Traits::StorageIndex StorageIndex; - #endif +#ifndef EIGEN_PARSED_BY_DOXYGEN + typedef typename Traits::IndicesType IndicesType; + typedef typename Traits::StorageIndex StorageIndex; +#endif - inline PermutationMatrix() - {} + inline PermutationMatrix() {} - /** Constructs an uninitialized permutation matrix of given size. - */ - explicit inline PermutationMatrix(Index size) : m_indices(size) - { - eigen_internal_assert(size <= NumTraits::highest()); - } + /** Constructs an uninitialized permutation matrix of given size. + */ + explicit inline PermutationMatrix(Index size) : m_indices(size) + { + eigen_internal_assert(size <= NumTraits::highest()); + } - /** Copy constructor. */ - template - inline PermutationMatrix(const PermutationBase& other) - : m_indices(other.indices()) {} - - #ifndef EIGEN_PARSED_BY_DOXYGEN - /** Standard copy constructor. Defined only to prevent a default copy constructor - * from hiding the other templated constructor */ - inline PermutationMatrix(const PermutationMatrix& other) : m_indices(other.indices()) {} - #endif - - /** Generic constructor from expression of the indices. The indices - * array has the meaning that the permutations sends each integer i to indices[i]. - * - * \warning It is your responsibility to check that the indices array that you passes actually - * describes a permutation, i.e., each value between 0 and n-1 occurs exactly once, where n is the - * array's size. - */ - template - explicit inline PermutationMatrix(const MatrixBase& indices) : m_indices(indices) - {} - - /** Convert the Transpositions \a tr to a permutation matrix */ - template - explicit PermutationMatrix(const TranspositionsBase& tr) - : m_indices(tr.size()) - { - *this = tr; - } + /** Copy constructor. */ + template + inline PermutationMatrix(const PermutationBase &other) : m_indices(other.indices()) + {} - /** Copies the other permutation into *this */ - template - PermutationMatrix& operator=(const PermutationBase& other) - { - m_indices = other.indices(); - return *this; - } +#ifndef EIGEN_PARSED_BY_DOXYGEN + /** Standard copy constructor. Defined only to prevent a default copy constructor + * from hiding the other templated constructor */ + inline PermutationMatrix(const PermutationMatrix &other) : m_indices(other.indices()) {} +#endif - /** Assignment from the Transpositions \a tr */ - template - PermutationMatrix& operator=(const TranspositionsBase& tr) - { - return Base::operator=(tr.derived()); - } + /** Generic constructor from expression of the indices. The indices + * array has the meaning that the permutations sends each integer i to indices[i]. + * + * \warning It is your responsibility to check that the indices array that you passes actually + * describes a permutation, i.e., each value between 0 and n-1 occurs exactly once, where n is the + * array's size. + */ + template explicit inline PermutationMatrix(const MatrixBase &indices) : m_indices(indices) {} + + /** Convert the Transpositions \a tr to a permutation matrix */ + template explicit PermutationMatrix(const TranspositionsBase &tr) : m_indices(tr.size()) + { + *this = tr; + } + + /** Copies the other permutation into *this */ + template PermutationMatrix &operator=(const PermutationBase &other) + { + m_indices = other.indices(); + return *this; + } + + /** Assignment from the Transpositions \a tr */ + template PermutationMatrix &operator=(const TranspositionsBase &tr) + { + return Base::operator=(tr.derived()); + } - #ifndef EIGEN_PARSED_BY_DOXYGEN - /** This is a special case of the templated operator=. Its purpose is to - * prevent a default operator= from hiding the templated operator=. - */ - PermutationMatrix& operator=(const PermutationMatrix& other) - { - m_indices = other.m_indices; - return *this; - } - #endif +#ifndef EIGEN_PARSED_BY_DOXYGEN + /** This is a special case of the templated operator=. Its purpose is to + * prevent a default operator= from hiding the templated operator=. + */ + PermutationMatrix &operator=(const PermutationMatrix &other) + { + m_indices = other.m_indices; + return *this; + } +#endif - /** const version of indices(). */ - const IndicesType& indices() const { return m_indices; } - /** \returns a reference to the stored array representing the permutation. */ - IndicesType& indices() { return m_indices; } + /** const version of indices(). */ + const IndicesType &indices() const { return m_indices; } + /** \returns a reference to the stored array representing the permutation. */ + IndicesType &indices() { return m_indices; } - /**** multiplication helpers to hopefully get RVO ****/ + /**** multiplication helpers to hopefully get RVO ****/ #ifndef EIGEN_PARSED_BY_DOXYGEN - template - PermutationMatrix(const InverseImpl& other) - : m_indices(other.derived().nestedExpression().size()) - { - eigen_internal_assert(m_indices.size() <= NumTraits::highest()); - StorageIndex end = StorageIndex(m_indices.size()); - for (StorageIndex i=0; i - PermutationMatrix(internal::PermPermProduct_t, const Lhs& lhs, const Rhs& rhs) - : m_indices(lhs.indices().size()) - { - Base::assignProduct(lhs,rhs); - } + template + PermutationMatrix(const InverseImpl &other) + : m_indices(other.derived().nestedExpression().size()) + { + eigen_internal_assert(m_indices.size() <= NumTraits::highest()); + StorageIndex end = StorageIndex(m_indices.size()); + for (StorageIndex i = 0; i < end; ++i) + m_indices.coeffRef(other.derived().nestedExpression().indices().coeff(i)) = i; + } + template + PermutationMatrix(internal::PermPermProduct_t, const Lhs &lhs, const Rhs &rhs) : m_indices(lhs.indices().size()) + { + Base::assignProduct(lhs, rhs); + } #endif - protected: - - IndicesType m_indices; +protected: + IndicesType m_indices; }; namespace internal { -template -struct traits,_PacketAccess> > - : traits > -{ - typedef PermutationStorage StorageKind; - typedef Map, _PacketAccess> IndicesType; - typedef _StorageIndex StorageIndex; - typedef void Scalar; -}; -} + template + struct traits, _PacketAccess>> + : traits> + { + typedef PermutationStorage StorageKind; + typedef Map, _PacketAccess> + IndicesType; + typedef _StorageIndex StorageIndex; + typedef void Scalar; + }; +}// namespace internal template -class Map,_PacketAccess> - : public PermutationBase,_PacketAccess> > +class Map, _PacketAccess> + : public PermutationBase< + Map, _PacketAccess>> { - typedef PermutationBase Base; - typedef internal::traits Traits; - public: - - #ifndef EIGEN_PARSED_BY_DOXYGEN - typedef typename Traits::IndicesType IndicesType; - typedef typename IndicesType::Scalar StorageIndex; - #endif - - inline Map(const StorageIndex* indicesPtr) - : m_indices(indicesPtr) - {} - - inline Map(const StorageIndex* indicesPtr, Index size) - : m_indices(indicesPtr,size) - {} - - /** Copies the other permutation into *this */ - template - Map& operator=(const PermutationBase& other) - { return Base::operator=(other.derived()); } - - /** Assignment from the Transpositions \a tr */ - template - Map& operator=(const TranspositionsBase& tr) - { return Base::operator=(tr.derived()); } - - #ifndef EIGEN_PARSED_BY_DOXYGEN - /** This is a special case of the templated operator=. Its purpose is to - * prevent a default operator= from hiding the templated operator=. - */ - Map& operator=(const Map& other) - { - m_indices = other.m_indices; - return *this; - } - #endif + typedef PermutationBase Base; + typedef internal::traits Traits; - /** const version of indices(). */ - const IndicesType& indices() const { return m_indices; } - /** \returns a reference to the stored array representing the permutation. */ - IndicesType& indices() { return m_indices; } +public: +#ifndef EIGEN_PARSED_BY_DOXYGEN + typedef typename Traits::IndicesType IndicesType; + typedef typename IndicesType::Scalar StorageIndex; +#endif + + inline Map(const StorageIndex *indicesPtr) : m_indices(indicesPtr) {} - protected: + inline Map(const StorageIndex *indicesPtr, Index size) : m_indices(indicesPtr, size) {} - IndicesType m_indices; + /** Copies the other permutation into *this */ + template Map &operator=(const PermutationBase &other) + { + return Base::operator=(other.derived()); + } + + /** Assignment from the Transpositions \a tr */ + template Map &operator=(const TranspositionsBase &tr) { return Base::operator=(tr.derived()); } + +#ifndef EIGEN_PARSED_BY_DOXYGEN + /** This is a special case of the templated operator=. Its purpose is to + * prevent a default operator= from hiding the templated operator=. + */ + Map &operator=(const Map &other) + { + m_indices = other.m_indices; + return *this; + } +#endif + + /** const version of indices(). */ + const IndicesType &indices() const { return m_indices; } + /** \returns a reference to the stored array representing the permutation. */ + IndicesType &indices() { return m_indices; } + +protected: + IndicesType m_indices; }; template class TranspositionsWrapper; namespace internal { -template -struct traits > -{ - typedef PermutationStorage StorageKind; - typedef void Scalar; - typedef typename _IndicesType::Scalar StorageIndex; - typedef _IndicesType IndicesType; - enum { - RowsAtCompileTime = _IndicesType::SizeAtCompileTime, - ColsAtCompileTime = _IndicesType::SizeAtCompileTime, - MaxRowsAtCompileTime = IndicesType::MaxSizeAtCompileTime, - MaxColsAtCompileTime = IndicesType::MaxSizeAtCompileTime, - Flags = 0 + template struct traits> + { + typedef PermutationStorage StorageKind; + typedef void Scalar; + typedef typename _IndicesType::Scalar StorageIndex; + typedef _IndicesType IndicesType; + enum { + RowsAtCompileTime = _IndicesType::SizeAtCompileTime, + ColsAtCompileTime = _IndicesType::SizeAtCompileTime, + MaxRowsAtCompileTime = IndicesType::MaxSizeAtCompileTime, + MaxColsAtCompileTime = IndicesType::MaxSizeAtCompileTime, + Flags = 0 + }; }; -}; -} +}// namespace internal /** \class PermutationWrapper - * \ingroup Core_Module - * - * \brief Class to view a vector of integers as a permutation matrix - * - * \tparam _IndicesType the type of the vector of integer (can be any compatible expression) - * - * This class allows to view any vector expression of integers as a permutation matrix. - * - * \sa class PermutationBase, class PermutationMatrix - */ -template -class PermutationWrapper : public PermutationBase > + * \ingroup Core_Module + * + * \brief Class to view a vector of integers as a permutation matrix + * + * \tparam _IndicesType the type of the vector of integer (can be any compatible expression) + * + * This class allows to view any vector expression of integers as a permutation matrix. + * + * \sa class PermutationBase, class PermutationMatrix + */ +template class PermutationWrapper : public PermutationBase> { - typedef PermutationBase Base; - typedef internal::traits Traits; - public: + typedef PermutationBase Base; + typedef internal::traits Traits; - #ifndef EIGEN_PARSED_BY_DOXYGEN - typedef typename Traits::IndicesType IndicesType; - #endif - - inline PermutationWrapper(const IndicesType& indices) - : m_indices(indices) - {} +public: +#ifndef EIGEN_PARSED_BY_DOXYGEN + typedef typename Traits::IndicesType IndicesType; +#endif - /** const version of indices(). */ - const typename internal::remove_all::type& - indices() const { return m_indices; } + inline PermutationWrapper(const IndicesType &indices) : m_indices(indices) {} - protected: + /** const version of indices(). */ + const typename internal::remove_all::type &indices() const { return m_indices; } - typename IndicesType::Nested m_indices; +protected: + typename IndicesType::Nested m_indices; }; /** \returns the matrix with the permutation applied to the columns. - */ + */ template -EIGEN_DEVICE_FUNC -const Product -operator*(const MatrixBase &matrix, - const PermutationBase& permutation) +EIGEN_DEVICE_FUNC const Product + operator*(const MatrixBase &matrix, const PermutationBase &permutation) { - return Product - (matrix.derived(), permutation.derived()); + return Product(matrix.derived(), permutation.derived()); } /** \returns the matrix with the permutation applied to the rows. - */ + */ template -EIGEN_DEVICE_FUNC -const Product -operator*(const PermutationBase &permutation, - const MatrixBase& matrix) +EIGEN_DEVICE_FUNC const Product + operator*(const PermutationBase &permutation, const MatrixBase &matrix) { - return Product - (permutation.derived(), matrix.derived()); + return Product(permutation.derived(), matrix.derived()); } template -class InverseImpl - : public EigenBase > +class InverseImpl : public EigenBase> { - typedef typename PermutationType::PlainPermutationType PlainPermutationType; - typedef internal::traits PermTraits; - protected: - InverseImpl() {} - public: - typedef Inverse InverseType; - using EigenBase >::derived; - - #ifndef EIGEN_PARSED_BY_DOXYGEN - typedef typename PermutationType::DenseMatrixType DenseMatrixType; - enum { - RowsAtCompileTime = PermTraits::RowsAtCompileTime, - ColsAtCompileTime = PermTraits::ColsAtCompileTime, - MaxRowsAtCompileTime = PermTraits::MaxRowsAtCompileTime, - MaxColsAtCompileTime = PermTraits::MaxColsAtCompileTime - }; - #endif - - #ifndef EIGEN_PARSED_BY_DOXYGEN - template - void evalTo(MatrixBase& other) const - { - other.setZero(); - for (Index i=0; i PermTraits; - /** \return the equivalent permutation matrix */ - PlainPermutationType eval() const { return derived(); } +protected: + InverseImpl() {} - DenseMatrixType toDenseMatrix() const { return derived(); } +public: + typedef Inverse InverseType; + using EigenBase>::derived; - /** \returns the matrix with the inverse permutation applied to the columns. - */ - template friend - const Product - operator*(const MatrixBase& matrix, const InverseType& trPerm) - { - return Product(matrix.derived(), trPerm.derived()); - } +#ifndef EIGEN_PARSED_BY_DOXYGEN + typedef typename PermutationType::DenseMatrixType DenseMatrixType; + enum { + RowsAtCompileTime = PermTraits::RowsAtCompileTime, + ColsAtCompileTime = PermTraits::ColsAtCompileTime, + MaxRowsAtCompileTime = PermTraits::MaxRowsAtCompileTime, + MaxColsAtCompileTime = PermTraits::MaxColsAtCompileTime + }; +#endif - /** \returns the matrix with the inverse permutation applied to the rows. - */ - template - const Product - operator*(const MatrixBase& matrix) const - { - return Product(derived(), matrix.derived()); - } +#ifndef EIGEN_PARSED_BY_DOXYGEN + template void evalTo(MatrixBase &other) const + { + other.setZero(); + for (Index i = 0; i < derived().rows(); ++i) + other.coeffRef(i, derived().nestedExpression().indices().coeff(i)) = typename DenseDerived::Scalar(1); + } +#endif + + /** \return the equivalent permutation matrix */ + PlainPermutationType eval() const { return derived(); } + + DenseMatrixType toDenseMatrix() const { return derived(); } + + /** \returns the matrix with the inverse permutation applied to the columns. + */ + template + friend const Product operator*(const MatrixBase &matrix, + const InverseType &trPerm) + { + return Product(matrix.derived(), trPerm.derived()); + } + + /** \returns the matrix with the inverse permutation applied to the rows. + */ + template + const Product operator*(const MatrixBase &matrix) const + { + return Product(derived(), matrix.derived()); + } }; -template -const PermutationWrapper MatrixBase::asPermutation() const +template const PermutationWrapper MatrixBase::asPermutation() const { return derived(); } namespace internal { -template<> struct AssignmentKind { typedef EigenBase2EigenBase Kind; }; + template<> struct AssignmentKind + { + typedef EigenBase2EigenBase Kind; + }; -} // end namespace internal +}// end namespace internal -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_PERMUTATIONMATRIX_H +#endif// EIGEN_PERMUTATIONMATRIX_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/PlainObjectBase.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/PlainObjectBase.h index 1dc7e223..920c5e9a 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/PlainObjectBase.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/PlainObjectBase.h @@ -12,1024 +12,1002 @@ #define EIGEN_DENSESTORAGEBASE_H #if defined(EIGEN_INITIALIZE_MATRICES_BY_ZERO) -# define EIGEN_INITIALIZE_COEFFS -# define EIGEN_INITIALIZE_COEFFS_IF_THAT_OPTION_IS_ENABLED for(int i=0;i::quiet_NaN(); +#define EIGEN_INITIALIZE_COEFFS +#define EIGEN_INITIALIZE_COEFFS_IF_THAT_OPTION_IS_ENABLED \ + for (int i = 0; i < base().size(); ++i) coeffRef(i) = std::numeric_limits::quiet_NaN(); #else -# undef EIGEN_INITIALIZE_COEFFS -# define EIGEN_INITIALIZE_COEFFS_IF_THAT_OPTION_IS_ENABLED +#undef EIGEN_INITIALIZE_COEFFS +#define EIGEN_INITIALIZE_COEFFS_IF_THAT_OPTION_IS_ENABLED #endif namespace Eigen { namespace internal { -template struct check_rows_cols_for_overflow { - template - EIGEN_DEVICE_FUNC - static EIGEN_ALWAYS_INLINE void run(Index, Index) + template struct check_rows_cols_for_overflow { - } -}; + template EIGEN_DEVICE_FUNC static EIGEN_ALWAYS_INLINE void run(Index, Index) {} + }; -template<> struct check_rows_cols_for_overflow { - template - EIGEN_DEVICE_FUNC - static EIGEN_ALWAYS_INLINE void run(Index rows, Index cols) - { - // http://hg.mozilla.org/mozilla-central/file/6c8a909977d3/xpcom/ds/CheckedInt.h#l242 - // we assume Index is signed - Index max_index = (std::size_t(1) << (8 * sizeof(Index) - 1)) - 1; // assume Index is signed - bool error = (rows == 0 || cols == 0) ? false - : (rows > max_index / cols); - if (error) - throw_std_bad_alloc(); - } -}; + template<> struct check_rows_cols_for_overflow + { + template EIGEN_DEVICE_FUNC static EIGEN_ALWAYS_INLINE void run(Index rows, Index cols) + { + // http://hg.mozilla.org/mozilla-central/file/6c8a909977d3/xpcom/ds/CheckedInt.h#l242 + // we assume Index is signed + Index max_index = (std::size_t(1) << (8 * sizeof(Index) - 1)) - 1;// assume Index is signed + bool error = (rows == 0 || cols == 0) ? false : (rows > max_index / cols); + if (error) throw_std_bad_alloc(); + } + }; -template -struct conservative_resize_like_impl; + template + struct conservative_resize_like_impl; -template struct matrix_swap_impl; + template struct matrix_swap_impl; -} // end namespace internal +}// end namespace internal #ifdef EIGEN_PARSED_BY_DOXYGEN namespace doxygen { -// This is a workaround to doxygen not being able to understand the inheritance logic -// when it is hidden by the dense_xpr_base helper struct. -// Moreover, doxygen fails to include members that are not documented in the declaration body of -// MatrixBase if we inherits MatrixBase >, -// this is why we simply inherits MatrixBase, though this does not make sense. - -/** This class is just a workaround for Doxygen and it does not not actually exist. */ -template struct dense_xpr_base_dispatcher; -/** This class is just a workaround for Doxygen and it does not not actually exist. */ -template -struct dense_xpr_base_dispatcher > - : public MatrixBase {}; -/** This class is just a workaround for Doxygen and it does not not actually exist. */ -template -struct dense_xpr_base_dispatcher > - : public ArrayBase {}; - -} // namespace doxygen + // This is a workaround to doxygen not being able to understand the inheritance logic + // when it is hidden by the dense_xpr_base helper struct. + // Moreover, doxygen fails to include members that are not documented in the declaration body of + // MatrixBase if we inherits MatrixBase >, + // this is why we simply inherits MatrixBase, though this does not make sense. + + /** This class is just a workaround for Doxygen and it does not not actually exist. */ + template struct dense_xpr_base_dispatcher; + /** This class is just a workaround for Doxygen and it does not not actually exist. */ + template + struct dense_xpr_base_dispatcher> : public MatrixBase + { + }; + /** This class is just a workaround for Doxygen and it does not not actually exist. */ + template + struct dense_xpr_base_dispatcher> : public ArrayBase + { + }; + +}// namespace doxygen /** \class PlainObjectBase - * \ingroup Core_Module - * \brief %Dense storage base class for matrices and arrays. - * - * This class can be extended with the help of the plugin mechanism described on the page - * \ref TopicCustomizing_Plugins by defining the preprocessor symbol \c EIGEN_PLAINOBJECTBASE_PLUGIN. - * - * \tparam Derived is the derived type, e.g., a Matrix or Array - * - * \sa \ref TopicClassHierarchy - */ -template -class PlainObjectBase : public doxygen::dense_xpr_base_dispatcher + * \ingroup Core_Module + * \brief %Dense storage base class for matrices and arrays. + * + * This class can be extended with the help of the plugin mechanism described on the page + * \ref TopicCustomizing_Plugins by defining the preprocessor symbol \c EIGEN_PLAINOBJECTBASE_PLUGIN. + * + * \tparam Derived is the derived type, e.g., a Matrix or Array + * + * \sa \ref TopicClassHierarchy + */ +template class PlainObjectBase : public doxygen::dense_xpr_base_dispatcher #else -template -class PlainObjectBase : public internal::dense_xpr_base::type +template class PlainObjectBase : public internal::dense_xpr_base::type #endif { - public: - enum { Options = internal::traits::Options }; - typedef typename internal::dense_xpr_base::type Base; - - typedef typename internal::traits::StorageKind StorageKind; - typedef typename internal::traits::Scalar Scalar; - - typedef typename internal::packet_traits::type PacketScalar; - typedef typename NumTraits::Real RealScalar; - typedef Derived DenseType; - - using Base::RowsAtCompileTime; - using Base::ColsAtCompileTime; - using Base::SizeAtCompileTime; - using Base::MaxRowsAtCompileTime; - using Base::MaxColsAtCompileTime; - using Base::MaxSizeAtCompileTime; - using Base::IsVectorAtCompileTime; - using Base::Flags; - - template friend class Eigen::Map; - friend class Eigen::Map; - typedef Eigen::Map MapType; - friend class Eigen::Map; - typedef const Eigen::Map ConstMapType; -#if EIGEN_MAX_ALIGN_BYTES>0 - // for EIGEN_MAX_ALIGN_BYTES==0, AlignedMax==Unaligned, and many compilers generate warnings for friend-ing a class twice. - friend class Eigen::Map; - friend class Eigen::Map; +public: + enum { Options = internal::traits::Options }; + typedef typename internal::dense_xpr_base::type Base; + + typedef typename internal::traits::StorageKind StorageKind; + typedef typename internal::traits::Scalar Scalar; + + typedef typename internal::packet_traits::type PacketScalar; + typedef typename NumTraits::Real RealScalar; + typedef Derived DenseType; + + using Base::RowsAtCompileTime; + using Base::ColsAtCompileTime; + using Base::SizeAtCompileTime; + using Base::MaxRowsAtCompileTime; + using Base::MaxColsAtCompileTime; + using Base::MaxSizeAtCompileTime; + using Base::IsVectorAtCompileTime; + using Base::Flags; + + template friend class Eigen::Map; + friend class Eigen::Map; + typedef Eigen::Map MapType; + friend class Eigen::Map; + typedef const Eigen::Map ConstMapType; +#if EIGEN_MAX_ALIGN_BYTES > 0 + // for EIGEN_MAX_ALIGN_BYTES==0, AlignedMax==Unaligned, and many compilers generate warnings for friend-ing a class + // twice. + friend class Eigen::Map; + friend class Eigen::Map; #endif - typedef Eigen::Map AlignedMapType; - typedef const Eigen::Map ConstAlignedMapType; - template struct StridedMapType { typedef Eigen::Map type; }; - template struct StridedConstMapType { typedef Eigen::Map type; }; - template struct StridedAlignedMapType { typedef Eigen::Map type; }; - template struct StridedConstAlignedMapType { typedef Eigen::Map type; }; - - protected: - DenseStorage m_storage; - - public: - enum { NeedsToAlign = (SizeAtCompileTime != Dynamic) && (internal::traits::Alignment>0) }; - EIGEN_MAKE_ALIGNED_OPERATOR_NEW_IF(NeedsToAlign) + typedef Eigen::Map AlignedMapType; + typedef const Eigen::Map ConstAlignedMapType; + template struct StridedMapType + { + typedef Eigen::Map type; + }; + template struct StridedConstMapType + { + typedef Eigen::Map type; + }; + template struct StridedAlignedMapType + { + typedef Eigen::Map type; + }; + template struct StridedConstAlignedMapType + { + typedef Eigen::Map type; + }; - EIGEN_DEVICE_FUNC - Base& base() { return *static_cast(this); } - EIGEN_DEVICE_FUNC - const Base& base() const { return *static_cast(this); } +protected: + DenseStorage m_storage; - EIGEN_DEVICE_FUNC - EIGEN_STRONG_INLINE Index rows() const { return m_storage.rows(); } - EIGEN_DEVICE_FUNC - EIGEN_STRONG_INLINE Index cols() const { return m_storage.cols(); } +public: + enum { NeedsToAlign = (SizeAtCompileTime != Dynamic) && (internal::traits::Alignment > 0) }; + EIGEN_MAKE_ALIGNED_OPERATOR_NEW_IF(NeedsToAlign) - /** This is an overloaded version of DenseCoeffsBase::coeff(Index,Index) const - * provided to by-pass the creation of an evaluator of the expression, thus saving compilation efforts. - * - * See DenseCoeffsBase::coeff(Index) const for details. */ - EIGEN_DEVICE_FUNC - EIGEN_STRONG_INLINE const Scalar& coeff(Index rowId, Index colId) const - { - if(Flags & RowMajorBit) - return m_storage.data()[colId + rowId * m_storage.cols()]; - else // column-major - return m_storage.data()[rowId + colId * m_storage.rows()]; - } - - /** This is an overloaded version of DenseCoeffsBase::coeff(Index) const - * provided to by-pass the creation of an evaluator of the expression, thus saving compilation efforts. - * - * See DenseCoeffsBase::coeff(Index) const for details. */ - EIGEN_DEVICE_FUNC - EIGEN_STRONG_INLINE const Scalar& coeff(Index index) const - { - return m_storage.data()[index]; - } + EIGEN_DEVICE_FUNC + Base &base() { return *static_cast(this); } + EIGEN_DEVICE_FUNC + const Base &base() const { return *static_cast(this); } - /** This is an overloaded version of DenseCoeffsBase::coeffRef(Index,Index) const - * provided to by-pass the creation of an evaluator of the expression, thus saving compilation efforts. - * - * See DenseCoeffsBase::coeffRef(Index,Index) const for details. */ - EIGEN_DEVICE_FUNC - EIGEN_STRONG_INLINE Scalar& coeffRef(Index rowId, Index colId) - { - if(Flags & RowMajorBit) - return m_storage.data()[colId + rowId * m_storage.cols()]; - else // column-major - return m_storage.data()[rowId + colId * m_storage.rows()]; - } + EIGEN_DEVICE_FUNC + EIGEN_STRONG_INLINE Index rows() const { return m_storage.rows(); } + EIGEN_DEVICE_FUNC + EIGEN_STRONG_INLINE Index cols() const { return m_storage.cols(); } - /** This is an overloaded version of DenseCoeffsBase::coeffRef(Index) const - * provided to by-pass the creation of an evaluator of the expression, thus saving compilation efforts. - * - * See DenseCoeffsBase::coeffRef(Index) const for details. */ - EIGEN_DEVICE_FUNC - EIGEN_STRONG_INLINE Scalar& coeffRef(Index index) - { - return m_storage.data()[index]; - } + /** This is an overloaded version of DenseCoeffsBase::coeff(Index,Index) const + * provided to by-pass the creation of an evaluator of the expression, thus saving compilation efforts. + * + * See DenseCoeffsBase::coeff(Index) const for details. */ + EIGEN_DEVICE_FUNC + EIGEN_STRONG_INLINE const Scalar &coeff(Index rowId, Index colId) const + { + if (Flags & RowMajorBit) + return m_storage.data()[colId + rowId * m_storage.cols()]; + else// column-major + return m_storage.data()[rowId + colId * m_storage.rows()]; + } - /** This is the const version of coeffRef(Index,Index) which is thus synonym of coeff(Index,Index). - * It is provided for convenience. */ - EIGEN_DEVICE_FUNC - EIGEN_STRONG_INLINE const Scalar& coeffRef(Index rowId, Index colId) const - { - if(Flags & RowMajorBit) - return m_storage.data()[colId + rowId * m_storage.cols()]; - else // column-major - return m_storage.data()[rowId + colId * m_storage.rows()]; - } + /** This is an overloaded version of DenseCoeffsBase::coeff(Index) const + * provided to by-pass the creation of an evaluator of the expression, thus saving compilation efforts. + * + * See DenseCoeffsBase::coeff(Index) const for details. */ + EIGEN_DEVICE_FUNC + EIGEN_STRONG_INLINE const Scalar &coeff(Index index) const { return m_storage.data()[index]; } - /** This is the const version of coeffRef(Index) which is thus synonym of coeff(Index). - * It is provided for convenience. */ - EIGEN_DEVICE_FUNC - EIGEN_STRONG_INLINE const Scalar& coeffRef(Index index) const - { - return m_storage.data()[index]; - } + /** This is an overloaded version of DenseCoeffsBase::coeffRef(Index,Index) const + * provided to by-pass the creation of an evaluator of the expression, thus saving compilation efforts. + * + * See DenseCoeffsBase::coeffRef(Index,Index) const for details. */ + EIGEN_DEVICE_FUNC + EIGEN_STRONG_INLINE Scalar &coeffRef(Index rowId, Index colId) + { + if (Flags & RowMajorBit) + return m_storage.data()[colId + rowId * m_storage.cols()]; + else// column-major + return m_storage.data()[rowId + colId * m_storage.rows()]; + } - /** \internal */ - template - EIGEN_STRONG_INLINE PacketScalar packet(Index rowId, Index colId) const - { - return internal::ploadt - (m_storage.data() + (Flags & RowMajorBit - ? colId + rowId * m_storage.cols() - : rowId + colId * m_storage.rows())); - } + /** This is an overloaded version of DenseCoeffsBase::coeffRef(Index) const + * provided to by-pass the creation of an evaluator of the expression, thus saving compilation efforts. + * + * See DenseCoeffsBase::coeffRef(Index) const for details. */ + EIGEN_DEVICE_FUNC + EIGEN_STRONG_INLINE Scalar &coeffRef(Index index) { return m_storage.data()[index]; } - /** \internal */ - template - EIGEN_STRONG_INLINE PacketScalar packet(Index index) const - { - return internal::ploadt(m_storage.data() + index); - } + /** This is the const version of coeffRef(Index,Index) which is thus synonym of coeff(Index,Index). + * It is provided for convenience. */ + EIGEN_DEVICE_FUNC + EIGEN_STRONG_INLINE const Scalar &coeffRef(Index rowId, Index colId) const + { + if (Flags & RowMajorBit) + return m_storage.data()[colId + rowId * m_storage.cols()]; + else// column-major + return m_storage.data()[rowId + colId * m_storage.rows()]; + } - /** \internal */ - template - EIGEN_STRONG_INLINE void writePacket(Index rowId, Index colId, const PacketScalar& val) - { - internal::pstoret - (m_storage.data() + (Flags & RowMajorBit - ? colId + rowId * m_storage.cols() - : rowId + colId * m_storage.rows()), val); - } + /** This is the const version of coeffRef(Index) which is thus synonym of coeff(Index). + * It is provided for convenience. */ + EIGEN_DEVICE_FUNC + EIGEN_STRONG_INLINE const Scalar &coeffRef(Index index) const { return m_storage.data()[index]; } - /** \internal */ - template - EIGEN_STRONG_INLINE void writePacket(Index index, const PacketScalar& val) - { - internal::pstoret(m_storage.data() + index, val); - } + /** \internal */ + template EIGEN_STRONG_INLINE PacketScalar packet(Index rowId, Index colId) const + { + return internal::ploadt( + m_storage.data() + (Flags & RowMajorBit ? colId + rowId * m_storage.cols() : rowId + colId * m_storage.rows())); + } - /** \returns a const pointer to the data array of this matrix */ - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const Scalar *data() const - { return m_storage.data(); } - - /** \returns a pointer to the data array of this matrix */ - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Scalar *data() - { return m_storage.data(); } - - /** Resizes \c *this to a \a rows x \a cols matrix. - * - * This method is intended for dynamic-size matrices, although it is legal to call it on any - * matrix as long as fixed dimensions are left unchanged. If you only want to change the number - * of rows and/or of columns, you can use resize(NoChange_t, Index), resize(Index, NoChange_t). - * - * If the current number of coefficients of \c *this exactly matches the - * product \a rows * \a cols, then no memory allocation is performed and - * the current values are left unchanged. In all other cases, including - * shrinking, the data is reallocated and all previous values are lost. - * - * Example: \include Matrix_resize_int_int.cpp - * Output: \verbinclude Matrix_resize_int_int.out - * - * \sa resize(Index) for vectors, resize(NoChange_t, Index), resize(Index, NoChange_t) - */ - EIGEN_DEVICE_FUNC - EIGEN_STRONG_INLINE void resize(Index rows, Index cols) - { - eigen_assert( EIGEN_IMPLIES(RowsAtCompileTime!=Dynamic,rows==RowsAtCompileTime) - && EIGEN_IMPLIES(ColsAtCompileTime!=Dynamic,cols==ColsAtCompileTime) - && EIGEN_IMPLIES(RowsAtCompileTime==Dynamic && MaxRowsAtCompileTime!=Dynamic,rows<=MaxRowsAtCompileTime) - && EIGEN_IMPLIES(ColsAtCompileTime==Dynamic && MaxColsAtCompileTime!=Dynamic,cols<=MaxColsAtCompileTime) - && rows>=0 && cols>=0 && "Invalid sizes when resizing a matrix or array."); - internal::check_rows_cols_for_overflow::run(rows, cols); - #ifdef EIGEN_INITIALIZE_COEFFS - Index size = rows*cols; - bool size_changed = size != this->size(); - m_storage.resize(size, rows, cols); - if(size_changed) EIGEN_INITIALIZE_COEFFS_IF_THAT_OPTION_IS_ENABLED - #else - m_storage.resize(rows*cols, rows, cols); - #endif - } + /** \internal */ + template EIGEN_STRONG_INLINE PacketScalar packet(Index index) const + { + return internal::ploadt(m_storage.data() + index); + } - /** Resizes \c *this to a vector of length \a size - * - * \only_for_vectors. This method does not work for - * partially dynamic matrices when the static dimension is anything other - * than 1. For example it will not work with Matrix. - * - * Example: \include Matrix_resize_int.cpp - * Output: \verbinclude Matrix_resize_int.out - * - * \sa resize(Index,Index), resize(NoChange_t, Index), resize(Index, NoChange_t) - */ - EIGEN_DEVICE_FUNC - inline void resize(Index size) - { - EIGEN_STATIC_ASSERT_VECTOR_ONLY(PlainObjectBase) - eigen_assert(((SizeAtCompileTime == Dynamic && (MaxSizeAtCompileTime==Dynamic || size<=MaxSizeAtCompileTime)) || SizeAtCompileTime == size) && size>=0); - #ifdef EIGEN_INITIALIZE_COEFFS - bool size_changed = size != this->size(); - #endif - if(RowsAtCompileTime == 1) - m_storage.resize(size, 1, size); - else - m_storage.resize(size, size, 1); - #ifdef EIGEN_INITIALIZE_COEFFS - if(size_changed) EIGEN_INITIALIZE_COEFFS_IF_THAT_OPTION_IS_ENABLED - #endif - } + /** \internal */ + template EIGEN_STRONG_INLINE void writePacket(Index rowId, Index colId, const PacketScalar &val) + { + internal::pstoret( + m_storage.data() + (Flags & RowMajorBit ? colId + rowId * m_storage.cols() : rowId + colId * m_storage.rows()), + val); + } - /** Resizes the matrix, changing only the number of columns. For the parameter of type NoChange_t, just pass the special value \c NoChange - * as in the example below. - * - * Example: \include Matrix_resize_NoChange_int.cpp - * Output: \verbinclude Matrix_resize_NoChange_int.out - * - * \sa resize(Index,Index) - */ - EIGEN_DEVICE_FUNC - inline void resize(NoChange_t, Index cols) - { - resize(rows(), cols); - } + /** \internal */ + template EIGEN_STRONG_INLINE void writePacket(Index index, const PacketScalar &val) + { + internal::pstoret(m_storage.data() + index, val); + } - /** Resizes the matrix, changing only the number of rows. For the parameter of type NoChange_t, just pass the special value \c NoChange - * as in the example below. - * - * Example: \include Matrix_resize_int_NoChange.cpp - * Output: \verbinclude Matrix_resize_int_NoChange.out - * - * \sa resize(Index,Index) - */ - EIGEN_DEVICE_FUNC - inline void resize(Index rows, NoChange_t) - { - resize(rows, cols()); - } + /** \returns a const pointer to the data array of this matrix */ + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const Scalar *data() const { return m_storage.data(); } + + /** \returns a pointer to the data array of this matrix */ + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Scalar *data() { return m_storage.data(); } + + /** Resizes \c *this to a \a rows x \a cols matrix. + * + * This method is intended for dynamic-size matrices, although it is legal to call it on any + * matrix as long as fixed dimensions are left unchanged. If you only want to change the number + * of rows and/or of columns, you can use resize(NoChange_t, Index), resize(Index, NoChange_t). + * + * If the current number of coefficients of \c *this exactly matches the + * product \a rows * \a cols, then no memory allocation is performed and + * the current values are left unchanged. In all other cases, including + * shrinking, the data is reallocated and all previous values are lost. + * + * Example: \include Matrix_resize_int_int.cpp + * Output: \verbinclude Matrix_resize_int_int.out + * + * \sa resize(Index) for vectors, resize(NoChange_t, Index), resize(Index, NoChange_t) + */ + EIGEN_DEVICE_FUNC + EIGEN_STRONG_INLINE void resize(Index rows, Index cols) + { + eigen_assert( + EIGEN_IMPLIES(RowsAtCompileTime != Dynamic, rows == RowsAtCompileTime) + && EIGEN_IMPLIES(ColsAtCompileTime != Dynamic, cols == ColsAtCompileTime) + && EIGEN_IMPLIES(RowsAtCompileTime == Dynamic && MaxRowsAtCompileTime != Dynamic, rows <= MaxRowsAtCompileTime) + && EIGEN_IMPLIES(ColsAtCompileTime == Dynamic && MaxColsAtCompileTime != Dynamic, cols <= MaxColsAtCompileTime) + && rows >= 0 && cols >= 0 && "Invalid sizes when resizing a matrix or array."); + internal::check_rows_cols_for_overflow::run(rows, cols); +#ifdef EIGEN_INITIALIZE_COEFFS + Index size = rows * cols; + bool size_changed = size != this->size(); + m_storage.resize(size, rows, cols); + if (size_changed) EIGEN_INITIALIZE_COEFFS_IF_THAT_OPTION_IS_ENABLED +#else + m_storage.resize(rows * cols, rows, cols); +#endif + } - /** Resizes \c *this to have the same dimensions as \a other. - * Takes care of doing all the checking that's needed. - * - * Note that copying a row-vector into a vector (and conversely) is allowed. - * The resizing, if any, is then done in the appropriate way so that row-vectors - * remain row-vectors and vectors remain vectors. - */ - template - EIGEN_DEVICE_FUNC - EIGEN_STRONG_INLINE void resizeLike(const EigenBase& _other) - { - const OtherDerived& other = _other.derived(); - internal::check_rows_cols_for_overflow::run(other.rows(), other.cols()); - const Index othersize = other.rows()*other.cols(); - if(RowsAtCompileTime == 1) - { - eigen_assert(other.rows() == 1 || other.cols() == 1); - resize(1, othersize); - } - else if(ColsAtCompileTime == 1) - { - eigen_assert(other.rows() == 1 || other.cols() == 1); - resize(othersize, 1); - } - else resize(other.rows(), other.cols()); - } + /** Resizes \c *this to a vector of length \a size + * + * \only_for_vectors. This method does not work for + * partially dynamic matrices when the static dimension is anything other + * than 1. For example it will not work with Matrix. + * + * Example: \include Matrix_resize_int.cpp + * Output: \verbinclude Matrix_resize_int.out + * + * \sa resize(Index,Index), resize(NoChange_t, Index), resize(Index, NoChange_t) + */ + EIGEN_DEVICE_FUNC + inline void resize(Index size) + { + EIGEN_STATIC_ASSERT_VECTOR_ONLY(PlainObjectBase) + eigen_assert(((SizeAtCompileTime == Dynamic && (MaxSizeAtCompileTime == Dynamic || size <= MaxSizeAtCompileTime)) + || SizeAtCompileTime == size) + && size >= 0); +#ifdef EIGEN_INITIALIZE_COEFFS + bool size_changed = size != this->size(); +#endif + if (RowsAtCompileTime == 1) + m_storage.resize(size, 1, size); + else + m_storage.resize(size, size, 1); +#ifdef EIGEN_INITIALIZE_COEFFS + if (size_changed) EIGEN_INITIALIZE_COEFFS_IF_THAT_OPTION_IS_ENABLED +#endif + } - /** Resizes the matrix to \a rows x \a cols while leaving old values untouched. - * - * The method is intended for matrices of dynamic size. If you only want to change the number - * of rows and/or of columns, you can use conservativeResize(NoChange_t, Index) or - * conservativeResize(Index, NoChange_t). - * - * Matrices are resized relative to the top-left element. In case values need to be - * appended to the matrix they will be uninitialized. - */ - EIGEN_DEVICE_FUNC - EIGEN_STRONG_INLINE void conservativeResize(Index rows, Index cols) - { - internal::conservative_resize_like_impl::run(*this, rows, cols); - } + /** Resizes the matrix, changing only the number of columns. For the parameter of type NoChange_t, just pass the + * special value \c NoChange as in the example below. + * + * Example: \include Matrix_resize_NoChange_int.cpp + * Output: \verbinclude Matrix_resize_NoChange_int.out + * + * \sa resize(Index,Index) + */ + EIGEN_DEVICE_FUNC + inline void resize(NoChange_t, Index cols) { resize(rows(), cols); } + + /** Resizes the matrix, changing only the number of rows. For the parameter of type NoChange_t, just pass the special + * value \c NoChange as in the example below. + * + * Example: \include Matrix_resize_int_NoChange.cpp + * Output: \verbinclude Matrix_resize_int_NoChange.out + * + * \sa resize(Index,Index) + */ + EIGEN_DEVICE_FUNC + inline void resize(Index rows, NoChange_t) { resize(rows, cols()); } + + /** Resizes \c *this to have the same dimensions as \a other. + * Takes care of doing all the checking that's needed. + * + * Note that copying a row-vector into a vector (and conversely) is allowed. + * The resizing, if any, is then done in the appropriate way so that row-vectors + * remain row-vectors and vectors remain vectors. + */ + template + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void resizeLike(const EigenBase &_other) + { + const OtherDerived &other = _other.derived(); + internal::check_rows_cols_for_overflow::run(other.rows(), other.cols()); + const Index othersize = other.rows() * other.cols(); + if (RowsAtCompileTime == 1) { + eigen_assert(other.rows() == 1 || other.cols() == 1); + resize(1, othersize); + } else if (ColsAtCompileTime == 1) { + eigen_assert(other.rows() == 1 || other.cols() == 1); + resize(othersize, 1); + } else + resize(other.rows(), other.cols()); + } - /** Resizes the matrix to \a rows x \a cols while leaving old values untouched. - * - * As opposed to conservativeResize(Index rows, Index cols), this version leaves - * the number of columns unchanged. - * - * In case the matrix is growing, new rows will be uninitialized. - */ - EIGEN_DEVICE_FUNC - EIGEN_STRONG_INLINE void conservativeResize(Index rows, NoChange_t) - { - // Note: see the comment in conservativeResize(Index,Index) - conservativeResize(rows, cols()); - } + /** Resizes the matrix to \a rows x \a cols while leaving old values untouched. + * + * The method is intended for matrices of dynamic size. If you only want to change the number + * of rows and/or of columns, you can use conservativeResize(NoChange_t, Index) or + * conservativeResize(Index, NoChange_t). + * + * Matrices are resized relative to the top-left element. In case values need to be + * appended to the matrix they will be uninitialized. + */ + EIGEN_DEVICE_FUNC + EIGEN_STRONG_INLINE void conservativeResize(Index rows, Index cols) + { + internal::conservative_resize_like_impl::run(*this, rows, cols); + } - /** Resizes the matrix to \a rows x \a cols while leaving old values untouched. - * - * As opposed to conservativeResize(Index rows, Index cols), this version leaves - * the number of rows unchanged. - * - * In case the matrix is growing, new columns will be uninitialized. - */ - EIGEN_DEVICE_FUNC - EIGEN_STRONG_INLINE void conservativeResize(NoChange_t, Index cols) - { - // Note: see the comment in conservativeResize(Index,Index) - conservativeResize(rows(), cols); - } + /** Resizes the matrix to \a rows x \a cols while leaving old values untouched. + * + * As opposed to conservativeResize(Index rows, Index cols), this version leaves + * the number of columns unchanged. + * + * In case the matrix is growing, new rows will be uninitialized. + */ + EIGEN_DEVICE_FUNC + EIGEN_STRONG_INLINE void conservativeResize(Index rows, NoChange_t) + { + // Note: see the comment in conservativeResize(Index,Index) + conservativeResize(rows, cols()); + } - /** Resizes the vector to \a size while retaining old values. - * - * \only_for_vectors. This method does not work for - * partially dynamic matrices when the static dimension is anything other - * than 1. For example it will not work with Matrix. - * - * When values are appended, they will be uninitialized. - */ - EIGEN_DEVICE_FUNC - EIGEN_STRONG_INLINE void conservativeResize(Index size) - { - internal::conservative_resize_like_impl::run(*this, size); - } + /** Resizes the matrix to \a rows x \a cols while leaving old values untouched. + * + * As opposed to conservativeResize(Index rows, Index cols), this version leaves + * the number of rows unchanged. + * + * In case the matrix is growing, new columns will be uninitialized. + */ + EIGEN_DEVICE_FUNC + EIGEN_STRONG_INLINE void conservativeResize(NoChange_t, Index cols) + { + // Note: see the comment in conservativeResize(Index,Index) + conservativeResize(rows(), cols); + } - /** Resizes the matrix to \a rows x \a cols of \c other, while leaving old values untouched. - * - * The method is intended for matrices of dynamic size. If you only want to change the number - * of rows and/or of columns, you can use conservativeResize(NoChange_t, Index) or - * conservativeResize(Index, NoChange_t). - * - * Matrices are resized relative to the top-left element. In case values need to be - * appended to the matrix they will copied from \c other. - */ - template - EIGEN_DEVICE_FUNC - EIGEN_STRONG_INLINE void conservativeResizeLike(const DenseBase& other) - { - internal::conservative_resize_like_impl::run(*this, other); - } + /** Resizes the vector to \a size while retaining old values. + * + * \only_for_vectors. This method does not work for + * partially dynamic matrices when the static dimension is anything other + * than 1. For example it will not work with Matrix. + * + * When values are appended, they will be uninitialized. + */ + EIGEN_DEVICE_FUNC + EIGEN_STRONG_INLINE void conservativeResize(Index size) + { + internal::conservative_resize_like_impl::run(*this, size); + } - /** This is a special case of the templated operator=. Its purpose is to - * prevent a default operator= from hiding the templated operator=. - */ - EIGEN_DEVICE_FUNC - EIGEN_STRONG_INLINE Derived& operator=(const PlainObjectBase& other) - { - return _set(other); - } + /** Resizes the matrix to \a rows x \a cols of \c other, while leaving old values untouched. + * + * The method is intended for matrices of dynamic size. If you only want to change the number + * of rows and/or of columns, you can use conservativeResize(NoChange_t, Index) or + * conservativeResize(Index, NoChange_t). + * + * Matrices are resized relative to the top-left element. In case values need to be + * appended to the matrix they will copied from \c other. + */ + template + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void conservativeResizeLike(const DenseBase &other) + { + internal::conservative_resize_like_impl::run(*this, other); + } - /** \sa MatrixBase::lazyAssign() */ - template - EIGEN_DEVICE_FUNC - EIGEN_STRONG_INLINE Derived& lazyAssign(const DenseBase& other) - { - _resize_to_match(other); - return Base::lazyAssign(other.derived()); - } + /** This is a special case of the templated operator=. Its purpose is to + * prevent a default operator= from hiding the templated operator=. + */ + EIGEN_DEVICE_FUNC + EIGEN_STRONG_INLINE Derived &operator=(const PlainObjectBase &other) { return _set(other); } - template - EIGEN_DEVICE_FUNC - EIGEN_STRONG_INLINE Derived& operator=(const ReturnByValue& func) - { - resize(func.rows(), func.cols()); - return Base::operator=(func); - } + /** \sa MatrixBase::lazyAssign() */ + template + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Derived &lazyAssign(const DenseBase &other) + { + _resize_to_match(other); + return Base::lazyAssign(other.derived()); + } - // Prevent user from trying to instantiate PlainObjectBase objects - // by making all its constructor protected. See bug 1074. - protected: + template + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Derived &operator=(const ReturnByValue &func) + { + resize(func.rows(), func.cols()); + return Base::operator=(func); + } - EIGEN_DEVICE_FUNC - EIGEN_STRONG_INLINE PlainObjectBase() : m_storage() - { -// _check_template_params(); -// EIGEN_INITIALIZE_COEFFS_IF_THAT_OPTION_IS_ENABLED - } + // Prevent user from trying to instantiate PlainObjectBase objects + // by making all its constructor protected. See bug 1074. +protected: + EIGEN_DEVICE_FUNC + EIGEN_STRONG_INLINE PlainObjectBase() : m_storage() + { + // _check_template_params(); + // EIGEN_INITIALIZE_COEFFS_IF_THAT_OPTION_IS_ENABLED + } #ifndef EIGEN_PARSED_BY_DOXYGEN - // FIXME is it still needed ? - /** \internal */ - EIGEN_DEVICE_FUNC - explicit PlainObjectBase(internal::constructor_without_unaligned_array_assert) - : m_storage(internal::constructor_without_unaligned_array_assert()) - { -// _check_template_params(); EIGEN_INITIALIZE_COEFFS_IF_THAT_OPTION_IS_ENABLED - } + // FIXME is it still needed ? + /** \internal */ + EIGEN_DEVICE_FUNC + explicit PlainObjectBase(internal::constructor_without_unaligned_array_assert) + : m_storage(internal::constructor_without_unaligned_array_assert()) + { + // _check_template_params(); EIGEN_INITIALIZE_COEFFS_IF_THAT_OPTION_IS_ENABLED + } #endif #if EIGEN_HAS_RVALUE_REFERENCES - EIGEN_DEVICE_FUNC - PlainObjectBase(PlainObjectBase&& other) EIGEN_NOEXCEPT - : m_storage( std::move(other.m_storage) ) - { - } + EIGEN_DEVICE_FUNC + PlainObjectBase(PlainObjectBase &&other) EIGEN_NOEXCEPT : m_storage(std::move(other.m_storage)) {} - EIGEN_DEVICE_FUNC - PlainObjectBase& operator=(PlainObjectBase&& other) EIGEN_NOEXCEPT - { - using std::swap; - swap(m_storage, other.m_storage); - return *this; - } + EIGEN_DEVICE_FUNC + PlainObjectBase &operator=(PlainObjectBase &&other) EIGEN_NOEXCEPT + { + using std::swap; + swap(m_storage, other.m_storage); + return *this; + } #endif - /** Copy constructor */ - EIGEN_DEVICE_FUNC - EIGEN_STRONG_INLINE PlainObjectBase(const PlainObjectBase& other) - : Base(), m_storage(other.m_storage) { } - EIGEN_DEVICE_FUNC - EIGEN_STRONG_INLINE PlainObjectBase(Index size, Index rows, Index cols) - : m_storage(size, rows, cols) - { -// _check_template_params(); -// EIGEN_INITIALIZE_COEFFS_IF_THAT_OPTION_IS_ENABLED - } + /** Copy constructor */ + EIGEN_DEVICE_FUNC + EIGEN_STRONG_INLINE PlainObjectBase(const PlainObjectBase &other) : Base(), m_storage(other.m_storage) {} + EIGEN_DEVICE_FUNC + EIGEN_STRONG_INLINE PlainObjectBase(Index size, Index rows, Index cols) : m_storage(size, rows, cols) + { + // _check_template_params(); + // EIGEN_INITIALIZE_COEFFS_IF_THAT_OPTION_IS_ENABLED + } - /** \sa PlainObjectBase::operator=(const EigenBase&) */ - template - EIGEN_DEVICE_FUNC - EIGEN_STRONG_INLINE PlainObjectBase(const DenseBase &other) - : m_storage() - { - _check_template_params(); - resizeLike(other); - _set_noalias(other); - } + /** \sa PlainObjectBase::operator=(const EigenBase&) */ + template + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE PlainObjectBase(const DenseBase &other) : m_storage() + { + _check_template_params(); + resizeLike(other); + _set_noalias(other); + } - /** \sa PlainObjectBase::operator=(const EigenBase&) */ - template - EIGEN_DEVICE_FUNC - EIGEN_STRONG_INLINE PlainObjectBase(const EigenBase &other) - : m_storage() - { - _check_template_params(); - resizeLike(other); - *this = other.derived(); - } - /** \brief Copy constructor with in-place evaluation */ - template - EIGEN_DEVICE_FUNC - EIGEN_STRONG_INLINE PlainObjectBase(const ReturnByValue& other) - { - _check_template_params(); - // FIXME this does not automatically transpose vectors if necessary - resize(other.rows(), other.cols()); - other.evalTo(this->derived()); - } + /** \sa PlainObjectBase::operator=(const EigenBase&) */ + template + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE PlainObjectBase(const EigenBase &other) : m_storage() + { + _check_template_params(); + resizeLike(other); + *this = other.derived(); + } + /** \brief Copy constructor with in-place evaluation */ + template + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE PlainObjectBase(const ReturnByValue &other) + { + _check_template_params(); + // FIXME this does not automatically transpose vectors if necessary + resize(other.rows(), other.cols()); + other.evalTo(this->derived()); + } - public: +public: + /** \brief Copies the generic expression \a other into *this. + * \copydetails DenseBase::operator=(const EigenBase &other) + */ + template + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Derived &operator=(const EigenBase &other) + { + _resize_to_match(other); + Base::operator=(other.derived()); + return this->derived(); + } - /** \brief Copies the generic expression \a other into *this. - * \copydetails DenseBase::operator=(const EigenBase &other) - */ - template - EIGEN_DEVICE_FUNC - EIGEN_STRONG_INLINE Derived& operator=(const EigenBase &other) - { - _resize_to_match(other); - Base::operator=(other.derived()); - return this->derived(); - } + /** \name Map + * These are convenience functions returning Map objects. The Map() static functions return unaligned Map objects, + * while the AlignedMap() functions return aligned Map objects and thus should be called only with 16-byte-aligned + * \a data pointers. + * + * Here is an example using strides: + * \include Matrix_Map_stride.cpp + * Output: \verbinclude Matrix_Map_stride.out + * + * \see class Map + */ + //@{ + static inline ConstMapType Map(const Scalar *data) { return ConstMapType(data); } + static inline MapType Map(Scalar *data) { return MapType(data); } + static inline ConstMapType Map(const Scalar *data, Index size) { return ConstMapType(data, size); } + static inline MapType Map(Scalar *data, Index size) { return MapType(data, size); } + static inline ConstMapType Map(const Scalar *data, Index rows, Index cols) { return ConstMapType(data, rows, cols); } + static inline MapType Map(Scalar *data, Index rows, Index cols) { return MapType(data, rows, cols); } + + static inline ConstAlignedMapType MapAligned(const Scalar *data) { return ConstAlignedMapType(data); } + static inline AlignedMapType MapAligned(Scalar *data) { return AlignedMapType(data); } + static inline ConstAlignedMapType MapAligned(const Scalar *data, Index size) + { + return ConstAlignedMapType(data, size); + } + static inline AlignedMapType MapAligned(Scalar *data, Index size) { return AlignedMapType(data, size); } + static inline ConstAlignedMapType MapAligned(const Scalar *data, Index rows, Index cols) + { + return ConstAlignedMapType(data, rows, cols); + } + static inline AlignedMapType MapAligned(Scalar *data, Index rows, Index cols) + { + return AlignedMapType(data, rows, cols); + } - /** \name Map - * These are convenience functions returning Map objects. The Map() static functions return unaligned Map objects, - * while the AlignedMap() functions return aligned Map objects and thus should be called only with 16-byte-aligned - * \a data pointers. - * - * Here is an example using strides: - * \include Matrix_Map_stride.cpp - * Output: \verbinclude Matrix_Map_stride.out - * - * \see class Map - */ - //@{ - static inline ConstMapType Map(const Scalar* data) - { return ConstMapType(data); } - static inline MapType Map(Scalar* data) - { return MapType(data); } - static inline ConstMapType Map(const Scalar* data, Index size) - { return ConstMapType(data, size); } - static inline MapType Map(Scalar* data, Index size) - { return MapType(data, size); } - static inline ConstMapType Map(const Scalar* data, Index rows, Index cols) - { return ConstMapType(data, rows, cols); } - static inline MapType Map(Scalar* data, Index rows, Index cols) - { return MapType(data, rows, cols); } - - static inline ConstAlignedMapType MapAligned(const Scalar* data) - { return ConstAlignedMapType(data); } - static inline AlignedMapType MapAligned(Scalar* data) - { return AlignedMapType(data); } - static inline ConstAlignedMapType MapAligned(const Scalar* data, Index size) - { return ConstAlignedMapType(data, size); } - static inline AlignedMapType MapAligned(Scalar* data, Index size) - { return AlignedMapType(data, size); } - static inline ConstAlignedMapType MapAligned(const Scalar* data, Index rows, Index cols) - { return ConstAlignedMapType(data, rows, cols); } - static inline AlignedMapType MapAligned(Scalar* data, Index rows, Index cols) - { return AlignedMapType(data, rows, cols); } - - template - static inline typename StridedConstMapType >::type Map(const Scalar* data, const Stride& stride) - { return typename StridedConstMapType >::type(data, stride); } - template - static inline typename StridedMapType >::type Map(Scalar* data, const Stride& stride) - { return typename StridedMapType >::type(data, stride); } - template - static inline typename StridedConstMapType >::type Map(const Scalar* data, Index size, const Stride& stride) - { return typename StridedConstMapType >::type(data, size, stride); } - template - static inline typename StridedMapType >::type Map(Scalar* data, Index size, const Stride& stride) - { return typename StridedMapType >::type(data, size, stride); } - template - static inline typename StridedConstMapType >::type Map(const Scalar* data, Index rows, Index cols, const Stride& stride) - { return typename StridedConstMapType >::type(data, rows, cols, stride); } - template - static inline typename StridedMapType >::type Map(Scalar* data, Index rows, Index cols, const Stride& stride) - { return typename StridedMapType >::type(data, rows, cols, stride); } - - template - static inline typename StridedConstAlignedMapType >::type MapAligned(const Scalar* data, const Stride& stride) - { return typename StridedConstAlignedMapType >::type(data, stride); } - template - static inline typename StridedAlignedMapType >::type MapAligned(Scalar* data, const Stride& stride) - { return typename StridedAlignedMapType >::type(data, stride); } - template - static inline typename StridedConstAlignedMapType >::type MapAligned(const Scalar* data, Index size, const Stride& stride) - { return typename StridedConstAlignedMapType >::type(data, size, stride); } - template - static inline typename StridedAlignedMapType >::type MapAligned(Scalar* data, Index size, const Stride& stride) - { return typename StridedAlignedMapType >::type(data, size, stride); } - template - static inline typename StridedConstAlignedMapType >::type MapAligned(const Scalar* data, Index rows, Index cols, const Stride& stride) - { return typename StridedConstAlignedMapType >::type(data, rows, cols, stride); } - template - static inline typename StridedAlignedMapType >::type MapAligned(Scalar* data, Index rows, Index cols, const Stride& stride) - { return typename StridedAlignedMapType >::type(data, rows, cols, stride); } - //@} - - using Base::setConstant; - EIGEN_DEVICE_FUNC Derived& setConstant(Index size, const Scalar& val); - EIGEN_DEVICE_FUNC Derived& setConstant(Index rows, Index cols, const Scalar& val); - - using Base::setZero; - EIGEN_DEVICE_FUNC Derived& setZero(Index size); - EIGEN_DEVICE_FUNC Derived& setZero(Index rows, Index cols); - - using Base::setOnes; - EIGEN_DEVICE_FUNC Derived& setOnes(Index size); - EIGEN_DEVICE_FUNC Derived& setOnes(Index rows, Index cols); - - using Base::setRandom; - Derived& setRandom(Index size); - Derived& setRandom(Index rows, Index cols); - - #ifdef EIGEN_PLAINOBJECTBASE_PLUGIN - #include EIGEN_PLAINOBJECTBASE_PLUGIN - #endif - - protected: - /** \internal Resizes *this in preparation for assigning \a other to it. - * Takes care of doing all the checking that's needed. - * - * Note that copying a row-vector into a vector (and conversely) is allowed. - * The resizing, if any, is then done in the appropriate way so that row-vectors - * remain row-vectors and vectors remain vectors. - */ - template - EIGEN_DEVICE_FUNC - EIGEN_STRONG_INLINE void _resize_to_match(const EigenBase& other) - { - #ifdef EIGEN_NO_AUTOMATIC_RESIZING - eigen_assert((this->size()==0 || (IsVectorAtCompileTime ? (this->size() == other.size()) - : (rows() == other.rows() && cols() == other.cols()))) - && "Size mismatch. Automatic resizing is disabled because EIGEN_NO_AUTOMATIC_RESIZING is defined"); - EIGEN_ONLY_USED_FOR_DEBUG(other); - #else - resizeLike(other); - #endif - } + template + static inline typename StridedConstMapType>::type Map(const Scalar *data, + const Stride &stride) + { + return typename StridedConstMapType>::type(data, stride); + } + template + static inline typename StridedMapType>::type Map(Scalar *data, + const Stride &stride) + { + return typename StridedMapType>::type(data, stride); + } + template + static inline typename StridedConstMapType>::type + Map(const Scalar *data, Index size, const Stride &stride) + { + return typename StridedConstMapType>::type(data, size, stride); + } + template + static inline typename StridedMapType>::type + Map(Scalar *data, Index size, const Stride &stride) + { + return typename StridedMapType>::type(data, size, stride); + } + template + static inline typename StridedConstMapType>::type + Map(const Scalar *data, Index rows, Index cols, const Stride &stride) + { + return typename StridedConstMapType>::type(data, rows, cols, stride); + } + template + static inline typename StridedMapType>::type + Map(Scalar *data, Index rows, Index cols, const Stride &stride) + { + return typename StridedMapType>::type(data, rows, cols, stride); + } - /** - * \brief Copies the value of the expression \a other into \c *this with automatic resizing. - * - * *this might be resized to match the dimensions of \a other. If *this was a null matrix (not already initialized), - * it will be initialized. - * - * Note that copying a row-vector into a vector (and conversely) is allowed. - * The resizing, if any, is then done in the appropriate way so that row-vectors - * remain row-vectors and vectors remain vectors. - * - * \sa operator=(const MatrixBase&), _set_noalias() - * - * \internal - */ - // aliasing is dealt once in internall::call_assignment - // so at this stage we have to assume aliasing... and resising has to be done later. - template - EIGEN_DEVICE_FUNC - EIGEN_STRONG_INLINE Derived& _set(const DenseBase& other) - { - internal::call_assignment(this->derived(), other.derived()); - return this->derived(); - } + template + static inline typename StridedConstAlignedMapType>::type MapAligned(const Scalar *data, + const Stride &stride) + { + return typename StridedConstAlignedMapType>::type(data, stride); + } + template + static inline typename StridedAlignedMapType>::type MapAligned(Scalar *data, + const Stride &stride) + { + return typename StridedAlignedMapType>::type(data, stride); + } + template + static inline typename StridedConstAlignedMapType>::type + MapAligned(const Scalar *data, Index size, const Stride &stride) + { + return typename StridedConstAlignedMapType>::type(data, size, stride); + } + template + static inline typename StridedAlignedMapType>::type + MapAligned(Scalar *data, Index size, const Stride &stride) + { + return typename StridedAlignedMapType>::type(data, size, stride); + } + template + static inline typename StridedConstAlignedMapType>::type + MapAligned(const Scalar *data, Index rows, Index cols, const Stride &stride) + { + return typename StridedConstAlignedMapType>::type(data, rows, cols, stride); + } + template + static inline typename StridedAlignedMapType>::type + MapAligned(Scalar *data, Index rows, Index cols, const Stride &stride) + { + return typename StridedAlignedMapType>::type(data, rows, cols, stride); + } + //@} - /** \internal Like _set() but additionally makes the assumption that no aliasing effect can happen (which - * is the case when creating a new matrix) so one can enforce lazy evaluation. - * - * \sa operator=(const MatrixBase&), _set() - */ - template - EIGEN_DEVICE_FUNC - EIGEN_STRONG_INLINE Derived& _set_noalias(const DenseBase& other) - { - // I don't think we need this resize call since the lazyAssign will anyways resize - // and lazyAssign will be called by the assign selector. - //_resize_to_match(other); - // the 'false' below means to enforce lazy evaluation. We don't use lazyAssign() because - // it wouldn't allow to copy a row-vector into a column-vector. - internal::call_assignment_no_alias(this->derived(), other.derived(), internal::assign_op()); - return this->derived(); - } + using Base::setConstant; + EIGEN_DEVICE_FUNC Derived &setConstant(Index size, const Scalar &val); + EIGEN_DEVICE_FUNC Derived &setConstant(Index rows, Index cols, const Scalar &val); - template - EIGEN_DEVICE_FUNC - EIGEN_STRONG_INLINE void _init2(Index rows, Index cols, typename internal::enable_if::type* = 0) - { - EIGEN_STATIC_ASSERT(bool(NumTraits::IsInteger) && - bool(NumTraits::IsInteger), - FLOATING_POINT_ARGUMENT_PASSED__INTEGER_WAS_EXPECTED) - resize(rows,cols); - } - - template - EIGEN_DEVICE_FUNC - EIGEN_STRONG_INLINE void _init2(const T0& val0, const T1& val1, typename internal::enable_if::type* = 0) - { - EIGEN_STATIC_ASSERT_VECTOR_SPECIFIC_SIZE(PlainObjectBase, 2) - m_storage.data()[0] = Scalar(val0); - m_storage.data()[1] = Scalar(val1); - } - - template - EIGEN_DEVICE_FUNC - EIGEN_STRONG_INLINE void _init2(const Index& val0, const Index& val1, - typename internal::enable_if< (!internal::is_same::value) - && (internal::is_same::value) - && (internal::is_same::value) - && Base::SizeAtCompileTime==2,T1>::type* = 0) - { - EIGEN_STATIC_ASSERT_VECTOR_SPECIFIC_SIZE(PlainObjectBase, 2) - m_storage.data()[0] = Scalar(val0); - m_storage.data()[1] = Scalar(val1); - } + using Base::setZero; + EIGEN_DEVICE_FUNC Derived &setZero(Index size); + EIGEN_DEVICE_FUNC Derived &setZero(Index rows, Index cols); - // The argument is convertible to the Index type and we either have a non 1x1 Matrix, or a dynamic-sized Array, - // then the argument is meant to be the size of the object. - template - EIGEN_DEVICE_FUNC - EIGEN_STRONG_INLINE void _init1(Index size, typename internal::enable_if< (Base::SizeAtCompileTime!=1 || !internal::is_convertible::value) - && ((!internal::is_same::XprKind,ArrayXpr>::value || Base::SizeAtCompileTime==Dynamic)),T>::type* = 0) - { - // NOTE MSVC 2008 complains if we directly put bool(NumTraits::IsInteger) as the EIGEN_STATIC_ASSERT argument. - const bool is_integer = NumTraits::IsInteger; - EIGEN_UNUSED_VARIABLE(is_integer); - EIGEN_STATIC_ASSERT(is_integer, - FLOATING_POINT_ARGUMENT_PASSED__INTEGER_WAS_EXPECTED) - resize(size); - } - - // We have a 1x1 matrix/array => the argument is interpreted as the value of the unique coefficient (case where scalar type can be implicitely converted) - template - EIGEN_DEVICE_FUNC - EIGEN_STRONG_INLINE void _init1(const Scalar& val0, typename internal::enable_if::value,T>::type* = 0) - { - EIGEN_STATIC_ASSERT_VECTOR_SPECIFIC_SIZE(PlainObjectBase, 1) - m_storage.data()[0] = val0; - } - - // We have a 1x1 matrix/array => the argument is interpreted as the value of the unique coefficient (case where scalar type match the index type) - template - EIGEN_DEVICE_FUNC - EIGEN_STRONG_INLINE void _init1(const Index& val0, - typename internal::enable_if< (!internal::is_same::value) - && (internal::is_same::value) - && Base::SizeAtCompileTime==1 - && internal::is_convertible::value,T*>::type* = 0) - { - EIGEN_STATIC_ASSERT_VECTOR_SPECIFIC_SIZE(PlainObjectBase, 1) - m_storage.data()[0] = Scalar(val0); - } + using Base::setOnes; + EIGEN_DEVICE_FUNC Derived &setOnes(Index size); + EIGEN_DEVICE_FUNC Derived &setOnes(Index rows, Index cols); - // Initialize a fixed size matrix from a pointer to raw data - template - EIGEN_DEVICE_FUNC - EIGEN_STRONG_INLINE void _init1(const Scalar* data){ - this->_set_noalias(ConstMapType(data)); - } + using Base::setRandom; + Derived &setRandom(Index size); + Derived &setRandom(Index rows, Index cols); - // Initialize an arbitrary matrix from a dense expression - template - EIGEN_DEVICE_FUNC - EIGEN_STRONG_INLINE void _init1(const DenseBase& other){ - this->_set_noalias(other); - } +#ifdef EIGEN_PLAINOBJECTBASE_PLUGIN +#include EIGEN_PLAINOBJECTBASE_PLUGIN +#endif - // Initialize an arbitrary matrix from an object convertible to the Derived type. - template - EIGEN_DEVICE_FUNC - EIGEN_STRONG_INLINE void _init1(const Derived& other){ - this->_set_noalias(other); - } +protected: + /** \internal Resizes *this in preparation for assigning \a other to it. + * Takes care of doing all the checking that's needed. + * + * Note that copying a row-vector into a vector (and conversely) is allowed. + * The resizing, if any, is then done in the appropriate way so that row-vectors + * remain row-vectors and vectors remain vectors. + */ + template + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void _resize_to_match(const EigenBase &other) + { +#ifdef EIGEN_NO_AUTOMATIC_RESIZING + eigen_assert((this->size() == 0 + || (IsVectorAtCompileTime ? (this->size() == other.size()) + : (rows() == other.rows() && cols() == other.cols()))) + && "Size mismatch. Automatic resizing is disabled because EIGEN_NO_AUTOMATIC_RESIZING is defined"); + EIGEN_ONLY_USED_FOR_DEBUG(other); +#else + resizeLike(other); +#endif + } - // Initialize an arbitrary matrix from a generic Eigen expression - template - EIGEN_DEVICE_FUNC - EIGEN_STRONG_INLINE void _init1(const EigenBase& other){ - this->derived() = other; - } + /** + * \brief Copies the value of the expression \a other into \c *this with automatic resizing. + * + * *this might be resized to match the dimensions of \a other. If *this was a null matrix (not already initialized), + * it will be initialized. + * + * Note that copying a row-vector into a vector (and conversely) is allowed. + * The resizing, if any, is then done in the appropriate way so that row-vectors + * remain row-vectors and vectors remain vectors. + * + * \sa operator=(const MatrixBase&), _set_noalias() + * + * \internal + */ + // aliasing is dealt once in internall::call_assignment + // so at this stage we have to assume aliasing... and resising has to be done later. + template + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Derived &_set(const DenseBase &other) + { + internal::call_assignment(this->derived(), other.derived()); + return this->derived(); + } - template - EIGEN_DEVICE_FUNC - EIGEN_STRONG_INLINE void _init1(const ReturnByValue& other) - { - resize(other.rows(), other.cols()); - other.evalTo(this->derived()); - } + /** \internal Like _set() but additionally makes the assumption that no aliasing effect can happen (which + * is the case when creating a new matrix) so one can enforce lazy evaluation. + * + * \sa operator=(const MatrixBase&), _set() + */ + template + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Derived &_set_noalias(const DenseBase &other) + { + // I don't think we need this resize call since the lazyAssign will anyways resize + // and lazyAssign will be called by the assign selector. + //_resize_to_match(other); + // the 'false' below means to enforce lazy evaluation. We don't use lazyAssign() because + // it wouldn't allow to copy a row-vector into a column-vector. + internal::call_assignment_no_alias( + this->derived(), other.derived(), internal::assign_op()); + return this->derived(); + } - template - EIGEN_DEVICE_FUNC - EIGEN_STRONG_INLINE void _init1(const RotationBase& r) - { - this->derived() = r; - } - - // For fixed-size Array - template - EIGEN_DEVICE_FUNC - EIGEN_STRONG_INLINE void _init1(const Scalar& val0, - typename internal::enable_if< Base::SizeAtCompileTime!=Dynamic - && Base::SizeAtCompileTime!=1 - && internal::is_convertible::value - && internal::is_same::XprKind,ArrayXpr>::value,T>::type* = 0) - { - Base::setConstant(val0); - } - - // For fixed-size Array - template - EIGEN_DEVICE_FUNC - EIGEN_STRONG_INLINE void _init1(const Index& val0, - typename internal::enable_if< (!internal::is_same::value) - && (internal::is_same::value) - && Base::SizeAtCompileTime!=Dynamic - && Base::SizeAtCompileTime!=1 - && internal::is_convertible::value - && internal::is_same::XprKind,ArrayXpr>::value,T*>::type* = 0) - { - Base::setConstant(val0); - } - - template - friend struct internal::matrix_swap_impl; + template + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void + _init2(Index rows, Index cols, typename internal::enable_if::type * = 0) + { + EIGEN_STATIC_ASSERT(bool(NumTraits::IsInteger) && bool(NumTraits::IsInteger), + FLOATING_POINT_ARGUMENT_PASSED__INTEGER_WAS_EXPECTED) + resize(rows, cols); + } - public: - -#ifndef EIGEN_PARSED_BY_DOXYGEN - /** \internal - * \brief Override DenseBase::swap() since for dynamic-sized matrices - * of same type it is enough to swap the data pointers. - */ - template - EIGEN_DEVICE_FUNC - void swap(DenseBase & other) - { - enum { SwapPointers = internal::is_same::value && Base::SizeAtCompileTime==Dynamic }; - internal::matrix_swap_impl::run(this->derived(), other.derived()); - } - - /** \internal - * \brief const version forwarded to DenseBase::swap - */ - template - EIGEN_DEVICE_FUNC - void swap(DenseBase const & other) - { Base::swap(other.derived()); } - - EIGEN_DEVICE_FUNC - static EIGEN_STRONG_INLINE void _check_template_params() - { - EIGEN_STATIC_ASSERT((EIGEN_IMPLIES(MaxRowsAtCompileTime==1 && MaxColsAtCompileTime!=1, (Options&RowMajor)==RowMajor) - && EIGEN_IMPLIES(MaxColsAtCompileTime==1 && MaxRowsAtCompileTime!=1, (Options&RowMajor)==0) - && ((RowsAtCompileTime == Dynamic) || (RowsAtCompileTime >= 0)) - && ((ColsAtCompileTime == Dynamic) || (ColsAtCompileTime >= 0)) - && ((MaxRowsAtCompileTime == Dynamic) || (MaxRowsAtCompileTime >= 0)) - && ((MaxColsAtCompileTime == Dynamic) || (MaxColsAtCompileTime >= 0)) - && (MaxRowsAtCompileTime == RowsAtCompileTime || RowsAtCompileTime==Dynamic) - && (MaxColsAtCompileTime == ColsAtCompileTime || ColsAtCompileTime==Dynamic) - && (Options & (DontAlign|RowMajor)) == Options), - INVALID_MATRIX_TEMPLATE_PARAMETERS) - } + template + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void + _init2(const T0 &val0, const T1 &val1, typename internal::enable_if::type * = 0) + { + EIGEN_STATIC_ASSERT_VECTOR_SPECIFIC_SIZE(PlainObjectBase, 2) + m_storage.data()[0] = Scalar(val0); + m_storage.data()[1] = Scalar(val1); + } - enum { IsPlainObjectBase = 1 }; -#endif -}; + template + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void _init2(const Index &val0, + const Index &val1, + typename internal::enable_if<(!internal::is_same::value) && (internal::is_same::value) + && (internal::is_same::value) && Base::SizeAtCompileTime == 2, + T1>::type * = 0) + { + EIGEN_STATIC_ASSERT_VECTOR_SPECIFIC_SIZE(PlainObjectBase, 2) + m_storage.data()[0] = Scalar(val0); + m_storage.data()[1] = Scalar(val1); + } -namespace internal { + // The argument is convertible to the Index type and we either have a non 1x1 Matrix, or a dynamic-sized Array, + // then the argument is meant to be the size of the object. + template + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void _init1(Index size, + typename internal::enable_if<(Base::SizeAtCompileTime != 1 || !internal::is_convertible::value) + && ((!internal::is_same::XprKind, ArrayXpr>::value + || Base::SizeAtCompileTime == Dynamic)), + T>::type * = 0) + { + // NOTE MSVC 2008 complains if we directly put bool(NumTraits::IsInteger) as the EIGEN_STATIC_ASSERT argument. + const bool is_integer = NumTraits::IsInteger; + EIGEN_UNUSED_VARIABLE(is_integer); + EIGEN_STATIC_ASSERT(is_integer, FLOATING_POINT_ARGUMENT_PASSED__INTEGER_WAS_EXPECTED) + resize(size); + } -template -struct conservative_resize_like_impl -{ - static void run(DenseBase& _this, Index rows, Index cols) + // We have a 1x1 matrix/array => the argument is interpreted as the value of the unique coefficient (case where scalar + // type can be implicitely converted) + template + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void _init1(const Scalar &val0, + typename internal::enable_if::value, T>::type + * = 0) { - if (_this.rows() == rows && _this.cols() == cols) return; - EIGEN_STATIC_ASSERT_DYNAMIC_SIZE(Derived) + EIGEN_STATIC_ASSERT_VECTOR_SPECIFIC_SIZE(PlainObjectBase, 1) + m_storage.data()[0] = val0; + } - if ( ( Derived::IsRowMajor && _this.cols() == cols) || // row-major and we change only the number of rows - (!Derived::IsRowMajor && _this.rows() == rows) ) // column-major and we change only the number of columns - { - internal::check_rows_cols_for_overflow::run(rows, cols); - _this.derived().m_storage.conservativeResize(rows*cols,rows,cols); - } - else - { - // The storage order does not allow us to use reallocation. - typename Derived::PlainObject tmp(rows,cols); - const Index common_rows = numext::mini(rows, _this.rows()); - const Index common_cols = numext::mini(cols, _this.cols()); - tmp.block(0,0,common_rows,common_cols) = _this.block(0,0,common_rows,common_cols); - _this.derived().swap(tmp); - } + // We have a 1x1 matrix/array => the argument is interpreted as the value of the unique coefficient (case where scalar + // type match the index type) + template + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void _init1(const Index &val0, + typename internal::enable_if<(!internal::is_same::value) && (internal::is_same::value) + && Base::SizeAtCompileTime == 1 && internal::is_convertible::value, + T *>::type * = 0) + { + EIGEN_STATIC_ASSERT_VECTOR_SPECIFIC_SIZE(PlainObjectBase, 1) + m_storage.data()[0] = Scalar(val0); } - static void run(DenseBase& _this, const DenseBase& other) + // Initialize a fixed size matrix from a pointer to raw data + template EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void _init1(const Scalar *data) { - if (_this.rows() == other.rows() && _this.cols() == other.cols()) return; + this->_set_noalias(ConstMapType(data)); + } - // Note: Here is space for improvement. Basically, for conservativeResize(Index,Index), - // neither RowsAtCompileTime or ColsAtCompileTime must be Dynamic. If only one of the - // dimensions is dynamic, one could use either conservativeResize(Index rows, NoChange_t) or - // conservativeResize(NoChange_t, Index cols). For these methods new static asserts like - // EIGEN_STATIC_ASSERT_DYNAMIC_ROWS and EIGEN_STATIC_ASSERT_DYNAMIC_COLS would be good. - EIGEN_STATIC_ASSERT_DYNAMIC_SIZE(Derived) - EIGEN_STATIC_ASSERT_DYNAMIC_SIZE(OtherDerived) + // Initialize an arbitrary matrix from a dense expression + template + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void _init1(const DenseBase &other) + { + this->_set_noalias(other); + } - if ( ( Derived::IsRowMajor && _this.cols() == other.cols()) || // row-major and we change only the number of rows - (!Derived::IsRowMajor && _this.rows() == other.rows()) ) // column-major and we change only the number of columns - { - const Index new_rows = other.rows() - _this.rows(); - const Index new_cols = other.cols() - _this.cols(); - _this.derived().m_storage.conservativeResize(other.size(),other.rows(),other.cols()); - if (new_rows>0) - _this.bottomRightCorner(new_rows, other.cols()) = other.bottomRows(new_rows); - else if (new_cols>0) - _this.bottomRightCorner(other.rows(), new_cols) = other.rightCols(new_cols); - } - else - { - // The storage order does not allow us to use reallocation. - typename Derived::PlainObject tmp(other); - const Index common_rows = numext::mini(tmp.rows(), _this.rows()); - const Index common_cols = numext::mini(tmp.cols(), _this.cols()); - tmp.block(0,0,common_rows,common_cols) = _this.block(0,0,common_rows,common_cols); - _this.derived().swap(tmp); - } + // Initialize an arbitrary matrix from an object convertible to the Derived type. + template EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void _init1(const Derived &other) + { + this->_set_noalias(other); } -}; -// Here, the specialization for vectors inherits from the general matrix case -// to allow calling .conservativeResize(rows,cols) on vectors. -template -struct conservative_resize_like_impl - : conservative_resize_like_impl -{ - using conservative_resize_like_impl::run; - - static void run(DenseBase& _this, Index size) + // Initialize an arbitrary matrix from a generic Eigen expression + template + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void _init1(const EigenBase &other) { - const Index new_rows = Derived::RowsAtCompileTime==1 ? 1 : size; - const Index new_cols = Derived::RowsAtCompileTime==1 ? size : 1; - _this.derived().m_storage.conservativeResize(size,new_rows,new_cols); + this->derived() = other; } - static void run(DenseBase& _this, const DenseBase& other) + template + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void _init1(const ReturnByValue &other) { - if (_this.rows() == other.rows() && _this.cols() == other.cols()) return; + resize(other.rows(), other.cols()); + other.evalTo(this->derived()); + } - const Index num_new_elements = other.size() - _this.size(); + template + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void _init1(const RotationBase &r) + { + this->derived() = r; + } - const Index new_rows = Derived::RowsAtCompileTime==1 ? 1 : other.rows(); - const Index new_cols = Derived::RowsAtCompileTime==1 ? other.cols() : 1; - _this.derived().m_storage.conservativeResize(other.size(),new_rows,new_cols); + // For fixed-size Array + template + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void _init1(const Scalar &val0, + typename internal::enable_if::value + && internal::is_same::XprKind, ArrayXpr>::value, + T>::type * = 0) + { + Base::setConstant(val0); + } - if (num_new_elements > 0) - _this.tail(num_new_elements) = other.tail(num_new_elements); + // For fixed-size Array + template + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void _init1(const Index &val0, + typename internal::enable_if<(!internal::is_same::value) && (internal::is_same::value) + && Base::SizeAtCompileTime != Dynamic && Base::SizeAtCompileTime != 1 + && internal::is_convertible::value + && internal::is_same::XprKind, ArrayXpr>::value, + T *>::type * = 0) + { + Base::setConstant(val0); } -}; -template -struct matrix_swap_impl -{ - EIGEN_DEVICE_FUNC - static inline void run(MatrixTypeA& a, MatrixTypeB& b) + template friend struct internal::matrix_swap_impl; + +public: +#ifndef EIGEN_PARSED_BY_DOXYGEN + /** \internal + * \brief Override DenseBase::swap() since for dynamic-sized matrices + * of same type it is enough to swap the data pointers. + */ + template EIGEN_DEVICE_FUNC void swap(DenseBase &other) { - a.base().swap(b); + enum { SwapPointers = internal::is_same::value && Base::SizeAtCompileTime == Dynamic }; + internal::matrix_swap_impl::run(this->derived(), other.derived()); + } + + /** \internal + * \brief const version forwarded to DenseBase::swap + */ + template EIGEN_DEVICE_FUNC void swap(DenseBase const &other) + { + Base::swap(other.derived()); } -}; -template -struct matrix_swap_impl -{ EIGEN_DEVICE_FUNC - static inline void run(MatrixTypeA& a, MatrixTypeB& b) + static EIGEN_STRONG_INLINE void _check_template_params() { - static_cast(a).m_storage.swap(static_cast(b).m_storage); + EIGEN_STATIC_ASSERT( + (EIGEN_IMPLIES(MaxRowsAtCompileTime == 1 && MaxColsAtCompileTime != 1, (Options & RowMajor) == RowMajor) + && EIGEN_IMPLIES(MaxColsAtCompileTime == 1 && MaxRowsAtCompileTime != 1, (Options & RowMajor) == 0) + && ((RowsAtCompileTime == Dynamic) || (RowsAtCompileTime >= 0)) + && ((ColsAtCompileTime == Dynamic) || (ColsAtCompileTime >= 0)) + && ((MaxRowsAtCompileTime == Dynamic) || (MaxRowsAtCompileTime >= 0)) + && ((MaxColsAtCompileTime == Dynamic) || (MaxColsAtCompileTime >= 0)) + && (MaxRowsAtCompileTime == RowsAtCompileTime || RowsAtCompileTime == Dynamic) + && (MaxColsAtCompileTime == ColsAtCompileTime || ColsAtCompileTime == Dynamic) + && (Options & (DontAlign | RowMajor)) == Options), + INVALID_MATRIX_TEMPLATE_PARAMETERS) } + + enum { IsPlainObjectBase = 1 }; +#endif }; -} // end namespace internal +namespace internal { + + template struct conservative_resize_like_impl + { + static void run(DenseBase &_this, Index rows, Index cols) + { + if (_this.rows() == rows && _this.cols() == cols) return; + EIGEN_STATIC_ASSERT_DYNAMIC_SIZE(Derived) + + if ((Derived::IsRowMajor && _this.cols() == cols) ||// row-major and we change only the number of rows + (!Derived::IsRowMajor && _this.rows() == rows))// column-major and we change only the number of columns + { + internal::check_rows_cols_for_overflow::run(rows, cols); + _this.derived().m_storage.conservativeResize(rows * cols, rows, cols); + } else { + // The storage order does not allow us to use reallocation. + typename Derived::PlainObject tmp(rows, cols); + const Index common_rows = numext::mini(rows, _this.rows()); + const Index common_cols = numext::mini(cols, _this.cols()); + tmp.block(0, 0, common_rows, common_cols) = _this.block(0, 0, common_rows, common_cols); + _this.derived().swap(tmp); + } + } + + static void run(DenseBase &_this, const DenseBase &other) + { + if (_this.rows() == other.rows() && _this.cols() == other.cols()) return; + + // Note: Here is space for improvement. Basically, for conservativeResize(Index,Index), + // neither RowsAtCompileTime or ColsAtCompileTime must be Dynamic. If only one of the + // dimensions is dynamic, one could use either conservativeResize(Index rows, NoChange_t) or + // conservativeResize(NoChange_t, Index cols). For these methods new static asserts like + // EIGEN_STATIC_ASSERT_DYNAMIC_ROWS and EIGEN_STATIC_ASSERT_DYNAMIC_COLS would be good. + EIGEN_STATIC_ASSERT_DYNAMIC_SIZE(Derived) + EIGEN_STATIC_ASSERT_DYNAMIC_SIZE(OtherDerived) + + if ((Derived::IsRowMajor && _this.cols() == other.cols()) ||// row-major and we change only the number of rows + (!Derived::IsRowMajor + && _this.rows() == other.rows()))// column-major and we change only the number of columns + { + const Index new_rows = other.rows() - _this.rows(); + const Index new_cols = other.cols() - _this.cols(); + _this.derived().m_storage.conservativeResize(other.size(), other.rows(), other.cols()); + if (new_rows > 0) + _this.bottomRightCorner(new_rows, other.cols()) = other.bottomRows(new_rows); + else if (new_cols > 0) + _this.bottomRightCorner(other.rows(), new_cols) = other.rightCols(new_cols); + } else { + // The storage order does not allow us to use reallocation. + typename Derived::PlainObject tmp(other); + const Index common_rows = numext::mini(tmp.rows(), _this.rows()); + const Index common_cols = numext::mini(tmp.cols(), _this.cols()); + tmp.block(0, 0, common_rows, common_cols) = _this.block(0, 0, common_rows, common_cols); + _this.derived().swap(tmp); + } + } + }; + + // Here, the specialization for vectors inherits from the general matrix case + // to allow calling .conservativeResize(rows,cols) on vectors. + template + struct conservative_resize_like_impl + : conservative_resize_like_impl + { + using conservative_resize_like_impl::run; + + static void run(DenseBase &_this, Index size) + { + const Index new_rows = Derived::RowsAtCompileTime == 1 ? 1 : size; + const Index new_cols = Derived::RowsAtCompileTime == 1 ? size : 1; + _this.derived().m_storage.conservativeResize(size, new_rows, new_cols); + } + + static void run(DenseBase &_this, const DenseBase &other) + { + if (_this.rows() == other.rows() && _this.cols() == other.cols()) return; + + const Index num_new_elements = other.size() - _this.size(); + + const Index new_rows = Derived::RowsAtCompileTime == 1 ? 1 : other.rows(); + const Index new_cols = Derived::RowsAtCompileTime == 1 ? other.cols() : 1; + _this.derived().m_storage.conservativeResize(other.size(), new_rows, new_cols); + + if (num_new_elements > 0) _this.tail(num_new_elements) = other.tail(num_new_elements); + } + }; + + template struct matrix_swap_impl + { + EIGEN_DEVICE_FUNC + static inline void run(MatrixTypeA &a, MatrixTypeB &b) { a.base().swap(b); } + }; + + template struct matrix_swap_impl + { + EIGEN_DEVICE_FUNC + static inline void run(MatrixTypeA &a, MatrixTypeB &b) + { + static_cast(a).m_storage.swap( + static_cast(b).m_storage); + } + }; + +}// end namespace internal -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_DENSESTORAGEBASE_H +#endif// EIGEN_DENSESTORAGEBASE_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/Product.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/Product.h index 676c4802..b02fc331 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/Product.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/Product.h @@ -16,171 +16,166 @@ template class Pro namespace internal { -template -struct traits > -{ - typedef typename remove_all::type LhsCleaned; - typedef typename remove_all::type RhsCleaned; - typedef traits LhsTraits; - typedef traits RhsTraits; - - typedef MatrixXpr XprKind; - - typedef typename ScalarBinaryOpTraits::Scalar, typename traits::Scalar>::ReturnType Scalar; - typedef typename product_promote_storage_type::ret>::ret StorageKind; - typedef typename promote_index_type::type StorageIndex; - - enum { - RowsAtCompileTime = LhsTraits::RowsAtCompileTime, - ColsAtCompileTime = RhsTraits::ColsAtCompileTime, - MaxRowsAtCompileTime = LhsTraits::MaxRowsAtCompileTime, - MaxColsAtCompileTime = RhsTraits::MaxColsAtCompileTime, - - // FIXME: only needed by GeneralMatrixMatrixTriangular - InnerSize = EIGEN_SIZE_MIN_PREFER_FIXED(LhsTraits::ColsAtCompileTime, RhsTraits::RowsAtCompileTime), - - // The storage order is somewhat arbitrary here. The correct one will be determined through the evaluator. - Flags = (MaxRowsAtCompileTime==1 && MaxColsAtCompileTime!=1) ? RowMajorBit - : (MaxColsAtCompileTime==1 && MaxRowsAtCompileTime!=1) ? 0 - : ( ((LhsTraits::Flags&NoPreferredStorageOrderBit) && (RhsTraits::Flags&RowMajorBit)) - || ((RhsTraits::Flags&NoPreferredStorageOrderBit) && (LhsTraits::Flags&RowMajorBit)) ) ? RowMajorBit - : NoPreferredStorageOrderBit + template struct traits> + { + typedef typename remove_all::type LhsCleaned; + typedef typename remove_all::type RhsCleaned; + typedef traits LhsTraits; + typedef traits RhsTraits; + + typedef MatrixXpr XprKind; + + typedef typename ScalarBinaryOpTraits::Scalar, + typename traits::Scalar>::ReturnType Scalar; + typedef typename product_promote_storage_type::ret>::ret StorageKind; + typedef typename promote_index_type::type + StorageIndex; + + enum { + RowsAtCompileTime = LhsTraits::RowsAtCompileTime, + ColsAtCompileTime = RhsTraits::ColsAtCompileTime, + MaxRowsAtCompileTime = LhsTraits::MaxRowsAtCompileTime, + MaxColsAtCompileTime = RhsTraits::MaxColsAtCompileTime, + + // FIXME: only needed by GeneralMatrixMatrixTriangular + InnerSize = EIGEN_SIZE_MIN_PREFER_FIXED(LhsTraits::ColsAtCompileTime, RhsTraits::RowsAtCompileTime), + + // The storage order is somewhat arbitrary here. The correct one will be determined through the evaluator. + Flags = (MaxRowsAtCompileTime == 1 && MaxColsAtCompileTime != 1) ? RowMajorBit + : (MaxColsAtCompileTime == 1 && MaxRowsAtCompileTime != 1) ? 0 + : (((LhsTraits::Flags & NoPreferredStorageOrderBit) && (RhsTraits::Flags & RowMajorBit)) + || ((RhsTraits::Flags & NoPreferredStorageOrderBit) && (LhsTraits::Flags & RowMajorBit))) + ? RowMajorBit + : NoPreferredStorageOrderBit + }; }; -}; -} // end namespace internal +}// end namespace internal /** \class Product - * \ingroup Core_Module - * - * \brief Expression of the product of two arbitrary matrices or vectors - * - * \tparam _Lhs the type of the left-hand side expression - * \tparam _Rhs the type of the right-hand side expression - * - * This class represents an expression of the product of two arbitrary matrices. - * - * The other template parameters are: - * \tparam Option can be DefaultProduct, AliasFreeProduct, or LazyProduct - * - */ + * \ingroup Core_Module + * + * \brief Expression of the product of two arbitrary matrices or vectors + * + * \tparam _Lhs the type of the left-hand side expression + * \tparam _Rhs the type of the right-hand side expression + * + * This class represents an expression of the product of two arbitrary matrices. + * + * The other template parameters are: + * \tparam Option can be DefaultProduct, AliasFreeProduct, or LazyProduct + * + */ template -class Product : public ProductImpl<_Lhs,_Rhs,Option, - typename internal::product_promote_storage_type::StorageKind, - typename internal::traits<_Rhs>::StorageKind, - internal::product_type<_Lhs,_Rhs>::ret>::ret> +class Product + : public ProductImpl<_Lhs, + _Rhs, + Option, + typename internal::product_promote_storage_type::StorageKind, + typename internal::traits<_Rhs>::StorageKind, + internal::product_type<_Lhs, _Rhs>::ret>::ret> { - public: - - typedef _Lhs Lhs; - typedef _Rhs Rhs; - - typedef typename ProductImpl< - Lhs, Rhs, Option, - typename internal::product_promote_storage_type::StorageKind, - typename internal::traits::StorageKind, - internal::product_type::ret>::ret>::Base Base; - EIGEN_GENERIC_PUBLIC_INTERFACE(Product) - - typedef typename internal::ref_selector::type LhsNested; - typedef typename internal::ref_selector::type RhsNested; - typedef typename internal::remove_all::type LhsNestedCleaned; - typedef typename internal::remove_all::type RhsNestedCleaned; - - EIGEN_DEVICE_FUNC Product(const Lhs& lhs, const Rhs& rhs) : m_lhs(lhs), m_rhs(rhs) - { - eigen_assert(lhs.cols() == rhs.rows() - && "invalid matrix product" - && "if you wanted a coeff-wise or a dot product use the respective explicit functions"); - } - - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Index rows() const { return m_lhs.rows(); } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Index cols() const { return m_rhs.cols(); } - - EIGEN_DEVICE_FUNC const LhsNestedCleaned& lhs() const { return m_lhs; } - EIGEN_DEVICE_FUNC const RhsNestedCleaned& rhs() const { return m_rhs; } - - protected: - - LhsNested m_lhs; - RhsNested m_rhs; +public: + typedef _Lhs Lhs; + typedef _Rhs Rhs; + + typedef typename ProductImpl::StorageKind, + typename internal::traits::StorageKind, + internal::product_type::ret>::ret>::Base Base; + EIGEN_GENERIC_PUBLIC_INTERFACE(Product) + + typedef typename internal::ref_selector::type LhsNested; + typedef typename internal::ref_selector::type RhsNested; + typedef typename internal::remove_all::type LhsNestedCleaned; + typedef typename internal::remove_all::type RhsNestedCleaned; + + EIGEN_DEVICE_FUNC Product(const Lhs &lhs, const Rhs &rhs) : m_lhs(lhs), m_rhs(rhs) + { + eigen_assert(lhs.cols() == rhs.rows() && "invalid matrix product" + && "if you wanted a coeff-wise or a dot product use the respective explicit functions"); + } + + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Index rows() const { return m_lhs.rows(); } + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Index cols() const { return m_rhs.cols(); } + + EIGEN_DEVICE_FUNC const LhsNestedCleaned &lhs() const { return m_lhs; } + EIGEN_DEVICE_FUNC const RhsNestedCleaned &rhs() const { return m_rhs; } + +protected: + LhsNested m_lhs; + RhsNested m_rhs; }; namespace internal { - -template::ret> -class dense_product_base - : public internal::dense_xpr_base >::type -{}; -/** Convertion to scalar for inner-products */ -template -class dense_product_base - : public internal::dense_xpr_base >::type -{ - typedef Product ProductXpr; - typedef typename internal::dense_xpr_base::type Base; -public: - using Base::derived; - typedef typename Base::Scalar Scalar; - - EIGEN_STRONG_INLINE operator const Scalar() const + template::ret> + class dense_product_base : public internal::dense_xpr_base>::type { - return internal::evaluator(derived()).coeff(0,0); - } -}; + }; + + /** Convertion to scalar for inner-products */ + template + class dense_product_base + : public internal::dense_xpr_base>::type + { + typedef Product ProductXpr; + typedef typename internal::dense_xpr_base::type Base; + + public: + using Base::derived; + typedef typename Base::Scalar Scalar; -} // namespace internal + EIGEN_STRONG_INLINE operator const Scalar() const { return internal::evaluator(derived()).coeff(0, 0); } + }; + +}// namespace internal // Generic API dispatcher template -class ProductImpl : public internal::generic_xpr_base, MatrixXpr, StorageKind>::type +class ProductImpl : public internal::generic_xpr_base, MatrixXpr, StorageKind>::type { - public: - typedef typename internal::generic_xpr_base, MatrixXpr, StorageKind>::type Base; +public: + typedef typename internal::generic_xpr_base, MatrixXpr, StorageKind>::type Base; }; template -class ProductImpl - : public internal::dense_product_base +class ProductImpl : public internal::dense_product_base { - typedef Product Derived; - - public: - - typedef typename internal::dense_product_base Base; - EIGEN_DENSE_PUBLIC_INTERFACE(Derived) - protected: - enum { - IsOneByOne = (RowsAtCompileTime == 1 || RowsAtCompileTime == Dynamic) && - (ColsAtCompileTime == 1 || ColsAtCompileTime == Dynamic), - EnableCoeff = IsOneByOne || Option==LazyProduct - }; - - public: - - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Scalar coeff(Index row, Index col) const - { - EIGEN_STATIC_ASSERT(EnableCoeff, THIS_METHOD_IS_ONLY_FOR_INNER_OR_LAZY_PRODUCTS); - eigen_assert( (Option==LazyProduct) || (this->rows() == 1 && this->cols() == 1) ); - - return internal::evaluator(derived()).coeff(row,col); - } - - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Scalar coeff(Index i) const - { - EIGEN_STATIC_ASSERT(EnableCoeff, THIS_METHOD_IS_ONLY_FOR_INNER_OR_LAZY_PRODUCTS); - eigen_assert( (Option==LazyProduct) || (this->rows() == 1 && this->cols() == 1) ); - - return internal::evaluator(derived()).coeff(i); - } - - + typedef Product Derived; + +public: + typedef typename internal::dense_product_base Base; + EIGEN_DENSE_PUBLIC_INTERFACE(Derived) +protected: + enum { + IsOneByOne = (RowsAtCompileTime == 1 || RowsAtCompileTime == Dynamic) + && (ColsAtCompileTime == 1 || ColsAtCompileTime == Dynamic), + EnableCoeff = IsOneByOne || Option == LazyProduct + }; + +public: + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Scalar coeff(Index row, Index col) const + { + EIGEN_STATIC_ASSERT(EnableCoeff, THIS_METHOD_IS_ONLY_FOR_INNER_OR_LAZY_PRODUCTS); + eigen_assert((Option == LazyProduct) || (this->rows() == 1 && this->cols() == 1)); + + return internal::evaluator(derived()).coeff(row, col); + } + + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Scalar coeff(Index i) const + { + EIGEN_STATIC_ASSERT(EnableCoeff, THIS_METHOD_IS_ONLY_FOR_INNER_OR_LAZY_PRODUCTS); + eigen_assert((Option == LazyProduct) || (this->rows() == 1 && this->cols() == 1)); + + return internal::evaluator(derived()).coeff(i); + } }; -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_PRODUCT_H +#endif// EIGEN_PRODUCT_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/ProductEvaluators.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/ProductEvaluators.h index 9b99bd76..e8bac639 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/ProductEvaluators.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/ProductEvaluators.h @@ -14,445 +14,500 @@ #define EIGEN_PRODUCTEVALUATORS_H namespace Eigen { - + namespace internal { -/** \internal - * Evaluator of a product expression. - * Since products require special treatments to handle all possible cases, - * we simply deffer the evaluation logic to a product_evaluator class - * which offers more partial specialization possibilities. - * - * \sa class product_evaluator - */ -template -struct evaluator > - : public product_evaluator > -{ - typedef Product XprType; - typedef product_evaluator Base; - - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE explicit evaluator(const XprType& xpr) : Base(xpr) {} -}; - -// Catch "scalar * ( A * B )" and transform it to "(A*scalar) * B" -// TODO we should apply that rule only if that's really helpful -template -struct evaluator_assume_aliasing, - const CwiseNullaryOp, Plain1>, - const Product > > -{ - static const bool value = true; -}; -template -struct evaluator, - const CwiseNullaryOp, Plain1>, - const Product > > - : public evaluator > -{ - typedef CwiseBinaryOp, - const CwiseNullaryOp, Plain1>, - const Product > XprType; - typedef evaluator > Base; - - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE explicit evaluator(const XprType& xpr) - : Base(xpr.lhs().functor().m_other * xpr.rhs().lhs() * xpr.rhs().rhs()) - {} -}; - - -template -struct evaluator, DiagIndex> > - : public evaluator, DiagIndex> > -{ - typedef Diagonal, DiagIndex> XprType; - typedef evaluator, DiagIndex> > Base; - - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE explicit evaluator(const XprType& xpr) - : Base(Diagonal, DiagIndex>( - Product(xpr.nestedExpression().lhs(), xpr.nestedExpression().rhs()), - xpr.index() )) - {} -}; - - -// Helper class to perform a matrix product with the destination at hand. -// Depending on the sizes of the factors, there are different evaluation strategies -// as controlled by internal::product_type. -template< typename Lhs, typename Rhs, - typename LhsShape = typename evaluator_traits::Shape, - typename RhsShape = typename evaluator_traits::Shape, - int ProductType = internal::product_type::value> -struct generic_product_impl; - -template -struct evaluator_assume_aliasing > { - static const bool value = true; -}; - -// This is the default evaluator implementation for products: -// It creates a temporary and call generic_product_impl -template -struct product_evaluator, ProductTag, LhsShape, RhsShape> - : public evaluator::PlainObject> -{ - typedef Product XprType; - typedef typename XprType::PlainObject PlainObject; - typedef evaluator Base; - enum { - Flags = Base::Flags | EvalBeforeNestingBit + /** \internal + * Evaluator of a product expression. + * Since products require special treatments to handle all possible cases, + * we simply deffer the evaluation logic to a product_evaluator class + * which offers more partial specialization possibilities. + * + * \sa class product_evaluator + */ + template + struct evaluator> : public product_evaluator> + { + typedef Product XprType; + typedef product_evaluator Base; + + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE explicit evaluator(const XprType &xpr) : Base(xpr) {} }; - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE - explicit product_evaluator(const XprType& xpr) - : m_result(xpr.rows(), xpr.cols()) + // Catch "scalar * ( A * B )" and transform it to "(A*scalar) * B" + // TODO we should apply that rule only if that's really helpful + template + struct evaluator_assume_aliasing, + const CwiseNullaryOp, Plain1>, + const Product>> { - ::new (static_cast(this)) Base(m_result); - -// FIXME shall we handle nested_eval here?, -// if so, then we must take care at removing the call to nested_eval in the specializations (e.g., in permutation_matrix_product, transposition_matrix_product, etc.) -// typedef typename internal::nested_eval::type LhsNested; -// typedef typename internal::nested_eval::type RhsNested; -// typedef typename internal::remove_all::type LhsNestedCleaned; -// typedef typename internal::remove_all::type RhsNestedCleaned; -// -// const LhsNested lhs(xpr.lhs()); -// const RhsNested rhs(xpr.rhs()); -// -// generic_product_impl::evalTo(m_result, lhs, rhs); - - generic_product_impl::evalTo(m_result, xpr.lhs(), xpr.rhs()); - } - -protected: - PlainObject m_result; -}; - -// The following three shortcuts are enabled only if the scalar types match excatly. -// TODO: we could enable them for different scalar types when the product is not vectorized. - -// Dense = Product -template< typename DstXprType, typename Lhs, typename Rhs, int Options, typename Scalar> -struct Assignment, internal::assign_op, Dense2Dense, - typename enable_if<(Options==DefaultProduct || Options==AliasFreeProduct)>::type> -{ - typedef Product SrcXprType; - static EIGEN_STRONG_INLINE - void run(DstXprType &dst, const SrcXprType &src, const internal::assign_op &) + static const bool value = true; + }; + template + struct evaluator, + const CwiseNullaryOp, Plain1>, + const Product>> + : public evaluator> { - Index dstRows = src.rows(); - Index dstCols = src.cols(); - if((dst.rows()!=dstRows) || (dst.cols()!=dstCols)) - dst.resize(dstRows, dstCols); - // FIXME shall we handle nested_eval here? - generic_product_impl::evalTo(dst, src.lhs(), src.rhs()); - } -}; - -// Dense += Product -template< typename DstXprType, typename Lhs, typename Rhs, int Options, typename Scalar> -struct Assignment, internal::add_assign_op, Dense2Dense, - typename enable_if<(Options==DefaultProduct || Options==AliasFreeProduct)>::type> -{ - typedef Product SrcXprType; - static EIGEN_STRONG_INLINE - void run(DstXprType &dst, const SrcXprType &src, const internal::add_assign_op &) + typedef CwiseBinaryOp, + const CwiseNullaryOp, Plain1>, + const Product> + XprType; + typedef evaluator> Base; + + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE explicit evaluator(const XprType &xpr) + : Base(xpr.lhs().functor().m_other * xpr.rhs().lhs() * xpr.rhs().rhs()) + {} + }; + + + template + struct evaluator, DiagIndex>> + : public evaluator, DiagIndex>> { - eigen_assert(dst.rows() == src.rows() && dst.cols() == src.cols()); - // FIXME shall we handle nested_eval here? - generic_product_impl::addTo(dst, src.lhs(), src.rhs()); - } -}; - -// Dense -= Product -template< typename DstXprType, typename Lhs, typename Rhs, int Options, typename Scalar> -struct Assignment, internal::sub_assign_op, Dense2Dense, - typename enable_if<(Options==DefaultProduct || Options==AliasFreeProduct)>::type> -{ - typedef Product SrcXprType; - static EIGEN_STRONG_INLINE - void run(DstXprType &dst, const SrcXprType &src, const internal::sub_assign_op &) + typedef Diagonal, DiagIndex> XprType; + typedef evaluator, DiagIndex>> Base; + + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE explicit evaluator(const XprType &xpr) + : Base(Diagonal, DiagIndex>( + Product(xpr.nestedExpression().lhs(), xpr.nestedExpression().rhs()), + xpr.index())) + {} + }; + + + // Helper class to perform a matrix product with the destination at hand. + // Depending on the sizes of the factors, there are different evaluation strategies + // as controlled by internal::product_type. + template::Shape, + typename RhsShape = typename evaluator_traits::Shape, + int ProductType = internal::product_type::value> + struct generic_product_impl; + + template struct evaluator_assume_aliasing> { - eigen_assert(dst.rows() == src.rows() && dst.cols() == src.cols()); - // FIXME shall we handle nested_eval here? - generic_product_impl::subTo(dst, src.lhs(), src.rhs()); - } -}; - - -// Dense ?= scalar * Product -// TODO we should apply that rule if that's really helpful -// for instance, this is not good for inner products -template< typename DstXprType, typename Lhs, typename Rhs, typename AssignFunc, typename Scalar, typename ScalarBis, typename Plain> -struct Assignment, const CwiseNullaryOp,Plain>, - const Product >, AssignFunc, Dense2Dense> -{ - typedef CwiseBinaryOp, - const CwiseNullaryOp,Plain>, - const Product > SrcXprType; - static EIGEN_STRONG_INLINE - void run(DstXprType &dst, const SrcXprType &src, const AssignFunc& func) + static const bool value = true; + }; + + // This is the default evaluator implementation for products: + // It creates a temporary and call generic_product_impl + template + struct product_evaluator, ProductTag, LhsShape, RhsShape> + : public evaluator::PlainObject> { - call_assignment_no_alias(dst, (src.lhs().functor().m_other * src.rhs().lhs())*src.rhs().rhs(), func); - } -}; - -//---------------------------------------- -// Catch "Dense ?= xpr + Product<>" expression to save one temporary -// FIXME we could probably enable these rules for any product, i.e., not only Dense and DefaultProduct - -template -struct evaluator_assume_aliasing::Scalar>, const OtherXpr, - const Product >, DenseShape > { - static const bool value = true; -}; - -template -struct evaluator_assume_aliasing::Scalar>, const OtherXpr, - const Product >, DenseShape > { - static const bool value = true; -}; - -template -struct assignment_from_xpr_op_product -{ - template - static EIGEN_STRONG_INLINE - void run(DstXprType &dst, const SrcXprType &src, const InitialFunc& /*func*/) + typedef Product XprType; + typedef typename XprType::PlainObject PlainObject; + typedef evaluator Base; + enum { Flags = Base::Flags | EvalBeforeNestingBit }; + + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE explicit product_evaluator(const XprType &xpr) + : m_result(xpr.rows(), xpr.cols()) + { + ::new (static_cast(this)) Base(m_result); + + // FIXME shall we handle nested_eval here?, + // if so, then we must take care at removing the call to nested_eval in the specializations (e.g., in + // permutation_matrix_product, transposition_matrix_product, etc.) + // typedef typename internal::nested_eval::type LhsNested; + // typedef typename internal::nested_eval::type RhsNested; + // typedef typename internal::remove_all::type LhsNestedCleaned; + // typedef typename internal::remove_all::type RhsNestedCleaned; + // + // const LhsNested lhs(xpr.lhs()); + // const RhsNested rhs(xpr.rhs()); + // + // generic_product_impl::evalTo(m_result, lhs, rhs); + + generic_product_impl::evalTo(m_result, xpr.lhs(), xpr.rhs()); + } + + protected: + PlainObject m_result; + }; + + // The following three shortcuts are enabled only if the scalar types match excatly. + // TODO: we could enable them for different scalar types when the product is not vectorized. + + // Dense = Product + template + struct Assignment, + internal::assign_op, + Dense2Dense, + typename enable_if<(Options == DefaultProduct || Options == AliasFreeProduct)>::type> { - call_assignment_no_alias(dst, src.lhs(), Func1()); - call_assignment_no_alias(dst, src.rhs(), Func2()); - } -}; - -#define EIGEN_CATCH_ASSIGN_XPR_OP_PRODUCT(ASSIGN_OP,BINOP,ASSIGN_OP2) \ - template< typename DstXprType, typename OtherXpr, typename Lhs, typename Rhs, typename DstScalar, typename SrcScalar, typename OtherScalar,typename ProdScalar> \ - struct Assignment, const OtherXpr, \ - const Product >, internal::ASSIGN_OP, Dense2Dense> \ - : assignment_from_xpr_op_product, internal::ASSIGN_OP, internal::ASSIGN_OP2 > \ - {} - -EIGEN_CATCH_ASSIGN_XPR_OP_PRODUCT(assign_op, scalar_sum_op,add_assign_op); -EIGEN_CATCH_ASSIGN_XPR_OP_PRODUCT(add_assign_op,scalar_sum_op,add_assign_op); -EIGEN_CATCH_ASSIGN_XPR_OP_PRODUCT(sub_assign_op,scalar_sum_op,sub_assign_op); - -EIGEN_CATCH_ASSIGN_XPR_OP_PRODUCT(assign_op, scalar_difference_op,sub_assign_op); -EIGEN_CATCH_ASSIGN_XPR_OP_PRODUCT(add_assign_op,scalar_difference_op,sub_assign_op); -EIGEN_CATCH_ASSIGN_XPR_OP_PRODUCT(sub_assign_op,scalar_difference_op,add_assign_op); - -//---------------------------------------- - -template -struct generic_product_impl -{ - template - static EIGEN_STRONG_INLINE void evalTo(Dst& dst, const Lhs& lhs, const Rhs& rhs) + typedef Product SrcXprType; + static EIGEN_STRONG_INLINE void + run(DstXprType &dst, const SrcXprType &src, const internal::assign_op &) + { + Index dstRows = src.rows(); + Index dstCols = src.cols(); + if ((dst.rows() != dstRows) || (dst.cols() != dstCols)) dst.resize(dstRows, dstCols); + // FIXME shall we handle nested_eval here? + generic_product_impl::evalTo(dst, src.lhs(), src.rhs()); + } + }; + + // Dense += Product + template + struct Assignment, + internal::add_assign_op, + Dense2Dense, + typename enable_if<(Options == DefaultProduct || Options == AliasFreeProduct)>::type> { - dst.coeffRef(0,0) = (lhs.transpose().cwiseProduct(rhs)).sum(); - } - - template - static EIGEN_STRONG_INLINE void addTo(Dst& dst, const Lhs& lhs, const Rhs& rhs) + typedef Product SrcXprType; + static EIGEN_STRONG_INLINE void + run(DstXprType &dst, const SrcXprType &src, const internal::add_assign_op &) + { + eigen_assert(dst.rows() == src.rows() && dst.cols() == src.cols()); + // FIXME shall we handle nested_eval here? + generic_product_impl::addTo(dst, src.lhs(), src.rhs()); + } + }; + + // Dense -= Product + template + struct Assignment, + internal::sub_assign_op, + Dense2Dense, + typename enable_if<(Options == DefaultProduct || Options == AliasFreeProduct)>::type> { - dst.coeffRef(0,0) += (lhs.transpose().cwiseProduct(rhs)).sum(); - } - - template - static EIGEN_STRONG_INLINE void subTo(Dst& dst, const Lhs& lhs, const Rhs& rhs) - { dst.coeffRef(0,0) -= (lhs.transpose().cwiseProduct(rhs)).sum(); } -}; - - -/*********************************************************************** -* Implementation of outer dense * dense vector product -***********************************************************************/ - -// Column major result -template -void outer_product_selector_run(Dst& dst, const Lhs &lhs, const Rhs &rhs, const Func& func, const false_type&) -{ - evaluator rhsEval(rhs); - typename nested_eval::type actual_lhs(lhs); - // FIXME if cols is large enough, then it might be useful to make sure that lhs is sequentially stored - // FIXME not very good if rhs is real and lhs complex while alpha is real too - const Index cols = dst.cols(); - for (Index j=0; j -void outer_product_selector_run(Dst& dst, const Lhs &lhs, const Rhs &rhs, const Func& func, const true_type&) -{ - evaluator lhsEval(lhs); - typename nested_eval::type actual_rhs(rhs); - // FIXME if rows is large enough, then it might be useful to make sure that rhs is sequentially stored - // FIXME not very good if lhs is real and rhs complex while alpha is real too - const Index rows = dst.rows(); - for (Index i=0; i -struct generic_product_impl -{ - template struct is_row_major : internal::conditional<(int(T::Flags)&RowMajorBit), internal::true_type, internal::false_type>::type {}; - typedef typename Product::Scalar Scalar; - - // TODO it would be nice to be able to exploit our *_assign_op functors for that purpose - struct set { template void operator()(const Dst& dst, const Src& src) const { dst.const_cast_derived() = src; } }; - struct add { template void operator()(const Dst& dst, const Src& src) const { dst.const_cast_derived() += src; } }; - struct sub { template void operator()(const Dst& dst, const Src& src) const { dst.const_cast_derived() -= src; } }; - struct adds { - Scalar m_scale; - explicit adds(const Scalar& s) : m_scale(s) {} - template void operator()(const Dst& dst, const Src& src) const { - dst.const_cast_derived() += m_scale * src; + typedef Product SrcXprType; + static EIGEN_STRONG_INLINE void + run(DstXprType &dst, const SrcXprType &src, const internal::sub_assign_op &) + { + eigen_assert(dst.rows() == src.rows() && dst.cols() == src.cols()); + // FIXME shall we handle nested_eval here? + generic_product_impl::subTo(dst, src.lhs(), src.rhs()); } }; - - template - static EIGEN_STRONG_INLINE void evalTo(Dst& dst, const Lhs& lhs, const Rhs& rhs) + + + // Dense ?= scalar * Product + // TODO we should apply that rule if that's really helpful + // for instance, this is not good for inner products + template + struct Assignment, + const CwiseNullaryOp, Plain>, + const Product>, + AssignFunc, + Dense2Dense> { - internal::outer_product_selector_run(dst, lhs, rhs, set(), is_row_major()); - } - - template - static EIGEN_STRONG_INLINE void addTo(Dst& dst, const Lhs& lhs, const Rhs& rhs) + typedef CwiseBinaryOp, + const CwiseNullaryOp, Plain>, + const Product> + SrcXprType; + static EIGEN_STRONG_INLINE void run(DstXprType &dst, const SrcXprType &src, const AssignFunc &func) + { + call_assignment_no_alias(dst, (src.lhs().functor().m_other * src.rhs().lhs()) * src.rhs().rhs(), func); + } + }; + + //---------------------------------------- + // Catch "Dense ?= xpr + Product<>" expression to save one temporary + // FIXME we could probably enable these rules for any product, i.e., not only Dense and DefaultProduct + + template + struct evaluator_assume_aliasing::Scalar>, + const OtherXpr, + const Product>, + DenseShape> { - internal::outer_product_selector_run(dst, lhs, rhs, add(), is_row_major()); - } - - template - static EIGEN_STRONG_INLINE void subTo(Dst& dst, const Lhs& lhs, const Rhs& rhs) + static const bool value = true; + }; + + template + struct evaluator_assume_aliasing::Scalar>, + const OtherXpr, + const Product>, + DenseShape> { - internal::outer_product_selector_run(dst, lhs, rhs, sub(), is_row_major()); - } - - template - static EIGEN_STRONG_INLINE void scaleAndAddTo(Dst& dst, const Lhs& lhs, const Rhs& rhs, const Scalar& alpha) + static const bool value = true; + }; + + template + struct assignment_from_xpr_op_product { - internal::outer_product_selector_run(dst, lhs, rhs, adds(alpha), is_row_major()); + template + static EIGEN_STRONG_INLINE void run(DstXprType &dst, const SrcXprType &src, const InitialFunc & /*func*/) + { + call_assignment_no_alias(dst, src.lhs(), Func1()); + call_assignment_no_alias(dst, src.rhs(), Func2()); + } + }; + +#define EIGEN_CATCH_ASSIGN_XPR_OP_PRODUCT(ASSIGN_OP, BINOP, ASSIGN_OP2) \ + template \ + struct Assignment, const OtherXpr, const Product>, \ + internal::ASSIGN_OP, \ + Dense2Dense> \ + : assignment_from_xpr_op_product, \ + internal::ASSIGN_OP, \ + internal::ASSIGN_OP2> \ + { \ } - -}; - - -// This base class provides default implementations for evalTo, addTo, subTo, in terms of scaleAndAddTo -template -struct generic_product_impl_base -{ - typedef typename Product::Scalar Scalar; - - template - static EIGEN_STRONG_INLINE void evalTo(Dst& dst, const Lhs& lhs, const Rhs& rhs) - { dst.setZero(); scaleAndAddTo(dst, lhs, rhs, Scalar(1)); } - - template - static EIGEN_STRONG_INLINE void addTo(Dst& dst, const Lhs& lhs, const Rhs& rhs) - { scaleAndAddTo(dst,lhs, rhs, Scalar(1)); } - - template - static EIGEN_STRONG_INLINE void subTo(Dst& dst, const Lhs& lhs, const Rhs& rhs) - { scaleAndAddTo(dst, lhs, rhs, Scalar(-1)); } - - template - static EIGEN_STRONG_INLINE void scaleAndAddTo(Dst& dst, const Lhs& lhs, const Rhs& rhs, const Scalar& alpha) - { Derived::scaleAndAddTo(dst,lhs,rhs,alpha); } - -}; - -template -struct generic_product_impl - : generic_product_impl_base > -{ - typedef typename nested_eval::type LhsNested; - typedef typename nested_eval::type RhsNested; - typedef typename Product::Scalar Scalar; - enum { Side = Lhs::IsVectorAtCompileTime ? OnTheLeft : OnTheRight }; - typedef typename internal::remove_all::type>::type MatrixType; - - template - static EIGEN_STRONG_INLINE void scaleAndAddTo(Dest& dst, const Lhs& lhs, const Rhs& rhs, const Scalar& alpha) + + EIGEN_CATCH_ASSIGN_XPR_OP_PRODUCT(assign_op, scalar_sum_op, add_assign_op); + EIGEN_CATCH_ASSIGN_XPR_OP_PRODUCT(add_assign_op, scalar_sum_op, add_assign_op); + EIGEN_CATCH_ASSIGN_XPR_OP_PRODUCT(sub_assign_op, scalar_sum_op, sub_assign_op); + + EIGEN_CATCH_ASSIGN_XPR_OP_PRODUCT(assign_op, scalar_difference_op, sub_assign_op); + EIGEN_CATCH_ASSIGN_XPR_OP_PRODUCT(add_assign_op, scalar_difference_op, sub_assign_op); + EIGEN_CATCH_ASSIGN_XPR_OP_PRODUCT(sub_assign_op, scalar_difference_op, add_assign_op); + + //---------------------------------------- + + template struct generic_product_impl { - LhsNested actual_lhs(lhs); - RhsNested actual_rhs(rhs); - internal::gemv_dense_selector::HasUsableDirectAccess) - >::run(actual_lhs, actual_rhs, dst, alpha); - } -}; - -template -struct generic_product_impl -{ - typedef typename Product::Scalar Scalar; - - template - static EIGEN_STRONG_INLINE void evalTo(Dst& dst, const Lhs& lhs, const Rhs& rhs) + template static EIGEN_STRONG_INLINE void evalTo(Dst &dst, const Lhs &lhs, const Rhs &rhs) + { + dst.coeffRef(0, 0) = (lhs.transpose().cwiseProduct(rhs)).sum(); + } + + template static EIGEN_STRONG_INLINE void addTo(Dst &dst, const Lhs &lhs, const Rhs &rhs) + { + dst.coeffRef(0, 0) += (lhs.transpose().cwiseProduct(rhs)).sum(); + } + + template static EIGEN_STRONG_INLINE void subTo(Dst &dst, const Lhs &lhs, const Rhs &rhs) + { + dst.coeffRef(0, 0) -= (lhs.transpose().cwiseProduct(rhs)).sum(); + } + }; + + + /*********************************************************************** + * Implementation of outer dense * dense vector product + ***********************************************************************/ + + // Column major result + template + void outer_product_selector_run(Dst &dst, const Lhs &lhs, const Rhs &rhs, const Func &func, const false_type &) { - // Same as: dst.noalias() = lhs.lazyProduct(rhs); - // but easier on the compiler side - call_assignment_no_alias(dst, lhs.lazyProduct(rhs), internal::assign_op()); + evaluator rhsEval(rhs); + typename nested_eval::type actual_lhs(lhs); + // FIXME if cols is large enough, then it might be useful to make sure that lhs is sequentially stored + // FIXME not very good if rhs is real and lhs complex while alpha is real too + const Index cols = dst.cols(); + for (Index j = 0; j < cols; ++j) func(dst.col(j), rhsEval.coeff(Index(0), j) * actual_lhs); } - - template - static EIGEN_STRONG_INLINE void addTo(Dst& dst, const Lhs& lhs, const Rhs& rhs) + + // Row major result + template + void outer_product_selector_run(Dst &dst, const Lhs &lhs, const Rhs &rhs, const Func &func, const true_type &) { - // dst.noalias() += lhs.lazyProduct(rhs); - call_assignment_no_alias(dst, lhs.lazyProduct(rhs), internal::add_assign_op()); + evaluator lhsEval(lhs); + typename nested_eval::type actual_rhs(rhs); + // FIXME if rows is large enough, then it might be useful to make sure that rhs is sequentially stored + // FIXME not very good if lhs is real and rhs complex while alpha is real too + const Index rows = dst.rows(); + for (Index i = 0; i < rows; ++i) func(dst.row(i), lhsEval.coeff(i, Index(0)) * actual_rhs); } - - template - static EIGEN_STRONG_INLINE void subTo(Dst& dst, const Lhs& lhs, const Rhs& rhs) + + template struct generic_product_impl { - // dst.noalias() -= lhs.lazyProduct(rhs); - call_assignment_no_alias(dst, lhs.lazyProduct(rhs), internal::sub_assign_op()); - } - -// template -// static inline void scaleAndAddTo(Dst& dst, const Lhs& lhs, const Rhs& rhs, const Scalar& alpha) -// { dst.noalias() += alpha * lhs.lazyProduct(rhs); } -}; - -// This specialization enforces the use of a coefficient-based evaluation strategy -template -struct generic_product_impl - : generic_product_impl {}; - -// Case 2: Evaluate coeff by coeff -// -// This is mostly taken from CoeffBasedProduct.h -// The main difference is that we add an extra argument to the etor_product_*_impl::run() function -// for the inner dimension of the product, because evaluator object do not know their size. - -template -struct etor_product_coeff_impl; - -template -struct etor_product_packet_impl; - -template -struct product_evaluator, ProductTag, DenseShape, DenseShape> - : evaluator_base > -{ - typedef Product XprType; - typedef typename XprType::Scalar Scalar; - typedef typename XprType::CoeffReturnType CoeffReturnType; - - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE - explicit product_evaluator(const XprType& xpr) - : m_lhs(xpr.lhs()), - m_rhs(xpr.rhs()), - m_lhsImpl(m_lhs), // FIXME the creation of the evaluator objects should result in a no-op, but check that! - m_rhsImpl(m_rhs), // Moreover, they are only useful for the packet path, so we could completely disable them when not needed, - // or perhaps declare them on the fly on the packet method... We have experiment to check what's best. - m_innerDim(xpr.lhs().cols()) + template + struct is_row_major + : internal::conditional<(int(T::Flags) & RowMajorBit), internal::true_type, internal::false_type>::type + { + }; + typedef typename Product::Scalar Scalar; + + // TODO it would be nice to be able to exploit our *_assign_op functors for that purpose + struct set + { + template void operator()(const Dst &dst, const Src &src) const + { + dst.const_cast_derived() = src; + } + }; + struct add + { + template void operator()(const Dst &dst, const Src &src) const + { + dst.const_cast_derived() += src; + } + }; + struct sub + { + template void operator()(const Dst &dst, const Src &src) const + { + dst.const_cast_derived() -= src; + } + }; + struct adds + { + Scalar m_scale; + explicit adds(const Scalar &s) : m_scale(s) {} + template void operator()(const Dst &dst, const Src &src) const + { + dst.const_cast_derived() += m_scale * src; + } + }; + + template static EIGEN_STRONG_INLINE void evalTo(Dst &dst, const Lhs &lhs, const Rhs &rhs) + { + internal::outer_product_selector_run(dst, lhs, rhs, set(), is_row_major()); + } + + template static EIGEN_STRONG_INLINE void addTo(Dst &dst, const Lhs &lhs, const Rhs &rhs) + { + internal::outer_product_selector_run(dst, lhs, rhs, add(), is_row_major()); + } + + template static EIGEN_STRONG_INLINE void subTo(Dst &dst, const Lhs &lhs, const Rhs &rhs) + { + internal::outer_product_selector_run(dst, lhs, rhs, sub(), is_row_major()); + } + + template + static EIGEN_STRONG_INLINE void scaleAndAddTo(Dst &dst, const Lhs &lhs, const Rhs &rhs, const Scalar &alpha) + { + internal::outer_product_selector_run(dst, lhs, rhs, adds(alpha), is_row_major()); + } + }; + + + // This base class provides default implementations for evalTo, addTo, subTo, in terms of scaleAndAddTo + template struct generic_product_impl_base + { + typedef typename Product::Scalar Scalar; + + template static EIGEN_STRONG_INLINE void evalTo(Dst &dst, const Lhs &lhs, const Rhs &rhs) + { + dst.setZero(); + scaleAndAddTo(dst, lhs, rhs, Scalar(1)); + } + + template static EIGEN_STRONG_INLINE void addTo(Dst &dst, const Lhs &lhs, const Rhs &rhs) + { + scaleAndAddTo(dst, lhs, rhs, Scalar(1)); + } + + template static EIGEN_STRONG_INLINE void subTo(Dst &dst, const Lhs &lhs, const Rhs &rhs) + { + scaleAndAddTo(dst, lhs, rhs, Scalar(-1)); + } + + template + static EIGEN_STRONG_INLINE void scaleAndAddTo(Dst &dst, const Lhs &lhs, const Rhs &rhs, const Scalar &alpha) + { + Derived::scaleAndAddTo(dst, lhs, rhs, alpha); + } + }; + + template + struct generic_product_impl + : generic_product_impl_base> + { + typedef typename nested_eval::type LhsNested; + typedef typename nested_eval::type RhsNested; + typedef typename Product::Scalar Scalar; + enum { Side = Lhs::IsVectorAtCompileTime ? OnTheLeft : OnTheRight }; + typedef typename internal::remove_all< + typename internal::conditional::type>::type MatrixType; + + template + static EIGEN_STRONG_INLINE void scaleAndAddTo(Dest &dst, const Lhs &lhs, const Rhs &rhs, const Scalar &alpha) + { + LhsNested actual_lhs(lhs); + RhsNested actual_rhs(rhs); + internal::gemv_dense_selector::HasUsableDirectAccess)>::run(actual_lhs, actual_rhs, dst, alpha); + } + }; + + template + struct generic_product_impl + { + typedef typename Product::Scalar Scalar; + + template static EIGEN_STRONG_INLINE void evalTo(Dst &dst, const Lhs &lhs, const Rhs &rhs) + { + // Same as: dst.noalias() = lhs.lazyProduct(rhs); + // but easier on the compiler side + call_assignment_no_alias(dst, lhs.lazyProduct(rhs), internal::assign_op()); + } + + template static EIGEN_STRONG_INLINE void addTo(Dst &dst, const Lhs &lhs, const Rhs &rhs) + { + // dst.noalias() += lhs.lazyProduct(rhs); + call_assignment_no_alias(dst, lhs.lazyProduct(rhs), internal::add_assign_op()); + } + + template static EIGEN_STRONG_INLINE void subTo(Dst &dst, const Lhs &lhs, const Rhs &rhs) + { + // dst.noalias() -= lhs.lazyProduct(rhs); + call_assignment_no_alias(dst, lhs.lazyProduct(rhs), internal::sub_assign_op()); + } + + // template + // static inline void scaleAndAddTo(Dst& dst, const Lhs& lhs, const Rhs& rhs, const Scalar& alpha) + // { dst.noalias() += alpha * lhs.lazyProduct(rhs); } + }; + + // This specialization enforces the use of a coefficient-based evaluation strategy + template + struct generic_product_impl + : generic_product_impl + { + }; + + // Case 2: Evaluate coeff by coeff + // + // This is mostly taken from CoeffBasedProduct.h + // The main difference is that we add an extra argument to the etor_product_*_impl::run() function + // for the inner dimension of the product, because evaluator object do not know their size. + + template + struct etor_product_coeff_impl; + + template + struct etor_product_packet_impl; + + template + struct product_evaluator, ProductTag, DenseShape, DenseShape> + : evaluator_base> { - EIGEN_INTERNAL_CHECK_COST_VALUE(NumTraits::MulCost); - EIGEN_INTERNAL_CHECK_COST_VALUE(NumTraits::AddCost); - EIGEN_INTERNAL_CHECK_COST_VALUE(CoeffReadCost); + typedef Product XprType; + typedef typename XprType::Scalar Scalar; + typedef typename XprType::CoeffReturnType CoeffReturnType; + + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE explicit product_evaluator(const XprType &xpr) + : m_lhs(xpr.lhs()), m_rhs(xpr.rhs()), + m_lhsImpl(m_lhs),// FIXME the creation of the evaluator objects should result in a no-op, but check that! + m_rhsImpl(m_rhs),// Moreover, they are only useful for the packet path, so we could completely disable + // them when not needed, or perhaps declare them on the fly on the packet method... We + // have experiment to check what's best. + m_innerDim(xpr.lhs().cols()) + { + EIGEN_INTERNAL_CHECK_COST_VALUE(NumTraits::MulCost); + EIGEN_INTERNAL_CHECK_COST_VALUE(NumTraits::AddCost); + EIGEN_INTERNAL_CHECK_COST_VALUE(CoeffReadCost); #if 0 std::cerr << "LhsOuterStrideBytes= " << LhsOuterStrideBytes << "\n"; std::cerr << "RhsOuterStrideBytes= " << RhsOuterStrideBytes << "\n"; @@ -465,648 +520,671 @@ struct product_evaluator, ProductTag, DenseShape, std::cerr << "Alignment= " << Alignment << "\n"; std::cerr << "Flags= " << Flags << "\n"; #endif - } + } - // Everything below here is taken from CoeffBasedProduct.h + // Everything below here is taken from CoeffBasedProduct.h - typedef typename internal::nested_eval::type LhsNested; - typedef typename internal::nested_eval::type RhsNested; - - typedef typename internal::remove_all::type LhsNestedCleaned; - typedef typename internal::remove_all::type RhsNestedCleaned; + typedef typename internal::nested_eval::type LhsNested; + typedef typename internal::nested_eval::type RhsNested; - typedef evaluator LhsEtorType; - typedef evaluator RhsEtorType; + typedef typename internal::remove_all::type LhsNestedCleaned; + typedef typename internal::remove_all::type RhsNestedCleaned; - enum { - RowsAtCompileTime = LhsNestedCleaned::RowsAtCompileTime, - ColsAtCompileTime = RhsNestedCleaned::ColsAtCompileTime, - InnerSize = EIGEN_SIZE_MIN_PREFER_FIXED(LhsNestedCleaned::ColsAtCompileTime, RhsNestedCleaned::RowsAtCompileTime), - MaxRowsAtCompileTime = LhsNestedCleaned::MaxRowsAtCompileTime, - MaxColsAtCompileTime = RhsNestedCleaned::MaxColsAtCompileTime - }; + typedef evaluator LhsEtorType; + typedef evaluator RhsEtorType; + + enum { + RowsAtCompileTime = LhsNestedCleaned::RowsAtCompileTime, + ColsAtCompileTime = RhsNestedCleaned::ColsAtCompileTime, + InnerSize = EIGEN_SIZE_MIN_PREFER_FIXED(LhsNestedCleaned::ColsAtCompileTime, RhsNestedCleaned::RowsAtCompileTime), + MaxRowsAtCompileTime = LhsNestedCleaned::MaxRowsAtCompileTime, + MaxColsAtCompileTime = RhsNestedCleaned::MaxColsAtCompileTime + }; + + typedef typename find_best_packet::type LhsVecPacketType; + typedef typename find_best_packet::type RhsVecPacketType; + + enum { + + LhsCoeffReadCost = LhsEtorType::CoeffReadCost, + RhsCoeffReadCost = RhsEtorType::CoeffReadCost, + CoeffReadCost = InnerSize == 0 ? NumTraits::ReadCost + : InnerSize == Dynamic + ? HugeCost + : InnerSize * (NumTraits::MulCost + LhsCoeffReadCost + RhsCoeffReadCost) + + (InnerSize - 1) * NumTraits::AddCost, + + Unroll = CoeffReadCost <= EIGEN_UNROLLING_LIMIT, + + LhsFlags = LhsEtorType::Flags, + RhsFlags = RhsEtorType::Flags, + + LhsRowMajor = LhsFlags & RowMajorBit, + RhsRowMajor = RhsFlags & RowMajorBit, + + LhsVecPacketSize = unpacket_traits::size, + RhsVecPacketSize = unpacket_traits::size, + + // Here, we don't care about alignment larger than the usable packet size. + LhsAlignment = + EIGEN_PLAIN_ENUM_MIN(LhsEtorType::Alignment, LhsVecPacketSize *int(sizeof(typename LhsNestedCleaned::Scalar))), + RhsAlignment = + EIGEN_PLAIN_ENUM_MIN(RhsEtorType::Alignment, RhsVecPacketSize *int(sizeof(typename RhsNestedCleaned::Scalar))), + + SameType = is_same::value, + + CanVectorizeRhs = bool(RhsRowMajor) && (RhsFlags & PacketAccessBit) && (ColsAtCompileTime != 1), + CanVectorizeLhs = (!LhsRowMajor) && (LhsFlags & PacketAccessBit) && (RowsAtCompileTime != 1), + + EvalToRowMajor = (MaxRowsAtCompileTime == 1 && MaxColsAtCompileTime != 1) ? 1 + : (MaxColsAtCompileTime == 1 && MaxRowsAtCompileTime != 1) + ? 0 + : (bool(RhsRowMajor) && !CanVectorizeLhs), + + Flags = ((unsigned int)(LhsFlags | RhsFlags) & HereditaryBits & ~RowMajorBit) + | (EvalToRowMajor ? RowMajorBit : 0) + // TODO enable vectorization for mixed types + | (SameType && (CanVectorizeLhs || CanVectorizeRhs) ? PacketAccessBit : 0) + | (XprType::IsVectorAtCompileTime ? LinearAccessBit : 0), + + LhsOuterStrideBytes = + int(LhsNestedCleaned::OuterStrideAtCompileTime) * int(sizeof(typename LhsNestedCleaned::Scalar)), + RhsOuterStrideBytes = + int(RhsNestedCleaned::OuterStrideAtCompileTime) * int(sizeof(typename RhsNestedCleaned::Scalar)), + + Alignment = + bool(CanVectorizeLhs) + ? (LhsOuterStrideBytes <= 0 || (int(LhsOuterStrideBytes) % EIGEN_PLAIN_ENUM_MAX(1, LhsAlignment)) != 0 + ? 0 + : LhsAlignment) + : bool(CanVectorizeRhs) + ? (RhsOuterStrideBytes <= 0 || (int(RhsOuterStrideBytes) % EIGEN_PLAIN_ENUM_MAX(1, RhsAlignment)) != 0 + ? 0 + : RhsAlignment) + : 0, + + /* CanVectorizeInner deserves special explanation. It does not affect the product flags. It is not used outside + * of Product. If the Product itself is not a packet-access expression, there is still a chance that the inner + * loop of the product might be vectorized. This is the meaning of CanVectorizeInner. Since it doesn't affect + * the Flags, it is safe to make this value depend on ActualPacketAccessBit, that doesn't affect the ABI. + */ + CanVectorizeInner = SameType && LhsRowMajor && (!RhsRowMajor) && (LhsFlags & RhsFlags & ActualPacketAccessBit) + && (InnerSize % packet_traits::size == 0) + }; - typedef typename find_best_packet::type LhsVecPacketType; - typedef typename find_best_packet::type RhsVecPacketType; - - enum { - - LhsCoeffReadCost = LhsEtorType::CoeffReadCost, - RhsCoeffReadCost = RhsEtorType::CoeffReadCost, - CoeffReadCost = InnerSize==0 ? NumTraits::ReadCost - : InnerSize == Dynamic ? HugeCost - : InnerSize * (NumTraits::MulCost + LhsCoeffReadCost + RhsCoeffReadCost) - + (InnerSize - 1) * NumTraits::AddCost, - - Unroll = CoeffReadCost <= EIGEN_UNROLLING_LIMIT, - - LhsFlags = LhsEtorType::Flags, - RhsFlags = RhsEtorType::Flags, - - LhsRowMajor = LhsFlags & RowMajorBit, - RhsRowMajor = RhsFlags & RowMajorBit, - - LhsVecPacketSize = unpacket_traits::size, - RhsVecPacketSize = unpacket_traits::size, - - // Here, we don't care about alignment larger than the usable packet size. - LhsAlignment = EIGEN_PLAIN_ENUM_MIN(LhsEtorType::Alignment,LhsVecPacketSize*int(sizeof(typename LhsNestedCleaned::Scalar))), - RhsAlignment = EIGEN_PLAIN_ENUM_MIN(RhsEtorType::Alignment,RhsVecPacketSize*int(sizeof(typename RhsNestedCleaned::Scalar))), - - SameType = is_same::value, - - CanVectorizeRhs = bool(RhsRowMajor) && (RhsFlags & PacketAccessBit) && (ColsAtCompileTime!=1), - CanVectorizeLhs = (!LhsRowMajor) && (LhsFlags & PacketAccessBit) && (RowsAtCompileTime!=1), - - EvalToRowMajor = (MaxRowsAtCompileTime==1&&MaxColsAtCompileTime!=1) ? 1 - : (MaxColsAtCompileTime==1&&MaxRowsAtCompileTime!=1) ? 0 - : (bool(RhsRowMajor) && !CanVectorizeLhs), - - Flags = ((unsigned int)(LhsFlags | RhsFlags) & HereditaryBits & ~RowMajorBit) - | (EvalToRowMajor ? RowMajorBit : 0) - // TODO enable vectorization for mixed types - | (SameType && (CanVectorizeLhs || CanVectorizeRhs) ? PacketAccessBit : 0) - | (XprType::IsVectorAtCompileTime ? LinearAccessBit : 0), - - LhsOuterStrideBytes = int(LhsNestedCleaned::OuterStrideAtCompileTime) * int(sizeof(typename LhsNestedCleaned::Scalar)), - RhsOuterStrideBytes = int(RhsNestedCleaned::OuterStrideAtCompileTime) * int(sizeof(typename RhsNestedCleaned::Scalar)), - - Alignment = bool(CanVectorizeLhs) ? (LhsOuterStrideBytes<=0 || (int(LhsOuterStrideBytes) % EIGEN_PLAIN_ENUM_MAX(1,LhsAlignment))!=0 ? 0 : LhsAlignment) - : bool(CanVectorizeRhs) ? (RhsOuterStrideBytes<=0 || (int(RhsOuterStrideBytes) % EIGEN_PLAIN_ENUM_MAX(1,RhsAlignment))!=0 ? 0 : RhsAlignment) - : 0, - - /* CanVectorizeInner deserves special explanation. It does not affect the product flags. It is not used outside - * of Product. If the Product itself is not a packet-access expression, there is still a chance that the inner - * loop of the product might be vectorized. This is the meaning of CanVectorizeInner. Since it doesn't affect - * the Flags, it is safe to make this value depend on ActualPacketAccessBit, that doesn't affect the ABI. + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const CoeffReturnType coeff(Index row, Index col) const + { + return (m_lhs.row(row).transpose().cwiseProduct(m_rhs.col(col))).sum(); + } + + /* Allow index-based non-packet access. It is impossible though to allow index-based packed access, + * which is why we don't set the LinearAccessBit. + * TODO: this seems possible when the result is a vector */ - CanVectorizeInner = SameType - && LhsRowMajor - && (!RhsRowMajor) - && (LhsFlags & RhsFlags & ActualPacketAccessBit) - && (InnerSize % packet_traits::size == 0) - }; - - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const CoeffReturnType coeff(Index row, Index col) const - { - return (m_lhs.row(row).transpose().cwiseProduct( m_rhs.col(col) )).sum(); - } + EIGEN_DEVICE_FUNC const CoeffReturnType coeff(Index index) const + { + const Index row = (RowsAtCompileTime == 1 || MaxRowsAtCompileTime == 1) ? 0 : index; + const Index col = (RowsAtCompileTime == 1 || MaxRowsAtCompileTime == 1) ? index : 0; + return (m_lhs.row(row).transpose().cwiseProduct(m_rhs.col(col))).sum(); + } - /* Allow index-based non-packet access. It is impossible though to allow index-based packed access, - * which is why we don't set the LinearAccessBit. - * TODO: this seems possible when the result is a vector - */ - EIGEN_DEVICE_FUNC const CoeffReturnType coeff(Index index) const - { - const Index row = (RowsAtCompileTime == 1 || MaxRowsAtCompileTime==1) ? 0 : index; - const Index col = (RowsAtCompileTime == 1 || MaxRowsAtCompileTime==1) ? index : 0; - return (m_lhs.row(row).transpose().cwiseProduct( m_rhs.col(col) )).sum(); - } + template const PacketType packet(Index row, Index col) const + { + PacketType res; + typedef etor_product_packet_impl + PacketImpl; + PacketImpl::run(row, col, m_lhsImpl, m_rhsImpl, m_innerDim, res); + return res; + } - template - const PacketType packet(Index row, Index col) const - { - PacketType res; - typedef etor_product_packet_impl PacketImpl; - PacketImpl::run(row, col, m_lhsImpl, m_rhsImpl, m_innerDim, res); - return res; - } + template const PacketType packet(Index index) const + { + const Index row = (RowsAtCompileTime == 1 || MaxRowsAtCompileTime == 1) ? 0 : index; + const Index col = (RowsAtCompileTime == 1 || MaxRowsAtCompileTime == 1) ? index : 0; + return packet(row, col); + } - template - const PacketType packet(Index index) const - { - const Index row = (RowsAtCompileTime == 1 || MaxRowsAtCompileTime==1) ? 0 : index; - const Index col = (RowsAtCompileTime == 1 || MaxRowsAtCompileTime==1) ? index : 0; - return packet(row,col); - } + protected: + typename internal::add_const_on_value_type::type m_lhs; + typename internal::add_const_on_value_type::type m_rhs; + + LhsEtorType m_lhsImpl; + RhsEtorType m_rhsImpl; -protected: - typename internal::add_const_on_value_type::type m_lhs; - typename internal::add_const_on_value_type::type m_rhs; - - LhsEtorType m_lhsImpl; - RhsEtorType m_rhsImpl; - - // TODO: Get rid of m_innerDim if known at compile time - Index m_innerDim; -}; - -template -struct product_evaluator, LazyCoeffBasedProductMode, DenseShape, DenseShape> - : product_evaluator, CoeffBasedProductMode, DenseShape, DenseShape> -{ - typedef Product XprType; - typedef Product BaseProduct; - typedef product_evaluator Base; - enum { - Flags = Base::Flags | EvalBeforeNestingBit + // TODO: Get rid of m_innerDim if known at compile time + Index m_innerDim; }; - EIGEN_DEVICE_FUNC explicit product_evaluator(const XprType& xpr) - : Base(BaseProduct(xpr.lhs(),xpr.rhs())) - {} -}; - -/**************************************** -*** Coeff based product, Packet path *** -****************************************/ - -template -struct etor_product_packet_impl -{ - static EIGEN_STRONG_INLINE void run(Index row, Index col, const Lhs& lhs, const Rhs& rhs, Index innerDim, Packet &res) - { - etor_product_packet_impl::run(row, col, lhs, rhs, innerDim, res); - res = pmadd(pset1(lhs.coeff(row, Index(UnrollingIndex-1))), rhs.template packet(Index(UnrollingIndex-1), col), res); - } -}; -template -struct etor_product_packet_impl -{ - static EIGEN_STRONG_INLINE void run(Index row, Index col, const Lhs& lhs, const Rhs& rhs, Index innerDim, Packet &res) + template + struct product_evaluator, LazyCoeffBasedProductMode, DenseShape, DenseShape> + : product_evaluator, CoeffBasedProductMode, DenseShape, DenseShape> { - etor_product_packet_impl::run(row, col, lhs, rhs, innerDim, res); - res = pmadd(lhs.template packet(row, Index(UnrollingIndex-1)), pset1(rhs.coeff(Index(UnrollingIndex-1), col)), res); - } -}; + typedef Product XprType; + typedef Product BaseProduct; + typedef product_evaluator Base; + enum { Flags = Base::Flags | EvalBeforeNestingBit }; + EIGEN_DEVICE_FUNC explicit product_evaluator(const XprType &xpr) : Base(BaseProduct(xpr.lhs(), xpr.rhs())) {} + }; -template -struct etor_product_packet_impl -{ - static EIGEN_STRONG_INLINE void run(Index row, Index col, const Lhs& lhs, const Rhs& rhs, Index /*innerDim*/, Packet &res) - { - res = pmul(pset1(lhs.coeff(row, Index(0))),rhs.template packet(Index(0), col)); - } -}; + /**************************************** + *** Coeff based product, Packet path *** + ****************************************/ -template -struct etor_product_packet_impl -{ - static EIGEN_STRONG_INLINE void run(Index row, Index col, const Lhs& lhs, const Rhs& rhs, Index /*innerDim*/, Packet &res) + template + struct etor_product_packet_impl { - res = pmul(lhs.template packet(row, Index(0)), pset1(rhs.coeff(Index(0), col))); - } -}; + static EIGEN_STRONG_INLINE void + run(Index row, Index col, const Lhs &lhs, const Rhs &rhs, Index innerDim, Packet &res) + { + etor_product_packet_impl::run( + row, col, lhs, rhs, innerDim, res); + res = pmadd(pset1(lhs.coeff(row, Index(UnrollingIndex - 1))), + rhs.template packet(Index(UnrollingIndex - 1), col), + res); + } + }; -template -struct etor_product_packet_impl -{ - static EIGEN_STRONG_INLINE void run(Index /*row*/, Index /*col*/, const Lhs& /*lhs*/, const Rhs& /*rhs*/, Index /*innerDim*/, Packet &res) + template + struct etor_product_packet_impl { - res = pset1(typename unpacket_traits::type(0)); - } -}; + static EIGEN_STRONG_INLINE void + run(Index row, Index col, const Lhs &lhs, const Rhs &rhs, Index innerDim, Packet &res) + { + etor_product_packet_impl::run( + row, col, lhs, rhs, innerDim, res); + res = pmadd(lhs.template packet(row, Index(UnrollingIndex - 1)), + pset1(rhs.coeff(Index(UnrollingIndex - 1), col)), + res); + } + }; -template -struct etor_product_packet_impl -{ - static EIGEN_STRONG_INLINE void run(Index /*row*/, Index /*col*/, const Lhs& /*lhs*/, const Rhs& /*rhs*/, Index /*innerDim*/, Packet &res) + template + struct etor_product_packet_impl { - res = pset1(typename unpacket_traits::type(0)); - } -}; + static EIGEN_STRONG_INLINE void + run(Index row, Index col, const Lhs &lhs, const Rhs &rhs, Index /*innerDim*/, Packet &res) + { + res = pmul(pset1(lhs.coeff(row, Index(0))), rhs.template packet(Index(0), col)); + } + }; -template -struct etor_product_packet_impl -{ - static EIGEN_STRONG_INLINE void run(Index row, Index col, const Lhs& lhs, const Rhs& rhs, Index innerDim, Packet& res) + template + struct etor_product_packet_impl { - res = pset1(typename unpacket_traits::type(0)); - for(Index i = 0; i < innerDim; ++i) - res = pmadd(pset1(lhs.coeff(row, i)), rhs.template packet(i, col), res); - } -}; + static EIGEN_STRONG_INLINE void + run(Index row, Index col, const Lhs &lhs, const Rhs &rhs, Index /*innerDim*/, Packet &res) + { + res = pmul(lhs.template packet(row, Index(0)), pset1(rhs.coeff(Index(0), col))); + } + }; -template -struct etor_product_packet_impl -{ - static EIGEN_STRONG_INLINE void run(Index row, Index col, const Lhs& lhs, const Rhs& rhs, Index innerDim, Packet& res) + template + struct etor_product_packet_impl { - res = pset1(typename unpacket_traits::type(0)); - for(Index i = 0; i < innerDim; ++i) - res = pmadd(lhs.template packet(row, i), pset1(rhs.coeff(i, col)), res); - } -}; - - -/*************************************************************************** -* Triangular products -***************************************************************************/ -template -struct triangular_product_impl; - -template -struct generic_product_impl - : generic_product_impl_base > -{ - typedef typename Product::Scalar Scalar; - - template - static void scaleAndAddTo(Dest& dst, const Lhs& lhs, const Rhs& rhs, const Scalar& alpha) + static EIGEN_STRONG_INLINE void + run(Index /*row*/, Index /*col*/, const Lhs & /*lhs*/, const Rhs & /*rhs*/, Index /*innerDim*/, Packet &res) + { + res = pset1(typename unpacket_traits::type(0)); + } + }; + + template + struct etor_product_packet_impl { - triangular_product_impl - ::run(dst, lhs.nestedExpression(), rhs, alpha); - } -}; - -template -struct generic_product_impl -: generic_product_impl_base > -{ - typedef typename Product::Scalar Scalar; - - template - static void scaleAndAddTo(Dest& dst, const Lhs& lhs, const Rhs& rhs, const Scalar& alpha) + static EIGEN_STRONG_INLINE void + run(Index /*row*/, Index /*col*/, const Lhs & /*lhs*/, const Rhs & /*rhs*/, Index /*innerDim*/, Packet &res) + { + res = pset1(typename unpacket_traits::type(0)); + } + }; + + template + struct etor_product_packet_impl { - triangular_product_impl::run(dst, lhs, rhs.nestedExpression(), alpha); - } -}; - - -/*************************************************************************** -* SelfAdjoint products -***************************************************************************/ -template -struct selfadjoint_product_impl; - -template -struct generic_product_impl - : generic_product_impl_base > -{ - typedef typename Product::Scalar Scalar; - - template - static void scaleAndAddTo(Dest& dst, const Lhs& lhs, const Rhs& rhs, const Scalar& alpha) + static EIGEN_STRONG_INLINE void + run(Index row, Index col, const Lhs &lhs, const Rhs &rhs, Index innerDim, Packet &res) + { + res = pset1(typename unpacket_traits::type(0)); + for (Index i = 0; i < innerDim; ++i) + res = pmadd(pset1(lhs.coeff(row, i)), rhs.template packet(i, col), res); + } + }; + + template + struct etor_product_packet_impl { - selfadjoint_product_impl::run(dst, lhs.nestedExpression(), rhs, alpha); - } -}; - -template -struct generic_product_impl -: generic_product_impl_base > -{ - typedef typename Product::Scalar Scalar; - - template - static void scaleAndAddTo(Dest& dst, const Lhs& lhs, const Rhs& rhs, const Scalar& alpha) + static EIGEN_STRONG_INLINE void + run(Index row, Index col, const Lhs &lhs, const Rhs &rhs, Index innerDim, Packet &res) + { + res = pset1(typename unpacket_traits::type(0)); + for (Index i = 0; i < innerDim; ++i) + res = pmadd(lhs.template packet(row, i), pset1(rhs.coeff(i, col)), res); + } + }; + + + /*************************************************************************** + * Triangular products + ***************************************************************************/ + template + struct triangular_product_impl; + + template + struct generic_product_impl + : generic_product_impl_base> { - selfadjoint_product_impl::run(dst, lhs, rhs.nestedExpression(), alpha); - } -}; - - -/*************************************************************************** -* Diagonal products -***************************************************************************/ - -template -struct diagonal_product_evaluator_base - : evaluator_base -{ - typedef typename ScalarBinaryOpTraits::ReturnType Scalar; -public: - enum { - CoeffReadCost = NumTraits::MulCost + evaluator::CoeffReadCost + evaluator::CoeffReadCost, - - MatrixFlags = evaluator::Flags, - DiagFlags = evaluator::Flags, - _StorageOrder = MatrixFlags & RowMajorBit ? RowMajor : ColMajor, - _ScalarAccessOnDiag = !((int(_StorageOrder) == ColMajor && int(ProductOrder) == OnTheLeft) - ||(int(_StorageOrder) == RowMajor && int(ProductOrder) == OnTheRight)), - _SameTypes = is_same::value, - // FIXME currently we need same types, but in the future the next rule should be the one - //_Vectorizable = bool(int(MatrixFlags)&PacketAccessBit) && ((!_PacketOnDiag) || (_SameTypes && bool(int(DiagFlags)&PacketAccessBit))), - _Vectorizable = bool(int(MatrixFlags)&PacketAccessBit) && _SameTypes && (_ScalarAccessOnDiag || (bool(int(DiagFlags)&PacketAccessBit))), - _LinearAccessMask = (MatrixType::RowsAtCompileTime==1 || MatrixType::ColsAtCompileTime==1) ? LinearAccessBit : 0, - Flags = ((HereditaryBits|_LinearAccessMask) & (unsigned int)(MatrixFlags)) | (_Vectorizable ? PacketAccessBit : 0), - Alignment = evaluator::Alignment, - - AsScalarProduct = (DiagonalType::SizeAtCompileTime==1) - || (DiagonalType::SizeAtCompileTime==Dynamic && MatrixType::RowsAtCompileTime==1 && ProductOrder==OnTheLeft) - || (DiagonalType::SizeAtCompileTime==Dynamic && MatrixType::ColsAtCompileTime==1 && ProductOrder==OnTheRight) + typedef typename Product::Scalar Scalar; + + template static void scaleAndAddTo(Dest &dst, const Lhs &lhs, const Rhs &rhs, const Scalar &alpha) + { + triangular_product_impl::run( + dst, lhs.nestedExpression(), rhs, alpha); + } }; - - diagonal_product_evaluator_base(const MatrixType &mat, const DiagonalType &diag) - : m_diagImpl(diag), m_matImpl(mat) + + template + struct generic_product_impl + : generic_product_impl_base> { - EIGEN_INTERNAL_CHECK_COST_VALUE(NumTraits::MulCost); - EIGEN_INTERNAL_CHECK_COST_VALUE(CoeffReadCost); - } - - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const Scalar coeff(Index idx) const + typedef typename Product::Scalar Scalar; + + template static void scaleAndAddTo(Dest &dst, const Lhs &lhs, const Rhs &rhs, const Scalar &alpha) + { + triangular_product_impl::run( + dst, lhs, rhs.nestedExpression(), alpha); + } + }; + + + /*************************************************************************** + * SelfAdjoint products + ***************************************************************************/ + template + struct selfadjoint_product_impl; + + template + struct generic_product_impl + : generic_product_impl_base> { - if(AsScalarProduct) - return m_diagImpl.coeff(0) * m_matImpl.coeff(idx); - else - return m_diagImpl.coeff(idx) * m_matImpl.coeff(idx); - } - -protected: - template - EIGEN_STRONG_INLINE PacketType packet_impl(Index row, Index col, Index id, internal::true_type) const + typedef typename Product::Scalar Scalar; + + template static void scaleAndAddTo(Dest &dst, const Lhs &lhs, const Rhs &rhs, const Scalar &alpha) + { + selfadjoint_product_impl::run( + dst, lhs.nestedExpression(), rhs, alpha); + } + }; + + template + struct generic_product_impl + : generic_product_impl_base> { - return internal::pmul(m_matImpl.template packet(row, col), - internal::pset1(m_diagImpl.coeff(id))); - } - - template - EIGEN_STRONG_INLINE PacketType packet_impl(Index row, Index col, Index id, internal::false_type) const + typedef typename Product::Scalar Scalar; + + template static void scaleAndAddTo(Dest &dst, const Lhs &lhs, const Rhs &rhs, const Scalar &alpha) + { + selfadjoint_product_impl::run( + dst, lhs, rhs.nestedExpression(), alpha); + } + }; + + + /*************************************************************************** + * Diagonal products + ***************************************************************************/ + + template + struct diagonal_product_evaluator_base : evaluator_base { + typedef + typename ScalarBinaryOpTraits::ReturnType Scalar; + + public: enum { - InnerSize = (MatrixType::Flags & RowMajorBit) ? MatrixType::ColsAtCompileTime : MatrixType::RowsAtCompileTime, - DiagonalPacketLoadMode = EIGEN_PLAIN_ENUM_MIN(LoadMode,((InnerSize%16) == 0) ? int(Aligned16) : int(evaluator::Alignment)) // FIXME hardcoded 16!! + CoeffReadCost = + NumTraits::MulCost + evaluator::CoeffReadCost + evaluator::CoeffReadCost, + + MatrixFlags = evaluator::Flags, + DiagFlags = evaluator::Flags, + _StorageOrder = MatrixFlags & RowMajorBit ? RowMajor : ColMajor, + _ScalarAccessOnDiag = !((int(_StorageOrder) == ColMajor && int(ProductOrder) == OnTheLeft) + || (int(_StorageOrder) == RowMajor && int(ProductOrder) == OnTheRight)), + _SameTypes = is_same::value, + // FIXME currently we need same types, but in the future the next rule should be the one + //_Vectorizable = bool(int(MatrixFlags)&PacketAccessBit) && ((!_PacketOnDiag) || (_SameTypes && + //bool(int(DiagFlags)&PacketAccessBit))), + _Vectorizable = bool(int(MatrixFlags) & PacketAccessBit) && _SameTypes + && (_ScalarAccessOnDiag || (bool(int(DiagFlags) & PacketAccessBit))), + _LinearAccessMask = + (MatrixType::RowsAtCompileTime == 1 || MatrixType::ColsAtCompileTime == 1) ? LinearAccessBit : 0, + Flags = + ((HereditaryBits | _LinearAccessMask) & (unsigned int)(MatrixFlags)) | (_Vectorizable ? PacketAccessBit : 0), + Alignment = evaluator::Alignment, + + AsScalarProduct = (DiagonalType::SizeAtCompileTime == 1) + || (DiagonalType::SizeAtCompileTime == Dynamic && MatrixType::RowsAtCompileTime == 1 + && ProductOrder == OnTheLeft) + || (DiagonalType::SizeAtCompileTime == Dynamic && MatrixType::ColsAtCompileTime == 1 + && ProductOrder == OnTheRight) }; - return internal::pmul(m_matImpl.template packet(row, col), - m_diagImpl.template packet(id)); - } - - evaluator m_diagImpl; - evaluator m_matImpl; -}; - -// diagonal * dense -template -struct product_evaluator, ProductTag, DiagonalShape, DenseShape> - : diagonal_product_evaluator_base, OnTheLeft> -{ - typedef diagonal_product_evaluator_base, OnTheLeft> Base; - using Base::m_diagImpl; - using Base::m_matImpl; - using Base::coeff; - typedef typename Base::Scalar Scalar; - - typedef Product XprType; - typedef typename XprType::PlainObject PlainObject; - - enum { - StorageOrder = int(Rhs::Flags) & RowMajorBit ? RowMajor : ColMajor + + diagonal_product_evaluator_base(const MatrixType &mat, const DiagonalType &diag) : m_diagImpl(diag), m_matImpl(mat) + { + EIGEN_INTERNAL_CHECK_COST_VALUE(NumTraits::MulCost); + EIGEN_INTERNAL_CHECK_COST_VALUE(CoeffReadCost); + } + + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const Scalar coeff(Index idx) const + { + if (AsScalarProduct) + return m_diagImpl.coeff(0) * m_matImpl.coeff(idx); + else + return m_diagImpl.coeff(idx) * m_matImpl.coeff(idx); + } + + protected: + template + EIGEN_STRONG_INLINE PacketType packet_impl(Index row, Index col, Index id, internal::true_type) const + { + return internal::pmul( + m_matImpl.template packet(row, col), internal::pset1(m_diagImpl.coeff(id))); + } + + template + EIGEN_STRONG_INLINE PacketType packet_impl(Index row, Index col, Index id, internal::false_type) const + { + enum { + InnerSize = (MatrixType::Flags & RowMajorBit) ? MatrixType::ColsAtCompileTime : MatrixType::RowsAtCompileTime, + DiagonalPacketLoadMode = EIGEN_PLAIN_ENUM_MIN(LoadMode, + ((InnerSize % 16) == 0) ? int(Aligned16) : int(evaluator::Alignment))// FIXME hardcoded 16!! + }; + return internal::pmul(m_matImpl.template packet(row, col), + m_diagImpl.template packet(id)); + } + + evaluator m_diagImpl; + evaluator m_matImpl; }; - EIGEN_DEVICE_FUNC explicit product_evaluator(const XprType& xpr) - : Base(xpr.rhs(), xpr.lhs().diagonal()) - { - } - - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const Scalar coeff(Index row, Index col) const + // diagonal * dense + template + struct product_evaluator, ProductTag, DiagonalShape, DenseShape> + : diagonal_product_evaluator_base, OnTheLeft> { - return m_diagImpl.coeff(row) * m_matImpl.coeff(row, col); - } - + typedef diagonal_product_evaluator_base, + OnTheLeft> + Base; + using Base::m_diagImpl; + using Base::m_matImpl; + using Base::coeff; + typedef typename Base::Scalar Scalar; + + typedef Product XprType; + typedef typename XprType::PlainObject PlainObject; + + enum { StorageOrder = int(Rhs::Flags) & RowMajorBit ? RowMajor : ColMajor }; + + EIGEN_DEVICE_FUNC explicit product_evaluator(const XprType &xpr) : Base(xpr.rhs(), xpr.lhs().diagonal()) {} + + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const Scalar coeff(Index row, Index col) const + { + return m_diagImpl.coeff(row) * m_matImpl.coeff(row, col); + } + #ifndef __CUDACC__ - template - EIGEN_STRONG_INLINE PacketType packet(Index row, Index col) const - { - // FIXME: NVCC used to complain about the template keyword, but we have to check whether this is still the case. - // See also similar calls below. - return this->template packet_impl(row,col, row, - typename internal::conditional::type()); - } - - template - EIGEN_STRONG_INLINE PacketType packet(Index idx) const - { - return packet(int(StorageOrder)==ColMajor?idx:0,int(StorageOrder)==ColMajor?0:idx); - } + template EIGEN_STRONG_INLINE PacketType packet(Index row, Index col) const + { + // FIXME: NVCC used to complain about the template keyword, but we have to check whether this is still the case. + // See also similar calls below. + return this->template packet_impl(row, + col, + row, + typename internal::conditional:: + type()); + } + + template EIGEN_STRONG_INLINE PacketType packet(Index idx) const + { + return packet( + int(StorageOrder) == ColMajor ? idx : 0, int(StorageOrder) == ColMajor ? 0 : idx); + } #endif -}; - -// dense * diagonal -template -struct product_evaluator, ProductTag, DenseShape, DiagonalShape> - : diagonal_product_evaluator_base, OnTheRight> -{ - typedef diagonal_product_evaluator_base, OnTheRight> Base; - using Base::m_diagImpl; - using Base::m_matImpl; - using Base::coeff; - typedef typename Base::Scalar Scalar; - - typedef Product XprType; - typedef typename XprType::PlainObject PlainObject; - - enum { StorageOrder = int(Lhs::Flags) & RowMajorBit ? RowMajor : ColMajor }; - - EIGEN_DEVICE_FUNC explicit product_evaluator(const XprType& xpr) - : Base(xpr.lhs(), xpr.rhs().diagonal()) - { - } - - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const Scalar coeff(Index row, Index col) const + }; + + // dense * diagonal + template + struct product_evaluator, ProductTag, DenseShape, DiagonalShape> + : diagonal_product_evaluator_base, OnTheRight> { - return m_matImpl.coeff(row, col) * m_diagImpl.coeff(col); - } - + typedef diagonal_product_evaluator_base, + OnTheRight> + Base; + using Base::m_diagImpl; + using Base::m_matImpl; + using Base::coeff; + typedef typename Base::Scalar Scalar; + + typedef Product XprType; + typedef typename XprType::PlainObject PlainObject; + + enum { StorageOrder = int(Lhs::Flags) & RowMajorBit ? RowMajor : ColMajor }; + + EIGEN_DEVICE_FUNC explicit product_evaluator(const XprType &xpr) : Base(xpr.lhs(), xpr.rhs().diagonal()) {} + + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const Scalar coeff(Index row, Index col) const + { + return m_matImpl.coeff(row, col) * m_diagImpl.coeff(col); + } + #ifndef __CUDACC__ - template - EIGEN_STRONG_INLINE PacketType packet(Index row, Index col) const - { - return this->template packet_impl(row,col, col, - typename internal::conditional::type()); - } - - template - EIGEN_STRONG_INLINE PacketType packet(Index idx) const - { - return packet(int(StorageOrder)==ColMajor?idx:0,int(StorageOrder)==ColMajor?0:idx); - } + template EIGEN_STRONG_INLINE PacketType packet(Index row, Index col) const + { + return this->template packet_impl(row, + col, + col, + typename internal::conditional:: + type()); + } + + template EIGEN_STRONG_INLINE PacketType packet(Index idx) const + { + return packet( + int(StorageOrder) == ColMajor ? idx : 0, int(StorageOrder) == ColMajor ? 0 : idx); + } #endif -}; - -/*************************************************************************** -* Products with permutation matrices -***************************************************************************/ - -/** \internal - * \class permutation_matrix_product - * Internal helper class implementing the product between a permutation matrix and a matrix. - * This class is specialized for DenseShape below and for SparseShape in SparseCore/SparsePermutation.h - */ -template -struct permutation_matrix_product; - -template -struct permutation_matrix_product -{ + }; + + /*************************************************************************** + * Products with permutation matrices + ***************************************************************************/ + + /** \internal + * \class permutation_matrix_product + * Internal helper class implementing the product between a permutation matrix and a matrix. + * This class is specialized for DenseShape below and for SparseShape in SparseCore/SparsePermutation.h + */ + template + struct permutation_matrix_product; + + template + struct permutation_matrix_product + { typedef typename nested_eval::type MatrixType; typedef typename remove_all::type MatrixTypeCleaned; template - static inline void run(Dest& dst, const PermutationType& perm, const ExpressionType& xpr) + static inline void run(Dest &dst, const PermutationType &perm, const ExpressionType &xpr) { MatrixType mat(xpr); - const Index n = Side==OnTheLeft ? mat.rows() : mat.cols(); + const Index n = Side == OnTheLeft ? mat.rows() : mat.cols(); // FIXME we need an is_same for expression that is not sensitive to constness. For instance // is_same_xpr, Block >::value should be true. - //if(is_same::value && extract_data(dst) == extract_data(mat)) - if(is_same_dense(dst, mat)) - { + // if(is_same::value && extract_data(dst) == extract_data(mat)) + if (is_same_dense(dst, mat)) { // apply the permutation inplace - Matrix mask(perm.size()); + Matrix mask(perm.size()); mask.fill(false); Index r = 0; - while(r < perm.size()) - { + while (r < perm.size()) { // search for the next seed - while(r=perm.size()) - break; + while (r < perm.size() && mask[r]) r++; + if (r >= perm.size()) break; // we got one, let's follow it until we are back to the seed Index k0 = r++; Index kPrev = k0; mask.coeffRef(k0) = true; - for(Index k=perm.indices().coeff(k0); k!=k0; k=perm.indices().coeff(k)) - { - Block(dst, k) - .swap(Block - (dst,((Side==OnTheLeft) ^ Transposed) ? k0 : kPrev)); + for (Index k = perm.indices().coeff(k0); k != k0; k = perm.indices().coeff(k)) { + Block(dst, k) + .swap(Block( + dst, ((Side == OnTheLeft) ^ Transposed) ? k0 : kPrev)); mask.coeffRef(k) = true; kPrev = k; } } - } - else - { - for(Index i = 0; i < n; ++i) - { - Block - (dst, ((Side==OnTheLeft) ^ Transposed) ? perm.indices().coeff(i) : i) - - = - - Block - (mat, ((Side==OnTheRight) ^ Transposed) ? perm.indices().coeff(i) : i); + } else { + for (Index i = 0; i < n; ++i) { + Block( + dst, ((Side == OnTheLeft) ^ Transposed) ? perm.indices().coeff(i) : i) + + = + + Block( + mat, ((Side == OnTheRight) ^ Transposed) ? perm.indices().coeff(i) : i); } } } -}; + }; -template -struct generic_product_impl -{ - template - static void evalTo(Dest& dst, const Lhs& lhs, const Rhs& rhs) + template + struct generic_product_impl { - permutation_matrix_product::run(dst, lhs, rhs); - } -}; + template static void evalTo(Dest &dst, const Lhs &lhs, const Rhs &rhs) + { + permutation_matrix_product::run(dst, lhs, rhs); + } + }; -template -struct generic_product_impl -{ - template - static void evalTo(Dest& dst, const Lhs& lhs, const Rhs& rhs) + template + struct generic_product_impl { - permutation_matrix_product::run(dst, rhs, lhs); - } -}; + template static void evalTo(Dest &dst, const Lhs &lhs, const Rhs &rhs) + { + permutation_matrix_product::run(dst, rhs, lhs); + } + }; -template -struct generic_product_impl, Rhs, PermutationShape, MatrixShape, ProductTag> -{ - template - static void evalTo(Dest& dst, const Inverse& lhs, const Rhs& rhs) + template + struct generic_product_impl, Rhs, PermutationShape, MatrixShape, ProductTag> { - permutation_matrix_product::run(dst, lhs.nestedExpression(), rhs); - } -}; + template static void evalTo(Dest &dst, const Inverse &lhs, const Rhs &rhs) + { + permutation_matrix_product::run(dst, lhs.nestedExpression(), rhs); + } + }; -template -struct generic_product_impl, MatrixShape, PermutationShape, ProductTag> -{ - template - static void evalTo(Dest& dst, const Lhs& lhs, const Inverse& rhs) - { - permutation_matrix_product::run(dst, rhs.nestedExpression(), lhs); - } -}; - - -/*************************************************************************** -* Products with transpositions matrices -***************************************************************************/ - -// FIXME could we unify Transpositions and Permutation into a single "shape"?? - -/** \internal - * \class transposition_matrix_product - * Internal helper class implementing the product between a permutation matrix and a matrix. - */ -template -struct transposition_matrix_product -{ - typedef typename nested_eval::type MatrixType; - typedef typename remove_all::type MatrixTypeCleaned; - - template - static inline void run(Dest& dst, const TranspositionType& tr, const ExpressionType& xpr) + template + struct generic_product_impl, MatrixShape, PermutationShape, ProductTag> { - MatrixType mat(xpr); - typedef typename TranspositionType::StorageIndex StorageIndex; - const Index size = tr.size(); - StorageIndex j = 0; + template static void evalTo(Dest &dst, const Lhs &lhs, const Inverse &rhs) + { + permutation_matrix_product::run(dst, rhs.nestedExpression(), lhs); + } + }; - if(!is_same_dense(dst,mat)) - dst = mat; - for(Index k=(Transposed?size-1:0) ; Transposed?k>=0:k -struct generic_product_impl -{ - template - static void evalTo(Dest& dst, const Lhs& lhs, const Rhs& rhs) + // FIXME could we unify Transpositions and Permutation into a single "shape"?? + + /** \internal + * \class transposition_matrix_product + * Internal helper class implementing the product between a permutation matrix and a matrix. + */ + template + struct transposition_matrix_product { - transposition_matrix_product::run(dst, lhs, rhs); - } -}; + typedef typename nested_eval::type MatrixType; + typedef typename remove_all::type MatrixTypeCleaned; -template -struct generic_product_impl -{ - template - static void evalTo(Dest& dst, const Lhs& lhs, const Rhs& rhs) + template + static inline void run(Dest &dst, const TranspositionType &tr, const ExpressionType &xpr) + { + MatrixType mat(xpr); + typedef typename TranspositionType::StorageIndex StorageIndex; + const Index size = tr.size(); + StorageIndex j = 0; + + if (!is_same_dense(dst, mat)) dst = mat; + + for (Index k = (Transposed ? size - 1 : 0); Transposed ? k >= 0 : k < size; Transposed ? --k : ++k) + if (Index(j = tr.coeff(k)) != k) { + if (Side == OnTheLeft) + dst.row(k).swap(dst.row(j)); + else if (Side == OnTheRight) + dst.col(k).swap(dst.col(j)); + } + } + }; + + template + struct generic_product_impl { - transposition_matrix_product::run(dst, rhs, lhs); - } -}; + template static void evalTo(Dest &dst, const Lhs &lhs, const Rhs &rhs) + { + transposition_matrix_product::run(dst, lhs, rhs); + } + }; + + template + struct generic_product_impl + { + template static void evalTo(Dest &dst, const Lhs &lhs, const Rhs &rhs) + { + transposition_matrix_product::run(dst, rhs, lhs); + } + }; -template -struct generic_product_impl, Rhs, TranspositionsShape, MatrixShape, ProductTag> -{ - template - static void evalTo(Dest& dst, const Transpose& lhs, const Rhs& rhs) + template + struct generic_product_impl, Rhs, TranspositionsShape, MatrixShape, ProductTag> { - transposition_matrix_product::run(dst, lhs.nestedExpression(), rhs); - } -}; + template static void evalTo(Dest &dst, const Transpose &lhs, const Rhs &rhs) + { + transposition_matrix_product::run(dst, lhs.nestedExpression(), rhs); + } + }; -template -struct generic_product_impl, MatrixShape, TranspositionsShape, ProductTag> -{ - template - static void evalTo(Dest& dst, const Lhs& lhs, const Transpose& rhs) + template + struct generic_product_impl, MatrixShape, TranspositionsShape, ProductTag> { - transposition_matrix_product::run(dst, rhs.nestedExpression(), lhs); - } -}; + template static void evalTo(Dest &dst, const Lhs &lhs, const Transpose &rhs) + { + transposition_matrix_product::run(dst, rhs.nestedExpression(), lhs); + } + }; -} // end namespace internal +}// end namespace internal -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_PRODUCT_EVALUATORS_H +#endif// EIGEN_PRODUCT_EVALUATORS_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/Random.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/Random.h index 6faf789c..86238fde 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/Random.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/Random.h @@ -10,173 +10,163 @@ #ifndef EIGEN_RANDOM_H #define EIGEN_RANDOM_H -namespace Eigen { +namespace Eigen { namespace internal { -template struct scalar_random_op { - EIGEN_EMPTY_STRUCT_CTOR(scalar_random_op) - inline const Scalar operator() () const { return random(); } -}; + template struct scalar_random_op + { + EIGEN_EMPTY_STRUCT_CTOR(scalar_random_op) + inline const Scalar operator()() const { return random(); } + }; -template -struct functor_traits > -{ enum { Cost = 5 * NumTraits::MulCost, PacketAccess = false, IsRepeatable = false }; }; + template struct functor_traits> + { + enum { Cost = 5 * NumTraits::MulCost, PacketAccess = false, IsRepeatable = false }; + }; -} // end namespace internal +}// end namespace internal /** \returns a random matrix expression - * - * Numbers are uniformly spread through their whole definition range for integer types, - * and in the [-1:1] range for floating point scalar types. - * - * The parameters \a rows and \a cols are the number of rows and of columns of - * the returned matrix. Must be compatible with this MatrixBase type. - * - * \not_reentrant - * - * This variant is meant to be used for dynamic-size matrix types. For fixed-size types, - * it is redundant to pass \a rows and \a cols as arguments, so Random() should be used - * instead. - * - * - * Example: \include MatrixBase_random_int_int.cpp - * Output: \verbinclude MatrixBase_random_int_int.out - * - * This expression has the "evaluate before nesting" flag so that it will be evaluated into - * a temporary matrix whenever it is nested in a larger expression. This prevents unexpected - * behavior with expressions involving random matrices. - * - * See DenseBase::NullaryExpr(Index, const CustomNullaryOp&) for an example using C++11 random generators. - * - * \sa DenseBase::setRandom(), DenseBase::Random(Index), DenseBase::Random() - */ + * + * Numbers are uniformly spread through their whole definition range for integer types, + * and in the [-1:1] range for floating point scalar types. + * + * The parameters \a rows and \a cols are the number of rows and of columns of + * the returned matrix. Must be compatible with this MatrixBase type. + * + * \not_reentrant + * + * This variant is meant to be used for dynamic-size matrix types. For fixed-size types, + * it is redundant to pass \a rows and \a cols as arguments, so Random() should be used + * instead. + * + * + * Example: \include MatrixBase_random_int_int.cpp + * Output: \verbinclude MatrixBase_random_int_int.out + * + * This expression has the "evaluate before nesting" flag so that it will be evaluated into + * a temporary matrix whenever it is nested in a larger expression. This prevents unexpected + * behavior with expressions involving random matrices. + * + * See DenseBase::NullaryExpr(Index, const CustomNullaryOp&) for an example using C++11 random generators. + * + * \sa DenseBase::setRandom(), DenseBase::Random(Index), DenseBase::Random() + */ template -inline const typename DenseBase::RandomReturnType -DenseBase::Random(Index rows, Index cols) +inline const typename DenseBase::RandomReturnType DenseBase::Random(Index rows, Index cols) { return NullaryExpr(rows, cols, internal::scalar_random_op()); } /** \returns a random vector expression - * - * Numbers are uniformly spread through their whole definition range for integer types, - * and in the [-1:1] range for floating point scalar types. - * - * The parameter \a size is the size of the returned vector. - * Must be compatible with this MatrixBase type. - * - * \only_for_vectors - * \not_reentrant - * - * This variant is meant to be used for dynamic-size vector types. For fixed-size types, - * it is redundant to pass \a size as argument, so Random() should be used - * instead. - * - * Example: \include MatrixBase_random_int.cpp - * Output: \verbinclude MatrixBase_random_int.out - * - * This expression has the "evaluate before nesting" flag so that it will be evaluated into - * a temporary vector whenever it is nested in a larger expression. This prevents unexpected - * behavior with expressions involving random matrices. - * - * \sa DenseBase::setRandom(), DenseBase::Random(Index,Index), DenseBase::Random() - */ + * + * Numbers are uniformly spread through their whole definition range for integer types, + * and in the [-1:1] range for floating point scalar types. + * + * The parameter \a size is the size of the returned vector. + * Must be compatible with this MatrixBase type. + * + * \only_for_vectors + * \not_reentrant + * + * This variant is meant to be used for dynamic-size vector types. For fixed-size types, + * it is redundant to pass \a size as argument, so Random() should be used + * instead. + * + * Example: \include MatrixBase_random_int.cpp + * Output: \verbinclude MatrixBase_random_int.out + * + * This expression has the "evaluate before nesting" flag so that it will be evaluated into + * a temporary vector whenever it is nested in a larger expression. This prevents unexpected + * behavior with expressions involving random matrices. + * + * \sa DenseBase::setRandom(), DenseBase::Random(Index,Index), DenseBase::Random() + */ template -inline const typename DenseBase::RandomReturnType -DenseBase::Random(Index size) +inline const typename DenseBase::RandomReturnType DenseBase::Random(Index size) { return NullaryExpr(size, internal::scalar_random_op()); } /** \returns a fixed-size random matrix or vector expression - * - * Numbers are uniformly spread through their whole definition range for integer types, - * and in the [-1:1] range for floating point scalar types. - * - * This variant is only for fixed-size MatrixBase types. For dynamic-size types, you - * need to use the variants taking size arguments. - * - * Example: \include MatrixBase_random.cpp - * Output: \verbinclude MatrixBase_random.out - * - * This expression has the "evaluate before nesting" flag so that it will be evaluated into - * a temporary matrix whenever it is nested in a larger expression. This prevents unexpected - * behavior with expressions involving random matrices. - * - * \not_reentrant - * - * \sa DenseBase::setRandom(), DenseBase::Random(Index,Index), DenseBase::Random(Index) - */ -template -inline const typename DenseBase::RandomReturnType -DenseBase::Random() + * + * Numbers are uniformly spread through their whole definition range for integer types, + * and in the [-1:1] range for floating point scalar types. + * + * This variant is only for fixed-size MatrixBase types. For dynamic-size types, you + * need to use the variants taking size arguments. + * + * Example: \include MatrixBase_random.cpp + * Output: \verbinclude MatrixBase_random.out + * + * This expression has the "evaluate before nesting" flag so that it will be evaluated into + * a temporary matrix whenever it is nested in a larger expression. This prevents unexpected + * behavior with expressions involving random matrices. + * + * \not_reentrant + * + * \sa DenseBase::setRandom(), DenseBase::Random(Index,Index), DenseBase::Random(Index) + */ +template inline const typename DenseBase::RandomReturnType DenseBase::Random() { return NullaryExpr(RowsAtCompileTime, ColsAtCompileTime, internal::scalar_random_op()); } /** Sets all coefficients in this expression to random values. - * - * Numbers are uniformly spread through their whole definition range for integer types, - * and in the [-1:1] range for floating point scalar types. - * - * \not_reentrant - * - * Example: \include MatrixBase_setRandom.cpp - * Output: \verbinclude MatrixBase_setRandom.out - * - * \sa class CwiseNullaryOp, setRandom(Index), setRandom(Index,Index) - */ -template -inline Derived& DenseBase::setRandom() -{ - return *this = Random(rows(), cols()); -} + * + * Numbers are uniformly spread through their whole definition range for integer types, + * and in the [-1:1] range for floating point scalar types. + * + * \not_reentrant + * + * Example: \include MatrixBase_setRandom.cpp + * Output: \verbinclude MatrixBase_setRandom.out + * + * \sa class CwiseNullaryOp, setRandom(Index), setRandom(Index,Index) + */ +template inline Derived &DenseBase::setRandom() { return *this = Random(rows(), cols()); } /** Resizes to the given \a newSize, and sets all coefficients in this expression to random values. - * - * Numbers are uniformly spread through their whole definition range for integer types, - * and in the [-1:1] range for floating point scalar types. - * - * \only_for_vectors - * \not_reentrant - * - * Example: \include Matrix_setRandom_int.cpp - * Output: \verbinclude Matrix_setRandom_int.out - * - * \sa DenseBase::setRandom(), setRandom(Index,Index), class CwiseNullaryOp, DenseBase::Random() - */ -template -EIGEN_STRONG_INLINE Derived& -PlainObjectBase::setRandom(Index newSize) + * + * Numbers are uniformly spread through their whole definition range for integer types, + * and in the [-1:1] range for floating point scalar types. + * + * \only_for_vectors + * \not_reentrant + * + * Example: \include Matrix_setRandom_int.cpp + * Output: \verbinclude Matrix_setRandom_int.out + * + * \sa DenseBase::setRandom(), setRandom(Index,Index), class CwiseNullaryOp, DenseBase::Random() + */ +template EIGEN_STRONG_INLINE Derived &PlainObjectBase::setRandom(Index newSize) { resize(newSize); return setRandom(); } /** Resizes to the given size, and sets all coefficients in this expression to random values. - * - * Numbers are uniformly spread through their whole definition range for integer types, - * and in the [-1:1] range for floating point scalar types. - * - * \not_reentrant - * - * \param rows the new number of rows - * \param cols the new number of columns - * - * Example: \include Matrix_setRandom_int_int.cpp - * Output: \verbinclude Matrix_setRandom_int_int.out - * - * \sa DenseBase::setRandom(), setRandom(Index), class CwiseNullaryOp, DenseBase::Random() - */ -template -EIGEN_STRONG_INLINE Derived& -PlainObjectBase::setRandom(Index rows, Index cols) + * + * Numbers are uniformly spread through their whole definition range for integer types, + * and in the [-1:1] range for floating point scalar types. + * + * \not_reentrant + * + * \param rows the new number of rows + * \param cols the new number of columns + * + * Example: \include Matrix_setRandom_int_int.cpp + * Output: \verbinclude Matrix_setRandom_int_int.out + * + * \sa DenseBase::setRandom(), setRandom(Index), class CwiseNullaryOp, DenseBase::Random() + */ +template EIGEN_STRONG_INLINE Derived &PlainObjectBase::setRandom(Index rows, Index cols) { resize(rows, cols); return setRandom(); } -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_RANDOM_H +#endif// EIGEN_RANDOM_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/Redux.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/Redux.h index 760e9f86..838af4f8 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/Redux.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/Redux.h @@ -11,495 +11,457 @@ #ifndef EIGEN_REDUX_H #define EIGEN_REDUX_H -namespace Eigen { +namespace Eigen { namespace internal { -// TODO -// * implement other kind of vectorization -// * factorize code + // TODO + // * implement other kind of vectorization + // * factorize code -/*************************************************************************** -* Part 1 : the logic deciding a strategy for vectorization and unrolling -***************************************************************************/ + /*************************************************************************** + * Part 1 : the logic deciding a strategy for vectorization and unrolling + ***************************************************************************/ -template -struct redux_traits -{ -public: - typedef typename find_best_packet::type PacketType; - enum { - PacketSize = unpacket_traits::size, - InnerMaxSize = int(Derived::IsRowMajor) - ? Derived::MaxColsAtCompileTime - : Derived::MaxRowsAtCompileTime - }; + template struct redux_traits + { + public: + typedef typename find_best_packet::type PacketType; + enum { + PacketSize = unpacket_traits::size, + InnerMaxSize = int(Derived::IsRowMajor) ? Derived::MaxColsAtCompileTime : Derived::MaxRowsAtCompileTime + }; - enum { - MightVectorize = (int(Derived::Flags)&ActualPacketAccessBit) - && (functor_traits::PacketAccess), - MayLinearVectorize = bool(MightVectorize) && (int(Derived::Flags)&LinearAccessBit), - MaySliceVectorize = bool(MightVectorize) && int(InnerMaxSize)>=3*PacketSize - }; + enum { + MightVectorize = (int(Derived::Flags) & ActualPacketAccessBit) && (functor_traits::PacketAccess), + MayLinearVectorize = bool(MightVectorize) && (int(Derived::Flags) & LinearAccessBit), + MaySliceVectorize = bool(MightVectorize) && int(InnerMaxSize) >= 3 * PacketSize + }; -public: - enum { - Traversal = int(MayLinearVectorize) ? int(LinearVectorizedTraversal) - : int(MaySliceVectorize) ? int(SliceVectorizedTraversal) - : int(DefaultTraversal) - }; + public: + enum { + Traversal = int(MayLinearVectorize) ? int(LinearVectorizedTraversal) + : int(MaySliceVectorize) ? int(SliceVectorizedTraversal) + : int(DefaultTraversal) + }; -public: - enum { - Cost = Derived::SizeAtCompileTime == Dynamic ? HugeCost - : Derived::SizeAtCompileTime * Derived::CoeffReadCost + (Derived::SizeAtCompileTime-1) * functor_traits::Cost, - UnrollingLimit = EIGEN_UNROLLING_LIMIT * (int(Traversal) == int(DefaultTraversal) ? 1 : int(PacketSize)) - }; + public: + enum { + Cost = Derived::SizeAtCompileTime == Dynamic ? HugeCost + : Derived::SizeAtCompileTime * Derived::CoeffReadCost + + (Derived::SizeAtCompileTime - 1) * functor_traits::Cost, + UnrollingLimit = EIGEN_UNROLLING_LIMIT * (int(Traversal) == int(DefaultTraversal) ? 1 : int(PacketSize)) + }; + + public: + enum { Unrolling = Cost <= UnrollingLimit ? CompleteUnrolling : NoUnrolling }; -public: - enum { - Unrolling = Cost <= UnrollingLimit ? CompleteUnrolling : NoUnrolling - }; - #ifdef EIGEN_DEBUG_ASSIGN - static void debug() - { - std::cerr << "Xpr: " << typeid(typename Derived::XprType).name() << std::endl; - std::cerr.setf(std::ios::hex, std::ios::basefield); - EIGEN_DEBUG_VAR(Derived::Flags) - std::cerr.unsetf(std::ios::hex); - EIGEN_DEBUG_VAR(InnerMaxSize) - EIGEN_DEBUG_VAR(PacketSize) - EIGEN_DEBUG_VAR(MightVectorize) - EIGEN_DEBUG_VAR(MayLinearVectorize) - EIGEN_DEBUG_VAR(MaySliceVectorize) - EIGEN_DEBUG_VAR(Traversal) - EIGEN_DEBUG_VAR(UnrollingLimit) - EIGEN_DEBUG_VAR(Unrolling) - std::cerr << std::endl; - } + static void debug() + { + std::cerr << "Xpr: " << typeid(typename Derived::XprType).name() << std::endl; + std::cerr.setf(std::ios::hex, std::ios::basefield); + EIGEN_DEBUG_VAR(Derived::Flags) + std::cerr.unsetf(std::ios::hex); + EIGEN_DEBUG_VAR(InnerMaxSize) + EIGEN_DEBUG_VAR(PacketSize) + EIGEN_DEBUG_VAR(MightVectorize) + EIGEN_DEBUG_VAR(MayLinearVectorize) + EIGEN_DEBUG_VAR(MaySliceVectorize) + EIGEN_DEBUG_VAR(Traversal) + EIGEN_DEBUG_VAR(UnrollingLimit) + EIGEN_DEBUG_VAR(Unrolling) + std::cerr << std::endl; + } #endif -}; + }; -/*************************************************************************** -* Part 2 : unrollers -***************************************************************************/ + /*************************************************************************** + * Part 2 : unrollers + ***************************************************************************/ -/*** no vectorization ***/ + /*** no vectorization ***/ -template -struct redux_novec_unroller -{ - enum { - HalfLength = Length/2 - }; + template struct redux_novec_unroller + { + enum { HalfLength = Length / 2 }; + + typedef typename Derived::Scalar Scalar; - typedef typename Derived::Scalar Scalar; + EIGEN_DEVICE_FUNC + static EIGEN_STRONG_INLINE Scalar run(const Derived &mat, const Func &func) + { + return func(redux_novec_unroller::run(mat, func), + redux_novec_unroller::run(mat, func)); + } + }; - EIGEN_DEVICE_FUNC - static EIGEN_STRONG_INLINE Scalar run(const Derived &mat, const Func& func) + template struct redux_novec_unroller { - return func(redux_novec_unroller::run(mat,func), - redux_novec_unroller::run(mat,func)); - } -}; + enum { outer = Start / Derived::InnerSizeAtCompileTime, inner = Start % Derived::InnerSizeAtCompileTime }; -template -struct redux_novec_unroller -{ - enum { - outer = Start / Derived::InnerSizeAtCompileTime, - inner = Start % Derived::InnerSizeAtCompileTime + typedef typename Derived::Scalar Scalar; + + EIGEN_DEVICE_FUNC + static EIGEN_STRONG_INLINE Scalar run(const Derived &mat, const Func &) + { + return mat.coeffByOuterInner(outer, inner); + } }; - typedef typename Derived::Scalar Scalar; + // This is actually dead code and will never be called. It is required + // to prevent false warnings regarding failed inlining though + // for 0 length run() will never be called at all. + template struct redux_novec_unroller + { + typedef typename Derived::Scalar Scalar; + EIGEN_DEVICE_FUNC + static EIGEN_STRONG_INLINE Scalar run(const Derived &, const Func &) { return Scalar(); } + }; + + /*** vectorization ***/ - EIGEN_DEVICE_FUNC - static EIGEN_STRONG_INLINE Scalar run(const Derived &mat, const Func&) + template struct redux_vec_unroller { - return mat.coeffByOuterInner(outer, inner); - } -}; - -// This is actually dead code and will never be called. It is required -// to prevent false warnings regarding failed inlining though -// for 0 length run() will never be called at all. -template -struct redux_novec_unroller -{ - typedef typename Derived::Scalar Scalar; - EIGEN_DEVICE_FUNC - static EIGEN_STRONG_INLINE Scalar run(const Derived&, const Func&) { return Scalar(); } -}; + enum { PacketSize = redux_traits::PacketSize, HalfLength = Length / 2 }; -/*** vectorization ***/ + typedef typename Derived::Scalar Scalar; + typedef typename redux_traits::PacketType PacketScalar; -template -struct redux_vec_unroller -{ - enum { - PacketSize = redux_traits::PacketSize, - HalfLength = Length/2 + static EIGEN_STRONG_INLINE PacketScalar run(const Derived &mat, const Func &func) + { + return func.packetOp(redux_vec_unroller::run(mat, func), + redux_vec_unroller::run(mat, func)); + } }; - typedef typename Derived::Scalar Scalar; - typedef typename redux_traits::PacketType PacketScalar; - - static EIGEN_STRONG_INLINE PacketScalar run(const Derived &mat, const Func& func) + template struct redux_vec_unroller { - return func.packetOp( - redux_vec_unroller::run(mat,func), - redux_vec_unroller::run(mat,func) ); - } -}; - -template -struct redux_vec_unroller -{ - enum { - index = Start * redux_traits::PacketSize, - outer = index / int(Derived::InnerSizeAtCompileTime), - inner = index % int(Derived::InnerSizeAtCompileTime), - alignment = Derived::Alignment - }; + enum { + index = Start * redux_traits::PacketSize, + outer = index / int(Derived::InnerSizeAtCompileTime), + inner = index % int(Derived::InnerSizeAtCompileTime), + alignment = Derived::Alignment + }; - typedef typename Derived::Scalar Scalar; - typedef typename redux_traits::PacketType PacketScalar; + typedef typename Derived::Scalar Scalar; + typedef typename redux_traits::PacketType PacketScalar; - static EIGEN_STRONG_INLINE PacketScalar run(const Derived &mat, const Func&) - { - return mat.template packetByOuterInner(outer, inner); - } -}; + static EIGEN_STRONG_INLINE PacketScalar run(const Derived &mat, const Func &) + { + return mat.template packetByOuterInner(outer, inner); + } + }; -/*************************************************************************** -* Part 3 : implementation of all cases -***************************************************************************/ + /*************************************************************************** + * Part 3 : implementation of all cases + ***************************************************************************/ -template::Traversal, - int Unrolling = redux_traits::Unrolling -> -struct redux_impl; + template::Traversal, + int Unrolling = redux_traits::Unrolling> + struct redux_impl; -template -struct redux_impl -{ - typedef typename Derived::Scalar Scalar; - EIGEN_DEVICE_FUNC - static EIGEN_STRONG_INLINE Scalar run(const Derived &mat, const Func& func) + template struct redux_impl { - eigen_assert(mat.rows()>0 && mat.cols()>0 && "you are using an empty matrix"); - Scalar res; - res = mat.coeffByOuterInner(0, 0); - for(Index i = 1; i < mat.innerSize(); ++i) - res = func(res, mat.coeffByOuterInner(0, i)); - for(Index i = 1; i < mat.outerSize(); ++i) - for(Index j = 0; j < mat.innerSize(); ++j) - res = func(res, mat.coeffByOuterInner(i, j)); - return res; - } -}; - -template -struct redux_impl - : public redux_novec_unroller -{}; - -template -struct redux_impl -{ - typedef typename Derived::Scalar Scalar; - typedef typename redux_traits::PacketType PacketScalar; + typedef typename Derived::Scalar Scalar; + EIGEN_DEVICE_FUNC + static EIGEN_STRONG_INLINE Scalar run(const Derived &mat, const Func &func) + { + eigen_assert(mat.rows() > 0 && mat.cols() > 0 && "you are using an empty matrix"); + Scalar res; + res = mat.coeffByOuterInner(0, 0); + for (Index i = 1; i < mat.innerSize(); ++i) res = func(res, mat.coeffByOuterInner(0, i)); + for (Index i = 1; i < mat.outerSize(); ++i) + for (Index j = 0; j < mat.innerSize(); ++j) res = func(res, mat.coeffByOuterInner(i, j)); + return res; + } + }; - static Scalar run(const Derived &mat, const Func& func) + template + struct redux_impl + : public redux_novec_unroller { - const Index size = mat.size(); - - const Index packetSize = redux_traits::PacketSize; - const int packetAlignment = unpacket_traits::alignment; - enum { - alignment0 = (bool(Derived::Flags & DirectAccessBit) && bool(packet_traits::AlignedOnScalar)) ? int(packetAlignment) : int(Unaligned), - alignment = EIGEN_PLAIN_ENUM_MAX(alignment0, Derived::Alignment) - }; - const Index alignedStart = internal::first_default_aligned(mat.nestedExpression()); - const Index alignedSize2 = ((size-alignedStart)/(2*packetSize))*(2*packetSize); - const Index alignedSize = ((size-alignedStart)/(packetSize))*(packetSize); - const Index alignedEnd2 = alignedStart + alignedSize2; - const Index alignedEnd = alignedStart + alignedSize; - Scalar res; - if(alignedSize) + }; + + template struct redux_impl + { + typedef typename Derived::Scalar Scalar; + typedef typename redux_traits::PacketType PacketScalar; + + static Scalar run(const Derived &mat, const Func &func) { - PacketScalar packet_res0 = mat.template packet(alignedStart); - if(alignedSize>packetSize) // we have at least two packets to partly unroll the loop - { - PacketScalar packet_res1 = mat.template packet(alignedStart+packetSize); - for(Index index = alignedStart + 2*packetSize; index < alignedEnd2; index += 2*packetSize) + const Index size = mat.size(); + + const Index packetSize = redux_traits::PacketSize; + const int packetAlignment = unpacket_traits::alignment; + enum { + alignment0 = (bool(Derived::Flags & DirectAccessBit) && bool(packet_traits::AlignedOnScalar)) + ? int(packetAlignment) + : int(Unaligned), + alignment = EIGEN_PLAIN_ENUM_MAX(alignment0, Derived::Alignment) + }; + const Index alignedStart = internal::first_default_aligned(mat.nestedExpression()); + const Index alignedSize2 = ((size - alignedStart) / (2 * packetSize)) * (2 * packetSize); + const Index alignedSize = ((size - alignedStart) / (packetSize)) * (packetSize); + const Index alignedEnd2 = alignedStart + alignedSize2; + const Index alignedEnd = alignedStart + alignedSize; + Scalar res; + if (alignedSize) { + PacketScalar packet_res0 = mat.template packet(alignedStart); + if (alignedSize > packetSize)// we have at least two packets to partly unroll the loop { - packet_res0 = func.packetOp(packet_res0, mat.template packet(index)); - packet_res1 = func.packetOp(packet_res1, mat.template packet(index+packetSize)); + PacketScalar packet_res1 = mat.template packet(alignedStart + packetSize); + for (Index index = alignedStart + 2 * packetSize; index < alignedEnd2; index += 2 * packetSize) { + packet_res0 = func.packetOp(packet_res0, mat.template packet(index)); + packet_res1 = func.packetOp(packet_res1, mat.template packet(index + packetSize)); + } + + packet_res0 = func.packetOp(packet_res0, packet_res1); + if (alignedEnd > alignedEnd2) + packet_res0 = func.packetOp(packet_res0, mat.template packet(alignedEnd2)); } + res = func.predux(packet_res0); - packet_res0 = func.packetOp(packet_res0,packet_res1); - if(alignedEnd>alignedEnd2) - packet_res0 = func.packetOp(packet_res0, mat.template packet(alignedEnd2)); - } - res = func.predux(packet_res0); + for (Index index = 0; index < alignedStart; ++index) res = func(res, mat.coeff(index)); - for(Index index = 0; index < alignedStart; ++index) - res = func(res,mat.coeff(index)); + for (Index index = alignedEnd; index < size; ++index) res = func(res, mat.coeff(index)); + } else// too small to vectorize anything. + // since this is dynamic-size hence inefficient anyway for such small sizes, don't try to optimize. + { + res = mat.coeff(0); + for (Index index = 1; index < size; ++index) res = func(res, mat.coeff(index)); + } - for(Index index = alignedEnd; index < size; ++index) - res = func(res,mat.coeff(index)); + return res; } - else // too small to vectorize anything. - // since this is dynamic-size hence inefficient anyway for such small sizes, don't try to optimize. + }; + + // NOTE: for SliceVectorizedTraversal we simply bypass unrolling + template + struct redux_impl + { + typedef typename Derived::Scalar Scalar; + typedef typename redux_traits::PacketType PacketType; + + EIGEN_DEVICE_FUNC static Scalar run(const Derived &mat, const Func &func) { - res = mat.coeff(0); - for(Index index = 1; index < size; ++index) - res = func(res,mat.coeff(index)); + eigen_assert(mat.rows() > 0 && mat.cols() > 0 && "you are using an empty matrix"); + const Index innerSize = mat.innerSize(); + const Index outerSize = mat.outerSize(); + enum { packetSize = redux_traits::PacketSize }; + const Index packetedInnerSize = ((innerSize) / packetSize) * packetSize; + Scalar res; + if (packetedInnerSize) { + PacketType packet_res = mat.template packet(0, 0); + for (Index j = 0; j < outerSize; ++j) + for (Index i = (j == 0 ? packetSize : 0); i < packetedInnerSize; i += Index(packetSize)) + packet_res = func.packetOp(packet_res, mat.template packetByOuterInner(j, i)); + + res = func.predux(packet_res); + for (Index j = 0; j < outerSize; ++j) + for (Index i = packetedInnerSize; i < innerSize; ++i) res = func(res, mat.coeffByOuterInner(j, i)); + } else// too small to vectorize anything. + // since this is dynamic-size hence inefficient anyway for such small sizes, don't try to optimize. + { + res = redux_impl::run(mat, func); + } + + return res; } + }; - return res; - } -}; + template + struct redux_impl + { + typedef typename Derived::Scalar Scalar; -// NOTE: for SliceVectorizedTraversal we simply bypass unrolling -template -struct redux_impl -{ - typedef typename Derived::Scalar Scalar; - typedef typename redux_traits::PacketType PacketType; + typedef typename redux_traits::PacketType PacketScalar; + enum { + PacketSize = redux_traits::PacketSize, + Size = Derived::SizeAtCompileTime, + VectorizedSize = (Size / PacketSize) * PacketSize + }; + EIGEN_DEVICE_FUNC static EIGEN_STRONG_INLINE Scalar run(const Derived &mat, const Func &func) + { + eigen_assert(mat.rows() > 0 && mat.cols() > 0 && "you are using an empty matrix"); + if (VectorizedSize > 0) { + Scalar res = func.predux(redux_vec_unroller::run(mat, func)); + if (VectorizedSize != Size) + res = func(res, redux_novec_unroller::run(mat, func)); + return res; + } else { + return redux_novec_unroller::run(mat, func); + } + } + }; - EIGEN_DEVICE_FUNC static Scalar run(const Derived &mat, const Func& func) + // evaluator adaptor + template class redux_evaluator { - eigen_assert(mat.rows()>0 && mat.cols()>0 && "you are using an empty matrix"); - const Index innerSize = mat.innerSize(); - const Index outerSize = mat.outerSize(); + public: + typedef _XprType XprType; + EIGEN_DEVICE_FUNC explicit redux_evaluator(const XprType &xpr) : m_evaluator(xpr), m_xpr(xpr) {} + + typedef typename XprType::Scalar Scalar; + typedef typename XprType::CoeffReturnType CoeffReturnType; + typedef typename XprType::PacketScalar PacketScalar; + typedef typename XprType::PacketReturnType PacketReturnType; + enum { - packetSize = redux_traits::PacketSize + MaxRowsAtCompileTime = XprType::MaxRowsAtCompileTime, + MaxColsAtCompileTime = XprType::MaxColsAtCompileTime, + // TODO we should not remove DirectAccessBit and rather find an elegant way to query the alignment offset at + // runtime from the evaluator + Flags = evaluator::Flags & ~DirectAccessBit, + IsRowMajor = XprType::IsRowMajor, + SizeAtCompileTime = XprType::SizeAtCompileTime, + InnerSizeAtCompileTime = XprType::InnerSizeAtCompileTime, + CoeffReadCost = evaluator::CoeffReadCost, + Alignment = evaluator::Alignment }; - const Index packetedInnerSize = ((innerSize)/packetSize)*packetSize; - Scalar res; - if(packetedInnerSize) + + EIGEN_DEVICE_FUNC Index rows() const { return m_xpr.rows(); } + EIGEN_DEVICE_FUNC Index cols() const { return m_xpr.cols(); } + EIGEN_DEVICE_FUNC Index size() const { return m_xpr.size(); } + EIGEN_DEVICE_FUNC Index innerSize() const { return m_xpr.innerSize(); } + EIGEN_DEVICE_FUNC Index outerSize() const { return m_xpr.outerSize(); } + + EIGEN_DEVICE_FUNC + CoeffReturnType coeff(Index row, Index col) const { return m_evaluator.coeff(row, col); } + + EIGEN_DEVICE_FUNC + CoeffReturnType coeff(Index index) const { return m_evaluator.coeff(index); } + + template PacketType packet(Index row, Index col) const + { + return m_evaluator.template packet(row, col); + } + + template PacketType packet(Index index) const { - PacketType packet_res = mat.template packet(0,0); - for(Index j=0; j(j,i)); - - res = func.predux(packet_res); - for(Index j=0; j(index); } - else // too small to vectorize anything. - // since this is dynamic-size hence inefficient anyway for such small sizes, don't try to optimize. + + EIGEN_DEVICE_FUNC + CoeffReturnType coeffByOuterInner(Index outer, Index inner) const { - res = redux_impl::run(mat, func); + return m_evaluator.coeff(IsRowMajor ? outer : inner, IsRowMajor ? inner : outer); } - return res; - } -}; + template PacketType packetByOuterInner(Index outer, Index inner) const + { + return m_evaluator.template packet(IsRowMajor ? outer : inner, IsRowMajor ? inner : outer); + } -template -struct redux_impl -{ - typedef typename Derived::Scalar Scalar; + const XprType &nestedExpression() const { return m_xpr; } - typedef typename redux_traits::PacketType PacketScalar; - enum { - PacketSize = redux_traits::PacketSize, - Size = Derived::SizeAtCompileTime, - VectorizedSize = (Size / PacketSize) * PacketSize + protected: + internal::evaluator m_evaluator; + const XprType &m_xpr; }; - EIGEN_DEVICE_FUNC static EIGEN_STRONG_INLINE Scalar run(const Derived &mat, const Func& func) - { - eigen_assert(mat.rows()>0 && mat.cols()>0 && "you are using an empty matrix"); - if (VectorizedSize > 0) { - Scalar res = func.predux(redux_vec_unroller::run(mat,func)); - if (VectorizedSize != Size) - res = func(res,redux_novec_unroller::run(mat,func)); - return res; - } - else { - return redux_novec_unroller::run(mat,func); - } - } -}; -// evaluator adaptor -template -class redux_evaluator -{ -public: - typedef _XprType XprType; - EIGEN_DEVICE_FUNC explicit redux_evaluator(const XprType &xpr) : m_evaluator(xpr), m_xpr(xpr) {} - - typedef typename XprType::Scalar Scalar; - typedef typename XprType::CoeffReturnType CoeffReturnType; - typedef typename XprType::PacketScalar PacketScalar; - typedef typename XprType::PacketReturnType PacketReturnType; - - enum { - MaxRowsAtCompileTime = XprType::MaxRowsAtCompileTime, - MaxColsAtCompileTime = XprType::MaxColsAtCompileTime, - // TODO we should not remove DirectAccessBit and rather find an elegant way to query the alignment offset at runtime from the evaluator - Flags = evaluator::Flags & ~DirectAccessBit, - IsRowMajor = XprType::IsRowMajor, - SizeAtCompileTime = XprType::SizeAtCompileTime, - InnerSizeAtCompileTime = XprType::InnerSizeAtCompileTime, - CoeffReadCost = evaluator::CoeffReadCost, - Alignment = evaluator::Alignment - }; - - EIGEN_DEVICE_FUNC Index rows() const { return m_xpr.rows(); } - EIGEN_DEVICE_FUNC Index cols() const { return m_xpr.cols(); } - EIGEN_DEVICE_FUNC Index size() const { return m_xpr.size(); } - EIGEN_DEVICE_FUNC Index innerSize() const { return m_xpr.innerSize(); } - EIGEN_DEVICE_FUNC Index outerSize() const { return m_xpr.outerSize(); } - - EIGEN_DEVICE_FUNC - CoeffReturnType coeff(Index row, Index col) const - { return m_evaluator.coeff(row, col); } - - EIGEN_DEVICE_FUNC - CoeffReturnType coeff(Index index) const - { return m_evaluator.coeff(index); } - - template - PacketType packet(Index row, Index col) const - { return m_evaluator.template packet(row, col); } - - template - PacketType packet(Index index) const - { return m_evaluator.template packet(index); } - - EIGEN_DEVICE_FUNC - CoeffReturnType coeffByOuterInner(Index outer, Index inner) const - { return m_evaluator.coeff(IsRowMajor ? outer : inner, IsRowMajor ? inner : outer); } - - template - PacketType packetByOuterInner(Index outer, Index inner) const - { return m_evaluator.template packet(IsRowMajor ? outer : inner, IsRowMajor ? inner : outer); } - - const XprType & nestedExpression() const { return m_xpr; } - -protected: - internal::evaluator m_evaluator; - const XprType &m_xpr; -}; - -} // end namespace internal +}// end namespace internal /*************************************************************************** -* Part 4 : public API -***************************************************************************/ + * Part 4 : public API + ***************************************************************************/ /** \returns the result of a full redux operation on the whole matrix or vector using \a func - * - * The template parameter \a BinaryOp is the type of the functor \a func which must be - * an associative operator. Both current C++98 and C++11 functor styles are handled. - * - * \sa DenseBase::sum(), DenseBase::minCoeff(), DenseBase::maxCoeff(), MatrixBase::colwise(), MatrixBase::rowwise() - */ + * + * The template parameter \a BinaryOp is the type of the functor \a func which must be + * an associative operator. Both current C++98 and C++11 functor styles are handled. + * + * \sa DenseBase::sum(), DenseBase::minCoeff(), DenseBase::maxCoeff(), MatrixBase::colwise(), MatrixBase::rowwise() + */ template template -EIGEN_STRONG_INLINE typename internal::traits::Scalar -DenseBase::redux(const Func& func) const +EIGEN_STRONG_INLINE typename internal::traits::Scalar DenseBase::redux(const Func &func) const { - eigen_assert(this->rows()>0 && this->cols()>0 && "you are using an empty matrix"); + eigen_assert(this->rows() > 0 && this->cols() > 0 && "you are using an empty matrix"); typedef typename internal::redux_evaluator ThisEvaluator; ThisEvaluator thisEval(derived()); - + return internal::redux_impl::run(thisEval, func); } /** \returns the minimum of all coefficients of \c *this. - * \warning the result is undefined if \c *this contains NaN. - */ + * \warning the result is undefined if \c *this contains NaN. + */ template -EIGEN_STRONG_INLINE typename internal::traits::Scalar -DenseBase::minCoeff() const +EIGEN_STRONG_INLINE typename internal::traits::Scalar DenseBase::minCoeff() const { - return derived().redux(Eigen::internal::scalar_min_op()); + return derived().redux(Eigen::internal::scalar_min_op()); } /** \returns the maximum of all coefficients of \c *this. - * \warning the result is undefined if \c *this contains NaN. - */ + * \warning the result is undefined if \c *this contains NaN. + */ template -EIGEN_STRONG_INLINE typename internal::traits::Scalar -DenseBase::maxCoeff() const +EIGEN_STRONG_INLINE typename internal::traits::Scalar DenseBase::maxCoeff() const { - return derived().redux(Eigen::internal::scalar_max_op()); + return derived().redux(Eigen::internal::scalar_max_op()); } /** \returns the sum of all coefficients of \c *this - * - * If \c *this is empty, then the value 0 is returned. - * - * \sa trace(), prod(), mean() - */ + * + * If \c *this is empty, then the value 0 is returned. + * + * \sa trace(), prod(), mean() + */ template -EIGEN_STRONG_INLINE typename internal::traits::Scalar -DenseBase::sum() const +EIGEN_STRONG_INLINE typename internal::traits::Scalar DenseBase::sum() const { - if(SizeAtCompileTime==0 || (SizeAtCompileTime==Dynamic && size()==0)) - return Scalar(0); - return derived().redux(Eigen::internal::scalar_sum_op()); + if (SizeAtCompileTime == 0 || (SizeAtCompileTime == Dynamic && size() == 0)) return Scalar(0); + return derived().redux(Eigen::internal::scalar_sum_op()); } /** \returns the mean of all coefficients of *this -* -* \sa trace(), prod(), sum() -*/ + * + * \sa trace(), prod(), sum() + */ template -EIGEN_STRONG_INLINE typename internal::traits::Scalar -DenseBase::mean() const +EIGEN_STRONG_INLINE typename internal::traits::Scalar DenseBase::mean() const { #ifdef __INTEL_COMPILER - #pragma warning push - #pragma warning ( disable : 2259 ) +#pragma warning push +#pragma warning(disable : 2259) #endif - return Scalar(derived().redux(Eigen::internal::scalar_sum_op())) / Scalar(this->size()); + return Scalar(derived().redux(Eigen::internal::scalar_sum_op())) / Scalar(this->size()); #ifdef __INTEL_COMPILER - #pragma warning pop +#pragma warning pop #endif } /** \returns the product of all coefficients of *this - * - * Example: \include MatrixBase_prod.cpp - * Output: \verbinclude MatrixBase_prod.out - * - * \sa sum(), mean(), trace() - */ + * + * Example: \include MatrixBase_prod.cpp + * Output: \verbinclude MatrixBase_prod.out + * + * \sa sum(), mean(), trace() + */ template -EIGEN_STRONG_INLINE typename internal::traits::Scalar -DenseBase::prod() const +EIGEN_STRONG_INLINE typename internal::traits::Scalar DenseBase::prod() const { - if(SizeAtCompileTime==0 || (SizeAtCompileTime==Dynamic && size()==0)) - return Scalar(1); + if (SizeAtCompileTime == 0 || (SizeAtCompileTime == Dynamic && size() == 0)) return Scalar(1); return derived().redux(Eigen::internal::scalar_product_op()); } /** \returns the trace of \c *this, i.e. the sum of the coefficients on the main diagonal. - * - * \c *this can be any matrix, not necessarily square. - * - * \sa diagonal(), sum() - */ + * + * \c *this can be any matrix, not necessarily square. + * + * \sa diagonal(), sum() + */ template -EIGEN_STRONG_INLINE typename internal::traits::Scalar -MatrixBase::trace() const +EIGEN_STRONG_INLINE typename internal::traits::Scalar MatrixBase::trace() const { return derived().diagonal().sum(); } -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_REDUX_H +#endif// EIGEN_REDUX_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/Ref.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/Ref.h index 9c6e3c5d..dca53037 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/Ref.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/Ref.h @@ -10,59 +10,63 @@ #ifndef EIGEN_REF_H #define EIGEN_REF_H -namespace Eigen { +namespace Eigen { namespace internal { -template -struct traits > - : public traits > -{ - typedef _PlainObjectType PlainObjectType; - typedef _StrideType StrideType; - enum { - Options = _Options, - Flags = traits >::Flags | NestByRefBit, - Alignment = traits >::Alignment - }; - - template struct match { + template + struct traits> + : public traits> + { + typedef _PlainObjectType PlainObjectType; + typedef _StrideType StrideType; enum { - HasDirectAccess = internal::has_direct_access::ret, - StorageOrderMatch = PlainObjectType::IsVectorAtCompileTime || Derived::IsVectorAtCompileTime || ((PlainObjectType::Flags&RowMajorBit)==(Derived::Flags&RowMajorBit)), - InnerStrideMatch = int(StrideType::InnerStrideAtCompileTime)==int(Dynamic) - || int(StrideType::InnerStrideAtCompileTime)==int(Derived::InnerStrideAtCompileTime) - || (int(StrideType::InnerStrideAtCompileTime)==0 && int(Derived::InnerStrideAtCompileTime)==1), - OuterStrideMatch = Derived::IsVectorAtCompileTime - || int(StrideType::OuterStrideAtCompileTime)==int(Dynamic) || int(StrideType::OuterStrideAtCompileTime)==int(Derived::OuterStrideAtCompileTime), - // NOTE, this indirection of evaluator::Alignment is needed - // to workaround a very strange bug in MSVC related to the instantiation - // of has_*ary_operator in evaluator. - // This line is surprisingly very sensitive. For instance, simply adding parenthesis - // as "DerivedAlignment = (int(evaluator::Alignment))," will make MSVC fail... - DerivedAlignment = int(evaluator::Alignment), - AlignmentMatch = (int(traits::Alignment)==int(Unaligned)) || (DerivedAlignment >= int(Alignment)), // FIXME the first condition is not very clear, it should be replaced by the required alignment - ScalarTypeMatch = internal::is_same::value, - MatchAtCompileTime = HasDirectAccess && StorageOrderMatch && InnerStrideMatch && OuterStrideMatch && AlignmentMatch && ScalarTypeMatch + Options = _Options, + Flags = traits>::Flags | NestByRefBit, + Alignment = traits>::Alignment + }; + + template struct match + { + enum { + HasDirectAccess = internal::has_direct_access::ret, + StorageOrderMatch = PlainObjectType::IsVectorAtCompileTime || Derived::IsVectorAtCompileTime + || ((PlainObjectType::Flags & RowMajorBit) == (Derived::Flags & RowMajorBit)), + InnerStrideMatch = + int(StrideType::InnerStrideAtCompileTime) == int(Dynamic) + || int(StrideType::InnerStrideAtCompileTime) == int(Derived::InnerStrideAtCompileTime) + || (int(StrideType::InnerStrideAtCompileTime) == 0 && int(Derived::InnerStrideAtCompileTime) == 1), + OuterStrideMatch = Derived::IsVectorAtCompileTime || int(StrideType::OuterStrideAtCompileTime) == int(Dynamic) + || int(StrideType::OuterStrideAtCompileTime) == int(Derived::OuterStrideAtCompileTime), + // NOTE, this indirection of evaluator::Alignment is needed + // to workaround a very strange bug in MSVC related to the instantiation + // of has_*ary_operator in evaluator. + // This line is surprisingly very sensitive. For instance, simply adding parenthesis + // as "DerivedAlignment = (int(evaluator::Alignment))," will make MSVC fail... + DerivedAlignment = int(evaluator::Alignment), + AlignmentMatch = (int(traits::Alignment) == int(Unaligned)) + || (DerivedAlignment >= int(Alignment)),// FIXME the first condition is not very clear, it + // should be replaced by the required alignment + ScalarTypeMatch = internal::is_same::value, + MatchAtCompileTime = HasDirectAccess && StorageOrderMatch && InnerStrideMatch && OuterStrideMatch + && AlignmentMatch && ScalarTypeMatch + }; + typedef typename internal::conditional::type type; }; - typedef typename internal::conditional::type type; }; - -}; -template -struct traits > : public traits {}; + template struct traits> : public traits + { + }; -} +}// namespace internal -template class RefBase - : public MapBase +template class RefBase : public MapBase { typedef typename internal::traits::PlainObjectType PlainObjectType; typedef typename internal::traits::StrideType StrideType; public: - typedef MapBase Base; EIGEN_DENSE_PUBLIC_INTERFACE(RefBase) @@ -74,210 +78,205 @@ template class RefBase EIGEN_DEVICE_FUNC inline Index outerStride() const { return StrideType::OuterStrideAtCompileTime != 0 ? m_stride.outer() - : IsVectorAtCompileTime ? this->size() - : int(Flags)&RowMajorBit ? this->cols() - : this->rows(); + : IsVectorAtCompileTime ? this->size() + : int(Flags) & RowMajorBit ? this->cols() + : this->rows(); } EIGEN_DEVICE_FUNC RefBase() - : Base(0,RowsAtCompileTime==Dynamic?0:RowsAtCompileTime,ColsAtCompileTime==Dynamic?0:ColsAtCompileTime), + : Base(0, + RowsAtCompileTime == Dynamic ? 0 : RowsAtCompileTime, + ColsAtCompileTime == Dynamic ? 0 : ColsAtCompileTime), // Stride<> does not allow default ctor for Dynamic strides, so let' initialize it with dummy values: - m_stride(StrideType::OuterStrideAtCompileTime==Dynamic?0:StrideType::OuterStrideAtCompileTime, - StrideType::InnerStrideAtCompileTime==Dynamic?0:StrideType::InnerStrideAtCompileTime) + m_stride(StrideType::OuterStrideAtCompileTime == Dynamic ? 0 : StrideType::OuterStrideAtCompileTime, + StrideType::InnerStrideAtCompileTime == Dynamic ? 0 : StrideType::InnerStrideAtCompileTime) {} - + EIGEN_INHERIT_ASSIGNMENT_OPERATORS(RefBase) protected: + typedef Stride StrideBase; - typedef Stride StrideBase; - - template - EIGEN_DEVICE_FUNC void construct(Expression& expr) + template EIGEN_DEVICE_FUNC void construct(Expression &expr) { - EIGEN_STATIC_ASSERT_SAME_MATRIX_SIZE(PlainObjectType,Expression); - - if(PlainObjectType::RowsAtCompileTime==1) - { - eigen_assert(expr.rows()==1 || expr.cols()==1); - ::new (static_cast(this)) Base(expr.data(), 1, expr.size()); - } - else if(PlainObjectType::ColsAtCompileTime==1) - { - eigen_assert(expr.rows()==1 || expr.cols()==1); - ::new (static_cast(this)) Base(expr.data(), expr.size(), 1); - } + EIGEN_STATIC_ASSERT_SAME_MATRIX_SIZE(PlainObjectType, Expression); + + if (PlainObjectType::RowsAtCompileTime == 1) { + eigen_assert(expr.rows() == 1 || expr.cols() == 1); + ::new (static_cast(this)) Base(expr.data(), 1, expr.size()); + } else if (PlainObjectType::ColsAtCompileTime == 1) { + eigen_assert(expr.rows() == 1 || expr.cols() == 1); + ::new (static_cast(this)) Base(expr.data(), expr.size(), 1); + } else + ::new (static_cast(this)) Base(expr.data(), expr.rows(), expr.cols()); + + if (Expression::IsVectorAtCompileTime && (!PlainObjectType::IsVectorAtCompileTime) + && ((Expression::Flags & RowMajorBit) != (PlainObjectType::Flags & RowMajorBit))) + ::new (&m_stride) StrideBase(expr.innerStride(), StrideType::InnerStrideAtCompileTime == 0 ? 0 : 1); else - ::new (static_cast(this)) Base(expr.data(), expr.rows(), expr.cols()); - - if(Expression::IsVectorAtCompileTime && (!PlainObjectType::IsVectorAtCompileTime) && ((Expression::Flags&RowMajorBit)!=(PlainObjectType::Flags&RowMajorBit))) - ::new (&m_stride) StrideBase(expr.innerStride(), StrideType::InnerStrideAtCompileTime==0?0:1); - else - ::new (&m_stride) StrideBase(StrideType::OuterStrideAtCompileTime==0?0:expr.outerStride(), - StrideType::InnerStrideAtCompileTime==0?0:expr.innerStride()); + ::new (&m_stride) StrideBase(StrideType::OuterStrideAtCompileTime == 0 ? 0 : expr.outerStride(), + StrideType::InnerStrideAtCompileTime == 0 ? 0 : expr.innerStride()); } StrideBase m_stride; }; /** \class Ref - * \ingroup Core_Module - * - * \brief A matrix or vector expression mapping an existing expression - * - * \tparam PlainObjectType the equivalent matrix type of the mapped data - * \tparam Options specifies the pointer alignment in bytes. It can be: \c #Aligned128, , \c #Aligned64, \c #Aligned32, \c #Aligned16, \c #Aligned8 or \c #Unaligned. - * The default is \c #Unaligned. - * \tparam StrideType optionally specifies strides. By default, Ref implies a contiguous storage along the inner dimension (inner stride==1), - * but accepts a variable outer stride (leading dimension). - * This can be overridden by specifying strides. - * The type passed here must be a specialization of the Stride template, see examples below. - * - * This class provides a way to write non-template functions taking Eigen objects as parameters while limiting the number of copies. - * A Ref<> object can represent either a const expression or a l-value: - * \code - * // in-out argument: - * void foo1(Ref x); - * - * // read-only const argument: - * void foo2(const Ref& x); - * \endcode - * - * In the in-out case, the input argument must satisfy the constraints of the actual Ref<> type, otherwise a compilation issue will be triggered. - * By default, a Ref can reference any dense vector expression of float having a contiguous memory layout. - * Likewise, a Ref can reference any column-major dense matrix expression of float whose column's elements are contiguously stored with - * the possibility to have a constant space in-between each column, i.e. the inner stride must be equal to 1, but the outer stride (or leading dimension) - * can be greater than the number of rows. - * - * In the const case, if the input expression does not match the above requirement, then it is evaluated into a temporary before being passed to the function. - * Here are some examples: - * \code - * MatrixXf A; - * VectorXf a; - * foo1(a.head()); // OK - * foo1(A.col()); // OK - * foo1(A.row()); // Compilation error because here innerstride!=1 - * foo2(A.row()); // Compilation error because A.row() is a 1xN object while foo2 is expecting a Nx1 object - * foo2(A.row().transpose()); // The row is copied into a contiguous temporary - * foo2(2*a); // The expression is evaluated into a temporary - * foo2(A.col().segment(2,4)); // No temporary - * \endcode - * - * The range of inputs that can be referenced without temporary can be enlarged using the last two template parameters. - * Here is an example accepting an innerstride!=1: - * \code - * // in-out argument: - * void foo3(Ref > x); - * foo3(A.row()); // OK - * \endcode - * The downside here is that the function foo3 might be significantly slower than foo1 because it won't be able to exploit vectorization, and will involve more - * expensive address computations even if the input is contiguously stored in memory. To overcome this issue, one might propose to overload internally calling a - * template function, e.g.: - * \code - * // in the .h: - * void foo(const Ref& A); - * void foo(const Ref >& A); - * - * // in the .cpp: - * template void foo_impl(const TypeOfA& A) { - * ... // crazy code goes here - * } - * void foo(const Ref& A) { foo_impl(A); } - * void foo(const Ref >& A) { foo_impl(A); } - * \endcode - * - * - * \sa PlainObjectBase::Map(), \ref TopicStorageOrders - */ -template class Ref - : public RefBase > + * \ingroup Core_Module + * + * \brief A matrix or vector expression mapping an existing expression + * + * \tparam PlainObjectType the equivalent matrix type of the mapped data + * \tparam Options specifies the pointer alignment in bytes. It can be: \c #Aligned128, , \c #Aligned64, \c #Aligned32, + * \c #Aligned16, \c #Aligned8 or \c #Unaligned. The default is \c #Unaligned. + * \tparam StrideType optionally specifies strides. By default, Ref implies a contiguous storage along the inner + * dimension (inner stride==1), but accepts a variable outer stride (leading dimension). This can be overridden by + * specifying strides. The type passed here must be a specialization of the Stride template, see examples below. + * + * This class provides a way to write non-template functions taking Eigen objects as parameters while limiting the + * number of copies. A Ref<> object can represent either a const expression or a l-value: + * \code + * // in-out argument: + * void foo1(Ref x); + * + * // read-only const argument: + * void foo2(const Ref& x); + * \endcode + * + * In the in-out case, the input argument must satisfy the constraints of the actual Ref<> type, otherwise a compilation + * issue will be triggered. By default, a Ref can reference any dense vector expression of float having a + * contiguous memory layout. Likewise, a Ref can reference any column-major dense matrix expression of float + * whose column's elements are contiguously stored with the possibility to have a constant space in-between each column, + * i.e. the inner stride must be equal to 1, but the outer stride (or leading dimension) can be greater than the number + * of rows. + * + * In the const case, if the input expression does not match the above requirement, then it is evaluated into a + * temporary before being passed to the function. Here are some examples: + * \code + * MatrixXf A; + * VectorXf a; + * foo1(a.head()); // OK + * foo1(A.col()); // OK + * foo1(A.row()); // Compilation error because here innerstride!=1 + * foo2(A.row()); // Compilation error because A.row() is a 1xN object while foo2 is expecting a Nx1 object + * foo2(A.row().transpose()); // The row is copied into a contiguous temporary + * foo2(2*a); // The expression is evaluated into a temporary + * foo2(A.col().segment(2,4)); // No temporary + * \endcode + * + * The range of inputs that can be referenced without temporary can be enlarged using the last two template parameters. + * Here is an example accepting an innerstride!=1: + * \code + * // in-out argument: + * void foo3(Ref > x); + * foo3(A.row()); // OK + * \endcode + * The downside here is that the function foo3 might be significantly slower than foo1 because it won't be able to + * exploit vectorization, and will involve more expensive address computations even if the input is contiguously stored + * in memory. To overcome this issue, one might propose to overload internally calling a template function, e.g.: + * \code + * // in the .h: + * void foo(const Ref& A); + * void foo(const Ref >& A); + * + * // in the .cpp: + * template void foo_impl(const TypeOfA& A) { + * ... // crazy code goes here + * } + * void foo(const Ref& A) { foo_impl(A); } + * void foo(const Ref >& A) { foo_impl(A); } + * \endcode + * + * + * \sa PlainObjectBase::Map(), \ref TopicStorageOrders + */ +template +class Ref : public RefBase> { - private: - typedef internal::traits Traits; - template - EIGEN_DEVICE_FUNC inline Ref(const PlainObjectBase& expr, - typename internal::enable_if::MatchAtCompileTime),Derived>::type* = 0); - public: - - typedef RefBase Base; - EIGEN_DENSE_PUBLIC_INTERFACE(Ref) +private: + typedef internal::traits Traits; + template + EIGEN_DEVICE_FUNC inline Ref(const PlainObjectBase &expr, + typename internal::enable_if::MatchAtCompileTime), Derived>::type * = 0); +public: + typedef RefBase Base; + EIGEN_DENSE_PUBLIC_INTERFACE(Ref) - #ifndef EIGEN_PARSED_BY_DOXYGEN - template - EIGEN_DEVICE_FUNC inline Ref(PlainObjectBase& expr, - typename internal::enable_if::MatchAtCompileTime),Derived>::type* = 0) - { - EIGEN_STATIC_ASSERT(bool(Traits::template match::MatchAtCompileTime), STORAGE_LAYOUT_DOES_NOT_MATCH); - Base::construct(expr.derived()); - } - template - EIGEN_DEVICE_FUNC inline Ref(const DenseBase& expr, - typename internal::enable_if::MatchAtCompileTime),Derived>::type* = 0) - #else - /** Implicit constructor from any dense expression */ - template - inline Ref(DenseBase& expr) - #endif - { - EIGEN_STATIC_ASSERT(bool(internal::is_lvalue::value), THIS_EXPRESSION_IS_NOT_A_LVALUE__IT_IS_READ_ONLY); - EIGEN_STATIC_ASSERT(bool(Traits::template match::MatchAtCompileTime), STORAGE_LAYOUT_DOES_NOT_MATCH); - EIGEN_STATIC_ASSERT(!Derived::IsPlainObjectBase,THIS_EXPRESSION_IS_NOT_A_LVALUE__IT_IS_READ_ONLY); - Base::construct(expr.const_cast_derived()); - } - EIGEN_INHERIT_ASSIGNMENT_OPERATORS(Ref) +#ifndef EIGEN_PARSED_BY_DOXYGEN + template + EIGEN_DEVICE_FUNC inline Ref(PlainObjectBase &expr, + typename internal::enable_if::MatchAtCompileTime), Derived>::type * = 0) + { + EIGEN_STATIC_ASSERT(bool(Traits::template match::MatchAtCompileTime), STORAGE_LAYOUT_DOES_NOT_MATCH); + Base::construct(expr.derived()); + } + template + EIGEN_DEVICE_FUNC inline Ref(const DenseBase &expr, + typename internal::enable_if::MatchAtCompileTime), Derived>::type * = 0) +#else + /** Implicit constructor from any dense expression */ + template inline Ref(DenseBase &expr) +#endif + { + EIGEN_STATIC_ASSERT(bool(internal::is_lvalue::value), THIS_EXPRESSION_IS_NOT_A_LVALUE__IT_IS_READ_ONLY); + EIGEN_STATIC_ASSERT(bool(Traits::template match::MatchAtCompileTime), STORAGE_LAYOUT_DOES_NOT_MATCH); + EIGEN_STATIC_ASSERT(!Derived::IsPlainObjectBase, THIS_EXPRESSION_IS_NOT_A_LVALUE__IT_IS_READ_ONLY); + Base::construct(expr.const_cast_derived()); + } + EIGEN_INHERIT_ASSIGNMENT_OPERATORS(Ref) }; // this is the const ref version -template class Ref - : public RefBase > +template +class Ref + : public RefBase> { - typedef internal::traits Traits; - public: - - typedef RefBase Base; - EIGEN_DENSE_PUBLIC_INTERFACE(Ref) + typedef internal::traits Traits; - template - EIGEN_DEVICE_FUNC inline Ref(const DenseBase& expr, - typename internal::enable_if::ScalarTypeMatch),Derived>::type* = 0) - { -// std::cout << match_helper::HasDirectAccess << "," << match_helper::OuterStrideMatch << "," << match_helper::InnerStrideMatch << "\n"; -// std::cout << int(StrideType::OuterStrideAtCompileTime) << " - " << int(Derived::OuterStrideAtCompileTime) << "\n"; -// std::cout << int(StrideType::InnerStrideAtCompileTime) << " - " << int(Derived::InnerStrideAtCompileTime) << "\n"; - construct(expr.derived(), typename Traits::template match::type()); - } +public: + typedef RefBase Base; + EIGEN_DENSE_PUBLIC_INTERFACE(Ref) - EIGEN_DEVICE_FUNC inline Ref(const Ref& other) : Base(other) { - // copy constructor shall not copy the m_object, to avoid unnecessary malloc and copy - } + template + EIGEN_DEVICE_FUNC inline Ref(const DenseBase &expr, + typename internal::enable_if::ScalarTypeMatch), Derived>::type * = 0) + { + // std::cout << match_helper::HasDirectAccess << "," << match_helper::OuterStrideMatch << "," + // << match_helper::InnerStrideMatch << "\n"; std::cout << int(StrideType::OuterStrideAtCompileTime) + // << " - " << int(Derived::OuterStrideAtCompileTime) << "\n"; std::cout << + // int(StrideType::InnerStrideAtCompileTime) << " - " << int(Derived::InnerStrideAtCompileTime) << "\n"; + construct(expr.derived(), typename Traits::template match::type()); + } - template - EIGEN_DEVICE_FUNC inline Ref(const RefBase& other) { - construct(other.derived(), typename Traits::template match::type()); - } + EIGEN_DEVICE_FUNC inline Ref(const Ref &other) : Base(other) + { + // copy constructor shall not copy the m_object, to avoid unnecessary malloc and copy + } - protected: + template EIGEN_DEVICE_FUNC inline Ref(const RefBase &other) + { + construct(other.derived(), typename Traits::template match::type()); + } - template - EIGEN_DEVICE_FUNC void construct(const Expression& expr,internal::true_type) - { - Base::construct(expr); - } +protected: + template EIGEN_DEVICE_FUNC void construct(const Expression &expr, internal::true_type) + { + Base::construct(expr); + } - template - EIGEN_DEVICE_FUNC void construct(const Expression& expr, internal::false_type) - { - internal::call_assignment_no_alias(m_object,expr,internal::assign_op()); - Base::construct(m_object); - } + template EIGEN_DEVICE_FUNC void construct(const Expression &expr, internal::false_type) + { + internal::call_assignment_no_alias(m_object, expr, internal::assign_op()); + Base::construct(m_object); + } - protected: - TPlainObjectType m_object; +protected: + TPlainObjectType m_object; }; -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_REF_H +#endif// EIGEN_REF_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/Replicate.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/Replicate.h index 9960ef88..3989a0b6 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/Replicate.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/Replicate.h @@ -10,133 +10,129 @@ #ifndef EIGEN_REPLICATE_H #define EIGEN_REPLICATE_H -namespace Eigen { +namespace Eigen { namespace internal { -template -struct traits > - : traits -{ - typedef typename MatrixType::Scalar Scalar; - typedef typename traits::StorageKind StorageKind; - typedef typename traits::XprKind XprKind; - typedef typename ref_selector::type MatrixTypeNested; - typedef typename remove_reference::type _MatrixTypeNested; - enum { - RowsAtCompileTime = RowFactor==Dynamic || int(MatrixType::RowsAtCompileTime)==Dynamic - ? Dynamic - : RowFactor * MatrixType::RowsAtCompileTime, - ColsAtCompileTime = ColFactor==Dynamic || int(MatrixType::ColsAtCompileTime)==Dynamic - ? Dynamic - : ColFactor * MatrixType::ColsAtCompileTime, - //FIXME we don't propagate the max sizes !!! - MaxRowsAtCompileTime = RowsAtCompileTime, - MaxColsAtCompileTime = ColsAtCompileTime, - IsRowMajor = MaxRowsAtCompileTime==1 && MaxColsAtCompileTime!=1 ? 1 - : MaxColsAtCompileTime==1 && MaxRowsAtCompileTime!=1 ? 0 - : (MatrixType::Flags & RowMajorBit) ? 1 : 0, - - // FIXME enable DirectAccess with negative strides? - Flags = IsRowMajor ? RowMajorBit : 0 + template + struct traits> : traits + { + typedef typename MatrixType::Scalar Scalar; + typedef typename traits::StorageKind StorageKind; + typedef typename traits::XprKind XprKind; + typedef typename ref_selector::type MatrixTypeNested; + typedef typename remove_reference::type _MatrixTypeNested; + enum { + RowsAtCompileTime = RowFactor == Dynamic || int(MatrixType::RowsAtCompileTime) == Dynamic + ? Dynamic + : RowFactor * MatrixType::RowsAtCompileTime, + ColsAtCompileTime = ColFactor == Dynamic || int(MatrixType::ColsAtCompileTime) == Dynamic + ? Dynamic + : ColFactor * MatrixType::ColsAtCompileTime, + // FIXME we don't propagate the max sizes !!! + MaxRowsAtCompileTime = RowsAtCompileTime, + MaxColsAtCompileTime = ColsAtCompileTime, + IsRowMajor = MaxRowsAtCompileTime == 1 && MaxColsAtCompileTime != 1 ? 1 + : MaxColsAtCompileTime == 1 && MaxRowsAtCompileTime != 1 ? 0 + : (MatrixType::Flags & RowMajorBit) ? 1 + : 0, + + // FIXME enable DirectAccess with negative strides? + Flags = IsRowMajor ? RowMajorBit : 0 + }; }; -}; -} +}// namespace internal /** - * \class Replicate - * \ingroup Core_Module - * - * \brief Expression of the multiple replication of a matrix or vector - * - * \tparam MatrixType the type of the object we are replicating - * \tparam RowFactor number of repetitions at compile time along the vertical direction, can be Dynamic. - * \tparam ColFactor number of repetitions at compile time along the horizontal direction, can be Dynamic. - * - * This class represents an expression of the multiple replication of a matrix or vector. - * It is the return type of DenseBase::replicate() and most of the time - * this is the only way it is used. - * - * \sa DenseBase::replicate() - */ -template class Replicate - : public internal::dense_xpr_base< Replicate >::type + * \class Replicate + * \ingroup Core_Module + * + * \brief Expression of the multiple replication of a matrix or vector + * + * \tparam MatrixType the type of the object we are replicating + * \tparam RowFactor number of repetitions at compile time along the vertical direction, can be Dynamic. + * \tparam ColFactor number of repetitions at compile time along the horizontal direction, can be Dynamic. + * + * This class represents an expression of the multiple replication of a matrix or vector. + * It is the return type of DenseBase::replicate() and most of the time + * this is the only way it is used. + * + * \sa DenseBase::replicate() + */ +template +class Replicate : public internal::dense_xpr_base>::type { - typedef typename internal::traits::MatrixTypeNested MatrixTypeNested; - typedef typename internal::traits::_MatrixTypeNested _MatrixTypeNested; - public: + typedef typename internal::traits::MatrixTypeNested MatrixTypeNested; + typedef typename internal::traits::_MatrixTypeNested _MatrixTypeNested; - typedef typename internal::dense_xpr_base::type Base; - EIGEN_DENSE_PUBLIC_INTERFACE(Replicate) - typedef typename internal::remove_all::type NestedExpression; +public: + typedef typename internal::dense_xpr_base::type Base; + EIGEN_DENSE_PUBLIC_INTERFACE(Replicate) + typedef typename internal::remove_all::type NestedExpression; - template - EIGEN_DEVICE_FUNC - inline explicit Replicate(const OriginalMatrixType& matrix) - : m_matrix(matrix), m_rowFactor(RowFactor), m_colFactor(ColFactor) - { - EIGEN_STATIC_ASSERT((internal::is_same::type,OriginalMatrixType>::value), - THE_MATRIX_OR_EXPRESSION_THAT_YOU_PASSED_DOES_NOT_HAVE_THE_EXPECTED_TYPE) - eigen_assert(RowFactor!=Dynamic && ColFactor!=Dynamic); - } + template + EIGEN_DEVICE_FUNC inline explicit Replicate(const OriginalMatrixType &matrix) + : m_matrix(matrix), m_rowFactor(RowFactor), m_colFactor(ColFactor) + { + EIGEN_STATIC_ASSERT( + (internal::is_same::type, OriginalMatrixType>::value), + THE_MATRIX_OR_EXPRESSION_THAT_YOU_PASSED_DOES_NOT_HAVE_THE_EXPECTED_TYPE) + eigen_assert(RowFactor != Dynamic && ColFactor != Dynamic); + } - template - EIGEN_DEVICE_FUNC - inline Replicate(const OriginalMatrixType& matrix, Index rowFactor, Index colFactor) - : m_matrix(matrix), m_rowFactor(rowFactor), m_colFactor(colFactor) - { - EIGEN_STATIC_ASSERT((internal::is_same::type,OriginalMatrixType>::value), - THE_MATRIX_OR_EXPRESSION_THAT_YOU_PASSED_DOES_NOT_HAVE_THE_EXPECTED_TYPE) - } + template + EIGEN_DEVICE_FUNC inline Replicate(const OriginalMatrixType &matrix, Index rowFactor, Index colFactor) + : m_matrix(matrix), m_rowFactor(rowFactor), + m_colFactor(colFactor){ EIGEN_STATIC_ASSERT( + (internal::is_same::type, OriginalMatrixType>::value), + THE_MATRIX_OR_EXPRESSION_THAT_YOU_PASSED_DOES_NOT_HAVE_THE_EXPECTED_TYPE) } - EIGEN_DEVICE_FUNC - inline Index rows() const { return m_matrix.rows() * m_rowFactor.value(); } - EIGEN_DEVICE_FUNC - inline Index cols() const { return m_matrix.cols() * m_colFactor.value(); } + EIGEN_DEVICE_FUNC inline Index rows() const + { + return m_matrix.rows() * m_rowFactor.value(); + } + EIGEN_DEVICE_FUNC + inline Index cols() const { return m_matrix.cols() * m_colFactor.value(); } - EIGEN_DEVICE_FUNC - const _MatrixTypeNested& nestedExpression() const - { - return m_matrix; - } + EIGEN_DEVICE_FUNC + const _MatrixTypeNested &nestedExpression() const { return m_matrix; } - protected: - MatrixTypeNested m_matrix; - const internal::variable_if_dynamic m_rowFactor; - const internal::variable_if_dynamic m_colFactor; +protected: + MatrixTypeNested m_matrix; + const internal::variable_if_dynamic m_rowFactor; + const internal::variable_if_dynamic m_colFactor; }; /** - * \return an expression of the replication of \c *this - * - * Example: \include MatrixBase_replicate.cpp - * Output: \verbinclude MatrixBase_replicate.out - * - * \sa VectorwiseOp::replicate(), DenseBase::replicate(Index,Index), class Replicate - */ + * \return an expression of the replication of \c *this + * + * Example: \include MatrixBase_replicate.cpp + * Output: \verbinclude MatrixBase_replicate.out + * + * \sa VectorwiseOp::replicate(), DenseBase::replicate(Index,Index), class Replicate + */ template template -const Replicate -DenseBase::replicate() const +const Replicate DenseBase::replicate() const { - return Replicate(derived()); + return Replicate(derived()); } /** - * \return an expression of the replication of each column (or row) of \c *this - * - * Example: \include DirectionWise_replicate_int.cpp - * Output: \verbinclude DirectionWise_replicate_int.out - * - * \sa VectorwiseOp::replicate(), DenseBase::replicate(), class Replicate - */ + * \return an expression of the replication of each column (or row) of \c *this + * + * Example: \include DirectionWise_replicate_int.cpp + * Output: \verbinclude DirectionWise_replicate_int.out + * + * \sa VectorwiseOp::replicate(), DenseBase::replicate(), class Replicate + */ template -const typename VectorwiseOp::ReplicateReturnType -VectorwiseOp::replicate(Index factor) const +const typename VectorwiseOp::ReplicateReturnType + VectorwiseOp::replicate(Index factor) const { - return typename VectorwiseOp::ReplicateReturnType - (_expression(),Direction==Vertical?factor:1,Direction==Horizontal?factor:1); + return typename VectorwiseOp::ReplicateReturnType( + _expression(), Direction == Vertical ? factor : 1, Direction == Horizontal ? factor : 1); } -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_REPLICATE_H +#endif// EIGEN_REPLICATE_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/ReturnByValue.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/ReturnByValue.h index c44b7673..8f59f7e4 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/ReturnByValue.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/ReturnByValue.h @@ -15,71 +15,71 @@ namespace Eigen { namespace internal { -template -struct traits > - : public traits::ReturnType> -{ - enum { - // We're disabling the DirectAccess because e.g. the constructor of - // the Block-with-DirectAccess expression requires to have a coeffRef method. - // Also, we don't want to have to implement the stride stuff. - Flags = (traits::ReturnType>::Flags - | EvalBeforeNestingBit) & ~DirectAccessBit + template struct traits> : public traits::ReturnType> + { + enum { + // We're disabling the DirectAccess because e.g. the constructor of + // the Block-with-DirectAccess expression requires to have a coeffRef method. + // Also, we don't want to have to implement the stride stuff. + Flags = (traits::ReturnType>::Flags | EvalBeforeNestingBit) & ~DirectAccessBit + }; }; -}; -/* The ReturnByValue object doesn't even have a coeff() method. - * So the only way that nesting it in an expression can work, is by evaluating it into a plain matrix. - * So internal::nested always gives the plain return matrix type. - * - * FIXME: I don't understand why we need this specialization: isn't this taken care of by the EvalBeforeNestingBit ?? - * Answer: EvalBeforeNestingBit should be deprecated since we have the evaluators - */ -template -struct nested_eval, n, PlainObject> -{ - typedef typename traits::ReturnType type; -}; + /* The ReturnByValue object doesn't even have a coeff() method. + * So the only way that nesting it in an expression can work, is by evaluating it into a plain matrix. + * So internal::nested always gives the plain return matrix type. + * + * FIXME: I don't understand why we need this specialization: isn't this taken care of by the EvalBeforeNestingBit ?? + * Answer: EvalBeforeNestingBit should be deprecated since we have the evaluators + */ + template struct nested_eval, n, PlainObject> + { + typedef typename traits::ReturnType type; + }; -} // end namespace internal +}// end namespace internal /** \class ReturnByValue - * \ingroup Core_Module - * - */ -template class ReturnByValue - : public internal::dense_xpr_base< ReturnByValue >::type, internal::no_assignment_operator + * \ingroup Core_Module + * + */ +template +class ReturnByValue + : public internal::dense_xpr_base>::type + , internal::no_assignment_operator { - public: - typedef typename internal::traits::ReturnType ReturnType; +public: + typedef typename internal::traits::ReturnType ReturnType; - typedef typename internal::dense_xpr_base::type Base; - EIGEN_DENSE_PUBLIC_INTERFACE(ReturnByValue) + typedef typename internal::dense_xpr_base::type Base; + EIGEN_DENSE_PUBLIC_INTERFACE(ReturnByValue) - template - EIGEN_DEVICE_FUNC - inline void evalTo(Dest& dst) const - { static_cast(this)->evalTo(dst); } - EIGEN_DEVICE_FUNC inline Index rows() const { return static_cast(this)->rows(); } - EIGEN_DEVICE_FUNC inline Index cols() const { return static_cast(this)->cols(); } + template EIGEN_DEVICE_FUNC inline void evalTo(Dest &dst) const + { + static_cast(this)->evalTo(dst); + } + EIGEN_DEVICE_FUNC inline Index rows() const { return static_cast(this)->rows(); } + EIGEN_DEVICE_FUNC inline Index cols() const { return static_cast(this)->cols(); } #ifndef EIGEN_PARSED_BY_DOXYGEN -#define Unusable YOU_ARE_TRYING_TO_ACCESS_A_SINGLE_COEFFICIENT_IN_A_SPECIAL_EXPRESSION_WHERE_THAT_IS_NOT_ALLOWED_BECAUSE_THAT_WOULD_BE_INEFFICIENT - class Unusable{ - Unusable(const Unusable&) {} - Unusable& operator=(const Unusable&) {return *this;} - }; - const Unusable& coeff(Index) const { return *reinterpret_cast(this); } - const Unusable& coeff(Index,Index) const { return *reinterpret_cast(this); } - Unusable& coeffRef(Index) { return *reinterpret_cast(this); } - Unusable& coeffRef(Index,Index) { return *reinterpret_cast(this); } +#define Unusable \ + YOU_ARE_TRYING_TO_ACCESS_A_SINGLE_COEFFICIENT_IN_A_SPECIAL_EXPRESSION_WHERE_THAT_IS_NOT_ALLOWED_BECAUSE_THAT_WOULD_BE_INEFFICIENT + class Unusable + { + Unusable(const Unusable &) {} + Unusable &operator=(const Unusable &) { return *this; } + }; + const Unusable &coeff(Index) const { return *reinterpret_cast(this); } + const Unusable &coeff(Index, Index) const { return *reinterpret_cast(this); } + Unusable &coeffRef(Index) { return *reinterpret_cast(this); } + Unusable &coeffRef(Index, Index) { return *reinterpret_cast(this); } #undef Unusable #endif }; template template -Derived& DenseBase::operator=(const ReturnByValue& other) +Derived &DenseBase::operator=(const ReturnByValue &other) { other.evalTo(derived()); return derived(); @@ -87,31 +87,29 @@ Derived& DenseBase::operator=(const ReturnByValue& other) namespace internal { -// Expression is evaluated in a temporary; default implementation of Assignment is bypassed so that -// when a ReturnByValue expression is assigned, the evaluator is not constructed. -// TODO: Finalize port to new regime; ReturnByValue should not exist in the expression world - -template -struct evaluator > - : public evaluator::ReturnType> -{ - typedef ReturnByValue XprType; - typedef typename internal::traits::ReturnType PlainObject; - typedef evaluator Base; - - EIGEN_DEVICE_FUNC explicit evaluator(const XprType& xpr) - : m_result(xpr.rows(), xpr.cols()) - { - ::new (static_cast(this)) Base(m_result); - xpr.evalTo(m_result); - } + // Expression is evaluated in a temporary; default implementation of Assignment is bypassed so that + // when a ReturnByValue expression is assigned, the evaluator is not constructed. + // TODO: Finalize port to new regime; ReturnByValue should not exist in the expression world -protected: - PlainObject m_result; -}; + template + struct evaluator> : public evaluator::ReturnType> + { + typedef ReturnByValue XprType; + typedef typename internal::traits::ReturnType PlainObject; + typedef evaluator Base; + + EIGEN_DEVICE_FUNC explicit evaluator(const XprType &xpr) : m_result(xpr.rows(), xpr.cols()) + { + ::new (static_cast(this)) Base(m_result); + xpr.evalTo(m_result); + } + + protected: + PlainObject m_result; + }; -} // end namespace internal +}// end namespace internal -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_RETURNBYVALUE_H +#endif// EIGEN_RETURNBYVALUE_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/Reverse.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/Reverse.h index 0640cda2..2d02f934 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/Reverse.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/Reverse.h @@ -12,200 +12,178 @@ #ifndef EIGEN_REVERSE_H #define EIGEN_REVERSE_H -namespace Eigen { +namespace Eigen { namespace internal { -template -struct traits > - : traits -{ - typedef typename MatrixType::Scalar Scalar; - typedef typename traits::StorageKind StorageKind; - typedef typename traits::XprKind XprKind; - typedef typename ref_selector::type MatrixTypeNested; - typedef typename remove_reference::type _MatrixTypeNested; - enum { - RowsAtCompileTime = MatrixType::RowsAtCompileTime, - ColsAtCompileTime = MatrixType::ColsAtCompileTime, - MaxRowsAtCompileTime = MatrixType::MaxRowsAtCompileTime, - MaxColsAtCompileTime = MatrixType::MaxColsAtCompileTime, - Flags = _MatrixTypeNested::Flags & (RowMajorBit | LvalueBit) + template struct traits> : traits + { + typedef typename MatrixType::Scalar Scalar; + typedef typename traits::StorageKind StorageKind; + typedef typename traits::XprKind XprKind; + typedef typename ref_selector::type MatrixTypeNested; + typedef typename remove_reference::type _MatrixTypeNested; + enum { + RowsAtCompileTime = MatrixType::RowsAtCompileTime, + ColsAtCompileTime = MatrixType::ColsAtCompileTime, + MaxRowsAtCompileTime = MatrixType::MaxRowsAtCompileTime, + MaxColsAtCompileTime = MatrixType::MaxColsAtCompileTime, + Flags = _MatrixTypeNested::Flags & (RowMajorBit | LvalueBit) + }; }; -}; -template struct reverse_packet_cond -{ - static inline PacketType run(const PacketType& x) { return preverse(x); } -}; + template struct reverse_packet_cond + { + static inline PacketType run(const PacketType &x) { return preverse(x); } + }; -template struct reverse_packet_cond -{ - static inline PacketType run(const PacketType& x) { return x; } -}; + template struct reverse_packet_cond + { + static inline PacketType run(const PacketType &x) { return x; } + }; -} // end namespace internal +}// end namespace internal /** \class Reverse - * \ingroup Core_Module - * - * \brief Expression of the reverse of a vector or matrix - * - * \tparam MatrixType the type of the object of which we are taking the reverse - * \tparam Direction defines the direction of the reverse operation, can be Vertical, Horizontal, or BothDirections - * - * This class represents an expression of the reverse of a vector. - * It is the return type of MatrixBase::reverse() and VectorwiseOp::reverse() - * and most of the time this is the only way it is used. - * - * \sa MatrixBase::reverse(), VectorwiseOp::reverse() - */ -template class Reverse - : public internal::dense_xpr_base< Reverse >::type + * \ingroup Core_Module + * + * \brief Expression of the reverse of a vector or matrix + * + * \tparam MatrixType the type of the object of which we are taking the reverse + * \tparam Direction defines the direction of the reverse operation, can be Vertical, Horizontal, or BothDirections + * + * This class represents an expression of the reverse of a vector. + * It is the return type of MatrixBase::reverse() and VectorwiseOp::reverse() + * and most of the time this is the only way it is used. + * + * \sa MatrixBase::reverse(), VectorwiseOp::reverse() + */ +template +class Reverse : public internal::dense_xpr_base>::type { - public: - - typedef typename internal::dense_xpr_base::type Base; - EIGEN_DENSE_PUBLIC_INTERFACE(Reverse) - typedef typename internal::remove_all::type NestedExpression; - using Base::IsRowMajor; +public: + typedef typename internal::dense_xpr_base::type Base; + EIGEN_DENSE_PUBLIC_INTERFACE(Reverse) + typedef typename internal::remove_all::type NestedExpression; + using Base::IsRowMajor; - protected: - enum { - PacketSize = internal::packet_traits::size, - IsColMajor = !IsRowMajor, - ReverseRow = (Direction == Vertical) || (Direction == BothDirections), - ReverseCol = (Direction == Horizontal) || (Direction == BothDirections), - OffsetRow = ReverseRow && IsColMajor ? PacketSize : 1, - OffsetCol = ReverseCol && IsRowMajor ? PacketSize : 1, - ReversePacket = (Direction == BothDirections) - || ((Direction == Vertical) && IsColMajor) +protected: + enum { + PacketSize = internal::packet_traits::size, + IsColMajor = !IsRowMajor, + ReverseRow = (Direction == Vertical) || (Direction == BothDirections), + ReverseCol = (Direction == Horizontal) || (Direction == BothDirections), + OffsetRow = ReverseRow && IsColMajor ? PacketSize : 1, + OffsetCol = ReverseCol && IsRowMajor ? PacketSize : 1, + ReversePacket = (Direction == BothDirections) || ((Direction == Vertical) && IsColMajor) || ((Direction == Horizontal) && IsRowMajor) - }; - typedef internal::reverse_packet_cond reverse_packet; - public: + }; + typedef internal::reverse_packet_cond reverse_packet; - EIGEN_DEVICE_FUNC explicit inline Reverse(const MatrixType& matrix) : m_matrix(matrix) { } +public: + EIGEN_DEVICE_FUNC explicit inline Reverse(const MatrixType &matrix) : m_matrix(matrix) {} - EIGEN_INHERIT_ASSIGNMENT_OPERATORS(Reverse) + EIGEN_INHERIT_ASSIGNMENT_OPERATORS(Reverse) - EIGEN_DEVICE_FUNC inline Index rows() const { return m_matrix.rows(); } - EIGEN_DEVICE_FUNC inline Index cols() const { return m_matrix.cols(); } + EIGEN_DEVICE_FUNC inline Index rows() const { return m_matrix.rows(); } + EIGEN_DEVICE_FUNC inline Index cols() const { return m_matrix.cols(); } - EIGEN_DEVICE_FUNC inline Index innerStride() const - { - return -m_matrix.innerStride(); - } + EIGEN_DEVICE_FUNC inline Index innerStride() const { return -m_matrix.innerStride(); } - EIGEN_DEVICE_FUNC const typename internal::remove_all::type& - nestedExpression() const - { - return m_matrix; - } + EIGEN_DEVICE_FUNC const typename internal::remove_all::type &nestedExpression() const + { + return m_matrix; + } - protected: - typename MatrixType::Nested m_matrix; +protected: + typename MatrixType::Nested m_matrix; }; /** \returns an expression of the reverse of *this. - * - * Example: \include MatrixBase_reverse.cpp - * Output: \verbinclude MatrixBase_reverse.out - * - */ -template -inline typename DenseBase::ReverseReturnType -DenseBase::reverse() + * + * Example: \include MatrixBase_reverse.cpp + * Output: \verbinclude MatrixBase_reverse.out + * + */ +template inline typename DenseBase::ReverseReturnType DenseBase::reverse() { return ReverseReturnType(derived()); } -//reverse const overload moved DenseBase.h due to a CUDA compiler bug +// reverse const overload moved DenseBase.h due to a CUDA compiler bug /** This is the "in place" version of reverse: it reverses \c *this. - * - * In most cases it is probably better to simply use the reversed expression - * of a matrix. However, when reversing the matrix data itself is really needed, - * then this "in-place" version is probably the right choice because it provides - * the following additional benefits: - * - less error prone: doing the same operation with .reverse() requires special care: - * \code m = m.reverse().eval(); \endcode - * - this API enables reverse operations without the need for a temporary - * - it allows future optimizations (cache friendliness, etc.) - * - * \sa VectorwiseOp::reverseInPlace(), reverse() */ -template -inline void DenseBase::reverseInPlace() + * + * In most cases it is probably better to simply use the reversed expression + * of a matrix. However, when reversing the matrix data itself is really needed, + * then this "in-place" version is probably the right choice because it provides + * the following additional benefits: + * - less error prone: doing the same operation with .reverse() requires special care: + * \code m = m.reverse().eval(); \endcode + * - this API enables reverse operations without the need for a temporary + * - it allows future optimizations (cache friendliness, etc.) + * + * \sa VectorwiseOp::reverseInPlace(), reverse() */ +template inline void DenseBase::reverseInPlace() { - if(cols()>rows()) - { - Index half = cols()/2; + if (cols() > rows()) { + Index half = cols() / 2; leftCols(half).swap(rightCols(half).reverse()); - if((cols()%2)==1) - { - Index half2 = rows()/2; + if ((cols() % 2) == 1) { + Index half2 = rows() / 2; col(half).head(half2).swap(col(half).tail(half2).reverse()); } - } - else - { - Index half = rows()/2; + } else { + Index half = rows() / 2; topRows(half).swap(bottomRows(half).reverse()); - if((rows()%2)==1) - { - Index half2 = cols()/2; + if ((rows() % 2) == 1) { + Index half2 = cols() / 2; row(half).head(half2).swap(row(half).tail(half2).reverse()); } } } namespace internal { - -template -struct vectorwise_reverse_inplace_impl; -template<> -struct vectorwise_reverse_inplace_impl -{ - template - static void run(ExpressionType &xpr) + template struct vectorwise_reverse_inplace_impl; + + template<> struct vectorwise_reverse_inplace_impl { - Index half = xpr.rows()/2; - xpr.topRows(half).swap(xpr.bottomRows(half).colwise().reverse()); - } -}; + template static void run(ExpressionType &xpr) + { + Index half = xpr.rows() / 2; + xpr.topRows(half).swap(xpr.bottomRows(half).colwise().reverse()); + } + }; -template<> -struct vectorwise_reverse_inplace_impl -{ - template - static void run(ExpressionType &xpr) + template<> struct vectorwise_reverse_inplace_impl { - Index half = xpr.cols()/2; - xpr.leftCols(half).swap(xpr.rightCols(half).rowwise().reverse()); - } -}; + template static void run(ExpressionType &xpr) + { + Index half = xpr.cols() / 2; + xpr.leftCols(half).swap(xpr.rightCols(half).rowwise().reverse()); + } + }; -} // end namespace internal +}// end namespace internal /** This is the "in place" version of VectorwiseOp::reverse: it reverses each column or row of \c *this. - * - * In most cases it is probably better to simply use the reversed expression - * of a matrix. However, when reversing the matrix data itself is really needed, - * then this "in-place" version is probably the right choice because it provides - * the following additional benefits: - * - less error prone: doing the same operation with .reverse() requires special care: - * \code m = m.reverse().eval(); \endcode - * - this API enables reverse operations without the need for a temporary - * - * \sa DenseBase::reverseInPlace(), reverse() */ -template -void VectorwiseOp::reverseInPlace() + * + * In most cases it is probably better to simply use the reversed expression + * of a matrix. However, when reversing the matrix data itself is really needed, + * then this "in-place" version is probably the right choice because it provides + * the following additional benefits: + * - less error prone: doing the same operation with .reverse() requires special care: + * \code m = m.reverse().eval(); \endcode + * - this API enables reverse operations without the need for a temporary + * + * \sa DenseBase::reverseInPlace(), reverse() */ +template void VectorwiseOp::reverseInPlace() { internal::vectorwise_reverse_inplace_impl::run(_expression().const_cast_derived()); } -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_REVERSE_H +#endif// EIGEN_REVERSE_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/Select.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/Select.h index 79eec1b5..daf8e967 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/Select.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/Select.h @@ -10,153 +10,139 @@ #ifndef EIGEN_SELECT_H #define EIGEN_SELECT_H -namespace Eigen { +namespace Eigen { /** \class Select - * \ingroup Core_Module - * - * \brief Expression of a coefficient wise version of the C++ ternary operator ?: - * - * \param ConditionMatrixType the type of the \em condition expression which must be a boolean matrix - * \param ThenMatrixType the type of the \em then expression - * \param ElseMatrixType the type of the \em else expression - * - * This class represents an expression of a coefficient wise version of the C++ ternary operator ?:. - * It is the return type of DenseBase::select() and most of the time this is the only way it is used. - * - * \sa DenseBase::select(const DenseBase&, const DenseBase&) const - */ + * \ingroup Core_Module + * + * \brief Expression of a coefficient wise version of the C++ ternary operator ?: + * + * \param ConditionMatrixType the type of the \em condition expression which must be a boolean matrix + * \param ThenMatrixType the type of the \em then expression + * \param ElseMatrixType the type of the \em else expression + * + * This class represents an expression of a coefficient wise version of the C++ ternary operator ?:. + * It is the return type of DenseBase::select() and most of the time this is the only way it is used. + * + * \sa DenseBase::select(const DenseBase&, const DenseBase&) const + */ namespace internal { -template -struct traits > - : traits -{ - typedef typename traits::Scalar Scalar; - typedef Dense StorageKind; - typedef typename traits::XprKind XprKind; - typedef typename ConditionMatrixType::Nested ConditionMatrixNested; - typedef typename ThenMatrixType::Nested ThenMatrixNested; - typedef typename ElseMatrixType::Nested ElseMatrixNested; - enum { - RowsAtCompileTime = ConditionMatrixType::RowsAtCompileTime, - ColsAtCompileTime = ConditionMatrixType::ColsAtCompileTime, - MaxRowsAtCompileTime = ConditionMatrixType::MaxRowsAtCompileTime, - MaxColsAtCompileTime = ConditionMatrixType::MaxColsAtCompileTime, - Flags = (unsigned int)ThenMatrixType::Flags & ElseMatrixType::Flags & RowMajorBit + template + struct traits> : traits + { + typedef typename traits::Scalar Scalar; + typedef Dense StorageKind; + typedef typename traits::XprKind XprKind; + typedef typename ConditionMatrixType::Nested ConditionMatrixNested; + typedef typename ThenMatrixType::Nested ThenMatrixNested; + typedef typename ElseMatrixType::Nested ElseMatrixNested; + enum { + RowsAtCompileTime = ConditionMatrixType::RowsAtCompileTime, + ColsAtCompileTime = ConditionMatrixType::ColsAtCompileTime, + MaxRowsAtCompileTime = ConditionMatrixType::MaxRowsAtCompileTime, + MaxColsAtCompileTime = ConditionMatrixType::MaxColsAtCompileTime, + Flags = (unsigned int)ThenMatrixType::Flags & ElseMatrixType::Flags & RowMajorBit + }; }; -}; -} +}// namespace internal template -class Select : public internal::dense_xpr_base< Select >::type, - internal::no_assignment_operator +class Select + : public internal::dense_xpr_base>::type + , internal::no_assignment_operator { - public: - - typedef typename internal::dense_xpr_base::type Base; + EIGEN_DENSE_PUBLIC_INTERFACE(Select) + + inline EIGEN_DEVICE_FUNC Select(const ConditionMatrixType &a_conditionMatrix, + const ThenMatrixType &a_thenMatrix, + const ElseMatrixType &a_elseMatrix) + : m_condition(a_conditionMatrix), m_then(a_thenMatrix), m_else(a_elseMatrix) + { + eigen_assert(m_condition.rows() == m_then.rows() && m_condition.rows() == m_else.rows()); + eigen_assert(m_condition.cols() == m_then.cols() && m_condition.cols() == m_else.cols()); + } + + inline EIGEN_DEVICE_FUNC Index rows() const { return m_condition.rows(); } + inline EIGEN_DEVICE_FUNC Index cols() const { return m_condition.cols(); } + + inline EIGEN_DEVICE_FUNC const Scalar coeff(Index i, Index j) const + { + if (m_condition.coeff(i, j)) + return m_then.coeff(i, j); + else + return m_else.coeff(i, j); + } + + inline EIGEN_DEVICE_FUNC const Scalar coeff(Index i) const + { + if (m_condition.coeff(i)) + return m_then.coeff(i); + else + return m_else.coeff(i); + } + + inline EIGEN_DEVICE_FUNC const ConditionMatrixType &conditionMatrix() const { return m_condition; } + + inline EIGEN_DEVICE_FUNC const ThenMatrixType &thenMatrix() const { return m_then; } + + inline EIGEN_DEVICE_FUNC const ElseMatrixType &elseMatrix() const { return m_else; } + +protected: + typename ConditionMatrixType::Nested m_condition; + typename ThenMatrixType::Nested m_then; + typename ElseMatrixType::Nested m_else; }; /** \returns a matrix where each coefficient (i,j) is equal to \a thenMatrix(i,j) - * if \c *this(i,j), and \a elseMatrix(i,j) otherwise. - * - * Example: \include MatrixBase_select.cpp - * Output: \verbinclude MatrixBase_select.out - * - * \sa class Select - */ + * if \c *this(i,j), and \a elseMatrix(i,j) otherwise. + * + * Example: \include MatrixBase_select.cpp + * Output: \verbinclude MatrixBase_select.out + * + * \sa class Select + */ template -template -inline const Select -DenseBase::select(const DenseBase& thenMatrix, - const DenseBase& elseMatrix) const +template +inline const Select + DenseBase::select(const DenseBase &thenMatrix, const DenseBase &elseMatrix) const { - return Select(derived(), thenMatrix.derived(), elseMatrix.derived()); + return Select(derived(), thenMatrix.derived(), elseMatrix.derived()); } /** Version of DenseBase::select(const DenseBase&, const DenseBase&) with - * the \em else expression being a scalar value. - * - * \sa DenseBase::select(const DenseBase&, const DenseBase&) const, class Select - */ + * the \em else expression being a scalar value. + * + * \sa DenseBase::select(const DenseBase&, const DenseBase&) const, class Select + */ template template -inline const Select -DenseBase::select(const DenseBase& thenMatrix, - const typename ThenDerived::Scalar& elseScalar) const +inline const Select DenseBase::select( + const DenseBase &thenMatrix, + const typename ThenDerived::Scalar &elseScalar) const { - return Select( - derived(), thenMatrix.derived(), ThenDerived::Constant(rows(),cols(),elseScalar)); + return Select( + derived(), thenMatrix.derived(), ThenDerived::Constant(rows(), cols(), elseScalar)); } /** Version of DenseBase::select(const DenseBase&, const DenseBase&) with - * the \em then expression being a scalar value. - * - * \sa DenseBase::select(const DenseBase&, const DenseBase&) const, class Select - */ + * the \em then expression being a scalar value. + * + * \sa DenseBase::select(const DenseBase&, const DenseBase&) const, class Select + */ template template -inline const Select -DenseBase::select(const typename ElseDerived::Scalar& thenScalar, - const DenseBase& elseMatrix) const +inline const Select DenseBase::select( + const typename ElseDerived::Scalar &thenScalar, + const DenseBase &elseMatrix) const { - return Select( - derived(), ElseDerived::Constant(rows(),cols(),thenScalar), elseMatrix.derived()); + return Select( + derived(), ElseDerived::Constant(rows(), cols(), thenScalar), elseMatrix.derived()); } -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_SELECT_H +#endif// EIGEN_SELECT_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/SelfAdjointView.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/SelfAdjointView.h index b2e51f37..7e6769a5 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/SelfAdjointView.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/SelfAdjointView.h @@ -10,247 +10,240 @@ #ifndef EIGEN_SELFADJOINTMATRIX_H #define EIGEN_SELFADJOINTMATRIX_H -namespace Eigen { +namespace Eigen { /** \class SelfAdjointView - * \ingroup Core_Module - * - * - * \brief Expression of a selfadjoint matrix from a triangular part of a dense matrix - * - * \param MatrixType the type of the dense matrix storing the coefficients - * \param TriangularPart can be either \c #Lower or \c #Upper - * - * This class is an expression of a sefladjoint matrix from a triangular part of a matrix - * with given dense storage of the coefficients. It is the return type of MatrixBase::selfadjointView() - * and most of the time this is the only way that it is used. - * - * \sa class TriangularBase, MatrixBase::selfadjointView() - */ + * \ingroup Core_Module + * + * + * \brief Expression of a selfadjoint matrix from a triangular part of a dense matrix + * + * \param MatrixType the type of the dense matrix storing the coefficients + * \param TriangularPart can be either \c #Lower or \c #Upper + * + * This class is an expression of a sefladjoint matrix from a triangular part of a matrix + * with given dense storage of the coefficients. It is the return type of MatrixBase::selfadjointView() + * and most of the time this is the only way that it is used. + * + * \sa class TriangularBase, MatrixBase::selfadjointView() + */ namespace internal { -template -struct traits > : traits -{ - typedef typename ref_selector::non_const_type MatrixTypeNested; - typedef typename remove_all::type MatrixTypeNestedCleaned; - typedef MatrixType ExpressionType; - typedef typename MatrixType::PlainObject FullMatrixType; - enum { - Mode = UpLo | SelfAdjoint, - FlagsLvalueBit = is_lvalue::value ? LvalueBit : 0, - Flags = MatrixTypeNestedCleaned::Flags & (HereditaryBits|FlagsLvalueBit) - & (~(PacketAccessBit | DirectAccessBit | LinearAccessBit)) // FIXME these flags should be preserved + template struct traits> : traits + { + typedef typename ref_selector::non_const_type MatrixTypeNested; + typedef typename remove_all::type MatrixTypeNestedCleaned; + typedef MatrixType ExpressionType; + typedef typename MatrixType::PlainObject FullMatrixType; + enum { + Mode = UpLo | SelfAdjoint, + FlagsLvalueBit = is_lvalue::value ? LvalueBit : 0, + Flags = MatrixTypeNestedCleaned::Flags & (HereditaryBits | FlagsLvalueBit) + & (~(PacketAccessBit | DirectAccessBit | LinearAccessBit))// FIXME these flags should be preserved + }; }; -}; -} +}// namespace internal -template class SelfAdjointView - : public TriangularBase > +template +class SelfAdjointView : public TriangularBase> { - public: +public: + typedef _MatrixType MatrixType; + typedef TriangularBase Base; + typedef typename internal::traits::MatrixTypeNested MatrixTypeNested; + typedef typename internal::traits::MatrixTypeNestedCleaned MatrixTypeNestedCleaned; + typedef MatrixTypeNestedCleaned NestedExpression; - typedef _MatrixType MatrixType; - typedef TriangularBase Base; - typedef typename internal::traits::MatrixTypeNested MatrixTypeNested; - typedef typename internal::traits::MatrixTypeNestedCleaned MatrixTypeNestedCleaned; - typedef MatrixTypeNestedCleaned NestedExpression; + /** \brief The type of coefficients in this matrix */ + typedef typename internal::traits::Scalar Scalar; + typedef typename MatrixType::StorageIndex StorageIndex; + typedef typename internal::remove_all::type MatrixConjugateReturnType; - /** \brief The type of coefficients in this matrix */ - typedef typename internal::traits::Scalar Scalar; - typedef typename MatrixType::StorageIndex StorageIndex; - typedef typename internal::remove_all::type MatrixConjugateReturnType; + enum { + Mode = internal::traits::Mode, + Flags = internal::traits::Flags, + TransposeMode = ((Mode & Upper) ? Lower : 0) | ((Mode & Lower) ? Upper : 0) + }; + typedef typename MatrixType::PlainObject PlainObject; - enum { - Mode = internal::traits::Mode, - Flags = internal::traits::Flags, - TransposeMode = ((Mode & Upper) ? Lower : 0) | ((Mode & Lower) ? Upper : 0) - }; - typedef typename MatrixType::PlainObject PlainObject; + EIGEN_DEVICE_FUNC + explicit inline SelfAdjointView(MatrixType &matrix) : m_matrix(matrix) + { + EIGEN_STATIC_ASSERT(UpLo == Lower || UpLo == Upper, SELFADJOINTVIEW_ACCEPTS_UPPER_AND_LOWER_MODE_ONLY); + } - EIGEN_DEVICE_FUNC - explicit inline SelfAdjointView(MatrixType& matrix) : m_matrix(matrix) - { - EIGEN_STATIC_ASSERT(UpLo==Lower || UpLo==Upper,SELFADJOINTVIEW_ACCEPTS_UPPER_AND_LOWER_MODE_ONLY); - } + EIGEN_DEVICE_FUNC + inline Index rows() const { return m_matrix.rows(); } + EIGEN_DEVICE_FUNC + inline Index cols() const { return m_matrix.cols(); } + EIGEN_DEVICE_FUNC + inline Index outerStride() const { return m_matrix.outerStride(); } + EIGEN_DEVICE_FUNC + inline Index innerStride() const { return m_matrix.innerStride(); } + + /** \sa MatrixBase::coeff() + * \warning the coordinates must fit into the referenced triangular part + */ + EIGEN_DEVICE_FUNC + inline Scalar coeff(Index row, Index col) const + { + Base::check_coordinates_internal(row, col); + return m_matrix.coeff(row, col); + } - EIGEN_DEVICE_FUNC - inline Index rows() const { return m_matrix.rows(); } - EIGEN_DEVICE_FUNC - inline Index cols() const { return m_matrix.cols(); } - EIGEN_DEVICE_FUNC - inline Index outerStride() const { return m_matrix.outerStride(); } - EIGEN_DEVICE_FUNC - inline Index innerStride() const { return m_matrix.innerStride(); } - - /** \sa MatrixBase::coeff() - * \warning the coordinates must fit into the referenced triangular part - */ - EIGEN_DEVICE_FUNC - inline Scalar coeff(Index row, Index col) const - { - Base::check_coordinates_internal(row, col); - return m_matrix.coeff(row, col); - } + /** \sa MatrixBase::coeffRef() + * \warning the coordinates must fit into the referenced triangular part + */ + EIGEN_DEVICE_FUNC + inline Scalar &coeffRef(Index row, Index col) + { + EIGEN_STATIC_ASSERT_LVALUE(SelfAdjointView); + Base::check_coordinates_internal(row, col); + return m_matrix.coeffRef(row, col); + } - /** \sa MatrixBase::coeffRef() - * \warning the coordinates must fit into the referenced triangular part - */ - EIGEN_DEVICE_FUNC - inline Scalar& coeffRef(Index row, Index col) - { - EIGEN_STATIC_ASSERT_LVALUE(SelfAdjointView); - Base::check_coordinates_internal(row, col); - return m_matrix.coeffRef(row, col); - } + /** \internal */ + EIGEN_DEVICE_FUNC + const MatrixTypeNestedCleaned &_expression() const { return m_matrix; } - /** \internal */ - EIGEN_DEVICE_FUNC - const MatrixTypeNestedCleaned& _expression() const { return m_matrix; } + EIGEN_DEVICE_FUNC + const MatrixTypeNestedCleaned &nestedExpression() const { return m_matrix; } + EIGEN_DEVICE_FUNC + MatrixTypeNestedCleaned &nestedExpression() { return m_matrix; } - EIGEN_DEVICE_FUNC - const MatrixTypeNestedCleaned& nestedExpression() const { return m_matrix; } - EIGEN_DEVICE_FUNC - MatrixTypeNestedCleaned& nestedExpression() { return m_matrix; } + /** Efficient triangular matrix times vector/matrix product */ + template + EIGEN_DEVICE_FUNC const Product operator*(const MatrixBase &rhs) const + { + return Product(*this, rhs.derived()); + } - /** Efficient triangular matrix times vector/matrix product */ - template - EIGEN_DEVICE_FUNC - const Product - operator*(const MatrixBase& rhs) const - { - return Product(*this, rhs.derived()); - } + /** Efficient vector/matrix times triangular matrix product */ + template + friend EIGEN_DEVICE_FUNC const Product operator*(const MatrixBase &lhs, + const SelfAdjointView &rhs) + { + return Product(lhs.derived(), rhs); + } - /** Efficient vector/matrix times triangular matrix product */ - template friend - EIGEN_DEVICE_FUNC - const Product - operator*(const MatrixBase& lhs, const SelfAdjointView& rhs) - { - return Product(lhs.derived(),rhs); - } - - friend EIGEN_DEVICE_FUNC - const SelfAdjointView - operator*(const Scalar& s, const SelfAdjointView& mat) - { - return (s*mat.nestedExpression()).template selfadjointView(); - } + friend EIGEN_DEVICE_FUNC const + SelfAdjointView + operator*(const Scalar &s, const SelfAdjointView &mat) + { + return (s * mat.nestedExpression()).template selfadjointView(); + } - /** Perform a symmetric rank 2 update of the selfadjoint matrix \c *this: - * \f$ this = this + \alpha u v^* + conj(\alpha) v u^* \f$ - * \returns a reference to \c *this - * - * The vectors \a u and \c v \b must be column vectors, however they can be - * a adjoint expression without any overhead. Only the meaningful triangular - * part of the matrix is updated, the rest is left unchanged. - * - * \sa rankUpdate(const MatrixBase&, Scalar) - */ - template - EIGEN_DEVICE_FUNC - SelfAdjointView& rankUpdate(const MatrixBase& u, const MatrixBase& v, const Scalar& alpha = Scalar(1)); - - /** Perform a symmetric rank K update of the selfadjoint matrix \c *this: - * \f$ this = this + \alpha ( u u^* ) \f$ where \a u is a vector or matrix. - * - * \returns a reference to \c *this - * - * Note that to perform \f$ this = this + \alpha ( u^* u ) \f$ you can simply - * call this function with u.adjoint(). - * - * \sa rankUpdate(const MatrixBase&, const MatrixBase&, Scalar) - */ - template - EIGEN_DEVICE_FUNC - SelfAdjointView& rankUpdate(const MatrixBase& u, const Scalar& alpha = Scalar(1)); - - /** \returns an expression of a triangular view extracted from the current selfadjoint view of a given triangular part - * - * The parameter \a TriMode can have the following values: \c #Upper, \c #StrictlyUpper, \c #UnitUpper, - * \c #Lower, \c #StrictlyLower, \c #UnitLower. - * - * If \c TriMode references the same triangular part than \c *this, then this method simply return a \c TriangularView of the nested expression, - * otherwise, the nested expression is first transposed, thus returning a \c TriangularView> object. - * - * \sa MatrixBase::triangularView(), class TriangularView - */ - template - EIGEN_DEVICE_FUNC - typename internal::conditional<(TriMode&(Upper|Lower))==(UpLo&(Upper|Lower)), - TriangularView, - TriangularView >::type + /** Perform a symmetric rank 2 update of the selfadjoint matrix \c *this: + * \f$ this = this + \alpha u v^* + conj(\alpha) v u^* \f$ + * \returns a reference to \c *this + * + * The vectors \a u and \c v \b must be column vectors, however they can be + * a adjoint expression without any overhead. Only the meaningful triangular + * part of the matrix is updated, the rest is left unchanged. + * + * \sa rankUpdate(const MatrixBase&, Scalar) + */ + template + EIGEN_DEVICE_FUNC SelfAdjointView & + rankUpdate(const MatrixBase &u, const MatrixBase &v, const Scalar &alpha = Scalar(1)); + + /** Perform a symmetric rank K update of the selfadjoint matrix \c *this: + * \f$ this = this + \alpha ( u u^* ) \f$ where \a u is a vector or matrix. + * + * \returns a reference to \c *this + * + * Note that to perform \f$ this = this + \alpha ( u^* u ) \f$ you can simply + * call this function with u.adjoint(). + * + * \sa rankUpdate(const MatrixBase&, const MatrixBase&, Scalar) + */ + template + EIGEN_DEVICE_FUNC SelfAdjointView &rankUpdate(const MatrixBase &u, const Scalar &alpha = Scalar(1)); + + /** \returns an expression of a triangular view extracted from the current selfadjoint view of a given triangular part + * + * The parameter \a TriMode can have the following values: \c #Upper, \c #StrictlyUpper, \c #UnitUpper, + * \c #Lower, \c #StrictlyLower, \c #UnitLower. + * + * If \c TriMode references the same triangular part than \c *this, then this method simply return a \c TriangularView + * of the nested expression, otherwise, the nested expression is first transposed, thus returning a \c + * TriangularView> object. + * + * \sa MatrixBase::triangularView(), class TriangularView + */ + template + EIGEN_DEVICE_FUNC typename internal::conditional<(TriMode & (Upper | Lower)) == (UpLo & (Upper | Lower)), + TriangularView, + TriangularView>::type triangularView() const - { - typename internal::conditional<(TriMode&(Upper|Lower))==(UpLo&(Upper|Lower)), MatrixType&, typename MatrixType::ConstTransposeReturnType>::type tmp1(m_matrix); - typename internal::conditional<(TriMode&(Upper|Lower))==(UpLo&(Upper|Lower)), MatrixType&, typename MatrixType::AdjointReturnType>::type tmp2(tmp1); - return typename internal::conditional<(TriMode&(Upper|Lower))==(UpLo&(Upper|Lower)), - TriangularView, - TriangularView >::type(tmp2); - } + { + typename internal::conditional<(TriMode & (Upper | Lower)) == (UpLo & (Upper | Lower)), + MatrixType &, + typename MatrixType::ConstTransposeReturnType>::type tmp1(m_matrix); + typename internal::conditional<(TriMode & (Upper | Lower)) == (UpLo & (Upper | Lower)), + MatrixType &, + typename MatrixType::AdjointReturnType>::type tmp2(tmp1); + return typename internal::conditional<(TriMode & (Upper | Lower)) == (UpLo & (Upper | Lower)), + TriangularView, + TriangularView>::type(tmp2); + } - typedef SelfAdjointView ConjugateReturnType; - /** \sa MatrixBase::conjugate() const */ - EIGEN_DEVICE_FUNC - inline const ConjugateReturnType conjugate() const - { return ConjugateReturnType(m_matrix.conjugate()); } - - typedef SelfAdjointView AdjointReturnType; - /** \sa MatrixBase::adjoint() const */ - EIGEN_DEVICE_FUNC - inline const AdjointReturnType adjoint() const - { return AdjointReturnType(m_matrix.adjoint()); } - - typedef SelfAdjointView TransposeReturnType; - /** \sa MatrixBase::transpose() */ - EIGEN_DEVICE_FUNC - inline TransposeReturnType transpose() - { - EIGEN_STATIC_ASSERT_LVALUE(MatrixType) - typename MatrixType::TransposeReturnType tmp(m_matrix); - return TransposeReturnType(tmp); - } + typedef SelfAdjointView ConjugateReturnType; + /** \sa MatrixBase::conjugate() const */ + EIGEN_DEVICE_FUNC + inline const ConjugateReturnType conjugate() const { return ConjugateReturnType(m_matrix.conjugate()); } - typedef SelfAdjointView ConstTransposeReturnType; - /** \sa MatrixBase::transpose() const */ - EIGEN_DEVICE_FUNC - inline const ConstTransposeReturnType transpose() const - { - return ConstTransposeReturnType(m_matrix.transpose()); - } + typedef SelfAdjointView AdjointReturnType; + /** \sa MatrixBase::adjoint() const */ + EIGEN_DEVICE_FUNC + inline const AdjointReturnType adjoint() const { return AdjointReturnType(m_matrix.adjoint()); } - /** \returns a const expression of the main diagonal of the matrix \c *this - * - * This method simply returns the diagonal of the nested expression, thus by-passing the SelfAdjointView decorator. - * - * \sa MatrixBase::diagonal(), class Diagonal */ - EIGEN_DEVICE_FUNC - typename MatrixType::ConstDiagonalReturnType diagonal() const - { - return typename MatrixType::ConstDiagonalReturnType(m_matrix); - } + typedef SelfAdjointView TransposeReturnType; + /** \sa MatrixBase::transpose() */ + EIGEN_DEVICE_FUNC + inline TransposeReturnType transpose() + { + EIGEN_STATIC_ASSERT_LVALUE(MatrixType) + typename MatrixType::TransposeReturnType tmp(m_matrix); + return TransposeReturnType(tmp); + } -/////////// Cholesky module /////////// + typedef SelfAdjointView ConstTransposeReturnType; + /** \sa MatrixBase::transpose() const */ + EIGEN_DEVICE_FUNC + inline const ConstTransposeReturnType transpose() const { return ConstTransposeReturnType(m_matrix.transpose()); } + + /** \returns a const expression of the main diagonal of the matrix \c *this + * + * This method simply returns the diagonal of the nested expression, thus by-passing the SelfAdjointView decorator. + * + * \sa MatrixBase::diagonal(), class Diagonal */ + EIGEN_DEVICE_FUNC + typename MatrixType::ConstDiagonalReturnType diagonal() const + { + return typename MatrixType::ConstDiagonalReturnType(m_matrix); + } - const LLT llt() const; - const LDLT ldlt() const; + /////////// Cholesky module /////////// -/////////// Eigenvalue module /////////// + const LLT llt() const; + const LDLT ldlt() const; - /** Real part of #Scalar */ - typedef typename NumTraits::Real RealScalar; - /** Return type of eigenvalues() */ - typedef Matrix::ColsAtCompileTime, 1> EigenvaluesReturnType; + /////////// Eigenvalue module /////////// - EIGEN_DEVICE_FUNC - EigenvaluesReturnType eigenvalues() const; - EIGEN_DEVICE_FUNC - RealScalar operatorNorm() const; + /** Real part of #Scalar */ + typedef typename NumTraits::Real RealScalar; + /** Return type of eigenvalues() */ + typedef Matrix::ColsAtCompileTime, 1> EigenvaluesReturnType; - protected: - MatrixTypeNested m_matrix; + EIGEN_DEVICE_FUNC + EigenvaluesReturnType eigenvalues() const; + EIGEN_DEVICE_FUNC + RealScalar operatorNorm() const; + +protected: + MatrixTypeNested m_matrix; }; @@ -258,95 +251,108 @@ template class SelfAdjointView // internal::selfadjoint_matrix_product_returntype > // operator*(const MatrixBase& lhs, const SelfAdjointView& rhs) // { -// return internal::matrix_selfadjoint_product_returntype >(lhs.derived(),rhs); +// return internal::matrix_selfadjoint_product_returntype +// >(lhs.derived(),rhs); // } // selfadjoint to dense matrix namespace internal { -// TODO currently a selfadjoint expression has the form SelfAdjointView<.,.> -// in the future selfadjoint-ness should be defined by the expression traits -// such that Transpose > is valid. (currently TriangularBase::transpose() is overloaded to make it work) -template -struct evaluator_traits > -{ - typedef typename storage_kind_to_evaluator_kind::Kind Kind; - typedef SelfAdjointShape Shape; -}; - -template -class triangular_dense_assignment_kernel - : public generic_dense_assignment_kernel -{ -protected: - typedef generic_dense_assignment_kernel Base; - typedef typename Base::DstXprType DstXprType; - typedef typename Base::SrcXprType SrcXprType; - using Base::m_dst; - using Base::m_src; - using Base::m_functor; -public: - - typedef typename Base::DstEvaluatorType DstEvaluatorType; - typedef typename Base::SrcEvaluatorType SrcEvaluatorType; - typedef typename Base::Scalar Scalar; - typedef typename Base::AssignmentTraits AssignmentTraits; - - - EIGEN_DEVICE_FUNC triangular_dense_assignment_kernel(DstEvaluatorType &dst, const SrcEvaluatorType &src, const Functor &func, DstXprType& dstExpr) - : Base(dst, src, func, dstExpr) - {} - - EIGEN_DEVICE_FUNC void assignCoeff(Index row, Index col) + // TODO currently a selfadjoint expression has the form SelfAdjointView<.,.> + // in the future selfadjoint-ness should be defined by the expression traits + // such that Transpose > is valid. (currently TriangularBase::transpose() is overloaded to + // make it work) + template struct evaluator_traits> { - eigen_internal_assert(row!=col); - Scalar tmp = m_src.coeff(row,col); - m_functor.assignCoeff(m_dst.coeffRef(row,col), tmp); - m_functor.assignCoeff(m_dst.coeffRef(col,row), numext::conj(tmp)); - } - - EIGEN_DEVICE_FUNC void assignDiagonalCoeff(Index id) + typedef typename storage_kind_to_evaluator_kind::Kind Kind; + typedef SelfAdjointShape Shape; + }; + + template + class triangular_dense_assignment_kernel : public generic_dense_assignment_kernel { - Base::assignCoeff(id,id); - } - - EIGEN_DEVICE_FUNC void assignOppositeCoeff(Index, Index) - { eigen_internal_assert(false && "should never be called"); } -}; + protected: + typedef generic_dense_assignment_kernel Base; + typedef typename Base::DstXprType DstXprType; + typedef typename Base::SrcXprType SrcXprType; + using Base::m_dst; + using Base::m_src; + using Base::m_functor; + + public: + typedef typename Base::DstEvaluatorType DstEvaluatorType; + typedef typename Base::SrcEvaluatorType SrcEvaluatorType; + typedef typename Base::Scalar Scalar; + typedef typename Base::AssignmentTraits AssignmentTraits; + + + EIGEN_DEVICE_FUNC triangular_dense_assignment_kernel(DstEvaluatorType &dst, + const SrcEvaluatorType &src, + const Functor &func, + DstXprType &dstExpr) + : Base(dst, src, func, dstExpr) + {} + + EIGEN_DEVICE_FUNC void assignCoeff(Index row, Index col) + { + eigen_internal_assert(row != col); + Scalar tmp = m_src.coeff(row, col); + m_functor.assignCoeff(m_dst.coeffRef(row, col), tmp); + m_functor.assignCoeff(m_dst.coeffRef(col, row), numext::conj(tmp)); + } + + EIGEN_DEVICE_FUNC void assignDiagonalCoeff(Index id) { Base::assignCoeff(id, id); } + + EIGEN_DEVICE_FUNC void assignOppositeCoeff(Index, Index) + { + eigen_internal_assert(false && "should never be called"); + } + }; -} // end namespace internal +}// end namespace internal /*************************************************************************** -* Implementation of MatrixBase methods -***************************************************************************/ + * Implementation of MatrixBase methods + ***************************************************************************/ /** This is the const version of MatrixBase::selfadjointView() */ template template typename MatrixBase::template ConstSelfAdjointViewReturnType::Type -MatrixBase::selfadjointView() const + MatrixBase::selfadjointView() const { return typename ConstSelfAdjointViewReturnType::Type(derived()); } -/** \returns an expression of a symmetric/self-adjoint view extracted from the upper or lower triangular part of the current matrix - * - * The parameter \a UpLo can be either \c #Upper or \c #Lower - * - * Example: \include MatrixBase_selfadjointView.cpp - * Output: \verbinclude MatrixBase_selfadjointView.out - * - * \sa class SelfAdjointView - */ +/** \returns an expression of a symmetric/self-adjoint view extracted from the upper or lower triangular part of the + * current matrix + * + * The parameter \a UpLo can be either \c #Upper or \c #Lower + * + * Example: \include MatrixBase_selfadjointView.cpp + * Output: \verbinclude MatrixBase_selfadjointView.out + * + * \sa class SelfAdjointView + */ template template -typename MatrixBase::template SelfAdjointViewReturnType::Type -MatrixBase::selfadjointView() +typename MatrixBase::template SelfAdjointViewReturnType::Type MatrixBase::selfadjointView() { return typename SelfAdjointViewReturnType::Type(derived()); } -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_SELFADJOINTMATRIX_H +#endif// EIGEN_SELFADJOINTMATRIX_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/SelfCwiseBinaryOp.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/SelfCwiseBinaryOp.h index 7c89c2e2..97c02015 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/SelfCwiseBinaryOp.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/SelfCwiseBinaryOp.h @@ -10,38 +10,42 @@ #ifndef EIGEN_SELFCWISEBINARYOP_H #define EIGEN_SELFCWISEBINARYOP_H -namespace Eigen { +namespace Eigen { // TODO generalize the scalar type of 'other' template -EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Derived& DenseBase::operator*=(const Scalar& other) +EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Derived &DenseBase::operator*=(const Scalar &other) { - internal::call_assignment(this->derived(), PlainObject::Constant(rows(),cols(),other), internal::mul_assign_op()); + internal::call_assignment( + this->derived(), PlainObject::Constant(rows(), cols(), other), internal::mul_assign_op()); return derived(); } template -EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Derived& ArrayBase::operator+=(const Scalar& other) +EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Derived &ArrayBase::operator+=(const Scalar &other) { - internal::call_assignment(this->derived(), PlainObject::Constant(rows(),cols(),other), internal::add_assign_op()); + internal::call_assignment( + this->derived(), PlainObject::Constant(rows(), cols(), other), internal::add_assign_op()); return derived(); } template -EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Derived& ArrayBase::operator-=(const Scalar& other) +EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Derived &ArrayBase::operator-=(const Scalar &other) { - internal::call_assignment(this->derived(), PlainObject::Constant(rows(),cols(),other), internal::sub_assign_op()); + internal::call_assignment( + this->derived(), PlainObject::Constant(rows(), cols(), other), internal::sub_assign_op()); return derived(); } template -EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Derived& DenseBase::operator/=(const Scalar& other) +EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Derived &DenseBase::operator/=(const Scalar &other) { - internal::call_assignment(this->derived(), PlainObject::Constant(rows(),cols(),other), internal::div_assign_op()); + internal::call_assignment( + this->derived(), PlainObject::Constant(rows(), cols(), other), internal::div_assign_op()); return derived(); } -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_SELFCWISEBINARYOP_H +#endif// EIGEN_SELFCWISEBINARYOP_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/Solve.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/Solve.h index a8daea51..bfe707ac 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/Solve.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/Solve.h @@ -13,176 +13,175 @@ namespace Eigen { template class SolveImpl; - + /** \class Solve - * \ingroup Core_Module - * - * \brief Pseudo expression representing a solving operation - * - * \tparam Decomposition the type of the matrix or decomposion object - * \tparam Rhstype the type of the right-hand side - * - * This class represents an expression of A.solve(B) - * and most of the time this is the only way it is used. - * - */ + * \ingroup Core_Module + * + * \brief Pseudo expression representing a solving operation + * + * \tparam Decomposition the type of the matrix or decomposion object + * \tparam Rhstype the type of the right-hand side + * + * This class represents an expression of A.solve(B) + * and most of the time this is the only way it is used. + * + */ namespace internal { -// this solve_traits class permits to determine the evaluation type with respect to storage kind (Dense vs Sparse) -template struct solve_traits; + // this solve_traits class permits to determine the evaluation type with respect to storage kind (Dense vs Sparse) + template struct solve_traits; -template -struct solve_traits -{ - typedef typename make_proper_matrix_type::type PlainObject; -}; + template struct solve_traits + { + typedef typename make_proper_matrix_type::type PlainObject; + }; -template -struct traits > - : traits::StorageKind>::PlainObject> -{ - typedef typename solve_traits::StorageKind>::PlainObject PlainObject; - typedef typename promote_index_type::type StorageIndex; - typedef traits BaseTraits; - enum { - Flags = BaseTraits::Flags & RowMajorBit, - CoeffReadCost = HugeCost + template + struct traits> + : traits< + typename solve_traits::StorageKind>::PlainObject> + { + typedef typename solve_traits::StorageKind>::PlainObject + PlainObject; + typedef typename promote_index_type::type + StorageIndex; + typedef traits BaseTraits; + enum { Flags = BaseTraits::Flags & RowMajorBit, CoeffReadCost = HugeCost }; }; -}; -} +}// namespace internal template -class Solve : public SolveImpl::StorageKind> +class Solve : public SolveImpl::StorageKind> { public: typedef typename internal::traits::PlainObject PlainObject; typedef typename internal::traits::StorageIndex StorageIndex; - - Solve(const Decomposition &dec, const RhsType &rhs) - : m_dec(dec), m_rhs(rhs) - {} - + + Solve(const Decomposition &dec, const RhsType &rhs) : m_dec(dec), m_rhs(rhs) {} + EIGEN_DEVICE_FUNC Index rows() const { return m_dec.cols(); } EIGEN_DEVICE_FUNC Index cols() const { return m_rhs.cols(); } - EIGEN_DEVICE_FUNC const Decomposition& dec() const { return m_dec; } - EIGEN_DEVICE_FUNC const RhsType& rhs() const { return m_rhs; } + EIGEN_DEVICE_FUNC const Decomposition &dec() const { return m_dec; } + EIGEN_DEVICE_FUNC const RhsType &rhs() const { return m_rhs; } protected: const Decomposition &m_dec; - const RhsType &m_rhs; + const RhsType &m_rhs; }; // Specialization of the Solve expression for dense results template -class SolveImpl - : public MatrixBase > +class SolveImpl : public MatrixBase> { - typedef Solve Derived; - + typedef Solve Derived; + public: - - typedef MatrixBase > Base; + typedef MatrixBase> Base; EIGEN_DENSE_PUBLIC_INTERFACE(Derived) private: - Scalar coeff(Index row, Index col) const; Scalar coeff(Index i) const; }; // Generic API dispatcher template -class SolveImpl : public internal::generic_xpr_base, MatrixXpr, StorageKind>::type +class SolveImpl : public internal::generic_xpr_base, MatrixXpr, StorageKind>::type { - public: - typedef typename internal::generic_xpr_base, MatrixXpr, StorageKind>::type Base; +public: + typedef typename internal::generic_xpr_base, MatrixXpr, StorageKind>::type Base; }; namespace internal { -// Evaluator of Solve -> eval into a temporary -template -struct evaluator > - : public evaluator::PlainObject> -{ - typedef Solve SolveType; - typedef typename SolveType::PlainObject PlainObject; - typedef evaluator Base; - - enum { Flags = Base::Flags | EvalBeforeNestingBit }; - - EIGEN_DEVICE_FUNC explicit evaluator(const SolveType& solve) - : m_result(solve.rows(), solve.cols()) + // Evaluator of Solve -> eval into a temporary + template + struct evaluator> + : public evaluator::PlainObject> { - ::new (static_cast(this)) Base(m_result); - solve.dec()._solve_impl(solve.rhs(), m_result); - } - -protected: - PlainObject m_result; -}; + typedef Solve SolveType; + typedef typename SolveType::PlainObject PlainObject; + typedef evaluator Base; -// Specialization for "dst = dec.solve(rhs)" -// NOTE we need to specialize it for Dense2Dense to avoid ambiguous specialization error and a Sparse2Sparse specialization must exist somewhere -template -struct Assignment, internal::assign_op, Dense2Dense> -{ - typedef Solve SrcXprType; - static void run(DstXprType &dst, const SrcXprType &src, const internal::assign_op &) - { - Index dstRows = src.rows(); - Index dstCols = src.cols(); - if((dst.rows()!=dstRows) || (dst.cols()!=dstCols)) - dst.resize(dstRows, dstCols); + enum { Flags = Base::Flags | EvalBeforeNestingBit }; - src.dec()._solve_impl(src.rhs(), dst); - } -}; + EIGEN_DEVICE_FUNC explicit evaluator(const SolveType &solve) : m_result(solve.rows(), solve.cols()) + { + ::new (static_cast(this)) Base(m_result); + solve.dec()._solve_impl(solve.rhs(), m_result); + } -// Specialization for "dst = dec.transpose().solve(rhs)" -template -struct Assignment,RhsType>, internal::assign_op, Dense2Dense> -{ - typedef Solve,RhsType> SrcXprType; - static void run(DstXprType &dst, const SrcXprType &src, const internal::assign_op &) + protected: + PlainObject m_result; + }; + + // Specialization for "dst = dec.solve(rhs)" + // NOTE we need to specialize it for Dense2Dense to avoid ambiguous specialization error and a Sparse2Sparse + // specialization must exist somewhere + template + struct Assignment, internal::assign_op, Dense2Dense> { - Index dstRows = src.rows(); - Index dstCols = src.cols(); - if((dst.rows()!=dstRows) || (dst.cols()!=dstCols)) - dst.resize(dstRows, dstCols); + typedef Solve SrcXprType; + static void run(DstXprType &dst, const SrcXprType &src, const internal::assign_op &) + { + Index dstRows = src.rows(); + Index dstCols = src.cols(); + if ((dst.rows() != dstRows) || (dst.cols() != dstCols)) dst.resize(dstRows, dstCols); + + src.dec()._solve_impl(src.rhs(), dst); + } + }; - src.dec().nestedExpression().template _solve_impl_transposed(src.rhs(), dst); - } -}; + // Specialization for "dst = dec.transpose().solve(rhs)" + template + struct Assignment, RhsType>, + internal::assign_op, + Dense2Dense> + { + typedef Solve, RhsType> SrcXprType; + static void run(DstXprType &dst, const SrcXprType &src, const internal::assign_op &) + { + Index dstRows = src.rows(); + Index dstCols = src.cols(); + if ((dst.rows() != dstRows) || (dst.cols() != dstCols)) dst.resize(dstRows, dstCols); + + src.dec().nestedExpression().template _solve_impl_transposed(src.rhs(), dst); + } + }; -// Specialization for "dst = dec.adjoint().solve(rhs)" -template -struct Assignment, const Transpose >,RhsType>, - internal::assign_op, Dense2Dense> -{ - typedef Solve, const Transpose >,RhsType> SrcXprType; - static void run(DstXprType &dst, const SrcXprType &src, const internal::assign_op &) + // Specialization for "dst = dec.adjoint().solve(rhs)" + template + struct Assignment, const Transpose>, + RhsType>, + internal::assign_op, + Dense2Dense> { - Index dstRows = src.rows(); - Index dstCols = src.cols(); - if((dst.rows()!=dstRows) || (dst.cols()!=dstCols)) - dst.resize(dstRows, dstCols); - - src.dec().nestedExpression().nestedExpression().template _solve_impl_transposed(src.rhs(), dst); - } -}; + typedef Solve, const Transpose>, + RhsType> + SrcXprType; + static void run(DstXprType &dst, const SrcXprType &src, const internal::assign_op &) + { + Index dstRows = src.rows(); + Index dstCols = src.cols(); + if ((dst.rows() != dstRows) || (dst.cols() != dstCols)) dst.resize(dstRows, dstCols); + + src.dec().nestedExpression().nestedExpression().template _solve_impl_transposed(src.rhs(), dst); + } + }; -} // end namepsace internal +}// namespace internal -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_SOLVE_H +#endif// EIGEN_SOLVE_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/SolveTriangular.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/SolveTriangular.h index 4652e2e1..d9d7581b 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/SolveTriangular.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/SolveTriangular.h @@ -10,226 +10,262 @@ #ifndef EIGEN_SOLVETRIANGULAR_H #define EIGEN_SOLVETRIANGULAR_H -namespace Eigen { +namespace Eigen { namespace internal { -// Forward declarations: -// The following two routines are implemented in the products/TriangularSolver*.h files -template -struct triangular_solve_vector; - -template -struct triangular_solve_matrix; - -// small helper struct extracting some traits on the underlying solver operation -template -class trsolve_traits -{ + // Forward declarations: + // The following two routines are implemented in the products/TriangularSolver*.h files + template + struct triangular_solve_vector; + + template + struct triangular_solve_matrix; + + // small helper struct extracting some traits on the underlying solver operation + template class trsolve_traits + { private: - enum { - RhsIsVectorAtCompileTime = (Side==OnTheLeft ? Rhs::ColsAtCompileTime : Rhs::RowsAtCompileTime)==1 - }; + enum { RhsIsVectorAtCompileTime = (Side == OnTheLeft ? Rhs::ColsAtCompileTime : Rhs::RowsAtCompileTime) == 1 }; + public: enum { - Unrolling = (RhsIsVectorAtCompileTime && Rhs::SizeAtCompileTime != Dynamic && Rhs::SizeAtCompileTime <= 8) - ? CompleteUnrolling : NoUnrolling, - RhsVectors = RhsIsVectorAtCompileTime ? 1 : Dynamic + Unrolling = (RhsIsVectorAtCompileTime && Rhs::SizeAtCompileTime != Dynamic && Rhs::SizeAtCompileTime <= 8) + ? CompleteUnrolling + : NoUnrolling, + RhsVectors = RhsIsVectorAtCompileTime ? 1 : Dynamic }; -}; - -template::Unrolling, - int RhsVectors = trsolve_traits::RhsVectors - > -struct triangular_solver_selector; - -template -struct triangular_solver_selector -{ - typedef typename Lhs::Scalar LhsScalar; - typedef typename Rhs::Scalar RhsScalar; - typedef blas_traits LhsProductTraits; - typedef typename LhsProductTraits::ExtractType ActualLhsType; - typedef Map, Aligned> MappedRhs; - static void run(const Lhs& lhs, Rhs& rhs) - { - ActualLhsType actualLhs = LhsProductTraits::extract(lhs); - - // FIXME find a way to allow an inner stride if packet_traits::size==1 - - bool useRhsDirectly = Rhs::InnerStrideAtCompileTime==1 || rhs.innerStride()==1; - - ei_declare_aligned_stack_constructed_variable(RhsScalar,actualRhs,rhs.size(), - (useRhsDirectly ? rhs.data() : 0)); - - if(!useRhsDirectly) - MappedRhs(actualRhs,rhs.size()) = rhs; - - triangular_solve_vector - ::run(actualLhs.cols(), actualLhs.data(), actualLhs.outerStride(), actualRhs); - - if(!useRhsDirectly) - rhs = MappedRhs(actualRhs, rhs.size()); - } -}; + }; -// the rhs is a matrix -template -struct triangular_solver_selector -{ - typedef typename Rhs::Scalar Scalar; - typedef blas_traits LhsProductTraits; - typedef typename LhsProductTraits::DirectLinearAccessType ActualLhsType; + template::Unrolling, + int RhsVectors = trsolve_traits::RhsVectors> + struct triangular_solver_selector; - static void run(const Lhs& lhs, Rhs& rhs) + template + struct triangular_solver_selector { - typename internal::add_const_on_value_type::type actualLhs = LhsProductTraits::extract(lhs); - - const Index size = lhs.rows(); - const Index othersize = Side==OnTheLeft? rhs.cols() : rhs.rows(); + typedef typename Lhs::Scalar LhsScalar; + typedef typename Rhs::Scalar RhsScalar; + typedef blas_traits LhsProductTraits; + typedef typename LhsProductTraits::ExtractType ActualLhsType; + typedef Map, Aligned> MappedRhs; + static void run(const Lhs &lhs, Rhs &rhs) + { + ActualLhsType actualLhs = LhsProductTraits::extract(lhs); + + // FIXME find a way to allow an inner stride if packet_traits::size==1 + + bool useRhsDirectly = Rhs::InnerStrideAtCompileTime == 1 || rhs.innerStride() == 1; + + ei_declare_aligned_stack_constructed_variable( + RhsScalar, actualRhs, rhs.size(), (useRhsDirectly ? rhs.data() : 0)); + + if (!useRhsDirectly) MappedRhs(actualRhs, rhs.size()) = rhs; + + triangular_solve_vector::run(actualLhs.cols(), + actualLhs.data(), + actualLhs.outerStride(), + actualRhs); + + if (!useRhsDirectly) rhs = MappedRhs(actualRhs, rhs.size()); + } + }; - typedef internal::gemm_blocking_space<(Rhs::Flags&RowMajorBit) ? RowMajor : ColMajor,Scalar,Scalar, - Rhs::MaxRowsAtCompileTime, Rhs::MaxColsAtCompileTime, Lhs::MaxRowsAtCompileTime,4> BlockingType; + // the rhs is a matrix + template + struct triangular_solver_selector + { + typedef typename Rhs::Scalar Scalar; + typedef blas_traits LhsProductTraits; + typedef typename LhsProductTraits::DirectLinearAccessType ActualLhsType; + + static void run(const Lhs &lhs, Rhs &rhs) + { + typename internal::add_const_on_value_type::type actualLhs = LhsProductTraits::extract(lhs); + + const Index size = lhs.rows(); + const Index othersize = Side == OnTheLeft ? rhs.cols() : rhs.rows(); + + typedef internal::gemm_blocking_space<(Rhs::Flags & RowMajorBit) ? RowMajor : ColMajor, + Scalar, + Scalar, + Rhs::MaxRowsAtCompileTime, + Rhs::MaxColsAtCompileTime, + Lhs::MaxRowsAtCompileTime, + 4> + BlockingType; + + BlockingType blocking(rhs.rows(), rhs.cols(), size, 1, false); + + triangular_solve_matrix::run(size, + othersize, + &actualLhs.coeffRef(0, 0), + actualLhs.outerStride(), + &rhs.coeffRef(0, 0), + rhs.outerStride(), + blocking); + } + }; - BlockingType blocking(rhs.rows(), rhs.cols(), size, 1, false); + /*************************************************************************** + * meta-unrolling implementation + ***************************************************************************/ - triangular_solve_matrix - ::run(size, othersize, &actualLhs.coeffRef(0,0), actualLhs.outerStride(), &rhs.coeffRef(0,0), rhs.outerStride(), blocking); - } -}; + template + struct triangular_solver_unroller; -/*************************************************************************** -* meta-unrolling implementation -***************************************************************************/ - -template -struct triangular_solver_unroller; + template + struct triangular_solver_unroller + { + enum { + IsLower = ((Mode & Lower) == Lower), + DiagIndex = IsLower ? LoopIndex : Size - LoopIndex - 1, + StartIndex = IsLower ? 0 : DiagIndex + 1 + }; + static void run(const Lhs &lhs, Rhs &rhs) + { + if (LoopIndex > 0) + rhs.coeffRef(DiagIndex) -= lhs.row(DiagIndex) + .template segment(StartIndex) + .transpose() + .cwiseProduct(rhs.template segment(StartIndex)) + .sum(); + + if (!(Mode & UnitDiag)) rhs.coeffRef(DiagIndex) /= lhs.coeff(DiagIndex, DiagIndex); + + triangular_solver_unroller::run(lhs, rhs); + } + }; -template -struct triangular_solver_unroller { - enum { - IsLower = ((Mode&Lower)==Lower), - DiagIndex = IsLower ? LoopIndex : Size - LoopIndex - 1, - StartIndex = IsLower ? 0 : DiagIndex+1 + template + struct triangular_solver_unroller + { + static void run(const Lhs &, Rhs &) {} }; - static void run(const Lhs& lhs, Rhs& rhs) + + template + struct triangular_solver_selector { - if (LoopIndex>0) - rhs.coeffRef(DiagIndex) -= lhs.row(DiagIndex).template segment(StartIndex).transpose() - .cwiseProduct(rhs.template segment(StartIndex)).sum(); - - if(!(Mode & UnitDiag)) - rhs.coeffRef(DiagIndex) /= lhs.coeff(DiagIndex,DiagIndex); - - triangular_solver_unroller::run(lhs,rhs); - } -}; - -template -struct triangular_solver_unroller { - static void run(const Lhs&, Rhs&) {} -}; - -template -struct triangular_solver_selector { - static void run(const Lhs& lhs, Rhs& rhs) - { triangular_solver_unroller::run(lhs,rhs); } -}; - -template -struct triangular_solver_selector { - static void run(const Lhs& lhs, Rhs& rhs) + static void run(const Lhs &lhs, Rhs &rhs) + { + triangular_solver_unroller::run(lhs, rhs); + } + }; + + template + struct triangular_solver_selector { - Transpose trLhs(lhs); - Transpose trRhs(rhs); - - triangular_solver_unroller,Transpose, - ((Mode&Upper)==Upper ? Lower : Upper) | (Mode&UnitDiag), - 0,Rhs::SizeAtCompileTime>::run(trLhs,trRhs); - } -}; + static void run(const Lhs &lhs, Rhs &rhs) + { + Transpose trLhs(lhs); + Transpose trRhs(rhs); + + triangular_solver_unroller, + Transpose, + ((Mode & Upper) == Upper ? Lower : Upper) | (Mode & UnitDiag), + 0, + Rhs::SizeAtCompileTime>::run(trLhs, trRhs); + } + }; -} // end namespace internal +}// end namespace internal /*************************************************************************** -* TriangularView methods -***************************************************************************/ + * TriangularView methods + ***************************************************************************/ #ifndef EIGEN_PARSED_BY_DOXYGEN template template -void TriangularViewImpl::solveInPlace(const MatrixBase& _other) const +void TriangularViewImpl::solveInPlace(const MatrixBase &_other) const { - OtherDerived& other = _other.const_cast_derived(); - eigen_assert( derived().cols() == derived().rows() && ((Side==OnTheLeft && derived().cols() == other.rows()) || (Side==OnTheRight && derived().cols() == other.cols())) ); - eigen_assert((!(Mode & ZeroDiag)) && bool(Mode & (Upper|Lower))); + OtherDerived &other = _other.const_cast_derived(); + eigen_assert(derived().cols() == derived().rows() + && ((Side == OnTheLeft && derived().cols() == other.rows()) + || (Side == OnTheRight && derived().cols() == other.cols()))); + eigen_assert((!(Mode & ZeroDiag)) && bool(Mode & (Upper | Lower))); // If solving for a 0x0 matrix, nothing to do, simply return. - if (derived().cols() == 0) - return; + if (derived().cols() == 0) return; - enum { copy = (internal::traits::Flags & RowMajorBit) && OtherDerived::IsVectorAtCompileTime && OtherDerived::SizeAtCompileTime!=1}; + enum { + copy = (internal::traits::Flags & RowMajorBit) && OtherDerived::IsVectorAtCompileTime + && OtherDerived::SizeAtCompileTime != 1 + }; typedef typename internal::conditional::type, OtherDerived&>::type OtherCopy; + typename internal::plain_matrix_type_column_major::type, + OtherDerived &>::type OtherCopy; OtherCopy otherCopy(other); - internal::triangular_solver_selector::type, - Side, Mode>::run(derived().nestedExpression(), otherCopy); + internal::triangular_solver_selector::type, Side, Mode>:: + run(derived().nestedExpression(), otherCopy); - if (copy) - other = otherCopy; + if (copy) other = otherCopy; } template template -const internal::triangular_solve_retval,Other> -TriangularViewImpl::solve(const MatrixBase& other) const +const internal::triangular_solve_retval, Other> + TriangularViewImpl::solve(const MatrixBase &other) const { - return internal::triangular_solve_retval(derived(), other.derived()); + return internal::triangular_solve_retval(derived(), other.derived()); } #endif namespace internal { -template -struct traits > -{ - typedef typename internal::plain_matrix_type_column_major::type ReturnType; -}; + template + struct traits> + { + typedef typename internal::plain_matrix_type_column_major::type ReturnType; + }; -template struct triangular_solve_retval - : public ReturnByValue > -{ - typedef typename remove_all::type RhsNestedCleaned; - typedef ReturnByValue Base; + template + struct triangular_solve_retval : public ReturnByValue> + { + typedef typename remove_all::type RhsNestedCleaned; + typedef ReturnByValue Base; - triangular_solve_retval(const TriangularType& tri, const Rhs& rhs) - : m_triangularMatrix(tri), m_rhs(rhs) - {} + triangular_solve_retval(const TriangularType &tri, const Rhs &rhs) : m_triangularMatrix(tri), m_rhs(rhs) {} - inline Index rows() const { return m_rhs.rows(); } - inline Index cols() const { return m_rhs.cols(); } + inline Index rows() const { return m_rhs.rows(); } + inline Index cols() const { return m_rhs.cols(); } - template inline void evalTo(Dest& dst) const - { - if(!is_same_dense(dst,m_rhs)) - dst = m_rhs; - m_triangularMatrix.template solveInPlace(dst); - } + template inline void evalTo(Dest &dst) const + { + if (!is_same_dense(dst, m_rhs)) dst = m_rhs; + m_triangularMatrix.template solveInPlace(dst); + } protected: - const TriangularType& m_triangularMatrix; + const TriangularType &m_triangularMatrix; typename Rhs::Nested m_rhs; -}; + }; -} // namespace internal +}// namespace internal -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_SOLVETRIANGULAR_H +#endif// EIGEN_SOLVETRIANGULAR_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/SolverBase.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/SolverBase.h index 8a4adc22..6b9502dc 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/SolverBase.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/SolverBase.h @@ -12,119 +12,101 @@ namespace Eigen { -namespace internal { - - - -} // end namespace internal +namespace internal {}// end namespace internal /** \class SolverBase - * \brief A base class for matrix decomposition and solvers - * - * \tparam Derived the actual type of the decomposition/solver. - * - * Any matrix decomposition inheriting this base class provide the following API: - * - * \code - * MatrixType A, b, x; - * DecompositionType dec(A); - * x = dec.solve(b); // solve A * x = b - * x = dec.transpose().solve(b); // solve A^T * x = b - * x = dec.adjoint().solve(b); // solve A' * x = b - * \endcode - * - * \warning Currently, any other usage of transpose() and adjoint() are not supported and will produce compilation errors. - * - * \sa class PartialPivLU, class FullPivLU - */ -template -class SolverBase : public EigenBase + * \brief A base class for matrix decomposition and solvers + * + * \tparam Derived the actual type of the decomposition/solver. + * + * Any matrix decomposition inheriting this base class provide the following API: + * + * \code + * MatrixType A, b, x; + * DecompositionType dec(A); + * x = dec.solve(b); // solve A * x = b + * x = dec.transpose().solve(b); // solve A^T * x = b + * x = dec.adjoint().solve(b); // solve A' * x = b + * \endcode + * + * \warning Currently, any other usage of transpose() and adjoint() are not supported and will produce compilation + * errors. + * + * \sa class PartialPivLU, class FullPivLU + */ +template class SolverBase : public EigenBase { - public: - - typedef EigenBase Base; - typedef typename internal::traits::Scalar Scalar; - typedef Scalar CoeffReturnType; - - enum { - RowsAtCompileTime = internal::traits::RowsAtCompileTime, - ColsAtCompileTime = internal::traits::ColsAtCompileTime, - SizeAtCompileTime = (internal::size_at_compile_time::RowsAtCompileTime, - internal::traits::ColsAtCompileTime>::ret), - MaxRowsAtCompileTime = internal::traits::MaxRowsAtCompileTime, - MaxColsAtCompileTime = internal::traits::MaxColsAtCompileTime, - MaxSizeAtCompileTime = (internal::size_at_compile_time::MaxRowsAtCompileTime, - internal::traits::MaxColsAtCompileTime>::ret), - IsVectorAtCompileTime = internal::traits::MaxRowsAtCompileTime == 1 - || internal::traits::MaxColsAtCompileTime == 1 - }; - - /** Default constructor */ - SolverBase() - {} - - ~SolverBase() - {} - - using Base::derived; - - /** \returns an expression of the solution x of \f$ A x = b \f$ using the current decomposition of A. - */ - template - inline const Solve - solve(const MatrixBase& b) const - { - eigen_assert(derived().rows()==b.rows() && "solve(): invalid number of rows of the right hand side matrix b"); - return Solve(derived(), b.derived()); - } - - /** \internal the return type of transpose() */ - typedef typename internal::add_const >::type ConstTransposeReturnType; - /** \returns an expression of the transposed of the factored matrix. - * - * A typical usage is to solve for the transposed problem A^T x = b: - * \code x = dec.transpose().solve(b); \endcode - * - * \sa adjoint(), solve() - */ - inline ConstTransposeReturnType transpose() const - { - return ConstTransposeReturnType(derived()); - } - - /** \internal the return type of adjoint() */ - typedef typename internal::conditional::IsComplex, - CwiseUnaryOp, ConstTransposeReturnType>, - ConstTransposeReturnType - >::type AdjointReturnType; - /** \returns an expression of the adjoint of the factored matrix - * - * A typical usage is to solve for the adjoint problem A' x = b: - * \code x = dec.adjoint().solve(b); \endcode - * - * For real scalar types, this function is equivalent to transpose(). - * - * \sa transpose(), solve() - */ - inline AdjointReturnType adjoint() const - { - return AdjointReturnType(derived().transpose()); - } - - protected: +public: + typedef EigenBase Base; + typedef typename internal::traits::Scalar Scalar; + typedef Scalar CoeffReturnType; + + enum { + RowsAtCompileTime = internal::traits::RowsAtCompileTime, + ColsAtCompileTime = internal::traits::ColsAtCompileTime, + SizeAtCompileTime = (internal::size_at_compile_time::RowsAtCompileTime, + internal::traits::ColsAtCompileTime>::ret), + MaxRowsAtCompileTime = internal::traits::MaxRowsAtCompileTime, + MaxColsAtCompileTime = internal::traits::MaxColsAtCompileTime, + MaxSizeAtCompileTime = (internal::size_at_compile_time::MaxRowsAtCompileTime, + internal::traits::MaxColsAtCompileTime>::ret), + IsVectorAtCompileTime = + internal::traits::MaxRowsAtCompileTime == 1 || internal::traits::MaxColsAtCompileTime == 1 + }; + + /** Default constructor */ + SolverBase() {} + + ~SolverBase() {} + + using Base::derived; + + /** \returns an expression of the solution x of \f$ A x = b \f$ using the current decomposition of A. + */ + template inline const Solve solve(const MatrixBase &b) const + { + eigen_assert(derived().rows() == b.rows() && "solve(): invalid number of rows of the right hand side matrix b"); + return Solve(derived(), b.derived()); + } + + /** \internal the return type of transpose() */ + typedef typename internal::add_const>::type ConstTransposeReturnType; + /** \returns an expression of the transposed of the factored matrix. + * + * A typical usage is to solve for the transposed problem A^T x = b: + * \code x = dec.transpose().solve(b); \endcode + * + * \sa adjoint(), solve() + */ + inline ConstTransposeReturnType transpose() const { return ConstTransposeReturnType(derived()); } + + /** \internal the return type of adjoint() */ + typedef typename internal::conditional::IsComplex, + CwiseUnaryOp, ConstTransposeReturnType>, + ConstTransposeReturnType>::type AdjointReturnType; + /** \returns an expression of the adjoint of the factored matrix + * + * A typical usage is to solve for the adjoint problem A' x = b: + * \code x = dec.adjoint().solve(b); \endcode + * + * For real scalar types, this function is equivalent to transpose(). + * + * \sa transpose(), solve() + */ + inline AdjointReturnType adjoint() const { return AdjointReturnType(derived().transpose()); } + +protected: }; namespace internal { -template -struct generic_xpr_base -{ - typedef SolverBase type; - -}; + template struct generic_xpr_base + { + typedef SolverBase type; + }; -} // end namespace internal +}// end namespace internal -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_SOLVERBASE_H +#endif// EIGEN_SOLVERBASE_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/StableNorm.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/StableNorm.h index 88c8d989..59729c47 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/StableNorm.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/StableNorm.h @@ -10,212 +10,196 @@ #ifndef EIGEN_STABLENORM_H #define EIGEN_STABLENORM_H -namespace Eigen { +namespace Eigen { namespace internal { -template -inline void stable_norm_kernel(const ExpressionType& bl, Scalar& ssq, Scalar& scale, Scalar& invScale) -{ - Scalar maxCoeff = bl.cwiseAbs().maxCoeff(); - - if(maxCoeff>scale) + template + inline void stable_norm_kernel(const ExpressionType &bl, Scalar &ssq, Scalar &scale, Scalar &invScale) { - ssq = ssq * numext::abs2(scale/maxCoeff); - Scalar tmp = Scalar(1)/maxCoeff; - if(tmp > NumTraits::highest()) - { - invScale = NumTraits::highest(); - scale = Scalar(1)/invScale; - } - else if(maxCoeff>NumTraits::highest()) // we got a INF - { - invScale = Scalar(1); - scale = maxCoeff; - } - else + Scalar maxCoeff = bl.cwiseAbs().maxCoeff(); + + if (maxCoeff > scale) { + ssq = ssq * numext::abs2(scale / maxCoeff); + Scalar tmp = Scalar(1) / maxCoeff; + if (tmp > NumTraits::highest()) { + invScale = NumTraits::highest(); + scale = Scalar(1) / invScale; + } else if (maxCoeff > NumTraits::highest())// we got a INF + { + invScale = Scalar(1); + scale = maxCoeff; + } else { + scale = maxCoeff; + invScale = tmp; + } + } else if (maxCoeff != maxCoeff)// we got a NaN { scale = maxCoeff; - invScale = tmp; } - } - else if(maxCoeff!=maxCoeff) // we got a NaN - { - scale = maxCoeff; - } - - // TODO if the maxCoeff is much much smaller than the current scale, - // then we can neglect this sub vector - if(scale>Scalar(0)) // if scale==0, then bl is 0 - ssq += (bl*invScale).squaredNorm(); -} -template -inline typename NumTraits::Scalar>::Real -blueNorm_impl(const EigenBase& _vec) -{ - typedef typename Derived::RealScalar RealScalar; - using std::pow; - using std::sqrt; - using std::abs; - const Derived& vec(_vec.derived()); - static bool initialized = false; - static RealScalar b1, b2, s1m, s2m, rbig, relerr; - if(!initialized) - { - int ibeta, it, iemin, iemax, iexp; - RealScalar eps; - // This program calculates the machine-dependent constants - // bl, b2, slm, s2m, relerr overfl - // from the "basic" machine-dependent numbers - // nbig, ibeta, it, iemin, iemax, rbig. - // The following define the basic machine-dependent constants. - // For portability, the PORT subprograms "ilmaeh" and "rlmach" - // are used. For any specific computer, each of the assignment - // statements can be replaced - ibeta = std::numeric_limits::radix; // base for floating-point numbers - it = std::numeric_limits::digits; // number of base-beta digits in mantissa - iemin = std::numeric_limits::min_exponent; // minimum exponent - iemax = std::numeric_limits::max_exponent; // maximum exponent - rbig = (std::numeric_limits::max)(); // largest floating-point number - - iexp = -((1-iemin)/2); - b1 = RealScalar(pow(RealScalar(ibeta),RealScalar(iexp))); // lower boundary of midrange - iexp = (iemax + 1 - it)/2; - b2 = RealScalar(pow(RealScalar(ibeta),RealScalar(iexp))); // upper boundary of midrange - - iexp = (2-iemin)/2; - s1m = RealScalar(pow(RealScalar(ibeta),RealScalar(iexp))); // scaling factor for lower range - iexp = - ((iemax+it)/2); - s2m = RealScalar(pow(RealScalar(ibeta),RealScalar(iexp))); // scaling factor for upper range - - eps = RealScalar(pow(double(ibeta), 1-it)); - relerr = sqrt(eps); // tolerance for neglecting asml - initialized = true; + // TODO if the maxCoeff is much much smaller than the current scale, + // then we can neglect this sub vector + if (scale > Scalar(0))// if scale==0, then bl is 0 + ssq += (bl * invScale).squaredNorm(); } - Index n = vec.size(); - RealScalar ab2 = b2 / RealScalar(n); - RealScalar asml = RealScalar(0); - RealScalar amed = RealScalar(0); - RealScalar abig = RealScalar(0); - for(typename Derived::InnerIterator it(vec, 0); it; ++it) - { - RealScalar ax = abs(it.value()); - if(ax > ab2) abig += numext::abs2(ax*s2m); - else if(ax < b1) asml += numext::abs2(ax*s1m); - else amed += numext::abs2(ax); - } - if(amed!=amed) - return amed; // we got a NaN - if(abig > RealScalar(0)) + + template + inline typename NumTraits::Scalar>::Real blueNorm_impl(const EigenBase &_vec) { - abig = sqrt(abig); - if(abig > rbig) // overflow, or *this contains INF values - return abig; // return INF - if(amed > RealScalar(0)) - { - abig = abig/s2m; - amed = sqrt(amed); + typedef typename Derived::RealScalar RealScalar; + using std::pow; + using std::sqrt; + using std::abs; + const Derived &vec(_vec.derived()); + static bool initialized = false; + static RealScalar b1, b2, s1m, s2m, rbig, relerr; + if (!initialized) { + int ibeta, it, iemin, iemax, iexp; + RealScalar eps; + // This program calculates the machine-dependent constants + // bl, b2, slm, s2m, relerr overfl + // from the "basic" machine-dependent numbers + // nbig, ibeta, it, iemin, iemax, rbig. + // The following define the basic machine-dependent constants. + // For portability, the PORT subprograms "ilmaeh" and "rlmach" + // are used. For any specific computer, each of the assignment + // statements can be replaced + ibeta = std::numeric_limits::radix;// base for floating-point numbers + it = std::numeric_limits::digits;// number of base-beta digits in mantissa + iemin = std::numeric_limits::min_exponent;// minimum exponent + iemax = std::numeric_limits::max_exponent;// maximum exponent + rbig = (std::numeric_limits::max)();// largest floating-point number + + iexp = -((1 - iemin) / 2); + b1 = RealScalar(pow(RealScalar(ibeta), RealScalar(iexp)));// lower boundary of midrange + iexp = (iemax + 1 - it) / 2; + b2 = RealScalar(pow(RealScalar(ibeta), RealScalar(iexp)));// upper boundary of midrange + + iexp = (2 - iemin) / 2; + s1m = RealScalar(pow(RealScalar(ibeta), RealScalar(iexp)));// scaling factor for lower range + iexp = -((iemax + it) / 2); + s2m = RealScalar(pow(RealScalar(ibeta), RealScalar(iexp)));// scaling factor for upper range + + eps = RealScalar(pow(double(ibeta), 1 - it)); + relerr = sqrt(eps);// tolerance for neglecting asml + initialized = true; } - else - return abig/s2m; - } - else if(asml > RealScalar(0)) - { - if (amed > RealScalar(0)) - { - abig = sqrt(amed); - amed = sqrt(asml) / s1m; + Index n = vec.size(); + RealScalar ab2 = b2 / RealScalar(n); + RealScalar asml = RealScalar(0); + RealScalar amed = RealScalar(0); + RealScalar abig = RealScalar(0); + for (typename Derived::InnerIterator it(vec, 0); it; ++it) { + RealScalar ax = abs(it.value()); + if (ax > ab2) + abig += numext::abs2(ax * s2m); + else if (ax < b1) + asml += numext::abs2(ax * s1m); + else + amed += numext::abs2(ax); } + if (amed != amed) return amed;// we got a NaN + if (abig > RealScalar(0)) { + abig = sqrt(abig); + if (abig > rbig)// overflow, or *this contains INF values + return abig;// return INF + if (amed > RealScalar(0)) { + abig = abig / s2m; + amed = sqrt(amed); + } else + return abig / s2m; + } else if (asml > RealScalar(0)) { + if (amed > RealScalar(0)) { + abig = sqrt(amed); + amed = sqrt(asml) / s1m; + } else + return sqrt(asml) / s1m; + } else + return sqrt(amed); + asml = numext::mini(abig, amed); + abig = numext::maxi(abig, amed); + if (asml <= abig * relerr) + return abig; else - return sqrt(asml)/s1m; + return abig * sqrt(RealScalar(1) + numext::abs2(asml / abig)); } - else - return sqrt(amed); - asml = numext::mini(abig, amed); - abig = numext::maxi(abig, amed); - if(asml <= abig*relerr) - return abig; - else - return abig * sqrt(RealScalar(1) + numext::abs2(asml/abig)); -} -} // end namespace internal +}// end namespace internal /** \returns the \em l2 norm of \c *this avoiding underflow and overflow. - * This version use a blockwise two passes algorithm: - * 1 - find the absolute largest coefficient \c s - * 2 - compute \f$ s \Vert \frac{*this}{s} \Vert \f$ in a standard way - * - * For architecture/scalar types supporting vectorization, this version - * is faster than blueNorm(). Otherwise the blueNorm() is much faster. - * - * \sa norm(), blueNorm(), hypotNorm() - */ + * This version use a blockwise two passes algorithm: + * 1 - find the absolute largest coefficient \c s + * 2 - compute \f$ s \Vert \frac{*this}{s} \Vert \f$ in a standard way + * + * For architecture/scalar types supporting vectorization, this version + * is faster than blueNorm(). Otherwise the blueNorm() is much faster. + * + * \sa norm(), blueNorm(), hypotNorm() + */ template -inline typename NumTraits::Scalar>::Real -MatrixBase::stableNorm() const +inline typename NumTraits::Scalar>::Real MatrixBase::stableNorm() const { using std::sqrt; using std::abs; const Index blockSize = 4096; RealScalar scale(0); RealScalar invScale(1); - RealScalar ssq(0); // sum of square - - typedef typename internal::nested_eval::type DerivedCopy; + RealScalar ssq(0);// sum of square + + typedef typename internal::nested_eval::type DerivedCopy; typedef typename internal::remove_all::type DerivedCopyClean; const DerivedCopy copy(derived()); - + enum { - CanAlign = ( (int(DerivedCopyClean::Flags)&DirectAccessBit) - || (int(internal::evaluator::Alignment)>0) // FIXME Alignment)>0 might not be enough - ) && (blockSize*sizeof(Scalar)*20) // if we cannot allocate on the stack, then let's not bother about this optimization + CanAlign = ((int(DerivedCopyClean::Flags) & DirectAccessBit) + || (int(internal::evaluator::Alignment) > 0)// FIXME Alignment)>0 might not be enough + ) + && (blockSize * sizeof(Scalar) * 2 < EIGEN_STACK_ALLOCATION_LIMIT) + && (EIGEN_MAX_STATIC_ALIGN_BYTES + > 0)// if we cannot allocate on the stack, then let's not bother about this optimization }; - typedef typename internal::conditional, internal::evaluator::Alignment>, - typename DerivedCopyClean::ConstSegmentReturnType>::type SegmentWrapper; + typedef typename internal::conditional, internal::evaluator::Alignment>, + typename DerivedCopyClean::ConstSegmentReturnType>::type SegmentWrapper; Index n = size(); - - if(n==1) - return abs(this->coeff(0)); - + + if (n == 1) return abs(this->coeff(0)); + Index bi = internal::first_default_aligned(copy); - if (bi>0) - internal::stable_norm_kernel(copy.head(bi), ssq, scale, invScale); - for (; bi 0) internal::stable_norm_kernel(copy.head(bi), ssq, scale, invScale); + for (; bi < n; bi += blockSize) + internal::stable_norm_kernel( + SegmentWrapper(copy.segment(bi, numext::mini(blockSize, n - bi))), ssq, scale, invScale); return scale * sqrt(ssq); } /** \returns the \em l2 norm of \c *this using the Blue's algorithm. - * A Portable Fortran Program to Find the Euclidean Norm of a Vector, - * ACM TOMS, Vol 4, Issue 1, 1978. - * - * For architecture/scalar types without vectorization, this version - * is much faster than stableNorm(). Otherwise the stableNorm() is faster. - * - * \sa norm(), stableNorm(), hypotNorm() - */ + * A Portable Fortran Program to Find the Euclidean Norm of a Vector, + * ACM TOMS, Vol 4, Issue 1, 1978. + * + * For architecture/scalar types without vectorization, this version + * is much faster than stableNorm(). Otherwise the stableNorm() is faster. + * + * \sa norm(), stableNorm(), hypotNorm() + */ template -inline typename NumTraits::Scalar>::Real -MatrixBase::blueNorm() const +inline typename NumTraits::Scalar>::Real MatrixBase::blueNorm() const { return internal::blueNorm_impl(*this); } /** \returns the \em l2 norm of \c *this avoiding undeflow and overflow. - * This version use a concatenation of hypot() calls, and it is very slow. - * - * \sa norm(), stableNorm() - */ + * This version use a concatenation of hypot() calls, and it is very slow. + * + * \sa norm(), stableNorm() + */ template -inline typename NumTraits::Scalar>::Real -MatrixBase::hypotNorm() const +inline typename NumTraits::Scalar>::Real MatrixBase::hypotNorm() const { return this->cwiseAbs().redux(internal::scalar_hypot_op()); } -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_STABLENORM_H +#endif// EIGEN_STABLENORM_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/Stride.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/Stride.h index 513742f3..35949abb 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/Stride.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/Stride.h @@ -10,102 +10,94 @@ #ifndef EIGEN_STRIDE_H #define EIGEN_STRIDE_H -namespace Eigen { +namespace Eigen { /** \class Stride - * \ingroup Core_Module - * - * \brief Holds strides information for Map - * - * This class holds the strides information for mapping arrays with strides with class Map. - * - * It holds two values: the inner stride and the outer stride. - * - * The inner stride is the pointer increment between two consecutive entries within a given row of a - * row-major matrix or within a given column of a column-major matrix. - * - * The outer stride is the pointer increment between two consecutive rows of a row-major matrix or - * between two consecutive columns of a column-major matrix. - * - * These two values can be passed either at compile-time as template parameters, or at runtime as - * arguments to the constructor. - * - * Indeed, this class takes two template parameters: - * \tparam _OuterStrideAtCompileTime the outer stride, or Dynamic if you want to specify it at runtime. - * \tparam _InnerStrideAtCompileTime the inner stride, or Dynamic if you want to specify it at runtime. - * - * Here is an example: - * \include Map_general_stride.cpp - * Output: \verbinclude Map_general_stride.out - * - * \sa class InnerStride, class OuterStride, \ref TopicStorageOrders - */ -template -class Stride + * \ingroup Core_Module + * + * \brief Holds strides information for Map + * + * This class holds the strides information for mapping arrays with strides with class Map. + * + * It holds two values: the inner stride and the outer stride. + * + * The inner stride is the pointer increment between two consecutive entries within a given row of a + * row-major matrix or within a given column of a column-major matrix. + * + * The outer stride is the pointer increment between two consecutive rows of a row-major matrix or + * between two consecutive columns of a column-major matrix. + * + * These two values can be passed either at compile-time as template parameters, or at runtime as + * arguments to the constructor. + * + * Indeed, this class takes two template parameters: + * \tparam _OuterStrideAtCompileTime the outer stride, or Dynamic if you want to specify it at runtime. + * \tparam _InnerStrideAtCompileTime the inner stride, or Dynamic if you want to specify it at runtime. + * + * Here is an example: + * \include Map_general_stride.cpp + * Output: \verbinclude Map_general_stride.out + * + * \sa class InnerStride, class OuterStride, \ref TopicStorageOrders + */ +template class Stride { - public: - typedef Eigen::Index Index; ///< \deprecated since Eigen 3.3 - enum { - InnerStrideAtCompileTime = _InnerStrideAtCompileTime, - OuterStrideAtCompileTime = _OuterStrideAtCompileTime - }; +public: + typedef Eigen::Index Index;///< \deprecated since Eigen 3.3 + enum { InnerStrideAtCompileTime = _InnerStrideAtCompileTime, OuterStrideAtCompileTime = _OuterStrideAtCompileTime }; - /** Default constructor, for use when strides are fixed at compile time */ - EIGEN_DEVICE_FUNC - Stride() - : m_outer(OuterStrideAtCompileTime), m_inner(InnerStrideAtCompileTime) - { - eigen_assert(InnerStrideAtCompileTime != Dynamic && OuterStrideAtCompileTime != Dynamic); - } + /** Default constructor, for use when strides are fixed at compile time */ + EIGEN_DEVICE_FUNC + Stride() : m_outer(OuterStrideAtCompileTime), m_inner(InnerStrideAtCompileTime) + { + eigen_assert(InnerStrideAtCompileTime != Dynamic && OuterStrideAtCompileTime != Dynamic); + } - /** Constructor allowing to pass the strides at runtime */ - EIGEN_DEVICE_FUNC - Stride(Index outerStride, Index innerStride) - : m_outer(outerStride), m_inner(innerStride) - { - eigen_assert(innerStride>=0 && outerStride>=0); - } + /** Constructor allowing to pass the strides at runtime */ + EIGEN_DEVICE_FUNC + Stride(Index outerStride, Index innerStride) : m_outer(outerStride), m_inner(innerStride) + { + eigen_assert(innerStride >= 0 && outerStride >= 0); + } - /** Copy constructor */ - EIGEN_DEVICE_FUNC - Stride(const Stride& other) - : m_outer(other.outer()), m_inner(other.inner()) - {} + /** Copy constructor */ + EIGEN_DEVICE_FUNC + Stride(const Stride &other) : m_outer(other.outer()), m_inner(other.inner()) {} - /** \returns the outer stride */ - EIGEN_DEVICE_FUNC - inline Index outer() const { return m_outer.value(); } - /** \returns the inner stride */ - EIGEN_DEVICE_FUNC - inline Index inner() const { return m_inner.value(); } + /** \returns the outer stride */ + EIGEN_DEVICE_FUNC + inline Index outer() const { return m_outer.value(); } + /** \returns the inner stride */ + EIGEN_DEVICE_FUNC + inline Index inner() const { return m_inner.value(); } - protected: - internal::variable_if_dynamic m_outer; - internal::variable_if_dynamic m_inner; +protected: + internal::variable_if_dynamic m_outer; + internal::variable_if_dynamic m_inner; }; /** \brief Convenience specialization of Stride to specify only an inner stride - * See class Map for some examples */ -template -class InnerStride : public Stride<0, Value> + * See class Map for some examples */ +template class InnerStride : public Stride<0, Value> { - typedef Stride<0, Value> Base; - public: - EIGEN_DEVICE_FUNC InnerStride() : Base() {} - EIGEN_DEVICE_FUNC InnerStride(Index v) : Base(0, v) {} // FIXME making this explicit could break valid code + typedef Stride<0, Value> Base; + +public: + EIGEN_DEVICE_FUNC InnerStride() : Base() {} + EIGEN_DEVICE_FUNC InnerStride(Index v) : Base(0, v) {}// FIXME making this explicit could break valid code }; /** \brief Convenience specialization of Stride to specify only an outer stride - * See class Map for some examples */ -template -class OuterStride : public Stride + * See class Map for some examples */ +template class OuterStride : public Stride { - typedef Stride Base; - public: - EIGEN_DEVICE_FUNC OuterStride() : Base() {} - EIGEN_DEVICE_FUNC OuterStride(Index v) : Base(v,0) {} // FIXME making this explicit could break valid code + typedef Stride Base; + +public: + EIGEN_DEVICE_FUNC OuterStride() : Base() {} + EIGEN_DEVICE_FUNC OuterStride(Index v) : Base(v, 0) {}// FIXME making this explicit could break valid code }; -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_STRIDE_H +#endif// EIGEN_STRIDE_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/Swap.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/Swap.h index d7020091..82bccf05 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/Swap.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/Swap.h @@ -10,58 +10,71 @@ #ifndef EIGEN_SWAP_H #define EIGEN_SWAP_H -namespace Eigen { +namespace Eigen { namespace internal { -// Overload default assignPacket behavior for swapping them -template -class generic_dense_assignment_kernel, Specialized> - : public generic_dense_assignment_kernel, BuiltIn> -{ -protected: - typedef generic_dense_assignment_kernel, BuiltIn> Base; - using Base::m_dst; - using Base::m_src; - using Base::m_functor; - -public: - typedef typename Base::Scalar Scalar; - typedef typename Base::DstXprType DstXprType; - typedef swap_assign_op Functor; - - EIGEN_DEVICE_FUNC generic_dense_assignment_kernel(DstEvaluatorTypeT &dst, const SrcEvaluatorTypeT &src, const Functor &func, DstXprType& dstExpr) - : Base(dst, src, func, dstExpr) - {} - - template - void assignPacket(Index row, Index col) + // Overload default assignPacket behavior for swapping them + template + class generic_dense_assignment_kernel, + Specialized> + : public generic_dense_assignment_kernel, + BuiltIn> { - PacketType tmp = m_src.template packet(row,col); - const_cast(m_src).template writePacket(row,col, m_dst.template packet(row,col)); - m_dst.template writePacket(row,col,tmp); - } - - template - void assignPacket(Index index) - { - PacketType tmp = m_src.template packet(index); - const_cast(m_src).template writePacket(index, m_dst.template packet(index)); - m_dst.template writePacket(index,tmp); - } - - // TODO find a simple way not to have to copy/paste this function from generic_dense_assignment_kernel, by simple I mean no CRTP (Gael) - template - void assignPacketByOuterInner(Index outer, Index inner) - { - Index row = Base::rowIndexByOuterInner(outer, inner); - Index col = Base::colIndexByOuterInner(outer, inner); - assignPacket(row, col); - } -}; + protected: + typedef generic_dense_assignment_kernel, + BuiltIn> + Base; + using Base::m_dst; + using Base::m_src; + using Base::m_functor; + + public: + typedef typename Base::Scalar Scalar; + typedef typename Base::DstXprType DstXprType; + typedef swap_assign_op Functor; + + EIGEN_DEVICE_FUNC generic_dense_assignment_kernel(DstEvaluatorTypeT &dst, + const SrcEvaluatorTypeT &src, + const Functor &func, + DstXprType &dstExpr) + : Base(dst, src, func, dstExpr) + {} + + template void assignPacket(Index row, Index col) + { + PacketType tmp = m_src.template packet(row, col); + const_cast(m_src).template writePacket( + row, col, m_dst.template packet(row, col)); + m_dst.template writePacket(row, col, tmp); + } + + template void assignPacket(Index index) + { + PacketType tmp = m_src.template packet(index); + const_cast(m_src).template writePacket( + index, m_dst.template packet(index)); + m_dst.template writePacket(index, tmp); + } + + // TODO find a simple way not to have to copy/paste this function from generic_dense_assignment_kernel, by simple I + // mean no CRTP (Gael) + template void assignPacketByOuterInner(Index outer, Index inner) + { + Index row = Base::rowIndexByOuterInner(outer, inner); + Index col = Base::colIndexByOuterInner(outer, inner); + assignPacket(row, col); + } + }; -} // namespace internal +}// namespace internal -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_SWAP_H +#endif// EIGEN_SWAP_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/Transpose.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/Transpose.h index 79b767bc..de721d70 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/Transpose.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/Transpose.h @@ -11,277 +11,257 @@ #ifndef EIGEN_TRANSPOSE_H #define EIGEN_TRANSPOSE_H -namespace Eigen { +namespace Eigen { namespace internal { -template -struct traits > : public traits -{ - typedef typename ref_selector::type MatrixTypeNested; - typedef typename remove_reference::type MatrixTypeNestedPlain; - enum { - RowsAtCompileTime = MatrixType::ColsAtCompileTime, - ColsAtCompileTime = MatrixType::RowsAtCompileTime, - MaxRowsAtCompileTime = MatrixType::MaxColsAtCompileTime, - MaxColsAtCompileTime = MatrixType::MaxRowsAtCompileTime, - FlagsLvalueBit = is_lvalue::value ? LvalueBit : 0, - Flags0 = traits::Flags & ~(LvalueBit | NestByRefBit), - Flags1 = Flags0 | FlagsLvalueBit, - Flags = Flags1 ^ RowMajorBit, - InnerStrideAtCompileTime = inner_stride_at_compile_time::ret, - OuterStrideAtCompileTime = outer_stride_at_compile_time::ret + template struct traits> : public traits + { + typedef typename ref_selector::type MatrixTypeNested; + typedef typename remove_reference::type MatrixTypeNestedPlain; + enum { + RowsAtCompileTime = MatrixType::ColsAtCompileTime, + ColsAtCompileTime = MatrixType::RowsAtCompileTime, + MaxRowsAtCompileTime = MatrixType::MaxColsAtCompileTime, + MaxColsAtCompileTime = MatrixType::MaxRowsAtCompileTime, + FlagsLvalueBit = is_lvalue::value ? LvalueBit : 0, + Flags0 = traits::Flags & ~(LvalueBit | NestByRefBit), + Flags1 = Flags0 | FlagsLvalueBit, + Flags = Flags1 ^ RowMajorBit, + InnerStrideAtCompileTime = inner_stride_at_compile_time::ret, + OuterStrideAtCompileTime = outer_stride_at_compile_time::ret + }; }; -}; -} +}// namespace internal template class TransposeImpl; /** \class Transpose - * \ingroup Core_Module - * - * \brief Expression of the transpose of a matrix - * - * \tparam MatrixType the type of the object of which we are taking the transpose - * - * This class represents an expression of the transpose of a matrix. - * It is the return type of MatrixBase::transpose() and MatrixBase::adjoint() - * and most of the time this is the only way it is used. - * - * \sa MatrixBase::transpose(), MatrixBase::adjoint() - */ -template class Transpose - : public TransposeImpl::StorageKind> + * \ingroup Core_Module + * + * \brief Expression of the transpose of a matrix + * + * \tparam MatrixType the type of the object of which we are taking the transpose + * + * This class represents an expression of the transpose of a matrix. + * It is the return type of MatrixBase::transpose() and MatrixBase::adjoint() + * and most of the time this is the only way it is used. + * + * \sa MatrixBase::transpose(), MatrixBase::adjoint() + */ +template +class Transpose : public TransposeImpl::StorageKind> { - public: - - typedef typename internal::ref_selector::non_const_type MatrixTypeNested; +public: + typedef typename internal::ref_selector::non_const_type MatrixTypeNested; - typedef typename TransposeImpl::StorageKind>::Base Base; - EIGEN_GENERIC_PUBLIC_INTERFACE(Transpose) - typedef typename internal::remove_all::type NestedExpression; + typedef typename TransposeImpl::StorageKind>::Base Base; + EIGEN_GENERIC_PUBLIC_INTERFACE(Transpose) + typedef typename internal::remove_all::type NestedExpression; - EIGEN_DEVICE_FUNC - explicit inline Transpose(MatrixType& matrix) : m_matrix(matrix) {} + EIGEN_DEVICE_FUNC + explicit inline Transpose(MatrixType &matrix) : m_matrix(matrix) {} - EIGEN_INHERIT_ASSIGNMENT_OPERATORS(Transpose) + EIGEN_INHERIT_ASSIGNMENT_OPERATORS(Transpose) - EIGEN_DEVICE_FUNC inline Index rows() const { return m_matrix.cols(); } - EIGEN_DEVICE_FUNC inline Index cols() const { return m_matrix.rows(); } + EIGEN_DEVICE_FUNC inline Index rows() const { return m_matrix.cols(); } + EIGEN_DEVICE_FUNC inline Index cols() const { return m_matrix.rows(); } - /** \returns the nested expression */ - EIGEN_DEVICE_FUNC - const typename internal::remove_all::type& - nestedExpression() const { return m_matrix; } + /** \returns the nested expression */ + EIGEN_DEVICE_FUNC + const typename internal::remove_all::type &nestedExpression() const { return m_matrix; } - /** \returns the nested expression */ - EIGEN_DEVICE_FUNC - typename internal::remove_reference::type& - nestedExpression() { return m_matrix; } + /** \returns the nested expression */ + EIGEN_DEVICE_FUNC + typename internal::remove_reference::type &nestedExpression() { return m_matrix; } - /** \internal */ - void resize(Index nrows, Index ncols) { - m_matrix.resize(ncols,nrows); - } + /** \internal */ + void resize(Index nrows, Index ncols) { m_matrix.resize(ncols, nrows); } - protected: - typename internal::ref_selector::non_const_type m_matrix; +protected: + typename internal::ref_selector::non_const_type m_matrix; }; namespace internal { -template::ret> -struct TransposeImpl_base -{ - typedef typename dense_xpr_base >::type type; -}; + template::ret> struct TransposeImpl_base + { + typedef typename dense_xpr_base>::type type; + }; -template -struct TransposeImpl_base -{ - typedef typename dense_xpr_base >::type type; -}; + template struct TransposeImpl_base + { + typedef typename dense_xpr_base>::type type; + }; -} // end namespace internal +}// end namespace internal // Generic API dispatcher template -class TransposeImpl - : public internal::generic_xpr_base >::type +class TransposeImpl : public internal::generic_xpr_base>::type { public: - typedef typename internal::generic_xpr_base >::type Base; + typedef typename internal::generic_xpr_base>::type Base; }; -template class TransposeImpl - : public internal::TransposeImpl_base::type +template +class TransposeImpl : public internal::TransposeImpl_base::type { - public: - - typedef typename internal::TransposeImpl_base::type Base; - using Base::coeffRef; - EIGEN_DENSE_PUBLIC_INTERFACE(Transpose) - EIGEN_INHERIT_ASSIGNMENT_OPERATORS(TransposeImpl) +public: + typedef typename internal::TransposeImpl_base::type Base; + using Base::coeffRef; + EIGEN_DENSE_PUBLIC_INTERFACE(Transpose) + EIGEN_INHERIT_ASSIGNMENT_OPERATORS(TransposeImpl) - EIGEN_DEVICE_FUNC inline Index innerStride() const { return derived().nestedExpression().innerStride(); } - EIGEN_DEVICE_FUNC inline Index outerStride() const { return derived().nestedExpression().outerStride(); } + EIGEN_DEVICE_FUNC inline Index innerStride() const { return derived().nestedExpression().innerStride(); } + EIGEN_DEVICE_FUNC inline Index outerStride() const { return derived().nestedExpression().outerStride(); } - typedef typename internal::conditional< - internal::is_lvalue::value, - Scalar, - const Scalar - >::type ScalarWithConstIfNotLvalue; + typedef typename internal::conditional::value, Scalar, const Scalar>::type + ScalarWithConstIfNotLvalue; - EIGEN_DEVICE_FUNC inline ScalarWithConstIfNotLvalue* data() { return derived().nestedExpression().data(); } - EIGEN_DEVICE_FUNC inline const Scalar* data() const { return derived().nestedExpression().data(); } + EIGEN_DEVICE_FUNC inline ScalarWithConstIfNotLvalue *data() { return derived().nestedExpression().data(); } + EIGEN_DEVICE_FUNC inline const Scalar *data() const { return derived().nestedExpression().data(); } - // FIXME: shall we keep the const version of coeffRef? - EIGEN_DEVICE_FUNC - inline const Scalar& coeffRef(Index rowId, Index colId) const - { - return derived().nestedExpression().coeffRef(colId, rowId); - } + // FIXME: shall we keep the const version of coeffRef? + EIGEN_DEVICE_FUNC + inline const Scalar &coeffRef(Index rowId, Index colId) const + { + return derived().nestedExpression().coeffRef(colId, rowId); + } - EIGEN_DEVICE_FUNC - inline const Scalar& coeffRef(Index index) const - { - return derived().nestedExpression().coeffRef(index); - } + EIGEN_DEVICE_FUNC + inline const Scalar &coeffRef(Index index) const { return derived().nestedExpression().coeffRef(index); } }; /** \returns an expression of the transpose of *this. - * - * Example: \include MatrixBase_transpose.cpp - * Output: \verbinclude MatrixBase_transpose.out - * - * \warning If you want to replace a matrix by its own transpose, do \b NOT do this: - * \code - * m = m.transpose(); // bug!!! caused by aliasing effect - * \endcode - * Instead, use the transposeInPlace() method: - * \code - * m.transposeInPlace(); - * \endcode - * which gives Eigen good opportunities for optimization, or alternatively you can also do: - * \code - * m = m.transpose().eval(); - * \endcode - * - * \sa transposeInPlace(), adjoint() */ -template -inline Transpose -DenseBase::transpose() + * + * Example: \include MatrixBase_transpose.cpp + * Output: \verbinclude MatrixBase_transpose.out + * + * \warning If you want to replace a matrix by its own transpose, do \b NOT do this: + * \code + * m = m.transpose(); // bug!!! caused by aliasing effect + * \endcode + * Instead, use the transposeInPlace() method: + * \code + * m.transposeInPlace(); + * \endcode + * which gives Eigen good opportunities for optimization, or alternatively you can also do: + * \code + * m = m.transpose().eval(); + * \endcode + * + * \sa transposeInPlace(), adjoint() */ +template inline Transpose DenseBase::transpose() { return TransposeReturnType(derived()); } /** This is the const version of transpose(). - * - * Make sure you read the warning for transpose() ! - * - * \sa transposeInPlace(), adjoint() */ + * + * Make sure you read the warning for transpose() ! + * + * \sa transposeInPlace(), adjoint() */ template -inline typename DenseBase::ConstTransposeReturnType -DenseBase::transpose() const +inline typename DenseBase::ConstTransposeReturnType DenseBase::transpose() const { return ConstTransposeReturnType(derived()); } /** \returns an expression of the adjoint (i.e. conjugate transpose) of *this. - * - * Example: \include MatrixBase_adjoint.cpp - * Output: \verbinclude MatrixBase_adjoint.out - * - * \warning If you want to replace a matrix by its own adjoint, do \b NOT do this: - * \code - * m = m.adjoint(); // bug!!! caused by aliasing effect - * \endcode - * Instead, use the adjointInPlace() method: - * \code - * m.adjointInPlace(); - * \endcode - * which gives Eigen good opportunities for optimization, or alternatively you can also do: - * \code - * m = m.adjoint().eval(); - * \endcode - * - * \sa adjointInPlace(), transpose(), conjugate(), class Transpose, class internal::scalar_conjugate_op */ + * + * Example: \include MatrixBase_adjoint.cpp + * Output: \verbinclude MatrixBase_adjoint.out + * + * \warning If you want to replace a matrix by its own adjoint, do \b NOT do this: + * \code + * m = m.adjoint(); // bug!!! caused by aliasing effect + * \endcode + * Instead, use the adjointInPlace() method: + * \code + * m.adjointInPlace(); + * \endcode + * which gives Eigen good opportunities for optimization, or alternatively you can also do: + * \code + * m = m.adjoint().eval(); + * \endcode + * + * \sa adjointInPlace(), transpose(), conjugate(), class Transpose, class internal::scalar_conjugate_op */ template -inline const typename MatrixBase::AdjointReturnType -MatrixBase::adjoint() const +inline const typename MatrixBase::AdjointReturnType MatrixBase::adjoint() const { return AdjointReturnType(this->transpose()); } /*************************************************************************** -* "in place" transpose implementation -***************************************************************************/ + * "in place" transpose implementation + ***************************************************************************/ namespace internal { -template::size)) - && (internal::evaluator::Flags&PacketAccessBit) > -struct inplace_transpose_selector; - -template -struct inplace_transpose_selector { // square matrix - static void run(MatrixType& m) { - m.matrix().template triangularView().swap(m.matrix().transpose()); - } -}; + template::size)) + && (internal::evaluator::Flags & PacketAccessBit)> + struct inplace_transpose_selector; + + template struct inplace_transpose_selector + {// square matrix + static void run(MatrixType &m) { m.matrix().template triangularView().swap(m.matrix().transpose()); } + }; -// TODO: vectorized path is currently limited to LargestPacketSize x LargestPacketSize cases only. -template -struct inplace_transpose_selector { // PacketSize x PacketSize - static void run(MatrixType& m) { - typedef typename MatrixType::Scalar Scalar; - typedef typename internal::packet_traits::type Packet; - const Index PacketSize = internal::packet_traits::size; - const Index Alignment = internal::evaluator::Alignment; - PacketBlock A; - for (Index i=0; i(i,0); - internal::ptranspose(A); - for (Index i=0; i(m.rowIndexByOuterInner(i,0), m.colIndexByOuterInner(i,0), A.packet[i]); - } -}; + // TODO: vectorized path is currently limited to LargestPacketSize x LargestPacketSize cases only. + template struct inplace_transpose_selector + {// PacketSize x PacketSize + static void run(MatrixType &m) + { + typedef typename MatrixType::Scalar Scalar; + typedef typename internal::packet_traits::type Packet; + const Index PacketSize = internal::packet_traits::size; + const Index Alignment = internal::evaluator::Alignment; + PacketBlock A; + for (Index i = 0; i < PacketSize; ++i) A.packet[i] = m.template packetByOuterInner(i, 0); + internal::ptranspose(A); + for (Index i = 0; i < PacketSize; ++i) + m.template writePacket(m.rowIndexByOuterInner(i, 0), m.colIndexByOuterInner(i, 0), A.packet[i]); + } + }; -template -struct inplace_transpose_selector { // non square matrix - static void run(MatrixType& m) { - if (m.rows()==m.cols()) - m.matrix().template triangularView().swap(m.matrix().transpose()); - else - m = m.transpose().eval(); - } -}; + template + struct inplace_transpose_selector + {// non square matrix + static void run(MatrixType &m) + { + if (m.rows() == m.cols()) + m.matrix().template triangularView().swap(m.matrix().transpose()); + else + m = m.transpose().eval(); + } + }; -} // end namespace internal +}// end namespace internal /** This is the "in place" version of transpose(): it replaces \c *this by its own transpose. - * Thus, doing - * \code - * m.transposeInPlace(); - * \endcode - * has the same effect on m as doing - * \code - * m = m.transpose().eval(); - * \endcode - * and is faster and also safer because in the latter line of code, forgetting the eval() results - * in a bug caused by \ref TopicAliasing "aliasing". - * - * Notice however that this method is only useful if you want to replace a matrix by its own transpose. - * If you just need the transpose of a matrix, use transpose(). - * - * \note if the matrix is not square, then \c *this must be a resizable matrix. - * This excludes (non-square) fixed-size matrices, block-expressions and maps. - * - * \sa transpose(), adjoint(), adjointInPlace() */ -template -inline void DenseBase::transposeInPlace() + * Thus, doing + * \code + * m.transposeInPlace(); + * \endcode + * has the same effect on m as doing + * \code + * m = m.transpose().eval(); + * \endcode + * and is faster and also safer because in the latter line of code, forgetting the eval() results + * in a bug caused by \ref TopicAliasing "aliasing". + * + * Notice however that this method is only useful if you want to replace a matrix by its own transpose. + * If you just need the transpose of a matrix, use transpose(). + * + * \note if the matrix is not square, then \c *this must be a resizable matrix. + * This excludes (non-square) fixed-size matrices, block-expressions and maps. + * + * \sa transpose(), adjoint(), adjointInPlace() */ +template inline void DenseBase::transposeInPlace() { eigen_assert((rows() == cols() || (RowsAtCompileTime == Dynamic && ColsAtCompileTime == Dynamic)) && "transposeInPlace() called on a non-square non-resizable matrix"); @@ -289,33 +269,29 @@ inline void DenseBase::transposeInPlace() } /*************************************************************************** -* "in place" adjoint implementation -***************************************************************************/ + * "in place" adjoint implementation + ***************************************************************************/ /** This is the "in place" version of adjoint(): it replaces \c *this by its own transpose. - * Thus, doing - * \code - * m.adjointInPlace(); - * \endcode - * has the same effect on m as doing - * \code - * m = m.adjoint().eval(); - * \endcode - * and is faster and also safer because in the latter line of code, forgetting the eval() results - * in a bug caused by aliasing. - * - * Notice however that this method is only useful if you want to replace a matrix by its own adjoint. - * If you just need the adjoint of a matrix, use adjoint(). - * - * \note if the matrix is not square, then \c *this must be a resizable matrix. - * This excludes (non-square) fixed-size matrices, block-expressions and maps. - * - * \sa transpose(), adjoint(), transposeInPlace() */ -template -inline void MatrixBase::adjointInPlace() -{ - derived() = adjoint().eval(); -} + * Thus, doing + * \code + * m.adjointInPlace(); + * \endcode + * has the same effect on m as doing + * \code + * m = m.adjoint().eval(); + * \endcode + * and is faster and also safer because in the latter line of code, forgetting the eval() results + * in a bug caused by aliasing. + * + * Notice however that this method is only useful if you want to replace a matrix by its own adjoint. + * If you just need the adjoint of a matrix, use adjoint(). + * + * \note if the matrix is not square, then \c *this must be a resizable matrix. + * This excludes (non-square) fixed-size matrices, block-expressions and maps. + * + * \sa transpose(), adjoint(), transposeInPlace() */ +template inline void MatrixBase::adjointInPlace() { derived() = adjoint().eval(); } #ifndef EIGEN_NO_DEBUG @@ -323,81 +299,78 @@ inline void MatrixBase::adjointInPlace() namespace internal { -template -struct check_transpose_aliasing_compile_time_selector -{ - enum { ret = bool(blas_traits::IsTransposed) != DestIsTransposed }; -}; + template struct check_transpose_aliasing_compile_time_selector + { + enum { ret = bool(blas_traits::IsTransposed) != DestIsTransposed }; + }; -template -struct check_transpose_aliasing_compile_time_selector > -{ - enum { ret = bool(blas_traits::IsTransposed) != DestIsTransposed - || bool(blas_traits::IsTransposed) != DestIsTransposed + template + struct check_transpose_aliasing_compile_time_selector> + { + enum { + ret = bool(blas_traits::IsTransposed) != DestIsTransposed + || bool(blas_traits::IsTransposed) != DestIsTransposed + }; }; -}; -template -struct check_transpose_aliasing_run_time_selector -{ - static bool run(const Scalar* dest, const OtherDerived& src) + template + struct check_transpose_aliasing_run_time_selector { - return (bool(blas_traits::IsTransposed) != DestIsTransposed) && (dest!=0 && dest==(const Scalar*)extract_data(src)); - } -}; + static bool run(const Scalar *dest, const OtherDerived &src) + { + return (bool(blas_traits::IsTransposed) != DestIsTransposed) + && (dest != 0 && dest == (const Scalar *)extract_data(src)); + } + }; -template -struct check_transpose_aliasing_run_time_selector > -{ - static bool run(const Scalar* dest, const CwiseBinaryOp& src) + template + struct check_transpose_aliasing_run_time_selector> { - return ((blas_traits::IsTransposed != DestIsTransposed) && (dest!=0 && dest==(const Scalar*)extract_data(src.lhs()))) - || ((blas_traits::IsTransposed != DestIsTransposed) && (dest!=0 && dest==(const Scalar*)extract_data(src.rhs()))); - } -}; + static bool run(const Scalar *dest, const CwiseBinaryOp &src) + { + return ((blas_traits::IsTransposed != DestIsTransposed) + && (dest != 0 && dest == (const Scalar *)extract_data(src.lhs()))) + || ((blas_traits::IsTransposed != DestIsTransposed) + && (dest != 0 && dest == (const Scalar *)extract_data(src.rhs()))); + } + }; -// the following selector, checkTransposeAliasing_impl, based on MightHaveTransposeAliasing, -// is because when the condition controlling the assert is known at compile time, ICC emits a warning. -// This is actually a good warning: in expressions that don't have any transposing, the condition is -// known at compile time to be false, and using that, we can avoid generating the code of the assert again -// and again for all these expressions that don't need it. - -template::IsTransposed,OtherDerived>::ret - > -struct checkTransposeAliasing_impl -{ - static void run(const Derived& dst, const OtherDerived& other) + // the following selector, checkTransposeAliasing_impl, based on MightHaveTransposeAliasing, + // is because when the condition controlling the assert is known at compile time, ICC emits a warning. + // This is actually a good warning: in expressions that don't have any transposing, the condition is + // known at compile time to be false, and using that, we can avoid generating the code of the assert again + // and again for all these expressions that don't need it. + + template::IsTransposed, OtherDerived>::ret> + struct checkTransposeAliasing_impl + { + static void run(const Derived &dst, const OtherDerived &other) { - eigen_assert((!check_transpose_aliasing_run_time_selector + eigen_assert((!check_transpose_aliasing_run_time_selector ::IsTransposed,OtherDerived> ::run(extract_data(dst), other)) && "aliasing detected during transposition, use transposeInPlace() " "or evaluate the rhs into a temporary using .eval()"); - } -}; + }; -template -struct checkTransposeAliasing_impl -{ - static void run(const Derived&, const OtherDerived&) - { - } -}; + template struct checkTransposeAliasing_impl + { + static void run(const Derived &, const OtherDerived &) {} + }; -template -void check_for_aliasing(const Dst &dst, const Src &src) -{ - internal::checkTransposeAliasing_impl::run(dst, src); -} + template void check_for_aliasing(const Dst &dst, const Src &src) + { + internal::checkTransposeAliasing_impl::run(dst, src); + } -} // end namespace internal +}// end namespace internal -#endif // EIGEN_NO_DEBUG +#endif// EIGEN_NO_DEBUG -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_TRANSPOSE_H +#endif// EIGEN_TRANSPOSE_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/Transpositions.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/Transpositions.h index 86da5af5..e1f68675 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/Transpositions.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/Transpositions.h @@ -10,398 +10,364 @@ #ifndef EIGEN_TRANSPOSITIONS_H #define EIGEN_TRANSPOSITIONS_H -namespace Eigen { +namespace Eigen { -template -class TranspositionsBase +template class TranspositionsBase { - typedef internal::traits Traits; - - public: - - typedef typename Traits::IndicesType IndicesType; - typedef typename IndicesType::Scalar StorageIndex; - typedef Eigen::Index Index; ///< \deprecated since Eigen 3.3 - - Derived& derived() { return *static_cast(this); } - const Derived& derived() const { return *static_cast(this); } - - /** Copies the \a other transpositions into \c *this */ - template - Derived& operator=(const TranspositionsBase& other) - { - indices() = other.indices(); - return derived(); - } - - #ifndef EIGEN_PARSED_BY_DOXYGEN - /** This is a special case of the templated operator=. Its purpose is to - * prevent a default operator= from hiding the templated operator=. - */ - Derived& operator=(const TranspositionsBase& other) - { - indices() = other.indices(); - return derived(); - } - #endif - - /** \returns the number of transpositions */ - Index size() const { return indices().size(); } - /** \returns the number of rows of the equivalent permutation matrix */ - Index rows() const { return indices().size(); } - /** \returns the number of columns of the equivalent permutation matrix */ - Index cols() const { return indices().size(); } - - /** Direct access to the underlying index vector */ - inline const StorageIndex& coeff(Index i) const { return indices().coeff(i); } - /** Direct access to the underlying index vector */ - inline StorageIndex& coeffRef(Index i) { return indices().coeffRef(i); } - /** Direct access to the underlying index vector */ - inline const StorageIndex& operator()(Index i) const { return indices()(i); } - /** Direct access to the underlying index vector */ - inline StorageIndex& operator()(Index i) { return indices()(i); } - /** Direct access to the underlying index vector */ - inline const StorageIndex& operator[](Index i) const { return indices()(i); } - /** Direct access to the underlying index vector */ - inline StorageIndex& operator[](Index i) { return indices()(i); } - - /** const version of indices(). */ - const IndicesType& indices() const { return derived().indices(); } - /** \returns a reference to the stored array representing the transpositions. */ - IndicesType& indices() { return derived().indices(); } - - /** Resizes to given size. */ - inline void resize(Index newSize) - { - indices().resize(newSize); - } - - /** Sets \c *this to represents an identity transformation */ - void setIdentity() - { - for(StorageIndex i = 0; i < indices().size(); ++i) - coeffRef(i) = i; - } - - // FIXME: do we want such methods ? - // might be usefull when the target matrix expression is complex, e.g.: - // object.matrix().block(..,..,..,..) = trans * object.matrix().block(..,..,..,..); - /* - template - void applyForwardToRows(MatrixType& mat) const - { - for(Index k=0 ; k - void applyBackwardToRows(MatrixType& mat) const - { - for(Index k=size()-1 ; k>=0 ; --k) - if(m_indices(k)!=k) - mat.row(k).swap(mat.row(m_indices(k))); - } - */ - - /** \returns the inverse transformation */ - inline Transpose inverse() const - { return Transpose(derived()); } - - /** \returns the tranpose transformation */ - inline Transpose transpose() const - { return Transpose(derived()); } - - protected: + typedef internal::traits Traits; + +public: + typedef typename Traits::IndicesType IndicesType; + typedef typename IndicesType::Scalar StorageIndex; + typedef Eigen::Index Index;///< \deprecated since Eigen 3.3 + + Derived &derived() { return *static_cast(this); } + const Derived &derived() const { return *static_cast(this); } + + /** Copies the \a other transpositions into \c *this */ + template Derived &operator=(const TranspositionsBase &other) + { + indices() = other.indices(); + return derived(); + } + +#ifndef EIGEN_PARSED_BY_DOXYGEN + /** This is a special case of the templated operator=. Its purpose is to + * prevent a default operator= from hiding the templated operator=. + */ + Derived &operator=(const TranspositionsBase &other) + { + indices() = other.indices(); + return derived(); + } +#endif + + /** \returns the number of transpositions */ + Index size() const { return indices().size(); } + /** \returns the number of rows of the equivalent permutation matrix */ + Index rows() const { return indices().size(); } + /** \returns the number of columns of the equivalent permutation matrix */ + Index cols() const { return indices().size(); } + + /** Direct access to the underlying index vector */ + inline const StorageIndex &coeff(Index i) const { return indices().coeff(i); } + /** Direct access to the underlying index vector */ + inline StorageIndex &coeffRef(Index i) { return indices().coeffRef(i); } + /** Direct access to the underlying index vector */ + inline const StorageIndex &operator()(Index i) const { return indices()(i); } + /** Direct access to the underlying index vector */ + inline StorageIndex &operator()(Index i) { return indices()(i); } + /** Direct access to the underlying index vector */ + inline const StorageIndex &operator[](Index i) const { return indices()(i); } + /** Direct access to the underlying index vector */ + inline StorageIndex &operator[](Index i) { return indices()(i); } + + /** const version of indices(). */ + const IndicesType &indices() const { return derived().indices(); } + /** \returns a reference to the stored array representing the transpositions. */ + IndicesType &indices() { return derived().indices(); } + + /** Resizes to given size. */ + inline void resize(Index newSize) { indices().resize(newSize); } + + /** Sets \c *this to represents an identity transformation */ + void setIdentity() + { + for (StorageIndex i = 0; i < indices().size(); ++i) coeffRef(i) = i; + } + + // FIXME: do we want such methods ? + // might be usefull when the target matrix expression is complex, e.g.: + // object.matrix().block(..,..,..,..) = trans * object.matrix().block(..,..,..,..); + /* + template + void applyForwardToRows(MatrixType& mat) const + { + for(Index k=0 ; k + void applyBackwardToRows(MatrixType& mat) const + { + for(Index k=size()-1 ; k>=0 ; --k) + if(m_indices(k)!=k) + mat.row(k).swap(mat.row(m_indices(k))); + } + */ + + /** \returns the inverse transformation */ + inline Transpose inverse() const { return Transpose(derived()); } + + /** \returns the tranpose transformation */ + inline Transpose transpose() const { return Transpose(derived()); } + +protected: }; namespace internal { -template -struct traits > - : traits > -{ - typedef Matrix<_StorageIndex, SizeAtCompileTime, 1, 0, MaxSizeAtCompileTime, 1> IndicesType; - typedef TranspositionsStorage StorageKind; -}; -} + template + struct traits> + : traits> + { + typedef Matrix<_StorageIndex, SizeAtCompileTime, 1, 0, MaxSizeAtCompileTime, 1> IndicesType; + typedef TranspositionsStorage StorageKind; + }; +}// namespace internal /** \class Transpositions - * \ingroup Core_Module - * - * \brief Represents a sequence of transpositions (row/column interchange) - * - * \tparam SizeAtCompileTime the number of transpositions, or Dynamic - * \tparam MaxSizeAtCompileTime the maximum number of transpositions, or Dynamic. This optional parameter defaults to SizeAtCompileTime. Most of the time, you should not have to specify it. - * - * This class represents a permutation transformation as a sequence of \em n transpositions - * \f$[T_{n-1} \ldots T_{i} \ldots T_{0}]\f$. It is internally stored as a vector of integers \c indices. - * Each transposition \f$ T_{i} \f$ applied on the left of a matrix (\f$ T_{i} M\f$) interchanges - * the rows \c i and \c indices[i] of the matrix \c M. - * A transposition applied on the right (e.g., \f$ M T_{i}\f$) yields a column interchange. - * - * Compared to the class PermutationMatrix, such a sequence of transpositions is what is - * computed during a decomposition with pivoting, and it is faster when applying the permutation in-place. - * - * To apply a sequence of transpositions to a matrix, simply use the operator * as in the following example: - * \code - * Transpositions tr; - * MatrixXf mat; - * mat = tr * mat; - * \endcode - * In this example, we detect that the matrix appears on both side, and so the transpositions - * are applied in-place without any temporary or extra copy. - * - * \sa class PermutationMatrix - */ + * \ingroup Core_Module + * + * \brief Represents a sequence of transpositions (row/column interchange) + * + * \tparam SizeAtCompileTime the number of transpositions, or Dynamic + * \tparam MaxSizeAtCompileTime the maximum number of transpositions, or Dynamic. This optional parameter defaults to + * SizeAtCompileTime. Most of the time, you should not have to specify it. + * + * This class represents a permutation transformation as a sequence of \em n transpositions + * \f$[T_{n-1} \ldots T_{i} \ldots T_{0}]\f$. It is internally stored as a vector of integers \c indices. + * Each transposition \f$ T_{i} \f$ applied on the left of a matrix (\f$ T_{i} M\f$) interchanges + * the rows \c i and \c indices[i] of the matrix \c M. + * A transposition applied on the right (e.g., \f$ M T_{i}\f$) yields a column interchange. + * + * Compared to the class PermutationMatrix, such a sequence of transpositions is what is + * computed during a decomposition with pivoting, and it is faster when applying the permutation in-place. + * + * To apply a sequence of transpositions to a matrix, simply use the operator * as in the following example: + * \code + * Transpositions tr; + * MatrixXf mat; + * mat = tr * mat; + * \endcode + * In this example, we detect that the matrix appears on both side, and so the transpositions + * are applied in-place without any temporary or extra copy. + * + * \sa class PermutationMatrix + */ template -class Transpositions : public TranspositionsBase > +class Transpositions : public TranspositionsBase> { - typedef internal::traits Traits; - public: - - typedef TranspositionsBase Base; - typedef typename Traits::IndicesType IndicesType; - typedef typename IndicesType::Scalar StorageIndex; - - inline Transpositions() {} - - /** Copy constructor. */ - template - inline Transpositions(const TranspositionsBase& other) - : m_indices(other.indices()) {} - - #ifndef EIGEN_PARSED_BY_DOXYGEN - /** Standard copy constructor. Defined only to prevent a default copy constructor - * from hiding the other templated constructor */ - inline Transpositions(const Transpositions& other) : m_indices(other.indices()) {} - #endif - - /** Generic constructor from expression of the transposition indices. */ - template - explicit inline Transpositions(const MatrixBase& indices) : m_indices(indices) - {} - - /** Copies the \a other transpositions into \c *this */ - template - Transpositions& operator=(const TranspositionsBase& other) - { - return Base::operator=(other); - } - - #ifndef EIGEN_PARSED_BY_DOXYGEN - /** This is a special case of the templated operator=. Its purpose is to - * prevent a default operator= from hiding the templated operator=. - */ - Transpositions& operator=(const Transpositions& other) - { - m_indices = other.m_indices; - return *this; - } - #endif - - /** Constructs an uninitialized permutation matrix of given size. - */ - inline Transpositions(Index size) : m_indices(size) - {} - - /** const version of indices(). */ - const IndicesType& indices() const { return m_indices; } - /** \returns a reference to the stored array representing the transpositions. */ - IndicesType& indices() { return m_indices; } - - protected: - - IndicesType m_indices; + typedef internal::traits Traits; + +public: + typedef TranspositionsBase Base; + typedef typename Traits::IndicesType IndicesType; + typedef typename IndicesType::Scalar StorageIndex; + + inline Transpositions() {} + + /** Copy constructor. */ + template + inline Transpositions(const TranspositionsBase &other) : m_indices(other.indices()) + {} + +#ifndef EIGEN_PARSED_BY_DOXYGEN + /** Standard copy constructor. Defined only to prevent a default copy constructor + * from hiding the other templated constructor */ + inline Transpositions(const Transpositions &other) : m_indices(other.indices()) {} +#endif + + /** Generic constructor from expression of the transposition indices. */ + template explicit inline Transpositions(const MatrixBase &indices) : m_indices(indices) {} + + /** Copies the \a other transpositions into \c *this */ + template Transpositions &operator=(const TranspositionsBase &other) + { + return Base::operator=(other); + } + +#ifndef EIGEN_PARSED_BY_DOXYGEN + /** This is a special case of the templated operator=. Its purpose is to + * prevent a default operator= from hiding the templated operator=. + */ + Transpositions &operator=(const Transpositions &other) + { + m_indices = other.m_indices; + return *this; + } +#endif + + /** Constructs an uninitialized permutation matrix of given size. + */ + inline Transpositions(Index size) : m_indices(size) {} + + /** const version of indices(). */ + const IndicesType &indices() const { return m_indices; } + /** \returns a reference to the stored array representing the transpositions. */ + IndicesType &indices() { return m_indices; } + +protected: + IndicesType m_indices; }; namespace internal { -template -struct traits,_PacketAccess> > - : traits > -{ - typedef Map, _PacketAccess> IndicesType; - typedef _StorageIndex StorageIndex; - typedef TranspositionsStorage StorageKind; -}; -} + template + struct traits, _PacketAccess>> + : traits> + { + typedef Map, _PacketAccess> + IndicesType; + typedef _StorageIndex StorageIndex; + typedef TranspositionsStorage StorageKind; + }; +}// namespace internal template -class Map,PacketAccess> - : public TranspositionsBase,PacketAccess> > +class Map, PacketAccess> + : public TranspositionsBase, PacketAccess>> { - typedef internal::traits Traits; - public: - - typedef TranspositionsBase Base; - typedef typename Traits::IndicesType IndicesType; - typedef typename IndicesType::Scalar StorageIndex; - - explicit inline Map(const StorageIndex* indicesPtr) - : m_indices(indicesPtr) - {} - - inline Map(const StorageIndex* indicesPtr, Index size) - : m_indices(indicesPtr,size) - {} - - /** Copies the \a other transpositions into \c *this */ - template - Map& operator=(const TranspositionsBase& other) - { - return Base::operator=(other); - } - - #ifndef EIGEN_PARSED_BY_DOXYGEN - /** This is a special case of the templated operator=. Its purpose is to - * prevent a default operator= from hiding the templated operator=. - */ - Map& operator=(const Map& other) - { - m_indices = other.m_indices; - return *this; - } - #endif - - /** const version of indices(). */ - const IndicesType& indices() const { return m_indices; } - - /** \returns a reference to the stored array representing the transpositions. */ - IndicesType& indices() { return m_indices; } - - protected: - - IndicesType m_indices; + typedef internal::traits Traits; + +public: + typedef TranspositionsBase Base; + typedef typename Traits::IndicesType IndicesType; + typedef typename IndicesType::Scalar StorageIndex; + + explicit inline Map(const StorageIndex *indicesPtr) : m_indices(indicesPtr) {} + + inline Map(const StorageIndex *indicesPtr, Index size) : m_indices(indicesPtr, size) {} + + /** Copies the \a other transpositions into \c *this */ + template Map &operator=(const TranspositionsBase &other) + { + return Base::operator=(other); + } + +#ifndef EIGEN_PARSED_BY_DOXYGEN + /** This is a special case of the templated operator=. Its purpose is to + * prevent a default operator= from hiding the templated operator=. + */ + Map &operator=(const Map &other) + { + m_indices = other.m_indices; + return *this; + } +#endif + + /** const version of indices(). */ + const IndicesType &indices() const { return m_indices; } + + /** \returns a reference to the stored array representing the transpositions. */ + IndicesType &indices() { return m_indices; } + +protected: + IndicesType m_indices; }; namespace internal { -template -struct traits > - : traits > -{ - typedef TranspositionsStorage StorageKind; -}; -} + template + struct traits> : traits> + { + typedef TranspositionsStorage StorageKind; + }; +}// namespace internal template -class TranspositionsWrapper - : public TranspositionsBase > +class TranspositionsWrapper : public TranspositionsBase> { - typedef internal::traits Traits; - public: - - typedef TranspositionsBase Base; - typedef typename Traits::IndicesType IndicesType; - typedef typename IndicesType::Scalar StorageIndex; - - explicit inline TranspositionsWrapper(IndicesType& indices) - : m_indices(indices) - {} - - /** Copies the \a other transpositions into \c *this */ - template - TranspositionsWrapper& operator=(const TranspositionsBase& other) - { - return Base::operator=(other); - } - - #ifndef EIGEN_PARSED_BY_DOXYGEN - /** This is a special case of the templated operator=. Its purpose is to - * prevent a default operator= from hiding the templated operator=. - */ - TranspositionsWrapper& operator=(const TranspositionsWrapper& other) - { - m_indices = other.m_indices; - return *this; - } - #endif - - /** const version of indices(). */ - const IndicesType& indices() const { return m_indices; } - - /** \returns a reference to the stored array representing the transpositions. */ - IndicesType& indices() { return m_indices; } - - protected: - - typename IndicesType::Nested m_indices; + typedef internal::traits Traits; + +public: + typedef TranspositionsBase Base; + typedef typename Traits::IndicesType IndicesType; + typedef typename IndicesType::Scalar StorageIndex; + + explicit inline TranspositionsWrapper(IndicesType &indices) : m_indices(indices) {} + + /** Copies the \a other transpositions into \c *this */ + template TranspositionsWrapper &operator=(const TranspositionsBase &other) + { + return Base::operator=(other); + } + +#ifndef EIGEN_PARSED_BY_DOXYGEN + /** This is a special case of the templated operator=. Its purpose is to + * prevent a default operator= from hiding the templated operator=. + */ + TranspositionsWrapper &operator=(const TranspositionsWrapper &other) + { + m_indices = other.m_indices; + return *this; + } +#endif + + /** const version of indices(). */ + const IndicesType &indices() const { return m_indices; } + + /** \returns a reference to the stored array representing the transpositions. */ + IndicesType &indices() { return m_indices; } + +protected: + typename IndicesType::Nested m_indices; }; - /** \returns the \a matrix with the \a transpositions applied to the columns. - */ + */ template -EIGEN_DEVICE_FUNC -const Product -operator*(const MatrixBase &matrix, - const TranspositionsBase& transpositions) +EIGEN_DEVICE_FUNC const Product + operator*(const MatrixBase &matrix, const TranspositionsBase &transpositions) { - return Product - (matrix.derived(), transpositions.derived()); + return Product(matrix.derived(), transpositions.derived()); } /** \returns the \a matrix with the \a transpositions applied to the rows. - */ + */ template -EIGEN_DEVICE_FUNC -const Product -operator*(const TranspositionsBase &transpositions, - const MatrixBase& matrix) +EIGEN_DEVICE_FUNC const Product + operator*(const TranspositionsBase &transpositions, const MatrixBase &matrix) { - return Product - (transpositions.derived(), matrix.derived()); + return Product(transpositions.derived(), matrix.derived()); } // Template partial specialization for transposed/inverse transpositions namespace internal { -template -struct traits > > - : traits -{}; + template struct traits>> : traits + { + }; -} // end namespace internal +}// end namespace internal -template -class Transpose > +template class Transpose> { - typedef TranspositionsDerived TranspositionType; - typedef typename TranspositionType::IndicesType IndicesType; - public: - - explicit Transpose(const TranspositionType& t) : m_transpositions(t) {} - - Index size() const { return m_transpositions.size(); } - Index rows() const { return m_transpositions.size(); } - Index cols() const { return m_transpositions.size(); } - - /** \returns the \a matrix with the inverse transpositions applied to the columns. - */ - template friend - const Product - operator*(const MatrixBase& matrix, const Transpose& trt) - { - return Product(matrix.derived(), trt); - } - - /** \returns the \a matrix with the inverse transpositions applied to the rows. - */ - template - const Product - operator*(const MatrixBase& matrix) const - { - return Product(*this, matrix.derived()); - } - - const TranspositionType& nestedExpression() const { return m_transpositions; } - - protected: - const TranspositionType& m_transpositions; + typedef TranspositionsDerived TranspositionType; + typedef typename TranspositionType::IndicesType IndicesType; + +public: + explicit Transpose(const TranspositionType &t) : m_transpositions(t) {} + + Index size() const { return m_transpositions.size(); } + Index rows() const { return m_transpositions.size(); } + Index cols() const { return m_transpositions.size(); } + + /** \returns the \a matrix with the inverse transpositions applied to the columns. + */ + template + friend const Product operator*(const MatrixBase &matrix, + const Transpose &trt) + { + return Product(matrix.derived(), trt); + } + + /** \returns the \a matrix with the inverse transpositions applied to the rows. + */ + template + const Product operator*(const MatrixBase &matrix) const + { + return Product(*this, matrix.derived()); + } + + const TranspositionType &nestedExpression() const { return m_transpositions; } + +protected: + const TranspositionType &m_transpositions; }; -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_TRANSPOSITIONS_H +#endif// EIGEN_TRANSPOSITIONS_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/TriangularMatrix.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/TriangularMatrix.h index 667ef09d..e6d0d562 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/TriangularMatrix.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/TriangularMatrix.h @@ -11,570 +11,537 @@ #ifndef EIGEN_TRIANGULARMATRIX_H #define EIGEN_TRIANGULARMATRIX_H -namespace Eigen { +namespace Eigen { namespace internal { - -template struct triangular_solve_retval; - + + template struct triangular_solve_retval; + } /** \class TriangularBase - * \ingroup Core_Module - * - * \brief Base class for triangular part in a matrix - */ + * \ingroup Core_Module + * + * \brief Base class for triangular part in a matrix + */ template class TriangularBase : public EigenBase { - public: +public: + enum { + Mode = internal::traits::Mode, + RowsAtCompileTime = internal::traits::RowsAtCompileTime, + ColsAtCompileTime = internal::traits::ColsAtCompileTime, + MaxRowsAtCompileTime = internal::traits::MaxRowsAtCompileTime, + MaxColsAtCompileTime = internal::traits::MaxColsAtCompileTime, - enum { - Mode = internal::traits::Mode, - RowsAtCompileTime = internal::traits::RowsAtCompileTime, - ColsAtCompileTime = internal::traits::ColsAtCompileTime, - MaxRowsAtCompileTime = internal::traits::MaxRowsAtCompileTime, - MaxColsAtCompileTime = internal::traits::MaxColsAtCompileTime, - - SizeAtCompileTime = (internal::size_at_compile_time::RowsAtCompileTime, - internal::traits::ColsAtCompileTime>::ret), - /**< This is equal to the number of coefficients, i.e. the number of - * rows times the number of columns, or to \a Dynamic if this is not - * known at compile-time. \sa RowsAtCompileTime, ColsAtCompileTime */ - - MaxSizeAtCompileTime = (internal::size_at_compile_time::MaxRowsAtCompileTime, - internal::traits::MaxColsAtCompileTime>::ret) - - }; - typedef typename internal::traits::Scalar Scalar; - typedef typename internal::traits::StorageKind StorageKind; - typedef typename internal::traits::StorageIndex StorageIndex; - typedef typename internal::traits::FullMatrixType DenseMatrixType; - typedef DenseMatrixType DenseType; - typedef Derived const& Nested; + SizeAtCompileTime = (internal::size_at_compile_time::RowsAtCompileTime, + internal::traits::ColsAtCompileTime>::ret), + /**< This is equal to the number of coefficients, i.e. the number of + * rows times the number of columns, or to \a Dynamic if this is not + * known at compile-time. \sa RowsAtCompileTime, ColsAtCompileTime */ - EIGEN_DEVICE_FUNC - inline TriangularBase() { eigen_assert(!((Mode&UnitDiag) && (Mode&ZeroDiag))); } + MaxSizeAtCompileTime = (internal::size_at_compile_time::MaxRowsAtCompileTime, + internal::traits::MaxColsAtCompileTime>::ret) - EIGEN_DEVICE_FUNC - inline Index rows() const { return derived().rows(); } - EIGEN_DEVICE_FUNC - inline Index cols() const { return derived().cols(); } - EIGEN_DEVICE_FUNC - inline Index outerStride() const { return derived().outerStride(); } - EIGEN_DEVICE_FUNC - inline Index innerStride() const { return derived().innerStride(); } - - // dummy resize function - void resize(Index rows, Index cols) - { - EIGEN_UNUSED_VARIABLE(rows); - EIGEN_UNUSED_VARIABLE(cols); - eigen_assert(rows==this->rows() && cols==this->cols()); - } + }; + typedef typename internal::traits::Scalar Scalar; + typedef typename internal::traits::StorageKind StorageKind; + typedef typename internal::traits::StorageIndex StorageIndex; + typedef typename internal::traits::FullMatrixType DenseMatrixType; + typedef DenseMatrixType DenseType; + typedef Derived const &Nested; - EIGEN_DEVICE_FUNC - inline Scalar coeff(Index row, Index col) const { return derived().coeff(row,col); } - EIGEN_DEVICE_FUNC - inline Scalar& coeffRef(Index row, Index col) { return derived().coeffRef(row,col); } + EIGEN_DEVICE_FUNC + inline TriangularBase() { eigen_assert(!((Mode & UnitDiag) && (Mode & ZeroDiag))); } - /** \see MatrixBase::copyCoeff(row,col) - */ - template - EIGEN_DEVICE_FUNC - EIGEN_STRONG_INLINE void copyCoeff(Index row, Index col, Other& other) - { - derived().coeffRef(row, col) = other.coeff(row, col); - } + EIGEN_DEVICE_FUNC + inline Index rows() const { return derived().rows(); } + EIGEN_DEVICE_FUNC + inline Index cols() const { return derived().cols(); } + EIGEN_DEVICE_FUNC + inline Index outerStride() const { return derived().outerStride(); } + EIGEN_DEVICE_FUNC + inline Index innerStride() const { return derived().innerStride(); } - EIGEN_DEVICE_FUNC - inline Scalar operator()(Index row, Index col) const - { - check_coordinates(row, col); - return coeff(row,col); - } - EIGEN_DEVICE_FUNC - inline Scalar& operator()(Index row, Index col) - { - check_coordinates(row, col); - return coeffRef(row,col); - } + // dummy resize function + void resize(Index rows, Index cols) + { + EIGEN_UNUSED_VARIABLE(rows); + EIGEN_UNUSED_VARIABLE(cols); + eigen_assert(rows == this->rows() && cols == this->cols()); + } - #ifndef EIGEN_PARSED_BY_DOXYGEN - EIGEN_DEVICE_FUNC - inline const Derived& derived() const { return *static_cast(this); } - EIGEN_DEVICE_FUNC - inline Derived& derived() { return *static_cast(this); } - #endif // not EIGEN_PARSED_BY_DOXYGEN + EIGEN_DEVICE_FUNC + inline Scalar coeff(Index row, Index col) const { return derived().coeff(row, col); } + EIGEN_DEVICE_FUNC + inline Scalar &coeffRef(Index row, Index col) { return derived().coeffRef(row, col); } - template - EIGEN_DEVICE_FUNC - void evalTo(MatrixBase &other) const; - template - EIGEN_DEVICE_FUNC - void evalToLazy(MatrixBase &other) const; + /** \see MatrixBase::copyCoeff(row,col) + */ + template EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void copyCoeff(Index row, Index col, Other &other) + { + derived().coeffRef(row, col) = other.coeff(row, col); + } - EIGEN_DEVICE_FUNC - DenseMatrixType toDenseMatrix() const - { - DenseMatrixType res(rows(), cols()); - evalToLazy(res); - return res; - } + EIGEN_DEVICE_FUNC + inline Scalar operator()(Index row, Index col) const + { + check_coordinates(row, col); + return coeff(row, col); + } + EIGEN_DEVICE_FUNC + inline Scalar &operator()(Index row, Index col) + { + check_coordinates(row, col); + return coeffRef(row, col); + } - protected: +#ifndef EIGEN_PARSED_BY_DOXYGEN + EIGEN_DEVICE_FUNC + inline const Derived &derived() const { return *static_cast(this); } + EIGEN_DEVICE_FUNC + inline Derived &derived() { return *static_cast(this); } +#endif// not EIGEN_PARSED_BY_DOXYGEN - void check_coordinates(Index row, Index col) const - { - EIGEN_ONLY_USED_FOR_DEBUG(row); - EIGEN_ONLY_USED_FOR_DEBUG(col); - eigen_assert(col>=0 && col=0 && row=row) - || (mode==Lower && col<=row) - || ((mode==StrictlyUpper || mode==UnitUpper) && col>row) - || ((mode==StrictlyLower || mode==UnitLower) && col EIGEN_DEVICE_FUNC void evalTo(MatrixBase &other) const; + template EIGEN_DEVICE_FUNC void evalToLazy(MatrixBase &other) const; - #ifdef EIGEN_INTERNAL_DEBUGGING - void check_coordinates_internal(Index row, Index col) const - { - check_coordinates(row, col); - } - #else - void check_coordinates_internal(Index , Index ) const {} - #endif + EIGEN_DEVICE_FUNC + DenseMatrixType toDenseMatrix() const + { + DenseMatrixType res(rows(), cols()); + evalToLazy(res); + return res; + } + +protected: + void check_coordinates(Index row, Index col) const + { + EIGEN_ONLY_USED_FOR_DEBUG(row); + EIGEN_ONLY_USED_FOR_DEBUG(col); + eigen_assert(col >= 0 && col < cols() && row >= 0 && row < rows()); + const int mode = int(Mode) & ~SelfAdjoint; + EIGEN_ONLY_USED_FOR_DEBUG(mode); + eigen_assert((mode == Upper && col >= row) || (mode == Lower && col <= row) + || ((mode == StrictlyUpper || mode == UnitUpper) && col > row) + || ((mode == StrictlyLower || mode == UnitLower) && col < row)); + } +#ifdef EIGEN_INTERNAL_DEBUGGING + void check_coordinates_internal(Index row, Index col) const { check_coordinates(row, col); } +#else + void check_coordinates_internal(Index, Index) const {} +#endif }; /** \class TriangularView - * \ingroup Core_Module - * - * \brief Expression of a triangular part in a matrix - * - * \param MatrixType the type of the object in which we are taking the triangular part - * \param Mode the kind of triangular matrix expression to construct. Can be #Upper, - * #Lower, #UnitUpper, #UnitLower, #StrictlyUpper, or #StrictlyLower. - * This is in fact a bit field; it must have either #Upper or #Lower, - * and additionally it may have #UnitDiag or #ZeroDiag or neither. - * - * This class represents a triangular part of a matrix, not necessarily square. Strictly speaking, for rectangular - * matrices one should speak of "trapezoid" parts. This class is the return type - * of MatrixBase::triangularView() and SparseMatrixBase::triangularView(), and most of the time this is the only way it is used. - * - * \sa MatrixBase::triangularView() - */ + * \ingroup Core_Module + * + * \brief Expression of a triangular part in a matrix + * + * \param MatrixType the type of the object in which we are taking the triangular part + * \param Mode the kind of triangular matrix expression to construct. Can be #Upper, + * #Lower, #UnitUpper, #UnitLower, #StrictlyUpper, or #StrictlyLower. + * This is in fact a bit field; it must have either #Upper or #Lower, + * and additionally it may have #UnitDiag or #ZeroDiag or neither. + * + * This class represents a triangular part of a matrix, not necessarily square. Strictly speaking, for rectangular + * matrices one should speak of "trapezoid" parts. This class is the return type + * of MatrixBase::triangularView() and SparseMatrixBase::triangularView(), and most of the time this is the only way it + * is used. + * + * \sa MatrixBase::triangularView() + */ namespace internal { -template -struct traits > : traits -{ - typedef typename ref_selector::non_const_type MatrixTypeNested; - typedef typename remove_reference::type MatrixTypeNestedNonRef; - typedef typename remove_all::type MatrixTypeNestedCleaned; - typedef typename MatrixType::PlainObject FullMatrixType; - typedef MatrixType ExpressionType; - enum { - Mode = _Mode, - FlagsLvalueBit = is_lvalue::value ? LvalueBit : 0, - Flags = (MatrixTypeNestedCleaned::Flags & (HereditaryBits | FlagsLvalueBit) & (~(PacketAccessBit | DirectAccessBit | LinearAccessBit))) + template + struct traits> : traits + { + typedef typename ref_selector::non_const_type MatrixTypeNested; + typedef typename remove_reference::type MatrixTypeNestedNonRef; + typedef typename remove_all::type MatrixTypeNestedCleaned; + typedef typename MatrixType::PlainObject FullMatrixType; + typedef MatrixType ExpressionType; + enum { + Mode = _Mode, + FlagsLvalueBit = is_lvalue::value ? LvalueBit : 0, + Flags = (MatrixTypeNestedCleaned::Flags & (HereditaryBits | FlagsLvalueBit) + & (~(PacketAccessBit | DirectAccessBit | LinearAccessBit))) + }; }; -}; -} +}// namespace internal template class TriangularViewImpl; -template class TriangularView - : public TriangularViewImpl<_MatrixType, _Mode, typename internal::traits<_MatrixType>::StorageKind > +template +class TriangularView + : public TriangularViewImpl<_MatrixType, _Mode, typename internal::traits<_MatrixType>::StorageKind> { - public: +public: + typedef TriangularViewImpl<_MatrixType, _Mode, typename internal::traits<_MatrixType>::StorageKind> Base; + typedef typename internal::traits::Scalar Scalar; + typedef _MatrixType MatrixType; - typedef TriangularViewImpl<_MatrixType, _Mode, typename internal::traits<_MatrixType>::StorageKind > Base; - typedef typename internal::traits::Scalar Scalar; - typedef _MatrixType MatrixType; +protected: + typedef typename internal::traits::MatrixTypeNested MatrixTypeNested; + typedef typename internal::traits::MatrixTypeNestedNonRef MatrixTypeNestedNonRef; - protected: - typedef typename internal::traits::MatrixTypeNested MatrixTypeNested; - typedef typename internal::traits::MatrixTypeNestedNonRef MatrixTypeNestedNonRef; + typedef typename internal::remove_all::type MatrixConjugateReturnType; - typedef typename internal::remove_all::type MatrixConjugateReturnType; - - public: +public: + typedef typename internal::traits::StorageKind StorageKind; + typedef typename internal::traits::MatrixTypeNestedCleaned NestedExpression; - typedef typename internal::traits::StorageKind StorageKind; - typedef typename internal::traits::MatrixTypeNestedCleaned NestedExpression; + enum { + Mode = _Mode, + Flags = internal::traits::Flags, + TransposeMode = (Mode & Upper ? Lower : 0) | (Mode & Lower ? Upper : 0) | (Mode & (UnitDiag)) | (Mode & (ZeroDiag)), + IsVectorAtCompileTime = false + }; - enum { - Mode = _Mode, - Flags = internal::traits::Flags, - TransposeMode = (Mode & Upper ? Lower : 0) - | (Mode & Lower ? Upper : 0) - | (Mode & (UnitDiag)) - | (Mode & (ZeroDiag)), - IsVectorAtCompileTime = false - }; + EIGEN_DEVICE_FUNC + explicit inline TriangularView(MatrixType &matrix) : m_matrix(matrix) {} - EIGEN_DEVICE_FUNC - explicit inline TriangularView(MatrixType& matrix) : m_matrix(matrix) - {} - - using Base::operator=; - TriangularView& operator=(const TriangularView &other) - { return Base::operator=(other); } + using Base::operator=; + TriangularView &operator=(const TriangularView &other) { return Base::operator=(other); } - /** \copydoc EigenBase::rows() */ - EIGEN_DEVICE_FUNC - inline Index rows() const { return m_matrix.rows(); } - /** \copydoc EigenBase::cols() */ - EIGEN_DEVICE_FUNC - inline Index cols() const { return m_matrix.cols(); } + /** \copydoc EigenBase::rows() */ + EIGEN_DEVICE_FUNC + inline Index rows() const { return m_matrix.rows(); } + /** \copydoc EigenBase::cols() */ + EIGEN_DEVICE_FUNC + inline Index cols() const { return m_matrix.cols(); } - /** \returns a const reference to the nested expression */ - EIGEN_DEVICE_FUNC - const NestedExpression& nestedExpression() const { return m_matrix; } + /** \returns a const reference to the nested expression */ + EIGEN_DEVICE_FUNC + const NestedExpression &nestedExpression() const { return m_matrix; } - /** \returns a reference to the nested expression */ - EIGEN_DEVICE_FUNC - NestedExpression& nestedExpression() { return m_matrix; } - - typedef TriangularView ConjugateReturnType; - /** \sa MatrixBase::conjugate() const */ - EIGEN_DEVICE_FUNC - inline const ConjugateReturnType conjugate() const - { return ConjugateReturnType(m_matrix.conjugate()); } + /** \returns a reference to the nested expression */ + EIGEN_DEVICE_FUNC + NestedExpression &nestedExpression() { return m_matrix; } - typedef TriangularView AdjointReturnType; - /** \sa MatrixBase::adjoint() const */ - EIGEN_DEVICE_FUNC - inline const AdjointReturnType adjoint() const - { return AdjointReturnType(m_matrix.adjoint()); } + typedef TriangularView ConjugateReturnType; + /** \sa MatrixBase::conjugate() const */ + EIGEN_DEVICE_FUNC + inline const ConjugateReturnType conjugate() const { return ConjugateReturnType(m_matrix.conjugate()); } - typedef TriangularView TransposeReturnType; - /** \sa MatrixBase::transpose() */ - EIGEN_DEVICE_FUNC - inline TransposeReturnType transpose() - { - EIGEN_STATIC_ASSERT_LVALUE(MatrixType) - typename MatrixType::TransposeReturnType tmp(m_matrix); - return TransposeReturnType(tmp); - } - - typedef TriangularView ConstTransposeReturnType; - /** \sa MatrixBase::transpose() const */ - EIGEN_DEVICE_FUNC - inline const ConstTransposeReturnType transpose() const - { - return ConstTransposeReturnType(m_matrix.transpose()); - } + typedef TriangularView AdjointReturnType; + /** \sa MatrixBase::adjoint() const */ + EIGEN_DEVICE_FUNC + inline const AdjointReturnType adjoint() const { return AdjointReturnType(m_matrix.adjoint()); } - template - EIGEN_DEVICE_FUNC - inline const Solve - solve(const MatrixBase& other) const - { return Solve(*this, other.derived()); } - - // workaround MSVC ICE - #if EIGEN_COMP_MSVC - template - EIGEN_DEVICE_FUNC - inline const internal::triangular_solve_retval - solve(const MatrixBase& other) const - { return Base::template solve(other); } - #else - using Base::solve; - #endif - - /** \returns a selfadjoint view of the referenced triangular part which must be either \c #Upper or \c #Lower. - * - * This is a shortcut for \code this->nestedExpression().selfadjointView<(*this)::Mode>() \endcode - * \sa MatrixBase::selfadjointView() */ - EIGEN_DEVICE_FUNC - SelfAdjointView selfadjointView() - { - EIGEN_STATIC_ASSERT((Mode&(UnitDiag|ZeroDiag))==0,PROGRAMMING_ERROR); - return SelfAdjointView(m_matrix); - } + typedef TriangularView TransposeReturnType; + /** \sa MatrixBase::transpose() */ + EIGEN_DEVICE_FUNC + inline TransposeReturnType transpose() + { + EIGEN_STATIC_ASSERT_LVALUE(MatrixType) + typename MatrixType::TransposeReturnType tmp(m_matrix); + return TransposeReturnType(tmp); + } - /** This is the const version of selfadjointView() */ - EIGEN_DEVICE_FUNC - const SelfAdjointView selfadjointView() const - { - EIGEN_STATIC_ASSERT((Mode&(UnitDiag|ZeroDiag))==0,PROGRAMMING_ERROR); - return SelfAdjointView(m_matrix); - } + typedef TriangularView ConstTransposeReturnType; + /** \sa MatrixBase::transpose() const */ + EIGEN_DEVICE_FUNC + inline const ConstTransposeReturnType transpose() const { return ConstTransposeReturnType(m_matrix.transpose()); } + template + EIGEN_DEVICE_FUNC inline const Solve solve(const MatrixBase &other) const + { + return Solve(*this, other.derived()); + } - /** \returns the determinant of the triangular matrix - * \sa MatrixBase::determinant() */ - EIGEN_DEVICE_FUNC - Scalar determinant() const - { - if (Mode & UnitDiag) - return 1; - else if (Mode & ZeroDiag) - return 0; - else - return m_matrix.diagonal().prod(); - } - - protected: +// workaround MSVC ICE +#if EIGEN_COMP_MSVC + template + EIGEN_DEVICE_FUNC inline const internal::triangular_solve_retval solve( + const MatrixBase &other) const + { + return Base::template solve(other); + } +#else + using Base::solve; +#endif + + /** \returns a selfadjoint view of the referenced triangular part which must be either \c #Upper or \c #Lower. + * + * This is a shortcut for \code this->nestedExpression().selfadjointView<(*this)::Mode>() \endcode + * \sa MatrixBase::selfadjointView() */ + EIGEN_DEVICE_FUNC + SelfAdjointView selfadjointView() + { + EIGEN_STATIC_ASSERT((Mode & (UnitDiag | ZeroDiag)) == 0, PROGRAMMING_ERROR); + return SelfAdjointView(m_matrix); + } + + /** This is the const version of selfadjointView() */ + EIGEN_DEVICE_FUNC + const SelfAdjointView selfadjointView() const + { + EIGEN_STATIC_ASSERT((Mode & (UnitDiag | ZeroDiag)) == 0, PROGRAMMING_ERROR); + return SelfAdjointView(m_matrix); + } + + + /** \returns the determinant of the triangular matrix + * \sa MatrixBase::determinant() */ + EIGEN_DEVICE_FUNC + Scalar determinant() const + { + if (Mode & UnitDiag) + return 1; + else if (Mode & ZeroDiag) + return 0; + else + return m_matrix.diagonal().prod(); + } - MatrixTypeNested m_matrix; +protected: + MatrixTypeNested m_matrix; }; /** \ingroup Core_Module - * - * \brief Base class for a triangular part in a \b dense matrix - * - * This class is an abstract base class of class TriangularView, and objects of type TriangularViewImpl cannot be instantiated. - * It extends class TriangularView with additional methods which available for dense expressions only. - * - * \sa class TriangularView, MatrixBase::triangularView() - */ -template class TriangularViewImpl<_MatrixType,_Mode,Dense> - : public TriangularBase > + * + * \brief Base class for a triangular part in a \b dense matrix + * + * This class is an abstract base class of class TriangularView, and objects of type TriangularViewImpl cannot be + * instantiated. It extends class TriangularView with additional methods which available for dense expressions only. + * + * \sa class TriangularView, MatrixBase::triangularView() + */ +template +class TriangularViewImpl<_MatrixType, _Mode, Dense> : public TriangularBase> { - public: +public: + typedef TriangularView<_MatrixType, _Mode> TriangularViewType; + typedef TriangularBase Base; + typedef typename internal::traits::Scalar Scalar; - typedef TriangularView<_MatrixType, _Mode> TriangularViewType; - typedef TriangularBase Base; - typedef typename internal::traits::Scalar Scalar; + typedef _MatrixType MatrixType; + typedef typename MatrixType::PlainObject DenseMatrixType; + typedef DenseMatrixType PlainObject; - typedef _MatrixType MatrixType; - typedef typename MatrixType::PlainObject DenseMatrixType; - typedef DenseMatrixType PlainObject; +public: + using Base::evalToLazy; + using Base::derived; - public: - using Base::evalToLazy; - using Base::derived; + typedef typename internal::traits::StorageKind StorageKind; - typedef typename internal::traits::StorageKind StorageKind; + enum { Mode = _Mode, Flags = internal::traits::Flags }; - enum { - Mode = _Mode, - Flags = internal::traits::Flags - }; + /** \returns the outer-stride of the underlying dense matrix + * \sa DenseCoeffsBase::outerStride() */ + EIGEN_DEVICE_FUNC + inline Index outerStride() const { return derived().nestedExpression().outerStride(); } + /** \returns the inner-stride of the underlying dense matrix + * \sa DenseCoeffsBase::innerStride() */ + EIGEN_DEVICE_FUNC + inline Index innerStride() const { return derived().nestedExpression().innerStride(); } - /** \returns the outer-stride of the underlying dense matrix - * \sa DenseCoeffsBase::outerStride() */ - EIGEN_DEVICE_FUNC - inline Index outerStride() const { return derived().nestedExpression().outerStride(); } - /** \returns the inner-stride of the underlying dense matrix - * \sa DenseCoeffsBase::innerStride() */ - EIGEN_DEVICE_FUNC - inline Index innerStride() const { return derived().nestedExpression().innerStride(); } + /** \sa MatrixBase::operator+=() */ + template EIGEN_DEVICE_FUNC TriangularViewType &operator+=(const DenseBase &other) + { + internal::call_assignment_no_alias( + derived(), other.derived(), internal::add_assign_op()); + return derived(); + } + /** \sa MatrixBase::operator-=() */ + template EIGEN_DEVICE_FUNC TriangularViewType &operator-=(const DenseBase &other) + { + internal::call_assignment_no_alias( + derived(), other.derived(), internal::sub_assign_op()); + return derived(); + } - /** \sa MatrixBase::operator+=() */ - template - EIGEN_DEVICE_FUNC - TriangularViewType& operator+=(const DenseBase& other) { - internal::call_assignment_no_alias(derived(), other.derived(), internal::add_assign_op()); - return derived(); - } - /** \sa MatrixBase::operator-=() */ - template - EIGEN_DEVICE_FUNC - TriangularViewType& operator-=(const DenseBase& other) { - internal::call_assignment_no_alias(derived(), other.derived(), internal::sub_assign_op()); - return derived(); - } - - /** \sa MatrixBase::operator*=() */ - EIGEN_DEVICE_FUNC - TriangularViewType& operator*=(const typename internal::traits::Scalar& other) { return *this = derived().nestedExpression() * other; } - /** \sa DenseBase::operator/=() */ - EIGEN_DEVICE_FUNC - TriangularViewType& operator/=(const typename internal::traits::Scalar& other) { return *this = derived().nestedExpression() / other; } + /** \sa MatrixBase::operator*=() */ + EIGEN_DEVICE_FUNC + TriangularViewType &operator*=(const typename internal::traits::Scalar &other) + { + return *this = derived().nestedExpression() * other; + } + /** \sa DenseBase::operator/=() */ + EIGEN_DEVICE_FUNC + TriangularViewType &operator/=(const typename internal::traits::Scalar &other) + { + return *this = derived().nestedExpression() / other; + } - /** \sa MatrixBase::fill() */ - EIGEN_DEVICE_FUNC - void fill(const Scalar& value) { setConstant(value); } - /** \sa MatrixBase::setConstant() */ - EIGEN_DEVICE_FUNC - TriangularViewType& setConstant(const Scalar& value) - { return *this = MatrixType::Constant(derived().rows(), derived().cols(), value); } - /** \sa MatrixBase::setZero() */ - EIGEN_DEVICE_FUNC - TriangularViewType& setZero() { return setConstant(Scalar(0)); } - /** \sa MatrixBase::setOnes() */ - EIGEN_DEVICE_FUNC - TriangularViewType& setOnes() { return setConstant(Scalar(1)); } + /** \sa MatrixBase::fill() */ + EIGEN_DEVICE_FUNC + void fill(const Scalar &value) { setConstant(value); } + /** \sa MatrixBase::setConstant() */ + EIGEN_DEVICE_FUNC + TriangularViewType &setConstant(const Scalar &value) + { + return *this = MatrixType::Constant(derived().rows(), derived().cols(), value); + } + /** \sa MatrixBase::setZero() */ + EIGEN_DEVICE_FUNC + TriangularViewType &setZero() { return setConstant(Scalar(0)); } + /** \sa MatrixBase::setOnes() */ + EIGEN_DEVICE_FUNC + TriangularViewType &setOnes() { return setConstant(Scalar(1)); } - /** \sa MatrixBase::coeff() - * \warning the coordinates must fit into the referenced triangular part - */ - EIGEN_DEVICE_FUNC - inline Scalar coeff(Index row, Index col) const - { - Base::check_coordinates_internal(row, col); - return derived().nestedExpression().coeff(row, col); - } + /** \sa MatrixBase::coeff() + * \warning the coordinates must fit into the referenced triangular part + */ + EIGEN_DEVICE_FUNC + inline Scalar coeff(Index row, Index col) const + { + Base::check_coordinates_internal(row, col); + return derived().nestedExpression().coeff(row, col); + } - /** \sa MatrixBase::coeffRef() - * \warning the coordinates must fit into the referenced triangular part - */ - EIGEN_DEVICE_FUNC - inline Scalar& coeffRef(Index row, Index col) - { - EIGEN_STATIC_ASSERT_LVALUE(TriangularViewType); - Base::check_coordinates_internal(row, col); - return derived().nestedExpression().coeffRef(row, col); - } + /** \sa MatrixBase::coeffRef() + * \warning the coordinates must fit into the referenced triangular part + */ + EIGEN_DEVICE_FUNC + inline Scalar &coeffRef(Index row, Index col) + { + EIGEN_STATIC_ASSERT_LVALUE(TriangularViewType); + Base::check_coordinates_internal(row, col); + return derived().nestedExpression().coeffRef(row, col); + } - /** Assigns a triangular matrix to a triangular part of a dense matrix */ - template - EIGEN_DEVICE_FUNC - TriangularViewType& operator=(const TriangularBase& other); + /** Assigns a triangular matrix to a triangular part of a dense matrix */ + template + EIGEN_DEVICE_FUNC TriangularViewType &operator=(const TriangularBase &other); - /** Shortcut for\code *this = other.other.triangularView<(*this)::Mode>() \endcode */ - template - EIGEN_DEVICE_FUNC - TriangularViewType& operator=(const MatrixBase& other); + /** Shortcut for\code *this = other.other.triangularView<(*this)::Mode>() \endcode */ + template + EIGEN_DEVICE_FUNC TriangularViewType &operator=(const MatrixBase &other); #ifndef EIGEN_PARSED_BY_DOXYGEN - EIGEN_DEVICE_FUNC - TriangularViewType& operator=(const TriangularViewImpl& other) - { return *this = other.derived().nestedExpression(); } + EIGEN_DEVICE_FUNC + TriangularViewType &operator=(const TriangularViewImpl &other) { return *this = other.derived().nestedExpression(); } - /** \deprecated */ - template - EIGEN_DEVICE_FUNC - void lazyAssign(const TriangularBase& other); + /** \deprecated */ + template EIGEN_DEVICE_FUNC void lazyAssign(const TriangularBase &other); - /** \deprecated */ - template - EIGEN_DEVICE_FUNC - void lazyAssign(const MatrixBase& other); + /** \deprecated */ + template EIGEN_DEVICE_FUNC void lazyAssign(const MatrixBase &other); #endif - /** Efficient triangular matrix times vector/matrix product */ - template - EIGEN_DEVICE_FUNC - const Product - operator*(const MatrixBase& rhs) const - { - return Product(derived(), rhs.derived()); - } - - /** Efficient vector/matrix times triangular matrix product */ - template friend - EIGEN_DEVICE_FUNC - const Product - operator*(const MatrixBase& lhs, const TriangularViewImpl& rhs) - { - return Product(lhs.derived(),rhs.derived()); - } + /** Efficient triangular matrix times vector/matrix product */ + template + EIGEN_DEVICE_FUNC const Product operator*(const MatrixBase &rhs) const + { + return Product(derived(), rhs.derived()); + } - /** \returns the product of the inverse of \c *this with \a other, \a *this being triangular. - * - * This function computes the inverse-matrix matrix product inverse(\c *this) * \a other if - * \a Side==OnTheLeft (the default), or the right-inverse-multiply \a other * inverse(\c *this) if - * \a Side==OnTheRight. - * - * Note that the template parameter \c Side can be ommitted, in which case \c Side==OnTheLeft - * - * The matrix \c *this must be triangular and invertible (i.e., all the coefficients of the - * diagonal must be non zero). It works as a forward (resp. backward) substitution if \c *this - * is an upper (resp. lower) triangular matrix. - * - * Example: \include Triangular_solve.cpp - * Output: \verbinclude Triangular_solve.out - * - * This function returns an expression of the inverse-multiply and can works in-place if it is assigned - * to the same matrix or vector \a other. - * - * For users coming from BLAS, this function (and more specifically solveInPlace()) offer - * all the operations supported by the \c *TRSV and \c *TRSM BLAS routines. - * - * \sa TriangularView::solveInPlace() - */ - template - EIGEN_DEVICE_FUNC - inline const internal::triangular_solve_retval - solve(const MatrixBase& other) const; - - /** "in-place" version of TriangularView::solve() where the result is written in \a other - * - * \warning The parameter is only marked 'const' to make the C++ compiler accept a temporary expression here. - * This function will const_cast it, so constness isn't honored here. - * - * Note that the template parameter \c Side can be ommitted, in which case \c Side==OnTheLeft - * - * See TriangularView:solve() for the details. - */ - template - EIGEN_DEVICE_FUNC - void solveInPlace(const MatrixBase& other) const; + /** Efficient vector/matrix times triangular matrix product */ + template + friend EIGEN_DEVICE_FUNC const Product + operator*(const MatrixBase &lhs, const TriangularViewImpl &rhs) + { + return Product(lhs.derived(), rhs.derived()); + } - template - EIGEN_DEVICE_FUNC - void solveInPlace(const MatrixBase& other) const - { return solveInPlace(other); } + /** \returns the product of the inverse of \c *this with \a other, \a *this being triangular. + * + * This function computes the inverse-matrix matrix product inverse(\c *this) * \a other if + * \a Side==OnTheLeft (the default), or the right-inverse-multiply \a other * inverse(\c *this) if + * \a Side==OnTheRight. + * + * Note that the template parameter \c Side can be ommitted, in which case \c Side==OnTheLeft + * + * The matrix \c *this must be triangular and invertible (i.e., all the coefficients of the + * diagonal must be non zero). It works as a forward (resp. backward) substitution if \c *this + * is an upper (resp. lower) triangular matrix. + * + * Example: \include Triangular_solve.cpp + * Output: \verbinclude Triangular_solve.out + * + * This function returns an expression of the inverse-multiply and can works in-place if it is assigned + * to the same matrix or vector \a other. + * + * For users coming from BLAS, this function (and more specifically solveInPlace()) offer + * all the operations supported by the \c *TRSV and \c *TRSM BLAS routines. + * + * \sa TriangularView::solveInPlace() + */ + template + EIGEN_DEVICE_FUNC inline const internal::triangular_solve_retval solve( + const MatrixBase &other) const; + + /** "in-place" version of TriangularView::solve() where the result is written in \a other + * + * \warning The parameter is only marked 'const' to make the C++ compiler accept a temporary expression here. + * This function will const_cast it, so constness isn't honored here. + * + * Note that the template parameter \c Side can be ommitted, in which case \c Side==OnTheLeft + * + * See TriangularView:solve() for the details. + */ + template + EIGEN_DEVICE_FUNC void solveInPlace(const MatrixBase &other) const; + + template EIGEN_DEVICE_FUNC void solveInPlace(const MatrixBase &other) const + { + return solveInPlace(other); + } - /** Swaps the coefficients of the common triangular parts of two matrices */ - template - EIGEN_DEVICE_FUNC + /** Swaps the coefficients of the common triangular parts of two matrices */ + template + EIGEN_DEVICE_FUNC #ifdef EIGEN_PARSED_BY_DOXYGEN void swap(TriangularBase &other) #else - void swap(TriangularBase const & other) + void swap(TriangularBase const &other) #endif - { - EIGEN_STATIC_ASSERT_LVALUE(OtherDerived); - call_assignment(derived(), other.const_cast_derived(), internal::swap_assign_op()); - } + { + EIGEN_STATIC_ASSERT_LVALUE(OtherDerived); + call_assignment(derived(), other.const_cast_derived(), internal::swap_assign_op()); + } - /** \deprecated - * Shortcut for \code (*this).swap(other.triangularView<(*this)::Mode>()) \endcode */ - template - EIGEN_DEVICE_FUNC - void swap(MatrixBase const & other) - { - EIGEN_STATIC_ASSERT_LVALUE(OtherDerived); - call_assignment(derived(), other.const_cast_derived(), internal::swap_assign_op()); - } + /** \deprecated + * Shortcut for \code (*this).swap(other.triangularView<(*this)::Mode>()) \endcode */ + template EIGEN_DEVICE_FUNC void swap(MatrixBase const &other) + { + EIGEN_STATIC_ASSERT_LVALUE(OtherDerived); + call_assignment(derived(), other.const_cast_derived(), internal::swap_assign_op()); + } - template - EIGEN_DEVICE_FUNC - EIGEN_STRONG_INLINE void _solve_impl(const RhsType &rhs, DstType &dst) const { - if(!internal::is_same_dense(dst,rhs)) - dst = rhs; - this->solveInPlace(dst); - } + template + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void _solve_impl(const RhsType &rhs, DstType &dst) const + { + if (!internal::is_same_dense(dst, rhs)) dst = rhs; + this->solveInPlace(dst); + } - template - EIGEN_DEVICE_FUNC - EIGEN_STRONG_INLINE TriangularViewType& _assignProduct(const ProductType& prod, const Scalar& alpha, bool beta); + template + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TriangularViewType & + _assignProduct(const ProductType &prod, const Scalar &alpha, bool beta); }; /*************************************************************************** -* Implementation of triangular evaluation/assignment -***************************************************************************/ + * Implementation of triangular evaluation/assignment + ***************************************************************************/ #ifndef EIGEN_PARSED_BY_DOXYGEN // FIXME should we keep that possibility template template -inline TriangularView& -TriangularViewImpl::operator=(const MatrixBase& other) +inline TriangularView &TriangularViewImpl::operator=( + const MatrixBase &other) { - internal::call_assignment_no_alias(derived(), other.derived(), internal::assign_op()); + internal::call_assignment_no_alias( + derived(), other.derived(), internal::assign_op()); return derived(); } // FIXME should we keep that possibility template template -void TriangularViewImpl::lazyAssign(const MatrixBase& other) +void TriangularViewImpl::lazyAssign(const MatrixBase &other) { internal::call_assignment_no_alias(derived(), other.template triangularView()); } - template template -inline TriangularView& -TriangularViewImpl::operator=(const TriangularBase& other) +inline TriangularView &TriangularViewImpl::operator=( + const TriangularBase &other) { eigen_assert(Mode == int(OtherDerived::Mode)); internal::call_assignment(derived(), other.derived()); @@ -583,7 +550,7 @@ TriangularViewImpl::operator=(const TriangularBase template -void TriangularViewImpl::lazyAssign(const TriangularBase& other) +void TriangularViewImpl::lazyAssign(const TriangularBase &other) { eigen_assert(Mode == int(OtherDerived::Mode)); internal::call_assignment_no_alias(derived(), other.derived()); @@ -591,11 +558,11 @@ void TriangularViewImpl::lazyAssign(const TriangularBas #endif /*************************************************************************** -* Implementation of TriangularBase methods -***************************************************************************/ + * Implementation of TriangularBase methods + ***************************************************************************/ /** Assigns a triangular or selfadjoint matrix to a dense matrix. - * If the matrix is triangular, the opposite part is set to zero. */ + * If the matrix is triangular, the opposite part is set to zero. */ template template void TriangularBase::evalTo(MatrixBase &other) const @@ -604,28 +571,27 @@ void TriangularBase::evalTo(MatrixBase &other) const } /*************************************************************************** -* Implementation of TriangularView methods -***************************************************************************/ + * Implementation of TriangularView methods + ***************************************************************************/ /*************************************************************************** -* Implementation of MatrixBase methods -***************************************************************************/ + * Implementation of MatrixBase methods + ***************************************************************************/ /** - * \returns an expression of a triangular view extracted from the current matrix - * - * The parameter \a Mode can have the following values: \c #Upper, \c #StrictlyUpper, \c #UnitUpper, - * \c #Lower, \c #StrictlyLower, \c #UnitLower. - * - * Example: \include MatrixBase_triangularView.cpp - * Output: \verbinclude MatrixBase_triangularView.out - * - * \sa class TriangularView - */ + * \returns an expression of a triangular view extracted from the current matrix + * + * The parameter \a Mode can have the following values: \c #Upper, \c #StrictlyUpper, \c #UnitUpper, + * \c #Lower, \c #StrictlyLower, \c #UnitLower. + * + * Example: \include MatrixBase_triangularView.cpp + * Output: \verbinclude MatrixBase_triangularView.out + * + * \sa class TriangularView + */ template template -typename MatrixBase::template TriangularViewReturnType::Type -MatrixBase::triangularView() +typename MatrixBase::template TriangularViewReturnType::Type MatrixBase::triangularView() { return typename TriangularViewReturnType::Type(derived()); } @@ -634,57 +600,51 @@ MatrixBase::triangularView() template template typename MatrixBase::template ConstTriangularViewReturnType::Type -MatrixBase::triangularView() const + MatrixBase::triangularView() const { return typename ConstTriangularViewReturnType::Type(derived()); } /** \returns true if *this is approximately equal to an upper triangular matrix, - * within the precision given by \a prec. - * - * \sa isLowerTriangular() - */ -template -bool MatrixBase::isUpperTriangular(const RealScalar& prec) const + * within the precision given by \a prec. + * + * \sa isLowerTriangular() + */ +template bool MatrixBase::isUpperTriangular(const RealScalar &prec) const { RealScalar maxAbsOnUpperPart = static_cast(-1); - for(Index j = 0; j < cols(); ++j) - { - Index maxi = numext::mini(j, rows()-1); - for(Index i = 0; i <= maxi; ++i) - { - RealScalar absValue = numext::abs(coeff(i,j)); - if(absValue > maxAbsOnUpperPart) maxAbsOnUpperPart = absValue; + for (Index j = 0; j < cols(); ++j) { + Index maxi = numext::mini(j, rows() - 1); + for (Index i = 0; i <= maxi; ++i) { + RealScalar absValue = numext::abs(coeff(i, j)); + if (absValue > maxAbsOnUpperPart) maxAbsOnUpperPart = absValue; } } RealScalar threshold = maxAbsOnUpperPart * prec; - for(Index j = 0; j < cols(); ++j) - for(Index i = j+1; i < rows(); ++i) - if(numext::abs(coeff(i, j)) > threshold) return false; + for (Index j = 0; j < cols(); ++j) + for (Index i = j + 1; i < rows(); ++i) + if (numext::abs(coeff(i, j)) > threshold) return false; return true; } /** \returns true if *this is approximately equal to a lower triangular matrix, - * within the precision given by \a prec. - * - * \sa isUpperTriangular() - */ -template -bool MatrixBase::isLowerTriangular(const RealScalar& prec) const + * within the precision given by \a prec. + * + * \sa isUpperTriangular() + */ +template bool MatrixBase::isLowerTriangular(const RealScalar &prec) const { RealScalar maxAbsOnLowerPart = static_cast(-1); - for(Index j = 0; j < cols(); ++j) - for(Index i = j; i < rows(); ++i) - { - RealScalar absValue = numext::abs(coeff(i,j)); - if(absValue > maxAbsOnLowerPart) maxAbsOnLowerPart = absValue; + for (Index j = 0; j < cols(); ++j) + for (Index i = j; i < rows(); ++i) { + RealScalar absValue = numext::abs(coeff(i, j)); + if (absValue > maxAbsOnLowerPart) maxAbsOnLowerPart = absValue; } RealScalar threshold = maxAbsOnLowerPart * prec; - for(Index j = 1; j < cols(); ++j) - { - Index maxi = numext::mini(j, rows()-1); - for(Index i = 0; i < maxi; ++i) - if(numext::abs(coeff(i, j)) > threshold) return false; + for (Index j = 1; j < cols(); ++j) { + Index maxi = numext::mini(j, rows() - 1); + for (Index i = 0; i < maxi; ++i) + if (numext::abs(coeff(i, j)) > threshold) return false; } return true; } @@ -698,286 +658,330 @@ bool MatrixBase::isLowerTriangular(const RealScalar& prec) const namespace internal { - -// TODO currently a triangular expression has the form TriangularView<.,.> -// in the future triangular-ness should be defined by the expression traits -// such that Transpose > is valid. (currently TriangularBase::transpose() is overloaded to make it work) -template -struct evaluator_traits > -{ - typedef typename storage_kind_to_evaluator_kind::Kind Kind; - typedef typename glue_shapes::Shape, TriangularShape>::type Shape; -}; -template -struct unary_evaluator, IndexBased> - : evaluator::type> -{ - typedef TriangularView XprType; - typedef evaluator::type> Base; - unary_evaluator(const XprType &xpr) : Base(xpr.nestedExpression()) {} -}; + // TODO currently a triangular expression has the form TriangularView<.,.> + // in the future triangular-ness should be defined by the expression traits + // such that Transpose > is valid. (currently TriangularBase::transpose() is overloaded to + // make it work) + template struct evaluator_traits> + { + typedef typename storage_kind_to_evaluator_kind::Kind Kind; + typedef typename glue_shapes::Shape, TriangularShape>::type Shape; + }; + + template + struct unary_evaluator, IndexBased> + : evaluator::type> + { + typedef TriangularView XprType; + typedef evaluator::type> Base; + unary_evaluator(const XprType &xpr) : Base(xpr.nestedExpression()) {} + }; -// Additional assignment kinds: -struct Triangular2Triangular {}; -struct Triangular2Dense {}; -struct Dense2Triangular {}; + // Additional assignment kinds: + struct Triangular2Triangular + { + }; + struct Triangular2Dense + { + }; + struct Dense2Triangular + { + }; -template struct triangular_assignment_loop; + template struct triangular_assignment_loop; + + + /** \internal Specialization of the dense assignment kernel for triangular matrices. + * The main difference is that the triangular, diagonal, and opposite parts are processed through three different + * functions. + * \tparam UpLo must be either Lower or Upper + * \tparam Mode must be either 0, UnitDiag, ZeroDiag, or SelfAdjoint + */ + template + class triangular_dense_assignment_kernel + : public generic_dense_assignment_kernel + { + protected: + typedef generic_dense_assignment_kernel Base; + typedef typename Base::DstXprType DstXprType; + typedef typename Base::SrcXprType SrcXprType; + using Base::m_dst; + using Base::m_src; + using Base::m_functor; + + public: + typedef typename Base::DstEvaluatorType DstEvaluatorType; + typedef typename Base::SrcEvaluatorType SrcEvaluatorType; + typedef typename Base::Scalar Scalar; + typedef typename Base::AssignmentTraits AssignmentTraits; + + + EIGEN_DEVICE_FUNC triangular_dense_assignment_kernel(DstEvaluatorType &dst, + const SrcEvaluatorType &src, + const Functor &func, + DstXprType &dstExpr) + : Base(dst, src, func, dstExpr) + {} - -/** \internal Specialization of the dense assignment kernel for triangular matrices. - * The main difference is that the triangular, diagonal, and opposite parts are processed through three different functions. - * \tparam UpLo must be either Lower or Upper - * \tparam Mode must be either 0, UnitDiag, ZeroDiag, or SelfAdjoint - */ -template -class triangular_dense_assignment_kernel : public generic_dense_assignment_kernel -{ -protected: - typedef generic_dense_assignment_kernel Base; - typedef typename Base::DstXprType DstXprType; - typedef typename Base::SrcXprType SrcXprType; - using Base::m_dst; - using Base::m_src; - using Base::m_functor; -public: - - typedef typename Base::DstEvaluatorType DstEvaluatorType; - typedef typename Base::SrcEvaluatorType SrcEvaluatorType; - typedef typename Base::Scalar Scalar; - typedef typename Base::AssignmentTraits AssignmentTraits; - - - EIGEN_DEVICE_FUNC triangular_dense_assignment_kernel(DstEvaluatorType &dst, const SrcEvaluatorType &src, const Functor &func, DstXprType& dstExpr) - : Base(dst, src, func, dstExpr) - {} - #ifdef EIGEN_INTERNAL_DEBUGGING - EIGEN_DEVICE_FUNC void assignCoeff(Index row, Index col) - { - eigen_internal_assert(row!=col); - Base::assignCoeff(row,col); - } + EIGEN_DEVICE_FUNC void assignCoeff(Index row, Index col) + { + eigen_internal_assert(row != col); + Base::assignCoeff(row, col); + } #else - using Base::assignCoeff; + using Base::assignCoeff; #endif - - EIGEN_DEVICE_FUNC void assignDiagonalCoeff(Index id) + + EIGEN_DEVICE_FUNC void assignDiagonalCoeff(Index id) + { + if (Mode == UnitDiag && SetOpposite) + m_functor.assignCoeff(m_dst.coeffRef(id, id), Scalar(1)); + else if (Mode == ZeroDiag && SetOpposite) + m_functor.assignCoeff(m_dst.coeffRef(id, id), Scalar(0)); + else if (Mode == 0) + Base::assignCoeff(id, id); + } + + EIGEN_DEVICE_FUNC void assignOppositeCoeff(Index row, Index col) + { + eigen_internal_assert(row != col); + if (SetOpposite) m_functor.assignCoeff(m_dst.coeffRef(row, col), Scalar(0)); + } + }; + + template + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void + call_triangular_assignment_loop(DstXprType &dst, const SrcXprType &src, const Functor &func) { - if(Mode==UnitDiag && SetOpposite) m_functor.assignCoeff(m_dst.coeffRef(id,id), Scalar(1)); - else if(Mode==ZeroDiag && SetOpposite) m_functor.assignCoeff(m_dst.coeffRef(id,id), Scalar(0)); - else if(Mode==0) Base::assignCoeff(id,id); - } - - EIGEN_DEVICE_FUNC void assignOppositeCoeff(Index row, Index col) - { - eigen_internal_assert(row!=col); - if(SetOpposite) - m_functor.assignCoeff(m_dst.coeffRef(row,col), Scalar(0)); - } -}; + typedef evaluator DstEvaluatorType; + typedef evaluator SrcEvaluatorType; -template -EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE -void call_triangular_assignment_loop(DstXprType& dst, const SrcXprType& src, const Functor &func) -{ - typedef evaluator DstEvaluatorType; - typedef evaluator SrcEvaluatorType; - - SrcEvaluatorType srcEvaluator(src); - - Index dstRows = src.rows(); - Index dstCols = src.cols(); - if((dst.rows()!=dstRows) || (dst.cols()!=dstCols)) - dst.resize(dstRows, dstCols); - DstEvaluatorType dstEvaluator(dst); - - typedef triangular_dense_assignment_kernel< Mode&(Lower|Upper),Mode&(UnitDiag|ZeroDiag|SelfAdjoint),SetOpposite, - DstEvaluatorType,SrcEvaluatorType,Functor> Kernel; - Kernel kernel(dstEvaluator, srcEvaluator, func, dst.const_cast_derived()); - - enum { - unroll = DstXprType::SizeAtCompileTime != Dynamic - && SrcEvaluatorType::CoeffReadCost < HugeCost - && DstXprType::SizeAtCompileTime * (DstEvaluatorType::CoeffReadCost+SrcEvaluatorType::CoeffReadCost) / 2 <= EIGEN_UNROLLING_LIMIT - }; - - triangular_assignment_loop::run(kernel); -} + SrcEvaluatorType srcEvaluator(src); -template -EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE -void call_triangular_assignment_loop(DstXprType& dst, const SrcXprType& src) -{ - call_triangular_assignment_loop(dst, src, internal::assign_op()); -} + Index dstRows = src.rows(); + Index dstCols = src.cols(); + if ((dst.rows() != dstRows) || (dst.cols() != dstCols)) dst.resize(dstRows, dstCols); + DstEvaluatorType dstEvaluator(dst); + + typedef triangular_dense_assignment_kernel + Kernel; + Kernel kernel(dstEvaluator, srcEvaluator, func, dst.const_cast_derived()); -template<> struct AssignmentKind { typedef Triangular2Triangular Kind; }; -template<> struct AssignmentKind { typedef Triangular2Dense Kind; }; -template<> struct AssignmentKind { typedef Dense2Triangular Kind; }; + enum { + unroll = + DstXprType::SizeAtCompileTime != Dynamic && SrcEvaluatorType::CoeffReadCost < HugeCost + && DstXprType::SizeAtCompileTime * (DstEvaluatorType::CoeffReadCost + SrcEvaluatorType::CoeffReadCost) / 2 + <= EIGEN_UNROLLING_LIMIT + }; + triangular_assignment_loop::run( + kernel); + } -template< typename DstXprType, typename SrcXprType, typename Functor> -struct Assignment -{ - EIGEN_DEVICE_FUNC static void run(DstXprType &dst, const SrcXprType &src, const Functor &func) + template + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void call_triangular_assignment_loop(DstXprType &dst, const SrcXprType &src) { - eigen_assert(int(DstXprType::Mode) == int(SrcXprType::Mode)); - - call_triangular_assignment_loop(dst, src, func); + call_triangular_assignment_loop( + dst, src, internal::assign_op()); } -}; -template< typename DstXprType, typename SrcXprType, typename Functor> -struct Assignment -{ - EIGEN_DEVICE_FUNC static void run(DstXprType &dst, const SrcXprType &src, const Functor &func) + template<> struct AssignmentKind { - call_triangular_assignment_loop(dst, src, func); - } -}; + typedef Triangular2Triangular Kind; + }; + template<> struct AssignmentKind + { + typedef Triangular2Dense Kind; + }; + template<> struct AssignmentKind + { + typedef Dense2Triangular Kind; + }; -template< typename DstXprType, typename SrcXprType, typename Functor> -struct Assignment -{ - EIGEN_DEVICE_FUNC static void run(DstXprType &dst, const SrcXprType &src, const Functor &func) + + template + struct Assignment { - call_triangular_assignment_loop(dst, src, func); - } -}; + EIGEN_DEVICE_FUNC static void run(DstXprType &dst, const SrcXprType &src, const Functor &func) + { + eigen_assert(int(DstXprType::Mode) == int(SrcXprType::Mode)); + call_triangular_assignment_loop(dst, src, func); + } + }; -template -struct triangular_assignment_loop -{ - // FIXME: this is not very clean, perhaps this information should be provided by the kernel? - typedef typename Kernel::DstEvaluatorType DstEvaluatorType; - typedef typename DstEvaluatorType::XprType DstXprType; - - enum { - col = (UnrollCount-1) / DstXprType::RowsAtCompileTime, - row = (UnrollCount-1) % DstXprType::RowsAtCompileTime + template + struct Assignment + { + EIGEN_DEVICE_FUNC static void run(DstXprType &dst, const SrcXprType &src, const Functor &func) + { + call_triangular_assignment_loop(dst, src, func); + } }; - - typedef typename Kernel::Scalar Scalar; - EIGEN_DEVICE_FUNC - static inline void run(Kernel &kernel) + template + struct Assignment { - triangular_assignment_loop::run(kernel); - - if(row==col) - kernel.assignDiagonalCoeff(row); - else if( ((Mode&Lower) && row>col) || ((Mode&Upper) && row(dst, src, func); + } + }; -// prevent buggy user code from causing an infinite recursion -template -struct triangular_assignment_loop -{ - EIGEN_DEVICE_FUNC - static inline void run(Kernel &) {} -}; + template struct triangular_assignment_loop + { + // FIXME: this is not very clean, perhaps this information should be provided by the kernel? + typedef typename Kernel::DstEvaluatorType DstEvaluatorType; + typedef typename DstEvaluatorType::XprType DstXprType; + enum { + col = (UnrollCount - 1) / DstXprType::RowsAtCompileTime, + row = (UnrollCount - 1) % DstXprType::RowsAtCompileTime + }; -// TODO: experiment with a recursive assignment procedure splitting the current -// triangular part into one rectangular and two triangular parts. + typedef typename Kernel::Scalar Scalar; + EIGEN_DEVICE_FUNC + static inline void run(Kernel &kernel) + { + triangular_assignment_loop::run(kernel); + + if (row == col) + kernel.assignDiagonalCoeff(row); + else if (((Mode & Lower) && row > col) || ((Mode & Upper) && row < col)) + kernel.assignCoeff(row, col); + else if (SetOpposite) + kernel.assignOppositeCoeff(row, col); + } + }; -template -struct triangular_assignment_loop -{ - typedef typename Kernel::Scalar Scalar; - EIGEN_DEVICE_FUNC - static inline void run(Kernel &kernel) + // prevent buggy user code from causing an infinite recursion + template + struct triangular_assignment_loop { - for(Index j = 0; j < kernel.cols(); ++j) + EIGEN_DEVICE_FUNC + static inline void run(Kernel &) {} + }; + + + // TODO: experiment with a recursive assignment procedure splitting the current + // triangular part into one rectangular and two triangular parts. + + + template + struct triangular_assignment_loop + { + typedef typename Kernel::Scalar Scalar; + EIGEN_DEVICE_FUNC + static inline void run(Kernel &kernel) { - Index maxi = numext::mini(j, kernel.rows()); - Index i = 0; - if (((Mode&Lower) && SetOpposite) || (Mode&Upper)) - { - for(; i < maxi; ++i) - if(Mode&Upper) kernel.assignCoeff(i, j); - else kernel.assignOppositeCoeff(i, j); - } - else - i = maxi; - - if(i template void TriangularBase::evalToLazy(MatrixBase &other) const { other.derived().resize(this->rows(), this->cols()); - internal::call_triangular_assignment_loop(other.derived(), derived().nestedExpression()); + internal::call_triangular_assignment_loop( + other.derived(), derived().nestedExpression()); } namespace internal { - -// Triangular = Product -template< typename DstXprType, typename Lhs, typename Rhs, typename Scalar> -struct Assignment, internal::assign_op::Scalar>, Dense2Triangular> -{ - typedef Product SrcXprType; - static void run(DstXprType &dst, const SrcXprType &src, const internal::assign_op &) + + // Triangular = Product + template + struct Assignment, + internal::assign_op::Scalar>, + Dense2Triangular> { - Index dstRows = src.rows(); - Index dstCols = src.cols(); - if((dst.rows()!=dstRows) || (dst.cols()!=dstCols)) - dst.resize(dstRows, dstCols); + typedef Product SrcXprType; + static void + run(DstXprType &dst, const SrcXprType &src, const internal::assign_op &) + { + Index dstRows = src.rows(); + Index dstCols = src.cols(); + if ((dst.rows() != dstRows) || (dst.cols() != dstCols)) dst.resize(dstRows, dstCols); - dst._assignProduct(src, 1, 0); - } -}; + dst._assignProduct(src, 1, 0); + } + }; -// Triangular += Product -template< typename DstXprType, typename Lhs, typename Rhs, typename Scalar> -struct Assignment, internal::add_assign_op::Scalar>, Dense2Triangular> -{ - typedef Product SrcXprType; - static void run(DstXprType &dst, const SrcXprType &src, const internal::add_assign_op &) + // Triangular += Product + template + struct Assignment, + internal::add_assign_op::Scalar>, + Dense2Triangular> { - dst._assignProduct(src, 1, 1); - } -}; + typedef Product SrcXprType; + static void + run(DstXprType &dst, const SrcXprType &src, const internal::add_assign_op &) + { + dst._assignProduct(src, 1, 1); + } + }; -// Triangular -= Product -template< typename DstXprType, typename Lhs, typename Rhs, typename Scalar> -struct Assignment, internal::sub_assign_op::Scalar>, Dense2Triangular> -{ - typedef Product SrcXprType; - static void run(DstXprType &dst, const SrcXprType &src, const internal::sub_assign_op &) + // Triangular -= Product + template + struct Assignment, + internal::sub_assign_op::Scalar>, + Dense2Triangular> { - dst._assignProduct(src, -1, 1); - } -}; + typedef Product SrcXprType; + static void + run(DstXprType &dst, const SrcXprType &src, const internal::sub_assign_op &) + { + dst._assignProduct(src, -1, 1); + } + }; -} // end namespace internal +}// end namespace internal -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_TRIANGULARMATRIX_H +#endif// EIGEN_TRIANGULARMATRIX_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/VectorBlock.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/VectorBlock.h index d72fbf7e..5fe88fe3 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/VectorBlock.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/VectorBlock.h @@ -11,86 +11,84 @@ #ifndef EIGEN_VECTORBLOCK_H #define EIGEN_VECTORBLOCK_H -namespace Eigen { +namespace Eigen { namespace internal { -template -struct traits > - : public traits::Flags & RowMajorBit ? 1 : Size, - traits::Flags & RowMajorBit ? Size : 1> > -{ -}; -} + template + struct traits> + : public traits::Flags & RowMajorBit ? 1 : Size, + traits::Flags & RowMajorBit ? Size : 1>> + { + }; +}// namespace internal /** \class VectorBlock - * \ingroup Core_Module - * - * \brief Expression of a fixed-size or dynamic-size sub-vector - * - * \tparam VectorType the type of the object in which we are taking a sub-vector - * \tparam Size size of the sub-vector we are taking at compile time (optional) - * - * This class represents an expression of either a fixed-size or dynamic-size sub-vector. - * It is the return type of DenseBase::segment(Index,Index) and DenseBase::segment(Index) and - * most of the time this is the only way it is used. - * - * However, if you want to directly maniputate sub-vector expressions, - * for instance if you want to write a function returning such an expression, you - * will need to use this class. - * - * Here is an example illustrating the dynamic case: - * \include class_VectorBlock.cpp - * Output: \verbinclude class_VectorBlock.out - * - * \note Even though this expression has dynamic size, in the case where \a VectorType - * has fixed size, this expression inherits a fixed maximal size which means that evaluating - * it does not cause a dynamic memory allocation. - * - * Here is an example illustrating the fixed-size case: - * \include class_FixedVectorBlock.cpp - * Output: \verbinclude class_FixedVectorBlock.out - * - * \sa class Block, DenseBase::segment(Index,Index,Index,Index), DenseBase::segment(Index,Index) - */ -template class VectorBlock + * \ingroup Core_Module + * + * \brief Expression of a fixed-size or dynamic-size sub-vector + * + * \tparam VectorType the type of the object in which we are taking a sub-vector + * \tparam Size size of the sub-vector we are taking at compile time (optional) + * + * This class represents an expression of either a fixed-size or dynamic-size sub-vector. + * It is the return type of DenseBase::segment(Index,Index) and DenseBase::segment(Index) and + * most of the time this is the only way it is used. + * + * However, if you want to directly maniputate sub-vector expressions, + * for instance if you want to write a function returning such an expression, you + * will need to use this class. + * + * Here is an example illustrating the dynamic case: + * \include class_VectorBlock.cpp + * Output: \verbinclude class_VectorBlock.out + * + * \note Even though this expression has dynamic size, in the case where \a VectorType + * has fixed size, this expression inherits a fixed maximal size which means that evaluating + * it does not cause a dynamic memory allocation. + * + * Here is an example illustrating the fixed-size case: + * \include class_FixedVectorBlock.cpp + * Output: \verbinclude class_FixedVectorBlock.out + * + * \sa class Block, DenseBase::segment(Index,Index,Index,Index), DenseBase::segment(Index,Index) + */ +template +class VectorBlock : public Block::Flags & RowMajorBit ? 1 : Size, - internal::traits::Flags & RowMajorBit ? Size : 1> + internal::traits::Flags & RowMajorBit ? 1 : Size, + internal::traits::Flags & RowMajorBit ? Size : 1> { - typedef Block::Flags & RowMajorBit ? 1 : Size, - internal::traits::Flags & RowMajorBit ? Size : 1> Base; - enum { - IsColVector = !(internal::traits::Flags & RowMajorBit) - }; - public: - EIGEN_DENSE_PUBLIC_INTERFACE(VectorBlock) + typedef Block::Flags & RowMajorBit ? 1 : Size, + internal::traits::Flags & RowMajorBit ? Size : 1> + Base; + enum { IsColVector = !(internal::traits::Flags & RowMajorBit) }; + +public: + EIGEN_DENSE_PUBLIC_INTERFACE(VectorBlock) - using Base::operator=; + using Base::operator=; - /** Dynamic-size constructor - */ - EIGEN_DEVICE_FUNC - inline VectorBlock(VectorType& vector, Index start, Index size) - : Base(vector, - IsColVector ? start : 0, IsColVector ? 0 : start, - IsColVector ? size : 1, IsColVector ? 1 : size) - { - EIGEN_STATIC_ASSERT_VECTOR_ONLY(VectorBlock); - } + /** Dynamic-size constructor + */ + EIGEN_DEVICE_FUNC + inline VectorBlock(VectorType &vector, Index start, Index size) + : Base(vector, IsColVector ? start : 0, IsColVector ? 0 : start, IsColVector ? size : 1, IsColVector ? 1 : size) + { + EIGEN_STATIC_ASSERT_VECTOR_ONLY(VectorBlock); + } - /** Fixed-size constructor - */ - EIGEN_DEVICE_FUNC - inline VectorBlock(VectorType& vector, Index start) - : Base(vector, IsColVector ? start : 0, IsColVector ? 0 : start) - { - EIGEN_STATIC_ASSERT_VECTOR_ONLY(VectorBlock); - } + /** Fixed-size constructor + */ + EIGEN_DEVICE_FUNC + inline VectorBlock(VectorType &vector, Index start) : Base(vector, IsColVector ? start : 0, IsColVector ? 0 : start) + { + EIGEN_STATIC_ASSERT_VECTOR_ONLY(VectorBlock); + } }; -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_VECTORBLOCK_H +#endif// EIGEN_VECTORBLOCK_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/VectorwiseOp.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/VectorwiseOp.h index 4fe267e9..0fecdd2f 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/VectorwiseOp.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/VectorwiseOp.h @@ -14,682 +14,634 @@ namespace Eigen { /** \class PartialReduxExpr - * \ingroup Core_Module - * - * \brief Generic expression of a partially reduxed matrix - * - * \tparam MatrixType the type of the matrix we are applying the redux operation - * \tparam MemberOp type of the member functor - * \tparam Direction indicates the direction of the redux (#Vertical or #Horizontal) - * - * This class represents an expression of a partial redux operator of a matrix. - * It is the return type of some VectorwiseOp functions, - * and most of the time this is the only way it is used. - * - * \sa class VectorwiseOp - */ - -template< typename MatrixType, typename MemberOp, int Direction> -class PartialReduxExpr; + * \ingroup Core_Module + * + * \brief Generic expression of a partially reduxed matrix + * + * \tparam MatrixType the type of the matrix we are applying the redux operation + * \tparam MemberOp type of the member functor + * \tparam Direction indicates the direction of the redux (#Vertical or #Horizontal) + * + * This class represents an expression of a partial redux operator of a matrix. + * It is the return type of some VectorwiseOp functions, + * and most of the time this is the only way it is used. + * + * \sa class VectorwiseOp + */ + +template class PartialReduxExpr; namespace internal { -template -struct traits > - : traits -{ - typedef typename MemberOp::result_type Scalar; - typedef typename traits::StorageKind StorageKind; - typedef typename traits::XprKind XprKind; - typedef typename MatrixType::Scalar InputScalar; - enum { - RowsAtCompileTime = Direction==Vertical ? 1 : MatrixType::RowsAtCompileTime, - ColsAtCompileTime = Direction==Horizontal ? 1 : MatrixType::ColsAtCompileTime, - MaxRowsAtCompileTime = Direction==Vertical ? 1 : MatrixType::MaxRowsAtCompileTime, - MaxColsAtCompileTime = Direction==Horizontal ? 1 : MatrixType::MaxColsAtCompileTime, - Flags = RowsAtCompileTime == 1 ? RowMajorBit : 0, - TraversalSize = Direction==Vertical ? MatrixType::RowsAtCompileTime : MatrixType::ColsAtCompileTime + template + struct traits> : traits + { + typedef typename MemberOp::result_type Scalar; + typedef typename traits::StorageKind StorageKind; + typedef typename traits::XprKind XprKind; + typedef typename MatrixType::Scalar InputScalar; + enum { + RowsAtCompileTime = Direction == Vertical ? 1 : MatrixType::RowsAtCompileTime, + ColsAtCompileTime = Direction == Horizontal ? 1 : MatrixType::ColsAtCompileTime, + MaxRowsAtCompileTime = Direction == Vertical ? 1 : MatrixType::MaxRowsAtCompileTime, + MaxColsAtCompileTime = Direction == Horizontal ? 1 : MatrixType::MaxColsAtCompileTime, + Flags = RowsAtCompileTime == 1 ? RowMajorBit : 0, + TraversalSize = Direction == Vertical ? MatrixType::RowsAtCompileTime : MatrixType::ColsAtCompileTime + }; }; -}; -} +}// namespace internal -template< typename MatrixType, typename MemberOp, int Direction> -class PartialReduxExpr : public internal::dense_xpr_base< PartialReduxExpr >::type, - internal::no_assignment_operator +template +class PartialReduxExpr + : public internal::dense_xpr_base>::type + , internal::no_assignment_operator { - public: - - typedef typename internal::dense_xpr_base::type Base; - EIGEN_DENSE_PUBLIC_INTERFACE(PartialReduxExpr) +public: + typedef typename internal::dense_xpr_base::type Base; + EIGEN_DENSE_PUBLIC_INTERFACE(PartialReduxExpr) - EIGEN_DEVICE_FUNC - explicit PartialReduxExpr(const MatrixType& mat, const MemberOp& func = MemberOp()) - : m_matrix(mat), m_functor(func) {} + EIGEN_DEVICE_FUNC + explicit PartialReduxExpr(const MatrixType &mat, const MemberOp &func = MemberOp()) : m_matrix(mat), m_functor(func) + {} - EIGEN_DEVICE_FUNC - Index rows() const { return (Direction==Vertical ? 1 : m_matrix.rows()); } - EIGEN_DEVICE_FUNC - Index cols() const { return (Direction==Horizontal ? 1 : m_matrix.cols()); } + EIGEN_DEVICE_FUNC + Index rows() const { return (Direction == Vertical ? 1 : m_matrix.rows()); } + EIGEN_DEVICE_FUNC + Index cols() const { return (Direction == Horizontal ? 1 : m_matrix.cols()); } - EIGEN_DEVICE_FUNC - typename MatrixType::Nested nestedExpression() const { return m_matrix; } + EIGEN_DEVICE_FUNC + typename MatrixType::Nested nestedExpression() const { return m_matrix; } - EIGEN_DEVICE_FUNC - const MemberOp& functor() const { return m_functor; } + EIGEN_DEVICE_FUNC + const MemberOp &functor() const { return m_functor; } - protected: - typename MatrixType::Nested m_matrix; - const MemberOp m_functor; +protected: + typename MatrixType::Nested m_matrix; + const MemberOp m_functor; }; -#define EIGEN_MEMBER_FUNCTOR(MEMBER,COST) \ - template \ - struct member_##MEMBER { \ - EIGEN_EMPTY_STRUCT_CTOR(member_##MEMBER) \ - typedef ResultType result_type; \ - template struct Cost \ - { enum { value = COST }; }; \ - template \ - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE \ - ResultType operator()(const XprType& mat) const \ - { return mat.MEMBER(); } \ +#define EIGEN_MEMBER_FUNCTOR(MEMBER, COST) \ + template struct member_##MEMBER \ + { \ + EIGEN_EMPTY_STRUCT_CTOR(member_##MEMBER) \ + typedef ResultType result_type; \ + template struct Cost \ + { \ + enum { value = COST }; \ + }; \ + template EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE ResultType operator()(const XprType &mat) const \ + { \ + return mat.MEMBER(); \ + } \ } namespace internal { -EIGEN_MEMBER_FUNCTOR(squaredNorm, Size * NumTraits::MulCost + (Size-1)*NumTraits::AddCost); -EIGEN_MEMBER_FUNCTOR(norm, (Size+5) * NumTraits::MulCost + (Size-1)*NumTraits::AddCost); -EIGEN_MEMBER_FUNCTOR(stableNorm, (Size+5) * NumTraits::MulCost + (Size-1)*NumTraits::AddCost); -EIGEN_MEMBER_FUNCTOR(blueNorm, (Size+5) * NumTraits::MulCost + (Size-1)*NumTraits::AddCost); -EIGEN_MEMBER_FUNCTOR(hypotNorm, (Size-1) * functor_traits >::Cost ); -EIGEN_MEMBER_FUNCTOR(sum, (Size-1)*NumTraits::AddCost); -EIGEN_MEMBER_FUNCTOR(mean, (Size-1)*NumTraits::AddCost + NumTraits::MulCost); -EIGEN_MEMBER_FUNCTOR(minCoeff, (Size-1)*NumTraits::AddCost); -EIGEN_MEMBER_FUNCTOR(maxCoeff, (Size-1)*NumTraits::AddCost); -EIGEN_MEMBER_FUNCTOR(all, (Size-1)*NumTraits::AddCost); -EIGEN_MEMBER_FUNCTOR(any, (Size-1)*NumTraits::AddCost); -EIGEN_MEMBER_FUNCTOR(count, (Size-1)*NumTraits::AddCost); -EIGEN_MEMBER_FUNCTOR(prod, (Size-1)*NumTraits::MulCost); - -template -struct member_lpnorm { - typedef ResultType result_type; - template struct Cost - { enum { value = (Size+5) * NumTraits::MulCost + (Size-1)*NumTraits::AddCost }; }; - EIGEN_DEVICE_FUNC member_lpnorm() {} - template - EIGEN_DEVICE_FUNC inline ResultType operator()(const XprType& mat) const - { return mat.template lpNorm

(); } -}; + EIGEN_MEMBER_FUNCTOR(squaredNorm, Size *NumTraits::MulCost + (Size - 1) * NumTraits::AddCost); + EIGEN_MEMBER_FUNCTOR(norm, (Size + 5) * NumTraits::MulCost + (Size - 1) * NumTraits::AddCost); + EIGEN_MEMBER_FUNCTOR(stableNorm, (Size + 5) * NumTraits::MulCost + (Size - 1) * NumTraits::AddCost); + EIGEN_MEMBER_FUNCTOR(blueNorm, (Size + 5) * NumTraits::MulCost + (Size - 1) * NumTraits::AddCost); + EIGEN_MEMBER_FUNCTOR(hypotNorm, (Size - 1) * functor_traits>::Cost); + EIGEN_MEMBER_FUNCTOR(sum, (Size - 1) * NumTraits::AddCost); + EIGEN_MEMBER_FUNCTOR(mean, (Size - 1) * NumTraits::AddCost + NumTraits::MulCost); + EIGEN_MEMBER_FUNCTOR(minCoeff, (Size - 1) * NumTraits::AddCost); + EIGEN_MEMBER_FUNCTOR(maxCoeff, (Size - 1) * NumTraits::AddCost); + EIGEN_MEMBER_FUNCTOR(all, (Size - 1) * NumTraits::AddCost); + EIGEN_MEMBER_FUNCTOR(any, (Size - 1) * NumTraits::AddCost); + EIGEN_MEMBER_FUNCTOR(count, (Size - 1) * NumTraits::AddCost); + EIGEN_MEMBER_FUNCTOR(prod, (Size - 1) * NumTraits::MulCost); + + template struct member_lpnorm + { + typedef ResultType result_type; + template struct Cost + { + enum { value = (Size + 5) * NumTraits::MulCost + (Size - 1) * NumTraits::AddCost }; + }; + EIGEN_DEVICE_FUNC member_lpnorm() {} + template EIGEN_DEVICE_FUNC inline ResultType operator()(const XprType &mat) const + { + return mat.template lpNorm

(); + } + }; -template -struct member_redux { - typedef typename result_of< - BinaryOp(const Scalar&,const Scalar&) - >::type result_type; - template struct Cost - { enum { value = (Size-1) * functor_traits::Cost }; }; - EIGEN_DEVICE_FUNC explicit member_redux(const BinaryOp func) : m_functor(func) {} - template - EIGEN_DEVICE_FUNC inline result_type operator()(const DenseBase& mat) const - { return mat.redux(m_functor); } - const BinaryOp m_functor; -}; -} + template struct member_redux + { + typedef typename result_of::type result_type; + template struct Cost + { + enum { value = (Size - 1) * functor_traits::Cost }; + }; + EIGEN_DEVICE_FUNC explicit member_redux(const BinaryOp func) : m_functor(func) {} + template EIGEN_DEVICE_FUNC inline result_type operator()(const DenseBase &mat) const + { + return mat.redux(m_functor); + } + const BinaryOp m_functor; + }; +}// namespace internal /** \class VectorwiseOp - * \ingroup Core_Module - * - * \brief Pseudo expression providing partial reduction operations - * - * \tparam ExpressionType the type of the object on which to do partial reductions - * \tparam Direction indicates the direction of the redux (#Vertical or #Horizontal) - * - * This class represents a pseudo expression with partial reduction features. - * It is the return type of DenseBase::colwise() and DenseBase::rowwise() - * and most of the time this is the only way it is used. - * - * Example: \include MatrixBase_colwise.cpp - * Output: \verbinclude MatrixBase_colwise.out - * - * \sa DenseBase::colwise(), DenseBase::rowwise(), class PartialReduxExpr - */ + * \ingroup Core_Module + * + * \brief Pseudo expression providing partial reduction operations + * + * \tparam ExpressionType the type of the object on which to do partial reductions + * \tparam Direction indicates the direction of the redux (#Vertical or #Horizontal) + * + * This class represents a pseudo expression with partial reduction features. + * It is the return type of DenseBase::colwise() and DenseBase::rowwise() + * and most of the time this is the only way it is used. + * + * Example: \include MatrixBase_colwise.cpp + * Output: \verbinclude MatrixBase_colwise.out + * + * \sa DenseBase::colwise(), DenseBase::rowwise(), class PartialReduxExpr + */ template class VectorwiseOp { - public: - - typedef typename ExpressionType::Scalar Scalar; - typedef typename ExpressionType::RealScalar RealScalar; - typedef Eigen::Index Index; ///< \deprecated since Eigen 3.3 - typedef typename internal::ref_selector::non_const_type ExpressionTypeNested; - typedef typename internal::remove_all::type ExpressionTypeNestedCleaned; +public: + typedef typename ExpressionType::Scalar Scalar; + typedef typename ExpressionType::RealScalar RealScalar; + typedef Eigen::Index Index;///< \deprecated since Eigen 3.3 + typedef typename internal::ref_selector::non_const_type ExpressionTypeNested; + typedef typename internal::remove_all::type ExpressionTypeNestedCleaned; + + template class Functor, typename Scalar_ = Scalar> struct ReturnType + { + typedef PartialReduxExpr, Direction> Type; + }; - template class Functor, - typename Scalar_=Scalar> struct ReturnType - { - typedef PartialReduxExpr, - Direction - > Type; - }; + template struct ReduxReturnType + { + typedef PartialReduxExpr, Direction> Type; + }; - template struct ReduxReturnType - { - typedef PartialReduxExpr, - Direction - > Type; - }; + enum { isVertical = (Direction == Vertical) ? 1 : 0, isHorizontal = (Direction == Horizontal) ? 1 : 0 }; + +protected: + typedef + typename internal::conditional::type + SubVector; + /** \internal + * \returns the i-th subvector according to the \c Direction */ + EIGEN_DEVICE_FUNC + SubVector subVector(Index i) { return SubVector(m_matrix.derived(), i); } + + /** \internal + * \returns the number of subvectors in the direction \c Direction */ + EIGEN_DEVICE_FUNC + Index subVectors() const { return isVertical ? m_matrix.cols() : m_matrix.rows(); } + + template struct ExtendedType + { + typedef Replicate + Type; + }; - enum { - isVertical = (Direction==Vertical) ? 1 : 0, - isHorizontal = (Direction==Horizontal) ? 1 : 0 - }; + /** \internal + * Replicates a vector to match the size of \c *this */ + template + EIGEN_DEVICE_FUNC typename ExtendedType::Type extendedTo(const DenseBase &other) const + { + EIGEN_STATIC_ASSERT(EIGEN_IMPLIES(isVertical, OtherDerived::MaxColsAtCompileTime == 1), + YOU_PASSED_A_ROW_VECTOR_BUT_A_COLUMN_VECTOR_WAS_EXPECTED) + EIGEN_STATIC_ASSERT(EIGEN_IMPLIES(isHorizontal, OtherDerived::MaxRowsAtCompileTime == 1), + YOU_PASSED_A_COLUMN_VECTOR_BUT_A_ROW_VECTOR_WAS_EXPECTED) + return typename ExtendedType::Type( + other.derived(), isVertical ? 1 : m_matrix.rows(), isHorizontal ? 1 : m_matrix.cols()); + } - protected: + template struct OppositeExtendedType + { + typedef Replicate + Type; + }; - typedef typename internal::conditional::type SubVector; - /** \internal - * \returns the i-th subvector according to the \c Direction */ - EIGEN_DEVICE_FUNC - SubVector subVector(Index i) - { - return SubVector(m_matrix.derived(),i); - } + /** \internal + * Replicates a vector in the opposite direction to match the size of \c *this */ + template + EIGEN_DEVICE_FUNC typename OppositeExtendedType::Type extendedToOpposite( + const DenseBase &other) const + { + EIGEN_STATIC_ASSERT(EIGEN_IMPLIES(isHorizontal, OtherDerived::MaxColsAtCompileTime == 1), + YOU_PASSED_A_ROW_VECTOR_BUT_A_COLUMN_VECTOR_WAS_EXPECTED) + EIGEN_STATIC_ASSERT(EIGEN_IMPLIES(isVertical, OtherDerived::MaxRowsAtCompileTime == 1), + YOU_PASSED_A_COLUMN_VECTOR_BUT_A_ROW_VECTOR_WAS_EXPECTED) + return typename OppositeExtendedType::Type( + other.derived(), isHorizontal ? 1 : m_matrix.rows(), isVertical ? 1 : m_matrix.cols()); + } - /** \internal - * \returns the number of subvectors in the direction \c Direction */ - EIGEN_DEVICE_FUNC - Index subVectors() const - { return isVertical?m_matrix.cols():m_matrix.rows(); } +public: + EIGEN_DEVICE_FUNC + explicit inline VectorwiseOp(ExpressionType &matrix) : m_matrix(matrix) {} + + /** \internal */ + EIGEN_DEVICE_FUNC + inline const ExpressionType &_expression() const { return m_matrix; } + + /** \returns a row or column vector expression of \c *this reduxed by \a func + * + * The template parameter \a BinaryOp is the type of the functor + * of the custom redux operator. Note that func must be an associative operator. + * + * \sa class VectorwiseOp, DenseBase::colwise(), DenseBase::rowwise() + */ + template + EIGEN_DEVICE_FUNC const typename ReduxReturnType::Type redux(const BinaryOp &func = BinaryOp()) const + { + return typename ReduxReturnType::Type(_expression(), internal::member_redux(func)); + } - template struct ExtendedType { - typedef Replicate Type; - }; + typedef typename ReturnType::Type MinCoeffReturnType; + typedef typename ReturnType::Type MaxCoeffReturnType; + typedef typename ReturnType::Type SquaredNormReturnType; + typedef typename ReturnType::Type NormReturnType; + typedef typename ReturnType::Type BlueNormReturnType; + typedef typename ReturnType::Type StableNormReturnType; + typedef typename ReturnType::Type HypotNormReturnType; + typedef typename ReturnType::Type SumReturnType; + typedef typename ReturnType::Type MeanReturnType; + typedef typename ReturnType::Type AllReturnType; + typedef typename ReturnType::Type AnyReturnType; + typedef PartialReduxExpr, Direction> CountReturnType; + typedef typename ReturnType::Type ProdReturnType; + typedef Reverse ConstReverseReturnType; + typedef Reverse ReverseReturnType; + + template struct LpNormReturnType + { + typedef PartialReduxExpr, Direction> Type; + }; - /** \internal - * Replicates a vector to match the size of \c *this */ - template - EIGEN_DEVICE_FUNC - typename ExtendedType::Type - extendedTo(const DenseBase& other) const - { - EIGEN_STATIC_ASSERT(EIGEN_IMPLIES(isVertical, OtherDerived::MaxColsAtCompileTime==1), - YOU_PASSED_A_ROW_VECTOR_BUT_A_COLUMN_VECTOR_WAS_EXPECTED) - EIGEN_STATIC_ASSERT(EIGEN_IMPLIES(isHorizontal, OtherDerived::MaxRowsAtCompileTime==1), - YOU_PASSED_A_COLUMN_VECTOR_BUT_A_ROW_VECTOR_WAS_EXPECTED) - return typename ExtendedType::Type - (other.derived(), - isVertical ? 1 : m_matrix.rows(), - isHorizontal ? 1 : m_matrix.cols()); - } + /** \returns a row (or column) vector expression of the smallest coefficient + * of each column (or row) of the referenced expression. + * + * \warning the result is undefined if \c *this contains NaN. + * + * Example: \include PartialRedux_minCoeff.cpp + * Output: \verbinclude PartialRedux_minCoeff.out + * + * \sa DenseBase::minCoeff() */ + EIGEN_DEVICE_FUNC + const MinCoeffReturnType minCoeff() const { return MinCoeffReturnType(_expression()); } + + /** \returns a row (or column) vector expression of the largest coefficient + * of each column (or row) of the referenced expression. + * + * \warning the result is undefined if \c *this contains NaN. + * + * Example: \include PartialRedux_maxCoeff.cpp + * Output: \verbinclude PartialRedux_maxCoeff.out + * + * \sa DenseBase::maxCoeff() */ + EIGEN_DEVICE_FUNC + const MaxCoeffReturnType maxCoeff() const { return MaxCoeffReturnType(_expression()); } + + /** \returns a row (or column) vector expression of the squared norm + * of each column (or row) of the referenced expression. + * This is a vector with real entries, even if the original matrix has complex entries. + * + * Example: \include PartialRedux_squaredNorm.cpp + * Output: \verbinclude PartialRedux_squaredNorm.out + * + * \sa DenseBase::squaredNorm() */ + EIGEN_DEVICE_FUNC + const SquaredNormReturnType squaredNorm() const { return SquaredNormReturnType(_expression()); } + + /** \returns a row (or column) vector expression of the norm + * of each column (or row) of the referenced expression. + * This is a vector with real entries, even if the original matrix has complex entries. + * + * Example: \include PartialRedux_norm.cpp + * Output: \verbinclude PartialRedux_norm.out + * + * \sa DenseBase::norm() */ + EIGEN_DEVICE_FUNC + const NormReturnType norm() const { return NormReturnType(_expression()); } + + /** \returns a row (or column) vector expression of the norm + * of each column (or row) of the referenced expression. + * This is a vector with real entries, even if the original matrix has complex entries. + * + * Example: \include PartialRedux_norm.cpp + * Output: \verbinclude PartialRedux_norm.out + * + * \sa DenseBase::norm() */ + template EIGEN_DEVICE_FUNC const typename LpNormReturnType

::Type lpNorm() const + { + return typename LpNormReturnType

::Type(_expression()); + } - template struct OppositeExtendedType { - typedef Replicate Type; - }; - /** \internal - * Replicates a vector in the opposite direction to match the size of \c *this */ - template - EIGEN_DEVICE_FUNC - typename OppositeExtendedType::Type - extendedToOpposite(const DenseBase& other) const - { - EIGEN_STATIC_ASSERT(EIGEN_IMPLIES(isHorizontal, OtherDerived::MaxColsAtCompileTime==1), - YOU_PASSED_A_ROW_VECTOR_BUT_A_COLUMN_VECTOR_WAS_EXPECTED) - EIGEN_STATIC_ASSERT(EIGEN_IMPLIES(isVertical, OtherDerived::MaxRowsAtCompileTime==1), - YOU_PASSED_A_COLUMN_VECTOR_BUT_A_ROW_VECTOR_WAS_EXPECTED) - return typename OppositeExtendedType::Type - (other.derived(), - isHorizontal ? 1 : m_matrix.rows(), - isVertical ? 1 : m_matrix.cols()); - } + /** \returns a row (or column) vector expression of the norm + * of each column (or row) of the referenced expression, using + * Blue's algorithm. + * This is a vector with real entries, even if the original matrix has complex entries. + * + * \sa DenseBase::blueNorm() */ + EIGEN_DEVICE_FUNC + const BlueNormReturnType blueNorm() const { return BlueNormReturnType(_expression()); } + + + /** \returns a row (or column) vector expression of the norm + * of each column (or row) of the referenced expression, avoiding + * underflow and overflow. + * This is a vector with real entries, even if the original matrix has complex entries. + * + * \sa DenseBase::stableNorm() */ + EIGEN_DEVICE_FUNC + const StableNormReturnType stableNorm() const { return StableNormReturnType(_expression()); } + + + /** \returns a row (or column) vector expression of the norm + * of each column (or row) of the referenced expression, avoiding + * underflow and overflow using a concatenation of hypot() calls. + * This is a vector with real entries, even if the original matrix has complex entries. + * + * \sa DenseBase::hypotNorm() */ + EIGEN_DEVICE_FUNC + const HypotNormReturnType hypotNorm() const { return HypotNormReturnType(_expression()); } + + /** \returns a row (or column) vector expression of the sum + * of each column (or row) of the referenced expression. + * + * Example: \include PartialRedux_sum.cpp + * Output: \verbinclude PartialRedux_sum.out + * + * \sa DenseBase::sum() */ + EIGEN_DEVICE_FUNC + const SumReturnType sum() const { return SumReturnType(_expression()); } + + /** \returns a row (or column) vector expression of the mean + * of each column (or row) of the referenced expression. + * + * \sa DenseBase::mean() */ + EIGEN_DEVICE_FUNC + const MeanReturnType mean() const { return MeanReturnType(_expression()); } + + /** \returns a row (or column) vector expression representing + * whether \b all coefficients of each respective column (or row) are \c true. + * This expression can be assigned to a vector with entries of type \c bool. + * + * \sa DenseBase::all() */ + EIGEN_DEVICE_FUNC + const AllReturnType all() const { return AllReturnType(_expression()); } + + /** \returns a row (or column) vector expression representing + * whether \b at \b least one coefficient of each respective column (or row) is \c true. + * This expression can be assigned to a vector with entries of type \c bool. + * + * \sa DenseBase::any() */ + EIGEN_DEVICE_FUNC + const AnyReturnType any() const { return AnyReturnType(_expression()); } + + /** \returns a row (or column) vector expression representing + * the number of \c true coefficients of each respective column (or row). + * This expression can be assigned to a vector whose entries have the same type as is used to + * index entries of the original matrix; for dense matrices, this is \c std::ptrdiff_t . + * + * Example: \include PartialRedux_count.cpp + * Output: \verbinclude PartialRedux_count.out + * + * \sa DenseBase::count() */ + EIGEN_DEVICE_FUNC + const CountReturnType count() const { return CountReturnType(_expression()); } + + /** \returns a row (or column) vector expression of the product + * of each column (or row) of the referenced expression. + * + * Example: \include PartialRedux_prod.cpp + * Output: \verbinclude PartialRedux_prod.out + * + * \sa DenseBase::prod() */ + EIGEN_DEVICE_FUNC + const ProdReturnType prod() const { return ProdReturnType(_expression()); } + + + /** \returns a matrix expression + * where each column (or row) are reversed. + * + * Example: \include Vectorwise_reverse.cpp + * Output: \verbinclude Vectorwise_reverse.out + * + * \sa DenseBase::reverse() */ + EIGEN_DEVICE_FUNC + const ConstReverseReturnType reverse() const { return ConstReverseReturnType(_expression()); } + + /** \returns a writable matrix expression + * where each column (or row) are reversed. + * + * \sa reverse() const */ + EIGEN_DEVICE_FUNC + ReverseReturnType reverse() { return ReverseReturnType(_expression()); } + + typedef Replicate ReplicateReturnType; + EIGEN_DEVICE_FUNC + const ReplicateReturnType replicate(Index factor) const; + + /** + * \return an expression of the replication of each column (or row) of \c *this + * + * Example: \include DirectionWise_replicate.cpp + * Output: \verbinclude DirectionWise_replicate.out + * + * \sa VectorwiseOp::replicate(Index), DenseBase::replicate(), class Replicate + */ + // NOTE implemented here because of sunstudio's compilation errors + // isVertical*Factor+isHorizontal instead of (isVertical?Factor:1) to handle CUDA bug with ternary operator + template + const Replicate + EIGEN_DEVICE_FUNC replicate(Index factor = Factor) const + { + return Replicate( + _expression(), isVertical ? factor : 1, isHorizontal ? factor : 1); + } - public: - EIGEN_DEVICE_FUNC - explicit inline VectorwiseOp(ExpressionType& matrix) : m_matrix(matrix) {} - - /** \internal */ - EIGEN_DEVICE_FUNC - inline const ExpressionType& _expression() const { return m_matrix; } - - /** \returns a row or column vector expression of \c *this reduxed by \a func - * - * The template parameter \a BinaryOp is the type of the functor - * of the custom redux operator. Note that func must be an associative operator. - * - * \sa class VectorwiseOp, DenseBase::colwise(), DenseBase::rowwise() - */ - template - EIGEN_DEVICE_FUNC - const typename ReduxReturnType::Type - redux(const BinaryOp& func = BinaryOp()) const - { return typename ReduxReturnType::Type(_expression(), internal::member_redux(func)); } - - typedef typename ReturnType::Type MinCoeffReturnType; - typedef typename ReturnType::Type MaxCoeffReturnType; - typedef typename ReturnType::Type SquaredNormReturnType; - typedef typename ReturnType::Type NormReturnType; - typedef typename ReturnType::Type BlueNormReturnType; - typedef typename ReturnType::Type StableNormReturnType; - typedef typename ReturnType::Type HypotNormReturnType; - typedef typename ReturnType::Type SumReturnType; - typedef typename ReturnType::Type MeanReturnType; - typedef typename ReturnType::Type AllReturnType; - typedef typename ReturnType::Type AnyReturnType; - typedef PartialReduxExpr, Direction> CountReturnType; - typedef typename ReturnType::Type ProdReturnType; - typedef Reverse ConstReverseReturnType; - typedef Reverse ReverseReturnType; - - template struct LpNormReturnType { - typedef PartialReduxExpr,Direction> Type; - }; + /////////// Artithmetic operators /////////// - /** \returns a row (or column) vector expression of the smallest coefficient - * of each column (or row) of the referenced expression. - * - * \warning the result is undefined if \c *this contains NaN. - * - * Example: \include PartialRedux_minCoeff.cpp - * Output: \verbinclude PartialRedux_minCoeff.out - * - * \sa DenseBase::minCoeff() */ - EIGEN_DEVICE_FUNC - const MinCoeffReturnType minCoeff() const - { return MinCoeffReturnType(_expression()); } - - /** \returns a row (or column) vector expression of the largest coefficient - * of each column (or row) of the referenced expression. - * - * \warning the result is undefined if \c *this contains NaN. - * - * Example: \include PartialRedux_maxCoeff.cpp - * Output: \verbinclude PartialRedux_maxCoeff.out - * - * \sa DenseBase::maxCoeff() */ - EIGEN_DEVICE_FUNC - const MaxCoeffReturnType maxCoeff() const - { return MaxCoeffReturnType(_expression()); } - - /** \returns a row (or column) vector expression of the squared norm - * of each column (or row) of the referenced expression. - * This is a vector with real entries, even if the original matrix has complex entries. - * - * Example: \include PartialRedux_squaredNorm.cpp - * Output: \verbinclude PartialRedux_squaredNorm.out - * - * \sa DenseBase::squaredNorm() */ - EIGEN_DEVICE_FUNC - const SquaredNormReturnType squaredNorm() const - { return SquaredNormReturnType(_expression()); } - - /** \returns a row (or column) vector expression of the norm - * of each column (or row) of the referenced expression. - * This is a vector with real entries, even if the original matrix has complex entries. - * - * Example: \include PartialRedux_norm.cpp - * Output: \verbinclude PartialRedux_norm.out - * - * \sa DenseBase::norm() */ - EIGEN_DEVICE_FUNC - const NormReturnType norm() const - { return NormReturnType(_expression()); } - - /** \returns a row (or column) vector expression of the norm - * of each column (or row) of the referenced expression. - * This is a vector with real entries, even if the original matrix has complex entries. - * - * Example: \include PartialRedux_norm.cpp - * Output: \verbinclude PartialRedux_norm.out - * - * \sa DenseBase::norm() */ - template - EIGEN_DEVICE_FUNC - const typename LpNormReturnType

::Type lpNorm() const - { return typename LpNormReturnType

::Type(_expression()); } - - - /** \returns a row (or column) vector expression of the norm - * of each column (or row) of the referenced expression, using - * Blue's algorithm. - * This is a vector with real entries, even if the original matrix has complex entries. - * - * \sa DenseBase::blueNorm() */ - EIGEN_DEVICE_FUNC - const BlueNormReturnType blueNorm() const - { return BlueNormReturnType(_expression()); } - - - /** \returns a row (or column) vector expression of the norm - * of each column (or row) of the referenced expression, avoiding - * underflow and overflow. - * This is a vector with real entries, even if the original matrix has complex entries. - * - * \sa DenseBase::stableNorm() */ - EIGEN_DEVICE_FUNC - const StableNormReturnType stableNorm() const - { return StableNormReturnType(_expression()); } - - - /** \returns a row (or column) vector expression of the norm - * of each column (or row) of the referenced expression, avoiding - * underflow and overflow using a concatenation of hypot() calls. - * This is a vector with real entries, even if the original matrix has complex entries. - * - * \sa DenseBase::hypotNorm() */ - EIGEN_DEVICE_FUNC - const HypotNormReturnType hypotNorm() const - { return HypotNormReturnType(_expression()); } - - /** \returns a row (or column) vector expression of the sum - * of each column (or row) of the referenced expression. - * - * Example: \include PartialRedux_sum.cpp - * Output: \verbinclude PartialRedux_sum.out - * - * \sa DenseBase::sum() */ - EIGEN_DEVICE_FUNC - const SumReturnType sum() const - { return SumReturnType(_expression()); } - - /** \returns a row (or column) vector expression of the mean - * of each column (or row) of the referenced expression. - * - * \sa DenseBase::mean() */ - EIGEN_DEVICE_FUNC - const MeanReturnType mean() const - { return MeanReturnType(_expression()); } - - /** \returns a row (or column) vector expression representing - * whether \b all coefficients of each respective column (or row) are \c true. - * This expression can be assigned to a vector with entries of type \c bool. - * - * \sa DenseBase::all() */ - EIGEN_DEVICE_FUNC - const AllReturnType all() const - { return AllReturnType(_expression()); } - - /** \returns a row (or column) vector expression representing - * whether \b at \b least one coefficient of each respective column (or row) is \c true. - * This expression can be assigned to a vector with entries of type \c bool. - * - * \sa DenseBase::any() */ - EIGEN_DEVICE_FUNC - const AnyReturnType any() const - { return AnyReturnType(_expression()); } - - /** \returns a row (or column) vector expression representing - * the number of \c true coefficients of each respective column (or row). - * This expression can be assigned to a vector whose entries have the same type as is used to - * index entries of the original matrix; for dense matrices, this is \c std::ptrdiff_t . - * - * Example: \include PartialRedux_count.cpp - * Output: \verbinclude PartialRedux_count.out - * - * \sa DenseBase::count() */ - EIGEN_DEVICE_FUNC - const CountReturnType count() const - { return CountReturnType(_expression()); } - - /** \returns a row (or column) vector expression of the product - * of each column (or row) of the referenced expression. - * - * Example: \include PartialRedux_prod.cpp - * Output: \verbinclude PartialRedux_prod.out - * - * \sa DenseBase::prod() */ - EIGEN_DEVICE_FUNC - const ProdReturnType prod() const - { return ProdReturnType(_expression()); } - - - /** \returns a matrix expression - * where each column (or row) are reversed. - * - * Example: \include Vectorwise_reverse.cpp - * Output: \verbinclude Vectorwise_reverse.out - * - * \sa DenseBase::reverse() */ - EIGEN_DEVICE_FUNC - const ConstReverseReturnType reverse() const - { return ConstReverseReturnType( _expression() ); } - - /** \returns a writable matrix expression - * where each column (or row) are reversed. - * - * \sa reverse() const */ - EIGEN_DEVICE_FUNC - ReverseReturnType reverse() - { return ReverseReturnType( _expression() ); } - - typedef Replicate ReplicateReturnType; - EIGEN_DEVICE_FUNC - const ReplicateReturnType replicate(Index factor) const; - - /** - * \return an expression of the replication of each column (or row) of \c *this - * - * Example: \include DirectionWise_replicate.cpp - * Output: \verbinclude DirectionWise_replicate.out - * - * \sa VectorwiseOp::replicate(Index), DenseBase::replicate(), class Replicate - */ - // NOTE implemented here because of sunstudio's compilation errors - // isVertical*Factor+isHorizontal instead of (isVertical?Factor:1) to handle CUDA bug with ternary operator - template const Replicate - EIGEN_DEVICE_FUNC - replicate(Index factor = Factor) const - { - return Replicate - (_expression(),isVertical?factor:1,isHorizontal?factor:1); - } + /** Copies the vector \a other to each subvector of \c *this */ + template EIGEN_DEVICE_FUNC ExpressionType &operator=(const DenseBase &other) + { + EIGEN_STATIC_ASSERT_VECTOR_ONLY(OtherDerived) + EIGEN_STATIC_ASSERT_SAME_XPR_KIND(ExpressionType, OtherDerived) + // eigen_assert((m_matrix.isNull()) == (other.isNull())); FIXME + return const_cast(m_matrix = extendedTo(other.derived())); + } -/////////// Artithmetic operators /////////// + /** Adds the vector \a other to each subvector of \c *this */ + template EIGEN_DEVICE_FUNC ExpressionType &operator+=(const DenseBase &other) + { + EIGEN_STATIC_ASSERT_VECTOR_ONLY(OtherDerived) + EIGEN_STATIC_ASSERT_SAME_XPR_KIND(ExpressionType, OtherDerived) + return const_cast(m_matrix += extendedTo(other.derived())); + } - /** Copies the vector \a other to each subvector of \c *this */ - template - EIGEN_DEVICE_FUNC - ExpressionType& operator=(const DenseBase& other) - { - EIGEN_STATIC_ASSERT_VECTOR_ONLY(OtherDerived) - EIGEN_STATIC_ASSERT_SAME_XPR_KIND(ExpressionType, OtherDerived) - //eigen_assert((m_matrix.isNull()) == (other.isNull())); FIXME - return const_cast(m_matrix = extendedTo(other.derived())); - } + /** Substracts the vector \a other to each subvector of \c *this */ + template EIGEN_DEVICE_FUNC ExpressionType &operator-=(const DenseBase &other) + { + EIGEN_STATIC_ASSERT_VECTOR_ONLY(OtherDerived) + EIGEN_STATIC_ASSERT_SAME_XPR_KIND(ExpressionType, OtherDerived) + return const_cast(m_matrix -= extendedTo(other.derived())); + } - /** Adds the vector \a other to each subvector of \c *this */ - template - EIGEN_DEVICE_FUNC - ExpressionType& operator+=(const DenseBase& other) - { - EIGEN_STATIC_ASSERT_VECTOR_ONLY(OtherDerived) - EIGEN_STATIC_ASSERT_SAME_XPR_KIND(ExpressionType, OtherDerived) - return const_cast(m_matrix += extendedTo(other.derived())); - } + /** Multiples each subvector of \c *this by the vector \a other */ + template EIGEN_DEVICE_FUNC ExpressionType &operator*=(const DenseBase &other) + { + EIGEN_STATIC_ASSERT_VECTOR_ONLY(OtherDerived) + EIGEN_STATIC_ASSERT_ARRAYXPR(ExpressionType) + EIGEN_STATIC_ASSERT_SAME_XPR_KIND(ExpressionType, OtherDerived) + m_matrix *= extendedTo(other.derived()); + return const_cast(m_matrix); + } - /** Substracts the vector \a other to each subvector of \c *this */ - template - EIGEN_DEVICE_FUNC - ExpressionType& operator-=(const DenseBase& other) - { - EIGEN_STATIC_ASSERT_VECTOR_ONLY(OtherDerived) - EIGEN_STATIC_ASSERT_SAME_XPR_KIND(ExpressionType, OtherDerived) - return const_cast(m_matrix -= extendedTo(other.derived())); - } + /** Divides each subvector of \c *this by the vector \a other */ + template EIGEN_DEVICE_FUNC ExpressionType &operator/=(const DenseBase &other) + { + EIGEN_STATIC_ASSERT_VECTOR_ONLY(OtherDerived) + EIGEN_STATIC_ASSERT_ARRAYXPR(ExpressionType) + EIGEN_STATIC_ASSERT_SAME_XPR_KIND(ExpressionType, OtherDerived) + m_matrix /= extendedTo(other.derived()); + return const_cast(m_matrix); + } - /** Multiples each subvector of \c *this by the vector \a other */ - template - EIGEN_DEVICE_FUNC - ExpressionType& operator*=(const DenseBase& other) - { - EIGEN_STATIC_ASSERT_VECTOR_ONLY(OtherDerived) - EIGEN_STATIC_ASSERT_ARRAYXPR(ExpressionType) - EIGEN_STATIC_ASSERT_SAME_XPR_KIND(ExpressionType, OtherDerived) - m_matrix *= extendedTo(other.derived()); - return const_cast(m_matrix); - } + /** Returns the expression of the sum of the vector \a other to each subvector of \c *this */ + template + EIGEN_STRONG_INLINE EIGEN_DEVICE_FUNC CwiseBinaryOp, + const ExpressionTypeNestedCleaned, + const typename ExtendedType::Type> + operator+(const DenseBase &other) const + { + EIGEN_STATIC_ASSERT_VECTOR_ONLY(OtherDerived) + EIGEN_STATIC_ASSERT_SAME_XPR_KIND(ExpressionType, OtherDerived) + return m_matrix + extendedTo(other.derived()); + } - /** Divides each subvector of \c *this by the vector \a other */ - template - EIGEN_DEVICE_FUNC - ExpressionType& operator/=(const DenseBase& other) - { - EIGEN_STATIC_ASSERT_VECTOR_ONLY(OtherDerived) - EIGEN_STATIC_ASSERT_ARRAYXPR(ExpressionType) - EIGEN_STATIC_ASSERT_SAME_XPR_KIND(ExpressionType, OtherDerived) - m_matrix /= extendedTo(other.derived()); - return const_cast(m_matrix); - } + /** Returns the expression of the difference between each subvector of \c *this and the vector \a other */ + template + EIGEN_DEVICE_FUNC CwiseBinaryOp, + const ExpressionTypeNestedCleaned, + const typename ExtendedType::Type> + operator-(const DenseBase &other) const + { + EIGEN_STATIC_ASSERT_VECTOR_ONLY(OtherDerived) + EIGEN_STATIC_ASSERT_SAME_XPR_KIND(ExpressionType, OtherDerived) + return m_matrix - extendedTo(other.derived()); + } - /** Returns the expression of the sum of the vector \a other to each subvector of \c *this */ - template EIGEN_STRONG_INLINE EIGEN_DEVICE_FUNC - CwiseBinaryOp, - const ExpressionTypeNestedCleaned, - const typename ExtendedType::Type> - operator+(const DenseBase& other) const - { - EIGEN_STATIC_ASSERT_VECTOR_ONLY(OtherDerived) - EIGEN_STATIC_ASSERT_SAME_XPR_KIND(ExpressionType, OtherDerived) - return m_matrix + extendedTo(other.derived()); - } + /** Returns the expression where each subvector is the product of the vector \a other + * by the corresponding subvector of \c *this */ + template + EIGEN_STRONG_INLINE EIGEN_DEVICE_FUNC CwiseBinaryOp, + const ExpressionTypeNestedCleaned, + const typename ExtendedType::Type> + EIGEN_DEVICE_FUNC operator*(const DenseBase &other) const + { + EIGEN_STATIC_ASSERT_VECTOR_ONLY(OtherDerived) + EIGEN_STATIC_ASSERT_ARRAYXPR(ExpressionType) + EIGEN_STATIC_ASSERT_SAME_XPR_KIND(ExpressionType, OtherDerived) + return m_matrix * extendedTo(other.derived()); + } - /** Returns the expression of the difference between each subvector of \c *this and the vector \a other */ - template - EIGEN_DEVICE_FUNC - CwiseBinaryOp, - const ExpressionTypeNestedCleaned, - const typename ExtendedType::Type> - operator-(const DenseBase& other) const - { - EIGEN_STATIC_ASSERT_VECTOR_ONLY(OtherDerived) - EIGEN_STATIC_ASSERT_SAME_XPR_KIND(ExpressionType, OtherDerived) - return m_matrix - extendedTo(other.derived()); - } + /** Returns the expression where each subvector is the quotient of the corresponding + * subvector of \c *this by the vector \a other */ + template + EIGEN_DEVICE_FUNC CwiseBinaryOp, + const ExpressionTypeNestedCleaned, + const typename ExtendedType::Type> + operator/(const DenseBase &other) const + { + EIGEN_STATIC_ASSERT_VECTOR_ONLY(OtherDerived) + EIGEN_STATIC_ASSERT_ARRAYXPR(ExpressionType) + EIGEN_STATIC_ASSERT_SAME_XPR_KIND(ExpressionType, OtherDerived) + return m_matrix / extendedTo(other.derived()); + } - /** Returns the expression where each subvector is the product of the vector \a other - * by the corresponding subvector of \c *this */ - template EIGEN_STRONG_INLINE EIGEN_DEVICE_FUNC - CwiseBinaryOp, - const ExpressionTypeNestedCleaned, - const typename ExtendedType::Type> - EIGEN_DEVICE_FUNC - operator*(const DenseBase& other) const - { - EIGEN_STATIC_ASSERT_VECTOR_ONLY(OtherDerived) - EIGEN_STATIC_ASSERT_ARRAYXPR(ExpressionType) - EIGEN_STATIC_ASSERT_SAME_XPR_KIND(ExpressionType, OtherDerived) - return m_matrix * extendedTo(other.derived()); - } + /** \returns an expression where each column (or row) of the referenced matrix are normalized. + * The referenced matrix is \b not modified. + * \sa MatrixBase::normalized(), normalize() + */ + EIGEN_DEVICE_FUNC + CwiseBinaryOp, + const ExpressionTypeNestedCleaned, + const typename OppositeExtendedType::Type>::Type> + normalized() const + { + return m_matrix.cwiseQuotient(extendedToOpposite(this->norm())); + } - /** Returns the expression where each subvector is the quotient of the corresponding - * subvector of \c *this by the vector \a other */ - template - EIGEN_DEVICE_FUNC - CwiseBinaryOp, - const ExpressionTypeNestedCleaned, - const typename ExtendedType::Type> - operator/(const DenseBase& other) const - { - EIGEN_STATIC_ASSERT_VECTOR_ONLY(OtherDerived) - EIGEN_STATIC_ASSERT_ARRAYXPR(ExpressionType) - EIGEN_STATIC_ASSERT_SAME_XPR_KIND(ExpressionType, OtherDerived) - return m_matrix / extendedTo(other.derived()); - } - /** \returns an expression where each column (or row) of the referenced matrix are normalized. - * The referenced matrix is \b not modified. - * \sa MatrixBase::normalized(), normalize() - */ - EIGEN_DEVICE_FUNC - CwiseBinaryOp, - const ExpressionTypeNestedCleaned, - const typename OppositeExtendedType::Type>::Type> - normalized() const { return m_matrix.cwiseQuotient(extendedToOpposite(this->norm())); } - - - /** Normalize in-place each row or columns of the referenced matrix. - * \sa MatrixBase::normalize(), normalized() - */ - EIGEN_DEVICE_FUNC void normalize() { - m_matrix = this->normalized(); - } + /** Normalize in-place each row or columns of the referenced matrix. + * \sa MatrixBase::normalize(), normalized() + */ + EIGEN_DEVICE_FUNC void normalize() { m_matrix = this->normalized(); } - EIGEN_DEVICE_FUNC inline void reverseInPlace(); + EIGEN_DEVICE_FUNC inline void reverseInPlace(); -/////////// Geometry module /////////// + /////////// Geometry module /////////// - typedef Homogeneous HomogeneousReturnType; - EIGEN_DEVICE_FUNC - HomogeneousReturnType homogeneous() const; + typedef Homogeneous HomogeneousReturnType; + EIGEN_DEVICE_FUNC + HomogeneousReturnType homogeneous() const; - typedef typename ExpressionType::PlainObject CrossReturnType; - template - EIGEN_DEVICE_FUNC - const CrossReturnType cross(const MatrixBase& other) const; + typedef typename ExpressionType::PlainObject CrossReturnType; + template + EIGEN_DEVICE_FUNC const CrossReturnType cross(const MatrixBase &other) const; - enum { - HNormalized_Size = Direction==Vertical ? internal::traits::RowsAtCompileTime + enum { + HNormalized_Size = Direction == Vertical ? internal::traits::RowsAtCompileTime : internal::traits::ColsAtCompileTime, - HNormalized_SizeMinusOne = HNormalized_Size==Dynamic ? Dynamic : HNormalized_Size-1 - }; - typedef Block::RowsAtCompileTime), - Direction==Horizontal ? int(HNormalized_SizeMinusOne) - : int(internal::traits::ColsAtCompileTime)> - HNormalized_Block; - typedef Block::RowsAtCompileTime), - Direction==Horizontal ? 1 : int(internal::traits::ColsAtCompileTime)> - HNormalized_Factors; - typedef CwiseBinaryOp::Scalar>, - const HNormalized_Block, - const Replicate > - HNormalizedReturnType; - - EIGEN_DEVICE_FUNC - const HNormalizedReturnType hnormalized() const; - - protected: - ExpressionTypeNested m_matrix; + HNormalized_SizeMinusOne = HNormalized_Size == Dynamic ? Dynamic : HNormalized_Size - 1 + }; + typedef Block::RowsAtCompileTime), + Direction == Horizontal ? int(HNormalized_SizeMinusOne) : int(internal::traits::ColsAtCompileTime)> + HNormalized_Block; + typedef Block::RowsAtCompileTime), + Direction == Horizontal ? 1 : int(internal::traits::ColsAtCompileTime)> + HNormalized_Factors; + typedef CwiseBinaryOp::Scalar>, + const HNormalized_Block, + const Replicate> + HNormalizedReturnType; + + EIGEN_DEVICE_FUNC + const HNormalizedReturnType hnormalized() const; + +protected: + ExpressionTypeNested m_matrix; }; -//const colwise moved to DenseBase.h due to CUDA compiler bug +// const colwise moved to DenseBase.h due to CUDA compiler bug /** \returns a writable VectorwiseOp wrapper of *this providing additional partial reduction operations - * - * \sa rowwise(), class VectorwiseOp, \ref TutorialReductionsVisitorsBroadcasting - */ -template -inline typename DenseBase::ColwiseReturnType -DenseBase::colwise() + * + * \sa rowwise(), class VectorwiseOp, \ref TutorialReductionsVisitorsBroadcasting + */ +template inline typename DenseBase::ColwiseReturnType DenseBase::colwise() { return ColwiseReturnType(derived()); } -//const rowwise moved to DenseBase.h due to CUDA compiler bug +// const rowwise moved to DenseBase.h due to CUDA compiler bug /** \returns a writable VectorwiseOp wrapper of *this providing additional partial reduction operations - * - * \sa colwise(), class VectorwiseOp, \ref TutorialReductionsVisitorsBroadcasting - */ -template -inline typename DenseBase::RowwiseReturnType -DenseBase::rowwise() + * + * \sa colwise(), class VectorwiseOp, \ref TutorialReductionsVisitorsBroadcasting + */ +template inline typename DenseBase::RowwiseReturnType DenseBase::rowwise() { return RowwiseReturnType(derived()); } -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_PARTIAL_REDUX_H +#endif// EIGEN_PARTIAL_REDUX_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/Visitor.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/Visitor.h index 54c1883d..74bf9b1e 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/Visitor.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/Visitor.h @@ -10,201 +10,180 @@ #ifndef EIGEN_VISITOR_H #define EIGEN_VISITOR_H -namespace Eigen { +namespace Eigen { namespace internal { -template -struct visitor_impl -{ - enum { - col = (UnrollCount-1) / Derived::RowsAtCompileTime, - row = (UnrollCount-1) % Derived::RowsAtCompileTime + template struct visitor_impl + { + enum { col = (UnrollCount - 1) / Derived::RowsAtCompileTime, row = (UnrollCount - 1) % Derived::RowsAtCompileTime }; + + EIGEN_DEVICE_FUNC + static inline void run(const Derived &mat, Visitor &visitor) + { + visitor_impl::run(mat, visitor); + visitor(mat.coeff(row, col), row, col); + } }; - EIGEN_DEVICE_FUNC - static inline void run(const Derived &mat, Visitor& visitor) + template struct visitor_impl { - visitor_impl::run(mat, visitor); - visitor(mat.coeff(row, col), row, col); - } -}; + EIGEN_DEVICE_FUNC + static inline void run(const Derived &mat, Visitor &visitor) { return visitor.init(mat.coeff(0, 0), 0, 0); } + }; -template -struct visitor_impl -{ - EIGEN_DEVICE_FUNC - static inline void run(const Derived &mat, Visitor& visitor) + template struct visitor_impl { - return visitor.init(mat.coeff(0, 0), 0, 0); - } -}; + EIGEN_DEVICE_FUNC + static inline void run(const Derived &mat, Visitor &visitor) + { + visitor.init(mat.coeff(0, 0), 0, 0); + for (Index i = 1; i < mat.rows(); ++i) visitor(mat.coeff(i, 0), i, 0); + for (Index j = 1; j < mat.cols(); ++j) + for (Index i = 0; i < mat.rows(); ++i) visitor(mat.coeff(i, j), i, j); + } + }; -template -struct visitor_impl -{ - EIGEN_DEVICE_FUNC - static inline void run(const Derived& mat, Visitor& visitor) + // evaluator adaptor + template class visitor_evaluator { - visitor.init(mat.coeff(0,0), 0, 0); - for(Index i = 1; i < mat.rows(); ++i) - visitor(mat.coeff(i, 0), i, 0); - for(Index j = 1; j < mat.cols(); ++j) - for(Index i = 0; i < mat.rows(); ++i) - visitor(mat.coeff(i, j), i, j); - } -}; + public: + EIGEN_DEVICE_FUNC + explicit visitor_evaluator(const XprType &xpr) : m_evaluator(xpr), m_xpr(xpr) {} -// evaluator adaptor -template -class visitor_evaluator -{ -public: - EIGEN_DEVICE_FUNC - explicit visitor_evaluator(const XprType &xpr) : m_evaluator(xpr), m_xpr(xpr) {} - - typedef typename XprType::Scalar Scalar; - typedef typename XprType::CoeffReturnType CoeffReturnType; - - enum { - RowsAtCompileTime = XprType::RowsAtCompileTime, - CoeffReadCost = internal::evaluator::CoeffReadCost - }; - - EIGEN_DEVICE_FUNC Index rows() const { return m_xpr.rows(); } - EIGEN_DEVICE_FUNC Index cols() const { return m_xpr.cols(); } - EIGEN_DEVICE_FUNC Index size() const { return m_xpr.size(); } + typedef typename XprType::Scalar Scalar; + typedef typename XprType::CoeffReturnType CoeffReturnType; + + enum { + RowsAtCompileTime = XprType::RowsAtCompileTime, + CoeffReadCost = internal::evaluator::CoeffReadCost + }; - EIGEN_DEVICE_FUNC CoeffReturnType coeff(Index row, Index col) const - { return m_evaluator.coeff(row, col); } - -protected: - internal::evaluator m_evaluator; - const XprType &m_xpr; -}; -} // end namespace internal + EIGEN_DEVICE_FUNC Index rows() const { return m_xpr.rows(); } + EIGEN_DEVICE_FUNC Index cols() const { return m_xpr.cols(); } + EIGEN_DEVICE_FUNC Index size() const { return m_xpr.size(); } + + EIGEN_DEVICE_FUNC CoeffReturnType coeff(Index row, Index col) const { return m_evaluator.coeff(row, col); } + + protected: + internal::evaluator m_evaluator; + const XprType &m_xpr; + }; +}// end namespace internal /** Applies the visitor \a visitor to the whole coefficients of the matrix or vector. - * - * The template parameter \a Visitor is the type of the visitor and provides the following interface: - * \code - * struct MyVisitor { - * // called for the first coefficient - * void init(const Scalar& value, Index i, Index j); - * // called for all other coefficients - * void operator() (const Scalar& value, Index i, Index j); - * }; - * \endcode - * - * \note compared to one or two \em for \em loops, visitors offer automatic - * unrolling for small fixed size matrix. - * - * \sa minCoeff(Index*,Index*), maxCoeff(Index*,Index*), DenseBase::redux() - */ + * + * The template parameter \a Visitor is the type of the visitor and provides the following interface: + * \code + * struct MyVisitor { + * // called for the first coefficient + * void init(const Scalar& value, Index i, Index j); + * // called for all other coefficients + * void operator() (const Scalar& value, Index i, Index j); + * }; + * \endcode + * + * \note compared to one or two \em for \em loops, visitors offer automatic + * unrolling for small fixed size matrix. + * + * \sa minCoeff(Index*,Index*), maxCoeff(Index*,Index*), DenseBase::redux() + */ template template -EIGEN_DEVICE_FUNC -void DenseBase::visit(Visitor& visitor) const +EIGEN_DEVICE_FUNC void DenseBase::visit(Visitor &visitor) const { typedef typename internal::visitor_evaluator ThisEvaluator; ThisEvaluator thisEval(derived()); - + enum { - unroll = SizeAtCompileTime != Dynamic - && SizeAtCompileTime * ThisEvaluator::CoeffReadCost + (SizeAtCompileTime-1) * internal::functor_traits::Cost <= EIGEN_UNROLLING_LIMIT + unroll = SizeAtCompileTime != Dynamic + && SizeAtCompileTime * ThisEvaluator::CoeffReadCost + + (SizeAtCompileTime - 1) * internal::functor_traits::Cost + <= EIGEN_UNROLLING_LIMIT }; - return internal::visitor_impl::run(thisEval, visitor); + return internal::visitor_impl < Visitor, ThisEvaluator, + unroll ? int(SizeAtCompileTime) : Dynamic > ::run(thisEval, visitor); } namespace internal { -/** \internal - * \brief Base class to implement min and max visitors - */ -template -struct coeff_visitor -{ - typedef typename Derived::Scalar Scalar; - Index row, col; - Scalar res; - EIGEN_DEVICE_FUNC - inline void init(const Scalar& value, Index i, Index j) + /** \internal + * \brief Base class to implement min and max visitors + */ + template struct coeff_visitor { - res = value; - row = i; - col = j; - } -}; + typedef typename Derived::Scalar Scalar; + Index row, col; + Scalar res; + EIGEN_DEVICE_FUNC + inline void init(const Scalar &value, Index i, Index j) + { + res = value; + row = i; + col = j; + } + }; -/** \internal - * \brief Visitor computing the min coefficient with its value and coordinates - * - * \sa DenseBase::minCoeff(Index*, Index*) - */ -template -struct min_coeff_visitor : coeff_visitor -{ - typedef typename Derived::Scalar Scalar; - EIGEN_DEVICE_FUNC - void operator() (const Scalar& value, Index i, Index j) + /** \internal + * \brief Visitor computing the min coefficient with its value and coordinates + * + * \sa DenseBase::minCoeff(Index*, Index*) + */ + template struct min_coeff_visitor : coeff_visitor { - if(value < this->res) + typedef typename Derived::Scalar Scalar; + EIGEN_DEVICE_FUNC + void operator()(const Scalar &value, Index i, Index j) { - this->res = value; - this->row = i; - this->col = j; + if (value < this->res) { + this->res = value; + this->row = i; + this->col = j; + } } - } -}; + }; -template -struct functor_traits > { - enum { - Cost = NumTraits::AddCost + template struct functor_traits> + { + enum { Cost = NumTraits::AddCost }; }; -}; -/** \internal - * \brief Visitor computing the max coefficient with its value and coordinates - * - * \sa DenseBase::maxCoeff(Index*, Index*) - */ -template -struct max_coeff_visitor : coeff_visitor -{ - typedef typename Derived::Scalar Scalar; - EIGEN_DEVICE_FUNC - void operator() (const Scalar& value, Index i, Index j) + /** \internal + * \brief Visitor computing the max coefficient with its value and coordinates + * + * \sa DenseBase::maxCoeff(Index*, Index*) + */ + template struct max_coeff_visitor : coeff_visitor { - if(value > this->res) + typedef typename Derived::Scalar Scalar; + EIGEN_DEVICE_FUNC + void operator()(const Scalar &value, Index i, Index j) { - this->res = value; - this->row = i; - this->col = j; + if (value > this->res) { + this->res = value; + this->row = i; + this->col = j; + } } - } -}; + }; -template -struct functor_traits > { - enum { - Cost = NumTraits::AddCost + template struct functor_traits> + { + enum { Cost = NumTraits::AddCost }; }; -}; -} // end namespace internal +}// end namespace internal /** \fn DenseBase::minCoeff(IndexType* rowId, IndexType* colId) const - * \returns the minimum of all coefficients of *this and puts in *row and *col its location. - * \warning the result is undefined if \c *this contains NaN. - * - * \sa DenseBase::minCoeff(Index*), DenseBase::maxCoeff(Index*,Index*), DenseBase::visit(), DenseBase::minCoeff() - */ + * \returns the minimum of all coefficients of *this and puts in *row and *col its location. + * \warning the result is undefined if \c *this contains NaN. + * + * \sa DenseBase::minCoeff(Index*), DenseBase::maxCoeff(Index*,Index*), DenseBase::visit(), DenseBase::minCoeff() + */ template template -EIGEN_DEVICE_FUNC -typename internal::traits::Scalar -DenseBase::minCoeff(IndexType* rowId, IndexType* colId) const +EIGEN_DEVICE_FUNC typename internal::traits::Scalar DenseBase::minCoeff(IndexType *rowId, + IndexType *colId) const { internal::min_coeff_visitor minVisitor; this->visit(minVisitor); @@ -214,34 +193,32 @@ DenseBase::minCoeff(IndexType* rowId, IndexType* colId) const } /** \returns the minimum of all coefficients of *this and puts in *index its location. - * \warning the result is undefined if \c *this contains NaN. - * - * \sa DenseBase::minCoeff(IndexType*,IndexType*), DenseBase::maxCoeff(IndexType*,IndexType*), DenseBase::visit(), DenseBase::minCoeff() - */ + * \warning the result is undefined if \c *this contains NaN. + * + * \sa DenseBase::minCoeff(IndexType*,IndexType*), DenseBase::maxCoeff(IndexType*,IndexType*), DenseBase::visit(), + * DenseBase::minCoeff() + */ template template -EIGEN_DEVICE_FUNC -typename internal::traits::Scalar -DenseBase::minCoeff(IndexType* index) const +EIGEN_DEVICE_FUNC typename internal::traits::Scalar DenseBase::minCoeff(IndexType *index) const { EIGEN_STATIC_ASSERT_VECTOR_ONLY(Derived) internal::min_coeff_visitor minVisitor; this->visit(minVisitor); - *index = IndexType((RowsAtCompileTime==1) ? minVisitor.col : minVisitor.row); + *index = IndexType((RowsAtCompileTime == 1) ? minVisitor.col : minVisitor.row); return minVisitor.res; } /** \fn DenseBase::maxCoeff(IndexType* rowId, IndexType* colId) const - * \returns the maximum of all coefficients of *this and puts in *row and *col its location. - * \warning the result is undefined if \c *this contains NaN. - * - * \sa DenseBase::minCoeff(IndexType*,IndexType*), DenseBase::visit(), DenseBase::maxCoeff() - */ + * \returns the maximum of all coefficients of *this and puts in *row and *col its location. + * \warning the result is undefined if \c *this contains NaN. + * + * \sa DenseBase::minCoeff(IndexType*,IndexType*), DenseBase::visit(), DenseBase::maxCoeff() + */ template template -EIGEN_DEVICE_FUNC -typename internal::traits::Scalar -DenseBase::maxCoeff(IndexType* rowPtr, IndexType* colPtr) const +EIGEN_DEVICE_FUNC typename internal::traits::Scalar DenseBase::maxCoeff(IndexType *rowPtr, + IndexType *colPtr) const { internal::max_coeff_visitor maxVisitor; this->visit(maxVisitor); @@ -251,23 +228,22 @@ DenseBase::maxCoeff(IndexType* rowPtr, IndexType* colPtr) const } /** \returns the maximum of all coefficients of *this and puts in *index its location. - * \warning the result is undefined if \c *this contains NaN. - * - * \sa DenseBase::maxCoeff(IndexType*,IndexType*), DenseBase::minCoeff(IndexType*,IndexType*), DenseBase::visitor(), DenseBase::maxCoeff() - */ + * \warning the result is undefined if \c *this contains NaN. + * + * \sa DenseBase::maxCoeff(IndexType*,IndexType*), DenseBase::minCoeff(IndexType*,IndexType*), DenseBase::visitor(), + * DenseBase::maxCoeff() + */ template template -EIGEN_DEVICE_FUNC -typename internal::traits::Scalar -DenseBase::maxCoeff(IndexType* index) const +EIGEN_DEVICE_FUNC typename internal::traits::Scalar DenseBase::maxCoeff(IndexType *index) const { EIGEN_STATIC_ASSERT_VECTOR_ONLY(Derived) internal::max_coeff_visitor maxVisitor; this->visit(maxVisitor); - *index = (RowsAtCompileTime==1) ? maxVisitor.col : maxVisitor.row; + *index = (RowsAtCompileTime == 1) ? maxVisitor.col : maxVisitor.row; return maxVisitor.res; } -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_VISITOR_H +#endif// EIGEN_VISITOR_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/arch/AVX/Complex.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/arch/AVX/Complex.h index 7fa61969..ad458b68 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/arch/AVX/Complex.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/arch/AVX/Complex.h @@ -14,438 +14,529 @@ namespace Eigen { namespace internal { -//---------- float ---------- -struct Packet4cf -{ - EIGEN_STRONG_INLINE Packet4cf() {} - EIGEN_STRONG_INLINE explicit Packet4cf(const __m256& a) : v(a) {} - __m256 v; -}; - -template<> struct packet_traits > : default_packet_traits -{ - typedef Packet4cf type; - typedef Packet2cf half; - enum { - Vectorizable = 1, - AlignedOnScalar = 1, - size = 4, - HasHalfPacket = 1, - - HasAdd = 1, - HasSub = 1, - HasMul = 1, - HasDiv = 1, - HasNegate = 1, - HasAbs = 0, - HasAbs2 = 0, - HasMin = 0, - HasMax = 0, - HasSetLinear = 0 + //---------- float ---------- + struct Packet4cf + { + EIGEN_STRONG_INLINE Packet4cf() {} + EIGEN_STRONG_INLINE explicit Packet4cf(const __m256 &a) : v(a) {} + __m256 v; + }; + + template<> struct packet_traits> : default_packet_traits + { + typedef Packet4cf type; + typedef Packet2cf half; + enum { + Vectorizable = 1, + AlignedOnScalar = 1, + size = 4, + HasHalfPacket = 1, + + HasAdd = 1, + HasSub = 1, + HasMul = 1, + HasDiv = 1, + HasNegate = 1, + HasAbs = 0, + HasAbs2 = 0, + HasMin = 0, + HasMax = 0, + HasSetLinear = 0 + }; }; -}; - -template<> struct unpacket_traits { typedef std::complex type; enum {size=4, alignment=Aligned32}; typedef Packet2cf half; }; - -template<> EIGEN_STRONG_INLINE Packet4cf padd(const Packet4cf& a, const Packet4cf& b) { return Packet4cf(_mm256_add_ps(a.v,b.v)); } -template<> EIGEN_STRONG_INLINE Packet4cf psub(const Packet4cf& a, const Packet4cf& b) { return Packet4cf(_mm256_sub_ps(a.v,b.v)); } -template<> EIGEN_STRONG_INLINE Packet4cf pnegate(const Packet4cf& a) -{ - return Packet4cf(pnegate(a.v)); -} -template<> EIGEN_STRONG_INLINE Packet4cf pconj(const Packet4cf& a) -{ - const __m256 mask = _mm256_castsi256_ps(_mm256_setr_epi32(0x00000000,0x80000000,0x00000000,0x80000000,0x00000000,0x80000000,0x00000000,0x80000000)); - return Packet4cf(_mm256_xor_ps(a.v,mask)); -} - -template<> EIGEN_STRONG_INLINE Packet4cf pmul(const Packet4cf& a, const Packet4cf& b) -{ - __m256 tmp1 = _mm256_mul_ps(_mm256_moveldup_ps(a.v), b.v); - __m256 tmp2 = _mm256_mul_ps(_mm256_movehdup_ps(a.v), _mm256_permute_ps(b.v, _MM_SHUFFLE(2,3,0,1))); - __m256 result = _mm256_addsub_ps(tmp1, tmp2); - return Packet4cf(result); -} - -template<> EIGEN_STRONG_INLINE Packet4cf pand (const Packet4cf& a, const Packet4cf& b) { return Packet4cf(_mm256_and_ps(a.v,b.v)); } -template<> EIGEN_STRONG_INLINE Packet4cf por (const Packet4cf& a, const Packet4cf& b) { return Packet4cf(_mm256_or_ps(a.v,b.v)); } -template<> EIGEN_STRONG_INLINE Packet4cf pxor (const Packet4cf& a, const Packet4cf& b) { return Packet4cf(_mm256_xor_ps(a.v,b.v)); } -template<> EIGEN_STRONG_INLINE Packet4cf pandnot(const Packet4cf& a, const Packet4cf& b) { return Packet4cf(_mm256_andnot_ps(a.v,b.v)); } - -template<> EIGEN_STRONG_INLINE Packet4cf pload (const std::complex* from) { EIGEN_DEBUG_ALIGNED_LOAD return Packet4cf(pload(&numext::real_ref(*from))); } -template<> EIGEN_STRONG_INLINE Packet4cf ploadu(const std::complex* from) { EIGEN_DEBUG_UNALIGNED_LOAD return Packet4cf(ploadu(&numext::real_ref(*from))); } - - -template<> EIGEN_STRONG_INLINE Packet4cf pset1(const std::complex& from) -{ - return Packet4cf(_mm256_castpd_ps(_mm256_broadcast_sd((const double*)(const void*)&from))); -} - -template<> EIGEN_STRONG_INLINE Packet4cf ploaddup(const std::complex* from) -{ - // FIXME The following might be optimized using _mm256_movedup_pd - Packet2cf a = ploaddup(from); - Packet2cf b = ploaddup(from+1); - return Packet4cf(_mm256_insertf128_ps(_mm256_castps128_ps256(a.v), b.v, 1)); -} - -template<> EIGEN_STRONG_INLINE void pstore >(std::complex* to, const Packet4cf& from) { EIGEN_DEBUG_ALIGNED_STORE pstore(&numext::real_ref(*to), from.v); } -template<> EIGEN_STRONG_INLINE void pstoreu >(std::complex* to, const Packet4cf& from) { EIGEN_DEBUG_UNALIGNED_STORE pstoreu(&numext::real_ref(*to), from.v); } - -template<> EIGEN_DEVICE_FUNC inline Packet4cf pgather, Packet4cf>(const std::complex* from, Index stride) -{ - return Packet4cf(_mm256_set_ps(std::imag(from[3*stride]), std::real(from[3*stride]), - std::imag(from[2*stride]), std::real(from[2*stride]), - std::imag(from[1*stride]), std::real(from[1*stride]), - std::imag(from[0*stride]), std::real(from[0*stride]))); -} - -template<> EIGEN_DEVICE_FUNC inline void pscatter, Packet4cf>(std::complex* to, const Packet4cf& from, Index stride) -{ - __m128 low = _mm256_extractf128_ps(from.v, 0); - to[stride*0] = std::complex(_mm_cvtss_f32(_mm_shuffle_ps(low, low, 0)), - _mm_cvtss_f32(_mm_shuffle_ps(low, low, 1))); - to[stride*1] = std::complex(_mm_cvtss_f32(_mm_shuffle_ps(low, low, 2)), - _mm_cvtss_f32(_mm_shuffle_ps(low, low, 3))); - - __m128 high = _mm256_extractf128_ps(from.v, 1); - to[stride*2] = std::complex(_mm_cvtss_f32(_mm_shuffle_ps(high, high, 0)), - _mm_cvtss_f32(_mm_shuffle_ps(high, high, 1))); - to[stride*3] = std::complex(_mm_cvtss_f32(_mm_shuffle_ps(high, high, 2)), - _mm_cvtss_f32(_mm_shuffle_ps(high, high, 3))); - -} - -template<> EIGEN_STRONG_INLINE std::complex pfirst(const Packet4cf& a) -{ - return pfirst(Packet2cf(_mm256_castps256_ps128(a.v))); -} - -template<> EIGEN_STRONG_INLINE Packet4cf preverse(const Packet4cf& a) { - __m128 low = _mm256_extractf128_ps(a.v, 0); - __m128 high = _mm256_extractf128_ps(a.v, 1); - __m128d lowd = _mm_castps_pd(low); - __m128d highd = _mm_castps_pd(high); - low = _mm_castpd_ps(_mm_shuffle_pd(lowd,lowd,0x1)); - high = _mm_castpd_ps(_mm_shuffle_pd(highd,highd,0x1)); - __m256 result = _mm256_setzero_ps(); - result = _mm256_insertf128_ps(result, low, 1); - result = _mm256_insertf128_ps(result, high, 0); - return Packet4cf(result); -} - -template<> EIGEN_STRONG_INLINE std::complex predux(const Packet4cf& a) -{ - return predux(padd(Packet2cf(_mm256_extractf128_ps(a.v,0)), - Packet2cf(_mm256_extractf128_ps(a.v,1)))); -} - -template<> EIGEN_STRONG_INLINE Packet4cf preduxp(const Packet4cf* vecs) -{ - Packet8f t0 = _mm256_shuffle_ps(vecs[0].v, vecs[0].v, _MM_SHUFFLE(3, 1, 2 ,0)); - Packet8f t1 = _mm256_shuffle_ps(vecs[1].v, vecs[1].v, _MM_SHUFFLE(3, 1, 2 ,0)); - t0 = _mm256_hadd_ps(t0,t1); - Packet8f t2 = _mm256_shuffle_ps(vecs[2].v, vecs[2].v, _MM_SHUFFLE(3, 1, 2 ,0)); - Packet8f t3 = _mm256_shuffle_ps(vecs[3].v, vecs[3].v, _MM_SHUFFLE(3, 1, 2 ,0)); - t2 = _mm256_hadd_ps(t2,t3); - - t1 = _mm256_permute2f128_ps(t0,t2, 0 + (2<<4)); - t3 = _mm256_permute2f128_ps(t0,t2, 1 + (3<<4)); - - return Packet4cf(_mm256_add_ps(t1,t3)); -} - -template<> EIGEN_STRONG_INLINE std::complex predux_mul(const Packet4cf& a) -{ - return predux_mul(pmul(Packet2cf(_mm256_extractf128_ps(a.v, 0)), - Packet2cf(_mm256_extractf128_ps(a.v, 1)))); -} - -template -struct palign_impl -{ - static EIGEN_STRONG_INLINE void run(Packet4cf& first, const Packet4cf& second) - { - if (Offset==0) return; - palign_impl::run(first.v, second.v); - } -}; - -template<> struct conj_helper -{ - EIGEN_STRONG_INLINE Packet4cf pmadd(const Packet4cf& x, const Packet4cf& y, const Packet4cf& c) const - { return padd(pmul(x,y),c); } - - EIGEN_STRONG_INLINE Packet4cf pmul(const Packet4cf& a, const Packet4cf& b) const - { - return internal::pmul(a, pconj(b)); - } -}; - -template<> struct conj_helper -{ - EIGEN_STRONG_INLINE Packet4cf pmadd(const Packet4cf& x, const Packet4cf& y, const Packet4cf& c) const - { return padd(pmul(x,y),c); } - - EIGEN_STRONG_INLINE Packet4cf pmul(const Packet4cf& a, const Packet4cf& b) const - { - return internal::pmul(pconj(a), b); - } -}; - -template<> struct conj_helper -{ - EIGEN_STRONG_INLINE Packet4cf pmadd(const Packet4cf& x, const Packet4cf& y, const Packet4cf& c) const - { return padd(pmul(x,y),c); } - - EIGEN_STRONG_INLINE Packet4cf pmul(const Packet4cf& a, const Packet4cf& b) const - { - return pconj(internal::pmul(a, b)); - } -}; - -EIGEN_MAKE_CONJ_HELPER_CPLX_REAL(Packet4cf,Packet8f) - -template<> EIGEN_STRONG_INLINE Packet4cf pdiv(const Packet4cf& a, const Packet4cf& b) -{ - Packet4cf num = pmul(a, pconj(b)); - __m256 tmp = _mm256_mul_ps(b.v, b.v); - __m256 tmp2 = _mm256_shuffle_ps(tmp,tmp,0xB1); - __m256 denom = _mm256_add_ps(tmp, tmp2); - return Packet4cf(_mm256_div_ps(num.v, denom)); -} - -template<> EIGEN_STRONG_INLINE Packet4cf pcplxflip(const Packet4cf& x) -{ - return Packet4cf(_mm256_shuffle_ps(x.v, x.v, _MM_SHUFFLE(2, 3, 0 ,1))); -} - -//---------- double ---------- -struct Packet2cd -{ - EIGEN_STRONG_INLINE Packet2cd() {} - EIGEN_STRONG_INLINE explicit Packet2cd(const __m256d& a) : v(a) {} - __m256d v; -}; - -template<> struct packet_traits > : default_packet_traits -{ - typedef Packet2cd type; - typedef Packet1cd half; - enum { - Vectorizable = 1, - AlignedOnScalar = 0, - size = 2, - HasHalfPacket = 1, - - HasAdd = 1, - HasSub = 1, - HasMul = 1, - HasDiv = 1, - HasNegate = 1, - HasAbs = 0, - HasAbs2 = 0, - HasMin = 0, - HasMax = 0, - HasSetLinear = 0 + + template<> struct unpacket_traits + { + typedef std::complex type; + enum { size = 4, alignment = Aligned32 }; + typedef Packet2cf half; }; -}; - -template<> struct unpacket_traits { typedef std::complex type; enum {size=2, alignment=Aligned32}; typedef Packet1cd half; }; - -template<> EIGEN_STRONG_INLINE Packet2cd padd(const Packet2cd& a, const Packet2cd& b) { return Packet2cd(_mm256_add_pd(a.v,b.v)); } -template<> EIGEN_STRONG_INLINE Packet2cd psub(const Packet2cd& a, const Packet2cd& b) { return Packet2cd(_mm256_sub_pd(a.v,b.v)); } -template<> EIGEN_STRONG_INLINE Packet2cd pnegate(const Packet2cd& a) { return Packet2cd(pnegate(a.v)); } -template<> EIGEN_STRONG_INLINE Packet2cd pconj(const Packet2cd& a) -{ - const __m256d mask = _mm256_castsi256_pd(_mm256_set_epi32(0x80000000,0x0,0x0,0x0,0x80000000,0x0,0x0,0x0)); - return Packet2cd(_mm256_xor_pd(a.v,mask)); -} - -template<> EIGEN_STRONG_INLINE Packet2cd pmul(const Packet2cd& a, const Packet2cd& b) -{ - __m256d tmp1 = _mm256_shuffle_pd(a.v,a.v,0x0); - __m256d even = _mm256_mul_pd(tmp1, b.v); - __m256d tmp2 = _mm256_shuffle_pd(a.v,a.v,0xF); - __m256d tmp3 = _mm256_shuffle_pd(b.v,b.v,0x5); - __m256d odd = _mm256_mul_pd(tmp2, tmp3); - return Packet2cd(_mm256_addsub_pd(even, odd)); -} - -template<> EIGEN_STRONG_INLINE Packet2cd pand (const Packet2cd& a, const Packet2cd& b) { return Packet2cd(_mm256_and_pd(a.v,b.v)); } -template<> EIGEN_STRONG_INLINE Packet2cd por (const Packet2cd& a, const Packet2cd& b) { return Packet2cd(_mm256_or_pd(a.v,b.v)); } -template<> EIGEN_STRONG_INLINE Packet2cd pxor (const Packet2cd& a, const Packet2cd& b) { return Packet2cd(_mm256_xor_pd(a.v,b.v)); } -template<> EIGEN_STRONG_INLINE Packet2cd pandnot(const Packet2cd& a, const Packet2cd& b) { return Packet2cd(_mm256_andnot_pd(a.v,b.v)); } - -template<> EIGEN_STRONG_INLINE Packet2cd pload (const std::complex* from) -{ EIGEN_DEBUG_ALIGNED_LOAD return Packet2cd(pload((const double*)from)); } -template<> EIGEN_STRONG_INLINE Packet2cd ploadu(const std::complex* from) -{ EIGEN_DEBUG_UNALIGNED_LOAD return Packet2cd(ploadu((const double*)from)); } - -template<> EIGEN_STRONG_INLINE Packet2cd pset1(const std::complex& from) -{ - // in case casting to a __m128d* is really not safe, then we can still fallback to this version: (much slower though) -// return Packet2cd(_mm256_loadu2_m128d((const double*)&from,(const double*)&from)); - return Packet2cd(_mm256_broadcast_pd((const __m128d*)(const void*)&from)); -} - -template<> EIGEN_STRONG_INLINE Packet2cd ploaddup(const std::complex* from) { return pset1(*from); } - -template<> EIGEN_STRONG_INLINE void pstore >(std::complex * to, const Packet2cd& from) { EIGEN_DEBUG_ALIGNED_STORE pstore((double*)to, from.v); } -template<> EIGEN_STRONG_INLINE void pstoreu >(std::complex * to, const Packet2cd& from) { EIGEN_DEBUG_UNALIGNED_STORE pstoreu((double*)to, from.v); } - -template<> EIGEN_DEVICE_FUNC inline Packet2cd pgather, Packet2cd>(const std::complex* from, Index stride) -{ - return Packet2cd(_mm256_set_pd(std::imag(from[1*stride]), std::real(from[1*stride]), - std::imag(from[0*stride]), std::real(from[0*stride]))); -} - -template<> EIGEN_DEVICE_FUNC inline void pscatter, Packet2cd>(std::complex* to, const Packet2cd& from, Index stride) -{ - __m128d low = _mm256_extractf128_pd(from.v, 0); - to[stride*0] = std::complex(_mm_cvtsd_f64(low), _mm_cvtsd_f64(_mm_shuffle_pd(low, low, 1))); - __m128d high = _mm256_extractf128_pd(from.v, 1); - to[stride*1] = std::complex(_mm_cvtsd_f64(high), _mm_cvtsd_f64(_mm_shuffle_pd(high, high, 1))); -} - -template<> EIGEN_STRONG_INLINE std::complex pfirst(const Packet2cd& a) -{ - __m128d low = _mm256_extractf128_pd(a.v, 0); - EIGEN_ALIGN16 double res[2]; - _mm_store_pd(res, low); - return std::complex(res[0],res[1]); -} - -template<> EIGEN_STRONG_INLINE Packet2cd preverse(const Packet2cd& a) { - __m256d result = _mm256_permute2f128_pd(a.v, a.v, 1); - return Packet2cd(result); -} - -template<> EIGEN_STRONG_INLINE std::complex predux(const Packet2cd& a) -{ - return predux(padd(Packet1cd(_mm256_extractf128_pd(a.v,0)), - Packet1cd(_mm256_extractf128_pd(a.v,1)))); -} - -template<> EIGEN_STRONG_INLINE Packet2cd preduxp(const Packet2cd* vecs) -{ - Packet4d t0 = _mm256_permute2f128_pd(vecs[0].v,vecs[1].v, 0 + (2<<4)); - Packet4d t1 = _mm256_permute2f128_pd(vecs[0].v,vecs[1].v, 1 + (3<<4)); - - return Packet2cd(_mm256_add_pd(t0,t1)); -} - -template<> EIGEN_STRONG_INLINE std::complex predux_mul(const Packet2cd& a) -{ - return predux(pmul(Packet1cd(_mm256_extractf128_pd(a.v,0)), - Packet1cd(_mm256_extractf128_pd(a.v,1)))); -} - -template -struct palign_impl -{ - static EIGEN_STRONG_INLINE void run(Packet2cd& first, const Packet2cd& second) - { - if (Offset==0) return; - palign_impl::run(first.v, second.v); - } -}; - -template<> struct conj_helper -{ - EIGEN_STRONG_INLINE Packet2cd pmadd(const Packet2cd& x, const Packet2cd& y, const Packet2cd& c) const - { return padd(pmul(x,y),c); } - - EIGEN_STRONG_INLINE Packet2cd pmul(const Packet2cd& a, const Packet2cd& b) const - { - return internal::pmul(a, pconj(b)); - } -}; - -template<> struct conj_helper -{ - EIGEN_STRONG_INLINE Packet2cd pmadd(const Packet2cd& x, const Packet2cd& y, const Packet2cd& c) const - { return padd(pmul(x,y),c); } - - EIGEN_STRONG_INLINE Packet2cd pmul(const Packet2cd& a, const Packet2cd& b) const - { - return internal::pmul(pconj(a), b); - } -}; - -template<> struct conj_helper -{ - EIGEN_STRONG_INLINE Packet2cd pmadd(const Packet2cd& x, const Packet2cd& y, const Packet2cd& c) const - { return padd(pmul(x,y),c); } - - EIGEN_STRONG_INLINE Packet2cd pmul(const Packet2cd& a, const Packet2cd& b) const - { - return pconj(internal::pmul(a, b)); - } -}; - -EIGEN_MAKE_CONJ_HELPER_CPLX_REAL(Packet2cd,Packet4d) - -template<> EIGEN_STRONG_INLINE Packet2cd pdiv(const Packet2cd& a, const Packet2cd& b) -{ - Packet2cd num = pmul(a, pconj(b)); - __m256d tmp = _mm256_mul_pd(b.v, b.v); - __m256d denom = _mm256_hadd_pd(tmp, tmp); - return Packet2cd(_mm256_div_pd(num.v, denom)); -} - -template<> EIGEN_STRONG_INLINE Packet2cd pcplxflip(const Packet2cd& x) -{ - return Packet2cd(_mm256_shuffle_pd(x.v, x.v, 0x5)); -} - -EIGEN_DEVICE_FUNC inline void -ptranspose(PacketBlock& kernel) { - __m256d P0 = _mm256_castps_pd(kernel.packet[0].v); - __m256d P1 = _mm256_castps_pd(kernel.packet[1].v); - __m256d P2 = _mm256_castps_pd(kernel.packet[2].v); - __m256d P3 = _mm256_castps_pd(kernel.packet[3].v); - - __m256d T0 = _mm256_shuffle_pd(P0, P1, 15); - __m256d T1 = _mm256_shuffle_pd(P0, P1, 0); - __m256d T2 = _mm256_shuffle_pd(P2, P3, 15); - __m256d T3 = _mm256_shuffle_pd(P2, P3, 0); - kernel.packet[1].v = _mm256_castpd_ps(_mm256_permute2f128_pd(T0, T2, 32)); - kernel.packet[3].v = _mm256_castpd_ps(_mm256_permute2f128_pd(T0, T2, 49)); - kernel.packet[0].v = _mm256_castpd_ps(_mm256_permute2f128_pd(T1, T3, 32)); - kernel.packet[2].v = _mm256_castpd_ps(_mm256_permute2f128_pd(T1, T3, 49)); -} + template<> EIGEN_STRONG_INLINE Packet4cf padd(const Packet4cf &a, const Packet4cf &b) + { + return Packet4cf(_mm256_add_ps(a.v, b.v)); + } + template<> EIGEN_STRONG_INLINE Packet4cf psub(const Packet4cf &a, const Packet4cf &b) + { + return Packet4cf(_mm256_sub_ps(a.v, b.v)); + } + template<> EIGEN_STRONG_INLINE Packet4cf pnegate(const Packet4cf &a) { return Packet4cf(pnegate(a.v)); } + template<> EIGEN_STRONG_INLINE Packet4cf pconj(const Packet4cf &a) + { + const __m256 mask = _mm256_castsi256_ps(_mm256_setr_epi32( + 0x00000000, 0x80000000, 0x00000000, 0x80000000, 0x00000000, 0x80000000, 0x00000000, 0x80000000)); + return Packet4cf(_mm256_xor_ps(a.v, mask)); + } + + template<> EIGEN_STRONG_INLINE Packet4cf pmul(const Packet4cf &a, const Packet4cf &b) + { + __m256 tmp1 = _mm256_mul_ps(_mm256_moveldup_ps(a.v), b.v); + __m256 tmp2 = _mm256_mul_ps(_mm256_movehdup_ps(a.v), _mm256_permute_ps(b.v, _MM_SHUFFLE(2, 3, 0, 1))); + __m256 result = _mm256_addsub_ps(tmp1, tmp2); + return Packet4cf(result); + } + + template<> EIGEN_STRONG_INLINE Packet4cf pand(const Packet4cf &a, const Packet4cf &b) + { + return Packet4cf(_mm256_and_ps(a.v, b.v)); + } + template<> EIGEN_STRONG_INLINE Packet4cf por(const Packet4cf &a, const Packet4cf &b) + { + return Packet4cf(_mm256_or_ps(a.v, b.v)); + } + template<> EIGEN_STRONG_INLINE Packet4cf pxor(const Packet4cf &a, const Packet4cf &b) + { + return Packet4cf(_mm256_xor_ps(a.v, b.v)); + } + template<> EIGEN_STRONG_INLINE Packet4cf pandnot(const Packet4cf &a, const Packet4cf &b) + { + return Packet4cf(_mm256_andnot_ps(a.v, b.v)); + } + + template<> EIGEN_STRONG_INLINE Packet4cf pload(const std::complex *from) + { + EIGEN_DEBUG_ALIGNED_LOAD return Packet4cf(pload(&numext::real_ref(*from))); + } + template<> EIGEN_STRONG_INLINE Packet4cf ploadu(const std::complex *from) + { + EIGEN_DEBUG_UNALIGNED_LOAD return Packet4cf(ploadu(&numext::real_ref(*from))); + } -EIGEN_DEVICE_FUNC inline void -ptranspose(PacketBlock& kernel) { - __m256d tmp = _mm256_permute2f128_pd(kernel.packet[0].v, kernel.packet[1].v, 0+(2<<4)); - kernel.packet[1].v = _mm256_permute2f128_pd(kernel.packet[0].v, kernel.packet[1].v, 1+(3<<4)); - kernel.packet[0].v = tmp; -} -template<> EIGEN_STRONG_INLINE Packet4cf pinsertfirst(const Packet4cf& a, std::complex b) -{ - return Packet4cf(_mm256_blend_ps(a.v,pset1(b).v,1|2)); -} + template<> EIGEN_STRONG_INLINE Packet4cf pset1(const std::complex &from) + { + return Packet4cf(_mm256_castpd_ps(_mm256_broadcast_sd((const double *)(const void *)&from))); + } -template<> EIGEN_STRONG_INLINE Packet2cd pinsertfirst(const Packet2cd& a, std::complex b) -{ - return Packet2cd(_mm256_blend_pd(a.v,pset1(b).v,1|2)); -} + template<> EIGEN_STRONG_INLINE Packet4cf ploaddup(const std::complex *from) + { + // FIXME The following might be optimized using _mm256_movedup_pd + Packet2cf a = ploaddup(from); + Packet2cf b = ploaddup(from + 1); + return Packet4cf(_mm256_insertf128_ps(_mm256_castps128_ps256(a.v), b.v, 1)); + } -template<> EIGEN_STRONG_INLINE Packet4cf pinsertlast(const Packet4cf& a, std::complex b) -{ - return Packet4cf(_mm256_blend_ps(a.v,pset1(b).v,(1<<7)|(1<<6))); -} + template<> EIGEN_STRONG_INLINE void pstore>(std::complex *to, const Packet4cf &from) + { + EIGEN_DEBUG_ALIGNED_STORE pstore(&numext::real_ref(*to), from.v); + } + template<> EIGEN_STRONG_INLINE void pstoreu>(std::complex *to, const Packet4cf &from) + { + EIGEN_DEBUG_UNALIGNED_STORE pstoreu(&numext::real_ref(*to), from.v); + } + + template<> + EIGEN_DEVICE_FUNC inline Packet4cf pgather, Packet4cf>(const std::complex *from, + Index stride) + { + return Packet4cf(_mm256_set_ps(std::imag(from[3 * stride]), + std::real(from[3 * stride]), + std::imag(from[2 * stride]), + std::real(from[2 * stride]), + std::imag(from[1 * stride]), + std::real(from[1 * stride]), + std::imag(from[0 * stride]), + std::real(from[0 * stride]))); + } + + template<> + EIGEN_DEVICE_FUNC inline void + pscatter, Packet4cf>(std::complex *to, const Packet4cf &from, Index stride) + { + __m128 low = _mm256_extractf128_ps(from.v, 0); + to[stride * 0] = + std::complex(_mm_cvtss_f32(_mm_shuffle_ps(low, low, 0)), _mm_cvtss_f32(_mm_shuffle_ps(low, low, 1))); + to[stride * 1] = + std::complex(_mm_cvtss_f32(_mm_shuffle_ps(low, low, 2)), _mm_cvtss_f32(_mm_shuffle_ps(low, low, 3))); + + __m128 high = _mm256_extractf128_ps(from.v, 1); + to[stride * 2] = + std::complex(_mm_cvtss_f32(_mm_shuffle_ps(high, high, 0)), _mm_cvtss_f32(_mm_shuffle_ps(high, high, 1))); + to[stride * 3] = + std::complex(_mm_cvtss_f32(_mm_shuffle_ps(high, high, 2)), _mm_cvtss_f32(_mm_shuffle_ps(high, high, 3))); + } + + template<> EIGEN_STRONG_INLINE std::complex pfirst(const Packet4cf &a) + { + return pfirst(Packet2cf(_mm256_castps256_ps128(a.v))); + } -template<> EIGEN_STRONG_INLINE Packet2cd pinsertlast(const Packet2cd& a, std::complex b) -{ - return Packet2cd(_mm256_blend_pd(a.v,pset1(b).v,(1<<3)|(1<<2))); -} + template<> EIGEN_STRONG_INLINE Packet4cf preverse(const Packet4cf &a) + { + __m128 low = _mm256_extractf128_ps(a.v, 0); + __m128 high = _mm256_extractf128_ps(a.v, 1); + __m128d lowd = _mm_castps_pd(low); + __m128d highd = _mm_castps_pd(high); + low = _mm_castpd_ps(_mm_shuffle_pd(lowd, lowd, 0x1)); + high = _mm_castpd_ps(_mm_shuffle_pd(highd, highd, 0x1)); + __m256 result = _mm256_setzero_ps(); + result = _mm256_insertf128_ps(result, low, 1); + result = _mm256_insertf128_ps(result, high, 0); + return Packet4cf(result); + } + + template<> EIGEN_STRONG_INLINE std::complex predux(const Packet4cf &a) + { + return predux(padd(Packet2cf(_mm256_extractf128_ps(a.v, 0)), Packet2cf(_mm256_extractf128_ps(a.v, 1)))); + } + + template<> EIGEN_STRONG_INLINE Packet4cf preduxp(const Packet4cf *vecs) + { + Packet8f t0 = _mm256_shuffle_ps(vecs[0].v, vecs[0].v, _MM_SHUFFLE(3, 1, 2, 0)); + Packet8f t1 = _mm256_shuffle_ps(vecs[1].v, vecs[1].v, _MM_SHUFFLE(3, 1, 2, 0)); + t0 = _mm256_hadd_ps(t0, t1); + Packet8f t2 = _mm256_shuffle_ps(vecs[2].v, vecs[2].v, _MM_SHUFFLE(3, 1, 2, 0)); + Packet8f t3 = _mm256_shuffle_ps(vecs[3].v, vecs[3].v, _MM_SHUFFLE(3, 1, 2, 0)); + t2 = _mm256_hadd_ps(t2, t3); + + t1 = _mm256_permute2f128_ps(t0, t2, 0 + (2 << 4)); + t3 = _mm256_permute2f128_ps(t0, t2, 1 + (3 << 4)); + + return Packet4cf(_mm256_add_ps(t1, t3)); + } + + template<> EIGEN_STRONG_INLINE std::complex predux_mul(const Packet4cf &a) + { + return predux_mul(pmul(Packet2cf(_mm256_extractf128_ps(a.v, 0)), Packet2cf(_mm256_extractf128_ps(a.v, 1)))); + } + + template struct palign_impl + { + static EIGEN_STRONG_INLINE void run(Packet4cf &first, const Packet4cf &second) + { + if (Offset == 0) return; + palign_impl::run(first.v, second.v); + } + }; + + template<> struct conj_helper + { + EIGEN_STRONG_INLINE Packet4cf pmadd(const Packet4cf &x, const Packet4cf &y, const Packet4cf &c) const + { + return padd(pmul(x, y), c); + } + + EIGEN_STRONG_INLINE Packet4cf pmul(const Packet4cf &a, const Packet4cf &b) const + { + return internal::pmul(a, pconj(b)); + } + }; + + template<> struct conj_helper + { + EIGEN_STRONG_INLINE Packet4cf pmadd(const Packet4cf &x, const Packet4cf &y, const Packet4cf &c) const + { + return padd(pmul(x, y), c); + } + + EIGEN_STRONG_INLINE Packet4cf pmul(const Packet4cf &a, const Packet4cf &b) const + { + return internal::pmul(pconj(a), b); + } + }; + + template<> struct conj_helper + { + EIGEN_STRONG_INLINE Packet4cf pmadd(const Packet4cf &x, const Packet4cf &y, const Packet4cf &c) const + { + return padd(pmul(x, y), c); + } + + EIGEN_STRONG_INLINE Packet4cf pmul(const Packet4cf &a, const Packet4cf &b) const + { + return pconj(internal::pmul(a, b)); + } + }; + + EIGEN_MAKE_CONJ_HELPER_CPLX_REAL(Packet4cf, Packet8f) + + template<> EIGEN_STRONG_INLINE Packet4cf pdiv(const Packet4cf &a, const Packet4cf &b) + { + Packet4cf num = pmul(a, pconj(b)); + __m256 tmp = _mm256_mul_ps(b.v, b.v); + __m256 tmp2 = _mm256_shuffle_ps(tmp, tmp, 0xB1); + __m256 denom = _mm256_add_ps(tmp, tmp2); + return Packet4cf(_mm256_div_ps(num.v, denom)); + } + + template<> EIGEN_STRONG_INLINE Packet4cf pcplxflip(const Packet4cf &x) + { + return Packet4cf(_mm256_shuffle_ps(x.v, x.v, _MM_SHUFFLE(2, 3, 0, 1))); + } + + //---------- double ---------- + struct Packet2cd + { + EIGEN_STRONG_INLINE Packet2cd() {} + EIGEN_STRONG_INLINE explicit Packet2cd(const __m256d &a) : v(a) {} + __m256d v; + }; + + template<> struct packet_traits> : default_packet_traits + { + typedef Packet2cd type; + typedef Packet1cd half; + enum { + Vectorizable = 1, + AlignedOnScalar = 0, + size = 2, + HasHalfPacket = 1, + + HasAdd = 1, + HasSub = 1, + HasMul = 1, + HasDiv = 1, + HasNegate = 1, + HasAbs = 0, + HasAbs2 = 0, + HasMin = 0, + HasMax = 0, + HasSetLinear = 0 + }; + }; + + template<> struct unpacket_traits + { + typedef std::complex type; + enum { size = 2, alignment = Aligned32 }; + typedef Packet1cd half; + }; + + template<> EIGEN_STRONG_INLINE Packet2cd padd(const Packet2cd &a, const Packet2cd &b) + { + return Packet2cd(_mm256_add_pd(a.v, b.v)); + } + template<> EIGEN_STRONG_INLINE Packet2cd psub(const Packet2cd &a, const Packet2cd &b) + { + return Packet2cd(_mm256_sub_pd(a.v, b.v)); + } + template<> EIGEN_STRONG_INLINE Packet2cd pnegate(const Packet2cd &a) { return Packet2cd(pnegate(a.v)); } + template<> EIGEN_STRONG_INLINE Packet2cd pconj(const Packet2cd &a) + { + const __m256d mask = _mm256_castsi256_pd(_mm256_set_epi32(0x80000000, 0x0, 0x0, 0x0, 0x80000000, 0x0, 0x0, 0x0)); + return Packet2cd(_mm256_xor_pd(a.v, mask)); + } + + template<> EIGEN_STRONG_INLINE Packet2cd pmul(const Packet2cd &a, const Packet2cd &b) + { + __m256d tmp1 = _mm256_shuffle_pd(a.v, a.v, 0x0); + __m256d even = _mm256_mul_pd(tmp1, b.v); + __m256d tmp2 = _mm256_shuffle_pd(a.v, a.v, 0xF); + __m256d tmp3 = _mm256_shuffle_pd(b.v, b.v, 0x5); + __m256d odd = _mm256_mul_pd(tmp2, tmp3); + return Packet2cd(_mm256_addsub_pd(even, odd)); + } + + template<> EIGEN_STRONG_INLINE Packet2cd pand(const Packet2cd &a, const Packet2cd &b) + { + return Packet2cd(_mm256_and_pd(a.v, b.v)); + } + template<> EIGEN_STRONG_INLINE Packet2cd por(const Packet2cd &a, const Packet2cd &b) + { + return Packet2cd(_mm256_or_pd(a.v, b.v)); + } + template<> EIGEN_STRONG_INLINE Packet2cd pxor(const Packet2cd &a, const Packet2cd &b) + { + return Packet2cd(_mm256_xor_pd(a.v, b.v)); + } + template<> EIGEN_STRONG_INLINE Packet2cd pandnot(const Packet2cd &a, const Packet2cd &b) + { + return Packet2cd(_mm256_andnot_pd(a.v, b.v)); + } + + template<> EIGEN_STRONG_INLINE Packet2cd pload(const std::complex *from) + { + EIGEN_DEBUG_ALIGNED_LOAD return Packet2cd(pload((const double *)from)); + } + template<> EIGEN_STRONG_INLINE Packet2cd ploadu(const std::complex *from) + { + EIGEN_DEBUG_UNALIGNED_LOAD return Packet2cd(ploadu((const double *)from)); + } + + template<> EIGEN_STRONG_INLINE Packet2cd pset1(const std::complex &from) + { + // in case casting to a __m128d* is really not safe, then we can still fallback to this version: (much slower + // though) + // return Packet2cd(_mm256_loadu2_m128d((const double*)&from,(const double*)&from)); + return Packet2cd(_mm256_broadcast_pd((const __m128d *)(const void *)&from)); + } + + template<> EIGEN_STRONG_INLINE Packet2cd ploaddup(const std::complex *from) + { + return pset1(*from); + } + + template<> EIGEN_STRONG_INLINE void pstore>(std::complex *to, const Packet2cd &from) + { + EIGEN_DEBUG_ALIGNED_STORE pstore((double *)to, from.v); + } + template<> EIGEN_STRONG_INLINE void pstoreu>(std::complex *to, const Packet2cd &from) + { + EIGEN_DEBUG_UNALIGNED_STORE pstoreu((double *)to, from.v); + } + + template<> + EIGEN_DEVICE_FUNC inline Packet2cd pgather, Packet2cd>(const std::complex *from, + Index stride) + { + return Packet2cd(_mm256_set_pd(std::imag(from[1 * stride]), + std::real(from[1 * stride]), + std::imag(from[0 * stride]), + std::real(from[0 * stride]))); + } + + template<> + EIGEN_DEVICE_FUNC inline void + pscatter, Packet2cd>(std::complex *to, const Packet2cd &from, Index stride) + { + __m128d low = _mm256_extractf128_pd(from.v, 0); + to[stride * 0] = std::complex(_mm_cvtsd_f64(low), _mm_cvtsd_f64(_mm_shuffle_pd(low, low, 1))); + __m128d high = _mm256_extractf128_pd(from.v, 1); + to[stride * 1] = std::complex(_mm_cvtsd_f64(high), _mm_cvtsd_f64(_mm_shuffle_pd(high, high, 1))); + } + + template<> EIGEN_STRONG_INLINE std::complex pfirst(const Packet2cd &a) + { + __m128d low = _mm256_extractf128_pd(a.v, 0); + EIGEN_ALIGN16 double res[2]; + _mm_store_pd(res, low); + return std::complex(res[0], res[1]); + } + + template<> EIGEN_STRONG_INLINE Packet2cd preverse(const Packet2cd &a) + { + __m256d result = _mm256_permute2f128_pd(a.v, a.v, 1); + return Packet2cd(result); + } + + template<> EIGEN_STRONG_INLINE std::complex predux(const Packet2cd &a) + { + return predux(padd(Packet1cd(_mm256_extractf128_pd(a.v, 0)), Packet1cd(_mm256_extractf128_pd(a.v, 1)))); + } + + template<> EIGEN_STRONG_INLINE Packet2cd preduxp(const Packet2cd *vecs) + { + Packet4d t0 = _mm256_permute2f128_pd(vecs[0].v, vecs[1].v, 0 + (2 << 4)); + Packet4d t1 = _mm256_permute2f128_pd(vecs[0].v, vecs[1].v, 1 + (3 << 4)); + + return Packet2cd(_mm256_add_pd(t0, t1)); + } + + template<> EIGEN_STRONG_INLINE std::complex predux_mul(const Packet2cd &a) + { + return predux(pmul(Packet1cd(_mm256_extractf128_pd(a.v, 0)), Packet1cd(_mm256_extractf128_pd(a.v, 1)))); + } + + template struct palign_impl + { + static EIGEN_STRONG_INLINE void run(Packet2cd &first, const Packet2cd &second) + { + if (Offset == 0) return; + palign_impl::run(first.v, second.v); + } + }; + + template<> struct conj_helper + { + EIGEN_STRONG_INLINE Packet2cd pmadd(const Packet2cd &x, const Packet2cd &y, const Packet2cd &c) const + { + return padd(pmul(x, y), c); + } + + EIGEN_STRONG_INLINE Packet2cd pmul(const Packet2cd &a, const Packet2cd &b) const + { + return internal::pmul(a, pconj(b)); + } + }; + + template<> struct conj_helper + { + EIGEN_STRONG_INLINE Packet2cd pmadd(const Packet2cd &x, const Packet2cd &y, const Packet2cd &c) const + { + return padd(pmul(x, y), c); + } + + EIGEN_STRONG_INLINE Packet2cd pmul(const Packet2cd &a, const Packet2cd &b) const + { + return internal::pmul(pconj(a), b); + } + }; + + template<> struct conj_helper + { + EIGEN_STRONG_INLINE Packet2cd pmadd(const Packet2cd &x, const Packet2cd &y, const Packet2cd &c) const + { + return padd(pmul(x, y), c); + } + + EIGEN_STRONG_INLINE Packet2cd pmul(const Packet2cd &a, const Packet2cd &b) const + { + return pconj(internal::pmul(a, b)); + } + }; + + EIGEN_MAKE_CONJ_HELPER_CPLX_REAL(Packet2cd, Packet4d) + + template<> EIGEN_STRONG_INLINE Packet2cd pdiv(const Packet2cd &a, const Packet2cd &b) + { + Packet2cd num = pmul(a, pconj(b)); + __m256d tmp = _mm256_mul_pd(b.v, b.v); + __m256d denom = _mm256_hadd_pd(tmp, tmp); + return Packet2cd(_mm256_div_pd(num.v, denom)); + } + + template<> EIGEN_STRONG_INLINE Packet2cd pcplxflip(const Packet2cd &x) + { + return Packet2cd(_mm256_shuffle_pd(x.v, x.v, 0x5)); + } + + EIGEN_DEVICE_FUNC inline void ptranspose(PacketBlock &kernel) + { + __m256d P0 = _mm256_castps_pd(kernel.packet[0].v); + __m256d P1 = _mm256_castps_pd(kernel.packet[1].v); + __m256d P2 = _mm256_castps_pd(kernel.packet[2].v); + __m256d P3 = _mm256_castps_pd(kernel.packet[3].v); + + __m256d T0 = _mm256_shuffle_pd(P0, P1, 15); + __m256d T1 = _mm256_shuffle_pd(P0, P1, 0); + __m256d T2 = _mm256_shuffle_pd(P2, P3, 15); + __m256d T3 = _mm256_shuffle_pd(P2, P3, 0); + + kernel.packet[1].v = _mm256_castpd_ps(_mm256_permute2f128_pd(T0, T2, 32)); + kernel.packet[3].v = _mm256_castpd_ps(_mm256_permute2f128_pd(T0, T2, 49)); + kernel.packet[0].v = _mm256_castpd_ps(_mm256_permute2f128_pd(T1, T3, 32)); + kernel.packet[2].v = _mm256_castpd_ps(_mm256_permute2f128_pd(T1, T3, 49)); + } + + EIGEN_DEVICE_FUNC inline void ptranspose(PacketBlock &kernel) + { + __m256d tmp = _mm256_permute2f128_pd(kernel.packet[0].v, kernel.packet[1].v, 0 + (2 << 4)); + kernel.packet[1].v = _mm256_permute2f128_pd(kernel.packet[0].v, kernel.packet[1].v, 1 + (3 << 4)); + kernel.packet[0].v = tmp; + } + + template<> EIGEN_STRONG_INLINE Packet4cf pinsertfirst(const Packet4cf &a, std::complex b) + { + return Packet4cf(_mm256_blend_ps(a.v, pset1(b).v, 1 | 2)); + } + + template<> EIGEN_STRONG_INLINE Packet2cd pinsertfirst(const Packet2cd &a, std::complex b) + { + return Packet2cd(_mm256_blend_pd(a.v, pset1(b).v, 1 | 2)); + } + + template<> EIGEN_STRONG_INLINE Packet4cf pinsertlast(const Packet4cf &a, std::complex b) + { + return Packet4cf(_mm256_blend_ps(a.v, pset1(b).v, (1 << 7) | (1 << 6))); + } + + template<> EIGEN_STRONG_INLINE Packet2cd pinsertlast(const Packet2cd &a, std::complex b) + { + return Packet2cd(_mm256_blend_pd(a.v, pset1(b).v, (1 << 3) | (1 << 2))); + } -} // end namespace internal +}// end namespace internal -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_COMPLEX_AVX_H +#endif// EIGEN_COMPLEX_AVX_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/arch/AVX/MathFunctions.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/arch/AVX/MathFunctions.h index 6af67ce2..776c64c5 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/arch/AVX/MathFunctions.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/arch/AVX/MathFunctions.h @@ -18,339 +18,339 @@ namespace Eigen { namespace internal { -inline Packet8i pshiftleft(Packet8i v, int n) -{ + inline Packet8i pshiftleft(Packet8i v, int n) + { #ifdef EIGEN_VECTORIZE_AVX2 - return _mm256_slli_epi32(v, n); + return _mm256_slli_epi32(v, n); #else - __m128i lo = _mm_slli_epi32(_mm256_extractf128_si256(v, 0), n); - __m128i hi = _mm_slli_epi32(_mm256_extractf128_si256(v, 1), n); - return _mm256_insertf128_si256(_mm256_castsi128_si256(lo), (hi), 1); + __m128i lo = _mm_slli_epi32(_mm256_extractf128_si256(v, 0), n); + __m128i hi = _mm_slli_epi32(_mm256_extractf128_si256(v, 1), n); + return _mm256_insertf128_si256(_mm256_castsi128_si256(lo), (hi), 1); #endif -} + } -inline Packet8f pshiftright(Packet8f v, int n) -{ + inline Packet8f pshiftright(Packet8f v, int n) + { #ifdef EIGEN_VECTORIZE_AVX2 - return _mm256_cvtepi32_ps(_mm256_srli_epi32(_mm256_castps_si256(v), n)); + return _mm256_cvtepi32_ps(_mm256_srli_epi32(_mm256_castps_si256(v), n)); #else - __m128i lo = _mm_srli_epi32(_mm256_extractf128_si256(_mm256_castps_si256(v), 0), n); - __m128i hi = _mm_srli_epi32(_mm256_extractf128_si256(_mm256_castps_si256(v), 1), n); - return _mm256_cvtepi32_ps(_mm256_insertf128_si256(_mm256_castsi128_si256(lo), (hi), 1)); + __m128i lo = _mm_srli_epi32(_mm256_extractf128_si256(_mm256_castps_si256(v), 0), n); + __m128i hi = _mm_srli_epi32(_mm256_extractf128_si256(_mm256_castps_si256(v), 1), n); + return _mm256_cvtepi32_ps(_mm256_insertf128_si256(_mm256_castsi128_si256(lo), (hi), 1)); #endif -} - -// Sine function -// Computes sin(x) by wrapping x to the interval [-Pi/4,3*Pi/4] and -// evaluating interpolants in [-Pi/4,Pi/4] or [Pi/4,3*Pi/4]. The interpolants -// are (anti-)symmetric and thus have only odd/even coefficients -template <> -EIGEN_DEFINE_FUNCTION_ALLOWING_MULTIPLE_DEFINITIONS EIGEN_UNUSED Packet8f -psin(const Packet8f& _x) { - Packet8f x = _x; - - // Some useful values. - _EIGEN_DECLARE_CONST_Packet8i(one, 1); - _EIGEN_DECLARE_CONST_Packet8f(one, 1.0f); - _EIGEN_DECLARE_CONST_Packet8f(two, 2.0f); - _EIGEN_DECLARE_CONST_Packet8f(one_over_four, 0.25f); - _EIGEN_DECLARE_CONST_Packet8f(one_over_pi, 3.183098861837907e-01f); - _EIGEN_DECLARE_CONST_Packet8f(neg_pi_first, -3.140625000000000e+00f); - _EIGEN_DECLARE_CONST_Packet8f(neg_pi_second, -9.670257568359375e-04f); - _EIGEN_DECLARE_CONST_Packet8f(neg_pi_third, -6.278329571784980e-07f); - _EIGEN_DECLARE_CONST_Packet8f(four_over_pi, 1.273239544735163e+00f); - - // Map x from [-Pi/4,3*Pi/4] to z in [-1,3] and subtract the shifted period. - Packet8f z = pmul(x, p8f_one_over_pi); - Packet8f shift = _mm256_floor_ps(padd(z, p8f_one_over_four)); - x = pmadd(shift, p8f_neg_pi_first, x); - x = pmadd(shift, p8f_neg_pi_second, x); - x = pmadd(shift, p8f_neg_pi_third, x); - z = pmul(x, p8f_four_over_pi); - - // Make a mask for the entries that need flipping, i.e. wherever the shift - // is odd. - Packet8i shift_ints = _mm256_cvtps_epi32(shift); - Packet8i shift_isodd = _mm256_castps_si256(_mm256_and_ps(_mm256_castsi256_ps(shift_ints), _mm256_castsi256_ps(p8i_one))); - Packet8i sign_flip_mask = pshiftleft(shift_isodd, 31); - - // Create a mask for which interpolant to use, i.e. if z > 1, then the mask - // is set to ones for that entry. - Packet8f ival_mask = _mm256_cmp_ps(z, p8f_one, _CMP_GT_OQ); - - // Evaluate the polynomial for the interval [1,3] in z. - _EIGEN_DECLARE_CONST_Packet8f(coeff_right_0, 9.999999724233232e-01f); - _EIGEN_DECLARE_CONST_Packet8f(coeff_right_2, -3.084242535619928e-01f); - _EIGEN_DECLARE_CONST_Packet8f(coeff_right_4, 1.584991525700324e-02f); - _EIGEN_DECLARE_CONST_Packet8f(coeff_right_6, -3.188805084631342e-04f); - Packet8f z_minus_two = psub(z, p8f_two); - Packet8f z_minus_two2 = pmul(z_minus_two, z_minus_two); - Packet8f right = pmadd(p8f_coeff_right_6, z_minus_two2, p8f_coeff_right_4); - right = pmadd(right, z_minus_two2, p8f_coeff_right_2); - right = pmadd(right, z_minus_two2, p8f_coeff_right_0); - - // Evaluate the polynomial for the interval [-1,1] in z. - _EIGEN_DECLARE_CONST_Packet8f(coeff_left_1, 7.853981525427295e-01f); - _EIGEN_DECLARE_CONST_Packet8f(coeff_left_3, -8.074536727092352e-02f); - _EIGEN_DECLARE_CONST_Packet8f(coeff_left_5, 2.489871967827018e-03f); - _EIGEN_DECLARE_CONST_Packet8f(coeff_left_7, -3.587725841214251e-05f); - Packet8f z2 = pmul(z, z); - Packet8f left = pmadd(p8f_coeff_left_7, z2, p8f_coeff_left_5); - left = pmadd(left, z2, p8f_coeff_left_3); - left = pmadd(left, z2, p8f_coeff_left_1); - left = pmul(left, z); - - // Assemble the results, i.e. select the left and right polynomials. - left = _mm256_andnot_ps(ival_mask, left); - right = _mm256_and_ps(ival_mask, right); - Packet8f res = _mm256_or_ps(left, right); - - // Flip the sign on the odd intervals and return the result. - res = _mm256_xor_ps(res, _mm256_castsi256_ps(sign_flip_mask)); - return res; -} - -// Natural logarithm -// Computes log(x) as log(2^e * m) = C*e + log(m), where the constant C =log(2) -// and m is in the range [sqrt(1/2),sqrt(2)). In this range, the logarithm can -// be easily approximated by a polynomial centered on m=1 for stability. -// TODO(gonnet): Further reduce the interval allowing for lower-degree -// polynomial interpolants -> ... -> profit! -template <> -EIGEN_DEFINE_FUNCTION_ALLOWING_MULTIPLE_DEFINITIONS EIGEN_UNUSED Packet8f -plog(const Packet8f& _x) { - Packet8f x = _x; - _EIGEN_DECLARE_CONST_Packet8f(1, 1.0f); - _EIGEN_DECLARE_CONST_Packet8f(half, 0.5f); - _EIGEN_DECLARE_CONST_Packet8f(126f, 126.0f); - - _EIGEN_DECLARE_CONST_Packet8f_FROM_INT(inv_mant_mask, ~0x7f800000); - - // The smallest non denormalized float number. - _EIGEN_DECLARE_CONST_Packet8f_FROM_INT(min_norm_pos, 0x00800000); - _EIGEN_DECLARE_CONST_Packet8f_FROM_INT(minus_inf, 0xff800000); - - // Polynomial coefficients. - _EIGEN_DECLARE_CONST_Packet8f(cephes_SQRTHF, 0.707106781186547524f); - _EIGEN_DECLARE_CONST_Packet8f(cephes_log_p0, 7.0376836292E-2f); - _EIGEN_DECLARE_CONST_Packet8f(cephes_log_p1, -1.1514610310E-1f); - _EIGEN_DECLARE_CONST_Packet8f(cephes_log_p2, 1.1676998740E-1f); - _EIGEN_DECLARE_CONST_Packet8f(cephes_log_p3, -1.2420140846E-1f); - _EIGEN_DECLARE_CONST_Packet8f(cephes_log_p4, +1.4249322787E-1f); - _EIGEN_DECLARE_CONST_Packet8f(cephes_log_p5, -1.6668057665E-1f); - _EIGEN_DECLARE_CONST_Packet8f(cephes_log_p6, +2.0000714765E-1f); - _EIGEN_DECLARE_CONST_Packet8f(cephes_log_p7, -2.4999993993E-1f); - _EIGEN_DECLARE_CONST_Packet8f(cephes_log_p8, +3.3333331174E-1f); - _EIGEN_DECLARE_CONST_Packet8f(cephes_log_q1, -2.12194440e-4f); - _EIGEN_DECLARE_CONST_Packet8f(cephes_log_q2, 0.693359375f); - - Packet8f invalid_mask = _mm256_cmp_ps(x, _mm256_setzero_ps(), _CMP_NGE_UQ); // not greater equal is true if x is NaN - Packet8f iszero_mask = _mm256_cmp_ps(x, _mm256_setzero_ps(), _CMP_EQ_OQ); - - // Truncate input values to the minimum positive normal. - x = pmax(x, p8f_min_norm_pos); - - Packet8f emm0 = pshiftright(x,23); - Packet8f e = _mm256_sub_ps(emm0, p8f_126f); - - // Set the exponents to -1, i.e. x are in the range [0.5,1). - x = _mm256_and_ps(x, p8f_inv_mant_mask); - x = _mm256_or_ps(x, p8f_half); - - // part2: Shift the inputs from the range [0.5,1) to [sqrt(1/2),sqrt(2)) - // and shift by -1. The values are then centered around 0, which improves - // the stability of the polynomial evaluation. - // if( x < SQRTHF ) { - // e -= 1; - // x = x + x - 1.0; - // } else { x = x - 1.0; } - Packet8f mask = _mm256_cmp_ps(x, p8f_cephes_SQRTHF, _CMP_LT_OQ); - Packet8f tmp = _mm256_and_ps(x, mask); - x = psub(x, p8f_1); - e = psub(e, _mm256_and_ps(p8f_1, mask)); - x = padd(x, tmp); - - Packet8f x2 = pmul(x, x); - Packet8f x3 = pmul(x2, x); - - // Evaluate the polynomial approximant of degree 8 in three parts, probably - // to improve instruction-level parallelism. - Packet8f y, y1, y2; - y = pmadd(p8f_cephes_log_p0, x, p8f_cephes_log_p1); - y1 = pmadd(p8f_cephes_log_p3, x, p8f_cephes_log_p4); - y2 = pmadd(p8f_cephes_log_p6, x, p8f_cephes_log_p7); - y = pmadd(y, x, p8f_cephes_log_p2); - y1 = pmadd(y1, x, p8f_cephes_log_p5); - y2 = pmadd(y2, x, p8f_cephes_log_p8); - y = pmadd(y, x3, y1); - y = pmadd(y, x3, y2); - y = pmul(y, x3); - - // Add the logarithm of the exponent back to the result of the interpolation. - y1 = pmul(e, p8f_cephes_log_q1); - tmp = pmul(x2, p8f_half); - y = padd(y, y1); - x = psub(x, tmp); - y2 = pmul(e, p8f_cephes_log_q2); - x = padd(x, y); - x = padd(x, y2); - - // Filter out invalid inputs, i.e. negative arg will be NAN, 0 will be -INF. - return _mm256_or_ps( - _mm256_andnot_ps(iszero_mask, _mm256_or_ps(x, invalid_mask)), - _mm256_and_ps(iszero_mask, p8f_minus_inf)); -} - -// Exponential function. Works by writing "x = m*log(2) + r" where -// "m = floor(x/log(2)+1/2)" and "r" is the remainder. The result is then -// "exp(x) = 2^m*exp(r)" where exp(r) is in the range [-1,1). -template <> -EIGEN_DEFINE_FUNCTION_ALLOWING_MULTIPLE_DEFINITIONS EIGEN_UNUSED Packet8f -pexp(const Packet8f& _x) { - _EIGEN_DECLARE_CONST_Packet8f(1, 1.0f); - _EIGEN_DECLARE_CONST_Packet8f(half, 0.5f); - _EIGEN_DECLARE_CONST_Packet8f(127, 127.0f); - - _EIGEN_DECLARE_CONST_Packet8f(exp_hi, 88.3762626647950f); - _EIGEN_DECLARE_CONST_Packet8f(exp_lo, -88.3762626647949f); - - _EIGEN_DECLARE_CONST_Packet8f(cephes_LOG2EF, 1.44269504088896341f); - - _EIGEN_DECLARE_CONST_Packet8f(cephes_exp_p0, 1.9875691500E-4f); - _EIGEN_DECLARE_CONST_Packet8f(cephes_exp_p1, 1.3981999507E-3f); - _EIGEN_DECLARE_CONST_Packet8f(cephes_exp_p2, 8.3334519073E-3f); - _EIGEN_DECLARE_CONST_Packet8f(cephes_exp_p3, 4.1665795894E-2f); - _EIGEN_DECLARE_CONST_Packet8f(cephes_exp_p4, 1.6666665459E-1f); - _EIGEN_DECLARE_CONST_Packet8f(cephes_exp_p5, 5.0000001201E-1f); - - // Clamp x. - Packet8f x = pmax(pmin(_x, p8f_exp_hi), p8f_exp_lo); - - // Express exp(x) as exp(m*ln(2) + r), start by extracting - // m = floor(x/ln(2) + 0.5). - Packet8f m = _mm256_floor_ps(pmadd(x, p8f_cephes_LOG2EF, p8f_half)); + } + + // Sine function + // Computes sin(x) by wrapping x to the interval [-Pi/4,3*Pi/4] and + // evaluating interpolants in [-Pi/4,Pi/4] or [Pi/4,3*Pi/4]. The interpolants + // are (anti-)symmetric and thus have only odd/even coefficients + template<> + EIGEN_DEFINE_FUNCTION_ALLOWING_MULTIPLE_DEFINITIONS EIGEN_UNUSED Packet8f psin(const Packet8f &_x) + { + Packet8f x = _x; + + // Some useful values. + _EIGEN_DECLARE_CONST_Packet8i(one, 1); + _EIGEN_DECLARE_CONST_Packet8f(one, 1.0f); + _EIGEN_DECLARE_CONST_Packet8f(two, 2.0f); + _EIGEN_DECLARE_CONST_Packet8f(one_over_four, 0.25f); + _EIGEN_DECLARE_CONST_Packet8f(one_over_pi, 3.183098861837907e-01f); + _EIGEN_DECLARE_CONST_Packet8f(neg_pi_first, -3.140625000000000e+00f); + _EIGEN_DECLARE_CONST_Packet8f(neg_pi_second, -9.670257568359375e-04f); + _EIGEN_DECLARE_CONST_Packet8f(neg_pi_third, -6.278329571784980e-07f); + _EIGEN_DECLARE_CONST_Packet8f(four_over_pi, 1.273239544735163e+00f); + + // Map x from [-Pi/4,3*Pi/4] to z in [-1,3] and subtract the shifted period. + Packet8f z = pmul(x, p8f_one_over_pi); + Packet8f shift = _mm256_floor_ps(padd(z, p8f_one_over_four)); + x = pmadd(shift, p8f_neg_pi_first, x); + x = pmadd(shift, p8f_neg_pi_second, x); + x = pmadd(shift, p8f_neg_pi_third, x); + z = pmul(x, p8f_four_over_pi); + + // Make a mask for the entries that need flipping, i.e. wherever the shift + // is odd. + Packet8i shift_ints = _mm256_cvtps_epi32(shift); + Packet8i shift_isodd = + _mm256_castps_si256(_mm256_and_ps(_mm256_castsi256_ps(shift_ints), _mm256_castsi256_ps(p8i_one))); + Packet8i sign_flip_mask = pshiftleft(shift_isodd, 31); + + // Create a mask for which interpolant to use, i.e. if z > 1, then the mask + // is set to ones for that entry. + Packet8f ival_mask = _mm256_cmp_ps(z, p8f_one, _CMP_GT_OQ); + + // Evaluate the polynomial for the interval [1,3] in z. + _EIGEN_DECLARE_CONST_Packet8f(coeff_right_0, 9.999999724233232e-01f); + _EIGEN_DECLARE_CONST_Packet8f(coeff_right_2, -3.084242535619928e-01f); + _EIGEN_DECLARE_CONST_Packet8f(coeff_right_4, 1.584991525700324e-02f); + _EIGEN_DECLARE_CONST_Packet8f(coeff_right_6, -3.188805084631342e-04f); + Packet8f z_minus_two = psub(z, p8f_two); + Packet8f z_minus_two2 = pmul(z_minus_two, z_minus_two); + Packet8f right = pmadd(p8f_coeff_right_6, z_minus_two2, p8f_coeff_right_4); + right = pmadd(right, z_minus_two2, p8f_coeff_right_2); + right = pmadd(right, z_minus_two2, p8f_coeff_right_0); + + // Evaluate the polynomial for the interval [-1,1] in z. + _EIGEN_DECLARE_CONST_Packet8f(coeff_left_1, 7.853981525427295e-01f); + _EIGEN_DECLARE_CONST_Packet8f(coeff_left_3, -8.074536727092352e-02f); + _EIGEN_DECLARE_CONST_Packet8f(coeff_left_5, 2.489871967827018e-03f); + _EIGEN_DECLARE_CONST_Packet8f(coeff_left_7, -3.587725841214251e-05f); + Packet8f z2 = pmul(z, z); + Packet8f left = pmadd(p8f_coeff_left_7, z2, p8f_coeff_left_5); + left = pmadd(left, z2, p8f_coeff_left_3); + left = pmadd(left, z2, p8f_coeff_left_1); + left = pmul(left, z); + + // Assemble the results, i.e. select the left and right polynomials. + left = _mm256_andnot_ps(ival_mask, left); + right = _mm256_and_ps(ival_mask, right); + Packet8f res = _mm256_or_ps(left, right); + + // Flip the sign on the odd intervals and return the result. + res = _mm256_xor_ps(res, _mm256_castsi256_ps(sign_flip_mask)); + return res; + } + + // Natural logarithm + // Computes log(x) as log(2^e * m) = C*e + log(m), where the constant C =log(2) + // and m is in the range [sqrt(1/2),sqrt(2)). In this range, the logarithm can + // be easily approximated by a polynomial centered on m=1 for stability. + // TODO(gonnet): Further reduce the interval allowing for lower-degree + // polynomial interpolants -> ... -> profit! + template<> + EIGEN_DEFINE_FUNCTION_ALLOWING_MULTIPLE_DEFINITIONS EIGEN_UNUSED Packet8f plog(const Packet8f &_x) + { + Packet8f x = _x; + _EIGEN_DECLARE_CONST_Packet8f(1, 1.0f); + _EIGEN_DECLARE_CONST_Packet8f(half, 0.5f); + _EIGEN_DECLARE_CONST_Packet8f(126f, 126.0f); + + _EIGEN_DECLARE_CONST_Packet8f_FROM_INT(inv_mant_mask, ~0x7f800000); + + // The smallest non denormalized float number. + _EIGEN_DECLARE_CONST_Packet8f_FROM_INT(min_norm_pos, 0x00800000); + _EIGEN_DECLARE_CONST_Packet8f_FROM_INT(minus_inf, 0xff800000); + + // Polynomial coefficients. + _EIGEN_DECLARE_CONST_Packet8f(cephes_SQRTHF, 0.707106781186547524f); + _EIGEN_DECLARE_CONST_Packet8f(cephes_log_p0, 7.0376836292E-2f); + _EIGEN_DECLARE_CONST_Packet8f(cephes_log_p1, -1.1514610310E-1f); + _EIGEN_DECLARE_CONST_Packet8f(cephes_log_p2, 1.1676998740E-1f); + _EIGEN_DECLARE_CONST_Packet8f(cephes_log_p3, -1.2420140846E-1f); + _EIGEN_DECLARE_CONST_Packet8f(cephes_log_p4, +1.4249322787E-1f); + _EIGEN_DECLARE_CONST_Packet8f(cephes_log_p5, -1.6668057665E-1f); + _EIGEN_DECLARE_CONST_Packet8f(cephes_log_p6, +2.0000714765E-1f); + _EIGEN_DECLARE_CONST_Packet8f(cephes_log_p7, -2.4999993993E-1f); + _EIGEN_DECLARE_CONST_Packet8f(cephes_log_p8, +3.3333331174E-1f); + _EIGEN_DECLARE_CONST_Packet8f(cephes_log_q1, -2.12194440e-4f); + _EIGEN_DECLARE_CONST_Packet8f(cephes_log_q2, 0.693359375f); + + Packet8f invalid_mask = _mm256_cmp_ps(x, _mm256_setzero_ps(), _CMP_NGE_UQ);// not greater equal is true if x is NaN + Packet8f iszero_mask = _mm256_cmp_ps(x, _mm256_setzero_ps(), _CMP_EQ_OQ); + + // Truncate input values to the minimum positive normal. + x = pmax(x, p8f_min_norm_pos); + + Packet8f emm0 = pshiftright(x, 23); + Packet8f e = _mm256_sub_ps(emm0, p8f_126f); + + // Set the exponents to -1, i.e. x are in the range [0.5,1). + x = _mm256_and_ps(x, p8f_inv_mant_mask); + x = _mm256_or_ps(x, p8f_half); + + // part2: Shift the inputs from the range [0.5,1) to [sqrt(1/2),sqrt(2)) + // and shift by -1. The values are then centered around 0, which improves + // the stability of the polynomial evaluation. + // if( x < SQRTHF ) { + // e -= 1; + // x = x + x - 1.0; + // } else { x = x - 1.0; } + Packet8f mask = _mm256_cmp_ps(x, p8f_cephes_SQRTHF, _CMP_LT_OQ); + Packet8f tmp = _mm256_and_ps(x, mask); + x = psub(x, p8f_1); + e = psub(e, _mm256_and_ps(p8f_1, mask)); + x = padd(x, tmp); + + Packet8f x2 = pmul(x, x); + Packet8f x3 = pmul(x2, x); + + // Evaluate the polynomial approximant of degree 8 in three parts, probably + // to improve instruction-level parallelism. + Packet8f y, y1, y2; + y = pmadd(p8f_cephes_log_p0, x, p8f_cephes_log_p1); + y1 = pmadd(p8f_cephes_log_p3, x, p8f_cephes_log_p4); + y2 = pmadd(p8f_cephes_log_p6, x, p8f_cephes_log_p7); + y = pmadd(y, x, p8f_cephes_log_p2); + y1 = pmadd(y1, x, p8f_cephes_log_p5); + y2 = pmadd(y2, x, p8f_cephes_log_p8); + y = pmadd(y, x3, y1); + y = pmadd(y, x3, y2); + y = pmul(y, x3); + + // Add the logarithm of the exponent back to the result of the interpolation. + y1 = pmul(e, p8f_cephes_log_q1); + tmp = pmul(x2, p8f_half); + y = padd(y, y1); + x = psub(x, tmp); + y2 = pmul(e, p8f_cephes_log_q2); + x = padd(x, y); + x = padd(x, y2); + + // Filter out invalid inputs, i.e. negative arg will be NAN, 0 will be -INF. + return _mm256_or_ps( + _mm256_andnot_ps(iszero_mask, _mm256_or_ps(x, invalid_mask)), _mm256_and_ps(iszero_mask, p8f_minus_inf)); + } + + // Exponential function. Works by writing "x = m*log(2) + r" where + // "m = floor(x/log(2)+1/2)" and "r" is the remainder. The result is then + // "exp(x) = 2^m*exp(r)" where exp(r) is in the range [-1,1). + template<> + EIGEN_DEFINE_FUNCTION_ALLOWING_MULTIPLE_DEFINITIONS EIGEN_UNUSED Packet8f pexp(const Packet8f &_x) + { + _EIGEN_DECLARE_CONST_Packet8f(1, 1.0f); + _EIGEN_DECLARE_CONST_Packet8f(half, 0.5f); + _EIGEN_DECLARE_CONST_Packet8f(127, 127.0f); + + _EIGEN_DECLARE_CONST_Packet8f(exp_hi, 88.3762626647950f); + _EIGEN_DECLARE_CONST_Packet8f(exp_lo, -88.3762626647949f); + + _EIGEN_DECLARE_CONST_Packet8f(cephes_LOG2EF, 1.44269504088896341f); + + _EIGEN_DECLARE_CONST_Packet8f(cephes_exp_p0, 1.9875691500E-4f); + _EIGEN_DECLARE_CONST_Packet8f(cephes_exp_p1, 1.3981999507E-3f); + _EIGEN_DECLARE_CONST_Packet8f(cephes_exp_p2, 8.3334519073E-3f); + _EIGEN_DECLARE_CONST_Packet8f(cephes_exp_p3, 4.1665795894E-2f); + _EIGEN_DECLARE_CONST_Packet8f(cephes_exp_p4, 1.6666665459E-1f); + _EIGEN_DECLARE_CONST_Packet8f(cephes_exp_p5, 5.0000001201E-1f); + + // Clamp x. + Packet8f x = pmax(pmin(_x, p8f_exp_hi), p8f_exp_lo); + + // Express exp(x) as exp(m*ln(2) + r), start by extracting + // m = floor(x/ln(2) + 0.5). + Packet8f m = _mm256_floor_ps(pmadd(x, p8f_cephes_LOG2EF, p8f_half)); // Get r = x - m*ln(2). If no FMA instructions are available, m*ln(2) is // subtracted out in two parts, m*C1+m*C2 = m*ln(2), to avoid accumulating // truncation errors. Note that we don't use the "pmadd" function here to // ensure that a precision-preserving FMA instruction is used. #ifdef EIGEN_VECTORIZE_FMA - _EIGEN_DECLARE_CONST_Packet8f(nln2, -0.6931471805599453f); - Packet8f r = _mm256_fmadd_ps(m, p8f_nln2, x); + _EIGEN_DECLARE_CONST_Packet8f(nln2, -0.6931471805599453f); + Packet8f r = _mm256_fmadd_ps(m, p8f_nln2, x); #else - _EIGEN_DECLARE_CONST_Packet8f(cephes_exp_C1, 0.693359375f); - _EIGEN_DECLARE_CONST_Packet8f(cephes_exp_C2, -2.12194440e-4f); - Packet8f r = psub(x, pmul(m, p8f_cephes_exp_C1)); - r = psub(r, pmul(m, p8f_cephes_exp_C2)); + _EIGEN_DECLARE_CONST_Packet8f(cephes_exp_C1, 0.693359375f); + _EIGEN_DECLARE_CONST_Packet8f(cephes_exp_C2, -2.12194440e-4f); + Packet8f r = psub(x, pmul(m, p8f_cephes_exp_C1)); + r = psub(r, pmul(m, p8f_cephes_exp_C2)); #endif - Packet8f r2 = pmul(r, r); - - // TODO(gonnet): Split into odd/even polynomials and try to exploit - // instruction-level parallelism. - Packet8f y = p8f_cephes_exp_p0; - y = pmadd(y, r, p8f_cephes_exp_p1); - y = pmadd(y, r, p8f_cephes_exp_p2); - y = pmadd(y, r, p8f_cephes_exp_p3); - y = pmadd(y, r, p8f_cephes_exp_p4); - y = pmadd(y, r, p8f_cephes_exp_p5); - y = pmadd(y, r2, r); - y = padd(y, p8f_1); - - // Build emm0 = 2^m. - Packet8i emm0 = _mm256_cvttps_epi32(padd(m, p8f_127)); - emm0 = pshiftleft(emm0, 23); - - // Return 2^m * exp(r). - return pmax(pmul(y, _mm256_castsi256_ps(emm0)), _x); -} - -// Hyperbolic Tangent function. -template <> -EIGEN_DEFINE_FUNCTION_ALLOWING_MULTIPLE_DEFINITIONS EIGEN_UNUSED Packet8f -ptanh(const Packet8f& x) { - return internal::generic_fast_tanh_float(x); -} - -template <> -EIGEN_DEFINE_FUNCTION_ALLOWING_MULTIPLE_DEFINITIONS EIGEN_UNUSED Packet4d -pexp(const Packet4d& _x) { - Packet4d x = _x; - - _EIGEN_DECLARE_CONST_Packet4d(1, 1.0); - _EIGEN_DECLARE_CONST_Packet4d(2, 2.0); - _EIGEN_DECLARE_CONST_Packet4d(half, 0.5); - - _EIGEN_DECLARE_CONST_Packet4d(exp_hi, 709.437); - _EIGEN_DECLARE_CONST_Packet4d(exp_lo, -709.436139303); - - _EIGEN_DECLARE_CONST_Packet4d(cephes_LOG2EF, 1.4426950408889634073599); - - _EIGEN_DECLARE_CONST_Packet4d(cephes_exp_p0, 1.26177193074810590878e-4); - _EIGEN_DECLARE_CONST_Packet4d(cephes_exp_p1, 3.02994407707441961300e-2); - _EIGEN_DECLARE_CONST_Packet4d(cephes_exp_p2, 9.99999999999999999910e-1); - - _EIGEN_DECLARE_CONST_Packet4d(cephes_exp_q0, 3.00198505138664455042e-6); - _EIGEN_DECLARE_CONST_Packet4d(cephes_exp_q1, 2.52448340349684104192e-3); - _EIGEN_DECLARE_CONST_Packet4d(cephes_exp_q2, 2.27265548208155028766e-1); - _EIGEN_DECLARE_CONST_Packet4d(cephes_exp_q3, 2.00000000000000000009e0); - - _EIGEN_DECLARE_CONST_Packet4d(cephes_exp_C1, 0.693145751953125); - _EIGEN_DECLARE_CONST_Packet4d(cephes_exp_C2, 1.42860682030941723212e-6); - _EIGEN_DECLARE_CONST_Packet4i(1023, 1023); - - Packet4d tmp, fx; - - // clamp x - x = pmax(pmin(x, p4d_exp_hi), p4d_exp_lo); - // Express exp(x) as exp(g + n*log(2)). - fx = pmadd(p4d_cephes_LOG2EF, x, p4d_half); - - // Get the integer modulus of log(2), i.e. the "n" described above. - fx = _mm256_floor_pd(fx); - - // Get the remainder modulo log(2), i.e. the "g" described above. Subtract - // n*log(2) out in two steps, i.e. n*C1 + n*C2, C1+C2=log2 to get the last - // digits right. - tmp = pmul(fx, p4d_cephes_exp_C1); - Packet4d z = pmul(fx, p4d_cephes_exp_C2); - x = psub(x, tmp); - x = psub(x, z); - - Packet4d x2 = pmul(x, x); - - // Evaluate the numerator polynomial of the rational interpolant. - Packet4d px = p4d_cephes_exp_p0; - px = pmadd(px, x2, p4d_cephes_exp_p1); - px = pmadd(px, x2, p4d_cephes_exp_p2); - px = pmul(px, x); - - // Evaluate the denominator polynomial of the rational interpolant. - Packet4d qx = p4d_cephes_exp_q0; - qx = pmadd(qx, x2, p4d_cephes_exp_q1); - qx = pmadd(qx, x2, p4d_cephes_exp_q2); - qx = pmadd(qx, x2, p4d_cephes_exp_q3); - - // I don't really get this bit, copied from the SSE2 routines, so... - // TODO(gonnet): Figure out what is going on here, perhaps find a better - // rational interpolant? - x = _mm256_div_pd(px, psub(qx, px)); - x = pmadd(p4d_2, x, p4d_1); - - // Build e=2^n by constructing the exponents in a 128-bit vector and - // shifting them to where they belong in double-precision values. - __m128i emm0 = _mm256_cvtpd_epi32(fx); - emm0 = _mm_add_epi32(emm0, p4i_1023); - emm0 = _mm_shuffle_epi32(emm0, _MM_SHUFFLE(3, 1, 2, 0)); - __m128i lo = _mm_slli_epi64(emm0, 52); - __m128i hi = _mm_slli_epi64(_mm_srli_epi64(emm0, 32), 52); - __m256i e = _mm256_insertf128_si256(_mm256_setzero_si256(), lo, 0); - e = _mm256_insertf128_si256(e, hi, 1); - - // Construct the result 2^n * exp(g) = e * x. The max is used to catch - // non-finite values in the input. - return pmax(pmul(x, _mm256_castsi256_pd(e)), _x); -} + Packet8f r2 = pmul(r, r); + + // TODO(gonnet): Split into odd/even polynomials and try to exploit + // instruction-level parallelism. + Packet8f y = p8f_cephes_exp_p0; + y = pmadd(y, r, p8f_cephes_exp_p1); + y = pmadd(y, r, p8f_cephes_exp_p2); + y = pmadd(y, r, p8f_cephes_exp_p3); + y = pmadd(y, r, p8f_cephes_exp_p4); + y = pmadd(y, r, p8f_cephes_exp_p5); + y = pmadd(y, r2, r); + y = padd(y, p8f_1); + + // Build emm0 = 2^m. + Packet8i emm0 = _mm256_cvttps_epi32(padd(m, p8f_127)); + emm0 = pshiftleft(emm0, 23); + + // Return 2^m * exp(r). + return pmax(pmul(y, _mm256_castsi256_ps(emm0)), _x); + } + + // Hyperbolic Tangent function. + template<> + EIGEN_DEFINE_FUNCTION_ALLOWING_MULTIPLE_DEFINITIONS EIGEN_UNUSED Packet8f ptanh(const Packet8f &x) + { + return internal::generic_fast_tanh_float(x); + } + + template<> + EIGEN_DEFINE_FUNCTION_ALLOWING_MULTIPLE_DEFINITIONS EIGEN_UNUSED Packet4d pexp(const Packet4d &_x) + { + Packet4d x = _x; + + _EIGEN_DECLARE_CONST_Packet4d(1, 1.0); + _EIGEN_DECLARE_CONST_Packet4d(2, 2.0); + _EIGEN_DECLARE_CONST_Packet4d(half, 0.5); + + _EIGEN_DECLARE_CONST_Packet4d(exp_hi, 709.437); + _EIGEN_DECLARE_CONST_Packet4d(exp_lo, -709.436139303); + + _EIGEN_DECLARE_CONST_Packet4d(cephes_LOG2EF, 1.4426950408889634073599); + + _EIGEN_DECLARE_CONST_Packet4d(cephes_exp_p0, 1.26177193074810590878e-4); + _EIGEN_DECLARE_CONST_Packet4d(cephes_exp_p1, 3.02994407707441961300e-2); + _EIGEN_DECLARE_CONST_Packet4d(cephes_exp_p2, 9.99999999999999999910e-1); + + _EIGEN_DECLARE_CONST_Packet4d(cephes_exp_q0, 3.00198505138664455042e-6); + _EIGEN_DECLARE_CONST_Packet4d(cephes_exp_q1, 2.52448340349684104192e-3); + _EIGEN_DECLARE_CONST_Packet4d(cephes_exp_q2, 2.27265548208155028766e-1); + _EIGEN_DECLARE_CONST_Packet4d(cephes_exp_q3, 2.00000000000000000009e0); + + _EIGEN_DECLARE_CONST_Packet4d(cephes_exp_C1, 0.693145751953125); + _EIGEN_DECLARE_CONST_Packet4d(cephes_exp_C2, 1.42860682030941723212e-6); + _EIGEN_DECLARE_CONST_Packet4i(1023, 1023); + + Packet4d tmp, fx; + + // clamp x + x = pmax(pmin(x, p4d_exp_hi), p4d_exp_lo); + // Express exp(x) as exp(g + n*log(2)). + fx = pmadd(p4d_cephes_LOG2EF, x, p4d_half); + + // Get the integer modulus of log(2), i.e. the "n" described above. + fx = _mm256_floor_pd(fx); + + // Get the remainder modulo log(2), i.e. the "g" described above. Subtract + // n*log(2) out in two steps, i.e. n*C1 + n*C2, C1+C2=log2 to get the last + // digits right. + tmp = pmul(fx, p4d_cephes_exp_C1); + Packet4d z = pmul(fx, p4d_cephes_exp_C2); + x = psub(x, tmp); + x = psub(x, z); + + Packet4d x2 = pmul(x, x); + + // Evaluate the numerator polynomial of the rational interpolant. + Packet4d px = p4d_cephes_exp_p0; + px = pmadd(px, x2, p4d_cephes_exp_p1); + px = pmadd(px, x2, p4d_cephes_exp_p2); + px = pmul(px, x); + + // Evaluate the denominator polynomial of the rational interpolant. + Packet4d qx = p4d_cephes_exp_q0; + qx = pmadd(qx, x2, p4d_cephes_exp_q1); + qx = pmadd(qx, x2, p4d_cephes_exp_q2); + qx = pmadd(qx, x2, p4d_cephes_exp_q3); + + // I don't really get this bit, copied from the SSE2 routines, so... + // TODO(gonnet): Figure out what is going on here, perhaps find a better + // rational interpolant? + x = _mm256_div_pd(px, psub(qx, px)); + x = pmadd(p4d_2, x, p4d_1); + + // Build e=2^n by constructing the exponents in a 128-bit vector and + // shifting them to where they belong in double-precision values. + __m128i emm0 = _mm256_cvtpd_epi32(fx); + emm0 = _mm_add_epi32(emm0, p4i_1023); + emm0 = _mm_shuffle_epi32(emm0, _MM_SHUFFLE(3, 1, 2, 0)); + __m128i lo = _mm_slli_epi64(emm0, 52); + __m128i hi = _mm_slli_epi64(_mm_srli_epi64(emm0, 32), 52); + __m256i e = _mm256_insertf128_si256(_mm256_setzero_si256(), lo, 0); + e = _mm256_insertf128_si256(e, hi, 1); + + // Construct the result 2^n * exp(g) = e * x. The max is used to catch + // non-finite values in the input. + return pmax(pmul(x, _mm256_castsi256_pd(e)), _x); + } // Functions for sqrt. // The EIGEN_FAST_MATH version uses the _mm_rsqrt_ps approximation and one step @@ -361,79 +361,82 @@ pexp(const Packet4d& _x) { // effective latency. This is similar to Quake3's fast inverse square root. // For detail see here: http://www.beyond3d.com/content/articles/8/ #if EIGEN_FAST_MATH -template <> -EIGEN_DEFINE_FUNCTION_ALLOWING_MULTIPLE_DEFINITIONS EIGEN_UNUSED Packet8f -psqrt(const Packet8f& _x) { - Packet8f half = pmul(_x, pset1(.5f)); - Packet8f denormal_mask = _mm256_and_ps( - _mm256_cmp_ps(_x, pset1((std::numeric_limits::min)()), - _CMP_LT_OQ), - _mm256_cmp_ps(_x, _mm256_setzero_ps(), _CMP_GE_OQ)); - - // Compute approximate reciprocal sqrt. - Packet8f x = _mm256_rsqrt_ps(_x); - // Do a single step of Newton's iteration. - x = pmul(x, psub(pset1(1.5f), pmul(half, pmul(x,x)))); - // Flush results for denormals to zero. - return _mm256_andnot_ps(denormal_mask, pmul(_x,x)); -} + template<> + EIGEN_DEFINE_FUNCTION_ALLOWING_MULTIPLE_DEFINITIONS EIGEN_UNUSED Packet8f psqrt(const Packet8f &_x) + { + Packet8f half = pmul(_x, pset1(.5f)); + Packet8f denormal_mask = + _mm256_and_ps(_mm256_cmp_ps(_x, pset1((std::numeric_limits::min)()), _CMP_LT_OQ), + _mm256_cmp_ps(_x, _mm256_setzero_ps(), _CMP_GE_OQ)); + + // Compute approximate reciprocal sqrt. + Packet8f x = _mm256_rsqrt_ps(_x); + // Do a single step of Newton's iteration. + x = pmul(x, psub(pset1(1.5f), pmul(half, pmul(x, x)))); + // Flush results for denormals to zero. + return _mm256_andnot_ps(denormal_mask, pmul(_x, x)); + } #else -template <> EIGEN_DEFINE_FUNCTION_ALLOWING_MULTIPLE_DEFINITIONS EIGEN_UNUSED -Packet8f psqrt(const Packet8f& x) { - return _mm256_sqrt_ps(x); -} + template<> + EIGEN_DEFINE_FUNCTION_ALLOWING_MULTIPLE_DEFINITIONS EIGEN_UNUSED Packet8f psqrt(const Packet8f &x) + { + return _mm256_sqrt_ps(x); + } #endif -template <> EIGEN_DEFINE_FUNCTION_ALLOWING_MULTIPLE_DEFINITIONS EIGEN_UNUSED -Packet4d psqrt(const Packet4d& x) { - return _mm256_sqrt_pd(x); -} + template<> + EIGEN_DEFINE_FUNCTION_ALLOWING_MULTIPLE_DEFINITIONS EIGEN_UNUSED Packet4d psqrt(const Packet4d &x) + { + return _mm256_sqrt_pd(x); + } #if EIGEN_FAST_MATH -template<> EIGEN_DEFINE_FUNCTION_ALLOWING_MULTIPLE_DEFINITIONS EIGEN_UNUSED -Packet8f prsqrt(const Packet8f& _x) { - _EIGEN_DECLARE_CONST_Packet8f_FROM_INT(inf, 0x7f800000); - _EIGEN_DECLARE_CONST_Packet8f_FROM_INT(nan, 0x7fc00000); - _EIGEN_DECLARE_CONST_Packet8f(one_point_five, 1.5f); - _EIGEN_DECLARE_CONST_Packet8f(minus_half, -0.5f); - _EIGEN_DECLARE_CONST_Packet8f_FROM_INT(flt_min, 0x00800000); + template<> + EIGEN_DEFINE_FUNCTION_ALLOWING_MULTIPLE_DEFINITIONS EIGEN_UNUSED Packet8f prsqrt(const Packet8f &_x) + { + _EIGEN_DECLARE_CONST_Packet8f_FROM_INT(inf, 0x7f800000); + _EIGEN_DECLARE_CONST_Packet8f_FROM_INT(nan, 0x7fc00000); + _EIGEN_DECLARE_CONST_Packet8f(one_point_five, 1.5f); + _EIGEN_DECLARE_CONST_Packet8f(minus_half, -0.5f); + _EIGEN_DECLARE_CONST_Packet8f_FROM_INT(flt_min, 0x00800000); - Packet8f neg_half = pmul(_x, p8f_minus_half); + Packet8f neg_half = pmul(_x, p8f_minus_half); - // select only the inverse sqrt of positive normal inputs (denormals are - // flushed to zero and cause infs as well). - Packet8f le_zero_mask = _mm256_cmp_ps(_x, p8f_flt_min, _CMP_LT_OQ); - Packet8f x = _mm256_andnot_ps(le_zero_mask, _mm256_rsqrt_ps(_x)); + // select only the inverse sqrt of positive normal inputs (denormals are + // flushed to zero and cause infs as well). + Packet8f le_zero_mask = _mm256_cmp_ps(_x, p8f_flt_min, _CMP_LT_OQ); + Packet8f x = _mm256_andnot_ps(le_zero_mask, _mm256_rsqrt_ps(_x)); - // Fill in NaNs and Infs for the negative/zero entries. - Packet8f neg_mask = _mm256_cmp_ps(_x, _mm256_setzero_ps(), _CMP_LT_OQ); - Packet8f zero_mask = _mm256_andnot_ps(neg_mask, le_zero_mask); - Packet8f infs_and_nans = _mm256_or_ps(_mm256_and_ps(neg_mask, p8f_nan), - _mm256_and_ps(zero_mask, p8f_inf)); + // Fill in NaNs and Infs for the negative/zero entries. + Packet8f neg_mask = _mm256_cmp_ps(_x, _mm256_setzero_ps(), _CMP_LT_OQ); + Packet8f zero_mask = _mm256_andnot_ps(neg_mask, le_zero_mask); + Packet8f infs_and_nans = _mm256_or_ps(_mm256_and_ps(neg_mask, p8f_nan), _mm256_and_ps(zero_mask, p8f_inf)); - // Do a single step of Newton's iteration. - x = pmul(x, pmadd(neg_half, pmul(x, x), p8f_one_point_five)); + // Do a single step of Newton's iteration. + x = pmul(x, pmadd(neg_half, pmul(x, x), p8f_one_point_five)); - // Insert NaNs and Infs in all the right places. - return _mm256_or_ps(x, infs_and_nans); -} + // Insert NaNs and Infs in all the right places. + return _mm256_or_ps(x, infs_and_nans); + } #else -template <> EIGEN_DEFINE_FUNCTION_ALLOWING_MULTIPLE_DEFINITIONS EIGEN_UNUSED -Packet8f prsqrt(const Packet8f& x) { - _EIGEN_DECLARE_CONST_Packet8f(one, 1.0f); - return _mm256_div_ps(p8f_one, _mm256_sqrt_ps(x)); -} + template<> + EIGEN_DEFINE_FUNCTION_ALLOWING_MULTIPLE_DEFINITIONS EIGEN_UNUSED Packet8f prsqrt(const Packet8f &x) + { + _EIGEN_DECLARE_CONST_Packet8f(one, 1.0f); + return _mm256_div_ps(p8f_one, _mm256_sqrt_ps(x)); + } #endif -template <> EIGEN_DEFINE_FUNCTION_ALLOWING_MULTIPLE_DEFINITIONS EIGEN_UNUSED -Packet4d prsqrt(const Packet4d& x) { - _EIGEN_DECLARE_CONST_Packet4d(one, 1.0); - return _mm256_div_pd(p4d_one, _mm256_sqrt_pd(x)); -} + template<> + EIGEN_DEFINE_FUNCTION_ALLOWING_MULTIPLE_DEFINITIONS EIGEN_UNUSED Packet4d prsqrt(const Packet4d &x) + { + _EIGEN_DECLARE_CONST_Packet4d(one, 1.0); + return _mm256_div_pd(p4d_one, _mm256_sqrt_pd(x)); + } -} // end namespace internal +}// end namespace internal -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_MATH_FUNCTIONS_AVX_H +#endif// EIGEN_MATH_FUNCTIONS_AVX_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/arch/AVX/PacketMath.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/arch/AVX/PacketMath.h index 923a124b..d44a5641 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/arch/AVX/PacketMath.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/arch/AVX/PacketMath.h @@ -19,7 +19,7 @@ namespace internal { #endif #ifndef EIGEN_ARCH_DEFAULT_NUMBER_OF_REGISTERS -#define EIGEN_ARCH_DEFAULT_NUMBER_OF_REGISTERS (2*sizeof(void*)) +#define EIGEN_ARCH_DEFAULT_NUMBER_OF_REGISTERS (2 * sizeof(void *)) #endif #ifdef __FMA__ @@ -28,336 +28,490 @@ namespace internal { #endif #endif -typedef __m256 Packet8f; -typedef __m256i Packet8i; -typedef __m256d Packet4d; + typedef __m256 Packet8f; + typedef __m256i Packet8i; + typedef __m256d Packet4d; -template<> struct is_arithmetic<__m256> { enum { value = true }; }; -template<> struct is_arithmetic<__m256i> { enum { value = true }; }; -template<> struct is_arithmetic<__m256d> { enum { value = true }; }; + template<> struct is_arithmetic<__m256> + { + enum { value = true }; + }; + template<> struct is_arithmetic<__m256i> + { + enum { value = true }; + }; + template<> struct is_arithmetic<__m256d> + { + enum { value = true }; + }; -#define _EIGEN_DECLARE_CONST_Packet8f(NAME,X) \ - const Packet8f p8f_##NAME = pset1(X) +#define _EIGEN_DECLARE_CONST_Packet8f(NAME, X) const Packet8f p8f_##NAME = pset1(X) -#define _EIGEN_DECLARE_CONST_Packet4d(NAME,X) \ - const Packet4d p4d_##NAME = pset1(X) +#define _EIGEN_DECLARE_CONST_Packet4d(NAME, X) const Packet4d p4d_##NAME = pset1(X) -#define _EIGEN_DECLARE_CONST_Packet8f_FROM_INT(NAME,X) \ +#define _EIGEN_DECLARE_CONST_Packet8f_FROM_INT(NAME, X) \ const Packet8f p8f_##NAME = _mm256_castsi256_ps(pset1(X)) -#define _EIGEN_DECLARE_CONST_Packet8i(NAME,X) \ - const Packet8i p8i_##NAME = pset1(X) +#define _EIGEN_DECLARE_CONST_Packet8i(NAME, X) const Packet8i p8i_##NAME = pset1(X) // Use the packet_traits defined in AVX512/PacketMath.h instead if we're going // to leverage AVX512 instructions. #ifndef EIGEN_VECTORIZE_AVX512 -template<> struct packet_traits : default_packet_traits -{ - typedef Packet8f type; - typedef Packet4f half; - enum { - Vectorizable = 1, - AlignedOnScalar = 1, - size=8, - HasHalfPacket = 1, - - HasDiv = 1, - HasSin = EIGEN_FAST_MATH, - HasCos = 0, - HasLog = 1, - HasExp = 1, - HasSqrt = 1, - HasRsqrt = 1, - HasTanh = EIGEN_FAST_MATH, - HasBlend = 1, - HasRound = 1, - HasFloor = 1, - HasCeil = 1 + template<> struct packet_traits : default_packet_traits + { + typedef Packet8f type; + typedef Packet4f half; + enum { + Vectorizable = 1, + AlignedOnScalar = 1, + size = 8, + HasHalfPacket = 1, + + HasDiv = 1, + HasSin = EIGEN_FAST_MATH, + HasCos = 0, + HasLog = 1, + HasExp = 1, + HasSqrt = 1, + HasRsqrt = 1, + HasTanh = EIGEN_FAST_MATH, + HasBlend = 1, + HasRound = 1, + HasFloor = 1, + HasCeil = 1 + }; }; -}; -template<> struct packet_traits : default_packet_traits -{ - typedef Packet4d type; - typedef Packet2d half; - enum { - Vectorizable = 1, - AlignedOnScalar = 1, - size=4, - HasHalfPacket = 1, - - HasDiv = 1, - HasExp = 1, - HasSqrt = 1, - HasRsqrt = 1, - HasBlend = 1, - HasRound = 1, - HasFloor = 1, - HasCeil = 1 + template<> struct packet_traits : default_packet_traits + { + typedef Packet4d type; + typedef Packet2d half; + enum { + Vectorizable = 1, + AlignedOnScalar = 1, + size = 4, + HasHalfPacket = 1, + + HasDiv = 1, + HasExp = 1, + HasSqrt = 1, + HasRsqrt = 1, + HasBlend = 1, + HasRound = 1, + HasFloor = 1, + HasCeil = 1 + }; }; -}; #endif -template<> struct scalar_div_cost { enum { value = 14 }; }; -template<> struct scalar_div_cost { enum { value = 16 }; }; - -/* Proper support for integers is only provided by AVX2. In the meantime, we'll - use SSE instructions and packets to deal with integers. -template<> struct packet_traits : default_packet_traits -{ - typedef Packet8i type; - enum { - Vectorizable = 1, - AlignedOnScalar = 1, - size=8 + template<> struct scalar_div_cost + { + enum { value = 14 }; + }; + template<> struct scalar_div_cost + { + enum { value = 16 }; }; -}; -*/ -template<> struct unpacket_traits { typedef float type; typedef Packet4f half; enum {size=8, alignment=Aligned32}; }; -template<> struct unpacket_traits { typedef double type; typedef Packet2d half; enum {size=4, alignment=Aligned32}; }; -template<> struct unpacket_traits { typedef int type; typedef Packet4i half; enum {size=8, alignment=Aligned32}; }; + /* Proper support for integers is only provided by AVX2. In the meantime, we'll + use SSE instructions and packets to deal with integers. + template<> struct packet_traits : default_packet_traits + { + typedef Packet8i type; + enum { + Vectorizable = 1, + AlignedOnScalar = 1, + size=8 + }; + }; + */ + + template<> struct unpacket_traits + { + typedef float type; + typedef Packet4f half; + enum { size = 8, alignment = Aligned32 }; + }; + template<> struct unpacket_traits + { + typedef double type; + typedef Packet2d half; + enum { size = 4, alignment = Aligned32 }; + }; + template<> struct unpacket_traits + { + typedef int type; + typedef Packet4i half; + enum { size = 8, alignment = Aligned32 }; + }; -template<> EIGEN_STRONG_INLINE Packet8f pset1(const float& from) { return _mm256_set1_ps(from); } -template<> EIGEN_STRONG_INLINE Packet4d pset1(const double& from) { return _mm256_set1_pd(from); } -template<> EIGEN_STRONG_INLINE Packet8i pset1(const int& from) { return _mm256_set1_epi32(from); } + template<> EIGEN_STRONG_INLINE Packet8f pset1(const float &from) { return _mm256_set1_ps(from); } + template<> EIGEN_STRONG_INLINE Packet4d pset1(const double &from) { return _mm256_set1_pd(from); } + template<> EIGEN_STRONG_INLINE Packet8i pset1(const int &from) { return _mm256_set1_epi32(from); } -template<> EIGEN_STRONG_INLINE Packet8f pload1(const float* from) { return _mm256_broadcast_ss(from); } -template<> EIGEN_STRONG_INLINE Packet4d pload1(const double* from) { return _mm256_broadcast_sd(from); } + template<> EIGEN_STRONG_INLINE Packet8f pload1(const float *from) { return _mm256_broadcast_ss(from); } + template<> EIGEN_STRONG_INLINE Packet4d pload1(const double *from) { return _mm256_broadcast_sd(from); } -template<> EIGEN_STRONG_INLINE Packet8f plset(const float& a) { return _mm256_add_ps(_mm256_set1_ps(a), _mm256_set_ps(7.0,6.0,5.0,4.0,3.0,2.0,1.0,0.0)); } -template<> EIGEN_STRONG_INLINE Packet4d plset(const double& a) { return _mm256_add_pd(_mm256_set1_pd(a), _mm256_set_pd(3.0,2.0,1.0,0.0)); } + template<> EIGEN_STRONG_INLINE Packet8f plset(const float &a) + { + return _mm256_add_ps(_mm256_set1_ps(a), _mm256_set_ps(7.0, 6.0, 5.0, 4.0, 3.0, 2.0, 1.0, 0.0)); + } + template<> EIGEN_STRONG_INLINE Packet4d plset(const double &a) + { + return _mm256_add_pd(_mm256_set1_pd(a), _mm256_set_pd(3.0, 2.0, 1.0, 0.0)); + } -template<> EIGEN_STRONG_INLINE Packet8f padd(const Packet8f& a, const Packet8f& b) { return _mm256_add_ps(a,b); } -template<> EIGEN_STRONG_INLINE Packet4d padd(const Packet4d& a, const Packet4d& b) { return _mm256_add_pd(a,b); } + template<> EIGEN_STRONG_INLINE Packet8f padd(const Packet8f &a, const Packet8f &b) + { + return _mm256_add_ps(a, b); + } + template<> EIGEN_STRONG_INLINE Packet4d padd(const Packet4d &a, const Packet4d &b) + { + return _mm256_add_pd(a, b); + } -template<> EIGEN_STRONG_INLINE Packet8f psub(const Packet8f& a, const Packet8f& b) { return _mm256_sub_ps(a,b); } -template<> EIGEN_STRONG_INLINE Packet4d psub(const Packet4d& a, const Packet4d& b) { return _mm256_sub_pd(a,b); } + template<> EIGEN_STRONG_INLINE Packet8f psub(const Packet8f &a, const Packet8f &b) + { + return _mm256_sub_ps(a, b); + } + template<> EIGEN_STRONG_INLINE Packet4d psub(const Packet4d &a, const Packet4d &b) + { + return _mm256_sub_pd(a, b); + } -template<> EIGEN_STRONG_INLINE Packet8f pnegate(const Packet8f& a) -{ - return _mm256_sub_ps(_mm256_set1_ps(0.0),a); -} -template<> EIGEN_STRONG_INLINE Packet4d pnegate(const Packet4d& a) -{ - return _mm256_sub_pd(_mm256_set1_pd(0.0),a); -} + template<> EIGEN_STRONG_INLINE Packet8f pnegate(const Packet8f &a) { return _mm256_sub_ps(_mm256_set1_ps(0.0), a); } + template<> EIGEN_STRONG_INLINE Packet4d pnegate(const Packet4d &a) { return _mm256_sub_pd(_mm256_set1_pd(0.0), a); } -template<> EIGEN_STRONG_INLINE Packet8f pconj(const Packet8f& a) { return a; } -template<> EIGEN_STRONG_INLINE Packet4d pconj(const Packet4d& a) { return a; } -template<> EIGEN_STRONG_INLINE Packet8i pconj(const Packet8i& a) { return a; } + template<> EIGEN_STRONG_INLINE Packet8f pconj(const Packet8f &a) { return a; } + template<> EIGEN_STRONG_INLINE Packet4d pconj(const Packet4d &a) { return a; } + template<> EIGEN_STRONG_INLINE Packet8i pconj(const Packet8i &a) { return a; } -template<> EIGEN_STRONG_INLINE Packet8f pmul(const Packet8f& a, const Packet8f& b) { return _mm256_mul_ps(a,b); } -template<> EIGEN_STRONG_INLINE Packet4d pmul(const Packet4d& a, const Packet4d& b) { return _mm256_mul_pd(a,b); } + template<> EIGEN_STRONG_INLINE Packet8f pmul(const Packet8f &a, const Packet8f &b) + { + return _mm256_mul_ps(a, b); + } + template<> EIGEN_STRONG_INLINE Packet4d pmul(const Packet4d &a, const Packet4d &b) + { + return _mm256_mul_pd(a, b); + } -template<> EIGEN_STRONG_INLINE Packet8f pdiv(const Packet8f& a, const Packet8f& b) { return _mm256_div_ps(a,b); } -template<> EIGEN_STRONG_INLINE Packet4d pdiv(const Packet4d& a, const Packet4d& b) { return _mm256_div_pd(a,b); } -template<> EIGEN_STRONG_INLINE Packet8i pdiv(const Packet8i& /*a*/, const Packet8i& /*b*/) -{ eigen_assert(false && "packet integer division are not supported by AVX"); - return pset1(0); -} + template<> EIGEN_STRONG_INLINE Packet8f pdiv(const Packet8f &a, const Packet8f &b) + { + return _mm256_div_ps(a, b); + } + template<> EIGEN_STRONG_INLINE Packet4d pdiv(const Packet4d &a, const Packet4d &b) + { + return _mm256_div_pd(a, b); + } + template<> EIGEN_STRONG_INLINE Packet8i pdiv(const Packet8i & /*a*/, const Packet8i & /*b*/) + { + eigen_assert(false && "packet integer division are not supported by AVX"); + return pset1(0); + } #ifdef __FMA__ -template<> EIGEN_STRONG_INLINE Packet8f pmadd(const Packet8f& a, const Packet8f& b, const Packet8f& c) { -#if ( (EIGEN_COMP_GNUC_STRICT && EIGEN_COMP_GNUC<80) || (EIGEN_COMP_CLANG) ) - // Clang stupidly generates a vfmadd213ps instruction plus some vmovaps on registers, - // and even register spilling with clang>=6.0 (bug 1637). - // Gcc stupidly generates a vfmadd132ps instruction. - // So let's enforce it to generate a vfmadd231ps instruction since the most common use - // case is to accumulate the result of the product. - Packet8f res = c; - __asm__("vfmadd231ps %[a], %[b], %[c]" : [c] "+x" (res) : [a] "x" (a), [b] "x" (b)); - return res; + template<> EIGEN_STRONG_INLINE Packet8f pmadd(const Packet8f &a, const Packet8f &b, const Packet8f &c) + { +#if ((EIGEN_COMP_GNUC_STRICT && EIGEN_COMP_GNUC < 80) || (EIGEN_COMP_CLANG)) + // Clang stupidly generates a vfmadd213ps instruction plus some vmovaps on registers, + // and even register spilling with clang>=6.0 (bug 1637). + // Gcc stupidly generates a vfmadd132ps instruction. + // So let's enforce it to generate a vfmadd231ps instruction since the most common use + // case is to accumulate the result of the product. + Packet8f res = c; + __asm__("vfmadd231ps %[a], %[b], %[c]" : [c] "+x"(res) : [a] "x"(a), [b] "x"(b)); + return res; #else - return _mm256_fmadd_ps(a,b,c); + return _mm256_fmadd_ps(a, b, c); #endif -} -template<> EIGEN_STRONG_INLINE Packet4d pmadd(const Packet4d& a, const Packet4d& b, const Packet4d& c) { -#if ( (EIGEN_COMP_GNUC_STRICT && EIGEN_COMP_GNUC<80) || (EIGEN_COMP_CLANG) ) - // see above - Packet4d res = c; - __asm__("vfmadd231pd %[a], %[b], %[c]" : [c] "+x" (res) : [a] "x" (a), [b] "x" (b)); - return res; + } + template<> EIGEN_STRONG_INLINE Packet4d pmadd(const Packet4d &a, const Packet4d &b, const Packet4d &c) + { +#if ((EIGEN_COMP_GNUC_STRICT && EIGEN_COMP_GNUC < 80) || (EIGEN_COMP_CLANG)) + // see above + Packet4d res = c; + __asm__("vfmadd231pd %[a], %[b], %[c]" : [c] "+x"(res) : [a] "x"(a), [b] "x"(b)); + return res; #else - return _mm256_fmadd_pd(a,b,c); + return _mm256_fmadd_pd(a, b, c); #endif -} + } #endif -template<> EIGEN_STRONG_INLINE Packet8f pmin(const Packet8f& a, const Packet8f& b) { return _mm256_min_ps(a,b); } -template<> EIGEN_STRONG_INLINE Packet4d pmin(const Packet4d& a, const Packet4d& b) { return _mm256_min_pd(a,b); } - -template<> EIGEN_STRONG_INLINE Packet8f pmax(const Packet8f& a, const Packet8f& b) { return _mm256_max_ps(a,b); } -template<> EIGEN_STRONG_INLINE Packet4d pmax(const Packet4d& a, const Packet4d& b) { return _mm256_max_pd(a,b); } - -template<> EIGEN_STRONG_INLINE Packet8f pround(const Packet8f& a) { return _mm256_round_ps(a, _MM_FROUND_CUR_DIRECTION); } -template<> EIGEN_STRONG_INLINE Packet4d pround(const Packet4d& a) { return _mm256_round_pd(a, _MM_FROUND_CUR_DIRECTION); } - -template<> EIGEN_STRONG_INLINE Packet8f pceil(const Packet8f& a) { return _mm256_ceil_ps(a); } -template<> EIGEN_STRONG_INLINE Packet4d pceil(const Packet4d& a) { return _mm256_ceil_pd(a); } - -template<> EIGEN_STRONG_INLINE Packet8f pfloor(const Packet8f& a) { return _mm256_floor_ps(a); } -template<> EIGEN_STRONG_INLINE Packet4d pfloor(const Packet4d& a) { return _mm256_floor_pd(a); } - -template<> EIGEN_STRONG_INLINE Packet8f pand(const Packet8f& a, const Packet8f& b) { return _mm256_and_ps(a,b); } -template<> EIGEN_STRONG_INLINE Packet4d pand(const Packet4d& a, const Packet4d& b) { return _mm256_and_pd(a,b); } - -template<> EIGEN_STRONG_INLINE Packet8f por(const Packet8f& a, const Packet8f& b) { return _mm256_or_ps(a,b); } -template<> EIGEN_STRONG_INLINE Packet4d por(const Packet4d& a, const Packet4d& b) { return _mm256_or_pd(a,b); } - -template<> EIGEN_STRONG_INLINE Packet8f pxor(const Packet8f& a, const Packet8f& b) { return _mm256_xor_ps(a,b); } -template<> EIGEN_STRONG_INLINE Packet4d pxor(const Packet4d& a, const Packet4d& b) { return _mm256_xor_pd(a,b); } - -template<> EIGEN_STRONG_INLINE Packet8f pandnot(const Packet8f& a, const Packet8f& b) { return _mm256_andnot_ps(a,b); } -template<> EIGEN_STRONG_INLINE Packet4d pandnot(const Packet4d& a, const Packet4d& b) { return _mm256_andnot_pd(a,b); } - -template<> EIGEN_STRONG_INLINE Packet8f pload(const float* from) { EIGEN_DEBUG_ALIGNED_LOAD return _mm256_load_ps(from); } -template<> EIGEN_STRONG_INLINE Packet4d pload(const double* from) { EIGEN_DEBUG_ALIGNED_LOAD return _mm256_load_pd(from); } -template<> EIGEN_STRONG_INLINE Packet8i pload(const int* from) { EIGEN_DEBUG_ALIGNED_LOAD return _mm256_load_si256(reinterpret_cast(from)); } - -template<> EIGEN_STRONG_INLINE Packet8f ploadu(const float* from) { EIGEN_DEBUG_UNALIGNED_LOAD return _mm256_loadu_ps(from); } -template<> EIGEN_STRONG_INLINE Packet4d ploadu(const double* from) { EIGEN_DEBUG_UNALIGNED_LOAD return _mm256_loadu_pd(from); } -template<> EIGEN_STRONG_INLINE Packet8i ploadu(const int* from) { EIGEN_DEBUG_UNALIGNED_LOAD return _mm256_loadu_si256(reinterpret_cast(from)); } - -// Loads 4 floats from memory a returns the packet {a0, a0 a1, a1, a2, a2, a3, a3} -template<> EIGEN_STRONG_INLINE Packet8f ploaddup(const float* from) -{ - // TODO try to find a way to avoid the need of a temporary register -// Packet8f tmp = _mm256_castps128_ps256(_mm_loadu_ps(from)); -// tmp = _mm256_insertf128_ps(tmp, _mm_movehl_ps(_mm256_castps256_ps128(tmp),_mm256_castps256_ps128(tmp)), 1); -// return _mm256_unpacklo_ps(tmp,tmp); - - // _mm256_insertf128_ps is very slow on Haswell, thus: - Packet8f tmp = _mm256_broadcast_ps((const __m128*)(const void*)from); - // mimic an "inplace" permutation of the lower 128bits using a blend - tmp = _mm256_blend_ps(tmp,_mm256_castps128_ps256(_mm_permute_ps( _mm256_castps256_ps128(tmp), _MM_SHUFFLE(1,0,1,0))), 15); - // then we can perform a consistent permutation on the global register to get everything in shape: - return _mm256_permute_ps(tmp, _MM_SHUFFLE(3,3,2,2)); -} -// Loads 2 doubles from memory a returns the packet {a0, a0 a1, a1} -template<> EIGEN_STRONG_INLINE Packet4d ploaddup(const double* from) -{ - Packet4d tmp = _mm256_broadcast_pd((const __m128d*)(const void*)from); - return _mm256_permute_pd(tmp, 3<<2); -} - -// Loads 2 floats from memory a returns the packet {a0, a0 a0, a0, a1, a1, a1, a1} -template<> EIGEN_STRONG_INLINE Packet8f ploadquad(const float* from) -{ - Packet8f tmp = _mm256_castps128_ps256(_mm_broadcast_ss(from)); - return _mm256_insertf128_ps(tmp, _mm_broadcast_ss(from+1), 1); -} - -template<> EIGEN_STRONG_INLINE void pstore(float* to, const Packet8f& from) { EIGEN_DEBUG_ALIGNED_STORE _mm256_store_ps(to, from); } -template<> EIGEN_STRONG_INLINE void pstore(double* to, const Packet4d& from) { EIGEN_DEBUG_ALIGNED_STORE _mm256_store_pd(to, from); } -template<> EIGEN_STRONG_INLINE void pstore(int* to, const Packet8i& from) { EIGEN_DEBUG_ALIGNED_STORE _mm256_storeu_si256(reinterpret_cast<__m256i*>(to), from); } - -template<> EIGEN_STRONG_INLINE void pstoreu(float* to, const Packet8f& from) { EIGEN_DEBUG_UNALIGNED_STORE _mm256_storeu_ps(to, from); } -template<> EIGEN_STRONG_INLINE void pstoreu(double* to, const Packet4d& from) { EIGEN_DEBUG_UNALIGNED_STORE _mm256_storeu_pd(to, from); } -template<> EIGEN_STRONG_INLINE void pstoreu(int* to, const Packet8i& from) { EIGEN_DEBUG_UNALIGNED_STORE _mm256_storeu_si256(reinterpret_cast<__m256i*>(to), from); } - -// NOTE: leverage _mm256_i32gather_ps and _mm256_i32gather_pd if AVX2 instructions are available -// NOTE: for the record the following seems to be slower: return _mm256_i32gather_ps(from, _mm256_set1_epi32(stride), 4); -template<> EIGEN_DEVICE_FUNC inline Packet8f pgather(const float* from, Index stride) -{ - return _mm256_set_ps(from[7*stride], from[6*stride], from[5*stride], from[4*stride], - from[3*stride], from[2*stride], from[1*stride], from[0*stride]); -} -template<> EIGEN_DEVICE_FUNC inline Packet4d pgather(const double* from, Index stride) -{ - return _mm256_set_pd(from[3*stride], from[2*stride], from[1*stride], from[0*stride]); -} - -template<> EIGEN_DEVICE_FUNC inline void pscatter(float* to, const Packet8f& from, Index stride) -{ - __m128 low = _mm256_extractf128_ps(from, 0); - to[stride*0] = _mm_cvtss_f32(low); - to[stride*1] = _mm_cvtss_f32(_mm_shuffle_ps(low, low, 1)); - to[stride*2] = _mm_cvtss_f32(_mm_shuffle_ps(low, low, 2)); - to[stride*3] = _mm_cvtss_f32(_mm_shuffle_ps(low, low, 3)); - - __m128 high = _mm256_extractf128_ps(from, 1); - to[stride*4] = _mm_cvtss_f32(high); - to[stride*5] = _mm_cvtss_f32(_mm_shuffle_ps(high, high, 1)); - to[stride*6] = _mm_cvtss_f32(_mm_shuffle_ps(high, high, 2)); - to[stride*7] = _mm_cvtss_f32(_mm_shuffle_ps(high, high, 3)); -} -template<> EIGEN_DEVICE_FUNC inline void pscatter(double* to, const Packet4d& from, Index stride) -{ - __m128d low = _mm256_extractf128_pd(from, 0); - to[stride*0] = _mm_cvtsd_f64(low); - to[stride*1] = _mm_cvtsd_f64(_mm_shuffle_pd(low, low, 1)); - __m128d high = _mm256_extractf128_pd(from, 1); - to[stride*2] = _mm_cvtsd_f64(high); - to[stride*3] = _mm_cvtsd_f64(_mm_shuffle_pd(high, high, 1)); -} - -template<> EIGEN_STRONG_INLINE void pstore1(float* to, const float& a) -{ - Packet8f pa = pset1(a); - pstore(to, pa); -} -template<> EIGEN_STRONG_INLINE void pstore1(double* to, const double& a) -{ - Packet4d pa = pset1(a); - pstore(to, pa); -} -template<> EIGEN_STRONG_INLINE void pstore1(int* to, const int& a) -{ - Packet8i pa = pset1(a); - pstore(to, pa); -} + template<> EIGEN_STRONG_INLINE Packet8f pmin(const Packet8f &a, const Packet8f &b) + { + return _mm256_min_ps(a, b); + } + template<> EIGEN_STRONG_INLINE Packet4d pmin(const Packet4d &a, const Packet4d &b) + { + return _mm256_min_pd(a, b); + } + + template<> EIGEN_STRONG_INLINE Packet8f pmax(const Packet8f &a, const Packet8f &b) + { + return _mm256_max_ps(a, b); + } + template<> EIGEN_STRONG_INLINE Packet4d pmax(const Packet4d &a, const Packet4d &b) + { + return _mm256_max_pd(a, b); + } + + template<> EIGEN_STRONG_INLINE Packet8f pround(const Packet8f &a) + { + return _mm256_round_ps(a, _MM_FROUND_CUR_DIRECTION); + } + template<> EIGEN_STRONG_INLINE Packet4d pround(const Packet4d &a) + { + return _mm256_round_pd(a, _MM_FROUND_CUR_DIRECTION); + } + + template<> EIGEN_STRONG_INLINE Packet8f pceil(const Packet8f &a) { return _mm256_ceil_ps(a); } + template<> EIGEN_STRONG_INLINE Packet4d pceil(const Packet4d &a) { return _mm256_ceil_pd(a); } + + template<> EIGEN_STRONG_INLINE Packet8f pfloor(const Packet8f &a) { return _mm256_floor_ps(a); } + template<> EIGEN_STRONG_INLINE Packet4d pfloor(const Packet4d &a) { return _mm256_floor_pd(a); } + + template<> EIGEN_STRONG_INLINE Packet8f pand(const Packet8f &a, const Packet8f &b) + { + return _mm256_and_ps(a, b); + } + template<> EIGEN_STRONG_INLINE Packet4d pand(const Packet4d &a, const Packet4d &b) + { + return _mm256_and_pd(a, b); + } + + template<> EIGEN_STRONG_INLINE Packet8f por(const Packet8f &a, const Packet8f &b) + { + return _mm256_or_ps(a, b); + } + template<> EIGEN_STRONG_INLINE Packet4d por(const Packet4d &a, const Packet4d &b) + { + return _mm256_or_pd(a, b); + } + + template<> EIGEN_STRONG_INLINE Packet8f pxor(const Packet8f &a, const Packet8f &b) + { + return _mm256_xor_ps(a, b); + } + template<> EIGEN_STRONG_INLINE Packet4d pxor(const Packet4d &a, const Packet4d &b) + { + return _mm256_xor_pd(a, b); + } + + template<> EIGEN_STRONG_INLINE Packet8f pandnot(const Packet8f &a, const Packet8f &b) + { + return _mm256_andnot_ps(a, b); + } + template<> EIGEN_STRONG_INLINE Packet4d pandnot(const Packet4d &a, const Packet4d &b) + { + return _mm256_andnot_pd(a, b); + } + + template<> EIGEN_STRONG_INLINE Packet8f pload(const float *from) + { + EIGEN_DEBUG_ALIGNED_LOAD return _mm256_load_ps(from); + } + template<> EIGEN_STRONG_INLINE Packet4d pload(const double *from) + { + EIGEN_DEBUG_ALIGNED_LOAD return _mm256_load_pd(from); + } + template<> EIGEN_STRONG_INLINE Packet8i pload(const int *from) + { + EIGEN_DEBUG_ALIGNED_LOAD return _mm256_load_si256(reinterpret_cast(from)); + } + + template<> EIGEN_STRONG_INLINE Packet8f ploadu(const float *from) + { + EIGEN_DEBUG_UNALIGNED_LOAD return _mm256_loadu_ps(from); + } + template<> EIGEN_STRONG_INLINE Packet4d ploadu(const double *from) + { + EIGEN_DEBUG_UNALIGNED_LOAD return _mm256_loadu_pd(from); + } + template<> EIGEN_STRONG_INLINE Packet8i ploadu(const int *from) + { + EIGEN_DEBUG_UNALIGNED_LOAD return _mm256_loadu_si256(reinterpret_cast(from)); + } + + // Loads 4 floats from memory a returns the packet {a0, a0 a1, a1, a2, a2, a3, a3} + template<> EIGEN_STRONG_INLINE Packet8f ploaddup(const float *from) + { + // TODO try to find a way to avoid the need of a temporary register + // Packet8f tmp = _mm256_castps128_ps256(_mm_loadu_ps(from)); + // tmp = _mm256_insertf128_ps(tmp, _mm_movehl_ps(_mm256_castps256_ps128(tmp),_mm256_castps256_ps128(tmp)), 1); + // return _mm256_unpacklo_ps(tmp,tmp); + + // _mm256_insertf128_ps is very slow on Haswell, thus: + Packet8f tmp = _mm256_broadcast_ps((const __m128 *)(const void *)from); + // mimic an "inplace" permutation of the lower 128bits using a blend + tmp = _mm256_blend_ps( + tmp, _mm256_castps128_ps256(_mm_permute_ps(_mm256_castps256_ps128(tmp), _MM_SHUFFLE(1, 0, 1, 0))), 15); + // then we can perform a consistent permutation on the global register to get everything in shape: + return _mm256_permute_ps(tmp, _MM_SHUFFLE(3, 3, 2, 2)); + } + // Loads 2 doubles from memory a returns the packet {a0, a0 a1, a1} + template<> EIGEN_STRONG_INLINE Packet4d ploaddup(const double *from) + { + Packet4d tmp = _mm256_broadcast_pd((const __m128d *)(const void *)from); + return _mm256_permute_pd(tmp, 3 << 2); + } + + // Loads 2 floats from memory a returns the packet {a0, a0 a0, a0, a1, a1, a1, a1} + template<> EIGEN_STRONG_INLINE Packet8f ploadquad(const float *from) + { + Packet8f tmp = _mm256_castps128_ps256(_mm_broadcast_ss(from)); + return _mm256_insertf128_ps(tmp, _mm_broadcast_ss(from + 1), 1); + } + + template<> EIGEN_STRONG_INLINE void pstore(float *to, const Packet8f &from) + { + EIGEN_DEBUG_ALIGNED_STORE _mm256_store_ps(to, from); + } + template<> EIGEN_STRONG_INLINE void pstore(double *to, const Packet4d &from) + { + EIGEN_DEBUG_ALIGNED_STORE _mm256_store_pd(to, from); + } + template<> EIGEN_STRONG_INLINE void pstore(int *to, const Packet8i &from) + { + EIGEN_DEBUG_ALIGNED_STORE _mm256_storeu_si256(reinterpret_cast<__m256i *>(to), from); + } + + template<> EIGEN_STRONG_INLINE void pstoreu(float *to, const Packet8f &from) + { + EIGEN_DEBUG_UNALIGNED_STORE _mm256_storeu_ps(to, from); + } + template<> EIGEN_STRONG_INLINE void pstoreu(double *to, const Packet4d &from) + { + EIGEN_DEBUG_UNALIGNED_STORE _mm256_storeu_pd(to, from); + } + template<> EIGEN_STRONG_INLINE void pstoreu(int *to, const Packet8i &from) + { + EIGEN_DEBUG_UNALIGNED_STORE _mm256_storeu_si256(reinterpret_cast<__m256i *>(to), from); + } + + // NOTE: leverage _mm256_i32gather_ps and _mm256_i32gather_pd if AVX2 instructions are available + // NOTE: for the record the following seems to be slower: return _mm256_i32gather_ps(from, _mm256_set1_epi32(stride), + // 4); + template<> EIGEN_DEVICE_FUNC inline Packet8f pgather(const float *from, Index stride) + { + return _mm256_set_ps(from[7 * stride], + from[6 * stride], + from[5 * stride], + from[4 * stride], + from[3 * stride], + from[2 * stride], + from[1 * stride], + from[0 * stride]); + } + template<> EIGEN_DEVICE_FUNC inline Packet4d pgather(const double *from, Index stride) + { + return _mm256_set_pd(from[3 * stride], from[2 * stride], from[1 * stride], from[0 * stride]); + } + + template<> EIGEN_DEVICE_FUNC inline void pscatter(float *to, const Packet8f &from, Index stride) + { + __m128 low = _mm256_extractf128_ps(from, 0); + to[stride * 0] = _mm_cvtss_f32(low); + to[stride * 1] = _mm_cvtss_f32(_mm_shuffle_ps(low, low, 1)); + to[stride * 2] = _mm_cvtss_f32(_mm_shuffle_ps(low, low, 2)); + to[stride * 3] = _mm_cvtss_f32(_mm_shuffle_ps(low, low, 3)); + + __m128 high = _mm256_extractf128_ps(from, 1); + to[stride * 4] = _mm_cvtss_f32(high); + to[stride * 5] = _mm_cvtss_f32(_mm_shuffle_ps(high, high, 1)); + to[stride * 6] = _mm_cvtss_f32(_mm_shuffle_ps(high, high, 2)); + to[stride * 7] = _mm_cvtss_f32(_mm_shuffle_ps(high, high, 3)); + } + template<> EIGEN_DEVICE_FUNC inline void pscatter(double *to, const Packet4d &from, Index stride) + { + __m128d low = _mm256_extractf128_pd(from, 0); + to[stride * 0] = _mm_cvtsd_f64(low); + to[stride * 1] = _mm_cvtsd_f64(_mm_shuffle_pd(low, low, 1)); + __m128d high = _mm256_extractf128_pd(from, 1); + to[stride * 2] = _mm_cvtsd_f64(high); + to[stride * 3] = _mm_cvtsd_f64(_mm_shuffle_pd(high, high, 1)); + } + + template<> EIGEN_STRONG_INLINE void pstore1(float *to, const float &a) + { + Packet8f pa = pset1(a); + pstore(to, pa); + } + template<> EIGEN_STRONG_INLINE void pstore1(double *to, const double &a) + { + Packet4d pa = pset1(a); + pstore(to, pa); + } + template<> EIGEN_STRONG_INLINE void pstore1(int *to, const int &a) + { + Packet8i pa = pset1(a); + pstore(to, pa); + } #ifndef EIGEN_VECTORIZE_AVX512 -template<> EIGEN_STRONG_INLINE void prefetch(const float* addr) { _mm_prefetch((SsePrefetchPtrType)(addr), _MM_HINT_T0); } -template<> EIGEN_STRONG_INLINE void prefetch(const double* addr) { _mm_prefetch((SsePrefetchPtrType)(addr), _MM_HINT_T0); } -template<> EIGEN_STRONG_INLINE void prefetch(const int* addr) { _mm_prefetch((SsePrefetchPtrType)(addr), _MM_HINT_T0); } + template<> EIGEN_STRONG_INLINE void prefetch(const float *addr) + { + _mm_prefetch((SsePrefetchPtrType)(addr), _MM_HINT_T0); + } + template<> EIGEN_STRONG_INLINE void prefetch(const double *addr) + { + _mm_prefetch((SsePrefetchPtrType)(addr), _MM_HINT_T0); + } + template<> EIGEN_STRONG_INLINE void prefetch(const int *addr) + { + _mm_prefetch((SsePrefetchPtrType)(addr), _MM_HINT_T0); + } #endif -template<> EIGEN_STRONG_INLINE float pfirst(const Packet8f& a) { - return _mm_cvtss_f32(_mm256_castps256_ps128(a)); -} -template<> EIGEN_STRONG_INLINE double pfirst(const Packet4d& a) { - return _mm_cvtsd_f64(_mm256_castpd256_pd128(a)); -} -template<> EIGEN_STRONG_INLINE int pfirst(const Packet8i& a) { - return _mm_cvtsi128_si32(_mm256_castsi256_si128(a)); -} - - -template<> EIGEN_STRONG_INLINE Packet8f preverse(const Packet8f& a) -{ - __m256 tmp = _mm256_shuffle_ps(a,a,0x1b); - return _mm256_permute2f128_ps(tmp, tmp, 1); -} -template<> EIGEN_STRONG_INLINE Packet4d preverse(const Packet4d& a) -{ - __m256d tmp = _mm256_shuffle_pd(a,a,5); - return _mm256_permute2f128_pd(tmp, tmp, 1); - #if 0 + template<> EIGEN_STRONG_INLINE float pfirst(const Packet8f &a) + { + return _mm_cvtss_f32(_mm256_castps256_ps128(a)); + } + template<> EIGEN_STRONG_INLINE double pfirst(const Packet4d &a) + { + return _mm_cvtsd_f64(_mm256_castpd256_pd128(a)); + } + template<> EIGEN_STRONG_INLINE int pfirst(const Packet8i &a) + { + return _mm_cvtsi128_si32(_mm256_castsi256_si128(a)); + } + + + template<> EIGEN_STRONG_INLINE Packet8f preverse(const Packet8f &a) + { + __m256 tmp = _mm256_shuffle_ps(a, a, 0x1b); + return _mm256_permute2f128_ps(tmp, tmp, 1); + } + template<> EIGEN_STRONG_INLINE Packet4d preverse(const Packet4d &a) + { + __m256d tmp = _mm256_shuffle_pd(a, a, 5); + return _mm256_permute2f128_pd(tmp, tmp, 1); +#if 0 // This version is unlikely to be faster as _mm256_shuffle_ps and _mm256_permute_pd // exhibit the same latency/throughput, but it is here for future reference/benchmarking... __m256d swap_halves = _mm256_permute2f128_pd(a,a,1); return _mm256_permute_pd(swap_halves,5); - #endif -} - -// pabs should be ok -template<> EIGEN_STRONG_INLINE Packet8f pabs(const Packet8f& a) -{ - const Packet8f mask = _mm256_castsi256_ps(_mm256_setr_epi32(0x7FFFFFFF,0x7FFFFFFF,0x7FFFFFFF,0x7FFFFFFF,0x7FFFFFFF,0x7FFFFFFF,0x7FFFFFFF,0x7FFFFFFF)); - return _mm256_and_ps(a,mask); -} -template<> EIGEN_STRONG_INLINE Packet4d pabs(const Packet4d& a) -{ - const Packet4d mask = _mm256_castsi256_pd(_mm256_setr_epi32(0xFFFFFFFF,0x7FFFFFFF,0xFFFFFFFF,0x7FFFFFFF,0xFFFFFFFF,0x7FFFFFFF,0xFFFFFFFF,0x7FFFFFFF)); - return _mm256_and_pd(a,mask); -} - -// preduxp should be ok -// FIXME: why is this ok? why isn't the simply implementation working as expected? -template<> EIGEN_STRONG_INLINE Packet8f preduxp(const Packet8f* vecs) -{ +#endif + } + + // pabs should be ok + template<> EIGEN_STRONG_INLINE Packet8f pabs(const Packet8f &a) + { + const Packet8f mask = _mm256_castsi256_ps(_mm256_setr_epi32( + 0x7FFFFFFF, 0x7FFFFFFF, 0x7FFFFFFF, 0x7FFFFFFF, 0x7FFFFFFF, 0x7FFFFFFF, 0x7FFFFFFF, 0x7FFFFFFF)); + return _mm256_and_ps(a, mask); + } + template<> EIGEN_STRONG_INLINE Packet4d pabs(const Packet4d &a) + { + const Packet4d mask = _mm256_castsi256_pd(_mm256_setr_epi32( + 0xFFFFFFFF, 0x7FFFFFFF, 0xFFFFFFFF, 0x7FFFFFFF, 0xFFFFFFFF, 0x7FFFFFFF, 0xFFFFFFFF, 0x7FFFFFFF)); + return _mm256_and_pd(a, mask); + } + + // preduxp should be ok + // FIXME: why is this ok? why isn't the simply implementation working as expected? + template<> EIGEN_STRONG_INLINE Packet8f preduxp(const Packet8f *vecs) + { __m256 hsum1 = _mm256_hadd_ps(vecs[0], vecs[1]); __m256 hsum2 = _mm256_hadd_ps(vecs[2], vecs[3]); __m256 hsum3 = _mm256_hadd_ps(vecs[4], vecs[5]); @@ -368,10 +522,10 @@ template<> EIGEN_STRONG_INLINE Packet8f preduxp(const Packet8f* vecs) __m256 hsum7 = _mm256_hadd_ps(hsum3, hsum3); __m256 hsum8 = _mm256_hadd_ps(hsum4, hsum4); - __m256 perm1 = _mm256_permute2f128_ps(hsum5, hsum5, 0x23); - __m256 perm2 = _mm256_permute2f128_ps(hsum6, hsum6, 0x23); - __m256 perm3 = _mm256_permute2f128_ps(hsum7, hsum7, 0x23); - __m256 perm4 = _mm256_permute2f128_ps(hsum8, hsum8, 0x23); + __m256 perm1 = _mm256_permute2f128_ps(hsum5, hsum5, 0x23); + __m256 perm2 = _mm256_permute2f128_ps(hsum6, hsum6, 0x23); + __m256 perm3 = _mm256_permute2f128_ps(hsum7, hsum7, 0x23); + __m256 perm4 = _mm256_permute2f128_ps(hsum8, hsum8, 0x23); __m256 sum1 = _mm256_add_ps(perm1, hsum5); __m256 sum2 = _mm256_add_ps(perm2, hsum6); @@ -383,255 +537,251 @@ template<> EIGEN_STRONG_INLINE Packet8f preduxp(const Packet8f* vecs) __m256 final = _mm256_blend_ps(blend1, blend2, 0xf0); return final; -} -template<> EIGEN_STRONG_INLINE Packet4d preduxp(const Packet4d* vecs) -{ - Packet4d tmp0, tmp1; - - tmp0 = _mm256_hadd_pd(vecs[0], vecs[1]); - tmp0 = _mm256_add_pd(tmp0, _mm256_permute2f128_pd(tmp0, tmp0, 1)); - - tmp1 = _mm256_hadd_pd(vecs[2], vecs[3]); - tmp1 = _mm256_add_pd(tmp1, _mm256_permute2f128_pd(tmp1, tmp1, 1)); - - return _mm256_blend_pd(tmp0, tmp1, 0xC); -} - -template<> EIGEN_STRONG_INLINE float predux(const Packet8f& a) -{ - return predux(Packet4f(_mm_add_ps(_mm256_castps256_ps128(a),_mm256_extractf128_ps(a,1)))); -} -template<> EIGEN_STRONG_INLINE double predux(const Packet4d& a) -{ - return predux(Packet2d(_mm_add_pd(_mm256_castpd256_pd128(a),_mm256_extractf128_pd(a,1)))); -} - -template<> EIGEN_STRONG_INLINE Packet4f predux_downto4(const Packet8f& a) -{ - return _mm_add_ps(_mm256_castps256_ps128(a),_mm256_extractf128_ps(a,1)); -} - -template<> EIGEN_STRONG_INLINE float predux_mul(const Packet8f& a) -{ - Packet8f tmp; - tmp = _mm256_mul_ps(a, _mm256_permute2f128_ps(a,a,1)); - tmp = _mm256_mul_ps(tmp, _mm256_shuffle_ps(tmp,tmp,_MM_SHUFFLE(1,0,3,2))); - return pfirst(_mm256_mul_ps(tmp, _mm256_shuffle_ps(tmp,tmp,1))); -} -template<> EIGEN_STRONG_INLINE double predux_mul(const Packet4d& a) -{ - Packet4d tmp; - tmp = _mm256_mul_pd(a, _mm256_permute2f128_pd(a,a,1)); - return pfirst(_mm256_mul_pd(tmp, _mm256_shuffle_pd(tmp,tmp,1))); -} - -template<> EIGEN_STRONG_INLINE float predux_min(const Packet8f& a) -{ - Packet8f tmp = _mm256_min_ps(a, _mm256_permute2f128_ps(a,a,1)); - tmp = _mm256_min_ps(tmp, _mm256_shuffle_ps(tmp,tmp,_MM_SHUFFLE(1,0,3,2))); - return pfirst(_mm256_min_ps(tmp, _mm256_shuffle_ps(tmp,tmp,1))); -} -template<> EIGEN_STRONG_INLINE double predux_min(const Packet4d& a) -{ - Packet4d tmp = _mm256_min_pd(a, _mm256_permute2f128_pd(a,a,1)); - return pfirst(_mm256_min_pd(tmp, _mm256_shuffle_pd(tmp, tmp, 1))); -} - -template<> EIGEN_STRONG_INLINE float predux_max(const Packet8f& a) -{ - Packet8f tmp = _mm256_max_ps(a, _mm256_permute2f128_ps(a,a,1)); - tmp = _mm256_max_ps(tmp, _mm256_shuffle_ps(tmp,tmp,_MM_SHUFFLE(1,0,3,2))); - return pfirst(_mm256_max_ps(tmp, _mm256_shuffle_ps(tmp,tmp,1))); -} - -template<> EIGEN_STRONG_INLINE double predux_max(const Packet4d& a) -{ - Packet4d tmp = _mm256_max_pd(a, _mm256_permute2f128_pd(a,a,1)); - return pfirst(_mm256_max_pd(tmp, _mm256_shuffle_pd(tmp, tmp, 1))); -} - - -template -struct palign_impl -{ - static EIGEN_STRONG_INLINE void run(Packet8f& first, const Packet8f& second) - { - if (Offset==1) - { - first = _mm256_blend_ps(first, second, 1); - Packet8f tmp1 = _mm256_permute_ps (first, _MM_SHUFFLE(0,3,2,1)); - Packet8f tmp2 = _mm256_permute2f128_ps (tmp1, tmp1, 1); - first = _mm256_blend_ps(tmp1, tmp2, 0x88); - } - else if (Offset==2) - { - first = _mm256_blend_ps(first, second, 3); - Packet8f tmp1 = _mm256_permute_ps (first, _MM_SHUFFLE(1,0,3,2)); - Packet8f tmp2 = _mm256_permute2f128_ps (tmp1, tmp1, 1); - first = _mm256_blend_ps(tmp1, tmp2, 0xcc); - } - else if (Offset==3) - { - first = _mm256_blend_ps(first, second, 7); - Packet8f tmp1 = _mm256_permute_ps (first, _MM_SHUFFLE(2,1,0,3)); - Packet8f tmp2 = _mm256_permute2f128_ps (tmp1, tmp1, 1); - first = _mm256_blend_ps(tmp1, tmp2, 0xee); - } - else if (Offset==4) - { - first = _mm256_blend_ps(first, second, 15); - Packet8f tmp1 = _mm256_permute_ps (first, _MM_SHUFFLE(3,2,1,0)); - Packet8f tmp2 = _mm256_permute2f128_ps (tmp1, tmp1, 1); - first = _mm256_permute_ps(tmp2, _MM_SHUFFLE(3,2,1,0)); - } - else if (Offset==5) - { - first = _mm256_blend_ps(first, second, 31); - first = _mm256_permute2f128_ps(first, first, 1); - Packet8f tmp = _mm256_permute_ps (first, _MM_SHUFFLE(0,3,2,1)); - first = _mm256_permute2f128_ps(tmp, tmp, 1); - first = _mm256_blend_ps(tmp, first, 0x88); - } - else if (Offset==6) - { - first = _mm256_blend_ps(first, second, 63); - first = _mm256_permute2f128_ps(first, first, 1); - Packet8f tmp = _mm256_permute_ps (first, _MM_SHUFFLE(1,0,3,2)); - first = _mm256_permute2f128_ps(tmp, tmp, 1); - first = _mm256_blend_ps(tmp, first, 0xcc); - } - else if (Offset==7) - { - first = _mm256_blend_ps(first, second, 127); - first = _mm256_permute2f128_ps(first, first, 1); - Packet8f tmp = _mm256_permute_ps (first, _MM_SHUFFLE(2,1,0,3)); - first = _mm256_permute2f128_ps(tmp, tmp, 1); - first = _mm256_blend_ps(tmp, first, 0xee); - } } -}; + template<> EIGEN_STRONG_INLINE Packet4d preduxp(const Packet4d *vecs) + { + Packet4d tmp0, tmp1; + + tmp0 = _mm256_hadd_pd(vecs[0], vecs[1]); + tmp0 = _mm256_add_pd(tmp0, _mm256_permute2f128_pd(tmp0, tmp0, 1)); + + tmp1 = _mm256_hadd_pd(vecs[2], vecs[3]); + tmp1 = _mm256_add_pd(tmp1, _mm256_permute2f128_pd(tmp1, tmp1, 1)); + + return _mm256_blend_pd(tmp0, tmp1, 0xC); + } -template -struct palign_impl -{ - static EIGEN_STRONG_INLINE void run(Packet4d& first, const Packet4d& second) + template<> EIGEN_STRONG_INLINE float predux(const Packet8f &a) { - if (Offset==1) - { - first = _mm256_blend_pd(first, second, 1); - __m256d tmp = _mm256_permute_pd(first, 5); - first = _mm256_permute2f128_pd(tmp, tmp, 1); - first = _mm256_blend_pd(tmp, first, 0xA); - } - else if (Offset==2) + return predux(Packet4f(_mm_add_ps(_mm256_castps256_ps128(a), _mm256_extractf128_ps(a, 1)))); + } + template<> EIGEN_STRONG_INLINE double predux(const Packet4d &a) + { + return predux(Packet2d(_mm_add_pd(_mm256_castpd256_pd128(a), _mm256_extractf128_pd(a, 1)))); + } + + template<> EIGEN_STRONG_INLINE Packet4f predux_downto4(const Packet8f &a) + { + return _mm_add_ps(_mm256_castps256_ps128(a), _mm256_extractf128_ps(a, 1)); + } + + template<> EIGEN_STRONG_INLINE float predux_mul(const Packet8f &a) + { + Packet8f tmp; + tmp = _mm256_mul_ps(a, _mm256_permute2f128_ps(a, a, 1)); + tmp = _mm256_mul_ps(tmp, _mm256_shuffle_ps(tmp, tmp, _MM_SHUFFLE(1, 0, 3, 2))); + return pfirst(_mm256_mul_ps(tmp, _mm256_shuffle_ps(tmp, tmp, 1))); + } + template<> EIGEN_STRONG_INLINE double predux_mul(const Packet4d &a) + { + Packet4d tmp; + tmp = _mm256_mul_pd(a, _mm256_permute2f128_pd(a, a, 1)); + return pfirst(_mm256_mul_pd(tmp, _mm256_shuffle_pd(tmp, tmp, 1))); + } + + template<> EIGEN_STRONG_INLINE float predux_min(const Packet8f &a) + { + Packet8f tmp = _mm256_min_ps(a, _mm256_permute2f128_ps(a, a, 1)); + tmp = _mm256_min_ps(tmp, _mm256_shuffle_ps(tmp, tmp, _MM_SHUFFLE(1, 0, 3, 2))); + return pfirst(_mm256_min_ps(tmp, _mm256_shuffle_ps(tmp, tmp, 1))); + } + template<> EIGEN_STRONG_INLINE double predux_min(const Packet4d &a) + { + Packet4d tmp = _mm256_min_pd(a, _mm256_permute2f128_pd(a, a, 1)); + return pfirst(_mm256_min_pd(tmp, _mm256_shuffle_pd(tmp, tmp, 1))); + } + + template<> EIGEN_STRONG_INLINE float predux_max(const Packet8f &a) + { + Packet8f tmp = _mm256_max_ps(a, _mm256_permute2f128_ps(a, a, 1)); + tmp = _mm256_max_ps(tmp, _mm256_shuffle_ps(tmp, tmp, _MM_SHUFFLE(1, 0, 3, 2))); + return pfirst(_mm256_max_ps(tmp, _mm256_shuffle_ps(tmp, tmp, 1))); + } + + template<> EIGEN_STRONG_INLINE double predux_max(const Packet4d &a) + { + Packet4d tmp = _mm256_max_pd(a, _mm256_permute2f128_pd(a, a, 1)); + return pfirst(_mm256_max_pd(tmp, _mm256_shuffle_pd(tmp, tmp, 1))); + } + + + template struct palign_impl + { + static EIGEN_STRONG_INLINE void run(Packet8f &first, const Packet8f &second) { - first = _mm256_blend_pd(first, second, 3); - first = _mm256_permute2f128_pd(first, first, 1); + if (Offset == 1) { + first = _mm256_blend_ps(first, second, 1); + Packet8f tmp1 = _mm256_permute_ps(first, _MM_SHUFFLE(0, 3, 2, 1)); + Packet8f tmp2 = _mm256_permute2f128_ps(tmp1, tmp1, 1); + first = _mm256_blend_ps(tmp1, tmp2, 0x88); + } else if (Offset == 2) { + first = _mm256_blend_ps(first, second, 3); + Packet8f tmp1 = _mm256_permute_ps(first, _MM_SHUFFLE(1, 0, 3, 2)); + Packet8f tmp2 = _mm256_permute2f128_ps(tmp1, tmp1, 1); + first = _mm256_blend_ps(tmp1, tmp2, 0xcc); + } else if (Offset == 3) { + first = _mm256_blend_ps(first, second, 7); + Packet8f tmp1 = _mm256_permute_ps(first, _MM_SHUFFLE(2, 1, 0, 3)); + Packet8f tmp2 = _mm256_permute2f128_ps(tmp1, tmp1, 1); + first = _mm256_blend_ps(tmp1, tmp2, 0xee); + } else if (Offset == 4) { + first = _mm256_blend_ps(first, second, 15); + Packet8f tmp1 = _mm256_permute_ps(first, _MM_SHUFFLE(3, 2, 1, 0)); + Packet8f tmp2 = _mm256_permute2f128_ps(tmp1, tmp1, 1); + first = _mm256_permute_ps(tmp2, _MM_SHUFFLE(3, 2, 1, 0)); + } else if (Offset == 5) { + first = _mm256_blend_ps(first, second, 31); + first = _mm256_permute2f128_ps(first, first, 1); + Packet8f tmp = _mm256_permute_ps(first, _MM_SHUFFLE(0, 3, 2, 1)); + first = _mm256_permute2f128_ps(tmp, tmp, 1); + first = _mm256_blend_ps(tmp, first, 0x88); + } else if (Offset == 6) { + first = _mm256_blend_ps(first, second, 63); + first = _mm256_permute2f128_ps(first, first, 1); + Packet8f tmp = _mm256_permute_ps(first, _MM_SHUFFLE(1, 0, 3, 2)); + first = _mm256_permute2f128_ps(tmp, tmp, 1); + first = _mm256_blend_ps(tmp, first, 0xcc); + } else if (Offset == 7) { + first = _mm256_blend_ps(first, second, 127); + first = _mm256_permute2f128_ps(first, first, 1); + Packet8f tmp = _mm256_permute_ps(first, _MM_SHUFFLE(2, 1, 0, 3)); + first = _mm256_permute2f128_ps(tmp, tmp, 1); + first = _mm256_blend_ps(tmp, first, 0xee); + } } - else if (Offset==3) + }; + + template struct palign_impl + { + static EIGEN_STRONG_INLINE void run(Packet4d &first, const Packet4d &second) { - first = _mm256_blend_pd(first, second, 7); - __m256d tmp = _mm256_permute_pd(first, 5); - first = _mm256_permute2f128_pd(tmp, tmp, 1); - first = _mm256_blend_pd(tmp, first, 5); + if (Offset == 1) { + first = _mm256_blend_pd(first, second, 1); + __m256d tmp = _mm256_permute_pd(first, 5); + first = _mm256_permute2f128_pd(tmp, tmp, 1); + first = _mm256_blend_pd(tmp, first, 0xA); + } else if (Offset == 2) { + first = _mm256_blend_pd(first, second, 3); + first = _mm256_permute2f128_pd(first, first, 1); + } else if (Offset == 3) { + first = _mm256_blend_pd(first, second, 7); + __m256d tmp = _mm256_permute_pd(first, 5); + first = _mm256_permute2f128_pd(tmp, tmp, 1); + first = _mm256_blend_pd(tmp, first, 5); + } } + }; + + EIGEN_DEVICE_FUNC inline void ptranspose(PacketBlock &kernel) + { + __m256 T0 = _mm256_unpacklo_ps(kernel.packet[0], kernel.packet[1]); + __m256 T1 = _mm256_unpackhi_ps(kernel.packet[0], kernel.packet[1]); + __m256 T2 = _mm256_unpacklo_ps(kernel.packet[2], kernel.packet[3]); + __m256 T3 = _mm256_unpackhi_ps(kernel.packet[2], kernel.packet[3]); + __m256 T4 = _mm256_unpacklo_ps(kernel.packet[4], kernel.packet[5]); + __m256 T5 = _mm256_unpackhi_ps(kernel.packet[4], kernel.packet[5]); + __m256 T6 = _mm256_unpacklo_ps(kernel.packet[6], kernel.packet[7]); + __m256 T7 = _mm256_unpackhi_ps(kernel.packet[6], kernel.packet[7]); + __m256 S0 = _mm256_shuffle_ps(T0, T2, _MM_SHUFFLE(1, 0, 1, 0)); + __m256 S1 = _mm256_shuffle_ps(T0, T2, _MM_SHUFFLE(3, 2, 3, 2)); + __m256 S2 = _mm256_shuffle_ps(T1, T3, _MM_SHUFFLE(1, 0, 1, 0)); + __m256 S3 = _mm256_shuffle_ps(T1, T3, _MM_SHUFFLE(3, 2, 3, 2)); + __m256 S4 = _mm256_shuffle_ps(T4, T6, _MM_SHUFFLE(1, 0, 1, 0)); + __m256 S5 = _mm256_shuffle_ps(T4, T6, _MM_SHUFFLE(3, 2, 3, 2)); + __m256 S6 = _mm256_shuffle_ps(T5, T7, _MM_SHUFFLE(1, 0, 1, 0)); + __m256 S7 = _mm256_shuffle_ps(T5, T7, _MM_SHUFFLE(3, 2, 3, 2)); + kernel.packet[0] = _mm256_permute2f128_ps(S0, S4, 0x20); + kernel.packet[1] = _mm256_permute2f128_ps(S1, S5, 0x20); + kernel.packet[2] = _mm256_permute2f128_ps(S2, S6, 0x20); + kernel.packet[3] = _mm256_permute2f128_ps(S3, S7, 0x20); + kernel.packet[4] = _mm256_permute2f128_ps(S0, S4, 0x31); + kernel.packet[5] = _mm256_permute2f128_ps(S1, S5, 0x31); + kernel.packet[6] = _mm256_permute2f128_ps(S2, S6, 0x31); + kernel.packet[7] = _mm256_permute2f128_ps(S3, S7, 0x31); + } + + EIGEN_DEVICE_FUNC inline void ptranspose(PacketBlock &kernel) + { + __m256 T0 = _mm256_unpacklo_ps(kernel.packet[0], kernel.packet[1]); + __m256 T1 = _mm256_unpackhi_ps(kernel.packet[0], kernel.packet[1]); + __m256 T2 = _mm256_unpacklo_ps(kernel.packet[2], kernel.packet[3]); + __m256 T3 = _mm256_unpackhi_ps(kernel.packet[2], kernel.packet[3]); + + __m256 S0 = _mm256_shuffle_ps(T0, T2, _MM_SHUFFLE(1, 0, 1, 0)); + __m256 S1 = _mm256_shuffle_ps(T0, T2, _MM_SHUFFLE(3, 2, 3, 2)); + __m256 S2 = _mm256_shuffle_ps(T1, T3, _MM_SHUFFLE(1, 0, 1, 0)); + __m256 S3 = _mm256_shuffle_ps(T1, T3, _MM_SHUFFLE(3, 2, 3, 2)); + + kernel.packet[0] = _mm256_permute2f128_ps(S0, S1, 0x20); + kernel.packet[1] = _mm256_permute2f128_ps(S2, S3, 0x20); + kernel.packet[2] = _mm256_permute2f128_ps(S0, S1, 0x31); + kernel.packet[3] = _mm256_permute2f128_ps(S2, S3, 0x31); } -}; - -EIGEN_DEVICE_FUNC inline void -ptranspose(PacketBlock& kernel) { - __m256 T0 = _mm256_unpacklo_ps(kernel.packet[0], kernel.packet[1]); - __m256 T1 = _mm256_unpackhi_ps(kernel.packet[0], kernel.packet[1]); - __m256 T2 = _mm256_unpacklo_ps(kernel.packet[2], kernel.packet[3]); - __m256 T3 = _mm256_unpackhi_ps(kernel.packet[2], kernel.packet[3]); - __m256 T4 = _mm256_unpacklo_ps(kernel.packet[4], kernel.packet[5]); - __m256 T5 = _mm256_unpackhi_ps(kernel.packet[4], kernel.packet[5]); - __m256 T6 = _mm256_unpacklo_ps(kernel.packet[6], kernel.packet[7]); - __m256 T7 = _mm256_unpackhi_ps(kernel.packet[6], kernel.packet[7]); - __m256 S0 = _mm256_shuffle_ps(T0,T2,_MM_SHUFFLE(1,0,1,0)); - __m256 S1 = _mm256_shuffle_ps(T0,T2,_MM_SHUFFLE(3,2,3,2)); - __m256 S2 = _mm256_shuffle_ps(T1,T3,_MM_SHUFFLE(1,0,1,0)); - __m256 S3 = _mm256_shuffle_ps(T1,T3,_MM_SHUFFLE(3,2,3,2)); - __m256 S4 = _mm256_shuffle_ps(T4,T6,_MM_SHUFFLE(1,0,1,0)); - __m256 S5 = _mm256_shuffle_ps(T4,T6,_MM_SHUFFLE(3,2,3,2)); - __m256 S6 = _mm256_shuffle_ps(T5,T7,_MM_SHUFFLE(1,0,1,0)); - __m256 S7 = _mm256_shuffle_ps(T5,T7,_MM_SHUFFLE(3,2,3,2)); - kernel.packet[0] = _mm256_permute2f128_ps(S0, S4, 0x20); - kernel.packet[1] = _mm256_permute2f128_ps(S1, S5, 0x20); - kernel.packet[2] = _mm256_permute2f128_ps(S2, S6, 0x20); - kernel.packet[3] = _mm256_permute2f128_ps(S3, S7, 0x20); - kernel.packet[4] = _mm256_permute2f128_ps(S0, S4, 0x31); - kernel.packet[5] = _mm256_permute2f128_ps(S1, S5, 0x31); - kernel.packet[6] = _mm256_permute2f128_ps(S2, S6, 0x31); - kernel.packet[7] = _mm256_permute2f128_ps(S3, S7, 0x31); -} - -EIGEN_DEVICE_FUNC inline void -ptranspose(PacketBlock& kernel) { - __m256 T0 = _mm256_unpacklo_ps(kernel.packet[0], kernel.packet[1]); - __m256 T1 = _mm256_unpackhi_ps(kernel.packet[0], kernel.packet[1]); - __m256 T2 = _mm256_unpacklo_ps(kernel.packet[2], kernel.packet[3]); - __m256 T3 = _mm256_unpackhi_ps(kernel.packet[2], kernel.packet[3]); - - __m256 S0 = _mm256_shuffle_ps(T0,T2,_MM_SHUFFLE(1,0,1,0)); - __m256 S1 = _mm256_shuffle_ps(T0,T2,_MM_SHUFFLE(3,2,3,2)); - __m256 S2 = _mm256_shuffle_ps(T1,T3,_MM_SHUFFLE(1,0,1,0)); - __m256 S3 = _mm256_shuffle_ps(T1,T3,_MM_SHUFFLE(3,2,3,2)); - - kernel.packet[0] = _mm256_permute2f128_ps(S0, S1, 0x20); - kernel.packet[1] = _mm256_permute2f128_ps(S2, S3, 0x20); - kernel.packet[2] = _mm256_permute2f128_ps(S0, S1, 0x31); - kernel.packet[3] = _mm256_permute2f128_ps(S2, S3, 0x31); -} - -EIGEN_DEVICE_FUNC inline void -ptranspose(PacketBlock& kernel) { - __m256d T0 = _mm256_shuffle_pd(kernel.packet[0], kernel.packet[1], 15); - __m256d T1 = _mm256_shuffle_pd(kernel.packet[0], kernel.packet[1], 0); - __m256d T2 = _mm256_shuffle_pd(kernel.packet[2], kernel.packet[3], 15); - __m256d T3 = _mm256_shuffle_pd(kernel.packet[2], kernel.packet[3], 0); - - kernel.packet[1] = _mm256_permute2f128_pd(T0, T2, 32); - kernel.packet[3] = _mm256_permute2f128_pd(T0, T2, 49); - kernel.packet[0] = _mm256_permute2f128_pd(T1, T3, 32); - kernel.packet[2] = _mm256_permute2f128_pd(T1, T3, 49); -} - -template<> EIGEN_STRONG_INLINE Packet8f pblend(const Selector<8>& ifPacket, const Packet8f& thenPacket, const Packet8f& elsePacket) { - const __m256 zero = _mm256_setzero_ps(); - const __m256 select = _mm256_set_ps(ifPacket.select[7], ifPacket.select[6], ifPacket.select[5], ifPacket.select[4], ifPacket.select[3], ifPacket.select[2], ifPacket.select[1], ifPacket.select[0]); - __m256 false_mask = _mm256_cmp_ps(select, zero, _CMP_EQ_UQ); - return _mm256_blendv_ps(thenPacket, elsePacket, false_mask); -} -template<> EIGEN_STRONG_INLINE Packet4d pblend(const Selector<4>& ifPacket, const Packet4d& thenPacket, const Packet4d& elsePacket) { - const __m256d zero = _mm256_setzero_pd(); - const __m256d select = _mm256_set_pd(ifPacket.select[3], ifPacket.select[2], ifPacket.select[1], ifPacket.select[0]); - __m256d false_mask = _mm256_cmp_pd(select, zero, _CMP_EQ_UQ); - return _mm256_blendv_pd(thenPacket, elsePacket, false_mask); -} - -template<> EIGEN_STRONG_INLINE Packet8f pinsertfirst(const Packet8f& a, float b) -{ - return _mm256_blend_ps(a,pset1(b),1); -} - -template<> EIGEN_STRONG_INLINE Packet4d pinsertfirst(const Packet4d& a, double b) -{ - return _mm256_blend_pd(a,pset1(b),1); -} - -template<> EIGEN_STRONG_INLINE Packet8f pinsertlast(const Packet8f& a, float b) -{ - return _mm256_blend_ps(a,pset1(b),(1<<7)); -} - -template<> EIGEN_STRONG_INLINE Packet4d pinsertlast(const Packet4d& a, double b) -{ - return _mm256_blend_pd(a,pset1(b),(1<<3)); -} - -} // end namespace internal - -} // end namespace Eigen - -#endif // EIGEN_PACKET_MATH_AVX_H + + EIGEN_DEVICE_FUNC inline void ptranspose(PacketBlock &kernel) + { + __m256d T0 = _mm256_shuffle_pd(kernel.packet[0], kernel.packet[1], 15); + __m256d T1 = _mm256_shuffle_pd(kernel.packet[0], kernel.packet[1], 0); + __m256d T2 = _mm256_shuffle_pd(kernel.packet[2], kernel.packet[3], 15); + __m256d T3 = _mm256_shuffle_pd(kernel.packet[2], kernel.packet[3], 0); + + kernel.packet[1] = _mm256_permute2f128_pd(T0, T2, 32); + kernel.packet[3] = _mm256_permute2f128_pd(T0, T2, 49); + kernel.packet[0] = _mm256_permute2f128_pd(T1, T3, 32); + kernel.packet[2] = _mm256_permute2f128_pd(T1, T3, 49); + } + + template<> + EIGEN_STRONG_INLINE Packet8f pblend(const Selector<8> &ifPacket, + const Packet8f &thenPacket, + const Packet8f &elsePacket) + { + const __m256 zero = _mm256_setzero_ps(); + const __m256 select = _mm256_set_ps(ifPacket.select[7], + ifPacket.select[6], + ifPacket.select[5], + ifPacket.select[4], + ifPacket.select[3], + ifPacket.select[2], + ifPacket.select[1], + ifPacket.select[0]); + __m256 false_mask = _mm256_cmp_ps(select, zero, _CMP_EQ_UQ); + return _mm256_blendv_ps(thenPacket, elsePacket, false_mask); + } + template<> + EIGEN_STRONG_INLINE Packet4d pblend(const Selector<4> &ifPacket, + const Packet4d &thenPacket, + const Packet4d &elsePacket) + { + const __m256d zero = _mm256_setzero_pd(); + const __m256d select = + _mm256_set_pd(ifPacket.select[3], ifPacket.select[2], ifPacket.select[1], ifPacket.select[0]); + __m256d false_mask = _mm256_cmp_pd(select, zero, _CMP_EQ_UQ); + return _mm256_blendv_pd(thenPacket, elsePacket, false_mask); + } + + template<> EIGEN_STRONG_INLINE Packet8f pinsertfirst(const Packet8f &a, float b) + { + return _mm256_blend_ps(a, pset1(b), 1); + } + + template<> EIGEN_STRONG_INLINE Packet4d pinsertfirst(const Packet4d &a, double b) + { + return _mm256_blend_pd(a, pset1(b), 1); + } + + template<> EIGEN_STRONG_INLINE Packet8f pinsertlast(const Packet8f &a, float b) + { + return _mm256_blend_ps(a, pset1(b), (1 << 7)); + } + + template<> EIGEN_STRONG_INLINE Packet4d pinsertlast(const Packet4d &a, double b) + { + return _mm256_blend_pd(a, pset1(b), (1 << 3)); + } + +}// end namespace internal + +}// end namespace Eigen + +#endif// EIGEN_PACKET_MATH_AVX_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/arch/AVX/TypeCasting.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/arch/AVX/TypeCasting.h index 83bfdc60..9f8c111a 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/arch/AVX/TypeCasting.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/arch/AVX/TypeCasting.h @@ -14,38 +14,25 @@ namespace Eigen { namespace internal { -// For now we use SSE to handle integers, so we can't use AVX instructions to cast -// from int to float -template <> -struct type_casting_traits { - enum { - VectorizedCast = 0, - SrcCoeffRatio = 1, - TgtCoeffRatio = 1 + // For now we use SSE to handle integers, so we can't use AVX instructions to cast + // from int to float + template<> struct type_casting_traits + { + enum { VectorizedCast = 0, SrcCoeffRatio = 1, TgtCoeffRatio = 1 }; }; -}; - -template <> -struct type_casting_traits { - enum { - VectorizedCast = 0, - SrcCoeffRatio = 1, - TgtCoeffRatio = 1 - }; -}; + template<> struct type_casting_traits + { + enum { VectorizedCast = 0, SrcCoeffRatio = 1, TgtCoeffRatio = 1 }; + }; -template<> EIGEN_STRONG_INLINE Packet8i pcast(const Packet8f& a) { - return _mm256_cvtps_epi32(a); -} + template<> EIGEN_STRONG_INLINE Packet8i pcast(const Packet8f &a) { return _mm256_cvtps_epi32(a); } -template<> EIGEN_STRONG_INLINE Packet8f pcast(const Packet8i& a) { - return _mm256_cvtepi32_ps(a); -} + template<> EIGEN_STRONG_INLINE Packet8f pcast(const Packet8i &a) { return _mm256_cvtepi32_ps(a); } -} // end namespace internal +}// end namespace internal -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_TYPE_CASTING_AVX_H +#endif// EIGEN_TYPE_CASTING_AVX_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/arch/AVX512/MathFunctions.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/arch/AVX512/MathFunctions.h index 9c1717f7..bcfca95a 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/arch/AVX512/MathFunctions.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/arch/AVX512/MathFunctions.h @@ -17,14 +17,11 @@ namespace internal { // Disable the code for older versions of gcc that don't support many of the required avx512 instrinsics. #if EIGEN_GNUC_AT_LEAST(5, 3) -#define _EIGEN_DECLARE_CONST_Packet16f(NAME, X) \ - const Packet16f p16f_##NAME = pset1(X) +#define _EIGEN_DECLARE_CONST_Packet16f(NAME, X) const Packet16f p16f_##NAME = pset1(X) -#define _EIGEN_DECLARE_CONST_Packet16f_FROM_INT(NAME, X) \ - const Packet16f p16f_##NAME = (__m512)pset1(X) +#define _EIGEN_DECLARE_CONST_Packet16f_FROM_INT(NAME, X) const Packet16f p16f_##NAME = (__m512)pset1(X) -#define _EIGEN_DECLARE_CONST_Packet8d(NAME, X) \ - const Packet8d p8d_##NAME = pset1(X) +#define _EIGEN_DECLARE_CONST_Packet8d(NAME, X) const Packet8d p8d_##NAME = pset1(X) #define _EIGEN_DECLARE_CONST_Packet8d_FROM_INT64(NAME, X) \ const Packet8d p8d_##NAME = _mm512_castsi512_pd(_mm512_set1_epi64(X)) @@ -34,150 +31,146 @@ namespace internal { // and m is in the range [sqrt(1/2),sqrt(2)). In this range, the logarithm can // be easily approximated by a polynomial centered on m=1 for stability. #if defined(EIGEN_VECTORIZE_AVX512DQ) -template <> -EIGEN_DEFINE_FUNCTION_ALLOWING_MULTIPLE_DEFINITIONS EIGEN_UNUSED Packet16f -plog(const Packet16f& _x) { - Packet16f x = _x; - _EIGEN_DECLARE_CONST_Packet16f(1, 1.0f); - _EIGEN_DECLARE_CONST_Packet16f(half, 0.5f); - _EIGEN_DECLARE_CONST_Packet16f(126f, 126.0f); - - _EIGEN_DECLARE_CONST_Packet16f_FROM_INT(inv_mant_mask, ~0x7f800000); - - // The smallest non denormalized float number. - _EIGEN_DECLARE_CONST_Packet16f_FROM_INT(min_norm_pos, 0x00800000); - _EIGEN_DECLARE_CONST_Packet16f_FROM_INT(minus_inf, 0xff800000); - _EIGEN_DECLARE_CONST_Packet16f_FROM_INT(nan, 0x7fc00000); - - // Polynomial coefficients. - _EIGEN_DECLARE_CONST_Packet16f(cephes_SQRTHF, 0.707106781186547524f); - _EIGEN_DECLARE_CONST_Packet16f(cephes_log_p0, 7.0376836292E-2f); - _EIGEN_DECLARE_CONST_Packet16f(cephes_log_p1, -1.1514610310E-1f); - _EIGEN_DECLARE_CONST_Packet16f(cephes_log_p2, 1.1676998740E-1f); - _EIGEN_DECLARE_CONST_Packet16f(cephes_log_p3, -1.2420140846E-1f); - _EIGEN_DECLARE_CONST_Packet16f(cephes_log_p4, +1.4249322787E-1f); - _EIGEN_DECLARE_CONST_Packet16f(cephes_log_p5, -1.6668057665E-1f); - _EIGEN_DECLARE_CONST_Packet16f(cephes_log_p6, +2.0000714765E-1f); - _EIGEN_DECLARE_CONST_Packet16f(cephes_log_p7, -2.4999993993E-1f); - _EIGEN_DECLARE_CONST_Packet16f(cephes_log_p8, +3.3333331174E-1f); - _EIGEN_DECLARE_CONST_Packet16f(cephes_log_q1, -2.12194440e-4f); - _EIGEN_DECLARE_CONST_Packet16f(cephes_log_q2, 0.693359375f); - - // invalid_mask is set to true when x is NaN - __mmask16 invalid_mask = - _mm512_cmp_ps_mask(x, _mm512_setzero_ps(), _CMP_NGE_UQ); - __mmask16 iszero_mask = - _mm512_cmp_ps_mask(x, _mm512_setzero_ps(), _CMP_EQ_UQ); - - // Truncate input values to the minimum positive normal. - x = pmax(x, p16f_min_norm_pos); - - // Extract the shifted exponents. - Packet16f emm0 = _mm512_cvtepi32_ps(_mm512_srli_epi32((__m512i)x, 23)); - Packet16f e = _mm512_sub_ps(emm0, p16f_126f); - - // Set the exponents to -1, i.e. x are in the range [0.5,1). - x = _mm512_and_ps(x, p16f_inv_mant_mask); - x = _mm512_or_ps(x, p16f_half); - - // part2: Shift the inputs from the range [0.5,1) to [sqrt(1/2),sqrt(2)) - // and shift by -1. The values are then centered around 0, which improves - // the stability of the polynomial evaluation. - // if( x < SQRTHF ) { - // e -= 1; - // x = x + x - 1.0; - // } else { x = x - 1.0; } - __mmask16 mask = _mm512_cmp_ps_mask(x, p16f_cephes_SQRTHF, _CMP_LT_OQ); - Packet16f tmp = _mm512_mask_blend_ps(mask, _mm512_setzero_ps(), x); - x = psub(x, p16f_1); - e = psub(e, _mm512_mask_blend_ps(mask, _mm512_setzero_ps(), p16f_1)); - x = padd(x, tmp); - - Packet16f x2 = pmul(x, x); - Packet16f x3 = pmul(x2, x); - - // Evaluate the polynomial approximant of degree 8 in three parts, probably - // to improve instruction-level parallelism. - Packet16f y, y1, y2; - y = pmadd(p16f_cephes_log_p0, x, p16f_cephes_log_p1); - y1 = pmadd(p16f_cephes_log_p3, x, p16f_cephes_log_p4); - y2 = pmadd(p16f_cephes_log_p6, x, p16f_cephes_log_p7); - y = pmadd(y, x, p16f_cephes_log_p2); - y1 = pmadd(y1, x, p16f_cephes_log_p5); - y2 = pmadd(y2, x, p16f_cephes_log_p8); - y = pmadd(y, x3, y1); - y = pmadd(y, x3, y2); - y = pmul(y, x3); - - // Add the logarithm of the exponent back to the result of the interpolation. - y1 = pmul(e, p16f_cephes_log_q1); - tmp = pmul(x2, p16f_half); - y = padd(y, y1); - x = psub(x, tmp); - y2 = pmul(e, p16f_cephes_log_q2); - x = padd(x, y); - x = padd(x, y2); - - // Filter out invalid inputs, i.e. negative arg will be NAN, 0 will be -INF. - return _mm512_mask_blend_ps(iszero_mask, - _mm512_mask_blend_ps(invalid_mask, x, p16f_nan), - p16f_minus_inf); -} + template<> + EIGEN_DEFINE_FUNCTION_ALLOWING_MULTIPLE_DEFINITIONS EIGEN_UNUSED Packet16f plog(const Packet16f &_x) + { + Packet16f x = _x; + _EIGEN_DECLARE_CONST_Packet16f(1, 1.0f); + _EIGEN_DECLARE_CONST_Packet16f(half, 0.5f); + _EIGEN_DECLARE_CONST_Packet16f(126f, 126.0f); + + _EIGEN_DECLARE_CONST_Packet16f_FROM_INT(inv_mant_mask, ~0x7f800000); + + // The smallest non denormalized float number. + _EIGEN_DECLARE_CONST_Packet16f_FROM_INT(min_norm_pos, 0x00800000); + _EIGEN_DECLARE_CONST_Packet16f_FROM_INT(minus_inf, 0xff800000); + _EIGEN_DECLARE_CONST_Packet16f_FROM_INT(nan, 0x7fc00000); + + // Polynomial coefficients. + _EIGEN_DECLARE_CONST_Packet16f(cephes_SQRTHF, 0.707106781186547524f); + _EIGEN_DECLARE_CONST_Packet16f(cephes_log_p0, 7.0376836292E-2f); + _EIGEN_DECLARE_CONST_Packet16f(cephes_log_p1, -1.1514610310E-1f); + _EIGEN_DECLARE_CONST_Packet16f(cephes_log_p2, 1.1676998740E-1f); + _EIGEN_DECLARE_CONST_Packet16f(cephes_log_p3, -1.2420140846E-1f); + _EIGEN_DECLARE_CONST_Packet16f(cephes_log_p4, +1.4249322787E-1f); + _EIGEN_DECLARE_CONST_Packet16f(cephes_log_p5, -1.6668057665E-1f); + _EIGEN_DECLARE_CONST_Packet16f(cephes_log_p6, +2.0000714765E-1f); + _EIGEN_DECLARE_CONST_Packet16f(cephes_log_p7, -2.4999993993E-1f); + _EIGEN_DECLARE_CONST_Packet16f(cephes_log_p8, +3.3333331174E-1f); + _EIGEN_DECLARE_CONST_Packet16f(cephes_log_q1, -2.12194440e-4f); + _EIGEN_DECLARE_CONST_Packet16f(cephes_log_q2, 0.693359375f); + + // invalid_mask is set to true when x is NaN + __mmask16 invalid_mask = _mm512_cmp_ps_mask(x, _mm512_setzero_ps(), _CMP_NGE_UQ); + __mmask16 iszero_mask = _mm512_cmp_ps_mask(x, _mm512_setzero_ps(), _CMP_EQ_UQ); + + // Truncate input values to the minimum positive normal. + x = pmax(x, p16f_min_norm_pos); + + // Extract the shifted exponents. + Packet16f emm0 = _mm512_cvtepi32_ps(_mm512_srli_epi32((__m512i)x, 23)); + Packet16f e = _mm512_sub_ps(emm0, p16f_126f); + + // Set the exponents to -1, i.e. x are in the range [0.5,1). + x = _mm512_and_ps(x, p16f_inv_mant_mask); + x = _mm512_or_ps(x, p16f_half); + + // part2: Shift the inputs from the range [0.5,1) to [sqrt(1/2),sqrt(2)) + // and shift by -1. The values are then centered around 0, which improves + // the stability of the polynomial evaluation. + // if( x < SQRTHF ) { + // e -= 1; + // x = x + x - 1.0; + // } else { x = x - 1.0; } + __mmask16 mask = _mm512_cmp_ps_mask(x, p16f_cephes_SQRTHF, _CMP_LT_OQ); + Packet16f tmp = _mm512_mask_blend_ps(mask, _mm512_setzero_ps(), x); + x = psub(x, p16f_1); + e = psub(e, _mm512_mask_blend_ps(mask, _mm512_setzero_ps(), p16f_1)); + x = padd(x, tmp); + + Packet16f x2 = pmul(x, x); + Packet16f x3 = pmul(x2, x); + + // Evaluate the polynomial approximant of degree 8 in three parts, probably + // to improve instruction-level parallelism. + Packet16f y, y1, y2; + y = pmadd(p16f_cephes_log_p0, x, p16f_cephes_log_p1); + y1 = pmadd(p16f_cephes_log_p3, x, p16f_cephes_log_p4); + y2 = pmadd(p16f_cephes_log_p6, x, p16f_cephes_log_p7); + y = pmadd(y, x, p16f_cephes_log_p2); + y1 = pmadd(y1, x, p16f_cephes_log_p5); + y2 = pmadd(y2, x, p16f_cephes_log_p8); + y = pmadd(y, x3, y1); + y = pmadd(y, x3, y2); + y = pmul(y, x3); + + // Add the logarithm of the exponent back to the result of the interpolation. + y1 = pmul(e, p16f_cephes_log_q1); + tmp = pmul(x2, p16f_half); + y = padd(y, y1); + x = psub(x, tmp); + y2 = pmul(e, p16f_cephes_log_q2); + x = padd(x, y); + x = padd(x, y2); + + // Filter out invalid inputs, i.e. negative arg will be NAN, 0 will be -INF. + return _mm512_mask_blend_ps(iszero_mask, _mm512_mask_blend_ps(invalid_mask, x, p16f_nan), p16f_minus_inf); + } #endif -// Exponential function. Works by writing "x = m*log(2) + r" where -// "m = floor(x/log(2)+1/2)" and "r" is the remainder. The result is then -// "exp(x) = 2^m*exp(r)" where exp(r) is in the range [-1,1). -template <> -EIGEN_DEFINE_FUNCTION_ALLOWING_MULTIPLE_DEFINITIONS EIGEN_UNUSED Packet16f -pexp(const Packet16f& _x) { - _EIGEN_DECLARE_CONST_Packet16f(1, 1.0f); - _EIGEN_DECLARE_CONST_Packet16f(half, 0.5f); - _EIGEN_DECLARE_CONST_Packet16f(127, 127.0f); - - _EIGEN_DECLARE_CONST_Packet16f(exp_hi, 88.3762626647950f); - _EIGEN_DECLARE_CONST_Packet16f(exp_lo, -88.3762626647949f); - - _EIGEN_DECLARE_CONST_Packet16f(cephes_LOG2EF, 1.44269504088896341f); - - _EIGEN_DECLARE_CONST_Packet16f(cephes_exp_p0, 1.9875691500E-4f); - _EIGEN_DECLARE_CONST_Packet16f(cephes_exp_p1, 1.3981999507E-3f); - _EIGEN_DECLARE_CONST_Packet16f(cephes_exp_p2, 8.3334519073E-3f); - _EIGEN_DECLARE_CONST_Packet16f(cephes_exp_p3, 4.1665795894E-2f); - _EIGEN_DECLARE_CONST_Packet16f(cephes_exp_p4, 1.6666665459E-1f); - _EIGEN_DECLARE_CONST_Packet16f(cephes_exp_p5, 5.0000001201E-1f); - - // Clamp x. - Packet16f x = pmax(pmin(_x, p16f_exp_hi), p16f_exp_lo); - - // Express exp(x) as exp(m*ln(2) + r), start by extracting - // m = floor(x/ln(2) + 0.5). - Packet16f m = _mm512_floor_ps(pmadd(x, p16f_cephes_LOG2EF, p16f_half)); - - // Get r = x - m*ln(2). Note that we can do this without losing more than one - // ulp precision due to the FMA instruction. - _EIGEN_DECLARE_CONST_Packet16f(nln2, -0.6931471805599453f); - Packet16f r = _mm512_fmadd_ps(m, p16f_nln2, x); - Packet16f r2 = pmul(r, r); - - // TODO(gonnet): Split into odd/even polynomials and try to exploit - // instruction-level parallelism. - Packet16f y = p16f_cephes_exp_p0; - y = pmadd(y, r, p16f_cephes_exp_p1); - y = pmadd(y, r, p16f_cephes_exp_p2); - y = pmadd(y, r, p16f_cephes_exp_p3); - y = pmadd(y, r, p16f_cephes_exp_p4); - y = pmadd(y, r, p16f_cephes_exp_p5); - y = pmadd(y, r2, r); - y = padd(y, p16f_1); - - // Build emm0 = 2^m. - Packet16i emm0 = _mm512_cvttps_epi32(padd(m, p16f_127)); - emm0 = _mm512_slli_epi32(emm0, 23); - - // Return 2^m * exp(r). - return pmax(pmul(y, _mm512_castsi512_ps(emm0)), _x); -} + // Exponential function. Works by writing "x = m*log(2) + r" where + // "m = floor(x/log(2)+1/2)" and "r" is the remainder. The result is then + // "exp(x) = 2^m*exp(r)" where exp(r) is in the range [-1,1). + template<> + EIGEN_DEFINE_FUNCTION_ALLOWING_MULTIPLE_DEFINITIONS EIGEN_UNUSED Packet16f pexp(const Packet16f &_x) + { + _EIGEN_DECLARE_CONST_Packet16f(1, 1.0f); + _EIGEN_DECLARE_CONST_Packet16f(half, 0.5f); + _EIGEN_DECLARE_CONST_Packet16f(127, 127.0f); + + _EIGEN_DECLARE_CONST_Packet16f(exp_hi, 88.3762626647950f); + _EIGEN_DECLARE_CONST_Packet16f(exp_lo, -88.3762626647949f); + + _EIGEN_DECLARE_CONST_Packet16f(cephes_LOG2EF, 1.44269504088896341f); + + _EIGEN_DECLARE_CONST_Packet16f(cephes_exp_p0, 1.9875691500E-4f); + _EIGEN_DECLARE_CONST_Packet16f(cephes_exp_p1, 1.3981999507E-3f); + _EIGEN_DECLARE_CONST_Packet16f(cephes_exp_p2, 8.3334519073E-3f); + _EIGEN_DECLARE_CONST_Packet16f(cephes_exp_p3, 4.1665795894E-2f); + _EIGEN_DECLARE_CONST_Packet16f(cephes_exp_p4, 1.6666665459E-1f); + _EIGEN_DECLARE_CONST_Packet16f(cephes_exp_p5, 5.0000001201E-1f); + + // Clamp x. + Packet16f x = pmax(pmin(_x, p16f_exp_hi), p16f_exp_lo); + + // Express exp(x) as exp(m*ln(2) + r), start by extracting + // m = floor(x/ln(2) + 0.5). + Packet16f m = _mm512_floor_ps(pmadd(x, p16f_cephes_LOG2EF, p16f_half)); + + // Get r = x - m*ln(2). Note that we can do this without losing more than one + // ulp precision due to the FMA instruction. + _EIGEN_DECLARE_CONST_Packet16f(nln2, -0.6931471805599453f); + Packet16f r = _mm512_fmadd_ps(m, p16f_nln2, x); + Packet16f r2 = pmul(r, r); + + // TODO(gonnet): Split into odd/even polynomials and try to exploit + // instruction-level parallelism. + Packet16f y = p16f_cephes_exp_p0; + y = pmadd(y, r, p16f_cephes_exp_p1); + y = pmadd(y, r, p16f_cephes_exp_p2); + y = pmadd(y, r, p16f_cephes_exp_p3); + y = pmadd(y, r, p16f_cephes_exp_p4); + y = pmadd(y, r, p16f_cephes_exp_p5); + y = pmadd(y, r2, r); + y = padd(y, p16f_1); + + // Build emm0 = 2^m. + Packet16i emm0 = _mm512_cvttps_epi32(padd(m, p16f_127)); + emm0 = _mm512_slli_epi32(emm0, 23); + + // Return 2^m * exp(r). + return pmax(pmul(y, _mm512_castsi512_ps(emm0)), _x); + } /*template <> EIGEN_DEFINE_FUNCTION_ALLOWING_MULTIPLE_DEFINITIONS EIGEN_UNUSED Packet8d @@ -255,61 +248,55 @@ pexp(const Packet8d& _x) { // also the fact that it can be inlined and pipelined with other computations, // further reducing its effective latency. #if EIGEN_FAST_MATH -template <> -EIGEN_DEFINE_FUNCTION_ALLOWING_MULTIPLE_DEFINITIONS EIGEN_UNUSED Packet16f -psqrt(const Packet16f& _x) { - _EIGEN_DECLARE_CONST_Packet16f(one_point_five, 1.5f); - _EIGEN_DECLARE_CONST_Packet16f(minus_half, -0.5f); - _EIGEN_DECLARE_CONST_Packet16f_FROM_INT(flt_min, 0x00800000); - - Packet16f neg_half = pmul(_x, p16f_minus_half); - - // select only the inverse sqrt of positive normal inputs (denormals are - // flushed to zero and cause infs as well). - __mmask16 non_zero_mask = _mm512_cmp_ps_mask(_x, p16f_flt_min, _CMP_GE_OQ); - Packet16f x = _mm512_mask_blend_ps(non_zero_mask, _mm512_setzero_ps(), _mm512_rsqrt14_ps(_x)); - - // Do a single step of Newton's iteration. - x = pmul(x, pmadd(neg_half, pmul(x, x), p16f_one_point_five)); - - // Multiply the original _x by it's reciprocal square root to extract the - // square root. - return pmul(_x, x); -} - -template <> -EIGEN_DEFINE_FUNCTION_ALLOWING_MULTIPLE_DEFINITIONS EIGEN_UNUSED Packet8d -psqrt(const Packet8d& _x) { - _EIGEN_DECLARE_CONST_Packet8d(one_point_five, 1.5); - _EIGEN_DECLARE_CONST_Packet8d(minus_half, -0.5); - _EIGEN_DECLARE_CONST_Packet8d_FROM_INT64(dbl_min, 0x0010000000000000LL); - - Packet8d neg_half = pmul(_x, p8d_minus_half); - - // select only the inverse sqrt of positive normal inputs (denormals are - // flushed to zero and cause infs as well). - __mmask8 non_zero_mask = _mm512_cmp_pd_mask(_x, p8d_dbl_min, _CMP_GE_OQ); - Packet8d x = _mm512_mask_blend_pd(non_zero_mask, _mm512_setzero_pd(), _mm512_rsqrt14_pd(_x)); - - // Do a first step of Newton's iteration. - x = pmul(x, pmadd(neg_half, pmul(x, x), p8d_one_point_five)); - - // Do a second step of Newton's iteration. - x = pmul(x, pmadd(neg_half, pmul(x, x), p8d_one_point_five)); - - // Multiply the original _x by it's reciprocal square root to extract the - // square root. - return pmul(_x, x); -} + template<> + EIGEN_DEFINE_FUNCTION_ALLOWING_MULTIPLE_DEFINITIONS EIGEN_UNUSED Packet16f psqrt(const Packet16f &_x) + { + _EIGEN_DECLARE_CONST_Packet16f(one_point_five, 1.5f); + _EIGEN_DECLARE_CONST_Packet16f(minus_half, -0.5f); + _EIGEN_DECLARE_CONST_Packet16f_FROM_INT(flt_min, 0x00800000); + + Packet16f neg_half = pmul(_x, p16f_minus_half); + + // select only the inverse sqrt of positive normal inputs (denormals are + // flushed to zero and cause infs as well). + __mmask16 non_zero_mask = _mm512_cmp_ps_mask(_x, p16f_flt_min, _CMP_GE_OQ); + Packet16f x = _mm512_mask_blend_ps(non_zero_mask, _mm512_setzero_ps(), _mm512_rsqrt14_ps(_x)); + + // Do a single step of Newton's iteration. + x = pmul(x, pmadd(neg_half, pmul(x, x), p16f_one_point_five)); + + // Multiply the original _x by it's reciprocal square root to extract the + // square root. + return pmul(_x, x); + } + + template<> + EIGEN_DEFINE_FUNCTION_ALLOWING_MULTIPLE_DEFINITIONS EIGEN_UNUSED Packet8d psqrt(const Packet8d &_x) + { + _EIGEN_DECLARE_CONST_Packet8d(one_point_five, 1.5); + _EIGEN_DECLARE_CONST_Packet8d(minus_half, -0.5); + _EIGEN_DECLARE_CONST_Packet8d_FROM_INT64(dbl_min, 0x0010000000000000LL); + + Packet8d neg_half = pmul(_x, p8d_minus_half); + + // select only the inverse sqrt of positive normal inputs (denormals are + // flushed to zero and cause infs as well). + __mmask8 non_zero_mask = _mm512_cmp_pd_mask(_x, p8d_dbl_min, _CMP_GE_OQ); + Packet8d x = _mm512_mask_blend_pd(non_zero_mask, _mm512_setzero_pd(), _mm512_rsqrt14_pd(_x)); + + // Do a first step of Newton's iteration. + x = pmul(x, pmadd(neg_half, pmul(x, x), p8d_one_point_five)); + + // Do a second step of Newton's iteration. + x = pmul(x, pmadd(neg_half, pmul(x, x), p8d_one_point_five)); + + // Multiply the original _x by it's reciprocal square root to extract the + // square root. + return pmul(_x, x); + } #else -template <> -EIGEN_STRONG_INLINE Packet16f psqrt(const Packet16f& x) { - return _mm512_sqrt_ps(x); -} -template <> -EIGEN_STRONG_INLINE Packet8d psqrt(const Packet8d& x) { - return _mm512_sqrt_pd(x); -} + template<> EIGEN_STRONG_INLINE Packet16f psqrt(const Packet16f &x) { return _mm512_sqrt_ps(x); } + template<> EIGEN_STRONG_INLINE Packet8d psqrt(const Packet8d &x) { return _mm512_sqrt_pd(x); } #endif // Functions for rsqrt. @@ -318,74 +305,71 @@ EIGEN_STRONG_INLINE Packet8d psqrt(const Packet8d& x) { // iterative version for doubles since there is no instruction for diretly // computing the reciprocal square root in AVX-512. #ifdef EIGEN_FAST_MATH -template <> -EIGEN_DEFINE_FUNCTION_ALLOWING_MULTIPLE_DEFINITIONS EIGEN_UNUSED Packet16f -prsqrt(const Packet16f& _x) { - _EIGEN_DECLARE_CONST_Packet16f_FROM_INT(inf, 0x7f800000); - _EIGEN_DECLARE_CONST_Packet16f_FROM_INT(nan, 0x7fc00000); - _EIGEN_DECLARE_CONST_Packet16f(one_point_five, 1.5f); - _EIGEN_DECLARE_CONST_Packet16f(minus_half, -0.5f); - _EIGEN_DECLARE_CONST_Packet16f_FROM_INT(flt_min, 0x00800000); - - Packet16f neg_half = pmul(_x, p16f_minus_half); - - // select only the inverse sqrt of positive normal inputs (denormals are - // flushed to zero and cause infs as well). - __mmask16 le_zero_mask = _mm512_cmp_ps_mask(_x, p16f_flt_min, _CMP_LT_OQ); - Packet16f x = _mm512_mask_blend_ps(le_zero_mask, _mm512_rsqrt14_ps(_x), _mm512_setzero_ps()); - - // Fill in NaNs and Infs for the negative/zero entries. - __mmask16 neg_mask = _mm512_cmp_ps_mask(_x, _mm512_setzero_ps(), _CMP_LT_OQ); - Packet16f infs_and_nans = _mm512_mask_blend_ps( - neg_mask, _mm512_mask_blend_ps(le_zero_mask, _mm512_setzero_ps(), p16f_inf), p16f_nan); - - // Do a single step of Newton's iteration. - x = pmul(x, pmadd(neg_half, pmul(x, x), p16f_one_point_five)); - - // Insert NaNs and Infs in all the right places. - return _mm512_mask_blend_ps(le_zero_mask, x, infs_and_nans); -} - -template <> -EIGEN_DEFINE_FUNCTION_ALLOWING_MULTIPLE_DEFINITIONS EIGEN_UNUSED Packet8d -prsqrt(const Packet8d& _x) { - _EIGEN_DECLARE_CONST_Packet8d_FROM_INT64(inf, 0x7ff0000000000000LL); - _EIGEN_DECLARE_CONST_Packet8d_FROM_INT64(nan, 0x7ff1000000000000LL); - _EIGEN_DECLARE_CONST_Packet8d(one_point_five, 1.5); - _EIGEN_DECLARE_CONST_Packet8d(minus_half, -0.5); - _EIGEN_DECLARE_CONST_Packet8d_FROM_INT64(dbl_min, 0x0010000000000000LL); - - Packet8d neg_half = pmul(_x, p8d_minus_half); - - // select only the inverse sqrt of positive normal inputs (denormals are - // flushed to zero and cause infs as well). - __mmask8 le_zero_mask = _mm512_cmp_pd_mask(_x, p8d_dbl_min, _CMP_LT_OQ); - Packet8d x = _mm512_mask_blend_pd(le_zero_mask, _mm512_rsqrt14_pd(_x), _mm512_setzero_pd()); - - // Fill in NaNs and Infs for the negative/zero entries. - __mmask8 neg_mask = _mm512_cmp_pd_mask(_x, _mm512_setzero_pd(), _CMP_LT_OQ); - Packet8d infs_and_nans = _mm512_mask_blend_pd( - neg_mask, _mm512_mask_blend_pd(le_zero_mask, _mm512_setzero_pd(), p8d_inf), p8d_nan); - - // Do a first step of Newton's iteration. - x = pmul(x, pmadd(neg_half, pmul(x, x), p8d_one_point_five)); - - // Do a second step of Newton's iteration. - x = pmul(x, pmadd(neg_half, pmul(x, x), p8d_one_point_five)); - - // Insert NaNs and Infs in all the right places. - return _mm512_mask_blend_pd(le_zero_mask, x, infs_and_nans); -} + template<> + EIGEN_DEFINE_FUNCTION_ALLOWING_MULTIPLE_DEFINITIONS EIGEN_UNUSED Packet16f prsqrt(const Packet16f &_x) + { + _EIGEN_DECLARE_CONST_Packet16f_FROM_INT(inf, 0x7f800000); + _EIGEN_DECLARE_CONST_Packet16f_FROM_INT(nan, 0x7fc00000); + _EIGEN_DECLARE_CONST_Packet16f(one_point_five, 1.5f); + _EIGEN_DECLARE_CONST_Packet16f(minus_half, -0.5f); + _EIGEN_DECLARE_CONST_Packet16f_FROM_INT(flt_min, 0x00800000); + + Packet16f neg_half = pmul(_x, p16f_minus_half); + + // select only the inverse sqrt of positive normal inputs (denormals are + // flushed to zero and cause infs as well). + __mmask16 le_zero_mask = _mm512_cmp_ps_mask(_x, p16f_flt_min, _CMP_LT_OQ); + Packet16f x = _mm512_mask_blend_ps(le_zero_mask, _mm512_rsqrt14_ps(_x), _mm512_setzero_ps()); + + // Fill in NaNs and Infs for the negative/zero entries. + __mmask16 neg_mask = _mm512_cmp_ps_mask(_x, _mm512_setzero_ps(), _CMP_LT_OQ); + Packet16f infs_and_nans = + _mm512_mask_blend_ps(neg_mask, _mm512_mask_blend_ps(le_zero_mask, _mm512_setzero_ps(), p16f_inf), p16f_nan); + + // Do a single step of Newton's iteration. + x = pmul(x, pmadd(neg_half, pmul(x, x), p16f_one_point_five)); + + // Insert NaNs and Infs in all the right places. + return _mm512_mask_blend_ps(le_zero_mask, x, infs_and_nans); + } + + template<> + EIGEN_DEFINE_FUNCTION_ALLOWING_MULTIPLE_DEFINITIONS EIGEN_UNUSED Packet8d prsqrt(const Packet8d &_x) + { + _EIGEN_DECLARE_CONST_Packet8d_FROM_INT64(inf, 0x7ff0000000000000LL); + _EIGEN_DECLARE_CONST_Packet8d_FROM_INT64(nan, 0x7ff1000000000000LL); + _EIGEN_DECLARE_CONST_Packet8d(one_point_five, 1.5); + _EIGEN_DECLARE_CONST_Packet8d(minus_half, -0.5); + _EIGEN_DECLARE_CONST_Packet8d_FROM_INT64(dbl_min, 0x0010000000000000LL); + + Packet8d neg_half = pmul(_x, p8d_minus_half); + + // select only the inverse sqrt of positive normal inputs (denormals are + // flushed to zero and cause infs as well). + __mmask8 le_zero_mask = _mm512_cmp_pd_mask(_x, p8d_dbl_min, _CMP_LT_OQ); + Packet8d x = _mm512_mask_blend_pd(le_zero_mask, _mm512_rsqrt14_pd(_x), _mm512_setzero_pd()); + + // Fill in NaNs and Infs for the negative/zero entries. + __mmask8 neg_mask = _mm512_cmp_pd_mask(_x, _mm512_setzero_pd(), _CMP_LT_OQ); + Packet8d infs_and_nans = + _mm512_mask_blend_pd(neg_mask, _mm512_mask_blend_pd(le_zero_mask, _mm512_setzero_pd(), p8d_inf), p8d_nan); + + // Do a first step of Newton's iteration. + x = pmul(x, pmadd(neg_half, pmul(x, x), p8d_one_point_five)); + + // Do a second step of Newton's iteration. + x = pmul(x, pmadd(neg_half, pmul(x, x), p8d_one_point_five)); + + // Insert NaNs and Infs in all the right places. + return _mm512_mask_blend_pd(le_zero_mask, x, infs_and_nans); + } #elif defined(EIGEN_VECTORIZE_AVX512ER) -template <> -EIGEN_STRONG_INLINE Packet16f prsqrt(const Packet16f& x) { - return _mm512_rsqrt28_ps(x); -} + template<> EIGEN_STRONG_INLINE Packet16f prsqrt(const Packet16f &x) { return _mm512_rsqrt28_ps(x); } #endif #endif -} // end namespace internal +}// end namespace internal -} // end namespace Eigen +}// end namespace Eigen -#endif // THIRD_PARTY_EIGEN3_EIGEN_SRC_CORE_ARCH_AVX512_MATHFUNCTIONS_H_ +#endif// THIRD_PARTY_EIGEN3_EIGEN_SRC_CORE_ARCH_AVX512_MATHFUNCTIONS_H_ diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/arch/AVX512/PacketMath.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/arch/AVX512/PacketMath.h index 5adddc7a..e05b58ba 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/arch/AVX512/PacketMath.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/arch/AVX512/PacketMath.h @@ -19,7 +19,7 @@ namespace internal { #endif #ifndef EIGEN_ARCH_DEFAULT_NUMBER_OF_REGISTERS -#define EIGEN_ARCH_DEFAULT_NUMBER_OF_REGISTERS (2*sizeof(void*)) +#define EIGEN_ARCH_DEFAULT_NUMBER_OF_REGISTERS (2 * sizeof(void *)) #endif #ifdef __FMA__ @@ -28,648 +28,589 @@ namespace internal { #endif #endif -typedef __m512 Packet16f; -typedef __m512i Packet16i; -typedef __m512d Packet8d; - -template <> -struct is_arithmetic<__m512> { - enum { value = true }; -}; -template <> -struct is_arithmetic<__m512i> { - enum { value = true }; -}; -template <> -struct is_arithmetic<__m512d> { - enum { value = true }; -}; - -template<> struct packet_traits : default_packet_traits -{ - typedef Packet16f type; - typedef Packet8f half; - enum { - Vectorizable = 1, - AlignedOnScalar = 1, - size = 16, - HasHalfPacket = 1, + typedef __m512 Packet16f; + typedef __m512i Packet16i; + typedef __m512d Packet8d; + + template<> struct is_arithmetic<__m512> + { + enum { value = true }; + }; + template<> struct is_arithmetic<__m512i> + { + enum { value = true }; + }; + template<> struct is_arithmetic<__m512d> + { + enum { value = true }; + }; + + template<> struct packet_traits : default_packet_traits + { + typedef Packet16f type; + typedef Packet8f half; + enum { + Vectorizable = 1, + AlignedOnScalar = 1, + size = 16, + HasHalfPacket = 1, #if EIGEN_GNUC_AT_LEAST(5, 3) #ifdef EIGEN_VECTORIZE_AVX512DQ - HasLog = 1, + HasLog = 1, #endif - HasExp = 1, - HasSqrt = 1, - HasRsqrt = 1, + HasExp = 1, + HasSqrt = 1, + HasRsqrt = 1, #endif - HasDiv = 1 + HasDiv = 1 + }; }; - }; -template<> struct packet_traits : default_packet_traits -{ - typedef Packet8d type; - typedef Packet4d half; - enum { - Vectorizable = 1, - AlignedOnScalar = 1, - size = 8, - HasHalfPacket = 1, + template<> struct packet_traits : default_packet_traits + { + typedef Packet8d type; + typedef Packet4d half; + enum { + Vectorizable = 1, + AlignedOnScalar = 1, + size = 8, + HasHalfPacket = 1, #if EIGEN_GNUC_AT_LEAST(5, 3) - HasSqrt = 1, - HasRsqrt = EIGEN_FAST_MATH, + HasSqrt = 1, + HasRsqrt = EIGEN_FAST_MATH, #endif - HasDiv = 1 + HasDiv = 1 + }; + }; + + /* TODO Implement AVX512 for integers + template<> struct packet_traits : default_packet_traits + { + typedef Packet16i type; + enum { + Vectorizable = 1, + AlignedOnScalar = 1, + size=8 + }; + }; + */ + + template<> struct unpacket_traits + { + typedef float type; + typedef Packet8f half; + enum { size = 16, alignment = Aligned64 }; }; -}; - -/* TODO Implement AVX512 for integers -template<> struct packet_traits : default_packet_traits -{ - typedef Packet16i type; - enum { - Vectorizable = 1, - AlignedOnScalar = 1, - size=8 + template<> struct unpacket_traits + { + typedef double type; + typedef Packet4d half; + enum { size = 8, alignment = Aligned64 }; }; -}; -*/ - -template <> -struct unpacket_traits { - typedef float type; - typedef Packet8f half; - enum { size = 16, alignment=Aligned64 }; -}; -template <> -struct unpacket_traits { - typedef double type; - typedef Packet4d half; - enum { size = 8, alignment=Aligned64 }; -}; -template <> -struct unpacket_traits { - typedef int type; - typedef Packet8i half; - enum { size = 16, alignment=Aligned64 }; -}; - -template <> -EIGEN_STRONG_INLINE Packet16f pset1(const float& from) { - return _mm512_set1_ps(from); -} -template <> -EIGEN_STRONG_INLINE Packet8d pset1(const double& from) { - return _mm512_set1_pd(from); -} -template <> -EIGEN_STRONG_INLINE Packet16i pset1(const int& from) { - return _mm512_set1_epi32(from); -} - -template <> -EIGEN_STRONG_INLINE Packet16f pload1(const float* from) { - return _mm512_broadcastss_ps(_mm_load_ps1(from)); -} -template <> -EIGEN_STRONG_INLINE Packet8d pload1(const double* from) { - return _mm512_broadcastsd_pd(_mm_load_pd1(from)); -} - -template <> -EIGEN_STRONG_INLINE Packet16f plset(const float& a) { - return _mm512_add_ps( - _mm512_set1_ps(a), - _mm512_set_ps(15.0f, 14.0f, 13.0f, 12.0f, 11.0f, 10.0f, 9.0f, 8.0f, 7.0f, 6.0f, 5.0f, - 4.0f, 3.0f, 2.0f, 1.0f, 0.0f)); -} -template <> -EIGEN_STRONG_INLINE Packet8d plset(const double& a) { - return _mm512_add_pd(_mm512_set1_pd(a), - _mm512_set_pd(7.0, 6.0, 5.0, 4.0, 3.0, 2.0, 1.0, 0.0)); -} - -template <> -EIGEN_STRONG_INLINE Packet16f padd(const Packet16f& a, - const Packet16f& b) { - return _mm512_add_ps(a, b); -} -template <> -EIGEN_STRONG_INLINE Packet8d padd(const Packet8d& a, - const Packet8d& b) { - return _mm512_add_pd(a, b); -} - -template <> -EIGEN_STRONG_INLINE Packet16f psub(const Packet16f& a, - const Packet16f& b) { - return _mm512_sub_ps(a, b); -} -template <> -EIGEN_STRONG_INLINE Packet8d psub(const Packet8d& a, - const Packet8d& b) { - return _mm512_sub_pd(a, b); -} - -template <> -EIGEN_STRONG_INLINE Packet16f pnegate(const Packet16f& a) { - return _mm512_sub_ps(_mm512_set1_ps(0.0), a); -} -template <> -EIGEN_STRONG_INLINE Packet8d pnegate(const Packet8d& a) { - return _mm512_sub_pd(_mm512_set1_pd(0.0), a); -} - -template <> -EIGEN_STRONG_INLINE Packet16f pconj(const Packet16f& a) { - return a; -} -template <> -EIGEN_STRONG_INLINE Packet8d pconj(const Packet8d& a) { - return a; -} -template <> -EIGEN_STRONG_INLINE Packet16i pconj(const Packet16i& a) { - return a; -} - -template <> -EIGEN_STRONG_INLINE Packet16f pmul(const Packet16f& a, - const Packet16f& b) { - return _mm512_mul_ps(a, b); -} -template <> -EIGEN_STRONG_INLINE Packet8d pmul(const Packet8d& a, - const Packet8d& b) { - return _mm512_mul_pd(a, b); -} - -template <> -EIGEN_STRONG_INLINE Packet16f pdiv(const Packet16f& a, - const Packet16f& b) { - return _mm512_div_ps(a, b); -} -template <> -EIGEN_STRONG_INLINE Packet8d pdiv(const Packet8d& a, - const Packet8d& b) { - return _mm512_div_pd(a, b); -} + template<> struct unpacket_traits + { + typedef int type; + typedef Packet8i half; + enum { size = 16, alignment = Aligned64 }; + }; + + template<> EIGEN_STRONG_INLINE Packet16f pset1(const float &from) { return _mm512_set1_ps(from); } + template<> EIGEN_STRONG_INLINE Packet8d pset1(const double &from) { return _mm512_set1_pd(from); } + template<> EIGEN_STRONG_INLINE Packet16i pset1(const int &from) { return _mm512_set1_epi32(from); } + + template<> EIGEN_STRONG_INLINE Packet16f pload1(const float *from) + { + return _mm512_broadcastss_ps(_mm_load_ps1(from)); + } + template<> EIGEN_STRONG_INLINE Packet8d pload1(const double *from) + { + return _mm512_broadcastsd_pd(_mm_load_pd1(from)); + } + + template<> EIGEN_STRONG_INLINE Packet16f plset(const float &a) + { + return _mm512_add_ps(_mm512_set1_ps(a), + _mm512_set_ps( + 15.0f, 14.0f, 13.0f, 12.0f, 11.0f, 10.0f, 9.0f, 8.0f, 7.0f, 6.0f, 5.0f, 4.0f, 3.0f, 2.0f, 1.0f, 0.0f)); + } + template<> EIGEN_STRONG_INLINE Packet8d plset(const double &a) + { + return _mm512_add_pd(_mm512_set1_pd(a), _mm512_set_pd(7.0, 6.0, 5.0, 4.0, 3.0, 2.0, 1.0, 0.0)); + } + + template<> EIGEN_STRONG_INLINE Packet16f padd(const Packet16f &a, const Packet16f &b) + { + return _mm512_add_ps(a, b); + } + template<> EIGEN_STRONG_INLINE Packet8d padd(const Packet8d &a, const Packet8d &b) + { + return _mm512_add_pd(a, b); + } + + template<> EIGEN_STRONG_INLINE Packet16f psub(const Packet16f &a, const Packet16f &b) + { + return _mm512_sub_ps(a, b); + } + template<> EIGEN_STRONG_INLINE Packet8d psub(const Packet8d &a, const Packet8d &b) + { + return _mm512_sub_pd(a, b); + } + + template<> EIGEN_STRONG_INLINE Packet16f pnegate(const Packet16f &a) { return _mm512_sub_ps(_mm512_set1_ps(0.0), a); } + template<> EIGEN_STRONG_INLINE Packet8d pnegate(const Packet8d &a) { return _mm512_sub_pd(_mm512_set1_pd(0.0), a); } + + template<> EIGEN_STRONG_INLINE Packet16f pconj(const Packet16f &a) { return a; } + template<> EIGEN_STRONG_INLINE Packet8d pconj(const Packet8d &a) { return a; } + template<> EIGEN_STRONG_INLINE Packet16i pconj(const Packet16i &a) { return a; } + + template<> EIGEN_STRONG_INLINE Packet16f pmul(const Packet16f &a, const Packet16f &b) + { + return _mm512_mul_ps(a, b); + } + template<> EIGEN_STRONG_INLINE Packet8d pmul(const Packet8d &a, const Packet8d &b) + { + return _mm512_mul_pd(a, b); + } + + template<> EIGEN_STRONG_INLINE Packet16f pdiv(const Packet16f &a, const Packet16f &b) + { + return _mm512_div_ps(a, b); + } + template<> EIGEN_STRONG_INLINE Packet8d pdiv(const Packet8d &a, const Packet8d &b) + { + return _mm512_div_pd(a, b); + } #ifdef __FMA__ -template <> -EIGEN_STRONG_INLINE Packet16f pmadd(const Packet16f& a, const Packet16f& b, - const Packet16f& c) { - return _mm512_fmadd_ps(a, b, c); -} -template <> -EIGEN_STRONG_INLINE Packet8d pmadd(const Packet8d& a, const Packet8d& b, - const Packet8d& c) { - return _mm512_fmadd_pd(a, b, c); -} + template<> EIGEN_STRONG_INLINE Packet16f pmadd(const Packet16f &a, const Packet16f &b, const Packet16f &c) + { + return _mm512_fmadd_ps(a, b, c); + } + template<> EIGEN_STRONG_INLINE Packet8d pmadd(const Packet8d &a, const Packet8d &b, const Packet8d &c) + { + return _mm512_fmadd_pd(a, b, c); + } #endif -template <> -EIGEN_STRONG_INLINE Packet16f pmin(const Packet16f& a, - const Packet16f& b) { - return _mm512_min_ps(a, b); -} -template <> -EIGEN_STRONG_INLINE Packet8d pmin(const Packet8d& a, - const Packet8d& b) { - return _mm512_min_pd(a, b); -} - -template <> -EIGEN_STRONG_INLINE Packet16f pmax(const Packet16f& a, - const Packet16f& b) { - return _mm512_max_ps(a, b); -} -template <> -EIGEN_STRONG_INLINE Packet8d pmax(const Packet8d& a, - const Packet8d& b) { - return _mm512_max_pd(a, b); -} - -template <> -EIGEN_STRONG_INLINE Packet16f pand(const Packet16f& a, - const Packet16f& b) { + template<> EIGEN_STRONG_INLINE Packet16f pmin(const Packet16f &a, const Packet16f &b) + { + return _mm512_min_ps(a, b); + } + template<> EIGEN_STRONG_INLINE Packet8d pmin(const Packet8d &a, const Packet8d &b) + { + return _mm512_min_pd(a, b); + } + + template<> EIGEN_STRONG_INLINE Packet16f pmax(const Packet16f &a, const Packet16f &b) + { + return _mm512_max_ps(a, b); + } + template<> EIGEN_STRONG_INLINE Packet8d pmax(const Packet8d &a, const Packet8d &b) + { + return _mm512_max_pd(a, b); + } + + template<> EIGEN_STRONG_INLINE Packet16f pand(const Packet16f &a, const Packet16f &b) + { #ifdef EIGEN_VECTORIZE_AVX512DQ - return _mm512_and_ps(a, b); + return _mm512_and_ps(a, b); #else - Packet16f res = _mm512_undefined_ps(); - Packet4f lane0_a = _mm512_extractf32x4_ps(a, 0); - Packet4f lane0_b = _mm512_extractf32x4_ps(b, 0); - res = _mm512_insertf32x4(res, _mm_and_ps(lane0_a, lane0_b), 0); + Packet16f res = _mm512_undefined_ps(); + Packet4f lane0_a = _mm512_extractf32x4_ps(a, 0); + Packet4f lane0_b = _mm512_extractf32x4_ps(b, 0); + res = _mm512_insertf32x4(res, _mm_and_ps(lane0_a, lane0_b), 0); - Packet4f lane1_a = _mm512_extractf32x4_ps(a, 1); - Packet4f lane1_b = _mm512_extractf32x4_ps(b, 1); - res = _mm512_insertf32x4(res, _mm_and_ps(lane1_a, lane1_b), 1); + Packet4f lane1_a = _mm512_extractf32x4_ps(a, 1); + Packet4f lane1_b = _mm512_extractf32x4_ps(b, 1); + res = _mm512_insertf32x4(res, _mm_and_ps(lane1_a, lane1_b), 1); - Packet4f lane2_a = _mm512_extractf32x4_ps(a, 2); - Packet4f lane2_b = _mm512_extractf32x4_ps(b, 2); - res = _mm512_insertf32x4(res, _mm_and_ps(lane2_a, lane2_b), 2); + Packet4f lane2_a = _mm512_extractf32x4_ps(a, 2); + Packet4f lane2_b = _mm512_extractf32x4_ps(b, 2); + res = _mm512_insertf32x4(res, _mm_and_ps(lane2_a, lane2_b), 2); - Packet4f lane3_a = _mm512_extractf32x4_ps(a, 3); - Packet4f lane3_b = _mm512_extractf32x4_ps(b, 3); - res = _mm512_insertf32x4(res, _mm_and_ps(lane3_a, lane3_b), 3); + Packet4f lane3_a = _mm512_extractf32x4_ps(a, 3); + Packet4f lane3_b = _mm512_extractf32x4_ps(b, 3); + res = _mm512_insertf32x4(res, _mm_and_ps(lane3_a, lane3_b), 3); - return res; + return res; #endif -} -template <> -EIGEN_STRONG_INLINE Packet8d pand(const Packet8d& a, - const Packet8d& b) { + } + template<> EIGEN_STRONG_INLINE Packet8d pand(const Packet8d &a, const Packet8d &b) + { #ifdef EIGEN_VECTORIZE_AVX512DQ - return _mm512_and_pd(a, b); + return _mm512_and_pd(a, b); #else - Packet8d res = _mm512_undefined_pd(); - Packet4d lane0_a = _mm512_extractf64x4_pd(a, 0); - Packet4d lane0_b = _mm512_extractf64x4_pd(b, 0); - res = _mm512_insertf64x4(res, _mm256_and_pd(lane0_a, lane0_b), 0); + Packet8d res = _mm512_undefined_pd(); + Packet4d lane0_a = _mm512_extractf64x4_pd(a, 0); + Packet4d lane0_b = _mm512_extractf64x4_pd(b, 0); + res = _mm512_insertf64x4(res, _mm256_and_pd(lane0_a, lane0_b), 0); - Packet4d lane1_a = _mm512_extractf64x4_pd(a, 1); - Packet4d lane1_b = _mm512_extractf64x4_pd(b, 1); - res = _mm512_insertf64x4(res, _mm256_and_pd(lane1_a, lane1_b), 1); + Packet4d lane1_a = _mm512_extractf64x4_pd(a, 1); + Packet4d lane1_b = _mm512_extractf64x4_pd(b, 1); + res = _mm512_insertf64x4(res, _mm256_and_pd(lane1_a, lane1_b), 1); - return res; + return res; #endif -} -template <> -EIGEN_STRONG_INLINE Packet16f por(const Packet16f& a, - const Packet16f& b) { + } + template<> EIGEN_STRONG_INLINE Packet16f por(const Packet16f &a, const Packet16f &b) + { #ifdef EIGEN_VECTORIZE_AVX512DQ - return _mm512_or_ps(a, b); + return _mm512_or_ps(a, b); #else - Packet16f res = _mm512_undefined_ps(); - Packet4f lane0_a = _mm512_extractf32x4_ps(a, 0); - Packet4f lane0_b = _mm512_extractf32x4_ps(b, 0); - res = _mm512_insertf32x4(res, _mm_or_ps(lane0_a, lane0_b), 0); + Packet16f res = _mm512_undefined_ps(); + Packet4f lane0_a = _mm512_extractf32x4_ps(a, 0); + Packet4f lane0_b = _mm512_extractf32x4_ps(b, 0); + res = _mm512_insertf32x4(res, _mm_or_ps(lane0_a, lane0_b), 0); - Packet4f lane1_a = _mm512_extractf32x4_ps(a, 1); - Packet4f lane1_b = _mm512_extractf32x4_ps(b, 1); - res = _mm512_insertf32x4(res, _mm_or_ps(lane1_a, lane1_b), 1); + Packet4f lane1_a = _mm512_extractf32x4_ps(a, 1); + Packet4f lane1_b = _mm512_extractf32x4_ps(b, 1); + res = _mm512_insertf32x4(res, _mm_or_ps(lane1_a, lane1_b), 1); - Packet4f lane2_a = _mm512_extractf32x4_ps(a, 2); - Packet4f lane2_b = _mm512_extractf32x4_ps(b, 2); - res = _mm512_insertf32x4(res, _mm_or_ps(lane2_a, lane2_b), 2); + Packet4f lane2_a = _mm512_extractf32x4_ps(a, 2); + Packet4f lane2_b = _mm512_extractf32x4_ps(b, 2); + res = _mm512_insertf32x4(res, _mm_or_ps(lane2_a, lane2_b), 2); - Packet4f lane3_a = _mm512_extractf32x4_ps(a, 3); - Packet4f lane3_b = _mm512_extractf32x4_ps(b, 3); - res = _mm512_insertf32x4(res, _mm_or_ps(lane3_a, lane3_b), 3); + Packet4f lane3_a = _mm512_extractf32x4_ps(a, 3); + Packet4f lane3_b = _mm512_extractf32x4_ps(b, 3); + res = _mm512_insertf32x4(res, _mm_or_ps(lane3_a, lane3_b), 3); - return res; + return res; #endif -} + } -template <> -EIGEN_STRONG_INLINE Packet8d por(const Packet8d& a, - const Packet8d& b) { + template<> EIGEN_STRONG_INLINE Packet8d por(const Packet8d &a, const Packet8d &b) + { #ifdef EIGEN_VECTORIZE_AVX512DQ - return _mm512_or_pd(a, b); + return _mm512_or_pd(a, b); #else - Packet8d res = _mm512_undefined_pd(); - Packet4d lane0_a = _mm512_extractf64x4_pd(a, 0); - Packet4d lane0_b = _mm512_extractf64x4_pd(b, 0); - res = _mm512_insertf64x4(res, _mm256_or_pd(lane0_a, lane0_b), 0); + Packet8d res = _mm512_undefined_pd(); + Packet4d lane0_a = _mm512_extractf64x4_pd(a, 0); + Packet4d lane0_b = _mm512_extractf64x4_pd(b, 0); + res = _mm512_insertf64x4(res, _mm256_or_pd(lane0_a, lane0_b), 0); - Packet4d lane1_a = _mm512_extractf64x4_pd(a, 1); - Packet4d lane1_b = _mm512_extractf64x4_pd(b, 1); - res = _mm512_insertf64x4(res, _mm256_or_pd(lane1_a, lane1_b), 1); + Packet4d lane1_a = _mm512_extractf64x4_pd(a, 1); + Packet4d lane1_b = _mm512_extractf64x4_pd(b, 1); + res = _mm512_insertf64x4(res, _mm256_or_pd(lane1_a, lane1_b), 1); - return res; + return res; #endif -} + } -template <> -EIGEN_STRONG_INLINE Packet16f pxor(const Packet16f& a, - const Packet16f& b) { + template<> EIGEN_STRONG_INLINE Packet16f pxor(const Packet16f &a, const Packet16f &b) + { #ifdef EIGEN_VECTORIZE_AVX512DQ - return _mm512_xor_ps(a, b); + return _mm512_xor_ps(a, b); #else - Packet16f res = _mm512_undefined_ps(); - Packet4f lane0_a = _mm512_extractf32x4_ps(a, 0); - Packet4f lane0_b = _mm512_extractf32x4_ps(b, 0); - res = _mm512_insertf32x4(res, _mm_xor_ps(lane0_a, lane0_b), 0); + Packet16f res = _mm512_undefined_ps(); + Packet4f lane0_a = _mm512_extractf32x4_ps(a, 0); + Packet4f lane0_b = _mm512_extractf32x4_ps(b, 0); + res = _mm512_insertf32x4(res, _mm_xor_ps(lane0_a, lane0_b), 0); - Packet4f lane1_a = _mm512_extractf32x4_ps(a, 1); - Packet4f lane1_b = _mm512_extractf32x4_ps(b, 1); - res = _mm512_insertf32x4(res, _mm_xor_ps(lane1_a, lane1_b), 1); + Packet4f lane1_a = _mm512_extractf32x4_ps(a, 1); + Packet4f lane1_b = _mm512_extractf32x4_ps(b, 1); + res = _mm512_insertf32x4(res, _mm_xor_ps(lane1_a, lane1_b), 1); - Packet4f lane2_a = _mm512_extractf32x4_ps(a, 2); - Packet4f lane2_b = _mm512_extractf32x4_ps(b, 2); - res = _mm512_insertf32x4(res, _mm_xor_ps(lane2_a, lane2_b), 2); + Packet4f lane2_a = _mm512_extractf32x4_ps(a, 2); + Packet4f lane2_b = _mm512_extractf32x4_ps(b, 2); + res = _mm512_insertf32x4(res, _mm_xor_ps(lane2_a, lane2_b), 2); - Packet4f lane3_a = _mm512_extractf32x4_ps(a, 3); - Packet4f lane3_b = _mm512_extractf32x4_ps(b, 3); - res = _mm512_insertf32x4(res, _mm_xor_ps(lane3_a, lane3_b), 3); + Packet4f lane3_a = _mm512_extractf32x4_ps(a, 3); + Packet4f lane3_b = _mm512_extractf32x4_ps(b, 3); + res = _mm512_insertf32x4(res, _mm_xor_ps(lane3_a, lane3_b), 3); - return res; + return res; #endif -} -template <> -EIGEN_STRONG_INLINE Packet8d pxor(const Packet8d& a, - const Packet8d& b) { + } + template<> EIGEN_STRONG_INLINE Packet8d pxor(const Packet8d &a, const Packet8d &b) + { #ifdef EIGEN_VECTORIZE_AVX512DQ - return _mm512_xor_pd(a, b); + return _mm512_xor_pd(a, b); #else - Packet8d res = _mm512_undefined_pd(); - Packet4d lane0_a = _mm512_extractf64x4_pd(a, 0); - Packet4d lane0_b = _mm512_extractf64x4_pd(b, 0); - res = _mm512_insertf64x4(res, _mm256_xor_pd(lane0_a, lane0_b), 0); + Packet8d res = _mm512_undefined_pd(); + Packet4d lane0_a = _mm512_extractf64x4_pd(a, 0); + Packet4d lane0_b = _mm512_extractf64x4_pd(b, 0); + res = _mm512_insertf64x4(res, _mm256_xor_pd(lane0_a, lane0_b), 0); - Packet4d lane1_a = _mm512_extractf64x4_pd(a, 1); - Packet4d lane1_b = _mm512_extractf64x4_pd(b, 1); - res = _mm512_insertf64x4(res, _mm256_xor_pd(lane1_a, lane1_b), 1); + Packet4d lane1_a = _mm512_extractf64x4_pd(a, 1); + Packet4d lane1_b = _mm512_extractf64x4_pd(b, 1); + res = _mm512_insertf64x4(res, _mm256_xor_pd(lane1_a, lane1_b), 1); - return res; + return res; #endif -} + } -template <> -EIGEN_STRONG_INLINE Packet16f pandnot(const Packet16f& a, - const Packet16f& b) { + template<> EIGEN_STRONG_INLINE Packet16f pandnot(const Packet16f &a, const Packet16f &b) + { #ifdef EIGEN_VECTORIZE_AVX512DQ - return _mm512_andnot_ps(a, b); + return _mm512_andnot_ps(a, b); #else - Packet16f res = _mm512_undefined_ps(); - Packet4f lane0_a = _mm512_extractf32x4_ps(a, 0); - Packet4f lane0_b = _mm512_extractf32x4_ps(b, 0); - res = _mm512_insertf32x4(res, _mm_andnot_ps(lane0_a, lane0_b), 0); + Packet16f res = _mm512_undefined_ps(); + Packet4f lane0_a = _mm512_extractf32x4_ps(a, 0); + Packet4f lane0_b = _mm512_extractf32x4_ps(b, 0); + res = _mm512_insertf32x4(res, _mm_andnot_ps(lane0_a, lane0_b), 0); - Packet4f lane1_a = _mm512_extractf32x4_ps(a, 1); - Packet4f lane1_b = _mm512_extractf32x4_ps(b, 1); - res = _mm512_insertf32x4(res, _mm_andnot_ps(lane1_a, lane1_b), 1); + Packet4f lane1_a = _mm512_extractf32x4_ps(a, 1); + Packet4f lane1_b = _mm512_extractf32x4_ps(b, 1); + res = _mm512_insertf32x4(res, _mm_andnot_ps(lane1_a, lane1_b), 1); - Packet4f lane2_a = _mm512_extractf32x4_ps(a, 2); - Packet4f lane2_b = _mm512_extractf32x4_ps(b, 2); - res = _mm512_insertf32x4(res, _mm_andnot_ps(lane2_a, lane2_b), 2); + Packet4f lane2_a = _mm512_extractf32x4_ps(a, 2); + Packet4f lane2_b = _mm512_extractf32x4_ps(b, 2); + res = _mm512_insertf32x4(res, _mm_andnot_ps(lane2_a, lane2_b), 2); - Packet4f lane3_a = _mm512_extractf32x4_ps(a, 3); - Packet4f lane3_b = _mm512_extractf32x4_ps(b, 3); - res = _mm512_insertf32x4(res, _mm_andnot_ps(lane3_a, lane3_b), 3); + Packet4f lane3_a = _mm512_extractf32x4_ps(a, 3); + Packet4f lane3_b = _mm512_extractf32x4_ps(b, 3); + res = _mm512_insertf32x4(res, _mm_andnot_ps(lane3_a, lane3_b), 3); - return res; + return res; #endif -} -template <> -EIGEN_STRONG_INLINE Packet8d pandnot(const Packet8d& a, - const Packet8d& b) { + } + template<> EIGEN_STRONG_INLINE Packet8d pandnot(const Packet8d &a, const Packet8d &b) + { #ifdef EIGEN_VECTORIZE_AVX512DQ - return _mm512_andnot_pd(a, b); + return _mm512_andnot_pd(a, b); #else - Packet8d res = _mm512_undefined_pd(); - Packet4d lane0_a = _mm512_extractf64x4_pd(a, 0); - Packet4d lane0_b = _mm512_extractf64x4_pd(b, 0); - res = _mm512_insertf64x4(res, _mm256_andnot_pd(lane0_a, lane0_b), 0); + Packet8d res = _mm512_undefined_pd(); + Packet4d lane0_a = _mm512_extractf64x4_pd(a, 0); + Packet4d lane0_b = _mm512_extractf64x4_pd(b, 0); + res = _mm512_insertf64x4(res, _mm256_andnot_pd(lane0_a, lane0_b), 0); - Packet4d lane1_a = _mm512_extractf64x4_pd(a, 1); - Packet4d lane1_b = _mm512_extractf64x4_pd(b, 1); - res = _mm512_insertf64x4(res, _mm256_andnot_pd(lane1_a, lane1_b), 1); + Packet4d lane1_a = _mm512_extractf64x4_pd(a, 1); + Packet4d lane1_b = _mm512_extractf64x4_pd(b, 1); + res = _mm512_insertf64x4(res, _mm256_andnot_pd(lane1_a, lane1_b), 1); - return res; + return res; #endif -} - -template <> -EIGEN_STRONG_INLINE Packet16f pload(const float* from) { - EIGEN_DEBUG_ALIGNED_LOAD return _mm512_load_ps(from); -} -template <> -EIGEN_STRONG_INLINE Packet8d pload(const double* from) { - EIGEN_DEBUG_ALIGNED_LOAD return _mm512_load_pd(from); -} -template <> -EIGEN_STRONG_INLINE Packet16i pload(const int* from) { - EIGEN_DEBUG_ALIGNED_LOAD return _mm512_load_si512( - reinterpret_cast(from)); -} - -template <> -EIGEN_STRONG_INLINE Packet16f ploadu(const float* from) { - EIGEN_DEBUG_UNALIGNED_LOAD return _mm512_loadu_ps(from); -} -template <> -EIGEN_STRONG_INLINE Packet8d ploadu(const double* from) { - EIGEN_DEBUG_UNALIGNED_LOAD return _mm512_loadu_pd(from); -} -template <> -EIGEN_STRONG_INLINE Packet16i ploadu(const int* from) { - EIGEN_DEBUG_UNALIGNED_LOAD return _mm512_loadu_si512( - reinterpret_cast(from)); -} - -// Loads 8 floats from memory a returns the packet -// {a0, a0 a1, a1, a2, a2, a3, a3, a4, a4, a5, a5, a6, a6, a7, a7} -template <> -EIGEN_STRONG_INLINE Packet16f ploaddup(const float* from) { - Packet8f lane0 = _mm256_broadcast_ps((const __m128*)(const void*)from); - // mimic an "inplace" permutation of the lower 128bits using a blend - lane0 = _mm256_blend_ps( - lane0, _mm256_castps128_ps256(_mm_permute_ps( - _mm256_castps256_ps128(lane0), _MM_SHUFFLE(1, 0, 1, 0))), - 15); - // then we can perform a consistent permutation on the global register to get - // everything in shape: - lane0 = _mm256_permute_ps(lane0, _MM_SHUFFLE(3, 3, 2, 2)); - - Packet8f lane1 = _mm256_broadcast_ps((const __m128*)(const void*)(from + 4)); - // mimic an "inplace" permutation of the lower 128bits using a blend - lane1 = _mm256_blend_ps( - lane1, _mm256_castps128_ps256(_mm_permute_ps( - _mm256_castps256_ps128(lane1), _MM_SHUFFLE(1, 0, 1, 0))), - 15); - // then we can perform a consistent permutation on the global register to get - // everything in shape: - lane1 = _mm256_permute_ps(lane1, _MM_SHUFFLE(3, 3, 2, 2)); + } + + template<> EIGEN_STRONG_INLINE Packet16f pload(const float *from) + { + EIGEN_DEBUG_ALIGNED_LOAD return _mm512_load_ps(from); + } + template<> EIGEN_STRONG_INLINE Packet8d pload(const double *from) + { + EIGEN_DEBUG_ALIGNED_LOAD return _mm512_load_pd(from); + } + template<> EIGEN_STRONG_INLINE Packet16i pload(const int *from) + { + EIGEN_DEBUG_ALIGNED_LOAD return _mm512_load_si512(reinterpret_cast(from)); + } + + template<> EIGEN_STRONG_INLINE Packet16f ploadu(const float *from) + { + EIGEN_DEBUG_UNALIGNED_LOAD return _mm512_loadu_ps(from); + } + template<> EIGEN_STRONG_INLINE Packet8d ploadu(const double *from) + { + EIGEN_DEBUG_UNALIGNED_LOAD return _mm512_loadu_pd(from); + } + template<> EIGEN_STRONG_INLINE Packet16i ploadu(const int *from) + { + EIGEN_DEBUG_UNALIGNED_LOAD return _mm512_loadu_si512(reinterpret_cast(from)); + } + + // Loads 8 floats from memory a returns the packet + // {a0, a0 a1, a1, a2, a2, a3, a3, a4, a4, a5, a5, a6, a6, a7, a7} + template<> EIGEN_STRONG_INLINE Packet16f ploaddup(const float *from) + { + Packet8f lane0 = _mm256_broadcast_ps((const __m128 *)(const void *)from); + // mimic an "inplace" permutation of the lower 128bits using a blend + lane0 = _mm256_blend_ps( + lane0, _mm256_castps128_ps256(_mm_permute_ps(_mm256_castps256_ps128(lane0), _MM_SHUFFLE(1, 0, 1, 0))), 15); + // then we can perform a consistent permutation on the global register to get + // everything in shape: + lane0 = _mm256_permute_ps(lane0, _MM_SHUFFLE(3, 3, 2, 2)); + + Packet8f lane1 = _mm256_broadcast_ps((const __m128 *)(const void *)(from + 4)); + // mimic an "inplace" permutation of the lower 128bits using a blend + lane1 = _mm256_blend_ps( + lane1, _mm256_castps128_ps256(_mm_permute_ps(_mm256_castps256_ps128(lane1), _MM_SHUFFLE(1, 0, 1, 0))), 15); + // then we can perform a consistent permutation on the global register to get + // everything in shape: + lane1 = _mm256_permute_ps(lane1, _MM_SHUFFLE(3, 3, 2, 2)); #ifdef EIGEN_VECTORIZE_AVX512DQ - Packet16f res = _mm512_undefined_ps(); - return _mm512_insertf32x8(res, lane0, 0); - return _mm512_insertf32x8(res, lane1, 1); - return res; + Packet16f res = _mm512_undefined_ps(); + return _mm512_insertf32x8(res, lane0, 0); + return _mm512_insertf32x8(res, lane1, 1); + return res; #else - Packet16f res = _mm512_undefined_ps(); - res = _mm512_insertf32x4(res, _mm256_extractf128_ps(lane0, 0), 0); - res = _mm512_insertf32x4(res, _mm256_extractf128_ps(lane0, 1), 1); - res = _mm512_insertf32x4(res, _mm256_extractf128_ps(lane1, 0), 2); - res = _mm512_insertf32x4(res, _mm256_extractf128_ps(lane1, 1), 3); - return res; + Packet16f res = _mm512_undefined_ps(); + res = _mm512_insertf32x4(res, _mm256_extractf128_ps(lane0, 0), 0); + res = _mm512_insertf32x4(res, _mm256_extractf128_ps(lane0, 1), 1); + res = _mm512_insertf32x4(res, _mm256_extractf128_ps(lane1, 0), 2); + res = _mm512_insertf32x4(res, _mm256_extractf128_ps(lane1, 1), 3); + return res; #endif -} -// Loads 4 doubles from memory a returns the packet {a0, a0 a1, a1, a2, a2, a3, -// a3} -template <> -EIGEN_STRONG_INLINE Packet8d ploaddup(const double* from) { - Packet4d lane0 = _mm256_broadcast_pd((const __m128d*)(const void*)from); - lane0 = _mm256_permute_pd(lane0, 3 << 2); - - Packet4d lane1 = _mm256_broadcast_pd((const __m128d*)(const void*)(from + 2)); - lane1 = _mm256_permute_pd(lane1, 3 << 2); - - Packet8d res = _mm512_undefined_pd(); - res = _mm512_insertf64x4(res, lane0, 0); - return _mm512_insertf64x4(res, lane1, 1); -} - -// Loads 4 floats from memory a returns the packet -// {a0, a0 a0, a0, a1, a1, a1, a1, a2, a2, a2, a2, a3, a3, a3, a3} -template <> -EIGEN_STRONG_INLINE Packet16f ploadquad(const float* from) { - Packet16f tmp = _mm512_undefined_ps(); - tmp = _mm512_insertf32x4(tmp, _mm_load_ps1(from), 0); - tmp = _mm512_insertf32x4(tmp, _mm_load_ps1(from + 1), 1); - tmp = _mm512_insertf32x4(tmp, _mm_load_ps1(from + 2), 2); - tmp = _mm512_insertf32x4(tmp, _mm_load_ps1(from + 3), 3); - return tmp; -} -// Loads 2 doubles from memory a returns the packet -// {a0, a0 a0, a0, a1, a1, a1, a1} -template <> -EIGEN_STRONG_INLINE Packet8d ploadquad(const double* from) { - Packet8d tmp = _mm512_undefined_pd(); - Packet2d tmp0 = _mm_load_pd1(from); - Packet2d tmp1 = _mm_load_pd1(from + 1); - Packet4d lane0 = _mm256_broadcastsd_pd(tmp0); - Packet4d lane1 = _mm256_broadcastsd_pd(tmp1); - tmp = _mm512_insertf64x4(tmp, lane0, 0); - return _mm512_insertf64x4(tmp, lane1, 1); -} - -template <> -EIGEN_STRONG_INLINE void pstore(float* to, const Packet16f& from) { - EIGEN_DEBUG_ALIGNED_STORE _mm512_store_ps(to, from); -} -template <> -EIGEN_STRONG_INLINE void pstore(double* to, const Packet8d& from) { - EIGEN_DEBUG_ALIGNED_STORE _mm512_store_pd(to, from); -} -template <> -EIGEN_STRONG_INLINE void pstore(int* to, const Packet16i& from) { - EIGEN_DEBUG_ALIGNED_STORE _mm512_storeu_si512(reinterpret_cast<__m512i*>(to), - from); -} - -template <> -EIGEN_STRONG_INLINE void pstoreu(float* to, const Packet16f& from) { - EIGEN_DEBUG_UNALIGNED_STORE _mm512_storeu_ps(to, from); -} -template <> -EIGEN_STRONG_INLINE void pstoreu(double* to, const Packet8d& from) { - EIGEN_DEBUG_UNALIGNED_STORE _mm512_storeu_pd(to, from); -} -template <> -EIGEN_STRONG_INLINE void pstoreu(int* to, const Packet16i& from) { - EIGEN_DEBUG_UNALIGNED_STORE _mm512_storeu_si512( - reinterpret_cast<__m512i*>(to), from); -} - -template <> -EIGEN_DEVICE_FUNC inline Packet16f pgather(const float* from, - Index stride) { - Packet16i stride_vector = _mm512_set1_epi32(stride); - Packet16i stride_multiplier = - _mm512_set_epi32(15, 14, 13, 12, 11, 10, 9, 8, 7, 6, 5, 4, 3, 2, 1, 0); - Packet16i indices = _mm512_mullo_epi32(stride_vector, stride_multiplier); - - return _mm512_i32gather_ps(indices, from, 4); -} -template <> -EIGEN_DEVICE_FUNC inline Packet8d pgather(const double* from, - Index stride) { - Packet8i stride_vector = _mm256_set1_epi32(stride); - Packet8i stride_multiplier = _mm256_set_epi32(7, 6, 5, 4, 3, 2, 1, 0); - Packet8i indices = _mm256_mullo_epi32(stride_vector, stride_multiplier); - - return _mm512_i32gather_pd(indices, from, 8); -} - -template <> -EIGEN_DEVICE_FUNC inline void pscatter(float* to, - const Packet16f& from, - Index stride) { - Packet16i stride_vector = _mm512_set1_epi32(stride); - Packet16i stride_multiplier = - _mm512_set_epi32(15, 14, 13, 12, 11, 10, 9, 8, 7, 6, 5, 4, 3, 2, 1, 0); - Packet16i indices = _mm512_mullo_epi32(stride_vector, stride_multiplier); - _mm512_i32scatter_ps(to, indices, from, 4); -} -template <> -EIGEN_DEVICE_FUNC inline void pscatter(double* to, - const Packet8d& from, - Index stride) { - Packet8i stride_vector = _mm256_set1_epi32(stride); - Packet8i stride_multiplier = _mm256_set_epi32(7, 6, 5, 4, 3, 2, 1, 0); - Packet8i indices = _mm256_mullo_epi32(stride_vector, stride_multiplier); - _mm512_i32scatter_pd(to, indices, from, 8); -} - -template <> -EIGEN_STRONG_INLINE void pstore1(float* to, const float& a) { - Packet16f pa = pset1(a); - pstore(to, pa); -} -template <> -EIGEN_STRONG_INLINE void pstore1(double* to, const double& a) { - Packet8d pa = pset1(a); - pstore(to, pa); -} -template <> -EIGEN_STRONG_INLINE void pstore1(int* to, const int& a) { - Packet16i pa = pset1(a); - pstore(to, pa); -} - -template<> EIGEN_STRONG_INLINE void prefetch(const float* addr) { _mm_prefetch((SsePrefetchPtrType)(addr), _MM_HINT_T0); } -template<> EIGEN_STRONG_INLINE void prefetch(const double* addr) { _mm_prefetch((SsePrefetchPtrType)(addr), _MM_HINT_T0); } -template<> EIGEN_STRONG_INLINE void prefetch(const int* addr) { _mm_prefetch((SsePrefetchPtrType)(addr), _MM_HINT_T0); } - -template <> -EIGEN_STRONG_INLINE float pfirst(const Packet16f& a) { - return _mm_cvtss_f32(_mm512_extractf32x4_ps(a, 0)); -} -template <> -EIGEN_STRONG_INLINE double pfirst(const Packet8d& a) { - return _mm_cvtsd_f64(_mm256_extractf128_pd(_mm512_extractf64x4_pd(a, 0), 0)); -} -template <> -EIGEN_STRONG_INLINE int pfirst(const Packet16i& a) { - return _mm_extract_epi32(_mm512_extracti32x4_epi32(a, 0), 0); -} - -template<> EIGEN_STRONG_INLINE Packet16f preverse(const Packet16f& a) -{ - return _mm512_permutexvar_ps(_mm512_set_epi32(0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15), a); -} - -template<> EIGEN_STRONG_INLINE Packet8d preverse(const Packet8d& a) -{ - return _mm512_permutexvar_pd(_mm512_set_epi32(0, 0, 0, 1, 0, 2, 0, 3, 0, 4, 0, 5, 0, 6, 0, 7), a); -} - -template<> EIGEN_STRONG_INLINE Packet16f pabs(const Packet16f& a) -{ - // _mm512_abs_ps intrinsic not found, so hack around it - return _mm512_castsi512_ps(_mm512_and_si512(_mm512_castps_si512(a), _mm512_set1_epi32(0x7fffffff))); -} -template <> -EIGEN_STRONG_INLINE Packet8d pabs(const Packet8d& a) { - // _mm512_abs_ps intrinsic not found, so hack around it - return _mm512_castsi512_pd(_mm512_and_si512(_mm512_castpd_si512(a), - _mm512_set1_epi64(0x7fffffffffffffff))); -} + } + // Loads 4 doubles from memory a returns the packet {a0, a0 a1, a1, a2, a2, a3, + // a3} + template<> EIGEN_STRONG_INLINE Packet8d ploaddup(const double *from) + { + Packet4d lane0 = _mm256_broadcast_pd((const __m128d *)(const void *)from); + lane0 = _mm256_permute_pd(lane0, 3 << 2); + + Packet4d lane1 = _mm256_broadcast_pd((const __m128d *)(const void *)(from + 2)); + lane1 = _mm256_permute_pd(lane1, 3 << 2); + + Packet8d res = _mm512_undefined_pd(); + res = _mm512_insertf64x4(res, lane0, 0); + return _mm512_insertf64x4(res, lane1, 1); + } + + // Loads 4 floats from memory a returns the packet + // {a0, a0 a0, a0, a1, a1, a1, a1, a2, a2, a2, a2, a3, a3, a3, a3} + template<> EIGEN_STRONG_INLINE Packet16f ploadquad(const float *from) + { + Packet16f tmp = _mm512_undefined_ps(); + tmp = _mm512_insertf32x4(tmp, _mm_load_ps1(from), 0); + tmp = _mm512_insertf32x4(tmp, _mm_load_ps1(from + 1), 1); + tmp = _mm512_insertf32x4(tmp, _mm_load_ps1(from + 2), 2); + tmp = _mm512_insertf32x4(tmp, _mm_load_ps1(from + 3), 3); + return tmp; + } + // Loads 2 doubles from memory a returns the packet + // {a0, a0 a0, a0, a1, a1, a1, a1} + template<> EIGEN_STRONG_INLINE Packet8d ploadquad(const double *from) + { + Packet8d tmp = _mm512_undefined_pd(); + Packet2d tmp0 = _mm_load_pd1(from); + Packet2d tmp1 = _mm_load_pd1(from + 1); + Packet4d lane0 = _mm256_broadcastsd_pd(tmp0); + Packet4d lane1 = _mm256_broadcastsd_pd(tmp1); + tmp = _mm512_insertf64x4(tmp, lane0, 0); + return _mm512_insertf64x4(tmp, lane1, 1); + } + + template<> EIGEN_STRONG_INLINE void pstore(float *to, const Packet16f &from) + { + EIGEN_DEBUG_ALIGNED_STORE _mm512_store_ps(to, from); + } + template<> EIGEN_STRONG_INLINE void pstore(double *to, const Packet8d &from) + { + EIGEN_DEBUG_ALIGNED_STORE _mm512_store_pd(to, from); + } + template<> EIGEN_STRONG_INLINE void pstore(int *to, const Packet16i &from) + { + EIGEN_DEBUG_ALIGNED_STORE _mm512_storeu_si512(reinterpret_cast<__m512i *>(to), from); + } + + template<> EIGEN_STRONG_INLINE void pstoreu(float *to, const Packet16f &from) + { + EIGEN_DEBUG_UNALIGNED_STORE _mm512_storeu_ps(to, from); + } + template<> EIGEN_STRONG_INLINE void pstoreu(double *to, const Packet8d &from) + { + EIGEN_DEBUG_UNALIGNED_STORE _mm512_storeu_pd(to, from); + } + template<> EIGEN_STRONG_INLINE void pstoreu(int *to, const Packet16i &from) + { + EIGEN_DEBUG_UNALIGNED_STORE _mm512_storeu_si512(reinterpret_cast<__m512i *>(to), from); + } + + template<> EIGEN_DEVICE_FUNC inline Packet16f pgather(const float *from, Index stride) + { + Packet16i stride_vector = _mm512_set1_epi32(stride); + Packet16i stride_multiplier = _mm512_set_epi32(15, 14, 13, 12, 11, 10, 9, 8, 7, 6, 5, 4, 3, 2, 1, 0); + Packet16i indices = _mm512_mullo_epi32(stride_vector, stride_multiplier); + + return _mm512_i32gather_ps(indices, from, 4); + } + template<> EIGEN_DEVICE_FUNC inline Packet8d pgather(const double *from, Index stride) + { + Packet8i stride_vector = _mm256_set1_epi32(stride); + Packet8i stride_multiplier = _mm256_set_epi32(7, 6, 5, 4, 3, 2, 1, 0); + Packet8i indices = _mm256_mullo_epi32(stride_vector, stride_multiplier); + + return _mm512_i32gather_pd(indices, from, 8); + } + + template<> EIGEN_DEVICE_FUNC inline void pscatter(float *to, const Packet16f &from, Index stride) + { + Packet16i stride_vector = _mm512_set1_epi32(stride); + Packet16i stride_multiplier = _mm512_set_epi32(15, 14, 13, 12, 11, 10, 9, 8, 7, 6, 5, 4, 3, 2, 1, 0); + Packet16i indices = _mm512_mullo_epi32(stride_vector, stride_multiplier); + _mm512_i32scatter_ps(to, indices, from, 4); + } + template<> EIGEN_DEVICE_FUNC inline void pscatter(double *to, const Packet8d &from, Index stride) + { + Packet8i stride_vector = _mm256_set1_epi32(stride); + Packet8i stride_multiplier = _mm256_set_epi32(7, 6, 5, 4, 3, 2, 1, 0); + Packet8i indices = _mm256_mullo_epi32(stride_vector, stride_multiplier); + _mm512_i32scatter_pd(to, indices, from, 8); + } + + template<> EIGEN_STRONG_INLINE void pstore1(float *to, const float &a) + { + Packet16f pa = pset1(a); + pstore(to, pa); + } + template<> EIGEN_STRONG_INLINE void pstore1(double *to, const double &a) + { + Packet8d pa = pset1(a); + pstore(to, pa); + } + template<> EIGEN_STRONG_INLINE void pstore1(int *to, const int &a) + { + Packet16i pa = pset1(a); + pstore(to, pa); + } + + template<> EIGEN_STRONG_INLINE void prefetch(const float *addr) + { + _mm_prefetch((SsePrefetchPtrType)(addr), _MM_HINT_T0); + } + template<> EIGEN_STRONG_INLINE void prefetch(const double *addr) + { + _mm_prefetch((SsePrefetchPtrType)(addr), _MM_HINT_T0); + } + template<> EIGEN_STRONG_INLINE void prefetch(const int *addr) + { + _mm_prefetch((SsePrefetchPtrType)(addr), _MM_HINT_T0); + } + + template<> EIGEN_STRONG_INLINE float pfirst(const Packet16f &a) + { + return _mm_cvtss_f32(_mm512_extractf32x4_ps(a, 0)); + } + template<> EIGEN_STRONG_INLINE double pfirst(const Packet8d &a) + { + return _mm_cvtsd_f64(_mm256_extractf128_pd(_mm512_extractf64x4_pd(a, 0), 0)); + } + template<> EIGEN_STRONG_INLINE int pfirst(const Packet16i &a) + { + return _mm_extract_epi32(_mm512_extracti32x4_epi32(a, 0), 0); + } + + template<> EIGEN_STRONG_INLINE Packet16f preverse(const Packet16f &a) + { + return _mm512_permutexvar_ps(_mm512_set_epi32(0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15), a); + } + + template<> EIGEN_STRONG_INLINE Packet8d preverse(const Packet8d &a) + { + return _mm512_permutexvar_pd(_mm512_set_epi32(0, 0, 0, 1, 0, 2, 0, 3, 0, 4, 0, 5, 0, 6, 0, 7), a); + } + + template<> EIGEN_STRONG_INLINE Packet16f pabs(const Packet16f &a) + { + // _mm512_abs_ps intrinsic not found, so hack around it + return _mm512_castsi512_ps(_mm512_and_si512(_mm512_castps_si512(a), _mm512_set1_epi32(0x7fffffff))); + } + template<> EIGEN_STRONG_INLINE Packet8d pabs(const Packet8d &a) + { + // _mm512_abs_ps intrinsic not found, so hack around it + return _mm512_castsi512_pd(_mm512_and_si512(_mm512_castpd_si512(a), _mm512_set1_epi64(0x7fffffffffffffff))); + } #ifdef EIGEN_VECTORIZE_AVX512DQ // AVX512F does not define _mm512_extractf32x8_ps to extract _m256 from _m512 -#define EIGEN_EXTRACT_8f_FROM_16f(INPUT, OUTPUT) \ - __m256 OUTPUT##_0 = _mm512_extractf32x8_ps(INPUT, 0) __m256 OUTPUT##_1 = \ - _mm512_extractf32x8_ps(INPUT, 1) +#define EIGEN_EXTRACT_8f_FROM_16f(INPUT, OUTPUT) \ + __m256 OUTPUT##_0 = _mm512_extractf32x8_ps(INPUT, 0) __m256 OUTPUT##_1 = _mm512_extractf32x8_ps(INPUT, 1) #else -#define EIGEN_EXTRACT_8f_FROM_16f(INPUT, OUTPUT) \ - __m256 OUTPUT##_0 = _mm256_insertf128_ps( \ - _mm256_castps128_ps256(_mm512_extractf32x4_ps(INPUT, 0)), \ - _mm512_extractf32x4_ps(INPUT, 1), 1); \ - __m256 OUTPUT##_1 = _mm256_insertf128_ps( \ - _mm256_castps128_ps256(_mm512_extractf32x4_ps(INPUT, 2)), \ - _mm512_extractf32x4_ps(INPUT, 3), 1); +#define EIGEN_EXTRACT_8f_FROM_16f(INPUT, OUTPUT) \ + __m256 OUTPUT##_0 = _mm256_insertf128_ps( \ + _mm256_castps128_ps256(_mm512_extractf32x4_ps(INPUT, 0)), _mm512_extractf32x4_ps(INPUT, 1), 1); \ + __m256 OUTPUT##_1 = _mm256_insertf128_ps( \ + _mm256_castps128_ps256(_mm512_extractf32x4_ps(INPUT, 2)), _mm512_extractf32x4_ps(INPUT, 3), 1); #endif #ifdef EIGEN_VECTORIZE_AVX512DQ @@ -683,200 +624,199 @@ EIGEN_STRONG_INLINE Packet8d pabs(const Packet8d& a) { OUTPUT = _mm512_insertf32x4(OUTPUT, _mm256_extractf128_ps(INPUTB, 0), 2); \ OUTPUT = _mm512_insertf32x4(OUTPUT, _mm256_extractf128_ps(INPUTB, 1), 3); #endif -template<> EIGEN_STRONG_INLINE Packet16f preduxp(const Packet16f* -vecs) -{ - EIGEN_EXTRACT_8f_FROM_16f(vecs[0], vecs0); - EIGEN_EXTRACT_8f_FROM_16f(vecs[1], vecs1); - EIGEN_EXTRACT_8f_FROM_16f(vecs[2], vecs2); - EIGEN_EXTRACT_8f_FROM_16f(vecs[3], vecs3); - EIGEN_EXTRACT_8f_FROM_16f(vecs[4], vecs4); - EIGEN_EXTRACT_8f_FROM_16f(vecs[5], vecs5); - EIGEN_EXTRACT_8f_FROM_16f(vecs[6], vecs6); - EIGEN_EXTRACT_8f_FROM_16f(vecs[7], vecs7); - EIGEN_EXTRACT_8f_FROM_16f(vecs[8], vecs8); - EIGEN_EXTRACT_8f_FROM_16f(vecs[9], vecs9); - EIGEN_EXTRACT_8f_FROM_16f(vecs[10], vecs10); - EIGEN_EXTRACT_8f_FROM_16f(vecs[11], vecs11); - EIGEN_EXTRACT_8f_FROM_16f(vecs[12], vecs12); - EIGEN_EXTRACT_8f_FROM_16f(vecs[13], vecs13); - EIGEN_EXTRACT_8f_FROM_16f(vecs[14], vecs14); - EIGEN_EXTRACT_8f_FROM_16f(vecs[15], vecs15); - - __m256 hsum1 = _mm256_hadd_ps(vecs0_0, vecs1_0); - __m256 hsum2 = _mm256_hadd_ps(vecs2_0, vecs3_0); - __m256 hsum3 = _mm256_hadd_ps(vecs4_0, vecs5_0); - __m256 hsum4 = _mm256_hadd_ps(vecs6_0, vecs7_0); - - __m256 hsum5 = _mm256_hadd_ps(hsum1, hsum1); - __m256 hsum6 = _mm256_hadd_ps(hsum2, hsum2); - __m256 hsum7 = _mm256_hadd_ps(hsum3, hsum3); - __m256 hsum8 = _mm256_hadd_ps(hsum4, hsum4); - - __m256 perm1 = _mm256_permute2f128_ps(hsum5, hsum5, 0x23); - __m256 perm2 = _mm256_permute2f128_ps(hsum6, hsum6, 0x23); - __m256 perm3 = _mm256_permute2f128_ps(hsum7, hsum7, 0x23); - __m256 perm4 = _mm256_permute2f128_ps(hsum8, hsum8, 0x23); - - __m256 sum1 = _mm256_add_ps(perm1, hsum5); - __m256 sum2 = _mm256_add_ps(perm2, hsum6); - __m256 sum3 = _mm256_add_ps(perm3, hsum7); - __m256 sum4 = _mm256_add_ps(perm4, hsum8); - - __m256 blend1 = _mm256_blend_ps(sum1, sum2, 0xcc); - __m256 blend2 = _mm256_blend_ps(sum3, sum4, 0xcc); - - __m256 final = _mm256_blend_ps(blend1, blend2, 0xf0); - - hsum1 = _mm256_hadd_ps(vecs0_1, vecs1_1); - hsum2 = _mm256_hadd_ps(vecs2_1, vecs3_1); - hsum3 = _mm256_hadd_ps(vecs4_1, vecs5_1); - hsum4 = _mm256_hadd_ps(vecs6_1, vecs7_1); - - hsum5 = _mm256_hadd_ps(hsum1, hsum1); - hsum6 = _mm256_hadd_ps(hsum2, hsum2); - hsum7 = _mm256_hadd_ps(hsum3, hsum3); - hsum8 = _mm256_hadd_ps(hsum4, hsum4); - - perm1 = _mm256_permute2f128_ps(hsum5, hsum5, 0x23); - perm2 = _mm256_permute2f128_ps(hsum6, hsum6, 0x23); - perm3 = _mm256_permute2f128_ps(hsum7, hsum7, 0x23); - perm4 = _mm256_permute2f128_ps(hsum8, hsum8, 0x23); - - sum1 = _mm256_add_ps(perm1, hsum5); - sum2 = _mm256_add_ps(perm2, hsum6); - sum3 = _mm256_add_ps(perm3, hsum7); - sum4 = _mm256_add_ps(perm4, hsum8); - - blend1 = _mm256_blend_ps(sum1, sum2, 0xcc); - blend2 = _mm256_blend_ps(sum3, sum4, 0xcc); - - final = padd(final, _mm256_blend_ps(blend1, blend2, 0xf0)); - - hsum1 = _mm256_hadd_ps(vecs8_0, vecs9_0); - hsum2 = _mm256_hadd_ps(vecs10_0, vecs11_0); - hsum3 = _mm256_hadd_ps(vecs12_0, vecs13_0); - hsum4 = _mm256_hadd_ps(vecs14_0, vecs15_0); - - hsum5 = _mm256_hadd_ps(hsum1, hsum1); - hsum6 = _mm256_hadd_ps(hsum2, hsum2); - hsum7 = _mm256_hadd_ps(hsum3, hsum3); - hsum8 = _mm256_hadd_ps(hsum4, hsum4); - - perm1 = _mm256_permute2f128_ps(hsum5, hsum5, 0x23); - perm2 = _mm256_permute2f128_ps(hsum6, hsum6, 0x23); - perm3 = _mm256_permute2f128_ps(hsum7, hsum7, 0x23); - perm4 = _mm256_permute2f128_ps(hsum8, hsum8, 0x23); - - sum1 = _mm256_add_ps(perm1, hsum5); - sum2 = _mm256_add_ps(perm2, hsum6); - sum3 = _mm256_add_ps(perm3, hsum7); - sum4 = _mm256_add_ps(perm4, hsum8); - - blend1 = _mm256_blend_ps(sum1, sum2, 0xcc); - blend2 = _mm256_blend_ps(sum3, sum4, 0xcc); - - __m256 final_1 = _mm256_blend_ps(blend1, blend2, 0xf0); - - hsum1 = _mm256_hadd_ps(vecs8_1, vecs9_1); - hsum2 = _mm256_hadd_ps(vecs10_1, vecs11_1); - hsum3 = _mm256_hadd_ps(vecs12_1, vecs13_1); - hsum4 = _mm256_hadd_ps(vecs14_1, vecs15_1); - - hsum5 = _mm256_hadd_ps(hsum1, hsum1); - hsum6 = _mm256_hadd_ps(hsum2, hsum2); - hsum7 = _mm256_hadd_ps(hsum3, hsum3); - hsum8 = _mm256_hadd_ps(hsum4, hsum4); - - perm1 = _mm256_permute2f128_ps(hsum5, hsum5, 0x23); - perm2 = _mm256_permute2f128_ps(hsum6, hsum6, 0x23); - perm3 = _mm256_permute2f128_ps(hsum7, hsum7, 0x23); - perm4 = _mm256_permute2f128_ps(hsum8, hsum8, 0x23); - - sum1 = _mm256_add_ps(perm1, hsum5); - sum2 = _mm256_add_ps(perm2, hsum6); - sum3 = _mm256_add_ps(perm3, hsum7); - sum4 = _mm256_add_ps(perm4, hsum8); - - blend1 = _mm256_blend_ps(sum1, sum2, 0xcc); - blend2 = _mm256_blend_ps(sum3, sum4, 0xcc); + template<> EIGEN_STRONG_INLINE Packet16f preduxp(const Packet16f *vecs) + { + EIGEN_EXTRACT_8f_FROM_16f(vecs[0], vecs0); + EIGEN_EXTRACT_8f_FROM_16f(vecs[1], vecs1); + EIGEN_EXTRACT_8f_FROM_16f(vecs[2], vecs2); + EIGEN_EXTRACT_8f_FROM_16f(vecs[3], vecs3); + EIGEN_EXTRACT_8f_FROM_16f(vecs[4], vecs4); + EIGEN_EXTRACT_8f_FROM_16f(vecs[5], vecs5); + EIGEN_EXTRACT_8f_FROM_16f(vecs[6], vecs6); + EIGEN_EXTRACT_8f_FROM_16f(vecs[7], vecs7); + EIGEN_EXTRACT_8f_FROM_16f(vecs[8], vecs8); + EIGEN_EXTRACT_8f_FROM_16f(vecs[9], vecs9); + EIGEN_EXTRACT_8f_FROM_16f(vecs[10], vecs10); + EIGEN_EXTRACT_8f_FROM_16f(vecs[11], vecs11); + EIGEN_EXTRACT_8f_FROM_16f(vecs[12], vecs12); + EIGEN_EXTRACT_8f_FROM_16f(vecs[13], vecs13); + EIGEN_EXTRACT_8f_FROM_16f(vecs[14], vecs14); + EIGEN_EXTRACT_8f_FROM_16f(vecs[15], vecs15); + + __m256 hsum1 = _mm256_hadd_ps(vecs0_0, vecs1_0); + __m256 hsum2 = _mm256_hadd_ps(vecs2_0, vecs3_0); + __m256 hsum3 = _mm256_hadd_ps(vecs4_0, vecs5_0); + __m256 hsum4 = _mm256_hadd_ps(vecs6_0, vecs7_0); + + __m256 hsum5 = _mm256_hadd_ps(hsum1, hsum1); + __m256 hsum6 = _mm256_hadd_ps(hsum2, hsum2); + __m256 hsum7 = _mm256_hadd_ps(hsum3, hsum3); + __m256 hsum8 = _mm256_hadd_ps(hsum4, hsum4); + + __m256 perm1 = _mm256_permute2f128_ps(hsum5, hsum5, 0x23); + __m256 perm2 = _mm256_permute2f128_ps(hsum6, hsum6, 0x23); + __m256 perm3 = _mm256_permute2f128_ps(hsum7, hsum7, 0x23); + __m256 perm4 = _mm256_permute2f128_ps(hsum8, hsum8, 0x23); + + __m256 sum1 = _mm256_add_ps(perm1, hsum5); + __m256 sum2 = _mm256_add_ps(perm2, hsum6); + __m256 sum3 = _mm256_add_ps(perm3, hsum7); + __m256 sum4 = _mm256_add_ps(perm4, hsum8); + + __m256 blend1 = _mm256_blend_ps(sum1, sum2, 0xcc); + __m256 blend2 = _mm256_blend_ps(sum3, sum4, 0xcc); + + __m256 final = _mm256_blend_ps(blend1, blend2, 0xf0); + + hsum1 = _mm256_hadd_ps(vecs0_1, vecs1_1); + hsum2 = _mm256_hadd_ps(vecs2_1, vecs3_1); + hsum3 = _mm256_hadd_ps(vecs4_1, vecs5_1); + hsum4 = _mm256_hadd_ps(vecs6_1, vecs7_1); + + hsum5 = _mm256_hadd_ps(hsum1, hsum1); + hsum6 = _mm256_hadd_ps(hsum2, hsum2); + hsum7 = _mm256_hadd_ps(hsum3, hsum3); + hsum8 = _mm256_hadd_ps(hsum4, hsum4); + + perm1 = _mm256_permute2f128_ps(hsum5, hsum5, 0x23); + perm2 = _mm256_permute2f128_ps(hsum6, hsum6, 0x23); + perm3 = _mm256_permute2f128_ps(hsum7, hsum7, 0x23); + perm4 = _mm256_permute2f128_ps(hsum8, hsum8, 0x23); + + sum1 = _mm256_add_ps(perm1, hsum5); + sum2 = _mm256_add_ps(perm2, hsum6); + sum3 = _mm256_add_ps(perm3, hsum7); + sum4 = _mm256_add_ps(perm4, hsum8); + + blend1 = _mm256_blend_ps(sum1, sum2, 0xcc); + blend2 = _mm256_blend_ps(sum3, sum4, 0xcc); + + final = padd(final, _mm256_blend_ps(blend1, blend2, 0xf0)); + + hsum1 = _mm256_hadd_ps(vecs8_0, vecs9_0); + hsum2 = _mm256_hadd_ps(vecs10_0, vecs11_0); + hsum3 = _mm256_hadd_ps(vecs12_0, vecs13_0); + hsum4 = _mm256_hadd_ps(vecs14_0, vecs15_0); + + hsum5 = _mm256_hadd_ps(hsum1, hsum1); + hsum6 = _mm256_hadd_ps(hsum2, hsum2); + hsum7 = _mm256_hadd_ps(hsum3, hsum3); + hsum8 = _mm256_hadd_ps(hsum4, hsum4); + + perm1 = _mm256_permute2f128_ps(hsum5, hsum5, 0x23); + perm2 = _mm256_permute2f128_ps(hsum6, hsum6, 0x23); + perm3 = _mm256_permute2f128_ps(hsum7, hsum7, 0x23); + perm4 = _mm256_permute2f128_ps(hsum8, hsum8, 0x23); + + sum1 = _mm256_add_ps(perm1, hsum5); + sum2 = _mm256_add_ps(perm2, hsum6); + sum3 = _mm256_add_ps(perm3, hsum7); + sum4 = _mm256_add_ps(perm4, hsum8); + + blend1 = _mm256_blend_ps(sum1, sum2, 0xcc); + blend2 = _mm256_blend_ps(sum3, sum4, 0xcc); + + __m256 final_1 = _mm256_blend_ps(blend1, blend2, 0xf0); + + hsum1 = _mm256_hadd_ps(vecs8_1, vecs9_1); + hsum2 = _mm256_hadd_ps(vecs10_1, vecs11_1); + hsum3 = _mm256_hadd_ps(vecs12_1, vecs13_1); + hsum4 = _mm256_hadd_ps(vecs14_1, vecs15_1); + + hsum5 = _mm256_hadd_ps(hsum1, hsum1); + hsum6 = _mm256_hadd_ps(hsum2, hsum2); + hsum7 = _mm256_hadd_ps(hsum3, hsum3); + hsum8 = _mm256_hadd_ps(hsum4, hsum4); + + perm1 = _mm256_permute2f128_ps(hsum5, hsum5, 0x23); + perm2 = _mm256_permute2f128_ps(hsum6, hsum6, 0x23); + perm3 = _mm256_permute2f128_ps(hsum7, hsum7, 0x23); + perm4 = _mm256_permute2f128_ps(hsum8, hsum8, 0x23); + + sum1 = _mm256_add_ps(perm1, hsum5); + sum2 = _mm256_add_ps(perm2, hsum6); + sum3 = _mm256_add_ps(perm3, hsum7); + sum4 = _mm256_add_ps(perm4, hsum8); + + blend1 = _mm256_blend_ps(sum1, sum2, 0xcc); + blend2 = _mm256_blend_ps(sum3, sum4, 0xcc); - final_1 = padd(final_1, _mm256_blend_ps(blend1, blend2, 0xf0)); + final_1 = padd(final_1, _mm256_blend_ps(blend1, blend2, 0xf0)); - __m512 final_output; + __m512 final_output; - EIGEN_INSERT_8f_INTO_16f(final_output, final, final_1); - return final_output; -} + EIGEN_INSERT_8f_INTO_16f(final_output, final, final_1); + return final_output; + } -template<> EIGEN_STRONG_INLINE Packet8d preduxp(const Packet8d* vecs) -{ - Packet4d vecs0_0 = _mm512_extractf64x4_pd(vecs[0], 0); - Packet4d vecs0_1 = _mm512_extractf64x4_pd(vecs[0], 1); + template<> EIGEN_STRONG_INLINE Packet8d preduxp(const Packet8d *vecs) + { + Packet4d vecs0_0 = _mm512_extractf64x4_pd(vecs[0], 0); + Packet4d vecs0_1 = _mm512_extractf64x4_pd(vecs[0], 1); - Packet4d vecs1_0 = _mm512_extractf64x4_pd(vecs[1], 0); - Packet4d vecs1_1 = _mm512_extractf64x4_pd(vecs[1], 1); + Packet4d vecs1_0 = _mm512_extractf64x4_pd(vecs[1], 0); + Packet4d vecs1_1 = _mm512_extractf64x4_pd(vecs[1], 1); - Packet4d vecs2_0 = _mm512_extractf64x4_pd(vecs[2], 0); - Packet4d vecs2_1 = _mm512_extractf64x4_pd(vecs[2], 1); + Packet4d vecs2_0 = _mm512_extractf64x4_pd(vecs[2], 0); + Packet4d vecs2_1 = _mm512_extractf64x4_pd(vecs[2], 1); - Packet4d vecs3_0 = _mm512_extractf64x4_pd(vecs[3], 0); - Packet4d vecs3_1 = _mm512_extractf64x4_pd(vecs[3], 1); + Packet4d vecs3_0 = _mm512_extractf64x4_pd(vecs[3], 0); + Packet4d vecs3_1 = _mm512_extractf64x4_pd(vecs[3], 1); - Packet4d vecs4_0 = _mm512_extractf64x4_pd(vecs[4], 0); - Packet4d vecs4_1 = _mm512_extractf64x4_pd(vecs[4], 1); + Packet4d vecs4_0 = _mm512_extractf64x4_pd(vecs[4], 0); + Packet4d vecs4_1 = _mm512_extractf64x4_pd(vecs[4], 1); - Packet4d vecs5_0 = _mm512_extractf64x4_pd(vecs[5], 0); - Packet4d vecs5_1 = _mm512_extractf64x4_pd(vecs[5], 1); + Packet4d vecs5_0 = _mm512_extractf64x4_pd(vecs[5], 0); + Packet4d vecs5_1 = _mm512_extractf64x4_pd(vecs[5], 1); - Packet4d vecs6_0 = _mm512_extractf64x4_pd(vecs[6], 0); - Packet4d vecs6_1 = _mm512_extractf64x4_pd(vecs[6], 1); + Packet4d vecs6_0 = _mm512_extractf64x4_pd(vecs[6], 0); + Packet4d vecs6_1 = _mm512_extractf64x4_pd(vecs[6], 1); - Packet4d vecs7_0 = _mm512_extractf64x4_pd(vecs[7], 0); - Packet4d vecs7_1 = _mm512_extractf64x4_pd(vecs[7], 1); + Packet4d vecs7_0 = _mm512_extractf64x4_pd(vecs[7], 0); + Packet4d vecs7_1 = _mm512_extractf64x4_pd(vecs[7], 1); - Packet4d tmp0, tmp1; + Packet4d tmp0, tmp1; - tmp0 = _mm256_hadd_pd(vecs0_0, vecs1_0); - tmp0 = _mm256_add_pd(tmp0, _mm256_permute2f128_pd(tmp0, tmp0, 1)); + tmp0 = _mm256_hadd_pd(vecs0_0, vecs1_0); + tmp0 = _mm256_add_pd(tmp0, _mm256_permute2f128_pd(tmp0, tmp0, 1)); - tmp1 = _mm256_hadd_pd(vecs2_0, vecs3_0); - tmp1 = _mm256_add_pd(tmp1, _mm256_permute2f128_pd(tmp1, tmp1, 1)); + tmp1 = _mm256_hadd_pd(vecs2_0, vecs3_0); + tmp1 = _mm256_add_pd(tmp1, _mm256_permute2f128_pd(tmp1, tmp1, 1)); - __m256d final_0 = _mm256_blend_pd(tmp0, tmp1, 0xC); + __m256d final_0 = _mm256_blend_pd(tmp0, tmp1, 0xC); - tmp0 = _mm256_hadd_pd(vecs0_1, vecs1_1); - tmp0 = _mm256_add_pd(tmp0, _mm256_permute2f128_pd(tmp0, tmp0, 1)); + tmp0 = _mm256_hadd_pd(vecs0_1, vecs1_1); + tmp0 = _mm256_add_pd(tmp0, _mm256_permute2f128_pd(tmp0, tmp0, 1)); - tmp1 = _mm256_hadd_pd(vecs2_1, vecs3_1); - tmp1 = _mm256_add_pd(tmp1, _mm256_permute2f128_pd(tmp1, tmp1, 1)); + tmp1 = _mm256_hadd_pd(vecs2_1, vecs3_1); + tmp1 = _mm256_add_pd(tmp1, _mm256_permute2f128_pd(tmp1, tmp1, 1)); - final_0 = padd(final_0, _mm256_blend_pd(tmp0, tmp1, 0xC)); + final_0 = padd(final_0, _mm256_blend_pd(tmp0, tmp1, 0xC)); - tmp0 = _mm256_hadd_pd(vecs4_0, vecs5_0); - tmp0 = _mm256_add_pd(tmp0, _mm256_permute2f128_pd(tmp0, tmp0, 1)); + tmp0 = _mm256_hadd_pd(vecs4_0, vecs5_0); + tmp0 = _mm256_add_pd(tmp0, _mm256_permute2f128_pd(tmp0, tmp0, 1)); - tmp1 = _mm256_hadd_pd(vecs6_0, vecs7_0); - tmp1 = _mm256_add_pd(tmp1, _mm256_permute2f128_pd(tmp1, tmp1, 1)); + tmp1 = _mm256_hadd_pd(vecs6_0, vecs7_0); + tmp1 = _mm256_add_pd(tmp1, _mm256_permute2f128_pd(tmp1, tmp1, 1)); - __m256d final_1 = _mm256_blend_pd(tmp0, tmp1, 0xC); + __m256d final_1 = _mm256_blend_pd(tmp0, tmp1, 0xC); - tmp0 = _mm256_hadd_pd(vecs4_1, vecs5_1); - tmp0 = _mm256_add_pd(tmp0, _mm256_permute2f128_pd(tmp0, tmp0, 1)); + tmp0 = _mm256_hadd_pd(vecs4_1, vecs5_1); + tmp0 = _mm256_add_pd(tmp0, _mm256_permute2f128_pd(tmp0, tmp0, 1)); - tmp1 = _mm256_hadd_pd(vecs6_1, vecs7_1); - tmp1 = _mm256_add_pd(tmp1, _mm256_permute2f128_pd(tmp1, tmp1, 1)); + tmp1 = _mm256_hadd_pd(vecs6_1, vecs7_1); + tmp1 = _mm256_add_pd(tmp1, _mm256_permute2f128_pd(tmp1, tmp1, 1)); - final_1 = padd(final_1, _mm256_blend_pd(tmp0, tmp1, 0xC)); + final_1 = padd(final_1, _mm256_blend_pd(tmp0, tmp1, 0xC)); - __m512d final_output = _mm512_insertf64x4(final_output, final_0, 0); + __m512d final_output = _mm512_insertf64x4(final_output, final_0, 0); - return _mm512_insertf64x4(final_output, final_1, 1); -} + return _mm512_insertf64x4(final_output, final_1, 1); + } -template <> -EIGEN_STRONG_INLINE float predux(const Packet16f& a) { - //#ifdef EIGEN_VECTORIZE_AVX512DQ + template<> EIGEN_STRONG_INLINE float predux(const Packet16f &a) + { + // #ifdef EIGEN_VECTORIZE_AVX512DQ #if 0 Packet8f lane0 = _mm512_extractf32x8_ps(a, 0); Packet8f lane1 = _mm512_extractf32x8_ps(a, 1); @@ -885,52 +825,52 @@ EIGEN_STRONG_INLINE float predux(const Packet16f& a) { tmp0 = _mm256_hadd_ps(tmp0, tmp0); return pfirst(_mm256_hadd_ps(tmp0, tmp0)); #else - Packet4f lane0 = _mm512_extractf32x4_ps(a, 0); - Packet4f lane1 = _mm512_extractf32x4_ps(a, 1); - Packet4f lane2 = _mm512_extractf32x4_ps(a, 2); - Packet4f lane3 = _mm512_extractf32x4_ps(a, 3); - Packet4f sum = padd(padd(lane0, lane1), padd(lane2, lane3)); - sum = _mm_hadd_ps(sum, sum); - sum = _mm_hadd_ps(sum, _mm_permute_ps(sum, 1)); - return pfirst(sum); + Packet4f lane0 = _mm512_extractf32x4_ps(a, 0); + Packet4f lane1 = _mm512_extractf32x4_ps(a, 1); + Packet4f lane2 = _mm512_extractf32x4_ps(a, 2); + Packet4f lane3 = _mm512_extractf32x4_ps(a, 3); + Packet4f sum = padd(padd(lane0, lane1), padd(lane2, lane3)); + sum = _mm_hadd_ps(sum, sum); + sum = _mm_hadd_ps(sum, _mm_permute_ps(sum, 1)); + return pfirst(sum); #endif -} -template <> -EIGEN_STRONG_INLINE double predux(const Packet8d& a) { - Packet4d lane0 = _mm512_extractf64x4_pd(a, 0); - Packet4d lane1 = _mm512_extractf64x4_pd(a, 1); - Packet4d sum = padd(lane0, lane1); - Packet4d tmp0 = _mm256_hadd_pd(sum, _mm256_permute2f128_pd(sum, sum, 1)); - return pfirst(_mm256_hadd_pd(tmp0, tmp0)); -} - -template <> -EIGEN_STRONG_INLINE Packet8f predux_downto4(const Packet16f& a) { + } + template<> EIGEN_STRONG_INLINE double predux(const Packet8d &a) + { + Packet4d lane0 = _mm512_extractf64x4_pd(a, 0); + Packet4d lane1 = _mm512_extractf64x4_pd(a, 1); + Packet4d sum = padd(lane0, lane1); + Packet4d tmp0 = _mm256_hadd_pd(sum, _mm256_permute2f128_pd(sum, sum, 1)); + return pfirst(_mm256_hadd_pd(tmp0, tmp0)); + } + + template<> EIGEN_STRONG_INLINE Packet8f predux_downto4(const Packet16f &a) + { #ifdef EIGEN_VECTORIZE_AVX512DQ - Packet8f lane0 = _mm512_extractf32x8_ps(a, 0); - Packet8f lane1 = _mm512_extractf32x8_ps(a, 1); - return padd(lane0, lane1); + Packet8f lane0 = _mm512_extractf32x8_ps(a, 0); + Packet8f lane1 = _mm512_extractf32x8_ps(a, 1); + return padd(lane0, lane1); #else - Packet4f lane0 = _mm512_extractf32x4_ps(a, 0); - Packet4f lane1 = _mm512_extractf32x4_ps(a, 1); - Packet4f lane2 = _mm512_extractf32x4_ps(a, 2); - Packet4f lane3 = _mm512_extractf32x4_ps(a, 3); - Packet4f sum0 = padd(lane0, lane2); - Packet4f sum1 = padd(lane1, lane3); - return _mm256_insertf128_ps(_mm256_castps128_ps256(sum0), sum1, 1); + Packet4f lane0 = _mm512_extractf32x4_ps(a, 0); + Packet4f lane1 = _mm512_extractf32x4_ps(a, 1); + Packet4f lane2 = _mm512_extractf32x4_ps(a, 2); + Packet4f lane3 = _mm512_extractf32x4_ps(a, 3); + Packet4f sum0 = padd(lane0, lane2); + Packet4f sum1 = padd(lane1, lane3); + return _mm256_insertf128_ps(_mm256_castps128_ps256(sum0), sum1, 1); #endif -} -template <> -EIGEN_STRONG_INLINE Packet4d predux_downto4(const Packet8d& a) { - Packet4d lane0 = _mm512_extractf64x4_pd(a, 0); - Packet4d lane1 = _mm512_extractf64x4_pd(a, 1); - Packet4d res = padd(lane0, lane1); - return res; -} - -template <> -EIGEN_STRONG_INLINE float predux_mul(const Packet16f& a) { -//#ifdef EIGEN_VECTORIZE_AVX512DQ + } + template<> EIGEN_STRONG_INLINE Packet4d predux_downto4(const Packet8d &a) + { + Packet4d lane0 = _mm512_extractf64x4_pd(a, 0); + Packet4d lane1 = _mm512_extractf64x4_pd(a, 1); + Packet4d res = padd(lane0, lane1); + return res; + } + + template<> EIGEN_STRONG_INLINE float predux_mul(const Packet16f &a) + { +// #ifdef EIGEN_VECTORIZE_AVX512DQ #if 0 Packet8f lane0 = _mm512_extractf32x8_ps(a, 0); Packet8f lane1 = _mm512_extractf32x8_ps(a, 1); @@ -939,261 +879,312 @@ EIGEN_STRONG_INLINE float predux_mul(const Packet16f& a) { res = pmul(res, _mm_permute_ps(res, _MM_SHUFFLE(0, 0, 3, 2))); return pfirst(pmul(res, _mm_permute_ps(res, _MM_SHUFFLE(0, 0, 0, 1)))); #else - Packet4f lane0 = _mm512_extractf32x4_ps(a, 0); - Packet4f lane1 = _mm512_extractf32x4_ps(a, 1); - Packet4f lane2 = _mm512_extractf32x4_ps(a, 2); - Packet4f lane3 = _mm512_extractf32x4_ps(a, 3); - Packet4f res = pmul(pmul(lane0, lane1), pmul(lane2, lane3)); - res = pmul(res, _mm_permute_ps(res, _MM_SHUFFLE(0, 0, 3, 2))); - return pfirst(pmul(res, _mm_permute_ps(res, _MM_SHUFFLE(0, 0, 0, 1)))); + Packet4f lane0 = _mm512_extractf32x4_ps(a, 0); + Packet4f lane1 = _mm512_extractf32x4_ps(a, 1); + Packet4f lane2 = _mm512_extractf32x4_ps(a, 2); + Packet4f lane3 = _mm512_extractf32x4_ps(a, 3); + Packet4f res = pmul(pmul(lane0, lane1), pmul(lane2, lane3)); + res = pmul(res, _mm_permute_ps(res, _MM_SHUFFLE(0, 0, 3, 2))); + return pfirst(pmul(res, _mm_permute_ps(res, _MM_SHUFFLE(0, 0, 0, 1)))); #endif -} -template <> -EIGEN_STRONG_INLINE double predux_mul(const Packet8d& a) { - Packet4d lane0 = _mm512_extractf64x4_pd(a, 0); - Packet4d lane1 = _mm512_extractf64x4_pd(a, 1); - Packet4d res = pmul(lane0, lane1); - res = pmul(res, _mm256_permute2f128_pd(res, res, 1)); - return pfirst(pmul(res, _mm256_shuffle_pd(res, res, 1))); -} - -template <> -EIGEN_STRONG_INLINE float predux_min(const Packet16f& a) { - Packet4f lane0 = _mm512_extractf32x4_ps(a, 0); - Packet4f lane1 = _mm512_extractf32x4_ps(a, 1); - Packet4f lane2 = _mm512_extractf32x4_ps(a, 2); - Packet4f lane3 = _mm512_extractf32x4_ps(a, 3); - Packet4f res = _mm_min_ps(_mm_min_ps(lane0, lane1), _mm_min_ps(lane2, lane3)); - res = _mm_min_ps(res, _mm_permute_ps(res, _MM_SHUFFLE(0, 0, 3, 2))); - return pfirst(_mm_min_ps(res, _mm_permute_ps(res, _MM_SHUFFLE(0, 0, 0, 1)))); -} -template <> -EIGEN_STRONG_INLINE double predux_min(const Packet8d& a) { - Packet4d lane0 = _mm512_extractf64x4_pd(a, 0); - Packet4d lane1 = _mm512_extractf64x4_pd(a, 1); - Packet4d res = _mm256_min_pd(lane0, lane1); - res = _mm256_min_pd(res, _mm256_permute2f128_pd(res, res, 1)); - return pfirst(_mm256_min_pd(res, _mm256_shuffle_pd(res, res, 1))); -} - -template <> -EIGEN_STRONG_INLINE float predux_max(const Packet16f& a) { - Packet4f lane0 = _mm512_extractf32x4_ps(a, 0); - Packet4f lane1 = _mm512_extractf32x4_ps(a, 1); - Packet4f lane2 = _mm512_extractf32x4_ps(a, 2); - Packet4f lane3 = _mm512_extractf32x4_ps(a, 3); - Packet4f res = _mm_max_ps(_mm_max_ps(lane0, lane1), _mm_max_ps(lane2, lane3)); - res = _mm_max_ps(res, _mm_permute_ps(res, _MM_SHUFFLE(0, 0, 3, 2))); - return pfirst(_mm_max_ps(res, _mm_permute_ps(res, _MM_SHUFFLE(0, 0, 0, 1)))); -} -template <> -EIGEN_STRONG_INLINE double predux_max(const Packet8d& a) { - Packet4d lane0 = _mm512_extractf64x4_pd(a, 0); - Packet4d lane1 = _mm512_extractf64x4_pd(a, 1); - Packet4d res = _mm256_max_pd(lane0, lane1); - res = _mm256_max_pd(res, _mm256_permute2f128_pd(res, res, 1)); - return pfirst(_mm256_max_pd(res, _mm256_shuffle_pd(res, res, 1))); -} - -template -struct palign_impl { - static EIGEN_STRONG_INLINE void run(Packet16f& first, - const Packet16f& second) { - if (Offset != 0) { - __m512i first_idx = _mm512_set_epi32( - Offset + 15, Offset + 14, Offset + 13, Offset + 12, Offset + 11, - Offset + 10, Offset + 9, Offset + 8, Offset + 7, Offset + 6, - Offset + 5, Offset + 4, Offset + 3, Offset + 2, Offset + 1, Offset); - - __m512i second_idx = - _mm512_set_epi32(Offset - 1, Offset - 2, Offset - 3, Offset - 4, - Offset - 5, Offset - 6, Offset - 7, Offset - 8, - Offset - 9, Offset - 10, Offset - 11, Offset - 12, - Offset - 13, Offset - 14, Offset - 15, Offset - 16); - - unsigned short mask = 0xFFFF; - mask <<= (16 - Offset); - - first = _mm512_permutexvar_ps(first_idx, first); - Packet16f tmp = _mm512_permutexvar_ps(second_idx, second); - first = _mm512_mask_blend_ps(mask, first, tmp); - } } -}; -template -struct palign_impl { - static EIGEN_STRONG_INLINE void run(Packet8d& first, const Packet8d& second) { - if (Offset != 0) { - __m512i first_idx = _mm512_set_epi32( - 0, Offset + 7, 0, Offset + 6, 0, Offset + 5, 0, Offset + 4, 0, - Offset + 3, 0, Offset + 2, 0, Offset + 1, 0, Offset); - - __m512i second_idx = _mm512_set_epi32( - 0, Offset - 1, 0, Offset - 2, 0, Offset - 3, 0, Offset - 4, 0, - Offset - 5, 0, Offset - 6, 0, Offset - 7, 0, Offset - 8); - - unsigned char mask = 0xFF; - mask <<= (8 - Offset); - - first = _mm512_permutexvar_pd(first_idx, first); - Packet8d tmp = _mm512_permutexvar_pd(second_idx, second); - first = _mm512_mask_blend_pd(mask, first, tmp); - } + template<> EIGEN_STRONG_INLINE double predux_mul(const Packet8d &a) + { + Packet4d lane0 = _mm512_extractf64x4_pd(a, 0); + Packet4d lane1 = _mm512_extractf64x4_pd(a, 1); + Packet4d res = pmul(lane0, lane1); + res = pmul(res, _mm256_permute2f128_pd(res, res, 1)); + return pfirst(pmul(res, _mm256_shuffle_pd(res, res, 1))); } -}; + + template<> EIGEN_STRONG_INLINE float predux_min(const Packet16f &a) + { + Packet4f lane0 = _mm512_extractf32x4_ps(a, 0); + Packet4f lane1 = _mm512_extractf32x4_ps(a, 1); + Packet4f lane2 = _mm512_extractf32x4_ps(a, 2); + Packet4f lane3 = _mm512_extractf32x4_ps(a, 3); + Packet4f res = _mm_min_ps(_mm_min_ps(lane0, lane1), _mm_min_ps(lane2, lane3)); + res = _mm_min_ps(res, _mm_permute_ps(res, _MM_SHUFFLE(0, 0, 3, 2))); + return pfirst(_mm_min_ps(res, _mm_permute_ps(res, _MM_SHUFFLE(0, 0, 0, 1)))); + } + template<> EIGEN_STRONG_INLINE double predux_min(const Packet8d &a) + { + Packet4d lane0 = _mm512_extractf64x4_pd(a, 0); + Packet4d lane1 = _mm512_extractf64x4_pd(a, 1); + Packet4d res = _mm256_min_pd(lane0, lane1); + res = _mm256_min_pd(res, _mm256_permute2f128_pd(res, res, 1)); + return pfirst(_mm256_min_pd(res, _mm256_shuffle_pd(res, res, 1))); + } + + template<> EIGEN_STRONG_INLINE float predux_max(const Packet16f &a) + { + Packet4f lane0 = _mm512_extractf32x4_ps(a, 0); + Packet4f lane1 = _mm512_extractf32x4_ps(a, 1); + Packet4f lane2 = _mm512_extractf32x4_ps(a, 2); + Packet4f lane3 = _mm512_extractf32x4_ps(a, 3); + Packet4f res = _mm_max_ps(_mm_max_ps(lane0, lane1), _mm_max_ps(lane2, lane3)); + res = _mm_max_ps(res, _mm_permute_ps(res, _MM_SHUFFLE(0, 0, 3, 2))); + return pfirst(_mm_max_ps(res, _mm_permute_ps(res, _MM_SHUFFLE(0, 0, 0, 1)))); + } + template<> EIGEN_STRONG_INLINE double predux_max(const Packet8d &a) + { + Packet4d lane0 = _mm512_extractf64x4_pd(a, 0); + Packet4d lane1 = _mm512_extractf64x4_pd(a, 1); + Packet4d res = _mm256_max_pd(lane0, lane1); + res = _mm256_max_pd(res, _mm256_permute2f128_pd(res, res, 1)); + return pfirst(_mm256_max_pd(res, _mm256_shuffle_pd(res, res, 1))); + } + + template struct palign_impl + { + static EIGEN_STRONG_INLINE void run(Packet16f &first, const Packet16f &second) + { + if (Offset != 0) { + __m512i first_idx = _mm512_set_epi32(Offset + 15, + Offset + 14, + Offset + 13, + Offset + 12, + Offset + 11, + Offset + 10, + Offset + 9, + Offset + 8, + Offset + 7, + Offset + 6, + Offset + 5, + Offset + 4, + Offset + 3, + Offset + 2, + Offset + 1, + Offset); + + __m512i second_idx = _mm512_set_epi32(Offset - 1, + Offset - 2, + Offset - 3, + Offset - 4, + Offset - 5, + Offset - 6, + Offset - 7, + Offset - 8, + Offset - 9, + Offset - 10, + Offset - 11, + Offset - 12, + Offset - 13, + Offset - 14, + Offset - 15, + Offset - 16); + + unsigned short mask = 0xFFFF; + mask <<= (16 - Offset); + + first = _mm512_permutexvar_ps(first_idx, first); + Packet16f tmp = _mm512_permutexvar_ps(second_idx, second); + first = _mm512_mask_blend_ps(mask, first, tmp); + } + } + }; + template struct palign_impl + { + static EIGEN_STRONG_INLINE void run(Packet8d &first, const Packet8d &second) + { + if (Offset != 0) { + __m512i first_idx = _mm512_set_epi32(0, + Offset + 7, + 0, + Offset + 6, + 0, + Offset + 5, + 0, + Offset + 4, + 0, + Offset + 3, + 0, + Offset + 2, + 0, + Offset + 1, + 0, + Offset); + + __m512i second_idx = _mm512_set_epi32(0, + Offset - 1, + 0, + Offset - 2, + 0, + Offset - 3, + 0, + Offset - 4, + 0, + Offset - 5, + 0, + Offset - 6, + 0, + Offset - 7, + 0, + Offset - 8); + + unsigned char mask = 0xFF; + mask <<= (8 - Offset); + + first = _mm512_permutexvar_pd(first_idx, first); + Packet8d tmp = _mm512_permutexvar_pd(second_idx, second); + first = _mm512_mask_blend_pd(mask, first, tmp); + } + } + }; #define PACK_OUTPUT(OUTPUT, INPUT, INDEX, STRIDE) \ EIGEN_INSERT_8f_INTO_16f(OUTPUT[INDEX], INPUT[INDEX], INPUT[INDEX + STRIDE]); -EIGEN_DEVICE_FUNC inline void ptranspose(PacketBlock& kernel) { - __m512 T0 = _mm512_unpacklo_ps(kernel.packet[0], kernel.packet[1]); - __m512 T1 = _mm512_unpackhi_ps(kernel.packet[0], kernel.packet[1]); - __m512 T2 = _mm512_unpacklo_ps(kernel.packet[2], kernel.packet[3]); - __m512 T3 = _mm512_unpackhi_ps(kernel.packet[2], kernel.packet[3]); - __m512 T4 = _mm512_unpacklo_ps(kernel.packet[4], kernel.packet[5]); - __m512 T5 = _mm512_unpackhi_ps(kernel.packet[4], kernel.packet[5]); - __m512 T6 = _mm512_unpacklo_ps(kernel.packet[6], kernel.packet[7]); - __m512 T7 = _mm512_unpackhi_ps(kernel.packet[6], kernel.packet[7]); - __m512 T8 = _mm512_unpacklo_ps(kernel.packet[8], kernel.packet[9]); - __m512 T9 = _mm512_unpackhi_ps(kernel.packet[8], kernel.packet[9]); - __m512 T10 = _mm512_unpacklo_ps(kernel.packet[10], kernel.packet[11]); - __m512 T11 = _mm512_unpackhi_ps(kernel.packet[10], kernel.packet[11]); - __m512 T12 = _mm512_unpacklo_ps(kernel.packet[12], kernel.packet[13]); - __m512 T13 = _mm512_unpackhi_ps(kernel.packet[12], kernel.packet[13]); - __m512 T14 = _mm512_unpacklo_ps(kernel.packet[14], kernel.packet[15]); - __m512 T15 = _mm512_unpackhi_ps(kernel.packet[14], kernel.packet[15]); - __m512 S0 = _mm512_shuffle_ps(T0, T2, _MM_SHUFFLE(1, 0, 1, 0)); - __m512 S1 = _mm512_shuffle_ps(T0, T2, _MM_SHUFFLE(3, 2, 3, 2)); - __m512 S2 = _mm512_shuffle_ps(T1, T3, _MM_SHUFFLE(1, 0, 1, 0)); - __m512 S3 = _mm512_shuffle_ps(T1, T3, _MM_SHUFFLE(3, 2, 3, 2)); - __m512 S4 = _mm512_shuffle_ps(T4, T6, _MM_SHUFFLE(1, 0, 1, 0)); - __m512 S5 = _mm512_shuffle_ps(T4, T6, _MM_SHUFFLE(3, 2, 3, 2)); - __m512 S6 = _mm512_shuffle_ps(T5, T7, _MM_SHUFFLE(1, 0, 1, 0)); - __m512 S7 = _mm512_shuffle_ps(T5, T7, _MM_SHUFFLE(3, 2, 3, 2)); - __m512 S8 = _mm512_shuffle_ps(T8, T10, _MM_SHUFFLE(1, 0, 1, 0)); - __m512 S9 = _mm512_shuffle_ps(T8, T10, _MM_SHUFFLE(3, 2, 3, 2)); - __m512 S10 = _mm512_shuffle_ps(T9, T11, _MM_SHUFFLE(1, 0, 1, 0)); - __m512 S11 = _mm512_shuffle_ps(T9, T11, _MM_SHUFFLE(3, 2, 3, 2)); - __m512 S12 = _mm512_shuffle_ps(T12, T14, _MM_SHUFFLE(1, 0, 1, 0)); - __m512 S13 = _mm512_shuffle_ps(T12, T14, _MM_SHUFFLE(3, 2, 3, 2)); - __m512 S14 = _mm512_shuffle_ps(T13, T15, _MM_SHUFFLE(1, 0, 1, 0)); - __m512 S15 = _mm512_shuffle_ps(T13, T15, _MM_SHUFFLE(3, 2, 3, 2)); - - EIGEN_EXTRACT_8f_FROM_16f(S0, S0); - EIGEN_EXTRACT_8f_FROM_16f(S1, S1); - EIGEN_EXTRACT_8f_FROM_16f(S2, S2); - EIGEN_EXTRACT_8f_FROM_16f(S3, S3); - EIGEN_EXTRACT_8f_FROM_16f(S4, S4); - EIGEN_EXTRACT_8f_FROM_16f(S5, S5); - EIGEN_EXTRACT_8f_FROM_16f(S6, S6); - EIGEN_EXTRACT_8f_FROM_16f(S7, S7); - EIGEN_EXTRACT_8f_FROM_16f(S8, S8); - EIGEN_EXTRACT_8f_FROM_16f(S9, S9); - EIGEN_EXTRACT_8f_FROM_16f(S10, S10); - EIGEN_EXTRACT_8f_FROM_16f(S11, S11); - EIGEN_EXTRACT_8f_FROM_16f(S12, S12); - EIGEN_EXTRACT_8f_FROM_16f(S13, S13); - EIGEN_EXTRACT_8f_FROM_16f(S14, S14); - EIGEN_EXTRACT_8f_FROM_16f(S15, S15); - - PacketBlock tmp; - - tmp.packet[0] = _mm256_permute2f128_ps(S0_0, S4_0, 0x20); - tmp.packet[1] = _mm256_permute2f128_ps(S1_0, S5_0, 0x20); - tmp.packet[2] = _mm256_permute2f128_ps(S2_0, S6_0, 0x20); - tmp.packet[3] = _mm256_permute2f128_ps(S3_0, S7_0, 0x20); - tmp.packet[4] = _mm256_permute2f128_ps(S0_0, S4_0, 0x31); - tmp.packet[5] = _mm256_permute2f128_ps(S1_0, S5_0, 0x31); - tmp.packet[6] = _mm256_permute2f128_ps(S2_0, S6_0, 0x31); - tmp.packet[7] = _mm256_permute2f128_ps(S3_0, S7_0, 0x31); - - tmp.packet[8] = _mm256_permute2f128_ps(S0_1, S4_1, 0x20); - tmp.packet[9] = _mm256_permute2f128_ps(S1_1, S5_1, 0x20); - tmp.packet[10] = _mm256_permute2f128_ps(S2_1, S6_1, 0x20); - tmp.packet[11] = _mm256_permute2f128_ps(S3_1, S7_1, 0x20); - tmp.packet[12] = _mm256_permute2f128_ps(S0_1, S4_1, 0x31); - tmp.packet[13] = _mm256_permute2f128_ps(S1_1, S5_1, 0x31); - tmp.packet[14] = _mm256_permute2f128_ps(S2_1, S6_1, 0x31); - tmp.packet[15] = _mm256_permute2f128_ps(S3_1, S7_1, 0x31); - - // Second set of _m256 outputs - tmp.packet[16] = _mm256_permute2f128_ps(S8_0, S12_0, 0x20); - tmp.packet[17] = _mm256_permute2f128_ps(S9_0, S13_0, 0x20); - tmp.packet[18] = _mm256_permute2f128_ps(S10_0, S14_0, 0x20); - tmp.packet[19] = _mm256_permute2f128_ps(S11_0, S15_0, 0x20); - tmp.packet[20] = _mm256_permute2f128_ps(S8_0, S12_0, 0x31); - tmp.packet[21] = _mm256_permute2f128_ps(S9_0, S13_0, 0x31); - tmp.packet[22] = _mm256_permute2f128_ps(S10_0, S14_0, 0x31); - tmp.packet[23] = _mm256_permute2f128_ps(S11_0, S15_0, 0x31); - - tmp.packet[24] = _mm256_permute2f128_ps(S8_1, S12_1, 0x20); - tmp.packet[25] = _mm256_permute2f128_ps(S9_1, S13_1, 0x20); - tmp.packet[26] = _mm256_permute2f128_ps(S10_1, S14_1, 0x20); - tmp.packet[27] = _mm256_permute2f128_ps(S11_1, S15_1, 0x20); - tmp.packet[28] = _mm256_permute2f128_ps(S8_1, S12_1, 0x31); - tmp.packet[29] = _mm256_permute2f128_ps(S9_1, S13_1, 0x31); - tmp.packet[30] = _mm256_permute2f128_ps(S10_1, S14_1, 0x31); - tmp.packet[31] = _mm256_permute2f128_ps(S11_1, S15_1, 0x31); - - // Pack them into the output - PACK_OUTPUT(kernel.packet, tmp.packet, 0, 16); - PACK_OUTPUT(kernel.packet, tmp.packet, 1, 16); - PACK_OUTPUT(kernel.packet, tmp.packet, 2, 16); - PACK_OUTPUT(kernel.packet, tmp.packet, 3, 16); - - PACK_OUTPUT(kernel.packet, tmp.packet, 4, 16); - PACK_OUTPUT(kernel.packet, tmp.packet, 5, 16); - PACK_OUTPUT(kernel.packet, tmp.packet, 6, 16); - PACK_OUTPUT(kernel.packet, tmp.packet, 7, 16); - - PACK_OUTPUT(kernel.packet, tmp.packet, 8, 16); - PACK_OUTPUT(kernel.packet, tmp.packet, 9, 16); - PACK_OUTPUT(kernel.packet, tmp.packet, 10, 16); - PACK_OUTPUT(kernel.packet, tmp.packet, 11, 16); - - PACK_OUTPUT(kernel.packet, tmp.packet, 12, 16); - PACK_OUTPUT(kernel.packet, tmp.packet, 13, 16); - PACK_OUTPUT(kernel.packet, tmp.packet, 14, 16); - PACK_OUTPUT(kernel.packet, tmp.packet, 15, 16); -} -#define PACK_OUTPUT_2(OUTPUT, INPUT, INDEX, STRIDE) \ - EIGEN_INSERT_8f_INTO_16f(OUTPUT[INDEX], INPUT[2 * INDEX], \ - INPUT[2 * INDEX + STRIDE]); - -EIGEN_DEVICE_FUNC inline void ptranspose(PacketBlock& kernel) { - __m512 T0 = _mm512_unpacklo_ps(kernel.packet[0], kernel.packet[1]); - __m512 T1 = _mm512_unpackhi_ps(kernel.packet[0], kernel.packet[1]); - __m512 T2 = _mm512_unpacklo_ps(kernel.packet[2], kernel.packet[3]); - __m512 T3 = _mm512_unpackhi_ps(kernel.packet[2], kernel.packet[3]); - - __m512 S0 = _mm512_shuffle_ps(T0, T2, _MM_SHUFFLE(1, 0, 1, 0)); - __m512 S1 = _mm512_shuffle_ps(T0, T2, _MM_SHUFFLE(3, 2, 3, 2)); - __m512 S2 = _mm512_shuffle_ps(T1, T3, _MM_SHUFFLE(1, 0, 1, 0)); - __m512 S3 = _mm512_shuffle_ps(T1, T3, _MM_SHUFFLE(3, 2, 3, 2)); - - EIGEN_EXTRACT_8f_FROM_16f(S0, S0); - EIGEN_EXTRACT_8f_FROM_16f(S1, S1); - EIGEN_EXTRACT_8f_FROM_16f(S2, S2); - EIGEN_EXTRACT_8f_FROM_16f(S3, S3); - - PacketBlock tmp; - - tmp.packet[0] = _mm256_permute2f128_ps(S0_0, S1_0, 0x20); - tmp.packet[1] = _mm256_permute2f128_ps(S2_0, S3_0, 0x20); - tmp.packet[2] = _mm256_permute2f128_ps(S0_0, S1_0, 0x31); - tmp.packet[3] = _mm256_permute2f128_ps(S2_0, S3_0, 0x31); - - tmp.packet[4] = _mm256_permute2f128_ps(S0_1, S1_1, 0x20); - tmp.packet[5] = _mm256_permute2f128_ps(S2_1, S3_1, 0x20); - tmp.packet[6] = _mm256_permute2f128_ps(S0_1, S1_1, 0x31); - tmp.packet[7] = _mm256_permute2f128_ps(S2_1, S3_1, 0x31); - - PACK_OUTPUT_2(kernel.packet, tmp.packet, 0, 1); - PACK_OUTPUT_2(kernel.packet, tmp.packet, 1, 1); - PACK_OUTPUT_2(kernel.packet, tmp.packet, 2, 1); - PACK_OUTPUT_2(kernel.packet, tmp.packet, 3, 1); -} + EIGEN_DEVICE_FUNC inline void ptranspose(PacketBlock &kernel) + { + __m512 T0 = _mm512_unpacklo_ps(kernel.packet[0], kernel.packet[1]); + __m512 T1 = _mm512_unpackhi_ps(kernel.packet[0], kernel.packet[1]); + __m512 T2 = _mm512_unpacklo_ps(kernel.packet[2], kernel.packet[3]); + __m512 T3 = _mm512_unpackhi_ps(kernel.packet[2], kernel.packet[3]); + __m512 T4 = _mm512_unpacklo_ps(kernel.packet[4], kernel.packet[5]); + __m512 T5 = _mm512_unpackhi_ps(kernel.packet[4], kernel.packet[5]); + __m512 T6 = _mm512_unpacklo_ps(kernel.packet[6], kernel.packet[7]); + __m512 T7 = _mm512_unpackhi_ps(kernel.packet[6], kernel.packet[7]); + __m512 T8 = _mm512_unpacklo_ps(kernel.packet[8], kernel.packet[9]); + __m512 T9 = _mm512_unpackhi_ps(kernel.packet[8], kernel.packet[9]); + __m512 T10 = _mm512_unpacklo_ps(kernel.packet[10], kernel.packet[11]); + __m512 T11 = _mm512_unpackhi_ps(kernel.packet[10], kernel.packet[11]); + __m512 T12 = _mm512_unpacklo_ps(kernel.packet[12], kernel.packet[13]); + __m512 T13 = _mm512_unpackhi_ps(kernel.packet[12], kernel.packet[13]); + __m512 T14 = _mm512_unpacklo_ps(kernel.packet[14], kernel.packet[15]); + __m512 T15 = _mm512_unpackhi_ps(kernel.packet[14], kernel.packet[15]); + __m512 S0 = _mm512_shuffle_ps(T0, T2, _MM_SHUFFLE(1, 0, 1, 0)); + __m512 S1 = _mm512_shuffle_ps(T0, T2, _MM_SHUFFLE(3, 2, 3, 2)); + __m512 S2 = _mm512_shuffle_ps(T1, T3, _MM_SHUFFLE(1, 0, 1, 0)); + __m512 S3 = _mm512_shuffle_ps(T1, T3, _MM_SHUFFLE(3, 2, 3, 2)); + __m512 S4 = _mm512_shuffle_ps(T4, T6, _MM_SHUFFLE(1, 0, 1, 0)); + __m512 S5 = _mm512_shuffle_ps(T4, T6, _MM_SHUFFLE(3, 2, 3, 2)); + __m512 S6 = _mm512_shuffle_ps(T5, T7, _MM_SHUFFLE(1, 0, 1, 0)); + __m512 S7 = _mm512_shuffle_ps(T5, T7, _MM_SHUFFLE(3, 2, 3, 2)); + __m512 S8 = _mm512_shuffle_ps(T8, T10, _MM_SHUFFLE(1, 0, 1, 0)); + __m512 S9 = _mm512_shuffle_ps(T8, T10, _MM_SHUFFLE(3, 2, 3, 2)); + __m512 S10 = _mm512_shuffle_ps(T9, T11, _MM_SHUFFLE(1, 0, 1, 0)); + __m512 S11 = _mm512_shuffle_ps(T9, T11, _MM_SHUFFLE(3, 2, 3, 2)); + __m512 S12 = _mm512_shuffle_ps(T12, T14, _MM_SHUFFLE(1, 0, 1, 0)); + __m512 S13 = _mm512_shuffle_ps(T12, T14, _MM_SHUFFLE(3, 2, 3, 2)); + __m512 S14 = _mm512_shuffle_ps(T13, T15, _MM_SHUFFLE(1, 0, 1, 0)); + __m512 S15 = _mm512_shuffle_ps(T13, T15, _MM_SHUFFLE(3, 2, 3, 2)); + + EIGEN_EXTRACT_8f_FROM_16f(S0, S0); + EIGEN_EXTRACT_8f_FROM_16f(S1, S1); + EIGEN_EXTRACT_8f_FROM_16f(S2, S2); + EIGEN_EXTRACT_8f_FROM_16f(S3, S3); + EIGEN_EXTRACT_8f_FROM_16f(S4, S4); + EIGEN_EXTRACT_8f_FROM_16f(S5, S5); + EIGEN_EXTRACT_8f_FROM_16f(S6, S6); + EIGEN_EXTRACT_8f_FROM_16f(S7, S7); + EIGEN_EXTRACT_8f_FROM_16f(S8, S8); + EIGEN_EXTRACT_8f_FROM_16f(S9, S9); + EIGEN_EXTRACT_8f_FROM_16f(S10, S10); + EIGEN_EXTRACT_8f_FROM_16f(S11, S11); + EIGEN_EXTRACT_8f_FROM_16f(S12, S12); + EIGEN_EXTRACT_8f_FROM_16f(S13, S13); + EIGEN_EXTRACT_8f_FROM_16f(S14, S14); + EIGEN_EXTRACT_8f_FROM_16f(S15, S15); + + PacketBlock tmp; + + tmp.packet[0] = _mm256_permute2f128_ps(S0_0, S4_0, 0x20); + tmp.packet[1] = _mm256_permute2f128_ps(S1_0, S5_0, 0x20); + tmp.packet[2] = _mm256_permute2f128_ps(S2_0, S6_0, 0x20); + tmp.packet[3] = _mm256_permute2f128_ps(S3_0, S7_0, 0x20); + tmp.packet[4] = _mm256_permute2f128_ps(S0_0, S4_0, 0x31); + tmp.packet[5] = _mm256_permute2f128_ps(S1_0, S5_0, 0x31); + tmp.packet[6] = _mm256_permute2f128_ps(S2_0, S6_0, 0x31); + tmp.packet[7] = _mm256_permute2f128_ps(S3_0, S7_0, 0x31); + + tmp.packet[8] = _mm256_permute2f128_ps(S0_1, S4_1, 0x20); + tmp.packet[9] = _mm256_permute2f128_ps(S1_1, S5_1, 0x20); + tmp.packet[10] = _mm256_permute2f128_ps(S2_1, S6_1, 0x20); + tmp.packet[11] = _mm256_permute2f128_ps(S3_1, S7_1, 0x20); + tmp.packet[12] = _mm256_permute2f128_ps(S0_1, S4_1, 0x31); + tmp.packet[13] = _mm256_permute2f128_ps(S1_1, S5_1, 0x31); + tmp.packet[14] = _mm256_permute2f128_ps(S2_1, S6_1, 0x31); + tmp.packet[15] = _mm256_permute2f128_ps(S3_1, S7_1, 0x31); + + // Second set of _m256 outputs + tmp.packet[16] = _mm256_permute2f128_ps(S8_0, S12_0, 0x20); + tmp.packet[17] = _mm256_permute2f128_ps(S9_0, S13_0, 0x20); + tmp.packet[18] = _mm256_permute2f128_ps(S10_0, S14_0, 0x20); + tmp.packet[19] = _mm256_permute2f128_ps(S11_0, S15_0, 0x20); + tmp.packet[20] = _mm256_permute2f128_ps(S8_0, S12_0, 0x31); + tmp.packet[21] = _mm256_permute2f128_ps(S9_0, S13_0, 0x31); + tmp.packet[22] = _mm256_permute2f128_ps(S10_0, S14_0, 0x31); + tmp.packet[23] = _mm256_permute2f128_ps(S11_0, S15_0, 0x31); + + tmp.packet[24] = _mm256_permute2f128_ps(S8_1, S12_1, 0x20); + tmp.packet[25] = _mm256_permute2f128_ps(S9_1, S13_1, 0x20); + tmp.packet[26] = _mm256_permute2f128_ps(S10_1, S14_1, 0x20); + tmp.packet[27] = _mm256_permute2f128_ps(S11_1, S15_1, 0x20); + tmp.packet[28] = _mm256_permute2f128_ps(S8_1, S12_1, 0x31); + tmp.packet[29] = _mm256_permute2f128_ps(S9_1, S13_1, 0x31); + tmp.packet[30] = _mm256_permute2f128_ps(S10_1, S14_1, 0x31); + tmp.packet[31] = _mm256_permute2f128_ps(S11_1, S15_1, 0x31); + + // Pack them into the output + PACK_OUTPUT(kernel.packet, tmp.packet, 0, 16); + PACK_OUTPUT(kernel.packet, tmp.packet, 1, 16); + PACK_OUTPUT(kernel.packet, tmp.packet, 2, 16); + PACK_OUTPUT(kernel.packet, tmp.packet, 3, 16); + + PACK_OUTPUT(kernel.packet, tmp.packet, 4, 16); + PACK_OUTPUT(kernel.packet, tmp.packet, 5, 16); + PACK_OUTPUT(kernel.packet, tmp.packet, 6, 16); + PACK_OUTPUT(kernel.packet, tmp.packet, 7, 16); + + PACK_OUTPUT(kernel.packet, tmp.packet, 8, 16); + PACK_OUTPUT(kernel.packet, tmp.packet, 9, 16); + PACK_OUTPUT(kernel.packet, tmp.packet, 10, 16); + PACK_OUTPUT(kernel.packet, tmp.packet, 11, 16); + + PACK_OUTPUT(kernel.packet, tmp.packet, 12, 16); + PACK_OUTPUT(kernel.packet, tmp.packet, 13, 16); + PACK_OUTPUT(kernel.packet, tmp.packet, 14, 16); + PACK_OUTPUT(kernel.packet, tmp.packet, 15, 16); + } +#define PACK_OUTPUT_2(OUTPUT, INPUT, INDEX, STRIDE) \ + EIGEN_INSERT_8f_INTO_16f(OUTPUT[INDEX], INPUT[2 * INDEX], INPUT[2 * INDEX + STRIDE]); + + EIGEN_DEVICE_FUNC inline void ptranspose(PacketBlock &kernel) + { + __m512 T0 = _mm512_unpacklo_ps(kernel.packet[0], kernel.packet[1]); + __m512 T1 = _mm512_unpackhi_ps(kernel.packet[0], kernel.packet[1]); + __m512 T2 = _mm512_unpacklo_ps(kernel.packet[2], kernel.packet[3]); + __m512 T3 = _mm512_unpackhi_ps(kernel.packet[2], kernel.packet[3]); + + __m512 S0 = _mm512_shuffle_ps(T0, T2, _MM_SHUFFLE(1, 0, 1, 0)); + __m512 S1 = _mm512_shuffle_ps(T0, T2, _MM_SHUFFLE(3, 2, 3, 2)); + __m512 S2 = _mm512_shuffle_ps(T1, T3, _MM_SHUFFLE(1, 0, 1, 0)); + __m512 S3 = _mm512_shuffle_ps(T1, T3, _MM_SHUFFLE(3, 2, 3, 2)); + + EIGEN_EXTRACT_8f_FROM_16f(S0, S0); + EIGEN_EXTRACT_8f_FROM_16f(S1, S1); + EIGEN_EXTRACT_8f_FROM_16f(S2, S2); + EIGEN_EXTRACT_8f_FROM_16f(S3, S3); + + PacketBlock tmp; + + tmp.packet[0] = _mm256_permute2f128_ps(S0_0, S1_0, 0x20); + tmp.packet[1] = _mm256_permute2f128_ps(S2_0, S3_0, 0x20); + tmp.packet[2] = _mm256_permute2f128_ps(S0_0, S1_0, 0x31); + tmp.packet[3] = _mm256_permute2f128_ps(S2_0, S3_0, 0x31); + + tmp.packet[4] = _mm256_permute2f128_ps(S0_1, S1_1, 0x20); + tmp.packet[5] = _mm256_permute2f128_ps(S2_1, S3_1, 0x20); + tmp.packet[6] = _mm256_permute2f128_ps(S0_1, S1_1, 0x31); + tmp.packet[7] = _mm256_permute2f128_ps(S2_1, S3_1, 0x31); + + PACK_OUTPUT_2(kernel.packet, tmp.packet, 0, 1); + PACK_OUTPUT_2(kernel.packet, tmp.packet, 1, 1); + PACK_OUTPUT_2(kernel.packet, tmp.packet, 2, 1); + PACK_OUTPUT_2(kernel.packet, tmp.packet, 3, 1); + } #define PACK_OUTPUT_SQ_D(OUTPUT, INPUT, INDEX, STRIDE) \ OUTPUT[INDEX] = _mm512_insertf64x4(OUTPUT[INDEX], INPUT[INDEX], 0); \ @@ -1201,116 +1192,95 @@ EIGEN_DEVICE_FUNC inline void ptranspose(PacketBlock& kernel) { #define PACK_OUTPUT_D(OUTPUT, INPUT, INDEX, STRIDE) \ OUTPUT[INDEX] = _mm512_insertf64x4(OUTPUT[INDEX], INPUT[(2 * INDEX)], 0); \ - OUTPUT[INDEX] = \ - _mm512_insertf64x4(OUTPUT[INDEX], INPUT[(2 * INDEX) + STRIDE], 1); - -EIGEN_DEVICE_FUNC inline void ptranspose(PacketBlock& kernel) { - __m512d T0 = _mm512_shuffle_pd(kernel.packet[0], kernel.packet[1], 0); - __m512d T1 = _mm512_shuffle_pd(kernel.packet[0], kernel.packet[1], 0xff); - __m512d T2 = _mm512_shuffle_pd(kernel.packet[2], kernel.packet[3], 0); - __m512d T3 = _mm512_shuffle_pd(kernel.packet[2], kernel.packet[3], 0xff); - - PacketBlock tmp; - - tmp.packet[0] = _mm256_permute2f128_pd(_mm512_extractf64x4_pd(T0, 0), - _mm512_extractf64x4_pd(T2, 0), 0x20); - tmp.packet[1] = _mm256_permute2f128_pd(_mm512_extractf64x4_pd(T1, 0), - _mm512_extractf64x4_pd(T3, 0), 0x20); - tmp.packet[2] = _mm256_permute2f128_pd(_mm512_extractf64x4_pd(T0, 0), - _mm512_extractf64x4_pd(T2, 0), 0x31); - tmp.packet[3] = _mm256_permute2f128_pd(_mm512_extractf64x4_pd(T1, 0), - _mm512_extractf64x4_pd(T3, 0), 0x31); - - tmp.packet[4] = _mm256_permute2f128_pd(_mm512_extractf64x4_pd(T0, 1), - _mm512_extractf64x4_pd(T2, 1), 0x20); - tmp.packet[5] = _mm256_permute2f128_pd(_mm512_extractf64x4_pd(T1, 1), - _mm512_extractf64x4_pd(T3, 1), 0x20); - tmp.packet[6] = _mm256_permute2f128_pd(_mm512_extractf64x4_pd(T0, 1), - _mm512_extractf64x4_pd(T2, 1), 0x31); - tmp.packet[7] = _mm256_permute2f128_pd(_mm512_extractf64x4_pd(T1, 1), - _mm512_extractf64x4_pd(T3, 1), 0x31); - - PACK_OUTPUT_D(kernel.packet, tmp.packet, 0, 1); - PACK_OUTPUT_D(kernel.packet, tmp.packet, 1, 1); - PACK_OUTPUT_D(kernel.packet, tmp.packet, 2, 1); - PACK_OUTPUT_D(kernel.packet, tmp.packet, 3, 1); -} - -EIGEN_DEVICE_FUNC inline void ptranspose(PacketBlock& kernel) { - __m512d T0 = _mm512_unpacklo_pd(kernel.packet[0], kernel.packet[1]); - __m512d T1 = _mm512_unpackhi_pd(kernel.packet[0], kernel.packet[1]); - __m512d T2 = _mm512_unpacklo_pd(kernel.packet[2], kernel.packet[3]); - __m512d T3 = _mm512_unpackhi_pd(kernel.packet[2], kernel.packet[3]); - __m512d T4 = _mm512_unpacklo_pd(kernel.packet[4], kernel.packet[5]); - __m512d T5 = _mm512_unpackhi_pd(kernel.packet[4], kernel.packet[5]); - __m512d T6 = _mm512_unpacklo_pd(kernel.packet[6], kernel.packet[7]); - __m512d T7 = _mm512_unpackhi_pd(kernel.packet[6], kernel.packet[7]); - - PacketBlock tmp; - - tmp.packet[0] = _mm256_permute2f128_pd(_mm512_extractf64x4_pd(T0, 0), - _mm512_extractf64x4_pd(T2, 0), 0x20); - tmp.packet[1] = _mm256_permute2f128_pd(_mm512_extractf64x4_pd(T1, 0), - _mm512_extractf64x4_pd(T3, 0), 0x20); - tmp.packet[2] = _mm256_permute2f128_pd(_mm512_extractf64x4_pd(T0, 0), - _mm512_extractf64x4_pd(T2, 0), 0x31); - tmp.packet[3] = _mm256_permute2f128_pd(_mm512_extractf64x4_pd(T1, 0), - _mm512_extractf64x4_pd(T3, 0), 0x31); - - tmp.packet[4] = _mm256_permute2f128_pd(_mm512_extractf64x4_pd(T0, 1), - _mm512_extractf64x4_pd(T2, 1), 0x20); - tmp.packet[5] = _mm256_permute2f128_pd(_mm512_extractf64x4_pd(T1, 1), - _mm512_extractf64x4_pd(T3, 1), 0x20); - tmp.packet[6] = _mm256_permute2f128_pd(_mm512_extractf64x4_pd(T0, 1), - _mm512_extractf64x4_pd(T2, 1), 0x31); - tmp.packet[7] = _mm256_permute2f128_pd(_mm512_extractf64x4_pd(T1, 1), - _mm512_extractf64x4_pd(T3, 1), 0x31); - - tmp.packet[8] = _mm256_permute2f128_pd(_mm512_extractf64x4_pd(T4, 0), - _mm512_extractf64x4_pd(T6, 0), 0x20); - tmp.packet[9] = _mm256_permute2f128_pd(_mm512_extractf64x4_pd(T5, 0), - _mm512_extractf64x4_pd(T7, 0), 0x20); - tmp.packet[10] = _mm256_permute2f128_pd(_mm512_extractf64x4_pd(T4, 0), - _mm512_extractf64x4_pd(T6, 0), 0x31); - tmp.packet[11] = _mm256_permute2f128_pd(_mm512_extractf64x4_pd(T5, 0), - _mm512_extractf64x4_pd(T7, 0), 0x31); - - tmp.packet[12] = _mm256_permute2f128_pd(_mm512_extractf64x4_pd(T4, 1), - _mm512_extractf64x4_pd(T6, 1), 0x20); - tmp.packet[13] = _mm256_permute2f128_pd(_mm512_extractf64x4_pd(T5, 1), - _mm512_extractf64x4_pd(T7, 1), 0x20); - tmp.packet[14] = _mm256_permute2f128_pd(_mm512_extractf64x4_pd(T4, 1), - _mm512_extractf64x4_pd(T6, 1), 0x31); - tmp.packet[15] = _mm256_permute2f128_pd(_mm512_extractf64x4_pd(T5, 1), - _mm512_extractf64x4_pd(T7, 1), 0x31); - - PACK_OUTPUT_SQ_D(kernel.packet, tmp.packet, 0, 8); - PACK_OUTPUT_SQ_D(kernel.packet, tmp.packet, 1, 8); - PACK_OUTPUT_SQ_D(kernel.packet, tmp.packet, 2, 8); - PACK_OUTPUT_SQ_D(kernel.packet, tmp.packet, 3, 8); - - PACK_OUTPUT_SQ_D(kernel.packet, tmp.packet, 4, 8); - PACK_OUTPUT_SQ_D(kernel.packet, tmp.packet, 5, 8); - PACK_OUTPUT_SQ_D(kernel.packet, tmp.packet, 6, 8); - PACK_OUTPUT_SQ_D(kernel.packet, tmp.packet, 7, 8); -} -template <> -EIGEN_STRONG_INLINE Packet16f pblend(const Selector<16>& /*ifPacket*/, - const Packet16f& /*thenPacket*/, - const Packet16f& /*elsePacket*/) { - assert(false && "To be implemented"); - return Packet16f(); -} -template <> -EIGEN_STRONG_INLINE Packet8d pblend(const Selector<8>& /*ifPacket*/, - const Packet8d& /*thenPacket*/, - const Packet8d& /*elsePacket*/) { - assert(false && "To be implemented"); - return Packet8d(); -} - -} // end namespace internal - -} // end namespace Eigen - -#endif // EIGEN_PACKET_MATH_AVX512_H + OUTPUT[INDEX] = _mm512_insertf64x4(OUTPUT[INDEX], INPUT[(2 * INDEX) + STRIDE], 1); + + EIGEN_DEVICE_FUNC inline void ptranspose(PacketBlock &kernel) + { + __m512d T0 = _mm512_shuffle_pd(kernel.packet[0], kernel.packet[1], 0); + __m512d T1 = _mm512_shuffle_pd(kernel.packet[0], kernel.packet[1], 0xff); + __m512d T2 = _mm512_shuffle_pd(kernel.packet[2], kernel.packet[3], 0); + __m512d T3 = _mm512_shuffle_pd(kernel.packet[2], kernel.packet[3], 0xff); + + PacketBlock tmp; + + tmp.packet[0] = _mm256_permute2f128_pd(_mm512_extractf64x4_pd(T0, 0), _mm512_extractf64x4_pd(T2, 0), 0x20); + tmp.packet[1] = _mm256_permute2f128_pd(_mm512_extractf64x4_pd(T1, 0), _mm512_extractf64x4_pd(T3, 0), 0x20); + tmp.packet[2] = _mm256_permute2f128_pd(_mm512_extractf64x4_pd(T0, 0), _mm512_extractf64x4_pd(T2, 0), 0x31); + tmp.packet[3] = _mm256_permute2f128_pd(_mm512_extractf64x4_pd(T1, 0), _mm512_extractf64x4_pd(T3, 0), 0x31); + + tmp.packet[4] = _mm256_permute2f128_pd(_mm512_extractf64x4_pd(T0, 1), _mm512_extractf64x4_pd(T2, 1), 0x20); + tmp.packet[5] = _mm256_permute2f128_pd(_mm512_extractf64x4_pd(T1, 1), _mm512_extractf64x4_pd(T3, 1), 0x20); + tmp.packet[6] = _mm256_permute2f128_pd(_mm512_extractf64x4_pd(T0, 1), _mm512_extractf64x4_pd(T2, 1), 0x31); + tmp.packet[7] = _mm256_permute2f128_pd(_mm512_extractf64x4_pd(T1, 1), _mm512_extractf64x4_pd(T3, 1), 0x31); + + PACK_OUTPUT_D(kernel.packet, tmp.packet, 0, 1); + PACK_OUTPUT_D(kernel.packet, tmp.packet, 1, 1); + PACK_OUTPUT_D(kernel.packet, tmp.packet, 2, 1); + PACK_OUTPUT_D(kernel.packet, tmp.packet, 3, 1); + } + + EIGEN_DEVICE_FUNC inline void ptranspose(PacketBlock &kernel) + { + __m512d T0 = _mm512_unpacklo_pd(kernel.packet[0], kernel.packet[1]); + __m512d T1 = _mm512_unpackhi_pd(kernel.packet[0], kernel.packet[1]); + __m512d T2 = _mm512_unpacklo_pd(kernel.packet[2], kernel.packet[3]); + __m512d T3 = _mm512_unpackhi_pd(kernel.packet[2], kernel.packet[3]); + __m512d T4 = _mm512_unpacklo_pd(kernel.packet[4], kernel.packet[5]); + __m512d T5 = _mm512_unpackhi_pd(kernel.packet[4], kernel.packet[5]); + __m512d T6 = _mm512_unpacklo_pd(kernel.packet[6], kernel.packet[7]); + __m512d T7 = _mm512_unpackhi_pd(kernel.packet[6], kernel.packet[7]); + + PacketBlock tmp; + + tmp.packet[0] = _mm256_permute2f128_pd(_mm512_extractf64x4_pd(T0, 0), _mm512_extractf64x4_pd(T2, 0), 0x20); + tmp.packet[1] = _mm256_permute2f128_pd(_mm512_extractf64x4_pd(T1, 0), _mm512_extractf64x4_pd(T3, 0), 0x20); + tmp.packet[2] = _mm256_permute2f128_pd(_mm512_extractf64x4_pd(T0, 0), _mm512_extractf64x4_pd(T2, 0), 0x31); + tmp.packet[3] = _mm256_permute2f128_pd(_mm512_extractf64x4_pd(T1, 0), _mm512_extractf64x4_pd(T3, 0), 0x31); + + tmp.packet[4] = _mm256_permute2f128_pd(_mm512_extractf64x4_pd(T0, 1), _mm512_extractf64x4_pd(T2, 1), 0x20); + tmp.packet[5] = _mm256_permute2f128_pd(_mm512_extractf64x4_pd(T1, 1), _mm512_extractf64x4_pd(T3, 1), 0x20); + tmp.packet[6] = _mm256_permute2f128_pd(_mm512_extractf64x4_pd(T0, 1), _mm512_extractf64x4_pd(T2, 1), 0x31); + tmp.packet[7] = _mm256_permute2f128_pd(_mm512_extractf64x4_pd(T1, 1), _mm512_extractf64x4_pd(T3, 1), 0x31); + + tmp.packet[8] = _mm256_permute2f128_pd(_mm512_extractf64x4_pd(T4, 0), _mm512_extractf64x4_pd(T6, 0), 0x20); + tmp.packet[9] = _mm256_permute2f128_pd(_mm512_extractf64x4_pd(T5, 0), _mm512_extractf64x4_pd(T7, 0), 0x20); + tmp.packet[10] = _mm256_permute2f128_pd(_mm512_extractf64x4_pd(T4, 0), _mm512_extractf64x4_pd(T6, 0), 0x31); + tmp.packet[11] = _mm256_permute2f128_pd(_mm512_extractf64x4_pd(T5, 0), _mm512_extractf64x4_pd(T7, 0), 0x31); + + tmp.packet[12] = _mm256_permute2f128_pd(_mm512_extractf64x4_pd(T4, 1), _mm512_extractf64x4_pd(T6, 1), 0x20); + tmp.packet[13] = _mm256_permute2f128_pd(_mm512_extractf64x4_pd(T5, 1), _mm512_extractf64x4_pd(T7, 1), 0x20); + tmp.packet[14] = _mm256_permute2f128_pd(_mm512_extractf64x4_pd(T4, 1), _mm512_extractf64x4_pd(T6, 1), 0x31); + tmp.packet[15] = _mm256_permute2f128_pd(_mm512_extractf64x4_pd(T5, 1), _mm512_extractf64x4_pd(T7, 1), 0x31); + + PACK_OUTPUT_SQ_D(kernel.packet, tmp.packet, 0, 8); + PACK_OUTPUT_SQ_D(kernel.packet, tmp.packet, 1, 8); + PACK_OUTPUT_SQ_D(kernel.packet, tmp.packet, 2, 8); + PACK_OUTPUT_SQ_D(kernel.packet, tmp.packet, 3, 8); + + PACK_OUTPUT_SQ_D(kernel.packet, tmp.packet, 4, 8); + PACK_OUTPUT_SQ_D(kernel.packet, tmp.packet, 5, 8); + PACK_OUTPUT_SQ_D(kernel.packet, tmp.packet, 6, 8); + PACK_OUTPUT_SQ_D(kernel.packet, tmp.packet, 7, 8); + } + template<> + EIGEN_STRONG_INLINE Packet16f pblend(const Selector<16> & /*ifPacket*/, + const Packet16f & /*thenPacket*/, + const Packet16f & /*elsePacket*/) + { + assert(false && "To be implemented"); + return Packet16f(); + } + template<> + EIGEN_STRONG_INLINE Packet8d pblend(const Selector<8> & /*ifPacket*/, + const Packet8d & /*thenPacket*/, + const Packet8d & /*elsePacket*/) + { + assert(false && "To be implemented"); + return Packet8d(); + } + +}// end namespace internal + +}// end namespace Eigen + +#endif// EIGEN_PACKET_MATH_AVX512_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/arch/AltiVec/Complex.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/arch/AltiVec/Complex.h index 3e665730..ff693daf 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/arch/AltiVec/Complex.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/arch/AltiVec/Complex.h @@ -15,416 +15,532 @@ namespace Eigen { namespace internal { -static Packet4ui p4ui_CONJ_XOR = vec_mergeh((Packet4ui)p4i_ZERO, (Packet4ui)p4f_MZERO);//{ 0x00000000, 0x80000000, 0x00000000, 0x80000000 }; + static Packet4ui p4ui_CONJ_XOR = + vec_mergeh((Packet4ui)p4i_ZERO, (Packet4ui)p4f_MZERO);//{ 0x00000000, 0x80000000, 0x00000000, 0x80000000 }; #ifdef __VSX__ #if defined(_BIG_ENDIAN) -static Packet2ul p2ul_CONJ_XOR1 = (Packet2ul) vec_sld((Packet4ui) p2d_MZERO, (Packet4ui) p2l_ZERO, 8);//{ 0x8000000000000000, 0x0000000000000000 }; -static Packet2ul p2ul_CONJ_XOR2 = (Packet2ul) vec_sld((Packet4ui) p2l_ZERO, (Packet4ui) p2d_MZERO, 8);//{ 0x8000000000000000, 0x0000000000000000 }; + static Packet2ul p2ul_CONJ_XOR1 = + (Packet2ul)vec_sld((Packet4ui)p2d_MZERO, (Packet4ui)p2l_ZERO, 8);//{ 0x8000000000000000, 0x0000000000000000 }; + static Packet2ul p2ul_CONJ_XOR2 = + (Packet2ul)vec_sld((Packet4ui)p2l_ZERO, (Packet4ui)p2d_MZERO, 8);//{ 0x8000000000000000, 0x0000000000000000 }; #else -static Packet2ul p2ul_CONJ_XOR1 = (Packet2ul) vec_sld((Packet4ui) p2l_ZERO, (Packet4ui) p2d_MZERO, 8);//{ 0x8000000000000000, 0x0000000000000000 }; -static Packet2ul p2ul_CONJ_XOR2 = (Packet2ul) vec_sld((Packet4ui) p2d_MZERO, (Packet4ui) p2l_ZERO, 8);//{ 0x8000000000000000, 0x0000000000000000 }; + static Packet2ul p2ul_CONJ_XOR1 = + (Packet2ul)vec_sld((Packet4ui)p2l_ZERO, (Packet4ui)p2d_MZERO, 8);//{ 0x8000000000000000, 0x0000000000000000 }; + static Packet2ul p2ul_CONJ_XOR2 = + (Packet2ul)vec_sld((Packet4ui)p2d_MZERO, (Packet4ui)p2l_ZERO, 8);//{ 0x8000000000000000, 0x0000000000000000 }; #endif #endif -//---------- float ---------- -struct Packet2cf -{ - EIGEN_STRONG_INLINE explicit Packet2cf() : v(p4f_ZERO) {} - EIGEN_STRONG_INLINE explicit Packet2cf(const Packet4f& a) : v(a) {} - Packet4f v; -}; - -template<> struct packet_traits > : default_packet_traits -{ - typedef Packet2cf type; - typedef Packet2cf half; - enum { - Vectorizable = 1, - AlignedOnScalar = 1, - size = 2, - HasHalfPacket = 0, - - HasAdd = 1, - HasSub = 1, - HasMul = 1, - HasDiv = 1, - HasNegate = 1, - HasAbs = 0, - HasAbs2 = 0, - HasMin = 0, - HasMax = 0, + //---------- float ---------- + struct Packet2cf + { + EIGEN_STRONG_INLINE explicit Packet2cf() : v(p4f_ZERO) {} + EIGEN_STRONG_INLINE explicit Packet2cf(const Packet4f &a) : v(a) {} + Packet4f v; + }; + + template<> struct packet_traits> : default_packet_traits + { + typedef Packet2cf type; + typedef Packet2cf half; + enum { + Vectorizable = 1, + AlignedOnScalar = 1, + size = 2, + HasHalfPacket = 0, + + HasAdd = 1, + HasSub = 1, + HasMul = 1, + HasDiv = 1, + HasNegate = 1, + HasAbs = 0, + HasAbs2 = 0, + HasMin = 0, + HasMax = 0, #ifdef __VSX__ - HasBlend = 1, + HasBlend = 1, #endif - HasSetLinear = 0 + HasSetLinear = 0 + }; }; -}; - -template<> struct unpacket_traits { typedef std::complex type; enum {size=2, alignment=Aligned16}; typedef Packet2cf half; }; - -template<> EIGEN_STRONG_INLINE Packet2cf pset1(const std::complex& from) -{ - Packet2cf res; - if((std::ptrdiff_t(&from) % 16) == 0) - res.v = pload((const float *)&from); - else - res.v = ploadu((const float *)&from); - res.v = vec_perm(res.v, res.v, p16uc_PSET64_HI); - return res; -} - -template<> EIGEN_STRONG_INLINE Packet2cf pload(const std::complex* from) { return Packet2cf(pload((const float *) from)); } -template<> EIGEN_STRONG_INLINE Packet2cf ploadu(const std::complex* from) { return Packet2cf(ploadu((const float*) from)); } -template<> EIGEN_STRONG_INLINE Packet2cf ploaddup(const std::complex* from) { return pset1(*from); } - -template<> EIGEN_STRONG_INLINE void pstore >(std::complex * to, const Packet2cf& from) { pstore((float*)to, from.v); } -template<> EIGEN_STRONG_INLINE void pstoreu >(std::complex * to, const Packet2cf& from) { pstoreu((float*)to, from.v); } - -template<> EIGEN_DEVICE_FUNC inline Packet2cf pgather, Packet2cf>(const std::complex* from, Index stride) -{ - std::complex EIGEN_ALIGN16 af[2]; - af[0] = from[0*stride]; - af[1] = from[1*stride]; - return pload(af); -} -template<> EIGEN_DEVICE_FUNC inline void pscatter, Packet2cf>(std::complex* to, const Packet2cf& from, Index stride) -{ - std::complex EIGEN_ALIGN16 af[2]; - pstore >((std::complex *) af, from); - to[0*stride] = af[0]; - to[1*stride] = af[1]; -} - -template<> EIGEN_STRONG_INLINE Packet2cf padd(const Packet2cf& a, const Packet2cf& b) { return Packet2cf(a.v + b.v); } -template<> EIGEN_STRONG_INLINE Packet2cf psub(const Packet2cf& a, const Packet2cf& b) { return Packet2cf(a.v - b.v); } -template<> EIGEN_STRONG_INLINE Packet2cf pnegate(const Packet2cf& a) { return Packet2cf(pnegate(a.v)); } -template<> EIGEN_STRONG_INLINE Packet2cf pconj(const Packet2cf& a) { return Packet2cf(pxor(a.v, reinterpret_cast(p4ui_CONJ_XOR))); } - -template<> EIGEN_STRONG_INLINE Packet2cf pmul(const Packet2cf& a, const Packet2cf& b) -{ - Packet4f v1, v2; - - // Permute and multiply the real parts of a and b - v1 = vec_perm(a.v, a.v, p16uc_PSET32_WODD); - // Get the imaginary parts of a - v2 = vec_perm(a.v, a.v, p16uc_PSET32_WEVEN); - // multiply a_re * b - v1 = vec_madd(v1, b.v, p4f_ZERO); - // multiply a_im * b and get the conjugate result - v2 = vec_madd(v2, b.v, p4f_ZERO); - v2 = reinterpret_cast(pxor(v2, reinterpret_cast(p4ui_CONJ_XOR))); - // permute back to a proper order - v2 = vec_perm(v2, v2, p16uc_COMPLEX32_REV); - - return Packet2cf(padd(v1, v2)); -} - -template<> EIGEN_STRONG_INLINE Packet2cf pand (const Packet2cf& a, const Packet2cf& b) { return Packet2cf(pand(a.v, b.v)); } -template<> EIGEN_STRONG_INLINE Packet2cf por (const Packet2cf& a, const Packet2cf& b) { return Packet2cf(por(a.v, b.v)); } -template<> EIGEN_STRONG_INLINE Packet2cf pxor (const Packet2cf& a, const Packet2cf& b) { return Packet2cf(pxor(a.v, b.v)); } -template<> EIGEN_STRONG_INLINE Packet2cf pandnot(const Packet2cf& a, const Packet2cf& b) { return Packet2cf(pandnot(a.v, b.v)); } - -template<> EIGEN_STRONG_INLINE void prefetch >(const std::complex * addr) { EIGEN_PPC_PREFETCH(addr); } - -template<> EIGEN_STRONG_INLINE std::complex pfirst(const Packet2cf& a) -{ - std::complex EIGEN_ALIGN16 res[2]; - pstore((float *)&res, a.v); - - return res[0]; -} - -template<> EIGEN_STRONG_INLINE Packet2cf preverse(const Packet2cf& a) -{ - Packet4f rev_a; - rev_a = vec_perm(a.v, a.v, p16uc_COMPLEX32_REV2); - return Packet2cf(rev_a); -} - -template<> EIGEN_STRONG_INLINE std::complex predux(const Packet2cf& a) -{ - Packet4f b; - b = vec_sld(a.v, a.v, 8); - b = padd(a.v, b); - return pfirst(Packet2cf(b)); -} - -template<> EIGEN_STRONG_INLINE Packet2cf preduxp(const Packet2cf* vecs) -{ - Packet4f b1, b2; -#ifdef _BIG_ENDIAN - b1 = vec_sld(vecs[0].v, vecs[1].v, 8); - b2 = vec_sld(vecs[1].v, vecs[0].v, 8); + + template<> struct unpacket_traits + { + typedef std::complex type; + enum { size = 2, alignment = Aligned16 }; + typedef Packet2cf half; + }; + + template<> EIGEN_STRONG_INLINE Packet2cf pset1(const std::complex &from) + { + Packet2cf res; + if ((std::ptrdiff_t(&from) % 16) == 0) + res.v = pload((const float *)&from); + else + res.v = ploadu((const float *)&from); + res.v = vec_perm(res.v, res.v, p16uc_PSET64_HI); + return res; + } + + template<> EIGEN_STRONG_INLINE Packet2cf pload(const std::complex *from) + { + return Packet2cf(pload((const float *)from)); + } + template<> EIGEN_STRONG_INLINE Packet2cf ploadu(const std::complex *from) + { + return Packet2cf(ploadu((const float *)from)); + } + template<> EIGEN_STRONG_INLINE Packet2cf ploaddup(const std::complex *from) + { + return pset1(*from); + } + + template<> EIGEN_STRONG_INLINE void pstore>(std::complex *to, const Packet2cf &from) + { + pstore((float *)to, from.v); + } + template<> EIGEN_STRONG_INLINE void pstoreu>(std::complex *to, const Packet2cf &from) + { + pstoreu((float *)to, from.v); + } + + template<> + EIGEN_DEVICE_FUNC inline Packet2cf pgather, Packet2cf>(const std::complex *from, + Index stride) + { + std::complex EIGEN_ALIGN16 af[2]; + af[0] = from[0 * stride]; + af[1] = from[1 * stride]; + return pload(af); + } + template<> + EIGEN_DEVICE_FUNC inline void + pscatter, Packet2cf>(std::complex *to, const Packet2cf &from, Index stride) + { + std::complex EIGEN_ALIGN16 af[2]; + pstore>((std::complex *)af, from); + to[0 * stride] = af[0]; + to[1 * stride] = af[1]; + } + + template<> EIGEN_STRONG_INLINE Packet2cf padd(const Packet2cf &a, const Packet2cf &b) + { + return Packet2cf(a.v + b.v); + } + template<> EIGEN_STRONG_INLINE Packet2cf psub(const Packet2cf &a, const Packet2cf &b) + { + return Packet2cf(a.v - b.v); + } + template<> EIGEN_STRONG_INLINE Packet2cf pnegate(const Packet2cf &a) { return Packet2cf(pnegate(a.v)); } + template<> EIGEN_STRONG_INLINE Packet2cf pconj(const Packet2cf &a) + { + return Packet2cf(pxor(a.v, reinterpret_cast(p4ui_CONJ_XOR))); + } + + template<> EIGEN_STRONG_INLINE Packet2cf pmul(const Packet2cf &a, const Packet2cf &b) + { + Packet4f v1, v2; + + // Permute and multiply the real parts of a and b + v1 = vec_perm(a.v, a.v, p16uc_PSET32_WODD); + // Get the imaginary parts of a + v2 = vec_perm(a.v, a.v, p16uc_PSET32_WEVEN); + // multiply a_re * b + v1 = vec_madd(v1, b.v, p4f_ZERO); + // multiply a_im * b and get the conjugate result + v2 = vec_madd(v2, b.v, p4f_ZERO); + v2 = reinterpret_cast(pxor(v2, reinterpret_cast(p4ui_CONJ_XOR))); + // permute back to a proper order + v2 = vec_perm(v2, v2, p16uc_COMPLEX32_REV); + + return Packet2cf(padd(v1, v2)); + } + + template<> EIGEN_STRONG_INLINE Packet2cf pand(const Packet2cf &a, const Packet2cf &b) + { + return Packet2cf(pand(a.v, b.v)); + } + template<> EIGEN_STRONG_INLINE Packet2cf por(const Packet2cf &a, const Packet2cf &b) + { + return Packet2cf(por(a.v, b.v)); + } + template<> EIGEN_STRONG_INLINE Packet2cf pxor(const Packet2cf &a, const Packet2cf &b) + { + return Packet2cf(pxor(a.v, b.v)); + } + template<> EIGEN_STRONG_INLINE Packet2cf pandnot(const Packet2cf &a, const Packet2cf &b) + { + return Packet2cf(pandnot(a.v, b.v)); + } + + template<> EIGEN_STRONG_INLINE void prefetch>(const std::complex *addr) + { + EIGEN_PPC_PREFETCH(addr); + } + + template<> EIGEN_STRONG_INLINE std::complex pfirst(const Packet2cf &a) + { + std::complex EIGEN_ALIGN16 res[2]; + pstore((float *)&res, a.v); + + return res[0]; + } + + template<> EIGEN_STRONG_INLINE Packet2cf preverse(const Packet2cf &a) + { + Packet4f rev_a; + rev_a = vec_perm(a.v, a.v, p16uc_COMPLEX32_REV2); + return Packet2cf(rev_a); + } + + template<> EIGEN_STRONG_INLINE std::complex predux(const Packet2cf &a) + { + Packet4f b; + b = vec_sld(a.v, a.v, 8); + b = padd(a.v, b); + return pfirst(Packet2cf(b)); + } + + template<> EIGEN_STRONG_INLINE Packet2cf preduxp(const Packet2cf *vecs) + { + Packet4f b1, b2; +#ifdef _BIG_ENDIAN + b1 = vec_sld(vecs[0].v, vecs[1].v, 8); + b2 = vec_sld(vecs[1].v, vecs[0].v, 8); #else - b1 = vec_sld(vecs[1].v, vecs[0].v, 8); - b2 = vec_sld(vecs[0].v, vecs[1].v, 8); + b1 = vec_sld(vecs[1].v, vecs[0].v, 8); + b2 = vec_sld(vecs[0].v, vecs[1].v, 8); #endif - b2 = vec_sld(b2, b2, 8); - b2 = padd(b1, b2); + b2 = vec_sld(b2, b2, 8); + b2 = padd(b1, b2); - return Packet2cf(b2); -} + return Packet2cf(b2); + } -template<> EIGEN_STRONG_INLINE std::complex predux_mul(const Packet2cf& a) -{ - Packet4f b; - Packet2cf prod; - b = vec_sld(a.v, a.v, 8); - prod = pmul(a, Packet2cf(b)); + template<> EIGEN_STRONG_INLINE std::complex predux_mul(const Packet2cf &a) + { + Packet4f b; + Packet2cf prod; + b = vec_sld(a.v, a.v, 8); + prod = pmul(a, Packet2cf(b)); - return pfirst(prod); -} + return pfirst(prod); + } -template -struct palign_impl -{ - static EIGEN_STRONG_INLINE void run(Packet2cf& first, const Packet2cf& second) + template struct palign_impl { - if (Offset==1) + static EIGEN_STRONG_INLINE void run(Packet2cf &first, const Packet2cf &second) { + if (Offset == 1) { #ifdef _BIG_ENDIAN - first.v = vec_sld(first.v, second.v, 8); + first.v = vec_sld(first.v, second.v, 8); #else - first.v = vec_sld(second.v, first.v, 8); + first.v = vec_sld(second.v, first.v, 8); #endif + } } - } -}; - -template<> struct conj_helper -{ - EIGEN_STRONG_INLINE Packet2cf pmadd(const Packet2cf& x, const Packet2cf& y, const Packet2cf& c) const - { return padd(pmul(x,y),c); } + }; - EIGEN_STRONG_INLINE Packet2cf pmul(const Packet2cf& a, const Packet2cf& b) const + template<> struct conj_helper { - return internal::pmul(a, pconj(b)); - } -}; + EIGEN_STRONG_INLINE Packet2cf pmadd(const Packet2cf &x, const Packet2cf &y, const Packet2cf &c) const + { + return padd(pmul(x, y), c); + } -template<> struct conj_helper -{ - EIGEN_STRONG_INLINE Packet2cf pmadd(const Packet2cf& x, const Packet2cf& y, const Packet2cf& c) const - { return padd(pmul(x,y),c); } + EIGEN_STRONG_INLINE Packet2cf pmul(const Packet2cf &a, const Packet2cf &b) const + { + return internal::pmul(a, pconj(b)); + } + }; - EIGEN_STRONG_INLINE Packet2cf pmul(const Packet2cf& a, const Packet2cf& b) const + template<> struct conj_helper { - return internal::pmul(pconj(a), b); - } -}; + EIGEN_STRONG_INLINE Packet2cf pmadd(const Packet2cf &x, const Packet2cf &y, const Packet2cf &c) const + { + return padd(pmul(x, y), c); + } -template<> struct conj_helper -{ - EIGEN_STRONG_INLINE Packet2cf pmadd(const Packet2cf& x, const Packet2cf& y, const Packet2cf& c) const - { return padd(pmul(x,y),c); } + EIGEN_STRONG_INLINE Packet2cf pmul(const Packet2cf &a, const Packet2cf &b) const + { + return internal::pmul(pconj(a), b); + } + }; - EIGEN_STRONG_INLINE Packet2cf pmul(const Packet2cf& a, const Packet2cf& b) const + template<> struct conj_helper { - return pconj(internal::pmul(a, b)); - } -}; + EIGEN_STRONG_INLINE Packet2cf pmadd(const Packet2cf &x, const Packet2cf &y, const Packet2cf &c) const + { + return padd(pmul(x, y), c); + } -EIGEN_MAKE_CONJ_HELPER_CPLX_REAL(Packet2cf,Packet4f) + EIGEN_STRONG_INLINE Packet2cf pmul(const Packet2cf &a, const Packet2cf &b) const + { + return pconj(internal::pmul(a, b)); + } + }; -template<> EIGEN_STRONG_INLINE Packet2cf pdiv(const Packet2cf& a, const Packet2cf& b) -{ - // TODO optimize it for AltiVec - Packet2cf res = conj_helper().pmul(a, b); - Packet4f s = pmul(b.v, b.v); - return Packet2cf(pdiv(res.v, padd(s, vec_perm(s, s, p16uc_COMPLEX32_REV)))); -} + EIGEN_MAKE_CONJ_HELPER_CPLX_REAL(Packet2cf, Packet4f) -template<> EIGEN_STRONG_INLINE Packet2cf pcplxflip(const Packet2cf& x) -{ - return Packet2cf(vec_perm(x.v, x.v, p16uc_COMPLEX32_REV)); -} + template<> EIGEN_STRONG_INLINE Packet2cf pdiv(const Packet2cf &a, const Packet2cf &b) + { + // TODO optimize it for AltiVec + Packet2cf res = conj_helper().pmul(a, b); + Packet4f s = pmul(b.v, b.v); + return Packet2cf(pdiv(res.v, padd(s, vec_perm(s, s, p16uc_COMPLEX32_REV)))); + } -EIGEN_STRONG_INLINE void ptranspose(PacketBlock& kernel) -{ - Packet4f tmp = vec_perm(kernel.packet[0].v, kernel.packet[1].v, p16uc_TRANSPOSE64_HI); - kernel.packet[1].v = vec_perm(kernel.packet[0].v, kernel.packet[1].v, p16uc_TRANSPOSE64_LO); - kernel.packet[0].v = tmp; -} + template<> EIGEN_STRONG_INLINE Packet2cf pcplxflip(const Packet2cf &x) + { + return Packet2cf(vec_perm(x.v, x.v, p16uc_COMPLEX32_REV)); + } + + EIGEN_STRONG_INLINE void ptranspose(PacketBlock &kernel) + { + Packet4f tmp = vec_perm(kernel.packet[0].v, kernel.packet[1].v, p16uc_TRANSPOSE64_HI); + kernel.packet[1].v = vec_perm(kernel.packet[0].v, kernel.packet[1].v, p16uc_TRANSPOSE64_LO); + kernel.packet[0].v = tmp; + } #ifdef __VSX__ -template<> EIGEN_STRONG_INLINE Packet2cf pblend(const Selector<2>& ifPacket, const Packet2cf& thenPacket, const Packet2cf& elsePacket) { - Packet2cf result; - result.v = reinterpret_cast(pblend(ifPacket, reinterpret_cast(thenPacket.v), reinterpret_cast(elsePacket.v))); - return result; -} + template<> + EIGEN_STRONG_INLINE + Packet2cf pblend(const Selector<2> &ifPacket, const Packet2cf &thenPacket, const Packet2cf &elsePacket) + { + Packet2cf result; + result.v = reinterpret_cast( + pblend(ifPacket, reinterpret_cast(thenPacket.v), reinterpret_cast(elsePacket.v))); + return result; + } #endif //---------- double ---------- #ifdef __VSX__ -struct Packet1cd -{ - EIGEN_STRONG_INLINE Packet1cd() {} - EIGEN_STRONG_INLINE explicit Packet1cd(const Packet2d& a) : v(a) {} - Packet2d v; -}; - -template<> struct packet_traits > : default_packet_traits -{ - typedef Packet1cd type; - typedef Packet1cd half; - enum { - Vectorizable = 1, - AlignedOnScalar = 0, - size = 1, - HasHalfPacket = 0, - - HasAdd = 1, - HasSub = 1, - HasMul = 1, - HasDiv = 1, - HasNegate = 1, - HasAbs = 0, - HasAbs2 = 0, - HasMin = 0, - HasMax = 0, - HasSetLinear = 0 + struct Packet1cd + { + EIGEN_STRONG_INLINE Packet1cd() {} + EIGEN_STRONG_INLINE explicit Packet1cd(const Packet2d &a) : v(a) {} + Packet2d v; }; -}; -template<> struct unpacket_traits { typedef std::complex type; enum {size=1, alignment=Aligned16}; typedef Packet1cd half; }; + template<> struct packet_traits> : default_packet_traits + { + typedef Packet1cd type; + typedef Packet1cd half; + enum { + Vectorizable = 1, + AlignedOnScalar = 0, + size = 1, + HasHalfPacket = 0, + + HasAdd = 1, + HasSub = 1, + HasMul = 1, + HasDiv = 1, + HasNegate = 1, + HasAbs = 0, + HasAbs2 = 0, + HasMin = 0, + HasMax = 0, + HasSetLinear = 0 + }; + }; -template<> EIGEN_STRONG_INLINE Packet1cd pload (const std::complex* from) { return Packet1cd(pload((const double*)from)); } -template<> EIGEN_STRONG_INLINE Packet1cd ploadu(const std::complex* from) { return Packet1cd(ploadu((const double*)from)); } -template<> EIGEN_STRONG_INLINE void pstore >(std::complex * to, const Packet1cd& from) { pstore((double*)to, from.v); } -template<> EIGEN_STRONG_INLINE void pstoreu >(std::complex * to, const Packet1cd& from) { pstoreu((double*)to, from.v); } + template<> struct unpacket_traits + { + typedef std::complex type; + enum { size = 1, alignment = Aligned16 }; + typedef Packet1cd half; + }; -template<> EIGEN_STRONG_INLINE Packet1cd pset1(const std::complex& from) -{ /* here we really have to use unaligned loads :( */ return ploadu(&from); } + template<> EIGEN_STRONG_INLINE Packet1cd pload(const std::complex *from) + { + return Packet1cd(pload((const double *)from)); + } + template<> EIGEN_STRONG_INLINE Packet1cd ploadu(const std::complex *from) + { + return Packet1cd(ploadu((const double *)from)); + } + template<> EIGEN_STRONG_INLINE void pstore>(std::complex *to, const Packet1cd &from) + { + pstore((double *)to, from.v); + } + template<> EIGEN_STRONG_INLINE void pstoreu>(std::complex *to, const Packet1cd &from) + { + pstoreu((double *)to, from.v); + } -template<> EIGEN_DEVICE_FUNC inline Packet1cd pgather, Packet1cd>(const std::complex* from, Index stride) -{ - std::complex EIGEN_ALIGN16 af[2]; - af[0] = from[0*stride]; - af[1] = from[1*stride]; - return pload(af); -} -template<> EIGEN_DEVICE_FUNC inline void pscatter, Packet1cd>(std::complex* to, const Packet1cd& from, Index stride) -{ - std::complex EIGEN_ALIGN16 af[2]; - pstore >(af, from); - to[0*stride] = af[0]; - to[1*stride] = af[1]; -} + template<> EIGEN_STRONG_INLINE Packet1cd pset1(const std::complex &from) + { /* here we really have to use unaligned loads :( */ + return ploadu(&from); + } -template<> EIGEN_STRONG_INLINE Packet1cd padd(const Packet1cd& a, const Packet1cd& b) { return Packet1cd(a.v + b.v); } -template<> EIGEN_STRONG_INLINE Packet1cd psub(const Packet1cd& a, const Packet1cd& b) { return Packet1cd(a.v - b.v); } -template<> EIGEN_STRONG_INLINE Packet1cd pnegate(const Packet1cd& a) { return Packet1cd(pnegate(Packet2d(a.v))); } -template<> EIGEN_STRONG_INLINE Packet1cd pconj(const Packet1cd& a) { return Packet1cd(pxor(a.v, reinterpret_cast(p2ul_CONJ_XOR2))); } + template<> + EIGEN_DEVICE_FUNC inline Packet1cd pgather, Packet1cd>(const std::complex *from, + Index stride) + { + std::complex EIGEN_ALIGN16 af[2]; + af[0] = from[0 * stride]; + af[1] = from[1 * stride]; + return pload(af); + } + template<> + EIGEN_DEVICE_FUNC inline void + pscatter, Packet1cd>(std::complex *to, const Packet1cd &from, Index stride) + { + std::complex EIGEN_ALIGN16 af[2]; + pstore>(af, from); + to[0 * stride] = af[0]; + to[1 * stride] = af[1]; + } -template<> EIGEN_STRONG_INLINE Packet1cd pmul(const Packet1cd& a, const Packet1cd& b) -{ - Packet2d a_re, a_im, v1, v2; + template<> EIGEN_STRONG_INLINE Packet1cd padd(const Packet1cd &a, const Packet1cd &b) + { + return Packet1cd(a.v + b.v); + } + template<> EIGEN_STRONG_INLINE Packet1cd psub(const Packet1cd &a, const Packet1cd &b) + { + return Packet1cd(a.v - b.v); + } + template<> EIGEN_STRONG_INLINE Packet1cd pnegate(const Packet1cd &a) { return Packet1cd(pnegate(Packet2d(a.v))); } + template<> EIGEN_STRONG_INLINE Packet1cd pconj(const Packet1cd &a) + { + return Packet1cd(pxor(a.v, reinterpret_cast(p2ul_CONJ_XOR2))); + } - // Permute and multiply the real parts of a and b - a_re = vec_perm(a.v, a.v, p16uc_PSET64_HI); - // Get the imaginary parts of a - a_im = vec_perm(a.v, a.v, p16uc_PSET64_LO); - // multiply a_re * b - v1 = vec_madd(a_re, b.v, p2d_ZERO); - // multiply a_im * b and get the conjugate result - v2 = vec_madd(a_im, b.v, p2d_ZERO); - v2 = reinterpret_cast(vec_sld(reinterpret_cast(v2), reinterpret_cast(v2), 8)); - v2 = pxor(v2, reinterpret_cast(p2ul_CONJ_XOR1)); + template<> EIGEN_STRONG_INLINE Packet1cd pmul(const Packet1cd &a, const Packet1cd &b) + { + Packet2d a_re, a_im, v1, v2; + + // Permute and multiply the real parts of a and b + a_re = vec_perm(a.v, a.v, p16uc_PSET64_HI); + // Get the imaginary parts of a + a_im = vec_perm(a.v, a.v, p16uc_PSET64_LO); + // multiply a_re * b + v1 = vec_madd(a_re, b.v, p2d_ZERO); + // multiply a_im * b and get the conjugate result + v2 = vec_madd(a_im, b.v, p2d_ZERO); + v2 = reinterpret_cast(vec_sld(reinterpret_cast(v2), reinterpret_cast(v2), 8)); + v2 = pxor(v2, reinterpret_cast(p2ul_CONJ_XOR1)); + + return Packet1cd(padd(v1, v2)); + } - return Packet1cd(padd(v1, v2)); -} + template<> EIGEN_STRONG_INLINE Packet1cd pand(const Packet1cd &a, const Packet1cd &b) + { + return Packet1cd(pand(a.v, b.v)); + } + template<> EIGEN_STRONG_INLINE Packet1cd por(const Packet1cd &a, const Packet1cd &b) + { + return Packet1cd(por(a.v, b.v)); + } + template<> EIGEN_STRONG_INLINE Packet1cd pxor(const Packet1cd &a, const Packet1cd &b) + { + return Packet1cd(pxor(a.v, b.v)); + } + template<> EIGEN_STRONG_INLINE Packet1cd pandnot(const Packet1cd &a, const Packet1cd &b) + { + return Packet1cd(pandnot(a.v, b.v)); + } -template<> EIGEN_STRONG_INLINE Packet1cd pand (const Packet1cd& a, const Packet1cd& b) { return Packet1cd(pand(a.v,b.v)); } -template<> EIGEN_STRONG_INLINE Packet1cd por (const Packet1cd& a, const Packet1cd& b) { return Packet1cd(por(a.v,b.v)); } -template<> EIGEN_STRONG_INLINE Packet1cd pxor (const Packet1cd& a, const Packet1cd& b) { return Packet1cd(pxor(a.v,b.v)); } -template<> EIGEN_STRONG_INLINE Packet1cd pandnot(const Packet1cd& a, const Packet1cd& b) { return Packet1cd(pandnot(a.v, b.v)); } + template<> EIGEN_STRONG_INLINE Packet1cd ploaddup(const std::complex *from) + { + return pset1(*from); + } -template<> EIGEN_STRONG_INLINE Packet1cd ploaddup(const std::complex* from) { return pset1(*from); } + template<> EIGEN_STRONG_INLINE void prefetch>(const std::complex *addr) + { + EIGEN_PPC_PREFETCH(addr); + } -template<> EIGEN_STRONG_INLINE void prefetch >(const std::complex * addr) { EIGEN_PPC_PREFETCH(addr); } + template<> EIGEN_STRONG_INLINE std::complex pfirst(const Packet1cd &a) + { + std::complex EIGEN_ALIGN16 res[2]; + pstore>(res, a); -template<> EIGEN_STRONG_INLINE std::complex pfirst(const Packet1cd& a) -{ - std::complex EIGEN_ALIGN16 res[2]; - pstore >(res, a); + return res[0]; + } - return res[0]; -} + template<> EIGEN_STRONG_INLINE Packet1cd preverse(const Packet1cd &a) { return a; } -template<> EIGEN_STRONG_INLINE Packet1cd preverse(const Packet1cd& a) { return a; } + template<> EIGEN_STRONG_INLINE std::complex predux(const Packet1cd &a) { return pfirst(a); } + template<> EIGEN_STRONG_INLINE Packet1cd preduxp(const Packet1cd *vecs) { return vecs[0]; } -template<> EIGEN_STRONG_INLINE std::complex predux(const Packet1cd& a) { return pfirst(a); } -template<> EIGEN_STRONG_INLINE Packet1cd preduxp(const Packet1cd* vecs) { return vecs[0]; } + template<> EIGEN_STRONG_INLINE std::complex predux_mul(const Packet1cd &a) { return pfirst(a); } -template<> EIGEN_STRONG_INLINE std::complex predux_mul(const Packet1cd& a) { return pfirst(a); } + template struct palign_impl + { + static EIGEN_STRONG_INLINE void run(Packet1cd & /*first*/, const Packet1cd & /*second*/) + { + // FIXME is it sure we never have to align a Packet1cd? + // Even though a std::complex has 16 bytes, it is not necessarily aligned on a 16 bytes boundary... + } + }; -template -struct palign_impl -{ - static EIGEN_STRONG_INLINE void run(Packet1cd& /*first*/, const Packet1cd& /*second*/) + template<> struct conj_helper { - // FIXME is it sure we never have to align a Packet1cd? - // Even though a std::complex has 16 bytes, it is not necessarily aligned on a 16 bytes boundary... - } -}; + EIGEN_STRONG_INLINE Packet1cd pmadd(const Packet1cd &x, const Packet1cd &y, const Packet1cd &c) const + { + return padd(pmul(x, y), c); + } -template<> struct conj_helper -{ - EIGEN_STRONG_INLINE Packet1cd pmadd(const Packet1cd& x, const Packet1cd& y, const Packet1cd& c) const - { return padd(pmul(x,y),c); } + EIGEN_STRONG_INLINE Packet1cd pmul(const Packet1cd &a, const Packet1cd &b) const + { + return internal::pmul(a, pconj(b)); + } + }; - EIGEN_STRONG_INLINE Packet1cd pmul(const Packet1cd& a, const Packet1cd& b) const + template<> struct conj_helper { - return internal::pmul(a, pconj(b)); - } -}; + EIGEN_STRONG_INLINE Packet1cd pmadd(const Packet1cd &x, const Packet1cd &y, const Packet1cd &c) const + { + return padd(pmul(x, y), c); + } -template<> struct conj_helper -{ - EIGEN_STRONG_INLINE Packet1cd pmadd(const Packet1cd& x, const Packet1cd& y, const Packet1cd& c) const - { return padd(pmul(x,y),c); } + EIGEN_STRONG_INLINE Packet1cd pmul(const Packet1cd &a, const Packet1cd &b) const + { + return internal::pmul(pconj(a), b); + } + }; - EIGEN_STRONG_INLINE Packet1cd pmul(const Packet1cd& a, const Packet1cd& b) const + template<> struct conj_helper { - return internal::pmul(pconj(a), b); - } -}; + EIGEN_STRONG_INLINE Packet1cd pmadd(const Packet1cd &x, const Packet1cd &y, const Packet1cd &c) const + { + return padd(pmul(x, y), c); + } -template<> struct conj_helper -{ - EIGEN_STRONG_INLINE Packet1cd pmadd(const Packet1cd& x, const Packet1cd& y, const Packet1cd& c) const - { return padd(pmul(x,y),c); } + EIGEN_STRONG_INLINE Packet1cd pmul(const Packet1cd &a, const Packet1cd &b) const + { + return pconj(internal::pmul(a, b)); + } + }; - EIGEN_STRONG_INLINE Packet1cd pmul(const Packet1cd& a, const Packet1cd& b) const + EIGEN_MAKE_CONJ_HELPER_CPLX_REAL(Packet1cd, Packet2d) + + template<> EIGEN_STRONG_INLINE Packet1cd pdiv(const Packet1cd &a, const Packet1cd &b) { - return pconj(internal::pmul(a, b)); + // TODO optimize it for AltiVec + Packet1cd res = conj_helper().pmul(a, b); + Packet2d s = pmul(b.v, b.v); + return Packet1cd(pdiv(res.v, padd(s, vec_perm(s, s, p16uc_REVERSE64)))); } -}; - -EIGEN_MAKE_CONJ_HELPER_CPLX_REAL(Packet1cd,Packet2d) -template<> EIGEN_STRONG_INLINE Packet1cd pdiv(const Packet1cd& a, const Packet1cd& b) -{ - // TODO optimize it for AltiVec - Packet1cd res = conj_helper().pmul(a,b); - Packet2d s = pmul(b.v, b.v); - return Packet1cd(pdiv(res.v, padd(s, vec_perm(s, s, p16uc_REVERSE64)))); -} - -EIGEN_STRONG_INLINE Packet1cd pcplxflip/**/(const Packet1cd& x) -{ - return Packet1cd(preverse(Packet2d(x.v))); -} + EIGEN_STRONG_INLINE Packet1cd pcplxflip /**/ (const Packet1cd &x) + { + return Packet1cd(preverse(Packet2d(x.v))); + } -EIGEN_STRONG_INLINE void ptranspose(PacketBlock& kernel) -{ - Packet2d tmp = vec_perm(kernel.packet[0].v, kernel.packet[1].v, p16uc_TRANSPOSE64_HI); - kernel.packet[1].v = vec_perm(kernel.packet[0].v, kernel.packet[1].v, p16uc_TRANSPOSE64_LO); - kernel.packet[0].v = tmp; -} -#endif // __VSX__ -} // end namespace internal + EIGEN_STRONG_INLINE void ptranspose(PacketBlock &kernel) + { + Packet2d tmp = vec_perm(kernel.packet[0].v, kernel.packet[1].v, p16uc_TRANSPOSE64_HI); + kernel.packet[1].v = vec_perm(kernel.packet[0].v, kernel.packet[1].v, p16uc_TRANSPOSE64_LO); + kernel.packet[0].v = tmp; + } +#endif// __VSX__ +}// end namespace internal -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_COMPLEX32_ALTIVEC_H +#endif// EIGEN_COMPLEX32_ALTIVEC_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/arch/AltiVec/MathFunctions.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/arch/AltiVec/MathFunctions.h index c5e4bede..57c4f7b2 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/arch/AltiVec/MathFunctions.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/arch/AltiVec/MathFunctions.h @@ -20,303 +20,299 @@ namespace Eigen { namespace internal { -static _EIGEN_DECLARE_CONST_Packet4f(1 , 1.0f); -static _EIGEN_DECLARE_CONST_Packet4f(half, 0.5f); -static _EIGEN_DECLARE_CONST_Packet4i(0x7f, 0x7f); -static _EIGEN_DECLARE_CONST_Packet4i(23, 23); - -static _EIGEN_DECLARE_CONST_Packet4f_FROM_INT(inv_mant_mask, ~0x7f800000); - -/* the smallest non denormalized float number */ -static _EIGEN_DECLARE_CONST_Packet4f_FROM_INT(min_norm_pos, 0x00800000); -static _EIGEN_DECLARE_CONST_Packet4f_FROM_INT(minus_inf, 0xff800000); // -1.f/0.f -static _EIGEN_DECLARE_CONST_Packet4f_FROM_INT(minus_nan, 0xffffffff); - -/* natural logarithm computed for 4 simultaneous float - return NaN for x <= 0 -*/ -static _EIGEN_DECLARE_CONST_Packet4f(cephes_SQRTHF, 0.707106781186547524f); -static _EIGEN_DECLARE_CONST_Packet4f(cephes_log_p0, 7.0376836292E-2f); -static _EIGEN_DECLARE_CONST_Packet4f(cephes_log_p1, - 1.1514610310E-1f); -static _EIGEN_DECLARE_CONST_Packet4f(cephes_log_p2, 1.1676998740E-1f); -static _EIGEN_DECLARE_CONST_Packet4f(cephes_log_p3, - 1.2420140846E-1f); -static _EIGEN_DECLARE_CONST_Packet4f(cephes_log_p4, + 1.4249322787E-1f); -static _EIGEN_DECLARE_CONST_Packet4f(cephes_log_p5, - 1.6668057665E-1f); -static _EIGEN_DECLARE_CONST_Packet4f(cephes_log_p6, + 2.0000714765E-1f); -static _EIGEN_DECLARE_CONST_Packet4f(cephes_log_p7, - 2.4999993993E-1f); -static _EIGEN_DECLARE_CONST_Packet4f(cephes_log_p8, + 3.3333331174E-1f); -static _EIGEN_DECLARE_CONST_Packet4f(cephes_log_q1, -2.12194440e-4f); -static _EIGEN_DECLARE_CONST_Packet4f(cephes_log_q2, 0.693359375f); - -static _EIGEN_DECLARE_CONST_Packet4f(exp_hi, 88.3762626647950f); -static _EIGEN_DECLARE_CONST_Packet4f(exp_lo, -88.3762626647949f); - -static _EIGEN_DECLARE_CONST_Packet4f(cephes_LOG2EF, 1.44269504088896341f); -static _EIGEN_DECLARE_CONST_Packet4f(cephes_exp_C1, 0.693359375f); -static _EIGEN_DECLARE_CONST_Packet4f(cephes_exp_C2, -2.12194440e-4f); - -static _EIGEN_DECLARE_CONST_Packet4f(cephes_exp_p0, 1.9875691500E-4f); -static _EIGEN_DECLARE_CONST_Packet4f(cephes_exp_p1, 1.3981999507E-3f); -static _EIGEN_DECLARE_CONST_Packet4f(cephes_exp_p2, 8.3334519073E-3f); -static _EIGEN_DECLARE_CONST_Packet4f(cephes_exp_p3, 4.1665795894E-2f); -static _EIGEN_DECLARE_CONST_Packet4f(cephes_exp_p4, 1.6666665459E-1f); -static _EIGEN_DECLARE_CONST_Packet4f(cephes_exp_p5, 5.0000001201E-1f); + static _EIGEN_DECLARE_CONST_Packet4f(1, 1.0f); + static _EIGEN_DECLARE_CONST_Packet4f(half, 0.5f); + static _EIGEN_DECLARE_CONST_Packet4i(0x7f, 0x7f); + static _EIGEN_DECLARE_CONST_Packet4i(23, 23); + + static _EIGEN_DECLARE_CONST_Packet4f_FROM_INT(inv_mant_mask, ~0x7f800000); + + /* the smallest non denormalized float number */ + static _EIGEN_DECLARE_CONST_Packet4f_FROM_INT(min_norm_pos, 0x00800000); + static _EIGEN_DECLARE_CONST_Packet4f_FROM_INT(minus_inf, 0xff800000);// -1.f/0.f + static _EIGEN_DECLARE_CONST_Packet4f_FROM_INT(minus_nan, 0xffffffff); + + /* natural logarithm computed for 4 simultaneous float + return NaN for x <= 0 + */ + static _EIGEN_DECLARE_CONST_Packet4f(cephes_SQRTHF, 0.707106781186547524f); + static _EIGEN_DECLARE_CONST_Packet4f(cephes_log_p0, 7.0376836292E-2f); + static _EIGEN_DECLARE_CONST_Packet4f(cephes_log_p1, -1.1514610310E-1f); + static _EIGEN_DECLARE_CONST_Packet4f(cephes_log_p2, 1.1676998740E-1f); + static _EIGEN_DECLARE_CONST_Packet4f(cephes_log_p3, -1.2420140846E-1f); + static _EIGEN_DECLARE_CONST_Packet4f(cephes_log_p4, +1.4249322787E-1f); + static _EIGEN_DECLARE_CONST_Packet4f(cephes_log_p5, -1.6668057665E-1f); + static _EIGEN_DECLARE_CONST_Packet4f(cephes_log_p6, +2.0000714765E-1f); + static _EIGEN_DECLARE_CONST_Packet4f(cephes_log_p7, -2.4999993993E-1f); + static _EIGEN_DECLARE_CONST_Packet4f(cephes_log_p8, +3.3333331174E-1f); + static _EIGEN_DECLARE_CONST_Packet4f(cephes_log_q1, -2.12194440e-4f); + static _EIGEN_DECLARE_CONST_Packet4f(cephes_log_q2, 0.693359375f); + + static _EIGEN_DECLARE_CONST_Packet4f(exp_hi, 88.3762626647950f); + static _EIGEN_DECLARE_CONST_Packet4f(exp_lo, -88.3762626647949f); + + static _EIGEN_DECLARE_CONST_Packet4f(cephes_LOG2EF, 1.44269504088896341f); + static _EIGEN_DECLARE_CONST_Packet4f(cephes_exp_C1, 0.693359375f); + static _EIGEN_DECLARE_CONST_Packet4f(cephes_exp_C2, -2.12194440e-4f); + + static _EIGEN_DECLARE_CONST_Packet4f(cephes_exp_p0, 1.9875691500E-4f); + static _EIGEN_DECLARE_CONST_Packet4f(cephes_exp_p1, 1.3981999507E-3f); + static _EIGEN_DECLARE_CONST_Packet4f(cephes_exp_p2, 8.3334519073E-3f); + static _EIGEN_DECLARE_CONST_Packet4f(cephes_exp_p3, 4.1665795894E-2f); + static _EIGEN_DECLARE_CONST_Packet4f(cephes_exp_p4, 1.6666665459E-1f); + static _EIGEN_DECLARE_CONST_Packet4f(cephes_exp_p5, 5.0000001201E-1f); #ifdef __VSX__ -static _EIGEN_DECLARE_CONST_Packet2d(1 , 1.0); -static _EIGEN_DECLARE_CONST_Packet2d(2 , 2.0); -static _EIGEN_DECLARE_CONST_Packet2d(half, 0.5); + static _EIGEN_DECLARE_CONST_Packet2d(1, 1.0); + static _EIGEN_DECLARE_CONST_Packet2d(2, 2.0); + static _EIGEN_DECLARE_CONST_Packet2d(half, 0.5); -static _EIGEN_DECLARE_CONST_Packet2d(exp_hi, 709.437); -static _EIGEN_DECLARE_CONST_Packet2d(exp_lo, -709.436139303); + static _EIGEN_DECLARE_CONST_Packet2d(exp_hi, 709.437); + static _EIGEN_DECLARE_CONST_Packet2d(exp_lo, -709.436139303); -static _EIGEN_DECLARE_CONST_Packet2d(cephes_LOG2EF, 1.4426950408889634073599); + static _EIGEN_DECLARE_CONST_Packet2d(cephes_LOG2EF, 1.4426950408889634073599); -static _EIGEN_DECLARE_CONST_Packet2d(cephes_exp_p0, 1.26177193074810590878e-4); -static _EIGEN_DECLARE_CONST_Packet2d(cephes_exp_p1, 3.02994407707441961300e-2); -static _EIGEN_DECLARE_CONST_Packet2d(cephes_exp_p2, 9.99999999999999999910e-1); + static _EIGEN_DECLARE_CONST_Packet2d(cephes_exp_p0, 1.26177193074810590878e-4); + static _EIGEN_DECLARE_CONST_Packet2d(cephes_exp_p1, 3.02994407707441961300e-2); + static _EIGEN_DECLARE_CONST_Packet2d(cephes_exp_p2, 9.99999999999999999910e-1); -static _EIGEN_DECLARE_CONST_Packet2d(cephes_exp_q0, 3.00198505138664455042e-6); -static _EIGEN_DECLARE_CONST_Packet2d(cephes_exp_q1, 2.52448340349684104192e-3); -static _EIGEN_DECLARE_CONST_Packet2d(cephes_exp_q2, 2.27265548208155028766e-1); -static _EIGEN_DECLARE_CONST_Packet2d(cephes_exp_q3, 2.00000000000000000009e0); + static _EIGEN_DECLARE_CONST_Packet2d(cephes_exp_q0, 3.00198505138664455042e-6); + static _EIGEN_DECLARE_CONST_Packet2d(cephes_exp_q1, 2.52448340349684104192e-3); + static _EIGEN_DECLARE_CONST_Packet2d(cephes_exp_q2, 2.27265548208155028766e-1); + static _EIGEN_DECLARE_CONST_Packet2d(cephes_exp_q3, 2.00000000000000000009e0); -static _EIGEN_DECLARE_CONST_Packet2d(cephes_exp_C1, 0.693145751953125); -static _EIGEN_DECLARE_CONST_Packet2d(cephes_exp_C2, 1.42860682030941723212e-6); + static _EIGEN_DECLARE_CONST_Packet2d(cephes_exp_C1, 0.693145751953125); + static _EIGEN_DECLARE_CONST_Packet2d(cephes_exp_C2, 1.42860682030941723212e-6); #ifdef __POWER8_VECTOR__ -static Packet2l p2l_1023 = { 1023, 1023 }; -static Packet2ul p2ul_52 = { 52, 52 }; + static Packet2l p2l_1023 = { 1023, 1023 }; + static Packet2ul p2ul_52 = { 52, 52 }; #endif #endif -template<> EIGEN_DEFINE_FUNCTION_ALLOWING_MULTIPLE_DEFINITIONS EIGEN_UNUSED -Packet4f plog(const Packet4f& _x) -{ - Packet4f x = _x; - - Packet4i emm0; - - /* isvalid_mask is 0 if x < 0 or x is NaN. */ - Packet4ui isvalid_mask = reinterpret_cast(vec_cmpge(x, p4f_ZERO)); - Packet4ui iszero_mask = reinterpret_cast(vec_cmpeq(x, p4f_ZERO)); - - x = pmax(x, p4f_min_norm_pos); /* cut off denormalized stuff */ - emm0 = vec_sr(reinterpret_cast(x), - reinterpret_cast(p4i_23)); - - /* keep only the fractional part */ - x = pand(x, p4f_inv_mant_mask); - x = por(x, p4f_half); - - emm0 = psub(emm0, p4i_0x7f); - Packet4f e = padd(vec_ctf(emm0, 0), p4f_1); - - /* part2: - if( x < SQRTHF ) { - e -= 1; - x = x + x - 1.0; - } else { x = x - 1.0; } - */ - Packet4f mask = reinterpret_cast(vec_cmplt(x, p4f_cephes_SQRTHF)); - Packet4f tmp = pand(x, mask); - x = psub(x, p4f_1); - e = psub(e, pand(p4f_1, mask)); - x = padd(x, tmp); - - Packet4f x2 = pmul(x,x); - Packet4f x3 = pmul(x2,x); - - Packet4f y, y1, y2; - y = pmadd(p4f_cephes_log_p0, x, p4f_cephes_log_p1); - y1 = pmadd(p4f_cephes_log_p3, x, p4f_cephes_log_p4); - y2 = pmadd(p4f_cephes_log_p6, x, p4f_cephes_log_p7); - y = pmadd(y , x, p4f_cephes_log_p2); - y1 = pmadd(y1, x, p4f_cephes_log_p5); - y2 = pmadd(y2, x, p4f_cephes_log_p8); - y = pmadd(y, x3, y1); - y = pmadd(y, x3, y2); - y = pmul(y, x3); - - y1 = pmul(e, p4f_cephes_log_q1); - tmp = pmul(x2, p4f_half); - y = padd(y, y1); - x = psub(x, tmp); - y2 = pmul(e, p4f_cephes_log_q2); - x = padd(x, y); - x = padd(x, y2); - // negative arg will be NAN, 0 will be -INF - x = vec_sel(x, p4f_minus_inf, iszero_mask); - x = vec_sel(p4f_minus_nan, x, isvalid_mask); - return x; -} - -template<> EIGEN_DEFINE_FUNCTION_ALLOWING_MULTIPLE_DEFINITIONS EIGEN_UNUSED -Packet4f pexp(const Packet4f& _x) -{ - Packet4f x = _x; - - Packet4f tmp, fx; - Packet4i emm0; - - // clamp x - x = pmax(pmin(x, p4f_exp_hi), p4f_exp_lo); - - // express exp(x) as exp(g + n*log(2)) - fx = pmadd(x, p4f_cephes_LOG2EF, p4f_half); - - fx = pfloor(fx); - - tmp = pmul(fx, p4f_cephes_exp_C1); - Packet4f z = pmul(fx, p4f_cephes_exp_C2); - x = psub(x, tmp); - x = psub(x, z); - - z = pmul(x,x); - - Packet4f y = p4f_cephes_exp_p0; - y = pmadd(y, x, p4f_cephes_exp_p1); - y = pmadd(y, x, p4f_cephes_exp_p2); - y = pmadd(y, x, p4f_cephes_exp_p3); - y = pmadd(y, x, p4f_cephes_exp_p4); - y = pmadd(y, x, p4f_cephes_exp_p5); - y = pmadd(y, z, x); - y = padd(y, p4f_1); - - // build 2^n - emm0 = vec_cts(fx, 0); - emm0 = vec_add(emm0, p4i_0x7f); - emm0 = vec_sl(emm0, reinterpret_cast(p4i_23)); - - // Altivec's max & min operators just drop silent NaNs. Check NaNs in - // inputs and return them unmodified. - Packet4ui isnumber_mask = reinterpret_cast(vec_cmpeq(_x, _x)); - return vec_sel(_x, pmax(pmul(y, reinterpret_cast(emm0)), _x), - isnumber_mask); -} + template<> + EIGEN_DEFINE_FUNCTION_ALLOWING_MULTIPLE_DEFINITIONS EIGEN_UNUSED Packet4f plog(const Packet4f &_x) + { + Packet4f x = _x; + + Packet4i emm0; + + /* isvalid_mask is 0 if x < 0 or x is NaN. */ + Packet4ui isvalid_mask = reinterpret_cast(vec_cmpge(x, p4f_ZERO)); + Packet4ui iszero_mask = reinterpret_cast(vec_cmpeq(x, p4f_ZERO)); + + x = pmax(x, p4f_min_norm_pos); /* cut off denormalized stuff */ + emm0 = vec_sr(reinterpret_cast(x), reinterpret_cast(p4i_23)); + + /* keep only the fractional part */ + x = pand(x, p4f_inv_mant_mask); + x = por(x, p4f_half); + + emm0 = psub(emm0, p4i_0x7f); + Packet4f e = padd(vec_ctf(emm0, 0), p4f_1); + + /* part2: + if( x < SQRTHF ) { + e -= 1; + x = x + x - 1.0; + } else { x = x - 1.0; } + */ + Packet4f mask = reinterpret_cast(vec_cmplt(x, p4f_cephes_SQRTHF)); + Packet4f tmp = pand(x, mask); + x = psub(x, p4f_1); + e = psub(e, pand(p4f_1, mask)); + x = padd(x, tmp); + + Packet4f x2 = pmul(x, x); + Packet4f x3 = pmul(x2, x); + + Packet4f y, y1, y2; + y = pmadd(p4f_cephes_log_p0, x, p4f_cephes_log_p1); + y1 = pmadd(p4f_cephes_log_p3, x, p4f_cephes_log_p4); + y2 = pmadd(p4f_cephes_log_p6, x, p4f_cephes_log_p7); + y = pmadd(y, x, p4f_cephes_log_p2); + y1 = pmadd(y1, x, p4f_cephes_log_p5); + y2 = pmadd(y2, x, p4f_cephes_log_p8); + y = pmadd(y, x3, y1); + y = pmadd(y, x3, y2); + y = pmul(y, x3); + + y1 = pmul(e, p4f_cephes_log_q1); + tmp = pmul(x2, p4f_half); + y = padd(y, y1); + x = psub(x, tmp); + y2 = pmul(e, p4f_cephes_log_q2); + x = padd(x, y); + x = padd(x, y2); + // negative arg will be NAN, 0 will be -INF + x = vec_sel(x, p4f_minus_inf, iszero_mask); + x = vec_sel(p4f_minus_nan, x, isvalid_mask); + return x; + } + + template<> + EIGEN_DEFINE_FUNCTION_ALLOWING_MULTIPLE_DEFINITIONS EIGEN_UNUSED Packet4f pexp(const Packet4f &_x) + { + Packet4f x = _x; + + Packet4f tmp, fx; + Packet4i emm0; + + // clamp x + x = pmax(pmin(x, p4f_exp_hi), p4f_exp_lo); + + // express exp(x) as exp(g + n*log(2)) + fx = pmadd(x, p4f_cephes_LOG2EF, p4f_half); + + fx = pfloor(fx); + + tmp = pmul(fx, p4f_cephes_exp_C1); + Packet4f z = pmul(fx, p4f_cephes_exp_C2); + x = psub(x, tmp); + x = psub(x, z); + + z = pmul(x, x); + + Packet4f y = p4f_cephes_exp_p0; + y = pmadd(y, x, p4f_cephes_exp_p1); + y = pmadd(y, x, p4f_cephes_exp_p2); + y = pmadd(y, x, p4f_cephes_exp_p3); + y = pmadd(y, x, p4f_cephes_exp_p4); + y = pmadd(y, x, p4f_cephes_exp_p5); + y = pmadd(y, z, x); + y = padd(y, p4f_1); + + // build 2^n + emm0 = vec_cts(fx, 0); + emm0 = vec_add(emm0, p4i_0x7f); + emm0 = vec_sl(emm0, reinterpret_cast(p4i_23)); + + // Altivec's max & min operators just drop silent NaNs. Check NaNs in + // inputs and return them unmodified. + Packet4ui isnumber_mask = reinterpret_cast(vec_cmpeq(_x, _x)); + return vec_sel(_x, pmax(pmul(y, reinterpret_cast(emm0)), _x), isnumber_mask); + } #ifndef EIGEN_COMP_CLANG -template<> EIGEN_DEFINE_FUNCTION_ALLOWING_MULTIPLE_DEFINITIONS EIGEN_UNUSED -Packet4f prsqrt(const Packet4f& x) -{ - return vec_rsqrt(x); -} + template<> + EIGEN_DEFINE_FUNCTION_ALLOWING_MULTIPLE_DEFINITIONS EIGEN_UNUSED Packet4f prsqrt(const Packet4f &x) + { + return vec_rsqrt(x); + } #endif #ifdef __VSX__ #ifndef EIGEN_COMP_CLANG -template<> EIGEN_DEFINE_FUNCTION_ALLOWING_MULTIPLE_DEFINITIONS EIGEN_UNUSED -Packet2d prsqrt(const Packet2d& x) -{ - return vec_rsqrt(x); -} + template<> + EIGEN_DEFINE_FUNCTION_ALLOWING_MULTIPLE_DEFINITIONS EIGEN_UNUSED Packet2d prsqrt(const Packet2d &x) + { + return vec_rsqrt(x); + } #endif -template<> EIGEN_DEFINE_FUNCTION_ALLOWING_MULTIPLE_DEFINITIONS EIGEN_UNUSED -Packet4f psqrt(const Packet4f& x) -{ - return vec_sqrt(x); -} - -template<> EIGEN_DEFINE_FUNCTION_ALLOWING_MULTIPLE_DEFINITIONS EIGEN_UNUSED -Packet2d psqrt(const Packet2d& x) -{ - return vec_sqrt(x); -} - -// VSX support varies between different compilers and even different -// versions of the same compiler. For gcc version >= 4.9.3, we can use -// vec_cts to efficiently convert Packet2d to Packet2l. Otherwise, use -// a slow version that works with older compilers. -// Update: apparently vec_cts/vec_ctf intrinsics for 64-bit doubles -// are buggy, https://gcc.gnu.org/bugzilla/show_bug.cgi?id=70963 -static inline Packet2l ConvertToPacket2l(const Packet2d& x) { -#if EIGEN_GNUC_AT_LEAST(5, 4) || \ - (EIGEN_GNUC_AT(6, 1) && __GNUC_PATCHLEVEL__ >= 1) - return vec_cts(x, 0); // TODO: check clang version. + template<> + EIGEN_DEFINE_FUNCTION_ALLOWING_MULTIPLE_DEFINITIONS EIGEN_UNUSED Packet4f psqrt(const Packet4f &x) + { + return vec_sqrt(x); + } + + template<> + EIGEN_DEFINE_FUNCTION_ALLOWING_MULTIPLE_DEFINITIONS EIGEN_UNUSED Packet2d psqrt(const Packet2d &x) + { + return vec_sqrt(x); + } + + // VSX support varies between different compilers and even different + // versions of the same compiler. For gcc version >= 4.9.3, we can use + // vec_cts to efficiently convert Packet2d to Packet2l. Otherwise, use + // a slow version that works with older compilers. + // Update: apparently vec_cts/vec_ctf intrinsics for 64-bit doubles + // are buggy, https://gcc.gnu.org/bugzilla/show_bug.cgi?id=70963 + static inline Packet2l ConvertToPacket2l(const Packet2d &x) + { +#if EIGEN_GNUC_AT_LEAST(5, 4) || (EIGEN_GNUC_AT(6, 1) && __GNUC_PATCHLEVEL__ >= 1) + return vec_cts(x, 0);// TODO: check clang version. #else - double tmp[2]; - memcpy(tmp, &x, sizeof(tmp)); - Packet2l l = { static_cast(tmp[0]), - static_cast(tmp[1]) }; - return l; + double tmp[2]; + memcpy(tmp, &x, sizeof(tmp)); + Packet2l l = { static_cast(tmp[0]), static_cast(tmp[1]) }; + return l; #endif -} + } -template<> EIGEN_DEFINE_FUNCTION_ALLOWING_MULTIPLE_DEFINITIONS EIGEN_UNUSED -Packet2d pexp(const Packet2d& _x) -{ - Packet2d x = _x; + template<> + EIGEN_DEFINE_FUNCTION_ALLOWING_MULTIPLE_DEFINITIONS EIGEN_UNUSED Packet2d pexp(const Packet2d &_x) + { + Packet2d x = _x; - Packet2d tmp, fx; - Packet2l emm0; + Packet2d tmp, fx; + Packet2l emm0; - // clamp x - x = pmax(pmin(x, p2d_exp_hi), p2d_exp_lo); + // clamp x + x = pmax(pmin(x, p2d_exp_hi), p2d_exp_lo); - /* express exp(x) as exp(g + n*log(2)) */ - fx = pmadd(x, p2d_cephes_LOG2EF, p2d_half); + /* express exp(x) as exp(g + n*log(2)) */ + fx = pmadd(x, p2d_cephes_LOG2EF, p2d_half); - fx = pfloor(fx); + fx = pfloor(fx); - tmp = pmul(fx, p2d_cephes_exp_C1); - Packet2d z = pmul(fx, p2d_cephes_exp_C2); - x = psub(x, tmp); - x = psub(x, z); + tmp = pmul(fx, p2d_cephes_exp_C1); + Packet2d z = pmul(fx, p2d_cephes_exp_C2); + x = psub(x, tmp); + x = psub(x, z); - Packet2d x2 = pmul(x,x); + Packet2d x2 = pmul(x, x); - Packet2d px = p2d_cephes_exp_p0; - px = pmadd(px, x2, p2d_cephes_exp_p1); - px = pmadd(px, x2, p2d_cephes_exp_p2); - px = pmul (px, x); + Packet2d px = p2d_cephes_exp_p0; + px = pmadd(px, x2, p2d_cephes_exp_p1); + px = pmadd(px, x2, p2d_cephes_exp_p2); + px = pmul(px, x); - Packet2d qx = p2d_cephes_exp_q0; - qx = pmadd(qx, x2, p2d_cephes_exp_q1); - qx = pmadd(qx, x2, p2d_cephes_exp_q2); - qx = pmadd(qx, x2, p2d_cephes_exp_q3); + Packet2d qx = p2d_cephes_exp_q0; + qx = pmadd(qx, x2, p2d_cephes_exp_q1); + qx = pmadd(qx, x2, p2d_cephes_exp_q2); + qx = pmadd(qx, x2, p2d_cephes_exp_q3); - x = pdiv(px,psub(qx,px)); - x = pmadd(p2d_2,x,p2d_1); + x = pdiv(px, psub(qx, px)); + x = pmadd(p2d_2, x, p2d_1); - // build 2^n - emm0 = ConvertToPacket2l(fx); + // build 2^n + emm0 = ConvertToPacket2l(fx); -#ifdef __POWER8_VECTOR__ - emm0 = vec_add(emm0, p2l_1023); - emm0 = vec_sl(emm0, p2ul_52); +#ifdef __POWER8_VECTOR__ + emm0 = vec_add(emm0, p2l_1023); + emm0 = vec_sl(emm0, p2ul_52); #else - // Code is a bit complex for POWER7. There is actually a - // vec_xxsldi intrinsic but it is not supported by some gcc versions. - // So we shift (52-32) bits and do a word swap with zeros. - _EIGEN_DECLARE_CONST_Packet4i(1023, 1023); - _EIGEN_DECLARE_CONST_Packet4i(20, 20); // 52 - 32 - - Packet4i emm04i = reinterpret_cast(emm0); - emm04i = vec_add(emm04i, p4i_1023); - emm04i = vec_sl(emm04i, reinterpret_cast(p4i_20)); - static const Packet16uc perm = { - 0x14, 0x15, 0x16, 0x17, 0x00, 0x01, 0x02, 0x03, - 0x1c, 0x1d, 0x1e, 0x1f, 0x08, 0x09, 0x0a, 0x0b }; -#ifdef _BIG_ENDIAN - emm0 = reinterpret_cast(vec_perm(p4i_ZERO, emm04i, perm)); + // Code is a bit complex for POWER7. There is actually a + // vec_xxsldi intrinsic but it is not supported by some gcc versions. + // So we shift (52-32) bits and do a word swap with zeros. + _EIGEN_DECLARE_CONST_Packet4i(1023, 1023); + _EIGEN_DECLARE_CONST_Packet4i(20, 20);// 52 - 32 + + Packet4i emm04i = reinterpret_cast(emm0); + emm04i = vec_add(emm04i, p4i_1023); + emm04i = vec_sl(emm04i, reinterpret_cast(p4i_20)); + static const Packet16uc perm = { + 0x14, 0x15, 0x16, 0x17, 0x00, 0x01, 0x02, 0x03, 0x1c, 0x1d, 0x1e, 0x1f, 0x08, 0x09, 0x0a, 0x0b + }; +#ifdef _BIG_ENDIAN + emm0 = reinterpret_cast(vec_perm(p4i_ZERO, emm04i, perm)); #else - emm0 = reinterpret_cast(vec_perm(emm04i, p4i_ZERO, perm)); + emm0 = reinterpret_cast(vec_perm(emm04i, p4i_ZERO, perm)); #endif #endif - // Altivec's max & min operators just drop silent NaNs. Check NaNs in - // inputs and return them unmodified. - Packet2ul isnumber_mask = reinterpret_cast(vec_cmpeq(_x, _x)); - return vec_sel(_x, pmax(pmul(x, reinterpret_cast(emm0)), _x), - isnumber_mask); -} + // Altivec's max & min operators just drop silent NaNs. Check NaNs in + // inputs and return them unmodified. + Packet2ul isnumber_mask = reinterpret_cast(vec_cmpeq(_x, _x)); + return vec_sel(_x, pmax(pmul(x, reinterpret_cast(emm0)), _x), isnumber_mask); + } #endif -} // end namespace internal +}// end namespace internal -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_MATH_FUNCTIONS_ALTIVEC_H +#endif// EIGEN_MATH_FUNCTIONS_ALTIVEC_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/arch/AltiVec/PacketMath.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/arch/AltiVec/PacketMath.h old mode 100755 new mode 100644 index 08a27d15..1fa58eeb --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/arch/AltiVec/PacketMath.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/arch/AltiVec/PacketMath.h @@ -28,1034 +28,1150 @@ namespace internal { // NOTE Altivec has 32 registers, but Eigen only accepts a value of 8 or 16 #ifndef EIGEN_ARCH_DEFAULT_NUMBER_OF_REGISTERS -#define EIGEN_ARCH_DEFAULT_NUMBER_OF_REGISTERS 32 +#define EIGEN_ARCH_DEFAULT_NUMBER_OF_REGISTERS 32 #endif -typedef __vector float Packet4f; -typedef __vector int Packet4i; -typedef __vector unsigned int Packet4ui; -typedef __vector __bool int Packet4bi; -typedef __vector short int Packet8i; -typedef __vector unsigned char Packet16uc; + typedef __vector float Packet4f; + typedef __vector int Packet4i; + typedef __vector unsigned int Packet4ui; + typedef __vector __bool int Packet4bi; + typedef __vector short int Packet8i; + typedef __vector unsigned char Packet16uc; -// We don't want to write the same code all the time, but we need to reuse the constants -// and it doesn't really work to declare them global, so we define macros instead + // We don't want to write the same code all the time, but we need to reuse the constants + // and it doesn't really work to declare them global, so we define macros instead -#define _EIGEN_DECLARE_CONST_FAST_Packet4f(NAME,X) \ - Packet4f p4f_##NAME = reinterpret_cast(vec_splat_s32(X)) +#define _EIGEN_DECLARE_CONST_FAST_Packet4f(NAME, X) Packet4f p4f_##NAME = reinterpret_cast(vec_splat_s32(X)) -#define _EIGEN_DECLARE_CONST_FAST_Packet4i(NAME,X) \ - Packet4i p4i_##NAME = vec_splat_s32(X) +#define _EIGEN_DECLARE_CONST_FAST_Packet4i(NAME, X) Packet4i p4i_##NAME = vec_splat_s32(X) -#define _EIGEN_DECLARE_CONST_Packet4f(NAME,X) \ - Packet4f p4f_##NAME = pset1(X) +#define _EIGEN_DECLARE_CONST_Packet4f(NAME, X) Packet4f p4f_##NAME = pset1(X) -#define _EIGEN_DECLARE_CONST_Packet4i(NAME,X) \ - Packet4i p4i_##NAME = pset1(X) +#define _EIGEN_DECLARE_CONST_Packet4i(NAME, X) Packet4i p4i_##NAME = pset1(X) -#define _EIGEN_DECLARE_CONST_Packet2d(NAME,X) \ - Packet2d p2d_##NAME = pset1(X) +#define _EIGEN_DECLARE_CONST_Packet2d(NAME, X) Packet2d p2d_##NAME = pset1(X) -#define _EIGEN_DECLARE_CONST_Packet2l(NAME,X) \ - Packet2l p2l_##NAME = pset1(X) +#define _EIGEN_DECLARE_CONST_Packet2l(NAME, X) Packet2l p2l_##NAME = pset1(X) -#define _EIGEN_DECLARE_CONST_Packet4f_FROM_INT(NAME,X) \ +#define _EIGEN_DECLARE_CONST_Packet4f_FROM_INT(NAME, X) \ const Packet4f p4f_##NAME = reinterpret_cast(pset1(X)) #define DST_CHAN 1 #define DST_CTRL(size, count, stride) (((size) << 24) | ((count) << 16) | (stride)) -// These constants are endian-agnostic -static _EIGEN_DECLARE_CONST_FAST_Packet4f(ZERO, 0); //{ 0.0, 0.0, 0.0, 0.0} -static _EIGEN_DECLARE_CONST_FAST_Packet4i(ZERO, 0); //{ 0, 0, 0, 0,} -static _EIGEN_DECLARE_CONST_FAST_Packet4i(ONE,1); //{ 1, 1, 1, 1} -static _EIGEN_DECLARE_CONST_FAST_Packet4i(MINUS16,-16); //{ -16, -16, -16, -16} -static _EIGEN_DECLARE_CONST_FAST_Packet4i(MINUS1,-1); //{ -1, -1, -1, -1} -static Packet4f p4f_MZERO = (Packet4f) vec_sl((Packet4ui)p4i_MINUS1, (Packet4ui)p4i_MINUS1); //{ 0x80000000, 0x80000000, 0x80000000, 0x80000000} + // These constants are endian-agnostic + static _EIGEN_DECLARE_CONST_FAST_Packet4f(ZERO, 0);//{ 0.0, 0.0, 0.0, 0.0} + static _EIGEN_DECLARE_CONST_FAST_Packet4i(ZERO, 0);//{ 0, 0, 0, 0,} + static _EIGEN_DECLARE_CONST_FAST_Packet4i(ONE, 1);//{ 1, 1, 1, 1} + static _EIGEN_DECLARE_CONST_FAST_Packet4i(MINUS16, -16);//{ -16, -16, -16, -16} + static _EIGEN_DECLARE_CONST_FAST_Packet4i(MINUS1, -1);//{ -1, -1, -1, -1} + static Packet4f p4f_MZERO = + (Packet4f)vec_sl((Packet4ui)p4i_MINUS1, (Packet4ui)p4i_MINUS1);//{ 0x80000000, 0x80000000, 0x80000000, 0x80000000} #ifndef __VSX__ -static Packet4f p4f_ONE = vec_ctf(p4i_ONE, 0); //{ 1.0, 1.0, 1.0, 1.0} + static Packet4f p4f_ONE = vec_ctf(p4i_ONE, 0);//{ 1.0, 1.0, 1.0, 1.0} #endif -static Packet4f p4f_COUNTDOWN = { 0.0, 1.0, 2.0, 3.0 }; -static Packet4i p4i_COUNTDOWN = { 0, 1, 2, 3 }; + static Packet4f p4f_COUNTDOWN = { 0.0, 1.0, 2.0, 3.0 }; + static Packet4i p4i_COUNTDOWN = { 0, 1, 2, 3 }; -static Packet16uc p16uc_REVERSE32 = { 12,13,14,15, 8,9,10,11, 4,5,6,7, 0,1,2,3 }; -static Packet16uc p16uc_DUPLICATE32_HI = { 0,1,2,3, 0,1,2,3, 4,5,6,7, 4,5,6,7 }; + static Packet16uc p16uc_REVERSE32 = { 12, 13, 14, 15, 8, 9, 10, 11, 4, 5, 6, 7, 0, 1, 2, 3 }; + static Packet16uc p16uc_DUPLICATE32_HI = { 0, 1, 2, 3, 0, 1, 2, 3, 4, 5, 6, 7, 4, 5, 6, 7 }; // Mask alignment #ifdef __PPC64__ -#define _EIGEN_MASK_ALIGNMENT 0xfffffffffffffff0 +#define _EIGEN_MASK_ALIGNMENT 0xfffffffffffffff0 #else -#define _EIGEN_MASK_ALIGNMENT 0xfffffff0 +#define _EIGEN_MASK_ALIGNMENT 0xfffffff0 #endif -#define _EIGEN_ALIGNED_PTR(x) ((std::ptrdiff_t)(x) & _EIGEN_MASK_ALIGNMENT) +#define _EIGEN_ALIGNED_PTR(x) ((std::ptrdiff_t)(x) & _EIGEN_MASK_ALIGNMENT) // Handle endianness properly while loading constants // Define global static constants: #ifdef _BIG_ENDIAN -static Packet16uc p16uc_FORWARD = vec_lvsl(0, (float*)0); + static Packet16uc p16uc_FORWARD = vec_lvsl(0, (float *)0); #ifdef __VSX__ -static Packet16uc p16uc_REVERSE64 = { 8,9,10,11, 12,13,14,15, 0,1,2,3, 4,5,6,7 }; + static Packet16uc p16uc_REVERSE64 = { 8, 9, 10, 11, 12, 13, 14, 15, 0, 1, 2, 3, 4, 5, 6, 7 }; #endif -static Packet16uc p16uc_PSET32_WODD = vec_sld((Packet16uc) vec_splat((Packet4ui)p16uc_FORWARD, 0), (Packet16uc) vec_splat((Packet4ui)p16uc_FORWARD, 2), 8);//{ 0,1,2,3, 0,1,2,3, 8,9,10,11, 8,9,10,11 }; -static Packet16uc p16uc_PSET32_WEVEN = vec_sld(p16uc_DUPLICATE32_HI, (Packet16uc) vec_splat((Packet4ui)p16uc_FORWARD, 3), 8);//{ 4,5,6,7, 4,5,6,7, 12,13,14,15, 12,13,14,15 }; -static Packet16uc p16uc_HALF64_0_16 = vec_sld((Packet16uc)p4i_ZERO, vec_splat((Packet16uc) vec_abs(p4i_MINUS16), 3), 8); //{ 0,0,0,0, 0,0,0,0, 16,16,16,16, 16,16,16,16}; + static Packet16uc p16uc_PSET32_WODD = vec_sld((Packet16uc)vec_splat((Packet4ui)p16uc_FORWARD, 0), + (Packet16uc)vec_splat((Packet4ui)p16uc_FORWARD, 2), + 8);//{ 0,1,2,3, 0,1,2,3, 8,9,10,11, 8,9,10,11 }; + static Packet16uc p16uc_PSET32_WEVEN = vec_sld(p16uc_DUPLICATE32_HI, + (Packet16uc)vec_splat((Packet4ui)p16uc_FORWARD, 3), + 8);//{ 4,5,6,7, 4,5,6,7, 12,13,14,15, 12,13,14,15 }; + static Packet16uc p16uc_HALF64_0_16 = vec_sld((Packet16uc)p4i_ZERO, + vec_splat((Packet16uc)vec_abs(p4i_MINUS16), 3), + 8);//{ 0,0,0,0, 0,0,0,0, 16,16,16,16, 16,16,16,16}; #else -static Packet16uc p16uc_FORWARD = p16uc_REVERSE32; -static Packet16uc p16uc_REVERSE64 = { 8,9,10,11, 12,13,14,15, 0,1,2,3, 4,5,6,7 }; -static Packet16uc p16uc_PSET32_WODD = vec_sld((Packet16uc) vec_splat((Packet4ui)p16uc_FORWARD, 1), (Packet16uc) vec_splat((Packet4ui)p16uc_FORWARD, 3), 8);//{ 0,1,2,3, 0,1,2,3, 8,9,10,11, 8,9,10,11 }; -static Packet16uc p16uc_PSET32_WEVEN = vec_sld((Packet16uc) vec_splat((Packet4ui)p16uc_FORWARD, 0), (Packet16uc) vec_splat((Packet4ui)p16uc_FORWARD, 2), 8);//{ 4,5,6,7, 4,5,6,7, 12,13,14,15, 12,13,14,15 }; -static Packet16uc p16uc_HALF64_0_16 = vec_sld(vec_splat((Packet16uc) vec_abs(p4i_MINUS16), 0), (Packet16uc)p4i_ZERO, 8); //{ 0,0,0,0, 0,0,0,0, 16,16,16,16, 16,16,16,16}; -#endif // _BIG_ENDIAN - -static Packet16uc p16uc_PSET64_HI = (Packet16uc) vec_mergeh((Packet4ui)p16uc_PSET32_WODD, (Packet4ui)p16uc_PSET32_WEVEN); //{ 0,1,2,3, 4,5,6,7, 0,1,2,3, 4,5,6,7 }; -static Packet16uc p16uc_PSET64_LO = (Packet16uc) vec_mergel((Packet4ui)p16uc_PSET32_WODD, (Packet4ui)p16uc_PSET32_WEVEN); //{ 8,9,10,11, 12,13,14,15, 8,9,10,11, 12,13,14,15 }; -static Packet16uc p16uc_TRANSPOSE64_HI = p16uc_PSET64_HI + p16uc_HALF64_0_16; //{ 0,1,2,3, 4,5,6,7, 16,17,18,19, 20,21,22,23}; -static Packet16uc p16uc_TRANSPOSE64_LO = p16uc_PSET64_LO + p16uc_HALF64_0_16; //{ 8,9,10,11, 12,13,14,15, 24,25,26,27, 28,29,30,31}; - -static Packet16uc p16uc_COMPLEX32_REV = vec_sld(p16uc_REVERSE32, p16uc_REVERSE32, 8); //{ 4,5,6,7, 0,1,2,3, 12,13,14,15, 8,9,10,11 }; + static Packet16uc p16uc_FORWARD = p16uc_REVERSE32; + static Packet16uc p16uc_REVERSE64 = { 8, 9, 10, 11, 12, 13, 14, 15, 0, 1, 2, 3, 4, 5, 6, 7 }; + static Packet16uc p16uc_PSET32_WODD = vec_sld((Packet16uc)vec_splat((Packet4ui)p16uc_FORWARD, 1), + (Packet16uc)vec_splat((Packet4ui)p16uc_FORWARD, 3), + 8);//{ 0,1,2,3, 0,1,2,3, 8,9,10,11, 8,9,10,11 }; + static Packet16uc p16uc_PSET32_WEVEN = vec_sld((Packet16uc)vec_splat((Packet4ui)p16uc_FORWARD, 0), + (Packet16uc)vec_splat((Packet4ui)p16uc_FORWARD, 2), + 8);//{ 4,5,6,7, 4,5,6,7, 12,13,14,15, 12,13,14,15 }; + static Packet16uc p16uc_HALF64_0_16 = vec_sld(vec_splat((Packet16uc)vec_abs(p4i_MINUS16), 0), + (Packet16uc)p4i_ZERO, + 8);//{ 0,0,0,0, 0,0,0,0, 16,16,16,16, 16,16,16,16}; +#endif// _BIG_ENDIAN + + static Packet16uc p16uc_PSET64_HI = (Packet16uc)vec_mergeh((Packet4ui)p16uc_PSET32_WODD, + (Packet4ui)p16uc_PSET32_WEVEN);//{ 0,1,2,3, 4,5,6,7, 0,1,2,3, 4,5,6,7 }; + static Packet16uc p16uc_PSET64_LO = (Packet16uc)vec_mergel((Packet4ui)p16uc_PSET32_WODD, + (Packet4ui)p16uc_PSET32_WEVEN);//{ 8,9,10,11, 12,13,14,15, 8,9,10,11, 12,13,14,15 }; + static Packet16uc p16uc_TRANSPOSE64_HI = + p16uc_PSET64_HI + p16uc_HALF64_0_16;//{ 0,1,2,3, 4,5,6,7, 16,17,18,19, 20,21,22,23}; + static Packet16uc p16uc_TRANSPOSE64_LO = + p16uc_PSET64_LO + p16uc_HALF64_0_16;//{ 8,9,10,11, 12,13,14,15, 24,25,26,27, 28,29,30,31}; + + static Packet16uc p16uc_COMPLEX32_REV = + vec_sld(p16uc_REVERSE32, p16uc_REVERSE32, 8);//{ 4,5,6,7, 0,1,2,3, 12,13,14,15, 8,9,10,11 }; #ifdef _BIG_ENDIAN -static Packet16uc p16uc_COMPLEX32_REV2 = vec_sld(p16uc_FORWARD, p16uc_FORWARD, 8); //{ 8,9,10,11, 12,13,14,15, 0,1,2,3, 4,5,6,7 }; + static Packet16uc p16uc_COMPLEX32_REV2 = + vec_sld(p16uc_FORWARD, p16uc_FORWARD, 8);//{ 8,9,10,11, 12,13,14,15, 0,1,2,3, 4,5,6,7 }; #else -static Packet16uc p16uc_COMPLEX32_REV2 = vec_sld(p16uc_PSET64_HI, p16uc_PSET64_LO, 8); //{ 8,9,10,11, 12,13,14,15, 0,1,2,3, 4,5,6,7 }; -#endif // _BIG_ENDIAN + static Packet16uc p16uc_COMPLEX32_REV2 = + vec_sld(p16uc_PSET64_HI, p16uc_PSET64_LO, 8);//{ 8,9,10,11, 12,13,14,15, 0,1,2,3, 4,5,6,7 }; +#endif// _BIG_ENDIAN #if EIGEN_HAS_BUILTIN(__builtin_prefetch) || EIGEN_COMP_GNUC - #define EIGEN_PPC_PREFETCH(ADDR) __builtin_prefetch(ADDR); +#define EIGEN_PPC_PREFETCH(ADDR) __builtin_prefetch(ADDR); #else - #define EIGEN_PPC_PREFETCH(ADDR) asm( " dcbt [%[addr]]\n" :: [addr] "r" (ADDR) : "cc" ); +#define EIGEN_PPC_PREFETCH(ADDR) asm(" dcbt [%[addr]]\n" ::[addr] "r"(ADDR) : "cc"); #endif -template<> struct packet_traits : default_packet_traits -{ - typedef Packet4f type; - typedef Packet4f half; - enum { - Vectorizable = 1, - AlignedOnScalar = 1, - size=4, - HasHalfPacket = 1, - - HasAdd = 1, - HasSub = 1, - HasMul = 1, - HasDiv = 1, - HasMin = 1, - HasMax = 1, - HasAbs = 1, - HasSin = 0, - HasCos = 0, - HasLog = 0, - HasExp = 1, + template<> struct packet_traits : default_packet_traits + { + typedef Packet4f type; + typedef Packet4f half; + enum { + Vectorizable = 1, + AlignedOnScalar = 1, + size = 4, + HasHalfPacket = 1, + + HasAdd = 1, + HasSub = 1, + HasMul = 1, + HasDiv = 1, + HasMin = 1, + HasMax = 1, + HasAbs = 1, + HasSin = 0, + HasCos = 0, + HasLog = 0, + HasExp = 1, #ifdef __VSX__ - HasSqrt = 1, + HasSqrt = 1, #if !EIGEN_COMP_CLANG - HasRsqrt = 1, + HasRsqrt = 1, #else - HasRsqrt = 0, + HasRsqrt = 0, #endif #else - HasSqrt = 0, - HasRsqrt = 0, + HasSqrt = 0, + HasRsqrt = 0, #endif - HasRound = 1, - HasFloor = 1, - HasCeil = 1, - HasNegate = 1, - HasBlend = 1 + HasRound = 1, + HasFloor = 1, + HasCeil = 1, + HasNegate = 1, + HasBlend = 1 + }; }; -}; -template<> struct packet_traits : default_packet_traits -{ - typedef Packet4i type; - typedef Packet4i half; - enum { - Vectorizable = 1, - AlignedOnScalar = 1, - size = 4, - HasHalfPacket = 0, - - HasAdd = 1, - HasSub = 1, - HasMul = 1, - HasDiv = 0, - HasBlend = 1 + template<> struct packet_traits : default_packet_traits + { + typedef Packet4i type; + typedef Packet4i half; + enum { + Vectorizable = 1, + AlignedOnScalar = 1, + size = 4, + HasHalfPacket = 0, + + HasAdd = 1, + HasSub = 1, + HasMul = 1, + HasDiv = 0, + HasBlend = 1 + }; }; -}; - - -template<> struct unpacket_traits { typedef float type; enum {size=4, alignment=Aligned16}; typedef Packet4f half; }; -template<> struct unpacket_traits { typedef int type; enum {size=4, alignment=Aligned16}; typedef Packet4i half; }; - -inline std::ostream & operator <<(std::ostream & s, const Packet16uc & v) -{ - union { - Packet16uc v; - unsigned char n[16]; - } vt; - vt.v = v; - for (int i=0; i< 16; i++) - s << (int)vt.n[i] << ", "; - return s; -} - -inline std::ostream & operator <<(std::ostream & s, const Packet4f & v) -{ - union { - Packet4f v; - float n[4]; - } vt; - vt.v = v; - s << vt.n[0] << ", " << vt.n[1] << ", " << vt.n[2] << ", " << vt.n[3]; - return s; -} - -inline std::ostream & operator <<(std::ostream & s, const Packet4i & v) -{ - union { - Packet4i v; - int n[4]; - } vt; - vt.v = v; - s << vt.n[0] << ", " << vt.n[1] << ", " << vt.n[2] << ", " << vt.n[3]; - return s; -} - -inline std::ostream & operator <<(std::ostream & s, const Packet4ui & v) -{ - union { - Packet4ui v; - unsigned int n[4]; - } vt; - vt.v = v; - s << vt.n[0] << ", " << vt.n[1] << ", " << vt.n[2] << ", " << vt.n[3]; - return s; -} - -// Need to define them first or we get specialization after instantiation errors -template<> EIGEN_STRONG_INLINE Packet4f pload(const float* from) -{ - EIGEN_DEBUG_ALIGNED_LOAD + + + template<> struct unpacket_traits + { + typedef float type; + enum { size = 4, alignment = Aligned16 }; + typedef Packet4f half; + }; + template<> struct unpacket_traits + { + typedef int type; + enum { size = 4, alignment = Aligned16 }; + typedef Packet4i half; + }; + + inline std::ostream &operator<<(std::ostream &s, const Packet16uc &v) + { + union { + Packet16uc v; + unsigned char n[16]; + } vt; + vt.v = v; + for (int i = 0; i < 16; i++) s << (int)vt.n[i] << ", "; + return s; + } + + inline std::ostream &operator<<(std::ostream &s, const Packet4f &v) + { + union { + Packet4f v; + float n[4]; + } vt; + vt.v = v; + s << vt.n[0] << ", " << vt.n[1] << ", " << vt.n[2] << ", " << vt.n[3]; + return s; + } + + inline std::ostream &operator<<(std::ostream &s, const Packet4i &v) + { + union { + Packet4i v; + int n[4]; + } vt; + vt.v = v; + s << vt.n[0] << ", " << vt.n[1] << ", " << vt.n[2] << ", " << vt.n[3]; + return s; + } + + inline std::ostream &operator<<(std::ostream &s, const Packet4ui &v) + { + union { + Packet4ui v; + unsigned int n[4]; + } vt; + vt.v = v; + s << vt.n[0] << ", " << vt.n[1] << ", " << vt.n[2] << ", " << vt.n[3]; + return s; + } + + // Need to define them first or we get specialization after instantiation errors + template<> EIGEN_STRONG_INLINE Packet4f pload(const float *from) + { + EIGEN_DEBUG_ALIGNED_LOAD #ifdef __VSX__ - return vec_vsx_ld(0, from); + return vec_vsx_ld(0, from); #else - return vec_ld(0, from); + return vec_ld(0, from); #endif -} + } -template<> EIGEN_STRONG_INLINE Packet4i pload(const int* from) -{ - EIGEN_DEBUG_ALIGNED_LOAD + template<> EIGEN_STRONG_INLINE Packet4i pload(const int *from) + { + EIGEN_DEBUG_ALIGNED_LOAD #ifdef __VSX__ - return vec_vsx_ld(0, from); + return vec_vsx_ld(0, from); #else - return vec_ld(0, from); + return vec_ld(0, from); #endif -} + } -template<> EIGEN_STRONG_INLINE void pstore(float* to, const Packet4f& from) -{ - EIGEN_DEBUG_ALIGNED_STORE + template<> EIGEN_STRONG_INLINE void pstore(float *to, const Packet4f &from) + { + EIGEN_DEBUG_ALIGNED_STORE #ifdef __VSX__ - vec_vsx_st(from, 0, to); + vec_vsx_st(from, 0, to); #else - vec_st(from, 0, to); + vec_st(from, 0, to); #endif -} + } -template<> EIGEN_STRONG_INLINE void pstore(int* to, const Packet4i& from) -{ - EIGEN_DEBUG_ALIGNED_STORE + template<> EIGEN_STRONG_INLINE void pstore(int *to, const Packet4i &from) + { + EIGEN_DEBUG_ALIGNED_STORE #ifdef __VSX__ - vec_vsx_st(from, 0, to); + vec_vsx_st(from, 0, to); #else - vec_st(from, 0, to); + vec_st(from, 0, to); #endif -} - -template<> EIGEN_STRONG_INLINE Packet4f pset1(const float& from) { - Packet4f v = {from, from, from, from}; - return v; -} - -template<> EIGEN_STRONG_INLINE Packet4i pset1(const int& from) { - Packet4i v = {from, from, from, from}; - return v; -} -template<> EIGEN_STRONG_INLINE void -pbroadcast4(const float *a, - Packet4f& a0, Packet4f& a1, Packet4f& a2, Packet4f& a3) -{ - a3 = pload(a); - a0 = vec_splat(a3, 0); - a1 = vec_splat(a3, 1); - a2 = vec_splat(a3, 2); - a3 = vec_splat(a3, 3); -} -template<> EIGEN_STRONG_INLINE void -pbroadcast4(const int *a, - Packet4i& a0, Packet4i& a1, Packet4i& a2, Packet4i& a3) -{ - a3 = pload(a); - a0 = vec_splat(a3, 0); - a1 = vec_splat(a3, 1); - a2 = vec_splat(a3, 2); - a3 = vec_splat(a3, 3); -} - -template<> EIGEN_DEVICE_FUNC inline Packet4f pgather(const float* from, Index stride) -{ - float EIGEN_ALIGN16 af[4]; - af[0] = from[0*stride]; - af[1] = from[1*stride]; - af[2] = from[2*stride]; - af[3] = from[3*stride]; - return pload(af); -} -template<> EIGEN_DEVICE_FUNC inline Packet4i pgather(const int* from, Index stride) -{ - int EIGEN_ALIGN16 ai[4]; - ai[0] = from[0*stride]; - ai[1] = from[1*stride]; - ai[2] = from[2*stride]; - ai[3] = from[3*stride]; - return pload(ai); -} -template<> EIGEN_DEVICE_FUNC inline void pscatter(float* to, const Packet4f& from, Index stride) -{ - float EIGEN_ALIGN16 af[4]; - pstore(af, from); - to[0*stride] = af[0]; - to[1*stride] = af[1]; - to[2*stride] = af[2]; - to[3*stride] = af[3]; -} -template<> EIGEN_DEVICE_FUNC inline void pscatter(int* to, const Packet4i& from, Index stride) -{ - int EIGEN_ALIGN16 ai[4]; - pstore((int *)ai, from); - to[0*stride] = ai[0]; - to[1*stride] = ai[1]; - to[2*stride] = ai[2]; - to[3*stride] = ai[3]; -} - -template<> EIGEN_STRONG_INLINE Packet4f plset(const float& a) { return pset1(a) + p4f_COUNTDOWN; } -template<> EIGEN_STRONG_INLINE Packet4i plset(const int& a) { return pset1(a) + p4i_COUNTDOWN; } - -template<> EIGEN_STRONG_INLINE Packet4f padd(const Packet4f& a, const Packet4f& b) { return a + b; } -template<> EIGEN_STRONG_INLINE Packet4i padd(const Packet4i& a, const Packet4i& b) { return a + b; } - -template<> EIGEN_STRONG_INLINE Packet4f psub(const Packet4f& a, const Packet4f& b) { return a - b; } -template<> EIGEN_STRONG_INLINE Packet4i psub(const Packet4i& a, const Packet4i& b) { return a - b; } - -template<> EIGEN_STRONG_INLINE Packet4f pnegate(const Packet4f& a) { return p4f_ZERO - a; } -template<> EIGEN_STRONG_INLINE Packet4i pnegate(const Packet4i& a) { return p4i_ZERO - a; } - -template<> EIGEN_STRONG_INLINE Packet4f pconj(const Packet4f& a) { return a; } -template<> EIGEN_STRONG_INLINE Packet4i pconj(const Packet4i& a) { return a; } - -template<> EIGEN_STRONG_INLINE Packet4f pmul(const Packet4f& a, const Packet4f& b) { return vec_madd(a,b, p4f_MZERO); } -template<> EIGEN_STRONG_INLINE Packet4i pmul(const Packet4i& a, const Packet4i& b) { return a * b; } - -template<> EIGEN_STRONG_INLINE Packet4f pdiv(const Packet4f& a, const Packet4f& b) -{ -#ifndef __VSX__ // VSX actually provides a div instruction - Packet4f t, y_0, y_1; - - // Altivec does not offer a divide instruction, we have to do a reciprocal approximation - y_0 = vec_re(b); - - // Do one Newton-Raphson iteration to get the needed accuracy - t = vec_nmsub(y_0, b, p4f_ONE); - y_1 = vec_madd(y_0, t, y_0); - - return vec_madd(a, y_1, p4f_MZERO); + } + + template<> EIGEN_STRONG_INLINE Packet4f pset1(const float &from) + { + Packet4f v = { from, from, from, from }; + return v; + } + + template<> EIGEN_STRONG_INLINE Packet4i pset1(const int &from) + { + Packet4i v = { from, from, from, from }; + return v; + } + template<> + EIGEN_STRONG_INLINE void pbroadcast4(const float *a, Packet4f &a0, Packet4f &a1, Packet4f &a2, Packet4f &a3) + { + a3 = pload(a); + a0 = vec_splat(a3, 0); + a1 = vec_splat(a3, 1); + a2 = vec_splat(a3, 2); + a3 = vec_splat(a3, 3); + } + template<> + EIGEN_STRONG_INLINE void pbroadcast4(const int *a, Packet4i &a0, Packet4i &a1, Packet4i &a2, Packet4i &a3) + { + a3 = pload(a); + a0 = vec_splat(a3, 0); + a1 = vec_splat(a3, 1); + a2 = vec_splat(a3, 2); + a3 = vec_splat(a3, 3); + } + + template<> EIGEN_DEVICE_FUNC inline Packet4f pgather(const float *from, Index stride) + { + float EIGEN_ALIGN16 af[4]; + af[0] = from[0 * stride]; + af[1] = from[1 * stride]; + af[2] = from[2 * stride]; + af[3] = from[3 * stride]; + return pload(af); + } + template<> EIGEN_DEVICE_FUNC inline Packet4i pgather(const int *from, Index stride) + { + int EIGEN_ALIGN16 ai[4]; + ai[0] = from[0 * stride]; + ai[1] = from[1 * stride]; + ai[2] = from[2 * stride]; + ai[3] = from[3 * stride]; + return pload(ai); + } + template<> EIGEN_DEVICE_FUNC inline void pscatter(float *to, const Packet4f &from, Index stride) + { + float EIGEN_ALIGN16 af[4]; + pstore(af, from); + to[0 * stride] = af[0]; + to[1 * stride] = af[1]; + to[2 * stride] = af[2]; + to[3 * stride] = af[3]; + } + template<> EIGEN_DEVICE_FUNC inline void pscatter(int *to, const Packet4i &from, Index stride) + { + int EIGEN_ALIGN16 ai[4]; + pstore((int *)ai, from); + to[0 * stride] = ai[0]; + to[1 * stride] = ai[1]; + to[2 * stride] = ai[2]; + to[3 * stride] = ai[3]; + } + + template<> EIGEN_STRONG_INLINE Packet4f plset(const float &a) { return pset1(a) + p4f_COUNTDOWN; } + template<> EIGEN_STRONG_INLINE Packet4i plset(const int &a) { return pset1(a) + p4i_COUNTDOWN; } + + template<> EIGEN_STRONG_INLINE Packet4f padd(const Packet4f &a, const Packet4f &b) { return a + b; } + template<> EIGEN_STRONG_INLINE Packet4i padd(const Packet4i &a, const Packet4i &b) { return a + b; } + + template<> EIGEN_STRONG_INLINE Packet4f psub(const Packet4f &a, const Packet4f &b) { return a - b; } + template<> EIGEN_STRONG_INLINE Packet4i psub(const Packet4i &a, const Packet4i &b) { return a - b; } + + template<> EIGEN_STRONG_INLINE Packet4f pnegate(const Packet4f &a) { return p4f_ZERO - a; } + template<> EIGEN_STRONG_INLINE Packet4i pnegate(const Packet4i &a) { return p4i_ZERO - a; } + + template<> EIGEN_STRONG_INLINE Packet4f pconj(const Packet4f &a) { return a; } + template<> EIGEN_STRONG_INLINE Packet4i pconj(const Packet4i &a) { return a; } + + template<> EIGEN_STRONG_INLINE Packet4f pmul(const Packet4f &a, const Packet4f &b) + { + return vec_madd(a, b, p4f_MZERO); + } + template<> EIGEN_STRONG_INLINE Packet4i pmul(const Packet4i &a, const Packet4i &b) { return a * b; } + + template<> EIGEN_STRONG_INLINE Packet4f pdiv(const Packet4f &a, const Packet4f &b) + { +#ifndef __VSX__// VSX actually provides a div instruction + Packet4f t, y_0, y_1; + + // Altivec does not offer a divide instruction, we have to do a reciprocal approximation + y_0 = vec_re(b); + + // Do one Newton-Raphson iteration to get the needed accuracy + t = vec_nmsub(y_0, b, p4f_ONE); + y_1 = vec_madd(y_0, t, y_0); + + return vec_madd(a, y_1, p4f_MZERO); #else - return vec_div(a, b); + return vec_div(a, b); #endif -} - -template<> EIGEN_STRONG_INLINE Packet4i pdiv(const Packet4i& /*a*/, const Packet4i& /*b*/) -{ eigen_assert(false && "packet integer division are not supported by AltiVec"); - return pset1(0); -} - -// for some weird raisons, it has to be overloaded for packet of integers -template<> EIGEN_STRONG_INLINE Packet4f pmadd(const Packet4f& a, const Packet4f& b, const Packet4f& c) { return vec_madd(a,b,c); } -template<> EIGEN_STRONG_INLINE Packet4i pmadd(const Packet4i& a, const Packet4i& b, const Packet4i& c) { return a*b + c; } - -template<> EIGEN_STRONG_INLINE Packet4f pmin(const Packet4f& a, const Packet4f& b) -{ - #ifdef __VSX__ - Packet4f ret; - __asm__ ("xvcmpgesp %x0,%x1,%x2\n\txxsel %x0,%x1,%x2,%x0" : "=&wa" (ret) : "wa" (a), "wa" (b)); - return ret; - #else - return vec_min(a, b); - #endif -} -template<> EIGEN_STRONG_INLINE Packet4i pmin(const Packet4i& a, const Packet4i& b) { return vec_min(a, b); } - -template<> EIGEN_STRONG_INLINE Packet4f pmax(const Packet4f& a, const Packet4f& b) -{ - #ifdef __VSX__ - Packet4f ret; - __asm__ ("xvcmpgtsp %x0,%x2,%x1\n\txxsel %x0,%x1,%x2,%x0" : "=&wa" (ret) : "wa" (a), "wa" (b)); - return ret; - #else - return vec_max(a, b); - #endif -} -template<> EIGEN_STRONG_INLINE Packet4i pmax(const Packet4i& a, const Packet4i& b) { return vec_max(a, b); } - -template<> EIGEN_STRONG_INLINE Packet4f pand(const Packet4f& a, const Packet4f& b) { return vec_and(a, b); } -template<> EIGEN_STRONG_INLINE Packet4i pand(const Packet4i& a, const Packet4i& b) { return vec_and(a, b); } - -template<> EIGEN_STRONG_INLINE Packet4f por(const Packet4f& a, const Packet4f& b) { return vec_or(a, b); } -template<> EIGEN_STRONG_INLINE Packet4i por(const Packet4i& a, const Packet4i& b) { return vec_or(a, b); } - -template<> EIGEN_STRONG_INLINE Packet4f pxor(const Packet4f& a, const Packet4f& b) { return vec_xor(a, b); } -template<> EIGEN_STRONG_INLINE Packet4i pxor(const Packet4i& a, const Packet4i& b) { return vec_xor(a, b); } - -template<> EIGEN_STRONG_INLINE Packet4f pandnot(const Packet4f& a, const Packet4f& b) { return vec_and(a, vec_nor(b, b)); } -template<> EIGEN_STRONG_INLINE Packet4i pandnot(const Packet4i& a, const Packet4i& b) { return vec_and(a, vec_nor(b, b)); } - -template<> EIGEN_STRONG_INLINE Packet4f pround(const Packet4f& a) { return vec_round(a); } -template<> EIGEN_STRONG_INLINE Packet4f pceil(const Packet4f& a) { return vec_ceil(a); } -template<> EIGEN_STRONG_INLINE Packet4f pfloor(const Packet4f& a) { return vec_floor(a); } + } -#ifdef _BIG_ENDIAN -template<> EIGEN_STRONG_INLINE Packet4f ploadu(const float* from) -{ - EIGEN_DEBUG_ALIGNED_LOAD - Packet16uc MSQ, LSQ; - Packet16uc mask; - MSQ = vec_ld(0, (unsigned char *)from); // most significant quadword - LSQ = vec_ld(15, (unsigned char *)from); // least significant quadword - mask = vec_lvsl(0, from); // create the permute mask - return (Packet4f) vec_perm(MSQ, LSQ, mask); // align the data - -} -template<> EIGEN_STRONG_INLINE Packet4i ploadu(const int* from) -{ - EIGEN_DEBUG_ALIGNED_LOAD - // Taken from http://developer.apple.com/hardwaredrivers/ve/alignment.html - Packet16uc MSQ, LSQ; - Packet16uc mask; - MSQ = vec_ld(0, (unsigned char *)from); // most significant quadword - LSQ = vec_ld(15, (unsigned char *)from); // least significant quadword - mask = vec_lvsl(0, from); // create the permute mask - return (Packet4i) vec_perm(MSQ, LSQ, mask); // align the data -} + template<> EIGEN_STRONG_INLINE Packet4i pdiv(const Packet4i & /*a*/, const Packet4i & /*b*/) + { + eigen_assert(false && "packet integer division are not supported by AltiVec"); + return pset1(0); + } + + // for some weird raisons, it has to be overloaded for packet of integers + template<> EIGEN_STRONG_INLINE Packet4f pmadd(const Packet4f &a, const Packet4f &b, const Packet4f &c) + { + return vec_madd(a, b, c); + } + template<> EIGEN_STRONG_INLINE Packet4i pmadd(const Packet4i &a, const Packet4i &b, const Packet4i &c) + { + return a * b + c; + } + + template<> EIGEN_STRONG_INLINE Packet4f pmin(const Packet4f &a, const Packet4f &b) + { +#ifdef __VSX__ + Packet4f ret; + __asm__("xvcmpgesp %x0,%x1,%x2\n\txxsel %x0,%x1,%x2,%x0" : "=&wa"(ret) : "wa"(a), "wa"(b)); + return ret; #else -// We also need ot redefine little endian loading of Packet4i/Packet4f using VSX -template<> EIGEN_STRONG_INLINE Packet4i ploadu(const int* from) -{ - EIGEN_DEBUG_UNALIGNED_LOAD - return (Packet4i) vec_vsx_ld((long)from & 15, (const int*) _EIGEN_ALIGNED_PTR(from)); -} -template<> EIGEN_STRONG_INLINE Packet4f ploadu(const float* from) -{ - EIGEN_DEBUG_UNALIGNED_LOAD - return (Packet4f) vec_vsx_ld((long)from & 15, (const float*) _EIGEN_ALIGNED_PTR(from)); -} + return vec_min(a, b); #endif + } + template<> EIGEN_STRONG_INLINE Packet4i pmin(const Packet4i &a, const Packet4i &b) { return vec_min(a, b); } + + template<> EIGEN_STRONG_INLINE Packet4f pmax(const Packet4f &a, const Packet4f &b) + { +#ifdef __VSX__ + Packet4f ret; + __asm__("xvcmpgtsp %x0,%x2,%x1\n\txxsel %x0,%x1,%x2,%x0" : "=&wa"(ret) : "wa"(a), "wa"(b)); + return ret; +#else + return vec_max(a, b); +#endif + } + template<> EIGEN_STRONG_INLINE Packet4i pmax(const Packet4i &a, const Packet4i &b) { return vec_max(a, b); } + + template<> EIGEN_STRONG_INLINE Packet4f pand(const Packet4f &a, const Packet4f &b) { return vec_and(a, b); } + template<> EIGEN_STRONG_INLINE Packet4i pand(const Packet4i &a, const Packet4i &b) { return vec_and(a, b); } + + template<> EIGEN_STRONG_INLINE Packet4f por(const Packet4f &a, const Packet4f &b) { return vec_or(a, b); } + template<> EIGEN_STRONG_INLINE Packet4i por(const Packet4i &a, const Packet4i &b) { return vec_or(a, b); } + + template<> EIGEN_STRONG_INLINE Packet4f pxor(const Packet4f &a, const Packet4f &b) { return vec_xor(a, b); } + template<> EIGEN_STRONG_INLINE Packet4i pxor(const Packet4i &a, const Packet4i &b) { return vec_xor(a, b); } -template<> EIGEN_STRONG_INLINE Packet4f ploaddup(const float* from) -{ - Packet4f p; - if((std::ptrdiff_t(from) % 16) == 0) p = pload(from); - else p = ploadu(from); - return vec_perm(p, p, p16uc_DUPLICATE32_HI); -} -template<> EIGEN_STRONG_INLINE Packet4i ploaddup(const int* from) -{ - Packet4i p; - if((std::ptrdiff_t(from) % 16) == 0) p = pload(from); - else p = ploadu(from); - return vec_perm(p, p, p16uc_DUPLICATE32_HI); -} + template<> EIGEN_STRONG_INLINE Packet4f pandnot(const Packet4f &a, const Packet4f &b) + { + return vec_and(a, vec_nor(b, b)); + } + template<> EIGEN_STRONG_INLINE Packet4i pandnot(const Packet4i &a, const Packet4i &b) + { + return vec_and(a, vec_nor(b, b)); + } + + template<> EIGEN_STRONG_INLINE Packet4f pround(const Packet4f &a) { return vec_round(a); } + template<> EIGEN_STRONG_INLINE Packet4f pceil(const Packet4f &a) { return vec_ceil(a); } + template<> EIGEN_STRONG_INLINE Packet4f pfloor(const Packet4f &a) { return vec_floor(a); } #ifdef _BIG_ENDIAN -template<> EIGEN_STRONG_INLINE void pstoreu(float* to, const Packet4f& from) -{ - EIGEN_DEBUG_UNALIGNED_STORE - // Taken from http://developer.apple.com/hardwaredrivers/ve/alignment.html - // Warning: not thread safe! - Packet16uc MSQ, LSQ, edges; - Packet16uc edgeAlign, align; - - MSQ = vec_ld(0, (unsigned char *)to); // most significant quadword - LSQ = vec_ld(15, (unsigned char *)to); // least significant quadword - edgeAlign = vec_lvsl(0, to); // permute map to extract edges - edges=vec_perm(LSQ,MSQ,edgeAlign); // extract the edges - align = vec_lvsr( 0, to ); // permute map to misalign data - MSQ = vec_perm(edges,(Packet16uc)from,align); // misalign the data (MSQ) - LSQ = vec_perm((Packet16uc)from,edges,align); // misalign the data (LSQ) - vec_st( LSQ, 15, (unsigned char *)to ); // Store the LSQ part first - vec_st( MSQ, 0, (unsigned char *)to ); // Store the MSQ part -} -template<> EIGEN_STRONG_INLINE void pstoreu(int* to, const Packet4i& from) -{ - EIGEN_DEBUG_UNALIGNED_STORE - // Taken from http://developer.apple.com/hardwaredrivers/ve/alignment.html - // Warning: not thread safe! - Packet16uc MSQ, LSQ, edges; - Packet16uc edgeAlign, align; - - MSQ = vec_ld(0, (unsigned char *)to); // most significant quadword - LSQ = vec_ld(15, (unsigned char *)to); // least significant quadword - edgeAlign = vec_lvsl(0, to); // permute map to extract edges - edges=vec_perm(LSQ, MSQ, edgeAlign); // extract the edges - align = vec_lvsr( 0, to ); // permute map to misalign data - MSQ = vec_perm(edges, (Packet16uc) from, align); // misalign the data (MSQ) - LSQ = vec_perm((Packet16uc) from, edges, align); // misalign the data (LSQ) - vec_st( LSQ, 15, (unsigned char *)to ); // Store the LSQ part first - vec_st( MSQ, 0, (unsigned char *)to ); // Store the MSQ part -} + template<> EIGEN_STRONG_INLINE Packet4f ploadu(const float *from) + { + EIGEN_DEBUG_ALIGNED_LOAD + Packet16uc MSQ, LSQ; + Packet16uc mask; + MSQ = vec_ld(0, (unsigned char *)from);// most significant quadword + LSQ = vec_ld(15, (unsigned char *)from);// least significant quadword + mask = vec_lvsl(0, from);// create the permute mask + return (Packet4f)vec_perm(MSQ, LSQ, mask);// align the data + } + template<> EIGEN_STRONG_INLINE Packet4i ploadu(const int *from) + { + EIGEN_DEBUG_ALIGNED_LOAD + // Taken from http://developer.apple.com/hardwaredrivers/ve/alignment.html + Packet16uc MSQ, LSQ; + Packet16uc mask; + MSQ = vec_ld(0, (unsigned char *)from);// most significant quadword + LSQ = vec_ld(15, (unsigned char *)from);// least significant quadword + mask = vec_lvsl(0, from);// create the permute mask + return (Packet4i)vec_perm(MSQ, LSQ, mask);// align the data + } #else -// We also need ot redefine little endian loading of Packet4i/Packet4f using VSX -template<> EIGEN_STRONG_INLINE void pstoreu(int* to, const Packet4i& from) -{ - EIGEN_DEBUG_ALIGNED_STORE - vec_vsx_st(from, (long)to & 15, (int*) _EIGEN_ALIGNED_PTR(to)); -} -template<> EIGEN_STRONG_INLINE void pstoreu(float* to, const Packet4f& from) -{ - EIGEN_DEBUG_ALIGNED_STORE - vec_vsx_st(from, (long)to & 15, (float*) _EIGEN_ALIGNED_PTR(to)); -} + // We also need ot redefine little endian loading of Packet4i/Packet4f using VSX + template<> EIGEN_STRONG_INLINE Packet4i ploadu(const int *from) + { + EIGEN_DEBUG_UNALIGNED_LOAD + return (Packet4i)vec_vsx_ld((long)from & 15, (const int *)_EIGEN_ALIGNED_PTR(from)); + } + template<> EIGEN_STRONG_INLINE Packet4f ploadu(const float *from) + { + EIGEN_DEBUG_UNALIGNED_LOAD + return (Packet4f)vec_vsx_ld((long)from & 15, (const float *)_EIGEN_ALIGNED_PTR(from)); + } #endif -template<> EIGEN_STRONG_INLINE void prefetch(const float* addr) { EIGEN_PPC_PREFETCH(addr); } -template<> EIGEN_STRONG_INLINE void prefetch(const int* addr) { EIGEN_PPC_PREFETCH(addr); } - -template<> EIGEN_STRONG_INLINE float pfirst(const Packet4f& a) { float EIGEN_ALIGN16 x; vec_ste(a, 0, &x); return x; } -template<> EIGEN_STRONG_INLINE int pfirst(const Packet4i& a) { int EIGEN_ALIGN16 x; vec_ste(a, 0, &x); return x; } - -template<> EIGEN_STRONG_INLINE Packet4f preverse(const Packet4f& a) -{ - return reinterpret_cast(vec_perm(reinterpret_cast(a), reinterpret_cast(a), p16uc_REVERSE32)); -} -template<> EIGEN_STRONG_INLINE Packet4i preverse(const Packet4i& a) -{ - return reinterpret_cast(vec_perm(reinterpret_cast(a), reinterpret_cast(a), p16uc_REVERSE32)); } - -template<> EIGEN_STRONG_INLINE Packet4f pabs(const Packet4f& a) { return vec_abs(a); } -template<> EIGEN_STRONG_INLINE Packet4i pabs(const Packet4i& a) { return vec_abs(a); } - -template<> EIGEN_STRONG_INLINE float predux(const Packet4f& a) -{ - Packet4f b, sum; - b = vec_sld(a, a, 8); - sum = a + b; - b = vec_sld(sum, sum, 4); - sum += b; - return pfirst(sum); -} - -template<> EIGEN_STRONG_INLINE Packet4f preduxp(const Packet4f* vecs) -{ - Packet4f v[4], sum[4]; - - // It's easier and faster to transpose then add as columns - // Check: http://www.freevec.org/function/matrix_4x4_transpose_floats for explanation - // Do the transpose, first set of moves - v[0] = vec_mergeh(vecs[0], vecs[2]); - v[1] = vec_mergel(vecs[0], vecs[2]); - v[2] = vec_mergeh(vecs[1], vecs[3]); - v[3] = vec_mergel(vecs[1], vecs[3]); - // Get the resulting vectors - sum[0] = vec_mergeh(v[0], v[2]); - sum[1] = vec_mergel(v[0], v[2]); - sum[2] = vec_mergeh(v[1], v[3]); - sum[3] = vec_mergel(v[1], v[3]); - - // Now do the summation: - // Lines 0+1 - sum[0] = sum[0] + sum[1]; - // Lines 2+3 - sum[1] = sum[2] + sum[3]; - // Add the results - sum[0] = sum[0] + sum[1]; - - return sum[0]; -} - -template<> EIGEN_STRONG_INLINE int predux(const Packet4i& a) -{ - Packet4i sum; - sum = vec_sums(a, p4i_ZERO); + template<> EIGEN_STRONG_INLINE Packet4f ploaddup(const float *from) + { + Packet4f p; + if ((std::ptrdiff_t(from) % 16) == 0) + p = pload(from); + else + p = ploadu(from); + return vec_perm(p, p, p16uc_DUPLICATE32_HI); + } + template<> EIGEN_STRONG_INLINE Packet4i ploaddup(const int *from) + { + Packet4i p; + if ((std::ptrdiff_t(from) % 16) == 0) + p = pload(from); + else + p = ploadu(from); + return vec_perm(p, p, p16uc_DUPLICATE32_HI); + } + #ifdef _BIG_ENDIAN - sum = vec_sld(sum, p4i_ZERO, 12); + template<> EIGEN_STRONG_INLINE void pstoreu(float *to, const Packet4f &from) + { + EIGEN_DEBUG_UNALIGNED_STORE + // Taken from http://developer.apple.com/hardwaredrivers/ve/alignment.html + // Warning: not thread safe! + Packet16uc MSQ, LSQ, edges; + Packet16uc edgeAlign, align; + + MSQ = vec_ld(0, (unsigned char *)to);// most significant quadword + LSQ = vec_ld(15, (unsigned char *)to);// least significant quadword + edgeAlign = vec_lvsl(0, to);// permute map to extract edges + edges = vec_perm(LSQ, MSQ, edgeAlign);// extract the edges + align = vec_lvsr(0, to);// permute map to misalign data + MSQ = vec_perm(edges, (Packet16uc)from, align);// misalign the data (MSQ) + LSQ = vec_perm((Packet16uc)from, edges, align);// misalign the data (LSQ) + vec_st(LSQ, 15, (unsigned char *)to);// Store the LSQ part first + vec_st(MSQ, 0, (unsigned char *)to);// Store the MSQ part + } + template<> EIGEN_STRONG_INLINE void pstoreu(int *to, const Packet4i &from) + { + EIGEN_DEBUG_UNALIGNED_STORE + // Taken from http://developer.apple.com/hardwaredrivers/ve/alignment.html + // Warning: not thread safe! + Packet16uc MSQ, LSQ, edges; + Packet16uc edgeAlign, align; + + MSQ = vec_ld(0, (unsigned char *)to);// most significant quadword + LSQ = vec_ld(15, (unsigned char *)to);// least significant quadword + edgeAlign = vec_lvsl(0, to);// permute map to extract edges + edges = vec_perm(LSQ, MSQ, edgeAlign);// extract the edges + align = vec_lvsr(0, to);// permute map to misalign data + MSQ = vec_perm(edges, (Packet16uc)from, align);// misalign the data (MSQ) + LSQ = vec_perm((Packet16uc)from, edges, align);// misalign the data (LSQ) + vec_st(LSQ, 15, (unsigned char *)to);// Store the LSQ part first + vec_st(MSQ, 0, (unsigned char *)to);// Store the MSQ part + } #else - sum = vec_sld(p4i_ZERO, sum, 4); + // We also need ot redefine little endian loading of Packet4i/Packet4f using VSX + template<> EIGEN_STRONG_INLINE void pstoreu(int *to, const Packet4i &from) + { + EIGEN_DEBUG_ALIGNED_STORE + vec_vsx_st(from, (long)to & 15, (int *)_EIGEN_ALIGNED_PTR(to)); + } + template<> EIGEN_STRONG_INLINE void pstoreu(float *to, const Packet4f &from) + { + EIGEN_DEBUG_ALIGNED_STORE + vec_vsx_st(from, (long)to & 15, (float *)_EIGEN_ALIGNED_PTR(to)); + } #endif - return pfirst(sum); -} - -template<> EIGEN_STRONG_INLINE Packet4i preduxp(const Packet4i* vecs) -{ - Packet4i v[4], sum[4]; - - // It's easier and faster to transpose then add as columns - // Check: http://www.freevec.org/function/matrix_4x4_transpose_floats for explanation - // Do the transpose, first set of moves - v[0] = vec_mergeh(vecs[0], vecs[2]); - v[1] = vec_mergel(vecs[0], vecs[2]); - v[2] = vec_mergeh(vecs[1], vecs[3]); - v[3] = vec_mergel(vecs[1], vecs[3]); - // Get the resulting vectors - sum[0] = vec_mergeh(v[0], v[2]); - sum[1] = vec_mergel(v[0], v[2]); - sum[2] = vec_mergeh(v[1], v[3]); - sum[3] = vec_mergel(v[1], v[3]); - - // Now do the summation: - // Lines 0+1 - sum[0] = sum[0] + sum[1]; - // Lines 2+3 - sum[1] = sum[2] + sum[3]; - // Add the results - sum[0] = sum[0] + sum[1]; - - return sum[0]; -} - -// Other reduction functions: -// mul -template<> EIGEN_STRONG_INLINE float predux_mul(const Packet4f& a) -{ - Packet4f prod; - prod = pmul(a, vec_sld(a, a, 8)); - return pfirst(pmul(prod, vec_sld(prod, prod, 4))); -} - -template<> EIGEN_STRONG_INLINE int predux_mul(const Packet4i& a) -{ - EIGEN_ALIGN16 int aux[4]; - pstore(aux, a); - return aux[0] * aux[1] * aux[2] * aux[3]; -} - -// min -template<> EIGEN_STRONG_INLINE float predux_min(const Packet4f& a) -{ - Packet4f b, res; - b = vec_min(a, vec_sld(a, a, 8)); - res = vec_min(b, vec_sld(b, b, 4)); - return pfirst(res); -} - -template<> EIGEN_STRONG_INLINE int predux_min(const Packet4i& a) -{ - Packet4i b, res; - b = vec_min(a, vec_sld(a, a, 8)); - res = vec_min(b, vec_sld(b, b, 4)); - return pfirst(res); -} - -// max -template<> EIGEN_STRONG_INLINE float predux_max(const Packet4f& a) -{ - Packet4f b, res; - b = vec_max(a, vec_sld(a, a, 8)); - res = vec_max(b, vec_sld(b, b, 4)); - return pfirst(res); -} - -template<> EIGEN_STRONG_INLINE int predux_max(const Packet4i& a) -{ - Packet4i b, res; - b = vec_max(a, vec_sld(a, a, 8)); - res = vec_max(b, vec_sld(b, b, 4)); - return pfirst(res); -} - -template -struct palign_impl -{ - static EIGEN_STRONG_INLINE void run(Packet4f& first, const Packet4f& second) + + template<> EIGEN_STRONG_INLINE void prefetch(const float *addr) { EIGEN_PPC_PREFETCH(addr); } + template<> EIGEN_STRONG_INLINE void prefetch(const int *addr) { EIGEN_PPC_PREFETCH(addr); } + + template<> EIGEN_STRONG_INLINE float pfirst(const Packet4f &a) + { + float EIGEN_ALIGN16 x; + vec_ste(a, 0, &x); + return x; + } + template<> EIGEN_STRONG_INLINE int pfirst(const Packet4i &a) + { + int EIGEN_ALIGN16 x; + vec_ste(a, 0, &x); + return x; + } + + template<> EIGEN_STRONG_INLINE Packet4f preverse(const Packet4f &a) + { + return reinterpret_cast( + vec_perm(reinterpret_cast(a), reinterpret_cast(a), p16uc_REVERSE32)); + } + template<> EIGEN_STRONG_INLINE Packet4i preverse(const Packet4i &a) + { + return reinterpret_cast( + vec_perm(reinterpret_cast(a), reinterpret_cast(a), p16uc_REVERSE32)); + } + + template<> EIGEN_STRONG_INLINE Packet4f pabs(const Packet4f &a) { return vec_abs(a); } + template<> EIGEN_STRONG_INLINE Packet4i pabs(const Packet4i &a) { return vec_abs(a); } + + template<> EIGEN_STRONG_INLINE float predux(const Packet4f &a) { + Packet4f b, sum; + b = vec_sld(a, a, 8); + sum = a + b; + b = vec_sld(sum, sum, 4); + sum += b; + return pfirst(sum); + } + + template<> EIGEN_STRONG_INLINE Packet4f preduxp(const Packet4f *vecs) + { + Packet4f v[4], sum[4]; + + // It's easier and faster to transpose then add as columns + // Check: http://www.freevec.org/function/matrix_4x4_transpose_floats for explanation + // Do the transpose, first set of moves + v[0] = vec_mergeh(vecs[0], vecs[2]); + v[1] = vec_mergel(vecs[0], vecs[2]); + v[2] = vec_mergeh(vecs[1], vecs[3]); + v[3] = vec_mergel(vecs[1], vecs[3]); + // Get the resulting vectors + sum[0] = vec_mergeh(v[0], v[2]); + sum[1] = vec_mergel(v[0], v[2]); + sum[2] = vec_mergeh(v[1], v[3]); + sum[3] = vec_mergel(v[1], v[3]); + + // Now do the summation: + // Lines 0+1 + sum[0] = sum[0] + sum[1]; + // Lines 2+3 + sum[1] = sum[2] + sum[3]; + // Add the results + sum[0] = sum[0] + sum[1]; + + return sum[0]; + } + + template<> EIGEN_STRONG_INLINE int predux(const Packet4i &a) + { + Packet4i sum; + sum = vec_sums(a, p4i_ZERO); #ifdef _BIG_ENDIAN - switch (Offset % 4) { - case 1: - first = vec_sld(first, second, 4); break; - case 2: - first = vec_sld(first, second, 8); break; - case 3: - first = vec_sld(first, second, 12); break; - } + sum = vec_sld(sum, p4i_ZERO, 12); #else - switch (Offset % 4) { - case 1: - first = vec_sld(second, first, 12); break; - case 2: - first = vec_sld(second, first, 8); break; - case 3: - first = vec_sld(second, first, 4); break; - } + sum = vec_sld(p4i_ZERO, sum, 4); #endif + return pfirst(sum); } -}; -template -struct palign_impl -{ - static EIGEN_STRONG_INLINE void run(Packet4i& first, const Packet4i& second) + template<> EIGEN_STRONG_INLINE Packet4i preduxp(const Packet4i *vecs) { + Packet4i v[4], sum[4]; + + // It's easier and faster to transpose then add as columns + // Check: http://www.freevec.org/function/matrix_4x4_transpose_floats for explanation + // Do the transpose, first set of moves + v[0] = vec_mergeh(vecs[0], vecs[2]); + v[1] = vec_mergel(vecs[0], vecs[2]); + v[2] = vec_mergeh(vecs[1], vecs[3]); + v[3] = vec_mergel(vecs[1], vecs[3]); + // Get the resulting vectors + sum[0] = vec_mergeh(v[0], v[2]); + sum[1] = vec_mergel(v[0], v[2]); + sum[2] = vec_mergeh(v[1], v[3]); + sum[3] = vec_mergel(v[1], v[3]); + + // Now do the summation: + // Lines 0+1 + sum[0] = sum[0] + sum[1]; + // Lines 2+3 + sum[1] = sum[2] + sum[3]; + // Add the results + sum[0] = sum[0] + sum[1]; + + return sum[0]; + } + + // Other reduction functions: + // mul + template<> EIGEN_STRONG_INLINE float predux_mul(const Packet4f &a) + { + Packet4f prod; + prod = pmul(a, vec_sld(a, a, 8)); + return pfirst(pmul(prod, vec_sld(prod, prod, 4))); + } + + template<> EIGEN_STRONG_INLINE int predux_mul(const Packet4i &a) + { + EIGEN_ALIGN16 int aux[4]; + pstore(aux, a); + return aux[0] * aux[1] * aux[2] * aux[3]; + } + + // min + template<> EIGEN_STRONG_INLINE float predux_min(const Packet4f &a) + { + Packet4f b, res; + b = vec_min(a, vec_sld(a, a, 8)); + res = vec_min(b, vec_sld(b, b, 4)); + return pfirst(res); + } + + template<> EIGEN_STRONG_INLINE int predux_min(const Packet4i &a) + { + Packet4i b, res; + b = vec_min(a, vec_sld(a, a, 8)); + res = vec_min(b, vec_sld(b, b, 4)); + return pfirst(res); + } + + // max + template<> EIGEN_STRONG_INLINE float predux_max(const Packet4f &a) + { + Packet4f b, res; + b = vec_max(a, vec_sld(a, a, 8)); + res = vec_max(b, vec_sld(b, b, 4)); + return pfirst(res); + } + + template<> EIGEN_STRONG_INLINE int predux_max(const Packet4i &a) + { + Packet4i b, res; + b = vec_max(a, vec_sld(a, a, 8)); + res = vec_max(b, vec_sld(b, b, 4)); + return pfirst(res); + } + + template struct palign_impl + { + static EIGEN_STRONG_INLINE void run(Packet4f &first, const Packet4f &second) + { #ifdef _BIG_ENDIAN - switch (Offset % 4) { - case 1: - first = vec_sld(first, second, 4); break; - case 2: - first = vec_sld(first, second, 8); break; - case 3: - first = vec_sld(first, second, 12); break; - } + switch (Offset % 4) { + case 1: + first = vec_sld(first, second, 4); + break; + case 2: + first = vec_sld(first, second, 8); + break; + case 3: + first = vec_sld(first, second, 12); + break; + } #else - switch (Offset % 4) { - case 1: - first = vec_sld(second, first, 12); break; - case 2: - first = vec_sld(second, first, 8); break; - case 3: - first = vec_sld(second, first, 4); break; + switch (Offset % 4) { + case 1: + first = vec_sld(second, first, 12); + break; + case 2: + first = vec_sld(second, first, 8); + break; + case 3: + first = vec_sld(second, first, 4); + break; + } +#endif } + }; + + template struct palign_impl + { + static EIGEN_STRONG_INLINE void run(Packet4i &first, const Packet4i &second) + { +#ifdef _BIG_ENDIAN + switch (Offset % 4) { + case 1: + first = vec_sld(first, second, 4); + break; + case 2: + first = vec_sld(first, second, 8); + break; + case 3: + first = vec_sld(first, second, 12); + break; + } +#else + switch (Offset % 4) { + case 1: + first = vec_sld(second, first, 12); + break; + case 2: + first = vec_sld(second, first, 8); + break; + case 3: + first = vec_sld(second, first, 4); + break; + } #endif + } + }; + + EIGEN_DEVICE_FUNC inline void ptranspose(PacketBlock &kernel) + { + Packet4f t0, t1, t2, t3; + t0 = vec_mergeh(kernel.packet[0], kernel.packet[2]); + t1 = vec_mergel(kernel.packet[0], kernel.packet[2]); + t2 = vec_mergeh(kernel.packet[1], kernel.packet[3]); + t3 = vec_mergel(kernel.packet[1], kernel.packet[3]); + kernel.packet[0] = vec_mergeh(t0, t2); + kernel.packet[1] = vec_mergel(t0, t2); + kernel.packet[2] = vec_mergeh(t1, t3); + kernel.packet[3] = vec_mergel(t1, t3); + } + + EIGEN_DEVICE_FUNC inline void ptranspose(PacketBlock &kernel) + { + Packet4i t0, t1, t2, t3; + t0 = vec_mergeh(kernel.packet[0], kernel.packet[2]); + t1 = vec_mergel(kernel.packet[0], kernel.packet[2]); + t2 = vec_mergeh(kernel.packet[1], kernel.packet[3]); + t3 = vec_mergel(kernel.packet[1], kernel.packet[3]); + kernel.packet[0] = vec_mergeh(t0, t2); + kernel.packet[1] = vec_mergel(t0, t2); + kernel.packet[2] = vec_mergeh(t1, t3); + kernel.packet[3] = vec_mergel(t1, t3); + } + + template<> + EIGEN_STRONG_INLINE Packet4i pblend(const Selector<4> &ifPacket, + const Packet4i &thenPacket, + const Packet4i &elsePacket) + { + Packet4ui select = { ifPacket.select[0], ifPacket.select[1], ifPacket.select[2], ifPacket.select[3] }; + Packet4ui mask = + reinterpret_cast(vec_cmpeq(reinterpret_cast(select), reinterpret_cast(p4i_ONE))); + return vec_sel(elsePacket, thenPacket, mask); + } + + template<> + EIGEN_STRONG_INLINE Packet4f pblend(const Selector<4> &ifPacket, + const Packet4f &thenPacket, + const Packet4f &elsePacket) + { + Packet4ui select = { ifPacket.select[0], ifPacket.select[1], ifPacket.select[2], ifPacket.select[3] }; + Packet4ui mask = + reinterpret_cast(vec_cmpeq(reinterpret_cast(select), reinterpret_cast(p4i_ONE))); + return vec_sel(elsePacket, thenPacket, mask); } -}; - -EIGEN_DEVICE_FUNC inline void -ptranspose(PacketBlock& kernel) { - Packet4f t0, t1, t2, t3; - t0 = vec_mergeh(kernel.packet[0], kernel.packet[2]); - t1 = vec_mergel(kernel.packet[0], kernel.packet[2]); - t2 = vec_mergeh(kernel.packet[1], kernel.packet[3]); - t3 = vec_mergel(kernel.packet[1], kernel.packet[3]); - kernel.packet[0] = vec_mergeh(t0, t2); - kernel.packet[1] = vec_mergel(t0, t2); - kernel.packet[2] = vec_mergeh(t1, t3); - kernel.packet[3] = vec_mergel(t1, t3); -} - -EIGEN_DEVICE_FUNC inline void -ptranspose(PacketBlock& kernel) { - Packet4i t0, t1, t2, t3; - t0 = vec_mergeh(kernel.packet[0], kernel.packet[2]); - t1 = vec_mergel(kernel.packet[0], kernel.packet[2]); - t2 = vec_mergeh(kernel.packet[1], kernel.packet[3]); - t3 = vec_mergel(kernel.packet[1], kernel.packet[3]); - kernel.packet[0] = vec_mergeh(t0, t2); - kernel.packet[1] = vec_mergel(t0, t2); - kernel.packet[2] = vec_mergeh(t1, t3); - kernel.packet[3] = vec_mergel(t1, t3); -} - -template<> EIGEN_STRONG_INLINE Packet4i pblend(const Selector<4>& ifPacket, const Packet4i& thenPacket, const Packet4i& elsePacket) { - Packet4ui select = { ifPacket.select[0], ifPacket.select[1], ifPacket.select[2], ifPacket.select[3] }; - Packet4ui mask = reinterpret_cast(vec_cmpeq(reinterpret_cast(select), reinterpret_cast(p4i_ONE))); - return vec_sel(elsePacket, thenPacket, mask); -} - -template<> EIGEN_STRONG_INLINE Packet4f pblend(const Selector<4>& ifPacket, const Packet4f& thenPacket, const Packet4f& elsePacket) { - Packet4ui select = { ifPacket.select[0], ifPacket.select[1], ifPacket.select[2], ifPacket.select[3] }; - Packet4ui mask = reinterpret_cast(vec_cmpeq(reinterpret_cast(select), reinterpret_cast(p4i_ONE))); - return vec_sel(elsePacket, thenPacket, mask); -} //---------- double ---------- #ifdef __VSX__ -typedef __vector double Packet2d; -typedef __vector unsigned long long Packet2ul; -typedef __vector long long Packet2l; + typedef __vector double Packet2d; + typedef __vector unsigned long long Packet2ul; + typedef __vector long long Packet2l; #if EIGEN_COMP_CLANG -typedef Packet2ul Packet2bl; + typedef Packet2ul Packet2bl; #else -typedef __vector __bool long Packet2bl; + typedef __vector __bool long Packet2bl; #endif -static Packet2l p2l_ONE = { 1, 1 }; -static Packet2l p2l_ZERO = reinterpret_cast(p4i_ZERO); -static Packet2d p2d_ONE = { 1.0, 1.0 }; -static Packet2d p2d_ZERO = reinterpret_cast(p4f_ZERO); -static Packet2d p2d_MZERO = { -0.0, -0.0 }; + static Packet2l p2l_ONE = { 1, 1 }; + static Packet2l p2l_ZERO = reinterpret_cast(p4i_ZERO); + static Packet2d p2d_ONE = { 1.0, 1.0 }; + static Packet2d p2d_ZERO = reinterpret_cast(p4f_ZERO); + static Packet2d p2d_MZERO = { -0.0, -0.0 }; #ifdef _BIG_ENDIAN -static Packet2d p2d_COUNTDOWN = reinterpret_cast(vec_sld(reinterpret_cast(p2d_ZERO), reinterpret_cast(p2d_ONE), 8)); + static Packet2d p2d_COUNTDOWN = + reinterpret_cast(vec_sld(reinterpret_cast(p2d_ZERO), reinterpret_cast(p2d_ONE), 8)); #else -static Packet2d p2d_COUNTDOWN = reinterpret_cast(vec_sld(reinterpret_cast(p2d_ONE), reinterpret_cast(p2d_ZERO), 8)); + static Packet2d p2d_COUNTDOWN = + reinterpret_cast(vec_sld(reinterpret_cast(p2d_ONE), reinterpret_cast(p2d_ZERO), 8)); #endif -template Packet2d vec_splat_dbl(Packet2d& a); - -template<> EIGEN_STRONG_INLINE Packet2d vec_splat_dbl<0>(Packet2d& a) -{ - return reinterpret_cast(vec_perm(a, a, p16uc_PSET64_HI)); -} - -template<> EIGEN_STRONG_INLINE Packet2d vec_splat_dbl<1>(Packet2d& a) -{ - return reinterpret_cast(vec_perm(a, a, p16uc_PSET64_LO)); -} - -template<> struct packet_traits : default_packet_traits -{ - typedef Packet2d type; - typedef Packet2d half; - enum { - Vectorizable = 1, - AlignedOnScalar = 1, - size=2, - HasHalfPacket = 1, - - HasAdd = 1, - HasSub = 1, - HasMul = 1, - HasDiv = 1, - HasMin = 1, - HasMax = 1, - HasAbs = 1, - HasSin = 0, - HasCos = 0, - HasLog = 0, - HasExp = 1, - HasSqrt = 1, - HasRsqrt = 1, - HasRound = 1, - HasFloor = 1, - HasCeil = 1, - HasNegate = 1, - HasBlend = 1 + template Packet2d vec_splat_dbl(Packet2d &a); + + template<> EIGEN_STRONG_INLINE Packet2d vec_splat_dbl<0>(Packet2d &a) + { + return reinterpret_cast(vec_perm(a, a, p16uc_PSET64_HI)); + } + + template<> EIGEN_STRONG_INLINE Packet2d vec_splat_dbl<1>(Packet2d &a) + { + return reinterpret_cast(vec_perm(a, a, p16uc_PSET64_LO)); + } + + template<> struct packet_traits : default_packet_traits + { + typedef Packet2d type; + typedef Packet2d half; + enum { + Vectorizable = 1, + AlignedOnScalar = 1, + size = 2, + HasHalfPacket = 1, + + HasAdd = 1, + HasSub = 1, + HasMul = 1, + HasDiv = 1, + HasMin = 1, + HasMax = 1, + HasAbs = 1, + HasSin = 0, + HasCos = 0, + HasLog = 0, + HasExp = 1, + HasSqrt = 1, + HasRsqrt = 1, + HasRound = 1, + HasFloor = 1, + HasCeil = 1, + HasNegate = 1, + HasBlend = 1 + }; + }; + + template<> struct unpacket_traits + { + typedef double type; + enum { size = 2, alignment = Aligned16 }; + typedef Packet2d half; }; -}; - -template<> struct unpacket_traits { typedef double type; enum {size=2, alignment=Aligned16}; typedef Packet2d half; }; - -inline std::ostream & operator <<(std::ostream & s, const Packet2l & v) -{ - union { - Packet2l v; - int64_t n[2]; - } vt; - vt.v = v; - s << vt.n[0] << ", " << vt.n[1]; - return s; -} - -inline std::ostream & operator <<(std::ostream & s, const Packet2d & v) -{ - union { - Packet2d v; - double n[2]; - } vt; - vt.v = v; - s << vt.n[0] << ", " << vt.n[1]; - return s; -} - -// Need to define them first or we get specialization after instantiation errors -template<> EIGEN_STRONG_INLINE Packet2d pload(const double* from) -{ - EIGEN_DEBUG_ALIGNED_LOAD + + inline std::ostream &operator<<(std::ostream &s, const Packet2l &v) + { + union { + Packet2l v; + int64_t n[2]; + } vt; + vt.v = v; + s << vt.n[0] << ", " << vt.n[1]; + return s; + } + + inline std::ostream &operator<<(std::ostream &s, const Packet2d &v) + { + union { + Packet2d v; + double n[2]; + } vt; + vt.v = v; + s << vt.n[0] << ", " << vt.n[1]; + return s; + } + + // Need to define them first or we get specialization after instantiation errors + template<> EIGEN_STRONG_INLINE Packet2d pload(const double *from) + { + EIGEN_DEBUG_ALIGNED_LOAD #ifdef __VSX__ - return vec_vsx_ld(0, from); + return vec_vsx_ld(0, from); #else - return vec_ld(0, from); + return vec_ld(0, from); #endif -} + } -template<> EIGEN_STRONG_INLINE void pstore(double* to, const Packet2d& from) -{ - EIGEN_DEBUG_ALIGNED_STORE + template<> EIGEN_STRONG_INLINE void pstore(double *to, const Packet2d &from) + { + EIGEN_DEBUG_ALIGNED_STORE #ifdef __VSX__ - vec_vsx_st(from, 0, to); + vec_vsx_st(from, 0, to); #else - vec_st(from, 0, to); + vec_st(from, 0, to); #endif -} - -template<> EIGEN_STRONG_INLINE Packet2d pset1(const double& from) { - Packet2d v = {from, from}; - return v; -} - -template<> EIGEN_STRONG_INLINE void -pbroadcast4(const double *a, - Packet2d& a0, Packet2d& a1, Packet2d& a2, Packet2d& a3) -{ - a1 = pload(a); - a0 = vec_splat_dbl<0>(a1); - a1 = vec_splat_dbl<1>(a1); - a3 = pload(a+2); - a2 = vec_splat_dbl<0>(a3); - a3 = vec_splat_dbl<1>(a3); -} - -template<> EIGEN_DEVICE_FUNC inline Packet2d pgather(const double* from, Index stride) -{ - double EIGEN_ALIGN16 af[2]; - af[0] = from[0*stride]; - af[1] = from[1*stride]; - return pload(af); -} -template<> EIGEN_DEVICE_FUNC inline void pscatter(double* to, const Packet2d& from, Index stride) -{ - double EIGEN_ALIGN16 af[2]; - pstore(af, from); - to[0*stride] = af[0]; - to[1*stride] = af[1]; -} - -template<> EIGEN_STRONG_INLINE Packet2d plset(const double& a) { return pset1(a) + p2d_COUNTDOWN; } - -template<> EIGEN_STRONG_INLINE Packet2d padd(const Packet2d& a, const Packet2d& b) { return a + b; } - -template<> EIGEN_STRONG_INLINE Packet2d psub(const Packet2d& a, const Packet2d& b) { return a - b; } - -template<> EIGEN_STRONG_INLINE Packet2d pnegate(const Packet2d& a) { return p2d_ZERO - a; } - -template<> EIGEN_STRONG_INLINE Packet2d pconj(const Packet2d& a) { return a; } - -template<> EIGEN_STRONG_INLINE Packet2d pmul(const Packet2d& a, const Packet2d& b) { return vec_madd(a,b,p2d_MZERO); } -template<> EIGEN_STRONG_INLINE Packet2d pdiv(const Packet2d& a, const Packet2d& b) { return vec_div(a,b); } - -// for some weird raisons, it has to be overloaded for packet of integers -template<> EIGEN_STRONG_INLINE Packet2d pmadd(const Packet2d& a, const Packet2d& b, const Packet2d& c) { return vec_madd(a, b, c); } - -template<> EIGEN_STRONG_INLINE Packet2d pmin(const Packet2d& a, const Packet2d& b) -{ - Packet2d ret; - __asm__ ("xvcmpgedp %x0,%x1,%x2\n\txxsel %x0,%x1,%x2,%x0" : "=&wa" (ret) : "wa" (a), "wa" (b)); - return ret; - } - -template<> EIGEN_STRONG_INLINE Packet2d pmax(const Packet2d& a, const Packet2d& b) -{ - Packet2d ret; - __asm__ ("xvcmpgtdp %x0,%x2,%x1\n\txxsel %x0,%x1,%x2,%x0" : "=&wa" (ret) : "wa" (a), "wa" (b)); - return ret; -} - -template<> EIGEN_STRONG_INLINE Packet2d pand(const Packet2d& a, const Packet2d& b) { return vec_and(a, b); } - -template<> EIGEN_STRONG_INLINE Packet2d por(const Packet2d& a, const Packet2d& b) { return vec_or(a, b); } - -template<> EIGEN_STRONG_INLINE Packet2d pxor(const Packet2d& a, const Packet2d& b) { return vec_xor(a, b); } - -template<> EIGEN_STRONG_INLINE Packet2d pandnot(const Packet2d& a, const Packet2d& b) { return vec_and(a, vec_nor(b, b)); } - -template<> EIGEN_STRONG_INLINE Packet2d pround(const Packet2d& a) { return vec_round(a); } -template<> EIGEN_STRONG_INLINE Packet2d pceil(const Packet2d& a) { return vec_ceil(a); } -template<> EIGEN_STRONG_INLINE Packet2d pfloor(const Packet2d& a) { return vec_floor(a); } - -template<> EIGEN_STRONG_INLINE Packet2d ploadu(const double* from) -{ - EIGEN_DEBUG_ALIGNED_LOAD - return (Packet2d) vec_vsx_ld((long)from & 15, (const double*) _EIGEN_ALIGNED_PTR(from)); -} - -template<> EIGEN_STRONG_INLINE Packet2d ploaddup(const double* from) -{ - Packet2d p; - if((std::ptrdiff_t(from) % 16) == 0) p = pload(from); - else p = ploadu(from); - return vec_splat_dbl<0>(p); -} - -template<> EIGEN_STRONG_INLINE void pstoreu(double* to, const Packet2d& from) -{ - EIGEN_DEBUG_ALIGNED_STORE - vec_vsx_st((Packet4f)from, (long)to & 15, (float*) _EIGEN_ALIGNED_PTR(to)); -} - -template<> EIGEN_STRONG_INLINE void prefetch(const double* addr) { EIGEN_PPC_PREFETCH(addr); } - -template<> EIGEN_STRONG_INLINE double pfirst(const Packet2d& a) { double EIGEN_ALIGN16 x[2]; pstore(x, a); return x[0]; } - -template<> EIGEN_STRONG_INLINE Packet2d preverse(const Packet2d& a) -{ - return reinterpret_cast(vec_perm(reinterpret_cast(a), reinterpret_cast(a), p16uc_REVERSE64)); -} -template<> EIGEN_STRONG_INLINE Packet2d pabs(const Packet2d& a) { return vec_abs(a); } - -template<> EIGEN_STRONG_INLINE double predux(const Packet2d& a) -{ - Packet2d b, sum; - b = reinterpret_cast(vec_sld(reinterpret_cast(a), reinterpret_cast(a), 8)); - sum = a + b; - return pfirst(sum); -} - -template<> EIGEN_STRONG_INLINE Packet2d preduxp(const Packet2d* vecs) -{ - Packet2d v[2], sum; - v[0] = vecs[0] + reinterpret_cast(vec_sld(reinterpret_cast(vecs[0]), reinterpret_cast(vecs[0]), 8)); - v[1] = vecs[1] + reinterpret_cast(vec_sld(reinterpret_cast(vecs[1]), reinterpret_cast(vecs[1]), 8)); + } + + template<> EIGEN_STRONG_INLINE Packet2d pset1(const double &from) + { + Packet2d v = { from, from }; + return v; + } + + template<> + EIGEN_STRONG_INLINE void + pbroadcast4(const double *a, Packet2d &a0, Packet2d &a1, Packet2d &a2, Packet2d &a3) + { + a1 = pload(a); + a0 = vec_splat_dbl<0>(a1); + a1 = vec_splat_dbl<1>(a1); + a3 = pload(a + 2); + a2 = vec_splat_dbl<0>(a3); + a3 = vec_splat_dbl<1>(a3); + } + + template<> EIGEN_DEVICE_FUNC inline Packet2d pgather(const double *from, Index stride) + { + double EIGEN_ALIGN16 af[2]; + af[0] = from[0 * stride]; + af[1] = from[1 * stride]; + return pload(af); + } + template<> EIGEN_DEVICE_FUNC inline void pscatter(double *to, const Packet2d &from, Index stride) + { + double EIGEN_ALIGN16 af[2]; + pstore(af, from); + to[0 * stride] = af[0]; + to[1 * stride] = af[1]; + } + + template<> EIGEN_STRONG_INLINE Packet2d plset(const double &a) + { + return pset1(a) + p2d_COUNTDOWN; + } + + template<> EIGEN_STRONG_INLINE Packet2d padd(const Packet2d &a, const Packet2d &b) { return a + b; } + + template<> EIGEN_STRONG_INLINE Packet2d psub(const Packet2d &a, const Packet2d &b) { return a - b; } + + template<> EIGEN_STRONG_INLINE Packet2d pnegate(const Packet2d &a) { return p2d_ZERO - a; } + + template<> EIGEN_STRONG_INLINE Packet2d pconj(const Packet2d &a) { return a; } + + template<> EIGEN_STRONG_INLINE Packet2d pmul(const Packet2d &a, const Packet2d &b) + { + return vec_madd(a, b, p2d_MZERO); + } + template<> EIGEN_STRONG_INLINE Packet2d pdiv(const Packet2d &a, const Packet2d &b) { return vec_div(a, b); } + + // for some weird raisons, it has to be overloaded for packet of integers + template<> EIGEN_STRONG_INLINE Packet2d pmadd(const Packet2d &a, const Packet2d &b, const Packet2d &c) + { + return vec_madd(a, b, c); + } + + template<> EIGEN_STRONG_INLINE Packet2d pmin(const Packet2d &a, const Packet2d &b) + { + Packet2d ret; + __asm__("xvcmpgedp %x0,%x1,%x2\n\txxsel %x0,%x1,%x2,%x0" : "=&wa"(ret) : "wa"(a), "wa"(b)); + return ret; + } + + template<> EIGEN_STRONG_INLINE Packet2d pmax(const Packet2d &a, const Packet2d &b) + { + Packet2d ret; + __asm__("xvcmpgtdp %x0,%x2,%x1\n\txxsel %x0,%x1,%x2,%x0" : "=&wa"(ret) : "wa"(a), "wa"(b)); + return ret; + } + + template<> EIGEN_STRONG_INLINE Packet2d pand(const Packet2d &a, const Packet2d &b) { return vec_and(a, b); } + + template<> EIGEN_STRONG_INLINE Packet2d por(const Packet2d &a, const Packet2d &b) { return vec_or(a, b); } + + template<> EIGEN_STRONG_INLINE Packet2d pxor(const Packet2d &a, const Packet2d &b) { return vec_xor(a, b); } + + template<> EIGEN_STRONG_INLINE Packet2d pandnot(const Packet2d &a, const Packet2d &b) + { + return vec_and(a, vec_nor(b, b)); + } + + template<> EIGEN_STRONG_INLINE Packet2d pround(const Packet2d &a) { return vec_round(a); } + template<> EIGEN_STRONG_INLINE Packet2d pceil(const Packet2d &a) { return vec_ceil(a); } + template<> EIGEN_STRONG_INLINE Packet2d pfloor(const Packet2d &a) { return vec_floor(a); } + + template<> EIGEN_STRONG_INLINE Packet2d ploadu(const double *from) + { + EIGEN_DEBUG_ALIGNED_LOAD + return (Packet2d)vec_vsx_ld((long)from & 15, (const double *)_EIGEN_ALIGNED_PTR(from)); + } + + template<> EIGEN_STRONG_INLINE Packet2d ploaddup(const double *from) + { + Packet2d p; + if ((std::ptrdiff_t(from) % 16) == 0) + p = pload(from); + else + p = ploadu(from); + return vec_splat_dbl<0>(p); + } + + template<> EIGEN_STRONG_INLINE void pstoreu(double *to, const Packet2d &from) + { + EIGEN_DEBUG_ALIGNED_STORE + vec_vsx_st((Packet4f)from, (long)to & 15, (float *)_EIGEN_ALIGNED_PTR(to)); + } + + template<> EIGEN_STRONG_INLINE void prefetch(const double *addr) { EIGEN_PPC_PREFETCH(addr); } + + template<> EIGEN_STRONG_INLINE double pfirst(const Packet2d &a) + { + double EIGEN_ALIGN16 x[2]; + pstore(x, a); + return x[0]; + } + + template<> EIGEN_STRONG_INLINE Packet2d preverse(const Packet2d &a) + { + return reinterpret_cast( + vec_perm(reinterpret_cast(a), reinterpret_cast(a), p16uc_REVERSE64)); + } + template<> EIGEN_STRONG_INLINE Packet2d pabs(const Packet2d &a) { return vec_abs(a); } + + template<> EIGEN_STRONG_INLINE double predux(const Packet2d &a) + { + Packet2d b, sum; + b = reinterpret_cast(vec_sld(reinterpret_cast(a), reinterpret_cast(a), 8)); + sum = a + b; + return pfirst(sum); + } + + template<> EIGEN_STRONG_INLINE Packet2d preduxp(const Packet2d *vecs) + { + Packet2d v[2], sum; + v[0] = vecs[0] + + reinterpret_cast( + vec_sld(reinterpret_cast(vecs[0]), reinterpret_cast(vecs[0]), 8)); + v[1] = vecs[1] + + reinterpret_cast( + vec_sld(reinterpret_cast(vecs[1]), reinterpret_cast(vecs[1]), 8)); #ifdef _BIG_ENDIAN - sum = reinterpret_cast(vec_sld(reinterpret_cast(v[0]), reinterpret_cast(v[1]), 8)); + sum = reinterpret_cast(vec_sld(reinterpret_cast(v[0]), reinterpret_cast(v[1]), 8)); #else - sum = reinterpret_cast(vec_sld(reinterpret_cast(v[1]), reinterpret_cast(v[0]), 8)); + sum = reinterpret_cast(vec_sld(reinterpret_cast(v[1]), reinterpret_cast(v[0]), 8)); #endif - return sum; -} -// Other reduction functions: -// mul -template<> EIGEN_STRONG_INLINE double predux_mul(const Packet2d& a) -{ - return pfirst(pmul(a, reinterpret_cast(vec_sld(reinterpret_cast(a), reinterpret_cast(a), 8)))); -} - -// min -template<> EIGEN_STRONG_INLINE double predux_min(const Packet2d& a) -{ - return pfirst(pmin(a, reinterpret_cast(vec_sld(reinterpret_cast(a), reinterpret_cast(a), 8)))); -} - -// max -template<> EIGEN_STRONG_INLINE double predux_max(const Packet2d& a) -{ - return pfirst(pmax(a, reinterpret_cast(vec_sld(reinterpret_cast(a), reinterpret_cast(a), 8)))); -} - -template -struct palign_impl -{ - static EIGEN_STRONG_INLINE void run(Packet2d& first, const Packet2d& second) - { - if (Offset == 1) + return sum; + } + // Other reduction functions: + // mul + template<> EIGEN_STRONG_INLINE double predux_mul(const Packet2d &a) + { + return pfirst( + pmul(a, reinterpret_cast(vec_sld(reinterpret_cast(a), reinterpret_cast(a), 8)))); + } + + // min + template<> EIGEN_STRONG_INLINE double predux_min(const Packet2d &a) + { + return pfirst( + pmin(a, reinterpret_cast(vec_sld(reinterpret_cast(a), reinterpret_cast(a), 8)))); + } + + // max + template<> EIGEN_STRONG_INLINE double predux_max(const Packet2d &a) + { + return pfirst( + pmax(a, reinterpret_cast(vec_sld(reinterpret_cast(a), reinterpret_cast(a), 8)))); + } + + template struct palign_impl + { + static EIGEN_STRONG_INLINE void run(Packet2d &first, const Packet2d &second) + { + if (Offset == 1) #ifdef _BIG_ENDIAN - first = reinterpret_cast(vec_sld(reinterpret_cast(first), reinterpret_cast(second), 8)); + first = reinterpret_cast( + vec_sld(reinterpret_cast(first), reinterpret_cast(second), 8)); #else - first = reinterpret_cast(vec_sld(reinterpret_cast(second), reinterpret_cast(first), 8)); + first = reinterpret_cast( + vec_sld(reinterpret_cast(second), reinterpret_cast(first), 8)); #endif - } -}; + } + }; -EIGEN_DEVICE_FUNC inline void -ptranspose(PacketBlock& kernel) { - Packet2d t0, t1; - t0 = vec_perm(kernel.packet[0], kernel.packet[1], p16uc_TRANSPOSE64_HI); - t1 = vec_perm(kernel.packet[0], kernel.packet[1], p16uc_TRANSPOSE64_LO); - kernel.packet[0] = t0; - kernel.packet[1] = t1; -} + EIGEN_DEVICE_FUNC inline void ptranspose(PacketBlock &kernel) + { + Packet2d t0, t1; + t0 = vec_perm(kernel.packet[0], kernel.packet[1], p16uc_TRANSPOSE64_HI); + t1 = vec_perm(kernel.packet[0], kernel.packet[1], p16uc_TRANSPOSE64_LO); + kernel.packet[0] = t0; + kernel.packet[1] = t1; + } -template<> EIGEN_STRONG_INLINE Packet2d pblend(const Selector<2>& ifPacket, const Packet2d& thenPacket, const Packet2d& elsePacket) { - Packet2l select = { ifPacket.select[0], ifPacket.select[1] }; - Packet2bl mask = reinterpret_cast( vec_cmpeq(reinterpret_cast(select), reinterpret_cast(p2l_ONE)) ); - return vec_sel(elsePacket, thenPacket, mask); -} -#endif // __VSX__ -} // end namespace internal + template<> + EIGEN_STRONG_INLINE Packet2d pblend(const Selector<2> &ifPacket, + const Packet2d &thenPacket, + const Packet2d &elsePacket) + { + Packet2l select = { ifPacket.select[0], ifPacket.select[1] }; + Packet2bl mask = + reinterpret_cast(vec_cmpeq(reinterpret_cast(select), reinterpret_cast(p2l_ONE))); + return vec_sel(elsePacket, thenPacket, mask); + } +#endif// __VSX__ +}// end namespace internal -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_PACKET_MATH_ALTIVEC_H +#endif// EIGEN_PACKET_MATH_ALTIVEC_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/arch/CUDA/Half.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/arch/CUDA/Half.h index 755e6209..12423f46 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/arch/CUDA/Half.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/arch/CUDA/Half.h @@ -50,106 +50,107 @@ struct half; namespace half_impl { #if !defined(EIGEN_HAS_CUDA_FP16) -// Make our own __half_raw definition that is similar to CUDA's. -struct __half_raw { - EIGEN_DEVICE_FUNC __half_raw() : x(0) {} - explicit EIGEN_DEVICE_FUNC __half_raw(unsigned short raw) : x(raw) {} - unsigned short x; -}; + // Make our own __half_raw definition that is similar to CUDA's. + struct __half_raw + { + EIGEN_DEVICE_FUNC __half_raw() : x(0) {} + explicit EIGEN_DEVICE_FUNC __half_raw(unsigned short raw) : x(raw) {} + unsigned short x; + }; #elif defined(EIGEN_CUDACC_VER) && EIGEN_CUDACC_VER < 90000 -// In CUDA < 9.0, __half is the equivalent of CUDA 9's __half_raw -typedef __half __half_raw; + // In CUDA < 9.0, __half is the equivalent of CUDA 9's __half_raw + typedef __half __half_raw; #endif -EIGEN_STRONG_INLINE EIGEN_DEVICE_FUNC __half_raw raw_uint16_to_half(unsigned short x); -EIGEN_STRONG_INLINE EIGEN_DEVICE_FUNC __half_raw float_to_half_rtne(float ff); -EIGEN_STRONG_INLINE EIGEN_DEVICE_FUNC float half_to_float(__half_raw h); + EIGEN_STRONG_INLINE EIGEN_DEVICE_FUNC __half_raw raw_uint16_to_half(unsigned short x); + EIGEN_STRONG_INLINE EIGEN_DEVICE_FUNC __half_raw float_to_half_rtne(float ff); + EIGEN_STRONG_INLINE EIGEN_DEVICE_FUNC float half_to_float(__half_raw h); -struct half_base : public __half_raw { - EIGEN_DEVICE_FUNC half_base() {} - EIGEN_DEVICE_FUNC half_base(const half_base& h) : __half_raw(h) {} - EIGEN_DEVICE_FUNC half_base(const __half_raw& h) : __half_raw(h) {} + struct half_base : public __half_raw + { + EIGEN_DEVICE_FUNC half_base() {} + EIGEN_DEVICE_FUNC half_base(const half_base &h) : __half_raw(h) {} + EIGEN_DEVICE_FUNC half_base(const __half_raw &h) : __half_raw(h) {} #if defined(EIGEN_HAS_CUDA_FP16) && defined(EIGEN_CUDACC_VER) && EIGEN_CUDACC_VER >= 90000 - EIGEN_DEVICE_FUNC half_base(const __half& h) : __half_raw(*(__half_raw*)&h) {} + EIGEN_DEVICE_FUNC half_base(const __half &h) : __half_raw(*(__half_raw *)&h) {} #endif -}; + }; -} // namespace half_impl +}// namespace half_impl // Class definition. -struct half : public half_impl::half_base { - #if !defined(EIGEN_HAS_CUDA_FP16) || (defined(EIGEN_CUDACC_VER) && EIGEN_CUDACC_VER < 90000) - typedef half_impl::__half_raw __half_raw; - #endif +struct half : public half_impl::half_base +{ +#if !defined(EIGEN_HAS_CUDA_FP16) || (defined(EIGEN_CUDACC_VER) && EIGEN_CUDACC_VER < 90000) + typedef half_impl::__half_raw __half_raw; +#endif EIGEN_DEVICE_FUNC half() {} - EIGEN_DEVICE_FUNC half(const __half_raw& h) : half_impl::half_base(h) {} - EIGEN_DEVICE_FUNC half(const half& h) : half_impl::half_base(h) {} + EIGEN_DEVICE_FUNC half(const __half_raw &h) : half_impl::half_base(h) {} + EIGEN_DEVICE_FUNC half(const half &h) : half_impl::half_base(h) {} #if defined(EIGEN_HAS_CUDA_FP16) && defined(EIGEN_CUDACC_VER) && EIGEN_CUDACC_VER >= 90000 - EIGEN_DEVICE_FUNC half(const __half& h) : half_impl::half_base(h) {} + EIGEN_DEVICE_FUNC half(const __half &h) : half_impl::half_base(h) {} #endif - explicit EIGEN_DEVICE_FUNC half(bool b) - : half_impl::half_base(half_impl::raw_uint16_to_half(b ? 0x3c00 : 0)) {} + explicit EIGEN_DEVICE_FUNC half(bool b) : half_impl::half_base(half_impl::raw_uint16_to_half(b ? 0x3c00 : 0)) {} template - explicit EIGEN_DEVICE_FUNC half(const T& val) - : half_impl::half_base(half_impl::float_to_half_rtne(static_cast(val))) {} - explicit EIGEN_DEVICE_FUNC half(float f) - : half_impl::half_base(half_impl::float_to_half_rtne(f)) {} + explicit EIGEN_DEVICE_FUNC half(const T &val) + : half_impl::half_base(half_impl::float_to_half_rtne(static_cast(val))) + {} + explicit EIGEN_DEVICE_FUNC half(float f) : half_impl::half_base(half_impl::float_to_half_rtne(f)) {} - EIGEN_DEVICE_FUNC EIGEN_EXPLICIT_CAST(bool) const { + EIGEN_DEVICE_FUNC EIGEN_EXPLICIT_CAST(bool) const + { // +0.0 and -0.0 become false, everything else becomes true. return (x & 0x7fff) != 0; } - EIGEN_DEVICE_FUNC EIGEN_EXPLICIT_CAST(signed char) const { + EIGEN_DEVICE_FUNC EIGEN_EXPLICIT_CAST(signed char) const + { return static_cast(half_impl::half_to_float(*this)); } - EIGEN_DEVICE_FUNC EIGEN_EXPLICIT_CAST(unsigned char) const { + EIGEN_DEVICE_FUNC EIGEN_EXPLICIT_CAST(unsigned char) const + { return static_cast(half_impl::half_to_float(*this)); } - EIGEN_DEVICE_FUNC EIGEN_EXPLICIT_CAST(short) const { - return static_cast(half_impl::half_to_float(*this)); - } - EIGEN_DEVICE_FUNC EIGEN_EXPLICIT_CAST(unsigned short) const { + EIGEN_DEVICE_FUNC EIGEN_EXPLICIT_CAST(short) const { return static_cast(half_impl::half_to_float(*this)); } + EIGEN_DEVICE_FUNC EIGEN_EXPLICIT_CAST(unsigned short) const + { return static_cast(half_impl::half_to_float(*this)); } - EIGEN_DEVICE_FUNC EIGEN_EXPLICIT_CAST(int) const { - return static_cast(half_impl::half_to_float(*this)); - } - EIGEN_DEVICE_FUNC EIGEN_EXPLICIT_CAST(unsigned int) const { + EIGEN_DEVICE_FUNC EIGEN_EXPLICIT_CAST(int) const { return static_cast(half_impl::half_to_float(*this)); } + EIGEN_DEVICE_FUNC EIGEN_EXPLICIT_CAST(unsigned int) const + { return static_cast(half_impl::half_to_float(*this)); } - EIGEN_DEVICE_FUNC EIGEN_EXPLICIT_CAST(long) const { - return static_cast(half_impl::half_to_float(*this)); - } - EIGEN_DEVICE_FUNC EIGEN_EXPLICIT_CAST(unsigned long) const { + EIGEN_DEVICE_FUNC EIGEN_EXPLICIT_CAST(long) const { return static_cast(half_impl::half_to_float(*this)); } + EIGEN_DEVICE_FUNC EIGEN_EXPLICIT_CAST(unsigned long) const + { return static_cast(half_impl::half_to_float(*this)); } - EIGEN_DEVICE_FUNC EIGEN_EXPLICIT_CAST(long long) const { + EIGEN_DEVICE_FUNC EIGEN_EXPLICIT_CAST(long long) const + { return static_cast(half_impl::half_to_float(*this)); } - EIGEN_DEVICE_FUNC EIGEN_EXPLICIT_CAST(unsigned long long) const { + EIGEN_DEVICE_FUNC EIGEN_EXPLICIT_CAST(unsigned long long) const + { return static_cast(half_to_float(*this)); } - EIGEN_DEVICE_FUNC EIGEN_EXPLICIT_CAST(float) const { - return half_impl::half_to_float(*this); - } - EIGEN_DEVICE_FUNC EIGEN_EXPLICIT_CAST(double) const { - return static_cast(half_impl::half_to_float(*this)); - } + EIGEN_DEVICE_FUNC EIGEN_EXPLICIT_CAST(float) const { return half_impl::half_to_float(*this); } + EIGEN_DEVICE_FUNC EIGEN_EXPLICIT_CAST(double) const { return static_cast(half_impl::half_to_float(*this)); } - EIGEN_DEVICE_FUNC half& operator=(const half& other) { + EIGEN_DEVICE_FUNC half &operator=(const half &other) + { x = other.x; return *this; } }; -} // end namespace Eigen +}// end namespace Eigen namespace std { -template<> -struct numeric_limits { +template<> struct numeric_limits +{ static const bool is_specialized = true; static const bool is_signed = true; static const bool is_integer = false; @@ -164,8 +165,10 @@ struct numeric_limits { static const bool is_bounded = false; static const bool is_modulo = false; static const int digits = 11; - static const int digits10 = 3; // according to http://half.sourceforge.net/structstd_1_1numeric__limits_3_01half__float_1_1half_01_4.html - static const int max_digits10 = 5; // according to http://half.sourceforge.net/structstd_1_1numeric__limits_3_01half__float_1_1half_01_4.html + static const int digits10 = + 3;// according to http://half.sourceforge.net/structstd_1_1numeric__limits_3_01half__float_1_1half_01_4.html + static const int max_digits10 = + 5;// according to http://half.sourceforge.net/structstd_1_1numeric__limits_3_01half__float_1_1half_01_4.html static const int radix = 2; static const int min_exponent = -13; static const int min_exponent10 = -4; @@ -174,9 +177,9 @@ struct numeric_limits { static const bool traps = true; static const bool tinyness_before = false; - static Eigen::half (min)() { return Eigen::half_impl::raw_uint16_to_half(0x400); } + static Eigen::half(min)() { return Eigen::half_impl::raw_uint16_to_half(0x400); } static Eigen::half lowest() { return Eigen::half_impl::raw_uint16_to_half(0xfbff); } - static Eigen::half (max)() { return Eigen::half_impl::raw_uint16_to_half(0x7bff); } + static Eigen::half(max)() { return Eigen::half_impl::raw_uint16_to_half(0x7bff); } static Eigen::half epsilon() { return Eigen::half_impl::raw_uint16_to_half(0x0800); } static Eigen::half round_error() { return Eigen::half(0.5); } static Eigen::half infinity() { return Eigen::half_impl::raw_uint16_to_half(0x7c00); } @@ -189,13 +192,16 @@ struct numeric_limits { // std::numeric_limits, std::numeric_limits, and // std::numeric_limits // https://stackoverflow.com/a/16519653/ -template<> -struct numeric_limits : numeric_limits {}; -template<> -struct numeric_limits : numeric_limits {}; -template<> -struct numeric_limits : numeric_limits {}; -} // end namespace std +template<> struct numeric_limits : numeric_limits +{ +}; +template<> struct numeric_limits : numeric_limits +{ +}; +template<> struct numeric_limits : numeric_limits +{ +}; +}// end namespace std namespace Eigen { @@ -203,444 +209,418 @@ namespace half_impl { #if defined(EIGEN_HAS_CUDA_FP16) && defined(EIGEN_CUDA_ARCH) && EIGEN_CUDA_ARCH >= 530 -// Intrinsics for native fp16 support. Note that on current hardware, -// these are no faster than float32_bits arithmetic (you need to use the half2 -// versions to get the ALU speed increased), but you do save the -// conversion steps back and forth. + // Intrinsics for native fp16 support. Note that on current hardware, + // these are no faster than float32_bits arithmetic (you need to use the half2 + // versions to get the ALU speed increased), but you do save the + // conversion steps back and forth. -EIGEN_STRONG_INLINE __device__ half operator + (const half& a, const half& b) { - return __hadd(a, b); -} -EIGEN_STRONG_INLINE __device__ half operator * (const half& a, const half& b) { - return __hmul(a, b); -} -EIGEN_STRONG_INLINE __device__ half operator - (const half& a, const half& b) { - return __hsub(a, b); -} -EIGEN_STRONG_INLINE __device__ half operator / (const half& a, const half& b) { - float num = __half2float(a); - float denom = __half2float(b); - return __float2half(num / denom); -} -EIGEN_STRONG_INLINE __device__ half operator - (const half& a) { - return __hneg(a); -} -EIGEN_STRONG_INLINE __device__ half& operator += (half& a, const half& b) { - a = a + b; - return a; -} -EIGEN_STRONG_INLINE __device__ half& operator *= (half& a, const half& b) { - a = a * b; - return a; -} -EIGEN_STRONG_INLINE __device__ half& operator -= (half& a, const half& b) { - a = a - b; - return a; -} -EIGEN_STRONG_INLINE __device__ half& operator /= (half& a, const half& b) { - a = a / b; - return a; -} -EIGEN_STRONG_INLINE __device__ bool operator == (const half& a, const half& b) { - return __heq(a, b); -} -EIGEN_STRONG_INLINE __device__ bool operator != (const half& a, const half& b) { - return __hne(a, b); -} -EIGEN_STRONG_INLINE __device__ bool operator < (const half& a, const half& b) { - return __hlt(a, b); -} -EIGEN_STRONG_INLINE __device__ bool operator <= (const half& a, const half& b) { - return __hle(a, b); -} -EIGEN_STRONG_INLINE __device__ bool operator > (const half& a, const half& b) { - return __hgt(a, b); -} -EIGEN_STRONG_INLINE __device__ bool operator >= (const half& a, const half& b) { - return __hge(a, b); -} + EIGEN_STRONG_INLINE __device__ half operator+(const half &a, const half &b) { return __hadd(a, b); } + EIGEN_STRONG_INLINE __device__ half operator*(const half &a, const half &b) { return __hmul(a, b); } + EIGEN_STRONG_INLINE __device__ half operator-(const half &a, const half &b) { return __hsub(a, b); } + EIGEN_STRONG_INLINE __device__ half operator/(const half &a, const half &b) + { + float num = __half2float(a); + float denom = __half2float(b); + return __float2half(num / denom); + } + EIGEN_STRONG_INLINE __device__ half operator-(const half &a) { return __hneg(a); } + EIGEN_STRONG_INLINE __device__ half &operator+=(half &a, const half &b) + { + a = a + b; + return a; + } + EIGEN_STRONG_INLINE __device__ half &operator*=(half &a, const half &b) + { + a = a * b; + return a; + } + EIGEN_STRONG_INLINE __device__ half &operator-=(half &a, const half &b) + { + a = a - b; + return a; + } + EIGEN_STRONG_INLINE __device__ half &operator/=(half &a, const half &b) + { + a = a / b; + return a; + } + EIGEN_STRONG_INLINE __device__ bool operator==(const half &a, const half &b) { return __heq(a, b); } + EIGEN_STRONG_INLINE __device__ bool operator!=(const half &a, const half &b) { return __hne(a, b); } + EIGEN_STRONG_INLINE __device__ bool operator<(const half &a, const half &b) { return __hlt(a, b); } + EIGEN_STRONG_INLINE __device__ bool operator<=(const half &a, const half &b) { return __hle(a, b); } + EIGEN_STRONG_INLINE __device__ bool operator>(const half &a, const half &b) { return __hgt(a, b); } + EIGEN_STRONG_INLINE __device__ bool operator>=(const half &a, const half &b) { return __hge(a, b); } -#else // Emulate support for half floats +#else// Emulate support for half floats -// Definitions for CPUs and older CUDA, mostly working through conversion -// to/from float32_bits. + // Definitions for CPUs and older CUDA, mostly working through conversion + // to/from float32_bits. -EIGEN_STRONG_INLINE EIGEN_DEVICE_FUNC half operator + (const half& a, const half& b) { - return half(float(a) + float(b)); -} -EIGEN_STRONG_INLINE EIGEN_DEVICE_FUNC half operator * (const half& a, const half& b) { - return half(float(a) * float(b)); -} -EIGEN_STRONG_INLINE EIGEN_DEVICE_FUNC half operator - (const half& a, const half& b) { - return half(float(a) - float(b)); -} -EIGEN_STRONG_INLINE EIGEN_DEVICE_FUNC half operator / (const half& a, const half& b) { - return half(float(a) / float(b)); -} -EIGEN_STRONG_INLINE EIGEN_DEVICE_FUNC half operator - (const half& a) { - half result; - result.x = a.x ^ 0x8000; - return result; -} -EIGEN_STRONG_INLINE EIGEN_DEVICE_FUNC half& operator += (half& a, const half& b) { - a = half(float(a) + float(b)); - return a; -} -EIGEN_STRONG_INLINE EIGEN_DEVICE_FUNC half& operator *= (half& a, const half& b) { - a = half(float(a) * float(b)); - return a; -} -EIGEN_STRONG_INLINE EIGEN_DEVICE_FUNC half& operator -= (half& a, const half& b) { - a = half(float(a) - float(b)); - return a; -} -EIGEN_STRONG_INLINE EIGEN_DEVICE_FUNC half& operator /= (half& a, const half& b) { - a = half(float(a) / float(b)); - return a; -} -EIGEN_STRONG_INLINE EIGEN_DEVICE_FUNC bool operator == (const half& a, const half& b) { - return numext::equal_strict(float(a),float(b)); -} -EIGEN_STRONG_INLINE EIGEN_DEVICE_FUNC bool operator != (const half& a, const half& b) { - return numext::not_equal_strict(float(a), float(b)); -} -EIGEN_STRONG_INLINE EIGEN_DEVICE_FUNC bool operator < (const half& a, const half& b) { - return float(a) < float(b); -} -EIGEN_STRONG_INLINE EIGEN_DEVICE_FUNC bool operator <= (const half& a, const half& b) { - return float(a) <= float(b); -} -EIGEN_STRONG_INLINE EIGEN_DEVICE_FUNC bool operator > (const half& a, const half& b) { - return float(a) > float(b); -} -EIGEN_STRONG_INLINE EIGEN_DEVICE_FUNC bool operator >= (const half& a, const half& b) { - return float(a) >= float(b); -} + EIGEN_STRONG_INLINE EIGEN_DEVICE_FUNC half operator+(const half &a, const half &b) + { + return half(float(a) + float(b)); + } + EIGEN_STRONG_INLINE EIGEN_DEVICE_FUNC half operator*(const half &a, const half &b) + { + return half(float(a) * float(b)); + } + EIGEN_STRONG_INLINE EIGEN_DEVICE_FUNC half operator-(const half &a, const half &b) + { + return half(float(a) - float(b)); + } + EIGEN_STRONG_INLINE EIGEN_DEVICE_FUNC half operator/(const half &a, const half &b) + { + return half(float(a) / float(b)); + } + EIGEN_STRONG_INLINE EIGEN_DEVICE_FUNC half operator-(const half &a) + { + half result; + result.x = a.x ^ 0x8000; + return result; + } + EIGEN_STRONG_INLINE EIGEN_DEVICE_FUNC half &operator+=(half &a, const half &b) + { + a = half(float(a) + float(b)); + return a; + } + EIGEN_STRONG_INLINE EIGEN_DEVICE_FUNC half &operator*=(half &a, const half &b) + { + a = half(float(a) * float(b)); + return a; + } + EIGEN_STRONG_INLINE EIGEN_DEVICE_FUNC half &operator-=(half &a, const half &b) + { + a = half(float(a) - float(b)); + return a; + } + EIGEN_STRONG_INLINE EIGEN_DEVICE_FUNC half &operator/=(half &a, const half &b) + { + a = half(float(a) / float(b)); + return a; + } + EIGEN_STRONG_INLINE EIGEN_DEVICE_FUNC bool operator==(const half &a, const half &b) + { + return numext::equal_strict(float(a), float(b)); + } + EIGEN_STRONG_INLINE EIGEN_DEVICE_FUNC bool operator!=(const half &a, const half &b) + { + return numext::not_equal_strict(float(a), float(b)); + } + EIGEN_STRONG_INLINE EIGEN_DEVICE_FUNC bool operator<(const half &a, const half &b) { return float(a) < float(b); } + EIGEN_STRONG_INLINE EIGEN_DEVICE_FUNC bool operator<=(const half &a, const half &b) { return float(a) <= float(b); } + EIGEN_STRONG_INLINE EIGEN_DEVICE_FUNC bool operator>(const half &a, const half &b) { return float(a) > float(b); } + EIGEN_STRONG_INLINE EIGEN_DEVICE_FUNC bool operator>=(const half &a, const half &b) { return float(a) >= float(b); } -#endif // Emulate support for half floats +#endif// Emulate support for half floats -// Division by an index. Do it in full float precision to avoid accuracy -// issues in converting the denominator to half. -EIGEN_STRONG_INLINE EIGEN_DEVICE_FUNC half operator / (const half& a, Index b) { - return half(static_cast(a) / static_cast(b)); -} + // Division by an index. Do it in full float precision to avoid accuracy + // issues in converting the denominator to half. + EIGEN_STRONG_INLINE EIGEN_DEVICE_FUNC half operator/(const half &a, Index b) + { + return half(static_cast(a) / static_cast(b)); + } -// Conversion routines, including fallbacks for the host or older CUDA. -// Note that newer Intel CPUs (Haswell or newer) have vectorized versions of -// these in hardware. If we need more performance on older/other CPUs, they are -// also possible to vectorize directly. + // Conversion routines, including fallbacks for the host or older CUDA. + // Note that newer Intel CPUs (Haswell or newer) have vectorized versions of + // these in hardware. If we need more performance on older/other CPUs, they are + // also possible to vectorize directly. -EIGEN_STRONG_INLINE EIGEN_DEVICE_FUNC __half_raw raw_uint16_to_half(unsigned short x) { - __half_raw h; - h.x = x; - return h; -} + EIGEN_STRONG_INLINE EIGEN_DEVICE_FUNC __half_raw raw_uint16_to_half(unsigned short x) + { + __half_raw h; + h.x = x; + return h; + } -union float32_bits { - unsigned int u; - float f; -}; + union float32_bits { + unsigned int u; + float f; + }; -EIGEN_STRONG_INLINE EIGEN_DEVICE_FUNC __half_raw float_to_half_rtne(float ff) { + EIGEN_STRONG_INLINE EIGEN_DEVICE_FUNC __half_raw float_to_half_rtne(float ff) + { #if defined(EIGEN_HAS_CUDA_FP16) && defined(EIGEN_CUDA_ARCH) && EIGEN_CUDA_ARCH >= 300 - __half tmp_ff = __float2half(ff); - return *(__half_raw*)&tmp_ff; + __half tmp_ff = __float2half(ff); + return *(__half_raw *)&tmp_ff; #elif defined(EIGEN_HAS_FP16_C) - __half_raw h; - h.x = _cvtss_sh(ff, 0); - return h; + __half_raw h; + h.x = _cvtss_sh(ff, 0); + return h; #else - float32_bits f; f.f = ff; - - const float32_bits f32infty = { 255 << 23 }; - const float32_bits f16max = { (127 + 16) << 23 }; - const float32_bits denorm_magic = { ((127 - 15) + (23 - 10) + 1) << 23 }; - unsigned int sign_mask = 0x80000000u; - __half_raw o; - o.x = static_cast(0x0u); - - unsigned int sign = f.u & sign_mask; - f.u ^= sign; - - // NOTE all the integer compares in this function can be safely - // compiled into signed compares since all operands are below - // 0x80000000. Important if you want fast straight SSE2 code - // (since there's no unsigned PCMPGTD). - - if (f.u >= f16max.u) { // result is Inf or NaN (all exponent bits set) - o.x = (f.u > f32infty.u) ? 0x7e00 : 0x7c00; // NaN->qNaN and Inf->Inf - } else { // (De)normalized number or zero - if (f.u < (113 << 23)) { // resulting FP16 is subnormal or zero - // use a magic value to align our 10 mantissa bits at the bottom of - // the float. as long as FP addition is round-to-nearest-even this - // just works. - f.f += denorm_magic.f; - - // and one integer subtract of the bias later, we have our final float! - o.x = static_cast(f.u - denorm_magic.u); - } else { - unsigned int mant_odd = (f.u >> 13) & 1; // resulting mantissa is odd - - // update exponent, rounding bias part 1 - f.u += ((unsigned int)(15 - 127) << 23) + 0xfff; - // rounding bias part 2 - f.u += mant_odd; - // take the bits! - o.x = static_cast(f.u >> 13); + float32_bits f; + f.f = ff; + + const float32_bits f32infty = { 255 << 23 }; + const float32_bits f16max = { (127 + 16) << 23 }; + const float32_bits denorm_magic = { ((127 - 15) + (23 - 10) + 1) << 23 }; + unsigned int sign_mask = 0x80000000u; + __half_raw o; + o.x = static_cast(0x0u); + + unsigned int sign = f.u & sign_mask; + f.u ^= sign; + + // NOTE all the integer compares in this function can be safely + // compiled into signed compares since all operands are below + // 0x80000000. Important if you want fast straight SSE2 code + // (since there's no unsigned PCMPGTD). + + if (f.u >= f16max.u) {// result is Inf or NaN (all exponent bits set) + o.x = (f.u > f32infty.u) ? 0x7e00 : 0x7c00;// NaN->qNaN and Inf->Inf + } else {// (De)normalized number or zero + if (f.u < (113 << 23)) {// resulting FP16 is subnormal or zero + // use a magic value to align our 10 mantissa bits at the bottom of + // the float. as long as FP addition is round-to-nearest-even this + // just works. + f.f += denorm_magic.f; + + // and one integer subtract of the bias later, we have our final float! + o.x = static_cast(f.u - denorm_magic.u); + } else { + unsigned int mant_odd = (f.u >> 13) & 1;// resulting mantissa is odd + + // update exponent, rounding bias part 1 + f.u += ((unsigned int)(15 - 127) << 23) + 0xfff; + // rounding bias part 2 + f.u += mant_odd; + // take the bits! + o.x = static_cast(f.u >> 13); + } } - } - o.x |= static_cast(sign >> 16); - return o; + o.x |= static_cast(sign >> 16); + return o; #endif -} + } -EIGEN_STRONG_INLINE EIGEN_DEVICE_FUNC float half_to_float(__half_raw h) { + EIGEN_STRONG_INLINE EIGEN_DEVICE_FUNC float half_to_float(__half_raw h) + { #if defined(EIGEN_HAS_CUDA_FP16) && defined(EIGEN_CUDA_ARCH) && EIGEN_CUDA_ARCH >= 300 - return __half2float(h); + return __half2float(h); #elif defined(EIGEN_HAS_FP16_C) - return _cvtsh_ss(h.x); + return _cvtsh_ss(h.x); #else - const float32_bits magic = { 113 << 23 }; - const unsigned int shifted_exp = 0x7c00 << 13; // exponent mask after shift - float32_bits o; - - o.u = (h.x & 0x7fff) << 13; // exponent/mantissa bits - unsigned int exp = shifted_exp & o.u; // just the exponent - o.u += (127 - 15) << 23; // exponent adjust - - // handle exponent special cases - if (exp == shifted_exp) { // Inf/NaN? - o.u += (128 - 16) << 23; // extra exp adjust - } else if (exp == 0) { // Zero/Denormal? - o.u += 1 << 23; // extra exp adjust - o.f -= magic.f; // renormalize - } + const float32_bits magic = { 113 << 23 }; + const unsigned int shifted_exp = 0x7c00 << 13;// exponent mask after shift + float32_bits o; + + o.u = (h.x & 0x7fff) << 13;// exponent/mantissa bits + unsigned int exp = shifted_exp & o.u;// just the exponent + o.u += (127 - 15) << 23;// exponent adjust + + // handle exponent special cases + if (exp == shifted_exp) {// Inf/NaN? + o.u += (128 - 16) << 23;// extra exp adjust + } else if (exp == 0) {// Zero/Denormal? + o.u += 1 << 23;// extra exp adjust + o.f -= magic.f;// renormalize + } - o.u |= (h.x & 0x8000) << 16; // sign bit - return o.f; + o.u |= (h.x & 0x8000) << 16;// sign bit + return o.f; #endif -} + } -// --- standard functions --- + // --- standard functions --- -EIGEN_STRONG_INLINE EIGEN_DEVICE_FUNC bool (isinf)(const half& a) { - return (a.x & 0x7fff) == 0x7c00; -} -EIGEN_STRONG_INLINE EIGEN_DEVICE_FUNC bool (isnan)(const half& a) { + EIGEN_STRONG_INLINE EIGEN_DEVICE_FUNC bool(isinf)(const half &a) { return (a.x & 0x7fff) == 0x7c00; } + EIGEN_STRONG_INLINE EIGEN_DEVICE_FUNC bool(isnan)(const half &a) + { #if defined(EIGEN_HAS_CUDA_FP16) && defined(EIGEN_CUDA_ARCH) && EIGEN_CUDA_ARCH >= 530 - return __hisnan(a); + return __hisnan(a); #else - return (a.x & 0x7fff) > 0x7c00; + return (a.x & 0x7fff) > 0x7c00; #endif -} -EIGEN_STRONG_INLINE EIGEN_DEVICE_FUNC bool (isfinite)(const half& a) { - return !(isinf EIGEN_NOT_A_MACRO (a)) && !(isnan EIGEN_NOT_A_MACRO (a)); -} + } + EIGEN_STRONG_INLINE EIGEN_DEVICE_FUNC bool(isfinite)(const half &a) + { + return !(isinf EIGEN_NOT_A_MACRO(a)) && !(isnan EIGEN_NOT_A_MACRO(a)); + } -EIGEN_STRONG_INLINE EIGEN_DEVICE_FUNC half abs(const half& a) { - half result; - result.x = a.x & 0x7FFF; - return result; -} -EIGEN_STRONG_INLINE EIGEN_DEVICE_FUNC half exp(const half& a) { + EIGEN_STRONG_INLINE EIGEN_DEVICE_FUNC half abs(const half &a) + { + half result; + result.x = a.x & 0x7FFF; + return result; + } + EIGEN_STRONG_INLINE EIGEN_DEVICE_FUNC half exp(const half &a) + { #if EIGEN_CUDACC_VER >= 80000 && defined EIGEN_CUDA_ARCH && EIGEN_CUDA_ARCH >= 530 - return half(hexp(a)); + return half(hexp(a)); #else - return half(::expf(float(a))); + return half(::expf(float(a))); #endif -} -EIGEN_STRONG_INLINE EIGEN_DEVICE_FUNC half log(const half& a) { + } + EIGEN_STRONG_INLINE EIGEN_DEVICE_FUNC half log(const half &a) + { #if defined(EIGEN_HAS_CUDA_FP16) && EIGEN_CUDACC_VER >= 80000 && defined(EIGEN_CUDA_ARCH) && EIGEN_CUDA_ARCH >= 530 - return half(::hlog(a)); + return half(::hlog(a)); #else - return half(::logf(float(a))); + return half(::logf(float(a))); #endif -} -EIGEN_STRONG_INLINE EIGEN_DEVICE_FUNC half log1p(const half& a) { - return half(numext::log1p(float(a))); -} -EIGEN_STRONG_INLINE EIGEN_DEVICE_FUNC half log10(const half& a) { - return half(::log10f(float(a))); -} -EIGEN_STRONG_INLINE EIGEN_DEVICE_FUNC half sqrt(const half& a) { + } + EIGEN_STRONG_INLINE EIGEN_DEVICE_FUNC half log1p(const half &a) { return half(numext::log1p(float(a))); } + EIGEN_STRONG_INLINE EIGEN_DEVICE_FUNC half log10(const half &a) { return half(::log10f(float(a))); } + EIGEN_STRONG_INLINE EIGEN_DEVICE_FUNC half sqrt(const half &a) + { #if EIGEN_CUDACC_VER >= 80000 && defined EIGEN_CUDA_ARCH && EIGEN_CUDA_ARCH >= 530 - return half(hsqrt(a)); + return half(hsqrt(a)); #else return half(::sqrtf(float(a))); #endif -} -EIGEN_STRONG_INLINE EIGEN_DEVICE_FUNC half pow(const half& a, const half& b) { - return half(::powf(float(a), float(b))); -} -EIGEN_STRONG_INLINE EIGEN_DEVICE_FUNC half sin(const half& a) { - return half(::sinf(float(a))); -} -EIGEN_STRONG_INLINE EIGEN_DEVICE_FUNC half cos(const half& a) { - return half(::cosf(float(a))); -} -EIGEN_STRONG_INLINE EIGEN_DEVICE_FUNC half tan(const half& a) { - return half(::tanf(float(a))); -} -EIGEN_STRONG_INLINE EIGEN_DEVICE_FUNC half tanh(const half& a) { - return half(::tanhf(float(a))); -} -EIGEN_STRONG_INLINE EIGEN_DEVICE_FUNC half floor(const half& a) { + } + EIGEN_STRONG_INLINE EIGEN_DEVICE_FUNC half pow(const half &a, const half &b) + { + return half(::powf(float(a), float(b))); + } + EIGEN_STRONG_INLINE EIGEN_DEVICE_FUNC half sin(const half &a) { return half(::sinf(float(a))); } + EIGEN_STRONG_INLINE EIGEN_DEVICE_FUNC half cos(const half &a) { return half(::cosf(float(a))); } + EIGEN_STRONG_INLINE EIGEN_DEVICE_FUNC half tan(const half &a) { return half(::tanf(float(a))); } + EIGEN_STRONG_INLINE EIGEN_DEVICE_FUNC half tanh(const half &a) { return half(::tanhf(float(a))); } + EIGEN_STRONG_INLINE EIGEN_DEVICE_FUNC half floor(const half &a) + { #if EIGEN_CUDACC_VER >= 80000 && defined EIGEN_CUDA_ARCH && EIGEN_CUDA_ARCH >= 300 - return half(hfloor(a)); + return half(hfloor(a)); #else - return half(::floorf(float(a))); + return half(::floorf(float(a))); #endif -} -EIGEN_STRONG_INLINE EIGEN_DEVICE_FUNC half ceil(const half& a) { + } + EIGEN_STRONG_INLINE EIGEN_DEVICE_FUNC half ceil(const half &a) + { #if EIGEN_CUDACC_VER >= 80000 && defined EIGEN_CUDA_ARCH && EIGEN_CUDA_ARCH >= 300 - return half(hceil(a)); + return half(hceil(a)); #else - return half(::ceilf(float(a))); + return half(::ceilf(float(a))); #endif -} + } -EIGEN_STRONG_INLINE EIGEN_DEVICE_FUNC half (min)(const half& a, const half& b) { + EIGEN_STRONG_INLINE EIGEN_DEVICE_FUNC half(min)(const half &a, const half &b) + { #if defined(EIGEN_HAS_CUDA_FP16) && defined(EIGEN_CUDA_ARCH) && EIGEN_CUDA_ARCH >= 530 - return __hlt(b, a) ? b : a; + return __hlt(b, a) ? b : a; #else - const float f1 = static_cast(a); - const float f2 = static_cast(b); - return f2 < f1 ? b : a; + const float f1 = static_cast(a); + const float f2 = static_cast(b); + return f2 < f1 ? b : a; #endif -} -EIGEN_STRONG_INLINE EIGEN_DEVICE_FUNC half (max)(const half& a, const half& b) { + } + EIGEN_STRONG_INLINE EIGEN_DEVICE_FUNC half(max)(const half &a, const half &b) + { #if defined(EIGEN_HAS_CUDA_FP16) && defined(EIGEN_CUDA_ARCH) && EIGEN_CUDA_ARCH >= 530 - return __hlt(a, b) ? b : a; + return __hlt(a, b) ? b : a; #else - const float f1 = static_cast(a); - const float f2 = static_cast(b); - return f1 < f2 ? b : a; + const float f1 = static_cast(a); + const float f2 = static_cast(b); + return f1 < f2 ? b : a; #endif -} + } -EIGEN_ALWAYS_INLINE std::ostream& operator << (std::ostream& os, const half& v) { - os << static_cast(v); - return os; -} + EIGEN_ALWAYS_INLINE std::ostream &operator<<(std::ostream &os, const half &v) + { + os << static_cast(v); + return os; + } -} // end namespace half_impl +}// end namespace half_impl // import Eigen::half_impl::half into Eigen namespace // using half_impl::half; namespace internal { -template<> -struct random_default_impl -{ - static inline half run(const half& x, const half& y) + template<> struct random_default_impl { - return x + (y-x) * half(float(std::rand()) / float(RAND_MAX)); - } - static inline half run() - { - return run(half(-1.f), half(1.f)); - } -}; + static inline half run(const half &x, const half &y) + { + return x + (y - x) * half(float(std::rand()) / float(RAND_MAX)); + } + static inline half run() { return run(half(-1.f), half(1.f)); } + }; -template<> struct is_arithmetic { enum { value = true }; }; + template<> struct is_arithmetic + { + enum { value = true }; + }; -} // end namespace internal +}// end namespace internal -template<> struct NumTraits - : GenericNumTraits +template<> struct NumTraits : GenericNumTraits { - enum { - IsSigned = true, - IsInteger = false, - IsComplex = false, - RequireInitialization = false - }; + enum { IsSigned = true, IsInteger = false, IsComplex = false, RequireInitialization = false }; - EIGEN_DEVICE_FUNC static EIGEN_STRONG_INLINE Eigen::half epsilon() { - return half_impl::raw_uint16_to_half(0x0800); - } + EIGEN_DEVICE_FUNC static EIGEN_STRONG_INLINE Eigen::half epsilon() { return half_impl::raw_uint16_to_half(0x0800); } EIGEN_DEVICE_FUNC static EIGEN_STRONG_INLINE Eigen::half dummy_precision() { return Eigen::half(1e-2f); } - EIGEN_DEVICE_FUNC static EIGEN_STRONG_INLINE Eigen::half highest() { - return half_impl::raw_uint16_to_half(0x7bff); - } - EIGEN_DEVICE_FUNC static EIGEN_STRONG_INLINE Eigen::half lowest() { - return half_impl::raw_uint16_to_half(0xfbff); - } - EIGEN_DEVICE_FUNC static EIGEN_STRONG_INLINE Eigen::half infinity() { - return half_impl::raw_uint16_to_half(0x7c00); - } - EIGEN_DEVICE_FUNC static EIGEN_STRONG_INLINE Eigen::half quiet_NaN() { - return half_impl::raw_uint16_to_half(0x7c01); - } + EIGEN_DEVICE_FUNC static EIGEN_STRONG_INLINE Eigen::half highest() { return half_impl::raw_uint16_to_half(0x7bff); } + EIGEN_DEVICE_FUNC static EIGEN_STRONG_INLINE Eigen::half lowest() { return half_impl::raw_uint16_to_half(0xfbff); } + EIGEN_DEVICE_FUNC static EIGEN_STRONG_INLINE Eigen::half infinity() { return half_impl::raw_uint16_to_half(0x7c00); } + EIGEN_DEVICE_FUNC static EIGEN_STRONG_INLINE Eigen::half quiet_NaN() { return half_impl::raw_uint16_to_half(0x7c01); } }; -} // end namespace Eigen +}// end namespace Eigen // C-like standard mathematical functions and trancendentals. -EIGEN_STRONG_INLINE EIGEN_DEVICE_FUNC Eigen::half fabsh(const Eigen::half& a) { +EIGEN_STRONG_INLINE EIGEN_DEVICE_FUNC Eigen::half fabsh(const Eigen::half &a) +{ Eigen::half result; result.x = a.x & 0x7FFF; return result; } -EIGEN_STRONG_INLINE EIGEN_DEVICE_FUNC Eigen::half exph(const Eigen::half& a) { - return Eigen::half(::expf(float(a))); -} -EIGEN_STRONG_INLINE EIGEN_DEVICE_FUNC Eigen::half logh(const Eigen::half& a) { +EIGEN_STRONG_INLINE EIGEN_DEVICE_FUNC Eigen::half exph(const Eigen::half &a) { return Eigen::half(::expf(float(a))); } +EIGEN_STRONG_INLINE EIGEN_DEVICE_FUNC Eigen::half logh(const Eigen::half &a) +{ #if EIGEN_CUDACC_VER >= 80000 && defined(EIGEN_CUDA_ARCH) && EIGEN_CUDA_ARCH >= 530 return Eigen::half(::hlog(a)); #else return Eigen::half(::logf(float(a))); #endif } -EIGEN_STRONG_INLINE EIGEN_DEVICE_FUNC Eigen::half sqrth(const Eigen::half& a) { - return Eigen::half(::sqrtf(float(a))); -} -EIGEN_STRONG_INLINE EIGEN_DEVICE_FUNC Eigen::half powh(const Eigen::half& a, const Eigen::half& b) { +EIGEN_STRONG_INLINE EIGEN_DEVICE_FUNC Eigen::half sqrth(const Eigen::half &a) { return Eigen::half(::sqrtf(float(a))); } +EIGEN_STRONG_INLINE EIGEN_DEVICE_FUNC Eigen::half powh(const Eigen::half &a, const Eigen::half &b) +{ return Eigen::half(::powf(float(a), float(b))); } -EIGEN_STRONG_INLINE EIGEN_DEVICE_FUNC Eigen::half floorh(const Eigen::half& a) { +EIGEN_STRONG_INLINE EIGEN_DEVICE_FUNC Eigen::half floorh(const Eigen::half &a) +{ return Eigen::half(::floorf(float(a))); } -EIGEN_STRONG_INLINE EIGEN_DEVICE_FUNC Eigen::half ceilh(const Eigen::half& a) { - return Eigen::half(::ceilf(float(a))); -} +EIGEN_STRONG_INLINE EIGEN_DEVICE_FUNC Eigen::half ceilh(const Eigen::half &a) { return Eigen::half(::ceilf(float(a))); } namespace std { #if __cplusplus > 199711L -template <> -struct hash { - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE std::size_t operator()(const Eigen::half& a) const { +template<> struct hash +{ + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE std::size_t operator()(const Eigen::half &a) const + { return static_cast(a.x); } }; #endif -} // end namespace std +}// end namespace std // Add the missing shfl_xor intrinsic #if defined(EIGEN_CUDA_ARCH) && EIGEN_CUDA_ARCH >= 300 -__device__ EIGEN_STRONG_INLINE Eigen::half __shfl_xor(Eigen::half var, int laneMask, int width=warpSize) { - #if EIGEN_CUDACC_VER < 90000 +__device__ EIGEN_STRONG_INLINE Eigen::half __shfl_xor(Eigen::half var, int laneMask, int width = warpSize) +{ +#if EIGEN_CUDACC_VER < 90000 return static_cast(__shfl_xor(static_cast(var), laneMask, width)); - #else +#else return static_cast(__shfl_xor_sync(0xFFFFFFFF, static_cast(var), laneMask, width)); - #endif +#endif } #endif // ldg() has an overload for __half_raw, but we also need one for Eigen::half. #if defined(EIGEN_CUDA_ARCH) && EIGEN_CUDA_ARCH >= 350 -EIGEN_STRONG_INLINE EIGEN_DEVICE_FUNC Eigen::half __ldg(const Eigen::half* ptr) { - return Eigen::half_impl::raw_uint16_to_half( - __ldg(reinterpret_cast(ptr))); +EIGEN_STRONG_INLINE EIGEN_DEVICE_FUNC Eigen::half __ldg(const Eigen::half *ptr) +{ + return Eigen::half_impl::raw_uint16_to_half(__ldg(reinterpret_cast(ptr))); } #endif @@ -649,26 +629,17 @@ EIGEN_STRONG_INLINE EIGEN_DEVICE_FUNC Eigen::half __ldg(const Eigen::half* ptr) namespace Eigen { namespace numext { -template<> -EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE -bool (isnan)(const Eigen::half& h) { - return (half_impl::isnan)(h); -} + template<> EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE bool(isnan)(const Eigen::half &h) { return (half_impl::isnan)(h); } -template<> -EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE -bool (isinf)(const Eigen::half& h) { - return (half_impl::isinf)(h); -} + template<> EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE bool(isinf)(const Eigen::half &h) { return (half_impl::isinf)(h); } -template<> -EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE -bool (isfinite)(const Eigen::half& h) { - return (half_impl::isfinite)(h); -} + template<> EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE bool(isfinite)(const Eigen::half &h) + { + return (half_impl::isfinite)(h); + } -} // namespace Eigen -} // namespace numext +}// namespace numext +}// namespace Eigen #endif -#endif // EIGEN_HALF_CUDA_H +#endif// EIGEN_HALF_CUDA_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/arch/CUDA/MathFunctions.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/arch/CUDA/MathFunctions.h index 0348b41d..b8f8d74d 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/arch/CUDA/MathFunctions.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/arch/CUDA/MathFunctions.h @@ -18,74 +18,64 @@ namespace internal { // introduce conflicts between these packet_traits definitions and the ones // we'll use on the host side (SSE, AVX, ...) #if defined(__CUDACC__) && defined(EIGEN_USE_GPU) -template<> EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE -float4 plog(const float4& a) -{ - return make_float4(logf(a.x), logf(a.y), logf(a.z), logf(a.w)); -} - -template<> EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE -double2 plog(const double2& a) -{ - using ::log; - return make_double2(log(a.x), log(a.y)); -} - -template<> EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE -float4 plog1p(const float4& a) -{ - return make_float4(log1pf(a.x), log1pf(a.y), log1pf(a.z), log1pf(a.w)); -} - -template<> EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE -double2 plog1p(const double2& a) -{ - return make_double2(log1p(a.x), log1p(a.y)); -} - -template<> EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE -float4 pexp(const float4& a) -{ - return make_float4(expf(a.x), expf(a.y), expf(a.z), expf(a.w)); -} - -template<> EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE -double2 pexp(const double2& a) -{ - using ::exp; - return make_double2(exp(a.x), exp(a.y)); -} - -template<> EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE -float4 psqrt(const float4& a) -{ - return make_float4(sqrtf(a.x), sqrtf(a.y), sqrtf(a.z), sqrtf(a.w)); -} - -template<> EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE -double2 psqrt(const double2& a) -{ - using ::sqrt; - return make_double2(sqrt(a.x), sqrt(a.y)); -} - -template<> EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE -float4 prsqrt(const float4& a) -{ - return make_float4(rsqrtf(a.x), rsqrtf(a.y), rsqrtf(a.z), rsqrtf(a.w)); -} - -template<> EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE -double2 prsqrt(const double2& a) -{ - return make_double2(rsqrt(a.x), rsqrt(a.y)); -} + template<> EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE float4 plog(const float4 &a) + { + return make_float4(logf(a.x), logf(a.y), logf(a.z), logf(a.w)); + } + + template<> EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE double2 plog(const double2 &a) + { + using ::log; + return make_double2(log(a.x), log(a.y)); + } + + template<> EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE float4 plog1p(const float4 &a) + { + return make_float4(log1pf(a.x), log1pf(a.y), log1pf(a.z), log1pf(a.w)); + } + + template<> EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE double2 plog1p(const double2 &a) + { + return make_double2(log1p(a.x), log1p(a.y)); + } + + template<> EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE float4 pexp(const float4 &a) + { + return make_float4(expf(a.x), expf(a.y), expf(a.z), expf(a.w)); + } + + template<> EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE double2 pexp(const double2 &a) + { + using ::exp; + return make_double2(exp(a.x), exp(a.y)); + } + + template<> EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE float4 psqrt(const float4 &a) + { + return make_float4(sqrtf(a.x), sqrtf(a.y), sqrtf(a.z), sqrtf(a.w)); + } + + template<> EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE double2 psqrt(const double2 &a) + { + using ::sqrt; + return make_double2(sqrt(a.x), sqrt(a.y)); + } + + template<> EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE float4 prsqrt(const float4 &a) + { + return make_float4(rsqrtf(a.x), rsqrtf(a.y), rsqrtf(a.z), rsqrtf(a.w)); + } + + template<> EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE double2 prsqrt(const double2 &a) + { + return make_double2(rsqrt(a.x), rsqrt(a.y)); + } #endif -} // end namespace internal +}// end namespace internal -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_MATH_FUNCTIONS_CUDA_H +#endif// EIGEN_MATH_FUNCTIONS_CUDA_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/arch/CUDA/PacketMath.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/arch/CUDA/PacketMath.h index 4dda6318..c7adeb5c 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/arch/CUDA/PacketMath.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/arch/CUDA/PacketMath.h @@ -18,316 +18,352 @@ namespace internal { // introduce conflicts between these packet_traits definitions and the ones // we'll use on the host side (SSE, AVX, ...) #if defined(__CUDACC__) && defined(EIGEN_USE_GPU) -template<> struct is_arithmetic { enum { value = true }; }; -template<> struct is_arithmetic { enum { value = true }; }; - -template<> struct packet_traits : default_packet_traits -{ - typedef float4 type; - typedef float4 half; - enum { - Vectorizable = 1, - AlignedOnScalar = 1, - size=4, - HasHalfPacket = 0, - - HasDiv = 1, - HasSin = 0, - HasCos = 0, - HasLog = 1, - HasExp = 1, - HasSqrt = 1, - HasRsqrt = 1, - HasLGamma = 1, - HasDiGamma = 1, - HasZeta = 1, - HasPolygamma = 1, - HasErf = 1, - HasErfc = 1, - HasIGamma = 1, - HasIGammac = 1, - HasBetaInc = 1, - - HasBlend = 0, + template<> struct is_arithmetic + { + enum { value = true }; }; -}; - -template<> struct packet_traits : default_packet_traits -{ - typedef double2 type; - typedef double2 half; - enum { - Vectorizable = 1, - AlignedOnScalar = 1, - size=2, - HasHalfPacket = 0, - - HasDiv = 1, - HasLog = 1, - HasExp = 1, - HasSqrt = 1, - HasRsqrt = 1, - HasLGamma = 1, - HasDiGamma = 1, - HasZeta = 1, - HasPolygamma = 1, - HasErf = 1, - HasErfc = 1, - HasIGamma = 1, - HasIGammac = 1, - HasBetaInc = 1, - - HasBlend = 0, + template<> struct is_arithmetic + { + enum { value = true }; }; -}; - - -template<> struct unpacket_traits { typedef float type; enum {size=4, alignment=Aligned16}; typedef float4 half; }; -template<> struct unpacket_traits { typedef double type; enum {size=2, alignment=Aligned16}; typedef double2 half; }; - -template<> EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE float4 pset1(const float& from) { - return make_float4(from, from, from, from); -} -template<> EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE double2 pset1(const double& from) { - return make_double2(from, from); -} - - -template<> EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE float4 plset(const float& a) { - return make_float4(a, a+1, a+2, a+3); -} -template<> EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE double2 plset(const double& a) { - return make_double2(a, a+1); -} - -template<> EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE float4 padd(const float4& a, const float4& b) { - return make_float4(a.x+b.x, a.y+b.y, a.z+b.z, a.w+b.w); -} -template<> EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE double2 padd(const double2& a, const double2& b) { - return make_double2(a.x+b.x, a.y+b.y); -} - -template<> EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE float4 psub(const float4& a, const float4& b) { - return make_float4(a.x-b.x, a.y-b.y, a.z-b.z, a.w-b.w); -} -template<> EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE double2 psub(const double2& a, const double2& b) { - return make_double2(a.x-b.x, a.y-b.y); -} - -template<> EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE float4 pnegate(const float4& a) { - return make_float4(-a.x, -a.y, -a.z, -a.w); -} -template<> EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE double2 pnegate(const double2& a) { - return make_double2(-a.x, -a.y); -} - -template<> EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE float4 pconj(const float4& a) { return a; } -template<> EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE double2 pconj(const double2& a) { return a; } - -template<> EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE float4 pmul(const float4& a, const float4& b) { - return make_float4(a.x*b.x, a.y*b.y, a.z*b.z, a.w*b.w); -} -template<> EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE double2 pmul(const double2& a, const double2& b) { - return make_double2(a.x*b.x, a.y*b.y); -} - -template<> EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE float4 pdiv(const float4& a, const float4& b) { - return make_float4(a.x/b.x, a.y/b.y, a.z/b.z, a.w/b.w); -} -template<> EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE double2 pdiv(const double2& a, const double2& b) { - return make_double2(a.x/b.x, a.y/b.y); -} - -template<> EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE float4 pmin(const float4& a, const float4& b) { - return make_float4(fminf(a.x, b.x), fminf(a.y, b.y), fminf(a.z, b.z), fminf(a.w, b.w)); -} -template<> EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE double2 pmin(const double2& a, const double2& b) { - return make_double2(fmin(a.x, b.x), fmin(a.y, b.y)); -} - -template<> EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE float4 pmax(const float4& a, const float4& b) { - return make_float4(fmaxf(a.x, b.x), fmaxf(a.y, b.y), fmaxf(a.z, b.z), fmaxf(a.w, b.w)); -} -template<> EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE double2 pmax(const double2& a, const double2& b) { - return make_double2(fmax(a.x, b.x), fmax(a.y, b.y)); -} - -template<> EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE float4 pload(const float* from) { - return *reinterpret_cast(from); -} - -template<> EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE double2 pload(const double* from) { - return *reinterpret_cast(from); -} - -template<> EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE float4 ploadu(const float* from) { - return make_float4(from[0], from[1], from[2], from[3]); -} -template<> EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE double2 ploadu(const double* from) { - return make_double2(from[0], from[1]); -} - -template<> EIGEN_STRONG_INLINE float4 ploaddup(const float* from) { - return make_float4(from[0], from[0], from[1], from[1]); -} -template<> EIGEN_STRONG_INLINE double2 ploaddup(const double* from) { - return make_double2(from[0], from[0]); -} - -template<> EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void pstore(float* to, const float4& from) { - *reinterpret_cast(to) = from; -} - -template<> EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void pstore(double* to, const double2& from) { - *reinterpret_cast(to) = from; -} - -template<> EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void pstoreu(float* to, const float4& from) { - to[0] = from.x; - to[1] = from.y; - to[2] = from.z; - to[3] = from.w; -} - -template<> EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void pstoreu(double* to, const double2& from) { - to[0] = from.x; - to[1] = from.y; -} - -template<> -EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE float4 ploadt_ro(const float* from) { + + template<> struct packet_traits : default_packet_traits + { + typedef float4 type; + typedef float4 half; + enum { + Vectorizable = 1, + AlignedOnScalar = 1, + size = 4, + HasHalfPacket = 0, + + HasDiv = 1, + HasSin = 0, + HasCos = 0, + HasLog = 1, + HasExp = 1, + HasSqrt = 1, + HasRsqrt = 1, + HasLGamma = 1, + HasDiGamma = 1, + HasZeta = 1, + HasPolygamma = 1, + HasErf = 1, + HasErfc = 1, + HasIGamma = 1, + HasIGammac = 1, + HasBetaInc = 1, + + HasBlend = 0, + }; + }; + + template<> struct packet_traits : default_packet_traits + { + typedef double2 type; + typedef double2 half; + enum { + Vectorizable = 1, + AlignedOnScalar = 1, + size = 2, + HasHalfPacket = 0, + + HasDiv = 1, + HasLog = 1, + HasExp = 1, + HasSqrt = 1, + HasRsqrt = 1, + HasLGamma = 1, + HasDiGamma = 1, + HasZeta = 1, + HasPolygamma = 1, + HasErf = 1, + HasErfc = 1, + HasIGamma = 1, + HasIGammac = 1, + HasBetaInc = 1, + + HasBlend = 0, + }; + }; + + + template<> struct unpacket_traits + { + typedef float type; + enum { size = 4, alignment = Aligned16 }; + typedef float4 half; + }; + template<> struct unpacket_traits + { + typedef double type; + enum { size = 2, alignment = Aligned16 }; + typedef double2 half; + }; + + template<> EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE float4 pset1(const float &from) + { + return make_float4(from, from, from, from); + } + template<> EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE double2 pset1(const double &from) + { + return make_double2(from, from); + } + + + template<> EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE float4 plset(const float &a) + { + return make_float4(a, a + 1, a + 2, a + 3); + } + template<> EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE double2 plset(const double &a) + { + return make_double2(a, a + 1); + } + + template<> EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE float4 padd(const float4 &a, const float4 &b) + { + return make_float4(a.x + b.x, a.y + b.y, a.z + b.z, a.w + b.w); + } + template<> EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE double2 padd(const double2 &a, const double2 &b) + { + return make_double2(a.x + b.x, a.y + b.y); + } + + template<> EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE float4 psub(const float4 &a, const float4 &b) + { + return make_float4(a.x - b.x, a.y - b.y, a.z - b.z, a.w - b.w); + } + template<> EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE double2 psub(const double2 &a, const double2 &b) + { + return make_double2(a.x - b.x, a.y - b.y); + } + + template<> EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE float4 pnegate(const float4 &a) + { + return make_float4(-a.x, -a.y, -a.z, -a.w); + } + template<> EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE double2 pnegate(const double2 &a) + { + return make_double2(-a.x, -a.y); + } + + template<> EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE float4 pconj(const float4 &a) { return a; } + template<> EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE double2 pconj(const double2 &a) { return a; } + + template<> EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE float4 pmul(const float4 &a, const float4 &b) + { + return make_float4(a.x * b.x, a.y * b.y, a.z * b.z, a.w * b.w); + } + template<> EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE double2 pmul(const double2 &a, const double2 &b) + { + return make_double2(a.x * b.x, a.y * b.y); + } + + template<> EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE float4 pdiv(const float4 &a, const float4 &b) + { + return make_float4(a.x / b.x, a.y / b.y, a.z / b.z, a.w / b.w); + } + template<> EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE double2 pdiv(const double2 &a, const double2 &b) + { + return make_double2(a.x / b.x, a.y / b.y); + } + + template<> EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE float4 pmin(const float4 &a, const float4 &b) + { + return make_float4(fminf(a.x, b.x), fminf(a.y, b.y), fminf(a.z, b.z), fminf(a.w, b.w)); + } + template<> EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE double2 pmin(const double2 &a, const double2 &b) + { + return make_double2(fmin(a.x, b.x), fmin(a.y, b.y)); + } + + template<> EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE float4 pmax(const float4 &a, const float4 &b) + { + return make_float4(fmaxf(a.x, b.x), fmaxf(a.y, b.y), fmaxf(a.z, b.z), fmaxf(a.w, b.w)); + } + template<> EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE double2 pmax(const double2 &a, const double2 &b) + { + return make_double2(fmax(a.x, b.x), fmax(a.y, b.y)); + } + + template<> EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE float4 pload(const float *from) + { + return *reinterpret_cast(from); + } + + template<> EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE double2 pload(const double *from) + { + return *reinterpret_cast(from); + } + + template<> EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE float4 ploadu(const float *from) + { + return make_float4(from[0], from[1], from[2], from[3]); + } + template<> EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE double2 ploadu(const double *from) + { + return make_double2(from[0], from[1]); + } + + template<> EIGEN_STRONG_INLINE float4 ploaddup(const float *from) + { + return make_float4(from[0], from[0], from[1], from[1]); + } + template<> EIGEN_STRONG_INLINE double2 ploaddup(const double *from) + { + return make_double2(from[0], from[0]); + } + + template<> EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void pstore(float *to, const float4 &from) + { + *reinterpret_cast(to) = from; + } + + template<> EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void pstore(double *to, const double2 &from) + { + *reinterpret_cast(to) = from; + } + + template<> EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void pstoreu(float *to, const float4 &from) + { + to[0] = from.x; + to[1] = from.y; + to[2] = from.z; + to[3] = from.w; + } + + template<> EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void pstoreu(double *to, const double2 &from) + { + to[0] = from.x; + to[1] = from.y; + } + + template<> EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE float4 ploadt_ro(const float *from) + { #if defined(__CUDA_ARCH__) && __CUDA_ARCH__ >= 350 - return __ldg((const float4*)from); + return __ldg((const float4 *)from); #else - return make_float4(from[0], from[1], from[2], from[3]); + return make_float4(from[0], from[1], from[2], from[3]); #endif -} -template<> -EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE double2 ploadt_ro(const double* from) { + } + template<> EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE double2 ploadt_ro(const double *from) + { #if defined(__CUDA_ARCH__) && __CUDA_ARCH__ >= 350 - return __ldg((const double2*)from); + return __ldg((const double2 *)from); #else - return make_double2(from[0], from[1]); + return make_double2(from[0], from[1]); #endif -} + } -template<> -EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE float4 ploadt_ro(const float* from) { + template<> EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE float4 ploadt_ro(const float *from) + { #if defined(__CUDA_ARCH__) && __CUDA_ARCH__ >= 350 - return make_float4(__ldg(from+0), __ldg(from+1), __ldg(from+2), __ldg(from+3)); + return make_float4(__ldg(from + 0), __ldg(from + 1), __ldg(from + 2), __ldg(from + 3)); #else - return make_float4(from[0], from[1], from[2], from[3]); + return make_float4(from[0], from[1], from[2], from[3]); #endif -} -template<> -EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE double2 ploadt_ro(const double* from) { + } + template<> EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE double2 ploadt_ro(const double *from) + { #if defined(__CUDA_ARCH__) && __CUDA_ARCH__ >= 350 - return make_double2(__ldg(from+0), __ldg(from+1)); + return make_double2(__ldg(from + 0), __ldg(from + 1)); #else - return make_double2(from[0], from[1]); + return make_double2(from[0], from[1]); #endif -} - -template<> EIGEN_DEVICE_FUNC inline float4 pgather(const float* from, Index stride) { - return make_float4(from[0*stride], from[1*stride], from[2*stride], from[3*stride]); -} - -template<> EIGEN_DEVICE_FUNC inline double2 pgather(const double* from, Index stride) { - return make_double2(from[0*stride], from[1*stride]); -} - -template<> EIGEN_DEVICE_FUNC inline void pscatter(float* to, const float4& from, Index stride) { - to[stride*0] = from.x; - to[stride*1] = from.y; - to[stride*2] = from.z; - to[stride*3] = from.w; -} -template<> EIGEN_DEVICE_FUNC inline void pscatter(double* to, const double2& from, Index stride) { - to[stride*0] = from.x; - to[stride*1] = from.y; -} - -template<> EIGEN_DEVICE_FUNC inline float pfirst(const float4& a) { - return a.x; -} -template<> EIGEN_DEVICE_FUNC inline double pfirst(const double2& a) { - return a.x; -} - -template<> EIGEN_DEVICE_FUNC inline float predux(const float4& a) { - return a.x + a.y + a.z + a.w; -} -template<> EIGEN_DEVICE_FUNC inline double predux(const double2& a) { - return a.x + a.y; -} - -template<> EIGEN_DEVICE_FUNC inline float predux_max(const float4& a) { - return fmaxf(fmaxf(a.x, a.y), fmaxf(a.z, a.w)); -} -template<> EIGEN_DEVICE_FUNC inline double predux_max(const double2& a) { - return fmax(a.x, a.y); -} - -template<> EIGEN_DEVICE_FUNC inline float predux_min(const float4& a) { - return fminf(fminf(a.x, a.y), fminf(a.z, a.w)); -} -template<> EIGEN_DEVICE_FUNC inline double predux_min(const double2& a) { - return fmin(a.x, a.y); -} - -template<> EIGEN_DEVICE_FUNC inline float predux_mul(const float4& a) { - return a.x * a.y * a.z * a.w; -} -template<> EIGEN_DEVICE_FUNC inline double predux_mul(const double2& a) { - return a.x * a.y; -} - -template<> EIGEN_DEVICE_FUNC inline float4 pabs(const float4& a) { - return make_float4(fabsf(a.x), fabsf(a.y), fabsf(a.z), fabsf(a.w)); -} -template<> EIGEN_DEVICE_FUNC inline double2 pabs(const double2& a) { - return make_double2(fabs(a.x), fabs(a.y)); -} - -EIGEN_DEVICE_FUNC inline void -ptranspose(PacketBlock& kernel) { - float tmp = kernel.packet[0].y; - kernel.packet[0].y = kernel.packet[1].x; - kernel.packet[1].x = tmp; - - tmp = kernel.packet[0].z; - kernel.packet[0].z = kernel.packet[2].x; - kernel.packet[2].x = tmp; - - tmp = kernel.packet[0].w; - kernel.packet[0].w = kernel.packet[3].x; - kernel.packet[3].x = tmp; - - tmp = kernel.packet[1].z; - kernel.packet[1].z = kernel.packet[2].y; - kernel.packet[2].y = tmp; - - tmp = kernel.packet[1].w; - kernel.packet[1].w = kernel.packet[3].y; - kernel.packet[3].y = tmp; - - tmp = kernel.packet[2].w; - kernel.packet[2].w = kernel.packet[3].z; - kernel.packet[3].z = tmp; -} - -EIGEN_DEVICE_FUNC inline void -ptranspose(PacketBlock& kernel) { - double tmp = kernel.packet[0].y; - kernel.packet[0].y = kernel.packet[1].x; - kernel.packet[1].x = tmp; -} + } + + template<> EIGEN_DEVICE_FUNC inline float4 pgather(const float *from, Index stride) + { + return make_float4(from[0 * stride], from[1 * stride], from[2 * stride], from[3 * stride]); + } + + template<> EIGEN_DEVICE_FUNC inline double2 pgather(const double *from, Index stride) + { + return make_double2(from[0 * stride], from[1 * stride]); + } + + template<> EIGEN_DEVICE_FUNC inline void pscatter(float *to, const float4 &from, Index stride) + { + to[stride * 0] = from.x; + to[stride * 1] = from.y; + to[stride * 2] = from.z; + to[stride * 3] = from.w; + } + template<> EIGEN_DEVICE_FUNC inline void pscatter(double *to, const double2 &from, Index stride) + { + to[stride * 0] = from.x; + to[stride * 1] = from.y; + } + + template<> EIGEN_DEVICE_FUNC inline float pfirst(const float4 &a) { return a.x; } + template<> EIGEN_DEVICE_FUNC inline double pfirst(const double2 &a) { return a.x; } + + template<> EIGEN_DEVICE_FUNC inline float predux(const float4 &a) { return a.x + a.y + a.z + a.w; } + template<> EIGEN_DEVICE_FUNC inline double predux(const double2 &a) { return a.x + a.y; } + + template<> EIGEN_DEVICE_FUNC inline float predux_max(const float4 &a) + { + return fmaxf(fmaxf(a.x, a.y), fmaxf(a.z, a.w)); + } + template<> EIGEN_DEVICE_FUNC inline double predux_max(const double2 &a) { return fmax(a.x, a.y); } + + template<> EIGEN_DEVICE_FUNC inline float predux_min(const float4 &a) + { + return fminf(fminf(a.x, a.y), fminf(a.z, a.w)); + } + template<> EIGEN_DEVICE_FUNC inline double predux_min(const double2 &a) { return fmin(a.x, a.y); } + + template<> EIGEN_DEVICE_FUNC inline float predux_mul(const float4 &a) { return a.x * a.y * a.z * a.w; } + template<> EIGEN_DEVICE_FUNC inline double predux_mul(const double2 &a) { return a.x * a.y; } + + template<> EIGEN_DEVICE_FUNC inline float4 pabs(const float4 &a) + { + return make_float4(fabsf(a.x), fabsf(a.y), fabsf(a.z), fabsf(a.w)); + } + template<> EIGEN_DEVICE_FUNC inline double2 pabs(const double2 &a) + { + return make_double2(fabs(a.x), fabs(a.y)); + } + + EIGEN_DEVICE_FUNC inline void ptranspose(PacketBlock &kernel) + { + float tmp = kernel.packet[0].y; + kernel.packet[0].y = kernel.packet[1].x; + kernel.packet[1].x = tmp; + + tmp = kernel.packet[0].z; + kernel.packet[0].z = kernel.packet[2].x; + kernel.packet[2].x = tmp; + + tmp = kernel.packet[0].w; + kernel.packet[0].w = kernel.packet[3].x; + kernel.packet[3].x = tmp; + + tmp = kernel.packet[1].z; + kernel.packet[1].z = kernel.packet[2].y; + kernel.packet[2].y = tmp; + + tmp = kernel.packet[1].w; + kernel.packet[1].w = kernel.packet[3].y; + kernel.packet[3].y = tmp; + + tmp = kernel.packet[2].w; + kernel.packet[2].w = kernel.packet[3].z; + kernel.packet[3].z = tmp; + } + + EIGEN_DEVICE_FUNC inline void ptranspose(PacketBlock &kernel) + { + double tmp = kernel.packet[0].y; + kernel.packet[0].y = kernel.packet[1].x; + kernel.packet[1].x = tmp; + } #endif -} // end namespace internal +}// end namespace internal -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_PACKET_MATH_CUDA_H +#endif// EIGEN_PACKET_MATH_CUDA_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/arch/CUDA/PacketMathHalf.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/arch/CUDA/PacketMathHalf.h index c66d3846..1285d12c 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/arch/CUDA/PacketMathHalf.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/arch/CUDA/PacketMathHalf.h @@ -17,1108 +17,1189 @@ namespace internal { // Most of the following operations require arch >= 3.0 #if defined(EIGEN_HAS_CUDA_FP16) && defined(__CUDACC__) && defined(__CUDA_ARCH__) && __CUDA_ARCH__ >= 300 -template<> struct is_arithmetic { enum { value = true }; }; - -template<> struct packet_traits : default_packet_traits -{ - typedef half2 type; - typedef half2 half; - enum { - Vectorizable = 1, - AlignedOnScalar = 1, - size=2, - HasHalfPacket = 0, - HasAdd = 1, - HasMul = 1, - HasDiv = 1, - HasSqrt = 1, - HasRsqrt = 1, - HasExp = 1, - HasLog = 1, - HasLog1p = 1 + template<> struct is_arithmetic + { + enum { value = true }; }; -}; -template<> struct unpacket_traits { typedef Eigen::half type; enum {size=2, alignment=Aligned16}; typedef half2 half; }; + template<> struct packet_traits : default_packet_traits + { + typedef half2 type; + typedef half2 half; + enum { + Vectorizable = 1, + AlignedOnScalar = 1, + size = 2, + HasHalfPacket = 0, + HasAdd = 1, + HasMul = 1, + HasDiv = 1, + HasSqrt = 1, + HasRsqrt = 1, + HasExp = 1, + HasLog = 1, + HasLog1p = 1 + }; + }; -template<> __device__ EIGEN_STRONG_INLINE half2 pset1(const Eigen::half& from) { - return __half2half2(from); -} + template<> struct unpacket_traits + { + typedef Eigen::half type; + enum { size = 2, alignment = Aligned16 }; + typedef half2 half; + }; -template<> __device__ EIGEN_STRONG_INLINE half2 pload(const Eigen::half* from) { - return *reinterpret_cast(from); -} + template<> __device__ EIGEN_STRONG_INLINE half2 pset1(const Eigen::half &from) { return __half2half2(from); } -template<> __device__ EIGEN_STRONG_INLINE half2 ploadu(const Eigen::half* from) { - return __halves2half2(from[0], from[1]); -} + template<> __device__ EIGEN_STRONG_INLINE half2 pload(const Eigen::half *from) + { + return *reinterpret_cast(from); + } -template<> EIGEN_STRONG_INLINE half2 ploaddup(const Eigen::half* from) { - return __halves2half2(from[0], from[0]); -} + template<> __device__ EIGEN_STRONG_INLINE half2 ploadu(const Eigen::half *from) + { + return __halves2half2(from[0], from[1]); + } -template<> __device__ EIGEN_STRONG_INLINE void pstore(Eigen::half* to, const half2& from) { - *reinterpret_cast(to) = from; -} + template<> EIGEN_STRONG_INLINE half2 ploaddup(const Eigen::half *from) + { + return __halves2half2(from[0], from[0]); + } -template<> __device__ EIGEN_STRONG_INLINE void pstoreu(Eigen::half* to, const half2& from) { - to[0] = __low2half(from); - to[1] = __high2half(from); -} + template<> __device__ EIGEN_STRONG_INLINE void pstore(Eigen::half *to, const half2 &from) + { + *reinterpret_cast(to) = from; + } -template<> - __device__ EIGEN_ALWAYS_INLINE half2 ploadt_ro(const Eigen::half* from) { + template<> __device__ EIGEN_STRONG_INLINE void pstoreu(Eigen::half *to, const half2 &from) + { + to[0] = __low2half(from); + to[1] = __high2half(from); + } + + template<> __device__ EIGEN_ALWAYS_INLINE half2 ploadt_ro(const Eigen::half *from) + { #if __CUDA_ARCH__ >= 350 - return __ldg((const half2*)from); + return __ldg((const half2 *)from); #else - return __halves2half2(*(from+0), *(from+1)); + return __halves2half2(*(from + 0), *(from + 1)); #endif -} + } -template<> -__device__ EIGEN_ALWAYS_INLINE half2 ploadt_ro(const Eigen::half* from) { + template<> __device__ EIGEN_ALWAYS_INLINE half2 ploadt_ro(const Eigen::half *from) + { #if __CUDA_ARCH__ >= 350 - return __halves2half2(__ldg(from+0), __ldg(from+1)); + return __halves2half2(__ldg(from + 0), __ldg(from + 1)); #else - return __halves2half2(*(from+0), *(from+1)); + return __halves2half2(*(from + 0), *(from + 1)); #endif -} - -template<> __device__ EIGEN_STRONG_INLINE half2 pgather(const Eigen::half* from, Index stride) { - return __halves2half2(from[0*stride], from[1*stride]); -} - -template<> __device__ EIGEN_STRONG_INLINE void pscatter(Eigen::half* to, const half2& from, Index stride) { - to[stride*0] = __low2half(from); - to[stride*1] = __high2half(from); -} - -template<> __device__ EIGEN_STRONG_INLINE Eigen::half pfirst(const half2& a) { - return __low2half(a); -} - -template<> __device__ EIGEN_STRONG_INLINE half2 pabs(const half2& a) { - half2 result; - unsigned temp = *(reinterpret_cast(&(a))); - *(reinterpret_cast(&(result))) = temp & 0x7FFF7FFF; - return result; -} - - -__device__ EIGEN_STRONG_INLINE void -ptranspose(PacketBlock& kernel) { - __half a1 = __low2half(kernel.packet[0]); - __half a2 = __high2half(kernel.packet[0]); - __half b1 = __low2half(kernel.packet[1]); - __half b2 = __high2half(kernel.packet[1]); - kernel.packet[0] = __halves2half2(a1, b1); - kernel.packet[1] = __halves2half2(a2, b2); -} - -template<> __device__ EIGEN_STRONG_INLINE half2 plset(const Eigen::half& a) { + } + + template<> __device__ EIGEN_STRONG_INLINE half2 pgather(const Eigen::half *from, Index stride) + { + return __halves2half2(from[0 * stride], from[1 * stride]); + } + + template<> + __device__ EIGEN_STRONG_INLINE void pscatter(Eigen::half *to, const half2 &from, Index stride) + { + to[stride * 0] = __low2half(from); + to[stride * 1] = __high2half(from); + } + + template<> __device__ EIGEN_STRONG_INLINE Eigen::half pfirst(const half2 &a) { return __low2half(a); } + + template<> __device__ EIGEN_STRONG_INLINE half2 pabs(const half2 &a) + { + half2 result; + unsigned temp = *(reinterpret_cast(&(a))); + *(reinterpret_cast(&(result))) = temp & 0x7FFF7FFF; + return result; + } + + + __device__ EIGEN_STRONG_INLINE void ptranspose(PacketBlock &kernel) + { + __half a1 = __low2half(kernel.packet[0]); + __half a2 = __high2half(kernel.packet[0]); + __half b1 = __low2half(kernel.packet[1]); + __half b2 = __high2half(kernel.packet[1]); + kernel.packet[0] = __halves2half2(a1, b1); + kernel.packet[1] = __halves2half2(a2, b2); + } + + template<> __device__ EIGEN_STRONG_INLINE half2 plset(const Eigen::half &a) + { #if __CUDA_ARCH__ >= 530 - return __halves2half2(a, __hadd(a, __float2half(1.0f))); + return __halves2half2(a, __hadd(a, __float2half(1.0f))); #else - float f = __half2float(a) + 1.0f; - return __halves2half2(a, __float2half(f)); + float f = __half2float(a) + 1.0f; + return __halves2half2(a, __float2half(f)); #endif -} + } -template<> __device__ EIGEN_STRONG_INLINE half2 padd(const half2& a, const half2& b) { + template<> __device__ EIGEN_STRONG_INLINE half2 padd(const half2 &a, const half2 &b) + { #if __CUDA_ARCH__ >= 530 - return __hadd2(a, b); + return __hadd2(a, b); #else - float a1 = __low2float(a); - float a2 = __high2float(a); - float b1 = __low2float(b); - float b2 = __high2float(b); - float r1 = a1 + b1; - float r2 = a2 + b2; - return __floats2half2_rn(r1, r2); + float a1 = __low2float(a); + float a2 = __high2float(a); + float b1 = __low2float(b); + float b2 = __high2float(b); + float r1 = a1 + b1; + float r2 = a2 + b2; + return __floats2half2_rn(r1, r2); #endif -} + } -template<> __device__ EIGEN_STRONG_INLINE half2 psub(const half2& a, const half2& b) { + template<> __device__ EIGEN_STRONG_INLINE half2 psub(const half2 &a, const half2 &b) + { #if __CUDA_ARCH__ >= 530 - return __hsub2(a, b); + return __hsub2(a, b); #else - float a1 = __low2float(a); - float a2 = __high2float(a); - float b1 = __low2float(b); - float b2 = __high2float(b); - float r1 = a1 - b1; - float r2 = a2 - b2; - return __floats2half2_rn(r1, r2); + float a1 = __low2float(a); + float a2 = __high2float(a); + float b1 = __low2float(b); + float b2 = __high2float(b); + float r1 = a1 - b1; + float r2 = a2 - b2; + return __floats2half2_rn(r1, r2); #endif -} + } -template<> __device__ EIGEN_STRONG_INLINE half2 pnegate(const half2& a) { + template<> __device__ EIGEN_STRONG_INLINE half2 pnegate(const half2 &a) + { #if __CUDA_ARCH__ >= 530 - return __hneg2(a); + return __hneg2(a); #else - float a1 = __low2float(a); - float a2 = __high2float(a); - return __floats2half2_rn(-a1, -a2); + float a1 = __low2float(a); + float a2 = __high2float(a); + return __floats2half2_rn(-a1, -a2); #endif -} + } -template<> __device__ EIGEN_STRONG_INLINE half2 pconj(const half2& a) { return a; } + template<> __device__ EIGEN_STRONG_INLINE half2 pconj(const half2 &a) { return a; } -template<> __device__ EIGEN_STRONG_INLINE half2 pmul(const half2& a, const half2& b) { + template<> __device__ EIGEN_STRONG_INLINE half2 pmul(const half2 &a, const half2 &b) + { #if __CUDA_ARCH__ >= 530 - return __hmul2(a, b); + return __hmul2(a, b); #else - float a1 = __low2float(a); - float a2 = __high2float(a); - float b1 = __low2float(b); - float b2 = __high2float(b); - float r1 = a1 * b1; - float r2 = a2 * b2; - return __floats2half2_rn(r1, r2); + float a1 = __low2float(a); + float a2 = __high2float(a); + float b1 = __low2float(b); + float b2 = __high2float(b); + float r1 = a1 * b1; + float r2 = a2 * b2; + return __floats2half2_rn(r1, r2); #endif -} + } -template<> __device__ EIGEN_STRONG_INLINE half2 pmadd(const half2& a, const half2& b, const half2& c) { + template<> __device__ EIGEN_STRONG_INLINE half2 pmadd(const half2 &a, const half2 &b, const half2 &c) + { #if __CUDA_ARCH__ >= 530 - return __hfma2(a, b, c); + return __hfma2(a, b, c); #else - float a1 = __low2float(a); - float a2 = __high2float(a); - float b1 = __low2float(b); - float b2 = __high2float(b); - float c1 = __low2float(c); - float c2 = __high2float(c); - float r1 = a1 * b1 + c1; - float r2 = a2 * b2 + c2; - return __floats2half2_rn(r1, r2); + float a1 = __low2float(a); + float a2 = __high2float(a); + float b1 = __low2float(b); + float b2 = __high2float(b); + float c1 = __low2float(c); + float c2 = __high2float(c); + float r1 = a1 * b1 + c1; + float r2 = a2 * b2 + c2; + return __floats2half2_rn(r1, r2); #endif -} - -template<> __device__ EIGEN_STRONG_INLINE half2 pdiv(const half2& a, const half2& b) { - float a1 = __low2float(a); - float a2 = __high2float(a); - float b1 = __low2float(b); - float b2 = __high2float(b); - float r1 = a1 / b1; - float r2 = a2 / b2; - return __floats2half2_rn(r1, r2); -} - -template<> __device__ EIGEN_STRONG_INLINE half2 pmin(const half2& a, const half2& b) { - float a1 = __low2float(a); - float a2 = __high2float(a); - float b1 = __low2float(b); - float b2 = __high2float(b); - __half r1 = a1 < b1 ? __low2half(a) : __low2half(b); - __half r2 = a2 < b2 ? __high2half(a) : __high2half(b); - return __halves2half2(r1, r2); -} - -template<> __device__ EIGEN_STRONG_INLINE half2 pmax(const half2& a, const half2& b) { - float a1 = __low2float(a); - float a2 = __high2float(a); - float b1 = __low2float(b); - float b2 = __high2float(b); - __half r1 = a1 > b1 ? __low2half(a) : __low2half(b); - __half r2 = a2 > b2 ? __high2half(a) : __high2half(b); - return __halves2half2(r1, r2); -} - -template<> __device__ EIGEN_STRONG_INLINE Eigen::half predux(const half2& a) { + } + + template<> __device__ EIGEN_STRONG_INLINE half2 pdiv(const half2 &a, const half2 &b) + { + float a1 = __low2float(a); + float a2 = __high2float(a); + float b1 = __low2float(b); + float b2 = __high2float(b); + float r1 = a1 / b1; + float r2 = a2 / b2; + return __floats2half2_rn(r1, r2); + } + + template<> __device__ EIGEN_STRONG_INLINE half2 pmin(const half2 &a, const half2 &b) + { + float a1 = __low2float(a); + float a2 = __high2float(a); + float b1 = __low2float(b); + float b2 = __high2float(b); + __half r1 = a1 < b1 ? __low2half(a) : __low2half(b); + __half r2 = a2 < b2 ? __high2half(a) : __high2half(b); + return __halves2half2(r1, r2); + } + + template<> __device__ EIGEN_STRONG_INLINE half2 pmax(const half2 &a, const half2 &b) + { + float a1 = __low2float(a); + float a2 = __high2float(a); + float b1 = __low2float(b); + float b2 = __high2float(b); + __half r1 = a1 > b1 ? __low2half(a) : __low2half(b); + __half r2 = a2 > b2 ? __high2half(a) : __high2half(b); + return __halves2half2(r1, r2); + } + + template<> __device__ EIGEN_STRONG_INLINE Eigen::half predux(const half2 &a) + { #if __CUDA_ARCH__ >= 530 - return __hadd(__low2half(a), __high2half(a)); + return __hadd(__low2half(a), __high2half(a)); #else - float a1 = __low2float(a); - float a2 = __high2float(a); - return Eigen::half(half_impl::raw_uint16_to_half(__float2half_rn(a1 + a2))); + float a1 = __low2float(a); + float a2 = __high2float(a); + return Eigen::half(half_impl::raw_uint16_to_half(__float2half_rn(a1 + a2))); #endif -} + } -template<> __device__ EIGEN_STRONG_INLINE Eigen::half predux_max(const half2& a) { + template<> __device__ EIGEN_STRONG_INLINE Eigen::half predux_max(const half2 &a) + { #if __CUDA_ARCH__ >= 530 - __half first = __low2half(a); - __half second = __high2half(a); - return __hgt(first, second) ? first : second; + __half first = __low2half(a); + __half second = __high2half(a); + return __hgt(first, second) ? first : second; #else - float a1 = __low2float(a); - float a2 = __high2float(a); - return a1 > a2 ? __low2half(a) : __high2half(a); + float a1 = __low2float(a); + float a2 = __high2float(a); + return a1 > a2 ? __low2half(a) : __high2half(a); #endif -} + } -template<> __device__ EIGEN_STRONG_INLINE Eigen::half predux_min(const half2& a) { + template<> __device__ EIGEN_STRONG_INLINE Eigen::half predux_min(const half2 &a) + { #if __CUDA_ARCH__ >= 530 - __half first = __low2half(a); - __half second = __high2half(a); - return __hlt(first, second) ? first : second; + __half first = __low2half(a); + __half second = __high2half(a); + return __hlt(first, second) ? first : second; #else - float a1 = __low2float(a); - float a2 = __high2float(a); - return a1 < a2 ? __low2half(a) : __high2half(a); + float a1 = __low2float(a); + float a2 = __high2float(a); + return a1 < a2 ? __low2half(a) : __high2half(a); #endif -} + } -template<> __device__ EIGEN_STRONG_INLINE Eigen::half predux_mul(const half2& a) { + template<> __device__ EIGEN_STRONG_INLINE Eigen::half predux_mul(const half2 &a) + { #if __CUDA_ARCH__ >= 530 - return __hmul(__low2half(a), __high2half(a)); + return __hmul(__low2half(a), __high2half(a)); #else - float a1 = __low2float(a); - float a2 = __high2float(a); - return Eigen::half(half_impl::raw_uint16_to_half(__float2half_rn(a1 * a2))); + float a1 = __low2float(a); + float a2 = __high2float(a); + return Eigen::half(half_impl::raw_uint16_to_half(__float2half_rn(a1 * a2))); #endif -} + } -template<> __device__ EIGEN_STRONG_INLINE half2 plog1p(const half2& a) { - float a1 = __low2float(a); - float a2 = __high2float(a); - float r1 = log1pf(a1); - float r2 = log1pf(a2); - return __floats2half2_rn(r1, r2); -} + template<> __device__ EIGEN_STRONG_INLINE half2 plog1p(const half2 &a) + { + float a1 = __low2float(a); + float a2 = __high2float(a); + float r1 = log1pf(a1); + float r2 = log1pf(a2); + return __floats2half2_rn(r1, r2); + } #if EIGEN_CUDACC_VER >= 80000 && defined EIGEN_CUDA_ARCH && EIGEN_CUDA_ARCH >= 530 -template<> __device__ EIGEN_STRONG_INLINE -half2 plog(const half2& a) { - return h2log(a); -} + template<> __device__ EIGEN_STRONG_INLINE half2 plog(const half2 &a) { return h2log(a); } -template<> __device__ EIGEN_STRONG_INLINE -half2 pexp(const half2& a) { - return h2exp(a); -} + template<> __device__ EIGEN_STRONG_INLINE half2 pexp(const half2 &a) { return h2exp(a); } -template<> __device__ EIGEN_STRONG_INLINE -half2 psqrt(const half2& a) { - return h2sqrt(a); -} + template<> __device__ EIGEN_STRONG_INLINE half2 psqrt(const half2 &a) { return h2sqrt(a); } -template<> __device__ EIGEN_STRONG_INLINE -half2 prsqrt(const half2& a) { - return h2rsqrt(a); -} + template<> __device__ EIGEN_STRONG_INLINE half2 prsqrt(const half2 &a) { return h2rsqrt(a); } #else -template<> __device__ EIGEN_STRONG_INLINE half2 plog(const half2& a) { - float a1 = __low2float(a); - float a2 = __high2float(a); - float r1 = logf(a1); - float r2 = logf(a2); - return __floats2half2_rn(r1, r2); -} - -template<> __device__ EIGEN_STRONG_INLINE half2 pexp(const half2& a) { - float a1 = __low2float(a); - float a2 = __high2float(a); - float r1 = expf(a1); - float r2 = expf(a2); - return __floats2half2_rn(r1, r2); -} - -template<> __device__ EIGEN_STRONG_INLINE half2 psqrt(const half2& a) { - float a1 = __low2float(a); - float a2 = __high2float(a); - float r1 = sqrtf(a1); - float r2 = sqrtf(a2); - return __floats2half2_rn(r1, r2); -} - -template<> __device__ EIGEN_STRONG_INLINE half2 prsqrt(const half2& a) { - float a1 = __low2float(a); - float a2 = __high2float(a); - float r1 = rsqrtf(a1); - float r2 = rsqrtf(a2); - return __floats2half2_rn(r1, r2); -} + template<> __device__ EIGEN_STRONG_INLINE half2 plog(const half2 &a) + { + float a1 = __low2float(a); + float a2 = __high2float(a); + float r1 = logf(a1); + float r2 = logf(a2); + return __floats2half2_rn(r1, r2); + } + + template<> __device__ EIGEN_STRONG_INLINE half2 pexp(const half2 &a) + { + float a1 = __low2float(a); + float a2 = __high2float(a); + float r1 = expf(a1); + float r2 = expf(a2); + return __floats2half2_rn(r1, r2); + } + + template<> __device__ EIGEN_STRONG_INLINE half2 psqrt(const half2 &a) + { + float a1 = __low2float(a); + float a2 = __high2float(a); + float r1 = sqrtf(a1); + float r2 = sqrtf(a2); + return __floats2half2_rn(r1, r2); + } + + template<> __device__ EIGEN_STRONG_INLINE half2 prsqrt(const half2 &a) + { + float a1 = __low2float(a); + float a2 = __high2float(a); + float r1 = rsqrtf(a1); + float r2 = rsqrtf(a2); + return __floats2half2_rn(r1, r2); + } #endif #elif defined EIGEN_VECTORIZE_AVX512 -typedef struct { - __m256i x; -} Packet16h; - - -template<> struct is_arithmetic { enum { value = true }; }; - -template <> -struct packet_traits : default_packet_traits { - typedef Packet16h type; - // There is no half-size packet for Packet16h. - typedef Packet16h half; - enum { - Vectorizable = 1, - AlignedOnScalar = 1, - size = 16, - HasHalfPacket = 0, - HasAdd = 0, - HasSub = 0, - HasMul = 0, - HasNegate = 0, - HasAbs = 0, - HasAbs2 = 0, - HasMin = 0, - HasMax = 0, - HasConj = 0, - HasSetLinear = 0, - HasDiv = 0, - HasSqrt = 0, - HasRsqrt = 0, - HasExp = 0, - HasLog = 0, - HasBlend = 0 + typedef struct + { + __m256i x; + } Packet16h; + + + template<> struct is_arithmetic + { + enum { value = true }; }; -}; - - -template<> struct unpacket_traits { typedef Eigen::half type; enum {size=16, alignment=Aligned32}; typedef Packet16h half; }; - -template<> EIGEN_STRONG_INLINE Packet16h pset1(const Eigen::half& from) { - Packet16h result; - result.x = _mm256_set1_epi16(from.x); - return result; -} - -template<> EIGEN_STRONG_INLINE Eigen::half pfirst(const Packet16h& from) { - return half_impl::raw_uint16_to_half(static_cast(_mm256_extract_epi16(from.x, 0))); -} - -template<> EIGEN_STRONG_INLINE Packet16h pload(const Eigen::half* from) { - Packet16h result; - result.x = _mm256_load_si256(reinterpret_cast(from)); - return result; -} - -template<> EIGEN_STRONG_INLINE Packet16h ploadu(const Eigen::half* from) { - Packet16h result; - result.x = _mm256_loadu_si256(reinterpret_cast(from)); - return result; -} - -template<> EIGEN_STRONG_INLINE void pstore(Eigen::half* to, const Packet16h& from) { - _mm256_store_si256((__m256i*)to, from.x); -} - -template<> EIGEN_STRONG_INLINE void pstoreu(Eigen::half* to, const Packet16h& from) { - _mm256_storeu_si256((__m256i*)to, from.x); -} - -template<> EIGEN_STRONG_INLINE Packet16h -ploadquad(const Eigen::half* from) { - Packet16h result; - unsigned short a = from[0].x; - unsigned short b = from[1].x; - unsigned short c = from[2].x; - unsigned short d = from[3].x; - result.x = _mm256_set_epi16(d, d, d, d, c, c, c, c, b, b, b, b, a, a, a, a); - return result; -} - -EIGEN_STRONG_INLINE Packet16f half2float(const Packet16h& a) { + + template<> struct packet_traits : default_packet_traits + { + typedef Packet16h type; + // There is no half-size packet for Packet16h. + typedef Packet16h half; + enum { + Vectorizable = 1, + AlignedOnScalar = 1, + size = 16, + HasHalfPacket = 0, + HasAdd = 0, + HasSub = 0, + HasMul = 0, + HasNegate = 0, + HasAbs = 0, + HasAbs2 = 0, + HasMin = 0, + HasMax = 0, + HasConj = 0, + HasSetLinear = 0, + HasDiv = 0, + HasSqrt = 0, + HasRsqrt = 0, + HasExp = 0, + HasLog = 0, + HasBlend = 0 + }; + }; + + + template<> struct unpacket_traits + { + typedef Eigen::half type; + enum { size = 16, alignment = Aligned32 }; + typedef Packet16h half; + }; + + template<> EIGEN_STRONG_INLINE Packet16h pset1(const Eigen::half &from) + { + Packet16h result; + result.x = _mm256_set1_epi16(from.x); + return result; + } + + template<> EIGEN_STRONG_INLINE Eigen::half pfirst(const Packet16h &from) + { + return half_impl::raw_uint16_to_half(static_cast(_mm256_extract_epi16(from.x, 0))); + } + + template<> EIGEN_STRONG_INLINE Packet16h pload(const Eigen::half *from) + { + Packet16h result; + result.x = _mm256_load_si256(reinterpret_cast(from)); + return result; + } + + template<> EIGEN_STRONG_INLINE Packet16h ploadu(const Eigen::half *from) + { + Packet16h result; + result.x = _mm256_loadu_si256(reinterpret_cast(from)); + return result; + } + + template<> EIGEN_STRONG_INLINE void pstore(Eigen::half *to, const Packet16h &from) + { + _mm256_store_si256((__m256i *)to, from.x); + } + + template<> EIGEN_STRONG_INLINE void pstoreu(Eigen::half *to, const Packet16h &from) + { + _mm256_storeu_si256((__m256i *)to, from.x); + } + + template<> EIGEN_STRONG_INLINE Packet16h ploadquad(const Eigen::half *from) + { + Packet16h result; + unsigned short a = from[0].x; + unsigned short b = from[1].x; + unsigned short c = from[2].x; + unsigned short d = from[3].x; + result.x = _mm256_set_epi16(d, d, d, d, c, c, c, c, b, b, b, b, a, a, a, a); + return result; + } + + EIGEN_STRONG_INLINE Packet16f half2float(const Packet16h &a) + { #ifdef EIGEN_HAS_FP16_C - return _mm512_cvtph_ps(a.x); + return _mm512_cvtph_ps(a.x); #else - EIGEN_ALIGN64 half aux[16]; - pstore(aux, a); - float f0(aux[0]); - float f1(aux[1]); - float f2(aux[2]); - float f3(aux[3]); - float f4(aux[4]); - float f5(aux[5]); - float f6(aux[6]); - float f7(aux[7]); - float f8(aux[8]); - float f9(aux[9]); - float fa(aux[10]); - float fb(aux[11]); - float fc(aux[12]); - float fd(aux[13]); - float fe(aux[14]); - float ff(aux[15]); - - return _mm512_set_ps( - ff, fe, fd, fc, fb, fa, f9, f8, f7, f6, f5, f4, f3, f2, f1, f0); + EIGEN_ALIGN64 half aux[16]; + pstore(aux, a); + float f0(aux[0]); + float f1(aux[1]); + float f2(aux[2]); + float f3(aux[3]); + float f4(aux[4]); + float f5(aux[5]); + float f6(aux[6]); + float f7(aux[7]); + float f8(aux[8]); + float f9(aux[9]); + float fa(aux[10]); + float fb(aux[11]); + float fc(aux[12]); + float fd(aux[13]); + float fe(aux[14]); + float ff(aux[15]); + + return _mm512_set_ps(ff, fe, fd, fc, fb, fa, f9, f8, f7, f6, f5, f4, f3, f2, f1, f0); #endif -} + } -EIGEN_STRONG_INLINE Packet16h float2half(const Packet16f& a) { + EIGEN_STRONG_INLINE Packet16h float2half(const Packet16f &a) + { #ifdef EIGEN_HAS_FP16_C - Packet16h result; - result.x = _mm512_cvtps_ph(a, _MM_FROUND_TO_NEAREST_INT|_MM_FROUND_NO_EXC); - return result; + Packet16h result; + result.x = _mm512_cvtps_ph(a, _MM_FROUND_TO_NEAREST_INT | _MM_FROUND_NO_EXC); + return result; #else - EIGEN_ALIGN64 float aux[16]; - pstore(aux, a); - half h0(aux[0]); - half h1(aux[1]); - half h2(aux[2]); - half h3(aux[3]); - half h4(aux[4]); - half h5(aux[5]); - half h6(aux[6]); - half h7(aux[7]); - half h8(aux[8]); - half h9(aux[9]); - half ha(aux[10]); - half hb(aux[11]); - half hc(aux[12]); - half hd(aux[13]); - half he(aux[14]); - half hf(aux[15]); - - Packet16h result; - result.x = _mm256_set_epi16( - hf.x, he.x, hd.x, hc.x, hb.x, ha.x, h9.x, h8.x, - h7.x, h6.x, h5.x, h4.x, h3.x, h2.x, h1.x, h0.x); - return result; + EIGEN_ALIGN64 float aux[16]; + pstore(aux, a); + half h0(aux[0]); + half h1(aux[1]); + half h2(aux[2]); + half h3(aux[3]); + half h4(aux[4]); + half h5(aux[5]); + half h6(aux[6]); + half h7(aux[7]); + half h8(aux[8]); + half h9(aux[9]); + half ha(aux[10]); + half hb(aux[11]); + half hc(aux[12]); + half hd(aux[13]); + half he(aux[14]); + half hf(aux[15]); + + Packet16h result; + result.x = + _mm256_set_epi16(hf.x, he.x, hd.x, hc.x, hb.x, ha.x, h9.x, h8.x, h7.x, h6.x, h5.x, h4.x, h3.x, h2.x, h1.x, h0.x); + return result; #endif -} - -template<> EIGEN_STRONG_INLINE Packet16h padd(const Packet16h& a, const Packet16h& b) { - Packet16f af = half2float(a); - Packet16f bf = half2float(b); - Packet16f rf = padd(af, bf); - return float2half(rf); -} - -template<> EIGEN_STRONG_INLINE Packet16h pmul(const Packet16h& a, const Packet16h& b) { - Packet16f af = half2float(a); - Packet16f bf = half2float(b); - Packet16f rf = pmul(af, bf); - return float2half(rf); -} - -template<> EIGEN_STRONG_INLINE half predux(const Packet16h& from) { - Packet16f from_float = half2float(from); - return half(predux(from_float)); -} - -template<> EIGEN_STRONG_INLINE Packet16h pgather(const Eigen::half* from, Index stride) -{ - Packet16h result; - result.x = _mm256_set_epi16( - from[15*stride].x, from[14*stride].x, from[13*stride].x, from[12*stride].x, - from[11*stride].x, from[10*stride].x, from[9*stride].x, from[8*stride].x, - from[7*stride].x, from[6*stride].x, from[5*stride].x, from[4*stride].x, - from[3*stride].x, from[2*stride].x, from[1*stride].x, from[0*stride].x); - return result; -} - -template<> EIGEN_STRONG_INLINE void pscatter(half* to, const Packet16h& from, Index stride) -{ - EIGEN_ALIGN64 half aux[16]; - pstore(aux, from); - to[stride*0].x = aux[0].x; - to[stride*1].x = aux[1].x; - to[stride*2].x = aux[2].x; - to[stride*3].x = aux[3].x; - to[stride*4].x = aux[4].x; - to[stride*5].x = aux[5].x; - to[stride*6].x = aux[6].x; - to[stride*7].x = aux[7].x; - to[stride*8].x = aux[8].x; - to[stride*9].x = aux[9].x; - to[stride*10].x = aux[10].x; - to[stride*11].x = aux[11].x; - to[stride*12].x = aux[12].x; - to[stride*13].x = aux[13].x; - to[stride*14].x = aux[14].x; - to[stride*15].x = aux[15].x; -} - -EIGEN_STRONG_INLINE void -ptranspose(PacketBlock& kernel) { - __m256i a = kernel.packet[0].x; - __m256i b = kernel.packet[1].x; - __m256i c = kernel.packet[2].x; - __m256i d = kernel.packet[3].x; - __m256i e = kernel.packet[4].x; - __m256i f = kernel.packet[5].x; - __m256i g = kernel.packet[6].x; - __m256i h = kernel.packet[7].x; - __m256i i = kernel.packet[8].x; - __m256i j = kernel.packet[9].x; - __m256i k = kernel.packet[10].x; - __m256i l = kernel.packet[11].x; - __m256i m = kernel.packet[12].x; - __m256i n = kernel.packet[13].x; - __m256i o = kernel.packet[14].x; - __m256i p = kernel.packet[15].x; - - __m256i ab_07 = _mm256_unpacklo_epi16(a, b); - __m256i cd_07 = _mm256_unpacklo_epi16(c, d); - __m256i ef_07 = _mm256_unpacklo_epi16(e, f); - __m256i gh_07 = _mm256_unpacklo_epi16(g, h); - __m256i ij_07 = _mm256_unpacklo_epi16(i, j); - __m256i kl_07 = _mm256_unpacklo_epi16(k, l); - __m256i mn_07 = _mm256_unpacklo_epi16(m, n); - __m256i op_07 = _mm256_unpacklo_epi16(o, p); - - __m256i ab_8f = _mm256_unpackhi_epi16(a, b); - __m256i cd_8f = _mm256_unpackhi_epi16(c, d); - __m256i ef_8f = _mm256_unpackhi_epi16(e, f); - __m256i gh_8f = _mm256_unpackhi_epi16(g, h); - __m256i ij_8f = _mm256_unpackhi_epi16(i, j); - __m256i kl_8f = _mm256_unpackhi_epi16(k, l); - __m256i mn_8f = _mm256_unpackhi_epi16(m, n); - __m256i op_8f = _mm256_unpackhi_epi16(o, p); - - __m256i abcd_03 = _mm256_unpacklo_epi32(ab_07, cd_07); - __m256i abcd_47 = _mm256_unpackhi_epi32(ab_07, cd_07); - __m256i efgh_03 = _mm256_unpacklo_epi32(ef_07, gh_07); - __m256i efgh_47 = _mm256_unpackhi_epi32(ef_07, gh_07); - __m256i ijkl_03 = _mm256_unpacklo_epi32(ij_07, kl_07); - __m256i ijkl_47 = _mm256_unpackhi_epi32(ij_07, kl_07); - __m256i mnop_03 = _mm256_unpacklo_epi32(mn_07, op_07); - __m256i mnop_47 = _mm256_unpackhi_epi32(mn_07, op_07); - - __m256i abcd_8b = _mm256_unpacklo_epi32(ab_8f, cd_8f); - __m256i abcd_cf = _mm256_unpackhi_epi32(ab_8f, cd_8f); - __m256i efgh_8b = _mm256_unpacklo_epi32(ef_8f, gh_8f); - __m256i efgh_cf = _mm256_unpackhi_epi32(ef_8f, gh_8f); - __m256i ijkl_8b = _mm256_unpacklo_epi32(ij_8f, kl_8f); - __m256i ijkl_cf = _mm256_unpackhi_epi32(ij_8f, kl_8f); - __m256i mnop_8b = _mm256_unpacklo_epi32(mn_8f, op_8f); - __m256i mnop_cf = _mm256_unpackhi_epi32(mn_8f, op_8f); - - __m256i abcdefgh_01 = _mm256_unpacklo_epi64(abcd_03, efgh_03); - __m256i abcdefgh_23 = _mm256_unpackhi_epi64(abcd_03, efgh_03); - __m256i ijklmnop_01 = _mm256_unpacklo_epi64(ijkl_03, mnop_03); - __m256i ijklmnop_23 = _mm256_unpackhi_epi64(ijkl_03, mnop_03); - __m256i abcdefgh_45 = _mm256_unpacklo_epi64(abcd_47, efgh_47); - __m256i abcdefgh_67 = _mm256_unpackhi_epi64(abcd_47, efgh_47); - __m256i ijklmnop_45 = _mm256_unpacklo_epi64(ijkl_47, mnop_47); - __m256i ijklmnop_67 = _mm256_unpackhi_epi64(ijkl_47, mnop_47); - __m256i abcdefgh_89 = _mm256_unpacklo_epi64(abcd_8b, efgh_8b); - __m256i abcdefgh_ab = _mm256_unpackhi_epi64(abcd_8b, efgh_8b); - __m256i ijklmnop_89 = _mm256_unpacklo_epi64(ijkl_8b, mnop_8b); - __m256i ijklmnop_ab = _mm256_unpackhi_epi64(ijkl_8b, mnop_8b); - __m256i abcdefgh_cd = _mm256_unpacklo_epi64(abcd_cf, efgh_cf); - __m256i abcdefgh_ef = _mm256_unpackhi_epi64(abcd_cf, efgh_cf); - __m256i ijklmnop_cd = _mm256_unpacklo_epi64(ijkl_cf, mnop_cf); - __m256i ijklmnop_ef = _mm256_unpackhi_epi64(ijkl_cf, mnop_cf); - - // NOTE: no unpacklo/hi instr in this case, so using permute instr. - __m256i a_p_0 = _mm256_permute2x128_si256(abcdefgh_01, ijklmnop_01, 0x20); - __m256i a_p_1 = _mm256_permute2x128_si256(abcdefgh_01, ijklmnop_01, 0x31); - __m256i a_p_2 = _mm256_permute2x128_si256(abcdefgh_23, ijklmnop_23, 0x20); - __m256i a_p_3 = _mm256_permute2x128_si256(abcdefgh_23, ijklmnop_23, 0x31); - __m256i a_p_4 = _mm256_permute2x128_si256(abcdefgh_45, ijklmnop_45, 0x20); - __m256i a_p_5 = _mm256_permute2x128_si256(abcdefgh_45, ijklmnop_45, 0x31); - __m256i a_p_6 = _mm256_permute2x128_si256(abcdefgh_67, ijklmnop_67, 0x20); - __m256i a_p_7 = _mm256_permute2x128_si256(abcdefgh_67, ijklmnop_67, 0x31); - __m256i a_p_8 = _mm256_permute2x128_si256(abcdefgh_89, ijklmnop_89, 0x20); - __m256i a_p_9 = _mm256_permute2x128_si256(abcdefgh_89, ijklmnop_89, 0x31); - __m256i a_p_a = _mm256_permute2x128_si256(abcdefgh_ab, ijklmnop_ab, 0x20); - __m256i a_p_b = _mm256_permute2x128_si256(abcdefgh_ab, ijklmnop_ab, 0x31); - __m256i a_p_c = _mm256_permute2x128_si256(abcdefgh_cd, ijklmnop_cd, 0x20); - __m256i a_p_d = _mm256_permute2x128_si256(abcdefgh_cd, ijklmnop_cd, 0x31); - __m256i a_p_e = _mm256_permute2x128_si256(abcdefgh_ef, ijklmnop_ef, 0x20); - __m256i a_p_f = _mm256_permute2x128_si256(abcdefgh_ef, ijklmnop_ef, 0x31); - - kernel.packet[0].x = a_p_0; - kernel.packet[1].x = a_p_1; - kernel.packet[2].x = a_p_2; - kernel.packet[3].x = a_p_3; - kernel.packet[4].x = a_p_4; - kernel.packet[5].x = a_p_5; - kernel.packet[6].x = a_p_6; - kernel.packet[7].x = a_p_7; - kernel.packet[8].x = a_p_8; - kernel.packet[9].x = a_p_9; - kernel.packet[10].x = a_p_a; - kernel.packet[11].x = a_p_b; - kernel.packet[12].x = a_p_c; - kernel.packet[13].x = a_p_d; - kernel.packet[14].x = a_p_e; - kernel.packet[15].x = a_p_f; -} - -EIGEN_STRONG_INLINE void -ptranspose(PacketBlock& kernel) { - EIGEN_ALIGN64 half in[8][16]; - pstore(in[0], kernel.packet[0]); - pstore(in[1], kernel.packet[1]); - pstore(in[2], kernel.packet[2]); - pstore(in[3], kernel.packet[3]); - pstore(in[4], kernel.packet[4]); - pstore(in[5], kernel.packet[5]); - pstore(in[6], kernel.packet[6]); - pstore(in[7], kernel.packet[7]); - - EIGEN_ALIGN64 half out[8][16]; - - for (int i = 0; i < 8; ++i) { - for (int j = 0; j < 8; ++j) { - out[i][j] = in[j][2*i]; - } - for (int j = 0; j < 8; ++j) { - out[i][j+8] = in[j][2*i+1]; - } } - kernel.packet[0] = pload(out[0]); - kernel.packet[1] = pload(out[1]); - kernel.packet[2] = pload(out[2]); - kernel.packet[3] = pload(out[3]); - kernel.packet[4] = pload(out[4]); - kernel.packet[5] = pload(out[5]); - kernel.packet[6] = pload(out[6]); - kernel.packet[7] = pload(out[7]); -} - -EIGEN_STRONG_INLINE void -ptranspose(PacketBlock& kernel) { - EIGEN_ALIGN64 half in[4][16]; - pstore(in[0], kernel.packet[0]); - pstore(in[1], kernel.packet[1]); - pstore(in[2], kernel.packet[2]); - pstore(in[3], kernel.packet[3]); - - EIGEN_ALIGN64 half out[4][16]; - - for (int i = 0; i < 4; ++i) { - for (int j = 0; j < 4; ++j) { - out[i][j] = in[j][4*i]; - } - for (int j = 0; j < 4; ++j) { - out[i][j+4] = in[j][4*i+1]; - } - for (int j = 0; j < 4; ++j) { - out[i][j+8] = in[j][4*i+2]; - } - for (int j = 0; j < 4; ++j) { - out[i][j+12] = in[j][4*i+3]; + template<> EIGEN_STRONG_INLINE Packet16h padd(const Packet16h &a, const Packet16h &b) + { + Packet16f af = half2float(a); + Packet16f bf = half2float(b); + Packet16f rf = padd(af, bf); + return float2half(rf); + } + + template<> EIGEN_STRONG_INLINE Packet16h pmul(const Packet16h &a, const Packet16h &b) + { + Packet16f af = half2float(a); + Packet16f bf = half2float(b); + Packet16f rf = pmul(af, bf); + return float2half(rf); + } + + template<> EIGEN_STRONG_INLINE half predux(const Packet16h &from) + { + Packet16f from_float = half2float(from); + return half(predux(from_float)); + } + + template<> EIGEN_STRONG_INLINE Packet16h pgather(const Eigen::half *from, Index stride) + { + Packet16h result; + result.x = _mm256_set_epi16(from[15 * stride].x, + from[14 * stride].x, + from[13 * stride].x, + from[12 * stride].x, + from[11 * stride].x, + from[10 * stride].x, + from[9 * stride].x, + from[8 * stride].x, + from[7 * stride].x, + from[6 * stride].x, + from[5 * stride].x, + from[4 * stride].x, + from[3 * stride].x, + from[2 * stride].x, + from[1 * stride].x, + from[0 * stride].x); + return result; + } + + template<> EIGEN_STRONG_INLINE void pscatter(half *to, const Packet16h &from, Index stride) + { + EIGEN_ALIGN64 half aux[16]; + pstore(aux, from); + to[stride * 0].x = aux[0].x; + to[stride * 1].x = aux[1].x; + to[stride * 2].x = aux[2].x; + to[stride * 3].x = aux[3].x; + to[stride * 4].x = aux[4].x; + to[stride * 5].x = aux[5].x; + to[stride * 6].x = aux[6].x; + to[stride * 7].x = aux[7].x; + to[stride * 8].x = aux[8].x; + to[stride * 9].x = aux[9].x; + to[stride * 10].x = aux[10].x; + to[stride * 11].x = aux[11].x; + to[stride * 12].x = aux[12].x; + to[stride * 13].x = aux[13].x; + to[stride * 14].x = aux[14].x; + to[stride * 15].x = aux[15].x; + } + + EIGEN_STRONG_INLINE void ptranspose(PacketBlock &kernel) + { + __m256i a = kernel.packet[0].x; + __m256i b = kernel.packet[1].x; + __m256i c = kernel.packet[2].x; + __m256i d = kernel.packet[3].x; + __m256i e = kernel.packet[4].x; + __m256i f = kernel.packet[5].x; + __m256i g = kernel.packet[6].x; + __m256i h = kernel.packet[7].x; + __m256i i = kernel.packet[8].x; + __m256i j = kernel.packet[9].x; + __m256i k = kernel.packet[10].x; + __m256i l = kernel.packet[11].x; + __m256i m = kernel.packet[12].x; + __m256i n = kernel.packet[13].x; + __m256i o = kernel.packet[14].x; + __m256i p = kernel.packet[15].x; + + __m256i ab_07 = _mm256_unpacklo_epi16(a, b); + __m256i cd_07 = _mm256_unpacklo_epi16(c, d); + __m256i ef_07 = _mm256_unpacklo_epi16(e, f); + __m256i gh_07 = _mm256_unpacklo_epi16(g, h); + __m256i ij_07 = _mm256_unpacklo_epi16(i, j); + __m256i kl_07 = _mm256_unpacklo_epi16(k, l); + __m256i mn_07 = _mm256_unpacklo_epi16(m, n); + __m256i op_07 = _mm256_unpacklo_epi16(o, p); + + __m256i ab_8f = _mm256_unpackhi_epi16(a, b); + __m256i cd_8f = _mm256_unpackhi_epi16(c, d); + __m256i ef_8f = _mm256_unpackhi_epi16(e, f); + __m256i gh_8f = _mm256_unpackhi_epi16(g, h); + __m256i ij_8f = _mm256_unpackhi_epi16(i, j); + __m256i kl_8f = _mm256_unpackhi_epi16(k, l); + __m256i mn_8f = _mm256_unpackhi_epi16(m, n); + __m256i op_8f = _mm256_unpackhi_epi16(o, p); + + __m256i abcd_03 = _mm256_unpacklo_epi32(ab_07, cd_07); + __m256i abcd_47 = _mm256_unpackhi_epi32(ab_07, cd_07); + __m256i efgh_03 = _mm256_unpacklo_epi32(ef_07, gh_07); + __m256i efgh_47 = _mm256_unpackhi_epi32(ef_07, gh_07); + __m256i ijkl_03 = _mm256_unpacklo_epi32(ij_07, kl_07); + __m256i ijkl_47 = _mm256_unpackhi_epi32(ij_07, kl_07); + __m256i mnop_03 = _mm256_unpacklo_epi32(mn_07, op_07); + __m256i mnop_47 = _mm256_unpackhi_epi32(mn_07, op_07); + + __m256i abcd_8b = _mm256_unpacklo_epi32(ab_8f, cd_8f); + __m256i abcd_cf = _mm256_unpackhi_epi32(ab_8f, cd_8f); + __m256i efgh_8b = _mm256_unpacklo_epi32(ef_8f, gh_8f); + __m256i efgh_cf = _mm256_unpackhi_epi32(ef_8f, gh_8f); + __m256i ijkl_8b = _mm256_unpacklo_epi32(ij_8f, kl_8f); + __m256i ijkl_cf = _mm256_unpackhi_epi32(ij_8f, kl_8f); + __m256i mnop_8b = _mm256_unpacklo_epi32(mn_8f, op_8f); + __m256i mnop_cf = _mm256_unpackhi_epi32(mn_8f, op_8f); + + __m256i abcdefgh_01 = _mm256_unpacklo_epi64(abcd_03, efgh_03); + __m256i abcdefgh_23 = _mm256_unpackhi_epi64(abcd_03, efgh_03); + __m256i ijklmnop_01 = _mm256_unpacklo_epi64(ijkl_03, mnop_03); + __m256i ijklmnop_23 = _mm256_unpackhi_epi64(ijkl_03, mnop_03); + __m256i abcdefgh_45 = _mm256_unpacklo_epi64(abcd_47, efgh_47); + __m256i abcdefgh_67 = _mm256_unpackhi_epi64(abcd_47, efgh_47); + __m256i ijklmnop_45 = _mm256_unpacklo_epi64(ijkl_47, mnop_47); + __m256i ijklmnop_67 = _mm256_unpackhi_epi64(ijkl_47, mnop_47); + __m256i abcdefgh_89 = _mm256_unpacklo_epi64(abcd_8b, efgh_8b); + __m256i abcdefgh_ab = _mm256_unpackhi_epi64(abcd_8b, efgh_8b); + __m256i ijklmnop_89 = _mm256_unpacklo_epi64(ijkl_8b, mnop_8b); + __m256i ijklmnop_ab = _mm256_unpackhi_epi64(ijkl_8b, mnop_8b); + __m256i abcdefgh_cd = _mm256_unpacklo_epi64(abcd_cf, efgh_cf); + __m256i abcdefgh_ef = _mm256_unpackhi_epi64(abcd_cf, efgh_cf); + __m256i ijklmnop_cd = _mm256_unpacklo_epi64(ijkl_cf, mnop_cf); + __m256i ijklmnop_ef = _mm256_unpackhi_epi64(ijkl_cf, mnop_cf); + + // NOTE: no unpacklo/hi instr in this case, so using permute instr. + __m256i a_p_0 = _mm256_permute2x128_si256(abcdefgh_01, ijklmnop_01, 0x20); + __m256i a_p_1 = _mm256_permute2x128_si256(abcdefgh_01, ijklmnop_01, 0x31); + __m256i a_p_2 = _mm256_permute2x128_si256(abcdefgh_23, ijklmnop_23, 0x20); + __m256i a_p_3 = _mm256_permute2x128_si256(abcdefgh_23, ijklmnop_23, 0x31); + __m256i a_p_4 = _mm256_permute2x128_si256(abcdefgh_45, ijklmnop_45, 0x20); + __m256i a_p_5 = _mm256_permute2x128_si256(abcdefgh_45, ijklmnop_45, 0x31); + __m256i a_p_6 = _mm256_permute2x128_si256(abcdefgh_67, ijklmnop_67, 0x20); + __m256i a_p_7 = _mm256_permute2x128_si256(abcdefgh_67, ijklmnop_67, 0x31); + __m256i a_p_8 = _mm256_permute2x128_si256(abcdefgh_89, ijklmnop_89, 0x20); + __m256i a_p_9 = _mm256_permute2x128_si256(abcdefgh_89, ijklmnop_89, 0x31); + __m256i a_p_a = _mm256_permute2x128_si256(abcdefgh_ab, ijklmnop_ab, 0x20); + __m256i a_p_b = _mm256_permute2x128_si256(abcdefgh_ab, ijklmnop_ab, 0x31); + __m256i a_p_c = _mm256_permute2x128_si256(abcdefgh_cd, ijklmnop_cd, 0x20); + __m256i a_p_d = _mm256_permute2x128_si256(abcdefgh_cd, ijklmnop_cd, 0x31); + __m256i a_p_e = _mm256_permute2x128_si256(abcdefgh_ef, ijklmnop_ef, 0x20); + __m256i a_p_f = _mm256_permute2x128_si256(abcdefgh_ef, ijklmnop_ef, 0x31); + + kernel.packet[0].x = a_p_0; + kernel.packet[1].x = a_p_1; + kernel.packet[2].x = a_p_2; + kernel.packet[3].x = a_p_3; + kernel.packet[4].x = a_p_4; + kernel.packet[5].x = a_p_5; + kernel.packet[6].x = a_p_6; + kernel.packet[7].x = a_p_7; + kernel.packet[8].x = a_p_8; + kernel.packet[9].x = a_p_9; + kernel.packet[10].x = a_p_a; + kernel.packet[11].x = a_p_b; + kernel.packet[12].x = a_p_c; + kernel.packet[13].x = a_p_d; + kernel.packet[14].x = a_p_e; + kernel.packet[15].x = a_p_f; + } + + EIGEN_STRONG_INLINE void ptranspose(PacketBlock &kernel) + { + EIGEN_ALIGN64 half in[8][16]; + pstore(in[0], kernel.packet[0]); + pstore(in[1], kernel.packet[1]); + pstore(in[2], kernel.packet[2]); + pstore(in[3], kernel.packet[3]); + pstore(in[4], kernel.packet[4]); + pstore(in[5], kernel.packet[5]); + pstore(in[6], kernel.packet[6]); + pstore(in[7], kernel.packet[7]); + + EIGEN_ALIGN64 half out[8][16]; + + for (int i = 0; i < 8; ++i) { + for (int j = 0; j < 8; ++j) { out[i][j] = in[j][2 * i]; } + for (int j = 0; j < 8; ++j) { out[i][j + 8] = in[j][2 * i + 1]; } } + + kernel.packet[0] = pload(out[0]); + kernel.packet[1] = pload(out[1]); + kernel.packet[2] = pload(out[2]); + kernel.packet[3] = pload(out[3]); + kernel.packet[4] = pload(out[4]); + kernel.packet[5] = pload(out[5]); + kernel.packet[6] = pload(out[6]); + kernel.packet[7] = pload(out[7]); } - kernel.packet[0] = pload(out[0]); - kernel.packet[1] = pload(out[1]); - kernel.packet[2] = pload(out[2]); - kernel.packet[3] = pload(out[3]); -} + EIGEN_STRONG_INLINE void ptranspose(PacketBlock &kernel) + { + EIGEN_ALIGN64 half in[4][16]; + pstore(in[0], kernel.packet[0]); + pstore(in[1], kernel.packet[1]); + pstore(in[2], kernel.packet[2]); + pstore(in[3], kernel.packet[3]); + + EIGEN_ALIGN64 half out[4][16]; + + for (int i = 0; i < 4; ++i) { + for (int j = 0; j < 4; ++j) { out[i][j] = in[j][4 * i]; } + for (int j = 0; j < 4; ++j) { out[i][j + 4] = in[j][4 * i + 1]; } + for (int j = 0; j < 4; ++j) { out[i][j + 8] = in[j][4 * i + 2]; } + for (int j = 0; j < 4; ++j) { out[i][j + 12] = in[j][4 * i + 3]; } + } + + kernel.packet[0] = pload(out[0]); + kernel.packet[1] = pload(out[1]); + kernel.packet[2] = pload(out[2]); + kernel.packet[3] = pload(out[3]); + } #elif defined EIGEN_VECTORIZE_AVX -typedef struct { - __m128i x; -} Packet8h; - - -template<> struct is_arithmetic { enum { value = true }; }; - -template <> -struct packet_traits : default_packet_traits { - typedef Packet8h type; - // There is no half-size packet for Packet8h. - typedef Packet8h half; - enum { - Vectorizable = 1, - AlignedOnScalar = 1, - size = 8, - HasHalfPacket = 0, - HasAdd = 0, - HasSub = 0, - HasMul = 0, - HasNegate = 0, - HasAbs = 0, - HasAbs2 = 0, - HasMin = 0, - HasMax = 0, - HasConj = 0, - HasSetLinear = 0, - HasDiv = 0, - HasSqrt = 0, - HasRsqrt = 0, - HasExp = 0, - HasLog = 0, - HasBlend = 0 + typedef struct + { + __m128i x; + } Packet8h; + + + template<> struct is_arithmetic + { + enum { value = true }; + }; + + template<> struct packet_traits : default_packet_traits + { + typedef Packet8h type; + // There is no half-size packet for Packet8h. + typedef Packet8h half; + enum { + Vectorizable = 1, + AlignedOnScalar = 1, + size = 8, + HasHalfPacket = 0, + HasAdd = 0, + HasSub = 0, + HasMul = 0, + HasNegate = 0, + HasAbs = 0, + HasAbs2 = 0, + HasMin = 0, + HasMax = 0, + HasConj = 0, + HasSetLinear = 0, + HasDiv = 0, + HasSqrt = 0, + HasRsqrt = 0, + HasExp = 0, + HasLog = 0, + HasBlend = 0 + }; + }; + + + template<> struct unpacket_traits + { + typedef Eigen::half type; + enum { size = 8, alignment = Aligned16 }; + typedef Packet8h half; }; -}; - - -template<> struct unpacket_traits { typedef Eigen::half type; enum {size=8, alignment=Aligned16}; typedef Packet8h half; }; - -template<> EIGEN_STRONG_INLINE Packet8h pset1(const Eigen::half& from) { - Packet8h result; - result.x = _mm_set1_epi16(from.x); - return result; -} - -template<> EIGEN_STRONG_INLINE Eigen::half pfirst(const Packet8h& from) { - return half_impl::raw_uint16_to_half(static_cast(_mm_extract_epi16(from.x, 0))); -} - -template<> EIGEN_STRONG_INLINE Packet8h pload(const Eigen::half* from) { - Packet8h result; - result.x = _mm_load_si128(reinterpret_cast(from)); - return result; -} - -template<> EIGEN_STRONG_INLINE Packet8h ploadu(const Eigen::half* from) { - Packet8h result; - result.x = _mm_loadu_si128(reinterpret_cast(from)); - return result; -} - -template<> EIGEN_STRONG_INLINE void pstore(Eigen::half* to, const Packet8h& from) { - _mm_store_si128(reinterpret_cast<__m128i*>(to), from.x); -} - -template<> EIGEN_STRONG_INLINE void pstoreu(Eigen::half* to, const Packet8h& from) { - _mm_storeu_si128(reinterpret_cast<__m128i*>(to), from.x); -} - -template<> EIGEN_STRONG_INLINE Packet8h -ploadquad(const Eigen::half* from) { - Packet8h result; - unsigned short a = from[0].x; - unsigned short b = from[1].x; - result.x = _mm_set_epi16(b, b, b, b, a, a, a, a); - return result; -} - -EIGEN_STRONG_INLINE Packet8f half2float(const Packet8h& a) { + + template<> EIGEN_STRONG_INLINE Packet8h pset1(const Eigen::half &from) + { + Packet8h result; + result.x = _mm_set1_epi16(from.x); + return result; + } + + template<> EIGEN_STRONG_INLINE Eigen::half pfirst(const Packet8h &from) + { + return half_impl::raw_uint16_to_half(static_cast(_mm_extract_epi16(from.x, 0))); + } + + template<> EIGEN_STRONG_INLINE Packet8h pload(const Eigen::half *from) + { + Packet8h result; + result.x = _mm_load_si128(reinterpret_cast(from)); + return result; + } + + template<> EIGEN_STRONG_INLINE Packet8h ploadu(const Eigen::half *from) + { + Packet8h result; + result.x = _mm_loadu_si128(reinterpret_cast(from)); + return result; + } + + template<> EIGEN_STRONG_INLINE void pstore(Eigen::half *to, const Packet8h &from) + { + _mm_store_si128(reinterpret_cast<__m128i *>(to), from.x); + } + + template<> EIGEN_STRONG_INLINE void pstoreu(Eigen::half *to, const Packet8h &from) + { + _mm_storeu_si128(reinterpret_cast<__m128i *>(to), from.x); + } + + template<> EIGEN_STRONG_INLINE Packet8h ploadquad(const Eigen::half *from) + { + Packet8h result; + unsigned short a = from[0].x; + unsigned short b = from[1].x; + result.x = _mm_set_epi16(b, b, b, b, a, a, a, a); + return result; + } + + EIGEN_STRONG_INLINE Packet8f half2float(const Packet8h &a) + { #ifdef EIGEN_HAS_FP16_C - return _mm256_cvtph_ps(a.x); + return _mm256_cvtph_ps(a.x); #else - EIGEN_ALIGN32 Eigen::half aux[8]; - pstore(aux, a); - float f0(aux[0]); - float f1(aux[1]); - float f2(aux[2]); - float f3(aux[3]); - float f4(aux[4]); - float f5(aux[5]); - float f6(aux[6]); - float f7(aux[7]); - - return _mm256_set_ps(f7, f6, f5, f4, f3, f2, f1, f0); + EIGEN_ALIGN32 Eigen::half aux[8]; + pstore(aux, a); + float f0(aux[0]); + float f1(aux[1]); + float f2(aux[2]); + float f3(aux[3]); + float f4(aux[4]); + float f5(aux[5]); + float f6(aux[6]); + float f7(aux[7]); + + return _mm256_set_ps(f7, f6, f5, f4, f3, f2, f1, f0); #endif -} + } -EIGEN_STRONG_INLINE Packet8h float2half(const Packet8f& a) { + EIGEN_STRONG_INLINE Packet8h float2half(const Packet8f &a) + { #ifdef EIGEN_HAS_FP16_C - Packet8h result; - result.x = _mm256_cvtps_ph(a, _MM_FROUND_TO_NEAREST_INT|_MM_FROUND_NO_EXC); - return result; + Packet8h result; + result.x = _mm256_cvtps_ph(a, _MM_FROUND_TO_NEAREST_INT | _MM_FROUND_NO_EXC); + return result; #else - EIGEN_ALIGN32 float aux[8]; - pstore(aux, a); - Eigen::half h0(aux[0]); - Eigen::half h1(aux[1]); - Eigen::half h2(aux[2]); - Eigen::half h3(aux[3]); - Eigen::half h4(aux[4]); - Eigen::half h5(aux[5]); - Eigen::half h6(aux[6]); - Eigen::half h7(aux[7]); - - Packet8h result; - result.x = _mm_set_epi16(h7.x, h6.x, h5.x, h4.x, h3.x, h2.x, h1.x, h0.x); - return result; + EIGEN_ALIGN32 float aux[8]; + pstore(aux, a); + Eigen::half h0(aux[0]); + Eigen::half h1(aux[1]); + Eigen::half h2(aux[2]); + Eigen::half h3(aux[3]); + Eigen::half h4(aux[4]); + Eigen::half h5(aux[5]); + Eigen::half h6(aux[6]); + Eigen::half h7(aux[7]); + + Packet8h result; + result.x = _mm_set_epi16(h7.x, h6.x, h5.x, h4.x, h3.x, h2.x, h1.x, h0.x); + return result; #endif -} - -template<> EIGEN_STRONG_INLINE Packet8h pconj(const Packet8h& a) { return a; } - -template<> EIGEN_STRONG_INLINE Packet8h padd(const Packet8h& a, const Packet8h& b) { - Packet8f af = half2float(a); - Packet8f bf = half2float(b); - Packet8f rf = padd(af, bf); - return float2half(rf); -} - -template<> EIGEN_STRONG_INLINE Packet8h pmul(const Packet8h& a, const Packet8h& b) { - Packet8f af = half2float(a); - Packet8f bf = half2float(b); - Packet8f rf = pmul(af, bf); - return float2half(rf); -} - -template<> EIGEN_STRONG_INLINE Packet8h pgather(const Eigen::half* from, Index stride) -{ - Packet8h result; - result.x = _mm_set_epi16(from[7*stride].x, from[6*stride].x, from[5*stride].x, from[4*stride].x, from[3*stride].x, from[2*stride].x, from[1*stride].x, from[0*stride].x); - return result; -} - -template<> EIGEN_STRONG_INLINE void pscatter(Eigen::half* to, const Packet8h& from, Index stride) -{ - EIGEN_ALIGN32 Eigen::half aux[8]; - pstore(aux, from); - to[stride*0].x = aux[0].x; - to[stride*1].x = aux[1].x; - to[stride*2].x = aux[2].x; - to[stride*3].x = aux[3].x; - to[stride*4].x = aux[4].x; - to[stride*5].x = aux[5].x; - to[stride*6].x = aux[6].x; - to[stride*7].x = aux[7].x; -} - -template<> EIGEN_STRONG_INLINE Eigen::half predux(const Packet8h& a) { - Packet8f af = half2float(a); - float reduced = predux(af); - return Eigen::half(reduced); -} - -template<> EIGEN_STRONG_INLINE Eigen::half predux_max(const Packet8h& a) { - Packet8f af = half2float(a); - float reduced = predux_max(af); - return Eigen::half(reduced); -} - -template<> EIGEN_STRONG_INLINE Eigen::half predux_min(const Packet8h& a) { - Packet8f af = half2float(a); - float reduced = predux_min(af); - return Eigen::half(reduced); -} - -template<> EIGEN_STRONG_INLINE Eigen::half predux_mul(const Packet8h& a) { - Packet8f af = half2float(a); - float reduced = predux_mul(af); - return Eigen::half(reduced); -} - -EIGEN_STRONG_INLINE void -ptranspose(PacketBlock& kernel) { - __m128i a = kernel.packet[0].x; - __m128i b = kernel.packet[1].x; - __m128i c = kernel.packet[2].x; - __m128i d = kernel.packet[3].x; - __m128i e = kernel.packet[4].x; - __m128i f = kernel.packet[5].x; - __m128i g = kernel.packet[6].x; - __m128i h = kernel.packet[7].x; - - __m128i a03b03 = _mm_unpacklo_epi16(a, b); - __m128i c03d03 = _mm_unpacklo_epi16(c, d); - __m128i e03f03 = _mm_unpacklo_epi16(e, f); - __m128i g03h03 = _mm_unpacklo_epi16(g, h); - __m128i a47b47 = _mm_unpackhi_epi16(a, b); - __m128i c47d47 = _mm_unpackhi_epi16(c, d); - __m128i e47f47 = _mm_unpackhi_epi16(e, f); - __m128i g47h47 = _mm_unpackhi_epi16(g, h); - - __m128i a01b01c01d01 = _mm_unpacklo_epi32(a03b03, c03d03); - __m128i a23b23c23d23 = _mm_unpackhi_epi32(a03b03, c03d03); - __m128i e01f01g01h01 = _mm_unpacklo_epi32(e03f03, g03h03); - __m128i e23f23g23h23 = _mm_unpackhi_epi32(e03f03, g03h03); - __m128i a45b45c45d45 = _mm_unpacklo_epi32(a47b47, c47d47); - __m128i a67b67c67d67 = _mm_unpackhi_epi32(a47b47, c47d47); - __m128i e45f45g45h45 = _mm_unpacklo_epi32(e47f47, g47h47); - __m128i e67f67g67h67 = _mm_unpackhi_epi32(e47f47, g47h47); - - __m128i a0b0c0d0e0f0g0h0 = _mm_unpacklo_epi64(a01b01c01d01, e01f01g01h01); - __m128i a1b1c1d1e1f1g1h1 = _mm_unpackhi_epi64(a01b01c01d01, e01f01g01h01); - __m128i a2b2c2d2e2f2g2h2 = _mm_unpacklo_epi64(a23b23c23d23, e23f23g23h23); - __m128i a3b3c3d3e3f3g3h3 = _mm_unpackhi_epi64(a23b23c23d23, e23f23g23h23); - __m128i a4b4c4d4e4f4g4h4 = _mm_unpacklo_epi64(a45b45c45d45, e45f45g45h45); - __m128i a5b5c5d5e5f5g5h5 = _mm_unpackhi_epi64(a45b45c45d45, e45f45g45h45); - __m128i a6b6c6d6e6f6g6h6 = _mm_unpacklo_epi64(a67b67c67d67, e67f67g67h67); - __m128i a7b7c7d7e7f7g7h7 = _mm_unpackhi_epi64(a67b67c67d67, e67f67g67h67); - - kernel.packet[0].x = a0b0c0d0e0f0g0h0; - kernel.packet[1].x = a1b1c1d1e1f1g1h1; - kernel.packet[2].x = a2b2c2d2e2f2g2h2; - kernel.packet[3].x = a3b3c3d3e3f3g3h3; - kernel.packet[4].x = a4b4c4d4e4f4g4h4; - kernel.packet[5].x = a5b5c5d5e5f5g5h5; - kernel.packet[6].x = a6b6c6d6e6f6g6h6; - kernel.packet[7].x = a7b7c7d7e7f7g7h7; -} - -EIGEN_STRONG_INLINE void -ptranspose(PacketBlock& kernel) { - EIGEN_ALIGN32 Eigen::half in[4][8]; - pstore(in[0], kernel.packet[0]); - pstore(in[1], kernel.packet[1]); - pstore(in[2], kernel.packet[2]); - pstore(in[3], kernel.packet[3]); - - EIGEN_ALIGN32 Eigen::half out[4][8]; - - for (int i = 0; i < 4; ++i) { - for (int j = 0; j < 4; ++j) { - out[i][j] = in[j][2*i]; - } - for (int j = 0; j < 4; ++j) { - out[i][j+4] = in[j][2*i+1]; - } } - kernel.packet[0] = pload(out[0]); - kernel.packet[1] = pload(out[1]); - kernel.packet[2] = pload(out[2]); - kernel.packet[3] = pload(out[3]); -} + template<> EIGEN_STRONG_INLINE Packet8h pconj(const Packet8h &a) { return a; } + + template<> EIGEN_STRONG_INLINE Packet8h padd(const Packet8h &a, const Packet8h &b) + { + Packet8f af = half2float(a); + Packet8f bf = half2float(b); + Packet8f rf = padd(af, bf); + return float2half(rf); + } + + template<> EIGEN_STRONG_INLINE Packet8h pmul(const Packet8h &a, const Packet8h &b) + { + Packet8f af = half2float(a); + Packet8f bf = half2float(b); + Packet8f rf = pmul(af, bf); + return float2half(rf); + } + + template<> EIGEN_STRONG_INLINE Packet8h pgather(const Eigen::half *from, Index stride) + { + Packet8h result; + result.x = _mm_set_epi16(from[7 * stride].x, + from[6 * stride].x, + from[5 * stride].x, + from[4 * stride].x, + from[3 * stride].x, + from[2 * stride].x, + from[1 * stride].x, + from[0 * stride].x); + return result; + } + + template<> + EIGEN_STRONG_INLINE void pscatter(Eigen::half *to, const Packet8h &from, Index stride) + { + EIGEN_ALIGN32 Eigen::half aux[8]; + pstore(aux, from); + to[stride * 0].x = aux[0].x; + to[stride * 1].x = aux[1].x; + to[stride * 2].x = aux[2].x; + to[stride * 3].x = aux[3].x; + to[stride * 4].x = aux[4].x; + to[stride * 5].x = aux[5].x; + to[stride * 6].x = aux[6].x; + to[stride * 7].x = aux[7].x; + } + + template<> EIGEN_STRONG_INLINE Eigen::half predux(const Packet8h &a) + { + Packet8f af = half2float(a); + float reduced = predux(af); + return Eigen::half(reduced); + } + + template<> EIGEN_STRONG_INLINE Eigen::half predux_max(const Packet8h &a) + { + Packet8f af = half2float(a); + float reduced = predux_max(af); + return Eigen::half(reduced); + } + + template<> EIGEN_STRONG_INLINE Eigen::half predux_min(const Packet8h &a) + { + Packet8f af = half2float(a); + float reduced = predux_min(af); + return Eigen::half(reduced); + } + + template<> EIGEN_STRONG_INLINE Eigen::half predux_mul(const Packet8h &a) + { + Packet8f af = half2float(a); + float reduced = predux_mul(af); + return Eigen::half(reduced); + } + + EIGEN_STRONG_INLINE void ptranspose(PacketBlock &kernel) + { + __m128i a = kernel.packet[0].x; + __m128i b = kernel.packet[1].x; + __m128i c = kernel.packet[2].x; + __m128i d = kernel.packet[3].x; + __m128i e = kernel.packet[4].x; + __m128i f = kernel.packet[5].x; + __m128i g = kernel.packet[6].x; + __m128i h = kernel.packet[7].x; + + __m128i a03b03 = _mm_unpacklo_epi16(a, b); + __m128i c03d03 = _mm_unpacklo_epi16(c, d); + __m128i e03f03 = _mm_unpacklo_epi16(e, f); + __m128i g03h03 = _mm_unpacklo_epi16(g, h); + __m128i a47b47 = _mm_unpackhi_epi16(a, b); + __m128i c47d47 = _mm_unpackhi_epi16(c, d); + __m128i e47f47 = _mm_unpackhi_epi16(e, f); + __m128i g47h47 = _mm_unpackhi_epi16(g, h); + + __m128i a01b01c01d01 = _mm_unpacklo_epi32(a03b03, c03d03); + __m128i a23b23c23d23 = _mm_unpackhi_epi32(a03b03, c03d03); + __m128i e01f01g01h01 = _mm_unpacklo_epi32(e03f03, g03h03); + __m128i e23f23g23h23 = _mm_unpackhi_epi32(e03f03, g03h03); + __m128i a45b45c45d45 = _mm_unpacklo_epi32(a47b47, c47d47); + __m128i a67b67c67d67 = _mm_unpackhi_epi32(a47b47, c47d47); + __m128i e45f45g45h45 = _mm_unpacklo_epi32(e47f47, g47h47); + __m128i e67f67g67h67 = _mm_unpackhi_epi32(e47f47, g47h47); + + __m128i a0b0c0d0e0f0g0h0 = _mm_unpacklo_epi64(a01b01c01d01, e01f01g01h01); + __m128i a1b1c1d1e1f1g1h1 = _mm_unpackhi_epi64(a01b01c01d01, e01f01g01h01); + __m128i a2b2c2d2e2f2g2h2 = _mm_unpacklo_epi64(a23b23c23d23, e23f23g23h23); + __m128i a3b3c3d3e3f3g3h3 = _mm_unpackhi_epi64(a23b23c23d23, e23f23g23h23); + __m128i a4b4c4d4e4f4g4h4 = _mm_unpacklo_epi64(a45b45c45d45, e45f45g45h45); + __m128i a5b5c5d5e5f5g5h5 = _mm_unpackhi_epi64(a45b45c45d45, e45f45g45h45); + __m128i a6b6c6d6e6f6g6h6 = _mm_unpacklo_epi64(a67b67c67d67, e67f67g67h67); + __m128i a7b7c7d7e7f7g7h7 = _mm_unpackhi_epi64(a67b67c67d67, e67f67g67h67); + + kernel.packet[0].x = a0b0c0d0e0f0g0h0; + kernel.packet[1].x = a1b1c1d1e1f1g1h1; + kernel.packet[2].x = a2b2c2d2e2f2g2h2; + kernel.packet[3].x = a3b3c3d3e3f3g3h3; + kernel.packet[4].x = a4b4c4d4e4f4g4h4; + kernel.packet[5].x = a5b5c5d5e5f5g5h5; + kernel.packet[6].x = a6b6c6d6e6f6g6h6; + kernel.packet[7].x = a7b7c7d7e7f7g7h7; + } + + EIGEN_STRONG_INLINE void ptranspose(PacketBlock &kernel) + { + EIGEN_ALIGN32 Eigen::half in[4][8]; + pstore(in[0], kernel.packet[0]); + pstore(in[1], kernel.packet[1]); + pstore(in[2], kernel.packet[2]); + pstore(in[3], kernel.packet[3]); + + EIGEN_ALIGN32 Eigen::half out[4][8]; + + for (int i = 0; i < 4; ++i) { + for (int j = 0; j < 4; ++j) { out[i][j] = in[j][2 * i]; } + for (int j = 0; j < 4; ++j) { out[i][j + 4] = in[j][2 * i + 1]; } + } + + kernel.packet[0] = pload(out[0]); + kernel.packet[1] = pload(out[1]); + kernel.packet[2] = pload(out[2]); + kernel.packet[3] = pload(out[3]); + } // Disable the following code since it's broken on too many platforms / compilers. -//#elif defined(EIGEN_VECTORIZE_SSE) && (!EIGEN_ARCH_x86_64) && (!EIGEN_COMP_MSVC) +// #elif defined(EIGEN_VECTORIZE_SSE) && (!EIGEN_ARCH_x86_64) && (!EIGEN_COMP_MSVC) #elif 0 -typedef struct { - __m64 x; -} Packet4h; - - -template<> struct is_arithmetic { enum { value = true }; }; - -template <> -struct packet_traits : default_packet_traits { - typedef Packet4h type; - // There is no half-size packet for Packet4h. - typedef Packet4h half; - enum { - Vectorizable = 1, - AlignedOnScalar = 1, - size = 4, - HasHalfPacket = 0, - HasAdd = 0, - HasSub = 0, - HasMul = 0, - HasNegate = 0, - HasAbs = 0, - HasAbs2 = 0, - HasMin = 0, - HasMax = 0, - HasConj = 0, - HasSetLinear = 0, - HasDiv = 0, - HasSqrt = 0, - HasRsqrt = 0, - HasExp = 0, - HasLog = 0, - HasBlend = 0 + typedef struct + { + __m64 x; + } Packet4h; + + + template<> struct is_arithmetic + { + enum { value = true }; + }; + + template<> struct packet_traits : default_packet_traits + { + typedef Packet4h type; + // There is no half-size packet for Packet4h. + typedef Packet4h half; + enum { + Vectorizable = 1, + AlignedOnScalar = 1, + size = 4, + HasHalfPacket = 0, + HasAdd = 0, + HasSub = 0, + HasMul = 0, + HasNegate = 0, + HasAbs = 0, + HasAbs2 = 0, + HasMin = 0, + HasMax = 0, + HasConj = 0, + HasSetLinear = 0, + HasDiv = 0, + HasSqrt = 0, + HasRsqrt = 0, + HasExp = 0, + HasLog = 0, + HasBlend = 0 + }; + }; + + + template<> struct unpacket_traits + { + typedef Eigen::half type; + enum { size = 4, alignment = Aligned16 }; + typedef Packet4h half; }; -}; - - -template<> struct unpacket_traits { typedef Eigen::half type; enum {size=4, alignment=Aligned16}; typedef Packet4h half; }; - -template<> EIGEN_STRONG_INLINE Packet4h pset1(const Eigen::half& from) { - Packet4h result; - result.x = _mm_set1_pi16(from.x); - return result; -} - -template<> EIGEN_STRONG_INLINE Eigen::half pfirst(const Packet4h& from) { - return half_impl::raw_uint16_to_half(static_cast(_mm_cvtsi64_si32(from.x))); -} - -template<> EIGEN_STRONG_INLINE Packet4h pconj(const Packet4h& a) { return a; } - -template<> EIGEN_STRONG_INLINE Packet4h padd(const Packet4h& a, const Packet4h& b) { - __int64_t a64 = _mm_cvtm64_si64(a.x); - __int64_t b64 = _mm_cvtm64_si64(b.x); - - Eigen::half h[4]; - - Eigen::half ha = half_impl::raw_uint16_to_half(static_cast(a64)); - Eigen::half hb = half_impl::raw_uint16_to_half(static_cast(b64)); - h[0] = ha + hb; - ha = half_impl::raw_uint16_to_half(static_cast(a64 >> 16)); - hb = half_impl::raw_uint16_to_half(static_cast(b64 >> 16)); - h[1] = ha + hb; - ha = half_impl::raw_uint16_to_half(static_cast(a64 >> 32)); - hb = half_impl::raw_uint16_to_half(static_cast(b64 >> 32)); - h[2] = ha + hb; - ha = half_impl::raw_uint16_to_half(static_cast(a64 >> 48)); - hb = half_impl::raw_uint16_to_half(static_cast(b64 >> 48)); - h[3] = ha + hb; - Packet4h result; - result.x = _mm_set_pi16(h[3].x, h[2].x, h[1].x, h[0].x); - return result; -} - -template<> EIGEN_STRONG_INLINE Packet4h pmul(const Packet4h& a, const Packet4h& b) { - __int64_t a64 = _mm_cvtm64_si64(a.x); - __int64_t b64 = _mm_cvtm64_si64(b.x); - - Eigen::half h[4]; - - Eigen::half ha = half_impl::raw_uint16_to_half(static_cast(a64)); - Eigen::half hb = half_impl::raw_uint16_to_half(static_cast(b64)); - h[0] = ha * hb; - ha = half_impl::raw_uint16_to_half(static_cast(a64 >> 16)); - hb = half_impl::raw_uint16_to_half(static_cast(b64 >> 16)); - h[1] = ha * hb; - ha = half_impl::raw_uint16_to_half(static_cast(a64 >> 32)); - hb = half_impl::raw_uint16_to_half(static_cast(b64 >> 32)); - h[2] = ha * hb; - ha = half_impl::raw_uint16_to_half(static_cast(a64 >> 48)); - hb = half_impl::raw_uint16_to_half(static_cast(b64 >> 48)); - h[3] = ha * hb; - Packet4h result; - result.x = _mm_set_pi16(h[3].x, h[2].x, h[1].x, h[0].x); - return result; -} - -template<> EIGEN_STRONG_INLINE Packet4h pload(const Eigen::half* from) { - Packet4h result; - result.x = _mm_cvtsi64_m64(*reinterpret_cast(from)); - return result; -} - -template<> EIGEN_STRONG_INLINE Packet4h ploadu(const Eigen::half* from) { - Packet4h result; - result.x = _mm_cvtsi64_m64(*reinterpret_cast(from)); - return result; -} - -template<> EIGEN_STRONG_INLINE void pstore(Eigen::half* to, const Packet4h& from) { - __int64_t r = _mm_cvtm64_si64(from.x); - *(reinterpret_cast<__int64_t*>(to)) = r; -} - -template<> EIGEN_STRONG_INLINE void pstoreu(Eigen::half* to, const Packet4h& from) { - __int64_t r = _mm_cvtm64_si64(from.x); - *(reinterpret_cast<__int64_t*>(to)) = r; -} - -template<> EIGEN_STRONG_INLINE Packet4h -ploadquad(const Eigen::half* from) { - return pset1(*from); -} - -template<> EIGEN_STRONG_INLINE Packet4h pgather(const Eigen::half* from, Index stride) -{ - Packet4h result; - result.x = _mm_set_pi16(from[3*stride].x, from[2*stride].x, from[1*stride].x, from[0*stride].x); - return result; -} - -template<> EIGEN_STRONG_INLINE void pscatter(Eigen::half* to, const Packet4h& from, Index stride) -{ - __int64_t a = _mm_cvtm64_si64(from.x); - to[stride*0].x = static_cast(a); - to[stride*1].x = static_cast(a >> 16); - to[stride*2].x = static_cast(a >> 32); - to[stride*3].x = static_cast(a >> 48); -} - -EIGEN_STRONG_INLINE void -ptranspose(PacketBlock& kernel) { - __m64 T0 = _mm_unpacklo_pi16(kernel.packet[0].x, kernel.packet[1].x); - __m64 T1 = _mm_unpacklo_pi16(kernel.packet[2].x, kernel.packet[3].x); - __m64 T2 = _mm_unpackhi_pi16(kernel.packet[0].x, kernel.packet[1].x); - __m64 T3 = _mm_unpackhi_pi16(kernel.packet[2].x, kernel.packet[3].x); - - kernel.packet[0].x = _mm_unpacklo_pi32(T0, T1); - kernel.packet[1].x = _mm_unpackhi_pi32(T0, T1); - kernel.packet[2].x = _mm_unpacklo_pi32(T2, T3); - kernel.packet[3].x = _mm_unpackhi_pi32(T2, T3); -} + + template<> EIGEN_STRONG_INLINE Packet4h pset1(const Eigen::half &from) + { + Packet4h result; + result.x = _mm_set1_pi16(from.x); + return result; + } + + template<> EIGEN_STRONG_INLINE Eigen::half pfirst(const Packet4h &from) + { + return half_impl::raw_uint16_to_half(static_cast(_mm_cvtsi64_si32(from.x))); + } + + template<> EIGEN_STRONG_INLINE Packet4h pconj(const Packet4h &a) { return a; } + + template<> EIGEN_STRONG_INLINE Packet4h padd(const Packet4h &a, const Packet4h &b) + { + __int64_t a64 = _mm_cvtm64_si64(a.x); + __int64_t b64 = _mm_cvtm64_si64(b.x); + + Eigen::half h[4]; + + Eigen::half ha = half_impl::raw_uint16_to_half(static_cast(a64)); + Eigen::half hb = half_impl::raw_uint16_to_half(static_cast(b64)); + h[0] = ha + hb; + ha = half_impl::raw_uint16_to_half(static_cast(a64 >> 16)); + hb = half_impl::raw_uint16_to_half(static_cast(b64 >> 16)); + h[1] = ha + hb; + ha = half_impl::raw_uint16_to_half(static_cast(a64 >> 32)); + hb = half_impl::raw_uint16_to_half(static_cast(b64 >> 32)); + h[2] = ha + hb; + ha = half_impl::raw_uint16_to_half(static_cast(a64 >> 48)); + hb = half_impl::raw_uint16_to_half(static_cast(b64 >> 48)); + h[3] = ha + hb; + Packet4h result; + result.x = _mm_set_pi16(h[3].x, h[2].x, h[1].x, h[0].x); + return result; + } + + template<> EIGEN_STRONG_INLINE Packet4h pmul(const Packet4h &a, const Packet4h &b) + { + __int64_t a64 = _mm_cvtm64_si64(a.x); + __int64_t b64 = _mm_cvtm64_si64(b.x); + + Eigen::half h[4]; + + Eigen::half ha = half_impl::raw_uint16_to_half(static_cast(a64)); + Eigen::half hb = half_impl::raw_uint16_to_half(static_cast(b64)); + h[0] = ha * hb; + ha = half_impl::raw_uint16_to_half(static_cast(a64 >> 16)); + hb = half_impl::raw_uint16_to_half(static_cast(b64 >> 16)); + h[1] = ha * hb; + ha = half_impl::raw_uint16_to_half(static_cast(a64 >> 32)); + hb = half_impl::raw_uint16_to_half(static_cast(b64 >> 32)); + h[2] = ha * hb; + ha = half_impl::raw_uint16_to_half(static_cast(a64 >> 48)); + hb = half_impl::raw_uint16_to_half(static_cast(b64 >> 48)); + h[3] = ha * hb; + Packet4h result; + result.x = _mm_set_pi16(h[3].x, h[2].x, h[1].x, h[0].x); + return result; + } + + template<> EIGEN_STRONG_INLINE Packet4h pload(const Eigen::half *from) + { + Packet4h result; + result.x = _mm_cvtsi64_m64(*reinterpret_cast(from)); + return result; + } + + template<> EIGEN_STRONG_INLINE Packet4h ploadu(const Eigen::half *from) + { + Packet4h result; + result.x = _mm_cvtsi64_m64(*reinterpret_cast(from)); + return result; + } + + template<> EIGEN_STRONG_INLINE void pstore(Eigen::half *to, const Packet4h &from) + { + __int64_t r = _mm_cvtm64_si64(from.x); + *(reinterpret_cast<__int64_t *>(to)) = r; + } + + template<> EIGEN_STRONG_INLINE void pstoreu(Eigen::half *to, const Packet4h &from) + { + __int64_t r = _mm_cvtm64_si64(from.x); + *(reinterpret_cast<__int64_t *>(to)) = r; + } + + template<> EIGEN_STRONG_INLINE Packet4h ploadquad(const Eigen::half *from) + { + return pset1(*from); + } + + template<> EIGEN_STRONG_INLINE Packet4h pgather(const Eigen::half *from, Index stride) + { + Packet4h result; + result.x = _mm_set_pi16(from[3 * stride].x, from[2 * stride].x, from[1 * stride].x, from[0 * stride].x); + return result; + } + + template<> + EIGEN_STRONG_INLINE void pscatter(Eigen::half *to, const Packet4h &from, Index stride) + { + __int64_t a = _mm_cvtm64_si64(from.x); + to[stride * 0].x = static_cast(a); + to[stride * 1].x = static_cast(a >> 16); + to[stride * 2].x = static_cast(a >> 32); + to[stride * 3].x = static_cast(a >> 48); + } + + EIGEN_STRONG_INLINE void ptranspose(PacketBlock &kernel) + { + __m64 T0 = _mm_unpacklo_pi16(kernel.packet[0].x, kernel.packet[1].x); + __m64 T1 = _mm_unpacklo_pi16(kernel.packet[2].x, kernel.packet[3].x); + __m64 T2 = _mm_unpackhi_pi16(kernel.packet[0].x, kernel.packet[1].x); + __m64 T3 = _mm_unpackhi_pi16(kernel.packet[2].x, kernel.packet[3].x); + + kernel.packet[0].x = _mm_unpacklo_pi32(T0, T1); + kernel.packet[1].x = _mm_unpackhi_pi32(T0, T1); + kernel.packet[2].x = _mm_unpacklo_pi32(T2, T3); + kernel.packet[3].x = _mm_unpackhi_pi32(T2, T3); + } #endif -} -} +}// namespace internal +}// namespace Eigen -#endif // EIGEN_PACKET_MATH_HALF_CUDA_H +#endif// EIGEN_PACKET_MATH_HALF_CUDA_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/arch/CUDA/TypeCasting.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/arch/CUDA/TypeCasting.h index aa5fbce8..146581a8 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/arch/CUDA/TypeCasting.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/arch/CUDA/TypeCasting.h @@ -14,199 +14,168 @@ namespace Eigen { namespace internal { -template<> -struct scalar_cast_op { - EIGEN_EMPTY_STRUCT_CTOR(scalar_cast_op) - typedef Eigen::half result_type; - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Eigen::half operator() (const float& a) const { - #if defined(EIGEN_HAS_CUDA_FP16) && defined(__CUDA_ARCH__) && __CUDA_ARCH__ >= 300 + template<> struct scalar_cast_op + { + EIGEN_EMPTY_STRUCT_CTOR(scalar_cast_op) + typedef Eigen::half result_type; + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Eigen::half operator()(const float &a) const + { +#if defined(EIGEN_HAS_CUDA_FP16) && defined(__CUDA_ARCH__) && __CUDA_ARCH__ >= 300 return __float2half(a); - #else +#else return Eigen::half(a); - #endif - } -}; +#endif + } + }; -template<> -struct functor_traits > -{ enum { Cost = NumTraits::AddCost, PacketAccess = false }; }; + template<> struct functor_traits> + { + enum { Cost = NumTraits::AddCost, PacketAccess = false }; + }; -template<> -struct scalar_cast_op { - EIGEN_EMPTY_STRUCT_CTOR(scalar_cast_op) - typedef Eigen::half result_type; - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Eigen::half operator() (const int& a) const { - #if defined(EIGEN_HAS_CUDA_FP16) && defined(__CUDA_ARCH__) && __CUDA_ARCH__ >= 300 + template<> struct scalar_cast_op + { + EIGEN_EMPTY_STRUCT_CTOR(scalar_cast_op) + typedef Eigen::half result_type; + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Eigen::half operator()(const int &a) const + { +#if defined(EIGEN_HAS_CUDA_FP16) && defined(__CUDA_ARCH__) && __CUDA_ARCH__ >= 300 return __float2half(static_cast(a)); - #else +#else return Eigen::half(static_cast(a)); - #endif - } -}; +#endif + } + }; -template<> -struct functor_traits > -{ enum { Cost = NumTraits::AddCost, PacketAccess = false }; }; + template<> struct functor_traits> + { + enum { Cost = NumTraits::AddCost, PacketAccess = false }; + }; -template<> -struct scalar_cast_op { - EIGEN_EMPTY_STRUCT_CTOR(scalar_cast_op) - typedef float result_type; - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE float operator() (const Eigen::half& a) const { - #if defined(EIGEN_HAS_CUDA_FP16) && defined(__CUDA_ARCH__) && __CUDA_ARCH__ >= 300 + template<> struct scalar_cast_op + { + EIGEN_EMPTY_STRUCT_CTOR(scalar_cast_op) + typedef float result_type; + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE float operator()(const Eigen::half &a) const + { +#if defined(EIGEN_HAS_CUDA_FP16) && defined(__CUDA_ARCH__) && __CUDA_ARCH__ >= 300 return __half2float(a); - #else +#else return static_cast(a); - #endif - } -}; - -template<> -struct functor_traits > -{ enum { Cost = NumTraits::AddCost, PacketAccess = false }; }; +#endif + } + }; + template<> struct functor_traits> + { + enum { Cost = NumTraits::AddCost, PacketAccess = false }; + }; #if defined(EIGEN_HAS_CUDA_FP16) && defined(__CUDA_ARCH__) && __CUDA_ARCH__ >= 300 -template <> -struct type_casting_traits { - enum { - VectorizedCast = 1, - SrcCoeffRatio = 2, - TgtCoeffRatio = 1 + template<> struct type_casting_traits + { + enum { VectorizedCast = 1, SrcCoeffRatio = 2, TgtCoeffRatio = 1 }; }; -}; - -template<> EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE float4 pcast(const half2& a, const half2& b) { - float2 r1 = __half22float2(a); - float2 r2 = __half22float2(b); - return make_float4(r1.x, r1.y, r2.x, r2.y); -} - -template <> -struct type_casting_traits { - enum { - VectorizedCast = 1, - SrcCoeffRatio = 1, - TgtCoeffRatio = 2 + + template<> EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE float4 pcast(const half2 &a, const half2 &b) + { + float2 r1 = __half22float2(a); + float2 r2 = __half22float2(b); + return make_float4(r1.x, r1.y, r2.x, r2.y); + } + + template<> struct type_casting_traits + { + enum { VectorizedCast = 1, SrcCoeffRatio = 1, TgtCoeffRatio = 2 }; }; -}; -template<> EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE half2 pcast(const float4& a) { - // Simply discard the second half of the input - return __floats2half2_rn(a.x, a.y); -} + template<> EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE half2 pcast(const float4 &a) + { + // Simply discard the second half of the input + return __floats2half2_rn(a.x, a.y); + } #elif defined EIGEN_VECTORIZE_AVX512 -template <> -struct type_casting_traits { - enum { - VectorizedCast = 1, - SrcCoeffRatio = 1, - TgtCoeffRatio = 1 + template<> struct type_casting_traits + { + enum { VectorizedCast = 1, SrcCoeffRatio = 1, TgtCoeffRatio = 1 }; }; -}; - -template<> EIGEN_STRONG_INLINE Packet16f pcast(const Packet16h& a) { - return half2float(a); -} - -template <> -struct type_casting_traits { - enum { - VectorizedCast = 1, - SrcCoeffRatio = 1, - TgtCoeffRatio = 1 + + template<> EIGEN_STRONG_INLINE Packet16f pcast(const Packet16h &a) { return half2float(a); } + + template<> struct type_casting_traits + { + enum { VectorizedCast = 1, SrcCoeffRatio = 1, TgtCoeffRatio = 1 }; }; -}; -template<> EIGEN_STRONG_INLINE Packet16h pcast(const Packet16f& a) { - return float2half(a); -} + template<> EIGEN_STRONG_INLINE Packet16h pcast(const Packet16f &a) { return float2half(a); } #elif defined EIGEN_VECTORIZE_AVX -template <> -struct type_casting_traits { - enum { - VectorizedCast = 1, - SrcCoeffRatio = 1, - TgtCoeffRatio = 1 + template<> struct type_casting_traits + { + enum { VectorizedCast = 1, SrcCoeffRatio = 1, TgtCoeffRatio = 1 }; }; -}; - -template<> EIGEN_STRONG_INLINE Packet8f pcast(const Packet8h& a) { - return half2float(a); -} - -template <> -struct type_casting_traits { - enum { - VectorizedCast = 1, - SrcCoeffRatio = 1, - TgtCoeffRatio = 1 + + template<> EIGEN_STRONG_INLINE Packet8f pcast(const Packet8h &a) { return half2float(a); } + + template<> struct type_casting_traits + { + enum { VectorizedCast = 1, SrcCoeffRatio = 1, TgtCoeffRatio = 1 }; }; -}; -template<> EIGEN_STRONG_INLINE Packet8h pcast(const Packet8f& a) { - return float2half(a); -} + template<> EIGEN_STRONG_INLINE Packet8h pcast(const Packet8f &a) { return float2half(a); } // Disable the following code since it's broken on too many platforms / compilers. -//#elif defined(EIGEN_VECTORIZE_SSE) && (!EIGEN_ARCH_x86_64) && (!EIGEN_COMP_MSVC) +// #elif defined(EIGEN_VECTORIZE_SSE) && (!EIGEN_ARCH_x86_64) && (!EIGEN_COMP_MSVC) #elif 0 -template <> -struct type_casting_traits { - enum { - VectorizedCast = 1, - SrcCoeffRatio = 1, - TgtCoeffRatio = 1 + template<> struct type_casting_traits + { + enum { VectorizedCast = 1, SrcCoeffRatio = 1, TgtCoeffRatio = 1 }; }; -}; - -template<> EIGEN_STRONG_INLINE Packet4f pcast(const Packet4h& a) { - __int64_t a64 = _mm_cvtm64_si64(a.x); - Eigen::half h = raw_uint16_to_half(static_cast(a64)); - float f1 = static_cast(h); - h = raw_uint16_to_half(static_cast(a64 >> 16)); - float f2 = static_cast(h); - h = raw_uint16_to_half(static_cast(a64 >> 32)); - float f3 = static_cast(h); - h = raw_uint16_to_half(static_cast(a64 >> 48)); - float f4 = static_cast(h); - return _mm_set_ps(f4, f3, f2, f1); -} - -template <> -struct type_casting_traits { - enum { - VectorizedCast = 1, - SrcCoeffRatio = 1, - TgtCoeffRatio = 1 - }; -}; -template<> EIGEN_STRONG_INLINE Packet4h pcast(const Packet4f& a) { - EIGEN_ALIGN16 float aux[4]; - pstore(aux, a); - Eigen::half h0(aux[0]); - Eigen::half h1(aux[1]); - Eigen::half h2(aux[2]); - Eigen::half h3(aux[3]); + template<> EIGEN_STRONG_INLINE Packet4f pcast(const Packet4h &a) + { + __int64_t a64 = _mm_cvtm64_si64(a.x); + Eigen::half h = raw_uint16_to_half(static_cast(a64)); + float f1 = static_cast(h); + h = raw_uint16_to_half(static_cast(a64 >> 16)); + float f2 = static_cast(h); + h = raw_uint16_to_half(static_cast(a64 >> 32)); + float f3 = static_cast(h); + h = raw_uint16_to_half(static_cast(a64 >> 48)); + float f4 = static_cast(h); + return _mm_set_ps(f4, f3, f2, f1); + } - Packet4h result; - result.x = _mm_set_pi16(h3.x, h2.x, h1.x, h0.x); - return result; -} + template<> struct type_casting_traits + { + enum { VectorizedCast = 1, SrcCoeffRatio = 1, TgtCoeffRatio = 1 }; + }; + + template<> EIGEN_STRONG_INLINE Packet4h pcast(const Packet4f &a) + { + EIGEN_ALIGN16 float aux[4]; + pstore(aux, a); + Eigen::half h0(aux[0]); + Eigen::half h1(aux[1]); + Eigen::half h2(aux[2]); + Eigen::half h3(aux[3]); + + Packet4h result; + result.x = _mm_set_pi16(h3.x, h2.x, h1.x, h0.x); + return result; + } #endif -} // end namespace internal +}// end namespace internal -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_TYPE_CASTING_CUDA_H +#endif// EIGEN_TYPE_CASTING_CUDA_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/arch/Default/ConjHelper.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/arch/Default/ConjHelper.h index 4cfe34e0..c964bbe5 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/arch/Default/ConjHelper.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/arch/Default/ConjHelper.h @@ -11,19 +11,29 @@ #ifndef EIGEN_ARCH_CONJ_HELPER_H #define EIGEN_ARCH_CONJ_HELPER_H -#define EIGEN_MAKE_CONJ_HELPER_CPLX_REAL(PACKET_CPLX, PACKET_REAL) \ - template<> struct conj_helper { \ - EIGEN_STRONG_INLINE PACKET_CPLX pmadd(const PACKET_REAL& x, const PACKET_CPLX& y, const PACKET_CPLX& c) const \ - { return padd(c, pmul(x,y)); } \ - EIGEN_STRONG_INLINE PACKET_CPLX pmul(const PACKET_REAL& x, const PACKET_CPLX& y) const \ - { return PACKET_CPLX(Eigen::internal::pmul(x, y.v)); } \ +#define EIGEN_MAKE_CONJ_HELPER_CPLX_REAL(PACKET_CPLX, PACKET_REAL) \ + template<> struct conj_helper \ + { \ + EIGEN_STRONG_INLINE PACKET_CPLX pmadd(const PACKET_REAL &x, const PACKET_CPLX &y, const PACKET_CPLX &c) const \ + { \ + return padd(c, pmul(x, y)); \ + } \ + EIGEN_STRONG_INLINE PACKET_CPLX pmul(const PACKET_REAL &x, const PACKET_CPLX &y) const \ + { \ + return PACKET_CPLX(Eigen::internal::pmul(x, y.v)); \ + } \ }; \ \ - template<> struct conj_helper { \ - EIGEN_STRONG_INLINE PACKET_CPLX pmadd(const PACKET_CPLX& x, const PACKET_REAL& y, const PACKET_CPLX& c) const \ - { return padd(c, pmul(x,y)); } \ - EIGEN_STRONG_INLINE PACKET_CPLX pmul(const PACKET_CPLX& x, const PACKET_REAL& y) const \ - { return PACKET_CPLX(Eigen::internal::pmul(x.v, y)); } \ + template<> struct conj_helper \ + { \ + EIGEN_STRONG_INLINE PACKET_CPLX pmadd(const PACKET_CPLX &x, const PACKET_REAL &y, const PACKET_CPLX &c) const \ + { \ + return padd(c, pmul(x, y)); \ + } \ + EIGEN_STRONG_INLINE PACKET_CPLX pmul(const PACKET_CPLX &x, const PACKET_REAL &y) const \ + { \ + return PACKET_CPLX(Eigen::internal::pmul(x.v, y)); \ + } \ }; -#endif // EIGEN_ARCH_CONJ_HELPER_H +#endif// EIGEN_ARCH_CONJ_HELPER_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/arch/Default/Settings.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/arch/Default/Settings.h index 097373c8..23b05917 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/arch/Default/Settings.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/arch/Default/Settings.h @@ -17,33 +17,33 @@ #define EIGEN_DEFAULT_SETTINGS_H /** Defines the maximal loop size to enable meta unrolling of loops. - * Note that the value here is expressed in Eigen's own notion of "number of FLOPS", - * it does not correspond to the number of iterations or the number of instructions - */ + * Note that the value here is expressed in Eigen's own notion of "number of FLOPS", + * it does not correspond to the number of iterations or the number of instructions + */ #ifndef EIGEN_UNROLLING_LIMIT #define EIGEN_UNROLLING_LIMIT 100 #endif /** Defines the threshold between a "small" and a "large" matrix. - * This threshold is mainly used to select the proper product implementation. - */ + * This threshold is mainly used to select the proper product implementation. + */ #ifndef EIGEN_CACHEFRIENDLY_PRODUCT_THRESHOLD #define EIGEN_CACHEFRIENDLY_PRODUCT_THRESHOLD 8 #endif /** Defines the maximal width of the blocks used in the triangular product and solver - * for vectors (level 2 blas xTRMV and xTRSV). The default is 8. - */ + * for vectors (level 2 blas xTRMV and xTRSV). The default is 8. + */ #ifndef EIGEN_TUNE_TRIANGULAR_PANEL_WIDTH #define EIGEN_TUNE_TRIANGULAR_PANEL_WIDTH 8 #endif /** Defines the default number of registers available for that architecture. - * Currently it must be 8 or 16. Other values will fail. - */ + * Currently it must be 8 or 16. Other values will fail. + */ #ifndef EIGEN_ARCH_DEFAULT_NUMBER_OF_REGISTERS #define EIGEN_ARCH_DEFAULT_NUMBER_OF_REGISTERS 8 #endif -#endif // EIGEN_DEFAULT_SETTINGS_H +#endif// EIGEN_DEFAULT_SETTINGS_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/arch/NEON/Complex.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/arch/NEON/Complex.h index 306a309b..6d45c4f1 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/arch/NEON/Complex.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/arch/NEON/Complex.h @@ -15,476 +15,556 @@ namespace Eigen { namespace internal { -inline uint32x4_t p4ui_CONJ_XOR() { + inline uint32x4_t p4ui_CONJ_XOR() + { // See bug 1325, clang fails to call vld1q_u64. #if EIGEN_COMP_CLANG - uint32x4_t ret = { 0x00000000, 0x80000000, 0x00000000, 0x80000000 }; - return ret; + uint32x4_t ret = { 0x00000000, 0x80000000, 0x00000000, 0x80000000 }; + return ret; #else - static const uint32_t conj_XOR_DATA[] = { 0x00000000, 0x80000000, 0x00000000, 0x80000000 }; - return vld1q_u32( conj_XOR_DATA ); + static const uint32_t conj_XOR_DATA[] = { 0x00000000, 0x80000000, 0x00000000, 0x80000000 }; + return vld1q_u32(conj_XOR_DATA); #endif -} - -inline uint32x2_t p2ui_CONJ_XOR() { - static const uint32_t conj_XOR_DATA[] = { 0x00000000, 0x80000000 }; - return vld1_u32( conj_XOR_DATA ); -} - -//---------- float ---------- -struct Packet2cf -{ - EIGEN_STRONG_INLINE Packet2cf() {} - EIGEN_STRONG_INLINE explicit Packet2cf(const Packet4f& a) : v(a) {} - Packet4f v; -}; - -template<> struct packet_traits > : default_packet_traits -{ - typedef Packet2cf type; - typedef Packet2cf half; - enum { - Vectorizable = 1, - AlignedOnScalar = 1, - size = 2, - HasHalfPacket = 0, - - HasAdd = 1, - HasSub = 1, - HasMul = 1, - HasDiv = 1, - HasNegate = 1, - HasAbs = 0, - HasAbs2 = 0, - HasMin = 0, - HasMax = 0, - HasSetLinear = 0 + } + + inline uint32x2_t p2ui_CONJ_XOR() + { + static const uint32_t conj_XOR_DATA[] = { 0x00000000, 0x80000000 }; + return vld1_u32(conj_XOR_DATA); + } + + //---------- float ---------- + struct Packet2cf + { + EIGEN_STRONG_INLINE Packet2cf() {} + EIGEN_STRONG_INLINE explicit Packet2cf(const Packet4f &a) : v(a) {} + Packet4f v; }; -}; - -template<> struct unpacket_traits { typedef std::complex type; enum {size=2, alignment=Aligned16}; typedef Packet2cf half; }; - -template<> EIGEN_STRONG_INLINE Packet2cf pset1(const std::complex& from) -{ - float32x2_t r64; - r64 = vld1_f32((const float *)&from); - - return Packet2cf(vcombine_f32(r64, r64)); -} - -template<> EIGEN_STRONG_INLINE Packet2cf padd(const Packet2cf& a, const Packet2cf& b) { return Packet2cf(padd(a.v,b.v)); } -template<> EIGEN_STRONG_INLINE Packet2cf psub(const Packet2cf& a, const Packet2cf& b) { return Packet2cf(psub(a.v,b.v)); } -template<> EIGEN_STRONG_INLINE Packet2cf pnegate(const Packet2cf& a) { return Packet2cf(pnegate(a.v)); } -template<> EIGEN_STRONG_INLINE Packet2cf pconj(const Packet2cf& a) -{ - Packet4ui b = vreinterpretq_u32_f32(a.v); - return Packet2cf(vreinterpretq_f32_u32(veorq_u32(b, p4ui_CONJ_XOR()))); -} - -template<> EIGEN_STRONG_INLINE Packet2cf pmul(const Packet2cf& a, const Packet2cf& b) -{ - Packet4f v1, v2; - - // Get the real values of a | a1_re | a1_re | a2_re | a2_re | - v1 = vcombine_f32(vdup_lane_f32(vget_low_f32(a.v), 0), vdup_lane_f32(vget_high_f32(a.v), 0)); - // Get the imag values of a | a1_im | a1_im | a2_im | a2_im | - v2 = vcombine_f32(vdup_lane_f32(vget_low_f32(a.v), 1), vdup_lane_f32(vget_high_f32(a.v), 1)); - // Multiply the real a with b - v1 = vmulq_f32(v1, b.v); - // Multiply the imag a with b - v2 = vmulq_f32(v2, b.v); - // Conjugate v2 - v2 = vreinterpretq_f32_u32(veorq_u32(vreinterpretq_u32_f32(v2), p4ui_CONJ_XOR())); - // Swap real/imag elements in v2. - v2 = vrev64q_f32(v2); - // Add and return the result - return Packet2cf(vaddq_f32(v1, v2)); -} - -template<> EIGEN_STRONG_INLINE Packet2cf pand (const Packet2cf& a, const Packet2cf& b) -{ - return Packet2cf(vreinterpretq_f32_u32(vandq_u32(vreinterpretq_u32_f32(a.v),vreinterpretq_u32_f32(b.v)))); -} -template<> EIGEN_STRONG_INLINE Packet2cf por (const Packet2cf& a, const Packet2cf& b) -{ - return Packet2cf(vreinterpretq_f32_u32(vorrq_u32(vreinterpretq_u32_f32(a.v),vreinterpretq_u32_f32(b.v)))); -} -template<> EIGEN_STRONG_INLINE Packet2cf pxor (const Packet2cf& a, const Packet2cf& b) -{ - return Packet2cf(vreinterpretq_f32_u32(veorq_u32(vreinterpretq_u32_f32(a.v),vreinterpretq_u32_f32(b.v)))); -} -template<> EIGEN_STRONG_INLINE Packet2cf pandnot(const Packet2cf& a, const Packet2cf& b) -{ - return Packet2cf(vreinterpretq_f32_u32(vbicq_u32(vreinterpretq_u32_f32(a.v),vreinterpretq_u32_f32(b.v)))); -} - -template<> EIGEN_STRONG_INLINE Packet2cf pload(const std::complex* from) { EIGEN_DEBUG_ALIGNED_LOAD return Packet2cf(pload((const float*)from)); } -template<> EIGEN_STRONG_INLINE Packet2cf ploadu(const std::complex* from) { EIGEN_DEBUG_UNALIGNED_LOAD return Packet2cf(ploadu((const float*)from)); } - -template<> EIGEN_STRONG_INLINE Packet2cf ploaddup(const std::complex* from) { return pset1(*from); } - -template<> EIGEN_STRONG_INLINE void pstore >(std::complex * to, const Packet2cf& from) { EIGEN_DEBUG_ALIGNED_STORE pstore((float*)to, from.v); } -template<> EIGEN_STRONG_INLINE void pstoreu >(std::complex * to, const Packet2cf& from) { EIGEN_DEBUG_UNALIGNED_STORE pstoreu((float*)to, from.v); } - -template<> EIGEN_DEVICE_FUNC inline Packet2cf pgather, Packet2cf>(const std::complex* from, Index stride) -{ - Packet4f res = pset1(0.f); - res = vsetq_lane_f32(std::real(from[0*stride]), res, 0); - res = vsetq_lane_f32(std::imag(from[0*stride]), res, 1); - res = vsetq_lane_f32(std::real(from[1*stride]), res, 2); - res = vsetq_lane_f32(std::imag(from[1*stride]), res, 3); - return Packet2cf(res); -} - -template<> EIGEN_DEVICE_FUNC inline void pscatter, Packet2cf>(std::complex* to, const Packet2cf& from, Index stride) -{ - to[stride*0] = std::complex(vgetq_lane_f32(from.v, 0), vgetq_lane_f32(from.v, 1)); - to[stride*1] = std::complex(vgetq_lane_f32(from.v, 2), vgetq_lane_f32(from.v, 3)); -} - -template<> EIGEN_STRONG_INLINE void prefetch >(const std::complex * addr) { EIGEN_ARM_PREFETCH((const float *)addr); } - -template<> EIGEN_STRONG_INLINE std::complex pfirst(const Packet2cf& a) -{ - std::complex EIGEN_ALIGN16 x[2]; - vst1q_f32((float *)x, a.v); - return x[0]; -} - -template<> EIGEN_STRONG_INLINE Packet2cf preverse(const Packet2cf& a) -{ - float32x2_t a_lo, a_hi; - Packet4f a_r128; - - a_lo = vget_low_f32(a.v); - a_hi = vget_high_f32(a.v); - a_r128 = vcombine_f32(a_hi, a_lo); - - return Packet2cf(a_r128); -} - -template<> EIGEN_STRONG_INLINE Packet2cf pcplxflip(const Packet2cf& a) -{ - return Packet2cf(vrev64q_f32(a.v)); -} - -template<> EIGEN_STRONG_INLINE std::complex predux(const Packet2cf& a) -{ - float32x2_t a1, a2; - std::complex s; - - a1 = vget_low_f32(a.v); - a2 = vget_high_f32(a.v); - a2 = vadd_f32(a1, a2); - vst1_f32((float *)&s, a2); - - return s; -} - -template<> EIGEN_STRONG_INLINE Packet2cf preduxp(const Packet2cf* vecs) -{ - Packet4f sum1, sum2, sum; - - // Add the first two 64-bit float32x2_t of vecs[0] - sum1 = vcombine_f32(vget_low_f32(vecs[0].v), vget_low_f32(vecs[1].v)); - sum2 = vcombine_f32(vget_high_f32(vecs[0].v), vget_high_f32(vecs[1].v)); - sum = vaddq_f32(sum1, sum2); - - return Packet2cf(sum); -} - -template<> EIGEN_STRONG_INLINE std::complex predux_mul(const Packet2cf& a) -{ - float32x2_t a1, a2, v1, v2, prod; - std::complex s; - - a1 = vget_low_f32(a.v); - a2 = vget_high_f32(a.v); - // Get the real values of a | a1_re | a1_re | a2_re | a2_re | - v1 = vdup_lane_f32(a1, 0); - // Get the real values of a | a1_im | a1_im | a2_im | a2_im | - v2 = vdup_lane_f32(a1, 1); - // Multiply the real a with b - v1 = vmul_f32(v1, a2); - // Multiply the imag a with b - v2 = vmul_f32(v2, a2); - // Conjugate v2 - v2 = vreinterpret_f32_u32(veor_u32(vreinterpret_u32_f32(v2), p2ui_CONJ_XOR())); - // Swap real/imag elements in v2. - v2 = vrev64_f32(v2); - // Add v1, v2 - prod = vadd_f32(v1, v2); - - vst1_f32((float *)&s, prod); - - return s; -} - -template -struct palign_impl -{ - EIGEN_STRONG_INLINE static void run(Packet2cf& first, const Packet2cf& second) - { - if (Offset==1) - { - first.v = vextq_f32(first.v, second.v, 2); - } + + template<> struct packet_traits> : default_packet_traits + { + typedef Packet2cf type; + typedef Packet2cf half; + enum { + Vectorizable = 1, + AlignedOnScalar = 1, + size = 2, + HasHalfPacket = 0, + + HasAdd = 1, + HasSub = 1, + HasMul = 1, + HasDiv = 1, + HasNegate = 1, + HasAbs = 0, + HasAbs2 = 0, + HasMin = 0, + HasMax = 0, + HasSetLinear = 0 + }; + }; + + template<> struct unpacket_traits + { + typedef std::complex type; + enum { size = 2, alignment = Aligned16 }; + typedef Packet2cf half; + }; + + template<> EIGEN_STRONG_INLINE Packet2cf pset1(const std::complex &from) + { + float32x2_t r64; + r64 = vld1_f32((const float *)&from); + + return Packet2cf(vcombine_f32(r64, r64)); } -}; -template<> struct conj_helper -{ - EIGEN_STRONG_INLINE Packet2cf pmadd(const Packet2cf& x, const Packet2cf& y, const Packet2cf& c) const - { return padd(pmul(x,y),c); } + template<> EIGEN_STRONG_INLINE Packet2cf padd(const Packet2cf &a, const Packet2cf &b) + { + return Packet2cf(padd(a.v, b.v)); + } + template<> EIGEN_STRONG_INLINE Packet2cf psub(const Packet2cf &a, const Packet2cf &b) + { + return Packet2cf(psub(a.v, b.v)); + } + template<> EIGEN_STRONG_INLINE Packet2cf pnegate(const Packet2cf &a) { return Packet2cf(pnegate(a.v)); } + template<> EIGEN_STRONG_INLINE Packet2cf pconj(const Packet2cf &a) + { + Packet4ui b = vreinterpretq_u32_f32(a.v); + return Packet2cf(vreinterpretq_f32_u32(veorq_u32(b, p4ui_CONJ_XOR()))); + } - EIGEN_STRONG_INLINE Packet2cf pmul(const Packet2cf& a, const Packet2cf& b) const + template<> EIGEN_STRONG_INLINE Packet2cf pmul(const Packet2cf &a, const Packet2cf &b) { - return internal::pmul(a, pconj(b)); + Packet4f v1, v2; + + // Get the real values of a | a1_re | a1_re | a2_re | a2_re | + v1 = vcombine_f32(vdup_lane_f32(vget_low_f32(a.v), 0), vdup_lane_f32(vget_high_f32(a.v), 0)); + // Get the imag values of a | a1_im | a1_im | a2_im | a2_im | + v2 = vcombine_f32(vdup_lane_f32(vget_low_f32(a.v), 1), vdup_lane_f32(vget_high_f32(a.v), 1)); + // Multiply the real a with b + v1 = vmulq_f32(v1, b.v); + // Multiply the imag a with b + v2 = vmulq_f32(v2, b.v); + // Conjugate v2 + v2 = vreinterpretq_f32_u32(veorq_u32(vreinterpretq_u32_f32(v2), p4ui_CONJ_XOR())); + // Swap real/imag elements in v2. + v2 = vrev64q_f32(v2); + // Add and return the result + return Packet2cf(vaddq_f32(v1, v2)); } -}; -template<> struct conj_helper -{ - EIGEN_STRONG_INLINE Packet2cf pmadd(const Packet2cf& x, const Packet2cf& y, const Packet2cf& c) const - { return padd(pmul(x,y),c); } + template<> EIGEN_STRONG_INLINE Packet2cf pand(const Packet2cf &a, const Packet2cf &b) + { + return Packet2cf(vreinterpretq_f32_u32(vandq_u32(vreinterpretq_u32_f32(a.v), vreinterpretq_u32_f32(b.v)))); + } + template<> EIGEN_STRONG_INLINE Packet2cf por(const Packet2cf &a, const Packet2cf &b) + { + return Packet2cf(vreinterpretq_f32_u32(vorrq_u32(vreinterpretq_u32_f32(a.v), vreinterpretq_u32_f32(b.v)))); + } + template<> EIGEN_STRONG_INLINE Packet2cf pxor(const Packet2cf &a, const Packet2cf &b) + { + return Packet2cf(vreinterpretq_f32_u32(veorq_u32(vreinterpretq_u32_f32(a.v), vreinterpretq_u32_f32(b.v)))); + } + template<> EIGEN_STRONG_INLINE Packet2cf pandnot(const Packet2cf &a, const Packet2cf &b) + { + return Packet2cf(vreinterpretq_f32_u32(vbicq_u32(vreinterpretq_u32_f32(a.v), vreinterpretq_u32_f32(b.v)))); + } - EIGEN_STRONG_INLINE Packet2cf pmul(const Packet2cf& a, const Packet2cf& b) const + template<> EIGEN_STRONG_INLINE Packet2cf pload(const std::complex *from) + { + EIGEN_DEBUG_ALIGNED_LOAD return Packet2cf(pload((const float *)from)); + } + template<> EIGEN_STRONG_INLINE Packet2cf ploadu(const std::complex *from) { - return internal::pmul(pconj(a), b); + EIGEN_DEBUG_UNALIGNED_LOAD return Packet2cf(ploadu((const float *)from)); } -}; -template<> struct conj_helper -{ - EIGEN_STRONG_INLINE Packet2cf pmadd(const Packet2cf& x, const Packet2cf& y, const Packet2cf& c) const - { return padd(pmul(x,y),c); } + template<> EIGEN_STRONG_INLINE Packet2cf ploaddup(const std::complex *from) + { + return pset1(*from); + } - EIGEN_STRONG_INLINE Packet2cf pmul(const Packet2cf& a, const Packet2cf& b) const + template<> EIGEN_STRONG_INLINE void pstore>(std::complex *to, const Packet2cf &from) + { + EIGEN_DEBUG_ALIGNED_STORE pstore((float *)to, from.v); + } + template<> EIGEN_STRONG_INLINE void pstoreu>(std::complex *to, const Packet2cf &from) { - return pconj(internal::pmul(a, b)); + EIGEN_DEBUG_UNALIGNED_STORE pstoreu((float *)to, from.v); } -}; -EIGEN_MAKE_CONJ_HELPER_CPLX_REAL(Packet2cf,Packet4f) + template<> + EIGEN_DEVICE_FUNC inline Packet2cf pgather, Packet2cf>(const std::complex *from, + Index stride) + { + Packet4f res = pset1(0.f); + res = vsetq_lane_f32(std::real(from[0 * stride]), res, 0); + res = vsetq_lane_f32(std::imag(from[0 * stride]), res, 1); + res = vsetq_lane_f32(std::real(from[1 * stride]), res, 2); + res = vsetq_lane_f32(std::imag(from[1 * stride]), res, 3); + return Packet2cf(res); + } -template<> EIGEN_STRONG_INLINE Packet2cf pdiv(const Packet2cf& a, const Packet2cf& b) -{ - // TODO optimize it for NEON - Packet2cf res = conj_helper().pmul(a,b); - Packet4f s, rev_s; + template<> + EIGEN_DEVICE_FUNC inline void + pscatter, Packet2cf>(std::complex *to, const Packet2cf &from, Index stride) + { + to[stride * 0] = std::complex(vgetq_lane_f32(from.v, 0), vgetq_lane_f32(from.v, 1)); + to[stride * 1] = std::complex(vgetq_lane_f32(from.v, 2), vgetq_lane_f32(from.v, 3)); + } - // this computes the norm - s = vmulq_f32(b.v, b.v); - rev_s = vrev64q_f32(s); + template<> EIGEN_STRONG_INLINE void prefetch>(const std::complex *addr) + { + EIGEN_ARM_PREFETCH((const float *)addr); + } - return Packet2cf(pdiv(res.v, vaddq_f32(s,rev_s))); -} + template<> EIGEN_STRONG_INLINE std::complex pfirst(const Packet2cf &a) + { + std::complex EIGEN_ALIGN16 x[2]; + vst1q_f32((float *)x, a.v); + return x[0]; + } + + template<> EIGEN_STRONG_INLINE Packet2cf preverse(const Packet2cf &a) + { + float32x2_t a_lo, a_hi; + Packet4f a_r128; + + a_lo = vget_low_f32(a.v); + a_hi = vget_high_f32(a.v); + a_r128 = vcombine_f32(a_hi, a_lo); + + return Packet2cf(a_r128); + } + + template<> EIGEN_STRONG_INLINE Packet2cf pcplxflip(const Packet2cf &a) + { + return Packet2cf(vrev64q_f32(a.v)); + } + + template<> EIGEN_STRONG_INLINE std::complex predux(const Packet2cf &a) + { + float32x2_t a1, a2; + std::complex s; + + a1 = vget_low_f32(a.v); + a2 = vget_high_f32(a.v); + a2 = vadd_f32(a1, a2); + vst1_f32((float *)&s, a2); + + return s; + } + + template<> EIGEN_STRONG_INLINE Packet2cf preduxp(const Packet2cf *vecs) + { + Packet4f sum1, sum2, sum; + + // Add the first two 64-bit float32x2_t of vecs[0] + sum1 = vcombine_f32(vget_low_f32(vecs[0].v), vget_low_f32(vecs[1].v)); + sum2 = vcombine_f32(vget_high_f32(vecs[0].v), vget_high_f32(vecs[1].v)); + sum = vaddq_f32(sum1, sum2); + + return Packet2cf(sum); + } + + template<> EIGEN_STRONG_INLINE std::complex predux_mul(const Packet2cf &a) + { + float32x2_t a1, a2, v1, v2, prod; + std::complex s; + + a1 = vget_low_f32(a.v); + a2 = vget_high_f32(a.v); + // Get the real values of a | a1_re | a1_re | a2_re | a2_re | + v1 = vdup_lane_f32(a1, 0); + // Get the real values of a | a1_im | a1_im | a2_im | a2_im | + v2 = vdup_lane_f32(a1, 1); + // Multiply the real a with b + v1 = vmul_f32(v1, a2); + // Multiply the imag a with b + v2 = vmul_f32(v2, a2); + // Conjugate v2 + v2 = vreinterpret_f32_u32(veor_u32(vreinterpret_u32_f32(v2), p2ui_CONJ_XOR())); + // Swap real/imag elements in v2. + v2 = vrev64_f32(v2); + // Add v1, v2 + prod = vadd_f32(v1, v2); + + vst1_f32((float *)&s, prod); + + return s; + } -EIGEN_DEVICE_FUNC inline void -ptranspose(PacketBlock& kernel) { - Packet4f tmp = vcombine_f32(vget_high_f32(kernel.packet[0].v), vget_high_f32(kernel.packet[1].v)); - kernel.packet[0].v = vcombine_f32(vget_low_f32(kernel.packet[0].v), vget_low_f32(kernel.packet[1].v)); - kernel.packet[1].v = tmp; -} + template struct palign_impl + { + EIGEN_STRONG_INLINE static void run(Packet2cf &first, const Packet2cf &second) + { + if (Offset == 1) { first.v = vextq_f32(first.v, second.v, 2); } + } + }; + + template<> struct conj_helper + { + EIGEN_STRONG_INLINE Packet2cf pmadd(const Packet2cf &x, const Packet2cf &y, const Packet2cf &c) const + { + return padd(pmul(x, y), c); + } + + EIGEN_STRONG_INLINE Packet2cf pmul(const Packet2cf &a, const Packet2cf &b) const + { + return internal::pmul(a, pconj(b)); + } + }; + + template<> struct conj_helper + { + EIGEN_STRONG_INLINE Packet2cf pmadd(const Packet2cf &x, const Packet2cf &y, const Packet2cf &c) const + { + return padd(pmul(x, y), c); + } + + EIGEN_STRONG_INLINE Packet2cf pmul(const Packet2cf &a, const Packet2cf &b) const + { + return internal::pmul(pconj(a), b); + } + }; + + template<> struct conj_helper + { + EIGEN_STRONG_INLINE Packet2cf pmadd(const Packet2cf &x, const Packet2cf &y, const Packet2cf &c) const + { + return padd(pmul(x, y), c); + } + + EIGEN_STRONG_INLINE Packet2cf pmul(const Packet2cf &a, const Packet2cf &b) const + { + return pconj(internal::pmul(a, b)); + } + }; + + EIGEN_MAKE_CONJ_HELPER_CPLX_REAL(Packet2cf, Packet4f) + + template<> EIGEN_STRONG_INLINE Packet2cf pdiv(const Packet2cf &a, const Packet2cf &b) + { + // TODO optimize it for NEON + Packet2cf res = conj_helper().pmul(a, b); + Packet4f s, rev_s; + + // this computes the norm + s = vmulq_f32(b.v, b.v); + rev_s = vrev64q_f32(s); + + return Packet2cf(pdiv(res.v, vaddq_f32(s, rev_s))); + } + + EIGEN_DEVICE_FUNC inline void ptranspose(PacketBlock &kernel) + { + Packet4f tmp = vcombine_f32(vget_high_f32(kernel.packet[0].v), vget_high_f32(kernel.packet[1].v)); + kernel.packet[0].v = vcombine_f32(vget_low_f32(kernel.packet[0].v), vget_low_f32(kernel.packet[1].v)); + kernel.packet[1].v = tmp; + } //---------- double ---------- #if EIGEN_ARCH_ARM64 && !EIGEN_APPLE_DOUBLE_NEON_BUG // See bug 1325, clang fails to call vld1q_u64. #if EIGEN_COMP_CLANG - static uint64x2_t p2ul_CONJ_XOR = {0x0, 0x8000000000000000}; + static uint64x2_t p2ul_CONJ_XOR = { 0x0, 0x8000000000000000 }; #else - const uint64_t p2ul_conj_XOR_DATA[] = { 0x0, 0x8000000000000000 }; - static uint64x2_t p2ul_CONJ_XOR = vld1q_u64( p2ul_conj_XOR_DATA ); + const uint64_t p2ul_conj_XOR_DATA[] = { 0x0, 0x8000000000000000 }; + static uint64x2_t p2ul_CONJ_XOR = vld1q_u64(p2ul_conj_XOR_DATA); #endif -struct Packet1cd -{ - EIGEN_STRONG_INLINE Packet1cd() {} - EIGEN_STRONG_INLINE explicit Packet1cd(const Packet2d& a) : v(a) {} - Packet2d v; -}; - -template<> struct packet_traits > : default_packet_traits -{ - typedef Packet1cd type; - typedef Packet1cd half; - enum { - Vectorizable = 1, - AlignedOnScalar = 0, - size = 1, - HasHalfPacket = 0, - - HasAdd = 1, - HasSub = 1, - HasMul = 1, - HasDiv = 1, - HasNegate = 1, - HasAbs = 0, - HasAbs2 = 0, - HasMin = 0, - HasMax = 0, - HasSetLinear = 0 + struct Packet1cd + { + EIGEN_STRONG_INLINE Packet1cd() {} + EIGEN_STRONG_INLINE explicit Packet1cd(const Packet2d &a) : v(a) {} + Packet2d v; + }; + + template<> struct packet_traits> : default_packet_traits + { + typedef Packet1cd type; + typedef Packet1cd half; + enum { + Vectorizable = 1, + AlignedOnScalar = 0, + size = 1, + HasHalfPacket = 0, + + HasAdd = 1, + HasSub = 1, + HasMul = 1, + HasDiv = 1, + HasNegate = 1, + HasAbs = 0, + HasAbs2 = 0, + HasMin = 0, + HasMax = 0, + HasSetLinear = 0 + }; + }; + + template<> struct unpacket_traits + { + typedef std::complex type; + enum { size = 1, alignment = Aligned16 }; + typedef Packet1cd half; }; -}; - -template<> struct unpacket_traits { typedef std::complex type; enum {size=1, alignment=Aligned16}; typedef Packet1cd half; }; - -template<> EIGEN_STRONG_INLINE Packet1cd pload(const std::complex* from) { EIGEN_DEBUG_ALIGNED_LOAD return Packet1cd(pload((const double*)from)); } -template<> EIGEN_STRONG_INLINE Packet1cd ploadu(const std::complex* from) { EIGEN_DEBUG_UNALIGNED_LOAD return Packet1cd(ploadu((const double*)from)); } - -template<> EIGEN_STRONG_INLINE Packet1cd pset1(const std::complex& from) -{ /* here we really have to use unaligned loads :( */ return ploadu(&from); } -template<> EIGEN_STRONG_INLINE Packet1cd padd(const Packet1cd& a, const Packet1cd& b) { return Packet1cd(padd(a.v,b.v)); } -template<> EIGEN_STRONG_INLINE Packet1cd psub(const Packet1cd& a, const Packet1cd& b) { return Packet1cd(psub(a.v,b.v)); } -template<> EIGEN_STRONG_INLINE Packet1cd pnegate(const Packet1cd& a) { return Packet1cd(pnegate(a.v)); } -template<> EIGEN_STRONG_INLINE Packet1cd pconj(const Packet1cd& a) { return Packet1cd(vreinterpretq_f64_u64(veorq_u64(vreinterpretq_u64_f64(a.v), p2ul_CONJ_XOR))); } + template<> EIGEN_STRONG_INLINE Packet1cd pload(const std::complex *from) + { + EIGEN_DEBUG_ALIGNED_LOAD return Packet1cd(pload((const double *)from)); + } + template<> EIGEN_STRONG_INLINE Packet1cd ploadu(const std::complex *from) + { + EIGEN_DEBUG_UNALIGNED_LOAD return Packet1cd(ploadu((const double *)from)); + } + + template<> EIGEN_STRONG_INLINE Packet1cd pset1(const std::complex &from) + { /* here we really have to use unaligned loads :( */ + return ploadu(&from); + } -template<> EIGEN_STRONG_INLINE Packet1cd pmul(const Packet1cd& a, const Packet1cd& b) -{ - Packet2d v1, v2; + template<> EIGEN_STRONG_INLINE Packet1cd padd(const Packet1cd &a, const Packet1cd &b) + { + return Packet1cd(padd(a.v, b.v)); + } + template<> EIGEN_STRONG_INLINE Packet1cd psub(const Packet1cd &a, const Packet1cd &b) + { + return Packet1cd(psub(a.v, b.v)); + } + template<> EIGEN_STRONG_INLINE Packet1cd pnegate(const Packet1cd &a) { return Packet1cd(pnegate(a.v)); } + template<> EIGEN_STRONG_INLINE Packet1cd pconj(const Packet1cd &a) + { + return Packet1cd(vreinterpretq_f64_u64(veorq_u64(vreinterpretq_u64_f64(a.v), p2ul_CONJ_XOR))); + } - // Get the real values of a - v1 = vdupq_lane_f64(vget_low_f64(a.v), 0); - // Get the imag values of a - v2 = vdupq_lane_f64(vget_high_f64(a.v), 0); - // Multiply the real a with b - v1 = vmulq_f64(v1, b.v); - // Multiply the imag a with b - v2 = vmulq_f64(v2, b.v); - // Conjugate v2 - v2 = vreinterpretq_f64_u64(veorq_u64(vreinterpretq_u64_f64(v2), p2ul_CONJ_XOR)); - // Swap real/imag elements in v2. - v2 = preverse(v2); - // Add and return the result - return Packet1cd(vaddq_f64(v1, v2)); -} + template<> EIGEN_STRONG_INLINE Packet1cd pmul(const Packet1cd &a, const Packet1cd &b) + { + Packet2d v1, v2; + + // Get the real values of a + v1 = vdupq_lane_f64(vget_low_f64(a.v), 0); + // Get the imag values of a + v2 = vdupq_lane_f64(vget_high_f64(a.v), 0); + // Multiply the real a with b + v1 = vmulq_f64(v1, b.v); + // Multiply the imag a with b + v2 = vmulq_f64(v2, b.v); + // Conjugate v2 + v2 = vreinterpretq_f64_u64(veorq_u64(vreinterpretq_u64_f64(v2), p2ul_CONJ_XOR)); + // Swap real/imag elements in v2. + v2 = preverse(v2); + // Add and return the result + return Packet1cd(vaddq_f64(v1, v2)); + } -template<> EIGEN_STRONG_INLINE Packet1cd pand (const Packet1cd& a, const Packet1cd& b) -{ - return Packet1cd(vreinterpretq_f64_u64(vandq_u64(vreinterpretq_u64_f64(a.v),vreinterpretq_u64_f64(b.v)))); -} -template<> EIGEN_STRONG_INLINE Packet1cd por (const Packet1cd& a, const Packet1cd& b) -{ - return Packet1cd(vreinterpretq_f64_u64(vorrq_u64(vreinterpretq_u64_f64(a.v),vreinterpretq_u64_f64(b.v)))); -} -template<> EIGEN_STRONG_INLINE Packet1cd pxor (const Packet1cd& a, const Packet1cd& b) -{ - return Packet1cd(vreinterpretq_f64_u64(veorq_u64(vreinterpretq_u64_f64(a.v),vreinterpretq_u64_f64(b.v)))); -} -template<> EIGEN_STRONG_INLINE Packet1cd pandnot(const Packet1cd& a, const Packet1cd& b) -{ - return Packet1cd(vreinterpretq_f64_u64(vbicq_u64(vreinterpretq_u64_f64(a.v),vreinterpretq_u64_f64(b.v)))); -} + template<> EIGEN_STRONG_INLINE Packet1cd pand(const Packet1cd &a, const Packet1cd &b) + { + return Packet1cd(vreinterpretq_f64_u64(vandq_u64(vreinterpretq_u64_f64(a.v), vreinterpretq_u64_f64(b.v)))); + } + template<> EIGEN_STRONG_INLINE Packet1cd por(const Packet1cd &a, const Packet1cd &b) + { + return Packet1cd(vreinterpretq_f64_u64(vorrq_u64(vreinterpretq_u64_f64(a.v), vreinterpretq_u64_f64(b.v)))); + } + template<> EIGEN_STRONG_INLINE Packet1cd pxor(const Packet1cd &a, const Packet1cd &b) + { + return Packet1cd(vreinterpretq_f64_u64(veorq_u64(vreinterpretq_u64_f64(a.v), vreinterpretq_u64_f64(b.v)))); + } + template<> EIGEN_STRONG_INLINE Packet1cd pandnot(const Packet1cd &a, const Packet1cd &b) + { + return Packet1cd(vreinterpretq_f64_u64(vbicq_u64(vreinterpretq_u64_f64(a.v), vreinterpretq_u64_f64(b.v)))); + } -template<> EIGEN_STRONG_INLINE Packet1cd ploaddup(const std::complex* from) { return pset1(*from); } + template<> EIGEN_STRONG_INLINE Packet1cd ploaddup(const std::complex *from) + { + return pset1(*from); + } -template<> EIGEN_STRONG_INLINE void pstore >(std::complex * to, const Packet1cd& from) { EIGEN_DEBUG_ALIGNED_STORE pstore((double*)to, from.v); } -template<> EIGEN_STRONG_INLINE void pstoreu >(std::complex * to, const Packet1cd& from) { EIGEN_DEBUG_UNALIGNED_STORE pstoreu((double*)to, from.v); } + template<> EIGEN_STRONG_INLINE void pstore>(std::complex *to, const Packet1cd &from) + { + EIGEN_DEBUG_ALIGNED_STORE pstore((double *)to, from.v); + } + template<> EIGEN_STRONG_INLINE void pstoreu>(std::complex *to, const Packet1cd &from) + { + EIGEN_DEBUG_UNALIGNED_STORE pstoreu((double *)to, from.v); + } -template<> EIGEN_STRONG_INLINE void prefetch >(const std::complex * addr) { EIGEN_ARM_PREFETCH((const double *)addr); } + template<> EIGEN_STRONG_INLINE void prefetch>(const std::complex *addr) + { + EIGEN_ARM_PREFETCH((const double *)addr); + } -template<> EIGEN_DEVICE_FUNC inline Packet1cd pgather, Packet1cd>(const std::complex* from, Index stride) -{ - Packet2d res = pset1(0.0); - res = vsetq_lane_f64(std::real(from[0*stride]), res, 0); - res = vsetq_lane_f64(std::imag(from[0*stride]), res, 1); - return Packet1cd(res); -} + template<> + EIGEN_DEVICE_FUNC inline Packet1cd pgather, Packet1cd>(const std::complex *from, + Index stride) + { + Packet2d res = pset1(0.0); + res = vsetq_lane_f64(std::real(from[0 * stride]), res, 0); + res = vsetq_lane_f64(std::imag(from[0 * stride]), res, 1); + return Packet1cd(res); + } -template<> EIGEN_DEVICE_FUNC inline void pscatter, Packet1cd>(std::complex* to, const Packet1cd& from, Index stride) -{ - to[stride*0] = std::complex(vgetq_lane_f64(from.v, 0), vgetq_lane_f64(from.v, 1)); -} + template<> + EIGEN_DEVICE_FUNC inline void + pscatter, Packet1cd>(std::complex *to, const Packet1cd &from, Index stride) + { + to[stride * 0] = std::complex(vgetq_lane_f64(from.v, 0), vgetq_lane_f64(from.v, 1)); + } -template<> EIGEN_STRONG_INLINE std::complex pfirst(const Packet1cd& a) -{ - std::complex EIGEN_ALIGN16 res; - pstore >(&res, a); + template<> EIGEN_STRONG_INLINE std::complex pfirst(const Packet1cd &a) + { + std::complex EIGEN_ALIGN16 res; + pstore>(&res, a); - return res; -} + return res; + } -template<> EIGEN_STRONG_INLINE Packet1cd preverse(const Packet1cd& a) { return a; } + template<> EIGEN_STRONG_INLINE Packet1cd preverse(const Packet1cd &a) { return a; } -template<> EIGEN_STRONG_INLINE std::complex predux(const Packet1cd& a) { return pfirst(a); } + template<> EIGEN_STRONG_INLINE std::complex predux(const Packet1cd &a) { return pfirst(a); } -template<> EIGEN_STRONG_INLINE Packet1cd preduxp(const Packet1cd* vecs) { return vecs[0]; } + template<> EIGEN_STRONG_INLINE Packet1cd preduxp(const Packet1cd *vecs) { return vecs[0]; } -template<> EIGEN_STRONG_INLINE std::complex predux_mul(const Packet1cd& a) { return pfirst(a); } + template<> EIGEN_STRONG_INLINE std::complex predux_mul(const Packet1cd &a) { return pfirst(a); } -template -struct palign_impl -{ - static EIGEN_STRONG_INLINE void run(Packet1cd& /*first*/, const Packet1cd& /*second*/) + template struct palign_impl { - // FIXME is it sure we never have to align a Packet1cd? - // Even though a std::complex has 16 bytes, it is not necessarily aligned on a 16 bytes boundary... - } -}; - -template<> struct conj_helper -{ - EIGEN_STRONG_INLINE Packet1cd pmadd(const Packet1cd& x, const Packet1cd& y, const Packet1cd& c) const - { return padd(pmul(x,y),c); } + static EIGEN_STRONG_INLINE void run(Packet1cd & /*first*/, const Packet1cd & /*second*/) + { + // FIXME is it sure we never have to align a Packet1cd? + // Even though a std::complex has 16 bytes, it is not necessarily aligned on a 16 bytes boundary... + } + }; - EIGEN_STRONG_INLINE Packet1cd pmul(const Packet1cd& a, const Packet1cd& b) const + template<> struct conj_helper { - return internal::pmul(a, pconj(b)); - } -}; + EIGEN_STRONG_INLINE Packet1cd pmadd(const Packet1cd &x, const Packet1cd &y, const Packet1cd &c) const + { + return padd(pmul(x, y), c); + } -template<> struct conj_helper -{ - EIGEN_STRONG_INLINE Packet1cd pmadd(const Packet1cd& x, const Packet1cd& y, const Packet1cd& c) const - { return padd(pmul(x,y),c); } + EIGEN_STRONG_INLINE Packet1cd pmul(const Packet1cd &a, const Packet1cd &b) const + { + return internal::pmul(a, pconj(b)); + } + }; - EIGEN_STRONG_INLINE Packet1cd pmul(const Packet1cd& a, const Packet1cd& b) const + template<> struct conj_helper { - return internal::pmul(pconj(a), b); - } -}; + EIGEN_STRONG_INLINE Packet1cd pmadd(const Packet1cd &x, const Packet1cd &y, const Packet1cd &c) const + { + return padd(pmul(x, y), c); + } -template<> struct conj_helper -{ - EIGEN_STRONG_INLINE Packet1cd pmadd(const Packet1cd& x, const Packet1cd& y, const Packet1cd& c) const - { return padd(pmul(x,y),c); } + EIGEN_STRONG_INLINE Packet1cd pmul(const Packet1cd &a, const Packet1cd &b) const + { + return internal::pmul(pconj(a), b); + } + }; - EIGEN_STRONG_INLINE Packet1cd pmul(const Packet1cd& a, const Packet1cd& b) const + template<> struct conj_helper { - return pconj(internal::pmul(a, b)); - } -}; + EIGEN_STRONG_INLINE Packet1cd pmadd(const Packet1cd &x, const Packet1cd &y, const Packet1cd &c) const + { + return padd(pmul(x, y), c); + } -EIGEN_MAKE_CONJ_HELPER_CPLX_REAL(Packet1cd,Packet2d) + EIGEN_STRONG_INLINE Packet1cd pmul(const Packet1cd &a, const Packet1cd &b) const + { + return pconj(internal::pmul(a, b)); + } + }; -template<> EIGEN_STRONG_INLINE Packet1cd pdiv(const Packet1cd& a, const Packet1cd& b) -{ - // TODO optimize it for NEON - Packet1cd res = conj_helper().pmul(a,b); - Packet2d s = pmul(b.v, b.v); - Packet2d rev_s = preverse(s); + EIGEN_MAKE_CONJ_HELPER_CPLX_REAL(Packet1cd, Packet2d) - return Packet1cd(pdiv(res.v, padd(s,rev_s))); -} + template<> EIGEN_STRONG_INLINE Packet1cd pdiv(const Packet1cd &a, const Packet1cd &b) + { + // TODO optimize it for NEON + Packet1cd res = conj_helper().pmul(a, b); + Packet2d s = pmul(b.v, b.v); + Packet2d rev_s = preverse(s); -EIGEN_STRONG_INLINE Packet1cd pcplxflip/**/(const Packet1cd& x) -{ - return Packet1cd(preverse(Packet2d(x.v))); -} + return Packet1cd(pdiv(res.v, padd(s, rev_s))); + } -EIGEN_STRONG_INLINE void ptranspose(PacketBlock& kernel) -{ - Packet2d tmp = vcombine_f64(vget_high_f64(kernel.packet[0].v), vget_high_f64(kernel.packet[1].v)); - kernel.packet[0].v = vcombine_f64(vget_low_f64(kernel.packet[0].v), vget_low_f64(kernel.packet[1].v)); - kernel.packet[1].v = tmp; -} -#endif // EIGEN_ARCH_ARM64 + EIGEN_STRONG_INLINE Packet1cd pcplxflip /**/ (const Packet1cd &x) + { + return Packet1cd(preverse(Packet2d(x.v))); + } + + EIGEN_STRONG_INLINE void ptranspose(PacketBlock &kernel) + { + Packet2d tmp = vcombine_f64(vget_high_f64(kernel.packet[0].v), vget_high_f64(kernel.packet[1].v)); + kernel.packet[0].v = vcombine_f64(vget_low_f64(kernel.packet[0].v), vget_low_f64(kernel.packet[1].v)); + kernel.packet[1].v = tmp; + } +#endif// EIGEN_ARCH_ARM64 -} // end namespace internal +}// end namespace internal -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_COMPLEX_NEON_H +#endif// EIGEN_COMPLEX_NEON_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/arch/NEON/MathFunctions.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/arch/NEON/MathFunctions.h index 6bb05bb9..b7fb9861 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/arch/NEON/MathFunctions.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/arch/NEON/MathFunctions.h @@ -16,76 +16,76 @@ namespace Eigen { namespace internal { -template<> EIGEN_DEFINE_FUNCTION_ALLOWING_MULTIPLE_DEFINITIONS EIGEN_UNUSED -Packet4f pexp(const Packet4f& _x) -{ - Packet4f x = _x; - Packet4f tmp, fx; - - _EIGEN_DECLARE_CONST_Packet4f(1 , 1.0f); - _EIGEN_DECLARE_CONST_Packet4f(half, 0.5f); - _EIGEN_DECLARE_CONST_Packet4i(0x7f, 0x7f); - _EIGEN_DECLARE_CONST_Packet4f(exp_hi, 88.3762626647950f); - _EIGEN_DECLARE_CONST_Packet4f(exp_lo, -88.3762626647949f); - _EIGEN_DECLARE_CONST_Packet4f(cephes_LOG2EF, 1.44269504088896341f); - _EIGEN_DECLARE_CONST_Packet4f(cephes_exp_C1, 0.693359375f); - _EIGEN_DECLARE_CONST_Packet4f(cephes_exp_C2, -2.12194440e-4f); - _EIGEN_DECLARE_CONST_Packet4f(cephes_exp_p0, 1.9875691500E-4f); - _EIGEN_DECLARE_CONST_Packet4f(cephes_exp_p1, 1.3981999507E-3f); - _EIGEN_DECLARE_CONST_Packet4f(cephes_exp_p2, 8.3334519073E-3f); - _EIGEN_DECLARE_CONST_Packet4f(cephes_exp_p3, 4.1665795894E-2f); - _EIGEN_DECLARE_CONST_Packet4f(cephes_exp_p4, 1.6666665459E-1f); - _EIGEN_DECLARE_CONST_Packet4f(cephes_exp_p5, 5.0000001201E-1f); - - x = vminq_f32(x, p4f_exp_hi); - x = vmaxq_f32(x, p4f_exp_lo); - - /* express exp(x) as exp(g + n*log(2)) */ - fx = vmlaq_f32(p4f_half, x, p4f_cephes_LOG2EF); - - /* perform a floorf */ - tmp = vcvtq_f32_s32(vcvtq_s32_f32(fx)); - - /* if greater, substract 1 */ - Packet4ui mask = vcgtq_f32(tmp, fx); - mask = vandq_u32(mask, vreinterpretq_u32_f32(p4f_1)); - - fx = vsubq_f32(tmp, vreinterpretq_f32_u32(mask)); - - tmp = vmulq_f32(fx, p4f_cephes_exp_C1); - Packet4f z = vmulq_f32(fx, p4f_cephes_exp_C2); - x = vsubq_f32(x, tmp); - x = vsubq_f32(x, z); - - Packet4f y = vmulq_f32(p4f_cephes_exp_p0, x); - z = vmulq_f32(x, x); - y = vaddq_f32(y, p4f_cephes_exp_p1); - y = vmulq_f32(y, x); - y = vaddq_f32(y, p4f_cephes_exp_p2); - y = vmulq_f32(y, x); - y = vaddq_f32(y, p4f_cephes_exp_p3); - y = vmulq_f32(y, x); - y = vaddq_f32(y, p4f_cephes_exp_p4); - y = vmulq_f32(y, x); - y = vaddq_f32(y, p4f_cephes_exp_p5); - - y = vmulq_f32(y, z); - y = vaddq_f32(y, x); - y = vaddq_f32(y, p4f_1); - - /* build 2^n */ - int32x4_t mm; - mm = vcvtq_s32_f32(fx); - mm = vaddq_s32(mm, p4i_0x7f); - mm = vshlq_n_s32(mm, 23); - Packet4f pow2n = vreinterpretq_f32_s32(mm); - - y = vmulq_f32(y, pow2n); - return y; -} - -} // end namespace internal - -} // end namespace Eigen - -#endif // EIGEN_MATH_FUNCTIONS_NEON_H + template<> + EIGEN_DEFINE_FUNCTION_ALLOWING_MULTIPLE_DEFINITIONS EIGEN_UNUSED Packet4f pexp(const Packet4f &_x) + { + Packet4f x = _x; + Packet4f tmp, fx; + + _EIGEN_DECLARE_CONST_Packet4f(1, 1.0f); + _EIGEN_DECLARE_CONST_Packet4f(half, 0.5f); + _EIGEN_DECLARE_CONST_Packet4i(0x7f, 0x7f); + _EIGEN_DECLARE_CONST_Packet4f(exp_hi, 88.3762626647950f); + _EIGEN_DECLARE_CONST_Packet4f(exp_lo, -88.3762626647949f); + _EIGEN_DECLARE_CONST_Packet4f(cephes_LOG2EF, 1.44269504088896341f); + _EIGEN_DECLARE_CONST_Packet4f(cephes_exp_C1, 0.693359375f); + _EIGEN_DECLARE_CONST_Packet4f(cephes_exp_C2, -2.12194440e-4f); + _EIGEN_DECLARE_CONST_Packet4f(cephes_exp_p0, 1.9875691500E-4f); + _EIGEN_DECLARE_CONST_Packet4f(cephes_exp_p1, 1.3981999507E-3f); + _EIGEN_DECLARE_CONST_Packet4f(cephes_exp_p2, 8.3334519073E-3f); + _EIGEN_DECLARE_CONST_Packet4f(cephes_exp_p3, 4.1665795894E-2f); + _EIGEN_DECLARE_CONST_Packet4f(cephes_exp_p4, 1.6666665459E-1f); + _EIGEN_DECLARE_CONST_Packet4f(cephes_exp_p5, 5.0000001201E-1f); + + x = vminq_f32(x, p4f_exp_hi); + x = vmaxq_f32(x, p4f_exp_lo); + + /* express exp(x) as exp(g + n*log(2)) */ + fx = vmlaq_f32(p4f_half, x, p4f_cephes_LOG2EF); + + /* perform a floorf */ + tmp = vcvtq_f32_s32(vcvtq_s32_f32(fx)); + + /* if greater, substract 1 */ + Packet4ui mask = vcgtq_f32(tmp, fx); + mask = vandq_u32(mask, vreinterpretq_u32_f32(p4f_1)); + + fx = vsubq_f32(tmp, vreinterpretq_f32_u32(mask)); + + tmp = vmulq_f32(fx, p4f_cephes_exp_C1); + Packet4f z = vmulq_f32(fx, p4f_cephes_exp_C2); + x = vsubq_f32(x, tmp); + x = vsubq_f32(x, z); + + Packet4f y = vmulq_f32(p4f_cephes_exp_p0, x); + z = vmulq_f32(x, x); + y = vaddq_f32(y, p4f_cephes_exp_p1); + y = vmulq_f32(y, x); + y = vaddq_f32(y, p4f_cephes_exp_p2); + y = vmulq_f32(y, x); + y = vaddq_f32(y, p4f_cephes_exp_p3); + y = vmulq_f32(y, x); + y = vaddq_f32(y, p4f_cephes_exp_p4); + y = vmulq_f32(y, x); + y = vaddq_f32(y, p4f_cephes_exp_p5); + + y = vmulq_f32(y, z); + y = vaddq_f32(y, x); + y = vaddq_f32(y, p4f_1); + + /* build 2^n */ + int32x4_t mm; + mm = vcvtq_s32_f32(fx); + mm = vaddq_s32(mm, p4i_0x7f); + mm = vshlq_n_s32(mm, 23); + Packet4f pow2n = vreinterpretq_f32_s32(mm); + + y = vmulq_f32(y, pow2n); + return y; + } + +}// end namespace internal + +}// end namespace Eigen + +#endif// EIGEN_MATH_FUNCTIONS_NEON_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/arch/NEON/PacketMath.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/arch/NEON/PacketMath.h index 3d5ed0d2..de169c69 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/arch/NEON/PacketMath.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/arch/NEON/PacketMath.h @@ -32,525 +32,612 @@ namespace internal { #if EIGEN_ARCH_ARM64 #define EIGEN_ARCH_DEFAULT_NUMBER_OF_REGISTERS 32 #else -#define EIGEN_ARCH_DEFAULT_NUMBER_OF_REGISTERS 16 +#define EIGEN_ARCH_DEFAULT_NUMBER_OF_REGISTERS 16 #endif #endif #if EIGEN_COMP_MSVC -// In MSVC's arm_neon.h header file, all NEON vector types -// are aliases to the same underlying type __n128. -// We thus have to wrap them to make them different C++ types. -// (See also bug 1428) - -template -struct eigen_packet_wrapper -{ - operator T&() { return m_val; } - operator const T&() const { return m_val; } - eigen_packet_wrapper() {} - eigen_packet_wrapper(const T &v) : m_val(v) {} - eigen_packet_wrapper& operator=(const T &v) { - m_val = v; - return *this; - } - - T m_val; -}; -typedef eigen_packet_wrapper Packet2f; -typedef eigen_packet_wrapper Packet4f; -typedef eigen_packet_wrapper Packet4i; -typedef eigen_packet_wrapper Packet2i; -typedef eigen_packet_wrapper Packet4ui; + // In MSVC's arm_neon.h header file, all NEON vector types + // are aliases to the same underlying type __n128. + // We thus have to wrap them to make them different C++ types. + // (See also bug 1428) + + template struct eigen_packet_wrapper + { + operator T &() { return m_val; } + operator const T &() const { return m_val; } + eigen_packet_wrapper() {} + eigen_packet_wrapper(const T &v) : m_val(v) {} + eigen_packet_wrapper &operator=(const T &v) + { + m_val = v; + return *this; + } + + T m_val; + }; + typedef eigen_packet_wrapper Packet2f; + typedef eigen_packet_wrapper Packet4f; + typedef eigen_packet_wrapper Packet4i; + typedef eigen_packet_wrapper Packet2i; + typedef eigen_packet_wrapper Packet4ui; #else -typedef float32x2_t Packet2f; -typedef float32x4_t Packet4f; -typedef int32x4_t Packet4i; -typedef int32x2_t Packet2i; -typedef uint32x4_t Packet4ui; + typedef float32x2_t Packet2f; + typedef float32x4_t Packet4f; + typedef int32x4_t Packet4i; + typedef int32x2_t Packet2i; + typedef uint32x4_t Packet4ui; -#endif // EIGEN_COMP_MSVC +#endif// EIGEN_COMP_MSVC -#define _EIGEN_DECLARE_CONST_Packet4f(NAME,X) \ - const Packet4f p4f_##NAME = pset1(X) +#define _EIGEN_DECLARE_CONST_Packet4f(NAME, X) const Packet4f p4f_##NAME = pset1(X) -#define _EIGEN_DECLARE_CONST_Packet4f_FROM_INT(NAME,X) \ +#define _EIGEN_DECLARE_CONST_Packet4f_FROM_INT(NAME, X) \ const Packet4f p4f_##NAME = vreinterpretq_f32_u32(pset1(X)) -#define _EIGEN_DECLARE_CONST_Packet4i(NAME,X) \ - const Packet4i p4i_##NAME = pset1(X) +#define _EIGEN_DECLARE_CONST_Packet4i(NAME, X) const Packet4i p4i_##NAME = pset1(X) #if EIGEN_ARCH_ARM64 - // __builtin_prefetch tends to do nothing on ARM64 compilers because the - // prefetch instructions there are too detailed for __builtin_prefetch to map - // meaningfully to them. - #define EIGEN_ARM_PREFETCH(ADDR) __asm__ __volatile__("prfm pldl1keep, [%[addr]]\n" ::[addr] "r"(ADDR) : ); +// __builtin_prefetch tends to do nothing on ARM64 compilers because the +// prefetch instructions there are too detailed for __builtin_prefetch to map +// meaningfully to them. +#define EIGEN_ARM_PREFETCH(ADDR) __asm__ __volatile__("prfm pldl1keep, [%[addr]]\n" ::[addr] "r"(ADDR) :); #elif EIGEN_HAS_BUILTIN(__builtin_prefetch) || EIGEN_COMP_GNUC - #define EIGEN_ARM_PREFETCH(ADDR) __builtin_prefetch(ADDR); +#define EIGEN_ARM_PREFETCH(ADDR) __builtin_prefetch(ADDR); #elif defined __pld - #define EIGEN_ARM_PREFETCH(ADDR) __pld(ADDR) +#define EIGEN_ARM_PREFETCH(ADDR) __pld(ADDR) #elif EIGEN_ARCH_ARM32 - #define EIGEN_ARM_PREFETCH(ADDR) __asm__ __volatile__ ("pld [%[addr]]\n" :: [addr] "r" (ADDR) : ); +#define EIGEN_ARM_PREFETCH(ADDR) __asm__ __volatile__("pld [%[addr]]\n" ::[addr] "r"(ADDR) :); #else - // by default no explicit prefetching - #define EIGEN_ARM_PREFETCH(ADDR) +// by default no explicit prefetching +#define EIGEN_ARM_PREFETCH(ADDR) #endif -template<> struct packet_traits : default_packet_traits -{ - typedef Packet4f type; - typedef Packet4f half; // Packet2f intrinsics not implemented yet - enum { - Vectorizable = 1, - AlignedOnScalar = 1, - size = 4, - HasHalfPacket=0, // Packet2f intrinsics not implemented yet - - HasDiv = 1, - // FIXME check the Has* - HasSin = 0, - HasCos = 0, - HasLog = 0, - HasExp = 1, - HasSqrt = 0 + template<> struct packet_traits : default_packet_traits + { + typedef Packet4f type; + typedef Packet4f half;// Packet2f intrinsics not implemented yet + enum { + Vectorizable = 1, + AlignedOnScalar = 1, + size = 4, + HasHalfPacket = 0,// Packet2f intrinsics not implemented yet + + HasDiv = 1, + // FIXME check the Has* + HasSin = 0, + HasCos = 0, + HasLog = 0, + HasExp = 1, + HasSqrt = 0 + }; }; -}; -template<> struct packet_traits : default_packet_traits -{ - typedef Packet4i type; - typedef Packet4i half; // Packet2i intrinsics not implemented yet - enum { - Vectorizable = 1, - AlignedOnScalar = 1, - size=4, - HasHalfPacket=0 // Packet2i intrinsics not implemented yet - // FIXME check the Has* + template<> struct packet_traits : default_packet_traits + { + typedef Packet4i type; + typedef Packet4i half;// Packet2i intrinsics not implemented yet + enum { + Vectorizable = 1, + AlignedOnScalar = 1, + size = 4, + HasHalfPacket = 0// Packet2i intrinsics not implemented yet + // FIXME check the Has* + }; }; -}; - -#if EIGEN_GNUC_AT_MOST(4,4) && !EIGEN_COMP_LLVM -// workaround gcc 4.2, 4.3 and 4.4 compilatin issue -EIGEN_STRONG_INLINE float32x4_t vld1q_f32(const float* x) { return ::vld1q_f32((const float32_t*)x); } -EIGEN_STRONG_INLINE float32x2_t vld1_f32 (const float* x) { return ::vld1_f32 ((const float32_t*)x); } -EIGEN_STRONG_INLINE float32x2_t vld1_dup_f32 (const float* x) { return ::vld1_dup_f32 ((const float32_t*)x); } -EIGEN_STRONG_INLINE void vst1q_f32(float* to, float32x4_t from) { ::vst1q_f32((float32_t*)to,from); } -EIGEN_STRONG_INLINE void vst1_f32 (float* to, float32x2_t from) { ::vst1_f32 ((float32_t*)to,from); } + +#if EIGEN_GNUC_AT_MOST(4, 4) && !EIGEN_COMP_LLVM + // workaround gcc 4.2, 4.3 and 4.4 compilatin issue + EIGEN_STRONG_INLINE float32x4_t vld1q_f32(const float *x) { return ::vld1q_f32((const float32_t *)x); } + EIGEN_STRONG_INLINE float32x2_t vld1_f32(const float *x) { return ::vld1_f32((const float32_t *)x); } + EIGEN_STRONG_INLINE float32x2_t vld1_dup_f32(const float *x) { return ::vld1_dup_f32((const float32_t *)x); } + EIGEN_STRONG_INLINE void vst1q_f32(float *to, float32x4_t from) { ::vst1q_f32((float32_t *)to, from); } + EIGEN_STRONG_INLINE void vst1_f32(float *to, float32x2_t from) { ::vst1_f32((float32_t *)to, from); } #endif -template<> struct unpacket_traits { typedef float type; enum {size=4, alignment=Aligned16}; typedef Packet4f half; }; -template<> struct unpacket_traits { typedef int32_t type; enum {size=4, alignment=Aligned16}; typedef Packet4i half; }; + template<> struct unpacket_traits + { + typedef float type; + enum { size = 4, alignment = Aligned16 }; + typedef Packet4f half; + }; + template<> struct unpacket_traits + { + typedef int32_t type; + enum { size = 4, alignment = Aligned16 }; + typedef Packet4i half; + }; -template<> EIGEN_STRONG_INLINE Packet4f pset1(const float& from) { return vdupq_n_f32(from); } -template<> EIGEN_STRONG_INLINE Packet4i pset1(const int32_t& from) { return vdupq_n_s32(from); } + template<> EIGEN_STRONG_INLINE Packet4f pset1(const float &from) { return vdupq_n_f32(from); } + template<> EIGEN_STRONG_INLINE Packet4i pset1(const int32_t &from) { return vdupq_n_s32(from); } -template<> EIGEN_STRONG_INLINE Packet4f plset(const float& a) -{ - const float f[] = {0, 1, 2, 3}; - Packet4f countdown = vld1q_f32(f); - return vaddq_f32(pset1(a), countdown); -} -template<> EIGEN_STRONG_INLINE Packet4i plset(const int32_t& a) -{ - const int32_t i[] = {0, 1, 2, 3}; - Packet4i countdown = vld1q_s32(i); - return vaddq_s32(pset1(a), countdown); -} + template<> EIGEN_STRONG_INLINE Packet4f plset(const float &a) + { + const float f[] = { 0, 1, 2, 3 }; + Packet4f countdown = vld1q_f32(f); + return vaddq_f32(pset1(a), countdown); + } + template<> EIGEN_STRONG_INLINE Packet4i plset(const int32_t &a) + { + const int32_t i[] = { 0, 1, 2, 3 }; + Packet4i countdown = vld1q_s32(i); + return vaddq_s32(pset1(a), countdown); + } -template<> EIGEN_STRONG_INLINE Packet4f padd(const Packet4f& a, const Packet4f& b) { return vaddq_f32(a,b); } -template<> EIGEN_STRONG_INLINE Packet4i padd(const Packet4i& a, const Packet4i& b) { return vaddq_s32(a,b); } + template<> EIGEN_STRONG_INLINE Packet4f padd(const Packet4f &a, const Packet4f &b) + { + return vaddq_f32(a, b); + } + template<> EIGEN_STRONG_INLINE Packet4i padd(const Packet4i &a, const Packet4i &b) + { + return vaddq_s32(a, b); + } -template<> EIGEN_STRONG_INLINE Packet4f psub(const Packet4f& a, const Packet4f& b) { return vsubq_f32(a,b); } -template<> EIGEN_STRONG_INLINE Packet4i psub(const Packet4i& a, const Packet4i& b) { return vsubq_s32(a,b); } + template<> EIGEN_STRONG_INLINE Packet4f psub(const Packet4f &a, const Packet4f &b) + { + return vsubq_f32(a, b); + } + template<> EIGEN_STRONG_INLINE Packet4i psub(const Packet4i &a, const Packet4i &b) + { + return vsubq_s32(a, b); + } -template<> EIGEN_STRONG_INLINE Packet4f pnegate(const Packet4f& a) { return vnegq_f32(a); } -template<> EIGEN_STRONG_INLINE Packet4i pnegate(const Packet4i& a) { return vnegq_s32(a); } + template<> EIGEN_STRONG_INLINE Packet4f pnegate(const Packet4f &a) { return vnegq_f32(a); } + template<> EIGEN_STRONG_INLINE Packet4i pnegate(const Packet4i &a) { return vnegq_s32(a); } -template<> EIGEN_STRONG_INLINE Packet4f pconj(const Packet4f& a) { return a; } -template<> EIGEN_STRONG_INLINE Packet4i pconj(const Packet4i& a) { return a; } + template<> EIGEN_STRONG_INLINE Packet4f pconj(const Packet4f &a) { return a; } + template<> EIGEN_STRONG_INLINE Packet4i pconj(const Packet4i &a) { return a; } -template<> EIGEN_STRONG_INLINE Packet4f pmul(const Packet4f& a, const Packet4f& b) { return vmulq_f32(a,b); } -template<> EIGEN_STRONG_INLINE Packet4i pmul(const Packet4i& a, const Packet4i& b) { return vmulq_s32(a,b); } + template<> EIGEN_STRONG_INLINE Packet4f pmul(const Packet4f &a, const Packet4f &b) + { + return vmulq_f32(a, b); + } + template<> EIGEN_STRONG_INLINE Packet4i pmul(const Packet4i &a, const Packet4i &b) + { + return vmulq_s32(a, b); + } -template<> EIGEN_STRONG_INLINE Packet4f pdiv(const Packet4f& a, const Packet4f& b) -{ + template<> EIGEN_STRONG_INLINE Packet4f pdiv(const Packet4f &a, const Packet4f &b) + { #if EIGEN_ARCH_ARM64 - return vdivq_f32(a,b); + return vdivq_f32(a, b); #else - Packet4f inv, restep, div; + Packet4f inv, restep, div; - // NEON does not offer a divide instruction, we have to do a reciprocal approximation - // However NEON in contrast to other SIMD engines (AltiVec/SSE), offers - // a reciprocal estimate AND a reciprocal step -which saves a few instructions - // vrecpeq_f32() returns an estimate to 1/b, which we will finetune with - // Newton-Raphson and vrecpsq_f32() - inv = vrecpeq_f32(b); + // NEON does not offer a divide instruction, we have to do a reciprocal approximation + // However NEON in contrast to other SIMD engines (AltiVec/SSE), offers + // a reciprocal estimate AND a reciprocal step -which saves a few instructions + // vrecpeq_f32() returns an estimate to 1/b, which we will finetune with + // Newton-Raphson and vrecpsq_f32() + inv = vrecpeq_f32(b); - // This returns a differential, by which we will have to multiply inv to get a better - // approximation of 1/b. - restep = vrecpsq_f32(b, inv); - inv = vmulq_f32(restep, inv); + // This returns a differential, by which we will have to multiply inv to get a better + // approximation of 1/b. + restep = vrecpsq_f32(b, inv); + inv = vmulq_f32(restep, inv); - // Finally, multiply a by 1/b and get the wanted result of the division. - div = vmulq_f32(a, inv); + // Finally, multiply a by 1/b and get the wanted result of the division. + div = vmulq_f32(a, inv); - return div; + return div; #endif -} + } -template<> EIGEN_STRONG_INLINE Packet4i pdiv(const Packet4i& /*a*/, const Packet4i& /*b*/) -{ eigen_assert(false && "packet integer division are not supported by NEON"); - return pset1(0); -} + template<> EIGEN_STRONG_INLINE Packet4i pdiv(const Packet4i & /*a*/, const Packet4i & /*b*/) + { + eigen_assert(false && "packet integer division are not supported by NEON"); + return pset1(0); + } // Clang/ARM wrongly advertises __ARM_FEATURE_FMA even when it's not available, // then implements a slow software scalar fallback calling fmaf()! // Filed LLVM bug: // https://llvm.org/bugs/show_bug.cgi?id=27216 #if (defined __ARM_FEATURE_FMA) && !(EIGEN_COMP_CLANG && EIGEN_ARCH_ARM) -// See bug 936. -// FMA is available on VFPv4 i.e. when compiling with -mfpu=neon-vfpv4. -// FMA is a true fused multiply-add i.e. only 1 rounding at the end, no intermediate rounding. -// MLA is not fused i.e. does 2 roundings. -// In addition to giving better accuracy, FMA also gives better performance here on a Krait (Nexus 4): -// MLA: 10 GFlop/s ; FMA: 12 GFlops/s. -template<> EIGEN_STRONG_INLINE Packet4f pmadd(const Packet4f& a, const Packet4f& b, const Packet4f& c) { return vfmaq_f32(c,a,b); } + // See bug 936. + // FMA is available on VFPv4 i.e. when compiling with -mfpu=neon-vfpv4. + // FMA is a true fused multiply-add i.e. only 1 rounding at the end, no intermediate rounding. + // MLA is not fused i.e. does 2 roundings. + // In addition to giving better accuracy, FMA also gives better performance here on a Krait (Nexus 4): + // MLA: 10 GFlop/s ; FMA: 12 GFlops/s. + template<> EIGEN_STRONG_INLINE Packet4f pmadd(const Packet4f &a, const Packet4f &b, const Packet4f &c) + { + return vfmaq_f32(c, a, b); + } #else -template<> EIGEN_STRONG_INLINE Packet4f pmadd(const Packet4f& a, const Packet4f& b, const Packet4f& c) { + template<> EIGEN_STRONG_INLINE Packet4f pmadd(const Packet4f &a, const Packet4f &b, const Packet4f &c) + { #if EIGEN_COMP_CLANG && EIGEN_ARCH_ARM - // Clang/ARM will replace VMLA by VMUL+VADD at least for some values of -mcpu, - // at least -mcpu=cortex-a8 and -mcpu=cortex-a7. Since the former is the default on - // -march=armv7-a, that is a very common case. - // See e.g. this thread: - // http://lists.llvm.org/pipermail/llvm-dev/2013-December/068806.html - // Filed LLVM bug: - // https://llvm.org/bugs/show_bug.cgi?id=27219 - Packet4f r = c; - asm volatile( - "vmla.f32 %q[r], %q[a], %q[b]" - : [r] "+w" (r) - : [a] "w" (a), - [b] "w" (b) - : ); - return r; + // Clang/ARM will replace VMLA by VMUL+VADD at least for some values of -mcpu, + // at least -mcpu=cortex-a8 and -mcpu=cortex-a7. Since the former is the default on + // -march=armv7-a, that is a very common case. + // See e.g. this thread: + // http://lists.llvm.org/pipermail/llvm-dev/2013-December/068806.html + // Filed LLVM bug: + // https://llvm.org/bugs/show_bug.cgi?id=27219 + Packet4f r = c; + asm volatile("vmla.f32 %q[r], %q[a], %q[b]" : [r] "+w"(r) : [a] "w"(a), [b] "w"(b) :); + return r; #else - return vmlaq_f32(c,a,b); + return vmlaq_f32(c, a, b); #endif -} + } #endif -// No FMA instruction for int, so use MLA unconditionally. -template<> EIGEN_STRONG_INLINE Packet4i pmadd(const Packet4i& a, const Packet4i& b, const Packet4i& c) { return vmlaq_s32(c,a,b); } - -template<> EIGEN_STRONG_INLINE Packet4f pmin(const Packet4f& a, const Packet4f& b) { return vminq_f32(a,b); } -template<> EIGEN_STRONG_INLINE Packet4i pmin(const Packet4i& a, const Packet4i& b) { return vminq_s32(a,b); } - -template<> EIGEN_STRONG_INLINE Packet4f pmax(const Packet4f& a, const Packet4f& b) { return vmaxq_f32(a,b); } -template<> EIGEN_STRONG_INLINE Packet4i pmax(const Packet4i& a, const Packet4i& b) { return vmaxq_s32(a,b); } - -// Logical Operations are not supported for float, so we have to reinterpret casts using NEON intrinsics -template<> EIGEN_STRONG_INLINE Packet4f pand(const Packet4f& a, const Packet4f& b) -{ - return vreinterpretq_f32_u32(vandq_u32(vreinterpretq_u32_f32(a),vreinterpretq_u32_f32(b))); -} -template<> EIGEN_STRONG_INLINE Packet4i pand(const Packet4i& a, const Packet4i& b) { return vandq_s32(a,b); } - -template<> EIGEN_STRONG_INLINE Packet4f por(const Packet4f& a, const Packet4f& b) -{ - return vreinterpretq_f32_u32(vorrq_u32(vreinterpretq_u32_f32(a),vreinterpretq_u32_f32(b))); -} -template<> EIGEN_STRONG_INLINE Packet4i por(const Packet4i& a, const Packet4i& b) { return vorrq_s32(a,b); } - -template<> EIGEN_STRONG_INLINE Packet4f pxor(const Packet4f& a, const Packet4f& b) -{ - return vreinterpretq_f32_u32(veorq_u32(vreinterpretq_u32_f32(a),vreinterpretq_u32_f32(b))); -} -template<> EIGEN_STRONG_INLINE Packet4i pxor(const Packet4i& a, const Packet4i& b) { return veorq_s32(a,b); } - -template<> EIGEN_STRONG_INLINE Packet4f pandnot(const Packet4f& a, const Packet4f& b) -{ - return vreinterpretq_f32_u32(vbicq_u32(vreinterpretq_u32_f32(a),vreinterpretq_u32_f32(b))); -} -template<> EIGEN_STRONG_INLINE Packet4i pandnot(const Packet4i& a, const Packet4i& b) { return vbicq_s32(a,b); } - -template<> EIGEN_STRONG_INLINE Packet4f pload(const float* from) { EIGEN_DEBUG_ALIGNED_LOAD return vld1q_f32(from); } -template<> EIGEN_STRONG_INLINE Packet4i pload(const int32_t* from) { EIGEN_DEBUG_ALIGNED_LOAD return vld1q_s32(from); } - -template<> EIGEN_STRONG_INLINE Packet4f ploadu(const float* from) { EIGEN_DEBUG_UNALIGNED_LOAD return vld1q_f32(from); } -template<> EIGEN_STRONG_INLINE Packet4i ploadu(const int32_t* from) { EIGEN_DEBUG_UNALIGNED_LOAD return vld1q_s32(from); } - -template<> EIGEN_STRONG_INLINE Packet4f ploaddup(const float* from) -{ - float32x2_t lo, hi; - lo = vld1_dup_f32(from); - hi = vld1_dup_f32(from+1); - return vcombine_f32(lo, hi); -} -template<> EIGEN_STRONG_INLINE Packet4i ploaddup(const int32_t* from) -{ - int32x2_t lo, hi; - lo = vld1_dup_s32(from); - hi = vld1_dup_s32(from+1); - return vcombine_s32(lo, hi); -} - -template<> EIGEN_STRONG_INLINE void pstore (float* to, const Packet4f& from) { EIGEN_DEBUG_ALIGNED_STORE vst1q_f32(to, from); } -template<> EIGEN_STRONG_INLINE void pstore(int32_t* to, const Packet4i& from) { EIGEN_DEBUG_ALIGNED_STORE vst1q_s32(to, from); } - -template<> EIGEN_STRONG_INLINE void pstoreu (float* to, const Packet4f& from) { EIGEN_DEBUG_UNALIGNED_STORE vst1q_f32(to, from); } -template<> EIGEN_STRONG_INLINE void pstoreu(int32_t* to, const Packet4i& from) { EIGEN_DEBUG_UNALIGNED_STORE vst1q_s32(to, from); } - -template<> EIGEN_DEVICE_FUNC inline Packet4f pgather(const float* from, Index stride) -{ - Packet4f res = pset1(0.f); - res = vsetq_lane_f32(from[0*stride], res, 0); - res = vsetq_lane_f32(from[1*stride], res, 1); - res = vsetq_lane_f32(from[2*stride], res, 2); - res = vsetq_lane_f32(from[3*stride], res, 3); - return res; -} -template<> EIGEN_DEVICE_FUNC inline Packet4i pgather(const int32_t* from, Index stride) -{ - Packet4i res = pset1(0); - res = vsetq_lane_s32(from[0*stride], res, 0); - res = vsetq_lane_s32(from[1*stride], res, 1); - res = vsetq_lane_s32(from[2*stride], res, 2); - res = vsetq_lane_s32(from[3*stride], res, 3); - return res; -} - -template<> EIGEN_DEVICE_FUNC inline void pscatter(float* to, const Packet4f& from, Index stride) -{ - to[stride*0] = vgetq_lane_f32(from, 0); - to[stride*1] = vgetq_lane_f32(from, 1); - to[stride*2] = vgetq_lane_f32(from, 2); - to[stride*3] = vgetq_lane_f32(from, 3); -} -template<> EIGEN_DEVICE_FUNC inline void pscatter(int32_t* to, const Packet4i& from, Index stride) -{ - to[stride*0] = vgetq_lane_s32(from, 0); - to[stride*1] = vgetq_lane_s32(from, 1); - to[stride*2] = vgetq_lane_s32(from, 2); - to[stride*3] = vgetq_lane_s32(from, 3); -} - -template<> EIGEN_STRONG_INLINE void prefetch (const float* addr) { EIGEN_ARM_PREFETCH(addr); } -template<> EIGEN_STRONG_INLINE void prefetch(const int32_t* addr) { EIGEN_ARM_PREFETCH(addr); } - -// FIXME only store the 2 first elements ? -template<> EIGEN_STRONG_INLINE float pfirst(const Packet4f& a) { float EIGEN_ALIGN16 x[4]; vst1q_f32(x, a); return x[0]; } -template<> EIGEN_STRONG_INLINE int32_t pfirst(const Packet4i& a) { int32_t EIGEN_ALIGN16 x[4]; vst1q_s32(x, a); return x[0]; } - -template<> EIGEN_STRONG_INLINE Packet4f preverse(const Packet4f& a) { - float32x2_t a_lo, a_hi; - Packet4f a_r64; - - a_r64 = vrev64q_f32(a); - a_lo = vget_low_f32(a_r64); - a_hi = vget_high_f32(a_r64); - return vcombine_f32(a_hi, a_lo); -} -template<> EIGEN_STRONG_INLINE Packet4i preverse(const Packet4i& a) { - int32x2_t a_lo, a_hi; - Packet4i a_r64; - - a_r64 = vrev64q_s32(a); - a_lo = vget_low_s32(a_r64); - a_hi = vget_high_s32(a_r64); - return vcombine_s32(a_hi, a_lo); -} - -template<> EIGEN_STRONG_INLINE Packet4f pabs(const Packet4f& a) { return vabsq_f32(a); } -template<> EIGEN_STRONG_INLINE Packet4i pabs(const Packet4i& a) { return vabsq_s32(a); } - -template<> EIGEN_STRONG_INLINE float predux(const Packet4f& a) -{ - float32x2_t a_lo, a_hi, sum; - - a_lo = vget_low_f32(a); - a_hi = vget_high_f32(a); - sum = vpadd_f32(a_lo, a_hi); - sum = vpadd_f32(sum, sum); - return vget_lane_f32(sum, 0); -} - -template<> EIGEN_STRONG_INLINE Packet4f preduxp(const Packet4f* vecs) -{ - float32x4x2_t vtrn1, vtrn2, res1, res2; - Packet4f sum1, sum2, sum; - - // NEON zip performs interleaving of the supplied vectors. - // We perform two interleaves in a row to acquire the transposed vector - vtrn1 = vzipq_f32(vecs[0], vecs[2]); - vtrn2 = vzipq_f32(vecs[1], vecs[3]); - res1 = vzipq_f32(vtrn1.val[0], vtrn2.val[0]); - res2 = vzipq_f32(vtrn1.val[1], vtrn2.val[1]); - - // Do the addition of the resulting vectors - sum1 = vaddq_f32(res1.val[0], res1.val[1]); - sum2 = vaddq_f32(res2.val[0], res2.val[1]); - sum = vaddq_f32(sum1, sum2); - - return sum; -} - -template<> EIGEN_STRONG_INLINE int32_t predux(const Packet4i& a) -{ - int32x2_t a_lo, a_hi, sum; - - a_lo = vget_low_s32(a); - a_hi = vget_high_s32(a); - sum = vpadd_s32(a_lo, a_hi); - sum = vpadd_s32(sum, sum); - return vget_lane_s32(sum, 0); -} - -template<> EIGEN_STRONG_INLINE Packet4i preduxp(const Packet4i* vecs) -{ - int32x4x2_t vtrn1, vtrn2, res1, res2; - Packet4i sum1, sum2, sum; - - // NEON zip performs interleaving of the supplied vectors. - // We perform two interleaves in a row to acquire the transposed vector - vtrn1 = vzipq_s32(vecs[0], vecs[2]); - vtrn2 = vzipq_s32(vecs[1], vecs[3]); - res1 = vzipq_s32(vtrn1.val[0], vtrn2.val[0]); - res2 = vzipq_s32(vtrn1.val[1], vtrn2.val[1]); - - // Do the addition of the resulting vectors - sum1 = vaddq_s32(res1.val[0], res1.val[1]); - sum2 = vaddq_s32(res2.val[0], res2.val[1]); - sum = vaddq_s32(sum1, sum2); - - return sum; -} + // No FMA instruction for int, so use MLA unconditionally. + template<> EIGEN_STRONG_INLINE Packet4i pmadd(const Packet4i &a, const Packet4i &b, const Packet4i &c) + { + return vmlaq_s32(c, a, b); + } -// Other reduction functions: -// mul -template<> EIGEN_STRONG_INLINE float predux_mul(const Packet4f& a) -{ - float32x2_t a_lo, a_hi, prod; - - // Get a_lo = |a1|a2| and a_hi = |a3|a4| - a_lo = vget_low_f32(a); - a_hi = vget_high_f32(a); - // Get the product of a_lo * a_hi -> |a1*a3|a2*a4| - prod = vmul_f32(a_lo, a_hi); - // Multiply prod with its swapped value |a2*a4|a1*a3| - prod = vmul_f32(prod, vrev64_f32(prod)); - - return vget_lane_f32(prod, 0); -} -template<> EIGEN_STRONG_INLINE int32_t predux_mul(const Packet4i& a) -{ - int32x2_t a_lo, a_hi, prod; - - // Get a_lo = |a1|a2| and a_hi = |a3|a4| - a_lo = vget_low_s32(a); - a_hi = vget_high_s32(a); - // Get the product of a_lo * a_hi -> |a1*a3|a2*a4| - prod = vmul_s32(a_lo, a_hi); - // Multiply prod with its swapped value |a2*a4|a1*a3| - prod = vmul_s32(prod, vrev64_s32(prod)); - - return vget_lane_s32(prod, 0); -} - -// min -template<> EIGEN_STRONG_INLINE float predux_min(const Packet4f& a) -{ - float32x2_t a_lo, a_hi, min; - - a_lo = vget_low_f32(a); - a_hi = vget_high_f32(a); - min = vpmin_f32(a_lo, a_hi); - min = vpmin_f32(min, min); - - return vget_lane_f32(min, 0); -} - -template<> EIGEN_STRONG_INLINE int32_t predux_min(const Packet4i& a) -{ - int32x2_t a_lo, a_hi, min; - - a_lo = vget_low_s32(a); - a_hi = vget_high_s32(a); - min = vpmin_s32(a_lo, a_hi); - min = vpmin_s32(min, min); - - return vget_lane_s32(min, 0); -} - -// max -template<> EIGEN_STRONG_INLINE float predux_max(const Packet4f& a) -{ - float32x2_t a_lo, a_hi, max; - - a_lo = vget_low_f32(a); - a_hi = vget_high_f32(a); - max = vpmax_f32(a_lo, a_hi); - max = vpmax_f32(max, max); - - return vget_lane_f32(max, 0); -} - -template<> EIGEN_STRONG_INLINE int32_t predux_max(const Packet4i& a) -{ - int32x2_t a_lo, a_hi, max; - - a_lo = vget_low_s32(a); - a_hi = vget_high_s32(a); - max = vpmax_s32(a_lo, a_hi); - max = vpmax_s32(max, max); - - return vget_lane_s32(max, 0); -} + template<> EIGEN_STRONG_INLINE Packet4f pmin(const Packet4f &a, const Packet4f &b) + { + return vminq_f32(a, b); + } + template<> EIGEN_STRONG_INLINE Packet4i pmin(const Packet4i &a, const Packet4i &b) + { + return vminq_s32(a, b); + } + + template<> EIGEN_STRONG_INLINE Packet4f pmax(const Packet4f &a, const Packet4f &b) + { + return vmaxq_f32(a, b); + } + template<> EIGEN_STRONG_INLINE Packet4i pmax(const Packet4i &a, const Packet4i &b) + { + return vmaxq_s32(a, b); + } + + // Logical Operations are not supported for float, so we have to reinterpret casts using NEON intrinsics + template<> EIGEN_STRONG_INLINE Packet4f pand(const Packet4f &a, const Packet4f &b) + { + return vreinterpretq_f32_u32(vandq_u32(vreinterpretq_u32_f32(a), vreinterpretq_u32_f32(b))); + } + template<> EIGEN_STRONG_INLINE Packet4i pand(const Packet4i &a, const Packet4i &b) + { + return vandq_s32(a, b); + } + + template<> EIGEN_STRONG_INLINE Packet4f por(const Packet4f &a, const Packet4f &b) + { + return vreinterpretq_f32_u32(vorrq_u32(vreinterpretq_u32_f32(a), vreinterpretq_u32_f32(b))); + } + template<> EIGEN_STRONG_INLINE Packet4i por(const Packet4i &a, const Packet4i &b) + { + return vorrq_s32(a, b); + } + + template<> EIGEN_STRONG_INLINE Packet4f pxor(const Packet4f &a, const Packet4f &b) + { + return vreinterpretq_f32_u32(veorq_u32(vreinterpretq_u32_f32(a), vreinterpretq_u32_f32(b))); + } + template<> EIGEN_STRONG_INLINE Packet4i pxor(const Packet4i &a, const Packet4i &b) + { + return veorq_s32(a, b); + } + + template<> EIGEN_STRONG_INLINE Packet4f pandnot(const Packet4f &a, const Packet4f &b) + { + return vreinterpretq_f32_u32(vbicq_u32(vreinterpretq_u32_f32(a), vreinterpretq_u32_f32(b))); + } + template<> EIGEN_STRONG_INLINE Packet4i pandnot(const Packet4i &a, const Packet4i &b) + { + return vbicq_s32(a, b); + } + + template<> EIGEN_STRONG_INLINE Packet4f pload(const float *from) + { + EIGEN_DEBUG_ALIGNED_LOAD return vld1q_f32(from); + } + template<> EIGEN_STRONG_INLINE Packet4i pload(const int32_t *from) + { + EIGEN_DEBUG_ALIGNED_LOAD return vld1q_s32(from); + } + + template<> EIGEN_STRONG_INLINE Packet4f ploadu(const float *from) + { + EIGEN_DEBUG_UNALIGNED_LOAD return vld1q_f32(from); + } + template<> EIGEN_STRONG_INLINE Packet4i ploadu(const int32_t *from) + { + EIGEN_DEBUG_UNALIGNED_LOAD return vld1q_s32(from); + } + + template<> EIGEN_STRONG_INLINE Packet4f ploaddup(const float *from) + { + float32x2_t lo, hi; + lo = vld1_dup_f32(from); + hi = vld1_dup_f32(from + 1); + return vcombine_f32(lo, hi); + } + template<> EIGEN_STRONG_INLINE Packet4i ploaddup(const int32_t *from) + { + int32x2_t lo, hi; + lo = vld1_dup_s32(from); + hi = vld1_dup_s32(from + 1); + return vcombine_s32(lo, hi); + } + + template<> EIGEN_STRONG_INLINE void pstore(float *to, const Packet4f &from) + { + EIGEN_DEBUG_ALIGNED_STORE vst1q_f32(to, from); + } + template<> EIGEN_STRONG_INLINE void pstore(int32_t *to, const Packet4i &from) + { + EIGEN_DEBUG_ALIGNED_STORE vst1q_s32(to, from); + } + + template<> EIGEN_STRONG_INLINE void pstoreu(float *to, const Packet4f &from) + { + EIGEN_DEBUG_UNALIGNED_STORE vst1q_f32(to, from); + } + template<> EIGEN_STRONG_INLINE void pstoreu(int32_t *to, const Packet4i &from) + { + EIGEN_DEBUG_UNALIGNED_STORE vst1q_s32(to, from); + } + + template<> EIGEN_DEVICE_FUNC inline Packet4f pgather(const float *from, Index stride) + { + Packet4f res = pset1(0.f); + res = vsetq_lane_f32(from[0 * stride], res, 0); + res = vsetq_lane_f32(from[1 * stride], res, 1); + res = vsetq_lane_f32(from[2 * stride], res, 2); + res = vsetq_lane_f32(from[3 * stride], res, 3); + return res; + } + template<> EIGEN_DEVICE_FUNC inline Packet4i pgather(const int32_t *from, Index stride) + { + Packet4i res = pset1(0); + res = vsetq_lane_s32(from[0 * stride], res, 0); + res = vsetq_lane_s32(from[1 * stride], res, 1); + res = vsetq_lane_s32(from[2 * stride], res, 2); + res = vsetq_lane_s32(from[3 * stride], res, 3); + return res; + } + + template<> EIGEN_DEVICE_FUNC inline void pscatter(float *to, const Packet4f &from, Index stride) + { + to[stride * 0] = vgetq_lane_f32(from, 0); + to[stride * 1] = vgetq_lane_f32(from, 1); + to[stride * 2] = vgetq_lane_f32(from, 2); + to[stride * 3] = vgetq_lane_f32(from, 3); + } + template<> EIGEN_DEVICE_FUNC inline void pscatter(int32_t *to, const Packet4i &from, Index stride) + { + to[stride * 0] = vgetq_lane_s32(from, 0); + to[stride * 1] = vgetq_lane_s32(from, 1); + to[stride * 2] = vgetq_lane_s32(from, 2); + to[stride * 3] = vgetq_lane_s32(from, 3); + } + + template<> EIGEN_STRONG_INLINE void prefetch(const float *addr) { EIGEN_ARM_PREFETCH(addr); } + template<> EIGEN_STRONG_INLINE void prefetch(const int32_t *addr) { EIGEN_ARM_PREFETCH(addr); } + + // FIXME only store the 2 first elements ? + template<> EIGEN_STRONG_INLINE float pfirst(const Packet4f &a) + { + float EIGEN_ALIGN16 x[4]; + vst1q_f32(x, a); + return x[0]; + } + template<> EIGEN_STRONG_INLINE int32_t pfirst(const Packet4i &a) + { + int32_t EIGEN_ALIGN16 x[4]; + vst1q_s32(x, a); + return x[0]; + } + + template<> EIGEN_STRONG_INLINE Packet4f preverse(const Packet4f &a) + { + float32x2_t a_lo, a_hi; + Packet4f a_r64; + + a_r64 = vrev64q_f32(a); + a_lo = vget_low_f32(a_r64); + a_hi = vget_high_f32(a_r64); + return vcombine_f32(a_hi, a_lo); + } + template<> EIGEN_STRONG_INLINE Packet4i preverse(const Packet4i &a) + { + int32x2_t a_lo, a_hi; + Packet4i a_r64; + + a_r64 = vrev64q_s32(a); + a_lo = vget_low_s32(a_r64); + a_hi = vget_high_s32(a_r64); + return vcombine_s32(a_hi, a_lo); + } + + template<> EIGEN_STRONG_INLINE Packet4f pabs(const Packet4f &a) { return vabsq_f32(a); } + template<> EIGEN_STRONG_INLINE Packet4i pabs(const Packet4i &a) { return vabsq_s32(a); } + + template<> EIGEN_STRONG_INLINE float predux(const Packet4f &a) + { + float32x2_t a_lo, a_hi, sum; + + a_lo = vget_low_f32(a); + a_hi = vget_high_f32(a); + sum = vpadd_f32(a_lo, a_hi); + sum = vpadd_f32(sum, sum); + return vget_lane_f32(sum, 0); + } + + template<> EIGEN_STRONG_INLINE Packet4f preduxp(const Packet4f *vecs) + { + float32x4x2_t vtrn1, vtrn2, res1, res2; + Packet4f sum1, sum2, sum; + + // NEON zip performs interleaving of the supplied vectors. + // We perform two interleaves in a row to acquire the transposed vector + vtrn1 = vzipq_f32(vecs[0], vecs[2]); + vtrn2 = vzipq_f32(vecs[1], vecs[3]); + res1 = vzipq_f32(vtrn1.val[0], vtrn2.val[0]); + res2 = vzipq_f32(vtrn1.val[1], vtrn2.val[1]); + + // Do the addition of the resulting vectors + sum1 = vaddq_f32(res1.val[0], res1.val[1]); + sum2 = vaddq_f32(res2.val[0], res2.val[1]); + sum = vaddq_f32(sum1, sum2); + + return sum; + } + + template<> EIGEN_STRONG_INLINE int32_t predux(const Packet4i &a) + { + int32x2_t a_lo, a_hi, sum; + + a_lo = vget_low_s32(a); + a_hi = vget_high_s32(a); + sum = vpadd_s32(a_lo, a_hi); + sum = vpadd_s32(sum, sum); + return vget_lane_s32(sum, 0); + } + + template<> EIGEN_STRONG_INLINE Packet4i preduxp(const Packet4i *vecs) + { + int32x4x2_t vtrn1, vtrn2, res1, res2; + Packet4i sum1, sum2, sum; + + // NEON zip performs interleaving of the supplied vectors. + // We perform two interleaves in a row to acquire the transposed vector + vtrn1 = vzipq_s32(vecs[0], vecs[2]); + vtrn2 = vzipq_s32(vecs[1], vecs[3]); + res1 = vzipq_s32(vtrn1.val[0], vtrn2.val[0]); + res2 = vzipq_s32(vtrn1.val[1], vtrn2.val[1]); + + // Do the addition of the resulting vectors + sum1 = vaddq_s32(res1.val[0], res1.val[1]); + sum2 = vaddq_s32(res2.val[0], res2.val[1]); + sum = vaddq_s32(sum1, sum2); + + return sum; + } + + // Other reduction functions: + // mul + template<> EIGEN_STRONG_INLINE float predux_mul(const Packet4f &a) + { + float32x2_t a_lo, a_hi, prod; + + // Get a_lo = |a1|a2| and a_hi = |a3|a4| + a_lo = vget_low_f32(a); + a_hi = vget_high_f32(a); + // Get the product of a_lo * a_hi -> |a1*a3|a2*a4| + prod = vmul_f32(a_lo, a_hi); + // Multiply prod with its swapped value |a2*a4|a1*a3| + prod = vmul_f32(prod, vrev64_f32(prod)); + + return vget_lane_f32(prod, 0); + } + template<> EIGEN_STRONG_INLINE int32_t predux_mul(const Packet4i &a) + { + int32x2_t a_lo, a_hi, prod; + + // Get a_lo = |a1|a2| and a_hi = |a3|a4| + a_lo = vget_low_s32(a); + a_hi = vget_high_s32(a); + // Get the product of a_lo * a_hi -> |a1*a3|a2*a4| + prod = vmul_s32(a_lo, a_hi); + // Multiply prod with its swapped value |a2*a4|a1*a3| + prod = vmul_s32(prod, vrev64_s32(prod)); + + return vget_lane_s32(prod, 0); + } + + // min + template<> EIGEN_STRONG_INLINE float predux_min(const Packet4f &a) + { + float32x2_t a_lo, a_hi, min; + + a_lo = vget_low_f32(a); + a_hi = vget_high_f32(a); + min = vpmin_f32(a_lo, a_hi); + min = vpmin_f32(min, min); + + return vget_lane_f32(min, 0); + } + + template<> EIGEN_STRONG_INLINE int32_t predux_min(const Packet4i &a) + { + int32x2_t a_lo, a_hi, min; + + a_lo = vget_low_s32(a); + a_hi = vget_high_s32(a); + min = vpmin_s32(a_lo, a_hi); + min = vpmin_s32(min, min); + + return vget_lane_s32(min, 0); + } + + // max + template<> EIGEN_STRONG_INLINE float predux_max(const Packet4f &a) + { + float32x2_t a_lo, a_hi, max; + + a_lo = vget_low_f32(a); + a_hi = vget_high_f32(a); + max = vpmax_f32(a_lo, a_hi); + max = vpmax_f32(max, max); + + return vget_lane_f32(max, 0); + } + + template<> EIGEN_STRONG_INLINE int32_t predux_max(const Packet4i &a) + { + int32x2_t a_lo, a_hi, max; + + a_lo = vget_low_s32(a); + a_hi = vget_high_s32(a); + max = vpmax_s32(a_lo, a_hi); + max = vpmax_s32(max, max); + + return vget_lane_s32(max, 0); + } // this PALIGN_NEON business is to work around a bug in LLVM Clang 3.0 causing incorrect compilation errors, // see bug 347 and this LLVM bug: http://llvm.org/bugs/show_bug.cgi?id=11074 -#define PALIGN_NEON(Offset,Type,Command) \ -template<>\ -struct palign_impl\ -{\ - EIGEN_STRONG_INLINE static void run(Type& first, const Type& second)\ - {\ - if (Offset!=0)\ - first = Command(first, second, Offset);\ - }\ -};\ - -PALIGN_NEON(0,Packet4f,vextq_f32) -PALIGN_NEON(1,Packet4f,vextq_f32) -PALIGN_NEON(2,Packet4f,vextq_f32) -PALIGN_NEON(3,Packet4f,vextq_f32) -PALIGN_NEON(0,Packet4i,vextq_s32) -PALIGN_NEON(1,Packet4i,vextq_s32) -PALIGN_NEON(2,Packet4i,vextq_s32) -PALIGN_NEON(3,Packet4i,vextq_s32) +#define PALIGN_NEON(Offset, Type, Command) \ + template<> struct palign_impl \ + { \ + EIGEN_STRONG_INLINE static void run(Type &first, const Type &second) \ + { \ + if (Offset != 0) first = Command(first, second, Offset); \ + } \ + }; + + PALIGN_NEON(0, Packet4f, vextq_f32) + PALIGN_NEON(1, Packet4f, vextq_f32) + PALIGN_NEON(2, Packet4f, vextq_f32) + PALIGN_NEON(3, Packet4f, vextq_f32) + PALIGN_NEON(0, Packet4i, vextq_s32) + PALIGN_NEON(1, Packet4i, vextq_s32) + PALIGN_NEON(2, Packet4i, vextq_s32) + PALIGN_NEON(3, Packet4i, vextq_s32) #undef PALIGN_NEON -EIGEN_DEVICE_FUNC inline void -ptranspose(PacketBlock& kernel) { - float32x4x2_t tmp1 = vzipq_f32(kernel.packet[0], kernel.packet[1]); - float32x4x2_t tmp2 = vzipq_f32(kernel.packet[2], kernel.packet[3]); - - kernel.packet[0] = vcombine_f32(vget_low_f32(tmp1.val[0]), vget_low_f32(tmp2.val[0])); - kernel.packet[1] = vcombine_f32(vget_high_f32(tmp1.val[0]), vget_high_f32(tmp2.val[0])); - kernel.packet[2] = vcombine_f32(vget_low_f32(tmp1.val[1]), vget_low_f32(tmp2.val[1])); - kernel.packet[3] = vcombine_f32(vget_high_f32(tmp1.val[1]), vget_high_f32(tmp2.val[1])); -} - -EIGEN_DEVICE_FUNC inline void -ptranspose(PacketBlock& kernel) { - int32x4x2_t tmp1 = vzipq_s32(kernel.packet[0], kernel.packet[1]); - int32x4x2_t tmp2 = vzipq_s32(kernel.packet[2], kernel.packet[3]); - kernel.packet[0] = vcombine_s32(vget_low_s32(tmp1.val[0]), vget_low_s32(tmp2.val[0])); - kernel.packet[1] = vcombine_s32(vget_high_s32(tmp1.val[0]), vget_high_s32(tmp2.val[0])); - kernel.packet[2] = vcombine_s32(vget_low_s32(tmp1.val[1]), vget_low_s32(tmp2.val[1])); - kernel.packet[3] = vcombine_s32(vget_high_s32(tmp1.val[1]), vget_high_s32(tmp2.val[1])); -} + EIGEN_DEVICE_FUNC inline void ptranspose(PacketBlock &kernel) + { + float32x4x2_t tmp1 = vzipq_f32(kernel.packet[0], kernel.packet[1]); + float32x4x2_t tmp2 = vzipq_f32(kernel.packet[2], kernel.packet[3]); + + kernel.packet[0] = vcombine_f32(vget_low_f32(tmp1.val[0]), vget_low_f32(tmp2.val[0])); + kernel.packet[1] = vcombine_f32(vget_high_f32(tmp1.val[0]), vget_high_f32(tmp2.val[0])); + kernel.packet[2] = vcombine_f32(vget_low_f32(tmp1.val[1]), vget_low_f32(tmp2.val[1])); + kernel.packet[3] = vcombine_f32(vget_high_f32(tmp1.val[1]), vget_high_f32(tmp2.val[1])); + } + + EIGEN_DEVICE_FUNC inline void ptranspose(PacketBlock &kernel) + { + int32x4x2_t tmp1 = vzipq_s32(kernel.packet[0], kernel.packet[1]); + int32x4x2_t tmp2 = vzipq_s32(kernel.packet[2], kernel.packet[3]); + kernel.packet[0] = vcombine_s32(vget_low_s32(tmp1.val[0]), vget_low_s32(tmp2.val[0])); + kernel.packet[1] = vcombine_s32(vget_high_s32(tmp1.val[0]), vget_high_s32(tmp2.val[0])); + kernel.packet[2] = vcombine_s32(vget_low_s32(tmp1.val[1]), vget_low_s32(tmp2.val[1])); + kernel.packet[3] = vcombine_s32(vget_high_s32(tmp1.val[1]), vget_high_s32(tmp2.val[1])); + } //---------- double ---------- @@ -567,194 +654,243 @@ ptranspose(PacketBlock& kernel) { #if EIGEN_ARCH_ARM64 && !EIGEN_APPLE_DOUBLE_NEON_BUG -// Bug 907: workaround missing declarations of the following two functions in the ADK -// Defining these functions as templates ensures that if these intrinsics are -// already defined in arm_neon.h, then our workaround doesn't cause a conflict -// and has lower priority in overload resolution. -template -uint64x2_t vreinterpretq_u64_f64(T a) -{ - return (uint64x2_t) a; -} - -template -float64x2_t vreinterpretq_f64_u64(T a) -{ - return (float64x2_t) a; -} - -typedef float64x2_t Packet2d; -typedef float64x1_t Packet1d; - -template<> struct packet_traits : default_packet_traits -{ - typedef Packet2d type; - typedef Packet2d half; - enum { - Vectorizable = 1, - AlignedOnScalar = 1, - size = 2, - HasHalfPacket=0, - - HasDiv = 1, - // FIXME check the Has* - HasSin = 0, - HasCos = 0, - HasLog = 0, - HasExp = 0, - HasSqrt = 0 + // Bug 907: workaround missing declarations of the following two functions in the ADK + // Defining these functions as templates ensures that if these intrinsics are + // already defined in arm_neon.h, then our workaround doesn't cause a conflict + // and has lower priority in overload resolution. + template uint64x2_t vreinterpretq_u64_f64(T a) { return (uint64x2_t)a; } + + template float64x2_t vreinterpretq_f64_u64(T a) { return (float64x2_t)a; } + + typedef float64x2_t Packet2d; + typedef float64x1_t Packet1d; + + template<> struct packet_traits : default_packet_traits + { + typedef Packet2d type; + typedef Packet2d half; + enum { + Vectorizable = 1, + AlignedOnScalar = 1, + size = 2, + HasHalfPacket = 0, + + HasDiv = 1, + // FIXME check the Has* + HasSin = 0, + HasCos = 0, + HasLog = 0, + HasExp = 0, + HasSqrt = 0 + }; }; -}; -template<> struct unpacket_traits { typedef double type; enum {size=2, alignment=Aligned16}; typedef Packet2d half; }; + template<> struct unpacket_traits + { + typedef double type; + enum { size = 2, alignment = Aligned16 }; + typedef Packet2d half; + }; -template<> EIGEN_STRONG_INLINE Packet2d pset1(const double& from) { return vdupq_n_f64(from); } + template<> EIGEN_STRONG_INLINE Packet2d pset1(const double &from) { return vdupq_n_f64(from); } -template<> EIGEN_STRONG_INLINE Packet2d plset(const double& a) -{ - const double countdown_raw[] = {0.0,1.0}; - const Packet2d countdown = vld1q_f64(countdown_raw); - return vaddq_f64(pset1(a), countdown); -} -template<> EIGEN_STRONG_INLINE Packet2d padd(const Packet2d& a, const Packet2d& b) { return vaddq_f64(a,b); } + template<> EIGEN_STRONG_INLINE Packet2d plset(const double &a) + { + const double countdown_raw[] = { 0.0, 1.0 }; + const Packet2d countdown = vld1q_f64(countdown_raw); + return vaddq_f64(pset1(a), countdown); + } + template<> EIGEN_STRONG_INLINE Packet2d padd(const Packet2d &a, const Packet2d &b) + { + return vaddq_f64(a, b); + } -template<> EIGEN_STRONG_INLINE Packet2d psub(const Packet2d& a, const Packet2d& b) { return vsubq_f64(a,b); } + template<> EIGEN_STRONG_INLINE Packet2d psub(const Packet2d &a, const Packet2d &b) + { + return vsubq_f64(a, b); + } -template<> EIGEN_STRONG_INLINE Packet2d pnegate(const Packet2d& a) { return vnegq_f64(a); } + template<> EIGEN_STRONG_INLINE Packet2d pnegate(const Packet2d &a) { return vnegq_f64(a); } -template<> EIGEN_STRONG_INLINE Packet2d pconj(const Packet2d& a) { return a; } + template<> EIGEN_STRONG_INLINE Packet2d pconj(const Packet2d &a) { return a; } -template<> EIGEN_STRONG_INLINE Packet2d pmul(const Packet2d& a, const Packet2d& b) { return vmulq_f64(a,b); } + template<> EIGEN_STRONG_INLINE Packet2d pmul(const Packet2d &a, const Packet2d &b) + { + return vmulq_f64(a, b); + } -template<> EIGEN_STRONG_INLINE Packet2d pdiv(const Packet2d& a, const Packet2d& b) { return vdivq_f64(a,b); } + template<> EIGEN_STRONG_INLINE Packet2d pdiv(const Packet2d &a, const Packet2d &b) + { + return vdivq_f64(a, b); + } #ifdef __ARM_FEATURE_FMA -// See bug 936. See above comment about FMA for float. -template<> EIGEN_STRONG_INLINE Packet2d pmadd(const Packet2d& a, const Packet2d& b, const Packet2d& c) { return vfmaq_f64(c,a,b); } + // See bug 936. See above comment about FMA for float. + template<> EIGEN_STRONG_INLINE Packet2d pmadd(const Packet2d &a, const Packet2d &b, const Packet2d &c) + { + return vfmaq_f64(c, a, b); + } #else -template<> EIGEN_STRONG_INLINE Packet2d pmadd(const Packet2d& a, const Packet2d& b, const Packet2d& c) { return vmlaq_f64(c,a,b); } + template<> EIGEN_STRONG_INLINE Packet2d pmadd(const Packet2d &a, const Packet2d &b, const Packet2d &c) + { + return vmlaq_f64(c, a, b); + } #endif -template<> EIGEN_STRONG_INLINE Packet2d pmin(const Packet2d& a, const Packet2d& b) { return vminq_f64(a,b); } + template<> EIGEN_STRONG_INLINE Packet2d pmin(const Packet2d &a, const Packet2d &b) + { + return vminq_f64(a, b); + } -template<> EIGEN_STRONG_INLINE Packet2d pmax(const Packet2d& a, const Packet2d& b) { return vmaxq_f64(a,b); } + template<> EIGEN_STRONG_INLINE Packet2d pmax(const Packet2d &a, const Packet2d &b) + { + return vmaxq_f64(a, b); + } -// Logical Operations are not supported for float, so we have to reinterpret casts using NEON intrinsics -template<> EIGEN_STRONG_INLINE Packet2d pand(const Packet2d& a, const Packet2d& b) -{ - return vreinterpretq_f64_u64(vandq_u64(vreinterpretq_u64_f64(a),vreinterpretq_u64_f64(b))); -} + // Logical Operations are not supported for float, so we have to reinterpret casts using NEON intrinsics + template<> EIGEN_STRONG_INLINE Packet2d pand(const Packet2d &a, const Packet2d &b) + { + return vreinterpretq_f64_u64(vandq_u64(vreinterpretq_u64_f64(a), vreinterpretq_u64_f64(b))); + } -template<> EIGEN_STRONG_INLINE Packet2d por(const Packet2d& a, const Packet2d& b) -{ - return vreinterpretq_f64_u64(vorrq_u64(vreinterpretq_u64_f64(a),vreinterpretq_u64_f64(b))); -} + template<> EIGEN_STRONG_INLINE Packet2d por(const Packet2d &a, const Packet2d &b) + { + return vreinterpretq_f64_u64(vorrq_u64(vreinterpretq_u64_f64(a), vreinterpretq_u64_f64(b))); + } -template<> EIGEN_STRONG_INLINE Packet2d pxor(const Packet2d& a, const Packet2d& b) -{ - return vreinterpretq_f64_u64(veorq_u64(vreinterpretq_u64_f64(a),vreinterpretq_u64_f64(b))); -} + template<> EIGEN_STRONG_INLINE Packet2d pxor(const Packet2d &a, const Packet2d &b) + { + return vreinterpretq_f64_u64(veorq_u64(vreinterpretq_u64_f64(a), vreinterpretq_u64_f64(b))); + } -template<> EIGEN_STRONG_INLINE Packet2d pandnot(const Packet2d& a, const Packet2d& b) -{ - return vreinterpretq_f64_u64(vbicq_u64(vreinterpretq_u64_f64(a),vreinterpretq_u64_f64(b))); -} + template<> EIGEN_STRONG_INLINE Packet2d pandnot(const Packet2d &a, const Packet2d &b) + { + return vreinterpretq_f64_u64(vbicq_u64(vreinterpretq_u64_f64(a), vreinterpretq_u64_f64(b))); + } -template<> EIGEN_STRONG_INLINE Packet2d pload(const double* from) { EIGEN_DEBUG_ALIGNED_LOAD return vld1q_f64(from); } + template<> EIGEN_STRONG_INLINE Packet2d pload(const double *from) + { + EIGEN_DEBUG_ALIGNED_LOAD return vld1q_f64(from); + } -template<> EIGEN_STRONG_INLINE Packet2d ploadu(const double* from) { EIGEN_DEBUG_UNALIGNED_LOAD return vld1q_f64(from); } + template<> EIGEN_STRONG_INLINE Packet2d ploadu(const double *from) + { + EIGEN_DEBUG_UNALIGNED_LOAD return vld1q_f64(from); + } -template<> EIGEN_STRONG_INLINE Packet2d ploaddup(const double* from) -{ - return vld1q_dup_f64(from); -} -template<> EIGEN_STRONG_INLINE void pstore(double* to, const Packet2d& from) { EIGEN_DEBUG_ALIGNED_STORE vst1q_f64(to, from); } + template<> EIGEN_STRONG_INLINE Packet2d ploaddup(const double *from) { return vld1q_dup_f64(from); } + template<> EIGEN_STRONG_INLINE void pstore(double *to, const Packet2d &from) + { + EIGEN_DEBUG_ALIGNED_STORE vst1q_f64(to, from); + } -template<> EIGEN_STRONG_INLINE void pstoreu(double* to, const Packet2d& from) { EIGEN_DEBUG_UNALIGNED_STORE vst1q_f64(to, from); } + template<> EIGEN_STRONG_INLINE void pstoreu(double *to, const Packet2d &from) + { + EIGEN_DEBUG_UNALIGNED_STORE vst1q_f64(to, from); + } -template<> EIGEN_DEVICE_FUNC inline Packet2d pgather(const double* from, Index stride) -{ - Packet2d res = pset1(0.0); - res = vsetq_lane_f64(from[0*stride], res, 0); - res = vsetq_lane_f64(from[1*stride], res, 1); - return res; -} -template<> EIGEN_DEVICE_FUNC inline void pscatter(double* to, const Packet2d& from, Index stride) -{ - to[stride*0] = vgetq_lane_f64(from, 0); - to[stride*1] = vgetq_lane_f64(from, 1); -} -template<> EIGEN_STRONG_INLINE void prefetch(const double* addr) { EIGEN_ARM_PREFETCH(addr); } + template<> EIGEN_DEVICE_FUNC inline Packet2d pgather(const double *from, Index stride) + { + Packet2d res = pset1(0.0); + res = vsetq_lane_f64(from[0 * stride], res, 0); + res = vsetq_lane_f64(from[1 * stride], res, 1); + return res; + } + template<> EIGEN_DEVICE_FUNC inline void pscatter(double *to, const Packet2d &from, Index stride) + { + to[stride * 0] = vgetq_lane_f64(from, 0); + to[stride * 1] = vgetq_lane_f64(from, 1); + } + template<> EIGEN_STRONG_INLINE void prefetch(const double *addr) { EIGEN_ARM_PREFETCH(addr); } -// FIXME only store the 2 first elements ? -template<> EIGEN_STRONG_INLINE double pfirst(const Packet2d& a) { return vgetq_lane_f64(a, 0); } + // FIXME only store the 2 first elements ? + template<> EIGEN_STRONG_INLINE double pfirst(const Packet2d &a) { return vgetq_lane_f64(a, 0); } -template<> EIGEN_STRONG_INLINE Packet2d preverse(const Packet2d& a) { return vcombine_f64(vget_high_f64(a), vget_low_f64(a)); } + template<> EIGEN_STRONG_INLINE Packet2d preverse(const Packet2d &a) + { + return vcombine_f64(vget_high_f64(a), vget_low_f64(a)); + } -template<> EIGEN_STRONG_INLINE Packet2d pabs(const Packet2d& a) { return vabsq_f64(a); } + template<> EIGEN_STRONG_INLINE Packet2d pabs(const Packet2d &a) { return vabsq_f64(a); } #if EIGEN_COMP_CLANG && defined(__apple_build_version__) -// workaround ICE, see bug 907 -template<> EIGEN_STRONG_INLINE double predux(const Packet2d& a) { return (vget_low_f64(a) + vget_high_f64(a))[0]; } + // workaround ICE, see bug 907 + template<> EIGEN_STRONG_INLINE double predux(const Packet2d &a) + { + return (vget_low_f64(a) + vget_high_f64(a))[0]; + } #else -template<> EIGEN_STRONG_INLINE double predux(const Packet2d& a) { return vget_lane_f64(vget_low_f64(a) + vget_high_f64(a), 0); } + template<> EIGEN_STRONG_INLINE double predux(const Packet2d &a) + { + return vget_lane_f64(vget_low_f64(a) + vget_high_f64(a), 0); + } #endif -template<> EIGEN_STRONG_INLINE Packet2d preduxp(const Packet2d* vecs) -{ - float64x2_t trn1, trn2; + template<> EIGEN_STRONG_INLINE Packet2d preduxp(const Packet2d *vecs) + { + float64x2_t trn1, trn2; - // NEON zip performs interleaving of the supplied vectors. - // We perform two interleaves in a row to acquire the transposed vector - trn1 = vzip1q_f64(vecs[0], vecs[1]); - trn2 = vzip2q_f64(vecs[0], vecs[1]); + // NEON zip performs interleaving of the supplied vectors. + // We perform two interleaves in a row to acquire the transposed vector + trn1 = vzip1q_f64(vecs[0], vecs[1]); + trn2 = vzip2q_f64(vecs[0], vecs[1]); - // Do the addition of the resulting vectors - return vaddq_f64(trn1, trn2); -} + // Do the addition of the resulting vectors + return vaddq_f64(trn1, trn2); + } // Other reduction functions: // mul #if EIGEN_COMP_CLANG && defined(__apple_build_version__) -template<> EIGEN_STRONG_INLINE double predux_mul(const Packet2d& a) { return (vget_low_f64(a) * vget_high_f64(a))[0]; } + template<> EIGEN_STRONG_INLINE double predux_mul(const Packet2d &a) + { + return (vget_low_f64(a) * vget_high_f64(a))[0]; + } #else -template<> EIGEN_STRONG_INLINE double predux_mul(const Packet2d& a) { return vget_lane_f64(vget_low_f64(a) * vget_high_f64(a), 0); } + template<> EIGEN_STRONG_INLINE double predux_mul(const Packet2d &a) + { + return vget_lane_f64(vget_low_f64(a) * vget_high_f64(a), 0); + } #endif -// min -template<> EIGEN_STRONG_INLINE double predux_min(const Packet2d& a) { return vgetq_lane_f64(vpminq_f64(a, a), 0); } + // min + template<> EIGEN_STRONG_INLINE double predux_min(const Packet2d &a) + { + return vgetq_lane_f64(vpminq_f64(a, a), 0); + } -// max -template<> EIGEN_STRONG_INLINE double predux_max(const Packet2d& a) { return vgetq_lane_f64(vpmaxq_f64(a, a), 0); } + // max + template<> EIGEN_STRONG_INLINE double predux_max(const Packet2d &a) + { + return vgetq_lane_f64(vpmaxq_f64(a, a), 0); + } // this PALIGN_NEON business is to work around a bug in LLVM Clang 3.0 causing incorrect compilation errors, // see bug 347 and this LLVM bug: http://llvm.org/bugs/show_bug.cgi?id=11074 -#define PALIGN_NEON(Offset,Type,Command) \ -template<>\ -struct palign_impl\ -{\ - EIGEN_STRONG_INLINE static void run(Type& first, const Type& second)\ - {\ - if (Offset!=0)\ - first = Command(first, second, Offset);\ - }\ -};\ - -PALIGN_NEON(0,Packet2d,vextq_f64) -PALIGN_NEON(1,Packet2d,vextq_f64) +#define PALIGN_NEON(Offset, Type, Command) \ + template<> struct palign_impl \ + { \ + EIGEN_STRONG_INLINE static void run(Type &first, const Type &second) \ + { \ + if (Offset != 0) first = Command(first, second, Offset); \ + } \ + }; + + PALIGN_NEON(0, Packet2d, vextq_f64) + PALIGN_NEON(1, Packet2d, vextq_f64) #undef PALIGN_NEON -EIGEN_DEVICE_FUNC inline void -ptranspose(PacketBlock& kernel) { - float64x2_t trn1 = vzip1q_f64(kernel.packet[0], kernel.packet[1]); - float64x2_t trn2 = vzip2q_f64(kernel.packet[0], kernel.packet[1]); + EIGEN_DEVICE_FUNC inline void ptranspose(PacketBlock &kernel) + { + float64x2_t trn1 = vzip1q_f64(kernel.packet[0], kernel.packet[1]); + float64x2_t trn2 = vzip2q_f64(kernel.packet[0], kernel.packet[1]); - kernel.packet[0] = trn1; - kernel.packet[1] = trn2; -} -#endif // EIGEN_ARCH_ARM64 + kernel.packet[0] = trn1; + kernel.packet[1] = trn2; + } +#endif// EIGEN_ARCH_ARM64 -} // end namespace internal +}// end namespace internal -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_PACKET_MATH_NEON_H +#endif// EIGEN_PACKET_MATH_NEON_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/arch/SSE/Complex.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/arch/SSE/Complex.h index d075043c..d9f7333b 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/arch/SSE/Complex.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/arch/SSE/Complex.h @@ -14,458 +14,543 @@ namespace Eigen { namespace internal { -//---------- float ---------- -struct Packet2cf -{ - EIGEN_STRONG_INLINE Packet2cf() {} - EIGEN_STRONG_INLINE explicit Packet2cf(const __m128& a) : v(a) {} - __m128 v; -}; + //---------- float ---------- + struct Packet2cf + { + EIGEN_STRONG_INLINE Packet2cf() {} + EIGEN_STRONG_INLINE explicit Packet2cf(const __m128 &a) : v(a) {} + __m128 v; + }; // Use the packet_traits defined in AVX/PacketMath.h instead if we're going // to leverage AVX instructions. #ifndef EIGEN_VECTORIZE_AVX -template<> struct packet_traits > : default_packet_traits -{ - typedef Packet2cf type; - typedef Packet2cf half; - enum { - Vectorizable = 1, - AlignedOnScalar = 1, - size = 2, - HasHalfPacket = 0, - - HasAdd = 1, - HasSub = 1, - HasMul = 1, - HasDiv = 1, - HasNegate = 1, - HasAbs = 0, - HasAbs2 = 0, - HasMin = 0, - HasMax = 0, - HasSetLinear = 0, - HasBlend = 1 + template<> struct packet_traits> : default_packet_traits + { + typedef Packet2cf type; + typedef Packet2cf half; + enum { + Vectorizable = 1, + AlignedOnScalar = 1, + size = 2, + HasHalfPacket = 0, + + HasAdd = 1, + HasSub = 1, + HasMul = 1, + HasDiv = 1, + HasNegate = 1, + HasAbs = 0, + HasAbs2 = 0, + HasMin = 0, + HasMax = 0, + HasSetLinear = 0, + HasBlend = 1 + }; + }; +#endif + + template<> struct unpacket_traits + { + typedef std::complex type; + enum { size = 2, alignment = Aligned16 }; + typedef Packet2cf half; + }; + + template<> EIGEN_STRONG_INLINE Packet2cf padd(const Packet2cf &a, const Packet2cf &b) + { + return Packet2cf(_mm_add_ps(a.v, b.v)); + } + template<> EIGEN_STRONG_INLINE Packet2cf psub(const Packet2cf &a, const Packet2cf &b) + { + return Packet2cf(_mm_sub_ps(a.v, b.v)); + } + template<> EIGEN_STRONG_INLINE Packet2cf pnegate(const Packet2cf &a) + { + const __m128 mask = _mm_castsi128_ps(_mm_setr_epi32(0x80000000, 0x80000000, 0x80000000, 0x80000000)); + return Packet2cf(_mm_xor_ps(a.v, mask)); + } + template<> EIGEN_STRONG_INLINE Packet2cf pconj(const Packet2cf &a) + { + const __m128 mask = _mm_castsi128_ps(_mm_setr_epi32(0x00000000, 0x80000000, 0x00000000, 0x80000000)); + return Packet2cf(_mm_xor_ps(a.v, mask)); + } + + template<> EIGEN_STRONG_INLINE Packet2cf pmul(const Packet2cf &a, const Packet2cf &b) + { +#ifdef EIGEN_VECTORIZE_SSE3 + return Packet2cf(_mm_addsub_ps( + _mm_mul_ps(_mm_moveldup_ps(a.v), b.v), _mm_mul_ps(_mm_movehdup_ps(a.v), vec4f_swizzle1(b.v, 1, 0, 3, 2)))); + // return Packet2cf(_mm_addsub_ps(_mm_mul_ps(vec4f_swizzle1(a.v, 0, 0, 2, 2), b.v), + // _mm_mul_ps(vec4f_swizzle1(a.v, 1, 1, 3, 3), + // vec4f_swizzle1(b.v, 1, 0, 3, 2)))); +#else + const __m128 mask = _mm_castsi128_ps(_mm_setr_epi32(0x80000000, 0x00000000, 0x80000000, 0x00000000)); + return Packet2cf(_mm_add_ps(_mm_mul_ps(vec4f_swizzle1(a.v, 0, 0, 2, 2), b.v), + _mm_xor_ps(_mm_mul_ps(vec4f_swizzle1(a.v, 1, 1, 3, 3), vec4f_swizzle1(b.v, 1, 0, 3, 2)), mask))); +#endif + } + + template<> EIGEN_STRONG_INLINE Packet2cf pand(const Packet2cf &a, const Packet2cf &b) + { + return Packet2cf(_mm_and_ps(a.v, b.v)); + } + template<> EIGEN_STRONG_INLINE Packet2cf por(const Packet2cf &a, const Packet2cf &b) + { + return Packet2cf(_mm_or_ps(a.v, b.v)); + } + template<> EIGEN_STRONG_INLINE Packet2cf pxor(const Packet2cf &a, const Packet2cf &b) + { + return Packet2cf(_mm_xor_ps(a.v, b.v)); + } + template<> EIGEN_STRONG_INLINE Packet2cf pandnot(const Packet2cf &a, const Packet2cf &b) + { + return Packet2cf(_mm_andnot_ps(a.v, b.v)); + } + + template<> EIGEN_STRONG_INLINE Packet2cf pload(const std::complex *from) + { + EIGEN_DEBUG_ALIGNED_LOAD return Packet2cf(pload(&numext::real_ref(*from))); + } + template<> EIGEN_STRONG_INLINE Packet2cf ploadu(const std::complex *from) + { + EIGEN_DEBUG_UNALIGNED_LOAD return Packet2cf(ploadu(&numext::real_ref(*from))); + } + + template<> EIGEN_STRONG_INLINE Packet2cf pset1(const std::complex &from) + { + Packet2cf res; +#if EIGEN_GNUC_AT_MOST(4, 2) + // Workaround annoying "may be used uninitialized in this function" warning with gcc 4.2 + res.v = _mm_loadl_pi(_mm_set1_ps(0.0f), reinterpret_cast(&from)); +#elif EIGEN_GNUC_AT_LEAST(4, 6) +// Suppress annoying "may be used uninitialized in this function" warning with gcc >= 4.6 +#pragma GCC diagnostic push +#pragma GCC diagnostic ignored "-Wuninitialized" + res.v = _mm_loadl_pi(res.v, (const __m64 *)&from); +#pragma GCC diagnostic pop +#else + res.v = _mm_loadl_pi(res.v, (const __m64 *)&from); +#endif + return Packet2cf(_mm_movelh_ps(res.v, res.v)); + } + + template<> EIGEN_STRONG_INLINE Packet2cf ploaddup(const std::complex *from) + { + return pset1(*from); + } + + template<> EIGEN_STRONG_INLINE void pstore>(std::complex *to, const Packet2cf &from) + { + EIGEN_DEBUG_ALIGNED_STORE pstore(&numext::real_ref(*to), Packet4f(from.v)); + } + template<> EIGEN_STRONG_INLINE void pstoreu>(std::complex *to, const Packet2cf &from) + { + EIGEN_DEBUG_UNALIGNED_STORE pstoreu(&numext::real_ref(*to), Packet4f(from.v)); + } + + + template<> + EIGEN_DEVICE_FUNC inline Packet2cf pgather, Packet2cf>(const std::complex *from, + Index stride) + { + return Packet2cf(_mm_set_ps(std::imag(from[1 * stride]), + std::real(from[1 * stride]), + std::imag(from[0 * stride]), + std::real(from[0 * stride]))); + } + + template<> + EIGEN_DEVICE_FUNC inline void + pscatter, Packet2cf>(std::complex *to, const Packet2cf &from, Index stride) + { + to[stride * 0] = std::complex( + _mm_cvtss_f32(_mm_shuffle_ps(from.v, from.v, 0)), _mm_cvtss_f32(_mm_shuffle_ps(from.v, from.v, 1))); + to[stride * 1] = std::complex( + _mm_cvtss_f32(_mm_shuffle_ps(from.v, from.v, 2)), _mm_cvtss_f32(_mm_shuffle_ps(from.v, from.v, 3))); + } + + template<> EIGEN_STRONG_INLINE void prefetch>(const std::complex *addr) + { + _mm_prefetch((SsePrefetchPtrType)(addr), _MM_HINT_T0); + } + + template<> EIGEN_STRONG_INLINE std::complex pfirst(const Packet2cf &a) + { +#if EIGEN_GNUC_AT_MOST(4, 3) + // Workaround gcc 4.2 ICE - this is not performance wise ideal, but who cares... + // This workaround also fix invalid code generation with gcc 4.3 + EIGEN_ALIGN16 std::complex res[2]; + _mm_store_ps((float *)res, a.v); + return res[0]; +#else + std::complex res; + _mm_storel_pi((__m64 *)&res, a.v); + return res; +#endif + } + + template<> EIGEN_STRONG_INLINE Packet2cf preverse(const Packet2cf &a) + { + return Packet2cf(_mm_castpd_ps(preverse(Packet2d(_mm_castps_pd(a.v))))); + } + + template<> EIGEN_STRONG_INLINE std::complex predux(const Packet2cf &a) + { + return pfirst(Packet2cf(_mm_add_ps(a.v, _mm_movehl_ps(a.v, a.v)))); + } + + template<> EIGEN_STRONG_INLINE Packet2cf preduxp(const Packet2cf *vecs) + { + return Packet2cf(_mm_add_ps(_mm_movelh_ps(vecs[0].v, vecs[1].v), _mm_movehl_ps(vecs[1].v, vecs[0].v))); + } + + template<> EIGEN_STRONG_INLINE std::complex predux_mul(const Packet2cf &a) + { + return pfirst(pmul(a, Packet2cf(_mm_movehl_ps(a.v, a.v)))); + } + + template struct palign_impl + { + static EIGEN_STRONG_INLINE void run(Packet2cf &first, const Packet2cf &second) + { + if (Offset == 1) { + first.v = _mm_movehl_ps(first.v, first.v); + first.v = _mm_movelh_ps(first.v, second.v); + } + } }; -}; + + template<> struct conj_helper + { + EIGEN_STRONG_INLINE Packet2cf pmadd(const Packet2cf &x, const Packet2cf &y, const Packet2cf &c) const + { + return padd(pmul(x, y), c); + } + + EIGEN_STRONG_INLINE Packet2cf pmul(const Packet2cf &a, const Packet2cf &b) const + { +#ifdef EIGEN_VECTORIZE_SSE3 + return internal::pmul(a, pconj(b)); +#else + const __m128 mask = _mm_castsi128_ps(_mm_setr_epi32(0x00000000, 0x80000000, 0x00000000, 0x80000000)); + return Packet2cf(_mm_add_ps(_mm_xor_ps(_mm_mul_ps(vec4f_swizzle1(a.v, 0, 0, 2, 2), b.v), mask), + _mm_mul_ps(vec4f_swizzle1(a.v, 1, 1, 3, 3), vec4f_swizzle1(b.v, 1, 0, 3, 2)))); #endif + } + }; + + template<> struct conj_helper + { + EIGEN_STRONG_INLINE Packet2cf pmadd(const Packet2cf &x, const Packet2cf &y, const Packet2cf &c) const + { + return padd(pmul(x, y), c); + } -template<> struct unpacket_traits { typedef std::complex type; enum {size=2, alignment=Aligned16}; typedef Packet2cf half; }; - -template<> EIGEN_STRONG_INLINE Packet2cf padd(const Packet2cf& a, const Packet2cf& b) { return Packet2cf(_mm_add_ps(a.v,b.v)); } -template<> EIGEN_STRONG_INLINE Packet2cf psub(const Packet2cf& a, const Packet2cf& b) { return Packet2cf(_mm_sub_ps(a.v,b.v)); } -template<> EIGEN_STRONG_INLINE Packet2cf pnegate(const Packet2cf& a) -{ - const __m128 mask = _mm_castsi128_ps(_mm_setr_epi32(0x80000000,0x80000000,0x80000000,0x80000000)); - return Packet2cf(_mm_xor_ps(a.v,mask)); -} -template<> EIGEN_STRONG_INLINE Packet2cf pconj(const Packet2cf& a) -{ - const __m128 mask = _mm_castsi128_ps(_mm_setr_epi32(0x00000000,0x80000000,0x00000000,0x80000000)); - return Packet2cf(_mm_xor_ps(a.v,mask)); -} - -template<> EIGEN_STRONG_INLINE Packet2cf pmul(const Packet2cf& a, const Packet2cf& b) -{ - #ifdef EIGEN_VECTORIZE_SSE3 - return Packet2cf(_mm_addsub_ps(_mm_mul_ps(_mm_moveldup_ps(a.v), b.v), - _mm_mul_ps(_mm_movehdup_ps(a.v), - vec4f_swizzle1(b.v, 1, 0, 3, 2)))); -// return Packet2cf(_mm_addsub_ps(_mm_mul_ps(vec4f_swizzle1(a.v, 0, 0, 2, 2), b.v), -// _mm_mul_ps(vec4f_swizzle1(a.v, 1, 1, 3, 3), -// vec4f_swizzle1(b.v, 1, 0, 3, 2)))); - #else - const __m128 mask = _mm_castsi128_ps(_mm_setr_epi32(0x80000000,0x00000000,0x80000000,0x00000000)); - return Packet2cf(_mm_add_ps(_mm_mul_ps(vec4f_swizzle1(a.v, 0, 0, 2, 2), b.v), - _mm_xor_ps(_mm_mul_ps(vec4f_swizzle1(a.v, 1, 1, 3, 3), - vec4f_swizzle1(b.v, 1, 0, 3, 2)), mask))); - #endif -} - -template<> EIGEN_STRONG_INLINE Packet2cf pand (const Packet2cf& a, const Packet2cf& b) { return Packet2cf(_mm_and_ps(a.v,b.v)); } -template<> EIGEN_STRONG_INLINE Packet2cf por (const Packet2cf& a, const Packet2cf& b) { return Packet2cf(_mm_or_ps(a.v,b.v)); } -template<> EIGEN_STRONG_INLINE Packet2cf pxor (const Packet2cf& a, const Packet2cf& b) { return Packet2cf(_mm_xor_ps(a.v,b.v)); } -template<> EIGEN_STRONG_INLINE Packet2cf pandnot(const Packet2cf& a, const Packet2cf& b) { return Packet2cf(_mm_andnot_ps(a.v,b.v)); } - -template<> EIGEN_STRONG_INLINE Packet2cf pload (const std::complex* from) { EIGEN_DEBUG_ALIGNED_LOAD return Packet2cf(pload(&numext::real_ref(*from))); } -template<> EIGEN_STRONG_INLINE Packet2cf ploadu(const std::complex* from) { EIGEN_DEBUG_UNALIGNED_LOAD return Packet2cf(ploadu(&numext::real_ref(*from))); } - -template<> EIGEN_STRONG_INLINE Packet2cf pset1(const std::complex& from) -{ - Packet2cf res; -#if EIGEN_GNUC_AT_MOST(4,2) - // Workaround annoying "may be used uninitialized in this function" warning with gcc 4.2 - res.v = _mm_loadl_pi(_mm_set1_ps(0.0f), reinterpret_cast(&from)); -#elif EIGEN_GNUC_AT_LEAST(4,6) - // Suppress annoying "may be used uninitialized in this function" warning with gcc >= 4.6 - #pragma GCC diagnostic push - #pragma GCC diagnostic ignored "-Wuninitialized" - res.v = _mm_loadl_pi(res.v, (const __m64*)&from); - #pragma GCC diagnostic pop + EIGEN_STRONG_INLINE Packet2cf pmul(const Packet2cf &a, const Packet2cf &b) const + { +#ifdef EIGEN_VECTORIZE_SSE3 + return internal::pmul(pconj(a), b); #else - res.v = _mm_loadl_pi(res.v, (const __m64*)&from); + const __m128 mask = _mm_castsi128_ps(_mm_setr_epi32(0x00000000, 0x80000000, 0x00000000, 0x80000000)); + return Packet2cf(_mm_add_ps(_mm_mul_ps(vec4f_swizzle1(a.v, 0, 0, 2, 2), b.v), + _mm_xor_ps(_mm_mul_ps(vec4f_swizzle1(a.v, 1, 1, 3, 3), vec4f_swizzle1(b.v, 1, 0, 3, 2)), mask))); #endif - return Packet2cf(_mm_movelh_ps(res.v,res.v)); -} - -template<> EIGEN_STRONG_INLINE Packet2cf ploaddup(const std::complex* from) { return pset1(*from); } - -template<> EIGEN_STRONG_INLINE void pstore >(std::complex * to, const Packet2cf& from) { EIGEN_DEBUG_ALIGNED_STORE pstore(&numext::real_ref(*to), Packet4f(from.v)); } -template<> EIGEN_STRONG_INLINE void pstoreu >(std::complex * to, const Packet2cf& from) { EIGEN_DEBUG_UNALIGNED_STORE pstoreu(&numext::real_ref(*to), Packet4f(from.v)); } - - -template<> EIGEN_DEVICE_FUNC inline Packet2cf pgather, Packet2cf>(const std::complex* from, Index stride) -{ - return Packet2cf(_mm_set_ps(std::imag(from[1*stride]), std::real(from[1*stride]), - std::imag(from[0*stride]), std::real(from[0*stride]))); -} - -template<> EIGEN_DEVICE_FUNC inline void pscatter, Packet2cf>(std::complex* to, const Packet2cf& from, Index stride) -{ - to[stride*0] = std::complex(_mm_cvtss_f32(_mm_shuffle_ps(from.v, from.v, 0)), - _mm_cvtss_f32(_mm_shuffle_ps(from.v, from.v, 1))); - to[stride*1] = std::complex(_mm_cvtss_f32(_mm_shuffle_ps(from.v, from.v, 2)), - _mm_cvtss_f32(_mm_shuffle_ps(from.v, from.v, 3))); -} - -template<> EIGEN_STRONG_INLINE void prefetch >(const std::complex * addr) { _mm_prefetch((SsePrefetchPtrType)(addr), _MM_HINT_T0); } - -template<> EIGEN_STRONG_INLINE std::complex pfirst(const Packet2cf& a) -{ - #if EIGEN_GNUC_AT_MOST(4,3) - // Workaround gcc 4.2 ICE - this is not performance wise ideal, but who cares... - // This workaround also fix invalid code generation with gcc 4.3 - EIGEN_ALIGN16 std::complex res[2]; - _mm_store_ps((float*)res, a.v); - return res[0]; - #else - std::complex res; - _mm_storel_pi((__m64*)&res, a.v); - return res; - #endif -} - -template<> EIGEN_STRONG_INLINE Packet2cf preverse(const Packet2cf& a) { return Packet2cf(_mm_castpd_ps(preverse(Packet2d(_mm_castps_pd(a.v))))); } - -template<> EIGEN_STRONG_INLINE std::complex predux(const Packet2cf& a) -{ - return pfirst(Packet2cf(_mm_add_ps(a.v, _mm_movehl_ps(a.v,a.v)))); -} - -template<> EIGEN_STRONG_INLINE Packet2cf preduxp(const Packet2cf* vecs) -{ - return Packet2cf(_mm_add_ps(_mm_movelh_ps(vecs[0].v,vecs[1].v), _mm_movehl_ps(vecs[1].v,vecs[0].v))); -} - -template<> EIGEN_STRONG_INLINE std::complex predux_mul(const Packet2cf& a) -{ - return pfirst(pmul(a, Packet2cf(_mm_movehl_ps(a.v,a.v)))); -} - -template -struct palign_impl -{ - static EIGEN_STRONG_INLINE void run(Packet2cf& first, const Packet2cf& second) - { - if (Offset==1) + } + }; + + template<> struct conj_helper + { + EIGEN_STRONG_INLINE Packet2cf pmadd(const Packet2cf &x, const Packet2cf &y, const Packet2cf &c) const { - first.v = _mm_movehl_ps(first.v, first.v); - first.v = _mm_movelh_ps(first.v, second.v); + return padd(pmul(x, y), c); } - } -}; -template<> struct conj_helper -{ - EIGEN_STRONG_INLINE Packet2cf pmadd(const Packet2cf& x, const Packet2cf& y, const Packet2cf& c) const - { return padd(pmul(x,y),c); } + EIGEN_STRONG_INLINE Packet2cf pmul(const Packet2cf &a, const Packet2cf &b) const + { +#ifdef EIGEN_VECTORIZE_SSE3 + return pconj(internal::pmul(a, b)); +#else + const __m128 mask = _mm_castsi128_ps(_mm_setr_epi32(0x00000000, 0x80000000, 0x00000000, 0x80000000)); + return Packet2cf(_mm_sub_ps(_mm_xor_ps(_mm_mul_ps(vec4f_swizzle1(a.v, 0, 0, 2, 2), b.v), mask), + _mm_mul_ps(vec4f_swizzle1(a.v, 1, 1, 3, 3), vec4f_swizzle1(b.v, 1, 0, 3, 2)))); +#endif + } + }; + + EIGEN_MAKE_CONJ_HELPER_CPLX_REAL(Packet2cf, Packet4f) + + template<> EIGEN_STRONG_INLINE Packet2cf pdiv(const Packet2cf &a, const Packet2cf &b) + { + // TODO optimize it for SSE3 and 4 + Packet2cf res = conj_helper().pmul(a, b); + __m128 s = _mm_mul_ps(b.v, b.v); + return Packet2cf(_mm_div_ps(res.v, _mm_add_ps(s, _mm_castsi128_ps(_mm_shuffle_epi32(_mm_castps_si128(s), 0xb1))))); + } - EIGEN_STRONG_INLINE Packet2cf pmul(const Packet2cf& a, const Packet2cf& b) const + EIGEN_STRONG_INLINE Packet2cf pcplxflip /* */ (const Packet2cf &x) { - #ifdef EIGEN_VECTORIZE_SSE3 - return internal::pmul(a, pconj(b)); - #else - const __m128 mask = _mm_castsi128_ps(_mm_setr_epi32(0x00000000,0x80000000,0x00000000,0x80000000)); - return Packet2cf(_mm_add_ps(_mm_xor_ps(_mm_mul_ps(vec4f_swizzle1(a.v, 0, 0, 2, 2), b.v), mask), - _mm_mul_ps(vec4f_swizzle1(a.v, 1, 1, 3, 3), - vec4f_swizzle1(b.v, 1, 0, 3, 2)))); - #endif + return Packet2cf(vec4f_swizzle1(x.v, 1, 0, 3, 2)); } -}; -template<> struct conj_helper -{ - EIGEN_STRONG_INLINE Packet2cf pmadd(const Packet2cf& x, const Packet2cf& y, const Packet2cf& c) const - { return padd(pmul(x,y),c); } - EIGEN_STRONG_INLINE Packet2cf pmul(const Packet2cf& a, const Packet2cf& b) const + //---------- double ---------- + struct Packet1cd { - #ifdef EIGEN_VECTORIZE_SSE3 - return internal::pmul(pconj(a), b); - #else - const __m128 mask = _mm_castsi128_ps(_mm_setr_epi32(0x00000000,0x80000000,0x00000000,0x80000000)); - return Packet2cf(_mm_add_ps(_mm_mul_ps(vec4f_swizzle1(a.v, 0, 0, 2, 2), b.v), - _mm_xor_ps(_mm_mul_ps(vec4f_swizzle1(a.v, 1, 1, 3, 3), - vec4f_swizzle1(b.v, 1, 0, 3, 2)), mask))); - #endif - } -}; - -template<> struct conj_helper -{ - EIGEN_STRONG_INLINE Packet2cf pmadd(const Packet2cf& x, const Packet2cf& y, const Packet2cf& c) const - { return padd(pmul(x,y),c); } - - EIGEN_STRONG_INLINE Packet2cf pmul(const Packet2cf& a, const Packet2cf& b) const - { - #ifdef EIGEN_VECTORIZE_SSE3 - return pconj(internal::pmul(a, b)); - #else - const __m128 mask = _mm_castsi128_ps(_mm_setr_epi32(0x00000000,0x80000000,0x00000000,0x80000000)); - return Packet2cf(_mm_sub_ps(_mm_xor_ps(_mm_mul_ps(vec4f_swizzle1(a.v, 0, 0, 2, 2), b.v), mask), - _mm_mul_ps(vec4f_swizzle1(a.v, 1, 1, 3, 3), - vec4f_swizzle1(b.v, 1, 0, 3, 2)))); - #endif - } -}; - -EIGEN_MAKE_CONJ_HELPER_CPLX_REAL(Packet2cf,Packet4f) - -template<> EIGEN_STRONG_INLINE Packet2cf pdiv(const Packet2cf& a, const Packet2cf& b) -{ - // TODO optimize it for SSE3 and 4 - Packet2cf res = conj_helper().pmul(a,b); - __m128 s = _mm_mul_ps(b.v,b.v); - return Packet2cf(_mm_div_ps(res.v,_mm_add_ps(s,_mm_castsi128_ps(_mm_shuffle_epi32(_mm_castps_si128(s), 0xb1))))); -} - -EIGEN_STRONG_INLINE Packet2cf pcplxflip/* */(const Packet2cf& x) -{ - return Packet2cf(vec4f_swizzle1(x.v, 1, 0, 3, 2)); -} - - -//---------- double ---------- -struct Packet1cd -{ - EIGEN_STRONG_INLINE Packet1cd() {} - EIGEN_STRONG_INLINE explicit Packet1cd(const __m128d& a) : v(a) {} - __m128d v; -}; + EIGEN_STRONG_INLINE Packet1cd() {} + EIGEN_STRONG_INLINE explicit Packet1cd(const __m128d &a) : v(a) {} + __m128d v; + }; // Use the packet_traits defined in AVX/PacketMath.h instead if we're going // to leverage AVX instructions. #ifndef EIGEN_VECTORIZE_AVX -template<> struct packet_traits > : default_packet_traits -{ - typedef Packet1cd type; - typedef Packet1cd half; - enum { - Vectorizable = 1, - AlignedOnScalar = 0, - size = 1, - HasHalfPacket = 0, - - HasAdd = 1, - HasSub = 1, - HasMul = 1, - HasDiv = 1, - HasNegate = 1, - HasAbs = 0, - HasAbs2 = 0, - HasMin = 0, - HasMax = 0, - HasSetLinear = 0 + template<> struct packet_traits> : default_packet_traits + { + typedef Packet1cd type; + typedef Packet1cd half; + enum { + Vectorizable = 1, + AlignedOnScalar = 0, + size = 1, + HasHalfPacket = 0, + + HasAdd = 1, + HasSub = 1, + HasMul = 1, + HasDiv = 1, + HasNegate = 1, + HasAbs = 0, + HasAbs2 = 0, + HasMin = 0, + HasMax = 0, + HasSetLinear = 0 + }; }; -}; #endif -template<> struct unpacket_traits { typedef std::complex type; enum {size=1, alignment=Aligned16}; typedef Packet1cd half; }; - -template<> EIGEN_STRONG_INLINE Packet1cd padd(const Packet1cd& a, const Packet1cd& b) { return Packet1cd(_mm_add_pd(a.v,b.v)); } -template<> EIGEN_STRONG_INLINE Packet1cd psub(const Packet1cd& a, const Packet1cd& b) { return Packet1cd(_mm_sub_pd(a.v,b.v)); } -template<> EIGEN_STRONG_INLINE Packet1cd pnegate(const Packet1cd& a) { return Packet1cd(pnegate(Packet2d(a.v))); } -template<> EIGEN_STRONG_INLINE Packet1cd pconj(const Packet1cd& a) -{ - const __m128d mask = _mm_castsi128_pd(_mm_set_epi32(0x80000000,0x0,0x0,0x0)); - return Packet1cd(_mm_xor_pd(a.v,mask)); -} - -template<> EIGEN_STRONG_INLINE Packet1cd pmul(const Packet1cd& a, const Packet1cd& b) -{ - #ifdef EIGEN_VECTORIZE_SSE3 - return Packet1cd(_mm_addsub_pd(_mm_mul_pd(_mm_movedup_pd(a.v), b.v), - _mm_mul_pd(vec2d_swizzle1(a.v, 1, 1), - vec2d_swizzle1(b.v, 1, 0)))); - #else - const __m128d mask = _mm_castsi128_pd(_mm_set_epi32(0x0,0x0,0x80000000,0x0)); - return Packet1cd(_mm_add_pd(_mm_mul_pd(vec2d_swizzle1(a.v, 0, 0), b.v), - _mm_xor_pd(_mm_mul_pd(vec2d_swizzle1(a.v, 1, 1), - vec2d_swizzle1(b.v, 1, 0)), mask))); - #endif -} - -template<> EIGEN_STRONG_INLINE Packet1cd pand (const Packet1cd& a, const Packet1cd& b) { return Packet1cd(_mm_and_pd(a.v,b.v)); } -template<> EIGEN_STRONG_INLINE Packet1cd por (const Packet1cd& a, const Packet1cd& b) { return Packet1cd(_mm_or_pd(a.v,b.v)); } -template<> EIGEN_STRONG_INLINE Packet1cd pxor (const Packet1cd& a, const Packet1cd& b) { return Packet1cd(_mm_xor_pd(a.v,b.v)); } -template<> EIGEN_STRONG_INLINE Packet1cd pandnot(const Packet1cd& a, const Packet1cd& b) { return Packet1cd(_mm_andnot_pd(a.v,b.v)); } - -// FIXME force unaligned load, this is a temporary fix -template<> EIGEN_STRONG_INLINE Packet1cd pload (const std::complex* from) -{ EIGEN_DEBUG_ALIGNED_LOAD return Packet1cd(pload((const double*)from)); } -template<> EIGEN_STRONG_INLINE Packet1cd ploadu(const std::complex* from) -{ EIGEN_DEBUG_UNALIGNED_LOAD return Packet1cd(ploadu((const double*)from)); } -template<> EIGEN_STRONG_INLINE Packet1cd pset1(const std::complex& from) -{ /* here we really have to use unaligned loads :( */ return ploadu(&from); } - -template<> EIGEN_STRONG_INLINE Packet1cd ploaddup(const std::complex* from) { return pset1(*from); } - -// FIXME force unaligned store, this is a temporary fix -template<> EIGEN_STRONG_INLINE void pstore >(std::complex * to, const Packet1cd& from) { EIGEN_DEBUG_ALIGNED_STORE pstore((double*)to, Packet2d(from.v)); } -template<> EIGEN_STRONG_INLINE void pstoreu >(std::complex * to, const Packet1cd& from) { EIGEN_DEBUG_UNALIGNED_STORE pstoreu((double*)to, Packet2d(from.v)); } - -template<> EIGEN_STRONG_INLINE void prefetch >(const std::complex * addr) { _mm_prefetch((SsePrefetchPtrType)(addr), _MM_HINT_T0); } - -template<> EIGEN_STRONG_INLINE std::complex pfirst(const Packet1cd& a) -{ - EIGEN_ALIGN16 double res[2]; - _mm_store_pd(res, a.v); - return std::complex(res[0],res[1]); -} - -template<> EIGEN_STRONG_INLINE Packet1cd preverse(const Packet1cd& a) { return a; } - -template<> EIGEN_STRONG_INLINE std::complex predux(const Packet1cd& a) -{ - return pfirst(a); -} - -template<> EIGEN_STRONG_INLINE Packet1cd preduxp(const Packet1cd* vecs) -{ - return vecs[0]; -} - -template<> EIGEN_STRONG_INLINE std::complex predux_mul(const Packet1cd& a) -{ - return pfirst(a); -} - -template -struct palign_impl -{ - static EIGEN_STRONG_INLINE void run(Packet1cd& /*first*/, const Packet1cd& /*second*/) - { - // FIXME is it sure we never have to align a Packet1cd? - // Even though a std::complex has 16 bytes, it is not necessarily aligned on a 16 bytes boundary... - } -}; - -template<> struct conj_helper -{ - EIGEN_STRONG_INLINE Packet1cd pmadd(const Packet1cd& x, const Packet1cd& y, const Packet1cd& c) const - { return padd(pmul(x,y),c); } - - EIGEN_STRONG_INLINE Packet1cd pmul(const Packet1cd& a, const Packet1cd& b) const - { - #ifdef EIGEN_VECTORIZE_SSE3 - return internal::pmul(a, pconj(b)); - #else - const __m128d mask = _mm_castsi128_pd(_mm_set_epi32(0x80000000,0x0,0x0,0x0)); - return Packet1cd(_mm_add_pd(_mm_xor_pd(_mm_mul_pd(vec2d_swizzle1(a.v, 0, 0), b.v), mask), - _mm_mul_pd(vec2d_swizzle1(a.v, 1, 1), - vec2d_swizzle1(b.v, 1, 0)))); - #endif - } -}; - -template<> struct conj_helper -{ - EIGEN_STRONG_INLINE Packet1cd pmadd(const Packet1cd& x, const Packet1cd& y, const Packet1cd& c) const - { return padd(pmul(x,y),c); } - - EIGEN_STRONG_INLINE Packet1cd pmul(const Packet1cd& a, const Packet1cd& b) const - { - #ifdef EIGEN_VECTORIZE_SSE3 - return internal::pmul(pconj(a), b); - #else - const __m128d mask = _mm_castsi128_pd(_mm_set_epi32(0x80000000,0x0,0x0,0x0)); + template<> struct unpacket_traits + { + typedef std::complex type; + enum { size = 1, alignment = Aligned16 }; + typedef Packet1cd half; + }; + + template<> EIGEN_STRONG_INLINE Packet1cd padd(const Packet1cd &a, const Packet1cd &b) + { + return Packet1cd(_mm_add_pd(a.v, b.v)); + } + template<> EIGEN_STRONG_INLINE Packet1cd psub(const Packet1cd &a, const Packet1cd &b) + { + return Packet1cd(_mm_sub_pd(a.v, b.v)); + } + template<> EIGEN_STRONG_INLINE Packet1cd pnegate(const Packet1cd &a) { return Packet1cd(pnegate(Packet2d(a.v))); } + template<> EIGEN_STRONG_INLINE Packet1cd pconj(const Packet1cd &a) + { + const __m128d mask = _mm_castsi128_pd(_mm_set_epi32(0x80000000, 0x0, 0x0, 0x0)); + return Packet1cd(_mm_xor_pd(a.v, mask)); + } + + template<> EIGEN_STRONG_INLINE Packet1cd pmul(const Packet1cd &a, const Packet1cd &b) + { +#ifdef EIGEN_VECTORIZE_SSE3 + return Packet1cd(_mm_addsub_pd( + _mm_mul_pd(_mm_movedup_pd(a.v), b.v), _mm_mul_pd(vec2d_swizzle1(a.v, 1, 1), vec2d_swizzle1(b.v, 1, 0)))); +#else + const __m128d mask = _mm_castsi128_pd(_mm_set_epi32(0x0, 0x0, 0x80000000, 0x0)); return Packet1cd(_mm_add_pd(_mm_mul_pd(vec2d_swizzle1(a.v, 0, 0), b.v), - _mm_xor_pd(_mm_mul_pd(vec2d_swizzle1(a.v, 1, 1), - vec2d_swizzle1(b.v, 1, 0)), mask))); - #endif - } -}; - -template<> struct conj_helper -{ - EIGEN_STRONG_INLINE Packet1cd pmadd(const Packet1cd& x, const Packet1cd& y, const Packet1cd& c) const - { return padd(pmul(x,y),c); } - - EIGEN_STRONG_INLINE Packet1cd pmul(const Packet1cd& a, const Packet1cd& b) const - { - #ifdef EIGEN_VECTORIZE_SSE3 - return pconj(internal::pmul(a, b)); - #else - const __m128d mask = _mm_castsi128_pd(_mm_set_epi32(0x80000000,0x0,0x0,0x0)); - return Packet1cd(_mm_sub_pd(_mm_xor_pd(_mm_mul_pd(vec2d_swizzle1(a.v, 0, 0), b.v), mask), - _mm_mul_pd(vec2d_swizzle1(a.v, 1, 1), - vec2d_swizzle1(b.v, 1, 0)))); - #endif - } -}; - -EIGEN_MAKE_CONJ_HELPER_CPLX_REAL(Packet1cd,Packet2d) - -template<> EIGEN_STRONG_INLINE Packet1cd pdiv(const Packet1cd& a, const Packet1cd& b) -{ - // TODO optimize it for SSE3 and 4 - Packet1cd res = conj_helper().pmul(a,b); - __m128d s = _mm_mul_pd(b.v,b.v); - return Packet1cd(_mm_div_pd(res.v, _mm_add_pd(s,_mm_shuffle_pd(s, s, 0x1)))); -} - -EIGEN_STRONG_INLINE Packet1cd pcplxflip/* */(const Packet1cd& x) -{ - return Packet1cd(preverse(Packet2d(x.v))); -} - -EIGEN_DEVICE_FUNC inline void -ptranspose(PacketBlock& kernel) { - __m128d w1 = _mm_castps_pd(kernel.packet[0].v); - __m128d w2 = _mm_castps_pd(kernel.packet[1].v); - - __m128 tmp = _mm_castpd_ps(_mm_unpackhi_pd(w1, w2)); - kernel.packet[0].v = _mm_castpd_ps(_mm_unpacklo_pd(w1, w2)); - kernel.packet[1].v = tmp; -} - -template<> EIGEN_STRONG_INLINE Packet2cf pblend(const Selector<2>& ifPacket, const Packet2cf& thenPacket, const Packet2cf& elsePacket) { - __m128d result = pblend(ifPacket, _mm_castps_pd(thenPacket.v), _mm_castps_pd(elsePacket.v)); - return Packet2cf(_mm_castpd_ps(result)); -} - -template<> EIGEN_STRONG_INLINE Packet2cf pinsertfirst(const Packet2cf& a, std::complex b) -{ - return Packet2cf(_mm_loadl_pi(a.v, reinterpret_cast(&b))); -} - -template<> EIGEN_STRONG_INLINE Packet1cd pinsertfirst(const Packet1cd&, std::complex b) -{ - return pset1(b); -} - -template<> EIGEN_STRONG_INLINE Packet2cf pinsertlast(const Packet2cf& a, std::complex b) -{ - return Packet2cf(_mm_loadh_pi(a.v, reinterpret_cast(&b))); -} - -template<> EIGEN_STRONG_INLINE Packet1cd pinsertlast(const Packet1cd&, std::complex b) -{ - return pset1(b); -} - -} // end namespace internal - -} // end namespace Eigen - -#endif // EIGEN_COMPLEX_SSE_H + _mm_xor_pd(_mm_mul_pd(vec2d_swizzle1(a.v, 1, 1), vec2d_swizzle1(b.v, 1, 0)), mask))); +#endif + } + + template<> EIGEN_STRONG_INLINE Packet1cd pand(const Packet1cd &a, const Packet1cd &b) + { + return Packet1cd(_mm_and_pd(a.v, b.v)); + } + template<> EIGEN_STRONG_INLINE Packet1cd por(const Packet1cd &a, const Packet1cd &b) + { + return Packet1cd(_mm_or_pd(a.v, b.v)); + } + template<> EIGEN_STRONG_INLINE Packet1cd pxor(const Packet1cd &a, const Packet1cd &b) + { + return Packet1cd(_mm_xor_pd(a.v, b.v)); + } + template<> EIGEN_STRONG_INLINE Packet1cd pandnot(const Packet1cd &a, const Packet1cd &b) + { + return Packet1cd(_mm_andnot_pd(a.v, b.v)); + } + + // FIXME force unaligned load, this is a temporary fix + template<> EIGEN_STRONG_INLINE Packet1cd pload(const std::complex *from) + { + EIGEN_DEBUG_ALIGNED_LOAD return Packet1cd(pload((const double *)from)); + } + template<> EIGEN_STRONG_INLINE Packet1cd ploadu(const std::complex *from) + { + EIGEN_DEBUG_UNALIGNED_LOAD return Packet1cd(ploadu((const double *)from)); + } + template<> EIGEN_STRONG_INLINE Packet1cd pset1(const std::complex &from) + { /* here we really have to use unaligned loads :( */ + return ploadu(&from); + } + + template<> EIGEN_STRONG_INLINE Packet1cd ploaddup(const std::complex *from) + { + return pset1(*from); + } + + // FIXME force unaligned store, this is a temporary fix + template<> EIGEN_STRONG_INLINE void pstore>(std::complex *to, const Packet1cd &from) + { + EIGEN_DEBUG_ALIGNED_STORE pstore((double *)to, Packet2d(from.v)); + } + template<> EIGEN_STRONG_INLINE void pstoreu>(std::complex *to, const Packet1cd &from) + { + EIGEN_DEBUG_UNALIGNED_STORE pstoreu((double *)to, Packet2d(from.v)); + } + + template<> EIGEN_STRONG_INLINE void prefetch>(const std::complex *addr) + { + _mm_prefetch((SsePrefetchPtrType)(addr), _MM_HINT_T0); + } + + template<> EIGEN_STRONG_INLINE std::complex pfirst(const Packet1cd &a) + { + EIGEN_ALIGN16 double res[2]; + _mm_store_pd(res, a.v); + return std::complex(res[0], res[1]); + } + + template<> EIGEN_STRONG_INLINE Packet1cd preverse(const Packet1cd &a) { return a; } + + template<> EIGEN_STRONG_INLINE std::complex predux(const Packet1cd &a) { return pfirst(a); } + + template<> EIGEN_STRONG_INLINE Packet1cd preduxp(const Packet1cd *vecs) { return vecs[0]; } + + template<> EIGEN_STRONG_INLINE std::complex predux_mul(const Packet1cd &a) { return pfirst(a); } + + template struct palign_impl + { + static EIGEN_STRONG_INLINE void run(Packet1cd & /*first*/, const Packet1cd & /*second*/) + { + // FIXME is it sure we never have to align a Packet1cd? + // Even though a std::complex has 16 bytes, it is not necessarily aligned on a 16 bytes boundary... + } + }; + + template<> struct conj_helper + { + EIGEN_STRONG_INLINE Packet1cd pmadd(const Packet1cd &x, const Packet1cd &y, const Packet1cd &c) const + { + return padd(pmul(x, y), c); + } + + EIGEN_STRONG_INLINE Packet1cd pmul(const Packet1cd &a, const Packet1cd &b) const + { +#ifdef EIGEN_VECTORIZE_SSE3 + return internal::pmul(a, pconj(b)); +#else + const __m128d mask = _mm_castsi128_pd(_mm_set_epi32(0x80000000, 0x0, 0x0, 0x0)); + return Packet1cd(_mm_add_pd(_mm_xor_pd(_mm_mul_pd(vec2d_swizzle1(a.v, 0, 0), b.v), mask), + _mm_mul_pd(vec2d_swizzle1(a.v, 1, 1), vec2d_swizzle1(b.v, 1, 0)))); +#endif + } + }; + + template<> struct conj_helper + { + EIGEN_STRONG_INLINE Packet1cd pmadd(const Packet1cd &x, const Packet1cd &y, const Packet1cd &c) const + { + return padd(pmul(x, y), c); + } + + EIGEN_STRONG_INLINE Packet1cd pmul(const Packet1cd &a, const Packet1cd &b) const + { +#ifdef EIGEN_VECTORIZE_SSE3 + return internal::pmul(pconj(a), b); +#else + const __m128d mask = _mm_castsi128_pd(_mm_set_epi32(0x80000000, 0x0, 0x0, 0x0)); + return Packet1cd(_mm_add_pd(_mm_mul_pd(vec2d_swizzle1(a.v, 0, 0), b.v), + _mm_xor_pd(_mm_mul_pd(vec2d_swizzle1(a.v, 1, 1), vec2d_swizzle1(b.v, 1, 0)), mask))); +#endif + } + }; + + template<> struct conj_helper + { + EIGEN_STRONG_INLINE Packet1cd pmadd(const Packet1cd &x, const Packet1cd &y, const Packet1cd &c) const + { + return padd(pmul(x, y), c); + } + + EIGEN_STRONG_INLINE Packet1cd pmul(const Packet1cd &a, const Packet1cd &b) const + { +#ifdef EIGEN_VECTORIZE_SSE3 + return pconj(internal::pmul(a, b)); +#else + const __m128d mask = _mm_castsi128_pd(_mm_set_epi32(0x80000000, 0x0, 0x0, 0x0)); + return Packet1cd(_mm_sub_pd(_mm_xor_pd(_mm_mul_pd(vec2d_swizzle1(a.v, 0, 0), b.v), mask), + _mm_mul_pd(vec2d_swizzle1(a.v, 1, 1), vec2d_swizzle1(b.v, 1, 0)))); +#endif + } + }; + + EIGEN_MAKE_CONJ_HELPER_CPLX_REAL(Packet1cd, Packet2d) + + template<> EIGEN_STRONG_INLINE Packet1cd pdiv(const Packet1cd &a, const Packet1cd &b) + { + // TODO optimize it for SSE3 and 4 + Packet1cd res = conj_helper().pmul(a, b); + __m128d s = _mm_mul_pd(b.v, b.v); + return Packet1cd(_mm_div_pd(res.v, _mm_add_pd(s, _mm_shuffle_pd(s, s, 0x1)))); + } + + EIGEN_STRONG_INLINE Packet1cd pcplxflip /* */ (const Packet1cd &x) + { + return Packet1cd(preverse(Packet2d(x.v))); + } + + EIGEN_DEVICE_FUNC inline void ptranspose(PacketBlock &kernel) + { + __m128d w1 = _mm_castps_pd(kernel.packet[0].v); + __m128d w2 = _mm_castps_pd(kernel.packet[1].v); + + __m128 tmp = _mm_castpd_ps(_mm_unpackhi_pd(w1, w2)); + kernel.packet[0].v = _mm_castpd_ps(_mm_unpacklo_pd(w1, w2)); + kernel.packet[1].v = tmp; + } + + template<> + EIGEN_STRONG_INLINE Packet2cf pblend(const Selector<2> &ifPacket, + const Packet2cf &thenPacket, + const Packet2cf &elsePacket) + { + __m128d result = pblend(ifPacket, _mm_castps_pd(thenPacket.v), _mm_castps_pd(elsePacket.v)); + return Packet2cf(_mm_castpd_ps(result)); + } + + template<> EIGEN_STRONG_INLINE Packet2cf pinsertfirst(const Packet2cf &a, std::complex b) + { + return Packet2cf(_mm_loadl_pi(a.v, reinterpret_cast(&b))); + } + + template<> EIGEN_STRONG_INLINE Packet1cd pinsertfirst(const Packet1cd &, std::complex b) + { + return pset1(b); + } + + template<> EIGEN_STRONG_INLINE Packet2cf pinsertlast(const Packet2cf &a, std::complex b) + { + return Packet2cf(_mm_loadh_pi(a.v, reinterpret_cast(&b))); + } + + template<> EIGEN_STRONG_INLINE Packet1cd pinsertlast(const Packet1cd &, std::complex b) + { + return pset1(b); + } + +}// end namespace internal + +}// end namespace Eigen + +#endif// EIGEN_COMPLEX_SSE_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/arch/SSE/MathFunctions.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/arch/SSE/MathFunctions.h index 7b5f948e..8a0b5903 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/arch/SSE/MathFunctions.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/arch/SSE/MathFunctions.h @@ -19,544 +19,546 @@ namespace Eigen { namespace internal { -template<> EIGEN_DEFINE_FUNCTION_ALLOWING_MULTIPLE_DEFINITIONS EIGEN_UNUSED -Packet4f plog(const Packet4f& _x) -{ - Packet4f x = _x; - _EIGEN_DECLARE_CONST_Packet4f(1 , 1.0f); - _EIGEN_DECLARE_CONST_Packet4f(half, 0.5f); - _EIGEN_DECLARE_CONST_Packet4i(0x7f, 0x7f); - - _EIGEN_DECLARE_CONST_Packet4f_FROM_INT(inv_mant_mask, ~0x7f800000); - - /* the smallest non denormalized float number */ - _EIGEN_DECLARE_CONST_Packet4f_FROM_INT(min_norm_pos, 0x00800000); - _EIGEN_DECLARE_CONST_Packet4f_FROM_INT(minus_inf, 0xff800000);//-1.f/0.f); - - /* natural logarithm computed for 4 simultaneous float - return NaN for x <= 0 - */ - _EIGEN_DECLARE_CONST_Packet4f(cephes_SQRTHF, 0.707106781186547524f); - _EIGEN_DECLARE_CONST_Packet4f(cephes_log_p0, 7.0376836292E-2f); - _EIGEN_DECLARE_CONST_Packet4f(cephes_log_p1, - 1.1514610310E-1f); - _EIGEN_DECLARE_CONST_Packet4f(cephes_log_p2, 1.1676998740E-1f); - _EIGEN_DECLARE_CONST_Packet4f(cephes_log_p3, - 1.2420140846E-1f); - _EIGEN_DECLARE_CONST_Packet4f(cephes_log_p4, + 1.4249322787E-1f); - _EIGEN_DECLARE_CONST_Packet4f(cephes_log_p5, - 1.6668057665E-1f); - _EIGEN_DECLARE_CONST_Packet4f(cephes_log_p6, + 2.0000714765E-1f); - _EIGEN_DECLARE_CONST_Packet4f(cephes_log_p7, - 2.4999993993E-1f); - _EIGEN_DECLARE_CONST_Packet4f(cephes_log_p8, + 3.3333331174E-1f); - _EIGEN_DECLARE_CONST_Packet4f(cephes_log_q1, -2.12194440e-4f); - _EIGEN_DECLARE_CONST_Packet4f(cephes_log_q2, 0.693359375f); - - - Packet4i emm0; - - Packet4f invalid_mask = _mm_cmpnge_ps(x, _mm_setzero_ps()); // not greater equal is true if x is NaN - Packet4f iszero_mask = _mm_cmpeq_ps(x, _mm_setzero_ps()); - - x = pmax(x, p4f_min_norm_pos); /* cut off denormalized stuff */ - emm0 = _mm_srli_epi32(_mm_castps_si128(x), 23); - - /* keep only the fractional part */ - x = _mm_and_ps(x, p4f_inv_mant_mask); - x = _mm_or_ps(x, p4f_half); - - emm0 = _mm_sub_epi32(emm0, p4i_0x7f); - Packet4f e = padd(Packet4f(_mm_cvtepi32_ps(emm0)), p4f_1); - - /* part2: - if( x < SQRTHF ) { - e -= 1; - x = x + x - 1.0; - } else { x = x - 1.0; } - */ - Packet4f mask = _mm_cmplt_ps(x, p4f_cephes_SQRTHF); - Packet4f tmp = pand(x, mask); - x = psub(x, p4f_1); - e = psub(e, pand(p4f_1, mask)); - x = padd(x, tmp); - - Packet4f x2 = pmul(x,x); - Packet4f x3 = pmul(x2,x); - - Packet4f y, y1, y2; - y = pmadd(p4f_cephes_log_p0, x, p4f_cephes_log_p1); - y1 = pmadd(p4f_cephes_log_p3, x, p4f_cephes_log_p4); - y2 = pmadd(p4f_cephes_log_p6, x, p4f_cephes_log_p7); - y = pmadd(y , x, p4f_cephes_log_p2); - y1 = pmadd(y1, x, p4f_cephes_log_p5); - y2 = pmadd(y2, x, p4f_cephes_log_p8); - y = pmadd(y, x3, y1); - y = pmadd(y, x3, y2); - y = pmul(y, x3); - - y1 = pmul(e, p4f_cephes_log_q1); - tmp = pmul(x2, p4f_half); - y = padd(y, y1); - x = psub(x, tmp); - y2 = pmul(e, p4f_cephes_log_q2); - x = padd(x, y); - x = padd(x, y2); - // negative arg will be NAN, 0 will be -INF - return _mm_or_ps(_mm_andnot_ps(iszero_mask, _mm_or_ps(x, invalid_mask)), - _mm_and_ps(iszero_mask, p4f_minus_inf)); -} - -template<> EIGEN_DEFINE_FUNCTION_ALLOWING_MULTIPLE_DEFINITIONS EIGEN_UNUSED -Packet4f pexp(const Packet4f& _x) -{ - Packet4f x = _x; - _EIGEN_DECLARE_CONST_Packet4f(1 , 1.0f); - _EIGEN_DECLARE_CONST_Packet4f(half, 0.5f); - _EIGEN_DECLARE_CONST_Packet4i(0x7f, 0x7f); - - - _EIGEN_DECLARE_CONST_Packet4f(exp_hi, 88.3762626647950f); - _EIGEN_DECLARE_CONST_Packet4f(exp_lo, -88.3762626647949f); - - _EIGEN_DECLARE_CONST_Packet4f(cephes_LOG2EF, 1.44269504088896341f); - _EIGEN_DECLARE_CONST_Packet4f(cephes_exp_C1, 0.693359375f); - _EIGEN_DECLARE_CONST_Packet4f(cephes_exp_C2, -2.12194440e-4f); - - _EIGEN_DECLARE_CONST_Packet4f(cephes_exp_p0, 1.9875691500E-4f); - _EIGEN_DECLARE_CONST_Packet4f(cephes_exp_p1, 1.3981999507E-3f); - _EIGEN_DECLARE_CONST_Packet4f(cephes_exp_p2, 8.3334519073E-3f); - _EIGEN_DECLARE_CONST_Packet4f(cephes_exp_p3, 4.1665795894E-2f); - _EIGEN_DECLARE_CONST_Packet4f(cephes_exp_p4, 1.6666665459E-1f); - _EIGEN_DECLARE_CONST_Packet4f(cephes_exp_p5, 5.0000001201E-1f); - - Packet4f tmp, fx; - Packet4i emm0; - - // clamp x - x = pmax(pmin(x, p4f_exp_hi), p4f_exp_lo); - - /* express exp(x) as exp(g + n*log(2)) */ - fx = pmadd(x, p4f_cephes_LOG2EF, p4f_half); + template<> + EIGEN_DEFINE_FUNCTION_ALLOWING_MULTIPLE_DEFINITIONS EIGEN_UNUSED Packet4f plog(const Packet4f &_x) + { + Packet4f x = _x; + _EIGEN_DECLARE_CONST_Packet4f(1, 1.0f); + _EIGEN_DECLARE_CONST_Packet4f(half, 0.5f); + _EIGEN_DECLARE_CONST_Packet4i(0x7f, 0x7f); + + _EIGEN_DECLARE_CONST_Packet4f_FROM_INT(inv_mant_mask, ~0x7f800000); + + /* the smallest non denormalized float number */ + _EIGEN_DECLARE_CONST_Packet4f_FROM_INT(min_norm_pos, 0x00800000); + _EIGEN_DECLARE_CONST_Packet4f_FROM_INT(minus_inf, 0xff800000);//-1.f/0.f); + + /* natural logarithm computed for 4 simultaneous float + return NaN for x <= 0 + */ + _EIGEN_DECLARE_CONST_Packet4f(cephes_SQRTHF, 0.707106781186547524f); + _EIGEN_DECLARE_CONST_Packet4f(cephes_log_p0, 7.0376836292E-2f); + _EIGEN_DECLARE_CONST_Packet4f(cephes_log_p1, -1.1514610310E-1f); + _EIGEN_DECLARE_CONST_Packet4f(cephes_log_p2, 1.1676998740E-1f); + _EIGEN_DECLARE_CONST_Packet4f(cephes_log_p3, -1.2420140846E-1f); + _EIGEN_DECLARE_CONST_Packet4f(cephes_log_p4, +1.4249322787E-1f); + _EIGEN_DECLARE_CONST_Packet4f(cephes_log_p5, -1.6668057665E-1f); + _EIGEN_DECLARE_CONST_Packet4f(cephes_log_p6, +2.0000714765E-1f); + _EIGEN_DECLARE_CONST_Packet4f(cephes_log_p7, -2.4999993993E-1f); + _EIGEN_DECLARE_CONST_Packet4f(cephes_log_p8, +3.3333331174E-1f); + _EIGEN_DECLARE_CONST_Packet4f(cephes_log_q1, -2.12194440e-4f); + _EIGEN_DECLARE_CONST_Packet4f(cephes_log_q2, 0.693359375f); + + + Packet4i emm0; + + Packet4f invalid_mask = _mm_cmpnge_ps(x, _mm_setzero_ps());// not greater equal is true if x is NaN + Packet4f iszero_mask = _mm_cmpeq_ps(x, _mm_setzero_ps()); + + x = pmax(x, p4f_min_norm_pos); /* cut off denormalized stuff */ + emm0 = _mm_srli_epi32(_mm_castps_si128(x), 23); + + /* keep only the fractional part */ + x = _mm_and_ps(x, p4f_inv_mant_mask); + x = _mm_or_ps(x, p4f_half); + + emm0 = _mm_sub_epi32(emm0, p4i_0x7f); + Packet4f e = padd(Packet4f(_mm_cvtepi32_ps(emm0)), p4f_1); + + /* part2: + if( x < SQRTHF ) { + e -= 1; + x = x + x - 1.0; + } else { x = x - 1.0; } + */ + Packet4f mask = _mm_cmplt_ps(x, p4f_cephes_SQRTHF); + Packet4f tmp = pand(x, mask); + x = psub(x, p4f_1); + e = psub(e, pand(p4f_1, mask)); + x = padd(x, tmp); + + Packet4f x2 = pmul(x, x); + Packet4f x3 = pmul(x2, x); + + Packet4f y, y1, y2; + y = pmadd(p4f_cephes_log_p0, x, p4f_cephes_log_p1); + y1 = pmadd(p4f_cephes_log_p3, x, p4f_cephes_log_p4); + y2 = pmadd(p4f_cephes_log_p6, x, p4f_cephes_log_p7); + y = pmadd(y, x, p4f_cephes_log_p2); + y1 = pmadd(y1, x, p4f_cephes_log_p5); + y2 = pmadd(y2, x, p4f_cephes_log_p8); + y = pmadd(y, x3, y1); + y = pmadd(y, x3, y2); + y = pmul(y, x3); + + y1 = pmul(e, p4f_cephes_log_q1); + tmp = pmul(x2, p4f_half); + y = padd(y, y1); + x = psub(x, tmp); + y2 = pmul(e, p4f_cephes_log_q2); + x = padd(x, y); + x = padd(x, y2); + // negative arg will be NAN, 0 will be -INF + return _mm_or_ps(_mm_andnot_ps(iszero_mask, _mm_or_ps(x, invalid_mask)), _mm_and_ps(iszero_mask, p4f_minus_inf)); + } + + template<> + EIGEN_DEFINE_FUNCTION_ALLOWING_MULTIPLE_DEFINITIONS EIGEN_UNUSED Packet4f pexp(const Packet4f &_x) + { + Packet4f x = _x; + _EIGEN_DECLARE_CONST_Packet4f(1, 1.0f); + _EIGEN_DECLARE_CONST_Packet4f(half, 0.5f); + _EIGEN_DECLARE_CONST_Packet4i(0x7f, 0x7f); + + + _EIGEN_DECLARE_CONST_Packet4f(exp_hi, 88.3762626647950f); + _EIGEN_DECLARE_CONST_Packet4f(exp_lo, -88.3762626647949f); + + _EIGEN_DECLARE_CONST_Packet4f(cephes_LOG2EF, 1.44269504088896341f); + _EIGEN_DECLARE_CONST_Packet4f(cephes_exp_C1, 0.693359375f); + _EIGEN_DECLARE_CONST_Packet4f(cephes_exp_C2, -2.12194440e-4f); + + _EIGEN_DECLARE_CONST_Packet4f(cephes_exp_p0, 1.9875691500E-4f); + _EIGEN_DECLARE_CONST_Packet4f(cephes_exp_p1, 1.3981999507E-3f); + _EIGEN_DECLARE_CONST_Packet4f(cephes_exp_p2, 8.3334519073E-3f); + _EIGEN_DECLARE_CONST_Packet4f(cephes_exp_p3, 4.1665795894E-2f); + _EIGEN_DECLARE_CONST_Packet4f(cephes_exp_p4, 1.6666665459E-1f); + _EIGEN_DECLARE_CONST_Packet4f(cephes_exp_p5, 5.0000001201E-1f); + + Packet4f tmp, fx; + Packet4i emm0; + + // clamp x + x = pmax(pmin(x, p4f_exp_hi), p4f_exp_lo); + + /* express exp(x) as exp(g + n*log(2)) */ + fx = pmadd(x, p4f_cephes_LOG2EF, p4f_half); #ifdef EIGEN_VECTORIZE_SSE4_1 - fx = _mm_floor_ps(fx); + fx = _mm_floor_ps(fx); #else - emm0 = _mm_cvttps_epi32(fx); - tmp = _mm_cvtepi32_ps(emm0); - /* if greater, substract 1 */ - Packet4f mask = _mm_cmpgt_ps(tmp, fx); - mask = _mm_and_ps(mask, p4f_1); - fx = psub(tmp, mask); + emm0 = _mm_cvttps_epi32(fx); + tmp = _mm_cvtepi32_ps(emm0); + /* if greater, substract 1 */ + Packet4f mask = _mm_cmpgt_ps(tmp, fx); + mask = _mm_and_ps(mask, p4f_1); + fx = psub(tmp, mask); #endif - tmp = pmul(fx, p4f_cephes_exp_C1); - Packet4f z = pmul(fx, p4f_cephes_exp_C2); - x = psub(x, tmp); - x = psub(x, z); - - z = pmul(x,x); - - Packet4f y = p4f_cephes_exp_p0; - y = pmadd(y, x, p4f_cephes_exp_p1); - y = pmadd(y, x, p4f_cephes_exp_p2); - y = pmadd(y, x, p4f_cephes_exp_p3); - y = pmadd(y, x, p4f_cephes_exp_p4); - y = pmadd(y, x, p4f_cephes_exp_p5); - y = pmadd(y, z, x); - y = padd(y, p4f_1); - - // build 2^n - emm0 = _mm_cvttps_epi32(fx); - emm0 = _mm_add_epi32(emm0, p4i_0x7f); - emm0 = _mm_slli_epi32(emm0, 23); - return pmax(pmul(y, Packet4f(_mm_castsi128_ps(emm0))), _x); -} -template<> EIGEN_DEFINE_FUNCTION_ALLOWING_MULTIPLE_DEFINITIONS EIGEN_UNUSED -Packet2d pexp(const Packet2d& _x) -{ - Packet2d x = _x; - - _EIGEN_DECLARE_CONST_Packet2d(1 , 1.0); - _EIGEN_DECLARE_CONST_Packet2d(2 , 2.0); - _EIGEN_DECLARE_CONST_Packet2d(half, 0.5); - - _EIGEN_DECLARE_CONST_Packet2d(exp_hi, 709.437); - _EIGEN_DECLARE_CONST_Packet2d(exp_lo, -709.436139303); - - _EIGEN_DECLARE_CONST_Packet2d(cephes_LOG2EF, 1.4426950408889634073599); - - _EIGEN_DECLARE_CONST_Packet2d(cephes_exp_p0, 1.26177193074810590878e-4); - _EIGEN_DECLARE_CONST_Packet2d(cephes_exp_p1, 3.02994407707441961300e-2); - _EIGEN_DECLARE_CONST_Packet2d(cephes_exp_p2, 9.99999999999999999910e-1); - - _EIGEN_DECLARE_CONST_Packet2d(cephes_exp_q0, 3.00198505138664455042e-6); - _EIGEN_DECLARE_CONST_Packet2d(cephes_exp_q1, 2.52448340349684104192e-3); - _EIGEN_DECLARE_CONST_Packet2d(cephes_exp_q2, 2.27265548208155028766e-1); - _EIGEN_DECLARE_CONST_Packet2d(cephes_exp_q3, 2.00000000000000000009e0); - - _EIGEN_DECLARE_CONST_Packet2d(cephes_exp_C1, 0.693145751953125); - _EIGEN_DECLARE_CONST_Packet2d(cephes_exp_C2, 1.42860682030941723212e-6); - static const __m128i p4i_1023_0 = _mm_setr_epi32(1023, 1023, 0, 0); - - Packet2d tmp, fx; - Packet4i emm0; - - // clamp x - x = pmax(pmin(x, p2d_exp_hi), p2d_exp_lo); - /* express exp(x) as exp(g + n*log(2)) */ - fx = pmadd(p2d_cephes_LOG2EF, x, p2d_half); + tmp = pmul(fx, p4f_cephes_exp_C1); + Packet4f z = pmul(fx, p4f_cephes_exp_C2); + x = psub(x, tmp); + x = psub(x, z); + + z = pmul(x, x); + + Packet4f y = p4f_cephes_exp_p0; + y = pmadd(y, x, p4f_cephes_exp_p1); + y = pmadd(y, x, p4f_cephes_exp_p2); + y = pmadd(y, x, p4f_cephes_exp_p3); + y = pmadd(y, x, p4f_cephes_exp_p4); + y = pmadd(y, x, p4f_cephes_exp_p5); + y = pmadd(y, z, x); + y = padd(y, p4f_1); + + // build 2^n + emm0 = _mm_cvttps_epi32(fx); + emm0 = _mm_add_epi32(emm0, p4i_0x7f); + emm0 = _mm_slli_epi32(emm0, 23); + return pmax(pmul(y, Packet4f(_mm_castsi128_ps(emm0))), _x); + } + template<> + EIGEN_DEFINE_FUNCTION_ALLOWING_MULTIPLE_DEFINITIONS EIGEN_UNUSED Packet2d pexp(const Packet2d &_x) + { + Packet2d x = _x; + + _EIGEN_DECLARE_CONST_Packet2d(1, 1.0); + _EIGEN_DECLARE_CONST_Packet2d(2, 2.0); + _EIGEN_DECLARE_CONST_Packet2d(half, 0.5); + + _EIGEN_DECLARE_CONST_Packet2d(exp_hi, 709.437); + _EIGEN_DECLARE_CONST_Packet2d(exp_lo, -709.436139303); + + _EIGEN_DECLARE_CONST_Packet2d(cephes_LOG2EF, 1.4426950408889634073599); + + _EIGEN_DECLARE_CONST_Packet2d(cephes_exp_p0, 1.26177193074810590878e-4); + _EIGEN_DECLARE_CONST_Packet2d(cephes_exp_p1, 3.02994407707441961300e-2); + _EIGEN_DECLARE_CONST_Packet2d(cephes_exp_p2, 9.99999999999999999910e-1); + + _EIGEN_DECLARE_CONST_Packet2d(cephes_exp_q0, 3.00198505138664455042e-6); + _EIGEN_DECLARE_CONST_Packet2d(cephes_exp_q1, 2.52448340349684104192e-3); + _EIGEN_DECLARE_CONST_Packet2d(cephes_exp_q2, 2.27265548208155028766e-1); + _EIGEN_DECLARE_CONST_Packet2d(cephes_exp_q3, 2.00000000000000000009e0); + + _EIGEN_DECLARE_CONST_Packet2d(cephes_exp_C1, 0.693145751953125); + _EIGEN_DECLARE_CONST_Packet2d(cephes_exp_C2, 1.42860682030941723212e-6); + static const __m128i p4i_1023_0 = _mm_setr_epi32(1023, 1023, 0, 0); + + Packet2d tmp, fx; + Packet4i emm0; + + // clamp x + x = pmax(pmin(x, p2d_exp_hi), p2d_exp_lo); + /* express exp(x) as exp(g + n*log(2)) */ + fx = pmadd(p2d_cephes_LOG2EF, x, p2d_half); #ifdef EIGEN_VECTORIZE_SSE4_1 - fx = _mm_floor_pd(fx); + fx = _mm_floor_pd(fx); #else - emm0 = _mm_cvttpd_epi32(fx); - tmp = _mm_cvtepi32_pd(emm0); - /* if greater, substract 1 */ - Packet2d mask = _mm_cmpgt_pd(tmp, fx); - mask = _mm_and_pd(mask, p2d_1); - fx = psub(tmp, mask); + emm0 = _mm_cvttpd_epi32(fx); + tmp = _mm_cvtepi32_pd(emm0); + /* if greater, substract 1 */ + Packet2d mask = _mm_cmpgt_pd(tmp, fx); + mask = _mm_and_pd(mask, p2d_1); + fx = psub(tmp, mask); #endif - tmp = pmul(fx, p2d_cephes_exp_C1); - Packet2d z = pmul(fx, p2d_cephes_exp_C2); - x = psub(x, tmp); - x = psub(x, z); - - Packet2d x2 = pmul(x,x); - - Packet2d px = p2d_cephes_exp_p0; - px = pmadd(px, x2, p2d_cephes_exp_p1); - px = pmadd(px, x2, p2d_cephes_exp_p2); - px = pmul (px, x); - - Packet2d qx = p2d_cephes_exp_q0; - qx = pmadd(qx, x2, p2d_cephes_exp_q1); - qx = pmadd(qx, x2, p2d_cephes_exp_q2); - qx = pmadd(qx, x2, p2d_cephes_exp_q3); - - x = pdiv(px,psub(qx,px)); - x = pmadd(p2d_2,x,p2d_1); - - // build 2^n - emm0 = _mm_cvttpd_epi32(fx); - emm0 = _mm_add_epi32(emm0, p4i_1023_0); - emm0 = _mm_slli_epi32(emm0, 20); - emm0 = _mm_shuffle_epi32(emm0, _MM_SHUFFLE(1,2,0,3)); - return pmax(pmul(x, Packet2d(_mm_castsi128_pd(emm0))), _x); -} - -/* evaluation of 4 sines at onces, using SSE2 intrinsics. - - The code is the exact rewriting of the cephes sinf function. - Precision is excellent as long as x < 8192 (I did not bother to - take into account the special handling they have for greater values - -- it does not return garbage for arguments over 8192, though, but - the extra precision is missing). - - Note that it is such that sinf((float)M_PI) = 8.74e-8, which is the - surprising but correct result. -*/ - -template<> EIGEN_DEFINE_FUNCTION_ALLOWING_MULTIPLE_DEFINITIONS EIGEN_UNUSED -Packet4f psin(const Packet4f& _x) -{ - Packet4f x = _x; - _EIGEN_DECLARE_CONST_Packet4f(1 , 1.0f); - _EIGEN_DECLARE_CONST_Packet4f(half, 0.5f); - - _EIGEN_DECLARE_CONST_Packet4i(1, 1); - _EIGEN_DECLARE_CONST_Packet4i(not1, ~1); - _EIGEN_DECLARE_CONST_Packet4i(2, 2); - _EIGEN_DECLARE_CONST_Packet4i(4, 4); - - _EIGEN_DECLARE_CONST_Packet4f_FROM_INT(sign_mask, 0x80000000); - - _EIGEN_DECLARE_CONST_Packet4f(minus_cephes_DP1,-0.78515625f); - _EIGEN_DECLARE_CONST_Packet4f(minus_cephes_DP2, -2.4187564849853515625e-4f); - _EIGEN_DECLARE_CONST_Packet4f(minus_cephes_DP3, -3.77489497744594108e-8f); - _EIGEN_DECLARE_CONST_Packet4f(sincof_p0, -1.9515295891E-4f); - _EIGEN_DECLARE_CONST_Packet4f(sincof_p1, 8.3321608736E-3f); - _EIGEN_DECLARE_CONST_Packet4f(sincof_p2, -1.6666654611E-1f); - _EIGEN_DECLARE_CONST_Packet4f(coscof_p0, 2.443315711809948E-005f); - _EIGEN_DECLARE_CONST_Packet4f(coscof_p1, -1.388731625493765E-003f); - _EIGEN_DECLARE_CONST_Packet4f(coscof_p2, 4.166664568298827E-002f); - _EIGEN_DECLARE_CONST_Packet4f(cephes_FOPI, 1.27323954473516f); // 4 / M_PI - - Packet4f xmm1, xmm2, xmm3, sign_bit, y; - - Packet4i emm0, emm2; - sign_bit = x; - /* take the absolute value */ - x = pabs(x); - - /* take the modulo */ - - /* extract the sign bit (upper one) */ - sign_bit = _mm_and_ps(sign_bit, p4f_sign_mask); - - /* scale by 4/Pi */ - y = pmul(x, p4f_cephes_FOPI); - - /* store the integer part of y in mm0 */ - emm2 = _mm_cvttps_epi32(y); - /* j=(j+1) & (~1) (see the cephes sources) */ - emm2 = _mm_add_epi32(emm2, p4i_1); - emm2 = _mm_and_si128(emm2, p4i_not1); - y = _mm_cvtepi32_ps(emm2); - /* get the swap sign flag */ - emm0 = _mm_and_si128(emm2, p4i_4); - emm0 = _mm_slli_epi32(emm0, 29); - /* get the polynom selection mask - there is one polynom for 0 <= x <= Pi/4 - and another one for Pi/4 EIGEN_DEFINE_FUNCTION_ALLOWING_MULTIPLE_DEFINITIONS EIGEN_UNUSED -Packet4f pcos(const Packet4f& _x) -{ - Packet4f x = _x; - _EIGEN_DECLARE_CONST_Packet4f(1 , 1.0f); - _EIGEN_DECLARE_CONST_Packet4f(half, 0.5f); - - _EIGEN_DECLARE_CONST_Packet4i(1, 1); - _EIGEN_DECLARE_CONST_Packet4i(not1, ~1); - _EIGEN_DECLARE_CONST_Packet4i(2, 2); - _EIGEN_DECLARE_CONST_Packet4i(4, 4); - - _EIGEN_DECLARE_CONST_Packet4f(minus_cephes_DP1,-0.78515625f); - _EIGEN_DECLARE_CONST_Packet4f(minus_cephes_DP2, -2.4187564849853515625e-4f); - _EIGEN_DECLARE_CONST_Packet4f(minus_cephes_DP3, -3.77489497744594108e-8f); - _EIGEN_DECLARE_CONST_Packet4f(sincof_p0, -1.9515295891E-4f); - _EIGEN_DECLARE_CONST_Packet4f(sincof_p1, 8.3321608736E-3f); - _EIGEN_DECLARE_CONST_Packet4f(sincof_p2, -1.6666654611E-1f); - _EIGEN_DECLARE_CONST_Packet4f(coscof_p0, 2.443315711809948E-005f); - _EIGEN_DECLARE_CONST_Packet4f(coscof_p1, -1.388731625493765E-003f); - _EIGEN_DECLARE_CONST_Packet4f(coscof_p2, 4.166664568298827E-002f); - _EIGEN_DECLARE_CONST_Packet4f(cephes_FOPI, 1.27323954473516f); // 4 / M_PI - - Packet4f xmm1, xmm2, xmm3, y; - Packet4i emm0, emm2; - - x = pabs(x); - - /* scale by 4/Pi */ - y = pmul(x, p4f_cephes_FOPI); - - /* get the integer part of y */ - emm2 = _mm_cvttps_epi32(y); - /* j=(j+1) & (~1) (see the cephes sources) */ - emm2 = _mm_add_epi32(emm2, p4i_1); - emm2 = _mm_and_si128(emm2, p4i_not1); - y = _mm_cvtepi32_ps(emm2); - - emm2 = _mm_sub_epi32(emm2, p4i_2); - - /* get the swap sign flag */ - emm0 = _mm_andnot_si128(emm2, p4i_4); - emm0 = _mm_slli_epi32(emm0, 29); - /* get the polynom selection mask */ - emm2 = _mm_and_si128(emm2, p4i_2); - emm2 = _mm_cmpeq_epi32(emm2, _mm_setzero_si128()); - - Packet4f sign_bit = _mm_castsi128_ps(emm0); - Packet4f poly_mask = _mm_castsi128_ps(emm2); - - /* The magic pass: "Extended precision modular arithmetic" - x = ((x - y * DP1) - y * DP2) - y * DP3; */ - xmm1 = pmul(y, p4f_minus_cephes_DP1); - xmm2 = pmul(y, p4f_minus_cephes_DP2); - xmm3 = pmul(y, p4f_minus_cephes_DP3); - x = padd(x, xmm1); - x = padd(x, xmm2); - x = padd(x, xmm3); - - /* Evaluate the first polynom (0 <= x <= Pi/4) */ - y = p4f_coscof_p0; - Packet4f z = pmul(x,x); - - y = pmadd(y,z,p4f_coscof_p1); - y = pmadd(y,z,p4f_coscof_p2); - y = pmul(y, z); - y = pmul(y, z); - Packet4f tmp = _mm_mul_ps(z, p4f_half); - y = psub(y, tmp); - y = padd(y, p4f_1); - - /* Evaluate the second polynom (Pi/4 <= x <= 0) */ - Packet4f y2 = p4f_sincof_p0; - y2 = pmadd(y2, z, p4f_sincof_p1); - y2 = pmadd(y2, z, p4f_sincof_p2); - y2 = pmul(y2, z); - y2 = pmadd(y2, x, x); - - /* select the correct result from the two polynoms */ - y2 = _mm_and_ps(poly_mask, y2); - y = _mm_andnot_ps(poly_mask, y); - y = _mm_or_ps(y,y2); - - /* update the sign */ - return _mm_xor_ps(y, sign_bit); -} + + template<> + EIGEN_DEFINE_FUNCTION_ALLOWING_MULTIPLE_DEFINITIONS EIGEN_UNUSED Packet4f psin(const Packet4f &_x) + { + Packet4f x = _x; + _EIGEN_DECLARE_CONST_Packet4f(1, 1.0f); + _EIGEN_DECLARE_CONST_Packet4f(half, 0.5f); + + _EIGEN_DECLARE_CONST_Packet4i(1, 1); + _EIGEN_DECLARE_CONST_Packet4i(not1, ~1); + _EIGEN_DECLARE_CONST_Packet4i(2, 2); + _EIGEN_DECLARE_CONST_Packet4i(4, 4); + + _EIGEN_DECLARE_CONST_Packet4f_FROM_INT(sign_mask, 0x80000000); + + _EIGEN_DECLARE_CONST_Packet4f(minus_cephes_DP1, -0.78515625f); + _EIGEN_DECLARE_CONST_Packet4f(minus_cephes_DP2, -2.4187564849853515625e-4f); + _EIGEN_DECLARE_CONST_Packet4f(minus_cephes_DP3, -3.77489497744594108e-8f); + _EIGEN_DECLARE_CONST_Packet4f(sincof_p0, -1.9515295891E-4f); + _EIGEN_DECLARE_CONST_Packet4f(sincof_p1, 8.3321608736E-3f); + _EIGEN_DECLARE_CONST_Packet4f(sincof_p2, -1.6666654611E-1f); + _EIGEN_DECLARE_CONST_Packet4f(coscof_p0, 2.443315711809948E-005f); + _EIGEN_DECLARE_CONST_Packet4f(coscof_p1, -1.388731625493765E-003f); + _EIGEN_DECLARE_CONST_Packet4f(coscof_p2, 4.166664568298827E-002f); + _EIGEN_DECLARE_CONST_Packet4f(cephes_FOPI, 1.27323954473516f);// 4 / M_PI + + Packet4f xmm1, xmm2, xmm3, sign_bit, y; + + Packet4i emm0, emm2; + sign_bit = x; + /* take the absolute value */ + x = pabs(x); + + /* take the modulo */ + + /* extract the sign bit (upper one) */ + sign_bit = _mm_and_ps(sign_bit, p4f_sign_mask); + + /* scale by 4/Pi */ + y = pmul(x, p4f_cephes_FOPI); + + /* store the integer part of y in mm0 */ + emm2 = _mm_cvttps_epi32(y); + /* j=(j+1) & (~1) (see the cephes sources) */ + emm2 = _mm_add_epi32(emm2, p4i_1); + emm2 = _mm_and_si128(emm2, p4i_not1); + y = _mm_cvtepi32_ps(emm2); + /* get the swap sign flag */ + emm0 = _mm_and_si128(emm2, p4i_4); + emm0 = _mm_slli_epi32(emm0, 29); + /* get the polynom selection mask + there is one polynom for 0 <= x <= Pi/4 + and another one for Pi/4 + EIGEN_DEFINE_FUNCTION_ALLOWING_MULTIPLE_DEFINITIONS EIGEN_UNUSED Packet4f pcos(const Packet4f &_x) + { + Packet4f x = _x; + _EIGEN_DECLARE_CONST_Packet4f(1, 1.0f); + _EIGEN_DECLARE_CONST_Packet4f(half, 0.5f); + + _EIGEN_DECLARE_CONST_Packet4i(1, 1); + _EIGEN_DECLARE_CONST_Packet4i(not1, ~1); + _EIGEN_DECLARE_CONST_Packet4i(2, 2); + _EIGEN_DECLARE_CONST_Packet4i(4, 4); + + _EIGEN_DECLARE_CONST_Packet4f(minus_cephes_DP1, -0.78515625f); + _EIGEN_DECLARE_CONST_Packet4f(minus_cephes_DP2, -2.4187564849853515625e-4f); + _EIGEN_DECLARE_CONST_Packet4f(minus_cephes_DP3, -3.77489497744594108e-8f); + _EIGEN_DECLARE_CONST_Packet4f(sincof_p0, -1.9515295891E-4f); + _EIGEN_DECLARE_CONST_Packet4f(sincof_p1, 8.3321608736E-3f); + _EIGEN_DECLARE_CONST_Packet4f(sincof_p2, -1.6666654611E-1f); + _EIGEN_DECLARE_CONST_Packet4f(coscof_p0, 2.443315711809948E-005f); + _EIGEN_DECLARE_CONST_Packet4f(coscof_p1, -1.388731625493765E-003f); + _EIGEN_DECLARE_CONST_Packet4f(coscof_p2, 4.166664568298827E-002f); + _EIGEN_DECLARE_CONST_Packet4f(cephes_FOPI, 1.27323954473516f);// 4 / M_PI + + Packet4f xmm1, xmm2, xmm3, y; + Packet4i emm0, emm2; + + x = pabs(x); + + /* scale by 4/Pi */ + y = pmul(x, p4f_cephes_FOPI); + + /* get the integer part of y */ + emm2 = _mm_cvttps_epi32(y); + /* j=(j+1) & (~1) (see the cephes sources) */ + emm2 = _mm_add_epi32(emm2, p4i_1); + emm2 = _mm_and_si128(emm2, p4i_not1); + y = _mm_cvtepi32_ps(emm2); + + emm2 = _mm_sub_epi32(emm2, p4i_2); + + /* get the swap sign flag */ + emm0 = _mm_andnot_si128(emm2, p4i_4); + emm0 = _mm_slli_epi32(emm0, 29); + /* get the polynom selection mask */ + emm2 = _mm_and_si128(emm2, p4i_2); + emm2 = _mm_cmpeq_epi32(emm2, _mm_setzero_si128()); + + Packet4f sign_bit = _mm_castsi128_ps(emm0); + Packet4f poly_mask = _mm_castsi128_ps(emm2); + + /* The magic pass: "Extended precision modular arithmetic" + x = ((x - y * DP1) - y * DP2) - y * DP3; */ + xmm1 = pmul(y, p4f_minus_cephes_DP1); + xmm2 = pmul(y, p4f_minus_cephes_DP2); + xmm3 = pmul(y, p4f_minus_cephes_DP3); + x = padd(x, xmm1); + x = padd(x, xmm2); + x = padd(x, xmm3); + + /* Evaluate the first polynom (0 <= x <= Pi/4) */ + y = p4f_coscof_p0; + Packet4f z = pmul(x, x); + + y = pmadd(y, z, p4f_coscof_p1); + y = pmadd(y, z, p4f_coscof_p2); + y = pmul(y, z); + y = pmul(y, z); + Packet4f tmp = _mm_mul_ps(z, p4f_half); + y = psub(y, tmp); + y = padd(y, p4f_1); + + /* Evaluate the second polynom (Pi/4 <= x <= 0) */ + Packet4f y2 = p4f_sincof_p0; + y2 = pmadd(y2, z, p4f_sincof_p1); + y2 = pmadd(y2, z, p4f_sincof_p2); + y2 = pmul(y2, z); + y2 = pmadd(y2, x, x); + + /* select the correct result from the two polynoms */ + y2 = _mm_and_ps(poly_mask, y2); + y = _mm_andnot_ps(poly_mask, y); + y = _mm_or_ps(y, y2); + + /* update the sign */ + return _mm_xor_ps(y, sign_bit); + } #if EIGEN_FAST_MATH -// Functions for sqrt. -// The EIGEN_FAST_MATH version uses the _mm_rsqrt_ps approximation and one step -// of Newton's method, at a cost of 1-2 bits of precision as opposed to the -// exact solution. It does not handle +inf, or denormalized numbers correctly. -// The main advantage of this approach is not just speed, but also the fact that -// it can be inlined and pipelined with other computations, further reducing its -// effective latency. This is similar to Quake3's fast inverse square root. -// For detail see here: http://www.beyond3d.com/content/articles/8/ -template<> EIGEN_DEFINE_FUNCTION_ALLOWING_MULTIPLE_DEFINITIONS EIGEN_UNUSED -Packet4f psqrt(const Packet4f& _x) -{ - Packet4f half = pmul(_x, pset1(.5f)); - Packet4f denormal_mask = _mm_and_ps( - _mm_cmpge_ps(_x, _mm_setzero_ps()), - _mm_cmplt_ps(_x, pset1((std::numeric_limits::min)()))); - - // Compute approximate reciprocal sqrt. - Packet4f x = _mm_rsqrt_ps(_x); - // Do a single step of Newton's iteration. - x = pmul(x, psub(pset1(1.5f), pmul(half, pmul(x,x)))); - // Flush results for denormals to zero. - return _mm_andnot_ps(denormal_mask, pmul(_x,x)); -} + // Functions for sqrt. + // The EIGEN_FAST_MATH version uses the _mm_rsqrt_ps approximation and one step + // of Newton's method, at a cost of 1-2 bits of precision as opposed to the + // exact solution. It does not handle +inf, or denormalized numbers correctly. + // The main advantage of this approach is not just speed, but also the fact that + // it can be inlined and pipelined with other computations, further reducing its + // effective latency. This is similar to Quake3's fast inverse square root. + // For detail see here: http://www.beyond3d.com/content/articles/8/ + template<> + EIGEN_DEFINE_FUNCTION_ALLOWING_MULTIPLE_DEFINITIONS EIGEN_UNUSED Packet4f psqrt(const Packet4f &_x) + { + Packet4f half = pmul(_x, pset1(.5f)); + Packet4f denormal_mask = _mm_and_ps( + _mm_cmpge_ps(_x, _mm_setzero_ps()), _mm_cmplt_ps(_x, pset1((std::numeric_limits::min)()))); + + // Compute approximate reciprocal sqrt. + Packet4f x = _mm_rsqrt_ps(_x); + // Do a single step of Newton's iteration. + x = pmul(x, psub(pset1(1.5f), pmul(half, pmul(x, x)))); + // Flush results for denormals to zero. + return _mm_andnot_ps(denormal_mask, pmul(_x, x)); + } #else -template<>EIGEN_DEFINE_FUNCTION_ALLOWING_MULTIPLE_DEFINITIONS EIGEN_UNUSED -Packet4f psqrt(const Packet4f& x) { return _mm_sqrt_ps(x); } + template<> + EIGEN_DEFINE_FUNCTION_ALLOWING_MULTIPLE_DEFINITIONS EIGEN_UNUSED Packet4f psqrt(const Packet4f &x) + { + return _mm_sqrt_ps(x); + } #endif -template<> EIGEN_DEFINE_FUNCTION_ALLOWING_MULTIPLE_DEFINITIONS EIGEN_UNUSED -Packet2d psqrt(const Packet2d& x) { return _mm_sqrt_pd(x); } + template<> + EIGEN_DEFINE_FUNCTION_ALLOWING_MULTIPLE_DEFINITIONS EIGEN_UNUSED Packet2d psqrt(const Packet2d &x) + { + return _mm_sqrt_pd(x); + } #if EIGEN_FAST_MATH -template<> EIGEN_DEFINE_FUNCTION_ALLOWING_MULTIPLE_DEFINITIONS EIGEN_UNUSED -Packet4f prsqrt(const Packet4f& _x) { - _EIGEN_DECLARE_CONST_Packet4f_FROM_INT(inf, 0x7f800000); - _EIGEN_DECLARE_CONST_Packet4f_FROM_INT(nan, 0x7fc00000); - _EIGEN_DECLARE_CONST_Packet4f(one_point_five, 1.5f); - _EIGEN_DECLARE_CONST_Packet4f(minus_half, -0.5f); - _EIGEN_DECLARE_CONST_Packet4f_FROM_INT(flt_min, 0x00800000); + template<> + EIGEN_DEFINE_FUNCTION_ALLOWING_MULTIPLE_DEFINITIONS EIGEN_UNUSED Packet4f prsqrt(const Packet4f &_x) + { + _EIGEN_DECLARE_CONST_Packet4f_FROM_INT(inf, 0x7f800000); + _EIGEN_DECLARE_CONST_Packet4f_FROM_INT(nan, 0x7fc00000); + _EIGEN_DECLARE_CONST_Packet4f(one_point_five, 1.5f); + _EIGEN_DECLARE_CONST_Packet4f(minus_half, -0.5f); + _EIGEN_DECLARE_CONST_Packet4f_FROM_INT(flt_min, 0x00800000); - Packet4f neg_half = pmul(_x, p4f_minus_half); + Packet4f neg_half = pmul(_x, p4f_minus_half); - // select only the inverse sqrt of positive normal inputs (denormals are - // flushed to zero and cause infs as well). - Packet4f le_zero_mask = _mm_cmple_ps(_x, p4f_flt_min); - Packet4f x = _mm_andnot_ps(le_zero_mask, _mm_rsqrt_ps(_x)); + // select only the inverse sqrt of positive normal inputs (denormals are + // flushed to zero and cause infs as well). + Packet4f le_zero_mask = _mm_cmple_ps(_x, p4f_flt_min); + Packet4f x = _mm_andnot_ps(le_zero_mask, _mm_rsqrt_ps(_x)); - // Fill in NaNs and Infs for the negative/zero entries. - Packet4f neg_mask = _mm_cmplt_ps(_x, _mm_setzero_ps()); - Packet4f zero_mask = _mm_andnot_ps(neg_mask, le_zero_mask); - Packet4f infs_and_nans = _mm_or_ps(_mm_and_ps(neg_mask, p4f_nan), - _mm_and_ps(zero_mask, p4f_inf)); + // Fill in NaNs and Infs for the negative/zero entries. + Packet4f neg_mask = _mm_cmplt_ps(_x, _mm_setzero_ps()); + Packet4f zero_mask = _mm_andnot_ps(neg_mask, le_zero_mask); + Packet4f infs_and_nans = _mm_or_ps(_mm_and_ps(neg_mask, p4f_nan), _mm_and_ps(zero_mask, p4f_inf)); - // Do a single step of Newton's iteration. - x = pmul(x, pmadd(neg_half, pmul(x, x), p4f_one_point_five)); + // Do a single step of Newton's iteration. + x = pmul(x, pmadd(neg_half, pmul(x, x), p4f_one_point_five)); - // Insert NaNs and Infs in all the right places. - return _mm_or_ps(x, infs_and_nans); -} + // Insert NaNs and Infs in all the right places. + return _mm_or_ps(x, infs_and_nans); + } #else -template<> EIGEN_DEFINE_FUNCTION_ALLOWING_MULTIPLE_DEFINITIONS EIGEN_UNUSED -Packet4f prsqrt(const Packet4f& x) { - // Unfortunately we can't use the much faster mm_rqsrt_ps since it only provides an approximation. - return _mm_div_ps(pset1(1.0f), _mm_sqrt_ps(x)); -} + template<> + EIGEN_DEFINE_FUNCTION_ALLOWING_MULTIPLE_DEFINITIONS EIGEN_UNUSED Packet4f prsqrt(const Packet4f &x) + { + // Unfortunately we can't use the much faster mm_rqsrt_ps since it only provides an approximation. + return _mm_div_ps(pset1(1.0f), _mm_sqrt_ps(x)); + } #endif -template<> EIGEN_DEFINE_FUNCTION_ALLOWING_MULTIPLE_DEFINITIONS EIGEN_UNUSED -Packet2d prsqrt(const Packet2d& x) { - // Unfortunately we can't use the much faster mm_rqsrt_pd since it only provides an approximation. - return _mm_div_pd(pset1(1.0), _mm_sqrt_pd(x)); -} + template<> + EIGEN_DEFINE_FUNCTION_ALLOWING_MULTIPLE_DEFINITIONS EIGEN_UNUSED Packet2d prsqrt(const Packet2d &x) + { + // Unfortunately we can't use the much faster mm_rqsrt_pd since it only provides an approximation. + return _mm_div_pd(pset1(1.0), _mm_sqrt_pd(x)); + } -// Hyperbolic Tangent function. -template <> -EIGEN_DEFINE_FUNCTION_ALLOWING_MULTIPLE_DEFINITIONS EIGEN_UNUSED Packet4f -ptanh(const Packet4f& x) { - return internal::generic_fast_tanh_float(x); -} + // Hyperbolic Tangent function. + template<> + EIGEN_DEFINE_FUNCTION_ALLOWING_MULTIPLE_DEFINITIONS EIGEN_UNUSED Packet4f ptanh(const Packet4f &x) + { + return internal::generic_fast_tanh_float(x); + } -} // end namespace internal +}// end namespace internal namespace numext { -template<> -EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE -float sqrt(const float &x) -{ - return internal::pfirst(internal::Packet4f(_mm_sqrt_ss(_mm_set_ss(x)))); -} - -template<> -EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE -double sqrt(const double &x) -{ + template<> EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE float sqrt(const float &x) + { + return internal::pfirst(internal::Packet4f(_mm_sqrt_ss(_mm_set_ss(x)))); + } + + template<> EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE double sqrt(const double &x) + { #if EIGEN_COMP_GNUC_STRICT - // This works around a GCC bug generating poor code for _mm_sqrt_pd - // See https://bitbucket.org/eigen/eigen/commits/14f468dba4d350d7c19c9b93072e19f7b3df563b - return internal::pfirst(internal::Packet2d(__builtin_ia32_sqrtsd(_mm_set_sd(x)))); + // This works around a GCC bug generating poor code for _mm_sqrt_pd + // See https://bitbucket.org/eigen/eigen/commits/14f468dba4d350d7c19c9b93072e19f7b3df563b + return internal::pfirst(internal::Packet2d(__builtin_ia32_sqrtsd(_mm_set_sd(x)))); #else - return internal::pfirst(internal::Packet2d(_mm_sqrt_pd(_mm_set_sd(x)))); + return internal::pfirst(internal::Packet2d(_mm_sqrt_pd(_mm_set_sd(x)))); #endif -} + } -} // end namespace numex +}// namespace numext -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_MATH_FUNCTIONS_SSE_H +#endif// EIGEN_MATH_FUNCTIONS_SSE_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/arch/SSE/PacketMath.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/arch/SSE/PacketMath.h old mode 100755 new mode 100644 index 60e2517e..657d379e --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/arch/SSE/PacketMath.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/arch/SSE/PacketMath.h @@ -19,7 +19,7 @@ namespace internal { #endif #ifndef EIGEN_ARCH_DEFAULT_NUMBER_OF_REGISTERS -#define EIGEN_ARCH_DEFAULT_NUMBER_OF_REGISTERS (2*sizeof(void*)) +#define EIGEN_ARCH_DEFAULT_NUMBER_OF_REGISTERS (2 * sizeof(void *)) #endif #ifdef __FMA__ @@ -28,156 +28,189 @@ namespace internal { #endif #endif -#if ((defined EIGEN_VECTORIZE_AVX) && (EIGEN_COMP_GNUC_STRICT || EIGEN_COMP_MINGW) && (__GXX_ABI_VERSION < 1004)) || EIGEN_OS_QNX -// With GCC's default ABI version, a __m128 or __m256 are the same types and therefore we cannot -// have overloads for both types without linking error. -// One solution is to increase ABI version using -fabi-version=4 (or greater). -// Otherwise, we workaround this inconvenience by wrapping 128bit types into the following helper -// structure: -template -struct eigen_packet_wrapper -{ - EIGEN_ALWAYS_INLINE operator T&() { return m_val; } - EIGEN_ALWAYS_INLINE operator const T&() const { return m_val; } - EIGEN_ALWAYS_INLINE eigen_packet_wrapper() {} - EIGEN_ALWAYS_INLINE eigen_packet_wrapper(const T &v) : m_val(v) {} - EIGEN_ALWAYS_INLINE eigen_packet_wrapper& operator=(const T &v) { - m_val = v; - return *this; - } - - T m_val; -}; -typedef eigen_packet_wrapper<__m128> Packet4f; -typedef eigen_packet_wrapper<__m128i> Packet4i; -typedef eigen_packet_wrapper<__m128d> Packet2d; +#if ((defined EIGEN_VECTORIZE_AVX) && (EIGEN_COMP_GNUC_STRICT || EIGEN_COMP_MINGW) && (__GXX_ABI_VERSION < 1004)) \ + || EIGEN_OS_QNX + // With GCC's default ABI version, a __m128 or __m256 are the same types and therefore we cannot + // have overloads for both types without linking error. + // One solution is to increase ABI version using -fabi-version=4 (or greater). + // Otherwise, we workaround this inconvenience by wrapping 128bit types into the following helper + // structure: + template struct eigen_packet_wrapper + { + EIGEN_ALWAYS_INLINE operator T &() { return m_val; } + EIGEN_ALWAYS_INLINE operator const T &() const { return m_val; } + EIGEN_ALWAYS_INLINE eigen_packet_wrapper() {} + EIGEN_ALWAYS_INLINE eigen_packet_wrapper(const T &v) : m_val(v) {} + EIGEN_ALWAYS_INLINE eigen_packet_wrapper &operator=(const T &v) + { + m_val = v; + return *this; + } + + T m_val; + }; + typedef eigen_packet_wrapper<__m128> Packet4f; + typedef eigen_packet_wrapper<__m128i> Packet4i; + typedef eigen_packet_wrapper<__m128d> Packet2d; #else -typedef __m128 Packet4f; -typedef __m128i Packet4i; -typedef __m128d Packet2d; + typedef __m128 Packet4f; + typedef __m128i Packet4i; + typedef __m128d Packet2d; #endif -template<> struct is_arithmetic<__m128> { enum { value = true }; }; -template<> struct is_arithmetic<__m128i> { enum { value = true }; }; -template<> struct is_arithmetic<__m128d> { enum { value = true }; }; + template<> struct is_arithmetic<__m128> + { + enum { value = true }; + }; + template<> struct is_arithmetic<__m128i> + { + enum { value = true }; + }; + template<> struct is_arithmetic<__m128d> + { + enum { value = true }; + }; + +#define vec4f_swizzle1(v, p, q, r, s) \ + (_mm_castsi128_ps(_mm_shuffle_epi32(_mm_castps_si128(v), ((s) << 6 | (r) << 4 | (q) << 2 | (p))))) -#define vec4f_swizzle1(v,p,q,r,s) \ - (_mm_castsi128_ps(_mm_shuffle_epi32( _mm_castps_si128(v), ((s)<<6|(r)<<4|(q)<<2|(p))))) +#define vec4i_swizzle1(v, p, q, r, s) (_mm_shuffle_epi32(v, ((s) << 6 | (r) << 4 | (q) << 2 | (p)))) -#define vec4i_swizzle1(v,p,q,r,s) \ - (_mm_shuffle_epi32( v, ((s)<<6|(r)<<4|(q)<<2|(p)))) +#define vec2d_swizzle1(v, p, q) \ + (_mm_castsi128_pd( \ + _mm_shuffle_epi32(_mm_castpd_si128(v), ((q * 2 + 1) << 6 | (q * 2) << 4 | (p * 2 + 1) << 2 | (p * 2))))) -#define vec2d_swizzle1(v,p,q) \ - (_mm_castsi128_pd(_mm_shuffle_epi32( _mm_castpd_si128(v), ((q*2+1)<<6|(q*2)<<4|(p*2+1)<<2|(p*2))))) - -#define vec4f_swizzle2(a,b,p,q,r,s) \ - (_mm_shuffle_ps( (a), (b), ((s)<<6|(r)<<4|(q)<<2|(p)))) +#define vec4f_swizzle2(a, b, p, q, r, s) (_mm_shuffle_ps((a), (b), ((s) << 6 | (r) << 4 | (q) << 2 | (p)))) -#define vec4i_swizzle2(a,b,p,q,r,s) \ - (_mm_castps_si128( (_mm_shuffle_ps( _mm_castsi128_ps(a), _mm_castsi128_ps(b), ((s)<<6|(r)<<4|(q)<<2|(p)))))) +#define vec4i_swizzle2(a, b, p, q, r, s) \ + (_mm_castps_si128((_mm_shuffle_ps(_mm_castsi128_ps(a), _mm_castsi128_ps(b), ((s) << 6 | (r) << 4 | (q) << 2 | (p)))))) -#define _EIGEN_DECLARE_CONST_Packet4f(NAME,X) \ - const Packet4f p4f_##NAME = pset1(X) +#define _EIGEN_DECLARE_CONST_Packet4f(NAME, X) const Packet4f p4f_##NAME = pset1(X) -#define _EIGEN_DECLARE_CONST_Packet2d(NAME,X) \ - const Packet2d p2d_##NAME = pset1(X) +#define _EIGEN_DECLARE_CONST_Packet2d(NAME, X) const Packet2d p2d_##NAME = pset1(X) -#define _EIGEN_DECLARE_CONST_Packet4f_FROM_INT(NAME,X) \ - const Packet4f p4f_##NAME = _mm_castsi128_ps(pset1(X)) +#define _EIGEN_DECLARE_CONST_Packet4f_FROM_INT(NAME, X) const Packet4f p4f_##NAME = _mm_castsi128_ps(pset1(X)) -#define _EIGEN_DECLARE_CONST_Packet4i(NAME,X) \ - const Packet4i p4i_##NAME = pset1(X) +#define _EIGEN_DECLARE_CONST_Packet4i(NAME, X) const Packet4i p4i_##NAME = pset1(X) // Use the packet_traits defined in AVX/PacketMath.h instead if we're going // to leverage AVX instructions. #ifndef EIGEN_VECTORIZE_AVX -template<> struct packet_traits : default_packet_traits -{ - typedef Packet4f type; - typedef Packet4f half; - enum { - Vectorizable = 1, - AlignedOnScalar = 1, - size=4, - HasHalfPacket = 0, - - HasDiv = 1, - HasSin = EIGEN_FAST_MATH, - HasCos = EIGEN_FAST_MATH, - HasLog = 1, - HasExp = 1, - HasSqrt = 1, - HasRsqrt = 1, - HasTanh = EIGEN_FAST_MATH, - HasBlend = 1 + template<> struct packet_traits : default_packet_traits + { + typedef Packet4f type; + typedef Packet4f half; + enum { + Vectorizable = 1, + AlignedOnScalar = 1, + size = 4, + HasHalfPacket = 0, + + HasDiv = 1, + HasSin = EIGEN_FAST_MATH, + HasCos = EIGEN_FAST_MATH, + HasLog = 1, + HasExp = 1, + HasSqrt = 1, + HasRsqrt = 1, + HasTanh = EIGEN_FAST_MATH, + HasBlend = 1 #ifdef EIGEN_VECTORIZE_SSE4_1 - , - HasRound = 1, - HasFloor = 1, - HasCeil = 1 + , + HasRound = 1, + HasFloor = 1, + HasCeil = 1 #endif + }; }; -}; -template<> struct packet_traits : default_packet_traits -{ - typedef Packet2d type; - typedef Packet2d half; - enum { - Vectorizable = 1, - AlignedOnScalar = 1, - size=2, - HasHalfPacket = 0, - - HasDiv = 1, - HasExp = 1, - HasSqrt = 1, - HasRsqrt = 1, - HasBlend = 1 + template<> struct packet_traits : default_packet_traits + { + typedef Packet2d type; + typedef Packet2d half; + enum { + Vectorizable = 1, + AlignedOnScalar = 1, + size = 2, + HasHalfPacket = 0, + + HasDiv = 1, + HasExp = 1, + HasSqrt = 1, + HasRsqrt = 1, + HasBlend = 1 #ifdef EIGEN_VECTORIZE_SSE4_1 - , - HasRound = 1, - HasFloor = 1, - HasCeil = 1 + , + HasRound = 1, + HasFloor = 1, + HasCeil = 1 #endif + }; }; -}; -#endif -template<> struct packet_traits : default_packet_traits -{ - typedef Packet4i type; - typedef Packet4i half; - enum { - Vectorizable = 1, - AlignedOnScalar = 1, - size=4, - - HasBlend = 1 +#endif + template<> struct packet_traits : default_packet_traits + { + typedef Packet4i type; + typedef Packet4i half; + enum { + Vectorizable = 1, + AlignedOnScalar = 1, + size = 4, + + HasBlend = 1 + }; }; -}; -template<> struct unpacket_traits { typedef float type; enum {size=4, alignment=Aligned16}; typedef Packet4f half; }; -template<> struct unpacket_traits { typedef double type; enum {size=2, alignment=Aligned16}; typedef Packet2d half; }; -template<> struct unpacket_traits { typedef int type; enum {size=4, alignment=Aligned16}; typedef Packet4i half; }; + template<> struct unpacket_traits + { + typedef float type; + enum { size = 4, alignment = Aligned16 }; + typedef Packet4f half; + }; + template<> struct unpacket_traits + { + typedef double type; + enum { size = 2, alignment = Aligned16 }; + typedef Packet2d half; + }; + template<> struct unpacket_traits + { + typedef int type; + enum { size = 4, alignment = Aligned16 }; + typedef Packet4i half; + }; #ifndef EIGEN_VECTORIZE_AVX -template<> struct scalar_div_cost { enum { value = 7 }; }; -template<> struct scalar_div_cost { enum { value = 8 }; }; + template<> struct scalar_div_cost + { + enum { value = 7 }; + }; + template<> struct scalar_div_cost + { + enum { value = 8 }; + }; #endif -#if EIGEN_COMP_MSVC==1500 -// Workaround MSVC 9 internal compiler error. -// TODO: It has been detected with win64 builds (amd64), so let's check whether it also happens in 32bits+SSE mode -// TODO: let's check whether there does not exist a better fix, like adding a pset0() function. (it crashed on pset1(0)). -template<> EIGEN_STRONG_INLINE Packet4f pset1(const float& from) { return _mm_set_ps(from,from,from,from); } -template<> EIGEN_STRONG_INLINE Packet2d pset1(const double& from) { return _mm_set_pd(from,from); } -template<> EIGEN_STRONG_INLINE Packet4i pset1(const int& from) { return _mm_set_epi32(from,from,from,from); } +#if EIGEN_COMP_MSVC == 1500 + // Workaround MSVC 9 internal compiler error. + // TODO: It has been detected with win64 builds (amd64), so let's check whether it also happens in 32bits+SSE mode + // TODO: let's check whether there does not exist a better fix, like adding a pset0() function. (it crashed on + // pset1(0)). + template<> EIGEN_STRONG_INLINE Packet4f pset1(const float &from) + { + return _mm_set_ps(from, from, from, from); + } + template<> EIGEN_STRONG_INLINE Packet2d pset1(const double &from) { return _mm_set_pd(from, from); } + template<> EIGEN_STRONG_INLINE Packet4i pset1(const int &from) + { + return _mm_set_epi32(from, from, from, from); + } #else -template<> EIGEN_STRONG_INLINE Packet4f pset1(const float& from) { return _mm_set_ps1(from); } -template<> EIGEN_STRONG_INLINE Packet2d pset1(const double& from) { return _mm_set1_pd(from); } -template<> EIGEN_STRONG_INLINE Packet4i pset1(const int& from) { return _mm_set1_epi32(from); } + template<> EIGEN_STRONG_INLINE Packet4f pset1(const float &from) { return _mm_set_ps1(from); } + template<> EIGEN_STRONG_INLINE Packet2d pset1(const double &from) { return _mm_set1_pd(from); } + template<> EIGEN_STRONG_INLINE Packet4i pset1(const int &from) { return _mm_set1_epi32(from); } #endif // GCC generates a shufps instruction for _mm_set1_ps/_mm_load1_ps instead of the more efficient pshufd instruction. @@ -186,485 +219,631 @@ template<> EIGEN_STRONG_INLINE Packet4i pset1(const int& from) { re // Therefore, we introduced the pload1 functions to be used in product kernels for which bug 203 does not apply. // Also note that with AVX, we want it to generate a vbroadcastss. #if EIGEN_COMP_GNUC_STRICT && (!defined __AVX__) -template<> EIGEN_STRONG_INLINE Packet4f pload1(const float *from) { - return vec4f_swizzle1(_mm_load_ss(from),0,0,0,0); -} -#endif - -template<> EIGEN_STRONG_INLINE Packet4f plset(const float& a) { return _mm_add_ps(pset1(a), _mm_set_ps(3,2,1,0)); } -template<> EIGEN_STRONG_INLINE Packet2d plset(const double& a) { return _mm_add_pd(pset1(a),_mm_set_pd(1,0)); } -template<> EIGEN_STRONG_INLINE Packet4i plset(const int& a) { return _mm_add_epi32(pset1(a),_mm_set_epi32(3,2,1,0)); } - -template<> EIGEN_STRONG_INLINE Packet4f padd(const Packet4f& a, const Packet4f& b) { return _mm_add_ps(a,b); } -template<> EIGEN_STRONG_INLINE Packet2d padd(const Packet2d& a, const Packet2d& b) { return _mm_add_pd(a,b); } -template<> EIGEN_STRONG_INLINE Packet4i padd(const Packet4i& a, const Packet4i& b) { return _mm_add_epi32(a,b); } - -template<> EIGEN_STRONG_INLINE Packet4f psub(const Packet4f& a, const Packet4f& b) { return _mm_sub_ps(a,b); } -template<> EIGEN_STRONG_INLINE Packet2d psub(const Packet2d& a, const Packet2d& b) { return _mm_sub_pd(a,b); } -template<> EIGEN_STRONG_INLINE Packet4i psub(const Packet4i& a, const Packet4i& b) { return _mm_sub_epi32(a,b); } - -template<> EIGEN_STRONG_INLINE Packet4f pnegate(const Packet4f& a) -{ - const Packet4f mask = _mm_castsi128_ps(_mm_setr_epi32(0x80000000,0x80000000,0x80000000,0x80000000)); - return _mm_xor_ps(a,mask); -} -template<> EIGEN_STRONG_INLINE Packet2d pnegate(const Packet2d& a) -{ - const Packet2d mask = _mm_castsi128_pd(_mm_setr_epi32(0x0,0x80000000,0x0,0x80000000)); - return _mm_xor_pd(a,mask); -} -template<> EIGEN_STRONG_INLINE Packet4i pnegate(const Packet4i& a) -{ - return psub(Packet4i(_mm_setr_epi32(0,0,0,0)), a); -} - -template<> EIGEN_STRONG_INLINE Packet4f pconj(const Packet4f& a) { return a; } -template<> EIGEN_STRONG_INLINE Packet2d pconj(const Packet2d& a) { return a; } -template<> EIGEN_STRONG_INLINE Packet4i pconj(const Packet4i& a) { return a; } - -template<> EIGEN_STRONG_INLINE Packet4f pmul(const Packet4f& a, const Packet4f& b) { return _mm_mul_ps(a,b); } -template<> EIGEN_STRONG_INLINE Packet2d pmul(const Packet2d& a, const Packet2d& b) { return _mm_mul_pd(a,b); } -template<> EIGEN_STRONG_INLINE Packet4i pmul(const Packet4i& a, const Packet4i& b) -{ + template<> EIGEN_STRONG_INLINE Packet4f pload1(const float *from) + { + return vec4f_swizzle1(_mm_load_ss(from), 0, 0, 0, 0); + } +#endif + + template<> EIGEN_STRONG_INLINE Packet4f plset(const float &a) + { + return _mm_add_ps(pset1(a), _mm_set_ps(3, 2, 1, 0)); + } + template<> EIGEN_STRONG_INLINE Packet2d plset(const double &a) + { + return _mm_add_pd(pset1(a), _mm_set_pd(1, 0)); + } + template<> EIGEN_STRONG_INLINE Packet4i plset(const int &a) + { + return _mm_add_epi32(pset1(a), _mm_set_epi32(3, 2, 1, 0)); + } + + template<> EIGEN_STRONG_INLINE Packet4f padd(const Packet4f &a, const Packet4f &b) + { + return _mm_add_ps(a, b); + } + template<> EIGEN_STRONG_INLINE Packet2d padd(const Packet2d &a, const Packet2d &b) + { + return _mm_add_pd(a, b); + } + template<> EIGEN_STRONG_INLINE Packet4i padd(const Packet4i &a, const Packet4i &b) + { + return _mm_add_epi32(a, b); + } + + template<> EIGEN_STRONG_INLINE Packet4f psub(const Packet4f &a, const Packet4f &b) + { + return _mm_sub_ps(a, b); + } + template<> EIGEN_STRONG_INLINE Packet2d psub(const Packet2d &a, const Packet2d &b) + { + return _mm_sub_pd(a, b); + } + template<> EIGEN_STRONG_INLINE Packet4i psub(const Packet4i &a, const Packet4i &b) + { + return _mm_sub_epi32(a, b); + } + + template<> EIGEN_STRONG_INLINE Packet4f pnegate(const Packet4f &a) + { + const Packet4f mask = _mm_castsi128_ps(_mm_setr_epi32(0x80000000, 0x80000000, 0x80000000, 0x80000000)); + return _mm_xor_ps(a, mask); + } + template<> EIGEN_STRONG_INLINE Packet2d pnegate(const Packet2d &a) + { + const Packet2d mask = _mm_castsi128_pd(_mm_setr_epi32(0x0, 0x80000000, 0x0, 0x80000000)); + return _mm_xor_pd(a, mask); + } + template<> EIGEN_STRONG_INLINE Packet4i pnegate(const Packet4i &a) + { + return psub(Packet4i(_mm_setr_epi32(0, 0, 0, 0)), a); + } + + template<> EIGEN_STRONG_INLINE Packet4f pconj(const Packet4f &a) { return a; } + template<> EIGEN_STRONG_INLINE Packet2d pconj(const Packet2d &a) { return a; } + template<> EIGEN_STRONG_INLINE Packet4i pconj(const Packet4i &a) { return a; } + + template<> EIGEN_STRONG_INLINE Packet4f pmul(const Packet4f &a, const Packet4f &b) + { + return _mm_mul_ps(a, b); + } + template<> EIGEN_STRONG_INLINE Packet2d pmul(const Packet2d &a, const Packet2d &b) + { + return _mm_mul_pd(a, b); + } + template<> EIGEN_STRONG_INLINE Packet4i pmul(const Packet4i &a, const Packet4i &b) + { #ifdef EIGEN_VECTORIZE_SSE4_1 - return _mm_mullo_epi32(a,b); + return _mm_mullo_epi32(a, b); #else - // this version is slightly faster than 4 scalar products - return vec4i_swizzle1( - vec4i_swizzle2( - _mm_mul_epu32(a,b), - _mm_mul_epu32(vec4i_swizzle1(a,1,0,3,2), - vec4i_swizzle1(b,1,0,3,2)), - 0,2,0,2), - 0,2,1,3); -#endif -} - -template<> EIGEN_STRONG_INLINE Packet4f pdiv(const Packet4f& a, const Packet4f& b) { return _mm_div_ps(a,b); } -template<> EIGEN_STRONG_INLINE Packet2d pdiv(const Packet2d& a, const Packet2d& b) { return _mm_div_pd(a,b); } - -// for some weird raisons, it has to be overloaded for packet of integers -template<> EIGEN_STRONG_INLINE Packet4i pmadd(const Packet4i& a, const Packet4i& b, const Packet4i& c) { return padd(pmul(a,b), c); } + // this version is slightly faster than 4 scalar products + return vec4i_swizzle1( + vec4i_swizzle2( + _mm_mul_epu32(a, b), _mm_mul_epu32(vec4i_swizzle1(a, 1, 0, 3, 2), vec4i_swizzle1(b, 1, 0, 3, 2)), 0, 2, 0, 2), + 0, + 2, + 1, + 3); +#endif + } + + template<> EIGEN_STRONG_INLINE Packet4f pdiv(const Packet4f &a, const Packet4f &b) + { + return _mm_div_ps(a, b); + } + template<> EIGEN_STRONG_INLINE Packet2d pdiv(const Packet2d &a, const Packet2d &b) + { + return _mm_div_pd(a, b); + } + + // for some weird raisons, it has to be overloaded for packet of integers + template<> EIGEN_STRONG_INLINE Packet4i pmadd(const Packet4i &a, const Packet4i &b, const Packet4i &c) + { + return padd(pmul(a, b), c); + } #ifdef __FMA__ -template<> EIGEN_STRONG_INLINE Packet4f pmadd(const Packet4f& a, const Packet4f& b, const Packet4f& c) { return _mm_fmadd_ps(a,b,c); } -template<> EIGEN_STRONG_INLINE Packet2d pmadd(const Packet2d& a, const Packet2d& b, const Packet2d& c) { return _mm_fmadd_pd(a,b,c); } + template<> EIGEN_STRONG_INLINE Packet4f pmadd(const Packet4f &a, const Packet4f &b, const Packet4f &c) + { + return _mm_fmadd_ps(a, b, c); + } + template<> EIGEN_STRONG_INLINE Packet2d pmadd(const Packet2d &a, const Packet2d &b, const Packet2d &c) + { + return _mm_fmadd_pd(a, b, c); + } #endif -template<> EIGEN_STRONG_INLINE Packet4f pmin(const Packet4f& a, const Packet4f& b) { return _mm_min_ps(a,b); } -template<> EIGEN_STRONG_INLINE Packet2d pmin(const Packet2d& a, const Packet2d& b) { return _mm_min_pd(a,b); } -template<> EIGEN_STRONG_INLINE Packet4i pmin(const Packet4i& a, const Packet4i& b) -{ + template<> EIGEN_STRONG_INLINE Packet4f pmin(const Packet4f &a, const Packet4f &b) + { + return _mm_min_ps(a, b); + } + template<> EIGEN_STRONG_INLINE Packet2d pmin(const Packet2d &a, const Packet2d &b) + { + return _mm_min_pd(a, b); + } + template<> EIGEN_STRONG_INLINE Packet4i pmin(const Packet4i &a, const Packet4i &b) + { #ifdef EIGEN_VECTORIZE_SSE4_1 - return _mm_min_epi32(a,b); + return _mm_min_epi32(a, b); #else - // after some bench, this version *is* faster than a scalar implementation - Packet4i mask = _mm_cmplt_epi32(a,b); - return _mm_or_si128(_mm_and_si128(mask,a),_mm_andnot_si128(mask,b)); + // after some bench, this version *is* faster than a scalar implementation + Packet4i mask = _mm_cmplt_epi32(a, b); + return _mm_or_si128(_mm_and_si128(mask, a), _mm_andnot_si128(mask, b)); #endif -} + } -template<> EIGEN_STRONG_INLINE Packet4f pmax(const Packet4f& a, const Packet4f& b) { return _mm_max_ps(a,b); } -template<> EIGEN_STRONG_INLINE Packet2d pmax(const Packet2d& a, const Packet2d& b) { return _mm_max_pd(a,b); } -template<> EIGEN_STRONG_INLINE Packet4i pmax(const Packet4i& a, const Packet4i& b) -{ + template<> EIGEN_STRONG_INLINE Packet4f pmax(const Packet4f &a, const Packet4f &b) + { + return _mm_max_ps(a, b); + } + template<> EIGEN_STRONG_INLINE Packet2d pmax(const Packet2d &a, const Packet2d &b) + { + return _mm_max_pd(a, b); + } + template<> EIGEN_STRONG_INLINE Packet4i pmax(const Packet4i &a, const Packet4i &b) + { #ifdef EIGEN_VECTORIZE_SSE4_1 - return _mm_max_epi32(a,b); + return _mm_max_epi32(a, b); #else - // after some bench, this version *is* faster than a scalar implementation - Packet4i mask = _mm_cmpgt_epi32(a,b); - return _mm_or_si128(_mm_and_si128(mask,a),_mm_andnot_si128(mask,b)); + // after some bench, this version *is* faster than a scalar implementation + Packet4i mask = _mm_cmpgt_epi32(a, b); + return _mm_or_si128(_mm_and_si128(mask, a), _mm_andnot_si128(mask, b)); #endif -} + } #ifdef EIGEN_VECTORIZE_SSE4_1 -template<> EIGEN_STRONG_INLINE Packet4f pround(const Packet4f& a) { return _mm_round_ps(a, 0); } -template<> EIGEN_STRONG_INLINE Packet2d pround(const Packet2d& a) { return _mm_round_pd(a, 0); } + template<> EIGEN_STRONG_INLINE Packet4f pround(const Packet4f &a) { return _mm_round_ps(a, 0); } + template<> EIGEN_STRONG_INLINE Packet2d pround(const Packet2d &a) { return _mm_round_pd(a, 0); } -template<> EIGEN_STRONG_INLINE Packet4f pceil(const Packet4f& a) { return _mm_ceil_ps(a); } -template<> EIGEN_STRONG_INLINE Packet2d pceil(const Packet2d& a) { return _mm_ceil_pd(a); } + template<> EIGEN_STRONG_INLINE Packet4f pceil(const Packet4f &a) { return _mm_ceil_ps(a); } + template<> EIGEN_STRONG_INLINE Packet2d pceil(const Packet2d &a) { return _mm_ceil_pd(a); } -template<> EIGEN_STRONG_INLINE Packet4f pfloor(const Packet4f& a) { return _mm_floor_ps(a); } -template<> EIGEN_STRONG_INLINE Packet2d pfloor(const Packet2d& a) { return _mm_floor_pd(a); } + template<> EIGEN_STRONG_INLINE Packet4f pfloor(const Packet4f &a) { return _mm_floor_ps(a); } + template<> EIGEN_STRONG_INLINE Packet2d pfloor(const Packet2d &a) { return _mm_floor_pd(a); } #endif -template<> EIGEN_STRONG_INLINE Packet4f pand(const Packet4f& a, const Packet4f& b) { return _mm_and_ps(a,b); } -template<> EIGEN_STRONG_INLINE Packet2d pand(const Packet2d& a, const Packet2d& b) { return _mm_and_pd(a,b); } -template<> EIGEN_STRONG_INLINE Packet4i pand(const Packet4i& a, const Packet4i& b) { return _mm_and_si128(a,b); } + template<> EIGEN_STRONG_INLINE Packet4f pand(const Packet4f &a, const Packet4f &b) + { + return _mm_and_ps(a, b); + } + template<> EIGEN_STRONG_INLINE Packet2d pand(const Packet2d &a, const Packet2d &b) + { + return _mm_and_pd(a, b); + } + template<> EIGEN_STRONG_INLINE Packet4i pand(const Packet4i &a, const Packet4i &b) + { + return _mm_and_si128(a, b); + } -template<> EIGEN_STRONG_INLINE Packet4f por(const Packet4f& a, const Packet4f& b) { return _mm_or_ps(a,b); } -template<> EIGEN_STRONG_INLINE Packet2d por(const Packet2d& a, const Packet2d& b) { return _mm_or_pd(a,b); } -template<> EIGEN_STRONG_INLINE Packet4i por(const Packet4i& a, const Packet4i& b) { return _mm_or_si128(a,b); } + template<> EIGEN_STRONG_INLINE Packet4f por(const Packet4f &a, const Packet4f &b) + { + return _mm_or_ps(a, b); + } + template<> EIGEN_STRONG_INLINE Packet2d por(const Packet2d &a, const Packet2d &b) + { + return _mm_or_pd(a, b); + } + template<> EIGEN_STRONG_INLINE Packet4i por(const Packet4i &a, const Packet4i &b) + { + return _mm_or_si128(a, b); + } -template<> EIGEN_STRONG_INLINE Packet4f pxor(const Packet4f& a, const Packet4f& b) { return _mm_xor_ps(a,b); } -template<> EIGEN_STRONG_INLINE Packet2d pxor(const Packet2d& a, const Packet2d& b) { return _mm_xor_pd(a,b); } -template<> EIGEN_STRONG_INLINE Packet4i pxor(const Packet4i& a, const Packet4i& b) { return _mm_xor_si128(a,b); } + template<> EIGEN_STRONG_INLINE Packet4f pxor(const Packet4f &a, const Packet4f &b) + { + return _mm_xor_ps(a, b); + } + template<> EIGEN_STRONG_INLINE Packet2d pxor(const Packet2d &a, const Packet2d &b) + { + return _mm_xor_pd(a, b); + } + template<> EIGEN_STRONG_INLINE Packet4i pxor(const Packet4i &a, const Packet4i &b) + { + return _mm_xor_si128(a, b); + } -template<> EIGEN_STRONG_INLINE Packet4f pandnot(const Packet4f& a, const Packet4f& b) { return _mm_andnot_ps(a,b); } -template<> EIGEN_STRONG_INLINE Packet2d pandnot(const Packet2d& a, const Packet2d& b) { return _mm_andnot_pd(a,b); } -template<> EIGEN_STRONG_INLINE Packet4i pandnot(const Packet4i& a, const Packet4i& b) { return _mm_andnot_si128(a,b); } + template<> EIGEN_STRONG_INLINE Packet4f pandnot(const Packet4f &a, const Packet4f &b) + { + return _mm_andnot_ps(a, b); + } + template<> EIGEN_STRONG_INLINE Packet2d pandnot(const Packet2d &a, const Packet2d &b) + { + return _mm_andnot_pd(a, b); + } + template<> EIGEN_STRONG_INLINE Packet4i pandnot(const Packet4i &a, const Packet4i &b) + { + return _mm_andnot_si128(a, b); + } -template<> EIGEN_STRONG_INLINE Packet4f pload(const float* from) { EIGEN_DEBUG_ALIGNED_LOAD return _mm_load_ps(from); } -template<> EIGEN_STRONG_INLINE Packet2d pload(const double* from) { EIGEN_DEBUG_ALIGNED_LOAD return _mm_load_pd(from); } -template<> EIGEN_STRONG_INLINE Packet4i pload(const int* from) { EIGEN_DEBUG_ALIGNED_LOAD return _mm_load_si128(reinterpret_cast(from)); } + template<> EIGEN_STRONG_INLINE Packet4f pload(const float *from) + { + EIGEN_DEBUG_ALIGNED_LOAD return _mm_load_ps(from); + } + template<> EIGEN_STRONG_INLINE Packet2d pload(const double *from) + { + EIGEN_DEBUG_ALIGNED_LOAD return _mm_load_pd(from); + } + template<> EIGEN_STRONG_INLINE Packet4i pload(const int *from) + { + EIGEN_DEBUG_ALIGNED_LOAD return _mm_load_si128(reinterpret_cast(from)); + } #if EIGEN_COMP_MSVC - template<> EIGEN_STRONG_INLINE Packet4f ploadu(const float* from) { + template<> EIGEN_STRONG_INLINE Packet4f ploadu(const float *from) + { EIGEN_DEBUG_UNALIGNED_LOAD - #if (EIGEN_COMP_MSVC==1600) +#if (EIGEN_COMP_MSVC == 1600) // NOTE Some version of MSVC10 generates bad code when using _mm_loadu_ps // (i.e., it does not generate an unaligned load!! - __m128 res = _mm_loadl_pi(_mm_set1_ps(0.0f), (const __m64*)(from)); - res = _mm_loadh_pi(res, (const __m64*)(from+2)); + __m128 res = _mm_loadl_pi(_mm_set1_ps(0.0f), (const __m64 *)(from)); + res = _mm_loadh_pi(res, (const __m64 *)(from + 2)); return res; - #else +#else return _mm_loadu_ps(from); - #endif +#endif } #else -// NOTE: with the code below, MSVC's compiler crashes! - -template<> EIGEN_STRONG_INLINE Packet4f ploadu(const float* from) -{ - EIGEN_DEBUG_UNALIGNED_LOAD - return _mm_loadu_ps(from); -} -#endif - -template<> EIGEN_STRONG_INLINE Packet2d ploadu(const double* from) -{ - EIGEN_DEBUG_UNALIGNED_LOAD - return _mm_loadu_pd(from); -} -template<> EIGEN_STRONG_INLINE Packet4i ploadu(const int* from) -{ - EIGEN_DEBUG_UNALIGNED_LOAD - return _mm_loadu_si128(reinterpret_cast(from)); -} - - -template<> EIGEN_STRONG_INLINE Packet4f ploaddup(const float* from) -{ - return vec4f_swizzle1(_mm_castpd_ps(_mm_load_sd(reinterpret_cast(from))), 0, 0, 1, 1); -} -template<> EIGEN_STRONG_INLINE Packet2d ploaddup(const double* from) -{ return pset1(from[0]); } -template<> EIGEN_STRONG_INLINE Packet4i ploaddup(const int* from) -{ - Packet4i tmp; - tmp = _mm_loadl_epi64(reinterpret_cast(from)); - return vec4i_swizzle1(tmp, 0, 0, 1, 1); -} - -template<> EIGEN_STRONG_INLINE void pstore(float* to, const Packet4f& from) { EIGEN_DEBUG_ALIGNED_STORE _mm_store_ps(to, from); } -template<> EIGEN_STRONG_INLINE void pstore(double* to, const Packet2d& from) { EIGEN_DEBUG_ALIGNED_STORE _mm_store_pd(to, from); } -template<> EIGEN_STRONG_INLINE void pstore(int* to, const Packet4i& from) { EIGEN_DEBUG_ALIGNED_STORE _mm_store_si128(reinterpret_cast<__m128i*>(to), from); } - -template<> EIGEN_STRONG_INLINE void pstoreu(double* to, const Packet2d& from) { EIGEN_DEBUG_UNALIGNED_STORE _mm_storeu_pd(to, from); } -template<> EIGEN_STRONG_INLINE void pstoreu(float* to, const Packet4f& from) { EIGEN_DEBUG_UNALIGNED_STORE _mm_storeu_ps(to, from); } -template<> EIGEN_STRONG_INLINE void pstoreu(int* to, const Packet4i& from) { EIGEN_DEBUG_UNALIGNED_STORE _mm_storeu_si128(reinterpret_cast<__m128i*>(to), from); } - -template<> EIGEN_DEVICE_FUNC inline Packet4f pgather(const float* from, Index stride) -{ - return _mm_set_ps(from[3*stride], from[2*stride], from[1*stride], from[0*stride]); -} -template<> EIGEN_DEVICE_FUNC inline Packet2d pgather(const double* from, Index stride) -{ - return _mm_set_pd(from[1*stride], from[0*stride]); -} -template<> EIGEN_DEVICE_FUNC inline Packet4i pgather(const int* from, Index stride) -{ - return _mm_set_epi32(from[3*stride], from[2*stride], from[1*stride], from[0*stride]); - } - -template<> EIGEN_DEVICE_FUNC inline void pscatter(float* to, const Packet4f& from, Index stride) -{ - to[stride*0] = _mm_cvtss_f32(from); - to[stride*1] = _mm_cvtss_f32(_mm_shuffle_ps(from, from, 1)); - to[stride*2] = _mm_cvtss_f32(_mm_shuffle_ps(from, from, 2)); - to[stride*3] = _mm_cvtss_f32(_mm_shuffle_ps(from, from, 3)); -} -template<> EIGEN_DEVICE_FUNC inline void pscatter(double* to, const Packet2d& from, Index stride) -{ - to[stride*0] = _mm_cvtsd_f64(from); - to[stride*1] = _mm_cvtsd_f64(_mm_shuffle_pd(from, from, 1)); -} -template<> EIGEN_DEVICE_FUNC inline void pscatter(int* to, const Packet4i& from, Index stride) -{ - to[stride*0] = _mm_cvtsi128_si32(from); - to[stride*1] = _mm_cvtsi128_si32(_mm_shuffle_epi32(from, 1)); - to[stride*2] = _mm_cvtsi128_si32(_mm_shuffle_epi32(from, 2)); - to[stride*3] = _mm_cvtsi128_si32(_mm_shuffle_epi32(from, 3)); -} - -// some compilers might be tempted to perform multiple moves instead of using a vector path. -template<> EIGEN_STRONG_INLINE void pstore1(float* to, const float& a) -{ - Packet4f pa = _mm_set_ss(a); - pstore(to, Packet4f(vec4f_swizzle1(pa,0,0,0,0))); -} -// some compilers might be tempted to perform multiple moves instead of using a vector path. -template<> EIGEN_STRONG_INLINE void pstore1(double* to, const double& a) -{ - Packet2d pa = _mm_set_sd(a); - pstore(to, Packet2d(vec2d_swizzle1(pa,0,0))); -} + // NOTE: with the code below, MSVC's compiler crashes! + + template<> EIGEN_STRONG_INLINE Packet4f ploadu(const float *from) + { + EIGEN_DEBUG_UNALIGNED_LOAD + return _mm_loadu_ps(from); + } +#endif + + template<> EIGEN_STRONG_INLINE Packet2d ploadu(const double *from) + { + EIGEN_DEBUG_UNALIGNED_LOAD + return _mm_loadu_pd(from); + } + template<> EIGEN_STRONG_INLINE Packet4i ploadu(const int *from) + { + EIGEN_DEBUG_UNALIGNED_LOAD + return _mm_loadu_si128(reinterpret_cast(from)); + } + + + template<> EIGEN_STRONG_INLINE Packet4f ploaddup(const float *from) + { + return vec4f_swizzle1(_mm_castpd_ps(_mm_load_sd(reinterpret_cast(from))), 0, 0, 1, 1); + } + template<> EIGEN_STRONG_INLINE Packet2d ploaddup(const double *from) { return pset1(from[0]); } + template<> EIGEN_STRONG_INLINE Packet4i ploaddup(const int *from) + { + Packet4i tmp; + tmp = _mm_loadl_epi64(reinterpret_cast(from)); + return vec4i_swizzle1(tmp, 0, 0, 1, 1); + } + + template<> EIGEN_STRONG_INLINE void pstore(float *to, const Packet4f &from) + { + EIGEN_DEBUG_ALIGNED_STORE _mm_store_ps(to, from); + } + template<> EIGEN_STRONG_INLINE void pstore(double *to, const Packet2d &from) + { + EIGEN_DEBUG_ALIGNED_STORE _mm_store_pd(to, from); + } + template<> EIGEN_STRONG_INLINE void pstore(int *to, const Packet4i &from) + { + EIGEN_DEBUG_ALIGNED_STORE _mm_store_si128(reinterpret_cast<__m128i *>(to), from); + } + + template<> EIGEN_STRONG_INLINE void pstoreu(double *to, const Packet2d &from) + { + EIGEN_DEBUG_UNALIGNED_STORE _mm_storeu_pd(to, from); + } + template<> EIGEN_STRONG_INLINE void pstoreu(float *to, const Packet4f &from) + { + EIGEN_DEBUG_UNALIGNED_STORE _mm_storeu_ps(to, from); + } + template<> EIGEN_STRONG_INLINE void pstoreu(int *to, const Packet4i &from) + { + EIGEN_DEBUG_UNALIGNED_STORE _mm_storeu_si128(reinterpret_cast<__m128i *>(to), from); + } + + template<> EIGEN_DEVICE_FUNC inline Packet4f pgather(const float *from, Index stride) + { + return _mm_set_ps(from[3 * stride], from[2 * stride], from[1 * stride], from[0 * stride]); + } + template<> EIGEN_DEVICE_FUNC inline Packet2d pgather(const double *from, Index stride) + { + return _mm_set_pd(from[1 * stride], from[0 * stride]); + } + template<> EIGEN_DEVICE_FUNC inline Packet4i pgather(const int *from, Index stride) + { + return _mm_set_epi32(from[3 * stride], from[2 * stride], from[1 * stride], from[0 * stride]); + } + + template<> EIGEN_DEVICE_FUNC inline void pscatter(float *to, const Packet4f &from, Index stride) + { + to[stride * 0] = _mm_cvtss_f32(from); + to[stride * 1] = _mm_cvtss_f32(_mm_shuffle_ps(from, from, 1)); + to[stride * 2] = _mm_cvtss_f32(_mm_shuffle_ps(from, from, 2)); + to[stride * 3] = _mm_cvtss_f32(_mm_shuffle_ps(from, from, 3)); + } + template<> EIGEN_DEVICE_FUNC inline void pscatter(double *to, const Packet2d &from, Index stride) + { + to[stride * 0] = _mm_cvtsd_f64(from); + to[stride * 1] = _mm_cvtsd_f64(_mm_shuffle_pd(from, from, 1)); + } + template<> EIGEN_DEVICE_FUNC inline void pscatter(int *to, const Packet4i &from, Index stride) + { + to[stride * 0] = _mm_cvtsi128_si32(from); + to[stride * 1] = _mm_cvtsi128_si32(_mm_shuffle_epi32(from, 1)); + to[stride * 2] = _mm_cvtsi128_si32(_mm_shuffle_epi32(from, 2)); + to[stride * 3] = _mm_cvtsi128_si32(_mm_shuffle_epi32(from, 3)); + } + + // some compilers might be tempted to perform multiple moves instead of using a vector path. + template<> EIGEN_STRONG_INLINE void pstore1(float *to, const float &a) + { + Packet4f pa = _mm_set_ss(a); + pstore(to, Packet4f(vec4f_swizzle1(pa, 0, 0, 0, 0))); + } + // some compilers might be tempted to perform multiple moves instead of using a vector path. + template<> EIGEN_STRONG_INLINE void pstore1(double *to, const double &a) + { + Packet2d pa = _mm_set_sd(a); + pstore(to, Packet2d(vec2d_swizzle1(pa, 0, 0))); + } #if EIGEN_COMP_PGI -typedef const void * SsePrefetchPtrType; + typedef const void *SsePrefetchPtrType; #else -typedef const char * SsePrefetchPtrType; + typedef const char *SsePrefetchPtrType; #endif #ifndef EIGEN_VECTORIZE_AVX -template<> EIGEN_STRONG_INLINE void prefetch(const float* addr) { _mm_prefetch((SsePrefetchPtrType)(addr), _MM_HINT_T0); } -template<> EIGEN_STRONG_INLINE void prefetch(const double* addr) { _mm_prefetch((SsePrefetchPtrType)(addr), _MM_HINT_T0); } -template<> EIGEN_STRONG_INLINE void prefetch(const int* addr) { _mm_prefetch((SsePrefetchPtrType)(addr), _MM_HINT_T0); } + template<> EIGEN_STRONG_INLINE void prefetch(const float *addr) + { + _mm_prefetch((SsePrefetchPtrType)(addr), _MM_HINT_T0); + } + template<> EIGEN_STRONG_INLINE void prefetch(const double *addr) + { + _mm_prefetch((SsePrefetchPtrType)(addr), _MM_HINT_T0); + } + template<> EIGEN_STRONG_INLINE void prefetch(const int *addr) + { + _mm_prefetch((SsePrefetchPtrType)(addr), _MM_HINT_T0); + } #endif #if EIGEN_COMP_MSVC_STRICT && EIGEN_OS_WIN64 -// The temporary variable fixes an internal compilation error in vs <= 2008 and a wrong-result bug in vs 2010 -// Direct of the struct members fixed bug #62. -template<> EIGEN_STRONG_INLINE float pfirst(const Packet4f& a) { return a.m128_f32[0]; } -template<> EIGEN_STRONG_INLINE double pfirst(const Packet2d& a) { return a.m128d_f64[0]; } -template<> EIGEN_STRONG_INLINE int pfirst(const Packet4i& a) { int x = _mm_cvtsi128_si32(a); return x; } + // The temporary variable fixes an internal compilation error in vs <= 2008 and a wrong-result bug in vs 2010 + // Direct of the struct members fixed bug #62. + template<> EIGEN_STRONG_INLINE float pfirst(const Packet4f &a) { return a.m128_f32[0]; } + template<> EIGEN_STRONG_INLINE double pfirst(const Packet2d &a) { return a.m128d_f64[0]; } + template<> EIGEN_STRONG_INLINE int pfirst(const Packet4i &a) + { + int x = _mm_cvtsi128_si32(a); + return x; + } #elif EIGEN_COMP_MSVC_STRICT -// The temporary variable fixes an internal compilation error in vs <= 2008 and a wrong-result bug in vs 2010 -template<> EIGEN_STRONG_INLINE float pfirst(const Packet4f& a) { float x = _mm_cvtss_f32(a); return x; } -template<> EIGEN_STRONG_INLINE double pfirst(const Packet2d& a) { double x = _mm_cvtsd_f64(a); return x; } -template<> EIGEN_STRONG_INLINE int pfirst(const Packet4i& a) { int x = _mm_cvtsi128_si32(a); return x; } + // The temporary variable fixes an internal compilation error in vs <= 2008 and a wrong-result bug in vs 2010 + template<> EIGEN_STRONG_INLINE float pfirst(const Packet4f &a) + { + float x = _mm_cvtss_f32(a); + return x; + } + template<> EIGEN_STRONG_INLINE double pfirst(const Packet2d &a) + { + double x = _mm_cvtsd_f64(a); + return x; + } + template<> EIGEN_STRONG_INLINE int pfirst(const Packet4i &a) + { + int x = _mm_cvtsi128_si32(a); + return x; + } #else -template<> EIGEN_STRONG_INLINE float pfirst(const Packet4f& a) { return _mm_cvtss_f32(a); } -template<> EIGEN_STRONG_INLINE double pfirst(const Packet2d& a) { return _mm_cvtsd_f64(a); } -template<> EIGEN_STRONG_INLINE int pfirst(const Packet4i& a) { return _mm_cvtsi128_si32(a); } -#endif - -template<> EIGEN_STRONG_INLINE Packet4f preverse(const Packet4f& a) -{ return _mm_shuffle_ps(a,a,0x1B); } -template<> EIGEN_STRONG_INLINE Packet2d preverse(const Packet2d& a) -{ return _mm_shuffle_pd(a,a,0x1); } -template<> EIGEN_STRONG_INLINE Packet4i preverse(const Packet4i& a) -{ return _mm_shuffle_epi32(a,0x1B); } - -template<> EIGEN_STRONG_INLINE Packet4f pabs(const Packet4f& a) -{ - const Packet4f mask = _mm_castsi128_ps(_mm_setr_epi32(0x7FFFFFFF,0x7FFFFFFF,0x7FFFFFFF,0x7FFFFFFF)); - return _mm_and_ps(a,mask); -} -template<> EIGEN_STRONG_INLINE Packet2d pabs(const Packet2d& a) -{ - const Packet2d mask = _mm_castsi128_pd(_mm_setr_epi32(0xFFFFFFFF,0x7FFFFFFF,0xFFFFFFFF,0x7FFFFFFF)); - return _mm_and_pd(a,mask); -} -template<> EIGEN_STRONG_INLINE Packet4i pabs(const Packet4i& a) -{ - #ifdef EIGEN_VECTORIZE_SSSE3 - return _mm_abs_epi32(a); - #else - Packet4i aux = _mm_srai_epi32(a,31); - return _mm_sub_epi32(_mm_xor_si128(a,aux),aux); - #endif -} + template<> EIGEN_STRONG_INLINE float pfirst(const Packet4f &a) { return _mm_cvtss_f32(a); } + template<> EIGEN_STRONG_INLINE double pfirst(const Packet2d &a) { return _mm_cvtsd_f64(a); } + template<> EIGEN_STRONG_INLINE int pfirst(const Packet4i &a) { return _mm_cvtsi128_si32(a); } +#endif + + template<> EIGEN_STRONG_INLINE Packet4f preverse(const Packet4f &a) { return _mm_shuffle_ps(a, a, 0x1B); } + template<> EIGEN_STRONG_INLINE Packet2d preverse(const Packet2d &a) { return _mm_shuffle_pd(a, a, 0x1); } + template<> EIGEN_STRONG_INLINE Packet4i preverse(const Packet4i &a) { return _mm_shuffle_epi32(a, 0x1B); } + + template<> EIGEN_STRONG_INLINE Packet4f pabs(const Packet4f &a) + { + const Packet4f mask = _mm_castsi128_ps(_mm_setr_epi32(0x7FFFFFFF, 0x7FFFFFFF, 0x7FFFFFFF, 0x7FFFFFFF)); + return _mm_and_ps(a, mask); + } + template<> EIGEN_STRONG_INLINE Packet2d pabs(const Packet2d &a) + { + const Packet2d mask = _mm_castsi128_pd(_mm_setr_epi32(0xFFFFFFFF, 0x7FFFFFFF, 0xFFFFFFFF, 0x7FFFFFFF)); + return _mm_and_pd(a, mask); + } + template<> EIGEN_STRONG_INLINE Packet4i pabs(const Packet4i &a) + { +#ifdef EIGEN_VECTORIZE_SSSE3 + return _mm_abs_epi32(a); +#else + Packet4i aux = _mm_srai_epi32(a, 31); + return _mm_sub_epi32(_mm_xor_si128(a, aux), aux); +#endif + } // with AVX, the default implementations based on pload1 are faster #ifndef __AVX__ -template<> EIGEN_STRONG_INLINE void -pbroadcast4(const float *a, - Packet4f& a0, Packet4f& a1, Packet4f& a2, Packet4f& a3) -{ - a3 = pload(a); - a0 = vec4f_swizzle1(a3, 0,0,0,0); - a1 = vec4f_swizzle1(a3, 1,1,1,1); - a2 = vec4f_swizzle1(a3, 2,2,2,2); - a3 = vec4f_swizzle1(a3, 3,3,3,3); -} -template<> EIGEN_STRONG_INLINE void -pbroadcast4(const double *a, - Packet2d& a0, Packet2d& a1, Packet2d& a2, Packet2d& a3) -{ + template<> + EIGEN_STRONG_INLINE void pbroadcast4(const float *a, Packet4f &a0, Packet4f &a1, Packet4f &a2, Packet4f &a3) + { + a3 = pload(a); + a0 = vec4f_swizzle1(a3, 0, 0, 0, 0); + a1 = vec4f_swizzle1(a3, 1, 1, 1, 1); + a2 = vec4f_swizzle1(a3, 2, 2, 2, 2); + a3 = vec4f_swizzle1(a3, 3, 3, 3, 3); + } + template<> + EIGEN_STRONG_INLINE void + pbroadcast4(const double *a, Packet2d &a0, Packet2d &a1, Packet2d &a2, Packet2d &a3) + { #ifdef EIGEN_VECTORIZE_SSE3 - a0 = _mm_loaddup_pd(a+0); - a1 = _mm_loaddup_pd(a+1); - a2 = _mm_loaddup_pd(a+2); - a3 = _mm_loaddup_pd(a+3); + a0 = _mm_loaddup_pd(a + 0); + a1 = _mm_loaddup_pd(a + 1); + a2 = _mm_loaddup_pd(a + 2); + a3 = _mm_loaddup_pd(a + 3); #else - a1 = pload(a); - a0 = vec2d_swizzle1(a1, 0,0); - a1 = vec2d_swizzle1(a1, 1,1); - a3 = pload(a+2); - a2 = vec2d_swizzle1(a3, 0,0); - a3 = vec2d_swizzle1(a3, 1,1); + a1 = pload(a); + a0 = vec2d_swizzle1(a1, 0, 0); + a1 = vec2d_swizzle1(a1, 1, 1); + a3 = pload(a + 2); + a2 = vec2d_swizzle1(a3, 0, 0); + a3 = vec2d_swizzle1(a3, 1, 1); #endif -} + } #endif -EIGEN_STRONG_INLINE void punpackp(Packet4f* vecs) -{ - vecs[1] = _mm_castsi128_ps(_mm_shuffle_epi32(_mm_castps_si128(vecs[0]), 0x55)); - vecs[2] = _mm_castsi128_ps(_mm_shuffle_epi32(_mm_castps_si128(vecs[0]), 0xAA)); - vecs[3] = _mm_castsi128_ps(_mm_shuffle_epi32(_mm_castps_si128(vecs[0]), 0xFF)); - vecs[0] = _mm_castsi128_ps(_mm_shuffle_epi32(_mm_castps_si128(vecs[0]), 0x00)); -} + EIGEN_STRONG_INLINE void punpackp(Packet4f *vecs) + { + vecs[1] = _mm_castsi128_ps(_mm_shuffle_epi32(_mm_castps_si128(vecs[0]), 0x55)); + vecs[2] = _mm_castsi128_ps(_mm_shuffle_epi32(_mm_castps_si128(vecs[0]), 0xAA)); + vecs[3] = _mm_castsi128_ps(_mm_shuffle_epi32(_mm_castps_si128(vecs[0]), 0xFF)); + vecs[0] = _mm_castsi128_ps(_mm_shuffle_epi32(_mm_castps_si128(vecs[0]), 0x00)); + } #ifdef EIGEN_VECTORIZE_SSE3 -template<> EIGEN_STRONG_INLINE Packet4f preduxp(const Packet4f* vecs) -{ - return _mm_hadd_ps(_mm_hadd_ps(vecs[0], vecs[1]),_mm_hadd_ps(vecs[2], vecs[3])); -} + template<> EIGEN_STRONG_INLINE Packet4f preduxp(const Packet4f *vecs) + { + return _mm_hadd_ps(_mm_hadd_ps(vecs[0], vecs[1]), _mm_hadd_ps(vecs[2], vecs[3])); + } -template<> EIGEN_STRONG_INLINE Packet2d preduxp(const Packet2d* vecs) -{ - return _mm_hadd_pd(vecs[0], vecs[1]); -} + template<> EIGEN_STRONG_INLINE Packet2d preduxp(const Packet2d *vecs) + { + return _mm_hadd_pd(vecs[0], vecs[1]); + } #else -template<> EIGEN_STRONG_INLINE Packet4f preduxp(const Packet4f* vecs) -{ - Packet4f tmp0, tmp1, tmp2; - tmp0 = _mm_unpacklo_ps(vecs[0], vecs[1]); - tmp1 = _mm_unpackhi_ps(vecs[0], vecs[1]); - tmp2 = _mm_unpackhi_ps(vecs[2], vecs[3]); - tmp0 = _mm_add_ps(tmp0, tmp1); - tmp1 = _mm_unpacklo_ps(vecs[2], vecs[3]); - tmp1 = _mm_add_ps(tmp1, tmp2); - tmp2 = _mm_movehl_ps(tmp1, tmp0); - tmp0 = _mm_movelh_ps(tmp0, tmp1); - return _mm_add_ps(tmp0, tmp2); -} - -template<> EIGEN_STRONG_INLINE Packet2d preduxp(const Packet2d* vecs) -{ - return _mm_add_pd(_mm_unpacklo_pd(vecs[0], vecs[1]), _mm_unpackhi_pd(vecs[0], vecs[1])); -} -#endif // SSE3 - -template<> EIGEN_STRONG_INLINE float predux(const Packet4f& a) -{ - // Disable SSE3 _mm_hadd_pd that is extremely slow on all existing Intel's architectures - // (from Nehalem to Haswell) -// #ifdef EIGEN_VECTORIZE_SSE3 -// Packet4f tmp = _mm_add_ps(a, vec4f_swizzle1(a,2,3,2,3)); -// return pfirst(_mm_hadd_ps(tmp, tmp)); -// #else - Packet4f tmp = _mm_add_ps(a, _mm_movehl_ps(a,a)); - return pfirst(_mm_add_ss(tmp, _mm_shuffle_ps(tmp,tmp, 1))); -// #endif -} - -template<> EIGEN_STRONG_INLINE double predux(const Packet2d& a) -{ - // Disable SSE3 _mm_hadd_pd that is extremely slow on all existing Intel's architectures - // (from Nehalem to Haswell) -// #ifdef EIGEN_VECTORIZE_SSE3 -// return pfirst(_mm_hadd_pd(a, a)); -// #else - return pfirst(_mm_add_sd(a, _mm_unpackhi_pd(a,a))); -// #endif -} + template<> EIGEN_STRONG_INLINE Packet4f preduxp(const Packet4f *vecs) + { + Packet4f tmp0, tmp1, tmp2; + tmp0 = _mm_unpacklo_ps(vecs[0], vecs[1]); + tmp1 = _mm_unpackhi_ps(vecs[0], vecs[1]); + tmp2 = _mm_unpackhi_ps(vecs[2], vecs[3]); + tmp0 = _mm_add_ps(tmp0, tmp1); + tmp1 = _mm_unpacklo_ps(vecs[2], vecs[3]); + tmp1 = _mm_add_ps(tmp1, tmp2); + tmp2 = _mm_movehl_ps(tmp1, tmp0); + tmp0 = _mm_movelh_ps(tmp0, tmp1); + return _mm_add_ps(tmp0, tmp2); + } + + template<> EIGEN_STRONG_INLINE Packet2d preduxp(const Packet2d *vecs) + { + return _mm_add_pd(_mm_unpacklo_pd(vecs[0], vecs[1]), _mm_unpackhi_pd(vecs[0], vecs[1])); + } +#endif// SSE3 + + template<> EIGEN_STRONG_INLINE float predux(const Packet4f &a) + { + // Disable SSE3 _mm_hadd_pd that is extremely slow on all existing Intel's architectures + // (from Nehalem to Haswell) + // #ifdef EIGEN_VECTORIZE_SSE3 + // Packet4f tmp = _mm_add_ps(a, vec4f_swizzle1(a,2,3,2,3)); + // return pfirst(_mm_hadd_ps(tmp, tmp)); + // #else + Packet4f tmp = _mm_add_ps(a, _mm_movehl_ps(a, a)); + return pfirst(_mm_add_ss(tmp, _mm_shuffle_ps(tmp, tmp, 1))); + // #endif + } + + template<> EIGEN_STRONG_INLINE double predux(const Packet2d &a) + { + // Disable SSE3 _mm_hadd_pd that is extremely slow on all existing Intel's architectures + // (from Nehalem to Haswell) + // #ifdef EIGEN_VECTORIZE_SSE3 + // return pfirst(_mm_hadd_pd(a, a)); + // #else + return pfirst(_mm_add_sd(a, _mm_unpackhi_pd(a, a))); + // #endif + } #ifdef EIGEN_VECTORIZE_SSSE3 -template<> EIGEN_STRONG_INLINE Packet4i preduxp(const Packet4i* vecs) -{ - return _mm_hadd_epi32(_mm_hadd_epi32(vecs[0], vecs[1]),_mm_hadd_epi32(vecs[2], vecs[3])); -} -template<> EIGEN_STRONG_INLINE int predux(const Packet4i& a) -{ - Packet4i tmp0 = _mm_hadd_epi32(a,a); - return pfirst(_mm_hadd_epi32(tmp0,tmp0)); -} + template<> EIGEN_STRONG_INLINE Packet4i preduxp(const Packet4i *vecs) + { + return _mm_hadd_epi32(_mm_hadd_epi32(vecs[0], vecs[1]), _mm_hadd_epi32(vecs[2], vecs[3])); + } + template<> EIGEN_STRONG_INLINE int predux(const Packet4i &a) + { + Packet4i tmp0 = _mm_hadd_epi32(a, a); + return pfirst(_mm_hadd_epi32(tmp0, tmp0)); + } #else -template<> EIGEN_STRONG_INLINE int predux(const Packet4i& a) -{ - Packet4i tmp = _mm_add_epi32(a, _mm_unpackhi_epi64(a,a)); - return pfirst(tmp) + pfirst(_mm_shuffle_epi32(tmp, 1)); -} - -template<> EIGEN_STRONG_INLINE Packet4i preduxp(const Packet4i* vecs) -{ - Packet4i tmp0, tmp1, tmp2; - tmp0 = _mm_unpacklo_epi32(vecs[0], vecs[1]); - tmp1 = _mm_unpackhi_epi32(vecs[0], vecs[1]); - tmp2 = _mm_unpackhi_epi32(vecs[2], vecs[3]); - tmp0 = _mm_add_epi32(tmp0, tmp1); - tmp1 = _mm_unpacklo_epi32(vecs[2], vecs[3]); - tmp1 = _mm_add_epi32(tmp1, tmp2); - tmp2 = _mm_unpacklo_epi64(tmp0, tmp1); - tmp0 = _mm_unpackhi_epi64(tmp0, tmp1); - return _mm_add_epi32(tmp0, tmp2); -} -#endif -// Other reduction functions: - -// mul -template<> EIGEN_STRONG_INLINE float predux_mul(const Packet4f& a) -{ - Packet4f tmp = _mm_mul_ps(a, _mm_movehl_ps(a,a)); - return pfirst(_mm_mul_ss(tmp, _mm_shuffle_ps(tmp,tmp, 1))); -} -template<> EIGEN_STRONG_INLINE double predux_mul(const Packet2d& a) -{ - return pfirst(_mm_mul_sd(a, _mm_unpackhi_pd(a,a))); -} -template<> EIGEN_STRONG_INLINE int predux_mul(const Packet4i& a) -{ - // after some experiments, it is seems this is the fastest way to implement it - // for GCC (eg., reusing pmul is very slow !) - // TODO try to call _mm_mul_epu32 directly - EIGEN_ALIGN16 int aux[4]; - pstore(aux, a); - return (aux[0] * aux[1]) * (aux[2] * aux[3]);; -} - -// min -template<> EIGEN_STRONG_INLINE float predux_min(const Packet4f& a) -{ - Packet4f tmp = _mm_min_ps(a, _mm_movehl_ps(a,a)); - return pfirst(_mm_min_ss(tmp, _mm_shuffle_ps(tmp,tmp, 1))); -} -template<> EIGEN_STRONG_INLINE double predux_min(const Packet2d& a) -{ - return pfirst(_mm_min_sd(a, _mm_unpackhi_pd(a,a))); -} -template<> EIGEN_STRONG_INLINE int predux_min(const Packet4i& a) -{ + template<> EIGEN_STRONG_INLINE int predux(const Packet4i &a) + { + Packet4i tmp = _mm_add_epi32(a, _mm_unpackhi_epi64(a, a)); + return pfirst(tmp) + pfirst(_mm_shuffle_epi32(tmp, 1)); + } + + template<> EIGEN_STRONG_INLINE Packet4i preduxp(const Packet4i *vecs) + { + Packet4i tmp0, tmp1, tmp2; + tmp0 = _mm_unpacklo_epi32(vecs[0], vecs[1]); + tmp1 = _mm_unpackhi_epi32(vecs[0], vecs[1]); + tmp2 = _mm_unpackhi_epi32(vecs[2], vecs[3]); + tmp0 = _mm_add_epi32(tmp0, tmp1); + tmp1 = _mm_unpacklo_epi32(vecs[2], vecs[3]); + tmp1 = _mm_add_epi32(tmp1, tmp2); + tmp2 = _mm_unpacklo_epi64(tmp0, tmp1); + tmp0 = _mm_unpackhi_epi64(tmp0, tmp1); + return _mm_add_epi32(tmp0, tmp2); + } +#endif + // Other reduction functions: + + // mul + template<> EIGEN_STRONG_INLINE float predux_mul(const Packet4f &a) + { + Packet4f tmp = _mm_mul_ps(a, _mm_movehl_ps(a, a)); + return pfirst(_mm_mul_ss(tmp, _mm_shuffle_ps(tmp, tmp, 1))); + } + template<> EIGEN_STRONG_INLINE double predux_mul(const Packet2d &a) + { + return pfirst(_mm_mul_sd(a, _mm_unpackhi_pd(a, a))); + } + template<> EIGEN_STRONG_INLINE int predux_mul(const Packet4i &a) + { + // after some experiments, it is seems this is the fastest way to implement it + // for GCC (eg., reusing pmul is very slow !) + // TODO try to call _mm_mul_epu32 directly + EIGEN_ALIGN16 int aux[4]; + pstore(aux, a); + return (aux[0] * aux[1]) * (aux[2] * aux[3]); + ; + } + + // min + template<> EIGEN_STRONG_INLINE float predux_min(const Packet4f &a) + { + Packet4f tmp = _mm_min_ps(a, _mm_movehl_ps(a, a)); + return pfirst(_mm_min_ss(tmp, _mm_shuffle_ps(tmp, tmp, 1))); + } + template<> EIGEN_STRONG_INLINE double predux_min(const Packet2d &a) + { + return pfirst(_mm_min_sd(a, _mm_unpackhi_pd(a, a))); + } + template<> EIGEN_STRONG_INLINE int predux_min(const Packet4i &a) + { #ifdef EIGEN_VECTORIZE_SSE4_1 - Packet4i tmp = _mm_min_epi32(a, _mm_shuffle_epi32(a, _MM_SHUFFLE(0,0,3,2))); - return pfirst(_mm_min_epi32(tmp,_mm_shuffle_epi32(tmp, 1))); + Packet4i tmp = _mm_min_epi32(a, _mm_shuffle_epi32(a, _MM_SHUFFLE(0, 0, 3, 2))); + return pfirst(_mm_min_epi32(tmp, _mm_shuffle_epi32(tmp, 1))); #else - // after some experiments, it is seems this is the fastest way to implement it - // for GCC (eg., it does not like using std::min after the pstore !!) - EIGEN_ALIGN16 int aux[4]; - pstore(aux, a); - int aux0 = aux[0] EIGEN_STRONG_INLINE float predux_max(const Packet4f& a) -{ - Packet4f tmp = _mm_max_ps(a, _mm_movehl_ps(a,a)); - return pfirst(_mm_max_ss(tmp, _mm_shuffle_ps(tmp,tmp, 1))); -} -template<> EIGEN_STRONG_INLINE double predux_max(const Packet2d& a) -{ - return pfirst(_mm_max_sd(a, _mm_unpackhi_pd(a,a))); -} -template<> EIGEN_STRONG_INLINE int predux_max(const Packet4i& a) -{ + // after some experiments, it is seems this is the fastest way to implement it + // for GCC (eg., it does not like using std::min after the pstore !!) + EIGEN_ALIGN16 int aux[4]; + pstore(aux, a); + int aux0 = aux[0] < aux[1] ? aux[0] : aux[1]; + int aux2 = aux[2] < aux[3] ? aux[2] : aux[3]; + return aux0 < aux2 ? aux0 : aux2; +#endif// EIGEN_VECTORIZE_SSE4_1 + } + + // max + template<> EIGEN_STRONG_INLINE float predux_max(const Packet4f &a) + { + Packet4f tmp = _mm_max_ps(a, _mm_movehl_ps(a, a)); + return pfirst(_mm_max_ss(tmp, _mm_shuffle_ps(tmp, tmp, 1))); + } + template<> EIGEN_STRONG_INLINE double predux_max(const Packet2d &a) + { + return pfirst(_mm_max_sd(a, _mm_unpackhi_pd(a, a))); + } + template<> EIGEN_STRONG_INLINE int predux_max(const Packet4i &a) + { #ifdef EIGEN_VECTORIZE_SSE4_1 - Packet4i tmp = _mm_max_epi32(a, _mm_shuffle_epi32(a, _MM_SHUFFLE(0,0,3,2))); - return pfirst(_mm_max_epi32(tmp,_mm_shuffle_epi32(tmp, 1))); + Packet4i tmp = _mm_max_epi32(a, _mm_shuffle_epi32(a, _MM_SHUFFLE(0, 0, 3, 2))); + return pfirst(_mm_max_epi32(tmp, _mm_shuffle_epi32(tmp, 1))); #else - // after some experiments, it is seems this is the fastest way to implement it - // for GCC (eg., it does not like using std::min after the pstore !!) - EIGEN_ALIGN16 int aux[4]; - pstore(aux, a); - int aux0 = aux[0]>aux[1] ? aux[0] : aux[1]; - int aux2 = aux[2]>aux[3] ? aux[2] : aux[3]; - return aux0>aux2 ? aux0 : aux2; -#endif // EIGEN_VECTORIZE_SSE4_1 -} + // after some experiments, it is seems this is the fastest way to implement it + // for GCC (eg., it does not like using std::min after the pstore !!) + EIGEN_ALIGN16 int aux[4]; + pstore(aux, a); + int aux0 = aux[0] > aux[1] ? aux[0] : aux[1]; + int aux2 = aux[2] > aux[3] ? aux[2] : aux[3]; + return aux0 > aux2 ? aux0 : aux2; +#endif// EIGEN_VECTORIZE_SSE4_1 + } #if EIGEN_COMP_GNUC // template <> EIGEN_STRONG_INLINE Packet4f pmadd(const Packet4f& a, const Packet4f& b, const Packet4f& c) @@ -682,214 +861,207 @@ template<> EIGEN_STRONG_INLINE int predux_max(const Packet4i& a) #endif #ifdef EIGEN_VECTORIZE_SSSE3 -// SSSE3 versions -template -struct palign_impl -{ - static EIGEN_STRONG_INLINE void run(Packet4f& first, const Packet4f& second) + // SSSE3 versions + template struct palign_impl { - if (Offset!=0) - first = _mm_castsi128_ps(_mm_alignr_epi8(_mm_castps_si128(second), _mm_castps_si128(first), Offset*4)); - } -}; + static EIGEN_STRONG_INLINE void run(Packet4f &first, const Packet4f &second) + { + if (Offset != 0) + first = _mm_castsi128_ps(_mm_alignr_epi8(_mm_castps_si128(second), _mm_castps_si128(first), Offset * 4)); + } + }; -template -struct palign_impl -{ - static EIGEN_STRONG_INLINE void run(Packet4i& first, const Packet4i& second) + template struct palign_impl { - if (Offset!=0) - first = _mm_alignr_epi8(second,first, Offset*4); - } -}; + static EIGEN_STRONG_INLINE void run(Packet4i &first, const Packet4i &second) + { + if (Offset != 0) first = _mm_alignr_epi8(second, first, Offset * 4); + } + }; -template -struct palign_impl -{ - static EIGEN_STRONG_INLINE void run(Packet2d& first, const Packet2d& second) + template struct palign_impl { - if (Offset==1) - first = _mm_castsi128_pd(_mm_alignr_epi8(_mm_castpd_si128(second), _mm_castpd_si128(first), 8)); - } -}; + static EIGEN_STRONG_INLINE void run(Packet2d &first, const Packet2d &second) + { + if (Offset == 1) first = _mm_castsi128_pd(_mm_alignr_epi8(_mm_castpd_si128(second), _mm_castpd_si128(first), 8)); + } + }; #else -// SSE2 versions -template -struct palign_impl -{ - static EIGEN_STRONG_INLINE void run(Packet4f& first, const Packet4f& second) + // SSE2 versions + template struct palign_impl { - if (Offset==1) + static EIGEN_STRONG_INLINE void run(Packet4f &first, const Packet4f &second) { - first = _mm_move_ss(first,second); - first = _mm_castsi128_ps(_mm_shuffle_epi32(_mm_castps_si128(first),0x39)); + if (Offset == 1) { + first = _mm_move_ss(first, second); + first = _mm_castsi128_ps(_mm_shuffle_epi32(_mm_castps_si128(first), 0x39)); + } else if (Offset == 2) { + first = _mm_movehl_ps(first, first); + first = _mm_movelh_ps(first, second); + } else if (Offset == 3) { + first = _mm_move_ss(first, second); + first = _mm_shuffle_ps(first, second, 0x93); + } } - else if (Offset==2) + }; + + template struct palign_impl + { + static EIGEN_STRONG_INLINE void run(Packet4i &first, const Packet4i &second) { - first = _mm_movehl_ps(first,first); - first = _mm_movelh_ps(first,second); + if (Offset == 1) { + first = _mm_castps_si128(_mm_move_ss(_mm_castsi128_ps(first), _mm_castsi128_ps(second))); + first = _mm_shuffle_epi32(first, 0x39); + } else if (Offset == 2) { + first = _mm_castps_si128(_mm_movehl_ps(_mm_castsi128_ps(first), _mm_castsi128_ps(first))); + first = _mm_castps_si128(_mm_movelh_ps(_mm_castsi128_ps(first), _mm_castsi128_ps(second))); + } else if (Offset == 3) { + first = _mm_castps_si128(_mm_move_ss(_mm_castsi128_ps(first), _mm_castsi128_ps(second))); + first = _mm_castps_si128(_mm_shuffle_ps(_mm_castsi128_ps(first), _mm_castsi128_ps(second), 0x93)); + } } - else if (Offset==3) + }; + + template struct palign_impl + { + static EIGEN_STRONG_INLINE void run(Packet2d &first, const Packet2d &second) { - first = _mm_move_ss(first,second); - first = _mm_shuffle_ps(first,second,0x93); + if (Offset == 1) { + first = _mm_castps_pd(_mm_movehl_ps(_mm_castpd_ps(first), _mm_castpd_ps(first))); + first = _mm_castps_pd(_mm_movelh_ps(_mm_castpd_ps(first), _mm_castpd_ps(second))); + } } + }; +#endif + + EIGEN_DEVICE_FUNC inline void ptranspose(PacketBlock &kernel) + { + _MM_TRANSPOSE4_PS(kernel.packet[0], kernel.packet[1], kernel.packet[2], kernel.packet[3]); } -}; -template -struct palign_impl -{ - static EIGEN_STRONG_INLINE void run(Packet4i& first, const Packet4i& second) + EIGEN_DEVICE_FUNC inline void ptranspose(PacketBlock &kernel) { - if (Offset==1) - { - first = _mm_castps_si128(_mm_move_ss(_mm_castsi128_ps(first),_mm_castsi128_ps(second))); - first = _mm_shuffle_epi32(first,0x39); - } - else if (Offset==2) - { - first = _mm_castps_si128(_mm_movehl_ps(_mm_castsi128_ps(first),_mm_castsi128_ps(first))); - first = _mm_castps_si128(_mm_movelh_ps(_mm_castsi128_ps(first),_mm_castsi128_ps(second))); - } - else if (Offset==3) - { - first = _mm_castps_si128(_mm_move_ss(_mm_castsi128_ps(first),_mm_castsi128_ps(second))); - first = _mm_castps_si128(_mm_shuffle_ps(_mm_castsi128_ps(first),_mm_castsi128_ps(second),0x93)); - } + __m128d tmp = _mm_unpackhi_pd(kernel.packet[0], kernel.packet[1]); + kernel.packet[0] = _mm_unpacklo_pd(kernel.packet[0], kernel.packet[1]); + kernel.packet[1] = tmp; } -}; -template -struct palign_impl -{ - static EIGEN_STRONG_INLINE void run(Packet2d& first, const Packet2d& second) + EIGEN_DEVICE_FUNC inline void ptranspose(PacketBlock &kernel) { - if (Offset==1) - { - first = _mm_castps_pd(_mm_movehl_ps(_mm_castpd_ps(first),_mm_castpd_ps(first))); - first = _mm_castps_pd(_mm_movelh_ps(_mm_castpd_ps(first),_mm_castpd_ps(second))); - } + __m128i T0 = _mm_unpacklo_epi32(kernel.packet[0], kernel.packet[1]); + __m128i T1 = _mm_unpacklo_epi32(kernel.packet[2], kernel.packet[3]); + __m128i T2 = _mm_unpackhi_epi32(kernel.packet[0], kernel.packet[1]); + __m128i T3 = _mm_unpackhi_epi32(kernel.packet[2], kernel.packet[3]); + + kernel.packet[0] = _mm_unpacklo_epi64(T0, T1); + kernel.packet[1] = _mm_unpackhi_epi64(T0, T1); + kernel.packet[2] = _mm_unpacklo_epi64(T2, T3); + kernel.packet[3] = _mm_unpackhi_epi64(T2, T3); } -}; -#endif - -EIGEN_DEVICE_FUNC inline void -ptranspose(PacketBlock& kernel) { - _MM_TRANSPOSE4_PS(kernel.packet[0], kernel.packet[1], kernel.packet[2], kernel.packet[3]); -} - -EIGEN_DEVICE_FUNC inline void -ptranspose(PacketBlock& kernel) { - __m128d tmp = _mm_unpackhi_pd(kernel.packet[0], kernel.packet[1]); - kernel.packet[0] = _mm_unpacklo_pd(kernel.packet[0], kernel.packet[1]); - kernel.packet[1] = tmp; -} - -EIGEN_DEVICE_FUNC inline void -ptranspose(PacketBlock& kernel) { - __m128i T0 = _mm_unpacklo_epi32(kernel.packet[0], kernel.packet[1]); - __m128i T1 = _mm_unpacklo_epi32(kernel.packet[2], kernel.packet[3]); - __m128i T2 = _mm_unpackhi_epi32(kernel.packet[0], kernel.packet[1]); - __m128i T3 = _mm_unpackhi_epi32(kernel.packet[2], kernel.packet[3]); - - kernel.packet[0] = _mm_unpacklo_epi64(T0, T1); - kernel.packet[1] = _mm_unpackhi_epi64(T0, T1); - kernel.packet[2] = _mm_unpacklo_epi64(T2, T3); - kernel.packet[3] = _mm_unpackhi_epi64(T2, T3); -} - -template<> EIGEN_STRONG_INLINE Packet4i pblend(const Selector<4>& ifPacket, const Packet4i& thenPacket, const Packet4i& elsePacket) { - const __m128i zero = _mm_setzero_si128(); - const __m128i select = _mm_set_epi32(ifPacket.select[3], ifPacket.select[2], ifPacket.select[1], ifPacket.select[0]); - __m128i false_mask = _mm_cmpeq_epi32(select, zero); + + template<> + EIGEN_STRONG_INLINE Packet4i pblend(const Selector<4> &ifPacket, + const Packet4i &thenPacket, + const Packet4i &elsePacket) + { + const __m128i zero = _mm_setzero_si128(); + const __m128i select = + _mm_set_epi32(ifPacket.select[3], ifPacket.select[2], ifPacket.select[1], ifPacket.select[0]); + __m128i false_mask = _mm_cmpeq_epi32(select, zero); #ifdef EIGEN_VECTORIZE_SSE4_1 - return _mm_blendv_epi8(thenPacket, elsePacket, false_mask); + return _mm_blendv_epi8(thenPacket, elsePacket, false_mask); #else - return _mm_or_si128(_mm_andnot_si128(false_mask, thenPacket), _mm_and_si128(false_mask, elsePacket)); + return _mm_or_si128(_mm_andnot_si128(false_mask, thenPacket), _mm_and_si128(false_mask, elsePacket)); #endif -} -template<> EIGEN_STRONG_INLINE Packet4f pblend(const Selector<4>& ifPacket, const Packet4f& thenPacket, const Packet4f& elsePacket) { - const __m128 zero = _mm_setzero_ps(); - const __m128 select = _mm_set_ps(ifPacket.select[3], ifPacket.select[2], ifPacket.select[1], ifPacket.select[0]); - __m128 false_mask = _mm_cmpeq_ps(select, zero); + } + template<> + EIGEN_STRONG_INLINE Packet4f pblend(const Selector<4> &ifPacket, + const Packet4f &thenPacket, + const Packet4f &elsePacket) + { + const __m128 zero = _mm_setzero_ps(); + const __m128 select = _mm_set_ps(ifPacket.select[3], ifPacket.select[2], ifPacket.select[1], ifPacket.select[0]); + __m128 false_mask = _mm_cmpeq_ps(select, zero); #ifdef EIGEN_VECTORIZE_SSE4_1 - return _mm_blendv_ps(thenPacket, elsePacket, false_mask); + return _mm_blendv_ps(thenPacket, elsePacket, false_mask); #else - return _mm_or_ps(_mm_andnot_ps(false_mask, thenPacket), _mm_and_ps(false_mask, elsePacket)); + return _mm_or_ps(_mm_andnot_ps(false_mask, thenPacket), _mm_and_ps(false_mask, elsePacket)); #endif -} -template<> EIGEN_STRONG_INLINE Packet2d pblend(const Selector<2>& ifPacket, const Packet2d& thenPacket, const Packet2d& elsePacket) { - const __m128d zero = _mm_setzero_pd(); - const __m128d select = _mm_set_pd(ifPacket.select[1], ifPacket.select[0]); - __m128d false_mask = _mm_cmpeq_pd(select, zero); + } + template<> + EIGEN_STRONG_INLINE Packet2d pblend(const Selector<2> &ifPacket, + const Packet2d &thenPacket, + const Packet2d &elsePacket) + { + const __m128d zero = _mm_setzero_pd(); + const __m128d select = _mm_set_pd(ifPacket.select[1], ifPacket.select[0]); + __m128d false_mask = _mm_cmpeq_pd(select, zero); #ifdef EIGEN_VECTORIZE_SSE4_1 - return _mm_blendv_pd(thenPacket, elsePacket, false_mask); + return _mm_blendv_pd(thenPacket, elsePacket, false_mask); #else - return _mm_or_pd(_mm_andnot_pd(false_mask, thenPacket), _mm_and_pd(false_mask, elsePacket)); + return _mm_or_pd(_mm_andnot_pd(false_mask, thenPacket), _mm_and_pd(false_mask, elsePacket)); #endif -} + } -template<> EIGEN_STRONG_INLINE Packet4f pinsertfirst(const Packet4f& a, float b) -{ + template<> EIGEN_STRONG_INLINE Packet4f pinsertfirst(const Packet4f &a, float b) + { #ifdef EIGEN_VECTORIZE_SSE4_1 - return _mm_blend_ps(a,pset1(b),1); + return _mm_blend_ps(a, pset1(b), 1); #else - return _mm_move_ss(a, _mm_load_ss(&b)); + return _mm_move_ss(a, _mm_load_ss(&b)); #endif -} + } -template<> EIGEN_STRONG_INLINE Packet2d pinsertfirst(const Packet2d& a, double b) -{ + template<> EIGEN_STRONG_INLINE Packet2d pinsertfirst(const Packet2d &a, double b) + { #ifdef EIGEN_VECTORIZE_SSE4_1 - return _mm_blend_pd(a,pset1(b),1); + return _mm_blend_pd(a, pset1(b), 1); #else - return _mm_move_sd(a, _mm_load_sd(&b)); + return _mm_move_sd(a, _mm_load_sd(&b)); #endif -} + } -template<> EIGEN_STRONG_INLINE Packet4f pinsertlast(const Packet4f& a, float b) -{ + template<> EIGEN_STRONG_INLINE Packet4f pinsertlast(const Packet4f &a, float b) + { #ifdef EIGEN_VECTORIZE_SSE4_1 - return _mm_blend_ps(a,pset1(b),(1<<3)); + return _mm_blend_ps(a, pset1(b), (1 << 3)); #else - const Packet4f mask = _mm_castsi128_ps(_mm_setr_epi32(0x0,0x0,0x0,0xFFFFFFFF)); - return _mm_or_ps(_mm_andnot_ps(mask, a), _mm_and_ps(mask, pset1(b))); + const Packet4f mask = _mm_castsi128_ps(_mm_setr_epi32(0x0, 0x0, 0x0, 0xFFFFFFFF)); + return _mm_or_ps(_mm_andnot_ps(mask, a), _mm_and_ps(mask, pset1(b))); #endif -} + } -template<> EIGEN_STRONG_INLINE Packet2d pinsertlast(const Packet2d& a, double b) -{ + template<> EIGEN_STRONG_INLINE Packet2d pinsertlast(const Packet2d &a, double b) + { #ifdef EIGEN_VECTORIZE_SSE4_1 - return _mm_blend_pd(a,pset1(b),(1<<1)); + return _mm_blend_pd(a, pset1(b), (1 << 1)); #else - const Packet2d mask = _mm_castsi128_pd(_mm_setr_epi32(0x0,0x0,0xFFFFFFFF,0xFFFFFFFF)); - return _mm_or_pd(_mm_andnot_pd(mask, a), _mm_and_pd(mask, pset1(b))); + const Packet2d mask = _mm_castsi128_pd(_mm_setr_epi32(0x0, 0x0, 0xFFFFFFFF, 0xFFFFFFFF)); + return _mm_or_pd(_mm_andnot_pd(mask, a), _mm_and_pd(mask, pset1(b))); #endif -} + } // Scalar path for pmadd with FMA to ensure consistency with vectorized path. #ifdef __FMA__ -template<> EIGEN_STRONG_INLINE float pmadd(const float& a, const float& b, const float& c) { - return ::fmaf(a,b,c); -} -template<> EIGEN_STRONG_INLINE double pmadd(const double& a, const double& b, const double& c) { - return ::fma(a,b,c); -} + template<> EIGEN_STRONG_INLINE float pmadd(const float &a, const float &b, const float &c) { return ::fmaf(a, b, c); } + template<> EIGEN_STRONG_INLINE double pmadd(const double &a, const double &b, const double &c) + { + return ::fma(a, b, c); + } #endif -} // end namespace internal +}// end namespace internal -} // end namespace Eigen +}// end namespace Eigen #if EIGEN_COMP_PGI // PGI++ does not define the following intrinsics in C++ mode. -static inline __m128 _mm_castpd_ps (__m128d x) { return reinterpret_cast<__m128&>(x); } -static inline __m128i _mm_castpd_si128(__m128d x) { return reinterpret_cast<__m128i&>(x); } -static inline __m128d _mm_castps_pd (__m128 x) { return reinterpret_cast<__m128d&>(x); } -static inline __m128i _mm_castps_si128(__m128 x) { return reinterpret_cast<__m128i&>(x); } -static inline __m128 _mm_castsi128_ps(__m128i x) { return reinterpret_cast<__m128&>(x); } -static inline __m128d _mm_castsi128_pd(__m128i x) { return reinterpret_cast<__m128d&>(x); } +static inline __m128 _mm_castpd_ps(__m128d x) { return reinterpret_cast<__m128 &>(x); } +static inline __m128i _mm_castpd_si128(__m128d x) { return reinterpret_cast<__m128i &>(x); } +static inline __m128d _mm_castps_pd(__m128 x) { return reinterpret_cast<__m128d &>(x); } +static inline __m128i _mm_castps_si128(__m128 x) { return reinterpret_cast<__m128i &>(x); } +static inline __m128 _mm_castsi128_ps(__m128i x) { return reinterpret_cast<__m128 &>(x); } +static inline __m128d _mm_castsi128_pd(__m128i x) { return reinterpret_cast<__m128d &>(x); } #endif -#endif // EIGEN_PACKET_MATH_SSE_H +#endif// EIGEN_PACKET_MATH_SSE_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/arch/SSE/TypeCasting.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/arch/SSE/TypeCasting.h index c6ca8c71..24925a2e 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/arch/SSE/TypeCasting.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/arch/SSE/TypeCasting.h @@ -15,63 +15,45 @@ namespace Eigen { namespace internal { #ifndef EIGEN_VECTORIZE_AVX -template <> -struct type_casting_traits { - enum { - VectorizedCast = 1, - SrcCoeffRatio = 1, - TgtCoeffRatio = 1 + template<> struct type_casting_traits + { + enum { VectorizedCast = 1, SrcCoeffRatio = 1, TgtCoeffRatio = 1 }; }; -}; -template <> -struct type_casting_traits { - enum { - VectorizedCast = 1, - SrcCoeffRatio = 1, - TgtCoeffRatio = 1 + template<> struct type_casting_traits + { + enum { VectorizedCast = 1, SrcCoeffRatio = 1, TgtCoeffRatio = 1 }; }; -}; -template <> -struct type_casting_traits { - enum { - VectorizedCast = 1, - SrcCoeffRatio = 2, - TgtCoeffRatio = 1 + template<> struct type_casting_traits + { + enum { VectorizedCast = 1, SrcCoeffRatio = 2, TgtCoeffRatio = 1 }; }; -}; -template <> -struct type_casting_traits { - enum { - VectorizedCast = 1, - SrcCoeffRatio = 1, - TgtCoeffRatio = 2 + template<> struct type_casting_traits + { + enum { VectorizedCast = 1, SrcCoeffRatio = 1, TgtCoeffRatio = 2 }; }; -}; #endif -template<> EIGEN_STRONG_INLINE Packet4i pcast(const Packet4f& a) { - return _mm_cvttps_epi32(a); -} + template<> EIGEN_STRONG_INLINE Packet4i pcast(const Packet4f &a) { return _mm_cvttps_epi32(a); } -template<> EIGEN_STRONG_INLINE Packet4f pcast(const Packet4i& a) { - return _mm_cvtepi32_ps(a); -} + template<> EIGEN_STRONG_INLINE Packet4f pcast(const Packet4i &a) { return _mm_cvtepi32_ps(a); } -template<> EIGEN_STRONG_INLINE Packet4f pcast(const Packet2d& a, const Packet2d& b) { - return _mm_shuffle_ps(_mm_cvtpd_ps(a), _mm_cvtpd_ps(b), (1 << 2) | (1 << 6)); -} + template<> EIGEN_STRONG_INLINE Packet4f pcast(const Packet2d &a, const Packet2d &b) + { + return _mm_shuffle_ps(_mm_cvtpd_ps(a), _mm_cvtpd_ps(b), (1 << 2) | (1 << 6)); + } -template<> EIGEN_STRONG_INLINE Packet2d pcast(const Packet4f& a) { - // Simply discard the second half of the input - return _mm_cvtps_pd(a); -} + template<> EIGEN_STRONG_INLINE Packet2d pcast(const Packet4f &a) + { + // Simply discard the second half of the input + return _mm_cvtps_pd(a); + } -} // end namespace internal +}// end namespace internal -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_TYPE_CASTING_SSE_H +#endif// EIGEN_TYPE_CASTING_SSE_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/arch/ZVector/Complex.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/arch/ZVector/Complex.h index 1bfb7339..6903dad7 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/arch/ZVector/Complex.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/arch/ZVector/Complex.h @@ -15,383 +15,488 @@ namespace Eigen { namespace internal { -static Packet2ul p2ul_CONJ_XOR1 = (Packet2ul) vec_sld((Packet4ui) p2d_ZERO_, (Packet4ui) p2l_ZERO, 8);//{ 0x8000000000000000, 0x0000000000000000 }; -static Packet2ul p2ul_CONJ_XOR2 = (Packet2ul) vec_sld((Packet4ui) p2l_ZERO, (Packet4ui) p2d_ZERO_, 8);//{ 0x8000000000000000, 0x0000000000000000 }; - -struct Packet1cd -{ - EIGEN_STRONG_INLINE Packet1cd() {} - EIGEN_STRONG_INLINE explicit Packet1cd(const Packet2d& a) : v(a) {} - Packet2d v; -}; - -struct Packet2cf -{ - EIGEN_STRONG_INLINE Packet2cf() {} - EIGEN_STRONG_INLINE explicit Packet2cf(const Packet4f& a) : v(a) {} - union { - Packet4f v; - Packet1cd cd[2]; + static Packet2ul p2ul_CONJ_XOR1 = + (Packet2ul)vec_sld((Packet4ui)p2d_ZERO_, (Packet4ui)p2l_ZERO, 8);//{ 0x8000000000000000, 0x0000000000000000 }; + static Packet2ul p2ul_CONJ_XOR2 = + (Packet2ul)vec_sld((Packet4ui)p2l_ZERO, (Packet4ui)p2d_ZERO_, 8);//{ 0x8000000000000000, 0x0000000000000000 }; + + struct Packet1cd + { + EIGEN_STRONG_INLINE Packet1cd() {} + EIGEN_STRONG_INLINE explicit Packet1cd(const Packet2d &a) : v(a) {} + Packet2d v; }; -}; - -template<> struct packet_traits > : default_packet_traits -{ - typedef Packet2cf type; - typedef Packet2cf half; - enum { - Vectorizable = 1, - AlignedOnScalar = 1, - size = 2, - HasHalfPacket = 0, - - HasAdd = 1, - HasSub = 1, - HasMul = 1, - HasDiv = 1, - HasNegate = 1, - HasAbs = 0, - HasAbs2 = 0, - HasMin = 0, - HasMax = 0, - HasBlend = 1, - HasSetLinear = 0 + + struct Packet2cf + { + EIGEN_STRONG_INLINE Packet2cf() {} + EIGEN_STRONG_INLINE explicit Packet2cf(const Packet4f &a) : v(a) {} + union { + Packet4f v; + Packet1cd cd[2]; + }; }; -}; - - -template<> struct packet_traits > : default_packet_traits -{ - typedef Packet1cd type; - typedef Packet1cd half; - enum { - Vectorizable = 1, - AlignedOnScalar = 1, - size = 1, - HasHalfPacket = 0, - - HasAdd = 1, - HasSub = 1, - HasMul = 1, - HasDiv = 1, - HasNegate = 1, - HasAbs = 0, - HasAbs2 = 0, - HasMin = 0, - HasMax = 0, - HasSetLinear = 0 + + template<> struct packet_traits> : default_packet_traits + { + typedef Packet2cf type; + typedef Packet2cf half; + enum { + Vectorizable = 1, + AlignedOnScalar = 1, + size = 2, + HasHalfPacket = 0, + + HasAdd = 1, + HasSub = 1, + HasMul = 1, + HasDiv = 1, + HasNegate = 1, + HasAbs = 0, + HasAbs2 = 0, + HasMin = 0, + HasMax = 0, + HasBlend = 1, + HasSetLinear = 0 + }; }; -}; - -template<> struct unpacket_traits { typedef std::complex type; enum {size=2, alignment=Aligned16}; typedef Packet2cf half; }; -template<> struct unpacket_traits { typedef std::complex type; enum {size=1, alignment=Aligned16}; typedef Packet1cd half; }; - -/* Forward declaration */ -EIGEN_STRONG_INLINE void ptranspose(PacketBlock& kernel); - -template<> EIGEN_STRONG_INLINE Packet2cf pload (const std::complex* from) { EIGEN_DEBUG_ALIGNED_LOAD return Packet2cf(pload((const float*)from)); } -template<> EIGEN_STRONG_INLINE Packet1cd pload (const std::complex* from) { EIGEN_DEBUG_ALIGNED_LOAD return Packet1cd(pload((const double*)from)); } -template<> EIGEN_STRONG_INLINE Packet2cf ploadu(const std::complex* from) { EIGEN_DEBUG_UNALIGNED_LOAD return Packet2cf(ploadu((const float*)from)); } -template<> EIGEN_STRONG_INLINE Packet1cd ploadu(const std::complex* from) { EIGEN_DEBUG_UNALIGNED_LOAD return Packet1cd(ploadu((const double*)from)); } -template<> EIGEN_STRONG_INLINE void pstore >(std::complex * to, const Packet2cf& from) { EIGEN_DEBUG_ALIGNED_STORE pstore((float*)to, from.v); } -template<> EIGEN_STRONG_INLINE void pstore >(std::complex * to, const Packet1cd& from) { EIGEN_DEBUG_ALIGNED_STORE pstore((double*)to, from.v); } -template<> EIGEN_STRONG_INLINE void pstoreu >(std::complex * to, const Packet2cf& from) { EIGEN_DEBUG_UNALIGNED_STORE pstoreu((float*)to, from.v); } -template<> EIGEN_STRONG_INLINE void pstoreu >(std::complex * to, const Packet1cd& from) { EIGEN_DEBUG_UNALIGNED_STORE pstoreu((double*)to, from.v); } - -template<> EIGEN_STRONG_INLINE Packet1cd pset1(const std::complex& from) -{ /* here we really have to use unaligned loads :( */ return ploadu(&from); } - -template<> EIGEN_STRONG_INLINE Packet2cf pset1(const std::complex& from) -{ - Packet2cf res; - res.cd[0] = Packet1cd(vec_ld2f((const float *)&from)); - res.cd[1] = res.cd[0]; - return res; -} -template<> EIGEN_DEVICE_FUNC inline Packet2cf pgather, Packet2cf>(const std::complex* from, Index stride) -{ - std::complex EIGEN_ALIGN16 af[2]; - af[0] = from[0*stride]; - af[1] = from[1*stride]; - return pload(af); -} -template<> EIGEN_DEVICE_FUNC inline Packet1cd pgather, Packet1cd>(const std::complex* from, Index stride EIGEN_UNUSED) -{ - return pload(from); -} -template<> EIGEN_DEVICE_FUNC inline void pscatter, Packet2cf>(std::complex* to, const Packet2cf& from, Index stride) -{ - std::complex EIGEN_ALIGN16 af[2]; - pstore >((std::complex *) af, from); - to[0*stride] = af[0]; - to[1*stride] = af[1]; -} -template<> EIGEN_DEVICE_FUNC inline void pscatter, Packet1cd>(std::complex* to, const Packet1cd& from, Index stride EIGEN_UNUSED) -{ - pstore >(to, from); -} - -template<> EIGEN_STRONG_INLINE Packet2cf padd(const Packet2cf& a, const Packet2cf& b) { return Packet2cf(padd(a.v, b.v)); } -template<> EIGEN_STRONG_INLINE Packet1cd padd(const Packet1cd& a, const Packet1cd& b) { return Packet1cd(a.v + b.v); } -template<> EIGEN_STRONG_INLINE Packet2cf psub(const Packet2cf& a, const Packet2cf& b) { return Packet2cf(psub(a.v, b.v)); } -template<> EIGEN_STRONG_INLINE Packet1cd psub(const Packet1cd& a, const Packet1cd& b) { return Packet1cd(a.v - b.v); } -template<> EIGEN_STRONG_INLINE Packet1cd pnegate(const Packet1cd& a) { return Packet1cd(pnegate(Packet2d(a.v))); } -template<> EIGEN_STRONG_INLINE Packet2cf pnegate(const Packet2cf& a) { return Packet2cf(pnegate(Packet4f(a.v))); } -template<> EIGEN_STRONG_INLINE Packet1cd pconj(const Packet1cd& a) { return Packet1cd((Packet2d)vec_xor((Packet2d)a.v, (Packet2d)p2ul_CONJ_XOR2)); } -template<> EIGEN_STRONG_INLINE Packet2cf pconj(const Packet2cf& a) -{ - Packet2cf res; - res.v.v4f[0] = pconj(Packet1cd(reinterpret_cast(a.v.v4f[0]))).v; - res.v.v4f[1] = pconj(Packet1cd(reinterpret_cast(a.v.v4f[1]))).v; - return res; -} - -template<> EIGEN_STRONG_INLINE Packet1cd pmul(const Packet1cd& a, const Packet1cd& b) -{ - Packet2d a_re, a_im, v1, v2; - - // Permute and multiply the real parts of a and b - a_re = vec_perm(a.v, a.v, p16uc_PSET64_HI); - // Get the imaginary parts of a - a_im = vec_perm(a.v, a.v, p16uc_PSET64_LO); - // multiply a_re * b - v1 = vec_madd(a_re, b.v, p2d_ZERO); - // multiply a_im * b and get the conjugate result - v2 = vec_madd(a_im, b.v, p2d_ZERO); - v2 = (Packet2d) vec_sld((Packet4ui)v2, (Packet4ui)v2, 8); - v2 = (Packet2d) vec_xor((Packet2d)v2, (Packet2d) p2ul_CONJ_XOR1); - - return Packet1cd(v1 + v2); -} -template<> EIGEN_STRONG_INLINE Packet2cf pmul(const Packet2cf& a, const Packet2cf& b) -{ - Packet2cf res; - res.v.v4f[0] = pmul(Packet1cd(reinterpret_cast(a.v.v4f[0])), Packet1cd(reinterpret_cast(b.v.v4f[0]))).v; - res.v.v4f[1] = pmul(Packet1cd(reinterpret_cast(a.v.v4f[1])), Packet1cd(reinterpret_cast(b.v.v4f[1]))).v; - return res; -} - -template<> EIGEN_STRONG_INLINE Packet1cd pand (const Packet1cd& a, const Packet1cd& b) { return Packet1cd(vec_and(a.v,b.v)); } -template<> EIGEN_STRONG_INLINE Packet2cf pand (const Packet2cf& a, const Packet2cf& b) { return Packet2cf(pand(a.v,b.v)); } -template<> EIGEN_STRONG_INLINE Packet1cd por (const Packet1cd& a, const Packet1cd& b) { return Packet1cd(vec_or(a.v,b.v)); } -template<> EIGEN_STRONG_INLINE Packet2cf por (const Packet2cf& a, const Packet2cf& b) { return Packet2cf(por(a.v,b.v)); } -template<> EIGEN_STRONG_INLINE Packet1cd pxor (const Packet1cd& a, const Packet1cd& b) { return Packet1cd(vec_xor(a.v,b.v)); } -template<> EIGEN_STRONG_INLINE Packet2cf pxor (const Packet2cf& a, const Packet2cf& b) { return Packet2cf(pxor(a.v,b.v)); } -template<> EIGEN_STRONG_INLINE Packet1cd pandnot(const Packet1cd& a, const Packet1cd& b) { return Packet1cd(vec_and(a.v, vec_nor(b.v,b.v))); } -template<> EIGEN_STRONG_INLINE Packet2cf pandnot(const Packet2cf& a, const Packet2cf& b) { return Packet2cf(pandnot(a.v,b.v)); } - -template<> EIGEN_STRONG_INLINE Packet1cd ploaddup(const std::complex* from) { return pset1(*from); } -template<> EIGEN_STRONG_INLINE Packet2cf ploaddup(const std::complex* from) { return pset1(*from); } - -template<> EIGEN_STRONG_INLINE void prefetch >(const std::complex * addr) { EIGEN_ZVECTOR_PREFETCH(addr); } -template<> EIGEN_STRONG_INLINE void prefetch >(const std::complex * addr) { EIGEN_ZVECTOR_PREFETCH(addr); } - -template<> EIGEN_STRONG_INLINE std::complex pfirst(const Packet1cd& a) -{ - std::complex EIGEN_ALIGN16 res; - pstore >(&res, a); - - return res; -} -template<> EIGEN_STRONG_INLINE std::complex pfirst(const Packet2cf& a) -{ - std::complex EIGEN_ALIGN16 res[2]; - pstore >(res, a); - - return res[0]; -} - -template<> EIGEN_STRONG_INLINE Packet1cd preverse(const Packet1cd& a) { return a; } -template<> EIGEN_STRONG_INLINE Packet2cf preverse(const Packet2cf& a) -{ - Packet2cf res; - res.cd[0] = a.cd[1]; - res.cd[1] = a.cd[0]; - return res; -} - -template<> EIGEN_STRONG_INLINE std::complex predux(const Packet1cd& a) -{ - return pfirst(a); -} -template<> EIGEN_STRONG_INLINE std::complex predux(const Packet2cf& a) -{ - std::complex res; - Packet1cd b = padd(a.cd[0], a.cd[1]); - vec_st2f(b.v, (float*)&res); - return res; -} - -template<> EIGEN_STRONG_INLINE Packet1cd preduxp(const Packet1cd* vecs) -{ - return vecs[0]; -} -template<> EIGEN_STRONG_INLINE Packet2cf preduxp(const Packet2cf* vecs) -{ - PacketBlock transpose; - transpose.packet[0] = vecs[0]; - transpose.packet[1] = vecs[1]; - ptranspose(transpose); - - return padd(transpose.packet[0], transpose.packet[1]); -} - -template<> EIGEN_STRONG_INLINE std::complex predux_mul(const Packet1cd& a) -{ - return pfirst(a); -} -template<> EIGEN_STRONG_INLINE std::complex predux_mul(const Packet2cf& a) -{ - std::complex res; - Packet1cd b = pmul(a.cd[0], a.cd[1]); - vec_st2f(b.v, (float*)&res); - return res; -} - -template -struct palign_impl -{ - static EIGEN_STRONG_INLINE void run(Packet1cd& /*first*/, const Packet1cd& /*second*/) - { - // FIXME is it sure we never have to align a Packet1cd? - // Even though a std::complex has 16 bytes, it is not necessarily aligned on a 16 bytes boundary... - } -}; - -template -struct palign_impl -{ - static EIGEN_STRONG_INLINE void run(Packet2cf& first, const Packet2cf& second) - { - if (Offset == 1) { - first.cd[0] = first.cd[1]; - first.cd[1] = second.cd[0]; - } + + + template<> struct packet_traits> : default_packet_traits + { + typedef Packet1cd type; + typedef Packet1cd half; + enum { + Vectorizable = 1, + AlignedOnScalar = 1, + size = 1, + HasHalfPacket = 0, + + HasAdd = 1, + HasSub = 1, + HasMul = 1, + HasDiv = 1, + HasNegate = 1, + HasAbs = 0, + HasAbs2 = 0, + HasMin = 0, + HasMax = 0, + HasSetLinear = 0 + }; + }; + + template<> struct unpacket_traits + { + typedef std::complex type; + enum { size = 2, alignment = Aligned16 }; + typedef Packet2cf half; + }; + template<> struct unpacket_traits + { + typedef std::complex type; + enum { size = 1, alignment = Aligned16 }; + typedef Packet1cd half; + }; + + /* Forward declaration */ + EIGEN_STRONG_INLINE void ptranspose(PacketBlock &kernel); + + template<> EIGEN_STRONG_INLINE Packet2cf pload(const std::complex *from) + { + EIGEN_DEBUG_ALIGNED_LOAD return Packet2cf(pload((const float *)from)); + } + template<> EIGEN_STRONG_INLINE Packet1cd pload(const std::complex *from) + { + EIGEN_DEBUG_ALIGNED_LOAD return Packet1cd(pload((const double *)from)); + } + template<> EIGEN_STRONG_INLINE Packet2cf ploadu(const std::complex *from) + { + EIGEN_DEBUG_UNALIGNED_LOAD return Packet2cf(ploadu((const float *)from)); + } + template<> EIGEN_STRONG_INLINE Packet1cd ploadu(const std::complex *from) + { + EIGEN_DEBUG_UNALIGNED_LOAD return Packet1cd(ploadu((const double *)from)); + } + template<> EIGEN_STRONG_INLINE void pstore>(std::complex *to, const Packet2cf &from) + { + EIGEN_DEBUG_ALIGNED_STORE pstore((float *)to, from.v); + } + template<> EIGEN_STRONG_INLINE void pstore>(std::complex *to, const Packet1cd &from) + { + EIGEN_DEBUG_ALIGNED_STORE pstore((double *)to, from.v); + } + template<> EIGEN_STRONG_INLINE void pstoreu>(std::complex *to, const Packet2cf &from) + { + EIGEN_DEBUG_UNALIGNED_STORE pstoreu((float *)to, from.v); + } + template<> EIGEN_STRONG_INLINE void pstoreu>(std::complex *to, const Packet1cd &from) + { + EIGEN_DEBUG_UNALIGNED_STORE pstoreu((double *)to, from.v); } -}; -template<> struct conj_helper -{ - EIGEN_STRONG_INLINE Packet1cd pmadd(const Packet1cd& x, const Packet1cd& y, const Packet1cd& c) const - { return padd(pmul(x,y),c); } + template<> EIGEN_STRONG_INLINE Packet1cd pset1(const std::complex &from) + { /* here we really have to use unaligned loads :( */ + return ploadu(&from); + } - EIGEN_STRONG_INLINE Packet1cd pmul(const Packet1cd& a, const Packet1cd& b) const + template<> EIGEN_STRONG_INLINE Packet2cf pset1(const std::complex &from) + { + Packet2cf res; + res.cd[0] = Packet1cd(vec_ld2f((const float *)&from)); + res.cd[1] = res.cd[0]; + return res; + } + template<> + EIGEN_DEVICE_FUNC inline Packet2cf pgather, Packet2cf>(const std::complex *from, + Index stride) { - return internal::pmul(a, pconj(b)); + std::complex EIGEN_ALIGN16 af[2]; + af[0] = from[0 * stride]; + af[1] = from[1 * stride]; + return pload(af); + } + template<> + EIGEN_DEVICE_FUNC inline Packet1cd pgather, Packet1cd>(const std::complex *from, + Index stride EIGEN_UNUSED) + { + return pload(from); + } + template<> + EIGEN_DEVICE_FUNC inline void + pscatter, Packet2cf>(std::complex *to, const Packet2cf &from, Index stride) + { + std::complex EIGEN_ALIGN16 af[2]; + pstore>((std::complex *)af, from); + to[0 * stride] = af[0]; + to[1 * stride] = af[1]; + } + template<> + EIGEN_DEVICE_FUNC inline void pscatter, Packet1cd>(std::complex *to, + const Packet1cd &from, + Index stride EIGEN_UNUSED) + { + pstore>(to, from); } -}; -template<> struct conj_helper -{ - EIGEN_STRONG_INLINE Packet1cd pmadd(const Packet1cd& x, const Packet1cd& y, const Packet1cd& c) const - { return padd(pmul(x,y),c); } + template<> EIGEN_STRONG_INLINE Packet2cf padd(const Packet2cf &a, const Packet2cf &b) + { + return Packet2cf(padd(a.v, b.v)); + } + template<> EIGEN_STRONG_INLINE Packet1cd padd(const Packet1cd &a, const Packet1cd &b) + { + return Packet1cd(a.v + b.v); + } + template<> EIGEN_STRONG_INLINE Packet2cf psub(const Packet2cf &a, const Packet2cf &b) + { + return Packet2cf(psub(a.v, b.v)); + } + template<> EIGEN_STRONG_INLINE Packet1cd psub(const Packet1cd &a, const Packet1cd &b) + { + return Packet1cd(a.v - b.v); + } + template<> EIGEN_STRONG_INLINE Packet1cd pnegate(const Packet1cd &a) { return Packet1cd(pnegate(Packet2d(a.v))); } + template<> EIGEN_STRONG_INLINE Packet2cf pnegate(const Packet2cf &a) { return Packet2cf(pnegate(Packet4f(a.v))); } + template<> EIGEN_STRONG_INLINE Packet1cd pconj(const Packet1cd &a) + { + return Packet1cd((Packet2d)vec_xor((Packet2d)a.v, (Packet2d)p2ul_CONJ_XOR2)); + } + template<> EIGEN_STRONG_INLINE Packet2cf pconj(const Packet2cf &a) + { + Packet2cf res; + res.v.v4f[0] = pconj(Packet1cd(reinterpret_cast(a.v.v4f[0]))).v; + res.v.v4f[1] = pconj(Packet1cd(reinterpret_cast(a.v.v4f[1]))).v; + return res; + } - EIGEN_STRONG_INLINE Packet1cd pmul(const Packet1cd& a, const Packet1cd& b) const + template<> EIGEN_STRONG_INLINE Packet1cd pmul(const Packet1cd &a, const Packet1cd &b) + { + Packet2d a_re, a_im, v1, v2; + + // Permute and multiply the real parts of a and b + a_re = vec_perm(a.v, a.v, p16uc_PSET64_HI); + // Get the imaginary parts of a + a_im = vec_perm(a.v, a.v, p16uc_PSET64_LO); + // multiply a_re * b + v1 = vec_madd(a_re, b.v, p2d_ZERO); + // multiply a_im * b and get the conjugate result + v2 = vec_madd(a_im, b.v, p2d_ZERO); + v2 = (Packet2d)vec_sld((Packet4ui)v2, (Packet4ui)v2, 8); + v2 = (Packet2d)vec_xor((Packet2d)v2, (Packet2d)p2ul_CONJ_XOR1); + + return Packet1cd(v1 + v2); + } + template<> EIGEN_STRONG_INLINE Packet2cf pmul(const Packet2cf &a, const Packet2cf &b) { - return internal::pmul(pconj(a), b); + Packet2cf res; + res.v.v4f[0] = + pmul(Packet1cd(reinterpret_cast(a.v.v4f[0])), Packet1cd(reinterpret_cast(b.v.v4f[0]))).v; + res.v.v4f[1] = + pmul(Packet1cd(reinterpret_cast(a.v.v4f[1])), Packet1cd(reinterpret_cast(b.v.v4f[1]))).v; + return res; } -}; -template<> struct conj_helper -{ - EIGEN_STRONG_INLINE Packet1cd pmadd(const Packet1cd& x, const Packet1cd& y, const Packet1cd& c) const - { return padd(pmul(x,y),c); } + template<> EIGEN_STRONG_INLINE Packet1cd pand(const Packet1cd &a, const Packet1cd &b) + { + return Packet1cd(vec_and(a.v, b.v)); + } + template<> EIGEN_STRONG_INLINE Packet2cf pand(const Packet2cf &a, const Packet2cf &b) + { + return Packet2cf(pand(a.v, b.v)); + } + template<> EIGEN_STRONG_INLINE Packet1cd por(const Packet1cd &a, const Packet1cd &b) + { + return Packet1cd(vec_or(a.v, b.v)); + } + template<> EIGEN_STRONG_INLINE Packet2cf por(const Packet2cf &a, const Packet2cf &b) + { + return Packet2cf(por(a.v, b.v)); + } + template<> EIGEN_STRONG_INLINE Packet1cd pxor(const Packet1cd &a, const Packet1cd &b) + { + return Packet1cd(vec_xor(a.v, b.v)); + } + template<> EIGEN_STRONG_INLINE Packet2cf pxor(const Packet2cf &a, const Packet2cf &b) + { + return Packet2cf(pxor(a.v, b.v)); + } + template<> EIGEN_STRONG_INLINE Packet1cd pandnot(const Packet1cd &a, const Packet1cd &b) + { + return Packet1cd(vec_and(a.v, vec_nor(b.v, b.v))); + } + template<> EIGEN_STRONG_INLINE Packet2cf pandnot(const Packet2cf &a, const Packet2cf &b) + { + return Packet2cf(pandnot(a.v, b.v)); + } - EIGEN_STRONG_INLINE Packet1cd pmul(const Packet1cd& a, const Packet1cd& b) const + template<> EIGEN_STRONG_INLINE Packet1cd ploaddup(const std::complex *from) { - return pconj(internal::pmul(a, b)); + return pset1(*from); + } + template<> EIGEN_STRONG_INLINE Packet2cf ploaddup(const std::complex *from) + { + return pset1(*from); } -}; -template<> struct conj_helper -{ - EIGEN_STRONG_INLINE Packet2cf pmadd(const Packet2cf& x, const Packet2cf& y, const Packet2cf& c) const - { return padd(pmul(x,y),c); } + template<> EIGEN_STRONG_INLINE void prefetch>(const std::complex *addr) + { + EIGEN_ZVECTOR_PREFETCH(addr); + } + template<> EIGEN_STRONG_INLINE void prefetch>(const std::complex *addr) + { + EIGEN_ZVECTOR_PREFETCH(addr); + } - EIGEN_STRONG_INLINE Packet2cf pmul(const Packet2cf& a, const Packet2cf& b) const + template<> EIGEN_STRONG_INLINE std::complex pfirst(const Packet1cd &a) { - return internal::pmul(a, pconj(b)); + std::complex EIGEN_ALIGN16 res; + pstore>(&res, a); + + return res; + } + template<> EIGEN_STRONG_INLINE std::complex pfirst(const Packet2cf &a) + { + std::complex EIGEN_ALIGN16 res[2]; + pstore>(res, a); + + return res[0]; } -}; -template<> struct conj_helper -{ - EIGEN_STRONG_INLINE Packet2cf pmadd(const Packet2cf& x, const Packet2cf& y, const Packet2cf& c) const - { return padd(pmul(x,y),c); } + template<> EIGEN_STRONG_INLINE Packet1cd preverse(const Packet1cd &a) { return a; } + template<> EIGEN_STRONG_INLINE Packet2cf preverse(const Packet2cf &a) + { + Packet2cf res; + res.cd[0] = a.cd[1]; + res.cd[1] = a.cd[0]; + return res; + } - EIGEN_STRONG_INLINE Packet2cf pmul(const Packet2cf& a, const Packet2cf& b) const + template<> EIGEN_STRONG_INLINE std::complex predux(const Packet1cd &a) { return pfirst(a); } + template<> EIGEN_STRONG_INLINE std::complex predux(const Packet2cf &a) { - return internal::pmul(pconj(a), b); + std::complex res; + Packet1cd b = padd(a.cd[0], a.cd[1]); + vec_st2f(b.v, (float *)&res); + return res; } -}; -template<> struct conj_helper -{ - EIGEN_STRONG_INLINE Packet2cf pmadd(const Packet2cf& x, const Packet2cf& y, const Packet2cf& c) const - { return padd(pmul(x,y),c); } + template<> EIGEN_STRONG_INLINE Packet1cd preduxp(const Packet1cd *vecs) { return vecs[0]; } + template<> EIGEN_STRONG_INLINE Packet2cf preduxp(const Packet2cf *vecs) + { + PacketBlock transpose; + transpose.packet[0] = vecs[0]; + transpose.packet[1] = vecs[1]; + ptranspose(transpose); + + return padd(transpose.packet[0], transpose.packet[1]); + } - EIGEN_STRONG_INLINE Packet2cf pmul(const Packet2cf& a, const Packet2cf& b) const + template<> EIGEN_STRONG_INLINE std::complex predux_mul(const Packet1cd &a) { return pfirst(a); } + template<> EIGEN_STRONG_INLINE std::complex predux_mul(const Packet2cf &a) { - return pconj(internal::pmul(a, b)); + std::complex res; + Packet1cd b = pmul(a.cd[0], a.cd[1]); + vec_st2f(b.v, (float *)&res); + return res; } -}; -EIGEN_MAKE_CONJ_HELPER_CPLX_REAL(Packet2cf,Packet4f) -EIGEN_MAKE_CONJ_HELPER_CPLX_REAL(Packet1cd,Packet2d) + template struct palign_impl + { + static EIGEN_STRONG_INLINE void run(Packet1cd & /*first*/, const Packet1cd & /*second*/) + { + // FIXME is it sure we never have to align a Packet1cd? + // Even though a std::complex has 16 bytes, it is not necessarily aligned on a 16 bytes boundary... + } + }; + + template struct palign_impl + { + static EIGEN_STRONG_INLINE void run(Packet2cf &first, const Packet2cf &second) + { + if (Offset == 1) { + first.cd[0] = first.cd[1]; + first.cd[1] = second.cd[0]; + } + } + }; + + template<> struct conj_helper + { + EIGEN_STRONG_INLINE Packet1cd pmadd(const Packet1cd &x, const Packet1cd &y, const Packet1cd &c) const + { + return padd(pmul(x, y), c); + } + + EIGEN_STRONG_INLINE Packet1cd pmul(const Packet1cd &a, const Packet1cd &b) const + { + return internal::pmul(a, pconj(b)); + } + }; + + template<> struct conj_helper + { + EIGEN_STRONG_INLINE Packet1cd pmadd(const Packet1cd &x, const Packet1cd &y, const Packet1cd &c) const + { + return padd(pmul(x, y), c); + } + + EIGEN_STRONG_INLINE Packet1cd pmul(const Packet1cd &a, const Packet1cd &b) const + { + return internal::pmul(pconj(a), b); + } + }; + + template<> struct conj_helper + { + EIGEN_STRONG_INLINE Packet1cd pmadd(const Packet1cd &x, const Packet1cd &y, const Packet1cd &c) const + { + return padd(pmul(x, y), c); + } + + EIGEN_STRONG_INLINE Packet1cd pmul(const Packet1cd &a, const Packet1cd &b) const + { + return pconj(internal::pmul(a, b)); + } + }; + + template<> struct conj_helper + { + EIGEN_STRONG_INLINE Packet2cf pmadd(const Packet2cf &x, const Packet2cf &y, const Packet2cf &c) const + { + return padd(pmul(x, y), c); + } + + EIGEN_STRONG_INLINE Packet2cf pmul(const Packet2cf &a, const Packet2cf &b) const + { + return internal::pmul(a, pconj(b)); + } + }; + + template<> struct conj_helper + { + EIGEN_STRONG_INLINE Packet2cf pmadd(const Packet2cf &x, const Packet2cf &y, const Packet2cf &c) const + { + return padd(pmul(x, y), c); + } + + EIGEN_STRONG_INLINE Packet2cf pmul(const Packet2cf &a, const Packet2cf &b) const + { + return internal::pmul(pconj(a), b); + } + }; + + template<> struct conj_helper + { + EIGEN_STRONG_INLINE Packet2cf pmadd(const Packet2cf &x, const Packet2cf &y, const Packet2cf &c) const + { + return padd(pmul(x, y), c); + } + + EIGEN_STRONG_INLINE Packet2cf pmul(const Packet2cf &a, const Packet2cf &b) const + { + return pconj(internal::pmul(a, b)); + } + }; -template<> EIGEN_STRONG_INLINE Packet1cd pdiv(const Packet1cd& a, const Packet1cd& b) -{ - // TODO optimize it for AltiVec - Packet1cd res = conj_helper().pmul(a,b); - Packet2d s = vec_madd(b.v, b.v, p2d_ZERO_); - return Packet1cd(pdiv(res.v, s + vec_perm(s, s, p16uc_REVERSE64))); -} + EIGEN_MAKE_CONJ_HELPER_CPLX_REAL(Packet2cf, Packet4f) + EIGEN_MAKE_CONJ_HELPER_CPLX_REAL(Packet1cd, Packet2d) -template<> EIGEN_STRONG_INLINE Packet2cf pdiv(const Packet2cf& a, const Packet2cf& b) -{ - // TODO optimize it for AltiVec - Packet2cf res; - res.cd[0] = pdiv(a.cd[0], b.cd[0]); - res.cd[1] = pdiv(a.cd[1], b.cd[1]); - return res; -} + template<> EIGEN_STRONG_INLINE Packet1cd pdiv(const Packet1cd &a, const Packet1cd &b) + { + // TODO optimize it for AltiVec + Packet1cd res = conj_helper().pmul(a, b); + Packet2d s = vec_madd(b.v, b.v, p2d_ZERO_); + return Packet1cd(pdiv(res.v, s + vec_perm(s, s, p16uc_REVERSE64))); + } -EIGEN_STRONG_INLINE Packet1cd pcplxflip/**/(const Packet1cd& x) -{ - return Packet1cd(preverse(Packet2d(x.v))); -} + template<> EIGEN_STRONG_INLINE Packet2cf pdiv(const Packet2cf &a, const Packet2cf &b) + { + // TODO optimize it for AltiVec + Packet2cf res; + res.cd[0] = pdiv(a.cd[0], b.cd[0]); + res.cd[1] = pdiv(a.cd[1], b.cd[1]); + return res; + } + + EIGEN_STRONG_INLINE Packet1cd pcplxflip /**/ (const Packet1cd &x) + { + return Packet1cd(preverse(Packet2d(x.v))); + } -EIGEN_STRONG_INLINE Packet2cf pcplxflip/**/(const Packet2cf& x) -{ - Packet2cf res; - res.cd[0] = pcplxflip(x.cd[0]); - res.cd[1] = pcplxflip(x.cd[1]); - return res; -} + EIGEN_STRONG_INLINE Packet2cf pcplxflip /**/ (const Packet2cf &x) + { + Packet2cf res; + res.cd[0] = pcplxflip(x.cd[0]); + res.cd[1] = pcplxflip(x.cd[1]); + return res; + } -EIGEN_STRONG_INLINE void ptranspose(PacketBlock& kernel) -{ - Packet2d tmp = vec_perm(kernel.packet[0].v, kernel.packet[1].v, p16uc_TRANSPOSE64_HI); - kernel.packet[1].v = vec_perm(kernel.packet[0].v, kernel.packet[1].v, p16uc_TRANSPOSE64_LO); - kernel.packet[0].v = tmp; -} + EIGEN_STRONG_INLINE void ptranspose(PacketBlock &kernel) + { + Packet2d tmp = vec_perm(kernel.packet[0].v, kernel.packet[1].v, p16uc_TRANSPOSE64_HI); + kernel.packet[1].v = vec_perm(kernel.packet[0].v, kernel.packet[1].v, p16uc_TRANSPOSE64_LO); + kernel.packet[0].v = tmp; + } -EIGEN_STRONG_INLINE void ptranspose(PacketBlock& kernel) -{ - Packet1cd tmp = kernel.packet[0].cd[1]; - kernel.packet[0].cd[1] = kernel.packet[1].cd[0]; - kernel.packet[1].cd[0] = tmp; -} + EIGEN_STRONG_INLINE void ptranspose(PacketBlock &kernel) + { + Packet1cd tmp = kernel.packet[0].cd[1]; + kernel.packet[0].cd[1] = kernel.packet[1].cd[0]; + kernel.packet[1].cd[0] = tmp; + } -template<> EIGEN_STRONG_INLINE Packet2cf pblend(const Selector<2>& ifPacket, const Packet2cf& thenPacket, const Packet2cf& elsePacket) { - Packet2cf result; - const Selector<4> ifPacket4 = { ifPacket.select[0], ifPacket.select[0], ifPacket.select[1], ifPacket.select[1] }; - result.v = pblend(ifPacket4, thenPacket.v, elsePacket.v); - return result; -} + template<> + EIGEN_STRONG_INLINE Packet2cf pblend(const Selector<2> &ifPacket, + const Packet2cf &thenPacket, + const Packet2cf &elsePacket) + { + Packet2cf result; + const Selector<4> ifPacket4 = { ifPacket.select[0], ifPacket.select[0], ifPacket.select[1], ifPacket.select[1] }; + result.v = pblend(ifPacket4, thenPacket.v, elsePacket.v); + return result; + } -} // end namespace internal +}// end namespace internal -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_COMPLEX32_ALTIVEC_H +#endif// EIGEN_COMPLEX32_ALTIVEC_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/arch/ZVector/MathFunctions.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/arch/ZVector/MathFunctions.h index 5c7aa725..c5d6af5b 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/arch/ZVector/MathFunctions.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/arch/ZVector/MathFunctions.h @@ -20,118 +20,118 @@ namespace Eigen { namespace internal { -static _EIGEN_DECLARE_CONST_Packet2d(1 , 1.0); -static _EIGEN_DECLARE_CONST_Packet2d(2 , 2.0); -static _EIGEN_DECLARE_CONST_Packet2d(half, 0.5); - -static _EIGEN_DECLARE_CONST_Packet2d(exp_hi, 709.437); -static _EIGEN_DECLARE_CONST_Packet2d(exp_lo, -709.436139303); - -static _EIGEN_DECLARE_CONST_Packet2d(cephes_LOG2EF, 1.4426950408889634073599); - -static _EIGEN_DECLARE_CONST_Packet2d(cephes_exp_p0, 1.26177193074810590878e-4); -static _EIGEN_DECLARE_CONST_Packet2d(cephes_exp_p1, 3.02994407707441961300e-2); -static _EIGEN_DECLARE_CONST_Packet2d(cephes_exp_p2, 9.99999999999999999910e-1); - -static _EIGEN_DECLARE_CONST_Packet2d(cephes_exp_q0, 3.00198505138664455042e-6); -static _EIGEN_DECLARE_CONST_Packet2d(cephes_exp_q1, 2.52448340349684104192e-3); -static _EIGEN_DECLARE_CONST_Packet2d(cephes_exp_q2, 2.27265548208155028766e-1); -static _EIGEN_DECLARE_CONST_Packet2d(cephes_exp_q3, 2.00000000000000000009e0); - -static _EIGEN_DECLARE_CONST_Packet2d(cephes_exp_C1, 0.693145751953125); -static _EIGEN_DECLARE_CONST_Packet2d(cephes_exp_C2, 1.42860682030941723212e-6); - -template<> EIGEN_DEFINE_FUNCTION_ALLOWING_MULTIPLE_DEFINITIONS EIGEN_UNUSED -Packet2d pexp(const Packet2d& _x) -{ - Packet2d x = _x; - - Packet2d tmp, fx; - Packet2l emm0; - - // clamp x - x = pmax(pmin(x, p2d_exp_hi), p2d_exp_lo); - /* express exp(x) as exp(g + n*log(2)) */ - fx = pmadd(p2d_cephes_LOG2EF, x, p2d_half); - - fx = vec_floor(fx); - - tmp = pmul(fx, p2d_cephes_exp_C1); - Packet2d z = pmul(fx, p2d_cephes_exp_C2); - x = psub(x, tmp); - x = psub(x, z); - - Packet2d x2 = pmul(x,x); - - Packet2d px = p2d_cephes_exp_p0; - px = pmadd(px, x2, p2d_cephes_exp_p1); - px = pmadd(px, x2, p2d_cephes_exp_p2); - px = pmul (px, x); - - Packet2d qx = p2d_cephes_exp_q0; - qx = pmadd(qx, x2, p2d_cephes_exp_q1); - qx = pmadd(qx, x2, p2d_cephes_exp_q2); - qx = pmadd(qx, x2, p2d_cephes_exp_q3); - - x = pdiv(px,psub(qx,px)); - x = pmadd(p2d_2,x,p2d_1); - - // build 2^n - emm0 = vec_ctsl(fx, 0); - - static const Packet2l p2l_1023 = { 1023, 1023 }; - static const Packet2ul p2ul_52 = { 52, 52 }; - - emm0 = emm0 + p2l_1023; - emm0 = emm0 << reinterpret_cast(p2ul_52); - - // Altivec's max & min operators just drop silent NaNs. Check NaNs in - // inputs and return them unmodified. - Packet2ul isnumber_mask = reinterpret_cast(vec_cmpeq(_x, _x)); - return vec_sel(_x, pmax(pmul(x, reinterpret_cast(emm0)), _x), - isnumber_mask); -} - -template<> EIGEN_DEFINE_FUNCTION_ALLOWING_MULTIPLE_DEFINITIONS EIGEN_UNUSED -Packet4f pexp(const Packet4f& x) -{ - Packet4f res; - res.v4f[0] = pexp(x.v4f[0]); - res.v4f[1] = pexp(x.v4f[1]); - return res; -} - -template<> EIGEN_DEFINE_FUNCTION_ALLOWING_MULTIPLE_DEFINITIONS EIGEN_UNUSED -Packet2d psqrt(const Packet2d& x) -{ - return __builtin_s390_vfsqdb(x); -} - -template<> EIGEN_DEFINE_FUNCTION_ALLOWING_MULTIPLE_DEFINITIONS EIGEN_UNUSED -Packet4f psqrt(const Packet4f& x) -{ - Packet4f res; - res.v4f[0] = psqrt(x.v4f[0]); - res.v4f[1] = psqrt(x.v4f[1]); - return res; -} - -template<> EIGEN_DEFINE_FUNCTION_ALLOWING_MULTIPLE_DEFINITIONS EIGEN_UNUSED -Packet2d prsqrt(const Packet2d& x) { - // Unfortunately we can't use the much faster mm_rqsrt_pd since it only provides an approximation. - return pset1(1.0) / psqrt(x); -} - -template<> EIGEN_DEFINE_FUNCTION_ALLOWING_MULTIPLE_DEFINITIONS EIGEN_UNUSED -Packet4f prsqrt(const Packet4f& x) { - Packet4f res; - res.v4f[0] = prsqrt(x.v4f[0]); - res.v4f[1] = prsqrt(x.v4f[1]); - return res; -} - -} // end namespace internal - -} // end namespace Eigen - -#endif // EIGEN_MATH_FUNCTIONS_ALTIVEC_H + static _EIGEN_DECLARE_CONST_Packet2d(1, 1.0); + static _EIGEN_DECLARE_CONST_Packet2d(2, 2.0); + static _EIGEN_DECLARE_CONST_Packet2d(half, 0.5); + + static _EIGEN_DECLARE_CONST_Packet2d(exp_hi, 709.437); + static _EIGEN_DECLARE_CONST_Packet2d(exp_lo, -709.436139303); + + static _EIGEN_DECLARE_CONST_Packet2d(cephes_LOG2EF, 1.4426950408889634073599); + + static _EIGEN_DECLARE_CONST_Packet2d(cephes_exp_p0, 1.26177193074810590878e-4); + static _EIGEN_DECLARE_CONST_Packet2d(cephes_exp_p1, 3.02994407707441961300e-2); + static _EIGEN_DECLARE_CONST_Packet2d(cephes_exp_p2, 9.99999999999999999910e-1); + + static _EIGEN_DECLARE_CONST_Packet2d(cephes_exp_q0, 3.00198505138664455042e-6); + static _EIGEN_DECLARE_CONST_Packet2d(cephes_exp_q1, 2.52448340349684104192e-3); + static _EIGEN_DECLARE_CONST_Packet2d(cephes_exp_q2, 2.27265548208155028766e-1); + static _EIGEN_DECLARE_CONST_Packet2d(cephes_exp_q3, 2.00000000000000000009e0); + + static _EIGEN_DECLARE_CONST_Packet2d(cephes_exp_C1, 0.693145751953125); + static _EIGEN_DECLARE_CONST_Packet2d(cephes_exp_C2, 1.42860682030941723212e-6); + + template<> + EIGEN_DEFINE_FUNCTION_ALLOWING_MULTIPLE_DEFINITIONS EIGEN_UNUSED Packet2d pexp(const Packet2d &_x) + { + Packet2d x = _x; + + Packet2d tmp, fx; + Packet2l emm0; + + // clamp x + x = pmax(pmin(x, p2d_exp_hi), p2d_exp_lo); + /* express exp(x) as exp(g + n*log(2)) */ + fx = pmadd(p2d_cephes_LOG2EF, x, p2d_half); + + fx = vec_floor(fx); + + tmp = pmul(fx, p2d_cephes_exp_C1); + Packet2d z = pmul(fx, p2d_cephes_exp_C2); + x = psub(x, tmp); + x = psub(x, z); + + Packet2d x2 = pmul(x, x); + + Packet2d px = p2d_cephes_exp_p0; + px = pmadd(px, x2, p2d_cephes_exp_p1); + px = pmadd(px, x2, p2d_cephes_exp_p2); + px = pmul(px, x); + + Packet2d qx = p2d_cephes_exp_q0; + qx = pmadd(qx, x2, p2d_cephes_exp_q1); + qx = pmadd(qx, x2, p2d_cephes_exp_q2); + qx = pmadd(qx, x2, p2d_cephes_exp_q3); + + x = pdiv(px, psub(qx, px)); + x = pmadd(p2d_2, x, p2d_1); + + // build 2^n + emm0 = vec_ctsl(fx, 0); + + static const Packet2l p2l_1023 = { 1023, 1023 }; + static const Packet2ul p2ul_52 = { 52, 52 }; + + emm0 = emm0 + p2l_1023; + emm0 = emm0 << reinterpret_cast(p2ul_52); + + // Altivec's max & min operators just drop silent NaNs. Check NaNs in + // inputs and return them unmodified. + Packet2ul isnumber_mask = reinterpret_cast(vec_cmpeq(_x, _x)); + return vec_sel(_x, pmax(pmul(x, reinterpret_cast(emm0)), _x), isnumber_mask); + } + + template<> EIGEN_DEFINE_FUNCTION_ALLOWING_MULTIPLE_DEFINITIONS EIGEN_UNUSED Packet4f pexp(const Packet4f &x) + { + Packet4f res; + res.v4f[0] = pexp(x.v4f[0]); + res.v4f[1] = pexp(x.v4f[1]); + return res; + } + + template<> + EIGEN_DEFINE_FUNCTION_ALLOWING_MULTIPLE_DEFINITIONS EIGEN_UNUSED Packet2d psqrt(const Packet2d &x) + { + return __builtin_s390_vfsqdb(x); + } + + template<> + EIGEN_DEFINE_FUNCTION_ALLOWING_MULTIPLE_DEFINITIONS EIGEN_UNUSED Packet4f psqrt(const Packet4f &x) + { + Packet4f res; + res.v4f[0] = psqrt(x.v4f[0]); + res.v4f[1] = psqrt(x.v4f[1]); + return res; + } + + template<> + EIGEN_DEFINE_FUNCTION_ALLOWING_MULTIPLE_DEFINITIONS EIGEN_UNUSED Packet2d prsqrt(const Packet2d &x) + { + // Unfortunately we can't use the much faster mm_rqsrt_pd since it only provides an approximation. + return pset1(1.0) / psqrt(x); + } + + template<> + EIGEN_DEFINE_FUNCTION_ALLOWING_MULTIPLE_DEFINITIONS EIGEN_UNUSED Packet4f prsqrt(const Packet4f &x) + { + Packet4f res; + res.v4f[0] = prsqrt(x.v4f[0]); + res.v4f[1] = prsqrt(x.v4f[1]); + return res; + } + +}// end namespace internal + +}// end namespace Eigen + +#endif// EIGEN_MATH_FUNCTIONS_ALTIVEC_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/arch/ZVector/PacketMath.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/arch/ZVector/PacketMath.h old mode 100755 new mode 100644 index 57b01fc6..3ce304cf --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/arch/ZVector/PacketMath.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/arch/ZVector/PacketMath.h @@ -29,917 +29,991 @@ namespace internal { #endif #ifndef EIGEN_ARCH_DEFAULT_NUMBER_OF_REGISTERS -#define EIGEN_ARCH_DEFAULT_NUMBER_OF_REGISTERS 16 +#define EIGEN_ARCH_DEFAULT_NUMBER_OF_REGISTERS 16 #endif -typedef __vector int Packet4i; -typedef __vector unsigned int Packet4ui; -typedef __vector __bool int Packet4bi; -typedef __vector short int Packet8i; -typedef __vector unsigned char Packet16uc; -typedef __vector double Packet2d; -typedef __vector unsigned long long Packet2ul; -typedef __vector long long Packet2l; + typedef __vector int Packet4i; + typedef __vector unsigned int Packet4ui; + typedef __vector __bool int Packet4bi; + typedef __vector short int Packet8i; + typedef __vector unsigned char Packet16uc; + typedef __vector double Packet2d; + typedef __vector unsigned long long Packet2ul; + typedef __vector long long Packet2l; -typedef struct { - Packet2d v4f[2]; -} Packet4f; + typedef struct + { + Packet2d v4f[2]; + } Packet4f; -typedef union { - int32_t i[4]; - uint32_t ui[4]; - int64_t l[2]; - uint64_t ul[2]; - double d[2]; - Packet4i v4i; - Packet4ui v4ui; - Packet2l v2l; - Packet2ul v2ul; - Packet2d v2d; -} Packet; + typedef union { + int32_t i[4]; + uint32_t ui[4]; + int64_t l[2]; + uint64_t ul[2]; + double d[2]; + Packet4i v4i; + Packet4ui v4ui; + Packet2l v2l; + Packet2ul v2ul; + Packet2d v2d; + } Packet; -// We don't want to write the same code all the time, but we need to reuse the constants -// and it doesn't really work to declare them global, so we define macros instead + // We don't want to write the same code all the time, but we need to reuse the constants + // and it doesn't really work to declare them global, so we define macros instead -#define _EIGEN_DECLARE_CONST_FAST_Packet4i(NAME,X) \ - Packet4i p4i_##NAME = reinterpret_cast(vec_splat_s32(X)) +#define _EIGEN_DECLARE_CONST_FAST_Packet4i(NAME, X) Packet4i p4i_##NAME = reinterpret_cast(vec_splat_s32(X)) -#define _EIGEN_DECLARE_CONST_FAST_Packet2d(NAME,X) \ - Packet2d p2d_##NAME = reinterpret_cast(vec_splat_s64(X)) +#define _EIGEN_DECLARE_CONST_FAST_Packet2d(NAME, X) Packet2d p2d_##NAME = reinterpret_cast(vec_splat_s64(X)) -#define _EIGEN_DECLARE_CONST_FAST_Packet2l(NAME,X) \ - Packet2l p2l_##NAME = reinterpret_cast(vec_splat_s64(X)) +#define _EIGEN_DECLARE_CONST_FAST_Packet2l(NAME, X) Packet2l p2l_##NAME = reinterpret_cast(vec_splat_s64(X)) -#define _EIGEN_DECLARE_CONST_Packet4i(NAME,X) \ - Packet4i p4i_##NAME = pset1(X) +#define _EIGEN_DECLARE_CONST_Packet4i(NAME, X) Packet4i p4i_##NAME = pset1(X) -#define _EIGEN_DECLARE_CONST_Packet2d(NAME,X) \ - Packet2d p2d_##NAME = pset1(X) +#define _EIGEN_DECLARE_CONST_Packet2d(NAME, X) Packet2d p2d_##NAME = pset1(X) -#define _EIGEN_DECLARE_CONST_Packet2l(NAME,X) \ - Packet2l p2l_##NAME = pset1(X) +#define _EIGEN_DECLARE_CONST_Packet2l(NAME, X) Packet2l p2l_##NAME = pset1(X) -// These constants are endian-agnostic -//static _EIGEN_DECLARE_CONST_FAST_Packet4i(ZERO, 0); //{ 0, 0, 0, 0,} -static _EIGEN_DECLARE_CONST_FAST_Packet4i(ONE, 1); //{ 1, 1, 1, 1} + // These constants are endian-agnostic + // static _EIGEN_DECLARE_CONST_FAST_Packet4i(ZERO, 0); //{ 0, 0, 0, 0,} + static _EIGEN_DECLARE_CONST_FAST_Packet4i(ONE, 1);//{ 1, 1, 1, 1} -static _EIGEN_DECLARE_CONST_FAST_Packet2d(ZERO, 0); -static _EIGEN_DECLARE_CONST_FAST_Packet2l(ZERO, 0); -static _EIGEN_DECLARE_CONST_FAST_Packet2l(ONE, 1); + static _EIGEN_DECLARE_CONST_FAST_Packet2d(ZERO, 0); + static _EIGEN_DECLARE_CONST_FAST_Packet2l(ZERO, 0); + static _EIGEN_DECLARE_CONST_FAST_Packet2l(ONE, 1); -static Packet2d p2d_ONE = { 1.0, 1.0 }; -static Packet2d p2d_ZERO_ = { -0.0, -0.0 }; + static Packet2d p2d_ONE = { 1.0, 1.0 }; + static Packet2d p2d_ZERO_ = { -0.0, -0.0 }; -static Packet4i p4i_COUNTDOWN = { 0, 1, 2, 3 }; -static Packet4f p4f_COUNTDOWN = { 0.0, 1.0, 2.0, 3.0 }; -static Packet2d p2d_COUNTDOWN = reinterpret_cast(vec_sld(reinterpret_cast(p2d_ZERO), reinterpret_cast(p2d_ONE), 8)); + static Packet4i p4i_COUNTDOWN = { 0, 1, 2, 3 }; + static Packet4f p4f_COUNTDOWN = { 0.0, 1.0, 2.0, 3.0 }; + static Packet2d p2d_COUNTDOWN = reinterpret_cast( + vec_sld(reinterpret_cast(p2d_ZERO), reinterpret_cast(p2d_ONE), 8)); -static Packet16uc p16uc_PSET64_HI = { 0,1,2,3, 4,5,6,7, 0,1,2,3, 4,5,6,7 }; -static Packet16uc p16uc_DUPLICATE32_HI = { 0,1,2,3, 0,1,2,3, 4,5,6,7, 4,5,6,7 }; + static Packet16uc p16uc_PSET64_HI = { 0, 1, 2, 3, 4, 5, 6, 7, 0, 1, 2, 3, 4, 5, 6, 7 }; + static Packet16uc p16uc_DUPLICATE32_HI = { 0, 1, 2, 3, 0, 1, 2, 3, 4, 5, 6, 7, 4, 5, 6, 7 }; // Mask alignment -#define _EIGEN_MASK_ALIGNMENT 0xfffffffffffffff0 +#define _EIGEN_MASK_ALIGNMENT 0xfffffffffffffff0 -#define _EIGEN_ALIGNED_PTR(x) ((std::ptrdiff_t)(x) & _EIGEN_MASK_ALIGNMENT) +#define _EIGEN_ALIGNED_PTR(x) ((std::ptrdiff_t)(x) & _EIGEN_MASK_ALIGNMENT) -// Handle endianness properly while loading constants -// Define global static constants: + // Handle endianness properly while loading constants + // Define global static constants: -static Packet16uc p16uc_FORWARD = { 0,1,2,3, 4,5,6,7, 8,9,10,11, 12,13,14,15 }; -static Packet16uc p16uc_REVERSE32 = { 12,13,14,15, 8,9,10,11, 4,5,6,7, 0,1,2,3 }; -static Packet16uc p16uc_REVERSE64 = { 8,9,10,11, 12,13,14,15, 0,1,2,3, 4,5,6,7 }; + static Packet16uc p16uc_FORWARD = { 0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15 }; + static Packet16uc p16uc_REVERSE32 = { 12, 13, 14, 15, 8, 9, 10, 11, 4, 5, 6, 7, 0, 1, 2, 3 }; + static Packet16uc p16uc_REVERSE64 = { 8, 9, 10, 11, 12, 13, 14, 15, 0, 1, 2, 3, 4, 5, 6, 7 }; -static Packet16uc p16uc_PSET32_WODD = vec_sld((Packet16uc) vec_splat((Packet4ui)p16uc_FORWARD, 0), (Packet16uc) vec_splat((Packet4ui)p16uc_FORWARD, 2), 8);//{ 0,1,2,3, 0,1,2,3, 8,9,10,11, 8,9,10,11 }; -static Packet16uc p16uc_PSET32_WEVEN = vec_sld(p16uc_DUPLICATE32_HI, (Packet16uc) vec_splat((Packet4ui)p16uc_FORWARD, 3), 8);//{ 4,5,6,7, 4,5,6,7, 12,13,14,15, 12,13,14,15 }; -/*static Packet16uc p16uc_HALF64_0_16 = vec_sld((Packet16uc)p4i_ZERO, vec_splat((Packet16uc) vec_abs(p4i_MINUS16), 3), 8); //{ 0,0,0,0, 0,0,0,0, 16,16,16,16, 16,16,16,16}; + static Packet16uc p16uc_PSET32_WODD = vec_sld((Packet16uc)vec_splat((Packet4ui)p16uc_FORWARD, 0), + (Packet16uc)vec_splat((Packet4ui)p16uc_FORWARD, 2), + 8);//{ 0,1,2,3, 0,1,2,3, 8,9,10,11, 8,9,10,11 }; + static Packet16uc p16uc_PSET32_WEVEN = vec_sld(p16uc_DUPLICATE32_HI, + (Packet16uc)vec_splat((Packet4ui)p16uc_FORWARD, 3), + 8);//{ 4,5,6,7, 4,5,6,7, 12,13,14,15, 12,13,14,15 }; + /*static Packet16uc p16uc_HALF64_0_16 = vec_sld((Packet16uc)p4i_ZERO, vec_splat((Packet16uc) vec_abs(p4i_MINUS16), 3), + 8); //{ 0,0,0,0, 0,0,0,0, 16,16,16,16, 16,16,16,16}; -static Packet16uc p16uc_PSET64_HI = (Packet16uc) vec_mergeh((Packet4ui)p16uc_PSET32_WODD, (Packet4ui)p16uc_PSET32_WEVEN); //{ 0,1,2,3, 4,5,6,7, 0,1,2,3, 4,5,6,7 };*/ -static Packet16uc p16uc_PSET64_LO = (Packet16uc) vec_mergel((Packet4ui)p16uc_PSET32_WODD, (Packet4ui)p16uc_PSET32_WEVEN); //{ 8,9,10,11, 12,13,14,15, 8,9,10,11, 12,13,14,15 }; -/*static Packet16uc p16uc_TRANSPOSE64_HI = vec_add(p16uc_PSET64_HI, p16uc_HALF64_0_16); //{ 0,1,2,3, 4,5,6,7, 16,17,18,19, 20,21,22,23}; -static Packet16uc p16uc_TRANSPOSE64_LO = vec_add(p16uc_PSET64_LO, p16uc_HALF64_0_16); //{ 8,9,10,11, 12,13,14,15, 24,25,26,27, 28,29,30,31};*/ -static Packet16uc p16uc_TRANSPOSE64_HI = { 0,1,2,3, 4,5,6,7, 16,17,18,19, 20,21,22,23}; -static Packet16uc p16uc_TRANSPOSE64_LO = { 8,9,10,11, 12,13,14,15, 24,25,26,27, 28,29,30,31}; + static Packet16uc p16uc_PSET64_HI = (Packet16uc) vec_mergeh((Packet4ui)p16uc_PSET32_WODD, + (Packet4ui)p16uc_PSET32_WEVEN); //{ 0,1,2,3, 4,5,6,7, 0,1,2,3, 4,5,6,7 };*/ + static Packet16uc p16uc_PSET64_LO = (Packet16uc)vec_mergel((Packet4ui)p16uc_PSET32_WODD, + (Packet4ui)p16uc_PSET32_WEVEN);//{ 8,9,10,11, 12,13,14,15, 8,9,10,11, 12,13,14,15 }; + /*static Packet16uc p16uc_TRANSPOSE64_HI = vec_add(p16uc_PSET64_HI, p16uc_HALF64_0_16); //{ 0,1,2,3, 4,5,6,7, + 16,17,18,19, 20,21,22,23}; static Packet16uc p16uc_TRANSPOSE64_LO = vec_add(p16uc_PSET64_LO, p16uc_HALF64_0_16); //{ + 8,9,10,11, 12,13,14,15, 24,25,26,27, 28,29,30,31};*/ + static Packet16uc p16uc_TRANSPOSE64_HI = { 0, 1, 2, 3, 4, 5, 6, 7, 16, 17, 18, 19, 20, 21, 22, 23 }; + static Packet16uc p16uc_TRANSPOSE64_LO = { 8, 9, 10, 11, 12, 13, 14, 15, 24, 25, 26, 27, 28, 29, 30, 31 }; -//static Packet16uc p16uc_COMPLEX32_REV = vec_sld(p16uc_REVERSE32, p16uc_REVERSE32, 8); //{ 4,5,6,7, 0,1,2,3, 12,13,14,15, 8,9,10,11 }; + // static Packet16uc p16uc_COMPLEX32_REV = vec_sld(p16uc_REVERSE32, p16uc_REVERSE32, 8); //{ 4,5,6,7, 0,1,2,3, + // 12,13,14,15, 8,9,10,11 }; -//static Packet16uc p16uc_COMPLEX32_REV2 = vec_sld(p16uc_FORWARD, p16uc_FORWARD, 8); //{ 8,9,10,11, 12,13,14,15, 0,1,2,3, 4,5,6,7 }; + // static Packet16uc p16uc_COMPLEX32_REV2 = vec_sld(p16uc_FORWARD, p16uc_FORWARD, 8); //{ 8,9,10,11, 12,13,14,15, + // 0,1,2,3, 4,5,6,7 }; #if EIGEN_HAS_BUILTIN(__builtin_prefetch) || EIGEN_COMP_GNUC - #define EIGEN_ZVECTOR_PREFETCH(ADDR) __builtin_prefetch(ADDR); +#define EIGEN_ZVECTOR_PREFETCH(ADDR) __builtin_prefetch(ADDR); #else - #define EIGEN_ZVECTOR_PREFETCH(ADDR) asm( " pfd [%[addr]]\n" :: [addr] "r" (ADDR) : "cc" ); +#define EIGEN_ZVECTOR_PREFETCH(ADDR) asm(" pfd [%[addr]]\n" ::[addr] "r"(ADDR) : "cc"); #endif -template<> struct packet_traits : default_packet_traits -{ - typedef Packet4i type; - typedef Packet4i half; - enum { - Vectorizable = 1, - AlignedOnScalar = 1, - size = 4, - HasHalfPacket = 0, - - HasAdd = 1, - HasSub = 1, - HasMul = 1, - HasDiv = 1, - HasBlend = 1 + template<> struct packet_traits : default_packet_traits + { + typedef Packet4i type; + typedef Packet4i half; + enum { + Vectorizable = 1, + AlignedOnScalar = 1, + size = 4, + HasHalfPacket = 0, + + HasAdd = 1, + HasSub = 1, + HasMul = 1, + HasDiv = 1, + HasBlend = 1 + }; }; -}; - -template<> struct packet_traits : default_packet_traits -{ - typedef Packet4f type; - typedef Packet4f half; - enum { - Vectorizable = 1, - AlignedOnScalar = 1, - size=4, - HasHalfPacket = 0, - - HasAdd = 1, - HasSub = 1, - HasMul = 1, - HasDiv = 1, - HasMin = 1, - HasMax = 1, - HasAbs = 1, - HasSin = 0, - HasCos = 0, - HasLog = 0, - HasExp = 1, - HasSqrt = 1, - HasRsqrt = 1, - HasRound = 1, - HasFloor = 1, - HasCeil = 1, - HasNegate = 1, - HasBlend = 1 + + template<> struct packet_traits : default_packet_traits + { + typedef Packet4f type; + typedef Packet4f half; + enum { + Vectorizable = 1, + AlignedOnScalar = 1, + size = 4, + HasHalfPacket = 0, + + HasAdd = 1, + HasSub = 1, + HasMul = 1, + HasDiv = 1, + HasMin = 1, + HasMax = 1, + HasAbs = 1, + HasSin = 0, + HasCos = 0, + HasLog = 0, + HasExp = 1, + HasSqrt = 1, + HasRsqrt = 1, + HasRound = 1, + HasFloor = 1, + HasCeil = 1, + HasNegate = 1, + HasBlend = 1 + }; }; -}; - -template<> struct packet_traits : default_packet_traits -{ - typedef Packet2d type; - typedef Packet2d half; - enum { - Vectorizable = 1, - AlignedOnScalar = 1, - size=2, - HasHalfPacket = 1, - - HasAdd = 1, - HasSub = 1, - HasMul = 1, - HasDiv = 1, - HasMin = 1, - HasMax = 1, - HasAbs = 1, - HasSin = 0, - HasCos = 0, - HasLog = 0, - HasExp = 1, - HasSqrt = 1, - HasRsqrt = 1, - HasRound = 1, - HasFloor = 1, - HasCeil = 1, - HasNegate = 1, - HasBlend = 1 + + template<> struct packet_traits : default_packet_traits + { + typedef Packet2d type; + typedef Packet2d half; + enum { + Vectorizable = 1, + AlignedOnScalar = 1, + size = 2, + HasHalfPacket = 1, + + HasAdd = 1, + HasSub = 1, + HasMul = 1, + HasDiv = 1, + HasMin = 1, + HasMax = 1, + HasAbs = 1, + HasSin = 0, + HasCos = 0, + HasLog = 0, + HasExp = 1, + HasSqrt = 1, + HasRsqrt = 1, + HasRound = 1, + HasFloor = 1, + HasCeil = 1, + HasNegate = 1, + HasBlend = 1 + }; }; -}; - -template<> struct unpacket_traits { typedef int type; enum {size=4, alignment=Aligned16}; typedef Packet4i half; }; -template<> struct unpacket_traits { typedef float type; enum {size=4, alignment=Aligned16}; typedef Packet4f half; }; -template<> struct unpacket_traits { typedef double type; enum {size=2, alignment=Aligned16}; typedef Packet2d half; }; - -/* Forward declaration */ -EIGEN_DEVICE_FUNC inline void ptranspose(PacketBlock& kernel); - -inline std::ostream & operator <<(std::ostream & s, const Packet4i & v) -{ - Packet vt; - vt.v4i = v; - s << vt.i[0] << ", " << vt.i[1] << ", " << vt.i[2] << ", " << vt.i[3]; - return s; -} - -inline std::ostream & operator <<(std::ostream & s, const Packet4ui & v) -{ - Packet vt; - vt.v4ui = v; - s << vt.ui[0] << ", " << vt.ui[1] << ", " << vt.ui[2] << ", " << vt.ui[3]; - return s; -} - -inline std::ostream & operator <<(std::ostream & s, const Packet2l & v) -{ - Packet vt; - vt.v2l = v; - s << vt.l[0] << ", " << vt.l[1]; - return s; -} - -inline std::ostream & operator <<(std::ostream & s, const Packet2ul & v) -{ - Packet vt; - vt.v2ul = v; - s << vt.ul[0] << ", " << vt.ul[1] ; - return s; -} - -inline std::ostream & operator <<(std::ostream & s, const Packet2d & v) -{ - Packet vt; - vt.v2d = v; - s << vt.d[0] << ", " << vt.d[1]; - return s; -} - -/* Helper function to simulate a vec_splat_packet4f - */ -template EIGEN_STRONG_INLINE Packet4f vec_splat_packet4f(const Packet4f& from) -{ - Packet4f splat; - switch (element) { - case 0: - splat.v4f[0] = vec_splat(from.v4f[0], 0); - splat.v4f[1] = splat.v4f[0]; - break; - case 1: - splat.v4f[0] = vec_splat(from.v4f[0], 1); - splat.v4f[1] = splat.v4f[0]; - break; - case 2: - splat.v4f[0] = vec_splat(from.v4f[1], 0); - splat.v4f[1] = splat.v4f[0]; - break; - case 3: - splat.v4f[0] = vec_splat(from.v4f[1], 1); - splat.v4f[1] = splat.v4f[0]; - break; - } - return splat; -} - -template -struct palign_impl -{ - static EIGEN_STRONG_INLINE void run(Packet4i& first, const Packet4i& second) - { - switch (Offset % 4) { - case 1: - first = vec_sld(first, second, 4); break; - case 2: - first = vec_sld(first, second, 8); break; - case 3: - first = vec_sld(first, second, 12); break; - } + + template<> struct unpacket_traits + { + typedef int type; + enum { size = 4, alignment = Aligned16 }; + typedef Packet4i half; + }; + template<> struct unpacket_traits + { + typedef float type; + enum { size = 4, alignment = Aligned16 }; + typedef Packet4f half; + }; + template<> struct unpacket_traits + { + typedef double type; + enum { size = 2, alignment = Aligned16 }; + typedef Packet2d half; + }; + + /* Forward declaration */ + EIGEN_DEVICE_FUNC inline void ptranspose(PacketBlock &kernel); + + inline std::ostream &operator<<(std::ostream &s, const Packet4i &v) + { + Packet vt; + vt.v4i = v; + s << vt.i[0] << ", " << vt.i[1] << ", " << vt.i[2] << ", " << vt.i[3]; + return s; + } + + inline std::ostream &operator<<(std::ostream &s, const Packet4ui &v) + { + Packet vt; + vt.v4ui = v; + s << vt.ui[0] << ", " << vt.ui[1] << ", " << vt.ui[2] << ", " << vt.ui[3]; + return s; + } + + inline std::ostream &operator<<(std::ostream &s, const Packet2l &v) + { + Packet vt; + vt.v2l = v; + s << vt.l[0] << ", " << vt.l[1]; + return s; + } + + inline std::ostream &operator<<(std::ostream &s, const Packet2ul &v) + { + Packet vt; + vt.v2ul = v; + s << vt.ul[0] << ", " << vt.ul[1]; + return s; + } + + inline std::ostream &operator<<(std::ostream &s, const Packet2d &v) + { + Packet vt; + vt.v2d = v; + s << vt.d[0] << ", " << vt.d[1]; + return s; } -}; -/* This is a tricky one, we have to translate float alignment to vector elements of sizeof double - */ -template -struct palign_impl -{ - static EIGEN_STRONG_INLINE void run(Packet4f& first, const Packet4f& second) + /* Helper function to simulate a vec_splat_packet4f + */ + template EIGEN_STRONG_INLINE Packet4f vec_splat_packet4f(const Packet4f &from) { - switch (Offset % 4) { + Packet4f splat; + switch (element) { + case 0: + splat.v4f[0] = vec_splat(from.v4f[0], 0); + splat.v4f[1] = splat.v4f[0]; + break; case 1: - first.v4f[0] = vec_sld(first.v4f[0], first.v4f[1], 8); - first.v4f[1] = vec_sld(first.v4f[1], second.v4f[0], 8); + splat.v4f[0] = vec_splat(from.v4f[0], 1); + splat.v4f[1] = splat.v4f[0]; break; case 2: - first.v4f[0] = first.v4f[1]; - first.v4f[1] = second.v4f[0]; + splat.v4f[0] = vec_splat(from.v4f[1], 0); + splat.v4f[1] = splat.v4f[0]; break; case 3: - first.v4f[0] = vec_sld(first.v4f[1], second.v4f[0], 8); - first.v4f[1] = vec_sld(second.v4f[0], second.v4f[1], 8); + splat.v4f[0] = vec_splat(from.v4f[1], 1); + splat.v4f[1] = splat.v4f[0]; break; } + return splat; + } + + template struct palign_impl + { + static EIGEN_STRONG_INLINE void run(Packet4i &first, const Packet4i &second) + { + switch (Offset % 4) { + case 1: + first = vec_sld(first, second, 4); + break; + case 2: + first = vec_sld(first, second, 8); + break; + case 3: + first = vec_sld(first, second, 12); + break; + } + } + }; + + /* This is a tricky one, we have to translate float alignment to vector elements of sizeof double + */ + template struct palign_impl + { + static EIGEN_STRONG_INLINE void run(Packet4f &first, const Packet4f &second) + { + switch (Offset % 4) { + case 1: + first.v4f[0] = vec_sld(first.v4f[0], first.v4f[1], 8); + first.v4f[1] = vec_sld(first.v4f[1], second.v4f[0], 8); + break; + case 2: + first.v4f[0] = first.v4f[1]; + first.v4f[1] = second.v4f[0]; + break; + case 3: + first.v4f[0] = vec_sld(first.v4f[1], second.v4f[0], 8); + first.v4f[1] = vec_sld(second.v4f[0], second.v4f[1], 8); + break; + } + } + }; + + + template struct palign_impl + { + static EIGEN_STRONG_INLINE void run(Packet2d &first, const Packet2d &second) + { + if (Offset == 1) + first = + reinterpret_cast(vec_sld(reinterpret_cast(first), reinterpret_cast(second), 8)); + } + }; + + template<> EIGEN_STRONG_INLINE Packet4i pload(const int *from) + { + // FIXME: No intrinsic yet + EIGEN_DEBUG_ALIGNED_LOAD + Packet *vfrom; + vfrom = (Packet *)from; + return vfrom->v4i; + } + + template<> EIGEN_STRONG_INLINE Packet4f pload(const float *from) + { + // FIXME: No intrinsic yet + EIGEN_DEBUG_ALIGNED_LOAD + Packet4f vfrom; + vfrom.v4f[0] = vec_ld2f(&from[0]); + vfrom.v4f[1] = vec_ld2f(&from[2]); + return vfrom; + } + + template<> EIGEN_STRONG_INLINE Packet2d pload(const double *from) + { + // FIXME: No intrinsic yet + EIGEN_DEBUG_ALIGNED_LOAD + Packet *vfrom; + vfrom = (Packet *)from; + return vfrom->v2d; + } + + template<> EIGEN_STRONG_INLINE void pstore(int *to, const Packet4i &from) + { + // FIXME: No intrinsic yet + EIGEN_DEBUG_ALIGNED_STORE + Packet *vto; + vto = (Packet *)to; + vto->v4i = from; + } + + template<> EIGEN_STRONG_INLINE void pstore(float *to, const Packet4f &from) + { + // FIXME: No intrinsic yet + EIGEN_DEBUG_ALIGNED_STORE + vec_st2f(from.v4f[0], &to[0]); + vec_st2f(from.v4f[1], &to[2]); } -}; - - -template -struct palign_impl -{ - static EIGEN_STRONG_INLINE void run(Packet2d& first, const Packet2d& second) - { - if (Offset == 1) - first = reinterpret_cast(vec_sld(reinterpret_cast(first), reinterpret_cast(second), 8)); - } -}; - -template<> EIGEN_STRONG_INLINE Packet4i pload(const int* from) -{ - // FIXME: No intrinsic yet - EIGEN_DEBUG_ALIGNED_LOAD - Packet *vfrom; - vfrom = (Packet *) from; - return vfrom->v4i; -} - -template<> EIGEN_STRONG_INLINE Packet4f pload(const float* from) -{ - // FIXME: No intrinsic yet - EIGEN_DEBUG_ALIGNED_LOAD - Packet4f vfrom; - vfrom.v4f[0] = vec_ld2f(&from[0]); - vfrom.v4f[1] = vec_ld2f(&from[2]); - return vfrom; -} - -template<> EIGEN_STRONG_INLINE Packet2d pload(const double* from) -{ - // FIXME: No intrinsic yet - EIGEN_DEBUG_ALIGNED_LOAD - Packet *vfrom; - vfrom = (Packet *) from; - return vfrom->v2d; -} - -template<> EIGEN_STRONG_INLINE void pstore(int* to, const Packet4i& from) -{ - // FIXME: No intrinsic yet - EIGEN_DEBUG_ALIGNED_STORE - Packet *vto; - vto = (Packet *) to; - vto->v4i = from; -} - -template<> EIGEN_STRONG_INLINE void pstore(float* to, const Packet4f& from) -{ - // FIXME: No intrinsic yet - EIGEN_DEBUG_ALIGNED_STORE - vec_st2f(from.v4f[0], &to[0]); - vec_st2f(from.v4f[1], &to[2]); -} - - -template<> EIGEN_STRONG_INLINE void pstore(double* to, const Packet2d& from) -{ - // FIXME: No intrinsic yet - EIGEN_DEBUG_ALIGNED_STORE - Packet *vto; - vto = (Packet *) to; - vto->v2d = from; -} - -template<> EIGEN_STRONG_INLINE Packet4i pset1(const int& from) -{ - return vec_splats(from); -} -template<> EIGEN_STRONG_INLINE Packet2d pset1(const double& from) { - return vec_splats(from); -} -template<> EIGEN_STRONG_INLINE Packet4f pset1(const float& from) -{ - Packet4f to; - to.v4f[0] = pset1(static_cast(from)); - to.v4f[1] = to.v4f[0]; - return to; -} - -template<> EIGEN_STRONG_INLINE void -pbroadcast4(const int *a, - Packet4i& a0, Packet4i& a1, Packet4i& a2, Packet4i& a3) -{ - a3 = pload(a); - a0 = vec_splat(a3, 0); - a1 = vec_splat(a3, 1); - a2 = vec_splat(a3, 2); - a3 = vec_splat(a3, 3); -} - -template<> EIGEN_STRONG_INLINE void -pbroadcast4(const float *a, - Packet4f& a0, Packet4f& a1, Packet4f& a2, Packet4f& a3) -{ - a3 = pload(a); - a0 = vec_splat_packet4f<0>(a3); - a1 = vec_splat_packet4f<1>(a3); - a2 = vec_splat_packet4f<2>(a3); - a3 = vec_splat_packet4f<3>(a3); -} - -template<> EIGEN_STRONG_INLINE void -pbroadcast4(const double *a, - Packet2d& a0, Packet2d& a1, Packet2d& a2, Packet2d& a3) -{ - a1 = pload(a); - a0 = vec_splat(a1, 0); - a1 = vec_splat(a1, 1); - a3 = pload(a+2); - a2 = vec_splat(a3, 0); - a3 = vec_splat(a3, 1); -} - -template<> EIGEN_DEVICE_FUNC inline Packet4i pgather(const int* from, Index stride) -{ - int EIGEN_ALIGN16 ai[4]; - ai[0] = from[0*stride]; - ai[1] = from[1*stride]; - ai[2] = from[2*stride]; - ai[3] = from[3*stride]; - return pload(ai); -} - -template<> EIGEN_DEVICE_FUNC inline Packet4f pgather(const float* from, Index stride) -{ - float EIGEN_ALIGN16 ai[4]; - ai[0] = from[0*stride]; - ai[1] = from[1*stride]; - ai[2] = from[2*stride]; - ai[3] = from[3*stride]; - return pload(ai); -} - -template<> EIGEN_DEVICE_FUNC inline Packet2d pgather(const double* from, Index stride) -{ - double EIGEN_ALIGN16 af[2]; - af[0] = from[0*stride]; - af[1] = from[1*stride]; - return pload(af); -} - -template<> EIGEN_DEVICE_FUNC inline void pscatter(int* to, const Packet4i& from, Index stride) -{ - int EIGEN_ALIGN16 ai[4]; - pstore((int *)ai, from); - to[0*stride] = ai[0]; - to[1*stride] = ai[1]; - to[2*stride] = ai[2]; - to[3*stride] = ai[3]; -} - -template<> EIGEN_DEVICE_FUNC inline void pscatter(float* to, const Packet4f& from, Index stride) -{ - float EIGEN_ALIGN16 ai[4]; - pstore((float *)ai, from); - to[0*stride] = ai[0]; - to[1*stride] = ai[1]; - to[2*stride] = ai[2]; - to[3*stride] = ai[3]; -} - -template<> EIGEN_DEVICE_FUNC inline void pscatter(double* to, const Packet2d& from, Index stride) -{ - double EIGEN_ALIGN16 af[2]; - pstore(af, from); - to[0*stride] = af[0]; - to[1*stride] = af[1]; -} - -template<> EIGEN_STRONG_INLINE Packet4i padd(const Packet4i& a, const Packet4i& b) { return (a + b); } -template<> EIGEN_STRONG_INLINE Packet4f padd(const Packet4f& a, const Packet4f& b) -{ - Packet4f c; - c.v4f[0] = a.v4f[0] + b.v4f[0]; - c.v4f[1] = a.v4f[1] + b.v4f[1]; - return c; -} -template<> EIGEN_STRONG_INLINE Packet2d padd(const Packet2d& a, const Packet2d& b) { return (a + b); } - -template<> EIGEN_STRONG_INLINE Packet4i psub(const Packet4i& a, const Packet4i& b) { return (a - b); } -template<> EIGEN_STRONG_INLINE Packet4f psub(const Packet4f& a, const Packet4f& b) -{ - Packet4f c; - c.v4f[0] = a.v4f[0] - b.v4f[0]; - c.v4f[1] = a.v4f[1] - b.v4f[1]; - return c; -} -template<> EIGEN_STRONG_INLINE Packet2d psub(const Packet2d& a, const Packet2d& b) { return (a - b); } - -template<> EIGEN_STRONG_INLINE Packet4i pmul(const Packet4i& a, const Packet4i& b) { return (a * b); } -template<> EIGEN_STRONG_INLINE Packet4f pmul(const Packet4f& a, const Packet4f& b) -{ - Packet4f c; - c.v4f[0] = a.v4f[0] * b.v4f[0]; - c.v4f[1] = a.v4f[1] * b.v4f[1]; - return c; -} -template<> EIGEN_STRONG_INLINE Packet2d pmul(const Packet2d& a, const Packet2d& b) { return (a * b); } - -template<> EIGEN_STRONG_INLINE Packet4i pdiv(const Packet4i& a, const Packet4i& b) { return (a / b); } -template<> EIGEN_STRONG_INLINE Packet4f pdiv(const Packet4f& a, const Packet4f& b) -{ - Packet4f c; - c.v4f[0] = a.v4f[0] / b.v4f[0]; - c.v4f[1] = a.v4f[1] / b.v4f[1]; - return c; -} -template<> EIGEN_STRONG_INLINE Packet2d pdiv(const Packet2d& a, const Packet2d& b) { return (a / b); } - -template<> EIGEN_STRONG_INLINE Packet4i pnegate(const Packet4i& a) { return (-a); } -template<> EIGEN_STRONG_INLINE Packet4f pnegate(const Packet4f& a) -{ - Packet4f c; - c.v4f[0] = -a.v4f[0]; - c.v4f[1] = -a.v4f[1]; - return c; -} -template<> EIGEN_STRONG_INLINE Packet2d pnegate(const Packet2d& a) { return (-a); } - -template<> EIGEN_STRONG_INLINE Packet4i pconj(const Packet4i& a) { return a; } -template<> EIGEN_STRONG_INLINE Packet4f pconj(const Packet4f& a) { return a; } -template<> EIGEN_STRONG_INLINE Packet2d pconj(const Packet2d& a) { return a; } - -template<> EIGEN_STRONG_INLINE Packet4i pmadd(const Packet4i& a, const Packet4i& b, const Packet4i& c) { return padd(pmul(a, b), c); } -template<> EIGEN_STRONG_INLINE Packet4f pmadd(const Packet4f& a, const Packet4f& b, const Packet4f& c) -{ - Packet4f res; - res.v4f[0] = vec_madd(a.v4f[0], b.v4f[0], c.v4f[0]); - res.v4f[1] = vec_madd(a.v4f[1], b.v4f[1], c.v4f[1]); - return res; -} -template<> EIGEN_STRONG_INLINE Packet2d pmadd(const Packet2d& a, const Packet2d& b, const Packet2d& c) { return vec_madd(a, b, c); } - -template<> EIGEN_STRONG_INLINE Packet4i plset(const int& a) { return padd(pset1(a), p4i_COUNTDOWN); } -template<> EIGEN_STRONG_INLINE Packet4f plset(const float& a) { return padd(pset1(a), p4f_COUNTDOWN); } -template<> EIGEN_STRONG_INLINE Packet2d plset(const double& a) { return padd(pset1(a), p2d_COUNTDOWN); } - -template<> EIGEN_STRONG_INLINE Packet4i pmin(const Packet4i& a, const Packet4i& b) { return vec_min(a, b); } -template<> EIGEN_STRONG_INLINE Packet2d pmin(const Packet2d& a, const Packet2d& b) { return vec_min(a, b); } -template<> EIGEN_STRONG_INLINE Packet4f pmin(const Packet4f& a, const Packet4f& b) -{ - Packet4f res; - res.v4f[0] = pmin(a.v4f[0], b.v4f[0]); - res.v4f[1] = pmin(a.v4f[1], b.v4f[1]); - return res; -} - -template<> EIGEN_STRONG_INLINE Packet4i pmax(const Packet4i& a, const Packet4i& b) { return vec_max(a, b); } -template<> EIGEN_STRONG_INLINE Packet2d pmax(const Packet2d& a, const Packet2d& b) { return vec_max(a, b); } -template<> EIGEN_STRONG_INLINE Packet4f pmax(const Packet4f& a, const Packet4f& b) -{ - Packet4f res; - res.v4f[0] = pmax(a.v4f[0], b.v4f[0]); - res.v4f[1] = pmax(a.v4f[1], b.v4f[1]); - return res; -} - -template<> EIGEN_STRONG_INLINE Packet4i pand(const Packet4i& a, const Packet4i& b) { return vec_and(a, b); } -template<> EIGEN_STRONG_INLINE Packet2d pand(const Packet2d& a, const Packet2d& b) { return vec_and(a, b); } -template<> EIGEN_STRONG_INLINE Packet4f pand(const Packet4f& a, const Packet4f& b) -{ - Packet4f res; - res.v4f[0] = pand(a.v4f[0], b.v4f[0]); - res.v4f[1] = pand(a.v4f[1], b.v4f[1]); - return res; -} - -template<> EIGEN_STRONG_INLINE Packet4i por(const Packet4i& a, const Packet4i& b) { return vec_or(a, b); } -template<> EIGEN_STRONG_INLINE Packet2d por(const Packet2d& a, const Packet2d& b) { return vec_or(a, b); } -template<> EIGEN_STRONG_INLINE Packet4f por(const Packet4f& a, const Packet4f& b) -{ - Packet4f res; - res.v4f[0] = pand(a.v4f[0], b.v4f[0]); - res.v4f[1] = pand(a.v4f[1], b.v4f[1]); - return res; -} - -template<> EIGEN_STRONG_INLINE Packet4i pxor(const Packet4i& a, const Packet4i& b) { return vec_xor(a, b); } -template<> EIGEN_STRONG_INLINE Packet2d pxor(const Packet2d& a, const Packet2d& b) { return vec_xor(a, b); } -template<> EIGEN_STRONG_INLINE Packet4f pxor(const Packet4f& a, const Packet4f& b) -{ - Packet4f res; - res.v4f[0] = pand(a.v4f[0], b.v4f[0]); - res.v4f[1] = pand(a.v4f[1], b.v4f[1]); - return res; -} - -template<> EIGEN_STRONG_INLINE Packet4i pandnot(const Packet4i& a, const Packet4i& b) { return pand(a, vec_nor(b, b)); } -template<> EIGEN_STRONG_INLINE Packet2d pandnot(const Packet2d& a, const Packet2d& b) { return vec_and(a, vec_nor(b, b)); } -template<> EIGEN_STRONG_INLINE Packet4f pandnot(const Packet4f& a, const Packet4f& b) -{ - Packet4f res; - res.v4f[0] = pandnot(a.v4f[0], b.v4f[0]); - res.v4f[1] = pandnot(a.v4f[1], b.v4f[1]); - return res; -} - -template<> EIGEN_STRONG_INLINE Packet4f pround(const Packet4f& a) -{ - Packet4f res; - res.v4f[0] = vec_round(a.v4f[0]); - res.v4f[1] = vec_round(a.v4f[1]); - return res; -} -template<> EIGEN_STRONG_INLINE Packet2d pround(const Packet2d& a) { return vec_round(a); } -template<> EIGEN_STRONG_INLINE Packet4f pceil(const Packet4f& a) -{ - Packet4f res; - res.v4f[0] = vec_ceil(a.v4f[0]); - res.v4f[1] = vec_ceil(a.v4f[1]); - return res; -} -template<> EIGEN_STRONG_INLINE Packet2d pceil(const Packet2d& a) { return vec_ceil(a); } -template<> EIGEN_STRONG_INLINE Packet4f pfloor(const Packet4f& a) -{ - Packet4f res; - res.v4f[0] = vec_floor(a.v4f[0]); - res.v4f[1] = vec_floor(a.v4f[1]); - return res; -} -template<> EIGEN_STRONG_INLINE Packet2d pfloor(const Packet2d& a) { return vec_floor(a); } - -template<> EIGEN_STRONG_INLINE Packet4i ploadu(const int* from) { return pload(from); } -template<> EIGEN_STRONG_INLINE Packet4f ploadu(const float* from) { return pload(from); } -template<> EIGEN_STRONG_INLINE Packet2d ploadu(const double* from) { return pload(from); } - - -template<> EIGEN_STRONG_INLINE Packet4i ploaddup(const int* from) -{ - Packet4i p = pload(from); - return vec_perm(p, p, p16uc_DUPLICATE32_HI); -} - -template<> EIGEN_STRONG_INLINE Packet4f ploaddup(const float* from) -{ - Packet4f p = pload(from); - p.v4f[1] = vec_splat(p.v4f[0], 1); - p.v4f[0] = vec_splat(p.v4f[0], 0); - return p; -} - -template<> EIGEN_STRONG_INLINE Packet2d ploaddup(const double* from) -{ - Packet2d p = pload(from); - return vec_perm(p, p, p16uc_PSET64_HI); -} - -template<> EIGEN_STRONG_INLINE void pstoreu(int* to, const Packet4i& from) { pstore(to, from); } -template<> EIGEN_STRONG_INLINE void pstoreu(float* to, const Packet4f& from) { pstore(to, from); } -template<> EIGEN_STRONG_INLINE void pstoreu(double* to, const Packet2d& from) { pstore(to, from); } - -template<> EIGEN_STRONG_INLINE void prefetch(const int* addr) { EIGEN_ZVECTOR_PREFETCH(addr); } -template<> EIGEN_STRONG_INLINE void prefetch(const float* addr) { EIGEN_ZVECTOR_PREFETCH(addr); } -template<> EIGEN_STRONG_INLINE void prefetch(const double* addr) { EIGEN_ZVECTOR_PREFETCH(addr); } - -template<> EIGEN_STRONG_INLINE int pfirst(const Packet4i& a) { int EIGEN_ALIGN16 x[4]; pstore(x, a); return x[0]; } -template<> EIGEN_STRONG_INLINE float pfirst(const Packet4f& a) { float EIGEN_ALIGN16 x[2]; vec_st2f(a.v4f[0], &x[0]); return x[0]; } -template<> EIGEN_STRONG_INLINE double pfirst(const Packet2d& a) { double EIGEN_ALIGN16 x[2]; pstore(x, a); return x[0]; } - -template<> EIGEN_STRONG_INLINE Packet4i preverse(const Packet4i& a) -{ - return reinterpret_cast(vec_perm(reinterpret_cast(a), reinterpret_cast(a), p16uc_REVERSE32)); -} - -template<> EIGEN_STRONG_INLINE Packet2d preverse(const Packet2d& a) -{ - return reinterpret_cast(vec_perm(reinterpret_cast(a), reinterpret_cast(a), p16uc_REVERSE64)); -} - -template<> EIGEN_STRONG_INLINE Packet4f preverse(const Packet4f& a) -{ - Packet4f rev; - rev.v4f[0] = preverse(a.v4f[1]); - rev.v4f[1] = preverse(a.v4f[0]); - return rev; -} - -template<> EIGEN_STRONG_INLINE Packet4i pabs(const Packet4i& a) { return vec_abs(a); } -template<> EIGEN_STRONG_INLINE Packet2d pabs(const Packet2d& a) { return vec_abs(a); } -template<> EIGEN_STRONG_INLINE Packet4f pabs(const Packet4f& a) -{ - Packet4f res; - res.v4f[0] = pabs(a.v4f[0]); - res.v4f[1] = pabs(a.v4f[1]); - return res; -} - -template<> EIGEN_STRONG_INLINE int predux(const Packet4i& a) -{ - Packet4i b, sum; - b = vec_sld(a, a, 8); - sum = padd(a, b); - b = vec_sld(sum, sum, 4); - sum = padd(sum, b); - return pfirst(sum); -} - -template<> EIGEN_STRONG_INLINE double predux(const Packet2d& a) -{ - Packet2d b, sum; - b = reinterpret_cast(vec_sld(reinterpret_cast(a), reinterpret_cast(a), 8)); - sum = padd(a, b); - return pfirst(sum); -} -template<> EIGEN_STRONG_INLINE float predux(const Packet4f& a) -{ - Packet2d sum; - sum = padd(a.v4f[0], a.v4f[1]); - double first = predux(sum); - return static_cast(first); -} - -template<> EIGEN_STRONG_INLINE Packet4i preduxp(const Packet4i* vecs) -{ - Packet4i v[4], sum[4]; - - // It's easier and faster to transpose then add as columns - // Check: http://www.freevec.org/function/matrix_4x4_transpose_floats for explanation - // Do the transpose, first set of moves - v[0] = vec_mergeh(vecs[0], vecs[2]); - v[1] = vec_mergel(vecs[0], vecs[2]); - v[2] = vec_mergeh(vecs[1], vecs[3]); - v[3] = vec_mergel(vecs[1], vecs[3]); - // Get the resulting vectors - sum[0] = vec_mergeh(v[0], v[2]); - sum[1] = vec_mergel(v[0], v[2]); - sum[2] = vec_mergeh(v[1], v[3]); - sum[3] = vec_mergel(v[1], v[3]); - - // Now do the summation: - // Lines 0+1 - sum[0] = padd(sum[0], sum[1]); - // Lines 2+3 - sum[1] = padd(sum[2], sum[3]); - // Add the results - sum[0] = padd(sum[0], sum[1]); - - return sum[0]; -} - -template<> EIGEN_STRONG_INLINE Packet2d preduxp(const Packet2d* vecs) -{ - Packet2d v[2], sum; - v[0] = padd(vecs[0], reinterpret_cast(vec_sld(reinterpret_cast(vecs[0]), reinterpret_cast(vecs[0]), 8))); - v[1] = padd(vecs[1], reinterpret_cast(vec_sld(reinterpret_cast(vecs[1]), reinterpret_cast(vecs[1]), 8))); - - sum = reinterpret_cast(vec_sld(reinterpret_cast(v[0]), reinterpret_cast(v[1]), 8)); - - return sum; -} - -template<> EIGEN_STRONG_INLINE Packet4f preduxp(const Packet4f* vecs) -{ - PacketBlock transpose; - transpose.packet[0] = vecs[0]; - transpose.packet[1] = vecs[1]; - transpose.packet[2] = vecs[2]; - transpose.packet[3] = vecs[3]; - ptranspose(transpose); - - Packet4f sum = padd(transpose.packet[0], transpose.packet[1]); - sum = padd(sum, transpose.packet[2]); - sum = padd(sum, transpose.packet[3]); - return sum; -} - -// Other reduction functions: -// mul -template<> EIGEN_STRONG_INLINE int predux_mul(const Packet4i& a) -{ - EIGEN_ALIGN16 int aux[4]; - pstore(aux, a); - return aux[0] * aux[1] * aux[2] * aux[3]; -} - -template<> EIGEN_STRONG_INLINE double predux_mul(const Packet2d& a) -{ - return pfirst(pmul(a, reinterpret_cast(vec_sld(reinterpret_cast(a), reinterpret_cast(a), 8)))); -} - -template<> EIGEN_STRONG_INLINE float predux_mul(const Packet4f& a) -{ - // Return predux_mul of the subvectors product - return static_cast(pfirst(predux_mul(pmul(a.v4f[0], a.v4f[1])))); -} - -// min -template<> EIGEN_STRONG_INLINE int predux_min(const Packet4i& a) -{ - Packet4i b, res; - b = pmin(a, vec_sld(a, a, 8)); - res = pmin(b, vec_sld(b, b, 4)); - return pfirst(res); -} - -template<> EIGEN_STRONG_INLINE double predux_min(const Packet2d& a) -{ - return pfirst(pmin(a, reinterpret_cast(vec_sld(reinterpret_cast(a), reinterpret_cast(a), 8)))); -} - -template<> EIGEN_STRONG_INLINE float predux_min(const Packet4f& a) -{ - Packet2d b, res; - b = pmin(a.v4f[0], a.v4f[1]); - res = pmin(b, reinterpret_cast(vec_sld(reinterpret_cast(b), reinterpret_cast(b), 8))); - return static_cast(pfirst(res)); -} - -// max -template<> EIGEN_STRONG_INLINE int predux_max(const Packet4i& a) -{ - Packet4i b, res; - b = pmax(a, vec_sld(a, a, 8)); - res = pmax(b, vec_sld(b, b, 4)); - return pfirst(res); -} - -// max -template<> EIGEN_STRONG_INLINE double predux_max(const Packet2d& a) -{ - return pfirst(pmax(a, reinterpret_cast(vec_sld(reinterpret_cast(a), reinterpret_cast(a), 8)))); -} - -template<> EIGEN_STRONG_INLINE float predux_max(const Packet4f& a) -{ - Packet2d b, res; - b = pmax(a.v4f[0], a.v4f[1]); - res = pmax(b, reinterpret_cast(vec_sld(reinterpret_cast(b), reinterpret_cast(b), 8))); - return static_cast(pfirst(res)); -} - -EIGEN_DEVICE_FUNC inline void -ptranspose(PacketBlock& kernel) { - Packet4i t0 = vec_mergeh(kernel.packet[0], kernel.packet[2]); - Packet4i t1 = vec_mergel(kernel.packet[0], kernel.packet[2]); - Packet4i t2 = vec_mergeh(kernel.packet[1], kernel.packet[3]); - Packet4i t3 = vec_mergel(kernel.packet[1], kernel.packet[3]); - kernel.packet[0] = vec_mergeh(t0, t2); - kernel.packet[1] = vec_mergel(t0, t2); - kernel.packet[2] = vec_mergeh(t1, t3); - kernel.packet[3] = vec_mergel(t1, t3); -} - -EIGEN_DEVICE_FUNC inline void -ptranspose(PacketBlock& kernel) { - Packet2d t0 = vec_perm(kernel.packet[0], kernel.packet[1], p16uc_TRANSPOSE64_HI); - Packet2d t1 = vec_perm(kernel.packet[0], kernel.packet[1], p16uc_TRANSPOSE64_LO); - kernel.packet[0] = t0; - kernel.packet[1] = t1; -} - -/* Split the Packet4f PacketBlock into 4 Packet2d PacketBlocks and transpose each one - */ -EIGEN_DEVICE_FUNC inline void -ptranspose(PacketBlock& kernel) { - PacketBlock t0,t1,t2,t3; - // copy top-left 2x2 Packet2d block - t0.packet[0] = kernel.packet[0].v4f[0]; - t0.packet[1] = kernel.packet[1].v4f[0]; - - // copy top-right 2x2 Packet2d block - t1.packet[0] = kernel.packet[0].v4f[1]; - t1.packet[1] = kernel.packet[1].v4f[1]; - - // copy bottom-left 2x2 Packet2d block - t2.packet[0] = kernel.packet[2].v4f[0]; - t2.packet[1] = kernel.packet[3].v4f[0]; - - // copy bottom-right 2x2 Packet2d block - t3.packet[0] = kernel.packet[2].v4f[1]; - t3.packet[1] = kernel.packet[3].v4f[1]; - - // Transpose all 2x2 blocks - ptranspose(t0); - ptranspose(t1); - ptranspose(t2); - ptranspose(t3); - - // Copy back transposed blocks, but exchange t1 and t2 due to transposition - kernel.packet[0].v4f[0] = t0.packet[0]; - kernel.packet[0].v4f[1] = t2.packet[0]; - kernel.packet[1].v4f[0] = t0.packet[1]; - kernel.packet[1].v4f[1] = t2.packet[1]; - kernel.packet[2].v4f[0] = t1.packet[0]; - kernel.packet[2].v4f[1] = t3.packet[0]; - kernel.packet[3].v4f[0] = t1.packet[1]; - kernel.packet[3].v4f[1] = t3.packet[1]; -} - -template<> EIGEN_STRONG_INLINE Packet4i pblend(const Selector<4>& ifPacket, const Packet4i& thenPacket, const Packet4i& elsePacket) { - Packet4ui select = { ifPacket.select[0], ifPacket.select[1], ifPacket.select[2], ifPacket.select[3] }; - Packet4ui mask = vec_cmpeq(select, reinterpret_cast(p4i_ONE)); - return vec_sel(elsePacket, thenPacket, mask); -} - -template<> EIGEN_STRONG_INLINE Packet4f pblend(const Selector<4>& ifPacket, const Packet4f& thenPacket, const Packet4f& elsePacket) { - Packet2ul select_hi = { ifPacket.select[0], ifPacket.select[1] }; - Packet2ul select_lo = { ifPacket.select[2], ifPacket.select[3] }; - Packet2ul mask_hi = vec_cmpeq(select_hi, reinterpret_cast(p2l_ONE)); - Packet2ul mask_lo = vec_cmpeq(select_lo, reinterpret_cast(p2l_ONE)); - Packet4f result; - result.v4f[0] = vec_sel(elsePacket.v4f[0], thenPacket.v4f[0], mask_hi); - result.v4f[1] = vec_sel(elsePacket.v4f[1], thenPacket.v4f[1], mask_lo); - return result; -} - -template<> EIGEN_STRONG_INLINE Packet2d pblend(const Selector<2>& ifPacket, const Packet2d& thenPacket, const Packet2d& elsePacket) { - Packet2ul select = { ifPacket.select[0], ifPacket.select[1] }; - Packet2ul mask = vec_cmpeq(select, reinterpret_cast(p2l_ONE)); - return vec_sel(elsePacket, thenPacket, mask); -} - -} // end namespace internal - -} // end namespace Eigen - -#endif // EIGEN_PACKET_MATH_ZVECTOR_H + + + template<> EIGEN_STRONG_INLINE void pstore(double *to, const Packet2d &from) + { + // FIXME: No intrinsic yet + EIGEN_DEBUG_ALIGNED_STORE + Packet *vto; + vto = (Packet *)to; + vto->v2d = from; + } + + template<> EIGEN_STRONG_INLINE Packet4i pset1(const int &from) { return vec_splats(from); } + template<> EIGEN_STRONG_INLINE Packet2d pset1(const double &from) { return vec_splats(from); } + template<> EIGEN_STRONG_INLINE Packet4f pset1(const float &from) + { + Packet4f to; + to.v4f[0] = pset1(static_cast(from)); + to.v4f[1] = to.v4f[0]; + return to; + } + + template<> + EIGEN_STRONG_INLINE void pbroadcast4(const int *a, Packet4i &a0, Packet4i &a1, Packet4i &a2, Packet4i &a3) + { + a3 = pload(a); + a0 = vec_splat(a3, 0); + a1 = vec_splat(a3, 1); + a2 = vec_splat(a3, 2); + a3 = vec_splat(a3, 3); + } + + template<> + EIGEN_STRONG_INLINE void pbroadcast4(const float *a, Packet4f &a0, Packet4f &a1, Packet4f &a2, Packet4f &a3) + { + a3 = pload(a); + a0 = vec_splat_packet4f<0>(a3); + a1 = vec_splat_packet4f<1>(a3); + a2 = vec_splat_packet4f<2>(a3); + a3 = vec_splat_packet4f<3>(a3); + } + + template<> + EIGEN_STRONG_INLINE void + pbroadcast4(const double *a, Packet2d &a0, Packet2d &a1, Packet2d &a2, Packet2d &a3) + { + a1 = pload(a); + a0 = vec_splat(a1, 0); + a1 = vec_splat(a1, 1); + a3 = pload(a + 2); + a2 = vec_splat(a3, 0); + a3 = vec_splat(a3, 1); + } + + template<> EIGEN_DEVICE_FUNC inline Packet4i pgather(const int *from, Index stride) + { + int EIGEN_ALIGN16 ai[4]; + ai[0] = from[0 * stride]; + ai[1] = from[1 * stride]; + ai[2] = from[2 * stride]; + ai[3] = from[3 * stride]; + return pload(ai); + } + + template<> EIGEN_DEVICE_FUNC inline Packet4f pgather(const float *from, Index stride) + { + float EIGEN_ALIGN16 ai[4]; + ai[0] = from[0 * stride]; + ai[1] = from[1 * stride]; + ai[2] = from[2 * stride]; + ai[3] = from[3 * stride]; + return pload(ai); + } + + template<> EIGEN_DEVICE_FUNC inline Packet2d pgather(const double *from, Index stride) + { + double EIGEN_ALIGN16 af[2]; + af[0] = from[0 * stride]; + af[1] = from[1 * stride]; + return pload(af); + } + + template<> EIGEN_DEVICE_FUNC inline void pscatter(int *to, const Packet4i &from, Index stride) + { + int EIGEN_ALIGN16 ai[4]; + pstore((int *)ai, from); + to[0 * stride] = ai[0]; + to[1 * stride] = ai[1]; + to[2 * stride] = ai[2]; + to[3 * stride] = ai[3]; + } + + template<> EIGEN_DEVICE_FUNC inline void pscatter(float *to, const Packet4f &from, Index stride) + { + float EIGEN_ALIGN16 ai[4]; + pstore((float *)ai, from); + to[0 * stride] = ai[0]; + to[1 * stride] = ai[1]; + to[2 * stride] = ai[2]; + to[3 * stride] = ai[3]; + } + + template<> EIGEN_DEVICE_FUNC inline void pscatter(double *to, const Packet2d &from, Index stride) + { + double EIGEN_ALIGN16 af[2]; + pstore(af, from); + to[0 * stride] = af[0]; + to[1 * stride] = af[1]; + } + + template<> EIGEN_STRONG_INLINE Packet4i padd(const Packet4i &a, const Packet4i &b) { return (a + b); } + template<> EIGEN_STRONG_INLINE Packet4f padd(const Packet4f &a, const Packet4f &b) + { + Packet4f c; + c.v4f[0] = a.v4f[0] + b.v4f[0]; + c.v4f[1] = a.v4f[1] + b.v4f[1]; + return c; + } + template<> EIGEN_STRONG_INLINE Packet2d padd(const Packet2d &a, const Packet2d &b) { return (a + b); } + + template<> EIGEN_STRONG_INLINE Packet4i psub(const Packet4i &a, const Packet4i &b) { return (a - b); } + template<> EIGEN_STRONG_INLINE Packet4f psub(const Packet4f &a, const Packet4f &b) + { + Packet4f c; + c.v4f[0] = a.v4f[0] - b.v4f[0]; + c.v4f[1] = a.v4f[1] - b.v4f[1]; + return c; + } + template<> EIGEN_STRONG_INLINE Packet2d psub(const Packet2d &a, const Packet2d &b) { return (a - b); } + + template<> EIGEN_STRONG_INLINE Packet4i pmul(const Packet4i &a, const Packet4i &b) { return (a * b); } + template<> EIGEN_STRONG_INLINE Packet4f pmul(const Packet4f &a, const Packet4f &b) + { + Packet4f c; + c.v4f[0] = a.v4f[0] * b.v4f[0]; + c.v4f[1] = a.v4f[1] * b.v4f[1]; + return c; + } + template<> EIGEN_STRONG_INLINE Packet2d pmul(const Packet2d &a, const Packet2d &b) { return (a * b); } + + template<> EIGEN_STRONG_INLINE Packet4i pdiv(const Packet4i &a, const Packet4i &b) { return (a / b); } + template<> EIGEN_STRONG_INLINE Packet4f pdiv(const Packet4f &a, const Packet4f &b) + { + Packet4f c; + c.v4f[0] = a.v4f[0] / b.v4f[0]; + c.v4f[1] = a.v4f[1] / b.v4f[1]; + return c; + } + template<> EIGEN_STRONG_INLINE Packet2d pdiv(const Packet2d &a, const Packet2d &b) { return (a / b); } + + template<> EIGEN_STRONG_INLINE Packet4i pnegate(const Packet4i &a) { return (-a); } + template<> EIGEN_STRONG_INLINE Packet4f pnegate(const Packet4f &a) + { + Packet4f c; + c.v4f[0] = -a.v4f[0]; + c.v4f[1] = -a.v4f[1]; + return c; + } + template<> EIGEN_STRONG_INLINE Packet2d pnegate(const Packet2d &a) { return (-a); } + + template<> EIGEN_STRONG_INLINE Packet4i pconj(const Packet4i &a) { return a; } + template<> EIGEN_STRONG_INLINE Packet4f pconj(const Packet4f &a) { return a; } + template<> EIGEN_STRONG_INLINE Packet2d pconj(const Packet2d &a) { return a; } + + template<> EIGEN_STRONG_INLINE Packet4i pmadd(const Packet4i &a, const Packet4i &b, const Packet4i &c) + { + return padd(pmul(a, b), c); + } + template<> EIGEN_STRONG_INLINE Packet4f pmadd(const Packet4f &a, const Packet4f &b, const Packet4f &c) + { + Packet4f res; + res.v4f[0] = vec_madd(a.v4f[0], b.v4f[0], c.v4f[0]); + res.v4f[1] = vec_madd(a.v4f[1], b.v4f[1], c.v4f[1]); + return res; + } + template<> EIGEN_STRONG_INLINE Packet2d pmadd(const Packet2d &a, const Packet2d &b, const Packet2d &c) + { + return vec_madd(a, b, c); + } + + template<> EIGEN_STRONG_INLINE Packet4i plset(const int &a) + { + return padd(pset1(a), p4i_COUNTDOWN); + } + template<> EIGEN_STRONG_INLINE Packet4f plset(const float &a) + { + return padd(pset1(a), p4f_COUNTDOWN); + } + template<> EIGEN_STRONG_INLINE Packet2d plset(const double &a) + { + return padd(pset1(a), p2d_COUNTDOWN); + } + + template<> EIGEN_STRONG_INLINE Packet4i pmin(const Packet4i &a, const Packet4i &b) { return vec_min(a, b); } + template<> EIGEN_STRONG_INLINE Packet2d pmin(const Packet2d &a, const Packet2d &b) { return vec_min(a, b); } + template<> EIGEN_STRONG_INLINE Packet4f pmin(const Packet4f &a, const Packet4f &b) + { + Packet4f res; + res.v4f[0] = pmin(a.v4f[0], b.v4f[0]); + res.v4f[1] = pmin(a.v4f[1], b.v4f[1]); + return res; + } + + template<> EIGEN_STRONG_INLINE Packet4i pmax(const Packet4i &a, const Packet4i &b) { return vec_max(a, b); } + template<> EIGEN_STRONG_INLINE Packet2d pmax(const Packet2d &a, const Packet2d &b) { return vec_max(a, b); } + template<> EIGEN_STRONG_INLINE Packet4f pmax(const Packet4f &a, const Packet4f &b) + { + Packet4f res; + res.v4f[0] = pmax(a.v4f[0], b.v4f[0]); + res.v4f[1] = pmax(a.v4f[1], b.v4f[1]); + return res; + } + + template<> EIGEN_STRONG_INLINE Packet4i pand(const Packet4i &a, const Packet4i &b) { return vec_and(a, b); } + template<> EIGEN_STRONG_INLINE Packet2d pand(const Packet2d &a, const Packet2d &b) { return vec_and(a, b); } + template<> EIGEN_STRONG_INLINE Packet4f pand(const Packet4f &a, const Packet4f &b) + { + Packet4f res; + res.v4f[0] = pand(a.v4f[0], b.v4f[0]); + res.v4f[1] = pand(a.v4f[1], b.v4f[1]); + return res; + } + + template<> EIGEN_STRONG_INLINE Packet4i por(const Packet4i &a, const Packet4i &b) { return vec_or(a, b); } + template<> EIGEN_STRONG_INLINE Packet2d por(const Packet2d &a, const Packet2d &b) { return vec_or(a, b); } + template<> EIGEN_STRONG_INLINE Packet4f por(const Packet4f &a, const Packet4f &b) + { + Packet4f res; + res.v4f[0] = pand(a.v4f[0], b.v4f[0]); + res.v4f[1] = pand(a.v4f[1], b.v4f[1]); + return res; + } + + template<> EIGEN_STRONG_INLINE Packet4i pxor(const Packet4i &a, const Packet4i &b) { return vec_xor(a, b); } + template<> EIGEN_STRONG_INLINE Packet2d pxor(const Packet2d &a, const Packet2d &b) { return vec_xor(a, b); } + template<> EIGEN_STRONG_INLINE Packet4f pxor(const Packet4f &a, const Packet4f &b) + { + Packet4f res; + res.v4f[0] = pand(a.v4f[0], b.v4f[0]); + res.v4f[1] = pand(a.v4f[1], b.v4f[1]); + return res; + } + + template<> EIGEN_STRONG_INLINE Packet4i pandnot(const Packet4i &a, const Packet4i &b) + { + return pand(a, vec_nor(b, b)); + } + template<> EIGEN_STRONG_INLINE Packet2d pandnot(const Packet2d &a, const Packet2d &b) + { + return vec_and(a, vec_nor(b, b)); + } + template<> EIGEN_STRONG_INLINE Packet4f pandnot(const Packet4f &a, const Packet4f &b) + { + Packet4f res; + res.v4f[0] = pandnot(a.v4f[0], b.v4f[0]); + res.v4f[1] = pandnot(a.v4f[1], b.v4f[1]); + return res; + } + + template<> EIGEN_STRONG_INLINE Packet4f pround(const Packet4f &a) + { + Packet4f res; + res.v4f[0] = vec_round(a.v4f[0]); + res.v4f[1] = vec_round(a.v4f[1]); + return res; + } + template<> EIGEN_STRONG_INLINE Packet2d pround(const Packet2d &a) { return vec_round(a); } + template<> EIGEN_STRONG_INLINE Packet4f pceil(const Packet4f &a) + { + Packet4f res; + res.v4f[0] = vec_ceil(a.v4f[0]); + res.v4f[1] = vec_ceil(a.v4f[1]); + return res; + } + template<> EIGEN_STRONG_INLINE Packet2d pceil(const Packet2d &a) { return vec_ceil(a); } + template<> EIGEN_STRONG_INLINE Packet4f pfloor(const Packet4f &a) + { + Packet4f res; + res.v4f[0] = vec_floor(a.v4f[0]); + res.v4f[1] = vec_floor(a.v4f[1]); + return res; + } + template<> EIGEN_STRONG_INLINE Packet2d pfloor(const Packet2d &a) { return vec_floor(a); } + + template<> EIGEN_STRONG_INLINE Packet4i ploadu(const int *from) { return pload(from); } + template<> EIGEN_STRONG_INLINE Packet4f ploadu(const float *from) { return pload(from); } + template<> EIGEN_STRONG_INLINE Packet2d ploadu(const double *from) { return pload(from); } + + + template<> EIGEN_STRONG_INLINE Packet4i ploaddup(const int *from) + { + Packet4i p = pload(from); + return vec_perm(p, p, p16uc_DUPLICATE32_HI); + } + + template<> EIGEN_STRONG_INLINE Packet4f ploaddup(const float *from) + { + Packet4f p = pload(from); + p.v4f[1] = vec_splat(p.v4f[0], 1); + p.v4f[0] = vec_splat(p.v4f[0], 0); + return p; + } + + template<> EIGEN_STRONG_INLINE Packet2d ploaddup(const double *from) + { + Packet2d p = pload(from); + return vec_perm(p, p, p16uc_PSET64_HI); + } + + template<> EIGEN_STRONG_INLINE void pstoreu(int *to, const Packet4i &from) { pstore(to, from); } + template<> EIGEN_STRONG_INLINE void pstoreu(float *to, const Packet4f &from) { pstore(to, from); } + template<> EIGEN_STRONG_INLINE void pstoreu(double *to, const Packet2d &from) { pstore(to, from); } + + template<> EIGEN_STRONG_INLINE void prefetch(const int *addr) { EIGEN_ZVECTOR_PREFETCH(addr); } + template<> EIGEN_STRONG_INLINE void prefetch(const float *addr) { EIGEN_ZVECTOR_PREFETCH(addr); } + template<> EIGEN_STRONG_INLINE void prefetch(const double *addr) { EIGEN_ZVECTOR_PREFETCH(addr); } + + template<> EIGEN_STRONG_INLINE int pfirst(const Packet4i &a) + { + int EIGEN_ALIGN16 x[4]; + pstore(x, a); + return x[0]; + } + template<> EIGEN_STRONG_INLINE float pfirst(const Packet4f &a) + { + float EIGEN_ALIGN16 x[2]; + vec_st2f(a.v4f[0], &x[0]); + return x[0]; + } + template<> EIGEN_STRONG_INLINE double pfirst(const Packet2d &a) + { + double EIGEN_ALIGN16 x[2]; + pstore(x, a); + return x[0]; + } + + template<> EIGEN_STRONG_INLINE Packet4i preverse(const Packet4i &a) + { + return reinterpret_cast( + vec_perm(reinterpret_cast(a), reinterpret_cast(a), p16uc_REVERSE32)); + } + + template<> EIGEN_STRONG_INLINE Packet2d preverse(const Packet2d &a) + { + return reinterpret_cast( + vec_perm(reinterpret_cast(a), reinterpret_cast(a), p16uc_REVERSE64)); + } + + template<> EIGEN_STRONG_INLINE Packet4f preverse(const Packet4f &a) + { + Packet4f rev; + rev.v4f[0] = preverse(a.v4f[1]); + rev.v4f[1] = preverse(a.v4f[0]); + return rev; + } + + template<> EIGEN_STRONG_INLINE Packet4i pabs(const Packet4i &a) { return vec_abs(a); } + template<> EIGEN_STRONG_INLINE Packet2d pabs(const Packet2d &a) { return vec_abs(a); } + template<> EIGEN_STRONG_INLINE Packet4f pabs(const Packet4f &a) + { + Packet4f res; + res.v4f[0] = pabs(a.v4f[0]); + res.v4f[1] = pabs(a.v4f[1]); + return res; + } + + template<> EIGEN_STRONG_INLINE int predux(const Packet4i &a) + { + Packet4i b, sum; + b = vec_sld(a, a, 8); + sum = padd(a, b); + b = vec_sld(sum, sum, 4); + sum = padd(sum, b); + return pfirst(sum); + } + + template<> EIGEN_STRONG_INLINE double predux(const Packet2d &a) + { + Packet2d b, sum; + b = reinterpret_cast(vec_sld(reinterpret_cast(a), reinterpret_cast(a), 8)); + sum = padd(a, b); + return pfirst(sum); + } + template<> EIGEN_STRONG_INLINE float predux(const Packet4f &a) + { + Packet2d sum; + sum = padd(a.v4f[0], a.v4f[1]); + double first = predux(sum); + return static_cast(first); + } + + template<> EIGEN_STRONG_INLINE Packet4i preduxp(const Packet4i *vecs) + { + Packet4i v[4], sum[4]; + + // It's easier and faster to transpose then add as columns + // Check: http://www.freevec.org/function/matrix_4x4_transpose_floats for explanation + // Do the transpose, first set of moves + v[0] = vec_mergeh(vecs[0], vecs[2]); + v[1] = vec_mergel(vecs[0], vecs[2]); + v[2] = vec_mergeh(vecs[1], vecs[3]); + v[3] = vec_mergel(vecs[1], vecs[3]); + // Get the resulting vectors + sum[0] = vec_mergeh(v[0], v[2]); + sum[1] = vec_mergel(v[0], v[2]); + sum[2] = vec_mergeh(v[1], v[3]); + sum[3] = vec_mergel(v[1], v[3]); + + // Now do the summation: + // Lines 0+1 + sum[0] = padd(sum[0], sum[1]); + // Lines 2+3 + sum[1] = padd(sum[2], sum[3]); + // Add the results + sum[0] = padd(sum[0], sum[1]); + + return sum[0]; + } + + template<> EIGEN_STRONG_INLINE Packet2d preduxp(const Packet2d *vecs) + { + Packet2d v[2], sum; + v[0] = padd(vecs[0], + reinterpret_cast( + vec_sld(reinterpret_cast(vecs[0]), reinterpret_cast(vecs[0]), 8))); + v[1] = padd(vecs[1], + reinterpret_cast( + vec_sld(reinterpret_cast(vecs[1]), reinterpret_cast(vecs[1]), 8))); + + sum = reinterpret_cast(vec_sld(reinterpret_cast(v[0]), reinterpret_cast(v[1]), 8)); + + return sum; + } + + template<> EIGEN_STRONG_INLINE Packet4f preduxp(const Packet4f *vecs) + { + PacketBlock transpose; + transpose.packet[0] = vecs[0]; + transpose.packet[1] = vecs[1]; + transpose.packet[2] = vecs[2]; + transpose.packet[3] = vecs[3]; + ptranspose(transpose); + + Packet4f sum = padd(transpose.packet[0], transpose.packet[1]); + sum = padd(sum, transpose.packet[2]); + sum = padd(sum, transpose.packet[3]); + return sum; + } + + // Other reduction functions: + // mul + template<> EIGEN_STRONG_INLINE int predux_mul(const Packet4i &a) + { + EIGEN_ALIGN16 int aux[4]; + pstore(aux, a); + return aux[0] * aux[1] * aux[2] * aux[3]; + } + + template<> EIGEN_STRONG_INLINE double predux_mul(const Packet2d &a) + { + return pfirst( + pmul(a, reinterpret_cast(vec_sld(reinterpret_cast(a), reinterpret_cast(a), 8)))); + } + + template<> EIGEN_STRONG_INLINE float predux_mul(const Packet4f &a) + { + // Return predux_mul of the subvectors product + return static_cast(pfirst(predux_mul(pmul(a.v4f[0], a.v4f[1])))); + } + + // min + template<> EIGEN_STRONG_INLINE int predux_min(const Packet4i &a) + { + Packet4i b, res; + b = pmin(a, vec_sld(a, a, 8)); + res = pmin(b, vec_sld(b, b, 4)); + return pfirst(res); + } + + template<> EIGEN_STRONG_INLINE double predux_min(const Packet2d &a) + { + return pfirst(pmin( + a, reinterpret_cast(vec_sld(reinterpret_cast(a), reinterpret_cast(a), 8)))); + } + + template<> EIGEN_STRONG_INLINE float predux_min(const Packet4f &a) + { + Packet2d b, res; + b = pmin(a.v4f[0], a.v4f[1]); + res = pmin( + b, reinterpret_cast(vec_sld(reinterpret_cast(b), reinterpret_cast(b), 8))); + return static_cast(pfirst(res)); + } + + // max + template<> EIGEN_STRONG_INLINE int predux_max(const Packet4i &a) + { + Packet4i b, res; + b = pmax(a, vec_sld(a, a, 8)); + res = pmax(b, vec_sld(b, b, 4)); + return pfirst(res); + } + + // max + template<> EIGEN_STRONG_INLINE double predux_max(const Packet2d &a) + { + return pfirst(pmax( + a, reinterpret_cast(vec_sld(reinterpret_cast(a), reinterpret_cast(a), 8)))); + } + + template<> EIGEN_STRONG_INLINE float predux_max(const Packet4f &a) + { + Packet2d b, res; + b = pmax(a.v4f[0], a.v4f[1]); + res = pmax( + b, reinterpret_cast(vec_sld(reinterpret_cast(b), reinterpret_cast(b), 8))); + return static_cast(pfirst(res)); + } + + EIGEN_DEVICE_FUNC inline void ptranspose(PacketBlock &kernel) + { + Packet4i t0 = vec_mergeh(kernel.packet[0], kernel.packet[2]); + Packet4i t1 = vec_mergel(kernel.packet[0], kernel.packet[2]); + Packet4i t2 = vec_mergeh(kernel.packet[1], kernel.packet[3]); + Packet4i t3 = vec_mergel(kernel.packet[1], kernel.packet[3]); + kernel.packet[0] = vec_mergeh(t0, t2); + kernel.packet[1] = vec_mergel(t0, t2); + kernel.packet[2] = vec_mergeh(t1, t3); + kernel.packet[3] = vec_mergel(t1, t3); + } + + EIGEN_DEVICE_FUNC inline void ptranspose(PacketBlock &kernel) + { + Packet2d t0 = vec_perm(kernel.packet[0], kernel.packet[1], p16uc_TRANSPOSE64_HI); + Packet2d t1 = vec_perm(kernel.packet[0], kernel.packet[1], p16uc_TRANSPOSE64_LO); + kernel.packet[0] = t0; + kernel.packet[1] = t1; + } + + /* Split the Packet4f PacketBlock into 4 Packet2d PacketBlocks and transpose each one + */ + EIGEN_DEVICE_FUNC inline void ptranspose(PacketBlock &kernel) + { + PacketBlock t0, t1, t2, t3; + // copy top-left 2x2 Packet2d block + t0.packet[0] = kernel.packet[0].v4f[0]; + t0.packet[1] = kernel.packet[1].v4f[0]; + + // copy top-right 2x2 Packet2d block + t1.packet[0] = kernel.packet[0].v4f[1]; + t1.packet[1] = kernel.packet[1].v4f[1]; + + // copy bottom-left 2x2 Packet2d block + t2.packet[0] = kernel.packet[2].v4f[0]; + t2.packet[1] = kernel.packet[3].v4f[0]; + + // copy bottom-right 2x2 Packet2d block + t3.packet[0] = kernel.packet[2].v4f[1]; + t3.packet[1] = kernel.packet[3].v4f[1]; + + // Transpose all 2x2 blocks + ptranspose(t0); + ptranspose(t1); + ptranspose(t2); + ptranspose(t3); + + // Copy back transposed blocks, but exchange t1 and t2 due to transposition + kernel.packet[0].v4f[0] = t0.packet[0]; + kernel.packet[0].v4f[1] = t2.packet[0]; + kernel.packet[1].v4f[0] = t0.packet[1]; + kernel.packet[1].v4f[1] = t2.packet[1]; + kernel.packet[2].v4f[0] = t1.packet[0]; + kernel.packet[2].v4f[1] = t3.packet[0]; + kernel.packet[3].v4f[0] = t1.packet[1]; + kernel.packet[3].v4f[1] = t3.packet[1]; + } + + template<> + EIGEN_STRONG_INLINE Packet4i pblend(const Selector<4> &ifPacket, + const Packet4i &thenPacket, + const Packet4i &elsePacket) + { + Packet4ui select = { ifPacket.select[0], ifPacket.select[1], ifPacket.select[2], ifPacket.select[3] }; + Packet4ui mask = vec_cmpeq(select, reinterpret_cast(p4i_ONE)); + return vec_sel(elsePacket, thenPacket, mask); + } + + template<> + EIGEN_STRONG_INLINE Packet4f pblend(const Selector<4> &ifPacket, + const Packet4f &thenPacket, + const Packet4f &elsePacket) + { + Packet2ul select_hi = { ifPacket.select[0], ifPacket.select[1] }; + Packet2ul select_lo = { ifPacket.select[2], ifPacket.select[3] }; + Packet2ul mask_hi = vec_cmpeq(select_hi, reinterpret_cast(p2l_ONE)); + Packet2ul mask_lo = vec_cmpeq(select_lo, reinterpret_cast(p2l_ONE)); + Packet4f result; + result.v4f[0] = vec_sel(elsePacket.v4f[0], thenPacket.v4f[0], mask_hi); + result.v4f[1] = vec_sel(elsePacket.v4f[1], thenPacket.v4f[1], mask_lo); + return result; + } + + template<> + EIGEN_STRONG_INLINE Packet2d pblend(const Selector<2> &ifPacket, + const Packet2d &thenPacket, + const Packet2d &elsePacket) + { + Packet2ul select = { ifPacket.select[0], ifPacket.select[1] }; + Packet2ul mask = vec_cmpeq(select, reinterpret_cast(p2l_ONE)); + return vec_sel(elsePacket, thenPacket, mask); + } + +}// end namespace internal + +}// end namespace Eigen + +#endif// EIGEN_PACKET_MATH_ZVECTOR_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/functors/AssignmentFunctors.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/functors/AssignmentFunctors.h index 4153b877..9115b4ae 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/functors/AssignmentFunctors.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/functors/AssignmentFunctors.h @@ -13,156 +13,168 @@ namespace Eigen { namespace internal { - -/** \internal - * \brief Template functor for scalar/packet assignment - * - */ -template struct assign_op { - - EIGEN_EMPTY_STRUCT_CTOR(assign_op) - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void assignCoeff(DstScalar& a, const SrcScalar& b) const { a = b; } - - template - EIGEN_STRONG_INLINE void assignPacket(DstScalar* a, const Packet& b) const - { internal::pstoret(a,b); } -}; - -// Empty overload for void type (used by PermutationMatrix) -template struct assign_op {}; - -template -struct functor_traits > { - enum { - Cost = NumTraits::ReadCost, - PacketAccess = is_same::value && packet_traits::Vectorizable && packet_traits::Vectorizable + + /** \internal + * \brief Template functor for scalar/packet assignment + * + */ + template struct assign_op + { + + EIGEN_EMPTY_STRUCT_CTOR(assign_op) + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void assignCoeff(DstScalar &a, const SrcScalar &b) const { a = b; } + + template EIGEN_STRONG_INLINE void assignPacket(DstScalar *a, const Packet &b) const + { + internal::pstoret(a, b); + } + }; + + // Empty overload for void type (used by PermutationMatrix) + template struct assign_op + { + }; + + template struct functor_traits> + { + enum { + Cost = NumTraits::ReadCost, + PacketAccess = is_same::value && packet_traits::Vectorizable + && packet_traits::Vectorizable + }; + }; + + /** \internal + * \brief Template functor for scalar/packet assignment with addition + * + */ + template struct add_assign_op + { + + EIGEN_EMPTY_STRUCT_CTOR(add_assign_op) + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void assignCoeff(DstScalar &a, const SrcScalar &b) const { a += b; } + + template EIGEN_STRONG_INLINE void assignPacket(DstScalar *a, const Packet &b) const + { + internal::pstoret(a, internal::padd(internal::ploadt(a), b)); + } + }; + template struct functor_traits> + { + enum { + Cost = NumTraits::ReadCost + NumTraits::AddCost, + PacketAccess = is_same::value && packet_traits::HasAdd + }; + }; + + /** \internal + * \brief Template functor for scalar/packet assignment with subtraction + * + */ + template struct sub_assign_op + { + + EIGEN_EMPTY_STRUCT_CTOR(sub_assign_op) + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void assignCoeff(DstScalar &a, const SrcScalar &b) const { a -= b; } + + template EIGEN_STRONG_INLINE void assignPacket(DstScalar *a, const Packet &b) const + { + internal::pstoret(a, internal::psub(internal::ploadt(a), b)); + } }; -}; - -/** \internal - * \brief Template functor for scalar/packet assignment with addition - * - */ -template struct add_assign_op { - - EIGEN_EMPTY_STRUCT_CTOR(add_assign_op) - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void assignCoeff(DstScalar& a, const SrcScalar& b) const { a += b; } - - template - EIGEN_STRONG_INLINE void assignPacket(DstScalar* a, const Packet& b) const - { internal::pstoret(a,internal::padd(internal::ploadt(a),b)); } -}; -template -struct functor_traits > { - enum { - Cost = NumTraits::ReadCost + NumTraits::AddCost, - PacketAccess = is_same::value && packet_traits::HasAdd + template struct functor_traits> + { + enum { + Cost = NumTraits::ReadCost + NumTraits::AddCost, + PacketAccess = is_same::value && packet_traits::HasSub + }; }; -}; - -/** \internal - * \brief Template functor for scalar/packet assignment with subtraction - * - */ -template struct sub_assign_op { - - EIGEN_EMPTY_STRUCT_CTOR(sub_assign_op) - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void assignCoeff(DstScalar& a, const SrcScalar& b) const { a -= b; } - - template - EIGEN_STRONG_INLINE void assignPacket(DstScalar* a, const Packet& b) const - { internal::pstoret(a,internal::psub(internal::ploadt(a),b)); } -}; -template -struct functor_traits > { - enum { - Cost = NumTraits::ReadCost + NumTraits::AddCost, - PacketAccess = is_same::value && packet_traits::HasSub + + /** \internal + * \brief Template functor for scalar/packet assignment with multiplication + * + */ + template struct mul_assign_op + { + + EIGEN_EMPTY_STRUCT_CTOR(mul_assign_op) + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void assignCoeff(DstScalar &a, const SrcScalar &b) const { a *= b; } + + template EIGEN_STRONG_INLINE void assignPacket(DstScalar *a, const Packet &b) const + { + internal::pstoret(a, internal::pmul(internal::ploadt(a), b)); + } }; -}; - -/** \internal - * \brief Template functor for scalar/packet assignment with multiplication - * - */ -template -struct mul_assign_op { - - EIGEN_EMPTY_STRUCT_CTOR(mul_assign_op) - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void assignCoeff(DstScalar& a, const SrcScalar& b) const { a *= b; } - - template - EIGEN_STRONG_INLINE void assignPacket(DstScalar* a, const Packet& b) const - { internal::pstoret(a,internal::pmul(internal::ploadt(a),b)); } -}; -template -struct functor_traits > { - enum { - Cost = NumTraits::ReadCost + NumTraits::MulCost, - PacketAccess = is_same::value && packet_traits::HasMul + template struct functor_traits> + { + enum { + Cost = NumTraits::ReadCost + NumTraits::MulCost, + PacketAccess = is_same::value && packet_traits::HasMul + }; }; -}; - -/** \internal - * \brief Template functor for scalar/packet assignment with diviving - * - */ -template struct div_assign_op { - - EIGEN_EMPTY_STRUCT_CTOR(div_assign_op) - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void assignCoeff(DstScalar& a, const SrcScalar& b) const { a /= b; } - - template - EIGEN_STRONG_INLINE void assignPacket(DstScalar* a, const Packet& b) const - { internal::pstoret(a,internal::pdiv(internal::ploadt(a),b)); } -}; -template -struct functor_traits > { - enum { - Cost = NumTraits::ReadCost + NumTraits::MulCost, - PacketAccess = is_same::value && packet_traits::HasDiv + + /** \internal + * \brief Template functor for scalar/packet assignment with diviving + * + */ + template struct div_assign_op + { + + EIGEN_EMPTY_STRUCT_CTOR(div_assign_op) + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void assignCoeff(DstScalar &a, const SrcScalar &b) const { a /= b; } + + template EIGEN_STRONG_INLINE void assignPacket(DstScalar *a, const Packet &b) const + { + internal::pstoret(a, internal::pdiv(internal::ploadt(a), b)); + } }; -}; - -/** \internal - * \brief Template functor for scalar/packet assignment with swapping - * - * It works as follow. For a non-vectorized evaluation loop, we have: - * for(i) func(A.coeffRef(i), B.coeff(i)); - * where B is a SwapWrapper expression. The trick is to make SwapWrapper::coeff behaves like a non-const coeffRef. - * Actually, SwapWrapper might not even be needed since even if B is a plain expression, since it has to be writable - * B.coeff already returns a const reference to the underlying scalar value. - * - * The case of a vectorized loop is more tricky: - * for(i,j) func.assignPacket(&A.coeffRef(i,j), B.packet(i,j)); - * Here, B must be a SwapWrapper whose packet function actually returns a proxy object holding a Scalar*, - * the actual alignment and Packet type. - * - */ -template struct swap_assign_op { - - EIGEN_EMPTY_STRUCT_CTOR(swap_assign_op) - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void assignCoeff(Scalar& a, const Scalar& b) const + template struct functor_traits> { + enum { + Cost = NumTraits::ReadCost + NumTraits::MulCost, + PacketAccess = is_same::value && packet_traits::HasDiv + }; + }; + + /** \internal + * \brief Template functor for scalar/packet assignment with swapping + * + * It works as follow. For a non-vectorized evaluation loop, we have: + * for(i) func(A.coeffRef(i), B.coeff(i)); + * where B is a SwapWrapper expression. The trick is to make SwapWrapper::coeff behaves like a non-const coeffRef. + * Actually, SwapWrapper might not even be needed since even if B is a plain expression, since it has to be writable + * B.coeff already returns a const reference to the underlying scalar value. + * + * The case of a vectorized loop is more tricky: + * for(i,j) func.assignPacket(&A.coeffRef(i,j), B.packet(i,j)); + * Here, B must be a SwapWrapper whose packet function actually returns a proxy object holding a Scalar*, + * the actual alignment and Packet type. + * + */ + template struct swap_assign_op + { + + EIGEN_EMPTY_STRUCT_CTOR(swap_assign_op) + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void assignCoeff(Scalar &a, const Scalar &b) const + { #ifdef __CUDACC__ - // FIXME is there some kind of cuda::swap? - Scalar t=b; const_cast(b)=a; a=t; + // FIXME is there some kind of cuda::swap? + Scalar t = b; + const_cast(b) = a; + a = t; #else - using std::swap; - swap(a,const_cast(b)); + using std::swap; + swap(a, const_cast(b)); #endif - } -}; -template -struct functor_traits > { - enum { - Cost = 3 * NumTraits::ReadCost, - PacketAccess = packet_traits::Vectorizable + } + }; + template struct functor_traits> + { + enum { Cost = 3 * NumTraits::ReadCost, PacketAccess = packet_traits::Vectorizable }; }; -}; -} // namespace internal +}// namespace internal -} // namespace Eigen +}// namespace Eigen -#endif // EIGEN_ASSIGNMENT_FUNCTORS_H +#endif// EIGEN_ASSIGNMENT_FUNCTORS_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/functors/BinaryFunctors.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/functors/BinaryFunctors.h index 3eae6b8c..d1bddd5e 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/functors/BinaryFunctors.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/functors/BinaryFunctors.h @@ -14,462 +14,513 @@ namespace Eigen { namespace internal { -//---------- associative binary functors ---------- - -template -struct binary_op_base -{ - typedef Arg1 first_argument_type; - typedef Arg2 second_argument_type; -}; - -/** \internal - * \brief Template functor to compute the sum of two scalars - * - * \sa class CwiseBinaryOp, MatrixBase::operator+, class VectorwiseOp, DenseBase::sum() - */ -template -struct scalar_sum_op : binary_op_base -{ - typedef typename ScalarBinaryOpTraits::ReturnType result_type; + //---------- associative binary functors ---------- + + template struct binary_op_base + { + typedef Arg1 first_argument_type; + typedef Arg2 second_argument_type; + }; + + /** \internal + * \brief Template functor to compute the sum of two scalars + * + * \sa class CwiseBinaryOp, MatrixBase::operator+, class VectorwiseOp, DenseBase::sum() + */ + template struct scalar_sum_op : binary_op_base + { + typedef typename ScalarBinaryOpTraits::ReturnType result_type; #ifndef EIGEN_SCALAR_BINARY_OP_PLUGIN - EIGEN_EMPTY_STRUCT_CTOR(scalar_sum_op) + EIGEN_EMPTY_STRUCT_CTOR(scalar_sum_op) #else - scalar_sum_op() { - EIGEN_SCALAR_BINARY_OP_PLUGIN - } + scalar_sum_op(){ EIGEN_SCALAR_BINARY_OP_PLUGIN } #endif - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const result_type operator() (const LhsScalar& a, const RhsScalar& b) const { return a + b; } - template - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const Packet packetOp(const Packet& a, const Packet& b) const - { return internal::padd(a,b); } - template - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const result_type predux(const Packet& a) const - { return internal::predux(a); } -}; -template -struct functor_traits > { - enum { - Cost = (NumTraits::AddCost+NumTraits::AddCost)/2, // rough estimate! - PacketAccess = is_same::value && packet_traits::HasAdd && packet_traits::HasAdd - // TODO vectorize mixed sum - }; -}; - -/** \internal - * \brief Template specialization to deprecate the summation of boolean expressions. - * This is required to solve Bug 426. - * \sa DenseBase::count(), DenseBase::any(), ArrayBase::cast(), MatrixBase::cast() - */ -template<> struct scalar_sum_op : scalar_sum_op { - EIGEN_DEPRECATED - scalar_sum_op() {} -}; - - -/** \internal - * \brief Template functor to compute the product of two scalars - * - * \sa class CwiseBinaryOp, Cwise::operator*(), class VectorwiseOp, MatrixBase::redux() - */ -template -struct scalar_product_op : binary_op_base -{ - typedef typename ScalarBinaryOpTraits::ReturnType result_type; + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const result_type operator()(const LhsScalar &a, const RhsScalar &b) const + { + return a + b; + } + template + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const Packet packetOp(const Packet &a, const Packet &b) const + { + return internal::padd(a, b); + } + template EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const result_type predux(const Packet &a) const + { + return internal::predux(a); + } + }; + template struct functor_traits> + { + enum { + Cost = (NumTraits::AddCost + NumTraits::AddCost) / 2,// rough estimate! + PacketAccess = + is_same::value && packet_traits::HasAdd && packet_traits::HasAdd + // TODO vectorize mixed sum + }; + }; + + /** \internal + * \brief Template specialization to deprecate the summation of boolean expressions. + * This is required to solve Bug 426. + * \sa DenseBase::count(), DenseBase::any(), ArrayBase::cast(), MatrixBase::cast() + */ + template<> struct scalar_sum_op : scalar_sum_op + { + EIGEN_DEPRECATED + scalar_sum_op() {} + }; + + + /** \internal + * \brief Template functor to compute the product of two scalars + * + * \sa class CwiseBinaryOp, Cwise::operator*(), class VectorwiseOp, MatrixBase::redux() + */ + template struct scalar_product_op : binary_op_base + { + typedef typename ScalarBinaryOpTraits::ReturnType result_type; #ifndef EIGEN_SCALAR_BINARY_OP_PLUGIN - EIGEN_EMPTY_STRUCT_CTOR(scalar_product_op) + EIGEN_EMPTY_STRUCT_CTOR(scalar_product_op) #else - scalar_product_op() { - EIGEN_SCALAR_BINARY_OP_PLUGIN - } + scalar_product_op(){ EIGEN_SCALAR_BINARY_OP_PLUGIN } #endif - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const result_type operator() (const LhsScalar& a, const RhsScalar& b) const { return a * b; } - template - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const Packet packetOp(const Packet& a, const Packet& b) const - { return internal::pmul(a,b); } - template - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const result_type predux(const Packet& a) const - { return internal::predux_mul(a); } -}; -template -struct functor_traits > { - enum { - Cost = (NumTraits::MulCost + NumTraits::MulCost)/2, // rough estimate! - PacketAccess = is_same::value && packet_traits::HasMul && packet_traits::HasMul - // TODO vectorize mixed product - }; -}; - -/** \internal - * \brief Template functor to compute the conjugate product of two scalars - * - * This is a short cut for conj(x) * y which is needed for optimization purpose; in Eigen2 support mode, this becomes x * conj(y) - */ -template -struct scalar_conj_product_op : binary_op_base -{ - - enum { - Conj = NumTraits::IsComplex - }; - - typedef typename ScalarBinaryOpTraits::ReturnType result_type; - - EIGEN_EMPTY_STRUCT_CTOR(scalar_conj_product_op) - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const result_type operator() (const LhsScalar& a, const RhsScalar& b) const - { return conj_helper().pmul(a,b); } - - template - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const Packet packetOp(const Packet& a, const Packet& b) const - { return conj_helper().pmul(a,b); } -}; -template -struct functor_traits > { - enum { - Cost = NumTraits::MulCost, - PacketAccess = internal::is_same::value && packet_traits::HasMul - }; -}; - -/** \internal - * \brief Template functor to compute the min of two scalars - * - * \sa class CwiseBinaryOp, MatrixBase::cwiseMin, class VectorwiseOp, MatrixBase::minCoeff() - */ -template -struct scalar_min_op : binary_op_base -{ - typedef typename ScalarBinaryOpTraits::ReturnType result_type; - EIGEN_EMPTY_STRUCT_CTOR(scalar_min_op) - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const result_type operator() (const LhsScalar& a, const RhsScalar& b) const { return numext::mini(a, b); } - template - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const Packet packetOp(const Packet& a, const Packet& b) const - { return internal::pmin(a,b); } - template - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const result_type predux(const Packet& a) const - { return internal::predux_min(a); } -}; -template -struct functor_traits > { - enum { - Cost = (NumTraits::AddCost+NumTraits::AddCost)/2, - PacketAccess = internal::is_same::value && packet_traits::HasMin - }; -}; - -/** \internal - * \brief Template functor to compute the max of two scalars - * - * \sa class CwiseBinaryOp, MatrixBase::cwiseMax, class VectorwiseOp, MatrixBase::maxCoeff() - */ -template -struct scalar_max_op : binary_op_base -{ - typedef typename ScalarBinaryOpTraits::ReturnType result_type; - EIGEN_EMPTY_STRUCT_CTOR(scalar_max_op) - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const result_type operator() (const LhsScalar& a, const RhsScalar& b) const { return numext::maxi(a, b); } - template - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const Packet packetOp(const Packet& a, const Packet& b) const - { return internal::pmax(a,b); } - template - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const result_type predux(const Packet& a) const - { return internal::predux_max(a); } -}; -template -struct functor_traits > { - enum { - Cost = (NumTraits::AddCost+NumTraits::AddCost)/2, - PacketAccess = internal::is_same::value && packet_traits::HasMax - }; -}; - -/** \internal - * \brief Template functors for comparison of two scalars - * \todo Implement packet-comparisons - */ -template struct scalar_cmp_op; - -template -struct functor_traits > { - enum { - Cost = (NumTraits::AddCost+NumTraits::AddCost)/2, - PacketAccess = false - }; -}; - -template -struct result_of(LhsScalar,RhsScalar)> { - typedef bool type; -}; - - -template -struct scalar_cmp_op : binary_op_base -{ - typedef bool result_type; - EIGEN_EMPTY_STRUCT_CTOR(scalar_cmp_op) - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE bool operator()(const LhsScalar& a, const RhsScalar& b) const {return a==b;} -}; -template -struct scalar_cmp_op : binary_op_base -{ - typedef bool result_type; - EIGEN_EMPTY_STRUCT_CTOR(scalar_cmp_op) - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE bool operator()(const LhsScalar& a, const RhsScalar& b) const {return a -struct scalar_cmp_op : binary_op_base -{ - typedef bool result_type; - EIGEN_EMPTY_STRUCT_CTOR(scalar_cmp_op) - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE bool operator()(const LhsScalar& a, const RhsScalar& b) const {return a<=b;} -}; -template -struct scalar_cmp_op : binary_op_base -{ - typedef bool result_type; - EIGEN_EMPTY_STRUCT_CTOR(scalar_cmp_op) - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE bool operator()(const LhsScalar& a, const RhsScalar& b) const {return a>b;} -}; -template -struct scalar_cmp_op : binary_op_base -{ - typedef bool result_type; - EIGEN_EMPTY_STRUCT_CTOR(scalar_cmp_op) - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE bool operator()(const LhsScalar& a, const RhsScalar& b) const {return a>=b;} -}; -template -struct scalar_cmp_op : binary_op_base -{ - typedef bool result_type; - EIGEN_EMPTY_STRUCT_CTOR(scalar_cmp_op) - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE bool operator()(const LhsScalar& a, const RhsScalar& b) const {return !(a<=b || b<=a);} -}; -template -struct scalar_cmp_op : binary_op_base -{ - typedef bool result_type; - EIGEN_EMPTY_STRUCT_CTOR(scalar_cmp_op) - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE bool operator()(const LhsScalar& a, const RhsScalar& b) const {return a!=b;} -}; - - -/** \internal - * \brief Template functor to compute the hypot of two \b positive \b and \b real scalars - * - * \sa MatrixBase::stableNorm(), class Redux - */ -template -struct scalar_hypot_op : binary_op_base -{ - EIGEN_EMPTY_STRUCT_CTOR(scalar_hypot_op) - - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const Scalar operator() (const Scalar &x, const Scalar &y) const - { - // This functor is used by hypotNorm only for which it is faster to first apply abs - // on all coefficients prior to reduction through hypot. - // This way we avoid calling abs on positive and real entries, and this also permits - // to seamlessly handle complexes. Otherwise we would have to handle both real and complexes - // through the same functor... - return internal::positive_real_hypot(x,y); - } -}; -template -struct functor_traits > { - enum - { - Cost = 3 * NumTraits::AddCost + - 2 * NumTraits::MulCost + - 2 * scalar_div_cost::value, - PacketAccess = false - }; -}; - -/** \internal - * \brief Template functor to compute the pow of two scalars - */ -template -struct scalar_pow_op : binary_op_base -{ - typedef typename ScalarBinaryOpTraits::ReturnType result_type; + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const result_type operator()(const LhsScalar &a, const RhsScalar &b) const + { + return a * b; + } + template + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const Packet packetOp(const Packet &a, const Packet &b) const + { + return internal::pmul(a, b); + } + template EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const result_type predux(const Packet &a) const + { + return internal::predux_mul(a); + } + }; + template struct functor_traits> + { + enum { + Cost = (NumTraits::MulCost + NumTraits::MulCost) / 2,// rough estimate! + PacketAccess = + is_same::value && packet_traits::HasMul && packet_traits::HasMul + // TODO vectorize mixed product + }; + }; + + /** \internal + * \brief Template functor to compute the conjugate product of two scalars + * + * This is a short cut for conj(x) * y which is needed for optimization purpose; in Eigen2 support mode, this becomes + * x * conj(y) + */ + template struct scalar_conj_product_op : binary_op_base + { + + enum { Conj = NumTraits::IsComplex }; + + typedef typename ScalarBinaryOpTraits::ReturnType result_type; + + EIGEN_EMPTY_STRUCT_CTOR(scalar_conj_product_op) + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const result_type operator()(const LhsScalar &a, const RhsScalar &b) const + { + return conj_helper().pmul(a, b); + } + + template + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const Packet packetOp(const Packet &a, const Packet &b) const + { + return conj_helper().pmul(a, b); + } + }; + template struct functor_traits> + { + enum { + Cost = NumTraits::MulCost, + PacketAccess = internal::is_same::value && packet_traits::HasMul + }; + }; + + /** \internal + * \brief Template functor to compute the min of two scalars + * + * \sa class CwiseBinaryOp, MatrixBase::cwiseMin, class VectorwiseOp, MatrixBase::minCoeff() + */ + template struct scalar_min_op : binary_op_base + { + typedef typename ScalarBinaryOpTraits::ReturnType result_type; + EIGEN_EMPTY_STRUCT_CTOR(scalar_min_op) + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const result_type operator()(const LhsScalar &a, const RhsScalar &b) const + { + return numext::mini(a, b); + } + template + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const Packet packetOp(const Packet &a, const Packet &b) const + { + return internal::pmin(a, b); + } + template EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const result_type predux(const Packet &a) const + { + return internal::predux_min(a); + } + }; + template struct functor_traits> + { + enum { + Cost = (NumTraits::AddCost + NumTraits::AddCost) / 2, + PacketAccess = internal::is_same::value && packet_traits::HasMin + }; + }; + + /** \internal + * \brief Template functor to compute the max of two scalars + * + * \sa class CwiseBinaryOp, MatrixBase::cwiseMax, class VectorwiseOp, MatrixBase::maxCoeff() + */ + template struct scalar_max_op : binary_op_base + { + typedef typename ScalarBinaryOpTraits::ReturnType result_type; + EIGEN_EMPTY_STRUCT_CTOR(scalar_max_op) + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const result_type operator()(const LhsScalar &a, const RhsScalar &b) const + { + return numext::maxi(a, b); + } + template + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const Packet packetOp(const Packet &a, const Packet &b) const + { + return internal::pmax(a, b); + } + template EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const result_type predux(const Packet &a) const + { + return internal::predux_max(a); + } + }; + template struct functor_traits> + { + enum { + Cost = (NumTraits::AddCost + NumTraits::AddCost) / 2, + PacketAccess = internal::is_same::value && packet_traits::HasMax + }; + }; + + /** \internal + * \brief Template functors for comparison of two scalars + * \todo Implement packet-comparisons + */ + template struct scalar_cmp_op; + + template + struct functor_traits> + { + enum { Cost = (NumTraits::AddCost + NumTraits::AddCost) / 2, PacketAccess = false }; + }; + + template + struct result_of(LhsScalar, RhsScalar)> + { + typedef bool type; + }; + + + template + struct scalar_cmp_op : binary_op_base + { + typedef bool result_type; + EIGEN_EMPTY_STRUCT_CTOR(scalar_cmp_op) + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE bool operator()(const LhsScalar &a, const RhsScalar &b) const + { + return a == b; + } + }; + template + struct scalar_cmp_op : binary_op_base + { + typedef bool result_type; + EIGEN_EMPTY_STRUCT_CTOR(scalar_cmp_op) + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE bool operator()(const LhsScalar &a, const RhsScalar &b) const + { + return a < b; + } + }; + template + struct scalar_cmp_op : binary_op_base + { + typedef bool result_type; + EIGEN_EMPTY_STRUCT_CTOR(scalar_cmp_op) + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE bool operator()(const LhsScalar &a, const RhsScalar &b) const + { + return a <= b; + } + }; + template + struct scalar_cmp_op : binary_op_base + { + typedef bool result_type; + EIGEN_EMPTY_STRUCT_CTOR(scalar_cmp_op) + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE bool operator()(const LhsScalar &a, const RhsScalar &b) const + { + return a > b; + } + }; + template + struct scalar_cmp_op : binary_op_base + { + typedef bool result_type; + EIGEN_EMPTY_STRUCT_CTOR(scalar_cmp_op) + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE bool operator()(const LhsScalar &a, const RhsScalar &b) const + { + return a >= b; + } + }; + template + struct scalar_cmp_op : binary_op_base + { + typedef bool result_type; + EIGEN_EMPTY_STRUCT_CTOR(scalar_cmp_op) + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE bool operator()(const LhsScalar &a, const RhsScalar &b) const + { + return !(a <= b || b <= a); + } + }; + template + struct scalar_cmp_op : binary_op_base + { + typedef bool result_type; + EIGEN_EMPTY_STRUCT_CTOR(scalar_cmp_op) + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE bool operator()(const LhsScalar &a, const RhsScalar &b) const + { + return a != b; + } + }; + + + /** \internal + * \brief Template functor to compute the hypot of two \b positive \b and \b real scalars + * + * \sa MatrixBase::stableNorm(), class Redux + */ + template struct scalar_hypot_op : binary_op_base + { + EIGEN_EMPTY_STRUCT_CTOR(scalar_hypot_op) + + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const Scalar operator()(const Scalar &x, const Scalar &y) const + { + // This functor is used by hypotNorm only for which it is faster to first apply abs + // on all coefficients prior to reduction through hypot. + // This way we avoid calling abs on positive and real entries, and this also permits + // to seamlessly handle complexes. Otherwise we would have to handle both real and complexes + // through the same functor... + return internal::positive_real_hypot(x, y); + } + }; + template struct functor_traits> + { + enum { + Cost = + 3 * NumTraits::AddCost + 2 * NumTraits::MulCost + 2 * scalar_div_cost::value, + PacketAccess = false + }; + }; + + /** \internal + * \brief Template functor to compute the pow of two scalars + */ + template struct scalar_pow_op : binary_op_base + { + typedef typename ScalarBinaryOpTraits::ReturnType result_type; #ifndef EIGEN_SCALAR_BINARY_OP_PLUGIN - EIGEN_EMPTY_STRUCT_CTOR(scalar_pow_op) + EIGEN_EMPTY_STRUCT_CTOR(scalar_pow_op) #else - scalar_pow_op() { - typedef Scalar LhsScalar; - typedef Exponent RhsScalar; - EIGEN_SCALAR_BINARY_OP_PLUGIN - } + scalar_pow_op() + { + typedef Scalar LhsScalar; + typedef Exponent RhsScalar; + EIGEN_SCALAR_BINARY_OP_PLUGIN + } #endif - EIGEN_DEVICE_FUNC - inline result_type operator() (const Scalar& a, const Exponent& b) const { return numext::pow(a, b); } -}; -template -struct functor_traits > { - enum { Cost = 5 * NumTraits::MulCost, PacketAccess = false }; -}; - - - -//---------- non associative binary functors ---------- - -/** \internal - * \brief Template functor to compute the difference of two scalars - * - * \sa class CwiseBinaryOp, MatrixBase::operator- - */ -template -struct scalar_difference_op : binary_op_base -{ - typedef typename ScalarBinaryOpTraits::ReturnType result_type; + EIGEN_DEVICE_FUNC + inline result_type operator()(const Scalar &a, const Exponent &b) const { return numext::pow(a, b); } + }; + template struct functor_traits> + { + enum { Cost = 5 * NumTraits::MulCost, PacketAccess = false }; + }; + + + //---------- non associative binary functors ---------- + + /** \internal + * \brief Template functor to compute the difference of two scalars + * + * \sa class CwiseBinaryOp, MatrixBase::operator- + */ + template struct scalar_difference_op : binary_op_base + { + typedef typename ScalarBinaryOpTraits::ReturnType result_type; #ifndef EIGEN_SCALAR_BINARY_OP_PLUGIN - EIGEN_EMPTY_STRUCT_CTOR(scalar_difference_op) + EIGEN_EMPTY_STRUCT_CTOR(scalar_difference_op) #else - scalar_difference_op() { - EIGEN_SCALAR_BINARY_OP_PLUGIN - } + scalar_difference_op(){ EIGEN_SCALAR_BINARY_OP_PLUGIN } #endif - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const result_type operator() (const LhsScalar& a, const RhsScalar& b) const { return a - b; } - template - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const Packet packetOp(const Packet& a, const Packet& b) const - { return internal::psub(a,b); } -}; -template -struct functor_traits > { - enum { - Cost = (NumTraits::AddCost+NumTraits::AddCost)/2, - PacketAccess = is_same::value && packet_traits::HasSub && packet_traits::HasSub - }; -}; - -/** \internal - * \brief Template functor to compute the quotient of two scalars - * - * \sa class CwiseBinaryOp, Cwise::operator/() - */ -template -struct scalar_quotient_op : binary_op_base -{ - typedef typename ScalarBinaryOpTraits::ReturnType result_type; + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const result_type operator()(const LhsScalar &a, const RhsScalar &b) const + { + return a - b; + } + template + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const Packet packetOp(const Packet &a, const Packet &b) const + { + return internal::psub(a, b); + } + }; + template struct functor_traits> + { + enum { + Cost = (NumTraits::AddCost + NumTraits::AddCost) / 2, + PacketAccess = + is_same::value && packet_traits::HasSub && packet_traits::HasSub + }; + }; + + /** \internal + * \brief Template functor to compute the quotient of two scalars + * + * \sa class CwiseBinaryOp, Cwise::operator/() + */ + template struct scalar_quotient_op : binary_op_base + { + typedef typename ScalarBinaryOpTraits::ReturnType result_type; #ifndef EIGEN_SCALAR_BINARY_OP_PLUGIN - EIGEN_EMPTY_STRUCT_CTOR(scalar_quotient_op) + EIGEN_EMPTY_STRUCT_CTOR(scalar_quotient_op) #else - scalar_quotient_op() { - EIGEN_SCALAR_BINARY_OP_PLUGIN - } + scalar_quotient_op(){ EIGEN_SCALAR_BINARY_OP_PLUGIN } #endif - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const result_type operator() (const LhsScalar& a, const RhsScalar& b) const { return a / b; } - template - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const Packet packetOp(const Packet& a, const Packet& b) const - { return internal::pdiv(a,b); } -}; -template -struct functor_traits > { - typedef typename scalar_quotient_op::result_type result_type; - enum { - PacketAccess = is_same::value && packet_traits::HasDiv && packet_traits::HasDiv, - Cost = scalar_div_cost::value + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const result_type operator()(const LhsScalar &a, const RhsScalar &b) const + { + return a / b; + } + template + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const Packet packetOp(const Packet &a, const Packet &b) const + { + return internal::pdiv(a, b); + } + }; + template struct functor_traits> + { + typedef typename scalar_quotient_op::result_type result_type; + enum { + PacketAccess = + is_same::value && packet_traits::HasDiv && packet_traits::HasDiv, + Cost = scalar_div_cost::value + }; }; -}; - -/** \internal - * \brief Template functor to compute the and of two booleans - * - * \sa class CwiseBinaryOp, ArrayBase::operator&& - */ -struct scalar_boolean_and_op { - EIGEN_EMPTY_STRUCT_CTOR(scalar_boolean_and_op) - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE bool operator() (const bool& a, const bool& b) const { return a && b; } -}; -template<> struct functor_traits { - enum { - Cost = NumTraits::AddCost, - PacketAccess = false + /** \internal + * \brief Template functor to compute the and of two booleans + * + * \sa class CwiseBinaryOp, ArrayBase::operator&& + */ + struct scalar_boolean_and_op + { + EIGEN_EMPTY_STRUCT_CTOR(scalar_boolean_and_op) + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE bool operator()(const bool &a, const bool &b) const { return a && b; } }; -}; - -/** \internal - * \brief Template functor to compute the or of two booleans - * - * \sa class CwiseBinaryOp, ArrayBase::operator|| - */ -struct scalar_boolean_or_op { - EIGEN_EMPTY_STRUCT_CTOR(scalar_boolean_or_op) - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE bool operator() (const bool& a, const bool& b) const { return a || b; } -}; -template<> struct functor_traits { - enum { - Cost = NumTraits::AddCost, - PacketAccess = false + template<> struct functor_traits + { + enum { Cost = NumTraits::AddCost, PacketAccess = false }; }; -}; -/** \internal - * \brief Template functor to compute the xor of two booleans - * - * \sa class CwiseBinaryOp, ArrayBase::operator^ - */ -struct scalar_boolean_xor_op { - EIGEN_EMPTY_STRUCT_CTOR(scalar_boolean_xor_op) - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE bool operator() (const bool& a, const bool& b) const { return a ^ b; } -}; -template<> struct functor_traits { - enum { - Cost = NumTraits::AddCost, - PacketAccess = false + /** \internal + * \brief Template functor to compute the or of two booleans + * + * \sa class CwiseBinaryOp, ArrayBase::operator|| + */ + struct scalar_boolean_or_op + { + EIGEN_EMPTY_STRUCT_CTOR(scalar_boolean_or_op) + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE bool operator()(const bool &a, const bool &b) const { return a || b; } + }; + template<> struct functor_traits + { + enum { Cost = NumTraits::AddCost, PacketAccess = false }; }; -}; + /** \internal + * \brief Template functor to compute the xor of two booleans + * + * \sa class CwiseBinaryOp, ArrayBase::operator^ + */ + struct scalar_boolean_xor_op + { + EIGEN_EMPTY_STRUCT_CTOR(scalar_boolean_xor_op) + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE bool operator()(const bool &a, const bool &b) const { return a ^ b; } + }; + template<> struct functor_traits + { + enum { Cost = NumTraits::AddCost, PacketAccess = false }; + }; -//---------- binary functors bound to a constant, thus appearing as a unary functor ---------- + //---------- binary functors bound to a constant, thus appearing as a unary functor ---------- -// The following two classes permits to turn any binary functor into a unary one with one argument bound to a constant value. -// They are analogues to std::binder1st/binder2nd but with the following differences: -// - they are compatible with packetOp -// - they are portable across C++ versions (the std::binder* are deprecated in C++11) -template struct bind1st_op : BinaryOp { + // The following two classes permits to turn any binary functor into a unary one with one argument bound to a constant + // value. They are analogues to std::binder1st/binder2nd but with the following differences: + // - they are compatible with packetOp + // - they are portable across C++ versions (the std::binder* are deprecated in C++11) + template struct bind1st_op : BinaryOp + { - typedef typename BinaryOp::first_argument_type first_argument_type; - typedef typename BinaryOp::second_argument_type second_argument_type; - typedef typename BinaryOp::result_type result_type; + typedef typename BinaryOp::first_argument_type first_argument_type; + typedef typename BinaryOp::second_argument_type second_argument_type; + typedef typename BinaryOp::result_type result_type; - bind1st_op(const first_argument_type &val) : m_value(val) {} + bind1st_op(const first_argument_type &val) : m_value(val) {} - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const result_type operator() (const second_argument_type& b) const { return BinaryOp::operator()(m_value,b); } + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const result_type operator()(const second_argument_type &b) const + { + return BinaryOp::operator()(m_value, b); + } - template - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const Packet packetOp(const Packet& b) const - { return BinaryOp::packetOp(internal::pset1(m_value), b); } + template EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const Packet packetOp(const Packet &b) const + { + return BinaryOp::packetOp(internal::pset1(m_value), b); + } - first_argument_type m_value; -}; -template struct functor_traits > : functor_traits {}; + first_argument_type m_value; + }; + template struct functor_traits> : functor_traits + { + }; -template struct bind2nd_op : BinaryOp { + template struct bind2nd_op : BinaryOp + { - typedef typename BinaryOp::first_argument_type first_argument_type; - typedef typename BinaryOp::second_argument_type second_argument_type; - typedef typename BinaryOp::result_type result_type; + typedef typename BinaryOp::first_argument_type first_argument_type; + typedef typename BinaryOp::second_argument_type second_argument_type; + typedef typename BinaryOp::result_type result_type; - bind2nd_op(const second_argument_type &val) : m_value(val) {} + bind2nd_op(const second_argument_type &val) : m_value(val) {} - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const result_type operator() (const first_argument_type& a) const { return BinaryOp::operator()(a,m_value); } + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const result_type operator()(const first_argument_type &a) const + { + return BinaryOp::operator()(a, m_value); + } - template - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const Packet packetOp(const Packet& a) const - { return BinaryOp::packetOp(a,internal::pset1(m_value)); } + template EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const Packet packetOp(const Packet &a) const + { + return BinaryOp::packetOp(a, internal::pset1(m_value)); + } - second_argument_type m_value; -}; -template struct functor_traits > : functor_traits {}; + second_argument_type m_value; + }; + template struct functor_traits> : functor_traits + { + }; -} // end namespace internal +}// end namespace internal -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_BINARY_FUNCTORS_H +#endif// EIGEN_BINARY_FUNCTORS_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/functors/NullaryFunctors.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/functors/NullaryFunctors.h index b03be026..6a40f5d3 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/functors/NullaryFunctors.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/functors/NullaryFunctors.h @@ -14,175 +14,216 @@ namespace Eigen { namespace internal { -template -struct scalar_constant_op { - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE scalar_constant_op(const scalar_constant_op& other) : m_other(other.m_other) { } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE scalar_constant_op(const Scalar& other) : m_other(other) { } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const Scalar operator() () const { return m_other; } - template - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const PacketType packetOp() const { return internal::pset1(m_other); } - const Scalar m_other; -}; -template -struct functor_traits > -{ enum { Cost = 0 /* as the constant value should be loaded in register only once for the whole expression */, - PacketAccess = packet_traits::Vectorizable, IsRepeatable = true }; }; - -template struct scalar_identity_op { - EIGEN_EMPTY_STRUCT_CTOR(scalar_identity_op) - template - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const Scalar operator() (IndexType row, IndexType col) const { return row==col ? Scalar(1) : Scalar(0); } -}; -template -struct functor_traits > -{ enum { Cost = NumTraits::AddCost, PacketAccess = false, IsRepeatable = true }; }; - -template struct linspaced_op_impl; - -template -struct linspaced_op_impl -{ - linspaced_op_impl(const Scalar& low, const Scalar& high, Index num_steps) : - m_low(low), m_high(high), m_size1(num_steps==1 ? 1 : num_steps-1), m_step(num_steps==1 ? Scalar() : (high-low)/Scalar(num_steps-1)), - m_flip(numext::abs(high) - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const Scalar operator() (IndexType i) const { - typedef typename NumTraits::Real RealScalar; - if(m_flip) - return (i==0)? m_low : (m_high - RealScalar(m_size1-i)*m_step); - else - return (i==m_size1)? m_high : (m_low + RealScalar(i)*m_step); - } - - template - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const Packet packetOp(IndexType i) const - { - // Principle: - // [low, ..., low] + ( [step, ..., step] * ( [i, ..., i] + [0, ..., size] ) ) - if(m_flip) + template struct scalar_constant_op + { + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE scalar_constant_op(const scalar_constant_op &other) : m_other(other.m_other) + {} + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE scalar_constant_op(const Scalar &other) : m_other(other) {} + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const Scalar operator()() const { return m_other; } + template EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const PacketType packetOp() const + { + return internal::pset1(m_other); + } + const Scalar m_other; + }; + template struct functor_traits> + { + enum { + Cost = 0 /* as the constant value should be loaded in register only once for the whole expression */, + PacketAccess = packet_traits::Vectorizable, + IsRepeatable = true + }; + }; + + template struct scalar_identity_op + { + EIGEN_EMPTY_STRUCT_CTOR(scalar_identity_op) + template + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const Scalar operator()(IndexType row, IndexType col) const + { + return row == col ? Scalar(1) : Scalar(0); + } + }; + template struct functor_traits> + { + enum { Cost = NumTraits::AddCost, PacketAccess = false, IsRepeatable = true }; + }; + + template struct linspaced_op_impl; + + template struct linspaced_op_impl + { + linspaced_op_impl(const Scalar &low, const Scalar &high, Index num_steps) + : m_low(low), m_high(high), m_size1(num_steps == 1 ? 1 : num_steps - 1), + m_step(num_steps == 1 ? Scalar() : (high - low) / Scalar(num_steps - 1)), + m_flip(numext::abs(high) < numext::abs(low)) + {} + + template EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const Scalar operator()(IndexType i) const { - Packet pi = plset(Scalar(i-m_size1)); - Packet res = padd(pset1(m_high), pmul(pset1(m_step), pi)); - if(i==0) - res = pinsertfirst(res, m_low); - return res; + typedef typename NumTraits::Real RealScalar; + if (m_flip) + return (i == 0) ? m_low : (m_high - RealScalar(m_size1 - i) * m_step); + else + return (i == m_size1) ? m_high : (m_low + RealScalar(i) * m_step); } - else + + template EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const Packet packetOp(IndexType i) const { - Packet pi = plset(Scalar(i)); - Packet res = padd(pset1(m_low), pmul(pset1(m_step), pi)); - if(i==m_size1-unpacket_traits::size+1) - res = pinsertlast(res, m_high); - return res; + // Principle: + // [low, ..., low] + ( [step, ..., step] * ( [i, ..., i] + [0, ..., size] ) ) + if (m_flip) { + Packet pi = plset(Scalar(i - m_size1)); + Packet res = padd(pset1(m_high), pmul(pset1(m_step), pi)); + if (i == 0) res = pinsertfirst(res, m_low); + return res; + } else { + Packet pi = plset(Scalar(i)); + Packet res = padd(pset1(m_low), pmul(pset1(m_step), pi)); + if (i == m_size1 - unpacket_traits::size + 1) res = pinsertlast(res, m_high); + return res; + } } - } - - const Scalar m_low; - const Scalar m_high; - const Index m_size1; - const Scalar m_step; - const bool m_flip; -}; - -template -struct linspaced_op_impl -{ - linspaced_op_impl(const Scalar& low, const Scalar& high, Index num_steps) : - m_low(low), - m_multiplier((high-low)/convert_index(num_steps<=1 ? 1 : num_steps-1)), - m_divisor(convert_index((high>=low?num_steps:-num_steps)+(high-low))/((numext::abs(high-low)+1)==0?1:(numext::abs(high-low)+1))), - m_use_divisor(num_steps>1 && (numext::abs(high-low)+1) - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE - const Scalar operator() (IndexType i) const - { - if(m_use_divisor) return m_low + convert_index(i)/m_divisor; - else return m_low + convert_index(i)*m_multiplier; - } - - const Scalar m_low; - const Scalar m_multiplier; - const Scalar m_divisor; - const bool m_use_divisor; -}; - -// ----- Linspace functor ---------------------------------------------------------------- - -// Forward declaration (we default to random access which does not really give -// us a speed gain when using packet access but it allows to use the functor in -// nested expressions). -template struct linspaced_op; -template struct functor_traits< linspaced_op > -{ - enum - { - Cost = 1, - PacketAccess = (!NumTraits::IsInteger) && packet_traits::HasSetLinear && packet_traits::HasBlend, - /*&& ((!NumTraits::IsInteger) || packet_traits::HasDiv),*/ // <- vectorization for integer is currently disabled - IsRepeatable = true - }; -}; -template struct linspaced_op -{ - linspaced_op(const Scalar& low, const Scalar& high, Index num_steps) - : impl((num_steps==1 ? high : low),high,num_steps) - {} - - template - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const Scalar operator() (IndexType i) const { return impl(i); } - - template - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const Packet packetOp(IndexType i) const { return impl.packetOp(i); } - - // This proxy object handles the actual required temporaries and the different - // implementations (integer vs. floating point). - const linspaced_op_impl::IsInteger> impl; -}; - -// Linear access is automatically determined from the operator() prototypes available for the given functor. -// If it exposes an operator()(i,j), then we assume the i and j coefficients are required independently -// and linear access is not possible. In all other cases, linear access is enabled. -// Users should not have to deal with this structure. -template struct functor_has_linear_access { enum { ret = !has_binary_operator::value }; }; + + const Scalar m_low; + const Scalar m_high; + const Index m_size1; + const Scalar m_step; + const bool m_flip; + }; + + template struct linspaced_op_impl + { + linspaced_op_impl(const Scalar &low, const Scalar &high, Index num_steps) + : m_low(low), m_multiplier((high - low) / convert_index(num_steps <= 1 ? 1 : num_steps - 1)), + m_divisor(convert_index((high >= low ? num_steps : -num_steps) + (high - low)) + / ((numext::abs(high - low) + 1) == 0 ? 1 : (numext::abs(high - low) + 1))), + m_use_divisor(num_steps > 1 && (numext::abs(high - low) + 1) < num_steps) + {} + + template EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const Scalar operator()(IndexType i) const + { + if (m_use_divisor) + return m_low + convert_index(i) / m_divisor; + else + return m_low + convert_index(i) * m_multiplier; + } + + const Scalar m_low; + const Scalar m_multiplier; + const Scalar m_divisor; + const bool m_use_divisor; + }; + + // ----- Linspace functor ---------------------------------------------------------------- + + // Forward declaration (we default to random access which does not really give + // us a speed gain when using packet access but it allows to use the functor in + // nested expressions). + template struct linspaced_op; + template struct functor_traits> + { + enum { + Cost = 1, + PacketAccess = + (!NumTraits::IsInteger) && packet_traits::HasSetLinear && packet_traits::HasBlend, + /*&& ((!NumTraits::IsInteger) || packet_traits::HasDiv),*/// <- vectorization for integer is + // currently disabled + IsRepeatable = true + }; + }; + template struct linspaced_op + { + linspaced_op(const Scalar &low, const Scalar &high, Index num_steps) + : impl((num_steps == 1 ? high : low), high, num_steps) + {} + + template EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const Scalar operator()(IndexType i) const + { + return impl(i); + } + + template + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const Packet packetOp(IndexType i) const + { + return impl.packetOp(i); + } + + // This proxy object handles the actual required temporaries and the different + // implementations (integer vs. floating point). + const linspaced_op_impl::IsInteger> impl; + }; + + // Linear access is automatically determined from the operator() prototypes available for the given functor. + // If it exposes an operator()(i,j), then we assume the i and j coefficients are required independently + // and linear access is not possible. In all other cases, linear access is enabled. + // Users should not have to deal with this structure. + template struct functor_has_linear_access + { + enum { ret = !has_binary_operator::value }; + }; // For unreliable compilers, let's specialize the has_*ary_operator // helpers so that at least built-in nullary functors work fine. -#if !( (EIGEN_COMP_MSVC>1600) || (EIGEN_GNUC_AT_LEAST(4,8)) || (EIGEN_COMP_ICC>=1600)) -template -struct has_nullary_operator,IndexType> { enum { value = 1}; }; -template -struct has_unary_operator,IndexType> { enum { value = 0}; }; -template -struct has_binary_operator,IndexType> { enum { value = 0}; }; - -template -struct has_nullary_operator,IndexType> { enum { value = 0}; }; -template -struct has_unary_operator,IndexType> { enum { value = 0}; }; -template -struct has_binary_operator,IndexType> { enum { value = 1}; }; - -template -struct has_nullary_operator,IndexType> { enum { value = 0}; }; -template -struct has_unary_operator,IndexType> { enum { value = 1}; }; -template -struct has_binary_operator,IndexType> { enum { value = 0}; }; - -template -struct has_nullary_operator,IndexType> { enum { value = 1}; }; -template -struct has_unary_operator,IndexType> { enum { value = 0}; }; -template -struct has_binary_operator,IndexType> { enum { value = 0}; }; +#if !((EIGEN_COMP_MSVC > 1600) || (EIGEN_GNUC_AT_LEAST(4, 8)) || (EIGEN_COMP_ICC >= 1600)) + template struct has_nullary_operator, IndexType> + { + enum { value = 1 }; + }; + template struct has_unary_operator, IndexType> + { + enum { value = 0 }; + }; + template struct has_binary_operator, IndexType> + { + enum { value = 0 }; + }; + + template struct has_nullary_operator, IndexType> + { + enum { value = 0 }; + }; + template struct has_unary_operator, IndexType> + { + enum { value = 0 }; + }; + template struct has_binary_operator, IndexType> + { + enum { value = 1 }; + }; + + template + struct has_nullary_operator, IndexType> + { + enum { value = 0 }; + }; + template + struct has_unary_operator, IndexType> + { + enum { value = 1 }; + }; + template + struct has_binary_operator, IndexType> + { + enum { value = 0 }; + }; + + template struct has_nullary_operator, IndexType> + { + enum { value = 1 }; + }; + template struct has_unary_operator, IndexType> + { + enum { value = 0 }; + }; + template struct has_binary_operator, IndexType> + { + enum { value = 0 }; + }; #endif -} // end namespace internal +}// end namespace internal -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_NULLARY_FUNCTORS_H +#endif// EIGEN_NULLARY_FUNCTORS_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/functors/StlFunctors.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/functors/StlFunctors.h index 9c1d7585..c5a88655 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/functors/StlFunctors.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/functors/StlFunctors.h @@ -14,114 +14,141 @@ namespace Eigen { namespace internal { -// default functor traits for STL functors: - -template -struct functor_traits > -{ enum { Cost = NumTraits::MulCost, PacketAccess = false }; }; - -template -struct functor_traits > -{ enum { Cost = NumTraits::MulCost, PacketAccess = false }; }; - -template -struct functor_traits > -{ enum { Cost = NumTraits::AddCost, PacketAccess = false }; }; - -template -struct functor_traits > -{ enum { Cost = NumTraits::AddCost, PacketAccess = false }; }; - -template -struct functor_traits > -{ enum { Cost = NumTraits::AddCost, PacketAccess = false }; }; - -template -struct functor_traits > -{ enum { Cost = 1, PacketAccess = false }; }; - -template -struct functor_traits > -{ enum { Cost = 1, PacketAccess = false }; }; - -template -struct functor_traits > -{ enum { Cost = 1, PacketAccess = false }; }; - -template -struct functor_traits > -{ enum { Cost = 1, PacketAccess = false }; }; - -template -struct functor_traits > -{ enum { Cost = 1, PacketAccess = false }; }; - -template -struct functor_traits > -{ enum { Cost = 1, PacketAccess = false }; }; - -template -struct functor_traits > -{ enum { Cost = 1, PacketAccess = false }; }; - -template -struct functor_traits > -{ enum { Cost = 1, PacketAccess = false }; }; - -template -struct functor_traits > -{ enum { Cost = 1, PacketAccess = false }; }; + // default functor traits for STL functors: + + template struct functor_traits> + { + enum { Cost = NumTraits::MulCost, PacketAccess = false }; + }; + + template struct functor_traits> + { + enum { Cost = NumTraits::MulCost, PacketAccess = false }; + }; + + template struct functor_traits> + { + enum { Cost = NumTraits::AddCost, PacketAccess = false }; + }; + + template struct functor_traits> + { + enum { Cost = NumTraits::AddCost, PacketAccess = false }; + }; + + template struct functor_traits> + { + enum { Cost = NumTraits::AddCost, PacketAccess = false }; + }; + + template struct functor_traits> + { + enum { Cost = 1, PacketAccess = false }; + }; + + template struct functor_traits> + { + enum { Cost = 1, PacketAccess = false }; + }; + + template struct functor_traits> + { + enum { Cost = 1, PacketAccess = false }; + }; + + template struct functor_traits> + { + enum { Cost = 1, PacketAccess = false }; + }; + + template struct functor_traits> + { + enum { Cost = 1, PacketAccess = false }; + }; + + template struct functor_traits> + { + enum { Cost = 1, PacketAccess = false }; + }; + + template struct functor_traits> + { + enum { Cost = 1, PacketAccess = false }; + }; + + template struct functor_traits> + { + enum { Cost = 1, PacketAccess = false }; + }; + + template struct functor_traits> + { + enum { Cost = 1, PacketAccess = false }; + }; #if (__cplusplus < 201103L) && (EIGEN_COMP_MSVC <= 1900) -// std::binder* are deprecated since c++11 and will be removed in c++17 -template -struct functor_traits > -{ enum { Cost = functor_traits::Cost, PacketAccess = false }; }; - -template -struct functor_traits > -{ enum { Cost = functor_traits::Cost, PacketAccess = false }; }; + // std::binder* are deprecated since c++11 and will be removed in c++17 + template struct functor_traits> + { + enum { Cost = functor_traits::Cost, PacketAccess = false }; + }; + + template struct functor_traits> + { + enum { Cost = functor_traits::Cost, PacketAccess = false }; + }; #endif #if (__cplusplus < 201703L) && (EIGEN_COMP_MSVC < 1910) -// std::unary_negate is deprecated since c++17 and will be removed in c++20 -template -struct functor_traits > -{ enum { Cost = 1 + functor_traits::Cost, PacketAccess = false }; }; - -// std::binary_negate is deprecated since c++17 and will be removed in c++20 -template -struct functor_traits > -{ enum { Cost = 1 + functor_traits::Cost, PacketAccess = false }; }; + // std::unary_negate is deprecated since c++17 and will be removed in c++20 + template struct functor_traits> + { + enum { Cost = 1 + functor_traits::Cost, PacketAccess = false }; + }; + + // std::binary_negate is deprecated since c++17 and will be removed in c++20 + template struct functor_traits> + { + enum { Cost = 1 + functor_traits::Cost, PacketAccess = false }; + }; #endif #ifdef EIGEN_STDEXT_SUPPORT -template -struct functor_traits > -{ enum { Cost = 0, PacketAccess = false }; }; - -template -struct functor_traits > -{ enum { Cost = 0, PacketAccess = false }; }; - -template -struct functor_traits > > -{ enum { Cost = 0, PacketAccess = false }; }; - -template -struct functor_traits > > -{ enum { Cost = 0, PacketAccess = false }; }; - -template -struct functor_traits > -{ enum { Cost = functor_traits::Cost + functor_traits::Cost, PacketAccess = false }; }; - -template -struct functor_traits > -{ enum { Cost = functor_traits::Cost + functor_traits::Cost + functor_traits::Cost, PacketAccess = false }; }; - -#endif // EIGEN_STDEXT_SUPPORT + template struct functor_traits> + { + enum { Cost = 0, PacketAccess = false }; + }; + + template struct functor_traits> + { + enum { Cost = 0, PacketAccess = false }; + }; + + template struct functor_traits>> + { + enum { Cost = 0, PacketAccess = false }; + }; + + template struct functor_traits>> + { + enum { Cost = 0, PacketAccess = false }; + }; + + template struct functor_traits> + { + enum { Cost = functor_traits::Cost + functor_traits::Cost, PacketAccess = false }; + }; + + template struct functor_traits> + { + enum { + Cost = functor_traits::Cost + functor_traits::Cost + functor_traits::Cost, + PacketAccess = false + }; + }; + +#endif// EIGEN_STDEXT_SUPPORT // allow to add new functors and specializations of functor_traits from outside Eigen. // this macro is really needed because functor_traits must be specialized after it is declared but before it is used... @@ -129,8 +156,8 @@ struct functor_traits > #include EIGEN_FUNCTORS_PLUGIN #endif -} // end namespace internal +}// end namespace internal -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_STL_FUNCTORS_H +#endif// EIGEN_STL_FUNCTORS_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/functors/TernaryFunctors.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/functors/TernaryFunctors.h index b254e96c..0d69a75d 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/functors/TernaryFunctors.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/functors/TernaryFunctors.h @@ -14,12 +14,11 @@ namespace Eigen { namespace internal { -//---------- associative ternary functors ---------- + //---------- associative ternary functors ---------- +}// end namespace internal -} // end namespace internal +}// end namespace Eigen -} // end namespace Eigen - -#endif // EIGEN_TERNARY_FUNCTORS_H +#endif// EIGEN_TERNARY_FUNCTORS_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/functors/UnaryFunctors.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/functors/UnaryFunctors.h index 2e6a00ff..e0237576 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/functors/UnaryFunctors.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/functors/UnaryFunctors.h @@ -14,779 +14,805 @@ namespace Eigen { namespace internal { -/** \internal - * \brief Template functor to compute the opposite of a scalar - * - * \sa class CwiseUnaryOp, MatrixBase::operator- - */ -template struct scalar_opposite_op { - EIGEN_EMPTY_STRUCT_CTOR(scalar_opposite_op) - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const Scalar operator() (const Scalar& a) const { return -a; } - template - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const Packet packetOp(const Packet& a) const - { return internal::pnegate(a); } -}; -template -struct functor_traits > -{ enum { - Cost = NumTraits::AddCost, - PacketAccess = packet_traits::HasNegate }; -}; - -/** \internal - * \brief Template functor to compute the absolute value of a scalar - * - * \sa class CwiseUnaryOp, Cwise::abs - */ -template struct scalar_abs_op { - EIGEN_EMPTY_STRUCT_CTOR(scalar_abs_op) - typedef typename NumTraits::Real result_type; - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const result_type operator() (const Scalar& a) const { return numext::abs(a); } - template - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const Packet packetOp(const Packet& a) const - { return internal::pabs(a); } -}; -template -struct functor_traits > -{ - enum { - Cost = NumTraits::AddCost, - PacketAccess = packet_traits::HasAbs - }; -}; - -/** \internal - * \brief Template functor to compute the score of a scalar, to chose a pivot - * - * \sa class CwiseUnaryOp - */ -template struct scalar_score_coeff_op : scalar_abs_op -{ - typedef void Score_is_abs; -}; -template -struct functor_traits > : functor_traits > {}; - -/* Avoid recomputing abs when we know the score and they are the same. Not a true Eigen functor. */ -template struct abs_knowing_score -{ - EIGEN_EMPTY_STRUCT_CTOR(abs_knowing_score) - typedef typename NumTraits::Real result_type; - template - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const result_type operator() (const Scalar& a, const Score&) const { return numext::abs(a); } -}; -template struct abs_knowing_score::Score_is_abs> -{ - EIGEN_EMPTY_STRUCT_CTOR(abs_knowing_score) - typedef typename NumTraits::Real result_type; - template - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const result_type operator() (const Scal&, const result_type& a) const { return a; } -}; - -/** \internal - * \brief Template functor to compute the squared absolute value of a scalar - * - * \sa class CwiseUnaryOp, Cwise::abs2 - */ -template struct scalar_abs2_op { - EIGEN_EMPTY_STRUCT_CTOR(scalar_abs2_op) - typedef typename NumTraits::Real result_type; - EIGEN_DEVICE_FUNC - EIGEN_STRONG_INLINE const result_type operator() (const Scalar& a) const { return numext::abs2(a); } - template - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const Packet packetOp(const Packet& a) const - { return internal::pmul(a,a); } -}; -template -struct functor_traits > -{ enum { Cost = NumTraits::MulCost, PacketAccess = packet_traits::HasAbs2 }; }; - -/** \internal - * \brief Template functor to compute the conjugate of a complex value - * - * \sa class CwiseUnaryOp, MatrixBase::conjugate() - */ -template struct scalar_conjugate_op { - EIGEN_EMPTY_STRUCT_CTOR(scalar_conjugate_op) - EIGEN_DEVICE_FUNC - EIGEN_STRONG_INLINE const Scalar operator() (const Scalar& a) const { using numext::conj; return conj(a); } - template - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const Packet packetOp(const Packet& a) const { return internal::pconj(a); } -}; -template -struct functor_traits > -{ - enum { - Cost = NumTraits::IsComplex ? NumTraits::AddCost : 0, - PacketAccess = packet_traits::HasConj - }; -}; - -/** \internal - * \brief Template functor to compute the phase angle of a complex - * - * \sa class CwiseUnaryOp, Cwise::arg - */ -template struct scalar_arg_op { - EIGEN_EMPTY_STRUCT_CTOR(scalar_arg_op) - typedef typename NumTraits::Real result_type; - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const result_type operator() (const Scalar& a) const { using numext::arg; return arg(a); } - template - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const Packet packetOp(const Packet& a) const - { return internal::parg(a); } -}; -template -struct functor_traits > -{ - enum { - Cost = NumTraits::IsComplex ? 5 * NumTraits::MulCost : NumTraits::AddCost, - PacketAccess = packet_traits::HasArg - }; -}; -/** \internal - * \brief Template functor to cast a scalar to another type - * - * \sa class CwiseUnaryOp, MatrixBase::cast() - */ -template -struct scalar_cast_op { - EIGEN_EMPTY_STRUCT_CTOR(scalar_cast_op) - typedef NewType result_type; - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const NewType operator() (const Scalar& a) const { return cast(a); } -}; -template -struct functor_traits > -{ enum { Cost = is_same::value ? 0 : NumTraits::AddCost, PacketAccess = false }; }; - -/** \internal - * \brief Template functor to extract the real part of a complex - * - * \sa class CwiseUnaryOp, MatrixBase::real() - */ -template -struct scalar_real_op { - EIGEN_EMPTY_STRUCT_CTOR(scalar_real_op) - typedef typename NumTraits::Real result_type; - EIGEN_DEVICE_FUNC - EIGEN_STRONG_INLINE result_type operator() (const Scalar& a) const { return numext::real(a); } -}; -template -struct functor_traits > -{ enum { Cost = 0, PacketAccess = false }; }; - -/** \internal - * \brief Template functor to extract the imaginary part of a complex - * - * \sa class CwiseUnaryOp, MatrixBase::imag() - */ -template -struct scalar_imag_op { - EIGEN_EMPTY_STRUCT_CTOR(scalar_imag_op) - typedef typename NumTraits::Real result_type; - EIGEN_DEVICE_FUNC - EIGEN_STRONG_INLINE result_type operator() (const Scalar& a) const { return numext::imag(a); } -}; -template -struct functor_traits > -{ enum { Cost = 0, PacketAccess = false }; }; - -/** \internal - * \brief Template functor to extract the real part of a complex as a reference - * - * \sa class CwiseUnaryOp, MatrixBase::real() - */ -template -struct scalar_real_ref_op { - EIGEN_EMPTY_STRUCT_CTOR(scalar_real_ref_op) - typedef typename NumTraits::Real result_type; - EIGEN_DEVICE_FUNC - EIGEN_STRONG_INLINE result_type& operator() (const Scalar& a) const { return numext::real_ref(*const_cast(&a)); } -}; -template -struct functor_traits > -{ enum { Cost = 0, PacketAccess = false }; }; - -/** \internal - * \brief Template functor to extract the imaginary part of a complex as a reference - * - * \sa class CwiseUnaryOp, MatrixBase::imag() - */ -template -struct scalar_imag_ref_op { - EIGEN_EMPTY_STRUCT_CTOR(scalar_imag_ref_op) - typedef typename NumTraits::Real result_type; - EIGEN_DEVICE_FUNC - EIGEN_STRONG_INLINE result_type& operator() (const Scalar& a) const { return numext::imag_ref(*const_cast(&a)); } -}; -template -struct functor_traits > -{ enum { Cost = 0, PacketAccess = false }; }; - -/** \internal - * - * \brief Template functor to compute the exponential of a scalar - * - * \sa class CwiseUnaryOp, Cwise::exp() - */ -template struct scalar_exp_op { - EIGEN_EMPTY_STRUCT_CTOR(scalar_exp_op) - EIGEN_DEVICE_FUNC inline const Scalar operator() (const Scalar& a) const { return numext::exp(a); } - template - EIGEN_DEVICE_FUNC inline Packet packetOp(const Packet& a) const { return internal::pexp(a); } -}; -template -struct functor_traits > { - enum { - PacketAccess = packet_traits::HasExp, + /** \internal + * \brief Template functor to compute the opposite of a scalar + * + * \sa class CwiseUnaryOp, MatrixBase::operator- + */ + template struct scalar_opposite_op + { + EIGEN_EMPTY_STRUCT_CTOR(scalar_opposite_op) + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const Scalar operator()(const Scalar &a) const { return -a; } + template EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const Packet packetOp(const Packet &a) const + { + return internal::pnegate(a); + } + }; + template struct functor_traits> + { + enum { Cost = NumTraits::AddCost, PacketAccess = packet_traits::HasNegate }; + }; + + /** \internal + * \brief Template functor to compute the absolute value of a scalar + * + * \sa class CwiseUnaryOp, Cwise::abs + */ + template struct scalar_abs_op + { + EIGEN_EMPTY_STRUCT_CTOR(scalar_abs_op) + typedef typename NumTraits::Real result_type; + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const result_type operator()(const Scalar &a) const { return numext::abs(a); } + template EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const Packet packetOp(const Packet &a) const + { + return internal::pabs(a); + } + }; + template struct functor_traits> + { + enum { Cost = NumTraits::AddCost, PacketAccess = packet_traits::HasAbs }; + }; + + /** \internal + * \brief Template functor to compute the score of a scalar, to chose a pivot + * + * \sa class CwiseUnaryOp + */ + template struct scalar_score_coeff_op : scalar_abs_op + { + typedef void Score_is_abs; + }; + template struct functor_traits> : functor_traits> + { + }; + + /* Avoid recomputing abs when we know the score and they are the same. Not a true Eigen functor. */ + template struct abs_knowing_score + { + EIGEN_EMPTY_STRUCT_CTOR(abs_knowing_score) + typedef typename NumTraits::Real result_type; + template + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const result_type operator()(const Scalar &a, const Score &) const + { + return numext::abs(a); + } + }; + template struct abs_knowing_score::Score_is_abs> + { + EIGEN_EMPTY_STRUCT_CTOR(abs_knowing_score) + typedef typename NumTraits::Real result_type; + template + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const result_type operator()(const Scal &, const result_type &a) const + { + return a; + } + }; + + /** \internal + * \brief Template functor to compute the squared absolute value of a scalar + * + * \sa class CwiseUnaryOp, Cwise::abs2 + */ + template struct scalar_abs2_op + { + EIGEN_EMPTY_STRUCT_CTOR(scalar_abs2_op) + typedef typename NumTraits::Real result_type; + EIGEN_DEVICE_FUNC + EIGEN_STRONG_INLINE const result_type operator()(const Scalar &a) const { return numext::abs2(a); } + template EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const Packet packetOp(const Packet &a) const + { + return internal::pmul(a, a); + } + }; + template struct functor_traits> + { + enum { Cost = NumTraits::MulCost, PacketAccess = packet_traits::HasAbs2 }; + }; + + /** \internal + * \brief Template functor to compute the conjugate of a complex value + * + * \sa class CwiseUnaryOp, MatrixBase::conjugate() + */ + template struct scalar_conjugate_op + { + EIGEN_EMPTY_STRUCT_CTOR(scalar_conjugate_op) + EIGEN_DEVICE_FUNC + EIGEN_STRONG_INLINE const Scalar operator()(const Scalar &a) const + { + using numext::conj; + return conj(a); + } + template EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const Packet packetOp(const Packet &a) const + { + return internal::pconj(a); + } + }; + template struct functor_traits> + { + enum { + Cost = NumTraits::IsComplex ? NumTraits::AddCost : 0, + PacketAccess = packet_traits::HasConj + }; + }; + + /** \internal + * \brief Template functor to compute the phase angle of a complex + * + * \sa class CwiseUnaryOp, Cwise::arg + */ + template struct scalar_arg_op + { + EIGEN_EMPTY_STRUCT_CTOR(scalar_arg_op) + typedef typename NumTraits::Real result_type; + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const result_type operator()(const Scalar &a) const + { + using numext::arg; + return arg(a); + } + template EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const Packet packetOp(const Packet &a) const + { + return internal::parg(a); + } + }; + template struct functor_traits> + { + enum { + Cost = NumTraits::IsComplex ? 5 * NumTraits::MulCost : NumTraits::AddCost, + PacketAccess = packet_traits::HasArg + }; + }; + /** \internal + * \brief Template functor to cast a scalar to another type + * + * \sa class CwiseUnaryOp, MatrixBase::cast() + */ + template struct scalar_cast_op + { + EIGEN_EMPTY_STRUCT_CTOR(scalar_cast_op) + typedef NewType result_type; + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const NewType operator()(const Scalar &a) const + { + return cast(a); + } + }; + template struct functor_traits> + { + enum { Cost = is_same::value ? 0 : NumTraits::AddCost, PacketAccess = false }; + }; + + /** \internal + * \brief Template functor to extract the real part of a complex + * + * \sa class CwiseUnaryOp, MatrixBase::real() + */ + template struct scalar_real_op + { + EIGEN_EMPTY_STRUCT_CTOR(scalar_real_op) + typedef typename NumTraits::Real result_type; + EIGEN_DEVICE_FUNC + EIGEN_STRONG_INLINE result_type operator()(const Scalar &a) const { return numext::real(a); } + }; + template struct functor_traits> + { + enum { Cost = 0, PacketAccess = false }; + }; + + /** \internal + * \brief Template functor to extract the imaginary part of a complex + * + * \sa class CwiseUnaryOp, MatrixBase::imag() + */ + template struct scalar_imag_op + { + EIGEN_EMPTY_STRUCT_CTOR(scalar_imag_op) + typedef typename NumTraits::Real result_type; + EIGEN_DEVICE_FUNC + EIGEN_STRONG_INLINE result_type operator()(const Scalar &a) const { return numext::imag(a); } + }; + template struct functor_traits> + { + enum { Cost = 0, PacketAccess = false }; + }; + + /** \internal + * \brief Template functor to extract the real part of a complex as a reference + * + * \sa class CwiseUnaryOp, MatrixBase::real() + */ + template struct scalar_real_ref_op + { + EIGEN_EMPTY_STRUCT_CTOR(scalar_real_ref_op) + typedef typename NumTraits::Real result_type; + EIGEN_DEVICE_FUNC + EIGEN_STRONG_INLINE result_type &operator()(const Scalar &a) const + { + return numext::real_ref(*const_cast(&a)); + } + }; + template struct functor_traits> + { + enum { Cost = 0, PacketAccess = false }; + }; + + /** \internal + * \brief Template functor to extract the imaginary part of a complex as a reference + * + * \sa class CwiseUnaryOp, MatrixBase::imag() + */ + template struct scalar_imag_ref_op + { + EIGEN_EMPTY_STRUCT_CTOR(scalar_imag_ref_op) + typedef typename NumTraits::Real result_type; + EIGEN_DEVICE_FUNC + EIGEN_STRONG_INLINE result_type &operator()(const Scalar &a) const + { + return numext::imag_ref(*const_cast(&a)); + } + }; + template struct functor_traits> + { + enum { Cost = 0, PacketAccess = false }; + }; + + /** \internal + * + * \brief Template functor to compute the exponential of a scalar + * + * \sa class CwiseUnaryOp, Cwise::exp() + */ + template struct scalar_exp_op + { + EIGEN_EMPTY_STRUCT_CTOR(scalar_exp_op) + EIGEN_DEVICE_FUNC inline const Scalar operator()(const Scalar &a) const { return numext::exp(a); } + template EIGEN_DEVICE_FUNC inline Packet packetOp(const Packet &a) const + { + return internal::pexp(a); + } + }; + template struct functor_traits> + { + enum { + PacketAccess = packet_traits::HasExp, // The following numbers are based on the AVX implementation. #ifdef EIGEN_VECTORIZE_FMA - // Haswell can issue 2 add/mul/madd per cycle. - Cost = - (sizeof(Scalar) == 4 - // float: 8 pmadd, 4 pmul, 2 padd/psub, 6 other - ? (8 * NumTraits::AddCost + 6 * NumTraits::MulCost) - // double: 7 pmadd, 5 pmul, 3 padd/psub, 1 div, 13 other - : (14 * NumTraits::AddCost + - 6 * NumTraits::MulCost + - scalar_div_cost::HasDiv>::value)) + // Haswell can issue 2 add/mul/madd per cycle. + Cost = (sizeof(Scalar) == 4 + // float: 8 pmadd, 4 pmul, 2 padd/psub, 6 other + ? (8 * NumTraits::AddCost + 6 * NumTraits::MulCost) + // double: 7 pmadd, 5 pmul, 3 padd/psub, 1 div, 13 other + : (14 * NumTraits::AddCost + 6 * NumTraits::MulCost + + scalar_div_cost::HasDiv>::value)) #else - Cost = - (sizeof(Scalar) == 4 - // float: 7 pmadd, 6 pmul, 4 padd/psub, 10 other - ? (21 * NumTraits::AddCost + 13 * NumTraits::MulCost) - // double: 7 pmadd, 5 pmul, 3 padd/psub, 1 div, 13 other - : (23 * NumTraits::AddCost + - 12 * NumTraits::MulCost + - scalar_div_cost::HasDiv>::value)) + Cost = (sizeof(Scalar) == 4 + // float: 7 pmadd, 6 pmul, 4 padd/psub, 10 other + ? (21 * NumTraits::AddCost + 13 * NumTraits::MulCost) + // double: 7 pmadd, 5 pmul, 3 padd/psub, 1 div, 13 other + : (23 * NumTraits::AddCost + 12 * NumTraits::MulCost + + scalar_div_cost::HasDiv>::value)) #endif + }; }; -}; - -/** \internal - * - * \brief Template functor to compute the logarithm of a scalar - * - * \sa class CwiseUnaryOp, ArrayBase::log() - */ -template struct scalar_log_op { - EIGEN_EMPTY_STRUCT_CTOR(scalar_log_op) - EIGEN_DEVICE_FUNC inline const Scalar operator() (const Scalar& a) const { return numext::log(a); } - template - EIGEN_DEVICE_FUNC inline Packet packetOp(const Packet& a) const { return internal::plog(a); } -}; -template -struct functor_traits > { - enum { - PacketAccess = packet_traits::HasLog, - Cost = - (PacketAccess - // The following numbers are based on the AVX implementation. + + /** \internal + * + * \brief Template functor to compute the logarithm of a scalar + * + * \sa class CwiseUnaryOp, ArrayBase::log() + */ + template struct scalar_log_op + { + EIGEN_EMPTY_STRUCT_CTOR(scalar_log_op) + EIGEN_DEVICE_FUNC inline const Scalar operator()(const Scalar &a) const { return numext::log(a); } + template EIGEN_DEVICE_FUNC inline Packet packetOp(const Packet &a) const + { + return internal::plog(a); + } + }; + template struct functor_traits> + { + enum { + PacketAccess = packet_traits::HasLog, + Cost = (PacketAccess + // The following numbers are based on the AVX implementation. #ifdef EIGEN_VECTORIZE_FMA - // 8 pmadd, 6 pmul, 8 padd/psub, 16 other, can issue 2 add/mul/madd per cycle. - ? (20 * NumTraits::AddCost + 7 * NumTraits::MulCost) + // 8 pmadd, 6 pmul, 8 padd/psub, 16 other, can issue 2 add/mul/madd per cycle. + ? (20 * NumTraits::AddCost + 7 * NumTraits::MulCost) #else - // 8 pmadd, 6 pmul, 8 padd/psub, 20 other - ? (36 * NumTraits::AddCost + 14 * NumTraits::MulCost) + // 8 pmadd, 6 pmul, 8 padd/psub, 20 other + ? (36 * NumTraits::AddCost + 14 * NumTraits::MulCost) #endif - // Measured cost of std::log. - : sizeof(Scalar)==4 ? 40 : 85) - }; -}; - -/** \internal - * - * \brief Template functor to compute the logarithm of 1 plus a scalar value - * - * \sa class CwiseUnaryOp, ArrayBase::log1p() - */ -template struct scalar_log1p_op { - EIGEN_EMPTY_STRUCT_CTOR(scalar_log1p_op) - EIGEN_DEVICE_FUNC inline const Scalar operator() (const Scalar& a) const { return numext::log1p(a); } - template - EIGEN_DEVICE_FUNC inline Packet packetOp(const Packet& a) const { return internal::plog1p(a); } -}; -template -struct functor_traits > { - enum { - PacketAccess = packet_traits::HasLog1p, - Cost = functor_traits >::Cost // TODO measure cost of log1p - }; -}; - -/** \internal - * - * \brief Template functor to compute the base-10 logarithm of a scalar - * - * \sa class CwiseUnaryOp, Cwise::log10() - */ -template struct scalar_log10_op { - EIGEN_EMPTY_STRUCT_CTOR(scalar_log10_op) - EIGEN_DEVICE_FUNC inline const Scalar operator() (const Scalar& a) const { EIGEN_USING_STD_MATH(log10) return log10(a); } - template - EIGEN_DEVICE_FUNC inline Packet packetOp(const Packet& a) const { return internal::plog10(a); } -}; -template -struct functor_traits > -{ enum { Cost = 5 * NumTraits::MulCost, PacketAccess = packet_traits::HasLog10 }; }; - -/** \internal - * \brief Template functor to compute the square root of a scalar - * \sa class CwiseUnaryOp, Cwise::sqrt() - */ -template struct scalar_sqrt_op { - EIGEN_EMPTY_STRUCT_CTOR(scalar_sqrt_op) - EIGEN_DEVICE_FUNC inline const Scalar operator() (const Scalar& a) const { return numext::sqrt(a); } - template - EIGEN_DEVICE_FUNC inline Packet packetOp(const Packet& a) const { return internal::psqrt(a); } -}; -template -struct functor_traits > { - enum { + // Measured cost of std::log. + : sizeof(Scalar) == 4 ? 40 : 85) + }; + }; + + /** \internal + * + * \brief Template functor to compute the logarithm of 1 plus a scalar value + * + * \sa class CwiseUnaryOp, ArrayBase::log1p() + */ + template struct scalar_log1p_op + { + EIGEN_EMPTY_STRUCT_CTOR(scalar_log1p_op) + EIGEN_DEVICE_FUNC inline const Scalar operator()(const Scalar &a) const { return numext::log1p(a); } + template EIGEN_DEVICE_FUNC inline Packet packetOp(const Packet &a) const + { + return internal::plog1p(a); + } + }; + template struct functor_traits> + { + enum { + PacketAccess = packet_traits::HasLog1p, + Cost = functor_traits>::Cost// TODO measure cost of log1p + }; + }; + + /** \internal + * + * \brief Template functor to compute the base-10 logarithm of a scalar + * + * \sa class CwiseUnaryOp, Cwise::log10() + */ + template struct scalar_log10_op + { + EIGEN_EMPTY_STRUCT_CTOR(scalar_log10_op) + EIGEN_DEVICE_FUNC inline const Scalar operator()(const Scalar &a) const + { + EIGEN_USING_STD_MATH(log10) return log10(a); + } + template EIGEN_DEVICE_FUNC inline Packet packetOp(const Packet &a) const + { + return internal::plog10(a); + } + }; + template struct functor_traits> + { + enum { Cost = 5 * NumTraits::MulCost, PacketAccess = packet_traits::HasLog10 }; + }; + + /** \internal + * \brief Template functor to compute the square root of a scalar + * \sa class CwiseUnaryOp, Cwise::sqrt() + */ + template struct scalar_sqrt_op + { + EIGEN_EMPTY_STRUCT_CTOR(scalar_sqrt_op) + EIGEN_DEVICE_FUNC inline const Scalar operator()(const Scalar &a) const { return numext::sqrt(a); } + template EIGEN_DEVICE_FUNC inline Packet packetOp(const Packet &a) const + { + return internal::psqrt(a); + } + }; + template struct functor_traits> + { + enum { #if EIGEN_FAST_MATH - // The following numbers are based on the AVX implementation. - Cost = (sizeof(Scalar) == 8 ? 28 - // 4 pmul, 1 pmadd, 3 other - : (3 * NumTraits::AddCost + - 5 * NumTraits::MulCost)), + // The following numbers are based on the AVX implementation. + Cost = (sizeof(Scalar) == 8 ? 28 + // 4 pmul, 1 pmadd, 3 other + : (3 * NumTraits::AddCost + 5 * NumTraits::MulCost)), #else - // The following numbers are based on min VSQRT throughput on Haswell. - Cost = (sizeof(Scalar) == 8 ? 28 : 14), + // The following numbers are based on min VSQRT throughput on Haswell. + Cost = (sizeof(Scalar) == 8 ? 28 : 14), #endif - PacketAccess = packet_traits::HasSqrt - }; -}; - -/** \internal - * \brief Template functor to compute the reciprocal square root of a scalar - * \sa class CwiseUnaryOp, Cwise::rsqrt() - */ -template struct scalar_rsqrt_op { - EIGEN_EMPTY_STRUCT_CTOR(scalar_rsqrt_op) - EIGEN_DEVICE_FUNC inline const Scalar operator() (const Scalar& a) const { return Scalar(1)/numext::sqrt(a); } - template - EIGEN_DEVICE_FUNC inline Packet packetOp(const Packet& a) const { return internal::prsqrt(a); } -}; - -template -struct functor_traits > -{ enum { - Cost = 5 * NumTraits::MulCost, - PacketAccess = packet_traits::HasRsqrt - }; -}; - -/** \internal - * \brief Template functor to compute the cosine of a scalar - * \sa class CwiseUnaryOp, ArrayBase::cos() - */ -template struct scalar_cos_op { - EIGEN_EMPTY_STRUCT_CTOR(scalar_cos_op) - EIGEN_DEVICE_FUNC inline Scalar operator() (const Scalar& a) const { return numext::cos(a); } - template - EIGEN_DEVICE_FUNC inline Packet packetOp(const Packet& a) const { return internal::pcos(a); } -}; -template -struct functor_traits > -{ - enum { - Cost = 5 * NumTraits::MulCost, - PacketAccess = packet_traits::HasCos - }; -}; - -/** \internal - * \brief Template functor to compute the sine of a scalar - * \sa class CwiseUnaryOp, ArrayBase::sin() - */ -template struct scalar_sin_op { - EIGEN_EMPTY_STRUCT_CTOR(scalar_sin_op) - EIGEN_DEVICE_FUNC inline const Scalar operator() (const Scalar& a) const { return numext::sin(a); } - template - EIGEN_DEVICE_FUNC inline Packet packetOp(const Packet& a) const { return internal::psin(a); } -}; -template -struct functor_traits > -{ - enum { - Cost = 5 * NumTraits::MulCost, - PacketAccess = packet_traits::HasSin - }; -}; - - -/** \internal - * \brief Template functor to compute the tan of a scalar - * \sa class CwiseUnaryOp, ArrayBase::tan() - */ -template struct scalar_tan_op { - EIGEN_EMPTY_STRUCT_CTOR(scalar_tan_op) - EIGEN_DEVICE_FUNC inline const Scalar operator() (const Scalar& a) const { return numext::tan(a); } - template - EIGEN_DEVICE_FUNC inline Packet packetOp(const Packet& a) const { return internal::ptan(a); } -}; -template -struct functor_traits > -{ - enum { - Cost = 5 * NumTraits::MulCost, - PacketAccess = packet_traits::HasTan - }; -}; - -/** \internal - * \brief Template functor to compute the arc cosine of a scalar - * \sa class CwiseUnaryOp, ArrayBase::acos() - */ -template struct scalar_acos_op { - EIGEN_EMPTY_STRUCT_CTOR(scalar_acos_op) - EIGEN_DEVICE_FUNC inline const Scalar operator() (const Scalar& a) const { return numext::acos(a); } - template - EIGEN_DEVICE_FUNC inline Packet packetOp(const Packet& a) const { return internal::pacos(a); } -}; -template -struct functor_traits > -{ - enum { - Cost = 5 * NumTraits::MulCost, - PacketAccess = packet_traits::HasACos - }; -}; - -/** \internal - * \brief Template functor to compute the arc sine of a scalar - * \sa class CwiseUnaryOp, ArrayBase::asin() - */ -template struct scalar_asin_op { - EIGEN_EMPTY_STRUCT_CTOR(scalar_asin_op) - EIGEN_DEVICE_FUNC inline const Scalar operator() (const Scalar& a) const { return numext::asin(a); } - template - EIGEN_DEVICE_FUNC inline Packet packetOp(const Packet& a) const { return internal::pasin(a); } -}; -template -struct functor_traits > -{ - enum { - Cost = 5 * NumTraits::MulCost, - PacketAccess = packet_traits::HasASin - }; -}; - - -/** \internal - * \brief Template functor to compute the atan of a scalar - * \sa class CwiseUnaryOp, ArrayBase::atan() - */ -template struct scalar_atan_op { - EIGEN_EMPTY_STRUCT_CTOR(scalar_atan_op) - EIGEN_DEVICE_FUNC inline const Scalar operator() (const Scalar& a) const { return numext::atan(a); } - template - EIGEN_DEVICE_FUNC inline Packet packetOp(const Packet& a) const { return internal::patan(a); } -}; -template -struct functor_traits > -{ - enum { - Cost = 5 * NumTraits::MulCost, - PacketAccess = packet_traits::HasATan - }; -}; - -/** \internal - * \brief Template functor to compute the tanh of a scalar - * \sa class CwiseUnaryOp, ArrayBase::tanh() - */ -template -struct scalar_tanh_op { - EIGEN_EMPTY_STRUCT_CTOR(scalar_tanh_op) - EIGEN_DEVICE_FUNC inline const Scalar operator()(const Scalar& a) const { return numext::tanh(a); } - template - EIGEN_DEVICE_FUNC inline Packet packetOp(const Packet& x) const { return ptanh(x); } -}; - -template -struct functor_traits > { - enum { - PacketAccess = packet_traits::HasTanh, - Cost = ( (EIGEN_FAST_MATH && is_same::value) + PacketAccess = packet_traits::HasSqrt + }; + }; + + /** \internal + * \brief Template functor to compute the reciprocal square root of a scalar + * \sa class CwiseUnaryOp, Cwise::rsqrt() + */ + template struct scalar_rsqrt_op + { + EIGEN_EMPTY_STRUCT_CTOR(scalar_rsqrt_op) + EIGEN_DEVICE_FUNC inline const Scalar operator()(const Scalar &a) const { return Scalar(1) / numext::sqrt(a); } + template EIGEN_DEVICE_FUNC inline Packet packetOp(const Packet &a) const + { + return internal::prsqrt(a); + } + }; + + template struct functor_traits> + { + enum { Cost = 5 * NumTraits::MulCost, PacketAccess = packet_traits::HasRsqrt }; + }; + + /** \internal + * \brief Template functor to compute the cosine of a scalar + * \sa class CwiseUnaryOp, ArrayBase::cos() + */ + template struct scalar_cos_op + { + EIGEN_EMPTY_STRUCT_CTOR(scalar_cos_op) + EIGEN_DEVICE_FUNC inline Scalar operator()(const Scalar &a) const { return numext::cos(a); } + template EIGEN_DEVICE_FUNC inline Packet packetOp(const Packet &a) const + { + return internal::pcos(a); + } + }; + template struct functor_traits> + { + enum { Cost = 5 * NumTraits::MulCost, PacketAccess = packet_traits::HasCos }; + }; + + /** \internal + * \brief Template functor to compute the sine of a scalar + * \sa class CwiseUnaryOp, ArrayBase::sin() + */ + template struct scalar_sin_op + { + EIGEN_EMPTY_STRUCT_CTOR(scalar_sin_op) + EIGEN_DEVICE_FUNC inline const Scalar operator()(const Scalar &a) const { return numext::sin(a); } + template EIGEN_DEVICE_FUNC inline Packet packetOp(const Packet &a) const + { + return internal::psin(a); + } + }; + template struct functor_traits> + { + enum { Cost = 5 * NumTraits::MulCost, PacketAccess = packet_traits::HasSin }; + }; + + + /** \internal + * \brief Template functor to compute the tan of a scalar + * \sa class CwiseUnaryOp, ArrayBase::tan() + */ + template struct scalar_tan_op + { + EIGEN_EMPTY_STRUCT_CTOR(scalar_tan_op) + EIGEN_DEVICE_FUNC inline const Scalar operator()(const Scalar &a) const { return numext::tan(a); } + template EIGEN_DEVICE_FUNC inline Packet packetOp(const Packet &a) const + { + return internal::ptan(a); + } + }; + template struct functor_traits> + { + enum { Cost = 5 * NumTraits::MulCost, PacketAccess = packet_traits::HasTan }; + }; + + /** \internal + * \brief Template functor to compute the arc cosine of a scalar + * \sa class CwiseUnaryOp, ArrayBase::acos() + */ + template struct scalar_acos_op + { + EIGEN_EMPTY_STRUCT_CTOR(scalar_acos_op) + EIGEN_DEVICE_FUNC inline const Scalar operator()(const Scalar &a) const { return numext::acos(a); } + template EIGEN_DEVICE_FUNC inline Packet packetOp(const Packet &a) const + { + return internal::pacos(a); + } + }; + template struct functor_traits> + { + enum { Cost = 5 * NumTraits::MulCost, PacketAccess = packet_traits::HasACos }; + }; + + /** \internal + * \brief Template functor to compute the arc sine of a scalar + * \sa class CwiseUnaryOp, ArrayBase::asin() + */ + template struct scalar_asin_op + { + EIGEN_EMPTY_STRUCT_CTOR(scalar_asin_op) + EIGEN_DEVICE_FUNC inline const Scalar operator()(const Scalar &a) const { return numext::asin(a); } + template EIGEN_DEVICE_FUNC inline Packet packetOp(const Packet &a) const + { + return internal::pasin(a); + } + }; + template struct functor_traits> + { + enum { Cost = 5 * NumTraits::MulCost, PacketAccess = packet_traits::HasASin }; + }; + + + /** \internal + * \brief Template functor to compute the atan of a scalar + * \sa class CwiseUnaryOp, ArrayBase::atan() + */ + template struct scalar_atan_op + { + EIGEN_EMPTY_STRUCT_CTOR(scalar_atan_op) + EIGEN_DEVICE_FUNC inline const Scalar operator()(const Scalar &a) const { return numext::atan(a); } + template EIGEN_DEVICE_FUNC inline Packet packetOp(const Packet &a) const + { + return internal::patan(a); + } + }; + template struct functor_traits> + { + enum { Cost = 5 * NumTraits::MulCost, PacketAccess = packet_traits::HasATan }; + }; + + /** \internal + * \brief Template functor to compute the tanh of a scalar + * \sa class CwiseUnaryOp, ArrayBase::tanh() + */ + template struct scalar_tanh_op + { + EIGEN_EMPTY_STRUCT_CTOR(scalar_tanh_op) + EIGEN_DEVICE_FUNC inline const Scalar operator()(const Scalar &a) const { return numext::tanh(a); } + template EIGEN_DEVICE_FUNC inline Packet packetOp(const Packet &x) const { return ptanh(x); } + }; + + template struct functor_traits> + { + enum { + PacketAccess = packet_traits::HasTanh, + Cost = ((EIGEN_FAST_MATH && is_same::value) // The following numbers are based on the AVX implementation, #ifdef EIGEN_VECTORIZE_FMA // Haswell can issue 2 add/mul/madd per cycle. // 9 pmadd, 2 pmul, 1 div, 2 other - ? (2 * NumTraits::AddCost + - 6 * NumTraits::MulCost + - scalar_div_cost::HasDiv>::value) + ? (2 * NumTraits::AddCost + 6 * NumTraits::MulCost + + scalar_div_cost::HasDiv>::value) #else - ? (11 * NumTraits::AddCost + - 11 * NumTraits::MulCost + - scalar_div_cost::HasDiv>::value) + ? (11 * NumTraits::AddCost + 11 * NumTraits::MulCost + + scalar_div_cost::HasDiv>::value) #endif // This number assumes a naive implementation of tanh - : (6 * NumTraits::AddCost + - 3 * NumTraits::MulCost + - 2 * scalar_div_cost::HasDiv>::value + - functor_traits >::Cost)) - }; -}; - -/** \internal - * \brief Template functor to compute the sinh of a scalar - * \sa class CwiseUnaryOp, ArrayBase::sinh() - */ -template struct scalar_sinh_op { - EIGEN_EMPTY_STRUCT_CTOR(scalar_sinh_op) - EIGEN_DEVICE_FUNC inline const Scalar operator() (const Scalar& a) const { return numext::sinh(a); } - template - EIGEN_DEVICE_FUNC inline Packet packetOp(const Packet& a) const { return internal::psinh(a); } -}; -template -struct functor_traits > -{ - enum { - Cost = 5 * NumTraits::MulCost, - PacketAccess = packet_traits::HasSinh - }; -}; - -/** \internal - * \brief Template functor to compute the cosh of a scalar - * \sa class CwiseUnaryOp, ArrayBase::cosh() - */ -template struct scalar_cosh_op { - EIGEN_EMPTY_STRUCT_CTOR(scalar_cosh_op) - EIGEN_DEVICE_FUNC inline const Scalar operator() (const Scalar& a) const { return numext::cosh(a); } - template - EIGEN_DEVICE_FUNC inline Packet packetOp(const Packet& a) const { return internal::pcosh(a); } -}; -template -struct functor_traits > -{ - enum { - Cost = 5 * NumTraits::MulCost, - PacketAccess = packet_traits::HasCosh - }; -}; - -/** \internal - * \brief Template functor to compute the inverse of a scalar - * \sa class CwiseUnaryOp, Cwise::inverse() - */ -template -struct scalar_inverse_op { - EIGEN_EMPTY_STRUCT_CTOR(scalar_inverse_op) - EIGEN_DEVICE_FUNC inline Scalar operator() (const Scalar& a) const { return Scalar(1)/a; } - template - EIGEN_DEVICE_FUNC inline const Packet packetOp(const Packet& a) const - { return internal::pdiv(pset1(Scalar(1)),a); } -}; -template -struct functor_traits > -{ enum { Cost = NumTraits::MulCost, PacketAccess = packet_traits::HasDiv }; }; - -/** \internal - * \brief Template functor to compute the square of a scalar - * \sa class CwiseUnaryOp, Cwise::square() - */ -template -struct scalar_square_op { - EIGEN_EMPTY_STRUCT_CTOR(scalar_square_op) - EIGEN_DEVICE_FUNC inline Scalar operator() (const Scalar& a) const { return a*a; } - template - EIGEN_DEVICE_FUNC inline const Packet packetOp(const Packet& a) const - { return internal::pmul(a,a); } -}; -template -struct functor_traits > -{ enum { Cost = NumTraits::MulCost, PacketAccess = packet_traits::HasMul }; }; - -/** \internal - * \brief Template functor to compute the cube of a scalar - * \sa class CwiseUnaryOp, Cwise::cube() - */ -template -struct scalar_cube_op { - EIGEN_EMPTY_STRUCT_CTOR(scalar_cube_op) - EIGEN_DEVICE_FUNC inline Scalar operator() (const Scalar& a) const { return a*a*a; } - template - EIGEN_DEVICE_FUNC inline const Packet packetOp(const Packet& a) const - { return internal::pmul(a,pmul(a,a)); } -}; -template -struct functor_traits > -{ enum { Cost = 2*NumTraits::MulCost, PacketAccess = packet_traits::HasMul }; }; - -/** \internal - * \brief Template functor to compute the rounded value of a scalar - * \sa class CwiseUnaryOp, ArrayBase::round() - */ -template struct scalar_round_op { - EIGEN_EMPTY_STRUCT_CTOR(scalar_round_op) - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const Scalar operator() (const Scalar& a) const { return numext::round(a); } - template - EIGEN_DEVICE_FUNC inline Packet packetOp(const Packet& a) const { return internal::pround(a); } -}; -template -struct functor_traits > -{ - enum { - Cost = NumTraits::MulCost, - PacketAccess = packet_traits::HasRound - }; -}; - -/** \internal - * \brief Template functor to compute the floor of a scalar - * \sa class CwiseUnaryOp, ArrayBase::floor() - */ -template struct scalar_floor_op { - EIGEN_EMPTY_STRUCT_CTOR(scalar_floor_op) - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const Scalar operator() (const Scalar& a) const { return numext::floor(a); } - template - EIGEN_DEVICE_FUNC inline Packet packetOp(const Packet& a) const { return internal::pfloor(a); } -}; -template -struct functor_traits > -{ - enum { - Cost = NumTraits::MulCost, - PacketAccess = packet_traits::HasFloor - }; -}; - -/** \internal - * \brief Template functor to compute the ceil of a scalar - * \sa class CwiseUnaryOp, ArrayBase::ceil() - */ -template struct scalar_ceil_op { - EIGEN_EMPTY_STRUCT_CTOR(scalar_ceil_op) - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const Scalar operator() (const Scalar& a) const { return numext::ceil(a); } - template - EIGEN_DEVICE_FUNC inline Packet packetOp(const Packet& a) const { return internal::pceil(a); } -}; -template -struct functor_traits > -{ - enum { - Cost = NumTraits::MulCost, - PacketAccess = packet_traits::HasCeil - }; -}; - -/** \internal - * \brief Template functor to compute whether a scalar is NaN - * \sa class CwiseUnaryOp, ArrayBase::isnan() - */ -template struct scalar_isnan_op { - EIGEN_EMPTY_STRUCT_CTOR(scalar_isnan_op) - typedef bool result_type; - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE result_type operator() (const Scalar& a) const { return (numext::isnan)(a); } -}; -template -struct functor_traits > -{ - enum { - Cost = NumTraits::MulCost, - PacketAccess = false - }; -}; - -/** \internal - * \brief Template functor to check whether a scalar is +/-inf - * \sa class CwiseUnaryOp, ArrayBase::isinf() - */ -template struct scalar_isinf_op { - EIGEN_EMPTY_STRUCT_CTOR(scalar_isinf_op) - typedef bool result_type; - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE result_type operator() (const Scalar& a) const { return (numext::isinf)(a); } -}; -template -struct functor_traits > -{ - enum { - Cost = NumTraits::MulCost, - PacketAccess = false - }; -}; - -/** \internal - * \brief Template functor to check whether a scalar has a finite value - * \sa class CwiseUnaryOp, ArrayBase::isfinite() - */ -template struct scalar_isfinite_op { - EIGEN_EMPTY_STRUCT_CTOR(scalar_isfinite_op) - typedef bool result_type; - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE result_type operator() (const Scalar& a) const { return (numext::isfinite)(a); } -}; -template -struct functor_traits > -{ - enum { - Cost = NumTraits::MulCost, - PacketAccess = false - }; -}; - -/** \internal - * \brief Template functor to compute the logical not of a boolean - * - * \sa class CwiseUnaryOp, ArrayBase::operator! - */ -template struct scalar_boolean_not_op { - EIGEN_EMPTY_STRUCT_CTOR(scalar_boolean_not_op) - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE bool operator() (const bool& a) const { return !a; } -}; -template -struct functor_traits > { - enum { - Cost = NumTraits::AddCost, - PacketAccess = false - }; -}; - -/** \internal - * \brief Template functor to compute the signum of a scalar - * \sa class CwiseUnaryOp, Cwise::sign() - */ -template::IsComplex!=0) > struct scalar_sign_op; -template -struct scalar_sign_op { - EIGEN_EMPTY_STRUCT_CTOR(scalar_sign_op) - EIGEN_DEVICE_FUNC inline const Scalar operator() (const Scalar& a) const - { - return Scalar( (a>Scalar(0)) - (a - //EIGEN_DEVICE_FUNC inline Packet packetOp(const Packet& a) const { return internal::psign(a); } -}; -template -struct scalar_sign_op { - EIGEN_EMPTY_STRUCT_CTOR(scalar_sign_op) - EIGEN_DEVICE_FUNC inline const Scalar operator() (const Scalar& a) const - { - typedef typename NumTraits::Real real_type; - real_type aa = numext::abs(a); - if (aa==real_type(0)) - return Scalar(0); - aa = real_type(1)/aa; - return Scalar(real(a)*aa, imag(a)*aa ); - } - //TODO - //template - //EIGEN_DEVICE_FUNC inline Packet packetOp(const Packet& a) const { return internal::psign(a); } -}; -template -struct functor_traits > -{ enum { - Cost = - NumTraits::IsComplex - ? ( 8*NumTraits::MulCost ) // roughly - : ( 3*NumTraits::AddCost), - PacketAccess = packet_traits::HasSign - }; -}; - -} // end namespace internal - -} // end namespace Eigen - -#endif // EIGEN_FUNCTORS_H + : (6 * NumTraits::AddCost + 3 * NumTraits::MulCost + + 2 * scalar_div_cost::HasDiv>::value + + functor_traits>::Cost)) + }; + }; + + /** \internal + * \brief Template functor to compute the sinh of a scalar + * \sa class CwiseUnaryOp, ArrayBase::sinh() + */ + template struct scalar_sinh_op + { + EIGEN_EMPTY_STRUCT_CTOR(scalar_sinh_op) + EIGEN_DEVICE_FUNC inline const Scalar operator()(const Scalar &a) const { return numext::sinh(a); } + template EIGEN_DEVICE_FUNC inline Packet packetOp(const Packet &a) const + { + return internal::psinh(a); + } + }; + template struct functor_traits> + { + enum { Cost = 5 * NumTraits::MulCost, PacketAccess = packet_traits::HasSinh }; + }; + + /** \internal + * \brief Template functor to compute the cosh of a scalar + * \sa class CwiseUnaryOp, ArrayBase::cosh() + */ + template struct scalar_cosh_op + { + EIGEN_EMPTY_STRUCT_CTOR(scalar_cosh_op) + EIGEN_DEVICE_FUNC inline const Scalar operator()(const Scalar &a) const { return numext::cosh(a); } + template EIGEN_DEVICE_FUNC inline Packet packetOp(const Packet &a) const + { + return internal::pcosh(a); + } + }; + template struct functor_traits> + { + enum { Cost = 5 * NumTraits::MulCost, PacketAccess = packet_traits::HasCosh }; + }; + + /** \internal + * \brief Template functor to compute the inverse of a scalar + * \sa class CwiseUnaryOp, Cwise::inverse() + */ + template struct scalar_inverse_op + { + EIGEN_EMPTY_STRUCT_CTOR(scalar_inverse_op) + EIGEN_DEVICE_FUNC inline Scalar operator()(const Scalar &a) const { return Scalar(1) / a; } + template EIGEN_DEVICE_FUNC inline const Packet packetOp(const Packet &a) const + { + return internal::pdiv(pset1(Scalar(1)), a); + } + }; + template struct functor_traits> + { + enum { Cost = NumTraits::MulCost, PacketAccess = packet_traits::HasDiv }; + }; + + /** \internal + * \brief Template functor to compute the square of a scalar + * \sa class CwiseUnaryOp, Cwise::square() + */ + template struct scalar_square_op + { + EIGEN_EMPTY_STRUCT_CTOR(scalar_square_op) + EIGEN_DEVICE_FUNC inline Scalar operator()(const Scalar &a) const { return a * a; } + template EIGEN_DEVICE_FUNC inline const Packet packetOp(const Packet &a) const + { + return internal::pmul(a, a); + } + }; + template struct functor_traits> + { + enum { Cost = NumTraits::MulCost, PacketAccess = packet_traits::HasMul }; + }; + + /** \internal + * \brief Template functor to compute the cube of a scalar + * \sa class CwiseUnaryOp, Cwise::cube() + */ + template struct scalar_cube_op + { + EIGEN_EMPTY_STRUCT_CTOR(scalar_cube_op) + EIGEN_DEVICE_FUNC inline Scalar operator()(const Scalar &a) const { return a * a * a; } + template EIGEN_DEVICE_FUNC inline const Packet packetOp(const Packet &a) const + { + return internal::pmul(a, pmul(a, a)); + } + }; + template struct functor_traits> + { + enum { Cost = 2 * NumTraits::MulCost, PacketAccess = packet_traits::HasMul }; + }; + + /** \internal + * \brief Template functor to compute the rounded value of a scalar + * \sa class CwiseUnaryOp, ArrayBase::round() + */ + template struct scalar_round_op + { + EIGEN_EMPTY_STRUCT_CTOR(scalar_round_op) + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const Scalar operator()(const Scalar &a) const { return numext::round(a); } + template EIGEN_DEVICE_FUNC inline Packet packetOp(const Packet &a) const + { + return internal::pround(a); + } + }; + template struct functor_traits> + { + enum { Cost = NumTraits::MulCost, PacketAccess = packet_traits::HasRound }; + }; + + /** \internal + * \brief Template functor to compute the floor of a scalar + * \sa class CwiseUnaryOp, ArrayBase::floor() + */ + template struct scalar_floor_op + { + EIGEN_EMPTY_STRUCT_CTOR(scalar_floor_op) + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const Scalar operator()(const Scalar &a) const { return numext::floor(a); } + template EIGEN_DEVICE_FUNC inline Packet packetOp(const Packet &a) const + { + return internal::pfloor(a); + } + }; + template struct functor_traits> + { + enum { Cost = NumTraits::MulCost, PacketAccess = packet_traits::HasFloor }; + }; + + /** \internal + * \brief Template functor to compute the ceil of a scalar + * \sa class CwiseUnaryOp, ArrayBase::ceil() + */ + template struct scalar_ceil_op + { + EIGEN_EMPTY_STRUCT_CTOR(scalar_ceil_op) + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const Scalar operator()(const Scalar &a) const { return numext::ceil(a); } + template EIGEN_DEVICE_FUNC inline Packet packetOp(const Packet &a) const + { + return internal::pceil(a); + } + }; + template struct functor_traits> + { + enum { Cost = NumTraits::MulCost, PacketAccess = packet_traits::HasCeil }; + }; + + /** \internal + * \brief Template functor to compute whether a scalar is NaN + * \sa class CwiseUnaryOp, ArrayBase::isnan() + */ + template struct scalar_isnan_op + { + EIGEN_EMPTY_STRUCT_CTOR(scalar_isnan_op) + typedef bool result_type; + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE result_type operator()(const Scalar &a) const { return (numext::isnan)(a); } + }; + template struct functor_traits> + { + enum { Cost = NumTraits::MulCost, PacketAccess = false }; + }; + + /** \internal + * \brief Template functor to check whether a scalar is +/-inf + * \sa class CwiseUnaryOp, ArrayBase::isinf() + */ + template struct scalar_isinf_op + { + EIGEN_EMPTY_STRUCT_CTOR(scalar_isinf_op) + typedef bool result_type; + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE result_type operator()(const Scalar &a) const { return (numext::isinf)(a); } + }; + template struct functor_traits> + { + enum { Cost = NumTraits::MulCost, PacketAccess = false }; + }; + + /** \internal + * \brief Template functor to check whether a scalar has a finite value + * \sa class CwiseUnaryOp, ArrayBase::isfinite() + */ + template struct scalar_isfinite_op + { + EIGEN_EMPTY_STRUCT_CTOR(scalar_isfinite_op) + typedef bool result_type; + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE result_type operator()(const Scalar &a) const + { + return (numext::isfinite)(a); + } + }; + template struct functor_traits> + { + enum { Cost = NumTraits::MulCost, PacketAccess = false }; + }; + + /** \internal + * \brief Template functor to compute the logical not of a boolean + * + * \sa class CwiseUnaryOp, ArrayBase::operator! + */ + template struct scalar_boolean_not_op + { + EIGEN_EMPTY_STRUCT_CTOR(scalar_boolean_not_op) + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE bool operator()(const bool &a) const { return !a; } + }; + template struct functor_traits> + { + enum { Cost = NumTraits::AddCost, PacketAccess = false }; + }; + + /** \internal + * \brief Template functor to compute the signum of a scalar + * \sa class CwiseUnaryOp, Cwise::sign() + */ + template::IsComplex != 0)> struct scalar_sign_op; + template struct scalar_sign_op + { + EIGEN_EMPTY_STRUCT_CTOR(scalar_sign_op) + EIGEN_DEVICE_FUNC inline const Scalar operator()(const Scalar &a) const + { + return Scalar((a > Scalar(0)) - (a < Scalar(0))); + } + // TODO + // template + // EIGEN_DEVICE_FUNC inline Packet packetOp(const Packet& a) const { return internal::psign(a); } + }; + template struct scalar_sign_op + { + EIGEN_EMPTY_STRUCT_CTOR(scalar_sign_op) + EIGEN_DEVICE_FUNC inline const Scalar operator()(const Scalar &a) const + { + typedef typename NumTraits::Real real_type; + real_type aa = numext::abs(a); + if (aa == real_type(0)) return Scalar(0); + aa = real_type(1) / aa; + return Scalar(real(a) * aa, imag(a) * aa); + } + // TODO + // template + // EIGEN_DEVICE_FUNC inline Packet packetOp(const Packet& a) const { return internal::psign(a); } + }; + template struct functor_traits> + { + enum { + Cost = NumTraits::IsComplex ? (8 * NumTraits::MulCost)// roughly + : (3 * NumTraits::AddCost), + PacketAccess = packet_traits::HasSign + }; + }; + +}// end namespace internal + +}// end namespace Eigen + +#endif// EIGEN_FUNCTORS_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/products/GeneralBlockPanelKernel.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/products/GeneralBlockPanelKernel.h index e3980f6f..f6bf92cd 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/products/GeneralBlockPanelKernel.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/products/GeneralBlockPanelKernel.h @@ -15,1358 +15,1331 @@ namespace Eigen { namespace internal { -template -class gebp_traits; + template class gebp_traits; -/** \internal \returns b if a<=0, and returns a otherwise. */ -inline std::ptrdiff_t manage_caching_sizes_helper(std::ptrdiff_t a, std::ptrdiff_t b) -{ - return a<=0 ? b : a; -} + /** \internal \returns b if a<=0, and returns a otherwise. */ + inline std::ptrdiff_t manage_caching_sizes_helper(std::ptrdiff_t a, std::ptrdiff_t b) { return a <= 0 ? b : a; } #if EIGEN_ARCH_i386_OR_x86_64 -const std::ptrdiff_t defaultL1CacheSize = 32*1024; -const std::ptrdiff_t defaultL2CacheSize = 256*1024; -const std::ptrdiff_t defaultL3CacheSize = 2*1024*1024; + const std::ptrdiff_t defaultL1CacheSize = 32 * 1024; + const std::ptrdiff_t defaultL2CacheSize = 256 * 1024; + const std::ptrdiff_t defaultL3CacheSize = 2 * 1024 * 1024; #else -const std::ptrdiff_t defaultL1CacheSize = 16*1024; -const std::ptrdiff_t defaultL2CacheSize = 512*1024; -const std::ptrdiff_t defaultL3CacheSize = 512*1024; + const std::ptrdiff_t defaultL1CacheSize = 16 * 1024; + const std::ptrdiff_t defaultL2CacheSize = 512 * 1024; + const std::ptrdiff_t defaultL3CacheSize = 512 * 1024; #endif -/** \internal */ -struct CacheSizes { - CacheSizes(): m_l1(-1),m_l2(-1),m_l3(-1) { - int l1CacheSize, l2CacheSize, l3CacheSize; - queryCacheSizes(l1CacheSize, l2CacheSize, l3CacheSize); - m_l1 = manage_caching_sizes_helper(l1CacheSize, defaultL1CacheSize); - m_l2 = manage_caching_sizes_helper(l2CacheSize, defaultL2CacheSize); - m_l3 = manage_caching_sizes_helper(l3CacheSize, defaultL3CacheSize); - } - - std::ptrdiff_t m_l1; - std::ptrdiff_t m_l2; - std::ptrdiff_t m_l3; -}; - - -/** \internal */ -inline void manage_caching_sizes(Action action, std::ptrdiff_t* l1, std::ptrdiff_t* l2, std::ptrdiff_t* l3) -{ - static CacheSizes m_cacheSizes; - - if(action==SetAction) - { - // set the cpu cache size and cache all block sizes from a global cache size in byte - eigen_internal_assert(l1!=0 && l2!=0); - m_cacheSizes.m_l1 = *l1; - m_cacheSizes.m_l2 = *l2; - m_cacheSizes.m_l3 = *l3; - } - else if(action==GetAction) + /** \internal */ + struct CacheSizes { - eigen_internal_assert(l1!=0 && l2!=0); - *l1 = m_cacheSizes.m_l1; - *l2 = m_cacheSizes.m_l2; - *l3 = m_cacheSizes.m_l3; - } - else - { - eigen_internal_assert(false); - } -} - -/* Helper for computeProductBlockingSizes. - * - * Given a m x k times k x n matrix product of scalar types \c LhsScalar and \c RhsScalar, - * this function computes the blocking size parameters along the respective dimensions - * for matrix products and related algorithms. The blocking sizes depends on various - * parameters: - * - the L1 and L2 cache sizes, - * - the register level blocking sizes defined by gebp_traits, - * - the number of scalars that fit into a packet (when vectorization is enabled). - * - * \sa setCpuCacheSizes */ - -template -void evaluateProductBlockingSizesHeuristic(Index& k, Index& m, Index& n, Index num_threads = 1) -{ - typedef gebp_traits Traits; + CacheSizes() : m_l1(-1), m_l2(-1), m_l3(-1) + { + int l1CacheSize, l2CacheSize, l3CacheSize; + queryCacheSizes(l1CacheSize, l2CacheSize, l3CacheSize); + m_l1 = manage_caching_sizes_helper(l1CacheSize, defaultL1CacheSize); + m_l2 = manage_caching_sizes_helper(l2CacheSize, defaultL2CacheSize); + m_l3 = manage_caching_sizes_helper(l3CacheSize, defaultL3CacheSize); + } - // Explanations: - // Let's recall that the product algorithms form mc x kc vertical panels A' on the lhs and - // kc x nc blocks B' on the rhs. B' has to fit into L2/L3 cache. Moreover, A' is processed - // per mr x kc horizontal small panels where mr is the blocking size along the m dimension - // at the register level. This small horizontal panel has to stay within L1 cache. - std::ptrdiff_t l1, l2, l3; - manage_caching_sizes(GetAction, &l1, &l2, &l3); + std::ptrdiff_t m_l1; + std::ptrdiff_t m_l2; + std::ptrdiff_t m_l3; + }; - if (num_threads > 1) { - typedef typename Traits::ResScalar ResScalar; - enum { - kdiv = KcFactor * (Traits::mr * sizeof(LhsScalar) + Traits::nr * sizeof(RhsScalar)), - ksub = Traits::mr * Traits::nr * sizeof(ResScalar), - kr = 8, - mr = Traits::mr, - nr = Traits::nr - }; - // Increasing k gives us more time to prefetch the content of the "C" - // registers. However once the latency is hidden there is no point in - // increasing the value of k, so we'll cap it at 320 (value determined - // experimentally). - const Index k_cache = (numext::mini)((l1-ksub)/kdiv, 320); - if (k_cache < k) { - k = k_cache - (k_cache % kr); - eigen_internal_assert(k > 0); - } - const Index n_cache = (l2-l1) / (nr * sizeof(RhsScalar) * k); - const Index n_per_thread = numext::div_ceil(n, num_threads); - if (n_cache <= n_per_thread) { - // Don't exceed the capacity of the l2 cache. - eigen_internal_assert(n_cache >= static_cast(nr)); - n = n_cache - (n_cache % nr); - eigen_internal_assert(n > 0); + /** \internal */ + inline void manage_caching_sizes(Action action, std::ptrdiff_t *l1, std::ptrdiff_t *l2, std::ptrdiff_t *l3) + { + static CacheSizes m_cacheSizes; + + if (action == SetAction) { + // set the cpu cache size and cache all block sizes from a global cache size in byte + eigen_internal_assert(l1 != 0 && l2 != 0); + m_cacheSizes.m_l1 = *l1; + m_cacheSizes.m_l2 = *l2; + m_cacheSizes.m_l3 = *l3; + } else if (action == GetAction) { + eigen_internal_assert(l1 != 0 && l2 != 0); + *l1 = m_cacheSizes.m_l1; + *l2 = m_cacheSizes.m_l2; + *l3 = m_cacheSizes.m_l3; } else { - n = (numext::mini)(n, (n_per_thread + nr - 1) - ((n_per_thread + nr - 1) % nr)); + eigen_internal_assert(false); } + } - if (l3 > l2) { - // l3 is shared between all cores, so we'll give each thread its own chunk of l3. - const Index m_cache = (l3-l2) / (sizeof(LhsScalar) * k * num_threads); - const Index m_per_thread = numext::div_ceil(m, num_threads); - if(m_cache < m_per_thread && m_cache >= static_cast(mr)) { - m = m_cache - (m_cache % mr); - eigen_internal_assert(m > 0); + /* Helper for computeProductBlockingSizes. + * + * Given a m x k times k x n matrix product of scalar types \c LhsScalar and \c RhsScalar, + * this function computes the blocking size parameters along the respective dimensions + * for matrix products and related algorithms. The blocking sizes depends on various + * parameters: + * - the L1 and L2 cache sizes, + * - the register level blocking sizes defined by gebp_traits, + * - the number of scalars that fit into a packet (when vectorization is enabled). + * + * \sa setCpuCacheSizes */ + + template + void evaluateProductBlockingSizesHeuristic(Index &k, Index &m, Index &n, Index num_threads = 1) + { + typedef gebp_traits Traits; + + // Explanations: + // Let's recall that the product algorithms form mc x kc vertical panels A' on the lhs and + // kc x nc blocks B' on the rhs. B' has to fit into L2/L3 cache. Moreover, A' is processed + // per mr x kc horizontal small panels where mr is the blocking size along the m dimension + // at the register level. This small horizontal panel has to stay within L1 cache. + std::ptrdiff_t l1, l2, l3; + manage_caching_sizes(GetAction, &l1, &l2, &l3); + + if (num_threads > 1) { + typedef typename Traits::ResScalar ResScalar; + enum { + kdiv = KcFactor * (Traits::mr * sizeof(LhsScalar) + Traits::nr * sizeof(RhsScalar)), + ksub = Traits::mr * Traits::nr * sizeof(ResScalar), + kr = 8, + mr = Traits::mr, + nr = Traits::nr + }; + // Increasing k gives us more time to prefetch the content of the "C" + // registers. However once the latency is hidden there is no point in + // increasing the value of k, so we'll cap it at 320 (value determined + // experimentally). + const Index k_cache = (numext::mini)((l1 - ksub) / kdiv, 320); + if (k_cache < k) { + k = k_cache - (k_cache % kr); + eigen_internal_assert(k > 0); + } + + const Index n_cache = (l2 - l1) / (nr * sizeof(RhsScalar) * k); + const Index n_per_thread = numext::div_ceil(n, num_threads); + if (n_cache <= n_per_thread) { + // Don't exceed the capacity of the l2 cache. + eigen_internal_assert(n_cache >= static_cast(nr)); + n = n_cache - (n_cache % nr); + eigen_internal_assert(n > 0); } else { - m = (numext::mini)(m, (m_per_thread + mr - 1) - ((m_per_thread + mr - 1) % mr)); + n = (numext::mini)(n, (n_per_thread + nr - 1) - ((n_per_thread + nr - 1) % nr)); } - } - } - else { - // In unit tests we do not want to use extra large matrices, - // so we reduce the cache size to check the blocking strategy is not flawed + + if (l3 > l2) { + // l3 is shared between all cores, so we'll give each thread its own chunk of l3. + const Index m_cache = (l3 - l2) / (sizeof(LhsScalar) * k * num_threads); + const Index m_per_thread = numext::div_ceil(m, num_threads); + if (m_cache < m_per_thread && m_cache >= static_cast(mr)) { + m = m_cache - (m_cache % mr); + eigen_internal_assert(m > 0); + } else { + m = (numext::mini)(m, (m_per_thread + mr - 1) - ((m_per_thread + mr - 1) % mr)); + } + } + } else { + // In unit tests we do not want to use extra large matrices, + // so we reduce the cache size to check the blocking strategy is not flawed #ifdef EIGEN_DEBUG_SMALL_PRODUCT_BLOCKS - l1 = 9*1024; - l2 = 32*1024; - l3 = 512*1024; + l1 = 9 * 1024; + l2 = 32 * 1024; + l3 = 512 * 1024; #endif - // Early return for small problems because the computation below are time consuming for small problems. - // Perhaps it would make more sense to consider k*n*m?? - // Note that for very tiny problem, this function should be bypassed anyway - // because we use the coefficient-based implementation for them. - if((numext::maxi)(k,(numext::maxi)(m,n))<48) - return; - - typedef typename Traits::ResScalar ResScalar; - enum { - k_peeling = 8, - k_div = KcFactor * (Traits::mr * sizeof(LhsScalar) + Traits::nr * sizeof(RhsScalar)), - k_sub = Traits::mr * Traits::nr * sizeof(ResScalar) - }; - - // ---- 1st level of blocking on L1, yields kc ---- + // Early return for small problems because the computation below are time consuming for small problems. + // Perhaps it would make more sense to consider k*n*m?? + // Note that for very tiny problem, this function should be bypassed anyway + // because we use the coefficient-based implementation for them. + if ((numext::maxi)(k, (numext::maxi)(m, n)) < 48) return; + + typedef typename Traits::ResScalar ResScalar; + enum { + k_peeling = 8, + k_div = KcFactor * (Traits::mr * sizeof(LhsScalar) + Traits::nr * sizeof(RhsScalar)), + k_sub = Traits::mr * Traits::nr * sizeof(ResScalar) + }; + + // ---- 1st level of blocking on L1, yields kc ---- + + // Blocking on the third dimension (i.e., k) is chosen so that an horizontal panel + // of size mr x kc of the lhs plus a vertical panel of kc x nr of the rhs both fits within L1 cache. + // We also include a register-level block of the result (mx x nr). + // (In an ideal world only the lhs panel would stay in L1) + // Moreover, kc has to be a multiple of 8 to be compatible with loop peeling, leading to a maximum blocking size + // of: + const Index max_kc = numext::maxi(((l1 - k_sub) / k_div) & (~(k_peeling - 1)), 1); + const Index old_k = k; + if (k > max_kc) { + // We are really blocking on the third dimension: + // -> reduce blocking size to make sure the last block is as large as possible + // while keeping the same number of sweeps over the result. + k = (k % max_kc) == 0 ? max_kc + : max_kc - k_peeling * ((max_kc - 1 - (k % max_kc)) / (k_peeling * (k / max_kc + 1))); + + eigen_internal_assert(((old_k / k) == (old_k / max_kc)) && "the number of sweeps has to remain the same"); + } - // Blocking on the third dimension (i.e., k) is chosen so that an horizontal panel - // of size mr x kc of the lhs plus a vertical panel of kc x nr of the rhs both fits within L1 cache. - // We also include a register-level block of the result (mx x nr). - // (In an ideal world only the lhs panel would stay in L1) - // Moreover, kc has to be a multiple of 8 to be compatible with loop peeling, leading to a maximum blocking size of: - const Index max_kc = numext::maxi(((l1-k_sub)/k_div) & (~(k_peeling-1)),1); - const Index old_k = k; - if(k>max_kc) - { - // We are really blocking on the third dimension: - // -> reduce blocking size to make sure the last block is as large as possible - // while keeping the same number of sweeps over the result. - k = (k%max_kc)==0 ? max_kc - : max_kc - k_peeling * ((max_kc-1-(k%max_kc))/(k_peeling*(k/max_kc+1))); +// ---- 2nd level of blocking on max(L2,L3), yields nc ---- - eigen_internal_assert(((old_k/k) == (old_k/max_kc)) && "the number of sweeps has to remain the same"); - } +// TODO find a reliable way to get the actual amount of cache per core to use for 2nd level blocking, that is: +// actual_l2 = max(l2, l3/nb_core_sharing_l3) +// The number below is quite conservative: it is better to underestimate the cache size rather than overestimating it) +// For instance, it corresponds to 6MB of L3 shared among 4 cores. +#ifdef EIGEN_DEBUG_SMALL_PRODUCT_BLOCKS + const Index actual_l2 = l3; +#else + const Index actual_l2 = 1572864;// == 1.5 MB +#endif - // ---- 2nd level of blocking on max(L2,L3), yields nc ---- - - // TODO find a reliable way to get the actual amount of cache per core to use for 2nd level blocking, that is: - // actual_l2 = max(l2, l3/nb_core_sharing_l3) - // The number below is quite conservative: it is better to underestimate the cache size rather than overestimating it) - // For instance, it corresponds to 6MB of L3 shared among 4 cores. - #ifdef EIGEN_DEBUG_SMALL_PRODUCT_BLOCKS - const Index actual_l2 = l3; - #else - const Index actual_l2 = 1572864; // == 1.5 MB - #endif - - // Here, nc is chosen such that a block of kc x nc of the rhs fit within half of L2. - // The second half is implicitly reserved to access the result and lhs coefficients. - // When k= Index(Traits::nr*sizeof(RhsScalar))*k) - { - // L1 blocking - max_nc = remaining_l1 / (k*sizeof(RhsScalar)); - } - else - { - // L2 blocking - max_nc = (3*actual_l2)/(2*2*max_kc*sizeof(RhsScalar)); - } - // WARNING Below, we assume that Traits::nr is a power of two. - Index nc = numext::mini(actual_l2/(2*k*sizeof(RhsScalar)), max_nc) & (~(Traits::nr-1)); - if(n>nc) - { - // We are really blocking over the columns: - // -> reduce blocking size to make sure the last block is as large as possible - // while keeping the same number of sweeps over the packed lhs. - // Here we allow one more sweep if this gives us a perfect match, thus the commented "-1" - n = (n%nc)==0 ? nc - : (nc - Traits::nr * ((nc/*-1*/-(n%nc))/(Traits::nr*(n/nc+1)))); - } - else if(old_k==k) - { - // So far, no blocking at all, i.e., kc==k, and nc==n. - // In this case, let's perform a blocking over the rows such that the packed lhs data is kept in cache L1/L2 - // TODO: part of this blocking strategy is now implemented within the kernel itself, so the L1-based heuristic here should be obsolete. - Index problem_size = k*n*sizeof(LhsScalar); - Index actual_lm = actual_l2; - Index max_mc = m; - if(problem_size<=1024) - { - // problem is small enough to keep in L1 - // Let's choose m such that lhs's block fit in 1/3 of L1 - actual_lm = l1; + // Here, nc is chosen such that a block of kc x nc of the rhs fit within half of L2. + // The second half is implicitly reserved to access the result and lhs coefficients. + // When k= Index(Traits::nr * sizeof(RhsScalar)) * k) { + // L1 blocking + max_nc = remaining_l1 / (k * sizeof(RhsScalar)); + } else { + // L2 blocking + max_nc = (3 * actual_l2) / (2 * 2 * max_kc * sizeof(RhsScalar)); } - else if(l3!=0 && problem_size<=32768) - { - // we have both L2 and L3, and problem is small enough to be kept in L2 - // Let's choose m such that lhs's block fit in 1/3 of L2 - actual_lm = l2; - max_mc = (numext::mini)(576,max_mc); + // WARNING Below, we assume that Traits::nr is a power of two. + Index nc = numext::mini(actual_l2 / (2 * k * sizeof(RhsScalar)), max_nc) & (~(Traits::nr - 1)); + if (n > nc) { + // We are really blocking over the columns: + // -> reduce blocking size to make sure the last block is as large as possible + // while keeping the same number of sweeps over the packed lhs. + // Here we allow one more sweep if this gives us a perfect match, thus the commented "-1" + n = (n % nc) == 0 ? nc : (nc - Traits::nr * ((nc /*-1*/ - (n % nc)) / (Traits::nr * (n / nc + 1)))); + } else if (old_k == k) { + // So far, no blocking at all, i.e., kc==k, and nc==n. + // In this case, let's perform a blocking over the rows such that the packed lhs data is kept in cache L1/L2 + // TODO: part of this blocking strategy is now implemented within the kernel itself, so the L1-based heuristic + // here should be obsolete. + Index problem_size = k * n * sizeof(LhsScalar); + Index actual_lm = actual_l2; + Index max_mc = m; + if (problem_size <= 1024) { + // problem is small enough to keep in L1 + // Let's choose m such that lhs's block fit in 1/3 of L1 + actual_lm = l1; + } else if (l3 != 0 && problem_size <= 32768) { + // we have both L2 and L3, and problem is small enough to be kept in L2 + // Let's choose m such that lhs's block fit in 1/3 of L2 + actual_lm = l2; + max_mc = (numext::mini)(576, max_mc); + } + Index mc = (numext::mini)(actual_lm / (3 * k * sizeof(LhsScalar)), max_mc); + if (mc > Traits::mr) + mc -= mc % Traits::mr; + else if (mc == 0) + return; + m = (m % mc) == 0 ? mc : (mc - Traits::mr * ((mc /*-1*/ - (m % mc)) / (Traits::mr * (m / mc + 1)))); } - Index mc = (numext::mini)(actual_lm/(3*k*sizeof(LhsScalar)), max_mc); - if (mc > Traits::mr) mc -= mc % Traits::mr; - else if (mc==0) return; - m = (m%mc)==0 ? mc - : (mc - Traits::mr * ((mc/*-1*/-(m%mc))/(Traits::mr*(m/mc+1)))); } } -} -template -inline bool useSpecificBlockingSizes(Index& k, Index& m, Index& n) -{ + template inline bool useSpecificBlockingSizes(Index &k, Index &m, Index &n) + { #ifdef EIGEN_TEST_SPECIFIC_BLOCKING_SIZES - if (EIGEN_TEST_SPECIFIC_BLOCKING_SIZES) { - k = numext::mini(k, EIGEN_TEST_SPECIFIC_BLOCKING_SIZE_K); - m = numext::mini(m, EIGEN_TEST_SPECIFIC_BLOCKING_SIZE_M); - n = numext::mini(n, EIGEN_TEST_SPECIFIC_BLOCKING_SIZE_N); - return true; - } + if (EIGEN_TEST_SPECIFIC_BLOCKING_SIZES) { + k = numext::mini(k, EIGEN_TEST_SPECIFIC_BLOCKING_SIZE_K); + m = numext::mini(m, EIGEN_TEST_SPECIFIC_BLOCKING_SIZE_M); + n = numext::mini(n, EIGEN_TEST_SPECIFIC_BLOCKING_SIZE_N); + return true; + } #else - EIGEN_UNUSED_VARIABLE(k) - EIGEN_UNUSED_VARIABLE(m) - EIGEN_UNUSED_VARIABLE(n) + EIGEN_UNUSED_VARIABLE(k) + EIGEN_UNUSED_VARIABLE(m) + EIGEN_UNUSED_VARIABLE(n) #endif - return false; -} + return false; + } -/** \brief Computes the blocking parameters for a m x k times k x n matrix product - * - * \param[in,out] k Input: the third dimension of the product. Output: the blocking size along the same dimension. - * \param[in,out] m Input: the number of rows of the left hand side. Output: the blocking size along the same dimension. - * \param[in,out] n Input: the number of columns of the right hand side. Output: the blocking size along the same dimension. - * - * Given a m x k times k x n matrix product of scalar types \c LhsScalar and \c RhsScalar, - * this function computes the blocking size parameters along the respective dimensions - * for matrix products and related algorithms. - * - * The blocking size parameters may be evaluated: - * - either by a heuristic based on cache sizes; - * - or using fixed prescribed values (for testing purposes). - * - * \sa setCpuCacheSizes */ - -template -void computeProductBlockingSizes(Index& k, Index& m, Index& n, Index num_threads = 1) -{ - if (!useSpecificBlockingSizes(k, m, n)) { - evaluateProductBlockingSizesHeuristic(k, m, n, num_threads); + /** \brief Computes the blocking parameters for a m x k times k x n matrix product + * + * \param[in,out] k Input: the third dimension of the product. Output: the blocking size along the same dimension. + * \param[in,out] m Input: the number of rows of the left hand side. Output: the blocking size along the same + * dimension. + * \param[in,out] n Input: the number of columns of the right hand side. Output: the blocking size along the same + * dimension. + * + * Given a m x k times k x n matrix product of scalar types \c LhsScalar and \c RhsScalar, + * this function computes the blocking size parameters along the respective dimensions + * for matrix products and related algorithms. + * + * The blocking size parameters may be evaluated: + * - either by a heuristic based on cache sizes; + * - or using fixed prescribed values (for testing purposes). + * + * \sa setCpuCacheSizes */ + + template + void computeProductBlockingSizes(Index &k, Index &m, Index &n, Index num_threads = 1) + { + if (!useSpecificBlockingSizes(k, m, n)) { + evaluateProductBlockingSizesHeuristic(k, m, n, num_threads); + } } -} -template -inline void computeProductBlockingSizes(Index& k, Index& m, Index& n, Index num_threads = 1) -{ - computeProductBlockingSizes(k, m, n, num_threads); -} + template + inline void computeProductBlockingSizes(Index &k, Index &m, Index &n, Index num_threads = 1) + { + computeProductBlockingSizes(k, m, n, num_threads); + } #ifdef EIGEN_HAS_SINGLE_INSTRUCTION_CJMADD - #define CJMADD(CJ,A,B,C,T) C = CJ.pmadd(A,B,C); +#define CJMADD(CJ, A, B, C, T) C = CJ.pmadd(A, B, C); #else // FIXME (a bit overkill maybe ?) - template struct gebp_madd_selector { - EIGEN_ALWAYS_INLINE static void run(const CJ& cj, A& a, B& b, C& c, T& /*t*/) - { - c = cj.pmadd(a,b,c); - } + template struct gebp_madd_selector + { + EIGEN_ALWAYS_INLINE static void run(const CJ &cj, A &a, B &b, C &c, T & /*t*/) { c = cj.pmadd(a, b, c); } }; - template struct gebp_madd_selector { - EIGEN_ALWAYS_INLINE static void run(const CJ& cj, T& a, T& b, T& c, T& t) + template struct gebp_madd_selector + { + EIGEN_ALWAYS_INLINE static void run(const CJ &cj, T &a, T &b, T &c, T &t) { - t = b; t = cj.pmul(a,t); c = padd(c,t); + t = b; + t = cj.pmul(a, t); + c = padd(c, t); } }; template - EIGEN_STRONG_INLINE void gebp_madd(const CJ& cj, A& a, B& b, C& c, T& t) + EIGEN_STRONG_INLINE void gebp_madd(const CJ &cj, A &a, B &b, C &c, T &t) { - gebp_madd_selector::run(cj,a,b,c,t); + gebp_madd_selector::run(cj, a, b, c, t); } - #define CJMADD(CJ,A,B,C,T) gebp_madd(CJ,A,B,C,T); +#define CJMADD(CJ, A, B, C, T) gebp_madd(CJ, A, B, C, T); // #define CJMADD(CJ,A,B,C,T) T = B; T = CJ.pmul(A,T); C = padd(C,T); #endif -/* Vectorization logic - * real*real: unpack rhs to constant packets, ... - * - * cd*cd : unpack rhs to (b_r,b_r), (b_i,b_i), mul to get (a_r b_r,a_i b_r) (a_r b_i,a_i b_i), - * storing each res packet into two packets (2x2), - * at the end combine them: swap the second and addsub them - * cf*cf : same but with 2x4 blocks - * cplx*real : unpack rhs to constant packets, ... - * real*cplx : load lhs as (a0,a0,a1,a1), and mul as usual - */ -template -class gebp_traits -{ -public: - typedef _LhsScalar LhsScalar; - typedef _RhsScalar RhsScalar; - typedef typename ScalarBinaryOpTraits::ReturnType ResScalar; - - enum { - ConjLhs = _ConjLhs, - ConjRhs = _ConjRhs, - Vectorizable = packet_traits::Vectorizable && packet_traits::Vectorizable, - LhsPacketSize = Vectorizable ? packet_traits::size : 1, - RhsPacketSize = Vectorizable ? packet_traits::size : 1, - ResPacketSize = Vectorizable ? packet_traits::size : 1, - - NumberOfRegisters = EIGEN_ARCH_DEFAULT_NUMBER_OF_REGISTERS, - - // register block size along the N direction must be 1 or 4 - nr = 4, - - // register block size along the M direction (currently, this one cannot be modified) - default_mr = (EIGEN_PLAIN_ENUM_MIN(16,NumberOfRegisters)/2/nr)*LhsPacketSize, -#if defined(EIGEN_HAS_SINGLE_INSTRUCTION_MADD) && !defined(EIGEN_VECTORIZE_ALTIVEC) && !defined(EIGEN_VECTORIZE_VSX) - // we assume 16 registers - // See bug 992, if the scalar type is not vectorizable but that EIGEN_HAS_SINGLE_INSTRUCTION_MADD is defined, - // then using 3*LhsPacketSize triggers non-implemented paths in syrk. - mr = Vectorizable ? 3*LhsPacketSize : default_mr, -#else - mr = default_mr, -#endif - - LhsProgress = LhsPacketSize, - RhsProgress = 1 - }; - - typedef typename packet_traits::type _LhsPacket; - typedef typename packet_traits::type _RhsPacket; - typedef typename packet_traits::type _ResPacket; - - typedef typename conditional::type LhsPacket; - typedef typename conditional::type RhsPacket; - typedef typename conditional::type ResPacket; - - typedef ResPacket AccPacket; - - EIGEN_STRONG_INLINE void initAcc(AccPacket& p) - { - p = pset1(ResScalar(0)); - } - - EIGEN_STRONG_INLINE void broadcastRhs(const RhsScalar* b, RhsPacket& b0, RhsPacket& b1, RhsPacket& b2, RhsPacket& b3) - { - pbroadcast4(b, b0, b1, b2, b3); - } - -// EIGEN_STRONG_INLINE void broadcastRhs(const RhsScalar* b, RhsPacket& b0, RhsPacket& b1) -// { -// pbroadcast2(b, b0, b1); -// } - - template - EIGEN_STRONG_INLINE void loadRhs(const RhsScalar* b, RhsPacketType& dest) const - { - dest = pset1(*b); - } - - EIGEN_STRONG_INLINE void loadRhsQuad(const RhsScalar* b, RhsPacket& dest) const + /* Vectorization logic + * real*real: unpack rhs to constant packets, ... + * + * cd*cd : unpack rhs to (b_r,b_r), (b_i,b_i), mul to get (a_r b_r,a_i b_r) (a_r b_i,a_i b_i), + * storing each res packet into two packets (2x2), + * at the end combine them: swap the second and addsub them + * cf*cf : same but with 2x4 blocks + * cplx*real : unpack rhs to constant packets, ... + * real*cplx : load lhs as (a0,a0,a1,a1), and mul as usual + */ + template class gebp_traits { - dest = ploadquad(b); - } + public: + typedef _LhsScalar LhsScalar; + typedef _RhsScalar RhsScalar; + typedef typename ScalarBinaryOpTraits::ReturnType ResScalar; - template - EIGEN_STRONG_INLINE void loadLhs(const LhsScalar* a, LhsPacketType& dest) const - { - dest = pload(a); - } - - template - EIGEN_STRONG_INLINE void loadLhsUnaligned(const LhsScalar* a, LhsPacketType& dest) const - { - dest = ploadu(a); - } - - template - EIGEN_STRONG_INLINE void madd(const LhsPacketType& a, const RhsPacketType& b, AccPacketType& c, AccPacketType& tmp) const - { - conj_helper cj; - // It would be a lot cleaner to call pmadd all the time. Unfortunately if we - // let gcc allocate the register in which to store the result of the pmul - // (in the case where there is no FMA) gcc fails to figure out how to avoid - // spilling register. -#ifdef EIGEN_HAS_SINGLE_INSTRUCTION_MADD - EIGEN_UNUSED_VARIABLE(tmp); - c = cj.pmadd(a,b,c); -#else - tmp = b; tmp = cj.pmul(a,tmp); c = padd(c,tmp); -#endif - } + enum { + ConjLhs = _ConjLhs, + ConjRhs = _ConjRhs, + Vectorizable = packet_traits::Vectorizable && packet_traits::Vectorizable, + LhsPacketSize = Vectorizable ? packet_traits::size : 1, + RhsPacketSize = Vectorizable ? packet_traits::size : 1, + ResPacketSize = Vectorizable ? packet_traits::size : 1, - EIGEN_STRONG_INLINE void acc(const AccPacket& c, const ResPacket& alpha, ResPacket& r) const - { - r = pmadd(c,alpha,r); - } - - template - EIGEN_STRONG_INLINE void acc(const ResPacketHalf& c, const ResPacketHalf& alpha, ResPacketHalf& r) const - { - r = pmadd(c,alpha,r); - } + NumberOfRegisters = EIGEN_ARCH_DEFAULT_NUMBER_OF_REGISTERS, -}; + // register block size along the N direction must be 1 or 4 + nr = 4, -template -class gebp_traits, RealScalar, _ConjLhs, false> -{ -public: - typedef std::complex LhsScalar; - typedef RealScalar RhsScalar; - typedef typename ScalarBinaryOpTraits::ReturnType ResScalar; - - enum { - ConjLhs = _ConjLhs, - ConjRhs = false, - Vectorizable = packet_traits::Vectorizable && packet_traits::Vectorizable, - LhsPacketSize = Vectorizable ? packet_traits::size : 1, - RhsPacketSize = Vectorizable ? packet_traits::size : 1, - ResPacketSize = Vectorizable ? packet_traits::size : 1, - - NumberOfRegisters = EIGEN_ARCH_DEFAULT_NUMBER_OF_REGISTERS, - nr = 4, + // register block size along the M direction (currently, this one cannot be modified) + default_mr = (EIGEN_PLAIN_ENUM_MIN(16, NumberOfRegisters) / 2 / nr) * LhsPacketSize, #if defined(EIGEN_HAS_SINGLE_INSTRUCTION_MADD) && !defined(EIGEN_VECTORIZE_ALTIVEC) && !defined(EIGEN_VECTORIZE_VSX) - // we assume 16 registers - mr = 3*LhsPacketSize, + // we assume 16 registers + // See bug 992, if the scalar type is not vectorizable but that EIGEN_HAS_SINGLE_INSTRUCTION_MADD is defined, + // then using 3*LhsPacketSize triggers non-implemented paths in syrk. + mr = Vectorizable ? 3 * LhsPacketSize : default_mr, #else - mr = (EIGEN_PLAIN_ENUM_MIN(16,NumberOfRegisters)/2/nr)*LhsPacketSize, + mr = default_mr, #endif - LhsProgress = LhsPacketSize, - RhsProgress = 1 - }; + LhsProgress = LhsPacketSize, + RhsProgress = 1 + }; - typedef typename packet_traits::type _LhsPacket; - typedef typename packet_traits::type _RhsPacket; - typedef typename packet_traits::type _ResPacket; + typedef typename packet_traits::type _LhsPacket; + typedef typename packet_traits::type _RhsPacket; + typedef typename packet_traits::type _ResPacket; - typedef typename conditional::type LhsPacket; - typedef typename conditional::type RhsPacket; - typedef typename conditional::type ResPacket; + typedef typename conditional::type LhsPacket; + typedef typename conditional::type RhsPacket; + typedef typename conditional::type ResPacket; - typedef ResPacket AccPacket; + typedef ResPacket AccPacket; - EIGEN_STRONG_INLINE void initAcc(AccPacket& p) - { - p = pset1(ResScalar(0)); - } + EIGEN_STRONG_INLINE void initAcc(AccPacket &p) { p = pset1(ResScalar(0)); } - EIGEN_STRONG_INLINE void loadRhs(const RhsScalar* b, RhsPacket& dest) const - { - dest = pset1(*b); - } - - EIGEN_STRONG_INLINE void loadRhsQuad(const RhsScalar* b, RhsPacket& dest) const - { - dest = pset1(*b); - } + EIGEN_STRONG_INLINE void + broadcastRhs(const RhsScalar *b, RhsPacket &b0, RhsPacket &b1, RhsPacket &b2, RhsPacket &b3) + { + pbroadcast4(b, b0, b1, b2, b3); + } - EIGEN_STRONG_INLINE void loadLhs(const LhsScalar* a, LhsPacket& dest) const - { - dest = pload(a); - } + // EIGEN_STRONG_INLINE void broadcastRhs(const RhsScalar* b, RhsPacket& b0, RhsPacket& b1) + // { + // pbroadcast2(b, b0, b1); + // } - EIGEN_STRONG_INLINE void loadLhsUnaligned(const LhsScalar* a, LhsPacket& dest) const - { - dest = ploadu(a); - } + template EIGEN_STRONG_INLINE void loadRhs(const RhsScalar *b, RhsPacketType &dest) const + { + dest = pset1(*b); + } - EIGEN_STRONG_INLINE void broadcastRhs(const RhsScalar* b, RhsPacket& b0, RhsPacket& b1, RhsPacket& b2, RhsPacket& b3) - { - pbroadcast4(b, b0, b1, b2, b3); - } - -// EIGEN_STRONG_INLINE void broadcastRhs(const RhsScalar* b, RhsPacket& b0, RhsPacket& b1) -// { -// pbroadcast2(b, b0, b1); -// } + EIGEN_STRONG_INLINE void loadRhsQuad(const RhsScalar *b, RhsPacket &dest) const { dest = ploadquad(b); } - EIGEN_STRONG_INLINE void madd(const LhsPacket& a, const RhsPacket& b, AccPacket& c, RhsPacket& tmp) const - { - madd_impl(a, b, c, tmp, typename conditional::type()); - } + template EIGEN_STRONG_INLINE void loadLhs(const LhsScalar *a, LhsPacketType &dest) const + { + dest = pload(a); + } - EIGEN_STRONG_INLINE void madd_impl(const LhsPacket& a, const RhsPacket& b, AccPacket& c, RhsPacket& tmp, const true_type&) const - { + template + EIGEN_STRONG_INLINE void loadLhsUnaligned(const LhsScalar *a, LhsPacketType &dest) const + { + dest = ploadu(a); + } + + template + EIGEN_STRONG_INLINE void + madd(const LhsPacketType &a, const RhsPacketType &b, AccPacketType &c, AccPacketType &tmp) const + { + conj_helper cj; + // It would be a lot cleaner to call pmadd all the time. Unfortunately if we + // let gcc allocate the register in which to store the result of the pmul + // (in the case where there is no FMA) gcc fails to figure out how to avoid + // spilling register. #ifdef EIGEN_HAS_SINGLE_INSTRUCTION_MADD - EIGEN_UNUSED_VARIABLE(tmp); - c.v = pmadd(a.v,b,c.v); + EIGEN_UNUSED_VARIABLE(tmp); + c = cj.pmadd(a, b, c); #else - tmp = b; tmp = pmul(a.v,tmp); c.v = padd(c.v,tmp); + tmp = b; + tmp = cj.pmul(a, tmp); + c = padd(c, tmp); #endif - } + } - EIGEN_STRONG_INLINE void madd_impl(const LhsScalar& a, const RhsScalar& b, ResScalar& c, RhsScalar& /*tmp*/, const false_type&) const - { - c += a * b; - } + EIGEN_STRONG_INLINE void acc(const AccPacket &c, const ResPacket &alpha, ResPacket &r) const + { + r = pmadd(c, alpha, r); + } - EIGEN_STRONG_INLINE void acc(const AccPacket& c, const ResPacket& alpha, ResPacket& r) const + template + EIGEN_STRONG_INLINE void acc(const ResPacketHalf &c, const ResPacketHalf &alpha, ResPacketHalf &r) const + { + r = pmadd(c, alpha, r); + } + }; + + template class gebp_traits, RealScalar, _ConjLhs, false> { - r = cj.pmadd(c,alpha,r); - } + public: + typedef std::complex LhsScalar; + typedef RealScalar RhsScalar; + typedef typename ScalarBinaryOpTraits::ReturnType ResScalar; -protected: - conj_helper cj; -}; + enum { + ConjLhs = _ConjLhs, + ConjRhs = false, + Vectorizable = packet_traits::Vectorizable && packet_traits::Vectorizable, + LhsPacketSize = Vectorizable ? packet_traits::size : 1, + RhsPacketSize = Vectorizable ? packet_traits::size : 1, + ResPacketSize = Vectorizable ? packet_traits::size : 1, + + NumberOfRegisters = EIGEN_ARCH_DEFAULT_NUMBER_OF_REGISTERS, + nr = 4, +#if defined(EIGEN_HAS_SINGLE_INSTRUCTION_MADD) && !defined(EIGEN_VECTORIZE_ALTIVEC) && !defined(EIGEN_VECTORIZE_VSX) + // we assume 16 registers + mr = 3 * LhsPacketSize, +#else + mr = (EIGEN_PLAIN_ENUM_MIN(16, NumberOfRegisters) / 2 / nr) * LhsPacketSize, +#endif -template -struct DoublePacket -{ - Packet first; - Packet second; -}; + LhsProgress = LhsPacketSize, + RhsProgress = 1 + }; -template -DoublePacket padd(const DoublePacket &a, const DoublePacket &b) -{ - DoublePacket res; - res.first = padd(a.first, b.first); - res.second = padd(a.second,b.second); - return res; -} + typedef typename packet_traits::type _LhsPacket; + typedef typename packet_traits::type _RhsPacket; + typedef typename packet_traits::type _ResPacket; -template -const DoublePacket& predux_downto4(const DoublePacket &a) -{ - return a; -} + typedef typename conditional::type LhsPacket; + typedef typename conditional::type RhsPacket; + typedef typename conditional::type ResPacket; -template struct unpacket_traits > { typedef DoublePacket half; }; -// template -// DoublePacket pmadd(const DoublePacket &a, const DoublePacket &b) -// { -// DoublePacket res; -// res.first = padd(a.first, b.first); -// res.second = padd(a.second,b.second); -// return res; -// } - -template -class gebp_traits, std::complex, _ConjLhs, _ConjRhs > -{ -public: - typedef std::complex Scalar; - typedef std::complex LhsScalar; - typedef std::complex RhsScalar; - typedef std::complex ResScalar; - - enum { - ConjLhs = _ConjLhs, - ConjRhs = _ConjRhs, - Vectorizable = packet_traits::Vectorizable - && packet_traits::Vectorizable, - RealPacketSize = Vectorizable ? packet_traits::size : 1, - ResPacketSize = Vectorizable ? packet_traits::size : 1, - LhsPacketSize = Vectorizable ? packet_traits::size : 1, - RhsPacketSize = Vectorizable ? packet_traits::size : 1, - - // FIXME: should depend on NumberOfRegisters - nr = 4, - mr = ResPacketSize, - - LhsProgress = ResPacketSize, - RhsProgress = 1 - }; - - typedef typename packet_traits::type RealPacket; - typedef typename packet_traits::type ScalarPacket; - typedef DoublePacket DoublePacketType; - - typedef typename conditional::type LhsPacket; - typedef typename conditional::type RhsPacket; - typedef typename conditional::type ResPacket; - typedef typename conditional::type AccPacket; - - EIGEN_STRONG_INLINE void initAcc(Scalar& p) { p = Scalar(0); } - - EIGEN_STRONG_INLINE void initAcc(DoublePacketType& p) - { - p.first = pset1(RealScalar(0)); - p.second = pset1(RealScalar(0)); - } + typedef ResPacket AccPacket; - // Scalar path - EIGEN_STRONG_INLINE void loadRhs(const RhsScalar* b, ResPacket& dest) const - { - dest = pset1(*b); - } + EIGEN_STRONG_INLINE void initAcc(AccPacket &p) { p = pset1(ResScalar(0)); } - // Vectorized path - EIGEN_STRONG_INLINE void loadRhs(const RhsScalar* b, DoublePacketType& dest) const - { - dest.first = pset1(real(*b)); - dest.second = pset1(imag(*b)); - } - - EIGEN_STRONG_INLINE void loadRhsQuad(const RhsScalar* b, ResPacket& dest) const - { - loadRhs(b,dest); - } - EIGEN_STRONG_INLINE void loadRhsQuad(const RhsScalar* b, DoublePacketType& dest) const - { - eigen_internal_assert(unpacket_traits::size<=4); - loadRhs(b,dest); - } - - EIGEN_STRONG_INLINE void broadcastRhs(const RhsScalar* b, RhsPacket& b0, RhsPacket& b1, RhsPacket& b2, RhsPacket& b3) - { - // FIXME not sure that's the best way to implement it! - loadRhs(b+0, b0); - loadRhs(b+1, b1); - loadRhs(b+2, b2); - loadRhs(b+3, b3); - } - - // Vectorized path - EIGEN_STRONG_INLINE void broadcastRhs(const RhsScalar* b, DoublePacketType& b0, DoublePacketType& b1) - { - // FIXME not sure that's the best way to implement it! - loadRhs(b+0, b0); - loadRhs(b+1, b1); - } - - // Scalar path - EIGEN_STRONG_INLINE void broadcastRhs(const RhsScalar* b, RhsScalar& b0, RhsScalar& b1) - { - // FIXME not sure that's the best way to implement it! - loadRhs(b+0, b0); - loadRhs(b+1, b1); - } + EIGEN_STRONG_INLINE void loadRhs(const RhsScalar *b, RhsPacket &dest) const { dest = pset1(*b); } - // nothing special here - EIGEN_STRONG_INLINE void loadLhs(const LhsScalar* a, LhsPacket& dest) const - { - dest = pload((const typename unpacket_traits::type*)(a)); - } + EIGEN_STRONG_INLINE void loadRhsQuad(const RhsScalar *b, RhsPacket &dest) const { dest = pset1(*b); } - EIGEN_STRONG_INLINE void loadLhsUnaligned(const LhsScalar* a, LhsPacket& dest) const - { - dest = ploadu((const typename unpacket_traits::type*)(a)); - } + EIGEN_STRONG_INLINE void loadLhs(const LhsScalar *a, LhsPacket &dest) const { dest = pload(a); } - EIGEN_STRONG_INLINE void madd(const LhsPacket& a, const RhsPacket& b, DoublePacketType& c, RhsPacket& /*tmp*/) const - { - c.first = padd(pmul(a,b.first), c.first); - c.second = padd(pmul(a,b.second),c.second); - } + EIGEN_STRONG_INLINE void loadLhsUnaligned(const LhsScalar *a, LhsPacket &dest) const + { + dest = ploadu(a); + } - EIGEN_STRONG_INLINE void madd(const LhsPacket& a, const RhsPacket& b, ResPacket& c, RhsPacket& /*tmp*/) const - { - c = cj.pmadd(a,b,c); - } - - EIGEN_STRONG_INLINE void acc(const Scalar& c, const Scalar& alpha, Scalar& r) const { r += alpha * c; } - - EIGEN_STRONG_INLINE void acc(const DoublePacketType& c, const ResPacket& alpha, ResPacket& r) const - { - // assemble c - ResPacket tmp; - if((!ConjLhs)&&(!ConjRhs)) + EIGEN_STRONG_INLINE void + broadcastRhs(const RhsScalar *b, RhsPacket &b0, RhsPacket &b1, RhsPacket &b2, RhsPacket &b3) { - tmp = pcplxflip(pconj(ResPacket(c.second))); - tmp = padd(ResPacket(c.first),tmp); + pbroadcast4(b, b0, b1, b2, b3); } - else if((!ConjLhs)&&(ConjRhs)) + + // EIGEN_STRONG_INLINE void broadcastRhs(const RhsScalar* b, RhsPacket& b0, RhsPacket& b1) + // { + // pbroadcast2(b, b0, b1); + // } + + EIGEN_STRONG_INLINE void madd(const LhsPacket &a, const RhsPacket &b, AccPacket &c, RhsPacket &tmp) const { - tmp = pconj(pcplxflip(ResPacket(c.second))); - tmp = padd(ResPacket(c.first),tmp); + madd_impl(a, b, c, tmp, typename conditional::type()); } - else if((ConjLhs)&&(!ConjRhs)) + + EIGEN_STRONG_INLINE void + madd_impl(const LhsPacket &a, const RhsPacket &b, AccPacket &c, RhsPacket &tmp, const true_type &) const { - tmp = pcplxflip(ResPacket(c.second)); - tmp = padd(pconj(ResPacket(c.first)),tmp); +#ifdef EIGEN_HAS_SINGLE_INSTRUCTION_MADD + EIGEN_UNUSED_VARIABLE(tmp); + c.v = pmadd(a.v, b, c.v); +#else + tmp = b; + tmp = pmul(a.v, tmp); + c.v = padd(c.v, tmp); +#endif } - else if((ConjLhs)&&(ConjRhs)) + + EIGEN_STRONG_INLINE void + madd_impl(const LhsScalar &a, const RhsScalar &b, ResScalar &c, RhsScalar & /*tmp*/, const false_type &) const { - tmp = pcplxflip(ResPacket(c.second)); - tmp = psub(pconj(ResPacket(c.first)),tmp); + c += a * b; } - - r = pmadd(tmp,alpha,r); - } -protected: - conj_helper cj; -}; + EIGEN_STRONG_INLINE void acc(const AccPacket &c, const ResPacket &alpha, ResPacket &r) const + { + r = cj.pmadd(c, alpha, r); + } -template -class gebp_traits, false, _ConjRhs > -{ -public: - typedef std::complex Scalar; - typedef RealScalar LhsScalar; - typedef Scalar RhsScalar; - typedef Scalar ResScalar; - - enum { - ConjLhs = false, - ConjRhs = _ConjRhs, - Vectorizable = packet_traits::Vectorizable - && packet_traits::Vectorizable, - LhsPacketSize = Vectorizable ? packet_traits::size : 1, - RhsPacketSize = Vectorizable ? packet_traits::size : 1, - ResPacketSize = Vectorizable ? packet_traits::size : 1, - - NumberOfRegisters = EIGEN_ARCH_DEFAULT_NUMBER_OF_REGISTERS, - // FIXME: should depend on NumberOfRegisters - nr = 4, - mr = (EIGEN_PLAIN_ENUM_MIN(16,NumberOfRegisters)/2/nr)*ResPacketSize, - - LhsProgress = ResPacketSize, - RhsProgress = 1 + protected: + conj_helper cj; }; - typedef typename packet_traits::type _LhsPacket; - typedef typename packet_traits::type _RhsPacket; - typedef typename packet_traits::type _ResPacket; - - typedef typename conditional::type LhsPacket; - typedef typename conditional::type RhsPacket; - typedef typename conditional::type ResPacket; - - typedef ResPacket AccPacket; - - EIGEN_STRONG_INLINE void initAcc(AccPacket& p) + template struct DoublePacket { - p = pset1(ResScalar(0)); - } + Packet first; + Packet second; + }; - EIGEN_STRONG_INLINE void loadRhs(const RhsScalar* b, RhsPacket& dest) const + template DoublePacket padd(const DoublePacket &a, const DoublePacket &b) { - dest = pset1(*b); + DoublePacket res; + res.first = padd(a.first, b.first); + res.second = padd(a.second, b.second); + return res; } - - void broadcastRhs(const RhsScalar* b, RhsPacket& b0, RhsPacket& b1, RhsPacket& b2, RhsPacket& b3) - { - pbroadcast4(b, b0, b1, b2, b3); - } - -// EIGEN_STRONG_INLINE void broadcastRhs(const RhsScalar* b, RhsPacket& b0, RhsPacket& b1) -// { -// // FIXME not sure that's the best way to implement it! -// b0 = pload1(b+0); -// b1 = pload1(b+1); -// } - - EIGEN_STRONG_INLINE void loadLhs(const LhsScalar* a, LhsPacket& dest) const + + template const DoublePacket &predux_downto4(const DoublePacket &a) { return a; } + + template struct unpacket_traits> { - dest = ploaddup(a); - } - - EIGEN_STRONG_INLINE void loadRhsQuad(const RhsScalar* b, RhsPacket& dest) const + typedef DoublePacket half; + }; + // template + // DoublePacket pmadd(const DoublePacket &a, const DoublePacket &b) + // { + // DoublePacket res; + // res.first = padd(a.first, b.first); + // res.second = padd(a.second,b.second); + // return res; + // } + + template + class gebp_traits, std::complex, _ConjLhs, _ConjRhs> { - eigen_internal_assert(unpacket_traits::size<=4); - loadRhs(b,dest); - } + public: + typedef std::complex Scalar; + typedef std::complex LhsScalar; + typedef std::complex RhsScalar; + typedef std::complex ResScalar; - EIGEN_STRONG_INLINE void loadLhsUnaligned(const LhsScalar* a, LhsPacket& dest) const - { - dest = ploaddup(a); - } + enum { + ConjLhs = _ConjLhs, + ConjRhs = _ConjRhs, + Vectorizable = packet_traits::Vectorizable && packet_traits::Vectorizable, + RealPacketSize = Vectorizable ? packet_traits::size : 1, + ResPacketSize = Vectorizable ? packet_traits::size : 1, + LhsPacketSize = Vectorizable ? packet_traits::size : 1, + RhsPacketSize = Vectorizable ? packet_traits::size : 1, + + // FIXME: should depend on NumberOfRegisters + nr = 4, + mr = ResPacketSize, + + LhsProgress = ResPacketSize, + RhsProgress = 1 + }; - EIGEN_STRONG_INLINE void madd(const LhsPacket& a, const RhsPacket& b, AccPacket& c, RhsPacket& tmp) const - { - madd_impl(a, b, c, tmp, typename conditional::type()); - } + typedef typename packet_traits::type RealPacket; + typedef typename packet_traits::type ScalarPacket; + typedef DoublePacket DoublePacketType; - EIGEN_STRONG_INLINE void madd_impl(const LhsPacket& a, const RhsPacket& b, AccPacket& c, RhsPacket& tmp, const true_type&) const - { -#ifdef EIGEN_HAS_SINGLE_INSTRUCTION_MADD - EIGEN_UNUSED_VARIABLE(tmp); - c.v = pmadd(a,b.v,c.v); -#else - tmp = b; tmp.v = pmul(a,tmp.v); c = padd(c,tmp); -#endif - - } + typedef typename conditional::type LhsPacket; + typedef typename conditional::type RhsPacket; + typedef typename conditional::type ResPacket; + typedef typename conditional::type AccPacket; - EIGEN_STRONG_INLINE void madd_impl(const LhsScalar& a, const RhsScalar& b, ResScalar& c, RhsScalar& /*tmp*/, const false_type&) const - { - c += a * b; - } + EIGEN_STRONG_INLINE void initAcc(Scalar &p) { p = Scalar(0); } - EIGEN_STRONG_INLINE void acc(const AccPacket& c, const ResPacket& alpha, ResPacket& r) const - { - r = cj.pmadd(alpha,c,r); - } + EIGEN_STRONG_INLINE void initAcc(DoublePacketType &p) + { + p.first = pset1(RealScalar(0)); + p.second = pset1(RealScalar(0)); + } -protected: - conj_helper cj; -}; + // Scalar path + EIGEN_STRONG_INLINE void loadRhs(const RhsScalar *b, ResPacket &dest) const { dest = pset1(*b); } -/* optimized GEneral packed Block * packed Panel product kernel - * - * Mixing type logic: C += A * B - * | A | B | comments - * |real |cplx | no vectorization yet, would require to pack A with duplication - * |cplx |real | easy vectorization - */ -template -struct gebp_kernel -{ - typedef gebp_traits Traits; - typedef typename Traits::ResScalar ResScalar; - typedef typename Traits::LhsPacket LhsPacket; - typedef typename Traits::RhsPacket RhsPacket; - typedef typename Traits::ResPacket ResPacket; - typedef typename Traits::AccPacket AccPacket; - - typedef gebp_traits SwappedTraits; - typedef typename SwappedTraits::ResScalar SResScalar; - typedef typename SwappedTraits::LhsPacket SLhsPacket; - typedef typename SwappedTraits::RhsPacket SRhsPacket; - typedef typename SwappedTraits::ResPacket SResPacket; - typedef typename SwappedTraits::AccPacket SAccPacket; - - typedef typename DataMapper::LinearMapper LinearMapper; - - enum { - Vectorizable = Traits::Vectorizable, - LhsProgress = Traits::LhsProgress, - RhsProgress = Traits::RhsProgress, - ResPacketSize = Traits::ResPacketSize - }; + // Vectorized path + EIGEN_STRONG_INLINE void loadRhs(const RhsScalar *b, DoublePacketType &dest) const + { + dest.first = pset1(real(*b)); + dest.second = pset1(imag(*b)); + } - EIGEN_DONT_INLINE - void operator()(const DataMapper& res, const LhsScalar* blockA, const RhsScalar* blockB, - Index rows, Index depth, Index cols, ResScalar alpha, - Index strideA=-1, Index strideB=-1, Index offsetA=0, Index offsetB=0); -}; - -template -EIGEN_DONT_INLINE -void gebp_kernel - ::operator()(const DataMapper& res, const LhsScalar* blockA, const RhsScalar* blockB, - Index rows, Index depth, Index cols, ResScalar alpha, - Index strideA, Index strideB, Index offsetA, Index offsetB) - { - Traits traits; - SwappedTraits straits; - - if(strideA==-1) strideA = depth; - if(strideB==-1) strideB = depth; - conj_helper cj; - Index packet_cols4 = nr>=4 ? (cols/4) * 4 : 0; - const Index peeled_mc3 = mr>=3*Traits::LhsProgress ? (rows/(3*LhsProgress))*(3*LhsProgress) : 0; - const Index peeled_mc2 = mr>=2*Traits::LhsProgress ? peeled_mc3+((rows-peeled_mc3)/(2*LhsProgress))*(2*LhsProgress) : 0; - const Index peeled_mc1 = mr>=1*Traits::LhsProgress ? (rows/(1*LhsProgress))*(1*LhsProgress) : 0; - enum { pk = 8 }; // NOTE Such a large peeling factor is important for large matrices (~ +5% when >1000 on Haswell) - const Index peeled_kc = depth & ~(pk-1); - const Index prefetch_res_offset = 32/sizeof(ResScalar); -// const Index depth2 = depth & ~1; + EIGEN_STRONG_INLINE void loadRhsQuad(const RhsScalar *b, ResPacket &dest) const { loadRhs(b, dest); } + EIGEN_STRONG_INLINE void loadRhsQuad(const RhsScalar *b, DoublePacketType &dest) const + { + eigen_internal_assert(unpacket_traits::size <= 4); + loadRhs(b, dest); + } - //---------- Process 3 * LhsProgress rows at once ---------- - // This corresponds to 3*LhsProgress x nr register blocks. - // Usually, make sense only with FMA - if(mr>=3*Traits::LhsProgress) + EIGEN_STRONG_INLINE void + broadcastRhs(const RhsScalar *b, RhsPacket &b0, RhsPacket &b1, RhsPacket &b2, RhsPacket &b3) { - // Here, the general idea is to loop on each largest micro horizontal panel of the lhs (3*Traits::LhsProgress x depth) - // and on each largest micro vertical panel of the rhs (depth * nr). - // Blocking sizes, i.e., 'depth' has been computed so that the micro horizontal panel of the lhs fit in L1. - // However, if depth is too small, we can extend the number of rows of these horizontal panels. - // This actual number of rows is computed as follow: - const Index l1 = defaultL1CacheSize; // in Bytes, TODO, l1 should be passed to this function. - // The max(1, ...) here is needed because we may be using blocking params larger than what our known l1 cache size - // suggests we should be using: either because our known l1 cache size is inaccurate (e.g. on Android, we can only guess), - // or because we are testing specific blocking sizes. - const Index actual_panel_rows = (3*LhsProgress) * std::max(1,( (l1 - sizeof(ResScalar)*mr*nr - depth*nr*sizeof(RhsScalar)) / (depth * sizeof(LhsScalar) * 3*LhsProgress) )); - for(Index i1=0; i1((const typename unpacket_traits::type *)(a)); + } - // performs "inner" products - const RhsScalar* blB = &blockB[j2*strideB+offsetB*nr]; - prefetch(&blB[0]); - LhsPacket A0, A1; + EIGEN_STRONG_INLINE void loadLhsUnaligned(const LhsScalar *a, LhsPacket &dest) const + { + dest = ploadu((const typename unpacket_traits::type *)(a)); + } - for(Index k=0; k(alpha); + EIGEN_STRONG_INLINE void acc(const Scalar &c, const Scalar &alpha, Scalar &r) const { r += alpha * c; } - R0 = r0.loadPacket(0 * Traits::ResPacketSize); - R1 = r0.loadPacket(1 * Traits::ResPacketSize); - R2 = r0.loadPacket(2 * Traits::ResPacketSize); - traits.acc(C0, alphav, R0); - traits.acc(C4, alphav, R1); - traits.acc(C8, alphav, R2); - r0.storePacket(0 * Traits::ResPacketSize, R0); - r0.storePacket(1 * Traits::ResPacketSize, R1); - r0.storePacket(2 * Traits::ResPacketSize, R2); - - R0 = r1.loadPacket(0 * Traits::ResPacketSize); - R1 = r1.loadPacket(1 * Traits::ResPacketSize); - R2 = r1.loadPacket(2 * Traits::ResPacketSize); - traits.acc(C1, alphav, R0); - traits.acc(C5, alphav, R1); - traits.acc(C9, alphav, R2); - r1.storePacket(0 * Traits::ResPacketSize, R0); - r1.storePacket(1 * Traits::ResPacketSize, R1); - r1.storePacket(2 * Traits::ResPacketSize, R2); + EIGEN_STRONG_INLINE void acc(const DoublePacketType &c, const ResPacket &alpha, ResPacket &r) const + { + // assemble c + ResPacket tmp; + if ((!ConjLhs) && (!ConjRhs)) { + tmp = pcplxflip(pconj(ResPacket(c.second))); + tmp = padd(ResPacket(c.first), tmp); + } else if ((!ConjLhs) && (ConjRhs)) { + tmp = pconj(pcplxflip(ResPacket(c.second))); + tmp = padd(ResPacket(c.first), tmp); + } else if ((ConjLhs) && (!ConjRhs)) { + tmp = pcplxflip(ResPacket(c.second)); + tmp = padd(pconj(ResPacket(c.first)), tmp); + } else if ((ConjLhs) && (ConjRhs)) { + tmp = pcplxflip(ResPacket(c.second)); + tmp = psub(pconj(ResPacket(c.first)), tmp); + } - R0 = r2.loadPacket(0 * Traits::ResPacketSize); - R1 = r2.loadPacket(1 * Traits::ResPacketSize); - R2 = r2.loadPacket(2 * Traits::ResPacketSize); - traits.acc(C2, alphav, R0); - traits.acc(C6, alphav, R1); - traits.acc(C10, alphav, R2); - r2.storePacket(0 * Traits::ResPacketSize, R0); - r2.storePacket(1 * Traits::ResPacketSize, R1); - r2.storePacket(2 * Traits::ResPacketSize, R2); - - R0 = r3.loadPacket(0 * Traits::ResPacketSize); - R1 = r3.loadPacket(1 * Traits::ResPacketSize); - R2 = r3.loadPacket(2 * Traits::ResPacketSize); - traits.acc(C3, alphav, R0); - traits.acc(C7, alphav, R1); - traits.acc(C11, alphav, R2); - r3.storePacket(0 * Traits::ResPacketSize, R0); - r3.storePacket(1 * Traits::ResPacketSize, R1); - r3.storePacket(2 * Traits::ResPacketSize, R2); - } - } + r = pmadd(tmp, alpha, r); + } - // Deal with remaining columns of the rhs - for(Index j2=packet_cols4; j2 cj; + }; - // gets res block as register - AccPacket C0, C4, C8; - traits.initAcc(C0); - traits.initAcc(C4); - traits.initAcc(C8); + template class gebp_traits, false, _ConjRhs> + { + public: + typedef std::complex Scalar; + typedef RealScalar LhsScalar; + typedef Scalar RhsScalar; + typedef Scalar ResScalar; - LinearMapper r0 = res.getLinearMapper(i, j2); - r0.prefetch(0); + enum { + ConjLhs = false, + ConjRhs = _ConjRhs, + Vectorizable = packet_traits::Vectorizable && packet_traits::Vectorizable, + LhsPacketSize = Vectorizable ? packet_traits::size : 1, + RhsPacketSize = Vectorizable ? packet_traits::size : 1, + ResPacketSize = Vectorizable ? packet_traits::size : 1, + + NumberOfRegisters = EIGEN_ARCH_DEFAULT_NUMBER_OF_REGISTERS, + // FIXME: should depend on NumberOfRegisters + nr = 4, + mr = (EIGEN_PLAIN_ENUM_MIN(16, NumberOfRegisters) / 2 / nr) * ResPacketSize, + + LhsProgress = ResPacketSize, + RhsProgress = 1 + }; - // performs "inner" products - const RhsScalar* blB = &blockB[j2*strideB+offsetB]; - LhsPacket A0, A1, A2; - - for(Index k=0; k::type _LhsPacket; + typedef typename packet_traits::type _RhsPacket; + typedef typename packet_traits::type _ResPacket; - blB += pk*RhsProgress; - blA += pk*3*Traits::LhsProgress; + typedef typename conditional::type LhsPacket; + typedef typename conditional::type RhsPacket; + typedef typename conditional::type ResPacket; - EIGEN_ASM_COMMENT("end gebp micro kernel 3pX1"); - } + typedef ResPacket AccPacket; - // process remaining peeled loop - for(Index k=peeled_kc; k(alpha); + EIGEN_STRONG_INLINE void initAcc(AccPacket &p) { p = pset1(ResScalar(0)); } - R0 = r0.loadPacket(0 * Traits::ResPacketSize); - R1 = r0.loadPacket(1 * Traits::ResPacketSize); - R2 = r0.loadPacket(2 * Traits::ResPacketSize); - traits.acc(C0, alphav, R0); - traits.acc(C4, alphav, R1); - traits.acc(C8, alphav, R2); - r0.storePacket(0 * Traits::ResPacketSize, R0); - r0.storePacket(1 * Traits::ResPacketSize, R1); - r0.storePacket(2 * Traits::ResPacketSize, R2); - } - } - } + EIGEN_STRONG_INLINE void loadRhs(const RhsScalar *b, RhsPacket &dest) const { dest = pset1(*b); } + + void broadcastRhs(const RhsScalar *b, RhsPacket &b0, RhsPacket &b1, RhsPacket &b2, RhsPacket &b3) + { + pbroadcast4(b, b0, b1, b2, b3); } - //---------- Process 2 * LhsProgress rows at once ---------- - if(mr>=2*Traits::LhsProgress) + // EIGEN_STRONG_INLINE void broadcastRhs(const RhsScalar* b, RhsPacket& b0, RhsPacket& b1) + // { + // // FIXME not sure that's the best way to implement it! + // b0 = pload1(b+0); + // b1 = pload1(b+1); + // } + + EIGEN_STRONG_INLINE void loadLhs(const LhsScalar *a, LhsPacket &dest) const { dest = ploaddup(a); } + + EIGEN_STRONG_INLINE void loadRhsQuad(const RhsScalar *b, RhsPacket &dest) const { - const Index l1 = defaultL1CacheSize; // in Bytes, TODO, l1 should be passed to this function. - // The max(1, ...) here is needed because we may be using blocking params larger than what our known l1 cache size - // suggests we should be using: either because our known l1 cache size is inaccurate (e.g. on Android, we can only guess), - // or because we are testing specific blocking sizes. - Index actual_panel_rows = (2*LhsProgress) * std::max(1,( (l1 - sizeof(ResScalar)*mr*nr - depth*nr*sizeof(RhsScalar)) / (depth * sizeof(LhsScalar) * 2*LhsProgress) )); - - for(Index i1=peeled_mc3; i1::size <= 4); + loadRhs(b, dest); + } - // gets res block as register - AccPacket C0, C1, C2, C3, - C4, C5, C6, C7; - traits.initAcc(C0); traits.initAcc(C1); traits.initAcc(C2); traits.initAcc(C3); - traits.initAcc(C4); traits.initAcc(C5); traits.initAcc(C6); traits.initAcc(C7); + EIGEN_STRONG_INLINE void loadLhsUnaligned(const LhsScalar *a, LhsPacket &dest) const + { + dest = ploaddup(a); + } - LinearMapper r0 = res.getLinearMapper(i, j2 + 0); - LinearMapper r1 = res.getLinearMapper(i, j2 + 1); - LinearMapper r2 = res.getLinearMapper(i, j2 + 2); - LinearMapper r3 = res.getLinearMapper(i, j2 + 3); + EIGEN_STRONG_INLINE void madd(const LhsPacket &a, const RhsPacket &b, AccPacket &c, RhsPacket &tmp) const + { + madd_impl(a, b, c, tmp, typename conditional::type()); + } - r0.prefetch(prefetch_res_offset); - r1.prefetch(prefetch_res_offset); - r2.prefetch(prefetch_res_offset); - r3.prefetch(prefetch_res_offset); + EIGEN_STRONG_INLINE void + madd_impl(const LhsPacket &a, const RhsPacket &b, AccPacket &c, RhsPacket &tmp, const true_type &) const + { +#ifdef EIGEN_HAS_SINGLE_INSTRUCTION_MADD + EIGEN_UNUSED_VARIABLE(tmp); + c.v = pmadd(a, b.v, c.v); +#else + tmp = b; + tmp.v = pmul(a, tmp.v); + c = padd(c, tmp); +#endif + } - // performs "inner" products - const RhsScalar* blB = &blockB[j2*strideB+offsetB*nr]; - prefetch(&blB[0]); - LhsPacket A0, A1; + EIGEN_STRONG_INLINE void + madd_impl(const LhsScalar &a, const RhsScalar &b, ResScalar &c, RhsScalar & /*tmp*/, const false_type &) const + { + c += a * b; + } - for(Index k=0; k=6 without FMA (bug 1637) - #if EIGEN_GNUC_AT_LEAST(6,0) && defined(EIGEN_VECTORIZE_SSE) - #define EIGEN_GEBP_2PX4_SPILLING_WORKAROUND __asm__ ("" : [a0] "+x,m" (A0),[a1] "+x,m" (A1)); - #else - #define EIGEN_GEBP_2PX4_SPILLING_WORKAROUND - #endif - #define EIGEN_GEBGP_ONESTEP(K) \ - do { \ - EIGEN_ASM_COMMENT("begin step of gebp micro kernel 2pX4"); \ - traits.loadLhs(&blA[(0+2*K)*LhsProgress], A0); \ - traits.loadLhs(&blA[(1+2*K)*LhsProgress], A1); \ - traits.broadcastRhs(&blB[(0+4*K)*RhsProgress], B_0, B1, B2, B3); \ - traits.madd(A0, B_0, C0, T0); \ - traits.madd(A1, B_0, C4, B_0); \ - traits.madd(A0, B1, C1, T0); \ - traits.madd(A1, B1, C5, B1); \ - traits.madd(A0, B2, C2, T0); \ - traits.madd(A1, B2, C6, B2); \ - traits.madd(A0, B3, C3, T0); \ - traits.madd(A1, B3, C7, B3); \ - EIGEN_GEBP_2PX4_SPILLING_WORKAROUND \ - EIGEN_ASM_COMMENT("end step of gebp micro kernel 2pX4"); \ - } while(false) - - internal::prefetch(blB+(48+0)); - EIGEN_GEBGP_ONESTEP(0); - EIGEN_GEBGP_ONESTEP(1); - EIGEN_GEBGP_ONESTEP(2); - EIGEN_GEBGP_ONESTEP(3); - internal::prefetch(blB+(48+16)); - EIGEN_GEBGP_ONESTEP(4); - EIGEN_GEBGP_ONESTEP(5); - EIGEN_GEBGP_ONESTEP(6); - EIGEN_GEBGP_ONESTEP(7); + EIGEN_STRONG_INLINE void acc(const AccPacket &c, const ResPacket &alpha, ResPacket &r) const + { + r = cj.pmadd(alpha, c, r); + } - blB += pk*4*RhsProgress; - blA += pk*(2*Traits::LhsProgress); + protected: + conj_helper cj; + }; - EIGEN_ASM_COMMENT("end gebp micro kernel 2pX4"); - } - // process remaining peeled loop - for(Index k=peeled_kc; k + struct gebp_kernel + { + typedef gebp_traits Traits; + typedef typename Traits::ResScalar ResScalar; + typedef typename Traits::LhsPacket LhsPacket; + typedef typename Traits::RhsPacket RhsPacket; + typedef typename Traits::ResPacket ResPacket; + typedef typename Traits::AccPacket AccPacket; - ResPacket R0, R1, R2, R3; - ResPacket alphav = pset1(alpha); + typedef gebp_traits SwappedTraits; + typedef typename SwappedTraits::ResScalar SResScalar; + typedef typename SwappedTraits::LhsPacket SLhsPacket; + typedef typename SwappedTraits::RhsPacket SRhsPacket; + typedef typename SwappedTraits::ResPacket SResPacket; + typedef typename SwappedTraits::AccPacket SAccPacket; - R0 = r0.loadPacket(0 * Traits::ResPacketSize); - R1 = r0.loadPacket(1 * Traits::ResPacketSize); - R2 = r1.loadPacket(0 * Traits::ResPacketSize); - R3 = r1.loadPacket(1 * Traits::ResPacketSize); - traits.acc(C0, alphav, R0); - traits.acc(C4, alphav, R1); - traits.acc(C1, alphav, R2); - traits.acc(C5, alphav, R3); - r0.storePacket(0 * Traits::ResPacketSize, R0); - r0.storePacket(1 * Traits::ResPacketSize, R1); - r1.storePacket(0 * Traits::ResPacketSize, R2); - r1.storePacket(1 * Traits::ResPacketSize, R3); + typedef typename DataMapper::LinearMapper LinearMapper; - R0 = r2.loadPacket(0 * Traits::ResPacketSize); - R1 = r2.loadPacket(1 * Traits::ResPacketSize); - R2 = r3.loadPacket(0 * Traits::ResPacketSize); - R3 = r3.loadPacket(1 * Traits::ResPacketSize); - traits.acc(C2, alphav, R0); - traits.acc(C6, alphav, R1); - traits.acc(C3, alphav, R2); - traits.acc(C7, alphav, R3); - r2.storePacket(0 * Traits::ResPacketSize, R0); - r2.storePacket(1 * Traits::ResPacketSize, R1); - r3.storePacket(0 * Traits::ResPacketSize, R2); - r3.storePacket(1 * Traits::ResPacketSize, R3); - } - } - - // Deal with remaining columns of the rhs - for(Index j2=packet_cols4; j2 + EIGEN_DONT_INLINE void + gebp_kernel::operator()( + const DataMapper &res, + const LhsScalar *blockA, + const RhsScalar *blockB, + Index rows, + Index depth, + Index cols, + ResScalar alpha, + Index strideA, + Index strideB, + Index offsetA, + Index offsetB) + { + Traits traits; + SwappedTraits straits; - // performs "inner" products - const RhsScalar* blB = &blockB[j2*strideB+offsetB]; - LhsPacket A0, A1; + if (strideA == -1) strideA = depth; + if (strideB == -1) strideB = depth; + conj_helper cj; + Index packet_cols4 = nr >= 4 ? (cols / 4) * 4 : 0; + const Index peeled_mc3 = mr >= 3 * Traits::LhsProgress ? (rows / (3 * LhsProgress)) * (3 * LhsProgress) : 0; + const Index peeled_mc2 = + mr >= 2 * Traits::LhsProgress ? peeled_mc3 + ((rows - peeled_mc3) / (2 * LhsProgress)) * (2 * LhsProgress) : 0; + const Index peeled_mc1 = mr >= 1 * Traits::LhsProgress ? (rows / (1 * LhsProgress)) * (1 * LhsProgress) : 0; + enum { pk = 8 };// NOTE Such a large peeling factor is important for large matrices (~ +5% when >1000 on Haswell) + const Index peeled_kc = depth & ~(pk - 1); + const Index prefetch_res_offset = 32 / sizeof(ResScalar); + // const Index depth2 = depth & ~1; - for(Index k=0; k= 3 * Traits::LhsProgress) { + // Here, the general idea is to loop on each largest micro horizontal panel of the lhs (3*Traits::LhsProgress x + // depth) and on each largest micro vertical panel of the rhs (depth * nr). Blocking sizes, i.e., 'depth' has been + // computed so that the micro horizontal panel of the lhs fit in L1. However, if depth is too small, we can extend + // the number of rows of these horizontal panels. This actual number of rows is computed as follow: + const Index l1 = defaultL1CacheSize;// in Bytes, TODO, l1 should be passed to this function. + // The max(1, ...) here is needed because we may be using blocking params larger than what our known l1 cache size + // suggests we should be using: either because our known l1 cache size is inaccurate (e.g. on Android, we can only + // guess), or because we are testing specific blocking sizes. + const Index actual_panel_rows = (3 * LhsProgress) + * std::max(1, + ((l1 - sizeof(ResScalar) * mr * nr - depth * nr * sizeof(RhsScalar)) + / (depth * sizeof(LhsScalar) * 3 * LhsProgress))); + for (Index i1 = 0; i1 < peeled_mc3; i1 += actual_panel_rows) { + const Index actual_panel_end = (std::min)(i1 + actual_panel_rows, peeled_mc3); + for (Index j2 = 0; j2 < packet_cols4; j2 += nr) { + for (Index i = i1; i < actual_panel_end; i += 3 * LhsProgress) { + + // We selected a 3*Traits::LhsProgress x nr micro block of res which is entirely + // stored into 3 x nr registers. + + const LhsScalar *blA = &blockA[i * strideA + offsetA * (3 * LhsProgress)]; + prefetch(&blA[0]); + + // gets res block as register + AccPacket C0, C1, C2, C3, C4, C5, C6, C7, C8, C9, C10, C11; + traits.initAcc(C0); + traits.initAcc(C1); + traits.initAcc(C2); + traits.initAcc(C3); + traits.initAcc(C4); + traits.initAcc(C5); + traits.initAcc(C6); + traits.initAcc(C7); + traits.initAcc(C8); + traits.initAcc(C9); + traits.initAcc(C10); + traits.initAcc(C11); + + LinearMapper r0 = res.getLinearMapper(i, j2 + 0); + LinearMapper r1 = res.getLinearMapper(i, j2 + 1); + LinearMapper r2 = res.getLinearMapper(i, j2 + 2); + LinearMapper r3 = res.getLinearMapper(i, j2 + 3); + + r0.prefetch(0); + r1.prefetch(0); + r2.prefetch(0); + r3.prefetch(0); + + // performs "inner" products + const RhsScalar *blB = &blockB[j2 * strideB + offsetB * nr]; + prefetch(&blB[0]); + LhsPacket A0, A1; + + for (Index k = 0; k < peeled_kc; k += pk) { + EIGEN_ASM_COMMENT("begin gebp micro kernel 3pX4"); + RhsPacket B_0, T0; + LhsPacket A2; + +#define EIGEN_GEBP_ONESTEP(K) \ + do { \ + EIGEN_ASM_COMMENT("begin step of gebp micro kernel 3pX4"); \ + EIGEN_ASM_COMMENT("Note: these asm comments work around bug 935!"); \ + internal::prefetch(blA + (3 * K + 16) * LhsProgress); \ + if (EIGEN_ARCH_ARM) { internal::prefetch(blB + (4 * K + 16) * RhsProgress); } /* Bug 953 */ \ + traits.loadLhs(&blA[(0 + 3 * K) * LhsProgress], A0); \ + traits.loadLhs(&blA[(1 + 3 * K) * LhsProgress], A1); \ + traits.loadLhs(&blA[(2 + 3 * K) * LhsProgress], A2); \ + traits.loadRhs(blB + (0 + 4 * K) * Traits::RhsProgress, B_0); \ + traits.madd(A0, B_0, C0, T0); \ + traits.madd(A1, B_0, C4, T0); \ + traits.madd(A2, B_0, C8, B_0); \ + traits.loadRhs(blB + (1 + 4 * K) * Traits::RhsProgress, B_0); \ + traits.madd(A0, B_0, C1, T0); \ + traits.madd(A1, B_0, C5, T0); \ + traits.madd(A2, B_0, C9, B_0); \ + traits.loadRhs(blB + (2 + 4 * K) * Traits::RhsProgress, B_0); \ + traits.madd(A0, B_0, C2, T0); \ + traits.madd(A1, B_0, C6, T0); \ + traits.madd(A2, B_0, C10, B_0); \ + traits.loadRhs(blB + (3 + 4 * K) * Traits::RhsProgress, B_0); \ + traits.madd(A0, B_0, C3, T0); \ + traits.madd(A1, B_0, C7, T0); \ + traits.madd(A2, B_0, C11, B_0); \ + EIGEN_ASM_COMMENT("end step of gebp micro kernel 3pX4"); \ + } while (false) + + internal::prefetch(blB); + EIGEN_GEBP_ONESTEP(0); + EIGEN_GEBP_ONESTEP(1); + EIGEN_GEBP_ONESTEP(2); + EIGEN_GEBP_ONESTEP(3); + EIGEN_GEBP_ONESTEP(4); + EIGEN_GEBP_ONESTEP(5); + EIGEN_GEBP_ONESTEP(6); + EIGEN_GEBP_ONESTEP(7); + + blB += pk * 4 * RhsProgress; + blA += pk * 3 * Traits::LhsProgress; + + EIGEN_ASM_COMMENT("end gebp micro kernel 3pX4"); + } + // process remaining peeled loop + for (Index k = peeled_kc; k < depth; k++) { + RhsPacket B_0, T0; + LhsPacket A2; + EIGEN_GEBP_ONESTEP(0); + blB += 4 * RhsProgress; + blA += 3 * Traits::LhsProgress; + } - blB += pk*RhsProgress; - blA += pk*2*Traits::LhsProgress; +#undef EIGEN_GEBP_ONESTEP - EIGEN_ASM_COMMENT("end gebp micro kernel 2pX1"); + ResPacket R0, R1, R2; + ResPacket alphav = pset1(alpha); + + R0 = r0.loadPacket(0 * Traits::ResPacketSize); + R1 = r0.loadPacket(1 * Traits::ResPacketSize); + R2 = r0.loadPacket(2 * Traits::ResPacketSize); + traits.acc(C0, alphav, R0); + traits.acc(C4, alphav, R1); + traits.acc(C8, alphav, R2); + r0.storePacket(0 * Traits::ResPacketSize, R0); + r0.storePacket(1 * Traits::ResPacketSize, R1); + r0.storePacket(2 * Traits::ResPacketSize, R2); + + R0 = r1.loadPacket(0 * Traits::ResPacketSize); + R1 = r1.loadPacket(1 * Traits::ResPacketSize); + R2 = r1.loadPacket(2 * Traits::ResPacketSize); + traits.acc(C1, alphav, R0); + traits.acc(C5, alphav, R1); + traits.acc(C9, alphav, R2); + r1.storePacket(0 * Traits::ResPacketSize, R0); + r1.storePacket(1 * Traits::ResPacketSize, R1); + r1.storePacket(2 * Traits::ResPacketSize, R2); + + R0 = r2.loadPacket(0 * Traits::ResPacketSize); + R1 = r2.loadPacket(1 * Traits::ResPacketSize); + R2 = r2.loadPacket(2 * Traits::ResPacketSize); + traits.acc(C2, alphav, R0); + traits.acc(C6, alphav, R1); + traits.acc(C10, alphav, R2); + r2.storePacket(0 * Traits::ResPacketSize, R0); + r2.storePacket(1 * Traits::ResPacketSize, R1); + r2.storePacket(2 * Traits::ResPacketSize, R2); + + R0 = r3.loadPacket(0 * Traits::ResPacketSize); + R1 = r3.loadPacket(1 * Traits::ResPacketSize); + R2 = r3.loadPacket(2 * Traits::ResPacketSize); + traits.acc(C3, alphav, R0); + traits.acc(C7, alphav, R1); + traits.acc(C11, alphav, R2); + r3.storePacket(0 * Traits::ResPacketSize, R0); + r3.storePacket(1 * Traits::ResPacketSize, R1); + r3.storePacket(2 * Traits::ResPacketSize, R2); } + } - // process remaining peeled loop - for(Index k=peeled_kc; k(alpha); + + R0 = r0.loadPacket(0 * Traits::ResPacketSize); + R1 = r0.loadPacket(1 * Traits::ResPacketSize); + R2 = r0.loadPacket(2 * Traits::ResPacketSize); + traits.acc(C0, alphav, R0); + traits.acc(C4, alphav, R1); + traits.acc(C8, alphav, R2); + r0.storePacket(0 * Traits::ResPacketSize, R0); + r0.storePacket(1 * Traits::ResPacketSize, R1); + r0.storePacket(2 * Traits::ResPacketSize, R2); } + } + } + } + + //---------- Process 2 * LhsProgress rows at once ---------- + if (mr >= 2 * Traits::LhsProgress) { + const Index l1 = defaultL1CacheSize;// in Bytes, TODO, l1 should be passed to this function. + // The max(1, ...) here is needed because we may be using blocking params larger than what our known l1 cache size + // suggests we should be using: either because our known l1 cache size is inaccurate (e.g. on Android, we can only + // guess), or because we are testing specific blocking sizes. + Index actual_panel_rows = (2 * LhsProgress) + * std::max(1, + ((l1 - sizeof(ResScalar) * mr * nr - depth * nr * sizeof(RhsScalar)) + / (depth * sizeof(LhsScalar) * 2 * LhsProgress))); + + for (Index i1 = peeled_mc3; i1 < peeled_mc2; i1 += actual_panel_rows) { + Index actual_panel_end = (std::min)(i1 + actual_panel_rows, peeled_mc2); + for (Index j2 = 0; j2 < packet_cols4; j2 += nr) { + for (Index i = i1; i < actual_panel_end; i += 2 * LhsProgress) { + + // We selected a 2*Traits::LhsProgress x nr micro block of res which is entirely + // stored into 2 x nr registers. + + const LhsScalar *blA = &blockA[i * strideA + offsetA * (2 * Traits::LhsProgress)]; + prefetch(&blA[0]); + + // gets res block as register + AccPacket C0, C1, C2, C3, C4, C5, C6, C7; + traits.initAcc(C0); + traits.initAcc(C1); + traits.initAcc(C2); + traits.initAcc(C3); + traits.initAcc(C4); + traits.initAcc(C5); + traits.initAcc(C6); + traits.initAcc(C7); + + LinearMapper r0 = res.getLinearMapper(i, j2 + 0); + LinearMapper r1 = res.getLinearMapper(i, j2 + 1); + LinearMapper r2 = res.getLinearMapper(i, j2 + 2); + LinearMapper r3 = res.getLinearMapper(i, j2 + 3); + + r0.prefetch(prefetch_res_offset); + r1.prefetch(prefetch_res_offset); + r2.prefetch(prefetch_res_offset); + r3.prefetch(prefetch_res_offset); + + // performs "inner" products + const RhsScalar *blB = &blockB[j2 * strideB + offsetB * nr]; + prefetch(&blB[0]); + LhsPacket A0, A1; + + for (Index k = 0; k < peeled_kc; k += pk) { + EIGEN_ASM_COMMENT("begin gebp micro kernel 2pX4"); + RhsPacket B_0, B1, B2, B3, T0; + +// NOTE: the begin/end asm comments below work around bug 935! +// but they are not enough for gcc>=6 without FMA (bug 1637) +#if EIGEN_GNUC_AT_LEAST(6, 0) && defined(EIGEN_VECTORIZE_SSE) +#define EIGEN_GEBP_2PX4_SPILLING_WORKAROUND __asm__("" : [a0] "+x,m"(A0), [a1] "+x,m"(A1)); +#else +#define EIGEN_GEBP_2PX4_SPILLING_WORKAROUND +#endif +#define EIGEN_GEBGP_ONESTEP(K) \ + do { \ + EIGEN_ASM_COMMENT("begin step of gebp micro kernel 2pX4"); \ + traits.loadLhs(&blA[(0 + 2 * K) * LhsProgress], A0); \ + traits.loadLhs(&blA[(1 + 2 * K) * LhsProgress], A1); \ + traits.broadcastRhs(&blB[(0 + 4 * K) * RhsProgress], B_0, B1, B2, B3); \ + traits.madd(A0, B_0, C0, T0); \ + traits.madd(A1, B_0, C4, B_0); \ + traits.madd(A0, B1, C1, T0); \ + traits.madd(A1, B1, C5, B1); \ + traits.madd(A0, B2, C2, T0); \ + traits.madd(A1, B2, C6, B2); \ + traits.madd(A0, B3, C3, T0); \ + traits.madd(A1, B3, C7, B3); \ + EIGEN_GEBP_2PX4_SPILLING_WORKAROUND \ + EIGEN_ASM_COMMENT("end step of gebp micro kernel 2pX4"); \ + } while (false) + + internal::prefetch(blB + (48 + 0)); + EIGEN_GEBGP_ONESTEP(0); + EIGEN_GEBGP_ONESTEP(1); + EIGEN_GEBGP_ONESTEP(2); + EIGEN_GEBGP_ONESTEP(3); + internal::prefetch(blB + (48 + 16)); + EIGEN_GEBGP_ONESTEP(4); + EIGEN_GEBGP_ONESTEP(5); + EIGEN_GEBGP_ONESTEP(6); + EIGEN_GEBGP_ONESTEP(7); + + blB += pk * 4 * RhsProgress; + blA += pk * (2 * Traits::LhsProgress); + + EIGEN_ASM_COMMENT("end gebp micro kernel 2pX4"); + } + // process remaining peeled loop + for (Index k = peeled_kc; k < depth; k++) { + RhsPacket B_0, B1, B2, B3, T0; + EIGEN_GEBGP_ONESTEP(0); + blB += 4 * RhsProgress; + blA += 2 * Traits::LhsProgress; + } #undef EIGEN_GEBGP_ONESTEP - ResPacket R0, R1; - ResPacket alphav = pset1(alpha); - R0 = r0.loadPacket(0 * Traits::ResPacketSize); - R1 = r0.loadPacket(1 * Traits::ResPacketSize); - traits.acc(C0, alphav, R0); - traits.acc(C4, alphav, R1); - r0.storePacket(0 * Traits::ResPacketSize, R0); - r0.storePacket(1 * Traits::ResPacketSize, R1); + ResPacket R0, R1, R2, R3; + ResPacket alphav = pset1(alpha); + + R0 = r0.loadPacket(0 * Traits::ResPacketSize); + R1 = r0.loadPacket(1 * Traits::ResPacketSize); + R2 = r1.loadPacket(0 * Traits::ResPacketSize); + R3 = r1.loadPacket(1 * Traits::ResPacketSize); + traits.acc(C0, alphav, R0); + traits.acc(C4, alphav, R1); + traits.acc(C1, alphav, R2); + traits.acc(C5, alphav, R3); + r0.storePacket(0 * Traits::ResPacketSize, R0); + r0.storePacket(1 * Traits::ResPacketSize, R1); + r1.storePacket(0 * Traits::ResPacketSize, R2); + r1.storePacket(1 * Traits::ResPacketSize, R3); + + R0 = r2.loadPacket(0 * Traits::ResPacketSize); + R1 = r2.loadPacket(1 * Traits::ResPacketSize); + R2 = r3.loadPacket(0 * Traits::ResPacketSize); + R3 = r3.loadPacket(1 * Traits::ResPacketSize); + traits.acc(C2, alphav, R0); + traits.acc(C6, alphav, R1); + traits.acc(C3, alphav, R2); + traits.acc(C7, alphav, R3); + r2.storePacket(0 * Traits::ResPacketSize, R0); + r2.storePacket(1 * Traits::ResPacketSize, R1); + r3.storePacket(0 * Traits::ResPacketSize, R2); + r3.storePacket(1 * Traits::ResPacketSize, R3); + } + } + + // Deal with remaining columns of the rhs + for (Index j2 = packet_cols4; j2 < cols; j2++) { + for (Index i = i1; i < actual_panel_end; i += 2 * LhsProgress) { + // One column at a time + const LhsScalar *blA = &blockA[i * strideA + offsetA * (2 * Traits::LhsProgress)]; + prefetch(&blA[0]); + + // gets res block as register + AccPacket C0, C4; + traits.initAcc(C0); + traits.initAcc(C4); + + LinearMapper r0 = res.getLinearMapper(i, j2); + r0.prefetch(prefetch_res_offset); + + // performs "inner" products + const RhsScalar *blB = &blockB[j2 * strideB + offsetB]; + LhsPacket A0, A1; + + for (Index k = 0; k < peeled_kc; k += pk) { + EIGEN_ASM_COMMENT("begin gebp micro kernel 2pX1"); + RhsPacket B_0, B1; + +#define EIGEN_GEBGP_ONESTEP(K) \ + do { \ + EIGEN_ASM_COMMENT("begin step of gebp micro kernel 2pX1"); \ + EIGEN_ASM_COMMENT("Note: these asm comments work around bug 935!"); \ + traits.loadLhs(&blA[(0 + 2 * K) * LhsProgress], A0); \ + traits.loadLhs(&blA[(1 + 2 * K) * LhsProgress], A1); \ + traits.loadRhs(&blB[(0 + K) * RhsProgress], B_0); \ + traits.madd(A0, B_0, C0, B1); \ + traits.madd(A1, B_0, C4, B_0); \ + EIGEN_ASM_COMMENT("end step of gebp micro kernel 2pX1"); \ + } while (false) + + EIGEN_GEBGP_ONESTEP(0); + EIGEN_GEBGP_ONESTEP(1); + EIGEN_GEBGP_ONESTEP(2); + EIGEN_GEBGP_ONESTEP(3); + EIGEN_GEBGP_ONESTEP(4); + EIGEN_GEBGP_ONESTEP(5); + EIGEN_GEBGP_ONESTEP(6); + EIGEN_GEBGP_ONESTEP(7); + + blB += pk * RhsProgress; + blA += pk * 2 * Traits::LhsProgress; + + EIGEN_ASM_COMMENT("end gebp micro kernel 2pX1"); + } + + // process remaining peeled loop + for (Index k = peeled_kc; k < depth; k++) { + RhsPacket B_0, B1; + EIGEN_GEBGP_ONESTEP(0); + blB += RhsProgress; + blA += 2 * Traits::LhsProgress; + } +#undef EIGEN_GEBGP_ONESTEP + ResPacket R0, R1; + ResPacket alphav = pset1(alpha); + + R0 = r0.loadPacket(0 * Traits::ResPacketSize); + R1 = r0.loadPacket(1 * Traits::ResPacketSize); + traits.acc(C0, alphav, R0); + traits.acc(C4, alphav, R1); + r0.storePacket(0 * Traits::ResPacketSize, R0); + r0.storePacket(1 * Traits::ResPacketSize, R1); } } } } //---------- Process 1 * LhsProgress rows at once ---------- - if(mr>=1*Traits::LhsProgress) - { + if (mr >= 1 * Traits::LhsProgress) { // loops on each largest micro horizontal panel of lhs (1*LhsProgress x depth) - for(Index i=peeled_mc2; i::half>::size; - if ((SwappedTraits::LhsProgress % 4) == 0 && - (SwappedTraits::LhsProgress <= 8) && - (SwappedTraits::LhsProgress!=8 || SResPacketHalfSize==nr)) - { + if ((SwappedTraits::LhsProgress % 4) == 0 && (SwappedTraits::LhsProgress <= 8) + && (SwappedTraits::LhsProgress != 8 || SResPacketHalfSize == nr)) { SAccPacket C0, C1, C2, C3; straits.initAcc(C0); straits.initAcc(C1); straits.initAcc(C2); straits.initAcc(C3); - const Index spk = (std::max)(1,SwappedTraits::LhsProgress/4); - const Index endk = (depth/spk)*spk; - const Index endk4 = (depth/(spk*4))*(spk*4); - - Index k=0; - for(; k=8,typename unpacket_traits::half,SResPacket>::type SResPacketHalf; - typedef typename conditional=8,typename unpacket_traits::half,SLhsPacket>::type SLhsPacketHalf; - typedef typename conditional=8,typename unpacket_traits::half,SRhsPacket>::type SRhsPacketHalf; - typedef typename conditional=8,typename unpacket_traits::half,SAccPacket>::type SAccPacketHalf; + typedef typename conditional= 8, + typename unpacket_traits::half, + SResPacket>::type SResPacketHalf; + typedef typename conditional= 8, + typename unpacket_traits::half, + SLhsPacket>::type SLhsPacketHalf; + typedef typename conditional= 8, + typename unpacket_traits::half, + SRhsPacket>::type SRhsPacketHalf; + typedef typename conditional= 8, + typename unpacket_traits::half, + SAccPacket>::type SAccPacketHalf; SResPacketHalf R = res.template gatherPacket(i, j2); SResPacketHalf alphav = pset1(alpha); - if(depth-endk>0) - { + if (depth - endk > 0) { // We have to handle the last row of the rhs which corresponds to a half-packet SLhsPacketHalf a0; SRhsPacketHalf b0; straits.loadLhsUnaligned(blB, a0); straits.loadRhs(blA, b0); SAccPacketHalf c0 = predux_downto4(C0); - straits.madd(a0,b0,c0,b0); + straits.madd(a0, b0, c0, b0); straits.acc(c0, alphav, R); - } - else - { + } else { straits.acc(predux_downto4(C0), alphav, R); } res.scatterPacket(i, j2, R); - } - else - { + } else { SResPacket R = res.template gatherPacket(i, j2); SResPacket alphav = pset1(alpha); straits.acc(C0, alphav, R); res.scatterPacket(i, j2, R); } - } - else // scalar path + } else// scalar path { // get a 1 x 4 res block as registers ResScalar C0(0), C1(0), C2(0), C3(0); - for(Index k=0; k -struct gemm_pack_lhs -{ - typedef typename DataMapper::LinearMapper LinearMapper; - EIGEN_DONT_INLINE void operator()(Scalar* blockA, const DataMapper& lhs, Index depth, Index rows, Index stride=0, Index offset=0); -}; + // pack a block of the lhs + // The traversal is as follow (mr==4): + // 0 4 8 12 ... + // 1 5 9 13 ... + // 2 6 10 14 ... + // 3 7 11 15 ... + // + // 16 20 24 28 ... + // 17 21 25 29 ... + // 18 22 26 30 ... + // 19 23 27 31 ... + // + // 32 33 34 35 ... + // 36 36 38 39 ... + template + struct gemm_pack_lhs + { + typedef typename DataMapper::LinearMapper LinearMapper; + EIGEN_DONT_INLINE void + operator()(Scalar *blockA, const DataMapper &lhs, Index depth, Index rows, Index stride = 0, Index offset = 0); + }; -template -EIGEN_DONT_INLINE void gemm_pack_lhs - ::operator()(Scalar* blockA, const DataMapper& lhs, Index depth, Index rows, Index stride, Index offset) -{ - typedef typename packet_traits::type Packet; - enum { PacketSize = packet_traits::size }; - - EIGEN_ASM_COMMENT("EIGEN PRODUCT PACK LHS"); - EIGEN_UNUSED_VARIABLE(stride); - EIGEN_UNUSED_VARIABLE(offset); - eigen_assert(((!PanelMode) && stride==0 && offset==0) || (PanelMode && stride>=depth && offset<=stride)); - eigen_assert( ((Pack1%PacketSize)==0 && Pack1<=4*PacketSize) || (Pack1<=4) ); - conj_if::IsComplex && Conjugate> cj; - Index count = 0; - - const Index peeled_mc3 = Pack1>=3*PacketSize ? (rows/(3*PacketSize))*(3*PacketSize) : 0; - const Index peeled_mc2 = Pack1>=2*PacketSize ? peeled_mc3+((rows-peeled_mc3)/(2*PacketSize))*(2*PacketSize) : 0; - const Index peeled_mc1 = Pack1>=1*PacketSize ? (rows/(1*PacketSize))*(1*PacketSize) : 0; - const Index peeled_mc0 = Pack2>=1*PacketSize ? peeled_mc1 - : Pack2>1 ? (rows/Pack2)*Pack2 : 0; - - Index i=0; - - // Pack 3 packets - if(Pack1>=3*PacketSize) + template + EIGEN_DONT_INLINE void + gemm_pack_lhs::operator()(Scalar *blockA, + const DataMapper &lhs, + Index depth, + Index rows, + Index stride, + Index offset) { - for(; i::type Packet; + enum { PacketSize = packet_traits::size }; + + EIGEN_ASM_COMMENT("EIGEN PRODUCT PACK LHS"); + EIGEN_UNUSED_VARIABLE(stride); + EIGEN_UNUSED_VARIABLE(offset); + eigen_assert(((!PanelMode) && stride == 0 && offset == 0) || (PanelMode && stride >= depth && offset <= stride)); + eigen_assert(((Pack1 % PacketSize) == 0 && Pack1 <= 4 * PacketSize) || (Pack1 <= 4)); + conj_if::IsComplex && Conjugate> cj; + Index count = 0; + + const Index peeled_mc3 = Pack1 >= 3 * PacketSize ? (rows / (3 * PacketSize)) * (3 * PacketSize) : 0; + const Index peeled_mc2 = + Pack1 >= 2 * PacketSize ? peeled_mc3 + ((rows - peeled_mc3) / (2 * PacketSize)) * (2 * PacketSize) : 0; + const Index peeled_mc1 = Pack1 >= 1 * PacketSize ? (rows / (1 * PacketSize)) * (1 * PacketSize) : 0; + const Index peeled_mc0 = Pack2 >= 1 * PacketSize ? peeled_mc1 : Pack2 > 1 ? (rows / Pack2) * Pack2 : 0; + + Index i = 0; + + // Pack 3 packets + if (Pack1 >= 3 * PacketSize) { + for (; i < peeled_mc3; i += 3 * PacketSize) { + if (PanelMode) count += (3 * PacketSize) * offset; + + for (Index k = 0; k < depth; k++) { + Packet A, B, C; + A = lhs.loadPacket(i + 0 * PacketSize, k); + B = lhs.loadPacket(i + 1 * PacketSize, k); + C = lhs.loadPacket(i + 2 * PacketSize, k); + pstore(blockA + count, cj.pconj(A)); + count += PacketSize; + pstore(blockA + count, cj.pconj(B)); + count += PacketSize; + pstore(blockA + count, cj.pconj(C)); + count += PacketSize; + } + if (PanelMode) count += (3 * PacketSize) * (stride - offset - depth); } - if(PanelMode) count += (3*PacketSize) * (stride-offset-depth); } - } - // Pack 2 packets - if(Pack1>=2*PacketSize) - { - for(; i= 2 * PacketSize) { + for (; i < peeled_mc2; i += 2 * PacketSize) { + if (PanelMode) count += (2 * PacketSize) * offset; + + for (Index k = 0; k < depth; k++) { + Packet A, B; + A = lhs.loadPacket(i + 0 * PacketSize, k); + B = lhs.loadPacket(i + 1 * PacketSize, k); + pstore(blockA + count, cj.pconj(A)); + count += PacketSize; + pstore(blockA + count, cj.pconj(B)); + count += PacketSize; + } + if (PanelMode) count += (2 * PacketSize) * (stride - offset - depth); } - if(PanelMode) count += (2*PacketSize) * (stride-offset-depth); } - } - // Pack 1 packets - if(Pack1>=1*PacketSize) - { - for(; i= 1 * PacketSize) { + for (; i < peeled_mc1; i += 1 * PacketSize) { + if (PanelMode) count += (1 * PacketSize) * offset; + + for (Index k = 0; k < depth; k++) { + Packet A; + A = lhs.loadPacket(i + 0 * PacketSize, k); + pstore(blockA + count, cj.pconj(A)); + count += PacketSize; + } + if (PanelMode) count += (1 * PacketSize) * (stride - offset - depth); } - if(PanelMode) count += (1*PacketSize) * (stride-offset-depth); } - } - // Pack scalars - if(Pack21) - { - for(; i 1) { + for (; i < peeled_mc0; i += Pack2) { + if (PanelMode) count += Pack2 * offset; - for(Index k=0; k -struct gemm_pack_lhs -{ - typedef typename DataMapper::LinearMapper LinearMapper; - EIGEN_DONT_INLINE void operator()(Scalar* blockA, const DataMapper& lhs, Index depth, Index rows, Index stride=0, Index offset=0); -}; -template -EIGEN_DONT_INLINE void gemm_pack_lhs - ::operator()(Scalar* blockA, const DataMapper& lhs, Index depth, Index rows, Index stride, Index offset) -{ - typedef typename packet_traits::type Packet; - enum { PacketSize = packet_traits::size }; - - EIGEN_ASM_COMMENT("EIGEN PRODUCT PACK LHS"); - EIGEN_UNUSED_VARIABLE(stride); - EIGEN_UNUSED_VARIABLE(offset); - eigen_assert(((!PanelMode) && stride==0 && offset==0) || (PanelMode && stride>=depth && offset<=stride)); - conj_if::IsComplex && Conjugate> cj; - Index count = 0; - -// const Index peeled_mc3 = Pack1>=3*PacketSize ? (rows/(3*PacketSize))*(3*PacketSize) : 0; -// const Index peeled_mc2 = Pack1>=2*PacketSize ? peeled_mc3+((rows-peeled_mc3)/(2*PacketSize))*(2*PacketSize) : 0; -// const Index peeled_mc1 = Pack1>=1*PacketSize ? (rows/(1*PacketSize))*(1*PacketSize) : 0; - - int pack = Pack1; - Index i = 0; - while(pack>0) + template + struct gemm_pack_lhs { - Index remaining_rows = rows-i; - Index peeled_mc = i+(remaining_rows/pack)*pack; - for(; i=PacketSize) - { - for(; k kernel; - for (int p = 0; p < PacketSize; ++p) kernel.packet[p] = lhs.loadPacket(i+p+m, k); - ptranspose(kernel); - for (int p = 0; p < PacketSize; ++p) pstore(blockA+count+m+(pack)*p, cj.pconj(kernel.packet[p])); + template + EIGEN_DONT_INLINE void + gemm_pack_lhs::operator()(Scalar *blockA, + const DataMapper &lhs, + Index depth, + Index rows, + Index stride, + Index offset) + { + typedef typename packet_traits::type Packet; + enum { PacketSize = packet_traits::size }; + + EIGEN_ASM_COMMENT("EIGEN PRODUCT PACK LHS"); + EIGEN_UNUSED_VARIABLE(stride); + EIGEN_UNUSED_VARIABLE(offset); + eigen_assert(((!PanelMode) && stride == 0 && offset == 0) || (PanelMode && stride >= depth && offset <= stride)); + conj_if::IsComplex && Conjugate> cj; + Index count = 0; + + // const Index peeled_mc3 = Pack1>=3*PacketSize ? (rows/(3*PacketSize))*(3*PacketSize) : 0; + // const Index peeled_mc2 = Pack1>=2*PacketSize ? peeled_mc3+((rows-peeled_mc3)/(2*PacketSize))*(2*PacketSize) : + // 0; const Index peeled_mc1 = Pack1>=1*PacketSize ? (rows/(1*PacketSize))*(1*PacketSize) : 0; + + int pack = Pack1; + Index i = 0; + while (pack > 0) { + Index remaining_rows = rows - i; + Index peeled_mc = i + (remaining_rows / pack) * pack; + for (; i < peeled_mc; i += pack) { + if (PanelMode) count += pack * offset; + + const Index peeled_k = (depth / PacketSize) * PacketSize; + Index k = 0; + if (pack >= PacketSize) { + for (; k < peeled_k; k += PacketSize) { + for (Index m = 0; m < pack; m += PacketSize) { + PacketBlock kernel; + for (int p = 0; p < PacketSize; ++p) kernel.packet[p] = lhs.loadPacket(i + p + m, k); + ptranspose(kernel); + for (int p = 0; p < PacketSize; ++p) pstore(blockA + count + m + (pack)*p, cj.pconj(kernel.packet[p])); + } + count += PacketSize * pack; } - count += PacketSize*pack; } - } - for(; k + struct gemm_pack_rhs { - if(PanelMode) count += offset; - for(Index k=0; k::type Packet; + typedef typename DataMapper::LinearMapper LinearMapper; + enum { PacketSize = packet_traits::size }; + EIGEN_DONT_INLINE void + operator()(Scalar *blockB, const DataMapper &rhs, Index depth, Index cols, Index stride = 0, Index offset = 0); + }; -// copy a complete panel of the rhs -// this version is optimized for column major matrices -// The traversal order is as follow: (nr==4): -// 0 1 2 3 12 13 14 15 24 27 -// 4 5 6 7 16 17 18 19 25 28 -// 8 9 10 11 20 21 22 23 26 29 -// . . . . . . . . . . -template -struct gemm_pack_rhs -{ - typedef typename packet_traits::type Packet; - typedef typename DataMapper::LinearMapper LinearMapper; - enum { PacketSize = packet_traits::size }; - EIGEN_DONT_INLINE void operator()(Scalar* blockB, const DataMapper& rhs, Index depth, Index cols, Index stride=0, Index offset=0); -}; - -template -EIGEN_DONT_INLINE void gemm_pack_rhs - ::operator()(Scalar* blockB, const DataMapper& rhs, Index depth, Index cols, Index stride, Index offset) -{ - EIGEN_ASM_COMMENT("EIGEN PRODUCT PACK RHS COLMAJOR"); - EIGEN_UNUSED_VARIABLE(stride); - EIGEN_UNUSED_VARIABLE(offset); - eigen_assert(((!PanelMode) && stride==0 && offset==0) || (PanelMode && stride>=depth && offset<=stride)); - conj_if::IsComplex && Conjugate> cj; - Index packet_cols8 = nr>=8 ? (cols/8) * 8 : 0; - Index packet_cols4 = nr>=4 ? (cols/4) * 4 : 0; - Index count = 0; - const Index peeled_k = (depth/PacketSize)*PacketSize; -// if(nr>=8) -// { -// for(Index j2=0; j2 kernel; -// for (int p = 0; p < PacketSize; ++p) { -// kernel.packet[p] = ploadu(&rhs[(j2+p)*rhsStride+k]); -// } -// ptranspose(kernel); -// for (int p = 0; p < PacketSize; ++p) { -// pstoreu(blockB+count, cj.pconj(kernel.packet[p])); -// count+=PacketSize; -// } -// } -// } -// for(; k=4) + template + EIGEN_DONT_INLINE void gemm_pack_rhs::operator()( + Scalar *blockB, + const DataMapper &rhs, + Index depth, + Index cols, + Index stride, + Index offset) { - for(Index j2=packet_cols8; j2 kernel; - kernel.packet[0] = dm0.loadPacket(k); - kernel.packet[1%PacketSize] = dm1.loadPacket(k); - kernel.packet[2%PacketSize] = dm2.loadPacket(k); - kernel.packet[3%PacketSize] = dm3.loadPacket(k); - ptranspose(kernel); - pstoreu(blockB+count+0*PacketSize, cj.pconj(kernel.packet[0])); - pstoreu(blockB+count+1*PacketSize, cj.pconj(kernel.packet[1%PacketSize])); - pstoreu(blockB+count+2*PacketSize, cj.pconj(kernel.packet[2%PacketSize])); - pstoreu(blockB+count+3*PacketSize, cj.pconj(kernel.packet[3%PacketSize])); - count+=4*PacketSize; + EIGEN_ASM_COMMENT("EIGEN PRODUCT PACK RHS COLMAJOR"); + EIGEN_UNUSED_VARIABLE(stride); + EIGEN_UNUSED_VARIABLE(offset); + eigen_assert(((!PanelMode) && stride == 0 && offset == 0) || (PanelMode && stride >= depth && offset <= stride)); + conj_if::IsComplex && Conjugate> cj; + Index packet_cols8 = nr >= 8 ? (cols / 8) * 8 : 0; + Index packet_cols4 = nr >= 4 ? (cols / 4) * 4 : 0; + Index count = 0; + const Index peeled_k = (depth / PacketSize) * PacketSize; + // if(nr>=8) + // { + // for(Index j2=0; j2 kernel; + // for (int p = 0; p < PacketSize; ++p) { + // kernel.packet[p] = ploadu(&rhs[(j2+p)*rhsStride+k]); + // } + // ptranspose(kernel); + // for (int p = 0; p < PacketSize; ++p) { + // pstoreu(blockB+count, cj.pconj(kernel.packet[p])); + // count+=PacketSize; + // } + // } + // } + // for(; k= 4) { + for (Index j2 = packet_cols8; j2 < packet_cols4; j2 += 4) { + // skip what we have before + if (PanelMode) count += 4 * offset; + const LinearMapper dm0 = rhs.getLinearMapper(0, j2 + 0); + const LinearMapper dm1 = rhs.getLinearMapper(0, j2 + 1); + const LinearMapper dm2 = rhs.getLinearMapper(0, j2 + 2); + const LinearMapper dm3 = rhs.getLinearMapper(0, j2 + 3); + + Index k = 0; + if ((PacketSize % 4) == 0)// TODO enable vectorized transposition for PacketSize==2 ?? + { + for (; k < peeled_k; k += PacketSize) { + PacketBlock kernel; + kernel.packet[0] = dm0.loadPacket(k); + kernel.packet[1 % PacketSize] = dm1.loadPacket(k); + kernel.packet[2 % PacketSize] = dm2.loadPacket(k); + kernel.packet[3 % PacketSize] = dm3.loadPacket(k); + ptranspose(kernel); + pstoreu(blockB + count + 0 * PacketSize, cj.pconj(kernel.packet[0])); + pstoreu(blockB + count + 1 * PacketSize, cj.pconj(kernel.packet[1 % PacketSize])); + pstoreu(blockB + count + 2 * PacketSize, cj.pconj(kernel.packet[2 % PacketSize])); + pstoreu(blockB + count + 3 * PacketSize, cj.pconj(kernel.packet[3 % PacketSize])); + count += 4 * PacketSize; + } + } + for (; k < depth; k++) { + blockB[count + 0] = cj(dm0(k)); + blockB[count + 1] = cj(dm1(k)); + blockB[count + 2] = cj(dm2(k)); + blockB[count + 3] = cj(dm3(k)); + count += 4; } + // skip what we have after + if (PanelMode) count += 4 * (stride - offset - depth); } - for(; k + struct gemm_pack_rhs { - if(PanelMode) count += offset; - const LinearMapper dm0 = rhs.getLinearMapper(0, j2); - for(Index k=0; k::type Packet; + typedef typename DataMapper::LinearMapper LinearMapper; + enum { PacketSize = packet_traits::size }; + EIGEN_DONT_INLINE void + operator()(Scalar *blockB, const DataMapper &rhs, Index depth, Index cols, Index stride = 0, Index offset = 0); + }; -// this version is optimized for row major matrices -template -struct gemm_pack_rhs -{ - typedef typename packet_traits::type Packet; - typedef typename DataMapper::LinearMapper LinearMapper; - enum { PacketSize = packet_traits::size }; - EIGEN_DONT_INLINE void operator()(Scalar* blockB, const DataMapper& rhs, Index depth, Index cols, Index stride=0, Index offset=0); -}; - -template -EIGEN_DONT_INLINE void gemm_pack_rhs - ::operator()(Scalar* blockB, const DataMapper& rhs, Index depth, Index cols, Index stride, Index offset) -{ - EIGEN_ASM_COMMENT("EIGEN PRODUCT PACK RHS ROWMAJOR"); - EIGEN_UNUSED_VARIABLE(stride); - EIGEN_UNUSED_VARIABLE(offset); - eigen_assert(((!PanelMode) && stride==0 && offset==0) || (PanelMode && stride>=depth && offset<=stride)); - conj_if::IsComplex && Conjugate> cj; - Index packet_cols8 = nr>=8 ? (cols/8) * 8 : 0; - Index packet_cols4 = nr>=4 ? (cols/4) * 4 : 0; - Index count = 0; - -// if(nr>=8) -// { -// for(Index j2=0; j2(&rhs[k*rhsStride + j2]); -// pstoreu(blockB+count, cj.pconj(A)); -// } else if (PacketSize==4) { -// Packet A = ploadu(&rhs[k*rhsStride + j2]); -// Packet B = ploadu(&rhs[k*rhsStride + j2 + PacketSize]); -// pstoreu(blockB+count, cj.pconj(A)); -// pstoreu(blockB+count+PacketSize, cj.pconj(B)); -// } else { -// const Scalar* b0 = &rhs[k*rhsStride + j2]; -// blockB[count+0] = cj(b0[0]); -// blockB[count+1] = cj(b0[1]); -// blockB[count+2] = cj(b0[2]); -// blockB[count+3] = cj(b0[3]); -// blockB[count+4] = cj(b0[4]); -// blockB[count+5] = cj(b0[5]); -// blockB[count+6] = cj(b0[6]); -// blockB[count+7] = cj(b0[7]); -// } -// count += 8; -// } -// // skip what we have after -// if(PanelMode) count += 8 * (stride-offset-depth); -// } -// } - if(nr>=4) + template + EIGEN_DONT_INLINE void gemm_pack_rhs::operator()( + Scalar *blockB, + const DataMapper &rhs, + Index depth, + Index cols, + Index stride, + Index offset) { - for(Index j2=packet_cols8; j2= depth && offset <= stride)); + conj_if::IsComplex && Conjugate> cj; + Index packet_cols8 = nr >= 8 ? (cols / 8) * 8 : 0; + Index packet_cols4 = nr >= 4 ? (cols / 4) * 4 : 0; + Index count = 0; + + // if(nr>=8) + // { + // for(Index j2=0; j2(&rhs[k*rhsStride + j2]); + // pstoreu(blockB+count, cj.pconj(A)); + // } else if (PacketSize==4) { + // Packet A = ploadu(&rhs[k*rhsStride + j2]); + // Packet B = ploadu(&rhs[k*rhsStride + j2 + PacketSize]); + // pstoreu(blockB+count, cj.pconj(A)); + // pstoreu(blockB+count+PacketSize, cj.pconj(B)); + // } else { + // const Scalar* b0 = &rhs[k*rhsStride + j2]; + // blockB[count+0] = cj(b0[0]); + // blockB[count+1] = cj(b0[1]); + // blockB[count+2] = cj(b0[2]); + // blockB[count+3] = cj(b0[3]); + // blockB[count+4] = cj(b0[4]); + // blockB[count+5] = cj(b0[5]); + // blockB[count+6] = cj(b0[6]); + // blockB[count+7] = cj(b0[7]); + // } + // count += 8; + // } + // // skip what we have after + // if(PanelMode) count += 8 * (stride-offset-depth); + // } + // } + if (nr >= 4) { + for (Index j2 = packet_cols8; j2 < packet_cols4; j2 += 4) { + // skip what we have before + if (PanelMode) count += 4 * offset; + for (Index k = 0; k < depth; k++) { + if (PacketSize == 4) { + Packet A = rhs.loadPacket(k, j2); + pstoreu(blockB + count, cj.pconj(A)); + count += PacketSize; + } else { + const LinearMapper dm0 = rhs.getLinearMapper(k, j2); + blockB[count + 0] = cj(dm0(0)); + blockB[count + 1] = cj(dm0(1)); + blockB[count + 2] = cj(dm0(2)); + blockB[count + 3] = cj(dm0(3)); + count += 4; + } } + // skip what we have after + if (PanelMode) count += 4 * (stride - offset - depth); } - // skip what we have after - if(PanelMode) count += 4 * (stride-offset-depth); } - } - // copy the remaining columns one at a time (nr==1) - for(Index j2=packet_cols4; j2 class level3_blocking; - -/* Specialization for a row-major destination matrix => simple transposition of the product */ -template< - typename Index, - typename LhsScalar, int LhsStorageOrder, bool ConjugateLhs, - typename RhsScalar, int RhsStorageOrder, bool ConjugateRhs> -struct general_matrix_matrix_product -{ - typedef gebp_traits Traits; - - typedef typename ScalarBinaryOpTraits::ReturnType ResScalar; - static EIGEN_STRONG_INLINE void run( - Index rows, Index cols, Index depth, - const LhsScalar* lhs, Index lhsStride, - const RhsScalar* rhs, Index rhsStride, - ResScalar* res, Index resStride, - ResScalar alpha, - level3_blocking& blocking, - GemmParallelInfo* info = 0) + template class level3_blocking; + + /* Specialization for a row-major destination matrix => simple transposition of the product */ + template + struct general_matrix_matrix_product { - // transpose the product such that the result is column major - general_matrix_matrix_product - ::run(cols,rows,depth,rhs,rhsStride,lhs,lhsStride,res,resStride,alpha,blocking,info); - } -}; - -/* Specialization for a col-major destination matrix - * => Blocking algorithm following Goto's paper */ -template< - typename Index, - typename LhsScalar, int LhsStorageOrder, bool ConjugateLhs, - typename RhsScalar, int RhsStorageOrder, bool ConjugateRhs> -struct general_matrix_matrix_product -{ - -typedef gebp_traits Traits; - -typedef typename ScalarBinaryOpTraits::ReturnType ResScalar; -static void run(Index rows, Index cols, Index depth, - const LhsScalar* _lhs, Index lhsStride, - const RhsScalar* _rhs, Index rhsStride, - ResScalar* _res, Index resStride, - ResScalar alpha, - level3_blocking& blocking, - GemmParallelInfo* info = 0) -{ - typedef const_blas_data_mapper LhsMapper; - typedef const_blas_data_mapper RhsMapper; - typedef blas_data_mapper ResMapper; - LhsMapper lhs(_lhs,lhsStride); - RhsMapper rhs(_rhs,rhsStride); - ResMapper res(_res, resStride); - - Index kc = blocking.kc(); // cache block size along the K direction - Index mc = (std::min)(rows,blocking.mc()); // cache block size along the M direction - Index nc = (std::min)(cols,blocking.nc()); // cache block size along the N direction - - gemm_pack_lhs pack_lhs; - gemm_pack_rhs pack_rhs; - gebp_kernel gebp; + typedef gebp_traits Traits; + + typedef typename ScalarBinaryOpTraits::ReturnType ResScalar; + static EIGEN_STRONG_INLINE void run(Index rows, + Index cols, + Index depth, + const LhsScalar *lhs, + Index lhsStride, + const RhsScalar *rhs, + Index rhsStride, + ResScalar *res, + Index resStride, + ResScalar alpha, + level3_blocking &blocking, + GemmParallelInfo *info = 0) + { + // transpose the product such that the result is column major + general_matrix_matrix_product::run(cols, rows, depth, rhs, rhsStride, lhs, lhsStride, res, resStride, alpha, blocking, info); + } + }; -#ifdef EIGEN_HAS_OPENMP - if(info) + /* Specialization for a col-major destination matrix + * => Blocking algorithm following Goto's paper */ + template + struct general_matrix_matrix_product { - // this is the parallel version! - int tid = omp_get_thread_num(); - int threads = omp_get_num_threads(); - - LhsScalar* blockA = blocking.blockA(); - eigen_internal_assert(blockA!=0); - std::size_t sizeB = kc*nc; - ei_declare_aligned_stack_constructed_variable(RhsScalar, blockB, sizeB, 0); - - // For each horizontal panel of the rhs, and corresponding vertical panel of the lhs... - for(Index k=0; k Traits; + + typedef typename ScalarBinaryOpTraits::ReturnType ResScalar; + static void run(Index rows, + Index cols, + Index depth, + const LhsScalar *_lhs, + Index lhsStride, + const RhsScalar *_rhs, + Index rhsStride, + ResScalar *_res, + Index resStride, + ResScalar alpha, + level3_blocking &blocking, + GemmParallelInfo *info = 0) { - const Index actual_kc = (std::min)(k+kc,depth)-k; // => rows of B', and cols of the A' + typedef const_blas_data_mapper LhsMapper; + typedef const_blas_data_mapper RhsMapper; + typedef blas_data_mapper ResMapper; + LhsMapper lhs(_lhs, lhsStride); + RhsMapper rhs(_rhs, rhsStride); + ResMapper res(_res, resStride); - // In order to reduce the chance that a thread has to wait for the other, - // let's start by packing B'. - pack_rhs(blockB, rhs.getSubMapper(k,0), actual_kc, nc); + Index kc = blocking.kc();// cache block size along the K direction + Index mc = (std::min)(rows, blocking.mc());// cache block size along the M direction + Index nc = (std::min)(cols, blocking.nc());// cache block size along the N direction - // Pack A_k to A' in a parallel fashion: - // each thread packs the sub block A_k,i to A'_i where i is the thread id. + gemm_pack_lhs pack_lhs; + gemm_pack_rhs pack_rhs; + gebp_kernel gebp; - // However, before copying to A'_i, we have to make sure that no other thread is still using it, - // i.e., we test that info[tid].users equals 0. - // Then, we set info[tid].users to the number of threads to mark that all other threads are going to use it. - while(info[tid].users!=0) {} - info[tid].users += threads; +#ifdef EIGEN_HAS_OPENMP + if (info) { + // this is the parallel version! + int tid = omp_get_thread_num(); + int threads = omp_get_num_threads(); + + LhsScalar *blockA = blocking.blockA(); + eigen_internal_assert(blockA != 0); + + std::size_t sizeB = kc * nc; + ei_declare_aligned_stack_constructed_variable(RhsScalar, blockB, sizeB, 0); + + // For each horizontal panel of the rhs, and corresponding vertical panel of the lhs... + for (Index k = 0; k < depth; k += kc) { + const Index actual_kc = (std::min)(k + kc, depth) - k;// => rows of B', and cols of the A' + + // In order to reduce the chance that a thread has to wait for the other, + // let's start by packing B'. + pack_rhs(blockB, rhs.getSubMapper(k, 0), actual_kc, nc); + + // Pack A_k to A' in a parallel fashion: + // each thread packs the sub block A_k,i to A'_i where i is the thread id. + + // However, before copying to A'_i, we have to make sure that no other thread is still using it, + // i.e., we test that info[tid].users equals 0. + // Then, we set info[tid].users to the number of threads to mark that all other threads are going to use it. + while (info[tid].users != 0) {} + info[tid].users += threads; + + pack_lhs(blockA + info[tid].lhs_start * actual_kc, + lhs.getSubMapper(info[tid].lhs_start, k), + actual_kc, + info[tid].lhs_length); + + // Notify the other threads that the part A'_i is ready to go. + info[tid].sync = k; + + // Computes C_i += A' * B' per A'_i + for (int shift = 0; shift < threads; ++shift) { + int i = (tid + shift) % threads; + + // At this point we have to make sure that A'_i has been updated by the thread i, + // we use testAndSetOrdered to mimic a volatile access. + // However, no need to wait for the B' part which has been updated by the current thread! + if (shift > 0) { + while (info[i].sync != k) {} + } + + gebp(res.getSubMapper(info[i].lhs_start, 0), + blockA + info[i].lhs_start * actual_kc, + blockB, + info[i].lhs_length, + actual_kc, + nc, + alpha); + } - pack_lhs(blockA+info[tid].lhs_start*actual_kc, lhs.getSubMapper(info[tid].lhs_start,k), actual_kc, info[tid].lhs_length); + // Then keep going as usual with the remaining B' + for (Index j = nc; j < cols; j += nc) { + const Index actual_nc = (std::min)(j + nc, cols) - j; - // Notify the other threads that the part A'_i is ready to go. - info[tid].sync = k; + // pack B_k,j to B' + pack_rhs(blockB, rhs.getSubMapper(k, j), actual_kc, actual_nc); - // Computes C_i += A' * B' per A'_i - for(int shift=0; shift0) { - while(info[i].sync!=k) { + // C_j += A' * B' + gebp(res.getSubMapper(0, j), blockA, blockB, rows, actual_kc, actual_nc, alpha); } - } - - gebp(res.getSubMapper(info[i].lhs_start, 0), blockA+info[i].lhs_start*actual_kc, blockB, info[i].lhs_length, actual_kc, nc, alpha); - } - // Then keep going as usual with the remaining B' - for(Index j=nc; j Pack lhs's panel into a sequential chunk of memory (L2/L3 caching) + // Note that this panel will be read as many times as the number of blocks in the rhs's + // horizontal panel which is, in practice, a very low number. + pack_lhs(blockA, lhs.getSubMapper(i2, k2), actual_kc, actual_mc); - // For each horizontal panel of the rhs, and corresponding panel of the lhs... - for(Index i2=0; i2 Pack lhs's panel into a sequential chunk of memory (L2/L3 caching) - // Note that this panel will be read as many times as the number of blocks in the rhs's - // horizontal panel which is, in practice, a very low number. - pack_lhs(blockA, lhs.getSubMapper(i2,k2), actual_kc, actual_mc); - - // For each kc x nc block of the rhs's horizontal panel... - for(Index j2=0; j2 -struct gemm_functor -{ - gemm_functor(const Lhs& lhs, const Rhs& rhs, Dest& dest, const Scalar& actualAlpha, BlockingType& blocking) - : m_lhs(lhs), m_rhs(rhs), m_dest(dest), m_actualAlpha(actualAlpha), m_blocking(blocking) - {} + }; - void initParallelSession(Index num_threads) const + /********************************************************************************* + * Specialization of generic_product_impl for "large" GEMM, i.e., + * implementation of the high level wrapper to general_matrix_matrix_product + **********************************************************************************/ + + template + struct gemm_functor { - m_blocking.initParallel(m_lhs.rows(), m_rhs.cols(), m_lhs.cols(), num_threads); - m_blocking.allocateA(); - } + gemm_functor(const Lhs &lhs, const Rhs &rhs, Dest &dest, const Scalar &actualAlpha, BlockingType &blocking) + : m_lhs(lhs), m_rhs(rhs), m_dest(dest), m_actualAlpha(actualAlpha), m_blocking(blocking) + {} - void operator() (Index row, Index rows, Index col=0, Index cols=-1, GemmParallelInfo* info=0) const - { - if(cols==-1) - cols = m_rhs.cols(); + void initParallelSession(Index num_threads) const + { + m_blocking.initParallel(m_lhs.rows(), m_rhs.cols(), m_lhs.cols(), num_threads); + m_blocking.allocateA(); + } - Gemm::run(rows, cols, m_lhs.cols(), - &m_lhs.coeffRef(row,0), m_lhs.outerStride(), - &m_rhs.coeffRef(0,col), m_rhs.outerStride(), - (Scalar*)&(m_dest.coeffRef(row,col)), m_dest.outerStride(), - m_actualAlpha, m_blocking, info); - } + void operator()(Index row, Index rows, Index col = 0, Index cols = -1, GemmParallelInfo *info = 0) const + { + if (cols == -1) cols = m_rhs.cols(); + + Gemm::run(rows, + cols, + m_lhs.cols(), + &m_lhs.coeffRef(row, 0), + m_lhs.outerStride(), + &m_rhs.coeffRef(0, col), + m_rhs.outerStride(), + (Scalar *)&(m_dest.coeffRef(row, col)), + m_dest.outerStride(), + m_actualAlpha, + m_blocking, + info); + } - typedef typename Gemm::Traits Traits; + typedef typename Gemm::Traits Traits; protected: - const Lhs& m_lhs; - const Rhs& m_rhs; - Dest& m_dest; + const Lhs &m_lhs; + const Rhs &m_rhs; + Dest &m_dest; Scalar m_actualAlpha; - BlockingType& m_blocking; -}; - -template class gemm_blocking_space; + BlockingType &m_blocking; + }; -template -class level3_blocking -{ + template + class gemm_blocking_space; + + template class level3_blocking + { typedef _LhsScalar LhsScalar; typedef _RhsScalar RhsScalar; protected: - LhsScalar* m_blockA; - RhsScalar* m_blockB; + LhsScalar *m_blockA; + RhsScalar *m_blockB; Index m_mc; Index m_nc; Index m_kc; public: - - level3_blocking() - : m_blockA(0), m_blockB(0), m_mc(0), m_nc(0), m_kc(0) - {} + level3_blocking() : m_blockA(0), m_blockB(0), m_mc(0), m_nc(0), m_kc(0) {} inline Index mc() const { return m_mc; } inline Index nc() const { return m_nc; } inline Index kc() const { return m_kc; } - inline LhsScalar* blockA() { return m_blockA; } - inline RhsScalar* blockB() { return m_blockB; } -}; + inline LhsScalar *blockA() { return m_blockA; } + inline RhsScalar *blockB() { return m_blockB; } + }; -template -class gemm_blocking_space - : public level3_blocking< - typename conditional::type, - typename conditional::type> -{ + template + class gemm_blocking_space + : public level3_blocking::type, + typename conditional::type> + { enum { - Transpose = StorageOrder==RowMajor, + Transpose = StorageOrder == RowMajor, ActualRows = Transpose ? MaxCols : MaxRows, ActualCols = Transpose ? MaxRows : MaxCols }; - typedef typename conditional::type LhsScalar; - typedef typename conditional::type RhsScalar; - typedef gebp_traits Traits; - enum { - SizeA = ActualRows * MaxDepth, - SizeB = ActualCols * MaxDepth - }; + typedef typename conditional::type LhsScalar; + typedef typename conditional::type RhsScalar; + typedef gebp_traits Traits; + enum { SizeA = ActualRows * MaxDepth, SizeB = ActualCols * MaxDepth }; #if EIGEN_MAX_STATIC_ALIGN_BYTES >= EIGEN_DEFAULT_ALIGN_BYTES EIGEN_ALIGN_MAX LhsScalar m_staticA[SizeA]; EIGEN_ALIGN_MAX RhsScalar m_staticB[SizeB]; #else - EIGEN_ALIGN_MAX char m_staticA[SizeA * sizeof(LhsScalar) + EIGEN_DEFAULT_ALIGN_BYTES-1]; - EIGEN_ALIGN_MAX char m_staticB[SizeB * sizeof(RhsScalar) + EIGEN_DEFAULT_ALIGN_BYTES-1]; + EIGEN_ALIGN_MAX char m_staticA[SizeA * sizeof(LhsScalar) + EIGEN_DEFAULT_ALIGN_BYTES - 1]; + EIGEN_ALIGN_MAX char m_staticB[SizeB * sizeof(RhsScalar) + EIGEN_DEFAULT_ALIGN_BYTES - 1]; #endif public: - - gemm_blocking_space(Index /*rows*/, Index /*cols*/, Index /*depth*/, Index /*num_threads*/, bool /*full_rows = false*/) + gemm_blocking_space(Index /*rows*/, + Index /*cols*/, + Index /*depth*/, + Index /*num_threads*/, + bool /*full_rows = false*/) { this->m_mc = ActualRows; this->m_nc = ActualCols; @@ -309,51 +366,52 @@ class gemm_blocking_spacem_blockA = m_staticA; this->m_blockB = m_staticB; #else - this->m_blockA = reinterpret_cast((internal::UIntPtr(m_staticA) + (EIGEN_DEFAULT_ALIGN_BYTES-1)) & ~std::size_t(EIGEN_DEFAULT_ALIGN_BYTES-1)); - this->m_blockB = reinterpret_cast((internal::UIntPtr(m_staticB) + (EIGEN_DEFAULT_ALIGN_BYTES-1)) & ~std::size_t(EIGEN_DEFAULT_ALIGN_BYTES-1)); + this->m_blockA = reinterpret_cast( + (internal::UIntPtr(m_staticA) + (EIGEN_DEFAULT_ALIGN_BYTES - 1)) & ~std::size_t(EIGEN_DEFAULT_ALIGN_BYTES - 1)); + this->m_blockB = reinterpret_cast( + (internal::UIntPtr(m_staticB) + (EIGEN_DEFAULT_ALIGN_BYTES - 1)) & ~std::size_t(EIGEN_DEFAULT_ALIGN_BYTES - 1)); #endif } - void initParallel(Index, Index, Index, Index) - {} + void initParallel(Index, Index, Index, Index) {} inline void allocateA() {} inline void allocateB() {} inline void allocateAll() {} -}; - -template -class gemm_blocking_space - : public level3_blocking< - typename conditional::type, - typename conditional::type> -{ - enum { - Transpose = StorageOrder==RowMajor - }; - typedef typename conditional::type LhsScalar; - typedef typename conditional::type RhsScalar; - typedef gebp_traits Traits; + }; + + template + class gemm_blocking_space + : public level3_blocking::type, + typename conditional::type> + { + enum { Transpose = StorageOrder == RowMajor }; + typedef typename conditional::type LhsScalar; + typedef typename conditional::type RhsScalar; + typedef gebp_traits Traits; Index m_sizeA; Index m_sizeB; public: - gemm_blocking_space(Index rows, Index cols, Index depth, Index num_threads, bool l3_blocking) { this->m_mc = Transpose ? cols : rows; this->m_nc = Transpose ? rows : cols; this->m_kc = depth; - if(l3_blocking) - { - computeProductBlockingSizes(this->m_kc, this->m_mc, this->m_nc, num_threads); - } - else // no l3 blocking + if (l3_blocking) { + computeProductBlockingSizes(this->m_kc, this->m_mc, this->m_nc, num_threads); + } else// no l3 blocking { Index n = this->m_nc; - computeProductBlockingSizes(this->m_kc, this->m_mc, n, num_threads); + computeProductBlockingSizes(this->m_kc, this->m_mc, n, num_threads); } m_sizeA = this->m_mc * this->m_kc; @@ -366,23 +424,21 @@ class gemm_blocking_spacem_nc = Transpose ? rows : cols; this->m_kc = depth; - eigen_internal_assert(this->m_blockA==0 && this->m_blockB==0); + eigen_internal_assert(this->m_blockA == 0 && this->m_blockB == 0); Index m = this->m_mc; - computeProductBlockingSizes(this->m_kc, m, this->m_nc, num_threads); + computeProductBlockingSizes(this->m_kc, m, this->m_nc, num_threads); m_sizeA = this->m_mc * this->m_kc; m_sizeB = this->m_kc * this->m_nc; } void allocateA() { - if(this->m_blockA==0) - this->m_blockA = aligned_new(m_sizeA); + if (this->m_blockA == 0) this->m_blockA = aligned_new(m_sizeA); } void allocateB() { - if(this->m_blockB==0) - this->m_blockB = aligned_new(m_sizeB); + if (this->m_blockB == 0) this->m_blockB = aligned_new(m_sizeB); } void allocateAll() @@ -396,97 +452,106 @@ class gemm_blocking_spacem_blockA, m_sizeA); aligned_delete(this->m_blockB, m_sizeB); } -}; + }; -} // end namespace internal +}// end namespace internal namespace internal { -template -struct generic_product_impl - : generic_product_impl_base > -{ - typedef typename Product::Scalar Scalar; - typedef typename Lhs::Scalar LhsScalar; - typedef typename Rhs::Scalar RhsScalar; + template + struct generic_product_impl + : generic_product_impl_base> + { + typedef typename Product::Scalar Scalar; + typedef typename Lhs::Scalar LhsScalar; + typedef typename Rhs::Scalar RhsScalar; - typedef internal::blas_traits LhsBlasTraits; - typedef typename LhsBlasTraits::DirectLinearAccessType ActualLhsType; - typedef typename internal::remove_all::type ActualLhsTypeCleaned; + typedef internal::blas_traits LhsBlasTraits; + typedef typename LhsBlasTraits::DirectLinearAccessType ActualLhsType; + typedef typename internal::remove_all::type ActualLhsTypeCleaned; - typedef internal::blas_traits RhsBlasTraits; - typedef typename RhsBlasTraits::DirectLinearAccessType ActualRhsType; - typedef typename internal::remove_all::type ActualRhsTypeCleaned; + typedef internal::blas_traits RhsBlasTraits; + typedef typename RhsBlasTraits::DirectLinearAccessType ActualRhsType; + typedef typename internal::remove_all::type ActualRhsTypeCleaned; - enum { - MaxDepthAtCompileTime = EIGEN_SIZE_MIN_PREFER_FIXED(Lhs::MaxColsAtCompileTime,Rhs::MaxRowsAtCompileTime) - }; + enum { MaxDepthAtCompileTime = EIGEN_SIZE_MIN_PREFER_FIXED(Lhs::MaxColsAtCompileTime, Rhs::MaxRowsAtCompileTime) }; - typedef generic_product_impl lazyproduct; + typedef generic_product_impl lazyproduct; - template - static void evalTo(Dst& dst, const Lhs& lhs, const Rhs& rhs) - { - if((rhs.rows()+dst.rows()+dst.cols())<20 && rhs.rows()>0) - lazyproduct::evalTo(dst, lhs, rhs); - else + template static void evalTo(Dst &dst, const Lhs &lhs, const Rhs &rhs) { - dst.setZero(); - scaleAndAddTo(dst, lhs, rhs, Scalar(1)); + if ((rhs.rows() + dst.rows() + dst.cols()) < 20 && rhs.rows() > 0) + lazyproduct::evalTo(dst, lhs, rhs); + else { + dst.setZero(); + scaleAndAddTo(dst, lhs, rhs, Scalar(1)); + } } - } - template - static void addTo(Dst& dst, const Lhs& lhs, const Rhs& rhs) - { - if((rhs.rows()+dst.rows()+dst.cols())<20 && rhs.rows()>0) - lazyproduct::addTo(dst, lhs, rhs); - else - scaleAndAddTo(dst,lhs, rhs, Scalar(1)); - } - - template - static void subTo(Dst& dst, const Lhs& lhs, const Rhs& rhs) - { - if((rhs.rows()+dst.rows()+dst.cols())<20 && rhs.rows()>0) - lazyproduct::subTo(dst, lhs, rhs); - else - scaleAndAddTo(dst, lhs, rhs, Scalar(-1)); - } - - template - static void scaleAndAddTo(Dest& dst, const Lhs& a_lhs, const Rhs& a_rhs, const Scalar& alpha) - { - eigen_assert(dst.rows()==a_lhs.rows() && dst.cols()==a_rhs.cols()); - if(a_lhs.cols()==0 || a_lhs.rows()==0 || a_rhs.cols()==0) - return; + template static void addTo(Dst &dst, const Lhs &lhs, const Rhs &rhs) + { + if ((rhs.rows() + dst.rows() + dst.cols()) < 20 && rhs.rows() > 0) + lazyproduct::addTo(dst, lhs, rhs); + else + scaleAndAddTo(dst, lhs, rhs, Scalar(1)); + } - typename internal::add_const_on_value_type::type lhs = LhsBlasTraits::extract(a_lhs); - typename internal::add_const_on_value_type::type rhs = RhsBlasTraits::extract(a_rhs); + template static void subTo(Dst &dst, const Lhs &lhs, const Rhs &rhs) + { + if ((rhs.rows() + dst.rows() + dst.cols()) < 20 && rhs.rows() > 0) + lazyproduct::subTo(dst, lhs, rhs); + else + scaleAndAddTo(dst, lhs, rhs, Scalar(-1)); + } - Scalar actualAlpha = alpha * LhsBlasTraits::extractScalarFactor(a_lhs) - * RhsBlasTraits::extractScalarFactor(a_rhs); + template + static void scaleAndAddTo(Dest &dst, const Lhs &a_lhs, const Rhs &a_rhs, const Scalar &alpha) + { + eigen_assert(dst.rows() == a_lhs.rows() && dst.cols() == a_rhs.cols()); + if (a_lhs.cols() == 0 || a_lhs.rows() == 0 || a_rhs.cols() == 0) return; - typedef internal::gemm_blocking_space<(Dest::Flags&RowMajorBit) ? RowMajor : ColMajor,LhsScalar,RhsScalar, - Dest::MaxRowsAtCompileTime,Dest::MaxColsAtCompileTime,MaxDepthAtCompileTime> BlockingType; + typename internal::add_const_on_value_type::type lhs = LhsBlasTraits::extract(a_lhs); + typename internal::add_const_on_value_type::type rhs = RhsBlasTraits::extract(a_rhs); - typedef internal::gemm_functor< - Scalar, Index, - internal::general_matrix_matrix_product< - Index, - LhsScalar, (ActualLhsTypeCleaned::Flags&RowMajorBit) ? RowMajor : ColMajor, bool(LhsBlasTraits::NeedToConjugate), - RhsScalar, (ActualRhsTypeCleaned::Flags&RowMajorBit) ? RowMajor : ColMajor, bool(RhsBlasTraits::NeedToConjugate), - (Dest::Flags&RowMajorBit) ? RowMajor : ColMajor>, - ActualLhsTypeCleaned, ActualRhsTypeCleaned, Dest, BlockingType> GemmFunctor; + Scalar actualAlpha = + alpha * LhsBlasTraits::extractScalarFactor(a_lhs) * RhsBlasTraits::extractScalarFactor(a_rhs); - BlockingType blocking(dst.rows(), dst.cols(), lhs.cols(), 1, true); - internal::parallelize_gemm<(Dest::MaxRowsAtCompileTime>32 || Dest::MaxRowsAtCompileTime==Dynamic)> - (GemmFunctor(lhs, rhs, dst, actualAlpha, blocking), a_lhs.rows(), a_rhs.cols(), a_lhs.cols(), Dest::Flags&RowMajorBit); - } -}; + typedef internal::gemm_blocking_space<(Dest::Flags & RowMajorBit) ? RowMajor : ColMajor, + LhsScalar, + RhsScalar, + Dest::MaxRowsAtCompileTime, + Dest::MaxColsAtCompileTime, + MaxDepthAtCompileTime> + BlockingType; + + typedef internal::gemm_functor, + ActualLhsTypeCleaned, + ActualRhsTypeCleaned, + Dest, + BlockingType> + GemmFunctor; + + BlockingType blocking(dst.rows(), dst.cols(), lhs.cols(), 1, true); + internal::parallelize_gemm<(Dest::MaxRowsAtCompileTime > 32 || Dest::MaxRowsAtCompileTime == Dynamic)>( + GemmFunctor(lhs, rhs, dst, actualAlpha, blocking), + a_lhs.rows(), + a_rhs.cols(), + a_lhs.cols(), + Dest::Flags & RowMajorBit); + } + }; -} // end namespace internal +}// end namespace internal -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_GENERAL_MATRIX_MATRIX_H +#endif// EIGEN_GENERAL_MATRIX_MATRIX_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/products/GeneralMatrixMatrixTriangular.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/products/GeneralMatrixMatrixTriangular.h index e844e37d..7f678c0b 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/products/GeneralMatrixMatrixTriangular.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/products/GeneralMatrixMatrixTriangular.h @@ -10,185 +10,265 @@ #ifndef EIGEN_GENERAL_MATRIX_MATRIX_TRIANGULAR_H #define EIGEN_GENERAL_MATRIX_MATRIX_TRIANGULAR_H -namespace Eigen { +namespace Eigen { template struct selfadjoint_rank1_update; namespace internal { -/********************************************************************** -* This file implements a general A * B product while -* evaluating only one triangular part of the product. -* This is a more general version of self adjoint product (C += A A^T) -* as the level 3 SYRK Blas routine. -**********************************************************************/ - -// forward declarations (defined at the end of this file) -template -struct tribb_kernel; - -/* Optimized matrix-matrix product evaluating only one triangular half */ -template -struct general_matrix_matrix_triangular_product; - -// as usual if the result is row major => we transpose the product -template -struct general_matrix_matrix_triangular_product -{ - typedef typename ScalarBinaryOpTraits::ReturnType ResScalar; - static EIGEN_STRONG_INLINE void run(Index size, Index depth,const LhsScalar* lhs, Index lhsStride, - const RhsScalar* rhs, Index rhsStride, ResScalar* res, Index resStride, - const ResScalar& alpha, level3_blocking& blocking) + /********************************************************************** + * This file implements a general A * B product while + * evaluating only one triangular part of the product. + * This is a more general version of self adjoint product (C += A A^T) + * as the level 3 SYRK Blas routine. + **********************************************************************/ + + // forward declarations (defined at the end of this file) + template + struct tribb_kernel; + + /* Optimized matrix-matrix product evaluating only one triangular half */ + template + struct general_matrix_matrix_triangular_product; + + // as usual if the result is row major => we transpose the product + template + struct general_matrix_matrix_triangular_product { - general_matrix_matrix_triangular_product - ::run(size,depth,rhs,rhsStride,lhs,lhsStride,res,resStride,alpha,blocking); - } -}; + typedef typename ScalarBinaryOpTraits::ReturnType ResScalar; + static EIGEN_STRONG_INLINE void run(Index size, + Index depth, + const LhsScalar *lhs, + Index lhsStride, + const RhsScalar *rhs, + Index rhsStride, + ResScalar *res, + Index resStride, + const ResScalar &alpha, + level3_blocking &blocking) + { + general_matrix_matrix_triangular_product::run(size, depth, rhs, rhsStride, lhs, lhsStride, res, resStride, alpha, blocking); + } + }; -template -struct general_matrix_matrix_triangular_product -{ - typedef typename ScalarBinaryOpTraits::ReturnType ResScalar; - static EIGEN_STRONG_INLINE void run(Index size, Index depth,const LhsScalar* _lhs, Index lhsStride, - const RhsScalar* _rhs, Index rhsStride, ResScalar* _res, Index resStride, - const ResScalar& alpha, level3_blocking& blocking) + template + struct general_matrix_matrix_triangular_product { - typedef gebp_traits Traits; - - typedef const_blas_data_mapper LhsMapper; - typedef const_blas_data_mapper RhsMapper; - typedef blas_data_mapper ResMapper; - LhsMapper lhs(_lhs,lhsStride); - RhsMapper rhs(_rhs,rhsStride); - ResMapper res(_res, resStride); - - Index kc = blocking.kc(); - Index mc = (std::min)(size,blocking.mc()); - - // !!! mc must be a multiple of nr: - if(mc > Traits::nr) - mc = (mc/Traits::nr)*Traits::nr; - - std::size_t sizeA = kc*mc; - std::size_t sizeB = kc*size; - - ei_declare_aligned_stack_constructed_variable(LhsScalar, blockA, sizeA, blocking.blockA()); - ei_declare_aligned_stack_constructed_variable(RhsScalar, blockB, sizeB, blocking.blockB()); - - gemm_pack_lhs pack_lhs; - gemm_pack_rhs pack_rhs; - gebp_kernel gebp; - tribb_kernel sybb; - - for(Index k2=0; k2::ReturnType ResScalar; + static EIGEN_STRONG_INLINE void run(Index size, + Index depth, + const LhsScalar *_lhs, + Index lhsStride, + const RhsScalar *_rhs, + Index rhsStride, + ResScalar *_res, + Index resStride, + const ResScalar &alpha, + level3_blocking &blocking) { - const Index actual_kc = (std::min)(k2+kc,depth)-k2; - - // note that the actual rhs is the transpose/adjoint of mat - pack_rhs(blockB, rhs.getSubMapper(k2,0), actual_kc, size); - - for(Index i2=0; i2 processed with gebp or skipped - // 2 - the actual_mc x actual_mc symmetric block => processed with a special kernel - // 3 - after the diagonal => processed with gebp or skipped - if (UpLo==Lower) - gebp(res.getSubMapper(i2, 0), blockA, blockB, actual_mc, actual_kc, - (std::min)(size,i2), alpha, -1, -1, 0, 0); - - - sybb(_res+resStride*i2 + i2, resStride, blockA, blockB + actual_kc*i2, actual_mc, actual_kc, alpha); - - if (UpLo==Upper) - { - Index j2 = i2+actual_mc; - gebp(res.getSubMapper(i2, j2), blockA, blockB+actual_kc*j2, actual_mc, - actual_kc, (std::max)(Index(0), size-j2), alpha, -1, -1, 0, 0); + typedef gebp_traits Traits; + + typedef const_blas_data_mapper LhsMapper; + typedef const_blas_data_mapper RhsMapper; + typedef blas_data_mapper ResMapper; + LhsMapper lhs(_lhs, lhsStride); + RhsMapper rhs(_rhs, rhsStride); + ResMapper res(_res, resStride); + + Index kc = blocking.kc(); + Index mc = (std::min)(size, blocking.mc()); + + // !!! mc must be a multiple of nr: + if (mc > Traits::nr) mc = (mc / Traits::nr) * Traits::nr; + + std::size_t sizeA = kc * mc; + std::size_t sizeB = kc * size; + + ei_declare_aligned_stack_constructed_variable(LhsScalar, blockA, sizeA, blocking.blockA()); + ei_declare_aligned_stack_constructed_variable(RhsScalar, blockB, sizeB, blocking.blockB()); + + gemm_pack_lhs pack_lhs; + gemm_pack_rhs pack_rhs; + gebp_kernel gebp; + tribb_kernel sybb; + + for (Index k2 = 0; k2 < depth; k2 += kc) { + const Index actual_kc = (std::min)(k2 + kc, depth) - k2; + + // note that the actual rhs is the transpose/adjoint of mat + pack_rhs(blockB, rhs.getSubMapper(k2, 0), actual_kc, size); + + for (Index i2 = 0; i2 < size; i2 += mc) { + const Index actual_mc = (std::min)(i2 + mc, size) - i2; + + pack_lhs(blockA, lhs.getSubMapper(i2, k2), actual_kc, actual_mc); + + // the selected actual_mc * size panel of res is split into three different part: + // 1 - before the diagonal => processed with gebp or skipped + // 2 - the actual_mc x actual_mc symmetric block => processed with a special kernel + // 3 - after the diagonal => processed with gebp or skipped + if (UpLo == Lower) + gebp( + res.getSubMapper(i2, 0), blockA, blockB, actual_mc, actual_kc, (std::min)(size, i2), alpha, -1, -1, 0, 0); + + + sybb(_res + resStride * i2 + i2, resStride, blockA, blockB + actual_kc * i2, actual_mc, actual_kc, alpha); + + if (UpLo == Upper) { + Index j2 = i2 + actual_mc; + gebp(res.getSubMapper(i2, j2), + blockA, + blockB + actual_kc * j2, + actual_mc, + actual_kc, + (std::max)(Index(0), size - j2), + alpha, + -1, + -1, + 0, + 0); + } } } } - } -}; - -// Optimized packed Block * packed Block product kernel evaluating only one given triangular part -// This kernel is built on top of the gebp kernel: -// - the current destination block is processed per panel of actual_mc x BlockSize -// where BlockSize is set to the minimal value allowing gebp to be as fast as possible -// - then, as usual, each panel is split into three parts along the diagonal, -// the sub blocks above and below the diagonal are processed as usual, -// while the triangular block overlapping the diagonal is evaluated into a -// small temporary buffer which is then accumulated into the result using a -// triangular traversal. -template -struct tribb_kernel -{ - typedef gebp_traits Traits; - typedef typename Traits::ResScalar ResScalar; - - enum { - BlockSize = meta_least_common_multiple::ret }; - void operator()(ResScalar* _res, Index resStride, const LhsScalar* blockA, const RhsScalar* blockB, Index size, Index depth, const ResScalar& alpha) + + // Optimized packed Block * packed Block product kernel evaluating only one given triangular part + // This kernel is built on top of the gebp kernel: + // - the current destination block is processed per panel of actual_mc x BlockSize + // where BlockSize is set to the minimal value allowing gebp to be as fast as possible + // - then, as usual, each panel is split into three parts along the diagonal, + // the sub blocks above and below the diagonal are processed as usual, + // while the triangular block overlapping the diagonal is evaluated into a + // small temporary buffer which is then accumulated into the result using a + // triangular traversal. + template + struct tribb_kernel { - typedef blas_data_mapper ResMapper; - ResMapper res(_res, resStride); - gebp_kernel gebp_kernel; + typedef gebp_traits Traits; + typedef typename Traits::ResScalar ResScalar; + + enum { BlockSize = meta_least_common_multiple::ret }; + void operator()(ResScalar *_res, + Index resStride, + const LhsScalar *blockA, + const RhsScalar *blockB, + Index size, + Index depth, + const ResScalar &alpha) + { + typedef blas_data_mapper ResMapper; + ResMapper res(_res, resStride); + gebp_kernel gebp_kernel; - Matrix buffer((internal::constructor_without_unaligned_array_assert())); + Matrix buffer( + (internal::constructor_without_unaligned_array_assert())); - // let's process the block per panel of actual_mc x BlockSize, - // again, each is split into three parts, etc. - for (Index j=0; j(BlockSize,size - j); - const RhsScalar* actual_b = blockB+j*depth; - - if(UpLo==Upper) - gebp_kernel(res.getSubMapper(0, j), blockA, actual_b, j, depth, actualBlockSize, alpha, - -1, -1, 0, 0); - - // selfadjoint micro block - { - Index i = j; - buffer.setZero(); - // 1 - apply the kernel on the temporary buffer - gebp_kernel(ResMapper(buffer.data(), BlockSize), blockA+depth*i, actual_b, actualBlockSize, depth, actualBlockSize, alpha, - -1, -1, 0, 0); - // 2 - triangular accumulation - for(Index j1=0; j1(BlockSize, size - j); + const RhsScalar *actual_b = blockB + j * depth; + + if (UpLo == Upper) + gebp_kernel(res.getSubMapper(0, j), blockA, actual_b, j, depth, actualBlockSize, alpha, -1, -1, 0, 0); + + // selfadjoint micro block { - ResScalar* r = &res(i, j + j1); - for(Index i1=UpLo==Lower ? j1 : 0; - UpLo==Lower ? i1 -struct general_product_to_triangular_selector +struct general_product_to_triangular_selector { - static void run(MatrixType& mat, const ProductType& prod, const typename MatrixType::Scalar& alpha, bool beta) + static void run(MatrixType &mat, const ProductType &prod, const typename MatrixType::Scalar &alpha, bool beta) { typedef typename MatrixType::Scalar Scalar; - + typedef typename internal::remove_all::type Lhs; typedef internal::blas_traits LhsBlasTraits; typedef typename LhsBlasTraits::DirectLinearAccessType ActualLhs; typedef typename internal::remove_all::type _ActualLhs; typename internal::add_const_on_value_type::type actualLhs = LhsBlasTraits::extract(prod.lhs()); - + typedef typename internal::remove_all::type Rhs; typedef internal::blas_traits RhsBlasTraits; typedef typename RhsBlasTraits::DirectLinearAccessType ActualRhs; typedef typename internal::remove_all::type _ActualRhs; typename internal::add_const_on_value_type::type actualRhs = RhsBlasTraits::extract(prod.rhs()); - Scalar actualAlpha = alpha * LhsBlasTraits::extractScalarFactor(prod.lhs().derived()) * RhsBlasTraits::extractScalarFactor(prod.rhs().derived()); + Scalar actualAlpha = alpha * LhsBlasTraits::extractScalarFactor(prod.lhs().derived()) + * RhsBlasTraits::extractScalarFactor(prod.rhs().derived()); - if(!beta) - mat.template triangularView().setZero(); + if (!beta) mat.template triangularView().setZero(); enum { - StorageOrder = (internal::traits::Flags&RowMajorBit) ? RowMajor : ColMajor, - UseLhsDirectly = _ActualLhs::InnerStrideAtCompileTime==1, - UseRhsDirectly = _ActualRhs::InnerStrideAtCompileTime==1 + StorageOrder = (internal::traits::Flags & RowMajorBit) ? RowMajor : ColMajor, + UseLhsDirectly = _ActualLhs::InnerStrideAtCompileTime == 1, + UseRhsDirectly = _ActualRhs::InnerStrideAtCompileTime == 1 }; - - internal::gemv_static_vector_if static_lhs; - ei_declare_aligned_stack_constructed_variable(Scalar, actualLhsPtr, actualLhs.size(), - (UseLhsDirectly ? const_cast(actualLhs.data()) : static_lhs.data())); - if(!UseLhsDirectly) Map(actualLhsPtr, actualLhs.size()) = actualLhs; - - internal::gemv_static_vector_if static_rhs; - ei_declare_aligned_stack_constructed_variable(Scalar, actualRhsPtr, actualRhs.size(), - (UseRhsDirectly ? const_cast(actualRhs.data()) : static_rhs.data())); - if(!UseRhsDirectly) Map(actualRhsPtr, actualRhs.size()) = actualRhs; - - - selfadjoint_rank1_update::IsComplex, - RhsBlasTraits::NeedToConjugate && NumTraits::IsComplex> - ::run(actualLhs.size(), mat.data(), mat.outerStride(), actualLhsPtr, actualRhsPtr, actualAlpha); + + internal::gemv_static_vector_if + static_lhs; + ei_declare_aligned_stack_constructed_variable(Scalar, + actualLhsPtr, + actualLhs.size(), + (UseLhsDirectly ? const_cast(actualLhs.data()) : static_lhs.data())); + if (!UseLhsDirectly) Map(actualLhsPtr, actualLhs.size()) = actualLhs; + + internal::gemv_static_vector_if + static_rhs; + ei_declare_aligned_stack_constructed_variable(Scalar, + actualRhsPtr, + actualRhs.size(), + (UseRhsDirectly ? const_cast(actualRhs.data()) : static_rhs.data())); + if (!UseRhsDirectly) Map(actualRhsPtr, actualRhs.size()) = actualRhs; + + + selfadjoint_rank1_update::IsComplex, + RhsBlasTraits::NeedToConjugate && NumTraits::IsComplex>::run(actualLhs.size(), + mat.data(), + mat.outerStride(), + actualLhsPtr, + actualRhsPtr, + actualAlpha); } }; template -struct general_product_to_triangular_selector +struct general_product_to_triangular_selector { - static void run(MatrixType& mat, const ProductType& prod, const typename MatrixType::Scalar& alpha, bool beta) + static void run(MatrixType &mat, const ProductType &prod, const typename MatrixType::Scalar &alpha, bool beta) { typedef typename internal::remove_all::type Lhs; typedef internal::blas_traits LhsBlasTraits; typedef typename LhsBlasTraits::DirectLinearAccessType ActualLhs; typedef typename internal::remove_all::type _ActualLhs; typename internal::add_const_on_value_type::type actualLhs = LhsBlasTraits::extract(prod.lhs()); - + typedef typename internal::remove_all::type Rhs; typedef internal::blas_traits RhsBlasTraits; typedef typename RhsBlasTraits::DirectLinearAccessType ActualRhs; typedef typename internal::remove_all::type _ActualRhs; typename internal::add_const_on_value_type::type actualRhs = RhsBlasTraits::extract(prod.rhs()); - typename ProductType::Scalar actualAlpha = alpha * LhsBlasTraits::extractScalarFactor(prod.lhs().derived()) * RhsBlasTraits::extractScalarFactor(prod.rhs().derived()); + typename ProductType::Scalar actualAlpha = alpha * LhsBlasTraits::extractScalarFactor(prod.lhs().derived()) + * RhsBlasTraits::extractScalarFactor(prod.rhs().derived()); - if(!beta) - mat.template triangularView().setZero(); + if (!beta) mat.template triangularView().setZero(); enum { - IsRowMajor = (internal::traits::Flags&RowMajorBit) ? 1 : 0, - LhsIsRowMajor = _ActualLhs::Flags&RowMajorBit ? 1 : 0, - RhsIsRowMajor = _ActualRhs::Flags&RowMajorBit ? 1 : 0, - SkipDiag = (UpLo&(UnitDiag|ZeroDiag))!=0 + IsRowMajor = (internal::traits::Flags & RowMajorBit) ? 1 : 0, + LhsIsRowMajor = _ActualLhs::Flags & RowMajorBit ? 1 : 0, + RhsIsRowMajor = _ActualRhs::Flags & RowMajorBit ? 1 : 0, + SkipDiag = (UpLo & (UnitDiag | ZeroDiag)) != 0 }; Index size = mat.cols(); - if(SkipDiag) - size--; + if (SkipDiag) size--; Index depth = actualLhs.cols(); - typedef internal::gemm_blocking_space BlockingType; + typedef internal::gemm_blocking_space + BlockingType; BlockingType blocking(size, size, depth, 1, false); internal::general_matrix_matrix_triangular_product - ::run(size, depth, - &actualLhs.coeffRef(SkipDiag&&(UpLo&Lower)==Lower ? 1 : 0,0), actualLhs.outerStride(), - &actualRhs.coeffRef(0,SkipDiag&&(UpLo&Upper)==Upper ? 1 : 0), actualRhs.outerStride(), - mat.data() + (SkipDiag ? (bool(IsRowMajor) != ((UpLo&Lower)==Lower) ? 1 : mat.outerStride() ) : 0), mat.outerStride(), actualAlpha, blocking); + typename Lhs::Scalar, + LhsIsRowMajor ? RowMajor : ColMajor, + LhsBlasTraits::NeedToConjugate, + typename Rhs::Scalar, + RhsIsRowMajor ? RowMajor : ColMajor, + RhsBlasTraits::NeedToConjugate, + IsRowMajor ? RowMajor : ColMajor, + UpLo &(Lower | Upper)>::run(size, + depth, + &actualLhs.coeffRef(SkipDiag && (UpLo & Lower) == Lower ? 1 : 0, 0), + actualLhs.outerStride(), + &actualRhs.coeffRef(0, SkipDiag && (UpLo & Upper) == Upper ? 1 : 0), + actualRhs.outerStride(), + mat.data() + (SkipDiag ? (bool(IsRowMajor) != ((UpLo & Lower) == Lower) ? 1 : mat.outerStride()) : 0), + mat.outerStride(), + actualAlpha, + blocking); } }; template template -TriangularView& TriangularViewImpl::_assignProduct(const ProductType& prod, const Scalar& alpha, bool beta) +TriangularView & + TriangularViewImpl::_assignProduct(const ProductType &prod, const Scalar &alpha, bool beta) { - EIGEN_STATIC_ASSERT((UpLo&UnitDiag)==0, WRITING_TO_TRIANGULAR_PART_WITH_UNIT_DIAGONAL_IS_NOT_SUPPORTED); + EIGEN_STATIC_ASSERT((UpLo & UnitDiag) == 0, WRITING_TO_TRIANGULAR_PART_WITH_UNIT_DIAGONAL_IS_NOT_SUPPORTED); eigen_assert(derived().nestedExpression().rows() == prod.rows() && derived().cols() == prod.cols()); - - general_product_to_triangular_selector::InnerSize==1>::run(derived().nestedExpression().const_cast_derived(), prod, alpha, beta); - + + general_product_to_triangular_selector::InnerSize == 1>:: + run(derived().nestedExpression().const_cast_derived(), prod, alpha, beta); + return derived(); } -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_GENERAL_MATRIX_MATRIX_TRIANGULAR_H +#endif// EIGEN_GENERAL_MATRIX_MATRIX_TRIANGULAR_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/products/GeneralMatrixMatrixTriangular_BLAS.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/products/GeneralMatrixMatrixTriangular_BLAS.h index f6f9ebec..a8097a0c 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/products/GeneralMatrixMatrixTriangular_BLAS.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/products/GeneralMatrixMatrixTriangular_BLAS.h @@ -37,109 +37,170 @@ namespace Eigen { namespace internal { -template -struct general_matrix_matrix_rankupdate : - general_matrix_matrix_triangular_product< - Index,Scalar,AStorageOrder,ConjugateA,Scalar,AStorageOrder,ConjugateA,ResStorageOrder,UpLo,BuiltIn> {}; + template + struct general_matrix_matrix_rankupdate + : general_matrix_matrix_triangular_product + { + }; // try to go to BLAS specialization -#define EIGEN_BLAS_RANKUPDATE_SPECIALIZE(Scalar) \ -template \ -struct general_matrix_matrix_triangular_product { \ - static EIGEN_STRONG_INLINE void run(Index size, Index depth,const Scalar* lhs, Index lhsStride, \ - const Scalar* rhs, Index rhsStride, Scalar* res, Index resStride, Scalar alpha, level3_blocking& blocking) \ - { \ - if ( lhs==rhs && ((UpLo&(Lower|Upper))==UpLo) ) { \ - general_matrix_matrix_rankupdate \ - ::run(size,depth,lhs,lhsStride,rhs,rhsStride,res,resStride,alpha,blocking); \ - } else { \ - general_matrix_matrix_triangular_product \ - ::run(size,depth,lhs,lhsStride,rhs,rhsStride,res,resStride,alpha,blocking); \ - } \ - } \ -}; - -EIGEN_BLAS_RANKUPDATE_SPECIALIZE(double) -EIGEN_BLAS_RANKUPDATE_SPECIALIZE(float) +#define EIGEN_BLAS_RANKUPDATE_SPECIALIZE(Scalar) \ + template \ + struct general_matrix_matrix_triangular_product \ + { \ + static EIGEN_STRONG_INLINE void run(Index size, \ + Index depth, \ + const Scalar *lhs, \ + Index lhsStride, \ + const Scalar *rhs, \ + Index rhsStride, \ + Scalar *res, \ + Index resStride, \ + Scalar alpha, \ + level3_blocking &blocking) \ + { \ + if (lhs == rhs && ((UpLo & (Lower | Upper)) == UpLo)) { \ + general_matrix_matrix_rankupdate::run( \ + size, depth, lhs, lhsStride, rhs, rhsStride, res, resStride, alpha, blocking); \ + } else { \ + general_matrix_matrix_triangular_product::run(size, depth, lhs, lhsStride, rhs, rhsStride, res, resStride, alpha, blocking); \ + } \ + } \ + }; + + EIGEN_BLAS_RANKUPDATE_SPECIALIZE(double) + EIGEN_BLAS_RANKUPDATE_SPECIALIZE(float) // TODO handle complex cases // EIGEN_BLAS_RANKUPDATE_SPECIALIZE(dcomplex) // EIGEN_BLAS_RANKUPDATE_SPECIALIZE(scomplex) // SYRK for float/double -#define EIGEN_BLAS_RANKUPDATE_R(EIGTYPE, BLASTYPE, BLASFUNC) \ -template \ -struct general_matrix_matrix_rankupdate { \ - enum { \ - IsLower = (UpLo&Lower) == Lower, \ - LowUp = IsLower ? Lower : Upper, \ - conjA = ((AStorageOrder==ColMajor) && ConjugateA) ? 1 : 0 \ - }; \ - static EIGEN_STRONG_INLINE void run(Index size, Index depth,const EIGTYPE* lhs, Index lhsStride, \ - const EIGTYPE* /*rhs*/, Index /*rhsStride*/, EIGTYPE* res, Index resStride, EIGTYPE alpha, level3_blocking& /*blocking*/) \ - { \ - /* typedef Matrix MatrixRhs;*/ \ -\ - BlasIndex lda=convert_index(lhsStride), ldc=convert_index(resStride), n=convert_index(size), k=convert_index(depth); \ - char uplo=((IsLower) ? 'L' : 'U'), trans=((AStorageOrder==RowMajor) ? 'T':'N'); \ - EIGTYPE beta(1); \ - BLASFUNC(&uplo, &trans, &n, &k, (const BLASTYPE*)&numext::real_ref(alpha), lhs, &lda, (const BLASTYPE*)&numext::real_ref(beta), res, &ldc); \ - } \ -}; +#define EIGEN_BLAS_RANKUPDATE_R(EIGTYPE, BLASTYPE, BLASFUNC) \ + template \ + struct general_matrix_matrix_rankupdate \ + { \ + enum { \ + IsLower = (UpLo & Lower) == Lower, \ + LowUp = IsLower ? Lower : Upper, \ + conjA = ((AStorageOrder == ColMajor) && ConjugateA) ? 1 : 0 \ + }; \ + static EIGEN_STRONG_INLINE void run(Index size, \ + Index depth, \ + const EIGTYPE *lhs, \ + Index lhsStride, \ + const EIGTYPE * /*rhs*/, \ + Index /*rhsStride*/, \ + EIGTYPE *res, \ + Index resStride, \ + EIGTYPE alpha, \ + level3_blocking & /*blocking*/) \ + { \ + /* typedef Matrix MatrixRhs;*/ \ + \ + BlasIndex lda = convert_index(lhsStride), ldc = convert_index(resStride), \ + n = convert_index(size), k = convert_index(depth); \ + char uplo = ((IsLower) ? 'L' : 'U'), trans = ((AStorageOrder == RowMajor) ? 'T' : 'N'); \ + EIGTYPE beta(1); \ + BLASFUNC(&uplo, \ + &trans, \ + &n, \ + &k, \ + (const BLASTYPE *)&numext::real_ref(alpha), \ + lhs, \ + &lda, \ + (const BLASTYPE *)&numext::real_ref(beta), \ + res, \ + &ldc); \ + } \ + }; // HERK for complex data -#define EIGEN_BLAS_RANKUPDATE_C(EIGTYPE, BLASTYPE, RTYPE, BLASFUNC) \ -template \ -struct general_matrix_matrix_rankupdate { \ - enum { \ - IsLower = (UpLo&Lower) == Lower, \ - LowUp = IsLower ? Lower : Upper, \ - conjA = (((AStorageOrder==ColMajor) && ConjugateA) || ((AStorageOrder==RowMajor) && !ConjugateA)) ? 1 : 0 \ - }; \ - static EIGEN_STRONG_INLINE void run(Index size, Index depth,const EIGTYPE* lhs, Index lhsStride, \ - const EIGTYPE* /*rhs*/, Index /*rhsStride*/, EIGTYPE* res, Index resStride, EIGTYPE alpha, level3_blocking& /*blocking*/) \ - { \ - typedef Matrix MatrixType; \ -\ - BlasIndex lda=convert_index(lhsStride), ldc=convert_index(resStride), n=convert_index(size), k=convert_index(depth); \ - char uplo=((IsLower) ? 'L' : 'U'), trans=((AStorageOrder==RowMajor) ? 'C':'N'); \ - RTYPE alpha_, beta_; \ - const EIGTYPE* a_ptr; \ -\ - alpha_ = alpha.real(); \ - beta_ = 1.0; \ -/* Copy with conjugation in some cases*/ \ - MatrixType a; \ - if (conjA) { \ - Map > mapA(lhs,n,k,OuterStride<>(lhsStride)); \ - a = mapA.conjugate(); \ - lda = a.outerStride(); \ - a_ptr = a.data(); \ - } else a_ptr=lhs; \ - BLASFUNC(&uplo, &trans, &n, &k, &alpha_, (BLASTYPE*)a_ptr, &lda, &beta_, (BLASTYPE*)res, &ldc); \ - } \ -}; +#define EIGEN_BLAS_RANKUPDATE_C(EIGTYPE, BLASTYPE, RTYPE, BLASFUNC) \ + template \ + struct general_matrix_matrix_rankupdate \ + { \ + enum { \ + IsLower = (UpLo & Lower) == Lower, \ + LowUp = IsLower ? Lower : Upper, \ + conjA = (((AStorageOrder == ColMajor) && ConjugateA) || ((AStorageOrder == RowMajor) && !ConjugateA)) ? 1 : 0 \ + }; \ + static EIGEN_STRONG_INLINE void run(Index size, \ + Index depth, \ + const EIGTYPE *lhs, \ + Index lhsStride, \ + const EIGTYPE * /*rhs*/, \ + Index /*rhsStride*/, \ + EIGTYPE *res, \ + Index resStride, \ + EIGTYPE alpha, \ + level3_blocking & /*blocking*/) \ + { \ + typedef Matrix MatrixType; \ + \ + BlasIndex lda = convert_index(lhsStride), ldc = convert_index(resStride), \ + n = convert_index(size), k = convert_index(depth); \ + char uplo = ((IsLower) ? 'L' : 'U'), trans = ((AStorageOrder == RowMajor) ? 'C' : 'N'); \ + RTYPE alpha_, beta_; \ + const EIGTYPE *a_ptr; \ + \ + alpha_ = alpha.real(); \ + beta_ = 1.0; \ + /* Copy with conjugation in some cases*/ \ + MatrixType a; \ + if (conjA) { \ + Map> mapA(lhs, n, k, OuterStride<>(lhsStride)); \ + a = mapA.conjugate(); \ + lda = a.outerStride(); \ + a_ptr = a.data(); \ + } else \ + a_ptr = lhs; \ + BLASFUNC(&uplo, &trans, &n, &k, &alpha_, (BLASTYPE *)a_ptr, &lda, &beta_, (BLASTYPE *)res, &ldc); \ + } \ + }; #ifdef EIGEN_USE_MKL -EIGEN_BLAS_RANKUPDATE_R(double, double, dsyrk) -EIGEN_BLAS_RANKUPDATE_R(float, float, ssyrk) + EIGEN_BLAS_RANKUPDATE_R(double, double, dsyrk) + EIGEN_BLAS_RANKUPDATE_R(float, float, ssyrk) #else -EIGEN_BLAS_RANKUPDATE_R(double, double, dsyrk_) -EIGEN_BLAS_RANKUPDATE_R(float, float, ssyrk_) + EIGEN_BLAS_RANKUPDATE_R(double, double, dsyrk_) + EIGEN_BLAS_RANKUPDATE_R(float, float, ssyrk_) #endif -// TODO hanlde complex cases -// EIGEN_BLAS_RANKUPDATE_C(dcomplex, double, double, zherk_) -// EIGEN_BLAS_RANKUPDATE_C(scomplex, float, float, cherk_) + // TODO hanlde complex cases + // EIGEN_BLAS_RANKUPDATE_C(dcomplex, double, double, zherk_) + // EIGEN_BLAS_RANKUPDATE_C(scomplex, float, float, cherk_) -} // end namespace internal +}// end namespace internal -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_GENERAL_MATRIX_MATRIX_TRIANGULAR_BLAS_H +#endif// EIGEN_GENERAL_MATRIX_MATRIX_TRIANGULAR_BLAS_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/products/GeneralMatrixMatrix_BLAS.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/products/GeneralMatrixMatrix_BLAS.h index b0f6b0d5..f95bb2ed 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/products/GeneralMatrixMatrix_BLAS.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/products/GeneralMatrixMatrix_BLAS.h @@ -33,90 +33,114 @@ #ifndef EIGEN_GENERAL_MATRIX_MATRIX_BLAS_H #define EIGEN_GENERAL_MATRIX_MATRIX_BLAS_H -namespace Eigen { +namespace Eigen { namespace internal { -/********************************************************************** -* This file implements general matrix-matrix multiplication using BLAS -* gemm function via partial specialization of -* general_matrix_matrix_product::run(..) method for float, double, -* std::complex and std::complex types -**********************************************************************/ + /********************************************************************** + * This file implements general matrix-matrix multiplication using BLAS + * gemm function via partial specialization of + * general_matrix_matrix_product::run(..) method for float, double, + * std::complex and std::complex types + **********************************************************************/ -// gemm specialization + // gemm specialization -#define GEMM_SPECIALIZATION(EIGTYPE, EIGPREFIX, BLASTYPE, BLASFUNC) \ -template< \ - typename Index, \ - int LhsStorageOrder, bool ConjugateLhs, \ - int RhsStorageOrder, bool ConjugateRhs> \ -struct general_matrix_matrix_product \ -{ \ -typedef gebp_traits Traits; \ -\ -static void run(Index rows, Index cols, Index depth, \ - const EIGTYPE* _lhs, Index lhsStride, \ - const EIGTYPE* _rhs, Index rhsStride, \ - EIGTYPE* res, Index resStride, \ - EIGTYPE alpha, \ - level3_blocking& /*blocking*/, \ - GemmParallelInfo* /*info = 0*/) \ -{ \ - using std::conj; \ -\ - char transa, transb; \ - BlasIndex m, n, k, lda, ldb, ldc; \ - const EIGTYPE *a, *b; \ - EIGTYPE beta(1); \ - MatrixX##EIGPREFIX a_tmp, b_tmp; \ -\ -/* Set transpose options */ \ - transa = (LhsStorageOrder==RowMajor) ? ((ConjugateLhs) ? 'C' : 'T') : 'N'; \ - transb = (RhsStorageOrder==RowMajor) ? ((ConjugateRhs) ? 'C' : 'T') : 'N'; \ -\ -/* Set m, n, k */ \ - m = convert_index(rows); \ - n = convert_index(cols); \ - k = convert_index(depth); \ -\ -/* Set lda, ldb, ldc */ \ - lda = convert_index(lhsStride); \ - ldb = convert_index(rhsStride); \ - ldc = convert_index(resStride); \ -\ -/* Set a, b, c */ \ - if ((LhsStorageOrder==ColMajor) && (ConjugateLhs)) { \ - Map > lhs(_lhs,m,k,OuterStride<>(lhsStride)); \ - a_tmp = lhs.conjugate(); \ - a = a_tmp.data(); \ - lda = convert_index(a_tmp.outerStride()); \ - } else a = _lhs; \ -\ - if ((RhsStorageOrder==ColMajor) && (ConjugateRhs)) { \ - Map > rhs(_rhs,k,n,OuterStride<>(rhsStride)); \ - b_tmp = rhs.conjugate(); \ - b = b_tmp.data(); \ - ldb = convert_index(b_tmp.outerStride()); \ - } else b = _rhs; \ -\ - BLASFUNC(&transa, &transb, &m, &n, &k, (const BLASTYPE*)&numext::real_ref(alpha), (const BLASTYPE*)a, &lda, (const BLASTYPE*)b, &ldb, (const BLASTYPE*)&numext::real_ref(beta), (BLASTYPE*)res, &ldc); \ -}}; +#define GEMM_SPECIALIZATION(EIGTYPE, EIGPREFIX, BLASTYPE, BLASFUNC) \ + template \ + struct general_matrix_matrix_product \ + { \ + typedef gebp_traits Traits; \ + \ + static void run(Index rows, \ + Index cols, \ + Index depth, \ + const EIGTYPE *_lhs, \ + Index lhsStride, \ + const EIGTYPE *_rhs, \ + Index rhsStride, \ + EIGTYPE *res, \ + Index resStride, \ + EIGTYPE alpha, \ + level3_blocking & /*blocking*/, \ + GemmParallelInfo * /*info = 0*/) \ + { \ + using std::conj; \ + \ + char transa, transb; \ + BlasIndex m, n, k, lda, ldb, ldc; \ + const EIGTYPE *a, *b; \ + EIGTYPE beta(1); \ + MatrixX##EIGPREFIX a_tmp, b_tmp; \ + \ + /* Set transpose options */ \ + transa = (LhsStorageOrder == RowMajor) ? ((ConjugateLhs) ? 'C' : 'T') : 'N'; \ + transb = (RhsStorageOrder == RowMajor) ? ((ConjugateRhs) ? 'C' : 'T') : 'N'; \ + \ + /* Set m, n, k */ \ + m = convert_index(rows); \ + n = convert_index(cols); \ + k = convert_index(depth); \ + \ + /* Set lda, ldb, ldc */ \ + lda = convert_index(lhsStride); \ + ldb = convert_index(rhsStride); \ + ldc = convert_index(resStride); \ + \ + /* Set a, b, c */ \ + if ((LhsStorageOrder == ColMajor) && (ConjugateLhs)) { \ + Map> lhs(_lhs, m, k, OuterStride<>(lhsStride)); \ + a_tmp = lhs.conjugate(); \ + a = a_tmp.data(); \ + lda = convert_index(a_tmp.outerStride()); \ + } else \ + a = _lhs; \ + \ + if ((RhsStorageOrder == ColMajor) && (ConjugateRhs)) { \ + Map> rhs(_rhs, k, n, OuterStride<>(rhsStride)); \ + b_tmp = rhs.conjugate(); \ + b = b_tmp.data(); \ + ldb = convert_index(b_tmp.outerStride()); \ + } else \ + b = _rhs; \ + \ + BLASFUNC(&transa, \ + &transb, \ + &m, \ + &n, \ + &k, \ + (const BLASTYPE *)&numext::real_ref(alpha), \ + (const BLASTYPE *)a, \ + &lda, \ + (const BLASTYPE *)b, \ + &ldb, \ + (const BLASTYPE *)&numext::real_ref(beta), \ + (BLASTYPE *)res, \ + &ldc); \ + } \ + }; #ifdef EIGEN_USE_MKL -GEMM_SPECIALIZATION(double, d, double, dgemm) -GEMM_SPECIALIZATION(float, f, float, sgemm) -GEMM_SPECIALIZATION(dcomplex, cd, MKL_Complex16, zgemm) -GEMM_SPECIALIZATION(scomplex, cf, MKL_Complex8, cgemm) + GEMM_SPECIALIZATION(double, d, double, dgemm) + GEMM_SPECIALIZATION(float, f, float, sgemm) + GEMM_SPECIALIZATION(dcomplex, cd, MKL_Complex16, zgemm) + GEMM_SPECIALIZATION(scomplex, cf, MKL_Complex8, cgemm) #else -GEMM_SPECIALIZATION(double, d, double, dgemm_) -GEMM_SPECIALIZATION(float, f, float, sgemm_) -GEMM_SPECIALIZATION(dcomplex, cd, double, zgemm_) -GEMM_SPECIALIZATION(scomplex, cf, float, cgemm_) + GEMM_SPECIALIZATION(double, d, double, dgemm_) + GEMM_SPECIALIZATION(float, f, float, sgemm_) + GEMM_SPECIALIZATION(dcomplex, cd, double, zgemm_) + GEMM_SPECIALIZATION(scomplex, cf, float, cgemm_) #endif -} // end namespase internal +}// namespace internal -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_GENERAL_MATRIX_MATRIX_BLAS_H +#endif// EIGEN_GENERAL_MATRIX_MATRIX_BLAS_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/products/GeneralMatrixVector.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/products/GeneralMatrixVector.h index a597c1f4..64a35e8c 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/products/GeneralMatrixVector.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/products/GeneralMatrixVector.h @@ -14,495 +14,524 @@ namespace Eigen { namespace internal { -/* Optimized col-major matrix * vector product: - * This algorithm processes 4 columns at onces that allows to both reduce - * the number of load/stores of the result by a factor 4 and to reduce - * the instruction dependency. Moreover, we know that all bands have the - * same alignment pattern. - * - * Mixing type logic: C += alpha * A * B - * | A | B |alpha| comments - * |real |cplx |cplx | no vectorization - * |real |cplx |real | alpha is converted to a cplx when calling the run function, no vectorization - * |cplx |real |cplx | invalid, the caller has to do tmp: = A * B; C += alpha*tmp - * |cplx |real |real | optimal case, vectorization possible via real-cplx mul - * - * Accesses to the matrix coefficients follow the following logic: - * - * - if all columns have the same alignment then - * - if the columns have the same alignment as the result vector, then easy! (-> AllAligned case) - * - otherwise perform unaligned loads only (-> NoneAligned case) - * - otherwise - * - if even columns have the same alignment then - * // odd columns are guaranteed to have the same alignment too - * - if even or odd columns have the same alignment as the result, then - * // for a register size of 2 scalars, this is guarantee to be the case (e.g., SSE with double) - * - perform half aligned and half unaligned loads (-> EvenAligned case) - * - otherwise perform unaligned loads only (-> NoneAligned case) - * - otherwise, if the register size is 4 scalars (e.g., SSE with float) then - * - one over 4 consecutive columns is guaranteed to be aligned with the result vector, - * perform simple aligned loads for this column and aligned loads plus re-alignment for the other. (-> FirstAligned case) - * // this re-alignment is done by the palign function implemented for SSE in Eigen/src/Core/arch/SSE/PacketMath.h - * - otherwise, - * // if we get here, this means the register size is greater than 4 (e.g., AVX with floats), - * // we currently fall back to the NoneAligned case - * - * The same reasoning apply for the transposed case. - * - * The last case (PacketSize>4) could probably be improved by generalizing the FirstAligned case, but since we do not support AVX yet... - * One might also wonder why in the EvenAligned case we perform unaligned loads instead of using the aligned-loads plus re-alignment - * strategy as in the FirstAligned case. The reason is that we observed that unaligned loads on a 8 byte boundary are not too slow - * compared to unaligned loads on a 4 byte boundary. - * - */ -template -struct general_matrix_vector_product -{ - typedef typename ScalarBinaryOpTraits::ReturnType ResScalar; - -enum { - Vectorizable = packet_traits::Vectorizable && packet_traits::Vectorizable - && int(packet_traits::size)==int(packet_traits::size), - LhsPacketSize = Vectorizable ? packet_traits::size : 1, - RhsPacketSize = Vectorizable ? packet_traits::size : 1, - ResPacketSize = Vectorizable ? packet_traits::size : 1 -}; - -typedef typename packet_traits::type _LhsPacket; -typedef typename packet_traits::type _RhsPacket; -typedef typename packet_traits::type _ResPacket; - -typedef typename conditional::type LhsPacket; -typedef typename conditional::type RhsPacket; -typedef typename conditional::type ResPacket; - -EIGEN_DONT_INLINE static void run( - Index rows, Index cols, - const LhsMapper& lhs, - const RhsMapper& rhs, - ResScalar* res, Index resIncr, - RhsScalar alpha); -}; - -template -EIGEN_DONT_INLINE void general_matrix_vector_product::run( - Index rows, Index cols, - const LhsMapper& lhs, - const RhsMapper& rhs, - ResScalar* res, Index resIncr, - RhsScalar alpha) -{ - EIGEN_UNUSED_VARIABLE(resIncr); - eigen_internal_assert(resIncr==1); - #ifdef _EIGEN_ACCUMULATE_PACKETS - #error _EIGEN_ACCUMULATE_PACKETS has already been defined - #endif - #define _EIGEN_ACCUMULATE_PACKETS(Alignment0,Alignment13,Alignment2) \ - pstore(&res[j], \ - padd(pload(&res[j]), \ - padd( \ - padd(pcj.pmul(lhs0.template load(j), ptmp0), \ - pcj.pmul(lhs1.template load(j), ptmp1)), \ - padd(pcj.pmul(lhs2.template load(j), ptmp2), \ - pcj.pmul(lhs3.template load(j), ptmp3)) ))) - - typedef typename LhsMapper::VectorMapper LhsScalars; - - conj_helper cj; - conj_helper pcj; - if(ConjugateRhs) - alpha = numext::conj(alpha); - - enum { AllAligned = 0, EvenAligned, FirstAligned, NoneAligned }; - const Index columnsAtOnce = 4; - const Index peels = 2; - const Index LhsPacketAlignedMask = LhsPacketSize-1; - const Index ResPacketAlignedMask = ResPacketSize-1; -// const Index PeelAlignedMask = ResPacketSize*peels-1; - const Index size = rows; - - const Index lhsStride = lhs.stride(); - - // How many coeffs of the result do we have to skip to be aligned. - // Here we assume data are at least aligned on the base scalar type. - Index alignedStart = internal::first_default_aligned(res,size); - Index alignedSize = ResPacketSize>1 ? alignedStart + ((size-alignedStart) & ~ResPacketAlignedMask) : 0; - const Index peeledSize = alignedSize - RhsPacketSize*peels - RhsPacketSize + 1; - - const Index alignmentStep = LhsPacketSize>1 ? (LhsPacketSize - lhsStride % LhsPacketSize) & LhsPacketAlignedMask : 0; - Index alignmentPattern = alignmentStep==0 ? AllAligned - : alignmentStep==(LhsPacketSize/2) ? EvenAligned - : FirstAligned; - - // we cannot assume the first element is aligned because of sub-matrices - const Index lhsAlignmentOffset = lhs.firstAligned(size); - - // find how many columns do we have to skip to be aligned with the result (if possible) - Index skipColumns = 0; - // if the data cannot be aligned (TODO add some compile time tests when possible, e.g. for floats) - if( (lhsAlignmentOffset < 0) || (lhsAlignmentOffset == size) || (UIntPtr(res)%sizeof(ResScalar)) ) + /* Optimized col-major matrix * vector product: + * This algorithm processes 4 columns at onces that allows to both reduce + * the number of load/stores of the result by a factor 4 and to reduce + * the instruction dependency. Moreover, we know that all bands have the + * same alignment pattern. + * + * Mixing type logic: C += alpha * A * B + * | A | B |alpha| comments + * |real |cplx |cplx | no vectorization + * |real |cplx |real | alpha is converted to a cplx when calling the run function, no vectorization + * |cplx |real |cplx | invalid, the caller has to do tmp: = A * B; C += alpha*tmp + * |cplx |real |real | optimal case, vectorization possible via real-cplx mul + * + * Accesses to the matrix coefficients follow the following logic: + * + * - if all columns have the same alignment then + * - if the columns have the same alignment as the result vector, then easy! (-> AllAligned case) + * - otherwise perform unaligned loads only (-> NoneAligned case) + * - otherwise + * - if even columns have the same alignment then + * // odd columns are guaranteed to have the same alignment too + * - if even or odd columns have the same alignment as the result, then + * // for a register size of 2 scalars, this is guarantee to be the case (e.g., SSE with double) + * - perform half aligned and half unaligned loads (-> EvenAligned case) + * - otherwise perform unaligned loads only (-> NoneAligned case) + * - otherwise, if the register size is 4 scalars (e.g., SSE with float) then + * - one over 4 consecutive columns is guaranteed to be aligned with the result vector, + * perform simple aligned loads for this column and aligned loads plus re-alignment for the other. (-> + * FirstAligned case) + * // this re-alignment is done by the palign function implemented for SSE in + * Eigen/src/Core/arch/SSE/PacketMath.h + * - otherwise, + * // if we get here, this means the register size is greater than 4 (e.g., AVX with floats), + * // we currently fall back to the NoneAligned case + * + * The same reasoning apply for the transposed case. + * + * The last case (PacketSize>4) could probably be improved by generalizing the FirstAligned case, but since we do not + * support AVX yet... One might also wonder why in the EvenAligned case we perform unaligned loads instead of using + * the aligned-loads plus re-alignment strategy as in the FirstAligned case. The reason is that we observed that + * unaligned loads on a 8 byte boundary are not too slow compared to unaligned loads on a 4 byte boundary. + * + */ + template + struct general_matrix_vector_product { - alignedSize = 0; - alignedStart = 0; - alignmentPattern = NoneAligned; - } - else if(LhsPacketSize > 4) - { - // TODO: extend the code to support aligned loads whenever possible when LhsPacketSize > 4. - // Currently, it seems to be better to perform unaligned loads anyway - alignmentPattern = NoneAligned; - } - else if (LhsPacketSize>1) + typedef typename ScalarBinaryOpTraits::ReturnType ResScalar; + + enum { + Vectorizable = packet_traits::Vectorizable && packet_traits::Vectorizable + && int(packet_traits::size) == int(packet_traits::size), + LhsPacketSize = Vectorizable ? packet_traits::size : 1, + RhsPacketSize = Vectorizable ? packet_traits::size : 1, + ResPacketSize = Vectorizable ? packet_traits::size : 1 + }; + + typedef typename packet_traits::type _LhsPacket; + typedef typename packet_traits::type _RhsPacket; + typedef typename packet_traits::type _ResPacket; + + typedef typename conditional::type LhsPacket; + typedef typename conditional::type RhsPacket; + typedef typename conditional::type ResPacket; + + EIGEN_DONT_INLINE static void run(Index rows, + Index cols, + const LhsMapper &lhs, + const RhsMapper &rhs, + ResScalar *res, + Index resIncr, + RhsScalar alpha); + }; + + template + EIGEN_DONT_INLINE void general_matrix_vector_product::run(Index rows, + Index cols, + const LhsMapper &lhs, + const RhsMapper &rhs, + ResScalar *res, + Index resIncr, + RhsScalar alpha) { - // eigen_internal_assert(size_t(firstLhs+lhsAlignmentOffset)%sizeof(LhsPacket)==0 || size(&res[j]), \ + padd(padd(pcj.pmul(lhs0.template load(j), ptmp0), \ + pcj.pmul(lhs1.template load(j), ptmp1)), \ + padd(pcj.pmul(lhs2.template load(j), ptmp2), \ + pcj.pmul(lhs3.template load(j), ptmp3))))) + + typedef typename LhsMapper::VectorMapper LhsScalars; + + conj_helper cj; + conj_helper pcj; + if (ConjugateRhs) alpha = numext::conj(alpha); + + enum { AllAligned = 0, EvenAligned, FirstAligned, NoneAligned }; + const Index columnsAtOnce = 4; + const Index peels = 2; + const Index LhsPacketAlignedMask = LhsPacketSize - 1; + const Index ResPacketAlignedMask = ResPacketSize - 1; + // const Index PeelAlignedMask = ResPacketSize*peels-1; + const Index size = rows; + + const Index lhsStride = lhs.stride(); + + // How many coeffs of the result do we have to skip to be aligned. + // Here we assume data are at least aligned on the base scalar type. + Index alignedStart = internal::first_default_aligned(res, size); + Index alignedSize = ResPacketSize > 1 ? alignedStart + ((size - alignedStart) & ~ResPacketAlignedMask) : 0; + const Index peeledSize = alignedSize - RhsPacketSize * peels - RhsPacketSize + 1; + + const Index alignmentStep = + LhsPacketSize > 1 ? (LhsPacketSize - lhsStride % LhsPacketSize) & LhsPacketAlignedMask : 0; + Index alignmentPattern = alignmentStep == 0 ? AllAligned + : alignmentStep == (LhsPacketSize / 2) ? EvenAligned + : FirstAligned; + + // we cannot assume the first element is aligned because of sub-matrices + const Index lhsAlignmentOffset = lhs.firstAligned(size); + + // find how many columns do we have to skip to be aligned with the result (if possible) + Index skipColumns = 0; + // if the data cannot be aligned (TODO add some compile time tests when possible, e.g. for floats) + if ((lhsAlignmentOffset < 0) || (lhsAlignmentOffset == size) || (UIntPtr(res) % sizeof(ResScalar))) { + alignedSize = 0; + alignedStart = 0; alignmentPattern = NoneAligned; - skipColumns = 0; - } - else - { - skipColumns = (std::min)(skipColumns,cols); - // note that the skiped columns are processed later. + } else if (LhsPacketSize > 4) { + // TODO: extend the code to support aligned loads whenever possible when LhsPacketSize > 4. + // Currently, it seems to be better to perform unaligned loads anyway + alignmentPattern = NoneAligned; + } else if (LhsPacketSize > 1) { + // eigen_internal_assert(size_t(firstLhs+lhsAlignmentOffset)%sizeof(LhsPacket)==0 || size= cols) + || LhsPacketSize > size + || (size_t(firstLhs+alignedStart+lhsStride*skipColumns)%sizeof(LhsPacket))==0);*/ + } else if (Vectorizable) { + alignedStart = 0; + alignedSize = size; + alignmentPattern = AllAligned; } - /* eigen_internal_assert( (alignmentPattern==NoneAligned) - || (skipColumns + columnsAtOnce >= cols) - || LhsPacketSize > size - || (size_t(firstLhs+alignedStart+lhsStride*skipColumns)%sizeof(LhsPacket))==0);*/ - } - else if(Vectorizable) - { - alignedStart = 0; - alignedSize = size; - alignmentPattern = AllAligned; - } + const Index offset1 = (alignmentPattern == FirstAligned && alignmentStep == 1) ? 3 : 1; + const Index offset3 = (alignmentPattern == FirstAligned && alignmentStep == 1) ? 1 : 3; - const Index offset1 = (alignmentPattern==FirstAligned && alignmentStep==1)?3:1; - const Index offset3 = (alignmentPattern==FirstAligned && alignmentStep==1)?1:3; + Index columnBound = ((cols - skipColumns) / columnsAtOnce) * columnsAtOnce + skipColumns; + for (Index i = skipColumns; i < columnBound; i += columnsAtOnce) { + RhsPacket ptmp0 = pset1(alpha * rhs(i, 0)), ptmp1 = pset1(alpha * rhs(i + offset1, 0)), + ptmp2 = pset1(alpha * rhs(i + 2, 0)), ptmp3 = pset1(alpha * rhs(i + offset3, 0)); - Index columnBound = ((cols-skipColumns)/columnsAtOnce)*columnsAtOnce + skipColumns; - for (Index i=skipColumns; i(alpha*rhs(i, 0)), - ptmp1 = pset1(alpha*rhs(i+offset1, 0)), - ptmp2 = pset1(alpha*rhs(i+2, 0)), - ptmp3 = pset1(alpha*rhs(i+offset3, 0)); - - // this helps a lot generating better binary code - const LhsScalars lhs0 = lhs.getVectorMapper(0, i+0), lhs1 = lhs.getVectorMapper(0, i+offset1), - lhs2 = lhs.getVectorMapper(0, i+2), lhs3 = lhs.getVectorMapper(0, i+offset3); - - if (Vectorizable) - { - /* explicit vectorization */ - // process initial unaligned coeffs - for (Index j=0; jalignedStart) - { - switch(alignmentPattern) - { + if (Vectorizable) { + /* explicit vectorization */ + // process initial unaligned coeffs + for (Index j = 0; j < alignedStart; ++j) { + res[j] = cj.pmadd(lhs0(j), pfirst(ptmp0), res[j]); + res[j] = cj.pmadd(lhs1(j), pfirst(ptmp1), res[j]); + res[j] = cj.pmadd(lhs2(j), pfirst(ptmp2), res[j]); + res[j] = cj.pmadd(lhs3(j), pfirst(ptmp3), res[j]); + } + + if (alignedSize > alignedStart) { + switch (alignmentPattern) { case AllAligned: - for (Index j = alignedStart; j1) - { + if (peels > 1) { LhsPacket A00, A01, A02, A03, A10, A11, A12, A13; ResPacket T0, T1; - A01 = lhs1.template load(alignedStart-1); - A02 = lhs2.template load(alignedStart-2); - A03 = lhs3.template load(alignedStart-3); + A01 = lhs1.template load(alignedStart - 1); + A02 = lhs2.template load(alignedStart - 2); + A03 = lhs3.template load(alignedStart - 3); - for (; j(j-1+LhsPacketSize); palign<1>(A01,A11); - A12 = lhs2.template load(j-2+LhsPacketSize); palign<2>(A02,A12); - A13 = lhs3.template load(j-3+LhsPacketSize); palign<3>(A03,A13); + for (; j < peeledSize; j += peels * ResPacketSize) { + A11 = lhs1.template load(j - 1 + LhsPacketSize); + palign<1>(A01, A11); + A12 = lhs2.template load(j - 2 + LhsPacketSize); + palign<2>(A02, A12); + A13 = lhs3.template load(j - 3 + LhsPacketSize); + palign<3>(A03, A13); A00 = lhs0.template load(j); - A10 = lhs0.template load(j+LhsPacketSize); - T0 = pcj.pmadd(A00, ptmp0, pload(&res[j])); - T1 = pcj.pmadd(A10, ptmp0, pload(&res[j+ResPacketSize])); - - T0 = pcj.pmadd(A01, ptmp1, T0); - A01 = lhs1.template load(j-1+2*LhsPacketSize); palign<1>(A11,A01); - T0 = pcj.pmadd(A02, ptmp2, T0); - A02 = lhs2.template load(j-2+2*LhsPacketSize); palign<2>(A12,A02); - T0 = pcj.pmadd(A03, ptmp3, T0); - pstore(&res[j],T0); - A03 = lhs3.template load(j-3+2*LhsPacketSize); palign<3>(A13,A03); - T1 = pcj.pmadd(A11, ptmp1, T1); - T1 = pcj.pmadd(A12, ptmp2, T1); - T1 = pcj.pmadd(A13, ptmp3, T1); - pstore(&res[j+ResPacketSize],T1); + A10 = lhs0.template load(j + LhsPacketSize); + T0 = pcj.pmadd(A00, ptmp0, pload(&res[j])); + T1 = pcj.pmadd(A10, ptmp0, pload(&res[j + ResPacketSize])); + + T0 = pcj.pmadd(A01, ptmp1, T0); + A01 = lhs1.template load(j - 1 + 2 * LhsPacketSize); + palign<1>(A11, A01); + T0 = pcj.pmadd(A02, ptmp2, T0); + A02 = lhs2.template load(j - 2 + 2 * LhsPacketSize); + palign<2>(A12, A02); + T0 = pcj.pmadd(A03, ptmp3, T0); + pstore(&res[j], T0); + A03 = lhs3.template load(j - 3 + 2 * LhsPacketSize); + palign<3>(A13, A03); + T1 = pcj.pmadd(A11, ptmp1, T1); + T1 = pcj.pmadd(A12, ptmp2, T1); + T1 = pcj.pmadd(A13, ptmp3, T1); + pstore(&res[j + ResPacketSize], T1); } } - for (; j(alpha*rhs(k, 0)); - const LhsScalars lhs0 = lhs.getVectorMapper(0, k); + // process remaining first and last columns (at most columnsAtOnce-1) + Index end = cols; + Index start = columnBound; + do { + for (Index k = start; k < end; ++k) { + RhsPacket ptmp0 = pset1(alpha * rhs(k, 0)); + const LhsScalars lhs0 = lhs.getVectorMapper(0, k); + + if (Vectorizable) { + /* explicit vectorization */ + // process first unaligned result's coeffs + for (Index j = 0; j < alignedStart; ++j) res[j] += cj.pmul(lhs0(j), pfirst(ptmp0)); + // process aligned result's coeffs + if (lhs0.template aligned(alignedStart)) + for (Index i = alignedStart; i < alignedSize; i += ResPacketSize) + pstore(&res[i], pcj.pmadd(lhs0.template load(i), ptmp0, pload(&res[i]))); + else + for (Index i = alignedStart; i < alignedSize; i += ResPacketSize) + pstore(&res[i], pcj.pmadd(lhs0.template load(i), ptmp0, pload(&res[i]))); + } - if (Vectorizable) - { - /* explicit vectorization */ - // process first unaligned result's coeffs - for (Index j=0; j(alignedStart)) - for (Index i = alignedStart;i(i), ptmp0, pload(&res[i]))); - else - for (Index i = alignedStart;i(i), ptmp0, pload(&res[i]))); + // process remaining scalars (or all if no explicit vectorization) + for (Index i = alignedSize; i < size; ++i) res[i] += cj.pmul(lhs0(i), pfirst(ptmp0)); } + if (skipColumns) { + start = 0; + end = skipColumns; + skipColumns = 0; + } else + break; + } while (Vectorizable); +#undef _EIGEN_ACCUMULATE_PACKETS + } - // process remaining scalars (or all if no explicit vectorization) - for (Index i=alignedSize; i -struct general_matrix_vector_product -{ -typedef typename ScalarBinaryOpTraits::ReturnType ResScalar; - -enum { - Vectorizable = packet_traits::Vectorizable && packet_traits::Vectorizable - && int(packet_traits::size)==int(packet_traits::size), - LhsPacketSize = Vectorizable ? packet_traits::size : 1, - RhsPacketSize = Vectorizable ? packet_traits::size : 1, - ResPacketSize = Vectorizable ? packet_traits::size : 1 -}; - -typedef typename packet_traits::type _LhsPacket; -typedef typename packet_traits::type _RhsPacket; -typedef typename packet_traits::type _ResPacket; - -typedef typename conditional::type LhsPacket; -typedef typename conditional::type RhsPacket; -typedef typename conditional::type ResPacket; - -EIGEN_DONT_INLINE static void run( - Index rows, Index cols, - const LhsMapper& lhs, - const RhsMapper& rhs, - ResScalar* res, Index resIncr, - ResScalar alpha); -}; - -template -EIGEN_DONT_INLINE void general_matrix_vector_product::run( - Index rows, Index cols, - const LhsMapper& lhs, - const RhsMapper& rhs, - ResScalar* res, Index resIncr, - ResScalar alpha) -{ - eigen_internal_assert(rhs.stride()==1); - - #ifdef _EIGEN_ACCUMULATE_PACKETS - #error _EIGEN_ACCUMULATE_PACKETS has already been defined - #endif - - #define _EIGEN_ACCUMULATE_PACKETS(Alignment0,Alignment13,Alignment2) {\ - RhsPacket b = rhs.getVectorMapper(j, 0).template load(0); \ - ptmp0 = pcj.pmadd(lhs0.template load(j), b, ptmp0); \ - ptmp1 = pcj.pmadd(lhs1.template load(j), b, ptmp1); \ - ptmp2 = pcj.pmadd(lhs2.template load(j), b, ptmp2); \ - ptmp3 = pcj.pmadd(lhs3.template load(j), b, ptmp3); } - - conj_helper cj; - conj_helper pcj; - - typedef typename LhsMapper::VectorMapper LhsScalars; - - enum { AllAligned=0, EvenAligned=1, FirstAligned=2, NoneAligned=3 }; - const Index rowsAtOnce = 4; - const Index peels = 2; - const Index RhsPacketAlignedMask = RhsPacketSize-1; - const Index LhsPacketAlignedMask = LhsPacketSize-1; - const Index depth = cols; - const Index lhsStride = lhs.stride(); - - // How many coeffs of the result do we have to skip to be aligned. - // Here we assume data are at least aligned on the base scalar type - // if that's not the case then vectorization is discarded, see below. - Index alignedStart = rhs.firstAligned(depth); - Index alignedSize = RhsPacketSize>1 ? alignedStart + ((depth-alignedStart) & ~RhsPacketAlignedMask) : 0; - const Index peeledSize = alignedSize - RhsPacketSize*peels - RhsPacketSize + 1; - - const Index alignmentStep = LhsPacketSize>1 ? (LhsPacketSize - lhsStride % LhsPacketSize) & LhsPacketAlignedMask : 0; - Index alignmentPattern = alignmentStep==0 ? AllAligned - : alignmentStep==(LhsPacketSize/2) ? EvenAligned - : FirstAligned; - - // we cannot assume the first element is aligned because of sub-matrices - const Index lhsAlignmentOffset = lhs.firstAligned(depth); - const Index rhsAlignmentOffset = rhs.firstAligned(rows); - - // find how many rows do we have to skip to be aligned with rhs (if possible) - Index skipRows = 0; - // if the data cannot be aligned (TODO add some compile time tests when possible, e.g. for floats) - if( (sizeof(LhsScalar)!=sizeof(RhsScalar)) || - (lhsAlignmentOffset < 0) || (lhsAlignmentOffset == depth) || - (rhsAlignmentOffset < 0) || (rhsAlignmentOffset == rows) ) + /* Optimized row-major matrix * vector product: + * This algorithm processes 4 rows at onces that allows to both reduce + * the number of load/stores of the result by a factor 4 and to reduce + * the instruction dependency. Moreover, we know that all bands have the + * same alignment pattern. + * + * Mixing type logic: + * - alpha is always a complex (or converted to a complex) + * - no vectorization + */ + template + struct general_matrix_vector_product { - alignedSize = 0; - alignedStart = 0; - alignmentPattern = NoneAligned; - } - else if(LhsPacketSize > 4) + typedef typename ScalarBinaryOpTraits::ReturnType ResScalar; + + enum { + Vectorizable = packet_traits::Vectorizable && packet_traits::Vectorizable + && int(packet_traits::size) == int(packet_traits::size), + LhsPacketSize = Vectorizable ? packet_traits::size : 1, + RhsPacketSize = Vectorizable ? packet_traits::size : 1, + ResPacketSize = Vectorizable ? packet_traits::size : 1 + }; + + typedef typename packet_traits::type _LhsPacket; + typedef typename packet_traits::type _RhsPacket; + typedef typename packet_traits::type _ResPacket; + + typedef typename conditional::type LhsPacket; + typedef typename conditional::type RhsPacket; + typedef typename conditional::type ResPacket; + + EIGEN_DONT_INLINE static void run(Index rows, + Index cols, + const LhsMapper &lhs, + const RhsMapper &rhs, + ResScalar *res, + Index resIncr, + ResScalar alpha); + }; + + template + EIGEN_DONT_INLINE void general_matrix_vector_product::run(Index rows, + Index cols, + const LhsMapper &lhs, + const RhsMapper &rhs, + ResScalar *res, + Index resIncr, + ResScalar alpha) { - // TODO: extend the code to support aligned loads whenever possible when LhsPacketSize > 4. - alignmentPattern = NoneAligned; + eigen_internal_assert(rhs.stride() == 1); + +#ifdef _EIGEN_ACCUMULATE_PACKETS +#error _EIGEN_ACCUMULATE_PACKETS has already been defined +#endif + +#define _EIGEN_ACCUMULATE_PACKETS(Alignment0, Alignment13, Alignment2) \ + { \ + RhsPacket b = rhs.getVectorMapper(j, 0).template load(0); \ + ptmp0 = pcj.pmadd(lhs0.template load(j), b, ptmp0); \ + ptmp1 = pcj.pmadd(lhs1.template load(j), b, ptmp1); \ + ptmp2 = pcj.pmadd(lhs2.template load(j), b, ptmp2); \ + ptmp3 = pcj.pmadd(lhs3.template load(j), b, ptmp3); \ } - else if (LhsPacketSize>1) - { - // eigen_internal_assert(size_t(firstLhs+lhsAlignmentOffset)%sizeof(LhsPacket)==0 || depth cj; + conj_helper pcj; + + typedef typename LhsMapper::VectorMapper LhsScalars; + + enum { AllAligned = 0, EvenAligned = 1, FirstAligned = 2, NoneAligned = 3 }; + const Index rowsAtOnce = 4; + const Index peels = 2; + const Index RhsPacketAlignedMask = RhsPacketSize - 1; + const Index LhsPacketAlignedMask = LhsPacketSize - 1; + const Index depth = cols; + const Index lhsStride = lhs.stride(); + + // How many coeffs of the result do we have to skip to be aligned. + // Here we assume data are at least aligned on the base scalar type + // if that's not the case then vectorization is discarded, see below. + Index alignedStart = rhs.firstAligned(depth); + Index alignedSize = RhsPacketSize > 1 ? alignedStart + ((depth - alignedStart) & ~RhsPacketAlignedMask) : 0; + const Index peeledSize = alignedSize - RhsPacketSize * peels - RhsPacketSize + 1; + + const Index alignmentStep = + LhsPacketSize > 1 ? (LhsPacketSize - lhsStride % LhsPacketSize) & LhsPacketAlignedMask : 0; + Index alignmentPattern = alignmentStep == 0 ? AllAligned + : alignmentStep == (LhsPacketSize / 2) ? EvenAligned + : FirstAligned; + + // we cannot assume the first element is aligned because of sub-matrices + const Index lhsAlignmentOffset = lhs.firstAligned(depth); + const Index rhsAlignmentOffset = rhs.firstAligned(rows); + + // find how many rows do we have to skip to be aligned with rhs (if possible) + Index skipRows = 0; + // if the data cannot be aligned (TODO add some compile time tests when possible, e.g. for floats) + if ((sizeof(LhsScalar) != sizeof(RhsScalar)) || (lhsAlignmentOffset < 0) || (lhsAlignmentOffset == depth) + || (rhsAlignmentOffset < 0) || (rhsAlignmentOffset == rows)) { + alignedSize = 0; + alignedStart = 0; alignmentPattern = NoneAligned; - skipRows = 0; - } - else - { - skipRows = (std::min)(skipRows,Index(rows)); - // note that the skiped columns are processed later. + } else if (LhsPacketSize > 4) { + // TODO: extend the code to support aligned loads whenever possible when LhsPacketSize > 4. + alignmentPattern = NoneAligned; + } else if (LhsPacketSize > 1) { + // eigen_internal_assert(size_t(firstLhs+lhsAlignmentOffset)%sizeof(LhsPacket)==0 || depth= rows) + || LhsPacketSize > depth + || (size_t(firstLhs+alignedStart+lhsStride*skipRows)%sizeof(LhsPacket))==0);*/ + } else if (Vectorizable) { + alignedStart = 0; + alignedSize = depth; + alignmentPattern = AllAligned; } - /* eigen_internal_assert( alignmentPattern==NoneAligned - || LhsPacketSize==1 - || (skipRows + rowsAtOnce >= rows) - || LhsPacketSize > depth - || (size_t(firstLhs+alignedStart+lhsStride*skipRows)%sizeof(LhsPacket))==0);*/ - } - else if(Vectorizable) - { - alignedStart = 0; - alignedSize = depth; - alignmentPattern = AllAligned; - } - - const Index offset1 = (alignmentPattern==FirstAligned && alignmentStep==1)?3:1; - const Index offset3 = (alignmentPattern==FirstAligned && alignmentStep==1)?1:3; - Index rowBound = ((rows-skipRows)/rowsAtOnce)*rowsAtOnce + skipRows; - for (Index i=skipRows; i(ResScalar(0)), ptmp1 = pset1(ResScalar(0)), - ptmp2 = pset1(ResScalar(0)), ptmp3 = pset1(ResScalar(0)); + // this helps the compiler generating good binary code + const LhsScalars lhs0 = lhs.getVectorMapper(i + 0, 0), lhs1 = lhs.getVectorMapper(i + offset1, 0), + lhs2 = lhs.getVectorMapper(i + 2, 0), lhs3 = lhs.getVectorMapper(i + offset3, 0); - // process initial unaligned coeffs - // FIXME this loop get vectorized by the compiler ! - for (Index j=0; j(ResScalar(0)), ptmp1 = pset1(ResScalar(0)), + ptmp2 = pset1(ResScalar(0)), ptmp3 = pset1(ResScalar(0)); + + // process initial unaligned coeffs + // FIXME this loop get vectorized by the compiler ! + for (Index j = 0; j < alignedStart; ++j) { + RhsScalar b = rhs(j, 0); + tmp0 += cj.pmul(lhs0(j), b); + tmp1 += cj.pmul(lhs1(j), b); + tmp2 += cj.pmul(lhs2(j), b); + tmp3 += cj.pmul(lhs3(j), b); + } - if (alignedSize>alignedStart) - { - switch(alignmentPattern) - { + if (alignedSize > alignedStart) { + switch (alignmentPattern) { case AllAligned: - for (Index j = alignedStart; j1) - { + if (peels > 1) { /* Here we proccess 4 rows with with two peeled iterations to hide * the overhead of unaligned loads. Moreover unaligned loads are handled * using special shift/move operations between the two aligned packets @@ -510,110 +539,112 @@ EIGEN_DONT_INLINE void general_matrix_vector_product(alignedStart-1); - A02 = lhs2.template load(alignedStart-2); - A03 = lhs3.template load(alignedStart-3); + A01 = lhs1.template load(alignedStart - 1); + A02 = lhs2.template load(alignedStart - 2); + A03 = lhs3.template load(alignedStart - 3); - for (; j(0); - A11 = lhs1.template load(j-1+LhsPacketSize); palign<1>(A01,A11); - A12 = lhs2.template load(j-2+LhsPacketSize); palign<2>(A02,A12); - A13 = lhs3.template load(j-3+LhsPacketSize); palign<3>(A03,A13); + A11 = lhs1.template load(j - 1 + LhsPacketSize); + palign<1>(A01, A11); + A12 = lhs2.template load(j - 2 + LhsPacketSize); + palign<2>(A02, A12); + A13 = lhs3.template load(j - 3 + LhsPacketSize); + palign<3>(A03, A13); ptmp0 = pcj.pmadd(lhs0.template load(j), b, ptmp0); ptmp1 = pcj.pmadd(A01, b, ptmp1); - A01 = lhs1.template load(j-1+2*LhsPacketSize); palign<1>(A11,A01); + A01 = lhs1.template load(j - 1 + 2 * LhsPacketSize); + palign<1>(A11, A01); ptmp2 = pcj.pmadd(A02, b, ptmp2); - A02 = lhs2.template load(j-2+2*LhsPacketSize); palign<2>(A12,A02); + A02 = lhs2.template load(j - 2 + 2 * LhsPacketSize); + palign<2>(A12, A02); ptmp3 = pcj.pmadd(A03, b, ptmp3); - A03 = lhs3.template load(j-3+2*LhsPacketSize); palign<3>(A13,A03); + A03 = lhs3.template load(j - 3 + 2 * LhsPacketSize); + palign<3>(A13, A03); - b = rhs.getVectorMapper(j+RhsPacketSize, 0).template load(0); - ptmp0 = pcj.pmadd(lhs0.template load(j+LhsPacketSize), b, ptmp0); + b = rhs.getVectorMapper(j + RhsPacketSize, 0).template load(0); + ptmp0 = pcj.pmadd(lhs0.template load(j + LhsPacketSize), b, ptmp0); ptmp1 = pcj.pmadd(A11, b, ptmp1); ptmp2 = pcj.pmadd(A12, b, ptmp2); ptmp3 = pcj.pmadd(A13, b, ptmp3); } } - for (; j(tmp0); - const LhsScalars lhs0 = lhs.getVectorMapper(i, 0); - // process first unaligned result's coeffs + // process remaining coeffs (or all if no explicit vectorization) // FIXME this loop get vectorized by the compiler ! - for (Index j=0; jalignedStart) - { - // process aligned rhs coeffs - if (lhs0.template aligned(alignedStart)) - for (Index j = alignedStart;j(j), rhs.getVectorMapper(j, 0).template load(0), ptmp0); - else - for (Index j = alignedStart;j(j), rhs.getVectorMapper(j, 0).template load(0), ptmp0); - tmp0 += predux(ptmp0); + for (Index j = alignedSize; j < depth; ++j) { + RhsScalar b = rhs(j, 0); + tmp0 += cj.pmul(lhs0(j), b); + tmp1 += cj.pmul(lhs1(j), b); + tmp2 += cj.pmul(lhs2(j), b); + tmp3 += cj.pmul(lhs3(j), b); } - - // process remaining scalars - // FIXME this loop get vectorized by the compiler ! - for (Index j=alignedSize; j(tmp0); + const LhsScalars lhs0 = lhs.getVectorMapper(i, 0); + // process first unaligned result's coeffs + // FIXME this loop get vectorized by the compiler ! + for (Index j = 0; j < alignedStart; ++j) tmp0 += cj.pmul(lhs0(j), rhs(j, 0)); + + if (alignedSize > alignedStart) { + // process aligned rhs coeffs + if (lhs0.template aligned(alignedStart)) + for (Index j = alignedStart; j < alignedSize; j += RhsPacketSize) + ptmp0 = pcj.pmadd(lhs0.template load(j), + rhs.getVectorMapper(j, 0).template load(0), + ptmp0); + else + for (Index j = alignedStart; j < alignedSize; j += RhsPacketSize) + ptmp0 = pcj.pmadd(lhs0.template load(j), + rhs.getVectorMapper(j, 0).template load(0), + ptmp0); + tmp0 += predux(ptmp0); + } + + // process remaining scalars + // FIXME this loop get vectorized by the compiler ! + for (Index j = alignedSize; j < depth; ++j) tmp0 += cj.pmul(lhs0(j), rhs(j, 0)); + res[i * resIncr] += alpha * tmp0; + } + if (skipRows) { + start = 0; + end = skipRows; + skipRows = 0; + } else + break; + } while (Vectorizable); + +#undef _EIGEN_ACCUMULATE_PACKETS + } -} // end namespace internal +}// end namespace internal -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_GENERAL_MATRIX_VECTOR_H +#endif// EIGEN_GENERAL_MATRIX_VECTOR_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/products/GeneralMatrixVector_BLAS.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/products/GeneralMatrixVector_BLAS.h index 6e36c2b3..088a3baf 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/products/GeneralMatrixVector_BLAS.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/products/GeneralMatrixVector_BLAS.h @@ -33,104 +33,154 @@ #ifndef EIGEN_GENERAL_MATRIX_VECTOR_BLAS_H #define EIGEN_GENERAL_MATRIX_VECTOR_BLAS_H -namespace Eigen { +namespace Eigen { namespace internal { -/********************************************************************** -* This file implements general matrix-vector multiplication using BLAS -* gemv function via partial specialization of -* general_matrix_vector_product::run(..) method for float, double, -* std::complex and std::complex types -**********************************************************************/ - -// gemv specialization - -template -struct general_matrix_vector_product_gemv; - -#define EIGEN_BLAS_GEMV_SPECIALIZE(Scalar) \ -template \ -struct general_matrix_vector_product,ColMajor,ConjugateLhs,Scalar,const_blas_data_mapper,ConjugateRhs,Specialized> { \ -static void run( \ - Index rows, Index cols, \ - const const_blas_data_mapper &lhs, \ - const const_blas_data_mapper &rhs, \ - Scalar* res, Index resIncr, Scalar alpha) \ -{ \ - if (ConjugateLhs) { \ - general_matrix_vector_product,ColMajor,ConjugateLhs,Scalar,const_blas_data_mapper,ConjugateRhs,BuiltIn>::run( \ - rows, cols, lhs, rhs, res, resIncr, alpha); \ - } else { \ - general_matrix_vector_product_gemv::run( \ - rows, cols, lhs.data(), lhs.stride(), rhs.data(), rhs.stride(), res, resIncr, alpha); \ - } \ -} \ -}; \ -template \ -struct general_matrix_vector_product,RowMajor,ConjugateLhs,Scalar,const_blas_data_mapper,ConjugateRhs,Specialized> { \ -static void run( \ - Index rows, Index cols, \ - const const_blas_data_mapper &lhs, \ - const const_blas_data_mapper &rhs, \ - Scalar* res, Index resIncr, Scalar alpha) \ -{ \ - general_matrix_vector_product_gemv::run( \ - rows, cols, lhs.data(), lhs.stride(), rhs.data(), rhs.stride(), res, resIncr, alpha); \ -} \ -}; \ - -EIGEN_BLAS_GEMV_SPECIALIZE(double) -EIGEN_BLAS_GEMV_SPECIALIZE(float) -EIGEN_BLAS_GEMV_SPECIALIZE(dcomplex) -EIGEN_BLAS_GEMV_SPECIALIZE(scomplex) - -#define EIGEN_BLAS_GEMV_SPECIALIZATION(EIGTYPE,BLASTYPE,BLASFUNC) \ -template \ -struct general_matrix_vector_product_gemv \ -{ \ -typedef Matrix GEMVVector;\ -\ -static void run( \ - Index rows, Index cols, \ - const EIGTYPE* lhs, Index lhsStride, \ - const EIGTYPE* rhs, Index rhsIncr, \ - EIGTYPE* res, Index resIncr, EIGTYPE alpha) \ -{ \ - BlasIndex m=convert_index(rows), n=convert_index(cols), \ - lda=convert_index(lhsStride), incx=convert_index(rhsIncr), incy=convert_index(resIncr); \ - const EIGTYPE beta(1); \ - const EIGTYPE *x_ptr; \ - char trans=(LhsStorageOrder==ColMajor) ? 'N' : (ConjugateLhs) ? 'C' : 'T'; \ - if (LhsStorageOrder==RowMajor) { \ - m = convert_index(cols); \ - n = convert_index(rows); \ - }\ - GEMVVector x_tmp; \ - if (ConjugateRhs) { \ - Map > map_x(rhs,cols,1,InnerStride<>(incx)); \ - x_tmp=map_x.conjugate(); \ - x_ptr=x_tmp.data(); \ - incx=1; \ - } else x_ptr=rhs; \ - BLASFUNC(&trans, &m, &n, (const BLASTYPE*)&numext::real_ref(alpha), (const BLASTYPE*)lhs, &lda, (const BLASTYPE*)x_ptr, &incx, (const BLASTYPE*)&numext::real_ref(beta), (BLASTYPE*)res, &incy); \ -}\ -}; + /********************************************************************** + * This file implements general matrix-vector multiplication using BLAS + * gemv function via partial specialization of + * general_matrix_vector_product::run(..) method for float, double, + * std::complex and std::complex types + **********************************************************************/ + + // gemv specialization + + template + struct general_matrix_vector_product_gemv; + +#define EIGEN_BLAS_GEMV_SPECIALIZE(Scalar) \ + template \ + struct general_matrix_vector_product, \ + ColMajor, \ + ConjugateLhs, \ + Scalar, \ + const_blas_data_mapper, \ + ConjugateRhs, \ + Specialized> \ + { \ + static void run(Index rows, \ + Index cols, \ + const const_blas_data_mapper &lhs, \ + const const_blas_data_mapper &rhs, \ + Scalar *res, \ + Index resIncr, \ + Scalar alpha) \ + { \ + if (ConjugateLhs) { \ + general_matrix_vector_product, \ + ColMajor, \ + ConjugateLhs, \ + Scalar, \ + const_blas_data_mapper, \ + ConjugateRhs, \ + BuiltIn>::run(rows, cols, lhs, rhs, res, resIncr, alpha); \ + } else { \ + general_matrix_vector_product_gemv::run( \ + rows, cols, lhs.data(), lhs.stride(), rhs.data(), rhs.stride(), res, resIncr, alpha); \ + } \ + } \ + }; \ + template \ + struct general_matrix_vector_product, \ + RowMajor, \ + ConjugateLhs, \ + Scalar, \ + const_blas_data_mapper, \ + ConjugateRhs, \ + Specialized> \ + { \ + static void run(Index rows, \ + Index cols, \ + const const_blas_data_mapper &lhs, \ + const const_blas_data_mapper &rhs, \ + Scalar *res, \ + Index resIncr, \ + Scalar alpha) \ + { \ + general_matrix_vector_product_gemv::run( \ + rows, cols, lhs.data(), lhs.stride(), rhs.data(), rhs.stride(), res, resIncr, alpha); \ + } \ + }; + + EIGEN_BLAS_GEMV_SPECIALIZE(double) + EIGEN_BLAS_GEMV_SPECIALIZE(float) + EIGEN_BLAS_GEMV_SPECIALIZE(dcomplex) + EIGEN_BLAS_GEMV_SPECIALIZE(scomplex) + +#define EIGEN_BLAS_GEMV_SPECIALIZATION(EIGTYPE, BLASTYPE, BLASFUNC) \ + template \ + struct general_matrix_vector_product_gemv \ + { \ + typedef Matrix GEMVVector; \ + \ + static void run(Index rows, \ + Index cols, \ + const EIGTYPE *lhs, \ + Index lhsStride, \ + const EIGTYPE *rhs, \ + Index rhsIncr, \ + EIGTYPE *res, \ + Index resIncr, \ + EIGTYPE alpha) \ + { \ + BlasIndex m = convert_index(rows), n = convert_index(cols), \ + lda = convert_index(lhsStride), incx = convert_index(rhsIncr), \ + incy = convert_index(resIncr); \ + const EIGTYPE beta(1); \ + const EIGTYPE *x_ptr; \ + char trans = (LhsStorageOrder == ColMajor) ? 'N' : (ConjugateLhs) ? 'C' : 'T'; \ + if (LhsStorageOrder == RowMajor) { \ + m = convert_index(cols); \ + n = convert_index(rows); \ + } \ + GEMVVector x_tmp; \ + if (ConjugateRhs) { \ + Map> map_x(rhs, cols, 1, InnerStride<>(incx)); \ + x_tmp = map_x.conjugate(); \ + x_ptr = x_tmp.data(); \ + incx = 1; \ + } else \ + x_ptr = rhs; \ + BLASFUNC(&trans, \ + &m, \ + &n, \ + (const BLASTYPE *)&numext::real_ref(alpha), \ + (const BLASTYPE *)lhs, \ + &lda, \ + (const BLASTYPE *)x_ptr, \ + &incx, \ + (const BLASTYPE *)&numext::real_ref(beta), \ + (BLASTYPE *)res, \ + &incy); \ + } \ + }; #ifdef EIGEN_USE_MKL -EIGEN_BLAS_GEMV_SPECIALIZATION(double, double, dgemv) -EIGEN_BLAS_GEMV_SPECIALIZATION(float, float, sgemv) -EIGEN_BLAS_GEMV_SPECIALIZATION(dcomplex, MKL_Complex16, zgemv) -EIGEN_BLAS_GEMV_SPECIALIZATION(scomplex, MKL_Complex8 , cgemv) + EIGEN_BLAS_GEMV_SPECIALIZATION(double, double, dgemv) + EIGEN_BLAS_GEMV_SPECIALIZATION(float, float, sgemv) + EIGEN_BLAS_GEMV_SPECIALIZATION(dcomplex, MKL_Complex16, zgemv) + EIGEN_BLAS_GEMV_SPECIALIZATION(scomplex, MKL_Complex8, cgemv) #else -EIGEN_BLAS_GEMV_SPECIALIZATION(double, double, dgemv_) -EIGEN_BLAS_GEMV_SPECIALIZATION(float, float, sgemv_) -EIGEN_BLAS_GEMV_SPECIALIZATION(dcomplex, double, zgemv_) -EIGEN_BLAS_GEMV_SPECIALIZATION(scomplex, float, cgemv_) + EIGEN_BLAS_GEMV_SPECIALIZATION(double, double, dgemv_) + EIGEN_BLAS_GEMV_SPECIALIZATION(float, float, sgemv_) + EIGEN_BLAS_GEMV_SPECIALIZATION(dcomplex, double, zgemv_) + EIGEN_BLAS_GEMV_SPECIALIZATION(scomplex, float, cgemv_) #endif -} // end namespase internal +}// namespace internal -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_GENERAL_MATRIX_VECTOR_BLAS_H +#endif// EIGEN_GENERAL_MATRIX_VECTOR_BLAS_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/products/Parallelizer.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/products/Parallelizer.h index c2f084c8..a49cd55a 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/products/Parallelizer.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/products/Parallelizer.h @@ -14,35 +14,30 @@ namespace Eigen { namespace internal { -/** \internal */ -inline void manage_multi_threading(Action action, int* v) -{ - static EIGEN_UNUSED int m_maxThreads = -1; - - if(action==SetAction) - { - eigen_internal_assert(v!=0); - m_maxThreads = *v; - } - else if(action==GetAction) - { - eigen_internal_assert(v!=0); - #ifdef EIGEN_HAS_OPENMP - if(m_maxThreads>0) - *v = m_maxThreads; - else - *v = omp_get_max_threads(); - #else - *v = 1; - #endif - } - else + /** \internal */ + inline void manage_multi_threading(Action action, int *v) { - eigen_internal_assert(false); + static EIGEN_UNUSED int m_maxThreads = -1; + + if (action == SetAction) { + eigen_internal_assert(v != 0); + m_maxThreads = *v; + } else if (action == GetAction) { + eigen_internal_assert(v != 0); +#ifdef EIGEN_HAS_OPENMP + if (m_maxThreads > 0) + *v = m_maxThreads; + else + *v = omp_get_max_threads(); +#else + *v = 1; +#endif + } else { + eigen_internal_assert(false); + } } -} -} +}// namespace internal /** Must be call first when calling Eigen from multiple threads */ inline void initParallel() @@ -54,7 +49,7 @@ inline void initParallel() } /** \returns the max number of threads reserved for Eigen - * \sa setNbThreads */ + * \sa setNbThreads */ inline int nbThreads() { int ret; @@ -63,101 +58,98 @@ inline int nbThreads() } /** Sets the max number of threads reserved for Eigen - * \sa nbThreads */ -inline void setNbThreads(int v) -{ - internal::manage_multi_threading(SetAction, &v); -} + * \sa nbThreads */ +inline void setNbThreads(int v) { internal::manage_multi_threading(SetAction, &v); } namespace internal { -template struct GemmParallelInfo -{ - GemmParallelInfo() : sync(-1), users(0), lhs_start(0), lhs_length(0) {} + template struct GemmParallelInfo + { + GemmParallelInfo() : sync(-1), users(0), lhs_start(0), lhs_length(0) {} - Index volatile sync; - int volatile users; + Index volatile sync; + int volatile users; - Index lhs_start; - Index lhs_length; -}; + Index lhs_start; + Index lhs_length; + }; -template -void parallelize_gemm(const Functor& func, Index rows, Index cols, Index depth, bool transpose) -{ - // TODO when EIGEN_USE_BLAS is defined, - // we should still enable OMP for other scalar types -#if !(defined (EIGEN_HAS_OPENMP)) || defined (EIGEN_USE_BLAS) - // FIXME the transpose variable is only needed to properly split - // the matrix product when multithreading is enabled. This is a temporary - // fix to support row-major destination matrices. This whole - // parallelizer mechanism has to be redisigned anyway. - EIGEN_UNUSED_VARIABLE(depth); - EIGEN_UNUSED_VARIABLE(transpose); - func(0,rows, 0,cols); + template + void parallelize_gemm(const Functor &func, Index rows, Index cols, Index depth, bool transpose) + { + // TODO when EIGEN_USE_BLAS is defined, + // we should still enable OMP for other scalar types +#if !(defined(EIGEN_HAS_OPENMP)) || defined(EIGEN_USE_BLAS) + // FIXME the transpose variable is only needed to properly split + // the matrix product when multithreading is enabled. This is a temporary + // fix to support row-major destination matrices. This whole + // parallelizer mechanism has to be redisigned anyway. + EIGEN_UNUSED_VARIABLE(depth); + EIGEN_UNUSED_VARIABLE(transpose); + func(0, rows, 0, cols); #else - // Dynamically check whether we should enable or disable OpenMP. - // The conditions are: - // - the max number of threads we can create is greater than 1 - // - we are not already in a parallel code - // - the sizes are large enough + // Dynamically check whether we should enable or disable OpenMP. + // The conditions are: + // - the max number of threads we can create is greater than 1 + // - we are not already in a parallel code + // - the sizes are large enough - // compute the maximal number of threads from the size of the product: - // This first heuristic takes into account that the product kernel is fully optimized when working with nr columns at once. - Index size = transpose ? rows : cols; - Index pb_max_threads = std::max(1,size / Functor::Traits::nr); + // compute the maximal number of threads from the size of the product: + // This first heuristic takes into account that the product kernel is fully optimized when working with nr columns + // at once. + Index size = transpose ? rows : cols; + Index pb_max_threads = std::max(1, size / Functor::Traits::nr); - // compute the maximal number of threads from the total amount of work: - double work = static_cast(rows) * static_cast(cols) * - static_cast(depth); - double kMinTaskSize = 50000; // FIXME improve this heuristic. - pb_max_threads = std::max(1, std::min(pb_max_threads, work / kMinTaskSize)); + // compute the maximal number of threads from the total amount of work: + double work = static_cast(rows) * static_cast(cols) * static_cast(depth); + double kMinTaskSize = 50000;// FIXME improve this heuristic. + pb_max_threads = std::max(1, std::min(pb_max_threads, work / kMinTaskSize)); - // compute the number of threads we are going to use - Index threads = std::min(nbThreads(), pb_max_threads); + // compute the number of threads we are going to use + Index threads = std::min(nbThreads(), pb_max_threads); - // if multi-threading is explicitely disabled, not useful, or if we already are in a parallel session, - // then abort multi-threading - // FIXME omp_get_num_threads()>1 only works for openmp, what if the user does not use openmp? - if((!Condition) || (threads==1) || (omp_get_num_threads()>1)) - return func(0,rows, 0,cols); + // if multi-threading is explicitely disabled, not useful, or if we already are in a parallel session, + // then abort multi-threading + // FIXME omp_get_num_threads()>1 only works for openmp, what if the user does not use openmp? + if ((!Condition) || (threads == 1) || (omp_get_num_threads() > 1)) return func(0, rows, 0, cols); - Eigen::initParallel(); - func.initParallelSession(threads); + Eigen::initParallel(); + func.initParallelSession(threads); - if(transpose) - std::swap(rows,cols); + if (transpose) std::swap(rows, cols); - ei_declare_aligned_stack_constructed_variable(GemmParallelInfo,info,threads,0); + ei_declare_aligned_stack_constructed_variable(GemmParallelInfo, info, threads, 0); - #pragma omp parallel num_threads(threads) - { - Index i = omp_get_thread_num(); - // Note that the actual number of threads might be lower than the number of request ones. - Index actual_threads = omp_get_num_threads(); +#pragma omp parallel num_threads(threads) + { + Index i = omp_get_thread_num(); + // Note that the actual number of threads might be lower than the number of request ones. + Index actual_threads = omp_get_num_threads(); - Index blockCols = (cols / actual_threads) & ~Index(0x3); - Index blockRows = (rows / actual_threads); - blockRows = (blockRows/Functor::Traits::mr)*Functor::Traits::mr; + Index blockCols = (cols / actual_threads) & ~Index(0x3); + Index blockRows = (rows / actual_threads); + blockRows = (blockRows / Functor::Traits::mr) * Functor::Traits::mr; - Index r0 = i*blockRows; - Index actualBlockRows = (i+1==actual_threads) ? rows-r0 : blockRows; + Index r0 = i * blockRows; + Index actualBlockRows = (i + 1 == actual_threads) ? rows - r0 : blockRows; - Index c0 = i*blockCols; - Index actualBlockCols = (i+1==actual_threads) ? cols-c0 : blockCols; + Index c0 = i * blockCols; + Index actualBlockCols = (i + 1 == actual_threads) ? cols - c0 : blockCols; - info[i].lhs_start = r0; - info[i].lhs_length = actualBlockRows; + info[i].lhs_start = r0; + info[i].lhs_length = actualBlockRows; - if(transpose) func(c0, actualBlockCols, 0, rows, info); - else func(0, rows, c0, actualBlockCols, info); - } + if (transpose) + func(c0, actualBlockCols, 0, rows, info); + else + func(0, rows, c0, actualBlockCols, info); + } #endif -} + } -} // end namespace internal +}// end namespace internal -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_PARALLELIZER_H +#endif// EIGEN_PARALLELIZER_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/products/SelfadjointMatrixMatrix.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/products/SelfadjointMatrixMatrix.h index da6f82ab..cc1867dd 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/products/SelfadjointMatrixMatrix.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/products/SelfadjointMatrixMatrix.h @@ -10,366 +10,391 @@ #ifndef EIGEN_SELFADJOINT_MATRIX_MATRIX_H #define EIGEN_SELFADJOINT_MATRIX_MATRIX_H -namespace Eigen { +namespace Eigen { namespace internal { -// pack a selfadjoint block diagonal for use with the gebp_kernel -template -struct symm_pack_lhs -{ - template inline - void pack(Scalar* blockA, const const_blas_data_mapper& lhs, Index cols, Index i, Index& count) + // pack a selfadjoint block diagonal for use with the gebp_kernel + template struct symm_pack_lhs { - // normal copy - for(Index k=0; k + inline void pack(Scalar *blockA, + const const_blas_data_mapper &lhs, + Index cols, + Index i, + Index &count) { - for(Index w=0; w::size }; - const_blas_data_mapper lhs(_lhs,lhsStride); - Index count = 0; - //Index peeled_mc3 = (rows/Pack1)*Pack1; - - const Index peeled_mc3 = Pack1>=3*PacketSize ? (rows/(3*PacketSize))*(3*PacketSize) : 0; - const Index peeled_mc2 = Pack1>=2*PacketSize ? peeled_mc3+((rows-peeled_mc3)/(2*PacketSize))*(2*PacketSize) : 0; - const Index peeled_mc1 = Pack1>=1*PacketSize ? (rows/(1*PacketSize))*(1*PacketSize) : 0; - - if(Pack1>=3*PacketSize) - for(Index i=0; i(blockA, lhs, cols, i, count); - - if(Pack1>=2*PacketSize) - for(Index i=peeled_mc3; i(blockA, lhs, cols, i, count); - - if(Pack1>=1*PacketSize) - for(Index i=peeled_mc2; i(blockA, lhs, cols, i, count); - - // do the same with mr==1 - for(Index i=peeled_mc1; i::size }; + const_blas_data_mapper lhs(_lhs, lhsStride); + Index count = 0; + // Index peeled_mc3 = (rows/Pack1)*Pack1; + + const Index peeled_mc3 = Pack1 >= 3 * PacketSize ? (rows / (3 * PacketSize)) * (3 * PacketSize) : 0; + const Index peeled_mc2 = + Pack1 >= 2 * PacketSize ? peeled_mc3 + ((rows - peeled_mc3) / (2 * PacketSize)) * (2 * PacketSize) : 0; + const Index peeled_mc1 = Pack1 >= 1 * PacketSize ? (rows / (1 * PacketSize)) * (1 * PacketSize) : 0; + + if (Pack1 >= 3 * PacketSize) + for (Index i = 0; i < peeled_mc3; i += 3 * PacketSize) pack<3 * PacketSize>(blockA, lhs, cols, i, count); + + if (Pack1 >= 2 * PacketSize) + for (Index i = peeled_mc3; i < peeled_mc2; i += 2 * PacketSize) + pack<2 * PacketSize>(blockA, lhs, cols, i, count); + + if (Pack1 >= 1 * PacketSize) + for (Index i = peeled_mc2; i < peeled_mc1; i += 1 * PacketSize) + pack<1 * PacketSize>(blockA, lhs, cols, i, count); + + // do the same with mr==1 + for (Index i = peeled_mc1; i < rows; i++) { + for (Index k = 0; k < i; k++) blockA[count++] = lhs(i, k);// normal - blockA[count++] = numext::real(lhs(i, i)); // real (diagonal) + blockA[count++] = numext::real(lhs(i, i));// real (diagonal) - for(Index k=i+1; k -struct symm_pack_rhs -{ - enum { PacketSize = packet_traits::size }; - void operator()(Scalar* blockB, const Scalar* _rhs, Index rhsStride, Index rows, Index cols, Index k2) + template struct symm_pack_rhs { - Index end_k = k2 + rows; - Index count = 0; - const_blas_data_mapper rhs(_rhs,rhsStride); - Index packet_cols8 = nr>=8 ? (cols/8) * 8 : 0; - Index packet_cols4 = nr>=4 ? (cols/4) * 4 : 0; - - // first part: normal case - for(Index j2=0; j2::size }; + void operator()(Scalar *blockB, const Scalar *_rhs, Index rhsStride, Index rows, Index cols, Index k2) { - for(Index k=k2; k=4) - { - blockB[count+2] = rhs(k,j2+2); - blockB[count+3] = rhs(k,j2+3); + Index end_k = k2 + rows; + Index count = 0; + const_blas_data_mapper rhs(_rhs, rhsStride); + Index packet_cols8 = nr >= 8 ? (cols / 8) * 8 : 0; + Index packet_cols4 = nr >= 4 ? (cols / 4) * 4 : 0; + + // first part: normal case + for (Index j2 = 0; j2 < k2; j2 += nr) { + for (Index k = k2; k < end_k; k++) { + blockB[count + 0] = rhs(k, j2 + 0); + blockB[count + 1] = rhs(k, j2 + 1); + if (nr >= 4) { + blockB[count + 2] = rhs(k, j2 + 2); + blockB[count + 3] = rhs(k, j2 + 3); + } + if (nr >= 8) { + blockB[count + 4] = rhs(k, j2 + 4); + blockB[count + 5] = rhs(k, j2 + 5); + blockB[count + 6] = rhs(k, j2 + 6); + blockB[count + 7] = rhs(k, j2 + 7); + } + count += nr; } - if (nr>=8) - { - blockB[count+4] = rhs(k,j2+4); - blockB[count+5] = rhs(k,j2+5); - blockB[count+6] = rhs(k,j2+6); - blockB[count+7] = rhs(k,j2+7); - } - count += nr; } - } - // second part: diagonal block - Index end8 = nr>=8 ? (std::min)(k2+rows,packet_cols8) : k2; - if(nr>=8) - { - for(Index j2=k2; j2= 8 ? (std::min)(k2 + rows, packet_cols8) : k2; + if (nr >= 8) { + for (Index j2 = k2; j2 < end8; j2 += 8) { + // again we can split vertically in three different parts (transpose, symmetric, normal) + // transpose + for (Index k = k2; k < j2; k++) { + blockB[count + 0] = numext::conj(rhs(j2 + 0, k)); + blockB[count + 1] = numext::conj(rhs(j2 + 1, k)); + blockB[count + 2] = numext::conj(rhs(j2 + 2, k)); + blockB[count + 3] = numext::conj(rhs(j2 + 3, k)); + blockB[count + 4] = numext::conj(rhs(j2 + 4, k)); + blockB[count + 5] = numext::conj(rhs(j2 + 5, k)); + blockB[count + 6] = numext::conj(rhs(j2 + 6, k)); + blockB[count + 7] = numext::conj(rhs(j2 + 7, k)); + count += 8; + } + // symmetric + Index h = 0; + for (Index k = j2; k < j2 + 8; k++) { + // normal + for (Index w = 0; w < h; ++w) blockB[count + w] = rhs(k, j2 + w); + + blockB[count + h] = numext::real(rhs(k, k)); + + // transpose + for (Index w = h + 1; w < 8; ++w) blockB[count + w] = numext::conj(rhs(j2 + w, k)); + count += 8; + ++h; + } + // normal + for (Index k = j2 + 8; k < end_k; k++) { + blockB[count + 0] = rhs(k, j2 + 0); + blockB[count + 1] = rhs(k, j2 + 1); + blockB[count + 2] = rhs(k, j2 + 2); + blockB[count + 3] = rhs(k, j2 + 3); + blockB[count + 4] = rhs(k, j2 + 4); + blockB[count + 5] = rhs(k, j2 + 5); + blockB[count + 6] = rhs(k, j2 + 6); + blockB[count + 7] = rhs(k, j2 + 7); + count += 8; + } } - // symmetric - Index h = 0; - for(Index k=j2; k= 4) { + for (Index j2 = end8; j2 < (std::min)(k2 + rows, packet_cols4); j2 += 4) { + // again we can split vertically in three different parts (transpose, symmetric, normal) + // transpose + for (Index k = k2; k < j2; k++) { + blockB[count + 0] = numext::conj(rhs(j2 + 0, k)); + blockB[count + 1] = numext::conj(rhs(j2 + 1, k)); + blockB[count + 2] = numext::conj(rhs(j2 + 2, k)); + blockB[count + 3] = numext::conj(rhs(j2 + 3, k)); + count += 4; + } + // symmetric + Index h = 0; + for (Index k = j2; k < j2 + 4; k++) { + // normal + for (Index w = 0; w < h; ++w) blockB[count + w] = rhs(k, j2 + w); + + blockB[count + h] = numext::real(rhs(k, k)); + + // transpose + for (Index w = h + 1; w < 4; ++w) blockB[count + w] = numext::conj(rhs(j2 + w, k)); + count += 4; + ++h; + } // normal - for (Index w=0 ; w= 8) { + for (Index j2 = k2 + rows; j2 < packet_cols8; j2 += 8) { + for (Index k = k2; k < end_k; k++) { + blockB[count + 0] = numext::conj(rhs(j2 + 0, k)); + blockB[count + 1] = numext::conj(rhs(j2 + 1, k)); + blockB[count + 2] = numext::conj(rhs(j2 + 2, k)); + blockB[count + 3] = numext::conj(rhs(j2 + 3, k)); + blockB[count + 4] = numext::conj(rhs(j2 + 4, k)); + blockB[count + 5] = numext::conj(rhs(j2 + 5, k)); + blockB[count + 6] = numext::conj(rhs(j2 + 6, k)); + blockB[count + 7] = numext::conj(rhs(j2 + 7, k)); + count += 8; + } } - // normal - for(Index k=j2+8; k= 4) { + for (Index j2 = (std::max)(packet_cols8, k2 + rows); j2 < packet_cols4; j2 += 4) { + for (Index k = k2; k < end_k; k++) { + blockB[count + 0] = numext::conj(rhs(j2 + 0, k)); + blockB[count + 1] = numext::conj(rhs(j2 + 1, k)); + blockB[count + 2] = numext::conj(rhs(j2 + 2, k)); + blockB[count + 3] = numext::conj(rhs(j2 + 3, k)); + count += 4; + } } } - } - if(nr>=4) - { - for(Index j2=end8; j2<(std::min)(k2+rows,packet_cols4); j2+=4) - { - // again we can split vertically in three different parts (transpose, symmetric, normal) + + // copy the remaining columns one at a time (=> the same with nr==1) + for (Index j2 = packet_cols4; j2 < cols; ++j2) { // transpose - for(Index k=k2; k=8) - { - for(Index j2=k2+rows; j2=4) + /* Optimized selfadjoint matrix * matrix (_SYMM) product built on top of + * the general matrix matrix product. + */ + template + struct product_selfadjoint_matrix; + + template + struct product_selfadjoint_matrix + { + + static EIGEN_STRONG_INLINE void run(Index rows, + Index cols, + const Scalar *lhs, + Index lhsStride, + const Scalar *rhs, + Index rhsStride, + Scalar *res, + Index resStride, + const Scalar &alpha, + level3_blocking &blocking) { - for(Index j2=(std::max)(packet_cols8,k2+rows); j2::IsComplex && EIGEN_LOGICAL_XOR(RhsSelfAdjoint, ConjugateRhs), + EIGEN_LOGICAL_XOR(LhsSelfAdjoint, LhsStorageOrder == RowMajor) ? ColMajor : RowMajor, + LhsSelfAdjoint, + NumTraits::IsComplex && EIGEN_LOGICAL_XOR(LhsSelfAdjoint, ConjugateLhs), + ColMajor>::run(cols, rows, rhs, rhsStride, lhs, lhsStride, res, resStride, alpha, blocking); } + }; - // copy the remaining columns one at a time (=> the same with nr==1) - for(Index j2=packet_cols4; j2 + struct product_selfadjoint_matrix + { - if(half==j2 && half &blocking); + }; - // normal - for(Index k=half+1; k -struct product_selfadjoint_matrix; - -template -struct product_selfadjoint_matrix -{ - - static EIGEN_STRONG_INLINE void run( - Index rows, Index cols, - const Scalar* lhs, Index lhsStride, - const Scalar* rhs, Index rhsStride, - Scalar* res, Index resStride, - const Scalar& alpha, level3_blocking& blocking) - { - product_selfadjoint_matrix::IsComplex && EIGEN_LOGICAL_XOR(RhsSelfAdjoint,ConjugateRhs), - EIGEN_LOGICAL_XOR(LhsSelfAdjoint,LhsStorageOrder==RowMajor) ? ColMajor : RowMajor, - LhsSelfAdjoint, NumTraits::IsComplex && EIGEN_LOGICAL_XOR(LhsSelfAdjoint,ConjugateLhs), - ColMajor> - ::run(cols, rows, rhs, rhsStride, lhs, lhsStride, res, resStride, alpha, blocking); - } -}; - -template -struct product_selfadjoint_matrix -{ - - static EIGEN_DONT_INLINE void run( - Index rows, Index cols, - const Scalar* _lhs, Index lhsStride, - const Scalar* _rhs, Index rhsStride, - Scalar* res, Index resStride, - const Scalar& alpha, level3_blocking& blocking); -}; - -template -EIGEN_DONT_INLINE void product_selfadjoint_matrix::run( - Index rows, Index cols, - const Scalar* _lhs, Index lhsStride, - const Scalar* _rhs, Index rhsStride, - Scalar* _res, Index resStride, - const Scalar& alpha, level3_blocking& blocking) + template + EIGEN_DONT_INLINE void product_selfadjoint_matrix::run(Index rows, + Index cols, + const Scalar *_lhs, + Index lhsStride, + const Scalar *_rhs, + Index rhsStride, + Scalar *_res, + Index resStride, + const Scalar &alpha, + level3_blocking &blocking) { Index size = rows; - typedef gebp_traits Traits; + typedef gebp_traits Traits; typedef const_blas_data_mapper LhsMapper; - typedef const_blas_data_mapper LhsTransposeMapper; + typedef const_blas_data_mapper + LhsTransposeMapper; typedef const_blas_data_mapper RhsMapper; typedef blas_data_mapper ResMapper; - LhsMapper lhs(_lhs,lhsStride); - LhsTransposeMapper lhs_transpose(_lhs,lhsStride); - RhsMapper rhs(_rhs,rhsStride); + LhsMapper lhs(_lhs, lhsStride); + LhsTransposeMapper lhs_transpose(_lhs, lhsStride); + RhsMapper rhs(_rhs, rhsStride); ResMapper res(_res, resStride); - Index kc = blocking.kc(); // cache block size along the K direction - Index mc = (std::min)(rows,blocking.mc()); // cache block size along the M direction + Index kc = blocking.kc();// cache block size along the K direction + Index mc = (std::min)(rows, blocking.mc());// cache block size along the M direction // kc must be smaller than mc - kc = (std::min)(kc,mc); - std::size_t sizeA = kc*mc; - std::size_t sizeB = kc*cols; + kc = (std::min)(kc, mc); + std::size_t sizeA = kc * mc; + std::size_t sizeB = kc * cols; ei_declare_aligned_stack_constructed_variable(Scalar, blockA, sizeA, blocking.blockA()); ei_declare_aligned_stack_constructed_variable(Scalar, blockB, sizeB, blocking.blockB()); gebp_kernel gebp_kernel; symm_pack_lhs pack_lhs; - gemm_pack_rhs pack_rhs; - gemm_pack_lhs pack_lhs_transposed; - - for(Index k2=0; k2 pack_rhs; + gemm_pack_lhs + pack_lhs_transposed; + + for (Index k2 = 0; k2 < size; k2 += kc) { + const Index actual_kc = (std::min)(k2 + kc, size) - k2; // we have selected one row panel of rhs and one column panel of lhs // pack rhs's panel into a sequential chunk of memory // and expand each coeff to a constant packet for further reuse - pack_rhs(blockB, rhs.getSubMapper(k2,0), actual_kc, cols); + pack_rhs(blockB, rhs.getSubMapper(k2, 0), actual_kc, cols); // the select lhs's panel has to be split in three different parts: // 1 - the transposed panel above the diagonal block => transposed packed copy // 2 - the diagonal block => special packed copy // 3 - the panel below the diagonal block => generic packed copy - for(Index i2=0; i2() - (blockA, lhs.getSubMapper(i2, k2), actual_kc, actual_mc); + for (Index i2 = k2 + kc; i2 < size; i2 += mc) { + const Index actual_mc = (std::min)(i2 + mc, size) - i2; + gemm_pack_lhs()( + blockA, lhs.getSubMapper(i2, k2), actual_kc, actual_mc); gebp_kernel(res.getSubMapper(i2, 0), blockA, blockB, actual_mc, actual_kc, cols, alpha); } } } -// matrix * selfadjoint product -template -struct product_selfadjoint_matrix -{ - - static EIGEN_DONT_INLINE void run( - Index rows, Index cols, - const Scalar* _lhs, Index lhsStride, - const Scalar* _rhs, Index rhsStride, - Scalar* res, Index resStride, - const Scalar& alpha, level3_blocking& blocking); -}; - -template -EIGEN_DONT_INLINE void product_selfadjoint_matrix::run( - Index rows, Index cols, - const Scalar* _lhs, Index lhsStride, - const Scalar* _rhs, Index rhsStride, - Scalar* _res, Index resStride, - const Scalar& alpha, level3_blocking& blocking) + // matrix * selfadjoint product + template + struct product_selfadjoint_matrix + { + + static EIGEN_DONT_INLINE void run(Index rows, + Index cols, + const Scalar *_lhs, + Index lhsStride, + const Scalar *_rhs, + Index rhsStride, + Scalar *res, + Index resStride, + const Scalar &alpha, + level3_blocking &blocking); + }; + + template + EIGEN_DONT_INLINE void product_selfadjoint_matrix::run(Index rows, + Index cols, + const Scalar *_lhs, + Index lhsStride, + const Scalar *_rhs, + Index rhsStride, + Scalar *_res, + Index resStride, + const Scalar &alpha, + level3_blocking &blocking) { Index size = cols; - typedef gebp_traits Traits; + typedef gebp_traits Traits; typedef const_blas_data_mapper LhsMapper; typedef blas_data_mapper ResMapper; - LhsMapper lhs(_lhs,lhsStride); - ResMapper res(_res,resStride); + LhsMapper lhs(_lhs, lhsStride); + ResMapper res(_res, resStride); - Index kc = blocking.kc(); // cache block size along the K direction - Index mc = (std::min)(rows,blocking.mc()); // cache block size along the M direction - std::size_t sizeA = kc*mc; - std::size_t sizeB = kc*cols; + Index kc = blocking.kc();// cache block size along the K direction + Index mc = (std::min)(rows, blocking.mc());// cache block size along the M direction + std::size_t sizeA = kc * mc; + std::size_t sizeB = kc * cols; ei_declare_aligned_stack_constructed_variable(Scalar, blockA, sizeA, blocking.blockA()); ei_declare_aligned_stack_constructed_variable(Scalar, blockB, sizeB, blocking.blockB()); gebp_kernel gebp_kernel; gemm_pack_lhs pack_lhs; - symm_pack_rhs pack_rhs; + symm_pack_rhs pack_rhs; - for(Index k2=0; k2 GEPP - for(Index i2=0; i2 -struct selfadjoint_product_impl -{ - typedef typename Product::Scalar Scalar; - - typedef internal::blas_traits LhsBlasTraits; - typedef typename LhsBlasTraits::DirectLinearAccessType ActualLhsType; - typedef internal::blas_traits RhsBlasTraits; - typedef typename RhsBlasTraits::DirectLinearAccessType ActualRhsType; - - enum { - LhsIsUpper = (LhsMode&(Upper|Lower))==Upper, - LhsIsSelfAdjoint = (LhsMode&SelfAdjoint)==SelfAdjoint, - RhsIsUpper = (RhsMode&(Upper|Lower))==Upper, - RhsIsSelfAdjoint = (RhsMode&SelfAdjoint)==SelfAdjoint - }; - - template - static void run(Dest &dst, const Lhs &a_lhs, const Rhs &a_rhs, const Scalar& alpha) + + template + struct selfadjoint_product_impl { - eigen_assert(dst.rows()==a_lhs.rows() && dst.cols()==a_rhs.cols()); - - typename internal::add_const_on_value_type::type lhs = LhsBlasTraits::extract(a_lhs); - typename internal::add_const_on_value_type::type rhs = RhsBlasTraits::extract(a_rhs); - - Scalar actualAlpha = alpha * LhsBlasTraits::extractScalarFactor(a_lhs) - * RhsBlasTraits::extractScalarFactor(a_rhs); - - typedef internal::gemm_blocking_space<(Dest::Flags&RowMajorBit) ? RowMajor : ColMajor,Scalar,Scalar, - Lhs::MaxRowsAtCompileTime, Rhs::MaxColsAtCompileTime, Lhs::MaxColsAtCompileTime,1> BlockingType; - - BlockingType blocking(lhs.rows(), rhs.cols(), lhs.cols(), 1, false); - - internal::product_selfadjoint_matrix::Flags &RowMajorBit) ? RowMajor : ColMajor, LhsIsSelfAdjoint, - NumTraits::IsComplex && EIGEN_LOGICAL_XOR(LhsIsUpper,bool(LhsBlasTraits::NeedToConjugate)), - EIGEN_LOGICAL_XOR(RhsIsUpper,internal::traits::Flags &RowMajorBit) ? RowMajor : ColMajor, RhsIsSelfAdjoint, - NumTraits::IsComplex && EIGEN_LOGICAL_XOR(RhsIsUpper,bool(RhsBlasTraits::NeedToConjugate)), - internal::traits::Flags&RowMajorBit ? RowMajor : ColMajor> - ::run( - lhs.rows(), rhs.cols(), // sizes - &lhs.coeffRef(0,0), lhs.outerStride(), // lhs info - &rhs.coeffRef(0,0), rhs.outerStride(), // rhs info - &dst.coeffRef(0,0), dst.outerStride(), // result info - actualAlpha, blocking // alpha + typedef typename Product::Scalar Scalar; + + typedef internal::blas_traits LhsBlasTraits; + typedef typename LhsBlasTraits::DirectLinearAccessType ActualLhsType; + typedef internal::blas_traits RhsBlasTraits; + typedef typename RhsBlasTraits::DirectLinearAccessType ActualRhsType; + + enum { + LhsIsUpper = (LhsMode & (Upper | Lower)) == Upper, + LhsIsSelfAdjoint = (LhsMode & SelfAdjoint) == SelfAdjoint, + RhsIsUpper = (RhsMode & (Upper | Lower)) == Upper, + RhsIsSelfAdjoint = (RhsMode & SelfAdjoint) == SelfAdjoint + }; + + template static void run(Dest &dst, const Lhs &a_lhs, const Rhs &a_rhs, const Scalar &alpha) + { + eigen_assert(dst.rows() == a_lhs.rows() && dst.cols() == a_rhs.cols()); + + typename internal::add_const_on_value_type::type lhs = LhsBlasTraits::extract(a_lhs); + typename internal::add_const_on_value_type::type rhs = RhsBlasTraits::extract(a_rhs); + + Scalar actualAlpha = + alpha * LhsBlasTraits::extractScalarFactor(a_lhs) * RhsBlasTraits::extractScalarFactor(a_rhs); + + typedef internal::gemm_blocking_space<(Dest::Flags & RowMajorBit) ? RowMajor : ColMajor, + Scalar, + Scalar, + Lhs::MaxRowsAtCompileTime, + Rhs::MaxColsAtCompileTime, + Lhs::MaxColsAtCompileTime, + 1> + BlockingType; + + BlockingType blocking(lhs.rows(), rhs.cols(), lhs.cols(), 1, false); + + internal::product_selfadjoint_matrix::Flags & RowMajorBit) ? RowMajor : ColMajor, + LhsIsSelfAdjoint, + NumTraits::IsComplex && EIGEN_LOGICAL_XOR(LhsIsUpper, bool(LhsBlasTraits::NeedToConjugate)), + EIGEN_LOGICAL_XOR(RhsIsUpper, internal::traits::Flags & RowMajorBit) ? RowMajor : ColMajor, + RhsIsSelfAdjoint, + NumTraits::IsComplex && EIGEN_LOGICAL_XOR(RhsIsUpper, bool(RhsBlasTraits::NeedToConjugate)), + internal::traits::Flags & RowMajorBit ? RowMajor : ColMajor>::run(lhs.rows(), + rhs.cols(),// sizes + &lhs.coeffRef(0, 0), + lhs.outerStride(),// lhs info + &rhs.coeffRef(0, 0), + rhs.outerStride(),// rhs info + &dst.coeffRef(0, 0), + dst.outerStride(),// result info + actualAlpha, + blocking// alpha ); - } -}; + } + }; -} // end namespace internal +}// end namespace internal -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_SELFADJOINT_MATRIX_MATRIX_H +#endif// EIGEN_SELFADJOINT_MATRIX_MATRIX_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/products/SelfadjointMatrixMatrix_BLAS.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/products/SelfadjointMatrixMatrix_BLAS.h index 9a531850..43496e45 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/products/SelfadjointMatrixMatrix_BLAS.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/products/SelfadjointMatrixMatrix_BLAS.h @@ -33,255 +33,340 @@ #ifndef EIGEN_SELFADJOINT_MATRIX_MATRIX_BLAS_H #define EIGEN_SELFADJOINT_MATRIX_MATRIX_BLAS_H -namespace Eigen { +namespace Eigen { namespace internal { -/* Optimized selfadjoint matrix * matrix (?SYMM/?HEMM) product */ + /* Optimized selfadjoint matrix * matrix (?SYMM/?HEMM) product */ -#define EIGEN_BLAS_SYMM_L(EIGTYPE, BLASTYPE, EIGPREFIX, BLASFUNC) \ -template \ -struct product_selfadjoint_matrix \ -{\ -\ - static void run( \ - Index rows, Index cols, \ - const EIGTYPE* _lhs, Index lhsStride, \ - const EIGTYPE* _rhs, Index rhsStride, \ - EIGTYPE* res, Index resStride, \ - EIGTYPE alpha, level3_blocking& /*blocking*/) \ - { \ - char side='L', uplo='L'; \ - BlasIndex m, n, lda, ldb, ldc; \ - const EIGTYPE *a, *b; \ - EIGTYPE beta(1); \ - MatrixX##EIGPREFIX b_tmp; \ -\ -/* Set transpose options */ \ -/* Set m, n, k */ \ - m = convert_index(rows); \ - n = convert_index(cols); \ -\ -/* Set lda, ldb, ldc */ \ - lda = convert_index(lhsStride); \ - ldb = convert_index(rhsStride); \ - ldc = convert_index(resStride); \ -\ -/* Set a, b, c */ \ - if (LhsStorageOrder==RowMajor) uplo='U'; \ - a = _lhs; \ -\ - if (RhsStorageOrder==RowMajor) { \ - Map > rhs(_rhs,n,m,OuterStride<>(rhsStride)); \ - b_tmp = rhs.adjoint(); \ - b = b_tmp.data(); \ - ldb = convert_index(b_tmp.outerStride()); \ - } else b = _rhs; \ -\ - BLASFUNC(&side, &uplo, &m, &n, (const BLASTYPE*)&numext::real_ref(alpha), (const BLASTYPE*)a, &lda, (const BLASTYPE*)b, &ldb, (const BLASTYPE*)&numext::real_ref(beta), (BLASTYPE*)res, &ldc); \ -\ - } \ -}; +#define EIGEN_BLAS_SYMM_L(EIGTYPE, BLASTYPE, EIGPREFIX, BLASFUNC) \ + template \ + struct product_selfadjoint_matrix \ + { \ + \ + static void run(Index rows, \ + Index cols, \ + const EIGTYPE *_lhs, \ + Index lhsStride, \ + const EIGTYPE *_rhs, \ + Index rhsStride, \ + EIGTYPE *res, \ + Index resStride, \ + EIGTYPE alpha, \ + level3_blocking & /*blocking*/) \ + { \ + char side = 'L', uplo = 'L'; \ + BlasIndex m, n, lda, ldb, ldc; \ + const EIGTYPE *a, *b; \ + EIGTYPE beta(1); \ + MatrixX##EIGPREFIX b_tmp; \ + \ + /* Set transpose options */ \ + /* Set m, n, k */ \ + m = convert_index(rows); \ + n = convert_index(cols); \ + \ + /* Set lda, ldb, ldc */ \ + lda = convert_index(lhsStride); \ + ldb = convert_index(rhsStride); \ + ldc = convert_index(resStride); \ + \ + /* Set a, b, c */ \ + if (LhsStorageOrder == RowMajor) uplo = 'U'; \ + a = _lhs; \ + \ + if (RhsStorageOrder == RowMajor) { \ + Map> rhs(_rhs, n, m, OuterStride<>(rhsStride)); \ + b_tmp = rhs.adjoint(); \ + b = b_tmp.data(); \ + ldb = convert_index(b_tmp.outerStride()); \ + } else \ + b = _rhs; \ + \ + BLASFUNC(&side, \ + &uplo, \ + &m, \ + &n, \ + (const BLASTYPE *)&numext::real_ref(alpha), \ + (const BLASTYPE *)a, \ + &lda, \ + (const BLASTYPE *)b, \ + &ldb, \ + (const BLASTYPE *)&numext::real_ref(beta), \ + (BLASTYPE *)res, \ + &ldc); \ + } \ + }; -#define EIGEN_BLAS_HEMM_L(EIGTYPE, BLASTYPE, EIGPREFIX, BLASFUNC) \ -template \ -struct product_selfadjoint_matrix \ -{\ - static void run( \ - Index rows, Index cols, \ - const EIGTYPE* _lhs, Index lhsStride, \ - const EIGTYPE* _rhs, Index rhsStride, \ - EIGTYPE* res, Index resStride, \ - EIGTYPE alpha, level3_blocking& /*blocking*/) \ - { \ - char side='L', uplo='L'; \ - BlasIndex m, n, lda, ldb, ldc; \ - const EIGTYPE *a, *b; \ - EIGTYPE beta(1); \ - MatrixX##EIGPREFIX b_tmp; \ - Matrix a_tmp; \ -\ -/* Set transpose options */ \ -/* Set m, n, k */ \ - m = convert_index(rows); \ - n = convert_index(cols); \ -\ -/* Set lda, ldb, ldc */ \ - lda = convert_index(lhsStride); \ - ldb = convert_index(rhsStride); \ - ldc = convert_index(resStride); \ -\ -/* Set a, b, c */ \ - if (((LhsStorageOrder==ColMajor) && ConjugateLhs) || ((LhsStorageOrder==RowMajor) && (!ConjugateLhs))) { \ - Map, 0, OuterStride<> > lhs(_lhs,m,m,OuterStride<>(lhsStride)); \ - a_tmp = lhs.conjugate(); \ - a = a_tmp.data(); \ - lda = convert_index(a_tmp.outerStride()); \ - } else a = _lhs; \ - if (LhsStorageOrder==RowMajor) uplo='U'; \ -\ - if (RhsStorageOrder==ColMajor && (!ConjugateRhs)) { \ - b = _rhs; } \ - else { \ - if (RhsStorageOrder==ColMajor && ConjugateRhs) { \ - Map > rhs(_rhs,m,n,OuterStride<>(rhsStride)); \ - b_tmp = rhs.conjugate(); \ - } else \ - if (ConjugateRhs) { \ - Map > rhs(_rhs,n,m,OuterStride<>(rhsStride)); \ - b_tmp = rhs.adjoint(); \ - } else { \ - Map > rhs(_rhs,n,m,OuterStride<>(rhsStride)); \ - b_tmp = rhs.transpose(); \ - } \ - b = b_tmp.data(); \ - ldb = convert_index(b_tmp.outerStride()); \ - } \ -\ - BLASFUNC(&side, &uplo, &m, &n, (const BLASTYPE*)&numext::real_ref(alpha), (const BLASTYPE*)a, &lda, (const BLASTYPE*)b, &ldb, (const BLASTYPE*)&numext::real_ref(beta), (BLASTYPE*)res, &ldc); \ -\ - } \ -}; +#define EIGEN_BLAS_HEMM_L(EIGTYPE, BLASTYPE, EIGPREFIX, BLASFUNC) \ + template \ + struct product_selfadjoint_matrix \ + { \ + static void run(Index rows, \ + Index cols, \ + const EIGTYPE *_lhs, \ + Index lhsStride, \ + const EIGTYPE *_rhs, \ + Index rhsStride, \ + EIGTYPE *res, \ + Index resStride, \ + EIGTYPE alpha, \ + level3_blocking & /*blocking*/) \ + { \ + char side = 'L', uplo = 'L'; \ + BlasIndex m, n, lda, ldb, ldc; \ + const EIGTYPE *a, *b; \ + EIGTYPE beta(1); \ + MatrixX##EIGPREFIX b_tmp; \ + Matrix a_tmp; \ + \ + /* Set transpose options */ \ + /* Set m, n, k */ \ + m = convert_index(rows); \ + n = convert_index(cols); \ + \ + /* Set lda, ldb, ldc */ \ + lda = convert_index(lhsStride); \ + ldb = convert_index(rhsStride); \ + ldc = convert_index(resStride); \ + \ + /* Set a, b, c */ \ + if (((LhsStorageOrder == ColMajor) && ConjugateLhs) || ((LhsStorageOrder == RowMajor) && (!ConjugateLhs))) { \ + Map, 0, OuterStride<>> lhs( \ + _lhs, m, m, OuterStride<>(lhsStride)); \ + a_tmp = lhs.conjugate(); \ + a = a_tmp.data(); \ + lda = convert_index(a_tmp.outerStride()); \ + } else \ + a = _lhs; \ + if (LhsStorageOrder == RowMajor) uplo = 'U'; \ + \ + if (RhsStorageOrder == ColMajor && (!ConjugateRhs)) { \ + b = _rhs; \ + } else { \ + if (RhsStorageOrder == ColMajor && ConjugateRhs) { \ + Map> rhs(_rhs, m, n, OuterStride<>(rhsStride)); \ + b_tmp = rhs.conjugate(); \ + } else if (ConjugateRhs) { \ + Map> rhs(_rhs, n, m, OuterStride<>(rhsStride)); \ + b_tmp = rhs.adjoint(); \ + } else { \ + Map> rhs(_rhs, n, m, OuterStride<>(rhsStride)); \ + b_tmp = rhs.transpose(); \ + } \ + b = b_tmp.data(); \ + ldb = convert_index(b_tmp.outerStride()); \ + } \ + \ + BLASFUNC(&side, \ + &uplo, \ + &m, \ + &n, \ + (const BLASTYPE *)&numext::real_ref(alpha), \ + (const BLASTYPE *)a, \ + &lda, \ + (const BLASTYPE *)b, \ + &ldb, \ + (const BLASTYPE *)&numext::real_ref(beta), \ + (BLASTYPE *)res, \ + &ldc); \ + } \ + }; #ifdef EIGEN_USE_MKL -EIGEN_BLAS_SYMM_L(double, double, d, dsymm) -EIGEN_BLAS_SYMM_L(float, float, f, ssymm) -EIGEN_BLAS_HEMM_L(dcomplex, MKL_Complex16, cd, zhemm) -EIGEN_BLAS_HEMM_L(scomplex, MKL_Complex8, cf, chemm) + EIGEN_BLAS_SYMM_L(double, double, d, dsymm) + EIGEN_BLAS_SYMM_L(float, float, f, ssymm) + EIGEN_BLAS_HEMM_L(dcomplex, MKL_Complex16, cd, zhemm) + EIGEN_BLAS_HEMM_L(scomplex, MKL_Complex8, cf, chemm) #else -EIGEN_BLAS_SYMM_L(double, double, d, dsymm_) -EIGEN_BLAS_SYMM_L(float, float, f, ssymm_) -EIGEN_BLAS_HEMM_L(dcomplex, double, cd, zhemm_) -EIGEN_BLAS_HEMM_L(scomplex, float, cf, chemm_) + EIGEN_BLAS_SYMM_L(double, double, d, dsymm_) + EIGEN_BLAS_SYMM_L(float, float, f, ssymm_) + EIGEN_BLAS_HEMM_L(dcomplex, double, cd, zhemm_) + EIGEN_BLAS_HEMM_L(scomplex, float, cf, chemm_) #endif -/* Optimized matrix * selfadjoint matrix (?SYMM/?HEMM) product */ + /* Optimized matrix * selfadjoint matrix (?SYMM/?HEMM) product */ -#define EIGEN_BLAS_SYMM_R(EIGTYPE, BLASTYPE, EIGPREFIX, BLASFUNC) \ -template \ -struct product_selfadjoint_matrix \ -{\ -\ - static void run( \ - Index rows, Index cols, \ - const EIGTYPE* _lhs, Index lhsStride, \ - const EIGTYPE* _rhs, Index rhsStride, \ - EIGTYPE* res, Index resStride, \ - EIGTYPE alpha, level3_blocking& /*blocking*/) \ - { \ - char side='R', uplo='L'; \ - BlasIndex m, n, lda, ldb, ldc; \ - const EIGTYPE *a, *b; \ - EIGTYPE beta(1); \ - MatrixX##EIGPREFIX b_tmp; \ -\ -/* Set m, n, k */ \ - m = convert_index(rows); \ - n = convert_index(cols); \ -\ -/* Set lda, ldb, ldc */ \ - lda = convert_index(rhsStride); \ - ldb = convert_index(lhsStride); \ - ldc = convert_index(resStride); \ -\ -/* Set a, b, c */ \ - if (RhsStorageOrder==RowMajor) uplo='U'; \ - a = _rhs; \ -\ - if (LhsStorageOrder==RowMajor) { \ - Map > lhs(_lhs,n,m,OuterStride<>(rhsStride)); \ - b_tmp = lhs.adjoint(); \ - b = b_tmp.data(); \ - ldb = convert_index(b_tmp.outerStride()); \ - } else b = _lhs; \ -\ - BLASFUNC(&side, &uplo, &m, &n, (const BLASTYPE*)&numext::real_ref(alpha), (const BLASTYPE*)a, &lda, (const BLASTYPE*)b, &ldb, (const BLASTYPE*)&numext::real_ref(beta), (BLASTYPE*)res, &ldc); \ -\ - } \ -}; +#define EIGEN_BLAS_SYMM_R(EIGTYPE, BLASTYPE, EIGPREFIX, BLASFUNC) \ + template \ + struct product_selfadjoint_matrix \ + { \ + \ + static void run(Index rows, \ + Index cols, \ + const EIGTYPE *_lhs, \ + Index lhsStride, \ + const EIGTYPE *_rhs, \ + Index rhsStride, \ + EIGTYPE *res, \ + Index resStride, \ + EIGTYPE alpha, \ + level3_blocking & /*blocking*/) \ + { \ + char side = 'R', uplo = 'L'; \ + BlasIndex m, n, lda, ldb, ldc; \ + const EIGTYPE *a, *b; \ + EIGTYPE beta(1); \ + MatrixX##EIGPREFIX b_tmp; \ + \ + /* Set m, n, k */ \ + m = convert_index(rows); \ + n = convert_index(cols); \ + \ + /* Set lda, ldb, ldc */ \ + lda = convert_index(rhsStride); \ + ldb = convert_index(lhsStride); \ + ldc = convert_index(resStride); \ + \ + /* Set a, b, c */ \ + if (RhsStorageOrder == RowMajor) uplo = 'U'; \ + a = _rhs; \ + \ + if (LhsStorageOrder == RowMajor) { \ + Map> lhs(_lhs, n, m, OuterStride<>(rhsStride)); \ + b_tmp = lhs.adjoint(); \ + b = b_tmp.data(); \ + ldb = convert_index(b_tmp.outerStride()); \ + } else \ + b = _lhs; \ + \ + BLASFUNC(&side, \ + &uplo, \ + &m, \ + &n, \ + (const BLASTYPE *)&numext::real_ref(alpha), \ + (const BLASTYPE *)a, \ + &lda, \ + (const BLASTYPE *)b, \ + &ldb, \ + (const BLASTYPE *)&numext::real_ref(beta), \ + (BLASTYPE *)res, \ + &ldc); \ + } \ + }; -#define EIGEN_BLAS_HEMM_R(EIGTYPE, BLASTYPE, EIGPREFIX, BLASFUNC) \ -template \ -struct product_selfadjoint_matrix \ -{\ - static void run( \ - Index rows, Index cols, \ - const EIGTYPE* _lhs, Index lhsStride, \ - const EIGTYPE* _rhs, Index rhsStride, \ - EIGTYPE* res, Index resStride, \ - EIGTYPE alpha, level3_blocking& /*blocking*/) \ - { \ - char side='R', uplo='L'; \ - BlasIndex m, n, lda, ldb, ldc; \ - const EIGTYPE *a, *b; \ - EIGTYPE beta(1); \ - MatrixX##EIGPREFIX b_tmp; \ - Matrix a_tmp; \ -\ -/* Set m, n, k */ \ - m = convert_index(rows); \ - n = convert_index(cols); \ -\ -/* Set lda, ldb, ldc */ \ - lda = convert_index(rhsStride); \ - ldb = convert_index(lhsStride); \ - ldc = convert_index(resStride); \ -\ -/* Set a, b, c */ \ - if (((RhsStorageOrder==ColMajor) && ConjugateRhs) || ((RhsStorageOrder==RowMajor) && (!ConjugateRhs))) { \ - Map, 0, OuterStride<> > rhs(_rhs,n,n,OuterStride<>(rhsStride)); \ - a_tmp = rhs.conjugate(); \ - a = a_tmp.data(); \ - lda = convert_index(a_tmp.outerStride()); \ - } else a = _rhs; \ - if (RhsStorageOrder==RowMajor) uplo='U'; \ -\ - if (LhsStorageOrder==ColMajor && (!ConjugateLhs)) { \ - b = _lhs; } \ - else { \ - if (LhsStorageOrder==ColMajor && ConjugateLhs) { \ - Map > lhs(_lhs,m,n,OuterStride<>(lhsStride)); \ - b_tmp = lhs.conjugate(); \ - } else \ - if (ConjugateLhs) { \ - Map > lhs(_lhs,n,m,OuterStride<>(lhsStride)); \ - b_tmp = lhs.adjoint(); \ - } else { \ - Map > lhs(_lhs,n,m,OuterStride<>(lhsStride)); \ - b_tmp = lhs.transpose(); \ - } \ - b = b_tmp.data(); \ - ldb = convert_index(b_tmp.outerStride()); \ - } \ -\ - BLASFUNC(&side, &uplo, &m, &n, (const BLASTYPE*)&numext::real_ref(alpha), (const BLASTYPE*)a, &lda, (const BLASTYPE*)b, &ldb, (const BLASTYPE*)&numext::real_ref(beta), (BLASTYPE*)res, &ldc); \ - } \ -}; +#define EIGEN_BLAS_HEMM_R(EIGTYPE, BLASTYPE, EIGPREFIX, BLASFUNC) \ + template \ + struct product_selfadjoint_matrix \ + { \ + static void run(Index rows, \ + Index cols, \ + const EIGTYPE *_lhs, \ + Index lhsStride, \ + const EIGTYPE *_rhs, \ + Index rhsStride, \ + EIGTYPE *res, \ + Index resStride, \ + EIGTYPE alpha, \ + level3_blocking & /*blocking*/) \ + { \ + char side = 'R', uplo = 'L'; \ + BlasIndex m, n, lda, ldb, ldc; \ + const EIGTYPE *a, *b; \ + EIGTYPE beta(1); \ + MatrixX##EIGPREFIX b_tmp; \ + Matrix a_tmp; \ + \ + /* Set m, n, k */ \ + m = convert_index(rows); \ + n = convert_index(cols); \ + \ + /* Set lda, ldb, ldc */ \ + lda = convert_index(rhsStride); \ + ldb = convert_index(lhsStride); \ + ldc = convert_index(resStride); \ + \ + /* Set a, b, c */ \ + if (((RhsStorageOrder == ColMajor) && ConjugateRhs) || ((RhsStorageOrder == RowMajor) && (!ConjugateRhs))) { \ + Map, 0, OuterStride<>> rhs( \ + _rhs, n, n, OuterStride<>(rhsStride)); \ + a_tmp = rhs.conjugate(); \ + a = a_tmp.data(); \ + lda = convert_index(a_tmp.outerStride()); \ + } else \ + a = _rhs; \ + if (RhsStorageOrder == RowMajor) uplo = 'U'; \ + \ + if (LhsStorageOrder == ColMajor && (!ConjugateLhs)) { \ + b = _lhs; \ + } else { \ + if (LhsStorageOrder == ColMajor && ConjugateLhs) { \ + Map> lhs(_lhs, m, n, OuterStride<>(lhsStride)); \ + b_tmp = lhs.conjugate(); \ + } else if (ConjugateLhs) { \ + Map> lhs(_lhs, n, m, OuterStride<>(lhsStride)); \ + b_tmp = lhs.adjoint(); \ + } else { \ + Map> lhs(_lhs, n, m, OuterStride<>(lhsStride)); \ + b_tmp = lhs.transpose(); \ + } \ + b = b_tmp.data(); \ + ldb = convert_index(b_tmp.outerStride()); \ + } \ + \ + BLASFUNC(&side, \ + &uplo, \ + &m, \ + &n, \ + (const BLASTYPE *)&numext::real_ref(alpha), \ + (const BLASTYPE *)a, \ + &lda, \ + (const BLASTYPE *)b, \ + &ldb, \ + (const BLASTYPE *)&numext::real_ref(beta), \ + (BLASTYPE *)res, \ + &ldc); \ + } \ + }; #ifdef EIGEN_USE_MKL -EIGEN_BLAS_SYMM_R(double, double, d, dsymm) -EIGEN_BLAS_SYMM_R(float, float, f, ssymm) -EIGEN_BLAS_HEMM_R(dcomplex, MKL_Complex16, cd, zhemm) -EIGEN_BLAS_HEMM_R(scomplex, MKL_Complex8, cf, chemm) + EIGEN_BLAS_SYMM_R(double, double, d, dsymm) + EIGEN_BLAS_SYMM_R(float, float, f, ssymm) + EIGEN_BLAS_HEMM_R(dcomplex, MKL_Complex16, cd, zhemm) + EIGEN_BLAS_HEMM_R(scomplex, MKL_Complex8, cf, chemm) #else -EIGEN_BLAS_SYMM_R(double, double, d, dsymm_) -EIGEN_BLAS_SYMM_R(float, float, f, ssymm_) -EIGEN_BLAS_HEMM_R(dcomplex, double, cd, zhemm_) -EIGEN_BLAS_HEMM_R(scomplex, float, cf, chemm_) + EIGEN_BLAS_SYMM_R(double, double, d, dsymm_) + EIGEN_BLAS_SYMM_R(float, float, f, ssymm_) + EIGEN_BLAS_HEMM_R(dcomplex, double, cd, zhemm_) + EIGEN_BLAS_HEMM_R(scomplex, float, cf, chemm_) #endif -} // end namespace internal +}// end namespace internal -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_SELFADJOINT_MATRIX_MATRIX_BLAS_H +#endif// EIGEN_SELFADJOINT_MATRIX_MATRIX_BLAS_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/products/SelfadjointMatrixVector.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/products/SelfadjointMatrixVector.h index 3fd180e6..eb21f6f9 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/products/SelfadjointMatrixVector.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/products/SelfadjointMatrixVector.h @@ -10,251 +10,283 @@ #ifndef EIGEN_SELFADJOINT_MATRIX_VECTOR_H #define EIGEN_SELFADJOINT_MATRIX_VECTOR_H -namespace Eigen { +namespace Eigen { namespace internal { -/* Optimized selfadjoint matrix * vector product: - * This algorithm processes 2 columns at onces that allows to both reduce - * the number of load/stores of the result by a factor 2 and to reduce - * the instruction dependency. - */ - -template -struct selfadjoint_matrix_vector_product; - -template -struct selfadjoint_matrix_vector_product - -{ -static EIGEN_DONT_INLINE void run( - Index size, - const Scalar* lhs, Index lhsStride, - const Scalar* rhs, - Scalar* res, - Scalar alpha); -}; - -template -EIGEN_DONT_INLINE void selfadjoint_matrix_vector_product::run( - Index size, - const Scalar* lhs, Index lhsStride, - const Scalar* rhs, - Scalar* res, - Scalar alpha) -{ - typedef typename packet_traits::type Packet; - typedef typename NumTraits::Real RealScalar; - const Index PacketSize = sizeof(Packet)/sizeof(Scalar); - - enum { - IsRowMajor = StorageOrder==RowMajor ? 1 : 0, - IsLower = UpLo == Lower ? 1 : 0, - FirstTriangular = IsRowMajor == IsLower - }; - - conj_helper::IsComplex && EIGEN_LOGICAL_XOR(ConjugateLhs, IsRowMajor), ConjugateRhs> cj0; - conj_helper::IsComplex && EIGEN_LOGICAL_XOR(ConjugateLhs, !IsRowMajor), ConjugateRhs> cj1; - conj_helper cjd; - - conj_helper::IsComplex && EIGEN_LOGICAL_XOR(ConjugateLhs, IsRowMajor), ConjugateRhs> pcj0; - conj_helper::IsComplex && EIGEN_LOGICAL_XOR(ConjugateLhs, !IsRowMajor), ConjugateRhs> pcj1; - - Scalar cjAlpha = ConjugateRhs ? numext::conj(alpha) : alpha; - + /* Optimized selfadjoint matrix * vector product: + * This algorithm processes 2 columns at onces that allows to both reduce + * the number of load/stores of the result by a factor 2 and to reduce + * the instruction dependency. + */ + + template + struct selfadjoint_matrix_vector_product; + + template + struct selfadjoint_matrix_vector_product - Index bound = (std::max)(Index(0),size-8) & 0xfffffffe; - if (FirstTriangular) - bound = size - bound; - - for (Index j=FirstTriangular ? bound : 0; - j<(FirstTriangular ? size : bound);j+=2) { - const Scalar* EIGEN_RESTRICT A0 = lhs + j*lhsStride; - const Scalar* EIGEN_RESTRICT A1 = lhs + (j+1)*lhsStride; - - Scalar t0 = cjAlpha * rhs[j]; - Packet ptmp0 = pset1(t0); - Scalar t1 = cjAlpha * rhs[j+1]; - Packet ptmp1 = pset1(t1); - - Scalar t2(0); - Packet ptmp2 = pset1(t2); - Scalar t3(0); - Packet ptmp3 = pset1(t3); - - Index starti = FirstTriangular ? 0 : j+2; - Index endi = FirstTriangular ? j : size; - Index alignedStart = (starti) + internal::first_default_aligned(&res[starti], endi-starti); - Index alignedEnd = alignedStart + ((endi-alignedStart)/(PacketSize))*(PacketSize); - - res[j] += cjd.pmul(numext::real(A0[j]), t0); - res[j+1] += cjd.pmul(numext::real(A1[j+1]), t1); - if(FirstTriangular) - { - res[j] += cj0.pmul(A1[j], t1); - t3 += cj1.pmul(A1[j], rhs[j]); - } - else - { - res[j+1] += cj0.pmul(A0[j+1],t0); - t2 += cj1.pmul(A0[j+1], rhs[j+1]); - } - - for (Index i=starti; i huge speed up) - // gcc 4.2 does this optimization automatically. - const Scalar* EIGEN_RESTRICT a0It = A0 + alignedStart; - const Scalar* EIGEN_RESTRICT a1It = A1 + alignedStart; - const Scalar* EIGEN_RESTRICT rhsIt = rhs + alignedStart; - Scalar* EIGEN_RESTRICT resIt = res + alignedStart; - for (Index i=alignedStart; i(a0It); a0It += PacketSize; - Packet A1i = ploadu(a1It); a1It += PacketSize; - Packet Bi = ploadu(rhsIt); rhsIt += PacketSize; // FIXME should be aligned in most cases - Packet Xi = pload (resIt); - - Xi = pcj0.pmadd(A0i,ptmp0, pcj0.pmadd(A1i,ptmp1,Xi)); - ptmp2 = pcj1.pmadd(A0i, Bi, ptmp2); - ptmp3 = pcj1.pmadd(A1i, Bi, ptmp3); - pstore(resIt,Xi); resIt += PacketSize; - } - for (Index i=alignedEnd; i + EIGEN_DONT_INLINE void + selfadjoint_matrix_vector_product::run( + Index size, + const Scalar *lhs, + Index lhsStride, + const Scalar *rhs, + Scalar *res, + Scalar alpha) { - const Scalar* EIGEN_RESTRICT A0 = lhs + j*lhsStride; + typedef typename packet_traits::type Packet; + typedef typename NumTraits::Real RealScalar; + const Index PacketSize = sizeof(Packet) / sizeof(Scalar); - Scalar t1 = cjAlpha * rhs[j]; - Scalar t2(0); - res[j] += cjd.pmul(numext::real(A0[j]), t1); - for (Index i=FirstTriangular ? 0 : j+1; i<(FirstTriangular ? j : size); i++) - { - res[i] += cj0.pmul(A0[i], t1); - t2 += cj1.pmul(A0[i], rhs[i]); + enum { + IsRowMajor = StorageOrder == RowMajor ? 1 : 0, + IsLower = UpLo == Lower ? 1 : 0, + FirstTriangular = IsRowMajor == IsLower + }; + + conj_helper::IsComplex && EIGEN_LOGICAL_XOR(ConjugateLhs, IsRowMajor), + ConjugateRhs> + cj0; + conj_helper::IsComplex && EIGEN_LOGICAL_XOR(ConjugateLhs, !IsRowMajor), + ConjugateRhs> + cj1; + conj_helper cjd; + + conj_helper::IsComplex && EIGEN_LOGICAL_XOR(ConjugateLhs, IsRowMajor), + ConjugateRhs> + pcj0; + conj_helper::IsComplex && EIGEN_LOGICAL_XOR(ConjugateLhs, !IsRowMajor), + ConjugateRhs> + pcj1; + + Scalar cjAlpha = ConjugateRhs ? numext::conj(alpha) : alpha; + + + Index bound = (std::max)(Index(0), size - 8) & 0xfffffffe; + if (FirstTriangular) bound = size - bound; + + for (Index j = FirstTriangular ? bound : 0; j < (FirstTriangular ? size : bound); j += 2) { + const Scalar *EIGEN_RESTRICT A0 = lhs + j * lhsStride; + const Scalar *EIGEN_RESTRICT A1 = lhs + (j + 1) * lhsStride; + + Scalar t0 = cjAlpha * rhs[j]; + Packet ptmp0 = pset1(t0); + Scalar t1 = cjAlpha * rhs[j + 1]; + Packet ptmp1 = pset1(t1); + + Scalar t2(0); + Packet ptmp2 = pset1(t2); + Scalar t3(0); + Packet ptmp3 = pset1(t3); + + Index starti = FirstTriangular ? 0 : j + 2; + Index endi = FirstTriangular ? j : size; + Index alignedStart = (starti) + internal::first_default_aligned(&res[starti], endi - starti); + Index alignedEnd = alignedStart + ((endi - alignedStart) / (PacketSize)) * (PacketSize); + + res[j] += cjd.pmul(numext::real(A0[j]), t0); + res[j + 1] += cjd.pmul(numext::real(A1[j + 1]), t1); + if (FirstTriangular) { + res[j] += cj0.pmul(A1[j], t1); + t3 += cj1.pmul(A1[j], rhs[j]); + } else { + res[j + 1] += cj0.pmul(A0[j + 1], t0); + t2 += cj1.pmul(A0[j + 1], rhs[j + 1]); + } + + for (Index i = starti; i < alignedStart; ++i) { + res[i] += cj0.pmul(A0[i], t0) + cj0.pmul(A1[i], t1); + t2 += cj1.pmul(A0[i], rhs[i]); + t3 += cj1.pmul(A1[i], rhs[i]); + } + // Yes this an optimization for gcc 4.3 and 4.4 (=> huge speed up) + // gcc 4.2 does this optimization automatically. + const Scalar *EIGEN_RESTRICT a0It = A0 + alignedStart; + const Scalar *EIGEN_RESTRICT a1It = A1 + alignedStart; + const Scalar *EIGEN_RESTRICT rhsIt = rhs + alignedStart; + Scalar *EIGEN_RESTRICT resIt = res + alignedStart; + for (Index i = alignedStart; i < alignedEnd; i += PacketSize) { + Packet A0i = ploadu(a0It); + a0It += PacketSize; + Packet A1i = ploadu(a1It); + a1It += PacketSize; + Packet Bi = ploadu(rhsIt); + rhsIt += PacketSize;// FIXME should be aligned in most cases + Packet Xi = pload(resIt); + + Xi = pcj0.pmadd(A0i, ptmp0, pcj0.pmadd(A1i, ptmp1, Xi)); + ptmp2 = pcj1.pmadd(A0i, Bi, ptmp2); + ptmp3 = pcj1.pmadd(A1i, Bi, ptmp3); + pstore(resIt, Xi); + resIt += PacketSize; + } + for (Index i = alignedEnd; i < endi; i++) { + res[i] += cj0.pmul(A0[i], t0) + cj0.pmul(A1[i], t1); + t2 += cj1.pmul(A0[i], rhs[i]); + t3 += cj1.pmul(A1[i], rhs[i]); + } + + res[j] += alpha * (t2 + predux(ptmp2)); + res[j + 1] += alpha * (t3 + predux(ptmp3)); + } + for (Index j = FirstTriangular ? 0 : bound; j < (FirstTriangular ? bound : size); j++) { + const Scalar *EIGEN_RESTRICT A0 = lhs + j * lhsStride; + + Scalar t1 = cjAlpha * rhs[j]; + Scalar t2(0); + res[j] += cjd.pmul(numext::real(A0[j]), t1); + for (Index i = FirstTriangular ? 0 : j + 1; i < (FirstTriangular ? j : size); i++) { + res[i] += cj0.pmul(A0[i], t1); + t2 += cj1.pmul(A0[i], rhs[i]); + } + res[j] += alpha * t2; } - res[j] += alpha * t2; } -} -} // end namespace internal +}// end namespace internal /*************************************************************************** -* Wrapper to product_selfadjoint_vector -***************************************************************************/ + * Wrapper to product_selfadjoint_vector + ***************************************************************************/ namespace internal { -template -struct selfadjoint_product_impl -{ - typedef typename Product::Scalar Scalar; - - typedef internal::blas_traits LhsBlasTraits; - typedef typename LhsBlasTraits::DirectLinearAccessType ActualLhsType; - typedef typename internal::remove_all::type ActualLhsTypeCleaned; - - typedef internal::blas_traits RhsBlasTraits; - typedef typename RhsBlasTraits::DirectLinearAccessType ActualRhsType; - typedef typename internal::remove_all::type ActualRhsTypeCleaned; - - enum { LhsUpLo = LhsMode&(Upper|Lower) }; - - template - static void run(Dest& dest, const Lhs &a_lhs, const Rhs &a_rhs, const Scalar& alpha) + template struct selfadjoint_product_impl { - typedef typename Dest::Scalar ResScalar; - typedef typename Rhs::Scalar RhsScalar; - typedef Map, EIGEN_PLAIN_ENUM_MIN(AlignedMax,internal::packet_traits::size)> MappedDest; - - eigen_assert(dest.rows()==a_lhs.rows() && dest.cols()==a_rhs.cols()); + typedef typename Product::Scalar Scalar; - typename internal::add_const_on_value_type::type lhs = LhsBlasTraits::extract(a_lhs); - typename internal::add_const_on_value_type::type rhs = RhsBlasTraits::extract(a_rhs); + typedef internal::blas_traits LhsBlasTraits; + typedef typename LhsBlasTraits::DirectLinearAccessType ActualLhsType; + typedef typename internal::remove_all::type ActualLhsTypeCleaned; - Scalar actualAlpha = alpha * LhsBlasTraits::extractScalarFactor(a_lhs) - * RhsBlasTraits::extractScalarFactor(a_rhs); + typedef internal::blas_traits RhsBlasTraits; + typedef typename RhsBlasTraits::DirectLinearAccessType ActualRhsType; + typedef typename internal::remove_all::type ActualRhsTypeCleaned; - enum { - EvalToDest = (Dest::InnerStrideAtCompileTime==1), - UseRhs = (ActualRhsTypeCleaned::InnerStrideAtCompileTime==1) - }; - - internal::gemv_static_vector_if static_dest; - internal::gemv_static_vector_if static_rhs; - - ei_declare_aligned_stack_constructed_variable(ResScalar,actualDestPtr,dest.size(), - EvalToDest ? dest.data() : static_dest.data()); - - ei_declare_aligned_stack_constructed_variable(RhsScalar,actualRhsPtr,rhs.size(), - UseRhs ? const_cast(rhs.data()) : static_rhs.data()); - - if(!EvalToDest) - { - #ifdef EIGEN_DENSE_STORAGE_CTOR_PLUGIN - Index size = dest.size(); - EIGEN_DENSE_STORAGE_CTOR_PLUGIN - #endif - MappedDest(actualDestPtr, dest.size()) = dest; - } - - if(!UseRhs) + enum { LhsUpLo = LhsMode & (Upper | Lower) }; + + template static void run(Dest &dest, const Lhs &a_lhs, const Rhs &a_rhs, const Scalar &alpha) { - #ifdef EIGEN_DENSE_STORAGE_CTOR_PLUGIN - Index size = rhs.size(); - EIGEN_DENSE_STORAGE_CTOR_PLUGIN - #endif - Map(actualRhsPtr, rhs.size()) = rhs; - } - - - internal::selfadjoint_matrix_vector_product::Flags&RowMajorBit) ? RowMajor : ColMajor, - int(LhsUpLo), bool(LhsBlasTraits::NeedToConjugate), bool(RhsBlasTraits::NeedToConjugate)>::run - ( - lhs.rows(), // size - &lhs.coeffRef(0,0), lhs.outerStride(), // lhs info - actualRhsPtr, // rhs info - actualDestPtr, // result info - actualAlpha // scale factor + typedef typename Dest::Scalar ResScalar; + typedef typename Rhs::Scalar RhsScalar; + typedef Map, + EIGEN_PLAIN_ENUM_MIN(AlignedMax, internal::packet_traits::size)> + MappedDest; + + eigen_assert(dest.rows() == a_lhs.rows() && dest.cols() == a_rhs.cols()); + + typename internal::add_const_on_value_type::type lhs = LhsBlasTraits::extract(a_lhs); + typename internal::add_const_on_value_type::type rhs = RhsBlasTraits::extract(a_rhs); + + Scalar actualAlpha = + alpha * LhsBlasTraits::extractScalarFactor(a_lhs) * RhsBlasTraits::extractScalarFactor(a_rhs); + + enum { + EvalToDest = (Dest::InnerStrideAtCompileTime == 1), + UseRhs = (ActualRhsTypeCleaned::InnerStrideAtCompileTime == 1) + }; + + internal::gemv_static_vector_if + static_dest; + internal::gemv_static_vector_if + static_rhs; + + ei_declare_aligned_stack_constructed_variable( + ResScalar, actualDestPtr, dest.size(), EvalToDest ? dest.data() : static_dest.data()); + + ei_declare_aligned_stack_constructed_variable( + RhsScalar, actualRhsPtr, rhs.size(), UseRhs ? const_cast(rhs.data()) : static_rhs.data()); + + if (!EvalToDest) { +#ifdef EIGEN_DENSE_STORAGE_CTOR_PLUGIN + Index size = dest.size(); + EIGEN_DENSE_STORAGE_CTOR_PLUGIN +#endif + MappedDest(actualDestPtr, dest.size()) = dest; + } + + if (!UseRhs) { +#ifdef EIGEN_DENSE_STORAGE_CTOR_PLUGIN + Index size = rhs.size(); + EIGEN_DENSE_STORAGE_CTOR_PLUGIN +#endif + Map(actualRhsPtr, rhs.size()) = rhs; + } + + + internal::selfadjoint_matrix_vector_product::Flags & RowMajorBit) ? RowMajor : ColMajor, + int(LhsUpLo), + bool(LhsBlasTraits::NeedToConjugate), + bool(RhsBlasTraits::NeedToConjugate)>::run(lhs.rows(),// size + &lhs.coeffRef(0, 0), + lhs.outerStride(),// lhs info + actualRhsPtr,// rhs info + actualDestPtr,// result info + actualAlpha// scale factor ); - - if(!EvalToDest) - dest = MappedDest(actualDestPtr, dest.size()); - } -}; -template -struct selfadjoint_product_impl -{ - typedef typename Product::Scalar Scalar; - enum { RhsUpLo = RhsMode&(Upper|Lower) }; + if (!EvalToDest) dest = MappedDest(actualDestPtr, dest.size()); + } + }; - template - static void run(Dest& dest, const Lhs &a_lhs, const Rhs &a_rhs, const Scalar& alpha) + template struct selfadjoint_product_impl { - // let's simply transpose the product - Transpose destT(dest); - selfadjoint_product_impl, int(RhsUpLo)==Upper ? Lower : Upper, false, - Transpose, 0, true>::run(destT, a_rhs.transpose(), a_lhs.transpose(), alpha); - } -}; + typedef typename Product::Scalar Scalar; + enum { RhsUpLo = RhsMode & (Upper | Lower) }; + + template static void run(Dest &dest, const Lhs &a_lhs, const Rhs &a_rhs, const Scalar &alpha) + { + // let's simply transpose the product + Transpose destT(dest); + selfadjoint_product_impl, + int(RhsUpLo) == Upper ? Lower : Upper, + false, + Transpose, + 0, + true>::run(destT, a_rhs.transpose(), a_lhs.transpose(), alpha); + } + }; -} // end namespace internal +}// end namespace internal -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_SELFADJOINT_MATRIX_VECTOR_H +#endif// EIGEN_SELFADJOINT_MATRIX_VECTOR_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/products/SelfadjointMatrixVector_BLAS.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/products/SelfadjointMatrixVector_BLAS.h index 1238345e..0b391983 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/products/SelfadjointMatrixVector_BLAS.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/products/SelfadjointMatrixVector_BLAS.h @@ -33,86 +33,91 @@ #ifndef EIGEN_SELFADJOINT_MATRIX_VECTOR_BLAS_H #define EIGEN_SELFADJOINT_MATRIX_VECTOR_BLAS_H -namespace Eigen { +namespace Eigen { namespace internal { -/********************************************************************** -* This file implements selfadjoint matrix-vector multiplication using BLAS -**********************************************************************/ - -// symv/hemv specialization - -template -struct selfadjoint_matrix_vector_product_symv : - selfadjoint_matrix_vector_product {}; - -#define EIGEN_BLAS_SYMV_SPECIALIZE(Scalar) \ -template \ -struct selfadjoint_matrix_vector_product { \ -static void run( \ - Index size, const Scalar* lhs, Index lhsStride, \ - const Scalar* _rhs, Scalar* res, Scalar alpha) { \ - enum {\ - IsColMajor = StorageOrder==ColMajor \ - }; \ - if (IsColMajor == ConjugateLhs) {\ - selfadjoint_matrix_vector_product::run( \ - size, lhs, lhsStride, _rhs, res, alpha); \ - } else {\ - selfadjoint_matrix_vector_product_symv::run( \ - size, lhs, lhsStride, _rhs, res, alpha); \ - }\ - } \ -}; \ - -EIGEN_BLAS_SYMV_SPECIALIZE(double) -EIGEN_BLAS_SYMV_SPECIALIZE(float) -EIGEN_BLAS_SYMV_SPECIALIZE(dcomplex) -EIGEN_BLAS_SYMV_SPECIALIZE(scomplex) - -#define EIGEN_BLAS_SYMV_SPECIALIZATION(EIGTYPE,BLASTYPE,BLASFUNC) \ -template \ -struct selfadjoint_matrix_vector_product_symv \ -{ \ -typedef Matrix SYMVVector;\ -\ -static void run( \ -Index size, const EIGTYPE* lhs, Index lhsStride, \ -const EIGTYPE* _rhs, EIGTYPE* res, EIGTYPE alpha) \ -{ \ - enum {\ - IsRowMajor = StorageOrder==RowMajor ? 1 : 0, \ - IsLower = UpLo == Lower ? 1 : 0 \ - }; \ - BlasIndex n=convert_index(size), lda=convert_index(lhsStride), incx=1, incy=1; \ - EIGTYPE beta(1); \ - const EIGTYPE *x_ptr; \ - char uplo=(IsRowMajor) ? (IsLower ? 'U' : 'L') : (IsLower ? 'L' : 'U'); \ - SYMVVector x_tmp; \ - if (ConjugateRhs) { \ - Map map_x(_rhs,size,1); \ - x_tmp=map_x.conjugate(); \ - x_ptr=x_tmp.data(); \ - } else x_ptr=_rhs; \ - BLASFUNC(&uplo, &n, (const BLASTYPE*)&numext::real_ref(alpha), (const BLASTYPE*)lhs, &lda, (const BLASTYPE*)x_ptr, &incx, (const BLASTYPE*)&numext::real_ref(beta), (BLASTYPE*)res, &incy); \ -}\ -}; + /********************************************************************** + * This file implements selfadjoint matrix-vector multiplication using BLAS + **********************************************************************/ + + // symv/hemv specialization + + template + struct selfadjoint_matrix_vector_product_symv + : selfadjoint_matrix_vector_product + { + }; + +#define EIGEN_BLAS_SYMV_SPECIALIZE(Scalar) \ + template \ + struct selfadjoint_matrix_vector_product \ + { \ + static void run(Index size, const Scalar *lhs, Index lhsStride, const Scalar *_rhs, Scalar *res, Scalar alpha) \ + { \ + enum { IsColMajor = StorageOrder == ColMajor }; \ + if (IsColMajor == ConjugateLhs) { \ + selfadjoint_matrix_vector_product:: \ + run(size, lhs, lhsStride, _rhs, res, alpha); \ + } else { \ + selfadjoint_matrix_vector_product_symv::run( \ + size, lhs, lhsStride, _rhs, res, alpha); \ + } \ + } \ + }; + + EIGEN_BLAS_SYMV_SPECIALIZE(double) + EIGEN_BLAS_SYMV_SPECIALIZE(float) + EIGEN_BLAS_SYMV_SPECIALIZE(dcomplex) + EIGEN_BLAS_SYMV_SPECIALIZE(scomplex) + +#define EIGEN_BLAS_SYMV_SPECIALIZATION(EIGTYPE, BLASTYPE, BLASFUNC) \ + template \ + struct selfadjoint_matrix_vector_product_symv \ + { \ + typedef Matrix SYMVVector; \ + \ + static void run(Index size, const EIGTYPE *lhs, Index lhsStride, const EIGTYPE *_rhs, EIGTYPE *res, EIGTYPE alpha) \ + { \ + enum { IsRowMajor = StorageOrder == RowMajor ? 1 : 0, IsLower = UpLo == Lower ? 1 : 0 }; \ + BlasIndex n = convert_index(size), lda = convert_index(lhsStride), incx = 1, incy = 1; \ + EIGTYPE beta(1); \ + const EIGTYPE *x_ptr; \ + char uplo = (IsRowMajor) ? (IsLower ? 'U' : 'L') : (IsLower ? 'L' : 'U'); \ + SYMVVector x_tmp; \ + if (ConjugateRhs) { \ + Map map_x(_rhs, size, 1); \ + x_tmp = map_x.conjugate(); \ + x_ptr = x_tmp.data(); \ + } else \ + x_ptr = _rhs; \ + BLASFUNC(&uplo, \ + &n, \ + (const BLASTYPE *)&numext::real_ref(alpha), \ + (const BLASTYPE *)lhs, \ + &lda, \ + (const BLASTYPE *)x_ptr, \ + &incx, \ + (const BLASTYPE *)&numext::real_ref(beta), \ + (BLASTYPE *)res, \ + &incy); \ + } \ + }; #ifdef EIGEN_USE_MKL -EIGEN_BLAS_SYMV_SPECIALIZATION(double, double, dsymv) -EIGEN_BLAS_SYMV_SPECIALIZATION(float, float, ssymv) -EIGEN_BLAS_SYMV_SPECIALIZATION(dcomplex, MKL_Complex16, zhemv) -EIGEN_BLAS_SYMV_SPECIALIZATION(scomplex, MKL_Complex8, chemv) + EIGEN_BLAS_SYMV_SPECIALIZATION(double, double, dsymv) + EIGEN_BLAS_SYMV_SPECIALIZATION(float, float, ssymv) + EIGEN_BLAS_SYMV_SPECIALIZATION(dcomplex, MKL_Complex16, zhemv) + EIGEN_BLAS_SYMV_SPECIALIZATION(scomplex, MKL_Complex8, chemv) #else -EIGEN_BLAS_SYMV_SPECIALIZATION(double, double, dsymv_) -EIGEN_BLAS_SYMV_SPECIALIZATION(float, float, ssymv_) -EIGEN_BLAS_SYMV_SPECIALIZATION(dcomplex, double, zhemv_) -EIGEN_BLAS_SYMV_SPECIALIZATION(scomplex, float, chemv_) + EIGEN_BLAS_SYMV_SPECIALIZATION(double, double, dsymv_) + EIGEN_BLAS_SYMV_SPECIALIZATION(float, float, ssymv_) + EIGEN_BLAS_SYMV_SPECIALIZATION(dcomplex, double, zhemv_) + EIGEN_BLAS_SYMV_SPECIALIZATION(scomplex, float, chemv_) #endif -} // end namespace internal +}// end namespace internal -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_SELFADJOINT_MATRIX_VECTOR_BLAS_H +#endif// EIGEN_SELFADJOINT_MATRIX_VECTOR_BLAS_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/products/SelfadjointProduct.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/products/SelfadjointProduct.h index f038d686..6605bfbb 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/products/SelfadjointProduct.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/products/SelfadjointProduct.h @@ -11,36 +11,39 @@ #define EIGEN_SELFADJOINT_PRODUCT_H /********************************************************************** -* This file implements a self adjoint product: C += A A^T updating only -* half of the selfadjoint matrix C. -* It corresponds to the level 3 SYRK and level 2 SYR Blas routines. -**********************************************************************/ + * This file implements a self adjoint product: C += A A^T updating only + * half of the selfadjoint matrix C. + * It corresponds to the level 3 SYRK and level 2 SYR Blas routines. + **********************************************************************/ -namespace Eigen { +namespace Eigen { template -struct selfadjoint_rank1_update +struct selfadjoint_rank1_update { - static void run(Index size, Scalar* mat, Index stride, const Scalar* vecX, const Scalar* vecY, const Scalar& alpha) + static void run(Index size, Scalar *mat, Index stride, const Scalar *vecX, const Scalar *vecY, const Scalar &alpha) { internal::conj_if cj; - typedef Map > OtherMap; - typedef typename internal::conditional::type ConjLhsType; - for (Index i=0; i >(mat+stride*i+(UpLo==Lower ? i : 0), (UpLo==Lower ? size-i : (i+1))) - += (alpha * cj(vecY[i])) * ConjLhsType(OtherMap(vecX+(UpLo==Lower ? i : 0),UpLo==Lower ? size-i : (i+1))); + typedef Map> OtherMap; + typedef typename internal::conditional::type + ConjLhsType; + for (Index i = 0; i < size; ++i) { + Map>( + mat + stride * i + (UpLo == Lower ? i : 0), (UpLo == Lower ? size - i : (i + 1))) += + (alpha * cj(vecY[i])) + * ConjLhsType(OtherMap(vecX + (UpLo == Lower ? i : 0), UpLo == Lower ? size - i : (i + 1))); } } }; template -struct selfadjoint_rank1_update +struct selfadjoint_rank1_update { - static void run(Index size, Scalar* mat, Index stride, const Scalar* vecX, const Scalar* vecY, const Scalar& alpha) + static void run(Index size, Scalar *mat, Index stride, const Scalar *vecX, const Scalar *vecY, const Scalar &alpha) { - selfadjoint_rank1_update::run(size,mat,stride,vecY,vecX,alpha); + selfadjoint_rank1_update::run( + size, mat, stride, vecY, vecX, alpha); } }; @@ -48,71 +51,100 @@ template -struct selfadjoint_product_selector +struct selfadjoint_product_selector { - static void run(MatrixType& mat, const OtherType& other, const typename MatrixType::Scalar& alpha) + static void run(MatrixType &mat, const OtherType &other, const typename MatrixType::Scalar &alpha) { typedef typename MatrixType::Scalar Scalar; typedef internal::blas_traits OtherBlasTraits; typedef typename OtherBlasTraits::DirectLinearAccessType ActualOtherType; typedef typename internal::remove_all::type _ActualOtherType; - typename internal::add_const_on_value_type::type actualOther = OtherBlasTraits::extract(other.derived()); + typename internal::add_const_on_value_type::type actualOther = + OtherBlasTraits::extract(other.derived()); Scalar actualAlpha = alpha * OtherBlasTraits::extractScalarFactor(other.derived()); enum { - StorageOrder = (internal::traits::Flags&RowMajorBit) ? RowMajor : ColMajor, - UseOtherDirectly = _ActualOtherType::InnerStrideAtCompileTime==1 + StorageOrder = (internal::traits::Flags & RowMajorBit) ? RowMajor : ColMajor, + UseOtherDirectly = _ActualOtherType::InnerStrideAtCompileTime == 1 }; - internal::gemv_static_vector_if static_other; + internal:: + gemv_static_vector_if + static_other; - ei_declare_aligned_stack_constructed_variable(Scalar, actualOtherPtr, other.size(), - (UseOtherDirectly ? const_cast(actualOther.data()) : static_other.data())); - - if(!UseOtherDirectly) + ei_declare_aligned_stack_constructed_variable(Scalar, + actualOtherPtr, + other.size(), + (UseOtherDirectly ? const_cast(actualOther.data()) : static_other.data())); + + if (!UseOtherDirectly) Map(actualOtherPtr, actualOther.size()) = actualOther; - - selfadjoint_rank1_update::IsComplex, - (!OtherBlasTraits::NeedToConjugate) && NumTraits::IsComplex> - ::run(other.size(), mat.data(), mat.outerStride(), actualOtherPtr, actualOtherPtr, actualAlpha); + + selfadjoint_rank1_update::IsComplex, + (!OtherBlasTraits::NeedToConjugate) && NumTraits::IsComplex>::run(other.size(), + mat.data(), + mat.outerStride(), + actualOtherPtr, + actualOtherPtr, + actualAlpha); } }; template -struct selfadjoint_product_selector +struct selfadjoint_product_selector { - static void run(MatrixType& mat, const OtherType& other, const typename MatrixType::Scalar& alpha) + static void run(MatrixType &mat, const OtherType &other, const typename MatrixType::Scalar &alpha) { typedef typename MatrixType::Scalar Scalar; typedef internal::blas_traits OtherBlasTraits; typedef typename OtherBlasTraits::DirectLinearAccessType ActualOtherType; typedef typename internal::remove_all::type _ActualOtherType; - typename internal::add_const_on_value_type::type actualOther = OtherBlasTraits::extract(other.derived()); + typename internal::add_const_on_value_type::type actualOther = + OtherBlasTraits::extract(other.derived()); Scalar actualAlpha = alpha * OtherBlasTraits::extractScalarFactor(other.derived()); enum { - IsRowMajor = (internal::traits::Flags&RowMajorBit) ? 1 : 0, - OtherIsRowMajor = _ActualOtherType::Flags&RowMajorBit ? 1 : 0 + IsRowMajor = (internal::traits::Flags & RowMajorBit) ? 1 : 0, + OtherIsRowMajor = _ActualOtherType::Flags & RowMajorBit ? 1 : 0 }; Index size = mat.cols(); Index depth = actualOther.cols(); - typedef internal::gemm_blocking_space BlockingType; + typedef internal::gemm_blocking_space + BlockingType; BlockingType blocking(size, size, depth, 1, false); internal::general_matrix_matrix_triangular_product::IsComplex, - Scalar, OtherIsRowMajor ? ColMajor : RowMajor, (!OtherBlasTraits::NeedToConjugate) && NumTraits::IsComplex, - IsRowMajor ? RowMajor : ColMajor, UpLo> - ::run(size, depth, - &actualOther.coeffRef(0,0), actualOther.outerStride(), &actualOther.coeffRef(0,0), actualOther.outerStride(), - mat.data(), mat.outerStride(), actualAlpha, blocking); + Scalar, + OtherIsRowMajor ? RowMajor : ColMajor, + OtherBlasTraits::NeedToConjugate && NumTraits::IsComplex, + Scalar, + OtherIsRowMajor ? ColMajor : RowMajor, + (!OtherBlasTraits::NeedToConjugate) && NumTraits::IsComplex, + IsRowMajor ? RowMajor : ColMajor, + UpLo>::run(size, + depth, + &actualOther.coeffRef(0, 0), + actualOther.outerStride(), + &actualOther.coeffRef(0, 0), + actualOther.outerStride(), + mat.data(), + mat.outerStride(), + actualAlpha, + blocking); } }; @@ -120,14 +152,14 @@ struct selfadjoint_product_selector template template -SelfAdjointView& SelfAdjointView -::rankUpdate(const MatrixBase& u, const Scalar& alpha) +SelfAdjointView &SelfAdjointView::rankUpdate(const MatrixBase &u, + const Scalar &alpha) { - selfadjoint_product_selector::run(_expression().const_cast_derived(), u.derived(), alpha); + selfadjoint_product_selector::run(_expression().const_cast_derived(), u.derived(), alpha); return *this; } -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_SELFADJOINT_PRODUCT_H +#endif// EIGEN_SELFADJOINT_PRODUCT_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/products/SelfadjointRank2Update.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/products/SelfadjointRank2Update.h index 2ae36411..40718326 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/products/SelfadjointRank2Update.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/products/SelfadjointRank2Update.h @@ -10,55 +10,56 @@ #ifndef EIGEN_SELFADJOINTRANK2UPTADE_H #define EIGEN_SELFADJOINTRANK2UPTADE_H -namespace Eigen { +namespace Eigen { namespace internal { -/* Optimized selfadjoint matrix += alpha * uv' + conj(alpha)*vu' - * It corresponds to the Level2 syr2 BLAS routine - */ + /* Optimized selfadjoint matrix += alpha * uv' + conj(alpha)*vu' + * It corresponds to the Level2 syr2 BLAS routine + */ -template -struct selfadjoint_rank2_update_selector; + template + struct selfadjoint_rank2_update_selector; -template -struct selfadjoint_rank2_update_selector -{ - static void run(Scalar* mat, Index stride, const UType& u, const VType& v, const Scalar& alpha) + template + struct selfadjoint_rank2_update_selector { - const Index size = u.size(); - for (Index i=0; i >(mat+stride*i+i, size-i) += - (numext::conj(alpha) * numext::conj(u.coeff(i))) * v.tail(size-i) - + (alpha * numext::conj(v.coeff(i))) * u.tail(size-i); + const Index size = u.size(); + for (Index i = 0; i < size; ++i) { + Map>(mat + stride * i + i, size - i) += + (numext::conj(alpha) * numext::conj(u.coeff(i))) * v.tail(size - i) + + (alpha * numext::conj(v.coeff(i))) * u.tail(size - i); + } } - } -}; + }; -template -struct selfadjoint_rank2_update_selector -{ - static void run(Scalar* mat, Index stride, const UType& u, const VType& v, const Scalar& alpha) + template + struct selfadjoint_rank2_update_selector { - const Index size = u.size(); - for (Index i=0; i >(mat+stride*i, i+1) += - (numext::conj(alpha) * numext::conj(u.coeff(i))) * v.head(i+1) - + (alpha * numext::conj(v.coeff(i))) * u.head(i+1); - } -}; + static void run(Scalar *mat, Index stride, const UType &u, const VType &v, const Scalar &alpha) + { + const Index size = u.size(); + for (Index i = 0; i < size; ++i) + Map>(mat + stride * i, i + 1) += + (numext::conj(alpha) * numext::conj(u.coeff(i))) * v.head(i + 1) + + (alpha * numext::conj(v.coeff(i))) * u.head(i + 1); + } + }; -template struct conj_expr_if - : conditional::Scalar>,T> > {}; + template + struct conj_expr_if : conditional::Scalar>, T>> + { + }; -} // end namespace internal +}// end namespace internal template template -SelfAdjointView& SelfAdjointView -::rankUpdate(const MatrixBase& u, const MatrixBase& v, const Scalar& alpha) +SelfAdjointView &SelfAdjointView::rankUpdate(const MatrixBase &u, + const MatrixBase &v, + const Scalar &alpha) { typedef internal::blas_traits UBlasTraits; typedef typename UBlasTraits::DirectLinearAccessType ActualUType; @@ -73,21 +74,28 @@ ::rankUpdate(const MatrixBase& u, const MatrixBase& v, const // If MatrixType is row major, then we use the routine for lower triangular in the upper triangular case and // vice versa, and take the complex conjugate of all coefficients and vector entries. - enum { IsRowMajor = (internal::traits::Flags&RowMajorBit) ? 1 : 0 }; - Scalar actualAlpha = alpha * UBlasTraits::extractScalarFactor(u.derived()) - * numext::conj(VBlasTraits::extractScalarFactor(v.derived())); - if (IsRowMajor) - actualAlpha = numext::conj(actualAlpha); - - typedef typename internal::remove_all::type>::type UType; - typedef typename internal::remove_all::type>::type VType; - internal::selfadjoint_rank2_update_selector - ::run(_expression().const_cast_derived().data(),_expression().outerStride(),UType(actualU),VType(actualV),actualAlpha); + enum { IsRowMajor = (internal::traits::Flags & RowMajorBit) ? 1 : 0 }; + Scalar actualAlpha = + alpha * UBlasTraits::extractScalarFactor(u.derived()) * numext::conj(VBlasTraits::extractScalarFactor(v.derived())); + if (IsRowMajor) actualAlpha = numext::conj(actualAlpha); + + typedef typename internal::remove_all< + typename internal::conj_expr_if::type>::type UType; + typedef typename internal::remove_all< + typename internal::conj_expr_if::type>::type VType; + internal::selfadjoint_rank2_update_selector::run(_expression().const_cast_derived().data(), + _expression().outerStride(), + UType(actualU), + VType(actualV), + actualAlpha); return *this; } -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_SELFADJOINTRANK2UPTADE_H +#endif// EIGEN_SELFADJOINTRANK2UPTADE_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/products/TriangularMatrixMatrix.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/products/TriangularMatrixMatrix.h index f784507e..c90ed34f 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/products/TriangularMatrixMatrix.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/products/TriangularMatrixMatrix.h @@ -10,129 +10,185 @@ #ifndef EIGEN_TRIANGULAR_MATRIX_MATRIX_H #define EIGEN_TRIANGULAR_MATRIX_MATRIX_H -namespace Eigen { +namespace Eigen { namespace internal { -// template -// struct gemm_pack_lhs_triangular -// { -// Matrix::IsComplex && Conjugate> cj; -// const_blas_data_mapper lhs(_lhs,lhsStride); -// int count = 0; -// const int peeled_mc = (rows/mr)*mr; -// for(int i=0; i -struct product_triangular_matrix_matrix; - -template -struct product_triangular_matrix_matrix -{ - static EIGEN_STRONG_INLINE void run( - Index rows, Index cols, Index depth, - const Scalar* lhs, Index lhsStride, - const Scalar* rhs, Index rhsStride, - Scalar* res, Index resStride, - const Scalar& alpha, level3_blocking& blocking) + // template + // struct gemm_pack_lhs_triangular + // { + // Matrix::IsComplex && Conjugate> cj; + // const_blas_data_mapper lhs(_lhs,lhsStride); + // int count = 0; + // const int peeled_mc = (rows/mr)*mr; + // for(int i=0; i + struct product_triangular_matrix_matrix; + + template + struct product_triangular_matrix_matrix { - product_triangular_matrix_matrix - ::run(cols, rows, depth, rhs, rhsStride, lhs, lhsStride, res, resStride, alpha, blocking); - } -}; - -// implements col-major += alpha * op(triangular) * op(general) -template -struct product_triangular_matrix_matrix -{ - - typedef gebp_traits Traits; - enum { - SmallPanelWidth = 2 * EIGEN_PLAIN_ENUM_MAX(Traits::mr,Traits::nr), - IsLower = (Mode&Lower) == Lower, - SetDiag = (Mode&(ZeroDiag|UnitDiag)) ? 0 : 1 + static EIGEN_STRONG_INLINE void run(Index rows, + Index cols, + Index depth, + const Scalar *lhs, + Index lhsStride, + const Scalar *rhs, + Index rhsStride, + Scalar *res, + Index resStride, + const Scalar &alpha, + level3_blocking &blocking) + { + product_triangular_matrix_matrix::run(cols, rows, depth, rhs, rhsStride, lhs, lhsStride, res, resStride, alpha, blocking); + } + }; + + // implements col-major += alpha * op(triangular) * op(general) + template + struct product_triangular_matrix_matrix + { + + typedef gebp_traits Traits; + enum { + SmallPanelWidth = 2 * EIGEN_PLAIN_ENUM_MAX(Traits::mr, Traits::nr), + IsLower = (Mode & Lower) == Lower, + SetDiag = (Mode & (ZeroDiag | UnitDiag)) ? 0 : 1 + }; + + static EIGEN_DONT_INLINE void run(Index _rows, + Index _cols, + Index _depth, + const Scalar *_lhs, + Index lhsStride, + const Scalar *_rhs, + Index rhsStride, + Scalar *res, + Index resStride, + const Scalar &alpha, + level3_blocking &blocking); }; - static EIGEN_DONT_INLINE void run( - Index _rows, Index _cols, Index _depth, - const Scalar* _lhs, Index lhsStride, - const Scalar* _rhs, Index rhsStride, - Scalar* res, Index resStride, - const Scalar& alpha, level3_blocking& blocking); -}; - -template -EIGEN_DONT_INLINE void product_triangular_matrix_matrix::run( - Index _rows, Index _cols, Index _depth, - const Scalar* _lhs, Index lhsStride, - const Scalar* _rhs, Index rhsStride, - Scalar* _res, Index resStride, - const Scalar& alpha, level3_blocking& blocking) + template + EIGEN_DONT_INLINE void product_triangular_matrix_matrix::run(Index _rows, + Index _cols, + Index _depth, + const Scalar *_lhs, + Index lhsStride, + const Scalar *_rhs, + Index rhsStride, + Scalar *_res, + Index resStride, + const Scalar &alpha, + level3_blocking &blocking) { // strip zeros - Index diagSize = (std::min)(_rows,_depth); - Index rows = IsLower ? _rows : diagSize; - Index depth = IsLower ? diagSize : _depth; - Index cols = _cols; - + Index diagSize = (std::min)(_rows, _depth); + Index rows = IsLower ? _rows : diagSize; + Index depth = IsLower ? diagSize : _depth; + Index cols = _cols; + typedef const_blas_data_mapper LhsMapper; typedef const_blas_data_mapper RhsMapper; typedef blas_data_mapper ResMapper; - LhsMapper lhs(_lhs,lhsStride); - RhsMapper rhs(_rhs,rhsStride); + LhsMapper lhs(_lhs, lhsStride); + RhsMapper rhs(_rhs, rhsStride); ResMapper res(_res, resStride); - Index kc = blocking.kc(); // cache block size along the K direction - Index mc = (std::min)(rows,blocking.mc()); // cache block size along the M direction + Index kc = blocking.kc();// cache block size along the K direction + Index mc = (std::min)(rows, blocking.mc());// cache block size along the M direction // The small panel size must not be larger than blocking size. // Usually this should never be the case because SmallPanelWidth^2 is very small // compared to L2 cache size, but let's be safe: - Index panelWidth = (std::min)(Index(SmallPanelWidth),(std::min)(kc,mc)); + Index panelWidth = (std::min)(Index(SmallPanelWidth), (std::min)(kc, mc)); - std::size_t sizeA = kc*mc; - std::size_t sizeB = kc*cols; + std::size_t sizeA = kc * mc; + std::size_t sizeB = kc * cols; ei_declare_aligned_stack_constructed_variable(Scalar, blockA, sizeA, blocking.blockA()); ei_declare_aligned_stack_constructed_variable(Scalar, blockB, sizeB, blocking.blockB()); @@ -143,32 +199,28 @@ EIGEN_DONT_INLINE void product_triangular_matrix_matrix triangularBuffer(a); + Matrix triangularBuffer(a); triangularBuffer.setZero(); - if((Mode&ZeroDiag)==ZeroDiag) + if ((Mode & ZeroDiag) == ZeroDiag) triangularBuffer.diagonal().setZero(); else triangularBuffer.diagonal().setOnes(); gebp_kernel gebp_kernel; gemm_pack_lhs pack_lhs; - gemm_pack_rhs pack_rhs; + gemm_pack_rhs pack_rhs; - for(Index k2=IsLower ? depth : 0; - IsLower ? k2>0 : k2 0 : k2 < depth; IsLower ? k2 -= kc : k2 += kc) { + Index actual_kc = (std::min)(IsLower ? k2 : depth - k2, kc); + Index actual_k2 = IsLower ? k2 - actual_kc : k2; // align blocks with the end of the triangular part for trapezoidal lhs - if((!IsLower)&&(k2rows)) - { - actual_kc = rows-k2; - k2 = k2+actual_kc-kc; + if ((!IsLower) && (k2 < rows) && (k2 + actual_kc > rows)) { + actual_kc = rows - k2; + k2 = k2 + actual_kc - kc; } - pack_rhs(blockB, rhs.getSubMapper(actual_k2,0), actual_kc, cols); + pack_rhs(blockB, rhs.getSubMapper(actual_k2, 0), actual_kc, cols); // the selected lhs's panel has to be split in three different parts: // 1 - the part which is zero => skip it @@ -176,291 +228,342 @@ EIGEN_DONT_INLINE void product_triangular_matrix_matrix GEPP // the block diagonal, if any: - if(IsLower || actual_k2(actual_kc-k1, panelWidth); - Index lengthTarget = IsLower ? actual_kc-k1-actualPanelWidth : k1; - Index startBlock = actual_k2+k1; + for (Index k1 = 0; k1 < actual_kc; k1 += panelWidth) { + Index actualPanelWidth = std::min(actual_kc - k1, panelWidth); + Index lengthTarget = IsLower ? actual_kc - k1 - actualPanelWidth : k1; + Index startBlock = actual_k2 + k1; Index blockBOffset = k1; // => GEBP with the micro triangular block // The trick is to pack this micro block while filling the opposite triangular part with zeros. // To this end we do an extra triangular copy to a small temporary buffer - for (Index k=0;k0) - { - Index startTarget = IsLower ? actual_k2+k1+actualPanelWidth : actual_k2; - - pack_lhs(blockA, lhs.getSubMapper(startTarget,startBlock), actualPanelWidth, lengthTarget); - - gebp_kernel(res.getSubMapper(startTarget, 0), blockA, blockB, - lengthTarget, actualPanelWidth, cols, alpha, - actualPanelWidth, actual_kc, 0, blockBOffset); + if (lengthTarget > 0) { + Index startTarget = IsLower ? actual_k2 + k1 + actualPanelWidth : actual_k2; + + pack_lhs(blockA, lhs.getSubMapper(startTarget, startBlock), actualPanelWidth, lengthTarget); + + gebp_kernel(res.getSubMapper(startTarget, 0), + blockA, + blockB, + lengthTarget, + actualPanelWidth, + cols, + alpha, + actualPanelWidth, + actual_kc, + 0, + blockBOffset); } } } // the part below (lower case) or above (upper case) the diagonal => GEPP { Index start = IsLower ? k2 : 0; - Index end = IsLower ? rows : (std::min)(actual_k2,rows); - for(Index i2=start; i2() - (blockA, lhs.getSubMapper(i2, actual_k2), actual_kc, actual_mc); - - gebp_kernel(res.getSubMapper(i2, 0), blockA, blockB, actual_mc, - actual_kc, cols, alpha, -1, -1, 0, 0); + Index end = IsLower ? rows : (std::min)(actual_k2, rows); + for (Index i2 = start; i2 < end; i2 += mc) { + const Index actual_mc = (std::min)(i2 + mc, end) - i2; + gemm_pack_lhs()( + blockA, lhs.getSubMapper(i2, actual_k2), actual_kc, actual_mc); + + gebp_kernel(res.getSubMapper(i2, 0), blockA, blockB, actual_mc, actual_kc, cols, alpha, -1, -1, 0, 0); } } } } -// implements col-major += alpha * op(general) * op(triangular) -template -struct product_triangular_matrix_matrix -{ - typedef gebp_traits Traits; - enum { - SmallPanelWidth = EIGEN_PLAIN_ENUM_MAX(Traits::mr,Traits::nr), - IsLower = (Mode&Lower) == Lower, - SetDiag = (Mode&(ZeroDiag|UnitDiag)) ? 0 : 1 + // implements col-major += alpha * op(general) * op(triangular) + template + struct product_triangular_matrix_matrix + { + typedef gebp_traits Traits; + enum { + SmallPanelWidth = EIGEN_PLAIN_ENUM_MAX(Traits::mr, Traits::nr), + IsLower = (Mode & Lower) == Lower, + SetDiag = (Mode & (ZeroDiag | UnitDiag)) ? 0 : 1 + }; + + static EIGEN_DONT_INLINE void run(Index _rows, + Index _cols, + Index _depth, + const Scalar *_lhs, + Index lhsStride, + const Scalar *_rhs, + Index rhsStride, + Scalar *res, + Index resStride, + const Scalar &alpha, + level3_blocking &blocking); }; - static EIGEN_DONT_INLINE void run( - Index _rows, Index _cols, Index _depth, - const Scalar* _lhs, Index lhsStride, - const Scalar* _rhs, Index rhsStride, - Scalar* res, Index resStride, - const Scalar& alpha, level3_blocking& blocking); -}; - -template -EIGEN_DONT_INLINE void product_triangular_matrix_matrix::run( - Index _rows, Index _cols, Index _depth, - const Scalar* _lhs, Index lhsStride, - const Scalar* _rhs, Index rhsStride, - Scalar* _res, Index resStride, - const Scalar& alpha, level3_blocking& blocking) + template + EIGEN_DONT_INLINE void product_triangular_matrix_matrix::run(Index _rows, + Index _cols, + Index _depth, + const Scalar *_lhs, + Index lhsStride, + const Scalar *_rhs, + Index rhsStride, + Scalar *_res, + Index resStride, + const Scalar &alpha, + level3_blocking &blocking) { - const Index PacketBytes = packet_traits::size*sizeof(Scalar); + const Index PacketBytes = packet_traits::size * sizeof(Scalar); // strip zeros - Index diagSize = (std::min)(_cols,_depth); - Index rows = _rows; - Index depth = IsLower ? _depth : diagSize; - Index cols = IsLower ? diagSize : _cols; - + Index diagSize = (std::min)(_cols, _depth); + Index rows = _rows; + Index depth = IsLower ? _depth : diagSize; + Index cols = IsLower ? diagSize : _cols; + typedef const_blas_data_mapper LhsMapper; typedef const_blas_data_mapper RhsMapper; typedef blas_data_mapper ResMapper; - LhsMapper lhs(_lhs,lhsStride); - RhsMapper rhs(_rhs,rhsStride); + LhsMapper lhs(_lhs, lhsStride); + RhsMapper rhs(_rhs, rhsStride); ResMapper res(_res, resStride); - Index kc = blocking.kc(); // cache block size along the K direction - Index mc = (std::min)(rows,blocking.mc()); // cache block size along the M direction + Index kc = blocking.kc();// cache block size along the K direction + Index mc = (std::min)(rows, blocking.mc());// cache block size along the M direction - std::size_t sizeA = kc*mc; - std::size_t sizeB = kc*cols+EIGEN_MAX_ALIGN_BYTES/sizeof(Scalar); + std::size_t sizeA = kc * mc; + std::size_t sizeB = kc * cols + EIGEN_MAX_ALIGN_BYTES / sizeof(Scalar); ei_declare_aligned_stack_constructed_variable(Scalar, blockA, sizeA, blocking.blockA()); ei_declare_aligned_stack_constructed_variable(Scalar, blockB, sizeB, blocking.blockB()); internal::constructor_without_unaligned_array_assert a; - Matrix triangularBuffer(a); + Matrix triangularBuffer(a); triangularBuffer.setZero(); - if((Mode&ZeroDiag)==ZeroDiag) + if ((Mode & ZeroDiag) == ZeroDiag) triangularBuffer.diagonal().setZero(); else triangularBuffer.diagonal().setOnes(); gebp_kernel gebp_kernel; gemm_pack_lhs pack_lhs; - gemm_pack_rhs pack_rhs; - gemm_pack_rhs pack_rhs_panel; + gemm_pack_rhs pack_rhs; + gemm_pack_rhs pack_rhs_panel; - for(Index k2=IsLower ? 0 : depth; - IsLower ? k20; - IsLower ? k2+=kc : k2-=kc) - { - Index actual_kc = (std::min)(IsLower ? depth-k2 : k2, kc); - Index actual_k2 = IsLower ? k2 : k2-actual_kc; + for (Index k2 = IsLower ? 0 : depth; IsLower ? k2 < depth : k2 > 0; IsLower ? k2 += kc : k2 -= kc) { + Index actual_kc = (std::min)(IsLower ? depth - k2 : k2, kc); + Index actual_k2 = IsLower ? k2 : k2 - actual_kc; // align blocks with the end of the triangular part for trapezoidal rhs - if(IsLower && (k2cols)) - { - actual_kc = cols-k2; + if (IsLower && (k2 < cols) && (actual_k2 + actual_kc > cols)) { + actual_kc = cols - k2; k2 = actual_k2 + actual_kc - kc; } // remaining size - Index rs = IsLower ? (std::min)(cols,actual_k2) : cols - k2; + Index rs = IsLower ? (std::min)(cols, actual_k2) : cols - k2; // size of the triangular part - Index ts = (IsLower && actual_k2>=cols) ? 0 : actual_kc; + Index ts = (IsLower && actual_k2 >= cols) ? 0 : actual_kc; - Scalar* geb = blockB+ts*ts; - geb = geb + internal::first_aligned(geb,PacketBytes/sizeof(Scalar)); + Scalar *geb = blockB + ts * ts; + geb = geb + internal::first_aligned(geb, PacketBytes / sizeof(Scalar)); - pack_rhs(geb, rhs.getSubMapper(actual_k2,IsLower ? 0 : k2), actual_kc, rs); + pack_rhs(geb, rhs.getSubMapper(actual_k2, IsLower ? 0 : k2), actual_kc, rs); // pack the triangular part of the rhs padding the unrolled blocks with zeros - if(ts>0) - { - for (Index j2=0; j2(actual_kc-j2, SmallPanelWidth); + if (ts > 0) { + for (Index j2 = 0; j2 < actual_kc; j2 += SmallPanelWidth) { + Index actualPanelWidth = std::min(actual_kc - j2, SmallPanelWidth); Index actual_j2 = actual_k2 + j2; - Index panelOffset = IsLower ? j2+actualPanelWidth : 0; - Index panelLength = IsLower ? actual_kc-j2-actualPanelWidth : j2; + Index panelOffset = IsLower ? j2 + actualPanelWidth : 0; + Index panelLength = IsLower ? actual_kc - j2 - actualPanelWidth : j2; // general part - pack_rhs_panel(blockB+j2*actual_kc, - rhs.getSubMapper(actual_k2+panelOffset, actual_j2), - panelLength, actualPanelWidth, - actual_kc, panelOffset); + pack_rhs_panel(blockB + j2 * actual_kc, + rhs.getSubMapper(actual_k2 + panelOffset, actual_j2), + panelLength, + actualPanelWidth, + actual_kc, + panelOffset); // append the triangular part via a temporary buffer - for (Index j=0;j0) - { - for (Index j2=0; j2(actual_kc-j2, SmallPanelWidth); - Index panelLength = IsLower ? actual_kc-j2 : j2+actualPanelWidth; + if (ts > 0) { + for (Index j2 = 0; j2 < actual_kc; j2 += SmallPanelWidth) { + Index actualPanelWidth = std::min(actual_kc - j2, SmallPanelWidth); + Index panelLength = IsLower ? actual_kc - j2 : j2 + actualPanelWidth; Index blockOffset = IsLower ? j2 : 0; gebp_kernel(res.getSubMapper(i2, actual_k2 + j2), - blockA, blockB+j2*actual_kc, - actual_mc, panelLength, actualPanelWidth, - alpha, - actual_kc, actual_kc, // strides - blockOffset, blockOffset);// offsets + blockA, + blockB + j2 * actual_kc, + actual_mc, + panelLength, + actualPanelWidth, + alpha, + actual_kc, + actual_kc,// strides + blockOffset, + blockOffset);// offsets } } - gebp_kernel(res.getSubMapper(i2, IsLower ? 0 : k2), - blockA, geb, actual_mc, actual_kc, rs, - alpha, - -1, -1, 0, 0); + gebp_kernel(res.getSubMapper(i2, IsLower ? 0 : k2), blockA, geb, actual_mc, actual_kc, rs, alpha, -1, -1, 0, 0); } } } -/*************************************************************************** -* Wrapper to product_triangular_matrix_matrix -***************************************************************************/ + /*************************************************************************** + * Wrapper to product_triangular_matrix_matrix + ***************************************************************************/ -} // end namespace internal +}// end namespace internal namespace internal { -template -struct triangular_product_impl -{ - template static void run(Dest& dst, const Lhs &a_lhs, const Rhs &a_rhs, const typename Dest::Scalar& alpha) + template + struct triangular_product_impl { - typedef typename Lhs::Scalar LhsScalar; - typedef typename Rhs::Scalar RhsScalar; - typedef typename Dest::Scalar Scalar; - - typedef internal::blas_traits LhsBlasTraits; - typedef typename LhsBlasTraits::DirectLinearAccessType ActualLhsType; - typedef typename internal::remove_all::type ActualLhsTypeCleaned; - typedef internal::blas_traits RhsBlasTraits; - typedef typename RhsBlasTraits::DirectLinearAccessType ActualRhsType; - typedef typename internal::remove_all::type ActualRhsTypeCleaned; - - typename internal::add_const_on_value_type::type lhs = LhsBlasTraits::extract(a_lhs); - typename internal::add_const_on_value_type::type rhs = RhsBlasTraits::extract(a_rhs); - - LhsScalar lhs_alpha = LhsBlasTraits::extractScalarFactor(a_lhs); - RhsScalar rhs_alpha = RhsBlasTraits::extractScalarFactor(a_rhs); - Scalar actualAlpha = alpha * lhs_alpha * rhs_alpha; - - typedef internal::gemm_blocking_space<(Dest::Flags&RowMajorBit) ? RowMajor : ColMajor,Scalar,Scalar, - Lhs::MaxRowsAtCompileTime, Rhs::MaxColsAtCompileTime, Lhs::MaxColsAtCompileTime,4> BlockingType; - - enum { IsLower = (Mode&Lower) == Lower }; - Index stripedRows = ((!LhsIsTriangular) || (IsLower)) ? lhs.rows() : (std::min)(lhs.rows(),lhs.cols()); - Index stripedCols = ((LhsIsTriangular) || (!IsLower)) ? rhs.cols() : (std::min)(rhs.cols(),rhs.rows()); - Index stripedDepth = LhsIsTriangular ? ((!IsLower) ? lhs.cols() : (std::min)(lhs.cols(),lhs.rows())) - : ((IsLower) ? rhs.rows() : (std::min)(rhs.rows(),rhs.cols())); - - BlockingType blocking(stripedRows, stripedCols, stripedDepth, 1, false); - - internal::product_triangular_matrix_matrix::Flags&RowMajorBit) ? RowMajor : ColMajor, LhsBlasTraits::NeedToConjugate, - (internal::traits::Flags&RowMajorBit) ? RowMajor : ColMajor, RhsBlasTraits::NeedToConjugate, - (internal::traits::Flags&RowMajorBit) ? RowMajor : ColMajor> - ::run( - stripedRows, stripedCols, stripedDepth, // sizes - &lhs.coeffRef(0,0), lhs.outerStride(), // lhs info - &rhs.coeffRef(0,0), rhs.outerStride(), // rhs info - &dst.coeffRef(0,0), dst.outerStride(), // result info - actualAlpha, blocking - ); - - // Apply correction if the diagonal is unit and a scalar factor was nested: - if ((Mode&UnitDiag)==UnitDiag) + template + static void run(Dest &dst, const Lhs &a_lhs, const Rhs &a_rhs, const typename Dest::Scalar &alpha) { - if (LhsIsTriangular && lhs_alpha!=LhsScalar(1)) - { - Index diagSize = (std::min)(lhs.rows(),lhs.cols()); - dst.topRows(diagSize) -= ((lhs_alpha-LhsScalar(1))*a_rhs).topRows(diagSize); - } - else if ((!LhsIsTriangular) && rhs_alpha!=RhsScalar(1)) - { - Index diagSize = (std::min)(rhs.rows(),rhs.cols()); - dst.leftCols(diagSize) -= (rhs_alpha-RhsScalar(1))*a_lhs.leftCols(diagSize); + typedef typename Lhs::Scalar LhsScalar; + typedef typename Rhs::Scalar RhsScalar; + typedef typename Dest::Scalar Scalar; + + typedef internal::blas_traits LhsBlasTraits; + typedef typename LhsBlasTraits::DirectLinearAccessType ActualLhsType; + typedef typename internal::remove_all::type ActualLhsTypeCleaned; + typedef internal::blas_traits RhsBlasTraits; + typedef typename RhsBlasTraits::DirectLinearAccessType ActualRhsType; + typedef typename internal::remove_all::type ActualRhsTypeCleaned; + + typename internal::add_const_on_value_type::type lhs = LhsBlasTraits::extract(a_lhs); + typename internal::add_const_on_value_type::type rhs = RhsBlasTraits::extract(a_rhs); + + LhsScalar lhs_alpha = LhsBlasTraits::extractScalarFactor(a_lhs); + RhsScalar rhs_alpha = RhsBlasTraits::extractScalarFactor(a_rhs); + Scalar actualAlpha = alpha * lhs_alpha * rhs_alpha; + + typedef internal::gemm_blocking_space<(Dest::Flags & RowMajorBit) ? RowMajor : ColMajor, + Scalar, + Scalar, + Lhs::MaxRowsAtCompileTime, + Rhs::MaxColsAtCompileTime, + Lhs::MaxColsAtCompileTime, + 4> + BlockingType; + + enum { IsLower = (Mode & Lower) == Lower }; + Index stripedRows = ((!LhsIsTriangular) || (IsLower)) ? lhs.rows() : (std::min)(lhs.rows(), lhs.cols()); + Index stripedCols = ((LhsIsTriangular) || (!IsLower)) ? rhs.cols() : (std::min)(rhs.cols(), rhs.rows()); + Index stripedDepth = LhsIsTriangular ? ((!IsLower) ? lhs.cols() : (std::min)(lhs.cols(), lhs.rows())) + : ((IsLower) ? rhs.rows() : (std::min)(rhs.rows(), rhs.cols())); + + BlockingType blocking(stripedRows, stripedCols, stripedDepth, 1, false); + + internal::product_triangular_matrix_matrix::Flags & RowMajorBit) ? RowMajor : ColMajor, + LhsBlasTraits::NeedToConjugate, + (internal::traits::Flags & RowMajorBit) ? RowMajor : ColMajor, + RhsBlasTraits::NeedToConjugate, + (internal::traits::Flags & RowMajorBit) ? RowMajor : ColMajor>::run(stripedRows, + stripedCols, + stripedDepth,// sizes + &lhs.coeffRef(0, 0), + lhs.outerStride(),// lhs info + &rhs.coeffRef(0, 0), + rhs.outerStride(),// rhs info + &dst.coeffRef(0, 0), + dst.outerStride(),// result info + actualAlpha, + blocking); + + // Apply correction if the diagonal is unit and a scalar factor was nested: + if ((Mode & UnitDiag) == UnitDiag) { + if (LhsIsTriangular && lhs_alpha != LhsScalar(1)) { + Index diagSize = (std::min)(lhs.rows(), lhs.cols()); + dst.topRows(diagSize) -= ((lhs_alpha - LhsScalar(1)) * a_rhs).topRows(diagSize); + } else if ((!LhsIsTriangular) && rhs_alpha != RhsScalar(1)) { + Index diagSize = (std::min)(rhs.rows(), rhs.cols()); + dst.leftCols(diagSize) -= (rhs_alpha - RhsScalar(1)) * a_lhs.leftCols(diagSize); + } } } - } -}; + }; -} // end namespace internal +}// end namespace internal -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_TRIANGULAR_MATRIX_MATRIX_H +#endif// EIGEN_TRIANGULAR_MATRIX_MATRIX_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/products/TriangularMatrixMatrix_BLAS.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/products/TriangularMatrixMatrix_BLAS.h index a25197ab..ca92bd00 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/products/TriangularMatrixMatrix_BLAS.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/products/TriangularMatrixMatrix_BLAS.h @@ -33,283 +33,420 @@ #ifndef EIGEN_TRIANGULAR_MATRIX_MATRIX_BLAS_H #define EIGEN_TRIANGULAR_MATRIX_MATRIX_BLAS_H -namespace Eigen { +namespace Eigen { namespace internal { -template -struct product_triangular_matrix_matrix_trmm : - product_triangular_matrix_matrix {}; + template + struct product_triangular_matrix_matrix_trmm + : product_triangular_matrix_matrix + { + }; // try to go to BLAS specialization -#define EIGEN_BLAS_TRMM_SPECIALIZE(Scalar, LhsIsTriangular) \ -template \ -struct product_triangular_matrix_matrix { \ - static inline void run(Index _rows, Index _cols, Index _depth, const Scalar* _lhs, Index lhsStride,\ - const Scalar* _rhs, Index rhsStride, Scalar* res, Index resStride, Scalar alpha, level3_blocking& blocking) { \ - product_triangular_matrix_matrix_trmm::run( \ - _rows, _cols, _depth, _lhs, lhsStride, _rhs, rhsStride, res, resStride, alpha, blocking); \ - } \ -}; +#define EIGEN_BLAS_TRMM_SPECIALIZE(Scalar, LhsIsTriangular) \ + template \ + struct product_triangular_matrix_matrix \ + { \ + static inline void run(Index _rows, \ + Index _cols, \ + Index _depth, \ + const Scalar *_lhs, \ + Index lhsStride, \ + const Scalar *_rhs, \ + Index rhsStride, \ + Scalar *res, \ + Index resStride, \ + Scalar alpha, \ + level3_blocking &blocking) \ + { \ + product_triangular_matrix_matrix_trmm::run(_rows, _cols, _depth, _lhs, lhsStride, _rhs, rhsStride, res, resStride, alpha, blocking); \ + } \ + }; -EIGEN_BLAS_TRMM_SPECIALIZE(double, true) -EIGEN_BLAS_TRMM_SPECIALIZE(double, false) -EIGEN_BLAS_TRMM_SPECIALIZE(dcomplex, true) -EIGEN_BLAS_TRMM_SPECIALIZE(dcomplex, false) -EIGEN_BLAS_TRMM_SPECIALIZE(float, true) -EIGEN_BLAS_TRMM_SPECIALIZE(float, false) -EIGEN_BLAS_TRMM_SPECIALIZE(scomplex, true) -EIGEN_BLAS_TRMM_SPECIALIZE(scomplex, false) + EIGEN_BLAS_TRMM_SPECIALIZE(double, true) + EIGEN_BLAS_TRMM_SPECIALIZE(double, false) + EIGEN_BLAS_TRMM_SPECIALIZE(dcomplex, true) + EIGEN_BLAS_TRMM_SPECIALIZE(dcomplex, false) + EIGEN_BLAS_TRMM_SPECIALIZE(float, true) + EIGEN_BLAS_TRMM_SPECIALIZE(float, false) + EIGEN_BLAS_TRMM_SPECIALIZE(scomplex, true) + EIGEN_BLAS_TRMM_SPECIALIZE(scomplex, false) // implements col-major += alpha * op(triangular) * op(general) -#define EIGEN_BLAS_TRMM_L(EIGTYPE, BLASTYPE, EIGPREFIX, BLASFUNC) \ -template \ -struct product_triangular_matrix_matrix_trmm \ -{ \ - enum { \ - IsLower = (Mode&Lower) == Lower, \ - SetDiag = (Mode&(ZeroDiag|UnitDiag)) ? 0 : 1, \ - IsUnitDiag = (Mode&UnitDiag) ? 1 : 0, \ - IsZeroDiag = (Mode&ZeroDiag) ? 1 : 0, \ - LowUp = IsLower ? Lower : Upper, \ - conjA = ((LhsStorageOrder==ColMajor) && ConjugateLhs) ? 1 : 0 \ - }; \ -\ - static void run( \ - Index _rows, Index _cols, Index _depth, \ - const EIGTYPE* _lhs, Index lhsStride, \ - const EIGTYPE* _rhs, Index rhsStride, \ - EIGTYPE* res, Index resStride, \ - EIGTYPE alpha, level3_blocking& blocking) \ - { \ - Index diagSize = (std::min)(_rows,_depth); \ - Index rows = IsLower ? _rows : diagSize; \ - Index depth = IsLower ? diagSize : _depth; \ - Index cols = _cols; \ -\ - typedef Matrix MatrixLhs; \ - typedef Matrix MatrixRhs; \ -\ -/* Non-square case - doesn't fit to BLAS ?TRMM. Fall to default triangular product or call BLAS ?GEMM*/ \ - if (rows != depth) { \ -\ - /* FIXME handle mkl_domain_get_max_threads */ \ - /*int nthr = mkl_domain_get_max_threads(EIGEN_BLAS_DOMAIN_BLAS);*/ int nthr = 1;\ -\ - if (((nthr==1) && (((std::max)(rows,depth)-diagSize)/(double)diagSize < 0.5))) { \ - /* Most likely no benefit to call TRMM or GEMM from BLAS */ \ - product_triangular_matrix_matrix::run( \ - _rows, _cols, _depth, _lhs, lhsStride, _rhs, rhsStride, res, resStride, alpha, blocking); \ - /*std::cout << "TRMM_L: A is not square! Go to Eigen TRMM implementation!\n";*/ \ - } else { \ - /* Make sense to call GEMM */ \ - Map > lhsMap(_lhs,rows,depth,OuterStride<>(lhsStride)); \ - MatrixLhs aa_tmp=lhsMap.template triangularView(); \ - BlasIndex aStride = convert_index(aa_tmp.outerStride()); \ - gemm_blocking_space gemm_blocking(_rows,_cols,_depth, 1, true); \ - general_matrix_matrix_product::run( \ - rows, cols, depth, aa_tmp.data(), aStride, _rhs, rhsStride, res, resStride, alpha, gemm_blocking, 0); \ -\ - /*std::cout << "TRMM_L: A is not square! Go to BLAS GEMM implementation! " << nthr<<" \n";*/ \ - } \ - return; \ - } \ - char side = 'L', transa, uplo, diag = 'N'; \ - EIGTYPE *b; \ - const EIGTYPE *a; \ - BlasIndex m, n, lda, ldb; \ -\ -/* Set m, n */ \ - m = convert_index(diagSize); \ - n = convert_index(cols); \ -\ -/* Set trans */ \ - transa = (LhsStorageOrder==RowMajor) ? ((ConjugateLhs) ? 'C' : 'T') : 'N'; \ -\ -/* Set b, ldb */ \ - Map > rhs(_rhs,depth,cols,OuterStride<>(rhsStride)); \ - MatrixX##EIGPREFIX b_tmp; \ -\ - if (ConjugateRhs) b_tmp = rhs.conjugate(); else b_tmp = rhs; \ - b = b_tmp.data(); \ - ldb = convert_index(b_tmp.outerStride()); \ -\ -/* Set uplo */ \ - uplo = IsLower ? 'L' : 'U'; \ - if (LhsStorageOrder==RowMajor) uplo = (uplo == 'L') ? 'U' : 'L'; \ -/* Set a, lda */ \ - Map > lhs(_lhs,rows,depth,OuterStride<>(lhsStride)); \ - MatrixLhs a_tmp; \ -\ - if ((conjA!=0) || (SetDiag==0)) { \ - if (conjA) a_tmp = lhs.conjugate(); else a_tmp = lhs; \ - if (IsZeroDiag) \ - a_tmp.diagonal().setZero(); \ - else if (IsUnitDiag) \ - a_tmp.diagonal().setOnes();\ - a = a_tmp.data(); \ - lda = convert_index(a_tmp.outerStride()); \ - } else { \ - a = _lhs; \ - lda = convert_index(lhsStride); \ - } \ - /*std::cout << "TRMM_L: A is square! Go to BLAS TRMM implementation! \n";*/ \ -/* call ?trmm*/ \ - BLASFUNC(&side, &uplo, &transa, &diag, &m, &n, (const BLASTYPE*)&numext::real_ref(alpha), (const BLASTYPE*)a, &lda, (BLASTYPE*)b, &ldb); \ -\ -/* Add op(a_triangular)*b into res*/ \ - Map > res_tmp(res,rows,cols,OuterStride<>(resStride)); \ - res_tmp=res_tmp+b_tmp; \ - } \ -}; +#define EIGEN_BLAS_TRMM_L(EIGTYPE, BLASTYPE, EIGPREFIX, BLASFUNC) \ + template \ + struct product_triangular_matrix_matrix_trmm \ + { \ + enum { \ + IsLower = (Mode & Lower) == Lower, \ + SetDiag = (Mode & (ZeroDiag | UnitDiag)) ? 0 : 1, \ + IsUnitDiag = (Mode & UnitDiag) ? 1 : 0, \ + IsZeroDiag = (Mode & ZeroDiag) ? 1 : 0, \ + LowUp = IsLower ? Lower : Upper, \ + conjA = ((LhsStorageOrder == ColMajor) && ConjugateLhs) ? 1 : 0 \ + }; \ + \ + static void run(Index _rows, \ + Index _cols, \ + Index _depth, \ + const EIGTYPE *_lhs, \ + Index lhsStride, \ + const EIGTYPE *_rhs, \ + Index rhsStride, \ + EIGTYPE *res, \ + Index resStride, \ + EIGTYPE alpha, \ + level3_blocking &blocking) \ + { \ + Index diagSize = (std::min)(_rows, _depth); \ + Index rows = IsLower ? _rows : diagSize; \ + Index depth = IsLower ? diagSize : _depth; \ + Index cols = _cols; \ + \ + typedef Matrix MatrixLhs; \ + typedef Matrix MatrixRhs; \ + \ + /* Non-square case - doesn't fit to BLAS ?TRMM. Fall to default triangular product or call BLAS ?GEMM*/ \ + if (rows != depth) { \ + \ + /* FIXME handle mkl_domain_get_max_threads */ \ + /*int nthr = mkl_domain_get_max_threads(EIGEN_BLAS_DOMAIN_BLAS);*/ int nthr = 1; \ + \ + if (((nthr == 1) && (((std::max)(rows, depth) - diagSize) / (double)diagSize < 0.5))) { \ + /* Most likely no benefit to call TRMM or GEMM from BLAS */ \ + product_triangular_matrix_matrix::run(_rows, _cols, _depth, _lhs, lhsStride, _rhs, rhsStride, res, resStride, alpha, blocking); \ + /*std::cout << "TRMM_L: A is not square! Go to Eigen TRMM implementation!\n";*/ \ + } else { \ + /* Make sense to call GEMM */ \ + Map> lhsMap(_lhs, rows, depth, OuterStride<>(lhsStride)); \ + MatrixLhs aa_tmp = lhsMap.template triangularView(); \ + BlasIndex aStride = convert_index(aa_tmp.outerStride()); \ + gemm_blocking_space gemm_blocking( \ + _rows, _cols, _depth, 1, true); \ + general_matrix_matrix_product::run(rows, \ + cols, \ + depth, \ + aa_tmp.data(), \ + aStride, \ + _rhs, \ + rhsStride, \ + res, \ + resStride, \ + alpha, \ + gemm_blocking, \ + 0); \ + \ + /*std::cout << "TRMM_L: A is not square! Go to BLAS GEMM implementation! " << nthr<<" \n";*/ \ + } \ + return; \ + } \ + char side = 'L', transa, uplo, diag = 'N'; \ + EIGTYPE *b; \ + const EIGTYPE *a; \ + BlasIndex m, n, lda, ldb; \ + \ + /* Set m, n */ \ + m = convert_index(diagSize); \ + n = convert_index(cols); \ + \ + /* Set trans */ \ + transa = (LhsStorageOrder == RowMajor) ? ((ConjugateLhs) ? 'C' : 'T') : 'N'; \ + \ + /* Set b, ldb */ \ + Map> rhs(_rhs, depth, cols, OuterStride<>(rhsStride)); \ + MatrixX##EIGPREFIX b_tmp; \ + \ + if (ConjugateRhs) \ + b_tmp = rhs.conjugate(); \ + else \ + b_tmp = rhs; \ + b = b_tmp.data(); \ + ldb = convert_index(b_tmp.outerStride()); \ + \ + /* Set uplo */ \ + uplo = IsLower ? 'L' : 'U'; \ + if (LhsStorageOrder == RowMajor) uplo = (uplo == 'L') ? 'U' : 'L'; \ + /* Set a, lda */ \ + Map> lhs(_lhs, rows, depth, OuterStride<>(lhsStride)); \ + MatrixLhs a_tmp; \ + \ + if ((conjA != 0) || (SetDiag == 0)) { \ + if (conjA) \ + a_tmp = lhs.conjugate(); \ + else \ + a_tmp = lhs; \ + if (IsZeroDiag) \ + a_tmp.diagonal().setZero(); \ + else if (IsUnitDiag) \ + a_tmp.diagonal().setOnes(); \ + a = a_tmp.data(); \ + lda = convert_index(a_tmp.outerStride()); \ + } else { \ + a = _lhs; \ + lda = convert_index(lhsStride); \ + } \ + /*std::cout << "TRMM_L: A is square! Go to BLAS TRMM implementation! \n";*/ \ + /* call ?trmm*/ \ + BLASFUNC(&side, \ + &uplo, \ + &transa, \ + &diag, \ + &m, \ + &n, \ + (const BLASTYPE *)&numext::real_ref(alpha), \ + (const BLASTYPE *)a, \ + &lda, \ + (BLASTYPE *)b, \ + &ldb); \ + \ + /* Add op(a_triangular)*b into res*/ \ + Map> res_tmp(res, rows, cols, OuterStride<>(resStride)); \ + res_tmp = res_tmp + b_tmp; \ + } \ + }; #ifdef EIGEN_USE_MKL -EIGEN_BLAS_TRMM_L(double, double, d, dtrmm) -EIGEN_BLAS_TRMM_L(dcomplex, MKL_Complex16, cd, ztrmm) -EIGEN_BLAS_TRMM_L(float, float, f, strmm) -EIGEN_BLAS_TRMM_L(scomplex, MKL_Complex8, cf, ctrmm) + EIGEN_BLAS_TRMM_L(double, double, d, dtrmm) + EIGEN_BLAS_TRMM_L(dcomplex, MKL_Complex16, cd, ztrmm) + EIGEN_BLAS_TRMM_L(float, float, f, strmm) + EIGEN_BLAS_TRMM_L(scomplex, MKL_Complex8, cf, ctrmm) #else -EIGEN_BLAS_TRMM_L(double, double, d, dtrmm_) -EIGEN_BLAS_TRMM_L(dcomplex, double, cd, ztrmm_) -EIGEN_BLAS_TRMM_L(float, float, f, strmm_) -EIGEN_BLAS_TRMM_L(scomplex, float, cf, ctrmm_) + EIGEN_BLAS_TRMM_L(double, double, d, dtrmm_) + EIGEN_BLAS_TRMM_L(dcomplex, double, cd, ztrmm_) + EIGEN_BLAS_TRMM_L(float, float, f, strmm_) + EIGEN_BLAS_TRMM_L(scomplex, float, cf, ctrmm_) #endif // implements col-major += alpha * op(general) * op(triangular) -#define EIGEN_BLAS_TRMM_R(EIGTYPE, BLASTYPE, EIGPREFIX, BLASFUNC) \ -template \ -struct product_triangular_matrix_matrix_trmm \ -{ \ - enum { \ - IsLower = (Mode&Lower) == Lower, \ - SetDiag = (Mode&(ZeroDiag|UnitDiag)) ? 0 : 1, \ - IsUnitDiag = (Mode&UnitDiag) ? 1 : 0, \ - IsZeroDiag = (Mode&ZeroDiag) ? 1 : 0, \ - LowUp = IsLower ? Lower : Upper, \ - conjA = ((RhsStorageOrder==ColMajor) && ConjugateRhs) ? 1 : 0 \ - }; \ -\ - static void run( \ - Index _rows, Index _cols, Index _depth, \ - const EIGTYPE* _lhs, Index lhsStride, \ - const EIGTYPE* _rhs, Index rhsStride, \ - EIGTYPE* res, Index resStride, \ - EIGTYPE alpha, level3_blocking& blocking) \ - { \ - Index diagSize = (std::min)(_cols,_depth); \ - Index rows = _rows; \ - Index depth = IsLower ? _depth : diagSize; \ - Index cols = IsLower ? diagSize : _cols; \ -\ - typedef Matrix MatrixLhs; \ - typedef Matrix MatrixRhs; \ -\ -/* Non-square case - doesn't fit to BLAS ?TRMM. Fall to default triangular product or call BLAS ?GEMM*/ \ - if (cols != depth) { \ -\ - int nthr = 1 /*mkl_domain_get_max_threads(EIGEN_BLAS_DOMAIN_BLAS)*/; \ -\ - if ((nthr==1) && (((std::max)(cols,depth)-diagSize)/(double)diagSize < 0.5)) { \ - /* Most likely no benefit to call TRMM or GEMM from BLAS*/ \ - product_triangular_matrix_matrix::run( \ - _rows, _cols, _depth, _lhs, lhsStride, _rhs, rhsStride, res, resStride, alpha, blocking); \ - /*std::cout << "TRMM_R: A is not square! Go to Eigen TRMM implementation!\n";*/ \ - } else { \ - /* Make sense to call GEMM */ \ - Map > rhsMap(_rhs,depth,cols, OuterStride<>(rhsStride)); \ - MatrixRhs aa_tmp=rhsMap.template triangularView(); \ - BlasIndex aStride = convert_index(aa_tmp.outerStride()); \ - gemm_blocking_space gemm_blocking(_rows,_cols,_depth, 1, true); \ - general_matrix_matrix_product::run( \ - rows, cols, depth, _lhs, lhsStride, aa_tmp.data(), aStride, res, resStride, alpha, gemm_blocking, 0); \ -\ - /*std::cout << "TRMM_R: A is not square! Go to BLAS GEMM implementation! " << nthr<<" \n";*/ \ - } \ - return; \ - } \ - char side = 'R', transa, uplo, diag = 'N'; \ - EIGTYPE *b; \ - const EIGTYPE *a; \ - BlasIndex m, n, lda, ldb; \ -\ -/* Set m, n */ \ - m = convert_index(rows); \ - n = convert_index(diagSize); \ -\ -/* Set trans */ \ - transa = (RhsStorageOrder==RowMajor) ? ((ConjugateRhs) ? 'C' : 'T') : 'N'; \ -\ -/* Set b, ldb */ \ - Map > lhs(_lhs,rows,depth,OuterStride<>(lhsStride)); \ - MatrixX##EIGPREFIX b_tmp; \ -\ - if (ConjugateLhs) b_tmp = lhs.conjugate(); else b_tmp = lhs; \ - b = b_tmp.data(); \ - ldb = convert_index(b_tmp.outerStride()); \ -\ -/* Set uplo */ \ - uplo = IsLower ? 'L' : 'U'; \ - if (RhsStorageOrder==RowMajor) uplo = (uplo == 'L') ? 'U' : 'L'; \ -/* Set a, lda */ \ - Map > rhs(_rhs,depth,cols, OuterStride<>(rhsStride)); \ - MatrixRhs a_tmp; \ -\ - if ((conjA!=0) || (SetDiag==0)) { \ - if (conjA) a_tmp = rhs.conjugate(); else a_tmp = rhs; \ - if (IsZeroDiag) \ - a_tmp.diagonal().setZero(); \ - else if (IsUnitDiag) \ - a_tmp.diagonal().setOnes();\ - a = a_tmp.data(); \ - lda = convert_index(a_tmp.outerStride()); \ - } else { \ - a = _rhs; \ - lda = convert_index(rhsStride); \ - } \ - /*std::cout << "TRMM_R: A is square! Go to BLAS TRMM implementation! \n";*/ \ -/* call ?trmm*/ \ - BLASFUNC(&side, &uplo, &transa, &diag, &m, &n, (const BLASTYPE*)&numext::real_ref(alpha), (const BLASTYPE*)a, &lda, (BLASTYPE*)b, &ldb); \ -\ -/* Add op(a_triangular)*b into res*/ \ - Map > res_tmp(res,rows,cols,OuterStride<>(resStride)); \ - res_tmp=res_tmp+b_tmp; \ - } \ -}; +#define EIGEN_BLAS_TRMM_R(EIGTYPE, BLASTYPE, EIGPREFIX, BLASFUNC) \ + template \ + struct product_triangular_matrix_matrix_trmm \ + { \ + enum { \ + IsLower = (Mode & Lower) == Lower, \ + SetDiag = (Mode & (ZeroDiag | UnitDiag)) ? 0 : 1, \ + IsUnitDiag = (Mode & UnitDiag) ? 1 : 0, \ + IsZeroDiag = (Mode & ZeroDiag) ? 1 : 0, \ + LowUp = IsLower ? Lower : Upper, \ + conjA = ((RhsStorageOrder == ColMajor) && ConjugateRhs) ? 1 : 0 \ + }; \ + \ + static void run(Index _rows, \ + Index _cols, \ + Index _depth, \ + const EIGTYPE *_lhs, \ + Index lhsStride, \ + const EIGTYPE *_rhs, \ + Index rhsStride, \ + EIGTYPE *res, \ + Index resStride, \ + EIGTYPE alpha, \ + level3_blocking &blocking) \ + { \ + Index diagSize = (std::min)(_cols, _depth); \ + Index rows = _rows; \ + Index depth = IsLower ? _depth : diagSize; \ + Index cols = IsLower ? diagSize : _cols; \ + \ + typedef Matrix MatrixLhs; \ + typedef Matrix MatrixRhs; \ + \ + /* Non-square case - doesn't fit to BLAS ?TRMM. Fall to default triangular product or call BLAS ?GEMM*/ \ + if (cols != depth) { \ + \ + int nthr = 1 /*mkl_domain_get_max_threads(EIGEN_BLAS_DOMAIN_BLAS)*/; \ + \ + if ((nthr == 1) && (((std::max)(cols, depth) - diagSize) / (double)diagSize < 0.5)) { \ + /* Most likely no benefit to call TRMM or GEMM from BLAS*/ \ + product_triangular_matrix_matrix::run(_rows, _cols, _depth, _lhs, lhsStride, _rhs, rhsStride, res, resStride, alpha, blocking); \ + /*std::cout << "TRMM_R: A is not square! Go to Eigen TRMM implementation!\n";*/ \ + } else { \ + /* Make sense to call GEMM */ \ + Map> rhsMap(_rhs, depth, cols, OuterStride<>(rhsStride)); \ + MatrixRhs aa_tmp = rhsMap.template triangularView(); \ + BlasIndex aStride = convert_index(aa_tmp.outerStride()); \ + gemm_blocking_space gemm_blocking( \ + _rows, _cols, _depth, 1, true); \ + general_matrix_matrix_product::run(rows, \ + cols, \ + depth, \ + _lhs, \ + lhsStride, \ + aa_tmp.data(), \ + aStride, \ + res, \ + resStride, \ + alpha, \ + gemm_blocking, \ + 0); \ + \ + /*std::cout << "TRMM_R: A is not square! Go to BLAS GEMM implementation! " << nthr<<" \n";*/ \ + } \ + return; \ + } \ + char side = 'R', transa, uplo, diag = 'N'; \ + EIGTYPE *b; \ + const EIGTYPE *a; \ + BlasIndex m, n, lda, ldb; \ + \ + /* Set m, n */ \ + m = convert_index(rows); \ + n = convert_index(diagSize); \ + \ + /* Set trans */ \ + transa = (RhsStorageOrder == RowMajor) ? ((ConjugateRhs) ? 'C' : 'T') : 'N'; \ + \ + /* Set b, ldb */ \ + Map> lhs(_lhs, rows, depth, OuterStride<>(lhsStride)); \ + MatrixX##EIGPREFIX b_tmp; \ + \ + if (ConjugateLhs) \ + b_tmp = lhs.conjugate(); \ + else \ + b_tmp = lhs; \ + b = b_tmp.data(); \ + ldb = convert_index(b_tmp.outerStride()); \ + \ + /* Set uplo */ \ + uplo = IsLower ? 'L' : 'U'; \ + if (RhsStorageOrder == RowMajor) uplo = (uplo == 'L') ? 'U' : 'L'; \ + /* Set a, lda */ \ + Map> rhs(_rhs, depth, cols, OuterStride<>(rhsStride)); \ + MatrixRhs a_tmp; \ + \ + if ((conjA != 0) || (SetDiag == 0)) { \ + if (conjA) \ + a_tmp = rhs.conjugate(); \ + else \ + a_tmp = rhs; \ + if (IsZeroDiag) \ + a_tmp.diagonal().setZero(); \ + else if (IsUnitDiag) \ + a_tmp.diagonal().setOnes(); \ + a = a_tmp.data(); \ + lda = convert_index(a_tmp.outerStride()); \ + } else { \ + a = _rhs; \ + lda = convert_index(rhsStride); \ + } \ + /*std::cout << "TRMM_R: A is square! Go to BLAS TRMM implementation! \n";*/ \ + /* call ?trmm*/ \ + BLASFUNC(&side, \ + &uplo, \ + &transa, \ + &diag, \ + &m, \ + &n, \ + (const BLASTYPE *)&numext::real_ref(alpha), \ + (const BLASTYPE *)a, \ + &lda, \ + (BLASTYPE *)b, \ + &ldb); \ + \ + /* Add op(a_triangular)*b into res*/ \ + Map> res_tmp(res, rows, cols, OuterStride<>(resStride)); \ + res_tmp = res_tmp + b_tmp; \ + } \ + }; #ifdef EIGEN_USE_MKL -EIGEN_BLAS_TRMM_R(double, double, d, dtrmm) -EIGEN_BLAS_TRMM_R(dcomplex, MKL_Complex16, cd, ztrmm) -EIGEN_BLAS_TRMM_R(float, float, f, strmm) -EIGEN_BLAS_TRMM_R(scomplex, MKL_Complex8, cf, ctrmm) + EIGEN_BLAS_TRMM_R(double, double, d, dtrmm) + EIGEN_BLAS_TRMM_R(dcomplex, MKL_Complex16, cd, ztrmm) + EIGEN_BLAS_TRMM_R(float, float, f, strmm) + EIGEN_BLAS_TRMM_R(scomplex, MKL_Complex8, cf, ctrmm) #else -EIGEN_BLAS_TRMM_R(double, double, d, dtrmm_) -EIGEN_BLAS_TRMM_R(dcomplex, double, cd, ztrmm_) -EIGEN_BLAS_TRMM_R(float, float, f, strmm_) -EIGEN_BLAS_TRMM_R(scomplex, float, cf, ctrmm_) + EIGEN_BLAS_TRMM_R(double, double, d, dtrmm_) + EIGEN_BLAS_TRMM_R(dcomplex, double, cd, ztrmm_) + EIGEN_BLAS_TRMM_R(float, float, f, strmm_) + EIGEN_BLAS_TRMM_R(scomplex, float, cf, ctrmm_) #endif -} // end namespace internal +}// end namespace internal -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_TRIANGULAR_MATRIX_MATRIX_BLAS_H +#endif// EIGEN_TRIANGULAR_MATRIX_MATRIX_BLAS_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/products/TriangularMatrixVector.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/products/TriangularMatrixVector.h index 76bfa159..1c5cb4ab 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/products/TriangularMatrixVector.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/products/TriangularMatrixVector.h @@ -14,337 +14,403 @@ namespace Eigen { namespace internal { -template -struct triangular_matrix_vector_product; - -template -struct triangular_matrix_vector_product -{ - typedef typename ScalarBinaryOpTraits::ReturnType ResScalar; - enum { - IsLower = ((Mode&Lower)==Lower), - HasUnitDiag = (Mode & UnitDiag)==UnitDiag, - HasZeroDiag = (Mode & ZeroDiag)==ZeroDiag + template + struct triangular_matrix_vector_product; + + template + struct triangular_matrix_vector_product + { + typedef typename ScalarBinaryOpTraits::ReturnType ResScalar; + enum { + IsLower = ((Mode & Lower) == Lower), + HasUnitDiag = (Mode & UnitDiag) == UnitDiag, + HasZeroDiag = (Mode & ZeroDiag) == ZeroDiag + }; + static EIGEN_DONT_INLINE void run(Index _rows, + Index _cols, + const LhsScalar *_lhs, + Index lhsStride, + const RhsScalar *_rhs, + Index rhsIncr, + ResScalar *_res, + Index resIncr, + const RhsScalar &alpha); }; - static EIGEN_DONT_INLINE void run(Index _rows, Index _cols, const LhsScalar* _lhs, Index lhsStride, - const RhsScalar* _rhs, Index rhsIncr, ResScalar* _res, Index resIncr, const RhsScalar& alpha); -}; - -template -EIGEN_DONT_INLINE void triangular_matrix_vector_product - ::run(Index _rows, Index _cols, const LhsScalar* _lhs, Index lhsStride, - const RhsScalar* _rhs, Index rhsIncr, ResScalar* _res, Index resIncr, const RhsScalar& alpha) + + template + EIGEN_DONT_INLINE void + triangular_matrix_vector_product::run( + Index _rows, + Index _cols, + const LhsScalar *_lhs, + Index lhsStride, + const RhsScalar *_rhs, + Index rhsIncr, + ResScalar *_res, + Index resIncr, + const RhsScalar &alpha) { static const Index PanelWidth = EIGEN_TUNE_TRIANGULAR_PANEL_WIDTH; - Index size = (std::min)(_rows,_cols); - Index rows = IsLower ? _rows : (std::min)(_rows,_cols); - Index cols = IsLower ? (std::min)(_rows,_cols) : _cols; + Index size = (std::min)(_rows, _cols); + Index rows = IsLower ? _rows : (std::min)(_rows, _cols); + Index cols = IsLower ? (std::min)(_rows, _cols) : _cols; - typedef Map, 0, OuterStride<> > LhsMap; - const LhsMap lhs(_lhs,rows,cols,OuterStride<>(lhsStride)); - typename conj_expr_if::type cjLhs(lhs); + typedef Map, 0, OuterStride<>> LhsMap; + const LhsMap lhs(_lhs, rows, cols, OuterStride<>(lhsStride)); + typename conj_expr_if::type cjLhs(lhs); - typedef Map, 0, InnerStride<> > RhsMap; - const RhsMap rhs(_rhs,cols,InnerStride<>(rhsIncr)); - typename conj_expr_if::type cjRhs(rhs); + typedef Map, 0, InnerStride<>> RhsMap; + const RhsMap rhs(_rhs, cols, InnerStride<>(rhsIncr)); + typename conj_expr_if::type cjRhs(rhs); - typedef Map > ResMap; - ResMap res(_res,rows); + typedef Map> ResMap; + ResMap res(_res, rows); - typedef const_blas_data_mapper LhsMapper; - typedef const_blas_data_mapper RhsMapper; + typedef const_blas_data_mapper LhsMapper; + typedef const_blas_data_mapper RhsMapper; - for (Index pi=0; pi0) - res.segment(s,r) += (alpha * cjRhs.coeff(i)) * cjLhs.col(i).segment(s,r); - if (HasUnitDiag) - res.coeffRef(i) += alpha * cjRhs.coeff(i); + Index s = IsLower ? ((HasUnitDiag || HasZeroDiag) ? i + 1 : i) : pi; + Index r = IsLower ? actualPanelWidth - k : k + 1; + if ((!(HasUnitDiag || HasZeroDiag)) || (--r) > 0) + res.segment(s, r) += (alpha * cjRhs.coeff(i)) * cjLhs.col(i).segment(s, r); + if (HasUnitDiag) res.coeffRef(i) += alpha * cjRhs.coeff(i); } Index r = IsLower ? rows - pi - actualPanelWidth : pi; - if (r>0) - { - Index s = IsLower ? pi+actualPanelWidth : 0; - general_matrix_vector_product::run( - r, actualPanelWidth, - LhsMapper(&lhs.coeffRef(s,pi), lhsStride), - RhsMapper(&rhs.coeffRef(pi), rhsIncr), - &res.coeffRef(s), resIncr, alpha); + if (r > 0) { + Index s = IsLower ? pi + actualPanelWidth : 0; + general_matrix_vector_product::run(r, + actualPanelWidth, + LhsMapper(&lhs.coeffRef(s, pi), lhsStride), + RhsMapper(&rhs.coeffRef(pi), rhsIncr), + &res.coeffRef(s), + resIncr, + alpha); } } - if((!IsLower) && cols>size) - { - general_matrix_vector_product::run( - rows, cols-size, - LhsMapper(&lhs.coeffRef(0,size), lhsStride), - RhsMapper(&rhs.coeffRef(size), rhsIncr), - _res, resIncr, alpha); + if ((!IsLower) && cols > size) { + general_matrix_vector_product::run( + rows, + cols - size, + LhsMapper(&lhs.coeffRef(0, size), lhsStride), + RhsMapper(&rhs.coeffRef(size), rhsIncr), + _res, + resIncr, + alpha); } } -template -struct triangular_matrix_vector_product -{ - typedef typename ScalarBinaryOpTraits::ReturnType ResScalar; - enum { - IsLower = ((Mode&Lower)==Lower), - HasUnitDiag = (Mode & UnitDiag)==UnitDiag, - HasZeroDiag = (Mode & ZeroDiag)==ZeroDiag + template + struct triangular_matrix_vector_product + { + typedef typename ScalarBinaryOpTraits::ReturnType ResScalar; + enum { + IsLower = ((Mode & Lower) == Lower), + HasUnitDiag = (Mode & UnitDiag) == UnitDiag, + HasZeroDiag = (Mode & ZeroDiag) == ZeroDiag + }; + static EIGEN_DONT_INLINE void run(Index _rows, + Index _cols, + const LhsScalar *_lhs, + Index lhsStride, + const RhsScalar *_rhs, + Index rhsIncr, + ResScalar *_res, + Index resIncr, + const ResScalar &alpha); }; - static EIGEN_DONT_INLINE void run(Index _rows, Index _cols, const LhsScalar* _lhs, Index lhsStride, - const RhsScalar* _rhs, Index rhsIncr, ResScalar* _res, Index resIncr, const ResScalar& alpha); -}; - -template -EIGEN_DONT_INLINE void triangular_matrix_vector_product - ::run(Index _rows, Index _cols, const LhsScalar* _lhs, Index lhsStride, - const RhsScalar* _rhs, Index rhsIncr, ResScalar* _res, Index resIncr, const ResScalar& alpha) + + template + EIGEN_DONT_INLINE void + triangular_matrix_vector_product::run( + Index _rows, + Index _cols, + const LhsScalar *_lhs, + Index lhsStride, + const RhsScalar *_rhs, + Index rhsIncr, + ResScalar *_res, + Index resIncr, + const ResScalar &alpha) { static const Index PanelWidth = EIGEN_TUNE_TRIANGULAR_PANEL_WIDTH; - Index diagSize = (std::min)(_rows,_cols); + Index diagSize = (std::min)(_rows, _cols); Index rows = IsLower ? _rows : diagSize; Index cols = IsLower ? diagSize : _cols; - typedef Map, 0, OuterStride<> > LhsMap; - const LhsMap lhs(_lhs,rows,cols,OuterStride<>(lhsStride)); - typename conj_expr_if::type cjLhs(lhs); + typedef Map, 0, OuterStride<>> LhsMap; + const LhsMap lhs(_lhs, rows, cols, OuterStride<>(lhsStride)); + typename conj_expr_if::type cjLhs(lhs); - typedef Map > RhsMap; - const RhsMap rhs(_rhs,cols); - typename conj_expr_if::type cjRhs(rhs); + typedef Map> RhsMap; + const RhsMap rhs(_rhs, cols); + typename conj_expr_if::type cjRhs(rhs); - typedef Map, 0, InnerStride<> > ResMap; - ResMap res(_res,rows,InnerStride<>(resIncr)); + typedef Map, 0, InnerStride<>> ResMap; + ResMap res(_res, rows, InnerStride<>(resIncr)); - typedef const_blas_data_mapper LhsMapper; - typedef const_blas_data_mapper RhsMapper; + typedef const_blas_data_mapper LhsMapper; + typedef const_blas_data_mapper RhsMapper; - for (Index pi=0; pi0) - res.coeffRef(i) += alpha * (cjLhs.row(i).segment(s,r).cwiseProduct(cjRhs.segment(s,r).transpose())).sum(); - if (HasUnitDiag) - res.coeffRef(i) += alpha * cjRhs.coeff(i); + Index s = IsLower ? pi : ((HasUnitDiag || HasZeroDiag) ? i + 1 : i); + Index r = IsLower ? k + 1 : actualPanelWidth - k; + if ((!(HasUnitDiag || HasZeroDiag)) || (--r) > 0) + res.coeffRef(i) += alpha * (cjLhs.row(i).segment(s, r).cwiseProduct(cjRhs.segment(s, r).transpose())).sum(); + if (HasUnitDiag) res.coeffRef(i) += alpha * cjRhs.coeff(i); } Index r = IsLower ? pi : cols - pi - actualPanelWidth; - if (r>0) - { + if (r > 0) { Index s = IsLower ? 0 : pi + actualPanelWidth; - general_matrix_vector_product::run( - actualPanelWidth, r, - LhsMapper(&lhs.coeffRef(pi,s), lhsStride), - RhsMapper(&rhs.coeffRef(s), rhsIncr), - &res.coeffRef(pi), resIncr, alpha); + general_matrix_vector_product::run(actualPanelWidth, + r, + LhsMapper(&lhs.coeffRef(pi, s), lhsStride), + RhsMapper(&rhs.coeffRef(s), rhsIncr), + &res.coeffRef(pi), + resIncr, + alpha); } } - if(IsLower && rows>diagSize) - { - general_matrix_vector_product::run( - rows-diagSize, cols, - LhsMapper(&lhs.coeffRef(diagSize,0), lhsStride), - RhsMapper(&rhs.coeffRef(0), rhsIncr), - &res.coeffRef(diagSize), resIncr, alpha); + if (IsLower && rows > diagSize) { + general_matrix_vector_product::run( + rows - diagSize, + cols, + LhsMapper(&lhs.coeffRef(diagSize, 0), lhsStride), + RhsMapper(&rhs.coeffRef(0), rhsIncr), + &res.coeffRef(diagSize), + resIncr, + alpha); } } -/*************************************************************************** -* Wrapper to product_triangular_vector -***************************************************************************/ + /*************************************************************************** + * Wrapper to product_triangular_vector + ***************************************************************************/ -template -struct trmv_selector; + template struct trmv_selector; -} // end namespace internal +}// end namespace internal namespace internal { -template -struct triangular_product_impl -{ - template static void run(Dest& dst, const Lhs &lhs, const Rhs &rhs, const typename Dest::Scalar& alpha) - { - eigen_assert(dst.rows()==lhs.rows() && dst.cols()==rhs.cols()); - - internal::trmv_selector::Flags)&RowMajorBit) ? RowMajor : ColMajor>::run(lhs, rhs, dst, alpha); - } -}; - -template -struct triangular_product_impl -{ - template static void run(Dest& dst, const Lhs &lhs, const Rhs &rhs, const typename Dest::Scalar& alpha) + template struct triangular_product_impl { - eigen_assert(dst.rows()==lhs.rows() && dst.cols()==rhs.cols()); - - Transpose dstT(dst); - internal::trmv_selector<(Mode & (UnitDiag|ZeroDiag)) | ((Mode & Lower) ? Upper : Lower), - (int(internal::traits::Flags)&RowMajorBit) ? ColMajor : RowMajor> - ::run(rhs.transpose(),lhs.transpose(), dstT, alpha); - } -}; - -} // end namespace internal + template + static void run(Dest &dst, const Lhs &lhs, const Rhs &rhs, const typename Dest::Scalar &alpha) + { + eigen_assert(dst.rows() == lhs.rows() && dst.cols() == rhs.cols()); -namespace internal { + internal::trmv_selector::Flags) & RowMajorBit) ? RowMajor : ColMajor>::run( + lhs, rhs, dst, alpha); + } + }; -// TODO: find a way to factorize this piece of code with gemv_selector since the logic is exactly the same. - -template struct trmv_selector -{ - template - static void run(const Lhs &lhs, const Rhs &rhs, Dest& dest, const typename Dest::Scalar& alpha) + template struct triangular_product_impl { - typedef typename Lhs::Scalar LhsScalar; - typedef typename Rhs::Scalar RhsScalar; - typedef typename Dest::Scalar ResScalar; - typedef typename Dest::RealScalar RealScalar; - - typedef internal::blas_traits LhsBlasTraits; - typedef typename LhsBlasTraits::DirectLinearAccessType ActualLhsType; - typedef internal::blas_traits RhsBlasTraits; - typedef typename RhsBlasTraits::DirectLinearAccessType ActualRhsType; - - typedef Map, EIGEN_PLAIN_ENUM_MIN(AlignedMax,internal::packet_traits::size)> MappedDest; - - typename internal::add_const_on_value_type::type actualLhs = LhsBlasTraits::extract(lhs); - typename internal::add_const_on_value_type::type actualRhs = RhsBlasTraits::extract(rhs); - - LhsScalar lhs_alpha = LhsBlasTraits::extractScalarFactor(lhs); - RhsScalar rhs_alpha = RhsBlasTraits::extractScalarFactor(rhs); - ResScalar actualAlpha = alpha * lhs_alpha * rhs_alpha; - - enum { - // FIXME find a way to allow an inner stride on the result if packet_traits::size==1 - // on, the other hand it is good for the cache to pack the vector anyways... - EvalToDestAtCompileTime = Dest::InnerStrideAtCompileTime==1, - ComplexByReal = (NumTraits::IsComplex) && (!NumTraits::IsComplex), - MightCannotUseDest = (Dest::InnerStrideAtCompileTime!=1) || ComplexByReal - }; - - gemv_static_vector_if static_dest; + template + static void run(Dest &dst, const Lhs &lhs, const Rhs &rhs, const typename Dest::Scalar &alpha) + { + eigen_assert(dst.rows() == lhs.rows() && dst.cols() == rhs.cols()); + + Transpose dstT(dst); + internal::trmv_selector<(Mode & (UnitDiag | ZeroDiag)) | ((Mode & Lower) ? Upper : Lower), + (int(internal::traits::Flags) & RowMajorBit) ? ColMajor : RowMajor>::run(rhs.transpose(), + lhs.transpose(), + dstT, + alpha); + } + }; - bool alphaIsCompatible = (!ComplexByReal) || (numext::imag(actualAlpha)==RealScalar(0)); - bool evalToDest = EvalToDestAtCompileTime && alphaIsCompatible; +}// end namespace internal - RhsScalar compatibleAlpha = get_factor::run(actualAlpha); +namespace internal { - ei_declare_aligned_stack_constructed_variable(ResScalar,actualDestPtr,dest.size(), - evalToDest ? dest.data() : static_dest.data()); + // TODO: find a way to factorize this piece of code with gemv_selector since the logic is exactly the same. - if(!evalToDest) + template struct trmv_selector + { + template + static void run(const Lhs &lhs, const Rhs &rhs, Dest &dest, const typename Dest::Scalar &alpha) { - #ifdef EIGEN_DENSE_STORAGE_CTOR_PLUGIN - Index size = dest.size(); - EIGEN_DENSE_STORAGE_CTOR_PLUGIN - #endif - if(!alphaIsCompatible) - { - MappedDest(actualDestPtr, dest.size()).setZero(); - compatibleAlpha = RhsScalar(1); + typedef typename Lhs::Scalar LhsScalar; + typedef typename Rhs::Scalar RhsScalar; + typedef typename Dest::Scalar ResScalar; + typedef typename Dest::RealScalar RealScalar; + + typedef internal::blas_traits LhsBlasTraits; + typedef typename LhsBlasTraits::DirectLinearAccessType ActualLhsType; + typedef internal::blas_traits RhsBlasTraits; + typedef typename RhsBlasTraits::DirectLinearAccessType ActualRhsType; + + typedef Map, + EIGEN_PLAIN_ENUM_MIN(AlignedMax, internal::packet_traits::size)> + MappedDest; + + typename internal::add_const_on_value_type::type actualLhs = LhsBlasTraits::extract(lhs); + typename internal::add_const_on_value_type::type actualRhs = RhsBlasTraits::extract(rhs); + + LhsScalar lhs_alpha = LhsBlasTraits::extractScalarFactor(lhs); + RhsScalar rhs_alpha = RhsBlasTraits::extractScalarFactor(rhs); + ResScalar actualAlpha = alpha * lhs_alpha * rhs_alpha; + + enum { + // FIXME find a way to allow an inner stride on the result if packet_traits::size==1 + // on, the other hand it is good for the cache to pack the vector anyways... + EvalToDestAtCompileTime = Dest::InnerStrideAtCompileTime == 1, + ComplexByReal = (NumTraits::IsComplex) && (!NumTraits::IsComplex), + MightCannotUseDest = (Dest::InnerStrideAtCompileTime != 1) || ComplexByReal + }; + + gemv_static_vector_if + static_dest; + + bool alphaIsCompatible = (!ComplexByReal) || (numext::imag(actualAlpha) == RealScalar(0)); + bool evalToDest = EvalToDestAtCompileTime && alphaIsCompatible; + + RhsScalar compatibleAlpha = get_factor::run(actualAlpha); + + ei_declare_aligned_stack_constructed_variable( + ResScalar, actualDestPtr, dest.size(), evalToDest ? dest.data() : static_dest.data()); + + if (!evalToDest) { +#ifdef EIGEN_DENSE_STORAGE_CTOR_PLUGIN + Index size = dest.size(); + EIGEN_DENSE_STORAGE_CTOR_PLUGIN +#endif + if (!alphaIsCompatible) { + MappedDest(actualDestPtr, dest.size()).setZero(); + compatibleAlpha = RhsScalar(1); + } else + MappedDest(actualDestPtr, dest.size()) = dest; } - else - MappedDest(actualDestPtr, dest.size()) = dest; - } - internal::triangular_matrix_vector_product - - ::run(actualLhs.rows(),actualLhs.cols(), - actualLhs.data(),actualLhs.outerStride(), - actualRhs.data(),actualRhs.innerStride(), - actualDestPtr,1,compatibleAlpha); - - if (!evalToDest) - { - if(!alphaIsCompatible) - dest += actualAlpha * MappedDest(actualDestPtr, dest.size()); - else - dest = MappedDest(actualDestPtr, dest.size()); - } + internal::triangular_matrix_vector_product::run(actualLhs.rows(), + actualLhs.cols(), + actualLhs.data(), + actualLhs.outerStride(), + actualRhs.data(), + actualRhs.innerStride(), + actualDestPtr, + 1, + compatibleAlpha); + + if (!evalToDest) { + if (!alphaIsCompatible) + dest += actualAlpha * MappedDest(actualDestPtr, dest.size()); + else + dest = MappedDest(actualDestPtr, dest.size()); + } - if ( ((Mode&UnitDiag)==UnitDiag) && (lhs_alpha!=LhsScalar(1)) ) - { - Index diagSize = (std::min)(lhs.rows(),lhs.cols()); - dest.head(diagSize) -= (lhs_alpha-LhsScalar(1))*rhs.head(diagSize); + if (((Mode & UnitDiag) == UnitDiag) && (lhs_alpha != LhsScalar(1))) { + Index diagSize = (std::min)(lhs.rows(), lhs.cols()); + dest.head(diagSize) -= (lhs_alpha - LhsScalar(1)) * rhs.head(diagSize); + } } - } -}; + }; -template struct trmv_selector -{ - template - static void run(const Lhs &lhs, const Rhs &rhs, Dest& dest, const typename Dest::Scalar& alpha) + template struct trmv_selector { - typedef typename Lhs::Scalar LhsScalar; - typedef typename Rhs::Scalar RhsScalar; - typedef typename Dest::Scalar ResScalar; - - typedef internal::blas_traits LhsBlasTraits; - typedef typename LhsBlasTraits::DirectLinearAccessType ActualLhsType; - typedef internal::blas_traits RhsBlasTraits; - typedef typename RhsBlasTraits::DirectLinearAccessType ActualRhsType; - typedef typename internal::remove_all::type ActualRhsTypeCleaned; - - typename add_const::type actualLhs = LhsBlasTraits::extract(lhs); - typename add_const::type actualRhs = RhsBlasTraits::extract(rhs); - - LhsScalar lhs_alpha = LhsBlasTraits::extractScalarFactor(lhs); - RhsScalar rhs_alpha = RhsBlasTraits::extractScalarFactor(rhs); - ResScalar actualAlpha = alpha * lhs_alpha * rhs_alpha; - - enum { - DirectlyUseRhs = ActualRhsTypeCleaned::InnerStrideAtCompileTime==1 - }; - - gemv_static_vector_if static_rhs; - - ei_declare_aligned_stack_constructed_variable(RhsScalar,actualRhsPtr,actualRhs.size(), - DirectlyUseRhs ? const_cast(actualRhs.data()) : static_rhs.data()); - - if(!DirectlyUseRhs) + template + static void run(const Lhs &lhs, const Rhs &rhs, Dest &dest, const typename Dest::Scalar &alpha) { - #ifdef EIGEN_DENSE_STORAGE_CTOR_PLUGIN - Index size = actualRhs.size(); - EIGEN_DENSE_STORAGE_CTOR_PLUGIN - #endif - Map(actualRhsPtr, actualRhs.size()) = actualRhs; - } + typedef typename Lhs::Scalar LhsScalar; + typedef typename Rhs::Scalar RhsScalar; + typedef typename Dest::Scalar ResScalar; + + typedef internal::blas_traits LhsBlasTraits; + typedef typename LhsBlasTraits::DirectLinearAccessType ActualLhsType; + typedef internal::blas_traits RhsBlasTraits; + typedef typename RhsBlasTraits::DirectLinearAccessType ActualRhsType; + typedef typename internal::remove_all::type ActualRhsTypeCleaned; + + typename add_const::type actualLhs = LhsBlasTraits::extract(lhs); + typename add_const::type actualRhs = RhsBlasTraits::extract(rhs); + + LhsScalar lhs_alpha = LhsBlasTraits::extractScalarFactor(lhs); + RhsScalar rhs_alpha = RhsBlasTraits::extractScalarFactor(rhs); + ResScalar actualAlpha = alpha * lhs_alpha * rhs_alpha; + + enum { DirectlyUseRhs = ActualRhsTypeCleaned::InnerStrideAtCompileTime == 1 }; + + gemv_static_vector_if + static_rhs; + + ei_declare_aligned_stack_constructed_variable(RhsScalar, + actualRhsPtr, + actualRhs.size(), + DirectlyUseRhs ? const_cast(actualRhs.data()) : static_rhs.data()); + + if (!DirectlyUseRhs) { +#ifdef EIGEN_DENSE_STORAGE_CTOR_PLUGIN + Index size = actualRhs.size(); + EIGEN_DENSE_STORAGE_CTOR_PLUGIN +#endif + Map(actualRhsPtr, actualRhs.size()) = actualRhs; + } - internal::triangular_matrix_vector_product - - ::run(actualLhs.rows(),actualLhs.cols(), - actualLhs.data(),actualLhs.outerStride(), - actualRhsPtr,1, - dest.data(),dest.innerStride(), - actualAlpha); - - if ( ((Mode&UnitDiag)==UnitDiag) && (lhs_alpha!=LhsScalar(1)) ) - { - Index diagSize = (std::min)(lhs.rows(),lhs.cols()); - dest.head(diagSize) -= (lhs_alpha-LhsScalar(1))*rhs.head(diagSize); + internal::triangular_matrix_vector_product::run(actualLhs.rows(), + actualLhs.cols(), + actualLhs.data(), + actualLhs.outerStride(), + actualRhsPtr, + 1, + dest.data(), + dest.innerStride(), + actualAlpha); + + if (((Mode & UnitDiag) == UnitDiag) && (lhs_alpha != LhsScalar(1))) { + Index diagSize = (std::min)(lhs.rows(), lhs.cols()); + dest.head(diagSize) -= (lhs_alpha - LhsScalar(1)) * rhs.head(diagSize); + } } - } -}; + }; -} // end namespace internal +}// end namespace internal -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_TRIANGULARMATRIXVECTOR_H +#endif// EIGEN_TRIANGULARMATRIXVECTOR_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/products/TriangularMatrixVector_BLAS.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/products/TriangularMatrixVector_BLAS.h index 3d47a2b9..d7910d8c 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/products/TriangularMatrixVector_BLAS.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/products/TriangularMatrixVector_BLAS.h @@ -33,223 +33,297 @@ #ifndef EIGEN_TRIANGULAR_MATRIX_VECTOR_BLAS_H #define EIGEN_TRIANGULAR_MATRIX_VECTOR_BLAS_H -namespace Eigen { +namespace Eigen { namespace internal { -/********************************************************************** -* This file implements triangular matrix-vector multiplication using BLAS -**********************************************************************/ - -// trmv/hemv specialization - -template -struct triangular_matrix_vector_product_trmv : - triangular_matrix_vector_product {}; - -#define EIGEN_BLAS_TRMV_SPECIALIZE(Scalar) \ -template \ -struct triangular_matrix_vector_product { \ - static void run(Index _rows, Index _cols, const Scalar* _lhs, Index lhsStride, \ - const Scalar* _rhs, Index rhsIncr, Scalar* _res, Index resIncr, Scalar alpha) { \ - triangular_matrix_vector_product_trmv::run( \ - _rows, _cols, _lhs, lhsStride, _rhs, rhsIncr, _res, resIncr, alpha); \ - } \ -}; \ -template \ -struct triangular_matrix_vector_product { \ - static void run(Index _rows, Index _cols, const Scalar* _lhs, Index lhsStride, \ - const Scalar* _rhs, Index rhsIncr, Scalar* _res, Index resIncr, Scalar alpha) { \ - triangular_matrix_vector_product_trmv::run( \ - _rows, _cols, _lhs, lhsStride, _rhs, rhsIncr, _res, resIncr, alpha); \ - } \ -}; - -EIGEN_BLAS_TRMV_SPECIALIZE(double) -EIGEN_BLAS_TRMV_SPECIALIZE(float) -EIGEN_BLAS_TRMV_SPECIALIZE(dcomplex) -EIGEN_BLAS_TRMV_SPECIALIZE(scomplex) + /********************************************************************** + * This file implements triangular matrix-vector multiplication using BLAS + **********************************************************************/ + + // trmv/hemv specialization + + template + struct triangular_matrix_vector_product_trmv + : triangular_matrix_vector_product + { + }; + +#define EIGEN_BLAS_TRMV_SPECIALIZE(Scalar) \ + template \ + struct triangular_matrix_vector_product \ + { \ + static void run(Index _rows, \ + Index _cols, \ + const Scalar *_lhs, \ + Index lhsStride, \ + const Scalar *_rhs, \ + Index rhsIncr, \ + Scalar *_res, \ + Index resIncr, \ + Scalar alpha) \ + { \ + triangular_matrix_vector_product_trmv::run( \ + _rows, _cols, _lhs, lhsStride, _rhs, rhsIncr, _res, resIncr, alpha); \ + } \ + }; \ + template \ + struct triangular_matrix_vector_product \ + { \ + static void run(Index _rows, \ + Index _cols, \ + const Scalar *_lhs, \ + Index lhsStride, \ + const Scalar *_rhs, \ + Index rhsIncr, \ + Scalar *_res, \ + Index resIncr, \ + Scalar alpha) \ + { \ + triangular_matrix_vector_product_trmv::run( \ + _rows, _cols, _lhs, lhsStride, _rhs, rhsIncr, _res, resIncr, alpha); \ + } \ + }; + + EIGEN_BLAS_TRMV_SPECIALIZE(double) + EIGEN_BLAS_TRMV_SPECIALIZE(float) + EIGEN_BLAS_TRMV_SPECIALIZE(dcomplex) + EIGEN_BLAS_TRMV_SPECIALIZE(scomplex) // implements col-major: res += alpha * op(triangular) * vector -#define EIGEN_BLAS_TRMV_CM(EIGTYPE, BLASTYPE, EIGPREFIX, BLASPREFIX, BLASPOSTFIX) \ -template \ -struct triangular_matrix_vector_product_trmv { \ - enum { \ - IsLower = (Mode&Lower) == Lower, \ - SetDiag = (Mode&(ZeroDiag|UnitDiag)) ? 0 : 1, \ - IsUnitDiag = (Mode&UnitDiag) ? 1 : 0, \ - IsZeroDiag = (Mode&ZeroDiag) ? 1 : 0, \ - LowUp = IsLower ? Lower : Upper \ - }; \ - static void run(Index _rows, Index _cols, const EIGTYPE* _lhs, Index lhsStride, \ - const EIGTYPE* _rhs, Index rhsIncr, EIGTYPE* _res, Index resIncr, EIGTYPE alpha) \ - { \ - if (ConjLhs || IsZeroDiag) { \ - triangular_matrix_vector_product::run( \ - _rows, _cols, _lhs, lhsStride, _rhs, rhsIncr, _res, resIncr, alpha); \ - return; \ - }\ - Index size = (std::min)(_rows,_cols); \ - Index rows = IsLower ? _rows : size; \ - Index cols = IsLower ? size : _cols; \ -\ - typedef VectorX##EIGPREFIX VectorRhs; \ - EIGTYPE *x, *y;\ -\ -/* Set x*/ \ - Map > rhs(_rhs,cols,InnerStride<>(rhsIncr)); \ - VectorRhs x_tmp; \ - if (ConjRhs) x_tmp = rhs.conjugate(); else x_tmp = rhs; \ - x = x_tmp.data(); \ -\ -/* Square part handling */\ -\ - char trans, uplo, diag; \ - BlasIndex m, n, lda, incx, incy; \ - EIGTYPE const *a; \ - EIGTYPE beta(1); \ -\ -/* Set m, n */ \ - n = convert_index(size); \ - lda = convert_index(lhsStride); \ - incx = 1; \ - incy = convert_index(resIncr); \ -\ -/* Set uplo, trans and diag*/ \ - trans = 'N'; \ - uplo = IsLower ? 'L' : 'U'; \ - diag = IsUnitDiag ? 'U' : 'N'; \ -\ -/* call ?TRMV*/ \ - BLASPREFIX##trmv##BLASPOSTFIX(&uplo, &trans, &diag, &n, (const BLASTYPE*)_lhs, &lda, (BLASTYPE*)x, &incx); \ -\ -/* Add op(a_tr)rhs into res*/ \ - BLASPREFIX##axpy##BLASPOSTFIX(&n, (const BLASTYPE*)&numext::real_ref(alpha),(const BLASTYPE*)x, &incx, (BLASTYPE*)_res, &incy); \ -/* Non-square case - doesn't fit to BLAS ?TRMV. Fall to default triangular product*/ \ - if (size<(std::max)(rows,cols)) { \ - if (ConjRhs) x_tmp = rhs.conjugate(); else x_tmp = rhs; \ - x = x_tmp.data(); \ - if (size(rows-size); \ - n = convert_index(size); \ - } \ - else { \ - x += size; \ - y = _res; \ - a = _lhs + size*lda; \ - m = convert_index(size); \ - n = convert_index(cols-size); \ - } \ - BLASPREFIX##gemv##BLASPOSTFIX(&trans, &m, &n, (const BLASTYPE*)&numext::real_ref(alpha), (const BLASTYPE*)a, &lda, (const BLASTYPE*)x, &incx, (const BLASTYPE*)&numext::real_ref(beta), (BLASTYPE*)y, &incy); \ - } \ - } \ -}; +#define EIGEN_BLAS_TRMV_CM(EIGTYPE, BLASTYPE, EIGPREFIX, BLASPREFIX, BLASPOSTFIX) \ + template \ + struct triangular_matrix_vector_product_trmv \ + { \ + enum { \ + IsLower = (Mode & Lower) == Lower, \ + SetDiag = (Mode & (ZeroDiag | UnitDiag)) ? 0 : 1, \ + IsUnitDiag = (Mode & UnitDiag) ? 1 : 0, \ + IsZeroDiag = (Mode & ZeroDiag) ? 1 : 0, \ + LowUp = IsLower ? Lower : Upper \ + }; \ + static void run(Index _rows, \ + Index _cols, \ + const EIGTYPE *_lhs, \ + Index lhsStride, \ + const EIGTYPE *_rhs, \ + Index rhsIncr, \ + EIGTYPE *_res, \ + Index resIncr, \ + EIGTYPE alpha) \ + { \ + if (ConjLhs || IsZeroDiag) { \ + triangular_matrix_vector_product::run( \ + _rows, _cols, _lhs, lhsStride, _rhs, rhsIncr, _res, resIncr, alpha); \ + return; \ + } \ + Index size = (std::min)(_rows, _cols); \ + Index rows = IsLower ? _rows : size; \ + Index cols = IsLower ? size : _cols; \ + \ + typedef VectorX##EIGPREFIX VectorRhs; \ + EIGTYPE *x, *y; \ + \ + /* Set x*/ \ + Map> rhs(_rhs, cols, InnerStride<>(rhsIncr)); \ + VectorRhs x_tmp; \ + if (ConjRhs) \ + x_tmp = rhs.conjugate(); \ + else \ + x_tmp = rhs; \ + x = x_tmp.data(); \ + \ + /* Square part handling */ \ + \ + char trans, uplo, diag; \ + BlasIndex m, n, lda, incx, incy; \ + EIGTYPE const *a; \ + EIGTYPE beta(1); \ + \ + /* Set m, n */ \ + n = convert_index(size); \ + lda = convert_index(lhsStride); \ + incx = 1; \ + incy = convert_index(resIncr); \ + \ + /* Set uplo, trans and diag*/ \ + trans = 'N'; \ + uplo = IsLower ? 'L' : 'U'; \ + diag = IsUnitDiag ? 'U' : 'N'; \ + \ + /* call ?TRMV*/ \ + BLASPREFIX##trmv##BLASPOSTFIX(&uplo, &trans, &diag, &n, (const BLASTYPE *)_lhs, &lda, (BLASTYPE *)x, &incx); \ + \ + /* Add op(a_tr)rhs into res*/ \ + BLASPREFIX##axpy##BLASPOSTFIX( \ + &n, (const BLASTYPE *)&numext::real_ref(alpha), (const BLASTYPE *)x, &incx, (BLASTYPE *)_res, &incy); \ + /* Non-square case - doesn't fit to BLAS ?TRMV. Fall to default triangular product*/ \ + if (size < (std::max)(rows, cols)) { \ + if (ConjRhs) \ + x_tmp = rhs.conjugate(); \ + else \ + x_tmp = rhs; \ + x = x_tmp.data(); \ + if (size < rows) { \ + y = _res + size * resIncr; \ + a = _lhs + size; \ + m = convert_index(rows - size); \ + n = convert_index(size); \ + } else { \ + x += size; \ + y = _res; \ + a = _lhs + size * lda; \ + m = convert_index(size); \ + n = convert_index(cols - size); \ + } \ + BLASPREFIX##gemv##BLASPOSTFIX(&trans, \ + &m, \ + &n, \ + (const BLASTYPE *)&numext::real_ref(alpha), \ + (const BLASTYPE *)a, \ + &lda, \ + (const BLASTYPE *)x, \ + &incx, \ + (const BLASTYPE *)&numext::real_ref(beta), \ + (BLASTYPE *)y, \ + &incy); \ + } \ + } \ + }; #ifdef EIGEN_USE_MKL -EIGEN_BLAS_TRMV_CM(double, double, d, d,) -EIGEN_BLAS_TRMV_CM(dcomplex, MKL_Complex16, cd, z,) -EIGEN_BLAS_TRMV_CM(float, float, f, s,) -EIGEN_BLAS_TRMV_CM(scomplex, MKL_Complex8, cf, c,) + EIGEN_BLAS_TRMV_CM(double, double, d, d, ) + EIGEN_BLAS_TRMV_CM(dcomplex, MKL_Complex16, cd, z, ) + EIGEN_BLAS_TRMV_CM(float, float, f, s, ) + EIGEN_BLAS_TRMV_CM(scomplex, MKL_Complex8, cf, c, ) #else -EIGEN_BLAS_TRMV_CM(double, double, d, d, _) -EIGEN_BLAS_TRMV_CM(dcomplex, double, cd, z, _) -EIGEN_BLAS_TRMV_CM(float, float, f, s, _) -EIGEN_BLAS_TRMV_CM(scomplex, float, cf, c, _) + EIGEN_BLAS_TRMV_CM(double, double, d, d, _) + EIGEN_BLAS_TRMV_CM(dcomplex, double, cd, z, _) + EIGEN_BLAS_TRMV_CM(float, float, f, s, _) + EIGEN_BLAS_TRMV_CM(scomplex, float, cf, c, _) #endif // implements row-major: res += alpha * op(triangular) * vector -#define EIGEN_BLAS_TRMV_RM(EIGTYPE, BLASTYPE, EIGPREFIX, BLASPREFIX, BLASPOSTFIX) \ -template \ -struct triangular_matrix_vector_product_trmv { \ - enum { \ - IsLower = (Mode&Lower) == Lower, \ - SetDiag = (Mode&(ZeroDiag|UnitDiag)) ? 0 : 1, \ - IsUnitDiag = (Mode&UnitDiag) ? 1 : 0, \ - IsZeroDiag = (Mode&ZeroDiag) ? 1 : 0, \ - LowUp = IsLower ? Lower : Upper \ - }; \ - static void run(Index _rows, Index _cols, const EIGTYPE* _lhs, Index lhsStride, \ - const EIGTYPE* _rhs, Index rhsIncr, EIGTYPE* _res, Index resIncr, EIGTYPE alpha) \ - { \ - if (IsZeroDiag) { \ - triangular_matrix_vector_product::run( \ - _rows, _cols, _lhs, lhsStride, _rhs, rhsIncr, _res, resIncr, alpha); \ - return; \ - }\ - Index size = (std::min)(_rows,_cols); \ - Index rows = IsLower ? _rows : size; \ - Index cols = IsLower ? size : _cols; \ -\ - typedef VectorX##EIGPREFIX VectorRhs; \ - EIGTYPE *x, *y;\ -\ -/* Set x*/ \ - Map > rhs(_rhs,cols,InnerStride<>(rhsIncr)); \ - VectorRhs x_tmp; \ - if (ConjRhs) x_tmp = rhs.conjugate(); else x_tmp = rhs; \ - x = x_tmp.data(); \ -\ -/* Square part handling */\ -\ - char trans, uplo, diag; \ - BlasIndex m, n, lda, incx, incy; \ - EIGTYPE const *a; \ - EIGTYPE beta(1); \ -\ -/* Set m, n */ \ - n = convert_index(size); \ - lda = convert_index(lhsStride); \ - incx = 1; \ - incy = convert_index(resIncr); \ -\ -/* Set uplo, trans and diag*/ \ - trans = ConjLhs ? 'C' : 'T'; \ - uplo = IsLower ? 'U' : 'L'; \ - diag = IsUnitDiag ? 'U' : 'N'; \ -\ -/* call ?TRMV*/ \ - BLASPREFIX##trmv##BLASPOSTFIX(&uplo, &trans, &diag, &n, (const BLASTYPE*)_lhs, &lda, (BLASTYPE*)x, &incx); \ -\ -/* Add op(a_tr)rhs into res*/ \ - BLASPREFIX##axpy##BLASPOSTFIX(&n, (const BLASTYPE*)&numext::real_ref(alpha),(const BLASTYPE*)x, &incx, (BLASTYPE*)_res, &incy); \ -/* Non-square case - doesn't fit to BLAS ?TRMV. Fall to default triangular product*/ \ - if (size<(std::max)(rows,cols)) { \ - if (ConjRhs) x_tmp = rhs.conjugate(); else x_tmp = rhs; \ - x = x_tmp.data(); \ - if (size(rows-size); \ - n = convert_index(size); \ - } \ - else { \ - x += size; \ - y = _res; \ - a = _lhs + size; \ - m = convert_index(size); \ - n = convert_index(cols-size); \ - } \ - BLASPREFIX##gemv##BLASPOSTFIX(&trans, &n, &m, (const BLASTYPE*)&numext::real_ref(alpha), (const BLASTYPE*)a, &lda, (const BLASTYPE*)x, &incx, (const BLASTYPE*)&numext::real_ref(beta), (BLASTYPE*)y, &incy); \ - } \ - } \ -}; +#define EIGEN_BLAS_TRMV_RM(EIGTYPE, BLASTYPE, EIGPREFIX, BLASPREFIX, BLASPOSTFIX) \ + template \ + struct triangular_matrix_vector_product_trmv \ + { \ + enum { \ + IsLower = (Mode & Lower) == Lower, \ + SetDiag = (Mode & (ZeroDiag | UnitDiag)) ? 0 : 1, \ + IsUnitDiag = (Mode & UnitDiag) ? 1 : 0, \ + IsZeroDiag = (Mode & ZeroDiag) ? 1 : 0, \ + LowUp = IsLower ? Lower : Upper \ + }; \ + static void run(Index _rows, \ + Index _cols, \ + const EIGTYPE *_lhs, \ + Index lhsStride, \ + const EIGTYPE *_rhs, \ + Index rhsIncr, \ + EIGTYPE *_res, \ + Index resIncr, \ + EIGTYPE alpha) \ + { \ + if (IsZeroDiag) { \ + triangular_matrix_vector_product::run( \ + _rows, _cols, _lhs, lhsStride, _rhs, rhsIncr, _res, resIncr, alpha); \ + return; \ + } \ + Index size = (std::min)(_rows, _cols); \ + Index rows = IsLower ? _rows : size; \ + Index cols = IsLower ? size : _cols; \ + \ + typedef VectorX##EIGPREFIX VectorRhs; \ + EIGTYPE *x, *y; \ + \ + /* Set x*/ \ + Map> rhs(_rhs, cols, InnerStride<>(rhsIncr)); \ + VectorRhs x_tmp; \ + if (ConjRhs) \ + x_tmp = rhs.conjugate(); \ + else \ + x_tmp = rhs; \ + x = x_tmp.data(); \ + \ + /* Square part handling */ \ + \ + char trans, uplo, diag; \ + BlasIndex m, n, lda, incx, incy; \ + EIGTYPE const *a; \ + EIGTYPE beta(1); \ + \ + /* Set m, n */ \ + n = convert_index(size); \ + lda = convert_index(lhsStride); \ + incx = 1; \ + incy = convert_index(resIncr); \ + \ + /* Set uplo, trans and diag*/ \ + trans = ConjLhs ? 'C' : 'T'; \ + uplo = IsLower ? 'U' : 'L'; \ + diag = IsUnitDiag ? 'U' : 'N'; \ + \ + /* call ?TRMV*/ \ + BLASPREFIX##trmv##BLASPOSTFIX(&uplo, &trans, &diag, &n, (const BLASTYPE *)_lhs, &lda, (BLASTYPE *)x, &incx); \ + \ + /* Add op(a_tr)rhs into res*/ \ + BLASPREFIX##axpy##BLASPOSTFIX( \ + &n, (const BLASTYPE *)&numext::real_ref(alpha), (const BLASTYPE *)x, &incx, (BLASTYPE *)_res, &incy); \ + /* Non-square case - doesn't fit to BLAS ?TRMV. Fall to default triangular product*/ \ + if (size < (std::max)(rows, cols)) { \ + if (ConjRhs) \ + x_tmp = rhs.conjugate(); \ + else \ + x_tmp = rhs; \ + x = x_tmp.data(); \ + if (size < rows) { \ + y = _res + size * resIncr; \ + a = _lhs + size * lda; \ + m = convert_index(rows - size); \ + n = convert_index(size); \ + } else { \ + x += size; \ + y = _res; \ + a = _lhs + size; \ + m = convert_index(size); \ + n = convert_index(cols - size); \ + } \ + BLASPREFIX##gemv##BLASPOSTFIX(&trans, \ + &n, \ + &m, \ + (const BLASTYPE *)&numext::real_ref(alpha), \ + (const BLASTYPE *)a, \ + &lda, \ + (const BLASTYPE *)x, \ + &incx, \ + (const BLASTYPE *)&numext::real_ref(beta), \ + (BLASTYPE *)y, \ + &incy); \ + } \ + } \ + }; #ifdef EIGEN_USE_MKL -EIGEN_BLAS_TRMV_RM(double, double, d, d,) -EIGEN_BLAS_TRMV_RM(dcomplex, MKL_Complex16, cd, z,) -EIGEN_BLAS_TRMV_RM(float, float, f, s,) -EIGEN_BLAS_TRMV_RM(scomplex, MKL_Complex8, cf, c,) + EIGEN_BLAS_TRMV_RM(double, double, d, d, ) + EIGEN_BLAS_TRMV_RM(dcomplex, MKL_Complex16, cd, z, ) + EIGEN_BLAS_TRMV_RM(float, float, f, s, ) + EIGEN_BLAS_TRMV_RM(scomplex, MKL_Complex8, cf, c, ) #else -EIGEN_BLAS_TRMV_RM(double, double, d, d,_) -EIGEN_BLAS_TRMV_RM(dcomplex, double, cd, z,_) -EIGEN_BLAS_TRMV_RM(float, float, f, s,_) -EIGEN_BLAS_TRMV_RM(scomplex, float, cf, c,_) + EIGEN_BLAS_TRMV_RM(double, double, d, d, _) + EIGEN_BLAS_TRMV_RM(dcomplex, double, cd, z, _) + EIGEN_BLAS_TRMV_RM(float, float, f, s, _) + EIGEN_BLAS_TRMV_RM(scomplex, float, cf, c, _) #endif -} // end namespase internal +}// namespace internal -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_TRIANGULAR_MATRIX_VECTOR_BLAS_H +#endif// EIGEN_TRIANGULAR_MATRIX_VECTOR_BLAS_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/products/TriangularSolverMatrix.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/products/TriangularSolverMatrix.h index 223c38b8..9fec7d49 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/products/TriangularSolverMatrix.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/products/TriangularSolverMatrix.h @@ -10,46 +10,54 @@ #ifndef EIGEN_TRIANGULAR_SOLVER_MATRIX_H #define EIGEN_TRIANGULAR_SOLVER_MATRIX_H -namespace Eigen { +namespace Eigen { namespace internal { -// if the rhs is row major, let's transpose the product -template -struct triangular_solve_matrix -{ - static void run( - Index size, Index cols, - const Scalar* tri, Index triStride, - Scalar* _other, Index otherStride, - level3_blocking& blocking) + // if the rhs is row major, let's transpose the product + template + struct triangular_solve_matrix { - triangular_solve_matrix< - Scalar, Index, Side==OnTheLeft?OnTheRight:OnTheLeft, - (Mode&UnitDiag) | ((Mode&Upper) ? Lower : Upper), - NumTraits::IsComplex && Conjugate, - TriStorageOrder==RowMajor ? ColMajor : RowMajor, ColMajor> - ::run(size, cols, tri, triStride, _other, otherStride, blocking); - } -}; - -/* Optimized triangular solver with multiple right hand side and the triangular matrix on the left - */ -template -struct triangular_solve_matrix -{ - static EIGEN_DONT_INLINE void run( - Index size, Index otherSize, - const Scalar* _tri, Index triStride, - Scalar* _other, Index otherStride, - level3_blocking& blocking); -}; -template -EIGEN_DONT_INLINE void triangular_solve_matrix::run( - Index size, Index otherSize, - const Scalar* _tri, Index triStride, - Scalar* _other, Index otherStride, - level3_blocking& blocking) + static void run(Index size, + Index cols, + const Scalar *tri, + Index triStride, + Scalar *_other, + Index otherStride, + level3_blocking &blocking) + { + triangular_solve_matrix::IsComplex && Conjugate, + TriStorageOrder == RowMajor ? ColMajor : RowMajor, + ColMajor>::run(size, cols, tri, triStride, _other, otherStride, blocking); + } + }; + + /* Optimized triangular solver with multiple right hand side and the triangular matrix on the left + */ + template + struct triangular_solve_matrix + { + static EIGEN_DONT_INLINE void run(Index size, + Index otherSize, + const Scalar *_tri, + Index triStride, + Scalar *_other, + Index otherStride, + level3_blocking &blocking); + }; + template + EIGEN_DONT_INLINE void + triangular_solve_matrix::run(Index size, + Index otherSize, + const Scalar *_tri, + Index triStride, + Scalar *_other, + Index otherStride, + level3_blocking &blocking) { Index cols = otherSize; @@ -58,18 +66,15 @@ EIGEN_DONT_INLINE void triangular_solve_matrix Traits; + typedef gebp_traits Traits; - enum { - SmallPanelWidth = EIGEN_PLAIN_ENUM_MAX(Traits::mr,Traits::nr), - IsLower = (Mode&Lower) == Lower - }; + enum { SmallPanelWidth = EIGEN_PLAIN_ENUM_MAX(Traits::mr, Traits::nr), IsLower = (Mode & Lower) == Lower }; - Index kc = blocking.kc(); // cache block size along the K direction - Index mc = (std::min)(size,blocking.mc()); // cache block size along the M direction + Index kc = blocking.kc();// cache block size along the K direction + Index mc = (std::min)(size, blocking.mc());// cache block size along the M direction - std::size_t sizeA = kc*mc; - std::size_t sizeB = kc*cols; + std::size_t sizeA = kc * mc; + std::size_t sizeB = kc * cols; ei_declare_aligned_stack_constructed_variable(Scalar, blockA, sizeA, blocking.blockA()); ei_declare_aligned_stack_constructed_variable(Scalar, blockB, sizeB, blocking.blockB()); @@ -83,14 +88,11 @@ EIGEN_DONT_INLINE void triangular_solve_matrix0 ? l2/(4 * sizeof(Scalar) * std::max(otherStride,size)) : 0; - subcols = std::max((subcols/Traits::nr)*Traits::nr, Traits::nr); + Index subcols = cols > 0 ? l2 / (4 * sizeof(Scalar) * std::max(otherStride, size)) : 0; + subcols = std::max((subcols / Traits::nr) * Traits::nr, Traits::nr); - for(Index k2=IsLower ? 0 : size; - IsLower ? k20; - IsLower ? k2+=kc : k2-=kc) - { - const Index actual_kc = (std::min)(IsLower ? size-k2 : k2, kc); + for (Index k2 = IsLower ? 0 : size; IsLower ? k2 < size : k2 > 0; IsLower ? k2 += kc : k2 -= kc) { + const Index actual_kc = (std::min)(IsLower ? size - k2 : k2, kc); // We have selected and packed a big horizontal panel R1 of rhs. Let B be the packed copy of this panel, // and R2 the remaining part of rhs. The corresponding vertical panel of lhs is split into @@ -105,101 +107,108 @@ EIGEN_DONT_INLINE void triangular_solve_matrix(actual_kc-k1, SmallPanelWidth); + for (Index k1 = 0; k1 < actual_kc; k1 += SmallPanelWidth) { + Index actualPanelWidth = std::min(actual_kc - k1, SmallPanelWidth); // tr solve - for (Index k=0; k0) - { - Index startTarget = IsLower ? k2+k1+actualPanelWidth : k2-actual_kc; - - pack_lhs(blockA, tri.getSubMapper(startTarget,startBlock), actualPanelWidth, lengthTarget); - - gebp_kernel(other.getSubMapper(startTarget,j2), blockA, blockB+actual_kc*j2, lengthTarget, actualPanelWidth, actual_cols, Scalar(-1), - actualPanelWidth, actual_kc, 0, blockBOffset); + if (lengthTarget > 0) { + Index startTarget = IsLower ? k2 + k1 + actualPanelWidth : k2 - actual_kc; + + pack_lhs(blockA, tri.getSubMapper(startTarget, startBlock), actualPanelWidth, lengthTarget); + + gebp_kernel(other.getSubMapper(startTarget, j2), + blockA, + blockB + actual_kc * j2, + lengthTarget, + actualPanelWidth, + actual_cols, + Scalar(-1), + actualPanelWidth, + actual_kc, + 0, + blockBOffset); } } } - + // R2 -= A21 * B => GEPP { - Index start = IsLower ? k2+kc : 0; - Index end = IsLower ? size : k2-kc; - for(Index i2=start; i20) - { - pack_lhs(blockA, tri.getSubMapper(i2, IsLower ? k2 : k2-kc), actual_kc, actual_mc); - - gebp_kernel(other.getSubMapper(i2, 0), blockA, blockB, actual_mc, actual_kc, cols, Scalar(-1), -1, -1, 0, 0); + Index start = IsLower ? k2 + kc : 0; + Index end = IsLower ? size : k2 - kc; + for (Index i2 = start; i2 < end; i2 += mc) { + const Index actual_mc = (std::min)(mc, end - i2); + if (actual_mc > 0) { + pack_lhs(blockA, tri.getSubMapper(i2, IsLower ? k2 : k2 - kc), actual_kc, actual_mc); + + gebp_kernel( + other.getSubMapper(i2, 0), blockA, blockB, actual_mc, actual_kc, cols, Scalar(-1), -1, -1, 0, 0); } } } } } -/* Optimized triangular solver with multiple left hand sides and the triangular matrix on the right - */ -template -struct triangular_solve_matrix -{ - static EIGEN_DONT_INLINE void run( - Index size, Index otherSize, - const Scalar* _tri, Index triStride, - Scalar* _other, Index otherStride, - level3_blocking& blocking); -}; -template -EIGEN_DONT_INLINE void triangular_solve_matrix::run( - Index size, Index otherSize, - const Scalar* _tri, Index triStride, - Scalar* _other, Index otherStride, - level3_blocking& blocking) + /* Optimized triangular solver with multiple left hand sides and the triangular matrix on the right + */ + template + struct triangular_solve_matrix + { + static EIGEN_DONT_INLINE void run(Index size, + Index otherSize, + const Scalar *_tri, + Index triStride, + Scalar *_other, + Index otherStride, + level3_blocking &blocking); + }; + template + EIGEN_DONT_INLINE void + triangular_solve_matrix::run(Index size, + Index otherSize, + const Scalar *_tri, + Index triStride, + Scalar *_other, + Index otherStride, + level3_blocking &blocking) { Index rows = otherSize; typedef typename NumTraits::Real RealScalar; @@ -209,18 +218,18 @@ EIGEN_DONT_INLINE void triangular_solve_matrix Traits; + typedef gebp_traits Traits; enum { - RhsStorageOrder = TriStorageOrder, - SmallPanelWidth = EIGEN_PLAIN_ENUM_MAX(Traits::mr,Traits::nr), - IsLower = (Mode&Lower) == Lower + RhsStorageOrder = TriStorageOrder, + SmallPanelWidth = EIGEN_PLAIN_ENUM_MAX(Traits::mr, Traits::nr), + IsLower = (Mode & Lower) == Lower }; - Index kc = blocking.kc(); // cache block size along the K direction - Index mc = (std::min)(rows,blocking.mc()); // cache block size along the M direction + Index kc = blocking.kc();// cache block size along the K direction + Index mc = (std::min)(rows, blocking.mc());// cache block size along the M direction - std::size_t sizeA = kc*mc; - std::size_t sizeB = kc*size; + std::size_t sizeA = kc * mc; + std::size_t sizeB = kc * size; ei_declare_aligned_stack_constructed_variable(Scalar, blockA, sizeA, blocking.blockA()); ei_declare_aligned_stack_constructed_variable(Scalar, blockB, sizeB, blocking.blockB()); @@ -228,108 +237,105 @@ EIGEN_DONT_INLINE void triangular_solve_matrix conj; gebp_kernel gebp_kernel; gemm_pack_rhs pack_rhs; - gemm_pack_rhs pack_rhs_panel; + gemm_pack_rhs pack_rhs_panel; gemm_pack_lhs pack_lhs_panel; - for(Index k2=IsLower ? size : 0; - IsLower ? k2>0 : k2 0 : k2 < size; IsLower ? k2 -= kc : k2 += kc) { + const Index actual_kc = (std::min)(IsLower ? k2 : size - k2, kc); + Index actual_k2 = IsLower ? k2 - actual_kc : k2; - Index startPanel = IsLower ? 0 : k2+actual_kc; + Index startPanel = IsLower ? 0 : k2 + actual_kc; Index rs = IsLower ? actual_k2 : size - actual_k2 - actual_kc; - Scalar* geb = blockB+actual_kc*actual_kc; + Scalar *geb = blockB + actual_kc * actual_kc; - if (rs>0) pack_rhs(geb, rhs.getSubMapper(actual_k2,startPanel), actual_kc, rs); + if (rs > 0) pack_rhs(geb, rhs.getSubMapper(actual_k2, startPanel), actual_kc, rs); // triangular packing (we only pack the panels off the diagonal, // neglecting the blocks overlapping the diagonal { - for (Index j2=0; j2(actual_kc-j2, SmallPanelWidth); + for (Index j2 = 0; j2 < actual_kc; j2 += SmallPanelWidth) { + Index actualPanelWidth = std::min(actual_kc - j2, SmallPanelWidth); Index actual_j2 = actual_k2 + j2; - Index panelOffset = IsLower ? j2+actualPanelWidth : 0; - Index panelLength = IsLower ? actual_kc-j2-actualPanelWidth : j2; - - if (panelLength>0) - pack_rhs_panel(blockB+j2*actual_kc, - rhs.getSubMapper(actual_k2+panelOffset, actual_j2), - panelLength, actualPanelWidth, - actual_kc, panelOffset); + Index panelOffset = IsLower ? j2 + actualPanelWidth : 0; + Index panelLength = IsLower ? actual_kc - j2 - actualPanelWidth : j2; + + if (panelLength > 0) + pack_rhs_panel(blockB + j2 * actual_kc, + rhs.getSubMapper(actual_k2 + panelOffset, actual_j2), + panelLength, + actualPanelWidth, + actual_kc, + panelOffset); } } - for(Index i2=0; i2 vertical panels of rhs) - for (Index j2 = IsLower - ? (actual_kc - ((actual_kc%SmallPanelWidth) ? Index(actual_kc%SmallPanelWidth) - : Index(SmallPanelWidth))) - : 0; - IsLower ? j2>=0 : j2(actual_kc-j2, SmallPanelWidth); + for (Index j2 = IsLower ? (actual_kc + - ((actual_kc % SmallPanelWidth) ? Index(actual_kc % SmallPanelWidth) + : Index(SmallPanelWidth))) + : 0; + IsLower ? j2 >= 0 : j2 < actual_kc; + IsLower ? j2 -= SmallPanelWidth : j2 += SmallPanelWidth) { + Index actualPanelWidth = std::min(actual_kc - j2, SmallPanelWidth); Index absolute_j2 = actual_k2 + j2; - Index panelOffset = IsLower ? j2+actualPanelWidth : 0; + Index panelOffset = IsLower ? j2 + actualPanelWidth : 0; Index panelLength = IsLower ? actual_kc - j2 - actualPanelWidth : j2; // GEBP - if(panelLength>0) - { - gebp_kernel(lhs.getSubMapper(i2,absolute_j2), - blockA, blockB+j2*actual_kc, - actual_mc, panelLength, actualPanelWidth, - Scalar(-1), - actual_kc, actual_kc, // strides - panelOffset, panelOffset); // offsets + if (panelLength > 0) { + gebp_kernel(lhs.getSubMapper(i2, absolute_j2), + blockA, + blockB + j2 * actual_kc, + actual_mc, + panelLength, + actualPanelWidth, + Scalar(-1), + actual_kc, + actual_kc,// strides + panelOffset, + panelOffset);// offsets } // unblocked triangular solve - for (Index k=0; k0) - gebp_kernel(lhs.getSubMapper(i2, startPanel), blockA, geb, - actual_mc, actual_kc, rs, Scalar(-1), - -1, -1, 0, 0); + if (rs > 0) + gebp_kernel( + lhs.getSubMapper(i2, startPanel), blockA, geb, actual_mc, actual_kc, rs, Scalar(-1), -1, -1, 0, 0); } } } -} // end namespace internal +}// end namespace internal -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_TRIANGULAR_SOLVER_MATRIX_H +#endif// EIGEN_TRIANGULAR_SOLVER_MATRIX_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/products/TriangularSolverMatrix_BLAS.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/products/TriangularSolverMatrix_BLAS.h index f0775116..79ceb646 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/products/TriangularSolverMatrix_BLAS.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/products/TriangularSolverMatrix_BLAS.h @@ -38,126 +38,152 @@ namespace Eigen { namespace internal { // implements LeftSide op(triangular)^-1 * general -#define EIGEN_BLAS_TRSM_L(EIGTYPE, BLASTYPE, BLASFUNC) \ -template \ -struct triangular_solve_matrix \ -{ \ - enum { \ - IsLower = (Mode&Lower) == Lower, \ - IsUnitDiag = (Mode&UnitDiag) ? 1 : 0, \ - IsZeroDiag = (Mode&ZeroDiag) ? 1 : 0, \ - conjA = ((TriStorageOrder==ColMajor) && Conjugate) ? 1 : 0 \ - }; \ - static void run( \ - Index size, Index otherSize, \ - const EIGTYPE* _tri, Index triStride, \ - EIGTYPE* _other, Index otherStride, level3_blocking& /*blocking*/) \ - { \ - BlasIndex m = convert_index(size), n = convert_index(otherSize), lda, ldb; \ - char side = 'L', uplo, diag='N', transa; \ - /* Set alpha_ */ \ - EIGTYPE alpha(1); \ - ldb = convert_index(otherStride);\ -\ - const EIGTYPE *a; \ -/* Set trans */ \ - transa = (TriStorageOrder==RowMajor) ? ((Conjugate) ? 'C' : 'T') : 'N'; \ -/* Set uplo */ \ - uplo = IsLower ? 'L' : 'U'; \ - if (TriStorageOrder==RowMajor) uplo = (uplo == 'L') ? 'U' : 'L'; \ -/* Set a, lda */ \ - typedef Matrix MatrixTri; \ - Map > tri(_tri,size,size,OuterStride<>(triStride)); \ - MatrixTri a_tmp; \ -\ - if (conjA) { \ - a_tmp = tri.conjugate(); \ - a = a_tmp.data(); \ - lda = convert_index(a_tmp.outerStride()); \ - } else { \ - a = _tri; \ - lda = convert_index(triStride); \ - } \ - if (IsUnitDiag) diag='U'; \ -/* call ?trsm*/ \ - BLASFUNC(&side, &uplo, &transa, &diag, &m, &n, (const BLASTYPE*)&numext::real_ref(alpha), (const BLASTYPE*)a, &lda, (BLASTYPE*)_other, &ldb); \ - } \ -}; +#define EIGEN_BLAS_TRSM_L(EIGTYPE, BLASTYPE, BLASFUNC) \ + template \ + struct triangular_solve_matrix \ + { \ + enum { \ + IsLower = (Mode & Lower) == Lower, \ + IsUnitDiag = (Mode & UnitDiag) ? 1 : 0, \ + IsZeroDiag = (Mode & ZeroDiag) ? 1 : 0, \ + conjA = ((TriStorageOrder == ColMajor) && Conjugate) ? 1 : 0 \ + }; \ + static void run(Index size, \ + Index otherSize, \ + const EIGTYPE *_tri, \ + Index triStride, \ + EIGTYPE *_other, \ + Index otherStride, \ + level3_blocking & /*blocking*/) \ + { \ + BlasIndex m = convert_index(size), n = convert_index(otherSize), lda, ldb; \ + char side = 'L', uplo, diag = 'N', transa; \ + /* Set alpha_ */ \ + EIGTYPE alpha(1); \ + ldb = convert_index(otherStride); \ + \ + const EIGTYPE *a; \ + /* Set trans */ \ + transa = (TriStorageOrder == RowMajor) ? ((Conjugate) ? 'C' : 'T') : 'N'; \ + /* Set uplo */ \ + uplo = IsLower ? 'L' : 'U'; \ + if (TriStorageOrder == RowMajor) uplo = (uplo == 'L') ? 'U' : 'L'; \ + /* Set a, lda */ \ + typedef Matrix MatrixTri; \ + Map> tri(_tri, size, size, OuterStride<>(triStride)); \ + MatrixTri a_tmp; \ + \ + if (conjA) { \ + a_tmp = tri.conjugate(); \ + a = a_tmp.data(); \ + lda = convert_index(a_tmp.outerStride()); \ + } else { \ + a = _tri; \ + lda = convert_index(triStride); \ + } \ + if (IsUnitDiag) diag = 'U'; \ + /* call ?trsm*/ \ + BLASFUNC(&side, \ + &uplo, \ + &transa, \ + &diag, \ + &m, \ + &n, \ + (const BLASTYPE *)&numext::real_ref(alpha), \ + (const BLASTYPE *)a, \ + &lda, \ + (BLASTYPE *)_other, \ + &ldb); \ + } \ + }; #ifdef EIGEN_USE_MKL -EIGEN_BLAS_TRSM_L(double, double, dtrsm) -EIGEN_BLAS_TRSM_L(dcomplex, MKL_Complex16, ztrsm) -EIGEN_BLAS_TRSM_L(float, float, strsm) -EIGEN_BLAS_TRSM_L(scomplex, MKL_Complex8, ctrsm) + EIGEN_BLAS_TRSM_L(double, double, dtrsm) + EIGEN_BLAS_TRSM_L(dcomplex, MKL_Complex16, ztrsm) + EIGEN_BLAS_TRSM_L(float, float, strsm) + EIGEN_BLAS_TRSM_L(scomplex, MKL_Complex8, ctrsm) #else -EIGEN_BLAS_TRSM_L(double, double, dtrsm_) -EIGEN_BLAS_TRSM_L(dcomplex, double, ztrsm_) -EIGEN_BLAS_TRSM_L(float, float, strsm_) -EIGEN_BLAS_TRSM_L(scomplex, float, ctrsm_) + EIGEN_BLAS_TRSM_L(double, double, dtrsm_) + EIGEN_BLAS_TRSM_L(dcomplex, double, ztrsm_) + EIGEN_BLAS_TRSM_L(float, float, strsm_) + EIGEN_BLAS_TRSM_L(scomplex, float, ctrsm_) #endif // implements RightSide general * op(triangular)^-1 -#define EIGEN_BLAS_TRSM_R(EIGTYPE, BLASTYPE, BLASFUNC) \ -template \ -struct triangular_solve_matrix \ -{ \ - enum { \ - IsLower = (Mode&Lower) == Lower, \ - IsUnitDiag = (Mode&UnitDiag) ? 1 : 0, \ - IsZeroDiag = (Mode&ZeroDiag) ? 1 : 0, \ - conjA = ((TriStorageOrder==ColMajor) && Conjugate) ? 1 : 0 \ - }; \ - static void run( \ - Index size, Index otherSize, \ - const EIGTYPE* _tri, Index triStride, \ - EIGTYPE* _other, Index otherStride, level3_blocking& /*blocking*/) \ - { \ - BlasIndex m = convert_index(otherSize), n = convert_index(size), lda, ldb; \ - char side = 'R', uplo, diag='N', transa; \ - /* Set alpha_ */ \ - EIGTYPE alpha(1); \ - ldb = convert_index(otherStride);\ -\ - const EIGTYPE *a; \ -/* Set trans */ \ - transa = (TriStorageOrder==RowMajor) ? ((Conjugate) ? 'C' : 'T') : 'N'; \ -/* Set uplo */ \ - uplo = IsLower ? 'L' : 'U'; \ - if (TriStorageOrder==RowMajor) uplo = (uplo == 'L') ? 'U' : 'L'; \ -/* Set a, lda */ \ - typedef Matrix MatrixTri; \ - Map > tri(_tri,size,size,OuterStride<>(triStride)); \ - MatrixTri a_tmp; \ -\ - if (conjA) { \ - a_tmp = tri.conjugate(); \ - a = a_tmp.data(); \ - lda = convert_index(a_tmp.outerStride()); \ - } else { \ - a = _tri; \ - lda = convert_index(triStride); \ - } \ - if (IsUnitDiag) diag='U'; \ -/* call ?trsm*/ \ - BLASFUNC(&side, &uplo, &transa, &diag, &m, &n, (const BLASTYPE*)&numext::real_ref(alpha), (const BLASTYPE*)a, &lda, (BLASTYPE*)_other, &ldb); \ - /*std::cout << "TRMS_L specialization!\n";*/ \ - } \ -}; +#define EIGEN_BLAS_TRSM_R(EIGTYPE, BLASTYPE, BLASFUNC) \ + template \ + struct triangular_solve_matrix \ + { \ + enum { \ + IsLower = (Mode & Lower) == Lower, \ + IsUnitDiag = (Mode & UnitDiag) ? 1 : 0, \ + IsZeroDiag = (Mode & ZeroDiag) ? 1 : 0, \ + conjA = ((TriStorageOrder == ColMajor) && Conjugate) ? 1 : 0 \ + }; \ + static void run(Index size, \ + Index otherSize, \ + const EIGTYPE *_tri, \ + Index triStride, \ + EIGTYPE *_other, \ + Index otherStride, \ + level3_blocking & /*blocking*/) \ + { \ + BlasIndex m = convert_index(otherSize), n = convert_index(size), lda, ldb; \ + char side = 'R', uplo, diag = 'N', transa; \ + /* Set alpha_ */ \ + EIGTYPE alpha(1); \ + ldb = convert_index(otherStride); \ + \ + const EIGTYPE *a; \ + /* Set trans */ \ + transa = (TriStorageOrder == RowMajor) ? ((Conjugate) ? 'C' : 'T') : 'N'; \ + /* Set uplo */ \ + uplo = IsLower ? 'L' : 'U'; \ + if (TriStorageOrder == RowMajor) uplo = (uplo == 'L') ? 'U' : 'L'; \ + /* Set a, lda */ \ + typedef Matrix MatrixTri; \ + Map> tri(_tri, size, size, OuterStride<>(triStride)); \ + MatrixTri a_tmp; \ + \ + if (conjA) { \ + a_tmp = tri.conjugate(); \ + a = a_tmp.data(); \ + lda = convert_index(a_tmp.outerStride()); \ + } else { \ + a = _tri; \ + lda = convert_index(triStride); \ + } \ + if (IsUnitDiag) diag = 'U'; \ + /* call ?trsm*/ \ + BLASFUNC(&side, \ + &uplo, \ + &transa, \ + &diag, \ + &m, \ + &n, \ + (const BLASTYPE *)&numext::real_ref(alpha), \ + (const BLASTYPE *)a, \ + &lda, \ + (BLASTYPE *)_other, \ + &ldb); \ + /*std::cout << "TRMS_L specialization!\n";*/ \ + } \ + }; #ifdef EIGEN_USE_MKL -EIGEN_BLAS_TRSM_R(double, double, dtrsm) -EIGEN_BLAS_TRSM_R(dcomplex, MKL_Complex16, ztrsm) -EIGEN_BLAS_TRSM_R(float, float, strsm) -EIGEN_BLAS_TRSM_R(scomplex, MKL_Complex8, ctrsm) + EIGEN_BLAS_TRSM_R(double, double, dtrsm) + EIGEN_BLAS_TRSM_R(dcomplex, MKL_Complex16, ztrsm) + EIGEN_BLAS_TRSM_R(float, float, strsm) + EIGEN_BLAS_TRSM_R(scomplex, MKL_Complex8, ctrsm) #else -EIGEN_BLAS_TRSM_R(double, double, dtrsm_) -EIGEN_BLAS_TRSM_R(dcomplex, double, ztrsm_) -EIGEN_BLAS_TRSM_R(float, float, strsm_) -EIGEN_BLAS_TRSM_R(scomplex, float, ctrsm_) + EIGEN_BLAS_TRSM_R(double, double, dtrsm_) + EIGEN_BLAS_TRSM_R(dcomplex, double, ztrsm_) + EIGEN_BLAS_TRSM_R(float, float, strsm_) + EIGEN_BLAS_TRSM_R(scomplex, float, ctrsm_) #endif -} // end namespace internal +}// end namespace internal -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_TRIANGULAR_SOLVER_MATRIX_BLAS_H +#endif// EIGEN_TRIANGULAR_SOLVER_MATRIX_BLAS_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/products/TriangularSolverVector.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/products/TriangularSolverVector.h index b994759b..740b6e44 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/products/TriangularSolverVector.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/products/TriangularSolverVector.h @@ -14,132 +14,122 @@ namespace Eigen { namespace internal { -template -struct triangular_solve_vector -{ - static void run(Index size, const LhsScalar* _lhs, Index lhsStride, RhsScalar* rhs) + template + struct triangular_solve_vector { - triangular_solve_vector::run(size, _lhs, lhsStride, rhs); - } -}; - -// forward and backward substitution, row-major, rhs is a vector -template -struct triangular_solve_vector -{ - enum { - IsLower = ((Mode&Lower)==Lower) + static void run(Index size, const LhsScalar *_lhs, Index lhsStride, RhsScalar *rhs) + { + triangular_solve_vector::run(size, _lhs, lhsStride, rhs); + } }; - static void run(Index size, const LhsScalar* _lhs, Index lhsStride, RhsScalar* rhs) + + // forward and backward substitution, row-major, rhs is a vector + template + struct triangular_solve_vector { - typedef Map, 0, OuterStride<> > LhsMap; - const LhsMap lhs(_lhs,size,size,OuterStride<>(lhsStride)); - - typedef const_blas_data_mapper LhsMapper; - typedef const_blas_data_mapper RhsMapper; - - typename internal::conditional< - Conjugate, - const CwiseUnaryOp,LhsMap>, - const LhsMap&> - ::type cjLhs(lhs); - static const Index PanelWidth = EIGEN_TUNE_TRIANGULAR_PANEL_WIDTH; - for(Index pi=IsLower ? 0 : size; - IsLower ? pi0; - IsLower ? pi+=PanelWidth : pi-=PanelWidth) + enum { IsLower = ((Mode & Lower) == Lower) }; + static void run(Index size, const LhsScalar *_lhs, Index lhsStride, RhsScalar *rhs) { - Index actualPanelWidth = (std::min)(IsLower ? size - pi : pi, PanelWidth); - - Index r = IsLower ? pi : size - pi; // remaining size - if (r > 0) - { - // let's directly call the low level product function because: - // 1 - it is faster to compile - // 2 - it is slighlty faster at runtime - Index startRow = IsLower ? pi : pi-actualPanelWidth; - Index startCol = IsLower ? 0 : pi; - - general_matrix_vector_product::run( - actualPanelWidth, r, - LhsMapper(&lhs.coeffRef(startRow,startCol), lhsStride), - RhsMapper(rhs + startCol, 1), - rhs + startRow, 1, - RhsScalar(-1)); - } - - for(Index k=0; k0) - rhs[i] -= (cjLhs.row(i).segment(s,k).transpose().cwiseProduct(Map >(rhs+s,k))).sum(); - - if(!(Mode & UnitDiag)) - rhs[i] /= cjLhs(i,i); + typedef Map, 0, OuterStride<>> LhsMap; + const LhsMap lhs(_lhs, size, size, OuterStride<>(lhsStride)); + + typedef const_blas_data_mapper LhsMapper; + typedef const_blas_data_mapper RhsMapper; + + typename internal::conditional, LhsMap>, + const LhsMap &>::type cjLhs(lhs); + static const Index PanelWidth = EIGEN_TUNE_TRIANGULAR_PANEL_WIDTH; + for (Index pi = IsLower ? 0 : size; IsLower ? pi < size : pi > 0; IsLower ? pi += PanelWidth : pi -= PanelWidth) { + Index actualPanelWidth = (std::min)(IsLower ? size - pi : pi, PanelWidth); + + Index r = IsLower ? pi : size - pi;// remaining size + if (r > 0) { + // let's directly call the low level product function because: + // 1 - it is faster to compile + // 2 - it is slighlty faster at runtime + Index startRow = IsLower ? pi : pi - actualPanelWidth; + Index startCol = IsLower ? 0 : pi; + + general_matrix_vector_product:: + run(actualPanelWidth, + r, + LhsMapper(&lhs.coeffRef(startRow, startCol), lhsStride), + RhsMapper(rhs + startCol, 1), + rhs + startRow, + 1, + RhsScalar(-1)); + } + + for (Index k = 0; k < actualPanelWidth; ++k) { + Index i = IsLower ? pi + k : pi - k - 1; + Index s = IsLower ? pi : i + 1; + if (k > 0) + rhs[i] -= (cjLhs.row(i).segment(s, k).transpose().cwiseProduct( + Map>(rhs + s, k))) + .sum(); + + if (!(Mode & UnitDiag)) rhs[i] /= cjLhs(i, i); + } } } - } -}; - -// forward and backward substitution, column-major, rhs is a vector -template -struct triangular_solve_vector -{ - enum { - IsLower = ((Mode&Lower)==Lower) }; - static void run(Index size, const LhsScalar* _lhs, Index lhsStride, RhsScalar* rhs) + + // forward and backward substitution, column-major, rhs is a vector + template + struct triangular_solve_vector { - typedef Map, 0, OuterStride<> > LhsMap; - const LhsMap lhs(_lhs,size,size,OuterStride<>(lhsStride)); - typedef const_blas_data_mapper LhsMapper; - typedef const_blas_data_mapper RhsMapper; - typename internal::conditional,LhsMap>, - const LhsMap& - >::type cjLhs(lhs); - static const Index PanelWidth = EIGEN_TUNE_TRIANGULAR_PANEL_WIDTH; - - for(Index pi=IsLower ? 0 : size; - IsLower ? pi0; - IsLower ? pi+=PanelWidth : pi-=PanelWidth) + enum { IsLower = ((Mode & Lower) == Lower) }; + static void run(Index size, const LhsScalar *_lhs, Index lhsStride, RhsScalar *rhs) { - Index actualPanelWidth = (std::min)(IsLower ? size - pi : pi, PanelWidth); - Index startBlock = IsLower ? pi : pi-actualPanelWidth; - Index endBlock = IsLower ? pi + actualPanelWidth : 0; - - for(Index k=0; k0) - Map >(rhs+s,r) -= rhs[i] * cjLhs.col(i).segment(s,r); - } - Index r = IsLower ? size - endBlock : startBlock; // remaining size - if (r > 0) - { - // let's directly call the low level product function because: - // 1 - it is faster to compile - // 2 - it is slighlty faster at runtime - general_matrix_vector_product::run( - r, actualPanelWidth, - LhsMapper(&lhs.coeffRef(endBlock,startBlock), lhsStride), - RhsMapper(rhs+startBlock, 1), - rhs+endBlock, 1, RhsScalar(-1)); + typedef Map, 0, OuterStride<>> LhsMap; + const LhsMap lhs(_lhs, size, size, OuterStride<>(lhsStride)); + typedef const_blas_data_mapper LhsMapper; + typedef const_blas_data_mapper RhsMapper; + typename internal::conditional, LhsMap>, + const LhsMap &>::type cjLhs(lhs); + static const Index PanelWidth = EIGEN_TUNE_TRIANGULAR_PANEL_WIDTH; + + for (Index pi = IsLower ? 0 : size; IsLower ? pi < size : pi > 0; IsLower ? pi += PanelWidth : pi -= PanelWidth) { + Index actualPanelWidth = (std::min)(IsLower ? size - pi : pi, PanelWidth); + Index startBlock = IsLower ? pi : pi - actualPanelWidth; + Index endBlock = IsLower ? pi + actualPanelWidth : 0; + + for (Index k = 0; k < actualPanelWidth; ++k) { + Index i = IsLower ? pi + k : pi - k - 1; + if (!(Mode & UnitDiag)) rhs[i] /= cjLhs.coeff(i, i); + + Index r = actualPanelWidth - k - 1;// remaining size + Index s = IsLower ? i + 1 : i - r; + if (r > 0) Map>(rhs + s, r) -= rhs[i] * cjLhs.col(i).segment(s, r); + } + Index r = IsLower ? size - endBlock : startBlock;// remaining size + if (r > 0) { + // let's directly call the low level product function because: + // 1 - it is faster to compile + // 2 - it is slighlty faster at runtime + general_matrix_vector_product:: + run(r, + actualPanelWidth, + LhsMapper(&lhs.coeffRef(endBlock, startBlock), lhsStride), + RhsMapper(rhs + startBlock, 1), + rhs + endBlock, + 1, + RhsScalar(-1)); + } } } - } -}; + }; -} // end namespace internal +}// end namespace internal -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_TRIANGULAR_SOLVER_VECTOR_H +#endif// EIGEN_TRIANGULAR_SOLVER_VECTOR_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/util/BlasUtil.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/util/BlasUtil.h old mode 100755 new mode 100644 index 6e6ee119..b8a6bfaf --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/util/BlasUtil.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/util/BlasUtil.h @@ -17,382 +17,437 @@ namespace Eigen { namespace internal { -// forward declarations -template -struct gebp_kernel; - -template -struct gemm_pack_rhs; - -template -struct gemm_pack_lhs; - -template< - typename Index, - typename LhsScalar, int LhsStorageOrder, bool ConjugateLhs, - typename RhsScalar, int RhsStorageOrder, bool ConjugateRhs, - int ResStorageOrder> -struct general_matrix_matrix_product; - -template -struct general_matrix_vector_product; - - -template struct conj_if; - -template<> struct conj_if { - template - inline T operator()(const T& x) const { return numext::conj(x); } - template - inline T pconj(const T& x) const { return internal::pconj(x); } -}; - -template<> struct conj_if { - template - inline const T& operator()(const T& x) const { return x; } - template - inline const T& pconj(const T& x) const { return x; } -}; - -// Generic implementation for custom complex types. -template -struct conj_helper -{ - typedef typename ScalarBinaryOpTraits::ReturnType Scalar; - - EIGEN_STRONG_INLINE Scalar pmadd(const LhsScalar& x, const RhsScalar& y, const Scalar& c) const - { return padd(c, pmul(x,y)); } - - EIGEN_STRONG_INLINE Scalar pmul(const LhsScalar& x, const RhsScalar& y) const - { return conj_if()(x) * conj_if()(y); } -}; - -template struct conj_helper -{ - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Scalar pmadd(const Scalar& x, const Scalar& y, const Scalar& c) const { return internal::pmadd(x,y,c); } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Scalar pmul(const Scalar& x, const Scalar& y) const { return internal::pmul(x,y); } -}; - -template struct conj_helper, std::complex, false,true> -{ - typedef std::complex Scalar; - EIGEN_STRONG_INLINE Scalar pmadd(const Scalar& x, const Scalar& y, const Scalar& c) const - { return c + pmul(x,y); } - - EIGEN_STRONG_INLINE Scalar pmul(const Scalar& x, const Scalar& y) const - { return Scalar(numext::real(x)*numext::real(y) + numext::imag(x)*numext::imag(y), numext::imag(x)*numext::real(y) - numext::real(x)*numext::imag(y)); } -}; - -template struct conj_helper, std::complex, true,false> -{ - typedef std::complex Scalar; - EIGEN_STRONG_INLINE Scalar pmadd(const Scalar& x, const Scalar& y, const Scalar& c) const - { return c + pmul(x,y); } - - EIGEN_STRONG_INLINE Scalar pmul(const Scalar& x, const Scalar& y) const - { return Scalar(numext::real(x)*numext::real(y) + numext::imag(x)*numext::imag(y), numext::real(x)*numext::imag(y) - numext::imag(x)*numext::real(y)); } -}; - -template struct conj_helper, std::complex, true,true> -{ - typedef std::complex Scalar; - EIGEN_STRONG_INLINE Scalar pmadd(const Scalar& x, const Scalar& y, const Scalar& c) const - { return c + pmul(x,y); } - - EIGEN_STRONG_INLINE Scalar pmul(const Scalar& x, const Scalar& y) const - { return Scalar(numext::real(x)*numext::real(y) - numext::imag(x)*numext::imag(y), - numext::real(x)*numext::imag(y) - numext::imag(x)*numext::real(y)); } -}; - -template struct conj_helper, RealScalar, Conj,false> -{ - typedef std::complex Scalar; - EIGEN_STRONG_INLINE Scalar pmadd(const Scalar& x, const RealScalar& y, const Scalar& c) const - { return padd(c, pmul(x,y)); } - EIGEN_STRONG_INLINE Scalar pmul(const Scalar& x, const RealScalar& y) const - { return conj_if()(x)*y; } -}; - -template struct conj_helper, false,Conj> -{ - typedef std::complex Scalar; - EIGEN_STRONG_INLINE Scalar pmadd(const RealScalar& x, const Scalar& y, const Scalar& c) const - { return padd(c, pmul(x,y)); } - EIGEN_STRONG_INLINE Scalar pmul(const RealScalar& x, const Scalar& y) const - { return x*conj_if()(y); } -}; - -template struct get_factor { - EIGEN_DEVICE_FUNC static EIGEN_STRONG_INLINE To run(const From& x) { return To(x); } -}; - -template struct get_factor::Real> { - EIGEN_DEVICE_FUNC - static EIGEN_STRONG_INLINE typename NumTraits::Real run(const Scalar& x) { return numext::real(x); } -}; - - -template -class BlasVectorMapper { + // forward declarations + template + struct gebp_kernel; + + template + struct gemm_pack_rhs; + + template + struct gemm_pack_lhs; + + template + struct general_matrix_matrix_product; + + template + struct general_matrix_vector_product; + + + template struct conj_if; + + template<> struct conj_if + { + template inline T operator()(const T &x) const { return numext::conj(x); } + template inline T pconj(const T &x) const { return internal::pconj(x); } + }; + + template<> struct conj_if + { + template inline const T &operator()(const T &x) const { return x; } + template inline const T &pconj(const T &x) const { return x; } + }; + + // Generic implementation for custom complex types. + template struct conj_helper + { + typedef typename ScalarBinaryOpTraits::ReturnType Scalar; + + EIGEN_STRONG_INLINE Scalar pmadd(const LhsScalar &x, const RhsScalar &y, const Scalar &c) const + { + return padd(c, pmul(x, y)); + } + + EIGEN_STRONG_INLINE Scalar pmul(const LhsScalar &x, const RhsScalar &y) const + { + return conj_if()(x) * conj_if()(y); + } + }; + + template struct conj_helper + { + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Scalar pmadd(const Scalar &x, const Scalar &y, const Scalar &c) const + { + return internal::pmadd(x, y, c); + } + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Scalar pmul(const Scalar &x, const Scalar &y) const + { + return internal::pmul(x, y); + } + }; + + template struct conj_helper, std::complex, false, true> + { + typedef std::complex Scalar; + EIGEN_STRONG_INLINE Scalar pmadd(const Scalar &x, const Scalar &y, const Scalar &c) const { return c + pmul(x, y); } + + EIGEN_STRONG_INLINE Scalar pmul(const Scalar &x, const Scalar &y) const + { + return Scalar(numext::real(x) * numext::real(y) + numext::imag(x) * numext::imag(y), + numext::imag(x) * numext::real(y) - numext::real(x) * numext::imag(y)); + } + }; + + template struct conj_helper, std::complex, true, false> + { + typedef std::complex Scalar; + EIGEN_STRONG_INLINE Scalar pmadd(const Scalar &x, const Scalar &y, const Scalar &c) const { return c + pmul(x, y); } + + EIGEN_STRONG_INLINE Scalar pmul(const Scalar &x, const Scalar &y) const + { + return Scalar(numext::real(x) * numext::real(y) + numext::imag(x) * numext::imag(y), + numext::real(x) * numext::imag(y) - numext::imag(x) * numext::real(y)); + } + }; + + template struct conj_helper, std::complex, true, true> + { + typedef std::complex Scalar; + EIGEN_STRONG_INLINE Scalar pmadd(const Scalar &x, const Scalar &y, const Scalar &c) const { return c + pmul(x, y); } + + EIGEN_STRONG_INLINE Scalar pmul(const Scalar &x, const Scalar &y) const + { + return Scalar(numext::real(x) * numext::real(y) - numext::imag(x) * numext::imag(y), + -numext::real(x) * numext::imag(y) - numext::imag(x) * numext::real(y)); + } + }; + + template struct conj_helper, RealScalar, Conj, false> + { + typedef std::complex Scalar; + EIGEN_STRONG_INLINE Scalar pmadd(const Scalar &x, const RealScalar &y, const Scalar &c) const + { + return padd(c, pmul(x, y)); + } + EIGEN_STRONG_INLINE Scalar pmul(const Scalar &x, const RealScalar &y) const { return conj_if()(x) * y; } + }; + + template struct conj_helper, false, Conj> + { + typedef std::complex Scalar; + EIGEN_STRONG_INLINE Scalar pmadd(const RealScalar &x, const Scalar &y, const Scalar &c) const + { + return padd(c, pmul(x, y)); + } + EIGEN_STRONG_INLINE Scalar pmul(const RealScalar &x, const Scalar &y) const { return x * conj_if()(y); } + }; + + template struct get_factor + { + EIGEN_DEVICE_FUNC static EIGEN_STRONG_INLINE To run(const From &x) { return To(x); } + }; + + template struct get_factor::Real> + { + EIGEN_DEVICE_FUNC + static EIGEN_STRONG_INLINE typename NumTraits::Real run(const Scalar &x) { return numext::real(x); } + }; + + + template class BlasVectorMapper + { public: - EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE BlasVectorMapper(Scalar *data) : m_data(data) {} + EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE BlasVectorMapper(Scalar *data) : m_data(data) {} - EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE Scalar operator()(Index i) const { - return m_data[i]; - } - template - EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE Packet load(Index i) const { - return ploadt(m_data + i); - } + EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE Scalar operator()(Index i) const { return m_data[i]; } + template EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE Packet load(Index i) const + { + return ploadt(m_data + i); + } - template - EIGEN_DEVICE_FUNC bool aligned(Index i) const { - return (UIntPtr(m_data+i)%sizeof(Packet))==0; - } + template EIGEN_DEVICE_FUNC bool aligned(Index i) const + { + return (UIntPtr(m_data + i) % sizeof(Packet)) == 0; + } protected: - Scalar* m_data; -}; + Scalar *m_data; + }; -template -class BlasLinearMapper { + template class BlasLinearMapper + { public: - typedef typename packet_traits::type Packet; - typedef typename packet_traits::half HalfPacket; + typedef typename packet_traits::type Packet; + typedef typename packet_traits::half HalfPacket; - EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE BlasLinearMapper(Scalar *data) : m_data(data) {} + EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE BlasLinearMapper(Scalar *data) : m_data(data) {} - EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE void prefetch(int i) const { - internal::prefetch(&operator()(i)); - } + EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE void prefetch(int i) const { internal::prefetch(&operator()(i)); } - EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE Scalar& operator()(Index i) const { - return m_data[i]; - } + EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE Scalar &operator()(Index i) const { return m_data[i]; } - EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE Packet loadPacket(Index i) const { - return ploadt(m_data + i); - } + EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE Packet loadPacket(Index i) const + { + return ploadt(m_data + i); + } - EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE HalfPacket loadHalfPacket(Index i) const { - return ploadt(m_data + i); - } + EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE HalfPacket loadHalfPacket(Index i) const + { + return ploadt(m_data + i); + } - EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE void storePacket(Index i, const Packet &p) const { - pstoret(m_data + i, p); - } + EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE void storePacket(Index i, const Packet &p) const + { + pstoret(m_data + i, p); + } protected: - Scalar *m_data; -}; + Scalar *m_data; + }; -// Lightweight helper class to access matrix coefficients. -template -class blas_data_mapper { + // Lightweight helper class to access matrix coefficients. + template class blas_data_mapper + { public: - typedef typename packet_traits::type Packet; - typedef typename packet_traits::half HalfPacket; + typedef typename packet_traits::type Packet; + typedef typename packet_traits::half HalfPacket; - typedef BlasLinearMapper LinearMapper; - typedef BlasVectorMapper VectorMapper; + typedef BlasLinearMapper LinearMapper; + typedef BlasVectorMapper VectorMapper; - EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE blas_data_mapper(Scalar* data, Index stride) : m_data(data), m_stride(stride) {} + EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE blas_data_mapper(Scalar *data, Index stride) : m_data(data), m_stride(stride) + {} - EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE blas_data_mapper - getSubMapper(Index i, Index j) const { - return blas_data_mapper(&operator()(i, j), m_stride); - } + EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE blas_data_mapper + getSubMapper(Index i, Index j) const + { + return blas_data_mapper(&operator()(i, j), m_stride); + } - EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE LinearMapper getLinearMapper(Index i, Index j) const { - return LinearMapper(&operator()(i, j)); - } + EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE LinearMapper getLinearMapper(Index i, Index j) const + { + return LinearMapper(&operator()(i, j)); + } - EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE VectorMapper getVectorMapper(Index i, Index j) const { - return VectorMapper(&operator()(i, j)); - } + EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE VectorMapper getVectorMapper(Index i, Index j) const + { + return VectorMapper(&operator()(i, j)); + } - EIGEN_DEVICE_FUNC - EIGEN_ALWAYS_INLINE Scalar& operator()(Index i, Index j) const { - return m_data[StorageOrder==RowMajor ? j + i*m_stride : i + j*m_stride]; - } + EIGEN_DEVICE_FUNC + EIGEN_ALWAYS_INLINE Scalar &operator()(Index i, Index j) const + { + return m_data[StorageOrder == RowMajor ? j + i * m_stride : i + j * m_stride]; + } - EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE Packet loadPacket(Index i, Index j) const { - return ploadt(&operator()(i, j)); - } + EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE Packet loadPacket(Index i, Index j) const + { + return ploadt(&operator()(i, j)); + } - EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE HalfPacket loadHalfPacket(Index i, Index j) const { - return ploadt(&operator()(i, j)); - } + EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE HalfPacket loadHalfPacket(Index i, Index j) const + { + return ploadt(&operator()(i, j)); + } - template - EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE void scatterPacket(Index i, Index j, const SubPacket &p) const { - pscatter(&operator()(i, j), p, m_stride); - } + template + EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE void scatterPacket(Index i, Index j, const SubPacket &p) const + { + pscatter(&operator()(i, j), p, m_stride); + } - template - EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE SubPacket gatherPacket(Index i, Index j) const { - return pgather(&operator()(i, j), m_stride); - } + template EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE SubPacket gatherPacket(Index i, Index j) const + { + return pgather(&operator()(i, j), m_stride); + } - EIGEN_DEVICE_FUNC const Index stride() const { return m_stride; } - EIGEN_DEVICE_FUNC const Scalar* data() const { return m_data; } + EIGEN_DEVICE_FUNC const Index stride() const { return m_stride; } + EIGEN_DEVICE_FUNC const Scalar *data() const { return m_data; } - EIGEN_DEVICE_FUNC Index firstAligned(Index size) const { - if (UIntPtr(m_data)%sizeof(Scalar)) { - return -1; + EIGEN_DEVICE_FUNC Index firstAligned(Index size) const + { + if (UIntPtr(m_data) % sizeof(Scalar)) { return -1; } + return internal::first_default_aligned(m_data, size); } - return internal::first_default_aligned(m_data, size); - } protected: - Scalar* EIGEN_RESTRICT m_data; - const Index m_stride; -}; + Scalar *EIGEN_RESTRICT m_data; + const Index m_stride; + }; -// lightweight helper class to access matrix coefficients (const version) -template -class const_blas_data_mapper : public blas_data_mapper { + // lightweight helper class to access matrix coefficients (const version) + template + class const_blas_data_mapper : public blas_data_mapper + { public: - EIGEN_ALWAYS_INLINE const_blas_data_mapper(const Scalar *data, Index stride) : blas_data_mapper(data, stride) {} - - EIGEN_ALWAYS_INLINE const_blas_data_mapper getSubMapper(Index i, Index j) const { - return const_blas_data_mapper(&(this->operator()(i, j)), this->m_stride); - } -}; - - -/* Helper class to analyze the factors of a Product expression. - * In particular it allows to pop out operator-, scalar multiples, - * and conjugate */ -template struct blas_traits -{ - typedef typename traits::Scalar Scalar; - typedef const XprType& ExtractType; - typedef XprType _ExtractType; - enum { - IsComplex = NumTraits::IsComplex, - IsTransposed = false, - NeedToConjugate = false, - HasUsableDirectAccess = ( (int(XprType::Flags)&DirectAccessBit) - && ( bool(XprType::IsVectorAtCompileTime) - || int(inner_stride_at_compile_time::ret) == 1) - ) ? 1 : 0 + EIGEN_ALWAYS_INLINE const_blas_data_mapper(const Scalar *data, Index stride) + : blas_data_mapper(data, stride) + {} + + EIGEN_ALWAYS_INLINE const_blas_data_mapper getSubMapper(Index i, Index j) const + { + return const_blas_data_mapper(&(this->operator()(i, j)), this->m_stride); + } + }; + + + /* Helper class to analyze the factors of a Product expression. + * In particular it allows to pop out operator-, scalar multiples, + * and conjugate */ + template struct blas_traits + { + typedef typename traits::Scalar Scalar; + typedef const XprType &ExtractType; + typedef XprType _ExtractType; + enum { + IsComplex = NumTraits::IsComplex, + IsTransposed = false, + NeedToConjugate = false, + HasUsableDirectAccess = + ((int(XprType::Flags) & DirectAccessBit) + && (bool(XprType::IsVectorAtCompileTime) || int(inner_stride_at_compile_time::ret) == 1)) + ? 1 + : 0 + }; + typedef typename conditional::type + DirectLinearAccessType; + static inline ExtractType extract(const XprType &x) { return x; } + static inline const Scalar extractScalarFactor(const XprType &) { return Scalar(1); } + }; + + // pop conjugate + template + struct blas_traits, NestedXpr>> : blas_traits + { + typedef blas_traits Base; + typedef CwiseUnaryOp, NestedXpr> XprType; + typedef typename Base::ExtractType ExtractType; + + enum { IsComplex = NumTraits::IsComplex, NeedToConjugate = Base::NeedToConjugate ? 0 : IsComplex }; + static inline ExtractType extract(const XprType &x) { return Base::extract(x.nestedExpression()); } + static inline Scalar extractScalarFactor(const XprType &x) + { + return conj(Base::extractScalarFactor(x.nestedExpression())); + } }; - typedef typename conditional::type DirectLinearAccessType; - static inline ExtractType extract(const XprType& x) { return x; } - static inline const Scalar extractScalarFactor(const XprType&) { return Scalar(1); } -}; - -// pop conjugate -template -struct blas_traits, NestedXpr> > - : blas_traits -{ - typedef blas_traits Base; - typedef CwiseUnaryOp, NestedXpr> XprType; - typedef typename Base::ExtractType ExtractType; - - enum { - IsComplex = NumTraits::IsComplex, - NeedToConjugate = Base::NeedToConjugate ? 0 : IsComplex + + // pop scalar multiple + template + struct blas_traits< + CwiseBinaryOp, const CwiseNullaryOp, Plain>, NestedXpr>> + : blas_traits + { + typedef blas_traits Base; + typedef CwiseBinaryOp, const CwiseNullaryOp, Plain>, NestedXpr> + XprType; + typedef typename Base::ExtractType ExtractType; + static inline ExtractType extract(const XprType &x) { return Base::extract(x.rhs()); } + static inline Scalar extractScalarFactor(const XprType &x) + { + return x.lhs().functor().m_other * Base::extractScalarFactor(x.rhs()); + } }; - static inline ExtractType extract(const XprType& x) { return Base::extract(x.nestedExpression()); } - static inline Scalar extractScalarFactor(const XprType& x) { return conj(Base::extractScalarFactor(x.nestedExpression())); } -}; - -// pop scalar multiple -template -struct blas_traits, const CwiseNullaryOp,Plain>, NestedXpr> > - : blas_traits -{ - typedef blas_traits Base; - typedef CwiseBinaryOp, const CwiseNullaryOp,Plain>, NestedXpr> XprType; - typedef typename Base::ExtractType ExtractType; - static inline ExtractType extract(const XprType& x) { return Base::extract(x.rhs()); } - static inline Scalar extractScalarFactor(const XprType& x) - { return x.lhs().functor().m_other * Base::extractScalarFactor(x.rhs()); } -}; -template -struct blas_traits, NestedXpr, const CwiseNullaryOp,Plain> > > - : blas_traits -{ - typedef blas_traits Base; - typedef CwiseBinaryOp, NestedXpr, const CwiseNullaryOp,Plain> > XprType; - typedef typename Base::ExtractType ExtractType; - static inline ExtractType extract(const XprType& x) { return Base::extract(x.lhs()); } - static inline Scalar extractScalarFactor(const XprType& x) - { return Base::extractScalarFactor(x.lhs()) * x.rhs().functor().m_other; } -}; -template -struct blas_traits, const CwiseNullaryOp,Plain1>, - const CwiseNullaryOp,Plain2> > > - : blas_traits,Plain1> > -{}; - -// pop opposite -template -struct blas_traits, NestedXpr> > - : blas_traits -{ - typedef blas_traits Base; - typedef CwiseUnaryOp, NestedXpr> XprType; - typedef typename Base::ExtractType ExtractType; - static inline ExtractType extract(const XprType& x) { return Base::extract(x.nestedExpression()); } - static inline Scalar extractScalarFactor(const XprType& x) - { return - Base::extractScalarFactor(x.nestedExpression()); } -}; - -// pop/push transpose -template -struct blas_traits > - : blas_traits -{ - typedef typename NestedXpr::Scalar Scalar; - typedef blas_traits Base; - typedef Transpose XprType; - typedef Transpose ExtractType; // const to get rid of a compile error; anyway blas traits are only used on the RHS - typedef Transpose _ExtractType; - typedef typename conditional::type DirectLinearAccessType; - enum { - IsTransposed = Base::IsTransposed ? 0 : 1 + template + struct blas_traits< + CwiseBinaryOp, NestedXpr, const CwiseNullaryOp, Plain>>> + : blas_traits + { + typedef blas_traits Base; + typedef CwiseBinaryOp, NestedXpr, const CwiseNullaryOp, Plain>> + XprType; + typedef typename Base::ExtractType ExtractType; + static inline ExtractType extract(const XprType &x) { return Base::extract(x.lhs()); } + static inline Scalar extractScalarFactor(const XprType &x) + { + return Base::extractScalarFactor(x.lhs()) * x.rhs().functor().m_other; + } + }; + template + struct blas_traits, + const CwiseNullaryOp, Plain1>, + const CwiseNullaryOp, Plain2>>> + : blas_traits, Plain1>> + { }; - static inline ExtractType extract(const XprType& x) { return ExtractType(Base::extract(x.nestedExpression())); } - static inline Scalar extractScalarFactor(const XprType& x) { return Base::extractScalarFactor(x.nestedExpression()); } -}; - -template -struct blas_traits - : blas_traits -{}; - -template::HasUsableDirectAccess> -struct extract_data_selector { - static const typename T::Scalar* run(const T& m) + + // pop opposite + template + struct blas_traits, NestedXpr>> : blas_traits { - return blas_traits::extract(m).data(); - } -}; + typedef blas_traits Base; + typedef CwiseUnaryOp, NestedXpr> XprType; + typedef typename Base::ExtractType ExtractType; + static inline ExtractType extract(const XprType &x) { return Base::extract(x.nestedExpression()); } + static inline Scalar extractScalarFactor(const XprType &x) + { + return -Base::extractScalarFactor(x.nestedExpression()); + } + }; -template -struct extract_data_selector { - static typename T::Scalar* run(const T&) { return 0; } -}; + // pop/push transpose + template struct blas_traits> : blas_traits + { + typedef typename NestedXpr::Scalar Scalar; + typedef blas_traits Base; + typedef Transpose XprType; + typedef Transpose + ExtractType;// const to get rid of a compile error; anyway blas traits are only used on the RHS + typedef Transpose _ExtractType; + typedef + typename conditional::type + DirectLinearAccessType; + enum { IsTransposed = Base::IsTransposed ? 0 : 1 }; + static inline ExtractType extract(const XprType &x) { return ExtractType(Base::extract(x.nestedExpression())); } + static inline Scalar extractScalarFactor(const XprType &x) + { + return Base::extractScalarFactor(x.nestedExpression()); + } + }; + + template struct blas_traits : blas_traits + { + }; + + template::HasUsableDirectAccess> struct extract_data_selector + { + static const typename T::Scalar *run(const T &m) { return blas_traits::extract(m).data(); } + }; + + template struct extract_data_selector + { + static typename T::Scalar *run(const T &) { return 0; } + }; -template const typename T::Scalar* extract_data(const T& m) -{ - return extract_data_selector::run(m); -} + template const typename T::Scalar *extract_data(const T &m) { return extract_data_selector::run(m); } -} // end namespace internal +}// end namespace internal -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_BLASUTIL_H +#endif// EIGEN_BLASUTIL_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/util/Constants.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/util/Constants.h index 7587d684..6b9db248 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/util/Constants.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/util/Constants.h @@ -14,162 +14,165 @@ namespace Eigen { /** This value means that a positive quantity (e.g., a size) is not known at compile-time, and that instead the value is - * stored in some runtime variable. - * - * Changing the value of Dynamic breaks the ABI, as Dynamic is often used as a template parameter for Matrix. - */ + * stored in some runtime variable. + * + * Changing the value of Dynamic breaks the ABI, as Dynamic is often used as a template parameter for Matrix. + */ const int Dynamic = -1; -/** This value means that a signed quantity (e.g., a signed index) is not known at compile-time, and that instead its value - * has to be specified at runtime. - */ +/** This value means that a signed quantity (e.g., a signed index) is not known at compile-time, and that instead its + * value has to be specified at runtime. + */ const int DynamicIndex = 0xffffff; /** This value means +Infinity; it is currently used only as the p parameter to MatrixBase::lpNorm(). - * The value Infinity there means the L-infinity norm. - */ + * The value Infinity there means the L-infinity norm. + */ const int Infinity = -1; /** This value means that the cost to evaluate an expression coefficient is either very expensive or - * cannot be known at compile time. - * - * This value has to be positive to (1) simplify cost computation, and (2) allow to distinguish between a very expensive and very very expensive expressions. - * It thus must also be large enough to make sure unrolling won't happen and that sub expressions will be evaluated, but not too large to avoid overflow. - */ + * cannot be known at compile time. + * + * This value has to be positive to (1) simplify cost computation, and (2) allow to distinguish between a very expensive + * and very very expensive expressions. It thus must also be large enough to make sure unrolling won't happen and that + * sub expressions will be evaluated, but not too large to avoid overflow. + */ const int HugeCost = 10000; /** \defgroup flags Flags - * \ingroup Core_Module - * - * These are the possible bits which can be OR'ed to constitute the flags of a matrix or - * expression. - * - * It is important to note that these flags are a purely compile-time notion. They are a compile-time property of - * an expression type, implemented as enum's. They are not stored in memory at runtime, and they do not incur any - * runtime overhead. - * - * \sa MatrixBase::Flags - */ + * \ingroup Core_Module + * + * These are the possible bits which can be OR'ed to constitute the flags of a matrix or + * expression. + * + * It is important to note that these flags are a purely compile-time notion. They are a compile-time property of + * an expression type, implemented as enum's. They are not stored in memory at runtime, and they do not incur any + * runtime overhead. + * + * \sa MatrixBase::Flags + */ /** \ingroup flags - * - * for a matrix, this means that the storage order is row-major. - * If this bit is not set, the storage order is column-major. - * For an expression, this determines the storage order of - * the matrix created by evaluation of that expression. - * \sa \blank \ref TopicStorageOrders */ + * + * for a matrix, this means that the storage order is row-major. + * If this bit is not set, the storage order is column-major. + * For an expression, this determines the storage order of + * the matrix created by evaluation of that expression. + * \sa \blank \ref TopicStorageOrders */ const unsigned int RowMajorBit = 0x1; /** \ingroup flags - * means the expression should be evaluated by the calling expression */ + * means the expression should be evaluated by the calling expression */ const unsigned int EvalBeforeNestingBit = 0x2; /** \ingroup flags - * \deprecated - * means the expression should be evaluated before any assignment */ + * \deprecated + * means the expression should be evaluated before any assignment */ EIGEN_DEPRECATED -const unsigned int EvalBeforeAssigningBit = 0x4; // FIXME deprecated +const unsigned int EvalBeforeAssigningBit = 0x4;// FIXME deprecated /** \ingroup flags - * - * Short version: means the expression might be vectorized - * - * Long version: means that the coefficients can be handled by packets - * and start at a memory location whose alignment meets the requirements - * of the present CPU architecture for optimized packet access. In the fixed-size - * case, there is the additional condition that it be possible to access all the - * coefficients by packets (this implies the requirement that the size be a multiple of 16 bytes, - * and that any nontrivial strides don't break the alignment). In the dynamic-size case, - * there is no such condition on the total size and strides, so it might not be possible to access - * all coeffs by packets. - * - * \note This bit can be set regardless of whether vectorization is actually enabled. - * To check for actual vectorizability, see \a ActualPacketAccessBit. - */ + * + * Short version: means the expression might be vectorized + * + * Long version: means that the coefficients can be handled by packets + * and start at a memory location whose alignment meets the requirements + * of the present CPU architecture for optimized packet access. In the fixed-size + * case, there is the additional condition that it be possible to access all the + * coefficients by packets (this implies the requirement that the size be a multiple of 16 bytes, + * and that any nontrivial strides don't break the alignment). In the dynamic-size case, + * there is no such condition on the total size and strides, so it might not be possible to access + * all coeffs by packets. + * + * \note This bit can be set regardless of whether vectorization is actually enabled. + * To check for actual vectorizability, see \a ActualPacketAccessBit. + */ const unsigned int PacketAccessBit = 0x8; #ifdef EIGEN_VECTORIZE /** \ingroup flags - * - * If vectorization is enabled (EIGEN_VECTORIZE is defined) this constant - * is set to the value \a PacketAccessBit. - * - * If vectorization is not enabled (EIGEN_VECTORIZE is not defined) this constant - * is set to the value 0. - */ + * + * If vectorization is enabled (EIGEN_VECTORIZE is defined) this constant + * is set to the value \a PacketAccessBit. + * + * If vectorization is not enabled (EIGEN_VECTORIZE is not defined) this constant + * is set to the value 0. + */ const unsigned int ActualPacketAccessBit = PacketAccessBit; #else const unsigned int ActualPacketAccessBit = 0x0; #endif /** \ingroup flags - * - * Short version: means the expression can be seen as 1D vector. - * - * Long version: means that one can access the coefficients - * of this expression by coeff(int), and coeffRef(int) in the case of a lvalue expression. These - * index-based access methods are guaranteed - * to not have to do any runtime computation of a (row, col)-pair from the index, so that it - * is guaranteed that whenever it is available, index-based access is at least as fast as - * (row,col)-based access. Expressions for which that isn't possible don't have the LinearAccessBit. - * - * If both PacketAccessBit and LinearAccessBit are set, then the - * packets of this expression can be accessed by packet(int), and writePacket(int) in the case of a - * lvalue expression. - * - * Typically, all vector expressions have the LinearAccessBit, but there is one exception: - * Product expressions don't have it, because it would be troublesome for vectorization, even when the - * Product is a vector expression. Thus, vector Product expressions allow index-based coefficient access but - * not index-based packet access, so they don't have the LinearAccessBit. - */ + * + * Short version: means the expression can be seen as 1D vector. + * + * Long version: means that one can access the coefficients + * of this expression by coeff(int), and coeffRef(int) in the case of a lvalue expression. These + * index-based access methods are guaranteed + * to not have to do any runtime computation of a (row, col)-pair from the index, so that it + * is guaranteed that whenever it is available, index-based access is at least as fast as + * (row,col)-based access. Expressions for which that isn't possible don't have the LinearAccessBit. + * + * If both PacketAccessBit and LinearAccessBit are set, then the + * packets of this expression can be accessed by packet(int), and writePacket(int) in the case of a + * lvalue expression. + * + * Typically, all vector expressions have the LinearAccessBit, but there is one exception: + * Product expressions don't have it, because it would be troublesome for vectorization, even when the + * Product is a vector expression. Thus, vector Product expressions allow index-based coefficient access but + * not index-based packet access, so they don't have the LinearAccessBit. + */ const unsigned int LinearAccessBit = 0x10; /** \ingroup flags - * - * Means the expression has a coeffRef() method, i.e. is writable as its individual coefficients are directly addressable. - * This rules out read-only expressions. - * - * Note that DirectAccessBit and LvalueBit are mutually orthogonal, as there are examples of expression having one but note - * the other: - * \li writable expressions that don't have a very simple memory layout as a strided array, have LvalueBit but not DirectAccessBit - * \li Map-to-const expressions, for example Map, have DirectAccessBit but not LvalueBit - * - * Expressions having LvalueBit also have their coeff() method returning a const reference instead of returning a new value. - */ + * + * Means the expression has a coeffRef() method, i.e. is writable as its individual coefficients are directly + * addressable. This rules out read-only expressions. + * + * Note that DirectAccessBit and LvalueBit are mutually orthogonal, as there are examples of expression having one but + * note the other: + * \li writable expressions that don't have a very simple memory layout as a strided array, have LvalueBit but not + * DirectAccessBit + * \li Map-to-const expressions, for example Map, have DirectAccessBit but not LvalueBit + * + * Expressions having LvalueBit also have their coeff() method returning a const reference instead of returning a new + * value. + */ const unsigned int LvalueBit = 0x20; /** \ingroup flags - * - * Means that the underlying array of coefficients can be directly accessed as a plain strided array. The memory layout - * of the array of coefficients must be exactly the natural one suggested by rows(), cols(), - * outerStride(), innerStride(), and the RowMajorBit. This rules out expressions such as Diagonal, whose coefficients, - * though referencable, do not have such a regular memory layout. - * - * See the comment on LvalueBit for an explanation of how LvalueBit and DirectAccessBit are mutually orthogonal. - */ + * + * Means that the underlying array of coefficients can be directly accessed as a plain strided array. The memory layout + * of the array of coefficients must be exactly the natural one suggested by rows(), cols(), + * outerStride(), innerStride(), and the RowMajorBit. This rules out expressions such as Diagonal, whose coefficients, + * though referencable, do not have such a regular memory layout. + * + * See the comment on LvalueBit for an explanation of how LvalueBit and DirectAccessBit are mutually orthogonal. + */ const unsigned int DirectAccessBit = 0x40; /** \deprecated \ingroup flags - * - * means the first coefficient packet is guaranteed to be aligned. - * An expression cannot has the AlignedBit without the PacketAccessBit flag. - * In other words, this means we are allow to perform an aligned packet access to the first element regardless - * of the expression kind: - * \code - * expression.packet(0); - * \endcode - */ + * + * means the first coefficient packet is guaranteed to be aligned. + * An expression cannot has the AlignedBit without the PacketAccessBit flag. + * In other words, this means we are allow to perform an aligned packet access to the first element regardless + * of the expression kind: + * \code + * expression.packet(0); + * \endcode + */ EIGEN_DEPRECATED const unsigned int AlignedBit = 0x80; const unsigned int NestByRefBit = 0x100; /** \ingroup flags - * - * for an expression, this means that the storage order - * can be either row-major or column-major. - * The precise choice will be decided at evaluation time or when - * combined with other expressions. - * \sa \blank \ref RowMajorBit, \ref TopicStorageOrders */ + * + * for an expression, this means that the storage order + * can be either row-major or column-major. + * The precise choice will be decided at evaluation time or when + * combined with other expressions. + * \sa \blank \ref RowMajorBit, \ref TopicStorageOrders */ const unsigned int NoPreferredStorageOrderBit = 0x200; /** \ingroup flags @@ -187,63 +190,62 @@ const unsigned int CompressedAccessBit = 0x400; // list of flags that are inherited by default -const unsigned int HereditaryBits = RowMajorBit - | EvalBeforeNestingBit; +const unsigned int HereditaryBits = RowMajorBit | EvalBeforeNestingBit; /** \defgroup enums Enumerations - * \ingroup Core_Module - * - * Various enumerations used in %Eigen. Many of these are used as template parameters. - */ + * \ingroup Core_Module + * + * Various enumerations used in %Eigen. Many of these are used as template parameters. + */ /** \ingroup enums - * Enum containing possible values for the \c Mode or \c UpLo parameter of - * MatrixBase::selfadjointView() and MatrixBase::triangularView(), and selfadjoint solvers. */ + * Enum containing possible values for the \c Mode or \c UpLo parameter of + * MatrixBase::selfadjointView() and MatrixBase::triangularView(), and selfadjoint solvers. */ enum UpLoType { /** View matrix as a lower triangular matrix. */ - Lower=0x1, + Lower = 0x1, /** View matrix as an upper triangular matrix. */ - Upper=0x2, + Upper = 0x2, /** %Matrix has ones on the diagonal; to be used in combination with #Lower or #Upper. */ - UnitDiag=0x4, + UnitDiag = 0x4, /** %Matrix has zeros on the diagonal; to be used in combination with #Lower or #Upper. */ - ZeroDiag=0x8, + ZeroDiag = 0x8, /** View matrix as a lower triangular matrix with ones on the diagonal. */ - UnitLower=UnitDiag|Lower, + UnitLower = UnitDiag | Lower, /** View matrix as an upper triangular matrix with ones on the diagonal. */ - UnitUpper=UnitDiag|Upper, + UnitUpper = UnitDiag | Upper, /** View matrix as a lower triangular matrix with zeros on the diagonal. */ - StrictlyLower=ZeroDiag|Lower, + StrictlyLower = ZeroDiag | Lower, /** View matrix as an upper triangular matrix with zeros on the diagonal. */ - StrictlyUpper=ZeroDiag|Upper, + StrictlyUpper = ZeroDiag | Upper, /** Used in BandMatrix and SelfAdjointView to indicate that the matrix is self-adjoint. */ - SelfAdjoint=0x10, + SelfAdjoint = 0x10, /** Used to support symmetric, non-selfadjoint, complex matrices. */ - Symmetric=0x20 + Symmetric = 0x20 }; /** \ingroup enums - * Enum for indicating whether a buffer is aligned or not. */ + * Enum for indicating whether a buffer is aligned or not. */ enum AlignmentType { - Unaligned=0, /**< Data pointer has no specific alignment. */ - Aligned8=8, /**< Data pointer is aligned on a 8 bytes boundary. */ - Aligned16=16, /**< Data pointer is aligned on a 16 bytes boundary. */ - Aligned32=32, /**< Data pointer is aligned on a 32 bytes boundary. */ - Aligned64=64, /**< Data pointer is aligned on a 64 bytes boundary. */ - Aligned128=128, /**< Data pointer is aligned on a 128 bytes boundary. */ - AlignedMask=255, - Aligned=16, /**< \deprecated Synonym for Aligned16. */ -#if EIGEN_MAX_ALIGN_BYTES==128 + Unaligned = 0, /**< Data pointer has no specific alignment. */ + Aligned8 = 8, /**< Data pointer is aligned on a 8 bytes boundary. */ + Aligned16 = 16, /**< Data pointer is aligned on a 16 bytes boundary. */ + Aligned32 = 32, /**< Data pointer is aligned on a 32 bytes boundary. */ + Aligned64 = 64, /**< Data pointer is aligned on a 64 bytes boundary. */ + Aligned128 = 128, /**< Data pointer is aligned on a 128 bytes boundary. */ + AlignedMask = 255, + Aligned = 16, /**< \deprecated Synonym for Aligned16. */ +#if EIGEN_MAX_ALIGN_BYTES == 128 AlignedMax = Aligned128 -#elif EIGEN_MAX_ALIGN_BYTES==64 +#elif EIGEN_MAX_ALIGN_BYTES == 64 AlignedMax = Aligned64 -#elif EIGEN_MAX_ALIGN_BYTES==32 +#elif EIGEN_MAX_ALIGN_BYTES == 32 AlignedMax = Aligned32 -#elif EIGEN_MAX_ALIGN_BYTES==16 +#elif EIGEN_MAX_ALIGN_BYTES == 16 AlignedMax = Aligned16 -#elif EIGEN_MAX_ALIGN_BYTES==8 +#elif EIGEN_MAX_ALIGN_BYTES == 8 AlignedMax = Aligned8 -#elif EIGEN_MAX_ALIGN_BYTES==0 +#elif EIGEN_MAX_ALIGN_BYTES == 0 AlignedMax = Unaligned #else #error Invalid value for EIGEN_MAX_ALIGN_BYTES @@ -257,35 +259,35 @@ enum AlignmentType { enum CornerType { TopLeft, TopRight, BottomLeft, BottomRight }; /** \ingroup enums - * Enum containing possible values for the \p Direction parameter of - * Reverse, PartialReduxExpr and VectorwiseOp. */ -enum DirectionType { - /** For Reverse, all columns are reversed; - * for PartialReduxExpr and VectorwiseOp, act on columns. */ - Vertical, - /** For Reverse, all rows are reversed; - * for PartialReduxExpr and VectorwiseOp, act on rows. */ - Horizontal, - /** For Reverse, both rows and columns are reversed; - * not used for PartialReduxExpr and VectorwiseOp. */ - BothDirections + * Enum containing possible values for the \p Direction parameter of + * Reverse, PartialReduxExpr and VectorwiseOp. */ +enum DirectionType { + /** For Reverse, all columns are reversed; + * for PartialReduxExpr and VectorwiseOp, act on columns. */ + Vertical, + /** For Reverse, all rows are reversed; + * for PartialReduxExpr and VectorwiseOp, act on rows. */ + Horizontal, + /** For Reverse, both rows and columns are reversed; + * not used for PartialReduxExpr and VectorwiseOp. */ + BothDirections }; /** \internal \ingroup enums - * Enum to specify how to traverse the entries of a matrix. */ + * Enum to specify how to traverse the entries of a matrix. */ enum TraversalType { /** \internal Default traversal, no vectorization, no index-based access */ DefaultTraversal, /** \internal No vectorization, use index-based access to have only one for loop instead of 2 nested loops */ LinearTraversal, /** \internal Equivalent to a slice vectorization for fixed-size matrices having good alignment - * and good size */ + * and good size */ InnerVectorizedTraversal, /** \internal Vectorization path using a single loop plus scalar loops for the - * unaligned boundaries */ + * unaligned boundaries */ LinearVectorizedTraversal, /** \internal Generic vectorization path using one vectorized loop per row/column with some - * scalar loops to handle the unaligned boundaries */ + * scalar loops to handle the unaligned boundaries */ SliceVectorizedTraversal, /** \internal Special case to properly handle incompatible scalar types or other defecting cases*/ InvalidTraversal, @@ -294,27 +296,24 @@ enum TraversalType { }; /** \internal \ingroup enums - * Enum to specify whether to unroll loops when traversing over the entries of a matrix. */ + * Enum to specify whether to unroll loops when traversing over the entries of a matrix. */ enum UnrollingType { /** \internal Do not unroll loops. */ NoUnrolling, /** \internal Unroll only the inner loop, but not the outer loop. */ InnerUnrolling, - /** \internal Unroll both the inner and the outer loop. If there is only one loop, - * because linear traversal is used, then unroll that loop. */ + /** \internal Unroll both the inner and the outer loop. If there is only one loop, + * because linear traversal is used, then unroll that loop. */ CompleteUnrolling }; /** \internal \ingroup enums - * Enum to specify whether to use the default (built-in) implementation or the specialization. */ -enum SpecializedType { - Specialized, - BuiltIn -}; + * Enum to specify whether to use the default (built-in) implementation or the specialization. */ +enum SpecializedType { Specialized, BuiltIn }; /** \ingroup enums - * Enum containing possible values for the \p _Options template parameter of - * Matrix, Array and BandMatrix. */ + * Enum containing possible values for the \p _Options template parameter of + * Matrix, Array and BandMatrix. */ enum StorageOptions { /** Storage order is column major (see \ref TopicStorageOrders). */ ColMajor = 0, @@ -327,12 +326,12 @@ enum StorageOptions { }; /** \ingroup enums - * Enum for specifying whether to apply or solve on the left or right. */ + * Enum for specifying whether to apply or solve on the left or right. */ enum SideType { /** Apply transformation on the left. */ - OnTheLeft = 1, + OnTheLeft = 1, /** Apply transformation on the right. */ - OnTheRight = 2 + OnTheRight = 2 }; /* the following used to be written as: @@ -342,74 +341,71 @@ enum SideType { * EIGEN_UNUSED NoChange_t NoChange; * } * - * on the ground that it feels dangerous to disambiguate overloaded functions on enum/integer types. + * on the ground that it feels dangerous to disambiguate overloaded functions on enum/integer types. * However, this leads to "variable declared but never referenced" warnings on Intel Composer XE, * and we do not know how to get rid of them (bug 450). */ -enum NoChange_t { NoChange }; +enum NoChange_t { NoChange }; enum Sequential_t { Sequential }; -enum Default_t { Default }; +enum Default_t { Default }; /** \internal \ingroup enums - * Used in AmbiVector. */ -enum AmbiVectorMode { - IsDense = 0, - IsSparse -}; + * Used in AmbiVector. */ +enum AmbiVectorMode { IsDense = 0, IsSparse }; /** \ingroup enums - * Used as template parameter in DenseCoeffBase and MapBase to indicate - * which accessors should be provided. */ + * Used as template parameter in DenseCoeffBase and MapBase to indicate + * which accessors should be provided. */ enum AccessorLevels { /** Read-only access via a member function. */ - ReadOnlyAccessors, + ReadOnlyAccessors, /** Read/write access via member functions. */ - WriteAccessors, + WriteAccessors, /** Direct read-only access to the coefficients. */ - DirectAccessors, + DirectAccessors, /** Direct read/write access to the coefficients. */ DirectWriteAccessors }; /** \ingroup enums - * Enum with options to give to various decompositions. */ + * Enum with options to give to various decompositions. */ enum DecompositionOptions { /** \internal Not used (meant for LDLT?). */ - Pivoting = 0x01, + Pivoting = 0x01, /** \internal Not used (meant for LDLT?). */ - NoPivoting = 0x02, + NoPivoting = 0x02, /** Used in JacobiSVD to indicate that the square matrix U is to be computed. */ - ComputeFullU = 0x04, + ComputeFullU = 0x04, /** Used in JacobiSVD to indicate that the thin matrix U is to be computed. */ - ComputeThinU = 0x08, + ComputeThinU = 0x08, /** Used in JacobiSVD to indicate that the square matrix V is to be computed. */ - ComputeFullV = 0x10, + ComputeFullV = 0x10, /** Used in JacobiSVD to indicate that the thin matrix V is to be computed. */ - ComputeThinV = 0x20, + ComputeThinV = 0x20, /** Used in SelfAdjointEigenSolver and GeneralizedSelfAdjointEigenSolver to specify - * that only the eigenvalues are to be computed and not the eigenvectors. */ - EigenvaluesOnly = 0x40, + * that only the eigenvalues are to be computed and not the eigenvectors. */ + EigenvaluesOnly = 0x40, /** Used in SelfAdjointEigenSolver and GeneralizedSelfAdjointEigenSolver to specify - * that both the eigenvalues and the eigenvectors are to be computed. */ + * that both the eigenvalues and the eigenvectors are to be computed. */ ComputeEigenvectors = 0x80, /** \internal */ EigVecMask = EigenvaluesOnly | ComputeEigenvectors, /** Used in GeneralizedSelfAdjointEigenSolver to indicate that it should - * solve the generalized eigenproblem \f$ Ax = \lambda B x \f$. */ - Ax_lBx = 0x100, + * solve the generalized eigenproblem \f$ Ax = \lambda B x \f$. */ + Ax_lBx = 0x100, /** Used in GeneralizedSelfAdjointEigenSolver to indicate that it should - * solve the generalized eigenproblem \f$ ABx = \lambda x \f$. */ - ABx_lx = 0x200, + * solve the generalized eigenproblem \f$ ABx = \lambda x \f$. */ + ABx_lx = 0x200, /** Used in GeneralizedSelfAdjointEigenSolver to indicate that it should - * solve the generalized eigenproblem \f$ BAx = \lambda x \f$. */ - BAx_lx = 0x400, + * solve the generalized eigenproblem \f$ BAx = \lambda x \f$. */ + BAx_lx = 0x400, /** \internal */ GenEigMask = Ax_lBx | ABx_lx | BAx_lx }; /** \ingroup enums - * Possible values for the \p QRPreconditioner template parameter of JacobiSVD. */ + * Possible values for the \p QRPreconditioner template parameter of JacobiSVD. */ enum QRPreconditioners { /** Do not specify what is to be done if the SVD of a non-square matrix is asked for. */ NoQRPreconditioner, @@ -426,38 +422,37 @@ enum QRPreconditioners { #endif /** \ingroup enums - * Enum for reporting the status of a computation. */ + * Enum for reporting the status of a computation. */ enum ComputationInfo { /** Computation was successful. */ - Success = 0, + Success = 0, /** The provided data did not satisfy the prerequisites. */ - NumericalIssue = 1, + NumericalIssue = 1, /** Iterative procedure did not converge. */ NoConvergence = 2, /** The inputs are invalid, or the algorithm has been improperly called. - * When assertions are enabled, such errors trigger an assert. */ + * When assertions are enabled, such errors trigger an assert. */ InvalidInput = 3 }; /** \ingroup enums - * Enum used to specify how a particular transformation is stored in a matrix. - * \sa Transform, Hyperplane::transform(). */ + * Enum used to specify how a particular transformation is stored in a matrix. + * \sa Transform, Hyperplane::transform(). */ enum TransformTraits { /** Transformation is an isometry. */ - Isometry = 0x1, - /** Transformation is an affine transformation stored as a (Dim+1)^2 matrix whose last row is - * assumed to be [0 ... 0 1]. */ - Affine = 0x2, + Isometry = 0x1, + /** Transformation is an affine transformation stored as a (Dim+1)^2 matrix whose last row is + * assumed to be [0 ... 0 1]. */ + Affine = 0x2, /** Transformation is an affine transformation stored as a (Dim) x (Dim+1) matrix. */ AffineCompact = 0x10 | Affine, /** Transformation is a general projective transformation stored as a (Dim+1)^2 matrix. */ - Projective = 0x20 + Projective = 0x20 }; /** \internal \ingroup enums - * Enum used to choose between implementation depending on the computer architecture. */ -namespace Architecture -{ + * Enum used to choose between implementation depending on the computer architecture. */ +namespace Architecture { enum Type { Generic = 0x0, SSE = 0x1, @@ -476,72 +471,121 @@ namespace Architecture Target = Generic #endif }; -} +}// namespace Architecture /** \internal \ingroup enums - * Enum used as template parameter in Product and product evaluators. */ -enum ProductImplType -{ DefaultProduct=0, LazyProduct, AliasFreeProduct, CoeffBasedProductMode, LazyCoeffBasedProductMode, OuterProduct, InnerProduct, GemvProduct, GemmProduct }; + * Enum used as template parameter in Product and product evaluators. */ +enum ProductImplType { + DefaultProduct = 0, + LazyProduct, + AliasFreeProduct, + CoeffBasedProductMode, + LazyCoeffBasedProductMode, + OuterProduct, + InnerProduct, + GemvProduct, + GemmProduct +}; /** \internal \ingroup enums - * Enum used in experimental parallel implementation. */ -enum Action {GetAction, SetAction}; + * Enum used in experimental parallel implementation. */ +enum Action { GetAction, SetAction }; /** The type used to identify a dense storage. */ -struct Dense {}; +struct Dense +{ +}; /** The type used to identify a general sparse storage. */ -struct Sparse {}; +struct Sparse +{ +}; /** The type used to identify a general solver (factored) storage. */ -struct SolverStorage {}; +struct SolverStorage +{ +}; /** The type used to identify a permutation storage. */ -struct PermutationStorage {}; +struct PermutationStorage +{ +}; /** The type used to identify a permutation storage. */ -struct TranspositionsStorage {}; +struct TranspositionsStorage +{ +}; /** The type used to identify a matrix expression */ -struct MatrixXpr {}; +struct MatrixXpr +{ +}; /** The type used to identify an array expression */ -struct ArrayXpr {}; +struct ArrayXpr +{ +}; // An evaluator must define its shape. By default, it can be one of the following: -struct DenseShape { static std::string debugName() { return "DenseShape"; } }; -struct SolverShape { static std::string debugName() { return "SolverShape"; } }; -struct HomogeneousShape { static std::string debugName() { return "HomogeneousShape"; } }; -struct DiagonalShape { static std::string debugName() { return "DiagonalShape"; } }; -struct BandShape { static std::string debugName() { return "BandShape"; } }; -struct TriangularShape { static std::string debugName() { return "TriangularShape"; } }; -struct SelfAdjointShape { static std::string debugName() { return "SelfAdjointShape"; } }; -struct PermutationShape { static std::string debugName() { return "PermutationShape"; } }; -struct TranspositionsShape { static std::string debugName() { return "TranspositionsShape"; } }; -struct SparseShape { static std::string debugName() { return "SparseShape"; } }; +struct DenseShape +{ + static std::string debugName() { return "DenseShape"; } +}; +struct SolverShape +{ + static std::string debugName() { return "SolverShape"; } +}; +struct HomogeneousShape +{ + static std::string debugName() { return "HomogeneousShape"; } +}; +struct DiagonalShape +{ + static std::string debugName() { return "DiagonalShape"; } +}; +struct BandShape +{ + static std::string debugName() { return "BandShape"; } +}; +struct TriangularShape +{ + static std::string debugName() { return "TriangularShape"; } +}; +struct SelfAdjointShape +{ + static std::string debugName() { return "SelfAdjointShape"; } +}; +struct PermutationShape +{ + static std::string debugName() { return "PermutationShape"; } +}; +struct TranspositionsShape +{ + static std::string debugName() { return "TranspositionsShape"; } +}; +struct SparseShape +{ + static std::string debugName() { return "SparseShape"; } +}; namespace internal { // random access iterators based on coeff*() accessors. -struct IndexBased {}; + struct IndexBased + { + }; -// evaluator based on iterators to access coefficients. -struct IteratorBased {}; + // evaluator based on iterators to access coefficients. + struct IteratorBased + { + }; -/** \internal - * Constants for comparison functors - */ -enum ComparisonName { - cmp_EQ = 0, - cmp_LT = 1, - cmp_LE = 2, - cmp_UNORD = 3, - cmp_NEQ = 4, - cmp_GT = 5, - cmp_GE = 6 -}; -} // end namespace internal + /** \internal + * Constants for comparison functors + */ + enum ComparisonName { cmp_EQ = 0, cmp_LT = 1, cmp_LE = 2, cmp_UNORD = 3, cmp_NEQ = 4, cmp_GT = 5, cmp_GE = 6 }; +}// end namespace internal -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_CONSTANTS_H +#endif// EIGEN_CONSTANTS_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/util/DisableStupidWarnings.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/util/DisableStupidWarnings.h old mode 100755 new mode 100644 index 351bd6c6..ea7b5ca4 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/util/DisableStupidWarnings.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/util/DisableStupidWarnings.h @@ -2,82 +2,84 @@ #define EIGEN_WARNINGS_DISABLED #ifdef _MSC_VER - // 4100 - unreferenced formal parameter (occurred e.g. in aligned_allocator::destroy(pointer p)) - // 4101 - unreferenced local variable - // 4127 - conditional expression is constant - // 4181 - qualifier applied to reference type ignored - // 4211 - nonstandard extension used : redefined extern to static - // 4244 - 'argument' : conversion from 'type1' to 'type2', possible loss of data - // 4273 - QtAlignedMalloc, inconsistent DLL linkage - // 4324 - structure was padded due to declspec(align()) - // 4503 - decorated name length exceeded, name was truncated - // 4512 - assignment operator could not be generated - // 4522 - 'class' : multiple assignment operators specified - // 4700 - uninitialized local variable 'xyz' used - // 4714 - function marked as __forceinline not inlined - // 4717 - 'function' : recursive on all control paths, function will cause runtime stack overflow - // 4800 - 'type' : forcing value to bool 'true' or 'false' (performance warning) - #ifndef EIGEN_PERMANENTLY_DISABLE_STUPID_WARNINGS - #pragma warning( push ) - #endif - #pragma warning( disable : 4100 4101 4127 4181 4211 4244 4273 4324 4503 4512 4522 4700 4714 4717 4800) +// 4100 - unreferenced formal parameter (occurred e.g. in aligned_allocator::destroy(pointer p)) +// 4101 - unreferenced local variable +// 4127 - conditional expression is constant +// 4181 - qualifier applied to reference type ignored +// 4211 - nonstandard extension used : redefined extern to static +// 4244 - 'argument' : conversion from 'type1' to 'type2', possible loss of data +// 4273 - QtAlignedMalloc, inconsistent DLL linkage +// 4324 - structure was padded due to declspec(align()) +// 4503 - decorated name length exceeded, name was truncated +// 4512 - assignment operator could not be generated +// 4522 - 'class' : multiple assignment operators specified +// 4700 - uninitialized local variable 'xyz' used +// 4714 - function marked as __forceinline not inlined +// 4717 - 'function' : recursive on all control paths, function will cause runtime stack overflow +// 4800 - 'type' : forcing value to bool 'true' or 'false' (performance warning) +#ifndef EIGEN_PERMANENTLY_DISABLE_STUPID_WARNINGS +#pragma warning(push) +#endif +#pragma warning(disable : 4100 4101 4127 4181 4211 4244 4273 4324 4503 4512 4522 4700 4714 4717 4800) #elif defined __INTEL_COMPILER - // 2196 - routine is both "inline" and "noinline" ("noinline" assumed) - // ICC 12 generates this warning even without any inline keyword, when defining class methods 'inline' i.e. inside of class body - // typedef that may be a reference type. - // 279 - controlling expression is constant - // ICC 12 generates this warning on assert(constant_expression_depending_on_template_params) and frankly this is a legitimate use case. - // 1684 - conversion from pointer to same-sized integral type (potential portability problem) - // 2259 - non-pointer conversion from "Eigen::Index={ptrdiff_t={long}}" to "int" may lose significant bits - #ifndef EIGEN_PERMANENTLY_DISABLE_STUPID_WARNINGS - #pragma warning push - #endif - #pragma warning disable 2196 279 1684 2259 +// 2196 - routine is both "inline" and "noinline" ("noinline" assumed) +// ICC 12 generates this warning even without any inline keyword, when defining class methods 'inline' i.e. +// inside of class body typedef that may be a reference type. +// 279 - controlling expression is constant +// ICC 12 generates this warning on assert(constant_expression_depending_on_template_params) and frankly this is +// a legitimate use case. +// 1684 - conversion from pointer to same-sized integral type (potential portability problem) +// 2259 - non-pointer conversion from "Eigen::Index={ptrdiff_t={long}}" to "int" may lose significant bits +#ifndef EIGEN_PERMANENTLY_DISABLE_STUPID_WARNINGS +#pragma warning push +#endif +#pragma warning disable 2196 279 1684 2259 #elif defined __clang__ - // -Wconstant-logical-operand - warning: use of logical && with constant operand; switch to bitwise & or remove constant - // this is really a stupid warning as it warns on compile-time expressions involving enums - #ifndef EIGEN_PERMANENTLY_DISABLE_STUPID_WARNINGS - #pragma clang diagnostic push - #endif - #pragma clang diagnostic ignored "-Wconstant-logical-operand" +// -Wconstant-logical-operand - warning: use of logical && with constant operand; switch to bitwise & or remove constant +// this is really a stupid warning as it warns on compile-time expressions involving enums +#ifndef EIGEN_PERMANENTLY_DISABLE_STUPID_WARNINGS +#pragma clang diagnostic push +#endif +#pragma clang diagnostic ignored "-Wconstant-logical-operand" #elif defined __GNUC__ - #if (!defined(EIGEN_PERMANENTLY_DISABLE_STUPID_WARNINGS)) && (__GNUC__ > 4 || (__GNUC__ == 4 && __GNUC_MINOR__ >= 6)) - #pragma GCC diagnostic push - #endif - // g++ warns about local variables shadowing member functions, which is too strict - #pragma GCC diagnostic ignored "-Wshadow" - #if __GNUC__ == 4 && __GNUC_MINOR__ < 8 - // Until g++-4.7 there are warnings when comparing unsigned int vs 0, even in templated functions: - #pragma GCC diagnostic ignored "-Wtype-limits" - #endif - #if __GNUC__>=6 - #pragma GCC diagnostic ignored "-Wignored-attributes" - #endif +#if (!defined(EIGEN_PERMANENTLY_DISABLE_STUPID_WARNINGS)) && (__GNUC__ > 4 || (__GNUC__ == 4 && __GNUC_MINOR__ >= 6)) +#pragma GCC diagnostic push +#endif +// g++ warns about local variables shadowing member functions, which is too strict +#pragma GCC diagnostic ignored "-Wshadow" +#if __GNUC__ == 4 && __GNUC_MINOR__ < 8 +// Until g++-4.7 there are warnings when comparing unsigned int vs 0, even in templated functions: +#pragma GCC diagnostic ignored "-Wtype-limits" +#endif +#if __GNUC__ >= 6 +#pragma GCC diagnostic ignored "-Wignored-attributes" +#endif #endif #if defined __NVCC__ - // Disable the "statement is unreachable" message - #pragma diag_suppress code_is_unreachable - // Disable the "dynamic initialization in unreachable code" message - #pragma diag_suppress initialization_not_reachable - // Disable the "invalid error number" message that we get with older versions of nvcc - #pragma diag_suppress 1222 - // Disable the "calling a __host__ function from a __host__ __device__ function is not allowed" messages (yes, there are many of them and they seem to change with every version of the compiler) - #pragma diag_suppress 2527 - #pragma diag_suppress 2529 - #pragma diag_suppress 2651 - #pragma diag_suppress 2653 - #pragma diag_suppress 2668 - #pragma diag_suppress 2669 - #pragma diag_suppress 2670 - #pragma diag_suppress 2671 - #pragma diag_suppress 2735 - #pragma diag_suppress 2737 +// Disable the "statement is unreachable" message +#pragma diag_suppress code_is_unreachable +// Disable the "dynamic initialization in unreachable code" message +#pragma diag_suppress initialization_not_reachable +// Disable the "invalid error number" message that we get with older versions of nvcc +#pragma diag_suppress 1222 +// Disable the "calling a __host__ function from a __host__ __device__ function is not allowed" messages (yes, there are +// many of them and they seem to change with every version of the compiler) +#pragma diag_suppress 2527 +#pragma diag_suppress 2529 +#pragma diag_suppress 2651 +#pragma diag_suppress 2653 +#pragma diag_suppress 2668 +#pragma diag_suppress 2669 +#pragma diag_suppress 2670 +#pragma diag_suppress 2671 +#pragma diag_suppress 2735 +#pragma diag_suppress 2737 #endif -#endif // not EIGEN_WARNINGS_DISABLED +#endif// not EIGEN_WARNINGS_DISABLED diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/util/ForwardDeclarations.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/util/ForwardDeclarations.h index ea107393..7ff4b3f3 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/util/ForwardDeclarations.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/util/ForwardDeclarations.h @@ -14,33 +14,36 @@ namespace Eigen { namespace internal { -template struct traits; - -// here we say once and for all that traits == traits -// When constness must affect traits, it has to be constness on template parameters on which T itself depends. -// For example, traits > != traits >, but -// traits > == traits > -template struct traits : traits {}; - -template struct has_direct_access -{ - enum { ret = (traits::Flags & DirectAccessBit) ? 1 : 0 }; -}; - -template struct accessors_level -{ - enum { has_direct_access = (traits::Flags & DirectAccessBit) ? 1 : 0, - has_write_access = (traits::Flags & LvalueBit) ? 1 : 0, - value = has_direct_access ? (has_write_access ? DirectWriteAccessors : DirectAccessors) - : (has_write_access ? WriteAccessors : ReadOnlyAccessors) + template struct traits; + + // here we say once and for all that traits == traits + // When constness must affect traits, it has to be constness on template parameters on which T itself depends. + // For example, traits > != traits >, but + // traits > == traits > + template struct traits : traits + { }; -}; -template struct evaluator_traits; + template struct has_direct_access + { + enum { ret = (traits::Flags & DirectAccessBit) ? 1 : 0 }; + }; + + template struct accessors_level + { + enum { + has_direct_access = (traits::Flags & DirectAccessBit) ? 1 : 0, + has_write_access = (traits::Flags & LvalueBit) ? 1 : 0, + value = has_direct_access ? (has_write_access ? DirectWriteAccessors : DirectAccessors) + : (has_write_access ? WriteAccessors : ReadOnlyAccessors) + }; + }; -template< typename T> struct evaluator; + template struct evaluator_traits; -} // end namespace internal + template struct evaluator; + +}// end namespace internal template struct NumTraits; @@ -49,76 +52,81 @@ template class DenseBase; template class PlainObjectBase; -template::value > -class DenseCoeffsBase; - -template::value> class DenseCoeffsBase; + +template class Matrix; + int _MaxRows = _Rows, + int _MaxCols = _Cols> +class Matrix; template class MatrixBase; template class ArrayBase; template class Flagged; -template class StorageBase > class NoAlias; +template class StorageBase> class NoAlias; template class NestByValue; template class ForceAlignedAccess; template class SwapWrapper; -template class Block; +template class Block; -template class VectorBlock; +template class VectorBlock; template class Transpose; template class Conjugate; -template class CwiseNullaryOp; -template class CwiseUnaryOp; -template class CwiseUnaryView; -template class CwiseBinaryOp; -template class CwiseTernaryOp; -template class Solve; -template class Inverse; +template class CwiseNullaryOp; +template class CwiseUnaryOp; +template class CwiseUnaryView; +template class CwiseBinaryOp; +template class CwiseTernaryOp; +template class Solve; +template class Inverse; template class Product; template class DiagonalBase; template class DiagonalWrapper; -template class DiagonalMatrix; +template class DiagonalMatrix; template class DiagonalProduct; template class Diagonal; -template class PermutationMatrix; -template class Transpositions; +template +class PermutationMatrix; +template +class Transpositions; template class PermutationBase; template class TranspositionsBase; template class PermutationWrapper; template class TranspositionsWrapper; template::has_write_access ? WriteAccessors : ReadOnlyAccessors -> class MapBase; + int Level = internal::accessors_level::has_write_access ? WriteAccessors : ReadOnlyAccessors> +class MapBase; template class Stride; template class InnerStride; template class OuterStride; -template > class Map; +template> class Map; template class RefBase; -template,OuterStride<> >::type > class Ref; +template, OuterStride<>>::type> +class Ref; template class TriangularBase; template class TriangularView; @@ -133,37 +141,42 @@ template class SolverBase; template class InnerIterator; namespace internal { -template struct kernel_retval_base; -template struct kernel_retval; -template struct image_retval_base; -template struct image_retval; -} // end namespace internal + template struct kernel_retval_base; + template struct kernel_retval; + template struct image_retval_base; + template struct image_retval; +}// end namespace internal namespace internal { -template class BandMatrix; + template + class BandMatrix; } namespace internal { -template struct product_type; - -template struct EnableIf; - -/** \internal - * \class product_evaluator - * Products need their own evaluator with more template arguments allowing for - * easier partial template specializations. - */ -template< typename T, - int ProductTag = internal::product_type::ret, - typename LhsShape = typename evaluator_traits::Shape, - typename RhsShape = typename evaluator_traits::Shape, - typename LhsScalar = typename traits::Scalar, - typename RhsScalar = typename traits::Scalar - > struct product_evaluator; -} - -template::value> + template struct product_type; + + template struct EnableIf; + + /** \internal + * \class product_evaluator + * Products need their own evaluator with more template arguments allowing for + * easier partial template specializations. + */ + template::ret, + typename LhsShape = typename evaluator_traits::Shape, + typename RhsShape = typename evaluator_traits::Shape, + typename LhsScalar = typename traits::Scalar, + typename RhsScalar = typename traits::Scalar> + struct product_evaluator; +}// namespace internal + +template::value> struct ProductReturnType; // this is a workaround for sun CC @@ -171,85 +184,89 @@ template struct LazyProductReturnType; namespace internal { -// Provides scalar/packet-wise product and product with accumulation -// with optional conjugation of the arguments. -template struct conj_helper; - -template struct scalar_sum_op; -template struct scalar_difference_op; -template struct scalar_conj_product_op; -template struct scalar_min_op; -template struct scalar_max_op; -template struct scalar_opposite_op; -template struct scalar_conjugate_op; -template struct scalar_real_op; -template struct scalar_imag_op; -template struct scalar_abs_op; -template struct scalar_abs2_op; -template struct scalar_sqrt_op; -template struct scalar_rsqrt_op; -template struct scalar_exp_op; -template struct scalar_log_op; -template struct scalar_cos_op; -template struct scalar_sin_op; -template struct scalar_acos_op; -template struct scalar_asin_op; -template struct scalar_tan_op; -template struct scalar_inverse_op; -template struct scalar_square_op; -template struct scalar_cube_op; -template struct scalar_cast_op; -template struct scalar_random_op; -template struct scalar_constant_op; -template struct scalar_identity_op; -template struct scalar_sign_op; -template struct scalar_pow_op; -template struct scalar_hypot_op; -template struct scalar_product_op; -template struct scalar_quotient_op; - -// SpecialFunctions module -template struct scalar_lgamma_op; -template struct scalar_digamma_op; -template struct scalar_erf_op; -template struct scalar_erfc_op; -template struct scalar_igamma_op; -template struct scalar_igammac_op; -template struct scalar_zeta_op; -template struct scalar_betainc_op; - -} // end namespace internal + // Provides scalar/packet-wise product and product with accumulation + // with optional conjugation of the arguments. + template struct conj_helper; + + template struct scalar_sum_op; + template struct scalar_difference_op; + template struct scalar_conj_product_op; + template struct scalar_min_op; + template struct scalar_max_op; + template struct scalar_opposite_op; + template struct scalar_conjugate_op; + template struct scalar_real_op; + template struct scalar_imag_op; + template struct scalar_abs_op; + template struct scalar_abs2_op; + template struct scalar_sqrt_op; + template struct scalar_rsqrt_op; + template struct scalar_exp_op; + template struct scalar_log_op; + template struct scalar_cos_op; + template struct scalar_sin_op; + template struct scalar_acos_op; + template struct scalar_asin_op; + template struct scalar_tan_op; + template struct scalar_inverse_op; + template struct scalar_square_op; + template struct scalar_cube_op; + template struct scalar_cast_op; + template struct scalar_random_op; + template struct scalar_constant_op; + template struct scalar_identity_op; + template struct scalar_sign_op; + template struct scalar_pow_op; + template struct scalar_hypot_op; + template struct scalar_product_op; + template struct scalar_quotient_op; + + // SpecialFunctions module + template struct scalar_lgamma_op; + template struct scalar_digamma_op; + template struct scalar_erf_op; + template struct scalar_erfc_op; + template struct scalar_igamma_op; + template struct scalar_igammac_op; + template struct scalar_zeta_op; + template struct scalar_betainc_op; + +}// end namespace internal struct IOFormat; // Array module -template class Array; + int _MaxRows = _Rows, + int _MaxCols = _Cols> +class Array; template class Select; template class PartialReduxExpr; template class VectorwiseOp; -template class Replicate; +template class Replicate; template class Reverse; template class FullPivLU; template class PartialPivLU; namespace internal { -template struct inverse_impl; + template struct inverse_impl; } template class HouseholderQR; template class ColPivHouseholderQR; @@ -259,8 +276,8 @@ template class BDCSVD; template class LLT; template class LDLT; -template class HouseholderSequence; -template class JacobiRotation; +template class HouseholderSequence; +template class JacobiRotation; // Geometry module: template class RotationBase; @@ -268,14 +285,14 @@ template class Cross; template class QuaternionBase; template class Rotation2D; template class AngleAxis; -template class Translation; -template class AlignedBox; +template class Translation; +template class AlignedBox; template class Quaternion; -template class Transform; -template class ParametrizedLine; -template class Hyperplane; +template class Transform; +template class ParametrizedLine; +template class Hyperplane; template class UniformScaling; -template class Homogeneous; +template class Homogeneous; // Sparse module: template class SparseMatrixBase; @@ -289,14 +306,13 @@ template class MatrixPowerReturnValue; template class MatrixComplexPowerReturnValue; namespace internal { -template -struct stem_function -{ - typedef std::complex::Real> ComplexScalar; - typedef ComplexScalar type(ComplexScalar, int); -}; -} + template struct stem_function + { + typedef std::complex::Real> ComplexScalar; + typedef ComplexScalar type(ComplexScalar, int); + }; +}// namespace internal -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_FORWARDDECLARATIONS_H +#endif// EIGEN_FORWARDDECLARATIONS_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/util/MKL_support.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/util/MKL_support.h old mode 100755 new mode 100644 index b7d6ecc7..a141ee59 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/util/MKL_support.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/util/MKL_support.h @@ -34,42 +34,43 @@ #define EIGEN_MKL_SUPPORT_H #ifdef EIGEN_USE_MKL_ALL - #ifndef EIGEN_USE_BLAS - #define EIGEN_USE_BLAS - #endif - #ifndef EIGEN_USE_LAPACKE - #define EIGEN_USE_LAPACKE - #endif - #ifndef EIGEN_USE_MKL_VML - #define EIGEN_USE_MKL_VML - #endif +#ifndef EIGEN_USE_BLAS +#define EIGEN_USE_BLAS +#endif +#ifndef EIGEN_USE_LAPACKE +#define EIGEN_USE_LAPACKE +#endif +#ifndef EIGEN_USE_MKL_VML +#define EIGEN_USE_MKL_VML +#endif #endif #ifdef EIGEN_USE_LAPACKE_STRICT - #define EIGEN_USE_LAPACKE +#define EIGEN_USE_LAPACKE #endif #if defined(EIGEN_USE_MKL_VML) && !defined(EIGEN_USE_MKL) - #define EIGEN_USE_MKL +#define EIGEN_USE_MKL #endif #if defined EIGEN_USE_MKL -# include +#include /*Check IMKL version for compatibility: < 10.3 is not usable with Eigen*/ -# ifndef INTEL_MKL_VERSION -# undef EIGEN_USE_MKL /* INTEL_MKL_VERSION is not even defined on older versions */ -# elif INTEL_MKL_VERSION < 100305 /* the intel-mkl-103-release-notes say this was when the lapacke.h interface was added*/ -# undef EIGEN_USE_MKL -# endif -# ifndef EIGEN_USE_MKL - /*If the MKL version is too old, undef everything*/ -# undef EIGEN_USE_MKL_ALL -# undef EIGEN_USE_LAPACKE -# undef EIGEN_USE_MKL_VML -# undef EIGEN_USE_LAPACKE_STRICT -# undef EIGEN_USE_LAPACKE -# endif +#ifndef INTEL_MKL_VERSION +#undef EIGEN_USE_MKL /* INTEL_MKL_VERSION is not even defined on older versions */ +#elif INTEL_MKL_VERSION \ + < 100305 /* the intel-mkl-103-release-notes say this was when the lapacke.h interface was added*/ +#undef EIGEN_USE_MKL +#endif +#ifndef EIGEN_USE_MKL +/*If the MKL version is too old, undef everything*/ +#undef EIGEN_USE_MKL_ALL +#undef EIGEN_USE_LAPACKE +#undef EIGEN_USE_MKL_VML +#undef EIGEN_USE_LAPACKE_STRICT +#undef EIGEN_USE_LAPACKE +#endif #endif #if defined EIGEN_USE_MKL @@ -116,7 +117,7 @@ namespace Eigen { typedef std::complex dcomplex; -typedef std::complex scomplex; +typedef std::complex scomplex; #if defined(EIGEN_USE_MKL) typedef MKL_INT BlasIndex; @@ -124,7 +125,7 @@ typedef MKL_INT BlasIndex; typedef int BlasIndex; #endif -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_MKL_SUPPORT_H +#endif// EIGEN_MKL_SUPPORT_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/util/Macros.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/util/Macros.h index aa054a0b..b1a0f810 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/util/Macros.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/util/Macros.h @@ -15,60 +15,61 @@ #define EIGEN_MAJOR_VERSION 3 #define EIGEN_MINOR_VERSION 7 -#define EIGEN_VERSION_AT_LEAST(x,y,z) (EIGEN_WORLD_VERSION>x || (EIGEN_WORLD_VERSION>=x && \ - (EIGEN_MAJOR_VERSION>y || (EIGEN_MAJOR_VERSION>=y && \ - EIGEN_MINOR_VERSION>=z)))) +#define EIGEN_VERSION_AT_LEAST(x, y, z) \ + (EIGEN_WORLD_VERSION > x \ + || (EIGEN_WORLD_VERSION >= x \ + && (EIGEN_MAJOR_VERSION > y || (EIGEN_MAJOR_VERSION >= y && EIGEN_MINOR_VERSION >= z)))) // Compiler identification, EIGEN_COMP_* /// \internal EIGEN_COMP_GNUC set to 1 for all compilers compatible with GCC #ifdef __GNUC__ - #define EIGEN_COMP_GNUC 1 +#define EIGEN_COMP_GNUC 1 #else - #define EIGEN_COMP_GNUC 0 +#define EIGEN_COMP_GNUC 0 #endif /// \internal EIGEN_COMP_CLANG set to major+minor version (e.g., 307 for clang 3.7) if the compiler is clang #if defined(__clang__) - #define EIGEN_COMP_CLANG (__clang_major__*100+__clang_minor__) +#define EIGEN_COMP_CLANG (__clang_major__ * 100 + __clang_minor__) #else - #define EIGEN_COMP_CLANG 0 +#define EIGEN_COMP_CLANG 0 #endif /// \internal EIGEN_COMP_LLVM set to 1 if the compiler backend is llvm #if defined(__llvm__) - #define EIGEN_COMP_LLVM 1 +#define EIGEN_COMP_LLVM 1 #else - #define EIGEN_COMP_LLVM 0 +#define EIGEN_COMP_LLVM 0 #endif /// \internal EIGEN_COMP_ICC set to __INTEL_COMPILER if the compiler is Intel compiler, 0 otherwise #if defined(__INTEL_COMPILER) - #define EIGEN_COMP_ICC __INTEL_COMPILER +#define EIGEN_COMP_ICC __INTEL_COMPILER #else - #define EIGEN_COMP_ICC 0 +#define EIGEN_COMP_ICC 0 #endif /// \internal EIGEN_COMP_MINGW set to 1 if the compiler is mingw #if defined(__MINGW32__) - #define EIGEN_COMP_MINGW 1 +#define EIGEN_COMP_MINGW 1 #else - #define EIGEN_COMP_MINGW 0 +#define EIGEN_COMP_MINGW 0 #endif /// \internal EIGEN_COMP_SUNCC set to 1 if the compiler is Solaris Studio #if defined(__SUNPRO_CC) - #define EIGEN_COMP_SUNCC 1 +#define EIGEN_COMP_SUNCC 1 #else - #define EIGEN_COMP_SUNCC 0 +#define EIGEN_COMP_SUNCC 0 #endif /// \internal EIGEN_COMP_MSVC set to _MSC_VER if the compiler is Microsoft Visual C++, 0 otherwise. #if defined(_MSC_VER) - #define EIGEN_COMP_MSVC _MSC_VER +#define EIGEN_COMP_MSVC _MSC_VER #else - #define EIGEN_COMP_MSVC 0 +#define EIGEN_COMP_MSVC 0 #endif // For the record, here is a table summarizing the possible values for EIGEN_COMP_MSVC: @@ -80,58 +81,62 @@ // 2015 14 1900 // "15" 15 1900 -/// \internal EIGEN_COMP_MSVC_STRICT set to 1 if the compiler is really Microsoft Visual C++ and not ,e.g., ICC or clang-cl +/// \internal EIGEN_COMP_MSVC_STRICT set to 1 if the compiler is really Microsoft Visual C++ and not ,e.g., ICC or +/// clang-cl #if EIGEN_COMP_MSVC && !(EIGEN_COMP_ICC || EIGEN_COMP_LLVM || EIGEN_COMP_CLANG) - #define EIGEN_COMP_MSVC_STRICT _MSC_VER +#define EIGEN_COMP_MSVC_STRICT _MSC_VER #else - #define EIGEN_COMP_MSVC_STRICT 0 +#define EIGEN_COMP_MSVC_STRICT 0 #endif /// \internal EIGEN_COMP_IBM set to 1 if the compiler is IBM XL C++ #if defined(__IBMCPP__) || defined(__xlc__) - #define EIGEN_COMP_IBM 1 +#define EIGEN_COMP_IBM 1 #else - #define EIGEN_COMP_IBM 0 +#define EIGEN_COMP_IBM 0 #endif /// \internal EIGEN_COMP_PGI set to 1 if the compiler is Portland Group Compiler #if defined(__PGI) - #define EIGEN_COMP_PGI 1 +#define EIGEN_COMP_PGI 1 #else - #define EIGEN_COMP_PGI 0 +#define EIGEN_COMP_PGI 0 #endif /// \internal EIGEN_COMP_ARM set to 1 if the compiler is ARM Compiler #if defined(__CC_ARM) || defined(__ARMCC_VERSION) - #define EIGEN_COMP_ARM 1 +#define EIGEN_COMP_ARM 1 #else - #define EIGEN_COMP_ARM 0 +#define EIGEN_COMP_ARM 0 #endif /// \internal EIGEN_COMP_ARM set to 1 if the compiler is ARM Compiler #if defined(__EMSCRIPTEN__) - #define EIGEN_COMP_EMSCRIPTEN 1 +#define EIGEN_COMP_EMSCRIPTEN 1 #else - #define EIGEN_COMP_EMSCRIPTEN 0 +#define EIGEN_COMP_EMSCRIPTEN 0 #endif -/// \internal EIGEN_GNUC_STRICT set to 1 if the compiler is really GCC and not a compatible compiler (e.g., ICC, clang, mingw, etc.) -#if EIGEN_COMP_GNUC && !(EIGEN_COMP_CLANG || EIGEN_COMP_ICC || EIGEN_COMP_MINGW || EIGEN_COMP_PGI || EIGEN_COMP_IBM || EIGEN_COMP_ARM || EIGEN_COMP_EMSCRIPTEN) - #define EIGEN_COMP_GNUC_STRICT 1 +/// \internal EIGEN_GNUC_STRICT set to 1 if the compiler is really GCC and not a compatible compiler (e.g., ICC, clang, +/// mingw, etc.) +#if EIGEN_COMP_GNUC \ + && !(EIGEN_COMP_CLANG || EIGEN_COMP_ICC || EIGEN_COMP_MINGW || EIGEN_COMP_PGI || EIGEN_COMP_IBM || EIGEN_COMP_ARM \ + || EIGEN_COMP_EMSCRIPTEN) +#define EIGEN_COMP_GNUC_STRICT 1 #else - #define EIGEN_COMP_GNUC_STRICT 0 +#define EIGEN_COMP_GNUC_STRICT 0 #endif #if EIGEN_COMP_GNUC - #define EIGEN_GNUC_AT_LEAST(x,y) ((__GNUC__==x && __GNUC_MINOR__>=y) || __GNUC__>x) - #define EIGEN_GNUC_AT_MOST(x,y) ((__GNUC__==x && __GNUC_MINOR__<=y) || __GNUC__= y) || __GNUC__ > x) +#define EIGEN_GNUC_AT_MOST(x, y) ((__GNUC__ == x && __GNUC_MINOR__ <= y) || __GNUC__ < x) +#define EIGEN_GNUC_AT(x, y) (__GNUC__ == x && __GNUC_MINOR__ == y) #else - #define EIGEN_GNUC_AT_LEAST(x,y) 0 - #define EIGEN_GNUC_AT_MOST(x,y) 0 - #define EIGEN_GNUC_AT(x,y) 0 +#define EIGEN_GNUC_AT_LEAST(x, y) 0 +#define EIGEN_GNUC_AT_MOST(x, y) 0 +#define EIGEN_GNUC_AT(x, y) 0 #endif // FIXME: could probably be removed as we do not support gcc 3.x anymore @@ -145,181 +150,179 @@ // Architecture identification, EIGEN_ARCH_* #if defined(__x86_64__) || defined(_M_X64) || defined(__amd64) - #define EIGEN_ARCH_x86_64 1 +#define EIGEN_ARCH_x86_64 1 #else - #define EIGEN_ARCH_x86_64 0 +#define EIGEN_ARCH_x86_64 0 #endif #if defined(__i386__) || defined(_M_IX86) || defined(_X86_) || defined(__i386) - #define EIGEN_ARCH_i386 1 +#define EIGEN_ARCH_i386 1 #else - #define EIGEN_ARCH_i386 0 +#define EIGEN_ARCH_i386 0 #endif #if EIGEN_ARCH_x86_64 || EIGEN_ARCH_i386 - #define EIGEN_ARCH_i386_OR_x86_64 1 +#define EIGEN_ARCH_i386_OR_x86_64 1 #else - #define EIGEN_ARCH_i386_OR_x86_64 0 +#define EIGEN_ARCH_i386_OR_x86_64 0 #endif /// \internal EIGEN_ARCH_ARM set to 1 if the architecture is ARM #if defined(__arm__) - #define EIGEN_ARCH_ARM 1 +#define EIGEN_ARCH_ARM 1 #else - #define EIGEN_ARCH_ARM 0 +#define EIGEN_ARCH_ARM 0 #endif /// \internal EIGEN_ARCH_ARM64 set to 1 if the architecture is ARM64 #if defined(__aarch64__) - #define EIGEN_ARCH_ARM64 1 +#define EIGEN_ARCH_ARM64 1 #else - #define EIGEN_ARCH_ARM64 0 +#define EIGEN_ARCH_ARM64 0 #endif #if EIGEN_ARCH_ARM || EIGEN_ARCH_ARM64 - #define EIGEN_ARCH_ARM_OR_ARM64 1 +#define EIGEN_ARCH_ARM_OR_ARM64 1 #else - #define EIGEN_ARCH_ARM_OR_ARM64 0 +#define EIGEN_ARCH_ARM_OR_ARM64 0 #endif /// \internal EIGEN_ARCH_MIPS set to 1 if the architecture is MIPS #if defined(__mips__) || defined(__mips) - #define EIGEN_ARCH_MIPS 1 +#define EIGEN_ARCH_MIPS 1 #else - #define EIGEN_ARCH_MIPS 0 +#define EIGEN_ARCH_MIPS 0 #endif /// \internal EIGEN_ARCH_SPARC set to 1 if the architecture is SPARC #if defined(__sparc__) || defined(__sparc) - #define EIGEN_ARCH_SPARC 1 +#define EIGEN_ARCH_SPARC 1 #else - #define EIGEN_ARCH_SPARC 0 +#define EIGEN_ARCH_SPARC 0 #endif /// \internal EIGEN_ARCH_IA64 set to 1 if the architecture is Intel Itanium #if defined(__ia64__) - #define EIGEN_ARCH_IA64 1 +#define EIGEN_ARCH_IA64 1 #else - #define EIGEN_ARCH_IA64 0 +#define EIGEN_ARCH_IA64 0 #endif /// \internal EIGEN_ARCH_PPC set to 1 if the architecture is PowerPC #if defined(__powerpc__) || defined(__ppc__) || defined(_M_PPC) - #define EIGEN_ARCH_PPC 1 +#define EIGEN_ARCH_PPC 1 #else - #define EIGEN_ARCH_PPC 0 +#define EIGEN_ARCH_PPC 0 #endif - // Operating system identification, EIGEN_OS_* /// \internal EIGEN_OS_UNIX set to 1 if the OS is a unix variant #if defined(__unix__) || defined(__unix) - #define EIGEN_OS_UNIX 1 +#define EIGEN_OS_UNIX 1 #else - #define EIGEN_OS_UNIX 0 +#define EIGEN_OS_UNIX 0 #endif /// \internal EIGEN_OS_LINUX set to 1 if the OS is based on Linux kernel #if defined(__linux__) - #define EIGEN_OS_LINUX 1 +#define EIGEN_OS_LINUX 1 #else - #define EIGEN_OS_LINUX 0 +#define EIGEN_OS_LINUX 0 #endif /// \internal EIGEN_OS_ANDROID set to 1 if the OS is Android // note: ANDROID is defined when using ndk_build, __ANDROID__ is defined when using a standalone toolchain. #if defined(__ANDROID__) || defined(ANDROID) - #define EIGEN_OS_ANDROID 1 +#define EIGEN_OS_ANDROID 1 #else - #define EIGEN_OS_ANDROID 0 +#define EIGEN_OS_ANDROID 0 #endif /// \internal EIGEN_OS_GNULINUX set to 1 if the OS is GNU Linux and not Linux-based OS (e.g., not android) #if defined(__gnu_linux__) && !(EIGEN_OS_ANDROID) - #define EIGEN_OS_GNULINUX 1 +#define EIGEN_OS_GNULINUX 1 #else - #define EIGEN_OS_GNULINUX 0 +#define EIGEN_OS_GNULINUX 0 #endif /// \internal EIGEN_OS_BSD set to 1 if the OS is a BSD variant #if defined(__FreeBSD__) || defined(__NetBSD__) || defined(__OpenBSD__) || defined(__bsdi__) || defined(__DragonFly__) - #define EIGEN_OS_BSD 1 +#define EIGEN_OS_BSD 1 #else - #define EIGEN_OS_BSD 0 +#define EIGEN_OS_BSD 0 #endif /// \internal EIGEN_OS_MAC set to 1 if the OS is MacOS #if defined(__APPLE__) - #define EIGEN_OS_MAC 1 +#define EIGEN_OS_MAC 1 #else - #define EIGEN_OS_MAC 0 +#define EIGEN_OS_MAC 0 #endif /// \internal EIGEN_OS_QNX set to 1 if the OS is QNX #if defined(__QNX__) - #define EIGEN_OS_QNX 1 +#define EIGEN_OS_QNX 1 #else - #define EIGEN_OS_QNX 0 +#define EIGEN_OS_QNX 0 #endif /// \internal EIGEN_OS_WIN set to 1 if the OS is Windows based #if defined(_WIN32) - #define EIGEN_OS_WIN 1 +#define EIGEN_OS_WIN 1 #else - #define EIGEN_OS_WIN 0 +#define EIGEN_OS_WIN 0 #endif /// \internal EIGEN_OS_WIN64 set to 1 if the OS is Windows 64bits #if defined(_WIN64) - #define EIGEN_OS_WIN64 1 +#define EIGEN_OS_WIN64 1 #else - #define EIGEN_OS_WIN64 0 +#define EIGEN_OS_WIN64 0 #endif /// \internal EIGEN_OS_WINCE set to 1 if the OS is Windows CE #if defined(_WIN32_WCE) - #define EIGEN_OS_WINCE 1 +#define EIGEN_OS_WINCE 1 #else - #define EIGEN_OS_WINCE 0 +#define EIGEN_OS_WINCE 0 #endif /// \internal EIGEN_OS_CYGWIN set to 1 if the OS is Windows/Cygwin #if defined(__CYGWIN__) - #define EIGEN_OS_CYGWIN 1 +#define EIGEN_OS_CYGWIN 1 #else - #define EIGEN_OS_CYGWIN 0 +#define EIGEN_OS_CYGWIN 0 #endif /// \internal EIGEN_OS_WIN_STRICT set to 1 if the OS is really Windows and not some variants -#if EIGEN_OS_WIN && !( EIGEN_OS_WINCE || EIGEN_OS_CYGWIN ) - #define EIGEN_OS_WIN_STRICT 1 +#if EIGEN_OS_WIN && !(EIGEN_OS_WINCE || EIGEN_OS_CYGWIN) +#define EIGEN_OS_WIN_STRICT 1 #else - #define EIGEN_OS_WIN_STRICT 0 +#define EIGEN_OS_WIN_STRICT 0 #endif /// \internal EIGEN_OS_SUN set to 1 if the OS is SUN #if (defined(sun) || defined(__sun)) && !(defined(__SVR4) || defined(__svr4__)) - #define EIGEN_OS_SUN 1 +#define EIGEN_OS_SUN 1 #else - #define EIGEN_OS_SUN 0 +#define EIGEN_OS_SUN 0 #endif /// \internal EIGEN_OS_SOLARIS set to 1 if the OS is Solaris #if (defined(sun) || defined(__sun)) && (defined(__SVR4) || defined(__svr4__)) - #define EIGEN_OS_SOLARIS 1 +#define EIGEN_OS_SOLARIS 1 #else - #define EIGEN_OS_SOLARIS 0 +#define EIGEN_OS_SOLARIS 0 #endif - -#if EIGEN_GNUC_AT_MOST(4,3) && !EIGEN_COMP_CLANG - // see bug 89 - #define EIGEN_SAFE_TO_USE_STANDARD_ASSERT_MACRO 0 +#if EIGEN_GNUC_AT_MOST(4, 3) && !EIGEN_COMP_CLANG +// see bug 89 +#define EIGEN_SAFE_TO_USE_STANDARD_ASSERT_MACRO 0 #else - #define EIGEN_SAFE_TO_USE_STANDARD_ASSERT_MACRO 1 +#define EIGEN_SAFE_TO_USE_STANDARD_ASSERT_MACRO 1 #endif // This macro can be used to prevent from macro expansion, e.g.: @@ -338,15 +341,15 @@ // Cross compiler wrapper around LLVM's __has_builtin #ifdef __has_builtin -# define EIGEN_HAS_BUILTIN(x) __has_builtin(x) +#define EIGEN_HAS_BUILTIN(x) __has_builtin(x) #else -# define EIGEN_HAS_BUILTIN(x) 0 +#define EIGEN_HAS_BUILTIN(x) 0 #endif // A Clang feature extension to determine compiler features. // We use it to determine 'cxx_rvalue_references' #ifndef __has_feature -# define __has_feature(x) 0 +#define __has_feature(x) 0 #endif // Upperbound on the C++ version to use. @@ -356,7 +359,7 @@ #define EIGEN_MAX_CPP_VER 99 #endif -#if EIGEN_MAX_CPP_VER>=11 && (defined(__cplusplus) && (__cplusplus >= 201103L) || EIGEN_COMP_MSVC >= 1900) +#if EIGEN_MAX_CPP_VER >= 11 && (defined(__cplusplus) && (__cplusplus >= 201103L) || EIGEN_COMP_MSVC >= 1900) #define EIGEN_HAS_CXX11 1 #else #define EIGEN_HAS_CXX11 0 @@ -365,31 +368,29 @@ // Do we support r-value references? #ifndef EIGEN_HAS_RVALUE_REFERENCES -#if EIGEN_MAX_CPP_VER>=11 && \ - (__has_feature(cxx_rvalue_references) || \ - (defined(__cplusplus) && __cplusplus >= 201103L) || \ - (EIGEN_COMP_MSVC >= 1600)) - #define EIGEN_HAS_RVALUE_REFERENCES 1 +#if EIGEN_MAX_CPP_VER >= 11 \ + && (__has_feature(cxx_rvalue_references) || (defined(__cplusplus) && __cplusplus >= 201103L) \ + || (EIGEN_COMP_MSVC >= 1600)) +#define EIGEN_HAS_RVALUE_REFERENCES 1 #else - #define EIGEN_HAS_RVALUE_REFERENCES 0 +#define EIGEN_HAS_RVALUE_REFERENCES 0 #endif #endif // Does the compiler support C99? #ifndef EIGEN_HAS_C99_MATH -#if EIGEN_MAX_CPP_VER>=11 && \ - ((defined(__STDC_VERSION__) && (__STDC_VERSION__ >= 199901)) \ - || (defined(__GNUC__) && defined(_GLIBCXX_USE_C99)) \ - || (defined(_LIBCPP_VERSION) && !defined(_MSC_VER))) - #define EIGEN_HAS_C99_MATH 1 +#if EIGEN_MAX_CPP_VER >= 11 \ + && ((defined(__STDC_VERSION__) && (__STDC_VERSION__ >= 199901)) || (defined(__GNUC__) && defined(_GLIBCXX_USE_C99)) \ + || (defined(_LIBCPP_VERSION) && !defined(_MSC_VER))) +#define EIGEN_HAS_C99_MATH 1 #else - #define EIGEN_HAS_C99_MATH 0 +#define EIGEN_HAS_C99_MATH 0 #endif #endif // Does the compiler support result_of? #ifndef EIGEN_HAS_STD_RESULT_OF -#if EIGEN_MAX_CPP_VER>=11 && ((__has_feature(cxx_lambdas) || (defined(__cplusplus) && __cplusplus >= 201103L))) +#if EIGEN_MAX_CPP_VER >= 11 && ((__has_feature(cxx_lambdas) || (defined(__cplusplus) && __cplusplus >= 201103L))) #define EIGEN_HAS_STD_RESULT_OF 1 #else #define EIGEN_HAS_STD_RESULT_OF 0 @@ -398,10 +399,10 @@ // Does the compiler support variadic templates? #ifndef EIGEN_HAS_VARIADIC_TEMPLATES -#if EIGEN_MAX_CPP_VER>=11 && (__cplusplus > 199711L || EIGEN_COMP_MSVC >= 1900) \ - && (!defined(__NVCC__) || !EIGEN_ARCH_ARM_OR_ARM64 || (EIGEN_CUDACC_VER >= 80000) ) - // ^^ Disable the use of variadic templates when compiling with versions of nvcc older than 8.0 on ARM devices: - // this prevents nvcc from crashing when compiling Eigen on Tegra X1 +#if EIGEN_MAX_CPP_VER >= 11 && (__cplusplus > 199711L || EIGEN_COMP_MSVC >= 1900) \ + && (!defined(__NVCC__) || !EIGEN_ARCH_ARM_OR_ARM64 || (EIGEN_CUDACC_VER >= 80000)) +// ^^ Disable the use of variadic templates when compiling with versions of nvcc older than 8.0 on ARM devices: +// this prevents nvcc from crashing when compiling Eigen on Tegra X1 #define EIGEN_HAS_VARIADIC_TEMPLATES 1 #else #define EIGEN_HAS_VARIADIC_TEMPLATES 0 @@ -413,11 +414,12 @@ #ifdef __CUDACC__ // Const expressions are supported provided that c++11 is enabled and we're using either clang or nvcc 7.5 or above -#if EIGEN_MAX_CPP_VER>=14 && (__cplusplus > 199711L && (EIGEN_COMP_CLANG || EIGEN_CUDACC_VER >= 70500)) - #define EIGEN_HAS_CONSTEXPR 1 +#if EIGEN_MAX_CPP_VER >= 14 && (__cplusplus > 199711L && (EIGEN_COMP_CLANG || EIGEN_CUDACC_VER >= 70500)) +#define EIGEN_HAS_CONSTEXPR 1 #endif -#elif EIGEN_MAX_CPP_VER>=14 && (__has_feature(cxx_relaxed_constexpr) || (defined(__cplusplus) && __cplusplus >= 201402L) || \ - (EIGEN_GNUC_AT_LEAST(4,8) && (__cplusplus > 199711L))) +#elif EIGEN_MAX_CPP_VER >= 14 \ + && (__has_feature(cxx_relaxed_constexpr) || (defined(__cplusplus) && __cplusplus >= 201402L) \ + || (EIGEN_GNUC_AT_LEAST(4, 8) && (__cplusplus > 199711L))) #define EIGEN_HAS_CONSTEXPR 1 #endif @@ -430,44 +432,45 @@ // Does the compiler support C++11 math? // Let's be conservative and enable the default C++11 implementation only if we are sure it exists #ifndef EIGEN_HAS_CXX11_MATH - #if EIGEN_MAX_CPP_VER>=11 && ((__cplusplus > 201103L) || (__cplusplus >= 201103L) && (EIGEN_COMP_GNUC_STRICT || EIGEN_COMP_CLANG || EIGEN_COMP_MSVC || EIGEN_COMP_ICC) \ - && (EIGEN_ARCH_i386_OR_x86_64) && (EIGEN_OS_GNULINUX || EIGEN_OS_WIN_STRICT || EIGEN_OS_MAC)) - #define EIGEN_HAS_CXX11_MATH 1 - #else - #define EIGEN_HAS_CXX11_MATH 0 - #endif +#if EIGEN_MAX_CPP_VER >= 11 \ + && ((__cplusplus > 201103L) \ + || (__cplusplus >= 201103L) && (EIGEN_COMP_GNUC_STRICT || EIGEN_COMP_CLANG || EIGEN_COMP_MSVC || EIGEN_COMP_ICC) \ + && (EIGEN_ARCH_i386_OR_x86_64) && (EIGEN_OS_GNULINUX || EIGEN_OS_WIN_STRICT || EIGEN_OS_MAC)) +#define EIGEN_HAS_CXX11_MATH 1 +#else +#define EIGEN_HAS_CXX11_MATH 0 +#endif #endif // Does the compiler support proper C++11 containers? #ifndef EIGEN_HAS_CXX11_CONTAINERS - #if EIGEN_MAX_CPP_VER>=11 && \ - ((__cplusplus > 201103L) \ - || ((__cplusplus >= 201103L) && (EIGEN_COMP_GNUC_STRICT || EIGEN_COMP_CLANG || EIGEN_COMP_ICC>=1400)) \ +#if EIGEN_MAX_CPP_VER >= 11 \ + && ((__cplusplus > 201103L) \ + || ((__cplusplus >= 201103L) && (EIGEN_COMP_GNUC_STRICT || EIGEN_COMP_CLANG || EIGEN_COMP_ICC >= 1400)) \ || EIGEN_COMP_MSVC >= 1900) - #define EIGEN_HAS_CXX11_CONTAINERS 1 - #else - #define EIGEN_HAS_CXX11_CONTAINERS 0 - #endif +#define EIGEN_HAS_CXX11_CONTAINERS 1 +#else +#define EIGEN_HAS_CXX11_CONTAINERS 0 +#endif #endif // Does the compiler support C++11 noexcept? #ifndef EIGEN_HAS_CXX11_NOEXCEPT - #if EIGEN_MAX_CPP_VER>=11 && \ - (__has_feature(cxx_noexcept) \ - || (__cplusplus > 201103L) \ - || ((__cplusplus >= 201103L) && (EIGEN_COMP_GNUC_STRICT || EIGEN_COMP_CLANG || EIGEN_COMP_ICC>=1400)) \ +#if EIGEN_MAX_CPP_VER >= 11 \ + && (__has_feature(cxx_noexcept) || (__cplusplus > 201103L) \ + || ((__cplusplus >= 201103L) && (EIGEN_COMP_GNUC_STRICT || EIGEN_COMP_CLANG || EIGEN_COMP_ICC >= 1400)) \ || EIGEN_COMP_MSVC >= 1900) - #define EIGEN_HAS_CXX11_NOEXCEPT 1 - #else - #define EIGEN_HAS_CXX11_NOEXCEPT 0 - #endif +#define EIGEN_HAS_CXX11_NOEXCEPT 1 +#else +#define EIGEN_HAS_CXX11_NOEXCEPT 0 +#endif #endif /** Allows to disable some optimizations which might affect the accuracy of the result. - * Such optimization are enabled by default, and set EIGEN_FAST_MATH to 0 to disable them. - * They currently include: - * - single precision ArrayBase::sin() and ArrayBase::cos() for SSE and AVX vectorization. - */ + * Such optimization are enabled by default, and set EIGEN_FAST_MATH to 0 to disable them. + * They currently include: + * - single precision ArrayBase::sin() and ArrayBase::cos() for SSE and AVX vectorization. + */ #ifndef EIGEN_FAST_MATH #define EIGEN_FAST_MATH 1 #endif @@ -475,8 +478,8 @@ #define EIGEN_DEBUG_VAR(x) std::cerr << #x << " = " << x << std::endl; // concatenate two tokens -#define EIGEN_CAT2(a,b) a ## b -#define EIGEN_CAT(a,b) EIGEN_CAT2(a,b) +#define EIGEN_CAT2(a, b) a##b +#define EIGEN_CAT(a, b) EIGEN_CAT2(a, b) #define EIGEN_COMMA , @@ -500,10 +503,11 @@ // it uses __attribute__((always_inline)) on GCC, which most of the time is useless and can severely harm compile times. // FIXME with the always_inline attribute, // gcc 3.4.x and 4.1 reports the following compilation error: -// Eval.h:91: sorry, unimplemented: inlining failed in call to 'const Eigen::Eval Eigen::MatrixBase::eval() const' +// Eval.h:91: sorry, unimplemented: inlining failed in call to 'const Eigen::Eval Eigen::MatrixBase::eval() const' // : function body not available // See also bug 1367 -#if EIGEN_GNUC_AT_LEAST(4,2) +#if EIGEN_GNUC_AT_LEAST(4, 2) #define EIGEN_ALWAYS_INLINE __attribute__((always_inline)) inline #else #define EIGEN_ALWAYS_INLINE EIGEN_STRONG_INLINE @@ -531,47 +535,48 @@ #define EIGEN_DEFINE_FUNCTION_ALLOWING_MULTIPLE_DEFINITIONS inline #ifdef NDEBUG -# ifndef EIGEN_NO_DEBUG -# define EIGEN_NO_DEBUG -# endif +#ifndef EIGEN_NO_DEBUG +#define EIGEN_NO_DEBUG +#endif #endif // eigen_plain_assert is where we implement the workaround for the assert() bug in GCC <= 4.3, see bug 89 #ifdef EIGEN_NO_DEBUG - #define eigen_plain_assert(x) -#else - #if EIGEN_SAFE_TO_USE_STANDARD_ASSERT_MACRO - namespace Eigen { - namespace internal { - inline bool copy_bool(bool b) { return b; } - } - } - #define eigen_plain_assert(x) assert(x) - #else - // work around bug 89 - #include // for abort - #include // for std::cerr - - namespace Eigen { - namespace internal { - // trivial function copying a bool. Must be EIGEN_DONT_INLINE, so we implement it after including Eigen headers. - // see bug 89. - namespace { +#define eigen_plain_assert(x) +#else +#if EIGEN_SAFE_TO_USE_STANDARD_ASSERT_MACRO +namespace Eigen { +namespace internal { + inline bool copy_bool(bool b) { return b; } +}// namespace internal +}// namespace Eigen +#define eigen_plain_assert(x) assert(x) +#else +// work around bug 89 +#include // for abort +#include // for std::cerr + +namespace Eigen { +namespace internal { + // trivial function copying a bool. Must be EIGEN_DONT_INLINE, so we implement it after including Eigen headers. + // see bug 89. + namespace { EIGEN_DONT_INLINE bool copy_bool(bool b) { return b; } - } - inline void assert_fail(const char *condition, const char *function, const char *file, int line) - { - std::cerr << "assertion failed: " << condition << " in function " << function << " at " << file << ":" << line << std::endl; - abort(); - } - } - } - #define eigen_plain_assert(x) \ - do { \ - if(!Eigen::internal::copy_bool(x)) \ - Eigen::internal::assert_fail(EIGEN_MAKESTRING(x), __PRETTY_FUNCTION__, __FILE__, __LINE__); \ - } while(false) - #endif + }// namespace + inline void assert_fail(const char *condition, const char *function, const char *file, int line) + { + std::cerr << "assertion failed: " << condition << " in function " << function << " at " << file << ":" << line + << std::endl; + abort(); + } +}// namespace internal +}// namespace Eigen +#define eigen_plain_assert(x) \ + do { \ + if (!Eigen::internal::copy_bool(x)) \ + Eigen::internal::assert_fail(EIGEN_MAKESTRING(x), __PRETTY_FUNCTION__, __FILE__, __LINE__); \ + } while (false) +#endif #endif // eigen_assert can be overridden @@ -592,15 +597,15 @@ #endif #ifndef EIGEN_NO_DEPRECATED_WARNING - #if EIGEN_COMP_GNUC - #define EIGEN_DEPRECATED __attribute__((deprecated)) - #elif EIGEN_COMP_MSVC - #define EIGEN_DEPRECATED __declspec(deprecated) - #else - #define EIGEN_DEPRECATED - #endif +#if EIGEN_COMP_GNUC +#define EIGEN_DEPRECATED __attribute__((deprecated)) +#elif EIGEN_COMP_MSVC +#define EIGEN_DEPRECATED __declspec(deprecated) +#else +#define EIGEN_DEPRECATED +#endif #else - #define EIGEN_DEPRECATED +#define EIGEN_DEPRECATED #endif #if EIGEN_COMP_GNUC @@ -611,18 +616,18 @@ // Suppresses 'unused variable' warnings. namespace Eigen { - namespace internal { - template EIGEN_DEVICE_FUNC void ignore_unused_variable(const T&) {} - } -} +namespace internal { + template EIGEN_DEVICE_FUNC void ignore_unused_variable(const T &) {} +}// namespace internal +}// namespace Eigen #define EIGEN_UNUSED_VARIABLE(var) Eigen::internal::ignore_unused_variable(var); #if !defined(EIGEN_ASM_COMMENT) - #if EIGEN_COMP_GNUC && (EIGEN_ARCH_i386_OR_x86_64 || EIGEN_ARCH_ARM_OR_ARM64) - #define EIGEN_ASM_COMMENT(X) __asm__("#" X) - #else - #define EIGEN_ASM_COMMENT(X) - #endif +#if EIGEN_COMP_GNUC && (EIGEN_ARCH_i386_OR_x86_64 || EIGEN_ARCH_ARM_OR_ARM64) +#define EIGEN_ASM_COMMENT(X) __asm__("#" X) +#else +#define EIGEN_ASM_COMMENT(X) +#endif #endif @@ -647,29 +652,29 @@ namespace Eigen { * vectorized and non-vectorized code. */ #if (defined __CUDACC__) - #define EIGEN_ALIGN_TO_BOUNDARY(n) __align__(n) +#define EIGEN_ALIGN_TO_BOUNDARY(n) __align__(n) #elif EIGEN_COMP_GNUC || EIGEN_COMP_PGI || EIGEN_COMP_IBM || EIGEN_COMP_ARM - #define EIGEN_ALIGN_TO_BOUNDARY(n) __attribute__((aligned(n))) +#define EIGEN_ALIGN_TO_BOUNDARY(n) __attribute__((aligned(n))) #elif EIGEN_COMP_MSVC - #define EIGEN_ALIGN_TO_BOUNDARY(n) __declspec(align(n)) +#define EIGEN_ALIGN_TO_BOUNDARY(n) __declspec(align(n)) #elif EIGEN_COMP_SUNCC - // FIXME not sure about this one: - #define EIGEN_ALIGN_TO_BOUNDARY(n) __attribute__((aligned(n))) + // FIXME not sure about this one: +#define EIGEN_ALIGN_TO_BOUNDARY(n) __attribute__((aligned(n))) #else - #error Please tell me what is the equivalent of __attribute__((aligned(n))) for your compiler +#error Please tell me what is the equivalent of __attribute__((aligned(n))) for your compiler #endif // If the user explicitly disable vectorization, then we also disable alignment #if defined(EIGEN_DONT_VECTORIZE) - #define EIGEN_IDEAL_MAX_ALIGN_BYTES 0 +#define EIGEN_IDEAL_MAX_ALIGN_BYTES 0 #elif defined(EIGEN_VECTORIZE_AVX512) - // 64 bytes static alignmeent is preferred only if really required - #define EIGEN_IDEAL_MAX_ALIGN_BYTES 64 + // 64 bytes static alignmeent is preferred only if really required +#define EIGEN_IDEAL_MAX_ALIGN_BYTES 64 #elif defined(__AVX__) - // 32 bytes static alignmeent is preferred only if really required - #define EIGEN_IDEAL_MAX_ALIGN_BYTES 32 + // 32 bytes static alignmeent is preferred only if really required +#define EIGEN_IDEAL_MAX_ALIGN_BYTES 32 #else - #define EIGEN_IDEAL_MAX_ALIGN_BYTES 16 +#define EIGEN_IDEAL_MAX_ALIGN_BYTES 16 #endif @@ -680,80 +685,78 @@ namespace Eigen { // that unless EIGEN_ALIGN is defined and not equal to 0, the data may not be // aligned at all regardless of the value of this #define. -#if (defined(EIGEN_DONT_ALIGN_STATICALLY) || defined(EIGEN_DONT_ALIGN)) && defined(EIGEN_MAX_STATIC_ALIGN_BYTES) && EIGEN_MAX_STATIC_ALIGN_BYTES>0 +#if (defined(EIGEN_DONT_ALIGN_STATICALLY) || defined(EIGEN_DONT_ALIGN)) && defined(EIGEN_MAX_STATIC_ALIGN_BYTES) \ + && EIGEN_MAX_STATIC_ALIGN_BYTES > 0 #error EIGEN_MAX_STATIC_ALIGN_BYTES and EIGEN_DONT_ALIGN[_STATICALLY] are both defined with EIGEN_MAX_STATIC_ALIGN_BYTES!=0. Use EIGEN_MAX_STATIC_ALIGN_BYTES=0 as a synonym of EIGEN_DONT_ALIGN_STATICALLY. #endif // EIGEN_DONT_ALIGN_STATICALLY and EIGEN_DONT_ALIGN are deprectated // They imply EIGEN_MAX_STATIC_ALIGN_BYTES=0 #if defined(EIGEN_DONT_ALIGN_STATICALLY) || defined(EIGEN_DONT_ALIGN) - #ifdef EIGEN_MAX_STATIC_ALIGN_BYTES - #undef EIGEN_MAX_STATIC_ALIGN_BYTES - #endif - #define EIGEN_MAX_STATIC_ALIGN_BYTES 0 +#ifdef EIGEN_MAX_STATIC_ALIGN_BYTES +#undef EIGEN_MAX_STATIC_ALIGN_BYTES +#endif +#define EIGEN_MAX_STATIC_ALIGN_BYTES 0 #endif #ifndef EIGEN_MAX_STATIC_ALIGN_BYTES - // Try to automatically guess what is the best default value for EIGEN_MAX_STATIC_ALIGN_BYTES - - // 16 byte alignment is only useful for vectorization. Since it affects the ABI, we need to enable - // 16 byte alignment on all platforms where vectorization might be enabled. In theory we could always - // enable alignment, but it can be a cause of problems on some platforms, so we just disable it in - // certain common platform (compiler+architecture combinations) to avoid these problems. - // Only static alignment is really problematic (relies on nonstandard compiler extensions), - // try to keep heap alignment even when we have to disable static alignment. - #if EIGEN_COMP_GNUC && !(EIGEN_ARCH_i386_OR_x86_64 || EIGEN_ARCH_ARM_OR_ARM64 || EIGEN_ARCH_PPC || EIGEN_ARCH_IA64) - #define EIGEN_GCC_AND_ARCH_DOESNT_WANT_STACK_ALIGNMENT 1 - #elif EIGEN_ARCH_ARM_OR_ARM64 && EIGEN_COMP_GNUC_STRICT && EIGEN_GNUC_AT_MOST(4, 6) - // Old versions of GCC on ARM, at least 4.4, were once seen to have buggy static alignment support. - // Not sure which version fixed it, hopefully it doesn't affect 4.7, which is still somewhat in use. - // 4.8 and newer seem definitely unaffected. - #define EIGEN_GCC_AND_ARCH_DOESNT_WANT_STACK_ALIGNMENT 1 - #else - #define EIGEN_GCC_AND_ARCH_DOESNT_WANT_STACK_ALIGNMENT 0 - #endif - - // static alignment is completely disabled with GCC 3, Sun Studio, and QCC/QNX - #if !EIGEN_GCC_AND_ARCH_DOESNT_WANT_STACK_ALIGNMENT \ - && !EIGEN_GCC3_OR_OLDER \ - && !EIGEN_COMP_SUNCC \ - && !EIGEN_OS_QNX - #define EIGEN_ARCH_WANTS_STACK_ALIGNMENT 1 - #else - #define EIGEN_ARCH_WANTS_STACK_ALIGNMENT 0 - #endif - - #if EIGEN_ARCH_WANTS_STACK_ALIGNMENT - #define EIGEN_MAX_STATIC_ALIGN_BYTES EIGEN_IDEAL_MAX_ALIGN_BYTES - #else - #define EIGEN_MAX_STATIC_ALIGN_BYTES 0 - #endif +// Try to automatically guess what is the best default value for EIGEN_MAX_STATIC_ALIGN_BYTES + +// 16 byte alignment is only useful for vectorization. Since it affects the ABI, we need to enable +// 16 byte alignment on all platforms where vectorization might be enabled. In theory we could always +// enable alignment, but it can be a cause of problems on some platforms, so we just disable it in +// certain common platform (compiler+architecture combinations) to avoid these problems. +// Only static alignment is really problematic (relies on nonstandard compiler extensions), +// try to keep heap alignment even when we have to disable static alignment. +#if EIGEN_COMP_GNUC && !(EIGEN_ARCH_i386_OR_x86_64 || EIGEN_ARCH_ARM_OR_ARM64 || EIGEN_ARCH_PPC || EIGEN_ARCH_IA64) +#define EIGEN_GCC_AND_ARCH_DOESNT_WANT_STACK_ALIGNMENT 1 +#elif EIGEN_ARCH_ARM_OR_ARM64 && EIGEN_COMP_GNUC_STRICT && EIGEN_GNUC_AT_MOST(4, 6) + // Old versions of GCC on ARM, at least 4.4, were once seen to have buggy static alignment support. +// Not sure which version fixed it, hopefully it doesn't affect 4.7, which is still somewhat in use. +// 4.8 and newer seem definitely unaffected. +#define EIGEN_GCC_AND_ARCH_DOESNT_WANT_STACK_ALIGNMENT 1 +#else +#define EIGEN_GCC_AND_ARCH_DOESNT_WANT_STACK_ALIGNMENT 0 +#endif + +// static alignment is completely disabled with GCC 3, Sun Studio, and QCC/QNX +#if !EIGEN_GCC_AND_ARCH_DOESNT_WANT_STACK_ALIGNMENT && !EIGEN_GCC3_OR_OLDER && !EIGEN_COMP_SUNCC && !EIGEN_OS_QNX +#define EIGEN_ARCH_WANTS_STACK_ALIGNMENT 1 +#else +#define EIGEN_ARCH_WANTS_STACK_ALIGNMENT 0 +#endif + +#if EIGEN_ARCH_WANTS_STACK_ALIGNMENT +#define EIGEN_MAX_STATIC_ALIGN_BYTES EIGEN_IDEAL_MAX_ALIGN_BYTES +#else +#define EIGEN_MAX_STATIC_ALIGN_BYTES 0 +#endif #endif // If EIGEN_MAX_ALIGN_BYTES is defined, then it is considered as an upper bound for EIGEN_MAX_ALIGN_BYTES -#if defined(EIGEN_MAX_ALIGN_BYTES) && EIGEN_MAX_ALIGN_BYTES0 is the true test whether we want to align arrays on the stack or not. -// It takes into account both the user choice to explicitly enable/disable alignment (by settting EIGEN_MAX_STATIC_ALIGN_BYTES) -// and the architecture config (EIGEN_ARCH_WANTS_STACK_ALIGNMENT). -// Henceforth, only EIGEN_MAX_STATIC_ALIGN_BYTES should be used. +// It takes into account both the user choice to explicitly enable/disable alignment (by settting +// EIGEN_MAX_STATIC_ALIGN_BYTES) and the architecture config (EIGEN_ARCH_WANTS_STACK_ALIGNMENT). Henceforth, only +// EIGEN_MAX_STATIC_ALIGN_BYTES should be used. // Shortcuts to EIGEN_ALIGN_TO_BOUNDARY -#define EIGEN_ALIGN8 EIGEN_ALIGN_TO_BOUNDARY(8) +#define EIGEN_ALIGN8 EIGEN_ALIGN_TO_BOUNDARY(8) #define EIGEN_ALIGN16 EIGEN_ALIGN_TO_BOUNDARY(16) #define EIGEN_ALIGN32 EIGEN_ALIGN_TO_BOUNDARY(32) #define EIGEN_ALIGN64 EIGEN_ALIGN_TO_BOUNDARY(64) -#if EIGEN_MAX_STATIC_ALIGN_BYTES>0 +#if EIGEN_MAX_STATIC_ALIGN_BYTES > 0 #define EIGEN_ALIGN_MAX EIGEN_ALIGN_TO_BOUNDARY(EIGEN_MAX_STATIC_ALIGN_BYTES) #else #define EIGEN_ALIGN_MAX @@ -762,17 +765,17 @@ namespace Eigen { // Dynamic alignment control -#if defined(EIGEN_DONT_ALIGN) && defined(EIGEN_MAX_ALIGN_BYTES) && EIGEN_MAX_ALIGN_BYTES>0 +#if defined(EIGEN_DONT_ALIGN) && defined(EIGEN_MAX_ALIGN_BYTES) && EIGEN_MAX_ALIGN_BYTES > 0 #error EIGEN_MAX_ALIGN_BYTES and EIGEN_DONT_ALIGN are both defined with EIGEN_MAX_ALIGN_BYTES!=0. Use EIGEN_MAX_ALIGN_BYTES=0 as a synonym of EIGEN_DONT_ALIGN. #endif #ifdef EIGEN_DONT_ALIGN - #ifdef EIGEN_MAX_ALIGN_BYTES - #undef EIGEN_MAX_ALIGN_BYTES - #endif - #define EIGEN_MAX_ALIGN_BYTES 0 +#ifdef EIGEN_MAX_ALIGN_BYTES +#undef EIGEN_MAX_ALIGN_BYTES +#endif +#define EIGEN_MAX_ALIGN_BYTES 0 #elif !defined(EIGEN_MAX_ALIGN_BYTES) - #define EIGEN_MAX_ALIGN_BYTES EIGEN_IDEAL_MAX_ALIGN_BYTES +#define EIGEN_MAX_ALIGN_BYTES EIGEN_IDEAL_MAX_ALIGN_BYTES #endif #if EIGEN_IDEAL_MAX_ALIGN_BYTES > EIGEN_MAX_ALIGN_BYTES @@ -790,10 +793,10 @@ namespace Eigen { #ifdef EIGEN_DONT_USE_RESTRICT_KEYWORD - #define EIGEN_RESTRICT +#define EIGEN_RESTRICT #endif #ifndef EIGEN_RESTRICT - #define EIGEN_RESTRICT __restrict +#define EIGEN_RESTRICT __restrict #endif #ifndef EIGEN_STACK_ALLOCATION_LIMIT @@ -814,188 +817,221 @@ namespace Eigen { // just an empty macro ! #define EIGEN_EMPTY -#if EIGEN_COMP_MSVC_STRICT && (EIGEN_COMP_MSVC < 1900 || EIGEN_CUDACC_VER>0) - // for older MSVC versions, as well as 1900 && CUDA 8, using the base operator is sufficient (cf Bugs 1000, 1324) - #define EIGEN_INHERIT_ASSIGNMENT_EQUAL_OPERATOR(Derived) \ - using Base::operator =; -#elif EIGEN_COMP_CLANG // workaround clang bug (see http://forum.kde.org/viewtopic.php?f=74&t=102653) - #define EIGEN_INHERIT_ASSIGNMENT_EQUAL_OPERATOR(Derived) \ - using Base::operator =; \ - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Derived& operator=(const Derived& other) { Base::operator=(other); return *this; } \ - template \ - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Derived& operator=(const DenseBase& other) { Base::operator=(other.derived()); return *this; } +#if EIGEN_COMP_MSVC_STRICT && (EIGEN_COMP_MSVC < 1900 || EIGEN_CUDACC_VER > 0) + // for older MSVC versions, as well as 1900 && CUDA 8, using the base operator is sufficient (cf Bugs 1000, 1324) +#define EIGEN_INHERIT_ASSIGNMENT_EQUAL_OPERATOR(Derived) using Base::operator=; +#elif EIGEN_COMP_CLANG// workaround clang bug (see http://forum.kde.org/viewtopic.php?f=74&t=102653) +#define EIGEN_INHERIT_ASSIGNMENT_EQUAL_OPERATOR(Derived) \ + using Base::operator=; \ + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Derived &operator=(const Derived &other) \ + { \ + Base::operator=(other); \ + return *this; \ + } \ + template \ + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Derived &operator=(const DenseBase &other) \ + { \ + Base::operator=(other.derived()); \ + return *this; \ + } #else - #define EIGEN_INHERIT_ASSIGNMENT_EQUAL_OPERATOR(Derived) \ - using Base::operator =; \ - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Derived& operator=(const Derived& other) \ - { \ - Base::operator=(other); \ - return *this; \ - } +#define EIGEN_INHERIT_ASSIGNMENT_EQUAL_OPERATOR(Derived) \ + using Base::operator=; \ + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Derived &operator=(const Derived &other) \ + { \ + Base::operator=(other); \ + return *this; \ + } #endif /** \internal * \brief Macro to manually inherit assignment operators. - * This is necessary, because the implicitly defined assignment operator gets deleted when a custom operator= is defined. + * This is necessary, because the implicitly defined assignment operator gets deleted when a custom operator= is + * defined. */ #define EIGEN_INHERIT_ASSIGNMENT_OPERATORS(Derived) EIGEN_INHERIT_ASSIGNMENT_EQUAL_OPERATOR(Derived) /** -* Just a side note. Commenting within defines works only by documenting -* behind the object (via '!<'). Comments cannot be multi-line and thus -* we have these extra long lines. What is confusing doxygen over here is -* that we use '\' and basically have a bunch of typedefs with their -* documentation in a single line. -**/ - -#define EIGEN_GENERIC_PUBLIC_INTERFACE(Derived) \ - typedef typename Eigen::internal::traits::Scalar Scalar; /*!< \brief Numeric type, e.g. float, double, int or std::complex. */ \ - typedef typename Eigen::NumTraits::Real RealScalar; /*!< \brief The underlying numeric type for composed scalar types. \details In cases where Scalar is e.g. std::complex, T were corresponding to RealScalar. */ \ - typedef typename Base::CoeffReturnType CoeffReturnType; /*!< \brief The return type for coefficient access. \details Depending on whether the object allows direct coefficient access (e.g. for a MatrixXd), this type is either 'const Scalar&' or simply 'Scalar' for objects that do not allow direct coefficient access. */ \ - typedef typename Eigen::internal::ref_selector::type Nested; \ - typedef typename Eigen::internal::traits::StorageKind StorageKind; \ - typedef typename Eigen::internal::traits::StorageIndex StorageIndex; \ - enum { RowsAtCompileTime = Eigen::internal::traits::RowsAtCompileTime, \ - ColsAtCompileTime = Eigen::internal::traits::ColsAtCompileTime, \ - Flags = Eigen::internal::traits::Flags, \ - SizeAtCompileTime = Base::SizeAtCompileTime, \ - MaxSizeAtCompileTime = Base::MaxSizeAtCompileTime, \ - IsVectorAtCompileTime = Base::IsVectorAtCompileTime }; \ - using Base::derived; \ + * Just a side note. Commenting within defines works only by documenting + * behind the object (via '!<'). Comments cannot be multi-line and thus + * we have these extra long lines. What is confusing doxygen over here is + * that we use '\' and basically have a bunch of typedefs with their + * documentation in a single line. + **/ + +#define EIGEN_GENERIC_PUBLIC_INTERFACE(Derived) \ + typedef typename Eigen::internal::traits::Scalar \ + Scalar; /*!< \brief Numeric type, e.g. float, double, int or std::complex. */ \ + typedef typename Eigen::NumTraits::Real \ + RealScalar; /*!< \brief The underlying numeric type for composed scalar types. \details In cases where Scalar is \ + e.g. std::complex, T were corresponding to RealScalar. */ \ + typedef typename Base::CoeffReturnType \ + CoeffReturnType; /*!< \brief The return type for coefficient access. \details Depending on whether the object \ + allows direct coefficient access (e.g. for a MatrixXd), this type is either 'const Scalar&' or \ + simply 'Scalar' for objects that do not allow direct coefficient access. */ \ + typedef typename Eigen::internal::ref_selector::type Nested; \ + typedef typename Eigen::internal::traits::StorageKind StorageKind; \ + typedef typename Eigen::internal::traits::StorageIndex StorageIndex; \ + enum { \ + RowsAtCompileTime = Eigen::internal::traits::RowsAtCompileTime, \ + ColsAtCompileTime = Eigen::internal::traits::ColsAtCompileTime, \ + Flags = Eigen::internal::traits::Flags, \ + SizeAtCompileTime = Base::SizeAtCompileTime, \ + MaxSizeAtCompileTime = Base::MaxSizeAtCompileTime, \ + IsVectorAtCompileTime = Base::IsVectorAtCompileTime \ + }; \ + using Base::derived; \ using Base::const_cast_derived; // FIXME Maybe the EIGEN_DENSE_PUBLIC_INTERFACE could be removed as importing PacketScalar is rarely needed #define EIGEN_DENSE_PUBLIC_INTERFACE(Derived) \ - EIGEN_GENERIC_PUBLIC_INTERFACE(Derived) \ + EIGEN_GENERIC_PUBLIC_INTERFACE(Derived) \ typedef typename Base::PacketScalar PacketScalar; -#define EIGEN_PLAIN_ENUM_MIN(a,b) (((int)a <= (int)b) ? (int)a : (int)b) -#define EIGEN_PLAIN_ENUM_MAX(a,b) (((int)a >= (int)b) ? (int)a : (int)b) +#define EIGEN_PLAIN_ENUM_MIN(a, b) (((int)a <= (int)b) ? (int)a : (int)b) +#define EIGEN_PLAIN_ENUM_MAX(a, b) (((int)a >= (int)b) ? (int)a : (int)b) // EIGEN_SIZE_MIN_PREFER_DYNAMIC gives the min between compile-time sizes. 0 has absolute priority, followed by 1, // followed by Dynamic, followed by other finite values. The reason for giving Dynamic the priority over // finite values is that min(3, Dynamic) should be Dynamic, since that could be anything between 0 and 3. -#define EIGEN_SIZE_MIN_PREFER_DYNAMIC(a,b) (((int)a == 0 || (int)b == 0) ? 0 \ - : ((int)a == 1 || (int)b == 1) ? 1 \ - : ((int)a == Dynamic || (int)b == Dynamic) ? Dynamic \ - : ((int)a <= (int)b) ? (int)a : (int)b) - -// EIGEN_SIZE_MIN_PREFER_FIXED is a variant of EIGEN_SIZE_MIN_PREFER_DYNAMIC comparing MaxSizes. The difference is that finite values -// now have priority over Dynamic, so that min(3, Dynamic) gives 3. Indeed, whatever the actual value is +#define EIGEN_SIZE_MIN_PREFER_DYNAMIC(a, b) \ + (((int)a == 0 || (int)b == 0) ? 0 \ + : ((int)a == 1 || (int)b == 1) ? 1 \ + : ((int)a == Dynamic || (int)b == Dynamic) ? Dynamic \ + : ((int)a <= (int)b) ? (int)a \ + : (int)b) + +// EIGEN_SIZE_MIN_PREFER_FIXED is a variant of EIGEN_SIZE_MIN_PREFER_DYNAMIC comparing MaxSizes. The difference is that +// finite values now have priority over Dynamic, so that min(3, Dynamic) gives 3. Indeed, whatever the actual value is // (between 0 and 3), it is not more than 3. -#define EIGEN_SIZE_MIN_PREFER_FIXED(a,b) (((int)a == 0 || (int)b == 0) ? 0 \ - : ((int)a == 1 || (int)b == 1) ? 1 \ - : ((int)a == Dynamic && (int)b == Dynamic) ? Dynamic \ - : ((int)a == Dynamic) ? (int)b \ - : ((int)b == Dynamic) ? (int)a \ - : ((int)a <= (int)b) ? (int)a : (int)b) +#define EIGEN_SIZE_MIN_PREFER_FIXED(a, b) \ + (((int)a == 0 || (int)b == 0) ? 0 \ + : ((int)a == 1 || (int)b == 1) ? 1 \ + : ((int)a == Dynamic && (int)b == Dynamic) ? Dynamic \ + : ((int)a == Dynamic) ? (int)b \ + : ((int)b == Dynamic) ? (int)a \ + : ((int)a <= (int)b) ? (int)a \ + : (int)b) // see EIGEN_SIZE_MIN_PREFER_DYNAMIC. No need for a separate variant for MaxSizes here. -#define EIGEN_SIZE_MAX(a,b) (((int)a == Dynamic || (int)b == Dynamic) ? Dynamic \ - : ((int)a >= (int)b) ? (int)a : (int)b) +#define EIGEN_SIZE_MAX(a, b) (((int)a == Dynamic || (int)b == Dynamic) ? Dynamic : ((int)a >= (int)b) ? (int)a : (int)b) -#define EIGEN_LOGICAL_XOR(a,b) (((a) || (b)) && !((a) && (b))) +#define EIGEN_LOGICAL_XOR(a, b) (((a) || (b)) && !((a) && (b))) -#define EIGEN_IMPLIES(a,b) (!(a) || (b)) +#define EIGEN_IMPLIES(a, b) (!(a) || (b)) // the expression type of a standard coefficient wise binary operation -#define EIGEN_CWISE_BINARY_RETURN_TYPE(LHS,RHS,OPNAME) \ - CwiseBinaryOp< \ - EIGEN_CAT(EIGEN_CAT(internal::scalar_,OPNAME),_op)< \ - typename internal::traits::Scalar, \ - typename internal::traits::Scalar \ - >, \ - const LHS, \ - const RHS \ - > - -#define EIGEN_MAKE_CWISE_BINARY_OP(METHOD,OPNAME) \ - template \ - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const EIGEN_CWISE_BINARY_RETURN_TYPE(Derived,OtherDerived,OPNAME) \ - (METHOD)(const EIGEN_CURRENT_STORAGE_BASE_CLASS &other) const \ - { \ - return EIGEN_CWISE_BINARY_RETURN_TYPE(Derived,OtherDerived,OPNAME)(derived(), other.derived()); \ +#define EIGEN_CWISE_BINARY_RETURN_TYPE(LHS, RHS, OPNAME) \ + CwiseBinaryOp::Scalar, \ + typename internal::traits::Scalar>, \ + const LHS, const RHS > + +#define EIGEN_MAKE_CWISE_BINARY_OP(METHOD, OPNAME) \ + template \ + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const EIGEN_CWISE_BINARY_RETURN_TYPE(Derived, OtherDerived, OPNAME)(METHOD)( \ + const EIGEN_CURRENT_STORAGE_BASE_CLASS &other) const \ + { \ + return EIGEN_CWISE_BINARY_RETURN_TYPE(Derived, OtherDerived, OPNAME)(derived(), other.derived()); \ } -#define EIGEN_SCALAR_BINARY_SUPPORTED(OPNAME,TYPEA,TYPEB) \ - (Eigen::internal::has_ReturnType > >::value) +#define EIGEN_SCALAR_BINARY_SUPPORTED(OPNAME, TYPEA, TYPEB) \ + (Eigen::internal::has_ReturnType> > ::value) -#define EIGEN_EXPR_BINARYOP_SCALAR_RETURN_TYPE(EXPR,SCALAR,OPNAME) \ - CwiseBinaryOp::Scalar,SCALAR>, const EXPR, \ - const typename internal::plain_constant_type::type> +#define EIGEN_EXPR_BINARYOP_SCALAR_RETURN_TYPE(EXPR, SCALAR, OPNAME) \ + CwiseBinaryOp::Scalar, \ + SCALAR>, \ + const EXPR, const typename internal::plain_constant_type::type > -#define EIGEN_SCALAR_BINARYOP_EXPR_RETURN_TYPE(SCALAR,EXPR,OPNAME) \ - CwiseBinaryOp::Scalar>, \ - const typename internal::plain_constant_type::type, const EXPR> +#define EIGEN_SCALAR_BINARYOP_EXPR_RETURN_TYPE(SCALAR, EXPR, OPNAME) \ + CwiseBinaryOp::Scalar>, \ + const typename internal::plain_constant_type::type, const EXPR > // Workaround for MSVC 2010 (see ML thread "patch with compile for for MSVC 2010") -#if EIGEN_COMP_MSVC_STRICT<=1600 -#define EIGEN_MSVC10_WORKAROUND_BINARYOP_RETURN_TYPE(X) typename internal::enable_if::type +#if EIGEN_COMP_MSVC_STRICT <= 1600 +#define EIGEN_MSVC10_WORKAROUND_BINARYOP_RETURN_TYPE(X) typename internal::enable_if::type #else #define EIGEN_MSVC10_WORKAROUND_BINARYOP_RETURN_TYPE(X) X #endif -#define EIGEN_MAKE_SCALAR_BINARY_OP_ONTHERIGHT(METHOD,OPNAME) \ - template EIGEN_DEVICE_FUNC inline \ - EIGEN_MSVC10_WORKAROUND_BINARYOP_RETURN_TYPE(const EIGEN_EXPR_BINARYOP_SCALAR_RETURN_TYPE(Derived,typename internal::promote_scalar_arg::type,OPNAME))\ - (METHOD)(const T& scalar) const { \ - typedef typename internal::promote_scalar_arg::type PromotedT; \ - return EIGEN_EXPR_BINARYOP_SCALAR_RETURN_TYPE(Derived,PromotedT,OPNAME)(derived(), \ - typename internal::plain_constant_type::type(derived().rows(), derived().cols(), internal::scalar_constant_op(scalar))); \ +#define EIGEN_MAKE_SCALAR_BINARY_OP_ONTHERIGHT(METHOD, OPNAME) \ + template \ + EIGEN_DEVICE_FUNC inline EIGEN_MSVC10_WORKAROUND_BINARYOP_RETURN_TYPE( \ + const EIGEN_EXPR_BINARYOP_SCALAR_RETURN_TYPE(Derived, \ + typename internal::promote_scalar_arg::type, \ + OPNAME))(METHOD)(const T &scalar) const \ + { \ + typedef typename internal::promote_scalar_arg::type \ + PromotedT; \ + return EIGEN_EXPR_BINARYOP_SCALAR_RETURN_TYPE(Derived, PromotedT, OPNAME)(derived(), \ + typename internal::plain_constant_type::type( \ + derived().rows(), derived().cols(), internal::scalar_constant_op(scalar))); \ } -#define EIGEN_MAKE_SCALAR_BINARY_OP_ONTHELEFT(METHOD,OPNAME) \ - template EIGEN_DEVICE_FUNC inline friend \ - EIGEN_MSVC10_WORKAROUND_BINARYOP_RETURN_TYPE(const EIGEN_SCALAR_BINARYOP_EXPR_RETURN_TYPE(typename internal::promote_scalar_arg::type,Derived,OPNAME)) \ - (METHOD)(const T& scalar, const StorageBaseType& matrix) { \ - typedef typename internal::promote_scalar_arg::type PromotedT; \ - return EIGEN_SCALAR_BINARYOP_EXPR_RETURN_TYPE(PromotedT,Derived,OPNAME)( \ - typename internal::plain_constant_type::type(matrix.derived().rows(), matrix.derived().cols(), internal::scalar_constant_op(scalar)), matrix.derived()); \ +#define EIGEN_MAKE_SCALAR_BINARY_OP_ONTHELEFT(METHOD, OPNAME) \ + template \ + EIGEN_DEVICE_FUNC inline friend EIGEN_MSVC10_WORKAROUND_BINARYOP_RETURN_TYPE( \ + const EIGEN_SCALAR_BINARYOP_EXPR_RETURN_TYPE( \ + typename internal::promote_scalar_arg::type, \ + Derived, \ + OPNAME))(METHOD)(const T &scalar, const StorageBaseType &matrix) \ + { \ + typedef typename internal::promote_scalar_arg::type \ + PromotedT; \ + return EIGEN_SCALAR_BINARYOP_EXPR_RETURN_TYPE(PromotedT, Derived, OPNAME)( \ + typename internal::plain_constant_type::type( \ + matrix.derived().rows(), matrix.derived().cols(), internal::scalar_constant_op(scalar)), \ + matrix.derived()); \ } -#define EIGEN_MAKE_SCALAR_BINARY_OP(METHOD,OPNAME) \ - EIGEN_MAKE_SCALAR_BINARY_OP_ONTHELEFT(METHOD,OPNAME) \ - EIGEN_MAKE_SCALAR_BINARY_OP_ONTHERIGHT(METHOD,OPNAME) +#define EIGEN_MAKE_SCALAR_BINARY_OP(METHOD, OPNAME) \ + EIGEN_MAKE_SCALAR_BINARY_OP_ONTHELEFT(METHOD, OPNAME) \ + EIGEN_MAKE_SCALAR_BINARY_OP_ONTHERIGHT(METHOD, OPNAME) #ifdef EIGEN_EXCEPTIONS -# define EIGEN_THROW_X(X) throw X -# define EIGEN_THROW throw -# define EIGEN_TRY try -# define EIGEN_CATCH(X) catch (X) +#define EIGEN_THROW_X(X) throw X +#define EIGEN_THROW throw +#define EIGEN_TRY try +#define EIGEN_CATCH(X) catch (X) +#else +#ifdef __CUDA_ARCH__ +#define EIGEN_THROW_X(X) asm("trap;") +#define EIGEN_THROW asm("trap;") #else -# ifdef __CUDA_ARCH__ -# define EIGEN_THROW_X(X) asm("trap;") -# define EIGEN_THROW asm("trap;") -# else -# define EIGEN_THROW_X(X) std::abort() -# define EIGEN_THROW std::abort() -# endif -# define EIGEN_TRY if (true) -# define EIGEN_CATCH(X) else +#define EIGEN_THROW_X(X) std::abort() +#define EIGEN_THROW std::abort() +#endif +#define EIGEN_TRY if (true) +#define EIGEN_CATCH(X) else #endif #if EIGEN_HAS_CXX11_NOEXCEPT -# define EIGEN_INCLUDE_TYPE_TRAITS -# define EIGEN_NOEXCEPT noexcept -# define EIGEN_NOEXCEPT_IF(x) noexcept(x) -# define EIGEN_NO_THROW noexcept(true) -# define EIGEN_EXCEPTION_SPEC(X) noexcept(false) -#else -# define EIGEN_NOEXCEPT -# define EIGEN_NOEXCEPT_IF(x) -# define EIGEN_NO_THROW throw() -# if EIGEN_COMP_MSVC - // MSVC does not support exception specifications (warning C4290), - // and they are deprecated in c++11 anyway. -# define EIGEN_EXCEPTION_SPEC(X) throw() -# else -# define EIGEN_EXCEPTION_SPEC(X) throw(X) -# endif -#endif - -#endif // EIGEN_MACROS_H +#define EIGEN_INCLUDE_TYPE_TRAITS +#define EIGEN_NOEXCEPT noexcept +#define EIGEN_NOEXCEPT_IF(x) noexcept(x) +#define EIGEN_NO_THROW noexcept(true) +#define EIGEN_EXCEPTION_SPEC(X) noexcept(false) +#else +#define EIGEN_NOEXCEPT +#define EIGEN_NOEXCEPT_IF(x) +#define EIGEN_NO_THROW throw() +#if EIGEN_COMP_MSVC + // MSVC does not support exception specifications (warning C4290), + // and they are deprecated in c++11 anyway. +#define EIGEN_EXCEPTION_SPEC(X) throw() +#else +#define EIGEN_EXCEPTION_SPEC(X) throw(X) +#endif +#endif + +#endif// EIGEN_MACROS_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/util/Memory.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/util/Memory.h index 291383c5..8ee60eb8 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/util/Memory.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/util/Memory.h @@ -31,11 +31,11 @@ // http://gcc.fyxm.net/summit/2003/Porting%20to%2064%20bit.pdf // page 114, "[The] LP64 model [...] is used by all 64-bit UNIX ports" so it's indeed // quite safe, at least within the context of glibc, to equate 64-bit with LP64. -#if defined(__GLIBC__) && ((__GLIBC__>=2 && __GLIBC_MINOR__ >= 8) || __GLIBC__>2) \ - && defined(__LP64__) && ! defined( __SANITIZE_ADDRESS__ ) && (EIGEN_DEFAULT_ALIGN_BYTES == 16) - #define EIGEN_GLIBC_MALLOC_ALREADY_ALIGNED 1 +#if defined(__GLIBC__) && ((__GLIBC__ >= 2 && __GLIBC_MINOR__ >= 8) || __GLIBC__ > 2) && defined(__LP64__) \ + && !defined(__SANITIZE_ADDRESS__) && (EIGEN_DEFAULT_ALIGN_BYTES == 16) +#define EIGEN_GLIBC_MALLOC_ALREADY_ALIGNED 1 #else - #define EIGEN_GLIBC_MALLOC_ALREADY_ALIGNED 0 +#define EIGEN_GLIBC_MALLOC_ALREADY_ALIGNED 0 #endif // FreeBSD 6 seems to have 16-byte aligned malloc @@ -43,18 +43,16 @@ // FreeBSD 7 seems to have 16-byte aligned malloc except on ARM and MIPS architectures // See http://svn.freebsd.org/viewvc/base/stable/7/lib/libc/stdlib/malloc.c?view=markup #if defined(__FreeBSD__) && !(EIGEN_ARCH_ARM || EIGEN_ARCH_MIPS) && (EIGEN_DEFAULT_ALIGN_BYTES == 16) - #define EIGEN_FREEBSD_MALLOC_ALREADY_ALIGNED 1 +#define EIGEN_FREEBSD_MALLOC_ALREADY_ALIGNED 1 #else - #define EIGEN_FREEBSD_MALLOC_ALREADY_ALIGNED 0 +#define EIGEN_FREEBSD_MALLOC_ALREADY_ALIGNED 0 #endif -#if (EIGEN_OS_MAC && (EIGEN_DEFAULT_ALIGN_BYTES == 16)) \ - || (EIGEN_OS_WIN64 && (EIGEN_DEFAULT_ALIGN_BYTES == 16)) \ - || EIGEN_GLIBC_MALLOC_ALREADY_ALIGNED \ - || EIGEN_FREEBSD_MALLOC_ALREADY_ALIGNED - #define EIGEN_MALLOC_ALREADY_ALIGNED 1 +#if (EIGEN_OS_MAC && (EIGEN_DEFAULT_ALIGN_BYTES == 16)) || (EIGEN_OS_WIN64 && (EIGEN_DEFAULT_ALIGN_BYTES == 16)) \ + || EIGEN_GLIBC_MALLOC_ALREADY_ALIGNED || EIGEN_FREEBSD_MALLOC_ALREADY_ALIGNED +#define EIGEN_MALLOC_ALREADY_ALIGNED 1 #else - #define EIGEN_MALLOC_ALREADY_ALIGNED 0 +#define EIGEN_MALLOC_ALREADY_ALIGNED 0 #endif #endif @@ -63,477 +61,452 @@ namespace Eigen { namespace internal { -EIGEN_DEVICE_FUNC -inline void throw_std_bad_alloc() -{ - #ifdef EIGEN_EXCEPTIONS + EIGEN_DEVICE_FUNC + inline void throw_std_bad_alloc() + { +#ifdef EIGEN_EXCEPTIONS throw std::bad_alloc(); - #else +#else std::size_t huge = static_cast(-1); ::operator new(huge); - #endif -} +#endif + } -/***************************************************************************** -*** Implementation of handmade aligned functions *** -*****************************************************************************/ + /***************************************************************************** + *** Implementation of handmade aligned functions *** + *****************************************************************************/ -/* ----- Hand made implementations of aligned malloc/free and realloc ----- */ + /* ----- Hand made implementations of aligned malloc/free and realloc ----- */ -/** \internal Like malloc, but the returned pointer is guaranteed to be 16-byte aligned. - * Fast, but wastes 16 additional bytes of memory. Does not throw any exception. - */ -inline void* handmade_aligned_malloc(std::size_t size) -{ - void *original = std::malloc(size+EIGEN_DEFAULT_ALIGN_BYTES); - if (original == 0) return 0; - void *aligned = reinterpret_cast((reinterpret_cast(original) & ~(std::size_t(EIGEN_DEFAULT_ALIGN_BYTES-1))) + EIGEN_DEFAULT_ALIGN_BYTES); - *(reinterpret_cast(aligned) - 1) = original; - return aligned; -} - -/** \internal Frees memory allocated with handmade_aligned_malloc */ -inline void handmade_aligned_free(void *ptr) -{ - if (ptr) std::free(*(reinterpret_cast(ptr) - 1)); -} + /** \internal Like malloc, but the returned pointer is guaranteed to be 16-byte aligned. + * Fast, but wastes 16 additional bytes of memory. Does not throw any exception. + */ + inline void *handmade_aligned_malloc(std::size_t size) + { + void *original = std::malloc(size + EIGEN_DEFAULT_ALIGN_BYTES); + if (original == 0) return 0; + void *aligned = + reinterpret_cast((reinterpret_cast(original) & ~(std::size_t(EIGEN_DEFAULT_ALIGN_BYTES - 1))) + + EIGEN_DEFAULT_ALIGN_BYTES); + *(reinterpret_cast(aligned) - 1) = original; + return aligned; + } -/** \internal - * \brief Reallocates aligned memory. - * Since we know that our handmade version is based on std::malloc - * we can use std::realloc to implement efficient reallocation. - */ -inline void* handmade_aligned_realloc(void* ptr, std::size_t size, std::size_t = 0) -{ - if (ptr == 0) return handmade_aligned_malloc(size); - void *original = *(reinterpret_cast(ptr) - 1); - std::ptrdiff_t previous_offset = static_cast(ptr)-static_cast(original); - original = std::realloc(original,size+EIGEN_DEFAULT_ALIGN_BYTES); - if (original == 0) return 0; - void *aligned = reinterpret_cast((reinterpret_cast(original) & ~(std::size_t(EIGEN_DEFAULT_ALIGN_BYTES-1))) + EIGEN_DEFAULT_ALIGN_BYTES); - void *previous_aligned = static_cast(original)+previous_offset; - if(aligned!=previous_aligned) - std::memmove(aligned, previous_aligned, size); - - *(reinterpret_cast(aligned) - 1) = original; - return aligned; -} + /** \internal Frees memory allocated with handmade_aligned_malloc */ + inline void handmade_aligned_free(void *ptr) + { + if (ptr) std::free(*(reinterpret_cast(ptr) - 1)); + } -/***************************************************************************** -*** Implementation of portable aligned versions of malloc/free/realloc *** -*****************************************************************************/ + /** \internal + * \brief Reallocates aligned memory. + * Since we know that our handmade version is based on std::malloc + * we can use std::realloc to implement efficient reallocation. + */ + inline void *handmade_aligned_realloc(void *ptr, std::size_t size, std::size_t = 0) + { + if (ptr == 0) return handmade_aligned_malloc(size); + void *original = *(reinterpret_cast(ptr) - 1); + std::ptrdiff_t previous_offset = static_cast(ptr) - static_cast(original); + original = std::realloc(original, size + EIGEN_DEFAULT_ALIGN_BYTES); + if (original == 0) return 0; + void *aligned = + reinterpret_cast((reinterpret_cast(original) & ~(std::size_t(EIGEN_DEFAULT_ALIGN_BYTES - 1))) + + EIGEN_DEFAULT_ALIGN_BYTES); + void *previous_aligned = static_cast(original) + previous_offset; + if (aligned != previous_aligned) std::memmove(aligned, previous_aligned, size); + + *(reinterpret_cast(aligned) - 1) = original; + return aligned; + } + + /***************************************************************************** + *** Implementation of portable aligned versions of malloc/free/realloc *** + *****************************************************************************/ #ifdef EIGEN_NO_MALLOC -EIGEN_DEVICE_FUNC inline void check_that_malloc_is_allowed() -{ - eigen_assert(false && "heap allocation is forbidden (EIGEN_NO_MALLOC is defined)"); -} + EIGEN_DEVICE_FUNC inline void check_that_malloc_is_allowed() + { + eigen_assert(false && "heap allocation is forbidden (EIGEN_NO_MALLOC is defined)"); + } #elif defined EIGEN_RUNTIME_NO_MALLOC -EIGEN_DEVICE_FUNC inline bool is_malloc_allowed_impl(bool update, bool new_value = false) -{ - static bool value = true; - if (update == 1) - value = new_value; - return value; -} -EIGEN_DEVICE_FUNC inline bool is_malloc_allowed() { return is_malloc_allowed_impl(false); } -EIGEN_DEVICE_FUNC inline bool set_is_malloc_allowed(bool new_value) { return is_malloc_allowed_impl(true, new_value); } -EIGEN_DEVICE_FUNC inline void check_that_malloc_is_allowed() -{ - eigen_assert(is_malloc_allowed() && "heap allocation is forbidden (EIGEN_RUNTIME_NO_MALLOC is defined and g_is_malloc_allowed is false)"); -} -#else -EIGEN_DEVICE_FUNC inline void check_that_malloc_is_allowed() -{} + EIGEN_DEVICE_FUNC inline bool is_malloc_allowed_impl(bool update, bool new_value = false) + { + static bool value = true; + if (update == 1) value = new_value; + return value; + } + EIGEN_DEVICE_FUNC inline bool is_malloc_allowed() { return is_malloc_allowed_impl(false); } + EIGEN_DEVICE_FUNC inline bool set_is_malloc_allowed(bool new_value) + { + return is_malloc_allowed_impl(true, new_value); + } + EIGEN_DEVICE_FUNC inline void check_that_malloc_is_allowed() + { + eigen_assert( + is_malloc_allowed() + && "heap allocation is forbidden (EIGEN_RUNTIME_NO_MALLOC is defined and g_is_malloc_allowed is false)"); + } +#else + EIGEN_DEVICE_FUNC inline void check_that_malloc_is_allowed() {} #endif -/** \internal Allocates \a size bytes. The returned pointer is guaranteed to have 16 or 32 bytes alignment depending on the requirements. - * On allocation error, the returned pointer is null, and std::bad_alloc is thrown. - */ -EIGEN_DEVICE_FUNC inline void* aligned_malloc(std::size_t size) -{ - check_that_malloc_is_allowed(); + /** \internal Allocates \a size bytes. The returned pointer is guaranteed to have 16 or 32 bytes alignment depending + * on the requirements. On allocation error, the returned pointer is null, and std::bad_alloc is thrown. + */ + EIGEN_DEVICE_FUNC inline void *aligned_malloc(std::size_t size) + { + check_that_malloc_is_allowed(); - void *result; - #if (EIGEN_DEFAULT_ALIGN_BYTES==0) || EIGEN_MALLOC_ALREADY_ALIGNED + void *result; +#if (EIGEN_DEFAULT_ALIGN_BYTES == 0) || EIGEN_MALLOC_ALREADY_ALIGNED result = std::malloc(size); - #if EIGEN_DEFAULT_ALIGN_BYTES==16 +#if EIGEN_DEFAULT_ALIGN_BYTES == 16 eigen_assert((size<16 || (std::size_t(result)%16)==0) && "System's malloc returned an unaligned pointer. Compile with EIGEN_MALLOC_ALREADY_ALIGNED=0 to fallback to handmade alignd memory allocator."); - #endif - #else +#endif +#else result = handmade_aligned_malloc(size); - #endif +#endif - if(!result && size) - throw_std_bad_alloc(); + if (!result && size) throw_std_bad_alloc(); - return result; -} + return result; + } -/** \internal Frees memory allocated with aligned_malloc. */ -EIGEN_DEVICE_FUNC inline void aligned_free(void *ptr) -{ - #if (EIGEN_DEFAULT_ALIGN_BYTES==0) || EIGEN_MALLOC_ALREADY_ALIGNED + /** \internal Frees memory allocated with aligned_malloc. */ + EIGEN_DEVICE_FUNC inline void aligned_free(void *ptr) + { +#if (EIGEN_DEFAULT_ALIGN_BYTES == 0) || EIGEN_MALLOC_ALREADY_ALIGNED std::free(ptr); - #else - handmade_aligned_free(ptr); - #endif -} - -/** - * \internal - * \brief Reallocates an aligned block of memory. - * \throws std::bad_alloc on allocation failure - */ -inline void* aligned_realloc(void *ptr, std::size_t new_size, std::size_t old_size) -{ - EIGEN_UNUSED_VARIABLE(old_size); - - void *result; -#if (EIGEN_DEFAULT_ALIGN_BYTES==0) || EIGEN_MALLOC_ALREADY_ALIGNED - result = std::realloc(ptr,new_size); #else - result = handmade_aligned_realloc(ptr,new_size,old_size); + handmade_aligned_free(ptr); #endif + } - if (!result && new_size) - throw_std_bad_alloc(); - - return result; -} - -/***************************************************************************** -*** Implementation of conditionally aligned functions *** -*****************************************************************************/ - -/** \internal Allocates \a size bytes. If Align is true, then the returned ptr is 16-byte-aligned. - * On allocation error, the returned pointer is null, and a std::bad_alloc is thrown. - */ -template EIGEN_DEVICE_FUNC inline void* conditional_aligned_malloc(std::size_t size) -{ - return aligned_malloc(size); -} - -template<> EIGEN_DEVICE_FUNC inline void* conditional_aligned_malloc(std::size_t size) -{ - check_that_malloc_is_allowed(); - - void *result = std::malloc(size); - if(!result && size) - throw_std_bad_alloc(); - return result; -} - -/** \internal Frees memory allocated with conditional_aligned_malloc */ -template EIGEN_DEVICE_FUNC inline void conditional_aligned_free(void *ptr) -{ - aligned_free(ptr); -} + /** + * \internal + * \brief Reallocates an aligned block of memory. + * \throws std::bad_alloc on allocation failure + */ + inline void *aligned_realloc(void *ptr, std::size_t new_size, std::size_t old_size) + { + EIGEN_UNUSED_VARIABLE(old_size); -template<> EIGEN_DEVICE_FUNC inline void conditional_aligned_free(void *ptr) -{ - std::free(ptr); -} + void *result; +#if (EIGEN_DEFAULT_ALIGN_BYTES == 0) || EIGEN_MALLOC_ALREADY_ALIGNED + result = std::realloc(ptr, new_size); +#else + result = handmade_aligned_realloc(ptr, new_size, old_size); +#endif -template inline void* conditional_aligned_realloc(void* ptr, std::size_t new_size, std::size_t old_size) -{ - return aligned_realloc(ptr, new_size, old_size); -} + if (!result && new_size) throw_std_bad_alloc(); -template<> inline void* conditional_aligned_realloc(void* ptr, std::size_t new_size, std::size_t) -{ - return std::realloc(ptr, new_size); -} + return result; + } -/***************************************************************************** -*** Construction/destruction of array elements *** -*****************************************************************************/ + /***************************************************************************** + *** Implementation of conditionally aligned functions *** + *****************************************************************************/ -/** \internal Destructs the elements of an array. - * The \a size parameters tells on how many objects to call the destructor of T. - */ -template EIGEN_DEVICE_FUNC inline void destruct_elements_of_array(T *ptr, std::size_t size) -{ - // always destruct an array starting from the end. - if(ptr) - while(size) ptr[--size].~T(); -} - -/** \internal Constructs the elements of an array. - * The \a size parameter tells on how many objects to call the constructor of T. - */ -template EIGEN_DEVICE_FUNC inline T* construct_elements_of_array(T *ptr, std::size_t size) -{ - std::size_t i; - EIGEN_TRY + /** \internal Allocates \a size bytes. If Align is true, then the returned ptr is 16-byte-aligned. + * On allocation error, the returned pointer is null, and a std::bad_alloc is thrown. + */ + template EIGEN_DEVICE_FUNC inline void *conditional_aligned_malloc(std::size_t size) { - for (i = 0; i < size; ++i) ::new (ptr + i) T; - return ptr; + return aligned_malloc(size); } - EIGEN_CATCH(...) + + template<> EIGEN_DEVICE_FUNC inline void *conditional_aligned_malloc(std::size_t size) { - destruct_elements_of_array(ptr, i); - EIGEN_THROW; + check_that_malloc_is_allowed(); + + void *result = std::malloc(size); + if (!result && size) throw_std_bad_alloc(); + return result; } - return NULL; -} -/***************************************************************************** -*** Implementation of aligned new/delete-like functions *** -*****************************************************************************/ + /** \internal Frees memory allocated with conditional_aligned_malloc */ + template EIGEN_DEVICE_FUNC inline void conditional_aligned_free(void *ptr) { aligned_free(ptr); } -template -EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE void check_size_for_overflow(std::size_t size) -{ - if(size > std::size_t(-1) / sizeof(T)) - throw_std_bad_alloc(); -} - -/** \internal Allocates \a size objects of type T. The returned pointer is guaranteed to have 16 bytes alignment. - * On allocation error, the returned pointer is undefined, but a std::bad_alloc is thrown. - * The default constructor of T is called. - */ -template EIGEN_DEVICE_FUNC inline T* aligned_new(std::size_t size) -{ - check_size_for_overflow(size); - T *result = reinterpret_cast(aligned_malloc(sizeof(T)*size)); - EIGEN_TRY - { - return construct_elements_of_array(result, size); - } - EIGEN_CATCH(...) + template<> EIGEN_DEVICE_FUNC inline void conditional_aligned_free(void *ptr) { std::free(ptr); } + + template inline void *conditional_aligned_realloc(void *ptr, std::size_t new_size, std::size_t old_size) { - aligned_free(result); - EIGEN_THROW; + return aligned_realloc(ptr, new_size, old_size); } - return result; -} -template EIGEN_DEVICE_FUNC inline T* conditional_aligned_new(std::size_t size) -{ - check_size_for_overflow(size); - T *result = reinterpret_cast(conditional_aligned_malloc(sizeof(T)*size)); - EIGEN_TRY + template<> inline void *conditional_aligned_realloc(void *ptr, std::size_t new_size, std::size_t) { - return construct_elements_of_array(result, size); + return std::realloc(ptr, new_size); } - EIGEN_CATCH(...) + + /***************************************************************************** + *** Construction/destruction of array elements *** + *****************************************************************************/ + + /** \internal Destructs the elements of an array. + * The \a size parameters tells on how many objects to call the destructor of T. + */ + template EIGEN_DEVICE_FUNC inline void destruct_elements_of_array(T *ptr, std::size_t size) { - conditional_aligned_free(result); - EIGEN_THROW; + // always destruct an array starting from the end. + if (ptr) + while (size) ptr[--size].~T(); } - return result; -} -/** \internal Deletes objects constructed with aligned_new - * The \a size parameters tells on how many objects to call the destructor of T. - */ -template EIGEN_DEVICE_FUNC inline void aligned_delete(T *ptr, std::size_t size) -{ - destruct_elements_of_array(ptr, size); - aligned_free(ptr); -} - -/** \internal Deletes objects constructed with conditional_aligned_new - * The \a size parameters tells on how many objects to call the destructor of T. - */ -template EIGEN_DEVICE_FUNC inline void conditional_aligned_delete(T *ptr, std::size_t size) -{ - destruct_elements_of_array(ptr, size); - conditional_aligned_free(ptr); -} - -template EIGEN_DEVICE_FUNC inline T* conditional_aligned_realloc_new(T* pts, std::size_t new_size, std::size_t old_size) -{ - check_size_for_overflow(new_size); - check_size_for_overflow(old_size); - if(new_size < old_size) - destruct_elements_of_array(pts+new_size, old_size-new_size); - T *result = reinterpret_cast(conditional_aligned_realloc(reinterpret_cast(pts), sizeof(T)*new_size, sizeof(T)*old_size)); - if(new_size > old_size) + /** \internal Constructs the elements of an array. + * The \a size parameter tells on how many objects to call the constructor of T. + */ + template EIGEN_DEVICE_FUNC inline T *construct_elements_of_array(T *ptr, std::size_t size) { + std::size_t i; EIGEN_TRY { - construct_elements_of_array(result+old_size, new_size-old_size); + for (i = 0; i < size; ++i) ::new (ptr + i) T; + return ptr; } EIGEN_CATCH(...) { - conditional_aligned_free(result); + destruct_elements_of_array(ptr, i); EIGEN_THROW; } + return NULL; } - return result; -} + /***************************************************************************** + *** Implementation of aligned new/delete-like functions *** + *****************************************************************************/ -template EIGEN_DEVICE_FUNC inline T* conditional_aligned_new_auto(std::size_t size) -{ - if(size==0) - return 0; // short-cut. Also fixes Bug 884 - check_size_for_overflow(size); - T *result = reinterpret_cast(conditional_aligned_malloc(sizeof(T)*size)); - if(NumTraits::RequireInitialization) + template EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE void check_size_for_overflow(std::size_t size) { - EIGEN_TRY - { - construct_elements_of_array(result, size); - } + if (size > std::size_t(-1) / sizeof(T)) throw_std_bad_alloc(); + } + + /** \internal Allocates \a size objects of type T. The returned pointer is guaranteed to have 16 bytes alignment. + * On allocation error, the returned pointer is undefined, but a std::bad_alloc is thrown. + * The default constructor of T is called. + */ + template EIGEN_DEVICE_FUNC inline T *aligned_new(std::size_t size) + { + check_size_for_overflow(size); + T *result = reinterpret_cast(aligned_malloc(sizeof(T) * size)); + EIGEN_TRY { return construct_elements_of_array(result, size); } EIGEN_CATCH(...) { - conditional_aligned_free(result); + aligned_free(result); EIGEN_THROW; } + return result; } - return result; -} -template inline T* conditional_aligned_realloc_new_auto(T* pts, std::size_t new_size, std::size_t old_size) -{ - check_size_for_overflow(new_size); - check_size_for_overflow(old_size); - if(NumTraits::RequireInitialization && (new_size < old_size)) - destruct_elements_of_array(pts+new_size, old_size-new_size); - T *result = reinterpret_cast(conditional_aligned_realloc(reinterpret_cast(pts), sizeof(T)*new_size, sizeof(T)*old_size)); - if(NumTraits::RequireInitialization && (new_size > old_size)) + template EIGEN_DEVICE_FUNC inline T *conditional_aligned_new(std::size_t size) { - EIGEN_TRY - { - construct_elements_of_array(result+old_size, new_size-old_size); - } + check_size_for_overflow(size); + T *result = reinterpret_cast(conditional_aligned_malloc(sizeof(T) * size)); + EIGEN_TRY { return construct_elements_of_array(result, size); } EIGEN_CATCH(...) { conditional_aligned_free(result); EIGEN_THROW; } + return result; } - return result; -} -template EIGEN_DEVICE_FUNC inline void conditional_aligned_delete_auto(T *ptr, std::size_t size) -{ - if(NumTraits::RequireInitialization) + /** \internal Deletes objects constructed with aligned_new + * The \a size parameters tells on how many objects to call the destructor of T. + */ + template EIGEN_DEVICE_FUNC inline void aligned_delete(T *ptr, std::size_t size) + { destruct_elements_of_array(ptr, size); - conditional_aligned_free(ptr); -} + aligned_free(ptr); + } -/****************************************************************************/ + /** \internal Deletes objects constructed with conditional_aligned_new + * The \a size parameters tells on how many objects to call the destructor of T. + */ + template EIGEN_DEVICE_FUNC inline void conditional_aligned_delete(T *ptr, std::size_t size) + { + destruct_elements_of_array(ptr, size); + conditional_aligned_free(ptr); + } -/** \internal Returns the index of the first element of the array that is well aligned with respect to the requested \a Alignment. - * - * \tparam Alignment requested alignment in Bytes. - * \param array the address of the start of the array - * \param size the size of the array - * - * \note If no element of the array is well aligned or the requested alignment is not a multiple of a scalar, - * the size of the array is returned. For example with SSE, the requested alignment is typically 16-bytes. If - * packet size for the given scalar type is 1, then everything is considered well-aligned. - * - * \note Otherwise, if the Alignment is larger that the scalar size, we rely on the assumptions that sizeof(Scalar) is a - * power of 2. On the other hand, we do not assume that the array address is a multiple of sizeof(Scalar), as that fails for - * example with Scalar=double on certain 32-bit platforms, see bug #79. - * - * There is also the variant first_aligned(const MatrixBase&) defined in DenseCoeffsBase.h. - * \sa first_default_aligned() - */ -template -EIGEN_DEVICE_FUNC inline Index first_aligned(const Scalar* array, Index size) -{ - const Index ScalarSize = sizeof(Scalar); - const Index AlignmentSize = Alignment / ScalarSize; - const Index AlignmentMask = AlignmentSize-1; + template + EIGEN_DEVICE_FUNC inline T *conditional_aligned_realloc_new(T *pts, std::size_t new_size, std::size_t old_size) + { + check_size_for_overflow(new_size); + check_size_for_overflow(old_size); + if (new_size < old_size) destruct_elements_of_array(pts + new_size, old_size - new_size); + T *result = reinterpret_cast( + conditional_aligned_realloc(reinterpret_cast(pts), sizeof(T) * new_size, sizeof(T) * old_size)); + if (new_size > old_size) { + EIGEN_TRY { construct_elements_of_array(result + old_size, new_size - old_size); } + EIGEN_CATCH(...) + { + conditional_aligned_free(result); + EIGEN_THROW; + } + } + return result; + } - if(AlignmentSize<=1) + + template EIGEN_DEVICE_FUNC inline T *conditional_aligned_new_auto(std::size_t size) + { + if (size == 0) return 0;// short-cut. Also fixes Bug 884 + check_size_for_overflow(size); + T *result = reinterpret_cast(conditional_aligned_malloc(sizeof(T) * size)); + if (NumTraits::RequireInitialization) { + EIGEN_TRY { construct_elements_of_array(result, size); } + EIGEN_CATCH(...) + { + conditional_aligned_free(result); + EIGEN_THROW; + } + } + return result; + } + + template + inline T *conditional_aligned_realloc_new_auto(T *pts, std::size_t new_size, std::size_t old_size) { - // Either the requested alignment if smaller than a scalar, or it exactly match a 1 scalar - // so that all elements of the array have the same alignment. - return 0; + check_size_for_overflow(new_size); + check_size_for_overflow(old_size); + if (NumTraits::RequireInitialization && (new_size < old_size)) + destruct_elements_of_array(pts + new_size, old_size - new_size); + T *result = reinterpret_cast( + conditional_aligned_realloc(reinterpret_cast(pts), sizeof(T) * new_size, sizeof(T) * old_size)); + if (NumTraits::RequireInitialization && (new_size > old_size)) { + EIGEN_TRY { construct_elements_of_array(result + old_size, new_size - old_size); } + EIGEN_CATCH(...) + { + conditional_aligned_free(result); + EIGEN_THROW; + } + } + return result; } - else if( (UIntPtr(array) & (sizeof(Scalar)-1)) || (Alignment%ScalarSize)!=0) + + template + EIGEN_DEVICE_FUNC inline void conditional_aligned_delete_auto(T *ptr, std::size_t size) { - // The array is not aligned to the size of a single scalar, or the requested alignment is not a multiple of the scalar size. - // Consequently, no element of the array is well aligned. - return size; + if (NumTraits::RequireInitialization) destruct_elements_of_array(ptr, size); + conditional_aligned_free(ptr); } - else + + /****************************************************************************/ + + /** \internal Returns the index of the first element of the array that is well aligned with respect to the requested + * \a Alignment. + * + * \tparam Alignment requested alignment in Bytes. + * \param array the address of the start of the array + * \param size the size of the array + * + * \note If no element of the array is well aligned or the requested alignment is not a multiple of a scalar, + * the size of the array is returned. For example with SSE, the requested alignment is typically 16-bytes. If + * packet size for the given scalar type is 1, then everything is considered well-aligned. + * + * \note Otherwise, if the Alignment is larger that the scalar size, we rely on the assumptions that sizeof(Scalar) is + * a power of 2. On the other hand, we do not assume that the array address is a multiple of sizeof(Scalar), as that + * fails for example with Scalar=double on certain 32-bit platforms, see bug #79. + * + * There is also the variant first_aligned(const MatrixBase&) defined in DenseCoeffsBase.h. + * \sa first_default_aligned() + */ + template + EIGEN_DEVICE_FUNC inline Index first_aligned(const Scalar *array, Index size) { - Index first = (AlignmentSize - (Index((UIntPtr(array)/sizeof(Scalar))) & AlignmentMask)) & AlignmentMask; - return (first < size) ? first : size; + const Index ScalarSize = sizeof(Scalar); + const Index AlignmentSize = Alignment / ScalarSize; + const Index AlignmentMask = AlignmentSize - 1; + + if (AlignmentSize <= 1) { + // Either the requested alignment if smaller than a scalar, or it exactly match a 1 scalar + // so that all elements of the array have the same alignment. + return 0; + } else if ((UIntPtr(array) & (sizeof(Scalar) - 1)) || (Alignment % ScalarSize) != 0) { + // The array is not aligned to the size of a single scalar, or the requested alignment is not a multiple of the + // scalar size. Consequently, no element of the array is well aligned. + return size; + } else { + Index first = (AlignmentSize - (Index((UIntPtr(array) / sizeof(Scalar))) & AlignmentMask)) & AlignmentMask; + return (first < size) ? first : size; + } } -} -/** \internal Returns the index of the first element of the array that is well aligned with respect the largest packet requirement. + /** \internal Returns the index of the first element of the array that is well aligned with respect the largest packet + * requirement. * \sa first_aligned(Scalar*,Index) and first_default_aligned(DenseBase) */ -template -EIGEN_DEVICE_FUNC inline Index first_default_aligned(const Scalar* array, Index size) -{ - typedef typename packet_traits::type DefaultPacketType; - return first_aligned::alignment>(array, size); -} - -/** \internal Returns the smallest integer multiple of \a base and greater or equal to \a size - */ -template -inline Index first_multiple(Index size, Index base) -{ - return ((size+base-1)/base)*base; -} + template + EIGEN_DEVICE_FUNC inline Index first_default_aligned(const Scalar *array, Index size) + { + typedef typename packet_traits::type DefaultPacketType; + return first_aligned::alignment>(array, size); + } -// std::copy is much slower than memcpy, so let's introduce a smart_copy which -// use memcpy on trivial types, i.e., on types that does not require an initialization ctor. -template struct smart_copy_helper; + /** \internal Returns the smallest integer multiple of \a base and greater or equal to \a size + */ + template inline Index first_multiple(Index size, Index base) + { + return ((size + base - 1) / base) * base; + } -template EIGEN_DEVICE_FUNC void smart_copy(const T* start, const T* end, T* target) -{ - smart_copy_helper::RequireInitialization>::run(start, end, target); -} + // std::copy is much slower than memcpy, so let's introduce a smart_copy which + // use memcpy on trivial types, i.e., on types that does not require an initialization ctor. + template struct smart_copy_helper; -template struct smart_copy_helper { - EIGEN_DEVICE_FUNC static inline void run(const T* start, const T* end, T* target) + template EIGEN_DEVICE_FUNC void smart_copy(const T *start, const T *end, T *target) { - IntPtr size = IntPtr(end)-IntPtr(start); - if(size==0) return; - eigen_internal_assert(start!=0 && end!=0 && target!=0); - std::memcpy(target, start, size); + smart_copy_helper::RequireInitialization>::run(start, end, target); } -}; -template struct smart_copy_helper { - EIGEN_DEVICE_FUNC static inline void run(const T* start, const T* end, T* target) - { std::copy(start, end, target); } -}; + template struct smart_copy_helper + { + EIGEN_DEVICE_FUNC static inline void run(const T *start, const T *end, T *target) + { + IntPtr size = IntPtr(end) - IntPtr(start); + if (size == 0) return; + eigen_internal_assert(start != 0 && end != 0 && target != 0); + std::memcpy(target, start, size); + } + }; -// intelligent memmove. falls back to std::memmove for POD types, uses std::copy otherwise. -template struct smart_memmove_helper; + template struct smart_copy_helper + { + EIGEN_DEVICE_FUNC static inline void run(const T *start, const T *end, T *target) { std::copy(start, end, target); } + }; -template void smart_memmove(const T* start, const T* end, T* target) -{ - smart_memmove_helper::RequireInitialization>::run(start, end, target); -} + // intelligent memmove. falls back to std::memmove for POD types, uses std::copy otherwise. + template struct smart_memmove_helper; -template struct smart_memmove_helper { - static inline void run(const T* start, const T* end, T* target) + template void smart_memmove(const T *start, const T *end, T *target) { - IntPtr size = IntPtr(end)-IntPtr(start); - if(size==0) return; - eigen_internal_assert(start!=0 && end!=0 && target!=0); - std::memmove(target, start, size); + smart_memmove_helper::RequireInitialization>::run(start, end, target); } -}; -template struct smart_memmove_helper { - static inline void run(const T* start, const T* end, T* target) - { - if (UIntPtr(target) < UIntPtr(start)) + template struct smart_memmove_helper + { + static inline void run(const T *start, const T *end, T *target) { - std::copy(start, end, target); + IntPtr size = IntPtr(end) - IntPtr(start); + if (size == 0) return; + eigen_internal_assert(start != 0 && end != 0 && target != 0); + std::memmove(target, start, size); } - else + }; + + template struct smart_memmove_helper + { + static inline void run(const T *start, const T *end, T *target) { - std::ptrdiff_t count = (std::ptrdiff_t(end)-std::ptrdiff_t(start)) / sizeof(T); - std::copy_backward(start, end, target + count); + if (UIntPtr(target) < UIntPtr(start)) { + std::copy(start, end, target); + } else { + std::ptrdiff_t count = (std::ptrdiff_t(end) - std::ptrdiff_t(start)) / sizeof(T); + std::copy_backward(start, end, target + count); + } } - } -}; + }; /***************************************************************************** @@ -543,109 +516,105 @@ template struct smart_memmove_helper { // you can overwrite Eigen's default behavior regarding alloca by defining EIGEN_ALLOCA // to the appropriate stack allocation function #ifndef EIGEN_ALLOCA - #if EIGEN_OS_LINUX || EIGEN_OS_MAC || (defined alloca) - #define EIGEN_ALLOCA alloca - #elif EIGEN_COMP_MSVC - #define EIGEN_ALLOCA _alloca - #endif +#if EIGEN_OS_LINUX || EIGEN_OS_MAC || (defined alloca) +#define EIGEN_ALLOCA alloca +#elif EIGEN_COMP_MSVC +#define EIGEN_ALLOCA _alloca +#endif #endif -// This helper class construct the allocated memory, and takes care of destructing and freeing the handled data -// at destruction time. In practice this helper class is mainly useful to avoid memory leak in case of exceptions. -template class aligned_stack_memory_handler : noncopyable -{ + // This helper class construct the allocated memory, and takes care of destructing and freeing the handled data + // at destruction time. In practice this helper class is mainly useful to avoid memory leak in case of exceptions. + template class aligned_stack_memory_handler : noncopyable + { public: /* Creates a stack_memory_handler responsible for the buffer \a ptr of size \a size. * Note that \a ptr can be 0 regardless of the other parameters. - * This constructor takes care of constructing/initializing the elements of the buffer if required by the scalar type T (see NumTraits::RequireInitialization). - * In this case, the buffer elements will also be destructed when this handler will be destructed. - * Finally, if \a dealloc is true, then the pointer \a ptr is freed. + * This constructor takes care of constructing/initializing the elements of the buffer if required by the scalar + * type T (see NumTraits::RequireInitialization). In this case, the buffer elements will also be destructed when + * this handler will be destructed. Finally, if \a dealloc is true, then the pointer \a ptr is freed. **/ - aligned_stack_memory_handler(T* ptr, std::size_t size, bool dealloc) + aligned_stack_memory_handler(T *ptr, std::size_t size, bool dealloc) : m_ptr(ptr), m_size(size), m_deallocate(dealloc) { - if(NumTraits::RequireInitialization && m_ptr) - Eigen::internal::construct_elements_of_array(m_ptr, size); + if (NumTraits::RequireInitialization && m_ptr) Eigen::internal::construct_elements_of_array(m_ptr, size); } ~aligned_stack_memory_handler() { - if(NumTraits::RequireInitialization && m_ptr) - Eigen::internal::destruct_elements_of_array(m_ptr, m_size); - if(m_deallocate) - Eigen::internal::aligned_free(m_ptr); + if (NumTraits::RequireInitialization && m_ptr) Eigen::internal::destruct_elements_of_array(m_ptr, m_size); + if (m_deallocate) Eigen::internal::aligned_free(m_ptr); } + protected: - T* m_ptr; + T *m_ptr; std::size_t m_size; bool m_deallocate; -}; + }; -template class scoped_array : noncopyable -{ - T* m_ptr; -public: - explicit scoped_array(std::ptrdiff_t size) - { - m_ptr = new T[size]; - } - ~scoped_array() + template class scoped_array : noncopyable { - delete[] m_ptr; - } - T& operator[](std::ptrdiff_t i) { return m_ptr[i]; } - const T& operator[](std::ptrdiff_t i) const { return m_ptr[i]; } - T* &ptr() { return m_ptr; } - const T* ptr() const { return m_ptr; } - operator const T*() const { return m_ptr; } -}; + T *m_ptr; -template void swap(scoped_array &a,scoped_array &b) -{ - std::swap(a.ptr(),b.ptr()); -} - -} // end namespace internal + public: + explicit scoped_array(std::ptrdiff_t size) { m_ptr = new T[size]; } + ~scoped_array() { delete[] m_ptr; } + T &operator[](std::ptrdiff_t i) { return m_ptr[i]; } + const T &operator[](std::ptrdiff_t i) const { return m_ptr[i]; } + T *&ptr() { return m_ptr; } + const T *ptr() const { return m_ptr; } + operator const T *() const { return m_ptr; } + }; + + template void swap(scoped_array &a, scoped_array &b) { std::swap(a.ptr(), b.ptr()); } + +}// end namespace internal /** \internal - * Declares, allocates and construct an aligned buffer named NAME of SIZE elements of type TYPE on the stack - * if SIZE is smaller than EIGEN_STACK_ALLOCATION_LIMIT, and if stack allocation is supported by the platform - * (currently, this is Linux and Visual Studio only). Otherwise the memory is allocated on the heap. - * The allocated buffer is automatically deleted when exiting the scope of this declaration. - * If BUFFER is non null, then the declared variable is simply an alias for BUFFER, and no allocation/deletion occurs. - * Here is an example: - * \code - * { - * ei_declare_aligned_stack_constructed_variable(float,data,size,0); - * // use data[0] to data[size-1] - * } - * \endcode - * The underlying stack allocation function can controlled with the EIGEN_ALLOCA preprocessor token. - */ + * Declares, allocates and construct an aligned buffer named NAME of SIZE elements of type TYPE on the stack + * if SIZE is smaller than EIGEN_STACK_ALLOCATION_LIMIT, and if stack allocation is supported by the platform + * (currently, this is Linux and Visual Studio only). Otherwise the memory is allocated on the heap. + * The allocated buffer is automatically deleted when exiting the scope of this declaration. + * If BUFFER is non null, then the declared variable is simply an alias for BUFFER, and no allocation/deletion occurs. + * Here is an example: + * \code + * { + * ei_declare_aligned_stack_constructed_variable(float,data,size,0); + * // use data[0] to data[size-1] + * } + * \endcode + * The underlying stack allocation function can controlled with the EIGEN_ALLOCA preprocessor token. + */ #ifdef EIGEN_ALLOCA - - #if EIGEN_DEFAULT_ALIGN_BYTES>0 - // We always manually re-align the result of EIGEN_ALLOCA. - // If alloca is already aligned, the compiler should be smart enough to optimize away the re-alignment. - #define EIGEN_ALIGNED_ALLOCA(SIZE) reinterpret_cast((internal::UIntPtr(EIGEN_ALLOCA(SIZE+EIGEN_DEFAULT_ALIGN_BYTES-1)) + EIGEN_DEFAULT_ALIGN_BYTES-1) & ~(std::size_t(EIGEN_DEFAULT_ALIGN_BYTES-1))) - #else - #define EIGEN_ALIGNED_ALLOCA(SIZE) EIGEN_ALLOCA(SIZE) - #endif - - #define ei_declare_aligned_stack_constructed_variable(TYPE,NAME,SIZE,BUFFER) \ - Eigen::internal::check_size_for_overflow(SIZE); \ - TYPE* NAME = (BUFFER)!=0 ? (BUFFER) \ - : reinterpret_cast( \ - (sizeof(TYPE)*SIZE<=EIGEN_STACK_ALLOCATION_LIMIT) ? EIGEN_ALIGNED_ALLOCA(sizeof(TYPE)*SIZE) \ - : Eigen::internal::aligned_malloc(sizeof(TYPE)*SIZE) ); \ - Eigen::internal::aligned_stack_memory_handler EIGEN_CAT(NAME,_stack_memory_destructor)((BUFFER)==0 ? NAME : 0,SIZE,sizeof(TYPE)*SIZE>EIGEN_STACK_ALLOCATION_LIMIT) +#if EIGEN_DEFAULT_ALIGN_BYTES > 0 + // We always manually re-align the result of EIGEN_ALLOCA. +// If alloca is already aligned, the compiler should be smart enough to optimize away the re-alignment. +#define EIGEN_ALIGNED_ALLOCA(SIZE) \ + reinterpret_cast( \ + (internal::UIntPtr(EIGEN_ALLOCA(SIZE + EIGEN_DEFAULT_ALIGN_BYTES - 1)) + EIGEN_DEFAULT_ALIGN_BYTES - 1) \ + & ~(std::size_t(EIGEN_DEFAULT_ALIGN_BYTES - 1))) #else +#define EIGEN_ALIGNED_ALLOCA(SIZE) EIGEN_ALLOCA(SIZE) +#endif + +#define ei_declare_aligned_stack_constructed_variable(TYPE, NAME, SIZE, BUFFER) \ + Eigen::internal::check_size_for_overflow(SIZE); \ + TYPE *NAME = (BUFFER) != 0 ? (BUFFER) \ + : reinterpret_cast((sizeof(TYPE) * SIZE <= EIGEN_STACK_ALLOCATION_LIMIT) \ + ? EIGEN_ALIGNED_ALLOCA(sizeof(TYPE) * SIZE) \ + : Eigen::internal::aligned_malloc(sizeof(TYPE) * SIZE)); \ + Eigen::internal::aligned_stack_memory_handler EIGEN_CAT(NAME, _stack_memory_destructor)( \ + (BUFFER) == 0 ? NAME : 0, SIZE, sizeof(TYPE) * SIZE > EIGEN_STACK_ALLOCATION_LIMIT) + +#else + +#define ei_declare_aligned_stack_constructed_variable(TYPE, NAME, SIZE, BUFFER) \ + Eigen::internal::check_size_for_overflow(SIZE); \ + TYPE *NAME = \ + (BUFFER) != 0 ? BUFFER : reinterpret_cast(Eigen::internal::aligned_malloc(sizeof(TYPE) * SIZE)); \ + Eigen::internal::aligned_stack_memory_handler EIGEN_CAT(NAME, _stack_memory_destructor)( \ + (BUFFER) == 0 ? NAME : 0, SIZE, true) - #define ei_declare_aligned_stack_constructed_variable(TYPE,NAME,SIZE,BUFFER) \ - Eigen::internal::check_size_for_overflow(SIZE); \ - TYPE* NAME = (BUFFER)!=0 ? BUFFER : reinterpret_cast(Eigen::internal::aligned_malloc(sizeof(TYPE)*SIZE)); \ - Eigen::internal::aligned_stack_memory_handler EIGEN_CAT(NAME,_stack_memory_destructor)((BUFFER)==0 ? NAME : 0,SIZE,true) - #endif @@ -653,341 +622,469 @@ template void swap(scoped_array &a,scoped_array &b) *** Implementation of EIGEN_MAKE_ALIGNED_OPERATOR_NEW [_IF] *** *****************************************************************************/ -#if EIGEN_MAX_ALIGN_BYTES!=0 - #define EIGEN_MAKE_ALIGNED_OPERATOR_NEW_NOTHROW(NeedsToAlign) \ - void* operator new(std::size_t size, const std::nothrow_t&) EIGEN_NO_THROW { \ - EIGEN_TRY { return Eigen::internal::conditional_aligned_malloc(size); } \ - EIGEN_CATCH (...) { return 0; } \ - } - #define EIGEN_MAKE_ALIGNED_OPERATOR_NEW_IF(NeedsToAlign) \ - void *operator new(std::size_t size) { \ - return Eigen::internal::conditional_aligned_malloc(size); \ - } \ - void *operator new[](std::size_t size) { \ - return Eigen::internal::conditional_aligned_malloc(size); \ - } \ - void operator delete(void * ptr) EIGEN_NO_THROW { Eigen::internal::conditional_aligned_free(ptr); } \ - void operator delete[](void * ptr) EIGEN_NO_THROW { Eigen::internal::conditional_aligned_free(ptr); } \ - void operator delete(void * ptr, std::size_t /* sz */) EIGEN_NO_THROW { Eigen::internal::conditional_aligned_free(ptr); } \ - void operator delete[](void * ptr, std::size_t /* sz */) EIGEN_NO_THROW { Eigen::internal::conditional_aligned_free(ptr); } \ - /* in-place new and delete. since (at least afaik) there is no actual */ \ - /* memory allocated we can safely let the default implementation handle */ \ - /* this particular case. */ \ - static void *operator new(std::size_t size, void *ptr) { return ::operator new(size,ptr); } \ - static void *operator new[](std::size_t size, void* ptr) { return ::operator new[](size,ptr); } \ - void operator delete(void * memory, void *ptr) EIGEN_NO_THROW { return ::operator delete(memory,ptr); } \ - void operator delete[](void * memory, void *ptr) EIGEN_NO_THROW { return ::operator delete[](memory,ptr); } \ - /* nothrow-new (returns zero instead of std::bad_alloc) */ \ - EIGEN_MAKE_ALIGNED_OPERATOR_NEW_NOTHROW(NeedsToAlign) \ - void operator delete(void *ptr, const std::nothrow_t&) EIGEN_NO_THROW { \ - Eigen::internal::conditional_aligned_free(ptr); \ - } \ - typedef void eigen_aligned_operator_new_marker_type; +#if EIGEN_MAX_ALIGN_BYTES != 0 +#define EIGEN_MAKE_ALIGNED_OPERATOR_NEW_NOTHROW(NeedsToAlign) \ + void *operator new(std::size_t size, const std::nothrow_t &) EIGEN_NO_THROW \ + { \ + EIGEN_TRY { return Eigen::internal::conditional_aligned_malloc(size); } \ + EIGEN_CATCH(...) { return 0; } \ + } +#define EIGEN_MAKE_ALIGNED_OPERATOR_NEW_IF(NeedsToAlign) \ + void *operator new(std::size_t size) { return Eigen::internal::conditional_aligned_malloc(size); } \ + void *operator new[](std::size_t size) { return Eigen::internal::conditional_aligned_malloc(size); } \ + void operator delete(void *ptr) EIGEN_NO_THROW { Eigen::internal::conditional_aligned_free(ptr); } \ + void operator delete[](void *ptr) EIGEN_NO_THROW { Eigen::internal::conditional_aligned_free(ptr); } \ + void operator delete(void *ptr, std::size_t /* sz */) EIGEN_NO_THROW \ + { \ + Eigen::internal::conditional_aligned_free(ptr); \ + } \ + void operator delete[](void *ptr, std::size_t /* sz */) EIGEN_NO_THROW \ + { \ + Eigen::internal::conditional_aligned_free(ptr); \ + } \ + /* in-place new and delete. since (at least afaik) there is no actual */ \ + /* memory allocated we can safely let the default implementation handle */ \ + /* this particular case. */ \ + static void *operator new(std::size_t size, void *ptr) { return ::operator new(size, ptr); } \ + static void *operator new[](std::size_t size, void *ptr) { return ::operator new[](size, ptr); } \ + void operator delete(void *memory, void *ptr) EIGEN_NO_THROW { return ::operator delete(memory, ptr); } \ + void operator delete[](void *memory, void *ptr) EIGEN_NO_THROW { return ::operator delete[](memory, ptr); } \ + /* nothrow-new (returns zero instead of std::bad_alloc) */ \ + EIGEN_MAKE_ALIGNED_OPERATOR_NEW_NOTHROW(NeedsToAlign) \ + void operator delete(void *ptr, const std::nothrow_t &) EIGEN_NO_THROW \ + { \ + Eigen::internal::conditional_aligned_free(ptr); \ + } \ + typedef void eigen_aligned_operator_new_marker_type; #else - #define EIGEN_MAKE_ALIGNED_OPERATOR_NEW_IF(NeedsToAlign) +#define EIGEN_MAKE_ALIGNED_OPERATOR_NEW_IF(NeedsToAlign) #endif #define EIGEN_MAKE_ALIGNED_OPERATOR_NEW EIGEN_MAKE_ALIGNED_OPERATOR_NEW_IF(true) -#define EIGEN_MAKE_ALIGNED_OPERATOR_NEW_IF_VECTORIZABLE_FIXED_SIZE(Scalar,Size) \ - EIGEN_MAKE_ALIGNED_OPERATOR_NEW_IF(bool(((Size)!=Eigen::Dynamic) && ((sizeof(Scalar)*(Size))%EIGEN_MAX_ALIGN_BYTES==0))) +#define EIGEN_MAKE_ALIGNED_OPERATOR_NEW_IF_VECTORIZABLE_FIXED_SIZE(Scalar, Size) \ + EIGEN_MAKE_ALIGNED_OPERATOR_NEW_IF( \ + bool(((Size) != Eigen::Dynamic) && ((sizeof(Scalar) * (Size)) % EIGEN_MAX_ALIGN_BYTES == 0))) /****************************************************************************/ /** \class aligned_allocator -* \ingroup Core_Module -* -* \brief STL compatible allocator to use with types requiring a non standrad alignment. -* -* The memory is aligned as for dynamically aligned matrix/array types such as MatrixXd. -* By default, it will thus provide at least 16 bytes alignment and more in following cases: -* - 32 bytes alignment if AVX is enabled. -* - 64 bytes alignment if AVX512 is enabled. -* -* This can be controled using the \c EIGEN_MAX_ALIGN_BYTES macro as documented -* \link TopicPreprocessorDirectivesPerformance there \endlink. -* -* Example: -* \code -* // Matrix4f requires 16 bytes alignment: -* std::map< int, Matrix4f, std::less, -* aligned_allocator > > my_map_mat4; -* // Vector3f does not require 16 bytes alignment, no need to use Eigen's allocator: -* std::map< int, Vector3f > my_map_vec3; -* \endcode -* -* \sa \blank \ref TopicStlContainers. -*/ -template -class aligned_allocator : public std::allocator + * \ingroup Core_Module + * + * \brief STL compatible allocator to use with types requiring a non standrad alignment. + * + * The memory is aligned as for dynamically aligned matrix/array types such as MatrixXd. + * By default, it will thus provide at least 16 bytes alignment and more in following cases: + * - 32 bytes alignment if AVX is enabled. + * - 64 bytes alignment if AVX512 is enabled. + * + * This can be controled using the \c EIGEN_MAX_ALIGN_BYTES macro as documented + * \link TopicPreprocessorDirectivesPerformance there \endlink. + * + * Example: + * \code + * // Matrix4f requires 16 bytes alignment: + * std::map< int, Matrix4f, std::less, + * aligned_allocator > > my_map_mat4; + * // Vector3f does not require 16 bytes alignment, no need to use Eigen's allocator: + * std::map< int, Vector3f > my_map_vec3; + * \endcode + * + * \sa \blank \ref TopicStlContainers. + */ +template class aligned_allocator : public std::allocator { public: - typedef std::size_t size_type; - typedef std::ptrdiff_t difference_type; - typedef T* pointer; - typedef const T* const_pointer; - typedef T& reference; - typedef const T& const_reference; - typedef T value_type; - - template - struct rebind + typedef std::size_t size_type; + typedef std::ptrdiff_t difference_type; + typedef T *pointer; + typedef const T *const_pointer; + typedef T &reference; + typedef const T &const_reference; + typedef T value_type; + + template struct rebind { typedef aligned_allocator other; }; aligned_allocator() : std::allocator() {} - aligned_allocator(const aligned_allocator& other) : std::allocator(other) {} + aligned_allocator(const aligned_allocator &other) : std::allocator(other) {} - template - aligned_allocator(const aligned_allocator& other) : std::allocator(other) {} + template aligned_allocator(const aligned_allocator &other) : std::allocator(other) {} ~aligned_allocator() {} - pointer allocate(size_type num, const void* /*hint*/ = 0) + pointer allocate(size_type num, const void * /*hint*/ = 0) { internal::check_size_for_overflow(num); size_type size = num * sizeof(T); -#if EIGEN_COMP_GNUC_STRICT && EIGEN_GNUC_AT_LEAST(7,0) +#if EIGEN_COMP_GNUC_STRICT && EIGEN_GNUC_AT_LEAST(7, 0) // workaround gcc bug https://gcc.gnu.org/bugzilla/show_bug.cgi?id=87544 - // It triggered eigen/Eigen/src/Core/util/Memory.h:189:12: warning: argument 1 value '18446744073709551612' exceeds maximum object size 9223372036854775807 - if(size>=std::size_t((std::numeric_limits::max)())) + // It triggered eigen/Eigen/src/Core/util/Memory.h:189:12: warning: argument 1 value '18446744073709551612' exceeds + // maximum object size 9223372036854775807 + if (size >= std::size_t((std::numeric_limits::max)())) return 0; else #endif - return static_cast( internal::aligned_malloc(size) ); + return static_cast(internal::aligned_malloc(size)); } - void deallocate(pointer p, size_type /*num*/) - { - internal::aligned_free(p); - } + void deallocate(pointer p, size_type /*num*/) { internal::aligned_free(p); } }; //---------- Cache sizes ---------- #if !defined(EIGEN_NO_CPUID) -# if EIGEN_COMP_GNUC && EIGEN_ARCH_i386_OR_x86_64 -# if defined(__PIC__) && EIGEN_ARCH_i386 - // Case for x86 with PIC -# define EIGEN_CPUID(abcd,func,id) \ - __asm__ __volatile__ ("xchgl %%ebx, %k1;cpuid; xchgl %%ebx,%k1": "=a" (abcd[0]), "=&r" (abcd[1]), "=c" (abcd[2]), "=d" (abcd[3]) : "a" (func), "c" (id)); -# elif defined(__PIC__) && EIGEN_ARCH_x86_64 - // Case for x64 with PIC. In theory this is only a problem with recent gcc and with medium or large code model, not with the default small code model. - // However, we cannot detect which code model is used, and the xchg overhead is negligible anyway. -# define EIGEN_CPUID(abcd,func,id) \ - __asm__ __volatile__ ("xchg{q}\t{%%}rbx, %q1; cpuid; xchg{q}\t{%%}rbx, %q1": "=a" (abcd[0]), "=&r" (abcd[1]), "=c" (abcd[2]), "=d" (abcd[3]) : "0" (func), "2" (id)); -# else - // Case for x86_64 or x86 w/o PIC -# define EIGEN_CPUID(abcd,func,id) \ - __asm__ __volatile__ ("cpuid": "=a" (abcd[0]), "=b" (abcd[1]), "=c" (abcd[2]), "=d" (abcd[3]) : "0" (func), "2" (id) ); -# endif -# elif EIGEN_COMP_MSVC -# if (EIGEN_COMP_MSVC > 1500) && EIGEN_ARCH_i386_OR_x86_64 -# define EIGEN_CPUID(abcd,func,id) __cpuidex((int*)abcd,func,id) -# endif -# endif +#if EIGEN_COMP_GNUC && EIGEN_ARCH_i386_OR_x86_64 +#if defined(__PIC__) && EIGEN_ARCH_i386 +// Case for x86 with PIC +#define EIGEN_CPUID(abcd, func, id) \ + __asm__ __volatile__("xchgl %%ebx, %k1;cpuid; xchgl %%ebx,%k1" \ + : "=a"(abcd[0]), "=&r"(abcd[1]), "=c"(abcd[2]), "=d"(abcd[3]) \ + : "a"(func), "c"(id)); +#elif defined(__PIC__) && EIGEN_ARCH_x86_64 +// Case for x64 with PIC. In theory this is only a problem with recent gcc and with medium or large code model, not with +// the default small code model. However, we cannot detect which code model is used, and the xchg overhead is negligible +// anyway. +#define EIGEN_CPUID(abcd, func, id) \ + __asm__ __volatile__("xchg{q}\t{%%}rbx, %q1; cpuid; xchg{q}\t{%%}rbx, %q1" \ + : "=a"(abcd[0]), "=&r"(abcd[1]), "=c"(abcd[2]), "=d"(abcd[3]) \ + : "0"(func), "2"(id)); +#else +// Case for x86_64 or x86 w/o PIC +#define EIGEN_CPUID(abcd, func, id) \ + __asm__ __volatile__("cpuid" : "=a"(abcd[0]), "=b"(abcd[1]), "=c"(abcd[2]), "=d"(abcd[3]) : "0"(func), "2"(id)); +#endif +#elif EIGEN_COMP_MSVC +#if (EIGEN_COMP_MSVC > 1500) && EIGEN_ARCH_i386_OR_x86_64 +#define EIGEN_CPUID(abcd, func, id) __cpuidex((int *)abcd, func, id) +#endif +#endif #endif namespace internal { #ifdef EIGEN_CPUID -inline bool cpuid_is_vendor(int abcd[4], const int vendor[3]) -{ - return abcd[1]==vendor[0] && abcd[3]==vendor[1] && abcd[2]==vendor[2]; -} - -inline void queryCacheSizes_intel_direct(int& l1, int& l2, int& l3) -{ - int abcd[4]; - l1 = l2 = l3 = 0; - int cache_id = 0; - int cache_type = 0; - do { - abcd[0] = abcd[1] = abcd[2] = abcd[3] = 0; - EIGEN_CPUID(abcd,0x4,cache_id); - cache_type = (abcd[0] & 0x0F) >> 0; - if(cache_type==1||cache_type==3) // data or unified cache - { - int cache_level = (abcd[0] & 0xE0) >> 5; // A[7:5] - int ways = (abcd[1] & 0xFFC00000) >> 22; // B[31:22] - int partitions = (abcd[1] & 0x003FF000) >> 12; // B[21:12] - int line_size = (abcd[1] & 0x00000FFF) >> 0; // B[11:0] - int sets = (abcd[2]); // C[31:0] - - int cache_size = (ways+1) * (partitions+1) * (line_size+1) * (sets+1); + inline bool cpuid_is_vendor(int abcd[4], const int vendor[3]) + { + return abcd[1] == vendor[0] && abcd[3] == vendor[1] && abcd[2] == vendor[2]; + } - switch(cache_level) + inline void queryCacheSizes_intel_direct(int &l1, int &l2, int &l3) + { + int abcd[4]; + l1 = l2 = l3 = 0; + int cache_id = 0; + int cache_type = 0; + do { + abcd[0] = abcd[1] = abcd[2] = abcd[3] = 0; + EIGEN_CPUID(abcd, 0x4, cache_id); + cache_type = (abcd[0] & 0x0F) >> 0; + if (cache_type == 1 || cache_type == 3)// data or unified cache { - case 1: l1 = cache_size; break; - case 2: l2 = cache_size; break; - case 3: l3 = cache_size; break; - default: break; + int cache_level = (abcd[0] & 0xE0) >> 5;// A[7:5] + int ways = (abcd[1] & 0xFFC00000) >> 22;// B[31:22] + int partitions = (abcd[1] & 0x003FF000) >> 12;// B[21:12] + int line_size = (abcd[1] & 0x00000FFF) >> 0;// B[11:0] + int sets = (abcd[2]);// C[31:0] + + int cache_size = (ways + 1) * (partitions + 1) * (line_size + 1) * (sets + 1); + + switch (cache_level) { + case 1: + l1 = cache_size; + break; + case 2: + l2 = cache_size; + break; + case 3: + l3 = cache_size; + break; + default: + break; + } } - } - cache_id++; - } while(cache_type>0 && cache_id<16); -} + cache_id++; + } while (cache_type > 0 && cache_id < 16); + } -inline void queryCacheSizes_intel_codes(int& l1, int& l2, int& l3) -{ - int abcd[4]; - abcd[0] = abcd[1] = abcd[2] = abcd[3] = 0; - l1 = l2 = l3 = 0; - EIGEN_CPUID(abcd,0x00000002,0); - unsigned char * bytes = reinterpret_cast(abcd)+2; - bool check_for_p2_core2 = false; - for(int i=0; i<14; ++i) - { - switch(bytes[i]) - { - case 0x0A: l1 = 8; break; // 0Ah data L1 cache, 8 KB, 2 ways, 32 byte lines - case 0x0C: l1 = 16; break; // 0Ch data L1 cache, 16 KB, 4 ways, 32 byte lines - case 0x0E: l1 = 24; break; // 0Eh data L1 cache, 24 KB, 6 ways, 64 byte lines - case 0x10: l1 = 16; break; // 10h data L1 cache, 16 KB, 4 ways, 32 byte lines (IA-64) - case 0x15: l1 = 16; break; // 15h code L1 cache, 16 KB, 4 ways, 32 byte lines (IA-64) - case 0x2C: l1 = 32; break; // 2Ch data L1 cache, 32 KB, 8 ways, 64 byte lines - case 0x30: l1 = 32; break; // 30h code L1 cache, 32 KB, 8 ways, 64 byte lines - case 0x60: l1 = 16; break; // 60h data L1 cache, 16 KB, 8 ways, 64 byte lines, sectored - case 0x66: l1 = 8; break; // 66h data L1 cache, 8 KB, 4 ways, 64 byte lines, sectored - case 0x67: l1 = 16; break; // 67h data L1 cache, 16 KB, 4 ways, 64 byte lines, sectored - case 0x68: l1 = 32; break; // 68h data L1 cache, 32 KB, 4 ways, 64 byte lines, sectored - case 0x1A: l2 = 96; break; // code and data L2 cache, 96 KB, 6 ways, 64 byte lines (IA-64) - case 0x22: l3 = 512; break; // code and data L3 cache, 512 KB, 4 ways (!), 64 byte lines, dual-sectored - case 0x23: l3 = 1024; break; // code and data L3 cache, 1024 KB, 8 ways, 64 byte lines, dual-sectored - case 0x25: l3 = 2048; break; // code and data L3 cache, 2048 KB, 8 ways, 64 byte lines, dual-sectored - case 0x29: l3 = 4096; break; // code and data L3 cache, 4096 KB, 8 ways, 64 byte lines, dual-sectored - case 0x39: l2 = 128; break; // code and data L2 cache, 128 KB, 4 ways, 64 byte lines, sectored - case 0x3A: l2 = 192; break; // code and data L2 cache, 192 KB, 6 ways, 64 byte lines, sectored - case 0x3B: l2 = 128; break; // code and data L2 cache, 128 KB, 2 ways, 64 byte lines, sectored - case 0x3C: l2 = 256; break; // code and data L2 cache, 256 KB, 4 ways, 64 byte lines, sectored - case 0x3D: l2 = 384; break; // code and data L2 cache, 384 KB, 6 ways, 64 byte lines, sectored - case 0x3E: l2 = 512; break; // code and data L2 cache, 512 KB, 4 ways, 64 byte lines, sectored - case 0x40: l2 = 0; break; // no integrated L2 cache (P6 core) or L3 cache (P4 core) - case 0x41: l2 = 128; break; // code and data L2 cache, 128 KB, 4 ways, 32 byte lines - case 0x42: l2 = 256; break; // code and data L2 cache, 256 KB, 4 ways, 32 byte lines - case 0x43: l2 = 512; break; // code and data L2 cache, 512 KB, 4 ways, 32 byte lines - case 0x44: l2 = 1024; break; // code and data L2 cache, 1024 KB, 4 ways, 32 byte lines - case 0x45: l2 = 2048; break; // code and data L2 cache, 2048 KB, 4 ways, 32 byte lines - case 0x46: l3 = 4096; break; // code and data L3 cache, 4096 KB, 4 ways, 64 byte lines - case 0x47: l3 = 8192; break; // code and data L3 cache, 8192 KB, 8 ways, 64 byte lines - case 0x48: l2 = 3072; break; // code and data L2 cache, 3072 KB, 12 ways, 64 byte lines - case 0x49: if(l2!=0) l3 = 4096; else {check_for_p2_core2=true; l3 = l2 = 4096;} break;// code and data L3 cache, 4096 KB, 16 ways, 64 byte lines (P4) or L2 for core2 - case 0x4A: l3 = 6144; break; // code and data L3 cache, 6144 KB, 12 ways, 64 byte lines - case 0x4B: l3 = 8192; break; // code and data L3 cache, 8192 KB, 16 ways, 64 byte lines - case 0x4C: l3 = 12288; break; // code and data L3 cache, 12288 KB, 12 ways, 64 byte lines - case 0x4D: l3 = 16384; break; // code and data L3 cache, 16384 KB, 16 ways, 64 byte lines - case 0x4E: l2 = 6144; break; // code and data L2 cache, 6144 KB, 24 ways, 64 byte lines - case 0x78: l2 = 1024; break; // code and data L2 cache, 1024 KB, 4 ways, 64 byte lines - case 0x79: l2 = 128; break; // code and data L2 cache, 128 KB, 8 ways, 64 byte lines, dual-sectored - case 0x7A: l2 = 256; break; // code and data L2 cache, 256 KB, 8 ways, 64 byte lines, dual-sectored - case 0x7B: l2 = 512; break; // code and data L2 cache, 512 KB, 8 ways, 64 byte lines, dual-sectored - case 0x7C: l2 = 1024; break; // code and data L2 cache, 1024 KB, 8 ways, 64 byte lines, dual-sectored - case 0x7D: l2 = 2048; break; // code and data L2 cache, 2048 KB, 8 ways, 64 byte lines - case 0x7E: l2 = 256; break; // code and data L2 cache, 256 KB, 8 ways, 128 byte lines, sect. (IA-64) - case 0x7F: l2 = 512; break; // code and data L2 cache, 512 KB, 2 ways, 64 byte lines - case 0x80: l2 = 512; break; // code and data L2 cache, 512 KB, 8 ways, 64 byte lines - case 0x81: l2 = 128; break; // code and data L2 cache, 128 KB, 8 ways, 32 byte lines - case 0x82: l2 = 256; break; // code and data L2 cache, 256 KB, 8 ways, 32 byte lines - case 0x83: l2 = 512; break; // code and data L2 cache, 512 KB, 8 ways, 32 byte lines - case 0x84: l2 = 1024; break; // code and data L2 cache, 1024 KB, 8 ways, 32 byte lines - case 0x85: l2 = 2048; break; // code and data L2 cache, 2048 KB, 8 ways, 32 byte lines - case 0x86: l2 = 512; break; // code and data L2 cache, 512 KB, 4 ways, 64 byte lines - case 0x87: l2 = 1024; break; // code and data L2 cache, 1024 KB, 8 ways, 64 byte lines - case 0x88: l3 = 2048; break; // code and data L3 cache, 2048 KB, 4 ways, 64 byte lines (IA-64) - case 0x89: l3 = 4096; break; // code and data L3 cache, 4096 KB, 4 ways, 64 byte lines (IA-64) - case 0x8A: l3 = 8192; break; // code and data L3 cache, 8192 KB, 4 ways, 64 byte lines (IA-64) - case 0x8D: l3 = 3072; break; // code and data L3 cache, 3072 KB, 12 ways, 128 byte lines (IA-64) - - default: break; + inline void queryCacheSizes_intel_codes(int &l1, int &l2, int &l3) + { + int abcd[4]; + abcd[0] = abcd[1] = abcd[2] = abcd[3] = 0; + l1 = l2 = l3 = 0; + EIGEN_CPUID(abcd, 0x00000002, 0); + unsigned char *bytes = reinterpret_cast(abcd) + 2; + bool check_for_p2_core2 = false; + for (int i = 0; i < 14; ++i) { + switch (bytes[i]) { + case 0x0A: + l1 = 8; + break;// 0Ah data L1 cache, 8 KB, 2 ways, 32 byte lines + case 0x0C: + l1 = 16; + break;// 0Ch data L1 cache, 16 KB, 4 ways, 32 byte lines + case 0x0E: + l1 = 24; + break;// 0Eh data L1 cache, 24 KB, 6 ways, 64 byte lines + case 0x10: + l1 = 16; + break;// 10h data L1 cache, 16 KB, 4 ways, 32 byte lines (IA-64) + case 0x15: + l1 = 16; + break;// 15h code L1 cache, 16 KB, 4 ways, 32 byte lines (IA-64) + case 0x2C: + l1 = 32; + break;// 2Ch data L1 cache, 32 KB, 8 ways, 64 byte lines + case 0x30: + l1 = 32; + break;// 30h code L1 cache, 32 KB, 8 ways, 64 byte lines + case 0x60: + l1 = 16; + break;// 60h data L1 cache, 16 KB, 8 ways, 64 byte lines, sectored + case 0x66: + l1 = 8; + break;// 66h data L1 cache, 8 KB, 4 ways, 64 byte lines, sectored + case 0x67: + l1 = 16; + break;// 67h data L1 cache, 16 KB, 4 ways, 64 byte lines, sectored + case 0x68: + l1 = 32; + break;// 68h data L1 cache, 32 KB, 4 ways, 64 byte lines, sectored + case 0x1A: + l2 = 96; + break;// code and data L2 cache, 96 KB, 6 ways, 64 byte lines (IA-64) + case 0x22: + l3 = 512; + break;// code and data L3 cache, 512 KB, 4 ways (!), 64 byte lines, dual-sectored + case 0x23: + l3 = 1024; + break;// code and data L3 cache, 1024 KB, 8 ways, 64 byte lines, dual-sectored + case 0x25: + l3 = 2048; + break;// code and data L3 cache, 2048 KB, 8 ways, 64 byte lines, dual-sectored + case 0x29: + l3 = 4096; + break;// code and data L3 cache, 4096 KB, 8 ways, 64 byte lines, dual-sectored + case 0x39: + l2 = 128; + break;// code and data L2 cache, 128 KB, 4 ways, 64 byte lines, sectored + case 0x3A: + l2 = 192; + break;// code and data L2 cache, 192 KB, 6 ways, 64 byte lines, sectored + case 0x3B: + l2 = 128; + break;// code and data L2 cache, 128 KB, 2 ways, 64 byte lines, sectored + case 0x3C: + l2 = 256; + break;// code and data L2 cache, 256 KB, 4 ways, 64 byte lines, sectored + case 0x3D: + l2 = 384; + break;// code and data L2 cache, 384 KB, 6 ways, 64 byte lines, sectored + case 0x3E: + l2 = 512; + break;// code and data L2 cache, 512 KB, 4 ways, 64 byte lines, sectored + case 0x40: + l2 = 0; + break;// no integrated L2 cache (P6 core) or L3 cache (P4 core) + case 0x41: + l2 = 128; + break;// code and data L2 cache, 128 KB, 4 ways, 32 byte lines + case 0x42: + l2 = 256; + break;// code and data L2 cache, 256 KB, 4 ways, 32 byte lines + case 0x43: + l2 = 512; + break;// code and data L2 cache, 512 KB, 4 ways, 32 byte lines + case 0x44: + l2 = 1024; + break;// code and data L2 cache, 1024 KB, 4 ways, 32 byte lines + case 0x45: + l2 = 2048; + break;// code and data L2 cache, 2048 KB, 4 ways, 32 byte lines + case 0x46: + l3 = 4096; + break;// code and data L3 cache, 4096 KB, 4 ways, 64 byte lines + case 0x47: + l3 = 8192; + break;// code and data L3 cache, 8192 KB, 8 ways, 64 byte lines + case 0x48: + l2 = 3072; + break;// code and data L2 cache, 3072 KB, 12 ways, 64 byte lines + case 0x49: + if (l2 != 0) + l3 = 4096; + else { + check_for_p2_core2 = true; + l3 = l2 = 4096; + } + break;// code and data L3 cache, 4096 KB, 16 ways, 64 byte lines (P4) or L2 for core2 + case 0x4A: + l3 = 6144; + break;// code and data L3 cache, 6144 KB, 12 ways, 64 byte lines + case 0x4B: + l3 = 8192; + break;// code and data L3 cache, 8192 KB, 16 ways, 64 byte lines + case 0x4C: + l3 = 12288; + break;// code and data L3 cache, 12288 KB, 12 ways, 64 byte lines + case 0x4D: + l3 = 16384; + break;// code and data L3 cache, 16384 KB, 16 ways, 64 byte lines + case 0x4E: + l2 = 6144; + break;// code and data L2 cache, 6144 KB, 24 ways, 64 byte lines + case 0x78: + l2 = 1024; + break;// code and data L2 cache, 1024 KB, 4 ways, 64 byte lines + case 0x79: + l2 = 128; + break;// code and data L2 cache, 128 KB, 8 ways, 64 byte lines, dual-sectored + case 0x7A: + l2 = 256; + break;// code and data L2 cache, 256 KB, 8 ways, 64 byte lines, dual-sectored + case 0x7B: + l2 = 512; + break;// code and data L2 cache, 512 KB, 8 ways, 64 byte lines, dual-sectored + case 0x7C: + l2 = 1024; + break;// code and data L2 cache, 1024 KB, 8 ways, 64 byte lines, dual-sectored + case 0x7D: + l2 = 2048; + break;// code and data L2 cache, 2048 KB, 8 ways, 64 byte lines + case 0x7E: + l2 = 256; + break;// code and data L2 cache, 256 KB, 8 ways, 128 byte lines, sect. (IA-64) + case 0x7F: + l2 = 512; + break;// code and data L2 cache, 512 KB, 2 ways, 64 byte lines + case 0x80: + l2 = 512; + break;// code and data L2 cache, 512 KB, 8 ways, 64 byte lines + case 0x81: + l2 = 128; + break;// code and data L2 cache, 128 KB, 8 ways, 32 byte lines + case 0x82: + l2 = 256; + break;// code and data L2 cache, 256 KB, 8 ways, 32 byte lines + case 0x83: + l2 = 512; + break;// code and data L2 cache, 512 KB, 8 ways, 32 byte lines + case 0x84: + l2 = 1024; + break;// code and data L2 cache, 1024 KB, 8 ways, 32 byte lines + case 0x85: + l2 = 2048; + break;// code and data L2 cache, 2048 KB, 8 ways, 32 byte lines + case 0x86: + l2 = 512; + break;// code and data L2 cache, 512 KB, 4 ways, 64 byte lines + case 0x87: + l2 = 1024; + break;// code and data L2 cache, 1024 KB, 8 ways, 64 byte lines + case 0x88: + l3 = 2048; + break;// code and data L3 cache, 2048 KB, 4 ways, 64 byte lines (IA-64) + case 0x89: + l3 = 4096; + break;// code and data L3 cache, 4096 KB, 4 ways, 64 byte lines (IA-64) + case 0x8A: + l3 = 8192; + break;// code and data L3 cache, 8192 KB, 4 ways, 64 byte lines (IA-64) + case 0x8D: + l3 = 3072; + break;// code and data L3 cache, 3072 KB, 12 ways, 128 byte lines (IA-64) + + default: + break; + } } + if (check_for_p2_core2 && l2 == l3) l3 = 0; + l1 *= 1024; + l2 *= 1024; + l3 *= 1024; } - if(check_for_p2_core2 && l2 == l3) - l3 = 0; - l1 *= 1024; - l2 *= 1024; - l3 *= 1024; -} -inline void queryCacheSizes_intel(int& l1, int& l2, int& l3, int max_std_funcs) -{ - if(max_std_funcs>=4) - queryCacheSizes_intel_direct(l1,l2,l3); - else - queryCacheSizes_intel_codes(l1,l2,l3); -} + inline void queryCacheSizes_intel(int &l1, int &l2, int &l3, int max_std_funcs) + { + if (max_std_funcs >= 4) + queryCacheSizes_intel_direct(l1, l2, l3); + else + queryCacheSizes_intel_codes(l1, l2, l3); + } -inline void queryCacheSizes_amd(int& l1, int& l2, int& l3) -{ - int abcd[4]; - abcd[0] = abcd[1] = abcd[2] = abcd[3] = 0; - EIGEN_CPUID(abcd,0x80000005,0); - l1 = (abcd[2] >> 24) * 1024; // C[31:24] = L1 size in KB - abcd[0] = abcd[1] = abcd[2] = abcd[3] = 0; - EIGEN_CPUID(abcd,0x80000006,0); - l2 = (abcd[2] >> 16) * 1024; // C[31;16] = l2 cache size in KB - l3 = ((abcd[3] & 0xFFFC000) >> 18) * 512 * 1024; // D[31;18] = l3 cache size in 512KB -} + inline void queryCacheSizes_amd(int &l1, int &l2, int &l3) + { + int abcd[4]; + abcd[0] = abcd[1] = abcd[2] = abcd[3] = 0; + EIGEN_CPUID(abcd, 0x80000005, 0); + l1 = (abcd[2] >> 24) * 1024;// C[31:24] = L1 size in KB + abcd[0] = abcd[1] = abcd[2] = abcd[3] = 0; + EIGEN_CPUID(abcd, 0x80000006, 0); + l2 = (abcd[2] >> 16) * 1024;// C[31;16] = l2 cache size in KB + l3 = ((abcd[3] & 0xFFFC000) >> 18) * 512 * 1024;// D[31;18] = l3 cache size in 512KB + } #endif -/** \internal - * Queries and returns the cache sizes in Bytes of the L1, L2, and L3 data caches respectively */ -inline void queryCacheSizes(int& l1, int& l2, int& l3) -{ - #ifdef EIGEN_CPUID - int abcd[4]; - const int GenuineIntel[] = {0x756e6547, 0x49656e69, 0x6c65746e}; - const int AuthenticAMD[] = {0x68747541, 0x69746e65, 0x444d4163}; - const int AMDisbetter_[] = {0x69444d41, 0x74656273, 0x21726574}; // "AMDisbetter!" - - // identify the CPU vendor - EIGEN_CPUID(abcd,0x0,0); - int max_std_funcs = abcd[1]; - if(cpuid_is_vendor(abcd,GenuineIntel)) - queryCacheSizes_intel(l1,l2,l3,max_std_funcs); - else if(cpuid_is_vendor(abcd,AuthenticAMD) || cpuid_is_vendor(abcd,AMDisbetter_)) - queryCacheSizes_amd(l1,l2,l3); - else - // by default let's use Intel's API - queryCacheSizes_intel(l1,l2,l3,max_std_funcs); - - // here is the list of other vendors: -// ||cpuid_is_vendor(abcd,"VIA VIA VIA ") -// ||cpuid_is_vendor(abcd,"CyrixInstead") -// ||cpuid_is_vendor(abcd,"CentaurHauls") -// ||cpuid_is_vendor(abcd,"GenuineTMx86") -// ||cpuid_is_vendor(abcd,"TransmetaCPU") -// ||cpuid_is_vendor(abcd,"RiseRiseRise") -// ||cpuid_is_vendor(abcd,"Geode by NSC") -// ||cpuid_is_vendor(abcd,"SiS SiS SiS ") -// ||cpuid_is_vendor(abcd,"UMC UMC UMC ") -// ||cpuid_is_vendor(abcd,"NexGenDriven") - #else - l1 = l2 = l3 = -1; - #endif -} + /** \internal + * Queries and returns the cache sizes in Bytes of the L1, L2, and L3 data caches respectively */ + inline void queryCacheSizes(int &l1, int &l2, int &l3) + { +#ifdef EIGEN_CPUID + int abcd[4]; + const int GenuineIntel[] = { 0x756e6547, 0x49656e69, 0x6c65746e }; + const int AuthenticAMD[] = { 0x68747541, 0x69746e65, 0x444d4163 }; + const int AMDisbetter_[] = { 0x69444d41, 0x74656273, 0x21726574 };// "AMDisbetter!" + + // identify the CPU vendor + EIGEN_CPUID(abcd, 0x0, 0); + int max_std_funcs = abcd[1]; + if (cpuid_is_vendor(abcd, GenuineIntel)) + queryCacheSizes_intel(l1, l2, l3, max_std_funcs); + else if (cpuid_is_vendor(abcd, AuthenticAMD) || cpuid_is_vendor(abcd, AMDisbetter_)) + queryCacheSizes_amd(l1, l2, l3); + else + // by default let's use Intel's API + queryCacheSizes_intel(l1, l2, l3, max_std_funcs); + + // here is the list of other vendors: + // ||cpuid_is_vendor(abcd,"VIA VIA VIA ") + // ||cpuid_is_vendor(abcd,"CyrixInstead") + // ||cpuid_is_vendor(abcd,"CentaurHauls") + // ||cpuid_is_vendor(abcd,"GenuineTMx86") + // ||cpuid_is_vendor(abcd,"TransmetaCPU") + // ||cpuid_is_vendor(abcd,"RiseRiseRise") + // ||cpuid_is_vendor(abcd,"Geode by NSC") + // ||cpuid_is_vendor(abcd,"SiS SiS SiS ") + // ||cpuid_is_vendor(abcd,"UMC UMC UMC ") + // ||cpuid_is_vendor(abcd,"NexGenDriven") +#else + l1 = l2 = l3 = -1; +#endif + } -/** \internal - * \returns the size in Bytes of the L1 data cache */ -inline int queryL1CacheSize() -{ - int l1(-1), l2, l3; - queryCacheSizes(l1,l2,l3); - return l1; -} + /** \internal + * \returns the size in Bytes of the L1 data cache */ + inline int queryL1CacheSize() + { + int l1(-1), l2, l3; + queryCacheSizes(l1, l2, l3); + return l1; + } -/** \internal - * \returns the size in Bytes of the L2 or L3 cache if this later is present */ -inline int queryTopLevelCacheSize() -{ - int l1, l2(-1), l3(-1); - queryCacheSizes(l1,l2,l3); - return (std::max)(l2,l3); -} + /** \internal + * \returns the size in Bytes of the L2 or L3 cache if this later is present */ + inline int queryTopLevelCacheSize() + { + int l1, l2(-1), l3(-1); + queryCacheSizes(l1, l2, l3); + return (std::max)(l2, l3); + } -} // end namespace internal +}// end namespace internal -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_MEMORY_H +#endif// EIGEN_MEMORY_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/util/Meta.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/util/Meta.h old mode 100755 new mode 100644 index d31e9541..4f6b2183 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/util/Meta.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/util/Meta.h @@ -16,7 +16,7 @@ #include #endif -#if EIGEN_COMP_ICC>=1600 && __cplusplus >= 201103L +#if EIGEN_COMP_ICC >= 1600 && __cplusplus >= 201103L #include #endif @@ -35,156 +35,357 @@ typedef EIGEN_DEFAULT_DENSE_INDEX_TYPE Index; namespace internal { /** \internal - * \file Meta.h - * This file contains generic metaprogramming classes which are not specifically related to Eigen. - * \note In case you wonder, yes we're aware that Boost already provides all these features, - * we however don't want to add a dependency to Boost. - */ + * \file Meta.h + * This file contains generic metaprogramming classes which are not specifically related to Eigen. + * \note In case you wonder, yes we're aware that Boost already provides all these features, + * we however don't want to add a dependency to Boost. + */ // Only recent versions of ICC complain about using ptrdiff_t to hold pointers, // and older versions do not provide *intptr_t types. -#if EIGEN_COMP_ICC>=1600 && __cplusplus >= 201103L -typedef std::intptr_t IntPtr; -typedef std::uintptr_t UIntPtr; +#if EIGEN_COMP_ICC >= 1600 && __cplusplus >= 201103L + typedef std::intptr_t IntPtr; + typedef std::uintptr_t UIntPtr; #else -typedef std::ptrdiff_t IntPtr; -typedef std::size_t UIntPtr; + typedef std::ptrdiff_t IntPtr; + typedef std::size_t UIntPtr; #endif -struct true_type { enum { value = 1 }; }; -struct false_type { enum { value = 0 }; }; - -template -struct conditional { typedef Then type; }; - -template -struct conditional { typedef Else type; }; - -template struct is_same { enum { value = 0 }; }; -template struct is_same { enum { value = 1 }; }; - -template struct remove_reference { typedef T type; }; -template struct remove_reference { typedef T type; }; - -template struct remove_pointer { typedef T type; }; -template struct remove_pointer { typedef T type; }; -template struct remove_pointer { typedef T type; }; - -template struct remove_const { typedef T type; }; -template struct remove_const { typedef T type; }; -template struct remove_const { typedef T type[]; }; -template struct remove_const { typedef T type[Size]; }; - -template struct remove_all { typedef T type; }; -template struct remove_all { typedef typename remove_all::type type; }; -template struct remove_all { typedef typename remove_all::type type; }; -template struct remove_all { typedef typename remove_all::type type; }; -template struct remove_all { typedef typename remove_all::type type; }; -template struct remove_all { typedef typename remove_all::type type; }; - -template struct is_arithmetic { enum { value = false }; }; -template<> struct is_arithmetic { enum { value = true }; }; -template<> struct is_arithmetic { enum { value = true }; }; -template<> struct is_arithmetic { enum { value = true }; }; -template<> struct is_arithmetic { enum { value = true }; }; -template<> struct is_arithmetic { enum { value = true }; }; -template<> struct is_arithmetic { enum { value = true }; }; -template<> struct is_arithmetic { enum { value = true }; }; -template<> struct is_arithmetic { enum { value = true }; }; -template<> struct is_arithmetic{ enum { value = true }; }; -template<> struct is_arithmetic { enum { value = true }; }; -template<> struct is_arithmetic { enum { value = true }; }; -template<> struct is_arithmetic { enum { value = true }; }; -template<> struct is_arithmetic { enum { value = true }; }; - -template struct is_integral { enum { value = false }; }; -template<> struct is_integral { enum { value = true }; }; -template<> struct is_integral { enum { value = true }; }; -template<> struct is_integral { enum { value = true }; }; -template<> struct is_integral { enum { value = true }; }; -template<> struct is_integral { enum { value = true }; }; -template<> struct is_integral { enum { value = true }; }; -template<> struct is_integral { enum { value = true }; }; -template<> struct is_integral { enum { value = true }; }; -template<> struct is_integral { enum { value = true }; }; -template<> struct is_integral { enum { value = true }; }; + struct true_type + { + enum { value = 1 }; + }; + struct false_type + { + enum { value = 0 }; + }; + + template struct conditional + { + typedef Then type; + }; + + template struct conditional + { + typedef Else type; + }; + + template struct is_same + { + enum { value = 0 }; + }; + template struct is_same + { + enum { value = 1 }; + }; + + template struct remove_reference + { + typedef T type; + }; + template struct remove_reference + { + typedef T type; + }; + + template struct remove_pointer + { + typedef T type; + }; + template struct remove_pointer + { + typedef T type; + }; + template struct remove_pointer + { + typedef T type; + }; + + template struct remove_const + { + typedef T type; + }; + template struct remove_const + { + typedef T type; + }; + template struct remove_const + { + typedef T type[]; + }; + template struct remove_const + { + typedef T type[Size]; + }; + + template struct remove_all + { + typedef T type; + }; + template struct remove_all + { + typedef typename remove_all::type type; + }; + template struct remove_all + { + typedef typename remove_all::type type; + }; + template struct remove_all + { + typedef typename remove_all::type type; + }; + template struct remove_all + { + typedef typename remove_all::type type; + }; + template struct remove_all + { + typedef typename remove_all::type type; + }; + + template struct is_arithmetic + { + enum { value = false }; + }; + template<> struct is_arithmetic + { + enum { value = true }; + }; + template<> struct is_arithmetic + { + enum { value = true }; + }; + template<> struct is_arithmetic + { + enum { value = true }; + }; + template<> struct is_arithmetic + { + enum { value = true }; + }; + template<> struct is_arithmetic + { + enum { value = true }; + }; + template<> struct is_arithmetic + { + enum { value = true }; + }; + template<> struct is_arithmetic + { + enum { value = true }; + }; + template<> struct is_arithmetic + { + enum { value = true }; + }; + template<> struct is_arithmetic + { + enum { value = true }; + }; + template<> struct is_arithmetic + { + enum { value = true }; + }; + template<> struct is_arithmetic + { + enum { value = true }; + }; + template<> struct is_arithmetic + { + enum { value = true }; + }; + template<> struct is_arithmetic + { + enum { value = true }; + }; + + template struct is_integral + { + enum { value = false }; + }; + template<> struct is_integral + { + enum { value = true }; + }; + template<> struct is_integral + { + enum { value = true }; + }; + template<> struct is_integral + { + enum { value = true }; + }; + template<> struct is_integral + { + enum { value = true }; + }; + template<> struct is_integral + { + enum { value = true }; + }; + template<> struct is_integral + { + enum { value = true }; + }; + template<> struct is_integral + { + enum { value = true }; + }; + template<> struct is_integral + { + enum { value = true }; + }; + template<> struct is_integral + { + enum { value = true }; + }; + template<> struct is_integral + { + enum { value = true }; + }; #if EIGEN_HAS_CXX11 -using std::make_unsigned; + using std::make_unsigned; #else -// TODO: Possibly improve this implementation of make_unsigned. -// It is currently used only by -// template struct random_default_impl. -template struct make_unsigned; -template<> struct make_unsigned { typedef unsigned char type; }; -template<> struct make_unsigned { typedef unsigned char type; }; -template<> struct make_unsigned { typedef unsigned char type; }; -template<> struct make_unsigned { typedef unsigned short type; }; -template<> struct make_unsigned { typedef unsigned short type; }; -template<> struct make_unsigned { typedef unsigned int type; }; -template<> struct make_unsigned { typedef unsigned int type; }; -template<> struct make_unsigned { typedef unsigned long type; }; -template<> struct make_unsigned { typedef unsigned long type; }; + // TODO: Possibly improve this implementation of make_unsigned. + // It is currently used only by + // template struct random_default_impl. + template struct make_unsigned; + template<> struct make_unsigned + { + typedef unsigned char type; + }; + template<> struct make_unsigned + { + typedef unsigned char type; + }; + template<> struct make_unsigned + { + typedef unsigned char type; + }; + template<> struct make_unsigned + { + typedef unsigned short type; + }; + template<> struct make_unsigned + { + typedef unsigned short type; + }; + template<> struct make_unsigned + { + typedef unsigned int type; + }; + template<> struct make_unsigned + { + typedef unsigned int type; + }; + template<> struct make_unsigned + { + typedef unsigned long type; + }; + template<> struct make_unsigned + { + typedef unsigned long type; + }; #if EIGEN_COMP_MSVC -template<> struct make_unsigned { typedef unsigned __int64 type; }; -template<> struct make_unsigned { typedef unsigned __int64 type; }; + template<> struct make_unsigned + { + typedef unsigned __int64 type; + }; + template<> struct make_unsigned + { + typedef unsigned __int64 type; + }; #endif #endif -template struct add_const { typedef const T type; }; -template struct add_const { typedef T& type; }; + template struct add_const + { + typedef const T type; + }; + template struct add_const + { + typedef T &type; + }; -template struct is_const { enum { value = 0 }; }; -template struct is_const { enum { value = 1 }; }; + template struct is_const + { + enum { value = 0 }; + }; + template struct is_const + { + enum { value = 1 }; + }; -template struct add_const_on_value_type { typedef const T type; }; -template struct add_const_on_value_type { typedef T const& type; }; -template struct add_const_on_value_type { typedef T const* type; }; -template struct add_const_on_value_type { typedef T const* const type; }; -template struct add_const_on_value_type { typedef T const* const type; }; + template struct add_const_on_value_type + { + typedef const T type; + }; + template struct add_const_on_value_type + { + typedef T const &type; + }; + template struct add_const_on_value_type + { + typedef T const *type; + }; + template struct add_const_on_value_type + { + typedef T const *const type; + }; + template struct add_const_on_value_type + { + typedef T const *const type; + }; -template -struct is_convertible_impl -{ -private: - struct any_conversion + template struct is_convertible_impl { - template any_conversion(const volatile T&); - template any_conversion(T&); - }; - struct yes {int a[1];}; - struct no {int a[2];}; + private: + struct any_conversion + { + template any_conversion(const volatile T &); + template any_conversion(T &); + }; + struct yes + { + int a[1]; + }; + struct no + { + int a[2]; + }; - static yes test(const To&, int); - static no test(any_conversion, ...); + static yes test(const To &, int); + static no test(any_conversion, ...); -public: - static From ms_from; + public: + static From ms_from; #ifdef __INTEL_COMPILER - #pragma warning push - #pragma warning ( disable : 2259 ) +#pragma warning push +#pragma warning(disable : 2259) #endif - enum { value = sizeof(test(ms_from, 0))==sizeof(yes) }; + enum { value = sizeof(test(ms_from, 0)) == sizeof(yes) }; #ifdef __INTEL_COMPILER - #pragma warning pop +#pragma warning pop #endif -}; + }; -template -struct is_convertible -{ - enum { value = is_convertible_impl::type, - typename remove_all::type>::value }; -}; + template struct is_convertible + { + enum { value = is_convertible_impl::type, typename remove_all::type>::value }; + }; -/** \internal Allows to enable/disable an overload - * according to a compile time condition. - */ -template struct enable_if; + /** \internal Allows to enable/disable an overload + * according to a compile time condition. + */ + template struct enable_if; -template struct enable_if -{ typedef T type; }; + template struct enable_if + { + typedef T type; + }; #if defined(__CUDA_ARCH__) #if !defined(__FLT_EPSILON__) @@ -192,343 +393,393 @@ template struct enable_if #define __DBL_EPSILON__ DBL_EPSILON #endif -namespace device { - -template struct numeric_limits -{ - EIGEN_DEVICE_FUNC - static T epsilon() { return 0; } - static T (max)() { assert(false && "Highest not supported for this type"); } - static T (min)() { assert(false && "Lowest not supported for this type"); } - static T infinity() { assert(false && "Infinity not supported for this type"); } - static T quiet_NaN() { assert(false && "quiet_NaN not supported for this type"); } -}; -template<> struct numeric_limits -{ - EIGEN_DEVICE_FUNC - static float epsilon() { return __FLT_EPSILON__; } - EIGEN_DEVICE_FUNC - static float (max)() { return CUDART_MAX_NORMAL_F; } - EIGEN_DEVICE_FUNC - static float (min)() { return FLT_MIN; } - EIGEN_DEVICE_FUNC - static float infinity() { return CUDART_INF_F; } - EIGEN_DEVICE_FUNC - static float quiet_NaN() { return CUDART_NAN_F; } -}; -template<> struct numeric_limits -{ - EIGEN_DEVICE_FUNC - static double epsilon() { return __DBL_EPSILON__; } - EIGEN_DEVICE_FUNC - static double (max)() { return DBL_MAX; } - EIGEN_DEVICE_FUNC - static double (min)() { return DBL_MIN; } - EIGEN_DEVICE_FUNC - static double infinity() { return CUDART_INF; } - EIGEN_DEVICE_FUNC - static double quiet_NaN() { return CUDART_NAN; } -}; -template<> struct numeric_limits -{ - EIGEN_DEVICE_FUNC - static int epsilon() { return 0; } - EIGEN_DEVICE_FUNC - static int (max)() { return INT_MAX; } - EIGEN_DEVICE_FUNC - static int (min)() { return INT_MIN; } -}; -template<> struct numeric_limits -{ - EIGEN_DEVICE_FUNC - static unsigned int epsilon() { return 0; } - EIGEN_DEVICE_FUNC - static unsigned int (max)() { return UINT_MAX; } - EIGEN_DEVICE_FUNC - static unsigned int (min)() { return 0; } -}; -template<> struct numeric_limits -{ - EIGEN_DEVICE_FUNC - static long epsilon() { return 0; } - EIGEN_DEVICE_FUNC - static long (max)() { return LONG_MAX; } - EIGEN_DEVICE_FUNC - static long (min)() { return LONG_MIN; } -}; -template<> struct numeric_limits -{ - EIGEN_DEVICE_FUNC - static unsigned long epsilon() { return 0; } - EIGEN_DEVICE_FUNC - static unsigned long (max)() { return ULONG_MAX; } - EIGEN_DEVICE_FUNC - static unsigned long (min)() { return 0; } -}; -template<> struct numeric_limits -{ - EIGEN_DEVICE_FUNC - static long long epsilon() { return 0; } - EIGEN_DEVICE_FUNC - static long long (max)() { return LLONG_MAX; } - EIGEN_DEVICE_FUNC - static long long (min)() { return LLONG_MIN; } -}; -template<> struct numeric_limits -{ - EIGEN_DEVICE_FUNC - static unsigned long long epsilon() { return 0; } - EIGEN_DEVICE_FUNC - static unsigned long long (max)() { return ULLONG_MAX; } - EIGEN_DEVICE_FUNC - static unsigned long long (min)() { return 0; } -}; - -} + namespace device { + + template struct numeric_limits + { + EIGEN_DEVICE_FUNC + static T epsilon() { return 0; } + static T(max)() { assert(false && "Highest not supported for this type"); } + static T(min)() { assert(false && "Lowest not supported for this type"); } + static T infinity() { assert(false && "Infinity not supported for this type"); } + static T quiet_NaN() { assert(false && "quiet_NaN not supported for this type"); } + }; + template<> struct numeric_limits + { + EIGEN_DEVICE_FUNC + static float epsilon() { return __FLT_EPSILON__; } + EIGEN_DEVICE_FUNC + static float(max)() { return CUDART_MAX_NORMAL_F; } + EIGEN_DEVICE_FUNC + static float(min)() { return FLT_MIN; } + EIGEN_DEVICE_FUNC + static float infinity() { return CUDART_INF_F; } + EIGEN_DEVICE_FUNC + static float quiet_NaN() { return CUDART_NAN_F; } + }; + template<> struct numeric_limits + { + EIGEN_DEVICE_FUNC + static double epsilon() { return __DBL_EPSILON__; } + EIGEN_DEVICE_FUNC + static double(max)() { return DBL_MAX; } + EIGEN_DEVICE_FUNC + static double(min)() { return DBL_MIN; } + EIGEN_DEVICE_FUNC + static double infinity() { return CUDART_INF; } + EIGEN_DEVICE_FUNC + static double quiet_NaN() { return CUDART_NAN; } + }; + template<> struct numeric_limits + { + EIGEN_DEVICE_FUNC + static int epsilon() { return 0; } + EIGEN_DEVICE_FUNC + static int(max)() { return INT_MAX; } + EIGEN_DEVICE_FUNC + static int(min)() { return INT_MIN; } + }; + template<> struct numeric_limits + { + EIGEN_DEVICE_FUNC + static unsigned int epsilon() { return 0; } + EIGEN_DEVICE_FUNC + static unsigned int(max)() { return UINT_MAX; } + EIGEN_DEVICE_FUNC + static unsigned int(min)() { return 0; } + }; + template<> struct numeric_limits + { + EIGEN_DEVICE_FUNC + static long epsilon() { return 0; } + EIGEN_DEVICE_FUNC + static long(max)() { return LONG_MAX; } + EIGEN_DEVICE_FUNC + static long(min)() { return LONG_MIN; } + }; + template<> struct numeric_limits + { + EIGEN_DEVICE_FUNC + static unsigned long epsilon() { return 0; } + EIGEN_DEVICE_FUNC + static unsigned long(max)() { return ULONG_MAX; } + EIGEN_DEVICE_FUNC + static unsigned long(min)() { return 0; } + }; + template<> struct numeric_limits + { + EIGEN_DEVICE_FUNC + static long long epsilon() { return 0; } + EIGEN_DEVICE_FUNC + static long long(max)() { return LLONG_MAX; } + EIGEN_DEVICE_FUNC + static long long(min)() { return LLONG_MIN; } + }; + template<> struct numeric_limits + { + EIGEN_DEVICE_FUNC + static unsigned long long epsilon() { return 0; } + EIGEN_DEVICE_FUNC + static unsigned long long(max)() { return ULLONG_MAX; } + EIGEN_DEVICE_FUNC + static unsigned long long(min)() { return 0; } + }; + + }// namespace device #endif -/** \internal - * A base class do disable default copy ctor and copy assignement operator. - */ -class noncopyable -{ - EIGEN_DEVICE_FUNC noncopyable(const noncopyable&); - EIGEN_DEVICE_FUNC const noncopyable& operator=(const noncopyable&); -protected: - EIGEN_DEVICE_FUNC noncopyable() {} - EIGEN_DEVICE_FUNC ~noncopyable() {} -}; + /** \internal + * A base class do disable default copy ctor and copy assignement operator. + */ + class noncopyable + { + EIGEN_DEVICE_FUNC noncopyable(const noncopyable &); + EIGEN_DEVICE_FUNC const noncopyable &operator=(const noncopyable &); + + protected: + EIGEN_DEVICE_FUNC noncopyable() {} + EIGEN_DEVICE_FUNC ~noncopyable() {} + }; /** \internal - * Convenient struct to get the result type of a unary or binary functor. - * - * It supports both the current STL mechanism (using the result_type member) as well as - * upcoming next STL generation (using a templated result member). - * If none of these members is provided, then the type of the first argument is returned. FIXME, that behavior is a pretty bad hack. - */ + * Convenient struct to get the result type of a unary or binary functor. + * + * It supports both the current STL mechanism (using the result_type member) as well as + * upcoming next STL generation (using a templated result member). + * If none of these members is provided, then the type of the first argument is returned. FIXME, that behavior is a + * pretty bad hack. + */ #if EIGEN_HAS_STD_RESULT_OF -template struct result_of { - typedef typename std::result_of::type type1; - typedef typename remove_all::type type; -}; + template struct result_of + { + typedef typename std::result_of::type type1; + typedef typename remove_all::type type; + }; #else -template struct result_of { }; + template struct result_of + { + }; -struct has_none {int a[1];}; -struct has_std_result_type {int a[2];}; -struct has_tr1_result {int a[3];}; + struct has_none + { + int a[1]; + }; + struct has_std_result_type + { + int a[2]; + }; + struct has_tr1_result + { + int a[3]; + }; -template -struct unary_result_of_select {typedef typename internal::remove_all::type type;}; + template struct unary_result_of_select + { + typedef typename internal::remove_all::type type; + }; -template -struct unary_result_of_select {typedef typename Func::result_type type;}; + template struct unary_result_of_select + { + typedef typename Func::result_type type; + }; -template -struct unary_result_of_select {typedef typename Func::template result::type type;}; + template struct unary_result_of_select + { + typedef typename Func::template result::type type; + }; -template -struct result_of { - template - static has_std_result_type testFunctor(T const *, typename T::result_type const * = 0); + template struct result_of + { + template static has_std_result_type testFunctor(T const *, typename T::result_type const * = 0); template - static has_tr1_result testFunctor(T const *, typename T::template result::type const * = 0); - static has_none testFunctor(...); + static has_tr1_result testFunctor(T const *, typename T::template result::type const * = 0); + static has_none testFunctor(...); // note that the following indirection is needed for gcc-3.3 - enum {FunctorType = sizeof(testFunctor(static_cast(0)))}; + enum { FunctorType = sizeof(testFunctor(static_cast(0))) }; typedef typename unary_result_of_select::type type; -}; + }; -template -struct binary_result_of_select {typedef typename internal::remove_all::type type;}; + template + struct binary_result_of_select + { + typedef typename internal::remove_all::type type; + }; -template -struct binary_result_of_select -{typedef typename Func::result_type type;}; + template + struct binary_result_of_select + { + typedef typename Func::result_type type; + }; -template -struct binary_result_of_select -{typedef typename Func::template result::type type;}; + template + struct binary_result_of_select + { + typedef typename Func::template result::type type; + }; -template -struct result_of { - template - static has_std_result_type testFunctor(T const *, typename T::result_type const * = 0); + template struct result_of + { + template static has_std_result_type testFunctor(T const *, typename T::result_type const * = 0); template - static has_tr1_result testFunctor(T const *, typename T::template result::type const * = 0); - static has_none testFunctor(...); + static has_tr1_result testFunctor(T const *, typename T::template result::type const * = 0); + static has_none testFunctor(...); // note that the following indirection is needed for gcc-3.3 - enum {FunctorType = sizeof(testFunctor(static_cast(0)))}; + enum { FunctorType = sizeof(testFunctor(static_cast(0))) }; typedef typename binary_result_of_select::type type; -}; + }; -template -struct ternary_result_of_select {typedef typename internal::remove_all::type type;}; + template + struct ternary_result_of_select + { + typedef typename internal::remove_all::type type; + }; -template -struct ternary_result_of_select -{typedef typename Func::result_type type;}; + template + struct ternary_result_of_select + { + typedef typename Func::result_type type; + }; -template -struct ternary_result_of_select -{typedef typename Func::template result::type type;}; + template + struct ternary_result_of_select + { + typedef typename Func::template result::type type; + }; -template -struct result_of { - template - static has_std_result_type testFunctor(T const *, typename T::result_type const * = 0); + template + struct result_of + { + template static has_std_result_type testFunctor(T const *, typename T::result_type const * = 0); template - static has_tr1_result testFunctor(T const *, typename T::template result::type const * = 0); - static has_none testFunctor(...); + static has_tr1_result testFunctor(T const *, + typename T::template result::type const * = 0); + static has_none testFunctor(...); // note that the following indirection is needed for gcc-3.3 - enum {FunctorType = sizeof(testFunctor(static_cast(0)))}; + enum { FunctorType = sizeof(testFunctor(static_cast(0))) }; typedef typename ternary_result_of_select::type type; -}; + }; #endif -struct meta_yes { char a[1]; }; -struct meta_no { char a[2]; }; - -// Check whether T::ReturnType does exist -template -struct has_ReturnType -{ - template static meta_yes testFunctor(typename C::ReturnType const *); - template static meta_no testFunctor(...); - - enum { value = sizeof(testFunctor(0)) == sizeof(meta_yes) }; -}; - -template const T* return_ptr(); - -template -struct has_nullary_operator -{ - template static meta_yes testFunctor(C const *,typename enable_if<(sizeof(return_ptr()->operator()())>0)>::type * = 0); - static meta_no testFunctor(...); - - enum { value = sizeof(testFunctor(static_cast(0))) == sizeof(meta_yes) }; -}; - -template -struct has_unary_operator -{ - template static meta_yes testFunctor(C const *,typename enable_if<(sizeof(return_ptr()->operator()(IndexType(0)))>0)>::type * = 0); - static meta_no testFunctor(...); - - enum { value = sizeof(testFunctor(static_cast(0))) == sizeof(meta_yes) }; -}; - -template -struct has_binary_operator -{ - template static meta_yes testFunctor(C const *,typename enable_if<(sizeof(return_ptr()->operator()(IndexType(0),IndexType(0)))>0)>::type * = 0); - static meta_no testFunctor(...); - - enum { value = sizeof(testFunctor(static_cast(0))) == sizeof(meta_yes) }; -}; - -/** \internal In short, it computes int(sqrt(\a Y)) with \a Y an integer. - * Usage example: \code meta_sqrt<1023>::ret \endcode - */ -template Y))) > - // use ?: instead of || just to shut up a stupid gcc 4.3 warning -class meta_sqrt -{ + struct meta_yes + { + char a[1]; + }; + struct meta_no + { + char a[2]; + }; + + // Check whether T::ReturnType does exist + template struct has_ReturnType + { + template static meta_yes testFunctor(typename C::ReturnType const *); + template static meta_no testFunctor(...); + + enum { value = sizeof(testFunctor(0)) == sizeof(meta_yes) }; + }; + + template const T *return_ptr(); + + template struct has_nullary_operator + { + template + static meta_yes testFunctor(C const *, typename enable_if<(sizeof(return_ptr()->operator()()) > 0)>::type * = 0); + static meta_no testFunctor(...); + + enum { value = sizeof(testFunctor(static_cast(0))) == sizeof(meta_yes) }; + }; + + template struct has_unary_operator + { + template + static meta_yes testFunctor(C const *, + typename enable_if<(sizeof(return_ptr()->operator()(IndexType(0))) > 0)>::type * = 0); + static meta_no testFunctor(...); + + enum { value = sizeof(testFunctor(static_cast(0))) == sizeof(meta_yes) }; + }; + + template struct has_binary_operator + { + template + static meta_yes testFunctor(C const *, + typename enable_if<(sizeof(return_ptr()->operator()(IndexType(0), IndexType(0))) > 0)>::type * = 0); + static meta_no testFunctor(...); + + enum { value = sizeof(testFunctor(static_cast(0))) == sizeof(meta_yes) }; + }; + + /** \internal In short, it computes int(sqrt(\a Y)) with \a Y an integer. + * Usage example: \code meta_sqrt<1023>::ret \endcode + */ + template Y)))> + // use ?: instead of || just to shut up a stupid gcc 4.3 warning + class meta_sqrt + { enum { - MidX = (InfX+SupX)/2, - TakeInf = MidX*MidX > Y ? 1 : 0, + MidX = (InfX + SupX) / 2, + TakeInf = MidX * MidX > Y ? 1 : 0, NewInf = int(TakeInf) ? InfX : int(MidX), NewSup = int(TakeInf) ? int(MidX) : SupX }; + + public: + enum { ret = meta_sqrt::ret }; + }; + + template class meta_sqrt + { public: - enum { ret = meta_sqrt::ret }; -}; - -template -class meta_sqrt { public: enum { ret = (SupX*SupX <= Y) ? SupX : InfX }; }; - - -/** \internal Computes the least common multiple of two positive integer A and B - * at compile-time. It implements a naive algorithm testing all multiples of A. - * It thus works better if A>=B. - */ -template -struct meta_least_common_multiple -{ - enum { ret = meta_least_common_multiple::ret }; -}; -template -struct meta_least_common_multiple -{ - enum { ret = A*K }; -}; - -/** \internal determines whether the product of two numeric types is allowed and what the return type is */ -template struct scalar_product_traits -{ - enum { Defined = 0 }; -}; - -// FIXME quick workaround around current limitation of result_of -// template -// struct result_of(ArgType0,ArgType1)> { -// typedef typename scalar_product_traits::type, typename remove_all::type>::ReturnType type; -// }; - -} // end namespace internal + enum { ret = (SupX * SupX <= Y) ? SupX : InfX }; + }; + + + /** \internal Computes the least common multiple of two positive integer A and B + * at compile-time. It implements a naive algorithm testing all multiples of A. + * It thus works better if A>=B. + */ + template struct meta_least_common_multiple + { + enum { ret = meta_least_common_multiple::ret }; + }; + template struct meta_least_common_multiple + { + enum { ret = A * K }; + }; + + /** \internal determines whether the product of two numeric types is allowed and what the return type is */ + template struct scalar_product_traits + { + enum { Defined = 0 }; + }; + + // FIXME quick workaround around current limitation of result_of + // template + // struct result_of(ArgType0,ArgType1)> { + // typedef typename scalar_product_traits::type, typename + // remove_all::type>::ReturnType type; + // }; + +}// end namespace internal namespace numext { - + #if defined(__CUDA_ARCH__) -template EIGEN_DEVICE_FUNC void swap(T &a, T &b) { T tmp = b; b = a; a = tmp; } + template EIGEN_DEVICE_FUNC void swap(T &a, T &b) + { + T tmp = b; + b = a; + a = tmp; + } #else -template EIGEN_STRONG_INLINE void swap(T &a, T &b) { std::swap(a,b); } + template EIGEN_STRONG_INLINE void swap(T &a, T &b) { std::swap(a, b); } #endif #if defined(__CUDA_ARCH__) -using internal::device::numeric_limits; + using internal::device::numeric_limits; #else -using std::numeric_limits; + using std::numeric_limits; #endif -// Integer division with rounding up. -// T is assumed to be an integer type with a>=0, and b>0 -template -T div_ceil(const T &a, const T &b) -{ - return (a+b-1) / b; -} + // Integer division with rounding up. + // T is assumed to be an integer type with a>=0, and b>0 + template T div_ceil(const T &a, const T &b) { return (a + b - 1) / b; } -// The aim of the following functions is to bypass -Wfloat-equal warnings -// when we really want a strict equality comparison on floating points. -template EIGEN_STRONG_INLINE -bool equal_strict(const X& x,const Y& y) { return x == y; } + // The aim of the following functions is to bypass -Wfloat-equal warnings + // when we really want a strict equality comparison on floating points. + template EIGEN_STRONG_INLINE bool equal_strict(const X &x, const Y &y) { return x == y; } -template<> EIGEN_STRONG_INLINE -bool equal_strict(const float& x,const float& y) { return std::equal_to()(x,y); } + template<> EIGEN_STRONG_INLINE bool equal_strict(const float &x, const float &y) + { + return std::equal_to()(x, y); + } -template<> EIGEN_STRONG_INLINE -bool equal_strict(const double& x,const double& y) { return std::equal_to()(x,y); } + template<> EIGEN_STRONG_INLINE bool equal_strict(const double &x, const double &y) + { + return std::equal_to()(x, y); + } -template EIGEN_STRONG_INLINE -bool not_equal_strict(const X& x,const Y& y) { return x != y; } + template EIGEN_STRONG_INLINE bool not_equal_strict(const X &x, const Y &y) { return x != y; } -template<> EIGEN_STRONG_INLINE -bool not_equal_strict(const float& x,const float& y) { return std::not_equal_to()(x,y); } + template<> EIGEN_STRONG_INLINE bool not_equal_strict(const float &x, const float &y) + { + return std::not_equal_to()(x, y); + } -template<> EIGEN_STRONG_INLINE -bool not_equal_strict(const double& x,const double& y) { return std::not_equal_to()(x,y); } + template<> EIGEN_STRONG_INLINE bool not_equal_strict(const double &x, const double &y) + { + return std::not_equal_to()(x, y); + } -} // end namespace numext +}// end namespace numext -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_META_H +#endif// EIGEN_META_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/util/ReenableStupidWarnings.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/util/ReenableStupidWarnings.h index ecc82b7c..367a6197 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/util/ReenableStupidWarnings.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/util/ReenableStupidWarnings.h @@ -2,17 +2,17 @@ #undef EIGEN_WARNINGS_DISABLED #ifndef EIGEN_PERMANENTLY_DISABLE_STUPID_WARNINGS - #ifdef _MSC_VER - #pragma warning( pop ) - #elif defined __INTEL_COMPILER - #pragma warning pop - #elif defined __clang__ - #pragma clang diagnostic pop - #elif defined __GNUC__ && (__GNUC__ > 4 || (__GNUC__ == 4 && __GNUC_MINOR__ >= 6)) - #pragma GCC diagnostic pop - #endif +#ifdef _MSC_VER +#pragma warning(pop) +#elif defined __INTEL_COMPILER +#pragma warning pop +#elif defined __clang__ +#pragma clang diagnostic pop +#elif defined __GNUC__ && (__GNUC__ > 4 || (__GNUC__ == 4 && __GNUC_MINOR__ >= 6)) +#pragma GCC diagnostic pop +#endif - #if defined __NVCC__ +#if defined __NVCC__ // Don't reenable the diagnostic messages, as it turns out these messages need // to be disabled at the point of the template instantiation (i.e the user code) // otherwise they'll be triggered by nvcc. @@ -20,8 +20,8 @@ // #pragma diag_default initialization_not_reachable // #pragma diag_default 2651 // #pragma diag_default 2653 - #endif +#endif #endif -#endif // EIGEN_WARNINGS_DISABLED +#endif// EIGEN_WARNINGS_DISABLED diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/util/StaticAssert.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/util/StaticAssert.h index 500e4779..74071455 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/util/StaticAssert.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/util/StaticAssert.h @@ -27,192 +27,188 @@ #ifndef EIGEN_STATIC_ASSERT #ifndef EIGEN_NO_STATIC_ASSERT - #if EIGEN_MAX_CPP_VER>=11 && (__has_feature(cxx_static_assert) || (defined(__cplusplus) && __cplusplus >= 201103L) || (EIGEN_COMP_MSVC >= 1600)) - - // if native static_assert is enabled, let's use it - #define EIGEN_STATIC_ASSERT(X,MSG) static_assert(X,#MSG); - - #else // not CXX0X - - namespace Eigen { - - namespace internal { - - template - struct static_assertion {}; - - template<> - struct static_assertion - { - enum { - YOU_TRIED_CALLING_A_VECTOR_METHOD_ON_A_MATRIX=1, - YOU_MIXED_VECTORS_OF_DIFFERENT_SIZES=1, - YOU_MIXED_MATRICES_OF_DIFFERENT_SIZES=1, - THIS_METHOD_IS_ONLY_FOR_VECTORS_OF_A_SPECIFIC_SIZE=1, - THIS_METHOD_IS_ONLY_FOR_MATRICES_OF_A_SPECIFIC_SIZE=1, - THIS_METHOD_IS_ONLY_FOR_OBJECTS_OF_A_SPECIFIC_SIZE=1, - OUT_OF_RANGE_ACCESS=1, - YOU_MADE_A_PROGRAMMING_MISTAKE=1, - EIGEN_INTERNAL_ERROR_PLEASE_FILE_A_BUG_REPORT=1, - EIGEN_INTERNAL_COMPILATION_ERROR_OR_YOU_MADE_A_PROGRAMMING_MISTAKE=1, - YOU_CALLED_A_FIXED_SIZE_METHOD_ON_A_DYNAMIC_SIZE_MATRIX_OR_VECTOR=1, - YOU_CALLED_A_DYNAMIC_SIZE_METHOD_ON_A_FIXED_SIZE_MATRIX_OR_VECTOR=1, - UNALIGNED_LOAD_AND_STORE_OPERATIONS_UNIMPLEMENTED_ON_ALTIVEC=1, - THIS_FUNCTION_IS_NOT_FOR_INTEGER_NUMERIC_TYPES=1, - FLOATING_POINT_ARGUMENT_PASSED__INTEGER_WAS_EXPECTED=1, - NUMERIC_TYPE_MUST_BE_REAL=1, - COEFFICIENT_WRITE_ACCESS_TO_SELFADJOINT_NOT_SUPPORTED=1, - WRITING_TO_TRIANGULAR_PART_WITH_UNIT_DIAGONAL_IS_NOT_SUPPORTED=1, - THIS_METHOD_IS_ONLY_FOR_FIXED_SIZE=1, - INVALID_MATRIX_PRODUCT=1, - INVALID_VECTOR_VECTOR_PRODUCT__IF_YOU_WANTED_A_DOT_OR_COEFF_WISE_PRODUCT_YOU_MUST_USE_THE_EXPLICIT_FUNCTIONS=1, - INVALID_MATRIX_PRODUCT__IF_YOU_WANTED_A_COEFF_WISE_PRODUCT_YOU_MUST_USE_THE_EXPLICIT_FUNCTION=1, - YOU_MIXED_DIFFERENT_NUMERIC_TYPES__YOU_NEED_TO_USE_THE_CAST_METHOD_OF_MATRIXBASE_TO_CAST_NUMERIC_TYPES_EXPLICITLY=1, - THIS_METHOD_IS_ONLY_FOR_COLUMN_MAJOR_MATRICES=1, - THIS_METHOD_IS_ONLY_FOR_ROW_MAJOR_MATRICES=1, - INVALID_MATRIX_TEMPLATE_PARAMETERS=1, - INVALID_MATRIXBASE_TEMPLATE_PARAMETERS=1, - BOTH_MATRICES_MUST_HAVE_THE_SAME_STORAGE_ORDER=1, - THIS_METHOD_IS_ONLY_FOR_DIAGONAL_MATRIX=1, - THE_MATRIX_OR_EXPRESSION_THAT_YOU_PASSED_DOES_NOT_HAVE_THE_EXPECTED_TYPE=1, - THIS_METHOD_IS_ONLY_FOR_EXPRESSIONS_WITH_DIRECT_MEMORY_ACCESS_SUCH_AS_MAP_OR_PLAIN_MATRICES=1, - YOU_ALREADY_SPECIFIED_THIS_STRIDE=1, - INVALID_STORAGE_ORDER_FOR_THIS_VECTOR_EXPRESSION=1, - THE_BRACKET_OPERATOR_IS_ONLY_FOR_VECTORS__USE_THE_PARENTHESIS_OPERATOR_INSTEAD=1, - PACKET_ACCESS_REQUIRES_TO_HAVE_INNER_STRIDE_FIXED_TO_1=1, - THIS_METHOD_IS_ONLY_FOR_SPECIFIC_TRANSFORMATIONS=1, - YOU_CANNOT_MIX_ARRAYS_AND_MATRICES=1, - YOU_PERFORMED_AN_INVALID_TRANSFORMATION_CONVERSION=1, - THIS_EXPRESSION_IS_NOT_A_LVALUE__IT_IS_READ_ONLY=1, - YOU_ARE_TRYING_TO_USE_AN_INDEX_BASED_ACCESSOR_ON_AN_EXPRESSION_THAT_DOES_NOT_SUPPORT_THAT=1, - THIS_METHOD_IS_ONLY_FOR_1x1_EXPRESSIONS=1, - THIS_METHOD_IS_ONLY_FOR_INNER_OR_LAZY_PRODUCTS=1, - THIS_METHOD_IS_ONLY_FOR_EXPRESSIONS_OF_BOOL=1, - THIS_METHOD_IS_ONLY_FOR_ARRAYS_NOT_MATRICES=1, - YOU_PASSED_A_ROW_VECTOR_BUT_A_COLUMN_VECTOR_WAS_EXPECTED=1, - YOU_PASSED_A_COLUMN_VECTOR_BUT_A_ROW_VECTOR_WAS_EXPECTED=1, - THE_INDEX_TYPE_MUST_BE_A_SIGNED_TYPE=1, - THE_STORAGE_ORDER_OF_BOTH_SIDES_MUST_MATCH=1, - OBJECT_ALLOCATED_ON_STACK_IS_TOO_BIG=1, - IMPLICIT_CONVERSION_TO_SCALAR_IS_FOR_INNER_PRODUCT_ONLY=1, - STORAGE_LAYOUT_DOES_NOT_MATCH=1, - EIGEN_INTERNAL_ERROR_PLEASE_FILE_A_BUG_REPORT__INVALID_COST_VALUE=1, - THIS_COEFFICIENT_ACCESSOR_TAKING_ONE_ACCESS_IS_ONLY_FOR_EXPRESSIONS_ALLOWING_LINEAR_ACCESS=1, - MATRIX_FREE_CONJUGATE_GRADIENT_IS_COMPATIBLE_WITH_UPPER_UNION_LOWER_MODE_ONLY=1, - THIS_TYPE_IS_NOT_SUPPORTED=1, - STORAGE_KIND_MUST_MATCH=1, - STORAGE_INDEX_MUST_MATCH=1, - CHOLMOD_SUPPORTS_DOUBLE_PRECISION_ONLY=1, - SELFADJOINTVIEW_ACCEPTS_UPPER_AND_LOWER_MODE_ONLY=1 - }; +#if EIGEN_MAX_CPP_VER >= 11 \ + && (__has_feature(cxx_static_assert) || (defined(__cplusplus) && __cplusplus >= 201103L) \ + || (EIGEN_COMP_MSVC >= 1600)) + +// if native static_assert is enabled, let's use it +#define EIGEN_STATIC_ASSERT(X, MSG) static_assert(X, #MSG); + +#else// not CXX0X + +namespace Eigen { + +namespace internal { + + template struct static_assertion + { + }; + + template<> struct static_assertion + { + enum { + YOU_TRIED_CALLING_A_VECTOR_METHOD_ON_A_MATRIX = 1, + YOU_MIXED_VECTORS_OF_DIFFERENT_SIZES = 1, + YOU_MIXED_MATRICES_OF_DIFFERENT_SIZES = 1, + THIS_METHOD_IS_ONLY_FOR_VECTORS_OF_A_SPECIFIC_SIZE = 1, + THIS_METHOD_IS_ONLY_FOR_MATRICES_OF_A_SPECIFIC_SIZE = 1, + THIS_METHOD_IS_ONLY_FOR_OBJECTS_OF_A_SPECIFIC_SIZE = 1, + OUT_OF_RANGE_ACCESS = 1, + YOU_MADE_A_PROGRAMMING_MISTAKE = 1, + EIGEN_INTERNAL_ERROR_PLEASE_FILE_A_BUG_REPORT = 1, + EIGEN_INTERNAL_COMPILATION_ERROR_OR_YOU_MADE_A_PROGRAMMING_MISTAKE = 1, + YOU_CALLED_A_FIXED_SIZE_METHOD_ON_A_DYNAMIC_SIZE_MATRIX_OR_VECTOR = 1, + YOU_CALLED_A_DYNAMIC_SIZE_METHOD_ON_A_FIXED_SIZE_MATRIX_OR_VECTOR = 1, + UNALIGNED_LOAD_AND_STORE_OPERATIONS_UNIMPLEMENTED_ON_ALTIVEC = 1, + THIS_FUNCTION_IS_NOT_FOR_INTEGER_NUMERIC_TYPES = 1, + FLOATING_POINT_ARGUMENT_PASSED__INTEGER_WAS_EXPECTED = 1, + NUMERIC_TYPE_MUST_BE_REAL = 1, + COEFFICIENT_WRITE_ACCESS_TO_SELFADJOINT_NOT_SUPPORTED = 1, + WRITING_TO_TRIANGULAR_PART_WITH_UNIT_DIAGONAL_IS_NOT_SUPPORTED = 1, + THIS_METHOD_IS_ONLY_FOR_FIXED_SIZE = 1, + INVALID_MATRIX_PRODUCT = 1, + INVALID_VECTOR_VECTOR_PRODUCT__IF_YOU_WANTED_A_DOT_OR_COEFF_WISE_PRODUCT_YOU_MUST_USE_THE_EXPLICIT_FUNCTIONS = 1, + INVALID_MATRIX_PRODUCT__IF_YOU_WANTED_A_COEFF_WISE_PRODUCT_YOU_MUST_USE_THE_EXPLICIT_FUNCTION = 1, + YOU_MIXED_DIFFERENT_NUMERIC_TYPES__YOU_NEED_TO_USE_THE_CAST_METHOD_OF_MATRIXBASE_TO_CAST_NUMERIC_TYPES_EXPLICITLY = + 1, + THIS_METHOD_IS_ONLY_FOR_COLUMN_MAJOR_MATRICES = 1, + THIS_METHOD_IS_ONLY_FOR_ROW_MAJOR_MATRICES = 1, + INVALID_MATRIX_TEMPLATE_PARAMETERS = 1, + INVALID_MATRIXBASE_TEMPLATE_PARAMETERS = 1, + BOTH_MATRICES_MUST_HAVE_THE_SAME_STORAGE_ORDER = 1, + THIS_METHOD_IS_ONLY_FOR_DIAGONAL_MATRIX = 1, + THE_MATRIX_OR_EXPRESSION_THAT_YOU_PASSED_DOES_NOT_HAVE_THE_EXPECTED_TYPE = 1, + THIS_METHOD_IS_ONLY_FOR_EXPRESSIONS_WITH_DIRECT_MEMORY_ACCESS_SUCH_AS_MAP_OR_PLAIN_MATRICES = 1, + YOU_ALREADY_SPECIFIED_THIS_STRIDE = 1, + INVALID_STORAGE_ORDER_FOR_THIS_VECTOR_EXPRESSION = 1, + THE_BRACKET_OPERATOR_IS_ONLY_FOR_VECTORS__USE_THE_PARENTHESIS_OPERATOR_INSTEAD = 1, + PACKET_ACCESS_REQUIRES_TO_HAVE_INNER_STRIDE_FIXED_TO_1 = 1, + THIS_METHOD_IS_ONLY_FOR_SPECIFIC_TRANSFORMATIONS = 1, + YOU_CANNOT_MIX_ARRAYS_AND_MATRICES = 1, + YOU_PERFORMED_AN_INVALID_TRANSFORMATION_CONVERSION = 1, + THIS_EXPRESSION_IS_NOT_A_LVALUE__IT_IS_READ_ONLY = 1, + YOU_ARE_TRYING_TO_USE_AN_INDEX_BASED_ACCESSOR_ON_AN_EXPRESSION_THAT_DOES_NOT_SUPPORT_THAT = 1, + THIS_METHOD_IS_ONLY_FOR_1x1_EXPRESSIONS = 1, + THIS_METHOD_IS_ONLY_FOR_INNER_OR_LAZY_PRODUCTS = 1, + THIS_METHOD_IS_ONLY_FOR_EXPRESSIONS_OF_BOOL = 1, + THIS_METHOD_IS_ONLY_FOR_ARRAYS_NOT_MATRICES = 1, + YOU_PASSED_A_ROW_VECTOR_BUT_A_COLUMN_VECTOR_WAS_EXPECTED = 1, + YOU_PASSED_A_COLUMN_VECTOR_BUT_A_ROW_VECTOR_WAS_EXPECTED = 1, + THE_INDEX_TYPE_MUST_BE_A_SIGNED_TYPE = 1, + THE_STORAGE_ORDER_OF_BOTH_SIDES_MUST_MATCH = 1, + OBJECT_ALLOCATED_ON_STACK_IS_TOO_BIG = 1, + IMPLICIT_CONVERSION_TO_SCALAR_IS_FOR_INNER_PRODUCT_ONLY = 1, + STORAGE_LAYOUT_DOES_NOT_MATCH = 1, + EIGEN_INTERNAL_ERROR_PLEASE_FILE_A_BUG_REPORT__INVALID_COST_VALUE = 1, + THIS_COEFFICIENT_ACCESSOR_TAKING_ONE_ACCESS_IS_ONLY_FOR_EXPRESSIONS_ALLOWING_LINEAR_ACCESS = 1, + MATRIX_FREE_CONJUGATE_GRADIENT_IS_COMPATIBLE_WITH_UPPER_UNION_LOWER_MODE_ONLY = 1, + THIS_TYPE_IS_NOT_SUPPORTED = 1, + STORAGE_KIND_MUST_MATCH = 1, + STORAGE_INDEX_MUST_MATCH = 1, + CHOLMOD_SUPPORTS_DOUBLE_PRECISION_ONLY = 1, + SELFADJOINTVIEW_ACCEPTS_UPPER_AND_LOWER_MODE_ONLY = 1 }; + }; - } // end namespace internal +}// end namespace internal - } // end namespace Eigen +}// end namespace Eigen - // Specialized implementation for MSVC to avoid "conditional - // expression is constant" warnings. This implementation doesn't - // appear to work under GCC, hence the multiple implementations. - #if EIGEN_COMP_MSVC +// Specialized implementation for MSVC to avoid "conditional +// expression is constant" warnings. This implementation doesn't +// appear to work under GCC, hence the multiple implementations. +#if EIGEN_COMP_MSVC - #define EIGEN_STATIC_ASSERT(CONDITION,MSG) \ - {Eigen::internal::static_assertion::MSG;} +#define EIGEN_STATIC_ASSERT(CONDITION, MSG) \ + { \ + Eigen::internal::static_assertion::MSG; \ + } - #else - // In some cases clang interprets bool(CONDITION) as function declaration - #define EIGEN_STATIC_ASSERT(CONDITION,MSG) \ - if (Eigen::internal::static_assertion(CONDITION)>::MSG) {} +#else + // In some cases clang interprets bool(CONDITION) as function declaration +#define EIGEN_STATIC_ASSERT(CONDITION, MSG) \ + if (Eigen::internal::static_assertion(CONDITION)>::MSG) {} - #endif +#endif - #endif // not CXX0X +#endif// not CXX0X -#else // EIGEN_NO_STATIC_ASSERT +#else// EIGEN_NO_STATIC_ASSERT - #define EIGEN_STATIC_ASSERT(CONDITION,MSG) eigen_assert((CONDITION) && #MSG); +#define EIGEN_STATIC_ASSERT(CONDITION, MSG) eigen_assert((CONDITION) && #MSG); -#endif // EIGEN_NO_STATIC_ASSERT -#endif // EIGEN_STATIC_ASSERT +#endif// EIGEN_NO_STATIC_ASSERT +#endif// EIGEN_STATIC_ASSERT // static assertion failing if the type \a TYPE is not a vector type #define EIGEN_STATIC_ASSERT_VECTOR_ONLY(TYPE) \ - EIGEN_STATIC_ASSERT(TYPE::IsVectorAtCompileTime, \ - YOU_TRIED_CALLING_A_VECTOR_METHOD_ON_A_MATRIX) + EIGEN_STATIC_ASSERT(TYPE::IsVectorAtCompileTime, YOU_TRIED_CALLING_A_VECTOR_METHOD_ON_A_MATRIX) // static assertion failing if the type \a TYPE is not fixed-size #define EIGEN_STATIC_ASSERT_FIXED_SIZE(TYPE) \ - EIGEN_STATIC_ASSERT(TYPE::SizeAtCompileTime!=Eigen::Dynamic, \ - YOU_CALLED_A_FIXED_SIZE_METHOD_ON_A_DYNAMIC_SIZE_MATRIX_OR_VECTOR) + EIGEN_STATIC_ASSERT( \ + TYPE::SizeAtCompileTime != Eigen::Dynamic, YOU_CALLED_A_FIXED_SIZE_METHOD_ON_A_DYNAMIC_SIZE_MATRIX_OR_VECTOR) // static assertion failing if the type \a TYPE is not dynamic-size #define EIGEN_STATIC_ASSERT_DYNAMIC_SIZE(TYPE) \ - EIGEN_STATIC_ASSERT(TYPE::SizeAtCompileTime==Eigen::Dynamic, \ - YOU_CALLED_A_DYNAMIC_SIZE_METHOD_ON_A_FIXED_SIZE_MATRIX_OR_VECTOR) + EIGEN_STATIC_ASSERT( \ + TYPE::SizeAtCompileTime == Eigen::Dynamic, YOU_CALLED_A_DYNAMIC_SIZE_METHOD_ON_A_FIXED_SIZE_MATRIX_OR_VECTOR) // static assertion failing if the type \a TYPE is not a vector type of the given size #define EIGEN_STATIC_ASSERT_VECTOR_SPECIFIC_SIZE(TYPE, SIZE) \ - EIGEN_STATIC_ASSERT(TYPE::IsVectorAtCompileTime && TYPE::SizeAtCompileTime==SIZE, \ - THIS_METHOD_IS_ONLY_FOR_VECTORS_OF_A_SPECIFIC_SIZE) + EIGEN_STATIC_ASSERT( \ + TYPE::IsVectorAtCompileTime &&TYPE::SizeAtCompileTime == SIZE, THIS_METHOD_IS_ONLY_FOR_VECTORS_OF_A_SPECIFIC_SIZE) // static assertion failing if the type \a TYPE is not a vector type of the given size -#define EIGEN_STATIC_ASSERT_MATRIX_SPECIFIC_SIZE(TYPE, ROWS, COLS) \ - EIGEN_STATIC_ASSERT(TYPE::RowsAtCompileTime==ROWS && TYPE::ColsAtCompileTime==COLS, \ - THIS_METHOD_IS_ONLY_FOR_MATRICES_OF_A_SPECIFIC_SIZE) +#define EIGEN_STATIC_ASSERT_MATRIX_SPECIFIC_SIZE(TYPE, ROWS, COLS) \ + EIGEN_STATIC_ASSERT(TYPE::RowsAtCompileTime == ROWS && TYPE::ColsAtCompileTime == COLS, \ + THIS_METHOD_IS_ONLY_FOR_MATRICES_OF_A_SPECIFIC_SIZE) // static assertion failing if the two vector expression types are not compatible (same fixed-size or dynamic size) -#define EIGEN_STATIC_ASSERT_SAME_VECTOR_SIZE(TYPE0,TYPE1) \ - EIGEN_STATIC_ASSERT( \ - (int(TYPE0::SizeAtCompileTime)==Eigen::Dynamic \ - || int(TYPE1::SizeAtCompileTime)==Eigen::Dynamic \ - || int(TYPE0::SizeAtCompileTime)==int(TYPE1::SizeAtCompileTime)),\ +#define EIGEN_STATIC_ASSERT_SAME_VECTOR_SIZE(TYPE0, TYPE1) \ + EIGEN_STATIC_ASSERT( \ + (int(TYPE0::SizeAtCompileTime) == Eigen::Dynamic || int(TYPE1::SizeAtCompileTime) == Eigen::Dynamic \ + || int(TYPE0::SizeAtCompileTime) == int(TYPE1::SizeAtCompileTime)), \ YOU_MIXED_VECTORS_OF_DIFFERENT_SIZES) -#define EIGEN_PREDICATE_SAME_MATRIX_SIZE(TYPE0,TYPE1) \ - ( \ - (int(Eigen::internal::size_of_xpr_at_compile_time::ret)==0 && int(Eigen::internal::size_of_xpr_at_compile_time::ret)==0) \ - || (\ - (int(TYPE0::RowsAtCompileTime)==Eigen::Dynamic \ - || int(TYPE1::RowsAtCompileTime)==Eigen::Dynamic \ - || int(TYPE0::RowsAtCompileTime)==int(TYPE1::RowsAtCompileTime)) \ - && (int(TYPE0::ColsAtCompileTime)==Eigen::Dynamic \ - || int(TYPE1::ColsAtCompileTime)==Eigen::Dynamic \ - || int(TYPE0::ColsAtCompileTime)==int(TYPE1::ColsAtCompileTime))\ - ) \ - ) +#define EIGEN_PREDICATE_SAME_MATRIX_SIZE(TYPE0, TYPE1) \ + ((int(Eigen::internal::size_of_xpr_at_compile_time::ret) == 0 \ + && int(Eigen::internal::size_of_xpr_at_compile_time::ret) == 0) \ + || ((int(TYPE0::RowsAtCompileTime) == Eigen::Dynamic || int(TYPE1::RowsAtCompileTime) == Eigen::Dynamic \ + || int(TYPE0::RowsAtCompileTime) == int(TYPE1::RowsAtCompileTime)) \ + && (int(TYPE0::ColsAtCompileTime) == Eigen::Dynamic || int(TYPE1::ColsAtCompileTime) == Eigen::Dynamic \ + || int(TYPE0::ColsAtCompileTime) == int(TYPE1::ColsAtCompileTime)))) #define EIGEN_STATIC_ASSERT_NON_INTEGER(TYPE) \ - EIGEN_STATIC_ASSERT(!NumTraits::IsInteger, THIS_FUNCTION_IS_NOT_FOR_INTEGER_NUMERIC_TYPES) + EIGEN_STATIC_ASSERT(!NumTraits::IsInteger, THIS_FUNCTION_IS_NOT_FOR_INTEGER_NUMERIC_TYPES) -// static assertion failing if it is guaranteed at compile-time that the two matrix expression types have different sizes -#define EIGEN_STATIC_ASSERT_SAME_MATRIX_SIZE(TYPE0,TYPE1) \ - EIGEN_STATIC_ASSERT( \ - EIGEN_PREDICATE_SAME_MATRIX_SIZE(TYPE0,TYPE1),\ - YOU_MIXED_MATRICES_OF_DIFFERENT_SIZES) +// static assertion failing if it is guaranteed at compile-time that the two matrix expression types have different +// sizes +#define EIGEN_STATIC_ASSERT_SAME_MATRIX_SIZE(TYPE0, TYPE1) \ + EIGEN_STATIC_ASSERT(EIGEN_PREDICATE_SAME_MATRIX_SIZE(TYPE0, TYPE1), YOU_MIXED_MATRICES_OF_DIFFERENT_SIZES) -#define EIGEN_STATIC_ASSERT_SIZE_1x1(TYPE) \ - EIGEN_STATIC_ASSERT((TYPE::RowsAtCompileTime == 1 || TYPE::RowsAtCompileTime == Dynamic) && \ - (TYPE::ColsAtCompileTime == 1 || TYPE::ColsAtCompileTime == Dynamic), \ - THIS_METHOD_IS_ONLY_FOR_1x1_EXPRESSIONS) +#define EIGEN_STATIC_ASSERT_SIZE_1x1(TYPE) \ + EIGEN_STATIC_ASSERT((TYPE::RowsAtCompileTime == 1 || TYPE::RowsAtCompileTime == Dynamic) \ + && (TYPE::ColsAtCompileTime == 1 || TYPE::ColsAtCompileTime == Dynamic), \ + THIS_METHOD_IS_ONLY_FOR_1x1_EXPRESSIONS) #define EIGEN_STATIC_ASSERT_LVALUE(Derived) \ - EIGEN_STATIC_ASSERT(Eigen::internal::is_lvalue::value, \ - THIS_EXPRESSION_IS_NOT_A_LVALUE__IT_IS_READ_ONLY) + EIGEN_STATIC_ASSERT(Eigen::internal::is_lvalue::value, THIS_EXPRESSION_IS_NOT_A_LVALUE__IT_IS_READ_ONLY) -#define EIGEN_STATIC_ASSERT_ARRAYXPR(Derived) \ - EIGEN_STATIC_ASSERT((Eigen::internal::is_same::XprKind, ArrayXpr>::value), \ - THIS_METHOD_IS_ONLY_FOR_ARRAYS_NOT_MATRICES) +#define EIGEN_STATIC_ASSERT_ARRAYXPR(Derived) \ + EIGEN_STATIC_ASSERT((Eigen::internal::is_same::XprKind, ArrayXpr>::value), \ + THIS_METHOD_IS_ONLY_FOR_ARRAYS_NOT_MATRICES) -#define EIGEN_STATIC_ASSERT_SAME_XPR_KIND(Derived1, Derived2) \ - EIGEN_STATIC_ASSERT((Eigen::internal::is_same::XprKind, \ - typename Eigen::internal::traits::XprKind \ - >::value), \ - YOU_CANNOT_MIX_ARRAYS_AND_MATRICES) +#define EIGEN_STATIC_ASSERT_SAME_XPR_KIND(Derived1, Derived2) \ + EIGEN_STATIC_ASSERT((Eigen::internal::is_same::XprKind, \ + typename Eigen::internal::traits::XprKind>::value), \ + YOU_CANNOT_MIX_ARRAYS_AND_MATRICES) // Check that a cost value is positive, and that is stay within a reasonable range // TODO this check could be enabled for internal debugging only #define EIGEN_INTERNAL_CHECK_COST_VALUE(C) \ - EIGEN_STATIC_ASSERT((C)>=0 && (C)<=HugeCost*HugeCost, EIGEN_INTERNAL_ERROR_PLEASE_FILE_A_BUG_REPORT__INVALID_COST_VALUE); + EIGEN_STATIC_ASSERT( \ + (C) >= 0 && (C) <= HugeCost * HugeCost, EIGEN_INTERNAL_ERROR_PLEASE_FILE_A_BUG_REPORT__INVALID_COST_VALUE); -#endif // EIGEN_STATIC_ASSERT_H +#endif// EIGEN_STATIC_ASSERT_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/util/XprHelper.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/util/XprHelper.h index ba5bd186..b25076bc 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/util/XprHelper.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Core/util/XprHelper.h @@ -14,726 +14,839 @@ // just a workaround because GCC seems to not really like empty structs // FIXME: gcc 4.3 generates bad code when strict-aliasing is enabled // so currently we simply disable this optimization for gcc 4.3 -#if EIGEN_COMP_GNUC && !EIGEN_GNUC_AT(4,3) - #define EIGEN_EMPTY_STRUCT_CTOR(X) \ - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE X() {} \ - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE X(const X& ) {} +#if EIGEN_COMP_GNUC && !EIGEN_GNUC_AT(4, 3) +#define EIGEN_EMPTY_STRUCT_CTOR(X) \ + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE X() {} \ + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE X(const X &) {} #else - #define EIGEN_EMPTY_STRUCT_CTOR(X) +#define EIGEN_EMPTY_STRUCT_CTOR(X) #endif namespace Eigen { namespace internal { -template -EIGEN_DEVICE_FUNC -inline IndexDest convert_index(const IndexSrc& idx) { - // for sizeof(IndexDest)>=sizeof(IndexSrc) compilers should be able to optimize this away: - eigen_internal_assert(idx <= NumTraits::highest() && "Index value to big for target type"); - return IndexDest(idx); -} - - -// promote_scalar_arg is an helper used in operation between an expression and a scalar, like: -// expression * scalar -// Its role is to determine how the type T of the scalar operand should be promoted given the scalar type ExprScalar of the given expression. -// The IsSupported template parameter must be provided by the caller as: internal::has_ReturnType >::value using the proper order for ExprScalar and T. -// Then the logic is as follows: -// - if the operation is natively supported as defined by IsSupported, then the scalar type is not promoted, and T is returned. -// - otherwise, NumTraits::Literal is returned if T is implicitly convertible to NumTraits::Literal AND that this does not imply a float to integer conversion. -// - otherwise, ExprScalar is returned if T is implicitly convertible to ExprScalar AND that this does not imply a float to integer conversion. -// - In all other cases, the promoted type is not defined, and the respective operation is thus invalid and not available (SFINAE). -template -struct promote_scalar_arg; - -template -struct promote_scalar_arg -{ - typedef T type; -}; - -// Recursively check safe conversion to PromotedType, and then ExprScalar if they are different. -template::value, - bool IsSafe = NumTraits::IsInteger || !NumTraits::IsInteger> -struct promote_scalar_arg_unsupported; + template EIGEN_DEVICE_FUNC inline IndexDest convert_index(const IndexSrc &idx) + { + // for sizeof(IndexDest)>=sizeof(IndexSrc) compilers should be able to optimize this away: + eigen_internal_assert(idx <= NumTraits::highest() && "Index value to big for target type"); + return IndexDest(idx); + } + + + // promote_scalar_arg is an helper used in operation between an expression and a scalar, like: + // expression * scalar + // Its role is to determine how the type T of the scalar operand should be promoted given the scalar type ExprScalar + // of the given expression. The IsSupported template parameter must be provided by the caller as: + // internal::has_ReturnType >::value using the proper order for ExprScalar and + // T. Then the logic is as follows: + // - if the operation is natively supported as defined by IsSupported, then the scalar type is not promoted, and T is + // returned. + // - otherwise, NumTraits::Literal is returned if T is implicitly convertible to + // NumTraits::Literal AND that this does not imply a float to integer conversion. + // - otherwise, ExprScalar is returned if T is implicitly convertible to ExprScalar AND that this does not imply a + // float to integer conversion. + // - In all other cases, the promoted type is not defined, and the respective operation is thus invalid and not + // available (SFINAE). + template struct promote_scalar_arg; + + template struct promote_scalar_arg + { + typedef T type; + }; -// Start recursion with NumTraits::Literal -template -struct promote_scalar_arg : promote_scalar_arg_unsupported::Literal> {}; + // Recursively check safe conversion to PromotedType, and then ExprScalar if they are different. + template::value, + bool IsSafe = NumTraits::IsInteger || !NumTraits::IsInteger> + struct promote_scalar_arg_unsupported; + + // Start recursion with NumTraits::Literal + template + struct promote_scalar_arg : promote_scalar_arg_unsupported::Literal> + { + }; -// We found a match! -template -struct promote_scalar_arg_unsupported -{ - typedef PromotedType type; -}; + // We found a match! + template + struct promote_scalar_arg_unsupported + { + typedef PromotedType type; + }; -// No match, but no real-to-integer issues, and ExprScalar and current PromotedType are different, -// so let's try to promote to ExprScalar -template -struct promote_scalar_arg_unsupported - : promote_scalar_arg_unsupported -{}; + // No match, but no real-to-integer issues, and ExprScalar and current PromotedType are different, + // so let's try to promote to ExprScalar + template + struct promote_scalar_arg_unsupported + : promote_scalar_arg_unsupported + { + }; -// Unsafe real-to-integer, let's stop. -template -struct promote_scalar_arg_unsupported {}; + // Unsafe real-to-integer, let's stop. + template + struct promote_scalar_arg_unsupported + { + }; -// T is not even convertible to ExprScalar, let's stop. -template -struct promote_scalar_arg_unsupported {}; + // T is not even convertible to ExprScalar, let's stop. + template struct promote_scalar_arg_unsupported + { + }; -//classes inheriting no_assignment_operator don't generate a default operator=. -class no_assignment_operator -{ + // classes inheriting no_assignment_operator don't generate a default operator=. + class no_assignment_operator + { private: - no_assignment_operator& operator=(const no_assignment_operator&); -}; + no_assignment_operator &operator=(const no_assignment_operator &); + }; -/** \internal return the index type with the largest number of bits */ -template -struct promote_index_type -{ - typedef typename conditional<(sizeof(I1)::type type; -}; + /** \internal return the index type with the largest number of bits */ + template struct promote_index_type + { + typedef typename conditional<(sizeof(I1) < sizeof(I2)), I2, I1>::type type; + }; -/** \internal If the template parameter Value is Dynamic, this class is just a wrapper around a T variable that - * can be accessed using value() and setValue(). - * Otherwise, this class is an empty structure and value() just returns the template parameter Value. - */ -template class variable_if_dynamic -{ + /** \internal If the template parameter Value is Dynamic, this class is just a wrapper around a T variable that + * can be accessed using value() and setValue(). + * Otherwise, this class is an empty structure and value() just returns the template parameter Value. + */ + template class variable_if_dynamic + { public: EIGEN_EMPTY_STRUCT_CTOR(variable_if_dynamic) - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE explicit variable_if_dynamic(T v) { EIGEN_ONLY_USED_FOR_DEBUG(v); eigen_assert(v == T(Value)); } + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE explicit variable_if_dynamic(T v) + { + EIGEN_ONLY_USED_FOR_DEBUG(v); + eigen_assert(v == T(Value)); + } EIGEN_DEVICE_FUNC static EIGEN_STRONG_INLINE T value() { return T(Value); } EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void setValue(T) {} -}; + }; -template class variable_if_dynamic -{ + template class variable_if_dynamic + { T m_value; EIGEN_DEVICE_FUNC variable_if_dynamic() { eigen_assert(false); } + public: EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE explicit variable_if_dynamic(T value) : m_value(value) {} EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE T value() const { return m_value; } EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void setValue(T value) { m_value = value; } -}; + }; -/** \internal like variable_if_dynamic but for DynamicIndex - */ -template class variable_if_dynamicindex -{ + /** \internal like variable_if_dynamic but for DynamicIndex + */ + template class variable_if_dynamicindex + { public: EIGEN_EMPTY_STRUCT_CTOR(variable_if_dynamicindex) - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE explicit variable_if_dynamicindex(T v) { EIGEN_ONLY_USED_FOR_DEBUG(v); eigen_assert(v == T(Value)); } + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE explicit variable_if_dynamicindex(T v) + { + EIGEN_ONLY_USED_FOR_DEBUG(v); + eigen_assert(v == T(Value)); + } EIGEN_DEVICE_FUNC static EIGEN_STRONG_INLINE T value() { return T(Value); } EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void setValue(T) {} -}; + }; -template class variable_if_dynamicindex -{ + template class variable_if_dynamicindex + { T m_value; EIGEN_DEVICE_FUNC variable_if_dynamicindex() { eigen_assert(false); } + public: EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE explicit variable_if_dynamicindex(T value) : m_value(value) {} EIGEN_DEVICE_FUNC T EIGEN_STRONG_INLINE value() const { return m_value; } EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void setValue(T value) { m_value = value; } -}; + }; -template struct functor_traits -{ - enum + template struct functor_traits { - Cost = 10, - PacketAccess = false, - IsRepeatable = false + enum { Cost = 10, PacketAccess = false, IsRepeatable = false }; }; -}; -template struct packet_traits; + template struct packet_traits; -template struct unpacket_traits -{ - typedef T type; - typedef T half; - enum + template struct unpacket_traits { - size = 1, - alignment = 1 + typedef T type; + typedef T half; + enum { size = 1, alignment = 1 }; }; -}; -template::size)==0 || is_same::half>::value> -struct find_best_packet_helper; + template::size) == 0 + || is_same::half>::value> + struct find_best_packet_helper; -template< int Size, typename PacketType> -struct find_best_packet_helper -{ - typedef PacketType type; -}; + template struct find_best_packet_helper + { + typedef PacketType type; + }; -template -struct find_best_packet_helper -{ - typedef typename find_best_packet_helper::half>::type type; -}; + template struct find_best_packet_helper + { + typedef typename find_best_packet_helper::half>::type type; + }; -template -struct find_best_packet -{ - typedef typename find_best_packet_helper::type>::type type; -}; + template struct find_best_packet + { + typedef typename find_best_packet_helper::type>::type type; + }; -#if EIGEN_MAX_STATIC_ALIGN_BYTES>0 -template -struct compute_default_alignment_helper -{ - enum { value = 0 }; -}; +#if EIGEN_MAX_STATIC_ALIGN_BYTES > 0 + template + struct compute_default_alignment_helper + { + enum { value = 0 }; + }; -template -struct compute_default_alignment_helper // Match -{ - enum { value = AlignmentBytes }; -}; + template + struct compute_default_alignment_helper// Match + { + enum { value = AlignmentBytes }; + }; -template -struct compute_default_alignment_helper // Try-half -{ - // current packet too large, try with an half-packet - enum { value = compute_default_alignment_helper::value }; -}; + template + struct compute_default_alignment_helper// Try-half + { + // current packet too large, try with an half-packet + enum { value = compute_default_alignment_helper::value }; + }; #else -// If static alignment is disabled, no need to bother. -// This also avoids a division by zero in "bool Match = bool((ArrayBytes%AlignmentBytes)==0)" -template -struct compute_default_alignment_helper -{ - enum { value = 0 }; -}; + // If static alignment is disabled, no need to bother. + // This also avoids a division by zero in "bool Match = bool((ArrayBytes%AlignmentBytes)==0)" + template struct compute_default_alignment_helper + { + enum { value = 0 }; + }; #endif -template struct compute_default_alignment { - enum { value = compute_default_alignment_helper::value }; -}; + template struct compute_default_alignment + { + enum { value = compute_default_alignment_helper::value }; + }; -template struct compute_default_alignment { - enum { value = EIGEN_MAX_ALIGN_BYTES }; -}; + template struct compute_default_alignment + { + enum { value = EIGEN_MAX_ALIGN_BYTES }; + }; -template class make_proper_matrix_type -{ + template + class make_proper_matrix_type + { enum { - IsColVector = _Cols==1 && _Rows!=1, - IsRowVector = _Rows==1 && _Cols!=1, - Options = IsColVector ? (_Options | ColMajor) & ~RowMajor - : IsRowVector ? (_Options | RowMajor) & ~ColMajor - : _Options + IsColVector = _Cols == 1 && _Rows != 1, + IsRowVector = _Rows == 1 && _Cols != 1, + Options = IsColVector ? (_Options | ColMajor) & ~RowMajor + : IsRowVector ? (_Options | RowMajor) & ~ColMajor + : _Options }; + public: typedef Matrix<_Scalar, _Rows, _Cols, Options, _MaxRows, _MaxCols> type; -}; + }; + + template class compute_matrix_flags + { + enum { row_major_bit = Options & RowMajor ? RowMajorBit : 0 }; -template -class compute_matrix_flags -{ - enum { row_major_bit = Options&RowMajor ? RowMajorBit : 0 }; public: // FIXME currently we still have to handle DirectAccessBit at the expression level to handle DenseCoeffsBase<> // and then propagate this information to the evaluator's flags. // However, I (Gael) think that DirectAccessBit should only matter at the evaluation stage. enum { ret = DirectAccessBit | LvalueBit | NestByRefBit | row_major_bit }; -}; - -template struct size_at_compile_time -{ - enum { ret = (_Rows==Dynamic || _Cols==Dynamic) ? Dynamic : _Rows * _Cols }; -}; - -template struct size_of_xpr_at_compile_time -{ - enum { ret = size_at_compile_time::RowsAtCompileTime,traits::ColsAtCompileTime>::ret }; -}; + }; -/* plain_matrix_type : the difference from eval is that plain_matrix_type is always a plain matrix type, - * whereas eval is a const reference in the case of a matrix - */ + template struct size_at_compile_time + { + enum { ret = (_Rows == Dynamic || _Cols == Dynamic) ? Dynamic : _Rows * _Cols }; + }; -template::StorageKind> struct plain_matrix_type; -template struct plain_matrix_type_dense; -template struct plain_matrix_type -{ - typedef typename plain_matrix_type_dense::XprKind, traits::Flags>::type type; -}; -template struct plain_matrix_type -{ - typedef typename T::PlainObject type; -}; + template struct size_of_xpr_at_compile_time + { + enum { ret = size_at_compile_time::RowsAtCompileTime, traits::ColsAtCompileTime>::ret }; + }; -template struct plain_matrix_type_dense -{ - typedef Matrix::Scalar, - traits::RowsAtCompileTime, - traits::ColsAtCompileTime, - AutoAlign | (Flags&RowMajorBit ? RowMajor : ColMajor), - traits::MaxRowsAtCompileTime, - traits::MaxColsAtCompileTime - > type; -}; + /* plain_matrix_type : the difference from eval is that plain_matrix_type is always a plain matrix type, + * whereas eval is a const reference in the case of a matrix + */ -template struct plain_matrix_type_dense -{ - typedef Array::Scalar, - traits::RowsAtCompileTime, - traits::ColsAtCompileTime, - AutoAlign | (Flags&RowMajorBit ? RowMajor : ColMajor), - traits::MaxRowsAtCompileTime, - traits::MaxColsAtCompileTime - > type; -}; + template::StorageKind> struct plain_matrix_type; + template struct plain_matrix_type_dense; + template struct plain_matrix_type + { + typedef typename plain_matrix_type_dense::XprKind, traits::Flags>::type type; + }; + template struct plain_matrix_type + { + typedef typename T::PlainObject type; + }; -/* eval : the return type of eval(). For matrices, this is just a const reference - * in order to avoid a useless copy - */ + template struct plain_matrix_type_dense + { + typedef Matrix::Scalar, + traits::RowsAtCompileTime, + traits::ColsAtCompileTime, + AutoAlign | (Flags & RowMajorBit ? RowMajor : ColMajor), + traits::MaxRowsAtCompileTime, + traits::MaxColsAtCompileTime> + type; + }; -template::StorageKind> struct eval; + template struct plain_matrix_type_dense + { + typedef Array::Scalar, + traits::RowsAtCompileTime, + traits::ColsAtCompileTime, + AutoAlign | (Flags & RowMajorBit ? RowMajor : ColMajor), + traits::MaxRowsAtCompileTime, + traits::MaxColsAtCompileTime> + type; + }; -template struct eval -{ - typedef typename plain_matrix_type::type type; -// typedef typename T::PlainObject type; -// typedef T::Matrix::Scalar, -// traits::RowsAtCompileTime, -// traits::ColsAtCompileTime, -// AutoAlign | (traits::Flags&RowMajorBit ? RowMajor : ColMajor), -// traits::MaxRowsAtCompileTime, -// traits::MaxColsAtCompileTime -// > type; -}; + /* eval : the return type of eval(). For matrices, this is just a const reference + * in order to avoid a useless copy + */ -template struct eval -{ - typedef typename plain_matrix_type::type type; -}; + template::StorageKind> struct eval; -// for matrices, no need to evaluate, just use a const reference to avoid a useless copy -template -struct eval, Dense> -{ - typedef const Matrix<_Scalar, _Rows, _Cols, _Options, _MaxRows, _MaxCols>& type; -}; + template struct eval + { + typedef typename plain_matrix_type::type type; + // typedef typename T::PlainObject type; + // typedef T::Matrix::Scalar, + // traits::RowsAtCompileTime, + // traits::ColsAtCompileTime, + // AutoAlign | (traits::Flags&RowMajorBit ? RowMajor : ColMajor), + // traits::MaxRowsAtCompileTime, + // traits::MaxColsAtCompileTime + // > type; + }; -template -struct eval, Dense> -{ - typedef const Array<_Scalar, _Rows, _Cols, _Options, _MaxRows, _MaxCols>& type; -}; + template struct eval + { + typedef typename plain_matrix_type::type type; + }; + // for matrices, no need to evaluate, just use a const reference to avoid a useless copy + template + struct eval, Dense> + { + typedef const Matrix<_Scalar, _Rows, _Cols, _Options, _MaxRows, _MaxCols> &type; + }; -/* similar to plain_matrix_type, but using the evaluator's Flags */ -template::StorageKind> struct plain_object_eval; + template + struct eval, Dense> + { + typedef const Array<_Scalar, _Rows, _Cols, _Options, _MaxRows, _MaxCols> &type; + }; -template -struct plain_object_eval -{ - typedef typename plain_matrix_type_dense::XprKind, evaluator::Flags>::type type; -}; + /* similar to plain_matrix_type, but using the evaluator's Flags */ + template::StorageKind> struct plain_object_eval; -/* plain_matrix_type_column_major : same as plain_matrix_type but guaranteed to be column-major - */ -template struct plain_matrix_type_column_major -{ - enum { Rows = traits::RowsAtCompileTime, - Cols = traits::ColsAtCompileTime, - MaxRows = traits::MaxRowsAtCompileTime, - MaxCols = traits::MaxColsAtCompileTime - }; - typedef Matrix::Scalar, - Rows, - Cols, - (MaxRows==1&&MaxCols!=1) ? RowMajor : ColMajor, - MaxRows, - MaxCols - > type; -}; + template struct plain_object_eval + { + typedef typename plain_matrix_type_dense::XprKind, evaluator::Flags>::type type; + }; -/* plain_matrix_type_row_major : same as plain_matrix_type but guaranteed to be row-major - */ -template struct plain_matrix_type_row_major -{ - enum { Rows = traits::RowsAtCompileTime, - Cols = traits::ColsAtCompileTime, - MaxRows = traits::MaxRowsAtCompileTime, - MaxCols = traits::MaxColsAtCompileTime - }; - typedef Matrix::Scalar, - Rows, - Cols, - (MaxCols==1&&MaxRows!=1) ? RowMajor : ColMajor, - MaxRows, - MaxCols - > type; -}; -/** \internal The reference selector for template expressions. The idea is that we don't - * need to use references for expressions since they are light weight proxy - * objects which should generate no copying overhead. */ -template -struct ref_selector -{ - typedef typename conditional< - bool(traits::Flags & NestByRefBit), - T const&, - const T - >::type type; - - typedef typename conditional< - bool(traits::Flags & NestByRefBit), - T &, - T - >::type non_const_type; -}; + /* plain_matrix_type_column_major : same as plain_matrix_type but guaranteed to be column-major + */ + template struct plain_matrix_type_column_major + { + enum { + Rows = traits::RowsAtCompileTime, + Cols = traits::ColsAtCompileTime, + MaxRows = traits::MaxRowsAtCompileTime, + MaxCols = traits::MaxColsAtCompileTime + }; + typedef Matrix::Scalar, + Rows, + Cols, + (MaxRows == 1 && MaxCols != 1) ? RowMajor : ColMajor, + MaxRows, + MaxCols> + type; + }; -/** \internal Adds the const qualifier on the value-type of T2 if and only if T1 is a const type */ -template -struct transfer_constness -{ - typedef typename conditional< - bool(internal::is_const::value), - typename internal::add_const_on_value_type::type, - T2 - >::type type; -}; + /* plain_matrix_type_row_major : same as plain_matrix_type but guaranteed to be row-major + */ + template struct plain_matrix_type_row_major + { + enum { + Rows = traits::RowsAtCompileTime, + Cols = traits::ColsAtCompileTime, + MaxRows = traits::MaxRowsAtCompileTime, + MaxCols = traits::MaxColsAtCompileTime + }; + typedef Matrix::Scalar, + Rows, + Cols, + (MaxCols == 1 && MaxRows != 1) ? RowMajor : ColMajor, + MaxRows, + MaxCols> + type; + }; + /** \internal The reference selector for template expressions. The idea is that we don't + * need to use references for expressions since they are light weight proxy + * objects which should generate no copying overhead. */ + template struct ref_selector + { + typedef typename conditional::Flags &NestByRefBit), T const &, const T>::type type; -// However, we still need a mechanism to detect whether an expression which is evaluated multiple time -// has to be evaluated into a temporary. -// That's the purpose of this new nested_eval helper: -/** \internal Determines how a given expression should be nested when evaluated multiple times. - * For example, when you do a * (b+c), Eigen will determine how the expression b+c should be - * evaluated into the bigger product expression. The choice is between nesting the expression b+c as-is, or - * evaluating that expression b+c into a temporary variable d, and nest d so that the resulting expression is - * a*d. Evaluating can be beneficial for example if every coefficient access in the resulting expression causes - * many coefficient accesses in the nested expressions -- as is the case with matrix product for example. - * - * \tparam T the type of the expression being nested. - * \tparam n the number of coefficient accesses in the nested expression for each coefficient access in the bigger expression. - * \tparam PlainObject the type of the temporary if needed. - */ -template::type> struct nested_eval -{ - enum { - ScalarReadCost = NumTraits::Scalar>::ReadCost, - CoeffReadCost = evaluator::CoeffReadCost, // NOTE What if an evaluator evaluate itself into a tempory? - // Then CoeffReadCost will be small (e.g., 1) but we still have to evaluate, especially if n>1. - // This situation is already taken care by the EvalBeforeNestingBit flag, which is turned ON - // for all evaluator creating a temporary. This flag is then propagated by the parent evaluators. - // Another solution could be to count the number of temps? - NAsInteger = n == Dynamic ? HugeCost : n, - CostEval = (NAsInteger+1) * ScalarReadCost + CoeffReadCost, - CostNoEval = NAsInteger * CoeffReadCost, - Evaluate = (int(evaluator::Flags) & EvalBeforeNestingBit) || (int(CostEval) < int(CostNoEval)) - }; - - typedef typename conditional::type>::type type; -}; + typedef typename conditional::Flags &NestByRefBit), T &, T>::type non_const_type; + }; -template -EIGEN_DEVICE_FUNC -inline T* const_cast_ptr(const T* ptr) -{ - return const_cast(ptr); -} + /** \internal Adds the const qualifier on the value-type of T2 if and only if T1 is a const type */ + template struct transfer_constness + { + typedef typename conditional::value), + typename internal::add_const_on_value_type::type, + T2>::type type; + }; -template::XprKind> -struct dense_xpr_base -{ - /* dense_xpr_base should only ever be used on dense expressions, thus falling either into the MatrixXpr or into the ArrayXpr cases */ -}; -template -struct dense_xpr_base -{ - typedef MatrixBase type; -}; + // However, we still need a mechanism to detect whether an expression which is evaluated multiple time + // has to be evaluated into a temporary. + // That's the purpose of this new nested_eval helper: + /** \internal Determines how a given expression should be nested when evaluated multiple times. + * For example, when you do a * (b+c), Eigen will determine how the expression b+c should be + * evaluated into the bigger product expression. The choice is between nesting the expression b+c as-is, or + * evaluating that expression b+c into a temporary variable d, and nest d so that the resulting expression is + * a*d. Evaluating can be beneficial for example if every coefficient access in the resulting expression causes + * many coefficient accesses in the nested expressions -- as is the case with matrix product for example. + * + * \tparam T the type of the expression being nested. + * \tparam n the number of coefficient accesses in the nested expression for each coefficient access in the bigger + * expression. + * \tparam PlainObject the type of the temporary if needed. + */ + template::type> struct nested_eval + { + enum { + ScalarReadCost = NumTraits::Scalar>::ReadCost, + CoeffReadCost = + evaluator::CoeffReadCost,// NOTE What if an evaluator evaluate itself into a tempory? + // Then CoeffReadCost will be small (e.g., 1) but we still have to evaluate, + // especially if n>1. This situation is already taken care by the + // EvalBeforeNestingBit flag, which is turned ON for all evaluator creating a + // temporary. This flag is then propagated by the parent evaluators. Another + // solution could be to count the number of temps? + NAsInteger = n == Dynamic ? HugeCost : n, + CostEval = (NAsInteger + 1) * ScalarReadCost + CoeffReadCost, + CostNoEval = NAsInteger * CoeffReadCost, + Evaluate = (int(evaluator::Flags) & EvalBeforeNestingBit) || (int(CostEval) < int(CostNoEval)) + }; -template -struct dense_xpr_base -{ - typedef ArrayBase type; -}; + typedef typename conditional::type>::type type; + }; -template::XprKind, typename StorageKind = typename traits::StorageKind> -struct generic_xpr_base; + template EIGEN_DEVICE_FUNC inline T *const_cast_ptr(const T *ptr) { return const_cast(ptr); } -template -struct generic_xpr_base -{ - typedef typename dense_xpr_base::type type; -}; + template::XprKind> struct dense_xpr_base + { + /* dense_xpr_base should only ever be used on dense expressions, thus falling either into the MatrixXpr or into the + * ArrayXpr cases */ + }; -template struct cast_return_type -{ - typedef typename XprType::Scalar CurrentScalarType; - typedef typename remove_all::type _CastType; - typedef typename _CastType::Scalar NewScalarType; - typedef typename conditional::value, - const XprType&,CastType>::type type; -}; + template struct dense_xpr_base + { + typedef MatrixBase type; + }; -template struct promote_storage_type; + template struct dense_xpr_base + { + typedef ArrayBase type; + }; -template struct promote_storage_type -{ - typedef A ret; -}; -template struct promote_storage_type -{ - typedef A ret; -}; -template struct promote_storage_type -{ - typedef A ret; -}; + template::XprKind, + typename StorageKind = typename traits::StorageKind> + struct generic_xpr_base; -/** \internal Specify the "storage kind" of applying a coefficient-wise - * binary operations between two expressions of kinds A and B respectively. - * The template parameter Functor permits to specialize the resulting storage kind wrt to - * the functor. - * The default rules are as follows: - * \code - * A op A -> A - * A op dense -> dense - * dense op B -> dense - * sparse op dense -> sparse - * dense op sparse -> sparse - * \endcode - */ -template struct cwise_promote_storage_type; + template struct generic_xpr_base + { + typedef typename dense_xpr_base::type type; + }; -template struct cwise_promote_storage_type { typedef A ret; }; -template struct cwise_promote_storage_type { typedef Dense ret; }; -template struct cwise_promote_storage_type { typedef Dense ret; }; -template struct cwise_promote_storage_type { typedef Dense ret; }; -template struct cwise_promote_storage_type { typedef Sparse ret; }; -template struct cwise_promote_storage_type { typedef Sparse ret; }; + template struct cast_return_type + { + typedef typename XprType::Scalar CurrentScalarType; + typedef typename remove_all::type _CastType; + typedef typename _CastType::Scalar NewScalarType; + typedef + typename conditional::value, const XprType &, CastType>::type type; + }; -template struct cwise_promote_storage_order { - enum { value = LhsOrder }; -}; + template struct promote_storage_type; -template struct cwise_promote_storage_order { enum { value = RhsOrder }; }; -template struct cwise_promote_storage_order { enum { value = LhsOrder }; }; -template struct cwise_promote_storage_order { enum { value = Order }; }; - - -/** \internal Specify the "storage kind" of multiplying an expression of kind A with kind B. - * The template parameter ProductTag permits to specialize the resulting storage kind wrt to - * some compile-time properties of the product: GemmProduct, GemvProduct, OuterProduct, InnerProduct. - * The default rules are as follows: - * \code - * K * K -> K - * dense * K -> dense - * K * dense -> dense - * diag * K -> K - * K * diag -> K - * Perm * K -> K - * K * Perm -> K - * \endcode - */ -template struct product_promote_storage_type; + template struct promote_storage_type + { + typedef A ret; + }; + template struct promote_storage_type + { + typedef A ret; + }; + template struct promote_storage_type + { + typedef A ret; + }; -template struct product_promote_storage_type { typedef A ret;}; -template struct product_promote_storage_type { typedef Dense ret;}; -template struct product_promote_storage_type { typedef Dense ret; }; -template struct product_promote_storage_type { typedef Dense ret; }; + /** \internal Specify the "storage kind" of applying a coefficient-wise + * binary operations between two expressions of kinds A and B respectively. + * The template parameter Functor permits to specialize the resulting storage kind wrt to + * the functor. + * The default rules are as follows: + * \code + * A op A -> A + * A op dense -> dense + * dense op B -> dense + * sparse op dense -> sparse + * dense op sparse -> sparse + * \endcode + */ + template struct cwise_promote_storage_type; + + template struct cwise_promote_storage_type + { + typedef A ret; + }; + template struct cwise_promote_storage_type + { + typedef Dense ret; + }; + template struct cwise_promote_storage_type + { + typedef Dense ret; + }; + template struct cwise_promote_storage_type + { + typedef Dense ret; + }; + template struct cwise_promote_storage_type + { + typedef Sparse ret; + }; + template struct cwise_promote_storage_type + { + typedef Sparse ret; + }; -template struct product_promote_storage_type { typedef A ret; }; -template struct product_promote_storage_type { typedef B ret; }; -template struct product_promote_storage_type { typedef Dense ret; }; -template struct product_promote_storage_type { typedef Dense ret; }; + template struct cwise_promote_storage_order + { + enum { value = LhsOrder }; + }; -template struct product_promote_storage_type { typedef A ret; }; -template struct product_promote_storage_type { typedef B ret; }; -template struct product_promote_storage_type { typedef Dense ret; }; -template struct product_promote_storage_type { typedef Dense ret; }; + template + struct cwise_promote_storage_order + { + enum { value = RhsOrder }; + }; + template + struct cwise_promote_storage_order + { + enum { value = LhsOrder }; + }; + template struct cwise_promote_storage_order + { + enum { value = Order }; + }; -/** \internal gives the plain matrix or array type to store a row/column/diagonal of a matrix type. - * \tparam Scalar optional parameter allowing to pass a different scalar type than the one of the MatrixType. - */ -template -struct plain_row_type -{ - typedef Matrix MatrixRowType; - typedef Array ArrayRowType; - - typedef typename conditional< - is_same< typename traits::XprKind, MatrixXpr >::value, - MatrixRowType, - ArrayRowType - >::type type; -}; -template -struct plain_col_type -{ - typedef Matrix MatrixColType; - typedef Array ArrayColType; - - typedef typename conditional< - is_same< typename traits::XprKind, MatrixXpr >::value, - MatrixColType, - ArrayColType - >::type type; -}; + /** \internal Specify the "storage kind" of multiplying an expression of kind A with kind B. + * The template parameter ProductTag permits to specialize the resulting storage kind wrt to + * some compile-time properties of the product: GemmProduct, GemvProduct, OuterProduct, InnerProduct. + * The default rules are as follows: + * \code + * K * K -> K + * dense * K -> dense + * K * dense -> dense + * diag * K -> K + * K * diag -> K + * Perm * K -> K + * K * Perm -> K + * \endcode + */ + template struct product_promote_storage_type; + + template struct product_promote_storage_type + { + typedef A ret; + }; + template struct product_promote_storage_type + { + typedef Dense ret; + }; + template struct product_promote_storage_type + { + typedef Dense ret; + }; + template struct product_promote_storage_type + { + typedef Dense ret; + }; -template -struct plain_diag_type -{ - enum { diag_size = EIGEN_SIZE_MIN_PREFER_DYNAMIC(ExpressionType::RowsAtCompileTime, ExpressionType::ColsAtCompileTime), - max_diag_size = EIGEN_SIZE_MIN_PREFER_FIXED(ExpressionType::MaxRowsAtCompileTime, ExpressionType::MaxColsAtCompileTime) + template struct product_promote_storage_type + { + typedef A ret; + }; + template struct product_promote_storage_type + { + typedef B ret; + }; + template struct product_promote_storage_type + { + typedef Dense ret; + }; + template struct product_promote_storage_type + { + typedef Dense ret; }; - typedef Matrix MatrixDiagType; - typedef Array ArrayDiagType; - typedef typename conditional< - is_same< typename traits::XprKind, MatrixXpr >::value, - MatrixDiagType, - ArrayDiagType - >::type type; -}; + template struct product_promote_storage_type + { + typedef A ret; + }; + template struct product_promote_storage_type + { + typedef B ret; + }; + template struct product_promote_storage_type + { + typedef Dense ret; + }; + template struct product_promote_storage_type + { + typedef Dense ret; + }; -template -struct plain_constant_type -{ - enum { Options = (traits::Flags&RowMajorBit)?RowMajor:0 }; + /** \internal gives the plain matrix or array type to store a row/column/diagonal of a matrix type. + * \tparam Scalar optional parameter allowing to pass a different scalar type than the one of the MatrixType. + */ + template struct plain_row_type + { + typedef Matrix + MatrixRowType; + typedef Array + ArrayRowType; + + typedef typename conditional::XprKind, MatrixXpr>::value, + MatrixRowType, + ArrayRowType>::type type; + }; - typedef Array::RowsAtCompileTime, traits::ColsAtCompileTime, - Options, traits::MaxRowsAtCompileTime,traits::MaxColsAtCompileTime> array_type; + template struct plain_col_type + { + typedef Matrix + MatrixColType; + typedef Array + ArrayColType; + + typedef typename conditional::XprKind, MatrixXpr>::value, + MatrixColType, + ArrayColType>::type type; + }; - typedef Matrix::RowsAtCompileTime, traits::ColsAtCompileTime, - Options, traits::MaxRowsAtCompileTime,traits::MaxColsAtCompileTime> matrix_type; + template struct plain_diag_type + { + enum { + diag_size = EIGEN_SIZE_MIN_PREFER_DYNAMIC(ExpressionType::RowsAtCompileTime, ExpressionType::ColsAtCompileTime), + max_diag_size = + EIGEN_SIZE_MIN_PREFER_FIXED(ExpressionType::MaxRowsAtCompileTime, ExpressionType::MaxColsAtCompileTime) + }; + typedef Matrix + MatrixDiagType; + typedef Array + ArrayDiagType; + + typedef typename conditional::XprKind, MatrixXpr>::value, + MatrixDiagType, + ArrayDiagType>::type type; + }; - typedef CwiseNullaryOp, const typename conditional::XprKind, MatrixXpr >::value, matrix_type, array_type>::type > type; -}; + template struct plain_constant_type + { + enum { Options = (traits::Flags & RowMajorBit) ? RowMajor : 0 }; + + typedef Array::RowsAtCompileTime, + traits::ColsAtCompileTime, + Options, + traits::MaxRowsAtCompileTime, + traits::MaxColsAtCompileTime> + array_type; + + typedef Matrix::RowsAtCompileTime, + traits::ColsAtCompileTime, + Options, + traits::MaxRowsAtCompileTime, + traits::MaxColsAtCompileTime> + matrix_type; + + typedef CwiseNullaryOp, + const typename conditional::XprKind, MatrixXpr>::value, matrix_type, array_type>:: + type> + type; + }; -template -struct is_lvalue -{ - enum { value = (!bool(is_const::value)) && - bool(traits::Flags & LvalueBit) }; -}; + template struct is_lvalue + { + enum { value = (!bool(is_const::value)) && bool(traits::Flags & LvalueBit) }; + }; -template struct is_diagonal -{ enum { ret = false }; }; + template struct is_diagonal + { + enum { ret = false }; + }; -template struct is_diagonal > -{ enum { ret = true }; }; + template struct is_diagonal> + { + enum { ret = true }; + }; -template struct is_diagonal > -{ enum { ret = true }; }; + template struct is_diagonal> + { + enum { ret = true }; + }; -template struct is_diagonal > -{ enum { ret = true }; }; + template struct is_diagonal> + { + enum { ret = true }; + }; -template struct glue_shapes; -template<> struct glue_shapes { typedef TriangularShape type; }; + template struct glue_shapes; + template<> struct glue_shapes + { + typedef TriangularShape type; + }; -template -bool is_same_dense(const T1 &mat1, const T2 &mat2, typename enable_if::ret&&has_direct_access::ret, T1>::type * = 0) -{ - return (mat1.data()==mat2.data()) && (mat1.innerStride()==mat2.innerStride()) && (mat1.outerStride()==mat2.outerStride()); -} + template + bool is_same_dense(const T1 &mat1, + const T2 &mat2, + typename enable_if::ret && has_direct_access::ret, T1>::type * = 0) + { + return (mat1.data() == mat2.data()) && (mat1.innerStride() == mat2.innerStride()) + && (mat1.outerStride() == mat2.outerStride()); + } + + template + bool is_same_dense(const T1 &, + const T2 &, + typename enable_if::ret && has_direct_access::ret), T1>::type * = 0) + { + return false; + } -template -bool is_same_dense(const T1 &, const T2 &, typename enable_if::ret&&has_direct_access::ret), T1>::type * = 0) -{ - return false; -} - -// Internal helper defining the cost of a scalar division for the type T. -// The default heuristic can be specialized for each scalar type and architecture. -template -struct scalar_div_cost { - enum { value = 8*NumTraits::MulCost }; -}; + // Internal helper defining the cost of a scalar division for the type T. + // The default heuristic can be specialized for each scalar type and architecture. + template struct scalar_div_cost + { + enum { value = 8 * NumTraits::MulCost }; + }; -template -struct scalar_div_cost, Vectorized> { - enum { value = 2*scalar_div_cost::value - + 6*NumTraits::MulCost - + 3*NumTraits::AddCost + template struct scalar_div_cost, Vectorized> + { + enum { value = 2 * scalar_div_cost::value + 6 * NumTraits::MulCost + 3 * NumTraits::AddCost }; }; -}; -template -struct scalar_div_cost::type> { enum { value = 24 }; }; -template -struct scalar_div_cost::type> { enum { value = 21 }; }; + template + struct scalar_div_cost::type> + { + enum { value = 24 }; + }; + template + struct scalar_div_cost::type> + { + enum { value = 21 }; + }; #ifdef EIGEN_DEBUG_ASSIGN -std::string demangle_traversal(int t) -{ - if(t==DefaultTraversal) return "DefaultTraversal"; - if(t==LinearTraversal) return "LinearTraversal"; - if(t==InnerVectorizedTraversal) return "InnerVectorizedTraversal"; - if(t==LinearVectorizedTraversal) return "LinearVectorizedTraversal"; - if(t==SliceVectorizedTraversal) return "SliceVectorizedTraversal"; - return "?"; -} -std::string demangle_unrolling(int t) -{ - if(t==NoUnrolling) return "NoUnrolling"; - if(t==InnerUnrolling) return "InnerUnrolling"; - if(t==CompleteUnrolling) return "CompleteUnrolling"; - return "?"; -} -std::string demangle_flags(int f) -{ - std::string res; - if(f&RowMajorBit) res += " | RowMajor"; - if(f&PacketAccessBit) res += " | Packet"; - if(f&LinearAccessBit) res += " | Linear"; - if(f&LvalueBit) res += " | Lvalue"; - if(f&DirectAccessBit) res += " | Direct"; - if(f&NestByRefBit) res += " | NestByRef"; - if(f&NoPreferredStorageOrderBit) res += " | NoPreferredStorageOrderBit"; - - return res; -} + std::string demangle_traversal(int t) + { + if (t == DefaultTraversal) return "DefaultTraversal"; + if (t == LinearTraversal) return "LinearTraversal"; + if (t == InnerVectorizedTraversal) return "InnerVectorizedTraversal"; + if (t == LinearVectorizedTraversal) return "LinearVectorizedTraversal"; + if (t == SliceVectorizedTraversal) return "SliceVectorizedTraversal"; + return "?"; + } + std::string demangle_unrolling(int t) + { + if (t == NoUnrolling) return "NoUnrolling"; + if (t == InnerUnrolling) return "InnerUnrolling"; + if (t == CompleteUnrolling) return "CompleteUnrolling"; + return "?"; + } + std::string demangle_flags(int f) + { + std::string res; + if (f & RowMajorBit) res += " | RowMajor"; + if (f & PacketAccessBit) res += " | Packet"; + if (f & LinearAccessBit) res += " | Linear"; + if (f & LvalueBit) res += " | Lvalue"; + if (f & DirectAccessBit) res += " | Direct"; + if (f & NestByRefBit) res += " | NestByRef"; + if (f & NoPreferredStorageOrderBit) res += " | NoPreferredStorageOrderBit"; + + return res; + } #endif -} // end namespace internal +}// end namespace internal /** \class ScalarBinaryOpTraits * \ingroup Core_Module * - * \brief Determines whether the given binary operation of two numeric types is allowed and what the scalar return type is. + * \brief Determines whether the given binary operation of two numeric types is allowed and what the scalar return type + is. * - * This class permits to control the scalar return type of any binary operation performed on two different scalar types through (partial) template specializations. + * This class permits to control the scalar return type of any binary operation performed on two different scalar types + through (partial) template specializations. * - * For instance, let \c U1, \c U2 and \c U3 be three user defined scalar types for which most operations between instances of \c U1 and \c U2 returns an \c U3. + * For instance, let \c U1, \c U2 and \c U3 be three user defined scalar types for which most operations between + instances of \c U1 and \c U2 returns an \c U3. * You can let %Eigen knows that by defining: \code template @@ -756,66 +869,68 @@ std::string demangle_flags(int f) - - +
ScalarAScalarBBinaryOpReturnTypeNote
\c T \c T \c * \c T
\c NumTraits::Real \c T \c * \c T Only if \c NumTraits::IsComplex
\c T \c NumTraits::Real \c * \c T Only if \c NumTraits::IsComplex
\c NumTraits::Real \c T \c * \c T Only if \c + NumTraits::IsComplex
\c T \c NumTraits::Real \c * \c T + Only if \c NumTraits::IsComplex
* * \sa CwiseBinaryOp */ -template > +template> struct ScalarBinaryOpTraits #ifndef EIGEN_PARSED_BY_DOXYGEN // for backward compatibility, use the hints given by the (deprecated) internal::scalar_product_traits class. - : internal::scalar_product_traits -#endif // EIGEN_PARSED_BY_DOXYGEN -{}; + : internal::scalar_product_traits +#endif// EIGEN_PARSED_BY_DOXYGEN +{ +}; -template -struct ScalarBinaryOpTraits +template struct ScalarBinaryOpTraits { typedef T ReturnType; }; -template -struct ScalarBinaryOpTraits::IsComplex,T>::type>::Real, BinaryOp> +template +struct ScalarBinaryOpTraits::IsComplex, T>::type>::Real, + BinaryOp> { typedef T ReturnType; }; -template -struct ScalarBinaryOpTraits::IsComplex,T>::type>::Real, T, BinaryOp> +template +struct ScalarBinaryOpTraits::IsComplex, T>::type>::Real, + T, + BinaryOp> { typedef T ReturnType; }; // For Matrix * Permutation -template -struct ScalarBinaryOpTraits +template struct ScalarBinaryOpTraits { typedef T ReturnType; }; // For Permutation * Matrix -template -struct ScalarBinaryOpTraits +template struct ScalarBinaryOpTraits { typedef T ReturnType; }; // for Permutation*Permutation -template -struct ScalarBinaryOpTraits +template struct ScalarBinaryOpTraits { typedef void ReturnType; }; // We require Lhs and Rhs to have "compatible" scalar types. -// It is tempting to always allow mixing different types but remember that this is often impossible in the vectorized paths. -// So allowing mixing different types gives very unexpected errors when enabling vectorization, when the user tries to -// add together a float matrix and a double matrix. -#define EIGEN_CHECK_BINARY_COMPATIBILIY(BINOP,LHS,RHS) \ - EIGEN_STATIC_ASSERT((Eigen::internal::has_ReturnType >::value), \ +// It is tempting to always allow mixing different types but remember that this is often impossible in the vectorized +// paths. So allowing mixing different types gives very unexpected errors when enabling vectorization, when the user +// tries to add together a float matrix and a double matrix. +#define EIGEN_CHECK_BINARY_COMPATIBILIY(BINOP, LHS, RHS) \ + EIGEN_STATIC_ASSERT((Eigen::internal::has_ReturnType>::value), \ YOU_MIXED_DIFFERENT_NUMERIC_TYPES__YOU_NEED_TO_USE_THE_CAST_METHOD_OF_MATRIXBASE_TO_CAST_NUMERIC_TYPES_EXPLICITLY) - -} // end namespace Eigen -#endif // EIGEN_XPRHELPER_H +}// end namespace Eigen + +#endif// EIGEN_XPRHELPER_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Eigenvalues/ComplexEigenSolver.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Eigenvalues/ComplexEigenSolver.h index dc5fae06..c25bca4b 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Eigenvalues/ComplexEigenSolver.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Eigenvalues/ComplexEigenSolver.h @@ -14,254 +14,238 @@ #include "./ComplexSchur.h" -namespace Eigen { +namespace Eigen { /** \eigenvalues_module \ingroup Eigenvalues_Module - * - * - * \class ComplexEigenSolver - * - * \brief Computes eigenvalues and eigenvectors of general complex matrices - * - * \tparam _MatrixType the type of the matrix of which we are - * computing the eigendecomposition; this is expected to be an - * instantiation of the Matrix class template. - * - * The eigenvalues and eigenvectors of a matrix \f$ A \f$ are scalars - * \f$ \lambda \f$ and vectors \f$ v \f$ such that \f$ Av = \lambda v - * \f$. If \f$ D \f$ is a diagonal matrix with the eigenvalues on - * the diagonal, and \f$ V \f$ is a matrix with the eigenvectors as - * its columns, then \f$ A V = V D \f$. The matrix \f$ V \f$ is - * almost always invertible, in which case we have \f$ A = V D V^{-1} - * \f$. This is called the eigendecomposition. - * - * The main function in this class is compute(), which computes the - * eigenvalues and eigenvectors of a given function. The - * documentation for that function contains an example showing the - * main features of the class. - * - * \sa class EigenSolver, class SelfAdjointEigenSolver - */ + * + * + * \class ComplexEigenSolver + * + * \brief Computes eigenvalues and eigenvectors of general complex matrices + * + * \tparam _MatrixType the type of the matrix of which we are + * computing the eigendecomposition; this is expected to be an + * instantiation of the Matrix class template. + * + * The eigenvalues and eigenvectors of a matrix \f$ A \f$ are scalars + * \f$ \lambda \f$ and vectors \f$ v \f$ such that \f$ Av = \lambda v + * \f$. If \f$ D \f$ is a diagonal matrix with the eigenvalues on + * the diagonal, and \f$ V \f$ is a matrix with the eigenvectors as + * its columns, then \f$ A V = V D \f$. The matrix \f$ V \f$ is + * almost always invertible, in which case we have \f$ A = V D V^{-1} + * \f$. This is called the eigendecomposition. + * + * The main function in this class is compute(), which computes the + * eigenvalues and eigenvectors of a given function. The + * documentation for that function contains an example showing the + * main features of the class. + * + * \sa class EigenSolver, class SelfAdjointEigenSolver + */ template class ComplexEigenSolver { - public: - - /** \brief Synonym for the template parameter \p _MatrixType. */ - typedef _MatrixType MatrixType; - - enum { - RowsAtCompileTime = MatrixType::RowsAtCompileTime, - ColsAtCompileTime = MatrixType::ColsAtCompileTime, - Options = MatrixType::Options, - MaxRowsAtCompileTime = MatrixType::MaxRowsAtCompileTime, - MaxColsAtCompileTime = MatrixType::MaxColsAtCompileTime - }; - - /** \brief Scalar type for matrices of type #MatrixType. */ - typedef typename MatrixType::Scalar Scalar; - typedef typename NumTraits::Real RealScalar; - typedef Eigen::Index Index; ///< \deprecated since Eigen 3.3 - - /** \brief Complex scalar type for #MatrixType. - * - * This is \c std::complex if #Scalar is real (e.g., - * \c float or \c double) and just \c Scalar if #Scalar is - * complex. - */ - typedef std::complex ComplexScalar; - - /** \brief Type for vector of eigenvalues as returned by eigenvalues(). - * - * This is a column vector with entries of type #ComplexScalar. - * The length of the vector is the size of #MatrixType. - */ - typedef Matrix EigenvalueType; - - /** \brief Type for matrix of eigenvectors as returned by eigenvectors(). - * - * This is a square matrix with entries of type #ComplexScalar. - * The size is the same as the size of #MatrixType. - */ - typedef Matrix EigenvectorType; - - /** \brief Default constructor. - * - * The default constructor is useful in cases in which the user intends to - * perform decompositions via compute(). - */ - ComplexEigenSolver() - : m_eivec(), - m_eivalues(), - m_schur(), - m_isInitialized(false), - m_eigenvectorsOk(false), - m_matX() - {} - - /** \brief Default Constructor with memory preallocation - * - * Like the default constructor but with preallocation of the internal data - * according to the specified problem \a size. - * \sa ComplexEigenSolver() - */ - explicit ComplexEigenSolver(Index size) - : m_eivec(size, size), - m_eivalues(size), - m_schur(size), - m_isInitialized(false), - m_eigenvectorsOk(false), - m_matX(size, size) - {} - - /** \brief Constructor; computes eigendecomposition of given matrix. - * - * \param[in] matrix Square matrix whose eigendecomposition is to be computed. - * \param[in] computeEigenvectors If true, both the eigenvectors and the - * eigenvalues are computed; if false, only the eigenvalues are - * computed. - * - * This constructor calls compute() to compute the eigendecomposition. - */ - template - explicit ComplexEigenSolver(const EigenBase& matrix, bool computeEigenvectors = true) - : m_eivec(matrix.rows(),matrix.cols()), - m_eivalues(matrix.cols()), - m_schur(matrix.rows()), - m_isInitialized(false), - m_eigenvectorsOk(false), - m_matX(matrix.rows(),matrix.cols()) - { - compute(matrix.derived(), computeEigenvectors); - } +public: + /** \brief Synonym for the template parameter \p _MatrixType. */ + typedef _MatrixType MatrixType; + + enum { + RowsAtCompileTime = MatrixType::RowsAtCompileTime, + ColsAtCompileTime = MatrixType::ColsAtCompileTime, + Options = MatrixType::Options, + MaxRowsAtCompileTime = MatrixType::MaxRowsAtCompileTime, + MaxColsAtCompileTime = MatrixType::MaxColsAtCompileTime + }; + + /** \brief Scalar type for matrices of type #MatrixType. */ + typedef typename MatrixType::Scalar Scalar; + typedef typename NumTraits::Real RealScalar; + typedef Eigen::Index Index;///< \deprecated since Eigen 3.3 + + /** \brief Complex scalar type for #MatrixType. + * + * This is \c std::complex if #Scalar is real (e.g., + * \c float or \c double) and just \c Scalar if #Scalar is + * complex. + */ + typedef std::complex ComplexScalar; + + /** \brief Type for vector of eigenvalues as returned by eigenvalues(). + * + * This is a column vector with entries of type #ComplexScalar. + * The length of the vector is the size of #MatrixType. + */ + typedef Matrix EigenvalueType; + + /** \brief Type for matrix of eigenvectors as returned by eigenvectors(). + * + * This is a square matrix with entries of type #ComplexScalar. + * The size is the same as the size of #MatrixType. + */ + typedef Matrix + EigenvectorType; + + /** \brief Default constructor. + * + * The default constructor is useful in cases in which the user intends to + * perform decompositions via compute(). + */ + ComplexEigenSolver() : m_eivec(), m_eivalues(), m_schur(), m_isInitialized(false), m_eigenvectorsOk(false), m_matX() + {} + + /** \brief Default Constructor with memory preallocation + * + * Like the default constructor but with preallocation of the internal data + * according to the specified problem \a size. + * \sa ComplexEigenSolver() + */ + explicit ComplexEigenSolver(Index size) + : m_eivec(size, size), m_eivalues(size), m_schur(size), m_isInitialized(false), m_eigenvectorsOk(false), + m_matX(size, size) + {} + + /** \brief Constructor; computes eigendecomposition of given matrix. + * + * \param[in] matrix Square matrix whose eigendecomposition is to be computed. + * \param[in] computeEigenvectors If true, both the eigenvectors and the + * eigenvalues are computed; if false, only the eigenvalues are + * computed. + * + * This constructor calls compute() to compute the eigendecomposition. + */ + template + explicit ComplexEigenSolver(const EigenBase &matrix, bool computeEigenvectors = true) + : m_eivec(matrix.rows(), matrix.cols()), m_eivalues(matrix.cols()), m_schur(matrix.rows()), m_isInitialized(false), + m_eigenvectorsOk(false), m_matX(matrix.rows(), matrix.cols()) + { + compute(matrix.derived(), computeEigenvectors); + } - /** \brief Returns the eigenvectors of given matrix. - * - * \returns A const reference to the matrix whose columns are the eigenvectors. - * - * \pre Either the constructor - * ComplexEigenSolver(const MatrixType& matrix, bool) or the member - * function compute(const MatrixType& matrix, bool) has been called before - * to compute the eigendecomposition of a matrix, and - * \p computeEigenvectors was set to true (the default). - * - * This function returns a matrix whose columns are the eigenvectors. Column - * \f$ k \f$ is an eigenvector corresponding to eigenvalue number \f$ k - * \f$ as returned by eigenvalues(). The eigenvectors are normalized to - * have (Euclidean) norm equal to one. The matrix returned by this - * function is the matrix \f$ V \f$ in the eigendecomposition \f$ A = V D - * V^{-1} \f$, if it exists. - * - * Example: \include ComplexEigenSolver_eigenvectors.cpp - * Output: \verbinclude ComplexEigenSolver_eigenvectors.out - */ - const EigenvectorType& eigenvectors() const - { - eigen_assert(m_isInitialized && "ComplexEigenSolver is not initialized."); - eigen_assert(m_eigenvectorsOk && "The eigenvectors have not been computed together with the eigenvalues."); - return m_eivec; - } + /** \brief Returns the eigenvectors of given matrix. + * + * \returns A const reference to the matrix whose columns are the eigenvectors. + * + * \pre Either the constructor + * ComplexEigenSolver(const MatrixType& matrix, bool) or the member + * function compute(const MatrixType& matrix, bool) has been called before + * to compute the eigendecomposition of a matrix, and + * \p computeEigenvectors was set to true (the default). + * + * This function returns a matrix whose columns are the eigenvectors. Column + * \f$ k \f$ is an eigenvector corresponding to eigenvalue number \f$ k + * \f$ as returned by eigenvalues(). The eigenvectors are normalized to + * have (Euclidean) norm equal to one. The matrix returned by this + * function is the matrix \f$ V \f$ in the eigendecomposition \f$ A = V D + * V^{-1} \f$, if it exists. + * + * Example: \include ComplexEigenSolver_eigenvectors.cpp + * Output: \verbinclude ComplexEigenSolver_eigenvectors.out + */ + const EigenvectorType &eigenvectors() const + { + eigen_assert(m_isInitialized && "ComplexEigenSolver is not initialized."); + eigen_assert(m_eigenvectorsOk && "The eigenvectors have not been computed together with the eigenvalues."); + return m_eivec; + } - /** \brief Returns the eigenvalues of given matrix. - * - * \returns A const reference to the column vector containing the eigenvalues. - * - * \pre Either the constructor - * ComplexEigenSolver(const MatrixType& matrix, bool) or the member - * function compute(const MatrixType& matrix, bool) has been called before - * to compute the eigendecomposition of a matrix. - * - * This function returns a column vector containing the - * eigenvalues. Eigenvalues are repeated according to their - * algebraic multiplicity, so there are as many eigenvalues as - * rows in the matrix. The eigenvalues are not sorted in any particular - * order. - * - * Example: \include ComplexEigenSolver_eigenvalues.cpp - * Output: \verbinclude ComplexEigenSolver_eigenvalues.out - */ - const EigenvalueType& eigenvalues() const - { - eigen_assert(m_isInitialized && "ComplexEigenSolver is not initialized."); - return m_eivalues; - } + /** \brief Returns the eigenvalues of given matrix. + * + * \returns A const reference to the column vector containing the eigenvalues. + * + * \pre Either the constructor + * ComplexEigenSolver(const MatrixType& matrix, bool) or the member + * function compute(const MatrixType& matrix, bool) has been called before + * to compute the eigendecomposition of a matrix. + * + * This function returns a column vector containing the + * eigenvalues. Eigenvalues are repeated according to their + * algebraic multiplicity, so there are as many eigenvalues as + * rows in the matrix. The eigenvalues are not sorted in any particular + * order. + * + * Example: \include ComplexEigenSolver_eigenvalues.cpp + * Output: \verbinclude ComplexEigenSolver_eigenvalues.out + */ + const EigenvalueType &eigenvalues() const + { + eigen_assert(m_isInitialized && "ComplexEigenSolver is not initialized."); + return m_eivalues; + } - /** \brief Computes eigendecomposition of given matrix. - * - * \param[in] matrix Square matrix whose eigendecomposition is to be computed. - * \param[in] computeEigenvectors If true, both the eigenvectors and the - * eigenvalues are computed; if false, only the eigenvalues are - * computed. - * \returns Reference to \c *this - * - * This function computes the eigenvalues of the complex matrix \p matrix. - * The eigenvalues() function can be used to retrieve them. If - * \p computeEigenvectors is true, then the eigenvectors are also computed - * and can be retrieved by calling eigenvectors(). - * - * The matrix is first reduced to Schur form using the - * ComplexSchur class. The Schur decomposition is then used to - * compute the eigenvalues and eigenvectors. - * - * The cost of the computation is dominated by the cost of the - * Schur decomposition, which is \f$ O(n^3) \f$ where \f$ n \f$ - * is the size of the matrix. - * - * Example: \include ComplexEigenSolver_compute.cpp - * Output: \verbinclude ComplexEigenSolver_compute.out - */ - template - ComplexEigenSolver& compute(const EigenBase& matrix, bool computeEigenvectors = true); - - /** \brief Reports whether previous computation was successful. - * - * \returns \c Success if computation was succesful, \c NoConvergence otherwise. - */ - ComputationInfo info() const - { - eigen_assert(m_isInitialized && "ComplexEigenSolver is not initialized."); - return m_schur.info(); - } + /** \brief Computes eigendecomposition of given matrix. + * + * \param[in] matrix Square matrix whose eigendecomposition is to be computed. + * \param[in] computeEigenvectors If true, both the eigenvectors and the + * eigenvalues are computed; if false, only the eigenvalues are + * computed. + * \returns Reference to \c *this + * + * This function computes the eigenvalues of the complex matrix \p matrix. + * The eigenvalues() function can be used to retrieve them. If + * \p computeEigenvectors is true, then the eigenvectors are also computed + * and can be retrieved by calling eigenvectors(). + * + * The matrix is first reduced to Schur form using the + * ComplexSchur class. The Schur decomposition is then used to + * compute the eigenvalues and eigenvectors. + * + * The cost of the computation is dominated by the cost of the + * Schur decomposition, which is \f$ O(n^3) \f$ where \f$ n \f$ + * is the size of the matrix. + * + * Example: \include ComplexEigenSolver_compute.cpp + * Output: \verbinclude ComplexEigenSolver_compute.out + */ + template + ComplexEigenSolver &compute(const EigenBase &matrix, bool computeEigenvectors = true); + + /** \brief Reports whether previous computation was successful. + * + * \returns \c Success if computation was succesful, \c NoConvergence otherwise. + */ + ComputationInfo info() const + { + eigen_assert(m_isInitialized && "ComplexEigenSolver is not initialized."); + return m_schur.info(); + } - /** \brief Sets the maximum number of iterations allowed. */ - ComplexEigenSolver& setMaxIterations(Index maxIters) - { - m_schur.setMaxIterations(maxIters); - return *this; - } + /** \brief Sets the maximum number of iterations allowed. */ + ComplexEigenSolver &setMaxIterations(Index maxIters) + { + m_schur.setMaxIterations(maxIters); + return *this; + } - /** \brief Returns the maximum number of iterations. */ - Index getMaxIterations() - { - return m_schur.getMaxIterations(); - } + /** \brief Returns the maximum number of iterations. */ + Index getMaxIterations() { return m_schur.getMaxIterations(); } - protected: - - static void check_template_parameters() - { - EIGEN_STATIC_ASSERT_NON_INTEGER(Scalar); - } - - EigenvectorType m_eivec; - EigenvalueType m_eivalues; - ComplexSchur m_schur; - bool m_isInitialized; - bool m_eigenvectorsOk; - EigenvectorType m_matX; - - private: - void doComputeEigenvectors(RealScalar matrixnorm); - void sortEigenvalues(bool computeEigenvectors); +protected: + static void check_template_parameters() { EIGEN_STATIC_ASSERT_NON_INTEGER(Scalar); } + + EigenvectorType m_eivec; + EigenvalueType m_eivalues; + ComplexSchur m_schur; + bool m_isInitialized; + bool m_eigenvectorsOk; + EigenvectorType m_matX; + +private: + void doComputeEigenvectors(RealScalar matrixnorm); + void sortEigenvalues(bool computeEigenvectors); }; template template -ComplexEigenSolver& -ComplexEigenSolver::compute(const EigenBase& matrix, bool computeEigenvectors) +ComplexEigenSolver &ComplexEigenSolver::compute(const EigenBase &matrix, + bool computeEigenvectors) { check_template_parameters(); - + // this code is inspired from Jampack eigen_assert(matrix.cols() == matrix.rows()); @@ -269,11 +253,9 @@ ComplexEigenSolver::compute(const EigenBase& matrix, bool // The eigenvalues are on the diagonal of T. m_schur.compute(matrix.derived(), computeEigenvectors); - if(m_schur.info() == Success) - { + if (m_schur.info() == Success) { m_eivalues = m_schur.matrixT().diagonal(); - if(computeEigenvectors) - doComputeEigenvectors(m_schur.matrixT().norm()); + if (computeEigenvectors) doComputeEigenvectors(m_schur.matrixT().norm()); sortEigenvalues(computeEigenvectors); } @@ -283,64 +265,54 @@ ComplexEigenSolver::compute(const EigenBase& matrix, bool } -template -void ComplexEigenSolver::doComputeEigenvectors(RealScalar matrixnorm) +template void ComplexEigenSolver::doComputeEigenvectors(RealScalar matrixnorm) { const Index n = m_eivalues.size(); - matrixnorm = numext::maxi(matrixnorm,(std::numeric_limits::min)()); + matrixnorm = numext::maxi(matrixnorm, (std::numeric_limits::min)()); // Compute X such that T = X D X^(-1), where D is the diagonal of T. // The matrix X is unit triangular. m_matX = EigenvectorType::Zero(n, n); - for(Index k=n-1 ; k>=0 ; k--) - { - m_matX.coeffRef(k,k) = ComplexScalar(1.0,0.0); + for (Index k = n - 1; k >= 0; k--) { + m_matX.coeffRef(k, k) = ComplexScalar(1.0, 0.0); // Compute X(i,k) using the (i,k) entry of the equation X T = D X - for(Index i=k-1 ; i>=0 ; i--) - { - m_matX.coeffRef(i,k) = -m_schur.matrixT().coeff(i,k); - if(k-i-1>0) - m_matX.coeffRef(i,k) -= (m_schur.matrixT().row(i).segment(i+1,k-i-1) * m_matX.col(k).segment(i+1,k-i-1)).value(); - ComplexScalar z = m_schur.matrixT().coeff(i,i) - m_schur.matrixT().coeff(k,k); - if(z==ComplexScalar(0)) - { + for (Index i = k - 1; i >= 0; i--) { + m_matX.coeffRef(i, k) = -m_schur.matrixT().coeff(i, k); + if (k - i - 1 > 0) + m_matX.coeffRef(i, k) -= + (m_schur.matrixT().row(i).segment(i + 1, k - i - 1) * m_matX.col(k).segment(i + 1, k - i - 1)).value(); + ComplexScalar z = m_schur.matrixT().coeff(i, i) - m_schur.matrixT().coeff(k, k); + if (z == ComplexScalar(0)) { // If the i-th and k-th eigenvalue are equal, then z equals 0. // Use a small value instead, to prevent division by zero. numext::real_ref(z) = NumTraits::epsilon() * matrixnorm; } - m_matX.coeffRef(i,k) = m_matX.coeff(i,k) / z; + m_matX.coeffRef(i, k) = m_matX.coeff(i, k) / z; } } // Compute V as V = U X; now A = U T U^* = U X D X^(-1) U^* = V D V^(-1) m_eivec.noalias() = m_schur.matrixU() * m_matX; // .. and normalize the eigenvectors - for(Index k=0 ; k -void ComplexEigenSolver::sortEigenvalues(bool computeEigenvectors) +template void ComplexEigenSolver::sortEigenvalues(bool computeEigenvectors) { - const Index n = m_eivalues.size(); - for (Index i=0; i struct complex_schur_reduce_to_hessenberg; + template struct complex_schur_reduce_to_hessenberg; } /** \eigenvalues_module \ingroup Eigenvalues_Module - * - * - * \class ComplexSchur - * - * \brief Performs a complex Schur decomposition of a real or complex square matrix - * - * \tparam _MatrixType the type of the matrix of which we are - * computing the Schur decomposition; this is expected to be an - * instantiation of the Matrix class template. - * - * Given a real or complex square matrix A, this class computes the - * Schur decomposition: \f$ A = U T U^*\f$ where U is a unitary - * complex matrix, and T is a complex upper triangular matrix. The - * diagonal of the matrix T corresponds to the eigenvalues of the - * matrix A. - * - * Call the function compute() to compute the Schur decomposition of - * a given matrix. Alternatively, you can use the - * ComplexSchur(const MatrixType&, bool) constructor which computes - * the Schur decomposition at construction time. Once the - * decomposition is computed, you can use the matrixU() and matrixT() - * functions to retrieve the matrices U and V in the decomposition. - * - * \note This code is inspired from Jampack - * - * \sa class RealSchur, class EigenSolver, class ComplexEigenSolver - */ + * + * + * \class ComplexSchur + * + * \brief Performs a complex Schur decomposition of a real or complex square matrix + * + * \tparam _MatrixType the type of the matrix of which we are + * computing the Schur decomposition; this is expected to be an + * instantiation of the Matrix class template. + * + * Given a real or complex square matrix A, this class computes the + * Schur decomposition: \f$ A = U T U^*\f$ where U is a unitary + * complex matrix, and T is a complex upper triangular matrix. The + * diagonal of the matrix T corresponds to the eigenvalues of the + * matrix A. + * + * Call the function compute() to compute the Schur decomposition of + * a given matrix. Alternatively, you can use the + * ComplexSchur(const MatrixType&, bool) constructor which computes + * the Schur decomposition at construction time. Once the + * decomposition is computed, you can use the matrixU() and matrixT() + * functions to retrieve the matrices U and V in the decomposition. + * + * \note This code is inspired from Jampack + * + * \sa class RealSchur, class EigenSolver, class ComplexEigenSolver + */ template class ComplexSchur { - public: - typedef _MatrixType MatrixType; - enum { - RowsAtCompileTime = MatrixType::RowsAtCompileTime, - ColsAtCompileTime = MatrixType::ColsAtCompileTime, - Options = MatrixType::Options, - MaxRowsAtCompileTime = MatrixType::MaxRowsAtCompileTime, - MaxColsAtCompileTime = MatrixType::MaxColsAtCompileTime - }; - - /** \brief Scalar type for matrices of type \p _MatrixType. */ - typedef typename MatrixType::Scalar Scalar; - typedef typename NumTraits::Real RealScalar; - typedef Eigen::Index Index; ///< \deprecated since Eigen 3.3 - - /** \brief Complex scalar type for \p _MatrixType. - * - * This is \c std::complex if #Scalar is real (e.g., - * \c float or \c double) and just \c Scalar if #Scalar is - * complex. - */ - typedef std::complex ComplexScalar; - - /** \brief Type for the matrices in the Schur decomposition. - * - * This is a square matrix with entries of type #ComplexScalar. - * The size is the same as the size of \p _MatrixType. - */ - typedef Matrix ComplexMatrixType; - - /** \brief Default constructor. - * - * \param [in] size Positive integer, size of the matrix whose Schur decomposition will be computed. - * - * The default constructor is useful in cases in which the user - * intends to perform decompositions via compute(). The \p size - * parameter is only used as a hint. It is not an error to give a - * wrong \p size, but it may impair performance. - * - * \sa compute() for an example. - */ - explicit ComplexSchur(Index size = RowsAtCompileTime==Dynamic ? 1 : RowsAtCompileTime) - : m_matT(size,size), - m_matU(size,size), - m_hess(size), - m_isInitialized(false), - m_matUisUptodate(false), - m_maxIters(-1) - {} - - /** \brief Constructor; computes Schur decomposition of given matrix. - * - * \param[in] matrix Square matrix whose Schur decomposition is to be computed. - * \param[in] computeU If true, both T and U are computed; if false, only T is computed. - * - * This constructor calls compute() to compute the Schur decomposition. - * - * \sa matrixT() and matrixU() for examples. - */ - template - explicit ComplexSchur(const EigenBase& matrix, bool computeU = true) - : m_matT(matrix.rows(),matrix.cols()), - m_matU(matrix.rows(),matrix.cols()), - m_hess(matrix.rows()), - m_isInitialized(false), - m_matUisUptodate(false), - m_maxIters(-1) - { - compute(matrix.derived(), computeU); - } - - /** \brief Returns the unitary matrix in the Schur decomposition. - * - * \returns A const reference to the matrix U. - * - * It is assumed that either the constructor - * ComplexSchur(const MatrixType& matrix, bool computeU) or the - * member function compute(const MatrixType& matrix, bool computeU) - * has been called before to compute the Schur decomposition of a - * matrix, and that \p computeU was set to true (the default - * value). - * - * Example: \include ComplexSchur_matrixU.cpp - * Output: \verbinclude ComplexSchur_matrixU.out - */ - const ComplexMatrixType& matrixU() const - { - eigen_assert(m_isInitialized && "ComplexSchur is not initialized."); - eigen_assert(m_matUisUptodate && "The matrix U has not been computed during the ComplexSchur decomposition."); - return m_matU; - } +public: + typedef _MatrixType MatrixType; + enum { + RowsAtCompileTime = MatrixType::RowsAtCompileTime, + ColsAtCompileTime = MatrixType::ColsAtCompileTime, + Options = MatrixType::Options, + MaxRowsAtCompileTime = MatrixType::MaxRowsAtCompileTime, + MaxColsAtCompileTime = MatrixType::MaxColsAtCompileTime + }; + + /** \brief Scalar type for matrices of type \p _MatrixType. */ + typedef typename MatrixType::Scalar Scalar; + typedef typename NumTraits::Real RealScalar; + typedef Eigen::Index Index;///< \deprecated since Eigen 3.3 + + /** \brief Complex scalar type for \p _MatrixType. + * + * This is \c std::complex if #Scalar is real (e.g., + * \c float or \c double) and just \c Scalar if #Scalar is + * complex. + */ + typedef std::complex ComplexScalar; + + /** \brief Type for the matrices in the Schur decomposition. + * + * This is a square matrix with entries of type #ComplexScalar. + * The size is the same as the size of \p _MatrixType. + */ + typedef Matrix + ComplexMatrixType; + + /** \brief Default constructor. + * + * \param [in] size Positive integer, size of the matrix whose Schur decomposition will be computed. + * + * The default constructor is useful in cases in which the user + * intends to perform decompositions via compute(). The \p size + * parameter is only used as a hint. It is not an error to give a + * wrong \p size, but it may impair performance. + * + * \sa compute() for an example. + */ + explicit ComplexSchur(Index size = RowsAtCompileTime == Dynamic ? 1 : RowsAtCompileTime) + : m_matT(size, size), m_matU(size, size), m_hess(size), m_isInitialized(false), m_matUisUptodate(false), + m_maxIters(-1) + {} + + /** \brief Constructor; computes Schur decomposition of given matrix. + * + * \param[in] matrix Square matrix whose Schur decomposition is to be computed. + * \param[in] computeU If true, both T and U are computed; if false, only T is computed. + * + * This constructor calls compute() to compute the Schur decomposition. + * + * \sa matrixT() and matrixU() for examples. + */ + template + explicit ComplexSchur(const EigenBase &matrix, bool computeU = true) + : m_matT(matrix.rows(), matrix.cols()), m_matU(matrix.rows(), matrix.cols()), m_hess(matrix.rows()), + m_isInitialized(false), m_matUisUptodate(false), m_maxIters(-1) + { + compute(matrix.derived(), computeU); + } - /** \brief Returns the triangular matrix in the Schur decomposition. - * - * \returns A const reference to the matrix T. - * - * It is assumed that either the constructor - * ComplexSchur(const MatrixType& matrix, bool computeU) or the - * member function compute(const MatrixType& matrix, bool computeU) - * has been called before to compute the Schur decomposition of a - * matrix. - * - * Note that this function returns a plain square matrix. If you want to reference - * only the upper triangular part, use: - * \code schur.matrixT().triangularView() \endcode - * - * Example: \include ComplexSchur_matrixT.cpp - * Output: \verbinclude ComplexSchur_matrixT.out - */ - const ComplexMatrixType& matrixT() const - { - eigen_assert(m_isInitialized && "ComplexSchur is not initialized."); - return m_matT; - } + /** \brief Returns the unitary matrix in the Schur decomposition. + * + * \returns A const reference to the matrix U. + * + * It is assumed that either the constructor + * ComplexSchur(const MatrixType& matrix, bool computeU) or the + * member function compute(const MatrixType& matrix, bool computeU) + * has been called before to compute the Schur decomposition of a + * matrix, and that \p computeU was set to true (the default + * value). + * + * Example: \include ComplexSchur_matrixU.cpp + * Output: \verbinclude ComplexSchur_matrixU.out + */ + const ComplexMatrixType &matrixU() const + { + eigen_assert(m_isInitialized && "ComplexSchur is not initialized."); + eigen_assert(m_matUisUptodate && "The matrix U has not been computed during the ComplexSchur decomposition."); + return m_matU; + } - /** \brief Computes Schur decomposition of given matrix. - * - * \param[in] matrix Square matrix whose Schur decomposition is to be computed. - * \param[in] computeU If true, both T and U are computed; if false, only T is computed. - - * \returns Reference to \c *this - * - * The Schur decomposition is computed by first reducing the - * matrix to Hessenberg form using the class - * HessenbergDecomposition. The Hessenberg matrix is then reduced - * to triangular form by performing QR iterations with a single - * shift. The cost of computing the Schur decomposition depends - * on the number of iterations; as a rough guide, it may be taken - * on the number of iterations; as a rough guide, it may be taken - * to be \f$25n^3\f$ complex flops, or \f$10n^3\f$ complex flops - * if \a computeU is false. - * - * Example: \include ComplexSchur_compute.cpp - * Output: \verbinclude ComplexSchur_compute.out - * - * \sa compute(const MatrixType&, bool, Index) - */ - template - ComplexSchur& compute(const EigenBase& matrix, bool computeU = true); - - /** \brief Compute Schur decomposition from a given Hessenberg matrix - * \param[in] matrixH Matrix in Hessenberg form H - * \param[in] matrixQ orthogonal matrix Q that transform a matrix A to H : A = Q H Q^T - * \param computeU Computes the matriX U of the Schur vectors - * \return Reference to \c *this - * - * This routine assumes that the matrix is already reduced in Hessenberg form matrixH - * using either the class HessenbergDecomposition or another mean. - * It computes the upper quasi-triangular matrix T of the Schur decomposition of H - * When computeU is true, this routine computes the matrix U such that - * A = U T U^T = (QZ) T (QZ)^T = Q H Q^T where A is the initial matrix - * - * NOTE Q is referenced if computeU is true; so, if the initial orthogonal matrix - * is not available, the user should give an identity matrix (Q.setIdentity()) - * - * \sa compute(const MatrixType&, bool) - */ - template - ComplexSchur& computeFromHessenberg(const HessMatrixType& matrixH, const OrthMatrixType& matrixQ, bool computeU=true); - - /** \brief Reports whether previous computation was successful. - * - * \returns \c Success if computation was succesful, \c NoConvergence otherwise. - */ - ComputationInfo info() const - { - eigen_assert(m_isInitialized && "ComplexSchur is not initialized."); - return m_info; - } + /** \brief Returns the triangular matrix in the Schur decomposition. + * + * \returns A const reference to the matrix T. + * + * It is assumed that either the constructor + * ComplexSchur(const MatrixType& matrix, bool computeU) or the + * member function compute(const MatrixType& matrix, bool computeU) + * has been called before to compute the Schur decomposition of a + * matrix. + * + * Note that this function returns a plain square matrix. If you want to reference + * only the upper triangular part, use: + * \code schur.matrixT().triangularView() \endcode + * + * Example: \include ComplexSchur_matrixT.cpp + * Output: \verbinclude ComplexSchur_matrixT.out + */ + const ComplexMatrixType &matrixT() const + { + eigen_assert(m_isInitialized && "ComplexSchur is not initialized."); + return m_matT; + } - /** \brief Sets the maximum number of iterations allowed. - * - * If not specified by the user, the maximum number of iterations is m_maxIterationsPerRow times the size - * of the matrix. - */ - ComplexSchur& setMaxIterations(Index maxIters) - { - m_maxIters = maxIters; - return *this; - } + /** \brief Computes Schur decomposition of given matrix. + * + * \param[in] matrix Square matrix whose Schur decomposition is to be computed. + * \param[in] computeU If true, both T and U are computed; if false, only T is computed. + + * \returns Reference to \c *this + * + * The Schur decomposition is computed by first reducing the + * matrix to Hessenberg form using the class + * HessenbergDecomposition. The Hessenberg matrix is then reduced + * to triangular form by performing QR iterations with a single + * shift. The cost of computing the Schur decomposition depends + * on the number of iterations; as a rough guide, it may be taken + * on the number of iterations; as a rough guide, it may be taken + * to be \f$25n^3\f$ complex flops, or \f$10n^3\f$ complex flops + * if \a computeU is false. + * + * Example: \include ComplexSchur_compute.cpp + * Output: \verbinclude ComplexSchur_compute.out + * + * \sa compute(const MatrixType&, bool, Index) + */ + template ComplexSchur &compute(const EigenBase &matrix, bool computeU = true); + + /** \brief Compute Schur decomposition from a given Hessenberg matrix + * \param[in] matrixH Matrix in Hessenberg form H + * \param[in] matrixQ orthogonal matrix Q that transform a matrix A to H : A = Q H Q^T + * \param computeU Computes the matriX U of the Schur vectors + * \return Reference to \c *this + * + * This routine assumes that the matrix is already reduced in Hessenberg form matrixH + * using either the class HessenbergDecomposition or another mean. + * It computes the upper quasi-triangular matrix T of the Schur decomposition of H + * When computeU is true, this routine computes the matrix U such that + * A = U T U^T = (QZ) T (QZ)^T = Q H Q^T where A is the initial matrix + * + * NOTE Q is referenced if computeU is true; so, if the initial orthogonal matrix + * is not available, the user should give an identity matrix (Q.setIdentity()) + * + * \sa compute(const MatrixType&, bool) + */ + template + ComplexSchur & + computeFromHessenberg(const HessMatrixType &matrixH, const OrthMatrixType &matrixQ, bool computeU = true); + + /** \brief Reports whether previous computation was successful. + * + * \returns \c Success if computation was succesful, \c NoConvergence otherwise. + */ + ComputationInfo info() const + { + eigen_assert(m_isInitialized && "ComplexSchur is not initialized."); + return m_info; + } - /** \brief Returns the maximum number of iterations. */ - Index getMaxIterations() - { - return m_maxIters; - } + /** \brief Sets the maximum number of iterations allowed. + * + * If not specified by the user, the maximum number of iterations is m_maxIterationsPerRow times the size + * of the matrix. + */ + ComplexSchur &setMaxIterations(Index maxIters) + { + m_maxIters = maxIters; + return *this; + } - /** \brief Maximum number of iterations per row. - * - * If not otherwise specified, the maximum number of iterations is this number times the size of the - * matrix. It is currently set to 30. - */ - static const int m_maxIterationsPerRow = 30; - - protected: - ComplexMatrixType m_matT, m_matU; - HessenbergDecomposition m_hess; - ComputationInfo m_info; - bool m_isInitialized; - bool m_matUisUptodate; - Index m_maxIters; - - private: - bool subdiagonalEntryIsNeglegible(Index i); - ComplexScalar computeShift(Index iu, Index iter); - void reduceToTriangularForm(bool computeU); - friend struct internal::complex_schur_reduce_to_hessenberg::IsComplex>; + /** \brief Returns the maximum number of iterations. */ + Index getMaxIterations() { return m_maxIters; } + + /** \brief Maximum number of iterations per row. + * + * If not otherwise specified, the maximum number of iterations is this number times the size of the + * matrix. It is currently set to 30. + */ + static const int m_maxIterationsPerRow = 30; + +protected: + ComplexMatrixType m_matT, m_matU; + HessenbergDecomposition m_hess; + ComputationInfo m_info; + bool m_isInitialized; + bool m_matUisUptodate; + Index m_maxIters; + +private: + bool subdiagonalEntryIsNeglegible(Index i); + ComplexScalar computeShift(Index iu, Index iter); + void reduceToTriangularForm(bool computeU); + friend struct internal::complex_schur_reduce_to_hessenberg::IsComplex>; }; /** If m_matT(i+1,i) is neglegible in floating point arithmetic - * compared to m_matT(i,i) and m_matT(j,j), then set it to zero and - * return true, else return false. */ -template -inline bool ComplexSchur::subdiagonalEntryIsNeglegible(Index i) + * compared to m_matT(i,i) and m_matT(j,j), then set it to zero and + * return true, else return false. */ +template inline bool ComplexSchur::subdiagonalEntryIsNeglegible(Index i) { - RealScalar d = numext::norm1(m_matT.coeff(i,i)) + numext::norm1(m_matT.coeff(i+1,i+1)); - RealScalar sd = numext::norm1(m_matT.coeff(i+1,i)); - if (internal::isMuchSmallerThan(sd, d, NumTraits::epsilon())) - { - m_matT.coeffRef(i+1,i) = ComplexScalar(0); + RealScalar d = numext::norm1(m_matT.coeff(i, i)) + numext::norm1(m_matT.coeff(i + 1, i + 1)); + RealScalar sd = numext::norm1(m_matT.coeff(i + 1, i)); + if (internal::isMuchSmallerThan(sd, d, NumTraits::epsilon())) { + m_matT.coeffRef(i + 1, i) = ComplexScalar(0); return true; } return false; @@ -281,33 +274,32 @@ template typename ComplexSchur::ComplexScalar ComplexSchur::computeShift(Index iu, Index iter) { using std::abs; - if (iter == 10 || iter == 20) - { + if (iter == 10 || iter == 20) { // exceptional shift, taken from http://www.netlib.org/eispack/comqr.f - return abs(numext::real(m_matT.coeff(iu,iu-1))) + abs(numext::real(m_matT.coeff(iu-1,iu-2))); + return abs(numext::real(m_matT.coeff(iu, iu - 1))) + abs(numext::real(m_matT.coeff(iu - 1, iu - 2))); } // compute the shift as one of the eigenvalues of t, the 2x2 // diagonal block on the bottom of the active submatrix - Matrix t = m_matT.template block<2,2>(iu-1,iu-1); + Matrix t = m_matT.template block<2, 2>(iu - 1, iu - 1); RealScalar normt = t.cwiseAbs().sum(); - t /= normt; // the normalization by sf is to avoid under/overflow + t /= normt;// the normalization by sf is to avoid under/overflow - ComplexScalar b = t.coeff(0,1) * t.coeff(1,0); - ComplexScalar c = t.coeff(0,0) - t.coeff(1,1); - ComplexScalar disc = sqrt(c*c + RealScalar(4)*b); - ComplexScalar det = t.coeff(0,0) * t.coeff(1,1) - b; - ComplexScalar trace = t.coeff(0,0) + t.coeff(1,1); + ComplexScalar b = t.coeff(0, 1) * t.coeff(1, 0); + ComplexScalar c = t.coeff(0, 0) - t.coeff(1, 1); + ComplexScalar disc = sqrt(c * c + RealScalar(4) * b); + ComplexScalar det = t.coeff(0, 0) * t.coeff(1, 1) - b; + ComplexScalar trace = t.coeff(0, 0) + t.coeff(1, 1); ComplexScalar eival1 = (trace + disc) / RealScalar(2); ComplexScalar eival2 = (trace - disc) / RealScalar(2); - if(numext::norm1(eival1) > numext::norm1(eival2)) + if (numext::norm1(eival1) > numext::norm1(eival2)) eival2 = det / eival1; else eival1 = det / eival2; // choose the eigenvalue closest to the bottom entry of the diagonal - if(numext::norm1(eival1-t.coeff(1,1)) < numext::norm1(eival2-t.coeff(1,1))) + if (numext::norm1(eival1 - t.coeff(1, 1)) < numext::norm1(eival2 - t.coeff(1, 1))) return normt * eival1; else return normt * eival2; @@ -316,113 +308,104 @@ typename ComplexSchur::ComplexScalar ComplexSchur::compu template template -ComplexSchur& ComplexSchur::compute(const EigenBase& matrix, bool computeU) +ComplexSchur &ComplexSchur::compute(const EigenBase &matrix, bool computeU) { m_matUisUptodate = false; eigen_assert(matrix.cols() == matrix.rows()); - if(matrix.cols() == 1) - { + if (matrix.cols() == 1) { m_matT = matrix.derived().template cast(); - if(computeU) m_matU = ComplexMatrixType::Identity(1,1); + if (computeU) m_matU = ComplexMatrixType::Identity(1, 1); m_info = Success; m_isInitialized = true; m_matUisUptodate = computeU; return *this; } - internal::complex_schur_reduce_to_hessenberg::IsComplex>::run(*this, matrix.derived(), computeU); + internal::complex_schur_reduce_to_hessenberg::IsComplex>::run( + *this, matrix.derived(), computeU); computeFromHessenberg(m_matT, m_matU, computeU); return *this; } template template -ComplexSchur& ComplexSchur::computeFromHessenberg(const HessMatrixType& matrixH, const OrthMatrixType& matrixQ, bool computeU) +ComplexSchur &ComplexSchur::computeFromHessenberg(const HessMatrixType &matrixH, + const OrthMatrixType &matrixQ, + bool computeU) { m_matT = matrixH; - if(computeU) - m_matU = matrixQ; + if (computeU) m_matU = matrixQ; reduceToTriangularForm(computeU); return *this; } namespace internal { -/* Reduce given matrix to Hessenberg form */ -template -struct complex_schur_reduce_to_hessenberg -{ - // this is the implementation for the case IsComplex = true - static void run(ComplexSchur& _this, const MatrixType& matrix, bool computeU) + /* Reduce given matrix to Hessenberg form */ + template struct complex_schur_reduce_to_hessenberg { - _this.m_hess.compute(matrix); - _this.m_matT = _this.m_hess.matrixH(); - if(computeU) _this.m_matU = _this.m_hess.matrixQ(); - } -}; + // this is the implementation for the case IsComplex = true + static void run(ComplexSchur &_this, const MatrixType &matrix, bool computeU) + { + _this.m_hess.compute(matrix); + _this.m_matT = _this.m_hess.matrixH(); + if (computeU) _this.m_matU = _this.m_hess.matrixQ(); + } + }; -template -struct complex_schur_reduce_to_hessenberg -{ - static void run(ComplexSchur& _this, const MatrixType& matrix, bool computeU) + template struct complex_schur_reduce_to_hessenberg { - typedef typename ComplexSchur::ComplexScalar ComplexScalar; - - // Note: m_hess is over RealScalar; m_matT and m_matU is over ComplexScalar - _this.m_hess.compute(matrix); - _this.m_matT = _this.m_hess.matrixH().template cast(); - if(computeU) + static void run(ComplexSchur &_this, const MatrixType &matrix, bool computeU) { - // This may cause an allocation which seems to be avoidable - MatrixType Q = _this.m_hess.matrixQ(); - _this.m_matU = Q.template cast(); + typedef typename ComplexSchur::ComplexScalar ComplexScalar; + + // Note: m_hess is over RealScalar; m_matT and m_matU is over ComplexScalar + _this.m_hess.compute(matrix); + _this.m_matT = _this.m_hess.matrixH().template cast(); + if (computeU) { + // This may cause an allocation which seems to be avoidable + MatrixType Q = _this.m_hess.matrixQ(); + _this.m_matU = Q.template cast(); + } } - } -}; + }; -} // end namespace internal +}// end namespace internal // Reduce the Hessenberg matrix m_matT to triangular form by QR iteration. -template -void ComplexSchur::reduceToTriangularForm(bool computeU) -{ +template void ComplexSchur::reduceToTriangularForm(bool computeU) +{ Index maxIters = m_maxIters; - if (maxIters == -1) - maxIters = m_maxIterationsPerRow * m_matT.rows(); + if (maxIters == -1) maxIters = m_maxIterationsPerRow * m_matT.rows(); - // The matrix m_matT is divided in three parts. - // Rows 0,...,il-1 are decoupled from the rest because m_matT(il,il-1) is zero. + // The matrix m_matT is divided in three parts. + // Rows 0,...,il-1 are decoupled from the rest because m_matT(il,il-1) is zero. // Rows il,...,iu is the part we are working on (the active submatrix). // Rows iu+1,...,end are already brought in triangular form. Index iu = m_matT.cols() - 1; Index il; - Index iter = 0; // number of iterations we are working on the (iu,iu) element - Index totalIter = 0; // number of iterations for whole matrix + Index iter = 0;// number of iterations we are working on the (iu,iu) element + Index totalIter = 0;// number of iterations for whole matrix - while(true) - { + while (true) { // find iu, the bottom row of the active submatrix - while(iu > 0) - { - if(!subdiagonalEntryIsNeglegible(iu-1)) break; + while (iu > 0) { + if (!subdiagonalEntryIsNeglegible(iu - 1)) break; iter = 0; --iu; } // if iu is zero then we are done; the whole matrix is triangularized - if(iu==0) break; + if (iu == 0) break; // if we spent too many iterations, we give up iter++; totalIter++; - if(totalIter > maxIters) break; + if (totalIter > maxIters) break; // find il, the top row of the active submatrix - il = iu-1; - while(il > 0 && !subdiagonalEntryIsNeglegible(il-1)) - { - --il; - } + il = iu - 1; + while (il > 0 && !subdiagonalEntryIsNeglegible(il - 1)) { --il; } /* perform the QR step using Givens rotations. The first rotation creates a bulge; the (il+2,il) element becomes nonzero. This @@ -430,22 +413,21 @@ void ComplexSchur::reduceToTriangularForm(bool computeU) ComplexScalar shift = computeShift(iu, iter); JacobiRotation rot; - rot.makeGivens(m_matT.coeff(il,il) - shift, m_matT.coeff(il+1,il)); - m_matT.rightCols(m_matT.cols()-il).applyOnTheLeft(il, il+1, rot.adjoint()); - m_matT.topRows((std::min)(il+2,iu)+1).applyOnTheRight(il, il+1, rot); - if(computeU) m_matU.applyOnTheRight(il, il+1, rot); - - for(Index i=il+1 ; i::reduceToTriangularForm(bool computeU) m_matUisUptodate = computeU; } -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_COMPLEX_SCHUR_H +#endif// EIGEN_COMPLEX_SCHUR_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Eigenvalues/ComplexSchur_LAPACKE.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Eigenvalues/ComplexSchur_LAPACKE.h index 4980a3ed..28190d31 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Eigenvalues/ComplexSchur_LAPACKE.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Eigenvalues/ComplexSchur_LAPACKE.h @@ -33,59 +33,70 @@ #ifndef EIGEN_COMPLEX_SCHUR_LAPACKE_H #define EIGEN_COMPLEX_SCHUR_LAPACKE_H -namespace Eigen { +namespace Eigen { /** \internal Specialization for the data types supported by LAPACKe */ -#define EIGEN_LAPACKE_SCHUR_COMPLEX(EIGTYPE, LAPACKE_TYPE, LAPACKE_PREFIX, LAPACKE_PREFIX_U, EIGCOLROW, LAPACKE_COLROW) \ -template<> template inline \ -ComplexSchur >& \ -ComplexSchur >::compute(const EigenBase& matrix, bool computeU) \ -{ \ - typedef Matrix MatrixType; \ - typedef MatrixType::RealScalar RealScalar; \ - typedef std::complex ComplexScalar; \ -\ - eigen_assert(matrix.cols() == matrix.rows()); \ -\ - m_matUisUptodate = false; \ - if(matrix.cols() == 1) \ - { \ - m_matT = matrix.derived().template cast(); \ - if(computeU) m_matU = ComplexMatrixType::Identity(1,1); \ - m_info = Success; \ - m_isInitialized = true; \ - m_matUisUptodate = computeU; \ - return *this; \ - } \ - lapack_int n = internal::convert_index(matrix.cols()), sdim, info; \ - lapack_int matrix_order = LAPACKE_COLROW; \ - char jobvs, sort='N'; \ - LAPACK_##LAPACKE_PREFIX_U##_SELECT1 select = 0; \ - jobvs = (computeU) ? 'V' : 'N'; \ - m_matU.resize(n, n); \ - lapack_int ldvs = internal::convert_index(m_matU.outerStride()); \ - m_matT = matrix; \ - lapack_int lda = internal::convert_index(m_matT.outerStride()); \ - Matrix w; \ - w.resize(n, 1);\ - info = LAPACKE_##LAPACKE_PREFIX##gees( matrix_order, jobvs, sort, select, n, (LAPACKE_TYPE*)m_matT.data(), lda, &sdim, (LAPACKE_TYPE*)w.data(), (LAPACKE_TYPE*)m_matU.data(), ldvs ); \ - if(info == 0) \ - m_info = Success; \ - else \ - m_info = NoConvergence; \ -\ - m_isInitialized = true; \ - m_matUisUptodate = computeU; \ - return *this; \ -\ -} +#define EIGEN_LAPACKE_SCHUR_COMPLEX( \ + EIGTYPE, LAPACKE_TYPE, LAPACKE_PREFIX, LAPACKE_PREFIX_U, EIGCOLROW, LAPACKE_COLROW) \ + template<> \ + template \ + inline ComplexSchur> & \ + ComplexSchur>::compute( \ + const EigenBase &matrix, bool computeU) \ + { \ + typedef Matrix MatrixType; \ + typedef MatrixType::RealScalar RealScalar; \ + typedef std::complex ComplexScalar; \ + \ + eigen_assert(matrix.cols() == matrix.rows()); \ + \ + m_matUisUptodate = false; \ + if (matrix.cols() == 1) { \ + m_matT = matrix.derived().template cast(); \ + if (computeU) m_matU = ComplexMatrixType::Identity(1, 1); \ + m_info = Success; \ + m_isInitialized = true; \ + m_matUisUptodate = computeU; \ + return *this; \ + } \ + lapack_int n = internal::convert_index(matrix.cols()), sdim, info; \ + lapack_int matrix_order = LAPACKE_COLROW; \ + char jobvs, sort = 'N'; \ + LAPACK_##LAPACKE_PREFIX_U##_SELECT1 select = 0; \ + jobvs = (computeU) ? 'V' : 'N'; \ + m_matU.resize(n, n); \ + lapack_int ldvs = internal::convert_index(m_matU.outerStride()); \ + m_matT = matrix; \ + lapack_int lda = internal::convert_index(m_matT.outerStride()); \ + Matrix w; \ + w.resize(n, 1); \ + info = LAPACKE_##LAPACKE_PREFIX##gees(matrix_order, \ + jobvs, \ + sort, \ + select, \ + n, \ + (LAPACKE_TYPE *)m_matT.data(), \ + lda, \ + &sdim, \ + (LAPACKE_TYPE *)w.data(), \ + (LAPACKE_TYPE *)m_matU.data(), \ + ldvs); \ + if (info == 0) \ + m_info = Success; \ + else \ + m_info = NoConvergence; \ + \ + m_isInitialized = true; \ + m_matUisUptodate = computeU; \ + return *this; \ + } EIGEN_LAPACKE_SCHUR_COMPLEX(dcomplex, lapack_complex_double, z, Z, ColMajor, LAPACK_COL_MAJOR) -EIGEN_LAPACKE_SCHUR_COMPLEX(scomplex, lapack_complex_float, c, C, ColMajor, LAPACK_COL_MAJOR) +EIGEN_LAPACKE_SCHUR_COMPLEX(scomplex, lapack_complex_float, c, C, ColMajor, LAPACK_COL_MAJOR) EIGEN_LAPACKE_SCHUR_COMPLEX(dcomplex, lapack_complex_double, z, Z, RowMajor, LAPACK_ROW_MAJOR) -EIGEN_LAPACKE_SCHUR_COMPLEX(scomplex, lapack_complex_float, c, C, RowMajor, LAPACK_ROW_MAJOR) +EIGEN_LAPACKE_SCHUR_COMPLEX(scomplex, lapack_complex_float, c, C, RowMajor, LAPACK_ROW_MAJOR) -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_COMPLEX_SCHUR_LAPACKE_H +#endif// EIGEN_COMPLEX_SCHUR_LAPACKE_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Eigenvalues/EigenSolver.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Eigenvalues/EigenSolver.h index f205b185..32e22a4d 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Eigenvalues/EigenSolver.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Eigenvalues/EigenSolver.h @@ -13,328 +13,316 @@ #include "./RealSchur.h" -namespace Eigen { +namespace Eigen { /** \eigenvalues_module \ingroup Eigenvalues_Module - * - * - * \class EigenSolver - * - * \brief Computes eigenvalues and eigenvectors of general matrices - * - * \tparam _MatrixType the type of the matrix of which we are computing the - * eigendecomposition; this is expected to be an instantiation of the Matrix - * class template. Currently, only real matrices are supported. - * - * The eigenvalues and eigenvectors of a matrix \f$ A \f$ are scalars - * \f$ \lambda \f$ and vectors \f$ v \f$ such that \f$ Av = \lambda v \f$. If - * \f$ D \f$ is a diagonal matrix with the eigenvalues on the diagonal, and - * \f$ V \f$ is a matrix with the eigenvectors as its columns, then \f$ A V = - * V D \f$. The matrix \f$ V \f$ is almost always invertible, in which case we - * have \f$ A = V D V^{-1} \f$. This is called the eigendecomposition. - * - * The eigenvalues and eigenvectors of a matrix may be complex, even when the - * matrix is real. However, we can choose real matrices \f$ V \f$ and \f$ D - * \f$ satisfying \f$ A V = V D \f$, just like the eigendecomposition, if the - * matrix \f$ D \f$ is not required to be diagonal, but if it is allowed to - * have blocks of the form - * \f[ \begin{bmatrix} u & v \\ -v & u \end{bmatrix} \f] - * (where \f$ u \f$ and \f$ v \f$ are real numbers) on the diagonal. These - * blocks correspond to complex eigenvalue pairs \f$ u \pm iv \f$. We call - * this variant of the eigendecomposition the pseudo-eigendecomposition. - * - * Call the function compute() to compute the eigenvalues and eigenvectors of - * a given matrix. Alternatively, you can use the - * EigenSolver(const MatrixType&, bool) constructor which computes the - * eigenvalues and eigenvectors at construction time. Once the eigenvalue and - * eigenvectors are computed, they can be retrieved with the eigenvalues() and - * eigenvectors() functions. The pseudoEigenvalueMatrix() and - * pseudoEigenvectors() methods allow the construction of the - * pseudo-eigendecomposition. - * - * The documentation for EigenSolver(const MatrixType&, bool) contains an - * example of the typical use of this class. - * - * \note The implementation is adapted from - * JAMA (public domain). - * Their code is based on EISPACK. - * - * \sa MatrixBase::eigenvalues(), class ComplexEigenSolver, class SelfAdjointEigenSolver - */ + * + * + * \class EigenSolver + * + * \brief Computes eigenvalues and eigenvectors of general matrices + * + * \tparam _MatrixType the type of the matrix of which we are computing the + * eigendecomposition; this is expected to be an instantiation of the Matrix + * class template. Currently, only real matrices are supported. + * + * The eigenvalues and eigenvectors of a matrix \f$ A \f$ are scalars + * \f$ \lambda \f$ and vectors \f$ v \f$ such that \f$ Av = \lambda v \f$. If + * \f$ D \f$ is a diagonal matrix with the eigenvalues on the diagonal, and + * \f$ V \f$ is a matrix with the eigenvectors as its columns, then \f$ A V = + * V D \f$. The matrix \f$ V \f$ is almost always invertible, in which case we + * have \f$ A = V D V^{-1} \f$. This is called the eigendecomposition. + * + * The eigenvalues and eigenvectors of a matrix may be complex, even when the + * matrix is real. However, we can choose real matrices \f$ V \f$ and \f$ D + * \f$ satisfying \f$ A V = V D \f$, just like the eigendecomposition, if the + * matrix \f$ D \f$ is not required to be diagonal, but if it is allowed to + * have blocks of the form + * \f[ \begin{bmatrix} u & v \\ -v & u \end{bmatrix} \f] + * (where \f$ u \f$ and \f$ v \f$ are real numbers) on the diagonal. These + * blocks correspond to complex eigenvalue pairs \f$ u \pm iv \f$. We call + * this variant of the eigendecomposition the pseudo-eigendecomposition. + * + * Call the function compute() to compute the eigenvalues and eigenvectors of + * a given matrix. Alternatively, you can use the + * EigenSolver(const MatrixType&, bool) constructor which computes the + * eigenvalues and eigenvectors at construction time. Once the eigenvalue and + * eigenvectors are computed, they can be retrieved with the eigenvalues() and + * eigenvectors() functions. The pseudoEigenvalueMatrix() and + * pseudoEigenvectors() methods allow the construction of the + * pseudo-eigendecomposition. + * + * The documentation for EigenSolver(const MatrixType&, bool) contains an + * example of the typical use of this class. + * + * \note The implementation is adapted from + * JAMA (public domain). + * Their code is based on EISPACK. + * + * \sa MatrixBase::eigenvalues(), class ComplexEigenSolver, class SelfAdjointEigenSolver + */ template class EigenSolver { - public: - - /** \brief Synonym for the template parameter \p _MatrixType. */ - typedef _MatrixType MatrixType; - - enum { - RowsAtCompileTime = MatrixType::RowsAtCompileTime, - ColsAtCompileTime = MatrixType::ColsAtCompileTime, - Options = MatrixType::Options, - MaxRowsAtCompileTime = MatrixType::MaxRowsAtCompileTime, - MaxColsAtCompileTime = MatrixType::MaxColsAtCompileTime - }; - - /** \brief Scalar type for matrices of type #MatrixType. */ - typedef typename MatrixType::Scalar Scalar; - typedef typename NumTraits::Real RealScalar; - typedef Eigen::Index Index; ///< \deprecated since Eigen 3.3 - - /** \brief Complex scalar type for #MatrixType. - * - * This is \c std::complex if #Scalar is real (e.g., - * \c float or \c double) and just \c Scalar if #Scalar is - * complex. - */ - typedef std::complex ComplexScalar; - - /** \brief Type for vector of eigenvalues as returned by eigenvalues(). - * - * This is a column vector with entries of type #ComplexScalar. - * The length of the vector is the size of #MatrixType. - */ - typedef Matrix EigenvalueType; - - /** \brief Type for matrix of eigenvectors as returned by eigenvectors(). - * - * This is a square matrix with entries of type #ComplexScalar. - * The size is the same as the size of #MatrixType. - */ - typedef Matrix EigenvectorsType; - - /** \brief Default constructor. - * - * The default constructor is useful in cases in which the user intends to - * perform decompositions via EigenSolver::compute(const MatrixType&, bool). - * - * \sa compute() for an example. - */ - EigenSolver() : m_eivec(), m_eivalues(), m_isInitialized(false), m_realSchur(), m_matT(), m_tmp() {} - - /** \brief Default constructor with memory preallocation - * - * Like the default constructor but with preallocation of the internal data - * according to the specified problem \a size. - * \sa EigenSolver() - */ - explicit EigenSolver(Index size) - : m_eivec(size, size), - m_eivalues(size), - m_isInitialized(false), - m_eigenvectorsOk(false), - m_realSchur(size), - m_matT(size, size), - m_tmp(size) - {} - - /** \brief Constructor; computes eigendecomposition of given matrix. - * - * \param[in] matrix Square matrix whose eigendecomposition is to be computed. - * \param[in] computeEigenvectors If true, both the eigenvectors and the - * eigenvalues are computed; if false, only the eigenvalues are - * computed. - * - * This constructor calls compute() to compute the eigenvalues - * and eigenvectors. - * - * Example: \include EigenSolver_EigenSolver_MatrixType.cpp - * Output: \verbinclude EigenSolver_EigenSolver_MatrixType.out - * - * \sa compute() - */ - template - explicit EigenSolver(const EigenBase& matrix, bool computeEigenvectors = true) - : m_eivec(matrix.rows(), matrix.cols()), - m_eivalues(matrix.cols()), - m_isInitialized(false), - m_eigenvectorsOk(false), - m_realSchur(matrix.cols()), - m_matT(matrix.rows(), matrix.cols()), - m_tmp(matrix.cols()) - { - compute(matrix.derived(), computeEigenvectors); - } +public: + /** \brief Synonym for the template parameter \p _MatrixType. */ + typedef _MatrixType MatrixType; + + enum { + RowsAtCompileTime = MatrixType::RowsAtCompileTime, + ColsAtCompileTime = MatrixType::ColsAtCompileTime, + Options = MatrixType::Options, + MaxRowsAtCompileTime = MatrixType::MaxRowsAtCompileTime, + MaxColsAtCompileTime = MatrixType::MaxColsAtCompileTime + }; + + /** \brief Scalar type for matrices of type #MatrixType. */ + typedef typename MatrixType::Scalar Scalar; + typedef typename NumTraits::Real RealScalar; + typedef Eigen::Index Index;///< \deprecated since Eigen 3.3 + + /** \brief Complex scalar type for #MatrixType. + * + * This is \c std::complex if #Scalar is real (e.g., + * \c float or \c double) and just \c Scalar if #Scalar is + * complex. + */ + typedef std::complex ComplexScalar; + + /** \brief Type for vector of eigenvalues as returned by eigenvalues(). + * + * This is a column vector with entries of type #ComplexScalar. + * The length of the vector is the size of #MatrixType. + */ + typedef Matrix EigenvalueType; + + /** \brief Type for matrix of eigenvectors as returned by eigenvectors(). + * + * This is a square matrix with entries of type #ComplexScalar. + * The size is the same as the size of #MatrixType. + */ + typedef Matrix + EigenvectorsType; + + /** \brief Default constructor. + * + * The default constructor is useful in cases in which the user intends to + * perform decompositions via EigenSolver::compute(const MatrixType&, bool). + * + * \sa compute() for an example. + */ + EigenSolver() : m_eivec(), m_eivalues(), m_isInitialized(false), m_realSchur(), m_matT(), m_tmp() {} + + /** \brief Default constructor with memory preallocation + * + * Like the default constructor but with preallocation of the internal data + * according to the specified problem \a size. + * \sa EigenSolver() + */ + explicit EigenSolver(Index size) + : m_eivec(size, size), m_eivalues(size), m_isInitialized(false), m_eigenvectorsOk(false), m_realSchur(size), + m_matT(size, size), m_tmp(size) + {} + + /** \brief Constructor; computes eigendecomposition of given matrix. + * + * \param[in] matrix Square matrix whose eigendecomposition is to be computed. + * \param[in] computeEigenvectors If true, both the eigenvectors and the + * eigenvalues are computed; if false, only the eigenvalues are + * computed. + * + * This constructor calls compute() to compute the eigenvalues + * and eigenvectors. + * + * Example: \include EigenSolver_EigenSolver_MatrixType.cpp + * Output: \verbinclude EigenSolver_EigenSolver_MatrixType.out + * + * \sa compute() + */ + template + explicit EigenSolver(const EigenBase &matrix, bool computeEigenvectors = true) + : m_eivec(matrix.rows(), matrix.cols()), m_eivalues(matrix.cols()), m_isInitialized(false), m_eigenvectorsOk(false), + m_realSchur(matrix.cols()), m_matT(matrix.rows(), matrix.cols()), m_tmp(matrix.cols()) + { + compute(matrix.derived(), computeEigenvectors); + } - /** \brief Returns the eigenvectors of given matrix. - * - * \returns %Matrix whose columns are the (possibly complex) eigenvectors. - * - * \pre Either the constructor - * EigenSolver(const MatrixType&,bool) or the member function - * compute(const MatrixType&, bool) has been called before, and - * \p computeEigenvectors was set to true (the default). - * - * Column \f$ k \f$ of the returned matrix is an eigenvector corresponding - * to eigenvalue number \f$ k \f$ as returned by eigenvalues(). The - * eigenvectors are normalized to have (Euclidean) norm equal to one. The - * matrix returned by this function is the matrix \f$ V \f$ in the - * eigendecomposition \f$ A = V D V^{-1} \f$, if it exists. - * - * Example: \include EigenSolver_eigenvectors.cpp - * Output: \verbinclude EigenSolver_eigenvectors.out - * - * \sa eigenvalues(), pseudoEigenvectors() - */ - EigenvectorsType eigenvectors() const; - - /** \brief Returns the pseudo-eigenvectors of given matrix. - * - * \returns Const reference to matrix whose columns are the pseudo-eigenvectors. - * - * \pre Either the constructor - * EigenSolver(const MatrixType&,bool) or the member function - * compute(const MatrixType&, bool) has been called before, and - * \p computeEigenvectors was set to true (the default). - * - * The real matrix \f$ V \f$ returned by this function and the - * block-diagonal matrix \f$ D \f$ returned by pseudoEigenvalueMatrix() - * satisfy \f$ AV = VD \f$. - * - * Example: \include EigenSolver_pseudoEigenvectors.cpp - * Output: \verbinclude EigenSolver_pseudoEigenvectors.out - * - * \sa pseudoEigenvalueMatrix(), eigenvectors() - */ - const MatrixType& pseudoEigenvectors() const - { - eigen_assert(m_isInitialized && "EigenSolver is not initialized."); - eigen_assert(m_eigenvectorsOk && "The eigenvectors have not been computed together with the eigenvalues."); - return m_eivec; - } + /** \brief Returns the eigenvectors of given matrix. + * + * \returns %Matrix whose columns are the (possibly complex) eigenvectors. + * + * \pre Either the constructor + * EigenSolver(const MatrixType&,bool) or the member function + * compute(const MatrixType&, bool) has been called before, and + * \p computeEigenvectors was set to true (the default). + * + * Column \f$ k \f$ of the returned matrix is an eigenvector corresponding + * to eigenvalue number \f$ k \f$ as returned by eigenvalues(). The + * eigenvectors are normalized to have (Euclidean) norm equal to one. The + * matrix returned by this function is the matrix \f$ V \f$ in the + * eigendecomposition \f$ A = V D V^{-1} \f$, if it exists. + * + * Example: \include EigenSolver_eigenvectors.cpp + * Output: \verbinclude EigenSolver_eigenvectors.out + * + * \sa eigenvalues(), pseudoEigenvectors() + */ + EigenvectorsType eigenvectors() const; + + /** \brief Returns the pseudo-eigenvectors of given matrix. + * + * \returns Const reference to matrix whose columns are the pseudo-eigenvectors. + * + * \pre Either the constructor + * EigenSolver(const MatrixType&,bool) or the member function + * compute(const MatrixType&, bool) has been called before, and + * \p computeEigenvectors was set to true (the default). + * + * The real matrix \f$ V \f$ returned by this function and the + * block-diagonal matrix \f$ D \f$ returned by pseudoEigenvalueMatrix() + * satisfy \f$ AV = VD \f$. + * + * Example: \include EigenSolver_pseudoEigenvectors.cpp + * Output: \verbinclude EigenSolver_pseudoEigenvectors.out + * + * \sa pseudoEigenvalueMatrix(), eigenvectors() + */ + const MatrixType &pseudoEigenvectors() const + { + eigen_assert(m_isInitialized && "EigenSolver is not initialized."); + eigen_assert(m_eigenvectorsOk && "The eigenvectors have not been computed together with the eigenvalues."); + return m_eivec; + } - /** \brief Returns the block-diagonal matrix in the pseudo-eigendecomposition. - * - * \returns A block-diagonal matrix. - * - * \pre Either the constructor - * EigenSolver(const MatrixType&,bool) or the member function - * compute(const MatrixType&, bool) has been called before. - * - * The matrix \f$ D \f$ returned by this function is real and - * block-diagonal. The blocks on the diagonal are either 1-by-1 or 2-by-2 - * blocks of the form - * \f$ \begin{bmatrix} u & v \\ -v & u \end{bmatrix} \f$. - * These blocks are not sorted in any particular order. - * The matrix \f$ D \f$ and the matrix \f$ V \f$ returned by - * pseudoEigenvectors() satisfy \f$ AV = VD \f$. - * - * \sa pseudoEigenvectors() for an example, eigenvalues() - */ - MatrixType pseudoEigenvalueMatrix() const; - - /** \brief Returns the eigenvalues of given matrix. - * - * \returns A const reference to the column vector containing the eigenvalues. - * - * \pre Either the constructor - * EigenSolver(const MatrixType&,bool) or the member function - * compute(const MatrixType&, bool) has been called before. - * - * The eigenvalues are repeated according to their algebraic multiplicity, - * so there are as many eigenvalues as rows in the matrix. The eigenvalues - * are not sorted in any particular order. - * - * Example: \include EigenSolver_eigenvalues.cpp - * Output: \verbinclude EigenSolver_eigenvalues.out - * - * \sa eigenvectors(), pseudoEigenvalueMatrix(), - * MatrixBase::eigenvalues() - */ - const EigenvalueType& eigenvalues() const - { - eigen_assert(m_isInitialized && "EigenSolver is not initialized."); - return m_eivalues; - } + /** \brief Returns the block-diagonal matrix in the pseudo-eigendecomposition. + * + * \returns A block-diagonal matrix. + * + * \pre Either the constructor + * EigenSolver(const MatrixType&,bool) or the member function + * compute(const MatrixType&, bool) has been called before. + * + * The matrix \f$ D \f$ returned by this function is real and + * block-diagonal. The blocks on the diagonal are either 1-by-1 or 2-by-2 + * blocks of the form + * \f$ \begin{bmatrix} u & v \\ -v & u \end{bmatrix} \f$. + * These blocks are not sorted in any particular order. + * The matrix \f$ D \f$ and the matrix \f$ V \f$ returned by + * pseudoEigenvectors() satisfy \f$ AV = VD \f$. + * + * \sa pseudoEigenvectors() for an example, eigenvalues() + */ + MatrixType pseudoEigenvalueMatrix() const; + + /** \brief Returns the eigenvalues of given matrix. + * + * \returns A const reference to the column vector containing the eigenvalues. + * + * \pre Either the constructor + * EigenSolver(const MatrixType&,bool) or the member function + * compute(const MatrixType&, bool) has been called before. + * + * The eigenvalues are repeated according to their algebraic multiplicity, + * so there are as many eigenvalues as rows in the matrix. The eigenvalues + * are not sorted in any particular order. + * + * Example: \include EigenSolver_eigenvalues.cpp + * Output: \verbinclude EigenSolver_eigenvalues.out + * + * \sa eigenvectors(), pseudoEigenvalueMatrix(), + * MatrixBase::eigenvalues() + */ + const EigenvalueType &eigenvalues() const + { + eigen_assert(m_isInitialized && "EigenSolver is not initialized."); + return m_eivalues; + } - /** \brief Computes eigendecomposition of given matrix. - * - * \param[in] matrix Square matrix whose eigendecomposition is to be computed. - * \param[in] computeEigenvectors If true, both the eigenvectors and the - * eigenvalues are computed; if false, only the eigenvalues are - * computed. - * \returns Reference to \c *this - * - * This function computes the eigenvalues of the real matrix \p matrix. - * The eigenvalues() function can be used to retrieve them. If - * \p computeEigenvectors is true, then the eigenvectors are also computed - * and can be retrieved by calling eigenvectors(). - * - * The matrix is first reduced to real Schur form using the RealSchur - * class. The Schur decomposition is then used to compute the eigenvalues - * and eigenvectors. - * - * The cost of the computation is dominated by the cost of the - * Schur decomposition, which is very approximately \f$ 25n^3 \f$ - * (where \f$ n \f$ is the size of the matrix) if \p computeEigenvectors - * is true, and \f$ 10n^3 \f$ if \p computeEigenvectors is false. - * - * This method reuses of the allocated data in the EigenSolver object. - * - * Example: \include EigenSolver_compute.cpp - * Output: \verbinclude EigenSolver_compute.out - */ - template - EigenSolver& compute(const EigenBase& matrix, bool computeEigenvectors = true); - - /** \returns NumericalIssue if the input contains INF or NaN values or overflow occured. Returns Success otherwise. */ - ComputationInfo info() const - { - eigen_assert(m_isInitialized && "EigenSolver is not initialized."); - return m_info; - } + /** \brief Computes eigendecomposition of given matrix. + * + * \param[in] matrix Square matrix whose eigendecomposition is to be computed. + * \param[in] computeEigenvectors If true, both the eigenvectors and the + * eigenvalues are computed; if false, only the eigenvalues are + * computed. + * \returns Reference to \c *this + * + * This function computes the eigenvalues of the real matrix \p matrix. + * The eigenvalues() function can be used to retrieve them. If + * \p computeEigenvectors is true, then the eigenvectors are also computed + * and can be retrieved by calling eigenvectors(). + * + * The matrix is first reduced to real Schur form using the RealSchur + * class. The Schur decomposition is then used to compute the eigenvalues + * and eigenvectors. + * + * The cost of the computation is dominated by the cost of the + * Schur decomposition, which is very approximately \f$ 25n^3 \f$ + * (where \f$ n \f$ is the size of the matrix) if \p computeEigenvectors + * is true, and \f$ 10n^3 \f$ if \p computeEigenvectors is false. + * + * This method reuses of the allocated data in the EigenSolver object. + * + * Example: \include EigenSolver_compute.cpp + * Output: \verbinclude EigenSolver_compute.out + */ + template + EigenSolver &compute(const EigenBase &matrix, bool computeEigenvectors = true); + + /** \returns NumericalIssue if the input contains INF or NaN values or overflow occured. Returns Success otherwise. */ + ComputationInfo info() const + { + eigen_assert(m_isInitialized && "EigenSolver is not initialized."); + return m_info; + } - /** \brief Sets the maximum number of iterations allowed. */ - EigenSolver& setMaxIterations(Index maxIters) - { - m_realSchur.setMaxIterations(maxIters); - return *this; - } + /** \brief Sets the maximum number of iterations allowed. */ + EigenSolver &setMaxIterations(Index maxIters) + { + m_realSchur.setMaxIterations(maxIters); + return *this; + } - /** \brief Returns the maximum number of iterations. */ - Index getMaxIterations() - { - return m_realSchur.getMaxIterations(); - } + /** \brief Returns the maximum number of iterations. */ + Index getMaxIterations() { return m_realSchur.getMaxIterations(); } - private: - void doComputeEigenvectors(); +private: + void doComputeEigenvectors(); - protected: - - static void check_template_parameters() - { - EIGEN_STATIC_ASSERT_NON_INTEGER(Scalar); - EIGEN_STATIC_ASSERT(!NumTraits::IsComplex, NUMERIC_TYPE_MUST_BE_REAL); - } - - MatrixType m_eivec; - EigenvalueType m_eivalues; - bool m_isInitialized; - bool m_eigenvectorsOk; - ComputationInfo m_info; - RealSchur m_realSchur; - MatrixType m_matT; - - typedef Matrix ColumnVectorType; - ColumnVectorType m_tmp; +protected: + static void check_template_parameters() + { + EIGEN_STATIC_ASSERT_NON_INTEGER(Scalar); + EIGEN_STATIC_ASSERT(!NumTraits::IsComplex, NUMERIC_TYPE_MUST_BE_REAL); + } + + MatrixType m_eivec; + EigenvalueType m_eivalues; + bool m_isInitialized; + bool m_eigenvectorsOk; + ComputationInfo m_info; + RealSchur m_realSchur; + MatrixType m_matT; + + typedef Matrix ColumnVectorType; + ColumnVectorType m_tmp; }; -template -MatrixType EigenSolver::pseudoEigenvalueMatrix() const +template MatrixType EigenSolver::pseudoEigenvalueMatrix() const { eigen_assert(m_isInitialized && "EigenSolver is not initialized."); - const RealScalar precision = RealScalar(2)*NumTraits::epsilon(); + const RealScalar precision = RealScalar(2) * NumTraits::epsilon(); Index n = m_eivalues.rows(); - MatrixType matD = MatrixType::Zero(n,n); - for (Index i=0; i(i,i) << numext::real(m_eivalues.coeff(i)), numext::imag(m_eivalues.coeff(i)), - -numext::imag(m_eivalues.coeff(i)), numext::real(m_eivalues.coeff(i)); + matD.coeffRef(i, i) = numext::real(m_eivalues.coeff(i)); + else { + matD.template block<2, 2>(i, i) << numext::real(m_eivalues.coeff(i)), numext::imag(m_eivalues.coeff(i)), + -numext::imag(m_eivalues.coeff(i)), numext::real(m_eivalues.coeff(i)); ++i; } } @@ -346,27 +334,23 @@ typename EigenSolver::EigenvectorsType EigenSolver::eige { eigen_assert(m_isInitialized && "EigenSolver is not initialized."); eigen_assert(m_eigenvectorsOk && "The eigenvectors have not been computed together with the eigenvalues."); - const RealScalar precision = RealScalar(2)*NumTraits::epsilon(); + const RealScalar precision = RealScalar(2) * NumTraits::epsilon(); Index n = m_eivec.cols(); - EigenvectorsType matV(n,n); - for (Index j=0; j(); matV.col(j).normalize(); - } - else - { + } else { // we have a pair of complex eigen values - for (Index i=0; i::EigenvectorsType EigenSolver::eige template template -EigenSolver& -EigenSolver::compute(const EigenBase& matrix, bool computeEigenvectors) +EigenSolver &EigenSolver::compute(const EigenBase &matrix, bool computeEigenvectors) { check_template_parameters(); - + using std::sqrt; using std::abs; using numext::isfinite; @@ -387,52 +370,44 @@ EigenSolver::compute(const EigenBase& matrix, bool comput // Reduce to real Schur form. m_realSchur.compute(matrix.derived(), computeEigenvectors); - + m_info = m_realSchur.info(); - if (m_info == Success) - { + if (m_info == Success) { m_matT = m_realSchur.matrixT(); - if (computeEigenvectors) - m_eivec = m_realSchur.matrixU(); - + if (computeEigenvectors) m_eivec = m_realSchur.matrixU(); + // Compute eigenvalues from matT m_eivalues.resize(matrix.cols()); Index i = 0; - while (i < matrix.cols()) - { - if (i == matrix.cols() - 1 || m_matT.coeff(i+1, i) == Scalar(0)) - { + while (i < matrix.cols()) { + if (i == matrix.cols() - 1 || m_matT.coeff(i + 1, i) == Scalar(0)) { m_eivalues.coeffRef(i) = m_matT.coeff(i, i); - if(!(isfinite)(m_eivalues.coeffRef(i))) - { + if (!(isfinite)(m_eivalues.coeffRef(i))) { m_isInitialized = true; m_eigenvectorsOk = false; m_info = NumericalIssue; return *this; } ++i; - } - else - { - Scalar p = Scalar(0.5) * (m_matT.coeff(i, i) - m_matT.coeff(i+1, i+1)); + } else { + Scalar p = Scalar(0.5) * (m_matT.coeff(i, i) - m_matT.coeff(i + 1, i + 1)); Scalar z; // Compute z = sqrt(abs(p * p + m_matT.coeff(i+1, i) * m_matT.coeff(i, i+1))); // without overflow { - Scalar t0 = m_matT.coeff(i+1, i); - Scalar t1 = m_matT.coeff(i, i+1); - Scalar maxval = numext::maxi(abs(p),numext::maxi(abs(t0),abs(t1))); + Scalar t0 = m_matT.coeff(i + 1, i); + Scalar t1 = m_matT.coeff(i, i + 1); + Scalar maxval = numext::maxi(abs(p), numext::maxi(abs(t0), abs(t1))); t0 /= maxval; t1 /= maxval; - Scalar p0 = p/maxval; + Scalar p0 = p / maxval; z = maxval * sqrt(abs(p0 * p0 + t0 * t1)); } - - m_eivalues.coeffRef(i) = ComplexScalar(m_matT.coeff(i+1, i+1) + p, z); - m_eivalues.coeffRef(i+1) = ComplexScalar(m_matT.coeff(i+1, i+1) + p, -z); - if(!((isfinite)(m_eivalues.coeffRef(i)) && (isfinite)(m_eivalues.coeffRef(i+1)))) - { + + m_eivalues.coeffRef(i) = ComplexScalar(m_matT.coeff(i + 1, i + 1) + p, z); + m_eivalues.coeffRef(i + 1) = ComplexScalar(m_matT.coeff(i + 1, i + 1) + p, -z); + if (!((isfinite)(m_eivalues.coeffRef(i)) && (isfinite)(m_eivalues.coeffRef(i + 1)))) { m_isInitialized = true; m_eigenvectorsOk = false; m_info = NumericalIssue; @@ -441,10 +416,9 @@ EigenSolver::compute(const EigenBase& matrix, bool comput i += 2; } } - + // Compute eigenvectors. - if (computeEigenvectors) - doComputeEigenvectors(); + if (computeEigenvectors) doComputeEigenvectors(); } m_isInitialized = true; @@ -454,8 +428,7 @@ EigenSolver::compute(const EigenBase& matrix, bool comput } -template -void EigenSolver::doComputeEigenvectors() +template void EigenSolver::doComputeEigenvectors() { using std::abs; const Index size = m_eivec.cols(); @@ -463,160 +436,133 @@ void EigenSolver::doComputeEigenvectors() // inefficient! this is already computed in RealSchur Scalar norm(0); - for (Index j = 0; j < size; ++j) - { - norm += m_matT.row(j).segment((std::max)(j-1,Index(0)), size-(std::max)(j-1,Index(0))).cwiseAbs().sum(); + for (Index j = 0; j < size; ++j) { + norm += m_matT.row(j).segment((std::max)(j - 1, Index(0)), size - (std::max)(j - 1, Index(0))).cwiseAbs().sum(); } - + // Backsubstitute to find vectors of upper triangular form - if (norm == Scalar(0)) - { - return; - } + if (norm == Scalar(0)) { return; } - for (Index n = size-1; n >= 0; n--) - { + for (Index n = size - 1; n >= 0; n--) { Scalar p = m_eivalues.coeff(n).real(); Scalar q = m_eivalues.coeff(n).imag(); // Scalar vector - if (q == Scalar(0)) - { + if (q == Scalar(0)) { Scalar lastr(0), lastw(0); Index l = n; - m_matT.coeffRef(n,n) = Scalar(1); - for (Index i = n-1; i >= 0; i--) - { - Scalar w = m_matT.coeff(i,i) - p; - Scalar r = m_matT.row(i).segment(l,n-l+1).dot(m_matT.col(n).segment(l, n-l+1)); + m_matT.coeffRef(n, n) = Scalar(1); + for (Index i = n - 1; i >= 0; i--) { + Scalar w = m_matT.coeff(i, i) - p; + Scalar r = m_matT.row(i).segment(l, n - l + 1).dot(m_matT.col(n).segment(l, n - l + 1)); - if (m_eivalues.coeff(i).imag() < Scalar(0)) - { + if (m_eivalues.coeff(i).imag() < Scalar(0)) { lastw = w; lastr = r; - } - else - { + } else { l = i; - if (m_eivalues.coeff(i).imag() == Scalar(0)) - { + if (m_eivalues.coeff(i).imag() == Scalar(0)) { if (w != Scalar(0)) - m_matT.coeffRef(i,n) = -r / w; + m_matT.coeffRef(i, n) = -r / w; else - m_matT.coeffRef(i,n) = -r / (eps * norm); - } - else // Solve real equations + m_matT.coeffRef(i, n) = -r / (eps * norm); + } else// Solve real equations { - Scalar x = m_matT.coeff(i,i+1); - Scalar y = m_matT.coeff(i+1,i); - Scalar denom = (m_eivalues.coeff(i).real() - p) * (m_eivalues.coeff(i).real() - p) + m_eivalues.coeff(i).imag() * m_eivalues.coeff(i).imag(); + Scalar x = m_matT.coeff(i, i + 1); + Scalar y = m_matT.coeff(i + 1, i); + Scalar denom = (m_eivalues.coeff(i).real() - p) * (m_eivalues.coeff(i).real() - p) + + m_eivalues.coeff(i).imag() * m_eivalues.coeff(i).imag(); Scalar t = (x * lastr - lastw * r) / denom; - m_matT.coeffRef(i,n) = t; + m_matT.coeffRef(i, n) = t; if (abs(x) > abs(lastw)) - m_matT.coeffRef(i+1,n) = (-r - w * t) / x; + m_matT.coeffRef(i + 1, n) = (-r - w * t) / x; else - m_matT.coeffRef(i+1,n) = (-lastr - y * t) / lastw; + m_matT.coeffRef(i + 1, n) = (-lastr - y * t) / lastw; } // Overflow control - Scalar t = abs(m_matT.coeff(i,n)); - if ((eps * t) * t > Scalar(1)) - m_matT.col(n).tail(size-i) /= t; + Scalar t = abs(m_matT.coeff(i, n)); + if ((eps * t) * t > Scalar(1)) m_matT.col(n).tail(size - i) /= t; } } - } - else if (q < Scalar(0) && n > 0) // Complex vector + } else if (q < Scalar(0) && n > 0)// Complex vector { Scalar lastra(0), lastsa(0), lastw(0); - Index l = n-1; + Index l = n - 1; // Last vector component imaginary so matrix is triangular - if (abs(m_matT.coeff(n,n-1)) > abs(m_matT.coeff(n-1,n))) - { - m_matT.coeffRef(n-1,n-1) = q / m_matT.coeff(n,n-1); - m_matT.coeffRef(n-1,n) = -(m_matT.coeff(n,n) - p) / m_matT.coeff(n,n-1); - } - else - { - ComplexScalar cc = ComplexScalar(Scalar(0),-m_matT.coeff(n-1,n)) / ComplexScalar(m_matT.coeff(n-1,n-1)-p,q); - m_matT.coeffRef(n-1,n-1) = numext::real(cc); - m_matT.coeffRef(n-1,n) = numext::imag(cc); + if (abs(m_matT.coeff(n, n - 1)) > abs(m_matT.coeff(n - 1, n))) { + m_matT.coeffRef(n - 1, n - 1) = q / m_matT.coeff(n, n - 1); + m_matT.coeffRef(n - 1, n) = -(m_matT.coeff(n, n) - p) / m_matT.coeff(n, n - 1); + } else { + ComplexScalar cc = + ComplexScalar(Scalar(0), -m_matT.coeff(n - 1, n)) / ComplexScalar(m_matT.coeff(n - 1, n - 1) - p, q); + m_matT.coeffRef(n - 1, n - 1) = numext::real(cc); + m_matT.coeffRef(n - 1, n) = numext::imag(cc); } - m_matT.coeffRef(n,n-1) = Scalar(0); - m_matT.coeffRef(n,n) = Scalar(1); - for (Index i = n-2; i >= 0; i--) - { - Scalar ra = m_matT.row(i).segment(l, n-l+1).dot(m_matT.col(n-1).segment(l, n-l+1)); - Scalar sa = m_matT.row(i).segment(l, n-l+1).dot(m_matT.col(n).segment(l, n-l+1)); - Scalar w = m_matT.coeff(i,i) - p; - - if (m_eivalues.coeff(i).imag() < Scalar(0)) - { + m_matT.coeffRef(n, n - 1) = Scalar(0); + m_matT.coeffRef(n, n) = Scalar(1); + for (Index i = n - 2; i >= 0; i--) { + Scalar ra = m_matT.row(i).segment(l, n - l + 1).dot(m_matT.col(n - 1).segment(l, n - l + 1)); + Scalar sa = m_matT.row(i).segment(l, n - l + 1).dot(m_matT.col(n).segment(l, n - l + 1)); + Scalar w = m_matT.coeff(i, i) - p; + + if (m_eivalues.coeff(i).imag() < Scalar(0)) { lastw = w; lastra = ra; lastsa = sa; - } - else - { + } else { l = i; - if (m_eivalues.coeff(i).imag() == RealScalar(0)) - { - ComplexScalar cc = ComplexScalar(-ra,-sa) / ComplexScalar(w,q); - m_matT.coeffRef(i,n-1) = numext::real(cc); - m_matT.coeffRef(i,n) = numext::imag(cc); - } - else - { + if (m_eivalues.coeff(i).imag() == RealScalar(0)) { + ComplexScalar cc = ComplexScalar(-ra, -sa) / ComplexScalar(w, q); + m_matT.coeffRef(i, n - 1) = numext::real(cc); + m_matT.coeffRef(i, n) = numext::imag(cc); + } else { // Solve complex equations - Scalar x = m_matT.coeff(i,i+1); - Scalar y = m_matT.coeff(i+1,i); - Scalar vr = (m_eivalues.coeff(i).real() - p) * (m_eivalues.coeff(i).real() - p) + m_eivalues.coeff(i).imag() * m_eivalues.coeff(i).imag() - q * q; + Scalar x = m_matT.coeff(i, i + 1); + Scalar y = m_matT.coeff(i + 1, i); + Scalar vr = (m_eivalues.coeff(i).real() - p) * (m_eivalues.coeff(i).real() - p) + + m_eivalues.coeff(i).imag() * m_eivalues.coeff(i).imag() - q * q; Scalar vi = (m_eivalues.coeff(i).real() - p) * Scalar(2) * q; if ((vr == Scalar(0)) && (vi == Scalar(0))) vr = eps * norm * (abs(w) + abs(q) + abs(x) + abs(y) + abs(lastw)); - ComplexScalar cc = ComplexScalar(x*lastra-lastw*ra+q*sa,x*lastsa-lastw*sa-q*ra) / ComplexScalar(vr,vi); - m_matT.coeffRef(i,n-1) = numext::real(cc); - m_matT.coeffRef(i,n) = numext::imag(cc); - if (abs(x) > (abs(lastw) + abs(q))) - { - m_matT.coeffRef(i+1,n-1) = (-ra - w * m_matT.coeff(i,n-1) + q * m_matT.coeff(i,n)) / x; - m_matT.coeffRef(i+1,n) = (-sa - w * m_matT.coeff(i,n) - q * m_matT.coeff(i,n-1)) / x; - } - else - { - cc = ComplexScalar(-lastra-y*m_matT.coeff(i,n-1),-lastsa-y*m_matT.coeff(i,n)) / ComplexScalar(lastw,q); - m_matT.coeffRef(i+1,n-1) = numext::real(cc); - m_matT.coeffRef(i+1,n) = numext::imag(cc); + ComplexScalar cc = + ComplexScalar(x * lastra - lastw * ra + q * sa, x * lastsa - lastw * sa - q * ra) / ComplexScalar(vr, vi); + m_matT.coeffRef(i, n - 1) = numext::real(cc); + m_matT.coeffRef(i, n) = numext::imag(cc); + if (abs(x) > (abs(lastw) + abs(q))) { + m_matT.coeffRef(i + 1, n - 1) = (-ra - w * m_matT.coeff(i, n - 1) + q * m_matT.coeff(i, n)) / x; + m_matT.coeffRef(i + 1, n) = (-sa - w * m_matT.coeff(i, n) - q * m_matT.coeff(i, n - 1)) / x; + } else { + cc = ComplexScalar(-lastra - y * m_matT.coeff(i, n - 1), -lastsa - y * m_matT.coeff(i, n)) + / ComplexScalar(lastw, q); + m_matT.coeffRef(i + 1, n - 1) = numext::real(cc); + m_matT.coeffRef(i + 1, n) = numext::imag(cc); } } // Overflow control - Scalar t = numext::maxi(abs(m_matT.coeff(i,n-1)),abs(m_matT.coeff(i,n))); - if ((eps * t) * t > Scalar(1)) - m_matT.block(i, n-1, size-i, 2) /= t; - + Scalar t = numext::maxi(abs(m_matT.coeff(i, n - 1)), abs(m_matT.coeff(i, n))); + if ((eps * t) * t > Scalar(1)) m_matT.block(i, n - 1, size - i, 2) /= t; } } - + // We handled a pair of complex conjugate eigenvalues, so need to skip them both n--; - } - else - { - eigen_assert(0 && "Internal bug in EigenSolver (INF or NaN has not been detected)"); // this should not happen + } else { + eigen_assert(0 && "Internal bug in EigenSolver (INF or NaN has not been detected)");// this should not happen } } // Back transformation to get eigenvectors of original matrix - for (Index j = size-1; j >= 0; j--) - { - m_tmp.noalias() = m_eivec.leftCols(j+1) * m_matT.col(j).segment(0, j+1); + for (Index j = size - 1; j >= 0; j--) { + m_tmp.noalias() = m_eivec.leftCols(j + 1) * m_matT.col(j).segment(0, j + 1); m_eivec.col(j) = m_tmp; } } -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_EIGENSOLVER_H +#endif// EIGEN_EIGENSOLVER_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Eigenvalues/GeneralizedEigenSolver.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Eigenvalues/GeneralizedEigenSolver.h index 87d789b3..9ab9e5d3 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Eigenvalues/GeneralizedEigenSolver.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Eigenvalues/GeneralizedEigenSolver.h @@ -14,280 +14,269 @@ #include "./RealQZ.h" -namespace Eigen { +namespace Eigen { /** \eigenvalues_module \ingroup Eigenvalues_Module - * - * - * \class GeneralizedEigenSolver - * - * \brief Computes the generalized eigenvalues and eigenvectors of a pair of general matrices - * - * \tparam _MatrixType the type of the matrices of which we are computing the - * eigen-decomposition; this is expected to be an instantiation of the Matrix - * class template. Currently, only real matrices are supported. - * - * The generalized eigenvalues and eigenvectors of a matrix pair \f$ A \f$ and \f$ B \f$ are scalars - * \f$ \lambda \f$ and vectors \f$ v \f$ such that \f$ Av = \lambda Bv \f$. If - * \f$ D \f$ is a diagonal matrix with the eigenvalues on the diagonal, and - * \f$ V \f$ is a matrix with the eigenvectors as its columns, then \f$ A V = - * B V D \f$. The matrix \f$ V \f$ is almost always invertible, in which case we - * have \f$ A = B V D V^{-1} \f$. This is called the generalized eigen-decomposition. - * - * The generalized eigenvalues and eigenvectors of a matrix pair may be complex, even when the - * matrices are real. Moreover, the generalized eigenvalue might be infinite if the matrix B is - * singular. To workaround this difficulty, the eigenvalues are provided as a pair of complex \f$ \alpha \f$ - * and real \f$ \beta \f$ such that: \f$ \lambda_i = \alpha_i / \beta_i \f$. If \f$ \beta_i \f$ is (nearly) zero, - * then one can consider the well defined left eigenvalue \f$ \mu = \beta_i / \alpha_i\f$ such that: - * \f$ \mu_i A v_i = B v_i \f$, or even \f$ \mu_i u_i^T A = u_i^T B \f$ where \f$ u_i \f$ is - * called the left eigenvector. - * - * Call the function compute() to compute the generalized eigenvalues and eigenvectors of - * a given matrix pair. Alternatively, you can use the - * GeneralizedEigenSolver(const MatrixType&, const MatrixType&, bool) constructor which computes the - * eigenvalues and eigenvectors at construction time. Once the eigenvalue and - * eigenvectors are computed, they can be retrieved with the eigenvalues() and - * eigenvectors() functions. - * - * Here is an usage example of this class: - * Example: \include GeneralizedEigenSolver.cpp - * Output: \verbinclude GeneralizedEigenSolver.out - * - * \sa MatrixBase::eigenvalues(), class ComplexEigenSolver, class SelfAdjointEigenSolver - */ + * + * + * \class GeneralizedEigenSolver + * + * \brief Computes the generalized eigenvalues and eigenvectors of a pair of general matrices + * + * \tparam _MatrixType the type of the matrices of which we are computing the + * eigen-decomposition; this is expected to be an instantiation of the Matrix + * class template. Currently, only real matrices are supported. + * + * The generalized eigenvalues and eigenvectors of a matrix pair \f$ A \f$ and \f$ B \f$ are scalars + * \f$ \lambda \f$ and vectors \f$ v \f$ such that \f$ Av = \lambda Bv \f$. If + * \f$ D \f$ is a diagonal matrix with the eigenvalues on the diagonal, and + * \f$ V \f$ is a matrix with the eigenvectors as its columns, then \f$ A V = + * B V D \f$. The matrix \f$ V \f$ is almost always invertible, in which case we + * have \f$ A = B V D V^{-1} \f$. This is called the generalized eigen-decomposition. + * + * The generalized eigenvalues and eigenvectors of a matrix pair may be complex, even when the + * matrices are real. Moreover, the generalized eigenvalue might be infinite if the matrix B is + * singular. To workaround this difficulty, the eigenvalues are provided as a pair of complex \f$ \alpha \f$ + * and real \f$ \beta \f$ such that: \f$ \lambda_i = \alpha_i / \beta_i \f$. If \f$ \beta_i \f$ is (nearly) zero, + * then one can consider the well defined left eigenvalue \f$ \mu = \beta_i / \alpha_i\f$ such that: + * \f$ \mu_i A v_i = B v_i \f$, or even \f$ \mu_i u_i^T A = u_i^T B \f$ where \f$ u_i \f$ is + * called the left eigenvector. + * + * Call the function compute() to compute the generalized eigenvalues and eigenvectors of + * a given matrix pair. Alternatively, you can use the + * GeneralizedEigenSolver(const MatrixType&, const MatrixType&, bool) constructor which computes the + * eigenvalues and eigenvectors at construction time. Once the eigenvalue and + * eigenvectors are computed, they can be retrieved with the eigenvalues() and + * eigenvectors() functions. + * + * Here is an usage example of this class: + * Example: \include GeneralizedEigenSolver.cpp + * Output: \verbinclude GeneralizedEigenSolver.out + * + * \sa MatrixBase::eigenvalues(), class ComplexEigenSolver, class SelfAdjointEigenSolver + */ template class GeneralizedEigenSolver { - public: - - /** \brief Synonym for the template parameter \p _MatrixType. */ - typedef _MatrixType MatrixType; - - enum { - RowsAtCompileTime = MatrixType::RowsAtCompileTime, - ColsAtCompileTime = MatrixType::ColsAtCompileTime, - Options = MatrixType::Options, - MaxRowsAtCompileTime = MatrixType::MaxRowsAtCompileTime, - MaxColsAtCompileTime = MatrixType::MaxColsAtCompileTime - }; - - /** \brief Scalar type for matrices of type #MatrixType. */ - typedef typename MatrixType::Scalar Scalar; - typedef typename NumTraits::Real RealScalar; - typedef Eigen::Index Index; ///< \deprecated since Eigen 3.3 - - /** \brief Complex scalar type for #MatrixType. - * - * This is \c std::complex if #Scalar is real (e.g., - * \c float or \c double) and just \c Scalar if #Scalar is - * complex. - */ - typedef std::complex ComplexScalar; - - /** \brief Type for vector of real scalar values eigenvalues as returned by betas(). - * - * This is a column vector with entries of type #Scalar. - * The length of the vector is the size of #MatrixType. - */ - typedef Matrix VectorType; - - /** \brief Type for vector of complex scalar values eigenvalues as returned by alphas(). - * - * This is a column vector with entries of type #ComplexScalar. - * The length of the vector is the size of #MatrixType. - */ - typedef Matrix ComplexVectorType; - - /** \brief Expression type for the eigenvalues as returned by eigenvalues(). - */ - typedef CwiseBinaryOp,ComplexVectorType,VectorType> EigenvalueType; - - /** \brief Type for matrix of eigenvectors as returned by eigenvectors(). - * - * This is a square matrix with entries of type #ComplexScalar. - * The size is the same as the size of #MatrixType. - */ - typedef Matrix EigenvectorsType; - - /** \brief Default constructor. - * - * The default constructor is useful in cases in which the user intends to - * perform decompositions via EigenSolver::compute(const MatrixType&, bool). - * - * \sa compute() for an example. - */ - GeneralizedEigenSolver() - : m_eivec(), - m_alphas(), - m_betas(), - m_valuesOkay(false), - m_vectorsOkay(false), - m_realQZ() - {} - - /** \brief Default constructor with memory preallocation - * - * Like the default constructor but with preallocation of the internal data - * according to the specified problem \a size. - * \sa GeneralizedEigenSolver() - */ - explicit GeneralizedEigenSolver(Index size) - : m_eivec(size, size), - m_alphas(size), - m_betas(size), - m_valuesOkay(false), - m_vectorsOkay(false), - m_realQZ(size), - m_tmp(size) - {} - - /** \brief Constructor; computes the generalized eigendecomposition of given matrix pair. - * - * \param[in] A Square matrix whose eigendecomposition is to be computed. - * \param[in] B Square matrix whose eigendecomposition is to be computed. - * \param[in] computeEigenvectors If true, both the eigenvectors and the - * eigenvalues are computed; if false, only the eigenvalues are computed. - * - * This constructor calls compute() to compute the generalized eigenvalues - * and eigenvectors. - * - * \sa compute() - */ - GeneralizedEigenSolver(const MatrixType& A, const MatrixType& B, bool computeEigenvectors = true) - : m_eivec(A.rows(), A.cols()), - m_alphas(A.cols()), - m_betas(A.cols()), - m_valuesOkay(false), - m_vectorsOkay(false), - m_realQZ(A.cols()), - m_tmp(A.cols()) - { - compute(A, B, computeEigenvectors); - } +public: + /** \brief Synonym for the template parameter \p _MatrixType. */ + typedef _MatrixType MatrixType; + + enum { + RowsAtCompileTime = MatrixType::RowsAtCompileTime, + ColsAtCompileTime = MatrixType::ColsAtCompileTime, + Options = MatrixType::Options, + MaxRowsAtCompileTime = MatrixType::MaxRowsAtCompileTime, + MaxColsAtCompileTime = MatrixType::MaxColsAtCompileTime + }; + + /** \brief Scalar type for matrices of type #MatrixType. */ + typedef typename MatrixType::Scalar Scalar; + typedef typename NumTraits::Real RealScalar; + typedef Eigen::Index Index;///< \deprecated since Eigen 3.3 + + /** \brief Complex scalar type for #MatrixType. + * + * This is \c std::complex if #Scalar is real (e.g., + * \c float or \c double) and just \c Scalar if #Scalar is + * complex. + */ + typedef std::complex ComplexScalar; + + /** \brief Type for vector of real scalar values eigenvalues as returned by betas(). + * + * This is a column vector with entries of type #Scalar. + * The length of the vector is the size of #MatrixType. + */ + typedef Matrix VectorType; + + /** \brief Type for vector of complex scalar values eigenvalues as returned by alphas(). + * + * This is a column vector with entries of type #ComplexScalar. + * The length of the vector is the size of #MatrixType. + */ + typedef Matrix ComplexVectorType; + + /** \brief Expression type for the eigenvalues as returned by eigenvalues(). + */ + typedef CwiseBinaryOp, ComplexVectorType, VectorType> + EigenvalueType; + + /** \brief Type for matrix of eigenvectors as returned by eigenvectors(). + * + * This is a square matrix with entries of type #ComplexScalar. + * The size is the same as the size of #MatrixType. + */ + typedef Matrix + EigenvectorsType; + + /** \brief Default constructor. + * + * The default constructor is useful in cases in which the user intends to + * perform decompositions via EigenSolver::compute(const MatrixType&, bool). + * + * \sa compute() for an example. + */ + GeneralizedEigenSolver() : m_eivec(), m_alphas(), m_betas(), m_valuesOkay(false), m_vectorsOkay(false), m_realQZ() {} + + /** \brief Default constructor with memory preallocation + * + * Like the default constructor but with preallocation of the internal data + * according to the specified problem \a size. + * \sa GeneralizedEigenSolver() + */ + explicit GeneralizedEigenSolver(Index size) + : m_eivec(size, size), m_alphas(size), m_betas(size), m_valuesOkay(false), m_vectorsOkay(false), m_realQZ(size), + m_tmp(size) + {} + + /** \brief Constructor; computes the generalized eigendecomposition of given matrix pair. + * + * \param[in] A Square matrix whose eigendecomposition is to be computed. + * \param[in] B Square matrix whose eigendecomposition is to be computed. + * \param[in] computeEigenvectors If true, both the eigenvectors and the + * eigenvalues are computed; if false, only the eigenvalues are computed. + * + * This constructor calls compute() to compute the generalized eigenvalues + * and eigenvectors. + * + * \sa compute() + */ + GeneralizedEigenSolver(const MatrixType &A, const MatrixType &B, bool computeEigenvectors = true) + : m_eivec(A.rows(), A.cols()), m_alphas(A.cols()), m_betas(A.cols()), m_valuesOkay(false), m_vectorsOkay(false), + m_realQZ(A.cols()), m_tmp(A.cols()) + { + compute(A, B, computeEigenvectors); + } - /* \brief Returns the computed generalized eigenvectors. - * - * \returns %Matrix whose columns are the (possibly complex) right eigenvectors. - * i.e. the eigenvectors that solve (A - l*B)x = 0. The ordering matches the eigenvalues. - * - * \pre Either the constructor - * GeneralizedEigenSolver(const MatrixType&,const MatrixType&, bool) or the member function - * compute(const MatrixType&, const MatrixType& bool) has been called before, and - * \p computeEigenvectors was set to true (the default). - * - * \sa eigenvalues() - */ - EigenvectorsType eigenvectors() const { - eigen_assert(m_vectorsOkay && "Eigenvectors for GeneralizedEigenSolver were not calculated."); - return m_eivec; - } + /* \brief Returns the computed generalized eigenvectors. + * + * \returns %Matrix whose columns are the (possibly complex) right eigenvectors. + * i.e. the eigenvectors that solve (A - l*B)x = 0. The ordering matches the eigenvalues. + * + * \pre Either the constructor + * GeneralizedEigenSolver(const MatrixType&,const MatrixType&, bool) or the member function + * compute(const MatrixType&, const MatrixType& bool) has been called before, and + * \p computeEigenvectors was set to true (the default). + * + * \sa eigenvalues() + */ + EigenvectorsType eigenvectors() const + { + eigen_assert(m_vectorsOkay && "Eigenvectors for GeneralizedEigenSolver were not calculated."); + return m_eivec; + } - /** \brief Returns an expression of the computed generalized eigenvalues. - * - * \returns An expression of the column vector containing the eigenvalues. - * - * It is a shortcut for \code this->alphas().cwiseQuotient(this->betas()); \endcode - * Not that betas might contain zeros. It is therefore not recommended to use this function, - * but rather directly deal with the alphas and betas vectors. - * - * \pre Either the constructor - * GeneralizedEigenSolver(const MatrixType&,const MatrixType&,bool) or the member function - * compute(const MatrixType&,const MatrixType&,bool) has been called before. - * - * The eigenvalues are repeated according to their algebraic multiplicity, - * so there are as many eigenvalues as rows in the matrix. The eigenvalues - * are not sorted in any particular order. - * - * \sa alphas(), betas(), eigenvectors() - */ - EigenvalueType eigenvalues() const - { - eigen_assert(m_valuesOkay && "GeneralizedEigenSolver is not initialized."); - return EigenvalueType(m_alphas,m_betas); - } + /** \brief Returns an expression of the computed generalized eigenvalues. + * + * \returns An expression of the column vector containing the eigenvalues. + * + * It is a shortcut for \code this->alphas().cwiseQuotient(this->betas()); \endcode + * Not that betas might contain zeros. It is therefore not recommended to use this function, + * but rather directly deal with the alphas and betas vectors. + * + * \pre Either the constructor + * GeneralizedEigenSolver(const MatrixType&,const MatrixType&,bool) or the member function + * compute(const MatrixType&,const MatrixType&,bool) has been called before. + * + * The eigenvalues are repeated according to their algebraic multiplicity, + * so there are as many eigenvalues as rows in the matrix. The eigenvalues + * are not sorted in any particular order. + * + * \sa alphas(), betas(), eigenvectors() + */ + EigenvalueType eigenvalues() const + { + eigen_assert(m_valuesOkay && "GeneralizedEigenSolver is not initialized."); + return EigenvalueType(m_alphas, m_betas); + } - /** \returns A const reference to the vectors containing the alpha values - * - * This vector permits to reconstruct the j-th eigenvalues as alphas(i)/betas(j). - * - * \sa betas(), eigenvalues() */ - ComplexVectorType alphas() const - { - eigen_assert(m_valuesOkay && "GeneralizedEigenSolver is not initialized."); - return m_alphas; - } + /** \returns A const reference to the vectors containing the alpha values + * + * This vector permits to reconstruct the j-th eigenvalues as alphas(i)/betas(j). + * + * \sa betas(), eigenvalues() */ + ComplexVectorType alphas() const + { + eigen_assert(m_valuesOkay && "GeneralizedEigenSolver is not initialized."); + return m_alphas; + } - /** \returns A const reference to the vectors containing the beta values - * - * This vector permits to reconstruct the j-th eigenvalues as alphas(i)/betas(j). - * - * \sa alphas(), eigenvalues() */ - VectorType betas() const - { - eigen_assert(m_valuesOkay && "GeneralizedEigenSolver is not initialized."); - return m_betas; - } + /** \returns A const reference to the vectors containing the beta values + * + * This vector permits to reconstruct the j-th eigenvalues as alphas(i)/betas(j). + * + * \sa alphas(), eigenvalues() */ + VectorType betas() const + { + eigen_assert(m_valuesOkay && "GeneralizedEigenSolver is not initialized."); + return m_betas; + } - /** \brief Computes generalized eigendecomposition of given matrix. - * - * \param[in] A Square matrix whose eigendecomposition is to be computed. - * \param[in] B Square matrix whose eigendecomposition is to be computed. - * \param[in] computeEigenvectors If true, both the eigenvectors and the - * eigenvalues are computed; if false, only the eigenvalues are - * computed. - * \returns Reference to \c *this - * - * This function computes the eigenvalues of the real matrix \p matrix. - * The eigenvalues() function can be used to retrieve them. If - * \p computeEigenvectors is true, then the eigenvectors are also computed - * and can be retrieved by calling eigenvectors(). - * - * The matrix is first reduced to real generalized Schur form using the RealQZ - * class. The generalized Schur decomposition is then used to compute the eigenvalues - * and eigenvectors. - * - * The cost of the computation is dominated by the cost of the - * generalized Schur decomposition. - * - * This method reuses of the allocated data in the GeneralizedEigenSolver object. - */ - GeneralizedEigenSolver& compute(const MatrixType& A, const MatrixType& B, bool computeEigenvectors = true); - - ComputationInfo info() const - { - eigen_assert(m_valuesOkay && "EigenSolver is not initialized."); - return m_realQZ.info(); - } + /** \brief Computes generalized eigendecomposition of given matrix. + * + * \param[in] A Square matrix whose eigendecomposition is to be computed. + * \param[in] B Square matrix whose eigendecomposition is to be computed. + * \param[in] computeEigenvectors If true, both the eigenvectors and the + * eigenvalues are computed; if false, only the eigenvalues are + * computed. + * \returns Reference to \c *this + * + * This function computes the eigenvalues of the real matrix \p matrix. + * The eigenvalues() function can be used to retrieve them. If + * \p computeEigenvectors is true, then the eigenvectors are also computed + * and can be retrieved by calling eigenvectors(). + * + * The matrix is first reduced to real generalized Schur form using the RealQZ + * class. The generalized Schur decomposition is then used to compute the eigenvalues + * and eigenvectors. + * + * The cost of the computation is dominated by the cost of the + * generalized Schur decomposition. + * + * This method reuses of the allocated data in the GeneralizedEigenSolver object. + */ + GeneralizedEigenSolver &compute(const MatrixType &A, const MatrixType &B, bool computeEigenvectors = true); + + ComputationInfo info() const + { + eigen_assert(m_valuesOkay && "EigenSolver is not initialized."); + return m_realQZ.info(); + } - /** Sets the maximal number of iterations allowed. - */ - GeneralizedEigenSolver& setMaxIterations(Index maxIters) - { - m_realQZ.setMaxIterations(maxIters); - return *this; - } + /** Sets the maximal number of iterations allowed. + */ + GeneralizedEigenSolver &setMaxIterations(Index maxIters) + { + m_realQZ.setMaxIterations(maxIters); + return *this; + } - protected: - - static void check_template_parameters() - { - EIGEN_STATIC_ASSERT_NON_INTEGER(Scalar); - EIGEN_STATIC_ASSERT(!NumTraits::IsComplex, NUMERIC_TYPE_MUST_BE_REAL); - } - - EigenvectorsType m_eivec; - ComplexVectorType m_alphas; - VectorType m_betas; - bool m_valuesOkay, m_vectorsOkay; - RealQZ m_realQZ; - ComplexVectorType m_tmp; +protected: + static void check_template_parameters() + { + EIGEN_STATIC_ASSERT_NON_INTEGER(Scalar); + EIGEN_STATIC_ASSERT(!NumTraits::IsComplex, NUMERIC_TYPE_MUST_BE_REAL); + } + + EigenvectorsType m_eivec; + ComplexVectorType m_alphas; + VectorType m_betas; + bool m_valuesOkay, m_vectorsOkay; + RealQZ m_realQZ; + ComplexVectorType m_tmp; }; template -GeneralizedEigenSolver& -GeneralizedEigenSolver::compute(const MatrixType& A, const MatrixType& B, bool computeEigenvectors) +GeneralizedEigenSolver & + GeneralizedEigenSolver::compute(const MatrixType &A, const MatrixType &B, bool computeEigenvectors) { check_template_parameters(); - + using std::sqrt; using std::abs; eigen_assert(A.cols() == A.rows() && B.cols() == A.rows() && B.cols() == B.rows()); @@ -297,56 +286,53 @@ GeneralizedEigenSolver::compute(const MatrixType& A, const MatrixTyp // Reduce to generalized real Schur form: // A = Q S Z and B = Q T Z m_realQZ.compute(A, B, computeEigenvectors); - if (m_realQZ.info() == Success) - { + if (m_realQZ.info() == Success) { // Resize storage m_alphas.resize(size); m_betas.resize(size); - if (computeEigenvectors) - { - m_eivec.resize(size,size); + if (computeEigenvectors) { + m_eivec.resize(size, size); m_tmp.resize(size); } // Aliases: - Map v(reinterpret_cast(m_tmp.data()), size); + Map v(reinterpret_cast(m_tmp.data()), size); ComplexVectorType &cv = m_tmp; const MatrixType &mS = m_realQZ.matrixS(); const MatrixType &mT = m_realQZ.matrixT(); Index i = 0; - while (i < size) - { - if (i == size - 1 || mS.coeff(i+1, i) == Scalar(0)) - { + while (i < size) { + if (i == size - 1 || mS.coeff(i + 1, i) == Scalar(0)) { // Real eigenvalue m_alphas.coeffRef(i) = mS.diagonal().coeff(i); - m_betas.coeffRef(i) = mT.diagonal().coeff(i); - if (computeEigenvectors) - { + m_betas.coeffRef(i) = mT.diagonal().coeff(i); + if (computeEigenvectors) { v.setConstant(Scalar(0.0)); v.coeffRef(i) = Scalar(1.0); // For singular eigenvalues do nothing more - if(abs(m_betas.coeffRef(i)) >= (std::numeric_limits::min)()) - { + if (abs(m_betas.coeffRef(i)) >= (std::numeric_limits::min)()) { // Non-singular eigenvalue const Scalar alpha = real(m_alphas.coeffRef(i)); const Scalar beta = m_betas.coeffRef(i); - for (Index j = i-1; j >= 0; j--) - { - const Index st = j+1; - const Index sz = i-j; - if (j > 0 && mS.coeff(j, j-1) != Scalar(0)) - { + for (Index j = i - 1; j >= 0; j--) { + const Index st = j + 1; + const Index sz = i - j; + if (j > 0 && mS.coeff(j, j - 1) != Scalar(0)) { // 2x2 block - Matrix rhs = (alpha*mT.template block<2,Dynamic>(j-1,st,2,sz) - beta*mS.template block<2,Dynamic>(j-1,st,2,sz)) .lazyProduct( v.segment(st,sz) ); - Matrix lhs = beta * mS.template block<2,2>(j-1,j-1) - alpha * mT.template block<2,2>(j-1,j-1); - v.template segment<2>(j-1) = lhs.partialPivLu().solve(rhs); + Matrix rhs = (alpha * mT.template block<2, Dynamic>(j - 1, st, 2, sz) + - beta * mS.template block<2, Dynamic>(j - 1, st, 2, sz)) + .lazyProduct(v.segment(st, sz)); + Matrix lhs = + beta * mS.template block<2, 2>(j - 1, j - 1) - alpha * mT.template block<2, 2>(j - 1, j - 1); + v.template segment<2>(j - 1) = lhs.partialPivLu().solve(rhs); j--; - } - else - { - v.coeffRef(j) = -v.segment(st,sz).transpose().cwiseProduct(beta*mS.block(j,st,1,sz) - alpha*mT.block(j,st,1,sz)).sum() / (beta*mS.coeffRef(j,j) - alpha*mT.coeffRef(j,j)); + } else { + v.coeffRef(j) = -v.segment(st, sz) + .transpose() + .cwiseProduct(beta * mS.block(j, st, 1, sz) - alpha * mT.block(j, st, 1, sz)) + .sum() + / (beta * mS.coeffRef(j, j) - alpha * mT.coeffRef(j, j)); } } } @@ -355,53 +341,55 @@ GeneralizedEigenSolver::compute(const MatrixType& A, const MatrixTyp m_eivec.col(i).imag().setConstant(0); } ++i; - } - else - { - // We need to extract the generalized eigenvalues of the pair of a general 2x2 block S and a positive diagonal 2x2 block T - // Then taking beta=T_00*T_11, we can avoid any division, and alpha is the eigenvalues of A = (U^-1 * S * U) * diag(T_11,T_00): + } else { + // We need to extract the generalized eigenvalues of the pair of a general 2x2 block S and a positive diagonal + // 2x2 block T Then taking beta=T_00*T_11, we can avoid any division, and alpha is the eigenvalues of A = (U^-1 + // * S * U) * diag(T_11,T_00): // T = [a 0] // [0 b] - RealScalar a = mT.diagonal().coeff(i), - b = mT.diagonal().coeff(i+1); - const RealScalar beta = m_betas.coeffRef(i) = m_betas.coeffRef(i+1) = a*b; + RealScalar a = mT.diagonal().coeff(i), b = mT.diagonal().coeff(i + 1); + const RealScalar beta = m_betas.coeffRef(i) = m_betas.coeffRef(i + 1) = a * b; // ^^ NOTE: using diagonal()(i) instead of coeff(i,i) workarounds a MSVC bug. - Matrix S2 = mS.template block<2,2>(i,i) * Matrix(b,a).asDiagonal(); + Matrix S2 = mS.template block<2, 2>(i, i) * Matrix(b, a).asDiagonal(); - Scalar p = Scalar(0.5) * (S2.coeff(0,0) - S2.coeff(1,1)); - Scalar z = sqrt(abs(p * p + S2.coeff(1,0) * S2.coeff(0,1))); - const ComplexScalar alpha = ComplexScalar(S2.coeff(1,1) + p, (beta > 0) ? z : -z); - m_alphas.coeffRef(i) = conj(alpha); - m_alphas.coeffRef(i+1) = alpha; + Scalar p = Scalar(0.5) * (S2.coeff(0, 0) - S2.coeff(1, 1)); + Scalar z = sqrt(abs(p * p + S2.coeff(1, 0) * S2.coeff(0, 1))); + const ComplexScalar alpha = ComplexScalar(S2.coeff(1, 1) + p, (beta > 0) ? z : -z); + m_alphas.coeffRef(i) = conj(alpha); + m_alphas.coeffRef(i + 1) = alpha; if (computeEigenvectors) { // Compute eigenvector in position (i+1) and then position (i) is just the conjugate cv.setZero(); - cv.coeffRef(i+1) = Scalar(1.0); + cv.coeffRef(i + 1) = Scalar(1.0); // here, the "static_cast" workaound expression template issues. - cv.coeffRef(i) = -(static_cast(beta*mS.coeffRef(i,i+1)) - alpha*mT.coeffRef(i,i+1)) - / (static_cast(beta*mS.coeffRef(i,i)) - alpha*mT.coeffRef(i,i)); - for (Index j = i-1; j >= 0; j--) - { - const Index st = j+1; - const Index sz = i+1-j; - if (j > 0 && mS.coeff(j, j-1) != Scalar(0)) - { + cv.coeffRef(i) = -(static_cast(beta * mS.coeffRef(i, i + 1)) - alpha * mT.coeffRef(i, i + 1)) + / (static_cast(beta * mS.coeffRef(i, i)) - alpha * mT.coeffRef(i, i)); + for (Index j = i - 1; j >= 0; j--) { + const Index st = j + 1; + const Index sz = i + 1 - j; + if (j > 0 && mS.coeff(j, j - 1) != Scalar(0)) { // 2x2 block - Matrix rhs = (alpha*mT.template block<2,Dynamic>(j-1,st,2,sz) - beta*mS.template block<2,Dynamic>(j-1,st,2,sz)) .lazyProduct( cv.segment(st,sz) ); - Matrix lhs = beta * mS.template block<2,2>(j-1,j-1) - alpha * mT.template block<2,2>(j-1,j-1); - cv.template segment<2>(j-1) = lhs.partialPivLu().solve(rhs); + Matrix rhs = (alpha * mT.template block<2, Dynamic>(j - 1, st, 2, sz) + - beta * mS.template block<2, Dynamic>(j - 1, st, 2, sz)) + .lazyProduct(cv.segment(st, sz)); + Matrix lhs = + beta * mS.template block<2, 2>(j - 1, j - 1) - alpha * mT.template block<2, 2>(j - 1, j - 1); + cv.template segment<2>(j - 1) = lhs.partialPivLu().solve(rhs); j--; } else { - cv.coeffRef(j) = cv.segment(st,sz).transpose().cwiseProduct(beta*mS.block(j,st,1,sz) - alpha*mT.block(j,st,1,sz)).sum() - / (alpha*mT.coeffRef(j,j) - static_cast(beta*mS.coeffRef(j,j))); + cv.coeffRef(j) = cv.segment(st, sz) + .transpose() + .cwiseProduct(beta * mS.block(j, st, 1, sz) - alpha * mT.block(j, st, 1, sz)) + .sum() + / (alpha * mT.coeffRef(j, j) - static_cast(beta * mS.coeffRef(j, j))); } } - m_eivec.col(i+1).noalias() = (m_realQZ.matrixZ().transpose() * cv); - m_eivec.col(i+1).normalize(); - m_eivec.col(i) = m_eivec.col(i+1).conjugate(); + m_eivec.col(i + 1).noalias() = (m_realQZ.matrixZ().transpose() * cv); + m_eivec.col(i + 1).normalize(); + m_eivec.col(i) = m_eivec.col(i + 1).conjugate(); } i += 2; } @@ -413,6 +401,6 @@ GeneralizedEigenSolver::compute(const MatrixType& A, const MatrixTyp return *this; } -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_GENERALIZEDEIGENSOLVER_H +#endif// EIGEN_GENERALIZEDEIGENSOLVER_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Eigenvalues/GeneralizedSelfAdjointEigenSolver.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Eigenvalues/GeneralizedSelfAdjointEigenSolver.h index 5f6bb828..d5b77854 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Eigenvalues/GeneralizedSelfAdjointEigenSolver.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Eigenvalues/GeneralizedSelfAdjointEigenSolver.h @@ -13,186 +13,177 @@ #include "./Tridiagonalization.h" -namespace Eigen { +namespace Eigen { /** \eigenvalues_module \ingroup Eigenvalues_Module - * - * - * \class GeneralizedSelfAdjointEigenSolver - * - * \brief Computes eigenvalues and eigenvectors of the generalized selfadjoint eigen problem - * - * \tparam _MatrixType the type of the matrix of which we are computing the - * eigendecomposition; this is expected to be an instantiation of the Matrix - * class template. - * - * This class solves the generalized eigenvalue problem - * \f$ Av = \lambda Bv \f$. In this case, the matrix \f$ A \f$ should be - * selfadjoint and the matrix \f$ B \f$ should be positive definite. - * - * Only the \b lower \b triangular \b part of the input matrix is referenced. - * - * Call the function compute() to compute the eigenvalues and eigenvectors of - * a given matrix. Alternatively, you can use the - * GeneralizedSelfAdjointEigenSolver(const MatrixType&, const MatrixType&, int) - * constructor which computes the eigenvalues and eigenvectors at construction time. - * Once the eigenvalue and eigenvectors are computed, they can be retrieved with the eigenvalues() - * and eigenvectors() functions. - * - * The documentation for GeneralizedSelfAdjointEigenSolver(const MatrixType&, const MatrixType&, int) - * contains an example of the typical use of this class. - * - * \sa class SelfAdjointEigenSolver, class EigenSolver, class ComplexEigenSolver - */ -template -class GeneralizedSelfAdjointEigenSolver : public SelfAdjointEigenSolver<_MatrixType> + * + * + * \class GeneralizedSelfAdjointEigenSolver + * + * \brief Computes eigenvalues and eigenvectors of the generalized selfadjoint eigen problem + * + * \tparam _MatrixType the type of the matrix of which we are computing the + * eigendecomposition; this is expected to be an instantiation of the Matrix + * class template. + * + * This class solves the generalized eigenvalue problem + * \f$ Av = \lambda Bv \f$. In this case, the matrix \f$ A \f$ should be + * selfadjoint and the matrix \f$ B \f$ should be positive definite. + * + * Only the \b lower \b triangular \b part of the input matrix is referenced. + * + * Call the function compute() to compute the eigenvalues and eigenvectors of + * a given matrix. Alternatively, you can use the + * GeneralizedSelfAdjointEigenSolver(const MatrixType&, const MatrixType&, int) + * constructor which computes the eigenvalues and eigenvectors at construction time. + * Once the eigenvalue and eigenvectors are computed, they can be retrieved with the eigenvalues() + * and eigenvectors() functions. + * + * The documentation for GeneralizedSelfAdjointEigenSolver(const MatrixType&, const MatrixType&, int) + * contains an example of the typical use of this class. + * + * \sa class SelfAdjointEigenSolver, class EigenSolver, class ComplexEigenSolver + */ +template class GeneralizedSelfAdjointEigenSolver : public SelfAdjointEigenSolver<_MatrixType> { - typedef SelfAdjointEigenSolver<_MatrixType> Base; - public: - - typedef _MatrixType MatrixType; - - /** \brief Default constructor for fixed-size matrices. - * - * The default constructor is useful in cases in which the user intends to - * perform decompositions via compute(). This constructor - * can only be used if \p _MatrixType is a fixed-size matrix; use - * GeneralizedSelfAdjointEigenSolver(Index) for dynamic-size matrices. - */ - GeneralizedSelfAdjointEigenSolver() : Base() {} - - /** \brief Constructor, pre-allocates memory for dynamic-size matrices. - * - * \param [in] size Positive integer, size of the matrix whose - * eigenvalues and eigenvectors will be computed. - * - * This constructor is useful for dynamic-size matrices, when the user - * intends to perform decompositions via compute(). The \p size - * parameter is only used as a hint. It is not an error to give a wrong - * \p size, but it may impair performance. - * - * \sa compute() for an example - */ - explicit GeneralizedSelfAdjointEigenSolver(Index size) - : Base(size) - {} - - /** \brief Constructor; computes generalized eigendecomposition of given matrix pencil. - * - * \param[in] matA Selfadjoint matrix in matrix pencil. - * Only the lower triangular part of the matrix is referenced. - * \param[in] matB Positive-definite matrix in matrix pencil. - * Only the lower triangular part of the matrix is referenced. - * \param[in] options A or-ed set of flags {#ComputeEigenvectors,#EigenvaluesOnly} | {#Ax_lBx,#ABx_lx,#BAx_lx}. - * Default is #ComputeEigenvectors|#Ax_lBx. - * - * This constructor calls compute(const MatrixType&, const MatrixType&, int) - * to compute the eigenvalues and (if requested) the eigenvectors of the - * generalized eigenproblem \f$ Ax = \lambda B x \f$ with \a matA the - * selfadjoint matrix \f$ A \f$ and \a matB the positive definite matrix - * \f$ B \f$. Each eigenvector \f$ x \f$ satisfies the property - * \f$ x^* B x = 1 \f$. The eigenvectors are computed if - * \a options contains ComputeEigenvectors. - * - * In addition, the two following variants can be solved via \p options: - * - \c ABx_lx: \f$ ABx = \lambda x \f$ - * - \c BAx_lx: \f$ BAx = \lambda x \f$ - * - * Example: \include SelfAdjointEigenSolver_SelfAdjointEigenSolver_MatrixType2.cpp - * Output: \verbinclude SelfAdjointEigenSolver_SelfAdjointEigenSolver_MatrixType2.out - * - * \sa compute(const MatrixType&, const MatrixType&, int) - */ - GeneralizedSelfAdjointEigenSolver(const MatrixType& matA, const MatrixType& matB, - int options = ComputeEigenvectors|Ax_lBx) - : Base(matA.cols()) - { - compute(matA, matB, options); - } - - /** \brief Computes generalized eigendecomposition of given matrix pencil. - * - * \param[in] matA Selfadjoint matrix in matrix pencil. - * Only the lower triangular part of the matrix is referenced. - * \param[in] matB Positive-definite matrix in matrix pencil. - * Only the lower triangular part of the matrix is referenced. - * \param[in] options A or-ed set of flags {#ComputeEigenvectors,#EigenvaluesOnly} | {#Ax_lBx,#ABx_lx,#BAx_lx}. - * Default is #ComputeEigenvectors|#Ax_lBx. - * - * \returns Reference to \c *this - * - * Accoring to \p options, this function computes eigenvalues and (if requested) - * the eigenvectors of one of the following three generalized eigenproblems: - * - \c Ax_lBx: \f$ Ax = \lambda B x \f$ - * - \c ABx_lx: \f$ ABx = \lambda x \f$ - * - \c BAx_lx: \f$ BAx = \lambda x \f$ - * with \a matA the selfadjoint matrix \f$ A \f$ and \a matB the positive definite - * matrix \f$ B \f$. - * In addition, each eigenvector \f$ x \f$ satisfies the property \f$ x^* B x = 1 \f$. - * - * The eigenvalues() function can be used to retrieve - * the eigenvalues. If \p options contains ComputeEigenvectors, then the - * eigenvectors are also computed and can be retrieved by calling - * eigenvectors(). - * - * The implementation uses LLT to compute the Cholesky decomposition - * \f$ B = LL^* \f$ and computes the classical eigendecomposition - * of the selfadjoint matrix \f$ L^{-1} A (L^*)^{-1} \f$ if \p options contains Ax_lBx - * and of \f$ L^{*} A L \f$ otherwise. This solves the - * generalized eigenproblem, because any solution of the generalized - * eigenproblem \f$ Ax = \lambda B x \f$ corresponds to a solution - * \f$ L^{-1} A (L^*)^{-1} (L^* x) = \lambda (L^* x) \f$ of the - * eigenproblem for \f$ L^{-1} A (L^*)^{-1} \f$. Similar statements - * can be made for the two other variants. - * - * Example: \include SelfAdjointEigenSolver_compute_MatrixType2.cpp - * Output: \verbinclude SelfAdjointEigenSolver_compute_MatrixType2.out - * - * \sa GeneralizedSelfAdjointEigenSolver(const MatrixType&, const MatrixType&, int) - */ - GeneralizedSelfAdjointEigenSolver& compute(const MatrixType& matA, const MatrixType& matB, - int options = ComputeEigenvectors|Ax_lBx); - - protected: + typedef SelfAdjointEigenSolver<_MatrixType> Base; + +public: + typedef _MatrixType MatrixType; + + /** \brief Default constructor for fixed-size matrices. + * + * The default constructor is useful in cases in which the user intends to + * perform decompositions via compute(). This constructor + * can only be used if \p _MatrixType is a fixed-size matrix; use + * GeneralizedSelfAdjointEigenSolver(Index) for dynamic-size matrices. + */ + GeneralizedSelfAdjointEigenSolver() : Base() {} + + /** \brief Constructor, pre-allocates memory for dynamic-size matrices. + * + * \param [in] size Positive integer, size of the matrix whose + * eigenvalues and eigenvectors will be computed. + * + * This constructor is useful for dynamic-size matrices, when the user + * intends to perform decompositions via compute(). The \p size + * parameter is only used as a hint. It is not an error to give a wrong + * \p size, but it may impair performance. + * + * \sa compute() for an example + */ + explicit GeneralizedSelfAdjointEigenSolver(Index size) : Base(size) {} + + /** \brief Constructor; computes generalized eigendecomposition of given matrix pencil. + * + * \param[in] matA Selfadjoint matrix in matrix pencil. + * Only the lower triangular part of the matrix is referenced. + * \param[in] matB Positive-definite matrix in matrix pencil. + * Only the lower triangular part of the matrix is referenced. + * \param[in] options A or-ed set of flags {#ComputeEigenvectors,#EigenvaluesOnly} | {#Ax_lBx,#ABx_lx,#BAx_lx}. + * Default is #ComputeEigenvectors|#Ax_lBx. + * + * This constructor calls compute(const MatrixType&, const MatrixType&, int) + * to compute the eigenvalues and (if requested) the eigenvectors of the + * generalized eigenproblem \f$ Ax = \lambda B x \f$ with \a matA the + * selfadjoint matrix \f$ A \f$ and \a matB the positive definite matrix + * \f$ B \f$. Each eigenvector \f$ x \f$ satisfies the property + * \f$ x^* B x = 1 \f$. The eigenvectors are computed if + * \a options contains ComputeEigenvectors. + * + * In addition, the two following variants can be solved via \p options: + * - \c ABx_lx: \f$ ABx = \lambda x \f$ + * - \c BAx_lx: \f$ BAx = \lambda x \f$ + * + * Example: \include SelfAdjointEigenSolver_SelfAdjointEigenSolver_MatrixType2.cpp + * Output: \verbinclude SelfAdjointEigenSolver_SelfAdjointEigenSolver_MatrixType2.out + * + * \sa compute(const MatrixType&, const MatrixType&, int) + */ + GeneralizedSelfAdjointEigenSolver(const MatrixType &matA, + const MatrixType &matB, + int options = ComputeEigenvectors | Ax_lBx) + : Base(matA.cols()) + { + compute(matA, matB, options); + } + /** \brief Computes generalized eigendecomposition of given matrix pencil. + * + * \param[in] matA Selfadjoint matrix in matrix pencil. + * Only the lower triangular part of the matrix is referenced. + * \param[in] matB Positive-definite matrix in matrix pencil. + * Only the lower triangular part of the matrix is referenced. + * \param[in] options A or-ed set of flags {#ComputeEigenvectors,#EigenvaluesOnly} | {#Ax_lBx,#ABx_lx,#BAx_lx}. + * Default is #ComputeEigenvectors|#Ax_lBx. + * + * \returns Reference to \c *this + * + * Accoring to \p options, this function computes eigenvalues and (if requested) + * the eigenvectors of one of the following three generalized eigenproblems: + * - \c Ax_lBx: \f$ Ax = \lambda B x \f$ + * - \c ABx_lx: \f$ ABx = \lambda x \f$ + * - \c BAx_lx: \f$ BAx = \lambda x \f$ + * with \a matA the selfadjoint matrix \f$ A \f$ and \a matB the positive definite + * matrix \f$ B \f$. + * In addition, each eigenvector \f$ x \f$ satisfies the property \f$ x^* B x = 1 \f$. + * + * The eigenvalues() function can be used to retrieve + * the eigenvalues. If \p options contains ComputeEigenvectors, then the + * eigenvectors are also computed and can be retrieved by calling + * eigenvectors(). + * + * The implementation uses LLT to compute the Cholesky decomposition + * \f$ B = LL^* \f$ and computes the classical eigendecomposition + * of the selfadjoint matrix \f$ L^{-1} A (L^*)^{-1} \f$ if \p options contains Ax_lBx + * and of \f$ L^{*} A L \f$ otherwise. This solves the + * generalized eigenproblem, because any solution of the generalized + * eigenproblem \f$ Ax = \lambda B x \f$ corresponds to a solution + * \f$ L^{-1} A (L^*)^{-1} (L^* x) = \lambda (L^* x) \f$ of the + * eigenproblem for \f$ L^{-1} A (L^*)^{-1} \f$. Similar statements + * can be made for the two other variants. + * + * Example: \include SelfAdjointEigenSolver_compute_MatrixType2.cpp + * Output: \verbinclude SelfAdjointEigenSolver_compute_MatrixType2.out + * + * \sa GeneralizedSelfAdjointEigenSolver(const MatrixType&, const MatrixType&, int) + */ + GeneralizedSelfAdjointEigenSolver & + compute(const MatrixType &matA, const MatrixType &matB, int options = ComputeEigenvectors | Ax_lBx); + +protected: }; template -GeneralizedSelfAdjointEigenSolver& GeneralizedSelfAdjointEigenSolver:: -compute(const MatrixType& matA, const MatrixType& matB, int options) +GeneralizedSelfAdjointEigenSolver & + GeneralizedSelfAdjointEigenSolver::compute(const MatrixType &matA, const MatrixType &matB, int options) { - eigen_assert(matA.cols()==matA.rows() && matB.rows()==matA.rows() && matB.cols()==matB.rows()); - eigen_assert((options&~(EigVecMask|GenEigMask))==0 - && (options&EigVecMask)!=EigVecMask - && ((options&GenEigMask)==0 || (options&GenEigMask)==Ax_lBx - || (options&GenEigMask)==ABx_lx || (options&GenEigMask)==BAx_lx) - && "invalid option parameter"); + eigen_assert(matA.cols() == matA.rows() && matB.rows() == matA.rows() && matB.cols() == matB.rows()); + eigen_assert((options & ~(EigVecMask | GenEigMask)) == 0 && (options & EigVecMask) != EigVecMask + && ((options & GenEigMask) == 0 || (options & GenEigMask) == Ax_lBx || (options & GenEigMask) == ABx_lx + || (options & GenEigMask) == BAx_lx) + && "invalid option parameter"); - bool computeEigVecs = ((options&EigVecMask)==0) || ((options&EigVecMask)==ComputeEigenvectors); + bool computeEigVecs = ((options & EigVecMask) == 0) || ((options & EigVecMask) == ComputeEigenvectors); // Compute the cholesky decomposition of matB = L L' = U'U LLT cholB(matB); - int type = (options&GenEigMask); - if(type==0) - type = Ax_lBx; + int type = (options & GenEigMask); + if (type == 0) type = Ax_lBx; - if(type==Ax_lBx) - { + if (type == Ax_lBx) { // compute C = inv(L) A inv(L') MatrixType matC = matA.template selfadjointView(); cholB.matrixL().template solveInPlace(matC); cholB.matrixU().template solveInPlace(matC); - Base::compute(matC, computeEigVecs ? ComputeEigenvectors : EigenvaluesOnly ); + Base::compute(matC, computeEigVecs ? ComputeEigenvectors : EigenvaluesOnly); // transform back the eigen vectors: evecs = inv(U) * evecs - if(computeEigVecs) - cholB.matrixU().solveInPlace(Base::m_eivec); - } - else if(type==ABx_lx) - { + if (computeEigVecs) cholB.matrixU().solveInPlace(Base::m_eivec); + } else if (type == ABx_lx) { // compute C = L' A L MatrixType matC = matA.template selfadjointView(); matC = matC * cholB.matrixL(); @@ -201,11 +192,8 @@ compute(const MatrixType& matA, const MatrixType& matB, int options) Base::compute(matC, computeEigVecs ? ComputeEigenvectors : EigenvaluesOnly); // transform back the eigen vectors: evecs = inv(U) * evecs - if(computeEigVecs) - cholB.matrixU().solveInPlace(Base::m_eivec); - } - else if(type==BAx_lx) - { + if (computeEigVecs) cholB.matrixU().solveInPlace(Base::m_eivec); + } else if (type == BAx_lx) { // compute C = L' A L MatrixType matC = matA.template selfadjointView(); matC = matC * cholB.matrixL(); @@ -214,13 +202,12 @@ compute(const MatrixType& matA, const MatrixType& matB, int options) Base::compute(matC, computeEigVecs ? ComputeEigenvectors : EigenvaluesOnly); // transform back the eigen vectors: evecs = L * evecs - if(computeEigVecs) - Base::m_eivec = cholB.matrixL() * Base::m_eivec; + if (computeEigVecs) Base::m_eivec = cholB.matrixL() * Base::m_eivec; } return *this; } -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_GENERALIZEDSELFADJOINTEIGENSOLVER_H +#endif// EIGEN_GENERALIZEDSELFADJOINTEIGENSOLVER_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Eigenvalues/HessenbergDecomposition.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Eigenvalues/HessenbergDecomposition.h index f647f69b..0e5b9e02 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Eigenvalues/HessenbergDecomposition.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Eigenvalues/HessenbergDecomposition.h @@ -11,299 +11,287 @@ #ifndef EIGEN_HESSENBERGDECOMPOSITION_H #define EIGEN_HESSENBERGDECOMPOSITION_H -namespace Eigen { +namespace Eigen { namespace internal { - -template struct HessenbergDecompositionMatrixHReturnType; -template -struct traits > -{ - typedef MatrixType ReturnType; -}; -} + template struct HessenbergDecompositionMatrixHReturnType; + template struct traits> + { + typedef MatrixType ReturnType; + }; + +}// namespace internal /** \eigenvalues_module \ingroup Eigenvalues_Module - * - * - * \class HessenbergDecomposition - * - * \brief Reduces a square matrix to Hessenberg form by an orthogonal similarity transformation - * - * \tparam _MatrixType the type of the matrix of which we are computing the Hessenberg decomposition - * - * This class performs an Hessenberg decomposition of a matrix \f$ A \f$. In - * the real case, the Hessenberg decomposition consists of an orthogonal - * matrix \f$ Q \f$ and a Hessenberg matrix \f$ H \f$ such that \f$ A = Q H - * Q^T \f$. An orthogonal matrix is a matrix whose inverse equals its - * transpose (\f$ Q^{-1} = Q^T \f$). A Hessenberg matrix has zeros below the - * subdiagonal, so it is almost upper triangular. The Hessenberg decomposition - * of a complex matrix is \f$ A = Q H Q^* \f$ with \f$ Q \f$ unitary (that is, - * \f$ Q^{-1} = Q^* \f$). - * - * Call the function compute() to compute the Hessenberg decomposition of a - * given matrix. Alternatively, you can use the - * HessenbergDecomposition(const MatrixType&) constructor which computes the - * Hessenberg decomposition at construction time. Once the decomposition is - * computed, you can use the matrixH() and matrixQ() functions to construct - * the matrices H and Q in the decomposition. - * - * The documentation for matrixH() contains an example of the typical use of - * this class. - * - * \sa class ComplexSchur, class Tridiagonalization, \ref QR_Module "QR Module" - */ + * + * + * \class HessenbergDecomposition + * + * \brief Reduces a square matrix to Hessenberg form by an orthogonal similarity transformation + * + * \tparam _MatrixType the type of the matrix of which we are computing the Hessenberg decomposition + * + * This class performs an Hessenberg decomposition of a matrix \f$ A \f$. In + * the real case, the Hessenberg decomposition consists of an orthogonal + * matrix \f$ Q \f$ and a Hessenberg matrix \f$ H \f$ such that \f$ A = Q H + * Q^T \f$. An orthogonal matrix is a matrix whose inverse equals its + * transpose (\f$ Q^{-1} = Q^T \f$). A Hessenberg matrix has zeros below the + * subdiagonal, so it is almost upper triangular. The Hessenberg decomposition + * of a complex matrix is \f$ A = Q H Q^* \f$ with \f$ Q \f$ unitary (that is, + * \f$ Q^{-1} = Q^* \f$). + * + * Call the function compute() to compute the Hessenberg decomposition of a + * given matrix. Alternatively, you can use the + * HessenbergDecomposition(const MatrixType&) constructor which computes the + * Hessenberg decomposition at construction time. Once the decomposition is + * computed, you can use the matrixH() and matrixQ() functions to construct + * the matrices H and Q in the decomposition. + * + * The documentation for matrixH() contains an example of the typical use of + * this class. + * + * \sa class ComplexSchur, class Tridiagonalization, \ref QR_Module "QR Module" + */ template class HessenbergDecomposition { - public: - - /** \brief Synonym for the template parameter \p _MatrixType. */ - typedef _MatrixType MatrixType; - - enum { - Size = MatrixType::RowsAtCompileTime, - SizeMinusOne = Size == Dynamic ? Dynamic : Size - 1, - Options = MatrixType::Options, - MaxSize = MatrixType::MaxRowsAtCompileTime, - MaxSizeMinusOne = MaxSize == Dynamic ? Dynamic : MaxSize - 1 - }; - - /** \brief Scalar type for matrices of type #MatrixType. */ - typedef typename MatrixType::Scalar Scalar; - typedef Eigen::Index Index; ///< \deprecated since Eigen 3.3 - - /** \brief Type for vector of Householder coefficients. - * - * This is column vector with entries of type #Scalar. The length of the - * vector is one less than the size of #MatrixType, if it is a fixed-side - * type. - */ - typedef Matrix CoeffVectorType; - - /** \brief Return type of matrixQ() */ - typedef HouseholderSequence::type> HouseholderSequenceType; - - typedef internal::HessenbergDecompositionMatrixHReturnType MatrixHReturnType; - - /** \brief Default constructor; the decomposition will be computed later. - * - * \param [in] size The size of the matrix whose Hessenberg decomposition will be computed. - * - * The default constructor is useful in cases in which the user intends to - * perform decompositions via compute(). The \p size parameter is only - * used as a hint. It is not an error to give a wrong \p size, but it may - * impair performance. - * - * \sa compute() for an example. - */ - explicit HessenbergDecomposition(Index size = Size==Dynamic ? 2 : Size) - : m_matrix(size,size), - m_temp(size), - m_isInitialized(false) - { - if(size>1) - m_hCoeffs.resize(size-1); - } +public: + /** \brief Synonym for the template parameter \p _MatrixType. */ + typedef _MatrixType MatrixType; + + enum { + Size = MatrixType::RowsAtCompileTime, + SizeMinusOne = Size == Dynamic ? Dynamic : Size - 1, + Options = MatrixType::Options, + MaxSize = MatrixType::MaxRowsAtCompileTime, + MaxSizeMinusOne = MaxSize == Dynamic ? Dynamic : MaxSize - 1 + }; + + /** \brief Scalar type for matrices of type #MatrixType. */ + typedef typename MatrixType::Scalar Scalar; + typedef Eigen::Index Index;///< \deprecated since Eigen 3.3 + + /** \brief Type for vector of Householder coefficients. + * + * This is column vector with entries of type #Scalar. The length of the + * vector is one less than the size of #MatrixType, if it is a fixed-side + * type. + */ + typedef Matrix CoeffVectorType; + + /** \brief Return type of matrixQ() */ + typedef HouseholderSequence::type> + HouseholderSequenceType; + + typedef internal::HessenbergDecompositionMatrixHReturnType MatrixHReturnType; + + /** \brief Default constructor; the decomposition will be computed later. + * + * \param [in] size The size of the matrix whose Hessenberg decomposition will be computed. + * + * The default constructor is useful in cases in which the user intends to + * perform decompositions via compute(). The \p size parameter is only + * used as a hint. It is not an error to give a wrong \p size, but it may + * impair performance. + * + * \sa compute() for an example. + */ + explicit HessenbergDecomposition(Index size = Size == Dynamic ? 2 : Size) + : m_matrix(size, size), m_temp(size), m_isInitialized(false) + { + if (size > 1) m_hCoeffs.resize(size - 1); + } - /** \brief Constructor; computes Hessenberg decomposition of given matrix. - * - * \param[in] matrix Square matrix whose Hessenberg decomposition is to be computed. - * - * This constructor calls compute() to compute the Hessenberg - * decomposition. - * - * \sa matrixH() for an example. - */ - template - explicit HessenbergDecomposition(const EigenBase& matrix) - : m_matrix(matrix.derived()), - m_temp(matrix.rows()), - m_isInitialized(false) - { - if(matrix.rows()<2) - { - m_isInitialized = true; - return; - } - m_hCoeffs.resize(matrix.rows()-1,1); - _compute(m_matrix, m_hCoeffs, m_temp); + /** \brief Constructor; computes Hessenberg decomposition of given matrix. + * + * \param[in] matrix Square matrix whose Hessenberg decomposition is to be computed. + * + * This constructor calls compute() to compute the Hessenberg + * decomposition. + * + * \sa matrixH() for an example. + */ + template + explicit HessenbergDecomposition(const EigenBase &matrix) + : m_matrix(matrix.derived()), m_temp(matrix.rows()), m_isInitialized(false) + { + if (matrix.rows() < 2) { m_isInitialized = true; + return; } + m_hCoeffs.resize(matrix.rows() - 1, 1); + _compute(m_matrix, m_hCoeffs, m_temp); + m_isInitialized = true; + } - /** \brief Computes Hessenberg decomposition of given matrix. - * - * \param[in] matrix Square matrix whose Hessenberg decomposition is to be computed. - * \returns Reference to \c *this - * - * The Hessenberg decomposition is computed by bringing the columns of the - * matrix successively in the required form using Householder reflections - * (see, e.g., Algorithm 7.4.2 in Golub \& Van Loan, %Matrix - * Computations). The cost is \f$ 10n^3/3 \f$ flops, where \f$ n \f$ - * denotes the size of the given matrix. - * - * This method reuses of the allocated data in the HessenbergDecomposition - * object. - * - * Example: \include HessenbergDecomposition_compute.cpp - * Output: \verbinclude HessenbergDecomposition_compute.out - */ - template - HessenbergDecomposition& compute(const EigenBase& matrix) - { - m_matrix = matrix.derived(); - if(matrix.rows()<2) - { - m_isInitialized = true; - return *this; - } - m_hCoeffs.resize(matrix.rows()-1,1); - _compute(m_matrix, m_hCoeffs, m_temp); + /** \brief Computes Hessenberg decomposition of given matrix. + * + * \param[in] matrix Square matrix whose Hessenberg decomposition is to be computed. + * \returns Reference to \c *this + * + * The Hessenberg decomposition is computed by bringing the columns of the + * matrix successively in the required form using Householder reflections + * (see, e.g., Algorithm 7.4.2 in Golub \& Van Loan, %Matrix + * Computations). The cost is \f$ 10n^3/3 \f$ flops, where \f$ n \f$ + * denotes the size of the given matrix. + * + * This method reuses of the allocated data in the HessenbergDecomposition + * object. + * + * Example: \include HessenbergDecomposition_compute.cpp + * Output: \verbinclude HessenbergDecomposition_compute.out + */ + template HessenbergDecomposition &compute(const EigenBase &matrix) + { + m_matrix = matrix.derived(); + if (matrix.rows() < 2) { m_isInitialized = true; return *this; } + m_hCoeffs.resize(matrix.rows() - 1, 1); + _compute(m_matrix, m_hCoeffs, m_temp); + m_isInitialized = true; + return *this; + } - /** \brief Returns the Householder coefficients. - * - * \returns a const reference to the vector of Householder coefficients - * - * \pre Either the constructor HessenbergDecomposition(const MatrixType&) - * or the member function compute(const MatrixType&) has been called - * before to compute the Hessenberg decomposition of a matrix. - * - * The Householder coefficients allow the reconstruction of the matrix - * \f$ Q \f$ in the Hessenberg decomposition from the packed data. - * - * \sa packedMatrix(), \ref Householder_Module "Householder module" - */ - const CoeffVectorType& householderCoefficients() const - { - eigen_assert(m_isInitialized && "HessenbergDecomposition is not initialized."); - return m_hCoeffs; - } - - /** \brief Returns the internal representation of the decomposition - * - * \returns a const reference to a matrix with the internal representation - * of the decomposition. - * - * \pre Either the constructor HessenbergDecomposition(const MatrixType&) - * or the member function compute(const MatrixType&) has been called - * before to compute the Hessenberg decomposition of a matrix. - * - * The returned matrix contains the following information: - * - the upper part and lower sub-diagonal represent the Hessenberg matrix H - * - the rest of the lower part contains the Householder vectors that, combined with - * Householder coefficients returned by householderCoefficients(), - * allows to reconstruct the matrix Q as - * \f$ Q = H_{N-1} \ldots H_1 H_0 \f$. - * Here, the matrices \f$ H_i \f$ are the Householder transformations - * \f$ H_i = (I - h_i v_i v_i^T) \f$ - * where \f$ h_i \f$ is the \f$ i \f$th Householder coefficient and - * \f$ v_i \f$ is the Householder vector defined by - * \f$ v_i = [ 0, \ldots, 0, 1, M(i+2,i), \ldots, M(N-1,i) ]^T \f$ - * with M the matrix returned by this function. - * - * See LAPACK for further details on this packed storage. - * - * Example: \include HessenbergDecomposition_packedMatrix.cpp - * Output: \verbinclude HessenbergDecomposition_packedMatrix.out - * - * \sa householderCoefficients() - */ - const MatrixType& packedMatrix() const - { - eigen_assert(m_isInitialized && "HessenbergDecomposition is not initialized."); - return m_matrix; - } + /** \brief Returns the Householder coefficients. + * + * \returns a const reference to the vector of Householder coefficients + * + * \pre Either the constructor HessenbergDecomposition(const MatrixType&) + * or the member function compute(const MatrixType&) has been called + * before to compute the Hessenberg decomposition of a matrix. + * + * The Householder coefficients allow the reconstruction of the matrix + * \f$ Q \f$ in the Hessenberg decomposition from the packed data. + * + * \sa packedMatrix(), \ref Householder_Module "Householder module" + */ + const CoeffVectorType &householderCoefficients() const + { + eigen_assert(m_isInitialized && "HessenbergDecomposition is not initialized."); + return m_hCoeffs; + } - /** \brief Reconstructs the orthogonal matrix Q in the decomposition - * - * \returns object representing the matrix Q - * - * \pre Either the constructor HessenbergDecomposition(const MatrixType&) - * or the member function compute(const MatrixType&) has been called - * before to compute the Hessenberg decomposition of a matrix. - * - * This function returns a light-weight object of template class - * HouseholderSequence. You can either apply it directly to a matrix or - * you can convert it to a matrix of type #MatrixType. - * - * \sa matrixH() for an example, class HouseholderSequence - */ - HouseholderSequenceType matrixQ() const - { - eigen_assert(m_isInitialized && "HessenbergDecomposition is not initialized."); - return HouseholderSequenceType(m_matrix, m_hCoeffs.conjugate()) - .setLength(m_matrix.rows() - 1) - .setShift(1); - } + /** \brief Returns the internal representation of the decomposition + * + * \returns a const reference to a matrix with the internal representation + * of the decomposition. + * + * \pre Either the constructor HessenbergDecomposition(const MatrixType&) + * or the member function compute(const MatrixType&) has been called + * before to compute the Hessenberg decomposition of a matrix. + * + * The returned matrix contains the following information: + * - the upper part and lower sub-diagonal represent the Hessenberg matrix H + * - the rest of the lower part contains the Householder vectors that, combined with + * Householder coefficients returned by householderCoefficients(), + * allows to reconstruct the matrix Q as + * \f$ Q = H_{N-1} \ldots H_1 H_0 \f$. + * Here, the matrices \f$ H_i \f$ are the Householder transformations + * \f$ H_i = (I - h_i v_i v_i^T) \f$ + * where \f$ h_i \f$ is the \f$ i \f$th Householder coefficient and + * \f$ v_i \f$ is the Householder vector defined by + * \f$ v_i = [ 0, \ldots, 0, 1, M(i+2,i), \ldots, M(N-1,i) ]^T \f$ + * with M the matrix returned by this function. + * + * See LAPACK for further details on this packed storage. + * + * Example: \include HessenbergDecomposition_packedMatrix.cpp + * Output: \verbinclude HessenbergDecomposition_packedMatrix.out + * + * \sa householderCoefficients() + */ + const MatrixType &packedMatrix() const + { + eigen_assert(m_isInitialized && "HessenbergDecomposition is not initialized."); + return m_matrix; + } - /** \brief Constructs the Hessenberg matrix H in the decomposition - * - * \returns expression object representing the matrix H - * - * \pre Either the constructor HessenbergDecomposition(const MatrixType&) - * or the member function compute(const MatrixType&) has been called - * before to compute the Hessenberg decomposition of a matrix. - * - * The object returned by this function constructs the Hessenberg matrix H - * when it is assigned to a matrix or otherwise evaluated. The matrix H is - * constructed from the packed matrix as returned by packedMatrix(): The - * upper part (including the subdiagonal) of the packed matrix contains - * the matrix H. It may sometimes be better to directly use the packed - * matrix instead of constructing the matrix H. - * - * Example: \include HessenbergDecomposition_matrixH.cpp - * Output: \verbinclude HessenbergDecomposition_matrixH.out - * - * \sa matrixQ(), packedMatrix() - */ - MatrixHReturnType matrixH() const - { - eigen_assert(m_isInitialized && "HessenbergDecomposition is not initialized."); - return MatrixHReturnType(*this); - } + /** \brief Reconstructs the orthogonal matrix Q in the decomposition + * + * \returns object representing the matrix Q + * + * \pre Either the constructor HessenbergDecomposition(const MatrixType&) + * or the member function compute(const MatrixType&) has been called + * before to compute the Hessenberg decomposition of a matrix. + * + * This function returns a light-weight object of template class + * HouseholderSequence. You can either apply it directly to a matrix or + * you can convert it to a matrix of type #MatrixType. + * + * \sa matrixH() for an example, class HouseholderSequence + */ + HouseholderSequenceType matrixQ() const + { + eigen_assert(m_isInitialized && "HessenbergDecomposition is not initialized."); + return HouseholderSequenceType(m_matrix, m_hCoeffs.conjugate()).setLength(m_matrix.rows() - 1).setShift(1); + } - private: + /** \brief Constructs the Hessenberg matrix H in the decomposition + * + * \returns expression object representing the matrix H + * + * \pre Either the constructor HessenbergDecomposition(const MatrixType&) + * or the member function compute(const MatrixType&) has been called + * before to compute the Hessenberg decomposition of a matrix. + * + * The object returned by this function constructs the Hessenberg matrix H + * when it is assigned to a matrix or otherwise evaluated. The matrix H is + * constructed from the packed matrix as returned by packedMatrix(): The + * upper part (including the subdiagonal) of the packed matrix contains + * the matrix H. It may sometimes be better to directly use the packed + * matrix instead of constructing the matrix H. + * + * Example: \include HessenbergDecomposition_matrixH.cpp + * Output: \verbinclude HessenbergDecomposition_matrixH.out + * + * \sa matrixQ(), packedMatrix() + */ + MatrixHReturnType matrixH() const + { + eigen_assert(m_isInitialized && "HessenbergDecomposition is not initialized."); + return MatrixHReturnType(*this); + } - typedef Matrix VectorType; - typedef typename NumTraits::Real RealScalar; - static void _compute(MatrixType& matA, CoeffVectorType& hCoeffs, VectorType& temp); +private: + typedef Matrix VectorType; + typedef typename NumTraits::Real RealScalar; + static void _compute(MatrixType &matA, CoeffVectorType &hCoeffs, VectorType &temp); - protected: - MatrixType m_matrix; - CoeffVectorType m_hCoeffs; - VectorType m_temp; - bool m_isInitialized; +protected: + MatrixType m_matrix; + CoeffVectorType m_hCoeffs; + VectorType m_temp; + bool m_isInitialized; }; /** \internal - * Performs a tridiagonal decomposition of \a matA in place. - * - * \param matA the input selfadjoint matrix - * \param hCoeffs returned Householder coefficients - * - * The result is written in the lower triangular part of \a matA. - * - * Implemented from Golub's "%Matrix Computations", algorithm 8.3.1. - * - * \sa packedMatrix() - */ + * Performs a tridiagonal decomposition of \a matA in place. + * + * \param matA the input selfadjoint matrix + * \param hCoeffs returned Householder coefficients + * + * The result is written in the lower triangular part of \a matA. + * + * Implemented from Golub's "%Matrix Computations", algorithm 8.3.1. + * + * \sa packedMatrix() + */ template -void HessenbergDecomposition::_compute(MatrixType& matA, CoeffVectorType& hCoeffs, VectorType& temp) +void HessenbergDecomposition::_compute(MatrixType &matA, CoeffVectorType &hCoeffs, VectorType &temp) { - eigen_assert(matA.rows()==matA.cols()); + eigen_assert(matA.rows() == matA.cols()); Index n = matA.rows(); temp.resize(n); - for (Index i = 0; i::_compute(MatrixType& matA, CoeffVector // A = H A matA.bottomRightCorner(remainingSize, remainingSize) - .applyHouseholderOnTheLeft(matA.col(i).tail(remainingSize-1), h, &temp.coeffRef(0)); + .applyHouseholderOnTheLeft(matA.col(i).tail(remainingSize - 1), h, &temp.coeffRef(0)); // A = A H' matA.rightCols(remainingSize) - .applyHouseholderOnTheRight(matA.col(i).tail(remainingSize-1).conjugate(), numext::conj(h), &temp.coeffRef(0)); + .applyHouseholderOnTheRight(matA.col(i).tail(remainingSize - 1).conjugate(), numext::conj(h), &temp.coeffRef(0)); } } namespace internal { -/** \eigenvalues_module \ingroup Eigenvalues_Module - * - * - * \brief Expression type for return value of HessenbergDecomposition::matrixH() - * - * \tparam MatrixType type of matrix in the Hessenberg decomposition - * - * Objects of this type represent the Hessenberg matrix in the Hessenberg - * decomposition of some matrix. The object holds a reference to the - * HessenbergDecomposition class until the it is assigned or evaluated for - * some other reason (the reference should remain valid during the life time - * of this object). This class is the return type of - * HessenbergDecomposition::matrixH(); there is probably no other use for this - * class. - */ -template struct HessenbergDecompositionMatrixHReturnType -: public ReturnByValue > -{ + /** \eigenvalues_module \ingroup Eigenvalues_Module + * + * + * \brief Expression type for return value of HessenbergDecomposition::matrixH() + * + * \tparam MatrixType type of matrix in the Hessenberg decomposition + * + * Objects of this type represent the Hessenberg matrix in the Hessenberg + * decomposition of some matrix. The object holds a reference to the + * HessenbergDecomposition class until the it is assigned or evaluated for + * some other reason (the reference should remain valid during the life time + * of this object). This class is the return type of + * HessenbergDecomposition::matrixH(); there is probably no other use for this + * class. + */ + template + struct HessenbergDecompositionMatrixHReturnType + : public ReturnByValue> + { public: /** \brief Constructor. - * - * \param[in] hess Hessenberg decomposition - */ - HessenbergDecompositionMatrixHReturnType(const HessenbergDecomposition& hess) : m_hess(hess) { } + * + * \param[in] hess Hessenberg decomposition + */ + HessenbergDecompositionMatrixHReturnType(const HessenbergDecomposition &hess) : m_hess(hess) {} /** \brief Hessenberg matrix in decomposition. - * - * \param[out] result Hessenberg matrix in decomposition \p hess which - * was passed to the constructor - */ - template - inline void evalTo(ResultType& result) const + * + * \param[out] result Hessenberg matrix in decomposition \p hess which + * was passed to the constructor + */ + template inline void evalTo(ResultType &result) const { result = m_hess.packedMatrix(); Index n = result.rows(); - if (n>2) - result.bottomLeftCorner(n-2, n-2).template triangularView().setZero(); + if (n > 2) result.bottomLeftCorner(n - 2, n - 2).template triangularView().setZero(); } Index rows() const { return m_hess.packedMatrix().rows(); } Index cols() const { return m_hess.packedMatrix().cols(); } protected: - const HessenbergDecomposition& m_hess; -}; + const HessenbergDecomposition &m_hess; + }; -} // end namespace internal +}// end namespace internal -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_HESSENBERGDECOMPOSITION_H +#endif// EIGEN_HESSENBERGDECOMPOSITION_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Eigenvalues/MatrixBaseEigenvalues.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Eigenvalues/MatrixBaseEigenvalues.h index e4e42607..143fae6e 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Eigenvalues/MatrixBaseEigenvalues.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Eigenvalues/MatrixBaseEigenvalues.h @@ -11,148 +11,134 @@ #ifndef EIGEN_MATRIXBASEEIGENVALUES_H #define EIGEN_MATRIXBASEEIGENVALUES_H -namespace Eigen { +namespace Eigen { namespace internal { -template -struct eigenvalues_selector -{ - // this is the implementation for the case IsComplex = true - static inline typename MatrixBase::EigenvaluesReturnType const - run(const MatrixBase& m) + template struct eigenvalues_selector { - typedef typename Derived::PlainObject PlainObject; - PlainObject m_eval(m); - return ComplexEigenSolver(m_eval, false).eigenvalues(); - } -}; + // this is the implementation for the case IsComplex = true + static inline typename MatrixBase::EigenvaluesReturnType const run(const MatrixBase &m) + { + typedef typename Derived::PlainObject PlainObject; + PlainObject m_eval(m); + return ComplexEigenSolver(m_eval, false).eigenvalues(); + } + }; -template -struct eigenvalues_selector -{ - static inline typename MatrixBase::EigenvaluesReturnType const - run(const MatrixBase& m) + template struct eigenvalues_selector { - typedef typename Derived::PlainObject PlainObject; - PlainObject m_eval(m); - return EigenSolver(m_eval, false).eigenvalues(); - } -}; + static inline typename MatrixBase::EigenvaluesReturnType const run(const MatrixBase &m) + { + typedef typename Derived::PlainObject PlainObject; + PlainObject m_eval(m); + return EigenSolver(m_eval, false).eigenvalues(); + } + }; -} // end namespace internal +}// end namespace internal -/** \brief Computes the eigenvalues of a matrix - * \returns Column vector containing the eigenvalues. - * - * \eigenvalues_module - * This function computes the eigenvalues with the help of the EigenSolver - * class (for real matrices) or the ComplexEigenSolver class (for complex - * matrices). - * - * The eigenvalues are repeated according to their algebraic multiplicity, - * so there are as many eigenvalues as rows in the matrix. - * - * The SelfAdjointView class provides a better algorithm for selfadjoint - * matrices. - * - * Example: \include MatrixBase_eigenvalues.cpp - * Output: \verbinclude MatrixBase_eigenvalues.out - * - * \sa EigenSolver::eigenvalues(), ComplexEigenSolver::eigenvalues(), - * SelfAdjointView::eigenvalues() - */ +/** \brief Computes the eigenvalues of a matrix + * \returns Column vector containing the eigenvalues. + * + * \eigenvalues_module + * This function computes the eigenvalues with the help of the EigenSolver + * class (for real matrices) or the ComplexEigenSolver class (for complex + * matrices). + * + * The eigenvalues are repeated according to their algebraic multiplicity, + * so there are as many eigenvalues as rows in the matrix. + * + * The SelfAdjointView class provides a better algorithm for selfadjoint + * matrices. + * + * Example: \include MatrixBase_eigenvalues.cpp + * Output: \verbinclude MatrixBase_eigenvalues.out + * + * \sa EigenSolver::eigenvalues(), ComplexEigenSolver::eigenvalues(), + * SelfAdjointView::eigenvalues() + */ template -inline typename MatrixBase::EigenvaluesReturnType -MatrixBase::eigenvalues() const +inline typename MatrixBase::EigenvaluesReturnType MatrixBase::eigenvalues() const { return internal::eigenvalues_selector::IsComplex>::run(derived()); } /** \brief Computes the eigenvalues of a matrix - * \returns Column vector containing the eigenvalues. - * - * \eigenvalues_module - * This function computes the eigenvalues with the help of the - * SelfAdjointEigenSolver class. The eigenvalues are repeated according to - * their algebraic multiplicity, so there are as many eigenvalues as rows in - * the matrix. - * - * Example: \include SelfAdjointView_eigenvalues.cpp - * Output: \verbinclude SelfAdjointView_eigenvalues.out - * - * \sa SelfAdjointEigenSolver::eigenvalues(), MatrixBase::eigenvalues() - */ -template + * \returns Column vector containing the eigenvalues. + * + * \eigenvalues_module + * This function computes the eigenvalues with the help of the + * SelfAdjointEigenSolver class. The eigenvalues are repeated according to + * their algebraic multiplicity, so there are as many eigenvalues as rows in + * the matrix. + * + * Example: \include SelfAdjointView_eigenvalues.cpp + * Output: \verbinclude SelfAdjointView_eigenvalues.out + * + * \sa SelfAdjointEigenSolver::eigenvalues(), MatrixBase::eigenvalues() + */ +template inline typename SelfAdjointView::EigenvaluesReturnType -SelfAdjointView::eigenvalues() const + SelfAdjointView::eigenvalues() const { PlainObject thisAsMatrix(*this); return SelfAdjointEigenSolver(thisAsMatrix, false).eigenvalues(); } - /** \brief Computes the L2 operator norm - * \returns Operator norm of the matrix. - * - * \eigenvalues_module - * This function computes the L2 operator norm of a matrix, which is also - * known as the spectral norm. The norm of a matrix \f$ A \f$ is defined to be - * \f[ \|A\|_2 = \max_x \frac{\|Ax\|_2}{\|x\|_2} \f] - * where the maximum is over all vectors and the norm on the right is the - * Euclidean vector norm. The norm equals the largest singular value, which is - * the square root of the largest eigenvalue of the positive semi-definite - * matrix \f$ A^*A \f$. - * - * The current implementation uses the eigenvalues of \f$ A^*A \f$, as computed - * by SelfAdjointView::eigenvalues(), to compute the operator norm of a - * matrix. The SelfAdjointView class provides a better algorithm for - * selfadjoint matrices. - * - * Example: \include MatrixBase_operatorNorm.cpp - * Output: \verbinclude MatrixBase_operatorNorm.out - * - * \sa SelfAdjointView::eigenvalues(), SelfAdjointView::operatorNorm() - */ -template -inline typename MatrixBase::RealScalar -MatrixBase::operatorNorm() const + * \returns Operator norm of the matrix. + * + * \eigenvalues_module + * This function computes the L2 operator norm of a matrix, which is also + * known as the spectral norm. The norm of a matrix \f$ A \f$ is defined to be + * \f[ \|A\|_2 = \max_x \frac{\|Ax\|_2}{\|x\|_2} \f] + * where the maximum is over all vectors and the norm on the right is the + * Euclidean vector norm. The norm equals the largest singular value, which is + * the square root of the largest eigenvalue of the positive semi-definite + * matrix \f$ A^*A \f$. + * + * The current implementation uses the eigenvalues of \f$ A^*A \f$, as computed + * by SelfAdjointView::eigenvalues(), to compute the operator norm of a + * matrix. The SelfAdjointView class provides a better algorithm for + * selfadjoint matrices. + * + * Example: \include MatrixBase_operatorNorm.cpp + * Output: \verbinclude MatrixBase_operatorNorm.out + * + * \sa SelfAdjointView::eigenvalues(), SelfAdjointView::operatorNorm() + */ +template inline typename MatrixBase::RealScalar MatrixBase::operatorNorm() const { using std::sqrt; typename Derived::PlainObject m_eval(derived()); // FIXME if it is really guaranteed that the eigenvalues are already sorted, // then we don't need to compute a maxCoeff() here, comparing the 1st and last ones is enough. - return sqrt((m_eval*m_eval.adjoint()) - .eval() - .template selfadjointView() - .eigenvalues() - .maxCoeff() - ); + return sqrt((m_eval * m_eval.adjoint()).eval().template selfadjointView().eigenvalues().maxCoeff()); } /** \brief Computes the L2 operator norm - * \returns Operator norm of the matrix. - * - * \eigenvalues_module - * This function computes the L2 operator norm of a self-adjoint matrix. For a - * self-adjoint matrix, the operator norm is the largest eigenvalue. - * - * The current implementation uses the eigenvalues of the matrix, as computed - * by eigenvalues(), to compute the operator norm of the matrix. - * - * Example: \include SelfAdjointView_operatorNorm.cpp - * Output: \verbinclude SelfAdjointView_operatorNorm.out - * - * \sa eigenvalues(), MatrixBase::operatorNorm() - */ + * \returns Operator norm of the matrix. + * + * \eigenvalues_module + * This function computes the L2 operator norm of a self-adjoint matrix. For a + * self-adjoint matrix, the operator norm is the largest eigenvalue. + * + * The current implementation uses the eigenvalues of the matrix, as computed + * by eigenvalues(), to compute the operator norm of the matrix. + * + * Example: \include SelfAdjointView_operatorNorm.cpp + * Output: \verbinclude SelfAdjointView_operatorNorm.out + * + * \sa eigenvalues(), MatrixBase::operatorNorm() + */ template -inline typename SelfAdjointView::RealScalar -SelfAdjointView::operatorNorm() const +inline typename SelfAdjointView::RealScalar SelfAdjointView::operatorNorm() const { return eigenvalues().cwiseAbs().maxCoeff(); } -} // end namespace Eigen +}// end namespace Eigen #endif diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Eigenvalues/RealQZ.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Eigenvalues/RealQZ.h index b3a910dd..46b7e6b1 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Eigenvalues/RealQZ.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Eigenvalues/RealQZ.h @@ -12,643 +12,574 @@ namespace Eigen { - /** \eigenvalues_module \ingroup Eigenvalues_Module +/** \eigenvalues_module \ingroup Eigenvalues_Module + * + * + * \class RealQZ + * + * \brief Performs a real QZ decomposition of a pair of square matrices + * + * \tparam _MatrixType the type of the matrix of which we are computing the + * real QZ decomposition; this is expected to be an instantiation of the + * Matrix class template. + * + * Given a real square matrices A and B, this class computes the real QZ + * decomposition: \f$ A = Q S Z \f$, \f$ B = Q T Z \f$ where Q and Z are + * real orthogonal matrixes, T is upper-triangular matrix, and S is upper + * quasi-triangular matrix. An orthogonal matrix is a matrix whose + * inverse is equal to its transpose, \f$ U^{-1} = U^T \f$. A quasi-triangular + * matrix is a block-triangular matrix whose diagonal consists of 1-by-1 + * blocks and 2-by-2 blocks where further reduction is impossible due to + * complex eigenvalues. + * + * The eigenvalues of the pencil \f$ A - z B \f$ can be obtained from + * 1x1 and 2x2 blocks on the diagonals of S and T. + * + * Call the function compute() to compute the real QZ decomposition of a + * given pair of matrices. Alternatively, you can use the + * RealQZ(const MatrixType& B, const MatrixType& B, bool computeQZ) + * constructor which computes the real QZ decomposition at construction + * time. Once the decomposition is computed, you can use the matrixS(), + * matrixT(), matrixQ() and matrixZ() functions to retrieve the matrices + * S, T, Q and Z in the decomposition. If computeQZ==false, some time + * is saved by not computing matrices Q and Z. + * + * Example: \include RealQZ_compute.cpp + * Output: \include RealQZ_compute.out + * + * \note The implementation is based on the algorithm in "Matrix Computations" + * by Gene H. Golub and Charles F. Van Loan, and a paper "An algorithm for + * generalized eigenvalue problems" by C.B.Moler and G.W.Stewart. + * + * \sa class RealSchur, class ComplexSchur, class EigenSolver, class ComplexEigenSolver + */ + +template class RealQZ +{ +public: + typedef _MatrixType MatrixType; + enum { + RowsAtCompileTime = MatrixType::RowsAtCompileTime, + ColsAtCompileTime = MatrixType::ColsAtCompileTime, + Options = MatrixType::Options, + MaxRowsAtCompileTime = MatrixType::MaxRowsAtCompileTime, + MaxColsAtCompileTime = MatrixType::MaxColsAtCompileTime + }; + typedef typename MatrixType::Scalar Scalar; + typedef std::complex::Real> ComplexScalar; + typedef Eigen::Index Index;///< \deprecated since Eigen 3.3 + + typedef Matrix EigenvalueType; + typedef Matrix ColumnVectorType; + + /** \brief Default constructor. * + * \param [in] size Positive integer, size of the matrix whose QZ decomposition will be computed. * - * \class RealQZ + * The default constructor is useful in cases in which the user intends to + * perform decompositions via compute(). The \p size parameter is only + * used as a hint. It is not an error to give a wrong \p size, but it may + * impair performance. * - * \brief Performs a real QZ decomposition of a pair of square matrices - * - * \tparam _MatrixType the type of the matrix of which we are computing the - * real QZ decomposition; this is expected to be an instantiation of the - * Matrix class template. - * - * Given a real square matrices A and B, this class computes the real QZ - * decomposition: \f$ A = Q S Z \f$, \f$ B = Q T Z \f$ where Q and Z are - * real orthogonal matrixes, T is upper-triangular matrix, and S is upper - * quasi-triangular matrix. An orthogonal matrix is a matrix whose - * inverse is equal to its transpose, \f$ U^{-1} = U^T \f$. A quasi-triangular - * matrix is a block-triangular matrix whose diagonal consists of 1-by-1 - * blocks and 2-by-2 blocks where further reduction is impossible due to - * complex eigenvalues. - * - * The eigenvalues of the pencil \f$ A - z B \f$ can be obtained from - * 1x1 and 2x2 blocks on the diagonals of S and T. - * - * Call the function compute() to compute the real QZ decomposition of a - * given pair of matrices. Alternatively, you can use the - * RealQZ(const MatrixType& B, const MatrixType& B, bool computeQZ) - * constructor which computes the real QZ decomposition at construction - * time. Once the decomposition is computed, you can use the matrixS(), - * matrixT(), matrixQ() and matrixZ() functions to retrieve the matrices - * S, T, Q and Z in the decomposition. If computeQZ==false, some time - * is saved by not computing matrices Q and Z. - * - * Example: \include RealQZ_compute.cpp - * Output: \include RealQZ_compute.out + * \sa compute() for an example. + */ + explicit RealQZ(Index size = RowsAtCompileTime == Dynamic ? 1 : RowsAtCompileTime) + : m_S(size, size), m_T(size, size), m_Q(size, size), m_Z(size, size), m_workspace(size * 2), m_maxIters(400), + m_isInitialized(false) + {} + + /** \brief Constructor; computes real QZ decomposition of given matrices * - * \note The implementation is based on the algorithm in "Matrix Computations" - * by Gene H. Golub and Charles F. Van Loan, and a paper "An algorithm for - * generalized eigenvalue problems" by C.B.Moler and G.W.Stewart. + * \param[in] A Matrix A. + * \param[in] B Matrix B. + * \param[in] computeQZ If false, A and Z are not computed. * - * \sa class RealSchur, class ComplexSchur, class EigenSolver, class ComplexEigenSolver + * This constructor calls compute() to compute the QZ decomposition. */ + RealQZ(const MatrixType &A, const MatrixType &B, bool computeQZ = true) + : m_S(A.rows(), A.cols()), m_T(A.rows(), A.cols()), m_Q(A.rows(), A.cols()), m_Z(A.rows(), A.cols()), + m_workspace(A.rows() * 2), m_maxIters(400), m_isInitialized(false) + { + compute(A, B, computeQZ); + } - template class RealQZ + /** \brief Returns matrix Q in the QZ decomposition. + * + * \returns A const reference to the matrix Q. + */ + const MatrixType &matrixQ() const { - public: - typedef _MatrixType MatrixType; - enum { - RowsAtCompileTime = MatrixType::RowsAtCompileTime, - ColsAtCompileTime = MatrixType::ColsAtCompileTime, - Options = MatrixType::Options, - MaxRowsAtCompileTime = MatrixType::MaxRowsAtCompileTime, - MaxColsAtCompileTime = MatrixType::MaxColsAtCompileTime - }; - typedef typename MatrixType::Scalar Scalar; - typedef std::complex::Real> ComplexScalar; - typedef Eigen::Index Index; ///< \deprecated since Eigen 3.3 - - typedef Matrix EigenvalueType; - typedef Matrix ColumnVectorType; - - /** \brief Default constructor. - * - * \param [in] size Positive integer, size of the matrix whose QZ decomposition will be computed. - * - * The default constructor is useful in cases in which the user intends to - * perform decompositions via compute(). The \p size parameter is only - * used as a hint. It is not an error to give a wrong \p size, but it may - * impair performance. - * - * \sa compute() for an example. - */ - explicit RealQZ(Index size = RowsAtCompileTime==Dynamic ? 1 : RowsAtCompileTime) : - m_S(size, size), - m_T(size, size), - m_Q(size, size), - m_Z(size, size), - m_workspace(size*2), - m_maxIters(400), - m_isInitialized(false) - { } - - /** \brief Constructor; computes real QZ decomposition of given matrices - * - * \param[in] A Matrix A. - * \param[in] B Matrix B. - * \param[in] computeQZ If false, A and Z are not computed. - * - * This constructor calls compute() to compute the QZ decomposition. - */ - RealQZ(const MatrixType& A, const MatrixType& B, bool computeQZ = true) : - m_S(A.rows(),A.cols()), - m_T(A.rows(),A.cols()), - m_Q(A.rows(),A.cols()), - m_Z(A.rows(),A.cols()), - m_workspace(A.rows()*2), - m_maxIters(400), - m_isInitialized(false) { - compute(A, B, computeQZ); - } + eigen_assert(m_isInitialized && "RealQZ is not initialized."); + eigen_assert(m_computeQZ && "The matrices Q and Z have not been computed during the QZ decomposition."); + return m_Q; + } - /** \brief Returns matrix Q in the QZ decomposition. - * - * \returns A const reference to the matrix Q. - */ - const MatrixType& matrixQ() const { - eigen_assert(m_isInitialized && "RealQZ is not initialized."); - eigen_assert(m_computeQZ && "The matrices Q and Z have not been computed during the QZ decomposition."); - return m_Q; - } + /** \brief Returns matrix Z in the QZ decomposition. + * + * \returns A const reference to the matrix Z. + */ + const MatrixType &matrixZ() const + { + eigen_assert(m_isInitialized && "RealQZ is not initialized."); + eigen_assert(m_computeQZ && "The matrices Q and Z have not been computed during the QZ decomposition."); + return m_Z; + } - /** \brief Returns matrix Z in the QZ decomposition. - * - * \returns A const reference to the matrix Z. - */ - const MatrixType& matrixZ() const { - eigen_assert(m_isInitialized && "RealQZ is not initialized."); - eigen_assert(m_computeQZ && "The matrices Q and Z have not been computed during the QZ decomposition."); - return m_Z; - } + /** \brief Returns matrix S in the QZ decomposition. + * + * \returns A const reference to the matrix S. + */ + const MatrixType &matrixS() const + { + eigen_assert(m_isInitialized && "RealQZ is not initialized."); + return m_S; + } - /** \brief Returns matrix S in the QZ decomposition. - * - * \returns A const reference to the matrix S. - */ - const MatrixType& matrixS() const { - eigen_assert(m_isInitialized && "RealQZ is not initialized."); - return m_S; - } + /** \brief Returns matrix S in the QZ decomposition. + * + * \returns A const reference to the matrix S. + */ + const MatrixType &matrixT() const + { + eigen_assert(m_isInitialized && "RealQZ is not initialized."); + return m_T; + } - /** \brief Returns matrix S in the QZ decomposition. - * - * \returns A const reference to the matrix S. - */ - const MatrixType& matrixT() const { - eigen_assert(m_isInitialized && "RealQZ is not initialized."); - return m_T; - } + /** \brief Computes QZ decomposition of given matrix. + * + * \param[in] A Matrix A. + * \param[in] B Matrix B. + * \param[in] computeQZ If false, A and Z are not computed. + * \returns Reference to \c *this + */ + RealQZ &compute(const MatrixType &A, const MatrixType &B, bool computeQZ = true); - /** \brief Computes QZ decomposition of given matrix. - * - * \param[in] A Matrix A. - * \param[in] B Matrix B. - * \param[in] computeQZ If false, A and Z are not computed. - * \returns Reference to \c *this - */ - RealQZ& compute(const MatrixType& A, const MatrixType& B, bool computeQZ = true); - - /** \brief Reports whether previous computation was successful. - * - * \returns \c Success if computation was succesful, \c NoConvergence otherwise. - */ - ComputationInfo info() const - { - eigen_assert(m_isInitialized && "RealQZ is not initialized."); - return m_info; - } + /** \brief Reports whether previous computation was successful. + * + * \returns \c Success if computation was succesful, \c NoConvergence otherwise. + */ + ComputationInfo info() const + { + eigen_assert(m_isInitialized && "RealQZ is not initialized."); + return m_info; + } - /** \brief Returns number of performed QR-like iterations. - */ - Index iterations() const - { - eigen_assert(m_isInitialized && "RealQZ is not initialized."); - return m_global_iter; - } + /** \brief Returns number of performed QR-like iterations. + */ + Index iterations() const + { + eigen_assert(m_isInitialized && "RealQZ is not initialized."); + return m_global_iter; + } - /** Sets the maximal number of iterations allowed to converge to one eigenvalue - * or decouple the problem. - */ - RealQZ& setMaxIterations(Index maxIters) - { - m_maxIters = maxIters; - return *this; + /** Sets the maximal number of iterations allowed to converge to one eigenvalue + * or decouple the problem. + */ + RealQZ &setMaxIterations(Index maxIters) + { + m_maxIters = maxIters; + return *this; + } + +private: + MatrixType m_S, m_T, m_Q, m_Z; + Matrix m_workspace; + ComputationInfo m_info; + Index m_maxIters; + bool m_isInitialized; + bool m_computeQZ; + Scalar m_normOfT, m_normOfS; + Index m_global_iter; + + typedef Matrix Vector3s; + typedef Matrix Vector2s; + typedef Matrix Matrix2s; + typedef JacobiRotation JRs; + + void hessenbergTriangular(); + void computeNorms(); + Index findSmallSubdiagEntry(Index iu); + Index findSmallDiagEntry(Index f, Index l); + void splitOffTwoRows(Index i); + void pushDownZero(Index z, Index f, Index l); + void step(Index f, Index l, Index iter); + +};// RealQZ + +/** \internal Reduces S and T to upper Hessenberg - triangular form */ +template void RealQZ::hessenbergTriangular() +{ + + const Index dim = m_S.cols(); + + // perform QR decomposition of T, overwrite T with R, save Q + HouseholderQR qrT(m_T); + m_T = qrT.matrixQR(); + m_T.template triangularView().setZero(); + m_Q = qrT.householderQ(); + // overwrite S with Q* S + m_S.applyOnTheLeft(m_Q.adjoint()); + // init Z as Identity + if (m_computeQZ) m_Z = MatrixType::Identity(dim, dim); + // reduce S to upper Hessenberg with Givens rotations + for (Index j = 0; j <= dim - 3; j++) { + for (Index i = dim - 1; i >= j + 2; i--) { + JRs G; + // kill S(i,j) + if (m_S.coeff(i, j) != 0) { + G.makeGivens(m_S.coeff(i - 1, j), m_S.coeff(i, j), &m_S.coeffRef(i - 1, j)); + m_S.coeffRef(i, j) = Scalar(0.0); + m_S.rightCols(dim - j - 1).applyOnTheLeft(i - 1, i, G.adjoint()); + m_T.rightCols(dim - i + 1).applyOnTheLeft(i - 1, i, G.adjoint()); + // update Q + if (m_computeQZ) m_Q.applyOnTheRight(i - 1, i, G); } - - private: - - MatrixType m_S, m_T, m_Q, m_Z; - Matrix m_workspace; - ComputationInfo m_info; - Index m_maxIters; - bool m_isInitialized; - bool m_computeQZ; - Scalar m_normOfT, m_normOfS; - Index m_global_iter; - - typedef Matrix Vector3s; - typedef Matrix Vector2s; - typedef Matrix Matrix2s; - typedef JacobiRotation JRs; - - void hessenbergTriangular(); - void computeNorms(); - Index findSmallSubdiagEntry(Index iu); - Index findSmallDiagEntry(Index f, Index l); - void splitOffTwoRows(Index i); - void pushDownZero(Index z, Index f, Index l); - void step(Index f, Index l, Index iter); - - }; // RealQZ - - /** \internal Reduces S and T to upper Hessenberg - triangular form */ - template - void RealQZ::hessenbergTriangular() - { - - const Index dim = m_S.cols(); - - // perform QR decomposition of T, overwrite T with R, save Q - HouseholderQR qrT(m_T); - m_T = qrT.matrixQR(); - m_T.template triangularView().setZero(); - m_Q = qrT.householderQ(); - // overwrite S with Q* S - m_S.applyOnTheLeft(m_Q.adjoint()); - // init Z as Identity - if (m_computeQZ) - m_Z = MatrixType::Identity(dim,dim); - // reduce S to upper Hessenberg with Givens rotations - for (Index j=0; j<=dim-3; j++) { - for (Index i=dim-1; i>=j+2; i--) { - JRs G; - // kill S(i,j) - if(m_S.coeff(i,j) != 0) - { - G.makeGivens(m_S.coeff(i-1,j), m_S.coeff(i,j), &m_S.coeffRef(i-1, j)); - m_S.coeffRef(i,j) = Scalar(0.0); - m_S.rightCols(dim-j-1).applyOnTheLeft(i-1,i,G.adjoint()); - m_T.rightCols(dim-i+1).applyOnTheLeft(i-1,i,G.adjoint()); - // update Q - if (m_computeQZ) - m_Q.applyOnTheRight(i-1,i,G); - } - // kill T(i,i-1) - if(m_T.coeff(i,i-1)!=Scalar(0)) - { - G.makeGivens(m_T.coeff(i,i), m_T.coeff(i,i-1), &m_T.coeffRef(i,i)); - m_T.coeffRef(i,i-1) = Scalar(0.0); - m_S.applyOnTheRight(i,i-1,G); - m_T.topRows(i).applyOnTheRight(i,i-1,G); - // update Z - if (m_computeQZ) - m_Z.applyOnTheLeft(i,i-1,G.adjoint()); - } - } + // kill T(i,i-1) + if (m_T.coeff(i, i - 1) != Scalar(0)) { + G.makeGivens(m_T.coeff(i, i), m_T.coeff(i, i - 1), &m_T.coeffRef(i, i)); + m_T.coeffRef(i, i - 1) = Scalar(0.0); + m_S.applyOnTheRight(i, i - 1, G); + m_T.topRows(i).applyOnTheRight(i, i - 1, G); + // update Z + if (m_computeQZ) m_Z.applyOnTheLeft(i, i - 1, G.adjoint()); } } + } +} + +/** \internal Computes vector L1 norms of S and T when in Hessenberg-Triangular form already */ +template inline void RealQZ::computeNorms() +{ + const Index size = m_S.cols(); + m_normOfS = Scalar(0.0); + m_normOfT = Scalar(0.0); + for (Index j = 0; j < size; ++j) { + m_normOfS += m_S.col(j).segment(0, (std::min)(size, j + 2)).cwiseAbs().sum(); + m_normOfT += m_T.row(j).segment(j, size - j).cwiseAbs().sum(); + } +} + + +/** \internal Look for single small sub-diagonal element S(res, res-1) and return res (or 0) */ +template inline Index RealQZ::findSmallSubdiagEntry(Index iu) +{ + using std::abs; + Index res = iu; + while (res > 0) { + Scalar s = abs(m_S.coeff(res - 1, res - 1)) + abs(m_S.coeff(res, res)); + if (s == Scalar(0.0)) s = m_normOfS; + if (abs(m_S.coeff(res, res - 1)) < NumTraits::epsilon() * s) break; + res--; + } + return res; +} + +/** \internal Look for single small diagonal element T(res, res) for res between f and l, and return res (or f-1) */ +template inline Index RealQZ::findSmallDiagEntry(Index f, Index l) +{ + using std::abs; + Index res = l; + while (res >= f) { + if (abs(m_T.coeff(res, res)) <= NumTraits::epsilon() * m_normOfT) break; + res--; + } + return res; +} + +/** \internal decouple 2x2 diagonal block in rows i, i+1 if eigenvalues are real */ +template inline void RealQZ::splitOffTwoRows(Index i) +{ + using std::abs; + using std::sqrt; + const Index dim = m_S.cols(); + if (abs(m_S.coeff(i + 1, i)) == Scalar(0)) return; + Index j = findSmallDiagEntry(i, i + 1); + if (j == i - 1) { + // block of (S T^{-1}) + Matrix2s STi = m_T.template block<2, 2>(i, i).template triangularView().template solve( + m_S.template block<2, 2>(i, i)); + Scalar p = Scalar(0.5) * (STi(0, 0) - STi(1, 1)); + Scalar q = p * p + STi(1, 0) * STi(0, 1); + if (q >= 0) { + Scalar z = sqrt(q); + // one QR-like iteration for ABi - lambda I + // is enough - when we know exact eigenvalue in advance, + // convergence is immediate + JRs G; + if (p >= 0) + G.makeGivens(p + z, STi(1, 0)); + else + G.makeGivens(p - z, STi(1, 0)); + m_S.rightCols(dim - i).applyOnTheLeft(i, i + 1, G.adjoint()); + m_T.rightCols(dim - i).applyOnTheLeft(i, i + 1, G.adjoint()); + // update Q + if (m_computeQZ) m_Q.applyOnTheRight(i, i + 1, G); + + G.makeGivens(m_T.coeff(i + 1, i + 1), m_T.coeff(i + 1, i)); + m_S.topRows(i + 2).applyOnTheRight(i + 1, i, G); + m_T.topRows(i + 2).applyOnTheRight(i + 1, i, G); + // update Z + if (m_computeQZ) m_Z.applyOnTheLeft(i + 1, i, G.adjoint()); - /** \internal Computes vector L1 norms of S and T when in Hessenberg-Triangular form already */ - template - inline void RealQZ::computeNorms() - { - const Index size = m_S.cols(); - m_normOfS = Scalar(0.0); - m_normOfT = Scalar(0.0); - for (Index j = 0; j < size; ++j) - { - m_normOfS += m_S.col(j).segment(0, (std::min)(size,j+2)).cwiseAbs().sum(); - m_normOfT += m_T.row(j).segment(j, size - j).cwiseAbs().sum(); - } + m_S.coeffRef(i + 1, i) = Scalar(0.0); + m_T.coeffRef(i + 1, i) = Scalar(0.0); } - - - /** \internal Look for single small sub-diagonal element S(res, res-1) and return res (or 0) */ - template - inline Index RealQZ::findSmallSubdiagEntry(Index iu) - { - using std::abs; - Index res = iu; - while (res > 0) - { - Scalar s = abs(m_S.coeff(res-1,res-1)) + abs(m_S.coeff(res,res)); - if (s == Scalar(0.0)) - s = m_normOfS; - if (abs(m_S.coeff(res,res-1)) < NumTraits::epsilon() * s) - break; - res--; - } - return res; + } else { + pushDownZero(j, i, i + 1); + } +} + +/** \internal use zero in T(z,z) to zero S(l,l-1), working in block f..l */ +template inline void RealQZ::pushDownZero(Index z, Index f, Index l) +{ + JRs G; + const Index dim = m_S.cols(); + for (Index zz = z; zz < l; zz++) { + // push 0 down + Index firstColS = zz > f ? (zz - 1) : zz; + G.makeGivens(m_T.coeff(zz, zz + 1), m_T.coeff(zz + 1, zz + 1)); + m_S.rightCols(dim - firstColS).applyOnTheLeft(zz, zz + 1, G.adjoint()); + m_T.rightCols(dim - zz).applyOnTheLeft(zz, zz + 1, G.adjoint()); + m_T.coeffRef(zz + 1, zz + 1) = Scalar(0.0); + // update Q + if (m_computeQZ) m_Q.applyOnTheRight(zz, zz + 1, G); + // kill S(zz+1, zz-1) + if (zz > f) { + G.makeGivens(m_S.coeff(zz + 1, zz), m_S.coeff(zz + 1, zz - 1)); + m_S.topRows(zz + 2).applyOnTheRight(zz, zz - 1, G); + m_T.topRows(zz + 1).applyOnTheRight(zz, zz - 1, G); + m_S.coeffRef(zz + 1, zz - 1) = Scalar(0.0); + // update Z + if (m_computeQZ) m_Z.applyOnTheLeft(zz, zz - 1, G.adjoint()); } - - /** \internal Look for single small diagonal element T(res, res) for res between f and l, and return res (or f-1) */ - template - inline Index RealQZ::findSmallDiagEntry(Index f, Index l) + } + // finally kill S(l,l-1) + G.makeGivens(m_S.coeff(l, l), m_S.coeff(l, l - 1)); + m_S.applyOnTheRight(l, l - 1, G); + m_T.applyOnTheRight(l, l - 1, G); + m_S.coeffRef(l, l - 1) = Scalar(0.0); + // update Z + if (m_computeQZ) m_Z.applyOnTheLeft(l, l - 1, G.adjoint()); +} + +/** \internal QR-like iterative step for block f..l */ +template inline void RealQZ::step(Index f, Index l, Index iter) +{ + using std::abs; + const Index dim = m_S.cols(); + + // x, y, z + Scalar x, y, z; + if (iter == 10) { + // Wilkinson ad hoc shift + const Scalar a11 = m_S.coeff(f + 0, f + 0), a12 = m_S.coeff(f + 0, f + 1), a21 = m_S.coeff(f + 1, f + 0), + a22 = m_S.coeff(f + 1, f + 1), a32 = m_S.coeff(f + 2, f + 1), b12 = m_T.coeff(f + 0, f + 1), + b11i = Scalar(1.0) / m_T.coeff(f + 0, f + 0), b22i = Scalar(1.0) / m_T.coeff(f + 1, f + 1), + a87 = m_S.coeff(l - 1, l - 2), a98 = m_S.coeff(l - 0, l - 1), + b77i = Scalar(1.0) / m_T.coeff(l - 2, l - 2), b88i = Scalar(1.0) / m_T.coeff(l - 1, l - 1); + Scalar ss = abs(a87 * b77i) + abs(a98 * b88i), lpl = Scalar(1.5) * ss, ll = ss * ss; + x = + ll + a11 * a11 * b11i * b11i - lpl * a11 * b11i + a12 * a21 * b11i * b22i - a11 * a21 * b12 * b11i * b11i * b22i; + y = a11 * a21 * b11i * b11i - lpl * a21 * b11i + a21 * a22 * b11i * b22i - a21 * a21 * b12 * b11i * b11i * b22i; + z = a21 * a32 * b11i * b22i; + } else if (iter == 16) { + // another exceptional shift + x = m_S.coeff(f, f) / m_T.coeff(f, f) - m_S.coeff(l, l) / m_T.coeff(l, l) + + m_S.coeff(l, l - 1) * m_T.coeff(l - 1, l) / (m_T.coeff(l - 1, l - 1) * m_T.coeff(l, l)); + y = m_S.coeff(f + 1, f) / m_T.coeff(f, f); + z = 0; + } else if (iter > 23 && !(iter % 8)) { + // extremely exceptional shift + x = internal::random(-1.0, 1.0); + y = internal::random(-1.0, 1.0); + z = internal::random(-1.0, 1.0); + } else { + // Compute the shifts: (x,y,z,0...) = (AB^-1 - l1 I) (AB^-1 - l2 I) e1 + // where l1 and l2 are the eigenvalues of the 2x2 matrix C = U V^-1 where + // U and V are 2x2 bottom right sub matrices of A and B. Thus: + // = AB^-1AB^-1 + l1 l2 I - (l1+l2)(AB^-1) + // = AB^-1AB^-1 + det(M) - tr(M)(AB^-1) + // Since we are only interested in having x, y, z with a correct ratio, we have: + const Scalar a11 = m_S.coeff(f, f), a12 = m_S.coeff(f, f + 1), a21 = m_S.coeff(f + 1, f), + a22 = m_S.coeff(f + 1, f + 1), a32 = m_S.coeff(f + 2, f + 1), + + a88 = m_S.coeff(l - 1, l - 1), a89 = m_S.coeff(l - 1, l), a98 = m_S.coeff(l, l - 1), + a99 = m_S.coeff(l, l), + + b11 = m_T.coeff(f, f), b12 = m_T.coeff(f, f + 1), b22 = m_T.coeff(f + 1, f + 1), + + b88 = m_T.coeff(l - 1, l - 1), b89 = m_T.coeff(l - 1, l), b99 = m_T.coeff(l, l); + + x = ((a88 / b88 - a11 / b11) * (a99 / b99 - a11 / b11) - (a89 / b99) * (a98 / b88) + + (a98 / b88) * (b89 / b99) * (a11 / b11)) + * (b11 / a21) + + a12 / b22 - (a11 / b11) * (b12 / b22); + y = (a22 / b22 - a11 / b11) - (a21 / b11) * (b12 / b22) - (a88 / b88 - a11 / b11) - (a99 / b99 - a11 / b11) + + (a98 / b88) * (b89 / b99); + z = a32 / b22; + } + + JRs G; + + for (Index k = f; k <= l - 2; k++) { + // variables for Householder reflections + Vector2s essential2; + Scalar tau, beta; + + Vector3s hr(x, y, z); + + // Q_k to annihilate S(k+1,k-1) and S(k+2,k-1) + hr.makeHouseholderInPlace(tau, beta); + essential2 = hr.template bottomRows<2>(); + Index fc = (std::max)(k - 1, Index(0));// first col to update + m_S.template middleRows<3>(k).rightCols(dim - fc).applyHouseholderOnTheLeft(essential2, tau, m_workspace.data()); + m_T.template middleRows<3>(k).rightCols(dim - fc).applyHouseholderOnTheLeft(essential2, tau, m_workspace.data()); + if (m_computeQZ) m_Q.template middleCols<3>(k).applyHouseholderOnTheRight(essential2, tau, m_workspace.data()); + if (k > f) m_S.coeffRef(k + 2, k - 1) = m_S.coeffRef(k + 1, k - 1) = Scalar(0.0); + + // Z_{k1} to annihilate T(k+2,k+1) and T(k+2,k) + hr << m_T.coeff(k + 2, k + 2), m_T.coeff(k + 2, k), m_T.coeff(k + 2, k + 1); + hr.makeHouseholderInPlace(tau, beta); + essential2 = hr.template bottomRows<2>(); { - using std::abs; - Index res = l; - while (res >= f) { - if (abs(m_T.coeff(res,res)) <= NumTraits::epsilon() * m_normOfT) - break; - res--; - } - return res; + Index lr = (std::min)(k + 4, dim);// last row to update + Map> tmp(m_workspace.data(), lr); + // S + tmp = m_S.template middleCols<2>(k).topRows(lr) * essential2; + tmp += m_S.col(k + 2).head(lr); + m_S.col(k + 2).head(lr) -= tau * tmp; + m_S.template middleCols<2>(k).topRows(lr) -= (tau * tmp) * essential2.adjoint(); + // T + tmp = m_T.template middleCols<2>(k).topRows(lr) * essential2; + tmp += m_T.col(k + 2).head(lr); + m_T.col(k + 2).head(lr) -= tau * tmp; + m_T.template middleCols<2>(k).topRows(lr) -= (tau * tmp) * essential2.adjoint(); } - - /** \internal decouple 2x2 diagonal block in rows i, i+1 if eigenvalues are real */ - template - inline void RealQZ::splitOffTwoRows(Index i) - { - using std::abs; - using std::sqrt; - const Index dim=m_S.cols(); - if (abs(m_S.coeff(i+1,i))==Scalar(0)) - return; - Index j = findSmallDiagEntry(i,i+1); - if (j==i-1) - { - // block of (S T^{-1}) - Matrix2s STi = m_T.template block<2,2>(i,i).template triangularView(). - template solve(m_S.template block<2,2>(i,i)); - Scalar p = Scalar(0.5)*(STi(0,0)-STi(1,1)); - Scalar q = p*p + STi(1,0)*STi(0,1); - if (q>=0) { - Scalar z = sqrt(q); - // one QR-like iteration for ABi - lambda I - // is enough - when we know exact eigenvalue in advance, - // convergence is immediate - JRs G; - if (p>=0) - G.makeGivens(p + z, STi(1,0)); - else - G.makeGivens(p - z, STi(1,0)); - m_S.rightCols(dim-i).applyOnTheLeft(i,i+1,G.adjoint()); - m_T.rightCols(dim-i).applyOnTheLeft(i,i+1,G.adjoint()); - // update Q - if (m_computeQZ) - m_Q.applyOnTheRight(i,i+1,G); - - G.makeGivens(m_T.coeff(i+1,i+1), m_T.coeff(i+1,i)); - m_S.topRows(i+2).applyOnTheRight(i+1,i,G); - m_T.topRows(i+2).applyOnTheRight(i+1,i,G); - // update Z - if (m_computeQZ) - m_Z.applyOnTheLeft(i+1,i,G.adjoint()); - - m_S.coeffRef(i+1,i) = Scalar(0.0); - m_T.coeffRef(i+1,i) = Scalar(0.0); - } - } - else - { - pushDownZero(j,i,i+1); - } + if (m_computeQZ) { + // Z + Map> tmp(m_workspace.data(), dim); + tmp = essential2.adjoint() * (m_Z.template middleRows<2>(k)); + tmp += m_Z.row(k + 2); + m_Z.row(k + 2) -= tau * tmp; + m_Z.template middleRows<2>(k) -= essential2 * (tau * tmp); } - - /** \internal use zero in T(z,z) to zero S(l,l-1), working in block f..l */ - template - inline void RealQZ::pushDownZero(Index z, Index f, Index l) + m_T.coeffRef(k + 2, k) = m_T.coeffRef(k + 2, k + 1) = Scalar(0.0); + + // Z_{k2} to annihilate T(k+1,k) + G.makeGivens(m_T.coeff(k + 1, k + 1), m_T.coeff(k + 1, k)); + m_S.applyOnTheRight(k + 1, k, G); + m_T.applyOnTheRight(k + 1, k, G); + // update Z + if (m_computeQZ) m_Z.applyOnTheLeft(k + 1, k, G.adjoint()); + m_T.coeffRef(k + 1, k) = Scalar(0.0); + + // update x,y,z + x = m_S.coeff(k + 1, k); + y = m_S.coeff(k + 2, k); + if (k < l - 2) z = m_S.coeff(k + 3, k); + }// loop over k + + // Q_{n-1} to annihilate y = S(l,l-2) + G.makeGivens(x, y); + m_S.applyOnTheLeft(l - 1, l, G.adjoint()); + m_T.applyOnTheLeft(l - 1, l, G.adjoint()); + if (m_computeQZ) m_Q.applyOnTheRight(l - 1, l, G); + m_S.coeffRef(l, l - 2) = Scalar(0.0); + + // Z_{n-1} to annihilate T(l,l-1) + G.makeGivens(m_T.coeff(l, l), m_T.coeff(l, l - 1)); + m_S.applyOnTheRight(l, l - 1, G); + m_T.applyOnTheRight(l, l - 1, G); + if (m_computeQZ) m_Z.applyOnTheLeft(l, l - 1, G.adjoint()); + m_T.coeffRef(l, l - 1) = Scalar(0.0); +} + +template +RealQZ &RealQZ::compute(const MatrixType &A_in, const MatrixType &B_in, bool computeQZ) +{ + + const Index dim = A_in.cols(); + + eigen_assert(A_in.rows() == dim && A_in.cols() == dim && B_in.rows() == dim && B_in.cols() == dim + && "Need square matrices of the same dimension"); + + m_isInitialized = true; + m_computeQZ = computeQZ; + m_S = A_in; + m_T = B_in; + m_workspace.resize(dim * 2); + m_global_iter = 0; + + // entrance point: hessenberg triangular decomposition + hessenbergTriangular(); + // compute L1 vector norms of T, S into m_normOfS, m_normOfT + computeNorms(); + + Index l = dim - 1, f, local_iter = 0; + + while (l > 0 && local_iter < m_maxIters) { + f = findSmallSubdiagEntry(l); + // now rows and columns f..l (including) decouple from the rest of the problem + if (f > 0) m_S.coeffRef(f, f - 1) = Scalar(0.0); + if (f == l)// One root found { - JRs G; - const Index dim = m_S.cols(); - for (Index zz=z; zzf ? (zz-1) : zz; - G.makeGivens(m_T.coeff(zz, zz+1), m_T.coeff(zz+1, zz+1)); - m_S.rightCols(dim-firstColS).applyOnTheLeft(zz,zz+1,G.adjoint()); - m_T.rightCols(dim-zz).applyOnTheLeft(zz,zz+1,G.adjoint()); - m_T.coeffRef(zz+1,zz+1) = Scalar(0.0); - // update Q - if (m_computeQZ) - m_Q.applyOnTheRight(zz,zz+1,G); - // kill S(zz+1, zz-1) - if (zz>f) - { - G.makeGivens(m_S.coeff(zz+1, zz), m_S.coeff(zz+1,zz-1)); - m_S.topRows(zz+2).applyOnTheRight(zz, zz-1,G); - m_T.topRows(zz+1).applyOnTheRight(zz, zz-1,G); - m_S.coeffRef(zz+1,zz-1) = Scalar(0.0); - // update Z - if (m_computeQZ) - m_Z.applyOnTheLeft(zz,zz-1,G.adjoint()); - } - } - // finally kill S(l,l-1) - G.makeGivens(m_S.coeff(l,l), m_S.coeff(l,l-1)); - m_S.applyOnTheRight(l,l-1,G); - m_T.applyOnTheRight(l,l-1,G); - m_S.coeffRef(l,l-1)=Scalar(0.0); - // update Z - if (m_computeQZ) - m_Z.applyOnTheLeft(l,l-1,G.adjoint()); - } - - /** \internal QR-like iterative step for block f..l */ - template - inline void RealQZ::step(Index f, Index l, Index iter) + l--; + local_iter = 0; + } else if (f == l - 1)// Two roots found { - using std::abs; - const Index dim = m_S.cols(); - - // x, y, z - Scalar x, y, z; - if (iter==10) - { - // Wilkinson ad hoc shift - const Scalar - a11=m_S.coeff(f+0,f+0), a12=m_S.coeff(f+0,f+1), - a21=m_S.coeff(f+1,f+0), a22=m_S.coeff(f+1,f+1), a32=m_S.coeff(f+2,f+1), - b12=m_T.coeff(f+0,f+1), - b11i=Scalar(1.0)/m_T.coeff(f+0,f+0), - b22i=Scalar(1.0)/m_T.coeff(f+1,f+1), - a87=m_S.coeff(l-1,l-2), - a98=m_S.coeff(l-0,l-1), - b77i=Scalar(1.0)/m_T.coeff(l-2,l-2), - b88i=Scalar(1.0)/m_T.coeff(l-1,l-1); - Scalar ss = abs(a87*b77i) + abs(a98*b88i), - lpl = Scalar(1.5)*ss, - ll = ss*ss; - x = ll + a11*a11*b11i*b11i - lpl*a11*b11i + a12*a21*b11i*b22i - - a11*a21*b12*b11i*b11i*b22i; - y = a11*a21*b11i*b11i - lpl*a21*b11i + a21*a22*b11i*b22i - - a21*a21*b12*b11i*b11i*b22i; - z = a21*a32*b11i*b22i; - } - else if (iter==16) - { - // another exceptional shift - x = m_S.coeff(f,f)/m_T.coeff(f,f)-m_S.coeff(l,l)/m_T.coeff(l,l) + m_S.coeff(l,l-1)*m_T.coeff(l-1,l) / - (m_T.coeff(l-1,l-1)*m_T.coeff(l,l)); - y = m_S.coeff(f+1,f)/m_T.coeff(f,f); - z = 0; - } - else if (iter>23 && !(iter%8)) - { - // extremely exceptional shift - x = internal::random(-1.0,1.0); - y = internal::random(-1.0,1.0); - z = internal::random(-1.0,1.0); - } - else - { - // Compute the shifts: (x,y,z,0...) = (AB^-1 - l1 I) (AB^-1 - l2 I) e1 - // where l1 and l2 are the eigenvalues of the 2x2 matrix C = U V^-1 where - // U and V are 2x2 bottom right sub matrices of A and B. Thus: - // = AB^-1AB^-1 + l1 l2 I - (l1+l2)(AB^-1) - // = AB^-1AB^-1 + det(M) - tr(M)(AB^-1) - // Since we are only interested in having x, y, z with a correct ratio, we have: - const Scalar - a11 = m_S.coeff(f,f), a12 = m_S.coeff(f,f+1), - a21 = m_S.coeff(f+1,f), a22 = m_S.coeff(f+1,f+1), - a32 = m_S.coeff(f+2,f+1), - - a88 = m_S.coeff(l-1,l-1), a89 = m_S.coeff(l-1,l), - a98 = m_S.coeff(l,l-1), a99 = m_S.coeff(l,l), - - b11 = m_T.coeff(f,f), b12 = m_T.coeff(f,f+1), - b22 = m_T.coeff(f+1,f+1), - - b88 = m_T.coeff(l-1,l-1), b89 = m_T.coeff(l-1,l), - b99 = m_T.coeff(l,l); - - x = ( (a88/b88 - a11/b11)*(a99/b99 - a11/b11) - (a89/b99)*(a98/b88) + (a98/b88)*(b89/b99)*(a11/b11) ) * (b11/a21) - + a12/b22 - (a11/b11)*(b12/b22); - y = (a22/b22-a11/b11) - (a21/b11)*(b12/b22) - (a88/b88-a11/b11) - (a99/b99-a11/b11) + (a98/b88)*(b89/b99); - z = a32/b22; - } - - JRs G; - - for (Index k=f; k<=l-2; k++) - { - // variables for Householder reflections - Vector2s essential2; - Scalar tau, beta; - - Vector3s hr(x,y,z); - - // Q_k to annihilate S(k+1,k-1) and S(k+2,k-1) - hr.makeHouseholderInPlace(tau, beta); - essential2 = hr.template bottomRows<2>(); - Index fc=(std::max)(k-1,Index(0)); // first col to update - m_S.template middleRows<3>(k).rightCols(dim-fc).applyHouseholderOnTheLeft(essential2, tau, m_workspace.data()); - m_T.template middleRows<3>(k).rightCols(dim-fc).applyHouseholderOnTheLeft(essential2, tau, m_workspace.data()); - if (m_computeQZ) - m_Q.template middleCols<3>(k).applyHouseholderOnTheRight(essential2, tau, m_workspace.data()); - if (k>f) - m_S.coeffRef(k+2,k-1) = m_S.coeffRef(k+1,k-1) = Scalar(0.0); - - // Z_{k1} to annihilate T(k+2,k+1) and T(k+2,k) - hr << m_T.coeff(k+2,k+2),m_T.coeff(k+2,k),m_T.coeff(k+2,k+1); - hr.makeHouseholderInPlace(tau, beta); - essential2 = hr.template bottomRows<2>(); - { - Index lr = (std::min)(k+4,dim); // last row to update - Map > tmp(m_workspace.data(),lr); - // S - tmp = m_S.template middleCols<2>(k).topRows(lr) * essential2; - tmp += m_S.col(k+2).head(lr); - m_S.col(k+2).head(lr) -= tau*tmp; - m_S.template middleCols<2>(k).topRows(lr) -= (tau*tmp) * essential2.adjoint(); - // T - tmp = m_T.template middleCols<2>(k).topRows(lr) * essential2; - tmp += m_T.col(k+2).head(lr); - m_T.col(k+2).head(lr) -= tau*tmp; - m_T.template middleCols<2>(k).topRows(lr) -= (tau*tmp) * essential2.adjoint(); - } - if (m_computeQZ) - { - // Z - Map > tmp(m_workspace.data(),dim); - tmp = essential2.adjoint()*(m_Z.template middleRows<2>(k)); - tmp += m_Z.row(k+2); - m_Z.row(k+2) -= tau*tmp; - m_Z.template middleRows<2>(k) -= essential2 * (tau*tmp); - } - m_T.coeffRef(k+2,k) = m_T.coeffRef(k+2,k+1) = Scalar(0.0); - - // Z_{k2} to annihilate T(k+1,k) - G.makeGivens(m_T.coeff(k+1,k+1), m_T.coeff(k+1,k)); - m_S.applyOnTheRight(k+1,k,G); - m_T.applyOnTheRight(k+1,k,G); - // update Z - if (m_computeQZ) - m_Z.applyOnTheLeft(k+1,k,G.adjoint()); - m_T.coeffRef(k+1,k) = Scalar(0.0); - - // update x,y,z - x = m_S.coeff(k+1,k); - y = m_S.coeff(k+2,k); - if (k < l-2) - z = m_S.coeff(k+3,k); - } // loop over k - - // Q_{n-1} to annihilate y = S(l,l-2) - G.makeGivens(x,y); - m_S.applyOnTheLeft(l-1,l,G.adjoint()); - m_T.applyOnTheLeft(l-1,l,G.adjoint()); - if (m_computeQZ) - m_Q.applyOnTheRight(l-1,l,G); - m_S.coeffRef(l,l-2) = Scalar(0.0); - - // Z_{n-1} to annihilate T(l,l-1) - G.makeGivens(m_T.coeff(l,l),m_T.coeff(l,l-1)); - m_S.applyOnTheRight(l,l-1,G); - m_T.applyOnTheRight(l,l-1,G); - if (m_computeQZ) - m_Z.applyOnTheLeft(l,l-1,G.adjoint()); - m_T.coeffRef(l,l-1) = Scalar(0.0); - } - - template - RealQZ& RealQZ::compute(const MatrixType& A_in, const MatrixType& B_in, bool computeQZ) + splitOffTwoRows(f); + l -= 2; + local_iter = 0; + } else// No convergence yet { - - const Index dim = A_in.cols(); - - eigen_assert (A_in.rows()==dim && A_in.cols()==dim - && B_in.rows()==dim && B_in.cols()==dim - && "Need square matrices of the same dimension"); - - m_isInitialized = true; - m_computeQZ = computeQZ; - m_S = A_in; m_T = B_in; - m_workspace.resize(dim*2); - m_global_iter = 0; - - // entrance point: hessenberg triangular decomposition - hessenbergTriangular(); - // compute L1 vector norms of T, S into m_normOfS, m_normOfT - computeNorms(); - - Index l = dim-1, - f, - local_iter = 0; - - while (l>0 && local_iter0) m_S.coeffRef(f,f-1) = Scalar(0.0); - if (f == l) // One root found - { - l--; - local_iter = 0; - } - else if (f == l-1) // Two roots found - { - splitOffTwoRows(f); - l -= 2; - local_iter = 0; - } - else // No convergence yet - { - // if there's zero on diagonal of T, we can isolate an eigenvalue with Givens rotations - Index z = findSmallDiagEntry(f,l); - if (z>=f) - { - // zero found - pushDownZero(z,f,l); - } - else - { - // We are sure now that S.block(f,f, l-f+1,l-f+1) is underuced upper-Hessenberg - // and T.block(f,f, l-f+1,l-f+1) is invertible uper-triangular, which allows to - // apply a QR-like iteration to rows and columns f..l. - step(f,l, local_iter); - local_iter++; - m_global_iter++; - } - } + // if there's zero on diagonal of T, we can isolate an eigenvalue with Givens rotations + Index z = findSmallDiagEntry(f, l); + if (z >= f) { + // zero found + pushDownZero(z, f, l); + } else { + // We are sure now that S.block(f,f, l-f+1,l-f+1) is underuced upper-Hessenberg + // and T.block(f,f, l-f+1,l-f+1) is invertible uper-triangular, which allows to + // apply a QR-like iteration to rows and columns f..l. + step(f, l, local_iter); + local_iter++; + m_global_iter++; } - // check if we converged before reaching iterations limit - m_info = (local_iter j_left, j_right; - internal::real_2x2_jacobi_svd(m_T, i, i+1, &j_left, &j_right); - - // Apply resulting Jacobi rotations - m_S.applyOnTheLeft(i,i+1,j_left); - m_S.applyOnTheRight(i,i+1,j_right); - m_T.applyOnTheLeft(i,i+1,j_left); - m_T.applyOnTheRight(i,i+1,j_right); - m_T(i+1,i) = m_T(i,i+1) = Scalar(0); - - if(m_computeQZ) { - m_Q.applyOnTheRight(i,i+1,j_left.transpose()); - m_Z.applyOnTheLeft(i,i+1,j_right.transpose()); - } - - i++; - } + } + } + // check if we converged before reaching iterations limit + m_info = (local_iter < m_maxIters) ? Success : NoConvergence; + + // For each non triangular 2x2 diagonal block of S, + // reduce the respective 2x2 diagonal block of T to positive diagonal form using 2x2 SVD. + // This step is not mandatory for QZ, but it does help further extraction of eigenvalues/eigenvectors, + // and is in par with Lapack/Matlab QZ. + if (m_info == Success) { + for (Index i = 0; i < dim - 1; ++i) { + if (m_S.coeff(i + 1, i) != Scalar(0)) { + JacobiRotation j_left, j_right; + internal::real_2x2_jacobi_svd(m_T, i, i + 1, &j_left, &j_right); + + // Apply resulting Jacobi rotations + m_S.applyOnTheLeft(i, i + 1, j_left); + m_S.applyOnTheRight(i, i + 1, j_right); + m_T.applyOnTheLeft(i, i + 1, j_left); + m_T.applyOnTheRight(i, i + 1, j_right); + m_T(i + 1, i) = m_T(i, i + 1) = Scalar(0); + + if (m_computeQZ) { + m_Q.applyOnTheRight(i, i + 1, j_left.transpose()); + m_Z.applyOnTheLeft(i, i + 1, j_right.transpose()); } + + i++; } + } + } - return *this; - } // end compute + return *this; +}// end compute -} // end namespace Eigen +}// end namespace Eigen -#endif //EIGEN_REAL_QZ +#endif// EIGEN_REAL_QZ diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Eigenvalues/RealSchur.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Eigenvalues/RealSchur.h index 17ea903f..2f3e2f31 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Eigenvalues/RealSchur.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Eigenvalues/RealSchur.h @@ -13,254 +13,241 @@ #include "./HessenbergDecomposition.h" -namespace Eigen { +namespace Eigen { /** \eigenvalues_module \ingroup Eigenvalues_Module - * - * - * \class RealSchur - * - * \brief Performs a real Schur decomposition of a square matrix - * - * \tparam _MatrixType the type of the matrix of which we are computing the - * real Schur decomposition; this is expected to be an instantiation of the - * Matrix class template. - * - * Given a real square matrix A, this class computes the real Schur - * decomposition: \f$ A = U T U^T \f$ where U is a real orthogonal matrix and - * T is a real quasi-triangular matrix. An orthogonal matrix is a matrix whose - * inverse is equal to its transpose, \f$ U^{-1} = U^T \f$. A quasi-triangular - * matrix is a block-triangular matrix whose diagonal consists of 1-by-1 - * blocks and 2-by-2 blocks with complex eigenvalues. The eigenvalues of the - * blocks on the diagonal of T are the same as the eigenvalues of the matrix - * A, and thus the real Schur decomposition is used in EigenSolver to compute - * the eigendecomposition of a matrix. - * - * Call the function compute() to compute the real Schur decomposition of a - * given matrix. Alternatively, you can use the RealSchur(const MatrixType&, bool) - * constructor which computes the real Schur decomposition at construction - * time. Once the decomposition is computed, you can use the matrixU() and - * matrixT() functions to retrieve the matrices U and T in the decomposition. - * - * The documentation of RealSchur(const MatrixType&, bool) contains an example - * of the typical use of this class. - * - * \note The implementation is adapted from - * JAMA (public domain). - * Their code is based on EISPACK. - * - * \sa class ComplexSchur, class EigenSolver, class ComplexEigenSolver - */ + * + * + * \class RealSchur + * + * \brief Performs a real Schur decomposition of a square matrix + * + * \tparam _MatrixType the type of the matrix of which we are computing the + * real Schur decomposition; this is expected to be an instantiation of the + * Matrix class template. + * + * Given a real square matrix A, this class computes the real Schur + * decomposition: \f$ A = U T U^T \f$ where U is a real orthogonal matrix and + * T is a real quasi-triangular matrix. An orthogonal matrix is a matrix whose + * inverse is equal to its transpose, \f$ U^{-1} = U^T \f$. A quasi-triangular + * matrix is a block-triangular matrix whose diagonal consists of 1-by-1 + * blocks and 2-by-2 blocks with complex eigenvalues. The eigenvalues of the + * blocks on the diagonal of T are the same as the eigenvalues of the matrix + * A, and thus the real Schur decomposition is used in EigenSolver to compute + * the eigendecomposition of a matrix. + * + * Call the function compute() to compute the real Schur decomposition of a + * given matrix. Alternatively, you can use the RealSchur(const MatrixType&, bool) + * constructor which computes the real Schur decomposition at construction + * time. Once the decomposition is computed, you can use the matrixU() and + * matrixT() functions to retrieve the matrices U and T in the decomposition. + * + * The documentation of RealSchur(const MatrixType&, bool) contains an example + * of the typical use of this class. + * + * \note The implementation is adapted from + * JAMA (public domain). + * Their code is based on EISPACK. + * + * \sa class ComplexSchur, class EigenSolver, class ComplexEigenSolver + */ template class RealSchur { - public: - typedef _MatrixType MatrixType; - enum { - RowsAtCompileTime = MatrixType::RowsAtCompileTime, - ColsAtCompileTime = MatrixType::ColsAtCompileTime, - Options = MatrixType::Options, - MaxRowsAtCompileTime = MatrixType::MaxRowsAtCompileTime, - MaxColsAtCompileTime = MatrixType::MaxColsAtCompileTime - }; - typedef typename MatrixType::Scalar Scalar; - typedef std::complex::Real> ComplexScalar; - typedef Eigen::Index Index; ///< \deprecated since Eigen 3.3 - - typedef Matrix EigenvalueType; - typedef Matrix ColumnVectorType; - - /** \brief Default constructor. - * - * \param [in] size Positive integer, size of the matrix whose Schur decomposition will be computed. - * - * The default constructor is useful in cases in which the user intends to - * perform decompositions via compute(). The \p size parameter is only - * used as a hint. It is not an error to give a wrong \p size, but it may - * impair performance. - * - * \sa compute() for an example. - */ - explicit RealSchur(Index size = RowsAtCompileTime==Dynamic ? 1 : RowsAtCompileTime) - : m_matT(size, size), - m_matU(size, size), - m_workspaceVector(size), - m_hess(size), - m_isInitialized(false), - m_matUisUptodate(false), - m_maxIters(-1) - { } - - /** \brief Constructor; computes real Schur decomposition of given matrix. - * - * \param[in] matrix Square matrix whose Schur decomposition is to be computed. - * \param[in] computeU If true, both T and U are computed; if false, only T is computed. - * - * This constructor calls compute() to compute the Schur decomposition. - * - * Example: \include RealSchur_RealSchur_MatrixType.cpp - * Output: \verbinclude RealSchur_RealSchur_MatrixType.out - */ - template - explicit RealSchur(const EigenBase& matrix, bool computeU = true) - : m_matT(matrix.rows(),matrix.cols()), - m_matU(matrix.rows(),matrix.cols()), - m_workspaceVector(matrix.rows()), - m_hess(matrix.rows()), - m_isInitialized(false), - m_matUisUptodate(false), - m_maxIters(-1) - { - compute(matrix.derived(), computeU); - } +public: + typedef _MatrixType MatrixType; + enum { + RowsAtCompileTime = MatrixType::RowsAtCompileTime, + ColsAtCompileTime = MatrixType::ColsAtCompileTime, + Options = MatrixType::Options, + MaxRowsAtCompileTime = MatrixType::MaxRowsAtCompileTime, + MaxColsAtCompileTime = MatrixType::MaxColsAtCompileTime + }; + typedef typename MatrixType::Scalar Scalar; + typedef std::complex::Real> ComplexScalar; + typedef Eigen::Index Index;///< \deprecated since Eigen 3.3 + + typedef Matrix EigenvalueType; + typedef Matrix ColumnVectorType; + + /** \brief Default constructor. + * + * \param [in] size Positive integer, size of the matrix whose Schur decomposition will be computed. + * + * The default constructor is useful in cases in which the user intends to + * perform decompositions via compute(). The \p size parameter is only + * used as a hint. It is not an error to give a wrong \p size, but it may + * impair performance. + * + * \sa compute() for an example. + */ + explicit RealSchur(Index size = RowsAtCompileTime == Dynamic ? 1 : RowsAtCompileTime) + : m_matT(size, size), m_matU(size, size), m_workspaceVector(size), m_hess(size), m_isInitialized(false), + m_matUisUptodate(false), m_maxIters(-1) + {} + + /** \brief Constructor; computes real Schur decomposition of given matrix. + * + * \param[in] matrix Square matrix whose Schur decomposition is to be computed. + * \param[in] computeU If true, both T and U are computed; if false, only T is computed. + * + * This constructor calls compute() to compute the Schur decomposition. + * + * Example: \include RealSchur_RealSchur_MatrixType.cpp + * Output: \verbinclude RealSchur_RealSchur_MatrixType.out + */ + template + explicit RealSchur(const EigenBase &matrix, bool computeU = true) + : m_matT(matrix.rows(), matrix.cols()), m_matU(matrix.rows(), matrix.cols()), m_workspaceVector(matrix.rows()), + m_hess(matrix.rows()), m_isInitialized(false), m_matUisUptodate(false), m_maxIters(-1) + { + compute(matrix.derived(), computeU); + } - /** \brief Returns the orthogonal matrix in the Schur decomposition. - * - * \returns A const reference to the matrix U. - * - * \pre Either the constructor RealSchur(const MatrixType&, bool) or the - * member function compute(const MatrixType&, bool) has been called before - * to compute the Schur decomposition of a matrix, and \p computeU was set - * to true (the default value). - * - * \sa RealSchur(const MatrixType&, bool) for an example - */ - const MatrixType& matrixU() const - { - eigen_assert(m_isInitialized && "RealSchur is not initialized."); - eigen_assert(m_matUisUptodate && "The matrix U has not been computed during the RealSchur decomposition."); - return m_matU; - } + /** \brief Returns the orthogonal matrix in the Schur decomposition. + * + * \returns A const reference to the matrix U. + * + * \pre Either the constructor RealSchur(const MatrixType&, bool) or the + * member function compute(const MatrixType&, bool) has been called before + * to compute the Schur decomposition of a matrix, and \p computeU was set + * to true (the default value). + * + * \sa RealSchur(const MatrixType&, bool) for an example + */ + const MatrixType &matrixU() const + { + eigen_assert(m_isInitialized && "RealSchur is not initialized."); + eigen_assert(m_matUisUptodate && "The matrix U has not been computed during the RealSchur decomposition."); + return m_matU; + } - /** \brief Returns the quasi-triangular matrix in the Schur decomposition. - * - * \returns A const reference to the matrix T. - * - * \pre Either the constructor RealSchur(const MatrixType&, bool) or the - * member function compute(const MatrixType&, bool) has been called before - * to compute the Schur decomposition of a matrix. - * - * \sa RealSchur(const MatrixType&, bool) for an example - */ - const MatrixType& matrixT() const - { - eigen_assert(m_isInitialized && "RealSchur is not initialized."); - return m_matT; - } - - /** \brief Computes Schur decomposition of given matrix. - * - * \param[in] matrix Square matrix whose Schur decomposition is to be computed. - * \param[in] computeU If true, both T and U are computed; if false, only T is computed. - * \returns Reference to \c *this - * - * The Schur decomposition is computed by first reducing the matrix to - * Hessenberg form using the class HessenbergDecomposition. The Hessenberg - * matrix is then reduced to triangular form by performing Francis QR - * iterations with implicit double shift. The cost of computing the Schur - * decomposition depends on the number of iterations; as a rough guide, it - * may be taken to be \f$25n^3\f$ flops if \a computeU is true and - * \f$10n^3\f$ flops if \a computeU is false. - * - * Example: \include RealSchur_compute.cpp - * Output: \verbinclude RealSchur_compute.out - * - * \sa compute(const MatrixType&, bool, Index) - */ - template - RealSchur& compute(const EigenBase& matrix, bool computeU = true); - - /** \brief Computes Schur decomposition of a Hessenberg matrix H = Z T Z^T - * \param[in] matrixH Matrix in Hessenberg form H - * \param[in] matrixQ orthogonal matrix Q that transform a matrix A to H : A = Q H Q^T - * \param computeU Computes the matriX U of the Schur vectors - * \return Reference to \c *this - * - * This routine assumes that the matrix is already reduced in Hessenberg form matrixH - * using either the class HessenbergDecomposition or another mean. - * It computes the upper quasi-triangular matrix T of the Schur decomposition of H - * When computeU is true, this routine computes the matrix U such that - * A = U T U^T = (QZ) T (QZ)^T = Q H Q^T where A is the initial matrix - * - * NOTE Q is referenced if computeU is true; so, if the initial orthogonal matrix - * is not available, the user should give an identity matrix (Q.setIdentity()) - * - * \sa compute(const MatrixType&, bool) - */ - template - RealSchur& computeFromHessenberg(const HessMatrixType& matrixH, const OrthMatrixType& matrixQ, bool computeU); - /** \brief Reports whether previous computation was successful. - * - * \returns \c Success if computation was succesful, \c NoConvergence otherwise. - */ - ComputationInfo info() const - { - eigen_assert(m_isInitialized && "RealSchur is not initialized."); - return m_info; - } + /** \brief Returns the quasi-triangular matrix in the Schur decomposition. + * + * \returns A const reference to the matrix T. + * + * \pre Either the constructor RealSchur(const MatrixType&, bool) or the + * member function compute(const MatrixType&, bool) has been called before + * to compute the Schur decomposition of a matrix. + * + * \sa RealSchur(const MatrixType&, bool) for an example + */ + const MatrixType &matrixT() const + { + eigen_assert(m_isInitialized && "RealSchur is not initialized."); + return m_matT; + } - /** \brief Sets the maximum number of iterations allowed. - * - * If not specified by the user, the maximum number of iterations is m_maxIterationsPerRow times the size - * of the matrix. - */ - RealSchur& setMaxIterations(Index maxIters) - { - m_maxIters = maxIters; - return *this; - } + /** \brief Computes Schur decomposition of given matrix. + * + * \param[in] matrix Square matrix whose Schur decomposition is to be computed. + * \param[in] computeU If true, both T and U are computed; if false, only T is computed. + * \returns Reference to \c *this + * + * The Schur decomposition is computed by first reducing the matrix to + * Hessenberg form using the class HessenbergDecomposition. The Hessenberg + * matrix is then reduced to triangular form by performing Francis QR + * iterations with implicit double shift. The cost of computing the Schur + * decomposition depends on the number of iterations; as a rough guide, it + * may be taken to be \f$25n^3\f$ flops if \a computeU is true and + * \f$10n^3\f$ flops if \a computeU is false. + * + * Example: \include RealSchur_compute.cpp + * Output: \verbinclude RealSchur_compute.out + * + * \sa compute(const MatrixType&, bool, Index) + */ + template RealSchur &compute(const EigenBase &matrix, bool computeU = true); + + /** \brief Computes Schur decomposition of a Hessenberg matrix H = Z T Z^T + * \param[in] matrixH Matrix in Hessenberg form H + * \param[in] matrixQ orthogonal matrix Q that transform a matrix A to H : A = Q H Q^T + * \param computeU Computes the matriX U of the Schur vectors + * \return Reference to \c *this + * + * This routine assumes that the matrix is already reduced in Hessenberg form matrixH + * using either the class HessenbergDecomposition or another mean. + * It computes the upper quasi-triangular matrix T of the Schur decomposition of H + * When computeU is true, this routine computes the matrix U such that + * A = U T U^T = (QZ) T (QZ)^T = Q H Q^T where A is the initial matrix + * + * NOTE Q is referenced if computeU is true; so, if the initial orthogonal matrix + * is not available, the user should give an identity matrix (Q.setIdentity()) + * + * \sa compute(const MatrixType&, bool) + */ + template + RealSchur &computeFromHessenberg(const HessMatrixType &matrixH, const OrthMatrixType &matrixQ, bool computeU); + /** \brief Reports whether previous computation was successful. + * + * \returns \c Success if computation was succesful, \c NoConvergence otherwise. + */ + ComputationInfo info() const + { + eigen_assert(m_isInitialized && "RealSchur is not initialized."); + return m_info; + } - /** \brief Returns the maximum number of iterations. */ - Index getMaxIterations() - { - return m_maxIters; - } + /** \brief Sets the maximum number of iterations allowed. + * + * If not specified by the user, the maximum number of iterations is m_maxIterationsPerRow times the size + * of the matrix. + */ + RealSchur &setMaxIterations(Index maxIters) + { + m_maxIters = maxIters; + return *this; + } - /** \brief Maximum number of iterations per row. - * - * If not otherwise specified, the maximum number of iterations is this number times the size of the - * matrix. It is currently set to 40. - */ - static const int m_maxIterationsPerRow = 40; - - private: - - MatrixType m_matT; - MatrixType m_matU; - ColumnVectorType m_workspaceVector; - HessenbergDecomposition m_hess; - ComputationInfo m_info; - bool m_isInitialized; - bool m_matUisUptodate; - Index m_maxIters; - - typedef Matrix Vector3s; - - Scalar computeNormOfT(); - Index findSmallSubdiagEntry(Index iu); - void splitOffTwoRows(Index iu, bool computeU, const Scalar& exshift); - void computeShift(Index iu, Index iter, Scalar& exshift, Vector3s& shiftInfo); - void initFrancisQRStep(Index il, Index iu, const Vector3s& shiftInfo, Index& im, Vector3s& firstHouseholderVector); - void performFrancisQRStep(Index il, Index im, Index iu, bool computeU, const Vector3s& firstHouseholderVector, Scalar* workspace); + /** \brief Returns the maximum number of iterations. */ + Index getMaxIterations() { return m_maxIters; } + + /** \brief Maximum number of iterations per row. + * + * If not otherwise specified, the maximum number of iterations is this number times the size of the + * matrix. It is currently set to 40. + */ + static const int m_maxIterationsPerRow = 40; + +private: + MatrixType m_matT; + MatrixType m_matU; + ColumnVectorType m_workspaceVector; + HessenbergDecomposition m_hess; + ComputationInfo m_info; + bool m_isInitialized; + bool m_matUisUptodate; + Index m_maxIters; + + typedef Matrix Vector3s; + + Scalar computeNormOfT(); + Index findSmallSubdiagEntry(Index iu); + void splitOffTwoRows(Index iu, bool computeU, const Scalar &exshift); + void computeShift(Index iu, Index iter, Scalar &exshift, Vector3s &shiftInfo); + void initFrancisQRStep(Index il, Index iu, const Vector3s &shiftInfo, Index &im, Vector3s &firstHouseholderVector); + void performFrancisQRStep(Index il, + Index im, + Index iu, + bool computeU, + const Vector3s &firstHouseholderVector, + Scalar *workspace); }; template template -RealSchur& RealSchur::compute(const EigenBase& matrix, bool computeU) +RealSchur &RealSchur::compute(const EigenBase &matrix, bool computeU) { const Scalar considerAsZero = (std::numeric_limits::min)(); eigen_assert(matrix.cols() == matrix.rows()); Index maxIters = m_maxIters; - if (maxIters == -1) - maxIters = m_maxIterationsPerRow * matrix.rows(); + if (maxIters == -1) maxIters = m_maxIterationsPerRow * matrix.rows(); Scalar scale = matrix.derived().cwiseAbs().maxCoeff(); - if(scale& RealSchur::compute(const EigenBase } // Step 1. Reduce to Hessenberg form - m_hess.compute(matrix.derived()/scale); + m_hess.compute(matrix.derived() / scale); - // Step 2. Reduce to real Schur form + // Step 2. Reduce to real Schur form computeFromHessenberg(m_hess.matrixH(), m_hess.matrixQ(), computeU); m_matT *= scale; - + return *this; } template template -RealSchur& RealSchur::computeFromHessenberg(const HessMatrixType& matrixH, const OrthMatrixType& matrixQ, bool computeU) +RealSchur &RealSchur::computeFromHessenberg(const HessMatrixType &matrixH, + const OrthMatrixType &matrixQ, + bool computeU) { using std::abs; m_matT = matrixH; - if(computeU) - m_matU = matrixQ; - + if (computeU) m_matU = matrixQ; + Index maxIters = m_maxIters; - if (maxIters == -1) - maxIters = m_maxIterationsPerRow * matrixH.rows(); + if (maxIters == -1) maxIters = m_maxIterationsPerRow * matrixH.rows(); m_workspaceVector.resize(m_matT.cols()); - Scalar* workspace = &m_workspaceVector.coeffRef(0); + Scalar *workspace = &m_workspaceVector.coeffRef(0); - // The matrix m_matT is divided in three parts. - // Rows 0,...,il-1 are decoupled from the rest because m_matT(il,il-1) is zero. + // The matrix m_matT is divided in three parts. + // Rows 0,...,il-1 are decoupled from the rest because m_matT(il,il-1) is zero. // Rows il,...,iu is the part we are working on (the active window). // Rows iu+1,...,end are already brought in triangular form. Index iu = m_matT.cols() - 1; - Index iter = 0; // iteration count for current eigenvalue - Index totalIter = 0; // iteration count for whole matrix - Scalar exshift(0); // sum of exceptional shifts + Index iter = 0;// iteration count for current eigenvalue + Index totalIter = 0;// iteration count for whole matrix + Scalar exshift(0);// sum of exceptional shifts Scalar norm = computeNormOfT(); - if(norm!=Scalar(0)) - { - while (iu >= 0) - { + if (norm != Scalar(0)) { + while (iu >= 0) { Index il = findSmallSubdiagEntry(iu); // Check for convergence - if (il == iu) // One root found + if (il == iu)// One root found { - m_matT.coeffRef(iu,iu) = m_matT.coeff(iu,iu) + exshift; - if (iu > 0) - m_matT.coeffRef(iu, iu-1) = Scalar(0); + m_matT.coeffRef(iu, iu) = m_matT.coeff(iu, iu) + exshift; + if (iu > 0) m_matT.coeffRef(iu, iu - 1) = Scalar(0); iu--; iter = 0; - } - else if (il == iu-1) // Two roots found + } else if (il == iu - 1)// Two roots found { splitOffTwoRows(iu, computeU, exshift); iu -= 2; iter = 0; - } - else // No convergence yet + } else// No convergence yet { - // The firstHouseholderVector vector has to be initialized to something to get rid of a silly GCC warning (-O1 -Wall -DNDEBUG ) + // The firstHouseholderVector vector has to be initialized to something to get rid of a silly GCC warning (-O1 + // -Wall -DNDEBUG ) Vector3s firstHouseholderVector = Vector3s::Zero(), shiftInfo; computeShift(iu, iter, exshift, shiftInfo); iter = iter + 1; @@ -338,7 +321,7 @@ RealSchur& RealSchur::computeFromHessenberg(const HessMa } } } - if(totalIter <= maxIters) + if (totalIter <= maxIters) m_info = Success; else m_info = NoConvergence; @@ -349,30 +332,25 @@ RealSchur& RealSchur::computeFromHessenberg(const HessMa } /** \internal Computes and returns vector L1 norm of T */ -template -inline typename MatrixType::Scalar RealSchur::computeNormOfT() +template inline typename MatrixType::Scalar RealSchur::computeNormOfT() { const Index size = m_matT.cols(); // FIXME to be efficient the following would requires a triangular reduxion code - // Scalar norm = m_matT.upper().cwiseAbs().sum() + // Scalar norm = m_matT.upper().cwiseAbs().sum() // + m_matT.bottomLeftCorner(size-1,size-1).diagonal().cwiseAbs().sum(); Scalar norm(0); - for (Index j = 0; j < size; ++j) - norm += m_matT.col(j).segment(0, (std::min)(size,j+2)).cwiseAbs().sum(); + for (Index j = 0; j < size; ++j) norm += m_matT.col(j).segment(0, (std::min)(size, j + 2)).cwiseAbs().sum(); return norm; } /** \internal Look for single small sub-diagonal element and returns its index */ -template -inline Index RealSchur::findSmallSubdiagEntry(Index iu) +template inline Index RealSchur::findSmallSubdiagEntry(Index iu) { using std::abs; Index res = iu; - while (res > 0) - { - Scalar s = abs(m_matT.coeff(res-1,res-1)) + abs(m_matT.coeff(res,res)); - if (abs(m_matT.coeff(res,res-1)) <= NumTraits::epsilon() * s) - break; + while (res > 0) { + Scalar s = abs(m_matT.coeff(res - 1, res - 1)) + abs(m_matT.coeff(res, res)); + if (abs(m_matT.coeff(res, res - 1)) <= NumTraits::epsilon() * s) break; res--; } return res; @@ -380,76 +358,68 @@ inline Index RealSchur::findSmallSubdiagEntry(Index iu) /** \internal Update T given that rows iu-1 and iu decouple from the rest. */ template -inline void RealSchur::splitOffTwoRows(Index iu, bool computeU, const Scalar& exshift) +inline void RealSchur::splitOffTwoRows(Index iu, bool computeU, const Scalar &exshift) { using std::sqrt; using std::abs; const Index size = m_matT.cols(); - // The eigenvalues of the 2x2 matrix [a b; c d] are + // The eigenvalues of the 2x2 matrix [a b; c d] are // trace +/- sqrt(discr/4) where discr = tr^2 - 4*det, tr = a + d, det = ad - bc - Scalar p = Scalar(0.5) * (m_matT.coeff(iu-1,iu-1) - m_matT.coeff(iu,iu)); - Scalar q = p * p + m_matT.coeff(iu,iu-1) * m_matT.coeff(iu-1,iu); // q = tr^2 / 4 - det = discr/4 - m_matT.coeffRef(iu,iu) += exshift; - m_matT.coeffRef(iu-1,iu-1) += exshift; + Scalar p = Scalar(0.5) * (m_matT.coeff(iu - 1, iu - 1) - m_matT.coeff(iu, iu)); + Scalar q = p * p + m_matT.coeff(iu, iu - 1) * m_matT.coeff(iu - 1, iu);// q = tr^2 / 4 - det = discr/4 + m_matT.coeffRef(iu, iu) += exshift; + m_matT.coeffRef(iu - 1, iu - 1) += exshift; - if (q >= Scalar(0)) // Two real eigenvalues + if (q >= Scalar(0))// Two real eigenvalues { Scalar z = sqrt(abs(q)); JacobiRotation rot; if (p >= Scalar(0)) - rot.makeGivens(p + z, m_matT.coeff(iu, iu-1)); + rot.makeGivens(p + z, m_matT.coeff(iu, iu - 1)); else - rot.makeGivens(p - z, m_matT.coeff(iu, iu-1)); + rot.makeGivens(p - z, m_matT.coeff(iu, iu - 1)); - m_matT.rightCols(size-iu+1).applyOnTheLeft(iu-1, iu, rot.adjoint()); - m_matT.topRows(iu+1).applyOnTheRight(iu-1, iu, rot); - m_matT.coeffRef(iu, iu-1) = Scalar(0); - if (computeU) - m_matU.applyOnTheRight(iu-1, iu, rot); + m_matT.rightCols(size - iu + 1).applyOnTheLeft(iu - 1, iu, rot.adjoint()); + m_matT.topRows(iu + 1).applyOnTheRight(iu - 1, iu, rot); + m_matT.coeffRef(iu, iu - 1) = Scalar(0); + if (computeU) m_matU.applyOnTheRight(iu - 1, iu, rot); } - if (iu > 1) - m_matT.coeffRef(iu-1, iu-2) = Scalar(0); + if (iu > 1) m_matT.coeffRef(iu - 1, iu - 2) = Scalar(0); } /** \internal Form shift in shiftInfo, and update exshift if an exceptional shift is performed. */ template -inline void RealSchur::computeShift(Index iu, Index iter, Scalar& exshift, Vector3s& shiftInfo) +inline void RealSchur::computeShift(Index iu, Index iter, Scalar &exshift, Vector3s &shiftInfo) { using std::sqrt; using std::abs; - shiftInfo.coeffRef(0) = m_matT.coeff(iu,iu); - shiftInfo.coeffRef(1) = m_matT.coeff(iu-1,iu-1); - shiftInfo.coeffRef(2) = m_matT.coeff(iu,iu-1) * m_matT.coeff(iu-1,iu); + shiftInfo.coeffRef(0) = m_matT.coeff(iu, iu); + shiftInfo.coeffRef(1) = m_matT.coeff(iu - 1, iu - 1); + shiftInfo.coeffRef(2) = m_matT.coeff(iu, iu - 1) * m_matT.coeff(iu - 1, iu); // Wilkinson's original ad hoc shift - if (iter == 10) - { + if (iter == 10) { exshift += shiftInfo.coeff(0); - for (Index i = 0; i <= iu; ++i) - m_matT.coeffRef(i,i) -= shiftInfo.coeff(0); - Scalar s = abs(m_matT.coeff(iu,iu-1)) + abs(m_matT.coeff(iu-1,iu-2)); + for (Index i = 0; i <= iu; ++i) m_matT.coeffRef(i, i) -= shiftInfo.coeff(0); + Scalar s = abs(m_matT.coeff(iu, iu - 1)) + abs(m_matT.coeff(iu - 1, iu - 2)); shiftInfo.coeffRef(0) = Scalar(0.75) * s; shiftInfo.coeffRef(1) = Scalar(0.75) * s; shiftInfo.coeffRef(2) = Scalar(-0.4375) * s * s; } // MATLAB's new ad hoc shift - if (iter == 30) - { + if (iter == 30) { Scalar s = (shiftInfo.coeff(1) - shiftInfo.coeff(0)) / Scalar(2.0); s = s * s + shiftInfo.coeff(2); - if (s > Scalar(0)) - { + if (s > Scalar(0)) { s = sqrt(s); - if (shiftInfo.coeff(1) < shiftInfo.coeff(0)) - s = -s; + if (shiftInfo.coeff(1) < shiftInfo.coeff(0)) s = -s; s = s + (shiftInfo.coeff(1) - shiftInfo.coeff(0)) / Scalar(2.0); s = shiftInfo.coeff(0) - shiftInfo.coeff(2) / s; exshift += s; - for (Index i = 0; i <= iu; ++i) - m_matT.coeffRef(i,i) -= s; + for (Index i = 0; i <= iu; ++i) m_matT.coeffRef(i, i) -= s; shiftInfo.setConstant(Scalar(0.964)); } } @@ -457,90 +427,90 @@ inline void RealSchur::computeShift(Index iu, Index iter, Scalar& ex /** \internal Compute index im at which Francis QR step starts and the first Householder vector. */ template -inline void RealSchur::initFrancisQRStep(Index il, Index iu, const Vector3s& shiftInfo, Index& im, Vector3s& firstHouseholderVector) +inline void RealSchur::initFrancisQRStep(Index il, + Index iu, + const Vector3s &shiftInfo, + Index &im, + Vector3s &firstHouseholderVector) { using std::abs; - Vector3s& v = firstHouseholderVector; // alias to save typing + Vector3s &v = firstHouseholderVector;// alias to save typing - for (im = iu-2; im >= il; --im) - { - const Scalar Tmm = m_matT.coeff(im,im); + for (im = iu - 2; im >= il; --im) { + const Scalar Tmm = m_matT.coeff(im, im); const Scalar r = shiftInfo.coeff(0) - Tmm; const Scalar s = shiftInfo.coeff(1) - Tmm; - v.coeffRef(0) = (r * s - shiftInfo.coeff(2)) / m_matT.coeff(im+1,im) + m_matT.coeff(im,im+1); - v.coeffRef(1) = m_matT.coeff(im+1,im+1) - Tmm - r - s; - v.coeffRef(2) = m_matT.coeff(im+2,im+1); - if (im == il) { - break; - } - const Scalar lhs = m_matT.coeff(im,im-1) * (abs(v.coeff(1)) + abs(v.coeff(2))); - const Scalar rhs = v.coeff(0) * (abs(m_matT.coeff(im-1,im-1)) + abs(Tmm) + abs(m_matT.coeff(im+1,im+1))); - if (abs(lhs) < NumTraits::epsilon() * rhs) - break; + v.coeffRef(0) = (r * s - shiftInfo.coeff(2)) / m_matT.coeff(im + 1, im) + m_matT.coeff(im, im + 1); + v.coeffRef(1) = m_matT.coeff(im + 1, im + 1) - Tmm - r - s; + v.coeffRef(2) = m_matT.coeff(im + 2, im + 1); + if (im == il) { break; } + const Scalar lhs = m_matT.coeff(im, im - 1) * (abs(v.coeff(1)) + abs(v.coeff(2))); + const Scalar rhs = v.coeff(0) * (abs(m_matT.coeff(im - 1, im - 1)) + abs(Tmm) + abs(m_matT.coeff(im + 1, im + 1))); + if (abs(lhs) < NumTraits::epsilon() * rhs) break; } } /** \internal Perform a Francis QR step involving rows il:iu and columns im:iu. */ template -inline void RealSchur::performFrancisQRStep(Index il, Index im, Index iu, bool computeU, const Vector3s& firstHouseholderVector, Scalar* workspace) +inline void RealSchur::performFrancisQRStep(Index il, + Index im, + Index iu, + bool computeU, + const Vector3s &firstHouseholderVector, + Scalar *workspace) { eigen_assert(im >= il); - eigen_assert(im <= iu-2); + eigen_assert(im <= iu - 2); const Index size = m_matT.cols(); - for (Index k = im; k <= iu-2; ++k) - { + for (Index k = im; k <= iu - 2; ++k) { bool firstIteration = (k == im); Vector3s v; if (firstIteration) v = firstHouseholderVector; else - v = m_matT.template block<3,1>(k,k-1); + v = m_matT.template block<3, 1>(k, k - 1); Scalar tau, beta; Matrix ess; v.makeHouseholder(ess, tau, beta); - - if (beta != Scalar(0)) // if v is not zero + + if (beta != Scalar(0))// if v is not zero { if (firstIteration && k > il) - m_matT.coeffRef(k,k-1) = -m_matT.coeff(k,k-1); + m_matT.coeffRef(k, k - 1) = -m_matT.coeff(k, k - 1); else if (!firstIteration) - m_matT.coeffRef(k,k-1) = beta; + m_matT.coeffRef(k, k - 1) = beta; // These Householder transformations form the O(n^3) part of the algorithm - m_matT.block(k, k, 3, size-k).applyHouseholderOnTheLeft(ess, tau, workspace); - m_matT.block(0, k, (std::min)(iu,k+3) + 1, 3).applyHouseholderOnTheRight(ess, tau, workspace); - if (computeU) - m_matU.block(0, k, size, 3).applyHouseholderOnTheRight(ess, tau, workspace); + m_matT.block(k, k, 3, size - k).applyHouseholderOnTheLeft(ess, tau, workspace); + m_matT.block(0, k, (std::min)(iu, k + 3) + 1, 3).applyHouseholderOnTheRight(ess, tau, workspace); + if (computeU) m_matU.block(0, k, size, 3).applyHouseholderOnTheRight(ess, tau, workspace); } } - Matrix v = m_matT.template block<2,1>(iu-1, iu-2); + Matrix v = m_matT.template block<2, 1>(iu - 1, iu - 2); Scalar tau, beta; Matrix ess; v.makeHouseholder(ess, tau, beta); - if (beta != Scalar(0)) // if v is not zero + if (beta != Scalar(0))// if v is not zero { - m_matT.coeffRef(iu-1, iu-2) = beta; - m_matT.block(iu-1, iu-1, 2, size-iu+1).applyHouseholderOnTheLeft(ess, tau, workspace); - m_matT.block(0, iu-1, iu+1, 2).applyHouseholderOnTheRight(ess, tau, workspace); - if (computeU) - m_matU.block(0, iu-1, size, 2).applyHouseholderOnTheRight(ess, tau, workspace); + m_matT.coeffRef(iu - 1, iu - 2) = beta; + m_matT.block(iu - 1, iu - 1, 2, size - iu + 1).applyHouseholderOnTheLeft(ess, tau, workspace); + m_matT.block(0, iu - 1, iu + 1, 2).applyHouseholderOnTheRight(ess, tau, workspace); + if (computeU) m_matU.block(0, iu - 1, size, 2).applyHouseholderOnTheRight(ess, tau, workspace); } // clean up pollution due to round-off errors - for (Index i = im+2; i <= iu; ++i) - { - m_matT.coeffRef(i,i-2) = Scalar(0); - if (i > im+2) - m_matT.coeffRef(i,i-3) = Scalar(0); + for (Index i = im + 2; i <= iu; ++i) { + m_matT.coeffRef(i, i - 2) = Scalar(0); + if (i > im + 2) m_matT.coeffRef(i, i - 3) = Scalar(0); } } -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_REAL_SCHUR_H +#endif// EIGEN_REAL_SCHUR_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Eigenvalues/RealSchur_LAPACKE.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Eigenvalues/RealSchur_LAPACKE.h index 2c225171..27421178 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Eigenvalues/RealSchur_LAPACKE.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Eigenvalues/RealSchur_LAPACKE.h @@ -33,45 +33,58 @@ #ifndef EIGEN_REAL_SCHUR_LAPACKE_H #define EIGEN_REAL_SCHUR_LAPACKE_H -namespace Eigen { +namespace Eigen { /** \internal Specialization for the data types supported by LAPACKe */ #define EIGEN_LAPACKE_SCHUR_REAL(EIGTYPE, LAPACKE_TYPE, LAPACKE_PREFIX, LAPACKE_PREFIX_U, EIGCOLROW, LAPACKE_COLROW) \ -template<> template inline \ -RealSchur >& \ -RealSchur >::compute(const EigenBase& matrix, bool computeU) \ -{ \ - eigen_assert(matrix.cols() == matrix.rows()); \ -\ - lapack_int n = internal::convert_index(matrix.cols()), sdim, info; \ - lapack_int matrix_order = LAPACKE_COLROW; \ - char jobvs, sort='N'; \ - LAPACK_##LAPACKE_PREFIX_U##_SELECT2 select = 0; \ - jobvs = (computeU) ? 'V' : 'N'; \ - m_matU.resize(n, n); \ - lapack_int ldvs = internal::convert_index(m_matU.outerStride()); \ - m_matT = matrix; \ - lapack_int lda = internal::convert_index(m_matT.outerStride()); \ - Matrix wr, wi; \ - wr.resize(n, 1); wi.resize(n, 1); \ - info = LAPACKE_##LAPACKE_PREFIX##gees( matrix_order, jobvs, sort, select, n, (LAPACKE_TYPE*)m_matT.data(), lda, &sdim, (LAPACKE_TYPE*)wr.data(), (LAPACKE_TYPE*)wi.data(), (LAPACKE_TYPE*)m_matU.data(), ldvs ); \ - if(info == 0) \ - m_info = Success; \ - else \ - m_info = NoConvergence; \ -\ - m_isInitialized = true; \ - m_matUisUptodate = computeU; \ - return *this; \ -\ -} + template<> \ + template \ + inline RealSchur> & \ + RealSchur>::compute( \ + const EigenBase &matrix, bool computeU) \ + { \ + eigen_assert(matrix.cols() == matrix.rows()); \ + \ + lapack_int n = internal::convert_index(matrix.cols()), sdim, info; \ + lapack_int matrix_order = LAPACKE_COLROW; \ + char jobvs, sort = 'N'; \ + LAPACK_##LAPACKE_PREFIX_U##_SELECT2 select = 0; \ + jobvs = (computeU) ? 'V' : 'N'; \ + m_matU.resize(n, n); \ + lapack_int ldvs = internal::convert_index(m_matU.outerStride()); \ + m_matT = matrix; \ + lapack_int lda = internal::convert_index(m_matT.outerStride()); \ + Matrix wr, wi; \ + wr.resize(n, 1); \ + wi.resize(n, 1); \ + info = LAPACKE_##LAPACKE_PREFIX##gees(matrix_order, \ + jobvs, \ + sort, \ + select, \ + n, \ + (LAPACKE_TYPE *)m_matT.data(), \ + lda, \ + &sdim, \ + (LAPACKE_TYPE *)wr.data(), \ + (LAPACKE_TYPE *)wi.data(), \ + (LAPACKE_TYPE *)m_matU.data(), \ + ldvs); \ + if (info == 0) \ + m_info = Success; \ + else \ + m_info = NoConvergence; \ + \ + m_isInitialized = true; \ + m_matUisUptodate = computeU; \ + return *this; \ + } -EIGEN_LAPACKE_SCHUR_REAL(double, double, d, D, ColMajor, LAPACK_COL_MAJOR) -EIGEN_LAPACKE_SCHUR_REAL(float, float, s, S, ColMajor, LAPACK_COL_MAJOR) -EIGEN_LAPACKE_SCHUR_REAL(double, double, d, D, RowMajor, LAPACK_ROW_MAJOR) -EIGEN_LAPACKE_SCHUR_REAL(float, float, s, S, RowMajor, LAPACK_ROW_MAJOR) +EIGEN_LAPACKE_SCHUR_REAL(double, double, d, D, ColMajor, LAPACK_COL_MAJOR) +EIGEN_LAPACKE_SCHUR_REAL(float, float, s, S, ColMajor, LAPACK_COL_MAJOR) +EIGEN_LAPACKE_SCHUR_REAL(double, double, d, D, RowMajor, LAPACK_ROW_MAJOR) +EIGEN_LAPACKE_SCHUR_REAL(float, float, s, S, RowMajor, LAPACK_ROW_MAJOR) -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_REAL_SCHUR_LAPACKE_H +#endif// EIGEN_REAL_SCHUR_LAPACKE_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Eigenvalues/SelfAdjointEigenSolver.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Eigenvalues/SelfAdjointEigenSolver.h index 9ddd553f..0e7df81f 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Eigenvalues/SelfAdjointEigenSolver.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Eigenvalues/SelfAdjointEigenSolver.h @@ -13,411 +13,399 @@ #include "./Tridiagonalization.h" -namespace Eigen { +namespace Eigen { -template -class GeneralizedSelfAdjointEigenSolver; +template class GeneralizedSelfAdjointEigenSolver; namespace internal { -template struct direct_selfadjoint_eigenvalues; -template -ComputationInfo computeFromTridiagonal_impl(DiagType& diag, SubDiagType& subdiag, const Index maxIterations, bool computeEigenvectors, MatrixType& eivec); -} + template struct direct_selfadjoint_eigenvalues; + template + ComputationInfo computeFromTridiagonal_impl(DiagType &diag, + SubDiagType &subdiag, + const Index maxIterations, + bool computeEigenvectors, + MatrixType &eivec); +}// namespace internal /** \eigenvalues_module \ingroup Eigenvalues_Module - * - * - * \class SelfAdjointEigenSolver - * - * \brief Computes eigenvalues and eigenvectors of selfadjoint matrices - * - * \tparam _MatrixType the type of the matrix of which we are computing the - * eigendecomposition; this is expected to be an instantiation of the Matrix - * class template. - * - * A matrix \f$ A \f$ is selfadjoint if it equals its adjoint. For real - * matrices, this means that the matrix is symmetric: it equals its - * transpose. This class computes the eigenvalues and eigenvectors of a - * selfadjoint matrix. These are the scalars \f$ \lambda \f$ and vectors - * \f$ v \f$ such that \f$ Av = \lambda v \f$. The eigenvalues of a - * selfadjoint matrix are always real. If \f$ D \f$ is a diagonal matrix with - * the eigenvalues on the diagonal, and \f$ V \f$ is a matrix with the - * eigenvectors as its columns, then \f$ A = V D V^{-1} \f$ (for selfadjoint - * matrices, the matrix \f$ V \f$ is always invertible). This is called the - * eigendecomposition. - * - * The algorithm exploits the fact that the matrix is selfadjoint, making it - * faster and more accurate than the general purpose eigenvalue algorithms - * implemented in EigenSolver and ComplexEigenSolver. - * - * Only the \b lower \b triangular \b part of the input matrix is referenced. - * - * Call the function compute() to compute the eigenvalues and eigenvectors of - * a given matrix. Alternatively, you can use the - * SelfAdjointEigenSolver(const MatrixType&, int) constructor which computes - * the eigenvalues and eigenvectors at construction time. Once the eigenvalue - * and eigenvectors are computed, they can be retrieved with the eigenvalues() - * and eigenvectors() functions. - * - * The documentation for SelfAdjointEigenSolver(const MatrixType&, int) - * contains an example of the typical use of this class. - * - * To solve the \em generalized eigenvalue problem \f$ Av = \lambda Bv \f$ and - * the likes, see the class GeneralizedSelfAdjointEigenSolver. - * - * \sa MatrixBase::eigenvalues(), class EigenSolver, class ComplexEigenSolver - */ + * + * + * \class SelfAdjointEigenSolver + * + * \brief Computes eigenvalues and eigenvectors of selfadjoint matrices + * + * \tparam _MatrixType the type of the matrix of which we are computing the + * eigendecomposition; this is expected to be an instantiation of the Matrix + * class template. + * + * A matrix \f$ A \f$ is selfadjoint if it equals its adjoint. For real + * matrices, this means that the matrix is symmetric: it equals its + * transpose. This class computes the eigenvalues and eigenvectors of a + * selfadjoint matrix. These are the scalars \f$ \lambda \f$ and vectors + * \f$ v \f$ such that \f$ Av = \lambda v \f$. The eigenvalues of a + * selfadjoint matrix are always real. If \f$ D \f$ is a diagonal matrix with + * the eigenvalues on the diagonal, and \f$ V \f$ is a matrix with the + * eigenvectors as its columns, then \f$ A = V D V^{-1} \f$ (for selfadjoint + * matrices, the matrix \f$ V \f$ is always invertible). This is called the + * eigendecomposition. + * + * The algorithm exploits the fact that the matrix is selfadjoint, making it + * faster and more accurate than the general purpose eigenvalue algorithms + * implemented in EigenSolver and ComplexEigenSolver. + * + * Only the \b lower \b triangular \b part of the input matrix is referenced. + * + * Call the function compute() to compute the eigenvalues and eigenvectors of + * a given matrix. Alternatively, you can use the + * SelfAdjointEigenSolver(const MatrixType&, int) constructor which computes + * the eigenvalues and eigenvectors at construction time. Once the eigenvalue + * and eigenvectors are computed, they can be retrieved with the eigenvalues() + * and eigenvectors() functions. + * + * The documentation for SelfAdjointEigenSolver(const MatrixType&, int) + * contains an example of the typical use of this class. + * + * To solve the \em generalized eigenvalue problem \f$ Av = \lambda Bv \f$ and + * the likes, see the class GeneralizedSelfAdjointEigenSolver. + * + * \sa MatrixBase::eigenvalues(), class EigenSolver, class ComplexEigenSolver + */ template class SelfAdjointEigenSolver { - public: - - typedef _MatrixType MatrixType; - enum { - Size = MatrixType::RowsAtCompileTime, - ColsAtCompileTime = MatrixType::ColsAtCompileTime, - Options = MatrixType::Options, - MaxColsAtCompileTime = MatrixType::MaxColsAtCompileTime - }; - - /** \brief Scalar type for matrices of type \p _MatrixType. */ - typedef typename MatrixType::Scalar Scalar; - typedef Eigen::Index Index; ///< \deprecated since Eigen 3.3 - - typedef Matrix EigenvectorsType; - - /** \brief Real scalar type for \p _MatrixType. - * - * This is just \c Scalar if #Scalar is real (e.g., \c float or - * \c double), and the type of the real part of \c Scalar if #Scalar is - * complex. - */ - typedef typename NumTraits::Real RealScalar; - - friend struct internal::direct_selfadjoint_eigenvalues::IsComplex>; - - /** \brief Type for vector of eigenvalues as returned by eigenvalues(). - * - * This is a column vector with entries of type #RealScalar. - * The length of the vector is the size of \p _MatrixType. - */ - typedef typename internal::plain_col_type::type RealVectorType; - typedef Tridiagonalization TridiagonalizationType; - typedef typename TridiagonalizationType::SubDiagonalType SubDiagonalType; - - /** \brief Default constructor for fixed-size matrices. - * - * The default constructor is useful in cases in which the user intends to - * perform decompositions via compute(). This constructor - * can only be used if \p _MatrixType is a fixed-size matrix; use - * SelfAdjointEigenSolver(Index) for dynamic-size matrices. - * - * Example: \include SelfAdjointEigenSolver_SelfAdjointEigenSolver.cpp - * Output: \verbinclude SelfAdjointEigenSolver_SelfAdjointEigenSolver.out - */ - EIGEN_DEVICE_FUNC - SelfAdjointEigenSolver() - : m_eivec(), - m_eivalues(), - m_subdiag(), - m_isInitialized(false) - { } - - /** \brief Constructor, pre-allocates memory for dynamic-size matrices. - * - * \param [in] size Positive integer, size of the matrix whose - * eigenvalues and eigenvectors will be computed. - * - * This constructor is useful for dynamic-size matrices, when the user - * intends to perform decompositions via compute(). The \p size - * parameter is only used as a hint. It is not an error to give a wrong - * \p size, but it may impair performance. - * - * \sa compute() for an example - */ - EIGEN_DEVICE_FUNC - explicit SelfAdjointEigenSolver(Index size) - : m_eivec(size, size), - m_eivalues(size), - m_subdiag(size > 1 ? size - 1 : 1), - m_isInitialized(false) - {} - - /** \brief Constructor; computes eigendecomposition of given matrix. - * - * \param[in] matrix Selfadjoint matrix whose eigendecomposition is to - * be computed. Only the lower triangular part of the matrix is referenced. - * \param[in] options Can be #ComputeEigenvectors (default) or #EigenvaluesOnly. - * - * This constructor calls compute(const MatrixType&, int) to compute the - * eigenvalues of the matrix \p matrix. The eigenvectors are computed if - * \p options equals #ComputeEigenvectors. - * - * Example: \include SelfAdjointEigenSolver_SelfAdjointEigenSolver_MatrixType.cpp - * Output: \verbinclude SelfAdjointEigenSolver_SelfAdjointEigenSolver_MatrixType.out - * - * \sa compute(const MatrixType&, int) - */ - template - EIGEN_DEVICE_FUNC - explicit SelfAdjointEigenSolver(const EigenBase& matrix, int options = ComputeEigenvectors) - : m_eivec(matrix.rows(), matrix.cols()), - m_eivalues(matrix.cols()), - m_subdiag(matrix.rows() > 1 ? matrix.rows() - 1 : 1), - m_isInitialized(false) - { - compute(matrix.derived(), options); - } +public: + typedef _MatrixType MatrixType; + enum { + Size = MatrixType::RowsAtCompileTime, + ColsAtCompileTime = MatrixType::ColsAtCompileTime, + Options = MatrixType::Options, + MaxColsAtCompileTime = MatrixType::MaxColsAtCompileTime + }; + + /** \brief Scalar type for matrices of type \p _MatrixType. */ + typedef typename MatrixType::Scalar Scalar; + typedef Eigen::Index Index;///< \deprecated since Eigen 3.3 - /** \brief Computes eigendecomposition of given matrix. - * - * \param[in] matrix Selfadjoint matrix whose eigendecomposition is to - * be computed. Only the lower triangular part of the matrix is referenced. - * \param[in] options Can be #ComputeEigenvectors (default) or #EigenvaluesOnly. - * \returns Reference to \c *this - * - * This function computes the eigenvalues of \p matrix. The eigenvalues() - * function can be used to retrieve them. If \p options equals #ComputeEigenvectors, - * then the eigenvectors are also computed and can be retrieved by - * calling eigenvectors(). - * - * This implementation uses a symmetric QR algorithm. The matrix is first - * reduced to tridiagonal form using the Tridiagonalization class. The - * tridiagonal matrix is then brought to diagonal form with implicit - * symmetric QR steps with Wilkinson shift. Details can be found in - * Section 8.3 of Golub \& Van Loan, %Matrix Computations. - * - * The cost of the computation is about \f$ 9n^3 \f$ if the eigenvectors - * are required and \f$ 4n^3/3 \f$ if they are not required. - * - * This method reuses the memory in the SelfAdjointEigenSolver object that - * was allocated when the object was constructed, if the size of the - * matrix does not change. - * - * Example: \include SelfAdjointEigenSolver_compute_MatrixType.cpp - * Output: \verbinclude SelfAdjointEigenSolver_compute_MatrixType.out - * - * \sa SelfAdjointEigenSolver(const MatrixType&, int) - */ - template - EIGEN_DEVICE_FUNC - SelfAdjointEigenSolver& compute(const EigenBase& matrix, int options = ComputeEigenvectors); - - /** \brief Computes eigendecomposition of given matrix using a closed-form algorithm - * - * This is a variant of compute(const MatrixType&, int options) which - * directly solves the underlying polynomial equation. - * - * Currently only 2x2 and 3x3 matrices for which the sizes are known at compile time are supported (e.g., Matrix3d). - * - * This method is usually significantly faster than the QR iterative algorithm - * but it might also be less accurate. It is also worth noting that - * for 3x3 matrices it involves trigonometric operations which are - * not necessarily available for all scalar types. - * - * For the 3x3 case, we observed the following worst case relative error regarding the eigenvalues: - * - double: 1e-8 - * - float: 1e-3 - * - * \sa compute(const MatrixType&, int options) - */ - EIGEN_DEVICE_FUNC - SelfAdjointEigenSolver& computeDirect(const MatrixType& matrix, int options = ComputeEigenvectors); - - /** - *\brief Computes the eigen decomposition from a tridiagonal symmetric matrix - * - * \param[in] diag The vector containing the diagonal of the matrix. - * \param[in] subdiag The subdiagonal of the matrix. - * \param[in] options Can be #ComputeEigenvectors (default) or #EigenvaluesOnly. - * \returns Reference to \c *this - * - * This function assumes that the matrix has been reduced to tridiagonal form. - * - * \sa compute(const MatrixType&, int) for more information - */ - SelfAdjointEigenSolver& computeFromTridiagonal(const RealVectorType& diag, const SubDiagonalType& subdiag , int options=ComputeEigenvectors); - - /** \brief Returns the eigenvectors of given matrix. - * - * \returns A const reference to the matrix whose columns are the eigenvectors. - * - * \pre The eigenvectors have been computed before. - * - * Column \f$ k \f$ of the returned matrix is an eigenvector corresponding - * to eigenvalue number \f$ k \f$ as returned by eigenvalues(). The - * eigenvectors are normalized to have (Euclidean) norm equal to one. If - * this object was used to solve the eigenproblem for the selfadjoint - * matrix \f$ A \f$, then the matrix returned by this function is the - * matrix \f$ V \f$ in the eigendecomposition \f$ A = V D V^{-1} \f$. - * - * Example: \include SelfAdjointEigenSolver_eigenvectors.cpp - * Output: \verbinclude SelfAdjointEigenSolver_eigenvectors.out - * - * \sa eigenvalues() - */ - EIGEN_DEVICE_FUNC - const EigenvectorsType& eigenvectors() const - { - eigen_assert(m_isInitialized && "SelfAdjointEigenSolver is not initialized."); - eigen_assert(m_eigenvectorsOk && "The eigenvectors have not been computed together with the eigenvalues."); - return m_eivec; - } + typedef Matrix EigenvectorsType; - /** \brief Returns the eigenvalues of given matrix. - * - * \returns A const reference to the column vector containing the eigenvalues. - * - * \pre The eigenvalues have been computed before. - * - * The eigenvalues are repeated according to their algebraic multiplicity, - * so there are as many eigenvalues as rows in the matrix. The eigenvalues - * are sorted in increasing order. - * - * Example: \include SelfAdjointEigenSolver_eigenvalues.cpp - * Output: \verbinclude SelfAdjointEigenSolver_eigenvalues.out - * - * \sa eigenvectors(), MatrixBase::eigenvalues() - */ - EIGEN_DEVICE_FUNC - const RealVectorType& eigenvalues() const - { - eigen_assert(m_isInitialized && "SelfAdjointEigenSolver is not initialized."); - return m_eivalues; - } + /** \brief Real scalar type for \p _MatrixType. + * + * This is just \c Scalar if #Scalar is real (e.g., \c float or + * \c double), and the type of the real part of \c Scalar if #Scalar is + * complex. + */ + typedef typename NumTraits::Real RealScalar; - /** \brief Computes the positive-definite square root of the matrix. - * - * \returns the positive-definite square root of the matrix - * - * \pre The eigenvalues and eigenvectors of a positive-definite matrix - * have been computed before. - * - * The square root of a positive-definite matrix \f$ A \f$ is the - * positive-definite matrix whose square equals \f$ A \f$. This function - * uses the eigendecomposition \f$ A = V D V^{-1} \f$ to compute the - * square root as \f$ A^{1/2} = V D^{1/2} V^{-1} \f$. - * - * Example: \include SelfAdjointEigenSolver_operatorSqrt.cpp - * Output: \verbinclude SelfAdjointEigenSolver_operatorSqrt.out - * - * \sa operatorInverseSqrt(), MatrixFunctions Module - */ - EIGEN_DEVICE_FUNC - MatrixType operatorSqrt() const - { - eigen_assert(m_isInitialized && "SelfAdjointEigenSolver is not initialized."); - eigen_assert(m_eigenvectorsOk && "The eigenvectors have not been computed together with the eigenvalues."); - return m_eivec * m_eivalues.cwiseSqrt().asDiagonal() * m_eivec.adjoint(); - } + friend struct internal::direct_selfadjoint_eigenvalues::IsComplex>; - /** \brief Computes the inverse square root of the matrix. - * - * \returns the inverse positive-definite square root of the matrix - * - * \pre The eigenvalues and eigenvectors of a positive-definite matrix - * have been computed before. - * - * This function uses the eigendecomposition \f$ A = V D V^{-1} \f$ to - * compute the inverse square root as \f$ V D^{-1/2} V^{-1} \f$. This is - * cheaper than first computing the square root with operatorSqrt() and - * then its inverse with MatrixBase::inverse(). - * - * Example: \include SelfAdjointEigenSolver_operatorInverseSqrt.cpp - * Output: \verbinclude SelfAdjointEigenSolver_operatorInverseSqrt.out - * - * \sa operatorSqrt(), MatrixBase::inverse(), MatrixFunctions Module - */ - EIGEN_DEVICE_FUNC - MatrixType operatorInverseSqrt() const - { - eigen_assert(m_isInitialized && "SelfAdjointEigenSolver is not initialized."); - eigen_assert(m_eigenvectorsOk && "The eigenvectors have not been computed together with the eigenvalues."); - return m_eivec * m_eivalues.cwiseInverse().cwiseSqrt().asDiagonal() * m_eivec.adjoint(); - } + /** \brief Type for vector of eigenvalues as returned by eigenvalues(). + * + * This is a column vector with entries of type #RealScalar. + * The length of the vector is the size of \p _MatrixType. + */ + typedef typename internal::plain_col_type::type RealVectorType; + typedef Tridiagonalization TridiagonalizationType; + typedef typename TridiagonalizationType::SubDiagonalType SubDiagonalType; + + /** \brief Default constructor for fixed-size matrices. + * + * The default constructor is useful in cases in which the user intends to + * perform decompositions via compute(). This constructor + * can only be used if \p _MatrixType is a fixed-size matrix; use + * SelfAdjointEigenSolver(Index) for dynamic-size matrices. + * + * Example: \include SelfAdjointEigenSolver_SelfAdjointEigenSolver.cpp + * Output: \verbinclude SelfAdjointEigenSolver_SelfAdjointEigenSolver.out + */ + EIGEN_DEVICE_FUNC + SelfAdjointEigenSolver() : m_eivec(), m_eivalues(), m_subdiag(), m_isInitialized(false) {} + + /** \brief Constructor, pre-allocates memory for dynamic-size matrices. + * + * \param [in] size Positive integer, size of the matrix whose + * eigenvalues and eigenvectors will be computed. + * + * This constructor is useful for dynamic-size matrices, when the user + * intends to perform decompositions via compute(). The \p size + * parameter is only used as a hint. It is not an error to give a wrong + * \p size, but it may impair performance. + * + * \sa compute() for an example + */ + EIGEN_DEVICE_FUNC + explicit SelfAdjointEigenSolver(Index size) + : m_eivec(size, size), m_eivalues(size), m_subdiag(size > 1 ? size - 1 : 1), m_isInitialized(false) + {} + + /** \brief Constructor; computes eigendecomposition of given matrix. + * + * \param[in] matrix Selfadjoint matrix whose eigendecomposition is to + * be computed. Only the lower triangular part of the matrix is referenced. + * \param[in] options Can be #ComputeEigenvectors (default) or #EigenvaluesOnly. + * + * This constructor calls compute(const MatrixType&, int) to compute the + * eigenvalues of the matrix \p matrix. The eigenvectors are computed if + * \p options equals #ComputeEigenvectors. + * + * Example: \include SelfAdjointEigenSolver_SelfAdjointEigenSolver_MatrixType.cpp + * Output: \verbinclude SelfAdjointEigenSolver_SelfAdjointEigenSolver_MatrixType.out + * + * \sa compute(const MatrixType&, int) + */ + template + EIGEN_DEVICE_FUNC explicit SelfAdjointEigenSolver(const EigenBase &matrix, + int options = ComputeEigenvectors) + : m_eivec(matrix.rows(), matrix.cols()), m_eivalues(matrix.cols()), + m_subdiag(matrix.rows() > 1 ? matrix.rows() - 1 : 1), m_isInitialized(false) + { + compute(matrix.derived(), options); + } - /** \brief Reports whether previous computation was successful. - * - * \returns \c Success if computation was succesful, \c NoConvergence otherwise. - */ - EIGEN_DEVICE_FUNC - ComputationInfo info() const - { - eigen_assert(m_isInitialized && "SelfAdjointEigenSolver is not initialized."); - return m_info; - } + /** \brief Computes eigendecomposition of given matrix. + * + * \param[in] matrix Selfadjoint matrix whose eigendecomposition is to + * be computed. Only the lower triangular part of the matrix is referenced. + * \param[in] options Can be #ComputeEigenvectors (default) or #EigenvaluesOnly. + * \returns Reference to \c *this + * + * This function computes the eigenvalues of \p matrix. The eigenvalues() + * function can be used to retrieve them. If \p options equals #ComputeEigenvectors, + * then the eigenvectors are also computed and can be retrieved by + * calling eigenvectors(). + * + * This implementation uses a symmetric QR algorithm. The matrix is first + * reduced to tridiagonal form using the Tridiagonalization class. The + * tridiagonal matrix is then brought to diagonal form with implicit + * symmetric QR steps with Wilkinson shift. Details can be found in + * Section 8.3 of Golub \& Van Loan, %Matrix Computations. + * + * The cost of the computation is about \f$ 9n^3 \f$ if the eigenvectors + * are required and \f$ 4n^3/3 \f$ if they are not required. + * + * This method reuses the memory in the SelfAdjointEigenSolver object that + * was allocated when the object was constructed, if the size of the + * matrix does not change. + * + * Example: \include SelfAdjointEigenSolver_compute_MatrixType.cpp + * Output: \verbinclude SelfAdjointEigenSolver_compute_MatrixType.out + * + * \sa SelfAdjointEigenSolver(const MatrixType&, int) + */ + template + EIGEN_DEVICE_FUNC SelfAdjointEigenSolver &compute(const EigenBase &matrix, + int options = ComputeEigenvectors); + + /** \brief Computes eigendecomposition of given matrix using a closed-form algorithm + * + * This is a variant of compute(const MatrixType&, int options) which + * directly solves the underlying polynomial equation. + * + * Currently only 2x2 and 3x3 matrices for which the sizes are known at compile time are supported (e.g., Matrix3d). + * + * This method is usually significantly faster than the QR iterative algorithm + * but it might also be less accurate. It is also worth noting that + * for 3x3 matrices it involves trigonometric operations which are + * not necessarily available for all scalar types. + * + * For the 3x3 case, we observed the following worst case relative error regarding the eigenvalues: + * - double: 1e-8 + * - float: 1e-3 + * + * \sa compute(const MatrixType&, int options) + */ + EIGEN_DEVICE_FUNC + SelfAdjointEigenSolver &computeDirect(const MatrixType &matrix, int options = ComputeEigenvectors); + + /** + *\brief Computes the eigen decomposition from a tridiagonal symmetric matrix + * + * \param[in] diag The vector containing the diagonal of the matrix. + * \param[in] subdiag The subdiagonal of the matrix. + * \param[in] options Can be #ComputeEigenvectors (default) or #EigenvaluesOnly. + * \returns Reference to \c *this + * + * This function assumes that the matrix has been reduced to tridiagonal form. + * + * \sa compute(const MatrixType&, int) for more information + */ + SelfAdjointEigenSolver &computeFromTridiagonal(const RealVectorType &diag, + const SubDiagonalType &subdiag, + int options = ComputeEigenvectors); + + /** \brief Returns the eigenvectors of given matrix. + * + * \returns A const reference to the matrix whose columns are the eigenvectors. + * + * \pre The eigenvectors have been computed before. + * + * Column \f$ k \f$ of the returned matrix is an eigenvector corresponding + * to eigenvalue number \f$ k \f$ as returned by eigenvalues(). The + * eigenvectors are normalized to have (Euclidean) norm equal to one. If + * this object was used to solve the eigenproblem for the selfadjoint + * matrix \f$ A \f$, then the matrix returned by this function is the + * matrix \f$ V \f$ in the eigendecomposition \f$ A = V D V^{-1} \f$. + * + * Example: \include SelfAdjointEigenSolver_eigenvectors.cpp + * Output: \verbinclude SelfAdjointEigenSolver_eigenvectors.out + * + * \sa eigenvalues() + */ + EIGEN_DEVICE_FUNC + const EigenvectorsType &eigenvectors() const + { + eigen_assert(m_isInitialized && "SelfAdjointEigenSolver is not initialized."); + eigen_assert(m_eigenvectorsOk && "The eigenvectors have not been computed together with the eigenvalues."); + return m_eivec; + } - /** \brief Maximum number of iterations. - * - * The algorithm terminates if it does not converge within m_maxIterations * n iterations, where n - * denotes the size of the matrix. This value is currently set to 30 (copied from LAPACK). - */ - static const int m_maxIterations = 30; + /** \brief Returns the eigenvalues of given matrix. + * + * \returns A const reference to the column vector containing the eigenvalues. + * + * \pre The eigenvalues have been computed before. + * + * The eigenvalues are repeated according to their algebraic multiplicity, + * so there are as many eigenvalues as rows in the matrix. The eigenvalues + * are sorted in increasing order. + * + * Example: \include SelfAdjointEigenSolver_eigenvalues.cpp + * Output: \verbinclude SelfAdjointEigenSolver_eigenvalues.out + * + * \sa eigenvectors(), MatrixBase::eigenvalues() + */ + EIGEN_DEVICE_FUNC + const RealVectorType &eigenvalues() const + { + eigen_assert(m_isInitialized && "SelfAdjointEigenSolver is not initialized."); + return m_eivalues; + } - protected: - static void check_template_parameters() - { - EIGEN_STATIC_ASSERT_NON_INTEGER(Scalar); - } - - EigenvectorsType m_eivec; - RealVectorType m_eivalues; - typename TridiagonalizationType::SubDiagonalType m_subdiag; - ComputationInfo m_info; - bool m_isInitialized; - bool m_eigenvectorsOk; + /** \brief Computes the positive-definite square root of the matrix. + * + * \returns the positive-definite square root of the matrix + * + * \pre The eigenvalues and eigenvectors of a positive-definite matrix + * have been computed before. + * + * The square root of a positive-definite matrix \f$ A \f$ is the + * positive-definite matrix whose square equals \f$ A \f$. This function + * uses the eigendecomposition \f$ A = V D V^{-1} \f$ to compute the + * square root as \f$ A^{1/2} = V D^{1/2} V^{-1} \f$. + * + * Example: \include SelfAdjointEigenSolver_operatorSqrt.cpp + * Output: \verbinclude SelfAdjointEigenSolver_operatorSqrt.out + * + * \sa operatorInverseSqrt(), MatrixFunctions Module + */ + EIGEN_DEVICE_FUNC + MatrixType operatorSqrt() const + { + eigen_assert(m_isInitialized && "SelfAdjointEigenSolver is not initialized."); + eigen_assert(m_eigenvectorsOk && "The eigenvectors have not been computed together with the eigenvalues."); + return m_eivec * m_eivalues.cwiseSqrt().asDiagonal() * m_eivec.adjoint(); + } + + /** \brief Computes the inverse square root of the matrix. + * + * \returns the inverse positive-definite square root of the matrix + * + * \pre The eigenvalues and eigenvectors of a positive-definite matrix + * have been computed before. + * + * This function uses the eigendecomposition \f$ A = V D V^{-1} \f$ to + * compute the inverse square root as \f$ V D^{-1/2} V^{-1} \f$. This is + * cheaper than first computing the square root with operatorSqrt() and + * then its inverse with MatrixBase::inverse(). + * + * Example: \include SelfAdjointEigenSolver_operatorInverseSqrt.cpp + * Output: \verbinclude SelfAdjointEigenSolver_operatorInverseSqrt.out + * + * \sa operatorSqrt(), MatrixBase::inverse(), MatrixFunctions Module + */ + EIGEN_DEVICE_FUNC + MatrixType operatorInverseSqrt() const + { + eigen_assert(m_isInitialized && "SelfAdjointEigenSolver is not initialized."); + eigen_assert(m_eigenvectorsOk && "The eigenvectors have not been computed together with the eigenvalues."); + return m_eivec * m_eivalues.cwiseInverse().cwiseSqrt().asDiagonal() * m_eivec.adjoint(); + } + + /** \brief Reports whether previous computation was successful. + * + * \returns \c Success if computation was succesful, \c NoConvergence otherwise. + */ + EIGEN_DEVICE_FUNC + ComputationInfo info() const + { + eigen_assert(m_isInitialized && "SelfAdjointEigenSolver is not initialized."); + return m_info; + } + + /** \brief Maximum number of iterations. + * + * The algorithm terminates if it does not converge within m_maxIterations * n iterations, where n + * denotes the size of the matrix. This value is currently set to 30 (copied from LAPACK). + */ + static const int m_maxIterations = 30; + +protected: + static void check_template_parameters() { EIGEN_STATIC_ASSERT_NON_INTEGER(Scalar); } + + EigenvectorsType m_eivec; + RealVectorType m_eivalues; + typename TridiagonalizationType::SubDiagonalType m_subdiag; + ComputationInfo m_info; + bool m_isInitialized; + bool m_eigenvectorsOk; }; namespace internal { -/** \internal - * - * \eigenvalues_module \ingroup Eigenvalues_Module - * - * Performs a QR step on a tridiagonal symmetric matrix represented as a - * pair of two vectors \a diag and \a subdiag. - * - * \param diag the diagonal part of the input selfadjoint tridiagonal matrix - * \param subdiag the sub-diagonal part of the input selfadjoint tridiagonal matrix - * \param start starting index of the submatrix to work on - * \param end last+1 index of the submatrix to work on - * \param matrixQ pointer to the column-major matrix holding the eigenvectors, can be 0 - * \param n size of the input matrix - * - * For compilation efficiency reasons, this procedure does not use eigen expression - * for its arguments. - * - * Implemented from Golub's "Matrix Computations", algorithm 8.3.2: - * "implicit symmetric QR step with Wilkinson shift" - */ -template -EIGEN_DEVICE_FUNC -static void tridiagonal_qr_step(RealScalar* diag, RealScalar* subdiag, Index start, Index end, Scalar* matrixQ, Index n); -} + /** \internal + * + * \eigenvalues_module \ingroup Eigenvalues_Module + * + * Performs a QR step on a tridiagonal symmetric matrix represented as a + * pair of two vectors \a diag and \a subdiag. + * + * \param diag the diagonal part of the input selfadjoint tridiagonal matrix + * \param subdiag the sub-diagonal part of the input selfadjoint tridiagonal matrix + * \param start starting index of the submatrix to work on + * \param end last+1 index of the submatrix to work on + * \param matrixQ pointer to the column-major matrix holding the eigenvectors, can be 0 + * \param n size of the input matrix + * + * For compilation efficiency reasons, this procedure does not use eigen expression + * for its arguments. + * + * Implemented from Golub's "Matrix Computations", algorithm 8.3.2: + * "implicit symmetric QR step with Wilkinson shift" + */ + template + EIGEN_DEVICE_FUNC static void + tridiagonal_qr_step(RealScalar *diag, RealScalar *subdiag, Index start, Index end, Scalar *matrixQ, Index n); +}// namespace internal template template -EIGEN_DEVICE_FUNC -SelfAdjointEigenSolver& SelfAdjointEigenSolver -::compute(const EigenBase& a_matrix, int options) +EIGEN_DEVICE_FUNC SelfAdjointEigenSolver & + SelfAdjointEigenSolver::compute(const EigenBase &a_matrix, int options) { check_template_parameters(); - + const InputType &matrix(a_matrix.derived()); - + using std::abs; eigen_assert(matrix.cols() == matrix.rows()); - eigen_assert((options&~(EigVecMask|GenEigMask))==0 - && (options&EigVecMask)!=EigVecMask - && "invalid option parameter"); - bool computeEigenvectors = (options&ComputeEigenvectors)==ComputeEigenvectors; + eigen_assert( + (options & ~(EigVecMask | GenEigMask)) == 0 && (options & EigVecMask) != EigVecMask && "invalid option parameter"); + bool computeEigenvectors = (options & ComputeEigenvectors) == ComputeEigenvectors; Index n = matrix.cols(); - m_eivalues.resize(n,1); + m_eivalues.resize(n, 1); - if(n==1) - { + if (n == 1) { m_eivec = matrix; - m_eivalues.coeffRef(0,0) = numext::real(m_eivec.coeff(0,0)); - if(computeEigenvectors) - m_eivec.setOnes(n,n); + m_eivalues.coeffRef(0, 0) = numext::real(m_eivec.coeff(0, 0)); + if (computeEigenvectors) m_eivec.setOnes(n, n); m_info = Success; m_isInitialized = true; m_eigenvectorsOk = computeEigenvectors; @@ -425,19 +413,19 @@ ::compute(const EigenBase& a_matrix, int options) } // declare some aliases - RealVectorType& diag = m_eivalues; - EigenvectorsType& mat = m_eivec; + RealVectorType &diag = m_eivalues; + EigenvectorsType &mat = m_eivec; // map the matrix coefficients to [-1:1] to avoid over- and underflow. mat = matrix.template triangularView(); RealScalar scale = mat.cwiseAbs().maxCoeff(); - if(scale==RealScalar(0)) scale = RealScalar(1); + if (scale == RealScalar(0)) scale = RealScalar(1); mat.template triangularView() /= scale; - m_subdiag.resize(n-1); + m_subdiag.resize(n - 1); internal::tridiagonalization_inplace(mat, diag, m_subdiag, computeEigenvectors); m_info = internal::computeFromTridiagonal_impl(diag, m_subdiag, m_maxIterations, computeEigenvectors, m_eivec); - + // scale back the eigen values m_eivalues *= scale; @@ -447,18 +435,17 @@ ::compute(const EigenBase& a_matrix, int options) } template -SelfAdjointEigenSolver& SelfAdjointEigenSolver -::computeFromTridiagonal(const RealVectorType& diag, const SubDiagonalType& subdiag , int options) +SelfAdjointEigenSolver &SelfAdjointEigenSolver::computeFromTridiagonal( + const RealVectorType &diag, + const SubDiagonalType &subdiag, + int options) { - //TODO : Add an option to scale the values beforehand - bool computeEigenvectors = (options&ComputeEigenvectors)==ComputeEigenvectors; + // TODO : Add an option to scale the values beforehand + bool computeEigenvectors = (options & ComputeEigenvectors) == ComputeEigenvectors; m_eivalues = diag; m_subdiag = subdiag; - if (computeEigenvectors) - { - m_eivec.setIdentity(diag.size(), diag.size()); - } + if (computeEigenvectors) { m_eivec.setIdentity(diag.size(), diag.size()); } m_info = internal::computeFromTridiagonal_impl(m_eivalues, m_subdiag, m_maxIterations, computeEigenvectors, m_eivec); m_isInitialized = true; @@ -467,404 +454,386 @@ ::computeFromTridiagonal(const RealVectorType& diag, const SubDiagonalType& subd } namespace internal { -/** - * \internal - * \brief Compute the eigendecomposition from a tridiagonal matrix - * - * \param[in,out] diag : On input, the diagonal of the matrix, on output the eigenvalues - * \param[in,out] subdiag : The subdiagonal part of the matrix (entries are modified during the decomposition) - * \param[in] maxIterations : the maximum number of iterations - * \param[in] computeEigenvectors : whether the eigenvectors have to be computed or not - * \param[out] eivec : The matrix to store the eigenvectors if computeEigenvectors==true. Must be allocated on input. - * \returns \c Success or \c NoConvergence - */ -template -ComputationInfo computeFromTridiagonal_impl(DiagType& diag, SubDiagType& subdiag, const Index maxIterations, bool computeEigenvectors, MatrixType& eivec) -{ - using std::abs; + /** + * \internal + * \brief Compute the eigendecomposition from a tridiagonal matrix + * + * \param[in,out] diag : On input, the diagonal of the matrix, on output the eigenvalues + * \param[in,out] subdiag : The subdiagonal part of the matrix (entries are modified during the decomposition) + * \param[in] maxIterations : the maximum number of iterations + * \param[in] computeEigenvectors : whether the eigenvectors have to be computed or not + * \param[out] eivec : The matrix to store the eigenvectors if computeEigenvectors==true. Must be allocated on input. + * \returns \c Success or \c NoConvergence + */ + template + ComputationInfo computeFromTridiagonal_impl(DiagType &diag, + SubDiagType &subdiag, + const Index maxIterations, + bool computeEigenvectors, + MatrixType &eivec) + { + using std::abs; - ComputationInfo info; - typedef typename MatrixType::Scalar Scalar; + ComputationInfo info; + typedef typename MatrixType::Scalar Scalar; - Index n = diag.size(); - Index end = n-1; - Index start = 0; - Index iter = 0; // total number of iterations - - typedef typename DiagType::RealScalar RealScalar; - const RealScalar considerAsZero = (std::numeric_limits::min)(); - const RealScalar precision = RealScalar(2)*NumTraits::epsilon(); - - while (end>0) - { - for (Index i = start; i0 && subdiag[end-1]==RealScalar(0)) - { - end--; - } - if (end<=0) - break; + typedef typename DiagType::RealScalar RealScalar; + const RealScalar considerAsZero = (std::numeric_limits::min)(); + const RealScalar precision = RealScalar(2) * NumTraits::epsilon(); - // if we spent too many iterations, we give up - iter++; - if(iter > maxIterations * n) break; + while (end > 0) { + for (Index i = start; i < end; ++i) + if (internal::isMuchSmallerThan(abs(subdiag[i]), (abs(diag[i]) + abs(diag[i + 1])), precision) + || abs(subdiag[i]) <= considerAsZero) + subdiag[i] = 0; - start = end - 1; - while (start>0 && subdiag[start-1]!=0) - start--; + // find the largest unreduced block + while (end > 0 && subdiag[end - 1] == RealScalar(0)) { end--; } + if (end <= 0) break; - internal::tridiagonal_qr_step(diag.data(), subdiag.data(), start, end, computeEigenvectors ? eivec.data() : (Scalar*)0, n); - } - if (iter <= maxIterations * n) - info = Success; - else - info = NoConvergence; - - // Sort eigenvalues and corresponding vectors. - // TODO make the sort optional ? - // TODO use a better sort algorithm !! - if (info == Success) - { - for (Index i = 0; i < n-1; ++i) - { - Index k; - diag.segment(i,n-i).minCoeff(&k); - if (k > 0) - { - std::swap(diag[i], diag[k+i]); - if(computeEigenvectors) - eivec.col(i).swap(eivec.col(k+i)); + // if we spent too many iterations, we give up + iter++; + if (iter > maxIterations * n) break; + + start = end - 1; + while (start > 0 && subdiag[start - 1] != 0) start--; + + internal::tridiagonal_qr_step( + diag.data(), subdiag.data(), start, end, computeEigenvectors ? eivec.data() : (Scalar *)0, n); + } + if (iter <= maxIterations * n) + info = Success; + else + info = NoConvergence; + + // Sort eigenvalues and corresponding vectors. + // TODO make the sort optional ? + // TODO use a better sort algorithm !! + if (info == Success) { + for (Index i = 0; i < n - 1; ++i) { + Index k; + diag.segment(i, n - i).minCoeff(&k); + if (k > 0) { + std::swap(diag[i], diag[k + i]); + if (computeEigenvectors) eivec.col(i).swap(eivec.col(k + i)); + } } } + return info; } - return info; -} - -template struct direct_selfadjoint_eigenvalues -{ - EIGEN_DEVICE_FUNC - static inline void run(SolverType& eig, const typename SolverType::MatrixType& A, int options) - { eig.compute(A,options); } -}; -template struct direct_selfadjoint_eigenvalues -{ - typedef typename SolverType::MatrixType MatrixType; - typedef typename SolverType::RealVectorType VectorType; - typedef typename SolverType::Scalar Scalar; - typedef typename SolverType::EigenvectorsType EigenvectorsType; - - - /** \internal - * Computes the roots of the characteristic polynomial of \a m. - * For numerical stability m.trace() should be near zero and to avoid over- or underflow m should be normalized. - */ - EIGEN_DEVICE_FUNC - static inline void computeRoots(const MatrixType& m, VectorType& roots) + template struct direct_selfadjoint_eigenvalues { - EIGEN_USING_STD_MATH(sqrt) - EIGEN_USING_STD_MATH(atan2) - EIGEN_USING_STD_MATH(cos) - EIGEN_USING_STD_MATH(sin) - const Scalar s_inv3 = Scalar(1)/Scalar(3); - const Scalar s_sqrt3 = sqrt(Scalar(3)); - - // The characteristic equation is x^3 - c2*x^2 + c1*x - c0 = 0. The - // eigenvalues are the roots to this equation, all guaranteed to be - // real-valued, because the matrix is symmetric. - Scalar c0 = m(0,0)*m(1,1)*m(2,2) + Scalar(2)*m(1,0)*m(2,0)*m(2,1) - m(0,0)*m(2,1)*m(2,1) - m(1,1)*m(2,0)*m(2,0) - m(2,2)*m(1,0)*m(1,0); - Scalar c1 = m(0,0)*m(1,1) - m(1,0)*m(1,0) + m(0,0)*m(2,2) - m(2,0)*m(2,0) + m(1,1)*m(2,2) - m(2,1)*m(2,1); - Scalar c2 = m(0,0) + m(1,1) + m(2,2); - - // Construct the parameters used in classifying the roots of the equation - // and in solving the equation for the roots in closed form. - Scalar c2_over_3 = c2*s_inv3; - Scalar a_over_3 = (c2*c2_over_3 - c1)*s_inv3; - a_over_3 = numext::maxi(a_over_3, Scalar(0)); - - Scalar half_b = Scalar(0.5)*(c0 + c2_over_3*(Scalar(2)*c2_over_3*c2_over_3 - c1)); - - Scalar q = a_over_3*a_over_3*a_over_3 - half_b*half_b; - q = numext::maxi(q, Scalar(0)); - - // Compute the eigenvalues by solving for the roots of the polynomial. - Scalar rho = sqrt(a_over_3); - Scalar theta = atan2(sqrt(q),half_b)*s_inv3; // since sqrt(q) > 0, atan2 is in [0, pi] and theta is in [0, pi/3] - Scalar cos_theta = cos(theta); - Scalar sin_theta = sin(theta); - // roots are already sorted, since cos is monotonically decreasing on [0, pi] - roots(0) = c2_over_3 - rho*(cos_theta + s_sqrt3*sin_theta); // == 2*rho*cos(theta+2pi/3) - roots(1) = c2_over_3 - rho*(cos_theta - s_sqrt3*sin_theta); // == 2*rho*cos(theta+ pi/3) - roots(2) = c2_over_3 + Scalar(2)*rho*cos_theta; - } + EIGEN_DEVICE_FUNC + static inline void run(SolverType &eig, const typename SolverType::MatrixType &A, int options) + { + eig.compute(A, options); + } + }; - EIGEN_DEVICE_FUNC - static inline bool extract_kernel(MatrixType& mat, Ref res, Ref representative) + template struct direct_selfadjoint_eigenvalues { - using std::abs; - Index i0; - // Find non-zero column i0 (by construction, there must exist a non zero coefficient on the diagonal): - mat.diagonal().cwiseAbs().maxCoeff(&i0); - // mat.col(i0) is a good candidate for an orthogonal vector to the current eigenvector, - // so let's save it: - representative = mat.col(i0); - Scalar n0, n1; - VectorType c0, c1; - n0 = (c0 = representative.cross(mat.col((i0+1)%3))).squaredNorm(); - n1 = (c1 = representative.cross(mat.col((i0+2)%3))).squaredNorm(); - if(n0>n1) res = c0/std::sqrt(n0); - else res = c1/std::sqrt(n1); - - return true; - } + typedef typename SolverType::MatrixType MatrixType; + typedef typename SolverType::RealVectorType VectorType; + typedef typename SolverType::Scalar Scalar; + typedef typename SolverType::EigenvectorsType EigenvectorsType; - EIGEN_DEVICE_FUNC - static inline void run(SolverType& solver, const MatrixType& mat, int options) - { - eigen_assert(mat.cols() == 3 && mat.cols() == mat.rows()); - eigen_assert((options&~(EigVecMask|GenEigMask))==0 - && (options&EigVecMask)!=EigVecMask - && "invalid option parameter"); - bool computeEigenvectors = (options&ComputeEigenvectors)==ComputeEigenvectors; - - EigenvectorsType& eivecs = solver.m_eivec; - VectorType& eivals = solver.m_eivalues; - - // Shift the matrix to the mean eigenvalue and map the matrix coefficients to [-1:1] to avoid over- and underflow. - Scalar shift = mat.trace() / Scalar(3); - // TODO Avoid this copy. Currently it is necessary to suppress bogus values when determining maxCoeff and for computing the eigenvectors later - MatrixType scaledMat = mat.template selfadjointView(); - scaledMat.diagonal().array() -= shift; - Scalar scale = scaledMat.cwiseAbs().maxCoeff(); - if(scale > 0) scaledMat /= scale; // TODO for scale==0 we could save the remaining operations - - // compute the eigenvalues - computeRoots(scaledMat,eivals); - - // compute the eigenvectors - if(computeEigenvectors) + + /** \internal + * Computes the roots of the characteristic polynomial of \a m. + * For numerical stability m.trace() should be near zero and to avoid over- or underflow m should be normalized. + */ + EIGEN_DEVICE_FUNC + static inline void computeRoots(const MatrixType &m, VectorType &roots) { - if((eivals(2)-eivals(0))<=Eigen::NumTraits::epsilon()) - { - // All three eigenvalues are numerically the same - eivecs.setIdentity(); - } + EIGEN_USING_STD_MATH(sqrt) + EIGEN_USING_STD_MATH(atan2) + EIGEN_USING_STD_MATH(cos) + EIGEN_USING_STD_MATH(sin) + const Scalar s_inv3 = Scalar(1) / Scalar(3); + const Scalar s_sqrt3 = sqrt(Scalar(3)); + + // The characteristic equation is x^3 - c2*x^2 + c1*x - c0 = 0. The + // eigenvalues are the roots to this equation, all guaranteed to be + // real-valued, because the matrix is symmetric. + Scalar c0 = m(0, 0) * m(1, 1) * m(2, 2) + Scalar(2) * m(1, 0) * m(2, 0) * m(2, 1) - m(0, 0) * m(2, 1) * m(2, 1) + - m(1, 1) * m(2, 0) * m(2, 0) - m(2, 2) * m(1, 0) * m(1, 0); + Scalar c1 = m(0, 0) * m(1, 1) - m(1, 0) * m(1, 0) + m(0, 0) * m(2, 2) - m(2, 0) * m(2, 0) + m(1, 1) * m(2, 2) + - m(2, 1) * m(2, 1); + Scalar c2 = m(0, 0) + m(1, 1) + m(2, 2); + + // Construct the parameters used in classifying the roots of the equation + // and in solving the equation for the roots in closed form. + Scalar c2_over_3 = c2 * s_inv3; + Scalar a_over_3 = (c2 * c2_over_3 - c1) * s_inv3; + a_over_3 = numext::maxi(a_over_3, Scalar(0)); + + Scalar half_b = Scalar(0.5) * (c0 + c2_over_3 * (Scalar(2) * c2_over_3 * c2_over_3 - c1)); + + Scalar q = a_over_3 * a_over_3 * a_over_3 - half_b * half_b; + q = numext::maxi(q, Scalar(0)); + + // Compute the eigenvalues by solving for the roots of the polynomial. + Scalar rho = sqrt(a_over_3); + Scalar theta = atan2(sqrt(q), half_b) * s_inv3;// since sqrt(q) > 0, atan2 is in [0, pi] and theta is in [0, pi/3] + Scalar cos_theta = cos(theta); + Scalar sin_theta = sin(theta); + // roots are already sorted, since cos is monotonically decreasing on [0, pi] + roots(0) = c2_over_3 - rho * (cos_theta + s_sqrt3 * sin_theta);// == 2*rho*cos(theta+2pi/3) + roots(1) = c2_over_3 - rho * (cos_theta - s_sqrt3 * sin_theta);// == 2*rho*cos(theta+ pi/3) + roots(2) = c2_over_3 + Scalar(2) * rho * cos_theta; + } + + EIGEN_DEVICE_FUNC + static inline bool extract_kernel(MatrixType &mat, Ref res, Ref representative) + { + using std::abs; + Index i0; + // Find non-zero column i0 (by construction, there must exist a non zero coefficient on the diagonal): + mat.diagonal().cwiseAbs().maxCoeff(&i0); + // mat.col(i0) is a good candidate for an orthogonal vector to the current eigenvector, + // so let's save it: + representative = mat.col(i0); + Scalar n0, n1; + VectorType c0, c1; + n0 = (c0 = representative.cross(mat.col((i0 + 1) % 3))).squaredNorm(); + n1 = (c1 = representative.cross(mat.col((i0 + 2) % 3))).squaredNorm(); + if (n0 > n1) + res = c0 / std::sqrt(n0); else - { - MatrixType tmp; - tmp = scaledMat; - - // Compute the eigenvector of the most distinct eigenvalue - Scalar d0 = eivals(2) - eivals(1); - Scalar d1 = eivals(1) - eivals(0); - Index k(0), l(2); - if(d0 > d1) - { - numext::swap(k,l); - d0 = d1; - } + res = c1 / std::sqrt(n1); - // Compute the eigenvector of index k - { - tmp.diagonal().array () -= eivals(k); - // By construction, 'tmp' is of rank 2, and its kernel corresponds to the respective eigenvector. - extract_kernel(tmp, eivecs.col(k), eivecs.col(l)); - } + return true; + } - // Compute eigenvector of index l - if(d0<=2*Eigen::NumTraits::epsilon()*d1) - { - // If d0 is too small, then the two other eigenvalues are numerically the same, - // and thus we only have to ortho-normalize the near orthogonal vector we saved above. - eivecs.col(l) -= eivecs.col(k).dot(eivecs.col(l))*eivecs.col(l); - eivecs.col(l).normalize(); - } - else - { + EIGEN_DEVICE_FUNC + static inline void run(SolverType &solver, const MatrixType &mat, int options) + { + eigen_assert(mat.cols() == 3 && mat.cols() == mat.rows()); + eigen_assert((options & ~(EigVecMask | GenEigMask)) == 0 && (options & EigVecMask) != EigVecMask + && "invalid option parameter"); + bool computeEigenvectors = (options & ComputeEigenvectors) == ComputeEigenvectors; + + EigenvectorsType &eivecs = solver.m_eivec; + VectorType &eivals = solver.m_eivalues; + + // Shift the matrix to the mean eigenvalue and map the matrix coefficients to [-1:1] to avoid over- and underflow. + Scalar shift = mat.trace() / Scalar(3); + // TODO Avoid this copy. Currently it is necessary to suppress bogus values when determining maxCoeff and for + // computing the eigenvectors later + MatrixType scaledMat = mat.template selfadjointView(); + scaledMat.diagonal().array() -= shift; + Scalar scale = scaledMat.cwiseAbs().maxCoeff(); + if (scale > 0) scaledMat /= scale;// TODO for scale==0 we could save the remaining operations + + // compute the eigenvalues + computeRoots(scaledMat, eivals); + + // compute the eigenvectors + if (computeEigenvectors) { + if ((eivals(2) - eivals(0)) <= Eigen::NumTraits::epsilon()) { + // All three eigenvalues are numerically the same + eivecs.setIdentity(); + } else { + MatrixType tmp; tmp = scaledMat; - tmp.diagonal().array () -= eivals(l); - VectorType dummy; - extract_kernel(tmp, eivecs.col(l), dummy); + // Compute the eigenvector of the most distinct eigenvalue + Scalar d0 = eivals(2) - eivals(1); + Scalar d1 = eivals(1) - eivals(0); + Index k(0), l(2); + if (d0 > d1) { + numext::swap(k, l); + d0 = d1; + } + + // Compute the eigenvector of index k + { + tmp.diagonal().array() -= eivals(k); + // By construction, 'tmp' is of rank 2, and its kernel corresponds to the respective eigenvector. + extract_kernel(tmp, eivecs.col(k), eivecs.col(l)); + } + + // Compute eigenvector of index l + if (d0 <= 2 * Eigen::NumTraits::epsilon() * d1) { + // If d0 is too small, then the two other eigenvalues are numerically the same, + // and thus we only have to ortho-normalize the near orthogonal vector we saved above. + eivecs.col(l) -= eivecs.col(k).dot(eivecs.col(l)) * eivecs.col(l); + eivecs.col(l).normalize(); + } else { + tmp = scaledMat; + tmp.diagonal().array() -= eivals(l); + + VectorType dummy; + extract_kernel(tmp, eivecs.col(l), dummy); + } + + // Compute last eigenvector from the other two + eivecs.col(1) = eivecs.col(2).cross(eivecs.col(0)).normalized(); } - - // Compute last eigenvector from the other two - eivecs.col(1) = eivecs.col(2).cross(eivecs.col(0)).normalized(); } - } - // Rescale back to the original size. - eivals *= scale; - eivals.array() += shift; - - solver.m_info = Success; - solver.m_isInitialized = true; - solver.m_eigenvectorsOk = computeEigenvectors; - } -}; + // Rescale back to the original size. + eivals *= scale; + eivals.array() += shift; -// 2x2 direct eigenvalues decomposition, code from Hauke Heibel -template -struct direct_selfadjoint_eigenvalues -{ - typedef typename SolverType::MatrixType MatrixType; - typedef typename SolverType::RealVectorType VectorType; - typedef typename SolverType::Scalar Scalar; - typedef typename SolverType::EigenvectorsType EigenvectorsType; - - EIGEN_DEVICE_FUNC - static inline void computeRoots(const MatrixType& m, VectorType& roots) - { - using std::sqrt; - const Scalar t0 = Scalar(0.5) * sqrt( numext::abs2(m(0,0)-m(1,1)) + Scalar(4)*numext::abs2(m(1,0))); - const Scalar t1 = Scalar(0.5) * (m(0,0) + m(1,1)); - roots(0) = t1 - t0; - roots(1) = t1 + t0; - } - - EIGEN_DEVICE_FUNC - static inline void run(SolverType& solver, const MatrixType& mat, int options) + solver.m_info = Success; + solver.m_isInitialized = true; + solver.m_eigenvectorsOk = computeEigenvectors; + } + }; + + // 2x2 direct eigenvalues decomposition, code from Hauke Heibel + template struct direct_selfadjoint_eigenvalues { - EIGEN_USING_STD_MATH(sqrt); - EIGEN_USING_STD_MATH(abs); - - eigen_assert(mat.cols() == 2 && mat.cols() == mat.rows()); - eigen_assert((options&~(EigVecMask|GenEigMask))==0 - && (options&EigVecMask)!=EigVecMask - && "invalid option parameter"); - bool computeEigenvectors = (options&ComputeEigenvectors)==ComputeEigenvectors; - - EigenvectorsType& eivecs = solver.m_eivec; - VectorType& eivals = solver.m_eivalues; - - // Shift the matrix to the mean eigenvalue and map the matrix coefficients to [-1:1] to avoid over- and underflow. - Scalar shift = mat.trace() / Scalar(2); - MatrixType scaledMat = mat; - scaledMat.coeffRef(0,1) = mat.coeff(1,0); - scaledMat.diagonal().array() -= shift; - Scalar scale = scaledMat.cwiseAbs().maxCoeff(); - if(scale > Scalar(0)) - scaledMat /= scale; - - // Compute the eigenvalues - computeRoots(scaledMat,eivals); - - // compute the eigen vectors - if(computeEigenvectors) + typedef typename SolverType::MatrixType MatrixType; + typedef typename SolverType::RealVectorType VectorType; + typedef typename SolverType::Scalar Scalar; + typedef typename SolverType::EigenvectorsType EigenvectorsType; + + EIGEN_DEVICE_FUNC + static inline void computeRoots(const MatrixType &m, VectorType &roots) { - if((eivals(1)-eivals(0))<=abs(eivals(1))*Eigen::NumTraits::epsilon()) - { - eivecs.setIdentity(); - } - else - { - scaledMat.diagonal().array () -= eivals(1); - Scalar a2 = numext::abs2(scaledMat(0,0)); - Scalar c2 = numext::abs2(scaledMat(1,1)); - Scalar b2 = numext::abs2(scaledMat(1,0)); - if(a2>c2) - { - eivecs.col(1) << -scaledMat(1,0), scaledMat(0,0); - eivecs.col(1) /= sqrt(a2+b2); - } - else - { - eivecs.col(1) << -scaledMat(1,1), scaledMat(1,0); - eivecs.col(1) /= sqrt(c2+b2); - } + using std::sqrt; + const Scalar t0 = Scalar(0.5) * sqrt(numext::abs2(m(0, 0) - m(1, 1)) + Scalar(4) * numext::abs2(m(1, 0))); + const Scalar t1 = Scalar(0.5) * (m(0, 0) + m(1, 1)); + roots(0) = t1 - t0; + roots(1) = t1 + t0; + } - eivecs.col(0) << eivecs.col(1).unitOrthogonal(); + EIGEN_DEVICE_FUNC + static inline void run(SolverType &solver, const MatrixType &mat, int options) + { + EIGEN_USING_STD_MATH(sqrt); + EIGEN_USING_STD_MATH(abs); + + eigen_assert(mat.cols() == 2 && mat.cols() == mat.rows()); + eigen_assert((options & ~(EigVecMask | GenEigMask)) == 0 && (options & EigVecMask) != EigVecMask + && "invalid option parameter"); + bool computeEigenvectors = (options & ComputeEigenvectors) == ComputeEigenvectors; + + EigenvectorsType &eivecs = solver.m_eivec; + VectorType &eivals = solver.m_eivalues; + + // Shift the matrix to the mean eigenvalue and map the matrix coefficients to [-1:1] to avoid over- and underflow. + Scalar shift = mat.trace() / Scalar(2); + MatrixType scaledMat = mat; + scaledMat.coeffRef(0, 1) = mat.coeff(1, 0); + scaledMat.diagonal().array() -= shift; + Scalar scale = scaledMat.cwiseAbs().maxCoeff(); + if (scale > Scalar(0)) scaledMat /= scale; + + // Compute the eigenvalues + computeRoots(scaledMat, eivals); + + // compute the eigen vectors + if (computeEigenvectors) { + if ((eivals(1) - eivals(0)) <= abs(eivals(1)) * Eigen::NumTraits::epsilon()) { + eivecs.setIdentity(); + } else { + scaledMat.diagonal().array() -= eivals(1); + Scalar a2 = numext::abs2(scaledMat(0, 0)); + Scalar c2 = numext::abs2(scaledMat(1, 1)); + Scalar b2 = numext::abs2(scaledMat(1, 0)); + if (a2 > c2) { + eivecs.col(1) << -scaledMat(1, 0), scaledMat(0, 0); + eivecs.col(1) /= sqrt(a2 + b2); + } else { + eivecs.col(1) << -scaledMat(1, 1), scaledMat(1, 0); + eivecs.col(1) /= sqrt(c2 + b2); + } + + eivecs.col(0) << eivecs.col(1).unitOrthogonal(); + } } - } - // Rescale back to the original size. - eivals *= scale; - eivals.array() += shift; + // Rescale back to the original size. + eivals *= scale; + eivals.array() += shift; - solver.m_info = Success; - solver.m_isInitialized = true; - solver.m_eigenvectorsOk = computeEigenvectors; - } -}; + solver.m_info = Success; + solver.m_isInitialized = true; + solver.m_eigenvectorsOk = computeEigenvectors; + } + }; -} +}// namespace internal template -EIGEN_DEVICE_FUNC -SelfAdjointEigenSolver& SelfAdjointEigenSolver -::computeDirect(const MatrixType& matrix, int options) +EIGEN_DEVICE_FUNC SelfAdjointEigenSolver & + SelfAdjointEigenSolver::computeDirect(const MatrixType &matrix, int options) { - internal::direct_selfadjoint_eigenvalues::IsComplex>::run(*this,matrix,options); + internal::direct_selfadjoint_eigenvalues::IsComplex>::run( + *this, matrix, options); return *this; } namespace internal { -template -EIGEN_DEVICE_FUNC -static void tridiagonal_qr_step(RealScalar* diag, RealScalar* subdiag, Index start, Index end, Scalar* matrixQ, Index n) -{ - using std::abs; - RealScalar td = (diag[end-1] - diag[end])*RealScalar(0.5); - RealScalar e = subdiag[end-1]; - // Note that thanks to scaling, e^2 or td^2 cannot overflow, however they can still - // underflow thus leading to inf/NaN values when using the following commented code: -// RealScalar e2 = numext::abs2(subdiag[end-1]); -// RealScalar mu = diag[end] - e2 / (td + (td>0 ? 1 : -1) * sqrt(td*td + e2)); - // This explain the following, somewhat more complicated, version: - RealScalar mu = diag[end]; - if(td==RealScalar(0)) - mu -= abs(e); - else - { - RealScalar e2 = numext::abs2(subdiag[end-1]); - RealScalar h = numext::hypot(td,e); - if(e2==RealScalar(0)) mu -= (e / (td + (td>RealScalar(0) ? RealScalar(1) : RealScalar(-1)))) * (e / h); - else mu -= e2 / (td + (td>RealScalar(0) ? h : -h)); - } - - RealScalar x = diag[start] - mu; - RealScalar z = subdiag[start]; - for (Index k = start; k < end; ++k) + template + EIGEN_DEVICE_FUNC static void + tridiagonal_qr_step(RealScalar *diag, RealScalar *subdiag, Index start, Index end, Scalar *matrixQ, Index n) { - JacobiRotation rot; - rot.makeGivens(x, z); + using std::abs; + RealScalar td = (diag[end - 1] - diag[end]) * RealScalar(0.5); + RealScalar e = subdiag[end - 1]; + // Note that thanks to scaling, e^2 or td^2 cannot overflow, however they can still + // underflow thus leading to inf/NaN values when using the following commented code: + // RealScalar e2 = numext::abs2(subdiag[end-1]); + // RealScalar mu = diag[end] - e2 / (td + (td>0 ? 1 : -1) * sqrt(td*td + e2)); + // This explain the following, somewhat more complicated, version: + RealScalar mu = diag[end]; + if (td == RealScalar(0)) + mu -= abs(e); + else { + RealScalar e2 = numext::abs2(subdiag[end - 1]); + RealScalar h = numext::hypot(td, e); + if (e2 == RealScalar(0)) + mu -= (e / (td + (td > RealScalar(0) ? RealScalar(1) : RealScalar(-1)))) * (e / h); + else + mu -= e2 / (td + (td > RealScalar(0) ? h : -h)); + } - // do T = G' T G - RealScalar sdk = rot.s() * diag[k] + rot.c() * subdiag[k]; - RealScalar dkp1 = rot.s() * subdiag[k] + rot.c() * diag[k+1]; + RealScalar x = diag[start] - mu; + RealScalar z = subdiag[start]; + for (Index k = start; k < end; ++k) { + JacobiRotation rot; + rot.makeGivens(x, z); - diag[k] = rot.c() * (rot.c() * diag[k] - rot.s() * subdiag[k]) - rot.s() * (rot.c() * subdiag[k] - rot.s() * diag[k+1]); - diag[k+1] = rot.s() * sdk + rot.c() * dkp1; - subdiag[k] = rot.c() * sdk - rot.s() * dkp1; - + // do T = G' T G + RealScalar sdk = rot.s() * diag[k] + rot.c() * subdiag[k]; + RealScalar dkp1 = rot.s() * subdiag[k] + rot.c() * diag[k + 1]; - if (k > start) - subdiag[k - 1] = rot.c() * subdiag[k-1] - rot.s() * z; + diag[k] = + rot.c() * (rot.c() * diag[k] - rot.s() * subdiag[k]) - rot.s() * (rot.c() * subdiag[k] - rot.s() * diag[k + 1]); + diag[k + 1] = rot.s() * sdk + rot.c() * dkp1; + subdiag[k] = rot.c() * sdk - rot.s() * dkp1; - x = subdiag[k]; - if (k < end - 1) - { - z = -rot.s() * subdiag[k+1]; - subdiag[k + 1] = rot.c() * subdiag[k+1]; - } - - // apply the givens rotation to the unit matrix Q = Q * G - if (matrixQ) - { - // FIXME if StorageOrder == RowMajor this operation is not very efficient - Map > q(matrixQ,n,n); - q.applyOnTheRight(k,k+1,rot); + if (k > start) subdiag[k - 1] = rot.c() * subdiag[k - 1] - rot.s() * z; + + x = subdiag[k]; + + if (k < end - 1) { + z = -rot.s() * subdiag[k + 1]; + subdiag[k + 1] = rot.c() * subdiag[k + 1]; + } + + // apply the givens rotation to the unit matrix Q = Q * G + if (matrixQ) { + // FIXME if StorageOrder == RowMajor this operation is not very efficient + Map> q(matrixQ, n, n); + q.applyOnTheRight(k, k + 1, rot); + } } } -} -} // end namespace internal +}// end namespace internal -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_SELFADJOINTEIGENSOLVER_H +#endif// EIGEN_SELFADJOINTEIGENSOLVER_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Eigenvalues/SelfAdjointEigenSolver_LAPACKE.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Eigenvalues/SelfAdjointEigenSolver_LAPACKE.h index b0c947dc..21b83af6 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Eigenvalues/SelfAdjointEigenSolver_LAPACKE.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Eigenvalues/SelfAdjointEigenSolver_LAPACKE.h @@ -33,55 +33,56 @@ #ifndef EIGEN_SAEIGENSOLVER_LAPACKE_H #define EIGEN_SAEIGENSOLVER_LAPACKE_H -namespace Eigen { +namespace Eigen { /** \internal Specialization for the data types supported by LAPACKe */ -#define EIGEN_LAPACKE_EIG_SELFADJ_2(EIGTYPE, LAPACKE_TYPE, LAPACKE_RTYPE, LAPACKE_NAME, EIGCOLROW ) \ -template<> template inline \ -SelfAdjointEigenSolver >& \ -SelfAdjointEigenSolver >::compute(const EigenBase& matrix, int options) \ -{ \ - eigen_assert(matrix.cols() == matrix.rows()); \ - eigen_assert((options&~(EigVecMask|GenEigMask))==0 \ - && (options&EigVecMask)!=EigVecMask \ - && "invalid option parameter"); \ - bool computeEigenvectors = (options&ComputeEigenvectors)==ComputeEigenvectors; \ - lapack_int n = internal::convert_index(matrix.cols()), lda, info; \ - m_eivalues.resize(n,1); \ - m_subdiag.resize(n-1); \ - m_eivec = matrix; \ -\ - if(n==1) \ - { \ - m_eivalues.coeffRef(0,0) = numext::real(m_eivec.coeff(0,0)); \ - if(computeEigenvectors) m_eivec.setOnes(n,n); \ - m_info = Success; \ - m_isInitialized = true; \ - m_eigenvectorsOk = computeEigenvectors; \ - return *this; \ - } \ -\ - lda = internal::convert_index(m_eivec.outerStride()); \ - char jobz, uplo='L'/*, range='A'*/; \ - jobz = computeEigenvectors ? 'V' : 'N'; \ -\ - info = LAPACKE_##LAPACKE_NAME( LAPACK_COL_MAJOR, jobz, uplo, n, (LAPACKE_TYPE*)m_eivec.data(), lda, (LAPACKE_RTYPE*)m_eivalues.data() ); \ - m_info = (info==0) ? Success : NoConvergence; \ - m_isInitialized = true; \ - m_eigenvectorsOk = computeEigenvectors; \ - return *this; \ -} +#define EIGEN_LAPACKE_EIG_SELFADJ_2(EIGTYPE, LAPACKE_TYPE, LAPACKE_RTYPE, LAPACKE_NAME, EIGCOLROW) \ + template<> \ + template \ + inline SelfAdjointEigenSolver> & \ + SelfAdjointEigenSolver>::compute( \ + const EigenBase &matrix, int options) \ + { \ + eigen_assert(matrix.cols() == matrix.rows()); \ + eigen_assert((options & ~(EigVecMask | GenEigMask)) == 0 && (options & EigVecMask) != EigVecMask \ + && "invalid option parameter"); \ + bool computeEigenvectors = (options & ComputeEigenvectors) == ComputeEigenvectors; \ + lapack_int n = internal::convert_index(matrix.cols()), lda, info; \ + m_eivalues.resize(n, 1); \ + m_subdiag.resize(n - 1); \ + m_eivec = matrix; \ + \ + if (n == 1) { \ + m_eivalues.coeffRef(0, 0) = numext::real(m_eivec.coeff(0, 0)); \ + if (computeEigenvectors) m_eivec.setOnes(n, n); \ + m_info = Success; \ + m_isInitialized = true; \ + m_eigenvectorsOk = computeEigenvectors; \ + return *this; \ + } \ + \ + lda = internal::convert_index(m_eivec.outerStride()); \ + char jobz, uplo = 'L' /*, range='A'*/; \ + jobz = computeEigenvectors ? 'V' : 'N'; \ + \ + info = LAPACKE_##LAPACKE_NAME( \ + LAPACK_COL_MAJOR, jobz, uplo, n, (LAPACKE_TYPE *)m_eivec.data(), lda, (LAPACKE_RTYPE *)m_eivalues.data()); \ + m_info = (info == 0) ? Success : NoConvergence; \ + m_isInitialized = true; \ + m_eigenvectorsOk = computeEigenvectors; \ + return *this; \ + } -#define EIGEN_LAPACKE_EIG_SELFADJ(EIGTYPE, LAPACKE_TYPE, LAPACKE_RTYPE, LAPACKE_NAME ) \ - EIGEN_LAPACKE_EIG_SELFADJ_2(EIGTYPE, LAPACKE_TYPE, LAPACKE_RTYPE, LAPACKE_NAME, ColMajor ) \ - EIGEN_LAPACKE_EIG_SELFADJ_2(EIGTYPE, LAPACKE_TYPE, LAPACKE_RTYPE, LAPACKE_NAME, RowMajor ) +#define EIGEN_LAPACKE_EIG_SELFADJ(EIGTYPE, LAPACKE_TYPE, LAPACKE_RTYPE, LAPACKE_NAME) \ + EIGEN_LAPACKE_EIG_SELFADJ_2(EIGTYPE, LAPACKE_TYPE, LAPACKE_RTYPE, LAPACKE_NAME, ColMajor) \ + EIGEN_LAPACKE_EIG_SELFADJ_2(EIGTYPE, LAPACKE_TYPE, LAPACKE_RTYPE, LAPACKE_NAME, RowMajor) -EIGEN_LAPACKE_EIG_SELFADJ(double, double, double, dsyev) -EIGEN_LAPACKE_EIG_SELFADJ(float, float, float, ssyev) +EIGEN_LAPACKE_EIG_SELFADJ(double, double, double, dsyev) +EIGEN_LAPACKE_EIG_SELFADJ(float, float, float, ssyev) EIGEN_LAPACKE_EIG_SELFADJ(dcomplex, lapack_complex_double, double, zheev) -EIGEN_LAPACKE_EIG_SELFADJ(scomplex, lapack_complex_float, float, cheev) +EIGEN_LAPACKE_EIG_SELFADJ(scomplex, lapack_complex_float, float, cheev) -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_SAEIGENSOLVER_H +#endif// EIGEN_SAEIGENSOLVER_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Eigenvalues/Tridiagonalization.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Eigenvalues/Tridiagonalization.h index 1d102c17..509a3e45 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Eigenvalues/Tridiagonalization.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Eigenvalues/Tridiagonalization.h @@ -11,308 +11,296 @@ #ifndef EIGEN_TRIDIAGONALIZATION_H #define EIGEN_TRIDIAGONALIZATION_H -namespace Eigen { +namespace Eigen { namespace internal { - -template struct TridiagonalizationMatrixTReturnType; -template -struct traits > - : public traits -{ - typedef typename MatrixType::PlainObject ReturnType; // FIXME shall it be a BandMatrix? - enum { Flags = 0 }; -}; -template -void tridiagonalization_inplace(MatrixType& matA, CoeffVectorType& hCoeffs); -} + template struct TridiagonalizationMatrixTReturnType; + template + struct traits> : public traits + { + typedef typename MatrixType::PlainObject ReturnType;// FIXME shall it be a BandMatrix? + enum { Flags = 0 }; + }; + + template + void tridiagonalization_inplace(MatrixType &matA, CoeffVectorType &hCoeffs); +}// namespace internal /** \eigenvalues_module \ingroup Eigenvalues_Module - * - * - * \class Tridiagonalization - * - * \brief Tridiagonal decomposition of a selfadjoint matrix - * - * \tparam _MatrixType the type of the matrix of which we are computing the - * tridiagonal decomposition; this is expected to be an instantiation of the - * Matrix class template. - * - * This class performs a tridiagonal decomposition of a selfadjoint matrix \f$ A \f$ such that: - * \f$ A = Q T Q^* \f$ where \f$ Q \f$ is unitary and \f$ T \f$ a real symmetric tridiagonal matrix. - * - * A tridiagonal matrix is a matrix which has nonzero elements only on the - * main diagonal and the first diagonal below and above it. The Hessenberg - * decomposition of a selfadjoint matrix is in fact a tridiagonal - * decomposition. This class is used in SelfAdjointEigenSolver to compute the - * eigenvalues and eigenvectors of a selfadjoint matrix. - * - * Call the function compute() to compute the tridiagonal decomposition of a - * given matrix. Alternatively, you can use the Tridiagonalization(const MatrixType&) - * constructor which computes the tridiagonal Schur decomposition at - * construction time. Once the decomposition is computed, you can use the - * matrixQ() and matrixT() functions to retrieve the matrices Q and T in the - * decomposition. - * - * The documentation of Tridiagonalization(const MatrixType&) contains an - * example of the typical use of this class. - * - * \sa class HessenbergDecomposition, class SelfAdjointEigenSolver - */ + * + * + * \class Tridiagonalization + * + * \brief Tridiagonal decomposition of a selfadjoint matrix + * + * \tparam _MatrixType the type of the matrix of which we are computing the + * tridiagonal decomposition; this is expected to be an instantiation of the + * Matrix class template. + * + * This class performs a tridiagonal decomposition of a selfadjoint matrix \f$ A \f$ such that: + * \f$ A = Q T Q^* \f$ where \f$ Q \f$ is unitary and \f$ T \f$ a real symmetric tridiagonal matrix. + * + * A tridiagonal matrix is a matrix which has nonzero elements only on the + * main diagonal and the first diagonal below and above it. The Hessenberg + * decomposition of a selfadjoint matrix is in fact a tridiagonal + * decomposition. This class is used in SelfAdjointEigenSolver to compute the + * eigenvalues and eigenvectors of a selfadjoint matrix. + * + * Call the function compute() to compute the tridiagonal decomposition of a + * given matrix. Alternatively, you can use the Tridiagonalization(const MatrixType&) + * constructor which computes the tridiagonal Schur decomposition at + * construction time. Once the decomposition is computed, you can use the + * matrixQ() and matrixT() functions to retrieve the matrices Q and T in the + * decomposition. + * + * The documentation of Tridiagonalization(const MatrixType&) contains an + * example of the typical use of this class. + * + * \sa class HessenbergDecomposition, class SelfAdjointEigenSolver + */ template class Tridiagonalization { - public: - - /** \brief Synonym for the template parameter \p _MatrixType. */ - typedef _MatrixType MatrixType; +public: + /** \brief Synonym for the template parameter \p _MatrixType. */ + typedef _MatrixType MatrixType; - typedef typename MatrixType::Scalar Scalar; - typedef typename NumTraits::Real RealScalar; - typedef Eigen::Index Index; ///< \deprecated since Eigen 3.3 - - enum { - Size = MatrixType::RowsAtCompileTime, - SizeMinusOne = Size == Dynamic ? Dynamic : (Size > 1 ? Size - 1 : 1), - Options = MatrixType::Options, - MaxSize = MatrixType::MaxRowsAtCompileTime, - MaxSizeMinusOne = MaxSize == Dynamic ? Dynamic : (MaxSize > 1 ? MaxSize - 1 : 1) - }; - - typedef Matrix CoeffVectorType; - typedef typename internal::plain_col_type::type DiagonalType; - typedef Matrix SubDiagonalType; - typedef typename internal::remove_all::type MatrixTypeRealView; - typedef internal::TridiagonalizationMatrixTReturnType MatrixTReturnType; - - typedef typename internal::conditional::IsComplex, - typename internal::add_const_on_value_type::RealReturnType>::type, - const Diagonal - >::type DiagonalReturnType; - - typedef typename internal::conditional::IsComplex, - typename internal::add_const_on_value_type::RealReturnType>::type, - const Diagonal - >::type SubDiagonalReturnType; - - /** \brief Return type of matrixQ() */ - typedef HouseholderSequence::type> HouseholderSequenceType; - - /** \brief Default constructor. - * - * \param [in] size Positive integer, size of the matrix whose tridiagonal - * decomposition will be computed. - * - * The default constructor is useful in cases in which the user intends to - * perform decompositions via compute(). The \p size parameter is only - * used as a hint. It is not an error to give a wrong \p size, but it may - * impair performance. - * - * \sa compute() for an example. - */ - explicit Tridiagonalization(Index size = Size==Dynamic ? 2 : Size) - : m_matrix(size,size), - m_hCoeffs(size > 1 ? size-1 : 1), - m_isInitialized(false) - {} - - /** \brief Constructor; computes tridiagonal decomposition of given matrix. - * - * \param[in] matrix Selfadjoint matrix whose tridiagonal decomposition - * is to be computed. - * - * This constructor calls compute() to compute the tridiagonal decomposition. - * - * Example: \include Tridiagonalization_Tridiagonalization_MatrixType.cpp - * Output: \verbinclude Tridiagonalization_Tridiagonalization_MatrixType.out - */ - template - explicit Tridiagonalization(const EigenBase& matrix) - : m_matrix(matrix.derived()), - m_hCoeffs(matrix.cols() > 1 ? matrix.cols()-1 : 1), - m_isInitialized(false) - { - internal::tridiagonalization_inplace(m_matrix, m_hCoeffs); - m_isInitialized = true; - } - - /** \brief Computes tridiagonal decomposition of given matrix. - * - * \param[in] matrix Selfadjoint matrix whose tridiagonal decomposition - * is to be computed. - * \returns Reference to \c *this - * - * The tridiagonal decomposition is computed by bringing the columns of - * the matrix successively in the required form using Householder - * reflections. The cost is \f$ 4n^3/3 \f$ flops, where \f$ n \f$ denotes - * the size of the given matrix. - * - * This method reuses of the allocated data in the Tridiagonalization - * object, if the size of the matrix does not change. - * - * Example: \include Tridiagonalization_compute.cpp - * Output: \verbinclude Tridiagonalization_compute.out - */ - template - Tridiagonalization& compute(const EigenBase& matrix) - { - m_matrix = matrix.derived(); - m_hCoeffs.resize(matrix.rows()-1, 1); - internal::tridiagonalization_inplace(m_matrix, m_hCoeffs); - m_isInitialized = true; - return *this; - } - - /** \brief Returns the Householder coefficients. - * - * \returns a const reference to the vector of Householder coefficients - * - * \pre Either the constructor Tridiagonalization(const MatrixType&) or - * the member function compute(const MatrixType&) has been called before - * to compute the tridiagonal decomposition of a matrix. - * - * The Householder coefficients allow the reconstruction of the matrix - * \f$ Q \f$ in the tridiagonal decomposition from the packed data. - * - * Example: \include Tridiagonalization_householderCoefficients.cpp - * Output: \verbinclude Tridiagonalization_householderCoefficients.out - * - * \sa packedMatrix(), \ref Householder_Module "Householder module" - */ - inline CoeffVectorType householderCoefficients() const - { - eigen_assert(m_isInitialized && "Tridiagonalization is not initialized."); - return m_hCoeffs; - } + typedef typename MatrixType::Scalar Scalar; + typedef typename NumTraits::Real RealScalar; + typedef Eigen::Index Index;///< \deprecated since Eigen 3.3 + + enum { + Size = MatrixType::RowsAtCompileTime, + SizeMinusOne = Size == Dynamic ? Dynamic : (Size > 1 ? Size - 1 : 1), + Options = MatrixType::Options, + MaxSize = MatrixType::MaxRowsAtCompileTime, + MaxSizeMinusOne = MaxSize == Dynamic ? Dynamic : (MaxSize > 1 ? MaxSize - 1 : 1) + }; + + typedef Matrix CoeffVectorType; + typedef typename internal::plain_col_type::type DiagonalType; + typedef Matrix SubDiagonalType; + typedef typename internal::remove_all::type MatrixTypeRealView; + typedef internal::TridiagonalizationMatrixTReturnType MatrixTReturnType; + + typedef typename internal::conditional::IsComplex, + typename internal::add_const_on_value_type::RealReturnType>::type, + const Diagonal>::type DiagonalReturnType; + + typedef typename internal::conditional::IsComplex, + typename internal::add_const_on_value_type::RealReturnType>::type, + const Diagonal>::type SubDiagonalReturnType; + + /** \brief Return type of matrixQ() */ + typedef HouseholderSequence::type> + HouseholderSequenceType; + + /** \brief Default constructor. + * + * \param [in] size Positive integer, size of the matrix whose tridiagonal + * decomposition will be computed. + * + * The default constructor is useful in cases in which the user intends to + * perform decompositions via compute(). The \p size parameter is only + * used as a hint. It is not an error to give a wrong \p size, but it may + * impair performance. + * + * \sa compute() for an example. + */ + explicit Tridiagonalization(Index size = Size == Dynamic ? 2 : Size) + : m_matrix(size, size), m_hCoeffs(size > 1 ? size - 1 : 1), m_isInitialized(false) + {} + + /** \brief Constructor; computes tridiagonal decomposition of given matrix. + * + * \param[in] matrix Selfadjoint matrix whose tridiagonal decomposition + * is to be computed. + * + * This constructor calls compute() to compute the tridiagonal decomposition. + * + * Example: \include Tridiagonalization_Tridiagonalization_MatrixType.cpp + * Output: \verbinclude Tridiagonalization_Tridiagonalization_MatrixType.out + */ + template + explicit Tridiagonalization(const EigenBase &matrix) + : m_matrix(matrix.derived()), m_hCoeffs(matrix.cols() > 1 ? matrix.cols() - 1 : 1), m_isInitialized(false) + { + internal::tridiagonalization_inplace(m_matrix, m_hCoeffs); + m_isInitialized = true; + } - /** \brief Returns the internal representation of the decomposition - * - * \returns a const reference to a matrix with the internal representation - * of the decomposition. - * - * \pre Either the constructor Tridiagonalization(const MatrixType&) or - * the member function compute(const MatrixType&) has been called before - * to compute the tridiagonal decomposition of a matrix. - * - * The returned matrix contains the following information: - * - the strict upper triangular part is equal to the input matrix A. - * - the diagonal and lower sub-diagonal represent the real tridiagonal - * symmetric matrix T. - * - the rest of the lower part contains the Householder vectors that, - * combined with Householder coefficients returned by - * householderCoefficients(), allows to reconstruct the matrix Q as - * \f$ Q = H_{N-1} \ldots H_1 H_0 \f$. - * Here, the matrices \f$ H_i \f$ are the Householder transformations - * \f$ H_i = (I - h_i v_i v_i^T) \f$ - * where \f$ h_i \f$ is the \f$ i \f$th Householder coefficient and - * \f$ v_i \f$ is the Householder vector defined by - * \f$ v_i = [ 0, \ldots, 0, 1, M(i+2,i), \ldots, M(N-1,i) ]^T \f$ - * with M the matrix returned by this function. - * - * See LAPACK for further details on this packed storage. - * - * Example: \include Tridiagonalization_packedMatrix.cpp - * Output: \verbinclude Tridiagonalization_packedMatrix.out - * - * \sa householderCoefficients() - */ - inline const MatrixType& packedMatrix() const - { - eigen_assert(m_isInitialized && "Tridiagonalization is not initialized."); - return m_matrix; - } + /** \brief Computes tridiagonal decomposition of given matrix. + * + * \param[in] matrix Selfadjoint matrix whose tridiagonal decomposition + * is to be computed. + * \returns Reference to \c *this + * + * The tridiagonal decomposition is computed by bringing the columns of + * the matrix successively in the required form using Householder + * reflections. The cost is \f$ 4n^3/3 \f$ flops, where \f$ n \f$ denotes + * the size of the given matrix. + * + * This method reuses of the allocated data in the Tridiagonalization + * object, if the size of the matrix does not change. + * + * Example: \include Tridiagonalization_compute.cpp + * Output: \verbinclude Tridiagonalization_compute.out + */ + template Tridiagonalization &compute(const EigenBase &matrix) + { + m_matrix = matrix.derived(); + m_hCoeffs.resize(matrix.rows() - 1, 1); + internal::tridiagonalization_inplace(m_matrix, m_hCoeffs); + m_isInitialized = true; + return *this; + } - /** \brief Returns the unitary matrix Q in the decomposition - * - * \returns object representing the matrix Q - * - * \pre Either the constructor Tridiagonalization(const MatrixType&) or - * the member function compute(const MatrixType&) has been called before - * to compute the tridiagonal decomposition of a matrix. - * - * This function returns a light-weight object of template class - * HouseholderSequence. You can either apply it directly to a matrix or - * you can convert it to a matrix of type #MatrixType. - * - * \sa Tridiagonalization(const MatrixType&) for an example, - * matrixT(), class HouseholderSequence - */ - HouseholderSequenceType matrixQ() const - { - eigen_assert(m_isInitialized && "Tridiagonalization is not initialized."); - return HouseholderSequenceType(m_matrix, m_hCoeffs.conjugate()) - .setLength(m_matrix.rows() - 1) - .setShift(1); - } + /** \brief Returns the Householder coefficients. + * + * \returns a const reference to the vector of Householder coefficients + * + * \pre Either the constructor Tridiagonalization(const MatrixType&) or + * the member function compute(const MatrixType&) has been called before + * to compute the tridiagonal decomposition of a matrix. + * + * The Householder coefficients allow the reconstruction of the matrix + * \f$ Q \f$ in the tridiagonal decomposition from the packed data. + * + * Example: \include Tridiagonalization_householderCoefficients.cpp + * Output: \verbinclude Tridiagonalization_householderCoefficients.out + * + * \sa packedMatrix(), \ref Householder_Module "Householder module" + */ + inline CoeffVectorType householderCoefficients() const + { + eigen_assert(m_isInitialized && "Tridiagonalization is not initialized."); + return m_hCoeffs; + } - /** \brief Returns an expression of the tridiagonal matrix T in the decomposition - * - * \returns expression object representing the matrix T - * - * \pre Either the constructor Tridiagonalization(const MatrixType&) or - * the member function compute(const MatrixType&) has been called before - * to compute the tridiagonal decomposition of a matrix. - * - * Currently, this function can be used to extract the matrix T from internal - * data and copy it to a dense matrix object. In most cases, it may be - * sufficient to directly use the packed matrix or the vector expressions - * returned by diagonal() and subDiagonal() instead of creating a new - * dense copy matrix with this function. - * - * \sa Tridiagonalization(const MatrixType&) for an example, - * matrixQ(), packedMatrix(), diagonal(), subDiagonal() - */ - MatrixTReturnType matrixT() const - { - eigen_assert(m_isInitialized && "Tridiagonalization is not initialized."); - return MatrixTReturnType(m_matrix.real()); - } + /** \brief Returns the internal representation of the decomposition + * + * \returns a const reference to a matrix with the internal representation + * of the decomposition. + * + * \pre Either the constructor Tridiagonalization(const MatrixType&) or + * the member function compute(const MatrixType&) has been called before + * to compute the tridiagonal decomposition of a matrix. + * + * The returned matrix contains the following information: + * - the strict upper triangular part is equal to the input matrix A. + * - the diagonal and lower sub-diagonal represent the real tridiagonal + * symmetric matrix T. + * - the rest of the lower part contains the Householder vectors that, + * combined with Householder coefficients returned by + * householderCoefficients(), allows to reconstruct the matrix Q as + * \f$ Q = H_{N-1} \ldots H_1 H_0 \f$. + * Here, the matrices \f$ H_i \f$ are the Householder transformations + * \f$ H_i = (I - h_i v_i v_i^T) \f$ + * where \f$ h_i \f$ is the \f$ i \f$th Householder coefficient and + * \f$ v_i \f$ is the Householder vector defined by + * \f$ v_i = [ 0, \ldots, 0, 1, M(i+2,i), \ldots, M(N-1,i) ]^T \f$ + * with M the matrix returned by this function. + * + * See LAPACK for further details on this packed storage. + * + * Example: \include Tridiagonalization_packedMatrix.cpp + * Output: \verbinclude Tridiagonalization_packedMatrix.out + * + * \sa householderCoefficients() + */ + inline const MatrixType &packedMatrix() const + { + eigen_assert(m_isInitialized && "Tridiagonalization is not initialized."); + return m_matrix; + } - /** \brief Returns the diagonal of the tridiagonal matrix T in the decomposition. - * - * \returns expression representing the diagonal of T - * - * \pre Either the constructor Tridiagonalization(const MatrixType&) or - * the member function compute(const MatrixType&) has been called before - * to compute the tridiagonal decomposition of a matrix. - * - * Example: \include Tridiagonalization_diagonal.cpp - * Output: \verbinclude Tridiagonalization_diagonal.out - * - * \sa matrixT(), subDiagonal() - */ - DiagonalReturnType diagonal() const; - - /** \brief Returns the subdiagonal of the tridiagonal matrix T in the decomposition. - * - * \returns expression representing the subdiagonal of T - * - * \pre Either the constructor Tridiagonalization(const MatrixType&) or - * the member function compute(const MatrixType&) has been called before - * to compute the tridiagonal decomposition of a matrix. - * - * \sa diagonal() for an example, matrixT() - */ - SubDiagonalReturnType subDiagonal() const; + /** \brief Returns the unitary matrix Q in the decomposition + * + * \returns object representing the matrix Q + * + * \pre Either the constructor Tridiagonalization(const MatrixType&) or + * the member function compute(const MatrixType&) has been called before + * to compute the tridiagonal decomposition of a matrix. + * + * This function returns a light-weight object of template class + * HouseholderSequence. You can either apply it directly to a matrix or + * you can convert it to a matrix of type #MatrixType. + * + * \sa Tridiagonalization(const MatrixType&) for an example, + * matrixT(), class HouseholderSequence + */ + HouseholderSequenceType matrixQ() const + { + eigen_assert(m_isInitialized && "Tridiagonalization is not initialized."); + return HouseholderSequenceType(m_matrix, m_hCoeffs.conjugate()).setLength(m_matrix.rows() - 1).setShift(1); + } - protected: + /** \brief Returns an expression of the tridiagonal matrix T in the decomposition + * + * \returns expression object representing the matrix T + * + * \pre Either the constructor Tridiagonalization(const MatrixType&) or + * the member function compute(const MatrixType&) has been called before + * to compute the tridiagonal decomposition of a matrix. + * + * Currently, this function can be used to extract the matrix T from internal + * data and copy it to a dense matrix object. In most cases, it may be + * sufficient to directly use the packed matrix or the vector expressions + * returned by diagonal() and subDiagonal() instead of creating a new + * dense copy matrix with this function. + * + * \sa Tridiagonalization(const MatrixType&) for an example, + * matrixQ(), packedMatrix(), diagonal(), subDiagonal() + */ + MatrixTReturnType matrixT() const + { + eigen_assert(m_isInitialized && "Tridiagonalization is not initialized."); + return MatrixTReturnType(m_matrix.real()); + } - MatrixType m_matrix; - CoeffVectorType m_hCoeffs; - bool m_isInitialized; + /** \brief Returns the diagonal of the tridiagonal matrix T in the decomposition. + * + * \returns expression representing the diagonal of T + * + * \pre Either the constructor Tridiagonalization(const MatrixType&) or + * the member function compute(const MatrixType&) has been called before + * to compute the tridiagonal decomposition of a matrix. + * + * Example: \include Tridiagonalization_diagonal.cpp + * Output: \verbinclude Tridiagonalization_diagonal.out + * + * \sa matrixT(), subDiagonal() + */ + DiagonalReturnType diagonal() const; + + /** \brief Returns the subdiagonal of the tridiagonal matrix T in the decomposition. + * + * \returns expression representing the subdiagonal of T + * + * \pre Either the constructor Tridiagonalization(const MatrixType&) or + * the member function compute(const MatrixType&) has been called before + * to compute the tridiagonal decomposition of a matrix. + * + * \sa diagonal() for an example, matrixT() + */ + SubDiagonalReturnType subDiagonal() const; + +protected: + MatrixType m_matrix; + CoeffVectorType m_hCoeffs; + bool m_isInitialized; }; template -typename Tridiagonalization::DiagonalReturnType -Tridiagonalization::diagonal() const +typename Tridiagonalization::DiagonalReturnType Tridiagonalization::diagonal() const { eigen_assert(m_isInitialized && "Tridiagonalization is not initialized."); return m_matrix.diagonal().real(); } template -typename Tridiagonalization::SubDiagonalReturnType -Tridiagonalization::subDiagonal() const +typename Tridiagonalization::SubDiagonalReturnType Tridiagonalization::subDiagonal() const { eigen_assert(m_isInitialized && "Tridiagonalization is not initialized."); return m_matrix.template diagonal<-1>().real(); @@ -320,221 +308,207 @@ Tridiagonalization::subDiagonal() const namespace internal { -/** \internal - * Performs a tridiagonal decomposition of the selfadjoint matrix \a matA in-place. - * - * \param[in,out] matA On input the selfadjoint matrix. Only the \b lower triangular part is referenced. - * On output, the strict upper part is left unchanged, and the lower triangular part - * represents the T and Q matrices in packed format has detailed below. - * \param[out] hCoeffs returned Householder coefficients (see below) - * - * On output, the tridiagonal selfadjoint matrix T is stored in the diagonal - * and lower sub-diagonal of the matrix \a matA. - * The unitary matrix Q is represented in a compact way as a product of - * Householder reflectors \f$ H_i \f$ such that: - * \f$ Q = H_{N-1} \ldots H_1 H_0 \f$. - * The Householder reflectors are defined as - * \f$ H_i = (I - h_i v_i v_i^T) \f$ - * where \f$ h_i = hCoeffs[i]\f$ is the \f$ i \f$th Householder coefficient and - * \f$ v_i \f$ is the Householder vector defined by - * \f$ v_i = [ 0, \ldots, 0, 1, matA(i+2,i), \ldots, matA(N-1,i) ]^T \f$. - * - * Implemented from Golub's "Matrix Computations", algorithm 8.3.1. - * - * \sa Tridiagonalization::packedMatrix() - */ -template -void tridiagonalization_inplace(MatrixType& matA, CoeffVectorType& hCoeffs) -{ - using numext::conj; - typedef typename MatrixType::Scalar Scalar; - typedef typename MatrixType::RealScalar RealScalar; - Index n = matA.rows(); - eigen_assert(n==matA.cols()); - eigen_assert(n==hCoeffs.size()+1 || n==1); - - for (Index i = 0; i + void tridiagonalization_inplace(MatrixType &matA, CoeffVectorType &hCoeffs) { - Index remainingSize = n-i-1; - RealScalar beta; - Scalar h; - matA.col(i).tail(remainingSize).makeHouseholderInPlace(h, beta); - - // Apply similarity transformation to remaining columns, - // i.e., A = H A H' where H = I - h v v' and v = matA.col(i).tail(n-i-1) - matA.col(i).coeffRef(i+1) = 1; - - hCoeffs.tail(n-i-1).noalias() = (matA.bottomRightCorner(remainingSize,remainingSize).template selfadjointView() - * (conj(h) * matA.col(i).tail(remainingSize))); - - hCoeffs.tail(n-i-1) += (conj(h)*RealScalar(-0.5)*(hCoeffs.tail(remainingSize).dot(matA.col(i).tail(remainingSize)))) * matA.col(i).tail(n-i-1); - - matA.bottomRightCorner(remainingSize, remainingSize).template selfadjointView() - .rankUpdate(matA.col(i).tail(remainingSize), hCoeffs.tail(remainingSize), Scalar(-1)); - - matA.col(i).coeffRef(i+1) = beta; - hCoeffs.coeffRef(i) = h; + using numext::conj; + typedef typename MatrixType::Scalar Scalar; + typedef typename MatrixType::RealScalar RealScalar; + Index n = matA.rows(); + eigen_assert(n == matA.cols()); + eigen_assert(n == hCoeffs.size() + 1 || n == 1); + + for (Index i = 0; i < n - 1; ++i) { + Index remainingSize = n - i - 1; + RealScalar beta; + Scalar h; + matA.col(i).tail(remainingSize).makeHouseholderInPlace(h, beta); + + // Apply similarity transformation to remaining columns, + // i.e., A = H A H' where H = I - h v v' and v = matA.col(i).tail(n-i-1) + matA.col(i).coeffRef(i + 1) = 1; + + hCoeffs.tail(n - i - 1).noalias() = + (matA.bottomRightCorner(remainingSize, remainingSize).template selfadjointView() + * (conj(h) * matA.col(i).tail(remainingSize))); + + hCoeffs.tail(n - i - 1) += + (conj(h) * RealScalar(-0.5) * (hCoeffs.tail(remainingSize).dot(matA.col(i).tail(remainingSize)))) + * matA.col(i).tail(n - i - 1); + + matA.bottomRightCorner(remainingSize, remainingSize) + .template selfadjointView() + .rankUpdate(matA.col(i).tail(remainingSize), hCoeffs.tail(remainingSize), Scalar(-1)); + + matA.col(i).coeffRef(i + 1) = beta; + hCoeffs.coeffRef(i) = h; + } } -} -// forward declaration, implementation at the end of this file -template::IsComplex> -struct tridiagonalization_inplace_selector; - -/** \brief Performs a full tridiagonalization in place - * - * \param[in,out] mat On input, the selfadjoint matrix whose tridiagonal - * decomposition is to be computed. Only the lower triangular part referenced. - * The rest is left unchanged. On output, the orthogonal matrix Q - * in the decomposition if \p extractQ is true. - * \param[out] diag The diagonal of the tridiagonal matrix T in the - * decomposition. - * \param[out] subdiag The subdiagonal of the tridiagonal matrix T in - * the decomposition. - * \param[in] extractQ If true, the orthogonal matrix Q in the - * decomposition is computed and stored in \p mat. - * - * Computes the tridiagonal decomposition of the selfadjoint matrix \p mat in place - * such that \f$ mat = Q T Q^* \f$ where \f$ Q \f$ is unitary and \f$ T \f$ a real - * symmetric tridiagonal matrix. - * - * The tridiagonal matrix T is passed to the output parameters \p diag and \p subdiag. If - * \p extractQ is true, then the orthogonal matrix Q is passed to \p mat. Otherwise the lower - * part of the matrix \p mat is destroyed. - * - * The vectors \p diag and \p subdiag are not resized. The function - * assumes that they are already of the correct size. The length of the - * vector \p diag should equal the number of rows in \p mat, and the - * length of the vector \p subdiag should be one left. - * - * This implementation contains an optimized path for 3-by-3 matrices - * which is especially useful for plane fitting. - * - * \note Currently, it requires two temporary vectors to hold the intermediate - * Householder coefficients, and to reconstruct the matrix Q from the Householder - * reflectors. - * - * Example (this uses the same matrix as the example in - * Tridiagonalization::Tridiagonalization(const MatrixType&)): - * \include Tridiagonalization_decomposeInPlace.cpp - * Output: \verbinclude Tridiagonalization_decomposeInPlace.out - * - * \sa class Tridiagonalization - */ -template -void tridiagonalization_inplace(MatrixType& mat, DiagonalType& diag, SubDiagonalType& subdiag, bool extractQ) -{ - eigen_assert(mat.cols()==mat.rows() && diag.size()==mat.rows() && subdiag.size()==mat.rows()-1); - tridiagonalization_inplace_selector::run(mat, diag, subdiag, extractQ); -} - -/** \internal - * General full tridiagonalization - */ -template -struct tridiagonalization_inplace_selector -{ - typedef typename Tridiagonalization::CoeffVectorType CoeffVectorType; - typedef typename Tridiagonalization::HouseholderSequenceType HouseholderSequenceType; - template - static void run(MatrixType& mat, DiagonalType& diag, SubDiagonalType& subdiag, bool extractQ) + // forward declaration, implementation at the end of this file + template::IsComplex> + struct tridiagonalization_inplace_selector; + + /** \brief Performs a full tridiagonalization in place + * + * \param[in,out] mat On input, the selfadjoint matrix whose tridiagonal + * decomposition is to be computed. Only the lower triangular part referenced. + * The rest is left unchanged. On output, the orthogonal matrix Q + * in the decomposition if \p extractQ is true. + * \param[out] diag The diagonal of the tridiagonal matrix T in the + * decomposition. + * \param[out] subdiag The subdiagonal of the tridiagonal matrix T in + * the decomposition. + * \param[in] extractQ If true, the orthogonal matrix Q in the + * decomposition is computed and stored in \p mat. + * + * Computes the tridiagonal decomposition of the selfadjoint matrix \p mat in place + * such that \f$ mat = Q T Q^* \f$ where \f$ Q \f$ is unitary and \f$ T \f$ a real + * symmetric tridiagonal matrix. + * + * The tridiagonal matrix T is passed to the output parameters \p diag and \p subdiag. If + * \p extractQ is true, then the orthogonal matrix Q is passed to \p mat. Otherwise the lower + * part of the matrix \p mat is destroyed. + * + * The vectors \p diag and \p subdiag are not resized. The function + * assumes that they are already of the correct size. The length of the + * vector \p diag should equal the number of rows in \p mat, and the + * length of the vector \p subdiag should be one left. + * + * This implementation contains an optimized path for 3-by-3 matrices + * which is especially useful for plane fitting. + * + * \note Currently, it requires two temporary vectors to hold the intermediate + * Householder coefficients, and to reconstruct the matrix Q from the Householder + * reflectors. + * + * Example (this uses the same matrix as the example in + * Tridiagonalization::Tridiagonalization(const MatrixType&)): + * \include Tridiagonalization_decomposeInPlace.cpp + * Output: \verbinclude Tridiagonalization_decomposeInPlace.out + * + * \sa class Tridiagonalization + */ + template + void tridiagonalization_inplace(MatrixType &mat, DiagonalType &diag, SubDiagonalType &subdiag, bool extractQ) { - CoeffVectorType hCoeffs(mat.cols()-1); - tridiagonalization_inplace(mat,hCoeffs); - diag = mat.diagonal().real(); - subdiag = mat.template diagonal<-1>().real(); - if(extractQ) - mat = HouseholderSequenceType(mat, hCoeffs.conjugate()) - .setLength(mat.rows() - 1) - .setShift(1); + eigen_assert(mat.cols() == mat.rows() && diag.size() == mat.rows() && subdiag.size() == mat.rows() - 1); + tridiagonalization_inplace_selector::run(mat, diag, subdiag, extractQ); } -}; - -/** \internal - * Specialization for 3x3 real matrices. - * Especially useful for plane fitting. - */ -template -struct tridiagonalization_inplace_selector -{ - typedef typename MatrixType::Scalar Scalar; - typedef typename MatrixType::RealScalar RealScalar; - template - static void run(MatrixType& mat, DiagonalType& diag, SubDiagonalType& subdiag, bool extractQ) + /** \internal + * General full tridiagonalization + */ + template struct tridiagonalization_inplace_selector { - using std::sqrt; - const RealScalar tol = (std::numeric_limits::min)(); - diag[0] = mat(0,0); - RealScalar v1norm2 = numext::abs2(mat(2,0)); - if(v1norm2 <= tol) + typedef typename Tridiagonalization::CoeffVectorType CoeffVectorType; + typedef typename Tridiagonalization::HouseholderSequenceType HouseholderSequenceType; + template + static void run(MatrixType &mat, DiagonalType &diag, SubDiagonalType &subdiag, bool extractQ) { - diag[1] = mat(1,1); - diag[2] = mat(2,2); - subdiag[0] = mat(1,0); - subdiag[1] = mat(2,1); - if (extractQ) - mat.setIdentity(); + CoeffVectorType hCoeffs(mat.cols() - 1); + tridiagonalization_inplace(mat, hCoeffs); + diag = mat.diagonal().real(); + subdiag = mat.template diagonal<-1>().real(); + if (extractQ) mat = HouseholderSequenceType(mat, hCoeffs.conjugate()).setLength(mat.rows() - 1).setShift(1); } - else + }; + + /** \internal + * Specialization for 3x3 real matrices. + * Especially useful for plane fitting. + */ + template struct tridiagonalization_inplace_selector + { + typedef typename MatrixType::Scalar Scalar; + typedef typename MatrixType::RealScalar RealScalar; + + template + static void run(MatrixType &mat, DiagonalType &diag, SubDiagonalType &subdiag, bool extractQ) { - RealScalar beta = sqrt(numext::abs2(mat(1,0)) + v1norm2); - RealScalar invBeta = RealScalar(1)/beta; - Scalar m01 = mat(1,0) * invBeta; - Scalar m02 = mat(2,0) * invBeta; - Scalar q = RealScalar(2)*m01*mat(2,1) + m02*(mat(2,2) - mat(1,1)); - diag[1] = mat(1,1) + m02*q; - diag[2] = mat(2,2) - m02*q; - subdiag[0] = beta; - subdiag[1] = mat(2,1) - m01 * q; - if (extractQ) - { - mat << 1, 0, 0, - 0, m01, m02, - 0, m02, -m01; + using std::sqrt; + const RealScalar tol = (std::numeric_limits::min)(); + diag[0] = mat(0, 0); + RealScalar v1norm2 = numext::abs2(mat(2, 0)); + if (v1norm2 <= tol) { + diag[1] = mat(1, 1); + diag[2] = mat(2, 2); + subdiag[0] = mat(1, 0); + subdiag[1] = mat(2, 1); + if (extractQ) mat.setIdentity(); + } else { + RealScalar beta = sqrt(numext::abs2(mat(1, 0)) + v1norm2); + RealScalar invBeta = RealScalar(1) / beta; + Scalar m01 = mat(1, 0) * invBeta; + Scalar m02 = mat(2, 0) * invBeta; + Scalar q = RealScalar(2) * m01 * mat(2, 1) + m02 * (mat(2, 2) - mat(1, 1)); + diag[1] = mat(1, 1) + m02 * q; + diag[2] = mat(2, 2) - m02 * q; + subdiag[0] = beta; + subdiag[1] = mat(2, 1) - m01 * q; + if (extractQ) { mat << 1, 0, 0, 0, m01, m02, 0, m02, -m01; } } } - } -}; - -/** \internal - * Trivial specialization for 1x1 matrices - */ -template -struct tridiagonalization_inplace_selector -{ - typedef typename MatrixType::Scalar Scalar; + }; - template - static void run(MatrixType& mat, DiagonalType& diag, SubDiagonalType&, bool extractQ) + /** \internal + * Trivial specialization for 1x1 matrices + */ + template struct tridiagonalization_inplace_selector { - diag(0,0) = numext::real(mat(0,0)); - if(extractQ) - mat(0,0) = Scalar(1); - } -}; + typedef typename MatrixType::Scalar Scalar; -/** \internal - * \eigenvalues_module \ingroup Eigenvalues_Module - * - * \brief Expression type for return value of Tridiagonalization::matrixT() - * - * \tparam MatrixType type of underlying dense matrix - */ -template struct TridiagonalizationMatrixTReturnType -: public ReturnByValue > -{ + template + static void run(MatrixType &mat, DiagonalType &diag, SubDiagonalType &, bool extractQ) + { + diag(0, 0) = numext::real(mat(0, 0)); + if (extractQ) mat(0, 0) = Scalar(1); + } + }; + + /** \internal + * \eigenvalues_module \ingroup Eigenvalues_Module + * + * \brief Expression type for return value of Tridiagonalization::matrixT() + * + * \tparam MatrixType type of underlying dense matrix + */ + template + struct TridiagonalizationMatrixTReturnType : public ReturnByValue> + { public: /** \brief Constructor. - * - * \param[in] mat The underlying dense matrix - */ - TridiagonalizationMatrixTReturnType(const MatrixType& mat) : m_matrix(mat) { } + * + * \param[in] mat The underlying dense matrix + */ + TridiagonalizationMatrixTReturnType(const MatrixType &mat) : m_matrix(mat) {} - template - inline void evalTo(ResultType& result) const + template inline void evalTo(ResultType &result) const { result.setZero(); result.template diagonal<1>() = m_matrix.template diagonal<-1>().conjugate(); @@ -547,10 +521,10 @@ template struct TridiagonalizationMatrixTReturnType protected: typename MatrixType::Nested m_matrix; -}; + }; -} // end namespace internal +}// end namespace internal -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_TRIDIAGONALIZATION_H +#endif// EIGEN_TRIDIAGONALIZATION_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Geometry/AlignedBox.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Geometry/AlignedBox.h index 066eae4f..49bd7e75 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Geometry/AlignedBox.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Geometry/AlignedBox.h @@ -10,79 +10,92 @@ #ifndef EIGEN_ALIGNEDBOX_H #define EIGEN_ALIGNEDBOX_H -namespace Eigen { +namespace Eigen { /** \geometry_module \ingroup Geometry_Module - * - * - * \class AlignedBox - * - * \brief An axis aligned box - * - * \tparam _Scalar the type of the scalar coefficients - * \tparam _AmbientDim the dimension of the ambient space, can be a compile time value or Dynamic. - * - * This class represents an axis aligned box as a pair of the minimal and maximal corners. - * \warning The result of most methods is undefined when applied to an empty box. You can check for empty boxes using isEmpty(). - * \sa alignedboxtypedefs - */ -template -class AlignedBox + * + * + * \class AlignedBox + * + * \brief An axis aligned box + * + * \tparam _Scalar the type of the scalar coefficients + * \tparam _AmbientDim the dimension of the ambient space, can be a compile time value or Dynamic. + * + * This class represents an axis aligned box as a pair of the minimal and maximal corners. + * \warning The result of most methods is undefined when applied to an empty box. You can check for empty boxes using + * isEmpty(). + * \sa alignedboxtypedefs + */ +template class AlignedBox { public: -EIGEN_MAKE_ALIGNED_OPERATOR_NEW_IF_VECTORIZABLE_FIXED_SIZE(_Scalar,_AmbientDim) + EIGEN_MAKE_ALIGNED_OPERATOR_NEW_IF_VECTORIZABLE_FIXED_SIZE(_Scalar, _AmbientDim) enum { AmbientDimAtCompileTime = _AmbientDim }; - typedef _Scalar Scalar; - typedef NumTraits ScalarTraits; - typedef Eigen::Index Index; ///< \deprecated since Eigen 3.3 - typedef typename ScalarTraits::Real RealScalar; - typedef typename ScalarTraits::NonInteger NonInteger; - typedef Matrix VectorType; + typedef _Scalar Scalar; + typedef NumTraits ScalarTraits; + typedef Eigen::Index Index;///< \deprecated since Eigen 3.3 + typedef typename ScalarTraits::Real RealScalar; + typedef typename ScalarTraits::NonInteger NonInteger; + typedef Matrix VectorType; typedef CwiseBinaryOp, const VectorType, const VectorType> VectorTypeSum; /** Define constants to name the corners of a 1D, 2D or 3D axis aligned bounding box */ - enum CornerType - { + enum CornerType { /** 1D names @{ */ - Min=0, Max=1, + Min = 0, + Max = 1, /** @} */ /** Identifier for 2D corner @{ */ - BottomLeft=0, BottomRight=1, - TopLeft=2, TopRight=3, + BottomLeft = 0, + BottomRight = 1, + TopLeft = 2, + TopRight = 3, /** @} */ /** Identifier for 3D corner @{ */ - BottomLeftFloor=0, BottomRightFloor=1, - TopLeftFloor=2, TopRightFloor=3, - BottomLeftCeil=4, BottomRightCeil=5, - TopLeftCeil=6, TopRightCeil=7 + BottomLeftFloor = 0, + BottomRightFloor = 1, + TopLeftFloor = 2, + TopRightFloor = 3, + BottomLeftCeil = 4, + BottomRightCeil = 5, + TopLeftCeil = 6, + TopRightCeil = 7 /** @} */ }; /** Default constructor initializing a null box. */ EIGEN_DEVICE_FUNC inline AlignedBox() - { if (AmbientDimAtCompileTime!=Dynamic) setEmpty(); } + { + if (AmbientDimAtCompileTime != Dynamic) setEmpty(); + } /** Constructs a null box with \a _dim the dimension of the ambient space. */ - EIGEN_DEVICE_FUNC inline explicit AlignedBox(Index _dim) : m_min(_dim), m_max(_dim) - { setEmpty(); } + EIGEN_DEVICE_FUNC inline explicit AlignedBox(Index _dim) : m_min(_dim), m_max(_dim) { setEmpty(); } /** Constructs a box with extremities \a _min and \a _max. - * \warning If either component of \a _min is larger than the same component of \a _max, the constructed box is empty. */ + * \warning If either component of \a _min is larger than the same component of \a _max, the constructed box is empty. + */ template - EIGEN_DEVICE_FUNC inline AlignedBox(const OtherVectorType1& _min, const OtherVectorType2& _max) : m_min(_min), m_max(_max) {} + EIGEN_DEVICE_FUNC inline AlignedBox(const OtherVectorType1 &_min, const OtherVectorType2 &_max) + : m_min(_min), m_max(_max) + {} /** Constructs a box containing a single point \a p. */ template - EIGEN_DEVICE_FUNC inline explicit AlignedBox(const MatrixBase& p) : m_min(p), m_max(m_min) - { } + EIGEN_DEVICE_FUNC inline explicit AlignedBox(const MatrixBase &p) : m_min(p), m_max(m_min) + {} EIGEN_DEVICE_FUNC ~AlignedBox() {} /** \returns the dimension in which the box holds */ - EIGEN_DEVICE_FUNC inline Index dim() const { return AmbientDimAtCompileTime==Dynamic ? m_min.size() : Index(AmbientDimAtCompileTime); } + EIGEN_DEVICE_FUNC inline Index dim() const + { + return AmbientDimAtCompileTime == Dynamic ? m_min.size() : Index(AmbientDimAtCompileTime); + } /** \deprecated use isEmpty() */ EIGEN_DEVICE_FUNC inline bool isNull() const { return isEmpty(); } @@ -98,51 +111,62 @@ EIGEN_MAKE_ALIGNED_OPERATOR_NEW_IF_VECTORIZABLE_FIXED_SIZE(_Scalar,_AmbientDim) * \sa isEmpty */ EIGEN_DEVICE_FUNC inline void setEmpty() { - m_min.setConstant( ScalarTraits::highest() ); - m_max.setConstant( ScalarTraits::lowest() ); + m_min.setConstant(ScalarTraits::highest()); + m_max.setConstant(ScalarTraits::lowest()); } /** \returns the minimal corner */ - EIGEN_DEVICE_FUNC inline const VectorType& (min)() const { return m_min; } + EIGEN_DEVICE_FUNC inline const VectorType &(min)() const { return m_min; } /** \returns a non const reference to the minimal corner */ - EIGEN_DEVICE_FUNC inline VectorType& (min)() { return m_min; } + EIGEN_DEVICE_FUNC inline VectorType &(min)() { return m_min; } /** \returns the maximal corner */ - EIGEN_DEVICE_FUNC inline const VectorType& (max)() const { return m_max; } + EIGEN_DEVICE_FUNC inline const VectorType &(max)() const { return m_max; } /** \returns a non const reference to the maximal corner */ - EIGEN_DEVICE_FUNC inline VectorType& (max)() { return m_max; } + EIGEN_DEVICE_FUNC inline VectorType &(max)() { return m_max; } /** \returns the center of the box */ EIGEN_DEVICE_FUNC inline const EIGEN_EXPR_BINARYOP_SCALAR_RETURN_TYPE(VectorTypeSum, RealScalar, quotient) - center() const - { return (m_min+m_max)/RealScalar(2); } + center() const + { + return (m_min + m_max) / RealScalar(2); + } /** \returns the lengths of the sides of the bounding box. - * Note that this function does not get the same - * result for integral or floating scalar types: see - */ - EIGEN_DEVICE_FUNC inline const CwiseBinaryOp< internal::scalar_difference_op, const VectorType, const VectorType> sizes() const - { return m_max - m_min; } + * Note that this function does not get the same + * result for integral or floating scalar types: see + */ + EIGEN_DEVICE_FUNC inline const CwiseBinaryOp, + const VectorType, + const VectorType> + sizes() const + { + return m_max - m_min; + } /** \returns the volume of the bounding box */ - EIGEN_DEVICE_FUNC inline Scalar volume() const - { return sizes().prod(); } + EIGEN_DEVICE_FUNC inline Scalar volume() const { return sizes().prod(); } /** \returns an expression for the bounding box diagonal vector - * if the length of the diagonal is needed: diagonal().norm() - * will provide it. - */ - EIGEN_DEVICE_FUNC inline CwiseBinaryOp< internal::scalar_difference_op, const VectorType, const VectorType> diagonal() const - { return sizes(); } + * if the length of the diagonal is needed: diagonal().norm() + * will provide it. + */ + EIGEN_DEVICE_FUNC inline CwiseBinaryOp, + const VectorType, + const VectorType> + diagonal() const + { + return sizes(); + } /** \returns the vertex of the bounding box at the corner defined by - * the corner-id corner. It works only for a 1D, 2D or 3D bounding box. - * For 1D bounding boxes corners are named by 2 enum constants: - * BottomLeft and BottomRight. - * For 2D bounding boxes, corners are named by 4 enum constants: - * BottomLeft, BottomRight, TopLeft, TopRight. - * For 3D bounding boxes, the following names are added: - * BottomLeftCeil, BottomRightCeil, TopLeftCeil, TopRightCeil. - */ + * the corner-id corner. It works only for a 1D, 2D or 3D bounding box. + * For 1D bounding boxes corners are named by 2 enum constants: + * BottomLeft and BottomRight. + * For 2D bounding boxes, corners are named by 4 enum constants: + * BottomLeft, BottomRight, TopLeft, TopRight. + * For 3D bounding boxes, the following names are added: + * BottomLeftCeil, BottomRightCeil, TopLeftCeil, TopRightCeil. + */ EIGEN_DEVICE_FUNC inline VectorType corner(CornerType corner) const { EIGEN_STATIC_ASSERT(_AmbientDim <= 3, THIS_METHOD_IS_ONLY_FOR_VECTORS_OF_A_SPECIFIC_SIZE); @@ -150,10 +174,11 @@ EIGEN_MAKE_ALIGNED_OPERATOR_NEW_IF_VECTORIZABLE_FIXED_SIZE(_Scalar,_AmbientDim) VectorType res; Index mult = 1; - for(Index d=0; d(Scalar(0), Scalar(1)); - } - else + for (Index d = 0; d < dim(); ++d) { + if (!ScalarTraits::IsInteger) { + r[d] = m_min[d] + (m_max[d] - m_min[d]) * internal::random(Scalar(0), Scalar(1)); + } else r[d] = internal::random(m_min[d], m_max[d]); } return r; } /** \returns true if the point \a p is inside the box \c *this. */ - template - EIGEN_DEVICE_FUNC inline bool contains(const MatrixBase& p) const + template EIGEN_DEVICE_FUNC inline bool contains(const MatrixBase &p) const { - typename internal::nested_eval::type p_n(p.derived()); - return (m_min.array()<=p_n.array()).all() && (p_n.array()<=m_max.array()).all(); + typename internal::nested_eval::type p_n(p.derived()); + return (m_min.array() <= p_n.array()).all() && (p_n.array() <= m_max.array()).all(); } /** \returns true if the box \a b is entirely inside the box \c *this. */ - EIGEN_DEVICE_FUNC inline bool contains(const AlignedBox& b) const - { return (m_min.array()<=(b.min)().array()).all() && ((b.max)().array()<=m_max.array()).all(); } + EIGEN_DEVICE_FUNC inline bool contains(const AlignedBox &b) const + { + return (m_min.array() <= (b.min)().array()).all() && ((b.max)().array() <= m_max.array()).all(); + } /** \returns true if the box \a b is intersecting the box \c *this. * \sa intersection, clamp */ - EIGEN_DEVICE_FUNC inline bool intersects(const AlignedBox& b) const - { return (m_min.array()<=(b.max)().array()).all() && ((b.min)().array()<=m_max.array()).all(); } + EIGEN_DEVICE_FUNC inline bool intersects(const AlignedBox &b) const + { + return (m_min.array() <= (b.max)().array()).all() && ((b.min)().array() <= m_max.array()).all(); + } /** Extends \c *this such that it contains the point \a p and returns a reference to \c *this. * \sa extend(const AlignedBox&) */ - template - EIGEN_DEVICE_FUNC inline AlignedBox& extend(const MatrixBase& p) + template EIGEN_DEVICE_FUNC inline AlignedBox &extend(const MatrixBase &p) { - typename internal::nested_eval::type p_n(p.derived()); + typename internal::nested_eval::type p_n(p.derived()); m_min = m_min.cwiseMin(p_n); m_max = m_max.cwiseMax(p_n); return *this; @@ -207,7 +230,7 @@ EIGEN_MAKE_ALIGNED_OPERATOR_NEW_IF_VECTORIZABLE_FIXED_SIZE(_Scalar,_AmbientDim) /** Extends \c *this such that it contains the box \a b and returns a reference to \c *this. * \sa merged, extend(const MatrixBase&) */ - EIGEN_DEVICE_FUNC inline AlignedBox& extend(const AlignedBox& b) + EIGEN_DEVICE_FUNC inline AlignedBox &extend(const AlignedBox &b) { m_min = m_min.cwiseMin(b.m_min); m_max = m_max.cwiseMax(b.m_max); @@ -217,7 +240,7 @@ EIGEN_MAKE_ALIGNED_OPERATOR_NEW_IF_VECTORIZABLE_FIXED_SIZE(_Scalar,_AmbientDim) /** Clamps \c *this by the box \a b and returns a reference to \c *this. * \note If the boxes don't intersect, the resulting box is empty. * \sa intersection(), intersects() */ - EIGEN_DEVICE_FUNC inline AlignedBox& clamp(const AlignedBox& b) + EIGEN_DEVICE_FUNC inline AlignedBox &clamp(const AlignedBox &b) { m_min = m_min.cwiseMax(b.m_min); m_max = m_max.cwiseMin(b.m_max); @@ -227,166 +250,168 @@ EIGEN_MAKE_ALIGNED_OPERATOR_NEW_IF_VECTORIZABLE_FIXED_SIZE(_Scalar,_AmbientDim) /** Returns an AlignedBox that is the intersection of \a b and \c *this * \note If the boxes don't intersect, the resulting box is empty. * \sa intersects(), clamp, contains() */ - EIGEN_DEVICE_FUNC inline AlignedBox intersection(const AlignedBox& b) const - {return AlignedBox(m_min.cwiseMax(b.m_min), m_max.cwiseMin(b.m_max)); } + EIGEN_DEVICE_FUNC inline AlignedBox intersection(const AlignedBox &b) const + { + return AlignedBox(m_min.cwiseMax(b.m_min), m_max.cwiseMin(b.m_max)); + } /** Returns an AlignedBox that is the union of \a b and \c *this. - * \note Merging with an empty box may result in a box bigger than \c *this. + * \note Merging with an empty box may result in a box bigger than \c *this. * \sa extend(const AlignedBox&) */ - EIGEN_DEVICE_FUNC inline AlignedBox merged(const AlignedBox& b) const - { return AlignedBox(m_min.cwiseMin(b.m_min), m_max.cwiseMax(b.m_max)); } + EIGEN_DEVICE_FUNC inline AlignedBox merged(const AlignedBox &b) const + { + return AlignedBox(m_min.cwiseMin(b.m_min), m_max.cwiseMax(b.m_max)); + } /** Translate \c *this by the vector \a t and returns a reference to \c *this. */ - template - EIGEN_DEVICE_FUNC inline AlignedBox& translate(const MatrixBase& a_t) + template EIGEN_DEVICE_FUNC inline AlignedBox &translate(const MatrixBase &a_t) { - const typename internal::nested_eval::type t(a_t.derived()); + const typename internal::nested_eval::type t(a_t.derived()); m_min += t; m_max += t; return *this; } /** \returns the squared distance between the point \a p and the box \c *this, - * and zero if \a p is inside the box. - * \sa exteriorDistance(const MatrixBase&), squaredExteriorDistance(const AlignedBox&) - */ + * and zero if \a p is inside the box. + * \sa exteriorDistance(const MatrixBase&), squaredExteriorDistance(const AlignedBox&) + */ template - EIGEN_DEVICE_FUNC inline Scalar squaredExteriorDistance(const MatrixBase& p) const; + EIGEN_DEVICE_FUNC inline Scalar squaredExteriorDistance(const MatrixBase &p) const; /** \returns the squared distance between the boxes \a b and \c *this, - * and zero if the boxes intersect. - * \sa exteriorDistance(const AlignedBox&), squaredExteriorDistance(const MatrixBase&) - */ - EIGEN_DEVICE_FUNC inline Scalar squaredExteriorDistance(const AlignedBox& b) const; + * and zero if the boxes intersect. + * \sa exteriorDistance(const AlignedBox&), squaredExteriorDistance(const MatrixBase&) + */ + EIGEN_DEVICE_FUNC inline Scalar squaredExteriorDistance(const AlignedBox &b) const; /** \returns the distance between the point \a p and the box \c *this, - * and zero if \a p is inside the box. - * \sa squaredExteriorDistance(const MatrixBase&), exteriorDistance(const AlignedBox&) - */ - template - EIGEN_DEVICE_FUNC inline NonInteger exteriorDistance(const MatrixBase& p) const - { EIGEN_USING_STD_MATH(sqrt) return sqrt(NonInteger(squaredExteriorDistance(p))); } + * and zero if \a p is inside the box. + * \sa squaredExteriorDistance(const MatrixBase&), exteriorDistance(const AlignedBox&) + */ + template EIGEN_DEVICE_FUNC inline NonInteger exteriorDistance(const MatrixBase &p) const + { + EIGEN_USING_STD_MATH(sqrt) return sqrt(NonInteger(squaredExteriorDistance(p))); + } /** \returns the distance between the boxes \a b and \c *this, - * and zero if the boxes intersect. - * \sa squaredExteriorDistance(const AlignedBox&), exteriorDistance(const MatrixBase&) - */ - EIGEN_DEVICE_FUNC inline NonInteger exteriorDistance(const AlignedBox& b) const - { EIGEN_USING_STD_MATH(sqrt) return sqrt(NonInteger(squaredExteriorDistance(b))); } + * and zero if the boxes intersect. + * \sa squaredExteriorDistance(const AlignedBox&), exteriorDistance(const MatrixBase&) + */ + EIGEN_DEVICE_FUNC inline NonInteger exteriorDistance(const AlignedBox &b) const + { + EIGEN_USING_STD_MATH(sqrt) return sqrt(NonInteger(squaredExteriorDistance(b))); + } /** \returns \c *this with scalar type casted to \a NewScalarType - * - * Note that if \a NewScalarType is equal to the current scalar type of \c *this - * then this function smartly returns a const reference to \c *this. - */ + * + * Note that if \a NewScalarType is equal to the current scalar type of \c *this + * then this function smartly returns a const reference to \c *this. + */ template - EIGEN_DEVICE_FUNC inline typename internal::cast_return_type >::type cast() const + EIGEN_DEVICE_FUNC inline + typename internal::cast_return_type>::type + cast() const { - return typename internal::cast_return_type >::type(*this); + return + typename internal::cast_return_type>::type(*this); } /** Copy constructor with scalar type conversion */ template - EIGEN_DEVICE_FUNC inline explicit AlignedBox(const AlignedBox& other) + EIGEN_DEVICE_FUNC inline explicit AlignedBox(const AlignedBox &other) { m_min = (other.min)().template cast(); m_max = (other.max)().template cast(); } /** \returns \c true if \c *this is approximately equal to \a other, within the precision - * determined by \a prec. - * - * \sa MatrixBase::isApprox() */ - EIGEN_DEVICE_FUNC bool isApprox(const AlignedBox& other, const RealScalar& prec = ScalarTraits::dummy_precision()) const - { return m_min.isApprox(other.m_min, prec) && m_max.isApprox(other.m_max, prec); } + * determined by \a prec. + * + * \sa MatrixBase::isApprox() */ + EIGEN_DEVICE_FUNC bool isApprox(const AlignedBox &other, + const RealScalar &prec = ScalarTraits::dummy_precision()) const + { + return m_min.isApprox(other.m_min, prec) && m_max.isApprox(other.m_max, prec); + } protected: - VectorType m_min, m_max; }; - -template +template template -EIGEN_DEVICE_FUNC inline Scalar AlignedBox::squaredExteriorDistance(const MatrixBase& a_p) const +EIGEN_DEVICE_FUNC inline Scalar AlignedBox::squaredExteriorDistance( + const MatrixBase &a_p) const { - typename internal::nested_eval::type p(a_p.derived()); + typename internal::nested_eval::type p(a_p.derived()); Scalar dist2(0); Scalar aux; - for (Index k=0; k p[k] ) - { + for (Index k = 0; k < dim(); ++k) { + if (m_min[k] > p[k]) { aux = m_min[k] - p[k]; - dist2 += aux*aux; - } - else if( p[k] > m_max[k] ) - { + dist2 += aux * aux; + } else if (p[k] > m_max[k]) { aux = p[k] - m_max[k]; - dist2 += aux*aux; + dist2 += aux * aux; } } return dist2; } -template -EIGEN_DEVICE_FUNC inline Scalar AlignedBox::squaredExteriorDistance(const AlignedBox& b) const +template +EIGEN_DEVICE_FUNC inline Scalar AlignedBox::squaredExteriorDistance(const AlignedBox &b) const { Scalar dist2(0); Scalar aux; - for (Index k=0; k b.m_max[k] ) - { + for (Index k = 0; k < dim(); ++k) { + if (m_min[k] > b.m_max[k]) { aux = m_min[k] - b.m_max[k]; - dist2 += aux*aux; - } - else if( b.m_min[k] > m_max[k] ) - { + dist2 += aux * aux; + } else if (b.m_min[k] > m_max[k]) { aux = b.m_min[k] - m_max[k]; - dist2 += aux*aux; + dist2 += aux * aux; } } return dist2; } /** \defgroup alignedboxtypedefs Global aligned box typedefs - * - * \ingroup Geometry_Module - * - * Eigen defines several typedef shortcuts for most common aligned box types. - * - * The general patterns are the following: - * - * \c AlignedBoxSizeType where \c Size can be \c 1, \c 2,\c 3,\c 4 for fixed size boxes or \c X for dynamic size, - * and where \c Type can be \c i for integer, \c f for float, \c d for double. - * - * For example, \c AlignedBox3d is a fixed-size 3x3 aligned box type of doubles, and \c AlignedBoxXf is a dynamic-size aligned box of floats. - * - * \sa class AlignedBox - */ - -#define EIGEN_MAKE_TYPEDEFS(Type, TypeSuffix, Size, SizeSuffix) \ -/** \ingroup alignedboxtypedefs */ \ -typedef AlignedBox AlignedBox##SizeSuffix##TypeSuffix; + * + * \ingroup Geometry_Module + * + * Eigen defines several typedef shortcuts for most common aligned box types. + * + * The general patterns are the following: + * + * \c AlignedBoxSizeType where \c Size can be \c 1, \c 2,\c 3,\c 4 for fixed size boxes or \c X for dynamic size, + * and where \c Type can be \c i for integer, \c f for float, \c d for double. + * + * For example, \c AlignedBox3d is a fixed-size 3x3 aligned box type of doubles, and \c AlignedBoxXf is a dynamic-size + * aligned box of floats. + * + * \sa class AlignedBox + */ + +#define EIGEN_MAKE_TYPEDEFS(Type, TypeSuffix, Size, SizeSuffix) \ + /** \ingroup alignedboxtypedefs */ \ + typedef AlignedBox AlignedBox##SizeSuffix##TypeSuffix; #define EIGEN_MAKE_TYPEDEFS_ALL_SIZES(Type, TypeSuffix) \ -EIGEN_MAKE_TYPEDEFS(Type, TypeSuffix, 1, 1) \ -EIGEN_MAKE_TYPEDEFS(Type, TypeSuffix, 2, 2) \ -EIGEN_MAKE_TYPEDEFS(Type, TypeSuffix, 3, 3) \ -EIGEN_MAKE_TYPEDEFS(Type, TypeSuffix, 4, 4) \ -EIGEN_MAKE_TYPEDEFS(Type, TypeSuffix, Dynamic, X) + EIGEN_MAKE_TYPEDEFS(Type, TypeSuffix, 1, 1) \ + EIGEN_MAKE_TYPEDEFS(Type, TypeSuffix, 2, 2) \ + EIGEN_MAKE_TYPEDEFS(Type, TypeSuffix, 3, 3) \ + EIGEN_MAKE_TYPEDEFS(Type, TypeSuffix, 4, 4) \ + EIGEN_MAKE_TYPEDEFS(Type, TypeSuffix, Dynamic, X) -EIGEN_MAKE_TYPEDEFS_ALL_SIZES(int, i) -EIGEN_MAKE_TYPEDEFS_ALL_SIZES(float, f) -EIGEN_MAKE_TYPEDEFS_ALL_SIZES(double, d) +EIGEN_MAKE_TYPEDEFS_ALL_SIZES(int, i) +EIGEN_MAKE_TYPEDEFS_ALL_SIZES(float, f) +EIGEN_MAKE_TYPEDEFS_ALL_SIZES(double, d) #undef EIGEN_MAKE_TYPEDEFS_ALL_SIZES #undef EIGEN_MAKE_TYPEDEFS -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_ALIGNEDBOX_H +#endif// EIGEN_ALIGNEDBOX_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Geometry/AngleAxis.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Geometry/AngleAxis.h index 83ee1be4..81be26f7 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Geometry/AngleAxis.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Geometry/AngleAxis.h @@ -10,133 +10,135 @@ #ifndef EIGEN_ANGLEAXIS_H #define EIGEN_ANGLEAXIS_H -namespace Eigen { +namespace Eigen { /** \geometry_module \ingroup Geometry_Module - * - * \class AngleAxis - * - * \brief Represents a 3D rotation as a rotation angle around an arbitrary 3D axis - * - * \param _Scalar the scalar type, i.e., the type of the coefficients. - * - * \warning When setting up an AngleAxis object, the axis vector \b must \b be \b normalized. - * - * The following two typedefs are provided for convenience: - * \li \c AngleAxisf for \c float - * \li \c AngleAxisd for \c double - * - * Combined with MatrixBase::Unit{X,Y,Z}, AngleAxis can be used to easily - * mimic Euler-angles. Here is an example: - * \include AngleAxis_mimic_euler.cpp - * Output: \verbinclude AngleAxis_mimic_euler.out - * - * \note This class is not aimed to be used to store a rotation transformation, - * but rather to make easier the creation of other rotation (Quaternion, rotation Matrix) - * and transformation objects. - * - * \sa class Quaternion, class Transform, MatrixBase::UnitX() - */ + * + * \class AngleAxis + * + * \brief Represents a 3D rotation as a rotation angle around an arbitrary 3D axis + * + * \param _Scalar the scalar type, i.e., the type of the coefficients. + * + * \warning When setting up an AngleAxis object, the axis vector \b must \b be \b normalized. + * + * The following two typedefs are provided for convenience: + * \li \c AngleAxisf for \c float + * \li \c AngleAxisd for \c double + * + * Combined with MatrixBase::Unit{X,Y,Z}, AngleAxis can be used to easily + * mimic Euler-angles. Here is an example: + * \include AngleAxis_mimic_euler.cpp + * Output: \verbinclude AngleAxis_mimic_euler.out + * + * \note This class is not aimed to be used to store a rotation transformation, + * but rather to make easier the creation of other rotation (Quaternion, rotation Matrix) + * and transformation objects. + * + * \sa class Quaternion, class Transform, MatrixBase::UnitX() + */ namespace internal { -template struct traits > -{ - typedef _Scalar Scalar; -}; -} + template struct traits> + { + typedef _Scalar Scalar; + }; +}// namespace internal -template -class AngleAxis : public RotationBase,3> +template class AngleAxis : public RotationBase, 3> { - typedef RotationBase,3> Base; + typedef RotationBase, 3> Base; public: - using Base::operator*; enum { Dim = 3 }; /** the scalar type of the coefficients */ typedef _Scalar Scalar; - typedef Matrix Matrix3; - typedef Matrix Vector3; + typedef Matrix Matrix3; + typedef Matrix Vector3; typedef Quaternion QuaternionType; protected: - Vector3 m_axis; Scalar m_angle; public: - /** Default constructor without initialization. */ EIGEN_DEVICE_FUNC AngleAxis() {} /** Constructs and initialize the angle-axis rotation from an \a angle in radian - * and an \a axis which \b must \b be \b normalized. - * - * \warning If the \a axis vector is not normalized, then the angle-axis object - * represents an invalid rotation. */ + * and an \a axis which \b must \b be \b normalized. + * + * \warning If the \a axis vector is not normalized, then the angle-axis object + * represents an invalid rotation. */ template - EIGEN_DEVICE_FUNC - inline AngleAxis(const Scalar& angle, const MatrixBase& axis) : m_axis(axis), m_angle(angle) {} + EIGEN_DEVICE_FUNC inline AngleAxis(const Scalar &angle, const MatrixBase &axis) + : m_axis(axis), m_angle(angle) + {} /** Constructs and initialize the angle-axis rotation from a quaternion \a q. - * This function implicitly normalizes the quaternion \a q. - */ - template - EIGEN_DEVICE_FUNC inline explicit AngleAxis(const QuaternionBase& q) { *this = q; } + * This function implicitly normalizes the quaternion \a q. + */ + template EIGEN_DEVICE_FUNC inline explicit AngleAxis(const QuaternionBase &q) + { + *this = q; + } /** Constructs and initialize the angle-axis rotation from a 3x3 rotation matrix. */ - template - EIGEN_DEVICE_FUNC inline explicit AngleAxis(const MatrixBase& m) { *this = m; } + template EIGEN_DEVICE_FUNC inline explicit AngleAxis(const MatrixBase &m) { *this = m; } /** \returns the value of the rotation angle in radian */ EIGEN_DEVICE_FUNC Scalar angle() const { return m_angle; } /** \returns a read-write reference to the stored angle in radian */ - EIGEN_DEVICE_FUNC Scalar& angle() { return m_angle; } + EIGEN_DEVICE_FUNC Scalar &angle() { return m_angle; } /** \returns the rotation axis */ - EIGEN_DEVICE_FUNC const Vector3& axis() const { return m_axis; } + EIGEN_DEVICE_FUNC const Vector3 &axis() const { return m_axis; } /** \returns a read-write reference to the stored rotation axis. - * - * \warning The rotation axis must remain a \b unit vector. - */ - EIGEN_DEVICE_FUNC Vector3& axis() { return m_axis; } + * + * \warning The rotation axis must remain a \b unit vector. + */ + EIGEN_DEVICE_FUNC Vector3 &axis() { return m_axis; } /** Concatenates two rotations */ - EIGEN_DEVICE_FUNC inline QuaternionType operator* (const AngleAxis& other) const - { return QuaternionType(*this) * QuaternionType(other); } + EIGEN_DEVICE_FUNC inline QuaternionType operator*(const AngleAxis &other) const + { + return QuaternionType(*this) * QuaternionType(other); + } /** Concatenates two rotations */ - EIGEN_DEVICE_FUNC inline QuaternionType operator* (const QuaternionType& other) const - { return QuaternionType(*this) * other; } + EIGEN_DEVICE_FUNC inline QuaternionType operator*(const QuaternionType &other) const + { + return QuaternionType(*this) * other; + } /** Concatenates two rotations */ - friend EIGEN_DEVICE_FUNC inline QuaternionType operator* (const QuaternionType& a, const AngleAxis& b) - { return a * QuaternionType(b); } + friend EIGEN_DEVICE_FUNC inline QuaternionType operator*(const QuaternionType &a, const AngleAxis &b) + { + return a * QuaternionType(b); + } /** \returns the inverse rotation, i.e., an angle-axis with opposite rotation angle */ - EIGEN_DEVICE_FUNC AngleAxis inverse() const - { return AngleAxis(-m_angle, m_axis); } + EIGEN_DEVICE_FUNC AngleAxis inverse() const { return AngleAxis(-m_angle, m_axis); } - template - EIGEN_DEVICE_FUNC AngleAxis& operator=(const QuaternionBase& q); - template - EIGEN_DEVICE_FUNC AngleAxis& operator=(const MatrixBase& m); + template EIGEN_DEVICE_FUNC AngleAxis &operator=(const QuaternionBase &q); + template EIGEN_DEVICE_FUNC AngleAxis &operator=(const MatrixBase &m); - template - EIGEN_DEVICE_FUNC AngleAxis& fromRotationMatrix(const MatrixBase& m); + template EIGEN_DEVICE_FUNC AngleAxis &fromRotationMatrix(const MatrixBase &m); EIGEN_DEVICE_FUNC Matrix3 toRotationMatrix(void) const; /** \returns \c *this with scalar type casted to \a NewScalarType - * - * Note that if \a NewScalarType is equal to the current scalar type of \c *this - * then this function smartly returns a const reference to \c *this. - */ + * + * Note that if \a NewScalarType is equal to the current scalar type of \c *this + * then this function smartly returns a const reference to \c *this. + */ template - EIGEN_DEVICE_FUNC inline typename internal::cast_return_type >::type cast() const - { return typename internal::cast_return_type >::type(*this); } + EIGEN_DEVICE_FUNC inline typename internal::cast_return_type>::type cast() const + { + return typename internal::cast_return_type>::type(*this); + } /** Copy constructor with scalar type conversion */ template - EIGEN_DEVICE_FUNC inline explicit AngleAxis(const AngleAxis& other) + EIGEN_DEVICE_FUNC inline explicit AngleAxis(const AngleAxis &other) { m_axis = other.axis().template cast(); m_angle = Scalar(other.angle()); @@ -145,45 +147,43 @@ class AngleAxis : public RotationBase,3> EIGEN_DEVICE_FUNC static inline const AngleAxis Identity() { return AngleAxis(Scalar(0), Vector3::UnitX()); } /** \returns \c true if \c *this is approximately equal to \a other, within the precision - * determined by \a prec. - * - * \sa MatrixBase::isApprox() */ - EIGEN_DEVICE_FUNC bool isApprox(const AngleAxis& other, const typename NumTraits::Real& prec = NumTraits::dummy_precision()) const - { return m_axis.isApprox(other.m_axis, prec) && internal::isApprox(m_angle,other.m_angle, prec); } + * determined by \a prec. + * + * \sa MatrixBase::isApprox() */ + EIGEN_DEVICE_FUNC bool isApprox(const AngleAxis &other, + const typename NumTraits::Real &prec = NumTraits::dummy_precision()) const + { + return m_axis.isApprox(other.m_axis, prec) && internal::isApprox(m_angle, other.m_angle, prec); + } }; /** \ingroup Geometry_Module - * single precision angle-axis type */ + * single precision angle-axis type */ typedef AngleAxis AngleAxisf; /** \ingroup Geometry_Module - * double precision angle-axis type */ + * double precision angle-axis type */ typedef AngleAxis AngleAxisd; /** Set \c *this from a \b unit quaternion. - * - * The resulting axis is normalized, and the computed angle is in the [0,pi] range. - * - * This function implicitly normalizes the quaternion \a q. - */ + * + * The resulting axis is normalized, and the computed angle is in the [0,pi] range. + * + * This function implicitly normalizes the quaternion \a q. + */ template template -EIGEN_DEVICE_FUNC AngleAxis& AngleAxis::operator=(const QuaternionBase& q) +EIGEN_DEVICE_FUNC AngleAxis &AngleAxis::operator=(const QuaternionBase &q) { EIGEN_USING_STD_MATH(atan2) EIGEN_USING_STD_MATH(abs) Scalar n = q.vec().norm(); - if(n::epsilon()) - n = q.vec().stableNorm(); + if (n < NumTraits::epsilon()) n = q.vec().stableNorm(); - if (n != Scalar(0)) - { - m_angle = Scalar(2)*atan2(n, abs(q.w())); - if(q.w() < Scalar(0)) - n = -n; - m_axis = q.vec() / n; - } - else - { + if (n != Scalar(0)) { + m_angle = Scalar(2) * atan2(n, abs(q.w())); + if (q.w() < Scalar(0)) n = -n; + m_axis = q.vec() / n; + } else { m_angle = Scalar(0); m_axis << Scalar(1), Scalar(0), Scalar(0); } @@ -191,10 +191,10 @@ EIGEN_DEVICE_FUNC AngleAxis& AngleAxis::operator=(const Quaterni } /** Set \c *this from a 3x3 rotation matrix \a mat. - */ + */ template template -EIGEN_DEVICE_FUNC AngleAxis& AngleAxis::operator=(const MatrixBase& mat) +EIGEN_DEVICE_FUNC AngleAxis &AngleAxis::operator=(const MatrixBase &mat) { // Since a direct conversion would not be really faster, // let's use the robust Quaternion implementation: @@ -202,46 +202,45 @@ EIGEN_DEVICE_FUNC AngleAxis& AngleAxis::operator=(const MatrixBa } /** -* \brief Sets \c *this from a 3x3 rotation matrix. -**/ + * \brief Sets \c *this from a 3x3 rotation matrix. + **/ template template -EIGEN_DEVICE_FUNC AngleAxis& AngleAxis::fromRotationMatrix(const MatrixBase& mat) +EIGEN_DEVICE_FUNC AngleAxis &AngleAxis::fromRotationMatrix(const MatrixBase &mat) { return *this = QuaternionType(mat); } /** Constructs and \returns an equivalent 3x3 rotation matrix. - */ + */ template -typename AngleAxis::Matrix3 -EIGEN_DEVICE_FUNC AngleAxis::toRotationMatrix(void) const +typename AngleAxis::Matrix3 EIGEN_DEVICE_FUNC AngleAxis::toRotationMatrix(void) const { EIGEN_USING_STD_MATH(sin) EIGEN_USING_STD_MATH(cos) Matrix3 res; - Vector3 sin_axis = sin(m_angle) * m_axis; + Vector3 sin_axis = sin(m_angle) * m_axis; Scalar c = cos(m_angle); - Vector3 cos1_axis = (Scalar(1)-c) * m_axis; + Vector3 cos1_axis = (Scalar(1) - c) * m_axis; Scalar tmp; tmp = cos1_axis.x() * m_axis.y(); - res.coeffRef(0,1) = tmp - sin_axis.z(); - res.coeffRef(1,0) = tmp + sin_axis.z(); + res.coeffRef(0, 1) = tmp - sin_axis.z(); + res.coeffRef(1, 0) = tmp + sin_axis.z(); tmp = cos1_axis.x() * m_axis.z(); - res.coeffRef(0,2) = tmp + sin_axis.y(); - res.coeffRef(2,0) = tmp - sin_axis.y(); + res.coeffRef(0, 2) = tmp + sin_axis.y(); + res.coeffRef(2, 0) = tmp - sin_axis.y(); tmp = cos1_axis.y() * m_axis.z(); - res.coeffRef(1,2) = tmp - sin_axis.x(); - res.coeffRef(2,1) = tmp + sin_axis.x(); + res.coeffRef(1, 2) = tmp - sin_axis.x(); + res.coeffRef(2, 1) = tmp + sin_axis.x(); res.diagonal() = (cos1_axis.cwiseProduct(m_axis)).array() + c; return res; } -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_ANGLEAXIS_H +#endif// EIGEN_ANGLEAXIS_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Geometry/EulerAngles.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Geometry/EulerAngles.h index c633268a..921d98ec 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Geometry/EulerAngles.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Geometry/EulerAngles.h @@ -10,105 +10,96 @@ #ifndef EIGEN_EULERANGLES_H #define EIGEN_EULERANGLES_H -namespace Eigen { +namespace Eigen { /** \geometry_module \ingroup Geometry_Module - * - * - * \returns the Euler-angles of the rotation matrix \c *this using the convention defined by the triplet (\a a0,\a a1,\a a2) - * - * Each of the three parameters \a a0,\a a1,\a a2 represents the respective rotation axis as an integer in {0,1,2}. - * For instance, in: - * \code Vector3f ea = mat.eulerAngles(2, 0, 2); \endcode - * "2" represents the z axis and "0" the x axis, etc. The returned angles are such that - * we have the following equality: - * \code - * mat == AngleAxisf(ea[0], Vector3f::UnitZ()) - * * AngleAxisf(ea[1], Vector3f::UnitX()) - * * AngleAxisf(ea[2], Vector3f::UnitZ()); \endcode - * This corresponds to the right-multiply conventions (with right hand side frames). - * - * The returned angles are in the ranges [0:pi]x[-pi:pi]x[-pi:pi]. - * - * \sa class AngleAxis - */ + * + * + * \returns the Euler-angles of the rotation matrix \c *this using the convention defined by the triplet (\a a0,\a a1,\a + * a2) + * + * Each of the three parameters \a a0,\a a1,\a a2 represents the respective rotation axis as an integer in {0,1,2}. + * For instance, in: + * \code Vector3f ea = mat.eulerAngles(2, 0, 2); \endcode + * "2" represents the z axis and "0" the x axis, etc. The returned angles are such that + * we have the following equality: + * \code + * mat == AngleAxisf(ea[0], Vector3f::UnitZ()) + * * AngleAxisf(ea[1], Vector3f::UnitX()) + * * AngleAxisf(ea[2], Vector3f::UnitZ()); \endcode + * This corresponds to the right-multiply conventions (with right hand side frames). + * + * The returned angles are in the ranges [0:pi]x[-pi:pi]x[-pi:pi]. + * + * \sa class AngleAxis + */ template -EIGEN_DEVICE_FUNC inline Matrix::Scalar,3,1> -MatrixBase::eulerAngles(Index a0, Index a1, Index a2) const +EIGEN_DEVICE_FUNC inline Matrix::Scalar, 3, 1> + MatrixBase::eulerAngles(Index a0, Index a1, Index a2) const { EIGEN_USING_STD_MATH(atan2) EIGEN_USING_STD_MATH(sin) EIGEN_USING_STD_MATH(cos) /* Implemented from Graphics Gems IV */ - EIGEN_STATIC_ASSERT_MATRIX_SPECIFIC_SIZE(Derived,3,3) + EIGEN_STATIC_ASSERT_MATRIX_SPECIFIC_SIZE(Derived, 3, 3) - Matrix res; - typedef Matrix Vector2; + Matrix res; + typedef Matrix Vector2; - const Index odd = ((a0+1)%3 == a1) ? 0 : 1; + const Index odd = ((a0 + 1) % 3 == a1) ? 0 : 1; const Index i = a0; - const Index j = (a0 + 1 + odd)%3; - const Index k = (a0 + 2 - odd)%3; - - if (a0==a2) - { - res[0] = atan2(coeff(j,i), coeff(k,i)); - if((odd && res[0]Scalar(0))) - { - if(res[0] > Scalar(0)) { + const Index j = (a0 + 1 + odd) % 3; + const Index k = (a0 + 2 - odd) % 3; + + if (a0 == a2) { + res[0] = atan2(coeff(j, i), coeff(k, i)); + if ((odd && res[0] < Scalar(0)) || ((!odd) && res[0] > Scalar(0))) { + if (res[0] > Scalar(0)) { res[0] -= Scalar(EIGEN_PI); - } - else { + } else { res[0] += Scalar(EIGEN_PI); } - Scalar s2 = Vector2(coeff(j,i), coeff(k,i)).norm(); - res[1] = -atan2(s2, coeff(i,i)); + Scalar s2 = Vector2(coeff(j, i), coeff(k, i)).norm(); + res[1] = -atan2(s2, coeff(i, i)); + } else { + Scalar s2 = Vector2(coeff(j, i), coeff(k, i)).norm(); + res[1] = atan2(s2, coeff(i, i)); } - else - { - Scalar s2 = Vector2(coeff(j,i), coeff(k,i)).norm(); - res[1] = atan2(s2, coeff(i,i)); - } - + // With a=(0,1,0), we have i=0; j=1; k=2, and after computing the first two angles, // we can compute their respective rotation, and apply its inverse to M. Since the result must // be a rotation around x, we have: // - // c2 s1.s2 c1.s2 1 0 0 + // c2 s1.s2 c1.s2 1 0 0 // 0 c1 -s1 * M = 0 c3 s3 // -s2 s1.c2 c1.c2 0 -s3 c3 // // Thus: m11.c1 - m21.s1 = c3 & m12.c1 - m22.s1 = s3 - + Scalar s1 = sin(res[0]); Scalar c1 = cos(res[0]); - res[2] = atan2(c1*coeff(j,k)-s1*coeff(k,k), c1*coeff(j,j) - s1 * coeff(k,j)); - } - else - { - res[0] = atan2(coeff(j,k), coeff(k,k)); - Scalar c2 = Vector2(coeff(i,i), coeff(i,j)).norm(); - if((odd && res[0]Scalar(0))) { - if(res[0] > Scalar(0)) { + res[2] = atan2(c1 * coeff(j, k) - s1 * coeff(k, k), c1 * coeff(j, j) - s1 * coeff(k, j)); + } else { + res[0] = atan2(coeff(j, k), coeff(k, k)); + Scalar c2 = Vector2(coeff(i, i), coeff(i, j)).norm(); + if ((odd && res[0] < Scalar(0)) || ((!odd) && res[0] > Scalar(0))) { + if (res[0] > Scalar(0)) { res[0] -= Scalar(EIGEN_PI); - } - else { + } else { res[0] += Scalar(EIGEN_PI); } - res[1] = atan2(-coeff(i,k), -c2); - } - else - res[1] = atan2(-coeff(i,k), c2); + res[1] = atan2(-coeff(i, k), -c2); + } else + res[1] = atan2(-coeff(i, k), c2); Scalar s1 = sin(res[0]); Scalar c1 = cos(res[0]); - res[2] = atan2(s1*coeff(k,i)-c1*coeff(j,i), c1*coeff(j,j) - s1 * coeff(k,j)); + res[2] = atan2(s1 * coeff(k, i) - c1 * coeff(j, i), c1 * coeff(j, j) - s1 * coeff(k, j)); } - if (!odd) - res = -res; - + if (!odd) res = -res; + return res; } -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_EULERANGLES_H +#endif// EIGEN_EULERANGLES_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Geometry/Homogeneous.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Geometry/Homogeneous.h index 5f0da1a9..e39ac5fe 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Geometry/Homogeneous.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Geometry/Homogeneous.h @@ -10,142 +10,137 @@ #ifndef EIGEN_HOMOGENEOUS_H #define EIGEN_HOMOGENEOUS_H -namespace Eigen { +namespace Eigen { /** \geometry_module \ingroup Geometry_Module - * - * \class Homogeneous - * - * \brief Expression of one (or a set of) homogeneous vector(s) - * - * \param MatrixType the type of the object in which we are making homogeneous - * - * This class represents an expression of one (or a set of) homogeneous vector(s). - * It is the return type of MatrixBase::homogeneous() and most of the time - * this is the only way it is used. - * - * \sa MatrixBase::homogeneous() - */ + * + * \class Homogeneous + * + * \brief Expression of one (or a set of) homogeneous vector(s) + * + * \param MatrixType the type of the object in which we are making homogeneous + * + * This class represents an expression of one (or a set of) homogeneous vector(s). + * It is the return type of MatrixBase::homogeneous() and most of the time + * this is the only way it is used. + * + * \sa MatrixBase::homogeneous() + */ namespace internal { -template -struct traits > - : traits -{ - typedef typename traits::StorageKind StorageKind; - typedef typename ref_selector::type MatrixTypeNested; - typedef typename remove_reference::type _MatrixTypeNested; - enum { - RowsPlusOne = (MatrixType::RowsAtCompileTime != Dynamic) ? - int(MatrixType::RowsAtCompileTime) + 1 : Dynamic, - ColsPlusOne = (MatrixType::ColsAtCompileTime != Dynamic) ? - int(MatrixType::ColsAtCompileTime) + 1 : Dynamic, - RowsAtCompileTime = Direction==Vertical ? RowsPlusOne : MatrixType::RowsAtCompileTime, - ColsAtCompileTime = Direction==Horizontal ? ColsPlusOne : MatrixType::ColsAtCompileTime, - MaxRowsAtCompileTime = RowsAtCompileTime, - MaxColsAtCompileTime = ColsAtCompileTime, - TmpFlags = _MatrixTypeNested::Flags & HereditaryBits, - Flags = ColsAtCompileTime==1 ? (TmpFlags & ~RowMajorBit) - : RowsAtCompileTime==1 ? (TmpFlags | RowMajorBit) - : TmpFlags + template struct traits> : traits + { + typedef typename traits::StorageKind StorageKind; + typedef typename ref_selector::type MatrixTypeNested; + typedef typename remove_reference::type _MatrixTypeNested; + enum { + RowsPlusOne = (MatrixType::RowsAtCompileTime != Dynamic) ? int(MatrixType::RowsAtCompileTime) + 1 : Dynamic, + ColsPlusOne = (MatrixType::ColsAtCompileTime != Dynamic) ? int(MatrixType::ColsAtCompileTime) + 1 : Dynamic, + RowsAtCompileTime = Direction == Vertical ? RowsPlusOne : MatrixType::RowsAtCompileTime, + ColsAtCompileTime = Direction == Horizontal ? ColsPlusOne : MatrixType::ColsAtCompileTime, + MaxRowsAtCompileTime = RowsAtCompileTime, + MaxColsAtCompileTime = ColsAtCompileTime, + TmpFlags = _MatrixTypeNested::Flags & HereditaryBits, + Flags = ColsAtCompileTime == 1 ? (TmpFlags & ~RowMajorBit) + : RowsAtCompileTime == 1 ? (TmpFlags | RowMajorBit) + : TmpFlags + }; }; -}; -template struct homogeneous_left_product_impl; -template struct homogeneous_right_product_impl; + template struct homogeneous_left_product_impl; + template struct homogeneous_right_product_impl; -} // end namespace internal +}// end namespace internal -template class Homogeneous - : public MatrixBase >, internal::no_assignment_operator +template +class Homogeneous + : public MatrixBase> + , internal::no_assignment_operator { - public: +public: + typedef MatrixType NestedExpression; + enum { Direction = _Direction }; - typedef MatrixType NestedExpression; - enum { Direction = _Direction }; + typedef MatrixBase Base; + EIGEN_DENSE_PUBLIC_INTERFACE(Homogeneous) - typedef MatrixBase Base; - EIGEN_DENSE_PUBLIC_INTERFACE(Homogeneous) + EIGEN_DEVICE_FUNC explicit inline Homogeneous(const MatrixType &matrix) : m_matrix(matrix) {} - EIGEN_DEVICE_FUNC explicit inline Homogeneous(const MatrixType& matrix) - : m_matrix(matrix) - {} + EIGEN_DEVICE_FUNC inline Index rows() const { return m_matrix.rows() + (int(Direction) == Vertical ? 1 : 0); } + EIGEN_DEVICE_FUNC inline Index cols() const { return m_matrix.cols() + (int(Direction) == Horizontal ? 1 : 0); } - EIGEN_DEVICE_FUNC inline Index rows() const { return m_matrix.rows() + (int(Direction)==Vertical ? 1 : 0); } - EIGEN_DEVICE_FUNC inline Index cols() const { return m_matrix.cols() + (int(Direction)==Horizontal ? 1 : 0); } - - EIGEN_DEVICE_FUNC const NestedExpression& nestedExpression() const { return m_matrix; } + EIGEN_DEVICE_FUNC const NestedExpression &nestedExpression() const { return m_matrix; } - template - EIGEN_DEVICE_FUNC inline const Product - operator* (const MatrixBase& rhs) const - { - eigen_assert(int(Direction)==Horizontal); - return Product(*this,rhs.derived()); - } + template + EIGEN_DEVICE_FUNC inline const Product operator*(const MatrixBase &rhs) const + { + eigen_assert(int(Direction) == Horizontal); + return Product(*this, rhs.derived()); + } - template friend - EIGEN_DEVICE_FUNC inline const Product - operator* (const MatrixBase& lhs, const Homogeneous& rhs) - { - eigen_assert(int(Direction)==Vertical); - return Product(lhs.derived(),rhs); - } + template + friend EIGEN_DEVICE_FUNC inline const Product operator*(const MatrixBase &lhs, + const Homogeneous &rhs) + { + eigen_assert(int(Direction) == Vertical); + return Product(lhs.derived(), rhs); + } - template friend - EIGEN_DEVICE_FUNC inline const Product, Homogeneous > - operator* (const Transform& lhs, const Homogeneous& rhs) - { - eigen_assert(int(Direction)==Vertical); - return Product, Homogeneous>(lhs,rhs); - } + template + friend EIGEN_DEVICE_FUNC inline const Product, Homogeneous> + operator*(const Transform &lhs, const Homogeneous &rhs) + { + eigen_assert(int(Direction) == Vertical); + return Product, Homogeneous>(lhs, rhs); + } - template - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE typename internal::result_of::type - redux(const Func& func) const - { - return func(m_matrix.redux(func), Scalar(1)); - } + template + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE typename internal::result_of::type redux( + const Func &func) const + { + return func(m_matrix.redux(func), Scalar(1)); + } - protected: - typename MatrixType::Nested m_matrix; +protected: + typename MatrixType::Nested m_matrix; }; /** \geometry_module \ingroup Geometry_Module - * - * \returns a vector expression that is one longer than the vector argument, with the value 1 symbolically appended as the last coefficient. - * - * This can be used to convert affine coordinates to homogeneous coordinates. - * - * \only_for_vectors - * - * Example: \include MatrixBase_homogeneous.cpp - * Output: \verbinclude MatrixBase_homogeneous.out - * - * \sa VectorwiseOp::homogeneous(), class Homogeneous - */ + * + * \returns a vector expression that is one longer than the vector argument, with the value 1 symbolically appended as + * the last coefficient. + * + * This can be used to convert affine coordinates to homogeneous coordinates. + * + * \only_for_vectors + * + * Example: \include MatrixBase_homogeneous.cpp + * Output: \verbinclude MatrixBase_homogeneous.out + * + * \sa VectorwiseOp::homogeneous(), class Homogeneous + */ template -EIGEN_DEVICE_FUNC inline typename MatrixBase::HomogeneousReturnType -MatrixBase::homogeneous() const +EIGEN_DEVICE_FUNC inline typename MatrixBase::HomogeneousReturnType MatrixBase::homogeneous() const { EIGEN_STATIC_ASSERT_VECTOR_ONLY(Derived); return HomogeneousReturnType(derived()); } /** \geometry_module \ingroup Geometry_Module - * - * \returns an expression where the value 1 is symbolically appended as the final coefficient to each column (or row) of the matrix. - * - * This can be used to convert affine coordinates to homogeneous coordinates. - * - * Example: \include VectorwiseOp_homogeneous.cpp - * Output: \verbinclude VectorwiseOp_homogeneous.out - * - * \sa MatrixBase::homogeneous(), class Homogeneous */ + * + * \returns an expression where the value 1 is symbolically appended as the final coefficient to each column (or row) of + * the matrix. + * + * This can be used to convert affine coordinates to homogeneous coordinates. + * + * Example: \include VectorwiseOp_homogeneous.cpp + * Output: \verbinclude VectorwiseOp_homogeneous.out + * + * \sa MatrixBase::homogeneous(), class Homogeneous */ template -EIGEN_DEVICE_FUNC inline Homogeneous -VectorwiseOp::homogeneous() const +EIGEN_DEVICE_FUNC inline Homogeneous + VectorwiseOp::homogeneous() const { return HomogeneousReturnType(_expression()); } @@ -169,329 +164,338 @@ VectorwiseOp::homogeneous() const * \sa VectorwiseOp::hnormalized() */ template EIGEN_DEVICE_FUNC inline const typename MatrixBase::HNormalizedReturnType -MatrixBase::hnormalized() const + MatrixBase::hnormalized() const { EIGEN_STATIC_ASSERT_VECTOR_ONLY(Derived); - return ConstStartMinusOne(derived(),0,0, - ColsAtCompileTime==1?size()-1:1, - ColsAtCompileTime==1?1:size()-1) / coeff(size()-1); + return ConstStartMinusOne( + derived(), 0, 0, ColsAtCompileTime == 1 ? size() - 1 : 1, ColsAtCompileTime == 1 ? 1 : size() - 1) + / coeff(size() - 1); } /** \geometry_module \ingroup Geometry_Module - * - * \brief column or row-wise homogeneous normalization - * - * \returns an expression of the first N-1 coefficients of each column (or row) of \c *this divided by the last coefficient of each column (or row). - * - * This can be used to convert homogeneous coordinates to affine coordinates. - * - * It is conceptually equivalent to calling MatrixBase::hnormalized() to each column (or row) of \c *this. - * - * Example: \include DirectionWise_hnormalized.cpp - * Output: \verbinclude DirectionWise_hnormalized.out - * - * \sa MatrixBase::hnormalized() */ + * + * \brief column or row-wise homogeneous normalization + * + * \returns an expression of the first N-1 coefficients of each column (or row) of \c *this divided by the last + * coefficient of each column (or row). + * + * This can be used to convert homogeneous coordinates to affine coordinates. + * + * It is conceptually equivalent to calling MatrixBase::hnormalized() to each column (or row) of \c *this. + * + * Example: \include DirectionWise_hnormalized.cpp + * Output: \verbinclude DirectionWise_hnormalized.out + * + * \sa MatrixBase::hnormalized() */ template -EIGEN_DEVICE_FUNC inline const typename VectorwiseOp::HNormalizedReturnType -VectorwiseOp::hnormalized() const +EIGEN_DEVICE_FUNC inline const typename VectorwiseOp::HNormalizedReturnType + VectorwiseOp::hnormalized() const { - return HNormalized_Block(_expression(),0,0, - Direction==Vertical ? _expression().rows()-1 : _expression().rows(), - Direction==Horizontal ? _expression().cols()-1 : _expression().cols()).cwiseQuotient( - Replicate - (HNormalized_Factors(_expression(), - Direction==Vertical ? _expression().rows()-1:0, - Direction==Horizontal ? _expression().cols()-1:0, - Direction==Vertical ? 1 : _expression().rows(), - Direction==Horizontal ? 1 : _expression().cols()), - Direction==Vertical ? _expression().rows()-1 : 1, - Direction==Horizontal ? _expression().cols()-1 : 1)); + return HNormalized_Block(_expression(), + 0, + 0, + Direction == Vertical ? _expression().rows() - 1 : _expression().rows(), + Direction == Horizontal ? _expression().cols() - 1 : _expression().cols()) + .cwiseQuotient(Replicate(HNormalized_Factors(_expression(), + Direction == Vertical ? _expression().rows() - 1 : 0, + Direction == Horizontal ? _expression().cols() - 1 : 0, + Direction == Vertical ? 1 : _expression().rows(), + Direction == Horizontal ? 1 : _expression().cols()), + Direction == Vertical ? _expression().rows() - 1 : 1, + Direction == Horizontal ? _expression().cols() - 1 : 1)); } namespace internal { -template -struct take_matrix_for_product -{ - typedef MatrixOrTransformType type; - EIGEN_DEVICE_FUNC static const type& run(const type &x) { return x; } -}; + template struct take_matrix_for_product + { + typedef MatrixOrTransformType type; + EIGEN_DEVICE_FUNC static const type &run(const type &x) { return x; } + }; -template -struct take_matrix_for_product > -{ - typedef Transform TransformType; - typedef typename internal::add_const::type type; - EIGEN_DEVICE_FUNC static type run (const TransformType& x) { return x.affine(); } -}; + template + struct take_matrix_for_product> + { + typedef Transform TransformType; + typedef typename internal::add_const::type type; + EIGEN_DEVICE_FUNC static type run(const TransformType &x) { return x.affine(); } + }; -template -struct take_matrix_for_product > -{ - typedef Transform TransformType; - typedef typename TransformType::MatrixType type; - EIGEN_DEVICE_FUNC static const type& run (const TransformType& x) { return x.matrix(); } -}; + template + struct take_matrix_for_product> + { + typedef Transform TransformType; + typedef typename TransformType::MatrixType type; + EIGEN_DEVICE_FUNC static const type &run(const TransformType &x) { return x.matrix(); } + }; -template -struct traits,Lhs> > -{ - typedef typename take_matrix_for_product::type LhsMatrixType; - typedef typename remove_all::type MatrixTypeCleaned; - typedef typename remove_all::type LhsMatrixTypeCleaned; - typedef typename make_proper_matrix_type< - typename traits::Scalar, - LhsMatrixTypeCleaned::RowsAtCompileTime, - MatrixTypeCleaned::ColsAtCompileTime, - MatrixTypeCleaned::PlainObject::Options, - LhsMatrixTypeCleaned::MaxRowsAtCompileTime, - MatrixTypeCleaned::MaxColsAtCompileTime>::type ReturnType; -}; + template + struct traits, Lhs>> + { + typedef typename take_matrix_for_product::type LhsMatrixType; + typedef typename remove_all::type MatrixTypeCleaned; + typedef typename remove_all::type LhsMatrixTypeCleaned; + typedef typename make_proper_matrix_type::Scalar, + LhsMatrixTypeCleaned::RowsAtCompileTime, + MatrixTypeCleaned::ColsAtCompileTime, + MatrixTypeCleaned::PlainObject::Options, + LhsMatrixTypeCleaned::MaxRowsAtCompileTime, + MatrixTypeCleaned::MaxColsAtCompileTime>::type ReturnType; + }; -template -struct homogeneous_left_product_impl,Lhs> - : public ReturnByValue,Lhs> > -{ - typedef typename traits::LhsMatrixType LhsMatrixType; - typedef typename remove_all::type LhsMatrixTypeCleaned; - typedef typename remove_all::type LhsMatrixTypeNested; - EIGEN_DEVICE_FUNC homogeneous_left_product_impl(const Lhs& lhs, const MatrixType& rhs) - : m_lhs(take_matrix_for_product::run(lhs)), - m_rhs(rhs) - {} - - EIGEN_DEVICE_FUNC inline Index rows() const { return m_lhs.rows(); } - EIGEN_DEVICE_FUNC inline Index cols() const { return m_rhs.cols(); } - - template EIGEN_DEVICE_FUNC void evalTo(Dest& dst) const + template + struct homogeneous_left_product_impl, Lhs> + : public ReturnByValue, Lhs>> { - // FIXME investigate how to allow lazy evaluation of this product when possible - dst = Block - (m_lhs,0,0,m_lhs.rows(),m_lhs.cols()-1) * m_rhs; - dst += m_lhs.col(m_lhs.cols()-1).rowwise() - .template replicate(m_rhs.cols()); - } + typedef typename traits::LhsMatrixType LhsMatrixType; + typedef typename remove_all::type LhsMatrixTypeCleaned; + typedef typename remove_all::type LhsMatrixTypeNested; + EIGEN_DEVICE_FUNC homogeneous_left_product_impl(const Lhs &lhs, const MatrixType &rhs) + : m_lhs(take_matrix_for_product::run(lhs)), m_rhs(rhs) + {} - typename LhsMatrixTypeCleaned::Nested m_lhs; - typename MatrixType::Nested m_rhs; -}; + EIGEN_DEVICE_FUNC inline Index rows() const { return m_lhs.rows(); } + EIGEN_DEVICE_FUNC inline Index cols() const { return m_rhs.cols(); } -template -struct traits,Rhs> > -{ - typedef typename make_proper_matrix_type::Scalar, - MatrixType::RowsAtCompileTime, - Rhs::ColsAtCompileTime, - MatrixType::PlainObject::Options, - MatrixType::MaxRowsAtCompileTime, - Rhs::MaxColsAtCompileTime>::type ReturnType; -}; + template EIGEN_DEVICE_FUNC void evalTo(Dest &dst) const + { + // FIXME investigate how to allow lazy evaluation of this product when possible + dst = Block( + m_lhs, 0, 0, m_lhs.rows(), m_lhs.cols() - 1) + * m_rhs; + dst += m_lhs.col(m_lhs.cols() - 1).rowwise().template replicate(m_rhs.cols()); + } -template -struct homogeneous_right_product_impl,Rhs> - : public ReturnByValue,Rhs> > -{ - typedef typename remove_all::type RhsNested; - EIGEN_DEVICE_FUNC homogeneous_right_product_impl(const MatrixType& lhs, const Rhs& rhs) - : m_lhs(lhs), m_rhs(rhs) - {} + typename LhsMatrixTypeCleaned::Nested m_lhs; + typename MatrixType::Nested m_rhs; + }; - EIGEN_DEVICE_FUNC inline Index rows() const { return m_lhs.rows(); } - EIGEN_DEVICE_FUNC inline Index cols() const { return m_rhs.cols(); } + template + struct traits, Rhs>> + { + typedef typename make_proper_matrix_type::Scalar, + MatrixType::RowsAtCompileTime, + Rhs::ColsAtCompileTime, + MatrixType::PlainObject::Options, + MatrixType::MaxRowsAtCompileTime, + Rhs::MaxColsAtCompileTime>::type ReturnType; + }; - template EIGEN_DEVICE_FUNC void evalTo(Dest& dst) const + template + struct homogeneous_right_product_impl, Rhs> + : public ReturnByValue, Rhs>> { - // FIXME investigate how to allow lazy evaluation of this product when possible - dst = m_lhs * Block - (m_rhs,0,0,m_rhs.rows()-1,m_rhs.cols()); - dst += m_rhs.row(m_rhs.rows()-1).colwise() - .template replicate(m_lhs.rows()); - } + typedef typename remove_all::type RhsNested; + EIGEN_DEVICE_FUNC homogeneous_right_product_impl(const MatrixType &lhs, const Rhs &rhs) : m_lhs(lhs), m_rhs(rhs) {} - typename MatrixType::Nested m_lhs; - typename Rhs::Nested m_rhs; -}; + EIGEN_DEVICE_FUNC inline Index rows() const { return m_lhs.rows(); } + EIGEN_DEVICE_FUNC inline Index cols() const { return m_rhs.cols(); } -template -struct evaluator_traits > -{ - typedef typename storage_kind_to_evaluator_kind::Kind Kind; - typedef HomogeneousShape Shape; -}; + template EIGEN_DEVICE_FUNC void evalTo(Dest &dst) const + { + // FIXME investigate how to allow lazy evaluation of this product when possible + dst = m_lhs + * Block(m_rhs, 0, 0, m_rhs.rows() - 1, m_rhs.cols()); + dst += m_rhs.row(m_rhs.rows() - 1).colwise().template replicate(m_lhs.rows()); + } -template<> struct AssignmentKind { typedef Dense2Dense Kind; }; + typename MatrixType::Nested m_lhs; + typename Rhs::Nested m_rhs; + }; + template struct evaluator_traits> + { + typedef typename storage_kind_to_evaluator_kind::Kind Kind; + typedef HomogeneousShape Shape; + }; + + template<> struct AssignmentKind + { + typedef Dense2Dense Kind; + }; -template -struct unary_evaluator, IndexBased> - : evaluator::PlainObject > -{ - typedef Homogeneous XprType; - typedef typename XprType::PlainObject PlainObject; - typedef evaluator Base; - EIGEN_DEVICE_FUNC explicit unary_evaluator(const XprType& op) - : Base(), m_temp(op) + template + struct unary_evaluator, IndexBased> + : evaluator::PlainObject> { - ::new (static_cast(this)) Base(m_temp); - } + typedef Homogeneous XprType; + typedef typename XprType::PlainObject PlainObject; + typedef evaluator Base; -protected: - PlainObject m_temp; -}; + EIGEN_DEVICE_FUNC explicit unary_evaluator(const XprType &op) : Base(), m_temp(op) + { + ::new (static_cast(this)) Base(m_temp); + } -// dense = homogeneous -template< typename DstXprType, typename ArgType, typename Scalar> -struct Assignment, internal::assign_op, Dense2Dense> -{ - typedef Homogeneous SrcXprType; - EIGEN_DEVICE_FUNC static void run(DstXprType &dst, const SrcXprType &src, const internal::assign_op &) + protected: + PlainObject m_temp; + }; + + // dense = homogeneous + template + struct Assignment, + internal::assign_op, + Dense2Dense> { - Index dstRows = src.rows(); - Index dstCols = src.cols(); - if((dst.rows()!=dstRows) || (dst.cols()!=dstCols)) - dst.resize(dstRows, dstCols); + typedef Homogeneous SrcXprType; + EIGEN_DEVICE_FUNC static void + run(DstXprType &dst, const SrcXprType &src, const internal::assign_op &) + { + Index dstRows = src.rows(); + Index dstCols = src.cols(); + if ((dst.rows() != dstRows) || (dst.cols() != dstCols)) dst.resize(dstRows, dstCols); - dst.template topRows(src.nestedExpression().rows()) = src.nestedExpression(); - dst.row(dst.rows()-1).setOnes(); - } -}; + dst.template topRows(src.nestedExpression().rows()) = src.nestedExpression(); + dst.row(dst.rows() - 1).setOnes(); + } + }; -// dense = homogeneous -template< typename DstXprType, typename ArgType, typename Scalar> -struct Assignment, internal::assign_op, Dense2Dense> -{ - typedef Homogeneous SrcXprType; - EIGEN_DEVICE_FUNC static void run(DstXprType &dst, const SrcXprType &src, const internal::assign_op &) + // dense = homogeneous + template + struct Assignment, + internal::assign_op, + Dense2Dense> { - Index dstRows = src.rows(); - Index dstCols = src.cols(); - if((dst.rows()!=dstRows) || (dst.cols()!=dstCols)) - dst.resize(dstRows, dstCols); + typedef Homogeneous SrcXprType; + EIGEN_DEVICE_FUNC static void + run(DstXprType &dst, const SrcXprType &src, const internal::assign_op &) + { + Index dstRows = src.rows(); + Index dstCols = src.cols(); + if ((dst.rows() != dstRows) || (dst.cols() != dstCols)) dst.resize(dstRows, dstCols); - dst.template leftCols(src.nestedExpression().cols()) = src.nestedExpression(); - dst.col(dst.cols()-1).setOnes(); - } -}; + dst.template leftCols(src.nestedExpression().cols()) = src.nestedExpression(); + dst.col(dst.cols() - 1).setOnes(); + } + }; -template -struct generic_product_impl, Rhs, HomogeneousShape, DenseShape, ProductTag> -{ - template - EIGEN_DEVICE_FUNC static void evalTo(Dest& dst, const Homogeneous& lhs, const Rhs& rhs) + template + struct generic_product_impl, Rhs, HomogeneousShape, DenseShape, ProductTag> { - homogeneous_right_product_impl, Rhs>(lhs.nestedExpression(), rhs).evalTo(dst); - } -}; + template + EIGEN_DEVICE_FUNC static void evalTo(Dest &dst, const Homogeneous &lhs, const Rhs &rhs) + { + homogeneous_right_product_impl, Rhs>(lhs.nestedExpression(), rhs).evalTo(dst); + } + }; -template -struct homogeneous_right_product_refactoring_helper -{ - enum { - Dim = Lhs::ColsAtCompileTime, - Rows = Lhs::RowsAtCompileTime + template struct homogeneous_right_product_refactoring_helper + { + enum { Dim = Lhs::ColsAtCompileTime, Rows = Lhs::RowsAtCompileTime }; + typedef typename Rhs::template ConstNRowsBlockXpr::Type LinearBlockConst; + typedef typename remove_const::type LinearBlock; + typedef typename Rhs::ConstRowXpr ConstantColumn; + typedef Replicate ConstantBlock; + typedef Product LinearProduct; + typedef CwiseBinaryOp, + const LinearProduct, + const ConstantBlock> + Xpr; }; - typedef typename Rhs::template ConstNRowsBlockXpr::Type LinearBlockConst; - typedef typename remove_const::type LinearBlock; - typedef typename Rhs::ConstRowXpr ConstantColumn; - typedef Replicate ConstantBlock; - typedef Product LinearProduct; - typedef CwiseBinaryOp, const LinearProduct, const ConstantBlock> Xpr; -}; -template -struct product_evaluator, ProductTag, HomogeneousShape, DenseShape> - : public evaluator::Xpr> -{ - typedef Product XprType; - typedef homogeneous_right_product_refactoring_helper helper; - typedef typename helper::ConstantBlock ConstantBlock; - typedef typename helper::Xpr RefactoredXpr; - typedef evaluator Base; - - EIGEN_DEVICE_FUNC explicit product_evaluator(const XprType& xpr) - : Base( xpr.lhs().nestedExpression() .lazyProduct( xpr.rhs().template topRows(xpr.lhs().nestedExpression().cols()) ) - + ConstantBlock(xpr.rhs().row(xpr.rhs().rows()-1),xpr.lhs().rows(), 1) ) - {} -}; + template + struct product_evaluator, ProductTag, HomogeneousShape, DenseShape> + : public evaluator::Xpr> + { + typedef Product XprType; + typedef homogeneous_right_product_refactoring_helper helper; + typedef typename helper::ConstantBlock ConstantBlock; + typedef typename helper::Xpr RefactoredXpr; + typedef evaluator Base; + + EIGEN_DEVICE_FUNC explicit product_evaluator(const XprType &xpr) + : Base(xpr.lhs().nestedExpression().lazyProduct( + xpr.rhs().template topRows(xpr.lhs().nestedExpression().cols())) + + ConstantBlock(xpr.rhs().row(xpr.rhs().rows() - 1), xpr.lhs().rows(), 1)) + {} + }; -template -struct generic_product_impl, DenseShape, HomogeneousShape, ProductTag> -{ - template - EIGEN_DEVICE_FUNC static void evalTo(Dest& dst, const Lhs& lhs, const Homogeneous& rhs) + template + struct generic_product_impl, DenseShape, HomogeneousShape, ProductTag> { - homogeneous_left_product_impl, Lhs>(lhs, rhs.nestedExpression()).evalTo(dst); - } -}; + template + EIGEN_DEVICE_FUNC static void evalTo(Dest &dst, const Lhs &lhs, const Homogeneous &rhs) + { + homogeneous_left_product_impl, Lhs>(lhs, rhs.nestedExpression()).evalTo(dst); + } + }; -// TODO: the following specialization is to address a regression from 3.2 to 3.3 -// In the future, this path should be optimized. -template -struct generic_product_impl, TriangularShape, HomogeneousShape, ProductTag> -{ - template - static void evalTo(Dest& dst, const Lhs& lhs, const Homogeneous& rhs) + // TODO: the following specialization is to address a regression from 3.2 to 3.3 + // In the future, this path should be optimized. + template + struct generic_product_impl, TriangularShape, HomogeneousShape, ProductTag> { - dst.noalias() = lhs * rhs.eval(); - } -}; + template static void evalTo(Dest &dst, const Lhs &lhs, const Homogeneous &rhs) + { + dst.noalias() = lhs * rhs.eval(); + } + }; -template -struct homogeneous_left_product_refactoring_helper -{ - enum { - Dim = Rhs::RowsAtCompileTime, - Cols = Rhs::ColsAtCompileTime + template struct homogeneous_left_product_refactoring_helper + { + enum { Dim = Rhs::RowsAtCompileTime, Cols = Rhs::ColsAtCompileTime }; + typedef typename Lhs::template ConstNColsBlockXpr::Type LinearBlockConst; + typedef typename remove_const::type LinearBlock; + typedef typename Lhs::ConstColXpr ConstantColumn; + typedef Replicate ConstantBlock; + typedef Product LinearProduct; + typedef CwiseBinaryOp, + const LinearProduct, + const ConstantBlock> + Xpr; }; - typedef typename Lhs::template ConstNColsBlockXpr::Type LinearBlockConst; - typedef typename remove_const::type LinearBlock; - typedef typename Lhs::ConstColXpr ConstantColumn; - typedef Replicate ConstantBlock; - typedef Product LinearProduct; - typedef CwiseBinaryOp, const LinearProduct, const ConstantBlock> Xpr; -}; -template -struct product_evaluator, ProductTag, DenseShape, HomogeneousShape> - : public evaluator::Xpr> -{ - typedef Product XprType; - typedef homogeneous_left_product_refactoring_helper helper; - typedef typename helper::ConstantBlock ConstantBlock; - typedef typename helper::Xpr RefactoredXpr; - typedef evaluator Base; - - EIGEN_DEVICE_FUNC explicit product_evaluator(const XprType& xpr) - : Base( xpr.lhs().template leftCols(xpr.rhs().nestedExpression().rows()) .lazyProduct( xpr.rhs().nestedExpression() ) - + ConstantBlock(xpr.lhs().col(xpr.lhs().cols()-1),1,xpr.rhs().cols()) ) - {} -}; + template + struct product_evaluator, ProductTag, DenseShape, HomogeneousShape> + : public evaluator::Xpr> + { + typedef Product XprType; + typedef homogeneous_left_product_refactoring_helper helper; + typedef typename helper::ConstantBlock ConstantBlock; + typedef typename helper::Xpr RefactoredXpr; + typedef evaluator Base; + + EIGEN_DEVICE_FUNC explicit product_evaluator(const XprType &xpr) + : Base(xpr.lhs() + .template leftCols(xpr.rhs().nestedExpression().rows()) + .lazyProduct(xpr.rhs().nestedExpression()) + + ConstantBlock(xpr.lhs().col(xpr.lhs().cols() - 1), 1, xpr.rhs().cols())) + {} + }; -template -struct generic_product_impl, Homogeneous, DenseShape, HomogeneousShape, ProductTag> -{ - typedef Transform TransformType; - template - EIGEN_DEVICE_FUNC static void evalTo(Dest& dst, const TransformType& lhs, const Homogeneous& rhs) + template + struct generic_product_impl, + Homogeneous, + DenseShape, + HomogeneousShape, + ProductTag> { - homogeneous_left_product_impl, TransformType>(lhs, rhs.nestedExpression()).evalTo(dst); - } -}; + typedef Transform TransformType; + template + EIGEN_DEVICE_FUNC static void evalTo(Dest &dst, const TransformType &lhs, const Homogeneous &rhs) + { + homogeneous_left_product_impl, TransformType>(lhs, rhs.nestedExpression()) + .evalTo(dst); + } + }; -template -struct permutation_matrix_product - : public permutation_matrix_product -{}; + template + struct permutation_matrix_product + : public permutation_matrix_product + { + }; -} // end namespace internal +}// end namespace internal -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_HOMOGENEOUS_H +#endif// EIGEN_HOMOGENEOUS_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Geometry/Hyperplane.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Geometry/Hyperplane.h index 05929b29..121b536f 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Geometry/Hyperplane.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Geometry/Hyperplane.h @@ -11,81 +11,78 @@ #ifndef EIGEN_HYPERPLANE_H #define EIGEN_HYPERPLANE_H -namespace Eigen { +namespace Eigen { /** \geometry_module \ingroup Geometry_Module - * - * \class Hyperplane - * - * \brief A hyperplane - * - * A hyperplane is an affine subspace of dimension n-1 in a space of dimension n. - * For example, a hyperplane in a plane is a line; a hyperplane in 3-space is a plane. - * - * \tparam _Scalar the scalar type, i.e., the type of the coefficients - * \tparam _AmbientDim the dimension of the ambient space, can be a compile time value or Dynamic. - * Notice that the dimension of the hyperplane is _AmbientDim-1. - * - * This class represents an hyperplane as the zero set of the implicit equation - * \f$ n \cdot x + d = 0 \f$ where \f$ n \f$ is a unit normal vector of the plane (linear part) - * and \f$ d \f$ is the distance (offset) to the origin. - */ -template -class Hyperplane + * + * \class Hyperplane + * + * \brief A hyperplane + * + * A hyperplane is an affine subspace of dimension n-1 in a space of dimension n. + * For example, a hyperplane in a plane is a line; a hyperplane in 3-space is a plane. + * + * \tparam _Scalar the scalar type, i.e., the type of the coefficients + * \tparam _AmbientDim the dimension of the ambient space, can be a compile time value or Dynamic. + * Notice that the dimension of the hyperplane is _AmbientDim-1. + * + * This class represents an hyperplane as the zero set of the implicit equation + * \f$ n \cdot x + d = 0 \f$ where \f$ n \f$ is a unit normal vector of the plane (linear part) + * and \f$ d \f$ is the distance (offset) to the origin. + */ +template class Hyperplane { public: - EIGEN_MAKE_ALIGNED_OPERATOR_NEW_IF_VECTORIZABLE_FIXED_SIZE(_Scalar,_AmbientDim==Dynamic ? Dynamic : _AmbientDim+1) - enum { - AmbientDimAtCompileTime = _AmbientDim, - Options = _Options - }; + EIGEN_MAKE_ALIGNED_OPERATOR_NEW_IF_VECTORIZABLE_FIXED_SIZE(_Scalar, + _AmbientDim == Dynamic ? Dynamic : _AmbientDim + 1) + enum { AmbientDimAtCompileTime = _AmbientDim, Options = _Options }; typedef _Scalar Scalar; typedef typename NumTraits::Real RealScalar; - typedef Eigen::Index Index; ///< \deprecated since Eigen 3.3 - typedef Matrix VectorType; - typedef Matrix Coefficients; - typedef Block NormalReturnType; - typedef const Block ConstNormalReturnType; + typedef Eigen::Index Index;///< \deprecated since Eigen 3.3 + typedef Matrix VectorType; + typedef Matrix + Coefficients; + typedef Block NormalReturnType; + typedef const Block ConstNormalReturnType; /** Default constructor without initialization */ EIGEN_DEVICE_FUNC inline Hyperplane() {} - + template - EIGEN_DEVICE_FUNC Hyperplane(const Hyperplane& other) - : m_coeffs(other.coeffs()) + EIGEN_DEVICE_FUNC Hyperplane(const Hyperplane &other) + : m_coeffs(other.coeffs()) {} /** Constructs a dynamic-size hyperplane with \a _dim the dimension - * of the ambient space */ - EIGEN_DEVICE_FUNC inline explicit Hyperplane(Index _dim) : m_coeffs(_dim+1) {} + * of the ambient space */ + EIGEN_DEVICE_FUNC inline explicit Hyperplane(Index _dim) : m_coeffs(_dim + 1) {} /** Construct a plane from its normal \a n and a point \a e onto the plane. - * \warning the vector normal is assumed to be normalized. - */ - EIGEN_DEVICE_FUNC inline Hyperplane(const VectorType& n, const VectorType& e) - : m_coeffs(n.size()+1) + * \warning the vector normal is assumed to be normalized. + */ + EIGEN_DEVICE_FUNC inline Hyperplane(const VectorType &n, const VectorType &e) : m_coeffs(n.size() + 1) { normal() = n; offset() = -n.dot(e); } /** Constructs a plane from its normal \a n and distance to the origin \a d - * such that the algebraic equation of the plane is \f$ n \cdot x + d = 0 \f$. - * \warning the vector normal is assumed to be normalized. - */ - EIGEN_DEVICE_FUNC inline Hyperplane(const VectorType& n, const Scalar& d) - : m_coeffs(n.size()+1) + * such that the algebraic equation of the plane is \f$ n \cdot x + d = 0 \f$. + * \warning the vector normal is assumed to be normalized. + */ + EIGEN_DEVICE_FUNC inline Hyperplane(const VectorType &n, const Scalar &d) : m_coeffs(n.size() + 1) { normal() = n; offset() = d; } /** Constructs a hyperplane passing through the two points. If the dimension of the ambient space - * is greater than 2, then there isn't uniqueness, so an arbitrary choice is made. - */ - EIGEN_DEVICE_FUNC static inline Hyperplane Through(const VectorType& p0, const VectorType& p1) + * is greater than 2, then there isn't uniqueness, so an arbitrary choice is made. + */ + EIGEN_DEVICE_FUNC static inline Hyperplane Through(const VectorType &p0, const VectorType &p1) { Hyperplane result(p0.size()); result.normal() = (p1 - p0).unitOrthogonal(); @@ -94,33 +91,32 @@ class Hyperplane } /** Constructs a hyperplane passing through the three points. The dimension of the ambient space - * is required to be exactly 3. - */ - EIGEN_DEVICE_FUNC static inline Hyperplane Through(const VectorType& p0, const VectorType& p1, const VectorType& p2) + * is required to be exactly 3. + */ + EIGEN_DEVICE_FUNC static inline Hyperplane Through(const VectorType &p0, const VectorType &p1, const VectorType &p2) { EIGEN_STATIC_ASSERT_VECTOR_SPECIFIC_SIZE(VectorType, 3) Hyperplane result(p0.size()); VectorType v0(p2 - p0), v1(p1 - p0); result.normal() = v0.cross(v1); RealScalar norm = result.normal().norm(); - if(norm <= v0.norm() * v1.norm() * NumTraits::epsilon()) - { - Matrix m; m << v0.transpose(), v1.transpose(); - JacobiSVD > svd(m, ComputeFullV); + if (norm <= v0.norm() * v1.norm() * NumTraits::epsilon()) { + Matrix m; + m << v0.transpose(), v1.transpose(); + JacobiSVD> svd(m, ComputeFullV); result.normal() = svd.matrixV().col(2); - } - else + } else result.normal() /= norm; result.offset() = -p0.dot(result.normal()); return result; } /** Constructs a hyperplane passing through the parametrized line \a parametrized. - * If the dimension of the ambient space is greater than 2, then there isn't uniqueness, - * so an arbitrary choice is made. - */ + * If the dimension of the ambient space is greater than 2, then there isn't uniqueness, + * so an arbitrary choice is made. + */ // FIXME to be consitent with the rest this could be implemented as a static Through function ?? - EIGEN_DEVICE_FUNC explicit Hyperplane(const ParametrizedLine& parametrized) + EIGEN_DEVICE_FUNC explicit Hyperplane(const ParametrizedLine ¶metrized) { normal() = parametrized.direction().unitOrthogonal(); offset() = -parametrized.origin().dot(normal()); @@ -129,117 +125,116 @@ class Hyperplane EIGEN_DEVICE_FUNC ~Hyperplane() {} /** \returns the dimension in which the plane holds */ - EIGEN_DEVICE_FUNC inline Index dim() const { return AmbientDimAtCompileTime==Dynamic ? m_coeffs.size()-1 : Index(AmbientDimAtCompileTime); } - - /** normalizes \c *this */ - EIGEN_DEVICE_FUNC void normalize(void) + EIGEN_DEVICE_FUNC inline Index dim() const { - m_coeffs /= normal().norm(); + return AmbientDimAtCompileTime == Dynamic ? m_coeffs.size() - 1 : Index(AmbientDimAtCompileTime); } + /** normalizes \c *this */ + EIGEN_DEVICE_FUNC void normalize(void) { m_coeffs /= normal().norm(); } + /** \returns the signed distance between the plane \c *this and a point \a p. - * \sa absDistance() - */ - EIGEN_DEVICE_FUNC inline Scalar signedDistance(const VectorType& p) const { return normal().dot(p) + offset(); } + * \sa absDistance() + */ + EIGEN_DEVICE_FUNC inline Scalar signedDistance(const VectorType &p) const { return normal().dot(p) + offset(); } /** \returns the absolute distance between the plane \c *this and a point \a p. - * \sa signedDistance() - */ - EIGEN_DEVICE_FUNC inline Scalar absDistance(const VectorType& p) const { return numext::abs(signedDistance(p)); } + * \sa signedDistance() + */ + EIGEN_DEVICE_FUNC inline Scalar absDistance(const VectorType &p) const { return numext::abs(signedDistance(p)); } /** \returns the projection of a point \a p onto the plane \c *this. - */ - EIGEN_DEVICE_FUNC inline VectorType projection(const VectorType& p) const { return p - signedDistance(p) * normal(); } + */ + EIGEN_DEVICE_FUNC inline VectorType projection(const VectorType &p) const { return p - signedDistance(p) * normal(); } /** \returns a constant reference to the unit normal vector of the plane, which corresponds - * to the linear part of the implicit equation. - */ - EIGEN_DEVICE_FUNC inline ConstNormalReturnType normal() const { return ConstNormalReturnType(m_coeffs,0,0,dim(),1); } + * to the linear part of the implicit equation. + */ + EIGEN_DEVICE_FUNC inline ConstNormalReturnType normal() const + { + return ConstNormalReturnType(m_coeffs, 0, 0, dim(), 1); + } /** \returns a non-constant reference to the unit normal vector of the plane, which corresponds - * to the linear part of the implicit equation. - */ - EIGEN_DEVICE_FUNC inline NormalReturnType normal() { return NormalReturnType(m_coeffs,0,0,dim(),1); } + * to the linear part of the implicit equation. + */ + EIGEN_DEVICE_FUNC inline NormalReturnType normal() { return NormalReturnType(m_coeffs, 0, 0, dim(), 1); } /** \returns the distance to the origin, which is also the "constant term" of the implicit equation - * \warning the vector normal is assumed to be normalized. - */ - EIGEN_DEVICE_FUNC inline const Scalar& offset() const { return m_coeffs.coeff(dim()); } + * \warning the vector normal is assumed to be normalized. + */ + EIGEN_DEVICE_FUNC inline const Scalar &offset() const { return m_coeffs.coeff(dim()); } /** \returns a non-constant reference to the distance to the origin, which is also the constant part - * of the implicit equation */ - EIGEN_DEVICE_FUNC inline Scalar& offset() { return m_coeffs(dim()); } + * of the implicit equation */ + EIGEN_DEVICE_FUNC inline Scalar &offset() { return m_coeffs(dim()); } /** \returns a constant reference to the coefficients c_i of the plane equation: - * \f$ c_0*x_0 + ... + c_{d-1}*x_{d-1} + c_d = 0 \f$ - */ - EIGEN_DEVICE_FUNC inline const Coefficients& coeffs() const { return m_coeffs; } + * \f$ c_0*x_0 + ... + c_{d-1}*x_{d-1} + c_d = 0 \f$ + */ + EIGEN_DEVICE_FUNC inline const Coefficients &coeffs() const { return m_coeffs; } /** \returns a non-constant reference to the coefficients c_i of the plane equation: - * \f$ c_0*x_0 + ... + c_{d-1}*x_{d-1} + c_d = 0 \f$ - */ - EIGEN_DEVICE_FUNC inline Coefficients& coeffs() { return m_coeffs; } + * \f$ c_0*x_0 + ... + c_{d-1}*x_{d-1} + c_d = 0 \f$ + */ + EIGEN_DEVICE_FUNC inline Coefficients &coeffs() { return m_coeffs; } /** \returns the intersection of *this with \a other. - * - * \warning The ambient space must be a plane, i.e. have dimension 2, so that \c *this and \a other are lines. - * - * \note If \a other is approximately parallel to *this, this method will return any point on *this. - */ - EIGEN_DEVICE_FUNC VectorType intersection(const Hyperplane& other) const + * + * \warning The ambient space must be a plane, i.e. have dimension 2, so that \c *this and \a other are lines. + * + * \note If \a other is approximately parallel to *this, this method will return any point on *this. + */ + EIGEN_DEVICE_FUNC VectorType intersection(const Hyperplane &other) const { EIGEN_STATIC_ASSERT_VECTOR_SPECIFIC_SIZE(VectorType, 2) Scalar det = coeffs().coeff(0) * other.coeffs().coeff(1) - coeffs().coeff(1) * other.coeffs().coeff(0); // since the line equations ax+by=c are normalized with a^2+b^2=1, the following tests // whether the two lines are approximately parallel. - if(internal::isMuchSmallerThan(det, Scalar(1))) - { // special case where the two lines are approximately parallel. Pick any point on the first line. - if(numext::abs(coeffs().coeff(1))>numext::abs(coeffs().coeff(0))) - return VectorType(coeffs().coeff(1), -coeffs().coeff(2)/coeffs().coeff(1)-coeffs().coeff(0)); - else - return VectorType(-coeffs().coeff(2)/coeffs().coeff(0)-coeffs().coeff(1), coeffs().coeff(0)); - } - else - { // general case - Scalar invdet = Scalar(1) / det; - return VectorType(invdet*(coeffs().coeff(1)*other.coeffs().coeff(2)-other.coeffs().coeff(1)*coeffs().coeff(2)), - invdet*(other.coeffs().coeff(0)*coeffs().coeff(2)-coeffs().coeff(0)*other.coeffs().coeff(2))); + if (internal::isMuchSmallerThan(det, + Scalar(1))) {// special case where the two lines are approximately parallel. Pick any point on the first line. + if (numext::abs(coeffs().coeff(1)) > numext::abs(coeffs().coeff(0))) + return VectorType(coeffs().coeff(1), -coeffs().coeff(2) / coeffs().coeff(1) - coeffs().coeff(0)); + else + return VectorType(-coeffs().coeff(2) / coeffs().coeff(0) - coeffs().coeff(1), coeffs().coeff(0)); + } else {// general case + Scalar invdet = Scalar(1) / det; + return VectorType( + invdet * (coeffs().coeff(1) * other.coeffs().coeff(2) - other.coeffs().coeff(1) * coeffs().coeff(2)), + invdet * (other.coeffs().coeff(0) * coeffs().coeff(2) - coeffs().coeff(0) * other.coeffs().coeff(2))); } } /** Applies the transformation matrix \a mat to \c *this and returns a reference to \c *this. - * - * \param mat the Dim x Dim transformation matrix - * \param traits specifies whether the matrix \a mat represents an #Isometry - * or a more generic #Affine transformation. The default is #Affine. - */ + * + * \param mat the Dim x Dim transformation matrix + * \param traits specifies whether the matrix \a mat represents an #Isometry + * or a more generic #Affine transformation. The default is #Affine. + */ template - EIGEN_DEVICE_FUNC inline Hyperplane& transform(const MatrixBase& mat, TransformTraits traits = Affine) + EIGEN_DEVICE_FUNC inline Hyperplane &transform(const MatrixBase &mat, TransformTraits traits = Affine) { - if (traits==Affine) - { + if (traits == Affine) { normal() = mat.inverse().transpose() * normal(); m_coeffs /= normal().norm(); - } - else if (traits==Isometry) + } else if (traits == Isometry) normal() = mat * normal(); - else - { + else { eigen_assert(0 && "invalid traits value in Hyperplane::transform()"); } return *this; } /** Applies the transformation \a t to \c *this and returns a reference to \c *this. - * - * \param t the transformation of dimension Dim - * \param traits specifies whether the transformation \a t represents an #Isometry - * or a more generic #Affine transformation. The default is #Affine. - * Other kind of transformations are not supported. - */ + * + * \param t the transformation of dimension Dim + * \param traits specifies whether the transformation \a t represents an #Isometry + * or a more generic #Affine transformation. The default is #Affine. + * Other kind of transformations are not supported. + */ template - EIGEN_DEVICE_FUNC inline Hyperplane& transform(const Transform& t, - TransformTraits traits = Affine) + EIGEN_DEVICE_FUNC inline Hyperplane &transform(const Transform &t, + TransformTraits traits = Affine) { transform(t.linear(), traits); offset() -= normal().dot(t.translation()); @@ -247,36 +242,42 @@ class Hyperplane } /** \returns \c *this with scalar type casted to \a NewScalarType - * - * Note that if \a NewScalarType is equal to the current scalar type of \c *this - * then this function smartly returns a const reference to \c *this. - */ + * + * Note that if \a NewScalarType is equal to the current scalar type of \c *this + * then this function smartly returns a const reference to \c *this. + */ template - EIGEN_DEVICE_FUNC inline typename internal::cast_return_type >::type cast() const + EIGEN_DEVICE_FUNC inline + typename internal::cast_return_type>::type + cast() const { return typename internal::cast_return_type >::type(*this); + Hyperplane>::type(*this); } /** Copy constructor with scalar type conversion */ - template - EIGEN_DEVICE_FUNC inline explicit Hyperplane(const Hyperplane& other) - { m_coeffs = other.coeffs().template cast(); } + template + EIGEN_DEVICE_FUNC inline explicit Hyperplane( + const Hyperplane &other) + { + m_coeffs = other.coeffs().template cast(); + } /** \returns \c true if \c *this is approximately equal to \a other, within the precision - * determined by \a prec. - * - * \sa MatrixBase::isApprox() */ + * determined by \a prec. + * + * \sa MatrixBase::isApprox() */ template - EIGEN_DEVICE_FUNC bool isApprox(const Hyperplane& other, const typename NumTraits::Real& prec = NumTraits::dummy_precision()) const - { return m_coeffs.isApprox(other.m_coeffs, prec); } + EIGEN_DEVICE_FUNC bool isApprox(const Hyperplane &other, + const typename NumTraits::Real &prec = NumTraits::dummy_precision()) const + { + return m_coeffs.isApprox(other.m_coeffs, prec); + } protected: - Coefficients m_coeffs; }; -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_HYPERPLANE_H +#endif// EIGEN_HYPERPLANE_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Geometry/OrthoMethods.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Geometry/OrthoMethods.h index a035e631..941866f2 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Geometry/OrthoMethods.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Geometry/OrthoMethods.h @@ -11,19 +11,20 @@ #ifndef EIGEN_ORTHOMETHODS_H #define EIGEN_ORTHOMETHODS_H -namespace Eigen { +namespace Eigen { /** \geometry_module \ingroup Geometry_Module - * - * \returns the cross product of \c *this and \a other - * - * Here is a very good explanation of cross-product: http://xkcd.com/199/ - * - * With complex numbers, the cross product is implemented as - * \f$ (\mathbf{a}+i\mathbf{b}) \times (\mathbf{c}+i\mathbf{d}) = (\mathbf{a} \times \mathbf{c} - \mathbf{b} \times \mathbf{d}) - i(\mathbf{a} \times \mathbf{d} - \mathbf{b} \times \mathbf{c})\f$ - * - * \sa MatrixBase::cross3() - */ + * + * \returns the cross product of \c *this and \a other + * + * Here is a very good explanation of cross-product: http://xkcd.com/199/ + * + * With complex numbers, the cross product is implemented as + * \f$ (\mathbf{a}+i\mathbf{b}) \times (\mathbf{c}+i\mathbf{d}) = (\mathbf{a} \times \mathbf{c} - \mathbf{b} \times + * \mathbf{d}) - i(\mathbf{a} \times \mathbf{d} - \mathbf{b} \times \mathbf{c})\f$ + * + * \sa MatrixBase::cross3() + */ template template #ifndef EIGEN_PARSED_BY_DOXYGEN @@ -31,102 +32,99 @@ EIGEN_DEVICE_FUNC inline typename MatrixBase::template cross_product_re #else inline typename MatrixBase::PlainObject #endif -MatrixBase::cross(const MatrixBase& other) const + MatrixBase::cross(const MatrixBase &other) const { - EIGEN_STATIC_ASSERT_VECTOR_SPECIFIC_SIZE(Derived,3) - EIGEN_STATIC_ASSERT_VECTOR_SPECIFIC_SIZE(OtherDerived,3) + EIGEN_STATIC_ASSERT_VECTOR_SPECIFIC_SIZE(Derived, 3) + EIGEN_STATIC_ASSERT_VECTOR_SPECIFIC_SIZE(OtherDerived, 3) // Note that there is no need for an expression here since the compiler // optimize such a small temporary very well (even within a complex expression) - typename internal::nested_eval::type lhs(derived()); - typename internal::nested_eval::type rhs(other.derived()); + typename internal::nested_eval::type lhs(derived()); + typename internal::nested_eval::type rhs(other.derived()); return typename cross_product_return_type::type( numext::conj(lhs.coeff(1) * rhs.coeff(2) - lhs.coeff(2) * rhs.coeff(1)), numext::conj(lhs.coeff(2) * rhs.coeff(0) - lhs.coeff(0) * rhs.coeff(2)), - numext::conj(lhs.coeff(0) * rhs.coeff(1) - lhs.coeff(1) * rhs.coeff(0)) - ); + numext::conj(lhs.coeff(0) * rhs.coeff(1) - lhs.coeff(1) * rhs.coeff(0))); } namespace internal { -template< int Arch,typename VectorLhs,typename VectorRhs, - typename Scalar = typename VectorLhs::Scalar, - bool Vectorizable = bool((VectorLhs::Flags&VectorRhs::Flags)&PacketAccessBit)> -struct cross3_impl { - EIGEN_DEVICE_FUNC static inline typename internal::plain_matrix_type::type - run(const VectorLhs& lhs, const VectorRhs& rhs) + template + struct cross3_impl { - return typename internal::plain_matrix_type::type( - numext::conj(lhs.coeff(1) * rhs.coeff(2) - lhs.coeff(2) * rhs.coeff(1)), - numext::conj(lhs.coeff(2) * rhs.coeff(0) - lhs.coeff(0) * rhs.coeff(2)), - numext::conj(lhs.coeff(0) * rhs.coeff(1) - lhs.coeff(1) * rhs.coeff(0)), - 0 - ); - } -}; + EIGEN_DEVICE_FUNC static inline typename internal::plain_matrix_type::type run(const VectorLhs &lhs, + const VectorRhs &rhs) + { + return typename internal::plain_matrix_type::type( + numext::conj(lhs.coeff(1) * rhs.coeff(2) - lhs.coeff(2) * rhs.coeff(1)), + numext::conj(lhs.coeff(2) * rhs.coeff(0) - lhs.coeff(0) * rhs.coeff(2)), + numext::conj(lhs.coeff(0) * rhs.coeff(1) - lhs.coeff(1) * rhs.coeff(0)), + 0); + } + }; -} +}// namespace internal /** \geometry_module \ingroup Geometry_Module - * - * \returns the cross product of \c *this and \a other using only the x, y, and z coefficients - * - * The size of \c *this and \a other must be four. This function is especially useful - * when using 4D vectors instead of 3D ones to get advantage of SSE/AltiVec vectorization. - * - * \sa MatrixBase::cross() - */ + * + * \returns the cross product of \c *this and \a other using only the x, y, and z coefficients + * + * The size of \c *this and \a other must be four. This function is especially useful + * when using 4D vectors instead of 3D ones to get advantage of SSE/AltiVec vectorization. + * + * \sa MatrixBase::cross() + */ template template -EIGEN_DEVICE_FUNC inline typename MatrixBase::PlainObject -MatrixBase::cross3(const MatrixBase& other) const +EIGEN_DEVICE_FUNC inline typename MatrixBase::PlainObject MatrixBase::cross3( + const MatrixBase &other) const { - EIGEN_STATIC_ASSERT_VECTOR_SPECIFIC_SIZE(Derived,4) - EIGEN_STATIC_ASSERT_VECTOR_SPECIFIC_SIZE(OtherDerived,4) + EIGEN_STATIC_ASSERT_VECTOR_SPECIFIC_SIZE(Derived, 4) + EIGEN_STATIC_ASSERT_VECTOR_SPECIFIC_SIZE(OtherDerived, 4) - typedef typename internal::nested_eval::type DerivedNested; - typedef typename internal::nested_eval::type OtherDerivedNested; + typedef typename internal::nested_eval::type DerivedNested; + typedef typename internal::nested_eval::type OtherDerivedNested; DerivedNested lhs(derived()); OtherDerivedNested rhs(other.derived()); return internal::cross3_impl::type, - typename internal::remove_all::type>::run(lhs,rhs); + typename internal::remove_all::type, + typename internal::remove_all::type>::run(lhs, rhs); } /** \geometry_module \ingroup Geometry_Module - * - * \returns a matrix expression of the cross product of each column or row - * of the referenced expression with the \a other vector. - * - * The referenced matrix must have one dimension equal to 3. - * The result matrix has the same dimensions than the referenced one. - * - * \sa MatrixBase::cross() */ + * + * \returns a matrix expression of the cross product of each column or row + * of the referenced expression with the \a other vector. + * + * The referenced matrix must have one dimension equal to 3. + * The result matrix has the same dimensions than the referenced one. + * + * \sa MatrixBase::cross() */ template template -EIGEN_DEVICE_FUNC -const typename VectorwiseOp::CrossReturnType -VectorwiseOp::cross(const MatrixBase& other) const +EIGEN_DEVICE_FUNC const typename VectorwiseOp::CrossReturnType + VectorwiseOp::cross(const MatrixBase &other) const { - EIGEN_STATIC_ASSERT_VECTOR_SPECIFIC_SIZE(OtherDerived,3) + EIGEN_STATIC_ASSERT_VECTOR_SPECIFIC_SIZE(OtherDerived, 3) EIGEN_STATIC_ASSERT((internal::is_same::value), YOU_MIXED_DIFFERENT_NUMERIC_TYPES__YOU_NEED_TO_USE_THE_CAST_METHOD_OF_MATRIXBASE_TO_CAST_NUMERIC_TYPES_EXPLICITLY) - - typename internal::nested_eval::type mat(_expression()); - typename internal::nested_eval::type vec(other.derived()); - CrossReturnType res(_expression().rows(),_expression().cols()); - if(Direction==Vertical) - { - eigen_assert(CrossReturnType::RowsAtCompileTime==3 && "the matrix must have exactly 3 rows"); + typename internal::nested_eval::type mat(_expression()); + typename internal::nested_eval::type vec(other.derived()); + + CrossReturnType res(_expression().rows(), _expression().cols()); + if (Direction == Vertical) { + eigen_assert(CrossReturnType::RowsAtCompileTime == 3 && "the matrix must have exactly 3 rows"); res.row(0) = (mat.row(1) * vec.coeff(2) - mat.row(2) * vec.coeff(1)).conjugate(); res.row(1) = (mat.row(2) * vec.coeff(0) - mat.row(0) * vec.coeff(2)).conjugate(); res.row(2) = (mat.row(0) * vec.coeff(1) - mat.row(1) * vec.coeff(0)).conjugate(); - } - else - { - eigen_assert(CrossReturnType::ColsAtCompileTime==3 && "the matrix must have exactly 3 columns"); + } else { + eigen_assert(CrossReturnType::ColsAtCompileTime == 3 && "the matrix must have exactly 3 columns"); res.col(0) = (mat.col(1) * vec.coeff(2) - mat.col(2) * vec.coeff(1)).conjugate(); res.col(1) = (mat.col(2) * vec.coeff(0) - mat.col(0) * vec.coeff(2)).conjugate(); res.col(2) = (mat.col(0) * vec.coeff(1) - mat.col(1) * vec.coeff(0)).conjugate(); @@ -136,99 +134,93 @@ VectorwiseOp::cross(const MatrixBase& ot namespace internal { -template -struct unitOrthogonal_selector -{ - typedef typename plain_matrix_type::type VectorType; - typedef typename traits::Scalar Scalar; - typedef typename NumTraits::Real RealScalar; - typedef Matrix Vector2; - EIGEN_DEVICE_FUNC - static inline VectorType run(const Derived& src) + template struct unitOrthogonal_selector { - VectorType perp = VectorType::Zero(src.size()); - Index maxi = 0; - Index sndi = 0; - src.cwiseAbs().maxCoeff(&maxi); - if (maxi==0) - sndi = 1; - RealScalar invnm = RealScalar(1)/(Vector2() << src.coeff(sndi),src.coeff(maxi)).finished().norm(); - perp.coeffRef(maxi) = -numext::conj(src.coeff(sndi)) * invnm; - perp.coeffRef(sndi) = numext::conj(src.coeff(maxi)) * invnm; - - return perp; - } -}; + typedef typename plain_matrix_type::type VectorType; + typedef typename traits::Scalar Scalar; + typedef typename NumTraits::Real RealScalar; + typedef Matrix Vector2; + EIGEN_DEVICE_FUNC + static inline VectorType run(const Derived &src) + { + VectorType perp = VectorType::Zero(src.size()); + Index maxi = 0; + Index sndi = 0; + src.cwiseAbs().maxCoeff(&maxi); + if (maxi == 0) sndi = 1; + RealScalar invnm = RealScalar(1) / (Vector2() << src.coeff(sndi), src.coeff(maxi)).finished().norm(); + perp.coeffRef(maxi) = -numext::conj(src.coeff(sndi)) * invnm; + perp.coeffRef(sndi) = numext::conj(src.coeff(maxi)) * invnm; + + return perp; + } + }; -template -struct unitOrthogonal_selector -{ - typedef typename plain_matrix_type::type VectorType; - typedef typename traits::Scalar Scalar; - typedef typename NumTraits::Real RealScalar; - EIGEN_DEVICE_FUNC - static inline VectorType run(const Derived& src) + template struct unitOrthogonal_selector { - VectorType perp; - /* Let us compute the crossed product of *this with a vector - * that is not too close to being colinear to *this. - */ - - /* unless the x and y coords are both close to zero, we can - * simply take ( -y, x, 0 ) and normalize it. - */ - if((!isMuchSmallerThan(src.x(), src.z())) - || (!isMuchSmallerThan(src.y(), src.z()))) + typedef typename plain_matrix_type::type VectorType; + typedef typename traits::Scalar Scalar; + typedef typename NumTraits::Real RealScalar; + EIGEN_DEVICE_FUNC + static inline VectorType run(const Derived &src) { - RealScalar invnm = RealScalar(1)/src.template head<2>().norm(); - perp.coeffRef(0) = -numext::conj(src.y())*invnm; - perp.coeffRef(1) = numext::conj(src.x())*invnm; - perp.coeffRef(2) = 0; + VectorType perp; + /* Let us compute the crossed product of *this with a vector + * that is not too close to being colinear to *this. + */ + + /* unless the x and y coords are both close to zero, we can + * simply take ( -y, x, 0 ) and normalize it. + */ + if ((!isMuchSmallerThan(src.x(), src.z())) || (!isMuchSmallerThan(src.y(), src.z()))) { + RealScalar invnm = RealScalar(1) / src.template head<2>().norm(); + perp.coeffRef(0) = -numext::conj(src.y()) * invnm; + perp.coeffRef(1) = numext::conj(src.x()) * invnm; + perp.coeffRef(2) = 0; + } + /* if both x and y are close to zero, then the vector is close + * to the z-axis, so it's far from colinear to the x-axis for instance. + * So we take the crossed product with (1,0,0) and normalize it. + */ + else { + RealScalar invnm = RealScalar(1) / src.template tail<2>().norm(); + perp.coeffRef(0) = 0; + perp.coeffRef(1) = -numext::conj(src.z()) * invnm; + perp.coeffRef(2) = numext::conj(src.y()) * invnm; + } + + return perp; } - /* if both x and y are close to zero, then the vector is close - * to the z-axis, so it's far from colinear to the x-axis for instance. - * So we take the crossed product with (1,0,0) and normalize it. - */ - else + }; + + template struct unitOrthogonal_selector + { + typedef typename plain_matrix_type::type VectorType; + EIGEN_DEVICE_FUNC + static inline VectorType run(const Derived &src) { - RealScalar invnm = RealScalar(1)/src.template tail<2>().norm(); - perp.coeffRef(0) = 0; - perp.coeffRef(1) = -numext::conj(src.z())*invnm; - perp.coeffRef(2) = numext::conj(src.y())*invnm; + return VectorType(-numext::conj(src.y()), numext::conj(src.x())).normalized(); } + }; - return perp; - } -}; - -template -struct unitOrthogonal_selector -{ - typedef typename plain_matrix_type::type VectorType; - EIGEN_DEVICE_FUNC - static inline VectorType run(const Derived& src) - { return VectorType(-numext::conj(src.y()), numext::conj(src.x())).normalized(); } -}; - -} // end namespace internal +}// end namespace internal /** \geometry_module \ingroup Geometry_Module - * - * \returns a unit vector which is orthogonal to \c *this - * - * The size of \c *this must be at least 2. If the size is exactly 2, - * then the returned vector is a counter clock wise rotation of \c *this, i.e., (-y,x).normalized(). - * - * \sa cross() - */ + * + * \returns a unit vector which is orthogonal to \c *this + * + * The size of \c *this must be at least 2. If the size is exactly 2, + * then the returned vector is a counter clock wise rotation of \c *this, i.e., (-y,x).normalized(). + * + * \sa cross() + */ template -EIGEN_DEVICE_FUNC typename MatrixBase::PlainObject -MatrixBase::unitOrthogonal() const +EIGEN_DEVICE_FUNC typename MatrixBase::PlainObject MatrixBase::unitOrthogonal() const { EIGEN_STATIC_ASSERT_VECTOR_ONLY(Derived) return internal::unitOrthogonal_selector::run(derived()); } -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_ORTHOMETHODS_H +#endif// EIGEN_ORTHOMETHODS_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Geometry/ParametrizedLine.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Geometry/ParametrizedLine.h index 1e985d8c..8fd04e0e 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Geometry/ParametrizedLine.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Geometry/ParametrizedLine.h @@ -11,185 +11,198 @@ #ifndef EIGEN_PARAMETRIZEDLINE_H #define EIGEN_PARAMETRIZEDLINE_H -namespace Eigen { +namespace Eigen { /** \geometry_module \ingroup Geometry_Module - * - * \class ParametrizedLine - * - * \brief A parametrized line - * - * A parametrized line is defined by an origin point \f$ \mathbf{o} \f$ and a unit - * direction vector \f$ \mathbf{d} \f$ such that the line corresponds to - * the set \f$ l(t) = \mathbf{o} + t \mathbf{d} \f$, \f$ t \in \mathbf{R} \f$. - * - * \tparam _Scalar the scalar type, i.e., the type of the coefficients - * \tparam _AmbientDim the dimension of the ambient space, can be a compile time value or Dynamic. - */ -template -class ParametrizedLine + * + * \class ParametrizedLine + * + * \brief A parametrized line + * + * A parametrized line is defined by an origin point \f$ \mathbf{o} \f$ and a unit + * direction vector \f$ \mathbf{d} \f$ such that the line corresponds to + * the set \f$ l(t) = \mathbf{o} + t \mathbf{d} \f$, \f$ t \in \mathbf{R} \f$. + * + * \tparam _Scalar the scalar type, i.e., the type of the coefficients + * \tparam _AmbientDim the dimension of the ambient space, can be a compile time value or Dynamic. + */ +template class ParametrizedLine { public: - EIGEN_MAKE_ALIGNED_OPERATOR_NEW_IF_VECTORIZABLE_FIXED_SIZE(_Scalar,_AmbientDim) - enum { - AmbientDimAtCompileTime = _AmbientDim, - Options = _Options - }; + EIGEN_MAKE_ALIGNED_OPERATOR_NEW_IF_VECTORIZABLE_FIXED_SIZE(_Scalar, _AmbientDim) + enum { AmbientDimAtCompileTime = _AmbientDim, Options = _Options }; typedef _Scalar Scalar; typedef typename NumTraits::Real RealScalar; - typedef Eigen::Index Index; ///< \deprecated since Eigen 3.3 - typedef Matrix VectorType; + typedef Eigen::Index Index;///< \deprecated since Eigen 3.3 + typedef Matrix VectorType; /** Default constructor without initialization */ EIGEN_DEVICE_FUNC inline ParametrizedLine() {} - + template - EIGEN_DEVICE_FUNC ParametrizedLine(const ParametrizedLine& other) - : m_origin(other.origin()), m_direction(other.direction()) + EIGEN_DEVICE_FUNC ParametrizedLine(const ParametrizedLine &other) + : m_origin(other.origin()), m_direction(other.direction()) {} /** Constructs a dynamic-size line with \a _dim the dimension - * of the ambient space */ + * of the ambient space */ EIGEN_DEVICE_FUNC inline explicit ParametrizedLine(Index _dim) : m_origin(_dim), m_direction(_dim) {} /** Initializes a parametrized line of direction \a direction and origin \a origin. - * \warning the vector direction is assumed to be normalized. - */ - EIGEN_DEVICE_FUNC ParametrizedLine(const VectorType& origin, const VectorType& direction) - : m_origin(origin), m_direction(direction) {} + * \warning the vector direction is assumed to be normalized. + */ + EIGEN_DEVICE_FUNC ParametrizedLine(const VectorType &origin, const VectorType &direction) + : m_origin(origin), m_direction(direction) + {} - template - EIGEN_DEVICE_FUNC explicit ParametrizedLine(const Hyperplane<_Scalar, _AmbientDim, OtherOptions>& hyperplane); + template + EIGEN_DEVICE_FUNC explicit ParametrizedLine(const Hyperplane<_Scalar, _AmbientDim, OtherOptions> &hyperplane); /** Constructs a parametrized line going from \a p0 to \a p1. */ - EIGEN_DEVICE_FUNC static inline ParametrizedLine Through(const VectorType& p0, const VectorType& p1) - { return ParametrizedLine(p0, (p1-p0).normalized()); } + EIGEN_DEVICE_FUNC static inline ParametrizedLine Through(const VectorType &p0, const VectorType &p1) + { + return ParametrizedLine(p0, (p1 - p0).normalized()); + } EIGEN_DEVICE_FUNC ~ParametrizedLine() {} /** \returns the dimension in which the line holds */ EIGEN_DEVICE_FUNC inline Index dim() const { return m_direction.size(); } - EIGEN_DEVICE_FUNC const VectorType& origin() const { return m_origin; } - EIGEN_DEVICE_FUNC VectorType& origin() { return m_origin; } + EIGEN_DEVICE_FUNC const VectorType &origin() const { return m_origin; } + EIGEN_DEVICE_FUNC VectorType &origin() { return m_origin; } - EIGEN_DEVICE_FUNC const VectorType& direction() const { return m_direction; } - EIGEN_DEVICE_FUNC VectorType& direction() { return m_direction; } + EIGEN_DEVICE_FUNC const VectorType &direction() const { return m_direction; } + EIGEN_DEVICE_FUNC VectorType &direction() { return m_direction; } /** \returns the squared distance of a point \a p to its projection onto the line \c *this. - * \sa distance() - */ - EIGEN_DEVICE_FUNC RealScalar squaredDistance(const VectorType& p) const + * \sa distance() + */ + EIGEN_DEVICE_FUNC RealScalar squaredDistance(const VectorType &p) const { VectorType diff = p - origin(); return (diff - direction().dot(diff) * direction()).squaredNorm(); } /** \returns the distance of a point \a p to its projection onto the line \c *this. - * \sa squaredDistance() - */ - EIGEN_DEVICE_FUNC RealScalar distance(const VectorType& p) const { EIGEN_USING_STD_MATH(sqrt) return sqrt(squaredDistance(p)); } + * \sa squaredDistance() + */ + EIGEN_DEVICE_FUNC RealScalar distance(const VectorType &p) const + { + EIGEN_USING_STD_MATH(sqrt) return sqrt(squaredDistance(p)); + } /** \returns the projection of a point \a p onto the line \c *this. */ - EIGEN_DEVICE_FUNC VectorType projection(const VectorType& p) const - { return origin() + direction().dot(p-origin()) * direction(); } - - EIGEN_DEVICE_FUNC VectorType pointAt(const Scalar& t) const; - - template - EIGEN_DEVICE_FUNC Scalar intersectionParameter(const Hyperplane<_Scalar, _AmbientDim, OtherOptions>& hyperplane) const; - - template - EIGEN_DEVICE_FUNC Scalar intersection(const Hyperplane<_Scalar, _AmbientDim, OtherOptions>& hyperplane) const; - - template - EIGEN_DEVICE_FUNC VectorType intersectionPoint(const Hyperplane<_Scalar, _AmbientDim, OtherOptions>& hyperplane) const; + EIGEN_DEVICE_FUNC VectorType projection(const VectorType &p) const + { + return origin() + direction().dot(p - origin()) * direction(); + } + + EIGEN_DEVICE_FUNC VectorType pointAt(const Scalar &t) const; + + template + EIGEN_DEVICE_FUNC Scalar intersectionParameter( + const Hyperplane<_Scalar, _AmbientDim, OtherOptions> &hyperplane) const; + + template + EIGEN_DEVICE_FUNC Scalar intersection(const Hyperplane<_Scalar, _AmbientDim, OtherOptions> &hyperplane) const; + + template + EIGEN_DEVICE_FUNC VectorType intersectionPoint( + const Hyperplane<_Scalar, _AmbientDim, OtherOptions> &hyperplane) const; /** \returns \c *this with scalar type casted to \a NewScalarType - * - * Note that if \a NewScalarType is equal to the current scalar type of \c *this - * then this function smartly returns a const reference to \c *this. - */ + * + * Note that if \a NewScalarType is equal to the current scalar type of \c *this + * then this function smartly returns a const reference to \c *this. + */ template EIGEN_DEVICE_FUNC inline typename internal::cast_return_type >::type cast() const + ParametrizedLine>::type + cast() const { return typename internal::cast_return_type >::type(*this); + ParametrizedLine>::type(*this); } /** Copy constructor with scalar type conversion */ - template - EIGEN_DEVICE_FUNC inline explicit ParametrizedLine(const ParametrizedLine& other) + template + EIGEN_DEVICE_FUNC inline explicit ParametrizedLine( + const ParametrizedLine &other) { m_origin = other.origin().template cast(); m_direction = other.direction().template cast(); } /** \returns \c true if \c *this is approximately equal to \a other, within the precision - * determined by \a prec. - * - * \sa MatrixBase::isApprox() */ - EIGEN_DEVICE_FUNC bool isApprox(const ParametrizedLine& other, const typename NumTraits::Real& prec = NumTraits::dummy_precision()) const - { return m_origin.isApprox(other.m_origin, prec) && m_direction.isApprox(other.m_direction, prec); } + * determined by \a prec. + * + * \sa MatrixBase::isApprox() */ + EIGEN_DEVICE_FUNC bool isApprox(const ParametrizedLine &other, + const typename NumTraits::Real &prec = NumTraits::dummy_precision()) const + { + return m_origin.isApprox(other.m_origin, prec) && m_direction.isApprox(other.m_direction, prec); + } protected: - VectorType m_origin, m_direction; }; /** Constructs a parametrized line from a 2D hyperplane - * - * \warning the ambient space must have dimension 2 such that the hyperplane actually describes a line - */ -template -template -EIGEN_DEVICE_FUNC inline ParametrizedLine<_Scalar, _AmbientDim,_Options>::ParametrizedLine(const Hyperplane<_Scalar, _AmbientDim,OtherOptions>& hyperplane) + * + * \warning the ambient space must have dimension 2 such that the hyperplane actually describes a line + */ +template +template +EIGEN_DEVICE_FUNC inline ParametrizedLine<_Scalar, _AmbientDim, _Options>::ParametrizedLine( + const Hyperplane<_Scalar, _AmbientDim, OtherOptions> &hyperplane) { EIGEN_STATIC_ASSERT_VECTOR_SPECIFIC_SIZE(VectorType, 2) direction() = hyperplane.normal().unitOrthogonal(); - origin() = -hyperplane.normal()*hyperplane.offset(); + origin() = -hyperplane.normal() * hyperplane.offset(); } /** \returns the point at \a t along this line - */ -template -EIGEN_DEVICE_FUNC inline typename ParametrizedLine<_Scalar, _AmbientDim,_Options>::VectorType -ParametrizedLine<_Scalar, _AmbientDim,_Options>::pointAt(const _Scalar& t) const + */ +template +EIGEN_DEVICE_FUNC inline typename ParametrizedLine<_Scalar, _AmbientDim, _Options>::VectorType + ParametrizedLine<_Scalar, _AmbientDim, _Options>::pointAt(const _Scalar &t) const { - return origin() + (direction()*t); + return origin() + (direction() * t); } /** \returns the parameter value of the intersection between \c *this and the given \a hyperplane - */ -template -template -EIGEN_DEVICE_FUNC inline _Scalar ParametrizedLine<_Scalar, _AmbientDim,_Options>::intersectionParameter(const Hyperplane<_Scalar, _AmbientDim, OtherOptions>& hyperplane) const + */ +template +template +EIGEN_DEVICE_FUNC inline _Scalar ParametrizedLine<_Scalar, _AmbientDim, _Options>::intersectionParameter( + const Hyperplane<_Scalar, _AmbientDim, OtherOptions> &hyperplane) const { - return -(hyperplane.offset()+hyperplane.normal().dot(origin())) - / hyperplane.normal().dot(direction()); + return -(hyperplane.offset() + hyperplane.normal().dot(origin())) / hyperplane.normal().dot(direction()); } /** \deprecated use intersectionParameter() - * \returns the parameter value of the intersection between \c *this and the given \a hyperplane - */ -template -template -EIGEN_DEVICE_FUNC inline _Scalar ParametrizedLine<_Scalar, _AmbientDim,_Options>::intersection(const Hyperplane<_Scalar, _AmbientDim, OtherOptions>& hyperplane) const + * \returns the parameter value of the intersection between \c *this and the given \a hyperplane + */ +template +template +EIGEN_DEVICE_FUNC inline _Scalar ParametrizedLine<_Scalar, _AmbientDim, _Options>::intersection( + const Hyperplane<_Scalar, _AmbientDim, OtherOptions> &hyperplane) const { return intersectionParameter(hyperplane); } /** \returns the point of the intersection between \c *this and the given hyperplane - */ -template -template -EIGEN_DEVICE_FUNC inline typename ParametrizedLine<_Scalar, _AmbientDim,_Options>::VectorType -ParametrizedLine<_Scalar, _AmbientDim,_Options>::intersectionPoint(const Hyperplane<_Scalar, _AmbientDim, OtherOptions>& hyperplane) const + */ +template +template +EIGEN_DEVICE_FUNC inline typename ParametrizedLine<_Scalar, _AmbientDim, _Options>::VectorType + ParametrizedLine<_Scalar, _AmbientDim, _Options>::intersectionPoint( + const Hyperplane<_Scalar, _AmbientDim, OtherOptions> &hyperplane) const { return pointAt(intersectionParameter(hyperplane)); } -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_PARAMETRIZEDLINE_H +#endif// EIGEN_PARAMETRIZEDLINE_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Geometry/Quaternion.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Geometry/Quaternion.h index c3fd8c3e..5843014a 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Geometry/Quaternion.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Geometry/Quaternion.h @@ -10,31 +10,28 @@ #ifndef EIGEN_QUATERNION_H #define EIGEN_QUATERNION_H -namespace Eigen { +namespace Eigen { /*************************************************************************** -* Definition of QuaternionBase -* The implementation is at the end of the file -***************************************************************************/ + * Definition of QuaternionBase + * The implementation is at the end of the file + ***************************************************************************/ namespace internal { -template -struct quaternionbase_assign_impl; + template + struct quaternionbase_assign_impl; } /** \geometry_module \ingroup Geometry_Module - * \class QuaternionBase - * \brief Base class for quaternion expressions - * \tparam Derived derived type (CRTP) - * \sa class Quaternion - */ -template -class QuaternionBase : public RotationBase + * \class QuaternionBase + * \brief Base class for quaternion expressions + * \tparam Derived derived type (CRTP) + * \sa class Quaternion + */ +template class QuaternionBase : public RotationBase { - public: +public: typedef RotationBase Base; using Base::operator*; @@ -44,24 +41,22 @@ class QuaternionBase : public RotationBase typedef typename NumTraits::Real RealScalar; typedef typename internal::traits::Coefficients Coefficients; typedef typename Coefficients::CoeffReturnType CoeffReturnType; - typedef typename internal::conditional::Flags&LvalueBit), - Scalar&, CoeffReturnType>::type NonConstCoeffReturnType; + typedef + typename internal::conditional::Flags &LvalueBit), Scalar &, CoeffReturnType>::type + NonConstCoeffReturnType; - enum { - Flags = Eigen::internal::traits::Flags - }; + enum { Flags = Eigen::internal::traits::Flags }; - // typedef typename Matrix Coefficients; + // typedef typename Matrix Coefficients; /** the type of a 3D vector */ - typedef Matrix Vector3; + typedef Matrix Vector3; /** the equivalent rotation matrix type */ - typedef Matrix Matrix3; + typedef Matrix Matrix3; /** the equivalent angle-axis type */ typedef AngleAxis AngleAxisType; - /** \returns the \c x coefficient */ EIGEN_DEVICE_FUNC inline CoeffReturnType x() const { return this->derived().coeffs().coeff(0); } /** \returns the \c y coefficient */ @@ -81,74 +76,91 @@ class QuaternionBase : public RotationBase EIGEN_DEVICE_FUNC inline NonConstCoeffReturnType w() { return this->derived().coeffs().w(); } /** \returns a read-only vector expression of the imaginary part (x,y,z) */ - EIGEN_DEVICE_FUNC inline const VectorBlock vec() const { return coeffs().template head<3>(); } + EIGEN_DEVICE_FUNC inline const VectorBlock vec() const { return coeffs().template head<3>(); } /** \returns a vector expression of the imaginary part (x,y,z) */ - EIGEN_DEVICE_FUNC inline VectorBlock vec() { return coeffs().template head<3>(); } + EIGEN_DEVICE_FUNC inline VectorBlock vec() { return coeffs().template head<3>(); } /** \returns a read-only vector expression of the coefficients (x,y,z,w) */ - EIGEN_DEVICE_FUNC inline const typename internal::traits::Coefficients& coeffs() const { return derived().coeffs(); } + EIGEN_DEVICE_FUNC inline const typename internal::traits::Coefficients &coeffs() const + { + return derived().coeffs(); + } /** \returns a vector expression of the coefficients (x,y,z,w) */ - EIGEN_DEVICE_FUNC inline typename internal::traits::Coefficients& coeffs() { return derived().coeffs(); } + EIGEN_DEVICE_FUNC inline typename internal::traits::Coefficients &coeffs() { return derived().coeffs(); } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE QuaternionBase& operator=(const QuaternionBase& other); - template EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Derived& operator=(const QuaternionBase& other); + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE QuaternionBase &operator=(const QuaternionBase &other); + template + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Derived &operator=(const QuaternionBase &other); -// disabled this copy operator as it is giving very strange compilation errors when compiling -// test_stdvector with GCC 4.4.2. This looks like a GCC bug though, so feel free to re-enable it if it's -// useful; however notice that we already have the templated operator= above and e.g. in MatrixBase -// we didn't have to add, in addition to templated operator=, such a non-templated copy operator. -// Derived& operator=(const QuaternionBase& other) -// { return operator=(other); } + // disabled this copy operator as it is giving very strange compilation errors when compiling + // test_stdvector with GCC 4.4.2. This looks like a GCC bug though, so feel free to re-enable it if it's + // useful; however notice that we already have the templated operator= above and e.g. in MatrixBase + // we didn't have to add, in addition to templated operator=, such a non-templated copy operator. + // Derived& operator=(const QuaternionBase& other) + // { return operator=(other); } - EIGEN_DEVICE_FUNC Derived& operator=(const AngleAxisType& aa); - template EIGEN_DEVICE_FUNC Derived& operator=(const MatrixBase& m); + EIGEN_DEVICE_FUNC Derived &operator=(const AngleAxisType &aa); + template EIGEN_DEVICE_FUNC Derived &operator=(const MatrixBase &m); /** \returns a quaternion representing an identity rotation - * \sa MatrixBase::Identity() - */ - EIGEN_DEVICE_FUNC static inline Quaternion Identity() { return Quaternion(Scalar(1), Scalar(0), Scalar(0), Scalar(0)); } + * \sa MatrixBase::Identity() + */ + EIGEN_DEVICE_FUNC static inline Quaternion Identity() + { + return Quaternion(Scalar(1), Scalar(0), Scalar(0), Scalar(0)); + } /** \sa QuaternionBase::Identity(), MatrixBase::setIdentity() - */ - EIGEN_DEVICE_FUNC inline QuaternionBase& setIdentity() { coeffs() << Scalar(0), Scalar(0), Scalar(0), Scalar(1); return *this; } + */ + EIGEN_DEVICE_FUNC inline QuaternionBase &setIdentity() + { + coeffs() << Scalar(0), Scalar(0), Scalar(0), Scalar(1); + return *this; + } /** \returns the squared norm of the quaternion's coefficients - * \sa QuaternionBase::norm(), MatrixBase::squaredNorm() - */ + * \sa QuaternionBase::norm(), MatrixBase::squaredNorm() + */ EIGEN_DEVICE_FUNC inline Scalar squaredNorm() const { return coeffs().squaredNorm(); } /** \returns the norm of the quaternion's coefficients - * \sa QuaternionBase::squaredNorm(), MatrixBase::norm() - */ + * \sa QuaternionBase::squaredNorm(), MatrixBase::norm() + */ EIGEN_DEVICE_FUNC inline Scalar norm() const { return coeffs().norm(); } /** Normalizes the quaternion \c *this - * \sa normalized(), MatrixBase::normalize() */ + * \sa normalized(), MatrixBase::normalize() */ EIGEN_DEVICE_FUNC inline void normalize() { coeffs().normalize(); } /** \returns a normalized copy of \c *this - * \sa normalize(), MatrixBase::normalized() */ + * \sa normalize(), MatrixBase::normalized() */ EIGEN_DEVICE_FUNC inline Quaternion normalized() const { return Quaternion(coeffs().normalized()); } - /** \returns the dot product of \c *this and \a other - * Geometrically speaking, the dot product of two unit quaternions - * corresponds to the cosine of half the angle between the two rotations. - * \sa angularDistance() - */ - template EIGEN_DEVICE_FUNC inline Scalar dot(const QuaternionBase& other) const { return coeffs().dot(other.coeffs()); } + /** \returns the dot product of \c *this and \a other + * Geometrically speaking, the dot product of two unit quaternions + * corresponds to the cosine of half the angle between the two rotations. + * \sa angularDistance() + */ + template EIGEN_DEVICE_FUNC inline Scalar dot(const QuaternionBase &other) const + { + return coeffs().dot(other.coeffs()); + } - template EIGEN_DEVICE_FUNC Scalar angularDistance(const QuaternionBase& other) const; + template + EIGEN_DEVICE_FUNC Scalar angularDistance(const QuaternionBase &other) const; /** \returns an equivalent 3x3 rotation matrix */ EIGEN_DEVICE_FUNC Matrix3 toRotationMatrix() const; /** \returns the quaternion which transform \a a into \a b through a rotation */ template - EIGEN_DEVICE_FUNC Derived& setFromTwoVectors(const MatrixBase& a, const MatrixBase& b); + EIGEN_DEVICE_FUNC Derived &setFromTwoVectors(const MatrixBase &a, const MatrixBase &b); - template EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Quaternion operator* (const QuaternionBase& q) const; - template EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Derived& operator*= (const QuaternionBase& q); + template + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Quaternion operator*(const QuaternionBase &q) const; + template + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Derived &operator*=(const QuaternionBase &q); /** \returns the quaternion describing the inverse rotation */ EIGEN_DEVICE_FUNC Quaternion inverse() const; @@ -156,84 +168,85 @@ class QuaternionBase : public RotationBase /** \returns the conjugated quaternion */ EIGEN_DEVICE_FUNC Quaternion conjugate() const; - template EIGEN_DEVICE_FUNC Quaternion slerp(const Scalar& t, const QuaternionBase& other) const; + template + EIGEN_DEVICE_FUNC Quaternion slerp(const Scalar &t, const QuaternionBase &other) const; /** \returns \c true if \c *this is approximately equal to \a other, within the precision - * determined by \a prec. - * - * \sa MatrixBase::isApprox() */ + * determined by \a prec. + * + * \sa MatrixBase::isApprox() */ template - EIGEN_DEVICE_FUNC bool isApprox(const QuaternionBase& other, const RealScalar& prec = NumTraits::dummy_precision()) const - { return coeffs().isApprox(other.coeffs(), prec); } + EIGEN_DEVICE_FUNC bool isApprox(const QuaternionBase &other, + const RealScalar &prec = NumTraits::dummy_precision()) const + { + return coeffs().isApprox(other.coeffs(), prec); + } /** return the result vector of \a v through the rotation*/ - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Vector3 _transformVector(const Vector3& v) const; + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Vector3 _transformVector(const Vector3 &v) const; /** \returns \c *this with scalar type casted to \a NewScalarType - * - * Note that if \a NewScalarType is equal to the current scalar type of \c *this - * then this function smartly returns a const reference to \c *this. - */ + * + * Note that if \a NewScalarType is equal to the current scalar type of \c *this + * then this function smartly returns a const reference to \c *this. + */ template - EIGEN_DEVICE_FUNC inline typename internal::cast_return_type >::type cast() const + EIGEN_DEVICE_FUNC inline typename internal::cast_return_type>::type cast() const { - return typename internal::cast_return_type >::type(derived()); + return typename internal::cast_return_type>::type(derived()); } #ifdef EIGEN_QUATERNIONBASE_PLUGIN -# include EIGEN_QUATERNIONBASE_PLUGIN +#include EIGEN_QUATERNIONBASE_PLUGIN #endif }; /*************************************************************************** -* Definition/implementation of Quaternion -***************************************************************************/ + * Definition/implementation of Quaternion + ***************************************************************************/ /** \geometry_module \ingroup Geometry_Module - * - * \class Quaternion - * - * \brief The quaternion class used to represent 3D orientations and rotations - * - * \tparam _Scalar the scalar type, i.e., the type of the coefficients - * \tparam _Options controls the memory alignment of the coefficients. Can be \# AutoAlign or \# DontAlign. Default is AutoAlign. - * - * This class represents a quaternion \f$ w+xi+yj+zk \f$ that is a convenient representation of - * orientations and rotations of objects in three dimensions. Compared to other representations - * like Euler angles or 3x3 matrices, quaternions offer the following advantages: - * \li \b compact storage (4 scalars) - * \li \b efficient to compose (28 flops), - * \li \b stable spherical interpolation - * - * The following two typedefs are provided for convenience: - * \li \c Quaternionf for \c float - * \li \c Quaterniond for \c double - * - * \warning Operations interpreting the quaternion as rotation have undefined behavior if the quaternion is not normalized. - * - * \sa class AngleAxis, class Transform - */ + * + * \class Quaternion + * + * \brief The quaternion class used to represent 3D orientations and rotations + * + * \tparam _Scalar the scalar type, i.e., the type of the coefficients + * \tparam _Options controls the memory alignment of the coefficients. Can be \# AutoAlign or \# DontAlign. Default is + * AutoAlign. + * + * This class represents a quaternion \f$ w+xi+yj+zk \f$ that is a convenient representation of + * orientations and rotations of objects in three dimensions. Compared to other representations + * like Euler angles or 3x3 matrices, quaternions offer the following advantages: + * \li \b compact storage (4 scalars) + * \li \b efficient to compose (28 flops), + * \li \b stable spherical interpolation + * + * The following two typedefs are provided for convenience: + * \li \c Quaternionf for \c float + * \li \c Quaterniond for \c double + * + * \warning Operations interpreting the quaternion as rotation have undefined behavior if the quaternion is not + * normalized. + * + * \sa class AngleAxis, class Transform + */ namespace internal { -template -struct traits > -{ - typedef Quaternion<_Scalar,_Options> PlainObject; - typedef _Scalar Scalar; - typedef Matrix<_Scalar,4,1,_Options> Coefficients; - enum{ - Alignment = internal::traits::Alignment, - Flags = LvalueBit + template struct traits> + { + typedef Quaternion<_Scalar, _Options> PlainObject; + typedef _Scalar Scalar; + typedef Matrix<_Scalar, 4, 1, _Options> Coefficients; + enum { Alignment = internal::traits::Alignment, Flags = LvalueBit }; }; -}; -} +}// namespace internal -template -class Quaternion : public QuaternionBase > +template class Quaternion : public QuaternionBase> { public: - typedef QuaternionBase > Base; - enum { NeedsAlignment = internal::traits::Alignment>0 }; + typedef QuaternionBase> Base; + enum { NeedsAlignment = internal::traits::Alignment > 0 }; typedef _Scalar Scalar; @@ -247,245 +260,253 @@ class Quaternion : public QuaternionBase > EIGEN_DEVICE_FUNC inline Quaternion() {} /** Constructs and initializes the quaternion \f$ w+xi+yj+zk \f$ from - * its four coefficients \a w, \a x, \a y and \a z. - * - * \warning Note the order of the arguments: the real \a w coefficient first, - * while internally the coefficients are stored in the following order: - * [\c x, \c y, \c z, \c w] - */ - EIGEN_DEVICE_FUNC inline Quaternion(const Scalar& w, const Scalar& x, const Scalar& y, const Scalar& z) : m_coeffs(x, y, z, w){} + * its four coefficients \a w, \a x, \a y and \a z. + * + * \warning Note the order of the arguments: the real \a w coefficient first, + * while internally the coefficients are stored in the following order: + * [\c x, \c y, \c z, \c w] + */ + EIGEN_DEVICE_FUNC inline Quaternion(const Scalar &w, const Scalar &x, const Scalar &y, const Scalar &z) + : m_coeffs(x, y, z, w) + {} /** Constructs and initialize a quaternion from the array data */ - EIGEN_DEVICE_FUNC explicit inline Quaternion(const Scalar* data) : m_coeffs(data) {} + EIGEN_DEVICE_FUNC explicit inline Quaternion(const Scalar *data) : m_coeffs(data) {} /** Copy constructor */ - template EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Quaternion(const QuaternionBase& other) { this->Base::operator=(other); } + template EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Quaternion(const QuaternionBase &other) + { + this->Base::operator=(other); + } /** Constructs and initializes a quaternion from the angle-axis \a aa */ - EIGEN_DEVICE_FUNC explicit inline Quaternion(const AngleAxisType& aa) { *this = aa; } + EIGEN_DEVICE_FUNC explicit inline Quaternion(const AngleAxisType &aa) { *this = aa; } /** Constructs and initializes a quaternion from either: - * - a rotation matrix expression, - * - a 4D vector expression representing quaternion coefficients. - */ - template - EIGEN_DEVICE_FUNC explicit inline Quaternion(const MatrixBase& other) { *this = other; } + * - a rotation matrix expression, + * - a 4D vector expression representing quaternion coefficients. + */ + template EIGEN_DEVICE_FUNC explicit inline Quaternion(const MatrixBase &other) + { + *this = other; + } /** Explicit copy constructor with scalar conversion */ template - EIGEN_DEVICE_FUNC explicit inline Quaternion(const Quaternion& other) - { m_coeffs = other.coeffs().template cast(); } + EIGEN_DEVICE_FUNC explicit inline Quaternion(const Quaternion &other) + { + m_coeffs = other.coeffs().template cast(); + } EIGEN_DEVICE_FUNC static Quaternion UnitRandom(); template - EIGEN_DEVICE_FUNC static Quaternion FromTwoVectors(const MatrixBase& a, const MatrixBase& b); + EIGEN_DEVICE_FUNC static Quaternion FromTwoVectors(const MatrixBase &a, const MatrixBase &b); - EIGEN_DEVICE_FUNC inline Coefficients& coeffs() { return m_coeffs;} - EIGEN_DEVICE_FUNC inline const Coefficients& coeffs() const { return m_coeffs;} + EIGEN_DEVICE_FUNC inline Coefficients &coeffs() { return m_coeffs; } + EIGEN_DEVICE_FUNC inline const Coefficients &coeffs() const { return m_coeffs; } EIGEN_MAKE_ALIGNED_OPERATOR_NEW_IF(bool(NeedsAlignment)) - + #ifdef EIGEN_QUATERNION_PLUGIN -# include EIGEN_QUATERNION_PLUGIN +#include EIGEN_QUATERNION_PLUGIN #endif protected: Coefficients m_coeffs; - + #ifndef EIGEN_PARSED_BY_DOXYGEN - static EIGEN_STRONG_INLINE void _check_template_params() - { - EIGEN_STATIC_ASSERT( (_Options & DontAlign) == _Options, - INVALID_MATRIX_TEMPLATE_PARAMETERS) - } + static EIGEN_STRONG_INLINE void _check_template_params() + { + EIGEN_STATIC_ASSERT((_Options & DontAlign) == _Options, INVALID_MATRIX_TEMPLATE_PARAMETERS) + } #endif }; /** \ingroup Geometry_Module - * single precision quaternion type */ + * single precision quaternion type */ typedef Quaternion Quaternionf; /** \ingroup Geometry_Module - * double precision quaternion type */ + * double precision quaternion type */ typedef Quaternion Quaterniond; /*************************************************************************** -* Specialization of Map> -***************************************************************************/ + * Specialization of Map> + ***************************************************************************/ namespace internal { template - struct traits, _Options> > : traits > + struct traits, _Options>> + : traits> { - typedef Map, _Options> Coefficients; + typedef Map, _Options> Coefficients; }; -} +}// namespace internal namespace internal { template - struct traits, _Options> > : traits > + struct traits, _Options>> + : traits> { - typedef Map, _Options> Coefficients; - typedef traits > TraitsBase; - enum { - Flags = TraitsBase::Flags & ~LvalueBit - }; + typedef Map, _Options> Coefficients; + typedef traits> TraitsBase; + enum { Flags = TraitsBase::Flags & ~LvalueBit }; }; -} +}// namespace internal /** \ingroup Geometry_Module - * \brief Quaternion expression mapping a constant memory buffer - * - * \tparam _Scalar the type of the Quaternion coefficients - * \tparam _Options see class Map - * - * This is a specialization of class Map for Quaternion. This class allows to view - * a 4 scalar memory buffer as an Eigen's Quaternion object. - * - * \sa class Map, class Quaternion, class QuaternionBase - */ + * \brief Quaternion expression mapping a constant memory buffer + * + * \tparam _Scalar the type of the Quaternion coefficients + * \tparam _Options see class Map + * + * This is a specialization of class Map for Quaternion. This class allows to view + * a 4 scalar memory buffer as an Eigen's Quaternion object. + * + * \sa class Map, class Quaternion, class QuaternionBase + */ template -class Map, _Options > - : public QuaternionBase, _Options> > +class Map, _Options> : public QuaternionBase, _Options>> { - public: - typedef QuaternionBase, _Options> > Base; +public: + typedef QuaternionBase, _Options>> Base; - typedef _Scalar Scalar; - typedef typename internal::traits::Coefficients Coefficients; - EIGEN_INHERIT_ASSIGNMENT_OPERATORS(Map) - using Base::operator*=; - - /** Constructs a Mapped Quaternion object from the pointer \a coeffs - * - * The pointer \a coeffs must reference the four coefficients of Quaternion in the following order: - * \code *coeffs == {x, y, z, w} \endcode - * - * If the template parameter _Options is set to #Aligned, then the pointer coeffs must be aligned. */ - EIGEN_DEVICE_FUNC explicit EIGEN_STRONG_INLINE Map(const Scalar* coeffs) : m_coeffs(coeffs) {} - - EIGEN_DEVICE_FUNC inline const Coefficients& coeffs() const { return m_coeffs;} - - protected: - const Coefficients m_coeffs; + typedef _Scalar Scalar; + typedef typename internal::traits::Coefficients Coefficients; + EIGEN_INHERIT_ASSIGNMENT_OPERATORS(Map) + using Base::operator*=; + + /** Constructs a Mapped Quaternion object from the pointer \a coeffs + * + * The pointer \a coeffs must reference the four coefficients of Quaternion in the following order: + * \code *coeffs == {x, y, z, w} \endcode + * + * If the template parameter _Options is set to #Aligned, then the pointer coeffs must be aligned. */ + EIGEN_DEVICE_FUNC explicit EIGEN_STRONG_INLINE Map(const Scalar *coeffs) : m_coeffs(coeffs) {} + + EIGEN_DEVICE_FUNC inline const Coefficients &coeffs() const { return m_coeffs; } + +protected: + const Coefficients m_coeffs; }; /** \ingroup Geometry_Module - * \brief Expression of a quaternion from a memory buffer - * - * \tparam _Scalar the type of the Quaternion coefficients - * \tparam _Options see class Map - * - * This is a specialization of class Map for Quaternion. This class allows to view - * a 4 scalar memory buffer as an Eigen's Quaternion object. - * - * \sa class Map, class Quaternion, class QuaternionBase - */ + * \brief Expression of a quaternion from a memory buffer + * + * \tparam _Scalar the type of the Quaternion coefficients + * \tparam _Options see class Map + * + * This is a specialization of class Map for Quaternion. This class allows to view + * a 4 scalar memory buffer as an Eigen's Quaternion object. + * + * \sa class Map, class Quaternion, class QuaternionBase + */ template -class Map, _Options > - : public QuaternionBase, _Options> > +class Map, _Options> : public QuaternionBase, _Options>> { - public: - typedef QuaternionBase, _Options> > Base; +public: + typedef QuaternionBase, _Options>> Base; - typedef _Scalar Scalar; - typedef typename internal::traits::Coefficients Coefficients; - EIGEN_INHERIT_ASSIGNMENT_OPERATORS(Map) - using Base::operator*=; - - /** Constructs a Mapped Quaternion object from the pointer \a coeffs - * - * The pointer \a coeffs must reference the four coefficients of Quaternion in the following order: - * \code *coeffs == {x, y, z, w} \endcode - * - * If the template parameter _Options is set to #Aligned, then the pointer coeffs must be aligned. */ - EIGEN_DEVICE_FUNC explicit EIGEN_STRONG_INLINE Map(Scalar* coeffs) : m_coeffs(coeffs) {} - - EIGEN_DEVICE_FUNC inline Coefficients& coeffs() { return m_coeffs; } - EIGEN_DEVICE_FUNC inline const Coefficients& coeffs() const { return m_coeffs; } - - protected: - Coefficients m_coeffs; + typedef _Scalar Scalar; + typedef typename internal::traits::Coefficients Coefficients; + EIGEN_INHERIT_ASSIGNMENT_OPERATORS(Map) + using Base::operator*=; + + /** Constructs a Mapped Quaternion object from the pointer \a coeffs + * + * The pointer \a coeffs must reference the four coefficients of Quaternion in the following order: + * \code *coeffs == {x, y, z, w} \endcode + * + * If the template parameter _Options is set to #Aligned, then the pointer coeffs must be aligned. */ + EIGEN_DEVICE_FUNC explicit EIGEN_STRONG_INLINE Map(Scalar *coeffs) : m_coeffs(coeffs) {} + + EIGEN_DEVICE_FUNC inline Coefficients &coeffs() { return m_coeffs; } + EIGEN_DEVICE_FUNC inline const Coefficients &coeffs() const { return m_coeffs; } + +protected: + Coefficients m_coeffs; }; /** \ingroup Geometry_Module - * Map an unaligned array of single precision scalars as a quaternion */ -typedef Map, 0> QuaternionMapf; + * Map an unaligned array of single precision scalars as a quaternion */ +typedef Map, 0> QuaternionMapf; /** \ingroup Geometry_Module - * Map an unaligned array of double precision scalars as a quaternion */ -typedef Map, 0> QuaternionMapd; + * Map an unaligned array of double precision scalars as a quaternion */ +typedef Map, 0> QuaternionMapd; /** \ingroup Geometry_Module - * Map a 16-byte aligned array of single precision scalars as a quaternion */ -typedef Map, Aligned> QuaternionMapAlignedf; + * Map a 16-byte aligned array of single precision scalars as a quaternion */ +typedef Map, Aligned> QuaternionMapAlignedf; /** \ingroup Geometry_Module - * Map a 16-byte aligned array of double precision scalars as a quaternion */ -typedef Map, Aligned> QuaternionMapAlignedd; + * Map a 16-byte aligned array of double precision scalars as a quaternion */ +typedef Map, Aligned> QuaternionMapAlignedd; /*************************************************************************** -* Implementation of QuaternionBase methods -***************************************************************************/ + * Implementation of QuaternionBase methods + ***************************************************************************/ // Generic Quaternion * Quaternion product // This product can be specialized for a given architecture via the Arch template argument. namespace internal { -template struct quat_product -{ - EIGEN_DEVICE_FUNC static EIGEN_STRONG_INLINE Quaternion run(const QuaternionBase& a, const QuaternionBase& b){ - return Quaternion - ( - a.w() * b.w() - a.x() * b.x() - a.y() * b.y() - a.z() * b.z(), - a.w() * b.x() + a.x() * b.w() + a.y() * b.z() - a.z() * b.y(), - a.w() * b.y() + a.y() * b.w() + a.z() * b.x() - a.x() * b.z(), - a.w() * b.z() + a.z() * b.w() + a.x() * b.y() - a.y() * b.x() - ); - } -}; -} + template struct quat_product + { + EIGEN_DEVICE_FUNC static EIGEN_STRONG_INLINE Quaternion run(const QuaternionBase &a, + const QuaternionBase &b) + { + return Quaternion(a.w() * b.w() - a.x() * b.x() - a.y() * b.y() - a.z() * b.z(), + a.w() * b.x() + a.x() * b.w() + a.y() * b.z() - a.z() * b.y(), + a.w() * b.y() + a.y() * b.w() + a.z() * b.x() - a.x() * b.z(), + a.w() * b.z() + a.z() * b.w() + a.x() * b.y() - a.y() * b.x()); + } + }; +}// namespace internal /** \returns the concatenation of two rotations as a quaternion-quaternion product */ -template -template +template +template EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Quaternion::Scalar> -QuaternionBase::operator* (const QuaternionBase& other) const + QuaternionBase::operator*(const QuaternionBase &other) const { EIGEN_STATIC_ASSERT((internal::is_same::value), - YOU_MIXED_DIFFERENT_NUMERIC_TYPES__YOU_NEED_TO_USE_THE_CAST_METHOD_OF_MATRIXBASE_TO_CAST_NUMERIC_TYPES_EXPLICITLY) - return internal::quat_product::Scalar>::run(*this, other); + YOU_MIXED_DIFFERENT_NUMERIC_TYPES__YOU_NEED_TO_USE_THE_CAST_METHOD_OF_MATRIXBASE_TO_CAST_NUMERIC_TYPES_EXPLICITLY) + return internal:: + quat_product::Scalar>::run( + *this, other); } /** \sa operator*(Quaternion) */ -template -template -EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Derived& QuaternionBase::operator*= (const QuaternionBase& other) +template +template +EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Derived &QuaternionBase::operator*=( + const QuaternionBase &other) { derived() = derived() * other.derived(); return derived(); } /** Rotation of a vector by a quaternion. - * \remarks If the quaternion is used to rotate several points (>1) - * then it is much more efficient to first convert it to a 3x3 Matrix. - * Comparison of the operation cost for n transformations: - * - Quaternion2: 30n - * - Via a Matrix3: 24 + 15n - */ -template + * \remarks If the quaternion is used to rotate several points (>1) + * then it is much more efficient to first convert it to a 3x3 Matrix. + * Comparison of the operation cost for n transformations: + * - Quaternion2: 30n + * - Via a Matrix3: 24 + 15n + */ +template EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE typename QuaternionBase::Vector3 -QuaternionBase::_transformVector(const Vector3& v) const + QuaternionBase::_transformVector(const Vector3 &v) const { - // Note that this algorithm comes from the optimization by hand - // of the conversion to a Matrix followed by a Matrix/Vector product. - // It appears to be much faster than the common algorithm found - // in the literature (30 versus 39 flops). It also requires two - // Vector3 as temporaries. - Vector3 uv = this->vec().cross(v); - uv += uv; - return v + this->w() * uv + this->vec().cross(uv); + // Note that this algorithm comes from the optimization by hand + // of the conversion to a Matrix followed by a Matrix/Vector product. + // It appears to be much faster than the common algorithm found + // in the literature (30 versus 39 flops). It also requires two + // Vector3 as temporaries. + Vector3 uv = this->vec().cross(v); + uv += uv; + return v + this->w() * uv + this->vec().cross(uv); } template -EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE QuaternionBase& QuaternionBase::operator=(const QuaternionBase& other) +EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE QuaternionBase &QuaternionBase::operator=( + const QuaternionBase &other) { coeffs() = other.coeffs(); return derived(); @@ -493,47 +514,47 @@ EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE QuaternionBase& QuaternionBase template -EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Derived& QuaternionBase::operator=(const QuaternionBase& other) +EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Derived &QuaternionBase::operator=( + const QuaternionBase &other) { coeffs() = other.coeffs(); return derived(); } /** Set \c *this from an angle-axis \a aa and returns a reference to \c *this - */ + */ template -EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Derived& QuaternionBase::operator=(const AngleAxisType& aa) +EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Derived &QuaternionBase::operator=(const AngleAxisType &aa) { EIGEN_USING_STD_MATH(cos) EIGEN_USING_STD_MATH(sin) - Scalar ha = Scalar(0.5)*aa.angle(); // Scalar(0.5) to suppress precision loss warnings + Scalar ha = Scalar(0.5) * aa.angle();// Scalar(0.5) to suppress precision loss warnings this->w() = cos(ha); this->vec() = sin(ha) * aa.axis(); return derived(); } /** Set \c *this from the expression \a xpr: - * - if \a xpr is a 4x1 vector, then \a xpr is assumed to be a quaternion - * - if \a xpr is a 3x3 matrix, then \a xpr is assumed to be rotation matrix - * and \a xpr is converted to a quaternion - */ + * - if \a xpr is a 4x1 vector, then \a xpr is assumed to be a quaternion + * - if \a xpr is a 3x3 matrix, then \a xpr is assumed to be rotation matrix + * and \a xpr is converted to a quaternion + */ template template -EIGEN_DEVICE_FUNC inline Derived& QuaternionBase::operator=(const MatrixBase& xpr) +EIGEN_DEVICE_FUNC inline Derived &QuaternionBase::operator=(const MatrixBase &xpr) { EIGEN_STATIC_ASSERT((internal::is_same::value), - YOU_MIXED_DIFFERENT_NUMERIC_TYPES__YOU_NEED_TO_USE_THE_CAST_METHOD_OF_MATRIXBASE_TO_CAST_NUMERIC_TYPES_EXPLICITLY) + YOU_MIXED_DIFFERENT_NUMERIC_TYPES__YOU_NEED_TO_USE_THE_CAST_METHOD_OF_MATRIXBASE_TO_CAST_NUMERIC_TYPES_EXPLICITLY) internal::quaternionbase_assign_impl::run(*this, xpr.derived()); return derived(); } /** Convert the quaternion to a 3x3 rotation matrix. The quaternion is required to - * be normalized, otherwise the result is undefined. - */ + * be normalized, otherwise the result is undefined. + */ template -EIGEN_DEVICE_FUNC inline typename QuaternionBase::Matrix3 -QuaternionBase::toRotationMatrix(void) const +EIGEN_DEVICE_FUNC inline typename QuaternionBase::Matrix3 QuaternionBase::toRotationMatrix(void) const { // NOTE if inlined, then gcc 4.2 and 4.4 get rid of the temporary (not gcc 4.3 !!) // if not inlined then the cost of the return by value is huge ~ +35%, @@ -541,45 +562,46 @@ QuaternionBase::toRotationMatrix(void) const // it has to be inlined, and so the return by value is not an issue Matrix3 res; - const Scalar tx = Scalar(2)*this->x(); - const Scalar ty = Scalar(2)*this->y(); - const Scalar tz = Scalar(2)*this->z(); - const Scalar twx = tx*this->w(); - const Scalar twy = ty*this->w(); - const Scalar twz = tz*this->w(); - const Scalar txx = tx*this->x(); - const Scalar txy = ty*this->x(); - const Scalar txz = tz*this->x(); - const Scalar tyy = ty*this->y(); - const Scalar tyz = tz*this->y(); - const Scalar tzz = tz*this->z(); - - res.coeffRef(0,0) = Scalar(1)-(tyy+tzz); - res.coeffRef(0,1) = txy-twz; - res.coeffRef(0,2) = txz+twy; - res.coeffRef(1,0) = txy+twz; - res.coeffRef(1,1) = Scalar(1)-(txx+tzz); - res.coeffRef(1,2) = tyz-twx; - res.coeffRef(2,0) = txz-twy; - res.coeffRef(2,1) = tyz+twx; - res.coeffRef(2,2) = Scalar(1)-(txx+tyy); + const Scalar tx = Scalar(2) * this->x(); + const Scalar ty = Scalar(2) * this->y(); + const Scalar tz = Scalar(2) * this->z(); + const Scalar twx = tx * this->w(); + const Scalar twy = ty * this->w(); + const Scalar twz = tz * this->w(); + const Scalar txx = tx * this->x(); + const Scalar txy = ty * this->x(); + const Scalar txz = tz * this->x(); + const Scalar tyy = ty * this->y(); + const Scalar tyz = tz * this->y(); + const Scalar tzz = tz * this->z(); + + res.coeffRef(0, 0) = Scalar(1) - (tyy + tzz); + res.coeffRef(0, 1) = txy - twz; + res.coeffRef(0, 2) = txz + twy; + res.coeffRef(1, 0) = txy + twz; + res.coeffRef(1, 1) = Scalar(1) - (txx + tzz); + res.coeffRef(1, 2) = tyz - twx; + res.coeffRef(2, 0) = txz - twy; + res.coeffRef(2, 1) = tyz + twx; + res.coeffRef(2, 2) = Scalar(1) - (txx + tyy); return res; } /** Sets \c *this to be a quaternion representing a rotation between - * the two arbitrary vectors \a a and \a b. In other words, the built - * rotation represent a rotation sending the line of direction \a a - * to the line of direction \a b, both lines passing through the origin. - * - * \returns a reference to \c *this. - * - * Note that the two input vectors do \b not have to be normalized, and - * do not need to have the same norm. - */ + * the two arbitrary vectors \a a and \a b. In other words, the built + * rotation represent a rotation sending the line of direction \a a + * to the line of direction \a b, both lines passing through the origin. + * + * \returns a reference to \c *this. + * + * Note that the two input vectors do \b not have to be normalized, and + * do not need to have the same norm. + */ template template -EIGEN_DEVICE_FUNC inline Derived& QuaternionBase::setFromTwoVectors(const MatrixBase& a, const MatrixBase& b) +EIGEN_DEVICE_FUNC inline Derived &QuaternionBase::setFromTwoVectors(const MatrixBase &a, + const MatrixBase &b) { EIGEN_USING_STD_MATH(sqrt) Vector3 v0 = a.normalized(); @@ -594,21 +616,21 @@ EIGEN_DEVICE_FUNC inline Derived& QuaternionBase::setFromTwoVectors(con // under the constraint: // ||x|| = 1 // which yields a singular value problem - if (c < Scalar(-1)+NumTraits::dummy_precision()) - { - c = numext::maxi(c,Scalar(-1)); - Matrix m; m << v0.transpose(), v1.transpose(); - JacobiSVD > svd(m, ComputeFullV); + if (c < Scalar(-1) + NumTraits::dummy_precision()) { + c = numext::maxi(c, Scalar(-1)); + Matrix m; + m << v0.transpose(), v1.transpose(); + JacobiSVD> svd(m, ComputeFullV); Vector3 axis = svd.matrixV().col(2); - Scalar w2 = (Scalar(1)+c)*Scalar(0.5); + Scalar w2 = (Scalar(1) + c) * Scalar(0.5); this->w() = sqrt(w2); this->vec() = axis * sqrt(Scalar(1) - w2); return derived(); } Vector3 axis = v0.cross(v1); - Scalar s = sqrt((Scalar(1)+c)*Scalar(2)); - Scalar invs = Scalar(1)/s; + Scalar s = sqrt((Scalar(1) + c) * Scalar(2)); + Scalar invs = Scalar(1) / s; this->vec() = axis * invs; this->w() = s * Scalar(0.5); @@ -616,59 +638,57 @@ EIGEN_DEVICE_FUNC inline Derived& QuaternionBase::setFromTwoVectors(con } /** \returns a random unit quaternion following a uniform distribution law on SO(3) - * - * \note The implementation is based on http://planning.cs.uiuc.edu/node198.html - */ + * + * \note The implementation is based on http://planning.cs.uiuc.edu/node198.html + */ template -EIGEN_DEVICE_FUNC Quaternion Quaternion::UnitRandom() +EIGEN_DEVICE_FUNC Quaternion Quaternion::UnitRandom() { EIGEN_USING_STD_MATH(sqrt) EIGEN_USING_STD_MATH(sin) EIGEN_USING_STD_MATH(cos) - const Scalar u1 = internal::random(0, 1), - u2 = internal::random(0, 2*EIGEN_PI), - u3 = internal::random(0, 2*EIGEN_PI); - const Scalar a = sqrt(1 - u1), - b = sqrt(u1); - return Quaternion (a * sin(u2), a * cos(u2), b * sin(u3), b * cos(u3)); + const Scalar u1 = internal::random(0, 1), u2 = internal::random(0, 2 * EIGEN_PI), + u3 = internal::random(0, 2 * EIGEN_PI); + const Scalar a = sqrt(1 - u1), b = sqrt(u1); + return Quaternion(a * sin(u2), a * cos(u2), b * sin(u3), b * cos(u3)); } /** Returns a quaternion representing a rotation between - * the two arbitrary vectors \a a and \a b. In other words, the built - * rotation represent a rotation sending the line of direction \a a - * to the line of direction \a b, both lines passing through the origin. - * - * \returns resulting quaternion - * - * Note that the two input vectors do \b not have to be normalized, and - * do not need to have the same norm. - */ + * the two arbitrary vectors \a a and \a b. In other words, the built + * rotation represent a rotation sending the line of direction \a a + * to the line of direction \a b, both lines passing through the origin. + * + * \returns resulting quaternion + * + * Note that the two input vectors do \b not have to be normalized, and + * do not need to have the same norm. + */ template template -EIGEN_DEVICE_FUNC Quaternion Quaternion::FromTwoVectors(const MatrixBase& a, const MatrixBase& b) +EIGEN_DEVICE_FUNC Quaternion Quaternion::FromTwoVectors(const MatrixBase &a, + const MatrixBase &b) { - Quaternion quat; - quat.setFromTwoVectors(a, b); - return quat; + Quaternion quat; + quat.setFromTwoVectors(a, b); + return quat; } /** \returns the multiplicative inverse of \c *this - * Note that in most cases, i.e., if you simply want the opposite rotation, - * and/or the quaternion is normalized, then it is enough to use the conjugate. - * - * \sa QuaternionBase::conjugate() - */ -template + * Note that in most cases, i.e., if you simply want the opposite rotation, + * and/or the quaternion is normalized, then it is enough to use the conjugate. + * + * \sa QuaternionBase::conjugate() + */ +template EIGEN_DEVICE_FUNC inline Quaternion::Scalar> QuaternionBase::inverse() const { // FIXME should this function be called multiplicativeInverse and conjugate() be called inverse() or opposite() ?? Scalar n2 = this->squaredNorm(); if (n2 > Scalar(0)) return Quaternion(conjugate().coeffs() / n2); - else - { + else { // return an invalid result to flag the error return Quaternion(Coefficients::Zero()); } @@ -676,54 +696,52 @@ EIGEN_DEVICE_FUNC inline Quaternion::Scalar> // Generic conjugate of a Quaternion namespace internal { -template struct quat_conj -{ - EIGEN_DEVICE_FUNC static EIGEN_STRONG_INLINE Quaternion run(const QuaternionBase& q){ - return Quaternion(q.w(),-q.x(),-q.y(),-q.z()); - } -}; -} - + template struct quat_conj + { + EIGEN_DEVICE_FUNC static EIGEN_STRONG_INLINE Quaternion run(const QuaternionBase &q) + { + return Quaternion(q.w(), -q.x(), -q.y(), -q.z()); + } + }; +}// namespace internal + /** \returns the conjugate of the \c *this which is equal to the multiplicative inverse - * if the quaternion is normalized. - * The conjugate of a quaternion represents the opposite rotation. - * - * \sa Quaternion2::inverse() - */ -template + * if the quaternion is normalized. + * The conjugate of a quaternion represents the opposite rotation. + * + * \sa Quaternion2::inverse() + */ +template EIGEN_DEVICE_FUNC inline Quaternion::Scalar> -QuaternionBase::conjugate() const + QuaternionBase::conjugate() const { - return internal::quat_conj::Scalar>::run(*this); - + return internal::quat_conj::Scalar>::run(*this); } /** \returns the angle (in radian) between two rotations - * \sa dot() - */ -template -template -EIGEN_DEVICE_FUNC inline typename internal::traits::Scalar -QuaternionBase::angularDistance(const QuaternionBase& other) const + * \sa dot() + */ +template +template +EIGEN_DEVICE_FUNC inline typename internal::traits::Scalar QuaternionBase::angularDistance( + const QuaternionBase &other) const { EIGEN_USING_STD_MATH(atan2) Quaternion d = (*this) * other.conjugate(); - return Scalar(2) * atan2( d.vec().norm(), numext::abs(d.w()) ); + return Scalar(2) * atan2(d.vec().norm(), numext::abs(d.w())); } - - + /** \returns the spherical linear interpolation between the two quaternions - * \c *this and \a other at the parameter \a t in [0;1]. - * - * This represents an interpolation for a constant motion between \c *this and \a other, - * see also http://en.wikipedia.org/wiki/Slerp. - */ -template -template -EIGEN_DEVICE_FUNC Quaternion::Scalar> -QuaternionBase::slerp(const Scalar& t, const QuaternionBase& other) const + * \c *this and \a other at the parameter \a t in [0;1]. + * + * This represents an interpolation for a constant motion between \c *this and \a other, + * see also http://en.wikipedia.org/wiki/Slerp. + */ +template +template +EIGEN_DEVICE_FUNC Quaternion::Scalar> QuaternionBase::slerp(const Scalar &t, + const QuaternionBase &other) const { EIGEN_USING_STD_MATH(acos) EIGEN_USING_STD_MATH(sin) @@ -734,81 +752,71 @@ QuaternionBase::slerp(const Scalar& t, const QuaternionBase=one) - { + if (absD >= one) { scale0 = Scalar(1) - t; scale1 = t; - } - else - { + } else { // theta is the angle between the 2 quaternions Scalar theta = acos(absD); Scalar sinTheta = sin(theta); - scale0 = sin( ( Scalar(1) - t ) * theta) / sinTheta; - scale1 = sin( ( t * theta) ) / sinTheta; + scale0 = sin((Scalar(1) - t) * theta) / sinTheta; + scale1 = sin((t * theta)) / sinTheta; } - if(d(scale0 * coeffs() + scale1 * other.coeffs()); } namespace internal { -// set from a rotation matrix -template -struct quaternionbase_assign_impl -{ - typedef typename Other::Scalar Scalar; - template EIGEN_DEVICE_FUNC static inline void run(QuaternionBase& q, const Other& a_mat) + // set from a rotation matrix + template struct quaternionbase_assign_impl { - const typename internal::nested_eval::type mat(a_mat); - EIGEN_USING_STD_MATH(sqrt) - // This algorithm comes from "Quaternion Calculus and Fast Animation", - // Ken Shoemake, 1987 SIGGRAPH course notes - Scalar t = mat.trace(); - if (t > Scalar(0)) - { - t = sqrt(t + Scalar(1.0)); - q.w() = Scalar(0.5)*t; - t = Scalar(0.5)/t; - q.x() = (mat.coeff(2,1) - mat.coeff(1,2)) * t; - q.y() = (mat.coeff(0,2) - mat.coeff(2,0)) * t; - q.z() = (mat.coeff(1,0) - mat.coeff(0,1)) * t; - } - else + typedef typename Other::Scalar Scalar; + template EIGEN_DEVICE_FUNC static inline void run(QuaternionBase &q, const Other &a_mat) { - Index i = 0; - if (mat.coeff(1,1) > mat.coeff(0,0)) - i = 1; - if (mat.coeff(2,2) > mat.coeff(i,i)) - i = 2; - Index j = (i+1)%3; - Index k = (j+1)%3; - - t = sqrt(mat.coeff(i,i)-mat.coeff(j,j)-mat.coeff(k,k) + Scalar(1.0)); - q.coeffs().coeffRef(i) = Scalar(0.5) * t; - t = Scalar(0.5)/t; - q.w() = (mat.coeff(k,j)-mat.coeff(j,k))*t; - q.coeffs().coeffRef(j) = (mat.coeff(j,i)+mat.coeff(i,j))*t; - q.coeffs().coeffRef(k) = (mat.coeff(k,i)+mat.coeff(i,k))*t; + const typename internal::nested_eval::type mat(a_mat); + EIGEN_USING_STD_MATH(sqrt) + // This algorithm comes from "Quaternion Calculus and Fast Animation", + // Ken Shoemake, 1987 SIGGRAPH course notes + Scalar t = mat.trace(); + if (t > Scalar(0)) { + t = sqrt(t + Scalar(1.0)); + q.w() = Scalar(0.5) * t; + t = Scalar(0.5) / t; + q.x() = (mat.coeff(2, 1) - mat.coeff(1, 2)) * t; + q.y() = (mat.coeff(0, 2) - mat.coeff(2, 0)) * t; + q.z() = (mat.coeff(1, 0) - mat.coeff(0, 1)) * t; + } else { + Index i = 0; + if (mat.coeff(1, 1) > mat.coeff(0, 0)) i = 1; + if (mat.coeff(2, 2) > mat.coeff(i, i)) i = 2; + Index j = (i + 1) % 3; + Index k = (j + 1) % 3; + + t = sqrt(mat.coeff(i, i) - mat.coeff(j, j) - mat.coeff(k, k) + Scalar(1.0)); + q.coeffs().coeffRef(i) = Scalar(0.5) * t; + t = Scalar(0.5) / t; + q.w() = (mat.coeff(k, j) - mat.coeff(j, k)) * t; + q.coeffs().coeffRef(j) = (mat.coeff(j, i) + mat.coeff(i, j)) * t; + q.coeffs().coeffRef(k) = (mat.coeff(k, i) + mat.coeff(i, k)) * t; + } } - } -}; + }; -// set from a vector of coefficients assumed to be a quaternion -template -struct quaternionbase_assign_impl -{ - typedef typename Other::Scalar Scalar; - template EIGEN_DEVICE_FUNC static inline void run(QuaternionBase& q, const Other& vec) + // set from a vector of coefficients assumed to be a quaternion + template struct quaternionbase_assign_impl { - q.coeffs() = vec; - } -}; + typedef typename Other::Scalar Scalar; + template EIGEN_DEVICE_FUNC static inline void run(QuaternionBase &q, const Other &vec) + { + q.coeffs() = vec; + } + }; -} // end namespace internal +}// end namespace internal -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_QUATERNION_H +#endif// EIGEN_QUATERNION_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Geometry/Rotation2D.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Geometry/Rotation2D.h index 884b7d0e..7c1e6cac 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Geometry/Rotation2D.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Geometry/Rotation2D.h @@ -10,66 +10,61 @@ #ifndef EIGEN_ROTATION2D_H #define EIGEN_ROTATION2D_H -namespace Eigen { +namespace Eigen { /** \geometry_module \ingroup Geometry_Module - * - * \class Rotation2D - * - * \brief Represents a rotation/orientation in a 2 dimensional space. - * - * \tparam _Scalar the scalar type, i.e., the type of the coefficients - * - * This class is equivalent to a single scalar representing a counter clock wise rotation - * as a single angle in radian. It provides some additional features such as the automatic - * conversion from/to a 2x2 rotation matrix. Moreover this class aims to provide a similar - * interface to Quaternion in order to facilitate the writing of generic algorithms - * dealing with rotations. - * - * \sa class Quaternion, class Transform - */ + * + * \class Rotation2D + * + * \brief Represents a rotation/orientation in a 2 dimensional space. + * + * \tparam _Scalar the scalar type, i.e., the type of the coefficients + * + * This class is equivalent to a single scalar representing a counter clock wise rotation + * as a single angle in radian. It provides some additional features such as the automatic + * conversion from/to a 2x2 rotation matrix. Moreover this class aims to provide a similar + * interface to Quaternion in order to facilitate the writing of generic algorithms + * dealing with rotations. + * + * \sa class Quaternion, class Transform + */ namespace internal { -template struct traits > -{ - typedef _Scalar Scalar; -}; -} // end namespace internal + template struct traits> + { + typedef _Scalar Scalar; + }; +}// end namespace internal -template -class Rotation2D : public RotationBase,2> +template class Rotation2D : public RotationBase, 2> { - typedef RotationBase,2> Base; + typedef RotationBase, 2> Base; public: - using Base::operator*; enum { Dim = 2 }; /** the scalar type of the coefficients */ typedef _Scalar Scalar; - typedef Matrix Vector2; - typedef Matrix Matrix2; + typedef Matrix Vector2; + typedef Matrix Matrix2; protected: - Scalar m_angle; public: - /** Construct a 2D counter clock wise rotation from the angle \a a in radian. */ - EIGEN_DEVICE_FUNC explicit inline Rotation2D(const Scalar& a) : m_angle(a) {} - + EIGEN_DEVICE_FUNC explicit inline Rotation2D(const Scalar &a) : m_angle(a) {} + /** Default constructor wihtout initialization. The represented rotation is undefined. */ EIGEN_DEVICE_FUNC Rotation2D() {} /** Construct a 2D rotation from a 2x2 rotation matrix \a mat. - * - * \sa fromRotationMatrix() - */ - template - EIGEN_DEVICE_FUNC explicit Rotation2D(const MatrixBase& m) + * + * \sa fromRotationMatrix() + */ + template EIGEN_DEVICE_FUNC explicit Rotation2D(const MatrixBase &m) { fromRotationMatrix(m.derived()); } @@ -78,19 +73,23 @@ class Rotation2D : public RotationBase,2> EIGEN_DEVICE_FUNC inline Scalar angle() const { return m_angle; } /** \returns a read-write reference to the rotation angle */ - EIGEN_DEVICE_FUNC inline Scalar& angle() { return m_angle; } - + EIGEN_DEVICE_FUNC inline Scalar &angle() { return m_angle; } + /** \returns the rotation angle in [0,2pi] */ - EIGEN_DEVICE_FUNC inline Scalar smallestPositiveAngle() const { - Scalar tmp = numext::fmod(m_angle,Scalar(2*EIGEN_PI)); - return tmpScalar(EIGEN_PI)) tmp -= Scalar(2*EIGEN_PI); - else if(tmp<-Scalar(EIGEN_PI)) tmp += Scalar(2*EIGEN_PI); + EIGEN_DEVICE_FUNC inline Scalar smallestAngle() const + { + Scalar tmp = numext::fmod(m_angle, Scalar(2 * EIGEN_PI)); + if (tmp > Scalar(EIGEN_PI)) + tmp -= Scalar(2 * EIGEN_PI); + else if (tmp < -Scalar(EIGEN_PI)) + tmp += Scalar(2 * EIGEN_PI); return tmp; } @@ -98,53 +97,59 @@ class Rotation2D : public RotationBase,2> EIGEN_DEVICE_FUNC inline Rotation2D inverse() const { return Rotation2D(-m_angle); } /** Concatenates two rotations */ - EIGEN_DEVICE_FUNC inline Rotation2D operator*(const Rotation2D& other) const - { return Rotation2D(m_angle + other.m_angle); } + EIGEN_DEVICE_FUNC inline Rotation2D operator*(const Rotation2D &other) const + { + return Rotation2D(m_angle + other.m_angle); + } /** Concatenates two rotations */ - EIGEN_DEVICE_FUNC inline Rotation2D& operator*=(const Rotation2D& other) - { m_angle += other.m_angle; return *this; } + EIGEN_DEVICE_FUNC inline Rotation2D &operator*=(const Rotation2D &other) + { + m_angle += other.m_angle; + return *this; + } /** Applies the rotation to a 2D vector */ - EIGEN_DEVICE_FUNC Vector2 operator* (const Vector2& vec) const - { return toRotationMatrix() * vec; } - - template - EIGEN_DEVICE_FUNC Rotation2D& fromRotationMatrix(const MatrixBase& m); + EIGEN_DEVICE_FUNC Vector2 operator*(const Vector2 &vec) const { return toRotationMatrix() * vec; } + + template EIGEN_DEVICE_FUNC Rotation2D &fromRotationMatrix(const MatrixBase &m); EIGEN_DEVICE_FUNC Matrix2 toRotationMatrix() const; /** Set \c *this from a 2x2 rotation matrix \a mat. - * In other words, this function extract the rotation angle from the rotation matrix. - * - * This method is an alias for fromRotationMatrix() - * - * \sa fromRotationMatrix() - */ - template - EIGEN_DEVICE_FUNC Rotation2D& operator=(const MatrixBase& m) - { return fromRotationMatrix(m.derived()); } + * In other words, this function extract the rotation angle from the rotation matrix. + * + * This method is an alias for fromRotationMatrix() + * + * \sa fromRotationMatrix() + */ + template EIGEN_DEVICE_FUNC Rotation2D &operator=(const MatrixBase &m) + { + return fromRotationMatrix(m.derived()); + } /** \returns the spherical interpolation between \c *this and \a other using - * parameter \a t. It is in fact equivalent to a linear interpolation. - */ - EIGEN_DEVICE_FUNC inline Rotation2D slerp(const Scalar& t, const Rotation2D& other) const + * parameter \a t. It is in fact equivalent to a linear interpolation. + */ + EIGEN_DEVICE_FUNC inline Rotation2D slerp(const Scalar &t, const Rotation2D &other) const { - Scalar dist = Rotation2D(other.m_angle-m_angle).smallestAngle(); - return Rotation2D(m_angle + dist*t); + Scalar dist = Rotation2D(other.m_angle - m_angle).smallestAngle(); + return Rotation2D(m_angle + dist * t); } /** \returns \c *this with scalar type casted to \a NewScalarType - * - * Note that if \a NewScalarType is equal to the current scalar type of \c *this - * then this function smartly returns a const reference to \c *this. - */ + * + * Note that if \a NewScalarType is equal to the current scalar type of \c *this + * then this function smartly returns a const reference to \c *this. + */ template - EIGEN_DEVICE_FUNC inline typename internal::cast_return_type >::type cast() const - { return typename internal::cast_return_type >::type(*this); } + EIGEN_DEVICE_FUNC inline typename internal::cast_return_type>::type cast() const + { + return typename internal::cast_return_type>::type(*this); + } /** Copy constructor with scalar type conversion */ template - EIGEN_DEVICE_FUNC inline explicit Rotation2D(const Rotation2D& other) + EIGEN_DEVICE_FUNC inline explicit Rotation2D(const Rotation2D &other) { m_angle = Scalar(other.angle()); } @@ -152,40 +157,42 @@ class Rotation2D : public RotationBase,2> EIGEN_DEVICE_FUNC static inline Rotation2D Identity() { return Rotation2D(0); } /** \returns \c true if \c *this is approximately equal to \a other, within the precision - * determined by \a prec. - * - * \sa MatrixBase::isApprox() */ - EIGEN_DEVICE_FUNC bool isApprox(const Rotation2D& other, const typename NumTraits::Real& prec = NumTraits::dummy_precision()) const - { return internal::isApprox(m_angle,other.m_angle, prec); } - + * determined by \a prec. + * + * \sa MatrixBase::isApprox() */ + EIGEN_DEVICE_FUNC bool isApprox(const Rotation2D &other, + const typename NumTraits::Real &prec = NumTraits::dummy_precision()) const + { + return internal::isApprox(m_angle, other.m_angle, prec); + } }; /** \ingroup Geometry_Module - * single precision 2D rotation type */ + * single precision 2D rotation type */ typedef Rotation2D Rotation2Df; /** \ingroup Geometry_Module - * double precision 2D rotation type */ + * double precision 2D rotation type */ typedef Rotation2D Rotation2Dd; /** Set \c *this from a 2x2 rotation matrix \a mat. - * In other words, this function extract the rotation angle - * from the rotation matrix. - */ + * In other words, this function extract the rotation angle + * from the rotation matrix. + */ template template -EIGEN_DEVICE_FUNC Rotation2D& Rotation2D::fromRotationMatrix(const MatrixBase& mat) +EIGEN_DEVICE_FUNC Rotation2D &Rotation2D::fromRotationMatrix(const MatrixBase &mat) { EIGEN_USING_STD_MATH(atan2) - EIGEN_STATIC_ASSERT(Derived::RowsAtCompileTime==2 && Derived::ColsAtCompileTime==2,YOU_MADE_A_PROGRAMMING_MISTAKE) - m_angle = atan2(mat.coeff(1,0), mat.coeff(0,0)); + EIGEN_STATIC_ASSERT( + Derived::RowsAtCompileTime == 2 && Derived::ColsAtCompileTime == 2, YOU_MADE_A_PROGRAMMING_MISTAKE) + m_angle = atan2(mat.coeff(1, 0), mat.coeff(0, 0)); return *this; } /** Constructs and \returns an equivalent 2x2 rotation matrix. - */ + */ template -typename Rotation2D::Matrix2 -EIGEN_DEVICE_FUNC Rotation2D::toRotationMatrix(void) const +typename Rotation2D::Matrix2 EIGEN_DEVICE_FUNC Rotation2D::toRotationMatrix(void) const { EIGEN_USING_STD_MATH(sin) EIGEN_USING_STD_MATH(cos) @@ -194,6 +201,6 @@ EIGEN_DEVICE_FUNC Rotation2D::toRotationMatrix(void) const return (Matrix2() << cosA, -sinA, sinA, cosA).finished(); } -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_ROTATION2D_H +#endif// EIGEN_ROTATION2D_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Geometry/RotationBase.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Geometry/RotationBase.h index f0ee0bd0..69bb10f0 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Geometry/RotationBase.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Geometry/RotationBase.h @@ -10,197 +10,213 @@ #ifndef EIGEN_ROTATIONBASE_H #define EIGEN_ROTATIONBASE_H -namespace Eigen { +namespace Eigen { // forward declaration namespace internal { -template -struct rotation_base_generic_product_selector; + template + struct rotation_base_generic_product_selector; } /** \class RotationBase - * - * \brief Common base class for compact rotation representations - * - * \tparam Derived is the derived type, i.e., a rotation type - * \tparam _Dim the dimension of the space - */ -template -class RotationBase + * + * \brief Common base class for compact rotation representations + * + * \tparam Derived is the derived type, i.e., a rotation type + * \tparam _Dim the dimension of the space + */ +template class RotationBase { - public: - enum { Dim = _Dim }; - /** the scalar type of the coefficients */ - typedef typename internal::traits::Scalar Scalar; - - /** corresponding linear transformation matrix type */ - typedef Matrix RotationMatrixType; - typedef Matrix VectorType; - - public: - EIGEN_DEVICE_FUNC inline const Derived& derived() const { return *static_cast(this); } - EIGEN_DEVICE_FUNC inline Derived& derived() { return *static_cast(this); } - - /** \returns an equivalent rotation matrix */ - EIGEN_DEVICE_FUNC inline RotationMatrixType toRotationMatrix() const { return derived().toRotationMatrix(); } - - /** \returns an equivalent rotation matrix - * This function is added to be conform with the Transform class' naming scheme. - */ - EIGEN_DEVICE_FUNC inline RotationMatrixType matrix() const { return derived().toRotationMatrix(); } - - /** \returns the inverse rotation */ - EIGEN_DEVICE_FUNC inline Derived inverse() const { return derived().inverse(); } - - /** \returns the concatenation of the rotation \c *this with a translation \a t */ - EIGEN_DEVICE_FUNC inline Transform operator*(const Translation& t) const - { return Transform(*this) * t; } - - /** \returns the concatenation of the rotation \c *this with a uniform scaling \a s */ - EIGEN_DEVICE_FUNC inline RotationMatrixType operator*(const UniformScaling& s) const - { return toRotationMatrix() * s.factor(); } - - /** \returns the concatenation of the rotation \c *this with a generic expression \a e - * \a e can be: - * - a DimxDim linear transformation matrix - * - a DimxDim diagonal matrix (axis aligned scaling) - * - a vector of size Dim - */ - template - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE typename internal::rotation_base_generic_product_selector::ReturnType - operator*(const EigenBase& e) const - { return internal::rotation_base_generic_product_selector::run(derived(), e.derived()); } - - /** \returns the concatenation of a linear transformation \a l with the rotation \a r */ - template friend - EIGEN_DEVICE_FUNC inline RotationMatrixType operator*(const EigenBase& l, const Derived& r) - { return l.derived() * r.toRotationMatrix(); } - - /** \returns the concatenation of a scaling \a l with the rotation \a r */ - EIGEN_DEVICE_FUNC friend inline Transform operator*(const DiagonalMatrix& l, const Derived& r) - { - Transform res(r); - res.linear().applyOnTheLeft(l); - return res; - } +public: + enum { Dim = _Dim }; + /** the scalar type of the coefficients */ + typedef typename internal::traits::Scalar Scalar; - /** \returns the concatenation of the rotation \c *this with a transformation \a t */ - template - EIGEN_DEVICE_FUNC inline Transform operator*(const Transform& t) const - { return toRotationMatrix() * t; } + /** corresponding linear transformation matrix type */ + typedef Matrix RotationMatrixType; + typedef Matrix VectorType; - template - EIGEN_DEVICE_FUNC inline VectorType _transformVector(const OtherVectorType& v) const - { return toRotationMatrix() * v; } -}; +public: + EIGEN_DEVICE_FUNC inline const Derived &derived() const { return *static_cast(this); } + EIGEN_DEVICE_FUNC inline Derived &derived() { return *static_cast(this); } -namespace internal { + /** \returns an equivalent rotation matrix */ + EIGEN_DEVICE_FUNC inline RotationMatrixType toRotationMatrix() const { return derived().toRotationMatrix(); } -// implementation of the generic product rotation * matrix -template -struct rotation_base_generic_product_selector -{ - enum { Dim = RotationDerived::Dim }; - typedef Matrix ReturnType; - EIGEN_DEVICE_FUNC static inline ReturnType run(const RotationDerived& r, const MatrixType& m) - { return r.toRotationMatrix() * m; } -}; + /** \returns an equivalent rotation matrix + * This function is added to be conform with the Transform class' naming scheme. + */ + EIGEN_DEVICE_FUNC inline RotationMatrixType matrix() const { return derived().toRotationMatrix(); } -template -struct rotation_base_generic_product_selector< RotationDerived, DiagonalMatrix, false > -{ - typedef Transform ReturnType; - EIGEN_DEVICE_FUNC static inline ReturnType run(const RotationDerived& r, const DiagonalMatrix& m) + /** \returns the inverse rotation */ + EIGEN_DEVICE_FUNC inline Derived inverse() const { return derived().inverse(); } + + /** \returns the concatenation of the rotation \c *this with a translation \a t */ + EIGEN_DEVICE_FUNC inline Transform operator*(const Translation &t) const + { + return Transform(*this) * t; + } + + /** \returns the concatenation of the rotation \c *this with a uniform scaling \a s */ + EIGEN_DEVICE_FUNC inline RotationMatrixType operator*(const UniformScaling &s) const { - ReturnType res(r); - res.linear() *= m; + return toRotationMatrix() * s.factor(); + } + + /** \returns the concatenation of the rotation \c *this with a generic expression \a e + * \a e can be: + * - a DimxDim linear transformation matrix + * - a DimxDim diagonal matrix (axis aligned scaling) + * - a vector of size Dim + */ + template + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE typename internal:: + rotation_base_generic_product_selector::ReturnType + operator*(const EigenBase &e) const + { + return internal::rotation_base_generic_product_selector::run(derived(), e.derived()); + } + + /** \returns the concatenation of a linear transformation \a l with the rotation \a r */ + template + friend EIGEN_DEVICE_FUNC inline RotationMatrixType operator*(const EigenBase &l, const Derived &r) + { + return l.derived() * r.toRotationMatrix(); + } + + /** \returns the concatenation of a scaling \a l with the rotation \a r */ + EIGEN_DEVICE_FUNC friend inline Transform operator*(const DiagonalMatrix &l, + const Derived &r) + { + Transform res(r); + res.linear().applyOnTheLeft(l); return res; } -}; -template -struct rotation_base_generic_product_selector -{ - enum { Dim = RotationDerived::Dim }; - typedef Matrix ReturnType; - EIGEN_DEVICE_FUNC static EIGEN_STRONG_INLINE ReturnType run(const RotationDerived& r, const OtherVectorType& v) + /** \returns the concatenation of the rotation \c *this with a transformation \a t */ + template + EIGEN_DEVICE_FUNC inline Transform operator*(const Transform &t) const + { + return toRotationMatrix() * t; + } + + template + EIGEN_DEVICE_FUNC inline VectorType _transformVector(const OtherVectorType &v) const { - return r._transformVector(v); + return toRotationMatrix() * v; } }; -} // end namespace internal +namespace internal { + + // implementation of the generic product rotation * matrix + template + struct rotation_base_generic_product_selector + { + enum { Dim = RotationDerived::Dim }; + typedef Matrix ReturnType; + EIGEN_DEVICE_FUNC static inline ReturnType run(const RotationDerived &r, const MatrixType &m) + { + return r.toRotationMatrix() * m; + } + }; + + template + struct rotation_base_generic_product_selector, false> + { + typedef Transform ReturnType; + EIGEN_DEVICE_FUNC static inline ReturnType run(const RotationDerived &r, + const DiagonalMatrix &m) + { + ReturnType res(r); + res.linear() *= m; + return res; + } + }; + + template + struct rotation_base_generic_product_selector + { + enum { Dim = RotationDerived::Dim }; + typedef Matrix ReturnType; + EIGEN_DEVICE_FUNC static EIGEN_STRONG_INLINE ReturnType run(const RotationDerived &r, const OtherVectorType &v) + { + return r._transformVector(v); + } + }; + +}// end namespace internal /** \geometry_module - * - * \brief Constructs a Dim x Dim rotation matrix from the rotation \a r - */ + * + * \brief Constructs a Dim x Dim rotation matrix from the rotation \a r + */ template template -EIGEN_DEVICE_FUNC Matrix<_Scalar, _Rows, _Cols, _Storage, _MaxRows, _MaxCols> -::Matrix(const RotationBase& r) +EIGEN_DEVICE_FUNC Matrix<_Scalar, _Rows, _Cols, _Storage, _MaxRows, _MaxCols>::Matrix( + const RotationBase &r) { - EIGEN_STATIC_ASSERT_MATRIX_SPECIFIC_SIZE(Matrix,int(OtherDerived::Dim),int(OtherDerived::Dim)) + EIGEN_STATIC_ASSERT_MATRIX_SPECIFIC_SIZE(Matrix, int(OtherDerived::Dim), int(OtherDerived::Dim)) *this = r.toRotationMatrix(); } /** \geometry_module - * - * \brief Set a Dim x Dim rotation matrix from the rotation \a r - */ + * + * \brief Set a Dim x Dim rotation matrix from the rotation \a r + */ template template -EIGEN_DEVICE_FUNC Matrix<_Scalar, _Rows, _Cols, _Storage, _MaxRows, _MaxCols>& -Matrix<_Scalar, _Rows, _Cols, _Storage, _MaxRows, _MaxCols> -::operator=(const RotationBase& r) +EIGEN_DEVICE_FUNC Matrix<_Scalar, _Rows, _Cols, _Storage, _MaxRows, _MaxCols> & + Matrix<_Scalar, _Rows, _Cols, _Storage, _MaxRows, _MaxCols>::operator=( + const RotationBase &r) { - EIGEN_STATIC_ASSERT_MATRIX_SPECIFIC_SIZE(Matrix,int(OtherDerived::Dim),int(OtherDerived::Dim)) + EIGEN_STATIC_ASSERT_MATRIX_SPECIFIC_SIZE(Matrix, int(OtherDerived::Dim), int(OtherDerived::Dim)) return *this = r.toRotationMatrix(); } namespace internal { -/** \internal - * - * Helper function to return an arbitrary rotation object to a rotation matrix. - * - * \tparam Scalar the numeric type of the matrix coefficients - * \tparam Dim the dimension of the current space - * - * It returns a Dim x Dim fixed size matrix. - * - * Default specializations are provided for: - * - any scalar type (2D), - * - any matrix expression, - * - any type based on RotationBase (e.g., Quaternion, AngleAxis, Rotation2D) - * - * Currently toRotationMatrix is only used by Transform. - * - * \sa class Transform, class Rotation2D, class Quaternion, class AngleAxis - */ -template -EIGEN_DEVICE_FUNC static inline Matrix toRotationMatrix(const Scalar& s) -{ - EIGEN_STATIC_ASSERT(Dim==2,YOU_MADE_A_PROGRAMMING_MISTAKE) - return Rotation2D(s).toRotationMatrix(); -} + /** \internal + * + * Helper function to return an arbitrary rotation object to a rotation matrix. + * + * \tparam Scalar the numeric type of the matrix coefficients + * \tparam Dim the dimension of the current space + * + * It returns a Dim x Dim fixed size matrix. + * + * Default specializations are provided for: + * - any scalar type (2D), + * - any matrix expression, + * - any type based on RotationBase (e.g., Quaternion, AngleAxis, Rotation2D) + * + * Currently toRotationMatrix is only used by Transform. + * + * \sa class Transform, class Rotation2D, class Quaternion, class AngleAxis + */ + template + EIGEN_DEVICE_FUNC static inline Matrix toRotationMatrix(const Scalar &s) + { + EIGEN_STATIC_ASSERT(Dim == 2, YOU_MADE_A_PROGRAMMING_MISTAKE) + return Rotation2D(s).toRotationMatrix(); + } -template -EIGEN_DEVICE_FUNC static inline Matrix toRotationMatrix(const RotationBase& r) -{ - return r.toRotationMatrix(); -} + template + EIGEN_DEVICE_FUNC static inline Matrix toRotationMatrix(const RotationBase &r) + { + return r.toRotationMatrix(); + } -template -EIGEN_DEVICE_FUNC static inline const MatrixBase& toRotationMatrix(const MatrixBase& mat) -{ - EIGEN_STATIC_ASSERT(OtherDerived::RowsAtCompileTime==Dim && OtherDerived::ColsAtCompileTime==Dim, - YOU_MADE_A_PROGRAMMING_MISTAKE) - return mat; -} + template + EIGEN_DEVICE_FUNC static inline const MatrixBase &toRotationMatrix(const MatrixBase &mat) + { + EIGEN_STATIC_ASSERT( + OtherDerived::RowsAtCompileTime == Dim && OtherDerived::ColsAtCompileTime == Dim, YOU_MADE_A_PROGRAMMING_MISTAKE) + return mat; + } -} // end namespace internal +}// end namespace internal -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_ROTATIONBASE_H +#endif// EIGEN_ROTATIONBASE_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Geometry/Scaling.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Geometry/Scaling.h old mode 100755 new mode 100644 index f58ca03d..67a73706 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Geometry/Scaling.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Geometry/Scaling.h @@ -10,59 +10,58 @@ #ifndef EIGEN_SCALING_H #define EIGEN_SCALING_H -namespace Eigen { +namespace Eigen { /** \geometry_module \ingroup Geometry_Module - * - * \class Scaling - * - * \brief Represents a generic uniform scaling transformation - * - * \tparam _Scalar the scalar type, i.e., the type of the coefficients. - * - * This class represent a uniform scaling transformation. It is the return - * type of Scaling(Scalar), and most of the time this is the only way it - * is used. In particular, this class is not aimed to be used to store a scaling transformation, - * but rather to make easier the constructions and updates of Transform objects. - * - * To represent an axis aligned scaling, use the DiagonalMatrix class. - * - * \sa Scaling(), class DiagonalMatrix, MatrixBase::asDiagonal(), class Translation, class Transform - */ -template -class UniformScaling + * + * \class Scaling + * + * \brief Represents a generic uniform scaling transformation + * + * \tparam _Scalar the scalar type, i.e., the type of the coefficients. + * + * This class represent a uniform scaling transformation. It is the return + * type of Scaling(Scalar), and most of the time this is the only way it + * is used. In particular, this class is not aimed to be used to store a scaling transformation, + * but rather to make easier the constructions and updates of Transform objects. + * + * To represent an axis aligned scaling, use the DiagonalMatrix class. + * + * \sa Scaling(), class DiagonalMatrix, MatrixBase::asDiagonal(), class Translation, class Transform + */ +template class UniformScaling { public: /** the scalar type of the coefficients */ typedef _Scalar Scalar; protected: - Scalar m_factor; public: - /** Default constructor without initialization. */ UniformScaling() {} /** Constructs and initialize a uniform scaling transformation */ - explicit inline UniformScaling(const Scalar& s) : m_factor(s) {} + explicit inline UniformScaling(const Scalar &s) : m_factor(s) {} - inline const Scalar& factor() const { return m_factor; } - inline Scalar& factor() { return m_factor; } + inline const Scalar &factor() const { return m_factor; } + inline Scalar &factor() { return m_factor; } /** Concatenates two uniform scaling */ - inline UniformScaling operator* (const UniformScaling& other) const - { return UniformScaling(m_factor * other.factor()); } + inline UniformScaling operator*(const UniformScaling &other) const + { + return UniformScaling(m_factor * other.factor()); + } /** Concatenates a uniform scaling and a translation */ - template - inline Transform operator* (const Translation& t) const; + template inline Transform operator*(const Translation &t) const; /** Concatenates a uniform scaling and an affine transformation */ template - inline Transform operator* (const Transform& t) const + inline Transform operator*( + const Transform &t) const { - Transform res = t; + Transform res = t; res.prescale(factor()); return res; } @@ -70,101 +69,113 @@ class UniformScaling /** Concatenates a uniform scaling and a linear transformation matrix */ // TODO returns an expression template - inline typename internal::plain_matrix_type::type operator* (const MatrixBase& other) const - { return other * m_factor; } + inline typename internal::plain_matrix_type::type operator*(const MatrixBase &other) const + { + return other * m_factor; + } - template - inline Matrix operator*(const RotationBase& r) const - { return r.toRotationMatrix() * m_factor; } + template + inline Matrix operator*(const RotationBase &r) const + { + return r.toRotationMatrix() * m_factor; + } /** \returns the inverse scaling */ - inline UniformScaling inverse() const - { return UniformScaling(Scalar(1)/m_factor); } + inline UniformScaling inverse() const { return UniformScaling(Scalar(1) / m_factor); } /** \returns \c *this with scalar type casted to \a NewScalarType - * - * Note that if \a NewScalarType is equal to the current scalar type of \c *this - * then this function smartly returns a const reference to \c *this. - */ - template - inline UniformScaling cast() const - { return UniformScaling(NewScalarType(m_factor)); } + * + * Note that if \a NewScalarType is equal to the current scalar type of \c *this + * then this function smartly returns a const reference to \c *this. + */ + template inline UniformScaling cast() const + { + return UniformScaling(NewScalarType(m_factor)); + } /** Copy constructor with scalar type conversion */ - template - inline explicit UniformScaling(const UniformScaling& other) - { m_factor = Scalar(other.factor()); } + template inline explicit UniformScaling(const UniformScaling &other) + { + m_factor = Scalar(other.factor()); + } /** \returns \c true if \c *this is approximately equal to \a other, within the precision - * determined by \a prec. - * - * \sa MatrixBase::isApprox() */ - bool isApprox(const UniformScaling& other, const typename NumTraits::Real& prec = NumTraits::dummy_precision()) const - { return internal::isApprox(m_factor, other.factor(), prec); } - + * determined by \a prec. + * + * \sa MatrixBase::isApprox() */ + bool isApprox(const UniformScaling &other, + const typename NumTraits::Real &prec = NumTraits::dummy_precision()) const + { + return internal::isApprox(m_factor, other.factor(), prec); + } }; /** \addtogroup Geometry_Module */ //@{ /** Concatenates a linear transformation matrix and a uniform scaling - * \relates UniformScaling - */ + * \relates UniformScaling + */ // NOTE this operator is defiend in MatrixBase and not as a friend function // of UniformScaling to fix an internal crash of Intel's ICC -template -EIGEN_EXPR_BINARYOP_SCALAR_RETURN_TYPE(Derived,Scalar,product) -operator*(const MatrixBase& matrix, const UniformScaling& s) -{ return matrix.derived() * s.factor(); } +template +EIGEN_EXPR_BINARYOP_SCALAR_RETURN_TYPE(Derived, Scalar, product) +operator*(const MatrixBase &matrix, const UniformScaling &s) +{ + return matrix.derived() * s.factor(); +} /** Constructs a uniform scaling from scale factor \a s */ inline UniformScaling Scaling(float s) { return UniformScaling(s); } /** Constructs a uniform scaling from scale factor \a s */ inline UniformScaling Scaling(double s) { return UniformScaling(s); } /** Constructs a uniform scaling from scale factor \a s */ -template -inline UniformScaling > Scaling(const std::complex& s) -{ return UniformScaling >(s); } +template inline UniformScaling> Scaling(const std::complex &s) +{ + return UniformScaling>(s); +} /** Constructs a 2D axis aligned scaling */ -template -inline DiagonalMatrix Scaling(const Scalar& sx, const Scalar& sy) -{ return DiagonalMatrix(sx, sy); } +template inline DiagonalMatrix Scaling(const Scalar &sx, const Scalar &sy) +{ + return DiagonalMatrix(sx, sy); +} /** Constructs a 3D axis aligned scaling */ -template -inline DiagonalMatrix Scaling(const Scalar& sx, const Scalar& sy, const Scalar& sz) -{ return DiagonalMatrix(sx, sy, sz); } +template inline DiagonalMatrix Scaling(const Scalar &sx, const Scalar &sy, const Scalar &sz) +{ + return DiagonalMatrix(sx, sy, sz); +} /** Constructs an axis aligned scaling expression from vector expression \a coeffs - * This is an alias for coeffs.asDiagonal() - */ -template -inline const DiagonalWrapper Scaling(const MatrixBase& coeffs) -{ return coeffs.asDiagonal(); } + * This is an alias for coeffs.asDiagonal() + */ +template inline const DiagonalWrapper Scaling(const MatrixBase &coeffs) +{ + return coeffs.asDiagonal(); +} /** \deprecated */ typedef DiagonalMatrix AlignedScaling2f; /** \deprecated */ -typedef DiagonalMatrix AlignedScaling2d; +typedef DiagonalMatrix AlignedScaling2d; /** \deprecated */ typedef DiagonalMatrix AlignedScaling3f; /** \deprecated */ -typedef DiagonalMatrix AlignedScaling3d; +typedef DiagonalMatrix AlignedScaling3d; //@} template template -inline Transform -UniformScaling::operator* (const Translation& t) const +inline Transform UniformScaling::operator*(const Translation &t) const { - Transform res; + Transform res; res.matrix().setZero(); res.linear().diagonal().fill(factor()); res.translation() = factor() * t.vector(); - res(Dim,Dim) = Scalar(1); + res(Dim, Dim) = Scalar(1); return res; } -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_SCALING_H +#endif// EIGEN_SCALING_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Geometry/Transform.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Geometry/Transform.h index 3f31ee45..fbe7ede0 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Geometry/Transform.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Geometry/Transform.h @@ -12,344 +12,335 @@ #ifndef EIGEN_TRANSFORM_H #define EIGEN_TRANSFORM_H -namespace Eigen { +namespace Eigen { namespace internal { -template -struct transform_traits -{ - enum + template struct transform_traits { - Dim = Transform::Dim, - HDim = Transform::HDim, - Mode = Transform::Mode, - IsProjective = (int(Mode)==int(Projective)) + enum { + Dim = Transform::Dim, + HDim = Transform::HDim, + Mode = Transform::Mode, + IsProjective = (int(Mode) == int(Projective)) + }; }; -}; -template< typename TransformType, - typename MatrixType, - int Case = transform_traits::IsProjective ? 0 - : int(MatrixType::RowsAtCompileTime) == int(transform_traits::HDim) ? 1 - : 2, - int RhsCols = MatrixType::ColsAtCompileTime> -struct transform_right_product_impl; - -template< typename Other, - int Mode, - int Options, - int Dim, - int HDim, - int OtherRows=Other::RowsAtCompileTime, - int OtherCols=Other::ColsAtCompileTime> -struct transform_left_product_impl; - -template< typename Lhs, - typename Rhs, - bool AnyProjective = - transform_traits::IsProjective || - transform_traits::IsProjective> -struct transform_transform_product_impl; - -template< typename Other, - int Mode, - int Options, - int Dim, - int HDim, - int OtherRows=Other::RowsAtCompileTime, - int OtherCols=Other::ColsAtCompileTime> -struct transform_construct_from_matrix; - -template struct transform_take_affine_part; - -template -struct traits > -{ - typedef _Scalar Scalar; - typedef Eigen::Index StorageIndex; - typedef Dense StorageKind; - enum { - Dim1 = _Dim==Dynamic ? _Dim : _Dim + 1, - RowsAtCompileTime = _Mode==Projective ? Dim1 : _Dim, - ColsAtCompileTime = Dim1, - MaxRowsAtCompileTime = RowsAtCompileTime, - MaxColsAtCompileTime = ColsAtCompileTime, - Flags = 0 + template::IsProjective ? 0 + : int(MatrixType::RowsAtCompileTime) == int(transform_traits::HDim) ? 1 + : 2, + int RhsCols = MatrixType::ColsAtCompileTime> + struct transform_right_product_impl; + + template + struct transform_left_product_impl; + + template::IsProjective || transform_traits::IsProjective> + struct transform_transform_product_impl; + + template + struct transform_construct_from_matrix; + + template struct transform_take_affine_part; + + template struct traits> + { + typedef _Scalar Scalar; + typedef Eigen::Index StorageIndex; + typedef Dense StorageKind; + enum { + Dim1 = _Dim == Dynamic ? _Dim : _Dim + 1, + RowsAtCompileTime = _Mode == Projective ? Dim1 : _Dim, + ColsAtCompileTime = Dim1, + MaxRowsAtCompileTime = RowsAtCompileTime, + MaxColsAtCompileTime = ColsAtCompileTime, + Flags = 0 + }; }; -}; -template struct transform_make_affine; + template struct transform_make_affine; -} // end namespace internal +}// end namespace internal /** \geometry_module \ingroup Geometry_Module - * - * \class Transform - * - * \brief Represents an homogeneous transformation in a N dimensional space - * - * \tparam _Scalar the scalar type, i.e., the type of the coefficients - * \tparam _Dim the dimension of the space - * \tparam _Mode the type of the transformation. Can be: - * - #Affine: the transformation is stored as a (Dim+1)^2 matrix, - * where the last row is assumed to be [0 ... 0 1]. - * - #AffineCompact: the transformation is stored as a (Dim)x(Dim+1) matrix. - * - #Projective: the transformation is stored as a (Dim+1)^2 matrix - * without any assumption. - * \tparam _Options has the same meaning as in class Matrix. It allows to specify DontAlign and/or RowMajor. - * These Options are passed directly to the underlying matrix type. - * - * The homography is internally represented and stored by a matrix which - * is available through the matrix() method. To understand the behavior of - * this class you have to think a Transform object as its internal - * matrix representation. The chosen convention is right multiply: - * - * \code v' = T * v \endcode - * - * Therefore, an affine transformation matrix M is shaped like this: - * - * \f$ \left( \begin{array}{cc} - * linear & translation\\ - * 0 ... 0 & 1 - * \end{array} \right) \f$ - * - * Note that for a projective transformation the last row can be anything, - * and then the interpretation of different parts might be sightly different. - * - * However, unlike a plain matrix, the Transform class provides many features - * simplifying both its assembly and usage. In particular, it can be composed - * with any other transformations (Transform,Translation,RotationBase,DiagonalMatrix) - * and can be directly used to transform implicit homogeneous vectors. All these - * operations are handled via the operator*. For the composition of transformations, - * its principle consists to first convert the right/left hand sides of the product - * to a compatible (Dim+1)^2 matrix and then perform a pure matrix product. - * Of course, internally, operator* tries to perform the minimal number of operations - * according to the nature of each terms. Likewise, when applying the transform - * to points, the latters are automatically promoted to homogeneous vectors - * before doing the matrix product. The conventions to homogeneous representations - * are performed as follow: - * - * \b Translation t (Dim)x(1): - * \f$ \left( \begin{array}{cc} - * I & t \\ - * 0\,...\,0 & 1 - * \end{array} \right) \f$ - * - * \b Rotation R (Dim)x(Dim): - * \f$ \left( \begin{array}{cc} - * R & 0\\ - * 0\,...\,0 & 1 - * \end{array} \right) \f$ - * - * \b Scaling \b DiagonalMatrix S (Dim)x(Dim): - * \f$ \left( \begin{array}{cc} - * S & 0\\ - * 0\,...\,0 & 1 - * \end{array} \right) \f$ - * - * \b Column \b point v (Dim)x(1): - * \f$ \left( \begin{array}{c} - * v\\ - * 1 - * \end{array} \right) \f$ - * - * \b Set \b of \b column \b points V1...Vn (Dim)x(n): - * \f$ \left( \begin{array}{ccc} - * v_1 & ... & v_n\\ - * 1 & ... & 1 - * \end{array} \right) \f$ - * - * The concatenation of a Transform object with any kind of other transformation - * always returns a Transform object. - * - * A little exception to the "as pure matrix product" rule is the case of the - * transformation of non homogeneous vectors by an affine transformation. In - * that case the last matrix row can be ignored, and the product returns non - * homogeneous vectors. - * - * Since, for instance, a Dim x Dim matrix is interpreted as a linear transformation, - * it is not possible to directly transform Dim vectors stored in a Dim x Dim matrix. - * The solution is either to use a Dim x Dynamic matrix or explicitly request a - * vector transformation by making the vector homogeneous: - * \code - * m' = T * m.colwise().homogeneous(); - * \endcode - * Note that there is zero overhead. - * - * Conversion methods from/to Qt's QMatrix and QTransform are available if the - * preprocessor token EIGEN_QT_SUPPORT is defined. - * - * This class can be extended with the help of the plugin mechanism described on the page - * \ref TopicCustomizing_Plugins by defining the preprocessor symbol \c EIGEN_TRANSFORM_PLUGIN. - * - * \sa class Matrix, class Quaternion - */ -template -class Transform + * + * \class Transform + * + * \brief Represents an homogeneous transformation in a N dimensional space + * + * \tparam _Scalar the scalar type, i.e., the type of the coefficients + * \tparam _Dim the dimension of the space + * \tparam _Mode the type of the transformation. Can be: + * - #Affine: the transformation is stored as a (Dim+1)^2 matrix, + * where the last row is assumed to be [0 ... 0 1]. + * - #AffineCompact: the transformation is stored as a (Dim)x(Dim+1) matrix. + * - #Projective: the transformation is stored as a (Dim+1)^2 matrix + * without any assumption. + * \tparam _Options has the same meaning as in class Matrix. It allows to specify DontAlign and/or RowMajor. + * These Options are passed directly to the underlying matrix type. + * + * The homography is internally represented and stored by a matrix which + * is available through the matrix() method. To understand the behavior of + * this class you have to think a Transform object as its internal + * matrix representation. The chosen convention is right multiply: + * + * \code v' = T * v \endcode + * + * Therefore, an affine transformation matrix M is shaped like this: + * + * \f$ \left( \begin{array}{cc} + * linear & translation\\ + * 0 ... 0 & 1 + * \end{array} \right) \f$ + * + * Note that for a projective transformation the last row can be anything, + * and then the interpretation of different parts might be sightly different. + * + * However, unlike a plain matrix, the Transform class provides many features + * simplifying both its assembly and usage. In particular, it can be composed + * with any other transformations (Transform,Translation,RotationBase,DiagonalMatrix) + * and can be directly used to transform implicit homogeneous vectors. All these + * operations are handled via the operator*. For the composition of transformations, + * its principle consists to first convert the right/left hand sides of the product + * to a compatible (Dim+1)^2 matrix and then perform a pure matrix product. + * Of course, internally, operator* tries to perform the minimal number of operations + * according to the nature of each terms. Likewise, when applying the transform + * to points, the latters are automatically promoted to homogeneous vectors + * before doing the matrix product. The conventions to homogeneous representations + * are performed as follow: + * + * \b Translation t (Dim)x(1): + * \f$ \left( \begin{array}{cc} + * I & t \\ + * 0\,...\,0 & 1 + * \end{array} \right) \f$ + * + * \b Rotation R (Dim)x(Dim): + * \f$ \left( \begin{array}{cc} + * R & 0\\ + * 0\,...\,0 & 1 + * \end{array} \right) \f$ + * + * \b Scaling \b DiagonalMatrix S (Dim)x(Dim): + * \f$ \left( \begin{array}{cc} + * S & 0\\ + * 0\,...\,0 & 1 + * \end{array} \right) \f$ + * + * \b Column \b point v (Dim)x(1): + * \f$ \left( \begin{array}{c} + * v\\ + * 1 + * \end{array} \right) \f$ + * + * \b Set \b of \b column \b points V1...Vn (Dim)x(n): + * \f$ \left( \begin{array}{ccc} + * v_1 & ... & v_n\\ + * 1 & ... & 1 + * \end{array} \right) \f$ + * + * The concatenation of a Transform object with any kind of other transformation + * always returns a Transform object. + * + * A little exception to the "as pure matrix product" rule is the case of the + * transformation of non homogeneous vectors by an affine transformation. In + * that case the last matrix row can be ignored, and the product returns non + * homogeneous vectors. + * + * Since, for instance, a Dim x Dim matrix is interpreted as a linear transformation, + * it is not possible to directly transform Dim vectors stored in a Dim x Dim matrix. + * The solution is either to use a Dim x Dynamic matrix or explicitly request a + * vector transformation by making the vector homogeneous: + * \code + * m' = T * m.colwise().homogeneous(); + * \endcode + * Note that there is zero overhead. + * + * Conversion methods from/to Qt's QMatrix and QTransform are available if the + * preprocessor token EIGEN_QT_SUPPORT is defined. + * + * This class can be extended with the help of the plugin mechanism described on the page + * \ref TopicCustomizing_Plugins by defining the preprocessor symbol \c EIGEN_TRANSFORM_PLUGIN. + * + * \sa class Matrix, class Quaternion + */ +template class Transform { public: - EIGEN_MAKE_ALIGNED_OPERATOR_NEW_IF_VECTORIZABLE_FIXED_SIZE(_Scalar,_Dim==Dynamic ? Dynamic : (_Dim+1)*(_Dim+1)) + EIGEN_MAKE_ALIGNED_OPERATOR_NEW_IF_VECTORIZABLE_FIXED_SIZE(_Scalar, + _Dim == Dynamic ? Dynamic : (_Dim + 1) * (_Dim + 1)) enum { Mode = _Mode, Options = _Options, - Dim = _Dim, ///< space dimension in which the transformation holds - HDim = _Dim+1, ///< size of a respective homogeneous vector - Rows = int(Mode)==(AffineCompact) ? Dim : HDim + Dim = _Dim,///< space dimension in which the transformation holds + HDim = _Dim + 1,///< size of a respective homogeneous vector + Rows = int(Mode) == (AffineCompact) ? Dim : HDim }; /** the scalar type of the coefficients */ typedef _Scalar Scalar; typedef Eigen::Index StorageIndex; - typedef Eigen::Index Index; ///< \deprecated since Eigen 3.3 + typedef Eigen::Index Index;///< \deprecated since Eigen 3.3 /** type of the matrix used to represent the transformation */ - typedef typename internal::make_proper_matrix_type::type MatrixType; + typedef typename internal::make_proper_matrix_type::type MatrixType; /** constified MatrixType */ typedef const MatrixType ConstMatrixType; /** type of the matrix used to represent the linear part of the transformation */ - typedef Matrix LinearMatrixType; + typedef Matrix LinearMatrixType; /** type of read/write reference to the linear part of the transformation */ - typedef Block LinearPart; + typedef Block LinearPart; /** type of read reference to the linear part of the transformation */ - typedef const Block ConstLinearPart; + typedef const Block + ConstLinearPart; /** type of read/write reference to the affine part of the transformation */ - typedef typename internal::conditional >::type AffinePart; + typedef + typename internal::conditional>::type + AffinePart; /** type of read reference to the affine part of the transformation */ - typedef typename internal::conditional >::type ConstAffinePart; + typedef typename internal::conditional>::type ConstAffinePart; /** type of a vector */ - typedef Matrix VectorType; + typedef Matrix VectorType; /** type of a read/write reference to the translation part of the rotation */ - typedef Block::Flags & RowMajorBit)> TranslationPart; + typedef Block::Flags & RowMajorBit)> TranslationPart; /** type of a read reference to the translation part of the rotation */ - typedef const Block::Flags & RowMajorBit)> ConstTranslationPart; + typedef const Block::Flags & RowMajorBit)> + ConstTranslationPart; /** corresponding translation type */ - typedef Translation TranslationType; - + typedef Translation TranslationType; + // this intermediate enum is needed to avoid an ICE with gcc 3.4 and 4.0 - enum { TransformTimeDiagonalMode = ((Mode==int(Isometry))?Affine:int(Mode)) }; + enum { TransformTimeDiagonalMode = ((Mode == int(Isometry)) ? Affine : int(Mode)) }; /** The return type of the product between a diagonal matrix and a transform */ - typedef Transform TransformTimeDiagonalReturnType; + typedef Transform TransformTimeDiagonalReturnType; protected: - MatrixType m_matrix; public: - /** Default constructor without initialization of the meaningful coefficients. - * If Mode==Affine, then the last row is set to [0 ... 0 1] */ + * If Mode==Affine, then the last row is set to [0 ... 0 1] */ EIGEN_DEVICE_FUNC inline Transform() { check_template_params(); - internal::transform_make_affine<(int(Mode)==Affine) ? Affine : AffineCompact>::run(m_matrix); + internal::transform_make_affine<(int(Mode) == Affine) ? Affine : AffineCompact>::run(m_matrix); } - EIGEN_DEVICE_FUNC inline Transform(const Transform& other) + EIGEN_DEVICE_FUNC inline Transform(const Transform &other) { check_template_params(); m_matrix = other.m_matrix; } - EIGEN_DEVICE_FUNC inline explicit Transform(const TranslationType& t) + EIGEN_DEVICE_FUNC inline explicit Transform(const TranslationType &t) { check_template_params(); *this = t; } - EIGEN_DEVICE_FUNC inline explicit Transform(const UniformScaling& s) + EIGEN_DEVICE_FUNC inline explicit Transform(const UniformScaling &s) { check_template_params(); *this = s; } - template - EIGEN_DEVICE_FUNC inline explicit Transform(const RotationBase& r) + template EIGEN_DEVICE_FUNC inline explicit Transform(const RotationBase &r) { check_template_params(); *this = r; } - EIGEN_DEVICE_FUNC inline Transform& operator=(const Transform& other) - { m_matrix = other.m_matrix; return *this; } + EIGEN_DEVICE_FUNC inline Transform &operator=(const Transform &other) + { + m_matrix = other.m_matrix; + return *this; + } typedef internal::transform_take_affine_part take_affine_part; /** Constructs and initializes a transformation from a Dim^2 or a (Dim+1)^2 matrix. */ - template - EIGEN_DEVICE_FUNC inline explicit Transform(const EigenBase& other) + template EIGEN_DEVICE_FUNC inline explicit Transform(const EigenBase &other) { - EIGEN_STATIC_ASSERT((internal::is_same::value), + EIGEN_STATIC_ASSERT((internal::is_same::value), YOU_MIXED_DIFFERENT_NUMERIC_TYPES__YOU_NEED_TO_USE_THE_CAST_METHOD_OF_MATRIXBASE_TO_CAST_NUMERIC_TYPES_EXPLICITLY); check_template_params(); - internal::transform_construct_from_matrix::run(this, other.derived()); + internal::transform_construct_from_matrix::run(this, other.derived()); } /** Set \c *this from a Dim^2 or (Dim+1)^2 matrix. */ - template - EIGEN_DEVICE_FUNC inline Transform& operator=(const EigenBase& other) + template EIGEN_DEVICE_FUNC inline Transform &operator=(const EigenBase &other) { - EIGEN_STATIC_ASSERT((internal::is_same::value), + EIGEN_STATIC_ASSERT((internal::is_same::value), YOU_MIXED_DIFFERENT_NUMERIC_TYPES__YOU_NEED_TO_USE_THE_CAST_METHOD_OF_MATRIXBASE_TO_CAST_NUMERIC_TYPES_EXPLICITLY); - internal::transform_construct_from_matrix::run(this, other.derived()); + internal::transform_construct_from_matrix::run(this, other.derived()); return *this; } - - template - EIGEN_DEVICE_FUNC inline Transform(const Transform& other) + + template EIGEN_DEVICE_FUNC inline Transform(const Transform &other) { check_template_params(); // only the options change, we can directly copy the matrices m_matrix = other.matrix(); } - template - EIGEN_DEVICE_FUNC inline Transform(const Transform& other) + template + EIGEN_DEVICE_FUNC inline Transform(const Transform &other) { check_template_params(); // prevent conversions as: // Affine | AffineCompact | Isometry = Projective - EIGEN_STATIC_ASSERT(EIGEN_IMPLIES(OtherMode==int(Projective), Mode==int(Projective)), - YOU_PERFORMED_AN_INVALID_TRANSFORMATION_CONVERSION) + EIGEN_STATIC_ASSERT(EIGEN_IMPLIES(OtherMode == int(Projective), Mode == int(Projective)), + YOU_PERFORMED_AN_INVALID_TRANSFORMATION_CONVERSION) // prevent conversions as: // Isometry = Affine | AffineCompact - EIGEN_STATIC_ASSERT(EIGEN_IMPLIES(OtherMode==int(Affine)||OtherMode==int(AffineCompact), Mode!=int(Isometry)), - YOU_PERFORMED_AN_INVALID_TRANSFORMATION_CONVERSION) + EIGEN_STATIC_ASSERT( + EIGEN_IMPLIES(OtherMode == int(Affine) || OtherMode == int(AffineCompact), Mode != int(Isometry)), + YOU_PERFORMED_AN_INVALID_TRANSFORMATION_CONVERSION) - enum { ModeIsAffineCompact = Mode == int(AffineCompact), - OtherModeIsAffineCompact = OtherMode == int(AffineCompact) + enum { + ModeIsAffineCompact = Mode == int(AffineCompact), + OtherModeIsAffineCompact = OtherMode == int(AffineCompact) }; - if(ModeIsAffineCompact == OtherModeIsAffineCompact) - { + if (ModeIsAffineCompact == OtherModeIsAffineCompact) { // We need the block expression because the code is compiled for all // combinations of transformations and will trigger a compile time error // if one tries to assign the matrices directly - m_matrix.template block(0,0) = other.matrix().template block(0,0); + m_matrix.template block(0, 0) = other.matrix().template block(0, 0); makeAffine(); - } - else if(OtherModeIsAffineCompact) - { - typedef typename Transform::MatrixType OtherMatrixType; - internal::transform_construct_from_matrix::run(this, other.matrix()); - } - else - { + } else if (OtherModeIsAffineCompact) { + typedef typename Transform::MatrixType OtherMatrixType; + internal::transform_construct_from_matrix::run(this, other.matrix()); + } else { // here we know that Mode == AffineCompact and OtherMode != AffineCompact. // if OtherMode were Projective, the static assert above would already have caught it. // So the only possibility is that OtherMode == Affine @@ -358,48 +349,49 @@ class Transform } } - template - EIGEN_DEVICE_FUNC Transform(const ReturnByValue& other) + template EIGEN_DEVICE_FUNC Transform(const ReturnByValue &other) { check_template_params(); other.evalTo(*this); } - template - EIGEN_DEVICE_FUNC Transform& operator=(const ReturnByValue& other) + template EIGEN_DEVICE_FUNC Transform &operator=(const ReturnByValue &other) { other.evalTo(*this); return *this; } - #ifdef EIGEN_QT_SUPPORT - inline Transform(const QMatrix& other); - inline Transform& operator=(const QMatrix& other); +#ifdef EIGEN_QT_SUPPORT + inline Transform(const QMatrix &other); + inline Transform &operator=(const QMatrix &other); inline QMatrix toQMatrix(void) const; - inline Transform(const QTransform& other); - inline Transform& operator=(const QTransform& other); + inline Transform(const QTransform &other); + inline Transform &operator=(const QTransform &other); inline QTransform toQTransform(void) const; - #endif - - EIGEN_DEVICE_FUNC Index rows() const { return int(Mode)==int(Projective) ? m_matrix.cols() : (m_matrix.cols()-1); } +#endif + + EIGEN_DEVICE_FUNC Index rows() const + { + return int(Mode) == int(Projective) ? m_matrix.cols() : (m_matrix.cols() - 1); + } EIGEN_DEVICE_FUNC Index cols() const { return m_matrix.cols(); } /** shortcut for m_matrix(row,col); - * \sa MatrixBase::operator(Index,Index) const */ - EIGEN_DEVICE_FUNC inline Scalar operator() (Index row, Index col) const { return m_matrix(row,col); } + * \sa MatrixBase::operator(Index,Index) const */ + EIGEN_DEVICE_FUNC inline Scalar operator()(Index row, Index col) const { return m_matrix(row, col); } /** shortcut for m_matrix(row,col); - * \sa MatrixBase::operator(Index,Index) */ - EIGEN_DEVICE_FUNC inline Scalar& operator() (Index row, Index col) { return m_matrix(row,col); } + * \sa MatrixBase::operator(Index,Index) */ + EIGEN_DEVICE_FUNC inline Scalar &operator()(Index row, Index col) { return m_matrix(row, col); } /** \returns a read-only expression of the transformation matrix */ - EIGEN_DEVICE_FUNC inline const MatrixType& matrix() const { return m_matrix; } + EIGEN_DEVICE_FUNC inline const MatrixType &matrix() const { return m_matrix; } /** \returns a writable expression of the transformation matrix */ - EIGEN_DEVICE_FUNC inline MatrixType& matrix() { return m_matrix; } + EIGEN_DEVICE_FUNC inline MatrixType &matrix() { return m_matrix; } /** \returns a read-only expression of the linear part of the transformation */ - EIGEN_DEVICE_FUNC inline ConstLinearPart linear() const { return ConstLinearPart(m_matrix,0,0); } + EIGEN_DEVICE_FUNC inline ConstLinearPart linear() const { return ConstLinearPart(m_matrix, 0, 0); } /** \returns a writable expression of the linear part of the transformation */ - EIGEN_DEVICE_FUNC inline LinearPart linear() { return LinearPart(m_matrix,0,0); } + EIGEN_DEVICE_FUNC inline LinearPart linear() { return LinearPart(m_matrix, 0, 0); } /** \returns a read-only expression of the Dim x HDim affine part of the transformation */ EIGEN_DEVICE_FUNC inline ConstAffinePart affine() const { return take_affine_part::run(m_matrix); } @@ -407,61 +399,67 @@ class Transform EIGEN_DEVICE_FUNC inline AffinePart affine() { return take_affine_part::run(m_matrix); } /** \returns a read-only expression of the translation vector of the transformation */ - EIGEN_DEVICE_FUNC inline ConstTranslationPart translation() const { return ConstTranslationPart(m_matrix,0,Dim); } + EIGEN_DEVICE_FUNC inline ConstTranslationPart translation() const { return ConstTranslationPart(m_matrix, 0, Dim); } /** \returns a writable expression of the translation vector of the transformation */ - EIGEN_DEVICE_FUNC inline TranslationPart translation() { return TranslationPart(m_matrix,0,Dim); } + EIGEN_DEVICE_FUNC inline TranslationPart translation() { return TranslationPart(m_matrix, 0, Dim); } /** \returns an expression of the product between the transform \c *this and a matrix expression \a other. - * - * The right-hand-side \a other can be either: - * \li an homogeneous vector of size Dim+1, - * \li a set of homogeneous vectors of size Dim+1 x N, - * \li a transformation matrix of size Dim+1 x Dim+1. - * - * Moreover, if \c *this represents an affine transformation (i.e., Mode!=Projective), then \a other can also be: - * \li a point of size Dim (computes: \code this->linear() * other + this->translation()\endcode), - * \li a set of N points as a Dim x N matrix (computes: \code (this->linear() * other).colwise() + this->translation()\endcode), - * - * In all cases, the return type is a matrix or vector of same sizes as the right-hand-side \a other. - * - * If you want to interpret \a other as a linear or affine transformation, then first convert it to a Transform<> type, - * or do your own cooking. - * - * Finally, if you want to apply Affine transformations to vectors, then explicitly apply the linear part only: - * \code - * Affine3f A; - * Vector3f v1, v2; - * v2 = A.linear() * v1; - * \endcode - * - */ + * + * The right-hand-side \a other can be either: + * \li an homogeneous vector of size Dim+1, + * \li a set of homogeneous vectors of size Dim+1 x N, + * \li a transformation matrix of size Dim+1 x Dim+1. + * + * Moreover, if \c *this represents an affine transformation (i.e., Mode!=Projective), then \a other can also be: + * \li a point of size Dim (computes: \code this->linear() * other + this->translation()\endcode), + * \li a set of N points as a Dim x N matrix (computes: \code (this->linear() * other).colwise() + + * this->translation()\endcode), + * + * In all cases, the return type is a matrix or vector of same sizes as the right-hand-side \a other. + * + * If you want to interpret \a other as a linear or affine transformation, then first convert it to a Transform<> + * type, or do your own cooking. + * + * Finally, if you want to apply Affine transformations to vectors, then explicitly apply the linear part only: + * \code + * Affine3f A; + * Vector3f v1, v2; + * v2 = A.linear() * v1; + * \endcode + * + */ // note: this function is defined here because some compilers cannot find the respective declaration template - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const typename internal::transform_right_product_impl::ResultType - operator * (const EigenBase &other) const - { return internal::transform_right_product_impl::run(*this,other.derived()); } + EIGEN_DEVICE_FUNC + EIGEN_STRONG_INLINE const typename internal::transform_right_product_impl::ResultType + operator*(const EigenBase &other) const + { + return internal::transform_right_product_impl::run(*this, other.derived()); + } /** \returns the product expression of a transformation matrix \a a times a transform \a b - * - * The left hand side \a other can be either: - * \li a linear transformation matrix of size Dim x Dim, - * \li an affine transformation matrix of size Dim x Dim+1, - * \li a general transformation matrix of size Dim+1 x Dim+1. - */ - template friend - EIGEN_DEVICE_FUNC inline const typename internal::transform_left_product_impl::ResultType - operator * (const EigenBase &a, const Transform &b) - { return internal::transform_left_product_impl::run(a.derived(),b); } + * + * The left hand side \a other can be either: + * \li a linear transformation matrix of size Dim x Dim, + * \li an affine transformation matrix of size Dim x Dim+1, + * \li a general transformation matrix of size Dim+1 x Dim+1. + */ + template + friend EIGEN_DEVICE_FUNC inline const typename internal:: + transform_left_product_impl::ResultType + operator*(const EigenBase &a, const Transform &b) + { + return internal::transform_left_product_impl::run(a.derived(), b); + } /** \returns The product expression of a transform \a a times a diagonal matrix \a b - * - * The rhs diagonal matrix is interpreted as an affine scaling transformation. The - * product results in a Transform of the same type (mode) as the lhs only if the lhs - * mode is no isometry. In that case, the returned transform is an affinity. - */ + * + * The rhs diagonal matrix is interpreted as an affine scaling transformation. The + * product results in a Transform of the same type (mode) as the lhs only if the lhs + * mode is no isometry. In that case, the returned transform is an affinity. + */ template - EIGEN_DEVICE_FUNC inline const TransformTimeDiagonalReturnType - operator * (const DiagonalBase &b) const + EIGEN_DEVICE_FUNC inline const TransformTimeDiagonalReturnType operator*(const DiagonalBase &b) const { TransformTimeDiagonalReturnType res(*this); res.linearExt() *= b; @@ -469,65 +467,70 @@ class Transform } /** \returns The product expression of a diagonal matrix \a a times a transform \a b - * - * The lhs diagonal matrix is interpreted as an affine scaling transformation. The - * product results in a Transform of the same type (mode) as the lhs only if the lhs - * mode is no isometry. In that case, the returned transform is an affinity. - */ + * + * The lhs diagonal matrix is interpreted as an affine scaling transformation. The + * product results in a Transform of the same type (mode) as the lhs only if the lhs + * mode is no isometry. In that case, the returned transform is an affinity. + */ template - EIGEN_DEVICE_FUNC friend inline TransformTimeDiagonalReturnType - operator * (const DiagonalBase &a, const Transform &b) + EIGEN_DEVICE_FUNC friend inline TransformTimeDiagonalReturnType operator*(const DiagonalBase &a, + const Transform &b) { TransformTimeDiagonalReturnType res; - res.linear().noalias() = a*b.linear(); - res.translation().noalias() = a*b.translation(); - if (Mode!=int(AffineCompact)) - res.matrix().row(Dim) = b.matrix().row(Dim); + res.linear().noalias() = a * b.linear(); + res.translation().noalias() = a * b.translation(); + if (Mode != int(AffineCompact)) res.matrix().row(Dim) = b.matrix().row(Dim); return res; } - template - EIGEN_DEVICE_FUNC inline Transform& operator*=(const EigenBase& other) { return *this = *this * other; } + template EIGEN_DEVICE_FUNC inline Transform &operator*=(const EigenBase &other) + { + return *this = *this * other; + } /** Concatenates two transformations */ - EIGEN_DEVICE_FUNC inline const Transform operator * (const Transform& other) const + EIGEN_DEVICE_FUNC inline const Transform operator*(const Transform &other) const { - return internal::transform_transform_product_impl::run(*this,other); + return internal::transform_transform_product_impl::run(*this, other); } - - #if EIGEN_COMP_ICC + +#if EIGEN_COMP_ICC private: // this intermediate structure permits to workaround a bug in ICC 11: // error: template instantiation resulted in unexpected function type of "Eigen::Transform // (const Eigen::Transform &) const" // (the meaning of a name may have changed since the template declaration -- the type of the template is: // "Eigen::internal::transform_transform_product_impl, - // Eigen::Transform, >::ResultType (const Eigen::Transform &) const") - // - template struct icc_11_workaround + // Eigen::Transform, >::ResultType (const Eigen::Transform &) const") + // + template struct icc_11_workaround { - typedef internal::transform_transform_product_impl > ProductType; + typedef internal::transform_transform_product_impl> + ProductType; typedef typename ProductType::ResultType ResultType; }; - + public: /** Concatenates two different transformations */ - template - inline typename icc_11_workaround::ResultType - operator * (const Transform& other) const + template + inline typename icc_11_workaround::ResultType operator*( + const Transform &other) const { - typedef typename icc_11_workaround::ProductType ProductType; - return ProductType::run(*this,other); + typedef typename icc_11_workaround::ProductType ProductType; + return ProductType::run(*this, other); } - #else +#else /** Concatenates two different transformations */ - template - EIGEN_DEVICE_FUNC inline typename internal::transform_transform_product_impl >::ResultType - operator * (const Transform& other) const + template + EIGEN_DEVICE_FUNC inline typename internal::transform_transform_product_impl>::ResultType + operator*(const Transform &other) const { - return internal::transform_transform_product_impl >::run(*this,other); + return internal::transform_transform_product_impl>::run( + *this, other); } - #endif +#endif /** \sa MatrixBase::setIdentity() */ EIGEN_DEVICE_FUNC void setIdentity() { m_matrix.setIdentity(); } @@ -536,56 +539,42 @@ class Transform * \brief Returns an identity transformation. * \todo In the future this function should be returning a Transform expression. */ - EIGEN_DEVICE_FUNC static const Transform Identity() - { - return Transform(MatrixType::Identity()); - } + EIGEN_DEVICE_FUNC static const Transform Identity() { return Transform(MatrixType::Identity()); } - template - EIGEN_DEVICE_FUNC - inline Transform& scale(const MatrixBase &other); + template EIGEN_DEVICE_FUNC inline Transform &scale(const MatrixBase &other); - template - EIGEN_DEVICE_FUNC - inline Transform& prescale(const MatrixBase &other); + template EIGEN_DEVICE_FUNC inline Transform &prescale(const MatrixBase &other); - EIGEN_DEVICE_FUNC inline Transform& scale(const Scalar& s); - EIGEN_DEVICE_FUNC inline Transform& prescale(const Scalar& s); + EIGEN_DEVICE_FUNC inline Transform &scale(const Scalar &s); + EIGEN_DEVICE_FUNC inline Transform &prescale(const Scalar &s); - template - EIGEN_DEVICE_FUNC - inline Transform& translate(const MatrixBase &other); + template EIGEN_DEVICE_FUNC inline Transform &translate(const MatrixBase &other); template - EIGEN_DEVICE_FUNC - inline Transform& pretranslate(const MatrixBase &other); + EIGEN_DEVICE_FUNC inline Transform &pretranslate(const MatrixBase &other); - template - EIGEN_DEVICE_FUNC - inline Transform& rotate(const RotationType& rotation); + template EIGEN_DEVICE_FUNC inline Transform &rotate(const RotationType &rotation); + + template EIGEN_DEVICE_FUNC inline Transform &prerotate(const RotationType &rotation); + + EIGEN_DEVICE_FUNC Transform &shear(const Scalar &sx, const Scalar &sy); + EIGEN_DEVICE_FUNC Transform &preshear(const Scalar &sx, const Scalar &sy); + + EIGEN_DEVICE_FUNC inline Transform &operator=(const TranslationType &t); - template EIGEN_DEVICE_FUNC - inline Transform& prerotate(const RotationType& rotation); + inline Transform &operator*=(const TranslationType &t) { return translate(t.vector()); } - EIGEN_DEVICE_FUNC Transform& shear(const Scalar& sx, const Scalar& sy); - EIGEN_DEVICE_FUNC Transform& preshear(const Scalar& sx, const Scalar& sy); + EIGEN_DEVICE_FUNC inline Transform operator*(const TranslationType &t) const; - EIGEN_DEVICE_FUNC inline Transform& operator=(const TranslationType& t); - EIGEN_DEVICE_FUNC - inline Transform& operator*=(const TranslationType& t) { return translate(t.vector()); } - - EIGEN_DEVICE_FUNC inline Transform operator*(const TranslationType& t) const; + inline Transform &operator=(const UniformScaling &t); - EIGEN_DEVICE_FUNC - inline Transform& operator=(const UniformScaling& t); - EIGEN_DEVICE_FUNC - inline Transform& operator*=(const UniformScaling& s) { return scale(s.factor()); } - + inline Transform &operator*=(const UniformScaling &s) { return scale(s.factor()); } + EIGEN_DEVICE_FUNC - inline TransformTimeDiagonalReturnType operator*(const UniformScaling& s) const + inline TransformTimeDiagonalReturnType operator*(const UniformScaling &s) const { TransformTimeDiagonalReturnType res = *this; res.scale(s.factor()); @@ -593,143 +582,156 @@ class Transform } EIGEN_DEVICE_FUNC - inline Transform& operator*=(const DiagonalMatrix& s) { linearExt() *= s; return *this; } + inline Transform &operator*=(const DiagonalMatrix &s) + { + linearExt() *= s; + return *this; + } - template - EIGEN_DEVICE_FUNC inline Transform& operator=(const RotationBase& r); - template - EIGEN_DEVICE_FUNC inline Transform& operator*=(const RotationBase& r) { return rotate(r.toRotationMatrix()); } - template - EIGEN_DEVICE_FUNC inline Transform operator*(const RotationBase& r) const; + template EIGEN_DEVICE_FUNC inline Transform &operator=(const RotationBase &r); + template EIGEN_DEVICE_FUNC inline Transform &operator*=(const RotationBase &r) + { + return rotate(r.toRotationMatrix()); + } + template EIGEN_DEVICE_FUNC inline Transform operator*(const RotationBase &r) const; EIGEN_DEVICE_FUNC const LinearMatrixType rotation() const; template - EIGEN_DEVICE_FUNC - void computeRotationScaling(RotationMatrixType *rotation, ScalingMatrixType *scaling) const; + EIGEN_DEVICE_FUNC void computeRotationScaling(RotationMatrixType *rotation, ScalingMatrixType *scaling) const; template - EIGEN_DEVICE_FUNC - void computeScalingRotation(ScalingMatrixType *scaling, RotationMatrixType *rotation) const; + EIGEN_DEVICE_FUNC void computeScalingRotation(ScalingMatrixType *scaling, RotationMatrixType *rotation) const; template - EIGEN_DEVICE_FUNC - Transform& fromPositionOrientationScale(const MatrixBase &position, - const OrientationType& orientation, const MatrixBase &scale); + EIGEN_DEVICE_FUNC Transform &fromPositionOrientationScale(const MatrixBase &position, + const OrientationType &orientation, + const MatrixBase &scale); EIGEN_DEVICE_FUNC inline Transform inverse(TransformTraits traits = (TransformTraits)Mode) const; /** \returns a const pointer to the column major internal matrix */ - EIGEN_DEVICE_FUNC const Scalar* data() const { return m_matrix.data(); } + EIGEN_DEVICE_FUNC const Scalar *data() const { return m_matrix.data(); } /** \returns a non-const pointer to the column major internal matrix */ - EIGEN_DEVICE_FUNC Scalar* data() { return m_matrix.data(); } + EIGEN_DEVICE_FUNC Scalar *data() { return m_matrix.data(); } /** \returns \c *this with scalar type casted to \a NewScalarType - * - * Note that if \a NewScalarType is equal to the current scalar type of \c *this - * then this function smartly returns a const reference to \c *this. - */ + * + * Note that if \a NewScalarType is equal to the current scalar type of \c *this + * then this function smartly returns a const reference to \c *this. + */ template - EIGEN_DEVICE_FUNC inline typename internal::cast_return_type >::type cast() const - { return typename internal::cast_return_type >::type(*this); } + EIGEN_DEVICE_FUNC inline + typename internal::cast_return_type>::type + cast() const + { + return typename internal::cast_return_type>::type(*this); + } /** Copy constructor with scalar type conversion */ template - EIGEN_DEVICE_FUNC inline explicit Transform(const Transform& other) + EIGEN_DEVICE_FUNC inline explicit Transform(const Transform &other) { check_template_params(); m_matrix = other.matrix().template cast(); } /** \returns \c true if \c *this is approximately equal to \a other, within the precision - * determined by \a prec. - * - * \sa MatrixBase::isApprox() */ - EIGEN_DEVICE_FUNC bool isApprox(const Transform& other, const typename NumTraits::Real& prec = NumTraits::dummy_precision()) const - { return m_matrix.isApprox(other.m_matrix, prec); } - - /** Sets the last row to [0 ... 0 1] - */ - EIGEN_DEVICE_FUNC void makeAffine() + * determined by \a prec. + * + * \sa MatrixBase::isApprox() */ + EIGEN_DEVICE_FUNC bool isApprox(const Transform &other, + const typename NumTraits::Real &prec = NumTraits::dummy_precision()) const { - internal::transform_make_affine::run(m_matrix); + return m_matrix.isApprox(other.m_matrix, prec); } + /** Sets the last row to [0 ... 0 1] + */ + EIGEN_DEVICE_FUNC void makeAffine() { internal::transform_make_affine::run(m_matrix); } + /** \internal - * \returns the Dim x Dim linear part if the transformation is affine, - * and the HDim x Dim part for projective transformations. - */ - EIGEN_DEVICE_FUNC inline Block linearExt() - { return m_matrix.template block(0,0); } + * \returns the Dim x Dim linear part if the transformation is affine, + * and the HDim x Dim part for projective transformations. + */ + EIGEN_DEVICE_FUNC inline Block linearExt() + { + return m_matrix.template block(0, 0); + } /** \internal - * \returns the Dim x Dim linear part if the transformation is affine, - * and the HDim x Dim part for projective transformations. - */ - EIGEN_DEVICE_FUNC inline const Block linearExt() const - { return m_matrix.template block(0,0); } + * \returns the Dim x Dim linear part if the transformation is affine, + * and the HDim x Dim part for projective transformations. + */ + EIGEN_DEVICE_FUNC inline const Block linearExt() const + { + return m_matrix.template block(0, 0); + } /** \internal - * \returns the translation part if the transformation is affine, - * and the last column for projective transformations. - */ - EIGEN_DEVICE_FUNC inline Block translationExt() - { return m_matrix.template block(0,Dim); } + * \returns the translation part if the transformation is affine, + * and the last column for projective transformations. + */ + EIGEN_DEVICE_FUNC inline Block translationExt() + { + return m_matrix.template block(0, Dim); + } /** \internal - * \returns the translation part if the transformation is affine, - * and the last column for projective transformations. - */ - EIGEN_DEVICE_FUNC inline const Block translationExt() const - { return m_matrix.template block(0,Dim); } + * \returns the translation part if the transformation is affine, + * and the last column for projective transformations. + */ + EIGEN_DEVICE_FUNC inline const Block translationExt() const + { + return m_matrix.template block(0, Dim); + } - #ifdef EIGEN_TRANSFORM_PLUGIN - #include EIGEN_TRANSFORM_PLUGIN - #endif - -protected: - #ifndef EIGEN_PARSED_BY_DOXYGEN - EIGEN_DEVICE_FUNC static EIGEN_STRONG_INLINE void check_template_params() - { - EIGEN_STATIC_ASSERT((Options & (DontAlign|RowMajor)) == Options, INVALID_MATRIX_TEMPLATE_PARAMETERS) - } - #endif +#ifdef EIGEN_TRANSFORM_PLUGIN +#include EIGEN_TRANSFORM_PLUGIN +#endif +protected: +#ifndef EIGEN_PARSED_BY_DOXYGEN + EIGEN_DEVICE_FUNC static EIGEN_STRONG_INLINE void check_template_params() + { + EIGEN_STATIC_ASSERT((Options & (DontAlign | RowMajor)) == Options, INVALID_MATRIX_TEMPLATE_PARAMETERS) + } +#endif }; /** \ingroup Geometry_Module */ -typedef Transform Isometry2f; +typedef Transform Isometry2f; /** \ingroup Geometry_Module */ -typedef Transform Isometry3f; +typedef Transform Isometry3f; /** \ingroup Geometry_Module */ -typedef Transform Isometry2d; +typedef Transform Isometry2d; /** \ingroup Geometry_Module */ -typedef Transform Isometry3d; +typedef Transform Isometry3d; /** \ingroup Geometry_Module */ -typedef Transform Affine2f; +typedef Transform Affine2f; /** \ingroup Geometry_Module */ -typedef Transform Affine3f; +typedef Transform Affine3f; /** \ingroup Geometry_Module */ -typedef Transform Affine2d; +typedef Transform Affine2d; /** \ingroup Geometry_Module */ -typedef Transform Affine3d; +typedef Transform Affine3d; /** \ingroup Geometry_Module */ -typedef Transform AffineCompact2f; +typedef Transform AffineCompact2f; /** \ingroup Geometry_Module */ -typedef Transform AffineCompact3f; +typedef Transform AffineCompact3f; /** \ingroup Geometry_Module */ -typedef Transform AffineCompact2d; +typedef Transform AffineCompact2d; /** \ingroup Geometry_Module */ -typedef Transform AffineCompact3d; +typedef Transform AffineCompact3d; /** \ingroup Geometry_Module */ -typedef Transform Projective2f; +typedef Transform Projective2f; /** \ingroup Geometry_Module */ -typedef Transform Projective3f; +typedef Transform Projective3f; /** \ingroup Geometry_Module */ -typedef Transform Projective2d; +typedef Transform Projective2d; /** \ingroup Geometry_Module */ -typedef Transform Projective3d; +typedef Transform Projective3d; /************************** *** Optional QT support *** @@ -737,96 +739,103 @@ typedef Transform Projective3d; #ifdef EIGEN_QT_SUPPORT /** Initializes \c *this from a QMatrix assuming the dimension is 2. - * - * This function is available only if the token EIGEN_QT_SUPPORT is defined. - */ -template -Transform::Transform(const QMatrix& other) + * + * This function is available only if the token EIGEN_QT_SUPPORT is defined. + */ +template +Transform::Transform(const QMatrix &other) { check_template_params(); *this = other; } /** Set \c *this from a QMatrix assuming the dimension is 2. - * - * This function is available only if the token EIGEN_QT_SUPPORT is defined. - */ -template -Transform& Transform::operator=(const QMatrix& other) + * + * This function is available only if the token EIGEN_QT_SUPPORT is defined. + */ +template +Transform &Transform::operator=(const QMatrix &other) { - EIGEN_STATIC_ASSERT(Dim==2, YOU_MADE_A_PROGRAMMING_MISTAKE) + EIGEN_STATIC_ASSERT(Dim == 2, YOU_MADE_A_PROGRAMMING_MISTAKE) if (Mode == int(AffineCompact)) - m_matrix << other.m11(), other.m21(), other.dx(), - other.m12(), other.m22(), other.dy(); + m_matrix << other.m11(), other.m21(), other.dx(), other.m12(), other.m22(), other.dy(); else - m_matrix << other.m11(), other.m21(), other.dx(), - other.m12(), other.m22(), other.dy(), - 0, 0, 1; + m_matrix << other.m11(), other.m21(), other.dx(), other.m12(), other.m22(), other.dy(), 0, 0, 1; return *this; } /** \returns a QMatrix from \c *this assuming the dimension is 2. - * - * \warning this conversion might loss data if \c *this is not affine - * - * This function is available only if the token EIGEN_QT_SUPPORT is defined. - */ + * + * \warning this conversion might loss data if \c *this is not affine + * + * This function is available only if the token EIGEN_QT_SUPPORT is defined. + */ template -QMatrix Transform::toQMatrix(void) const +QMatrix Transform::toQMatrix(void) const { check_template_params(); - EIGEN_STATIC_ASSERT(Dim==2, YOU_MADE_A_PROGRAMMING_MISTAKE) - return QMatrix(m_matrix.coeff(0,0), m_matrix.coeff(1,0), - m_matrix.coeff(0,1), m_matrix.coeff(1,1), - m_matrix.coeff(0,2), m_matrix.coeff(1,2)); + EIGEN_STATIC_ASSERT(Dim == 2, YOU_MADE_A_PROGRAMMING_MISTAKE) + return QMatrix(m_matrix.coeff(0, 0), + m_matrix.coeff(1, 0), + m_matrix.coeff(0, 1), + m_matrix.coeff(1, 1), + m_matrix.coeff(0, 2), + m_matrix.coeff(1, 2)); } /** Initializes \c *this from a QTransform assuming the dimension is 2. - * - * This function is available only if the token EIGEN_QT_SUPPORT is defined. - */ -template -Transform::Transform(const QTransform& other) + * + * This function is available only if the token EIGEN_QT_SUPPORT is defined. + */ +template +Transform::Transform(const QTransform &other) { check_template_params(); *this = other; } /** Set \c *this from a QTransform assuming the dimension is 2. - * - * This function is available only if the token EIGEN_QT_SUPPORT is defined. - */ + * + * This function is available only if the token EIGEN_QT_SUPPORT is defined. + */ template -Transform& Transform::operator=(const QTransform& other) +Transform &Transform::operator=(const QTransform &other) { check_template_params(); - EIGEN_STATIC_ASSERT(Dim==2, YOU_MADE_A_PROGRAMMING_MISTAKE) + EIGEN_STATIC_ASSERT(Dim == 2, YOU_MADE_A_PROGRAMMING_MISTAKE) if (Mode == int(AffineCompact)) - m_matrix << other.m11(), other.m21(), other.dx(), - other.m12(), other.m22(), other.dy(); + m_matrix << other.m11(), other.m21(), other.dx(), other.m12(), other.m22(), other.dy(); else - m_matrix << other.m11(), other.m21(), other.dx(), - other.m12(), other.m22(), other.dy(), - other.m13(), other.m23(), other.m33(); + m_matrix << other.m11(), other.m21(), other.dx(), other.m12(), other.m22(), other.dy(), other.m13(), other.m23(), + other.m33(); return *this; } /** \returns a QTransform from \c *this assuming the dimension is 2. - * - * This function is available only if the token EIGEN_QT_SUPPORT is defined. - */ + * + * This function is available only if the token EIGEN_QT_SUPPORT is defined. + */ template -QTransform Transform::toQTransform(void) const +QTransform Transform::toQTransform(void) const { - EIGEN_STATIC_ASSERT(Dim==2, YOU_MADE_A_PROGRAMMING_MISTAKE) + EIGEN_STATIC_ASSERT(Dim == 2, YOU_MADE_A_PROGRAMMING_MISTAKE) if (Mode == int(AffineCompact)) - return QTransform(m_matrix.coeff(0,0), m_matrix.coeff(1,0), - m_matrix.coeff(0,1), m_matrix.coeff(1,1), - m_matrix.coeff(0,2), m_matrix.coeff(1,2)); + return QTransform(m_matrix.coeff(0, 0), + m_matrix.coeff(1, 0), + m_matrix.coeff(0, 1), + m_matrix.coeff(1, 1), + m_matrix.coeff(0, 2), + m_matrix.coeff(1, 2)); else - return QTransform(m_matrix.coeff(0,0), m_matrix.coeff(1,0), m_matrix.coeff(2,0), - m_matrix.coeff(0,1), m_matrix.coeff(1,1), m_matrix.coeff(2,1), - m_matrix.coeff(0,2), m_matrix.coeff(1,2), m_matrix.coeff(2,2)); + return QTransform(m_matrix.coeff(0, 0), + m_matrix.coeff(1, 0), + m_matrix.coeff(2, 0), + m_matrix.coeff(0, 1), + m_matrix.coeff(1, 1), + m_matrix.coeff(2, 1), + m_matrix.coeff(0, 2), + m_matrix.coeff(1, 2), + m_matrix.coeff(2, 2)); } #endif @@ -835,84 +844,86 @@ QTransform Transform::toQTransform(void) const *********************/ /** Applies on the right the non uniform scale transformation represented - * by the vector \a other to \c *this and returns a reference to \c *this. - * \sa prescale() - */ + * by the vector \a other to \c *this and returns a reference to \c *this. + * \sa prescale() + */ template template -EIGEN_DEVICE_FUNC Transform& -Transform::scale(const MatrixBase &other) +EIGEN_DEVICE_FUNC Transform &Transform::scale( + const MatrixBase &other) { - EIGEN_STATIC_ASSERT_VECTOR_SPECIFIC_SIZE(OtherDerived,int(Dim)) - EIGEN_STATIC_ASSERT(Mode!=int(Isometry), THIS_METHOD_IS_ONLY_FOR_SPECIFIC_TRANSFORMATIONS) + EIGEN_STATIC_ASSERT_VECTOR_SPECIFIC_SIZE(OtherDerived, int(Dim)) + EIGEN_STATIC_ASSERT(Mode != int(Isometry), THIS_METHOD_IS_ONLY_FOR_SPECIFIC_TRANSFORMATIONS) linearExt().noalias() = (linearExt() * other.asDiagonal()); return *this; } /** Applies on the right a uniform scale of a factor \a c to \c *this - * and returns a reference to \c *this. - * \sa prescale(Scalar) - */ + * and returns a reference to \c *this. + * \sa prescale(Scalar) + */ template -EIGEN_DEVICE_FUNC inline Transform& Transform::scale(const Scalar& s) +EIGEN_DEVICE_FUNC inline Transform &Transform::scale( + const Scalar &s) { - EIGEN_STATIC_ASSERT(Mode!=int(Isometry), THIS_METHOD_IS_ONLY_FOR_SPECIFIC_TRANSFORMATIONS) + EIGEN_STATIC_ASSERT(Mode != int(Isometry), THIS_METHOD_IS_ONLY_FOR_SPECIFIC_TRANSFORMATIONS) linearExt() *= s; return *this; } /** Applies on the left the non uniform scale transformation represented - * by the vector \a other to \c *this and returns a reference to \c *this. - * \sa scale() - */ + * by the vector \a other to \c *this and returns a reference to \c *this. + * \sa scale() + */ template template -EIGEN_DEVICE_FUNC Transform& -Transform::prescale(const MatrixBase &other) +EIGEN_DEVICE_FUNC Transform &Transform::prescale( + const MatrixBase &other) { - EIGEN_STATIC_ASSERT_VECTOR_SPECIFIC_SIZE(OtherDerived,int(Dim)) - EIGEN_STATIC_ASSERT(Mode!=int(Isometry), THIS_METHOD_IS_ONLY_FOR_SPECIFIC_TRANSFORMATIONS) + EIGEN_STATIC_ASSERT_VECTOR_SPECIFIC_SIZE(OtherDerived, int(Dim)) + EIGEN_STATIC_ASSERT(Mode != int(Isometry), THIS_METHOD_IS_ONLY_FOR_SPECIFIC_TRANSFORMATIONS) affine().noalias() = (other.asDiagonal() * affine()); return *this; } /** Applies on the left a uniform scale of a factor \a c to \c *this - * and returns a reference to \c *this. - * \sa scale(Scalar) - */ + * and returns a reference to \c *this. + * \sa scale(Scalar) + */ template -EIGEN_DEVICE_FUNC inline Transform& Transform::prescale(const Scalar& s) +EIGEN_DEVICE_FUNC inline Transform &Transform::prescale( + const Scalar &s) { - EIGEN_STATIC_ASSERT(Mode!=int(Isometry), THIS_METHOD_IS_ONLY_FOR_SPECIFIC_TRANSFORMATIONS) + EIGEN_STATIC_ASSERT(Mode != int(Isometry), THIS_METHOD_IS_ONLY_FOR_SPECIFIC_TRANSFORMATIONS) m_matrix.template topRows() *= s; return *this; } /** Applies on the right the translation matrix represented by the vector \a other - * to \c *this and returns a reference to \c *this. - * \sa pretranslate() - */ + * to \c *this and returns a reference to \c *this. + * \sa pretranslate() + */ template template -EIGEN_DEVICE_FUNC Transform& -Transform::translate(const MatrixBase &other) +EIGEN_DEVICE_FUNC Transform &Transform::translate( + const MatrixBase &other) { - EIGEN_STATIC_ASSERT_VECTOR_SPECIFIC_SIZE(OtherDerived,int(Dim)) + EIGEN_STATIC_ASSERT_VECTOR_SPECIFIC_SIZE(OtherDerived, int(Dim)) translationExt() += linearExt() * other; return *this; } /** Applies on the left the translation matrix represented by the vector \a other - * to \c *this and returns a reference to \c *this. - * \sa translate() - */ + * to \c *this and returns a reference to \c *this. + * \sa translate() + */ template template -EIGEN_DEVICE_FUNC Transform& -Transform::pretranslate(const MatrixBase &other) +EIGEN_DEVICE_FUNC Transform &Transform::pretranslate( + const MatrixBase &other) { - EIGEN_STATIC_ASSERT_VECTOR_SPECIFIC_SIZE(OtherDerived,int(Dim)) - if(int(Mode)==int(Projective)) + EIGEN_STATIC_ASSERT_VECTOR_SPECIFIC_SIZE(OtherDerived, int(Dim)) + if (int(Mode) == int(Projective)) affine() += other * m_matrix.row(Dim); else translation() += other; @@ -920,76 +931,76 @@ Transform::pretranslate(const MatrixBase } /** Applies on the right the rotation represented by the rotation \a rotation - * to \c *this and returns a reference to \c *this. - * - * The template parameter \a RotationType is the type of the rotation which - * must be known by internal::toRotationMatrix<>. - * - * Natively supported types includes: - * - any scalar (2D), - * - a Dim x Dim matrix expression, - * - a Quaternion (3D), - * - a AngleAxis (3D) - * - * This mechanism is easily extendable to support user types such as Euler angles, - * or a pair of Quaternion for 4D rotations. - * - * \sa rotate(Scalar), class Quaternion, class AngleAxis, prerotate(RotationType) - */ + * to \c *this and returns a reference to \c *this. + * + * The template parameter \a RotationType is the type of the rotation which + * must be known by internal::toRotationMatrix<>. + * + * Natively supported types includes: + * - any scalar (2D), + * - a Dim x Dim matrix expression, + * - a Quaternion (3D), + * - a AngleAxis (3D) + * + * This mechanism is easily extendable to support user types such as Euler angles, + * or a pair of Quaternion for 4D rotations. + * + * \sa rotate(Scalar), class Quaternion, class AngleAxis, prerotate(RotationType) + */ template template -EIGEN_DEVICE_FUNC Transform& -Transform::rotate(const RotationType& rotation) +EIGEN_DEVICE_FUNC Transform &Transform::rotate( + const RotationType &rotation) { - linearExt() *= internal::toRotationMatrix(rotation); + linearExt() *= internal::toRotationMatrix(rotation); return *this; } /** Applies on the left the rotation represented by the rotation \a rotation - * to \c *this and returns a reference to \c *this. - * - * See rotate() for further details. - * - * \sa rotate() - */ + * to \c *this and returns a reference to \c *this. + * + * See rotate() for further details. + * + * \sa rotate() + */ template template -EIGEN_DEVICE_FUNC Transform& -Transform::prerotate(const RotationType& rotation) +EIGEN_DEVICE_FUNC Transform &Transform::prerotate( + const RotationType &rotation) { - m_matrix.template block(0,0) = internal::toRotationMatrix(rotation) - * m_matrix.template block(0,0); + m_matrix.template block(0, 0) = + internal::toRotationMatrix(rotation) * m_matrix.template block(0, 0); return *this; } /** Applies on the right the shear transformation represented - * by the vector \a other to \c *this and returns a reference to \c *this. - * \warning 2D only. - * \sa preshear() - */ + * by the vector \a other to \c *this and returns a reference to \c *this. + * \warning 2D only. + * \sa preshear() + */ template -EIGEN_DEVICE_FUNC Transform& -Transform::shear(const Scalar& sx, const Scalar& sy) +EIGEN_DEVICE_FUNC Transform &Transform::shear(const Scalar &sx, + const Scalar &sy) { - EIGEN_STATIC_ASSERT(int(Dim)==2, YOU_MADE_A_PROGRAMMING_MISTAKE) - EIGEN_STATIC_ASSERT(Mode!=int(Isometry), THIS_METHOD_IS_ONLY_FOR_SPECIFIC_TRANSFORMATIONS) - VectorType tmp = linear().col(0)*sy + linear().col(1); - linear() << linear().col(0) + linear().col(1)*sx, tmp; + EIGEN_STATIC_ASSERT(int(Dim) == 2, YOU_MADE_A_PROGRAMMING_MISTAKE) + EIGEN_STATIC_ASSERT(Mode != int(Isometry), THIS_METHOD_IS_ONLY_FOR_SPECIFIC_TRANSFORMATIONS) + VectorType tmp = linear().col(0) * sy + linear().col(1); + linear() << linear().col(0) + linear().col(1) * sx, tmp; return *this; } /** Applies on the left the shear transformation represented - * by the vector \a other to \c *this and returns a reference to \c *this. - * \warning 2D only. - * \sa shear() - */ + * by the vector \a other to \c *this and returns a reference to \c *this. + * \warning 2D only. + * \sa shear() + */ template -EIGEN_DEVICE_FUNC Transform& -Transform::preshear(const Scalar& sx, const Scalar& sy) +EIGEN_DEVICE_FUNC Transform & + Transform::preshear(const Scalar &sx, const Scalar &sy) { - EIGEN_STATIC_ASSERT(int(Dim)==2, YOU_MADE_A_PROGRAMMING_MISTAKE) - EIGEN_STATIC_ASSERT(Mode!=int(Isometry), THIS_METHOD_IS_ONLY_FOR_SPECIFIC_TRANSFORMATIONS) - m_matrix.template block(0,0) = LinearMatrixType(1, sx, sy, 1) * m_matrix.template block(0,0); + EIGEN_STATIC_ASSERT(int(Dim) == 2, YOU_MADE_A_PROGRAMMING_MISTAKE) + EIGEN_STATIC_ASSERT(Mode != int(Isometry), THIS_METHOD_IS_ONLY_FOR_SPECIFIC_TRANSFORMATIONS) + m_matrix.template block(0, 0) = LinearMatrixType(1, sx, sy, 1) * m_matrix.template block(0, 0); return *this; } @@ -998,7 +1009,8 @@ Transform::preshear(const Scalar& sx, const Scalar& sy) ******************************************************/ template -EIGEN_DEVICE_FUNC inline Transform& Transform::operator=(const TranslationType& t) +EIGEN_DEVICE_FUNC inline Transform &Transform::operator=( + const TranslationType &t) { linear().setIdentity(); translation() = t.vector(); @@ -1007,7 +1019,8 @@ EIGEN_DEVICE_FUNC inline Transform& Transform -EIGEN_DEVICE_FUNC inline Transform Transform::operator*(const TranslationType& t) const +EIGEN_DEVICE_FUNC inline Transform Transform::operator*( + const TranslationType &t) const { Transform res = *this; res.translate(t.vector()); @@ -1015,7 +1028,8 @@ EIGEN_DEVICE_FUNC inline Transform Transform -EIGEN_DEVICE_FUNC inline Transform& Transform::operator=(const UniformScaling& s) +EIGEN_DEVICE_FUNC inline Transform &Transform::operator=( + const UniformScaling &s) { m_matrix.setZero(); linear().diagonal().fill(s.factor()); @@ -1025,9 +1039,10 @@ EIGEN_DEVICE_FUNC inline Transform& Transform template -EIGEN_DEVICE_FUNC inline Transform& Transform::operator=(const RotationBase& r) +EIGEN_DEVICE_FUNC inline Transform &Transform::operator=( + const RotationBase &r) { - linear() = internal::toRotationMatrix(r); + linear() = internal::toRotationMatrix(r); translation().setZero(); makeAffine(); return *this; @@ -1035,7 +1050,8 @@ EIGEN_DEVICE_FUNC inline Transform& Transform template -EIGEN_DEVICE_FUNC inline Transform Transform::operator*(const RotationBase& r) const +EIGEN_DEVICE_FUNC inline Transform Transform::operator*( + const RotationBase &r) const { Transform res = *this; res.rotate(r.derived()); @@ -1047,45 +1063,45 @@ EIGEN_DEVICE_FUNC inline Transform Transform -EIGEN_DEVICE_FUNC const typename Transform::LinearMatrixType -Transform::rotation() const +EIGEN_DEVICE_FUNC const typename Transform::LinearMatrixType + Transform::rotation() const { LinearMatrixType result; - computeRotationScaling(&result, (LinearMatrixType*)0); + computeRotationScaling(&result, (LinearMatrixType *)0); return result; } /** decomposes the linear part of the transformation as a product rotation x scaling, the scaling being - * not necessarily positive. - * - * If either pointer is zero, the corresponding computation is skipped. - * - * - * - * \svd_module - * - * \sa computeScalingRotation(), rotation(), class SVD - */ + * not necessarily positive. + * + * If either pointer is zero, the corresponding computation is skipped. + * + * + * + * \svd_module + * + * \sa computeScalingRotation(), rotation(), class SVD + */ template template -EIGEN_DEVICE_FUNC void Transform::computeRotationScaling(RotationMatrixType *rotation, ScalingMatrixType *scaling) const +EIGEN_DEVICE_FUNC void Transform::computeRotationScaling(RotationMatrixType *rotation, + ScalingMatrixType *scaling) const { JacobiSVD svd(linear(), ComputeFullU | ComputeFullV); - Scalar x = (svd.matrixU() * svd.matrixV().adjoint()).determinant(); // so x has absolute value 1 + Scalar x = (svd.matrixU() * svd.matrixV().adjoint()).determinant();// so x has absolute value 1 VectorType sv(svd.singularValues()); sv.coeffRef(0) *= x; - if(scaling) scaling->lazyAssign(svd.matrixV() * sv.asDiagonal() * svd.matrixV().adjoint()); - if(rotation) - { + if (scaling) scaling->lazyAssign(svd.matrixV() * sv.asDiagonal() * svd.matrixV().adjoint()); + if (rotation) { LinearMatrixType m(svd.matrixU()); m.col(0) /= x; rotation->lazyAssign(m * svd.matrixV().adjoint()); @@ -1093,28 +1109,28 @@ EIGEN_DEVICE_FUNC void Transform::computeRotationScalin } /** decomposes the linear part of the transformation as a product scaling x rotation, the scaling being - * not necessarily positive. - * - * If either pointer is zero, the corresponding computation is skipped. - * - * - * - * \svd_module - * - * \sa computeRotationScaling(), rotation(), class SVD - */ + * not necessarily positive. + * + * If either pointer is zero, the corresponding computation is skipped. + * + * + * + * \svd_module + * + * \sa computeRotationScaling(), rotation(), class SVD + */ template template -EIGEN_DEVICE_FUNC void Transform::computeScalingRotation(ScalingMatrixType *scaling, RotationMatrixType *rotation) const +EIGEN_DEVICE_FUNC void Transform::computeScalingRotation(ScalingMatrixType *scaling, + RotationMatrixType *rotation) const { JacobiSVD svd(linear(), ComputeFullU | ComputeFullV); - Scalar x = (svd.matrixU() * svd.matrixV().adjoint()).determinant(); // so x has absolute value 1 + Scalar x = (svd.matrixU() * svd.matrixV().adjoint()).determinant();// so x has absolute value 1 VectorType sv(svd.singularValues()); sv.coeffRef(0) *= x; - if(scaling) scaling->lazyAssign(svd.matrixU() * sv.asDiagonal() * svd.matrixU().adjoint()); - if(rotation) - { + if (scaling) scaling->lazyAssign(svd.matrixU() * sv.asDiagonal() * svd.matrixU().adjoint()); + if (rotation) { LinearMatrixType m(svd.matrixU()); m.col(0) /= x; rotation->lazyAssign(m * svd.matrixV().adjoint()); @@ -1122,15 +1138,16 @@ EIGEN_DEVICE_FUNC void Transform::computeScalingRotatio } /** Convenient method to set \c *this from a position, orientation and scale - * of a 3D object. - */ + * of a 3D object. + */ template template -EIGEN_DEVICE_FUNC Transform& -Transform::fromPositionOrientationScale(const MatrixBase &position, - const OrientationType& orientation, const MatrixBase &scale) +EIGEN_DEVICE_FUNC Transform & + Transform::fromPositionOrientationScale(const MatrixBase &position, + const OrientationType &orientation, + const MatrixBase &scale) { - linear() = internal::toRotationMatrix(orientation); + linear() = internal::toRotationMatrix(orientation); linear() *= scale.asDiagonal(); translation() = position; makeAffine(); @@ -1139,404 +1156,402 @@ Transform::fromPositionOrientationScale(const MatrixBas namespace internal { -template -struct transform_make_affine -{ - template - EIGEN_DEVICE_FUNC static void run(MatrixType &mat) + template struct transform_make_affine { - static const int Dim = MatrixType::ColsAtCompileTime-1; - mat.template block<1,Dim>(Dim,0).setZero(); - mat.coeffRef(Dim,Dim) = typename MatrixType::Scalar(1); - } -}; + template EIGEN_DEVICE_FUNC static void run(MatrixType &mat) + { + static const int Dim = MatrixType::ColsAtCompileTime - 1; + mat.template block<1, Dim>(Dim, 0).setZero(); + mat.coeffRef(Dim, Dim) = typename MatrixType::Scalar(1); + } + }; -template<> -struct transform_make_affine -{ - template EIGEN_DEVICE_FUNC static void run(MatrixType &) { } -}; - -// selector needed to avoid taking the inverse of a 3x4 matrix -template -struct projective_transform_inverse -{ - EIGEN_DEVICE_FUNC static inline void run(const TransformType&, TransformType&) - {} -}; + template<> struct transform_make_affine + { + template EIGEN_DEVICE_FUNC static void run(MatrixType &) {} + }; -template -struct projective_transform_inverse -{ - EIGEN_DEVICE_FUNC static inline void run(const TransformType& m, TransformType& res) + // selector needed to avoid taking the inverse of a 3x4 matrix + template struct projective_transform_inverse { - res.matrix() = m.matrix().inverse(); - } -}; + EIGEN_DEVICE_FUNC static inline void run(const TransformType &, TransformType &) {} + }; + + template struct projective_transform_inverse + { + EIGEN_DEVICE_FUNC static inline void run(const TransformType &m, TransformType &res) + { + res.matrix() = m.matrix().inverse(); + } + }; -} // end namespace internal +}// end namespace internal /** - * - * \returns the inverse transformation according to some given knowledge - * on \c *this. - * - * \param hint allows to optimize the inversion process when the transformation - * is known to be not a general transformation (optional). The possible values are: - * - #Projective if the transformation is not necessarily affine, i.e., if the - * last row is not guaranteed to be [0 ... 0 1] - * - #Affine if the last row can be assumed to be [0 ... 0 1] - * - #Isometry if the transformation is only a concatenations of translations - * and rotations. - * The default is the template class parameter \c Mode. - * - * \warning unless \a traits is always set to NoShear or NoScaling, this function - * requires the generic inverse method of MatrixBase defined in the LU module. If - * you forget to include this module, then you will get hard to debug linking errors. - * - * \sa MatrixBase::inverse() - */ + * + * \returns the inverse transformation according to some given knowledge + * on \c *this. + * + * \param hint allows to optimize the inversion process when the transformation + * is known to be not a general transformation (optional). The possible values are: + * - #Projective if the transformation is not necessarily affine, i.e., if the + * last row is not guaranteed to be [0 ... 0 1] + * - #Affine if the last row can be assumed to be [0 ... 0 1] + * - #Isometry if the transformation is only a concatenations of translations + * and rotations. + * The default is the template class parameter \c Mode. + * + * \warning unless \a traits is always set to NoShear or NoScaling, this function + * requires the generic inverse method of MatrixBase defined in the LU module. If + * you forget to include this module, then you will get hard to debug linking errors. + * + * \sa MatrixBase::inverse() + */ template -EIGEN_DEVICE_FUNC Transform -Transform::inverse(TransformTraits hint) const +EIGEN_DEVICE_FUNC Transform Transform::inverse( + TransformTraits hint) const { Transform res; - if (hint == Projective) - { + if (hint == Projective) { internal::projective_transform_inverse::run(*this, res); - } - else - { - if (hint == Isometry) - { - res.matrix().template topLeftCorner() = linear().transpose(); - } - else if(hint&Affine) - { - res.matrix().template topLeftCorner() = linear().inverse(); - } - else - { + } else { + if (hint == Isometry) { + res.matrix().template topLeftCorner() = linear().transpose(); + } else if (hint & Affine) { + res.matrix().template topLeftCorner() = linear().inverse(); + } else { eigen_assert(false && "Invalid transform traits in Transform::Inverse"); } // translation and remaining parts - res.matrix().template topRightCorner() - = - res.matrix().template topLeftCorner() * translation(); - res.makeAffine(); // we do need this, because in the beginning res is uninitialized + res.matrix().template topRightCorner() = -res.matrix().template topLeftCorner() * translation(); + res.makeAffine();// we do need this, because in the beginning res is uninitialized } return res; } namespace internal { -/***************************************************** -*** Specializations of take affine part *** -*****************************************************/ - -template struct transform_take_affine_part { - typedef typename TransformType::MatrixType MatrixType; - typedef typename TransformType::AffinePart AffinePart; - typedef typename TransformType::ConstAffinePart ConstAffinePart; - static inline AffinePart run(MatrixType& m) - { return m.template block(0,0); } - static inline ConstAffinePart run(const MatrixType& m) - { return m.template block(0,0); } -}; + /***************************************************** + *** Specializations of take affine part *** + *****************************************************/ -template -struct transform_take_affine_part > { - typedef typename Transform::MatrixType MatrixType; - static inline MatrixType& run(MatrixType& m) { return m; } - static inline const MatrixType& run(const MatrixType& m) { return m; } -}; + template struct transform_take_affine_part + { + typedef typename TransformType::MatrixType MatrixType; + typedef typename TransformType::AffinePart AffinePart; + typedef typename TransformType::ConstAffinePart ConstAffinePart; + static inline AffinePart run(MatrixType &m) + { + return m.template block(0, 0); + } + static inline ConstAffinePart run(const MatrixType &m) + { + return m.template block(0, 0); + } + }; -/***************************************************** -*** Specializations of construct from matrix *** -*****************************************************/ + template + struct transform_take_affine_part> + { + typedef typename Transform::MatrixType MatrixType; + static inline MatrixType &run(MatrixType &m) { return m; } + static inline const MatrixType &run(const MatrixType &m) { return m; } + }; -template -struct transform_construct_from_matrix -{ - static inline void run(Transform *transform, const Other& other) + /***************************************************** + *** Specializations of construct from matrix *** + *****************************************************/ + + template + struct transform_construct_from_matrix { - transform->linear() = other; - transform->translation().setZero(); - transform->makeAffine(); - } -}; + static inline void run(Transform *transform, const Other &other) + { + transform->linear() = other; + transform->translation().setZero(); + transform->makeAffine(); + } + }; -template -struct transform_construct_from_matrix -{ - static inline void run(Transform *transform, const Other& other) + template + struct transform_construct_from_matrix { - transform->affine() = other; - transform->makeAffine(); - } -}; + static inline void run(Transform *transform, const Other &other) + { + transform->affine() = other; + transform->makeAffine(); + } + }; -template -struct transform_construct_from_matrix -{ - static inline void run(Transform *transform, const Other& other) - { transform->matrix() = other; } -}; + template + struct transform_construct_from_matrix + { + static inline void run(Transform *transform, const Other &other) + { + transform->matrix() = other; + } + }; -template -struct transform_construct_from_matrix -{ - static inline void run(Transform *transform, const Other& other) - { transform->matrix() = other.template block(0,0); } -}; + template + struct transform_construct_from_matrix + { + static inline void run(Transform *transform, + const Other &other) + { + transform->matrix() = other.template block(0, 0); + } + }; -/********************************************************** -*** Specializations of operator* with rhs EigenBase *** -**********************************************************/ + /********************************************************** + *** Specializations of operator* with rhs EigenBase *** + **********************************************************/ -template -struct transform_product_result -{ - enum - { - Mode = - (LhsMode == (int)Projective || RhsMode == (int)Projective ) ? Projective : - (LhsMode == (int)Affine || RhsMode == (int)Affine ) ? Affine : - (LhsMode == (int)AffineCompact || RhsMode == (int)AffineCompact ) ? AffineCompact : - (LhsMode == (int)Isometry || RhsMode == (int)Isometry ) ? Isometry : Projective + template struct transform_product_result + { + enum { + Mode = (LhsMode == (int)Projective || RhsMode == (int)Projective) ? Projective + : (LhsMode == (int)Affine || RhsMode == (int)Affine) ? Affine + : (LhsMode == (int)AffineCompact || RhsMode == (int)AffineCompact) ? AffineCompact + : (LhsMode == (int)Isometry || RhsMode == (int)Isometry) ? Isometry + : Projective + }; }; -}; -template< typename TransformType, typename MatrixType, int RhsCols> -struct transform_right_product_impl< TransformType, MatrixType, 0, RhsCols> -{ - typedef typename MatrixType::PlainObject ResultType; - - static EIGEN_STRONG_INLINE ResultType run(const TransformType& T, const MatrixType& other) + template + struct transform_right_product_impl { - return T.matrix() * other; - } -}; + typedef typename MatrixType::PlainObject ResultType; -template< typename TransformType, typename MatrixType, int RhsCols> -struct transform_right_product_impl< TransformType, MatrixType, 1, RhsCols> -{ - enum { - Dim = TransformType::Dim, - HDim = TransformType::HDim, - OtherRows = MatrixType::RowsAtCompileTime, - OtherCols = MatrixType::ColsAtCompileTime + static EIGEN_STRONG_INLINE ResultType run(const TransformType &T, const MatrixType &other) + { + return T.matrix() * other; + } }; - typedef typename MatrixType::PlainObject ResultType; - - static EIGEN_STRONG_INLINE ResultType run(const TransformType& T, const MatrixType& other) + template + struct transform_right_product_impl { - EIGEN_STATIC_ASSERT(OtherRows==HDim, YOU_MIXED_MATRICES_OF_DIFFERENT_SIZES); + enum { + Dim = TransformType::Dim, + HDim = TransformType::HDim, + OtherRows = MatrixType::RowsAtCompileTime, + OtherCols = MatrixType::ColsAtCompileTime + }; - typedef Block TopLeftLhs; + typedef typename MatrixType::PlainObject ResultType; - ResultType res(other.rows(),other.cols()); - TopLeftLhs(res, 0, 0, Dim, other.cols()).noalias() = T.affine() * other; - res.row(OtherRows-1) = other.row(OtherRows-1); - - return res; - } -}; + static EIGEN_STRONG_INLINE ResultType run(const TransformType &T, const MatrixType &other) + { + EIGEN_STATIC_ASSERT(OtherRows == HDim, YOU_MIXED_MATRICES_OF_DIFFERENT_SIZES); -template< typename TransformType, typename MatrixType, int RhsCols> -struct transform_right_product_impl< TransformType, MatrixType, 2, RhsCols> -{ - enum { - Dim = TransformType::Dim, - HDim = TransformType::HDim, - OtherRows = MatrixType::RowsAtCompileTime, - OtherCols = MatrixType::ColsAtCompileTime - }; + typedef Block TopLeftLhs; - typedef typename MatrixType::PlainObject ResultType; + ResultType res(other.rows(), other.cols()); + TopLeftLhs(res, 0, 0, Dim, other.cols()).noalias() = T.affine() * other; + res.row(OtherRows - 1) = other.row(OtherRows - 1); + + return res; + } + }; - static EIGEN_STRONG_INLINE ResultType run(const TransformType& T, const MatrixType& other) + template + struct transform_right_product_impl { - EIGEN_STATIC_ASSERT(OtherRows==Dim, YOU_MIXED_MATRICES_OF_DIFFERENT_SIZES); + enum { + Dim = TransformType::Dim, + HDim = TransformType::HDim, + OtherRows = MatrixType::RowsAtCompileTime, + OtherCols = MatrixType::ColsAtCompileTime + }; - typedef Block TopLeftLhs; - ResultType res(Replicate(T.translation(),1,other.cols())); - TopLeftLhs(res, 0, 0, Dim, other.cols()).noalias() += T.linear() * other; + typedef typename MatrixType::PlainObject ResultType; - return res; - } -}; + static EIGEN_STRONG_INLINE ResultType run(const TransformType &T, const MatrixType &other) + { + EIGEN_STATIC_ASSERT(OtherRows == Dim, YOU_MIXED_MATRICES_OF_DIFFERENT_SIZES); -template< typename TransformType, typename MatrixType > -struct transform_right_product_impl< TransformType, MatrixType, 2, 1> // rhs is a vector of size Dim -{ - typedef typename TransformType::MatrixType TransformMatrix; - enum { - Dim = TransformType::Dim, - HDim = TransformType::HDim, - OtherRows = MatrixType::RowsAtCompileTime, - WorkingRows = EIGEN_PLAIN_ENUM_MIN(TransformMatrix::RowsAtCompileTime,HDim) - }; + typedef Block TopLeftLhs; + ResultType res( + Replicate(T.translation(), 1, other.cols())); + TopLeftLhs(res, 0, 0, Dim, other.cols()).noalias() += T.linear() * other; - typedef typename MatrixType::PlainObject ResultType; + return res; + } + }; - static EIGEN_STRONG_INLINE ResultType run(const TransformType& T, const MatrixType& other) + template + struct transform_right_product_impl// rhs is a vector of size Dim { - EIGEN_STATIC_ASSERT(OtherRows==Dim, YOU_MIXED_MATRICES_OF_DIFFERENT_SIZES); + typedef typename TransformType::MatrixType TransformMatrix; + enum { + Dim = TransformType::Dim, + HDim = TransformType::HDim, + OtherRows = MatrixType::RowsAtCompileTime, + WorkingRows = EIGEN_PLAIN_ENUM_MIN(TransformMatrix::RowsAtCompileTime, HDim) + }; - Matrix rhs; - rhs.template head() = other; rhs[Dim] = typename ResultType::Scalar(1); - Matrix res(T.matrix() * rhs); - return res.template head(); - } -}; + typedef typename MatrixType::PlainObject ResultType; -/********************************************************** -*** Specializations of operator* with lhs EigenBase *** -**********************************************************/ + static EIGEN_STRONG_INLINE ResultType run(const TransformType &T, const MatrixType &other) + { + EIGEN_STATIC_ASSERT(OtherRows == Dim, YOU_MIXED_MATRICES_OF_DIFFERENT_SIZES); -// generic HDim x HDim matrix * T => Projective -template -struct transform_left_product_impl -{ - typedef Transform TransformType; - typedef typename TransformType::MatrixType MatrixType; - typedef Transform ResultType; - static ResultType run(const Other& other,const TransformType& tr) - { return ResultType(other * tr.matrix()); } -}; + Matrix rhs; + rhs.template head() = other; + rhs[Dim] = typename ResultType::Scalar(1); + Matrix res(T.matrix() * rhs); + return res.template head(); + } + }; -// generic HDim x HDim matrix * AffineCompact => Projective -template -struct transform_left_product_impl -{ - typedef Transform TransformType; - typedef typename TransformType::MatrixType MatrixType; - typedef Transform ResultType; - static ResultType run(const Other& other,const TransformType& tr) - { - ResultType res; - res.matrix().noalias() = other.template block(0,0) * tr.matrix(); - res.matrix().col(Dim) += other.col(Dim); - return res; - } -}; + /********************************************************** + *** Specializations of operator* with lhs EigenBase *** + **********************************************************/ -// affine matrix * T -template -struct transform_left_product_impl -{ - typedef Transform TransformType; - typedef typename TransformType::MatrixType MatrixType; - typedef TransformType ResultType; - static ResultType run(const Other& other,const TransformType& tr) + // generic HDim x HDim matrix * T => Projective + template + struct transform_left_product_impl { - ResultType res; - res.affine().noalias() = other * tr.matrix(); - res.matrix().row(Dim) = tr.matrix().row(Dim); - return res; - } -}; + typedef Transform TransformType; + typedef typename TransformType::MatrixType MatrixType; + typedef Transform ResultType; + static ResultType run(const Other &other, const TransformType &tr) { return ResultType(other * tr.matrix()); } + }; -// affine matrix * AffineCompact -template -struct transform_left_product_impl -{ - typedef Transform TransformType; - typedef typename TransformType::MatrixType MatrixType; - typedef TransformType ResultType; - static ResultType run(const Other& other,const TransformType& tr) + // generic HDim x HDim matrix * AffineCompact => Projective + template + struct transform_left_product_impl { - ResultType res; - res.matrix().noalias() = other.template block(0,0) * tr.matrix(); - res.translation() += other.col(Dim); - return res; - } -}; + typedef Transform TransformType; + typedef typename TransformType::MatrixType MatrixType; + typedef Transform ResultType; + static ResultType run(const Other &other, const TransformType &tr) + { + ResultType res; + res.matrix().noalias() = other.template block(0, 0) * tr.matrix(); + res.matrix().col(Dim) += other.col(Dim); + return res; + } + }; -// linear matrix * T -template -struct transform_left_product_impl -{ - typedef Transform TransformType; - typedef typename TransformType::MatrixType MatrixType; - typedef TransformType ResultType; - static ResultType run(const Other& other, const TransformType& tr) + // affine matrix * T + template + struct transform_left_product_impl { - TransformType res; - if(Mode!=int(AffineCompact)) + typedef Transform TransformType; + typedef typename TransformType::MatrixType MatrixType; + typedef TransformType ResultType; + static ResultType run(const Other &other, const TransformType &tr) + { + ResultType res; + res.affine().noalias() = other * tr.matrix(); res.matrix().row(Dim) = tr.matrix().row(Dim); - res.matrix().template topRows().noalias() - = other * tr.matrix().template topRows(); - return res; - } -}; + return res; + } + }; + + // affine matrix * AffineCompact + template + struct transform_left_product_impl + { + typedef Transform TransformType; + typedef typename TransformType::MatrixType MatrixType; + typedef TransformType ResultType; + static ResultType run(const Other &other, const TransformType &tr) + { + ResultType res; + res.matrix().noalias() = other.template block(0, 0) * tr.matrix(); + res.translation() += other.col(Dim); + return res; + } + }; -/********************************************************** -*** Specializations of operator* with another Transform *** -**********************************************************/ + // linear matrix * T + template + struct transform_left_product_impl + { + typedef Transform TransformType; + typedef typename TransformType::MatrixType MatrixType; + typedef TransformType ResultType; + static ResultType run(const Other &other, const TransformType &tr) + { + TransformType res; + if (Mode != int(AffineCompact)) res.matrix().row(Dim) = tr.matrix().row(Dim); + res.matrix().template topRows().noalias() = other * tr.matrix().template topRows(); + return res; + } + }; -template -struct transform_transform_product_impl,Transform,false > -{ - enum { ResultMode = transform_product_result::Mode }; - typedef Transform Lhs; - typedef Transform Rhs; - typedef Transform ResultType; - static ResultType run(const Lhs& lhs, const Rhs& rhs) + /********************************************************** + *** Specializations of operator* with another Transform *** + **********************************************************/ + + template + struct transform_transform_product_impl, + Transform, + false> { - ResultType res; - res.linear() = lhs.linear() * rhs.linear(); - res.translation() = lhs.linear() * rhs.translation() + lhs.translation(); - res.makeAffine(); - return res; - } -}; + enum { ResultMode = transform_product_result::Mode }; + typedef Transform Lhs; + typedef Transform Rhs; + typedef Transform ResultType; + static ResultType run(const Lhs &lhs, const Rhs &rhs) + { + ResultType res; + res.linear() = lhs.linear() * rhs.linear(); + res.translation() = lhs.linear() * rhs.translation() + lhs.translation(); + res.makeAffine(); + return res; + } + }; -template -struct transform_transform_product_impl,Transform,true > -{ - typedef Transform Lhs; - typedef Transform Rhs; - typedef Transform ResultType; - static ResultType run(const Lhs& lhs, const Rhs& rhs) + template + struct transform_transform_product_impl, + Transform, + true> { - return ResultType( lhs.matrix() * rhs.matrix() ); - } -}; + typedef Transform Lhs; + typedef Transform Rhs; + typedef Transform ResultType; + static ResultType run(const Lhs &lhs, const Rhs &rhs) { return ResultType(lhs.matrix() * rhs.matrix()); } + }; -template -struct transform_transform_product_impl,Transform,true > -{ - typedef Transform Lhs; - typedef Transform Rhs; - typedef Transform ResultType; - static ResultType run(const Lhs& lhs, const Rhs& rhs) + template + struct transform_transform_product_impl, + Transform, + true> { - ResultType res; - res.matrix().template topRows() = lhs.matrix() * rhs.matrix(); - res.matrix().row(Dim) = rhs.matrix().row(Dim); - return res; - } -}; + typedef Transform Lhs; + typedef Transform Rhs; + typedef Transform ResultType; + static ResultType run(const Lhs &lhs, const Rhs &rhs) + { + ResultType res; + res.matrix().template topRows() = lhs.matrix() * rhs.matrix(); + res.matrix().row(Dim) = rhs.matrix().row(Dim); + return res; + } + }; -template -struct transform_transform_product_impl,Transform,true > -{ - typedef Transform Lhs; - typedef Transform Rhs; - typedef Transform ResultType; - static ResultType run(const Lhs& lhs, const Rhs& rhs) + template + struct transform_transform_product_impl, + Transform, + true> { - ResultType res(lhs.matrix().template leftCols() * rhs.matrix()); - res.matrix().col(Dim) += lhs.matrix().col(Dim); - return res; - } -}; + typedef Transform Lhs; + typedef Transform Rhs; + typedef Transform ResultType; + static ResultType run(const Lhs &lhs, const Rhs &rhs) + { + ResultType res(lhs.matrix().template leftCols() * rhs.matrix()); + res.matrix().col(Dim) += lhs.matrix().col(Dim); + return res; + } + }; -} // end namespace internal +}// end namespace internal -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_TRANSFORM_H +#endif// EIGEN_TRANSFORM_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Geometry/Translation.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Geometry/Translation.h index 51d9a82e..2317e96f 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Geometry/Translation.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Geometry/Translation.h @@ -10,65 +10,62 @@ #ifndef EIGEN_TRANSLATION_H #define EIGEN_TRANSLATION_H -namespace Eigen { +namespace Eigen { /** \geometry_module \ingroup Geometry_Module - * - * \class Translation - * - * \brief Represents a translation transformation - * - * \tparam _Scalar the scalar type, i.e., the type of the coefficients. - * \tparam _Dim the dimension of the space, can be a compile time value or Dynamic - * - * \note This class is not aimed to be used to store a translation transformation, - * but rather to make easier the constructions and updates of Transform objects. - * - * \sa class Scaling, class Transform - */ -template -class Translation + * + * \class Translation + * + * \brief Represents a translation transformation + * + * \tparam _Scalar the scalar type, i.e., the type of the coefficients. + * \tparam _Dim the dimension of the space, can be a compile time value or Dynamic + * + * \note This class is not aimed to be used to store a translation transformation, + * but rather to make easier the constructions and updates of Transform objects. + * + * \sa class Scaling, class Transform + */ +template class Translation { public: - EIGEN_MAKE_ALIGNED_OPERATOR_NEW_IF_VECTORIZABLE_FIXED_SIZE(_Scalar,_Dim) + EIGEN_MAKE_ALIGNED_OPERATOR_NEW_IF_VECTORIZABLE_FIXED_SIZE(_Scalar, _Dim) /** dimension of the space */ enum { Dim = _Dim }; /** the scalar type of the coefficients */ typedef _Scalar Scalar; /** corresponding vector type */ - typedef Matrix VectorType; + typedef Matrix VectorType; /** corresponding linear transformation matrix type */ - typedef Matrix LinearMatrixType; + typedef Matrix LinearMatrixType; /** corresponding affine transformation type */ - typedef Transform AffineTransformType; + typedef Transform AffineTransformType; /** corresponding isometric transformation type */ - typedef Transform IsometryTransformType; + typedef Transform IsometryTransformType; protected: - VectorType m_coeffs; public: - /** Default constructor without initialization. */ EIGEN_DEVICE_FUNC Translation() {} /** */ - EIGEN_DEVICE_FUNC inline Translation(const Scalar& sx, const Scalar& sy) + EIGEN_DEVICE_FUNC inline Translation(const Scalar &sx, const Scalar &sy) { - eigen_assert(Dim==2); + eigen_assert(Dim == 2); m_coeffs.x() = sx; m_coeffs.y() = sy; } /** */ - EIGEN_DEVICE_FUNC inline Translation(const Scalar& sx, const Scalar& sy, const Scalar& sz) + EIGEN_DEVICE_FUNC inline Translation(const Scalar &sx, const Scalar &sy, const Scalar &sz) { - eigen_assert(Dim==3); + eigen_assert(Dim == 3); m_coeffs.x() = sx; m_coeffs.y() = sy; m_coeffs.z() = sz; } /** Constructs and initialize the translation transformation from a vector of translation coefficients */ - EIGEN_DEVICE_FUNC explicit inline Translation(const VectorType& vector) : m_coeffs(vector) {} + EIGEN_DEVICE_FUNC explicit inline Translation(const VectorType &vector) : m_coeffs(vector) {} /** \brief Retruns the x-translation by value. **/ EIGEN_DEVICE_FUNC inline Scalar x() const { return m_coeffs.x(); } @@ -78,67 +75,74 @@ class Translation EIGEN_DEVICE_FUNC inline Scalar z() const { return m_coeffs.z(); } /** \brief Retruns the x-translation as a reference. **/ - EIGEN_DEVICE_FUNC inline Scalar& x() { return m_coeffs.x(); } + EIGEN_DEVICE_FUNC inline Scalar &x() { return m_coeffs.x(); } /** \brief Retruns the y-translation as a reference. **/ - EIGEN_DEVICE_FUNC inline Scalar& y() { return m_coeffs.y(); } + EIGEN_DEVICE_FUNC inline Scalar &y() { return m_coeffs.y(); } /** \brief Retruns the z-translation as a reference. **/ - EIGEN_DEVICE_FUNC inline Scalar& z() { return m_coeffs.z(); } + EIGEN_DEVICE_FUNC inline Scalar &z() { return m_coeffs.z(); } - EIGEN_DEVICE_FUNC const VectorType& vector() const { return m_coeffs; } - EIGEN_DEVICE_FUNC VectorType& vector() { return m_coeffs; } + EIGEN_DEVICE_FUNC const VectorType &vector() const { return m_coeffs; } + EIGEN_DEVICE_FUNC VectorType &vector() { return m_coeffs; } - EIGEN_DEVICE_FUNC const VectorType& translation() const { return m_coeffs; } - EIGEN_DEVICE_FUNC VectorType& translation() { return m_coeffs; } + EIGEN_DEVICE_FUNC const VectorType &translation() const { return m_coeffs; } + EIGEN_DEVICE_FUNC VectorType &translation() { return m_coeffs; } /** Concatenates two translation */ - EIGEN_DEVICE_FUNC inline Translation operator* (const Translation& other) const - { return Translation(m_coeffs + other.m_coeffs); } + EIGEN_DEVICE_FUNC inline Translation operator*(const Translation &other) const + { + return Translation(m_coeffs + other.m_coeffs); + } /** Concatenates a translation and a uniform scaling */ - EIGEN_DEVICE_FUNC inline AffineTransformType operator* (const UniformScaling& other) const; + EIGEN_DEVICE_FUNC inline AffineTransformType operator*(const UniformScaling &other) const; /** Concatenates a translation and a linear transformation */ template - EIGEN_DEVICE_FUNC inline AffineTransformType operator* (const EigenBase& linear) const; + EIGEN_DEVICE_FUNC inline AffineTransformType operator*(const EigenBase &linear) const; /** Concatenates a translation and a rotation */ template - EIGEN_DEVICE_FUNC inline IsometryTransformType operator*(const RotationBase& r) const - { return *this * IsometryTransformType(r); } + EIGEN_DEVICE_FUNC inline IsometryTransformType operator*(const RotationBase &r) const + { + return *this * IsometryTransformType(r); + } /** \returns the concatenation of a linear transformation \a l with the translation \a t */ // its a nightmare to define a templated friend function outside its declaration - template friend - EIGEN_DEVICE_FUNC inline AffineTransformType operator*(const EigenBase& linear, const Translation& t) + template + friend EIGEN_DEVICE_FUNC inline AffineTransformType operator*(const EigenBase &linear, + const Translation &t) { AffineTransformType res; res.matrix().setZero(); res.linear() = linear.derived(); res.translation() = linear.derived() * t.m_coeffs; res.matrix().row(Dim).setZero(); - res(Dim,Dim) = Scalar(1); + res(Dim, Dim) = Scalar(1); return res; } /** Concatenates a translation and a transformation */ template - EIGEN_DEVICE_FUNC inline Transform operator* (const Transform& t) const + EIGEN_DEVICE_FUNC inline Transform operator*(const Transform &t) const { - Transform res = t; + Transform res = t; res.pretranslate(m_coeffs); return res; } /** Applies translation to vector */ template - inline typename internal::enable_if::type - operator* (const MatrixBase& vec) const - { return m_coeffs + vec.derived(); } + inline typename internal::enable_if::type operator*( + const MatrixBase &vec) const + { + return m_coeffs + vec.derived(); + } /** \returns the inverse translation (opposite) */ Translation inverse() const { return Translation(-m_coeffs); } - Translation& operator=(const Translation& other) + Translation &operator=(const Translation &other) { m_coeffs = other.m_coeffs; return *this; @@ -147,62 +151,69 @@ class Translation static const Translation Identity() { return Translation(VectorType::Zero()); } /** \returns \c *this with scalar type casted to \a NewScalarType - * - * Note that if \a NewScalarType is equal to the current scalar type of \c *this - * then this function smartly returns a const reference to \c *this. - */ + * + * Note that if \a NewScalarType is equal to the current scalar type of \c *this + * then this function smartly returns a const reference to \c *this. + */ template - EIGEN_DEVICE_FUNC inline typename internal::cast_return_type >::type cast() const - { return typename internal::cast_return_type >::type(*this); } + EIGEN_DEVICE_FUNC inline typename internal::cast_return_type>::type + cast() const + { + return typename internal::cast_return_type>::type(*this); + } /** Copy constructor with scalar type conversion */ template - EIGEN_DEVICE_FUNC inline explicit Translation(const Translation& other) - { m_coeffs = other.vector().template cast(); } + EIGEN_DEVICE_FUNC inline explicit Translation(const Translation &other) + { + m_coeffs = other.vector().template cast(); + } /** \returns \c true if \c *this is approximately equal to \a other, within the precision - * determined by \a prec. - * - * \sa MatrixBase::isApprox() */ - EIGEN_DEVICE_FUNC bool isApprox(const Translation& other, const typename NumTraits::Real& prec = NumTraits::dummy_precision()) const - { return m_coeffs.isApprox(other.m_coeffs, prec); } - + * determined by \a prec. + * + * \sa MatrixBase::isApprox() */ + EIGEN_DEVICE_FUNC bool isApprox(const Translation &other, + const typename NumTraits::Real &prec = NumTraits::dummy_precision()) const + { + return m_coeffs.isApprox(other.m_coeffs, prec); + } }; /** \addtogroup Geometry_Module */ //@{ typedef Translation Translation2f; -typedef Translation Translation2d; +typedef Translation Translation2d; typedef Translation Translation3f; -typedef Translation Translation3d; +typedef Translation Translation3d; //@} template -EIGEN_DEVICE_FUNC inline typename Translation::AffineTransformType -Translation::operator* (const UniformScaling& other) const +EIGEN_DEVICE_FUNC inline typename Translation::AffineTransformType Translation::operator*( + const UniformScaling &other) const { AffineTransformType res; res.matrix().setZero(); res.linear().diagonal().fill(other.factor()); res.translation() = m_coeffs; - res(Dim,Dim) = Scalar(1); + res(Dim, Dim) = Scalar(1); return res; } template template -EIGEN_DEVICE_FUNC inline typename Translation::AffineTransformType -Translation::operator* (const EigenBase& linear) const +EIGEN_DEVICE_FUNC inline typename Translation::AffineTransformType Translation::operator*( + const EigenBase &linear) const { AffineTransformType res; res.matrix().setZero(); res.linear() = linear.derived(); res.translation() = m_coeffs; res.matrix().row(Dim).setZero(); - res(Dim,Dim) = Scalar(1); + res(Dim, Dim) = Scalar(1); return res; } -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_TRANSLATION_H +#endif// EIGEN_TRANSLATION_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Geometry/Umeyama.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Geometry/Umeyama.h index 7e933fca..52fd9ba5 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Geometry/Umeyama.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Geometry/Umeyama.h @@ -10,13 +10,13 @@ #ifndef EIGEN_UMEYAMA_H #define EIGEN_UMEYAMA_H -// This file requires the user to include +// This file requires the user to include // * Eigen/Core -// * Eigen/LU +// * Eigen/LU // * Eigen/SVD // * Eigen/Array -namespace Eigen { +namespace Eigen { #ifndef EIGEN_PARSED_BY_DOXYGEN @@ -25,74 +25,74 @@ namespace Eigen { // cannot trivially be deduced when float and double types are mixed. namespace internal { -// Compile time return type deduction for different MatrixBase types. -// Different means here different alignment and parameters but the same underlying -// real scalar type. -template -struct umeyama_transform_matrix_type -{ - enum { - MinRowsAtCompileTime = EIGEN_SIZE_MIN_PREFER_DYNAMIC(MatrixType::RowsAtCompileTime, OtherMatrixType::RowsAtCompileTime), - - // When possible we want to choose some small fixed size value since the result - // is likely to fit on the stack. So here, EIGEN_SIZE_MIN_PREFER_DYNAMIC is not what we want. - HomogeneousDimension = int(MinRowsAtCompileTime) == Dynamic ? Dynamic : int(MinRowsAtCompileTime)+1 + // Compile time return type deduction for different MatrixBase types. + // Different means here different alignment and parameters but the same underlying + // real scalar type. + template struct umeyama_transform_matrix_type + { + enum { + MinRowsAtCompileTime = + EIGEN_SIZE_MIN_PREFER_DYNAMIC(MatrixType::RowsAtCompileTime, OtherMatrixType::RowsAtCompileTime), + + // When possible we want to choose some small fixed size value since the result + // is likely to fit on the stack. So here, EIGEN_SIZE_MIN_PREFER_DYNAMIC is not what we want. + HomogeneousDimension = int(MinRowsAtCompileTime) == Dynamic ? Dynamic : int(MinRowsAtCompileTime) + 1 + }; + + typedef Matrix::Scalar, + HomogeneousDimension, + HomogeneousDimension, + AutoAlign | (traits::Flags & RowMajorBit ? RowMajor : ColMajor), + HomogeneousDimension, + HomogeneousDimension> + type; }; - typedef Matrix::Scalar, - HomogeneousDimension, - HomogeneousDimension, - AutoAlign | (traits::Flags & RowMajorBit ? RowMajor : ColMajor), - HomogeneousDimension, - HomogeneousDimension - > type; -}; - -} +}// namespace internal #endif /** -* \geometry_module \ingroup Geometry_Module -* -* \brief Returns the transformation between two point sets. -* -* The algorithm is based on: -* "Least-squares estimation of transformation parameters between two point patterns", -* Shinji Umeyama, PAMI 1991, DOI: 10.1109/34.88573 -* -* It estimates parameters \f$ c, \mathbf{R}, \f$ and \f$ \mathbf{t} \f$ such that -* \f{align*} -* \frac{1}{n} \sum_{i=1}^n \vert\vert y_i - (c\mathbf{R}x_i + \mathbf{t}) \vert\vert_2^2 -* \f} -* is minimized. -* -* The algorithm is based on the analysis of the covariance matrix -* \f$ \Sigma_{\mathbf{x}\mathbf{y}} \in \mathbb{R}^{d \times d} \f$ -* of the input point sets \f$ \mathbf{x} \f$ and \f$ \mathbf{y} \f$ where -* \f$d\f$ is corresponding to the dimension (which is typically small). -* The analysis is involving the SVD having a complexity of \f$O(d^3)\f$ -* though the actual computational effort lies in the covariance -* matrix computation which has an asymptotic lower bound of \f$O(dm)\f$ when -* the input point sets have dimension \f$d \times m\f$. -* -* Currently the method is working only for floating point matrices. -* -* \todo Should the return type of umeyama() become a Transform? -* -* \param src Source points \f$ \mathbf{x} = \left( x_1, \hdots, x_n \right) \f$. -* \param dst Destination points \f$ \mathbf{y} = \left( y_1, \hdots, y_n \right) \f$. -* \param with_scaling Sets \f$ c=1 \f$ when false is passed. -* \return The homogeneous transformation -* \f{align*} -* T = \begin{bmatrix} c\mathbf{R} & \mathbf{t} \\ \mathbf{0} & 1 \end{bmatrix} -* \f} -* minimizing the resudiual above. This transformation is always returned as an -* Eigen::Matrix. -*/ -template + * \geometry_module \ingroup Geometry_Module + * + * \brief Returns the transformation between two point sets. + * + * The algorithm is based on: + * "Least-squares estimation of transformation parameters between two point patterns", + * Shinji Umeyama, PAMI 1991, DOI: 10.1109/34.88573 + * + * It estimates parameters \f$ c, \mathbf{R}, \f$ and \f$ \mathbf{t} \f$ such that + * \f{align*} + * \frac{1}{n} \sum_{i=1}^n \vert\vert y_i - (c\mathbf{R}x_i + \mathbf{t}) \vert\vert_2^2 + * \f} + * is minimized. + * + * The algorithm is based on the analysis of the covariance matrix + * \f$ \Sigma_{\mathbf{x}\mathbf{y}} \in \mathbb{R}^{d \times d} \f$ + * of the input point sets \f$ \mathbf{x} \f$ and \f$ \mathbf{y} \f$ where + * \f$d\f$ is corresponding to the dimension (which is typically small). + * The analysis is involving the SVD having a complexity of \f$O(d^3)\f$ + * though the actual computational effort lies in the covariance + * matrix computation which has an asymptotic lower bound of \f$O(dm)\f$ when + * the input point sets have dimension \f$d \times m\f$. + * + * Currently the method is working only for floating point matrices. + * + * \todo Should the return type of umeyama() become a Transform? + * + * \param src Source points \f$ \mathbf{x} = \left( x_1, \hdots, x_n \right) \f$. + * \param dst Destination points \f$ \mathbf{y} = \left( y_1, \hdots, y_n \right) \f$. + * \param with_scaling Sets \f$ c=1 \f$ when false is passed. + * \return The homogeneous transformation + * \f{align*} + * T = \begin{bmatrix} c\mathbf{R} & \mathbf{t} \\ \mathbf{0} & 1 \end{bmatrix} + * \f} + * minimizing the resudiual above. This transformation is always returned as an + * Eigen::Matrix. + */ +template typename internal::umeyama_transform_matrix_type::type -umeyama(const MatrixBase& src, const MatrixBase& dst, bool with_scaling = true) + umeyama(const MatrixBase &src, const MatrixBase &dst, bool with_scaling = true) { typedef typename internal::umeyama_transform_matrix_type::type TransformationMatrixType; typedef typename internal::traits::Scalar Scalar; @@ -108,8 +108,8 @@ umeyama(const MatrixBase& src, const MatrixBase& dst, boo typedef Matrix MatrixType; typedef typename internal::plain_matrix_type_row_major::type RowMajorMatrixType; - const Index m = src.rows(); // dimension - const Index n = src.cols(); // number of measurements + const Index m = src.rows();// dimension + const Index n = src.cols();// number of measurements // required for demeaning ... const RealScalar one_over_n = RealScalar(1) / static_cast(n); @@ -131,36 +131,32 @@ umeyama(const MatrixBase& src, const MatrixBase& dst, boo JacobiSVD svd(sigma, ComputeFullU | ComputeFullV); // Initialize the resulting transformation with an identity matrix... - TransformationMatrixType Rt = TransformationMatrixType::Identity(m+1,m+1); + TransformationMatrixType Rt = TransformationMatrixType::Identity(m + 1, m + 1); // Eq. (39) VectorType S = VectorType::Ones(m); - if ( svd.matrixU().determinant() * svd.matrixV().determinant() < 0 ) - S(m-1) = -1; + if (svd.matrixU().determinant() * svd.matrixV().determinant() < 0) S(m - 1) = -1; // Eq. (40) and (43) - Rt.block(0,0,m,m).noalias() = svd.matrixU() * S.asDiagonal() * svd.matrixV().transpose(); + Rt.block(0, 0, m, m).noalias() = svd.matrixU() * S.asDiagonal() * svd.matrixV().transpose(); - if (with_scaling) - { + if (with_scaling) { // Eq. (42) - const Scalar c = Scalar(1)/src_var * svd.singularValues().dot(S); + const Scalar c = Scalar(1) / src_var * svd.singularValues().dot(S); // Eq. (41) Rt.col(m).head(m) = dst_mean; - Rt.col(m).head(m).noalias() -= c*Rt.topLeftCorner(m,m)*src_mean; - Rt.block(0,0,m,m) *= c; - } - else - { + Rt.col(m).head(m).noalias() -= c * Rt.topLeftCorner(m, m) * src_mean; + Rt.block(0, 0, m, m) *= c; + } else { Rt.col(m).head(m) = dst_mean; - Rt.col(m).head(m).noalias() -= Rt.topLeftCorner(m,m)*src_mean; + Rt.col(m).head(m).noalias() -= Rt.topLeftCorner(m, m) * src_mean; } return Rt; } -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_UMEYAMA_H +#endif// EIGEN_UMEYAMA_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Geometry/arch/Geometry_SSE.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Geometry/arch/Geometry_SSE.h index f68cab58..d9fdf332 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Geometry/arch/Geometry_SSE.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Geometry/arch/Geometry_SSE.h @@ -11,151 +11,136 @@ #ifndef EIGEN_GEOMETRY_SSE_H #define EIGEN_GEOMETRY_SSE_H -namespace Eigen { +namespace Eigen { namespace internal { -template -struct quat_product -{ - enum { - AAlignment = traits::Alignment, - BAlignment = traits::Alignment, - ResAlignment = traits >::Alignment - }; - static inline Quaternion run(const QuaternionBase& _a, const QuaternionBase& _b) + template struct quat_product { - Quaternion res; - const __m128 mask = _mm_setr_ps(0.f,0.f,0.f,-0.f); - __m128 a = _a.coeffs().template packet(0); - __m128 b = _b.coeffs().template packet(0); - __m128 s1 = _mm_mul_ps(vec4f_swizzle1(a,1,2,0,2),vec4f_swizzle1(b,2,0,1,2)); - __m128 s2 = _mm_mul_ps(vec4f_swizzle1(a,3,3,3,1),vec4f_swizzle1(b,0,1,2,1)); - pstoret( - &res.x(), - _mm_add_ps(_mm_sub_ps(_mm_mul_ps(a,vec4f_swizzle1(b,3,3,3,3)), - _mm_mul_ps(vec4f_swizzle1(a,2,0,1,0), - vec4f_swizzle1(b,1,2,0,0))), - _mm_xor_ps(mask,_mm_add_ps(s1,s2)))); - - return res; - } -}; - -template -struct quat_conj -{ - enum { - ResAlignment = traits >::Alignment + enum { + AAlignment = traits::Alignment, + BAlignment = traits::Alignment, + ResAlignment = traits>::Alignment + }; + static inline Quaternion run(const QuaternionBase &_a, const QuaternionBase &_b) + { + Quaternion res; + const __m128 mask = _mm_setr_ps(0.f, 0.f, 0.f, -0.f); + __m128 a = _a.coeffs().template packet(0); + __m128 b = _b.coeffs().template packet(0); + __m128 s1 = _mm_mul_ps(vec4f_swizzle1(a, 1, 2, 0, 2), vec4f_swizzle1(b, 2, 0, 1, 2)); + __m128 s2 = _mm_mul_ps(vec4f_swizzle1(a, 3, 3, 3, 1), vec4f_swizzle1(b, 0, 1, 2, 1)); + pstoret(&res.x(), + _mm_add_ps(_mm_sub_ps(_mm_mul_ps(a, vec4f_swizzle1(b, 3, 3, 3, 3)), + _mm_mul_ps(vec4f_swizzle1(a, 2, 0, 1, 0), vec4f_swizzle1(b, 1, 2, 0, 0))), + _mm_xor_ps(mask, _mm_add_ps(s1, s2)))); + + return res; + } }; - static inline Quaternion run(const QuaternionBase& q) + + template struct quat_conj { - Quaternion res; - const __m128 mask = _mm_setr_ps(-0.f,-0.f,-0.f,0.f); - pstoret(&res.x(), _mm_xor_ps(mask, q.coeffs().template packet::Alignment>(0))); - return res; - } -}; - - -template -struct cross3_impl -{ - enum { - ResAlignment = traits::type>::Alignment + enum { ResAlignment = traits>::Alignment }; + static inline Quaternion run(const QuaternionBase &q) + { + Quaternion res; + const __m128 mask = _mm_setr_ps(-0.f, -0.f, -0.f, 0.f); + pstoret( + &res.x(), _mm_xor_ps(mask, q.coeffs().template packet::Alignment>(0))); + return res; + } }; - static inline typename plain_matrix_type::type - run(const VectorLhs& lhs, const VectorRhs& rhs) - { - __m128 a = lhs.template packet::Alignment>(0); - __m128 b = rhs.template packet::Alignment>(0); - __m128 mul1=_mm_mul_ps(vec4f_swizzle1(a,1,2,0,3),vec4f_swizzle1(b,2,0,1,3)); - __m128 mul2=_mm_mul_ps(vec4f_swizzle1(a,2,0,1,3),vec4f_swizzle1(b,1,2,0,3)); - typename plain_matrix_type::type res; - pstoret(&res.x(),_mm_sub_ps(mul1,mul2)); - return res; - } -}; - - -template -struct quat_product -{ - enum { - BAlignment = traits::Alignment, - ResAlignment = traits >::Alignment + template + struct cross3_impl + { + enum { ResAlignment = traits::type>::Alignment }; + static inline typename plain_matrix_type::type run(const VectorLhs &lhs, const VectorRhs &rhs) + { + __m128 a = lhs.template packet::Alignment>(0); + __m128 b = rhs.template packet::Alignment>(0); + __m128 mul1 = _mm_mul_ps(vec4f_swizzle1(a, 1, 2, 0, 3), vec4f_swizzle1(b, 2, 0, 1, 3)); + __m128 mul2 = _mm_mul_ps(vec4f_swizzle1(a, 2, 0, 1, 3), vec4f_swizzle1(b, 1, 2, 0, 3)); + typename plain_matrix_type::type res; + pstoret(&res.x(), _mm_sub_ps(mul1, mul2)); + return res; + } }; - static inline Quaternion run(const QuaternionBase& _a, const QuaternionBase& _b) + + template struct quat_product { - const Packet2d mask = _mm_castsi128_pd(_mm_set_epi32(0x0,0x0,0x80000000,0x0)); - - Quaternion res; - - const double* a = _a.coeffs().data(); - Packet2d b_xy = _b.coeffs().template packet(0); - Packet2d b_zw = _b.coeffs().template packet(2); - Packet2d a_xx = pset1(a[0]); - Packet2d a_yy = pset1(a[1]); - Packet2d a_zz = pset1(a[2]); - Packet2d a_ww = pset1(a[3]); - - // two temporaries: - Packet2d t1, t2; - - /* - * t1 = ww*xy + yy*zw - * t2 = zz*xy - xx*zw - * res.xy = t1 +/- swap(t2) - */ - t1 = padd(pmul(a_ww, b_xy), pmul(a_yy, b_zw)); - t2 = psub(pmul(a_zz, b_xy), pmul(a_xx, b_zw)); + enum { BAlignment = traits::Alignment, ResAlignment = traits>::Alignment }; + + static inline Quaternion run(const QuaternionBase &_a, const QuaternionBase &_b) + { + const Packet2d mask = _mm_castsi128_pd(_mm_set_epi32(0x0, 0x0, 0x80000000, 0x0)); + + Quaternion res; + + const double *a = _a.coeffs().data(); + Packet2d b_xy = _b.coeffs().template packet(0); + Packet2d b_zw = _b.coeffs().template packet(2); + Packet2d a_xx = pset1(a[0]); + Packet2d a_yy = pset1(a[1]); + Packet2d a_zz = pset1(a[2]); + Packet2d a_ww = pset1(a[3]); + + // two temporaries: + Packet2d t1, t2; + + /* + * t1 = ww*xy + yy*zw + * t2 = zz*xy - xx*zw + * res.xy = t1 +/- swap(t2) + */ + t1 = padd(pmul(a_ww, b_xy), pmul(a_yy, b_zw)); + t2 = psub(pmul(a_zz, b_xy), pmul(a_xx, b_zw)); #ifdef EIGEN_VECTORIZE_SSE3 - EIGEN_UNUSED_VARIABLE(mask) - pstoret(&res.x(), _mm_addsub_pd(t1, preverse(t2))); + EIGEN_UNUSED_VARIABLE(mask) + pstoret(&res.x(), _mm_addsub_pd(t1, preverse(t2))); #else - pstoret(&res.x(), padd(t1, pxor(mask,preverse(t2)))); + pstoret(&res.x(), padd(t1, pxor(mask, preverse(t2)))); #endif - - /* - * t1 = ww*zw - yy*xy - * t2 = zz*zw + xx*xy - * res.zw = t1 -/+ swap(t2) = swap( swap(t1) +/- t2) - */ - t1 = psub(pmul(a_ww, b_zw), pmul(a_yy, b_xy)); - t2 = padd(pmul(a_zz, b_zw), pmul(a_xx, b_xy)); + + /* + * t1 = ww*zw - yy*xy + * t2 = zz*zw + xx*xy + * res.zw = t1 -/+ swap(t2) = swap( swap(t1) +/- t2) + */ + t1 = psub(pmul(a_ww, b_zw), pmul(a_yy, b_xy)); + t2 = padd(pmul(a_zz, b_zw), pmul(a_xx, b_xy)); #ifdef EIGEN_VECTORIZE_SSE3 - EIGEN_UNUSED_VARIABLE(mask) - pstoret(&res.z(), preverse(_mm_addsub_pd(preverse(t1), t2))); + EIGEN_UNUSED_VARIABLE(mask) + pstoret(&res.z(), preverse(_mm_addsub_pd(preverse(t1), t2))); #else - pstoret(&res.z(), psub(t1, pxor(mask,preverse(t2)))); + pstoret(&res.z(), psub(t1, pxor(mask, preverse(t2)))); #endif - return res; -} -}; - -template -struct quat_conj -{ - enum { - ResAlignment = traits >::Alignment + return res; + } }; - static inline Quaternion run(const QuaternionBase& q) + + template struct quat_conj { - Quaternion res; - const __m128d mask0 = _mm_setr_pd(-0.,-0.); - const __m128d mask2 = _mm_setr_pd(-0.,0.); - pstoret(&res.x(), _mm_xor_pd(mask0, q.coeffs().template packet::Alignment>(0))); - pstoret(&res.z(), _mm_xor_pd(mask2, q.coeffs().template packet::Alignment>(2))); - return res; - } -}; + enum { ResAlignment = traits>::Alignment }; + static inline Quaternion run(const QuaternionBase &q) + { + Quaternion res; + const __m128d mask0 = _mm_setr_pd(-0., -0.); + const __m128d mask2 = _mm_setr_pd(-0., 0.); + pstoret( + &res.x(), _mm_xor_pd(mask0, q.coeffs().template packet::Alignment>(0))); + pstoret( + &res.z(), _mm_xor_pd(mask2, q.coeffs().template packet::Alignment>(2))); + return res; + } + }; -} // end namespace internal +}// end namespace internal -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_GEOMETRY_SSE_H +#endif// EIGEN_GEOMETRY_SSE_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Householder/BlockHouseholder.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Householder/BlockHouseholder.h index 01a7ed18..843df5d6 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Householder/BlockHouseholder.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Householder/BlockHouseholder.h @@ -13,91 +13,103 @@ // This file contains some helper function to deal with block householder reflectors -namespace Eigen { +namespace Eigen { namespace internal { - -/** \internal */ -// template -// void make_block_householder_triangular_factor(TriangularFactorType& triFactor, const VectorsType& vectors, const CoeffsType& hCoeffs) -// { -// typedef typename VectorsType::Scalar Scalar; -// const Index nbVecs = vectors.cols(); -// eigen_assert(triFactor.rows() == nbVecs && triFactor.cols() == nbVecs && vectors.rows()>=nbVecs); -// -// for(Index i = 0; i < nbVecs; i++) -// { -// Index rs = vectors.rows() - i; -// // Warning, note that hCoeffs may alias with vectors. -// // It is then necessary to copy it before modifying vectors(i,i). -// typename CoeffsType::Scalar h = hCoeffs(i); -// // This hack permits to pass trough nested Block<> and Transpose<> expressions. -// Scalar *Vii_ptr = const_cast(vectors.data() + vectors.outerStride()*i + vectors.innerStride()*i); -// Scalar Vii = *Vii_ptr; -// *Vii_ptr = Scalar(1); -// triFactor.col(i).head(i).noalias() = -h * vectors.block(i, 0, rs, i).adjoint() -// * vectors.col(i).tail(rs); -// *Vii_ptr = Vii; -// // FIXME add .noalias() once the triangular product can work inplace -// triFactor.col(i).head(i) = triFactor.block(0,0,i,i).template triangularView() -// * triFactor.col(i).head(i); -// triFactor(i,i) = hCoeffs(i); -// } -// } -/** \internal */ -// This variant avoid modifications in vectors -template -void make_block_householder_triangular_factor(TriangularFactorType& triFactor, const VectorsType& vectors, const CoeffsType& hCoeffs) -{ - const Index nbVecs = vectors.cols(); - eigen_assert(triFactor.rows() == nbVecs && triFactor.cols() == nbVecs && vectors.rows()>=nbVecs); + /** \internal */ + // template + // void make_block_householder_triangular_factor(TriangularFactorType& triFactor, const VectorsType& vectors, const + // CoeffsType& hCoeffs) + // { + // typedef typename VectorsType::Scalar Scalar; + // const Index nbVecs = vectors.cols(); + // eigen_assert(triFactor.rows() == nbVecs && triFactor.cols() == nbVecs && vectors.rows()>=nbVecs); + // + // for(Index i = 0; i < nbVecs; i++) + // { + // Index rs = vectors.rows() - i; + // // Warning, note that hCoeffs may alias with vectors. + // // It is then necessary to copy it before modifying vectors(i,i). + // typename CoeffsType::Scalar h = hCoeffs(i); + // // This hack permits to pass trough nested Block<> and Transpose<> expressions. + // Scalar *Vii_ptr = const_cast(vectors.data() + vectors.outerStride()*i + vectors.innerStride()*i); + // Scalar Vii = *Vii_ptr; + // *Vii_ptr = Scalar(1); + // triFactor.col(i).head(i).noalias() = -h * vectors.block(i, 0, rs, i).adjoint() + // * vectors.col(i).tail(rs); + // *Vii_ptr = Vii; + // // FIXME add .noalias() once the triangular product can work inplace + // triFactor.col(i).head(i) = triFactor.block(0,0,i,i).template triangularView() + // * triFactor.col(i).head(i); + // triFactor(i,i) = hCoeffs(i); + // } + // } - for(Index i = nbVecs-1; i >=0 ; --i) + /** \internal */ + // This variant avoid modifications in vectors + template + void make_block_householder_triangular_factor(TriangularFactorType &triFactor, + const VectorsType &vectors, + const CoeffsType &hCoeffs) { - Index rs = vectors.rows() - i - 1; - Index rt = nbVecs-i-1; + const Index nbVecs = vectors.cols(); + eigen_assert(triFactor.rows() == nbVecs && triFactor.cols() == nbVecs && vectors.rows() >= nbVecs); - if(rt>0) - { - triFactor.row(i).tail(rt).noalias() = -hCoeffs(i) * vectors.col(i).tail(rs).adjoint() - * vectors.bottomRightCorner(rs, rt).template triangularView(); - - // FIXME add .noalias() once the triangular product can work inplace - triFactor.row(i).tail(rt) = triFactor.row(i).tail(rt) * triFactor.bottomRightCorner(rt,rt).template triangularView(); - + for (Index i = nbVecs - 1; i >= 0; --i) { + Index rs = vectors.rows() - i - 1; + Index rt = nbVecs - i - 1; + + if (rt > 0) { + triFactor.row(i).tail(rt).noalias() = -hCoeffs(i) * vectors.col(i).tail(rs).adjoint() + * vectors.bottomRightCorner(rs, rt).template triangularView(); + + // FIXME add .noalias() once the triangular product can work inplace + triFactor.row(i).tail(rt) = + triFactor.row(i).tail(rt) * triFactor.bottomRightCorner(rt, rt).template triangularView(); + } + triFactor(i, i) = hCoeffs(i); } - triFactor(i,i) = hCoeffs(i); } -} -/** \internal - * if forward then perform mat = H0 * H1 * H2 * mat - * otherwise perform mat = H2 * H1 * H0 * mat - */ -template -void apply_block_householder_on_the_left(MatrixType& mat, const VectorsType& vectors, const CoeffsType& hCoeffs, bool forward) -{ - enum { TFactorSize = MatrixType::ColsAtCompileTime }; - Index nbVecs = vectors.cols(); - Matrix T(nbVecs,nbVecs); - - if(forward) make_block_householder_triangular_factor(T, vectors, hCoeffs); - else make_block_householder_triangular_factor(T, vectors, hCoeffs.conjugate()); - const TriangularView V(vectors); + /** \internal + * if forward then perform mat = H0 * H1 * H2 * mat + * otherwise perform mat = H2 * H1 * H0 * mat + */ + template + void apply_block_householder_on_the_left(MatrixType &mat, + const VectorsType &vectors, + const CoeffsType &hCoeffs, + bool forward) + { + enum { TFactorSize = MatrixType::ColsAtCompileTime }; + Index nbVecs = vectors.cols(); + Matrix T(nbVecs, nbVecs); - // A -= V T V^* A - Matrix tmp = V.adjoint() * mat; - // FIXME add .noalias() once the triangular product can work inplace - if(forward) tmp = T.template triangularView() * tmp; - else tmp = T.template triangularView().adjoint() * tmp; - mat.noalias() -= V * tmp; -} + if (forward) + make_block_householder_triangular_factor(T, vectors, hCoeffs); + else + make_block_householder_triangular_factor(T, vectors, hCoeffs.conjugate()); + const TriangularView V(vectors); + + // A -= V T V^* A + Matrix + tmp = V.adjoint() * mat; + // FIXME add .noalias() once the triangular product can work inplace + if (forward) + tmp = T.template triangularView() * tmp; + else + tmp = T.template triangularView().adjoint() * tmp; + mat.noalias() -= V * tmp; + } -} // end namespace internal +}// end namespace internal -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_BLOCK_HOUSEHOLDER_H +#endif// EIGEN_BLOCK_HOUSEHOLDER_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Householder/Householder.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Householder/Householder.h index 80de2c30..5a07b925 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Householder/Householder.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Householder/Householder.h @@ -11,118 +11,105 @@ #ifndef EIGEN_HOUSEHOLDER_H #define EIGEN_HOUSEHOLDER_H -namespace Eigen { +namespace Eigen { namespace internal { -template struct decrement_size -{ - enum { - ret = n==Dynamic ? n : n-1 + template struct decrement_size + { + enum { ret = n == Dynamic ? n : n - 1 }; }; -}; -} +}// namespace internal /** Computes the elementary reflector H such that: - * \f$ H *this = [ beta 0 ... 0]^T \f$ - * where the transformation H is: - * \f$ H = I - tau v v^*\f$ - * and the vector v is: - * \f$ v^T = [1 essential^T] \f$ - * - * The essential part of the vector \c v is stored in *this. - * - * On output: - * \param tau the scaling factor of the Householder transformation - * \param beta the result of H * \c *this - * - * \sa MatrixBase::makeHouseholder(), MatrixBase::applyHouseholderOnTheLeft(), - * MatrixBase::applyHouseholderOnTheRight() - */ -template -void MatrixBase::makeHouseholderInPlace(Scalar& tau, RealScalar& beta) + * \f$ H *this = [ beta 0 ... 0]^T \f$ + * where the transformation H is: + * \f$ H = I - tau v v^*\f$ + * and the vector v is: + * \f$ v^T = [1 essential^T] \f$ + * + * The essential part of the vector \c v is stored in *this. + * + * On output: + * \param tau the scaling factor of the Householder transformation + * \param beta the result of H * \c *this + * + * \sa MatrixBase::makeHouseholder(), MatrixBase::applyHouseholderOnTheLeft(), + * MatrixBase::applyHouseholderOnTheRight() + */ +template void MatrixBase::makeHouseholderInPlace(Scalar &tau, RealScalar &beta) { - VectorBlock::ret> essentialPart(derived(), 1, size()-1); + VectorBlock::ret> essentialPart(derived(), 1, size() - 1); makeHouseholder(essentialPart, tau, beta); } /** Computes the elementary reflector H such that: - * \f$ H *this = [ beta 0 ... 0]^T \f$ - * where the transformation H is: - * \f$ H = I - tau v v^*\f$ - * and the vector v is: - * \f$ v^T = [1 essential^T] \f$ - * - * On output: - * \param essential the essential part of the vector \c v - * \param tau the scaling factor of the Householder transformation - * \param beta the result of H * \c *this - * - * \sa MatrixBase::makeHouseholderInPlace(), MatrixBase::applyHouseholderOnTheLeft(), - * MatrixBase::applyHouseholderOnTheRight() - */ + * \f$ H *this = [ beta 0 ... 0]^T \f$ + * where the transformation H is: + * \f$ H = I - tau v v^*\f$ + * and the vector v is: + * \f$ v^T = [1 essential^T] \f$ + * + * On output: + * \param essential the essential part of the vector \c v + * \param tau the scaling factor of the Householder transformation + * \param beta the result of H * \c *this + * + * \sa MatrixBase::makeHouseholderInPlace(), MatrixBase::applyHouseholderOnTheLeft(), + * MatrixBase::applyHouseholderOnTheRight() + */ template template -void MatrixBase::makeHouseholder( - EssentialPart& essential, - Scalar& tau, - RealScalar& beta) const +void MatrixBase::makeHouseholder(EssentialPart &essential, Scalar &tau, RealScalar &beta) const { using std::sqrt; using numext::conj; - + EIGEN_STATIC_ASSERT_VECTOR_ONLY(EssentialPart) - VectorBlock tail(derived(), 1, size()-1); - - RealScalar tailSqNorm = size()==1 ? RealScalar(0) : tail.squaredNorm(); + VectorBlock tail(derived(), 1, size() - 1); + + RealScalar tailSqNorm = size() == 1 ? RealScalar(0) : tail.squaredNorm(); Scalar c0 = coeff(0); const RealScalar tol = (std::numeric_limits::min)(); - if(tailSqNorm <= tol && numext::abs2(numext::imag(c0))<=tol) - { + if (tailSqNorm <= tol && numext::abs2(numext::imag(c0)) <= tol) { tau = RealScalar(0); beta = numext::real(c0); essential.setZero(); - } - else - { + } else { beta = sqrt(numext::abs2(c0) + tailSqNorm); - if (numext::real(c0)>=RealScalar(0)) - beta = -beta; + if (numext::real(c0) >= RealScalar(0)) beta = -beta; essential = tail / (c0 - beta); tau = conj((beta - c0) / beta); } } /** Apply the elementary reflector H given by - * \f$ H = I - tau v v^*\f$ - * with - * \f$ v^T = [1 essential^T] \f$ - * from the left to a vector or matrix. - * - * On input: - * \param essential the essential part of the vector \c v - * \param tau the scaling factor of the Householder transformation - * \param workspace a pointer to working space with at least - * this->cols() * essential.size() entries - * - * \sa MatrixBase::makeHouseholder(), MatrixBase::makeHouseholderInPlace(), - * MatrixBase::applyHouseholderOnTheRight() - */ + * \f$ H = I - tau v v^*\f$ + * with + * \f$ v^T = [1 essential^T] \f$ + * from the left to a vector or matrix. + * + * On input: + * \param essential the essential part of the vector \c v + * \param tau the scaling factor of the Householder transformation + * \param workspace a pointer to working space with at least + * this->cols() * essential.size() entries + * + * \sa MatrixBase::makeHouseholder(), MatrixBase::makeHouseholderInPlace(), + * MatrixBase::applyHouseholderOnTheRight() + */ template template -void MatrixBase::applyHouseholderOnTheLeft( - const EssentialPart& essential, - const Scalar& tau, - Scalar* workspace) +void MatrixBase::applyHouseholderOnTheLeft(const EssentialPart &essential, + const Scalar &tau, + Scalar *workspace) { - if(rows() == 1) - { - *this *= Scalar(1)-tau; - } - else if(tau!=Scalar(0)) - { - Map::type> tmp(workspace,cols()); - Block bottom(derived(), 1, 0, rows()-1, cols()); + if (rows() == 1) { + *this *= Scalar(1) - tau; + } else if (tau != Scalar(0)) { + Map::type> tmp(workspace, cols()); + Block bottom( + derived(), 1, 0, rows() - 1, cols()); tmp.noalias() = essential.adjoint() * bottom; tmp += this->row(0); this->row(0) -= tau * tmp; @@ -131,35 +118,32 @@ void MatrixBase::applyHouseholderOnTheLeft( } /** Apply the elementary reflector H given by - * \f$ H = I - tau v v^*\f$ - * with - * \f$ v^T = [1 essential^T] \f$ - * from the right to a vector or matrix. - * - * On input: - * \param essential the essential part of the vector \c v - * \param tau the scaling factor of the Householder transformation - * \param workspace a pointer to working space with at least - * this->cols() * essential.size() entries - * - * \sa MatrixBase::makeHouseholder(), MatrixBase::makeHouseholderInPlace(), - * MatrixBase::applyHouseholderOnTheLeft() - */ + * \f$ H = I - tau v v^*\f$ + * with + * \f$ v^T = [1 essential^T] \f$ + * from the right to a vector or matrix. + * + * On input: + * \param essential the essential part of the vector \c v + * \param tau the scaling factor of the Householder transformation + * \param workspace a pointer to working space with at least + * this->cols() * essential.size() entries + * + * \sa MatrixBase::makeHouseholder(), MatrixBase::makeHouseholderInPlace(), + * MatrixBase::applyHouseholderOnTheLeft() + */ template template -void MatrixBase::applyHouseholderOnTheRight( - const EssentialPart& essential, - const Scalar& tau, - Scalar* workspace) +void MatrixBase::applyHouseholderOnTheRight(const EssentialPart &essential, + const Scalar &tau, + Scalar *workspace) { - if(cols() == 1) - { - *this *= Scalar(1)-tau; - } - else if(tau!=Scalar(0)) - { - Map::type> tmp(workspace,rows()); - Block right(derived(), 0, 1, rows(), cols()-1); + if (cols() == 1) { + *this *= Scalar(1) - tau; + } else if (tau != Scalar(0)) { + Map::type> tmp(workspace, rows()); + Block right( + derived(), 0, 1, rows(), cols() - 1); tmp.noalias() = right * essential.conjugate(); tmp += this->col(0); this->col(0) -= tau * tmp; @@ -167,6 +151,6 @@ void MatrixBase::applyHouseholderOnTheRight( } } -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_HOUSEHOLDER_H +#endif// EIGEN_HOUSEHOLDER_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Householder/HouseholderSequence.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Householder/HouseholderSequence.h index 3ce0a693..0328f7e6 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Householder/HouseholderSequence.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Householder/HouseholderSequence.h @@ -11,460 +11,444 @@ #ifndef EIGEN_HOUSEHOLDER_SEQUENCE_H #define EIGEN_HOUSEHOLDER_SEQUENCE_H -namespace Eigen { +namespace Eigen { /** \ingroup Householder_Module - * \householder_module - * \class HouseholderSequence - * \brief Sequence of Householder reflections acting on subspaces with decreasing size - * \tparam VectorsType type of matrix containing the Householder vectors - * \tparam CoeffsType type of vector containing the Householder coefficients - * \tparam Side either OnTheLeft (the default) or OnTheRight - * - * This class represents a product sequence of Householder reflections where the first Householder reflection - * acts on the whole space, the second Householder reflection leaves the one-dimensional subspace spanned by - * the first unit vector invariant, the third Householder reflection leaves the two-dimensional subspace - * spanned by the first two unit vectors invariant, and so on up to the last reflection which leaves all but - * one dimensions invariant and acts only on the last dimension. Such sequences of Householder reflections - * are used in several algorithms to zero out certain parts of a matrix. Indeed, the methods - * HessenbergDecomposition::matrixQ(), Tridiagonalization::matrixQ(), HouseholderQR::householderQ(), - * and ColPivHouseholderQR::householderQ() all return a %HouseholderSequence. - * - * More precisely, the class %HouseholderSequence represents an \f$ n \times n \f$ matrix \f$ H \f$ of the - * form \f$ H = \prod_{i=0}^{n-1} H_i \f$ where the i-th Householder reflection is \f$ H_i = I - h_i v_i - * v_i^* \f$. The i-th Householder coefficient \f$ h_i \f$ is a scalar and the i-th Householder vector \f$ - * v_i \f$ is a vector of the form - * \f[ - * v_i = [\underbrace{0, \ldots, 0}_{i-1\mbox{ zeros}}, 1, \underbrace{*, \ldots,*}_{n-i\mbox{ arbitrary entries}} ]. - * \f] - * The last \f$ n-i \f$ entries of \f$ v_i \f$ are called the essential part of the Householder vector. - * - * Typical usages are listed below, where H is a HouseholderSequence: - * \code - * A.applyOnTheRight(H); // A = A * H - * A.applyOnTheLeft(H); // A = H * A - * A.applyOnTheRight(H.adjoint()); // A = A * H^* - * A.applyOnTheLeft(H.adjoint()); // A = H^* * A - * MatrixXd Q = H; // conversion to a dense matrix - * \endcode - * In addition to the adjoint, you can also apply the inverse (=adjoint), the transpose, and the conjugate operators. - * - * See the documentation for HouseholderSequence(const VectorsType&, const CoeffsType&) for an example. - * - * \sa MatrixBase::applyOnTheLeft(), MatrixBase::applyOnTheRight() - */ + * \householder_module + * \class HouseholderSequence + * \brief Sequence of Householder reflections acting on subspaces with decreasing size + * \tparam VectorsType type of matrix containing the Householder vectors + * \tparam CoeffsType type of vector containing the Householder coefficients + * \tparam Side either OnTheLeft (the default) or OnTheRight + * + * This class represents a product sequence of Householder reflections where the first Householder reflection + * acts on the whole space, the second Householder reflection leaves the one-dimensional subspace spanned by + * the first unit vector invariant, the third Householder reflection leaves the two-dimensional subspace + * spanned by the first two unit vectors invariant, and so on up to the last reflection which leaves all but + * one dimensions invariant and acts only on the last dimension. Such sequences of Householder reflections + * are used in several algorithms to zero out certain parts of a matrix. Indeed, the methods + * HessenbergDecomposition::matrixQ(), Tridiagonalization::matrixQ(), HouseholderQR::householderQ(), + * and ColPivHouseholderQR::householderQ() all return a %HouseholderSequence. + * + * More precisely, the class %HouseholderSequence represents an \f$ n \times n \f$ matrix \f$ H \f$ of the + * form \f$ H = \prod_{i=0}^{n-1} H_i \f$ where the i-th Householder reflection is \f$ H_i = I - h_i v_i + * v_i^* \f$. The i-th Householder coefficient \f$ h_i \f$ is a scalar and the i-th Householder vector \f$ + * v_i \f$ is a vector of the form + * \f[ + * v_i = [\underbrace{0, \ldots, 0}_{i-1\mbox{ zeros}}, 1, \underbrace{*, \ldots,*}_{n-i\mbox{ arbitrary entries}} ]. + * \f] + * The last \f$ n-i \f$ entries of \f$ v_i \f$ are called the essential part of the Householder vector. + * + * Typical usages are listed below, where H is a HouseholderSequence: + * \code + * A.applyOnTheRight(H); // A = A * H + * A.applyOnTheLeft(H); // A = H * A + * A.applyOnTheRight(H.adjoint()); // A = A * H^* + * A.applyOnTheLeft(H.adjoint()); // A = H^* * A + * MatrixXd Q = H; // conversion to a dense matrix + * \endcode + * In addition to the adjoint, you can also apply the inverse (=adjoint), the transpose, and the conjugate operators. + * + * See the documentation for HouseholderSequence(const VectorsType&, const CoeffsType&) for an example. + * + * \sa MatrixBase::applyOnTheLeft(), MatrixBase::applyOnTheRight() + */ namespace internal { -template -struct traits > -{ - typedef typename VectorsType::Scalar Scalar; - typedef typename VectorsType::StorageIndex StorageIndex; - typedef typename VectorsType::StorageKind StorageKind; - enum { - RowsAtCompileTime = Side==OnTheLeft ? traits::RowsAtCompileTime - : traits::ColsAtCompileTime, - ColsAtCompileTime = RowsAtCompileTime, - MaxRowsAtCompileTime = Side==OnTheLeft ? traits::MaxRowsAtCompileTime - : traits::MaxColsAtCompileTime, - MaxColsAtCompileTime = MaxRowsAtCompileTime, - Flags = 0 + template + struct traits> + { + typedef typename VectorsType::Scalar Scalar; + typedef typename VectorsType::StorageIndex StorageIndex; + typedef typename VectorsType::StorageKind StorageKind; + enum { + RowsAtCompileTime = + Side == OnTheLeft ? traits::RowsAtCompileTime : traits::ColsAtCompileTime, + ColsAtCompileTime = RowsAtCompileTime, + MaxRowsAtCompileTime = + Side == OnTheLeft ? traits::MaxRowsAtCompileTime : traits::MaxColsAtCompileTime, + MaxColsAtCompileTime = MaxRowsAtCompileTime, + Flags = 0 + }; }; -}; - -struct HouseholderSequenceShape {}; - -template -struct evaluator_traits > - : public evaluator_traits_base > -{ - typedef HouseholderSequenceShape Shape; -}; -template -struct hseq_side_dependent_impl -{ - typedef Block EssentialVectorType; - typedef HouseholderSequence HouseholderSequenceType; - static inline const EssentialVectorType essentialVector(const HouseholderSequenceType& h, Index k) + struct HouseholderSequenceShape { - Index start = k+1+h.m_shift; - return Block(h.m_vectors, start, k, h.rows()-start, 1); - } -}; + }; -template -struct hseq_side_dependent_impl -{ - typedef Transpose > EssentialVectorType; - typedef HouseholderSequence HouseholderSequenceType; - static inline const EssentialVectorType essentialVector(const HouseholderSequenceType& h, Index k) + template + struct evaluator_traits> + : public evaluator_traits_base> { - Index start = k+1+h.m_shift; - return Block(h.m_vectors, k, start, 1, h.rows()-start).transpose(); - } -}; - -template struct matrix_type_times_scalar_type -{ - typedef typename ScalarBinaryOpTraits::ReturnType - ResultScalar; - typedef Matrix Type; -}; - -} // end namespace internal - -template class HouseholderSequence - : public EigenBase > -{ - typedef typename internal::hseq_side_dependent_impl::EssentialVectorType EssentialVectorType; - - public: - enum { - RowsAtCompileTime = internal::traits::RowsAtCompileTime, - ColsAtCompileTime = internal::traits::ColsAtCompileTime, - MaxRowsAtCompileTime = internal::traits::MaxRowsAtCompileTime, - MaxColsAtCompileTime = internal::traits::MaxColsAtCompileTime - }; - typedef typename internal::traits::Scalar Scalar; - - typedef HouseholderSequence< - typename internal::conditional::IsComplex, - typename internal::remove_all::type, - VectorsType>::type, - typename internal::conditional::IsComplex, - typename internal::remove_all::type, - CoeffsType>::type, - Side - > ConjugateReturnType; - - /** \brief Constructor. - * \param[in] v %Matrix containing the essential parts of the Householder vectors - * \param[in] h Vector containing the Householder coefficients - * - * Constructs the Householder sequence with coefficients given by \p h and vectors given by \p v. The - * i-th Householder coefficient \f$ h_i \f$ is given by \p h(i) and the essential part of the i-th - * Householder vector \f$ v_i \f$ is given by \p v(k,i) with \p k > \p i (the subdiagonal part of the - * i-th column). If \p v has fewer columns than rows, then the Householder sequence contains as many - * Householder reflections as there are columns. - * - * \note The %HouseholderSequence object stores \p v and \p h by reference. - * - * Example: \include HouseholderSequence_HouseholderSequence.cpp - * Output: \verbinclude HouseholderSequence_HouseholderSequence.out - * - * \sa setLength(), setShift() - */ - HouseholderSequence(const VectorsType& v, const CoeffsType& h) - : m_vectors(v), m_coeffs(h), m_trans(false), m_length(v.diagonalSize()), - m_shift(0) - { - } + typedef HouseholderSequenceShape Shape; + }; - /** \brief Copy constructor. */ - HouseholderSequence(const HouseholderSequence& other) - : m_vectors(other.m_vectors), - m_coeffs(other.m_coeffs), - m_trans(other.m_trans), - m_length(other.m_length), - m_shift(other.m_shift) + template struct hseq_side_dependent_impl + { + typedef Block EssentialVectorType; + typedef HouseholderSequence HouseholderSequenceType; + static inline const EssentialVectorType essentialVector(const HouseholderSequenceType &h, Index k) { + Index start = k + 1 + h.m_shift; + return Block(h.m_vectors, start, k, h.rows() - start, 1); } + }; - /** \brief Number of rows of transformation viewed as a matrix. - * \returns Number of rows - * \details This equals the dimension of the space that the transformation acts on. - */ - Index rows() const { return Side==OnTheLeft ? m_vectors.rows() : m_vectors.cols(); } - - /** \brief Number of columns of transformation viewed as a matrix. - * \returns Number of columns - * \details This equals the dimension of the space that the transformation acts on. - */ - Index cols() const { return rows(); } - - /** \brief Essential part of a Householder vector. - * \param[in] k Index of Householder reflection - * \returns Vector containing non-trivial entries of k-th Householder vector - * - * This function returns the essential part of the Householder vector \f$ v_i \f$. This is a vector of - * length \f$ n-i \f$ containing the last \f$ n-i \f$ entries of the vector - * \f[ - * v_i = [\underbrace{0, \ldots, 0}_{i-1\mbox{ zeros}}, 1, \underbrace{*, \ldots,*}_{n-i\mbox{ arbitrary entries}} ]. - * \f] - * The index \f$ i \f$ equals \p k + shift(), corresponding to the k-th column of the matrix \p v - * passed to the constructor. - * - * \sa setShift(), shift() - */ - const EssentialVectorType essentialVector(Index k) const + template + struct hseq_side_dependent_impl + { + typedef Transpose> EssentialVectorType; + typedef HouseholderSequence HouseholderSequenceType; + static inline const EssentialVectorType essentialVector(const HouseholderSequenceType &h, Index k) { - eigen_assert(k >= 0 && k < m_length); - return internal::hseq_side_dependent_impl::essentialVector(*this, k); + Index start = k + 1 + h.m_shift; + return Block(h.m_vectors, k, start, 1, h.rows() - start).transpose(); } + }; - /** \brief %Transpose of the Householder sequence. */ - HouseholderSequence transpose() const - { - return HouseholderSequence(*this).setTrans(!m_trans); - } + template struct matrix_type_times_scalar_type + { + typedef typename ScalarBinaryOpTraits::ReturnType ResultScalar; + typedef Matrix + Type; + }; - /** \brief Complex conjugate of the Householder sequence. */ - ConjugateReturnType conjugate() const - { - return ConjugateReturnType(m_vectors.conjugate(), m_coeffs.conjugate()) - .setTrans(m_trans) - .setLength(m_length) - .setShift(m_shift); - } +}// end namespace internal - /** \brief Adjoint (conjugate transpose) of the Householder sequence. */ - ConjugateReturnType adjoint() const - { - return conjugate().setTrans(!m_trans); - } +template +class HouseholderSequence : public EigenBase> +{ + typedef + typename internal::hseq_side_dependent_impl::EssentialVectorType EssentialVectorType; - /** \brief Inverse of the Householder sequence (equals the adjoint). */ - ConjugateReturnType inverse() const { return adjoint(); } +public: + enum { + RowsAtCompileTime = internal::traits::RowsAtCompileTime, + ColsAtCompileTime = internal::traits::ColsAtCompileTime, + MaxRowsAtCompileTime = internal::traits::MaxRowsAtCompileTime, + MaxColsAtCompileTime = internal::traits::MaxColsAtCompileTime + }; + typedef typename internal::traits::Scalar Scalar; + + typedef HouseholderSequence::IsComplex, + typename internal::remove_all::type, + VectorsType>::type, + typename internal::conditional::IsComplex, + typename internal::remove_all::type, + CoeffsType>::type, + Side> + ConjugateReturnType; + + /** \brief Constructor. + * \param[in] v %Matrix containing the essential parts of the Householder vectors + * \param[in] h Vector containing the Householder coefficients + * + * Constructs the Householder sequence with coefficients given by \p h and vectors given by \p v. The + * i-th Householder coefficient \f$ h_i \f$ is given by \p h(i) and the essential part of the i-th + * Householder vector \f$ v_i \f$ is given by \p v(k,i) with \p k > \p i (the subdiagonal part of the + * i-th column). If \p v has fewer columns than rows, then the Householder sequence contains as many + * Householder reflections as there are columns. + * + * \note The %HouseholderSequence object stores \p v and \p h by reference. + * + * Example: \include HouseholderSequence_HouseholderSequence.cpp + * Output: \verbinclude HouseholderSequence_HouseholderSequence.out + * + * \sa setLength(), setShift() + */ + HouseholderSequence(const VectorsType &v, const CoeffsType &h) + : m_vectors(v), m_coeffs(h), m_trans(false), m_length(v.diagonalSize()), m_shift(0) + {} + + /** \brief Copy constructor. */ + HouseholderSequence(const HouseholderSequence &other) + : m_vectors(other.m_vectors), m_coeffs(other.m_coeffs), m_trans(other.m_trans), m_length(other.m_length), + m_shift(other.m_shift) + {} + + /** \brief Number of rows of transformation viewed as a matrix. + * \returns Number of rows + * \details This equals the dimension of the space that the transformation acts on. + */ + Index rows() const { return Side == OnTheLeft ? m_vectors.rows() : m_vectors.cols(); } + + /** \brief Number of columns of transformation viewed as a matrix. + * \returns Number of columns + * \details This equals the dimension of the space that the transformation acts on. + */ + Index cols() const { return rows(); } + + /** \brief Essential part of a Householder vector. + * \param[in] k Index of Householder reflection + * \returns Vector containing non-trivial entries of k-th Householder vector + * + * This function returns the essential part of the Householder vector \f$ v_i \f$. This is a vector of + * length \f$ n-i \f$ containing the last \f$ n-i \f$ entries of the vector + * \f[ + * v_i = [\underbrace{0, \ldots, 0}_{i-1\mbox{ zeros}}, 1, \underbrace{*, \ldots,*}_{n-i\mbox{ arbitrary entries}} ]. + * \f] + * The index \f$ i \f$ equals \p k + shift(), corresponding to the k-th column of the matrix \p v + * passed to the constructor. + * + * \sa setShift(), shift() + */ + const EssentialVectorType essentialVector(Index k) const + { + eigen_assert(k >= 0 && k < m_length); + return internal::hseq_side_dependent_impl::essentialVector(*this, k); + } - /** \internal */ - template inline void evalTo(DestType& dst) const - { - Matrix workspace(rows()); - evalTo(dst, workspace); - } + /** \brief %Transpose of the Householder sequence. */ + HouseholderSequence transpose() const { return HouseholderSequence(*this).setTrans(!m_trans); } - /** \internal */ - template - void evalTo(Dest& dst, Workspace& workspace) const - { - workspace.resize(rows()); - Index vecs = m_length; - if(internal::is_same_dense(dst,m_vectors)) - { - // in-place - dst.diagonal().setOnes(); - dst.template triangularView().setZero(); - for(Index k = vecs-1; k >= 0; --k) - { - Index cornerSize = rows() - k - m_shift; - if(m_trans) - dst.bottomRightCorner(cornerSize, cornerSize) - .applyHouseholderOnTheRight(essentialVector(k), m_coeffs.coeff(k), workspace.data()); - else - dst.bottomRightCorner(cornerSize, cornerSize) - .applyHouseholderOnTheLeft(essentialVector(k), m_coeffs.coeff(k), workspace.data()); - - // clear the off diagonal vector - dst.col(k).tail(rows()-k-1).setZero(); - } - // clear the remaining columns if needed - for(Index k = 0; k= 0; --k) - { - Index cornerSize = rows() - k - m_shift; - if(m_trans) - dst.bottomRightCorner(cornerSize, cornerSize) - .applyHouseholderOnTheRight(essentialVector(k), m_coeffs.coeff(k), &workspace.coeffRef(0)); - else - dst.bottomRightCorner(cornerSize, cornerSize) - .applyHouseholderOnTheLeft(essentialVector(k), m_coeffs.coeff(k), &workspace.coeffRef(0)); - } - } - } + /** \brief Complex conjugate of the Householder sequence. */ + ConjugateReturnType conjugate() const + { + return ConjugateReturnType(m_vectors.conjugate(), m_coeffs.conjugate()) + .setTrans(m_trans) + .setLength(m_length) + .setShift(m_shift); + } - /** \internal */ - template inline void applyThisOnTheRight(Dest& dst) const - { - Matrix workspace(dst.rows()); - applyThisOnTheRight(dst, workspace); - } + /** \brief Adjoint (conjugate transpose) of the Householder sequence. */ + ConjugateReturnType adjoint() const { return conjugate().setTrans(!m_trans); } - /** \internal */ - template - inline void applyThisOnTheRight(Dest& dst, Workspace& workspace) const - { - workspace.resize(dst.rows()); - for(Index k = 0; k < m_length; ++k) - { - Index actual_k = m_trans ? m_length-k-1 : k; - dst.rightCols(rows()-m_shift-actual_k) - .applyHouseholderOnTheRight(essentialVector(actual_k), m_coeffs.coeff(actual_k), workspace.data()); - } - } + /** \brief Inverse of the Householder sequence (equals the adjoint). */ + ConjugateReturnType inverse() const { return adjoint(); } - /** \internal */ - template inline void applyThisOnTheLeft(Dest& dst) const - { - Matrix workspace; - applyThisOnTheLeft(dst, workspace); - } + /** \internal */ + template inline void evalTo(DestType &dst) const + { + Matrix workspace( + rows()); + evalTo(dst, workspace); + } - /** \internal */ - template - inline void applyThisOnTheLeft(Dest& dst, Workspace& workspace) const - { - const Index BlockSize = 48; - // if the entries are large enough, then apply the reflectors by block - if(m_length>=BlockSize && dst.cols()>1) - { - for(Index i = 0; i < m_length; i+=BlockSize) - { - Index end = m_trans ? (std::min)(m_length,i+BlockSize) : m_length-i; - Index k = m_trans ? i : (std::max)(Index(0),end-BlockSize); - Index bs = end-k; - Index start = k + m_shift; - - typedef Block::type,Dynamic,Dynamic> SubVectorsType; - SubVectorsType sub_vecs1(m_vectors.const_cast_derived(), Side==OnTheRight ? k : start, - Side==OnTheRight ? start : k, - Side==OnTheRight ? bs : m_vectors.rows()-start, - Side==OnTheRight ? m_vectors.cols()-start : bs); - typename internal::conditional, SubVectorsType&>::type sub_vecs(sub_vecs1); - Block sub_dst(dst,dst.rows()-rows()+m_shift+k,0, rows()-m_shift-k,dst.cols()); - apply_block_householder_on_the_left(sub_dst, sub_vecs, m_coeffs.segment(k, bs), !m_trans); - } + /** \internal */ + template void evalTo(Dest &dst, Workspace &workspace) const + { + workspace.resize(rows()); + Index vecs = m_length; + if (internal::is_same_dense(dst, m_vectors)) { + // in-place + dst.diagonal().setOnes(); + dst.template triangularView().setZero(); + for (Index k = vecs - 1; k >= 0; --k) { + Index cornerSize = rows() - k - m_shift; + if (m_trans) + dst.bottomRightCorner(cornerSize, cornerSize) + .applyHouseholderOnTheRight(essentialVector(k), m_coeffs.coeff(k), workspace.data()); + else + dst.bottomRightCorner(cornerSize, cornerSize) + .applyHouseholderOnTheLeft(essentialVector(k), m_coeffs.coeff(k), workspace.data()); + + // clear the off diagonal vector + dst.col(k).tail(rows() - k - 1).setZero(); } - else - { - workspace.resize(dst.cols()); - for(Index k = 0; k < m_length; ++k) - { - Index actual_k = m_trans ? k : m_length-k-1; - dst.bottomRows(rows()-m_shift-actual_k) - .applyHouseholderOnTheLeft(essentialVector(actual_k), m_coeffs.coeff(actual_k), workspace.data()); - } + // clear the remaining columns if needed + for (Index k = 0; k < cols() - vecs; ++k) dst.col(k).tail(rows() - k - 1).setZero(); + } else { + dst.setIdentity(rows(), rows()); + for (Index k = vecs - 1; k >= 0; --k) { + Index cornerSize = rows() - k - m_shift; + if (m_trans) + dst.bottomRightCorner(cornerSize, cornerSize) + .applyHouseholderOnTheRight(essentialVector(k), m_coeffs.coeff(k), &workspace.coeffRef(0)); + else + dst.bottomRightCorner(cornerSize, cornerSize) + .applyHouseholderOnTheLeft(essentialVector(k), m_coeffs.coeff(k), &workspace.coeffRef(0)); } } + } - /** \brief Computes the product of a Householder sequence with a matrix. - * \param[in] other %Matrix being multiplied. - * \returns Expression object representing the product. - * - * This function computes \f$ HM \f$ where \f$ H \f$ is the Householder sequence represented by \p *this - * and \f$ M \f$ is the matrix \p other. - */ - template - typename internal::matrix_type_times_scalar_type::Type operator*(const MatrixBase& other) const - { - typename internal::matrix_type_times_scalar_type::Type - res(other.template cast::ResultScalar>()); - applyThisOnTheLeft(res); - return res; - } + /** \internal */ + template inline void applyThisOnTheRight(Dest &dst) const + { + Matrix workspace(dst.rows()); + applyThisOnTheRight(dst, workspace); + } - template friend struct internal::hseq_side_dependent_impl; - - /** \brief Sets the length of the Householder sequence. - * \param [in] length New value for the length. - * - * By default, the length \f$ n \f$ of the Householder sequence \f$ H = H_0 H_1 \ldots H_{n-1} \f$ is set - * to the number of columns of the matrix \p v passed to the constructor, or the number of rows if that - * is smaller. After this function is called, the length equals \p length. - * - * \sa length() - */ - HouseholderSequence& setLength(Index length) - { - m_length = length; - return *this; + /** \internal */ + template inline void applyThisOnTheRight(Dest &dst, Workspace &workspace) const + { + workspace.resize(dst.rows()); + for (Index k = 0; k < m_length; ++k) { + Index actual_k = m_trans ? m_length - k - 1 : k; + dst.rightCols(rows() - m_shift - actual_k) + .applyHouseholderOnTheRight(essentialVector(actual_k), m_coeffs.coeff(actual_k), workspace.data()); } + } - /** \brief Sets the shift of the Householder sequence. - * \param [in] shift New value for the shift. - * - * By default, a %HouseholderSequence object represents \f$ H = H_0 H_1 \ldots H_{n-1} \f$ and the i-th - * column of the matrix \p v passed to the constructor corresponds to the i-th Householder - * reflection. After this function is called, the object represents \f$ H = H_{\mathrm{shift}} - * H_{\mathrm{shift}+1} \ldots H_{n-1} \f$ and the i-th column of \p v corresponds to the (shift+i)-th - * Householder reflection. - * - * \sa shift() - */ - HouseholderSequence& setShift(Index shift) - { - m_shift = shift; - return *this; + /** \internal */ + template inline void applyThisOnTheLeft(Dest &dst) const + { + Matrix workspace; + applyThisOnTheLeft(dst, workspace); + } + + /** \internal */ + template inline void applyThisOnTheLeft(Dest &dst, Workspace &workspace) const + { + const Index BlockSize = 48; + // if the entries are large enough, then apply the reflectors by block + if (m_length >= BlockSize && dst.cols() > 1) { + for (Index i = 0; i < m_length; i += BlockSize) { + Index end = m_trans ? (std::min)(m_length, i + BlockSize) : m_length - i; + Index k = m_trans ? i : (std::max)(Index(0), end - BlockSize); + Index bs = end - k; + Index start = k + m_shift; + + typedef Block::type, Dynamic, Dynamic> SubVectorsType; + SubVectorsType sub_vecs1(m_vectors.const_cast_derived(), + Side == OnTheRight ? k : start, + Side == OnTheRight ? start : k, + Side == OnTheRight ? bs : m_vectors.rows() - start, + Side == OnTheRight ? m_vectors.cols() - start : bs); + typename internal::conditional, SubVectorsType &>::type sub_vecs( + sub_vecs1); + Block sub_dst( + dst, dst.rows() - rows() + m_shift + k, 0, rows() - m_shift - k, dst.cols()); + apply_block_householder_on_the_left(sub_dst, sub_vecs, m_coeffs.segment(k, bs), !m_trans); + } + } else { + workspace.resize(dst.cols()); + for (Index k = 0; k < m_length; ++k) { + Index actual_k = m_trans ? k : m_length - k - 1; + dst.bottomRows(rows() - m_shift - actual_k) + .applyHouseholderOnTheLeft(essentialVector(actual_k), m_coeffs.coeff(actual_k), workspace.data()); + } } + } - Index length() const { return m_length; } /**< \brief Returns the length of the Householder sequence. */ - Index shift() const { return m_shift; } /**< \brief Returns the shift of the Householder sequence. */ + /** \brief Computes the product of a Householder sequence with a matrix. + * \param[in] other %Matrix being multiplied. + * \returns Expression object representing the product. + * + * This function computes \f$ HM \f$ where \f$ H \f$ is the Householder sequence represented by \p *this + * and \f$ M \f$ is the matrix \p other. + */ + template + typename internal::matrix_type_times_scalar_type::Type operator*( + const MatrixBase &other) const + { + typename internal::matrix_type_times_scalar_type::Type res( + other.template cast::ResultScalar>()); + applyThisOnTheLeft(res); + return res; + } - /* Necessary for .adjoint() and .conjugate() */ - template friend class HouseholderSequence; + template friend struct internal::hseq_side_dependent_impl; + + /** \brief Sets the length of the Householder sequence. + * \param [in] length New value for the length. + * + * By default, the length \f$ n \f$ of the Householder sequence \f$ H = H_0 H_1 \ldots H_{n-1} \f$ is set + * to the number of columns of the matrix \p v passed to the constructor, or the number of rows if that + * is smaller. After this function is called, the length equals \p length. + * + * \sa length() + */ + HouseholderSequence &setLength(Index length) + { + m_length = length; + return *this; + } - protected: + /** \brief Sets the shift of the Householder sequence. + * \param [in] shift New value for the shift. + * + * By default, a %HouseholderSequence object represents \f$ H = H_0 H_1 \ldots H_{n-1} \f$ and the i-th + * column of the matrix \p v passed to the constructor corresponds to the i-th Householder + * reflection. After this function is called, the object represents \f$ H = H_{\mathrm{shift}} + * H_{\mathrm{shift}+1} \ldots H_{n-1} \f$ and the i-th column of \p v corresponds to the (shift+i)-th + * Householder reflection. + * + * \sa shift() + */ + HouseholderSequence &setShift(Index shift) + { + m_shift = shift; + return *this; + } - /** \brief Sets the transpose flag. - * \param [in] trans New value of the transpose flag. - * - * By default, the transpose flag is not set. If the transpose flag is set, then this object represents - * \f$ H^T = H_{n-1}^T \ldots H_1^T H_0^T \f$ instead of \f$ H = H_0 H_1 \ldots H_{n-1} \f$. - * - * \sa trans() - */ - HouseholderSequence& setTrans(bool trans) - { - m_trans = trans; - return *this; - } + Index length() const { return m_length; } /**< \brief Returns the length of the Householder sequence. */ + Index shift() const { return m_shift; } /**< \brief Returns the shift of the Householder sequence. */ + + /* Necessary for .adjoint() and .conjugate() */ + template friend class HouseholderSequence; + +protected: + /** \brief Sets the transpose flag. + * \param [in] trans New value of the transpose flag. + * + * By default, the transpose flag is not set. If the transpose flag is set, then this object represents + * \f$ H^T = H_{n-1}^T \ldots H_1^T H_0^T \f$ instead of \f$ H = H_0 H_1 \ldots H_{n-1} \f$. + * + * \sa trans() + */ + HouseholderSequence &setTrans(bool trans) + { + m_trans = trans; + return *this; + } - bool trans() const { return m_trans; } /**< \brief Returns the transpose flag. */ + bool trans() const { return m_trans; } /**< \brief Returns the transpose flag. */ - typename VectorsType::Nested m_vectors; - typename CoeffsType::Nested m_coeffs; - bool m_trans; - Index m_length; - Index m_shift; + typename VectorsType::Nested m_vectors; + typename CoeffsType::Nested m_coeffs; + bool m_trans; + Index m_length; + Index m_shift; }; /** \brief Computes the product of a matrix with a Householder sequence. - * \param[in] other %Matrix being multiplied. - * \param[in] h %HouseholderSequence being multiplied. - * \returns Expression object representing the product. - * - * This function computes \f$ MH \f$ where \f$ M \f$ is the matrix \p other and \f$ H \f$ is the - * Householder sequence represented by \p h. - */ + * \param[in] other %Matrix being multiplied. + * \param[in] h %HouseholderSequence being multiplied. + * \returns Expression object representing the product. + * + * This function computes \f$ MH \f$ where \f$ M \f$ is the matrix \p other and \f$ H \f$ is the + * Householder sequence represented by \p h. + */ template -typename internal::matrix_type_times_scalar_type::Type operator*(const MatrixBase& other, const HouseholderSequence& h) +typename internal::matrix_type_times_scalar_type::Type + operator*(const MatrixBase &other, const HouseholderSequence &h) { - typename internal::matrix_type_times_scalar_type::Type - res(other.template cast::ResultScalar>()); + typename internal::matrix_type_times_scalar_type::Type res( + other.template cast< + typename internal::matrix_type_times_scalar_type::ResultScalar>()); h.applyThisOnTheRight(res); return res; } /** \ingroup Householder_Module \householder_module - * \brief Convenience function for constructing a Householder sequence. - * \returns A HouseholderSequence constructed from the specified arguments. - */ + * \brief Convenience function for constructing a Householder sequence. + * \returns A HouseholderSequence constructed from the specified arguments. + */ template -HouseholderSequence householderSequence(const VectorsType& v, const CoeffsType& h) +HouseholderSequence householderSequence(const VectorsType &v, const CoeffsType &h) { - return HouseholderSequence(v, h); + return HouseholderSequence(v, h); } /** \ingroup Householder_Module \householder_module - * \brief Convenience function for constructing a Householder sequence. - * \returns A HouseholderSequence constructed from the specified arguments. - * \details This function differs from householderSequence() in that the template argument \p OnTheSide of - * the constructed HouseholderSequence is set to OnTheRight, instead of the default OnTheLeft. - */ + * \brief Convenience function for constructing a Householder sequence. + * \returns A HouseholderSequence constructed from the specified arguments. + * \details This function differs from householderSequence() in that the template argument \p OnTheSide of + * the constructed HouseholderSequence is set to OnTheRight, instead of the default OnTheLeft. + */ template -HouseholderSequence rightHouseholderSequence(const VectorsType& v, const CoeffsType& h) +HouseholderSequence rightHouseholderSequence(const VectorsType &v, + const CoeffsType &h) { - return HouseholderSequence(v, h); + return HouseholderSequence(v, h); } -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_HOUSEHOLDER_SEQUENCE_H +#endif// EIGEN_HOUSEHOLDER_SEQUENCE_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/IterativeLinearSolvers/BasicPreconditioners.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/IterativeLinearSolvers/BasicPreconditioners.h index f66c846e..3f48cee2 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/IterativeLinearSolvers/BasicPreconditioners.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/IterativeLinearSolvers/BasicPreconditioners.h @@ -10,7 +10,7 @@ #ifndef EIGEN_BASIC_PRECONDITIONERS_H #define EIGEN_BASIC_PRECONDITIONERS_H -namespace Eigen { +namespace Eigen { /** \ingroup IterativeLinearSolvers_Module * \brief A preconditioner based on the digonal entries @@ -32,79 +32,63 @@ namespace Eigen { * * \sa class LeastSquareDiagonalPreconditioner, class ConjugateGradient */ -template -class DiagonalPreconditioner +template class DiagonalPreconditioner { - typedef _Scalar Scalar; - typedef Matrix Vector; - public: - typedef typename Vector::StorageIndex StorageIndex; - enum { - ColsAtCompileTime = Dynamic, - MaxColsAtCompileTime = Dynamic - }; - - DiagonalPreconditioner() : m_isInitialized(false) {} - - template - explicit DiagonalPreconditioner(const MatType& mat) : m_invdiag(mat.cols()) - { - compute(mat); - } + typedef _Scalar Scalar; + typedef Matrix Vector; - Index rows() const { return m_invdiag.size(); } - Index cols() const { return m_invdiag.size(); } - - template - DiagonalPreconditioner& analyzePattern(const MatType& ) - { - return *this; - } - - template - DiagonalPreconditioner& factorize(const MatType& mat) - { - m_invdiag.resize(mat.cols()); - for(int j=0; j - DiagonalPreconditioner& compute(const MatType& mat) - { - return factorize(mat); - } +public: + typedef typename Vector::StorageIndex StorageIndex; + enum { ColsAtCompileTime = Dynamic, MaxColsAtCompileTime = Dynamic }; - /** \internal */ - template - void _solve_impl(const Rhs& b, Dest& x) const - { - x = m_invdiag.array() * b.array() ; - } + DiagonalPreconditioner() : m_isInitialized(false) {} - template inline const Solve - solve(const MatrixBase& b) const - { - eigen_assert(m_isInitialized && "DiagonalPreconditioner is not initialized."); - eigen_assert(m_invdiag.size()==b.rows() - && "DiagonalPreconditioner::solve(): invalid number of rows of the right hand side matrix b"); - return Solve(*this, b.derived()); - } - - ComputationInfo info() { return Success; } + template explicit DiagonalPreconditioner(const MatType &mat) : m_invdiag(mat.cols()) + { + compute(mat); + } - protected: - Vector m_invdiag; - bool m_isInitialized; + Index rows() const { return m_invdiag.size(); } + Index cols() const { return m_invdiag.size(); } + + template DiagonalPreconditioner &analyzePattern(const MatType &) { return *this; } + + template DiagonalPreconditioner &factorize(const MatType &mat) + { + m_invdiag.resize(mat.cols()); + for (int j = 0; j < mat.outerSize(); ++j) { + typename MatType::InnerIterator it(mat, j); + while (it && it.index() != j) ++it; + if (it && it.index() == j && it.value() != Scalar(0)) + m_invdiag(j) = Scalar(1) / it.value(); + else + m_invdiag(j) = Scalar(1); + } + m_isInitialized = true; + return *this; + } + + template DiagonalPreconditioner &compute(const MatType &mat) { return factorize(mat); } + + /** \internal */ + template void _solve_impl(const Rhs &b, Dest &x) const + { + x = m_invdiag.array() * b.array(); + } + + template inline const Solve solve(const MatrixBase &b) const + { + eigen_assert(m_isInitialized && "DiagonalPreconditioner is not initialized."); + eigen_assert(m_invdiag.size() == b.rows() + && "DiagonalPreconditioner::solve(): invalid number of rows of the right hand side matrix b"); + return Solve(*this, b.derived()); + } + + ComputationInfo info() { return Success; } + +protected: + Vector m_invdiag; + bool m_isInitialized; }; /** \ingroup IterativeLinearSolvers_Module @@ -121,106 +105,79 @@ class DiagonalPreconditioner * \implsparsesolverconcept * * The diagonal entries are pre-inverted and stored into a dense vector. - * + * * \sa class LeastSquaresConjugateGradient, class DiagonalPreconditioner */ -template -class LeastSquareDiagonalPreconditioner : public DiagonalPreconditioner<_Scalar> +template class LeastSquareDiagonalPreconditioner : public DiagonalPreconditioner<_Scalar> { - typedef _Scalar Scalar; - typedef typename NumTraits::Real RealScalar; - typedef DiagonalPreconditioner<_Scalar> Base; - using Base::m_invdiag; - public: - - LeastSquareDiagonalPreconditioner() : Base() {} - - template - explicit LeastSquareDiagonalPreconditioner(const MatType& mat) : Base() - { - compute(mat); - } - - template - LeastSquareDiagonalPreconditioner& analyzePattern(const MatType& ) - { - return *this; - } - - template - LeastSquareDiagonalPreconditioner& factorize(const MatType& mat) - { - // Compute the inverse squared-norm of each column of mat - m_invdiag.resize(mat.cols()); - if(MatType::IsRowMajor) - { - m_invdiag.setZero(); - for(Index j=0; jRealScalar(0)) - m_invdiag(j) = RealScalar(1)/numext::real(m_invdiag(j)); + typedef _Scalar Scalar; + typedef typename NumTraits::Real RealScalar; + typedef DiagonalPreconditioner<_Scalar> Base; + using Base::m_invdiag; + +public: + LeastSquareDiagonalPreconditioner() : Base() {} + + template explicit LeastSquareDiagonalPreconditioner(const MatType &mat) : Base() { compute(mat); } + + template LeastSquareDiagonalPreconditioner &analyzePattern(const MatType &) { return *this; } + + template LeastSquareDiagonalPreconditioner &factorize(const MatType &mat) + { + // Compute the inverse squared-norm of each column of mat + m_invdiag.resize(mat.cols()); + if (MatType::IsRowMajor) { + m_invdiag.setZero(); + for (Index j = 0; j < mat.outerSize(); ++j) { + for (typename MatType::InnerIterator it(mat, j); it; ++it) m_invdiag(it.index()) += numext::abs2(it.value()); } - else - { - for(Index j=0; jRealScalar(0)) - m_invdiag(j) = RealScalar(1)/sum; - else - m_invdiag(j) = RealScalar(1); - } + for (Index j = 0; j < mat.cols(); ++j) + if (numext::real(m_invdiag(j)) > RealScalar(0)) m_invdiag(j) = RealScalar(1) / numext::real(m_invdiag(j)); + } else { + for (Index j = 0; j < mat.outerSize(); ++j) { + RealScalar sum = mat.col(j).squaredNorm(); + if (sum > RealScalar(0)) + m_invdiag(j) = RealScalar(1) / sum; + else + m_invdiag(j) = RealScalar(1); } - Base::m_isInitialized = true; - return *this; } - - template - LeastSquareDiagonalPreconditioner& compute(const MatType& mat) - { - return factorize(mat); - } - - ComputationInfo info() { return Success; } + Base::m_isInitialized = true; + return *this; + } + + template LeastSquareDiagonalPreconditioner &compute(const MatType &mat) { return factorize(mat); } - protected: + ComputationInfo info() { return Success; } + +protected: }; /** \ingroup IterativeLinearSolvers_Module - * \brief A naive preconditioner which approximates any matrix as the identity matrix - * - * \implsparsesolverconcept - * - * \sa class DiagonalPreconditioner - */ + * \brief A naive preconditioner which approximates any matrix as the identity matrix + * + * \implsparsesolverconcept + * + * \sa class DiagonalPreconditioner + */ class IdentityPreconditioner { - public: - - IdentityPreconditioner() {} - - template - explicit IdentityPreconditioner(const MatrixType& ) {} - - template - IdentityPreconditioner& analyzePattern(const MatrixType& ) { return *this; } - - template - IdentityPreconditioner& factorize(const MatrixType& ) { return *this; } - - template - IdentityPreconditioner& compute(const MatrixType& ) { return *this; } - - template - inline const Rhs& solve(const Rhs& b) const { return b; } - - ComputationInfo info() { return Success; } +public: + IdentityPreconditioner() {} + + template explicit IdentityPreconditioner(const MatrixType &) {} + + template IdentityPreconditioner &analyzePattern(const MatrixType &) { return *this; } + + template IdentityPreconditioner &factorize(const MatrixType &) { return *this; } + + template IdentityPreconditioner &compute(const MatrixType &) { return *this; } + + template inline const Rhs &solve(const Rhs &b) const { return b; } + + ComputationInfo info() { return Success; } }; -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_BASIC_PRECONDITIONERS_H +#endif// EIGEN_BASIC_PRECONDITIONERS_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/IterativeLinearSolvers/BiCGSTAB.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/IterativeLinearSolvers/BiCGSTAB.h index 454f4681..173236fb 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/IterativeLinearSolvers/BiCGSTAB.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/IterativeLinearSolvers/BiCGSTAB.h @@ -11,151 +11,148 @@ #ifndef EIGEN_BICGSTAB_H #define EIGEN_BICGSTAB_H -namespace Eigen { +namespace Eigen { namespace internal { -/** \internal Low-level bi conjugate gradient stabilized algorithm - * \param mat The matrix A - * \param rhs The right hand side vector b - * \param x On input and initial solution, on output the computed solution. - * \param precond A preconditioner being able to efficiently solve for an - * approximation of Ax=b (regardless of b) - * \param iters On input the max number of iteration, on output the number of performed iterations. - * \param tol_error On input the tolerance error, on output an estimation of the relative error. - * \return false in the case of numerical issue, for example a break down of BiCGSTAB. - */ -template -bool bicgstab(const MatrixType& mat, const Rhs& rhs, Dest& x, - const Preconditioner& precond, Index& iters, - typename Dest::RealScalar& tol_error) -{ - using std::sqrt; - using std::abs; - typedef typename Dest::RealScalar RealScalar; - typedef typename Dest::Scalar Scalar; - typedef Matrix VectorType; - RealScalar tol = tol_error; - Index maxIters = iters; - - Index n = mat.cols(); - VectorType r = rhs - mat * x; - VectorType r0 = r; - - RealScalar r0_sqnorm = r0.squaredNorm(); - RealScalar rhs_sqnorm = rhs.squaredNorm(); - if(rhs_sqnorm == 0) - { - x.setZero(); - return true; - } - Scalar rho = 1; - Scalar alpha = 1; - Scalar w = 1; - - VectorType v = VectorType::Zero(n), p = VectorType::Zero(n); - VectorType y(n), z(n); - VectorType kt(n), ks(n); - - VectorType s(n), t(n); - - RealScalar tol2 = tol*tol*rhs_sqnorm; - RealScalar eps2 = NumTraits::epsilon()*NumTraits::epsilon(); - Index i = 0; - Index restarts = 0; - - while ( r.squaredNorm() > tol2 && i + bool bicgstab(const MatrixType &mat, + const Rhs &rhs, + Dest &x, + const Preconditioner &precond, + Index &iters, + typename Dest::RealScalar &tol_error) { - Scalar rho_old = rho; - - rho = r0.dot(r); - if (abs(rho) < eps2*r0_sqnorm) - { - // The new residual vector became too orthogonal to the arbitrarily chosen direction r0 - // Let's restart with a new r0: - r = rhs - mat * x; - r0 = r; - rho = r0_sqnorm = r.squaredNorm(); - if(restarts++ == 0) - i = 0; + using std::sqrt; + using std::abs; + typedef typename Dest::RealScalar RealScalar; + typedef typename Dest::Scalar Scalar; + typedef Matrix VectorType; + RealScalar tol = tol_error; + Index maxIters = iters; + + Index n = mat.cols(); + VectorType r = rhs - mat * x; + VectorType r0 = r; + + RealScalar r0_sqnorm = r0.squaredNorm(); + RealScalar rhs_sqnorm = rhs.squaredNorm(); + if (rhs_sqnorm == 0) { + x.setZero(); + return true; } - Scalar beta = (rho/rho_old) * (alpha / w); - p = r + beta * (p - w * v); - - y = precond.solve(p); - - v.noalias() = mat * y; - - alpha = rho / r0.dot(v); - s = r - alpha * v; - - z = precond.solve(s); - t.noalias() = mat * z; - - RealScalar tmp = t.squaredNorm(); - if(tmp>RealScalar(0)) - w = t.dot(s) / tmp; - else - w = Scalar(0); - x += alpha * y + w * z; - r = s - w * t; - ++i; + Scalar rho = 1; + Scalar alpha = 1; + Scalar w = 1; + + VectorType v = VectorType::Zero(n), p = VectorType::Zero(n); + VectorType y(n), z(n); + VectorType kt(n), ks(n); + + VectorType s(n), t(n); + + RealScalar tol2 = tol * tol * rhs_sqnorm; + RealScalar eps2 = NumTraits::epsilon() * NumTraits::epsilon(); + Index i = 0; + Index restarts = 0; + + while (r.squaredNorm() > tol2 && i < maxIters) { + Scalar rho_old = rho; + + rho = r0.dot(r); + if (abs(rho) < eps2 * r0_sqnorm) { + // The new residual vector became too orthogonal to the arbitrarily chosen direction r0 + // Let's restart with a new r0: + r = rhs - mat * x; + r0 = r; + rho = r0_sqnorm = r.squaredNorm(); + if (restarts++ == 0) i = 0; + } + Scalar beta = (rho / rho_old) * (alpha / w); + p = r + beta * (p - w * v); + + y = precond.solve(p); + + v.noalias() = mat * y; + + alpha = rho / r0.dot(v); + s = r - alpha * v; + + z = precond.solve(s); + t.noalias() = mat * z; + + RealScalar tmp = t.squaredNorm(); + if (tmp > RealScalar(0)) + w = t.dot(s) / tmp; + else + w = Scalar(0); + x += alpha * y + w * z; + r = s - w * t; + ++i; + } + tol_error = sqrt(r.squaredNorm() / rhs_sqnorm); + iters = i; + return true; } - tol_error = sqrt(r.squaredNorm()/rhs_sqnorm); - iters = i; - return true; -} -} +}// namespace internal -template< typename _MatrixType, - typename _Preconditioner = DiagonalPreconditioner > +template> class BiCGSTAB; namespace internal { -template< typename _MatrixType, typename _Preconditioner> -struct traits > -{ - typedef _MatrixType MatrixType; - typedef _Preconditioner Preconditioner; -}; + template struct traits> + { + typedef _MatrixType MatrixType; + typedef _Preconditioner Preconditioner; + }; -} +}// namespace internal /** \ingroup IterativeLinearSolvers_Module - * \brief A bi conjugate gradient stabilized solver for sparse square problems - * - * This class allows to solve for A.x = b sparse linear problems using a bi conjugate gradient - * stabilized algorithm. The vectors x and b can be either dense or sparse. - * - * \tparam _MatrixType the type of the sparse matrix A, can be a dense or a sparse matrix. - * \tparam _Preconditioner the type of the preconditioner. Default is DiagonalPreconditioner - * - * \implsparsesolverconcept - * - * The maximal number of iterations and tolerance value can be controlled via the setMaxIterations() - * and setTolerance() methods. The defaults are the size of the problem for the maximal number of iterations - * and NumTraits::epsilon() for the tolerance. - * - * The tolerance corresponds to the relative residual error: |Ax-b|/|b| - * - * \b Performance: when using sparse matrices, best performance is achied for a row-major sparse matrix format. - * Moreover, in this case multi-threading can be exploited if the user code is compiled with OpenMP enabled. - * See \ref TopicMultiThreading for details. - * - * This class can be used as the direct solver classes. Here is a typical usage example: - * \include BiCGSTAB_simple.cpp - * - * By default the iterations start with x=0 as an initial guess of the solution. - * One can control the start using the solveWithGuess() method. - * - * BiCGSTAB can also be used in a matrix-free context, see the following \link MatrixfreeSolverExample example \endlink. - * - * \sa class SimplicialCholesky, DiagonalPreconditioner, IdentityPreconditioner - */ -template< typename _MatrixType, typename _Preconditioner> -class BiCGSTAB : public IterativeSolverBase > + * \brief A bi conjugate gradient stabilized solver for sparse square problems + * + * This class allows to solve for A.x = b sparse linear problems using a bi conjugate gradient + * stabilized algorithm. The vectors x and b can be either dense or sparse. + * + * \tparam _MatrixType the type of the sparse matrix A, can be a dense or a sparse matrix. + * \tparam _Preconditioner the type of the preconditioner. Default is DiagonalPreconditioner + * + * \implsparsesolverconcept + * + * The maximal number of iterations and tolerance value can be controlled via the setMaxIterations() + * and setTolerance() methods. The defaults are the size of the problem for the maximal number of iterations + * and NumTraits::epsilon() for the tolerance. + * + * The tolerance corresponds to the relative residual error: |Ax-b|/|b| + * + * \b Performance: when using sparse matrices, best performance is achied for a row-major sparse matrix format. + * Moreover, in this case multi-threading can be exploited if the user code is compiled with OpenMP enabled. + * See \ref TopicMultiThreading for details. + * + * This class can be used as the direct solver classes. Here is a typical usage example: + * \include BiCGSTAB_simple.cpp + * + * By default the iterations start with x=0 as an initial guess of the solution. + * One can control the start using the solveWithGuess() method. + * + * BiCGSTAB can also be used in a matrix-free context, see the following \link MatrixfreeSolverExample example \endlink. + * + * \sa class SimplicialCholesky, DiagonalPreconditioner, IdentityPreconditioner + */ +template +class BiCGSTAB : public IterativeSolverBase> { typedef IterativeSolverBase Base; using Base::matrix; @@ -163,6 +160,7 @@ class BiCGSTAB : public IterativeSolverBase - explicit BiCGSTAB(const EigenBase& A) : Base(A.derived()) {} + * + * This constructor is a shortcut for the default constructor followed + * by a call to compute(). + * + * \warning this class stores a reference to the matrix A as well as some + * precomputed values that depend on it. Therefore, if \a A is changed + * this class becomes invalid. Call compute() to update it with the new + * matrix A, or modify a copy of A. + */ + template explicit BiCGSTAB(const EigenBase &A) : Base(A.derived()) {} ~BiCGSTAB() {} /** \internal */ - template - void _solve_with_guess_impl(const Rhs& b, Dest& x) const - { + template void _solve_with_guess_impl(const Rhs &b, Dest &x) const + { bool failed = false; - for(Index j=0; j - void _solve_impl(const MatrixBase& b, Dest& x) const + template void _solve_impl(const MatrixBase &b, Dest &x) const { - x.resize(this->rows(),b.cols()); + x.resize(this->rows(), b.cols()); x.setZero(); - _solve_with_guess_impl(b,x); + _solve_with_guess_impl(b, x); } protected: - }; -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_BICGSTAB_H +#endif// EIGEN_BICGSTAB_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/IterativeLinearSolvers/ConjugateGradient.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/IterativeLinearSolvers/ConjugateGradient.h index f7ce4713..aaca3df4 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/IterativeLinearSolvers/ConjugateGradient.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/IterativeLinearSolvers/ConjugateGradient.h @@ -10,102 +10,101 @@ #ifndef EIGEN_CONJUGATE_GRADIENT_H #define EIGEN_CONJUGATE_GRADIENT_H -namespace Eigen { +namespace Eigen { namespace internal { -/** \internal Low-level conjugate gradient algorithm - * \param mat The matrix A - * \param rhs The right hand side vector b - * \param x On input and initial solution, on output the computed solution. - * \param precond A preconditioner being able to efficiently solve for an - * approximation of Ax=b (regardless of b) - * \param iters On input the max number of iteration, on output the number of performed iterations. - * \param tol_error On input the tolerance error, on output an estimation of the relative error. - */ -template -EIGEN_DONT_INLINE -void conjugate_gradient(const MatrixType& mat, const Rhs& rhs, Dest& x, - const Preconditioner& precond, Index& iters, - typename Dest::RealScalar& tol_error) -{ - using std::sqrt; - using std::abs; - typedef typename Dest::RealScalar RealScalar; - typedef typename Dest::Scalar Scalar; - typedef Matrix VectorType; - - RealScalar tol = tol_error; - Index maxIters = iters; - - Index n = mat.cols(); - - VectorType residual = rhs - mat * x; //initial residual - - RealScalar rhsNorm2 = rhs.squaredNorm(); - if(rhsNorm2 == 0) - { - x.setZero(); - iters = 0; - tol_error = 0; - return; - } - const RealScalar considerAsZero = (std::numeric_limits::min)(); - RealScalar threshold = numext::maxi(tol*tol*rhsNorm2,considerAsZero); - RealScalar residualNorm2 = residual.squaredNorm(); - if (residualNorm2 < threshold) + /** \internal Low-level conjugate gradient algorithm + * \param mat The matrix A + * \param rhs The right hand side vector b + * \param x On input and initial solution, on output the computed solution. + * \param precond A preconditioner being able to efficiently solve for an + * approximation of Ax=b (regardless of b) + * \param iters On input the max number of iteration, on output the number of performed iterations. + * \param tol_error On input the tolerance error, on output an estimation of the relative error. + */ + template + EIGEN_DONT_INLINE void conjugate_gradient(const MatrixType &mat, + const Rhs &rhs, + Dest &x, + const Preconditioner &precond, + Index &iters, + typename Dest::RealScalar &tol_error) { - iters = 0; - tol_error = sqrt(residualNorm2 / rhsNorm2); - return; - } + using std::sqrt; + using std::abs; + typedef typename Dest::RealScalar RealScalar; + typedef typename Dest::Scalar Scalar; + typedef Matrix VectorType; - VectorType p(n); - p = precond.solve(residual); // initial search direction + RealScalar tol = tol_error; + Index maxIters = iters; - VectorType z(n), tmp(n); - RealScalar absNew = numext::real(residual.dot(p)); // the square of the absolute value of r scaled by invM - Index i = 0; - while(i < maxIters) - { - tmp.noalias() = mat * p; // the bottleneck of the algorithm - - Scalar alpha = absNew / p.dot(tmp); // the amount we travel on dir - x += alpha * p; // update solution - residual -= alpha * tmp; // update residual - - residualNorm2 = residual.squaredNorm(); - if(residualNorm2 < threshold) - break; - - z = precond.solve(residual); // approximately solve for "A z = residual" - - RealScalar absOld = absNew; - absNew = numext::real(residual.dot(z)); // update the absolute value of r - RealScalar beta = absNew / absOld; // calculate the Gram-Schmidt value used to create the new search direction - p = z + beta * p; // update search direction - i++; + Index n = mat.cols(); + + VectorType residual = rhs - mat * x;// initial residual + + RealScalar rhsNorm2 = rhs.squaredNorm(); + if (rhsNorm2 == 0) { + x.setZero(); + iters = 0; + tol_error = 0; + return; + } + const RealScalar considerAsZero = (std::numeric_limits::min)(); + RealScalar threshold = numext::maxi(tol * tol * rhsNorm2, considerAsZero); + RealScalar residualNorm2 = residual.squaredNorm(); + if (residualNorm2 < threshold) { + iters = 0; + tol_error = sqrt(residualNorm2 / rhsNorm2); + return; + } + + VectorType p(n); + p = precond.solve(residual);// initial search direction + + VectorType z(n), tmp(n); + RealScalar absNew = numext::real(residual.dot(p));// the square of the absolute value of r scaled by invM + Index i = 0; + while (i < maxIters) { + tmp.noalias() = mat * p;// the bottleneck of the algorithm + + Scalar alpha = absNew / p.dot(tmp);// the amount we travel on dir + x += alpha * p;// update solution + residual -= alpha * tmp;// update residual + + residualNorm2 = residual.squaredNorm(); + if (residualNorm2 < threshold) break; + + z = precond.solve(residual);// approximately solve for "A z = residual" + + RealScalar absOld = absNew; + absNew = numext::real(residual.dot(z));// update the absolute value of r + RealScalar beta = absNew / absOld;// calculate the Gram-Schmidt value used to create the new search direction + p = z + beta * p;// update search direction + i++; + } + tol_error = sqrt(residualNorm2 / rhsNorm2); + iters = i; } - tol_error = sqrt(residualNorm2 / rhsNorm2); - iters = i; -} -} +}// namespace internal -template< typename _MatrixType, int _UpLo=Lower, - typename _Preconditioner = DiagonalPreconditioner > +template> class ConjugateGradient; namespace internal { -template< typename _MatrixType, int _UpLo, typename _Preconditioner> -struct traits > -{ - typedef _MatrixType MatrixType; - typedef _Preconditioner Preconditioner; -}; + template + struct traits> + { + typedef _MatrixType MatrixType; + typedef _Preconditioner Preconditioner; + }; -} +}// namespace internal /** \ingroup IterativeLinearSolvers_Module * \brief A conjugate gradient solver for sparse (or dense) self-adjoint problems @@ -124,14 +123,14 @@ struct traits > * The maximal number of iterations and tolerance value can be controlled via the setMaxIterations() * and setTolerance() methods. The defaults are the size of the problem for the maximal number of iterations * and NumTraits::epsilon() for the tolerance. - * + * * The tolerance corresponds to the relative residual error: |Ax-b|/|b| - * + * * \b Performance: Even though the default value of \c _UpLo is \c Lower, significantly higher performance is * achieved when using a complete matrix and \b Lower|Upper as the \a _UpLo template parameter. Moreover, in this * case multi-threading can be exploited if the user code is compiled with OpenMP enabled. * See \ref TopicMultiThreading for details. - * + * * This class can be used as the direct solver classes. Here is a typical usage example: \code int n = 10000; @@ -146,16 +145,17 @@ struct traits > // update b, and solve again x = cg.solve(b); \endcode - * + * * By default the iterations start with x=0 as an initial guess of the solution. * One can control the start using the solveWithGuess() method. - * - * ConjugateGradient can also be used in a matrix-free context, see the following \link MatrixfreeSolverExample example \endlink. + * + * ConjugateGradient can also be used in a matrix-free context, see the following \link MatrixfreeSolverExample example + \endlink. * * \sa class LeastSquaresConjugateGradient, class SimplicialCholesky, DiagonalPreconditioner, IdentityPreconditioner */ -template< typename _MatrixType, int _UpLo, typename _Preconditioner> -class ConjugateGradient : public IterativeSolverBase > +template +class ConjugateGradient : public IterativeSolverBase> { typedef IterativeSolverBase Base; using Base::matrix; @@ -163,84 +163,78 @@ class ConjugateGradient : public IterativeSolverBase - explicit ConjugateGradient(const EigenBase& A) : Base(A.derived()) {} + * + * This constructor is a shortcut for the default constructor followed + * by a call to compute(). + * + * \warning this class stores a reference to the matrix A as well as some + * precomputed values that depend on it. Therefore, if \a A is changed + * this class becomes invalid. Call compute() to update it with the new + * matrix A, or modify a copy of A. + */ + template explicit ConjugateGradient(const EigenBase &A) : Base(A.derived()) {} ~ConjugateGradient() {} /** \internal */ - template - void _solve_with_guess_impl(const Rhs& b, Dest& x) const + template void _solve_with_guess_impl(const Rhs &b, Dest &x) const { typedef typename Base::MatrixWrapper MatrixWrapper; typedef typename Base::ActualMatrixType ActualMatrixType; enum { - TransposeInput = (!MatrixWrapper::MatrixFree) - && (UpLo==(Lower|Upper)) - && (!MatrixType::IsRowMajor) - && (!NumTraits::IsComplex) + TransposeInput = (!MatrixWrapper::MatrixFree) && (UpLo == (Lower | Upper)) && (!MatrixType::IsRowMajor) + && (!NumTraits::IsComplex) }; - typedef typename internal::conditional, ActualMatrixType const&>::type RowMajorWrapper; - EIGEN_STATIC_ASSERT(EIGEN_IMPLIES(MatrixWrapper::MatrixFree,UpLo==(Lower|Upper)),MATRIX_FREE_CONJUGATE_GRADIENT_IS_COMPATIBLE_WITH_UPPER_UNION_LOWER_MODE_ONLY); - typedef typename internal::conditional::Type - >::type SelfAdjointWrapper; + typedef + typename internal::conditional, ActualMatrixType const &>::type + RowMajorWrapper; + EIGEN_STATIC_ASSERT(EIGEN_IMPLIES(MatrixWrapper::MatrixFree, UpLo == (Lower | Upper)), + MATRIX_FREE_CONJUGATE_GRADIENT_IS_COMPATIBLE_WITH_UPPER_UNION_LOWER_MODE_ONLY); + typedef typename internal::conditional::Type>::type SelfAdjointWrapper; m_iterations = Base::maxIterations(); m_error = Base::m_tolerance; - for(Index j=0; j - void _solve_impl(const MatrixBase& b, Dest& x) const + template void _solve_impl(const MatrixBase &b, Dest &x) const { x.setZero(); - _solve_with_guess_impl(b.derived(),x); + _solve_with_guess_impl(b.derived(), x); } protected: - }; -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_CONJUGATE_GRADIENT_H +#endif// EIGEN_CONJUGATE_GRADIENT_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/IterativeLinearSolvers/IncompleteCholesky.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/IterativeLinearSolvers/IncompleteCholesky.h index e45c272b..4bc20141 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/IterativeLinearSolvers/IncompleteCholesky.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/IterativeLinearSolvers/IncompleteCholesky.h @@ -11,181 +11,200 @@ #ifndef EIGEN_INCOMPLETE_CHOlESKY_H #define EIGEN_INCOMPLETE_CHOlESKY_H -#include #include +#include -namespace Eigen { -/** - * \brief Modified Incomplete Cholesky with dual threshold - * - * References : C-J. Lin and J. J. Moré, Incomplete Cholesky Factorizations with - * Limited memory, SIAM J. Sci. Comput. 21(1), pp. 24-45, 1999 - * - * \tparam Scalar the scalar type of the input matrices - * \tparam _UpLo The triangular part that will be used for the computations. It can be Lower - * or Upper. Default is Lower. - * \tparam _OrderingType The ordering method to use, either AMDOrdering<> or NaturalOrdering<>. Default is AMDOrdering, - * unless EIGEN_MPL2_ONLY is defined, in which case the default is NaturalOrdering. - * - * \implsparsesolverconcept - * - * It performs the following incomplete factorization: \f$ S P A P' S \approx L L' \f$ - * where L is a lower triangular factor, S is a diagonal scaling matrix, and P is a - * fill-in reducing permutation as computed by the ordering method. - * - * \b Shifting \b strategy: Let \f$ B = S P A P' S \f$ be the scaled matrix on which the factorization is carried out, - * and \f$ \beta \f$ be the minimum value of the diagonal. If \f$ \beta > 0 \f$ then, the factorization is directly performed - * on the matrix B. Otherwise, the factorization is performed on the shifted matrix \f$ B + (\sigma+|\beta| I \f$ where - * \f$ \sigma \f$ is the initial shift value as returned and set by setInitialShift() method. The default value is \f$ \sigma = 10^{-3} \f$. - * If the factorization fails, then the shift in doubled until it succeed or a maximum of ten attempts. If it still fails, as returned by - * the info() method, then you can either increase the initial shift, or better use another preconditioning technique. - * - */ -template or NaturalOrdering<>. Default is + * AMDOrdering, unless EIGEN_MPL2_ONLY is defined, in which case the default is NaturalOrdering. + * + * \implsparsesolverconcept + * + * It performs the following incomplete factorization: \f$ S P A P' S \approx L L' \f$ + * where L is a lower triangular factor, S is a diagonal scaling matrix, and P is a + * fill-in reducing permutation as computed by the ordering method. + * + * \b Shifting \b strategy: Let \f$ B = S P A P' S \f$ be the scaled matrix on which the factorization is carried out, + * and \f$ \beta \f$ be the minimum value of the diagonal. If \f$ \beta > 0 \f$ then, the factorization is directly + * performed on the matrix B. Otherwise, the factorization is performed on the shifted matrix \f$ B + (\sigma+|\beta| I + * \f$ where + * \f$ \sigma \f$ is the initial shift value as returned and set by setInitialShift() method. The default value is \f$ + * \sigma = 10^{-3} \f$. If the factorization fails, then the shift in doubled until it succeed or a maximum of ten + * attempts. If it still fails, as returned by the info() method, then you can either increase the initial shift, or + * better use another preconditioning technique. + * + */ +template + AMDOrdering #else -NaturalOrdering + NaturalOrdering #endif -> -class IncompleteCholesky : public SparseSolverBase > + > +class IncompleteCholesky : public SparseSolverBase> { - protected: - typedef SparseSolverBase > Base; - using Base::m_isInitialized; - public: - typedef typename NumTraits::Real RealScalar; - typedef _OrderingType OrderingType; - typedef typename OrderingType::PermutationType PermutationType; - typedef typename PermutationType::StorageIndex StorageIndex; - typedef SparseMatrix FactorType; - typedef Matrix VectorSx; - typedef Matrix VectorRx; - typedef Matrix VectorIx; - typedef std::vector > VectorList; - enum { UpLo = _UpLo }; - enum { - ColsAtCompileTime = Dynamic, - MaxColsAtCompileTime = Dynamic - }; - public: - - /** Default constructor leaving the object in a partly non-initialized stage. - * - * You must call compute() or the pair analyzePattern()/factorize() to make it valid. - * - * \sa IncompleteCholesky(const MatrixType&) - */ - IncompleteCholesky() : m_initialShift(1e-3),m_factorizationIsOk(false) {} - - /** Constructor computing the incomplete factorization for the given matrix \a matrix. - */ - template - IncompleteCholesky(const MatrixType& matrix) : m_initialShift(1e-3),m_factorizationIsOk(false) - { - compute(matrix); - } - - /** \returns number of rows of the factored matrix */ - Index rows() const { return m_L.rows(); } - - /** \returns number of columns of the factored matrix */ - Index cols() const { return m_L.cols(); } - - - /** \brief Reports whether previous computation was successful. - * - * It triggers an assertion if \c *this has not been initialized through the respective constructor, - * or a call to compute() or analyzePattern(). - * - * \returns \c Success if computation was successful, - * \c NumericalIssue if the matrix appears to be negative. - */ - ComputationInfo info() const - { - eigen_assert(m_isInitialized && "IncompleteCholesky is not initialized."); - return m_info; - } - - /** \brief Set the initial shift parameter \f$ \sigma \f$. - */ - void setInitialShift(RealScalar shift) { m_initialShift = shift; } - - /** \brief Computes the fill reducing permutation vector using the sparsity pattern of \a mat - */ - template - void analyzePattern(const MatrixType& mat) - { - OrderingType ord; - PermutationType pinv; - ord(mat.template selfadjointView(), pinv); - if(pinv.size()>0) m_perm = pinv.inverse(); - else m_perm.resize(0); - m_L.resize(mat.rows(), mat.cols()); - m_analysisIsOk = true; - m_isInitialized = true; - m_info = Success; - } - - /** \brief Performs the numerical factorization of the input matrix \a mat - * - * The method analyzePattern() or compute() must have been called beforehand - * with a matrix having the same pattern. - * - * \sa compute(), analyzePattern() - */ - template - void factorize(const MatrixType& mat); - - /** Computes or re-computes the incomplete Cholesky factorization of the input matrix \a mat - * - * It is a shortcut for a sequential call to the analyzePattern() and factorize() methods. - * - * \sa analyzePattern(), factorize() - */ - template - void compute(const MatrixType& mat) - { - analyzePattern(mat); - factorize(mat); - } - - // internal - template - void _solve_impl(const Rhs& b, Dest& x) const - { - eigen_assert(m_factorizationIsOk && "factorize() should be called first"); - if (m_perm.rows() == b.rows()) x = m_perm * b; - else x = b; - x = m_scale.asDiagonal() * x; - x = m_L.template triangularView().solve(x); - x = m_L.adjoint().template triangularView().solve(x); - x = m_scale.asDiagonal() * x; - if (m_perm.rows() == b.rows()) - x = m_perm.inverse() * x; - } +protected: + typedef SparseSolverBase> Base; + using Base::m_isInitialized; + +public: + typedef typename NumTraits::Real RealScalar; + typedef _OrderingType OrderingType; + typedef typename OrderingType::PermutationType PermutationType; + typedef typename PermutationType::StorageIndex StorageIndex; + typedef SparseMatrix FactorType; + typedef Matrix VectorSx; + typedef Matrix VectorRx; + typedef Matrix VectorIx; + typedef std::vector> VectorList; + enum { UpLo = _UpLo }; + enum { ColsAtCompileTime = Dynamic, MaxColsAtCompileTime = Dynamic }; + +public: + /** Default constructor leaving the object in a partly non-initialized stage. + * + * You must call compute() or the pair analyzePattern()/factorize() to make it valid. + * + * \sa IncompleteCholesky(const MatrixType&) + */ + IncompleteCholesky() : m_initialShift(1e-3), m_factorizationIsOk(false) {} + + /** Constructor computing the incomplete factorization for the given matrix \a matrix. + */ + template + IncompleteCholesky(const MatrixType &matrix) : m_initialShift(1e-3), m_factorizationIsOk(false) + { + compute(matrix); + } + + /** \returns number of rows of the factored matrix */ + Index rows() const { return m_L.rows(); } + + /** \returns number of columns of the factored matrix */ + Index cols() const { return m_L.cols(); } + + + /** \brief Reports whether previous computation was successful. + * + * It triggers an assertion if \c *this has not been initialized through the respective constructor, + * or a call to compute() or analyzePattern(). + * + * \returns \c Success if computation was successful, + * \c NumericalIssue if the matrix appears to be negative. + */ + ComputationInfo info() const + { + eigen_assert(m_isInitialized && "IncompleteCholesky is not initialized."); + return m_info; + } + + /** \brief Set the initial shift parameter \f$ \sigma \f$. + */ + void setInitialShift(RealScalar shift) { m_initialShift = shift; } + + /** \brief Computes the fill reducing permutation vector using the sparsity pattern of \a mat + */ + template void analyzePattern(const MatrixType &mat) + { + OrderingType ord; + PermutationType pinv; + ord(mat.template selfadjointView(), pinv); + if (pinv.size() > 0) + m_perm = pinv.inverse(); + else + m_perm.resize(0); + m_L.resize(mat.rows(), mat.cols()); + m_analysisIsOk = true; + m_isInitialized = true; + m_info = Success; + } + + /** \brief Performs the numerical factorization of the input matrix \a mat + * + * The method analyzePattern() or compute() must have been called beforehand + * with a matrix having the same pattern. + * + * \sa compute(), analyzePattern() + */ + template void factorize(const MatrixType &mat); + + /** Computes or re-computes the incomplete Cholesky factorization of the input matrix \a mat + * + * It is a shortcut for a sequential call to the analyzePattern() and factorize() methods. + * + * \sa analyzePattern(), factorize() + */ + template void compute(const MatrixType &mat) + { + analyzePattern(mat); + factorize(mat); + } - /** \returns the sparse lower triangular factor L */ - const FactorType& matrixL() const { eigen_assert("m_factorizationIsOk"); return m_L; } + // internal + template void _solve_impl(const Rhs &b, Dest &x) const + { + eigen_assert(m_factorizationIsOk && "factorize() should be called first"); + if (m_perm.rows() == b.rows()) + x = m_perm * b; + else + x = b; + x = m_scale.asDiagonal() * x; + x = m_L.template triangularView().solve(x); + x = m_L.adjoint().template triangularView().solve(x); + x = m_scale.asDiagonal() * x; + if (m_perm.rows() == b.rows()) x = m_perm.inverse() * x; + } - /** \returns a vector representing the scaling factor S */ - const VectorRx& scalingS() const { eigen_assert("m_factorizationIsOk"); return m_scale; } + /** \returns the sparse lower triangular factor L */ + const FactorType &matrixL() const + { + eigen_assert("m_factorizationIsOk"); + return m_L; + } + + /** \returns a vector representing the scaling factor S */ + const VectorRx &scalingS() const + { + eigen_assert("m_factorizationIsOk"); + return m_scale; + } - /** \returns the fill-in reducing permutation P (can be empty for a natural ordering) */ - const PermutationType& permutationP() const { eigen_assert("m_analysisIsOk"); return m_perm; } + /** \returns the fill-in reducing permutation P (can be empty for a natural ordering) */ + const PermutationType &permutationP() const + { + eigen_assert("m_analysisIsOk"); + return m_perm; + } - protected: - FactorType m_L; // The lower part stored in CSC - VectorRx m_scale; // The vector for scaling the matrix - RealScalar m_initialShift; // The initial shift parameter - bool m_analysisIsOk; - bool m_factorizationIsOk; - ComputationInfo m_info; - PermutationType m_perm; +protected: + FactorType m_L;// The lower part stored in CSC + VectorRx m_scale;// The vector for scaling the matrix + RealScalar m_initialShift;// The initial shift parameter + bool m_analysisIsOk; + bool m_factorizationIsOk; + ComputationInfo m_info; + PermutationType m_perm; - private: - inline void updateList(Ref colPtr, Ref rowIdx, Ref vals, const Index& col, const Index& jk, VectorIx& firstElt, VectorList& listCol); -}; +private: + inline void updateList(Ref colPtr, + Ref rowIdx, + Ref vals, + const Index &col, + const Index &jk, + VectorIx &firstElt, + VectorList &listCol); +}; // Based on the following paper: // C-J. Lin and J. J. Moré, Incomplete Cholesky Factorizations with @@ -193,97 +212,88 @@ class IncompleteCholesky : public SparseSolverBase template -void IncompleteCholesky::factorize(const _MatrixType& mat) +void IncompleteCholesky::factorize(const _MatrixType &mat) { using std::sqrt; - eigen_assert(m_analysisIsOk && "analyzePattern() should be called first"); - - // Dropping strategy : Keep only the p largest elements per column, where p is the number of elements in the column of the original matrix. Other strategies will be added - + eigen_assert(m_analysisIsOk && "analyzePattern() should be called first"); + + // Dropping strategy : Keep only the p largest elements per column, where p is the number of elements in the column of + // the original matrix. Other strategies will be added + // Apply the fill-reducing permutation computed in analyzePattern() - if (m_perm.rows() == mat.rows() ) // To detect the null permutation + if (m_perm.rows() == mat.rows())// To detect the null permutation { // The temporary is needed to make sure that the diagonal entry is properly sorted FactorType tmp(mat.rows(), mat.cols()); tmp = mat.template selfadjointView<_UpLo>().twistedBy(m_perm); m_L.template selfadjointView() = tmp.template selfadjointView(); - } - else - { + } else { m_L.template selfadjointView() = mat.template selfadjointView<_UpLo>(); } - - Index n = m_L.cols(); + + Index n = m_L.cols(); Index nnz = m_L.nonZeros(); - Map vals(m_L.valuePtr(), nnz); //values - Map rowIdx(m_L.innerIndexPtr(), nnz); //Row indices - Map colPtr( m_L.outerIndexPtr(), n+1); // Pointer to the beginning of each row - VectorIx firstElt(n-1); // for each j, points to the next entry in vals that will be used in the factorization - VectorList listCol(n); // listCol(j) is a linked list of columns to update column j - VectorSx col_vals(n); // Store a nonzero values in each column - VectorIx col_irow(n); // Row indices of nonzero elements in each column + Map vals(m_L.valuePtr(), nnz);// values + Map rowIdx(m_L.innerIndexPtr(), nnz);// Row indices + Map colPtr(m_L.outerIndexPtr(), n + 1);// Pointer to the beginning of each row + VectorIx firstElt(n - 1);// for each j, points to the next entry in vals that will be used in the factorization + VectorList listCol(n);// listCol(j) is a linked list of columns to update column j + VectorSx col_vals(n);// Store a nonzero values in each column + VectorIx col_irow(n);// Row indices of nonzero elements in each column VectorIx col_pattern(n); col_pattern.fill(-1); StorageIndex col_nnz; - - - // Computes the scaling factors + + + // Computes the scaling factors m_scale.resize(n); m_scale.setZero(); for (Index j = 0; j < n; j++) - for (Index k = colPtr[j]; k < colPtr[j+1]; k++) - { + for (Index k = colPtr[j]; k < colPtr[j + 1]; k++) { m_scale(j) += numext::abs2(vals(k)); - if(rowIdx[k]!=j) - m_scale(rowIdx[k]) += numext::abs2(vals(k)); + if (rowIdx[k] != j) m_scale(rowIdx[k]) += numext::abs2(vals(k)); } - + m_scale = m_scale.cwiseSqrt().cwiseSqrt(); for (Index j = 0; j < n; ++j) - if(m_scale(j)>(std::numeric_limits::min)()) - m_scale(j) = RealScalar(1)/m_scale(j); + if (m_scale(j) > (std::numeric_limits::min)()) + m_scale(j) = RealScalar(1) / m_scale(j); else m_scale(j) = 1; // TODO disable scaling if not needed, i.e., if it is roughly uniform? (this will make solve() faster) - - // Scale and compute the shift for the matrix + + // Scale and compute the shift for the matrix RealScalar mindiag = NumTraits::highest(); - for (Index j = 0; j < n; j++) - { - for (Index k = colPtr[j]; k < colPtr[j+1]; k++) - vals[k] *= (m_scale(j)*m_scale(rowIdx[k])); - eigen_internal_assert(rowIdx[colPtr[j]]==j && "IncompleteCholesky: only the lower triangular part must be stored"); + for (Index j = 0; j < n; j++) { + for (Index k = colPtr[j]; k < colPtr[j + 1]; k++) vals[k] *= (m_scale(j) * m_scale(rowIdx[k])); + eigen_internal_assert( + rowIdx[colPtr[j]] == j && "IncompleteCholesky: only the lower triangular part must be stored"); mindiag = numext::mini(numext::real(vals[colPtr[j]]), mindiag); } FactorType L_save = m_L; - + RealScalar shift = 0; - if(mindiag <= RealScalar(0.)) - shift = m_initialShift - mindiag; + if (mindiag <= RealScalar(0.)) shift = m_initialShift - mindiag; m_info = NumericalIssue; // Try to perform the incomplete factorization using the current shift int iter = 0; - do - { + do { // Apply the shift to the diagonal elements of the matrix - for (Index j = 0; j < n; j++) - vals[colPtr[j]] += shift; + for (Index j = 0; j < n; j++) vals[colPtr[j]] += shift; // jki version of the Cholesky factorization - Index j=0; - for (; j < n; ++j) - { + Index j = 0; + for (; j < n; ++j) { // Left-looking factorization of the j-th column // First, load the j-th column into col_vals - Scalar diag = vals[colPtr[j]]; // It is assumed that only the lower part is stored + Scalar diag = vals[colPtr[j]];// It is assumed that only the lower part is stored col_nnz = 0; - for (Index i = colPtr[j] + 1; i < colPtr[j+1]; i++) - { + for (Index i = colPtr[j] + 1; i < colPtr[j + 1]; i++) { StorageIndex l = rowIdx[i]; col_vals(col_nnz) = vals[i]; col_irow(col_nnz) = l; @@ -293,69 +303,60 @@ void IncompleteCholesky::factorize(const _MatrixType { typename std::list::iterator k; // Browse all previous columns that will update column j - for(k = listCol[j].begin(); k != listCol[j].end(); k++) - { - Index jk = firstElt(*k); // First element to use in the column - eigen_internal_assert(rowIdx[jk]==j); + for (k = listCol[j].begin(); k != listCol[j].end(); k++) { + Index jk = firstElt(*k);// First element to use in the column + eigen_internal_assert(rowIdx[jk] == j); Scalar v_j_jk = numext::conj(vals[jk]); jk += 1; - for (Index i = jk; i < colPtr[*k+1]; i++) - { + for (Index i = jk; i < colPtr[*k + 1]; i++) { StorageIndex l = rowIdx[i]; - if(col_pattern[l]<0) - { + if (col_pattern[l] < 0) { col_vals(col_nnz) = vals[i] * v_j_jk; col_irow[col_nnz] = l; col_pattern(l) = col_nnz; col_nnz++; - } - else + } else col_vals(col_pattern[l]) -= vals[i] * v_j_jk; } - updateList(colPtr,rowIdx,vals, *k, jk, firstElt, listCol); + updateList(colPtr, rowIdx, vals, *k, jk, firstElt, listCol); } } // Scale the current column - if(numext::real(diag) <= 0) - { - if(++iter>=10) - return; + if (numext::real(diag) <= 0) { + if (++iter >= 10) return; // increase shift - shift = numext::maxi(m_initialShift,RealScalar(2)*shift); + shift = numext::maxi(m_initialShift, RealScalar(2) * shift); // restore m_L, col_pattern, and listCol vals = Map(L_save.valuePtr(), nnz); rowIdx = Map(L_save.innerIndexPtr(), nnz); - colPtr = Map(L_save.outerIndexPtr(), n+1); + colPtr = Map(L_save.outerIndexPtr(), n + 1); col_pattern.fill(-1); - for(Index i=0; i cvals = col_vals.head(col_nnz); Ref cirow = col_irow.head(col_nnz); - internal::QuickSplit(cvals,cirow, p); + internal::QuickSplit(cvals, cirow, p); // Insert the largest p elements in the matrix Index cpt = 0; - for (Index i = colPtr[j]+1; i < colPtr[j+1]; i++) - { + for (Index i = colPtr[j] + 1; i < colPtr[j + 1]; i++) { vals[i] = col_vals(cpt); rowIdx[i] = col_irow(cpt); // restore col_pattern: @@ -363,38 +364,41 @@ void IncompleteCholesky::factorize(const _MatrixType cpt++; } // Get the first smallest row index and put it after the diagonal element - Index jk = colPtr(j)+1; - updateList(colPtr,rowIdx,vals,j,jk,firstElt,listCol); + Index jk = colPtr(j) + 1; + updateList(colPtr, rowIdx, vals, j, jk, firstElt, listCol); } - if(j==n) - { + if (j == n) { m_factorizationIsOk = true; m_info = Success; } - } while(m_info!=Success); + } while (m_info != Success); } template -inline void IncompleteCholesky::updateList(Ref colPtr, Ref rowIdx, Ref vals, const Index& col, const Index& jk, VectorIx& firstElt, VectorList& listCol) +inline void IncompleteCholesky::updateList(Ref colPtr, + Ref rowIdx, + Ref vals, + const Index &col, + const Index &jk, + VectorIx &firstElt, + VectorList &listCol) { - if (jk < colPtr(col+1) ) - { - Index p = colPtr(col+1) - jk; - Index minpos; - rowIdx.segment(jk,p).minCoeff(&minpos); + if (jk < colPtr(col + 1)) { + Index p = colPtr(col + 1) - jk; + Index minpos; + rowIdx.segment(jk, p).minCoeff(&minpos); minpos += jk; - if (rowIdx(minpos) != rowIdx(jk)) - { - //Swap - std::swap(rowIdx(jk),rowIdx(minpos)); - std::swap(vals(jk),vals(minpos)); + if (rowIdx(minpos) != rowIdx(jk)) { + // Swap + std::swap(rowIdx(jk), rowIdx(minpos)); + std::swap(vals(jk), vals(minpos)); } - firstElt(col) = internal::convert_index(jk); - listCol[rowIdx(jk)].push_back(internal::convert_index(col)); + firstElt(col) = internal::convert_index(jk); + listCol[rowIdx(jk)].push_back(internal::convert_index(col)); } } -} // end namespace Eigen +}// end namespace Eigen #endif diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/IterativeLinearSolvers/IncompleteLUT.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/IterativeLinearSolvers/IncompleteLUT.h index 338e6f10..981f1fb7 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/IterativeLinearSolvers/IncompleteLUT.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/IterativeLinearSolvers/IncompleteLUT.h @@ -12,234 +12,224 @@ #define EIGEN_INCOMPLETE_LUT_H -namespace Eigen { +namespace Eigen { namespace internal { - -/** \internal - * Compute a quick-sort split of a vector - * On output, the vector row is permuted such that its elements satisfy - * abs(row(i)) >= abs(row(ncut)) if incut - * \param row The vector of values - * \param ind The array of index for the elements in @p row - * \param ncut The number of largest elements to keep - **/ -template -Index QuickSplit(VectorV &row, VectorI &ind, Index ncut) -{ - typedef typename VectorV::RealScalar RealScalar; - using std::swap; - using std::abs; - Index mid; - Index n = row.size(); /* length of the vector */ - Index first, last ; - - ncut--; /* to fit the zero-based indices */ - first = 0; - last = n-1; - if (ncut < first || ncut > last ) return 0; - - do { - mid = first; - RealScalar abskey = abs(row(mid)); - for (Index j = first + 1; j <= last; j++) { - if ( abs(row(j)) > abskey) { - ++mid; - swap(row(mid), row(j)); - swap(ind(mid), ind(j)); + + /** \internal + * Compute a quick-sort split of a vector + * On output, the vector row is permuted such that its elements satisfy + * abs(row(i)) >= abs(row(ncut)) if incut + * \param row The vector of values + * \param ind The array of index for the elements in @p row + * \param ncut The number of largest elements to keep + **/ + template Index QuickSplit(VectorV &row, VectorI &ind, Index ncut) + { + typedef typename VectorV::RealScalar RealScalar; + using std::swap; + using std::abs; + Index mid; + Index n = row.size(); /* length of the vector */ + Index first, last; + + ncut--; /* to fit the zero-based indices */ + first = 0; + last = n - 1; + if (ncut < first || ncut > last) return 0; + + do { + mid = first; + RealScalar abskey = abs(row(mid)); + for (Index j = first + 1; j <= last; j++) { + if (abs(row(j)) > abskey) { + ++mid; + swap(row(mid), row(j)); + swap(ind(mid), ind(j)); + } } - } - /* Interchange for the pivot element */ - swap(row(mid), row(first)); - swap(ind(mid), ind(first)); - - if (mid > ncut) last = mid - 1; - else if (mid < ncut ) first = mid + 1; - } while (mid != ncut ); - - return 0; /* mid is equal to ncut */ -} + /* Interchange for the pivot element */ + swap(row(mid), row(first)); + swap(ind(mid), ind(first)); + + if (mid > ncut) + last = mid - 1; + else if (mid < ncut) + first = mid + 1; + } while (mid != ncut); + + return 0; /* mid is equal to ncut */ + } }// end namespace internal /** \ingroup IterativeLinearSolvers_Module - * \class IncompleteLUT - * \brief Incomplete LU factorization with dual-threshold strategy - * - * \implsparsesolverconcept - * - * During the numerical factorization, two dropping rules are used : - * 1) any element whose magnitude is less than some tolerance is dropped. - * This tolerance is obtained by multiplying the input tolerance @p droptol - * by the average magnitude of all the original elements in the current row. - * 2) After the elimination of the row, only the @p fill largest elements in - * the L part and the @p fill largest elements in the U part are kept - * (in addition to the diagonal element ). Note that @p fill is computed from - * the input parameter @p fillfactor which is used the ratio to control the fill_in - * relatively to the initial number of nonzero elements. - * - * The two extreme cases are when @p droptol=0 (to keep all the @p fill*2 largest elements) - * and when @p fill=n/2 with @p droptol being different to zero. - * - * References : Yousef Saad, ILUT: A dual threshold incomplete LU factorization, - * Numerical Linear Algebra with Applications, 1(4), pp 387-402, 1994. - * - * NOTE : The following implementation is derived from the ILUT implementation - * in the SPARSKIT package, Copyright (C) 2005, the Regents of the University of Minnesota - * released under the terms of the GNU LGPL: - * http://www-users.cs.umn.edu/~saad/software/SPARSKIT/README - * However, Yousef Saad gave us permission to relicense his ILUT code to MPL2. - * See the Eigen mailing list archive, thread: ILUT, date: July 8, 2012: - * http://listengine.tuxfamily.org/lists.tuxfamily.org/eigen/2012/07/msg00064.html - * alternatively, on GMANE: - * http://comments.gmane.org/gmane.comp.lib.eigen/3302 - */ -template -class IncompleteLUT : public SparseSolverBase > + * \class IncompleteLUT + * \brief Incomplete LU factorization with dual-threshold strategy + * + * \implsparsesolverconcept + * + * During the numerical factorization, two dropping rules are used : + * 1) any element whose magnitude is less than some tolerance is dropped. + * This tolerance is obtained by multiplying the input tolerance @p droptol + * by the average magnitude of all the original elements in the current row. + * 2) After the elimination of the row, only the @p fill largest elements in + * the L part and the @p fill largest elements in the U part are kept + * (in addition to the diagonal element ). Note that @p fill is computed from + * the input parameter @p fillfactor which is used the ratio to control the fill_in + * relatively to the initial number of nonzero elements. + * + * The two extreme cases are when @p droptol=0 (to keep all the @p fill*2 largest elements) + * and when @p fill=n/2 with @p droptol being different to zero. + * + * References : Yousef Saad, ILUT: A dual threshold incomplete LU factorization, + * Numerical Linear Algebra with Applications, 1(4), pp 387-402, 1994. + * + * NOTE : The following implementation is derived from the ILUT implementation + * in the SPARSKIT package, Copyright (C) 2005, the Regents of the University of Minnesota + * released under the terms of the GNU LGPL: + * http://www-users.cs.umn.edu/~saad/software/SPARSKIT/README + * However, Yousef Saad gave us permission to relicense his ILUT code to MPL2. + * See the Eigen mailing list archive, thread: ILUT, date: July 8, 2012: + * http://listengine.tuxfamily.org/lists.tuxfamily.org/eigen/2012/07/msg00064.html + * alternatively, on GMANE: + * http://comments.gmane.org/gmane.comp.lib.eigen/3302 + */ +template +class IncompleteLUT : public SparseSolverBase> { - protected: - typedef SparseSolverBase Base; - using Base::m_isInitialized; - public: - typedef _Scalar Scalar; - typedef _StorageIndex StorageIndex; - typedef typename NumTraits::Real RealScalar; - typedef Matrix Vector; - typedef Matrix VectorI; - typedef SparseMatrix FactorType; - - enum { - ColsAtCompileTime = Dynamic, - MaxColsAtCompileTime = Dynamic - }; - - public: - - IncompleteLUT() - : m_droptol(NumTraits::dummy_precision()), m_fillfactor(10), - m_analysisIsOk(false), m_factorizationIsOk(false) - {} - - template - explicit IncompleteLUT(const MatrixType& mat, const RealScalar& droptol=NumTraits::dummy_precision(), int fillfactor = 10) - : m_droptol(droptol),m_fillfactor(fillfactor), - m_analysisIsOk(false),m_factorizationIsOk(false) - { - eigen_assert(fillfactor != 0); - compute(mat); - } - - Index rows() const { return m_lu.rows(); } - - Index cols() const { return m_lu.cols(); } - - /** \brief Reports whether previous computation was successful. - * - * \returns \c Success if computation was succesful, - * \c NumericalIssue if the matrix.appears to be negative. - */ - ComputationInfo info() const - { - eigen_assert(m_isInitialized && "IncompleteLUT is not initialized."); - return m_info; - } - - template - void analyzePattern(const MatrixType& amat); - - template - void factorize(const MatrixType& amat); - - /** - * Compute an incomplete LU factorization with dual threshold on the matrix mat - * No pivoting is done in this version - * - **/ - template - IncompleteLUT& compute(const MatrixType& amat) - { - analyzePattern(amat); - factorize(amat); - return *this; - } +protected: + typedef SparseSolverBase Base; + using Base::m_isInitialized; + +public: + typedef _Scalar Scalar; + typedef _StorageIndex StorageIndex; + typedef typename NumTraits::Real RealScalar; + typedef Matrix Vector; + typedef Matrix VectorI; + typedef SparseMatrix FactorType; + + enum { ColsAtCompileTime = Dynamic, MaxColsAtCompileTime = Dynamic }; + +public: + IncompleteLUT() + : m_droptol(NumTraits::dummy_precision()), m_fillfactor(10), m_analysisIsOk(false), + m_factorizationIsOk(false) + {} + + template + explicit IncompleteLUT(const MatrixType &mat, + const RealScalar &droptol = NumTraits::dummy_precision(), + int fillfactor = 10) + : m_droptol(droptol), m_fillfactor(fillfactor), m_analysisIsOk(false), m_factorizationIsOk(false) + { + eigen_assert(fillfactor != 0); + compute(mat); + } - void setDroptol(const RealScalar& droptol); - void setFillfactor(int fillfactor); - - template - void _solve_impl(const Rhs& b, Dest& x) const - { - x = m_Pinv * b; - x = m_lu.template triangularView().solve(x); - x = m_lu.template triangularView().solve(x); - x = m_P * x; - } + Index rows() const { return m_lu.rows(); } -protected: + Index cols() const { return m_lu.cols(); } - /** keeps off-diagonal entries; drops diagonal entries */ - struct keep_diag { - inline bool operator() (const Index& row, const Index& col, const Scalar&) const - { - return row!=col; - } - }; + /** \brief Reports whether previous computation was successful. + * + * \returns \c Success if computation was succesful, + * \c NumericalIssue if the matrix.appears to be negative. + */ + ComputationInfo info() const + { + eigen_assert(m_isInitialized && "IncompleteLUT is not initialized."); + return m_info; + } + + template void analyzePattern(const MatrixType &amat); + + template void factorize(const MatrixType &amat); + + /** + * Compute an incomplete LU factorization with dual threshold on the matrix mat + * No pivoting is done in this version + * + **/ + template IncompleteLUT &compute(const MatrixType &amat) + { + analyzePattern(amat); + factorize(amat); + return *this; + } + + void setDroptol(const RealScalar &droptol); + void setFillfactor(int fillfactor); + + template void _solve_impl(const Rhs &b, Dest &x) const + { + x = m_Pinv * b; + x = m_lu.template triangularView().solve(x); + x = m_lu.template triangularView().solve(x); + x = m_P * x; + } protected: + /** keeps off-diagonal entries; drops diagonal entries */ + struct keep_diag + { + inline bool operator()(const Index &row, const Index &col, const Scalar &) const { return row != col; } + }; - FactorType m_lu; - RealScalar m_droptol; - int m_fillfactor; - bool m_analysisIsOk; - bool m_factorizationIsOk; - ComputationInfo m_info; - PermutationMatrix m_P; // Fill-reducing permutation - PermutationMatrix m_Pinv; // Inverse permutation +protected: + FactorType m_lu; + RealScalar m_droptol; + int m_fillfactor; + bool m_analysisIsOk; + bool m_factorizationIsOk; + ComputationInfo m_info; + PermutationMatrix m_P;// Fill-reducing permutation + PermutationMatrix m_Pinv;// Inverse permutation }; /** * Set control parameter droptol - * \param droptol Drop any element whose magnitude is less than this tolerance - **/ + * \param droptol Drop any element whose magnitude is less than this tolerance + **/ template -void IncompleteLUT::setDroptol(const RealScalar& droptol) +void IncompleteLUT::setDroptol(const RealScalar &droptol) { - this->m_droptol = droptol; + this->m_droptol = droptol; } /** * Set control parameter fillfactor - * \param fillfactor This is used to compute the number @p fill_in of largest elements to keep on each row. - **/ -template -void IncompleteLUT::setFillfactor(int fillfactor) + * \param fillfactor This is used to compute the number @p fill_in of largest elements to keep on each row. + **/ +template void IncompleteLUT::setFillfactor(int fillfactor) { - this->m_fillfactor = fillfactor; + this->m_fillfactor = fillfactor; } -template +template template -void IncompleteLUT::analyzePattern(const _MatrixType& amat) +void IncompleteLUT::analyzePattern(const _MatrixType &amat) { // Compute the Fill-reducing permutation // Since ILUT does not perform any numerical pivoting, // it is highly preferable to keep the diagonal through symmetric permutations. #ifndef EIGEN_MPL2_ONLY // To this end, let's symmetrize the pattern and perform AMD on it. - SparseMatrix mat1 = amat; - SparseMatrix mat2 = amat.transpose(); + SparseMatrix mat1 = amat; + SparseMatrix mat2 = amat.transpose(); // FIXME for a matrix with nearly symmetric pattern, mat2+mat1 is the appropriate choice. // on the other hand for a really non-symmetric pattern, mat2*mat1 should be prefered... - SparseMatrix AtA = mat2 + mat1; + SparseMatrix AtA = mat2 + mat1; AMDOrdering ordering; - ordering(AtA,m_P); - m_Pinv = m_P.inverse(); // cache the inverse permutation + ordering(AtA, m_P); + m_Pinv = m_P.inverse();// cache the inverse permutation #else // If AMD is not available, (MPL2-only), then let's use the slower COLAMD routine. - SparseMatrix mat1 = amat; + SparseMatrix mat1 = amat; COLAMDOrdering ordering; - ordering(mat1,m_Pinv); + ordering(mat1, m_Pinv); m_P = m_Pinv.inverse(); #endif @@ -248,9 +238,9 @@ void IncompleteLUT::analyzePattern(const _MatrixType& amat) m_isInitialized = true; } -template +template template -void IncompleteLUT::factorize(const _MatrixType& amat) +void IncompleteLUT::factorize(const _MatrixType &amat) { using std::sqrt; using std::swap; @@ -258,16 +248,16 @@ void IncompleteLUT::factorize(const _MatrixType& amat) using internal::convert_index; eigen_assert((amat.rows() == amat.cols()) && "The factorization should be done on a square matrix"); - Index n = amat.cols(); // Size of the matrix - m_lu.resize(n,n); + Index n = amat.cols();// Size of the matrix + m_lu.resize(n, n); // Declare Working vectors and variables - Vector u(n) ; // real values of the row -- maximum size is n -- - VectorI ju(n); // column position of the values in u -- maximum size is n - VectorI jr(n); // Indicate the position of the nonzero elements in the vector u -- A zero location is indicated by -1 + Vector u(n);// real values of the row -- maximum size is n -- + VectorI ju(n);// column position of the values in u -- maximum size is n + VectorI jr(n);// Indicate the position of the nonzero elements in the vector u -- A zero location is indicated by -1 // Apply the fill-reducing permutation eigen_assert(m_analysisIsOk && "You must first call analyzePattern()"); - SparseMatrix mat; + SparseMatrix mat; mat = amat.twistedBy(m_Pinv); // Initialization @@ -276,44 +266,37 @@ void IncompleteLUT::factorize(const _MatrixType& amat) u.fill(0); // number of largest elements to keep in each row: - Index fill_in = (amat.nonZeros()*m_fillfactor)/n + 1; + Index fill_in = (amat.nonZeros() * m_fillfactor) / n + 1; if (fill_in > n) fill_in = n; // number of largest nonzero elements to keep in the L and the U part of the current row: - Index nnzL = fill_in/2; + Index nnzL = fill_in / 2; Index nnzU = nnzL; m_lu.reserve(n * (nnzL + nnzU + 1)); // global loop over the rows of the sparse matrix - for (Index ii = 0; ii < n; ii++) - { + for (Index ii = 0; ii < n; ii++) { // 1 - copy the lower and the upper part of the row i of mat in the working vector u - Index sizeu = 1; // number of nonzero elements in the upper part of the current row - Index sizel = 0; // number of nonzero elements in the lower part of the current row - ju(ii) = convert_index(ii); - u(ii) = 0; - jr(ii) = convert_index(ii); + Index sizeu = 1;// number of nonzero elements in the upper part of the current row + Index sizel = 0;// number of nonzero elements in the lower part of the current row + ju(ii) = convert_index(ii); + u(ii) = 0; + jr(ii) = convert_index(ii); RealScalar rownorm = 0; - typename FactorType::InnerIterator j_it(mat, ii); // Iterate through the current row ii - for (; j_it; ++j_it) - { + typename FactorType::InnerIterator j_it(mat, ii);// Iterate through the current row ii + for (; j_it; ++j_it) { Index k = j_it.index(); - if (k < ii) - { + if (k < ii) { // copy the lower part ju(sizel) = convert_index(k); u(sizel) = j_it.value(); jr(k) = convert_index(sizel); ++sizel; - } - else if (k == ii) - { + } else if (k == ii) { u(ii) = j_it.value(); - } - else - { + } else { // copy the upper part Index jpos = ii + sizeu; ju(jpos) = convert_index(k); @@ -325,8 +308,7 @@ void IncompleteLUT::factorize(const _MatrixType& amat) } // 2 - detect possible zero row - if(rownorm==0) - { + if (rownorm == 0) { m_info = NumericalIssue; return; } @@ -336,15 +318,13 @@ void IncompleteLUT::factorize(const _MatrixType& amat) // 3 - eliminate the previous nonzero rows Index jj = 0; Index len = 0; - while (jj < sizel) - { + while (jj < sizel) { // In order to eliminate in the correct order, // we must select first the smallest column index among ju(jj:sizel) Index k; - Index minrow = ju.segment(jj,sizel-jj).minCoeff(&k); // k is relative to the segment + Index minrow = ju.segment(jj, sizel - jj).minCoeff(&k);// k is relative to the segment k += jj; - if (minrow != ju(jj)) - { + if (minrow != ju(jj)) { // swap the two locations Index j = ju(jj); swap(ju(jj), ju(k)); @@ -358,55 +338,51 @@ void IncompleteLUT::factorize(const _MatrixType& amat) // Start elimination typename FactorType::InnerIterator ki_it(m_lu, minrow); while (ki_it && ki_it.index() < minrow) ++ki_it; - eigen_internal_assert(ki_it && ki_it.col()==minrow); + eigen_internal_assert(ki_it && ki_it.col() == minrow); Scalar fact = u(jj) / ki_it.value(); // drop too small elements - if(abs(fact) <= m_droptol) - { + if (abs(fact) <= m_droptol) { jj++; continue; } // linear combination of the current row ii and the row minrow ++ki_it; - for (; ki_it; ++ki_it) - { + for (; ki_it; ++ki_it) { Scalar prod = fact * ki_it.value(); - Index j = ki_it.index(); - Index jpos = jr(j); - if (jpos == -1) // fill-in element + Index j = ki_it.index(); + Index jpos = jr(j); + if (jpos == -1)// fill-in element { Index newpos; - if (j >= ii) // dealing with the upper part + if (j >= ii)// dealing with the upper part { newpos = ii + sizeu; sizeu++; - eigen_internal_assert(sizeu<=n); - } - else // dealing with the lower part + eigen_internal_assert(sizeu <= n); + } else// dealing with the lower part { newpos = sizel; sizel++; - eigen_internal_assert(sizel<=ii); + eigen_internal_assert(sizel <= ii); } ju(newpos) = convert_index(j); u(newpos) = -prod; jr(j) = convert_index(newpos); - } - else + } else u(jpos) -= prod; } // store the pivot element - u(len) = fact; + u(len) = fact; ju(len) = convert_index(minrow); ++len; jj++; - } // end of the elimination on the row ii + }// end of the elimination on the row ii // reset the upper part of the pointer jr to zero - for(Index k = 0; k ::factorize(const _MatrixType& amat) // store the largest m_fill elements of the L part m_lu.startVec(ii); - for(Index k = 0; k < len; k++) - m_lu.insertBackByOuterInnerUnordered(ii,ju(k)) = u(k); + for (Index k = 0; k < len; k++) m_lu.insertBackByOuterInnerUnordered(ii, ju(k)) = u(k); // store the diagonal element // apply a shifting rule to avoid zero pivots (we are doing an incomplete factorization) - if (u(ii) == Scalar(0)) - u(ii) = sqrt(m_droptol) * rownorm; + if (u(ii) == Scalar(0)) u(ii) = sqrt(m_droptol) * rownorm; m_lu.insertBackByOuterInnerUnordered(ii, ii) = u(ii); // sort the U-part of the row // apply the dropping rule first len = 0; - for(Index k = 1; k < sizeu; k++) - { - if(abs(u(ii+k)) > m_droptol * rownorm ) - { + for (Index k = 1; k < sizeu; k++) { + if (abs(u(ii + k)) > m_droptol * rownorm) { ++len; - u(ii + len) = u(ii + k); + u(ii + len) = u(ii + k); ju(ii + len) = ju(ii + k); } } - sizeu = len + 1; // +1 to take into account the diagonal element + sizeu = len + 1;// +1 to take into account the diagonal element len = (std::min)(sizeu, nnzU); - typename Vector::SegmentReturnType uu(u.segment(ii+1, sizeu-1)); - typename VectorI::SegmentReturnType juu(ju.segment(ii+1, sizeu-1)); + typename Vector::SegmentReturnType uu(u.segment(ii + 1, sizeu - 1)); + typename VectorI::SegmentReturnType juu(ju.segment(ii + 1, sizeu - 1)); internal::QuickSplit(uu, juu, len); // store the largest elements of the U part - for(Index k = ii + 1; k < ii + len; k++) - m_lu.insertBackByOuterInnerUnordered(ii,ju(k)) = u(k); + for (Index k = ii + 1; k < ii + len; k++) m_lu.insertBackByOuterInnerUnordered(ii, ju(k)) = u(k); } m_lu.finalize(); m_lu.makeCompressed(); @@ -457,6 +428,6 @@ void IncompleteLUT::factorize(const _MatrixType& amat) m_info = Success; } -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_INCOMPLETE_LUT_H +#endif// EIGEN_INCOMPLETE_LUT_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/IterativeLinearSolvers/IterativeSolverBase.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/IterativeLinearSolvers/IterativeSolverBase.h index 7c2326eb..ccc1ef07 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/IterativeLinearSolvers/IterativeSolverBase.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/IterativeLinearSolvers/IterativeSolverBase.h @@ -10,142 +10,117 @@ #ifndef EIGEN_ITERATIVE_SOLVER_BASE_H #define EIGEN_ITERATIVE_SOLVER_BASE_H -namespace Eigen { +namespace Eigen { namespace internal { -template -struct is_ref_compatible_impl -{ -private: - template - struct any_conversion + template struct is_ref_compatible_impl { - template any_conversion(const volatile T&); - template any_conversion(T&); - }; - struct yes {int a[1];}; - struct no {int a[2];}; - - template - static yes test(const Ref&, int); - template - static no test(any_conversion, ...); - -public: - static MatrixType ms_from; - enum { value = sizeof(test(ms_from, 0))==sizeof(yes) }; -}; - -template -struct is_ref_compatible -{ - enum { value = is_ref_compatible_impl::type>::value }; -}; + private: + template struct any_conversion + { + template any_conversion(const volatile T &); + template any_conversion(T &); + }; + struct yes + { + int a[1]; + }; + struct no + { + int a[2]; + }; -template::value> -class generic_matrix_wrapper; + template static yes test(const Ref &, int); + template static no test(any_conversion, ...); -// We have an explicit matrix at hand, compatible with Ref<> -template -class generic_matrix_wrapper -{ -public: - typedef Ref ActualMatrixType; - template struct ConstSelfAdjointViewReturnType { - typedef typename ActualMatrixType::template ConstSelfAdjointViewReturnType::Type Type; + public: + static MatrixType ms_from; + enum { value = sizeof(test(ms_from, 0)) == sizeof(yes) }; }; - enum { - MatrixFree = false + template struct is_ref_compatible + { + enum { value = is_ref_compatible_impl::type>::value }; }; - generic_matrix_wrapper() - : m_dummy(0,0), m_matrix(m_dummy) - {} - - template - generic_matrix_wrapper(const InputType &mat) - : m_matrix(mat) - {} + template::value> + class generic_matrix_wrapper; - const ActualMatrixType& matrix() const + // We have an explicit matrix at hand, compatible with Ref<> + template class generic_matrix_wrapper { - return m_matrix; - } + public: + typedef Ref ActualMatrixType; + template struct ConstSelfAdjointViewReturnType + { + typedef typename ActualMatrixType::template ConstSelfAdjointViewReturnType::Type Type; + }; - template - void grab(const EigenBase &mat) - { - m_matrix.~Ref(); - ::new (&m_matrix) Ref(mat.derived()); - } + enum { MatrixFree = false }; - void grab(const Ref &mat) - { - if(&(mat.derived()) != &m_matrix) + generic_matrix_wrapper() : m_dummy(0, 0), m_matrix(m_dummy) {} + + template generic_matrix_wrapper(const InputType &mat) : m_matrix(mat) {} + + const ActualMatrixType &matrix() const { return m_matrix; } + + template void grab(const EigenBase &mat) { m_matrix.~Ref(); - ::new (&m_matrix) Ref(mat); + ::new (&m_matrix) Ref(mat.derived()); } - } -protected: - MatrixType m_dummy; // used to default initialize the Ref<> object - ActualMatrixType m_matrix; -}; + void grab(const Ref &mat) + { + if (&(mat.derived()) != &m_matrix) { + m_matrix.~Ref(); + ::new (&m_matrix) Ref(mat); + } + } -// MatrixType is not compatible with Ref<> -> matrix-free wrapper -template -class generic_matrix_wrapper -{ -public: - typedef MatrixType ActualMatrixType; - template struct ConstSelfAdjointViewReturnType - { - typedef ActualMatrixType Type; + protected: + MatrixType m_dummy;// used to default initialize the Ref<> object + ActualMatrixType m_matrix; }; - enum { - MatrixFree = true - }; + // MatrixType is not compatible with Ref<> -> matrix-free wrapper + template class generic_matrix_wrapper + { + public: + typedef MatrixType ActualMatrixType; + template struct ConstSelfAdjointViewReturnType + { + typedef ActualMatrixType Type; + }; - generic_matrix_wrapper() - : mp_matrix(0) - {} + enum { MatrixFree = true }; - generic_matrix_wrapper(const MatrixType &mat) - : mp_matrix(&mat) - {} + generic_matrix_wrapper() : mp_matrix(0) {} - const ActualMatrixType& matrix() const - { - return *mp_matrix; - } + generic_matrix_wrapper(const MatrixType &mat) : mp_matrix(&mat) {} - void grab(const MatrixType &mat) - { - mp_matrix = &mat; - } + const ActualMatrixType &matrix() const { return *mp_matrix; } -protected: - const ActualMatrixType *mp_matrix; -}; + void grab(const MatrixType &mat) { mp_matrix = &mat; } + + protected: + const ActualMatrixType *mp_matrix; + }; -} +}// namespace internal /** \ingroup IterativeLinearSolvers_Module - * \brief Base class for linear iterative solvers - * - * \sa class SimplicialCholesky, DiagonalPreconditioner, IdentityPreconditioner - */ -template< typename Derived> -class IterativeSolverBase : public SparseSolverBase + * \brief Base class for linear iterative solvers + * + * \sa class SimplicialCholesky, DiagonalPreconditioner, IdentityPreconditioner + */ +template class IterativeSolverBase : public SparseSolverBase { protected: typedef SparseSolverBase Base; using Base::m_isInitialized; - + public: typedef typename internal::traits::MatrixType MatrixType; typedef typename internal::traits::Preconditioner Preconditioner; @@ -153,48 +128,39 @@ class IterativeSolverBase : public SparseSolverBase typedef typename MatrixType::StorageIndex StorageIndex; typedef typename MatrixType::RealScalar RealScalar; - enum { - ColsAtCompileTime = MatrixType::ColsAtCompileTime, - MaxColsAtCompileTime = MatrixType::MaxColsAtCompileTime - }; + enum { ColsAtCompileTime = MatrixType::ColsAtCompileTime, MaxColsAtCompileTime = MatrixType::MaxColsAtCompileTime }; public: - using Base::derived; /** Default constructor. */ - IterativeSolverBase() - { - init(); - } + IterativeSolverBase() { init(); } /** Initialize the solver with matrix \a A for further \c Ax=b solving. - * - * This constructor is a shortcut for the default constructor followed - * by a call to compute(). - * - * \warning this class stores a reference to the matrix A as well as some - * precomputed values that depend on it. Therefore, if \a A is changed - * this class becomes invalid. Call compute() to update it with the new - * matrix A, or modify a copy of A. - */ + * + * This constructor is a shortcut for the default constructor followed + * by a call to compute(). + * + * \warning this class stores a reference to the matrix A as well as some + * precomputed values that depend on it. Therefore, if \a A is changed + * this class becomes invalid. Call compute() to update it with the new + * matrix A, or modify a copy of A. + */ template - explicit IterativeSolverBase(const EigenBase& A) - : m_matrixWrapper(A.derived()) + explicit IterativeSolverBase(const EigenBase &A) : m_matrixWrapper(A.derived()) { init(); compute(matrix()); } ~IterativeSolverBase() {} - + /** Initializes the iterative solver for the sparsity pattern of the matrix \a A for further solving \c Ax=b problems. - * - * Currently, this function mostly calls analyzePattern on the preconditioner. In the future - * we might, for instance, implement column reordering for faster matrix vector products. - */ - template - Derived& analyzePattern(const EigenBase& A) + * + * Currently, this function mostly calls analyzePattern on the preconditioner. In the future + * we might, for instance, implement column reordering for faster matrix vector products. + */ + template Derived &analyzePattern(const EigenBase &A) { grab(A.derived()); m_preconditioner.analyzePattern(matrix()); @@ -203,20 +169,20 @@ class IterativeSolverBase : public SparseSolverBase m_info = m_preconditioner.info(); return derived(); } - - /** Initializes the iterative solver with the numerical values of the matrix \a A for further solving \c Ax=b problems. - * - * Currently, this function mostly calls factorize on the preconditioner. - * - * \warning this class stores a reference to the matrix A as well as some - * precomputed values that depend on it. Therefore, if \a A is changed - * this class becomes invalid. Call compute() to update it with the new - * matrix A, or modify a copy of A. - */ - template - Derived& factorize(const EigenBase& A) + + /** Initializes the iterative solver with the numerical values of the matrix \a A for further solving \c Ax=b + * problems. + * + * Currently, this function mostly calls factorize on the preconditioner. + * + * \warning this class stores a reference to the matrix A as well as some + * precomputed values that depend on it. Therefore, if \a A is changed + * this class becomes invalid. Call compute() to update it with the new + * matrix A, or modify a copy of A. + */ + template Derived &factorize(const EigenBase &A) { - eigen_assert(m_analysisIsOk && "You must first call analyzePattern()"); + eigen_assert(m_analysisIsOk && "You must first call analyzePattern()"); grab(A.derived()); m_preconditioner.factorize(matrix()); m_factorizationIsOk = true; @@ -225,17 +191,16 @@ class IterativeSolverBase : public SparseSolverBase } /** Initializes the iterative solver with the matrix \a A for further solving \c Ax=b problems. - * - * Currently, this function mostly initializes/computes the preconditioner. In the future - * we might, for instance, implement column reordering for faster matrix vector products. - * - * \warning this class stores a reference to the matrix A as well as some - * precomputed values that depend on it. Therefore, if \a A is changed - * this class becomes invalid. Call compute() to update it with the new - * matrix A, or modify a copy of A. - */ - template - Derived& compute(const EigenBase& A) + * + * Currently, this function mostly initializes/computes the preconditioner. In the future + * we might, for instance, implement column reordering for faster matrix vector products. + * + * \warning this class stores a reference to the matrix A as well as some + * precomputed values that depend on it. Therefore, if \a A is changed + * this class becomes invalid. Call compute() to update it with the new + * matrix A, or modify a copy of A. + */ + template Derived &compute(const EigenBase &A) { grab(A.derived()); m_preconditioner.compute(matrix()); @@ -253,40 +218,37 @@ class IterativeSolverBase : public SparseSolverBase Index cols() const { return matrix().cols(); } /** \returns the tolerance threshold used by the stopping criteria. - * \sa setTolerance() - */ + * \sa setTolerance() + */ RealScalar tolerance() const { return m_tolerance; } - + /** Sets the tolerance threshold used by the stopping criteria. - * - * This value is used as an upper bound to the relative residual error: |Ax-b|/|b|. - * The default value is the machine precision given by NumTraits::epsilon() - */ - Derived& setTolerance(const RealScalar& tolerance) + * + * This value is used as an upper bound to the relative residual error: |Ax-b|/|b|. + * The default value is the machine precision given by NumTraits::epsilon() + */ + Derived &setTolerance(const RealScalar &tolerance) { m_tolerance = tolerance; return derived(); } /** \returns a read-write reference to the preconditioner for custom configuration. */ - Preconditioner& preconditioner() { return m_preconditioner; } - + Preconditioner &preconditioner() { return m_preconditioner; } + /** \returns a read-only reference to the preconditioner. */ - const Preconditioner& preconditioner() const { return m_preconditioner; } + const Preconditioner &preconditioner() const { return m_preconditioner; } /** \returns the max number of iterations. - * It is either the value setted by setMaxIterations or, by default, - * twice the number of columns of the matrix. - */ - Index maxIterations() const - { - return (m_maxIterations<0) ? 2*matrix().cols() : m_maxIterations; - } - + * It is either the value setted by setMaxIterations or, by default, + * twice the number of columns of the matrix. + */ + Index maxIterations() const { return (m_maxIterations < 0) ? 2 * matrix().cols() : m_maxIterations; } + /** Sets the max number of iterations. - * Default is twice the number of columns of the matrix. - */ - Derived& setMaxIterations(Index maxIters) + * Default is twice the number of columns of the matrix. + */ + Derived &setMaxIterations(Index maxIters) { m_maxIterations = maxIters; return derived(); @@ -300,8 +262,8 @@ class IterativeSolverBase : public SparseSolverBase } /** \returns the tolerance error reached during the last solve. - * It is a close approximation of the true relative residual error |Ax-b|/|b|. - */ + * It is a close approximation of the true relative residual error |Ax-b|/|b|. + */ RealScalar error() const { eigen_assert(m_isInitialized && "ConjugateGradient is not initialized."); @@ -309,16 +271,15 @@ class IterativeSolverBase : public SparseSolverBase } /** \returns the solution x of \f$ A x = b \f$ using the current decomposition of A - * and \a x0 as an initial solution. - * - * \sa solve(), compute() - */ - template - inline const SolveWithGuess - solveWithGuess(const MatrixBase& b, const Guess& x0) const + * and \a x0 as an initial solution. + * + * \sa solve(), compute() + */ + template + inline const SolveWithGuess solveWithGuess(const MatrixBase &b, const Guess &x0) const { eigen_assert(m_isInitialized && "Solver is not initialized."); - eigen_assert(derived().rows()==b.rows() && "solve(): invalid number of rows of the right hand side matrix b"); + eigen_assert(derived().rows() == b.rows() && "solve(): invalid number of rows of the right hand side matrix b"); return SolveWithGuess(derived(), b.derived(), x0); } @@ -328,24 +289,24 @@ class IterativeSolverBase : public SparseSolverBase eigen_assert(m_isInitialized && "IterativeSolverBase is not initialized."); return m_info; } - + /** \internal */ template - void _solve_impl(const Rhs& b, SparseMatrixBase &aDest) const + void _solve_impl(const Rhs &b, SparseMatrixBase &aDest) const { - eigen_assert(rows()==b.rows()); - + eigen_assert(rows() == b.rows()); + Index rhsCols = b.cols(); Index size = b.rows(); - DestDerived& dest(aDest.derived()); + DestDerived &dest(aDest.derived()); typedef typename DestDerived::Scalar DestScalar; - Eigen::Matrix tb(size); - Eigen::Matrix tx(cols()); + Eigen::Matrix tb(size); + Eigen::Matrix tx(cols()); // We do not directly fill dest because sparse expressions have to be free of aliasing issue. - // For non square least-square problems, b and dest might not have the same size whereas they might alias each-other. - typename DestDerived::PlainObject tmp(cols(),rhsCols); - for(Index k=0; k typedef internal::generic_matrix_wrapper MatrixWrapper; typedef typename MatrixWrapper::ActualMatrixType ActualMatrixType; - const ActualMatrixType& matrix() const - { - return m_matrixWrapper.matrix(); - } - - template - void grab(const InputType &A) - { - m_matrixWrapper.grab(A); - } - + const ActualMatrixType &matrix() const { return m_matrixWrapper.matrix(); } + + template void grab(const InputType &A) { m_matrixWrapper.grab(A); } + MatrixWrapper m_matrixWrapper; Preconditioner m_preconditioner; Index m_maxIterations; RealScalar m_tolerance; - + mutable RealScalar m_error; mutable Index m_iterations; mutable ComputationInfo m_info; mutable bool m_analysisIsOk, m_factorizationIsOk; }; -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_ITERATIVE_SOLVER_BASE_H +#endif// EIGEN_ITERATIVE_SOLVER_BASE_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/IterativeLinearSolvers/LeastSquareConjugateGradient.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/IterativeLinearSolvers/LeastSquareConjugateGradient.h index 0aea0e09..03fcbc22 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/IterativeLinearSolvers/LeastSquareConjugateGradient.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/IterativeLinearSolvers/LeastSquareConjugateGradient.h @@ -10,109 +10,108 @@ #ifndef EIGEN_LEAST_SQUARE_CONJUGATE_GRADIENT_H #define EIGEN_LEAST_SQUARE_CONJUGATE_GRADIENT_H -namespace Eigen { +namespace Eigen { namespace internal { -/** \internal Low-level conjugate gradient algorithm for least-square problems - * \param mat The matrix A - * \param rhs The right hand side vector b - * \param x On input and initial solution, on output the computed solution. - * \param precond A preconditioner being able to efficiently solve for an - * approximation of A'Ax=b (regardless of b) - * \param iters On input the max number of iteration, on output the number of performed iterations. - * \param tol_error On input the tolerance error, on output an estimation of the relative error. - */ -template -EIGEN_DONT_INLINE -void least_square_conjugate_gradient(const MatrixType& mat, const Rhs& rhs, Dest& x, - const Preconditioner& precond, Index& iters, - typename Dest::RealScalar& tol_error) -{ - using std::sqrt; - using std::abs; - typedef typename Dest::RealScalar RealScalar; - typedef typename Dest::Scalar Scalar; - typedef Matrix VectorType; - - RealScalar tol = tol_error; - Index maxIters = iters; - - Index m = mat.rows(), n = mat.cols(); - - VectorType residual = rhs - mat * x; - VectorType normal_residual = mat.adjoint() * residual; - - RealScalar rhsNorm2 = (mat.adjoint()*rhs).squaredNorm(); - if(rhsNorm2 == 0) - { - x.setZero(); - iters = 0; - tol_error = 0; - return; - } - RealScalar threshold = tol*tol*rhsNorm2; - RealScalar residualNorm2 = normal_residual.squaredNorm(); - if (residualNorm2 < threshold) + /** \internal Low-level conjugate gradient algorithm for least-square problems + * \param mat The matrix A + * \param rhs The right hand side vector b + * \param x On input and initial solution, on output the computed solution. + * \param precond A preconditioner being able to efficiently solve for an + * approximation of A'Ax=b (regardless of b) + * \param iters On input the max number of iteration, on output the number of performed iterations. + * \param tol_error On input the tolerance error, on output an estimation of the relative error. + */ + template + EIGEN_DONT_INLINE void least_square_conjugate_gradient(const MatrixType &mat, + const Rhs &rhs, + Dest &x, + const Preconditioner &precond, + Index &iters, + typename Dest::RealScalar &tol_error) { - iters = 0; + using std::sqrt; + using std::abs; + typedef typename Dest::RealScalar RealScalar; + typedef typename Dest::Scalar Scalar; + typedef Matrix VectorType; + + RealScalar tol = tol_error; + Index maxIters = iters; + + Index m = mat.rows(), n = mat.cols(); + + VectorType residual = rhs - mat * x; + VectorType normal_residual = mat.adjoint() * residual; + + RealScalar rhsNorm2 = (mat.adjoint() * rhs).squaredNorm(); + if (rhsNorm2 == 0) { + x.setZero(); + iters = 0; + tol_error = 0; + return; + } + RealScalar threshold = tol * tol * rhsNorm2; + RealScalar residualNorm2 = normal_residual.squaredNorm(); + if (residualNorm2 < threshold) { + iters = 0; + tol_error = sqrt(residualNorm2 / rhsNorm2); + return; + } + + VectorType p(n); + p = precond.solve(normal_residual);// initial search direction + + VectorType z(n), tmp(m); + RealScalar absNew = numext::real(normal_residual.dot(p));// the square of the absolute value of r scaled by invM + Index i = 0; + while (i < maxIters) { + tmp.noalias() = mat * p; + + Scalar alpha = absNew / tmp.squaredNorm();// the amount we travel on dir + x += alpha * p;// update solution + residual -= alpha * tmp;// update residual + normal_residual = mat.adjoint() * residual;// update residual of the normal equation + + residualNorm2 = normal_residual.squaredNorm(); + if (residualNorm2 < threshold) break; + + z = precond.solve(normal_residual);// approximately solve for "A'A z = normal_residual" + + RealScalar absOld = absNew; + absNew = numext::real(normal_residual.dot(z));// update the absolute value of r + RealScalar beta = absNew / absOld;// calculate the Gram-Schmidt value used to create the new search direction + p = z + beta * p;// update search direction + i++; + } tol_error = sqrt(residualNorm2 / rhsNorm2); - return; + iters = i; } - - VectorType p(n); - p = precond.solve(normal_residual); // initial search direction - - VectorType z(n), tmp(m); - RealScalar absNew = numext::real(normal_residual.dot(p)); // the square of the absolute value of r scaled by invM - Index i = 0; - while(i < maxIters) - { - tmp.noalias() = mat * p; - - Scalar alpha = absNew / tmp.squaredNorm(); // the amount we travel on dir - x += alpha * p; // update solution - residual -= alpha * tmp; // update residual - normal_residual = mat.adjoint() * residual; // update residual of the normal equation - - residualNorm2 = normal_residual.squaredNorm(); - if(residualNorm2 < threshold) - break; - - z = precond.solve(normal_residual); // approximately solve for "A'A z = normal_residual" - - RealScalar absOld = absNew; - absNew = numext::real(normal_residual.dot(z)); // update the absolute value of r - RealScalar beta = absNew / absOld; // calculate the Gram-Schmidt value used to create the new search direction - p = z + beta * p; // update search direction - i++; - } - tol_error = sqrt(residualNorm2 / rhsNorm2); - iters = i; -} -} +}// namespace internal -template< typename _MatrixType, - typename _Preconditioner = LeastSquareDiagonalPreconditioner > +template> class LeastSquaresConjugateGradient; namespace internal { -template< typename _MatrixType, typename _Preconditioner> -struct traits > -{ - typedef _MatrixType MatrixType; - typedef _Preconditioner Preconditioner; -}; + template + struct traits> + { + typedef _MatrixType MatrixType; + typedef _Preconditioner Preconditioner; + }; -} +}// namespace internal /** \ingroup IterativeLinearSolvers_Module * \brief A conjugate gradient solver for sparse (or dense) least-square problems * * This class allows to solve for A x = b linear problems using an iterative conjugate gradient algorithm. - * The matrix A can be non symmetric and rectangular, but the matrix A' A should be positive-definite to guaranty stability. + * The matrix A can be non symmetric and rectangular, but the matrix A' A should be positive-definite to guaranty + stability. * Otherwise, the SparseLU or SparseQR classes might be preferable. * The matrix A and the vectors x and b can be either dense or sparse. * @@ -120,11 +119,11 @@ struct traits > * \tparam _Preconditioner the type of the preconditioner. Default is LeastSquareDiagonalPreconditioner * * \implsparsesolverconcept - * + * * The maximal number of iterations and tolerance value can be controlled via the setMaxIterations() * and setTolerance() methods. The defaults are the size of the problem for the maximal number of iterations * and NumTraits::epsilon() for the tolerance. - * + * * This class can be used as the direct solver classes. Here is a typical usage example: \code int m=1000000, n = 10000; @@ -139,14 +138,15 @@ struct traits > // update b, and solve again x = lscg.solve(b); \endcode - * + * * By default the iterations start with x=0 as an initial guess of the solution. * One can control the start using the solveWithGuess() method. - * + * * \sa class ConjugateGradient, SparseLU, SparseQR */ -template< typename _MatrixType, typename _Preconditioner> -class LeastSquaresConjugateGradient : public IterativeSolverBase > +template +class LeastSquaresConjugateGradient + : public IterativeSolverBase> { typedef IterativeSolverBase Base; using Base::matrix; @@ -154,6 +154,7 @@ class LeastSquaresConjugateGradient : public IterativeSolverBase - explicit LeastSquaresConjugateGradient(const EigenBase& A) : Base(A.derived()) {} + explicit LeastSquaresConjugateGradient(const EigenBase &A) : Base(A.derived()) + {} ~LeastSquaresConjugateGradient() {} /** \internal */ - template - void _solve_with_guess_impl(const Rhs& b, Dest& x) const + template void _solve_with_guess_impl(const Rhs &b, Dest &x) const { m_iterations = Base::maxIterations(); m_error = Base::m_tolerance; - for(Index j=0; j - void _solve_impl(const MatrixBase& b, Dest& x) const + template void _solve_impl(const MatrixBase &b, Dest &x) const { x.setZero(); - _solve_with_guess_impl(b.derived(),x); + _solve_with_guess_impl(b.derived(), x); } - }; -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_LEAST_SQUARE_CONJUGATE_GRADIENT_H +#endif// EIGEN_LEAST_SQUARE_CONJUGATE_GRADIENT_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/IterativeLinearSolvers/SolveWithGuess.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/IterativeLinearSolvers/SolveWithGuess.h index 0ace4517..a84562e5 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/IterativeLinearSolvers/SolveWithGuess.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/IterativeLinearSolvers/SolveWithGuess.h @@ -13,55 +13,60 @@ namespace Eigen { template class SolveWithGuess; - + /** \class SolveWithGuess - * \ingroup IterativeLinearSolvers_Module - * - * \brief Pseudo expression representing a solving operation - * - * \tparam Decomposition the type of the matrix or decomposion object - * \tparam Rhstype the type of the right-hand side - * - * This class represents an expression of A.solve(B) - * and most of the time this is the only way it is used. - * - */ + * \ingroup IterativeLinearSolvers_Module + * + * \brief Pseudo expression representing a solving operation + * + * \tparam Decomposition the type of the matrix or decomposion object + * \tparam Rhstype the type of the right-hand side + * + * This class represents an expression of A.solve(B) + * and most of the time this is the only way it is used. + * + */ namespace internal { -template -struct traits > - : traits > -{}; + template + struct traits> : traits> + { + }; -} +}// namespace internal template -class SolveWithGuess : public internal::generic_xpr_base, MatrixXpr, typename internal::traits::StorageKind>::type +class SolveWithGuess + : public internal::generic_xpr_base, + MatrixXpr, + typename internal::traits::StorageKind>::type { public: typedef typename internal::traits::Scalar Scalar; typedef typename internal::traits::PlainObject PlainObject; - typedef typename internal::generic_xpr_base, MatrixXpr, typename internal::traits::StorageKind>::type Base; + typedef typename internal::generic_xpr_base, + MatrixXpr, + typename internal::traits::StorageKind>::type Base; typedef typename internal::ref_selector::type Nested; - + SolveWithGuess(const Decomposition &dec, const RhsType &rhs, const GuessType &guess) : m_dec(dec), m_rhs(rhs), m_guess(guess) {} - + EIGEN_DEVICE_FUNC Index rows() const { return m_dec.cols(); } EIGEN_DEVICE_FUNC Index cols() const { return m_rhs.cols(); } - EIGEN_DEVICE_FUNC const Decomposition& dec() const { return m_dec; } - EIGEN_DEVICE_FUNC const RhsType& rhs() const { return m_rhs; } - EIGEN_DEVICE_FUNC const GuessType& guess() const { return m_guess; } + EIGEN_DEVICE_FUNC const Decomposition &dec() const { return m_dec; } + EIGEN_DEVICE_FUNC const RhsType &rhs() const { return m_rhs; } + EIGEN_DEVICE_FUNC const GuessType &guess() const { return m_guess; } protected: const Decomposition &m_dec; - const RhsType &m_rhs; - const GuessType &m_guess; - + const RhsType &m_rhs; + const GuessType &m_guess; + private: Scalar coeff(Index row, Index col) const; Scalar coeff(Index i) const; @@ -69,47 +74,49 @@ class SolveWithGuess : public internal::generic_xpr_base eval into a temporary -template -struct evaluator > - : public evaluator::PlainObject> -{ - typedef SolveWithGuess SolveType; - typedef typename SolveType::PlainObject PlainObject; - typedef evaluator Base; - - evaluator(const SolveType& solve) - : m_result(solve.rows(), solve.cols()) + // Evaluator of SolveWithGuess -> eval into a temporary + template + struct evaluator> + : public evaluator::PlainObject> { - ::new (static_cast(this)) Base(m_result); - m_result = solve.guess(); - solve.dec()._solve_with_guess_impl(solve.rhs(), m_result); - } - -protected: - PlainObject m_result; -}; - -// Specialization for "dst = dec.solveWithGuess(rhs)" -// NOTE we need to specialize it for Dense2Dense to avoid ambiguous specialization error and a Sparse2Sparse specialization must exist somewhere -template -struct Assignment, internal::assign_op, Dense2Dense> -{ - typedef SolveWithGuess SrcXprType; - static void run(DstXprType &dst, const SrcXprType &src, const internal::assign_op &) + typedef SolveWithGuess SolveType; + typedef typename SolveType::PlainObject PlainObject; + typedef evaluator Base; + + evaluator(const SolveType &solve) : m_result(solve.rows(), solve.cols()) + { + ::new (static_cast(this)) Base(m_result); + m_result = solve.guess(); + solve.dec()._solve_with_guess_impl(solve.rhs(), m_result); + } + + protected: + PlainObject m_result; + }; + + // Specialization for "dst = dec.solveWithGuess(rhs)" + // NOTE we need to specialize it for Dense2Dense to avoid ambiguous specialization error and a Sparse2Sparse + // specialization must exist somewhere + template + struct Assignment, + internal::assign_op, + Dense2Dense> { - Index dstRows = src.rows(); - Index dstCols = src.cols(); - if((dst.rows()!=dstRows) || (dst.cols()!=dstCols)) - dst.resize(dstRows, dstCols); - - dst = src.guess(); - src.dec()._solve_with_guess_impl(src.rhs(), dst/*, src.guess()*/); - } -}; + typedef SolveWithGuess SrcXprType; + static void run(DstXprType &dst, const SrcXprType &src, const internal::assign_op &) + { + Index dstRows = src.rows(); + Index dstCols = src.cols(); + if ((dst.rows() != dstRows) || (dst.cols() != dstCols)) dst.resize(dstRows, dstCols); + + dst = src.guess(); + src.dec()._solve_with_guess_impl(src.rhs(), dst /*, src.guess()*/); + } + }; -} // end namepsace internal +}// namespace internal -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_SOLVEWITHGUESS_H +#endif// EIGEN_SOLVEWITHGUESS_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Jacobi/Jacobi.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Jacobi/Jacobi.h index 1998c632..6e4fec1a 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/Jacobi/Jacobi.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/Jacobi/Jacobi.h @@ -11,270 +11,261 @@ #ifndef EIGEN_JACOBI_H #define EIGEN_JACOBI_H -namespace Eigen { +namespace Eigen { /** \ingroup Jacobi_Module - * \jacobi_module - * \class JacobiRotation - * \brief Rotation given by a cosine-sine pair. - * - * This class represents a Jacobi or Givens rotation. - * This is a 2D rotation in the plane \c J of angle \f$ \theta \f$ defined by - * its cosine \c c and sine \c s as follow: - * \f$ J = \left ( \begin{array}{cc} c & \overline s \\ -s & \overline c \end{array} \right ) \f$ - * - * You can apply the respective counter-clockwise rotation to a column vector \c v by - * applying its adjoint on the left: \f$ v = J^* v \f$ that translates to the following Eigen code: - * \code - * v.applyOnTheLeft(J.adjoint()); - * \endcode - * - * \sa MatrixBase::applyOnTheLeft(), MatrixBase::applyOnTheRight() - */ + * \jacobi_module + * \class JacobiRotation + * \brief Rotation given by a cosine-sine pair. + * + * This class represents a Jacobi or Givens rotation. + * This is a 2D rotation in the plane \c J of angle \f$ \theta \f$ defined by + * its cosine \c c and sine \c s as follow: + * \f$ J = \left ( \begin{array}{cc} c & \overline s \\ -s & \overline c \end{array} \right ) \f$ + * + * You can apply the respective counter-clockwise rotation to a column vector \c v by + * applying its adjoint on the left: \f$ v = J^* v \f$ that translates to the following Eigen code: + * \code + * v.applyOnTheLeft(J.adjoint()); + * \endcode + * + * \sa MatrixBase::applyOnTheLeft(), MatrixBase::applyOnTheRight() + */ template class JacobiRotation { - public: - typedef typename NumTraits::Real RealScalar; +public: + typedef typename NumTraits::Real RealScalar; - /** Default constructor without any initialization. */ - JacobiRotation() {} + /** Default constructor without any initialization. */ + JacobiRotation() {} - /** Construct a planar rotation from a cosine-sine pair (\a c, \c s). */ - JacobiRotation(const Scalar& c, const Scalar& s) : m_c(c), m_s(s) {} + /** Construct a planar rotation from a cosine-sine pair (\a c, \c s). */ + JacobiRotation(const Scalar &c, const Scalar &s) : m_c(c), m_s(s) {} - Scalar& c() { return m_c; } - Scalar c() const { return m_c; } - Scalar& s() { return m_s; } - Scalar s() const { return m_s; } + Scalar &c() { return m_c; } + Scalar c() const { return m_c; } + Scalar &s() { return m_s; } + Scalar s() const { return m_s; } - /** Concatenates two planar rotation */ - JacobiRotation operator*(const JacobiRotation& other) - { - using numext::conj; - return JacobiRotation(m_c * other.m_c - conj(m_s) * other.m_s, - conj(m_c * conj(other.m_s) + conj(m_s) * conj(other.m_c))); - } + /** Concatenates two planar rotation */ + JacobiRotation operator*(const JacobiRotation &other) + { + using numext::conj; + return JacobiRotation( + m_c * other.m_c - conj(m_s) * other.m_s, conj(m_c * conj(other.m_s) + conj(m_s) * conj(other.m_c))); + } - /** Returns the transposed transformation */ - JacobiRotation transpose() const { using numext::conj; return JacobiRotation(m_c, -conj(m_s)); } + /** Returns the transposed transformation */ + JacobiRotation transpose() const + { + using numext::conj; + return JacobiRotation(m_c, -conj(m_s)); + } - /** Returns the adjoint transformation */ - JacobiRotation adjoint() const { using numext::conj; return JacobiRotation(conj(m_c), -m_s); } + /** Returns the adjoint transformation */ + JacobiRotation adjoint() const + { + using numext::conj; + return JacobiRotation(conj(m_c), -m_s); + } - template - bool makeJacobi(const MatrixBase&, Index p, Index q); - bool makeJacobi(const RealScalar& x, const Scalar& y, const RealScalar& z); + template bool makeJacobi(const MatrixBase &, Index p, Index q); + bool makeJacobi(const RealScalar &x, const Scalar &y, const RealScalar &z); - void makeGivens(const Scalar& p, const Scalar& q, Scalar* r=0); + void makeGivens(const Scalar &p, const Scalar &q, Scalar *r = 0); - protected: - void makeGivens(const Scalar& p, const Scalar& q, Scalar* r, internal::true_type); - void makeGivens(const Scalar& p, const Scalar& q, Scalar* r, internal::false_type); +protected: + void makeGivens(const Scalar &p, const Scalar &q, Scalar *r, internal::true_type); + void makeGivens(const Scalar &p, const Scalar &q, Scalar *r, internal::false_type); - Scalar m_c, m_s; + Scalar m_c, m_s; }; -/** Makes \c *this as a Jacobi rotation \a J such that applying \a J on both the right and left sides of the selfadjoint 2x2 matrix - * \f$ B = \left ( \begin{array}{cc} x & y \\ \overline y & z \end{array} \right )\f$ yields a diagonal matrix \f$ A = J^* B J \f$ - * - * \sa MatrixBase::makeJacobi(const MatrixBase&, Index, Index), MatrixBase::applyOnTheLeft(), MatrixBase::applyOnTheRight() - */ +/** Makes \c *this as a Jacobi rotation \a J such that applying \a J on both the right and left sides of the selfadjoint + * 2x2 matrix + * \f$ B = \left ( \begin{array}{cc} x & y \\ \overline y & z \end{array} \right )\f$ yields a diagonal matrix \f$ A = + * J^* B J \f$ + * + * \sa MatrixBase::makeJacobi(const MatrixBase&, Index, Index), MatrixBase::applyOnTheLeft(), + * MatrixBase::applyOnTheRight() + */ template -bool JacobiRotation::makeJacobi(const RealScalar& x, const Scalar& y, const RealScalar& z) +bool JacobiRotation::makeJacobi(const RealScalar &x, const Scalar &y, const RealScalar &z) { using std::sqrt; using std::abs; - RealScalar deno = RealScalar(2)*abs(y); - if(deno < (std::numeric_limits::min)()) - { + RealScalar deno = RealScalar(2) * abs(y); + if (deno < (std::numeric_limits::min)()) { m_c = Scalar(1); m_s = Scalar(0); return false; - } - else - { - RealScalar tau = (x-z)/deno; + } else { + RealScalar tau = (x - z) / deno; RealScalar w = sqrt(numext::abs2(tau) + RealScalar(1)); RealScalar t; - if(tau>RealScalar(0)) - { + if (tau > RealScalar(0)) { t = RealScalar(1) / (tau + w); - } - else - { + } else { t = RealScalar(1) / (tau - w); } RealScalar sign_t = t > RealScalar(0) ? RealScalar(1) : RealScalar(-1); - RealScalar n = RealScalar(1) / sqrt(numext::abs2(t)+RealScalar(1)); - m_s = - sign_t * (numext::conj(y) / abs(y)) * abs(t) * n; + RealScalar n = RealScalar(1) / sqrt(numext::abs2(t) + RealScalar(1)); + m_s = -sign_t * (numext::conj(y) / abs(y)) * abs(t) * n; m_c = n; return true; } } -/** Makes \c *this as a Jacobi rotation \c J such that applying \a J on both the right and left sides of the 2x2 selfadjoint matrix - * \f$ B = \left ( \begin{array}{cc} \text{this}_{pp} & \text{this}_{pq} \\ (\text{this}_{pq})^* & \text{this}_{qq} \end{array} \right )\f$ yields - * a diagonal matrix \f$ A = J^* B J \f$ - * - * Example: \include Jacobi_makeJacobi.cpp - * Output: \verbinclude Jacobi_makeJacobi.out - * - * \sa JacobiRotation::makeJacobi(RealScalar, Scalar, RealScalar), MatrixBase::applyOnTheLeft(), MatrixBase::applyOnTheRight() - */ +/** Makes \c *this as a Jacobi rotation \c J such that applying \a J on both the right and left sides of the 2x2 + * selfadjoint matrix + * \f$ B = \left ( \begin{array}{cc} \text{this}_{pp} & \text{this}_{pq} \\ (\text{this}_{pq})^* & \text{this}_{qq} + * \end{array} \right )\f$ yields a diagonal matrix \f$ A = J^* B J \f$ + * + * Example: \include Jacobi_makeJacobi.cpp + * Output: \verbinclude Jacobi_makeJacobi.out + * + * \sa JacobiRotation::makeJacobi(RealScalar, Scalar, RealScalar), MatrixBase::applyOnTheLeft(), + * MatrixBase::applyOnTheRight() + */ template template -inline bool JacobiRotation::makeJacobi(const MatrixBase& m, Index p, Index q) +inline bool JacobiRotation::makeJacobi(const MatrixBase &m, Index p, Index q) { - return makeJacobi(numext::real(m.coeff(p,p)), m.coeff(p,q), numext::real(m.coeff(q,q))); + return makeJacobi(numext::real(m.coeff(p, p)), m.coeff(p, q), numext::real(m.coeff(q, q))); } /** Makes \c *this as a Givens rotation \c G such that applying \f$ G^* \f$ to the left of the vector - * \f$ V = \left ( \begin{array}{c} p \\ q \end{array} \right )\f$ yields: - * \f$ G^* V = \left ( \begin{array}{c} r \\ 0 \end{array} \right )\f$. - * - * The value of \a r is returned if \a r is not null (the default is null). - * Also note that G is built such that the cosine is always real. - * - * Example: \include Jacobi_makeGivens.cpp - * Output: \verbinclude Jacobi_makeGivens.out - * - * This function implements the continuous Givens rotation generation algorithm - * found in Anderson (2000), Discontinuous Plane Rotations and the Symmetric Eigenvalue Problem. - * LAPACK Working Note 150, University of Tennessee, UT-CS-00-454, December 4, 2000. - * - * \sa MatrixBase::applyOnTheLeft(), MatrixBase::applyOnTheRight() - */ -template -void JacobiRotation::makeGivens(const Scalar& p, const Scalar& q, Scalar* r) + * \f$ V = \left ( \begin{array}{c} p \\ q \end{array} \right )\f$ yields: + * \f$ G^* V = \left ( \begin{array}{c} r \\ 0 \end{array} \right )\f$. + * + * The value of \a r is returned if \a r is not null (the default is null). + * Also note that G is built such that the cosine is always real. + * + * Example: \include Jacobi_makeGivens.cpp + * Output: \verbinclude Jacobi_makeGivens.out + * + * This function implements the continuous Givens rotation generation algorithm + * found in Anderson (2000), Discontinuous Plane Rotations and the Symmetric Eigenvalue Problem. + * LAPACK Working Note 150, University of Tennessee, UT-CS-00-454, December 4, 2000. + * + * \sa MatrixBase::applyOnTheLeft(), MatrixBase::applyOnTheRight() + */ +template void JacobiRotation::makeGivens(const Scalar &p, const Scalar &q, Scalar *r) { - makeGivens(p, q, r, typename internal::conditional::IsComplex, internal::true_type, internal::false_type>::type()); + makeGivens(p, + q, + r, + typename internal::conditional::IsComplex, internal::true_type, internal::false_type>::type()); } // specialization for complexes template -void JacobiRotation::makeGivens(const Scalar& p, const Scalar& q, Scalar* r, internal::true_type) +void JacobiRotation::makeGivens(const Scalar &p, const Scalar &q, Scalar *r, internal::true_type) { using std::sqrt; using std::abs; using numext::conj; - - if(q==Scalar(0)) - { - m_c = numext::real(p)<0 ? Scalar(-1) : Scalar(1); + + if (q == Scalar(0)) { + m_c = numext::real(p) < 0 ? Scalar(-1) : Scalar(1); m_s = 0; - if(r) *r = m_c * p; - } - else if(p==Scalar(0)) - { + if (r) *r = m_c * p; + } else if (p == Scalar(0)) { m_c = 0; - m_s = -q/abs(q); - if(r) *r = abs(q); - } - else - { + m_s = -q / abs(q); + if (r) *r = abs(q); + } else { RealScalar p1 = numext::norm1(p); RealScalar q1 = numext::norm1(q); - if(p1>=q1) - { + if (p1 >= q1) { Scalar ps = p / p1; RealScalar p2 = numext::abs2(ps); Scalar qs = q / p1; RealScalar q2 = numext::abs2(qs); - RealScalar u = sqrt(RealScalar(1) + q2/p2); - if(numext::real(p) -void JacobiRotation::makeGivens(const Scalar& p, const Scalar& q, Scalar* r, internal::false_type) +void JacobiRotation::makeGivens(const Scalar &p, const Scalar &q, Scalar *r, internal::false_type) { using std::sqrt; using std::abs; - if(q==Scalar(0)) - { - m_c = p abs(q)) - { - Scalar t = q/p; + m_s = q < Scalar(0) ? Scalar(1) : Scalar(-1); + if (r) *r = abs(q); + } else if (abs(p) > abs(q)) { + Scalar t = q / p; Scalar u = sqrt(Scalar(1) + numext::abs2(t)); - if(p -void apply_rotation_in_the_plane(DenseBase& xpr_x, DenseBase& xpr_y, const JacobiRotation& j); -} + /** \jacobi_module + * Applies the clock wise 2D rotation \a j to the set of 2D vectors of cordinates \a x and \a y: + * \f$ \left ( \begin{array}{cc} x \\ y \end{array} \right ) = J \left ( \begin{array}{cc} x \\ y \end{array} \right + * ) \f$ + * + * \sa MatrixBase::applyOnTheLeft(), MatrixBase::applyOnTheRight() + */ + template + void apply_rotation_in_the_plane(DenseBase &xpr_x, + DenseBase &xpr_y, + const JacobiRotation &j); +}// namespace internal /** \jacobi_module - * Applies the rotation in the plane \a j to the rows \a p and \a q of \c *this, i.e., it computes B = J * B, - * with \f$ B = \left ( \begin{array}{cc} \text{*this.row}(p) \\ \text{*this.row}(q) \end{array} \right ) \f$. - * - * \sa class JacobiRotation, MatrixBase::applyOnTheRight(), internal::apply_rotation_in_the_plane() - */ + * Applies the rotation in the plane \a j to the rows \a p and \a q of \c *this, i.e., it computes B = J * B, + * with \f$ B = \left ( \begin{array}{cc} \text{*this.row}(p) \\ \text{*this.row}(q) \end{array} \right ) \f$. + * + * \sa class JacobiRotation, MatrixBase::applyOnTheRight(), internal::apply_rotation_in_the_plane() + */ template template -inline void MatrixBase::applyOnTheLeft(Index p, Index q, const JacobiRotation& j) +inline void MatrixBase::applyOnTheLeft(Index p, Index q, const JacobiRotation &j) { RowXpr x(this->row(p)); RowXpr y(this->row(q)); @@ -282,14 +273,14 @@ inline void MatrixBase::applyOnTheLeft(Index p, Index q, const JacobiRo } /** \ingroup Jacobi_Module - * Applies the rotation in the plane \a j to the columns \a p and \a q of \c *this, i.e., it computes B = B * J - * with \f$ B = \left ( \begin{array}{cc} \text{*this.col}(p) & \text{*this.col}(q) \end{array} \right ) \f$. - * - * \sa class JacobiRotation, MatrixBase::applyOnTheLeft(), internal::apply_rotation_in_the_plane() - */ + * Applies the rotation in the plane \a j to the columns \a p and \a q of \c *this, i.e., it computes B = B * J + * with \f$ B = \left ( \begin{array}{cc} \text{*this.col}(p) & \text{*this.col}(q) \end{array} \right ) \f$. + * + * \sa class JacobiRotation, MatrixBase::applyOnTheLeft(), internal::apply_rotation_in_the_plane() + */ template template -inline void MatrixBase::applyOnTheRight(Index p, Index q, const JacobiRotation& j) +inline void MatrixBase::applyOnTheRight(Index p, Index q, const JacobiRotation &j) { ColXpr x(this->col(p)); ColXpr y(this->col(q)); @@ -298,165 +289,154 @@ inline void MatrixBase::applyOnTheRight(Index p, Index q, const JacobiR namespace internal { -template -struct apply_rotation_in_the_plane_selector -{ - static inline void run(Scalar *x, Index incrx, Scalar *y, Index incry, Index size, OtherScalar c, OtherScalar s) + template + struct apply_rotation_in_the_plane_selector { - for(Index i=0; i -struct apply_rotation_in_the_plane_selector -{ - static inline void run(Scalar *x, Index incrx, Scalar *y, Index incry, Index size, OtherScalar c, OtherScalar s) + }; + + template + struct apply_rotation_in_the_plane_selector { - enum { - PacketSize = packet_traits::size, - OtherPacketSize = packet_traits::size - }; - typedef typename packet_traits::type Packet; - typedef typename packet_traits::type OtherPacket; - - /*** dynamic-size vectorized paths ***/ - if(SizeAtCompileTime == Dynamic && ((incrx==1 && incry==1) || PacketSize == 1)) + static inline void run(Scalar *x, Index incrx, Scalar *y, Index incry, Index size, OtherScalar c, OtherScalar s) { - // both vectors are sequentially stored in memory => vectorization - enum { Peeling = 2 }; - - Index alignedStart = internal::first_default_aligned(y, size); - Index alignedEnd = alignedStart + ((size-alignedStart)/PacketSize)*PacketSize; + enum { PacketSize = packet_traits::size, OtherPacketSize = packet_traits::size }; + typedef typename packet_traits::type Packet; + typedef typename packet_traits::type OtherPacket; + + /*** dynamic-size vectorized paths ***/ + if (SizeAtCompileTime == Dynamic && ((incrx == 1 && incry == 1) || PacketSize == 1)) { + // both vectors are sequentially stored in memory => vectorization + enum { Peeling = 2 }; + + Index alignedStart = internal::first_default_aligned(y, size); + Index alignedEnd = alignedStart + ((size - alignedStart) / PacketSize) * PacketSize; + + const OtherPacket pc = pset1(c); + const OtherPacket ps = pset1(s); + conj_helper::IsComplex, false> pcj; + conj_helper pm; + + for (Index i = 0; i < alignedStart; ++i) { + Scalar xi = x[i]; + Scalar yi = y[i]; + x[i] = c * xi + numext::conj(s) * yi; + y[i] = -s * xi + numext::conj(c) * yi; + } - const OtherPacket pc = pset1(c); - const OtherPacket ps = pset1(s); - conj_helper::IsComplex,false> pcj; - conj_helper pm; + Scalar *EIGEN_RESTRICT px = x + alignedStart; + Scalar *EIGEN_RESTRICT py = y + alignedStart; + + if (internal::first_default_aligned(x, size) == alignedStart) { + for (Index i = alignedStart; i < alignedEnd; i += PacketSize) { + Packet xi = pload(px); + Packet yi = pload(py); + pstore(px, padd(pm.pmul(pc, xi), pcj.pmul(ps, yi))); + pstore(py, psub(pcj.pmul(pc, yi), pm.pmul(ps, xi))); + px += PacketSize; + py += PacketSize; + } + } else { + Index peelingEnd = alignedStart + ((size - alignedStart) / (Peeling * PacketSize)) * (Peeling * PacketSize); + for (Index i = alignedStart; i < peelingEnd; i += Peeling * PacketSize) { + Packet xi = ploadu(px); + Packet xi1 = ploadu(px + PacketSize); + Packet yi = pload(py); + Packet yi1 = pload(py + PacketSize); + pstoreu(px, padd(pm.pmul(pc, xi), pcj.pmul(ps, yi))); + pstoreu(px + PacketSize, padd(pm.pmul(pc, xi1), pcj.pmul(ps, yi1))); + pstore(py, psub(pcj.pmul(pc, yi), pm.pmul(ps, xi))); + pstore(py + PacketSize, psub(pcj.pmul(pc, yi1), pm.pmul(ps, xi1))); + px += Peeling * PacketSize; + py += Peeling * PacketSize; + } + if (alignedEnd != peelingEnd) { + Packet xi = ploadu(x + peelingEnd); + Packet yi = pload(y + peelingEnd); + pstoreu(x + peelingEnd, padd(pm.pmul(pc, xi), pcj.pmul(ps, yi))); + pstore(y + peelingEnd, psub(pcj.pmul(pc, yi), pm.pmul(ps, xi))); + } + } - for(Index i=0; i 0)// FIXME should be compared to the required alignment { - for(Index i=alignedStart; i(c); + const OtherPacket ps = pset1(s); + conj_helper::IsComplex, false> pcj; + conj_helper pm; + Scalar *EIGEN_RESTRICT px = x; + Scalar *EIGEN_RESTRICT py = y; + for (Index i = 0; i < size; i += PacketSize) { Packet xi = pload(px); Packet yi = pload(py); - pstore(px, padd(pm.pmul(pc,xi),pcj.pmul(ps,yi))); - pstore(py, psub(pcj.pmul(pc,yi),pm.pmul(ps,xi))); + pstore(px, padd(pm.pmul(pc, xi), pcj.pmul(ps, yi))); + pstore(py, psub(pcj.pmul(pc, yi), pm.pmul(ps, xi))); px += PacketSize; py += PacketSize; } } - else - { - Index peelingEnd = alignedStart + ((size-alignedStart)/(Peeling*PacketSize))*(Peeling*PacketSize); - for(Index i=alignedStart; i(px); - Packet xi1 = ploadu(px+PacketSize); - Packet yi = pload (py); - Packet yi1 = pload (py+PacketSize); - pstoreu(px, padd(pm.pmul(pc,xi),pcj.pmul(ps,yi))); - pstoreu(px+PacketSize, padd(pm.pmul(pc,xi1),pcj.pmul(ps,yi1))); - pstore (py, psub(pcj.pmul(pc,yi),pm.pmul(ps,xi))); - pstore (py+PacketSize, psub(pcj.pmul(pc,yi1),pm.pmul(ps,xi1))); - px += Peeling*PacketSize; - py += Peeling*PacketSize; - } - if(alignedEnd!=peelingEnd) - { - Packet xi = ploadu(x+peelingEnd); - Packet yi = pload (y+peelingEnd); - pstoreu(x+peelingEnd, padd(pm.pmul(pc,xi),pcj.pmul(ps,yi))); - pstore (y+peelingEnd, psub(pcj.pmul(pc,yi),pm.pmul(ps,xi))); - } - } - - for(Index i=alignedEnd; i0) // FIXME should be compared to the required alignment - { - const OtherPacket pc = pset1(c); - const OtherPacket ps = pset1(s); - conj_helper::IsComplex,false> pcj; - conj_helper pm; - Scalar* EIGEN_RESTRICT px = x; - Scalar* EIGEN_RESTRICT py = y; - for(Index i=0; i(px); - Packet yi = pload(py); - pstore(px, padd(pm.pmul(pc,xi),pcj.pmul(ps,yi))); - pstore(py, psub(pcj.pmul(pc,yi),pm.pmul(ps,xi))); - px += PacketSize; - py += PacketSize; + /*** non-vectorized path ***/ + else { + apply_rotation_in_the_plane_selector::run( + x, incrx, y, incry, size, c, s); } } + }; - /*** non-vectorized path ***/ - else - { - apply_rotation_in_the_plane_selector::run(x,incrx,y,incry,size,c,s); - } + template + void /*EIGEN_DONT_INLINE*/ apply_rotation_in_the_plane(DenseBase &xpr_x, + DenseBase &xpr_y, + const JacobiRotation &j) + { + typedef typename VectorX::Scalar Scalar; + const bool Vectorizable = (VectorX::Flags & VectorY::Flags & PacketAccessBit) + && (int(packet_traits::size) == int(packet_traits::size)); + + eigen_assert(xpr_x.size() == xpr_y.size()); + Index size = xpr_x.size(); + Index incrx = xpr_x.derived().innerStride(); + Index incry = xpr_y.derived().innerStride(); + + Scalar *EIGEN_RESTRICT x = &xpr_x.derived().coeffRef(0); + Scalar *EIGEN_RESTRICT y = &xpr_y.derived().coeffRef(0); + + OtherScalar c = j.c(); + OtherScalar s = j.s(); + if (c == OtherScalar(1) && s == OtherScalar(0)) return; + + apply_rotation_in_the_plane_selector::Alignment, evaluator::Alignment), + Vectorizable>::run(x, incrx, y, incry, size, c, s); } -}; - -template -void /*EIGEN_DONT_INLINE*/ apply_rotation_in_the_plane(DenseBase& xpr_x, DenseBase& xpr_y, const JacobiRotation& j) -{ - typedef typename VectorX::Scalar Scalar; - const bool Vectorizable = (VectorX::Flags & VectorY::Flags & PacketAccessBit) - && (int(packet_traits::size) == int(packet_traits::size)); - - eigen_assert(xpr_x.size() == xpr_y.size()); - Index size = xpr_x.size(); - Index incrx = xpr_x.derived().innerStride(); - Index incry = xpr_y.derived().innerStride(); - - Scalar* EIGEN_RESTRICT x = &xpr_x.derived().coeffRef(0); - Scalar* EIGEN_RESTRICT y = &xpr_y.derived().coeffRef(0); - - OtherScalar c = j.c(); - OtherScalar s = j.s(); - if (c==OtherScalar(1) && s==OtherScalar(0)) - return; - - apply_rotation_in_the_plane_selector< - Scalar,OtherScalar, - VectorX::SizeAtCompileTime, - EIGEN_PLAIN_ENUM_MIN(evaluator::Alignment, evaluator::Alignment), - Vectorizable>::run(x,incrx,y,incry,size,c,s); -} -} // end namespace internal +}// end namespace internal -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_JACOBI_H +#endif// EIGEN_JACOBI_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/LU/Determinant.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/LU/Determinant.h index d6a3c1e5..2ee65a73 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/LU/Determinant.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/LU/Determinant.h @@ -10,92 +10,78 @@ #ifndef EIGEN_DETERMINANT_H #define EIGEN_DETERMINANT_H -namespace Eigen { +namespace Eigen { namespace internal { -template -inline const typename Derived::Scalar bruteforce_det3_helper -(const MatrixBase& matrix, int a, int b, int c) -{ - return matrix.coeff(0,a) - * (matrix.coeff(1,b) * matrix.coeff(2,c) - matrix.coeff(1,c) * matrix.coeff(2,b)); -} - -template -const typename Derived::Scalar bruteforce_det4_helper -(const MatrixBase& matrix, int j, int k, int m, int n) -{ - return (matrix.coeff(j,0) * matrix.coeff(k,1) - matrix.coeff(k,0) * matrix.coeff(j,1)) - * (matrix.coeff(m,2) * matrix.coeff(n,3) - matrix.coeff(n,2) * matrix.coeff(m,3)); -} - -template struct determinant_impl -{ - static inline typename traits::Scalar run(const Derived& m) + template + inline const typename Derived::Scalar bruteforce_det3_helper(const MatrixBase &matrix, int a, int b, int c) { - if(Derived::ColsAtCompileTime==Dynamic && m.rows()==0) - return typename traits::Scalar(1); - return m.partialPivLu().determinant(); + return matrix.coeff(0, a) * (matrix.coeff(1, b) * matrix.coeff(2, c) - matrix.coeff(1, c) * matrix.coeff(2, b)); } -}; -template struct determinant_impl -{ - static inline typename traits::Scalar run(const Derived& m) + template + const typename Derived::Scalar bruteforce_det4_helper(const MatrixBase &matrix, int j, int k, int m, int n) { - return m.coeff(0,0); + return (matrix.coeff(j, 0) * matrix.coeff(k, 1) - matrix.coeff(k, 0) * matrix.coeff(j, 1)) + * (matrix.coeff(m, 2) * matrix.coeff(n, 3) - matrix.coeff(n, 2) * matrix.coeff(m, 3)); } -}; -template struct determinant_impl -{ - static inline typename traits::Scalar run(const Derived& m) + template struct determinant_impl { - return m.coeff(0,0) * m.coeff(1,1) - m.coeff(1,0) * m.coeff(0,1); - } -}; + static inline typename traits::Scalar run(const Derived &m) + { + if (Derived::ColsAtCompileTime == Dynamic && m.rows() == 0) return typename traits::Scalar(1); + return m.partialPivLu().determinant(); + } + }; -template struct determinant_impl -{ - static inline typename traits::Scalar run(const Derived& m) + template struct determinant_impl { - return bruteforce_det3_helper(m,0,1,2) - - bruteforce_det3_helper(m,1,0,2) - + bruteforce_det3_helper(m,2,0,1); - } -}; + static inline typename traits::Scalar run(const Derived &m) { return m.coeff(0, 0); } + }; -template struct determinant_impl -{ - static typename traits::Scalar run(const Derived& m) + template struct determinant_impl { - // trick by Martin Costabel to compute 4x4 det with only 30 muls - return bruteforce_det4_helper(m,0,1,2,3) - - bruteforce_det4_helper(m,0,2,1,3) - + bruteforce_det4_helper(m,0,3,1,2) - + bruteforce_det4_helper(m,1,2,0,3) - - bruteforce_det4_helper(m,1,3,0,2) - + bruteforce_det4_helper(m,2,3,0,1); - } -}; + static inline typename traits::Scalar run(const Derived &m) + { + return m.coeff(0, 0) * m.coeff(1, 1) - m.coeff(1, 0) * m.coeff(0, 1); + } + }; + + template struct determinant_impl + { + static inline typename traits::Scalar run(const Derived &m) + { + return bruteforce_det3_helper(m, 0, 1, 2) - bruteforce_det3_helper(m, 1, 0, 2) + + bruteforce_det3_helper(m, 2, 0, 1); + } + }; + + template struct determinant_impl + { + static typename traits::Scalar run(const Derived &m) + { + // trick by Martin Costabel to compute 4x4 det with only 30 muls + return bruteforce_det4_helper(m, 0, 1, 2, 3) - bruteforce_det4_helper(m, 0, 2, 1, 3) + + bruteforce_det4_helper(m, 0, 3, 1, 2) + bruteforce_det4_helper(m, 1, 2, 0, 3) + - bruteforce_det4_helper(m, 1, 3, 0, 2) + bruteforce_det4_helper(m, 2, 3, 0, 1); + } + }; -} // end namespace internal +}// end namespace internal /** \lu_module - * - * \returns the determinant of this matrix - */ -template -inline typename internal::traits::Scalar MatrixBase::determinant() const + * + * \returns the determinant of this matrix + */ +template inline typename internal::traits::Scalar MatrixBase::determinant() const { eigen_assert(rows() == cols()); - typedef typename internal::nested_eval::type Nested; + typedef typename internal::nested_eval::type Nested; return internal::determinant_impl::type>::run(derived()); } -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_DETERMINANT_H +#endif// EIGEN_DETERMINANT_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/LU/FullPivLU.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/LU/FullPivLU.h index 03b6af70..e40a86c0 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/LU/FullPivLU.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/LU/FullPivLU.h @@ -13,486 +13,454 @@ namespace Eigen { namespace internal { -template struct traits > - : traits<_MatrixType> -{ - typedef MatrixXpr XprKind; - typedef SolverStorage StorageKind; - enum { Flags = 0 }; -}; + template struct traits> : traits<_MatrixType> + { + typedef MatrixXpr XprKind; + typedef SolverStorage StorageKind; + enum { Flags = 0 }; + }; -} // end namespace internal +}// end namespace internal /** \ingroup LU_Module - * - * \class FullPivLU - * - * \brief LU decomposition of a matrix with complete pivoting, and related features - * - * \tparam _MatrixType the type of the matrix of which we are computing the LU decomposition - * - * This class represents a LU decomposition of any matrix, with complete pivoting: the matrix A is - * decomposed as \f$ A = P^{-1} L U Q^{-1} \f$ where L is unit-lower-triangular, U is - * upper-triangular, and P and Q are permutation matrices. This is a rank-revealing LU - * decomposition. The eigenvalues (diagonal coefficients) of U are sorted in such a way that any - * zeros are at the end. - * - * This decomposition provides the generic approach to solving systems of linear equations, computing - * the rank, invertibility, inverse, kernel, and determinant. - * - * This LU decomposition is very stable and well tested with large matrices. However there are use cases where the SVD - * decomposition is inherently more stable and/or flexible. For example, when computing the kernel of a matrix, - * working with the SVD allows to select the smallest singular values of the matrix, something that - * the LU decomposition doesn't see. - * - * The data of the LU decomposition can be directly accessed through the methods matrixLU(), - * permutationP(), permutationQ(). - * - * As an exemple, here is how the original matrix can be retrieved: - * \include class_FullPivLU.cpp - * Output: \verbinclude class_FullPivLU.out - * - * This class supports the \link InplaceDecomposition inplace decomposition \endlink mechanism. - * - * \sa MatrixBase::fullPivLu(), MatrixBase::determinant(), MatrixBase::inverse() - */ -template class FullPivLU - : public SolverBase > + * + * \class FullPivLU + * + * \brief LU decomposition of a matrix with complete pivoting, and related features + * + * \tparam _MatrixType the type of the matrix of which we are computing the LU decomposition + * + * This class represents a LU decomposition of any matrix, with complete pivoting: the matrix A is + * decomposed as \f$ A = P^{-1} L U Q^{-1} \f$ where L is unit-lower-triangular, U is + * upper-triangular, and P and Q are permutation matrices. This is a rank-revealing LU + * decomposition. The eigenvalues (diagonal coefficients) of U are sorted in such a way that any + * zeros are at the end. + * + * This decomposition provides the generic approach to solving systems of linear equations, computing + * the rank, invertibility, inverse, kernel, and determinant. + * + * This LU decomposition is very stable and well tested with large matrices. However there are use cases where the SVD + * decomposition is inherently more stable and/or flexible. For example, when computing the kernel of a matrix, + * working with the SVD allows to select the smallest singular values of the matrix, something that + * the LU decomposition doesn't see. + * + * The data of the LU decomposition can be directly accessed through the methods matrixLU(), + * permutationP(), permutationQ(). + * + * As an exemple, here is how the original matrix can be retrieved: + * \include class_FullPivLU.cpp + * Output: \verbinclude class_FullPivLU.out + * + * This class supports the \link InplaceDecomposition inplace decomposition \endlink mechanism. + * + * \sa MatrixBase::fullPivLu(), MatrixBase::determinant(), MatrixBase::inverse() + */ +template class FullPivLU : public SolverBase> { - public: - typedef _MatrixType MatrixType; - typedef SolverBase Base; - - EIGEN_GENERIC_PUBLIC_INTERFACE(FullPivLU) - // FIXME StorageIndex defined in EIGEN_GENERIC_PUBLIC_INTERFACE should be int - enum { - MaxRowsAtCompileTime = MatrixType::MaxRowsAtCompileTime, - MaxColsAtCompileTime = MatrixType::MaxColsAtCompileTime - }; - typedef typename internal::plain_row_type::type IntRowVectorType; - typedef typename internal::plain_col_type::type IntColVectorType; - typedef PermutationMatrix PermutationQType; - typedef PermutationMatrix PermutationPType; - typedef typename MatrixType::PlainObject PlainObject; - - /** - * \brief Default Constructor. - * - * The default constructor is useful in cases in which the user intends to - * perform decompositions via LU::compute(const MatrixType&). - */ - FullPivLU(); - - /** \brief Default Constructor with memory preallocation - * - * Like the default constructor but with preallocation of the internal data - * according to the specified problem \a size. - * \sa FullPivLU() - */ - FullPivLU(Index rows, Index cols); - - /** Constructor. - * - * \param matrix the matrix of which to compute the LU decomposition. - * It is required to be nonzero. - */ - template - explicit FullPivLU(const EigenBase& matrix); - - /** \brief Constructs a LU factorization from a given matrix - * - * This overloaded constructor is provided for \link InplaceDecomposition inplace decomposition \endlink when \c MatrixType is a Eigen::Ref. - * - * \sa FullPivLU(const EigenBase&) - */ - template - explicit FullPivLU(EigenBase& matrix); - - /** Computes the LU decomposition of the given matrix. - * - * \param matrix the matrix of which to compute the LU decomposition. - * It is required to be nonzero. - * - * \returns a reference to *this - */ - template - FullPivLU& compute(const EigenBase& matrix) { - m_lu = matrix.derived(); - computeInPlace(); - return *this; - } +public: + typedef _MatrixType MatrixType; + typedef SolverBase Base; + + EIGEN_GENERIC_PUBLIC_INTERFACE(FullPivLU) + // FIXME StorageIndex defined in EIGEN_GENERIC_PUBLIC_INTERFACE should be int + enum { + MaxRowsAtCompileTime = MatrixType::MaxRowsAtCompileTime, + MaxColsAtCompileTime = MatrixType::MaxColsAtCompileTime + }; + typedef typename internal::plain_row_type::type IntRowVectorType; + typedef typename internal::plain_col_type::type IntColVectorType; + typedef PermutationMatrix PermutationQType; + typedef PermutationMatrix PermutationPType; + typedef typename MatrixType::PlainObject PlainObject; + + /** + * \brief Default Constructor. + * + * The default constructor is useful in cases in which the user intends to + * perform decompositions via LU::compute(const MatrixType&). + */ + FullPivLU(); - /** \returns the LU decomposition matrix: the upper-triangular part is U, the - * unit-lower-triangular part is L (at least for square matrices; in the non-square - * case, special care is needed, see the documentation of class FullPivLU). - * - * \sa matrixL(), matrixU() - */ - inline const MatrixType& matrixLU() const - { - eigen_assert(m_isInitialized && "LU is not initialized."); - return m_lu; - } + /** \brief Default Constructor with memory preallocation + * + * Like the default constructor but with preallocation of the internal data + * according to the specified problem \a size. + * \sa FullPivLU() + */ + FullPivLU(Index rows, Index cols); - /** \returns the number of nonzero pivots in the LU decomposition. - * Here nonzero is meant in the exact sense, not in a fuzzy sense. - * So that notion isn't really intrinsically interesting, but it is - * still useful when implementing algorithms. - * - * \sa rank() - */ - inline Index nonzeroPivots() const - { - eigen_assert(m_isInitialized && "LU is not initialized."); - return m_nonzero_pivots; - } + /** Constructor. + * + * \param matrix the matrix of which to compute the LU decomposition. + * It is required to be nonzero. + */ + template explicit FullPivLU(const EigenBase &matrix); + + /** \brief Constructs a LU factorization from a given matrix + * + * This overloaded constructor is provided for \link InplaceDecomposition inplace decomposition \endlink when \c + * MatrixType is a Eigen::Ref. + * + * \sa FullPivLU(const EigenBase&) + */ + template explicit FullPivLU(EigenBase &matrix); + + /** Computes the LU decomposition of the given matrix. + * + * \param matrix the matrix of which to compute the LU decomposition. + * It is required to be nonzero. + * + * \returns a reference to *this + */ + template FullPivLU &compute(const EigenBase &matrix) + { + m_lu = matrix.derived(); + computeInPlace(); + return *this; + } - /** \returns the absolute value of the biggest pivot, i.e. the biggest - * diagonal coefficient of U. - */ - RealScalar maxPivot() const { return m_maxpivot; } + /** \returns the LU decomposition matrix: the upper-triangular part is U, the + * unit-lower-triangular part is L (at least for square matrices; in the non-square + * case, special care is needed, see the documentation of class FullPivLU). + * + * \sa matrixL(), matrixU() + */ + inline const MatrixType &matrixLU() const + { + eigen_assert(m_isInitialized && "LU is not initialized."); + return m_lu; + } - /** \returns the permutation matrix P - * - * \sa permutationQ() - */ - EIGEN_DEVICE_FUNC inline const PermutationPType& permutationP() const - { - eigen_assert(m_isInitialized && "LU is not initialized."); - return m_p; - } + /** \returns the number of nonzero pivots in the LU decomposition. + * Here nonzero is meant in the exact sense, not in a fuzzy sense. + * So that notion isn't really intrinsically interesting, but it is + * still useful when implementing algorithms. + * + * \sa rank() + */ + inline Index nonzeroPivots() const + { + eigen_assert(m_isInitialized && "LU is not initialized."); + return m_nonzero_pivots; + } - /** \returns the permutation matrix Q - * - * \sa permutationP() - */ - inline const PermutationQType& permutationQ() const - { - eigen_assert(m_isInitialized && "LU is not initialized."); - return m_q; - } + /** \returns the absolute value of the biggest pivot, i.e. the biggest + * diagonal coefficient of U. + */ + RealScalar maxPivot() const { return m_maxpivot; } - /** \returns the kernel of the matrix, also called its null-space. The columns of the returned matrix - * will form a basis of the kernel. - * - * \note If the kernel has dimension zero, then the returned matrix is a column-vector filled with zeros. - * - * \note This method has to determine which pivots should be considered nonzero. - * For that, it uses the threshold value that you can control by calling - * setThreshold(const RealScalar&). - * - * Example: \include FullPivLU_kernel.cpp - * Output: \verbinclude FullPivLU_kernel.out - * - * \sa image() - */ - inline const internal::kernel_retval kernel() const - { - eigen_assert(m_isInitialized && "LU is not initialized."); - return internal::kernel_retval(*this); - } + /** \returns the permutation matrix P + * + * \sa permutationQ() + */ + EIGEN_DEVICE_FUNC inline const PermutationPType &permutationP() const + { + eigen_assert(m_isInitialized && "LU is not initialized."); + return m_p; + } - /** \returns the image of the matrix, also called its column-space. The columns of the returned matrix - * will form a basis of the image (column-space). - * - * \param originalMatrix the original matrix, of which *this is the LU decomposition. - * The reason why it is needed to pass it here, is that this allows - * a large optimization, as otherwise this method would need to reconstruct it - * from the LU decomposition. - * - * \note If the image has dimension zero, then the returned matrix is a column-vector filled with zeros. - * - * \note This method has to determine which pivots should be considered nonzero. - * For that, it uses the threshold value that you can control by calling - * setThreshold(const RealScalar&). - * - * Example: \include FullPivLU_image.cpp - * Output: \verbinclude FullPivLU_image.out - * - * \sa kernel() - */ - inline const internal::image_retval - image(const MatrixType& originalMatrix) const - { - eigen_assert(m_isInitialized && "LU is not initialized."); - return internal::image_retval(*this, originalMatrix); - } + /** \returns the permutation matrix Q + * + * \sa permutationP() + */ + inline const PermutationQType &permutationQ() const + { + eigen_assert(m_isInitialized && "LU is not initialized."); + return m_q; + } - /** \return a solution x to the equation Ax=b, where A is the matrix of which - * *this is the LU decomposition. - * - * \param b the right-hand-side of the equation to solve. Can be a vector or a matrix, - * the only requirement in order for the equation to make sense is that - * b.rows()==A.rows(), where A is the matrix of which *this is the LU decomposition. - * - * \returns a solution. - * - * \note_about_checking_solutions - * - * \note_about_arbitrary_choice_of_solution - * \note_about_using_kernel_to_study_multiple_solutions - * - * Example: \include FullPivLU_solve.cpp - * Output: \verbinclude FullPivLU_solve.out - * - * \sa TriangularView::solve(), kernel(), inverse() - */ - // FIXME this is a copy-paste of the base-class member to add the isInitialized assertion. - template - inline const Solve - solve(const MatrixBase& b) const - { - eigen_assert(m_isInitialized && "LU is not initialized."); - return Solve(*this, b.derived()); - } + /** \returns the kernel of the matrix, also called its null-space. The columns of the returned matrix + * will form a basis of the kernel. + * + * \note If the kernel has dimension zero, then the returned matrix is a column-vector filled with zeros. + * + * \note This method has to determine which pivots should be considered nonzero. + * For that, it uses the threshold value that you can control by calling + * setThreshold(const RealScalar&). + * + * Example: \include FullPivLU_kernel.cpp + * Output: \verbinclude FullPivLU_kernel.out + * + * \sa image() + */ + inline const internal::kernel_retval kernel() const + { + eigen_assert(m_isInitialized && "LU is not initialized."); + return internal::kernel_retval(*this); + } - /** \returns an estimate of the reciprocal condition number of the matrix of which \c *this is - the LU decomposition. - */ - inline RealScalar rcond() const - { - eigen_assert(m_isInitialized && "PartialPivLU is not initialized."); - return internal::rcond_estimate_helper(m_l1_norm, *this); - } + /** \returns the image of the matrix, also called its column-space. The columns of the returned matrix + * will form a basis of the image (column-space). + * + * \param originalMatrix the original matrix, of which *this is the LU decomposition. + * The reason why it is needed to pass it here, is that this allows + * a large optimization, as otherwise this method would need to reconstruct it + * from the LU decomposition. + * + * \note If the image has dimension zero, then the returned matrix is a column-vector filled with zeros. + * + * \note This method has to determine which pivots should be considered nonzero. + * For that, it uses the threshold value that you can control by calling + * setThreshold(const RealScalar&). + * + * Example: \include FullPivLU_image.cpp + * Output: \verbinclude FullPivLU_image.out + * + * \sa kernel() + */ + inline const internal::image_retval image(const MatrixType &originalMatrix) const + { + eigen_assert(m_isInitialized && "LU is not initialized."); + return internal::image_retval(*this, originalMatrix); + } - /** \returns the determinant of the matrix of which - * *this is the LU decomposition. It has only linear complexity - * (that is, O(n) where n is the dimension of the square matrix) - * as the LU decomposition has already been computed. - * - * \note This is only for square matrices. - * - * \note For fixed-size matrices of size up to 4, MatrixBase::determinant() offers - * optimized paths. - * - * \warning a determinant can be very big or small, so for matrices - * of large enough dimension, there is a risk of overflow/underflow. - * - * \sa MatrixBase::determinant() - */ - typename internal::traits::Scalar determinant() const; - - /** Allows to prescribe a threshold to be used by certain methods, such as rank(), - * who need to determine when pivots are to be considered nonzero. This is not used for the - * LU decomposition itself. - * - * When it needs to get the threshold value, Eigen calls threshold(). By default, this - * uses a formula to automatically determine a reasonable threshold. - * Once you have called the present method setThreshold(const RealScalar&), - * your value is used instead. - * - * \param threshold The new value to use as the threshold. - * - * A pivot will be considered nonzero if its absolute value is strictly greater than - * \f$ \vert pivot \vert \leqslant threshold \times \vert maxpivot \vert \f$ - * where maxpivot is the biggest pivot. - * - * If you want to come back to the default behavior, call setThreshold(Default_t) - */ - FullPivLU& setThreshold(const RealScalar& threshold) - { - m_usePrescribedThreshold = true; - m_prescribedThreshold = threshold; - return *this; - } + /** \return a solution x to the equation Ax=b, where A is the matrix of which + * *this is the LU decomposition. + * + * \param b the right-hand-side of the equation to solve. Can be a vector or a matrix, + * the only requirement in order for the equation to make sense is that + * b.rows()==A.rows(), where A is the matrix of which *this is the LU decomposition. + * + * \returns a solution. + * + * \note_about_checking_solutions + * + * \note_about_arbitrary_choice_of_solution + * \note_about_using_kernel_to_study_multiple_solutions + * + * Example: \include FullPivLU_solve.cpp + * Output: \verbinclude FullPivLU_solve.out + * + * \sa TriangularView::solve(), kernel(), inverse() + */ + // FIXME this is a copy-paste of the base-class member to add the isInitialized assertion. + template inline const Solve solve(const MatrixBase &b) const + { + eigen_assert(m_isInitialized && "LU is not initialized."); + return Solve(*this, b.derived()); + } - /** Allows to come back to the default behavior, letting Eigen use its default formula for - * determining the threshold. - * - * You should pass the special object Eigen::Default as parameter here. - * \code lu.setThreshold(Eigen::Default); \endcode - * - * See the documentation of setThreshold(const RealScalar&). - */ - FullPivLU& setThreshold(Default_t) - { - m_usePrescribedThreshold = false; - return *this; - } + /** \returns an estimate of the reciprocal condition number of the matrix of which \c *this is + the LU decomposition. + */ + inline RealScalar rcond() const + { + eigen_assert(m_isInitialized && "PartialPivLU is not initialized."); + return internal::rcond_estimate_helper(m_l1_norm, *this); + } - /** Returns the threshold that will be used by certain methods such as rank(). - * - * See the documentation of setThreshold(const RealScalar&). - */ - RealScalar threshold() const - { - eigen_assert(m_isInitialized || m_usePrescribedThreshold); - return m_usePrescribedThreshold ? m_prescribedThreshold - // this formula comes from experimenting (see "LU precision tuning" thread on the list) - // and turns out to be identical to Higham's formula used already in LDLt. - : NumTraits::epsilon() * m_lu.diagonalSize(); - } + /** \returns the determinant of the matrix of which + * *this is the LU decomposition. It has only linear complexity + * (that is, O(n) where n is the dimension of the square matrix) + * as the LU decomposition has already been computed. + * + * \note This is only for square matrices. + * + * \note For fixed-size matrices of size up to 4, MatrixBase::determinant() offers + * optimized paths. + * + * \warning a determinant can be very big or small, so for matrices + * of large enough dimension, there is a risk of overflow/underflow. + * + * \sa MatrixBase::determinant() + */ + typename internal::traits::Scalar determinant() const; + + /** Allows to prescribe a threshold to be used by certain methods, such as rank(), + * who need to determine when pivots are to be considered nonzero. This is not used for the + * LU decomposition itself. + * + * When it needs to get the threshold value, Eigen calls threshold(). By default, this + * uses a formula to automatically determine a reasonable threshold. + * Once you have called the present method setThreshold(const RealScalar&), + * your value is used instead. + * + * \param threshold The new value to use as the threshold. + * + * A pivot will be considered nonzero if its absolute value is strictly greater than + * \f$ \vert pivot \vert \leqslant threshold \times \vert maxpivot \vert \f$ + * where maxpivot is the biggest pivot. + * + * If you want to come back to the default behavior, call setThreshold(Default_t) + */ + FullPivLU &setThreshold(const RealScalar &threshold) + { + m_usePrescribedThreshold = true; + m_prescribedThreshold = threshold; + return *this; + } - /** \returns the rank of the matrix of which *this is the LU decomposition. - * - * \note This method has to determine which pivots should be considered nonzero. - * For that, it uses the threshold value that you can control by calling - * setThreshold(const RealScalar&). - */ - inline Index rank() const - { - using std::abs; - eigen_assert(m_isInitialized && "LU is not initialized."); - RealScalar premultiplied_threshold = abs(m_maxpivot) * threshold(); - Index result = 0; - for(Index i = 0; i < m_nonzero_pivots; ++i) - result += (abs(m_lu.coeff(i,i)) > premultiplied_threshold); - return result; - } + /** Allows to come back to the default behavior, letting Eigen use its default formula for + * determining the threshold. + * + * You should pass the special object Eigen::Default as parameter here. + * \code lu.setThreshold(Eigen::Default); \endcode + * + * See the documentation of setThreshold(const RealScalar&). + */ + FullPivLU &setThreshold(Default_t) + { + m_usePrescribedThreshold = false; + return *this; + } - /** \returns the dimension of the kernel of the matrix of which *this is the LU decomposition. - * - * \note This method has to determine which pivots should be considered nonzero. - * For that, it uses the threshold value that you can control by calling - * setThreshold(const RealScalar&). - */ - inline Index dimensionOfKernel() const - { - eigen_assert(m_isInitialized && "LU is not initialized."); - return cols() - rank(); - } + /** Returns the threshold that will be used by certain methods such as rank(). + * + * See the documentation of setThreshold(const RealScalar&). + */ + RealScalar threshold() const + { + eigen_assert(m_isInitialized || m_usePrescribedThreshold); + return m_usePrescribedThreshold ? m_prescribedThreshold + // this formula comes from experimenting (see "LU precision tuning" thread on the + // list) and turns out to be identical to Higham's formula used already in LDLt. + : NumTraits::epsilon() * m_lu.diagonalSize(); + } - /** \returns true if the matrix of which *this is the LU decomposition represents an injective - * linear map, i.e. has trivial kernel; false otherwise. - * - * \note This method has to determine which pivots should be considered nonzero. - * For that, it uses the threshold value that you can control by calling - * setThreshold(const RealScalar&). - */ - inline bool isInjective() const - { - eigen_assert(m_isInitialized && "LU is not initialized."); - return rank() == cols(); - } + /** \returns the rank of the matrix of which *this is the LU decomposition. + * + * \note This method has to determine which pivots should be considered nonzero. + * For that, it uses the threshold value that you can control by calling + * setThreshold(const RealScalar&). + */ + inline Index rank() const + { + using std::abs; + eigen_assert(m_isInitialized && "LU is not initialized."); + RealScalar premultiplied_threshold = abs(m_maxpivot) * threshold(); + Index result = 0; + for (Index i = 0; i < m_nonzero_pivots; ++i) result += (abs(m_lu.coeff(i, i)) > premultiplied_threshold); + return result; + } - /** \returns true if the matrix of which *this is the LU decomposition represents a surjective - * linear map; false otherwise. - * - * \note This method has to determine which pivots should be considered nonzero. - * For that, it uses the threshold value that you can control by calling - * setThreshold(const RealScalar&). - */ - inline bool isSurjective() const - { - eigen_assert(m_isInitialized && "LU is not initialized."); - return rank() == rows(); - } + /** \returns the dimension of the kernel of the matrix of which *this is the LU decomposition. + * + * \note This method has to determine which pivots should be considered nonzero. + * For that, it uses the threshold value that you can control by calling + * setThreshold(const RealScalar&). + */ + inline Index dimensionOfKernel() const + { + eigen_assert(m_isInitialized && "LU is not initialized."); + return cols() - rank(); + } - /** \returns true if the matrix of which *this is the LU decomposition is invertible. - * - * \note This method has to determine which pivots should be considered nonzero. - * For that, it uses the threshold value that you can control by calling - * setThreshold(const RealScalar&). - */ - inline bool isInvertible() const - { - eigen_assert(m_isInitialized && "LU is not initialized."); - return isInjective() && (m_lu.rows() == m_lu.cols()); - } + /** \returns true if the matrix of which *this is the LU decomposition represents an injective + * linear map, i.e. has trivial kernel; false otherwise. + * + * \note This method has to determine which pivots should be considered nonzero. + * For that, it uses the threshold value that you can control by calling + * setThreshold(const RealScalar&). + */ + inline bool isInjective() const + { + eigen_assert(m_isInitialized && "LU is not initialized."); + return rank() == cols(); + } - /** \returns the inverse of the matrix of which *this is the LU decomposition. - * - * \note If this matrix is not invertible, the returned matrix has undefined coefficients. - * Use isInvertible() to first determine whether this matrix is invertible. - * - * \sa MatrixBase::inverse() - */ - inline const Inverse inverse() const - { - eigen_assert(m_isInitialized && "LU is not initialized."); - eigen_assert(m_lu.rows() == m_lu.cols() && "You can't take the inverse of a non-square matrix!"); - return Inverse(*this); - } + /** \returns true if the matrix of which *this is the LU decomposition represents a surjective + * linear map; false otherwise. + * + * \note This method has to determine which pivots should be considered nonzero. + * For that, it uses the threshold value that you can control by calling + * setThreshold(const RealScalar&). + */ + inline bool isSurjective() const + { + eigen_assert(m_isInitialized && "LU is not initialized."); + return rank() == rows(); + } - MatrixType reconstructedMatrix() const; + /** \returns true if the matrix of which *this is the LU decomposition is invertible. + * + * \note This method has to determine which pivots should be considered nonzero. + * For that, it uses the threshold value that you can control by calling + * setThreshold(const RealScalar&). + */ + inline bool isInvertible() const + { + eigen_assert(m_isInitialized && "LU is not initialized."); + return isInjective() && (m_lu.rows() == m_lu.cols()); + } - EIGEN_DEVICE_FUNC inline Index rows() const { return m_lu.rows(); } - EIGEN_DEVICE_FUNC inline Index cols() const { return m_lu.cols(); } + /** \returns the inverse of the matrix of which *this is the LU decomposition. + * + * \note If this matrix is not invertible, the returned matrix has undefined coefficients. + * Use isInvertible() to first determine whether this matrix is invertible. + * + * \sa MatrixBase::inverse() + */ + inline const Inverse inverse() const + { + eigen_assert(m_isInitialized && "LU is not initialized."); + eigen_assert(m_lu.rows() == m_lu.cols() && "You can't take the inverse of a non-square matrix!"); + return Inverse(*this); + } - #ifndef EIGEN_PARSED_BY_DOXYGEN - template - EIGEN_DEVICE_FUNC - void _solve_impl(const RhsType &rhs, DstType &dst) const; + MatrixType reconstructedMatrix() const; - template - EIGEN_DEVICE_FUNC - void _solve_impl_transposed(const RhsType &rhs, DstType &dst) const; - #endif + EIGEN_DEVICE_FUNC inline Index rows() const { return m_lu.rows(); } + EIGEN_DEVICE_FUNC inline Index cols() const { return m_lu.cols(); } - protected: +#ifndef EIGEN_PARSED_BY_DOXYGEN + template + EIGEN_DEVICE_FUNC void _solve_impl(const RhsType &rhs, DstType &dst) const; - static void check_template_parameters() - { - EIGEN_STATIC_ASSERT_NON_INTEGER(Scalar); - } + template + EIGEN_DEVICE_FUNC void _solve_impl_transposed(const RhsType &rhs, DstType &dst) const; +#endif - void computeInPlace(); - - MatrixType m_lu; - PermutationPType m_p; - PermutationQType m_q; - IntColVectorType m_rowsTranspositions; - IntRowVectorType m_colsTranspositions; - Index m_nonzero_pivots; - RealScalar m_l1_norm; - RealScalar m_maxpivot, m_prescribedThreshold; - signed char m_det_pq; - bool m_isInitialized, m_usePrescribedThreshold; +protected: + static void check_template_parameters() { EIGEN_STATIC_ASSERT_NON_INTEGER(Scalar); } + + void computeInPlace(); + + MatrixType m_lu; + PermutationPType m_p; + PermutationQType m_q; + IntColVectorType m_rowsTranspositions; + IntRowVectorType m_colsTranspositions; + Index m_nonzero_pivots; + RealScalar m_l1_norm; + RealScalar m_maxpivot, m_prescribedThreshold; + signed char m_det_pq; + bool m_isInitialized, m_usePrescribedThreshold; }; template -FullPivLU::FullPivLU() - : m_isInitialized(false), m_usePrescribedThreshold(false) -{ -} +FullPivLU::FullPivLU() : m_isInitialized(false), m_usePrescribedThreshold(false) +{} template FullPivLU::FullPivLU(Index rows, Index cols) - : m_lu(rows, cols), - m_p(rows), - m_q(cols), - m_rowsTranspositions(rows), - m_colsTranspositions(cols), - m_isInitialized(false), - m_usePrescribedThreshold(false) -{ -} + : m_lu(rows, cols), m_p(rows), m_q(cols), m_rowsTranspositions(rows), m_colsTranspositions(cols), + m_isInitialized(false), m_usePrescribedThreshold(false) +{} template template -FullPivLU::FullPivLU(const EigenBase& matrix) - : m_lu(matrix.rows(), matrix.cols()), - m_p(matrix.rows()), - m_q(matrix.cols()), - m_rowsTranspositions(matrix.rows()), - m_colsTranspositions(matrix.cols()), - m_isInitialized(false), - m_usePrescribedThreshold(false) +FullPivLU::FullPivLU(const EigenBase &matrix) + : m_lu(matrix.rows(), matrix.cols()), m_p(matrix.rows()), m_q(matrix.cols()), m_rowsTranspositions(matrix.rows()), + m_colsTranspositions(matrix.cols()), m_isInitialized(false), m_usePrescribedThreshold(false) { compute(matrix.derived()); } template template -FullPivLU::FullPivLU(EigenBase& matrix) - : m_lu(matrix.derived()), - m_p(matrix.rows()), - m_q(matrix.cols()), - m_rowsTranspositions(matrix.rows()), - m_colsTranspositions(matrix.cols()), - m_isInitialized(false), - m_usePrescribedThreshold(false) +FullPivLU::FullPivLU(EigenBase &matrix) + : m_lu(matrix.derived()), m_p(matrix.rows()), m_q(matrix.cols()), m_rowsTranspositions(matrix.rows()), + m_colsTranspositions(matrix.cols()), m_isInitialized(false), m_usePrescribedThreshold(false) { computeInPlace(); } -template -void FullPivLU::computeInPlace() +template void FullPivLU::computeInPlace() { check_template_parameters(); // the permutations are stored as int indices, so just to be sure: - eigen_assert(m_lu.rows()<=NumTraits::highest() && m_lu.cols()<=NumTraits::highest()); + eigen_assert(m_lu.rows() <= NumTraits::highest() && m_lu.cols() <= NumTraits::highest()); m_l1_norm = m_lu.cwiseAbs().colwise().sum().maxCoeff(); @@ -504,13 +472,12 @@ void FullPivLU::computeInPlace() // can't accumulate on-the-fly because that will be done in reverse order for the rows. m_rowsTranspositions.resize(m_lu.rows()); m_colsTranspositions.resize(m_lu.cols()); - Index number_of_transpositions = 0; // number of NONTRIVIAL transpositions, i.e. m_rowsTranspositions[i]!=i + Index number_of_transpositions = 0;// number of NONTRIVIAL transpositions, i.e. m_rowsTranspositions[i]!=i - m_nonzero_pivots = size; // the generic case is that in which all pivots are nonzero (invertible case) + m_nonzero_pivots = size;// the generic case is that in which all pivots are nonzero (invertible case) m_maxpivot = RealScalar(0); - for(Index k = 0; k < size; ++k) - { + for (Index k = 0; k < size; ++k) { // First, we need to find the pivot. // biggest coefficient in the remaining bottom-right corner (starting at row k, col k) @@ -518,38 +485,37 @@ void FullPivLU::computeInPlace() typedef internal::scalar_score_coeff_op Scoring; typedef typename Scoring::result_type Score; Score biggest_in_corner; - biggest_in_corner = m_lu.bottomRightCorner(rows-k, cols-k) - .unaryExpr(Scoring()) - .maxCoeff(&row_of_biggest_in_corner, &col_of_biggest_in_corner); - row_of_biggest_in_corner += k; // correct the values! since they were computed in the corner, - col_of_biggest_in_corner += k; // need to add k to them. + biggest_in_corner = m_lu.bottomRightCorner(rows - k, cols - k) + .unaryExpr(Scoring()) + .maxCoeff(&row_of_biggest_in_corner, &col_of_biggest_in_corner); + row_of_biggest_in_corner += k;// correct the values! since they were computed in the corner, + col_of_biggest_in_corner += k;// need to add k to them. - if(biggest_in_corner==Score(0)) - { + if (biggest_in_corner == Score(0)) { // before exiting, make sure to initialize the still uninitialized transpositions // in a sane state without destroying what we already have. m_nonzero_pivots = k; - for(Index i = k; i < size; ++i) - { + for (Index i = k; i < size; ++i) { m_rowsTranspositions.coeffRef(i) = i; m_colsTranspositions.coeffRef(i) = i; } break; } - RealScalar abs_pivot = internal::abs_knowing_score()(m_lu(row_of_biggest_in_corner, col_of_biggest_in_corner), biggest_in_corner); - if(abs_pivot > m_maxpivot) m_maxpivot = abs_pivot; + RealScalar abs_pivot = internal::abs_knowing_score()( + m_lu(row_of_biggest_in_corner, col_of_biggest_in_corner), biggest_in_corner); + if (abs_pivot > m_maxpivot) m_maxpivot = abs_pivot; // Now that we've found the pivot, we need to apply the row/col swaps to // bring it to the location (k,k). m_rowsTranspositions.coeffRef(k) = row_of_biggest_in_corner; m_colsTranspositions.coeffRef(k) = col_of_biggest_in_corner; - if(k != row_of_biggest_in_corner) { + if (k != row_of_biggest_in_corner) { m_lu.row(k).swap(m_lu.row(row_of_biggest_in_corner)); ++number_of_transpositions; } - if(k != col_of_biggest_in_corner) { + if (k != col_of_biggest_in_corner) { m_lu.col(k).swap(m_lu.col(col_of_biggest_in_corner)); ++number_of_transpositions; } @@ -557,30 +523,27 @@ void FullPivLU::computeInPlace() // Now that the pivot is at the right location, we update the remaining // bottom-right corner by Gaussian elimination. - if(k= 0; --k) - m_p.applyTranspositionOnTheRight(k, m_rowsTranspositions.coeff(k)); + for (Index k = size - 1; k >= 0; --k) m_p.applyTranspositionOnTheRight(k, m_rowsTranspositions.coeff(k)); m_q.setIdentity(cols); - for(Index k = 0; k < size; ++k) - m_q.applyTranspositionOnTheRight(k, m_colsTranspositions.coeff(k)); + for (Index k = 0; k < size; ++k) m_q.applyTranspositionOnTheRight(k, m_colsTranspositions.coeff(k)); - m_det_pq = (number_of_transpositions%2) ? -1 : 1; + m_det_pq = (number_of_transpositions % 2) ? -1 : 1; m_isInitialized = true; } -template -typename internal::traits::Scalar FullPivLU::determinant() const +template typename internal::traits::Scalar FullPivLU::determinant() const { eigen_assert(m_isInitialized && "LU is not initialized."); eigen_assert(m_lu.rows() == m_lu.cols() && "You can't take the determinant of a non-square matrix!"); @@ -590,18 +553,15 @@ typename internal::traits::Scalar FullPivLU::determinant /** \returns the matrix represented by the decomposition, * i.e., it returns the product: \f$ P^{-1} L U Q^{-1} \f$. * This function is provided for debug purposes. */ -template -MatrixType FullPivLU::reconstructedMatrix() const +template MatrixType FullPivLU::reconstructedMatrix() const { eigen_assert(m_isInitialized && "LU is not initialized."); const Index smalldim = (std::min)(m_lu.rows(), m_lu.cols()); // LU - MatrixType res(m_lu.rows(),m_lu.cols()); + MatrixType res(m_lu.rows(), m_lu.cols()); // FIXME the .toDenseMatrix() should not be needed... - res = m_lu.leftCols(smalldim) - .template triangularView().toDenseMatrix() - * m_lu.topRows(smalldim) - .template triangularView().toDenseMatrix(); + res = m_lu.leftCols(smalldim).template triangularView().toDenseMatrix() + * m_lu.topRows(smalldim).template triangularView().toDenseMatrix(); // P^{-1}(LU) res = m_p.inverse() * res; @@ -615,131 +575,122 @@ MatrixType FullPivLU::reconstructedMatrix() const /********* Implementation of kernel() **************************************************/ namespace internal { -template -struct kernel_retval > - : kernel_retval_base > -{ - EIGEN_MAKE_KERNEL_HELPERS(FullPivLU<_MatrixType>) + template + struct kernel_retval> : kernel_retval_base> + { + EIGEN_MAKE_KERNEL_HELPERS(FullPivLU<_MatrixType>) - enum { MaxSmallDimAtCompileTime = EIGEN_SIZE_MIN_PREFER_FIXED( - MatrixType::MaxColsAtCompileTime, - MatrixType::MaxRowsAtCompileTime) - }; + enum { + MaxSmallDimAtCompileTime = + EIGEN_SIZE_MIN_PREFER_FIXED(MatrixType::MaxColsAtCompileTime, MatrixType::MaxRowsAtCompileTime) + }; - template void evalTo(Dest& dst) const - { - using std::abs; - const Index cols = dec().matrixLU().cols(), dimker = cols - rank(); - if(dimker == 0) + template void evalTo(Dest &dst) const { - // The Kernel is just {0}, so it doesn't have a basis properly speaking, but let's - // avoid crashing/asserting as that depends on floating point calculations. Let's - // just return a single column vector filled with zeros. - dst.setZero(); - return; - } + using std::abs; + const Index cols = dec().matrixLU().cols(), dimker = cols - rank(); + if (dimker == 0) { + // The Kernel is just {0}, so it doesn't have a basis properly speaking, but let's + // avoid crashing/asserting as that depends on floating point calculations. Let's + // just return a single column vector filled with zeros. + dst.setZero(); + return; + } - /* Let us use the following lemma: - * - * Lemma: If the matrix A has the LU decomposition PAQ = LU, - * then Ker A = Q(Ker U). - * - * Proof: trivial: just keep in mind that P, Q, L are invertible. - */ - - /* Thus, all we need to do is to compute Ker U, and then apply Q. - * - * U is upper triangular, with eigenvalues sorted so that any zeros appear at the end. - * Thus, the diagonal of U ends with exactly - * dimKer zero's. Let us use that to construct dimKer linearly - * independent vectors in Ker U. - */ - - Matrix pivots(rank()); - RealScalar premultiplied_threshold = dec().maxPivot() * dec().threshold(); - Index p = 0; - for(Index i = 0; i < dec().nonzeroPivots(); ++i) - if(abs(dec().matrixLU().coeff(i,i)) > premultiplied_threshold) - pivots.coeffRef(p++) = i; - eigen_internal_assert(p == rank()); - - // we construct a temporaty trapezoid matrix m, by taking the U matrix and - // permuting the rows and cols to bring the nonnegligible pivots to the top of - // the main diagonal. We need that to be able to apply our triangular solvers. - // FIXME when we get triangularView-for-rectangular-matrices, this can be simplified - Matrix - m(dec().matrixLU().block(0, 0, rank(), cols)); - for(Index i = 0; i < rank(); ++i) - { - if(i) m.row(i).head(i).setZero(); - m.row(i).tail(cols-i) = dec().matrixLU().row(pivots.coeff(i)).tail(cols-i); + /* Let us use the following lemma: + * + * Lemma: If the matrix A has the LU decomposition PAQ = LU, + * then Ker A = Q(Ker U). + * + * Proof: trivial: just keep in mind that P, Q, L are invertible. + */ + + /* Thus, all we need to do is to compute Ker U, and then apply Q. + * + * U is upper triangular, with eigenvalues sorted so that any zeros appear at the end. + * Thus, the diagonal of U ends with exactly + * dimKer zero's. Let us use that to construct dimKer linearly + * independent vectors in Ker U. + */ + + Matrix pivots(rank()); + RealScalar premultiplied_threshold = dec().maxPivot() * dec().threshold(); + Index p = 0; + for (Index i = 0; i < dec().nonzeroPivots(); ++i) + if (abs(dec().matrixLU().coeff(i, i)) > premultiplied_threshold) pivots.coeffRef(p++) = i; + eigen_internal_assert(p == rank()); + + // we construct a temporaty trapezoid matrix m, by taking the U matrix and + // permuting the rows and cols to bring the nonnegligible pivots to the top of + // the main diagonal. We need that to be able to apply our triangular solvers. + // FIXME when we get triangularView-for-rectangular-matrices, this can be simplified + Matrix + m(dec().matrixLU().block(0, 0, rank(), cols)); + for (Index i = 0; i < rank(); ++i) { + if (i) m.row(i).head(i).setZero(); + m.row(i).tail(cols - i) = dec().matrixLU().row(pivots.coeff(i)).tail(cols - i); + } + m.block(0, 0, rank(), rank()); + m.block(0, 0, rank(), rank()).template triangularView().setZero(); + for (Index i = 0; i < rank(); ++i) m.col(i).swap(m.col(pivots.coeff(i))); + + // ok, we have our trapezoid matrix, we can apply the triangular solver. + // notice that the math behind this suggests that we should apply this to the + // negative of the RHS, but for performance we just put the negative sign elsewhere, see below. + m.topLeftCorner(rank(), rank()).template triangularView().solveInPlace(m.topRightCorner(rank(), dimker)); + + // now we must undo the column permutation that we had applied! + for (Index i = rank() - 1; i >= 0; --i) m.col(i).swap(m.col(pivots.coeff(i))); + + // see the negative sign in the next line, that's what we were talking about above. + for (Index i = 0; i < rank(); ++i) dst.row(dec().permutationQ().indices().coeff(i)) = -m.row(i).tail(dimker); + for (Index i = rank(); i < cols; ++i) dst.row(dec().permutationQ().indices().coeff(i)).setZero(); + for (Index k = 0; k < dimker; ++k) dst.coeffRef(dec().permutationQ().indices().coeff(rank() + k), k) = Scalar(1); } - m.block(0, 0, rank(), rank()); - m.block(0, 0, rank(), rank()).template triangularView().setZero(); - for(Index i = 0; i < rank(); ++i) - m.col(i).swap(m.col(pivots.coeff(i))); - - // ok, we have our trapezoid matrix, we can apply the triangular solver. - // notice that the math behind this suggests that we should apply this to the - // negative of the RHS, but for performance we just put the negative sign elsewhere, see below. - m.topLeftCorner(rank(), rank()) - .template triangularView().solveInPlace( - m.topRightCorner(rank(), dimker) - ); - - // now we must undo the column permutation that we had applied! - for(Index i = rank()-1; i >= 0; --i) - m.col(i).swap(m.col(pivots.coeff(i))); - - // see the negative sign in the next line, that's what we were talking about above. - for(Index i = 0; i < rank(); ++i) dst.row(dec().permutationQ().indices().coeff(i)) = -m.row(i).tail(dimker); - for(Index i = rank(); i < cols; ++i) dst.row(dec().permutationQ().indices().coeff(i)).setZero(); - for(Index k = 0; k < dimker; ++k) dst.coeffRef(dec().permutationQ().indices().coeff(rank()+k), k) = Scalar(1); - } -}; + }; -/***** Implementation of image() *****************************************************/ + /***** Implementation of image() *****************************************************/ -template -struct image_retval > - : image_retval_base > -{ - EIGEN_MAKE_IMAGE_HELPERS(FullPivLU<_MatrixType>) + template struct image_retval> : image_retval_base> + { + EIGEN_MAKE_IMAGE_HELPERS(FullPivLU<_MatrixType>) - enum { MaxSmallDimAtCompileTime = EIGEN_SIZE_MIN_PREFER_FIXED( - MatrixType::MaxColsAtCompileTime, - MatrixType::MaxRowsAtCompileTime) - }; + enum { + MaxSmallDimAtCompileTime = + EIGEN_SIZE_MIN_PREFER_FIXED(MatrixType::MaxColsAtCompileTime, MatrixType::MaxRowsAtCompileTime) + }; - template void evalTo(Dest& dst) const - { - using std::abs; - if(rank() == 0) + template void evalTo(Dest &dst) const { - // The Image is just {0}, so it doesn't have a basis properly speaking, but let's - // avoid crashing/asserting as that depends on floating point calculations. Let's - // just return a single column vector filled with zeros. - dst.setZero(); - return; - } + using std::abs; + if (rank() == 0) { + // The Image is just {0}, so it doesn't have a basis properly speaking, but let's + // avoid crashing/asserting as that depends on floating point calculations. Let's + // just return a single column vector filled with zeros. + dst.setZero(); + return; + } - Matrix pivots(rank()); - RealScalar premultiplied_threshold = dec().maxPivot() * dec().threshold(); - Index p = 0; - for(Index i = 0; i < dec().nonzeroPivots(); ++i) - if(abs(dec().matrixLU().coeff(i,i)) > premultiplied_threshold) - pivots.coeffRef(p++) = i; - eigen_internal_assert(p == rank()); + Matrix pivots(rank()); + RealScalar premultiplied_threshold = dec().maxPivot() * dec().threshold(); + Index p = 0; + for (Index i = 0; i < dec().nonzeroPivots(); ++i) + if (abs(dec().matrixLU().coeff(i, i)) > premultiplied_threshold) pivots.coeffRef(p++) = i; + eigen_internal_assert(p == rank()); - for(Index i = 0; i < rank(); ++i) - dst.col(i) = originalMatrix().col(dec().permutationQ().indices().coeff(pivots.coeff(i))); - } -}; + for (Index i = 0; i < rank(); ++i) + dst.col(i) = originalMatrix().col(dec().permutationQ().indices().coeff(pivots.coeff(i))); + } + }; -/***** Implementation of solve() *****************************************************/ + /***** Implementation of solve() *****************************************************/ -} // end namespace internal +}// end namespace internal #ifndef EIGEN_PARSED_BY_DOXYGEN template @@ -747,21 +698,18 @@ template void FullPivLU<_MatrixType>::_solve_impl(const RhsType &rhs, DstType &dst) const { /* The decomposition PAQ = LU can be rewritten as A = P^{-1} L U Q^{-1}. - * So we proceed as follows: - * Step 1: compute c = P * rhs. - * Step 2: replace c by the solution x to Lx = c. Exists because L is invertible. - * Step 3: replace c by the solution x to Ux = c. May or may not exist. - * Step 4: result = Q * c; - */ - - const Index rows = this->rows(), - cols = this->cols(), - nonzero_pivots = this->rank(); + * So we proceed as follows: + * Step 1: compute c = P * rhs. + * Step 2: replace c by the solution x to Lx = c. Exists because L is invertible. + * Step 3: replace c by the solution x to Ux = c. May or may not exist. + * Step 4: result = Q * c; + */ + + const Index rows = this->rows(), cols = this->cols(), nonzero_pivots = this->rank(); eigen_assert(rhs.rows() == rows); const Index smalldim = (std::min)(rows, cols); - if(nonzero_pivots == 0) - { + if (nonzero_pivots == 0) { dst.setZero(); return; } @@ -772,22 +720,17 @@ void FullPivLU<_MatrixType>::_solve_impl(const RhsType &rhs, DstType &dst) const c = permutationP() * rhs; // Step 2 - m_lu.topLeftCorner(smalldim,smalldim) - .template triangularView() - .solveInPlace(c.topRows(smalldim)); - if(rows>cols) - c.bottomRows(rows-cols) -= m_lu.bottomRows(rows-cols) * c.topRows(cols); + m_lu.topLeftCorner(smalldim, smalldim).template triangularView().solveInPlace(c.topRows(smalldim)); + if (rows > cols) c.bottomRows(rows - cols) -= m_lu.bottomRows(rows - cols) * c.topRows(cols); // Step 3 m_lu.topLeftCorner(nonzero_pivots, nonzero_pivots) - .template triangularView() - .solveInPlace(c.topRows(nonzero_pivots)); + .template triangularView() + .solveInPlace(c.topRows(nonzero_pivots)); // Step 4 - for(Index i = 0; i < nonzero_pivots; ++i) - dst.row(permutationQ().indices().coeff(i)) = c.row(i); - for(Index i = nonzero_pivots; i < m_lu.cols(); ++i) - dst.row(permutationQ().indices().coeff(i)).setZero(); + for (Index i = 0; i < nonzero_pivots; ++i) dst.row(permutationQ().indices().coeff(i)) = c.row(i); + for (Index i = nonzero_pivots; i < m_lu.cols(); ++i) dst.row(permutationQ().indices().coeff(i)).setZero(); } template @@ -805,13 +748,11 @@ void FullPivLU<_MatrixType>::_solve_impl_transposed(const RhsType &rhs, DstType * If Conjugate is true, replace "^T" by "^*" above. */ - const Index rows = this->rows(), cols = this->cols(), - nonzero_pivots = this->rank(); - eigen_assert(rhs.rows() == cols); + const Index rows = this->rows(), cols = this->cols(), nonzero_pivots = this->rank(); + eigen_assert(rhs.rows() == cols); const Index smalldim = (std::min)(rows, cols); - if(nonzero_pivots == 0) - { + if (nonzero_pivots == 0) { dst.setZero(); return; } @@ -824,33 +765,31 @@ void FullPivLU<_MatrixType>::_solve_impl_transposed(const RhsType &rhs, DstType if (Conjugate) { // Step 2 m_lu.topLeftCorner(nonzero_pivots, nonzero_pivots) - .template triangularView() - .adjoint() - .solveInPlace(c.topRows(nonzero_pivots)); + .template triangularView() + .adjoint() + .solveInPlace(c.topRows(nonzero_pivots)); // Step 3 m_lu.topLeftCorner(smalldim, smalldim) - .template triangularView() - .adjoint() - .solveInPlace(c.topRows(smalldim)); + .template triangularView() + .adjoint() + .solveInPlace(c.topRows(smalldim)); } else { // Step 2 m_lu.topLeftCorner(nonzero_pivots, nonzero_pivots) - .template triangularView() - .transpose() - .solveInPlace(c.topRows(nonzero_pivots)); + .template triangularView() + .transpose() + .solveInPlace(c.topRows(nonzero_pivots)); // Step 3 m_lu.topLeftCorner(smalldim, smalldim) - .template triangularView() - .transpose() - .solveInPlace(c.topRows(smalldim)); + .template triangularView() + .transpose() + .solveInPlace(c.topRows(smalldim)); } // Step 4 PermutationPType invp = permutationP().inverse().eval(); - for(Index i = 0; i < smalldim; ++i) - dst.row(invp.indices().coeff(i)) = c.row(i); - for(Index i = smalldim; i < rows; ++i) - dst.row(invp.indices().coeff(i)).setZero(); + for (Index i = 0; i < smalldim; ++i) dst.row(invp.indices().coeff(i)) = c.row(i); + for (Index i = smalldim; i < rows; ++i) dst.row(invp.indices().coeff(i)).setZero(); } #endif @@ -858,34 +797,38 @@ void FullPivLU<_MatrixType>::_solve_impl_transposed(const RhsType &rhs, DstType namespace internal { -/***** Implementation of inverse() *****************************************************/ -template -struct Assignment >, internal::assign_op::Scalar>, Dense2Dense> -{ - typedef FullPivLU LuType; - typedef Inverse SrcXprType; - static void run(DstXprType &dst, const SrcXprType &src, const internal::assign_op &) + /***** Implementation of inverse() *****************************************************/ + template + struct Assignment>, + internal::assign_op::Scalar>, + Dense2Dense> { - dst = src.nestedExpression().solve(MatrixType::Identity(src.rows(), src.cols())); - } -}; -} // end namespace internal + typedef FullPivLU LuType; + typedef Inverse SrcXprType; + static void run(DstXprType &dst, + const SrcXprType &src, + const internal::assign_op &) + { + dst = src.nestedExpression().solve(MatrixType::Identity(src.rows(), src.cols())); + } + }; +}// end namespace internal /******* MatrixBase methods *****************************************************************/ /** \lu_module - * - * \return the full-pivoting LU decomposition of \c *this. - * - * \sa class FullPivLU - */ + * + * \return the full-pivoting LU decomposition of \c *this. + * + * \sa class FullPivLU + */ template -inline const FullPivLU::PlainObject> -MatrixBase::fullPivLu() const +inline const FullPivLU::PlainObject> MatrixBase::fullPivLu() const { return FullPivLU(eval()); } -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_LU_H +#endif// EIGEN_LU_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/LU/InverseImpl.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/LU/InverseImpl.h index f49f2336..298c3ee6 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/LU/InverseImpl.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/LU/InverseImpl.h @@ -11,405 +11,367 @@ #ifndef EIGEN_INVERSE_IMPL_H #define EIGEN_INVERSE_IMPL_H -namespace Eigen { +namespace Eigen { namespace internal { -/********************************** -*** General case implementation *** -**********************************/ + /********************************** + *** General case implementation *** + **********************************/ -template -struct compute_inverse -{ - EIGEN_DEVICE_FUNC - static inline void run(const MatrixType& matrix, ResultType& result) + template struct compute_inverse { - result = matrix.partialPivLu().inverse(); - } -}; + EIGEN_DEVICE_FUNC + static inline void run(const MatrixType &matrix, ResultType &result) { result = matrix.partialPivLu().inverse(); } + }; -template -struct compute_inverse_and_det_with_check { /* nothing! general case not supported. */ }; + template + struct compute_inverse_and_det_with_check + { /* nothing! general case not supported. */ + }; -/**************************** -*** Size 1 implementation *** -****************************/ + /**************************** + *** Size 1 implementation *** + ****************************/ -template -struct compute_inverse -{ - EIGEN_DEVICE_FUNC - static inline void run(const MatrixType& matrix, ResultType& result) + template struct compute_inverse { - typedef typename MatrixType::Scalar Scalar; - internal::evaluator matrixEval(matrix); - result.coeffRef(0,0) = Scalar(1) / matrixEval.coeff(0,0); - } -}; + EIGEN_DEVICE_FUNC + static inline void run(const MatrixType &matrix, ResultType &result) + { + typedef typename MatrixType::Scalar Scalar; + internal::evaluator matrixEval(matrix); + result.coeffRef(0, 0) = Scalar(1) / matrixEval.coeff(0, 0); + } + }; -template -struct compute_inverse_and_det_with_check -{ - EIGEN_DEVICE_FUNC - static inline void run( - const MatrixType& matrix, - const typename MatrixType::RealScalar& absDeterminantThreshold, - ResultType& result, - typename ResultType::Scalar& determinant, - bool& invertible - ) + template + struct compute_inverse_and_det_with_check + { + EIGEN_DEVICE_FUNC + static inline void run(const MatrixType &matrix, + const typename MatrixType::RealScalar &absDeterminantThreshold, + ResultType &result, + typename ResultType::Scalar &determinant, + bool &invertible) + { + using std::abs; + determinant = matrix.coeff(0, 0); + invertible = abs(determinant) > absDeterminantThreshold; + if (invertible) result.coeffRef(0, 0) = typename ResultType::Scalar(1) / determinant; + } + }; + + /**************************** + *** Size 2 implementation *** + ****************************/ + + template + EIGEN_DEVICE_FUNC inline void compute_inverse_size2_helper(const MatrixType &matrix, + const typename ResultType::Scalar &invdet, + ResultType &result) { - using std::abs; - determinant = matrix.coeff(0,0); - invertible = abs(determinant) > absDeterminantThreshold; - if(invertible) result.coeffRef(0,0) = typename ResultType::Scalar(1) / determinant; + result.coeffRef(0, 0) = matrix.coeff(1, 1) * invdet; + result.coeffRef(1, 0) = -matrix.coeff(1, 0) * invdet; + result.coeffRef(0, 1) = -matrix.coeff(0, 1) * invdet; + result.coeffRef(1, 1) = matrix.coeff(0, 0) * invdet; } -}; -/**************************** -*** Size 2 implementation *** -****************************/ + template struct compute_inverse + { + EIGEN_DEVICE_FUNC + static inline void run(const MatrixType &matrix, ResultType &result) + { + typedef typename ResultType::Scalar Scalar; + const Scalar invdet = typename MatrixType::Scalar(1) / matrix.determinant(); + compute_inverse_size2_helper(matrix, invdet, result); + } + }; + + template + struct compute_inverse_and_det_with_check + { + EIGEN_DEVICE_FUNC + static inline void run(const MatrixType &matrix, + const typename MatrixType::RealScalar &absDeterminantThreshold, + ResultType &inverse, + typename ResultType::Scalar &determinant, + bool &invertible) + { + using std::abs; + typedef typename ResultType::Scalar Scalar; + determinant = matrix.determinant(); + invertible = abs(determinant) > absDeterminantThreshold; + if (!invertible) return; + const Scalar invdet = Scalar(1) / determinant; + compute_inverse_size2_helper(matrix, invdet, inverse); + } + }; -template -EIGEN_DEVICE_FUNC -inline void compute_inverse_size2_helper( - const MatrixType& matrix, const typename ResultType::Scalar& invdet, - ResultType& result) -{ - result.coeffRef(0,0) = matrix.coeff(1,1) * invdet; - result.coeffRef(1,0) = -matrix.coeff(1,0) * invdet; - result.coeffRef(0,1) = -matrix.coeff(0,1) * invdet; - result.coeffRef(1,1) = matrix.coeff(0,0) * invdet; -} + /**************************** + *** Size 3 implementation *** + ****************************/ -template -struct compute_inverse -{ - EIGEN_DEVICE_FUNC - static inline void run(const MatrixType& matrix, ResultType& result) + template + EIGEN_DEVICE_FUNC inline typename MatrixType::Scalar cofactor_3x3(const MatrixType &m) { - typedef typename ResultType::Scalar Scalar; - const Scalar invdet = typename MatrixType::Scalar(1) / matrix.determinant(); - compute_inverse_size2_helper(matrix, invdet, result); + enum { i1 = (i + 1) % 3, i2 = (i + 2) % 3, j1 = (j + 1) % 3, j2 = (j + 2) % 3 }; + return m.coeff(i1, j1) * m.coeff(i2, j2) - m.coeff(i1, j2) * m.coeff(i2, j1); } -}; -template -struct compute_inverse_and_det_with_check -{ - EIGEN_DEVICE_FUNC - static inline void run( - const MatrixType& matrix, - const typename MatrixType::RealScalar& absDeterminantThreshold, - ResultType& inverse, - typename ResultType::Scalar& determinant, - bool& invertible - ) + template + EIGEN_DEVICE_FUNC inline void compute_inverse_size3_helper(const MatrixType &matrix, + const typename ResultType::Scalar &invdet, + const Matrix &cofactors_col0, + ResultType &result) { - using std::abs; - typedef typename ResultType::Scalar Scalar; - determinant = matrix.determinant(); - invertible = abs(determinant) > absDeterminantThreshold; - if(!invertible) return; - const Scalar invdet = Scalar(1) / determinant; - compute_inverse_size2_helper(matrix, invdet, inverse); + result.row(0) = cofactors_col0 * invdet; + result.coeffRef(1, 0) = cofactor_3x3(matrix) * invdet; + result.coeffRef(1, 1) = cofactor_3x3(matrix) * invdet; + result.coeffRef(1, 2) = cofactor_3x3(matrix) * invdet; + result.coeffRef(2, 0) = cofactor_3x3(matrix) * invdet; + result.coeffRef(2, 1) = cofactor_3x3(matrix) * invdet; + result.coeffRef(2, 2) = cofactor_3x3(matrix) * invdet; } -}; -/**************************** -*** Size 3 implementation *** -****************************/ + template struct compute_inverse + { + EIGEN_DEVICE_FUNC + static inline void run(const MatrixType &matrix, ResultType &result) + { + typedef typename ResultType::Scalar Scalar; + Matrix cofactors_col0; + cofactors_col0.coeffRef(0) = cofactor_3x3(matrix); + cofactors_col0.coeffRef(1) = cofactor_3x3(matrix); + cofactors_col0.coeffRef(2) = cofactor_3x3(matrix); + const Scalar det = (cofactors_col0.cwiseProduct(matrix.col(0))).sum(); + const Scalar invdet = Scalar(1) / det; + compute_inverse_size3_helper(matrix, invdet, cofactors_col0, result); + } + }; -template -EIGEN_DEVICE_FUNC -inline typename MatrixType::Scalar cofactor_3x3(const MatrixType& m) -{ - enum { - i1 = (i+1) % 3, - i2 = (i+2) % 3, - j1 = (j+1) % 3, - j2 = (j+2) % 3 + template + struct compute_inverse_and_det_with_check + { + EIGEN_DEVICE_FUNC + static inline void run(const MatrixType &matrix, + const typename MatrixType::RealScalar &absDeterminantThreshold, + ResultType &inverse, + typename ResultType::Scalar &determinant, + bool &invertible) + { + using std::abs; + typedef typename ResultType::Scalar Scalar; + Matrix cofactors_col0; + cofactors_col0.coeffRef(0) = cofactor_3x3(matrix); + cofactors_col0.coeffRef(1) = cofactor_3x3(matrix); + cofactors_col0.coeffRef(2) = cofactor_3x3(matrix); + determinant = (cofactors_col0.cwiseProduct(matrix.col(0))).sum(); + invertible = abs(determinant) > absDeterminantThreshold; + if (!invertible) return; + const Scalar invdet = Scalar(1) / determinant; + compute_inverse_size3_helper(matrix, invdet, cofactors_col0, inverse); + } }; - return m.coeff(i1, j1) * m.coeff(i2, j2) - - m.coeff(i1, j2) * m.coeff(i2, j1); -} -template -EIGEN_DEVICE_FUNC -inline void compute_inverse_size3_helper( - const MatrixType& matrix, - const typename ResultType::Scalar& invdet, - const Matrix& cofactors_col0, - ResultType& result) -{ - result.row(0) = cofactors_col0 * invdet; - result.coeffRef(1,0) = cofactor_3x3(matrix) * invdet; - result.coeffRef(1,1) = cofactor_3x3(matrix) * invdet; - result.coeffRef(1,2) = cofactor_3x3(matrix) * invdet; - result.coeffRef(2,0) = cofactor_3x3(matrix) * invdet; - result.coeffRef(2,1) = cofactor_3x3(matrix) * invdet; - result.coeffRef(2,2) = cofactor_3x3(matrix) * invdet; -} + /**************************** + *** Size 4 implementation *** + ****************************/ -template -struct compute_inverse -{ - EIGEN_DEVICE_FUNC - static inline void run(const MatrixType& matrix, ResultType& result) + template + EIGEN_DEVICE_FUNC inline const typename Derived::Scalar + general_det3_helper(const MatrixBase &matrix, int i1, int i2, int i3, int j1, int j2, int j3) { - typedef typename ResultType::Scalar Scalar; - Matrix cofactors_col0; - cofactors_col0.coeffRef(0) = cofactor_3x3(matrix); - cofactors_col0.coeffRef(1) = cofactor_3x3(matrix); - cofactors_col0.coeffRef(2) = cofactor_3x3(matrix); - const Scalar det = (cofactors_col0.cwiseProduct(matrix.col(0))).sum(); - const Scalar invdet = Scalar(1) / det; - compute_inverse_size3_helper(matrix, invdet, cofactors_col0, result); + return matrix.coeff(i1, j1) + * (matrix.coeff(i2, j2) * matrix.coeff(i3, j3) - matrix.coeff(i2, j3) * matrix.coeff(i3, j2)); } -}; -template -struct compute_inverse_and_det_with_check -{ - EIGEN_DEVICE_FUNC - static inline void run( - const MatrixType& matrix, - const typename MatrixType::RealScalar& absDeterminantThreshold, - ResultType& inverse, - typename ResultType::Scalar& determinant, - bool& invertible - ) + template + EIGEN_DEVICE_FUNC inline typename MatrixType::Scalar cofactor_4x4(const MatrixType &matrix) { - using std::abs; - typedef typename ResultType::Scalar Scalar; - Matrix cofactors_col0; - cofactors_col0.coeffRef(0) = cofactor_3x3(matrix); - cofactors_col0.coeffRef(1) = cofactor_3x3(matrix); - cofactors_col0.coeffRef(2) = cofactor_3x3(matrix); - determinant = (cofactors_col0.cwiseProduct(matrix.col(0))).sum(); - invertible = abs(determinant) > absDeterminantThreshold; - if(!invertible) return; - const Scalar invdet = Scalar(1) / determinant; - compute_inverse_size3_helper(matrix, invdet, cofactors_col0, inverse); + enum { i1 = (i + 1) % 4, i2 = (i + 2) % 4, i3 = (i + 3) % 4, j1 = (j + 1) % 4, j2 = (j + 2) % 4, j3 = (j + 3) % 4 }; + return general_det3_helper(matrix, i1, i2, i3, j1, j2, j3) + general_det3_helper(matrix, i2, i3, i1, j1, j2, j3) + + general_det3_helper(matrix, i3, i1, i2, j1, j2, j3); } -}; -/**************************** -*** Size 4 implementation *** -****************************/ - -template -EIGEN_DEVICE_FUNC -inline const typename Derived::Scalar general_det3_helper -(const MatrixBase& matrix, int i1, int i2, int i3, int j1, int j2, int j3) -{ - return matrix.coeff(i1,j1) - * (matrix.coeff(i2,j2) * matrix.coeff(i3,j3) - matrix.coeff(i2,j3) * matrix.coeff(i3,j2)); -} - -template -EIGEN_DEVICE_FUNC -inline typename MatrixType::Scalar cofactor_4x4(const MatrixType& matrix) -{ - enum { - i1 = (i+1) % 4, - i2 = (i+2) % 4, - i3 = (i+3) % 4, - j1 = (j+1) % 4, - j2 = (j+2) % 4, - j3 = (j+3) % 4 + template struct compute_inverse_size4 + { + EIGEN_DEVICE_FUNC + static void run(const MatrixType &matrix, ResultType &result) + { + result.coeffRef(0, 0) = cofactor_4x4(matrix); + result.coeffRef(1, 0) = -cofactor_4x4(matrix); + result.coeffRef(2, 0) = cofactor_4x4(matrix); + result.coeffRef(3, 0) = -cofactor_4x4(matrix); + result.coeffRef(0, 2) = cofactor_4x4(matrix); + result.coeffRef(1, 2) = -cofactor_4x4(matrix); + result.coeffRef(2, 2) = cofactor_4x4(matrix); + result.coeffRef(3, 2) = -cofactor_4x4(matrix); + result.coeffRef(0, 1) = -cofactor_4x4(matrix); + result.coeffRef(1, 1) = cofactor_4x4(matrix); + result.coeffRef(2, 1) = -cofactor_4x4(matrix); + result.coeffRef(3, 1) = cofactor_4x4(matrix); + result.coeffRef(0, 3) = -cofactor_4x4(matrix); + result.coeffRef(1, 3) = cofactor_4x4(matrix); + result.coeffRef(2, 3) = -cofactor_4x4(matrix); + result.coeffRef(3, 3) = cofactor_4x4(matrix); + result /= (matrix.col(0).cwiseProduct(result.row(0).transpose())).sum(); + } }; - return general_det3_helper(matrix, i1, i2, i3, j1, j2, j3) - + general_det3_helper(matrix, i2, i3, i1, j1, j2, j3) - + general_det3_helper(matrix, i3, i1, i2, j1, j2, j3); -} -template -struct compute_inverse_size4 -{ - EIGEN_DEVICE_FUNC - static void run(const MatrixType& matrix, ResultType& result) + template + struct compute_inverse + : compute_inverse_size4 { - result.coeffRef(0,0) = cofactor_4x4(matrix); - result.coeffRef(1,0) = -cofactor_4x4(matrix); - result.coeffRef(2,0) = cofactor_4x4(matrix); - result.coeffRef(3,0) = -cofactor_4x4(matrix); - result.coeffRef(0,2) = cofactor_4x4(matrix); - result.coeffRef(1,2) = -cofactor_4x4(matrix); - result.coeffRef(2,2) = cofactor_4x4(matrix); - result.coeffRef(3,2) = -cofactor_4x4(matrix); - result.coeffRef(0,1) = -cofactor_4x4(matrix); - result.coeffRef(1,1) = cofactor_4x4(matrix); - result.coeffRef(2,1) = -cofactor_4x4(matrix); - result.coeffRef(3,1) = cofactor_4x4(matrix); - result.coeffRef(0,3) = -cofactor_4x4(matrix); - result.coeffRef(1,3) = cofactor_4x4(matrix); - result.coeffRef(2,3) = -cofactor_4x4(matrix); - result.coeffRef(3,3) = cofactor_4x4(matrix); - result /= (matrix.col(0).cwiseProduct(result.row(0).transpose())).sum(); - } -}; - -template -struct compute_inverse - : compute_inverse_size4 -{ -}; + }; -template -struct compute_inverse_and_det_with_check -{ - EIGEN_DEVICE_FUNC - static inline void run( - const MatrixType& matrix, - const typename MatrixType::RealScalar& absDeterminantThreshold, - ResultType& inverse, - typename ResultType::Scalar& determinant, - bool& invertible - ) + template + struct compute_inverse_and_det_with_check { - using std::abs; - determinant = matrix.determinant(); - invertible = abs(determinant) > absDeterminantThreshold; - if(invertible) compute_inverse::run(matrix, inverse); - } -}; + EIGEN_DEVICE_FUNC + static inline void run(const MatrixType &matrix, + const typename MatrixType::RealScalar &absDeterminantThreshold, + ResultType &inverse, + typename ResultType::Scalar &determinant, + bool &invertible) + { + using std::abs; + determinant = matrix.determinant(); + invertible = abs(determinant) > absDeterminantThreshold; + if (invertible) compute_inverse::run(matrix, inverse); + } + }; -/************************* -*** MatrixBase methods *** -*************************/ + /************************* + *** MatrixBase methods *** + *************************/ -} // end namespace internal +}// end namespace internal namespace internal { -// Specialization for "dense = dense_xpr.inverse()" -template -struct Assignment, internal::assign_op, Dense2Dense> -{ - typedef Inverse SrcXprType; - static void run(DstXprType &dst, const SrcXprType &src, const internal::assign_op &) + // Specialization for "dense = dense_xpr.inverse()" + template + struct Assignment, + internal::assign_op, + Dense2Dense> { - Index dstRows = src.rows(); - Index dstCols = src.cols(); - if((dst.rows()!=dstRows) || (dst.cols()!=dstCols)) - dst.resize(dstRows, dstCols); - - const int Size = EIGEN_PLAIN_ENUM_MIN(XprType::ColsAtCompileTime,DstXprType::ColsAtCompileTime); - EIGEN_ONLY_USED_FOR_DEBUG(Size); - eigen_assert(( (Size<=1) || (Size>4) || (extract_data(src.nestedExpression())!=extract_data(dst))) - && "Aliasing problem detected in inverse(), you need to do inverse().eval() here."); - - typedef typename internal::nested_eval::type ActualXprType; - typedef typename internal::remove_all::type ActualXprTypeCleanded; - - ActualXprType actual_xpr(src.nestedExpression()); - - compute_inverse::run(actual_xpr, dst); - } -}; + typedef Inverse SrcXprType; + static void run(DstXprType &dst, + const SrcXprType &src, + const internal::assign_op &) + { + Index dstRows = src.rows(); + Index dstCols = src.cols(); + if ((dst.rows() != dstRows) || (dst.cols() != dstCols)) dst.resize(dstRows, dstCols); + + const int Size = EIGEN_PLAIN_ENUM_MIN(XprType::ColsAtCompileTime, DstXprType::ColsAtCompileTime); + EIGEN_ONLY_USED_FOR_DEBUG(Size); + eigen_assert(((Size <= 1) || (Size > 4) || (extract_data(src.nestedExpression()) != extract_data(dst))) + && "Aliasing problem detected in inverse(), you need to do inverse().eval() here."); + + typedef typename internal::nested_eval::type ActualXprType; + typedef typename internal::remove_all::type ActualXprTypeCleanded; + + ActualXprType actual_xpr(src.nestedExpression()); + + compute_inverse::run(actual_xpr, dst); + } + }; + - -} // end namespace internal +}// end namespace internal /** \lu_module - * - * \returns the matrix inverse of this matrix. - * - * For small fixed sizes up to 4x4, this method uses cofactors. - * In the general case, this method uses class PartialPivLU. - * - * \note This matrix must be invertible, otherwise the result is undefined. If you need an - * invertibility check, do the following: - * \li for fixed sizes up to 4x4, use computeInverseAndDetWithCheck(). - * \li for the general case, use class FullPivLU. - * - * Example: \include MatrixBase_inverse.cpp - * Output: \verbinclude MatrixBase_inverse.out - * - * \sa computeInverseAndDetWithCheck() - */ -template -inline const Inverse MatrixBase::inverse() const + * + * \returns the matrix inverse of this matrix. + * + * For small fixed sizes up to 4x4, this method uses cofactors. + * In the general case, this method uses class PartialPivLU. + * + * \note This matrix must be invertible, otherwise the result is undefined. If you need an + * invertibility check, do the following: + * \li for fixed sizes up to 4x4, use computeInverseAndDetWithCheck(). + * \li for the general case, use class FullPivLU. + * + * Example: \include MatrixBase_inverse.cpp + * Output: \verbinclude MatrixBase_inverse.out + * + * \sa computeInverseAndDetWithCheck() + */ +template inline const Inverse MatrixBase::inverse() const { - EIGEN_STATIC_ASSERT(!NumTraits::IsInteger,THIS_FUNCTION_IS_NOT_FOR_INTEGER_NUMERIC_TYPES) + EIGEN_STATIC_ASSERT(!NumTraits::IsInteger, THIS_FUNCTION_IS_NOT_FOR_INTEGER_NUMERIC_TYPES) eigen_assert(rows() == cols()); return Inverse(derived()); } /** \lu_module - * - * Computation of matrix inverse and determinant, with invertibility check. - * - * This is only for fixed-size square matrices of size up to 4x4. - * - * \param inverse Reference to the matrix in which to store the inverse. - * \param determinant Reference to the variable in which to store the determinant. - * \param invertible Reference to the bool variable in which to store whether the matrix is invertible. - * \param absDeterminantThreshold Optional parameter controlling the invertibility check. - * The matrix will be declared invertible if the absolute value of its - * determinant is greater than this threshold. - * - * Example: \include MatrixBase_computeInverseAndDetWithCheck.cpp - * Output: \verbinclude MatrixBase_computeInverseAndDetWithCheck.out - * - * \sa inverse(), computeInverseWithCheck() - */ + * + * Computation of matrix inverse and determinant, with invertibility check. + * + * This is only for fixed-size square matrices of size up to 4x4. + * + * \param inverse Reference to the matrix in which to store the inverse. + * \param determinant Reference to the variable in which to store the determinant. + * \param invertible Reference to the bool variable in which to store whether the matrix is invertible. + * \param absDeterminantThreshold Optional parameter controlling the invertibility check. + * The matrix will be declared invertible if the absolute value of its + * determinant is greater than this threshold. + * + * Example: \include MatrixBase_computeInverseAndDetWithCheck.cpp + * Output: \verbinclude MatrixBase_computeInverseAndDetWithCheck.out + * + * \sa inverse(), computeInverseWithCheck() + */ template template -inline void MatrixBase::computeInverseAndDetWithCheck( - ResultType& inverse, - typename ResultType::Scalar& determinant, - bool& invertible, - const RealScalar& absDeterminantThreshold - ) const +inline void MatrixBase::computeInverseAndDetWithCheck(ResultType &inverse, + typename ResultType::Scalar &determinant, + bool &invertible, + const RealScalar &absDeterminantThreshold) const { // i'd love to put some static assertions there, but SFINAE means that they have no effect... eigen_assert(rows() == cols()); // for 2x2, it's worth giving a chance to avoid evaluating. // for larger sizes, evaluating has negligible cost and limits code size. - typedef typename internal::conditional< - RowsAtCompileTime == 2, + typedef typename internal::conditional::type>::type, - PlainObject - >::type MatrixType; - internal::compute_inverse_and_det_with_check::run - (derived(), absDeterminantThreshold, inverse, determinant, invertible); + PlainObject>::type MatrixType; + internal::compute_inverse_and_det_with_check::run( + derived(), absDeterminantThreshold, inverse, determinant, invertible); } /** \lu_module - * - * Computation of matrix inverse, with invertibility check. - * - * This is only for fixed-size square matrices of size up to 4x4. - * - * \param inverse Reference to the matrix in which to store the inverse. - * \param invertible Reference to the bool variable in which to store whether the matrix is invertible. - * \param absDeterminantThreshold Optional parameter controlling the invertibility check. - * The matrix will be declared invertible if the absolute value of its - * determinant is greater than this threshold. - * - * Example: \include MatrixBase_computeInverseWithCheck.cpp - * Output: \verbinclude MatrixBase_computeInverseWithCheck.out - * - * \sa inverse(), computeInverseAndDetWithCheck() - */ + * + * Computation of matrix inverse, with invertibility check. + * + * This is only for fixed-size square matrices of size up to 4x4. + * + * \param inverse Reference to the matrix in which to store the inverse. + * \param invertible Reference to the bool variable in which to store whether the matrix is invertible. + * \param absDeterminantThreshold Optional parameter controlling the invertibility check. + * The matrix will be declared invertible if the absolute value of its + * determinant is greater than this threshold. + * + * Example: \include MatrixBase_computeInverseWithCheck.cpp + * Output: \verbinclude MatrixBase_computeInverseWithCheck.out + * + * \sa inverse(), computeInverseAndDetWithCheck() + */ template template -inline void MatrixBase::computeInverseWithCheck( - ResultType& inverse, - bool& invertible, - const RealScalar& absDeterminantThreshold - ) const +inline void MatrixBase::computeInverseWithCheck(ResultType &inverse, + bool &invertible, + const RealScalar &absDeterminantThreshold) const { Scalar determinant; // i'd love to put some static assertions there, but SFINAE means that they have no effect... eigen_assert(rows() == cols()); - computeInverseAndDetWithCheck(inverse,determinant,invertible,absDeterminantThreshold); + computeInverseAndDetWithCheck(inverse, determinant, invertible, absDeterminantThreshold); } -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_INVERSE_IMPL_H +#endif// EIGEN_INVERSE_IMPL_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/LU/PartialPivLU.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/LU/PartialPivLU.h index d4396188..153304f4 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/LU/PartialPivLU.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/LU/PartialPivLU.h @@ -14,323 +14,289 @@ namespace Eigen { namespace internal { -template struct traits > - : traits<_MatrixType> -{ - typedef MatrixXpr XprKind; - typedef SolverStorage StorageKind; - typedef traits<_MatrixType> BaseTraits; - enum { - Flags = BaseTraits::Flags & RowMajorBit, - CoeffReadCost = Dynamic + template struct traits> : traits<_MatrixType> + { + typedef MatrixXpr XprKind; + typedef SolverStorage StorageKind; + typedef traits<_MatrixType> BaseTraits; + enum { Flags = BaseTraits::Flags & RowMajorBit, CoeffReadCost = Dynamic }; }; -}; -template -struct enable_if_ref; -// { -// typedef Derived type; -// }; + template struct enable_if_ref; + // { + // typedef Derived type; + // }; -template -struct enable_if_ref,Derived> { - typedef Derived type; -}; + template struct enable_if_ref, Derived> + { + typedef Derived type; + }; -} // end namespace internal +}// end namespace internal /** \ingroup LU_Module - * - * \class PartialPivLU - * - * \brief LU decomposition of a matrix with partial pivoting, and related features - * - * \tparam _MatrixType the type of the matrix of which we are computing the LU decomposition - * - * This class represents a LU decomposition of a \b square \b invertible matrix, with partial pivoting: the matrix A - * is decomposed as A = PLU where L is unit-lower-triangular, U is upper-triangular, and P - * is a permutation matrix. - * - * Typically, partial pivoting LU decomposition is only considered numerically stable for square invertible - * matrices. Thus LAPACK's dgesv and dgesvx require the matrix to be square and invertible. The present class - * does the same. It will assert that the matrix is square, but it won't (actually it can't) check that the - * matrix is invertible: it is your task to check that you only use this decomposition on invertible matrices. - * - * The guaranteed safe alternative, working for all matrices, is the full pivoting LU decomposition, provided - * by class FullPivLU. - * - * This is \b not a rank-revealing LU decomposition. Many features are intentionally absent from this class, - * such as rank computation. If you need these features, use class FullPivLU. - * - * This LU decomposition is suitable to invert invertible matrices. It is what MatrixBase::inverse() uses - * in the general case. - * On the other hand, it is \b not suitable to determine whether a given matrix is invertible. - * - * The data of the LU decomposition can be directly accessed through the methods matrixLU(), permutationP(). - * - * This class supports the \link InplaceDecomposition inplace decomposition \endlink mechanism. - * - * \sa MatrixBase::partialPivLu(), MatrixBase::determinant(), MatrixBase::inverse(), MatrixBase::computeInverse(), class FullPivLU - */ -template class PartialPivLU - : public SolverBase > + * + * \class PartialPivLU + * + * \brief LU decomposition of a matrix with partial pivoting, and related features + * + * \tparam _MatrixType the type of the matrix of which we are computing the LU decomposition + * + * This class represents a LU decomposition of a \b square \b invertible matrix, with partial pivoting: the matrix A + * is decomposed as A = PLU where L is unit-lower-triangular, U is upper-triangular, and P + * is a permutation matrix. + * + * Typically, partial pivoting LU decomposition is only considered numerically stable for square invertible + * matrices. Thus LAPACK's dgesv and dgesvx require the matrix to be square and invertible. The present class + * does the same. It will assert that the matrix is square, but it won't (actually it can't) check that the + * matrix is invertible: it is your task to check that you only use this decomposition on invertible matrices. + * + * The guaranteed safe alternative, working for all matrices, is the full pivoting LU decomposition, provided + * by class FullPivLU. + * + * This is \b not a rank-revealing LU decomposition. Many features are intentionally absent from this class, + * such as rank computation. If you need these features, use class FullPivLU. + * + * This LU decomposition is suitable to invert invertible matrices. It is what MatrixBase::inverse() uses + * in the general case. + * On the other hand, it is \b not suitable to determine whether a given matrix is invertible. + * + * The data of the LU decomposition can be directly accessed through the methods matrixLU(), permutationP(). + * + * This class supports the \link InplaceDecomposition inplace decomposition \endlink mechanism. + * + * \sa MatrixBase::partialPivLu(), MatrixBase::determinant(), MatrixBase::inverse(), MatrixBase::computeInverse(), class + * FullPivLU + */ +template class PartialPivLU : public SolverBase> { - public: - - typedef _MatrixType MatrixType; - typedef SolverBase Base; - EIGEN_GENERIC_PUBLIC_INTERFACE(PartialPivLU) - // FIXME StorageIndex defined in EIGEN_GENERIC_PUBLIC_INTERFACE should be int - enum { - MaxRowsAtCompileTime = MatrixType::MaxRowsAtCompileTime, - MaxColsAtCompileTime = MatrixType::MaxColsAtCompileTime - }; - typedef PermutationMatrix PermutationType; - typedef Transpositions TranspositionType; - typedef typename MatrixType::PlainObject PlainObject; - - /** - * \brief Default Constructor. - * - * The default constructor is useful in cases in which the user intends to - * perform decompositions via PartialPivLU::compute(const MatrixType&). - */ - PartialPivLU(); - - /** \brief Default Constructor with memory preallocation - * - * Like the default constructor but with preallocation of the internal data - * according to the specified problem \a size. - * \sa PartialPivLU() - */ - explicit PartialPivLU(Index size); - - /** Constructor. - * - * \param matrix the matrix of which to compute the LU decomposition. - * - * \warning The matrix should have full rank (e.g. if it's square, it should be invertible). - * If you need to deal with non-full rank, use class FullPivLU instead. - */ - template - explicit PartialPivLU(const EigenBase& matrix); - - /** Constructor for \link InplaceDecomposition inplace decomposition \endlink - * - * \param matrix the matrix of which to compute the LU decomposition. - * - * \warning The matrix should have full rank (e.g. if it's square, it should be invertible). - * If you need to deal with non-full rank, use class FullPivLU instead. - */ - template - explicit PartialPivLU(EigenBase& matrix); - - template - PartialPivLU& compute(const EigenBase& matrix) { - m_lu = matrix.derived(); - compute(); - return *this; - } +public: + typedef _MatrixType MatrixType; + typedef SolverBase Base; + EIGEN_GENERIC_PUBLIC_INTERFACE(PartialPivLU) + // FIXME StorageIndex defined in EIGEN_GENERIC_PUBLIC_INTERFACE should be int + enum { + MaxRowsAtCompileTime = MatrixType::MaxRowsAtCompileTime, + MaxColsAtCompileTime = MatrixType::MaxColsAtCompileTime + }; + typedef PermutationMatrix PermutationType; + typedef Transpositions TranspositionType; + typedef typename MatrixType::PlainObject PlainObject; + + /** + * \brief Default Constructor. + * + * The default constructor is useful in cases in which the user intends to + * perform decompositions via PartialPivLU::compute(const MatrixType&). + */ + PartialPivLU(); + + /** \brief Default Constructor with memory preallocation + * + * Like the default constructor but with preallocation of the internal data + * according to the specified problem \a size. + * \sa PartialPivLU() + */ + explicit PartialPivLU(Index size); + + /** Constructor. + * + * \param matrix the matrix of which to compute the LU decomposition. + * + * \warning The matrix should have full rank (e.g. if it's square, it should be invertible). + * If you need to deal with non-full rank, use class FullPivLU instead. + */ + template explicit PartialPivLU(const EigenBase &matrix); + + /** Constructor for \link InplaceDecomposition inplace decomposition \endlink + * + * \param matrix the matrix of which to compute the LU decomposition. + * + * \warning The matrix should have full rank (e.g. if it's square, it should be invertible). + * If you need to deal with non-full rank, use class FullPivLU instead. + */ + template explicit PartialPivLU(EigenBase &matrix); + + template PartialPivLU &compute(const EigenBase &matrix) + { + m_lu = matrix.derived(); + compute(); + return *this; + } - /** \returns the LU decomposition matrix: the upper-triangular part is U, the - * unit-lower-triangular part is L (at least for square matrices; in the non-square - * case, special care is needed, see the documentation of class FullPivLU). - * - * \sa matrixL(), matrixU() - */ - inline const MatrixType& matrixLU() const - { - eigen_assert(m_isInitialized && "PartialPivLU is not initialized."); - return m_lu; - } + /** \returns the LU decomposition matrix: the upper-triangular part is U, the + * unit-lower-triangular part is L (at least for square matrices; in the non-square + * case, special care is needed, see the documentation of class FullPivLU). + * + * \sa matrixL(), matrixU() + */ + inline const MatrixType &matrixLU() const + { + eigen_assert(m_isInitialized && "PartialPivLU is not initialized."); + return m_lu; + } - /** \returns the permutation matrix P. - */ - inline const PermutationType& permutationP() const - { - eigen_assert(m_isInitialized && "PartialPivLU is not initialized."); - return m_p; - } + /** \returns the permutation matrix P. + */ + inline const PermutationType &permutationP() const + { + eigen_assert(m_isInitialized && "PartialPivLU is not initialized."); + return m_p; + } - /** This method returns the solution x to the equation Ax=b, where A is the matrix of which - * *this is the LU decomposition. - * - * \param b the right-hand-side of the equation to solve. Can be a vector or a matrix, - * the only requirement in order for the equation to make sense is that - * b.rows()==A.rows(), where A is the matrix of which *this is the LU decomposition. - * - * \returns the solution. - * - * Example: \include PartialPivLU_solve.cpp - * Output: \verbinclude PartialPivLU_solve.out - * - * Since this PartialPivLU class assumes anyway that the matrix A is invertible, the solution - * theoretically exists and is unique regardless of b. - * - * \sa TriangularView::solve(), inverse(), computeInverse() - */ - // FIXME this is a copy-paste of the base-class member to add the isInitialized assertion. - template - inline const Solve - solve(const MatrixBase& b) const - { - eigen_assert(m_isInitialized && "PartialPivLU is not initialized."); - return Solve(*this, b.derived()); - } + /** This method returns the solution x to the equation Ax=b, where A is the matrix of which + * *this is the LU decomposition. + * + * \param b the right-hand-side of the equation to solve. Can be a vector or a matrix, + * the only requirement in order for the equation to make sense is that + * b.rows()==A.rows(), where A is the matrix of which *this is the LU decomposition. + * + * \returns the solution. + * + * Example: \include PartialPivLU_solve.cpp + * Output: \verbinclude PartialPivLU_solve.out + * + * Since this PartialPivLU class assumes anyway that the matrix A is invertible, the solution + * theoretically exists and is unique regardless of b. + * + * \sa TriangularView::solve(), inverse(), computeInverse() + */ + // FIXME this is a copy-paste of the base-class member to add the isInitialized assertion. + template inline const Solve solve(const MatrixBase &b) const + { + eigen_assert(m_isInitialized && "PartialPivLU is not initialized."); + return Solve(*this, b.derived()); + } - /** \returns an estimate of the reciprocal condition number of the matrix of which \c *this is - the LU decomposition. - */ - inline RealScalar rcond() const - { - eigen_assert(m_isInitialized && "PartialPivLU is not initialized."); - return internal::rcond_estimate_helper(m_l1_norm, *this); - } + /** \returns an estimate of the reciprocal condition number of the matrix of which \c *this is + the LU decomposition. + */ + inline RealScalar rcond() const + { + eigen_assert(m_isInitialized && "PartialPivLU is not initialized."); + return internal::rcond_estimate_helper(m_l1_norm, *this); + } - /** \returns the inverse of the matrix of which *this is the LU decomposition. - * - * \warning The matrix being decomposed here is assumed to be invertible. If you need to check for - * invertibility, use class FullPivLU instead. - * - * \sa MatrixBase::inverse(), LU::inverse() - */ - inline const Inverse inverse() const - { - eigen_assert(m_isInitialized && "PartialPivLU is not initialized."); - return Inverse(*this); - } + /** \returns the inverse of the matrix of which *this is the LU decomposition. + * + * \warning The matrix being decomposed here is assumed to be invertible. If you need to check for + * invertibility, use class FullPivLU instead. + * + * \sa MatrixBase::inverse(), LU::inverse() + */ + inline const Inverse inverse() const + { + eigen_assert(m_isInitialized && "PartialPivLU is not initialized."); + return Inverse(*this); + } - /** \returns the determinant of the matrix of which - * *this is the LU decomposition. It has only linear complexity - * (that is, O(n) where n is the dimension of the square matrix) - * as the LU decomposition has already been computed. - * - * \note For fixed-size matrices of size up to 4, MatrixBase::determinant() offers - * optimized paths. - * - * \warning a determinant can be very big or small, so for matrices - * of large enough dimension, there is a risk of overflow/underflow. - * - * \sa MatrixBase::determinant() - */ - Scalar determinant() const; - - MatrixType reconstructedMatrix() const; - - inline Index rows() const { return m_lu.rows(); } - inline Index cols() const { return m_lu.cols(); } - - #ifndef EIGEN_PARSED_BY_DOXYGEN - template - EIGEN_DEVICE_FUNC - void _solve_impl(const RhsType &rhs, DstType &dst) const { - /* The decomposition PA = LU can be rewritten as A = P^{-1} L U. - * So we proceed as follows: - * Step 1: compute c = Pb. - * Step 2: replace c by the solution x to Lx = c. - * Step 3: replace c by the solution x to Ux = c. - */ - - eigen_assert(rhs.rows() == m_lu.rows()); + /** \returns the determinant of the matrix of which + * *this is the LU decomposition. It has only linear complexity + * (that is, O(n) where n is the dimension of the square matrix) + * as the LU decomposition has already been computed. + * + * \note For fixed-size matrices of size up to 4, MatrixBase::determinant() offers + * optimized paths. + * + * \warning a determinant can be very big or small, so for matrices + * of large enough dimension, there is a risk of overflow/underflow. + * + * \sa MatrixBase::determinant() + */ + Scalar determinant() const; + + MatrixType reconstructedMatrix() const; + + inline Index rows() const { return m_lu.rows(); } + inline Index cols() const { return m_lu.cols(); } + +#ifndef EIGEN_PARSED_BY_DOXYGEN + template + EIGEN_DEVICE_FUNC void _solve_impl(const RhsType &rhs, DstType &dst) const + { + /* The decomposition PA = LU can be rewritten as A = P^{-1} L U. + * So we proceed as follows: + * Step 1: compute c = Pb. + * Step 2: replace c by the solution x to Lx = c. + * Step 3: replace c by the solution x to Ux = c. + */ - // Step 1 - dst = permutationP() * rhs; + eigen_assert(rhs.rows() == m_lu.rows()); - // Step 2 - m_lu.template triangularView().solveInPlace(dst); + // Step 1 + dst = permutationP() * rhs; - // Step 3 - m_lu.template triangularView().solveInPlace(dst); - } + // Step 2 + m_lu.template triangularView().solveInPlace(dst); - template - EIGEN_DEVICE_FUNC - void _solve_impl_transposed(const RhsType &rhs, DstType &dst) const { - /* The decomposition PA = LU can be rewritten as A = P^{-1} L U. - * So we proceed as follows: - * Step 1: compute c = Pb. - * Step 2: replace c by the solution x to Lx = c. - * Step 3: replace c by the solution x to Ux = c. - */ - - eigen_assert(rhs.rows() == m_lu.cols()); - - if (Conjugate) { - // Step 1 - dst = m_lu.template triangularView().adjoint().solve(rhs); - // Step 2 - m_lu.template triangularView().adjoint().solveInPlace(dst); - } else { - // Step 1 - dst = m_lu.template triangularView().transpose().solve(rhs); - // Step 2 - m_lu.template triangularView().transpose().solveInPlace(dst); - } - // Step 3 - dst = permutationP().transpose() * dst; - } - #endif + // Step 3 + m_lu.template triangularView().solveInPlace(dst); + } - protected: + template + EIGEN_DEVICE_FUNC void _solve_impl_transposed(const RhsType &rhs, DstType &dst) const + { + /* The decomposition PA = LU can be rewritten as A = P^{-1} L U. + * So we proceed as follows: + * Step 1: compute c = Pb. + * Step 2: replace c by the solution x to Lx = c. + * Step 3: replace c by the solution x to Ux = c. + */ - static void check_template_parameters() - { - EIGEN_STATIC_ASSERT_NON_INTEGER(Scalar); + eigen_assert(rhs.rows() == m_lu.cols()); + + if (Conjugate) { + // Step 1 + dst = m_lu.template triangularView().adjoint().solve(rhs); + // Step 2 + m_lu.template triangularView().adjoint().solveInPlace(dst); + } else { + // Step 1 + dst = m_lu.template triangularView().transpose().solve(rhs); + // Step 2 + m_lu.template triangularView().transpose().solveInPlace(dst); } + // Step 3 + dst = permutationP().transpose() * dst; + } +#endif - void compute(); +protected: + static void check_template_parameters() { EIGEN_STATIC_ASSERT_NON_INTEGER(Scalar); } - MatrixType m_lu; - PermutationType m_p; - TranspositionType m_rowsTranspositions; - RealScalar m_l1_norm; - signed char m_det_p; - bool m_isInitialized; + void compute(); + + MatrixType m_lu; + PermutationType m_p; + TranspositionType m_rowsTranspositions; + RealScalar m_l1_norm; + signed char m_det_p; + bool m_isInitialized; }; template PartialPivLU::PartialPivLU() - : m_lu(), - m_p(), - m_rowsTranspositions(), - m_l1_norm(0), - m_det_p(0), - m_isInitialized(false) -{ -} + : m_lu(), m_p(), m_rowsTranspositions(), m_l1_norm(0), m_det_p(0), m_isInitialized(false) +{} template PartialPivLU::PartialPivLU(Index size) - : m_lu(size, size), - m_p(size), - m_rowsTranspositions(size), - m_l1_norm(0), - m_det_p(0), - m_isInitialized(false) -{ -} + : m_lu(size, size), m_p(size), m_rowsTranspositions(size), m_l1_norm(0), m_det_p(0), m_isInitialized(false) +{} template template -PartialPivLU::PartialPivLU(const EigenBase& matrix) - : m_lu(matrix.rows(),matrix.cols()), - m_p(matrix.rows()), - m_rowsTranspositions(matrix.rows()), - m_l1_norm(0), - m_det_p(0), - m_isInitialized(false) +PartialPivLU::PartialPivLU(const EigenBase &matrix) + : m_lu(matrix.rows(), matrix.cols()), m_p(matrix.rows()), m_rowsTranspositions(matrix.rows()), m_l1_norm(0), + m_det_p(0), m_isInitialized(false) { compute(matrix.derived()); } template template -PartialPivLU::PartialPivLU(EigenBase& matrix) - : m_lu(matrix.derived()), - m_p(matrix.rows()), - m_rowsTranspositions(matrix.rows()), - m_l1_norm(0), - m_det_p(0), +PartialPivLU::PartialPivLU(EigenBase &matrix) + : m_lu(matrix.derived()), m_p(matrix.rows()), m_rowsTranspositions(matrix.rows()), m_l1_norm(0), m_det_p(0), m_isInitialized(false) { compute(); @@ -338,186 +304,182 @@ PartialPivLU::PartialPivLU(EigenBase& matrix) namespace internal { -/** \internal This is the blocked version of fullpivlu_unblocked() */ -template -struct partial_lu_impl -{ - // FIXME add a stride to Map, so that the following mapping becomes easier, - // another option would be to create an expression being able to automatically - // warp any Map, Matrix, and Block expressions as a unique type, but since that's exactly - // a Map + stride, why not adding a stride to Map, and convenient ctors from a Matrix, - // and Block. - typedef Map > MapLU; - typedef Block MatrixType; - typedef Block BlockType; - typedef typename MatrixType::RealScalar RealScalar; - - /** \internal performs the LU decomposition in-place of the matrix \a lu - * using an unblocked algorithm. - * - * In addition, this function returns the row transpositions in the - * vector \a row_transpositions which must have a size equal to the number - * of columns of the matrix \a lu, and an integer \a nb_transpositions - * which returns the actual number of transpositions. - * - * \returns The index of the first pivot which is exactly zero if any, or a negative number otherwise. - */ - static Index unblocked_lu(MatrixType& lu, PivIndex* row_transpositions, PivIndex& nb_transpositions) + /** \internal This is the blocked version of fullpivlu_unblocked() */ + template struct partial_lu_impl { - typedef scalar_score_coeff_op Scoring; - typedef typename Scoring::result_type Score; - const Index rows = lu.rows(); - const Index cols = lu.cols(); - const Index size = (std::min)(rows,cols); - nb_transpositions = 0; - Index first_zero_pivot = -1; - for(Index k = 0; k < size; ++k) + // FIXME add a stride to Map, so that the following mapping becomes easier, + // another option would be to create an expression being able to automatically + // warp any Map, Matrix, and Block expressions as a unique type, but since that's exactly + // a Map + stride, why not adding a stride to Map, and convenient ctors from a Matrix, + // and Block. + typedef Map> MapLU; + typedef Block MatrixType; + typedef Block BlockType; + typedef typename MatrixType::RealScalar RealScalar; + + /** \internal performs the LU decomposition in-place of the matrix \a lu + * using an unblocked algorithm. + * + * In addition, this function returns the row transpositions in the + * vector \a row_transpositions which must have a size equal to the number + * of columns of the matrix \a lu, and an integer \a nb_transpositions + * which returns the actual number of transpositions. + * + * \returns The index of the first pivot which is exactly zero if any, or a negative number otherwise. + */ + static Index unblocked_lu(MatrixType &lu, PivIndex *row_transpositions, PivIndex &nb_transpositions) { - Index rrows = rows-k-1; - Index rcols = cols-k-1; - - Index row_of_biggest_in_col; - Score biggest_in_corner - = lu.col(k).tail(rows-k).unaryExpr(Scoring()).maxCoeff(&row_of_biggest_in_col); - row_of_biggest_in_col += k; - - row_transpositions[k] = PivIndex(row_of_biggest_in_col); - - if(biggest_in_corner != Score(0)) - { - if(k != row_of_biggest_in_col) - { - lu.row(k).swap(lu.row(row_of_biggest_in_col)); - ++nb_transpositions; + typedef scalar_score_coeff_op Scoring; + typedef typename Scoring::result_type Score; + const Index rows = lu.rows(); + const Index cols = lu.cols(); + const Index size = (std::min)(rows, cols); + nb_transpositions = 0; + Index first_zero_pivot = -1; + for (Index k = 0; k < size; ++k) { + Index rrows = rows - k - 1; + Index rcols = cols - k - 1; + + Index row_of_biggest_in_col; + Score biggest_in_corner = lu.col(k).tail(rows - k).unaryExpr(Scoring()).maxCoeff(&row_of_biggest_in_col); + row_of_biggest_in_col += k; + + row_transpositions[k] = PivIndex(row_of_biggest_in_col); + + if (biggest_in_corner != Score(0)) { + if (k != row_of_biggest_in_col) { + lu.row(k).swap(lu.row(row_of_biggest_in_col)); + ++nb_transpositions; + } + + // FIXME shall we introduce a safe quotient expression in cas 1/lu.coeff(k,k) + // overflow but not the actual quotient? + lu.col(k).tail(rrows) /= lu.coeff(k, k); + } else if (first_zero_pivot == -1) { + // the pivot is exactly zero, we record the index of the first pivot which is exactly 0, + // and continue the factorization such we still have A = PLU + first_zero_pivot = k; } - // FIXME shall we introduce a safe quotient expression in cas 1/lu.coeff(k,k) - // overflow but not the actual quotient? - lu.col(k).tail(rrows) /= lu.coeff(k,k); - } - else if(first_zero_pivot==-1) - { - // the pivot is exactly zero, we record the index of the first pivot which is exactly 0, - // and continue the factorization such we still have A = PLU - first_zero_pivot = k; + if (k < rows - 1) lu.bottomRightCorner(rrows, rcols).noalias() -= lu.col(k).tail(rrows) * lu.row(k).tail(rcols); } - - if(k > > - */ - static Index blocked_lu(Index rows, Index cols, Scalar* lu_data, Index luStride, PivIndex* row_transpositions, PivIndex& nb_transpositions, Index maxBlockSize=256) - { - MapLU lu1(lu_data,StorageOrder==RowMajor?rows:luStride,StorageOrder==RowMajor?luStride:cols); - MatrixType lu(lu1,0,0,rows,cols); - - const Index size = (std::min)(rows,cols); - - // if the matrix is too small, no blocking: - if(size<=16) + /** \internal performs the LU decomposition in-place of the matrix represented + * by the variables \a rows, \a cols, \a lu_data, and \a lu_stride using a + * recursive, blocked algorithm. + * + * In addition, this function returns the row transpositions in the + * vector \a row_transpositions which must have a size equal to the number + * of columns of the matrix \a lu, and an integer \a nb_transpositions + * which returns the actual number of transpositions. + * + * \returns The index of the first pivot which is exactly zero if any, or a negative number otherwise. + * + * \note This very low level interface using pointers, etc. is to: + * 1 - reduce the number of instanciations to the strict minimum + * 2 - avoid infinite recursion of the instanciations with Block > > + */ + static Index blocked_lu(Index rows, + Index cols, + Scalar *lu_data, + Index luStride, + PivIndex *row_transpositions, + PivIndex &nb_transpositions, + Index maxBlockSize = 256) { - return unblocked_lu(lu, row_transpositions, nb_transpositions); - } + MapLU lu1(lu_data, StorageOrder == RowMajor ? rows : luStride, StorageOrder == RowMajor ? luStride : cols); + MatrixType lu(lu1, 0, 0, rows, cols); - // automatically adjust the number of subdivisions to the size - // of the matrix so that there is enough sub blocks: - Index blockSize; - { - blockSize = size/8; - blockSize = (blockSize/16)*16; - blockSize = (std::min)((std::max)(blockSize,Index(8)), maxBlockSize); - } + const Index size = (std::min)(rows, cols); - nb_transpositions = 0; - Index first_zero_pivot = -1; - for(Index k = 0; k < size; k+=blockSize) - { - Index bs = (std::min)(size-k,blockSize); // actual size of the block - Index trows = rows - k - bs; // trailing rows - Index tsize = size - k - bs; // trailing size - - // partition the matrix: - // A00 | A01 | A02 - // lu = A_0 | A_1 | A_2 = A10 | A11 | A12 - // A20 | A21 | A22 - BlockType A_0(lu,0,0,rows,k); - BlockType A_2(lu,0,k+bs,rows,tsize); - BlockType A11(lu,k,k,bs,bs); - BlockType A12(lu,k,k+bs,bs,tsize); - BlockType A21(lu,k+bs,k,trows,bs); - BlockType A22(lu,k+bs,k+bs,trows,tsize); - - PivIndex nb_transpositions_in_panel; - // recursively call the blocked LU algorithm on [A11^T A21^T]^T - // with a very small blocking size: - Index ret = blocked_lu(trows+bs, bs, &lu.coeffRef(k,k), luStride, - row_transpositions+k, nb_transpositions_in_panel, 16); - if(ret>=0 && first_zero_pivot==-1) - first_zero_pivot = k+ret; - - nb_transpositions += nb_transpositions_in_panel; - // update permutations and apply them to A_0 - for(Index i=k; i(k)); - A_0.row(i).swap(A_0.row(piv)); + blockSize = size / 8; + blockSize = (blockSize / 16) * 16; + blockSize = (std::min)((std::max)(blockSize, Index(8)), maxBlockSize); } - if(trows) - { - // apply permutations to A_2 - for(Index i=k;i= 0 && first_zero_pivot == -1) first_zero_pivot = k + ret; + + nb_transpositions += nb_transpositions_in_panel; + // update permutations and apply them to A_0 + for (Index i = k; i < k + bs; ++i) { + Index piv = (row_transpositions[i] += internal::convert_index(k)); + A_0.row(i).swap(A_0.row(piv)); + } + + if (trows) { + // apply permutations to A_2 + for (Index i = k; i < k + bs; ++i) A_2.row(i).swap(A_2.row(row_transpositions[i])); - // A12 = A11^-1 A12 - A11.template triangularView().solveInPlace(A12); + // A12 = A11^-1 A12 + A11.template triangularView().solveInPlace(A12); - A22.noalias() -= A21 * A12; + A22.noalias() -= A21 * A12; + } } + return first_zero_pivot; } - return first_zero_pivot; - } -}; - -/** \internal performs the LU decomposition with partial pivoting in-place. - */ -template -void partial_lu_inplace(MatrixType& lu, TranspositionType& row_transpositions, typename TranspositionType::StorageIndex& nb_transpositions) -{ - eigen_assert(lu.cols() == row_transpositions.size()); - eigen_assert((&row_transpositions.coeffRef(1)-&row_transpositions.coeffRef(0)) == 1); + }; - partial_lu_impl - - ::blocked_lu(lu.rows(), lu.cols(), &lu.coeffRef(0,0), lu.outerStride(), &row_transpositions.coeffRef(0), nb_transpositions); -} + /** \internal performs the LU decomposition with partial pivoting in-place. + */ + template + void partial_lu_inplace(MatrixType &lu, + TranspositionType &row_transpositions, + typename TranspositionType::StorageIndex &nb_transpositions) + { + eigen_assert(lu.cols() == row_transpositions.size()); + eigen_assert((&row_transpositions.coeffRef(1) - &row_transpositions.coeffRef(0)) == 1); + + partial_lu_impl::blocked_lu(lu.rows(), + lu.cols(), + &lu.coeffRef(0, 0), + lu.outerStride(), + &row_transpositions.coeffRef(0), + nb_transpositions); + } -} // end namespace internal +}// end namespace internal -template -void PartialPivLU::compute() +template void PartialPivLU::compute() { check_template_parameters(); // the row permutation is stored as int indices, so just to be sure: - eigen_assert(m_lu.rows()::highest()); + eigen_assert(m_lu.rows() < NumTraits::highest()); m_l1_norm = m_lu.cwiseAbs().colwise().sum().maxCoeff(); @@ -528,15 +490,14 @@ void PartialPivLU::compute() typename TranspositionType::StorageIndex nb_transpositions; internal::partial_lu_inplace(m_lu, m_rowsTranspositions, nb_transpositions); - m_det_p = (nb_transpositions%2) ? -1 : 1; + m_det_p = (nb_transpositions % 2) ? -1 : 1; m_p = m_rowsTranspositions; m_isInitialized = true; } -template -typename PartialPivLU::Scalar PartialPivLU::determinant() const +template typename PartialPivLU::Scalar PartialPivLU::determinant() const { eigen_assert(m_isInitialized && "PartialPivLU is not initialized."); return Scalar(m_det_p) * m_lu.diagonal().prod(); @@ -545,13 +506,11 @@ typename PartialPivLU::Scalar PartialPivLU::determinant( /** \returns the matrix represented by the decomposition, * i.e., it returns the product: P^{-1} L U. * This function is provided for debug purpose. */ -template -MatrixType PartialPivLU::reconstructedMatrix() const +template MatrixType PartialPivLU::reconstructedMatrix() const { eigen_assert(m_isInitialized && "LU is not initialized."); // LU - MatrixType res = m_lu.template triangularView().toDenseMatrix() - * m_lu.template triangularView(); + MatrixType res = m_lu.template triangularView().toDenseMatrix() * m_lu.template triangularView(); // P^{-1}(LU) res = m_p.inverse() * res; @@ -563,49 +522,52 @@ MatrixType PartialPivLU::reconstructedMatrix() const namespace internal { -/***** Implementation of inverse() *****************************************************/ -template -struct Assignment >, internal::assign_op::Scalar>, Dense2Dense> -{ - typedef PartialPivLU LuType; - typedef Inverse SrcXprType; - static void run(DstXprType &dst, const SrcXprType &src, const internal::assign_op &) + /***** Implementation of inverse() *****************************************************/ + template + struct Assignment>, + internal::assign_op::Scalar>, + Dense2Dense> { - dst = src.nestedExpression().solve(MatrixType::Identity(src.rows(), src.cols())); - } -}; -} // end namespace internal + typedef PartialPivLU LuType; + typedef Inverse SrcXprType; + static void run(DstXprType &dst, + const SrcXprType &src, + const internal::assign_op &) + { + dst = src.nestedExpression().solve(MatrixType::Identity(src.rows(), src.cols())); + } + }; +}// end namespace internal /******** MatrixBase methods *******/ /** \lu_module - * - * \return the partial-pivoting LU decomposition of \c *this. - * - * \sa class PartialPivLU - */ + * + * \return the partial-pivoting LU decomposition of \c *this. + * + * \sa class PartialPivLU + */ template -inline const PartialPivLU::PlainObject> -MatrixBase::partialPivLu() const +inline const PartialPivLU::PlainObject> MatrixBase::partialPivLu() const { return PartialPivLU(eval()); } /** \lu_module - * - * Synonym of partialPivLu(). - * - * \return the partial-pivoting LU decomposition of \c *this. - * - * \sa class PartialPivLU - */ + * + * Synonym of partialPivLu(). + * + * \return the partial-pivoting LU decomposition of \c *this. + * + * \sa class PartialPivLU + */ template -inline const PartialPivLU::PlainObject> -MatrixBase::lu() const +inline const PartialPivLU::PlainObject> MatrixBase::lu() const { return PartialPivLU(eval()); } -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_PARTIALLU_H +#endif// EIGEN_PARTIALLU_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/LU/PartialPivLU_LAPACKE.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/LU/PartialPivLU_LAPACKE.h index 755168a9..d1aaed3d 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/LU/PartialPivLU_LAPACKE.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/LU/PartialPivLU_LAPACKE.h @@ -33,51 +33,59 @@ #ifndef EIGEN_PARTIALLU_LAPACK_H #define EIGEN_PARTIALLU_LAPACK_H -namespace Eigen { +namespace Eigen { namespace internal { -/** \internal Specialization for the data types supported by LAPACKe */ + /** \internal Specialization for the data types supported by LAPACKe */ -#define EIGEN_LAPACKE_LU_PARTPIV(EIGTYPE, LAPACKE_TYPE, LAPACKE_PREFIX) \ -template \ -struct partial_lu_impl \ -{ \ - /* \internal performs the LU decomposition in-place of the matrix represented */ \ - static lapack_int blocked_lu(Index rows, Index cols, EIGTYPE* lu_data, Index luStride, lapack_int* row_transpositions, lapack_int& nb_transpositions, lapack_int maxBlockSize=256) \ - { \ - EIGEN_UNUSED_VARIABLE(maxBlockSize);\ - lapack_int matrix_order, first_zero_pivot; \ - lapack_int m, n, lda, *ipiv, info; \ - EIGTYPE* a; \ -/* Set up parameters for ?getrf */ \ - matrix_order = StorageOrder==RowMajor ? LAPACK_ROW_MAJOR : LAPACK_COL_MAJOR; \ - lda = convert_index(luStride); \ - a = lu_data; \ - ipiv = row_transpositions; \ - m = convert_index(rows); \ - n = convert_index(cols); \ - nb_transpositions = 0; \ -\ - info = LAPACKE_##LAPACKE_PREFIX##getrf( matrix_order, m, n, (LAPACKE_TYPE*)a, lda, ipiv ); \ -\ - for(int i=0;i= 0); \ -/* something should be done with nb_transpositions */ \ -\ - first_zero_pivot = info; \ - return first_zero_pivot; \ - } \ -}; +#define EIGEN_LAPACKE_LU_PARTPIV(EIGTYPE, LAPACKE_TYPE, LAPACKE_PREFIX) \ + template struct partial_lu_impl \ + { \ + /* \internal performs the LU decomposition in-place of the matrix represented */ \ + static lapack_int blocked_lu(Index rows, \ + Index cols, \ + EIGTYPE *lu_data, \ + Index luStride, \ + lapack_int *row_transpositions, \ + lapack_int &nb_transpositions, \ + lapack_int maxBlockSize = 256) \ + { \ + EIGEN_UNUSED_VARIABLE(maxBlockSize); \ + lapack_int matrix_order, first_zero_pivot; \ + lapack_int m, n, lda, *ipiv, info; \ + EIGTYPE *a; \ + /* Set up parameters for ?getrf */ \ + matrix_order = StorageOrder == RowMajor ? LAPACK_ROW_MAJOR : LAPACK_COL_MAJOR; \ + lda = convert_index(luStride); \ + a = lu_data; \ + ipiv = row_transpositions; \ + m = convert_index(rows); \ + n = convert_index(cols); \ + nb_transpositions = 0; \ + \ + info = LAPACKE_##LAPACKE_PREFIX##getrf(matrix_order, m, n, (LAPACKE_TYPE *)a, lda, ipiv); \ + \ + for (int i = 0; i < m; i++) { \ + ipiv[i]--; \ + if (ipiv[i] != i) nb_transpositions++; \ + } \ + \ + eigen_assert(info >= 0); \ + /* something should be done with nb_transpositions */ \ + \ + first_zero_pivot = info; \ + return first_zero_pivot; \ + } \ + }; -EIGEN_LAPACKE_LU_PARTPIV(double, double, d) -EIGEN_LAPACKE_LU_PARTPIV(float, float, s) -EIGEN_LAPACKE_LU_PARTPIV(dcomplex, lapack_complex_double, z) -EIGEN_LAPACKE_LU_PARTPIV(scomplex, lapack_complex_float, c) + EIGEN_LAPACKE_LU_PARTPIV(double, double, d) + EIGEN_LAPACKE_LU_PARTPIV(float, float, s) + EIGEN_LAPACKE_LU_PARTPIV(dcomplex, lapack_complex_double, z) + EIGEN_LAPACKE_LU_PARTPIV(scomplex, lapack_complex_float, c) -} // end namespace internal +}// end namespace internal -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_PARTIALLU_LAPACK_H +#endif// EIGEN_PARTIALLU_LAPACK_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/LU/arch/Inverse_SSE.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/LU/arch/Inverse_SSE.h index ebb64a62..ffba7720 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/LU/arch/Inverse_SSE.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/LU/arch/Inverse_SSE.h @@ -27,312 +27,317 @@ #ifndef EIGEN_INVERSE_SSE_H #define EIGEN_INVERSE_SSE_H -namespace Eigen { +namespace Eigen { namespace internal { -template -struct compute_inverse_size4 -{ - enum { - MatrixAlignment = traits::Alignment, - ResultAlignment = traits::Alignment, - StorageOrdersMatch = (MatrixType::Flags&RowMajorBit) == (ResultType::Flags&RowMajorBit) - }; - typedef typename conditional<(MatrixType::Flags&LinearAccessBit),MatrixType const &,typename MatrixType::PlainObject>::type ActualMatrixType; - - static void run(const MatrixType& mat, ResultType& result) + template + struct compute_inverse_size4 { - ActualMatrixType matrix(mat); - EIGEN_ALIGN16 const unsigned int _Sign_PNNP[4] = { 0x00000000, 0x80000000, 0x80000000, 0x00000000 }; - - // Load the full matrix into registers - __m128 _L1 = matrix.template packet( 0); - __m128 _L2 = matrix.template packet( 4); - __m128 _L3 = matrix.template packet( 8); - __m128 _L4 = matrix.template packet(12); - - // The inverse is calculated using "Divide and Conquer" technique. The - // original matrix is divide into four 2x2 sub-matrices. Since each - // register holds four matrix element, the smaller matrices are - // represented as a registers. Hence we get a better locality of the - // calculations. - - __m128 A, B, C, D; // the four sub-matrices - if(!StorageOrdersMatch) - { - A = _mm_unpacklo_ps(_L1, _L2); - B = _mm_unpacklo_ps(_L3, _L4); - C = _mm_unpackhi_ps(_L1, _L2); - D = _mm_unpackhi_ps(_L3, _L4); - } - else + enum { + MatrixAlignment = traits::Alignment, + ResultAlignment = traits::Alignment, + StorageOrdersMatch = (MatrixType::Flags & RowMajorBit) == (ResultType::Flags & RowMajorBit) + }; + typedef typename conditional<(MatrixType::Flags & LinearAccessBit), + MatrixType const &, + typename MatrixType::PlainObject>::type ActualMatrixType; + + static void run(const MatrixType &mat, ResultType &result) { - A = _mm_movelh_ps(_L1, _L2); - B = _mm_movehl_ps(_L2, _L1); - C = _mm_movelh_ps(_L3, _L4); - D = _mm_movehl_ps(_L4, _L3); + ActualMatrixType matrix(mat); + EIGEN_ALIGN16 const unsigned int _Sign_PNNP[4] = { 0x00000000, 0x80000000, 0x80000000, 0x00000000 }; + + // Load the full matrix into registers + __m128 _L1 = matrix.template packet(0); + __m128 _L2 = matrix.template packet(4); + __m128 _L3 = matrix.template packet(8); + __m128 _L4 = matrix.template packet(12); + + // The inverse is calculated using "Divide and Conquer" technique. The + // original matrix is divide into four 2x2 sub-matrices. Since each + // register holds four matrix element, the smaller matrices are + // represented as a registers. Hence we get a better locality of the + // calculations. + + __m128 A, B, C, D;// the four sub-matrices + if (!StorageOrdersMatch) { + A = _mm_unpacklo_ps(_L1, _L2); + B = _mm_unpacklo_ps(_L3, _L4); + C = _mm_unpackhi_ps(_L1, _L2); + D = _mm_unpackhi_ps(_L3, _L4); + } else { + A = _mm_movelh_ps(_L1, _L2); + B = _mm_movehl_ps(_L2, _L1); + C = _mm_movelh_ps(_L3, _L4); + D = _mm_movehl_ps(_L4, _L3); + } + + __m128 iA, iB, iC, iD,// partial inverse of the sub-matrices + DC, AB; + __m128 dA, dB, dC, dD;// determinant of the sub-matrices + __m128 det, d, d1, d2; + __m128 rd;// reciprocal of the determinant + + // AB = A# * B + AB = _mm_mul_ps(_mm_shuffle_ps(A, A, 0x0F), B); + AB = _mm_sub_ps(AB, _mm_mul_ps(_mm_shuffle_ps(A, A, 0xA5), _mm_shuffle_ps(B, B, 0x4E))); + // DC = D# * C + DC = _mm_mul_ps(_mm_shuffle_ps(D, D, 0x0F), C); + DC = _mm_sub_ps(DC, _mm_mul_ps(_mm_shuffle_ps(D, D, 0xA5), _mm_shuffle_ps(C, C, 0x4E))); + + // dA = |A| + dA = _mm_mul_ps(_mm_shuffle_ps(A, A, 0x5F), A); + dA = _mm_sub_ss(dA, _mm_movehl_ps(dA, dA)); + // dB = |B| + dB = _mm_mul_ps(_mm_shuffle_ps(B, B, 0x5F), B); + dB = _mm_sub_ss(dB, _mm_movehl_ps(dB, dB)); + + // dC = |C| + dC = _mm_mul_ps(_mm_shuffle_ps(C, C, 0x5F), C); + dC = _mm_sub_ss(dC, _mm_movehl_ps(dC, dC)); + // dD = |D| + dD = _mm_mul_ps(_mm_shuffle_ps(D, D, 0x5F), D); + dD = _mm_sub_ss(dD, _mm_movehl_ps(dD, dD)); + + // d = trace(AB*DC) = trace(A#*B*D#*C) + d = _mm_mul_ps(_mm_shuffle_ps(DC, DC, 0xD8), AB); + + // iD = C*A#*B + iD = _mm_mul_ps(_mm_shuffle_ps(C, C, 0xA0), _mm_movelh_ps(AB, AB)); + iD = _mm_add_ps(iD, _mm_mul_ps(_mm_shuffle_ps(C, C, 0xF5), _mm_movehl_ps(AB, AB))); + // iA = B*D#*C + iA = _mm_mul_ps(_mm_shuffle_ps(B, B, 0xA0), _mm_movelh_ps(DC, DC)); + iA = _mm_add_ps(iA, _mm_mul_ps(_mm_shuffle_ps(B, B, 0xF5), _mm_movehl_ps(DC, DC))); + + // d = trace(AB*DC) = trace(A#*B*D#*C) [continue] + d = _mm_add_ps(d, _mm_movehl_ps(d, d)); + d = _mm_add_ss(d, _mm_shuffle_ps(d, d, 1)); + d1 = _mm_mul_ss(dA, dD); + d2 = _mm_mul_ss(dB, dC); + + // iD = D*|A| - C*A#*B + iD = _mm_sub_ps(_mm_mul_ps(D, _mm_shuffle_ps(dA, dA, 0)), iD); + + // iA = A*|D| - B*D#*C; + iA = _mm_sub_ps(_mm_mul_ps(A, _mm_shuffle_ps(dD, dD, 0)), iA); + + // det = |A|*|D| + |B|*|C| - trace(A#*B*D#*C) + det = _mm_sub_ss(_mm_add_ss(d1, d2), d); + rd = _mm_div_ss(_mm_set_ss(1.0f), det); + + // #ifdef ZERO_SINGULAR + // rd = _mm_and_ps(_mm_cmpneq_ss(det,_mm_setzero_ps()), rd); + // #endif + + // iB = D * (A#B)# = D*B#*A + iB = _mm_mul_ps(D, _mm_shuffle_ps(AB, AB, 0x33)); + iB = _mm_sub_ps(iB, _mm_mul_ps(_mm_shuffle_ps(D, D, 0xB1), _mm_shuffle_ps(AB, AB, 0x66))); + // iC = A * (D#C)# = A*C#*D + iC = _mm_mul_ps(A, _mm_shuffle_ps(DC, DC, 0x33)); + iC = _mm_sub_ps(iC, _mm_mul_ps(_mm_shuffle_ps(A, A, 0xB1), _mm_shuffle_ps(DC, DC, 0x66))); + + rd = _mm_shuffle_ps(rd, rd, 0); + rd = _mm_xor_ps(rd, _mm_load_ps((float *)_Sign_PNNP)); + + // iB = C*|B| - D*B#*A + iB = _mm_sub_ps(_mm_mul_ps(C, _mm_shuffle_ps(dB, dB, 0)), iB); + + // iC = B*|C| - A*C#*D; + iC = _mm_sub_ps(_mm_mul_ps(B, _mm_shuffle_ps(dC, dC, 0)), iC); + + // iX = iX / det + iA = _mm_mul_ps(rd, iA); + iB = _mm_mul_ps(rd, iB); + iC = _mm_mul_ps(rd, iC); + iD = _mm_mul_ps(rd, iD); + + Index res_stride = result.outerStride(); + float *res = result.data(); + pstoret(res + 0, _mm_shuffle_ps(iA, iB, 0x77)); + pstoret(res + res_stride, _mm_shuffle_ps(iA, iB, 0x22)); + pstoret(res + 2 * res_stride, _mm_shuffle_ps(iC, iD, 0x77)); + pstoret(res + 3 * res_stride, _mm_shuffle_ps(iC, iD, 0x22)); } - - __m128 iA, iB, iC, iD, // partial inverse of the sub-matrices - DC, AB; - __m128 dA, dB, dC, dD; // determinant of the sub-matrices - __m128 det, d, d1, d2; - __m128 rd; // reciprocal of the determinant - - // AB = A# * B - AB = _mm_mul_ps(_mm_shuffle_ps(A,A,0x0F), B); - AB = _mm_sub_ps(AB,_mm_mul_ps(_mm_shuffle_ps(A,A,0xA5), _mm_shuffle_ps(B,B,0x4E))); - // DC = D# * C - DC = _mm_mul_ps(_mm_shuffle_ps(D,D,0x0F), C); - DC = _mm_sub_ps(DC,_mm_mul_ps(_mm_shuffle_ps(D,D,0xA5), _mm_shuffle_ps(C,C,0x4E))); - - // dA = |A| - dA = _mm_mul_ps(_mm_shuffle_ps(A, A, 0x5F),A); - dA = _mm_sub_ss(dA, _mm_movehl_ps(dA,dA)); - // dB = |B| - dB = _mm_mul_ps(_mm_shuffle_ps(B, B, 0x5F),B); - dB = _mm_sub_ss(dB, _mm_movehl_ps(dB,dB)); - - // dC = |C| - dC = _mm_mul_ps(_mm_shuffle_ps(C, C, 0x5F),C); - dC = _mm_sub_ss(dC, _mm_movehl_ps(dC,dC)); - // dD = |D| - dD = _mm_mul_ps(_mm_shuffle_ps(D, D, 0x5F),D); - dD = _mm_sub_ss(dD, _mm_movehl_ps(dD,dD)); - - // d = trace(AB*DC) = trace(A#*B*D#*C) - d = _mm_mul_ps(_mm_shuffle_ps(DC,DC,0xD8),AB); - - // iD = C*A#*B - iD = _mm_mul_ps(_mm_shuffle_ps(C,C,0xA0), _mm_movelh_ps(AB,AB)); - iD = _mm_add_ps(iD,_mm_mul_ps(_mm_shuffle_ps(C,C,0xF5), _mm_movehl_ps(AB,AB))); - // iA = B*D#*C - iA = _mm_mul_ps(_mm_shuffle_ps(B,B,0xA0), _mm_movelh_ps(DC,DC)); - iA = _mm_add_ps(iA,_mm_mul_ps(_mm_shuffle_ps(B,B,0xF5), _mm_movehl_ps(DC,DC))); - - // d = trace(AB*DC) = trace(A#*B*D#*C) [continue] - d = _mm_add_ps(d, _mm_movehl_ps(d, d)); - d = _mm_add_ss(d, _mm_shuffle_ps(d, d, 1)); - d1 = _mm_mul_ss(dA,dD); - d2 = _mm_mul_ss(dB,dC); - - // iD = D*|A| - C*A#*B - iD = _mm_sub_ps(_mm_mul_ps(D,_mm_shuffle_ps(dA,dA,0)), iD); - - // iA = A*|D| - B*D#*C; - iA = _mm_sub_ps(_mm_mul_ps(A,_mm_shuffle_ps(dD,dD,0)), iA); - - // det = |A|*|D| + |B|*|C| - trace(A#*B*D#*C) - det = _mm_sub_ss(_mm_add_ss(d1,d2),d); - rd = _mm_div_ss(_mm_set_ss(1.0f), det); - -// #ifdef ZERO_SINGULAR -// rd = _mm_and_ps(_mm_cmpneq_ss(det,_mm_setzero_ps()), rd); -// #endif - - // iB = D * (A#B)# = D*B#*A - iB = _mm_mul_ps(D, _mm_shuffle_ps(AB,AB,0x33)); - iB = _mm_sub_ps(iB, _mm_mul_ps(_mm_shuffle_ps(D,D,0xB1), _mm_shuffle_ps(AB,AB,0x66))); - // iC = A * (D#C)# = A*C#*D - iC = _mm_mul_ps(A, _mm_shuffle_ps(DC,DC,0x33)); - iC = _mm_sub_ps(iC, _mm_mul_ps(_mm_shuffle_ps(A,A,0xB1), _mm_shuffle_ps(DC,DC,0x66))); - - rd = _mm_shuffle_ps(rd,rd,0); - rd = _mm_xor_ps(rd, _mm_load_ps((float*)_Sign_PNNP)); - - // iB = C*|B| - D*B#*A - iB = _mm_sub_ps(_mm_mul_ps(C,_mm_shuffle_ps(dB,dB,0)), iB); - - // iC = B*|C| - A*C#*D; - iC = _mm_sub_ps(_mm_mul_ps(B,_mm_shuffle_ps(dC,dC,0)), iC); - - // iX = iX / det - iA = _mm_mul_ps(rd,iA); - iB = _mm_mul_ps(rd,iB); - iC = _mm_mul_ps(rd,iC); - iD = _mm_mul_ps(rd,iD); - - Index res_stride = result.outerStride(); - float* res = result.data(); - pstoret(res+0, _mm_shuffle_ps(iA,iB,0x77)); - pstoret(res+res_stride, _mm_shuffle_ps(iA,iB,0x22)); - pstoret(res+2*res_stride, _mm_shuffle_ps(iC,iD,0x77)); - pstoret(res+3*res_stride, _mm_shuffle_ps(iC,iD,0x22)); - } - -}; - -template -struct compute_inverse_size4 -{ - enum { - MatrixAlignment = traits::Alignment, - ResultAlignment = traits::Alignment, - StorageOrdersMatch = (MatrixType::Flags&RowMajorBit) == (ResultType::Flags&RowMajorBit) }; - typedef typename conditional<(MatrixType::Flags&LinearAccessBit),MatrixType const &,typename MatrixType::PlainObject>::type ActualMatrixType; - - static void run(const MatrixType& mat, ResultType& result) + + template + struct compute_inverse_size4 { - ActualMatrixType matrix(mat); - const __m128d _Sign_NP = _mm_castsi128_pd(_mm_set_epi32(0x0,0x0,0x80000000,0x0)); - const __m128d _Sign_PN = _mm_castsi128_pd(_mm_set_epi32(0x80000000,0x0,0x0,0x0)); - - // The inverse is calculated using "Divide and Conquer" technique. The - // original matrix is divide into four 2x2 sub-matrices. Since each - // register of the matrix holds two elements, the smaller matrices are - // consisted of two registers. Hence we get a better locality of the - // calculations. - - // the four sub-matrices - __m128d A1, A2, B1, B2, C1, C2, D1, D2; - - if(StorageOrdersMatch) + enum { + MatrixAlignment = traits::Alignment, + ResultAlignment = traits::Alignment, + StorageOrdersMatch = (MatrixType::Flags & RowMajorBit) == (ResultType::Flags & RowMajorBit) + }; + typedef typename conditional<(MatrixType::Flags & LinearAccessBit), + MatrixType const &, + typename MatrixType::PlainObject>::type ActualMatrixType; + + static void run(const MatrixType &mat, ResultType &result) { - A1 = matrix.template packet( 0); B1 = matrix.template packet( 2); - A2 = matrix.template packet( 4); B2 = matrix.template packet( 6); - C1 = matrix.template packet( 8); D1 = matrix.template packet(10); - C2 = matrix.template packet(12); D2 = matrix.template packet(14); + ActualMatrixType matrix(mat); + const __m128d _Sign_NP = _mm_castsi128_pd(_mm_set_epi32(0x0, 0x0, 0x80000000, 0x0)); + const __m128d _Sign_PN = _mm_castsi128_pd(_mm_set_epi32(0x80000000, 0x0, 0x0, 0x0)); + + // The inverse is calculated using "Divide and Conquer" technique. The + // original matrix is divide into four 2x2 sub-matrices. Since each + // register of the matrix holds two elements, the smaller matrices are + // consisted of two registers. Hence we get a better locality of the + // calculations. + + // the four sub-matrices + __m128d A1, A2, B1, B2, C1, C2, D1, D2; + + if (StorageOrdersMatch) { + A1 = matrix.template packet(0); + B1 = matrix.template packet(2); + A2 = matrix.template packet(4); + B2 = matrix.template packet(6); + C1 = matrix.template packet(8); + D1 = matrix.template packet(10); + C2 = matrix.template packet(12); + D2 = matrix.template packet(14); + } else { + __m128d tmp; + A1 = matrix.template packet(0); + C1 = matrix.template packet(2); + A2 = matrix.template packet(4); + C2 = matrix.template packet(6); + tmp = A1; + A1 = _mm_unpacklo_pd(A1, A2); + A2 = _mm_unpackhi_pd(tmp, A2); + tmp = C1; + C1 = _mm_unpacklo_pd(C1, C2); + C2 = _mm_unpackhi_pd(tmp, C2); + + B1 = matrix.template packet(8); + D1 = matrix.template packet(10); + B2 = matrix.template packet(12); + D2 = matrix.template packet(14); + tmp = B1; + B1 = _mm_unpacklo_pd(B1, B2); + B2 = _mm_unpackhi_pd(tmp, B2); + tmp = D1; + D1 = _mm_unpacklo_pd(D1, D2); + D2 = _mm_unpackhi_pd(tmp, D2); + } + + __m128d iA1, iA2, iB1, iB2, iC1, iC2, iD1, iD2,// partial invese of the sub-matrices + DC1, DC2, AB1, AB2; + __m128d dA, dB, dC, dD;// determinant of the sub-matrices + __m128d det, d1, d2, rd; + + // dA = |A| + dA = _mm_shuffle_pd(A2, A2, 1); + dA = _mm_mul_pd(A1, dA); + dA = _mm_sub_sd(dA, _mm_shuffle_pd(dA, dA, 3)); + // dB = |B| + dB = _mm_shuffle_pd(B2, B2, 1); + dB = _mm_mul_pd(B1, dB); + dB = _mm_sub_sd(dB, _mm_shuffle_pd(dB, dB, 3)); + + // AB = A# * B + AB1 = _mm_mul_pd(B1, _mm_shuffle_pd(A2, A2, 3)); + AB2 = _mm_mul_pd(B2, _mm_shuffle_pd(A1, A1, 0)); + AB1 = _mm_sub_pd(AB1, _mm_mul_pd(B2, _mm_shuffle_pd(A1, A1, 3))); + AB2 = _mm_sub_pd(AB2, _mm_mul_pd(B1, _mm_shuffle_pd(A2, A2, 0))); + + // dC = |C| + dC = _mm_shuffle_pd(C2, C2, 1); + dC = _mm_mul_pd(C1, dC); + dC = _mm_sub_sd(dC, _mm_shuffle_pd(dC, dC, 3)); + // dD = |D| + dD = _mm_shuffle_pd(D2, D2, 1); + dD = _mm_mul_pd(D1, dD); + dD = _mm_sub_sd(dD, _mm_shuffle_pd(dD, dD, 3)); + + // DC = D# * C + DC1 = _mm_mul_pd(C1, _mm_shuffle_pd(D2, D2, 3)); + DC2 = _mm_mul_pd(C2, _mm_shuffle_pd(D1, D1, 0)); + DC1 = _mm_sub_pd(DC1, _mm_mul_pd(C2, _mm_shuffle_pd(D1, D1, 3))); + DC2 = _mm_sub_pd(DC2, _mm_mul_pd(C1, _mm_shuffle_pd(D2, D2, 0))); + + // rd = trace(AB*DC) = trace(A#*B*D#*C) + d1 = _mm_mul_pd(AB1, _mm_shuffle_pd(DC1, DC2, 0)); + d2 = _mm_mul_pd(AB2, _mm_shuffle_pd(DC1, DC2, 3)); + rd = _mm_add_pd(d1, d2); + rd = _mm_add_sd(rd, _mm_shuffle_pd(rd, rd, 3)); + + // iD = C*A#*B + iD1 = _mm_mul_pd(AB1, _mm_shuffle_pd(C1, C1, 0)); + iD2 = _mm_mul_pd(AB1, _mm_shuffle_pd(C2, C2, 0)); + iD1 = _mm_add_pd(iD1, _mm_mul_pd(AB2, _mm_shuffle_pd(C1, C1, 3))); + iD2 = _mm_add_pd(iD2, _mm_mul_pd(AB2, _mm_shuffle_pd(C2, C2, 3))); + + // iA = B*D#*C + iA1 = _mm_mul_pd(DC1, _mm_shuffle_pd(B1, B1, 0)); + iA2 = _mm_mul_pd(DC1, _mm_shuffle_pd(B2, B2, 0)); + iA1 = _mm_add_pd(iA1, _mm_mul_pd(DC2, _mm_shuffle_pd(B1, B1, 3))); + iA2 = _mm_add_pd(iA2, _mm_mul_pd(DC2, _mm_shuffle_pd(B2, B2, 3))); + + // iD = D*|A| - C*A#*B + dA = _mm_shuffle_pd(dA, dA, 0); + iD1 = _mm_sub_pd(_mm_mul_pd(D1, dA), iD1); + iD2 = _mm_sub_pd(_mm_mul_pd(D2, dA), iD2); + + // iA = A*|D| - B*D#*C; + dD = _mm_shuffle_pd(dD, dD, 0); + iA1 = _mm_sub_pd(_mm_mul_pd(A1, dD), iA1); + iA2 = _mm_sub_pd(_mm_mul_pd(A2, dD), iA2); + + d1 = _mm_mul_sd(dA, dD); + d2 = _mm_mul_sd(dB, dC); + + // iB = D * (A#B)# = D*B#*A + iB1 = _mm_mul_pd(D1, _mm_shuffle_pd(AB2, AB1, 1)); + iB2 = _mm_mul_pd(D2, _mm_shuffle_pd(AB2, AB1, 1)); + iB1 = _mm_sub_pd(iB1, _mm_mul_pd(_mm_shuffle_pd(D1, D1, 1), _mm_shuffle_pd(AB2, AB1, 2))); + iB2 = _mm_sub_pd(iB2, _mm_mul_pd(_mm_shuffle_pd(D2, D2, 1), _mm_shuffle_pd(AB2, AB1, 2))); + + // det = |A|*|D| + |B|*|C| - trace(A#*B*D#*C) + det = _mm_add_sd(d1, d2); + det = _mm_sub_sd(det, rd); + + // iC = A * (D#C)# = A*C#*D + iC1 = _mm_mul_pd(A1, _mm_shuffle_pd(DC2, DC1, 1)); + iC2 = _mm_mul_pd(A2, _mm_shuffle_pd(DC2, DC1, 1)); + iC1 = _mm_sub_pd(iC1, _mm_mul_pd(_mm_shuffle_pd(A1, A1, 1), _mm_shuffle_pd(DC2, DC1, 2))); + iC2 = _mm_sub_pd(iC2, _mm_mul_pd(_mm_shuffle_pd(A2, A2, 1), _mm_shuffle_pd(DC2, DC1, 2))); + + rd = _mm_div_sd(_mm_set_sd(1.0), det); + // #ifdef ZERO_SINGULAR + // rd = _mm_and_pd(_mm_cmpneq_sd(det,_mm_setzero_pd()), rd); + // #endif + rd = _mm_shuffle_pd(rd, rd, 0); + + // iB = C*|B| - D*B#*A + dB = _mm_shuffle_pd(dB, dB, 0); + iB1 = _mm_sub_pd(_mm_mul_pd(C1, dB), iB1); + iB2 = _mm_sub_pd(_mm_mul_pd(C2, dB), iB2); + + d1 = _mm_xor_pd(rd, _Sign_PN); + d2 = _mm_xor_pd(rd, _Sign_NP); + + // iC = B*|C| - A*C#*D; + dC = _mm_shuffle_pd(dC, dC, 0); + iC1 = _mm_sub_pd(_mm_mul_pd(B1, dC), iC1); + iC2 = _mm_sub_pd(_mm_mul_pd(B2, dC), iC2); + + Index res_stride = result.outerStride(); + double *res = result.data(); + pstoret(res + 0, _mm_mul_pd(_mm_shuffle_pd(iA2, iA1, 3), d1)); + pstoret(res + res_stride, _mm_mul_pd(_mm_shuffle_pd(iA2, iA1, 0), d2)); + pstoret(res + 2, _mm_mul_pd(_mm_shuffle_pd(iB2, iB1, 3), d1)); + pstoret(res + res_stride + 2, _mm_mul_pd(_mm_shuffle_pd(iB2, iB1, 0), d2)); + pstoret(res + 2 * res_stride, _mm_mul_pd(_mm_shuffle_pd(iC2, iC1, 3), d1)); + pstoret(res + 3 * res_stride, _mm_mul_pd(_mm_shuffle_pd(iC2, iC1, 0), d2)); + pstoret(res + 2 * res_stride + 2, _mm_mul_pd(_mm_shuffle_pd(iD2, iD1, 3), d1)); + pstoret(res + 3 * res_stride + 2, _mm_mul_pd(_mm_shuffle_pd(iD2, iD1, 0), d2)); } - else - { - __m128d tmp; - A1 = matrix.template packet( 0); C1 = matrix.template packet( 2); - A2 = matrix.template packet( 4); C2 = matrix.template packet( 6); - tmp = A1; - A1 = _mm_unpacklo_pd(A1,A2); - A2 = _mm_unpackhi_pd(tmp,A2); - tmp = C1; - C1 = _mm_unpacklo_pd(C1,C2); - C2 = _mm_unpackhi_pd(tmp,C2); - - B1 = matrix.template packet( 8); D1 = matrix.template packet(10); - B2 = matrix.template packet(12); D2 = matrix.template packet(14); - tmp = B1; - B1 = _mm_unpacklo_pd(B1,B2); - B2 = _mm_unpackhi_pd(tmp,B2); - tmp = D1; - D1 = _mm_unpacklo_pd(D1,D2); - D2 = _mm_unpackhi_pd(tmp,D2); - } - - __m128d iA1, iA2, iB1, iB2, iC1, iC2, iD1, iD2, // partial invese of the sub-matrices - DC1, DC2, AB1, AB2; - __m128d dA, dB, dC, dD; // determinant of the sub-matrices - __m128d det, d1, d2, rd; - - // dA = |A| - dA = _mm_shuffle_pd(A2, A2, 1); - dA = _mm_mul_pd(A1, dA); - dA = _mm_sub_sd(dA, _mm_shuffle_pd(dA,dA,3)); - // dB = |B| - dB = _mm_shuffle_pd(B2, B2, 1); - dB = _mm_mul_pd(B1, dB); - dB = _mm_sub_sd(dB, _mm_shuffle_pd(dB,dB,3)); - - // AB = A# * B - AB1 = _mm_mul_pd(B1, _mm_shuffle_pd(A2,A2,3)); - AB2 = _mm_mul_pd(B2, _mm_shuffle_pd(A1,A1,0)); - AB1 = _mm_sub_pd(AB1, _mm_mul_pd(B2, _mm_shuffle_pd(A1,A1,3))); - AB2 = _mm_sub_pd(AB2, _mm_mul_pd(B1, _mm_shuffle_pd(A2,A2,0))); - - // dC = |C| - dC = _mm_shuffle_pd(C2, C2, 1); - dC = _mm_mul_pd(C1, dC); - dC = _mm_sub_sd(dC, _mm_shuffle_pd(dC,dC,3)); - // dD = |D| - dD = _mm_shuffle_pd(D2, D2, 1); - dD = _mm_mul_pd(D1, dD); - dD = _mm_sub_sd(dD, _mm_shuffle_pd(dD,dD,3)); - - // DC = D# * C - DC1 = _mm_mul_pd(C1, _mm_shuffle_pd(D2,D2,3)); - DC2 = _mm_mul_pd(C2, _mm_shuffle_pd(D1,D1,0)); - DC1 = _mm_sub_pd(DC1, _mm_mul_pd(C2, _mm_shuffle_pd(D1,D1,3))); - DC2 = _mm_sub_pd(DC2, _mm_mul_pd(C1, _mm_shuffle_pd(D2,D2,0))); - - // rd = trace(AB*DC) = trace(A#*B*D#*C) - d1 = _mm_mul_pd(AB1, _mm_shuffle_pd(DC1, DC2, 0)); - d2 = _mm_mul_pd(AB2, _mm_shuffle_pd(DC1, DC2, 3)); - rd = _mm_add_pd(d1, d2); - rd = _mm_add_sd(rd, _mm_shuffle_pd(rd, rd,3)); - - // iD = C*A#*B - iD1 = _mm_mul_pd(AB1, _mm_shuffle_pd(C1,C1,0)); - iD2 = _mm_mul_pd(AB1, _mm_shuffle_pd(C2,C2,0)); - iD1 = _mm_add_pd(iD1, _mm_mul_pd(AB2, _mm_shuffle_pd(C1,C1,3))); - iD2 = _mm_add_pd(iD2, _mm_mul_pd(AB2, _mm_shuffle_pd(C2,C2,3))); - - // iA = B*D#*C - iA1 = _mm_mul_pd(DC1, _mm_shuffle_pd(B1,B1,0)); - iA2 = _mm_mul_pd(DC1, _mm_shuffle_pd(B2,B2,0)); - iA1 = _mm_add_pd(iA1, _mm_mul_pd(DC2, _mm_shuffle_pd(B1,B1,3))); - iA2 = _mm_add_pd(iA2, _mm_mul_pd(DC2, _mm_shuffle_pd(B2,B2,3))); - - // iD = D*|A| - C*A#*B - dA = _mm_shuffle_pd(dA,dA,0); - iD1 = _mm_sub_pd(_mm_mul_pd(D1, dA), iD1); - iD2 = _mm_sub_pd(_mm_mul_pd(D2, dA), iD2); - - // iA = A*|D| - B*D#*C; - dD = _mm_shuffle_pd(dD,dD,0); - iA1 = _mm_sub_pd(_mm_mul_pd(A1, dD), iA1); - iA2 = _mm_sub_pd(_mm_mul_pd(A2, dD), iA2); - - d1 = _mm_mul_sd(dA, dD); - d2 = _mm_mul_sd(dB, dC); - - // iB = D * (A#B)# = D*B#*A - iB1 = _mm_mul_pd(D1, _mm_shuffle_pd(AB2,AB1,1)); - iB2 = _mm_mul_pd(D2, _mm_shuffle_pd(AB2,AB1,1)); - iB1 = _mm_sub_pd(iB1, _mm_mul_pd(_mm_shuffle_pd(D1,D1,1), _mm_shuffle_pd(AB2,AB1,2))); - iB2 = _mm_sub_pd(iB2, _mm_mul_pd(_mm_shuffle_pd(D2,D2,1), _mm_shuffle_pd(AB2,AB1,2))); - - // det = |A|*|D| + |B|*|C| - trace(A#*B*D#*C) - det = _mm_add_sd(d1, d2); - det = _mm_sub_sd(det, rd); - - // iC = A * (D#C)# = A*C#*D - iC1 = _mm_mul_pd(A1, _mm_shuffle_pd(DC2,DC1,1)); - iC2 = _mm_mul_pd(A2, _mm_shuffle_pd(DC2,DC1,1)); - iC1 = _mm_sub_pd(iC1, _mm_mul_pd(_mm_shuffle_pd(A1,A1,1), _mm_shuffle_pd(DC2,DC1,2))); - iC2 = _mm_sub_pd(iC2, _mm_mul_pd(_mm_shuffle_pd(A2,A2,1), _mm_shuffle_pd(DC2,DC1,2))); - - rd = _mm_div_sd(_mm_set_sd(1.0), det); -// #ifdef ZERO_SINGULAR -// rd = _mm_and_pd(_mm_cmpneq_sd(det,_mm_setzero_pd()), rd); -// #endif - rd = _mm_shuffle_pd(rd,rd,0); - - // iB = C*|B| - D*B#*A - dB = _mm_shuffle_pd(dB,dB,0); - iB1 = _mm_sub_pd(_mm_mul_pd(C1, dB), iB1); - iB2 = _mm_sub_pd(_mm_mul_pd(C2, dB), iB2); - - d1 = _mm_xor_pd(rd, _Sign_PN); - d2 = _mm_xor_pd(rd, _Sign_NP); - - // iC = B*|C| - A*C#*D; - dC = _mm_shuffle_pd(dC,dC,0); - iC1 = _mm_sub_pd(_mm_mul_pd(B1, dC), iC1); - iC2 = _mm_sub_pd(_mm_mul_pd(B2, dC), iC2); - - Index res_stride = result.outerStride(); - double* res = result.data(); - pstoret(res+0, _mm_mul_pd(_mm_shuffle_pd(iA2, iA1, 3), d1)); - pstoret(res+res_stride, _mm_mul_pd(_mm_shuffle_pd(iA2, iA1, 0), d2)); - pstoret(res+2, _mm_mul_pd(_mm_shuffle_pd(iB2, iB1, 3), d1)); - pstoret(res+res_stride+2, _mm_mul_pd(_mm_shuffle_pd(iB2, iB1, 0), d2)); - pstoret(res+2*res_stride, _mm_mul_pd(_mm_shuffle_pd(iC2, iC1, 3), d1)); - pstoret(res+3*res_stride, _mm_mul_pd(_mm_shuffle_pd(iC2, iC1, 0), d2)); - pstoret(res+2*res_stride+2,_mm_mul_pd(_mm_shuffle_pd(iD2, iD1, 3), d1)); - pstoret(res+3*res_stride+2,_mm_mul_pd(_mm_shuffle_pd(iD2, iD1, 0), d2)); - } -}; - -} // end namespace internal - -} // end namespace Eigen - -#endif // EIGEN_INVERSE_SSE_H + }; + +}// end namespace internal + +}// end namespace Eigen + +#endif// EIGEN_INVERSE_SSE_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/MetisSupport/MetisSupport.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/MetisSupport/MetisSupport.h index 4c15304a..7373226b 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/MetisSupport/MetisSupport.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/MetisSupport/MetisSupport.h @@ -12,126 +12,111 @@ namespace Eigen { /** * Get the fill-reducing ordering from the METIS package - * - * If A is the original matrix and Ap is the permuted matrix, + * + * If A is the original matrix and Ap is the permuted matrix, * the fill-reducing permutation is defined as follows : - * Row (column) i of A is the matperm(i) row (column) of Ap. + * Row (column) i of A is the matperm(i) row (column) of Ap. * WARNING: As computed by METIS, this corresponds to the vector iperm (instead of perm) */ -template -class MetisOrdering +template class MetisOrdering { public: - typedef PermutationMatrix PermutationType; - typedef Matrix IndexVector; - - template - void get_symmetrized_graph(const MatrixType& A) + typedef PermutationMatrix PermutationType; + typedef Matrix IndexVector; + + template void get_symmetrized_graph(const MatrixType &A) { - Index m = A.cols(); + Index m = A.cols(); eigen_assert((A.rows() == A.cols()) && "ONLY FOR SQUARED MATRICES"); - // Get the transpose of the input matrix - MatrixType At = A.transpose(); + // Get the transpose of the input matrix + MatrixType At = A.transpose(); // Get the number of nonzeros elements in each row/col of At+A - Index TotNz = 0; - IndexVector visited(m); - visited.setConstant(-1); - for (StorageIndex j = 0; j < m; j++) - { + Index TotNz = 0; + IndexVector visited(m); + visited.setConstant(-1); + for (StorageIndex j = 0; j < m; j++) { // Compute the union structure of of A(j,:) and At(j,:) - visited(j) = j; // Do not include the diagonal element + visited(j) = j;// Do not include the diagonal element // Get the nonzeros in row/column j of A - for (typename MatrixType::InnerIterator it(A, j); it; ++it) - { - Index idx = it.index(); // Get the row index (for column major) or column index (for row major) - if (visited(idx) != j ) - { - visited(idx) = j; - ++TotNz; + for (typename MatrixType::InnerIterator it(A, j); it; ++it) { + Index idx = it.index();// Get the row index (for column major) or column index (for row major) + if (visited(idx) != j) { + visited(idx) = j; + ++TotNz; } } - //Get the nonzeros in row/column j of At - for (typename MatrixType::InnerIterator it(At, j); it; ++it) - { - Index idx = it.index(); - if(visited(idx) != j) - { - visited(idx) = j; - ++TotNz; + // Get the nonzeros in row/column j of At + for (typename MatrixType::InnerIterator it(At, j); it; ++it) { + Index idx = it.index(); + if (visited(idx) != j) { + visited(idx) = j; + ++TotNz; } } } // Reserve place for A + At - m_indexPtr.resize(m+1); - m_innerIndices.resize(TotNz); + m_indexPtr.resize(m + 1); + m_innerIndices.resize(TotNz); + + // Now compute the real adjacency list of each column/row + visited.setConstant(-1); + StorageIndex CurNz = 0; + for (StorageIndex j = 0; j < m; j++) { + m_indexPtr(j) = CurNz; - // Now compute the real adjacency list of each column/row - visited.setConstant(-1); - StorageIndex CurNz = 0; - for (StorageIndex j = 0; j < m; j++) - { - m_indexPtr(j) = CurNz; - - visited(j) = j; // Do not include the diagonal element + visited(j) = j;// Do not include the diagonal element // Add the pattern of row/column j of A to A+At - for (typename MatrixType::InnerIterator it(A,j); it; ++it) - { - StorageIndex idx = it.index(); // Get the row index (for column major) or column index (for row major) - if (visited(idx) != j ) - { - visited(idx) = j; - m_innerIndices(CurNz) = idx; - CurNz++; + for (typename MatrixType::InnerIterator it(A, j); it; ++it) { + StorageIndex idx = it.index();// Get the row index (for column major) or column index (for row major) + if (visited(idx) != j) { + visited(idx) = j; + m_innerIndices(CurNz) = idx; + CurNz++; } } - //Add the pattern of row/column j of At to A+At - for (typename MatrixType::InnerIterator it(At, j); it; ++it) - { - StorageIndex idx = it.index(); - if(visited(idx) != j) - { - visited(idx) = j; - m_innerIndices(CurNz) = idx; - ++CurNz; + // Add the pattern of row/column j of At to A+At + for (typename MatrixType::InnerIterator it(At, j); it; ++it) { + StorageIndex idx = it.index(); + if (visited(idx) != j) { + visited(idx) = j; + m_innerIndices(CurNz) = idx; + ++CurNz; } } } - m_indexPtr(m) = CurNz; + m_indexPtr(m) = CurNz; } - - template - void operator() (const MatrixType& A, PermutationType& matperm) + + template void operator()(const MatrixType &A, PermutationType &matperm) { - StorageIndex m = internal::convert_index(A.cols()); // must be StorageIndex, because it is passed by address to METIS - IndexVector perm(m),iperm(m); - // First, symmetrize the matrix graph. - get_symmetrized_graph(A); - int output_error; - - // Call the fill-reducing routine from METIS - output_error = METIS_NodeND(&m, m_indexPtr.data(), m_innerIndices.data(), NULL, NULL, perm.data(), iperm.data()); - - if(output_error != METIS_OK) - { - //FIXME The ordering interface should define a class of possible errors - std::cerr << "ERROR WHILE CALLING THE METIS PACKAGE \n"; - return; + StorageIndex m = + internal::convert_index(A.cols());// must be StorageIndex, because it is passed by address to METIS + IndexVector perm(m), iperm(m); + // First, symmetrize the matrix graph. + get_symmetrized_graph(A); + int output_error; + + // Call the fill-reducing routine from METIS + output_error = METIS_NodeND(&m, m_indexPtr.data(), m_innerIndices.data(), NULL, NULL, perm.data(), iperm.data()); + + if (output_error != METIS_OK) { + // FIXME The ordering interface should define a class of possible errors + std::cerr << "ERROR WHILE CALLING THE METIS PACKAGE \n"; + return; } - - // Get the fill-reducing permutation - //NOTE: If Ap is the permuted matrix then perm and iperm vectors are defined as follows + + // Get the fill-reducing permutation + // NOTE: If Ap is the permuted matrix then perm and iperm vectors are defined as follows // Row (column) i of Ap is the perm(i) row(column) of A, and row (column) i of A is the iperm(i) row(column) of Ap - - matperm.resize(m); - for (int j = 0; j < m; j++) - matperm.indices()(iperm(j)) = j; - + + matperm.resize(m); + for (int j = 0; j < m; j++) matperm.indices()(iperm(j)) = j; } - - protected: - IndexVector m_indexPtr; // Pointer to the adjacenccy list of each row/column - IndexVector m_innerIndices; // Adjacency list + +protected: + IndexVector m_indexPtr;// Pointer to the adjacenccy list of each row/column + IndexVector m_innerIndices;// Adjacency list }; -}// end namespace eigen +}// namespace Eigen #endif diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/OrderingMethods/Amd.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/OrderingMethods/Amd.h index f91ecb24..831320e2 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/OrderingMethods/Amd.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/OrderingMethods/Amd.h @@ -31,415 +31,385 @@ Foundation, Inc., 51 Franklin St, Fifth Floor, Boston, MA 02110-1301 USA #ifndef EIGEN_SPARSE_AMD_H #define EIGEN_SPARSE_AMD_H -namespace Eigen { +namespace Eigen { namespace internal { - -template inline T amd_flip(const T& i) { return -i-2; } -template inline T amd_unflip(const T& i) { return i<0 ? amd_flip(i) : i; } -template inline bool amd_marked(const T0* w, const T1& j) { return w[j]<0; } -template inline void amd_mark(const T0* w, const T1& j) { return w[j] = amd_flip(w[j]); } - -/* clear w */ -template -static StorageIndex cs_wclear (StorageIndex mark, StorageIndex lemax, StorageIndex *w, StorageIndex n) -{ - StorageIndex k; - if(mark < 2 || (mark + lemax < 0)) - { - for(k = 0; k < n; k++) - if(w[k] != 0) - w[k] = 1; - mark = 2; - } - return (mark); /* at this point, w[0..n-1] < mark holds */ -} - -/* depth-first search and postorder of a tree rooted at node j */ -template -StorageIndex cs_tdfs(StorageIndex j, StorageIndex k, StorageIndex *head, const StorageIndex *next, StorageIndex *post, StorageIndex *stack) -{ - StorageIndex i, p, top = 0; - if(!head || !next || !post || !stack) return (-1); /* check inputs */ - stack[0] = j; /* place j on the stack */ - while (top >= 0) /* while (stack is not empty) */ + + template inline T amd_flip(const T &i) { return -i - 2; } + template inline T amd_unflip(const T &i) { return i < 0 ? amd_flip(i) : i; } + template inline bool amd_marked(const T0 *w, const T1 &j) { return w[j] < 0; } + template inline void amd_mark(const T0 *w, const T1 &j) { return w[j] = amd_flip(w[j]); } + + /* clear w */ + template + static StorageIndex cs_wclear(StorageIndex mark, StorageIndex lemax, StorageIndex *w, StorageIndex n) { - p = stack[top]; /* p = top of stack */ - i = head[p]; /* i = youngest child of p */ - if(i == -1) - { - top--; /* p has no unordered children left */ - post[k++] = p; /* node p is the kth postordered node */ - } - else - { - head[p] = next[i]; /* remove i from children of p */ - stack[++top] = i; /* start dfs on child node i */ + StorageIndex k; + if (mark < 2 || (mark + lemax < 0)) { + for (k = 0; k < n; k++) + if (w[k] != 0) w[k] = 1; + mark = 2; } + return (mark); /* at this point, w[0..n-1] < mark holds */ } - return k; -} - - -/** \internal - * \ingroup OrderingMethods_Module - * Approximate minimum degree ordering algorithm. - * - * \param[in] C the input selfadjoint matrix stored in compressed column major format. - * \param[out] perm the permutation P reducing the fill-in of the input matrix \a C - * - * Note that the input matrix \a C must be complete, that is both the upper and lower parts have to be stored, as well as the diagonal entries. - * On exit the values of C are destroyed */ -template -void minimum_degree_ordering(SparseMatrix& C, PermutationMatrix& perm) -{ - using std::sqrt; - - StorageIndex d, dk, dext, lemax = 0, e, elenk, eln, i, j, k, k1, - k2, k3, jlast, ln, dense, nzmax, mindeg = 0, nvi, nvj, nvk, mark, wnvi, - ok, nel = 0, p, p1, p2, p3, p4, pj, pk, pk1, pk2, pn, q, t, h; - - StorageIndex n = StorageIndex(C.cols()); - dense = std::max (16, StorageIndex(10 * sqrt(double(n)))); /* find dense threshold */ - dense = (std::min)(n-2, dense); - - StorageIndex cnz = StorageIndex(C.nonZeros()); - perm.resize(n+1); - t = cnz + cnz/5 + 2*n; /* add elbow room to C */ - C.resizeNonZeros(t); - - // get workspace - ei_declare_aligned_stack_constructed_variable(StorageIndex,W,8*(n+1),0); - StorageIndex* len = W; - StorageIndex* nv = W + (n+1); - StorageIndex* next = W + 2*(n+1); - StorageIndex* head = W + 3*(n+1); - StorageIndex* elen = W + 4*(n+1); - StorageIndex* degree = W + 5*(n+1); - StorageIndex* w = W + 6*(n+1); - StorageIndex* hhead = W + 7*(n+1); - StorageIndex* last = perm.indices().data(); /* use P as workspace for last */ - - /* --- Initialize quotient graph ---------------------------------------- */ - StorageIndex* Cp = C.outerIndexPtr(); - StorageIndex* Ci = C.innerIndexPtr(); - for(k = 0; k < n; k++) - len[k] = Cp[k+1] - Cp[k]; - len[n] = 0; - nzmax = t; - - for(i = 0; i <= n; i++) - { - head[i] = -1; // degree list i is empty - last[i] = -1; - next[i] = -1; - hhead[i] = -1; // hash list i is empty - nv[i] = 1; // node i is just one node - w[i] = 1; // node i is alive - elen[i] = 0; // Ek of node i is empty - degree[i] = len[i]; // degree of node i - } - mark = internal::cs_wclear(0, 0, w, n); /* clear w */ - - /* --- Initialize degree lists ------------------------------------------ */ - for(i = 0; i < n; i++) + + /* depth-first search and postorder of a tree rooted at node j */ + template + StorageIndex cs_tdfs(StorageIndex j, + StorageIndex k, + StorageIndex *head, + const StorageIndex *next, + StorageIndex *post, + StorageIndex *stack) { - bool has_diag = false; - for(p = Cp[i]; p= 0) /* while (stack is not empty) */ { - elen[i] = -2; /* element i is dead */ - nel++; - Cp[i] = -1; /* i is a root of assembly tree */ - w[i] = 0; - } - else if(d > dense || !has_diag) /* node i is dense or has no structural diagonal element */ - { - nv[i] = 0; /* absorb i into element n */ - elen[i] = -1; /* node i is dead */ - nel++; - Cp[i] = amd_flip (n); - nv[n]++; - } - else - { - if(head[d] != -1) last[head[d]] = i; - next[i] = head[d]; /* put node i in degree list d */ - head[d] = i; + p = stack[top]; /* p = top of stack */ + i = head[p]; /* i = youngest child of p */ + if (i == -1) { + top--; /* p has no unordered children left */ + post[k++] = p; /* node p is the kth postordered node */ + } else { + head[p] = next[i]; /* remove i from children of p */ + stack[++top] = i; /* start dfs on child node i */ + } } + return k; } - - elen[n] = -2; /* n is a dead element */ - Cp[n] = -1; /* n is a root of assembly tree */ - w[n] = 0; /* n is a dead element */ - - while (nel < n) /* while (selecting pivots) do */ + + + /** \internal + * \ingroup OrderingMethods_Module + * Approximate minimum degree ordering algorithm. + * + * \param[in] C the input selfadjoint matrix stored in compressed column major format. + * \param[out] perm the permutation P reducing the fill-in of the input matrix \a C + * + * Note that the input matrix \a C must be complete, that is both the upper and lower parts have to be stored, as well + * as the diagonal entries. On exit the values of C are destroyed */ + template + void minimum_degree_ordering(SparseMatrix &C, + PermutationMatrix &perm) { - /* --- Select node of minimum approximate degree -------------------- */ - for(k = -1; mindeg < n && (k = head[mindeg]) == -1; mindeg++) {} - if(next[k] != -1) last[next[k]] = -1; - head[mindeg] = next[k]; /* remove k from degree list */ - elenk = elen[k]; /* elenk = |Ek| */ - nvk = nv[k]; /* # of nodes k represents */ - nel += nvk; /* nv[k] nodes of A eliminated */ - - /* --- Garbage collection ------------------------------------------- */ - if(elenk > 0 && cnz + mindeg >= nzmax) - { - for(j = 0; j < n; j++) - { - if((p = Cp[j]) >= 0) /* j is a live node or element */ - { - Cp[j] = Ci[p]; /* save first entry of object */ - Ci[p] = amd_flip (j); /* first entry is now amd_flip(j) */ + using std::sqrt; + + StorageIndex d, dk, dext, lemax = 0, e, elenk, eln, i, j, k, k1, k2, k3, jlast, ln, dense, nzmax, mindeg = 0, nvi, + nvj, nvk, mark, wnvi, ok, nel = 0, p, p1, p2, p3, p4, pj, pk, pk1, pk2, pn, q, t, h; + + StorageIndex n = StorageIndex(C.cols()); + dense = std::max(16, StorageIndex(10 * sqrt(double(n)))); /* find dense threshold */ + dense = (std::min)(n - 2, dense); + + StorageIndex cnz = StorageIndex(C.nonZeros()); + perm.resize(n + 1); + t = cnz + cnz / 5 + 2 * n; /* add elbow room to C */ + C.resizeNonZeros(t); + + // get workspace + ei_declare_aligned_stack_constructed_variable(StorageIndex, W, 8 * (n + 1), 0); + StorageIndex *len = W; + StorageIndex *nv = W + (n + 1); + StorageIndex *next = W + 2 * (n + 1); + StorageIndex *head = W + 3 * (n + 1); + StorageIndex *elen = W + 4 * (n + 1); + StorageIndex *degree = W + 5 * (n + 1); + StorageIndex *w = W + 6 * (n + 1); + StorageIndex *hhead = W + 7 * (n + 1); + StorageIndex *last = perm.indices().data(); /* use P as workspace for last */ + + /* --- Initialize quotient graph ---------------------------------------- */ + StorageIndex *Cp = C.outerIndexPtr(); + StorageIndex *Ci = C.innerIndexPtr(); + for (k = 0; k < n; k++) len[k] = Cp[k + 1] - Cp[k]; + len[n] = 0; + nzmax = t; + + for (i = 0; i <= n; i++) { + head[i] = -1;// degree list i is empty + last[i] = -1; + next[i] = -1; + hhead[i] = -1;// hash list i is empty + nv[i] = 1;// node i is just one node + w[i] = 1;// node i is alive + elen[i] = 0;// Ek of node i is empty + degree[i] = len[i];// degree of node i + } + mark = internal::cs_wclear(0, 0, w, n); /* clear w */ + + /* --- Initialize degree lists ------------------------------------------ */ + for (i = 0; i < n; i++) { + bool has_diag = false; + for (p = Cp[i]; p < Cp[i + 1]; ++p) + if (Ci[p] == i) { + has_diag = true; + break; } - } - for(q = 0, p = 0; p < cnz; ) /* scan all of memory */ + + d = degree[i]; + if (d == 1 && has_diag) /* node i is empty */ { - if((j = amd_flip (Ci[p++])) >= 0) /* found object j */ - { - Ci[q] = Cp[j]; /* restore first entry of object */ - Cp[j] = q++; /* new pointer to object j */ - for(k3 = 0; k3 < len[j]-1; k3++) Ci[q++] = Ci[p++]; - } + elen[i] = -2; /* element i is dead */ + nel++; + Cp[i] = -1; /* i is a root of assembly tree */ + w[i] = 0; + } else if (d > dense || !has_diag) /* node i is dense or has no structural diagonal element */ + { + nv[i] = 0; /* absorb i into element n */ + elen[i] = -1; /* node i is dead */ + nel++; + Cp[i] = amd_flip(n); + nv[n]++; + } else { + if (head[d] != -1) last[head[d]] = i; + next[i] = head[d]; /* put node i in degree list d */ + head[d] = i; } - cnz = q; /* Ci[cnz...nzmax-1] now free */ } - - /* --- Construct new element ---------------------------------------- */ - dk = 0; - nv[k] = -nvk; /* flag k as in Lk */ - p = Cp[k]; - pk1 = (elenk == 0) ? p : cnz; /* do in place if elen[k] == 0 */ - pk2 = pk1; - for(k1 = 1; k1 <= elenk + 1; k1++) + + elen[n] = -2; /* n is a dead element */ + Cp[n] = -1; /* n is a root of assembly tree */ + w[n] = 0; /* n is a dead element */ + + while (nel < n) /* while (selecting pivots) do */ { - if(k1 > elenk) - { - e = k; /* search the nodes in k */ - pj = p; /* list of nodes starts at Ci[pj]*/ - ln = len[k] - elenk; /* length of list of nodes in k */ - } - else - { - e = Ci[p++]; /* search the nodes in e */ - pj = Cp[e]; - ln = len[e]; /* length of list of nodes in e */ - } - for(k2 = 1; k2 <= ln; k2++) - { - i = Ci[pj++]; - if((nvi = nv[i]) <= 0) continue; /* node i dead, or seen */ - dk += nvi; /* degree[Lk] += size of node i */ - nv[i] = -nvi; /* negate nv[i] to denote i in Lk*/ - Ci[pk2++] = i; /* place i in Lk */ - if(next[i] != -1) last[next[i]] = last[i]; - if(last[i] != -1) /* remove i from degree list */ - { - next[last[i]] = next[i]; + /* --- Select node of minimum approximate degree -------------------- */ + for (k = -1; mindeg < n && (k = head[mindeg]) == -1; mindeg++) {} + if (next[k] != -1) last[next[k]] = -1; + head[mindeg] = next[k]; /* remove k from degree list */ + elenk = elen[k]; /* elenk = |Ek| */ + nvk = nv[k]; /* # of nodes k represents */ + nel += nvk; /* nv[k] nodes of A eliminated */ + + /* --- Garbage collection ------------------------------------------- */ + if (elenk > 0 && cnz + mindeg >= nzmax) { + for (j = 0; j < n; j++) { + if ((p = Cp[j]) >= 0) /* j is a live node or element */ + { + Cp[j] = Ci[p]; /* save first entry of object */ + Ci[p] = amd_flip(j); /* first entry is now amd_flip(j) */ + } } - else + for (q = 0, p = 0; p < cnz;) /* scan all of memory */ { - head[degree[i]] = next[i]; + if ((j = amd_flip(Ci[p++])) >= 0) /* found object j */ + { + Ci[q] = Cp[j]; /* restore first entry of object */ + Cp[j] = q++; /* new pointer to object j */ + for (k3 = 0; k3 < len[j] - 1; k3++) Ci[q++] = Ci[p++]; + } } + cnz = q; /* Ci[cnz...nzmax-1] now free */ } - if(e != k) - { - Cp[e] = amd_flip (k); /* absorb e into k */ - w[e] = 0; /* e is now a dead element */ - } - } - if(elenk != 0) cnz = pk2; /* Ci[cnz...nzmax] is free */ - degree[k] = dk; /* external degree of k - |Lk\i| */ - Cp[k] = pk1; /* element k is in Ci[pk1..pk2-1] */ - len[k] = pk2 - pk1; - elen[k] = -2; /* k is now an element */ - - /* --- Find set differences ----------------------------------------- */ - mark = internal::cs_wclear(mark, lemax, w, n); /* clear w if necessary */ - for(pk = pk1; pk < pk2; pk++) /* scan 1: find |Le\Lk| */ - { - i = Ci[pk]; - if((eln = elen[i]) <= 0) continue;/* skip if elen[i] empty */ - nvi = -nv[i]; /* nv[i] was negated */ - wnvi = mark - nvi; - for(p = Cp[i]; p <= Cp[i] + eln - 1; p++) /* scan Ei */ - { - e = Ci[p]; - if(w[e] >= mark) - { - w[e] -= nvi; /* decrement |Le\Lk| */ + + /* --- Construct new element ---------------------------------------- */ + dk = 0; + nv[k] = -nvk; /* flag k as in Lk */ + p = Cp[k]; + pk1 = (elenk == 0) ? p : cnz; /* do in place if elen[k] == 0 */ + pk2 = pk1; + for (k1 = 1; k1 <= elenk + 1; k1++) { + if (k1 > elenk) { + e = k; /* search the nodes in k */ + pj = p; /* list of nodes starts at Ci[pj]*/ + ln = len[k] - elenk; /* length of list of nodes in k */ + } else { + e = Ci[p++]; /* search the nodes in e */ + pj = Cp[e]; + ln = len[e]; /* length of list of nodes in e */ } - else if(w[e] != 0) /* ensure e is a live element */ - { - w[e] = degree[e] + wnvi; /* 1st time e seen in scan 1 */ + for (k2 = 1; k2 <= ln; k2++) { + i = Ci[pj++]; + if ((nvi = nv[i]) <= 0) continue; /* node i dead, or seen */ + dk += nvi; /* degree[Lk] += size of node i */ + nv[i] = -nvi; /* negate nv[i] to denote i in Lk*/ + Ci[pk2++] = i; /* place i in Lk */ + if (next[i] != -1) last[next[i]] = last[i]; + if (last[i] != -1) /* remove i from degree list */ + { + next[last[i]] = next[i]; + } else { + head[degree[i]] = next[i]; + } + } + if (e != k) { + Cp[e] = amd_flip(k); /* absorb e into k */ + w[e] = 0; /* e is now a dead element */ } } - } - - /* --- Degree update ------------------------------------------------ */ - for(pk = pk1; pk < pk2; pk++) /* scan2: degree update */ - { - i = Ci[pk]; /* consider node i in Lk */ - p1 = Cp[i]; - p2 = p1 + elen[i] - 1; - pn = p1; - for(h = 0, d = 0, p = p1; p <= p2; p++) /* scan Ei */ + if (elenk != 0) cnz = pk2; /* Ci[cnz...nzmax] is free */ + degree[k] = dk; /* external degree of k - |Lk\i| */ + Cp[k] = pk1; /* element k is in Ci[pk1..pk2-1] */ + len[k] = pk2 - pk1; + elen[k] = -2; /* k is now an element */ + + /* --- Find set differences ----------------------------------------- */ + mark = internal::cs_wclear(mark, lemax, w, n); /* clear w if necessary */ + for (pk = pk1; pk < pk2; pk++) /* scan 1: find |Le\Lk| */ { - e = Ci[p]; - if(w[e] != 0) /* e is an unabsorbed element */ + i = Ci[pk]; + if ((eln = elen[i]) <= 0) continue; /* skip if elen[i] empty */ + nvi = -nv[i]; /* nv[i] was negated */ + wnvi = mark - nvi; + for (p = Cp[i]; p <= Cp[i] + eln - 1; p++) /* scan Ei */ { - dext = w[e] - mark; /* dext = |Le\Lk| */ - if(dext > 0) + e = Ci[p]; + if (w[e] >= mark) { + w[e] -= nvi; /* decrement |Le\Lk| */ + } else if (w[e] != 0) /* ensure e is a live element */ { - d += dext; /* sum up the set differences */ - Ci[pn++] = e; /* keep e in Ei */ - h += e; /* compute the hash of node i */ - } - else - { - Cp[e] = amd_flip (k); /* aggressive absorb. e->k */ - w[e] = 0; /* e is a dead element */ + w[e] = degree[e] + wnvi; /* 1st time e seen in scan 1 */ } } } - elen[i] = pn - p1 + 1; /* elen[i] = |Ei| */ - p3 = pn; - p4 = p1 + len[i]; - for(p = p2 + 1; p < p4; p++) /* prune edges in Ai */ - { - j = Ci[p]; - if((nvj = nv[j]) <= 0) continue; /* node j dead or in Lk */ - d += nvj; /* degree(i) += |j| */ - Ci[pn++] = j; /* place j in node list of i */ - h += j; /* compute hash for node i */ - } - if(d == 0) /* check for mass elimination */ - { - Cp[i] = amd_flip (k); /* absorb i into k */ - nvi = -nv[i]; - dk -= nvi; /* |Lk| -= |i| */ - nvk += nvi; /* |k| += nv[i] */ - nel += nvi; - nv[i] = 0; - elen[i] = -1; /* node i is dead */ - } - else - { - degree[i] = std::min (degree[i], d); /* update degree(i) */ - Ci[pn] = Ci[p3]; /* move first node to end */ - Ci[p3] = Ci[p1]; /* move 1st el. to end of Ei */ - Ci[p1] = k; /* add k as 1st element in of Ei */ - len[i] = pn - p1 + 1; /* new len of adj. list of node i */ - h %= n; /* finalize hash of i */ - next[i] = hhead[h]; /* place i in hash bucket */ - hhead[h] = i; - last[i] = h; /* save hash of i in last[i] */ - } - } /* scan2 is done */ - degree[k] = dk; /* finalize |Lk| */ - lemax = std::max(lemax, dk); - mark = internal::cs_wclear(mark+lemax, lemax, w, n); /* clear w */ - - /* --- Supernode detection ------------------------------------------ */ - for(pk = pk1; pk < pk2; pk++) - { - i = Ci[pk]; - if(nv[i] >= 0) continue; /* skip if i is dead */ - h = last[i]; /* scan hash bucket of node i */ - i = hhead[h]; - hhead[h] = -1; /* hash bucket will be empty */ - for(; i != -1 && next[i] != -1; i = next[i], mark++) + + /* --- Degree update ------------------------------------------------ */ + for (pk = pk1; pk < pk2; pk++) /* scan2: degree update */ { - ln = len[i]; - eln = elen[i]; - for(p = Cp[i]+1; p <= Cp[i] + ln-1; p++) w[Ci[p]] = mark; - jlast = i; - for(j = next[i]; j != -1; ) /* compare i with all j */ + i = Ci[pk]; /* consider node i in Lk */ + p1 = Cp[i]; + p2 = p1 + elen[i] - 1; + pn = p1; + for (h = 0, d = 0, p = p1; p <= p2; p++) /* scan Ei */ { - ok = (len[j] == ln) && (elen[j] == eln); - for(p = Cp[j] + 1; ok && p <= Cp[j] + ln - 1; p++) - { - if(w[Ci[p]] != mark) ok = 0; /* compare i and j*/ - } - if(ok) /* i and j are identical */ + e = Ci[p]; + if (w[e] != 0) /* e is an unabsorbed element */ { - Cp[j] = amd_flip (i); /* absorb j into i */ - nv[i] += nv[j]; - nv[j] = 0; - elen[j] = -1; /* node j is dead */ - j = next[j]; /* delete j from hash bucket */ - next[jlast] = j; + dext = w[e] - mark; /* dext = |Le\Lk| */ + if (dext > 0) { + d += dext; /* sum up the set differences */ + Ci[pn++] = e; /* keep e in Ei */ + h += e; /* compute the hash of node i */ + } else { + Cp[e] = amd_flip(k); /* aggressive absorb. e->k */ + w[e] = 0; /* e is a dead element */ + } } - else + } + elen[i] = pn - p1 + 1; /* elen[i] = |Ei| */ + p3 = pn; + p4 = p1 + len[i]; + for (p = p2 + 1; p < p4; p++) /* prune edges in Ai */ + { + j = Ci[p]; + if ((nvj = nv[j]) <= 0) continue; /* node j dead or in Lk */ + d += nvj; /* degree(i) += |j| */ + Ci[pn++] = j; /* place j in node list of i */ + h += j; /* compute hash for node i */ + } + if (d == 0) /* check for mass elimination */ + { + Cp[i] = amd_flip(k); /* absorb i into k */ + nvi = -nv[i]; + dk -= nvi; /* |Lk| -= |i| */ + nvk += nvi; /* |k| += nv[i] */ + nel += nvi; + nv[i] = 0; + elen[i] = -1; /* node i is dead */ + } else { + degree[i] = std::min(degree[i], d); /* update degree(i) */ + Ci[pn] = Ci[p3]; /* move first node to end */ + Ci[p3] = Ci[p1]; /* move 1st el. to end of Ei */ + Ci[p1] = k; /* add k as 1st element in of Ei */ + len[i] = pn - p1 + 1; /* new len of adj. list of node i */ + h %= n; /* finalize hash of i */ + next[i] = hhead[h]; /* place i in hash bucket */ + hhead[h] = i; + last[i] = h; /* save hash of i in last[i] */ + } + } /* scan2 is done */ + degree[k] = dk; /* finalize |Lk| */ + lemax = std::max(lemax, dk); + mark = internal::cs_wclear(mark + lemax, lemax, w, n); /* clear w */ + + /* --- Supernode detection ------------------------------------------ */ + for (pk = pk1; pk < pk2; pk++) { + i = Ci[pk]; + if (nv[i] >= 0) continue; /* skip if i is dead */ + h = last[i]; /* scan hash bucket of node i */ + i = hhead[h]; + hhead[h] = -1; /* hash bucket will be empty */ + for (; i != -1 && next[i] != -1; i = next[i], mark++) { + ln = len[i]; + eln = elen[i]; + for (p = Cp[i] + 1; p <= Cp[i] + ln - 1; p++) w[Ci[p]] = mark; + jlast = i; + for (j = next[i]; j != -1;) /* compare i with all j */ { - jlast = j; /* j and i are different */ - j = next[j]; + ok = (len[j] == ln) && (elen[j] == eln); + for (p = Cp[j] + 1; ok && p <= Cp[j] + ln - 1; p++) { + if (w[Ci[p]] != mark) ok = 0; /* compare i and j*/ + } + if (ok) /* i and j are identical */ + { + Cp[j] = amd_flip(i); /* absorb j into i */ + nv[i] += nv[j]; + nv[j] = 0; + elen[j] = -1; /* node j is dead */ + j = next[j]; /* delete j from hash bucket */ + next[jlast] = j; + } else { + jlast = j; /* j and i are different */ + j = next[j]; + } } } } + + /* --- Finalize new element------------------------------------------ */ + for (p = pk1, pk = pk1; pk < pk2; pk++) /* finalize Lk */ + { + i = Ci[pk]; + if ((nvi = -nv[i]) <= 0) continue; /* skip if i is dead */ + nv[i] = nvi; /* restore nv[i] */ + d = degree[i] + dk - nvi; /* compute external degree(i) */ + d = std::min(d, n - nel - nvi); + if (head[d] != -1) last[head[d]] = i; + next[i] = head[d]; /* put i back in degree list */ + last[i] = -1; + head[d] = i; + mindeg = std::min(mindeg, d); /* find new minimum degree */ + degree[i] = d; + Ci[p++] = i; /* place i in Lk */ + } + nv[k] = nvk; /* # nodes absorbed into k */ + if ((len[k] = p - pk1) == 0) /* length of adj list of element k*/ + { + Cp[k] = -1; /* k is a root of the tree */ + w[k] = 0; /* k is now a dead element */ + } + if (elenk != 0) cnz = p; /* free unused space in Lk */ } - - /* --- Finalize new element------------------------------------------ */ - for(p = pk1, pk = pk1; pk < pk2; pk++) /* finalize Lk */ + + /* --- Postordering ----------------------------------------------------- */ + for (i = 0; i < n; i++) Cp[i] = amd_flip(Cp[i]); /* fix assembly tree */ + for (j = 0; j <= n; j++) head[j] = -1; + for (j = n; j >= 0; j--) /* place unordered nodes in lists */ { - i = Ci[pk]; - if((nvi = -nv[i]) <= 0) continue;/* skip if i is dead */ - nv[i] = nvi; /* restore nv[i] */ - d = degree[i] + dk - nvi; /* compute external degree(i) */ - d = std::min (d, n - nel - nvi); - if(head[d] != -1) last[head[d]] = i; - next[i] = head[d]; /* put i back in degree list */ - last[i] = -1; - head[d] = i; - mindeg = std::min (mindeg, d); /* find new minimum degree */ - degree[i] = d; - Ci[p++] = i; /* place i in Lk */ + if (nv[j] > 0) continue; /* skip if j is an element */ + next[j] = head[Cp[j]]; /* place j in list of its parent */ + head[Cp[j]] = j; } - nv[k] = nvk; /* # nodes absorbed into k */ - if((len[k] = p-pk1) == 0) /* length of adj list of element k*/ + for (e = n; e >= 0; e--) /* place elements in lists */ { - Cp[k] = -1; /* k is a root of the tree */ - w[k] = 0; /* k is now a dead element */ + if (nv[e] <= 0) continue; /* skip unless e is an element */ + if (Cp[e] != -1) { + next[e] = head[Cp[e]]; /* place e in list of its parent */ + head[Cp[e]] = e; + } } - if(elenk != 0) cnz = p; /* free unused space in Lk */ - } - - /* --- Postordering ----------------------------------------------------- */ - for(i = 0; i < n; i++) Cp[i] = amd_flip (Cp[i]);/* fix assembly tree */ - for(j = 0; j <= n; j++) head[j] = -1; - for(j = n; j >= 0; j--) /* place unordered nodes in lists */ - { - if(nv[j] > 0) continue; /* skip if j is an element */ - next[j] = head[Cp[j]]; /* place j in list of its parent */ - head[Cp[j]] = j; - } - for(e = n; e >= 0; e--) /* place elements in lists */ - { - if(nv[e] <= 0) continue; /* skip unless e is an element */ - if(Cp[e] != -1) + for (k = 0, i = 0; i <= n; i++) /* postorder the assembly tree */ { - next[e] = head[Cp[e]]; /* place e in list of its parent */ - head[Cp[e]] = e; + if (Cp[i] == -1) k = internal::cs_tdfs(i, k, head, next, perm.indices().data(), w); } + + perm.indices().conservativeResize(n); } - for(k = 0, i = 0; i <= n; i++) /* postorder the assembly tree */ - { - if(Cp[i] == -1) k = internal::cs_tdfs(i, k, head, next, perm.indices().data(), w); - } - - perm.indices().conservativeResize(n); -} -} // namespace internal +}// namespace internal -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_SPARSE_AMD_H +#endif// EIGEN_SPARSE_AMD_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/OrderingMethods/Eigen_Colamd.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/OrderingMethods/Eigen_Colamd.h index da85b4d6..89227a75 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/OrderingMethods/Eigen_Colamd.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/OrderingMethods/Eigen_Colamd.h @@ -13,37 +13,37 @@ // Davis (davis@cise.ufl.edu), University of Florida. The algorithm was // developed in collaboration with John Gilbert, Xerox PARC, and Esmond // Ng, Oak Ridge National Laboratory. -// +// // Date: -// +// // September 8, 2003. Version 2.3. -// +// // Acknowledgements: -// +// // This work was supported by the National Science Foundation, under // grants DMS-9504974 and DMS-9803599. -// +// // Notice: -// +// // Copyright (c) 1998-2003 by the University of Florida. // All Rights Reserved. -// +// // THIS MATERIAL IS PROVIDED AS IS, WITH ABSOLUTELY NO WARRANTY // EXPRESSED OR IMPLIED. ANY USE IS AT YOUR OWN RISK. -// +// // Permission is hereby granted to use, copy, modify, and/or distribute // this program, provided that the Copyright, this License, and the // Availability of the original version is retained on all copies and made // accessible to the end-user of any code or package that includes COLAMD -// or any modified version of COLAMD. -// +// or any modified version of COLAMD. +// // Availability: -// +// // The colamd/symamd library is available at -// +// // http://www.suitesparse.com - + #ifndef EIGEN_COLAMD_H #define EIGEN_COLAMD_H @@ -60,7 +60,7 @@ namespace internal { #define COLAMD_KNOBS 20 /* number of output statistics. Only stats [0..6] are currently used. */ -#define COLAMD_STATS 20 +#define COLAMD_STATS 20 /* knobs [0] and stats [0]: dense row knob and output statistic. */ #define COLAMD_DENSE_ROW 0 @@ -74,31 +74,31 @@ namespace internal { /* stats [3]: colamd status: zero OK, > 0 warning or notice, < 0 error */ #define COLAMD_STATUS 3 -/* stats [4..6]: error info, or info on jumbled columns */ +/* stats [4..6]: error info, or info on jumbled columns */ #define COLAMD_INFO1 4 #define COLAMD_INFO2 5 #define COLAMD_INFO3 6 /* error codes returned in stats [3]: */ -#define COLAMD_OK (0) -#define COLAMD_OK_BUT_JUMBLED (1) -#define COLAMD_ERROR_A_not_present (-1) -#define COLAMD_ERROR_p_not_present (-2) -#define COLAMD_ERROR_nrow_negative (-3) -#define COLAMD_ERROR_ncol_negative (-4) -#define COLAMD_ERROR_nnz_negative (-5) -#define COLAMD_ERROR_p0_nonzero (-6) -#define COLAMD_ERROR_A_too_small (-7) -#define COLAMD_ERROR_col_length_negative (-8) -#define COLAMD_ERROR_row_index_out_of_bounds (-9) -#define COLAMD_ERROR_out_of_memory (-10) -#define COLAMD_ERROR_internal_error (-999) +#define COLAMD_OK (0) +#define COLAMD_OK_BUT_JUMBLED (1) +#define COLAMD_ERROR_A_not_present (-1) +#define COLAMD_ERROR_p_not_present (-2) +#define COLAMD_ERROR_nrow_negative (-3) +#define COLAMD_ERROR_ncol_negative (-4) +#define COLAMD_ERROR_nnz_negative (-5) +#define COLAMD_ERROR_p0_nonzero (-6) +#define COLAMD_ERROR_A_too_small (-7) +#define COLAMD_ERROR_col_length_negative (-8) +#define COLAMD_ERROR_row_index_out_of_bounds (-9) +#define COLAMD_ERROR_out_of_memory (-10) +#define COLAMD_ERROR_internal_error (-999) /* ========================================================================== */ /* === Definitions ========================================================== */ /* ========================================================================== */ -#define ONES_COMPLEMENT(r) (-(r)-1) +#define ONES_COMPLEMENT(r) (-(r) - 1) /* -------------------------------------------------------------------------- */ @@ -106,84 +106,83 @@ namespace internal { /* Row and column status */ #define ALIVE (0) -#define DEAD (-1) +#define DEAD (-1) /* Column status */ -#define DEAD_PRINCIPAL (-1) -#define DEAD_NON_PRINCIPAL (-2) +#define DEAD_PRINCIPAL (-1) +#define DEAD_NON_PRINCIPAL (-2) /* Macros for row and column status update and checking. */ -#define ROW_IS_DEAD(r) ROW_IS_MARKED_DEAD (Row[r].shared2.mark) -#define ROW_IS_MARKED_DEAD(row_mark) (row_mark < ALIVE) -#define ROW_IS_ALIVE(r) (Row [r].shared2.mark >= ALIVE) -#define COL_IS_DEAD(c) (Col [c].start < ALIVE) -#define COL_IS_ALIVE(c) (Col [c].start >= ALIVE) -#define COL_IS_DEAD_PRINCIPAL(c) (Col [c].start == DEAD_PRINCIPAL) -#define KILL_ROW(r) { Row [r].shared2.mark = DEAD ; } -#define KILL_PRINCIPAL_COL(c) { Col [c].start = DEAD_PRINCIPAL ; } -#define KILL_NON_PRINCIPAL_COL(c) { Col [c].start = DEAD_NON_PRINCIPAL ; } +#define ROW_IS_DEAD(r) ROW_IS_MARKED_DEAD(Row[r].shared2.mark) +#define ROW_IS_MARKED_DEAD(row_mark) (row_mark < ALIVE) +#define ROW_IS_ALIVE(r) (Row[r].shared2.mark >= ALIVE) +#define COL_IS_DEAD(c) (Col[c].start < ALIVE) +#define COL_IS_ALIVE(c) (Col[c].start >= ALIVE) +#define COL_IS_DEAD_PRINCIPAL(c) (Col[c].start == DEAD_PRINCIPAL) +#define KILL_ROW(r) \ + { \ + Row[r].shared2.mark = DEAD; \ + } +#define KILL_PRINCIPAL_COL(c) \ + { \ + Col[c].start = DEAD_PRINCIPAL; \ + } +#define KILL_NON_PRINCIPAL_COL(c) \ + { \ + Col[c].start = DEAD_NON_PRINCIPAL; \ + } /* ========================================================================== */ /* === Colamd reporting mechanism =========================================== */ /* ========================================================================== */ // == Row and Column structures == -template -struct colamd_col +template struct colamd_col { - IndexType start ; /* index for A of first row in this column, or DEAD */ + IndexType start; /* index for A of first row in this column, or DEAD */ /* if column is dead */ - IndexType length ; /* number of rows in this column */ - union - { - IndexType thickness ; /* number of original columns represented by this */ + IndexType length; /* number of rows in this column */ + union { + IndexType thickness; /* number of original columns represented by this */ /* col, if the column is alive */ - IndexType parent ; /* parent in parent tree super-column structure, if */ + IndexType parent; /* parent in parent tree super-column structure, if */ /* the column is dead */ - } shared1 ; - union - { - IndexType score ; /* the score used to maintain heap, if col is alive */ - IndexType order ; /* pivot ordering of this column, if col is dead */ - } shared2 ; - union - { - IndexType headhash ; /* head of a hash bucket, if col is at the head of */ + } shared1; + union { + IndexType score; /* the score used to maintain heap, if col is alive */ + IndexType order; /* pivot ordering of this column, if col is dead */ + } shared2; + union { + IndexType headhash; /* head of a hash bucket, if col is at the head of */ /* a degree list */ - IndexType hash ; /* hash value, if col is not in a degree list */ - IndexType prev ; /* previous column in degree list, if col is in a */ + IndexType hash; /* hash value, if col is not in a degree list */ + IndexType prev; /* previous column in degree list, if col is in a */ /* degree list (but not at the head of a degree list) */ - } shared3 ; - union - { - IndexType degree_next ; /* next column, if col is in a degree list */ - IndexType hash_next ; /* next column, if col is in a hash list */ - } shared4 ; - + } shared3; + union { + IndexType degree_next; /* next column, if col is in a degree list */ + IndexType hash_next; /* next column, if col is in a hash list */ + } shared4; }; - -template -struct Colamd_Row + +template struct Colamd_Row { - IndexType start ; /* index for A of first col in this row */ - IndexType length ; /* number of principal columns in this row */ - union - { - IndexType degree ; /* number of principal & non-principal columns in row */ - IndexType p ; /* used as a row pointer in init_rows_cols () */ - } shared1 ; - union - { - IndexType mark ; /* for computing set differences and marking dead rows*/ - IndexType first_column ;/* first column in row (used in garbage collection) */ - } shared2 ; - + IndexType start; /* index for A of first col in this row */ + IndexType length; /* number of principal columns in this row */ + union { + IndexType degree; /* number of principal & non-principal columns in row */ + IndexType p; /* used as a row pointer in init_rows_cols () */ + } shared1; + union { + IndexType mark; /* for computing set differences and marking dead rows*/ + IndexType first_column; /* first column in row (used in garbage collection) */ + } shared2; }; - + /* ========================================================================== */ /* === Colamd recommended memory size ======================================= */ /* ========================================================================== */ - + /* The recommended length Alen of the array A passed to colamd is given by the COLAMD_RECOMMENDED (nnz, n_row, n_col) macro. It returns -1 if any @@ -192,41 +191,74 @@ struct Colamd_Row required for the Col and Row arrays, respectively, which are internal to colamd. An additional n_col space is the minimal amount of "elbow room", and nnz/5 more space is recommended for run time efficiency. - + This macro is not needed when using symamd. - + Explicit typecast to IndexType added Sept. 23, 2002, COLAMD version 2.2, to avoid gcc -pedantic warning messages. */ -template -inline IndexType colamd_c(IndexType n_col) -{ return IndexType( ((n_col) + 1) * sizeof (colamd_col) / sizeof (IndexType) ) ; } +template inline IndexType colamd_c(IndexType n_col) +{ + return IndexType(((n_col) + 1) * sizeof(colamd_col) / sizeof(IndexType)); +} -template -inline IndexType colamd_r(IndexType n_row) -{ return IndexType(((n_row) + 1) * sizeof (Colamd_Row) / sizeof (IndexType)); } +template inline IndexType colamd_r(IndexType n_row) +{ + return IndexType(((n_row) + 1) * sizeof(Colamd_Row) / sizeof(IndexType)); +} // Prototypes of non-user callable routines -template -static IndexType init_rows_cols (IndexType n_row, IndexType n_col, Colamd_Row Row [], colamd_col col [], IndexType A [], IndexType p [], IndexType stats[COLAMD_STATS] ); - -template -static void init_scoring (IndexType n_row, IndexType n_col, Colamd_Row Row [], colamd_col Col [], IndexType A [], IndexType head [], double knobs[COLAMD_KNOBS], IndexType *p_n_row2, IndexType *p_n_col2, IndexType *p_max_deg); - -template -static IndexType find_ordering (IndexType n_row, IndexType n_col, IndexType Alen, Colamd_Row Row [], colamd_col Col [], IndexType A [], IndexType head [], IndexType n_col2, IndexType max_deg, IndexType pfree); - -template -static void order_children (IndexType n_col, colamd_col Col [], IndexType p []); - -template -static void detect_super_cols (colamd_col Col [], IndexType A [], IndexType head [], IndexType row_start, IndexType row_length ) ; - -template -static IndexType garbage_collection (IndexType n_row, IndexType n_col, Colamd_Row Row [], colamd_col Col [], IndexType A [], IndexType *pfree) ; - -template -static inline IndexType clear_mark (IndexType n_row, Colamd_Row Row [] ) ; +template +static IndexType init_rows_cols(IndexType n_row, + IndexType n_col, + Colamd_Row Row[], + colamd_col col[], + IndexType A[], + IndexType p[], + IndexType stats[COLAMD_STATS]); + +template +static void init_scoring(IndexType n_row, + IndexType n_col, + Colamd_Row Row[], + colamd_col Col[], + IndexType A[], + IndexType head[], + double knobs[COLAMD_KNOBS], + IndexType *p_n_row2, + IndexType *p_n_col2, + IndexType *p_max_deg); + +template +static IndexType find_ordering(IndexType n_row, + IndexType n_col, + IndexType Alen, + Colamd_Row Row[], + colamd_col Col[], + IndexType A[], + IndexType head[], + IndexType n_col2, + IndexType max_deg, + IndexType pfree); + +template static void order_children(IndexType n_col, colamd_col Col[], IndexType p[]); + +template +static void detect_super_cols(colamd_col Col[], + IndexType A[], + IndexType head[], + IndexType row_start, + IndexType row_length); + +template +static IndexType garbage_collection(IndexType n_row, + IndexType n_col, + Colamd_Row Row[], + colamd_col Col[], + IndexType A[], + IndexType *pfree); + +template static inline IndexType clear_mark(IndexType n_row, Colamd_Row Row[]); /* === No debugging ========================================================= */ @@ -236,39 +268,38 @@ static inline IndexType clear_mark (IndexType n_row, Colamd_Row Row #define COLAMD_DEBUG3(params) ; #define COLAMD_DEBUG4(params) ; -#define COLAMD_ASSERT(expression) ((void) 0) +#define COLAMD_ASSERT(expression) ((void)0) /** - * \brief Returns the recommended value of Alen - * - * Returns recommended value of Alen for use by colamd. - * Returns -1 if any input argument is negative. - * The use of this routine or macro is optional. - * Note that the macro uses its arguments more than once, - * so be careful for side effects, if you pass expressions as arguments to COLAMD_RECOMMENDED. - * + * \brief Returns the recommended value of Alen + * + * Returns recommended value of Alen for use by colamd. + * Returns -1 if any input argument is negative. + * The use of this routine or macro is optional. + * Note that the macro uses its arguments more than once, + * so be careful for side effects, if you pass expressions as arguments to COLAMD_RECOMMENDED. + * * \param nnz nonzeros in A * \param n_row number of rows in A * \param n_col number of columns in A * \return recommended value of Alen for use by colamd */ -template -inline IndexType colamd_recommended ( IndexType nnz, IndexType n_row, IndexType n_col) +template inline IndexType colamd_recommended(IndexType nnz, IndexType n_row, IndexType n_col) { if ((nnz) < 0 || (n_row) < 0 || (n_col) < 0) return (-1); else - return (2 * (nnz) + colamd_c (n_col) + colamd_r (n_row) + (n_col) + ((nnz) / 5)); + return (2 * (nnz) + colamd_c(n_col) + colamd_r(n_row) + (n_col) + ((nnz) / 5)); } /** * \brief set default parameters The use of this routine is optional. - * + * * Colamd: rows with more than (knobs [COLAMD_DENSE_ROW] * n_col) * entries are removed prior to ordering. Columns with more than * (knobs [COLAMD_DENSE_COL] * n_row) entries are removed prior to - * ordering, and placed last in the output column ordering. + * ordering, and placed last in the output column ordering. * * COLAMD_DENSE_ROW and COLAMD_DENSE_COL are defined as 0 and 1, * respectively, in colamd.h. Default values of these two knobs @@ -279,37 +310,31 @@ inline IndexType colamd_recommended ( IndexType nnz, IndexType n_row, IndexType * not need to change, assuming that you either use * colamd_set_defaults, or pass a (double *) NULL pointer as the * knobs array to colamd or symamd. - * + * * \param knobs parameter settings for colamd */ static inline void colamd_set_defaults(double knobs[COLAMD_KNOBS]) { /* === Local variables ================================================== */ - - int i ; - if (!knobs) - { - return ; /* no knobs to initialize */ - } - for (i = 0 ; i < COLAMD_KNOBS ; i++) - { - knobs [i] = 0 ; - } - knobs [COLAMD_DENSE_ROW] = 0.5 ; /* ignore rows over 50% dense */ - knobs [COLAMD_DENSE_COL] = 0.5 ; /* ignore columns over 50% dense */ + int i; + + if (!knobs) { return; /* no knobs to initialize */ } + for (i = 0; i < COLAMD_KNOBS; i++) { knobs[i] = 0; } + knobs[COLAMD_DENSE_ROW] = 0.5; /* ignore rows over 50% dense */ + knobs[COLAMD_DENSE_COL] = 0.5; /* ignore columns over 50% dense */ } -/** +/** * \brief Computes a column ordering using the column approximate minimum degree ordering - * + * * Computes a column ordering (Q) of A such that P(AQ)=LU or * (AQ)'AQ=LL' have less fill-in and require fewer floating point * operations than factorizing the unpermuted matrix A or A'A, * respectively. - * - * + * + * * \param n_row number of rows in A * \param n_col number of columns in A * \param Alen, size of the array A @@ -318,145 +343,141 @@ static inline void colamd_set_defaults(double knobs[COLAMD_KNOBS]) * \param knobs parameter settings for colamd * \param stats colamd output statistics and error codes */ -template -static bool colamd(IndexType n_row, IndexType n_col, IndexType Alen, IndexType *A, IndexType *p, double knobs[COLAMD_KNOBS], IndexType stats[COLAMD_STATS]) +template +static bool colamd(IndexType n_row, + IndexType n_col, + IndexType Alen, + IndexType *A, + IndexType *p, + double knobs[COLAMD_KNOBS], + IndexType stats[COLAMD_STATS]) { /* === Local variables ================================================== */ - - IndexType i ; /* loop index */ - IndexType nnz ; /* nonzeros in A */ - IndexType Row_size ; /* size of Row [], in integers */ - IndexType Col_size ; /* size of Col [], in integers */ - IndexType need ; /* minimum required length of A */ - Colamd_Row *Row ; /* pointer into A of Row [0..n_row] array */ - colamd_col *Col ; /* pointer into A of Col [0..n_col] array */ - IndexType n_col2 ; /* number of non-dense, non-empty columns */ - IndexType n_row2 ; /* number of non-dense, non-empty rows */ - IndexType ngarbage ; /* number of garbage collections performed */ - IndexType max_deg ; /* maximum row degree */ - double default_knobs [COLAMD_KNOBS] ; /* default knobs array */ - - + + IndexType i; /* loop index */ + IndexType nnz; /* nonzeros in A */ + IndexType Row_size; /* size of Row [], in integers */ + IndexType Col_size; /* size of Col [], in integers */ + IndexType need; /* minimum required length of A */ + Colamd_Row *Row; /* pointer into A of Row [0..n_row] array */ + colamd_col *Col; /* pointer into A of Col [0..n_col] array */ + IndexType n_col2; /* number of non-dense, non-empty columns */ + IndexType n_row2; /* number of non-dense, non-empty rows */ + IndexType ngarbage; /* number of garbage collections performed */ + IndexType max_deg; /* maximum row degree */ + double default_knobs[COLAMD_KNOBS]; /* default knobs array */ + + /* === Check the input arguments ======================================== */ - - if (!stats) - { - COLAMD_DEBUG0 (("colamd: stats not present\n")) ; - return (false) ; - } - for (i = 0 ; i < COLAMD_STATS ; i++) - { - stats [i] = 0 ; + + if (!stats) { + COLAMD_DEBUG0(("colamd: stats not present\n")); + return (false); } - stats [COLAMD_STATUS] = COLAMD_OK ; - stats [COLAMD_INFO1] = -1 ; - stats [COLAMD_INFO2] = -1 ; - - if (!A) /* A is not present */ + for (i = 0; i < COLAMD_STATS; i++) { stats[i] = 0; } + stats[COLAMD_STATUS] = COLAMD_OK; + stats[COLAMD_INFO1] = -1; + stats[COLAMD_INFO2] = -1; + + if (!A) /* A is not present */ { - stats [COLAMD_STATUS] = COLAMD_ERROR_A_not_present ; - COLAMD_DEBUG0 (("colamd: A not present\n")) ; - return (false) ; + stats[COLAMD_STATUS] = COLAMD_ERROR_A_not_present; + COLAMD_DEBUG0(("colamd: A not present\n")); + return (false); } - - if (!p) /* p is not present */ + + if (!p) /* p is not present */ { - stats [COLAMD_STATUS] = COLAMD_ERROR_p_not_present ; - COLAMD_DEBUG0 (("colamd: p not present\n")) ; - return (false) ; + stats[COLAMD_STATUS] = COLAMD_ERROR_p_not_present; + COLAMD_DEBUG0(("colamd: p not present\n")); + return (false); } - - if (n_row < 0) /* n_row must be >= 0 */ + + if (n_row < 0) /* n_row must be >= 0 */ { - stats [COLAMD_STATUS] = COLAMD_ERROR_nrow_negative ; - stats [COLAMD_INFO1] = n_row ; - COLAMD_DEBUG0 (("colamd: nrow negative %d\n", n_row)) ; - return (false) ; + stats[COLAMD_STATUS] = COLAMD_ERROR_nrow_negative; + stats[COLAMD_INFO1] = n_row; + COLAMD_DEBUG0(("colamd: nrow negative %d\n", n_row)); + return (false); } - - if (n_col < 0) /* n_col must be >= 0 */ + + if (n_col < 0) /* n_col must be >= 0 */ { - stats [COLAMD_STATUS] = COLAMD_ERROR_ncol_negative ; - stats [COLAMD_INFO1] = n_col ; - COLAMD_DEBUG0 (("colamd: ncol negative %d\n", n_col)) ; - return (false) ; + stats[COLAMD_STATUS] = COLAMD_ERROR_ncol_negative; + stats[COLAMD_INFO1] = n_col; + COLAMD_DEBUG0(("colamd: ncol negative %d\n", n_col)); + return (false); } - - nnz = p [n_col] ; - if (nnz < 0) /* nnz must be >= 0 */ + + nnz = p[n_col]; + if (nnz < 0) /* nnz must be >= 0 */ { - stats [COLAMD_STATUS] = COLAMD_ERROR_nnz_negative ; - stats [COLAMD_INFO1] = nnz ; - COLAMD_DEBUG0 (("colamd: number of entries negative %d\n", nnz)) ; - return (false) ; + stats[COLAMD_STATUS] = COLAMD_ERROR_nnz_negative; + stats[COLAMD_INFO1] = nnz; + COLAMD_DEBUG0(("colamd: number of entries negative %d\n", nnz)); + return (false); } - - if (p [0] != 0) - { - stats [COLAMD_STATUS] = COLAMD_ERROR_p0_nonzero ; - stats [COLAMD_INFO1] = p [0] ; - COLAMD_DEBUG0 (("colamd: p[0] not zero %d\n", p [0])) ; - return (false) ; + + if (p[0] != 0) { + stats[COLAMD_STATUS] = COLAMD_ERROR_p0_nonzero; + stats[COLAMD_INFO1] = p[0]; + COLAMD_DEBUG0(("colamd: p[0] not zero %d\n", p[0])); + return (false); } - + /* === If no knobs, set default knobs =================================== */ - - if (!knobs) - { - colamd_set_defaults (default_knobs) ; - knobs = default_knobs ; + + if (!knobs) { + colamd_set_defaults(default_knobs); + knobs = default_knobs; } - + /* === Allocate the Row and Col arrays from array A ===================== */ - - Col_size = colamd_c (n_col) ; - Row_size = colamd_r (n_row) ; - need = 2*nnz + n_col + Col_size + Row_size ; - - if (need > Alen) - { + + Col_size = colamd_c(n_col); + Row_size = colamd_r(n_row); + need = 2 * nnz + n_col + Col_size + Row_size; + + if (need > Alen) { /* not enough space in array A to perform the ordering */ - stats [COLAMD_STATUS] = COLAMD_ERROR_A_too_small ; - stats [COLAMD_INFO1] = need ; - stats [COLAMD_INFO2] = Alen ; - COLAMD_DEBUG0 (("colamd: Need Alen >= %d, given only Alen = %d\n", need,Alen)); - return (false) ; + stats[COLAMD_STATUS] = COLAMD_ERROR_A_too_small; + stats[COLAMD_INFO1] = need; + stats[COLAMD_INFO2] = Alen; + COLAMD_DEBUG0(("colamd: Need Alen >= %d, given only Alen = %d\n", need, Alen)); + return (false); } - - Alen -= Col_size + Row_size ; - Col = (colamd_col *) &A [Alen] ; - Row = (Colamd_Row *) &A [Alen + Col_size] ; + + Alen -= Col_size + Row_size; + Col = (colamd_col *)&A[Alen]; + Row = (Colamd_Row *)&A[Alen + Col_size]; /* === Construct the row and column data structures ===================== */ - - if (!Eigen::internal::init_rows_cols (n_row, n_col, Row, Col, A, p, stats)) - { + + if (!Eigen::internal::init_rows_cols(n_row, n_col, Row, Col, A, p, stats)) { /* input matrix is invalid */ - COLAMD_DEBUG0 (("colamd: Matrix invalid\n")) ; - return (false) ; + COLAMD_DEBUG0(("colamd: Matrix invalid\n")); + return (false); } - + /* === Initialize scores, kill dense rows/columns ======================= */ - Eigen::internal::init_scoring (n_row, n_col, Row, Col, A, p, knobs, - &n_row2, &n_col2, &max_deg) ; - + Eigen::internal::init_scoring(n_row, n_col, Row, Col, A, p, knobs, &n_row2, &n_col2, &max_deg); + /* === Order the supercolumns =========================================== */ - - ngarbage = Eigen::internal::find_ordering (n_row, n_col, Alen, Row, Col, A, p, - n_col2, max_deg, 2*nnz) ; - + + ngarbage = Eigen::internal::find_ordering(n_row, n_col, Alen, Row, Col, A, p, n_col2, max_deg, 2 * nnz); + /* === Order the non-principal columns ================================== */ - - Eigen::internal::order_children (n_col, Col, p) ; - + + Eigen::internal::order_children(n_col, Col, p); + /* === Return statistics in stats ======================================= */ - - stats [COLAMD_DENSE_ROW] = n_row - n_row2 ; - stats [COLAMD_DENSE_COL] = n_col - n_col2 ; - stats [COLAMD_DEFRAG_COUNT] = ngarbage ; - COLAMD_DEBUG0 (("colamd: done.\n")) ; - return (true) ; + + stats[COLAMD_DENSE_ROW] = n_row - n_row2; + stats[COLAMD_DENSE_COL] = n_col - n_col2; + stats[COLAMD_DEFRAG_COUNT] = ngarbage; + COLAMD_DEBUG0(("colamd: done.\n")); + return (true); } /* ========================================================================== */ @@ -478,113 +499,104 @@ static bool colamd(IndexType n_row, IndexType n_col, IndexType Alen, IndexType * column form of the matrix. Returns false if the matrix is invalid, true otherwise. Not user-callable. */ -template -static IndexType init_rows_cols /* returns true if OK, or false otherwise */ +template +static IndexType init_rows_cols /* returns true if OK, or false otherwise */ ( /* === Parameters ======================================================= */ - IndexType n_row, /* number of rows of A */ - IndexType n_col, /* number of columns of A */ - Colamd_Row Row [], /* of size n_row+1 */ - colamd_col Col [], /* of size n_col+1 */ - IndexType A [], /* row indices of A, of size Alen */ - IndexType p [], /* pointers to columns in A, of size n_col+1 */ - IndexType stats [COLAMD_STATS] /* colamd statistics */ - ) + IndexType n_row, /* number of rows of A */ + IndexType n_col, /* number of columns of A */ + Colamd_Row Row[], /* of size n_row+1 */ + colamd_col Col[], /* of size n_col+1 */ + IndexType A[], /* row indices of A, of size Alen */ + IndexType p[], /* pointers to columns in A, of size n_col+1 */ + IndexType stats[COLAMD_STATS] /* colamd statistics */ + ) { /* === Local variables ================================================== */ - IndexType col ; /* a column index */ - IndexType row ; /* a row index */ - IndexType *cp ; /* a column pointer */ - IndexType *cp_end ; /* a pointer to the end of a column */ - IndexType *rp ; /* a row pointer */ - IndexType *rp_end ; /* a pointer to the end of a row */ - IndexType last_row ; /* previous row */ + IndexType col; /* a column index */ + IndexType row; /* a row index */ + IndexType *cp; /* a column pointer */ + IndexType *cp_end; /* a pointer to the end of a column */ + IndexType *rp; /* a row pointer */ + IndexType *rp_end; /* a pointer to the end of a row */ + IndexType last_row; /* previous row */ /* === Initialize columns, and check column pointers ==================== */ - for (col = 0 ; col < n_col ; col++) - { - Col [col].start = p [col] ; - Col [col].length = p [col+1] - p [col] ; + for (col = 0; col < n_col; col++) { + Col[col].start = p[col]; + Col[col].length = p[col + 1] - p[col]; - if ((Col [col].length) < 0) // extra parentheses to work-around gcc bug 10200 + if ((Col[col].length) < 0)// extra parentheses to work-around gcc bug 10200 { /* column pointers must be non-decreasing */ - stats [COLAMD_STATUS] = COLAMD_ERROR_col_length_negative ; - stats [COLAMD_INFO1] = col ; - stats [COLAMD_INFO2] = Col [col].length ; - COLAMD_DEBUG0 (("colamd: col %d length %d < 0\n", col, Col [col].length)) ; - return (false) ; + stats[COLAMD_STATUS] = COLAMD_ERROR_col_length_negative; + stats[COLAMD_INFO1] = col; + stats[COLAMD_INFO2] = Col[col].length; + COLAMD_DEBUG0(("colamd: col %d length %d < 0\n", col, Col[col].length)); + return (false); } - Col [col].shared1.thickness = 1 ; - Col [col].shared2.score = 0 ; - Col [col].shared3.prev = COLAMD_EMPTY ; - Col [col].shared4.degree_next = COLAMD_EMPTY ; + Col[col].shared1.thickness = 1; + Col[col].shared2.score = 0; + Col[col].shared3.prev = COLAMD_EMPTY; + Col[col].shared4.degree_next = COLAMD_EMPTY; } /* p [0..n_col] no longer needed, used as "head" in subsequent routines */ /* === Scan columns, compute row degrees, and check row indices ========= */ - stats [COLAMD_INFO3] = 0 ; /* number of duplicate or unsorted row indices*/ + stats[COLAMD_INFO3] = 0; /* number of duplicate or unsorted row indices*/ - for (row = 0 ; row < n_row ; row++) - { - Row [row].length = 0 ; - Row [row].shared2.mark = -1 ; + for (row = 0; row < n_row; row++) { + Row[row].length = 0; + Row[row].shared2.mark = -1; } - for (col = 0 ; col < n_col ; col++) - { - last_row = -1 ; + for (col = 0; col < n_col; col++) { + last_row = -1; - cp = &A [p [col]] ; - cp_end = &A [p [col+1]] ; + cp = &A[p[col]]; + cp_end = &A[p[col + 1]]; - while (cp < cp_end) - { - row = *cp++ ; + while (cp < cp_end) { + row = *cp++; /* make sure row indices within range */ - if (row < 0 || row >= n_row) - { - stats [COLAMD_STATUS] = COLAMD_ERROR_row_index_out_of_bounds ; - stats [COLAMD_INFO1] = col ; - stats [COLAMD_INFO2] = row ; - stats [COLAMD_INFO3] = n_row ; - COLAMD_DEBUG0 (("colamd: row %d col %d out of bounds\n", row, col)) ; - return (false) ; + if (row < 0 || row >= n_row) { + stats[COLAMD_STATUS] = COLAMD_ERROR_row_index_out_of_bounds; + stats[COLAMD_INFO1] = col; + stats[COLAMD_INFO2] = row; + stats[COLAMD_INFO3] = n_row; + COLAMD_DEBUG0(("colamd: row %d col %d out of bounds\n", row, col)); + return (false); } - if (row <= last_row || Row [row].shared2.mark == col) - { - /* row index are unsorted or repeated (or both), thus col */ - /* is jumbled. This is a notice, not an error condition. */ - stats [COLAMD_STATUS] = COLAMD_OK_BUT_JUMBLED ; - stats [COLAMD_INFO1] = col ; - stats [COLAMD_INFO2] = row ; - (stats [COLAMD_INFO3]) ++ ; - COLAMD_DEBUG1 (("colamd: row %d col %d unsorted/duplicate\n",row,col)); + if (row <= last_row || Row[row].shared2.mark == col) { + /* row index are unsorted or repeated (or both), thus col */ + /* is jumbled. This is a notice, not an error condition. */ + stats[COLAMD_STATUS] = COLAMD_OK_BUT_JUMBLED; + stats[COLAMD_INFO1] = col; + stats[COLAMD_INFO2] = row; + (stats[COLAMD_INFO3])++; + COLAMD_DEBUG1(("colamd: row %d col %d unsorted/duplicate\n", row, col)); } - if (Row [row].shared2.mark != col) - { - Row [row].length++ ; - } - else - { - /* this is a repeated entry in the column, */ - /* it will be removed */ - Col [col].length-- ; + if (Row[row].shared2.mark != col) { + Row[row].length++; + } else { + /* this is a repeated entry in the column, */ + /* it will be removed */ + Col[col].length--; } /* mark the row as having been seen in this column */ - Row [row].shared2.mark = col ; + Row[row].shared2.mark = col; - last_row = row ; + last_row = row; } } @@ -592,63 +604,50 @@ static IndexType init_rows_cols /* returns true if OK, or false otherwise */ /* row form of the matrix starts directly after the column */ /* form of matrix in A */ - Row [0].start = p [n_col] ; - Row [0].shared1.p = Row [0].start ; - Row [0].shared2.mark = -1 ; - for (row = 1 ; row < n_row ; row++) - { - Row [row].start = Row [row-1].start + Row [row-1].length ; - Row [row].shared1.p = Row [row].start ; - Row [row].shared2.mark = -1 ; + Row[0].start = p[n_col]; + Row[0].shared1.p = Row[0].start; + Row[0].shared2.mark = -1; + for (row = 1; row < n_row; row++) { + Row[row].start = Row[row - 1].start + Row[row - 1].length; + Row[row].shared1.p = Row[row].start; + Row[row].shared2.mark = -1; } /* === Create row form ================================================== */ - if (stats [COLAMD_STATUS] == COLAMD_OK_BUT_JUMBLED) - { + if (stats[COLAMD_STATUS] == COLAMD_OK_BUT_JUMBLED) { /* if cols jumbled, watch for repeated row indices */ - for (col = 0 ; col < n_col ; col++) - { - cp = &A [p [col]] ; - cp_end = &A [p [col+1]] ; - while (cp < cp_end) - { - row = *cp++ ; - if (Row [row].shared2.mark != col) - { - A [(Row [row].shared1.p)++] = col ; - Row [row].shared2.mark = col ; - } + for (col = 0; col < n_col; col++) { + cp = &A[p[col]]; + cp_end = &A[p[col + 1]]; + while (cp < cp_end) { + row = *cp++; + if (Row[row].shared2.mark != col) { + A[(Row[row].shared1.p)++] = col; + Row[row].shared2.mark = col; + } } } - } - else - { + } else { /* if cols not jumbled, we don't need the mark (this is faster) */ - for (col = 0 ; col < n_col ; col++) - { - cp = &A [p [col]] ; - cp_end = &A [p [col+1]] ; - while (cp < cp_end) - { - A [(Row [*cp++].shared1.p)++] = col ; - } + for (col = 0; col < n_col; col++) { + cp = &A[p[col]]; + cp_end = &A[p[col + 1]]; + while (cp < cp_end) { A[(Row[*cp++].shared1.p)++] = col; } } } /* === Clear the row marks and set row degrees ========================== */ - for (row = 0 ; row < n_row ; row++) - { - Row [row].shared2.mark = 0 ; - Row [row].shared1.degree = Row [row].length ; + for (row = 0; row < n_row; row++) { + Row[row].shared2.mark = 0; + Row[row].shared1.degree = Row[row].length; } /* === See if we need to re-create columns ============================== */ - if (stats [COLAMD_STATUS] == COLAMD_OK_BUT_JUMBLED) - { - COLAMD_DEBUG0 (("colamd: reconstructing column form, matrix jumbled\n")) ; + if (stats[COLAMD_STATUS] == COLAMD_OK_BUT_JUMBLED) { + COLAMD_DEBUG0(("colamd: reconstructing column form, matrix jumbled\n")); /* === Compute col pointers ========================================= */ @@ -657,32 +656,27 @@ static IndexType init_rows_cols /* returns true if OK, or false otherwise */ /* Note, we may have a gap between the col form and the row */ /* form if there were duplicate entries, if so, it will be */ /* removed upon the first garbage collection */ - Col [0].start = 0 ; - p [0] = Col [0].start ; - for (col = 1 ; col < n_col ; col++) - { + Col[0].start = 0; + p[0] = Col[0].start; + for (col = 1; col < n_col; col++) { /* note that the lengths here are for pruned columns, i.e. */ /* no duplicate row indices will exist for these columns */ - Col [col].start = Col [col-1].start + Col [col-1].length ; - p [col] = Col [col].start ; + Col[col].start = Col[col - 1].start + Col[col - 1].length; + p[col] = Col[col].start; } /* === Re-create col form =========================================== */ - for (row = 0 ; row < n_row ; row++) - { - rp = &A [Row [row].start] ; - rp_end = rp + Row [row].length ; - while (rp < rp_end) - { - A [(p [*rp++])++] = row ; - } + for (row = 0; row < n_row; row++) { + rp = &A[Row[row].start]; + rp_end = rp + Row[row].length; + while (rp < rp_end) { A[(p[*rp++])++] = row; } } } /* === Done. Matrix is not (or no longer) jumbled ====================== */ - return (true) ; + return (true); } @@ -694,113 +688,98 @@ static IndexType init_rows_cols /* returns true if OK, or false otherwise */ Kills dense or empty columns and rows, calculates an initial score for each column, and places all columns in the degree lists. Not user-callable. */ -template -static void init_scoring - ( - /* === Parameters ======================================================= */ +template +static void init_scoring( + /* === Parameters ======================================================= */ - IndexType n_row, /* number of rows of A */ - IndexType n_col, /* number of columns of A */ - Colamd_Row Row [], /* of size n_row+1 */ - colamd_col Col [], /* of size n_col+1 */ - IndexType A [], /* column form and row form of A */ - IndexType head [], /* of size n_col+1 */ - double knobs [COLAMD_KNOBS],/* parameters */ - IndexType *p_n_row2, /* number of non-dense, non-empty rows */ - IndexType *p_n_col2, /* number of non-dense, non-empty columns */ - IndexType *p_max_deg /* maximum row degree */ - ) + IndexType n_row, /* number of rows of A */ + IndexType n_col, /* number of columns of A */ + Colamd_Row Row[], /* of size n_row+1 */ + colamd_col Col[], /* of size n_col+1 */ + IndexType A[], /* column form and row form of A */ + IndexType head[], /* of size n_col+1 */ + double knobs[COLAMD_KNOBS], /* parameters */ + IndexType *p_n_row2, /* number of non-dense, non-empty rows */ + IndexType *p_n_col2, /* number of non-dense, non-empty columns */ + IndexType *p_max_deg /* maximum row degree */ +) { /* === Local variables ================================================== */ - IndexType c ; /* a column index */ - IndexType r, row ; /* a row index */ - IndexType *cp ; /* a column pointer */ - IndexType deg ; /* degree of a row or column */ - IndexType *cp_end ; /* a pointer to the end of a column */ - IndexType *new_cp ; /* new column pointer */ - IndexType col_length ; /* length of pruned column */ - IndexType score ; /* current column score */ - IndexType n_col2 ; /* number of non-dense, non-empty columns */ - IndexType n_row2 ; /* number of non-dense, non-empty rows */ - IndexType dense_row_count ; /* remove rows with more entries than this */ - IndexType dense_col_count ; /* remove cols with more entries than this */ - IndexType min_score ; /* smallest column score */ - IndexType max_deg ; /* maximum row degree */ - IndexType next_col ; /* Used to add to degree list.*/ + IndexType c; /* a column index */ + IndexType r, row; /* a row index */ + IndexType *cp; /* a column pointer */ + IndexType deg; /* degree of a row or column */ + IndexType *cp_end; /* a pointer to the end of a column */ + IndexType *new_cp; /* new column pointer */ + IndexType col_length; /* length of pruned column */ + IndexType score; /* current column score */ + IndexType n_col2; /* number of non-dense, non-empty columns */ + IndexType n_row2; /* number of non-dense, non-empty rows */ + IndexType dense_row_count; /* remove rows with more entries than this */ + IndexType dense_col_count; /* remove cols with more entries than this */ + IndexType min_score; /* smallest column score */ + IndexType max_deg; /* maximum row degree */ + IndexType next_col; /* Used to add to degree list.*/ /* === Extract knobs ==================================================== */ - dense_row_count = numext::maxi(IndexType(0), numext::mini(IndexType(knobs [COLAMD_DENSE_ROW] * n_col), n_col)) ; - dense_col_count = numext::maxi(IndexType(0), numext::mini(IndexType(knobs [COLAMD_DENSE_COL] * n_row), n_row)) ; - COLAMD_DEBUG1 (("colamd: densecount: %d %d\n", dense_row_count, dense_col_count)) ; - max_deg = 0 ; - n_col2 = n_col ; - n_row2 = n_row ; + dense_row_count = numext::maxi(IndexType(0), numext::mini(IndexType(knobs[COLAMD_DENSE_ROW] * n_col), n_col)); + dense_col_count = numext::maxi(IndexType(0), numext::mini(IndexType(knobs[COLAMD_DENSE_COL] * n_row), n_row)); + COLAMD_DEBUG1(("colamd: densecount: %d %d\n", dense_row_count, dense_col_count)); + max_deg = 0; + n_col2 = n_col; + n_row2 = n_row; /* === Kill empty columns =============================================== */ /* Put the empty columns at the end in their natural order, so that LU */ /* factorization can proceed as far as possible. */ - for (c = n_col-1 ; c >= 0 ; c--) - { - deg = Col [c].length ; - if (deg == 0) - { + for (c = n_col - 1; c >= 0; c--) { + deg = Col[c].length; + if (deg == 0) { /* this is a empty column, kill and order it last */ - Col [c].shared2.order = --n_col2 ; - KILL_PRINCIPAL_COL (c) ; + Col[c].shared2.order = --n_col2; + KILL_PRINCIPAL_COL(c); } } - COLAMD_DEBUG1 (("colamd: null columns killed: %d\n", n_col - n_col2)) ; + COLAMD_DEBUG1(("colamd: null columns killed: %d\n", n_col - n_col2)); /* === Kill dense columns =============================================== */ /* Put the dense columns at the end, in their natural order */ - for (c = n_col-1 ; c >= 0 ; c--) - { + for (c = n_col - 1; c >= 0; c--) { /* skip any dead columns */ - if (COL_IS_DEAD (c)) - { - continue ; - } - deg = Col [c].length ; - if (deg > dense_col_count) - { + if (COL_IS_DEAD(c)) { continue; } + deg = Col[c].length; + if (deg > dense_col_count) { /* this is a dense column, kill and order it last */ - Col [c].shared2.order = --n_col2 ; + Col[c].shared2.order = --n_col2; /* decrement the row degrees */ - cp = &A [Col [c].start] ; - cp_end = cp + Col [c].length ; - while (cp < cp_end) - { - Row [*cp++].shared1.degree-- ; - } - KILL_PRINCIPAL_COL (c) ; + cp = &A[Col[c].start]; + cp_end = cp + Col[c].length; + while (cp < cp_end) { Row[*cp++].shared1.degree--; } + KILL_PRINCIPAL_COL(c); } } - COLAMD_DEBUG1 (("colamd: Dense and null columns killed: %d\n", n_col - n_col2)) ; + COLAMD_DEBUG1(("colamd: Dense and null columns killed: %d\n", n_col - n_col2)); /* === Kill dense and empty rows ======================================== */ - for (r = 0 ; r < n_row ; r++) - { - deg = Row [r].shared1.degree ; - COLAMD_ASSERT (deg >= 0 && deg <= n_col) ; - if (deg > dense_row_count || deg == 0) - { + for (r = 0; r < n_row; r++) { + deg = Row[r].shared1.degree; + COLAMD_ASSERT(deg >= 0 && deg <= n_col); + if (deg > dense_row_count || deg == 0) { /* kill a dense or empty row */ - KILL_ROW (r) ; - --n_row2 ; - } - else - { + KILL_ROW(r); + --n_row2; + } else { /* keep track of max degree of remaining rows */ - max_deg = numext::maxi(max_deg, deg) ; + max_deg = numext::maxi(max_deg, deg); } } - COLAMD_DEBUG1 (("colamd: Dense and null rows killed: %d\n", n_row - n_row2)) ; + COLAMD_DEBUG1(("colamd: Dense and null rows killed: %d\n", n_row - n_row2)); /* === Compute initial column scores ==================================== */ @@ -810,54 +789,42 @@ static void init_scoring /* pruned in the code below. */ /* now find the initial matlab score for each column */ - for (c = n_col-1 ; c >= 0 ; c--) - { + for (c = n_col - 1; c >= 0; c--) { /* skip dead column */ - if (COL_IS_DEAD (c)) - { - continue ; - } - score = 0 ; - cp = &A [Col [c].start] ; - new_cp = cp ; - cp_end = cp + Col [c].length ; - while (cp < cp_end) - { + if (COL_IS_DEAD(c)) { continue; } + score = 0; + cp = &A[Col[c].start]; + new_cp = cp; + cp_end = cp + Col[c].length; + while (cp < cp_end) { /* get a row */ - row = *cp++ ; + row = *cp++; /* skip if dead */ - if (ROW_IS_DEAD (row)) - { - continue ; - } + if (ROW_IS_DEAD(row)) { continue; } /* compact the column */ - *new_cp++ = row ; + *new_cp++ = row; /* add row's external degree */ - score += Row [row].shared1.degree - 1 ; + score += Row[row].shared1.degree - 1; /* guard against integer overflow */ - score = numext::mini(score, n_col) ; + score = numext::mini(score, n_col); } /* determine pruned column length */ - col_length = (IndexType) (new_cp - &A [Col [c].start]) ; - if (col_length == 0) - { + col_length = (IndexType)(new_cp - &A[Col[c].start]); + if (col_length == 0) { /* a newly-made null column (all rows in this col are "dense" */ /* and have already been killed) */ - COLAMD_DEBUG2 (("Newly null killed: %d\n", c)) ; - Col [c].shared2.order = --n_col2 ; - KILL_PRINCIPAL_COL (c) ; - } - else - { + COLAMD_DEBUG2(("Newly null killed: %d\n", c)); + Col[c].shared2.order = --n_col2; + KILL_PRINCIPAL_COL(c); + } else { /* set column length and set score */ - COLAMD_ASSERT (score >= 0) ; - COLAMD_ASSERT (score <= n_col) ; - Col [c].length = col_length ; - Col [c].shared2.score = score ; + COLAMD_ASSERT(score >= 0); + COLAMD_ASSERT(score <= n_col); + Col[c].length = col_length; + Col[c].shared2.score = score; } } - COLAMD_DEBUG1 (("colamd: Dense, null, and newly-null columns killed: %d\n", - n_col-n_col2)) ; + COLAMD_DEBUG1(("colamd: Dense, null, and newly-null columns killed: %d\n", n_col - n_col2)); /* At this point, all empty rows and columns are dead. All live columns */ /* are "clean" (containing no dead rows) and simplicial (no supercolumns */ @@ -868,57 +835,46 @@ static void init_scoring /* clear the hash buckets */ - for (c = 0 ; c <= n_col ; c++) - { - head [c] = COLAMD_EMPTY ; - } - min_score = n_col ; + for (c = 0; c <= n_col; c++) { head[c] = COLAMD_EMPTY; } + min_score = n_col; /* place in reverse order, so low column indices are at the front */ /* of the lists. This is to encourage natural tie-breaking */ - for (c = n_col-1 ; c >= 0 ; c--) - { + for (c = n_col - 1; c >= 0; c--) { /* only add principal columns to degree lists */ - if (COL_IS_ALIVE (c)) - { - COLAMD_DEBUG4 (("place %d score %d minscore %d ncol %d\n", - c, Col [c].shared2.score, min_score, n_col)) ; + if (COL_IS_ALIVE(c)) { + COLAMD_DEBUG4(("place %d score %d minscore %d ncol %d\n", c, Col[c].shared2.score, min_score, n_col)); /* === Add columns score to DList =============================== */ - score = Col [c].shared2.score ; + score = Col[c].shared2.score; - COLAMD_ASSERT (min_score >= 0) ; - COLAMD_ASSERT (min_score <= n_col) ; - COLAMD_ASSERT (score >= 0) ; - COLAMD_ASSERT (score <= n_col) ; - COLAMD_ASSERT (head [score] >= COLAMD_EMPTY) ; + COLAMD_ASSERT(min_score >= 0); + COLAMD_ASSERT(min_score <= n_col); + COLAMD_ASSERT(score >= 0); + COLAMD_ASSERT(score <= n_col); + COLAMD_ASSERT(head[score] >= COLAMD_EMPTY); /* now add this column to dList at proper score location */ - next_col = head [score] ; - Col [c].shared3.prev = COLAMD_EMPTY ; - Col [c].shared4.degree_next = next_col ; + next_col = head[score]; + Col[c].shared3.prev = COLAMD_EMPTY; + Col[c].shared4.degree_next = next_col; /* if there already was a column with the same score, set its */ /* previous pointer to this new column */ - if (next_col != COLAMD_EMPTY) - { - Col [next_col].shared3.prev = c ; - } - head [score] = c ; + if (next_col != COLAMD_EMPTY) { Col[next_col].shared3.prev = c; } + head[score] = c; /* see if this score is less than current min */ - min_score = numext::mini(min_score, score) ; - - + min_score = numext::mini(min_score, score); } } /* === Return number of remaining columns, and max row degree =========== */ - *p_n_col2 = n_col2 ; - *p_n_row2 = n_row2 ; - *p_max_deg = max_deg ; + *p_n_col2 = n_col2; + *p_n_row2 = n_row2; + *p_max_deg = max_deg; } @@ -931,199 +887,180 @@ static void init_scoring (no supercolumns on input). Uses a minimum approximate column minimum degree ordering method. Not user-callable. */ -template +template static IndexType find_ordering /* return the number of garbage collections */ ( /* === Parameters ======================================================= */ - IndexType n_row, /* number of rows of A */ - IndexType n_col, /* number of columns of A */ - IndexType Alen, /* size of A, 2*nnz + n_col or larger */ - Colamd_Row Row [], /* of size n_row+1 */ - colamd_col Col [], /* of size n_col+1 */ - IndexType A [], /* column form and row form of A */ - IndexType head [], /* of size n_col+1 */ - IndexType n_col2, /* Remaining columns to order */ - IndexType max_deg, /* Maximum row degree */ - IndexType pfree /* index of first free slot (2*nnz on entry) */ - ) + IndexType n_row, /* number of rows of A */ + IndexType n_col, /* number of columns of A */ + IndexType Alen, /* size of A, 2*nnz + n_col or larger */ + Colamd_Row Row[], /* of size n_row+1 */ + colamd_col Col[], /* of size n_col+1 */ + IndexType A[], /* column form and row form of A */ + IndexType head[], /* of size n_col+1 */ + IndexType n_col2, /* Remaining columns to order */ + IndexType max_deg, /* Maximum row degree */ + IndexType pfree /* index of first free slot (2*nnz on entry) */ + ) { /* === Local variables ================================================== */ - IndexType k ; /* current pivot ordering step */ - IndexType pivot_col ; /* current pivot column */ - IndexType *cp ; /* a column pointer */ - IndexType *rp ; /* a row pointer */ - IndexType pivot_row ; /* current pivot row */ - IndexType *new_cp ; /* modified column pointer */ - IndexType *new_rp ; /* modified row pointer */ - IndexType pivot_row_start ; /* pointer to start of pivot row */ - IndexType pivot_row_degree ; /* number of columns in pivot row */ - IndexType pivot_row_length ; /* number of supercolumns in pivot row */ - IndexType pivot_col_score ; /* score of pivot column */ - IndexType needed_memory ; /* free space needed for pivot row */ - IndexType *cp_end ; /* pointer to the end of a column */ - IndexType *rp_end ; /* pointer to the end of a row */ - IndexType row ; /* a row index */ - IndexType col ; /* a column index */ - IndexType max_score ; /* maximum possible score */ - IndexType cur_score ; /* score of current column */ - unsigned int hash ; /* hash value for supernode detection */ - IndexType head_column ; /* head of hash bucket */ - IndexType first_col ; /* first column in hash bucket */ - IndexType tag_mark ; /* marker value for mark array */ - IndexType row_mark ; /* Row [row].shared2.mark */ - IndexType set_difference ; /* set difference size of row with pivot row */ - IndexType min_score ; /* smallest column score */ - IndexType col_thickness ; /* "thickness" (no. of columns in a supercol) */ - IndexType max_mark ; /* maximum value of tag_mark */ - IndexType pivot_col_thickness ; /* number of columns represented by pivot col */ - IndexType prev_col ; /* Used by Dlist operations. */ - IndexType next_col ; /* Used by Dlist operations. */ - IndexType ngarbage ; /* number of garbage collections performed */ + IndexType k; /* current pivot ordering step */ + IndexType pivot_col; /* current pivot column */ + IndexType *cp; /* a column pointer */ + IndexType *rp; /* a row pointer */ + IndexType pivot_row; /* current pivot row */ + IndexType *new_cp; /* modified column pointer */ + IndexType *new_rp; /* modified row pointer */ + IndexType pivot_row_start; /* pointer to start of pivot row */ + IndexType pivot_row_degree; /* number of columns in pivot row */ + IndexType pivot_row_length; /* number of supercolumns in pivot row */ + IndexType pivot_col_score; /* score of pivot column */ + IndexType needed_memory; /* free space needed for pivot row */ + IndexType *cp_end; /* pointer to the end of a column */ + IndexType *rp_end; /* pointer to the end of a row */ + IndexType row; /* a row index */ + IndexType col; /* a column index */ + IndexType max_score; /* maximum possible score */ + IndexType cur_score; /* score of current column */ + unsigned int hash; /* hash value for supernode detection */ + IndexType head_column; /* head of hash bucket */ + IndexType first_col; /* first column in hash bucket */ + IndexType tag_mark; /* marker value for mark array */ + IndexType row_mark; /* Row [row].shared2.mark */ + IndexType set_difference; /* set difference size of row with pivot row */ + IndexType min_score; /* smallest column score */ + IndexType col_thickness; /* "thickness" (no. of columns in a supercol) */ + IndexType max_mark; /* maximum value of tag_mark */ + IndexType pivot_col_thickness; /* number of columns represented by pivot col */ + IndexType prev_col; /* Used by Dlist operations. */ + IndexType next_col; /* Used by Dlist operations. */ + IndexType ngarbage; /* number of garbage collections performed */ /* === Initialization and clear mark ==================================== */ - max_mark = INT_MAX - n_col ; /* INT_MAX defined in */ - tag_mark = Eigen::internal::clear_mark (n_row, Row) ; - min_score = 0 ; - ngarbage = 0 ; - COLAMD_DEBUG1 (("colamd: Ordering, n_col2=%d\n", n_col2)) ; + max_mark = INT_MAX - n_col; /* INT_MAX defined in */ + tag_mark = Eigen::internal::clear_mark(n_row, Row); + min_score = 0; + ngarbage = 0; + COLAMD_DEBUG1(("colamd: Ordering, n_col2=%d\n", n_col2)); /* === Order the columns ================================================ */ - for (k = 0 ; k < n_col2 ; /* 'k' is incremented below */) - { + for (k = 0; k < n_col2; /* 'k' is incremented below */) { /* === Select pivot column, and order it ============================ */ /* make sure degree list isn't empty */ - COLAMD_ASSERT (min_score >= 0) ; - COLAMD_ASSERT (min_score <= n_col) ; - COLAMD_ASSERT (head [min_score] >= COLAMD_EMPTY) ; + COLAMD_ASSERT(min_score >= 0); + COLAMD_ASSERT(min_score <= n_col); + COLAMD_ASSERT(head[min_score] >= COLAMD_EMPTY); /* get pivot column from head of minimum degree list */ - while (min_score < n_col && head [min_score] == COLAMD_EMPTY) - { - min_score++ ; - } - pivot_col = head [min_score] ; - COLAMD_ASSERT (pivot_col >= 0 && pivot_col <= n_col) ; - next_col = Col [pivot_col].shared4.degree_next ; - head [min_score] = next_col ; - if (next_col != COLAMD_EMPTY) - { - Col [next_col].shared3.prev = COLAMD_EMPTY ; - } + while (min_score < n_col && head[min_score] == COLAMD_EMPTY) { min_score++; } + pivot_col = head[min_score]; + COLAMD_ASSERT(pivot_col >= 0 && pivot_col <= n_col); + next_col = Col[pivot_col].shared4.degree_next; + head[min_score] = next_col; + if (next_col != COLAMD_EMPTY) { Col[next_col].shared3.prev = COLAMD_EMPTY; } - COLAMD_ASSERT (COL_IS_ALIVE (pivot_col)) ; - COLAMD_DEBUG3 (("Pivot col: %d\n", pivot_col)) ; + COLAMD_ASSERT(COL_IS_ALIVE(pivot_col)); + COLAMD_DEBUG3(("Pivot col: %d\n", pivot_col)); /* remember score for defrag check */ - pivot_col_score = Col [pivot_col].shared2.score ; + pivot_col_score = Col[pivot_col].shared2.score; /* the pivot column is the kth column in the pivot order */ - Col [pivot_col].shared2.order = k ; + Col[pivot_col].shared2.order = k; /* increment order count by column thickness */ - pivot_col_thickness = Col [pivot_col].shared1.thickness ; - k += pivot_col_thickness ; - COLAMD_ASSERT (pivot_col_thickness > 0) ; + pivot_col_thickness = Col[pivot_col].shared1.thickness; + k += pivot_col_thickness; + COLAMD_ASSERT(pivot_col_thickness > 0); /* === Garbage_collection, if necessary ============================= */ - needed_memory = numext::mini(pivot_col_score, n_col - k) ; - if (pfree + needed_memory >= Alen) - { - pfree = Eigen::internal::garbage_collection (n_row, n_col, Row, Col, A, &A [pfree]) ; - ngarbage++ ; + needed_memory = numext::mini(pivot_col_score, n_col - k); + if (pfree + needed_memory >= Alen) { + pfree = Eigen::internal::garbage_collection(n_row, n_col, Row, Col, A, &A[pfree]); + ngarbage++; /* after garbage collection we will have enough */ - COLAMD_ASSERT (pfree + needed_memory < Alen) ; + COLAMD_ASSERT(pfree + needed_memory < Alen); /* garbage collection has wiped out the Row[].shared2.mark array */ - tag_mark = Eigen::internal::clear_mark (n_row, Row) ; - + tag_mark = Eigen::internal::clear_mark(n_row, Row); } /* === Compute pivot row pattern ==================================== */ /* get starting location for this new merged row */ - pivot_row_start = pfree ; + pivot_row_start = pfree; /* initialize new row counts to zero */ - pivot_row_degree = 0 ; + pivot_row_degree = 0; /* tag pivot column as having been visited so it isn't included */ /* in merged pivot row */ - Col [pivot_col].shared1.thickness = -pivot_col_thickness ; + Col[pivot_col].shared1.thickness = -pivot_col_thickness; /* pivot row is the union of all rows in the pivot column pattern */ - cp = &A [Col [pivot_col].start] ; - cp_end = cp + Col [pivot_col].length ; - while (cp < cp_end) - { + cp = &A[Col[pivot_col].start]; + cp_end = cp + Col[pivot_col].length; + while (cp < cp_end) { /* get a row */ - row = *cp++ ; - COLAMD_DEBUG4 (("Pivot col pattern %d %d\n", ROW_IS_ALIVE (row), row)) ; + row = *cp++; + COLAMD_DEBUG4(("Pivot col pattern %d %d\n", ROW_IS_ALIVE(row), row)); /* skip if row is dead */ - if (ROW_IS_DEAD (row)) - { - continue ; - } - rp = &A [Row [row].start] ; - rp_end = rp + Row [row].length ; - while (rp < rp_end) - { - /* get a column */ - col = *rp++ ; - /* add the column, if alive and untagged */ - col_thickness = Col [col].shared1.thickness ; - if (col_thickness > 0 && COL_IS_ALIVE (col)) - { - /* tag column in pivot row */ - Col [col].shared1.thickness = -col_thickness ; - COLAMD_ASSERT (pfree < Alen) ; - /* place column in pivot row */ - A [pfree++] = col ; - pivot_row_degree += col_thickness ; - } + if (ROW_IS_DEAD(row)) { continue; } + rp = &A[Row[row].start]; + rp_end = rp + Row[row].length; + while (rp < rp_end) { + /* get a column */ + col = *rp++; + /* add the column, if alive and untagged */ + col_thickness = Col[col].shared1.thickness; + if (col_thickness > 0 && COL_IS_ALIVE(col)) { + /* tag column in pivot row */ + Col[col].shared1.thickness = -col_thickness; + COLAMD_ASSERT(pfree < Alen); + /* place column in pivot row */ + A[pfree++] = col; + pivot_row_degree += col_thickness; + } } } /* clear tag on pivot column */ - Col [pivot_col].shared1.thickness = pivot_col_thickness ; - max_deg = numext::maxi(max_deg, pivot_row_degree) ; + Col[pivot_col].shared1.thickness = pivot_col_thickness; + max_deg = numext::maxi(max_deg, pivot_row_degree); /* === Kill all rows used to construct pivot row ==================== */ /* also kill pivot row, temporarily */ - cp = &A [Col [pivot_col].start] ; - cp_end = cp + Col [pivot_col].length ; - while (cp < cp_end) - { + cp = &A[Col[pivot_col].start]; + cp_end = cp + Col[pivot_col].length; + while (cp < cp_end) { /* may be killing an already dead row */ - row = *cp++ ; - COLAMD_DEBUG3 (("Kill row in pivot col: %d\n", row)) ; - KILL_ROW (row) ; + row = *cp++; + COLAMD_DEBUG3(("Kill row in pivot col: %d\n", row)); + KILL_ROW(row); } /* === Select a row index to use as the new pivot row =============== */ - pivot_row_length = pfree - pivot_row_start ; - if (pivot_row_length > 0) - { + pivot_row_length = pfree - pivot_row_start; + if (pivot_row_length > 0) { /* pick the "pivot" row arbitrarily (first row in col) */ - pivot_row = A [Col [pivot_col].start] ; - COLAMD_DEBUG3 (("Pivotal row is %d\n", pivot_row)) ; - } - else - { + pivot_row = A[Col[pivot_col].start]; + COLAMD_DEBUG3(("Pivotal row is %d\n", pivot_row)); + } else { /* there is no pivot row, since it is of zero length */ - pivot_row = COLAMD_EMPTY ; - COLAMD_ASSERT (pivot_row_length == 0) ; + pivot_row = COLAMD_EMPTY; + COLAMD_ASSERT(pivot_row_length == 0); } - COLAMD_ASSERT (Col [pivot_col].length > 0 || pivot_row_length == 0) ; + COLAMD_ASSERT(Col[pivot_col].length > 0 || pivot_row_length == 0); /* === Approximate degree computation =============================== */ @@ -1146,180 +1083,154 @@ static IndexType find_ordering /* return the number of garbage collections */ /* === Compute set differences ====================================== */ - COLAMD_DEBUG3 (("** Computing set differences phase. **\n")) ; + COLAMD_DEBUG3(("** Computing set differences phase. **\n")); /* pivot row is currently dead - it will be revived later. */ - COLAMD_DEBUG3 (("Pivot row: ")) ; + COLAMD_DEBUG3(("Pivot row: ")); /* for each column in pivot row */ - rp = &A [pivot_row_start] ; - rp_end = rp + pivot_row_length ; - while (rp < rp_end) - { - col = *rp++ ; - COLAMD_ASSERT (COL_IS_ALIVE (col) && col != pivot_col) ; - COLAMD_DEBUG3 (("Col: %d\n", col)) ; + rp = &A[pivot_row_start]; + rp_end = rp + pivot_row_length; + while (rp < rp_end) { + col = *rp++; + COLAMD_ASSERT(COL_IS_ALIVE(col) && col != pivot_col); + COLAMD_DEBUG3(("Col: %d\n", col)); /* clear tags used to construct pivot row pattern */ - col_thickness = -Col [col].shared1.thickness ; - COLAMD_ASSERT (col_thickness > 0) ; - Col [col].shared1.thickness = col_thickness ; + col_thickness = -Col[col].shared1.thickness; + COLAMD_ASSERT(col_thickness > 0); + Col[col].shared1.thickness = col_thickness; /* === Remove column from degree list =========================== */ - cur_score = Col [col].shared2.score ; - prev_col = Col [col].shared3.prev ; - next_col = Col [col].shared4.degree_next ; - COLAMD_ASSERT (cur_score >= 0) ; - COLAMD_ASSERT (cur_score <= n_col) ; - COLAMD_ASSERT (cur_score >= COLAMD_EMPTY) ; - if (prev_col == COLAMD_EMPTY) - { - head [cur_score] = next_col ; - } - else - { - Col [prev_col].shared4.degree_next = next_col ; - } - if (next_col != COLAMD_EMPTY) - { - Col [next_col].shared3.prev = prev_col ; + cur_score = Col[col].shared2.score; + prev_col = Col[col].shared3.prev; + next_col = Col[col].shared4.degree_next; + COLAMD_ASSERT(cur_score >= 0); + COLAMD_ASSERT(cur_score <= n_col); + COLAMD_ASSERT(cur_score >= COLAMD_EMPTY); + if (prev_col == COLAMD_EMPTY) { + head[cur_score] = next_col; + } else { + Col[prev_col].shared4.degree_next = next_col; } + if (next_col != COLAMD_EMPTY) { Col[next_col].shared3.prev = prev_col; } /* === Scan the column ========================================== */ - cp = &A [Col [col].start] ; - cp_end = cp + Col [col].length ; - while (cp < cp_end) - { - /* get a row */ - row = *cp++ ; - row_mark = Row [row].shared2.mark ; - /* skip if dead */ - if (ROW_IS_MARKED_DEAD (row_mark)) - { - continue ; - } - COLAMD_ASSERT (row != pivot_row) ; - set_difference = row_mark - tag_mark ; - /* check if the row has been seen yet */ - if (set_difference < 0) - { - COLAMD_ASSERT (Row [row].shared1.degree <= max_deg) ; - set_difference = Row [row].shared1.degree ; - } - /* subtract column thickness from this row's set difference */ - set_difference -= col_thickness ; - COLAMD_ASSERT (set_difference >= 0) ; - /* absorb this row if the set difference becomes zero */ - if (set_difference == 0) - { - COLAMD_DEBUG3 (("aggressive absorption. Row: %d\n", row)) ; - KILL_ROW (row) ; - } - else - { - /* save the new mark */ - Row [row].shared2.mark = set_difference + tag_mark ; - } + cp = &A[Col[col].start]; + cp_end = cp + Col[col].length; + while (cp < cp_end) { + /* get a row */ + row = *cp++; + row_mark = Row[row].shared2.mark; + /* skip if dead */ + if (ROW_IS_MARKED_DEAD(row_mark)) { continue; } + COLAMD_ASSERT(row != pivot_row); + set_difference = row_mark - tag_mark; + /* check if the row has been seen yet */ + if (set_difference < 0) { + COLAMD_ASSERT(Row[row].shared1.degree <= max_deg); + set_difference = Row[row].shared1.degree; + } + /* subtract column thickness from this row's set difference */ + set_difference -= col_thickness; + COLAMD_ASSERT(set_difference >= 0); + /* absorb this row if the set difference becomes zero */ + if (set_difference == 0) { + COLAMD_DEBUG3(("aggressive absorption. Row: %d\n", row)); + KILL_ROW(row); + } else { + /* save the new mark */ + Row[row].shared2.mark = set_difference + tag_mark; + } } } /* === Add up set differences for each column ======================= */ - COLAMD_DEBUG3 (("** Adding set differences phase. **\n")) ; + COLAMD_DEBUG3(("** Adding set differences phase. **\n")); /* for each column in pivot row */ - rp = &A [pivot_row_start] ; - rp_end = rp + pivot_row_length ; - while (rp < rp_end) - { + rp = &A[pivot_row_start]; + rp_end = rp + pivot_row_length; + while (rp < rp_end) { /* get a column */ - col = *rp++ ; - COLAMD_ASSERT (COL_IS_ALIVE (col) && col != pivot_col) ; - hash = 0 ; - cur_score = 0 ; - cp = &A [Col [col].start] ; + col = *rp++; + COLAMD_ASSERT(COL_IS_ALIVE(col) && col != pivot_col); + hash = 0; + cur_score = 0; + cp = &A[Col[col].start]; /* compact the column */ - new_cp = cp ; - cp_end = cp + Col [col].length ; - - COLAMD_DEBUG4 (("Adding set diffs for Col: %d.\n", col)) ; - - while (cp < cp_end) - { - /* get a row */ - row = *cp++ ; - COLAMD_ASSERT(row >= 0 && row < n_row) ; - row_mark = Row [row].shared2.mark ; - /* skip if dead */ - if (ROW_IS_MARKED_DEAD (row_mark)) - { - continue ; - } - COLAMD_ASSERT (row_mark > tag_mark) ; - /* compact the column */ - *new_cp++ = row ; - /* compute hash function */ - hash += row ; - /* add set difference */ - cur_score += row_mark - tag_mark ; - /* integer overflow... */ - cur_score = numext::mini(cur_score, n_col) ; + new_cp = cp; + cp_end = cp + Col[col].length; + + COLAMD_DEBUG4(("Adding set diffs for Col: %d.\n", col)); + + while (cp < cp_end) { + /* get a row */ + row = *cp++; + COLAMD_ASSERT(row >= 0 && row < n_row); + row_mark = Row[row].shared2.mark; + /* skip if dead */ + if (ROW_IS_MARKED_DEAD(row_mark)) { continue; } + COLAMD_ASSERT(row_mark > tag_mark); + /* compact the column */ + *new_cp++ = row; + /* compute hash function */ + hash += row; + /* add set difference */ + cur_score += row_mark - tag_mark; + /* integer overflow... */ + cur_score = numext::mini(cur_score, n_col); } /* recompute the column's length */ - Col [col].length = (IndexType) (new_cp - &A [Col [col].start]) ; + Col[col].length = (IndexType)(new_cp - &A[Col[col].start]); /* === Further mass elimination ================================= */ - if (Col [col].length == 0) - { - COLAMD_DEBUG4 (("further mass elimination. Col: %d\n", col)) ; - /* nothing left but the pivot row in this column */ - KILL_PRINCIPAL_COL (col) ; - pivot_row_degree -= Col [col].shared1.thickness ; - COLAMD_ASSERT (pivot_row_degree >= 0) ; - /* order it */ - Col [col].shared2.order = k ; - /* increment order count by column thickness */ - k += Col [col].shared1.thickness ; - } - else - { - /* === Prepare for supercolumn detection ==================== */ - - COLAMD_DEBUG4 (("Preparing supercol detection for Col: %d.\n", col)) ; - - /* save score so far */ - Col [col].shared2.score = cur_score ; - - /* add column to hash table, for supercolumn detection */ - hash %= n_col + 1 ; - - COLAMD_DEBUG4 ((" Hash = %d, n_col = %d.\n", hash, n_col)) ; - COLAMD_ASSERT (hash <= n_col) ; - - head_column = head [hash] ; - if (head_column > COLAMD_EMPTY) - { - /* degree list "hash" is non-empty, use prev (shared3) of */ - /* first column in degree list as head of hash bucket */ - first_col = Col [head_column].shared3.headhash ; - Col [head_column].shared3.headhash = col ; - } - else - { - /* degree list "hash" is empty, use head as hash bucket */ - first_col = - (head_column + 2) ; - head [hash] = - (col + 2) ; - } - Col [col].shared4.hash_next = first_col ; - - /* save hash function in Col [col].shared3.hash */ - Col [col].shared3.hash = (IndexType) hash ; - COLAMD_ASSERT (COL_IS_ALIVE (col)) ; + if (Col[col].length == 0) { + COLAMD_DEBUG4(("further mass elimination. Col: %d\n", col)); + /* nothing left but the pivot row in this column */ + KILL_PRINCIPAL_COL(col); + pivot_row_degree -= Col[col].shared1.thickness; + COLAMD_ASSERT(pivot_row_degree >= 0); + /* order it */ + Col[col].shared2.order = k; + /* increment order count by column thickness */ + k += Col[col].shared1.thickness; + } else { + /* === Prepare for supercolumn detection ==================== */ + + COLAMD_DEBUG4(("Preparing supercol detection for Col: %d.\n", col)); + + /* save score so far */ + Col[col].shared2.score = cur_score; + + /* add column to hash table, for supercolumn detection */ + hash %= n_col + 1; + + COLAMD_DEBUG4((" Hash = %d, n_col = %d.\n", hash, n_col)); + COLAMD_ASSERT(hash <= n_col); + + head_column = head[hash]; + if (head_column > COLAMD_EMPTY) { + /* degree list "hash" is non-empty, use prev (shared3) of */ + /* first column in degree list as head of hash bucket */ + first_col = Col[head_column].shared3.headhash; + Col[head_column].shared3.headhash = col; + } else { + /* degree list "hash" is empty, use head as hash bucket */ + first_col = -(head_column + 2); + head[hash] = -(col + 2); + } + Col[col].shared4.hash_next = first_col; + + /* save hash function in Col [col].shared3.hash */ + Col[col].shared3.hash = (IndexType)hash; + COLAMD_ASSERT(COL_IS_ALIVE(col)); } } @@ -1327,102 +1238,92 @@ static IndexType find_ordering /* return the number of garbage collections */ /* === Supercolumn detection ======================================== */ - COLAMD_DEBUG3 (("** Supercolumn detection phase. **\n")) ; + COLAMD_DEBUG3(("** Supercolumn detection phase. **\n")); - Eigen::internal::detect_super_cols (Col, A, head, pivot_row_start, pivot_row_length) ; + Eigen::internal::detect_super_cols(Col, A, head, pivot_row_start, pivot_row_length); /* === Kill the pivotal column ====================================== */ - KILL_PRINCIPAL_COL (pivot_col) ; + KILL_PRINCIPAL_COL(pivot_col); /* === Clear mark =================================================== */ - tag_mark += (max_deg + 1) ; - if (tag_mark >= max_mark) - { - COLAMD_DEBUG2 (("clearing tag_mark\n")) ; - tag_mark = Eigen::internal::clear_mark (n_row, Row) ; + tag_mark += (max_deg + 1); + if (tag_mark >= max_mark) { + COLAMD_DEBUG2(("clearing tag_mark\n")); + tag_mark = Eigen::internal::clear_mark(n_row, Row); } /* === Finalize the new pivot row, and column scores ================ */ - COLAMD_DEBUG3 (("** Finalize scores phase. **\n")) ; + COLAMD_DEBUG3(("** Finalize scores phase. **\n")); /* for each column in pivot row */ - rp = &A [pivot_row_start] ; + rp = &A[pivot_row_start]; /* compact the pivot row */ - new_rp = rp ; - rp_end = rp + pivot_row_length ; - while (rp < rp_end) - { - col = *rp++ ; + new_rp = rp; + rp_end = rp + pivot_row_length; + while (rp < rp_end) { + col = *rp++; /* skip dead columns */ - if (COL_IS_DEAD (col)) - { - continue ; - } - *new_rp++ = col ; + if (COL_IS_DEAD(col)) { continue; } + *new_rp++ = col; /* add new pivot row to column */ - A [Col [col].start + (Col [col].length++)] = pivot_row ; + A[Col[col].start + (Col[col].length++)] = pivot_row; /* retrieve score so far and add on pivot row's degree. */ /* (we wait until here for this in case the pivot */ /* row's degree was reduced due to mass elimination). */ - cur_score = Col [col].shared2.score + pivot_row_degree ; + cur_score = Col[col].shared2.score + pivot_row_degree; /* calculate the max possible score as the number of */ /* external columns minus the 'k' value minus the */ /* columns thickness */ - max_score = n_col - k - Col [col].shared1.thickness ; + max_score = n_col - k - Col[col].shared1.thickness; /* make the score the external degree of the union-of-rows */ - cur_score -= Col [col].shared1.thickness ; + cur_score -= Col[col].shared1.thickness; /* make sure score is less or equal than the max score */ - cur_score = numext::mini(cur_score, max_score) ; - COLAMD_ASSERT (cur_score >= 0) ; + cur_score = numext::mini(cur_score, max_score); + COLAMD_ASSERT(cur_score >= 0); /* store updated score */ - Col [col].shared2.score = cur_score ; + Col[col].shared2.score = cur_score; /* === Place column back in degree list ========================= */ - COLAMD_ASSERT (min_score >= 0) ; - COLAMD_ASSERT (min_score <= n_col) ; - COLAMD_ASSERT (cur_score >= 0) ; - COLAMD_ASSERT (cur_score <= n_col) ; - COLAMD_ASSERT (head [cur_score] >= COLAMD_EMPTY) ; - next_col = head [cur_score] ; - Col [col].shared4.degree_next = next_col ; - Col [col].shared3.prev = COLAMD_EMPTY ; - if (next_col != COLAMD_EMPTY) - { - Col [next_col].shared3.prev = col ; - } - head [cur_score] = col ; + COLAMD_ASSERT(min_score >= 0); + COLAMD_ASSERT(min_score <= n_col); + COLAMD_ASSERT(cur_score >= 0); + COLAMD_ASSERT(cur_score <= n_col); + COLAMD_ASSERT(head[cur_score] >= COLAMD_EMPTY); + next_col = head[cur_score]; + Col[col].shared4.degree_next = next_col; + Col[col].shared3.prev = COLAMD_EMPTY; + if (next_col != COLAMD_EMPTY) { Col[next_col].shared3.prev = col; } + head[cur_score] = col; /* see if this score is less than current min */ - min_score = numext::mini(min_score, cur_score) ; - + min_score = numext::mini(min_score, cur_score); } /* === Resurrect the new pivot row ================================== */ - if (pivot_row_degree > 0) - { + if (pivot_row_degree > 0) { /* update pivot row length to reflect any cols that were killed */ /* during super-col detection and mass elimination */ - Row [pivot_row].start = pivot_row_start ; - Row [pivot_row].length = (IndexType) (new_rp - &A[pivot_row_start]) ; - Row [pivot_row].shared1.degree = pivot_row_degree ; - Row [pivot_row].shared2.mark = 0 ; + Row[pivot_row].start = pivot_row_start; + Row[pivot_row].length = (IndexType)(new_rp - &A[pivot_row_start]); + Row[pivot_row].shared1.degree = pivot_row_degree; + Row[pivot_row].shared2.mark = 0; /* pivot row is no longer dead */ } } /* === All principal columns have now been ordered ====================== */ - return (ngarbage) ; + return (ngarbage); } @@ -1442,72 +1343,64 @@ static IndexType find_ordering /* return the number of garbage collections */ taken by this routine is O (n_col), that is, linear in the number of columns. Not user-callable. */ -template -static inline void order_children -( +template +static inline void order_children( /* === Parameters ======================================================= */ - IndexType n_col, /* number of columns of A */ - colamd_col Col [], /* of size n_col+1 */ - IndexType p [] /* p [0 ... n_col-1] is the column permutation*/ - ) + IndexType n_col, /* number of columns of A */ + colamd_col Col[], /* of size n_col+1 */ + IndexType p[] /* p [0 ... n_col-1] is the column permutation*/ +) { /* === Local variables ================================================== */ - IndexType i ; /* loop counter for all columns */ - IndexType c ; /* column index */ - IndexType parent ; /* index of column's parent */ - IndexType order ; /* column's order */ + IndexType i; /* loop counter for all columns */ + IndexType c; /* column index */ + IndexType parent; /* index of column's parent */ + IndexType order; /* column's order */ /* === Order each non-principal column ================================== */ - for (i = 0 ; i < n_col ; i++) - { + for (i = 0; i < n_col; i++) { /* find an un-ordered non-principal column */ - COLAMD_ASSERT (COL_IS_DEAD (i)) ; - if (!COL_IS_DEAD_PRINCIPAL (i) && Col [i].shared2.order == COLAMD_EMPTY) - { - parent = i ; + COLAMD_ASSERT(COL_IS_DEAD(i)); + if (!COL_IS_DEAD_PRINCIPAL(i) && Col[i].shared2.order == COLAMD_EMPTY) { + parent = i; /* once found, find its principal parent */ - do - { - parent = Col [parent].shared1.parent ; - } while (!COL_IS_DEAD_PRINCIPAL (parent)) ; + do { + parent = Col[parent].shared1.parent; + } while (!COL_IS_DEAD_PRINCIPAL(parent)); /* now, order all un-ordered non-principal columns along path */ /* to this parent. collapse tree at the same time */ - c = i ; + c = i; /* get order of parent */ - order = Col [parent].shared2.order ; + order = Col[parent].shared2.order; - do - { - COLAMD_ASSERT (Col [c].shared2.order == COLAMD_EMPTY) ; + do { + COLAMD_ASSERT(Col[c].shared2.order == COLAMD_EMPTY); - /* order this column */ - Col [c].shared2.order = order++ ; - /* collaps tree */ - Col [c].shared1.parent = parent ; + /* order this column */ + Col[c].shared2.order = order++; + /* collaps tree */ + Col[c].shared1.parent = parent; - /* get immediate parent of this column */ - c = Col [c].shared1.parent ; + /* get immediate parent of this column */ + c = Col[c].shared1.parent; - /* continue until we hit an ordered column. There are */ - /* guarranteed not to be anymore unordered columns */ - /* above an ordered column */ - } while (Col [c].shared2.order == COLAMD_EMPTY) ; + /* continue until we hit an ordered column. There are */ + /* guarranteed not to be anymore unordered columns */ + /* above an ordered column */ + } while (Col[c].shared2.order == COLAMD_EMPTY); /* re-order the super_col parent to largest order for this group */ - Col [parent].shared2.order = order ; + Col[parent].shared2.order = order; } } /* === Generate the permutation ========================================= */ - for (c = 0 ; c < n_col ; c++) - { - p [Col [c].shared2.order] = c ; - } + for (c = 0; c < n_col; c++) { p[Col[c].shared2.order] = c; } } @@ -1543,140 +1436,118 @@ static inline void order_children just been computed in the approximate degree computation. Not user-callable. */ -template -static void detect_super_cols -( +template +static void detect_super_cols( /* === Parameters ======================================================= */ - - colamd_col Col [], /* of size n_col+1 */ - IndexType A [], /* row indices of A */ - IndexType head [], /* head of degree lists and hash buckets */ - IndexType row_start, /* pointer to set of columns to check */ - IndexType row_length /* number of columns to check */ + + colamd_col Col[], /* of size n_col+1 */ + IndexType A[], /* row indices of A */ + IndexType head[], /* head of degree lists and hash buckets */ + IndexType row_start, /* pointer to set of columns to check */ + IndexType row_length /* number of columns to check */ ) { /* === Local variables ================================================== */ - IndexType hash ; /* hash value for a column */ - IndexType *rp ; /* pointer to a row */ - IndexType c ; /* a column index */ - IndexType super_c ; /* column index of the column to absorb into */ - IndexType *cp1 ; /* column pointer for column super_c */ - IndexType *cp2 ; /* column pointer for column c */ - IndexType length ; /* length of column super_c */ - IndexType prev_c ; /* column preceding c in hash bucket */ - IndexType i ; /* loop counter */ - IndexType *rp_end ; /* pointer to the end of the row */ - IndexType col ; /* a column index in the row to check */ - IndexType head_column ; /* first column in hash bucket or degree list */ - IndexType first_col ; /* first column in hash bucket */ + IndexType hash; /* hash value for a column */ + IndexType *rp; /* pointer to a row */ + IndexType c; /* a column index */ + IndexType super_c; /* column index of the column to absorb into */ + IndexType *cp1; /* column pointer for column super_c */ + IndexType *cp2; /* column pointer for column c */ + IndexType length; /* length of column super_c */ + IndexType prev_c; /* column preceding c in hash bucket */ + IndexType i; /* loop counter */ + IndexType *rp_end; /* pointer to the end of the row */ + IndexType col; /* a column index in the row to check */ + IndexType head_column; /* first column in hash bucket or degree list */ + IndexType first_col; /* first column in hash bucket */ /* === Consider each column in the row ================================== */ - rp = &A [row_start] ; - rp_end = rp + row_length ; - while (rp < rp_end) - { - col = *rp++ ; - if (COL_IS_DEAD (col)) - { - continue ; - } + rp = &A[row_start]; + rp_end = rp + row_length; + while (rp < rp_end) { + col = *rp++; + if (COL_IS_DEAD(col)) { continue; } /* get hash number for this column */ - hash = Col [col].shared3.hash ; - COLAMD_ASSERT (hash <= n_col) ; + hash = Col[col].shared3.hash; + COLAMD_ASSERT(hash <= n_col); /* === Get the first column in this hash bucket ===================== */ - head_column = head [hash] ; - if (head_column > COLAMD_EMPTY) - { - first_col = Col [head_column].shared3.headhash ; - } - else - { - first_col = - (head_column + 2) ; + head_column = head[hash]; + if (head_column > COLAMD_EMPTY) { + first_col = Col[head_column].shared3.headhash; + } else { + first_col = -(head_column + 2); } /* === Consider each column in the hash bucket ====================== */ - for (super_c = first_col ; super_c != COLAMD_EMPTY ; - super_c = Col [super_c].shared4.hash_next) - { - COLAMD_ASSERT (COL_IS_ALIVE (super_c)) ; - COLAMD_ASSERT (Col [super_c].shared3.hash == hash) ; - length = Col [super_c].length ; + for (super_c = first_col; super_c != COLAMD_EMPTY; super_c = Col[super_c].shared4.hash_next) { + COLAMD_ASSERT(COL_IS_ALIVE(super_c)); + COLAMD_ASSERT(Col[super_c].shared3.hash == hash); + length = Col[super_c].length; /* prev_c is the column preceding column c in the hash bucket */ - prev_c = super_c ; + prev_c = super_c; /* === Compare super_c with all columns after it ================ */ - for (c = Col [super_c].shared4.hash_next ; - c != COLAMD_EMPTY ; c = Col [c].shared4.hash_next) - { - COLAMD_ASSERT (c != super_c) ; - COLAMD_ASSERT (COL_IS_ALIVE (c)) ; - COLAMD_ASSERT (Col [c].shared3.hash == hash) ; - - /* not identical if lengths or scores are different */ - if (Col [c].length != length || - Col [c].shared2.score != Col [super_c].shared2.score) - { - prev_c = c ; - continue ; - } - - /* compare the two columns */ - cp1 = &A [Col [super_c].start] ; - cp2 = &A [Col [c].start] ; - - for (i = 0 ; i < length ; i++) - { - /* the columns are "clean" (no dead rows) */ - COLAMD_ASSERT (ROW_IS_ALIVE (*cp1)) ; - COLAMD_ASSERT (ROW_IS_ALIVE (*cp2)) ; - /* row indices will same order for both supercols, */ - /* no gather scatter nessasary */ - if (*cp1++ != *cp2++) - { - break ; - } - } - - /* the two columns are different if the for-loop "broke" */ - if (i != length) - { - prev_c = c ; - continue ; - } - - /* === Got it! two columns are identical =================== */ - - COLAMD_ASSERT (Col [c].shared2.score == Col [super_c].shared2.score) ; - - Col [super_c].shared1.thickness += Col [c].shared1.thickness ; - Col [c].shared1.parent = super_c ; - KILL_NON_PRINCIPAL_COL (c) ; - /* order c later, in order_children() */ - Col [c].shared2.order = COLAMD_EMPTY ; - /* remove c from hash bucket */ - Col [prev_c].shared4.hash_next = Col [c].shared4.hash_next ; + for (c = Col[super_c].shared4.hash_next; c != COLAMD_EMPTY; c = Col[c].shared4.hash_next) { + COLAMD_ASSERT(c != super_c); + COLAMD_ASSERT(COL_IS_ALIVE(c)); + COLAMD_ASSERT(Col[c].shared3.hash == hash); + + /* not identical if lengths or scores are different */ + if (Col[c].length != length || Col[c].shared2.score != Col[super_c].shared2.score) { + prev_c = c; + continue; + } + + /* compare the two columns */ + cp1 = &A[Col[super_c].start]; + cp2 = &A[Col[c].start]; + + for (i = 0; i < length; i++) { + /* the columns are "clean" (no dead rows) */ + COLAMD_ASSERT(ROW_IS_ALIVE(*cp1)); + COLAMD_ASSERT(ROW_IS_ALIVE(*cp2)); + /* row indices will same order for both supercols, */ + /* no gather scatter nessasary */ + if (*cp1++ != *cp2++) { break; } + } + + /* the two columns are different if the for-loop "broke" */ + if (i != length) { + prev_c = c; + continue; + } + + /* === Got it! two columns are identical =================== */ + + COLAMD_ASSERT(Col[c].shared2.score == Col[super_c].shared2.score); + + Col[super_c].shared1.thickness += Col[c].shared1.thickness; + Col[c].shared1.parent = super_c; + KILL_NON_PRINCIPAL_COL(c); + /* order c later, in order_children() */ + Col[c].shared2.order = COLAMD_EMPTY; + /* remove c from hash bucket */ + Col[prev_c].shared4.hash_next = Col[c].shared4.hash_next; } } /* === Empty this hash bucket ======================================= */ - if (head_column > COLAMD_EMPTY) - { + if (head_column > COLAMD_EMPTY) { /* corresponding degree list "hash" is not empty */ - Col [head_column].shared3.headhash = COLAMD_EMPTY ; - } - else - { + Col[head_column].shared3.headhash = COLAMD_EMPTY; + } else { /* corresponding degree list "hash" is empty */ - head [hash] = COLAMD_EMPTY ; + head[hash] = COLAMD_EMPTY; } } } @@ -1694,116 +1565,97 @@ static void detect_super_cols itself linear in the number of nonzeros in the input matrix. Not user-callable. */ -template -static IndexType garbage_collection /* returns the new value of pfree */ +template +static IndexType garbage_collection /* returns the new value of pfree */ ( /* === Parameters ======================================================= */ - - IndexType n_row, /* number of rows */ - IndexType n_col, /* number of columns */ - Colamd_Row Row [], /* row info */ - colamd_col Col [], /* column info */ - IndexType A [], /* A [0 ... Alen-1] holds the matrix */ - IndexType *pfree /* &A [0] ... pfree is in use */ - ) + + IndexType n_row, /* number of rows */ + IndexType n_col, /* number of columns */ + Colamd_Row Row[], /* row info */ + colamd_col Col[], /* column info */ + IndexType A[], /* A [0 ... Alen-1] holds the matrix */ + IndexType *pfree /* &A [0] ... pfree is in use */ + ) { /* === Local variables ================================================== */ - IndexType *psrc ; /* source pointer */ - IndexType *pdest ; /* destination pointer */ - IndexType j ; /* counter */ - IndexType r ; /* a row index */ - IndexType c ; /* a column index */ - IndexType length ; /* length of a row or column */ + IndexType *psrc; /* source pointer */ + IndexType *pdest; /* destination pointer */ + IndexType j; /* counter */ + IndexType r; /* a row index */ + IndexType c; /* a column index */ + IndexType length; /* length of a row or column */ /* === Defragment the columns =========================================== */ - pdest = &A[0] ; - for (c = 0 ; c < n_col ; c++) - { - if (COL_IS_ALIVE (c)) - { - psrc = &A [Col [c].start] ; + pdest = &A[0]; + for (c = 0; c < n_col; c++) { + if (COL_IS_ALIVE(c)) { + psrc = &A[Col[c].start]; /* move and compact the column */ - COLAMD_ASSERT (pdest <= psrc) ; - Col [c].start = (IndexType) (pdest - &A [0]) ; - length = Col [c].length ; - for (j = 0 ; j < length ; j++) - { - r = *psrc++ ; - if (ROW_IS_ALIVE (r)) - { - *pdest++ = r ; - } + COLAMD_ASSERT(pdest <= psrc); + Col[c].start = (IndexType)(pdest - &A[0]); + length = Col[c].length; + for (j = 0; j < length; j++) { + r = *psrc++; + if (ROW_IS_ALIVE(r)) { *pdest++ = r; } } - Col [c].length = (IndexType) (pdest - &A [Col [c].start]) ; + Col[c].length = (IndexType)(pdest - &A[Col[c].start]); } } /* === Prepare to defragment the rows =================================== */ - for (r = 0 ; r < n_row ; r++) - { - if (ROW_IS_ALIVE (r)) - { - if (Row [r].length == 0) - { - /* this row is of zero length. cannot compact it, so kill it */ - COLAMD_DEBUG3 (("Defrag row kill\n")) ; - KILL_ROW (r) ; - } - else - { - /* save first column index in Row [r].shared2.first_column */ - psrc = &A [Row [r].start] ; - Row [r].shared2.first_column = *psrc ; - COLAMD_ASSERT (ROW_IS_ALIVE (r)) ; - /* flag the start of the row with the one's complement of row */ - *psrc = ONES_COMPLEMENT (r) ; - + for (r = 0; r < n_row; r++) { + if (ROW_IS_ALIVE(r)) { + if (Row[r].length == 0) { + /* this row is of zero length. cannot compact it, so kill it */ + COLAMD_DEBUG3(("Defrag row kill\n")); + KILL_ROW(r); + } else { + /* save first column index in Row [r].shared2.first_column */ + psrc = &A[Row[r].start]; + Row[r].shared2.first_column = *psrc; + COLAMD_ASSERT(ROW_IS_ALIVE(r)); + /* flag the start of the row with the one's complement of row */ + *psrc = ONES_COMPLEMENT(r); } } } /* === Defragment the rows ============================================== */ - psrc = pdest ; - while (psrc < pfree) - { + psrc = pdest; + while (psrc < pfree) { /* find a negative number ... the start of a row */ - if (*psrc++ < 0) - { - psrc-- ; + if (*psrc++ < 0) { + psrc--; /* get the row index */ - r = ONES_COMPLEMENT (*psrc) ; - COLAMD_ASSERT (r >= 0 && r < n_row) ; + r = ONES_COMPLEMENT(*psrc); + COLAMD_ASSERT(r >= 0 && r < n_row); /* restore first column index */ - *psrc = Row [r].shared2.first_column ; - COLAMD_ASSERT (ROW_IS_ALIVE (r)) ; + *psrc = Row[r].shared2.first_column; + COLAMD_ASSERT(ROW_IS_ALIVE(r)); /* move and compact the row */ - COLAMD_ASSERT (pdest <= psrc) ; - Row [r].start = (IndexType) (pdest - &A [0]) ; - length = Row [r].length ; - for (j = 0 ; j < length ; j++) - { - c = *psrc++ ; - if (COL_IS_ALIVE (c)) - { - *pdest++ = c ; - } + COLAMD_ASSERT(pdest <= psrc); + Row[r].start = (IndexType)(pdest - &A[0]); + length = Row[r].length; + for (j = 0; j < length; j++) { + c = *psrc++; + if (COL_IS_ALIVE(c)) { *pdest++ = c; } } - Row [r].length = (IndexType) (pdest - &A [Row [r].start]) ; - + Row[r].length = (IndexType)(pdest - &A[Row[r].start]); } } /* ensure we found all the rows */ - COLAMD_ASSERT (debug_rows == 0) ; + COLAMD_ASSERT(debug_rows == 0); /* === Return the new value of pfree ==================================== */ - return ((IndexType) (pdest - &A [0])) ; + return ((IndexType)(pdest - &A[0])); } @@ -1815,29 +1667,25 @@ static IndexType garbage_collection /* returns the new value of pfree */ Clears the Row [].shared2.mark array, and returns the new tag_mark. Return value is the new tag_mark. Not user-callable. */ -template -static inline IndexType clear_mark /* return the new value for tag_mark */ +template +static inline IndexType clear_mark /* return the new value for tag_mark */ ( - /* === Parameters ======================================================= */ + /* === Parameters ======================================================= */ - IndexType n_row, /* number of rows in A */ - Colamd_Row Row [] /* Row [0 ... n_row-1].shared2.mark is set to zero */ - ) + IndexType n_row, /* number of rows in A */ + Colamd_Row Row[] /* Row [0 ... n_row-1].shared2.mark is set to zero */ + ) { /* === Local variables ================================================== */ - IndexType r ; + IndexType r; - for (r = 0 ; r < n_row ; r++) - { - if (ROW_IS_ALIVE (r)) - { - Row [r].shared2.mark = 0 ; - } + for (r = 0; r < n_row; r++) { + if (ROW_IS_ALIVE(r)) { Row[r].shared2.mark = 0; } } - return (1) ; + return (1); } -} // namespace internal +}// namespace internal #endif diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/OrderingMethods/Ordering.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/OrderingMethods/Ordering.h index 7ea9b14d..673070bf 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/OrderingMethods/Ordering.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/OrderingMethods/Ordering.h @@ -1,4 +1,4 @@ - + // This file is part of Eigen, a lightweight C++ template library // for linear algebra. // @@ -12,146 +12,134 @@ #define EIGEN_ORDERING_H namespace Eigen { - + #include "Eigen_Colamd.h" namespace internal { - -/** \internal - * \ingroup OrderingMethods_Module - * \param[in] A the input non-symmetric matrix - * \param[out] symmat the symmetric pattern A^T+A from the input matrix \a A. - * FIXME: The values should not be considered here - */ -template -void ordering_helper_at_plus_a(const MatrixType& A, MatrixType& symmat) -{ - MatrixType C; - C = A.transpose(); // NOTE: Could be costly - for (int i = 0; i < C.rows(); i++) + + /** \internal + * \ingroup OrderingMethods_Module + * \param[in] A the input non-symmetric matrix + * \param[out] symmat the symmetric pattern A^T+A from the input matrix \a A. + * FIXME: The values should not be considered here + */ + template void ordering_helper_at_plus_a(const MatrixType &A, MatrixType &symmat) { - for (typename MatrixType::InnerIterator it(C, i); it; ++it) - it.valueRef() = 0.0; + MatrixType C; + C = A.transpose();// NOTE: Could be costly + for (int i = 0; i < C.rows(); i++) { + for (typename MatrixType::InnerIterator it(C, i); it; ++it) it.valueRef() = 0.0; + } + symmat = C + A; } - symmat = C + A; -} - -} + +}// namespace internal #ifndef EIGEN_MPL2_ONLY /** \ingroup OrderingMethods_Module - * \class AMDOrdering - * - * Functor computing the \em approximate \em minimum \em degree ordering - * If the matrix is not structurally symmetric, an ordering of A^T+A is computed - * \tparam StorageIndex The type of indices of the matrix - * \sa COLAMDOrdering - */ -template -class AMDOrdering + * \class AMDOrdering + * + * Functor computing the \em approximate \em minimum \em degree ordering + * If the matrix is not structurally symmetric, an ordering of A^T+A is computed + * \tparam StorageIndex The type of indices of the matrix + * \sa COLAMDOrdering + */ +template class AMDOrdering { - public: - typedef PermutationMatrix PermutationType; - - /** Compute the permutation vector from a sparse matrix - * This routine is much faster if the input matrix is column-major - */ - template - void operator()(const MatrixType& mat, PermutationType& perm) - { - // Compute the symmetric pattern - SparseMatrix symm; - internal::ordering_helper_at_plus_a(mat,symm); - - // Call the AMD routine - //m_mat.prune(keep_diag()); - internal::minimum_degree_ordering(symm, perm); - } - - /** Compute the permutation with a selfadjoint matrix */ - template - void operator()(const SparseSelfAdjointView& mat, PermutationType& perm) - { - SparseMatrix C; C = mat; - - // Call the AMD routine - // m_mat.prune(keep_diag()); //Remove the diagonal elements - internal::minimum_degree_ordering(C, perm); - } +public: + typedef PermutationMatrix PermutationType; + + /** Compute the permutation vector from a sparse matrix + * This routine is much faster if the input matrix is column-major + */ + template void operator()(const MatrixType &mat, PermutationType &perm) + { + // Compute the symmetric pattern + SparseMatrix symm; + internal::ordering_helper_at_plus_a(mat, symm); + + // Call the AMD routine + // m_mat.prune(keep_diag()); + internal::minimum_degree_ordering(symm, perm); + } + + /** Compute the permutation with a selfadjoint matrix */ + template + void operator()(const SparseSelfAdjointView &mat, PermutationType &perm) + { + SparseMatrix C; + C = mat; + + // Call the AMD routine + // m_mat.prune(keep_diag()); //Remove the diagonal elements + internal::minimum_degree_ordering(C, perm); + } }; -#endif // EIGEN_MPL2_ONLY +#endif// EIGEN_MPL2_ONLY /** \ingroup OrderingMethods_Module - * \class NaturalOrdering - * - * Functor computing the natural ordering (identity) - * - * \note Returns an empty permutation matrix - * \tparam StorageIndex The type of indices of the matrix - */ -template -class NaturalOrdering + * \class NaturalOrdering + * + * Functor computing the natural ordering (identity) + * + * \note Returns an empty permutation matrix + * \tparam StorageIndex The type of indices of the matrix + */ +template class NaturalOrdering { - public: - typedef PermutationMatrix PermutationType; - - /** Compute the permutation vector from a column-major sparse matrix */ - template - void operator()(const MatrixType& /*mat*/, PermutationType& perm) - { - perm.resize(0); - } - +public: + typedef PermutationMatrix PermutationType; + + /** Compute the permutation vector from a column-major sparse matrix */ + template void operator()(const MatrixType & /*mat*/, PermutationType &perm) { perm.resize(0); } }; /** \ingroup OrderingMethods_Module - * \class COLAMDOrdering - * - * \tparam StorageIndex The type of indices of the matrix - * - * Functor computing the \em column \em approximate \em minimum \em degree ordering - * The matrix should be in column-major and \b compressed format (see SparseMatrix::makeCompressed()). - */ -template -class COLAMDOrdering + * \class COLAMDOrdering + * + * \tparam StorageIndex The type of indices of the matrix + * + * Functor computing the \em column \em approximate \em minimum \em degree ordering + * The matrix should be in column-major and \b compressed format (see SparseMatrix::makeCompressed()). + */ +template class COLAMDOrdering { - public: - typedef PermutationMatrix PermutationType; - typedef Matrix IndexVector; - - /** Compute the permutation vector \a perm form the sparse matrix \a mat - * \warning The input sparse matrix \a mat must be in compressed mode (see SparseMatrix::makeCompressed()). - */ - template - void operator() (const MatrixType& mat, PermutationType& perm) - { - eigen_assert(mat.isCompressed() && "COLAMDOrdering requires a sparse matrix in compressed mode. Call .makeCompressed() before passing it to COLAMDOrdering"); - - StorageIndex m = StorageIndex(mat.rows()); - StorageIndex n = StorageIndex(mat.cols()); - StorageIndex nnz = StorageIndex(mat.nonZeros()); - // Get the recommended value of Alen to be used by colamd - StorageIndex Alen = internal::colamd_recommended(nnz, m, n); - // Set the default parameters - double knobs [COLAMD_KNOBS]; - StorageIndex stats [COLAMD_STATS]; - internal::colamd_set_defaults(knobs); - - IndexVector p(n+1), A(Alen); - for(StorageIndex i=0; i <= n; i++) p(i) = mat.outerIndexPtr()[i]; - for(StorageIndex i=0; i < nnz; i++) A(i) = mat.innerIndexPtr()[i]; - // Call Colamd routine to compute the ordering - StorageIndex info = internal::colamd(m, n, Alen, A.data(), p.data(), knobs, stats); - EIGEN_UNUSED_VARIABLE(info); - eigen_assert( info && "COLAMD failed " ); - - perm.resize(n); - for (StorageIndex i = 0; i < n; i++) perm.indices()(p(i)) = i; - } +public: + typedef PermutationMatrix PermutationType; + typedef Matrix IndexVector; + + /** Compute the permutation vector \a perm form the sparse matrix \a mat + * \warning The input sparse matrix \a mat must be in compressed mode (see SparseMatrix::makeCompressed()). + */ + template void operator()(const MatrixType &mat, PermutationType &perm) + { + eigen_assert(mat.isCompressed() && "COLAMDOrdering requires a sparse matrix in compressed mode. Call .makeCompressed() before passing it to COLAMDOrdering"); + + StorageIndex m = StorageIndex(mat.rows()); + StorageIndex n = StorageIndex(mat.cols()); + StorageIndex nnz = StorageIndex(mat.nonZeros()); + // Get the recommended value of Alen to be used by colamd + StorageIndex Alen = internal::colamd_recommended(nnz, m, n); + // Set the default parameters + double knobs[COLAMD_KNOBS]; + StorageIndex stats[COLAMD_STATS]; + internal::colamd_set_defaults(knobs); + + IndexVector p(n + 1), A(Alen); + for (StorageIndex i = 0; i <= n; i++) p(i) = mat.outerIndexPtr()[i]; + for (StorageIndex i = 0; i < nnz; i++) A(i) = mat.innerIndexPtr()[i]; + // Call Colamd routine to compute the ordering + StorageIndex info = internal::colamd(m, n, Alen, A.data(), p.data(), knobs, stats); + EIGEN_UNUSED_VARIABLE(info); + eigen_assert(info && "COLAMD failed "); + + perm.resize(n); + for (StorageIndex i = 0; i < n; i++) perm.indices()(p(i)) = i; + } }; -} // end namespace Eigen +}// end namespace Eigen #endif diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/PaStiXSupport/PaStiXSupport.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/PaStiXSupport/PaStiXSupport.h index 160d8a52..03dee034 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/PaStiXSupport/PaStiXSupport.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/PaStiXSupport/PaStiXSupport.h @@ -10,35 +10,33 @@ #ifndef EIGEN_PASTIXSUPPORT_H #define EIGEN_PASTIXSUPPORT_H -namespace Eigen { +namespace Eigen { #if defined(DCOMPLEX) - #define PASTIX_COMPLEX COMPLEX - #define PASTIX_DCOMPLEX DCOMPLEX +#define PASTIX_COMPLEX COMPLEX +#define PASTIX_DCOMPLEX DCOMPLEX #else - #define PASTIX_COMPLEX std::complex - #define PASTIX_DCOMPLEX std::complex +#define PASTIX_COMPLEX std::complex +#define PASTIX_DCOMPLEX std::complex #endif /** \ingroup PaStiXSupport_Module - * \brief Interface to the PaStix solver - * - * This class is used to solve the linear systems A.X = B via the PaStix library. - * The matrix can be either real or complex, symmetric or not. - * - * \sa TutorialSparseDirectSolvers - */ + * \brief Interface to the PaStix solver + * + * This class is used to solve the linear systems A.X = B via the PaStix library. + * The matrix can be either real or complex, symmetric or not. + * + * \sa TutorialSparseDirectSolvers + */ template class PastixLU; template class PastixLLT; template class PastixLDLT; -namespace internal -{ - +namespace internal { + template struct pastix_traits; - template - struct pastix_traits< PastixLU<_MatrixType> > + template struct pastix_traits> { typedef _MatrixType MatrixType; typedef typename _MatrixType::Scalar Scalar; @@ -46,8 +44,7 @@ namespace internal typedef typename _MatrixType::StorageIndex StorageIndex; }; - template - struct pastix_traits< PastixLLT<_MatrixType,Options> > + template struct pastix_traits> { typedef _MatrixType MatrixType; typedef typename _MatrixType::Scalar Scalar; @@ -55,306 +52,386 @@ namespace internal typedef typename _MatrixType::StorageIndex StorageIndex; }; - template - struct pastix_traits< PastixLDLT<_MatrixType,Options> > + template struct pastix_traits> { typedef _MatrixType MatrixType; typedef typename _MatrixType::Scalar Scalar; typedef typename _MatrixType::RealScalar RealScalar; typedef typename _MatrixType::StorageIndex StorageIndex; }; - - inline void eigen_pastix(pastix_data_t **pastix_data, int pastix_comm, int n, int *ptr, int *idx, float *vals, int *perm, int * invp, float *x, int nbrhs, int *iparm, double *dparm) + + inline void eigen_pastix(pastix_data_t **pastix_data, + int pastix_comm, + int n, + int *ptr, + int *idx, + float *vals, + int *perm, + int *invp, + float *x, + int nbrhs, + int *iparm, + double *dparm) { - if (n == 0) { ptr = NULL; idx = NULL; vals = NULL; } - if (nbrhs == 0) {x = NULL; nbrhs=1;} - s_pastix(pastix_data, pastix_comm, n, ptr, idx, vals, perm, invp, x, nbrhs, iparm, dparm); + if (n == 0) { + ptr = NULL; + idx = NULL; + vals = NULL; + } + if (nbrhs == 0) { + x = NULL; + nbrhs = 1; + } + s_pastix(pastix_data, pastix_comm, n, ptr, idx, vals, perm, invp, x, nbrhs, iparm, dparm); } - - inline void eigen_pastix(pastix_data_t **pastix_data, int pastix_comm, int n, int *ptr, int *idx, double *vals, int *perm, int * invp, double *x, int nbrhs, int *iparm, double *dparm) + + inline void eigen_pastix(pastix_data_t **pastix_data, + int pastix_comm, + int n, + int *ptr, + int *idx, + double *vals, + int *perm, + int *invp, + double *x, + int nbrhs, + int *iparm, + double *dparm) { - if (n == 0) { ptr = NULL; idx = NULL; vals = NULL; } - if (nbrhs == 0) {x = NULL; nbrhs=1;} - d_pastix(pastix_data, pastix_comm, n, ptr, idx, vals, perm, invp, x, nbrhs, iparm, dparm); + if (n == 0) { + ptr = NULL; + idx = NULL; + vals = NULL; + } + if (nbrhs == 0) { + x = NULL; + nbrhs = 1; + } + d_pastix(pastix_data, pastix_comm, n, ptr, idx, vals, perm, invp, x, nbrhs, iparm, dparm); } - - inline void eigen_pastix(pastix_data_t **pastix_data, int pastix_comm, int n, int *ptr, int *idx, std::complex *vals, int *perm, int * invp, std::complex *x, int nbrhs, int *iparm, double *dparm) + + inline void eigen_pastix(pastix_data_t **pastix_data, + int pastix_comm, + int n, + int *ptr, + int *idx, + std::complex *vals, + int *perm, + int *invp, + std::complex *x, + int nbrhs, + int *iparm, + double *dparm) { - if (n == 0) { ptr = NULL; idx = NULL; vals = NULL; } - if (nbrhs == 0) {x = NULL; nbrhs=1;} - c_pastix(pastix_data, pastix_comm, n, ptr, idx, reinterpret_cast(vals), perm, invp, reinterpret_cast(x), nbrhs, iparm, dparm); + if (n == 0) { + ptr = NULL; + idx = NULL; + vals = NULL; + } + if (nbrhs == 0) { + x = NULL; + nbrhs = 1; + } + c_pastix(pastix_data, + pastix_comm, + n, + ptr, + idx, + reinterpret_cast(vals), + perm, + invp, + reinterpret_cast(x), + nbrhs, + iparm, + dparm); } - - inline void eigen_pastix(pastix_data_t **pastix_data, int pastix_comm, int n, int *ptr, int *idx, std::complex *vals, int *perm, int * invp, std::complex *x, int nbrhs, int *iparm, double *dparm) + + inline void eigen_pastix(pastix_data_t **pastix_data, + int pastix_comm, + int n, + int *ptr, + int *idx, + std::complex *vals, + int *perm, + int *invp, + std::complex *x, + int nbrhs, + int *iparm, + double *dparm) { - if (n == 0) { ptr = NULL; idx = NULL; vals = NULL; } - if (nbrhs == 0) {x = NULL; nbrhs=1;} - z_pastix(pastix_data, pastix_comm, n, ptr, idx, reinterpret_cast(vals), perm, invp, reinterpret_cast(x), nbrhs, iparm, dparm); + if (n == 0) { + ptr = NULL; + idx = NULL; + vals = NULL; + } + if (nbrhs == 0) { + x = NULL; + nbrhs = 1; + } + z_pastix(pastix_data, + pastix_comm, + n, + ptr, + idx, + reinterpret_cast(vals), + perm, + invp, + reinterpret_cast(x), + nbrhs, + iparm, + dparm); } // Convert the matrix to Fortran-style Numbering - template - void c_to_fortran_numbering (MatrixType& mat) + template void c_to_fortran_numbering(MatrixType &mat) { - if ( !(mat.outerIndexPtr()[0]) ) - { + if (!(mat.outerIndexPtr()[0])) { int i; - for(i = 0; i <= mat.rows(); ++i) - ++mat.outerIndexPtr()[i]; - for(i = 0; i < mat.nonZeros(); ++i) - ++mat.innerIndexPtr()[i]; + for (i = 0; i <= mat.rows(); ++i) ++mat.outerIndexPtr()[i]; + for (i = 0; i < mat.nonZeros(); ++i) ++mat.innerIndexPtr()[i]; } } - + // Convert to C-style Numbering - template - void fortran_to_c_numbering (MatrixType& mat) + template void fortran_to_c_numbering(MatrixType &mat) { // Check the Numbering - if ( mat.outerIndexPtr()[0] == 1 ) - { // Convert to C-style numbering + if (mat.outerIndexPtr()[0] == 1) {// Convert to C-style numbering int i; - for(i = 0; i <= mat.rows(); ++i) - --mat.outerIndexPtr()[i]; - for(i = 0; i < mat.nonZeros(); ++i) - --mat.innerIndexPtr()[i]; + for (i = 0; i <= mat.rows(); ++i) --mat.outerIndexPtr()[i]; + for (i = 0; i < mat.nonZeros(); ++i) --mat.innerIndexPtr()[i]; } } -} +}// namespace internal -// This is the base class to interface with PaStiX functions. -// Users should not used this class directly. -template -class PastixBase : public SparseSolverBase +// This is the base class to interface with PaStiX functions. +// Users should not used this class directly. +template class PastixBase : public SparseSolverBase { - protected: - typedef SparseSolverBase Base; - using Base::derived; - using Base::m_isInitialized; - public: - using Base::_solve_impl; - - typedef typename internal::pastix_traits::MatrixType _MatrixType; - typedef _MatrixType MatrixType; - typedef typename MatrixType::Scalar Scalar; - typedef typename MatrixType::RealScalar RealScalar; - typedef typename MatrixType::StorageIndex StorageIndex; - typedef Matrix Vector; - typedef SparseMatrix ColSpMatrix; - enum { - ColsAtCompileTime = MatrixType::ColsAtCompileTime, - MaxColsAtCompileTime = MatrixType::MaxColsAtCompileTime - }; - - public: - - PastixBase() : m_initisOk(false), m_analysisIsOk(false), m_factorizationIsOk(false), m_pastixdata(0), m_size(0) - { - init(); - } - - ~PastixBase() - { - clean(); - } - - template - bool _solve_impl(const MatrixBase &b, MatrixBase &x) const; - - /** Returns a reference to the integer vector IPARM of PaStiX parameters - * to modify the default parameters. - * The statistics related to the different phases of factorization and solve are saved here as well - * \sa analyzePattern() factorize() - */ - Array& iparm() - { - return m_iparm; - } - - /** Return a reference to a particular index parameter of the IPARM vector - * \sa iparm() - */ - - int& iparm(int idxparam) - { - return m_iparm(idxparam); - } - - /** Returns a reference to the double vector DPARM of PaStiX parameters - * The statistics related to the different phases of factorization and solve are saved here as well - * \sa analyzePattern() factorize() - */ - Array& dparm() - { - return m_dparm; - } - - - /** Return a reference to a particular index parameter of the DPARM vector - * \sa dparm() - */ - double& dparm(int idxparam) - { - return m_dparm(idxparam); - } - - inline Index cols() const { return m_size; } - inline Index rows() const { return m_size; } - - /** \brief Reports whether previous computation was successful. - * - * \returns \c Success if computation was succesful, - * \c NumericalIssue if the PaStiX reports a problem - * \c InvalidInput if the input matrix is invalid - * - * \sa iparm() - */ - ComputationInfo info() const - { - eigen_assert(m_isInitialized && "Decomposition is not initialized."); - return m_info; - } - - protected: - - // Initialize the Pastix data structure, check the matrix - void init(); - - // Compute the ordering and the symbolic factorization - void analyzePattern(ColSpMatrix& mat); - - // Compute the numerical factorization - void factorize(ColSpMatrix& mat); - - // Free all the data allocated by Pastix - void clean() - { - eigen_assert(m_initisOk && "The Pastix structure should be allocated first"); - m_iparm(IPARM_START_TASK) = API_TASK_CLEAN; - m_iparm(IPARM_END_TASK) = API_TASK_CLEAN; - internal::eigen_pastix(&m_pastixdata, MPI_COMM_WORLD, 0, 0, 0, (Scalar*)0, - m_perm.data(), m_invp.data(), 0, 0, m_iparm.data(), m_dparm.data()); - } - - void compute(ColSpMatrix& mat); - - int m_initisOk; - int m_analysisIsOk; - int m_factorizationIsOk; - mutable ComputationInfo m_info; - mutable pastix_data_t *m_pastixdata; // Data structure for pastix - mutable int m_comm; // The MPI communicator identifier - mutable Array m_iparm; // integer vector for the input parameters - mutable Array m_dparm; // Scalar vector for the input parameters - mutable Matrix m_perm; // Permutation vector - mutable Matrix m_invp; // Inverse permutation vector - mutable int m_size; // Size of the matrix -}; - - /** Initialize the PaStiX data structure. - *A first call to this function fills iparm and dparm with the default PaStiX parameters - * \sa iparm() dparm() +protected: + typedef SparseSolverBase Base; + using Base::derived; + using Base::m_isInitialized; + +public: + using Base::_solve_impl; + + typedef typename internal::pastix_traits::MatrixType _MatrixType; + typedef _MatrixType MatrixType; + typedef typename MatrixType::Scalar Scalar; + typedef typename MatrixType::RealScalar RealScalar; + typedef typename MatrixType::StorageIndex StorageIndex; + typedef Matrix Vector; + typedef SparseMatrix ColSpMatrix; + enum { ColsAtCompileTime = MatrixType::ColsAtCompileTime, MaxColsAtCompileTime = MatrixType::MaxColsAtCompileTime }; + +public: + PastixBase() : m_initisOk(false), m_analysisIsOk(false), m_factorizationIsOk(false), m_pastixdata(0), m_size(0) + { + init(); + } + + ~PastixBase() { clean(); } + + template bool _solve_impl(const MatrixBase &b, MatrixBase &x) const; + + /** Returns a reference to the integer vector IPARM of PaStiX parameters + * to modify the default parameters. + * The statistics related to the different phases of factorization and solve are saved here as well + * \sa analyzePattern() factorize() + */ + Array &iparm() { return m_iparm; } + + /** Return a reference to a particular index parameter of the IPARM vector + * \sa iparm() + */ + + int &iparm(int idxparam) { return m_iparm(idxparam); } + + /** Returns a reference to the double vector DPARM of PaStiX parameters + * The statistics related to the different phases of factorization and solve are saved here as well + * \sa analyzePattern() factorize() */ -template -void PastixBase::init() + Array &dparm() { return m_dparm; } + + + /** Return a reference to a particular index parameter of the DPARM vector + * \sa dparm() + */ + double &dparm(int idxparam) { return m_dparm(idxparam); } + + inline Index cols() const { return m_size; } + inline Index rows() const { return m_size; } + + /** \brief Reports whether previous computation was successful. + * + * \returns \c Success if computation was succesful, + * \c NumericalIssue if the PaStiX reports a problem + * \c InvalidInput if the input matrix is invalid + * + * \sa iparm() + */ + ComputationInfo info() const + { + eigen_assert(m_isInitialized && "Decomposition is not initialized."); + return m_info; + } + +protected: + // Initialize the Pastix data structure, check the matrix + void init(); + + // Compute the ordering and the symbolic factorization + void analyzePattern(ColSpMatrix &mat); + + // Compute the numerical factorization + void factorize(ColSpMatrix &mat); + + // Free all the data allocated by Pastix + void clean() + { + eigen_assert(m_initisOk && "The Pastix structure should be allocated first"); + m_iparm(IPARM_START_TASK) = API_TASK_CLEAN; + m_iparm(IPARM_END_TASK) = API_TASK_CLEAN; + internal::eigen_pastix(&m_pastixdata, + MPI_COMM_WORLD, + 0, + 0, + 0, + (Scalar *)0, + m_perm.data(), + m_invp.data(), + 0, + 0, + m_iparm.data(), + m_dparm.data()); + } + + void compute(ColSpMatrix &mat); + + int m_initisOk; + int m_analysisIsOk; + int m_factorizationIsOk; + mutable ComputationInfo m_info; + mutable pastix_data_t *m_pastixdata;// Data structure for pastix + mutable int m_comm;// The MPI communicator identifier + mutable Array m_iparm;// integer vector for the input parameters + mutable Array m_dparm;// Scalar vector for the input parameters + mutable Matrix m_perm;// Permutation vector + mutable Matrix m_invp;// Inverse permutation vector + mutable int m_size;// Size of the matrix +}; + +/** Initialize the PaStiX data structure. + *A first call to this function fills iparm and dparm with the default PaStiX parameters + * \sa iparm() dparm() + */ +template void PastixBase::init() { - m_size = 0; + m_size = 0; m_iparm.setZero(IPARM_SIZE); m_dparm.setZero(DPARM_SIZE); - + m_iparm(IPARM_MODIFY_PARAMETER) = API_NO; - pastix(&m_pastixdata, MPI_COMM_WORLD, - 0, 0, 0, 0, - 0, 0, 0, 1, m_iparm.data(), m_dparm.data()); - + pastix(&m_pastixdata, MPI_COMM_WORLD, 0, 0, 0, 0, 0, 0, 0, 1, m_iparm.data(), m_dparm.data()); + m_iparm[IPARM_MATRIX_VERIFICATION] = API_NO; - m_iparm[IPARM_VERBOSE] = API_VERBOSE_NOT; - m_iparm[IPARM_ORDERING] = API_ORDER_SCOTCH; - m_iparm[IPARM_INCOMPLETE] = API_NO; - m_iparm[IPARM_OOC_LIMIT] = 2000; - m_iparm[IPARM_RHS_MAKING] = API_RHS_B; + m_iparm[IPARM_VERBOSE] = API_VERBOSE_NOT; + m_iparm[IPARM_ORDERING] = API_ORDER_SCOTCH; + m_iparm[IPARM_INCOMPLETE] = API_NO; + m_iparm[IPARM_OOC_LIMIT] = 2000; + m_iparm[IPARM_RHS_MAKING] = API_RHS_B; m_iparm(IPARM_MATRIX_VERIFICATION) = API_NO; - + m_iparm(IPARM_START_TASK) = API_TASK_INIT; m_iparm(IPARM_END_TASK) = API_TASK_INIT; - internal::eigen_pastix(&m_pastixdata, MPI_COMM_WORLD, 0, 0, 0, (Scalar*)0, - 0, 0, 0, 0, m_iparm.data(), m_dparm.data()); - + internal::eigen_pastix( + &m_pastixdata, MPI_COMM_WORLD, 0, 0, 0, (Scalar *)0, 0, 0, 0, 0, m_iparm.data(), m_dparm.data()); + // Check the returned error - if(m_iparm(IPARM_ERROR_NUMBER)) { + if (m_iparm(IPARM_ERROR_NUMBER)) { m_info = InvalidInput; m_initisOk = false; - } - else { + } else { m_info = Success; m_initisOk = true; } } -template -void PastixBase::compute(ColSpMatrix& mat) +template void PastixBase::compute(ColSpMatrix &mat) { eigen_assert(mat.rows() == mat.cols() && "The input matrix should be squared"); - - analyzePattern(mat); + + analyzePattern(mat); factorize(mat); - + m_iparm(IPARM_MATRIX_VERIFICATION) = API_NO; } -template -void PastixBase::analyzePattern(ColSpMatrix& mat) -{ +template void PastixBase::analyzePattern(ColSpMatrix &mat) +{ eigen_assert(m_initisOk && "The initialization of PaSTiX failed"); - + // clean previous calls - if(m_size>0) - clean(); - + if (m_size > 0) clean(); + m_size = internal::convert_index(mat.rows()); m_perm.resize(m_size); m_invp.resize(m_size); - + m_iparm(IPARM_START_TASK) = API_TASK_ORDERING; m_iparm(IPARM_END_TASK) = API_TASK_ANALYSE; - internal::eigen_pastix(&m_pastixdata, MPI_COMM_WORLD, m_size, mat.outerIndexPtr(), mat.innerIndexPtr(), - mat.valuePtr(), m_perm.data(), m_invp.data(), 0, 0, m_iparm.data(), m_dparm.data()); - + internal::eigen_pastix(&m_pastixdata, + MPI_COMM_WORLD, + m_size, + mat.outerIndexPtr(), + mat.innerIndexPtr(), + mat.valuePtr(), + m_perm.data(), + m_invp.data(), + 0, + 0, + m_iparm.data(), + m_dparm.data()); + // Check the returned error - if(m_iparm(IPARM_ERROR_NUMBER)) - { + if (m_iparm(IPARM_ERROR_NUMBER)) { m_info = NumericalIssue; m_analysisIsOk = false; - } - else - { + } else { m_info = Success; m_analysisIsOk = true; } } -template -void PastixBase::factorize(ColSpMatrix& mat) +template void PastixBase::factorize(ColSpMatrix &mat) { -// if(&m_cpyMat != &mat) m_cpyMat = mat; + // if(&m_cpyMat != &mat) m_cpyMat = mat; eigen_assert(m_analysisIsOk && "The analysis phase should be called before the factorization phase"); m_iparm(IPARM_START_TASK) = API_TASK_NUMFACT; m_iparm(IPARM_END_TASK) = API_TASK_NUMFACT; m_size = internal::convert_index(mat.rows()); - - internal::eigen_pastix(&m_pastixdata, MPI_COMM_WORLD, m_size, mat.outerIndexPtr(), mat.innerIndexPtr(), - mat.valuePtr(), m_perm.data(), m_invp.data(), 0, 0, m_iparm.data(), m_dparm.data()); - + + internal::eigen_pastix(&m_pastixdata, + MPI_COMM_WORLD, + m_size, + mat.outerIndexPtr(), + mat.innerIndexPtr(), + mat.valuePtr(), + m_perm.data(), + m_invp.data(), + 0, + 0, + m_iparm.data(), + m_dparm.data()); + // Check the returned error - if(m_iparm(IPARM_ERROR_NUMBER)) - { + if (m_iparm(IPARM_ERROR_NUMBER)) { m_info = NumericalIssue; m_factorizationIsOk = false; m_isInitialized = false; - } - else - { + } else { m_info = Success; m_factorizationIsOk = true; m_isInitialized = true; @@ -363,316 +440,311 @@ void PastixBase::factorize(ColSpMatrix& mat) /* Solve the system */ template -template +template bool PastixBase::_solve_impl(const MatrixBase &b, MatrixBase &x) const { eigen_assert(m_isInitialized && "The matrix should be factorized first"); - EIGEN_STATIC_ASSERT((Dest::Flags&RowMajorBit)==0, - THIS_METHOD_IS_ONLY_FOR_COLUMN_MAJOR_MATRICES); + EIGEN_STATIC_ASSERT((Dest::Flags & RowMajorBit) == 0, THIS_METHOD_IS_ONLY_FOR_COLUMN_MAJOR_MATRICES); int rhs = 1; - + x = b; /* on return, x is overwritten by the computed solution */ - - for (int i = 0; i < b.cols(); i++){ - m_iparm[IPARM_START_TASK] = API_TASK_SOLVE; - m_iparm[IPARM_END_TASK] = API_TASK_REFINE; - - internal::eigen_pastix(&m_pastixdata, MPI_COMM_WORLD, internal::convert_index(x.rows()), 0, 0, 0, - m_perm.data(), m_invp.data(), &x(0, i), rhs, m_iparm.data(), m_dparm.data()); + + for (int i = 0; i < b.cols(); i++) { + m_iparm[IPARM_START_TASK] = API_TASK_SOLVE; + m_iparm[IPARM_END_TASK] = API_TASK_REFINE; + + internal::eigen_pastix(&m_pastixdata, + MPI_COMM_WORLD, + internal::convert_index(x.rows()), + 0, + 0, + 0, + m_perm.data(), + m_invp.data(), + &x(0, i), + rhs, + m_iparm.data(), + m_dparm.data()); } - + // Check the returned error - m_info = m_iparm(IPARM_ERROR_NUMBER)==0 ? Success : NumericalIssue; - - return m_iparm(IPARM_ERROR_NUMBER)==0; + m_info = m_iparm(IPARM_ERROR_NUMBER) == 0 ? Success : NumericalIssue; + + return m_iparm(IPARM_ERROR_NUMBER) == 0; } /** \ingroup PaStiXSupport_Module - * \class PastixLU - * \brief Sparse direct LU solver based on PaStiX library - * - * This class is used to solve the linear systems A.X = B with a supernodal LU - * factorization in the PaStiX library. The matrix A should be squared and nonsingular - * PaStiX requires that the matrix A has a symmetric structural pattern. - * This interface can symmetrize the input matrix otherwise. - * The vectors or matrices X and B can be either dense or sparse. - * - * \tparam _MatrixType the type of the sparse matrix A, it must be a SparseMatrix<> - * \tparam IsStrSym Indicates if the input matrix has a symmetric pattern, default is false - * NOTE : Note that if the analysis and factorization phase are called separately, - * the input matrix will be symmetrized at each call, hence it is advised to - * symmetrize the matrix in a end-user program and set \p IsStrSym to true - * - * \implsparsesolverconcept - * - * \sa \ref TutorialSparseSolverConcept, class SparseLU - * - */ -template -class PastixLU : public PastixBase< PastixLU<_MatrixType> > + * \class PastixLU + * \brief Sparse direct LU solver based on PaStiX library + * + * This class is used to solve the linear systems A.X = B with a supernodal LU + * factorization in the PaStiX library. The matrix A should be squared and nonsingular + * PaStiX requires that the matrix A has a symmetric structural pattern. + * This interface can symmetrize the input matrix otherwise. + * The vectors or matrices X and B can be either dense or sparse. + * + * \tparam _MatrixType the type of the sparse matrix A, it must be a SparseMatrix<> + * \tparam IsStrSym Indicates if the input matrix has a symmetric pattern, default is false + * NOTE : Note that if the analysis and factorization phase are called separately, + * the input matrix will be symmetrized at each call, hence it is advised to + * symmetrize the matrix in a end-user program and set \p IsStrSym to true + * + * \implsparsesolverconcept + * + * \sa \ref TutorialSparseSolverConcept, class SparseLU + * + */ +template class PastixLU : public PastixBase> { - public: - typedef _MatrixType MatrixType; - typedef PastixBase > Base; - typedef typename Base::ColSpMatrix ColSpMatrix; - typedef typename MatrixType::StorageIndex StorageIndex; - - public: - PastixLU() : Base() - { - init(); - } - - explicit PastixLU(const MatrixType& matrix):Base() - { - init(); - compute(matrix); - } - /** Compute the LU supernodal factorization of \p matrix. - * iparm and dparm can be used to tune the PaStiX parameters. - * see the PaStiX user's manual - * \sa analyzePattern() factorize() - */ - void compute (const MatrixType& matrix) - { - m_structureIsUptodate = false; - ColSpMatrix temp; - grabMatrix(matrix, temp); - Base::compute(temp); - } - /** Compute the LU symbolic factorization of \p matrix using its sparsity pattern. - * Several ordering methods can be used at this step. See the PaStiX user's manual. - * The result of this operation can be used with successive matrices having the same pattern as \p matrix - * \sa factorize() - */ - void analyzePattern(const MatrixType& matrix) - { - m_structureIsUptodate = false; - ColSpMatrix temp; - grabMatrix(matrix, temp); - Base::analyzePattern(temp); - } +public: + typedef _MatrixType MatrixType; + typedef PastixBase> Base; + typedef typename Base::ColSpMatrix ColSpMatrix; + typedef typename MatrixType::StorageIndex StorageIndex; - /** Compute the LU supernodal factorization of \p matrix - * WARNING The matrix \p matrix should have the same structural pattern - * as the same used in the analysis phase. - * \sa analyzePattern() - */ - void factorize(const MatrixType& matrix) - { - ColSpMatrix temp; - grabMatrix(matrix, temp); - Base::factorize(temp); - } - protected: - - void init() - { - m_structureIsUptodate = false; - m_iparm(IPARM_SYM) = API_SYM_NO; - m_iparm(IPARM_FACTORIZATION) = API_FACT_LU; - } - - void grabMatrix(const MatrixType& matrix, ColSpMatrix& out) - { - if(IsStrSym) - out = matrix; - else - { - if(!m_structureIsUptodate) - { - // update the transposed structure - m_transposedStructure = matrix.transpose(); - - // Set the elements of the matrix to zero - for (Index j=0; j - * \tparam UpLo The part of the matrix to use : Lower or Upper. The default is Lower as required by PaStiX - * - * \implsparsesolverconcept - * - * \sa \ref TutorialSparseSolverConcept, class SimplicialLLT - */ -template -class PastixLLT : public PastixBase< PastixLLT<_MatrixType, _UpLo> > + * \class PastixLLT + * \brief A sparse direct supernodal Cholesky (LLT) factorization and solver based on the PaStiX library + * + * This class is used to solve the linear systems A.X = B via a LL^T supernodal Cholesky factorization + * available in the PaStiX library. The matrix A should be symmetric and positive definite + * WARNING Selfadjoint complex matrices are not supported in the current version of PaStiX + * The vectors or matrices X and B can be either dense or sparse + * + * \tparam MatrixType the type of the sparse matrix A, it must be a SparseMatrix<> + * \tparam UpLo The part of the matrix to use : Lower or Upper. The default is Lower as required by PaStiX + * + * \implsparsesolverconcept + * + * \sa \ref TutorialSparseSolverConcept, class SimplicialLLT + */ +template class PastixLLT : public PastixBase> { - public: - typedef _MatrixType MatrixType; - typedef PastixBase > Base; - typedef typename Base::ColSpMatrix ColSpMatrix; - - public: - enum { UpLo = _UpLo }; - PastixLLT() : Base() - { - init(); - } - - explicit PastixLLT(const MatrixType& matrix):Base() - { - init(); - compute(matrix); - } +public: + typedef _MatrixType MatrixType; + typedef PastixBase> Base; + typedef typename Base::ColSpMatrix ColSpMatrix; - /** Compute the L factor of the LL^T supernodal factorization of \p matrix - * \sa analyzePattern() factorize() - */ - void compute (const MatrixType& matrix) - { - ColSpMatrix temp; - grabMatrix(matrix, temp); - Base::compute(temp); - } +public: + enum { UpLo = _UpLo }; + PastixLLT() : Base() { init(); } - /** Compute the LL^T symbolic factorization of \p matrix using its sparsity pattern - * The result of this operation can be used with successive matrices having the same pattern as \p matrix - * \sa factorize() - */ - void analyzePattern(const MatrixType& matrix) - { - ColSpMatrix temp; - grabMatrix(matrix, temp); - Base::analyzePattern(temp); - } - /** Compute the LL^T supernodal numerical factorization of \p matrix - * \sa analyzePattern() - */ - void factorize(const MatrixType& matrix) - { - ColSpMatrix temp; - grabMatrix(matrix, temp); - Base::factorize(temp); - } - protected: - using Base::m_iparm; - - void init() - { - m_iparm(IPARM_SYM) = API_SYM_YES; - m_iparm(IPARM_FACTORIZATION) = API_FACT_LLT; - } - - void grabMatrix(const MatrixType& matrix, ColSpMatrix& out) - { - out.resize(matrix.rows(), matrix.cols()); - // Pastix supports only lower, column-major matrices - out.template selfadjointView() = matrix.template selfadjointView(); - internal::c_to_fortran_numbering(out); - } + explicit PastixLLT(const MatrixType &matrix) : Base() + { + init(); + compute(matrix); + } + + /** Compute the L factor of the LL^T supernodal factorization of \p matrix + * \sa analyzePattern() factorize() + */ + void compute(const MatrixType &matrix) + { + ColSpMatrix temp; + grabMatrix(matrix, temp); + Base::compute(temp); + } + + /** Compute the LL^T symbolic factorization of \p matrix using its sparsity pattern + * The result of this operation can be used with successive matrices having the same pattern as \p matrix + * \sa factorize() + */ + void analyzePattern(const MatrixType &matrix) + { + ColSpMatrix temp; + grabMatrix(matrix, temp); + Base::analyzePattern(temp); + } + /** Compute the LL^T supernodal numerical factorization of \p matrix + * \sa analyzePattern() + */ + void factorize(const MatrixType &matrix) + { + ColSpMatrix temp; + grabMatrix(matrix, temp); + Base::factorize(temp); + } + +protected: + using Base::m_iparm; + + void init() + { + m_iparm(IPARM_SYM) = API_SYM_YES; + m_iparm(IPARM_FACTORIZATION) = API_FACT_LLT; + } + + void grabMatrix(const MatrixType &matrix, ColSpMatrix &out) + { + out.resize(matrix.rows(), matrix.cols()); + // Pastix supports only lower, column-major matrices + out.template selfadjointView() = matrix.template selfadjointView(); + internal::c_to_fortran_numbering(out); + } }; /** \ingroup PaStiXSupport_Module - * \class PastixLDLT - * \brief A sparse direct supernodal Cholesky (LLT) factorization and solver based on the PaStiX library - * - * This class is used to solve the linear systems A.X = B via a LDL^T supernodal Cholesky factorization - * available in the PaStiX library. The matrix A should be symmetric and positive definite - * WARNING Selfadjoint complex matrices are not supported in the current version of PaStiX - * The vectors or matrices X and B can be either dense or sparse - * - * \tparam MatrixType the type of the sparse matrix A, it must be a SparseMatrix<> - * \tparam UpLo The part of the matrix to use : Lower or Upper. The default is Lower as required by PaStiX - * - * \implsparsesolverconcept - * - * \sa \ref TutorialSparseSolverConcept, class SimplicialLDLT - */ -template -class PastixLDLT : public PastixBase< PastixLDLT<_MatrixType, _UpLo> > + * \class PastixLDLT + * \brief A sparse direct supernodal Cholesky (LLT) factorization and solver based on the PaStiX library + * + * This class is used to solve the linear systems A.X = B via a LDL^T supernodal Cholesky factorization + * available in the PaStiX library. The matrix A should be symmetric and positive definite + * WARNING Selfadjoint complex matrices are not supported in the current version of PaStiX + * The vectors or matrices X and B can be either dense or sparse + * + * \tparam MatrixType the type of the sparse matrix A, it must be a SparseMatrix<> + * \tparam UpLo The part of the matrix to use : Lower or Upper. The default is Lower as required by PaStiX + * + * \implsparsesolverconcept + * + * \sa \ref TutorialSparseSolverConcept, class SimplicialLDLT + */ +template class PastixLDLT : public PastixBase> { - public: - typedef _MatrixType MatrixType; - typedef PastixBase > Base; - typedef typename Base::ColSpMatrix ColSpMatrix; - - public: - enum { UpLo = _UpLo }; - PastixLDLT():Base() - { - init(); - } - - explicit PastixLDLT(const MatrixType& matrix):Base() - { - init(); - compute(matrix); - } +public: + typedef _MatrixType MatrixType; + typedef PastixBase> Base; + typedef typename Base::ColSpMatrix ColSpMatrix; - /** Compute the L and D factors of the LDL^T factorization of \p matrix - * \sa analyzePattern() factorize() - */ - void compute (const MatrixType& matrix) - { - ColSpMatrix temp; - grabMatrix(matrix, temp); - Base::compute(temp); - } +public: + enum { UpLo = _UpLo }; + PastixLDLT() : Base() { init(); } - /** Compute the LDL^T symbolic factorization of \p matrix using its sparsity pattern - * The result of this operation can be used with successive matrices having the same pattern as \p matrix - * \sa factorize() - */ - void analyzePattern(const MatrixType& matrix) - { - ColSpMatrix temp; - grabMatrix(matrix, temp); - Base::analyzePattern(temp); - } - /** Compute the LDL^T supernodal numerical factorization of \p matrix - * - */ - void factorize(const MatrixType& matrix) - { - ColSpMatrix temp; - grabMatrix(matrix, temp); - Base::factorize(temp); - } + explicit PastixLDLT(const MatrixType &matrix) : Base() + { + init(); + compute(matrix); + } - protected: - using Base::m_iparm; - - void init() - { - m_iparm(IPARM_SYM) = API_SYM_YES; - m_iparm(IPARM_FACTORIZATION) = API_FACT_LDLT; - } - - void grabMatrix(const MatrixType& matrix, ColSpMatrix& out) - { - // Pastix supports only lower, column-major matrices - out.resize(matrix.rows(), matrix.cols()); - out.template selfadjointView() = matrix.template selfadjointView(); - internal::c_to_fortran_numbering(out); - } + /** Compute the L and D factors of the LDL^T factorization of \p matrix + * \sa analyzePattern() factorize() + */ + void compute(const MatrixType &matrix) + { + ColSpMatrix temp; + grabMatrix(matrix, temp); + Base::compute(temp); + } + + /** Compute the LDL^T symbolic factorization of \p matrix using its sparsity pattern + * The result of this operation can be used with successive matrices having the same pattern as \p matrix + * \sa factorize() + */ + void analyzePattern(const MatrixType &matrix) + { + ColSpMatrix temp; + grabMatrix(matrix, temp); + Base::analyzePattern(temp); + } + /** Compute the LDL^T supernodal numerical factorization of \p matrix + * + */ + void factorize(const MatrixType &matrix) + { + ColSpMatrix temp; + grabMatrix(matrix, temp); + Base::factorize(temp); + } + +protected: + using Base::m_iparm; + + void init() + { + m_iparm(IPARM_SYM) = API_SYM_YES; + m_iparm(IPARM_FACTORIZATION) = API_FACT_LDLT; + } + + void grabMatrix(const MatrixType &matrix, ColSpMatrix &out) + { + // Pastix supports only lower, column-major matrices + out.resize(matrix.rows(), matrix.cols()); + out.template selfadjointView() = matrix.template selfadjointView(); + internal::c_to_fortran_numbering(out); + } }; -} // end namespace Eigen +}// end namespace Eigen #endif diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/PardisoSupport/PardisoSupport.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/PardisoSupport/PardisoSupport.h index 091c3970..9bf1751e 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/PardisoSupport/PardisoSupport.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/PardisoSupport/PardisoSupport.h @@ -32,31 +32,54 @@ #ifndef EIGEN_PARDISOSUPPORT_H #define EIGEN_PARDISOSUPPORT_H -namespace Eigen { +namespace Eigen { template class PardisoLU; -template class PardisoLLT; -template class PardisoLDLT; +template class PardisoLLT; +template class PardisoLDLT; -namespace internal -{ - template - struct pardiso_run_selector +namespace internal { + template struct pardiso_run_selector { - static IndexType run( _MKL_DSS_HANDLE_t pt, IndexType maxfct, IndexType mnum, IndexType type, IndexType phase, IndexType n, void *a, - IndexType *ia, IndexType *ja, IndexType *perm, IndexType nrhs, IndexType *iparm, IndexType msglvl, void *b, void *x) + static IndexType run(_MKL_DSS_HANDLE_t pt, + IndexType maxfct, + IndexType mnum, + IndexType type, + IndexType phase, + IndexType n, + void *a, + IndexType *ia, + IndexType *ja, + IndexType *perm, + IndexType nrhs, + IndexType *iparm, + IndexType msglvl, + void *b, + void *x) { IndexType error = 0; ::pardiso(pt, &maxfct, &mnum, &type, &phase, &n, a, ia, ja, perm, &nrhs, iparm, &msglvl, b, x, &error); return error; } }; - template<> - struct pardiso_run_selector + template<> struct pardiso_run_selector { typedef long long int IndexType; - static IndexType run( _MKL_DSS_HANDLE_t pt, IndexType maxfct, IndexType mnum, IndexType type, IndexType phase, IndexType n, void *a, - IndexType *ia, IndexType *ja, IndexType *perm, IndexType nrhs, IndexType *iparm, IndexType msglvl, void *b, void *x) + static IndexType run(_MKL_DSS_HANDLE_t pt, + IndexType maxfct, + IndexType mnum, + IndexType type, + IndexType phase, + IndexType n, + void *a, + IndexType *ia, + IndexType *ja, + IndexType *perm, + IndexType nrhs, + IndexType *iparm, + IndexType msglvl, + void *b, + void *x) { IndexType error = 0; ::pardiso_64(pt, &maxfct, &mnum, &type, &phase, &n, a, ia, ja, perm, &nrhs, iparm, &msglvl, b, x, &error); @@ -66,8 +89,7 @@ namespace internal template struct pardiso_traits; - template - struct pardiso_traits< PardisoLU<_MatrixType> > + template struct pardiso_traits> { typedef _MatrixType MatrixType; typedef typename _MatrixType::Scalar Scalar; @@ -75,8 +97,7 @@ namespace internal typedef typename _MatrixType::StorageIndex StorageIndex; }; - template - struct pardiso_traits< PardisoLLT<_MatrixType, Options> > + template struct pardiso_traits> { typedef _MatrixType MatrixType; typedef typename _MatrixType::Scalar Scalar; @@ -84,179 +105,178 @@ namespace internal typedef typename _MatrixType::StorageIndex StorageIndex; }; - template - struct pardiso_traits< PardisoLDLT<_MatrixType, Options> > + template struct pardiso_traits> { typedef _MatrixType MatrixType; typedef typename _MatrixType::Scalar Scalar; typedef typename _MatrixType::RealScalar RealScalar; - typedef typename _MatrixType::StorageIndex StorageIndex; + typedef typename _MatrixType::StorageIndex StorageIndex; }; -} // end namespace internal +}// end namespace internal -template -class PardisoImpl : public SparseSolverBase +template class PardisoImpl : public SparseSolverBase { - protected: - typedef SparseSolverBase Base; - using Base::derived; - using Base::m_isInitialized; - - typedef internal::pardiso_traits Traits; - public: - using Base::_solve_impl; - - typedef typename Traits::MatrixType MatrixType; - typedef typename Traits::Scalar Scalar; - typedef typename Traits::RealScalar RealScalar; - typedef typename Traits::StorageIndex StorageIndex; - typedef SparseMatrix SparseMatrixType; - typedef Matrix VectorType; - typedef Matrix IntRowVectorType; - typedef Matrix IntColVectorType; - typedef Array ParameterType; - enum { - ScalarIsComplex = NumTraits::IsComplex, - ColsAtCompileTime = Dynamic, - MaxColsAtCompileTime = Dynamic - }; - - PardisoImpl() - { - eigen_assert((sizeof(StorageIndex) >= sizeof(_INTEGER_t) && sizeof(StorageIndex) <= 8) && "Non-supported index type"); - m_iparm.setZero(); - m_msglvl = 0; // No output - m_isInitialized = false; - } +protected: + typedef SparseSolverBase Base; + using Base::derived; + using Base::m_isInitialized; + + typedef internal::pardiso_traits Traits; + +public: + using Base::_solve_impl; + + typedef typename Traits::MatrixType MatrixType; + typedef typename Traits::Scalar Scalar; + typedef typename Traits::RealScalar RealScalar; + typedef typename Traits::StorageIndex StorageIndex; + typedef SparseMatrix SparseMatrixType; + typedef Matrix VectorType; + typedef Matrix IntRowVectorType; + typedef Matrix IntColVectorType; + typedef Array ParameterType; + enum { ScalarIsComplex = NumTraits::IsComplex, ColsAtCompileTime = Dynamic, MaxColsAtCompileTime = Dynamic }; + + PardisoImpl() + { + eigen_assert( + (sizeof(StorageIndex) >= sizeof(_INTEGER_t) && sizeof(StorageIndex) <= 8) && "Non-supported index type"); + m_iparm.setZero(); + m_msglvl = 0;// No output + m_isInitialized = false; + } - ~PardisoImpl() - { - pardisoRelease(); - } + ~PardisoImpl() { pardisoRelease(); } - inline Index cols() const { return m_size; } - inline Index rows() const { return m_size; } - - /** \brief Reports whether previous computation was successful. - * - * \returns \c Success if computation was succesful, - * \c NumericalIssue if the matrix appears to be negative. - */ - ComputationInfo info() const - { - eigen_assert(m_isInitialized && "Decomposition is not initialized."); - return m_info; - } + inline Index cols() const { return m_size; } + inline Index rows() const { return m_size; } - /** \warning for advanced usage only. - * \returns a reference to the parameter array controlling PARDISO. - * See the PARDISO manual to know how to use it. */ - ParameterType& pardisoParameterArray() - { - return m_iparm; - } - - /** Performs a symbolic decomposition on the sparcity of \a matrix. - * - * This function is particularly useful when solving for several problems having the same structure. - * - * \sa factorize() - */ - Derived& analyzePattern(const MatrixType& matrix); - - /** Performs a numeric decomposition of \a matrix - * - * The given matrix must has the same sparcity than the matrix on which the symbolic decomposition has been performed. - * - * \sa analyzePattern() - */ - Derived& factorize(const MatrixType& matrix); - - Derived& compute(const MatrixType& matrix); - - template - void _solve_impl(const MatrixBase &b, MatrixBase &dest) const; - - protected: - void pardisoRelease() - { - if(m_isInitialized) // Factorization ran at least once - { - internal::pardiso_run_selector::run(m_pt, 1, 1, m_type, -1, internal::convert_index(m_size),0, 0, 0, m_perm.data(), 0, - m_iparm.data(), m_msglvl, NULL, NULL); - m_isInitialized = false; - } - } + /** \brief Reports whether previous computation was successful. + * + * \returns \c Success if computation was succesful, + * \c NumericalIssue if the matrix appears to be negative. + */ + ComputationInfo info() const + { + eigen_assert(m_isInitialized && "Decomposition is not initialized."); + return m_info; + } - void pardisoInit(int type) + /** \warning for advanced usage only. + * \returns a reference to the parameter array controlling PARDISO. + * See the PARDISO manual to know how to use it. */ + ParameterType &pardisoParameterArray() { return m_iparm; } + + /** Performs a symbolic decomposition on the sparcity of \a matrix. + * + * This function is particularly useful when solving for several problems having the same structure. + * + * \sa factorize() + */ + Derived &analyzePattern(const MatrixType &matrix); + + /** Performs a numeric decomposition of \a matrix + * + * The given matrix must has the same sparcity than the matrix on which the symbolic decomposition has been performed. + * + * \sa analyzePattern() + */ + Derived &factorize(const MatrixType &matrix); + + Derived &compute(const MatrixType &matrix); + + template void _solve_impl(const MatrixBase &b, MatrixBase &dest) const; + +protected: + void pardisoRelease() + { + if (m_isInitialized)// Factorization ran at least once { - m_type = type; - bool symmetric = std::abs(m_type) < 10; - m_iparm[0] = 1; // No solver default - m_iparm[1] = 2; // use Metis for the ordering - m_iparm[2] = 0; // Reserved. Set to zero. (??Numbers of processors, value of OMP_NUM_THREADS??) - m_iparm[3] = 0; // No iterative-direct algorithm - m_iparm[4] = 0; // No user fill-in reducing permutation - m_iparm[5] = 0; // Write solution into x, b is left unchanged - m_iparm[6] = 0; // Not in use - m_iparm[7] = 2; // Max numbers of iterative refinement steps - m_iparm[8] = 0; // Not in use - m_iparm[9] = 13; // Perturb the pivot elements with 1E-13 - m_iparm[10] = symmetric ? 0 : 1; // Use nonsymmetric permutation and scaling MPS - m_iparm[11] = 0; // Not in use - m_iparm[12] = symmetric ? 0 : 1; // Maximum weighted matching algorithm is switched-off (default for symmetric). - // Try m_iparm[12] = 1 in case of inappropriate accuracy - m_iparm[13] = 0; // Output: Number of perturbed pivots - m_iparm[14] = 0; // Not in use - m_iparm[15] = 0; // Not in use - m_iparm[16] = 0; // Not in use - m_iparm[17] = -1; // Output: Number of nonzeros in the factor LU - m_iparm[18] = -1; // Output: Mflops for LU factorization - m_iparm[19] = 0; // Output: Numbers of CG Iterations - - m_iparm[20] = 0; // 1x1 pivoting - m_iparm[26] = 0; // No matrix checker - m_iparm[27] = (sizeof(RealScalar) == 4) ? 1 : 0; - m_iparm[34] = 1; // C indexing - m_iparm[36] = 0; // CSR - m_iparm[59] = 0; // 0 - In-Core ; 1 - Automatic switch between In-Core and Out-of-Core modes ; 2 - Out-of-Core - - memset(m_pt, 0, sizeof(m_pt)); + internal::pardiso_run_selector::run(m_pt, + 1, + 1, + m_type, + -1, + internal::convert_index(m_size), + 0, + 0, + 0, + m_perm.data(), + 0, + m_iparm.data(), + m_msglvl, + NULL, + NULL); + m_isInitialized = false; } + } - protected: - // cached data to reduce reallocation, etc. - - void manageErrorCode(Index error) const - { - switch(error) - { - case 0: - m_info = Success; - break; - case -4: - case -7: - m_info = NumericalIssue; - break; - default: - m_info = InvalidInput; - } + void pardisoInit(int type) + { + m_type = type; + bool symmetric = std::abs(m_type) < 10; + m_iparm[0] = 1;// No solver default + m_iparm[1] = 2;// use Metis for the ordering + m_iparm[2] = 0;// Reserved. Set to zero. (??Numbers of processors, value of OMP_NUM_THREADS??) + m_iparm[3] = 0;// No iterative-direct algorithm + m_iparm[4] = 0;// No user fill-in reducing permutation + m_iparm[5] = 0;// Write solution into x, b is left unchanged + m_iparm[6] = 0;// Not in use + m_iparm[7] = 2;// Max numbers of iterative refinement steps + m_iparm[8] = 0;// Not in use + m_iparm[9] = 13;// Perturb the pivot elements with 1E-13 + m_iparm[10] = symmetric ? 0 : 1;// Use nonsymmetric permutation and scaling MPS + m_iparm[11] = 0;// Not in use + m_iparm[12] = symmetric ? 0 : 1;// Maximum weighted matching algorithm is switched-off (default for symmetric). + // Try m_iparm[12] = 1 in case of inappropriate accuracy + m_iparm[13] = 0;// Output: Number of perturbed pivots + m_iparm[14] = 0;// Not in use + m_iparm[15] = 0;// Not in use + m_iparm[16] = 0;// Not in use + m_iparm[17] = -1;// Output: Number of nonzeros in the factor LU + m_iparm[18] = -1;// Output: Mflops for LU factorization + m_iparm[19] = 0;// Output: Numbers of CG Iterations + + m_iparm[20] = 0;// 1x1 pivoting + m_iparm[26] = 0;// No matrix checker + m_iparm[27] = (sizeof(RealScalar) == 4) ? 1 : 0; + m_iparm[34] = 1;// C indexing + m_iparm[36] = 0;// CSR + m_iparm[59] = 0;// 0 - In-Core ; 1 - Automatic switch between In-Core and Out-of-Core modes ; 2 - Out-of-Core + + memset(m_pt, 0, sizeof(m_pt)); + } + +protected: + // cached data to reduce reallocation, etc. + + void manageErrorCode(Index error) const + { + switch (error) { + case 0: + m_info = Success; + break; + case -4: + case -7: + m_info = NumericalIssue; + break; + default: + m_info = InvalidInput; } + } - mutable SparseMatrixType m_matrix; - mutable ComputationInfo m_info; - bool m_analysisIsOk, m_factorizationIsOk; - StorageIndex m_type, m_msglvl; - mutable void *m_pt[64]; - mutable ParameterType m_iparm; - mutable IntColVectorType m_perm; - Index m_size; - + mutable SparseMatrixType m_matrix; + mutable ComputationInfo m_info; + bool m_analysisIsOk, m_factorizationIsOk; + StorageIndex m_type, m_msglvl; + mutable void *m_pt[64]; + mutable ParameterType m_iparm; + mutable IntColVectorType m_perm; + Index m_size; }; -template -Derived& PardisoImpl::compute(const MatrixType& a) +template Derived &PardisoImpl::compute(const MatrixType &a) { m_size = a.rows(); eigen_assert(a.rows() == a.cols()); @@ -264,11 +284,23 @@ Derived& PardisoImpl::compute(const MatrixType& a) pardisoRelease(); m_perm.setZero(m_size); derived().getMatrix(a); - + Index error; - error = internal::pardiso_run_selector::run(m_pt, 1, 1, m_type, 12, internal::convert_index(m_size), - m_matrix.valuePtr(), m_matrix.outerIndexPtr(), m_matrix.innerIndexPtr(), - m_perm.data(), 0, m_iparm.data(), m_msglvl, NULL, NULL); + error = internal::pardiso_run_selector::run(m_pt, + 1, + 1, + m_type, + 12, + internal::convert_index(m_size), + m_matrix.valuePtr(), + m_matrix.outerIndexPtr(), + m_matrix.innerIndexPtr(), + m_perm.data(), + 0, + m_iparm.data(), + m_msglvl, + NULL, + NULL); manageErrorCode(error); m_analysisIsOk = true; m_factorizationIsOk = true; @@ -276,8 +308,7 @@ Derived& PardisoImpl::compute(const MatrixType& a) return derived(); } -template -Derived& PardisoImpl::analyzePattern(const MatrixType& a) +template Derived &PardisoImpl::analyzePattern(const MatrixType &a) { m_size = a.rows(); eigen_assert(m_size == a.cols()); @@ -285,12 +316,24 @@ Derived& PardisoImpl::analyzePattern(const MatrixType& a) pardisoRelease(); m_perm.setZero(m_size); derived().getMatrix(a); - + Index error; - error = internal::pardiso_run_selector::run(m_pt, 1, 1, m_type, 11, internal::convert_index(m_size), - m_matrix.valuePtr(), m_matrix.outerIndexPtr(), m_matrix.innerIndexPtr(), - m_perm.data(), 0, m_iparm.data(), m_msglvl, NULL, NULL); - + error = internal::pardiso_run_selector::run(m_pt, + 1, + 1, + m_type, + 11, + internal::convert_index(m_size), + m_matrix.valuePtr(), + m_matrix.outerIndexPtr(), + m_matrix.innerIndexPtr(), + m_perm.data(), + 0, + m_iparm.data(), + m_msglvl, + NULL, + NULL); + manageErrorCode(error); m_analysisIsOk = true; m_factorizationIsOk = false; @@ -298,246 +341,249 @@ Derived& PardisoImpl::analyzePattern(const MatrixType& a) return derived(); } -template -Derived& PardisoImpl::factorize(const MatrixType& a) +template Derived &PardisoImpl::factorize(const MatrixType &a) { eigen_assert(m_analysisIsOk && "You must first call analyzePattern()"); eigen_assert(m_size == a.rows() && m_size == a.cols()); - + derived().getMatrix(a); Index error; - error = internal::pardiso_run_selector::run(m_pt, 1, 1, m_type, 22, internal::convert_index(m_size), - m_matrix.valuePtr(), m_matrix.outerIndexPtr(), m_matrix.innerIndexPtr(), - m_perm.data(), 0, m_iparm.data(), m_msglvl, NULL, NULL); - + error = internal::pardiso_run_selector::run(m_pt, + 1, + 1, + m_type, + 22, + internal::convert_index(m_size), + m_matrix.valuePtr(), + m_matrix.outerIndexPtr(), + m_matrix.innerIndexPtr(), + m_perm.data(), + 0, + m_iparm.data(), + m_msglvl, + NULL, + NULL); + manageErrorCode(error); m_factorizationIsOk = true; return derived(); } template -template -void PardisoImpl::_solve_impl(const MatrixBase &b, MatrixBase& x) const +template +void PardisoImpl::_solve_impl(const MatrixBase &b, MatrixBase &x) const { - if(m_iparm[0] == 0) // Factorization was not computed + if (m_iparm[0] == 0)// Factorization was not computed { m_info = InvalidInput; return; } - //Index n = m_matrix.rows(); + // Index n = m_matrix.rows(); Index nrhs = Index(b.cols()); - eigen_assert(m_size==b.rows()); - eigen_assert(((MatrixBase::Flags & RowMajorBit) == 0 || nrhs == 1) && "Row-major right hand sides are not supported"); - eigen_assert(((MatrixBase::Flags & RowMajorBit) == 0 || nrhs == 1) && "Row-major matrices of unknowns are not supported"); + eigen_assert(m_size == b.rows()); + eigen_assert( + ((MatrixBase::Flags & RowMajorBit) == 0 || nrhs == 1) && "Row-major right hand sides are not supported"); + eigen_assert(((MatrixBase::Flags & RowMajorBit) == 0 || nrhs == 1) + && "Row-major matrices of unknowns are not supported"); eigen_assert(((nrhs == 1) || b.outerStride() == b.rows())); -// switch (transposed) { -// case SvNoTrans : m_iparm[11] = 0 ; break; -// case SvTranspose : m_iparm[11] = 2 ; break; -// case SvAdjoint : m_iparm[11] = 1 ; break; -// default: -// //std::cerr << "Eigen: transposition option \"" << transposed << "\" not supported by the PARDISO backend\n"; -// m_iparm[11] = 0; -// } + // switch (transposed) { + // case SvNoTrans : m_iparm[11] = 0 ; break; + // case SvTranspose : m_iparm[11] = 2 ; break; + // case SvAdjoint : m_iparm[11] = 1 ; break; + // default: + // //std::cerr << "Eigen: transposition option \"" << transposed << "\" not supported by the PARDISO backend\n"; + // m_iparm[11] = 0; + // } + + Scalar *rhs_ptr = const_cast(b.derived().data()); + Matrix tmp; - Scalar* rhs_ptr = const_cast(b.derived().data()); - Matrix tmp; - // Pardiso cannot solve in-place - if(rhs_ptr == x.derived().data()) - { + if (rhs_ptr == x.derived().data()) { tmp = b; rhs_ptr = tmp.data(); } - + Index error; - error = internal::pardiso_run_selector::run(m_pt, 1, 1, m_type, 33, internal::convert_index(m_size), - m_matrix.valuePtr(), m_matrix.outerIndexPtr(), m_matrix.innerIndexPtr(), - m_perm.data(), internal::convert_index(nrhs), m_iparm.data(), m_msglvl, - rhs_ptr, x.derived().data()); + error = internal::pardiso_run_selector::run(m_pt, + 1, + 1, + m_type, + 33, + internal::convert_index(m_size), + m_matrix.valuePtr(), + m_matrix.outerIndexPtr(), + m_matrix.innerIndexPtr(), + m_perm.data(), + internal::convert_index(nrhs), + m_iparm.data(), + m_msglvl, + rhs_ptr, + x.derived().data()); manageErrorCode(error); } /** \ingroup PardisoSupport_Module - * \class PardisoLU - * \brief A sparse direct LU factorization and solver based on the PARDISO library - * - * This class allows to solve for A.X = B sparse linear problems via a direct LU factorization - * using the Intel MKL PARDISO library. The sparse matrix A must be squared and invertible. - * The vectors or matrices X and B can be either dense or sparse. - * - * By default, it runs in in-core mode. To enable PARDISO's out-of-core feature, set: - * \code solver.pardisoParameterArray()[59] = 1; \endcode - * - * \tparam _MatrixType the type of the sparse matrix A, it must be a SparseMatrix<> - * - * \implsparsesolverconcept - * - * \sa \ref TutorialSparseSolverConcept, class SparseLU - */ -template -class PardisoLU : public PardisoImpl< PardisoLU > + * \class PardisoLU + * \brief A sparse direct LU factorization and solver based on the PARDISO library + * + * This class allows to solve for A.X = B sparse linear problems via a direct LU factorization + * using the Intel MKL PARDISO library. The sparse matrix A must be squared and invertible. + * The vectors or matrices X and B can be either dense or sparse. + * + * By default, it runs in in-core mode. To enable PARDISO's out-of-core feature, set: + * \code solver.pardisoParameterArray()[59] = 1; \endcode + * + * \tparam _MatrixType the type of the sparse matrix A, it must be a SparseMatrix<> + * + * \implsparsesolverconcept + * + * \sa \ref TutorialSparseSolverConcept, class SparseLU + */ +template class PardisoLU : public PardisoImpl> { - protected: - typedef PardisoImpl Base; - typedef typename Base::Scalar Scalar; - typedef typename Base::RealScalar RealScalar; - using Base::pardisoInit; - using Base::m_matrix; - friend class PardisoImpl< PardisoLU >; +protected: + typedef PardisoImpl Base; + typedef typename Base::Scalar Scalar; + typedef typename Base::RealScalar RealScalar; + using Base::pardisoInit; + using Base::m_matrix; + friend class PardisoImpl>; - public: +public: + using Base::compute; + using Base::solve; - using Base::compute; - using Base::solve; + PardisoLU() : Base() { pardisoInit(Base::ScalarIsComplex ? 13 : 11); } - PardisoLU() - : Base() - { - pardisoInit(Base::ScalarIsComplex ? 13 : 11); - } + explicit PardisoLU(const MatrixType &matrix) : Base() + { + pardisoInit(Base::ScalarIsComplex ? 13 : 11); + compute(matrix); + } - explicit PardisoLU(const MatrixType& matrix) - : Base() - { - pardisoInit(Base::ScalarIsComplex ? 13 : 11); - compute(matrix); - } - protected: - void getMatrix(const MatrixType& matrix) - { - m_matrix = matrix; - m_matrix.makeCompressed(); - } +protected: + void getMatrix(const MatrixType &matrix) + { + m_matrix = matrix; + m_matrix.makeCompressed(); + } }; /** \ingroup PardisoSupport_Module - * \class PardisoLLT - * \brief A sparse direct Cholesky (LLT) factorization and solver based on the PARDISO library - * - * This class allows to solve for A.X = B sparse linear problems via a LL^T Cholesky factorization - * using the Intel MKL PARDISO library. The sparse matrix A must be selfajoint and positive definite. - * The vectors or matrices X and B can be either dense or sparse. - * - * By default, it runs in in-core mode. To enable PARDISO's out-of-core feature, set: - * \code solver.pardisoParameterArray()[59] = 1; \endcode - * - * \tparam MatrixType the type of the sparse matrix A, it must be a SparseMatrix<> - * \tparam UpLo can be any bitwise combination of Upper, Lower. The default is Upper, meaning only the upper triangular part has to be used. - * Upper|Lower can be used to tell both triangular parts can be used as input. - * - * \implsparsesolverconcept - * - * \sa \ref TutorialSparseSolverConcept, class SimplicialLLT - */ -template -class PardisoLLT : public PardisoImpl< PardisoLLT > + * \class PardisoLLT + * \brief A sparse direct Cholesky (LLT) factorization and solver based on the PARDISO library + * + * This class allows to solve for A.X = B sparse linear problems via a LL^T Cholesky factorization + * using the Intel MKL PARDISO library. The sparse matrix A must be selfajoint and positive definite. + * The vectors or matrices X and B can be either dense or sparse. + * + * By default, it runs in in-core mode. To enable PARDISO's out-of-core feature, set: + * \code solver.pardisoParameterArray()[59] = 1; \endcode + * + * \tparam MatrixType the type of the sparse matrix A, it must be a SparseMatrix<> + * \tparam UpLo can be any bitwise combination of Upper, Lower. The default is Upper, meaning only the upper triangular + * part has to be used. Upper|Lower can be used to tell both triangular parts can be used as input. + * + * \implsparsesolverconcept + * + * \sa \ref TutorialSparseSolverConcept, class SimplicialLLT + */ +template class PardisoLLT : public PardisoImpl> { - protected: - typedef PardisoImpl< PardisoLLT > Base; - typedef typename Base::Scalar Scalar; - typedef typename Base::RealScalar RealScalar; - using Base::pardisoInit; - using Base::m_matrix; - friend class PardisoImpl< PardisoLLT >; - - public: - - typedef typename Base::StorageIndex StorageIndex; - enum { UpLo = _UpLo }; - using Base::compute; - - PardisoLLT() - : Base() - { - pardisoInit(Base::ScalarIsComplex ? 4 : 2); - } +protected: + typedef PardisoImpl> Base; + typedef typename Base::Scalar Scalar; + typedef typename Base::RealScalar RealScalar; + using Base::pardisoInit; + using Base::m_matrix; + friend class PardisoImpl>; + +public: + typedef typename Base::StorageIndex StorageIndex; + enum { UpLo = _UpLo }; + using Base::compute; + + PardisoLLT() : Base() { pardisoInit(Base::ScalarIsComplex ? 4 : 2); } + + explicit PardisoLLT(const MatrixType &matrix) : Base() + { + pardisoInit(Base::ScalarIsComplex ? 4 : 2); + compute(matrix); + } - explicit PardisoLLT(const MatrixType& matrix) - : Base() - { - pardisoInit(Base::ScalarIsComplex ? 4 : 2); - compute(matrix); - } - - protected: - - void getMatrix(const MatrixType& matrix) - { - // PARDISO supports only upper, row-major matrices - PermutationMatrix p_null; - m_matrix.resize(matrix.rows(), matrix.cols()); - m_matrix.template selfadjointView() = matrix.template selfadjointView().twistedBy(p_null); - m_matrix.makeCompressed(); - } +protected: + void getMatrix(const MatrixType &matrix) + { + // PARDISO supports only upper, row-major matrices + PermutationMatrix p_null; + m_matrix.resize(matrix.rows(), matrix.cols()); + m_matrix.template selfadjointView() = matrix.template selfadjointView().twistedBy(p_null); + m_matrix.makeCompressed(); + } }; /** \ingroup PardisoSupport_Module - * \class PardisoLDLT - * \brief A sparse direct Cholesky (LDLT) factorization and solver based on the PARDISO library - * - * This class allows to solve for A.X = B sparse linear problems via a LDL^T Cholesky factorization - * using the Intel MKL PARDISO library. The sparse matrix A is assumed to be selfajoint and positive definite. - * For complex matrices, A can also be symmetric only, see the \a Options template parameter. - * The vectors or matrices X and B can be either dense or sparse. - * - * By default, it runs in in-core mode. To enable PARDISO's out-of-core feature, set: - * \code solver.pardisoParameterArray()[59] = 1; \endcode - * - * \tparam MatrixType the type of the sparse matrix A, it must be a SparseMatrix<> - * \tparam Options can be any bitwise combination of Upper, Lower, and Symmetric. The default is Upper, meaning only the upper triangular part has to be used. - * Symmetric can be used for symmetric, non-selfadjoint complex matrices, the default being to assume a selfadjoint matrix. - * Upper|Lower can be used to tell both triangular parts can be used as input. - * - * \implsparsesolverconcept - * - * \sa \ref TutorialSparseSolverConcept, class SimplicialLDLT - */ -template -class PardisoLDLT : public PardisoImpl< PardisoLDLT > + * \class PardisoLDLT + * \brief A sparse direct Cholesky (LDLT) factorization and solver based on the PARDISO library + * + * This class allows to solve for A.X = B sparse linear problems via a LDL^T Cholesky factorization + * using the Intel MKL PARDISO library. The sparse matrix A is assumed to be selfajoint and positive definite. + * For complex matrices, A can also be symmetric only, see the \a Options template parameter. + * The vectors or matrices X and B can be either dense or sparse. + * + * By default, it runs in in-core mode. To enable PARDISO's out-of-core feature, set: + * \code solver.pardisoParameterArray()[59] = 1; \endcode + * + * \tparam MatrixType the type of the sparse matrix A, it must be a SparseMatrix<> + * \tparam Options can be any bitwise combination of Upper, Lower, and Symmetric. The default is Upper, meaning only the + * upper triangular part has to be used. Symmetric can be used for symmetric, non-selfadjoint complex matrices, the + * default being to assume a selfadjoint matrix. Upper|Lower can be used to tell both triangular parts can be used as + * input. + * + * \implsparsesolverconcept + * + * \sa \ref TutorialSparseSolverConcept, class SimplicialLDLT + */ +template class PardisoLDLT : public PardisoImpl> { - protected: - typedef PardisoImpl< PardisoLDLT > Base; - typedef typename Base::Scalar Scalar; - typedef typename Base::RealScalar RealScalar; - using Base::pardisoInit; - using Base::m_matrix; - friend class PardisoImpl< PardisoLDLT >; - - public: - - typedef typename Base::StorageIndex StorageIndex; - using Base::compute; - enum { UpLo = Options&(Upper|Lower) }; - - PardisoLDLT() - : Base() - { - pardisoInit(Base::ScalarIsComplex ? ( bool(Options&Symmetric) ? 6 : -4 ) : -2); - } +protected: + typedef PardisoImpl> Base; + typedef typename Base::Scalar Scalar; + typedef typename Base::RealScalar RealScalar; + using Base::pardisoInit; + using Base::m_matrix; + friend class PardisoImpl>; + +public: + typedef typename Base::StorageIndex StorageIndex; + using Base::compute; + enum { UpLo = Options & (Upper | Lower) }; + + PardisoLDLT() : Base() { pardisoInit(Base::ScalarIsComplex ? (bool(Options & Symmetric) ? 6 : -4) : -2); } + + explicit PardisoLDLT(const MatrixType &matrix) : Base() + { + pardisoInit(Base::ScalarIsComplex ? (bool(Options & Symmetric) ? 6 : -4) : -2); + compute(matrix); + } - explicit PardisoLDLT(const MatrixType& matrix) - : Base() - { - pardisoInit(Base::ScalarIsComplex ? ( bool(Options&Symmetric) ? 6 : -4 ) : -2); - compute(matrix); - } - - void getMatrix(const MatrixType& matrix) - { - // PARDISO supports only upper, row-major matrices - PermutationMatrix p_null; - m_matrix.resize(matrix.rows(), matrix.cols()); - m_matrix.template selfadjointView() = matrix.template selfadjointView().twistedBy(p_null); - m_matrix.makeCompressed(); - } + void getMatrix(const MatrixType &matrix) + { + // PARDISO supports only upper, row-major matrices + PermutationMatrix p_null; + m_matrix.resize(matrix.rows(), matrix.cols()); + m_matrix.template selfadjointView() = matrix.template selfadjointView().twistedBy(p_null); + m_matrix.makeCompressed(); + } }; -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_PARDISOSUPPORT_H +#endif// EIGEN_PARDISOSUPPORT_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/QR/ColPivHouseholderQR.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/QR/ColPivHouseholderQR.h index a7b47d55..4bb483c2 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/QR/ColPivHouseholderQR.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/QR/ColPivHouseholderQR.h @@ -14,438 +14,403 @@ namespace Eigen { namespace internal { -template struct traits > - : traits<_MatrixType> -{ - enum { Flags = 0 }; -}; + template struct traits> : traits<_MatrixType> + { + enum { Flags = 0 }; + }; -} // end namespace internal +}// end namespace internal /** \ingroup QR_Module - * - * \class ColPivHouseholderQR - * - * \brief Householder rank-revealing QR decomposition of a matrix with column-pivoting - * - * \tparam _MatrixType the type of the matrix of which we are computing the QR decomposition - * - * This class performs a rank-revealing QR decomposition of a matrix \b A into matrices \b P, \b Q and \b R - * such that - * \f[ - * \mathbf{A} \, \mathbf{P} = \mathbf{Q} \, \mathbf{R} - * \f] - * by using Householder transformations. Here, \b P is a permutation matrix, \b Q a unitary matrix and \b R an - * upper triangular matrix. - * - * This decomposition performs column pivoting in order to be rank-revealing and improve - * numerical stability. It is slower than HouseholderQR, and faster than FullPivHouseholderQR. - * - * This class supports the \link InplaceDecomposition inplace decomposition \endlink mechanism. - * - * \sa MatrixBase::colPivHouseholderQr() - */ + * + * \class ColPivHouseholderQR + * + * \brief Householder rank-revealing QR decomposition of a matrix with column-pivoting + * + * \tparam _MatrixType the type of the matrix of which we are computing the QR decomposition + * + * This class performs a rank-revealing QR decomposition of a matrix \b A into matrices \b P, \b Q and \b R + * such that + * \f[ + * \mathbf{A} \, \mathbf{P} = \mathbf{Q} \, \mathbf{R} + * \f] + * by using Householder transformations. Here, \b P is a permutation matrix, \b Q a unitary matrix and \b R an + * upper triangular matrix. + * + * This decomposition performs column pivoting in order to be rank-revealing and improve + * numerical stability. It is slower than HouseholderQR, and faster than FullPivHouseholderQR. + * + * This class supports the \link InplaceDecomposition inplace decomposition \endlink mechanism. + * + * \sa MatrixBase::colPivHouseholderQr() + */ template class ColPivHouseholderQR { - public: - - typedef _MatrixType MatrixType; - enum { - RowsAtCompileTime = MatrixType::RowsAtCompileTime, - ColsAtCompileTime = MatrixType::ColsAtCompileTime, - MaxRowsAtCompileTime = MatrixType::MaxRowsAtCompileTime, - MaxColsAtCompileTime = MatrixType::MaxColsAtCompileTime - }; - typedef typename MatrixType::Scalar Scalar; - typedef typename MatrixType::RealScalar RealScalar; - // FIXME should be int - typedef typename MatrixType::StorageIndex StorageIndex; - typedef typename internal::plain_diag_type::type HCoeffsType; - typedef PermutationMatrix PermutationType; - typedef typename internal::plain_row_type::type IntRowVectorType; - typedef typename internal::plain_row_type::type RowVectorType; - typedef typename internal::plain_row_type::type RealRowVectorType; - typedef HouseholderSequence::type> HouseholderSequenceType; - typedef typename MatrixType::PlainObject PlainObject; - - private: - - typedef typename PermutationType::StorageIndex PermIndexType; - - public: - - /** - * \brief Default Constructor. - * - * The default constructor is useful in cases in which the user intends to - * perform decompositions via ColPivHouseholderQR::compute(const MatrixType&). - */ - ColPivHouseholderQR() - : m_qr(), - m_hCoeffs(), - m_colsPermutation(), - m_colsTranspositions(), - m_temp(), - m_colNormsUpdated(), - m_colNormsDirect(), - m_isInitialized(false), - m_usePrescribedThreshold(false) {} - - /** \brief Default Constructor with memory preallocation - * - * Like the default constructor but with preallocation of the internal data - * according to the specified problem \a size. - * \sa ColPivHouseholderQR() - */ - ColPivHouseholderQR(Index rows, Index cols) - : m_qr(rows, cols), - m_hCoeffs((std::min)(rows,cols)), - m_colsPermutation(PermIndexType(cols)), - m_colsTranspositions(cols), - m_temp(cols), - m_colNormsUpdated(cols), - m_colNormsDirect(cols), - m_isInitialized(false), - m_usePrescribedThreshold(false) {} - - /** \brief Constructs a QR factorization from a given matrix - * - * This constructor computes the QR factorization of the matrix \a matrix by calling - * the method compute(). It is a short cut for: - * - * \code - * ColPivHouseholderQR qr(matrix.rows(), matrix.cols()); - * qr.compute(matrix); - * \endcode - * - * \sa compute() - */ - template - explicit ColPivHouseholderQR(const EigenBase& matrix) - : m_qr(matrix.rows(), matrix.cols()), - m_hCoeffs((std::min)(matrix.rows(),matrix.cols())), - m_colsPermutation(PermIndexType(matrix.cols())), - m_colsTranspositions(matrix.cols()), - m_temp(matrix.cols()), - m_colNormsUpdated(matrix.cols()), - m_colNormsDirect(matrix.cols()), - m_isInitialized(false), - m_usePrescribedThreshold(false) - { - compute(matrix.derived()); - } +public: + typedef _MatrixType MatrixType; + enum { + RowsAtCompileTime = MatrixType::RowsAtCompileTime, + ColsAtCompileTime = MatrixType::ColsAtCompileTime, + MaxRowsAtCompileTime = MatrixType::MaxRowsAtCompileTime, + MaxColsAtCompileTime = MatrixType::MaxColsAtCompileTime + }; + typedef typename MatrixType::Scalar Scalar; + typedef typename MatrixType::RealScalar RealScalar; + // FIXME should be int + typedef typename MatrixType::StorageIndex StorageIndex; + typedef typename internal::plain_diag_type::type HCoeffsType; + typedef PermutationMatrix PermutationType; + typedef typename internal::plain_row_type::type IntRowVectorType; + typedef typename internal::plain_row_type::type RowVectorType; + typedef typename internal::plain_row_type::type RealRowVectorType; + typedef HouseholderSequence::type> + HouseholderSequenceType; + typedef typename MatrixType::PlainObject PlainObject; + +private: + typedef typename PermutationType::StorageIndex PermIndexType; + +public: + /** + * \brief Default Constructor. + * + * The default constructor is useful in cases in which the user intends to + * perform decompositions via ColPivHouseholderQR::compute(const MatrixType&). + */ + ColPivHouseholderQR() + : m_qr(), m_hCoeffs(), m_colsPermutation(), m_colsTranspositions(), m_temp(), m_colNormsUpdated(), + m_colNormsDirect(), m_isInitialized(false), m_usePrescribedThreshold(false) + {} + + /** \brief Default Constructor with memory preallocation + * + * Like the default constructor but with preallocation of the internal data + * according to the specified problem \a size. + * \sa ColPivHouseholderQR() + */ + ColPivHouseholderQR(Index rows, Index cols) + : m_qr(rows, cols), m_hCoeffs((std::min)(rows, cols)), m_colsPermutation(PermIndexType(cols)), + m_colsTranspositions(cols), m_temp(cols), m_colNormsUpdated(cols), m_colNormsDirect(cols), m_isInitialized(false), + m_usePrescribedThreshold(false) + {} + + /** \brief Constructs a QR factorization from a given matrix + * + * This constructor computes the QR factorization of the matrix \a matrix by calling + * the method compute(). It is a short cut for: + * + * \code + * ColPivHouseholderQR qr(matrix.rows(), matrix.cols()); + * qr.compute(matrix); + * \endcode + * + * \sa compute() + */ + template + explicit ColPivHouseholderQR(const EigenBase &matrix) + : m_qr(matrix.rows(), matrix.cols()), m_hCoeffs((std::min)(matrix.rows(), matrix.cols())), + m_colsPermutation(PermIndexType(matrix.cols())), m_colsTranspositions(matrix.cols()), m_temp(matrix.cols()), + m_colNormsUpdated(matrix.cols()), m_colNormsDirect(matrix.cols()), m_isInitialized(false), + m_usePrescribedThreshold(false) + { + compute(matrix.derived()); + } - /** \brief Constructs a QR factorization from a given matrix - * - * This overloaded constructor is provided for \link InplaceDecomposition inplace decomposition \endlink when \c MatrixType is a Eigen::Ref. - * - * \sa ColPivHouseholderQR(const EigenBase&) - */ - template - explicit ColPivHouseholderQR(EigenBase& matrix) - : m_qr(matrix.derived()), - m_hCoeffs((std::min)(matrix.rows(),matrix.cols())), - m_colsPermutation(PermIndexType(matrix.cols())), - m_colsTranspositions(matrix.cols()), - m_temp(matrix.cols()), - m_colNormsUpdated(matrix.cols()), - m_colNormsDirect(matrix.cols()), - m_isInitialized(false), - m_usePrescribedThreshold(false) - { - computeInPlace(); - } + /** \brief Constructs a QR factorization from a given matrix + * + * This overloaded constructor is provided for \link InplaceDecomposition inplace decomposition \endlink when \c + * MatrixType is a Eigen::Ref. + * + * \sa ColPivHouseholderQR(const EigenBase&) + */ + template + explicit ColPivHouseholderQR(EigenBase &matrix) + : m_qr(matrix.derived()), m_hCoeffs((std::min)(matrix.rows(), matrix.cols())), + m_colsPermutation(PermIndexType(matrix.cols())), m_colsTranspositions(matrix.cols()), m_temp(matrix.cols()), + m_colNormsUpdated(matrix.cols()), m_colNormsDirect(matrix.cols()), m_isInitialized(false), + m_usePrescribedThreshold(false) + { + computeInPlace(); + } - /** This method finds a solution x to the equation Ax=b, where A is the matrix of which - * *this is the QR decomposition, if any exists. - * - * \param b the right-hand-side of the equation to solve. - * - * \returns a solution. - * - * \note_about_checking_solutions - * - * \note_about_arbitrary_choice_of_solution - * - * Example: \include ColPivHouseholderQR_solve.cpp - * Output: \verbinclude ColPivHouseholderQR_solve.out - */ - template - inline const Solve - solve(const MatrixBase& b) const - { - eigen_assert(m_isInitialized && "ColPivHouseholderQR is not initialized."); - return Solve(*this, b.derived()); - } + /** This method finds a solution x to the equation Ax=b, where A is the matrix of which + * *this is the QR decomposition, if any exists. + * + * \param b the right-hand-side of the equation to solve. + * + * \returns a solution. + * + * \note_about_checking_solutions + * + * \note_about_arbitrary_choice_of_solution + * + * Example: \include ColPivHouseholderQR_solve.cpp + * Output: \verbinclude ColPivHouseholderQR_solve.out + */ + template inline const Solve solve(const MatrixBase &b) const + { + eigen_assert(m_isInitialized && "ColPivHouseholderQR is not initialized."); + return Solve(*this, b.derived()); + } - HouseholderSequenceType householderQ() const; - HouseholderSequenceType matrixQ() const - { - return householderQ(); - } + HouseholderSequenceType householderQ() const; + HouseholderSequenceType matrixQ() const { return householderQ(); } - /** \returns a reference to the matrix where the Householder QR decomposition is stored - */ - const MatrixType& matrixQR() const - { - eigen_assert(m_isInitialized && "ColPivHouseholderQR is not initialized."); - return m_qr; - } + /** \returns a reference to the matrix where the Householder QR decomposition is stored + */ + const MatrixType &matrixQR() const + { + eigen_assert(m_isInitialized && "ColPivHouseholderQR is not initialized."); + return m_qr; + } - /** \returns a reference to the matrix where the result Householder QR is stored - * \warning The strict lower part of this matrix contains internal values. - * Only the upper triangular part should be referenced. To get it, use - * \code matrixR().template triangularView() \endcode - * For rank-deficient matrices, use - * \code - * matrixR().topLeftCorner(rank(), rank()).template triangularView() - * \endcode - */ - const MatrixType& matrixR() const - { - eigen_assert(m_isInitialized && "ColPivHouseholderQR is not initialized."); - return m_qr; - } + /** \returns a reference to the matrix where the result Householder QR is stored + * \warning The strict lower part of this matrix contains internal values. + * Only the upper triangular part should be referenced. To get it, use + * \code matrixR().template triangularView() \endcode + * For rank-deficient matrices, use + * \code + * matrixR().topLeftCorner(rank(), rank()).template triangularView() + * \endcode + */ + const MatrixType &matrixR() const + { + eigen_assert(m_isInitialized && "ColPivHouseholderQR is not initialized."); + return m_qr; + } - template - ColPivHouseholderQR& compute(const EigenBase& matrix); + template ColPivHouseholderQR &compute(const EigenBase &matrix); - /** \returns a const reference to the column permutation matrix */ - const PermutationType& colsPermutation() const - { - eigen_assert(m_isInitialized && "ColPivHouseholderQR is not initialized."); - return m_colsPermutation; - } + /** \returns a const reference to the column permutation matrix */ + const PermutationType &colsPermutation() const + { + eigen_assert(m_isInitialized && "ColPivHouseholderQR is not initialized."); + return m_colsPermutation; + } - /** \returns the absolute value of the determinant of the matrix of which - * *this is the QR decomposition. It has only linear complexity - * (that is, O(n) where n is the dimension of the square matrix) - * as the QR decomposition has already been computed. - * - * \note This is only for square matrices. - * - * \warning a determinant can be very big or small, so for matrices - * of large enough dimension, there is a risk of overflow/underflow. - * One way to work around that is to use logAbsDeterminant() instead. - * - * \sa logAbsDeterminant(), MatrixBase::determinant() - */ - typename MatrixType::RealScalar absDeterminant() const; - - /** \returns the natural log of the absolute value of the determinant of the matrix of which - * *this is the QR decomposition. It has only linear complexity - * (that is, O(n) where n is the dimension of the square matrix) - * as the QR decomposition has already been computed. - * - * \note This is only for square matrices. - * - * \note This method is useful to work around the risk of overflow/underflow that's inherent - * to determinant computation. - * - * \sa absDeterminant(), MatrixBase::determinant() - */ - typename MatrixType::RealScalar logAbsDeterminant() const; - - /** \returns the rank of the matrix of which *this is the QR decomposition. - * - * \note This method has to determine which pivots should be considered nonzero. - * For that, it uses the threshold value that you can control by calling - * setThreshold(const RealScalar&). - */ - inline Index rank() const - { - using std::abs; - eigen_assert(m_isInitialized && "ColPivHouseholderQR is not initialized."); - RealScalar premultiplied_threshold = abs(m_maxpivot) * threshold(); - Index result = 0; - for(Index i = 0; i < m_nonzero_pivots; ++i) - result += (abs(m_qr.coeff(i,i)) > premultiplied_threshold); - return result; - } + /** \returns the absolute value of the determinant of the matrix of which + * *this is the QR decomposition. It has only linear complexity + * (that is, O(n) where n is the dimension of the square matrix) + * as the QR decomposition has already been computed. + * + * \note This is only for square matrices. + * + * \warning a determinant can be very big or small, so for matrices + * of large enough dimension, there is a risk of overflow/underflow. + * One way to work around that is to use logAbsDeterminant() instead. + * + * \sa logAbsDeterminant(), MatrixBase::determinant() + */ + typename MatrixType::RealScalar absDeterminant() const; + + /** \returns the natural log of the absolute value of the determinant of the matrix of which + * *this is the QR decomposition. It has only linear complexity + * (that is, O(n) where n is the dimension of the square matrix) + * as the QR decomposition has already been computed. + * + * \note This is only for square matrices. + * + * \note This method is useful to work around the risk of overflow/underflow that's inherent + * to determinant computation. + * + * \sa absDeterminant(), MatrixBase::determinant() + */ + typename MatrixType::RealScalar logAbsDeterminant() const; + + /** \returns the rank of the matrix of which *this is the QR decomposition. + * + * \note This method has to determine which pivots should be considered nonzero. + * For that, it uses the threshold value that you can control by calling + * setThreshold(const RealScalar&). + */ + inline Index rank() const + { + using std::abs; + eigen_assert(m_isInitialized && "ColPivHouseholderQR is not initialized."); + RealScalar premultiplied_threshold = abs(m_maxpivot) * threshold(); + Index result = 0; + for (Index i = 0; i < m_nonzero_pivots; ++i) result += (abs(m_qr.coeff(i, i)) > premultiplied_threshold); + return result; + } - /** \returns the dimension of the kernel of the matrix of which *this is the QR decomposition. - * - * \note This method has to determine which pivots should be considered nonzero. - * For that, it uses the threshold value that you can control by calling - * setThreshold(const RealScalar&). - */ - inline Index dimensionOfKernel() const - { - eigen_assert(m_isInitialized && "ColPivHouseholderQR is not initialized."); - return cols() - rank(); - } + /** \returns the dimension of the kernel of the matrix of which *this is the QR decomposition. + * + * \note This method has to determine which pivots should be considered nonzero. + * For that, it uses the threshold value that you can control by calling + * setThreshold(const RealScalar&). + */ + inline Index dimensionOfKernel() const + { + eigen_assert(m_isInitialized && "ColPivHouseholderQR is not initialized."); + return cols() - rank(); + } - /** \returns true if the matrix of which *this is the QR decomposition represents an injective - * linear map, i.e. has trivial kernel; false otherwise. - * - * \note This method has to determine which pivots should be considered nonzero. - * For that, it uses the threshold value that you can control by calling - * setThreshold(const RealScalar&). - */ - inline bool isInjective() const - { - eigen_assert(m_isInitialized && "ColPivHouseholderQR is not initialized."); - return rank() == cols(); - } + /** \returns true if the matrix of which *this is the QR decomposition represents an injective + * linear map, i.e. has trivial kernel; false otherwise. + * + * \note This method has to determine which pivots should be considered nonzero. + * For that, it uses the threshold value that you can control by calling + * setThreshold(const RealScalar&). + */ + inline bool isInjective() const + { + eigen_assert(m_isInitialized && "ColPivHouseholderQR is not initialized."); + return rank() == cols(); + } - /** \returns true if the matrix of which *this is the QR decomposition represents a surjective - * linear map; false otherwise. - * - * \note This method has to determine which pivots should be considered nonzero. - * For that, it uses the threshold value that you can control by calling - * setThreshold(const RealScalar&). - */ - inline bool isSurjective() const - { - eigen_assert(m_isInitialized && "ColPivHouseholderQR is not initialized."); - return rank() == rows(); - } + /** \returns true if the matrix of which *this is the QR decomposition represents a surjective + * linear map; false otherwise. + * + * \note This method has to determine which pivots should be considered nonzero. + * For that, it uses the threshold value that you can control by calling + * setThreshold(const RealScalar&). + */ + inline bool isSurjective() const + { + eigen_assert(m_isInitialized && "ColPivHouseholderQR is not initialized."); + return rank() == rows(); + } - /** \returns true if the matrix of which *this is the QR decomposition is invertible. - * - * \note This method has to determine which pivots should be considered nonzero. - * For that, it uses the threshold value that you can control by calling - * setThreshold(const RealScalar&). - */ - inline bool isInvertible() const - { - eigen_assert(m_isInitialized && "ColPivHouseholderQR is not initialized."); - return isInjective() && isSurjective(); - } + /** \returns true if the matrix of which *this is the QR decomposition is invertible. + * + * \note This method has to determine which pivots should be considered nonzero. + * For that, it uses the threshold value that you can control by calling + * setThreshold(const RealScalar&). + */ + inline bool isInvertible() const + { + eigen_assert(m_isInitialized && "ColPivHouseholderQR is not initialized."); + return isInjective() && isSurjective(); + } - /** \returns the inverse of the matrix of which *this is the QR decomposition. - * - * \note If this matrix is not invertible, the returned matrix has undefined coefficients. - * Use isInvertible() to first determine whether this matrix is invertible. - */ - inline const Inverse inverse() const - { - eigen_assert(m_isInitialized && "ColPivHouseholderQR is not initialized."); - return Inverse(*this); - } + /** \returns the inverse of the matrix of which *this is the QR decomposition. + * + * \note If this matrix is not invertible, the returned matrix has undefined coefficients. + * Use isInvertible() to first determine whether this matrix is invertible. + */ + inline const Inverse inverse() const + { + eigen_assert(m_isInitialized && "ColPivHouseholderQR is not initialized."); + return Inverse(*this); + } - inline Index rows() const { return m_qr.rows(); } - inline Index cols() const { return m_qr.cols(); } - - /** \returns a const reference to the vector of Householder coefficients used to represent the factor \c Q. - * - * For advanced uses only. - */ - const HCoeffsType& hCoeffs() const { return m_hCoeffs; } - - /** Allows to prescribe a threshold to be used by certain methods, such as rank(), - * who need to determine when pivots are to be considered nonzero. This is not used for the - * QR decomposition itself. - * - * When it needs to get the threshold value, Eigen calls threshold(). By default, this - * uses a formula to automatically determine a reasonable threshold. - * Once you have called the present method setThreshold(const RealScalar&), - * your value is used instead. - * - * \param threshold The new value to use as the threshold. - * - * A pivot will be considered nonzero if its absolute value is strictly greater than - * \f$ \vert pivot \vert \leqslant threshold \times \vert maxpivot \vert \f$ - * where maxpivot is the biggest pivot. - * - * If you want to come back to the default behavior, call setThreshold(Default_t) - */ - ColPivHouseholderQR& setThreshold(const RealScalar& threshold) - { - m_usePrescribedThreshold = true; - m_prescribedThreshold = threshold; - return *this; - } + inline Index rows() const { return m_qr.rows(); } + inline Index cols() const { return m_qr.cols(); } + + /** \returns a const reference to the vector of Householder coefficients used to represent the factor \c Q. + * + * For advanced uses only. + */ + const HCoeffsType &hCoeffs() const { return m_hCoeffs; } + + /** Allows to prescribe a threshold to be used by certain methods, such as rank(), + * who need to determine when pivots are to be considered nonzero. This is not used for the + * QR decomposition itself. + * + * When it needs to get the threshold value, Eigen calls threshold(). By default, this + * uses a formula to automatically determine a reasonable threshold. + * Once you have called the present method setThreshold(const RealScalar&), + * your value is used instead. + * + * \param threshold The new value to use as the threshold. + * + * A pivot will be considered nonzero if its absolute value is strictly greater than + * \f$ \vert pivot \vert \leqslant threshold \times \vert maxpivot \vert \f$ + * where maxpivot is the biggest pivot. + * + * If you want to come back to the default behavior, call setThreshold(Default_t) + */ + ColPivHouseholderQR &setThreshold(const RealScalar &threshold) + { + m_usePrescribedThreshold = true; + m_prescribedThreshold = threshold; + return *this; + } - /** Allows to come back to the default behavior, letting Eigen use its default formula for - * determining the threshold. - * - * You should pass the special object Eigen::Default as parameter here. - * \code qr.setThreshold(Eigen::Default); \endcode - * - * See the documentation of setThreshold(const RealScalar&). - */ - ColPivHouseholderQR& setThreshold(Default_t) - { - m_usePrescribedThreshold = false; - return *this; - } + /** Allows to come back to the default behavior, letting Eigen use its default formula for + * determining the threshold. + * + * You should pass the special object Eigen::Default as parameter here. + * \code qr.setThreshold(Eigen::Default); \endcode + * + * See the documentation of setThreshold(const RealScalar&). + */ + ColPivHouseholderQR &setThreshold(Default_t) + { + m_usePrescribedThreshold = false; + return *this; + } - /** Returns the threshold that will be used by certain methods such as rank(). - * - * See the documentation of setThreshold(const RealScalar&). - */ - RealScalar threshold() const - { - eigen_assert(m_isInitialized || m_usePrescribedThreshold); - return m_usePrescribedThreshold ? m_prescribedThreshold - // this formula comes from experimenting (see "LU precision tuning" thread on the list) - // and turns out to be identical to Higham's formula used already in LDLt. - : NumTraits::epsilon() * RealScalar(m_qr.diagonalSize()); - } + /** Returns the threshold that will be used by certain methods such as rank(). + * + * See the documentation of setThreshold(const RealScalar&). + */ + RealScalar threshold() const + { + eigen_assert(m_isInitialized || m_usePrescribedThreshold); + return m_usePrescribedThreshold ? m_prescribedThreshold + // this formula comes from experimenting (see "LU precision tuning" thread on the + // list) and turns out to be identical to Higham's formula used already in LDLt. + : NumTraits::epsilon() * RealScalar(m_qr.diagonalSize()); + } - /** \returns the number of nonzero pivots in the QR decomposition. - * Here nonzero is meant in the exact sense, not in a fuzzy sense. - * So that notion isn't really intrinsically interesting, but it is - * still useful when implementing algorithms. - * - * \sa rank() - */ - inline Index nonzeroPivots() const - { - eigen_assert(m_isInitialized && "ColPivHouseholderQR is not initialized."); - return m_nonzero_pivots; - } + /** \returns the number of nonzero pivots in the QR decomposition. + * Here nonzero is meant in the exact sense, not in a fuzzy sense. + * So that notion isn't really intrinsically interesting, but it is + * still useful when implementing algorithms. + * + * \sa rank() + */ + inline Index nonzeroPivots() const + { + eigen_assert(m_isInitialized && "ColPivHouseholderQR is not initialized."); + return m_nonzero_pivots; + } - /** \returns the absolute value of the biggest pivot, i.e. the biggest - * diagonal coefficient of R. - */ - RealScalar maxPivot() const { return m_maxpivot; } - - /** \brief Reports whether the QR factorization was succesful. - * - * \note This function always returns \c Success. It is provided for compatibility - * with other factorization routines. - * \returns \c Success - */ - ComputationInfo info() const - { - eigen_assert(m_isInitialized && "Decomposition is not initialized."); - return Success; - } + /** \returns the absolute value of the biggest pivot, i.e. the biggest + * diagonal coefficient of R. + */ + RealScalar maxPivot() const { return m_maxpivot; } + + /** \brief Reports whether the QR factorization was succesful. + * + * \note This function always returns \c Success. It is provided for compatibility + * with other factorization routines. + * \returns \c Success + */ + ComputationInfo info() const + { + eigen_assert(m_isInitialized && "Decomposition is not initialized."); + return Success; + } - #ifndef EIGEN_PARSED_BY_DOXYGEN - template - EIGEN_DEVICE_FUNC - void _solve_impl(const RhsType &rhs, DstType &dst) const; - #endif +#ifndef EIGEN_PARSED_BY_DOXYGEN + template + EIGEN_DEVICE_FUNC void _solve_impl(const RhsType &rhs, DstType &dst) const; +#endif - protected: +protected: + friend class CompleteOrthogonalDecomposition; - friend class CompleteOrthogonalDecomposition; + static void check_template_parameters() { EIGEN_STATIC_ASSERT_NON_INTEGER(Scalar); } - static void check_template_parameters() - { - EIGEN_STATIC_ASSERT_NON_INTEGER(Scalar); - } + void computeInPlace(); - void computeInPlace(); - - MatrixType m_qr; - HCoeffsType m_hCoeffs; - PermutationType m_colsPermutation; - IntRowVectorType m_colsTranspositions; - RowVectorType m_temp; - RealRowVectorType m_colNormsUpdated; - RealRowVectorType m_colNormsDirect; - bool m_isInitialized, m_usePrescribedThreshold; - RealScalar m_prescribedThreshold, m_maxpivot; - Index m_nonzero_pivots; - Index m_det_pq; + MatrixType m_qr; + HCoeffsType m_hCoeffs; + PermutationType m_colsPermutation; + IntRowVectorType m_colsTranspositions; + RowVectorType m_temp; + RealRowVectorType m_colNormsUpdated; + RealRowVectorType m_colNormsDirect; + bool m_isInitialized, m_usePrescribedThreshold; + RealScalar m_prescribedThreshold, m_maxpivot; + Index m_nonzero_pivots; + Index m_det_pq; }; -template -typename MatrixType::RealScalar ColPivHouseholderQR::absDeterminant() const +template typename MatrixType::RealScalar ColPivHouseholderQR::absDeterminant() const { using std::abs; eigen_assert(m_isInitialized && "ColPivHouseholderQR is not initialized."); @@ -453,8 +418,7 @@ typename MatrixType::RealScalar ColPivHouseholderQR::absDeterminant( return abs(m_qr.diagonal().prod()); } -template -typename MatrixType::RealScalar ColPivHouseholderQR::logAbsDeterminant() const +template typename MatrixType::RealScalar ColPivHouseholderQR::logAbsDeterminant() const { eigen_assert(m_isInitialized && "ColPivHouseholderQR is not initialized."); eigen_assert(m_qr.rows() == m_qr.cols() && "You can't take the determinant of a non-square matrix!"); @@ -462,27 +426,26 @@ typename MatrixType::RealScalar ColPivHouseholderQR::logAbsDetermina } /** Performs the QR factorization of the given matrix \a matrix. The result of - * the factorization is stored into \c *this, and a reference to \c *this - * is returned. - * - * \sa class ColPivHouseholderQR, ColPivHouseholderQR(const MatrixType&) - */ + * the factorization is stored into \c *this, and a reference to \c *this + * is returned. + * + * \sa class ColPivHouseholderQR, ColPivHouseholderQR(const MatrixType&) + */ template template -ColPivHouseholderQR& ColPivHouseholderQR::compute(const EigenBase& matrix) +ColPivHouseholderQR &ColPivHouseholderQR::compute(const EigenBase &matrix) { m_qr = matrix.derived(); computeInPlace(); return *this; } -template -void ColPivHouseholderQR::computeInPlace() +template void ColPivHouseholderQR::computeInPlace() { check_template_parameters(); // the column permutation is stored as int indices, so just to be sure: - eigen_assert(m_qr.cols()<=NumTraits::highest()); + eigen_assert(m_qr.cols() <= NumTraits::highest()); using std::abs; @@ -506,27 +469,26 @@ void ColPivHouseholderQR::computeInPlace() m_colNormsUpdated.coeffRef(k) = m_colNormsDirect.coeffRef(k); } - RealScalar threshold_helper = numext::abs2(m_colNormsUpdated.maxCoeff() * NumTraits::epsilon()) / RealScalar(rows); + RealScalar threshold_helper = + numext::abs2(m_colNormsUpdated.maxCoeff() * NumTraits::epsilon()) / RealScalar(rows); RealScalar norm_downdate_threshold = numext::sqrt(NumTraits::epsilon()); - m_nonzero_pivots = size; // the generic case is that in which all pivots are nonzero (invertible case) + m_nonzero_pivots = size;// the generic case is that in which all pivots are nonzero (invertible case) m_maxpivot = RealScalar(0); - for(Index k = 0; k < size; ++k) - { + for (Index k = 0; k < size; ++k) { // first, we look up in our table m_colNormsUpdated which column has the biggest norm Index biggest_col_index; - RealScalar biggest_col_sq_norm = numext::abs2(m_colNormsUpdated.tail(cols-k).maxCoeff(&biggest_col_index)); + RealScalar biggest_col_sq_norm = numext::abs2(m_colNormsUpdated.tail(cols - k).maxCoeff(&biggest_col_index)); biggest_col_index += k; // Track the number of meaningful pivots but do not stop the decomposition to make // sure that the initial matrix is properly reproduced. See bug 941. - if(m_nonzero_pivots==size && biggest_col_sq_norm < threshold_helper * RealScalar(rows-k)) - m_nonzero_pivots = k; + if (m_nonzero_pivots == size && biggest_col_sq_norm < threshold_helper * RealScalar(rows - k)) m_nonzero_pivots = k; // apply the transposition to the columns m_colsTranspositions.coeffRef(k) = biggest_col_index; - if(k != biggest_col_index) { + if (k != biggest_col_index) { m_qr.col(k).swap(m_qr.col(biggest_col_index)); std::swap(m_colNormsUpdated.coeffRef(k), m_colNormsUpdated.coeffRef(biggest_col_index)); std::swap(m_colNormsDirect.coeffRef(k), m_colNormsDirect.coeffRef(biggest_col_index)); @@ -535,17 +497,17 @@ void ColPivHouseholderQR::computeInPlace() // generate the householder vector, store it below the diagonal RealScalar beta; - m_qr.col(k).tail(rows-k).makeHouseholderInPlace(m_hCoeffs.coeffRef(k), beta); + m_qr.col(k).tail(rows - k).makeHouseholderInPlace(m_hCoeffs.coeffRef(k), beta); // apply the householder transformation to the diagonal coefficient - m_qr.coeffRef(k,k) = beta; + m_qr.coeffRef(k, k) = beta; // remember the maximum absolute value of diagonal coefficients - if(abs(beta) > m_maxpivot) m_maxpivot = abs(beta); + if (abs(beta) > m_maxpivot) m_maxpivot = abs(beta); // apply the householder transformation - m_qr.bottomRightCorner(rows-k, cols-k-1) - .applyHouseholderOnTheLeft(m_qr.col(k).tail(rows-k-1), m_hCoeffs.coeffRef(k), &m_temp.coeffRef(k+1)); + m_qr.bottomRightCorner(rows - k, cols - k - 1) + .applyHouseholderOnTheLeft(m_qr.col(k).tail(rows - k - 1), m_hCoeffs.coeffRef(k), &m_temp.coeffRef(k + 1)); // update our table of norms of the columns for (Index j = k + 1; j < cols; ++j) { @@ -556,9 +518,9 @@ void ColPivHouseholderQR::computeInPlace() if (m_colNormsUpdated.coeffRef(j) != RealScalar(0)) { RealScalar temp = abs(m_qr.coeffRef(k, j)) / m_colNormsUpdated.coeffRef(j); temp = (RealScalar(1) + temp) * (RealScalar(1) - temp); - temp = temp < RealScalar(0) ? RealScalar(0) : temp; - RealScalar temp2 = temp * numext::abs2(m_colNormsUpdated.coeffRef(j) / - m_colNormsDirect.coeffRef(j)); + temp = temp < RealScalar(0) ? RealScalar(0) : temp; + RealScalar temp2 = + temp * numext::abs2(m_colNormsUpdated.coeffRef(j) / m_colNormsDirect.coeffRef(j)); if (temp2 <= norm_downdate_threshold) { // The updated norm has become too inaccurate so re-compute the column // norm directly. @@ -572,10 +534,10 @@ void ColPivHouseholderQR::computeInPlace() } m_colsPermutation.setIdentity(PermIndexType(cols)); - for(PermIndexType k = 0; k < size/*m_nonzero_pivots*/; ++k) + for (PermIndexType k = 0; k < size /*m_nonzero_pivots*/; ++k) m_colsPermutation.applyTranspositionOnTheRight(k, PermIndexType(m_colsTranspositions.coeff(k))); - m_det_pq = (number_of_transpositions%2) ? -1 : 1; + m_det_pq = (number_of_transpositions % 2) ? -1 : 1; m_isInitialized = true; } @@ -588,8 +550,7 @@ void ColPivHouseholderQR<_MatrixType>::_solve_impl(const RhsType &rhs, DstType & const Index nonzero_pivots = nonzeroPivots(); - if(nonzero_pivots == 0) - { + if (nonzero_pivots == 0) { dst.setZero(); return; } @@ -597,57 +558,57 @@ void ColPivHouseholderQR<_MatrixType>::_solve_impl(const RhsType &rhs, DstType & typename RhsType::PlainObject c(rhs); // Note that the matrix Q = H_0^* H_1^*... so its inverse is Q^* = (H_0 H_1 ...)^T - c.applyOnTheLeft(householderSequence(m_qr, m_hCoeffs) - .setLength(nonzero_pivots) - .transpose() - ); + c.applyOnTheLeft(householderSequence(m_qr, m_hCoeffs).setLength(nonzero_pivots).transpose()); m_qr.topLeftCorner(nonzero_pivots, nonzero_pivots) - .template triangularView() - .solveInPlace(c.topRows(nonzero_pivots)); + .template triangularView() + .solveInPlace(c.topRows(nonzero_pivots)); - for(Index i = 0; i < nonzero_pivots; ++i) dst.row(m_colsPermutation.indices().coeff(i)) = c.row(i); - for(Index i = nonzero_pivots; i < cols(); ++i) dst.row(m_colsPermutation.indices().coeff(i)).setZero(); + for (Index i = 0; i < nonzero_pivots; ++i) dst.row(m_colsPermutation.indices().coeff(i)) = c.row(i); + for (Index i = nonzero_pivots; i < cols(); ++i) dst.row(m_colsPermutation.indices().coeff(i)).setZero(); } #endif namespace internal { -template -struct Assignment >, internal::assign_op::Scalar>, Dense2Dense> -{ - typedef ColPivHouseholderQR QrType; - typedef Inverse SrcXprType; - static void run(DstXprType &dst, const SrcXprType &src, const internal::assign_op &) + template + struct Assignment>, + internal::assign_op::Scalar>, + Dense2Dense> { - dst = src.nestedExpression().solve(MatrixType::Identity(src.rows(), src.cols())); - } -}; + typedef ColPivHouseholderQR QrType; + typedef Inverse SrcXprType; + static void run(DstXprType &dst, + const SrcXprType &src, + const internal::assign_op &) + { + dst = src.nestedExpression().solve(MatrixType::Identity(src.rows(), src.cols())); + } + }; -} // end namespace internal +}// end namespace internal /** \returns the matrix Q as a sequence of householder transformations. - * You can extract the meaningful part only by using: - * \code qr.householderQ().setLength(qr.nonzeroPivots()) \endcode*/ + * You can extract the meaningful part only by using: + * \code qr.householderQ().setLength(qr.nonzeroPivots()) \endcode*/ template -typename ColPivHouseholderQR::HouseholderSequenceType ColPivHouseholderQR - ::householderQ() const +typename ColPivHouseholderQR::HouseholderSequenceType ColPivHouseholderQR::householderQ() const { eigen_assert(m_isInitialized && "ColPivHouseholderQR is not initialized."); return HouseholderSequenceType(m_qr, m_hCoeffs.conjugate()); } /** \return the column-pivoting Householder QR decomposition of \c *this. - * - * \sa class ColPivHouseholderQR - */ + * + * \sa class ColPivHouseholderQR + */ template -const ColPivHouseholderQR::PlainObject> -MatrixBase::colPivHouseholderQr() const +const ColPivHouseholderQR::PlainObject> MatrixBase::colPivHouseholderQr() const { return ColPivHouseholderQR(eval()); } -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_COLPIVOTINGHOUSEHOLDERQR_H +#endif// EIGEN_COLPIVOTINGHOUSEHOLDERQR_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/QR/ColPivHouseholderQR_LAPACKE.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/QR/ColPivHouseholderQR_LAPACKE.h index 4e9651f8..adf4cfef 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/QR/ColPivHouseholderQR_LAPACKE.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/QR/ColPivHouseholderQR_LAPACKE.h @@ -34,64 +34,69 @@ #ifndef EIGEN_COLPIVOTINGHOUSEHOLDERQR_LAPACKE_H #define EIGEN_COLPIVOTINGHOUSEHOLDERQR_LAPACKE_H -namespace Eigen { +namespace Eigen { /** \internal Specialization for the data types supported by LAPACKe */ -#define EIGEN_LAPACKE_QR_COLPIV(EIGTYPE, LAPACKE_TYPE, LAPACKE_PREFIX, EIGCOLROW, LAPACKE_COLROW) \ -template<> template inline \ -ColPivHouseholderQR >& \ -ColPivHouseholderQR >::compute( \ - const EigenBase& matrix) \ -\ -{ \ - using std::abs; \ - typedef Matrix MatrixType; \ - typedef MatrixType::RealScalar RealScalar; \ - Index rows = matrix.rows();\ - Index cols = matrix.cols();\ -\ - m_qr = matrix;\ - Index size = m_qr.diagonalSize();\ - m_hCoeffs.resize(size);\ -\ - m_colsTranspositions.resize(cols);\ - /*Index number_of_transpositions = 0;*/ \ -\ - m_nonzero_pivots = 0; \ - m_maxpivot = RealScalar(0);\ - m_colsPermutation.resize(cols); \ - m_colsPermutation.indices().setZero(); \ -\ - lapack_int lda = internal::convert_index(m_qr.outerStride()); \ - lapack_int matrix_order = LAPACKE_COLROW; \ - LAPACKE_##LAPACKE_PREFIX##geqp3( matrix_order, internal::convert_index(rows), internal::convert_index(cols), \ - (LAPACKE_TYPE*)m_qr.data(), lda, (lapack_int*)m_colsPermutation.indices().data(), (LAPACKE_TYPE*)m_hCoeffs.data()); \ - m_isInitialized = true; \ - m_maxpivot=m_qr.diagonal().cwiseAbs().maxCoeff(); \ - m_hCoeffs.adjointInPlace(); \ - RealScalar premultiplied_threshold = abs(m_maxpivot) * threshold(); \ - lapack_int *perm = m_colsPermutation.indices().data(); \ - for(Index i=0;i premultiplied_threshold);\ - } \ - for(Index i=0;i \ + template \ + inline ColPivHouseholderQR> & \ + ColPivHouseholderQR>::compute( \ + const EigenBase &matrix) \ + \ + { \ + using std::abs; \ + typedef Matrix MatrixType; \ + typedef MatrixType::RealScalar RealScalar; \ + Index rows = matrix.rows(); \ + Index cols = matrix.cols(); \ + \ + m_qr = matrix; \ + Index size = m_qr.diagonalSize(); \ + m_hCoeffs.resize(size); \ + \ + m_colsTranspositions.resize(cols); \ + /*Index number_of_transpositions = 0;*/ \ + \ + m_nonzero_pivots = 0; \ + m_maxpivot = RealScalar(0); \ + m_colsPermutation.resize(cols); \ + m_colsPermutation.indices().setZero(); \ + \ + lapack_int lda = internal::convert_index(m_qr.outerStride()); \ + lapack_int matrix_order = LAPACKE_COLROW; \ + LAPACKE_##LAPACKE_PREFIX##geqp3(matrix_order, \ + internal::convert_index(rows), \ + internal::convert_index(cols), \ + (LAPACKE_TYPE *)m_qr.data(), \ + lda, \ + (lapack_int *)m_colsPermutation.indices().data(), \ + (LAPACKE_TYPE *)m_hCoeffs.data()); \ + m_isInitialized = true; \ + m_maxpivot = m_qr.diagonal().cwiseAbs().maxCoeff(); \ + m_hCoeffs.adjointInPlace(); \ + RealScalar premultiplied_threshold = abs(m_maxpivot) * threshold(); \ + lapack_int *perm = m_colsPermutation.indices().data(); \ + for (Index i = 0; i < size; i++) { m_nonzero_pivots += (abs(m_qr.coeff(i, i)) > premultiplied_threshold); } \ + for (Index i = 0; i < cols; i++) perm[i]--; \ + \ + /*m_det_pq = (number_of_transpositions%2) ? -1 : 1; // TODO: It's not needed now; fix upon availability in Eigen \ + */ \ + \ + return *this; \ + } -EIGEN_LAPACKE_QR_COLPIV(double, double, d, ColMajor, LAPACK_COL_MAJOR) -EIGEN_LAPACKE_QR_COLPIV(float, float, s, ColMajor, LAPACK_COL_MAJOR) +EIGEN_LAPACKE_QR_COLPIV(double, double, d, ColMajor, LAPACK_COL_MAJOR) +EIGEN_LAPACKE_QR_COLPIV(float, float, s, ColMajor, LAPACK_COL_MAJOR) EIGEN_LAPACKE_QR_COLPIV(dcomplex, lapack_complex_double, z, ColMajor, LAPACK_COL_MAJOR) -EIGEN_LAPACKE_QR_COLPIV(scomplex, lapack_complex_float, c, ColMajor, LAPACK_COL_MAJOR) +EIGEN_LAPACKE_QR_COLPIV(scomplex, lapack_complex_float, c, ColMajor, LAPACK_COL_MAJOR) -EIGEN_LAPACKE_QR_COLPIV(double, double, d, RowMajor, LAPACK_ROW_MAJOR) -EIGEN_LAPACKE_QR_COLPIV(float, float, s, RowMajor, LAPACK_ROW_MAJOR) +EIGEN_LAPACKE_QR_COLPIV(double, double, d, RowMajor, LAPACK_ROW_MAJOR) +EIGEN_LAPACKE_QR_COLPIV(float, float, s, RowMajor, LAPACK_ROW_MAJOR) EIGEN_LAPACKE_QR_COLPIV(dcomplex, lapack_complex_double, z, RowMajor, LAPACK_ROW_MAJOR) -EIGEN_LAPACKE_QR_COLPIV(scomplex, lapack_complex_float, c, RowMajor, LAPACK_ROW_MAJOR) +EIGEN_LAPACKE_QR_COLPIV(scomplex, lapack_complex_float, c, RowMajor, LAPACK_ROW_MAJOR) -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_COLPIVOTINGHOUSEHOLDERQR_LAPACKE_H +#endif// EIGEN_COLPIVOTINGHOUSEHOLDERQR_LAPACKE_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/QR/CompleteOrthogonalDecomposition.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/QR/CompleteOrthogonalDecomposition.h index 34c637b7..1288a8ab 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/QR/CompleteOrthogonalDecomposition.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/QR/CompleteOrthogonalDecomposition.h @@ -13,40 +13,39 @@ namespace Eigen { namespace internal { -template -struct traits > - : traits<_MatrixType> { - enum { Flags = 0 }; -}; + template struct traits> : traits<_MatrixType> + { + enum { Flags = 0 }; + }; -} // end namespace internal +}// end namespace internal /** \ingroup QR_Module - * - * \class CompleteOrthogonalDecomposition - * - * \brief Complete orthogonal decomposition (COD) of a matrix. - * - * \param MatrixType the type of the matrix of which we are computing the COD. - * - * This class performs a rank-revealing complete orthogonal decomposition of a - * matrix \b A into matrices \b P, \b Q, \b T, and \b Z such that - * \f[ - * \mathbf{A} \, \mathbf{P} = \mathbf{Q} \, - * \begin{bmatrix} \mathbf{T} & \mathbf{0} \\ - * \mathbf{0} & \mathbf{0} \end{bmatrix} \, \mathbf{Z} - * \f] - * by using Householder transformations. Here, \b P is a permutation matrix, - * \b Q and \b Z are unitary matrices and \b T an upper triangular matrix of - * size rank-by-rank. \b A may be rank deficient. - * - * This class supports the \link InplaceDecomposition inplace decomposition \endlink mechanism. - * - * \sa MatrixBase::completeOrthogonalDecomposition() - */ -template -class CompleteOrthogonalDecomposition { - public: + * + * \class CompleteOrthogonalDecomposition + * + * \brief Complete orthogonal decomposition (COD) of a matrix. + * + * \param MatrixType the type of the matrix of which we are computing the COD. + * + * This class performs a rank-revealing complete orthogonal decomposition of a + * matrix \b A into matrices \b P, \b Q, \b T, and \b Z such that + * \f[ + * \mathbf{A} \, \mathbf{P} = \mathbf{Q} \, + * \begin{bmatrix} \mathbf{T} & \mathbf{0} \\ + * \mathbf{0} & \mathbf{0} \end{bmatrix} \, \mathbf{Z} + * \f] + * by using Householder transformations. Here, \b P is a permutation matrix, + * \b Q and \b Z are unitary matrices and \b T an upper triangular matrix of + * size rank-by-rank. \b A may be rank deficient. + * + * This class supports the \link InplaceDecomposition inplace decomposition \endlink mechanism. + * + * \sa MatrixBase::completeOrthogonalDecomposition() + */ +template class CompleteOrthogonalDecomposition +{ +public: typedef _MatrixType MatrixType; enum { RowsAtCompileTime = MatrixType::RowsAtCompileTime, @@ -58,23 +57,19 @@ class CompleteOrthogonalDecomposition { typedef typename MatrixType::RealScalar RealScalar; typedef typename MatrixType::StorageIndex StorageIndex; typedef typename internal::plain_diag_type::type HCoeffsType; - typedef PermutationMatrix - PermutationType; - typedef typename internal::plain_row_type::type - IntRowVectorType; + typedef PermutationMatrix PermutationType; + typedef typename internal::plain_row_type::type IntRowVectorType; typedef typename internal::plain_row_type::type RowVectorType; - typedef typename internal::plain_row_type::type - RealRowVectorType; - typedef HouseholderSequence< - MatrixType, typename internal::remove_all< - typename HCoeffsType::ConjugateReturnType>::type> - HouseholderSequenceType; + typedef typename internal::plain_row_type::type RealRowVectorType; + typedef HouseholderSequence::type> + HouseholderSequenceType; typedef typename MatrixType::PlainObject PlainObject; - private: +private: typedef typename PermutationType::Index PermIndexType; - public: +public: /** * \brief Default Constructor. * @@ -91,7 +86,8 @@ class CompleteOrthogonalDecomposition { * \sa CompleteOrthogonalDecomposition() */ CompleteOrthogonalDecomposition(Index rows, Index cols) - : m_cpqr(rows, cols), m_zCoeffs((std::min)(rows, cols)), m_temp(cols) {} + : m_cpqr(rows, cols), m_zCoeffs((std::min)(rows, cols)), m_temp(cols) + {} /** \brief Constructs a complete orthogonal decomposition from a given * matrix. @@ -109,26 +105,23 @@ class CompleteOrthogonalDecomposition { * * \sa compute() */ - template - explicit CompleteOrthogonalDecomposition(const EigenBase& matrix) - : m_cpqr(matrix.rows(), matrix.cols()), - m_zCoeffs((std::min)(matrix.rows(), matrix.cols())), - m_temp(matrix.cols()) + template + explicit CompleteOrthogonalDecomposition(const EigenBase &matrix) + : m_cpqr(matrix.rows(), matrix.cols()), m_zCoeffs((std::min)(matrix.rows(), matrix.cols())), m_temp(matrix.cols()) { compute(matrix.derived()); } /** \brief Constructs a complete orthogonal decomposition from a given matrix - * - * This overloaded constructor is provided for \link InplaceDecomposition inplace decomposition \endlink when \c MatrixType is a Eigen::Ref. - * - * \sa CompleteOrthogonalDecomposition(const EigenBase&) - */ + * + * This overloaded constructor is provided for \link InplaceDecomposition inplace decomposition \endlink when \c + * MatrixType is a Eigen::Ref. + * + * \sa CompleteOrthogonalDecomposition(const EigenBase&) + */ template - explicit CompleteOrthogonalDecomposition(EigenBase& matrix) - : m_cpqr(matrix.derived()), - m_zCoeffs((std::min)(matrix.rows(), matrix.cols())), - m_temp(matrix.cols()) + explicit CompleteOrthogonalDecomposition(EigenBase &matrix) + : m_cpqr(matrix.derived()), m_zCoeffs((std::min)(matrix.rows(), matrix.cols())), m_temp(matrix.cols()) { computeInPlace(); } @@ -143,11 +136,9 @@ class CompleteOrthogonalDecomposition { * \returns a solution. * */ - template - inline const Solve solve( - const MatrixBase& b) const { - eigen_assert(m_cpqr.m_isInitialized && - "CompleteOrthogonalDecomposition is not initialized."); + template inline const Solve solve(const MatrixBase &b) const + { + eigen_assert(m_cpqr.m_isInitialized && "CompleteOrthogonalDecomposition is not initialized."); return Solve(*this, b.derived()); } @@ -156,7 +147,8 @@ class CompleteOrthogonalDecomposition { /** \returns the matrix \b Z. */ - MatrixType matrixZ() const { + MatrixType matrixZ() const + { MatrixType Z = MatrixType::Identity(m_cpqr.cols(), m_cpqr.cols()); applyZAdjointOnTheLeftInPlace(Z); return Z.adjoint(); @@ -165,7 +157,7 @@ class CompleteOrthogonalDecomposition { /** \returns a reference to the matrix where the complete orthogonal * decomposition is stored */ - const MatrixType& matrixQTZ() const { return m_cpqr.matrixQR(); } + const MatrixType &matrixQTZ() const { return m_cpqr.matrixQR(); } /** \returns a reference to the matrix where the complete orthogonal * decomposition is stored. @@ -178,10 +170,10 @@ class CompleteOrthogonalDecomposition { * matrixR().topLeftCorner(rank(), rank()).template triangularView() * \endcode */ - const MatrixType& matrixT() const { return m_cpqr.matrixQR(); } + const MatrixType &matrixT() const { return m_cpqr.matrixQR(); } - template - CompleteOrthogonalDecomposition& compute(const EigenBase& matrix) { + template CompleteOrthogonalDecomposition &compute(const EigenBase &matrix) + { // Compute the column pivoted QR factorization A P = Q R. m_cpqr.compute(matrix); computeInPlace(); @@ -189,9 +181,7 @@ class CompleteOrthogonalDecomposition { } /** \returns a const reference to the column permutation matrix */ - const PermutationType& colsPermutation() const { - return m_cpqr.colsPermutation(); - } + const PermutationType &colsPermutation() const { return m_cpqr.colsPermutation(); } /** \returns the absolute value of the determinant of the matrix of which * *this is the complete orthogonal decomposition. It has only linear @@ -286,14 +276,14 @@ class CompleteOrthogonalDecomposition { * * For advanced uses only. */ - inline const HCoeffsType& hCoeffs() const { return m_cpqr.hCoeffs(); } + inline const HCoeffsType &hCoeffs() const { return m_cpqr.hCoeffs(); } /** \returns a const reference to the vector of Householder coefficients * used to represent the factor \c Z. * * For advanced uses only. */ - const HCoeffsType& zCoeffs() const { return m_zCoeffs; } + const HCoeffsType &zCoeffs() const { return m_zCoeffs; } /** Allows to prescribe a threshold to be used by certain methods, such as * rank(), who need to determine when pivots are to be considered nonzero. @@ -314,7 +304,8 @@ class CompleteOrthogonalDecomposition { * If you want to come back to the default behavior, call * setThreshold(Default_t) */ - CompleteOrthogonalDecomposition& setThreshold(const RealScalar& threshold) { + CompleteOrthogonalDecomposition &setThreshold(const RealScalar &threshold) + { m_cpqr.setThreshold(threshold); return *this; } @@ -327,7 +318,8 @@ class CompleteOrthogonalDecomposition { * * See the documentation of setThreshold(const RealScalar&). */ - CompleteOrthogonalDecomposition& setThreshold(Default_t) { + CompleteOrthogonalDecomposition &setThreshold(Default_t) + { m_cpqr.setThreshold(Default); return *this; } @@ -360,42 +352,40 @@ class CompleteOrthogonalDecomposition { * with other factorization routines. * \returns \c Success */ - ComputationInfo info() const { + ComputationInfo info() const + { eigen_assert(m_cpqr.m_isInitialized && "Decomposition is not initialized."); return Success; } #ifndef EIGEN_PARSED_BY_DOXYGEN - template - EIGEN_DEVICE_FUNC void _solve_impl(const RhsType& rhs, DstType& dst) const; + template + EIGEN_DEVICE_FUNC void _solve_impl(const RhsType &rhs, DstType &dst) const; #endif - protected: - static void check_template_parameters() { - EIGEN_STATIC_ASSERT_NON_INTEGER(Scalar); - } +protected: + static void check_template_parameters() { EIGEN_STATIC_ASSERT_NON_INTEGER(Scalar); } void computeInPlace(); /** Overwrites \b rhs with \f$ \mathbf{Z}^* * \mathbf{rhs} \f$. */ - template - void applyZAdjointOnTheLeftInPlace(Rhs& rhs) const; + template void applyZAdjointOnTheLeftInPlace(Rhs &rhs) const; ColPivHouseholderQR m_cpqr; HCoeffsType m_zCoeffs; RowVectorType m_temp; }; -template -typename MatrixType::RealScalar -CompleteOrthogonalDecomposition::absDeterminant() const { +template +typename MatrixType::RealScalar CompleteOrthogonalDecomposition::absDeterminant() const +{ return m_cpqr.absDeterminant(); } -template -typename MatrixType::RealScalar -CompleteOrthogonalDecomposition::logAbsDeterminant() const { +template +typename MatrixType::RealScalar CompleteOrthogonalDecomposition::logAbsDeterminant() const +{ return m_cpqr.logAbsDeterminant(); } @@ -406,8 +396,7 @@ CompleteOrthogonalDecomposition::logAbsDeterminant() const { * \sa class CompleteOrthogonalDecomposition, * CompleteOrthogonalDecomposition(const MatrixType&) */ -template -void CompleteOrthogonalDecomposition::computeInPlace() +template void CompleteOrthogonalDecomposition::computeInPlace() { check_template_parameters(); @@ -437,60 +426,48 @@ void CompleteOrthogonalDecomposition::computeInPlace() // Given the API for Householder reflectors, it is more convenient if // we swap the leading parts of columns k and r-1 (zero-based) to form // the matrix X_k = [X(0:k, k), X(0:k, r:n)] - m_cpqr.m_qr.col(k).head(k + 1).swap( - m_cpqr.m_qr.col(rank - 1).head(k + 1)); + m_cpqr.m_qr.col(k).head(k + 1).swap(m_cpqr.m_qr.col(rank - 1).head(k + 1)); } // Construct Householder reflector Z(k) to zero out the last row of X_k, // i.e. choose Z(k) such that // [X(k, k), X(k, r:n)] * Z(k) = [beta, 0, .., 0]. RealScalar beta; - m_cpqr.m_qr.row(k) - .tail(cols - rank + 1) - .makeHouseholderInPlace(m_zCoeffs(k), beta); + m_cpqr.m_qr.row(k).tail(cols - rank + 1).makeHouseholderInPlace(m_zCoeffs(k), beta); m_cpqr.m_qr(k, rank - 1) = beta; if (k > 0) { // Apply Z(k) to the first k rows of X_k m_cpqr.m_qr.topRightCorner(k, cols - rank + 1) - .applyHouseholderOnTheRight( - m_cpqr.m_qr.row(k).tail(cols - rank).transpose(), m_zCoeffs(k), - &m_temp(0)); + .applyHouseholderOnTheRight(m_cpqr.m_qr.row(k).tail(cols - rank).transpose(), m_zCoeffs(k), &m_temp(0)); } if (k != rank - 1) { // Swap X(0:k,k) back to its proper location. - m_cpqr.m_qr.col(k).head(k + 1).swap( - m_cpqr.m_qr.col(rank - 1).head(k + 1)); + m_cpqr.m_qr.col(k).head(k + 1).swap(m_cpqr.m_qr.col(rank - 1).head(k + 1)); } } } } -template -template -void CompleteOrthogonalDecomposition::applyZAdjointOnTheLeftInPlace( - Rhs& rhs) const { +template +template +void CompleteOrthogonalDecomposition::applyZAdjointOnTheLeftInPlace(Rhs &rhs) const +{ const Index cols = this->cols(); const Index nrhs = rhs.cols(); const Index rank = this->rank(); Matrix temp((std::max)(cols, nrhs)); for (Index k = 0; k < rank; ++k) { - if (k != rank - 1) { - rhs.row(k).swap(rhs.row(rank - 1)); - } + if (k != rank - 1) { rhs.row(k).swap(rhs.row(rank - 1)); } rhs.middleRows(rank - 1, cols - rank + 1) - .applyHouseholderOnTheLeft( - matrixQTZ().row(k).tail(cols - rank).adjoint(), zCoeffs()(k), - &temp(0)); - if (k != rank - 1) { - rhs.row(k).swap(rhs.row(rank - 1)); - } + .applyHouseholderOnTheLeft(matrixQTZ().row(k).tail(cols - rank).adjoint(), zCoeffs()(k), &temp(0)); + if (k != rank - 1) { rhs.row(k).swap(rhs.row(rank - 1)); } } } #ifndef EIGEN_PARSED_BY_DOXYGEN -template -template -void CompleteOrthogonalDecomposition<_MatrixType>::_solve_impl( - const RhsType& rhs, DstType& dst) const { +template +template +void CompleteOrthogonalDecomposition<_MatrixType>::_solve_impl(const RhsType &rhs, DstType &dst) const +{ eigen_assert(rhs.rows() == this->rows()); const Index rank = this->rank(); @@ -503,14 +480,10 @@ void CompleteOrthogonalDecomposition<_MatrixType>::_solve_impl( // Note that the matrix Q = H_0^* H_1^*... so its inverse is // Q^* = (H_0 H_1 ...)^T typename RhsType::PlainObject c(rhs); - c.applyOnTheLeft( - householderSequence(matrixQTZ(), hCoeffs()).setLength(rank).transpose()); + c.applyOnTheLeft(householderSequence(matrixQTZ(), hCoeffs()).setLength(rank).transpose()); // Solve T z = c(1:rank, :) - dst.topRows(rank) = matrixT() - .topLeftCorner(rank, rank) - .template triangularView() - .solve(c.topRows(rank)); + dst.topRows(rank) = matrixT().topLeftCorner(rank, rank).template triangularView().solve(c.topRows(rank)); const Index cols = this->cols(); if (rank < cols) { @@ -527,36 +500,43 @@ void CompleteOrthogonalDecomposition<_MatrixType>::_solve_impl( namespace internal { -template -struct Assignment >, internal::assign_op::Scalar>, Dense2Dense> -{ - typedef CompleteOrthogonalDecomposition CodType; - typedef Inverse SrcXprType; - static void run(DstXprType &dst, const SrcXprType &src, const internal::assign_op &) + template + struct Assignment>, + internal::assign_op::Scalar>, + Dense2Dense> { - dst = src.nestedExpression().solve(MatrixType::Identity(src.rows(), src.rows())); - } -}; + typedef CompleteOrthogonalDecomposition CodType; + typedef Inverse SrcXprType; + static void run(DstXprType &dst, + const SrcXprType &src, + const internal::assign_op &) + { + dst = src.nestedExpression().solve(MatrixType::Identity(src.rows(), src.rows())); + } + }; -} // end namespace internal +}// end namespace internal /** \returns the matrix Q as a sequence of householder transformations */ -template +template typename CompleteOrthogonalDecomposition::HouseholderSequenceType -CompleteOrthogonalDecomposition::householderQ() const { + CompleteOrthogonalDecomposition::householderQ() const +{ return m_cpqr.householderQ(); } /** \return the complete orthogonal decomposition of \c *this. - * - * \sa class CompleteOrthogonalDecomposition - */ -template + * + * \sa class CompleteOrthogonalDecomposition + */ +template const CompleteOrthogonalDecomposition::PlainObject> -MatrixBase::completeOrthogonalDecomposition() const { + MatrixBase::completeOrthogonalDecomposition() const +{ return CompleteOrthogonalDecomposition(eval()); } -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_COMPLETEORTHOGONALDECOMPOSITION_H +#endif// EIGEN_COMPLETEORTHOGONALDECOMPOSITION_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/QR/FullPivHouseholderQR.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/QR/FullPivHouseholderQR.h index e489bddc..4b2912bc 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/QR/FullPivHouseholderQR.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/QR/FullPivHouseholderQR.h @@ -11,418 +11,393 @@ #ifndef EIGEN_FULLPIVOTINGHOUSEHOLDERQR_H #define EIGEN_FULLPIVOTINGHOUSEHOLDERQR_H -namespace Eigen { +namespace Eigen { namespace internal { -template struct traits > - : traits<_MatrixType> -{ - enum { Flags = 0 }; -}; + template struct traits> : traits<_MatrixType> + { + enum { Flags = 0 }; + }; -template struct FullPivHouseholderQRMatrixQReturnType; + template struct FullPivHouseholderQRMatrixQReturnType; -template -struct traits > -{ - typedef typename MatrixType::PlainObject ReturnType; -}; + template struct traits> + { + typedef typename MatrixType::PlainObject ReturnType; + }; -} // end namespace internal +}// end namespace internal /** \ingroup QR_Module - * - * \class FullPivHouseholderQR - * - * \brief Householder rank-revealing QR decomposition of a matrix with full pivoting - * - * \tparam _MatrixType the type of the matrix of which we are computing the QR decomposition - * - * This class performs a rank-revealing QR decomposition of a matrix \b A into matrices \b P, \b P', \b Q and \b R - * such that - * \f[ - * \mathbf{P} \, \mathbf{A} \, \mathbf{P}' = \mathbf{Q} \, \mathbf{R} - * \f] - * by using Householder transformations. Here, \b P and \b P' are permutation matrices, \b Q a unitary matrix - * and \b R an upper triangular matrix. - * - * This decomposition performs a very prudent full pivoting in order to be rank-revealing and achieve optimal - * numerical stability. The trade-off is that it is slower than HouseholderQR and ColPivHouseholderQR. - * - * This class supports the \link InplaceDecomposition inplace decomposition \endlink mechanism. - * - * \sa MatrixBase::fullPivHouseholderQr() - */ + * + * \class FullPivHouseholderQR + * + * \brief Householder rank-revealing QR decomposition of a matrix with full pivoting + * + * \tparam _MatrixType the type of the matrix of which we are computing the QR decomposition + * + * This class performs a rank-revealing QR decomposition of a matrix \b A into matrices \b P, \b P', \b Q and \b R + * such that + * \f[ + * \mathbf{P} \, \mathbf{A} \, \mathbf{P}' = \mathbf{Q} \, \mathbf{R} + * \f] + * by using Householder transformations. Here, \b P and \b P' are permutation matrices, \b Q a unitary matrix + * and \b R an upper triangular matrix. + * + * This decomposition performs a very prudent full pivoting in order to be rank-revealing and achieve optimal + * numerical stability. The trade-off is that it is slower than HouseholderQR and ColPivHouseholderQR. + * + * This class supports the \link InplaceDecomposition inplace decomposition \endlink mechanism. + * + * \sa MatrixBase::fullPivHouseholderQr() + */ template class FullPivHouseholderQR { - public: +public: + typedef _MatrixType MatrixType; + enum { + RowsAtCompileTime = MatrixType::RowsAtCompileTime, + ColsAtCompileTime = MatrixType::ColsAtCompileTime, + MaxRowsAtCompileTime = MatrixType::MaxRowsAtCompileTime, + MaxColsAtCompileTime = MatrixType::MaxColsAtCompileTime + }; + typedef typename MatrixType::Scalar Scalar; + typedef typename MatrixType::RealScalar RealScalar; + // FIXME should be int + typedef typename MatrixType::StorageIndex StorageIndex; + typedef internal::FullPivHouseholderQRMatrixQReturnType MatrixQReturnType; + typedef typename internal::plain_diag_type::type HCoeffsType; + typedef Matrix + IntDiagSizeVectorType; + typedef PermutationMatrix PermutationType; + typedef typename internal::plain_row_type::type RowVectorType; + typedef typename internal::plain_col_type::type ColVectorType; + typedef typename MatrixType::PlainObject PlainObject; + + /** \brief Default Constructor. + * + * The default constructor is useful in cases in which the user intends to + * perform decompositions via FullPivHouseholderQR::compute(const MatrixType&). + */ + FullPivHouseholderQR() + : m_qr(), m_hCoeffs(), m_rows_transpositions(), m_cols_transpositions(), m_cols_permutation(), m_temp(), + m_isInitialized(false), m_usePrescribedThreshold(false) + {} - typedef _MatrixType MatrixType; - enum { - RowsAtCompileTime = MatrixType::RowsAtCompileTime, - ColsAtCompileTime = MatrixType::ColsAtCompileTime, - MaxRowsAtCompileTime = MatrixType::MaxRowsAtCompileTime, - MaxColsAtCompileTime = MatrixType::MaxColsAtCompileTime - }; - typedef typename MatrixType::Scalar Scalar; - typedef typename MatrixType::RealScalar RealScalar; - // FIXME should be int - typedef typename MatrixType::StorageIndex StorageIndex; - typedef internal::FullPivHouseholderQRMatrixQReturnType MatrixQReturnType; - typedef typename internal::plain_diag_type::type HCoeffsType; - typedef Matrix IntDiagSizeVectorType; - typedef PermutationMatrix PermutationType; - typedef typename internal::plain_row_type::type RowVectorType; - typedef typename internal::plain_col_type::type ColVectorType; - typedef typename MatrixType::PlainObject PlainObject; - - /** \brief Default Constructor. - * - * The default constructor is useful in cases in which the user intends to - * perform decompositions via FullPivHouseholderQR::compute(const MatrixType&). - */ - FullPivHouseholderQR() - : m_qr(), - m_hCoeffs(), - m_rows_transpositions(), - m_cols_transpositions(), - m_cols_permutation(), - m_temp(), - m_isInitialized(false), - m_usePrescribedThreshold(false) {} - - /** \brief Default Constructor with memory preallocation - * - * Like the default constructor but with preallocation of the internal data - * according to the specified problem \a size. - * \sa FullPivHouseholderQR() - */ - FullPivHouseholderQR(Index rows, Index cols) - : m_qr(rows, cols), - m_hCoeffs((std::min)(rows,cols)), - m_rows_transpositions((std::min)(rows,cols)), - m_cols_transpositions((std::min)(rows,cols)), - m_cols_permutation(cols), - m_temp(cols), - m_isInitialized(false), - m_usePrescribedThreshold(false) {} - - /** \brief Constructs a QR factorization from a given matrix - * - * This constructor computes the QR factorization of the matrix \a matrix by calling - * the method compute(). It is a short cut for: - * - * \code - * FullPivHouseholderQR qr(matrix.rows(), matrix.cols()); - * qr.compute(matrix); - * \endcode - * - * \sa compute() - */ - template - explicit FullPivHouseholderQR(const EigenBase& matrix) - : m_qr(matrix.rows(), matrix.cols()), - m_hCoeffs((std::min)(matrix.rows(), matrix.cols())), - m_rows_transpositions((std::min)(matrix.rows(), matrix.cols())), - m_cols_transpositions((std::min)(matrix.rows(), matrix.cols())), - m_cols_permutation(matrix.cols()), - m_temp(matrix.cols()), - m_isInitialized(false), - m_usePrescribedThreshold(false) - { - compute(matrix.derived()); - } + /** \brief Default Constructor with memory preallocation + * + * Like the default constructor but with preallocation of the internal data + * according to the specified problem \a size. + * \sa FullPivHouseholderQR() + */ + FullPivHouseholderQR(Index rows, Index cols) + : m_qr(rows, cols), m_hCoeffs((std::min)(rows, cols)), m_rows_transpositions((std::min)(rows, cols)), + m_cols_transpositions((std::min)(rows, cols)), m_cols_permutation(cols), m_temp(cols), m_isInitialized(false), + m_usePrescribedThreshold(false) + {} - /** \brief Constructs a QR factorization from a given matrix - * - * This overloaded constructor is provided for \link InplaceDecomposition inplace decomposition \endlink when \c MatrixType is a Eigen::Ref. - * - * \sa FullPivHouseholderQR(const EigenBase&) - */ - template - explicit FullPivHouseholderQR(EigenBase& matrix) - : m_qr(matrix.derived()), - m_hCoeffs((std::min)(matrix.rows(), matrix.cols())), - m_rows_transpositions((std::min)(matrix.rows(), matrix.cols())), - m_cols_transpositions((std::min)(matrix.rows(), matrix.cols())), - m_cols_permutation(matrix.cols()), - m_temp(matrix.cols()), - m_isInitialized(false), - m_usePrescribedThreshold(false) - { - computeInPlace(); - } + /** \brief Constructs a QR factorization from a given matrix + * + * This constructor computes the QR factorization of the matrix \a matrix by calling + * the method compute(). It is a short cut for: + * + * \code + * FullPivHouseholderQR qr(matrix.rows(), matrix.cols()); + * qr.compute(matrix); + * \endcode + * + * \sa compute() + */ + template + explicit FullPivHouseholderQR(const EigenBase &matrix) + : m_qr(matrix.rows(), matrix.cols()), m_hCoeffs((std::min)(matrix.rows(), matrix.cols())), + m_rows_transpositions((std::min)(matrix.rows(), matrix.cols())), + m_cols_transpositions((std::min)(matrix.rows(), matrix.cols())), m_cols_permutation(matrix.cols()), + m_temp(matrix.cols()), m_isInitialized(false), m_usePrescribedThreshold(false) + { + compute(matrix.derived()); + } - /** This method finds a solution x to the equation Ax=b, where A is the matrix of which - * \c *this is the QR decomposition. - * - * \param b the right-hand-side of the equation to solve. - * - * \returns the exact or least-square solution if the rank is greater or equal to the number of columns of A, - * and an arbitrary solution otherwise. - * - * \note_about_checking_solutions - * - * \note_about_arbitrary_choice_of_solution - * - * Example: \include FullPivHouseholderQR_solve.cpp - * Output: \verbinclude FullPivHouseholderQR_solve.out - */ - template - inline const Solve - solve(const MatrixBase& b) const - { - eigen_assert(m_isInitialized && "FullPivHouseholderQR is not initialized."); - return Solve(*this, b.derived()); - } + /** \brief Constructs a QR factorization from a given matrix + * + * This overloaded constructor is provided for \link InplaceDecomposition inplace decomposition \endlink when \c + * MatrixType is a Eigen::Ref. + * + * \sa FullPivHouseholderQR(const EigenBase&) + */ + template + explicit FullPivHouseholderQR(EigenBase &matrix) + : m_qr(matrix.derived()), m_hCoeffs((std::min)(matrix.rows(), matrix.cols())), + m_rows_transpositions((std::min)(matrix.rows(), matrix.cols())), + m_cols_transpositions((std::min)(matrix.rows(), matrix.cols())), m_cols_permutation(matrix.cols()), + m_temp(matrix.cols()), m_isInitialized(false), m_usePrescribedThreshold(false) + { + computeInPlace(); + } - /** \returns Expression object representing the matrix Q - */ - MatrixQReturnType matrixQ(void) const; + /** This method finds a solution x to the equation Ax=b, where A is the matrix of which + * \c *this is the QR decomposition. + * + * \param b the right-hand-side of the equation to solve. + * + * \returns the exact or least-square solution if the rank is greater or equal to the number of columns of A, + * and an arbitrary solution otherwise. + * + * \note_about_checking_solutions + * + * \note_about_arbitrary_choice_of_solution + * + * Example: \include FullPivHouseholderQR_solve.cpp + * Output: \verbinclude FullPivHouseholderQR_solve.out + */ + template inline const Solve solve(const MatrixBase &b) const + { + eigen_assert(m_isInitialized && "FullPivHouseholderQR is not initialized."); + return Solve(*this, b.derived()); + } - /** \returns a reference to the matrix where the Householder QR decomposition is stored - */ - const MatrixType& matrixQR() const - { - eigen_assert(m_isInitialized && "FullPivHouseholderQR is not initialized."); - return m_qr; - } + /** \returns Expression object representing the matrix Q + */ + MatrixQReturnType matrixQ(void) const; - template - FullPivHouseholderQR& compute(const EigenBase& matrix); + /** \returns a reference to the matrix where the Householder QR decomposition is stored + */ + const MatrixType &matrixQR() const + { + eigen_assert(m_isInitialized && "FullPivHouseholderQR is not initialized."); + return m_qr; + } - /** \returns a const reference to the column permutation matrix */ - const PermutationType& colsPermutation() const - { - eigen_assert(m_isInitialized && "FullPivHouseholderQR is not initialized."); - return m_cols_permutation; - } + template FullPivHouseholderQR &compute(const EigenBase &matrix); - /** \returns a const reference to the vector of indices representing the rows transpositions */ - const IntDiagSizeVectorType& rowsTranspositions() const - { - eigen_assert(m_isInitialized && "FullPivHouseholderQR is not initialized."); - return m_rows_transpositions; - } + /** \returns a const reference to the column permutation matrix */ + const PermutationType &colsPermutation() const + { + eigen_assert(m_isInitialized && "FullPivHouseholderQR is not initialized."); + return m_cols_permutation; + } - /** \returns the absolute value of the determinant of the matrix of which - * *this is the QR decomposition. It has only linear complexity - * (that is, O(n) where n is the dimension of the square matrix) - * as the QR decomposition has already been computed. - * - * \note This is only for square matrices. - * - * \warning a determinant can be very big or small, so for matrices - * of large enough dimension, there is a risk of overflow/underflow. - * One way to work around that is to use logAbsDeterminant() instead. - * - * \sa logAbsDeterminant(), MatrixBase::determinant() - */ - typename MatrixType::RealScalar absDeterminant() const; - - /** \returns the natural log of the absolute value of the determinant of the matrix of which - * *this is the QR decomposition. It has only linear complexity - * (that is, O(n) where n is the dimension of the square matrix) - * as the QR decomposition has already been computed. - * - * \note This is only for square matrices. - * - * \note This method is useful to work around the risk of overflow/underflow that's inherent - * to determinant computation. - * - * \sa absDeterminant(), MatrixBase::determinant() - */ - typename MatrixType::RealScalar logAbsDeterminant() const; - - /** \returns the rank of the matrix of which *this is the QR decomposition. - * - * \note This method has to determine which pivots should be considered nonzero. - * For that, it uses the threshold value that you can control by calling - * setThreshold(const RealScalar&). - */ - inline Index rank() const - { - using std::abs; - eigen_assert(m_isInitialized && "FullPivHouseholderQR is not initialized."); - RealScalar premultiplied_threshold = abs(m_maxpivot) * threshold(); - Index result = 0; - for(Index i = 0; i < m_nonzero_pivots; ++i) - result += (abs(m_qr.coeff(i,i)) > premultiplied_threshold); - return result; - } + /** \returns a const reference to the vector of indices representing the rows transpositions */ + const IntDiagSizeVectorType &rowsTranspositions() const + { + eigen_assert(m_isInitialized && "FullPivHouseholderQR is not initialized."); + return m_rows_transpositions; + } - /** \returns the dimension of the kernel of the matrix of which *this is the QR decomposition. - * - * \note This method has to determine which pivots should be considered nonzero. - * For that, it uses the threshold value that you can control by calling - * setThreshold(const RealScalar&). - */ - inline Index dimensionOfKernel() const - { - eigen_assert(m_isInitialized && "FullPivHouseholderQR is not initialized."); - return cols() - rank(); - } + /** \returns the absolute value of the determinant of the matrix of which + * *this is the QR decomposition. It has only linear complexity + * (that is, O(n) where n is the dimension of the square matrix) + * as the QR decomposition has already been computed. + * + * \note This is only for square matrices. + * + * \warning a determinant can be very big or small, so for matrices + * of large enough dimension, there is a risk of overflow/underflow. + * One way to work around that is to use logAbsDeterminant() instead. + * + * \sa logAbsDeterminant(), MatrixBase::determinant() + */ + typename MatrixType::RealScalar absDeterminant() const; + + /** \returns the natural log of the absolute value of the determinant of the matrix of which + * *this is the QR decomposition. It has only linear complexity + * (that is, O(n) where n is the dimension of the square matrix) + * as the QR decomposition has already been computed. + * + * \note This is only for square matrices. + * + * \note This method is useful to work around the risk of overflow/underflow that's inherent + * to determinant computation. + * + * \sa absDeterminant(), MatrixBase::determinant() + */ + typename MatrixType::RealScalar logAbsDeterminant() const; + + /** \returns the rank of the matrix of which *this is the QR decomposition. + * + * \note This method has to determine which pivots should be considered nonzero. + * For that, it uses the threshold value that you can control by calling + * setThreshold(const RealScalar&). + */ + inline Index rank() const + { + using std::abs; + eigen_assert(m_isInitialized && "FullPivHouseholderQR is not initialized."); + RealScalar premultiplied_threshold = abs(m_maxpivot) * threshold(); + Index result = 0; + for (Index i = 0; i < m_nonzero_pivots; ++i) result += (abs(m_qr.coeff(i, i)) > premultiplied_threshold); + return result; + } - /** \returns true if the matrix of which *this is the QR decomposition represents an injective - * linear map, i.e. has trivial kernel; false otherwise. - * - * \note This method has to determine which pivots should be considered nonzero. - * For that, it uses the threshold value that you can control by calling - * setThreshold(const RealScalar&). - */ - inline bool isInjective() const - { - eigen_assert(m_isInitialized && "FullPivHouseholderQR is not initialized."); - return rank() == cols(); - } + /** \returns the dimension of the kernel of the matrix of which *this is the QR decomposition. + * + * \note This method has to determine which pivots should be considered nonzero. + * For that, it uses the threshold value that you can control by calling + * setThreshold(const RealScalar&). + */ + inline Index dimensionOfKernel() const + { + eigen_assert(m_isInitialized && "FullPivHouseholderQR is not initialized."); + return cols() - rank(); + } - /** \returns true if the matrix of which *this is the QR decomposition represents a surjective - * linear map; false otherwise. - * - * \note This method has to determine which pivots should be considered nonzero. - * For that, it uses the threshold value that you can control by calling - * setThreshold(const RealScalar&). - */ - inline bool isSurjective() const - { - eigen_assert(m_isInitialized && "FullPivHouseholderQR is not initialized."); - return rank() == rows(); - } + /** \returns true if the matrix of which *this is the QR decomposition represents an injective + * linear map, i.e. has trivial kernel; false otherwise. + * + * \note This method has to determine which pivots should be considered nonzero. + * For that, it uses the threshold value that you can control by calling + * setThreshold(const RealScalar&). + */ + inline bool isInjective() const + { + eigen_assert(m_isInitialized && "FullPivHouseholderQR is not initialized."); + return rank() == cols(); + } - /** \returns true if the matrix of which *this is the QR decomposition is invertible. - * - * \note This method has to determine which pivots should be considered nonzero. - * For that, it uses the threshold value that you can control by calling - * setThreshold(const RealScalar&). - */ - inline bool isInvertible() const - { - eigen_assert(m_isInitialized && "FullPivHouseholderQR is not initialized."); - return isInjective() && isSurjective(); - } + /** \returns true if the matrix of which *this is the QR decomposition represents a surjective + * linear map; false otherwise. + * + * \note This method has to determine which pivots should be considered nonzero. + * For that, it uses the threshold value that you can control by calling + * setThreshold(const RealScalar&). + */ + inline bool isSurjective() const + { + eigen_assert(m_isInitialized && "FullPivHouseholderQR is not initialized."); + return rank() == rows(); + } - /** \returns the inverse of the matrix of which *this is the QR decomposition. - * - * \note If this matrix is not invertible, the returned matrix has undefined coefficients. - * Use isInvertible() to first determine whether this matrix is invertible. - */ - inline const Inverse inverse() const - { - eigen_assert(m_isInitialized && "FullPivHouseholderQR is not initialized."); - return Inverse(*this); - } + /** \returns true if the matrix of which *this is the QR decomposition is invertible. + * + * \note This method has to determine which pivots should be considered nonzero. + * For that, it uses the threshold value that you can control by calling + * setThreshold(const RealScalar&). + */ + inline bool isInvertible() const + { + eigen_assert(m_isInitialized && "FullPivHouseholderQR is not initialized."); + return isInjective() && isSurjective(); + } - inline Index rows() const { return m_qr.rows(); } - inline Index cols() const { return m_qr.cols(); } - - /** \returns a const reference to the vector of Householder coefficients used to represent the factor \c Q. - * - * For advanced uses only. - */ - const HCoeffsType& hCoeffs() const { return m_hCoeffs; } - - /** Allows to prescribe a threshold to be used by certain methods, such as rank(), - * who need to determine when pivots are to be considered nonzero. This is not used for the - * QR decomposition itself. - * - * When it needs to get the threshold value, Eigen calls threshold(). By default, this - * uses a formula to automatically determine a reasonable threshold. - * Once you have called the present method setThreshold(const RealScalar&), - * your value is used instead. - * - * \param threshold The new value to use as the threshold. - * - * A pivot will be considered nonzero if its absolute value is strictly greater than - * \f$ \vert pivot \vert \leqslant threshold \times \vert maxpivot \vert \f$ - * where maxpivot is the biggest pivot. - * - * If you want to come back to the default behavior, call setThreshold(Default_t) - */ - FullPivHouseholderQR& setThreshold(const RealScalar& threshold) - { - m_usePrescribedThreshold = true; - m_prescribedThreshold = threshold; - return *this; - } + /** \returns the inverse of the matrix of which *this is the QR decomposition. + * + * \note If this matrix is not invertible, the returned matrix has undefined coefficients. + * Use isInvertible() to first determine whether this matrix is invertible. + */ + inline const Inverse inverse() const + { + eigen_assert(m_isInitialized && "FullPivHouseholderQR is not initialized."); + return Inverse(*this); + } - /** Allows to come back to the default behavior, letting Eigen use its default formula for - * determining the threshold. - * - * You should pass the special object Eigen::Default as parameter here. - * \code qr.setThreshold(Eigen::Default); \endcode - * - * See the documentation of setThreshold(const RealScalar&). - */ - FullPivHouseholderQR& setThreshold(Default_t) - { - m_usePrescribedThreshold = false; - return *this; - } + inline Index rows() const { return m_qr.rows(); } + inline Index cols() const { return m_qr.cols(); } + + /** \returns a const reference to the vector of Householder coefficients used to represent the factor \c Q. + * + * For advanced uses only. + */ + const HCoeffsType &hCoeffs() const { return m_hCoeffs; } + + /** Allows to prescribe a threshold to be used by certain methods, such as rank(), + * who need to determine when pivots are to be considered nonzero. This is not used for the + * QR decomposition itself. + * + * When it needs to get the threshold value, Eigen calls threshold(). By default, this + * uses a formula to automatically determine a reasonable threshold. + * Once you have called the present method setThreshold(const RealScalar&), + * your value is used instead. + * + * \param threshold The new value to use as the threshold. + * + * A pivot will be considered nonzero if its absolute value is strictly greater than + * \f$ \vert pivot \vert \leqslant threshold \times \vert maxpivot \vert \f$ + * where maxpivot is the biggest pivot. + * + * If you want to come back to the default behavior, call setThreshold(Default_t) + */ + FullPivHouseholderQR &setThreshold(const RealScalar &threshold) + { + m_usePrescribedThreshold = true; + m_prescribedThreshold = threshold; + return *this; + } - /** Returns the threshold that will be used by certain methods such as rank(). - * - * See the documentation of setThreshold(const RealScalar&). - */ - RealScalar threshold() const - { - eigen_assert(m_isInitialized || m_usePrescribedThreshold); - return m_usePrescribedThreshold ? m_prescribedThreshold - // this formula comes from experimenting (see "LU precision tuning" thread on the list) - // and turns out to be identical to Higham's formula used already in LDLt. - : NumTraits::epsilon() * RealScalar(m_qr.diagonalSize()); - } + /** Allows to come back to the default behavior, letting Eigen use its default formula for + * determining the threshold. + * + * You should pass the special object Eigen::Default as parameter here. + * \code qr.setThreshold(Eigen::Default); \endcode + * + * See the documentation of setThreshold(const RealScalar&). + */ + FullPivHouseholderQR &setThreshold(Default_t) + { + m_usePrescribedThreshold = false; + return *this; + } - /** \returns the number of nonzero pivots in the QR decomposition. - * Here nonzero is meant in the exact sense, not in a fuzzy sense. - * So that notion isn't really intrinsically interesting, but it is - * still useful when implementing algorithms. - * - * \sa rank() - */ - inline Index nonzeroPivots() const - { - eigen_assert(m_isInitialized && "LU is not initialized."); - return m_nonzero_pivots; - } + /** Returns the threshold that will be used by certain methods such as rank(). + * + * See the documentation of setThreshold(const RealScalar&). + */ + RealScalar threshold() const + { + eigen_assert(m_isInitialized || m_usePrescribedThreshold); + return m_usePrescribedThreshold ? m_prescribedThreshold + // this formula comes from experimenting (see "LU precision tuning" thread on the + // list) and turns out to be identical to Higham's formula used already in LDLt. + : NumTraits::epsilon() * RealScalar(m_qr.diagonalSize()); + } - /** \returns the absolute value of the biggest pivot, i.e. the biggest - * diagonal coefficient of U. - */ - RealScalar maxPivot() const { return m_maxpivot; } - - #ifndef EIGEN_PARSED_BY_DOXYGEN - template - EIGEN_DEVICE_FUNC - void _solve_impl(const RhsType &rhs, DstType &dst) const; - #endif + /** \returns the number of nonzero pivots in the QR decomposition. + * Here nonzero is meant in the exact sense, not in a fuzzy sense. + * So that notion isn't really intrinsically interesting, but it is + * still useful when implementing algorithms. + * + * \sa rank() + */ + inline Index nonzeroPivots() const + { + eigen_assert(m_isInitialized && "LU is not initialized."); + return m_nonzero_pivots; + } - protected: - - static void check_template_parameters() - { - EIGEN_STATIC_ASSERT_NON_INTEGER(Scalar); - } - - void computeInPlace(); - - MatrixType m_qr; - HCoeffsType m_hCoeffs; - IntDiagSizeVectorType m_rows_transpositions; - IntDiagSizeVectorType m_cols_transpositions; - PermutationType m_cols_permutation; - RowVectorType m_temp; - bool m_isInitialized, m_usePrescribedThreshold; - RealScalar m_prescribedThreshold, m_maxpivot; - Index m_nonzero_pivots; - RealScalar m_precision; - Index m_det_pq; + /** \returns the absolute value of the biggest pivot, i.e. the biggest + * diagonal coefficient of U. + */ + RealScalar maxPivot() const { return m_maxpivot; } + +#ifndef EIGEN_PARSED_BY_DOXYGEN + template + EIGEN_DEVICE_FUNC void _solve_impl(const RhsType &rhs, DstType &dst) const; +#endif + +protected: + static void check_template_parameters() { EIGEN_STATIC_ASSERT_NON_INTEGER(Scalar); } + + void computeInPlace(); + + MatrixType m_qr; + HCoeffsType m_hCoeffs; + IntDiagSizeVectorType m_rows_transpositions; + IntDiagSizeVectorType m_cols_transpositions; + PermutationType m_cols_permutation; + RowVectorType m_temp; + bool m_isInitialized, m_usePrescribedThreshold; + RealScalar m_prescribedThreshold, m_maxpivot; + Index m_nonzero_pivots; + RealScalar m_precision; + Index m_det_pq; }; -template -typename MatrixType::RealScalar FullPivHouseholderQR::absDeterminant() const +template typename MatrixType::RealScalar FullPivHouseholderQR::absDeterminant() const { using std::abs; eigen_assert(m_isInitialized && "FullPivHouseholderQR is not initialized."); @@ -439,31 +414,30 @@ typename MatrixType::RealScalar FullPivHouseholderQR::logAbsDetermin } /** Performs the QR factorization of the given matrix \a matrix. The result of - * the factorization is stored into \c *this, and a reference to \c *this - * is returned. - * - * \sa class FullPivHouseholderQR, FullPivHouseholderQR(const MatrixType&) - */ + * the factorization is stored into \c *this, and a reference to \c *this + * is returned. + * + * \sa class FullPivHouseholderQR, FullPivHouseholderQR(const MatrixType&) + */ template template -FullPivHouseholderQR& FullPivHouseholderQR::compute(const EigenBase& matrix) +FullPivHouseholderQR &FullPivHouseholderQR::compute(const EigenBase &matrix) { m_qr = matrix.derived(); computeInPlace(); return *this; } -template -void FullPivHouseholderQR::computeInPlace() +template void FullPivHouseholderQR::computeInPlace() { check_template_parameters(); using std::abs; Index rows = m_qr.rows(); Index cols = m_qr.cols(); - Index size = (std::min)(rows,cols); + Index size = (std::min)(rows, cols); + - m_hCoeffs.resize(size); m_temp.resize(cols); @@ -476,29 +450,27 @@ void FullPivHouseholderQR::computeInPlace() RealScalar biggest(0); - m_nonzero_pivots = size; // the generic case is that in which all pivots are nonzero (invertible case) + m_nonzero_pivots = size;// the generic case is that in which all pivots are nonzero (invertible case) m_maxpivot = RealScalar(0); - for (Index k = 0; k < size; ++k) - { + for (Index k = 0; k < size; ++k) { Index row_of_biggest_in_corner, col_of_biggest_in_corner; typedef internal::scalar_score_coeff_op Scoring; typedef typename Scoring::result_type Score; - Score score = m_qr.bottomRightCorner(rows-k, cols-k) - .unaryExpr(Scoring()) - .maxCoeff(&row_of_biggest_in_corner, &col_of_biggest_in_corner); + Score score = m_qr.bottomRightCorner(rows - k, cols - k) + .unaryExpr(Scoring()) + .maxCoeff(&row_of_biggest_in_corner, &col_of_biggest_in_corner); row_of_biggest_in_corner += k; col_of_biggest_in_corner += k; - RealScalar biggest_in_corner = internal::abs_knowing_score()(m_qr(row_of_biggest_in_corner, col_of_biggest_in_corner), score); - if(k==0) biggest = biggest_in_corner; + RealScalar biggest_in_corner = + internal::abs_knowing_score()(m_qr(row_of_biggest_in_corner, col_of_biggest_in_corner), score); + if (k == 0) biggest = biggest_in_corner; // if the corner is negligible, then we have less than full rank, and we can finish early - if(internal::isMuchSmallerThan(biggest_in_corner, biggest, m_precision)) - { + if (internal::isMuchSmallerThan(biggest_in_corner, biggest, m_precision)) { m_nonzero_pivots = k; - for(Index i = k; i < size; i++) - { + for (Index i = k; i < size; i++) { m_rows_transpositions.coeffRef(i) = i; m_cols_transpositions.coeffRef(i) = i; m_hCoeffs.coeffRef(i) = Scalar(0); @@ -508,31 +480,30 @@ void FullPivHouseholderQR::computeInPlace() m_rows_transpositions.coeffRef(k) = row_of_biggest_in_corner; m_cols_transpositions.coeffRef(k) = col_of_biggest_in_corner; - if(k != row_of_biggest_in_corner) { - m_qr.row(k).tail(cols-k).swap(m_qr.row(row_of_biggest_in_corner).tail(cols-k)); + if (k != row_of_biggest_in_corner) { + m_qr.row(k).tail(cols - k).swap(m_qr.row(row_of_biggest_in_corner).tail(cols - k)); ++number_of_transpositions; } - if(k != col_of_biggest_in_corner) { + if (k != col_of_biggest_in_corner) { m_qr.col(k).swap(m_qr.col(col_of_biggest_in_corner)); ++number_of_transpositions; } RealScalar beta; - m_qr.col(k).tail(rows-k).makeHouseholderInPlace(m_hCoeffs.coeffRef(k), beta); - m_qr.coeffRef(k,k) = beta; + m_qr.col(k).tail(rows - k).makeHouseholderInPlace(m_hCoeffs.coeffRef(k), beta); + m_qr.coeffRef(k, k) = beta; // remember the maximum absolute value of diagonal coefficients - if(abs(beta) > m_maxpivot) m_maxpivot = abs(beta); + if (abs(beta) > m_maxpivot) m_maxpivot = abs(beta); - m_qr.bottomRightCorner(rows-k, cols-k-1) - .applyHouseholderOnTheLeft(m_qr.col(k).tail(rows-k-1), m_hCoeffs.coeffRef(k), &m_temp.coeffRef(k+1)); + m_qr.bottomRightCorner(rows - k, cols - k - 1) + .applyHouseholderOnTheLeft(m_qr.col(k).tail(rows - k - 1), m_hCoeffs.coeffRef(k), &m_temp.coeffRef(k + 1)); } m_cols_permutation.setIdentity(cols); - for(Index k = 0; k < size; ++k) - m_cols_permutation.applyTranspositionOnTheRight(k, m_cols_transpositions.coeff(k)); + for (Index k = 0; k < size; ++k) m_cols_permutation.applyTranspositionOnTheRight(k, m_cols_transpositions.coeff(k)); - m_det_pq = (number_of_transpositions%2) ? -1 : 1; + m_det_pq = (number_of_transpositions % 2) ? -1 : 1; m_isInitialized = true; } @@ -546,112 +517,112 @@ void FullPivHouseholderQR<_MatrixType>::_solve_impl(const RhsType &rhs, DstType // FIXME introduce nonzeroPivots() and use it here. and more generally, // make the same improvements in this dec as in FullPivLU. - if(l_rank==0) - { + if (l_rank == 0) { dst.setZero(); return; } typename RhsType::PlainObject c(rhs); - Matrix temp(rhs.cols()); - for (Index k = 0; k < l_rank; ++k) - { - Index remainingSize = rows()-k; + Matrix temp(rhs.cols()); + for (Index k = 0; k < l_rank; ++k) { + Index remainingSize = rows() - k; c.row(k).swap(c.row(m_rows_transpositions.coeff(k))); c.bottomRightCorner(remainingSize, rhs.cols()) - .applyHouseholderOnTheLeft(m_qr.col(k).tail(remainingSize-1), - m_hCoeffs.coeff(k), &temp.coeffRef(0)); + .applyHouseholderOnTheLeft(m_qr.col(k).tail(remainingSize - 1), m_hCoeffs.coeff(k), &temp.coeffRef(0)); } - m_qr.topLeftCorner(l_rank, l_rank) - .template triangularView() - .solveInPlace(c.topRows(l_rank)); + m_qr.topLeftCorner(l_rank, l_rank).template triangularView().solveInPlace(c.topRows(l_rank)); - for(Index i = 0; i < l_rank; ++i) dst.row(m_cols_permutation.indices().coeff(i)) = c.row(i); - for(Index i = l_rank; i < cols(); ++i) dst.row(m_cols_permutation.indices().coeff(i)).setZero(); + for (Index i = 0; i < l_rank; ++i) dst.row(m_cols_permutation.indices().coeff(i)) = c.row(i); + for (Index i = l_rank; i < cols(); ++i) dst.row(m_cols_permutation.indices().coeff(i)).setZero(); } #endif namespace internal { - -template -struct Assignment >, internal::assign_op::Scalar>, Dense2Dense> -{ - typedef FullPivHouseholderQR QrType; - typedef Inverse SrcXprType; - static void run(DstXprType &dst, const SrcXprType &src, const internal::assign_op &) - { - dst = src.nestedExpression().solve(MatrixType::Identity(src.rows(), src.cols())); - } -}; - -/** \ingroup QR_Module - * - * \brief Expression type for return value of FullPivHouseholderQR::matrixQ() - * - * \tparam MatrixType type of underlying dense matrix - */ -template struct FullPivHouseholderQRMatrixQReturnType - : public ReturnByValue > -{ -public: - typedef typename FullPivHouseholderQR::IntDiagSizeVectorType IntDiagSizeVectorType; - typedef typename internal::plain_diag_type::type HCoeffsType; - typedef Matrix WorkVectorType; - - FullPivHouseholderQRMatrixQReturnType(const MatrixType& qr, - const HCoeffsType& hCoeffs, - const IntDiagSizeVectorType& rowsTranspositions) - : m_qr(qr), - m_hCoeffs(hCoeffs), - m_rowsTranspositions(rowsTranspositions) - {} - template - void evalTo(ResultType& result) const + template + struct Assignment>, + internal::assign_op::Scalar>, + Dense2Dense> { - const Index rows = m_qr.rows(); - WorkVectorType workspace(rows); - evalTo(result, workspace); - } - - template - void evalTo(ResultType& result, WorkVectorType& workspace) const + typedef FullPivHouseholderQR QrType; + typedef Inverse SrcXprType; + static void run(DstXprType &dst, + const SrcXprType &src, + const internal::assign_op &) + { + dst = src.nestedExpression().solve(MatrixType::Identity(src.rows(), src.cols())); + } + }; + + /** \ingroup QR_Module + * + * \brief Expression type for return value of FullPivHouseholderQR::matrixQ() + * + * \tparam MatrixType type of underlying dense matrix + */ + template + struct FullPivHouseholderQRMatrixQReturnType : public ReturnByValue> { - using numext::conj; - // compute the product H'_0 H'_1 ... H'_n-1, - // where H_k is the k-th Householder transformation I - h_k v_k v_k' - // and v_k is the k-th Householder vector [1,m_qr(k+1,k), m_qr(k+2,k), ...] - const Index rows = m_qr.rows(); - const Index cols = m_qr.cols(); - const Index size = (std::min)(rows, cols); - workspace.resize(rows); - result.setIdentity(rows, rows); - for (Index k = size-1; k >= 0; k--) + public: + typedef typename FullPivHouseholderQR::IntDiagSizeVectorType IntDiagSizeVectorType; + typedef typename internal::plain_diag_type::type HCoeffsType; + typedef Matrix + WorkVectorType; + + FullPivHouseholderQRMatrixQReturnType(const MatrixType &qr, + const HCoeffsType &hCoeffs, + const IntDiagSizeVectorType &rowsTranspositions) + : m_qr(qr), m_hCoeffs(hCoeffs), m_rowsTranspositions(rowsTranspositions) + {} + + template void evalTo(ResultType &result) const { - result.block(k, k, rows-k, rows-k) - .applyHouseholderOnTheLeft(m_qr.col(k).tail(rows-k-1), conj(m_hCoeffs.coeff(k)), &workspace.coeffRef(k)); - result.row(k).swap(result.row(m_rowsTranspositions.coeff(k))); + const Index rows = m_qr.rows(); + WorkVectorType workspace(rows); + evalTo(result, workspace); } - } - Index rows() const { return m_qr.rows(); } - Index cols() const { return m_qr.rows(); } + template void evalTo(ResultType &result, WorkVectorType &workspace) const + { + using numext::conj; + // compute the product H'_0 H'_1 ... H'_n-1, + // where H_k is the k-th Householder transformation I - h_k v_k v_k' + // and v_k is the k-th Householder vector [1,m_qr(k+1,k), m_qr(k+2,k), ...] + const Index rows = m_qr.rows(); + const Index cols = m_qr.cols(); + const Index size = (std::min)(rows, cols); + workspace.resize(rows); + result.setIdentity(rows, rows); + for (Index k = size - 1; k >= 0; k--) { + result.block(k, k, rows - k, rows - k) + .applyHouseholderOnTheLeft(m_qr.col(k).tail(rows - k - 1), conj(m_hCoeffs.coeff(k)), &workspace.coeffRef(k)); + result.row(k).swap(result.row(m_rowsTranspositions.coeff(k))); + } + } -protected: - typename MatrixType::Nested m_qr; - typename HCoeffsType::Nested m_hCoeffs; - typename IntDiagSizeVectorType::Nested m_rowsTranspositions; -}; + Index rows() const { return m_qr.rows(); } + Index cols() const { return m_qr.rows(); } + + protected: + typename MatrixType::Nested m_qr; + typename HCoeffsType::Nested m_hCoeffs; + typename IntDiagSizeVectorType::Nested m_rowsTranspositions; + }; -// template -// struct evaluator > -// : public evaluator > > -// {}; + // template + // struct evaluator > + // : public evaluator > > + // {}; -} // end namespace internal +}// end namespace internal template inline typename FullPivHouseholderQR::MatrixQReturnType FullPivHouseholderQR::matrixQ() const @@ -661,16 +632,15 @@ inline typename FullPivHouseholderQR::MatrixQReturnType FullPivHouse } /** \return the full-pivoting Householder QR decomposition of \c *this. - * - * \sa class FullPivHouseholderQR - */ + * + * \sa class FullPivHouseholderQR + */ template -const FullPivHouseholderQR::PlainObject> -MatrixBase::fullPivHouseholderQr() const +const FullPivHouseholderQR::PlainObject> MatrixBase::fullPivHouseholderQr() const { return FullPivHouseholderQR(eval()); } -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_FULLPIVOTINGHOUSEHOLDERQR_H +#endif// EIGEN_FULLPIVOTINGHOUSEHOLDERQR_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/QR/HouseholderQR.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/QR/HouseholderQR.h index 3513d995..43fc6a9b 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/QR/HouseholderQR.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/QR/HouseholderQR.h @@ -12,228 +12,222 @@ #ifndef EIGEN_QR_H #define EIGEN_QR_H -namespace Eigen { +namespace Eigen { /** \ingroup QR_Module - * - * - * \class HouseholderQR - * - * \brief Householder QR decomposition of a matrix - * - * \tparam _MatrixType the type of the matrix of which we are computing the QR decomposition - * - * This class performs a QR decomposition of a matrix \b A into matrices \b Q and \b R - * such that - * \f[ - * \mathbf{A} = \mathbf{Q} \, \mathbf{R} - * \f] - * by using Householder transformations. Here, \b Q a unitary matrix and \b R an upper triangular matrix. - * The result is stored in a compact way compatible with LAPACK. - * - * Note that no pivoting is performed. This is \b not a rank-revealing decomposition. - * If you want that feature, use FullPivHouseholderQR or ColPivHouseholderQR instead. - * - * This Householder QR decomposition is faster, but less numerically stable and less feature-full than - * FullPivHouseholderQR or ColPivHouseholderQR. - * - * This class supports the \link InplaceDecomposition inplace decomposition \endlink mechanism. - * - * \sa MatrixBase::householderQr() - */ + * + * + * \class HouseholderQR + * + * \brief Householder QR decomposition of a matrix + * + * \tparam _MatrixType the type of the matrix of which we are computing the QR decomposition + * + * This class performs a QR decomposition of a matrix \b A into matrices \b Q and \b R + * such that + * \f[ + * \mathbf{A} = \mathbf{Q} \, \mathbf{R} + * \f] + * by using Householder transformations. Here, \b Q a unitary matrix and \b R an upper triangular matrix. + * The result is stored in a compact way compatible with LAPACK. + * + * Note that no pivoting is performed. This is \b not a rank-revealing decomposition. + * If you want that feature, use FullPivHouseholderQR or ColPivHouseholderQR instead. + * + * This Householder QR decomposition is faster, but less numerically stable and less feature-full than + * FullPivHouseholderQR or ColPivHouseholderQR. + * + * This class supports the \link InplaceDecomposition inplace decomposition \endlink mechanism. + * + * \sa MatrixBase::householderQr() + */ template class HouseholderQR { - public: - - typedef _MatrixType MatrixType; - enum { - RowsAtCompileTime = MatrixType::RowsAtCompileTime, - ColsAtCompileTime = MatrixType::ColsAtCompileTime, - MaxRowsAtCompileTime = MatrixType::MaxRowsAtCompileTime, - MaxColsAtCompileTime = MatrixType::MaxColsAtCompileTime - }; - typedef typename MatrixType::Scalar Scalar; - typedef typename MatrixType::RealScalar RealScalar; - // FIXME should be int - typedef typename MatrixType::StorageIndex StorageIndex; - typedef Matrix MatrixQType; - typedef typename internal::plain_diag_type::type HCoeffsType; - typedef typename internal::plain_row_type::type RowVectorType; - typedef HouseholderSequence::type> HouseholderSequenceType; - - /** - * \brief Default Constructor. - * - * The default constructor is useful in cases in which the user intends to - * perform decompositions via HouseholderQR::compute(const MatrixType&). - */ - HouseholderQR() : m_qr(), m_hCoeffs(), m_temp(), m_isInitialized(false) {} - - /** \brief Default Constructor with memory preallocation - * - * Like the default constructor but with preallocation of the internal data - * according to the specified problem \a size. - * \sa HouseholderQR() - */ - HouseholderQR(Index rows, Index cols) - : m_qr(rows, cols), - m_hCoeffs((std::min)(rows,cols)), - m_temp(cols), - m_isInitialized(false) {} - - /** \brief Constructs a QR factorization from a given matrix - * - * This constructor computes the QR factorization of the matrix \a matrix by calling - * the method compute(). It is a short cut for: - * - * \code - * HouseholderQR qr(matrix.rows(), matrix.cols()); - * qr.compute(matrix); - * \endcode - * - * \sa compute() - */ - template - explicit HouseholderQR(const EigenBase& matrix) - : m_qr(matrix.rows(), matrix.cols()), - m_hCoeffs((std::min)(matrix.rows(),matrix.cols())), - m_temp(matrix.cols()), - m_isInitialized(false) - { - compute(matrix.derived()); - } +public: + typedef _MatrixType MatrixType; + enum { + RowsAtCompileTime = MatrixType::RowsAtCompileTime, + ColsAtCompileTime = MatrixType::ColsAtCompileTime, + MaxRowsAtCompileTime = MatrixType::MaxRowsAtCompileTime, + MaxColsAtCompileTime = MatrixType::MaxColsAtCompileTime + }; + typedef typename MatrixType::Scalar Scalar; + typedef typename MatrixType::RealScalar RealScalar; + // FIXME should be int + typedef typename MatrixType::StorageIndex StorageIndex; + typedef Matrix + MatrixQType; + typedef typename internal::plain_diag_type::type HCoeffsType; + typedef typename internal::plain_row_type::type RowVectorType; + typedef HouseholderSequence::type> + HouseholderSequenceType; + + /** + * \brief Default Constructor. + * + * The default constructor is useful in cases in which the user intends to + * perform decompositions via HouseholderQR::compute(const MatrixType&). + */ + HouseholderQR() : m_qr(), m_hCoeffs(), m_temp(), m_isInitialized(false) {} + + /** \brief Default Constructor with memory preallocation + * + * Like the default constructor but with preallocation of the internal data + * according to the specified problem \a size. + * \sa HouseholderQR() + */ + HouseholderQR(Index rows, Index cols) + : m_qr(rows, cols), m_hCoeffs((std::min)(rows, cols)), m_temp(cols), m_isInitialized(false) + {} + + /** \brief Constructs a QR factorization from a given matrix + * + * This constructor computes the QR factorization of the matrix \a matrix by calling + * the method compute(). It is a short cut for: + * + * \code + * HouseholderQR qr(matrix.rows(), matrix.cols()); + * qr.compute(matrix); + * \endcode + * + * \sa compute() + */ + template + explicit HouseholderQR(const EigenBase &matrix) + : m_qr(matrix.rows(), matrix.cols()), m_hCoeffs((std::min)(matrix.rows(), matrix.cols())), m_temp(matrix.cols()), + m_isInitialized(false) + { + compute(matrix.derived()); + } - /** \brief Constructs a QR factorization from a given matrix - * - * This overloaded constructor is provided for \link InplaceDecomposition inplace decomposition \endlink when - * \c MatrixType is a Eigen::Ref. - * - * \sa HouseholderQR(const EigenBase&) - */ - template - explicit HouseholderQR(EigenBase& matrix) - : m_qr(matrix.derived()), - m_hCoeffs((std::min)(matrix.rows(),matrix.cols())), - m_temp(matrix.cols()), - m_isInitialized(false) - { - computeInPlace(); - } + /** \brief Constructs a QR factorization from a given matrix + * + * This overloaded constructor is provided for \link InplaceDecomposition inplace decomposition \endlink when + * \c MatrixType is a Eigen::Ref. + * + * \sa HouseholderQR(const EigenBase&) + */ + template + explicit HouseholderQR(EigenBase &matrix) + : m_qr(matrix.derived()), m_hCoeffs((std::min)(matrix.rows(), matrix.cols())), m_temp(matrix.cols()), + m_isInitialized(false) + { + computeInPlace(); + } - /** This method finds a solution x to the equation Ax=b, where A is the matrix of which - * *this is the QR decomposition, if any exists. - * - * \param b the right-hand-side of the equation to solve. - * - * \returns a solution. - * - * \note_about_checking_solutions - * - * \note_about_arbitrary_choice_of_solution - * - * Example: \include HouseholderQR_solve.cpp - * Output: \verbinclude HouseholderQR_solve.out - */ - template - inline const Solve - solve(const MatrixBase& b) const - { - eigen_assert(m_isInitialized && "HouseholderQR is not initialized."); - return Solve(*this, b.derived()); - } + /** This method finds a solution x to the equation Ax=b, where A is the matrix of which + * *this is the QR decomposition, if any exists. + * + * \param b the right-hand-side of the equation to solve. + * + * \returns a solution. + * + * \note_about_checking_solutions + * + * \note_about_arbitrary_choice_of_solution + * + * Example: \include HouseholderQR_solve.cpp + * Output: \verbinclude HouseholderQR_solve.out + */ + template inline const Solve solve(const MatrixBase &b) const + { + eigen_assert(m_isInitialized && "HouseholderQR is not initialized."); + return Solve(*this, b.derived()); + } - /** This method returns an expression of the unitary matrix Q as a sequence of Householder transformations. - * - * The returned expression can directly be used to perform matrix products. It can also be assigned to a dense Matrix object. - * Here is an example showing how to recover the full or thin matrix Q, as well as how to perform matrix products using operator*: - * - * Example: \include HouseholderQR_householderQ.cpp - * Output: \verbinclude HouseholderQR_householderQ.out - */ - HouseholderSequenceType householderQ() const - { - eigen_assert(m_isInitialized && "HouseholderQR is not initialized."); - return HouseholderSequenceType(m_qr, m_hCoeffs.conjugate()); - } + /** This method returns an expression of the unitary matrix Q as a sequence of Householder transformations. + * + * The returned expression can directly be used to perform matrix products. It can also be assigned to a dense Matrix + * object. Here is an example showing how to recover the full or thin matrix Q, as well as how to perform matrix + * products using operator*: + * + * Example: \include HouseholderQR_householderQ.cpp + * Output: \verbinclude HouseholderQR_householderQ.out + */ + HouseholderSequenceType householderQ() const + { + eigen_assert(m_isInitialized && "HouseholderQR is not initialized."); + return HouseholderSequenceType(m_qr, m_hCoeffs.conjugate()); + } - /** \returns a reference to the matrix where the Householder QR decomposition is stored - * in a LAPACK-compatible way. - */ - const MatrixType& matrixQR() const - { - eigen_assert(m_isInitialized && "HouseholderQR is not initialized."); - return m_qr; - } + /** \returns a reference to the matrix where the Householder QR decomposition is stored + * in a LAPACK-compatible way. + */ + const MatrixType &matrixQR() const + { + eigen_assert(m_isInitialized && "HouseholderQR is not initialized."); + return m_qr; + } - template - HouseholderQR& compute(const EigenBase& matrix) { - m_qr = matrix.derived(); - computeInPlace(); - return *this; - } + template HouseholderQR &compute(const EigenBase &matrix) + { + m_qr = matrix.derived(); + computeInPlace(); + return *this; + } - /** \returns the absolute value of the determinant of the matrix of which - * *this is the QR decomposition. It has only linear complexity - * (that is, O(n) where n is the dimension of the square matrix) - * as the QR decomposition has already been computed. - * - * \note This is only for square matrices. - * - * \warning a determinant can be very big or small, so for matrices - * of large enough dimension, there is a risk of overflow/underflow. - * One way to work around that is to use logAbsDeterminant() instead. - * - * \sa logAbsDeterminant(), MatrixBase::determinant() - */ - typename MatrixType::RealScalar absDeterminant() const; - - /** \returns the natural log of the absolute value of the determinant of the matrix of which - * *this is the QR decomposition. It has only linear complexity - * (that is, O(n) where n is the dimension of the square matrix) - * as the QR decomposition has already been computed. - * - * \note This is only for square matrices. - * - * \note This method is useful to work around the risk of overflow/underflow that's inherent - * to determinant computation. - * - * \sa absDeterminant(), MatrixBase::determinant() - */ - typename MatrixType::RealScalar logAbsDeterminant() const; - - inline Index rows() const { return m_qr.rows(); } - inline Index cols() const { return m_qr.cols(); } - - /** \returns a const reference to the vector of Householder coefficients used to represent the factor \c Q. - * - * For advanced uses only. - */ - const HCoeffsType& hCoeffs() const { return m_hCoeffs; } - - #ifndef EIGEN_PARSED_BY_DOXYGEN - template - EIGEN_DEVICE_FUNC - void _solve_impl(const RhsType &rhs, DstType &dst) const; - #endif - - protected: - - static void check_template_parameters() - { - EIGEN_STATIC_ASSERT_NON_INTEGER(Scalar); - } + /** \returns the absolute value of the determinant of the matrix of which + * *this is the QR decomposition. It has only linear complexity + * (that is, O(n) where n is the dimension of the square matrix) + * as the QR decomposition has already been computed. + * + * \note This is only for square matrices. + * + * \warning a determinant can be very big or small, so for matrices + * of large enough dimension, there is a risk of overflow/underflow. + * One way to work around that is to use logAbsDeterminant() instead. + * + * \sa logAbsDeterminant(), MatrixBase::determinant() + */ + typename MatrixType::RealScalar absDeterminant() const; + + /** \returns the natural log of the absolute value of the determinant of the matrix of which + * *this is the QR decomposition. It has only linear complexity + * (that is, O(n) where n is the dimension of the square matrix) + * as the QR decomposition has already been computed. + * + * \note This is only for square matrices. + * + * \note This method is useful to work around the risk of overflow/underflow that's inherent + * to determinant computation. + * + * \sa absDeterminant(), MatrixBase::determinant() + */ + typename MatrixType::RealScalar logAbsDeterminant() const; + + inline Index rows() const { return m_qr.rows(); } + inline Index cols() const { return m_qr.cols(); } + + /** \returns a const reference to the vector of Householder coefficients used to represent the factor \c Q. + * + * For advanced uses only. + */ + const HCoeffsType &hCoeffs() const { return m_hCoeffs; } + +#ifndef EIGEN_PARSED_BY_DOXYGEN + template + EIGEN_DEVICE_FUNC void _solve_impl(const RhsType &rhs, DstType &dst) const; +#endif - void computeInPlace(); - - MatrixType m_qr; - HCoeffsType m_hCoeffs; - RowVectorType m_temp; - bool m_isInitialized; +protected: + static void check_template_parameters() { EIGEN_STATIC_ASSERT_NON_INTEGER(Scalar); } + + void computeInPlace(); + + MatrixType m_qr; + HCoeffsType m_hCoeffs; + RowVectorType m_temp; + bool m_isInitialized; }; -template -typename MatrixType::RealScalar HouseholderQR::absDeterminant() const +template typename MatrixType::RealScalar HouseholderQR::absDeterminant() const { using std::abs; eigen_assert(m_isInitialized && "HouseholderQR is not initialized."); @@ -241,8 +235,7 @@ typename MatrixType::RealScalar HouseholderQR::absDeterminant() cons return abs(m_qr.diagonal().prod()); } -template -typename MatrixType::RealScalar HouseholderQR::logAbsDeterminant() const +template typename MatrixType::RealScalar HouseholderQR::logAbsDeterminant() const { eigen_assert(m_isInitialized && "HouseholderQR is not initialized."); eigen_assert(m_qr.rows() == m_qr.cols() && "You can't take the determinant of a non-square matrix!"); @@ -251,98 +244,93 @@ typename MatrixType::RealScalar HouseholderQR::logAbsDeterminant() c namespace internal { -/** \internal */ -template -void householder_qr_inplace_unblocked(MatrixQR& mat, HCoeffs& hCoeffs, typename MatrixQR::Scalar* tempData = 0) -{ - typedef typename MatrixQR::Scalar Scalar; - typedef typename MatrixQR::RealScalar RealScalar; - Index rows = mat.rows(); - Index cols = mat.cols(); - Index size = (std::min)(rows,cols); - - eigen_assert(hCoeffs.size() == size); - - typedef Matrix TempType; - TempType tempVector; - if(tempData==0) - { - tempVector.resize(cols); - tempData = tempVector.data(); - } - - for(Index k = 0; k < size; ++k) - { - Index remainingRows = rows - k; - Index remainingCols = cols - k - 1; - - RealScalar beta; - mat.col(k).tail(remainingRows).makeHouseholderInPlace(hCoeffs.coeffRef(k), beta); - mat.coeffRef(k,k) = beta; - - // apply H to remaining part of m_qr from the left - mat.bottomRightCorner(remainingRows, remainingCols) - .applyHouseholderOnTheLeft(mat.col(k).tail(remainingRows-1), hCoeffs.coeffRef(k), tempData+k+1); - } -} - -/** \internal */ -template -struct householder_qr_inplace_blocked -{ - // This is specialized for MKL-supported Scalar types in HouseholderQR_MKL.h - static void run(MatrixQR& mat, HCoeffs& hCoeffs, Index maxBlockSize=32, - typename MatrixQR::Scalar* tempData = 0) + /** \internal */ + template + void householder_qr_inplace_unblocked(MatrixQR &mat, HCoeffs &hCoeffs, typename MatrixQR::Scalar *tempData = 0) { typedef typename MatrixQR::Scalar Scalar; - typedef Block BlockType; - + typedef typename MatrixQR::RealScalar RealScalar; Index rows = mat.rows(); Index cols = mat.cols(); Index size = (std::min)(rows, cols); - typedef Matrix TempType; + eigen_assert(hCoeffs.size() == size); + + typedef Matrix TempType; TempType tempVector; - if(tempData==0) - { + if (tempData == 0) { tempVector.resize(cols); tempData = tempVector.data(); } - Index blockSize = (std::min)(maxBlockSize,size); + for (Index k = 0; k < size; ++k) { + Index remainingRows = rows - k; + Index remainingCols = cols - k - 1; - Index k = 0; - for (k = 0; k < size; k += blockSize) + RealScalar beta; + mat.col(k).tail(remainingRows).makeHouseholderInPlace(hCoeffs.coeffRef(k), beta); + mat.coeffRef(k, k) = beta; + + // apply H to remaining part of m_qr from the left + mat.bottomRightCorner(remainingRows, remainingCols) + .applyHouseholderOnTheLeft(mat.col(k).tail(remainingRows - 1), hCoeffs.coeffRef(k), tempData + k + 1); + } + } + + /** \internal */ + template + struct householder_qr_inplace_blocked + { + // This is specialized for MKL-supported Scalar types in HouseholderQR_MKL.h + static void run(MatrixQR &mat, HCoeffs &hCoeffs, Index maxBlockSize = 32, typename MatrixQR::Scalar *tempData = 0) { - Index bs = (std::min)(size-k,blockSize); // actual size of the block - Index tcols = cols - k - bs; // trailing columns - Index brows = rows-k; // rows of the block - - // partition the matrix: - // A00 | A01 | A02 - // mat = A10 | A11 | A12 - // A20 | A21 | A22 - // and performs the qr dec of [A11^T A12^T]^T - // and update [A21^T A22^T]^T using level 3 operations. - // Finally, the algorithm continue on A22 - - BlockType A11_21 = mat.block(k,k,brows,bs); - Block hCoeffsSegment = hCoeffs.segment(k,bs); - - householder_qr_inplace_unblocked(A11_21, hCoeffsSegment, tempData); - - if(tcols) - { - BlockType A21_22 = mat.block(k,k+bs,brows,tcols); - apply_block_householder_on_the_left(A21_22,A11_21,hCoeffsSegment, false); // false == backward + typedef typename MatrixQR::Scalar Scalar; + typedef Block BlockType; + + Index rows = mat.rows(); + Index cols = mat.cols(); + Index size = (std::min)(rows, cols); + + typedef Matrix TempType; + TempType tempVector; + if (tempData == 0) { + tempVector.resize(cols); + tempData = tempVector.data(); + } + + Index blockSize = (std::min)(maxBlockSize, size); + + Index k = 0; + for (k = 0; k < size; k += blockSize) { + Index bs = (std::min)(size - k, blockSize);// actual size of the block + Index tcols = cols - k - bs;// trailing columns + Index brows = rows - k;// rows of the block + + // partition the matrix: + // A00 | A01 | A02 + // mat = A10 | A11 | A12 + // A20 | A21 | A22 + // and performs the qr dec of [A11^T A12^T]^T + // and update [A21^T A22^T]^T using level 3 operations. + // Finally, the algorithm continue on A22 + + BlockType A11_21 = mat.block(k, k, brows, bs); + Block hCoeffsSegment = hCoeffs.segment(k, bs); + + householder_qr_inplace_unblocked(A11_21, hCoeffsSegment, tempData); + + if (tcols) { + BlockType A21_22 = mat.block(k, k + bs, brows, tcols); + apply_block_householder_on_the_left(A21_22, A11_21, hCoeffsSegment, false);// false == backward + } } } - } -}; + }; -} // end namespace internal +}// end namespace internal #ifndef EIGEN_PARSED_BY_DOXYGEN template @@ -355,34 +343,28 @@ void HouseholderQR<_MatrixType>::_solve_impl(const RhsType &rhs, DstType &dst) c typename RhsType::PlainObject c(rhs); // Note that the matrix Q = H_0^* H_1^*... so its inverse is Q^* = (H_0 H_1 ...)^T - c.applyOnTheLeft(householderSequence( - m_qr.leftCols(rank), - m_hCoeffs.head(rank)).transpose() - ); + c.applyOnTheLeft(householderSequence(m_qr.leftCols(rank), m_hCoeffs.head(rank)).transpose()); - m_qr.topLeftCorner(rank, rank) - .template triangularView() - .solveInPlace(c.topRows(rank)); + m_qr.topLeftCorner(rank, rank).template triangularView().solveInPlace(c.topRows(rank)); dst.topRows(rank) = c.topRows(rank); - dst.bottomRows(cols()-rank).setZero(); + dst.bottomRows(cols() - rank).setZero(); } #endif /** Performs the QR factorization of the given matrix \a matrix. The result of - * the factorization is stored into \c *this, and a reference to \c *this - * is returned. - * - * \sa class HouseholderQR, HouseholderQR(const MatrixType&) - */ -template -void HouseholderQR::computeInPlace() + * the factorization is stored into \c *this, and a reference to \c *this + * is returned. + * + * \sa class HouseholderQR, HouseholderQR(const MatrixType&) + */ +template void HouseholderQR::computeInPlace() { check_template_parameters(); - + Index rows = m_qr.rows(); Index cols = m_qr.cols(); - Index size = (std::min)(rows,cols); + Index size = (std::min)(rows, cols); m_hCoeffs.resize(size); @@ -394,16 +376,15 @@ void HouseholderQR::computeInPlace() } /** \return the Householder QR decomposition of \c *this. - * - * \sa class HouseholderQR - */ + * + * \sa class HouseholderQR + */ template -const HouseholderQR::PlainObject> -MatrixBase::householderQr() const +const HouseholderQR::PlainObject> MatrixBase::householderQr() const { return HouseholderQR(eval()); } -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_QR_H +#endif// EIGEN_QR_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/QR/HouseholderQR_LAPACKE.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/QR/HouseholderQR_LAPACKE.h index 1dc7d536..2b4e190b 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/QR/HouseholderQR_LAPACKE.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/QR/HouseholderQR_LAPACKE.h @@ -34,35 +34,35 @@ #ifndef EIGEN_QR_LAPACKE_H #define EIGEN_QR_LAPACKE_H -namespace Eigen { +namespace Eigen { namespace internal { -/** \internal Specialization for the data types supported by LAPACKe */ + /** \internal Specialization for the data types supported by LAPACKe */ -#define EIGEN_LAPACKE_QR_NOPIV(EIGTYPE, LAPACKE_TYPE, LAPACKE_PREFIX) \ -template \ -struct householder_qr_inplace_blocked \ -{ \ - static void run(MatrixQR& mat, HCoeffs& hCoeffs, Index = 32, \ - typename MatrixQR::Scalar* = 0) \ - { \ - lapack_int m = (lapack_int) mat.rows(); \ - lapack_int n = (lapack_int) mat.cols(); \ - lapack_int lda = (lapack_int) mat.outerStride(); \ - lapack_int matrix_order = (MatrixQR::IsRowMajor) ? LAPACK_ROW_MAJOR : LAPACK_COL_MAJOR; \ - LAPACKE_##LAPACKE_PREFIX##geqrf( matrix_order, m, n, (LAPACKE_TYPE*)mat.data(), lda, (LAPACKE_TYPE*)hCoeffs.data()); \ - hCoeffs.adjointInPlace(); \ - } \ -}; +#define EIGEN_LAPACKE_QR_NOPIV(EIGTYPE, LAPACKE_TYPE, LAPACKE_PREFIX) \ + template \ + struct householder_qr_inplace_blocked \ + { \ + static void run(MatrixQR &mat, HCoeffs &hCoeffs, Index = 32, typename MatrixQR::Scalar * = 0) \ + { \ + lapack_int m = (lapack_int)mat.rows(); \ + lapack_int n = (lapack_int)mat.cols(); \ + lapack_int lda = (lapack_int)mat.outerStride(); \ + lapack_int matrix_order = (MatrixQR::IsRowMajor) ? LAPACK_ROW_MAJOR : LAPACK_COL_MAJOR; \ + LAPACKE_##LAPACKE_PREFIX##geqrf( \ + matrix_order, m, n, (LAPACKE_TYPE *)mat.data(), lda, (LAPACKE_TYPE *)hCoeffs.data()); \ + hCoeffs.adjointInPlace(); \ + } \ + }; -EIGEN_LAPACKE_QR_NOPIV(double, double, d) -EIGEN_LAPACKE_QR_NOPIV(float, float, s) -EIGEN_LAPACKE_QR_NOPIV(dcomplex, lapack_complex_double, z) -EIGEN_LAPACKE_QR_NOPIV(scomplex, lapack_complex_float, c) + EIGEN_LAPACKE_QR_NOPIV(double, double, d) + EIGEN_LAPACKE_QR_NOPIV(float, float, s) + EIGEN_LAPACKE_QR_NOPIV(dcomplex, lapack_complex_double, z) + EIGEN_LAPACKE_QR_NOPIV(scomplex, lapack_complex_float, c) -} // end namespace internal +}// end namespace internal -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_QR_LAPACKE_H +#endif// EIGEN_QR_LAPACKE_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/SPQRSupport/SuiteSparseQRSupport.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/SPQRSupport/SuiteSparseQRSupport.h index 953d57c9..f23ddd79 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/SPQRSupport/SuiteSparseQRSupport.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/SPQRSupport/SuiteSparseQRSupport.h @@ -12,301 +12,289 @@ #define EIGEN_SUITESPARSEQRSUPPORT_H namespace Eigen { - - template class SPQR; - template struct SPQRMatrixQReturnType; - template struct SPQRMatrixQTransposeReturnType; - template struct SPQR_QProduct; - namespace internal { - template struct traits > - { - typedef typename SPQRType::MatrixType ReturnType; - }; - template struct traits > - { - typedef typename SPQRType::MatrixType ReturnType; - }; - template struct traits > - { - typedef typename Derived::PlainObject ReturnType; - }; - } // End namespace internal - + +template class SPQR; +template struct SPQRMatrixQReturnType; +template struct SPQRMatrixQTransposeReturnType; +template struct SPQR_QProduct; +namespace internal { + template struct traits> + { + typedef typename SPQRType::MatrixType ReturnType; + }; + template struct traits> + { + typedef typename SPQRType::MatrixType ReturnType; + }; + template struct traits> + { + typedef typename Derived::PlainObject ReturnType; + }; +}// End namespace internal + /** - * \ingroup SPQRSupport_Module - * \class SPQR - * \brief Sparse QR factorization based on SuiteSparseQR library - * - * This class is used to perform a multithreaded and multifrontal rank-revealing QR decomposition - * of sparse matrices. The result is then used to solve linear leasts_square systems. - * Clearly, a QR factorization is returned such that A*P = Q*R where : - * - * P is the column permutation. Use colsPermutation() to get it. - * - * Q is the orthogonal matrix represented as Householder reflectors. - * Use matrixQ() to get an expression and matrixQ().transpose() to get the transpose. - * You can then apply it to a vector. - * - * R is the sparse triangular factor. Use matrixQR() to get it as SparseMatrix. - * NOTE : The Index type of R is always SuiteSparse_long. You can get it with SPQR::Index - * - * \tparam _MatrixType The type of the sparse matrix A, must be a column-major SparseMatrix<> - * - * \implsparsesolverconcept - * - * - */ -template -class SPQR : public SparseSolverBase > + * \ingroup SPQRSupport_Module + * \class SPQR + * \brief Sparse QR factorization based on SuiteSparseQR library + * + * This class is used to perform a multithreaded and multifrontal rank-revealing QR decomposition + * of sparse matrices. The result is then used to solve linear leasts_square systems. + * Clearly, a QR factorization is returned such that A*P = Q*R where : + * + * P is the column permutation. Use colsPermutation() to get it. + * + * Q is the orthogonal matrix represented as Householder reflectors. + * Use matrixQ() to get an expression and matrixQ().transpose() to get the transpose. + * You can then apply it to a vector. + * + * R is the sparse triangular factor. Use matrixQR() to get it as SparseMatrix. + * NOTE : The Index type of R is always SuiteSparse_long. You can get it with SPQR::Index + * + * \tparam _MatrixType The type of the sparse matrix A, must be a column-major SparseMatrix<> + * + * \implsparsesolverconcept + * + * + */ +template class SPQR : public SparseSolverBase> { - protected: - typedef SparseSolverBase > Base; - using Base::m_isInitialized; - public: - typedef typename _MatrixType::Scalar Scalar; - typedef typename _MatrixType::RealScalar RealScalar; - typedef SuiteSparse_long StorageIndex ; - typedef SparseMatrix MatrixType; - typedef Map > PermutationType; - enum { - ColsAtCompileTime = Dynamic, - MaxColsAtCompileTime = Dynamic - }; - public: - SPQR() - : m_ordering(SPQR_ORDERING_DEFAULT), m_allow_tol(SPQR_DEFAULT_TOL), m_tolerance (NumTraits::epsilon()), m_useDefaultThreshold(true) - { - cholmod_l_start(&m_cc); - } - - explicit SPQR(const _MatrixType& matrix) - : m_ordering(SPQR_ORDERING_DEFAULT), m_allow_tol(SPQR_DEFAULT_TOL), m_tolerance (NumTraits::epsilon()), m_useDefaultThreshold(true) - { - cholmod_l_start(&m_cc); - compute(matrix); - } - - ~SPQR() - { - SPQR_free(); - cholmod_l_finish(&m_cc); - } - void SPQR_free() - { - cholmod_l_free_sparse(&m_H, &m_cc); - cholmod_l_free_sparse(&m_cR, &m_cc); - cholmod_l_free_dense(&m_HTau, &m_cc); - std::free(m_E); - std::free(m_HPinv); - } +protected: + typedef SparseSolverBase> Base; + using Base::m_isInitialized; - void compute(const _MatrixType& matrix) - { - if(m_isInitialized) SPQR_free(); +public: + typedef typename _MatrixType::Scalar Scalar; + typedef typename _MatrixType::RealScalar RealScalar; + typedef SuiteSparse_long StorageIndex; + typedef SparseMatrix MatrixType; + typedef Map> PermutationType; + enum { ColsAtCompileTime = Dynamic, MaxColsAtCompileTime = Dynamic }; - MatrixType mat(matrix); - - /* Compute the default threshold as in MatLab, see: - * Tim Davis, "Algorithm 915, SuiteSparseQR: Multifrontal Multithreaded Rank-Revealing - * Sparse QR Factorization, ACM Trans. on Math. Soft. 38(1), 2011, Page 8:3 - */ - RealScalar pivotThreshold = m_tolerance; - if(m_useDefaultThreshold) - { - RealScalar max2Norm = 0.0; - for (int j = 0; j < mat.cols(); j++) max2Norm = numext::maxi(max2Norm, mat.col(j).norm()); - if(max2Norm==RealScalar(0)) - max2Norm = RealScalar(1); - pivotThreshold = 20 * (mat.rows() + mat.cols()) * max2Norm * NumTraits::epsilon(); - } - cholmod_sparse A; - A = viewAsCholmod(mat); - m_rows = matrix.rows(); - Index col = matrix.cols(); - m_rank = SuiteSparseQR(m_ordering, pivotThreshold, col, &A, - &m_cR, &m_E, &m_H, &m_HPinv, &m_HTau, &m_cc); +public: + SPQR() + : m_ordering(SPQR_ORDERING_DEFAULT), m_allow_tol(SPQR_DEFAULT_TOL), m_tolerance(NumTraits::epsilon()), + m_useDefaultThreshold(true) + { + cholmod_l_start(&m_cc); + } - if (!m_cR) - { - m_info = NumericalIssue; - m_isInitialized = false; - return; - } - m_info = Success; - m_isInitialized = true; - m_isRUpToDate = false; - } - /** - * Get the number of rows of the input matrix and the Q matrix - */ - inline Index rows() const {return m_rows; } - - /** - * Get the number of columns of the input matrix. - */ - inline Index cols() const { return m_cR->ncol; } - - template - void _solve_impl(const MatrixBase &b, MatrixBase &dest) const - { - eigen_assert(m_isInitialized && " The QR factorization should be computed first, call compute()"); - eigen_assert(b.cols()==1 && "This method is for vectors only"); + explicit SPQR(const _MatrixType &matrix) + : m_ordering(SPQR_ORDERING_DEFAULT), m_allow_tol(SPQR_DEFAULT_TOL), m_tolerance(NumTraits::epsilon()), + m_useDefaultThreshold(true) + { + cholmod_l_start(&m_cc); + compute(matrix); + } - //Compute Q^T * b - typename Dest::PlainObject y, y2; - y = matrixQ().transpose() * b; - - // Solves with the triangular matrix R - Index rk = this->rank(); - y2 = y; - y.resize((std::max)(cols(),Index(y.rows())),y.cols()); - y.topRows(rk) = this->matrixR().topLeftCorner(rk, rk).template triangularView().solve(y2.topRows(rk)); + ~SPQR() + { + SPQR_free(); + cholmod_l_finish(&m_cc); + } + void SPQR_free() + { + cholmod_l_free_sparse(&m_H, &m_cc); + cholmod_l_free_sparse(&m_cR, &m_cc); + cholmod_l_free_dense(&m_HTau, &m_cc); + std::free(m_E); + std::free(m_HPinv); + } - // Apply the column permutation - // colsPermutation() performs a copy of the permutation, - // so let's apply it manually: - for(Index i = 0; i < rk; ++i) dest.row(m_E[i]) = y.row(i); - for(Index i = rk; i < cols(); ++i) dest.row(m_E[i]).setZero(); - -// y.bottomRows(y.rows()-rk).setZero(); -// dest = colsPermutation() * y.topRows(cols()); - - m_info = Success; - } - - /** \returns the sparse triangular factor R. It is a sparse matrix - */ - const MatrixType matrixR() const - { - eigen_assert(m_isInitialized && " The QR factorization should be computed first, call compute()"); - if(!m_isRUpToDate) { - m_R = viewAsEigen(*m_cR); - m_isRUpToDate = true; - } - return m_R; - } - /// Get an expression of the matrix Q - SPQRMatrixQReturnType matrixQ() const - { - return SPQRMatrixQReturnType(*this); - } - /// Get the permutation that was applied to columns of A - PermutationType colsPermutation() const - { - eigen_assert(m_isInitialized && "Decomposition is not initialized."); - return PermutationType(m_E, m_cR->ncol); - } - /** - * Gets the rank of the matrix. - * It should be equal to matrixQR().cols if the matrix is full-rank + void compute(const _MatrixType &matrix) + { + if (m_isInitialized) SPQR_free(); + + MatrixType mat(matrix); + + /* Compute the default threshold as in MatLab, see: + * Tim Davis, "Algorithm 915, SuiteSparseQR: Multifrontal Multithreaded Rank-Revealing + * Sparse QR Factorization, ACM Trans. on Math. Soft. 38(1), 2011, Page 8:3 */ - Index rank() const - { - eigen_assert(m_isInitialized && "Decomposition is not initialized."); - return m_cc.SPQR_istat[4]; + RealScalar pivotThreshold = m_tolerance; + if (m_useDefaultThreshold) { + RealScalar max2Norm = 0.0; + for (int j = 0; j < mat.cols(); j++) max2Norm = numext::maxi(max2Norm, mat.col(j).norm()); + if (max2Norm == RealScalar(0)) max2Norm = RealScalar(1); + pivotThreshold = 20 * (mat.rows() + mat.cols()) * max2Norm * NumTraits::epsilon(); } - /// Set the fill-reducing ordering method to be used - void setSPQROrdering(int ord) { m_ordering = ord;} - /// Set the tolerance tol to treat columns with 2-norm < =tol as zero - void setPivotThreshold(const RealScalar& tol) - { - m_useDefaultThreshold = false; - m_tolerance = tol; + cholmod_sparse A; + A = viewAsCholmod(mat); + m_rows = matrix.rows(); + Index col = matrix.cols(); + m_rank = SuiteSparseQR(m_ordering, pivotThreshold, col, &A, &m_cR, &m_E, &m_H, &m_HPinv, &m_HTau, &m_cc); + + if (!m_cR) { + m_info = NumericalIssue; + m_isInitialized = false; + return; } - - /** \returns a pointer to the SPQR workspace */ - cholmod_common *cholmodCommon() const { return &m_cc; } - - - /** \brief Reports whether previous computation was successful. - * - * \returns \c Success if computation was succesful, - * \c NumericalIssue if the sparse QR can not be computed - */ - ComputationInfo info() const - { - eigen_assert(m_isInitialized && "Decomposition is not initialized."); - return m_info; + m_info = Success; + m_isInitialized = true; + m_isRUpToDate = false; + } + /** + * Get the number of rows of the input matrix and the Q matrix + */ + inline Index rows() const { return m_rows; } + + /** + * Get the number of columns of the input matrix. + */ + inline Index cols() const { return m_cR->ncol; } + + template void _solve_impl(const MatrixBase &b, MatrixBase &dest) const + { + eigen_assert(m_isInitialized && " The QR factorization should be computed first, call compute()"); + eigen_assert(b.cols() == 1 && "This method is for vectors only"); + + // Compute Q^T * b + typename Dest::PlainObject y, y2; + y = matrixQ().transpose() * b; + + // Solves with the triangular matrix R + Index rk = this->rank(); + y2 = y; + y.resize((std::max)(cols(), Index(y.rows())), y.cols()); + y.topRows(rk) = this->matrixR().topLeftCorner(rk, rk).template triangularView().solve(y2.topRows(rk)); + + // Apply the column permutation + // colsPermutation() performs a copy of the permutation, + // so let's apply it manually: + for (Index i = 0; i < rk; ++i) dest.row(m_E[i]) = y.row(i); + for (Index i = rk; i < cols(); ++i) dest.row(m_E[i]).setZero(); + + // y.bottomRows(y.rows()-rk).setZero(); + // dest = colsPermutation() * y.topRows(cols()); + + m_info = Success; + } + + /** \returns the sparse triangular factor R. It is a sparse matrix + */ + const MatrixType matrixR() const + { + eigen_assert(m_isInitialized && " The QR factorization should be computed first, call compute()"); + if (!m_isRUpToDate) { + m_R = viewAsEigen(*m_cR); + m_isRUpToDate = true; } - protected: - bool m_analysisIsOk; - bool m_factorizationIsOk; - mutable bool m_isRUpToDate; - mutable ComputationInfo m_info; - int m_ordering; // Ordering method to use, see SPQR's manual - int m_allow_tol; // Allow to use some tolerance during numerical factorization. - RealScalar m_tolerance; // treat columns with 2-norm below this tolerance as zero - mutable cholmod_sparse *m_cR; // The sparse R factor in cholmod format - mutable MatrixType m_R; // The sparse matrix R in Eigen format - mutable StorageIndex *m_E; // The permutation applied to columns - mutable cholmod_sparse *m_H; //The householder vectors - mutable StorageIndex *m_HPinv; // The row permutation of H - mutable cholmod_dense *m_HTau; // The Householder coefficients - mutable Index m_rank; // The rank of the matrix - mutable cholmod_common m_cc; // Workspace and parameters - bool m_useDefaultThreshold; // Use default threshold - Index m_rows; - template friend struct SPQR_QProduct; + return m_R; + } + /// Get an expression of the matrix Q + SPQRMatrixQReturnType matrixQ() const { return SPQRMatrixQReturnType(*this); } + /// Get the permutation that was applied to columns of A + PermutationType colsPermutation() const + { + eigen_assert(m_isInitialized && "Decomposition is not initialized."); + return PermutationType(m_E, m_cR->ncol); + } + /** + * Gets the rank of the matrix. + * It should be equal to matrixQR().cols if the matrix is full-rank + */ + Index rank() const + { + eigen_assert(m_isInitialized && "Decomposition is not initialized."); + return m_cc.SPQR_istat[4]; + } + /// Set the fill-reducing ordering method to be used + void setSPQROrdering(int ord) { m_ordering = ord; } + /// Set the tolerance tol to treat columns with 2-norm < =tol as zero + void setPivotThreshold(const RealScalar &tol) + { + m_useDefaultThreshold = false; + m_tolerance = tol; + } + + /** \returns a pointer to the SPQR workspace */ + cholmod_common *cholmodCommon() const { return &m_cc; } + + + /** \brief Reports whether previous computation was successful. + * + * \returns \c Success if computation was succesful, + * \c NumericalIssue if the sparse QR can not be computed + */ + ComputationInfo info() const + { + eigen_assert(m_isInitialized && "Decomposition is not initialized."); + return m_info; + } + +protected: + bool m_analysisIsOk; + bool m_factorizationIsOk; + mutable bool m_isRUpToDate; + mutable ComputationInfo m_info; + int m_ordering;// Ordering method to use, see SPQR's manual + int m_allow_tol;// Allow to use some tolerance during numerical factorization. + RealScalar m_tolerance;// treat columns with 2-norm below this tolerance as zero + mutable cholmod_sparse *m_cR;// The sparse R factor in cholmod format + mutable MatrixType m_R;// The sparse matrix R in Eigen format + mutable StorageIndex *m_E;// The permutation applied to columns + mutable cholmod_sparse *m_H;// The householder vectors + mutable StorageIndex *m_HPinv;// The row permutation of H + mutable cholmod_dense *m_HTau;// The Householder coefficients + mutable Index m_rank;// The rank of the matrix + mutable cholmod_common m_cc;// Workspace and parameters + bool m_useDefaultThreshold;// Use default threshold + Index m_rows; + template friend struct SPQR_QProduct; }; -template -struct SPQR_QProduct : ReturnByValue > +template struct SPQR_QProduct : ReturnByValue> { typedef typename SPQRType::Scalar Scalar; typedef typename SPQRType::StorageIndex StorageIndex; - //Define the constructor to get reference to argument types - SPQR_QProduct(const SPQRType& spqr, const Derived& other, bool transpose) : m_spqr(spqr),m_other(other),m_transpose(transpose) {} - + // Define the constructor to get reference to argument types + SPQR_QProduct(const SPQRType &spqr, const Derived &other, bool transpose) + : m_spqr(spqr), m_other(other), m_transpose(transpose) + {} + inline Index rows() const { return m_transpose ? m_spqr.rows() : m_spqr.cols(); } inline Index cols() const { return m_other.cols(); } // Assign to a vector - template - void evalTo(ResType& res) const + template void evalTo(ResType &res) const { cholmod_dense y_cd; - cholmod_dense *x_cd; - int method = m_transpose ? SPQR_QTX : SPQR_QX; + cholmod_dense *x_cd; + int method = m_transpose ? SPQR_QTX : SPQR_QX; cholmod_common *cc = m_spqr.cholmodCommon(); y_cd = viewAsCholmod(m_other.const_cast_derived()); x_cd = SuiteSparseQR_qmult(method, m_spqr.m_H, m_spqr.m_HTau, m_spqr.m_HPinv, &y_cd, cc); - res = Matrix::Map(reinterpret_cast(x_cd->x), x_cd->nrow, x_cd->ncol); + res = Matrix::Map( + reinterpret_cast(x_cd->x), x_cd->nrow, x_cd->ncol); cholmod_l_free_dense(&x_cd, cc); } - const SPQRType& m_spqr; - const Derived& m_other; - bool m_transpose; - + const SPQRType &m_spqr; + const Derived &m_other; + bool m_transpose; }; -template -struct SPQRMatrixQReturnType{ - - SPQRMatrixQReturnType(const SPQRType& spqr) : m_spqr(spqr) {} - template - SPQR_QProduct operator*(const MatrixBase& other) - { - return SPQR_QProduct(m_spqr,other.derived(),false); - } - SPQRMatrixQTransposeReturnType adjoint() const +template struct SPQRMatrixQReturnType +{ + + SPQRMatrixQReturnType(const SPQRType &spqr) : m_spqr(spqr) {} + template SPQR_QProduct operator*(const MatrixBase &other) { - return SPQRMatrixQTransposeReturnType(m_spqr); + return SPQR_QProduct(m_spqr, other.derived(), false); } + SPQRMatrixQTransposeReturnType adjoint() const { return SPQRMatrixQTransposeReturnType(m_spqr); } // To use for operations with the transpose of Q SPQRMatrixQTransposeReturnType transpose() const { return SPQRMatrixQTransposeReturnType(m_spqr); } - const SPQRType& m_spqr; + const SPQRType &m_spqr; }; -template -struct SPQRMatrixQTransposeReturnType{ - SPQRMatrixQTransposeReturnType(const SPQRType& spqr) : m_spqr(spqr) {} - template - SPQR_QProduct operator*(const MatrixBase& other) +template struct SPQRMatrixQTransposeReturnType +{ + SPQRMatrixQTransposeReturnType(const SPQRType &spqr) : m_spqr(spqr) {} + template SPQR_QProduct operator*(const MatrixBase &other) { - return SPQR_QProduct(m_spqr,other.derived(), true); + return SPQR_QProduct(m_spqr, other.derived(), true); } - const SPQRType& m_spqr; + const SPQRType &m_spqr; }; }// End namespace Eigen diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/SVD/BDCSVD.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/SVD/BDCSVD.h index 1134d66e..0f2ff7a1 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/SVD/BDCSVD.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/SVD/BDCSVD.h @@ -1,9 +1,9 @@ // This file is part of Eigen, a lightweight C++ template library // for linear algebra. -// +// // We used the "A Divide-And-Conquer Algorithm for the Bidiagonal SVD" // research report written by Ming Gu and Stanley C.Eisenstat -// The code variable names correspond to the names they used in their +// The code variable names correspond to the names they used in their // report // // Copyright (C) 2013 Gauthier Brun @@ -27,20 +27,19 @@ namespace Eigen { #ifdef EIGEN_BDCSVD_DEBUG_VERBOSE IOFormat bdcsvdfmt(8, 0, ", ", "\n", " [", "]"); #endif - + template class BDCSVD; namespace internal { -template -struct traits > -{ - typedef _MatrixType MatrixType; -}; + template struct traits> + { + typedef _MatrixType MatrixType; + }; + +}// end namespace internal + -} // end namespace internal - - /** \ingroup SVD_Module * * @@ -63,40 +62,39 @@ struct traits > * * \sa class JacobiSVD */ -template -class BDCSVD : public SVDBase > +template class BDCSVD : public SVDBase> { typedef SVDBase Base; - + public: using Base::rows; using Base::cols; using Base::computeU; using Base::computeV; - + typedef _MatrixType MatrixType; typedef typename MatrixType::Scalar Scalar; typedef typename NumTraits::Real RealScalar; typedef typename NumTraits::Literal Literal; enum { - RowsAtCompileTime = MatrixType::RowsAtCompileTime, - ColsAtCompileTime = MatrixType::ColsAtCompileTime, - DiagSizeAtCompileTime = EIGEN_SIZE_MIN_PREFER_DYNAMIC(RowsAtCompileTime, ColsAtCompileTime), - MaxRowsAtCompileTime = MatrixType::MaxRowsAtCompileTime, - MaxColsAtCompileTime = MatrixType::MaxColsAtCompileTime, - MaxDiagSizeAtCompileTime = EIGEN_SIZE_MIN_PREFER_FIXED(MaxRowsAtCompileTime, MaxColsAtCompileTime), + RowsAtCompileTime = MatrixType::RowsAtCompileTime, + ColsAtCompileTime = MatrixType::ColsAtCompileTime, + DiagSizeAtCompileTime = EIGEN_SIZE_MIN_PREFER_DYNAMIC(RowsAtCompileTime, ColsAtCompileTime), + MaxRowsAtCompileTime = MatrixType::MaxRowsAtCompileTime, + MaxColsAtCompileTime = MatrixType::MaxColsAtCompileTime, + MaxDiagSizeAtCompileTime = EIGEN_SIZE_MIN_PREFER_FIXED(MaxRowsAtCompileTime, MaxColsAtCompileTime), MatrixOptions = MatrixType::Options }; typedef typename Base::MatrixUType MatrixUType; typedef typename Base::MatrixVType MatrixVType; typedef typename Base::SingularValuesType SingularValuesType; - + typedef Matrix MatrixX; typedef Matrix MatrixXr; typedef Matrix VectorType; typedef Array ArrayXr; - typedef Array ArrayXi; + typedef Array ArrayXi; typedef Ref ArrayRef; typedef Ref IndicesRef; @@ -105,8 +103,7 @@ class BDCSVD : public SVDBase > * The default constructor is useful in cases in which the user intends to * perform decompositions via BDCSVD::compute(const MatrixType&). */ - BDCSVD() : m_algoswap(16), m_numIters(0) - {} + BDCSVD() : m_algoswap(16), m_numIters(0) {} /** \brief Default Constructor with memory preallocation @@ -115,8 +112,7 @@ class BDCSVD : public SVDBase > * according to the specified problem size. * \sa BDCSVD() */ - BDCSVD(Index rows, Index cols, unsigned int computationOptions = 0) - : m_algoswap(16), m_numIters(0) + BDCSVD(Index rows, Index cols, unsigned int computationOptions = 0) : m_algoswap(16), m_numIters(0) { allocate(rows, cols, computationOptions); } @@ -124,66 +120,87 @@ class BDCSVD : public SVDBase > /** \brief Constructor performing the decomposition of given matrix. * * \param matrix the matrix to decompose - * \param computationOptions optional parameter allowing to specify if you want full or thin U or V unitaries to be computed. - * By default, none is computed. This is a bit - field, the possible bits are #ComputeFullU, #ComputeThinU, + * \param computationOptions optional parameter allowing to specify if you want full or thin U or V unitaries to be + * computed. By default, none is computed. This is a bit - field, the possible bits are #ComputeFullU, #ComputeThinU, * #ComputeFullV, #ComputeThinV. * - * Thin unitaries are only available if your matrix type has a Dynamic number of columns (for example MatrixXf). They also are not - * available with the (non - default) FullPivHouseholderQR preconditioner. + * Thin unitaries are only available if your matrix type has a Dynamic number of columns (for example MatrixXf). They + * also are not available with the (non - default) FullPivHouseholderQR preconditioner. */ - BDCSVD(const MatrixType& matrix, unsigned int computationOptions = 0) - : m_algoswap(16), m_numIters(0) + BDCSVD(const MatrixType &matrix, unsigned int computationOptions = 0) : m_algoswap(16), m_numIters(0) { compute(matrix, computationOptions); } - ~BDCSVD() - { - } - + ~BDCSVD() {} + /** \brief Method performing the decomposition of given matrix using custom options. * * \param matrix the matrix to decompose - * \param computationOptions optional parameter allowing to specify if you want full or thin U or V unitaries to be computed. - * By default, none is computed. This is a bit - field, the possible bits are #ComputeFullU, #ComputeThinU, + * \param computationOptions optional parameter allowing to specify if you want full or thin U or V unitaries to be + * computed. By default, none is computed. This is a bit - field, the possible bits are #ComputeFullU, #ComputeThinU, * #ComputeFullV, #ComputeThinV. * - * Thin unitaries are only available if your matrix type has a Dynamic number of columns (for example MatrixXf). They also are not - * available with the (non - default) FullPivHouseholderQR preconditioner. + * Thin unitaries are only available if your matrix type has a Dynamic number of columns (for example MatrixXf). They + * also are not available with the (non - default) FullPivHouseholderQR preconditioner. */ - BDCSVD& compute(const MatrixType& matrix, unsigned int computationOptions); + BDCSVD &compute(const MatrixType &matrix, unsigned int computationOptions); /** \brief Method performing the decomposition of given matrix using current options. * * \param matrix the matrix to decompose * - * This method uses the current \a computationOptions, as already passed to the constructor or to compute(const MatrixType&, unsigned int). + * This method uses the current \a computationOptions, as already passed to the constructor or to compute(const + * MatrixType&, unsigned int). */ - BDCSVD& compute(const MatrixType& matrix) - { - return compute(matrix, this->m_computationOptions); - } + BDCSVD &compute(const MatrixType &matrix) { return compute(matrix, this->m_computationOptions); } - void setSwitchSize(int s) + void setSwitchSize(int s) { - eigen_assert(s>3 && "BDCSVD the size of the algo switch has to be greater than 3"); + eigen_assert(s > 3 && "BDCSVD the size of the algo switch has to be greater than 3"); m_algoswap = s; } - + private: void allocate(Index rows, Index cols, unsigned int computationOptions); void divide(Index firstCol, Index lastCol, Index firstRowW, Index firstColW, Index shift); - void computeSVDofM(Index firstCol, Index n, MatrixXr& U, VectorType& singVals, MatrixXr& V); - void computeSingVals(const ArrayRef& col0, const ArrayRef& diag, const IndicesRef& perm, VectorType& singVals, ArrayRef shifts, ArrayRef mus); - void perturbCol0(const ArrayRef& col0, const ArrayRef& diag, const IndicesRef& perm, const VectorType& singVals, const ArrayRef& shifts, const ArrayRef& mus, ArrayRef zhat); - void computeSingVecs(const ArrayRef& zhat, const ArrayRef& diag, const IndicesRef& perm, const VectorType& singVals, const ArrayRef& shifts, const ArrayRef& mus, MatrixXr& U, MatrixXr& V); + void computeSVDofM(Index firstCol, Index n, MatrixXr &U, VectorType &singVals, MatrixXr &V); + void computeSingVals(const ArrayRef &col0, + const ArrayRef &diag, + const IndicesRef &perm, + VectorType &singVals, + ArrayRef shifts, + ArrayRef mus); + void perturbCol0(const ArrayRef &col0, + const ArrayRef &diag, + const IndicesRef &perm, + const VectorType &singVals, + const ArrayRef &shifts, + const ArrayRef &mus, + ArrayRef zhat); + void computeSingVecs(const ArrayRef &zhat, + const ArrayRef &diag, + const IndicesRef &perm, + const VectorType &singVals, + const ArrayRef &shifts, + const ArrayRef &mus, + MatrixXr &U, + MatrixXr &V); void deflation43(Index firstCol, Index shift, Index i, Index size); - void deflation44(Index firstColu , Index firstColm, Index firstRowW, Index firstColW, Index i, Index j, Index size); + void deflation44(Index firstColu, Index firstColm, Index firstRowW, Index firstColW, Index i, Index j, Index size); void deflation(Index firstCol, Index lastCol, Index k, Index firstRowW, Index firstColW, Index shift); template - void copyUV(const HouseholderU &householderU, const HouseholderV &householderV, const NaiveU &naiveU, const NaiveV &naivev); - void structured_update(Block A, const MatrixXr &B, Index n1); - static RealScalar secularEq(RealScalar x, const ArrayRef& col0, const ArrayRef& diag, const IndicesRef &perm, const ArrayRef& diagShifted, RealScalar shift); + void copyUV(const HouseholderU &householderU, + const HouseholderV &householderV, + const NaiveU &naiveU, + const NaiveV &naivev); + void structured_update(Block A, const MatrixXr &B, Index n1); + static RealScalar secularEq(RealScalar x, + const ArrayRef &col0, + const ArrayRef &diag, + const IndicesRef &perm, + const ArrayRef &diagShifted, + RealScalar shift); protected: MatrixXr m_naiveU, m_naiveV; @@ -193,7 +210,7 @@ class BDCSVD : public SVDBase > ArrayXi m_workspaceI; int m_algoswap; bool m_isTranspose, m_compU, m_compV; - + using Base::m_singularValues; using Base::m_diagSize; using Base::m_computeFullU; @@ -205,66 +222,67 @@ class BDCSVD : public SVDBase > using Base::m_isInitialized; using Base::m_nonzeroSingularValues; -public: +public: int m_numIters; -}; //end class BDCSVD +};// end class BDCSVD // Method to allocate and initialize matrix and attributes -template -void BDCSVD::allocate(Index rows, Index cols, unsigned int computationOptions) +template void BDCSVD::allocate(Index rows, Index cols, unsigned int computationOptions) { m_isTranspose = (cols > rows); - if (Base::allocate(rows, cols, computationOptions)) - return; - - m_computed = MatrixXr::Zero(m_diagSize + 1, m_diagSize ); + if (Base::allocate(rows, cols, computationOptions)) return; + + m_computed = MatrixXr::Zero(m_diagSize + 1, m_diagSize); m_compU = computeV(); m_compV = computeU(); - if (m_isTranspose) - std::swap(m_compU, m_compV); - - if (m_compU) m_naiveU = MatrixXr::Zero(m_diagSize + 1, m_diagSize + 1 ); - else m_naiveU = MatrixXr::Zero(2, m_diagSize + 1 ); - + if (m_isTranspose) std::swap(m_compU, m_compV); + + if (m_compU) + m_naiveU = MatrixXr::Zero(m_diagSize + 1, m_diagSize + 1); + else + m_naiveU = MatrixXr::Zero(2, m_diagSize + 1); + if (m_compV) m_naiveV = MatrixXr::Zero(m_diagSize, m_diagSize); - - m_workspace.resize((m_diagSize+1)*(m_diagSize+1)*3); - m_workspaceI.resize(3*m_diagSize); + + m_workspace.resize((m_diagSize + 1) * (m_diagSize + 1) * 3); + m_workspaceI.resize(3 * m_diagSize); }// end allocate template -BDCSVD& BDCSVD::compute(const MatrixType& matrix, unsigned int computationOptions) +BDCSVD &BDCSVD::compute(const MatrixType &matrix, unsigned int computationOptions) { #ifdef EIGEN_BDCSVD_DEBUG_VERBOSE - std::cout << "\n\n\n======================================================================================================================\n\n\n"; + std::cout << "\n\n\n=================================================================================================" + "=====================\n\n\n"; #endif allocate(matrix.rows(), matrix.cols(), computationOptions); using std::abs; const RealScalar considerZero = (std::numeric_limits::min)(); - + //**** step -1 - If the problem is too small, directly falls back to JacobiSVD and return - if(matrix.cols() < m_algoswap) - { + if (matrix.cols() < m_algoswap) { // FIXME this line involves temporaries - JacobiSVD jsvd(matrix,computationOptions); - if(computeU()) m_matrixU = jsvd.matrixU(); - if(computeV()) m_matrixV = jsvd.matrixV(); + JacobiSVD jsvd(matrix, computationOptions); + if (computeU()) m_matrixU = jsvd.matrixU(); + if (computeV()) m_matrixV = jsvd.matrixV(); m_singularValues = jsvd.singularValues(); m_nonzeroSingularValues = jsvd.nonzeroSingularValues(); m_isInitialized = true; return *this; } - + //**** step 0 - Copy the input matrix and apply scaling to reduce over/under-flows RealScalar scale = matrix.cwiseAbs().maxCoeff(); - if(scale==Literal(0)) scale = Literal(1); + if (scale == Literal(0)) scale = Literal(1); MatrixX copy; - if (m_isTranspose) copy = matrix.adjoint()/scale; - else copy = matrix/scale; - + if (m_isTranspose) + copy = matrix.adjoint() / scale; + else + copy = matrix / scale; + //**** step 1 - Bidiagonalization // FIXME this line involves temporaries internal::UpperBidiagonalization bid(copy); @@ -278,18 +296,14 @@ BDCSVD& BDCSVD::compute(const MatrixType& matrix, unsign divide(0, m_diagSize - 1, 0, 0, 0); //**** step 3 - Copy singular values and vectors - for (int i=0; i& BDCSVD::compute(const MatrixType& matrix, unsign // std::cout << "m_naiveU\n" << m_naiveU << "\n\n"; // std::cout << "m_naiveV\n" << m_naiveV << "\n\n"; #endif - if(m_isTranspose) copyUV(bid.householderV(), bid.householderU(), m_naiveV, m_naiveU); - else copyUV(bid.householderU(), bid.householderV(), m_naiveU, m_naiveV); + if (m_isTranspose) + copyUV(bid.householderV(), bid.householderU(), m_naiveV, m_naiveU); + else + copyUV(bid.householderU(), bid.householderV(), m_naiveU, m_naiveV); m_isInitialized = true; return *this; @@ -309,109 +325,107 @@ BDCSVD& BDCSVD::compute(const MatrixType& matrix, unsign template template -void BDCSVD::copyUV(const HouseholderU &householderU, const HouseholderV &householderV, const NaiveU &naiveU, const NaiveV &naiveV) +void BDCSVD::copyUV(const HouseholderU &householderU, + const HouseholderV &householderV, + const NaiveU &naiveU, + const NaiveV &naiveV) { // Note exchange of U and V: m_matrixU is set from m_naiveV and vice versa - if (computeU()) - { + if (computeU()) { Index Ucols = m_computeThinU ? m_diagSize : householderU.cols(); m_matrixU = MatrixX::Identity(householderU.cols(), Ucols); - m_matrixU.topLeftCorner(m_diagSize, m_diagSize) = naiveV.template cast().topLeftCorner(m_diagSize, m_diagSize); - householderU.applyThisOnTheLeft(m_matrixU); // FIXME this line involves a temporary buffer + m_matrixU.topLeftCorner(m_diagSize, m_diagSize) = + naiveV.template cast().topLeftCorner(m_diagSize, m_diagSize); + householderU.applyThisOnTheLeft(m_matrixU);// FIXME this line involves a temporary buffer } - if (computeV()) - { + if (computeV()) { Index Vcols = m_computeThinV ? m_diagSize : householderV.cols(); m_matrixV = MatrixX::Identity(householderV.cols(), Vcols); - m_matrixV.topLeftCorner(m_diagSize, m_diagSize) = naiveU.template cast().topLeftCorner(m_diagSize, m_diagSize); - householderV.applyThisOnTheLeft(m_matrixV); // FIXME this line involves a temporary buffer + m_matrixV.topLeftCorner(m_diagSize, m_diagSize) = + naiveU.template cast().topLeftCorner(m_diagSize, m_diagSize); + householderV.applyThisOnTheLeft(m_matrixV);// FIXME this line involves a temporary buffer } } /** \internal - * Performs A = A * B exploiting the special structure of the matrix A. Splitting A as: - * A = [A1] - * [A2] - * such that A1.rows()==n1, then we assume that at least half of the columns of A1 and A2 are zeros. - * We can thus pack them prior to the the matrix product. However, this is only worth the effort if the matrix is large - * enough. - */ + * Performs A = A * B exploiting the special structure of the matrix A. Splitting A as: + * A = [A1] + * [A2] + * such that A1.rows()==n1, then we assume that at least half of the columns of A1 and A2 are zeros. + * We can thus pack them prior to the the matrix product. However, this is only worth the effort if the matrix is large + * enough. + */ template -void BDCSVD::structured_update(Block A, const MatrixXr &B, Index n1) +void BDCSVD::structured_update(Block A, const MatrixXr &B, Index n1) { Index n = A.rows(); - if(n>100) - { + if (n > 100) { // If the matrices are large enough, let's exploit the sparse structure of A by // splitting it in half (wrt n1), and packing the non-zero columns. Index n2 = n - n1; - Map A1(m_workspace.data() , n1, n); - Map A2(m_workspace.data()+ n1*n, n2, n); - Map B1(m_workspace.data()+ n*n, n, n); - Map B2(m_workspace.data()+2*n*n, n, n); - Index k1=0, k2=0; - for(Index j=0; j A1(m_workspace.data(), n1, n); + Map A2(m_workspace.data() + n1 * n, n2, n); + Map B1(m_workspace.data() + n * n, n, n); + Map B2(m_workspace.data() + 2 * n * n, n, n); + Index k1 = 0, k2 = 0; + for (Index j = 0; j < n; ++j) { + if ((A.col(j).head(n1).array() != Literal(0)).any()) { A1.col(k1) = A.col(j).head(n1); B1.row(k1) = B.row(j); ++k1; } - if( (A.col(j).tail(n2).array()!=Literal(0)).any() ) - { + if ((A.col(j).tail(n2).array() != Literal(0)).any()) { A2.col(k2) = A.col(j).tail(n2); B2.row(k2) = B.row(j); ++k2; } } - - A.topRows(n1).noalias() = A1.leftCols(k1) * B1.topRows(k1); + + A.topRows(n1).noalias() = A1.leftCols(k1) * B1.topRows(k1); A.bottomRows(n2).noalias() = A2.leftCols(k2) * B2.topRows(k2); - } - else - { - Map tmp(m_workspace.data(),n,n); - tmp.noalias() = A*B; + } else { + Map tmp(m_workspace.data(), n, n); + tmp.noalias() = A * B; A = tmp; } } -// The divide algorithm is done "in place", we are always working on subsets of the same matrix. The divide methods takes as argument the -// place of the submatrix we are currently working on. +// The divide algorithm is done "in place", we are always working on subsets of the same matrix. The divide methods +// takes as argument the place of the submatrix we are currently working on. //@param firstCol : The Index of the first column of the submatrix of m_computed and for m_naiveU; -//@param lastCol : The Index of the last column of the submatrix of m_computed and for m_naiveU; +//@param lastCol : The Index of the last column of the submatrix of m_computed and for m_naiveU; // lastCol + 1 - firstCol is the size of the submatrix. -//@param firstRowW : The Index of the first row of the matrix W that we are to change. (see the reference paper section 1 for more information on W) +//@param firstRowW : The Index of the first row of the matrix W that we are to change. (see the reference paper section +//1 for more information on W) //@param firstRowW : Same as firstRowW with the column. -//@param shift : Each time one takes the left submatrix, one must add 1 to the shift. Why? Because! We actually want the last column of the U submatrix -// to become the first column (*coeff) and to shift all the other columns to the right. There are more details on the reference paper. +//@param shift : Each time one takes the left submatrix, one must add 1 to the shift. Why? Because! We actually want the +//last column of the U submatrix +// to become the first column (*coeff) and to shift all the other columns to the right. There are more details on the +// reference paper. template -void BDCSVD::divide (Index firstCol, Index lastCol, Index firstRowW, Index firstColW, Index shift) +void BDCSVD::divide(Index firstCol, Index lastCol, Index firstRowW, Index firstColW, Index shift) { // requires rows = cols + 1; using std::pow; using std::sqrt; using std::abs; const Index n = lastCol - firstCol + 1; - const Index k = n/2; + const Index k = n / 2; const RealScalar considerZero = (std::numeric_limits::min)(); RealScalar alphaK; - RealScalar betaK; - RealScalar r0; + RealScalar betaK; + RealScalar r0; RealScalar lambda, phi, c0, s0; VectorType l, f; - // We use the other algorithm which is more efficient for small + // We use the other algorithm which is more efficient for small // matrices. - if (n < m_algoswap) - { + if (n < m_algoswap) { // FIXME this line involves temporaries JacobiSVD b(m_computed.block(firstCol, firstCol, n + 1, n), ComputeFullU | (m_compV ? ComputeFullV : 0)); if (m_compU) m_naiveU.block(firstCol, firstCol, n + 1, n + 1).real() = b.matrixU(); - else - { + else { m_naiveU.row(0).segment(firstCol, n + 1).real() = b.matrixU().row(0); m_naiveU.row(1).segment(firstCol, n + 1).real() = b.matrixU().row(n); } @@ -421,140 +435,127 @@ void BDCSVD::divide (Index firstCol, Index lastCol, Index firstRowW, return; } // We use the divide and conquer algorithm - alphaK = m_computed(firstCol + k, firstCol + k); + alphaK = m_computed(firstCol + k, firstCol + k); betaK = m_computed(firstCol + k + 1, firstCol + k); // The divide must be done in that order in order to have good results. Divide change the data inside the submatrices - // and the divide of the right submatrice reads one column of the left submatrice. That's why we need to treat the - // right submatrix before the left one. + // and the divide of the right submatrice reads one column of the left submatrice. That's why we need to treat the + // right submatrix before the left one. divide(k + 1 + firstCol, lastCol, k + 1 + firstRowW, k + 1 + firstColW, shift); divide(firstCol, k - 1 + firstCol, firstRowW, firstColW + 1, shift + 1); - if (m_compU) - { + if (m_compU) { lambda = m_naiveU(firstCol + k, firstCol + k); phi = m_naiveU(firstCol + k + 1, lastCol + 1); - } - else - { + } else { lambda = m_naiveU(1, firstCol + k); phi = m_naiveU(0, lastCol + 1); } r0 = sqrt((abs(alphaK * lambda) * abs(alphaK * lambda)) + abs(betaK * phi) * abs(betaK * phi)); - if (m_compU) - { + if (m_compU) { l = m_naiveU.row(firstCol + k).segment(firstCol, k); f = m_naiveU.row(firstCol + k + 1).segment(firstCol + k + 1, n - k - 1); - } - else - { + } else { l = m_naiveU.row(1).segment(firstCol, k); f = m_naiveU.row(0).segment(firstCol + k + 1, n - k - 1); } - if (m_compV) m_naiveV(firstRowW+k, firstColW) = Literal(1); - if (r0= firstCol; i--) + for (Index i = firstCol + k - 1; i >= firstCol; i--) m_naiveU.col(i + 1).segment(firstCol, k + 1) = m_naiveU.col(i).segment(firstCol, k + 1); // we shift q1 at the left with a factor c0 - m_naiveU.col(firstCol).segment( firstCol, k + 1) = (q1 * c0); + m_naiveU.col(firstCol).segment(firstCol, k + 1) = (q1 * c0); // last column = q1 * - s0 - m_naiveU.col(lastCol + 1).segment(firstCol, k + 1) = (q1 * ( - s0)); + m_naiveU.col(lastCol + 1).segment(firstCol, k + 1) = (q1 * (-s0)); // first column = q2 * s0 - m_naiveU.col(firstCol).segment(firstCol + k + 1, n - k) = m_naiveU.col(lastCol + 1).segment(firstCol + k + 1, n - k) * s0; + m_naiveU.col(firstCol).segment(firstCol + k + 1, n - k) = + m_naiveU.col(lastCol + 1).segment(firstCol + k + 1, n - k) * s0; // q2 *= c0 m_naiveU.col(lastCol + 1).segment(firstCol + k + 1, n - k) *= c0; - } - else - { + } else { RealScalar q1 = m_naiveU(0, firstCol + k); // we shift Q1 to the right - for (Index i = firstCol + k - 1; i >= firstCol; i--) - m_naiveU(0, i + 1) = m_naiveU(0, i); + for (Index i = firstCol + k - 1; i >= firstCol; i--) m_naiveU(0, i + 1) = m_naiveU(0, i); // we shift q1 at the left with a factor c0 m_naiveU(0, firstCol) = (q1 * c0); // last column = q1 * - s0 - m_naiveU(0, lastCol + 1) = (q1 * ( - s0)); + m_naiveU(0, lastCol + 1) = (q1 * (-s0)); // first column = q2 * s0 - m_naiveU(1, firstCol) = m_naiveU(1, lastCol + 1) *s0; + m_naiveU(1, firstCol) = m_naiveU(1, lastCol + 1) * s0; // q2 *= c0 m_naiveU(1, lastCol + 1) *= c0; m_naiveU.row(1).segment(firstCol + 1, k).setZero(); m_naiveU.row(0).segment(firstCol + k + 1, n - k - 1).setZero(); } - + #ifdef EIGEN_BDCSVD_SANITY_CHECKS assert(m_naiveU.allFinite()); assert(m_naiveV.allFinite()); assert(m_computed.allFinite()); #endif - + m_computed(firstCol + shift, firstCol + shift) = r0; m_computed.col(firstCol + shift).segment(firstCol + shift + 1, k) = alphaK * l.transpose().real(); m_computed.col(firstCol + shift).segment(firstCol + shift + k + 1, n - k - 1) = betaK * f.transpose().real(); #ifdef EIGEN_BDCSVD_DEBUG_VERBOSE - ArrayXr tmp1 = (m_computed.block(firstCol+shift, firstCol+shift, n, n)).jacobiSvd().singularValues(); + ArrayXr tmp1 = (m_computed.block(firstCol + shift, firstCol + shift, n, n)).jacobiSvd().singularValues(); #endif // Second part: try to deflate singular values in combined matrix deflation(firstCol, lastCol, k, firstRowW, firstColW, shift); #ifdef EIGEN_BDCSVD_DEBUG_VERBOSE - ArrayXr tmp2 = (m_computed.block(firstCol+shift, firstCol+shift, n, n)).jacobiSvd().singularValues(); + ArrayXr tmp2 = (m_computed.block(firstCol + shift, firstCol + shift, n, n)).jacobiSvd().singularValues(); std::cout << "\n\nj1 = " << tmp1.transpose().format(bdcsvdfmt) << "\n"; std::cout << "j2 = " << tmp2.transpose().format(bdcsvdfmt) << "\n\n"; - std::cout << "err: " << ((tmp1-tmp2).abs()>1e-12*tmp2.abs()).transpose() << "\n"; + std::cout << "err: " << ((tmp1 - tmp2).abs() > 1e-12 * tmp2.abs()).transpose() << "\n"; static int count = 0; std::cout << "# " << ++count << "\n\n"; - assert((tmp1-tmp2).matrix().norm() < 1e-14*tmp2.matrix().norm()); + assert((tmp1 - tmp2).matrix().norm() < 1e-14 * tmp2.matrix().norm()); // assert(count<681); // assert(((tmp1-tmp2).abs()<1e-13*tmp2.abs()).all()); #endif - + // Third part: compute SVD of combined matrix MatrixXr UofSVD, VofSVD; VectorType singVals; computeSVDofM(firstCol + shift, n, UofSVD, singVals, VofSVD); - + #ifdef EIGEN_BDCSVD_SANITY_CHECKS assert(UofSVD.allFinite()); assert(VofSVD.allFinite()); #endif - + if (m_compU) - structured_update(m_naiveU.block(firstCol, firstCol, n + 1, n + 1), UofSVD, (n+2)/2); - else - { - Map,Aligned> tmp(m_workspace.data(),2,n+1); - tmp.noalias() = m_naiveU.middleCols(firstCol, n+1) * UofSVD; + structured_update(m_naiveU.block(firstCol, firstCol, n + 1, n + 1), UofSVD, (n + 2) / 2); + else { + Map, Aligned> tmp(m_workspace.data(), 2, n + 1); + tmp.noalias() = m_naiveU.middleCols(firstCol, n + 1) * UofSVD; m_naiveU.middleCols(firstCol, n + 1) = tmp; } - - if (m_compV) structured_update(m_naiveV.block(firstRowW, firstColW, n, n), VofSVD, (n+1)/2); - + + if (m_compV) structured_update(m_naiveV.block(firstRowW, firstColW, n, n), VofSVD, (n + 1) / 2); + #ifdef EIGEN_BDCSVD_SANITY_CHECKS assert(m_naiveU.allFinite()); assert(m_naiveV.allFinite()); assert(m_computed.allFinite()); #endif - + m_computed.block(firstCol + shift, firstCol + shift, n, n).setZero(); m_computed.block(firstCol + shift, firstCol + shift, n, n).diagonal() = singVals; }// end divide @@ -566,147 +567,160 @@ void BDCSVD::divide (Index firstCol, Index lastCol, Index firstRowW, // // TODO Opportunities for optimization: better root finding algo, better stopping criterion, better // handling of round-off errors, be consistent in ordering -// For instance, to solve the secular equation using FMM, see http://www.stat.uchicago.edu/~lekheng/courses/302/classics/greengard-rokhlin.pdf -template -void BDCSVD::computeSVDofM(Index firstCol, Index n, MatrixXr& U, VectorType& singVals, MatrixXr& V) +// For instance, to solve the secular equation using FMM, see +// http://www.stat.uchicago.edu/~lekheng/courses/302/classics/greengard-rokhlin.pdf +template +void BDCSVD::computeSVDofM(Index firstCol, Index n, MatrixXr &U, VectorType &singVals, MatrixXr &V) { const RealScalar considerZero = (std::numeric_limits::min)(); using std::abs; ArrayRef col0 = m_computed.col(firstCol).segment(firstCol, n); - m_workspace.head(n) = m_computed.block(firstCol, firstCol, n, n).diagonal(); + m_workspace.head(n) = m_computed.block(firstCol, firstCol, n, n).diagonal(); ArrayRef diag = m_workspace.head(n); diag(0) = Literal(0); // Allocate space for singular values and vectors singVals.resize(n); - U.resize(n+1, n+1); + U.resize(n + 1, n + 1); if (m_compV) V.resize(n, n); #ifdef EIGEN_BDCSVD_DEBUG_VERBOSE - if (col0.hasNaN() || diag.hasNaN()) - std::cout << "\n\nHAS NAN\n\n"; + if (col0.hasNaN() || diag.hasNaN()) std::cout << "\n\nHAS NAN\n\n"; #endif - + // Many singular values might have been deflated, the zero ones have been moved to the end, // but others are interleaved and we must ignore them at this stage. // To this end, let's compute a permutation skipping them: Index actual_n = n; - while(actual_n>1 && diag(actual_n-1)==Literal(0)) --actual_n; - Index m = 0; // size of the deflated problem - for(Index k=0;kconsiderZero) - m_workspaceI(m++) = k; - Map perm(m_workspaceI.data(),m); - - Map shifts(m_workspace.data()+1*n, n); - Map mus(m_workspace.data()+2*n, n); - Map zhat(m_workspace.data()+3*n, n); + while (actual_n > 1 && diag(actual_n - 1) == Literal(0)) --actual_n; + Index m = 0;// size of the deflated problem + for (Index k = 0; k < actual_n; ++k) + if (abs(col0(k)) > considerZero) m_workspaceI(m++) = k; + Map perm(m_workspaceI.data(), m); + + Map shifts(m_workspace.data() + 1 * n, n); + Map mus(m_workspace.data() + 2 * n, n); + Map zhat(m_workspace.data() + 3 * n, n); #ifdef EIGEN_BDCSVD_DEBUG_VERBOSE std::cout << "computeSVDofM using:\n"; std::cout << " z: " << col0.transpose() << "\n"; std::cout << " d: " << diag.transpose() << "\n"; #endif - + // Compute singVals, shifts, and mus computeSingVals(col0, diag, perm, singVals, shifts, mus); - + #ifdef EIGEN_BDCSVD_DEBUG_VERBOSE - std::cout << " j: " << (m_computed.block(firstCol, firstCol, n, n)).jacobiSvd().singularValues().transpose().reverse() << "\n\n"; + std::cout << " j: " + << (m_computed.block(firstCol, firstCol, n, n)).jacobiSvd().singularValues().transpose().reverse() + << "\n\n"; std::cout << " sing-val: " << singVals.transpose() << "\n"; std::cout << " mu: " << mus.transpose() << "\n"; std::cout << " shift: " << shifts.transpose() << "\n"; - + { Index actual_n = n; - while(actual_n>1 && abs(col0(actual_n-1)) 1 && abs(col0(actual_n - 1)) < considerZero) --actual_n; std::cout << "\n\n mus: " << mus.head(actual_n).transpose() << "\n\n"; - std::cout << " check1 (expect0) : " << ((singVals.array()-(shifts+mus)) / singVals.array()).head(actual_n).transpose() << "\n\n"; - std::cout << " check2 (>0) : " << ((singVals.array()-diag) / singVals.array()).head(actual_n).transpose() << "\n\n"; - std::cout << " check3 (>0) : " << ((diag.segment(1,actual_n-1)-singVals.head(actual_n-1).array()) / singVals.head(actual_n-1).array()).transpose() << "\n\n\n"; - std::cout << " check4 (>0) : " << ((singVals.segment(1,actual_n-1)-singVals.head(actual_n-1))).transpose() << "\n\n\n"; + std::cout << " check1 (expect0) : " + << ((singVals.array() - (shifts + mus)) / singVals.array()).head(actual_n).transpose() << "\n\n"; + std::cout << " check2 (>0) : " << ((singVals.array() - diag) / singVals.array()).head(actual_n).transpose() + << "\n\n"; + std::cout << " check3 (>0) : " + << ((diag.segment(1, actual_n - 1) - singVals.head(actual_n - 1).array()) + / singVals.head(actual_n - 1).array()) + .transpose() + << "\n\n\n"; + std::cout << " check4 (>0) : " + << ((singVals.segment(1, actual_n - 1) - singVals.head(actual_n - 1))).transpose() << "\n\n\n"; } #endif - + #ifdef EIGEN_BDCSVD_SANITY_CHECKS assert(singVals.allFinite()); assert(mus.allFinite()); assert(shifts.allFinite()); #endif - + // Compute zhat perturbCol0(col0, diag, perm, singVals, shifts, mus, zhat); -#ifdef EIGEN_BDCSVD_DEBUG_VERBOSE +#ifdef EIGEN_BDCSVD_DEBUG_VERBOSE std::cout << " zhat: " << zhat.transpose() << "\n"; #endif - + #ifdef EIGEN_BDCSVD_SANITY_CHECKS assert(zhat.allFinite()); #endif - + computeSingVecs(zhat, diag, perm, singVals, shifts, mus, U, V); - -#ifdef EIGEN_BDCSVD_DEBUG_VERBOSE - std::cout << "U^T U: " << (U.transpose() * U - MatrixXr(MatrixXr::Identity(U.cols(),U.cols()))).norm() << "\n"; - std::cout << "V^T V: " << (V.transpose() * V - MatrixXr(MatrixXr::Identity(V.cols(),V.cols()))).norm() << "\n"; + +#ifdef EIGEN_BDCSVD_DEBUG_VERBOSE + std::cout << "U^T U: " << (U.transpose() * U - MatrixXr(MatrixXr::Identity(U.cols(), U.cols()))).norm() << "\n"; + std::cout << "V^T V: " << (V.transpose() * V - MatrixXr(MatrixXr::Identity(V.cols(), V.cols()))).norm() << "\n"; #endif - + #ifdef EIGEN_BDCSVD_SANITY_CHECKS assert(U.allFinite()); assert(V.allFinite()); - assert((U.transpose() * U - MatrixXr(MatrixXr::Identity(U.cols(),U.cols()))).norm() < 1e-14 * n); - assert((V.transpose() * V - MatrixXr(MatrixXr::Identity(V.cols(),V.cols()))).norm() < 1e-14 * n); + assert((U.transpose() * U - MatrixXr(MatrixXr::Identity(U.cols(), U.cols()))).norm() < 1e-14 * n); + assert((V.transpose() * V - MatrixXr(MatrixXr::Identity(V.cols(), V.cols()))).norm() < 1e-14 * n); assert(m_naiveU.allFinite()); assert(m_naiveV.allFinite()); assert(m_computed.allFinite()); #endif - + // Because of deflation, the singular values might not be completely sorted. // Fortunately, reordering them is a O(n) problem - for(Index i=0; isingVals(i+1)) - { + for (Index i = 0; i < actual_n - 1; ++i) { + if (singVals(i) > singVals(i + 1)) { using std::swap; - swap(singVals(i),singVals(i+1)); - U.col(i).swap(U.col(i+1)); - if(m_compV) V.col(i).swap(V.col(i+1)); + swap(singVals(i), singVals(i + 1)); + U.col(i).swap(U.col(i + 1)); + if (m_compV) V.col(i).swap(V.col(i + 1)); } } - + // Reverse order so that singular values in increased order // Because of deflation, the zeros singular-values are already at the end singVals.head(actual_n).reverseInPlace(); U.leftCols(actual_n).rowwise().reverseInPlace(); if (m_compV) V.leftCols(actual_n).rowwise().reverseInPlace(); - + #ifdef EIGEN_BDCSVD_DEBUG_VERBOSE - JacobiSVD jsvd(m_computed.block(firstCol, firstCol, n, n) ); + JacobiSVD jsvd(m_computed.block(firstCol, firstCol, n, n)); std::cout << " * j: " << jsvd.singularValues().transpose() << "\n\n"; std::cout << " * sing-val: " << singVals.transpose() << "\n"; // std::cout << " * err: " << ((jsvd.singularValues()-singVals)>1e-13*singVals.norm()).transpose() << "\n"; #endif } -template -typename BDCSVD::RealScalar BDCSVD::secularEq(RealScalar mu, const ArrayRef& col0, const ArrayRef& diag, const IndicesRef &perm, const ArrayRef& diagShifted, RealScalar shift) +template +typename BDCSVD::RealScalar BDCSVD::secularEq(RealScalar mu, + const ArrayRef &col0, + const ArrayRef &diag, + const IndicesRef &perm, + const ArrayRef &diagShifted, + RealScalar shift) { Index m = perm.size(); RealScalar res = Literal(1); - for(Index i=0; i -void BDCSVD::computeSingVals(const ArrayRef& col0, const ArrayRef& diag, const IndicesRef &perm, - VectorType& singVals, ArrayRef shifts, ArrayRef mus) +template +void BDCSVD::computeSingVals(const ArrayRef &col0, + const ArrayRef &diag, + const IndicesRef &perm, + VectorType &singVals, + ArrayRef shifts, + ArrayRef mus) { using std::abs; using std::swap; @@ -716,159 +730,159 @@ void BDCSVD::computeSingVals(const ArrayRef& col0, const ArrayRef& d Index actual_n = n; // Note that here actual_n is computed based on col0(i)==0 instead of diag(i)==0 as above // because 1) we have diag(i)==0 => col0(i)==0 and 2) if col0(i)==0, then diag(i) is already a singular value. - while(actual_n>1 && col0(actual_n-1)==Literal(0)) --actual_n; + while (actual_n > 1 && col0(actual_n - 1) == Literal(0)) --actual_n; - for (Index k = 0; k < n; ++k) - { - if (col0(k) == Literal(0) || actual_n==1) - { + for (Index k = 0; k < n; ++k) { + if (col0(k) == Literal(0) || actual_n == 1) { // if col0(k) == 0, then entry is deflated, so singular value is on diagonal // if actual_n==1, then the deflated problem is already diagonalized - singVals(k) = k==0 ? col0(0) : diag(k); + singVals(k) = k == 0 ? col0(0) : diag(k); mus(k) = Literal(0); - shifts(k) = k==0 ? col0(0) : diag(k); + shifts(k) = k == 0 ? col0(0) : diag(k); continue; - } + } // otherwise, use secular equation to find singular value RealScalar left = diag(k); - RealScalar right; // was: = (k != actual_n-1) ? diag(k+1) : (diag(actual_n-1) + col0.matrix().norm()); - if(k==actual_n-1) - right = (diag(actual_n-1) + col0.matrix().norm()); - else - { + RealScalar right;// was: = (k != actual_n-1) ? diag(k+1) : (diag(actual_n-1) + col0.matrix().norm()); + if (k == actual_n - 1) + right = (diag(actual_n - 1) + col0.matrix().norm()); + else { // Skip deflated singular values, // recall that at this stage we assume that z[j]!=0 and all entries for which z[j]==0 have been put aside. // This should be equivalent to using perm[] - Index l = k+1; - while(col0(l)==Literal(0)) { ++l; eigen_internal_assert(l Literal(0)) ? left : right; - + RealScalar shift = (k == actual_n - 1 || fMid > Literal(0)) ? left : right; + // measure everything relative to shift - Map diagShifted(m_workspace.data()+4*n, n); + Map diagShifted(m_workspace.data() + 4 * n, n); diagShifted = diag - shift; - + // initial guess RealScalar muPrev, muCur; - if (shift == left) - { + if (shift == left) { muPrev = (right - left) * RealScalar(0.1); - if (k == actual_n-1) muCur = right - left; - else muCur = (right - left) * RealScalar(0.5); - } - else - { + if (k == actual_n - 1) + muCur = right - left; + else + muCur = (right - left) * RealScalar(0.5); + } else { muPrev = -(right - left) * RealScalar(0.1); muCur = -(right - left) * RealScalar(0.5); } RealScalar fPrev = secularEq(muPrev, col0, diag, perm, diagShifted, shift); RealScalar fCur = secularEq(muCur, col0, diag, perm, diagShifted, shift); - if (abs(fPrev) < abs(fCur)) - { + if (abs(fPrev) < abs(fCur)) { swap(fPrev, fCur); swap(muPrev, muCur); } // rational interpolation: fit a function of the form a / mu + b through the two previous // iterates and use its zero to compute the next iterate - bool useBisection = fPrev*fCur>Literal(0); - while (fCur!=Literal(0) && abs(muCur - muPrev) > Literal(8) * NumTraits::epsilon() * numext::maxi(abs(muCur), abs(muPrev)) && abs(fCur - fPrev)>NumTraits::epsilon() && !useBisection) - { + bool useBisection = fPrev * fCur > Literal(0); + while (fCur != Literal(0) + && abs(muCur - muPrev) + > Literal(8) * NumTraits::epsilon() * numext::maxi(abs(muCur), abs(muPrev)) + && abs(fCur - fPrev) > NumTraits::epsilon() && !useBisection) { ++m_numIters; // Find a and b such that the function f(mu) = a / mu + b matches the current and previous samples. - RealScalar a = (fCur - fPrev) / (Literal(1)/muCur - Literal(1)/muPrev); + RealScalar a = (fCur - fPrev) / (Literal(1) / muCur - Literal(1) / muPrev); RealScalar b = fCur - a / muCur; // And find mu such that f(mu)==0: - RealScalar muZero = -a/b; + RealScalar muZero = -a / b; RealScalar fZero = secularEq(muZero, col0, diag, perm, diagShifted, shift); - + muPrev = muCur; fPrev = fCur; muCur = muZero; fCur = fZero; - - - if (shift == left && (muCur < Literal(0) || muCur > right - left)) useBisection = true; + + + if (shift == left && (muCur < Literal(0) || muCur > right - left)) useBisection = true; if (shift == right && (muCur < -(right - left) || muCur > Literal(0))) useBisection = true; - if (abs(fCur)>abs(fPrev)) useBisection = true; + if (abs(fCur) > abs(fPrev)) useBisection = true; } // fall back on bisection method if rational interpolation did not work - if (useBisection) - { -#ifdef EIGEN_BDCSVD_DEBUG_VERBOSE + if (useBisection) { +#ifdef EIGEN_BDCSVD_DEBUG_VERBOSE std::cout << "useBisection for k = " << k << ", actual_n = " << actual_n << "\n"; #endif RealScalar leftShifted, rightShifted; - if (shift == left) - { + if (shift == left) { // to avoid overflow, we must have mu > max(real_min, |z(k)|/sqrt(real_max)), // the factor 2 is to be more conservative - leftShifted = numext::maxi( (std::numeric_limits::min)(), Literal(2) * abs(col0(k)) / sqrt((std::numeric_limits::max)()) ); + leftShifted = numext::maxi((std::numeric_limits::min)(), + Literal(2) * abs(col0(k)) / sqrt((std::numeric_limits::max)())); // check that we did it right: - eigen_internal_assert( (numext::isfinite)( (col0(k)/leftShifted)*(col0(k)/(diag(k)+shift+leftShifted)) ) ); + eigen_internal_assert( + (numext::isfinite)((col0(k) / leftShifted) * (col0(k) / (diag(k) + shift + leftShifted)))); // I don't understand why the case k==0 would be special there: // if (k == 0) rightShifted = right - left; else - rightShifted = (k==actual_n-1) ? right : ((right - left) * RealScalar(0.51)); // theoretically we can take 0.5, but let's be safe - } - else - { + rightShifted = (k == actual_n - 1) + ? right + : ((right - left) * RealScalar(0.51));// theoretically we can take 0.5, but let's be safe + } else { leftShifted = -(right - left) * RealScalar(0.51); - if(k+1( (std::numeric_limits::min)(), abs(col0(k+1)) / sqrt((std::numeric_limits::max)()) ); + if (k + 1 < n) + rightShifted = -numext::maxi((std::numeric_limits::min)(), + abs(col0(k + 1)) / sqrt((std::numeric_limits::max)())); else rightShifted = -(std::numeric_limits::min)(); } - + RealScalar fLeft = secularEq(leftShifted, col0, diag, perm, diagShifted, shift); #if defined EIGEN_INTERNAL_DEBUGGING || defined EIGEN_BDCSVD_DEBUG_VERBOSE RealScalar fRight = secularEq(rightShifted, col0, diag, perm, diagShifted, shift); #endif -#ifdef EIGEN_BDCSVD_DEBUG_VERBOSE - if(!(fLeft * fRight<0)) - { - std::cout << "fLeft: " << leftShifted << " - " << diagShifted.head(10).transpose() << "\n ; " << bool(left==shift) << " " << (left-shift) << "\n"; - std::cout << k << " : " << fLeft << " * " << fRight << " == " << fLeft * fRight << " ; " << left << " - " << right << " -> " << leftShifted << " " << rightShifted << " shift=" << shift << "\n"; +#ifdef EIGEN_BDCSVD_DEBUG_VERBOSE + if (!(fLeft * fRight < 0)) { + std::cout << "fLeft: " << leftShifted << " - " << diagShifted.head(10).transpose() << "\n ; " + << bool(left == shift) << " " << (left - shift) << "\n"; + std::cout << k << " : " << fLeft << " * " << fRight << " == " << fLeft * fRight << " ; " << left << " - " + << right << " -> " << leftShifted << " " << rightShifted << " shift=" << shift << "\n"; } #endif eigen_internal_assert(fLeft * fRight < Literal(0)); - - while (rightShifted - leftShifted > Literal(2) * NumTraits::epsilon() * numext::maxi(abs(leftShifted), abs(rightShifted))) - { + + while (rightShifted - leftShifted > Literal(2) * NumTraits::epsilon() + * numext::maxi(abs(leftShifted), abs(rightShifted))) { RealScalar midShifted = (leftShifted + rightShifted) / Literal(2); fMid = secularEq(midShifted, col0, diag, perm, diagShifted, shift); - if (fLeft * fMid < Literal(0)) - { + if (fLeft * fMid < Literal(0)) { rightShifted = midShifted; - } - else - { + } else { leftShifted = midShifted; fLeft = fMid; } @@ -876,7 +890,7 @@ void BDCSVD::computeSingVals(const ArrayRef& col0, const ArrayRef& d muCur = (leftShifted + rightShifted) / Literal(2); } - + singVals[k] = shift + muCur; shifts[k] = shift; mus[k] = muCur; @@ -884,54 +898,58 @@ void BDCSVD::computeSingVals(const ArrayRef& col0, const ArrayRef& d // perturb singular value slightly if it equals diagonal entry to avoid division by zero later // (deflation is supposed to avoid this from happening) // - this does no seem to be necessary anymore - -// if (singVals[k] == left) singVals[k] *= 1 + NumTraits::epsilon(); -// if (singVals[k] == right) singVals[k] *= 1 - NumTraits::epsilon(); + // if (singVals[k] == left) singVals[k] *= 1 + NumTraits::epsilon(); + // if (singVals[k] == right) singVals[k] *= 1 - NumTraits::epsilon(); } } // zhat is perturbation of col0 for which singular vectors can be computed stably (see Section 3.1) -template -void BDCSVD::perturbCol0 - (const ArrayRef& col0, const ArrayRef& diag, const IndicesRef &perm, const VectorType& singVals, - const ArrayRef& shifts, const ArrayRef& mus, ArrayRef zhat) +template +void BDCSVD::perturbCol0(const ArrayRef &col0, + const ArrayRef &diag, + const IndicesRef &perm, + const VectorType &singVals, + const ArrayRef &shifts, + const ArrayRef &mus, + ArrayRef zhat) { using std::sqrt; Index n = col0.size(); Index m = perm.size(); - if(m==0) - { + if (m == 0) { zhat.setZero(); return; } - Index last = perm(m-1); + Index last = perm(m - 1); // The offset permits to skip deflated entries while computing zhat - for (Index k = 0; k < n; ++k) - { - if (col0(k) == Literal(0)) // deflated + for (Index k = 0; k < n; ++k) { + if (col0(k) == Literal(0))// deflated zhat(k) = Literal(0); - else - { + else { // see equation (3.6) RealScalar dk = diag(k); RealScalar prod = (singVals(last) + dk) * (mus(last) + (shifts(last) - dk)); - for(Index l = 0; l 0.9 ) - std::cout << " " << ((singVals(j)+dk)*(mus(j)+(shifts(j)-dk)))/((diag(i)+dk)*(diag(i)-dk)) << " == (" << (singVals(j)+dk) << " * " << (mus(j)+(shifts(j)-dk)) - << ") / (" << (diag(i)+dk) << " * " << (diag(i)-dk) << ")\n"; + if (i != k + && std::abs(((singVals(j) + dk) * (mus(j) + (shifts(j) - dk))) / ((diag(i) + dk) * (diag(i) - dk)) - 1) + > 0.9) + std::cout << " " + << ((singVals(j) + dk) * (mus(j) + (shifts(j) - dk))) / ((diag(i) + dk) * (diag(i) - dk)) + << " == (" << (singVals(j) + dk) << " * " << (mus(j) + (shifts(j) - dk)) << ") / (" + << (diag(i) + dk) << " * " << (diag(i) - dk) << ")\n"; #endif } } #ifdef EIGEN_BDCSVD_DEBUG_VERBOSE - std::cout << "zhat(" << k << ") = sqrt( " << prod << ") ; " << (singVals(last) + dk) << " * " << mus(last) + shifts(last) << " - " << dk << "\n"; + std::cout << "zhat(" << k << ") = sqrt( " << prod << ") ; " << (singVals(last) + dk) << " * " + << mus(last) + shifts(last) << " - " << dk << "\n"; #endif RealScalar tmp = sqrt(prod); zhat(k) = col0(k) > Literal(0) ? tmp : -tmp; @@ -940,74 +958,72 @@ void BDCSVD::perturbCol0 } // compute singular vectors -template -void BDCSVD::computeSingVecs - (const ArrayRef& zhat, const ArrayRef& diag, const IndicesRef &perm, const VectorType& singVals, - const ArrayRef& shifts, const ArrayRef& mus, MatrixXr& U, MatrixXr& V) +template +void BDCSVD::computeSingVecs(const ArrayRef &zhat, + const ArrayRef &diag, + const IndicesRef &perm, + const VectorType &singVals, + const ArrayRef &shifts, + const ArrayRef &mus, + MatrixXr &U, + MatrixXr &V) { Index n = zhat.size(); Index m = perm.size(); - - for (Index k = 0; k < n; ++k) - { - if (zhat(k) == Literal(0)) - { - U.col(k) = VectorType::Unit(n+1, k); + + for (Index k = 0; k < n; ++k) { + if (zhat(k) == Literal(0)) { + U.col(k) = VectorType::Unit(n + 1, k); if (m_compV) V.col(k) = VectorType::Unit(n, k); - } - else - { + } else { U.col(k).setZero(); - for(Index l=0;l= 1, di almost null and zi non null. // We use a rotation to zero out zi applied to the left of M -template -void BDCSVD::deflation43(Index firstCol, Index shift, Index i, Index size) +template void BDCSVD::deflation43(Index firstCol, Index shift, Index i, Index size) { using std::abs; using std::sqrt; using std::pow; Index start = firstCol + shift; RealScalar c = m_computed(start, start); - RealScalar s = m_computed(start+i, start); - RealScalar r = numext::hypot(c,s); - if (r == Literal(0)) - { - m_computed(start+i, start+i) = Literal(0); + RealScalar s = m_computed(start + i, start); + RealScalar r = numext::hypot(c, s); + if (r == Literal(0)) { + m_computed(start + i, start + i) = Literal(0); return; } - m_computed(start,start) = r; - m_computed(start+i, start) = Literal(0); - m_computed(start+i, start+i) = Literal(0); - - JacobiRotation J(c/r,-s/r); - if (m_compU) m_naiveU.middleRows(firstCol, size+1).applyOnTheRight(firstCol, firstCol+i, J); - else m_naiveU.applyOnTheRight(firstCol, firstCol+i, J); + m_computed(start, start) = r; + m_computed(start + i, start) = Literal(0); + m_computed(start + i, start + i) = Literal(0); + + JacobiRotation J(c / r, -s / r); + if (m_compU) + m_naiveU.middleRows(firstCol, size + 1).applyOnTheRight(firstCol, firstCol + i, J); + else + m_naiveU.applyOnTheRight(firstCol, firstCol + i, J); }// end deflation 43 @@ -1015,97 +1031,107 @@ void BDCSVD::deflation43(Index firstCol, Index shift, Index i, Index // i,j >= 1, i!=j and |di - dj| < epsilon * norm2(M) // We apply two rotations to have zj = 0; // TODO deflation44 is still broken and not properly tested -template -void BDCSVD::deflation44(Index firstColu , Index firstColm, Index firstRowW, Index firstColW, Index i, Index j, Index size) +template +void BDCSVD::deflation44(Index firstColu, + Index firstColm, + Index firstRowW, + Index firstColW, + Index i, + Index j, + Index size) { using std::abs; using std::sqrt; using std::conj; using std::pow; - RealScalar c = m_computed(firstColm+i, firstColm); - RealScalar s = m_computed(firstColm+j, firstColm); + RealScalar c = m_computed(firstColm + i, firstColm); + RealScalar s = m_computed(firstColm + j, firstColm); RealScalar r = sqrt(numext::abs2(c) + numext::abs2(s)); -#ifdef EIGEN_BDCSVD_DEBUG_VERBOSE +#ifdef EIGEN_BDCSVD_DEBUG_VERBOSE std::cout << "deflation 4.4: " << i << "," << j << " -> " << c << " " << s << " " << r << " ; " - << m_computed(firstColm + i-1, firstColm) << " " - << m_computed(firstColm + i, firstColm) << " " - << m_computed(firstColm + i+1, firstColm) << " " - << m_computed(firstColm + i+2, firstColm) << "\n"; - std::cout << m_computed(firstColm + i-1, firstColm + i-1) << " " - << m_computed(firstColm + i, firstColm+i) << " " - << m_computed(firstColm + i+1, firstColm+i+1) << " " - << m_computed(firstColm + i+2, firstColm+i+2) << "\n"; + << m_computed(firstColm + i - 1, firstColm) << " " << m_computed(firstColm + i, firstColm) << " " + << m_computed(firstColm + i + 1, firstColm) << " " << m_computed(firstColm + i + 2, firstColm) << "\n"; + std::cout << m_computed(firstColm + i - 1, firstColm + i - 1) << " " << m_computed(firstColm + i, firstColm + i) + << " " << m_computed(firstColm + i + 1, firstColm + i + 1) << " " + << m_computed(firstColm + i + 2, firstColm + i + 2) << "\n"; #endif - if (r==Literal(0)) - { + if (r == Literal(0)) { m_computed(firstColm + i, firstColm + i) = m_computed(firstColm + j, firstColm + j); return; } - c/=r; - s/=r; - m_computed(firstColm + i, firstColm) = r; + c /= r; + s /= r; + m_computed(firstColm + i, firstColm) = r; m_computed(firstColm + j, firstColm + j) = m_computed(firstColm + i, firstColm + i); m_computed(firstColm + j, firstColm) = Literal(0); - JacobiRotation J(c,-s); - if (m_compU) m_naiveU.middleRows(firstColu, size+1).applyOnTheRight(firstColu + i, firstColu + j, J); - else m_naiveU.applyOnTheRight(firstColu+i, firstColu+j, J); - if (m_compV) m_naiveV.middleRows(firstRowW, size).applyOnTheRight(firstColW + i, firstColW + j, J); + JacobiRotation J(c, -s); + if (m_compU) + m_naiveU.middleRows(firstColu, size + 1).applyOnTheRight(firstColu + i, firstColu + j, J); + else + m_naiveU.applyOnTheRight(firstColu + i, firstColu + j, J); + if (m_compV) m_naiveV.middleRows(firstRowW, size).applyOnTheRight(firstColW + i, firstColW + j, J); }// end deflation 44 // acts on block from (firstCol+shift, firstCol+shift) to (lastCol+shift, lastCol+shift) [inclusive] -template -void BDCSVD::deflation(Index firstCol, Index lastCol, Index k, Index firstRowW, Index firstColW, Index shift) +template +void BDCSVD::deflation(Index firstCol, + Index lastCol, + Index k, + Index firstRowW, + Index firstColW, + Index shift) { using std::sqrt; using std::abs; const Index length = lastCol + 1 - firstCol; - - Block col0(m_computed, firstCol+shift, firstCol+shift, length, 1); + + Block col0(m_computed, firstCol + shift, firstCol + shift, length, 1); Diagonal fulldiag(m_computed); - VectorBlock,Dynamic> diag(fulldiag, firstCol+shift, length); - + VectorBlock, Dynamic> diag(fulldiag, firstCol + shift, length); + const RealScalar considerZero = (std::numeric_limits::min)(); - RealScalar maxDiag = diag.tail((std::max)(Index(1),length-1)).cwiseAbs().maxCoeff(); - RealScalar epsilon_strict = numext::maxi(considerZero,NumTraits::epsilon() * maxDiag); - RealScalar epsilon_coarse = Literal(8) * NumTraits::epsilon() * numext::maxi(col0.cwiseAbs().maxCoeff(), maxDiag); - + RealScalar maxDiag = diag.tail((std::max)(Index(1), length - 1)).cwiseAbs().maxCoeff(); + RealScalar epsilon_strict = numext::maxi(considerZero, NumTraits::epsilon() * maxDiag); + RealScalar epsilon_coarse = + Literal(8) * NumTraits::epsilon() * numext::maxi(col0.cwiseAbs().maxCoeff(), maxDiag); + #ifdef EIGEN_BDCSVD_SANITY_CHECKS assert(m_naiveU.allFinite()); assert(m_naiveV.allFinite()); assert(m_computed.allFinite()); #endif -#ifdef EIGEN_BDCSVD_DEBUG_VERBOSE - std::cout << "\ndeflate:" << diag.head(k+1).transpose() << " | " << diag.segment(k+1,length-k-1).transpose() << "\n"; +#ifdef EIGEN_BDCSVD_DEBUG_VERBOSE + std::cout << "\ndeflate:" << diag.head(k + 1).transpose() << " | " + << diag.segment(k + 1, length - k - 1).transpose() << "\n"; #endif - - //condition 4.1 - if (diag(0) < epsilon_coarse) - { -#ifdef EIGEN_BDCSVD_DEBUG_VERBOSE + + // condition 4.1 + if (diag(0) < epsilon_coarse) { +#ifdef EIGEN_BDCSVD_DEBUG_VERBOSE std::cout << "deflation 4.1, because " << diag(0) << " < " << epsilon_coarse << "\n"; #endif diag(0) = epsilon_coarse; } - //condition 4.2 - for (Index i=1;i::deflation(Index firstCol, Index lastCol, Index k, Index { // Check for total deflation // If we have a total deflation, then we have to consider col0(0)==diag(0) as a singular value during sorting - bool total_deflation = (col0.tail(length-1).array() k) permutation[p] = j++; - else if (j >= length) permutation[p] = i++; - else if (diag(i) < diag(j)) permutation[p] = j++; - else permutation[p] = i++; + for (Index i = 1; i < length; ++i) + if (abs(diag(i)) < considerZero) permutation[p++] = i; + + Index i = 1, j = k + 1; + for (; p < length; ++p) { + if (i > k) + permutation[p] = j++; + else if (j >= length) + permutation[p] = i++; + else if (diag(i) < diag(j)) + permutation[p] = j++; + else + permutation[p] = i++; } } - + // If we have a total deflation, then we have to insert diag(0) at the right place - if(total_deflation) - { - for(Index i=1; i::deflation(Index firstCol, Index lastCol, Index k, Index std::cout << "sorted: " << diag.transpose().format(bdcsvdfmt) << "\n"; std::cout << " : " << col0.transpose() << "\n\n"; #endif - - //condition 4.4 + + // condition 4.4 { - Index i = length-1; - while(i>0 && (abs(diag(i))1;--i) - if( (diag(i) - diag(i-1)) < NumTraits::epsilon()*maxDiag ) - { + Index i = length - 1; + while (i > 0 && (abs(diag(i)) < considerZero || abs(col0(i)) < considerZero)) --i; + for (; i > 1; --i) + if ((diag(i) - diag(i - 1)) < NumTraits::epsilon() * maxDiag) { #ifdef EIGEN_BDCSVD_DEBUG_VERBOSE - std::cout << "deflation 4.4 with i = " << i << " because " << (diag(i) - diag(i-1)) << " < " << NumTraits::epsilon()*diag(i) << "\n"; + std::cout << "deflation 4.4 with i = " << i << " because " << (diag(i) - diag(i - 1)) << " < " + << NumTraits::epsilon() * diag(i) << "\n"; #endif - eigen_internal_assert(abs(diag(i) - diag(i-1)) -BDCSVD::PlainObject> -MatrixBase::bdcSvd(unsigned int computationOptions) const +BDCSVD::PlainObject> MatrixBase::bdcSvd(unsigned int computationOptions) const { return BDCSVD(*this, computationOptions); } #endif -} // end namespace Eigen +}// end namespace Eigen #endif diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/SVD/JacobiSVD.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/SVD/JacobiSVD.h index 43488b1e..7598c957 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/SVD/JacobiSVD.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/SVD/JacobiSVD.h @@ -11,602 +11,597 @@ #ifndef EIGEN_JACOBISVD_H #define EIGEN_JACOBISVD_H -namespace Eigen { +namespace Eigen { namespace internal { -// forward declaration (needed by ICC) -// the empty body is required by MSVC -template::IsComplex> -struct svd_precondition_2x2_block_to_be_real {}; - -/*** QR preconditioners (R-SVD) - *** - *** Their role is to reduce the problem of computing the SVD to the case of a square matrix. - *** This approach, known as R-SVD, is an optimization for rectangular-enough matrices, and is a requirement for - *** JacobiSVD which by itself is only able to work on square matrices. - ***/ - -enum { PreconditionIfMoreColsThanRows, PreconditionIfMoreRowsThanCols }; - -template -struct qr_preconditioner_should_do_anything -{ - enum { a = MatrixType::RowsAtCompileTime != Dynamic && - MatrixType::ColsAtCompileTime != Dynamic && - MatrixType::ColsAtCompileTime <= MatrixType::RowsAtCompileTime, - b = MatrixType::RowsAtCompileTime != Dynamic && - MatrixType::ColsAtCompileTime != Dynamic && - MatrixType::RowsAtCompileTime <= MatrixType::ColsAtCompileTime, - ret = !( (QRPreconditioner == NoQRPreconditioner) || - (Case == PreconditionIfMoreColsThanRows && bool(a)) || - (Case == PreconditionIfMoreRowsThanCols && bool(b)) ) + // forward declaration (needed by ICC) + // the empty body is required by MSVC + template::IsComplex> + struct svd_precondition_2x2_block_to_be_real + { }; -}; - -template::ret -> struct qr_preconditioner_impl {}; -template -class qr_preconditioner_impl -{ -public: - void allocate(const JacobiSVD&) {} - bool run(JacobiSVD&, const MatrixType&) - { - return false; - } -}; + /*** QR preconditioners (R-SVD) + *** + *** Their role is to reduce the problem of computing the SVD to the case of a square matrix. + *** This approach, known as R-SVD, is an optimization for rectangular-enough matrices, and is a requirement for + *** JacobiSVD which by itself is only able to work on square matrices. + ***/ -/*** preconditioner using FullPivHouseholderQR ***/ + enum { PreconditionIfMoreColsThanRows, PreconditionIfMoreRowsThanCols }; -template -class qr_preconditioner_impl -{ -public: - typedef typename MatrixType::Scalar Scalar; - enum + template struct qr_preconditioner_should_do_anything { - RowsAtCompileTime = MatrixType::RowsAtCompileTime, - MaxRowsAtCompileTime = MatrixType::MaxRowsAtCompileTime + enum { + a = MatrixType::RowsAtCompileTime != Dynamic && MatrixType::ColsAtCompileTime != Dynamic + && MatrixType::ColsAtCompileTime <= MatrixType::RowsAtCompileTime, + b = MatrixType::RowsAtCompileTime != Dynamic && MatrixType::ColsAtCompileTime != Dynamic + && MatrixType::RowsAtCompileTime <= MatrixType::ColsAtCompileTime, + ret = !((QRPreconditioner == NoQRPreconditioner) || (Case == PreconditionIfMoreColsThanRows && bool(a)) + || (Case == PreconditionIfMoreRowsThanCols && bool(b))) + }; }; - typedef Matrix WorkspaceType; - void allocate(const JacobiSVD& svd) + template::ret> + struct qr_preconditioner_impl { - if (svd.rows() != m_qr.rows() || svd.cols() != m_qr.cols()) - { - m_qr.~QRType(); - ::new (&m_qr) QRType(svd.rows(), svd.cols()); - } - if (svd.m_computeFullU) m_workspace.resize(svd.rows()); - } + }; - bool run(JacobiSVD& svd, const MatrixType& matrix) + template + class qr_preconditioner_impl { - if(matrix.rows() > matrix.cols()) - { - m_qr.compute(matrix); - svd.m_workMatrix = m_qr.matrixQR().block(0,0,matrix.cols(),matrix.cols()).template triangularView(); - if(svd.m_computeFullU) m_qr.matrixQ().evalTo(svd.m_matrixU, m_workspace); - if(svd.computeV()) svd.m_matrixV = m_qr.colsPermutation(); - return true; - } - return false; - } -private: - typedef FullPivHouseholderQR QRType; - QRType m_qr; - WorkspaceType m_workspace; -}; - -template -class qr_preconditioner_impl -{ -public: - typedef typename MatrixType::Scalar Scalar; - enum - { - RowsAtCompileTime = MatrixType::RowsAtCompileTime, - ColsAtCompileTime = MatrixType::ColsAtCompileTime, - MaxRowsAtCompileTime = MatrixType::MaxRowsAtCompileTime, - MaxColsAtCompileTime = MatrixType::MaxColsAtCompileTime, - TrOptions = RowsAtCompileTime==1 ? (MatrixType::Options & ~(RowMajor)) - : ColsAtCompileTime==1 ? (MatrixType::Options | RowMajor) - : MatrixType::Options + public: + void allocate(const JacobiSVD &) {} + bool run(JacobiSVD &, const MatrixType &) { return false; } }; - typedef Matrix - TransposeTypeWithSameStorageOrder; - void allocate(const JacobiSVD& svd) - { - if (svd.cols() != m_qr.rows() || svd.rows() != m_qr.cols()) - { - m_qr.~QRType(); - ::new (&m_qr) QRType(svd.cols(), svd.rows()); - } - m_adjoint.resize(svd.cols(), svd.rows()); - if (svd.m_computeFullV) m_workspace.resize(svd.cols()); - } + /*** preconditioner using FullPivHouseholderQR ***/ - bool run(JacobiSVD& svd, const MatrixType& matrix) + template + class qr_preconditioner_impl { - if(matrix.cols() > matrix.rows()) - { - m_adjoint = matrix.adjoint(); - m_qr.compute(m_adjoint); - svd.m_workMatrix = m_qr.matrixQR().block(0,0,matrix.rows(),matrix.rows()).template triangularView().adjoint(); - if(svd.m_computeFullV) m_qr.matrixQ().evalTo(svd.m_matrixV, m_workspace); - if(svd.computeU()) svd.m_matrixU = m_qr.colsPermutation(); - return true; - } - else return false; - } -private: - typedef FullPivHouseholderQR QRType; - QRType m_qr; - TransposeTypeWithSameStorageOrder m_adjoint; - typename internal::plain_row_type::type m_workspace; -}; - -/*** preconditioner using ColPivHouseholderQR ***/ + public: + typedef typename MatrixType::Scalar Scalar; + enum { RowsAtCompileTime = MatrixType::RowsAtCompileTime, MaxRowsAtCompileTime = MatrixType::MaxRowsAtCompileTime }; + typedef Matrix WorkspaceType; -template -class qr_preconditioner_impl -{ -public: - void allocate(const JacobiSVD& svd) - { - if (svd.rows() != m_qr.rows() || svd.cols() != m_qr.cols()) + void allocate(const JacobiSVD &svd) { - m_qr.~QRType(); - ::new (&m_qr) QRType(svd.rows(), svd.cols()); + if (svd.rows() != m_qr.rows() || svd.cols() != m_qr.cols()) { + m_qr.~QRType(); + ::new (&m_qr) QRType(svd.rows(), svd.cols()); + } + if (svd.m_computeFullU) m_workspace.resize(svd.rows()); } - if (svd.m_computeFullU) m_workspace.resize(svd.rows()); - else if (svd.m_computeThinU) m_workspace.resize(svd.cols()); - } - bool run(JacobiSVD& svd, const MatrixType& matrix) - { - if(matrix.rows() > matrix.cols()) + bool run(JacobiSVD &svd, const MatrixType &matrix) { - m_qr.compute(matrix); - svd.m_workMatrix = m_qr.matrixQR().block(0,0,matrix.cols(),matrix.cols()).template triangularView(); - if(svd.m_computeFullU) m_qr.householderQ().evalTo(svd.m_matrixU, m_workspace); - else if(svd.m_computeThinU) - { - svd.m_matrixU.setIdentity(matrix.rows(), matrix.cols()); - m_qr.householderQ().applyThisOnTheLeft(svd.m_matrixU, m_workspace); + if (matrix.rows() > matrix.cols()) { + m_qr.compute(matrix); + svd.m_workMatrix = m_qr.matrixQR().block(0, 0, matrix.cols(), matrix.cols()).template triangularView(); + if (svd.m_computeFullU) m_qr.matrixQ().evalTo(svd.m_matrixU, m_workspace); + if (svd.computeV()) svd.m_matrixV = m_qr.colsPermutation(); + return true; } - if(svd.computeV()) svd.m_matrixV = m_qr.colsPermutation(); - return true; + return false; } - return false; - } -private: - typedef ColPivHouseholderQR QRType; - QRType m_qr; - typename internal::plain_col_type::type m_workspace; -}; - -template -class qr_preconditioner_impl -{ -public: - typedef typename MatrixType::Scalar Scalar; - enum - { - RowsAtCompileTime = MatrixType::RowsAtCompileTime, - ColsAtCompileTime = MatrixType::ColsAtCompileTime, - MaxRowsAtCompileTime = MatrixType::MaxRowsAtCompileTime, - MaxColsAtCompileTime = MatrixType::MaxColsAtCompileTime, - TrOptions = RowsAtCompileTime==1 ? (MatrixType::Options & ~(RowMajor)) - : ColsAtCompileTime==1 ? (MatrixType::Options | RowMajor) - : MatrixType::Options + private: + typedef FullPivHouseholderQR QRType; + QRType m_qr; + WorkspaceType m_workspace; }; - typedef Matrix - TransposeTypeWithSameStorageOrder; - - void allocate(const JacobiSVD& svd) + template + class qr_preconditioner_impl { - if (svd.cols() != m_qr.rows() || svd.rows() != m_qr.cols()) + public: + typedef typename MatrixType::Scalar Scalar; + enum { + RowsAtCompileTime = MatrixType::RowsAtCompileTime, + ColsAtCompileTime = MatrixType::ColsAtCompileTime, + MaxRowsAtCompileTime = MatrixType::MaxRowsAtCompileTime, + MaxColsAtCompileTime = MatrixType::MaxColsAtCompileTime, + TrOptions = RowsAtCompileTime == 1 ? (MatrixType::Options & ~(RowMajor)) + : ColsAtCompileTime == 1 ? (MatrixType::Options | RowMajor) + : MatrixType::Options + }; + typedef Matrix + TransposeTypeWithSameStorageOrder; + + void allocate(const JacobiSVD &svd) { - m_qr.~QRType(); - ::new (&m_qr) QRType(svd.cols(), svd.rows()); + if (svd.cols() != m_qr.rows() || svd.rows() != m_qr.cols()) { + m_qr.~QRType(); + ::new (&m_qr) QRType(svd.cols(), svd.rows()); + } + m_adjoint.resize(svd.cols(), svd.rows()); + if (svd.m_computeFullV) m_workspace.resize(svd.cols()); } - if (svd.m_computeFullV) m_workspace.resize(svd.cols()); - else if (svd.m_computeThinV) m_workspace.resize(svd.rows()); - m_adjoint.resize(svd.cols(), svd.rows()); - } - bool run(JacobiSVD& svd, const MatrixType& matrix) - { - if(matrix.cols() > matrix.rows()) + bool run(JacobiSVD &svd, const MatrixType &matrix) { - m_adjoint = matrix.adjoint(); - m_qr.compute(m_adjoint); - - svd.m_workMatrix = m_qr.matrixQR().block(0,0,matrix.rows(),matrix.rows()).template triangularView().adjoint(); - if(svd.m_computeFullV) m_qr.householderQ().evalTo(svd.m_matrixV, m_workspace); - else if(svd.m_computeThinV) - { - svd.m_matrixV.setIdentity(matrix.cols(), matrix.rows()); - m_qr.householderQ().applyThisOnTheLeft(svd.m_matrixV, m_workspace); - } - if(svd.computeU()) svd.m_matrixU = m_qr.colsPermutation(); - return true; + if (matrix.cols() > matrix.rows()) { + m_adjoint = matrix.adjoint(); + m_qr.compute(m_adjoint); + svd.m_workMatrix = + m_qr.matrixQR().block(0, 0, matrix.rows(), matrix.rows()).template triangularView().adjoint(); + if (svd.m_computeFullV) m_qr.matrixQ().evalTo(svd.m_matrixV, m_workspace); + if (svd.computeU()) svd.m_matrixU = m_qr.colsPermutation(); + return true; + } else + return false; } - else return false; - } -private: - typedef ColPivHouseholderQR QRType; - QRType m_qr; - TransposeTypeWithSameStorageOrder m_adjoint; - typename internal::plain_row_type::type m_workspace; -}; + private: + typedef FullPivHouseholderQR QRType; + QRType m_qr; + TransposeTypeWithSameStorageOrder m_adjoint; + typename internal::plain_row_type::type m_workspace; + }; -/*** preconditioner using HouseholderQR ***/ + /*** preconditioner using ColPivHouseholderQR ***/ -template -class qr_preconditioner_impl -{ -public: - void allocate(const JacobiSVD& svd) + template + class qr_preconditioner_impl { - if (svd.rows() != m_qr.rows() || svd.cols() != m_qr.cols()) + public: + void allocate(const JacobiSVD &svd) { - m_qr.~QRType(); - ::new (&m_qr) QRType(svd.rows(), svd.cols()); + if (svd.rows() != m_qr.rows() || svd.cols() != m_qr.cols()) { + m_qr.~QRType(); + ::new (&m_qr) QRType(svd.rows(), svd.cols()); + } + if (svd.m_computeFullU) + m_workspace.resize(svd.rows()); + else if (svd.m_computeThinU) + m_workspace.resize(svd.cols()); } - if (svd.m_computeFullU) m_workspace.resize(svd.rows()); - else if (svd.m_computeThinU) m_workspace.resize(svd.cols()); - } - bool run(JacobiSVD& svd, const MatrixType& matrix) - { - if(matrix.rows() > matrix.cols()) + bool run(JacobiSVD &svd, const MatrixType &matrix) { - m_qr.compute(matrix); - svd.m_workMatrix = m_qr.matrixQR().block(0,0,matrix.cols(),matrix.cols()).template triangularView(); - if(svd.m_computeFullU) m_qr.householderQ().evalTo(svd.m_matrixU, m_workspace); - else if(svd.m_computeThinU) - { - svd.m_matrixU.setIdentity(matrix.rows(), matrix.cols()); - m_qr.householderQ().applyThisOnTheLeft(svd.m_matrixU, m_workspace); + if (matrix.rows() > matrix.cols()) { + m_qr.compute(matrix); + svd.m_workMatrix = m_qr.matrixQR().block(0, 0, matrix.cols(), matrix.cols()).template triangularView(); + if (svd.m_computeFullU) + m_qr.householderQ().evalTo(svd.m_matrixU, m_workspace); + else if (svd.m_computeThinU) { + svd.m_matrixU.setIdentity(matrix.rows(), matrix.cols()); + m_qr.householderQ().applyThisOnTheLeft(svd.m_matrixU, m_workspace); + } + if (svd.computeV()) svd.m_matrixV = m_qr.colsPermutation(); + return true; } - if(svd.computeV()) svd.m_matrixV.setIdentity(matrix.cols(), matrix.cols()); - return true; + return false; } - return false; - } -private: - typedef HouseholderQR QRType; - QRType m_qr; - typename internal::plain_col_type::type m_workspace; -}; -template -class qr_preconditioner_impl -{ -public: - typedef typename MatrixType::Scalar Scalar; - enum - { - RowsAtCompileTime = MatrixType::RowsAtCompileTime, - ColsAtCompileTime = MatrixType::ColsAtCompileTime, - MaxRowsAtCompileTime = MatrixType::MaxRowsAtCompileTime, - MaxColsAtCompileTime = MatrixType::MaxColsAtCompileTime, - Options = MatrixType::Options + private: + typedef ColPivHouseholderQR QRType; + QRType m_qr; + typename internal::plain_col_type::type m_workspace; }; - typedef Matrix - TransposeTypeWithSameStorageOrder; - - void allocate(const JacobiSVD& svd) + template + class qr_preconditioner_impl { - if (svd.cols() != m_qr.rows() || svd.rows() != m_qr.cols()) - { - m_qr.~QRType(); - ::new (&m_qr) QRType(svd.cols(), svd.rows()); - } - if (svd.m_computeFullV) m_workspace.resize(svd.cols()); - else if (svd.m_computeThinV) m_workspace.resize(svd.rows()); - m_adjoint.resize(svd.cols(), svd.rows()); - } + public: + typedef typename MatrixType::Scalar Scalar; + enum { + RowsAtCompileTime = MatrixType::RowsAtCompileTime, + ColsAtCompileTime = MatrixType::ColsAtCompileTime, + MaxRowsAtCompileTime = MatrixType::MaxRowsAtCompileTime, + MaxColsAtCompileTime = MatrixType::MaxColsAtCompileTime, + TrOptions = RowsAtCompileTime == 1 ? (MatrixType::Options & ~(RowMajor)) + : ColsAtCompileTime == 1 ? (MatrixType::Options | RowMajor) + : MatrixType::Options + }; - bool run(JacobiSVD& svd, const MatrixType& matrix) - { - if(matrix.cols() > matrix.rows()) + typedef Matrix + TransposeTypeWithSameStorageOrder; + + void allocate(const JacobiSVD &svd) { - m_adjoint = matrix.adjoint(); - m_qr.compute(m_adjoint); - - svd.m_workMatrix = m_qr.matrixQR().block(0,0,matrix.rows(),matrix.rows()).template triangularView().adjoint(); - if(svd.m_computeFullV) m_qr.householderQ().evalTo(svd.m_matrixV, m_workspace); - else if(svd.m_computeThinV) - { - svd.m_matrixV.setIdentity(matrix.cols(), matrix.rows()); - m_qr.householderQ().applyThisOnTheLeft(svd.m_matrixV, m_workspace); + if (svd.cols() != m_qr.rows() || svd.rows() != m_qr.cols()) { + m_qr.~QRType(); + ::new (&m_qr) QRType(svd.cols(), svd.rows()); } - if(svd.computeU()) svd.m_matrixU.setIdentity(matrix.rows(), matrix.rows()); - return true; + if (svd.m_computeFullV) + m_workspace.resize(svd.cols()); + else if (svd.m_computeThinV) + m_workspace.resize(svd.rows()); + m_adjoint.resize(svd.cols(), svd.rows()); } - else return false; - } -private: - typedef HouseholderQR QRType; - QRType m_qr; - TransposeTypeWithSameStorageOrder m_adjoint; - typename internal::plain_row_type::type m_workspace; -}; + bool run(JacobiSVD &svd, const MatrixType &matrix) + { + if (matrix.cols() > matrix.rows()) { + m_adjoint = matrix.adjoint(); + m_qr.compute(m_adjoint); + + svd.m_workMatrix = + m_qr.matrixQR().block(0, 0, matrix.rows(), matrix.rows()).template triangularView().adjoint(); + if (svd.m_computeFullV) + m_qr.householderQ().evalTo(svd.m_matrixV, m_workspace); + else if (svd.m_computeThinV) { + svd.m_matrixV.setIdentity(matrix.cols(), matrix.rows()); + m_qr.householderQ().applyThisOnTheLeft(svd.m_matrixV, m_workspace); + } + if (svd.computeU()) svd.m_matrixU = m_qr.colsPermutation(); + return true; + } else + return false; + } -/*** 2x2 SVD implementation - *** - *** JacobiSVD consists in performing a series of 2x2 SVD subproblems - ***/ + private: + typedef ColPivHouseholderQR QRType; + QRType m_qr; + TransposeTypeWithSameStorageOrder m_adjoint; + typename internal::plain_row_type::type m_workspace; + }; -template -struct svd_precondition_2x2_block_to_be_real -{ - typedef JacobiSVD SVD; - typedef typename MatrixType::RealScalar RealScalar; - static bool run(typename SVD::WorkMatrixType&, SVD&, Index, Index, RealScalar&) { return true; } -}; + /*** preconditioner using HouseholderQR ***/ -template -struct svd_precondition_2x2_block_to_be_real -{ - typedef JacobiSVD SVD; - typedef typename MatrixType::Scalar Scalar; - typedef typename MatrixType::RealScalar RealScalar; - static bool run(typename SVD::WorkMatrixType& work_matrix, SVD& svd, Index p, Index q, RealScalar& maxDiagEntry) + template + class qr_preconditioner_impl { - using std::sqrt; - using std::abs; - Scalar z; - JacobiRotation rot; - RealScalar n = sqrt(numext::abs2(work_matrix.coeff(p,p)) + numext::abs2(work_matrix.coeff(q,p))); - - const RealScalar considerAsZero = (std::numeric_limits::min)(); - const RealScalar precision = NumTraits::epsilon(); - - if(n==0) + public: + void allocate(const JacobiSVD &svd) { - // make sure first column is zero - work_matrix.coeffRef(p,p) = work_matrix.coeffRef(q,p) = Scalar(0); - - if(abs(numext::imag(work_matrix.coeff(p,q)))>considerAsZero) - { - // work_matrix.coeff(p,q) can be zero if work_matrix.coeff(q,p) is not zero but small enough to underflow when computing n - z = abs(work_matrix.coeff(p,q)) / work_matrix.coeff(p,q); - work_matrix.row(p) *= z; - if(svd.computeU()) svd.m_matrixU.col(p) *= conj(z); - } - if(abs(numext::imag(work_matrix.coeff(q,q)))>considerAsZero) - { - z = abs(work_matrix.coeff(q,q)) / work_matrix.coeff(q,q); - work_matrix.row(q) *= z; - if(svd.computeU()) svd.m_matrixU.col(q) *= conj(z); + if (svd.rows() != m_qr.rows() || svd.cols() != m_qr.cols()) { + m_qr.~QRType(); + ::new (&m_qr) QRType(svd.rows(), svd.cols()); } - // otherwise the second row is already zero, so we have nothing to do. + if (svd.m_computeFullU) + m_workspace.resize(svd.rows()); + else if (svd.m_computeThinU) + m_workspace.resize(svd.cols()); } - else + + bool run(JacobiSVD &svd, const MatrixType &matrix) { - rot.c() = conj(work_matrix.coeff(p,p)) / n; - rot.s() = work_matrix.coeff(q,p) / n; - work_matrix.applyOnTheLeft(p,q,rot); - if(svd.computeU()) svd.m_matrixU.applyOnTheRight(p,q,rot.adjoint()); - if(abs(numext::imag(work_matrix.coeff(p,q)))>considerAsZero) - { - z = abs(work_matrix.coeff(p,q)) / work_matrix.coeff(p,q); - work_matrix.col(q) *= z; - if(svd.computeV()) svd.m_matrixV.col(q) *= z; - } - if(abs(numext::imag(work_matrix.coeff(q,q)))>considerAsZero) - { - z = abs(work_matrix.coeff(q,q)) / work_matrix.coeff(q,q); - work_matrix.row(q) *= z; - if(svd.computeU()) svd.m_matrixU.col(q) *= conj(z); + if (matrix.rows() > matrix.cols()) { + m_qr.compute(matrix); + svd.m_workMatrix = m_qr.matrixQR().block(0, 0, matrix.cols(), matrix.cols()).template triangularView(); + if (svd.m_computeFullU) + m_qr.householderQ().evalTo(svd.m_matrixU, m_workspace); + else if (svd.m_computeThinU) { + svd.m_matrixU.setIdentity(matrix.rows(), matrix.cols()); + m_qr.householderQ().applyThisOnTheLeft(svd.m_matrixU, m_workspace); + } + if (svd.computeV()) svd.m_matrixV.setIdentity(matrix.cols(), matrix.cols()); + return true; } + return false; } - // update largest diagonal entry - maxDiagEntry = numext::maxi(maxDiagEntry,numext::maxi(abs(work_matrix.coeff(p,p)), abs(work_matrix.coeff(q,q)))); - // and check whether the 2x2 block is already diagonal - RealScalar threshold = numext::maxi(considerAsZero, precision * maxDiagEntry); - return abs(work_matrix.coeff(p,q))>threshold || abs(work_matrix.coeff(q,p)) > threshold; - } -}; - -template -struct traits > -{ - typedef _MatrixType MatrixType; -}; - -} // end namespace internal + private: + typedef HouseholderQR QRType; + QRType m_qr; + typename internal::plain_col_type::type m_workspace; + }; -/** \ingroup SVD_Module - * - * - * \class JacobiSVD - * - * \brief Two-sided Jacobi SVD decomposition of a rectangular matrix - * - * \tparam _MatrixType the type of the matrix of which we are computing the SVD decomposition - * \tparam QRPreconditioner this optional parameter allows to specify the type of QR decomposition that will be used internally - * for the R-SVD step for non-square matrices. See discussion of possible values below. - * - * SVD decomposition consists in decomposing any n-by-p matrix \a A as a product - * \f[ A = U S V^* \f] - * where \a U is a n-by-n unitary, \a V is a p-by-p unitary, and \a S is a n-by-p real positive matrix which is zero outside of its main diagonal; - * the diagonal entries of S are known as the \em singular \em values of \a A and the columns of \a U and \a V are known as the left - * and right \em singular \em vectors of \a A respectively. - * - * Singular values are always sorted in decreasing order. - * - * This JacobiSVD decomposition computes only the singular values by default. If you want \a U or \a V, you need to ask for them explicitly. - * - * You can ask for only \em thin \a U or \a V to be computed, meaning the following. In case of a rectangular n-by-p matrix, letting \a m be the - * smaller value among \a n and \a p, there are only \a m singular vectors; the remaining columns of \a U and \a V do not correspond to actual - * singular vectors. Asking for \em thin \a U or \a V means asking for only their \a m first columns to be formed. So \a U is then a n-by-m matrix, - * and \a V is then a p-by-m matrix. Notice that thin \a U and \a V are all you need for (least squares) solving. - * - * Here's an example demonstrating basic usage: - * \include JacobiSVD_basic.cpp - * Output: \verbinclude JacobiSVD_basic.out - * - * This JacobiSVD class is a two-sided Jacobi R-SVD decomposition, ensuring optimal reliability and accuracy. The downside is that it's slower than - * bidiagonalizing SVD algorithms for large square matrices; however its complexity is still \f$ O(n^2p) \f$ where \a n is the smaller dimension and - * \a p is the greater dimension, meaning that it is still of the same order of complexity as the faster bidiagonalizing R-SVD algorithms. - * In particular, like any R-SVD, it takes advantage of non-squareness in that its complexity is only linear in the greater dimension. - * - * If the input matrix has inf or nan coefficients, the result of the computation is undefined, but the computation is guaranteed to - * terminate in finite (and reasonable) time. - * - * The possible values for QRPreconditioner are: - * \li ColPivHouseholderQRPreconditioner is the default. In practice it's very safe. It uses column-pivoting QR. - * \li FullPivHouseholderQRPreconditioner, is the safest and slowest. It uses full-pivoting QR. - * Contrary to other QRs, it doesn't allow computing thin unitaries. - * \li HouseholderQRPreconditioner is the fastest, and less safe and accurate than the pivoting variants. It uses non-pivoting QR. - * This is very similar in safety and accuracy to the bidiagonalization process used by bidiagonalizing SVD algorithms (since bidiagonalization - * is inherently non-pivoting). However the resulting SVD is still more reliable than bidiagonalizing SVDs because the Jacobi-based iterarive - * process is more reliable than the optimized bidiagonal SVD iterations. - * \li NoQRPreconditioner allows not to use a QR preconditioner at all. This is useful if you know that you will only be computing - * JacobiSVD decompositions of square matrices. Non-square matrices require a QR preconditioner. Using this option will result in - * faster compilation and smaller executable code. It won't significantly speed up computation, since JacobiSVD is always checking - * if QR preconditioning is needed before applying it anyway. - * - * \sa MatrixBase::jacobiSvd() - */ -template class JacobiSVD - : public SVDBase > -{ - typedef SVDBase Base; + template + class qr_preconditioner_impl + { public: - - typedef _MatrixType MatrixType; typedef typename MatrixType::Scalar Scalar; - typedef typename NumTraits::Real RealScalar; enum { RowsAtCompileTime = MatrixType::RowsAtCompileTime, ColsAtCompileTime = MatrixType::ColsAtCompileTime, - DiagSizeAtCompileTime = EIGEN_SIZE_MIN_PREFER_DYNAMIC(RowsAtCompileTime,ColsAtCompileTime), MaxRowsAtCompileTime = MatrixType::MaxRowsAtCompileTime, MaxColsAtCompileTime = MatrixType::MaxColsAtCompileTime, - MaxDiagSizeAtCompileTime = EIGEN_SIZE_MIN_PREFER_FIXED(MaxRowsAtCompileTime,MaxColsAtCompileTime), - MatrixOptions = MatrixType::Options + Options = MatrixType::Options }; - typedef typename Base::MatrixUType MatrixUType; - typedef typename Base::MatrixVType MatrixVType; - typedef typename Base::SingularValuesType SingularValuesType; - - typedef typename internal::plain_row_type::type RowType; - typedef typename internal::plain_col_type::type ColType; - typedef Matrix - WorkMatrixType; - - /** \brief Default Constructor. - * - * The default constructor is useful in cases in which the user intends to - * perform decompositions via JacobiSVD::compute(const MatrixType&). - */ - JacobiSVD() - {} - - - /** \brief Default Constructor with memory preallocation - * - * Like the default constructor but with preallocation of the internal data - * according to the specified problem size. - * \sa JacobiSVD() - */ - JacobiSVD(Index rows, Index cols, unsigned int computationOptions = 0) + typedef Matrix + TransposeTypeWithSameStorageOrder; + + void allocate(const JacobiSVD &svd) { - allocate(rows, cols, computationOptions); + if (svd.cols() != m_qr.rows() || svd.rows() != m_qr.cols()) { + m_qr.~QRType(); + ::new (&m_qr) QRType(svd.cols(), svd.rows()); + } + if (svd.m_computeFullV) + m_workspace.resize(svd.cols()); + else if (svd.m_computeThinV) + m_workspace.resize(svd.rows()); + m_adjoint.resize(svd.cols(), svd.rows()); } - /** \brief Constructor performing the decomposition of given matrix. - * - * \param matrix the matrix to decompose - * \param computationOptions optional parameter allowing to specify if you want full or thin U or V unitaries to be computed. - * By default, none is computed. This is a bit-field, the possible bits are #ComputeFullU, #ComputeThinU, - * #ComputeFullV, #ComputeThinV. - * - * Thin unitaries are only available if your matrix type has a Dynamic number of columns (for example MatrixXf). They also are not - * available with the (non-default) FullPivHouseholderQR preconditioner. - */ - explicit JacobiSVD(const MatrixType& matrix, unsigned int computationOptions = 0) + bool run(JacobiSVD &svd, const MatrixType &matrix) { - compute(matrix, computationOptions); + if (matrix.cols() > matrix.rows()) { + m_adjoint = matrix.adjoint(); + m_qr.compute(m_adjoint); + + svd.m_workMatrix = + m_qr.matrixQR().block(0, 0, matrix.rows(), matrix.rows()).template triangularView().adjoint(); + if (svd.m_computeFullV) + m_qr.householderQ().evalTo(svd.m_matrixV, m_workspace); + else if (svd.m_computeThinV) { + svd.m_matrixV.setIdentity(matrix.cols(), matrix.rows()); + m_qr.householderQ().applyThisOnTheLeft(svd.m_matrixV, m_workspace); + } + if (svd.computeU()) svd.m_matrixU.setIdentity(matrix.rows(), matrix.rows()); + return true; + } else + return false; } - /** \brief Method performing the decomposition of given matrix using custom options. - * - * \param matrix the matrix to decompose - * \param computationOptions optional parameter allowing to specify if you want full or thin U or V unitaries to be computed. - * By default, none is computed. This is a bit-field, the possible bits are #ComputeFullU, #ComputeThinU, - * #ComputeFullV, #ComputeThinV. - * - * Thin unitaries are only available if your matrix type has a Dynamic number of columns (for example MatrixXf). They also are not - * available with the (non-default) FullPivHouseholderQR preconditioner. - */ - JacobiSVD& compute(const MatrixType& matrix, unsigned int computationOptions); - - /** \brief Method performing the decomposition of given matrix using current options. - * - * \param matrix the matrix to decompose - * - * This method uses the current \a computationOptions, as already passed to the constructor or to compute(const MatrixType&, unsigned int). - */ - JacobiSVD& compute(const MatrixType& matrix) + private: + typedef HouseholderQR QRType; + QRType m_qr; + TransposeTypeWithSameStorageOrder m_adjoint; + typename internal::plain_row_type::type m_workspace; + }; + + /*** 2x2 SVD implementation + *** + *** JacobiSVD consists in performing a series of 2x2 SVD subproblems + ***/ + + template + struct svd_precondition_2x2_block_to_be_real + { + typedef JacobiSVD SVD; + typedef typename MatrixType::RealScalar RealScalar; + static bool run(typename SVD::WorkMatrixType &, SVD &, Index, Index, RealScalar &) { return true; } + }; + + template + struct svd_precondition_2x2_block_to_be_real + { + typedef JacobiSVD SVD; + typedef typename MatrixType::Scalar Scalar; + typedef typename MatrixType::RealScalar RealScalar; + static bool run(typename SVD::WorkMatrixType &work_matrix, SVD &svd, Index p, Index q, RealScalar &maxDiagEntry) { - return compute(matrix, m_computationOptions); + using std::sqrt; + using std::abs; + Scalar z; + JacobiRotation rot; + RealScalar n = sqrt(numext::abs2(work_matrix.coeff(p, p)) + numext::abs2(work_matrix.coeff(q, p))); + + const RealScalar considerAsZero = (std::numeric_limits::min)(); + const RealScalar precision = NumTraits::epsilon(); + + if (n == 0) { + // make sure first column is zero + work_matrix.coeffRef(p, p) = work_matrix.coeffRef(q, p) = Scalar(0); + + if (abs(numext::imag(work_matrix.coeff(p, q))) > considerAsZero) { + // work_matrix.coeff(p,q) can be zero if work_matrix.coeff(q,p) is not zero but small enough to underflow when + // computing n + z = abs(work_matrix.coeff(p, q)) / work_matrix.coeff(p, q); + work_matrix.row(p) *= z; + if (svd.computeU()) svd.m_matrixU.col(p) *= conj(z); + } + if (abs(numext::imag(work_matrix.coeff(q, q))) > considerAsZero) { + z = abs(work_matrix.coeff(q, q)) / work_matrix.coeff(q, q); + work_matrix.row(q) *= z; + if (svd.computeU()) svd.m_matrixU.col(q) *= conj(z); + } + // otherwise the second row is already zero, so we have nothing to do. + } else { + rot.c() = conj(work_matrix.coeff(p, p)) / n; + rot.s() = work_matrix.coeff(q, p) / n; + work_matrix.applyOnTheLeft(p, q, rot); + if (svd.computeU()) svd.m_matrixU.applyOnTheRight(p, q, rot.adjoint()); + if (abs(numext::imag(work_matrix.coeff(p, q))) > considerAsZero) { + z = abs(work_matrix.coeff(p, q)) / work_matrix.coeff(p, q); + work_matrix.col(q) *= z; + if (svd.computeV()) svd.m_matrixV.col(q) *= z; + } + if (abs(numext::imag(work_matrix.coeff(q, q))) > considerAsZero) { + z = abs(work_matrix.coeff(q, q)) / work_matrix.coeff(q, q); + work_matrix.row(q) *= z; + if (svd.computeU()) svd.m_matrixU.col(q) *= conj(z); + } + } + + // update largest diagonal entry + maxDiagEntry = numext::maxi( + maxDiagEntry, numext::maxi(abs(work_matrix.coeff(p, p)), abs(work_matrix.coeff(q, q)))); + // and check whether the 2x2 block is already diagonal + RealScalar threshold = numext::maxi(considerAsZero, precision * maxDiagEntry); + return abs(work_matrix.coeff(p, q)) > threshold || abs(work_matrix.coeff(q, p)) > threshold; } + }; + + template struct traits> + { + typedef _MatrixType MatrixType; + }; - using Base::computeU; - using Base::computeV; - using Base::rows; - using Base::cols; - using Base::rank; +}// end namespace internal - private: - void allocate(Index rows, Index cols, unsigned int computationOptions); - - protected: - using Base::m_matrixU; - using Base::m_matrixV; - using Base::m_singularValues; - using Base::m_isInitialized; - using Base::m_isAllocated; - using Base::m_usePrescribedThreshold; - using Base::m_computeFullU; - using Base::m_computeThinU; - using Base::m_computeFullV; - using Base::m_computeThinV; - using Base::m_computationOptions; - using Base::m_nonzeroSingularValues; - using Base::m_rows; - using Base::m_cols; - using Base::m_diagSize; - using Base::m_prescribedThreshold; - WorkMatrixType m_workMatrix; - - template - friend struct internal::svd_precondition_2x2_block_to_be_real; - template - friend struct internal::qr_preconditioner_impl; - - internal::qr_preconditioner_impl m_qr_precond_morecols; - internal::qr_preconditioner_impl m_qr_precond_morerows; - MatrixType m_scaledMatrix; +/** \ingroup SVD_Module + * + * + * \class JacobiSVD + * + * \brief Two-sided Jacobi SVD decomposition of a rectangular matrix + * + * \tparam _MatrixType the type of the matrix of which we are computing the SVD decomposition + * \tparam QRPreconditioner this optional parameter allows to specify the type of QR decomposition that will be used + * internally for the R-SVD step for non-square matrices. See discussion of possible values below. + * + * SVD decomposition consists in decomposing any n-by-p matrix \a A as a product + * \f[ A = U S V^* \f] + * where \a U is a n-by-n unitary, \a V is a p-by-p unitary, and \a S is a n-by-p real positive matrix which is zero + * outside of its main diagonal; the diagonal entries of S are known as the \em singular \em values of \a A and the + * columns of \a U and \a V are known as the left and right \em singular \em vectors of \a A respectively. + * + * Singular values are always sorted in decreasing order. + * + * This JacobiSVD decomposition computes only the singular values by default. If you want \a U or \a V, you need to ask + * for them explicitly. + * + * You can ask for only \em thin \a U or \a V to be computed, meaning the following. In case of a rectangular n-by-p + * matrix, letting \a m be the smaller value among \a n and \a p, there are only \a m singular vectors; the remaining + * columns of \a U and \a V do not correspond to actual singular vectors. Asking for \em thin \a U or \a V means asking + * for only their \a m first columns to be formed. So \a U is then a n-by-m matrix, and \a V is then a p-by-m matrix. + * Notice that thin \a U and \a V are all you need for (least squares) solving. + * + * Here's an example demonstrating basic usage: + * \include JacobiSVD_basic.cpp + * Output: \verbinclude JacobiSVD_basic.out + * + * This JacobiSVD class is a two-sided Jacobi R-SVD decomposition, ensuring optimal reliability and accuracy. The + * downside is that it's slower than bidiagonalizing SVD algorithms for large square matrices; however its complexity is + * still \f$ O(n^2p) \f$ where \a n is the smaller dimension and + * \a p is the greater dimension, meaning that it is still of the same order of complexity as the faster bidiagonalizing + * R-SVD algorithms. In particular, like any R-SVD, it takes advantage of non-squareness in that its complexity is only + * linear in the greater dimension. + * + * If the input matrix has inf or nan coefficients, the result of the computation is undefined, but the computation is + * guaranteed to terminate in finite (and reasonable) time. + * + * The possible values for QRPreconditioner are: + * \li ColPivHouseholderQRPreconditioner is the default. In practice it's very safe. It uses column-pivoting QR. + * \li FullPivHouseholderQRPreconditioner, is the safest and slowest. It uses full-pivoting QR. + * Contrary to other QRs, it doesn't allow computing thin unitaries. + * \li HouseholderQRPreconditioner is the fastest, and less safe and accurate than the pivoting variants. It uses + * non-pivoting QR. This is very similar in safety and accuracy to the bidiagonalization process used by bidiagonalizing + * SVD algorithms (since bidiagonalization is inherently non-pivoting). However the resulting SVD is still more reliable + * than bidiagonalizing SVDs because the Jacobi-based iterarive process is more reliable than the optimized bidiagonal + * SVD iterations. + * \li NoQRPreconditioner allows not to use a QR preconditioner at all. This is useful if you know that you will only be + * computing JacobiSVD decompositions of square matrices. Non-square matrices require a QR preconditioner. Using this + * option will result in faster compilation and smaller executable code. It won't significantly speed up computation, + * since JacobiSVD is always checking if QR preconditioning is needed before applying it anyway. + * + * \sa MatrixBase::jacobiSvd() + */ +template +class JacobiSVD : public SVDBase> +{ + typedef SVDBase Base; + +public: + typedef _MatrixType MatrixType; + typedef typename MatrixType::Scalar Scalar; + typedef typename NumTraits::Real RealScalar; + enum { + RowsAtCompileTime = MatrixType::RowsAtCompileTime, + ColsAtCompileTime = MatrixType::ColsAtCompileTime, + DiagSizeAtCompileTime = EIGEN_SIZE_MIN_PREFER_DYNAMIC(RowsAtCompileTime, ColsAtCompileTime), + MaxRowsAtCompileTime = MatrixType::MaxRowsAtCompileTime, + MaxColsAtCompileTime = MatrixType::MaxColsAtCompileTime, + MaxDiagSizeAtCompileTime = EIGEN_SIZE_MIN_PREFER_FIXED(MaxRowsAtCompileTime, MaxColsAtCompileTime), + MatrixOptions = MatrixType::Options + }; + + typedef typename Base::MatrixUType MatrixUType; + typedef typename Base::MatrixVType MatrixVType; + typedef typename Base::SingularValuesType SingularValuesType; + + typedef typename internal::plain_row_type::type RowType; + typedef typename internal::plain_col_type::type ColType; + typedef Matrix + WorkMatrixType; + + /** \brief Default Constructor. + * + * The default constructor is useful in cases in which the user intends to + * perform decompositions via JacobiSVD::compute(const MatrixType&). + */ + JacobiSVD() {} + + + /** \brief Default Constructor with memory preallocation + * + * Like the default constructor but with preallocation of the internal data + * according to the specified problem size. + * \sa JacobiSVD() + */ + JacobiSVD(Index rows, Index cols, unsigned int computationOptions = 0) { allocate(rows, cols, computationOptions); } + + /** \brief Constructor performing the decomposition of given matrix. + * + * \param matrix the matrix to decompose + * \param computationOptions optional parameter allowing to specify if you want full or thin U or V unitaries to be + * computed. By default, none is computed. This is a bit-field, the possible bits are #ComputeFullU, #ComputeThinU, + * #ComputeFullV, #ComputeThinV. + * + * Thin unitaries are only available if your matrix type has a Dynamic number of columns (for example MatrixXf). They + * also are not available with the (non-default) FullPivHouseholderQR preconditioner. + */ + explicit JacobiSVD(const MatrixType &matrix, unsigned int computationOptions = 0) + { + compute(matrix, computationOptions); + } + + /** \brief Method performing the decomposition of given matrix using custom options. + * + * \param matrix the matrix to decompose + * \param computationOptions optional parameter allowing to specify if you want full or thin U or V unitaries to be + * computed. By default, none is computed. This is a bit-field, the possible bits are #ComputeFullU, #ComputeThinU, + * #ComputeFullV, #ComputeThinV. + * + * Thin unitaries are only available if your matrix type has a Dynamic number of columns (for example MatrixXf). They + * also are not available with the (non-default) FullPivHouseholderQR preconditioner. + */ + JacobiSVD &compute(const MatrixType &matrix, unsigned int computationOptions); + + /** \brief Method performing the decomposition of given matrix using current options. + * + * \param matrix the matrix to decompose + * + * This method uses the current \a computationOptions, as already passed to the constructor or to compute(const + * MatrixType&, unsigned int). + */ + JacobiSVD &compute(const MatrixType &matrix) { return compute(matrix, m_computationOptions); } + + using Base::computeU; + using Base::computeV; + using Base::rows; + using Base::cols; + using Base::rank; + +private: + void allocate(Index rows, Index cols, unsigned int computationOptions); + +protected: + using Base::m_matrixU; + using Base::m_matrixV; + using Base::m_singularValues; + using Base::m_isInitialized; + using Base::m_isAllocated; + using Base::m_usePrescribedThreshold; + using Base::m_computeFullU; + using Base::m_computeThinU; + using Base::m_computeFullV; + using Base::m_computeThinV; + using Base::m_computationOptions; + using Base::m_nonzeroSingularValues; + using Base::m_rows; + using Base::m_cols; + using Base::m_diagSize; + using Base::m_prescribedThreshold; + WorkMatrixType m_workMatrix; + + template + friend struct internal::svd_precondition_2x2_block_to_be_real; + template + friend struct internal::qr_preconditioner_impl; + + internal::qr_preconditioner_impl + m_qr_precond_morecols; + internal::qr_preconditioner_impl + m_qr_precond_morerows; + MatrixType m_scaledMatrix; }; template @@ -614,13 +609,7 @@ void JacobiSVD::allocate(Index rows, Index cols, u { eigen_assert(rows >= 0 && cols >= 0); - if (m_isAllocated && - rows == m_rows && - cols == m_cols && - computationOptions == m_computationOptions) - { - return; - } + if (m_isAllocated && rows == m_rows && cols == m_cols && computationOptions == m_computationOptions) { return; } m_rows = rows; m_cols = cols; @@ -633,40 +622,33 @@ void JacobiSVD::allocate(Index rows, Index cols, u m_computeThinV = (computationOptions & ComputeThinV) != 0; eigen_assert(!(m_computeFullU && m_computeThinU) && "JacobiSVD: you can't ask for both full and thin U"); eigen_assert(!(m_computeFullV && m_computeThinV) && "JacobiSVD: you can't ask for both full and thin V"); - eigen_assert(EIGEN_IMPLIES(m_computeThinU || m_computeThinV, MatrixType::ColsAtCompileTime==Dynamic) && - "JacobiSVD: thin U and V are only available when your matrix has a dynamic number of columns."); - if (QRPreconditioner == FullPivHouseholderQRPreconditioner) - { - eigen_assert(!(m_computeThinU || m_computeThinV) && + eigen_assert(EIGEN_IMPLIES(m_computeThinU || m_computeThinV, MatrixType::ColsAtCompileTime == Dynamic) + && "JacobiSVD: thin U and V are only available when your matrix has a dynamic number of columns."); + if (QRPreconditioner == FullPivHouseholderQRPreconditioner) { + eigen_assert(!(m_computeThinU || m_computeThinV) && "JacobiSVD: can't compute thin U or thin V with the FullPivHouseholderQR preconditioner. " "Use the ColPivHouseholderQR preconditioner instead."); } m_diagSize = (std::min)(m_rows, m_cols); m_singularValues.resize(m_diagSize); - if(RowsAtCompileTime==Dynamic) - m_matrixU.resize(m_rows, m_computeFullU ? m_rows - : m_computeThinU ? m_diagSize - : 0); - if(ColsAtCompileTime==Dynamic) - m_matrixV.resize(m_cols, m_computeFullV ? m_cols - : m_computeThinV ? m_diagSize - : 0); + if (RowsAtCompileTime == Dynamic) m_matrixU.resize(m_rows, m_computeFullU ? m_rows : m_computeThinU ? m_diagSize : 0); + if (ColsAtCompileTime == Dynamic) m_matrixV.resize(m_cols, m_computeFullV ? m_cols : m_computeThinV ? m_diagSize : 0); m_workMatrix.resize(m_diagSize, m_diagSize); - - if(m_cols>m_rows) m_qr_precond_morecols.allocate(*this); - if(m_rows>m_cols) m_qr_precond_morerows.allocate(*this); - if(m_rows!=m_cols) m_scaledMatrix.resize(rows,cols); + + if (m_cols > m_rows) m_qr_precond_morecols.allocate(*this); + if (m_rows > m_cols) m_qr_precond_morerows.allocate(*this); + if (m_rows != m_cols) m_scaledMatrix.resize(rows, cols); } template -JacobiSVD& -JacobiSVD::compute(const MatrixType& matrix, unsigned int computationOptions) +JacobiSVD &JacobiSVD::compute(const MatrixType &matrix, + unsigned int computationOptions) { using std::abs; allocate(matrix.rows(), matrix.cols(), computationOptions); - // currently we stop when we reach precision 2*epsilon as the last bit of precision can require an unreasonable number of iterations, - // only worsening the precision of U and V as we accumulate more rotations + // currently we stop when we reach precision 2*epsilon as the last bit of precision can require an unreasonable number + // of iterations, only worsening the precision of U and V as we accumulate more rotations const RealScalar precision = RealScalar(2) * NumTraits::epsilon(); // limit for denormal numbers to be considered zero in order to avoid infinite loops (see bug 286) @@ -674,110 +656,98 @@ JacobiSVD::compute(const MatrixType& matrix, unsig // Scaling factor to reduce over/under-flows RealScalar scale = matrix.cwiseAbs().maxCoeff(); - if(scale==RealScalar(0)) scale = RealScalar(1); - + if (scale == RealScalar(0)) scale = RealScalar(1); + /*** step 1. The R-SVD step: we use a QR decomposition to reduce to the case of a square matrix */ - if(m_rows!=m_cols) - { + if (m_rows != m_cols) { m_scaledMatrix = matrix / scale; m_qr_precond_morecols.run(*this, m_scaledMatrix); m_qr_precond_morerows.run(*this, m_scaledMatrix); - } - else - { - m_workMatrix = matrix.block(0,0,m_diagSize,m_diagSize) / scale; - if(m_computeFullU) m_matrixU.setIdentity(m_rows,m_rows); - if(m_computeThinU) m_matrixU.setIdentity(m_rows,m_diagSize); - if(m_computeFullV) m_matrixV.setIdentity(m_cols,m_cols); - if(m_computeThinV) m_matrixV.setIdentity(m_cols, m_diagSize); + } else { + m_workMatrix = matrix.block(0, 0, m_diagSize, m_diagSize) / scale; + if (m_computeFullU) m_matrixU.setIdentity(m_rows, m_rows); + if (m_computeThinU) m_matrixU.setIdentity(m_rows, m_diagSize); + if (m_computeFullV) m_matrixV.setIdentity(m_cols, m_cols); + if (m_computeThinV) m_matrixV.setIdentity(m_cols, m_diagSize); } /*** step 2. The main Jacobi SVD iteration. ***/ RealScalar maxDiagEntry = m_workMatrix.cwiseAbs().diagonal().maxCoeff(); bool finished = false; - while(!finished) - { + while (!finished) { finished = true; // do a sweep: for all index pairs (p,q), perform SVD of the corresponding 2x2 sub-matrix - for(Index p = 1; p < m_diagSize; ++p) - { - for(Index q = 0; q < p; ++q) - { + for (Index p = 1; p < m_diagSize; ++p) { + for (Index q = 0; q < p; ++q) { // if this 2x2 sub-matrix is not diagonal already... // notice that this comparison will evaluate to false if any NaN is involved, ensuring that NaN's don't // keep us iterating forever. Similarly, small denormal numbers are considered zero. RealScalar threshold = numext::maxi(considerAsZero, precision * maxDiagEntry); - if(abs(m_workMatrix.coeff(p,q))>threshold || abs(m_workMatrix.coeff(q,p)) > threshold) - { + if (abs(m_workMatrix.coeff(p, q)) > threshold || abs(m_workMatrix.coeff(q, p)) > threshold) { finished = false; // perform SVD decomposition of 2x2 sub-matrix corresponding to indices p,q to make it diagonal // the complex to real operation returns true if the updated 2x2 block is not already diagonal - if(internal::svd_precondition_2x2_block_to_be_real::run(m_workMatrix, *this, p, q, maxDiagEntry)) - { + if (internal::svd_precondition_2x2_block_to_be_real::run( + m_workMatrix, *this, p, q, maxDiagEntry)) { JacobiRotation j_left, j_right; internal::real_2x2_jacobi_svd(m_workMatrix, p, q, &j_left, &j_right); // accumulate resulting Jacobi rotations - m_workMatrix.applyOnTheLeft(p,q,j_left); - if(computeU()) m_matrixU.applyOnTheRight(p,q,j_left.transpose()); + m_workMatrix.applyOnTheLeft(p, q, j_left); + if (computeU()) m_matrixU.applyOnTheRight(p, q, j_left.transpose()); - m_workMatrix.applyOnTheRight(p,q,j_right); - if(computeV()) m_matrixV.applyOnTheRight(p,q,j_right); + m_workMatrix.applyOnTheRight(p, q, j_right); + if (computeV()) m_matrixV.applyOnTheRight(p, q, j_right); // keep track of the largest diagonal coefficient - maxDiagEntry = numext::maxi(maxDiagEntry,numext::maxi(abs(m_workMatrix.coeff(p,p)), abs(m_workMatrix.coeff(q,q)))); + maxDiagEntry = numext::maxi( + maxDiagEntry, numext::maxi(abs(m_workMatrix.coeff(p, p)), abs(m_workMatrix.coeff(q, q)))); } } } } } - /*** step 3. The work matrix is now diagonal, so ensure it's positive so its diagonal entries are the singular values ***/ + /*** step 3. The work matrix is now diagonal, so ensure it's positive so its diagonal entries are the singular values + * ***/ - for(Index i = 0; i < m_diagSize; ++i) - { + for (Index i = 0; i < m_diagSize; ++i) { // For a complex matrix, some diagonal coefficients might note have been // treated by svd_precondition_2x2_block_to_be_real, and the imaginary part // of some diagonal entry might not be null. - if(NumTraits::IsComplex && abs(numext::imag(m_workMatrix.coeff(i,i)))>considerAsZero) - { - RealScalar a = abs(m_workMatrix.coeff(i,i)); + if (NumTraits::IsComplex && abs(numext::imag(m_workMatrix.coeff(i, i))) > considerAsZero) { + RealScalar a = abs(m_workMatrix.coeff(i, i)); m_singularValues.coeffRef(i) = abs(a); - if(computeU()) m_matrixU.col(i) *= m_workMatrix.coeff(i,i)/a; - } - else - { + if (computeU()) m_matrixU.col(i) *= m_workMatrix.coeff(i, i) / a; + } else { // m_workMatrix.coeff(i,i) is already real, no difficulty: - RealScalar a = numext::real(m_workMatrix.coeff(i,i)); + RealScalar a = numext::real(m_workMatrix.coeff(i, i)); m_singularValues.coeffRef(i) = abs(a); - if(computeU() && (a::compute(const MatrixType& matrix, unsig } /** \svd_module - * - * \return the singular value decomposition of \c *this computed by two-sided - * Jacobi transformations. - * - * \sa class JacobiSVD - */ + * + * \return the singular value decomposition of \c *this computed by two-sided + * Jacobi transformations. + * + * \sa class JacobiSVD + */ template -JacobiSVD::PlainObject> -MatrixBase::jacobiSvd(unsigned int computationOptions) const +JacobiSVD::PlainObject> MatrixBase::jacobiSvd( + unsigned int computationOptions) const { return JacobiSVD(*this, computationOptions); } -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_JACOBISVD_H +#endif// EIGEN_JACOBISVD_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/SVD/JacobiSVD_LAPACKE.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/SVD/JacobiSVD_LAPACKE.h index ff0516f6..412f57ba 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/SVD/JacobiSVD_LAPACKE.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/SVD/JacobiSVD_LAPACKE.h @@ -33,59 +33,86 @@ #ifndef EIGEN_JACOBISVD_LAPACKE_H #define EIGEN_JACOBISVD_LAPACKE_H -namespace Eigen { +namespace Eigen { /** \internal Specialization for the data types supported by LAPACKe */ -#define EIGEN_LAPACKE_SVD(EIGTYPE, LAPACKE_TYPE, LAPACKE_RTYPE, LAPACKE_PREFIX, EIGCOLROW, LAPACKE_COLROW) \ -template<> inline \ -JacobiSVD, ColPivHouseholderQRPreconditioner>& \ -JacobiSVD, ColPivHouseholderQRPreconditioner>::compute(const Matrix& matrix, unsigned int computationOptions) \ -{ \ - typedef Matrix MatrixType; \ - /*typedef MatrixType::Scalar Scalar;*/ \ - /*typedef MatrixType::RealScalar RealScalar;*/ \ - allocate(matrix.rows(), matrix.cols(), computationOptions); \ -\ - /*const RealScalar precision = RealScalar(2) * NumTraits::epsilon();*/ \ - m_nonzeroSingularValues = m_diagSize; \ -\ - lapack_int lda = internal::convert_index(matrix.outerStride()), ldu, ldvt; \ - lapack_int matrix_order = LAPACKE_COLROW; \ - char jobu, jobvt; \ - LAPACKE_TYPE *u, *vt, dummy; \ - jobu = (m_computeFullU) ? 'A' : (m_computeThinU) ? 'S' : 'N'; \ - jobvt = (m_computeFullV) ? 'A' : (m_computeThinV) ? 'S' : 'N'; \ - if (computeU()) { \ - ldu = internal::convert_index(m_matrixU.outerStride()); \ - u = (LAPACKE_TYPE*)m_matrixU.data(); \ - } else { ldu=1; u=&dummy; }\ - MatrixType localV; \ - lapack_int vt_rows = (m_computeFullV) ? internal::convert_index(m_cols) : (m_computeThinV) ? internal::convert_index(m_diagSize) : 1; \ - if (computeV()) { \ - localV.resize(vt_rows, m_cols); \ - ldvt = internal::convert_index(localV.outerStride()); \ - vt = (LAPACKE_TYPE*)localV.data(); \ - } else { ldvt=1; vt=&dummy; }\ - Matrix superb; superb.resize(m_diagSize, 1); \ - MatrixType m_temp; m_temp = matrix; \ - LAPACKE_##LAPACKE_PREFIX##gesvd( matrix_order, jobu, jobvt, internal::convert_index(m_rows), internal::convert_index(m_cols), (LAPACKE_TYPE*)m_temp.data(), lda, (LAPACKE_RTYPE*)m_singularValues.data(), u, ldu, vt, ldvt, superb.data()); \ - if (computeV()) m_matrixV = localV.adjoint(); \ - /* for(int i=0;i \ + inline JacobiSVD, \ + ColPivHouseholderQRPreconditioner> & \ + JacobiSVD, \ + ColPivHouseholderQRPreconditioner>::compute(const Matrix \ + &matrix, \ + unsigned int computationOptions) \ + { \ + typedef Matrix MatrixType; \ + /*typedef MatrixType::Scalar Scalar;*/ \ + /*typedef MatrixType::RealScalar RealScalar;*/ \ + allocate(matrix.rows(), matrix.cols(), computationOptions); \ + \ + /*const RealScalar precision = RealScalar(2) * NumTraits::epsilon();*/ \ + m_nonzeroSingularValues = m_diagSize; \ + \ + lapack_int lda = internal::convert_index(matrix.outerStride()), ldu, ldvt; \ + lapack_int matrix_order = LAPACKE_COLROW; \ + char jobu, jobvt; \ + LAPACKE_TYPE *u, *vt, dummy; \ + jobu = (m_computeFullU) ? 'A' : (m_computeThinU) ? 'S' : 'N'; \ + jobvt = (m_computeFullV) ? 'A' : (m_computeThinV) ? 'S' : 'N'; \ + if (computeU()) { \ + ldu = internal::convert_index(m_matrixU.outerStride()); \ + u = (LAPACKE_TYPE *)m_matrixU.data(); \ + } else { \ + ldu = 1; \ + u = &dummy; \ + } \ + MatrixType localV; \ + lapack_int vt_rows = (m_computeFullV) ? internal::convert_index(m_cols) \ + : (m_computeThinV) ? internal::convert_index(m_diagSize) \ + : 1; \ + if (computeV()) { \ + localV.resize(vt_rows, m_cols); \ + ldvt = internal::convert_index(localV.outerStride()); \ + vt = (LAPACKE_TYPE *)localV.data(); \ + } else { \ + ldvt = 1; \ + vt = &dummy; \ + } \ + Matrix superb; \ + superb.resize(m_diagSize, 1); \ + MatrixType m_temp; \ + m_temp = matrix; \ + LAPACKE_##LAPACKE_PREFIX##gesvd(matrix_order, \ + jobu, \ + jobvt, \ + internal::convert_index(m_rows), \ + internal::convert_index(m_cols), \ + (LAPACKE_TYPE *)m_temp.data(), \ + lda, \ + (LAPACKE_RTYPE *)m_singularValues.data(), \ + u, \ + ldu, \ + vt, \ + ldvt, \ + superb.data()); \ + if (computeV()) m_matrixV = localV.adjoint(); \ + /* for(int i=0;i -class SVDBase +template class SVDBase { public: @@ -53,34 +53,47 @@ class SVDBase typedef typename MatrixType::Scalar Scalar; typedef typename NumTraits::Real RealScalar; typedef typename MatrixType::StorageIndex StorageIndex; - typedef Eigen::Index Index; ///< \deprecated since Eigen 3.3 + typedef Eigen::Index Index;///< \deprecated since Eigen 3.3 enum { RowsAtCompileTime = MatrixType::RowsAtCompileTime, ColsAtCompileTime = MatrixType::ColsAtCompileTime, - DiagSizeAtCompileTime = EIGEN_SIZE_MIN_PREFER_DYNAMIC(RowsAtCompileTime,ColsAtCompileTime), + DiagSizeAtCompileTime = EIGEN_SIZE_MIN_PREFER_DYNAMIC(RowsAtCompileTime, ColsAtCompileTime), MaxRowsAtCompileTime = MatrixType::MaxRowsAtCompileTime, MaxColsAtCompileTime = MatrixType::MaxColsAtCompileTime, - MaxDiagSizeAtCompileTime = EIGEN_SIZE_MIN_PREFER_FIXED(MaxRowsAtCompileTime,MaxColsAtCompileTime), + MaxDiagSizeAtCompileTime = EIGEN_SIZE_MIN_PREFER_FIXED(MaxRowsAtCompileTime, MaxColsAtCompileTime), MatrixOptions = MatrixType::Options }; - typedef Matrix MatrixUType; - typedef Matrix MatrixVType; + typedef Matrix + MatrixUType; + typedef Matrix + MatrixVType; typedef typename internal::plain_diag_type::type SingularValuesType; - - Derived& derived() { return *static_cast(this); } - const Derived& derived() const { return *static_cast(this); } + + Derived &derived() { return *static_cast(this); } + const Derived &derived() const { return *static_cast(this); } /** \returns the \a U matrix. * * For the SVD decomposition of a n-by-p matrix, letting \a m be the minimum of \a n and \a p, - * the U matrix is n-by-n if you asked for \link Eigen::ComputeFullU ComputeFullU \endlink, and is n-by-m if you asked for \link Eigen::ComputeThinU ComputeThinU \endlink. + * the U matrix is n-by-n if you asked for \link Eigen::ComputeFullU ComputeFullU \endlink, and is n-by-m if you asked + * for \link Eigen::ComputeThinU ComputeThinU \endlink. * * The \a m first columns of \a U are the left singular vectors of the matrix being decomposed. * * This method asserts that you asked for \a U to be computed. */ - const MatrixUType& matrixU() const + const MatrixUType &matrixU() const { eigen_assert(m_isInitialized && "SVD is not initialized."); eigen_assert(computeU() && "This SVD decomposition didn't compute U. Did you ask for it?"); @@ -90,13 +103,14 @@ class SVDBase /** \returns the \a V matrix. * * For the SVD decomposition of a n-by-p matrix, letting \a m be the minimum of \a n and \a p, - * the V matrix is p-by-p if you asked for \link Eigen::ComputeFullV ComputeFullV \endlink, and is p-by-m if you asked for \link Eigen::ComputeThinV ComputeThinV \endlink. + * the V matrix is p-by-p if you asked for \link Eigen::ComputeFullV ComputeFullV \endlink, and is p-by-m if you asked + * for \link Eigen::ComputeThinV ComputeThinV \endlink. * * The \a m first columns of \a V are the right singular vectors of the matrix being decomposed. * * This method asserts that you asked for \a V to be computed. */ - const MatrixVType& matrixV() const + const MatrixVType &matrixV() const { eigen_assert(m_isInitialized && "SVD is not initialized."); eigen_assert(computeV() && "This SVD decomposition didn't compute V. Did you ask for it?"); @@ -108,7 +122,7 @@ class SVDBase * For the SVD decomposition of a n-by-p matrix, letting \a m be the minimum of \a n and \a p, the * returned vector has size \a m. Singular values are always sorted in decreasing order. */ - const SingularValuesType& singularValues() const + const SingularValuesType &singularValues() const { eigen_assert(m_isInitialized && "SVD is not initialized."); return m_singularValues; @@ -120,39 +134,40 @@ class SVDBase eigen_assert(m_isInitialized && "SVD is not initialized."); return m_nonzeroSingularValues; } - + /** \returns the rank of the matrix of which \c *this is the SVD. - * - * \note This method has to determine which singular values should be considered nonzero. - * For that, it uses the threshold value that you can control by calling - * setThreshold(const RealScalar&). - */ + * + * \note This method has to determine which singular values should be considered nonzero. + * For that, it uses the threshold value that you can control by calling + * setThreshold(const RealScalar&). + */ inline Index rank() const { using std::abs; eigen_assert(m_isInitialized && "JacobiSVD is not initialized."); - if(m_singularValues.size()==0) return 0; - RealScalar premultiplied_threshold = numext::maxi(m_singularValues.coeff(0) * threshold(), (std::numeric_limits::min)()); - Index i = m_nonzeroSingularValues-1; - while(i>=0 && m_singularValues.coeff(i) < premultiplied_threshold) --i; - return i+1; + if (m_singularValues.size() == 0) return 0; + RealScalar premultiplied_threshold = + numext::maxi(m_singularValues.coeff(0) * threshold(), (std::numeric_limits::min)()); + Index i = m_nonzeroSingularValues - 1; + while (i >= 0 && m_singularValues.coeff(i) < premultiplied_threshold) --i; + return i + 1; } - + /** Allows to prescribe a threshold to be used by certain methods, such as rank() and solve(), - * which need to determine when singular values are to be considered nonzero. - * This is not used for the SVD decomposition itself. - * - * When it needs to get the threshold value, Eigen calls threshold(). - * The default is \c NumTraits::epsilon() - * - * \param threshold The new value to use as the threshold. - * - * A singular value will be considered nonzero if its value is strictly greater than - * \f$ \vert singular value \vert \leqslant threshold \times \vert max singular value \vert \f$. - * - * If you want to come back to the default behavior, call setThreshold(Default_t) - */ - Derived& setThreshold(const RealScalar& threshold) + * which need to determine when singular values are to be considered nonzero. + * This is not used for the SVD decomposition itself. + * + * When it needs to get the threshold value, Eigen calls threshold(). + * The default is \c NumTraits::epsilon() + * + * \param threshold The new value to use as the threshold. + * + * A singular value will be considered nonzero if its value is strictly greater than + * \f$ \vert singular value \vert \leqslant threshold \times \vert max singular value \vert \f$. + * + * If you want to come back to the default behavior, call setThreshold(Default_t) + */ + Derived &setThreshold(const RealScalar &threshold) { m_usePrescribedThreshold = true; m_prescribedThreshold = threshold; @@ -160,30 +175,29 @@ class SVDBase } /** Allows to come back to the default behavior, letting Eigen use its default formula for - * determining the threshold. - * - * You should pass the special object Eigen::Default as parameter here. - * \code svd.setThreshold(Eigen::Default); \endcode - * - * See the documentation of setThreshold(const RealScalar&). - */ - Derived& setThreshold(Default_t) + * determining the threshold. + * + * You should pass the special object Eigen::Default as parameter here. + * \code svd.setThreshold(Eigen::Default); \endcode + * + * See the documentation of setThreshold(const RealScalar&). + */ + Derived &setThreshold(Default_t) { m_usePrescribedThreshold = false; return derived(); } /** Returns the threshold that will be used by certain methods such as rank(). - * - * See the documentation of setThreshold(const RealScalar&). - */ + * + * See the documentation of setThreshold(const RealScalar&). + */ RealScalar threshold() const { eigen_assert(m_isInitialized || m_usePrescribedThreshold); // this temporary is needed to workaround a MSVC issue - Index diagSize = (std::max)(1,m_diagSize); - return m_usePrescribedThreshold ? m_prescribedThreshold - : diagSize*NumTraits::epsilon(); + Index diagSize = (std::max)(1, m_diagSize); + return m_usePrescribedThreshold ? m_prescribedThreshold : diagSize * NumTraits::epsilon(); } /** \returns true if \a U (full or thin) is asked for in this SVD decomposition */ @@ -193,40 +207,35 @@ class SVDBase inline Index rows() const { return m_rows; } inline Index cols() const { return m_cols; } - + /** \returns a (least squares) solution of \f$ A x = b \f$ using the current SVD decomposition of A. - * - * \param b the right-hand-side of the equation to solve. - * - * \note Solving requires both U and V to be computed. Thin U and V are enough, there is no need for full U or V. - * - * \note SVD solving is implicitly least-squares. Thus, this method serves both purposes of exact solving and least-squares solving. - * In other words, the returned solution is guaranteed to minimize the Euclidean norm \f$ \Vert A x - b \Vert \f$. - */ - template - inline const Solve - solve(const MatrixBase& b) const + * + * \param b the right-hand-side of the equation to solve. + * + * \note Solving requires both U and V to be computed. Thin U and V are enough, there is no need for full U or V. + * + * \note SVD solving is implicitly least-squares. Thus, this method serves both purposes of exact solving and + * least-squares solving. In other words, the returned solution is guaranteed to minimize the Euclidean norm \f$ \Vert + * A x - b \Vert \f$. + */ + template inline const Solve solve(const MatrixBase &b) const { eigen_assert(m_isInitialized && "SVD is not initialized."); - eigen_assert(computeU() && computeV() && "SVD::solve() requires both unitaries U and V to be computed (thin unitaries suffice)."); + eigen_assert(computeU() && computeV() + && "SVD::solve() requires both unitaries U and V to be computed (thin unitaries suffice)."); return Solve(derived(), b.derived()); } - - #ifndef EIGEN_PARSED_BY_DOXYGEN + +#ifndef EIGEN_PARSED_BY_DOXYGEN template - EIGEN_DEVICE_FUNC - void _solve_impl(const RhsType &rhs, DstType &dst) const; - #endif + EIGEN_DEVICE_FUNC void _solve_impl(const RhsType &rhs, DstType &dst) const; +#endif protected: - - static void check_template_parameters() - { - EIGEN_STATIC_ASSERT_NON_INTEGER(Scalar); - } - + static void check_template_parameters() { EIGEN_STATIC_ASSERT_NON_INTEGER(Scalar); } + // return true if already allocated - bool allocate(Index rows, Index cols, unsigned int computationOptions) ; + bool allocate(Index rows, Index cols, unsigned int computationOptions); MatrixUType m_matrixU; MatrixVType m_matrixV; @@ -243,16 +252,11 @@ class SVDBase * Default constructor of SVDBase */ SVDBase() - : m_isInitialized(false), - m_isAllocated(false), - m_usePrescribedThreshold(false), - m_computationOptions(0), + : m_isInitialized(false), m_isAllocated(false), m_usePrescribedThreshold(false), m_computationOptions(0), m_rows(-1), m_cols(-1), m_diagSize(0) { check_template_parameters(); } - - }; #ifndef EIGEN_PARSED_BY_DOXYGEN @@ -265,9 +269,15 @@ void SVDBase::_solve_impl(const RhsType &rhs, DstType &dst) const // A = U S V^* // So A^{-1} = V S^{-1} U^* - Matrix tmp; + Matrix + tmp; Index l_rank = rank(); - tmp.noalias() = m_matrixU.leftCols(l_rank).adjoint() * rhs; + tmp.noalias() = m_matrixU.leftCols(l_rank).adjoint() * rhs; tmp = m_singularValues.head(l_rank).asDiagonal().inverse() * tmp; dst = m_matrixV.leftCols(l_rank) * tmp; } @@ -278,13 +288,7 @@ bool SVDBase::allocate(Index rows, Index cols, unsigned int computat { eigen_assert(rows >= 0 && cols >= 0); - if (m_isAllocated && - rows == m_rows && - cols == m_cols && - computationOptions == m_computationOptions) - { - return true; - } + if (m_isAllocated && rows == m_rows && cols == m_cols && computationOptions == m_computationOptions) { return true; } m_rows = rows; m_cols = cols; @@ -297,19 +301,17 @@ bool SVDBase::allocate(Index rows, Index cols, unsigned int computat m_computeThinV = (computationOptions & ComputeThinV) != 0; eigen_assert(!(m_computeFullU && m_computeThinU) && "SVDBase: you can't ask for both full and thin U"); eigen_assert(!(m_computeFullV && m_computeThinV) && "SVDBase: you can't ask for both full and thin V"); - eigen_assert(EIGEN_IMPLIES(m_computeThinU || m_computeThinV, MatrixType::ColsAtCompileTime==Dynamic) && - "SVDBase: thin U and V are only available when your matrix has a dynamic number of columns."); + eigen_assert(EIGEN_IMPLIES(m_computeThinU || m_computeThinV, MatrixType::ColsAtCompileTime == Dynamic) + && "SVDBase: thin U and V are only available when your matrix has a dynamic number of columns."); m_diagSize = (std::min)(m_rows, m_cols); m_singularValues.resize(m_diagSize); - if(RowsAtCompileTime==Dynamic) - m_matrixU.resize(m_rows, m_computeFullU ? m_rows : m_computeThinU ? m_diagSize : 0); - if(ColsAtCompileTime==Dynamic) - m_matrixV.resize(m_cols, m_computeFullV ? m_cols : m_computeThinV ? m_diagSize : 0); + if (RowsAtCompileTime == Dynamic) m_matrixU.resize(m_rows, m_computeFullU ? m_rows : m_computeThinU ? m_diagSize : 0); + if (ColsAtCompileTime == Dynamic) m_matrixV.resize(m_cols, m_computeFullV ? m_cols : m_computeThinV ? m_diagSize : 0); return false; } -}// end namespace +}// namespace Eigen -#endif // EIGEN_SVDBASE_H +#endif// EIGEN_SVDBASE_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/SVD/UpperBidiagonalization.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/SVD/UpperBidiagonalization.h index 11ac847e..b8a75d3b 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/SVD/UpperBidiagonalization.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/SVD/UpperBidiagonalization.h @@ -11,16 +11,15 @@ #ifndef EIGEN_BIDIAGONALIZATION_H #define EIGEN_BIDIAGONALIZATION_H -namespace Eigen { +namespace Eigen { namespace internal { -// UpperBidiagonalization will probably be replaced by a Bidiagonalization class, don't want to make it stable API. -// At the same time, it's useful to keep for now as it's about the only thing that is testing the BandMatrix class. + // UpperBidiagonalization will probably be replaced by a Bidiagonalization class, don't want to make it stable API. + // At the same time, it's useful to keep for now as it's about the only thing that is testing the BandMatrix class. -template class UpperBidiagonalization -{ + template class UpperBidiagonalization + { public: - typedef _MatrixType MatrixType; enum { RowsAtCompileTime = MatrixType::RowsAtCompileTime, @@ -29,370 +28,343 @@ template class UpperBidiagonalization }; typedef typename MatrixType::Scalar Scalar; typedef typename MatrixType::RealScalar RealScalar; - typedef Eigen::Index Index; ///< \deprecated since Eigen 3.3 + typedef Eigen::Index Index;///< \deprecated since Eigen 3.3 typedef Matrix RowVectorType; typedef Matrix ColVectorType; typedef BandMatrix BidiagonalType; typedef Matrix DiagVectorType; typedef Matrix SuperDiagVectorType; - typedef HouseholderSequence< - const MatrixType, - const typename internal::remove_all::ConjugateReturnType>::type - > HouseholderUSequenceType; - typedef HouseholderSequence< - const typename internal::remove_all::type, - Diagonal, - OnTheRight - > HouseholderVSequenceType; - + typedef HouseholderSequence::ConjugateReturnType>::type> + HouseholderUSequenceType; + typedef HouseholderSequence::type, + Diagonal, + OnTheRight> + HouseholderVSequenceType; + /** - * \brief Default Constructor. - * - * The default constructor is useful in cases in which the user intends to - * perform decompositions via Bidiagonalization::compute(const MatrixType&). - */ + * \brief Default Constructor. + * + * The default constructor is useful in cases in which the user intends to + * perform decompositions via Bidiagonalization::compute(const MatrixType&). + */ UpperBidiagonalization() : m_householder(), m_bidiagonal(), m_isInitialized(false) {} - explicit UpperBidiagonalization(const MatrixType& matrix) - : m_householder(matrix.rows(), matrix.cols()), - m_bidiagonal(matrix.cols(), matrix.cols()), - m_isInitialized(false) + explicit UpperBidiagonalization(const MatrixType &matrix) + : m_householder(matrix.rows(), matrix.cols()), m_bidiagonal(matrix.cols(), matrix.cols()), m_isInitialized(false) { compute(matrix); } - - UpperBidiagonalization& compute(const MatrixType& matrix); - UpperBidiagonalization& computeUnblocked(const MatrixType& matrix); - - const MatrixType& householder() const { return m_householder; } - const BidiagonalType& bidiagonal() const { return m_bidiagonal; } - + + UpperBidiagonalization &compute(const MatrixType &matrix); + UpperBidiagonalization &computeUnblocked(const MatrixType &matrix); + + const MatrixType &householder() const { return m_householder; } + const BidiagonalType &bidiagonal() const { return m_bidiagonal; } + const HouseholderUSequenceType householderU() const { eigen_assert(m_isInitialized && "UpperBidiagonalization is not initialized."); return HouseholderUSequenceType(m_householder, m_householder.diagonal().conjugate()); } - const HouseholderVSequenceType householderV() // const here gives nasty errors and i'm lazy + const HouseholderVSequenceType householderV()// const here gives nasty errors and i'm lazy { eigen_assert(m_isInitialized && "UpperBidiagonalization is not initialized."); return HouseholderVSequenceType(m_householder.conjugate(), m_householder.const_derived().template diagonal<1>()) - .setLength(m_householder.cols()-1) - .setShift(1); + .setLength(m_householder.cols() - 1) + .setShift(1); } - + protected: MatrixType m_householder; BidiagonalType m_bidiagonal; bool m_isInitialized; -}; - -// Standard upper bidiagonalization without fancy optimizations -// This version should be faster for small matrix size -template -void upperbidiagonalization_inplace_unblocked(MatrixType& mat, - typename MatrixType::RealScalar *diagonal, - typename MatrixType::RealScalar *upper_diagonal, - typename MatrixType::Scalar* tempData = 0) -{ - typedef typename MatrixType::Scalar Scalar; - - Index rows = mat.rows(); - Index cols = mat.cols(); - - typedef Matrix TempType; - TempType tempVector; - if(tempData==0) + }; + + // Standard upper bidiagonalization without fancy optimizations + // This version should be faster for small matrix size + template + void upperbidiagonalization_inplace_unblocked(MatrixType &mat, + typename MatrixType::RealScalar *diagonal, + typename MatrixType::RealScalar *upper_diagonal, + typename MatrixType::Scalar *tempData = 0) { - tempVector.resize(rows); - tempData = tempVector.data(); - } - - for (Index k = 0; /* breaks at k==cols-1 below */ ; ++k) - { - Index remainingRows = rows - k; - Index remainingCols = cols - k - 1; - - // construct left householder transform in-place in A - mat.col(k).tail(remainingRows) - .makeHouseholderInPlace(mat.coeffRef(k,k), diagonal[k]); - // apply householder transform to remaining part of A on the left - mat.bottomRightCorner(remainingRows, remainingCols) - .applyHouseholderOnTheLeft(mat.col(k).tail(remainingRows-1), mat.coeff(k,k), tempData); - - if(k == cols-1) break; - - // construct right householder transform in-place in mat - mat.row(k).tail(remainingCols) - .makeHouseholderInPlace(mat.coeffRef(k,k+1), upper_diagonal[k]); - // apply householder transform to remaining part of mat on the left - mat.bottomRightCorner(remainingRows-1, remainingCols) - .applyHouseholderOnTheRight(mat.row(k).tail(remainingCols-1).transpose(), mat.coeff(k,k+1), tempData); - } -} + typedef typename MatrixType::Scalar Scalar; -/** \internal - * Helper routine for the block reduction to upper bidiagonal form. - * - * Let's partition the matrix A: - * - * | A00 A01 | - * A = | | - * | A10 A11 | - * - * This function reduces to bidiagonal form the left \c rows x \a blockSize vertical panel [A00/A10] - * and the \a blockSize x \c cols horizontal panel [A00 A01] of the matrix \a A. The bottom-right block A11 - * is updated using matrix-matrix products: - * A22 -= V * Y^T - X * U^T - * where V and U contains the left and right Householder vectors. U and V are stored in A10, and A01 - * respectively, and the update matrices X and Y are computed during the reduction. - * - */ -template -void upperbidiagonalization_blocked_helper(MatrixType& A, - typename MatrixType::RealScalar *diagonal, - typename MatrixType::RealScalar *upper_diagonal, - Index bs, - Ref::Flags & RowMajorBit> > X, - Ref::Flags & RowMajorBit> > Y) -{ - typedef typename MatrixType::Scalar Scalar; - typedef typename MatrixType::RealScalar RealScalar; - typedef typename NumTraits::Literal Literal; - enum { StorageOrder = traits::Flags & RowMajorBit }; - typedef InnerStride ColInnerStride; - typedef InnerStride RowInnerStride; - typedef Ref, 0, ColInnerStride> SubColumnType; - typedef Ref, 0, RowInnerStride> SubRowType; - typedef Ref > SubMatType; - - Index brows = A.rows(); - Index bcols = A.cols(); - - Scalar tau_u, tau_u_prev(0), tau_v; - - for(Index k = 0; k < bs; ++k) - { - Index remainingRows = brows - k; - Index remainingCols = bcols - k - 1; - - SubMatType X_k1( X.block(k,0, remainingRows,k) ); - SubMatType V_k1( A.block(k,0, remainingRows,k) ); - - // 1 - update the k-th column of A - SubColumnType v_k = A.col(k).tail(remainingRows); - v_k -= V_k1 * Y.row(k).head(k).adjoint(); - if(k) v_k -= X_k1 * A.col(k).head(k); - - // 2 - construct left Householder transform in-place - v_k.makeHouseholderInPlace(tau_v, diagonal[k]); - - if(k+1 TempType; + TempType tempVector; + if (tempData == 0) { + tempVector.resize(rows); + tempData = tempVector.data(); + } - // 5 - construct right Householder transform in-place - u_k.makeHouseholderInPlace(tau_u, upper_diagonal[k]); + for (Index k = 0; /* breaks at k==cols-1 below */; ++k) { + Index remainingRows = rows - k; + Index remainingCols = cols - k - 1; - // this eases the application of Householder transformations - // A(k,k+1) will store tau_u later - A(k,k+1) = Scalar(1); + // construct left householder transform in-place in A + mat.col(k).tail(remainingRows).makeHouseholderInPlace(mat.coeffRef(k, k), diagonal[k]); + // apply householder transform to remaining part of A on the left + mat.bottomRightCorner(remainingRows, remainingCols) + .applyHouseholderOnTheLeft(mat.col(k).tail(remainingRows - 1), mat.coeff(k, k), tempData); - // 6 - Compute x_k = tau_u * ( A*u_k - X_k-1*U_k-1^T*u_k - V_k*Y_k^T*u_k ) - { - SubColumnType x_k ( X.col(k).tail(remainingRows-1) ); - - // let's use the begining of column k of X as a temporary vectors - // note that tmp0 and tmp1 overlaps - SubColumnType tmp0 ( X.col(k).head(k) ), - tmp1 ( X.col(k).head(k+1) ); - - x_k.noalias() = A.block(k+1,k+1, remainingRows-1,remainingCols) * u_k.transpose(); // bottleneck - tmp0.noalias() = U_k1 * u_k.transpose(); - x_k.noalias() -= X_k1.bottomRows(remainingRows-1) * tmp0; - tmp1.noalias() = Y_k.adjoint() * u_k.transpose(); - x_k.noalias() -= A.block(k+1,0, remainingRows-1,k+1) * tmp1; - x_k *= numext::conj(tau_u); - tau_u = numext::conj(tau_u); - u_k = u_k.conjugate(); - } + if (k == cols - 1) break; - if(k>0) A.coeffRef(k-1,k) = tau_u_prev; - tau_u_prev = tau_u; + // construct right householder transform in-place in mat + mat.row(k).tail(remainingCols).makeHouseholderInPlace(mat.coeffRef(k, k + 1), upper_diagonal[k]); + // apply householder transform to remaining part of mat on the left + mat.bottomRightCorner(remainingRows - 1, remainingCols) + .applyHouseholderOnTheRight(mat.row(k).tail(remainingCols - 1).transpose(), mat.coeff(k, k + 1), tempData); } - else - A.coeffRef(k-1,k) = tau_u_prev; - - A.coeffRef(k,k) = tau_v; } - - if(bsbs && brows>bs) + /** \internal + * Helper routine for the block reduction to upper bidiagonal form. + * + * Let's partition the matrix A: + * + * | A00 A01 | + * A = | | + * | A10 A11 | + * + * This function reduces to bidiagonal form the left \c rows x \a blockSize vertical panel [A00/A10] + * and the \a blockSize x \c cols horizontal panel [A00 A01] of the matrix \a A. The bottom-right block A11 + * is updated using matrix-matrix products: + * A22 -= V * Y^T - X * U^T + * where V and U contains the left and right Householder vectors. U and V are stored in A10, and A01 + * respectively, and the update matrices X and Y are computed during the reduction. + * + */ + template + void upperbidiagonalization_blocked_helper(MatrixType &A, + typename MatrixType::RealScalar *diagonal, + typename MatrixType::RealScalar *upper_diagonal, + Index bs, + Ref::Flags & RowMajorBit>> X, + Ref::Flags & RowMajorBit>> Y) { - SubMatType A11( A.bottomRightCorner(brows-bs,bcols-bs) ); - SubMatType A10( A.block(bs,0, brows-bs,bs) ); - SubMatType A01( A.block(0,bs, bs,bcols-bs) ); - Scalar tmp = A01(bs-1,0); - A01(bs-1,0) = Literal(1); - A11.noalias() -= A10 * Y.topLeftCorner(bcols,bs).bottomRows(bcols-bs).adjoint(); - A11.noalias() -= X.topLeftCorner(brows,bs).bottomRows(brows-bs) * A01; - A01(bs-1,0) = tmp; + typedef typename MatrixType::Scalar Scalar; + typedef typename MatrixType::RealScalar RealScalar; + typedef typename NumTraits::Literal Literal; + enum { StorageOrder = traits::Flags & RowMajorBit }; + typedef InnerStride ColInnerStride; + typedef InnerStride RowInnerStride; + typedef Ref, 0, ColInnerStride> SubColumnType; + typedef Ref, 0, RowInnerStride> SubRowType; + typedef Ref> SubMatType; + + Index brows = A.rows(); + Index bcols = A.cols(); + + Scalar tau_u, tau_u_prev(0), tau_v; + + for (Index k = 0; k < bs; ++k) { + Index remainingRows = brows - k; + Index remainingCols = bcols - k - 1; + + SubMatType X_k1(X.block(k, 0, remainingRows, k)); + SubMatType V_k1(A.block(k, 0, remainingRows, k)); + + // 1 - update the k-th column of A + SubColumnType v_k = A.col(k).tail(remainingRows); + v_k -= V_k1 * Y.row(k).head(k).adjoint(); + if (k) v_k -= X_k1 * A.col(k).head(k); + + // 2 - construct left Householder transform in-place + v_k.makeHouseholderInPlace(tau_v, diagonal[k]); + + if (k + 1 < bcols) { + SubMatType Y_k(Y.block(k + 1, 0, remainingCols, k + 1)); + SubMatType U_k1(A.block(0, k + 1, k, remainingCols)); + + // this eases the application of Householder transforAions + // A(k,k) will store tau_v later + A(k, k) = Scalar(1); + + // 3 - Compute y_k^T = tau_v * ( A^T*v_k - Y_k-1*V_k-1^T*v_k - U_k-1*X_k-1^T*v_k ) + { + SubColumnType y_k(Y.col(k).tail(remainingCols)); + + // let's use the begining of column k of Y as a temporary vector + SubColumnType tmp(Y.col(k).head(k)); + y_k.noalias() = A.block(k, k + 1, remainingRows, remainingCols).adjoint() * v_k;// bottleneck + tmp.noalias() = V_k1.adjoint() * v_k; + y_k.noalias() -= Y_k.leftCols(k) * tmp; + tmp.noalias() = X_k1.adjoint() * v_k; + y_k.noalias() -= U_k1.adjoint() * tmp; + y_k *= numext::conj(tau_v); + } + + // 4 - update k-th row of A (it will become u_k) + SubRowType u_k(A.row(k).tail(remainingCols)); + u_k = u_k.conjugate(); + { + u_k -= Y_k * A.row(k).head(k + 1).adjoint(); + if (k) u_k -= U_k1.adjoint() * X.row(k).head(k).adjoint(); + } + + // 5 - construct right Householder transform in-place + u_k.makeHouseholderInPlace(tau_u, upper_diagonal[k]); + + // this eases the application of Householder transformations + // A(k,k+1) will store tau_u later + A(k, k + 1) = Scalar(1); + + // 6 - Compute x_k = tau_u * ( A*u_k - X_k-1*U_k-1^T*u_k - V_k*Y_k^T*u_k ) + { + SubColumnType x_k(X.col(k).tail(remainingRows - 1)); + + // let's use the begining of column k of X as a temporary vectors + // note that tmp0 and tmp1 overlaps + SubColumnType tmp0(X.col(k).head(k)), tmp1(X.col(k).head(k + 1)); + + x_k.noalias() = A.block(k + 1, k + 1, remainingRows - 1, remainingCols) * u_k.transpose();// bottleneck + tmp0.noalias() = U_k1 * u_k.transpose(); + x_k.noalias() -= X_k1.bottomRows(remainingRows - 1) * tmp0; + tmp1.noalias() = Y_k.adjoint() * u_k.transpose(); + x_k.noalias() -= A.block(k + 1, 0, remainingRows - 1, k + 1) * tmp1; + x_k *= numext::conj(tau_u); + tau_u = numext::conj(tau_u); + u_k = u_k.conjugate(); + } + + if (k > 0) A.coeffRef(k - 1, k) = tau_u_prev; + tau_u_prev = tau_u; + } else + A.coeffRef(k - 1, k) = tau_u_prev; + + A.coeffRef(k, k) = tau_v; + } + + if (bs < bcols) A.coeffRef(bs - 1, bs) = tau_u_prev; + + // update A22 + if (bcols > bs && brows > bs) { + SubMatType A11(A.bottomRightCorner(brows - bs, bcols - bs)); + SubMatType A10(A.block(bs, 0, brows - bs, bs)); + SubMatType A01(A.block(0, bs, bs, bcols - bs)); + Scalar tmp = A01(bs - 1, 0); + A01(bs - 1, 0) = Literal(1); + A11.noalias() -= A10 * Y.topLeftCorner(bcols, bs).bottomRows(bcols - bs).adjoint(); + A11.noalias() -= X.topLeftCorner(brows, bs).bottomRows(brows - bs) * A01; + A01(bs - 1, 0) = tmp; + } } -} -/** \internal - * - * Implementation of a block-bidiagonal reduction. - * It is based on the following paper: - * The Design of a Parallel Dense Linear Algebra Software Library: Reduction to Hessenberg, Tridiagonal, and Bidiagonal Form. - * by Jaeyoung Choi, Jack J. Dongarra, David W. Walker. (1995) - * section 3.3 - */ -template -void upperbidiagonalization_inplace_blocked(MatrixType& A, BidiagType& bidiagonal, - Index maxBlockSize=32, - typename MatrixType::Scalar* /*tempData*/ = 0) -{ - typedef typename MatrixType::Scalar Scalar; - typedef Block BlockType; - - Index rows = A.rows(); - Index cols = A.cols(); - Index size = (std::min)(rows, cols); - - // X and Y are work space - enum { StorageOrder = traits::Flags & RowMajorBit }; - Matrix X(rows,maxBlockSize); - Matrix Y(cols,maxBlockSize); - Index blockSize = (std::min)(maxBlockSize,size); - - Index k = 0; - for(k = 0; k < size; k += blockSize) + /** \internal + * + * Implementation of a block-bidiagonal reduction. + * It is based on the following paper: + * The Design of a Parallel Dense Linear Algebra Software Library: Reduction to Hessenberg, Tridiagonal, and + * Bidiagonal Form. by Jaeyoung Choi, Jack J. Dongarra, David W. Walker. (1995) section 3.3 + */ + template + void upperbidiagonalization_inplace_blocked(MatrixType &A, + BidiagType &bidiagonal, + Index maxBlockSize = 32, + typename MatrixType::Scalar * /*tempData*/ = 0) { - Index bs = (std::min)(size-k,blockSize); // actual size of the block - Index brows = rows - k; // rows of the block - Index bcols = cols - k; // columns of the block - - // partition the matrix A: - // - // | A00 A01 A02 | - // | | - // A = | A10 A11 A12 | - // | | - // | A20 A21 A22 | - // - // where A11 is a bs x bs diagonal block, - // and let: - // | A11 A12 | - // B = | | - // | A21 A22 | - - BlockType B = A.block(k,k,brows,bcols); - - // This stage performs the bidiagonalization of A11, A21, A12, and updating of A22. - // Finally, the algorithm continue on the updated A22. - // - // However, if B is too small, or A22 empty, then let's use an unblocked strategy - if(k+bs==cols || bcols<48) // somewhat arbitrary threshold - { - upperbidiagonalization_inplace_unblocked(B, - &(bidiagonal.template diagonal<0>().coeffRef(k)), - &(bidiagonal.template diagonal<1>().coeffRef(k)), - X.data() - ); - break; // We're done - } - else - { - upperbidiagonalization_blocked_helper( B, - &(bidiagonal.template diagonal<0>().coeffRef(k)), - &(bidiagonal.template diagonal<1>().coeffRef(k)), - bs, - X.topLeftCorner(brows,bs), - Y.topLeftCorner(bcols,bs) - ); + typedef typename MatrixType::Scalar Scalar; + typedef Block BlockType; + + Index rows = A.rows(); + Index cols = A.cols(); + Index size = (std::min)(rows, cols); + + // X and Y are work space + enum { StorageOrder = traits::Flags & RowMajorBit }; + Matrix X( + rows, maxBlockSize); + Matrix Y( + cols, maxBlockSize); + Index blockSize = (std::min)(maxBlockSize, size); + + Index k = 0; + for (k = 0; k < size; k += blockSize) { + Index bs = (std::min)(size - k, blockSize);// actual size of the block + Index brows = rows - k;// rows of the block + Index bcols = cols - k;// columns of the block + + // partition the matrix A: + // + // | A00 A01 A02 | + // | | + // A = | A10 A11 A12 | + // | | + // | A20 A21 A22 | + // + // where A11 is a bs x bs diagonal block, + // and let: + // | A11 A12 | + // B = | | + // | A21 A22 | + + BlockType B = A.block(k, k, brows, bcols); + + // This stage performs the bidiagonalization of A11, A21, A12, and updating of A22. + // Finally, the algorithm continue on the updated A22. + // + // However, if B is too small, or A22 empty, then let's use an unblocked strategy + if (k + bs == cols || bcols < 48)// somewhat arbitrary threshold + { + upperbidiagonalization_inplace_unblocked(B, + &(bidiagonal.template diagonal<0>().coeffRef(k)), + &(bidiagonal.template diagonal<1>().coeffRef(k)), + X.data()); + break;// We're done + } else { + upperbidiagonalization_blocked_helper(B, + &(bidiagonal.template diagonal<0>().coeffRef(k)), + &(bidiagonal.template diagonal<1>().coeffRef(k)), + bs, + X.topLeftCorner(brows, bs), + Y.topLeftCorner(bcols, bs)); + } } } -} -template -UpperBidiagonalization<_MatrixType>& UpperBidiagonalization<_MatrixType>::computeUnblocked(const _MatrixType& matrix) -{ - Index rows = matrix.rows(); - Index cols = matrix.cols(); - EIGEN_ONLY_USED_FOR_DEBUG(cols); + template + UpperBidiagonalization<_MatrixType> &UpperBidiagonalization<_MatrixType>::computeUnblocked(const _MatrixType &matrix) + { + Index rows = matrix.rows(); + Index cols = matrix.cols(); + EIGEN_ONLY_USED_FOR_DEBUG(cols); - eigen_assert(rows >= cols && "UpperBidiagonalization is only for Arices satisfying rows>=cols."); + eigen_assert(rows >= cols && "UpperBidiagonalization is only for Arices satisfying rows>=cols."); - m_householder = matrix; + m_householder = matrix; - ColVectorType temp(rows); + ColVectorType temp(rows); - upperbidiagonalization_inplace_unblocked(m_householder, - &(m_bidiagonal.template diagonal<0>().coeffRef(0)), - &(m_bidiagonal.template diagonal<1>().coeffRef(0)), - temp.data()); + upperbidiagonalization_inplace_unblocked(m_householder, + &(m_bidiagonal.template diagonal<0>().coeffRef(0)), + &(m_bidiagonal.template diagonal<1>().coeffRef(0)), + temp.data()); - m_isInitialized = true; - return *this; -} + m_isInitialized = true; + return *this; + } -template -UpperBidiagonalization<_MatrixType>& UpperBidiagonalization<_MatrixType>::compute(const _MatrixType& matrix) -{ - Index rows = matrix.rows(); - Index cols = matrix.cols(); - EIGEN_ONLY_USED_FOR_DEBUG(rows); - EIGEN_ONLY_USED_FOR_DEBUG(cols); - - eigen_assert(rows >= cols && "UpperBidiagonalization is only for Arices satisfying rows>=cols."); - - m_householder = matrix; - upperbidiagonalization_inplace_blocked(m_householder, m_bidiagonal); - - m_isInitialized = true; - return *this; -} + template + UpperBidiagonalization<_MatrixType> &UpperBidiagonalization<_MatrixType>::compute(const _MatrixType &matrix) + { + Index rows = matrix.rows(); + Index cols = matrix.cols(); + EIGEN_ONLY_USED_FOR_DEBUG(rows); + EIGEN_ONLY_USED_FOR_DEBUG(cols); + + eigen_assert(rows >= cols && "UpperBidiagonalization is only for Arices satisfying rows>=cols."); + + m_householder = matrix; + upperbidiagonalization_inplace_blocked(m_householder, m_bidiagonal); + + m_isInitialized = true; + return *this; + } #if 0 /** \return the Householder QR decomposition of \c *this. @@ -407,8 +379,8 @@ MatrixBase::bidiagonalization() const } #endif -} // end namespace internal +}// end namespace internal -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_BIDIAGONALIZATION_H +#endif// EIGEN_BIDIAGONALIZATION_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/SparseCholesky/SimplicialCholesky.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/SparseCholesky/SimplicialCholesky.h index 2907f652..d9c228e5 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/SparseCholesky/SimplicialCholesky.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/SparseCholesky/SimplicialCholesky.h @@ -10,680 +10,632 @@ #ifndef EIGEN_SIMPLICIAL_CHOLESKY_H #define EIGEN_SIMPLICIAL_CHOLESKY_H -namespace Eigen { +namespace Eigen { -enum SimplicialCholeskyMode { - SimplicialCholeskyLLT, - SimplicialCholeskyLDLT -}; +enum SimplicialCholeskyMode { SimplicialCholeskyLLT, SimplicialCholeskyLDLT }; namespace internal { - template - struct simplicial_cholesky_grab_input { - typedef CholMatrixType const * ConstCholMatrixPtr; - static void run(const InputMatrixType& input, ConstCholMatrixPtr &pmat, CholMatrixType &tmp) + template struct simplicial_cholesky_grab_input + { + typedef CholMatrixType const *ConstCholMatrixPtr; + static void run(const InputMatrixType &input, ConstCholMatrixPtr &pmat, CholMatrixType &tmp) { tmp = input; pmat = &tmp; } }; - - template - struct simplicial_cholesky_grab_input { - typedef MatrixType const * ConstMatrixPtr; - static void run(const MatrixType& input, ConstMatrixPtr &pmat, MatrixType &/*tmp*/) - { - pmat = &input; - } + + template struct simplicial_cholesky_grab_input + { + typedef MatrixType const *ConstMatrixPtr; + static void run(const MatrixType &input, ConstMatrixPtr &pmat, MatrixType & /*tmp*/) { pmat = &input; } }; -} // end namespace internal +}// end namespace internal /** \ingroup SparseCholesky_Module - * \brief A base class for direct sparse Cholesky factorizations - * - * This is a base class for LL^T and LDL^T Cholesky factorizations of sparse matrices that are - * selfadjoint and positive definite. These factorizations allow for solving A.X = B where - * X and B can be either dense or sparse. - * - * In order to reduce the fill-in, a symmetric permutation P is applied prior to the factorization - * such that the factorized matrix is P A P^-1. - * - * \tparam Derived the type of the derived class, that is the actual factorization type. - * - */ -template -class SimplicialCholeskyBase : public SparseSolverBase + * \brief A base class for direct sparse Cholesky factorizations + * + * This is a base class for LL^T and LDL^T Cholesky factorizations of sparse matrices that are + * selfadjoint and positive definite. These factorizations allow for solving A.X = B where + * X and B can be either dense or sparse. + * + * In order to reduce the fill-in, a symmetric permutation P is applied prior to the factorization + * such that the factorized matrix is P A P^-1. + * + * \tparam Derived the type of the derived class, that is the actual factorization type. + * + */ +template class SimplicialCholeskyBase : public SparseSolverBase { - typedef SparseSolverBase Base; - using Base::m_isInitialized; - - public: - typedef typename internal::traits::MatrixType MatrixType; - typedef typename internal::traits::OrderingType OrderingType; - enum { UpLo = internal::traits::UpLo }; - typedef typename MatrixType::Scalar Scalar; - typedef typename MatrixType::RealScalar RealScalar; - typedef typename MatrixType::StorageIndex StorageIndex; - typedef SparseMatrix CholMatrixType; - typedef CholMatrixType const * ConstCholMatrixPtr; - typedef Matrix VectorType; - typedef Matrix VectorI; - - enum { - ColsAtCompileTime = MatrixType::ColsAtCompileTime, - MaxColsAtCompileTime = MatrixType::MaxColsAtCompileTime - }; - - public: - - using Base::derived; - - /** Default constructor */ - SimplicialCholeskyBase() - : m_info(Success), m_shiftOffset(0), m_shiftScale(1) - {} - - explicit SimplicialCholeskyBase(const MatrixType& matrix) - : m_info(Success), m_shiftOffset(0), m_shiftScale(1) - { - derived().compute(matrix); - } + typedef SparseSolverBase Base; + using Base::m_isInitialized; - ~SimplicialCholeskyBase() - { - } +public: + typedef typename internal::traits::MatrixType MatrixType; + typedef typename internal::traits::OrderingType OrderingType; + enum { UpLo = internal::traits::UpLo }; + typedef typename MatrixType::Scalar Scalar; + typedef typename MatrixType::RealScalar RealScalar; + typedef typename MatrixType::StorageIndex StorageIndex; + typedef SparseMatrix CholMatrixType; + typedef CholMatrixType const *ConstCholMatrixPtr; + typedef Matrix VectorType; + typedef Matrix VectorI; + + enum { ColsAtCompileTime = MatrixType::ColsAtCompileTime, MaxColsAtCompileTime = MatrixType::MaxColsAtCompileTime }; - Derived& derived() { return *static_cast(this); } - const Derived& derived() const { return *static_cast(this); } - - inline Index cols() const { return m_matrix.cols(); } - inline Index rows() const { return m_matrix.rows(); } - - /** \brief Reports whether previous computation was successful. - * - * \returns \c Success if computation was succesful, - * \c NumericalIssue if the matrix.appears to be negative. - */ - ComputationInfo info() const - { - eigen_assert(m_isInitialized && "Decomposition is not initialized."); - return m_info; - } - - /** \returns the permutation P - * \sa permutationPinv() */ - const PermutationMatrix& permutationP() const - { return m_P; } - - /** \returns the inverse P^-1 of the permutation P - * \sa permutationP() */ - const PermutationMatrix& permutationPinv() const - { return m_Pinv; } - - /** Sets the shift parameters that will be used to adjust the diagonal coefficients during the numerical factorization. - * - * During the numerical factorization, the diagonal coefficients are transformed by the following linear model:\n - * \c d_ii = \a offset + \a scale * \c d_ii - * - * The default is the identity transformation with \a offset=0, and \a scale=1. - * - * \returns a reference to \c *this. - */ - Derived& setShift(const RealScalar& offset, const RealScalar& scale = 1) - { - m_shiftOffset = offset; - m_shiftScale = scale; - return derived(); - } +public: + using Base::derived; + + /** Default constructor */ + SimplicialCholeskyBase() : m_info(Success), m_shiftOffset(0), m_shiftScale(1) {} + + explicit SimplicialCholeskyBase(const MatrixType &matrix) : m_info(Success), m_shiftOffset(0), m_shiftScale(1) + { + derived().compute(matrix); + } + + ~SimplicialCholeskyBase() {} + + Derived &derived() { return *static_cast(this); } + const Derived &derived() const { return *static_cast(this); } + + inline Index cols() const { return m_matrix.cols(); } + inline Index rows() const { return m_matrix.rows(); } + + /** \brief Reports whether previous computation was successful. + * + * \returns \c Success if computation was succesful, + * \c NumericalIssue if the matrix.appears to be negative. + */ + ComputationInfo info() const + { + eigen_assert(m_isInitialized && "Decomposition is not initialized."); + return m_info; + } + + /** \returns the permutation P + * \sa permutationPinv() */ + const PermutationMatrix &permutationP() const { return m_P; } + + /** \returns the inverse P^-1 of the permutation P + * \sa permutationP() */ + const PermutationMatrix &permutationPinv() const { return m_Pinv; } + + /** Sets the shift parameters that will be used to adjust the diagonal coefficients during the numerical + * factorization. + * + * During the numerical factorization, the diagonal coefficients are transformed by the following linear model:\n + * \c d_ii = \a offset + \a scale * \c d_ii + * + * The default is the identity transformation with \a offset=0, and \a scale=1. + * + * \returns a reference to \c *this. + */ + Derived &setShift(const RealScalar &offset, const RealScalar &scale = 1) + { + m_shiftOffset = offset; + m_shiftScale = scale; + return derived(); + } #ifndef EIGEN_PARSED_BY_DOXYGEN - /** \internal */ - template - void dumpMemory(Stream& s) - { - int total = 0; - s << " L: " << ((total+=(m_matrix.cols()+1) * sizeof(int) + m_matrix.nonZeros()*(sizeof(int)+sizeof(Scalar))) >> 20) << "Mb" << "\n"; - s << " diag: " << ((total+=m_diag.size() * sizeof(Scalar)) >> 20) << "Mb" << "\n"; - s << " tree: " << ((total+=m_parent.size() * sizeof(int)) >> 20) << "Mb" << "\n"; - s << " nonzeros: " << ((total+=m_nonZerosPerCol.size() * sizeof(int)) >> 20) << "Mb" << "\n"; - s << " perm: " << ((total+=m_P.size() * sizeof(int)) >> 20) << "Mb" << "\n"; - s << " perm^-1: " << ((total+=m_Pinv.size() * sizeof(int)) >> 20) << "Mb" << "\n"; - s << " TOTAL: " << (total>> 20) << "Mb" << "\n"; - } + /** \internal */ + template void dumpMemory(Stream &s) + { + int total = 0; + s << " L: " + << ((total += (m_matrix.cols() + 1) * sizeof(int) + m_matrix.nonZeros() * (sizeof(int) + sizeof(Scalar))) >> 20) + << "Mb" << "\n"; + s << " diag: " << ((total += m_diag.size() * sizeof(Scalar)) >> 20) << "Mb" << "\n"; + s << " tree: " << ((total += m_parent.size() * sizeof(int)) >> 20) << "Mb" << "\n"; + s << " nonzeros: " << ((total += m_nonZerosPerCol.size() * sizeof(int)) >> 20) << "Mb" << "\n"; + s << " perm: " << ((total += m_P.size() * sizeof(int)) >> 20) << "Mb" << "\n"; + s << " perm^-1: " << ((total += m_Pinv.size() * sizeof(int)) >> 20) << "Mb" << "\n"; + s << " TOTAL: " << (total >> 20) << "Mb" << "\n"; + } - /** \internal */ - template - void _solve_impl(const MatrixBase &b, MatrixBase &dest) const - { - eigen_assert(m_factorizationIsOk && "The decomposition is not in a valid state for solving, you must first call either compute() or symbolic()/numeric()"); - eigen_assert(m_matrix.rows()==b.rows()); + /** \internal */ + template void _solve_impl(const MatrixBase &b, MatrixBase &dest) const + { + eigen_assert(m_factorizationIsOk && "The decomposition is not in a valid state for solving, you must first call either compute() or symbolic()/numeric()"); + eigen_assert(m_matrix.rows() == b.rows()); - if(m_info!=Success) - return; + if (m_info != Success) return; - if(m_P.size()>0) - dest = m_P * b; - else - dest = b; + if (m_P.size() > 0) + dest = m_P * b; + else + dest = b; - if(m_matrix.nonZeros()>0) // otherwise L==I - derived().matrixL().solveInPlace(dest); + if (m_matrix.nonZeros() > 0)// otherwise L==I + derived().matrixL().solveInPlace(dest); - if(m_diag.size()>0) - dest = m_diag.asDiagonal().inverse() * dest; + if (m_diag.size() > 0) dest = m_diag.asDiagonal().inverse() * dest; - if (m_matrix.nonZeros()>0) // otherwise U==I - derived().matrixU().solveInPlace(dest); + if (m_matrix.nonZeros() > 0)// otherwise U==I + derived().matrixU().solveInPlace(dest); - if(m_P.size()>0) - dest = m_Pinv * dest; - } - - template - void _solve_impl(const SparseMatrixBase &b, SparseMatrixBase &dest) const - { - internal::solve_sparse_through_dense_panels(derived(), b, dest); - } + if (m_P.size() > 0) dest = m_Pinv * dest; + } -#endif // EIGEN_PARSED_BY_DOXYGEN + template + void _solve_impl(const SparseMatrixBase &b, SparseMatrixBase &dest) const + { + internal::solve_sparse_through_dense_panels(derived(), b, dest); + } - protected: - - /** Computes the sparse Cholesky decomposition of \a matrix */ - template - void compute(const MatrixType& matrix) - { - eigen_assert(matrix.rows()==matrix.cols()); - Index size = matrix.cols(); - CholMatrixType tmp(size,size); - ConstCholMatrixPtr pmat; - ordering(matrix, pmat, tmp); - analyzePattern_preordered(*pmat, DoLDLT); - factorize_preordered(*pmat); - } - - template - void factorize(const MatrixType& a) - { - eigen_assert(a.rows()==a.cols()); - Index size = a.cols(); - CholMatrixType tmp(size,size); - ConstCholMatrixPtr pmat; - - if(m_P.size()==0 && (UpLo&Upper)==Upper) - { - // If there is no ordering, try to directly use the input matrix without any copy - internal::simplicial_cholesky_grab_input::run(a, pmat, tmp); - } - else - { - tmp.template selfadjointView() = a.template selfadjointView().twistedBy(m_P); - pmat = &tmp; - } - - factorize_preordered(*pmat); - } +#endif// EIGEN_PARSED_BY_DOXYGEN - template - void factorize_preordered(const CholMatrixType& a); +protected: + /** Computes the sparse Cholesky decomposition of \a matrix */ + template void compute(const MatrixType &matrix) + { + eigen_assert(matrix.rows() == matrix.cols()); + Index size = matrix.cols(); + CholMatrixType tmp(size, size); + ConstCholMatrixPtr pmat; + ordering(matrix, pmat, tmp); + analyzePattern_preordered(*pmat, DoLDLT); + factorize_preordered(*pmat); + } - void analyzePattern(const MatrixType& a, bool doLDLT) - { - eigen_assert(a.rows()==a.cols()); - Index size = a.cols(); - CholMatrixType tmp(size,size); - ConstCholMatrixPtr pmat; - ordering(a, pmat, tmp); - analyzePattern_preordered(*pmat,doLDLT); + template void factorize(const MatrixType &a) + { + eigen_assert(a.rows() == a.cols()); + Index size = a.cols(); + CholMatrixType tmp(size, size); + ConstCholMatrixPtr pmat; + + if (m_P.size() == 0 && (UpLo & Upper) == Upper) { + // If there is no ordering, try to directly use the input matrix without any copy + internal::simplicial_cholesky_grab_input::run(a, pmat, tmp); + } else { + tmp.template selfadjointView() = a.template selfadjointView().twistedBy(m_P); + pmat = &tmp; } - void analyzePattern_preordered(const CholMatrixType& a, bool doLDLT); - - void ordering(const MatrixType& a, ConstCholMatrixPtr &pmat, CholMatrixType& ap); - - /** keeps off-diagonal entries; drops diagonal entries */ - struct keep_diag { - inline bool operator() (const Index& row, const Index& col, const Scalar&) const - { - return row!=col; - } - }; - - mutable ComputationInfo m_info; - bool m_factorizationIsOk; - bool m_analysisIsOk; - - CholMatrixType m_matrix; - VectorType m_diag; // the diagonal coefficients (LDLT mode) - VectorI m_parent; // elimination tree - VectorI m_nonZerosPerCol; - PermutationMatrix m_P; // the permutation - PermutationMatrix m_Pinv; // the inverse permutation - - RealScalar m_shiftOffset; - RealScalar m_shiftScale; + + factorize_preordered(*pmat); + } + + template void factorize_preordered(const CholMatrixType &a); + + void analyzePattern(const MatrixType &a, bool doLDLT) + { + eigen_assert(a.rows() == a.cols()); + Index size = a.cols(); + CholMatrixType tmp(size, size); + ConstCholMatrixPtr pmat; + ordering(a, pmat, tmp); + analyzePattern_preordered(*pmat, doLDLT); + } + void analyzePattern_preordered(const CholMatrixType &a, bool doLDLT); + + void ordering(const MatrixType &a, ConstCholMatrixPtr &pmat, CholMatrixType &ap); + + /** keeps off-diagonal entries; drops diagonal entries */ + struct keep_diag + { + inline bool operator()(const Index &row, const Index &col, const Scalar &) const { return row != col; } + }; + + mutable ComputationInfo m_info; + bool m_factorizationIsOk; + bool m_analysisIsOk; + + CholMatrixType m_matrix; + VectorType m_diag;// the diagonal coefficients (LDLT mode) + VectorI m_parent;// elimination tree + VectorI m_nonZerosPerCol; + PermutationMatrix m_P;// the permutation + PermutationMatrix m_Pinv;// the inverse permutation + + RealScalar m_shiftOffset; + RealScalar m_shiftScale; }; -template > class SimplicialLLT; -template > class SimplicialLDLT; -template > class SimplicialCholesky; +template> +class SimplicialLLT; +template> +class SimplicialLDLT; +template> +class SimplicialCholesky; namespace internal { -template struct traits > -{ - typedef _MatrixType MatrixType; - typedef _Ordering OrderingType; - enum { UpLo = _UpLo }; - typedef typename MatrixType::Scalar Scalar; - typedef typename MatrixType::StorageIndex StorageIndex; - typedef SparseMatrix CholMatrixType; - typedef TriangularView MatrixL; - typedef TriangularView MatrixU; - static inline MatrixL getL(const MatrixType& m) { return MatrixL(m); } - static inline MatrixU getU(const MatrixType& m) { return MatrixU(m.adjoint()); } -}; + template + struct traits> + { + typedef _MatrixType MatrixType; + typedef _Ordering OrderingType; + enum { UpLo = _UpLo }; + typedef typename MatrixType::Scalar Scalar; + typedef typename MatrixType::StorageIndex StorageIndex; + typedef SparseMatrix CholMatrixType; + typedef TriangularView MatrixL; + typedef TriangularView MatrixU; + static inline MatrixL getL(const MatrixType &m) { return MatrixL(m); } + static inline MatrixU getU(const MatrixType &m) { return MatrixU(m.adjoint()); } + }; -template struct traits > -{ - typedef _MatrixType MatrixType; - typedef _Ordering OrderingType; - enum { UpLo = _UpLo }; - typedef typename MatrixType::Scalar Scalar; - typedef typename MatrixType::StorageIndex StorageIndex; - typedef SparseMatrix CholMatrixType; - typedef TriangularView MatrixL; - typedef TriangularView MatrixU; - static inline MatrixL getL(const MatrixType& m) { return MatrixL(m); } - static inline MatrixU getU(const MatrixType& m) { return MatrixU(m.adjoint()); } -}; + template + struct traits> + { + typedef _MatrixType MatrixType; + typedef _Ordering OrderingType; + enum { UpLo = _UpLo }; + typedef typename MatrixType::Scalar Scalar; + typedef typename MatrixType::StorageIndex StorageIndex; + typedef SparseMatrix CholMatrixType; + typedef TriangularView MatrixL; + typedef TriangularView MatrixU; + static inline MatrixL getL(const MatrixType &m) { return MatrixL(m); } + static inline MatrixU getU(const MatrixType &m) { return MatrixU(m.adjoint()); } + }; -template struct traits > -{ - typedef _MatrixType MatrixType; - typedef _Ordering OrderingType; - enum { UpLo = _UpLo }; -}; + template + struct traits> + { + typedef _MatrixType MatrixType; + typedef _Ordering OrderingType; + enum { UpLo = _UpLo }; + }; -} +}// namespace internal /** \ingroup SparseCholesky_Module - * \class SimplicialLLT - * \brief A direct sparse LLT Cholesky factorizations - * - * This class provides a LL^T Cholesky factorizations of sparse matrices that are - * selfadjoint and positive definite. The factorization allows for solving A.X = B where - * X and B can be either dense or sparse. - * - * In order to reduce the fill-in, a symmetric permutation P is applied prior to the factorization - * such that the factorized matrix is P A P^-1. - * - * \tparam _MatrixType the type of the sparse matrix A, it must be a SparseMatrix<> - * \tparam _UpLo the triangular part that will be used for the computations. It can be Lower - * or Upper. Default is Lower. - * \tparam _Ordering The ordering method to use, either AMDOrdering<> or NaturalOrdering<>. Default is AMDOrdering<> - * - * \implsparsesolverconcept - * - * \sa class SimplicialLDLT, class AMDOrdering, class NaturalOrdering - */ + * \class SimplicialLLT + * \brief A direct sparse LLT Cholesky factorizations + * + * This class provides a LL^T Cholesky factorizations of sparse matrices that are + * selfadjoint and positive definite. The factorization allows for solving A.X = B where + * X and B can be either dense or sparse. + * + * In order to reduce the fill-in, a symmetric permutation P is applied prior to the factorization + * such that the factorized matrix is P A P^-1. + * + * \tparam _MatrixType the type of the sparse matrix A, it must be a SparseMatrix<> + * \tparam _UpLo the triangular part that will be used for the computations. It can be Lower + * or Upper. Default is Lower. + * \tparam _Ordering The ordering method to use, either AMDOrdering<> or NaturalOrdering<>. Default is AMDOrdering<> + * + * \implsparsesolverconcept + * + * \sa class SimplicialLDLT, class AMDOrdering, class NaturalOrdering + */ template - class SimplicialLLT : public SimplicialCholeskyBase > +class SimplicialLLT : public SimplicialCholeskyBase> { public: - typedef _MatrixType MatrixType; - enum { UpLo = _UpLo }; - typedef SimplicialCholeskyBase Base; - typedef typename MatrixType::Scalar Scalar; - typedef typename MatrixType::RealScalar RealScalar; - typedef typename MatrixType::StorageIndex StorageIndex; - typedef SparseMatrix CholMatrixType; - typedef Matrix VectorType; - typedef internal::traits Traits; - typedef typename Traits::MatrixL MatrixL; - typedef typename Traits::MatrixU MatrixU; + typedef _MatrixType MatrixType; + enum { UpLo = _UpLo }; + typedef SimplicialCholeskyBase Base; + typedef typename MatrixType::Scalar Scalar; + typedef typename MatrixType::RealScalar RealScalar; + typedef typename MatrixType::StorageIndex StorageIndex; + typedef SparseMatrix CholMatrixType; + typedef Matrix VectorType; + typedef internal::traits Traits; + typedef typename Traits::MatrixL MatrixL; + typedef typename Traits::MatrixU MatrixU; + public: - /** Default constructor */ - SimplicialLLT() : Base() {} - /** Constructs and performs the LLT factorization of \a matrix */ - explicit SimplicialLLT(const MatrixType& matrix) - : Base(matrix) {} - - /** \returns an expression of the factor L */ - inline const MatrixL matrixL() const { - eigen_assert(Base::m_factorizationIsOk && "Simplicial LLT not factorized"); - return Traits::getL(Base::m_matrix); - } + /** Default constructor */ + SimplicialLLT() : Base() {} + /** Constructs and performs the LLT factorization of \a matrix */ + explicit SimplicialLLT(const MatrixType &matrix) : Base(matrix) {} - /** \returns an expression of the factor U (= L^*) */ - inline const MatrixU matrixU() const { - eigen_assert(Base::m_factorizationIsOk && "Simplicial LLT not factorized"); - return Traits::getU(Base::m_matrix); - } - - /** Computes the sparse Cholesky decomposition of \a matrix */ - SimplicialLLT& compute(const MatrixType& matrix) - { - Base::template compute(matrix); - return *this; - } + /** \returns an expression of the factor L */ + inline const MatrixL matrixL() const + { + eigen_assert(Base::m_factorizationIsOk && "Simplicial LLT not factorized"); + return Traits::getL(Base::m_matrix); + } - /** Performs a symbolic decomposition on the sparcity of \a matrix. - * - * This function is particularly useful when solving for several problems having the same structure. - * - * \sa factorize() - */ - void analyzePattern(const MatrixType& a) - { - Base::analyzePattern(a, false); - } + /** \returns an expression of the factor U (= L^*) */ + inline const MatrixU matrixU() const + { + eigen_assert(Base::m_factorizationIsOk && "Simplicial LLT not factorized"); + return Traits::getU(Base::m_matrix); + } - /** Performs a numeric decomposition of \a matrix - * - * The given matrix must has the same sparcity than the matrix on which the symbolic decomposition has been performed. - * - * \sa analyzePattern() - */ - void factorize(const MatrixType& a) - { - Base::template factorize(a); - } + /** Computes the sparse Cholesky decomposition of \a matrix */ + SimplicialLLT &compute(const MatrixType &matrix) + { + Base::template compute(matrix); + return *this; + } - /** \returns the determinant of the underlying matrix from the current factorization */ - Scalar determinant() const - { - Scalar detL = Base::m_matrix.diagonal().prod(); - return numext::abs2(detL); - } + /** Performs a symbolic decomposition on the sparcity of \a matrix. + * + * This function is particularly useful when solving for several problems having the same structure. + * + * \sa factorize() + */ + void analyzePattern(const MatrixType &a) { Base::analyzePattern(a, false); } + + /** Performs a numeric decomposition of \a matrix + * + * The given matrix must has the same sparcity than the matrix on which the symbolic decomposition has been performed. + * + * \sa analyzePattern() + */ + void factorize(const MatrixType &a) { Base::template factorize(a); } + + /** \returns the determinant of the underlying matrix from the current factorization */ + Scalar determinant() const + { + Scalar detL = Base::m_matrix.diagonal().prod(); + return numext::abs2(detL); + } }; /** \ingroup SparseCholesky_Module - * \class SimplicialLDLT - * \brief A direct sparse LDLT Cholesky factorizations without square root. - * - * This class provides a LDL^T Cholesky factorizations without square root of sparse matrices that are - * selfadjoint and positive definite. The factorization allows for solving A.X = B where - * X and B can be either dense or sparse. - * - * In order to reduce the fill-in, a symmetric permutation P is applied prior to the factorization - * such that the factorized matrix is P A P^-1. - * - * \tparam _MatrixType the type of the sparse matrix A, it must be a SparseMatrix<> - * \tparam _UpLo the triangular part that will be used for the computations. It can be Lower - * or Upper. Default is Lower. - * \tparam _Ordering The ordering method to use, either AMDOrdering<> or NaturalOrdering<>. Default is AMDOrdering<> - * - * \implsparsesolverconcept - * - * \sa class SimplicialLLT, class AMDOrdering, class NaturalOrdering - */ + * \class SimplicialLDLT + * \brief A direct sparse LDLT Cholesky factorizations without square root. + * + * This class provides a LDL^T Cholesky factorizations without square root of sparse matrices that are + * selfadjoint and positive definite. The factorization allows for solving A.X = B where + * X and B can be either dense or sparse. + * + * In order to reduce the fill-in, a symmetric permutation P is applied prior to the factorization + * such that the factorized matrix is P A P^-1. + * + * \tparam _MatrixType the type of the sparse matrix A, it must be a SparseMatrix<> + * \tparam _UpLo the triangular part that will be used for the computations. It can be Lower + * or Upper. Default is Lower. + * \tparam _Ordering The ordering method to use, either AMDOrdering<> or NaturalOrdering<>. Default is AMDOrdering<> + * + * \implsparsesolverconcept + * + * \sa class SimplicialLLT, class AMDOrdering, class NaturalOrdering + */ template - class SimplicialLDLT : public SimplicialCholeskyBase > +class SimplicialLDLT : public SimplicialCholeskyBase> { public: - typedef _MatrixType MatrixType; - enum { UpLo = _UpLo }; - typedef SimplicialCholeskyBase Base; - typedef typename MatrixType::Scalar Scalar; - typedef typename MatrixType::RealScalar RealScalar; - typedef typename MatrixType::StorageIndex StorageIndex; - typedef SparseMatrix CholMatrixType; - typedef Matrix VectorType; - typedef internal::traits Traits; - typedef typename Traits::MatrixL MatrixL; - typedef typename Traits::MatrixU MatrixU; -public: - /** Default constructor */ - SimplicialLDLT() : Base() {} + typedef _MatrixType MatrixType; + enum { UpLo = _UpLo }; + typedef SimplicialCholeskyBase Base; + typedef typename MatrixType::Scalar Scalar; + typedef typename MatrixType::RealScalar RealScalar; + typedef typename MatrixType::StorageIndex StorageIndex; + typedef SparseMatrix CholMatrixType; + typedef Matrix VectorType; + typedef internal::traits Traits; + typedef typename Traits::MatrixL MatrixL; + typedef typename Traits::MatrixU MatrixU; - /** Constructs and performs the LLT factorization of \a matrix */ - explicit SimplicialLDLT(const MatrixType& matrix) - : Base(matrix) {} +public: + /** Default constructor */ + SimplicialLDLT() : Base() {} - /** \returns a vector expression of the diagonal D */ - inline const VectorType vectorD() const { - eigen_assert(Base::m_factorizationIsOk && "Simplicial LDLT not factorized"); - return Base::m_diag; - } - /** \returns an expression of the factor L */ - inline const MatrixL matrixL() const { - eigen_assert(Base::m_factorizationIsOk && "Simplicial LDLT not factorized"); - return Traits::getL(Base::m_matrix); - } + /** Constructs and performs the LLT factorization of \a matrix */ + explicit SimplicialLDLT(const MatrixType &matrix) : Base(matrix) {} - /** \returns an expression of the factor U (= L^*) */ - inline const MatrixU matrixU() const { - eigen_assert(Base::m_factorizationIsOk && "Simplicial LDLT not factorized"); - return Traits::getU(Base::m_matrix); - } + /** \returns a vector expression of the diagonal D */ + inline const VectorType vectorD() const + { + eigen_assert(Base::m_factorizationIsOk && "Simplicial LDLT not factorized"); + return Base::m_diag; + } + /** \returns an expression of the factor L */ + inline const MatrixL matrixL() const + { + eigen_assert(Base::m_factorizationIsOk && "Simplicial LDLT not factorized"); + return Traits::getL(Base::m_matrix); + } - /** Computes the sparse Cholesky decomposition of \a matrix */ - SimplicialLDLT& compute(const MatrixType& matrix) - { - Base::template compute(matrix); - return *this; - } - - /** Performs a symbolic decomposition on the sparcity of \a matrix. - * - * This function is particularly useful when solving for several problems having the same structure. - * - * \sa factorize() - */ - void analyzePattern(const MatrixType& a) - { - Base::analyzePattern(a, true); - } + /** \returns an expression of the factor U (= L^*) */ + inline const MatrixU matrixU() const + { + eigen_assert(Base::m_factorizationIsOk && "Simplicial LDLT not factorized"); + return Traits::getU(Base::m_matrix); + } - /** Performs a numeric decomposition of \a matrix - * - * The given matrix must has the same sparcity than the matrix on which the symbolic decomposition has been performed. - * - * \sa analyzePattern() - */ - void factorize(const MatrixType& a) - { - Base::template factorize(a); - } + /** Computes the sparse Cholesky decomposition of \a matrix */ + SimplicialLDLT &compute(const MatrixType &matrix) + { + Base::template compute(matrix); + return *this; + } - /** \returns the determinant of the underlying matrix from the current factorization */ - Scalar determinant() const - { - return Base::m_diag.prod(); - } + /** Performs a symbolic decomposition on the sparcity of \a matrix. + * + * This function is particularly useful when solving for several problems having the same structure. + * + * \sa factorize() + */ + void analyzePattern(const MatrixType &a) { Base::analyzePattern(a, true); } + + /** Performs a numeric decomposition of \a matrix + * + * The given matrix must has the same sparcity than the matrix on which the symbolic decomposition has been performed. + * + * \sa analyzePattern() + */ + void factorize(const MatrixType &a) { Base::template factorize(a); } + + /** \returns the determinant of the underlying matrix from the current factorization */ + Scalar determinant() const { return Base::m_diag.prod(); } }; /** \deprecated use SimplicialLDLT or class SimplicialLLT - * \ingroup SparseCholesky_Module - * \class SimplicialCholesky - * - * \sa class SimplicialLDLT, class SimplicialLLT - */ + * \ingroup SparseCholesky_Module + * \class SimplicialCholesky + * + * \sa class SimplicialLDLT, class SimplicialLLT + */ template - class SimplicialCholesky : public SimplicialCholeskyBase > +class SimplicialCholesky : public SimplicialCholeskyBase> { public: - typedef _MatrixType MatrixType; - enum { UpLo = _UpLo }; - typedef SimplicialCholeskyBase Base; - typedef typename MatrixType::Scalar Scalar; - typedef typename MatrixType::RealScalar RealScalar; - typedef typename MatrixType::StorageIndex StorageIndex; - typedef SparseMatrix CholMatrixType; - typedef Matrix VectorType; - typedef internal::traits Traits; - typedef internal::traits > LDLTTraits; - typedef internal::traits > LLTTraits; - public: - SimplicialCholesky() : Base(), m_LDLT(true) {} - - explicit SimplicialCholesky(const MatrixType& matrix) - : Base(), m_LDLT(true) - { - compute(matrix); - } + typedef _MatrixType MatrixType; + enum { UpLo = _UpLo }; + typedef SimplicialCholeskyBase Base; + typedef typename MatrixType::Scalar Scalar; + typedef typename MatrixType::RealScalar RealScalar; + typedef typename MatrixType::StorageIndex StorageIndex; + typedef SparseMatrix CholMatrixType; + typedef Matrix VectorType; + typedef internal::traits Traits; + typedef internal::traits> LDLTTraits; + typedef internal::traits> LLTTraits; - SimplicialCholesky& setMode(SimplicialCholeskyMode mode) - { - switch(mode) - { - case SimplicialCholeskyLLT: - m_LDLT = false; - break; - case SimplicialCholeskyLDLT: - m_LDLT = true; - break; - default: - break; - } - - return *this; - } +public: + SimplicialCholesky() : Base(), m_LDLT(true) {} - inline const VectorType vectorD() const { - eigen_assert(Base::m_factorizationIsOk && "Simplicial Cholesky not factorized"); - return Base::m_diag; - } - inline const CholMatrixType rawMatrix() const { - eigen_assert(Base::m_factorizationIsOk && "Simplicial Cholesky not factorized"); - return Base::m_matrix; - } - - /** Computes the sparse Cholesky decomposition of \a matrix */ - SimplicialCholesky& compute(const MatrixType& matrix) + explicit SimplicialCholesky(const MatrixType &matrix) : Base(), m_LDLT(true) { compute(matrix); } + + SimplicialCholesky &setMode(SimplicialCholeskyMode mode) + { + switch (mode) { + case SimplicialCholeskyLLT: + m_LDLT = false; + break; + case SimplicialCholeskyLDLT: + m_LDLT = true; + break; + default: + break; + } + + return *this; + } + + inline const VectorType vectorD() const + { + eigen_assert(Base::m_factorizationIsOk && "Simplicial Cholesky not factorized"); + return Base::m_diag; + } + inline const CholMatrixType rawMatrix() const + { + eigen_assert(Base::m_factorizationIsOk && "Simplicial Cholesky not factorized"); + return Base::m_matrix; + } + + /** Computes the sparse Cholesky decomposition of \a matrix */ + SimplicialCholesky &compute(const MatrixType &matrix) + { + if (m_LDLT) + Base::template compute(matrix); + else + Base::template compute(matrix); + return *this; + } + + /** Performs a symbolic decomposition on the sparcity of \a matrix. + * + * This function is particularly useful when solving for several problems having the same structure. + * + * \sa factorize() + */ + void analyzePattern(const MatrixType &a) { Base::analyzePattern(a, m_LDLT); } + + /** Performs a numeric decomposition of \a matrix + * + * The given matrix must has the same sparcity than the matrix on which the symbolic decomposition has been performed. + * + * \sa analyzePattern() + */ + void factorize(const MatrixType &a) + { + if (m_LDLT) + Base::template factorize(a); + else + Base::template factorize(a); + } + + /** \internal */ + template void _solve_impl(const MatrixBase &b, MatrixBase &dest) const + { + eigen_assert(Base::m_factorizationIsOk && "The decomposition is not in a valid state for solving, you must first call either compute() or symbolic()/numeric()"); + eigen_assert(Base::m_matrix.rows() == b.rows()); + + if (Base::m_info != Success) return; + + if (Base::m_P.size() > 0) + dest = Base::m_P * b; + else + dest = b; + + if (Base::m_matrix.nonZeros() > 0)// otherwise L==I { - if(m_LDLT) - Base::template compute(matrix); + if (m_LDLT) + LDLTTraits::getL(Base::m_matrix).solveInPlace(dest); else - Base::template compute(matrix); - return *this; + LLTTraits::getL(Base::m_matrix).solveInPlace(dest); } - /** Performs a symbolic decomposition on the sparcity of \a matrix. - * - * This function is particularly useful when solving for several problems having the same structure. - * - * \sa factorize() - */ - void analyzePattern(const MatrixType& a) - { - Base::analyzePattern(a, m_LDLT); - } + if (Base::m_diag.size() > 0) dest = Base::m_diag.asDiagonal().inverse() * dest; - /** Performs a numeric decomposition of \a matrix - * - * The given matrix must has the same sparcity than the matrix on which the symbolic decomposition has been performed. - * - * \sa analyzePattern() - */ - void factorize(const MatrixType& a) + if (Base::m_matrix.nonZeros() > 0)// otherwise I==I { - if(m_LDLT) - Base::template factorize(a); + if (m_LDLT) + LDLTTraits::getU(Base::m_matrix).solveInPlace(dest); else - Base::template factorize(a); + LLTTraits::getU(Base::m_matrix).solveInPlace(dest); } - /** \internal */ - template - void _solve_impl(const MatrixBase &b, MatrixBase &dest) const - { - eigen_assert(Base::m_factorizationIsOk && "The decomposition is not in a valid state for solving, you must first call either compute() or symbolic()/numeric()"); - eigen_assert(Base::m_matrix.rows()==b.rows()); + if (Base::m_P.size() > 0) dest = Base::m_Pinv * dest; + } - if(Base::m_info!=Success) - return; + /** \internal */ + template + void _solve_impl(const SparseMatrixBase &b, SparseMatrixBase &dest) const + { + internal::solve_sparse_through_dense_panels(*this, b, dest); + } - if(Base::m_P.size()>0) - dest = Base::m_P * b; - else - dest = b; - - if(Base::m_matrix.nonZeros()>0) // otherwise L==I - { - if(m_LDLT) - LDLTTraits::getL(Base::m_matrix).solveInPlace(dest); - else - LLTTraits::getL(Base::m_matrix).solveInPlace(dest); - } - - if(Base::m_diag.size()>0) - dest = Base::m_diag.asDiagonal().inverse() * dest; - - if (Base::m_matrix.nonZeros()>0) // otherwise I==I - { - if(m_LDLT) - LDLTTraits::getU(Base::m_matrix).solveInPlace(dest); - else - LLTTraits::getU(Base::m_matrix).solveInPlace(dest); - } - - if(Base::m_P.size()>0) - dest = Base::m_Pinv * dest; - } - - /** \internal */ - template - void _solve_impl(const SparseMatrixBase &b, SparseMatrixBase &dest) const - { - internal::solve_sparse_through_dense_panels(*this, b, dest); - } - - Scalar determinant() const - { - if(m_LDLT) - { - return Base::m_diag.prod(); - } - else - { - Scalar detL = Diagonal(Base::m_matrix).prod(); - return numext::abs2(detL); - } + Scalar determinant() const + { + if (m_LDLT) { + return Base::m_diag.prod(); + } else { + Scalar detL = Diagonal(Base::m_matrix).prod(); + return numext::abs2(detL); } - - protected: - bool m_LDLT; + } + +protected: + bool m_LDLT; }; template -void SimplicialCholeskyBase::ordering(const MatrixType& a, ConstCholMatrixPtr &pmat, CholMatrixType& ap) +void SimplicialCholeskyBase::ordering(const MatrixType &a, ConstCholMatrixPtr &pmat, CholMatrixType &ap) { - eigen_assert(a.rows()==a.cols()); + eigen_assert(a.rows() == a.cols()); const Index size = a.rows(); pmat = ≈ // Note that ordering methods compute the inverse permutation - if(!internal::is_same >::value) - { + if (!internal::is_same>::value) { { CholMatrixType C; C = a.template selfadjointView(); - + OrderingType ordering; - ordering(C,m_Pinv); + ordering(C, m_Pinv); } - if(m_Pinv.size()>0) m_P = m_Pinv.inverse(); - else m_P.resize(0); - - ap.resize(size,size); + if (m_Pinv.size() > 0) + m_P = m_Pinv.inverse(); + else + m_P.resize(0); + + ap.resize(size, size); ap.template selfadjointView() = a.template selfadjointView().twistedBy(m_P); - } - else - { + } else { m_Pinv.resize(0); m_P.resize(0); - if(int(UpLo)==int(Lower) || MatrixType::IsRowMajor) - { + if (int(UpLo) == int(Lower) || MatrixType::IsRowMajor) { // we have to transpose the lower part to to the upper one - ap.resize(size,size); + ap.resize(size, size); ap.template selfadjointView() = a.template selfadjointView(); - } - else - internal::simplicial_cholesky_grab_input::run(a, pmat, ap); - } + } else + internal::simplicial_cholesky_grab_input::run(a, pmat, ap); + } } -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_SIMPLICIAL_CHOLESKY_H +#endif// EIGEN_SIMPLICIAL_CHOLESKY_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/SparseCholesky/SimplicialCholesky_impl.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/SparseCholesky/SimplicialCholesky_impl.h index 31e06995..f3b4c87b 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/SparseCholesky/SimplicialCholesky_impl.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/SparseCholesky/SimplicialCholesky_impl.h @@ -48,7 +48,7 @@ LDL License: namespace Eigen { template -void SimplicialCholeskyBase::analyzePattern_preordered(const CholMatrixType& ap, bool doLDLT) +void SimplicialCholeskyBase::analyzePattern_preordered(const CholMatrixType &ap, bool doLDLT) { const StorageIndex size = StorageIndex(ap.rows()); m_matrix.resize(size, size); @@ -57,136 +57,120 @@ void SimplicialCholeskyBase::analyzePattern_preordered(const CholMatrix ei_declare_aligned_stack_constructed_variable(StorageIndex, tags, size, 0); - for(StorageIndex k = 0; k < size; ++k) - { + for (StorageIndex k = 0; k < size; ++k) { /* L(k,:) pattern: all nodes reachable in etree from nz in A(0:k-1,k) */ - m_parent[k] = -1; /* parent of k is not yet known */ - tags[k] = k; /* mark node k as visited */ - m_nonZerosPerCol[k] = 0; /* count of nonzeros in column k of L */ - for(typename CholMatrixType::InnerIterator it(ap,k); it; ++it) - { + m_parent[k] = -1; /* parent of k is not yet known */ + tags[k] = k; /* mark node k as visited */ + m_nonZerosPerCol[k] = 0; /* count of nonzeros in column k of L */ + for (typename CholMatrixType::InnerIterator it(ap, k); it; ++it) { StorageIndex i = it.index(); - if(i < k) - { + if (i < k) { /* follow path from i to root of etree, stop at flagged node */ - for(; tags[i] != k; i = m_parent[i]) - { + for (; tags[i] != k; i = m_parent[i]) { /* find parent of i if not yet determined */ - if (m_parent[i] == -1) - m_parent[i] = k; - m_nonZerosPerCol[i]++; /* L (k,i) is nonzero */ - tags[i] = k; /* mark i as visited */ + if (m_parent[i] == -1) m_parent[i] = k; + m_nonZerosPerCol[i]++; /* L (k,i) is nonzero */ + tags[i] = k; /* mark i as visited */ } } } } /* construct Lp index array from m_nonZerosPerCol column counts */ - StorageIndex* Lp = m_matrix.outerIndexPtr(); + StorageIndex *Lp = m_matrix.outerIndexPtr(); Lp[0] = 0; - for(StorageIndex k = 0; k < size; ++k) - Lp[k+1] = Lp[k] + m_nonZerosPerCol[k] + (doLDLT ? 0 : 1); + for (StorageIndex k = 0; k < size; ++k) Lp[k + 1] = Lp[k] + m_nonZerosPerCol[k] + (doLDLT ? 0 : 1); m_matrix.resizeNonZeros(Lp[size]); - m_isInitialized = true; - m_info = Success; - m_analysisIsOk = true; + m_isInitialized = true; + m_info = Success; + m_analysisIsOk = true; m_factorizationIsOk = false; } template template -void SimplicialCholeskyBase::factorize_preordered(const CholMatrixType& ap) +void SimplicialCholeskyBase::factorize_preordered(const CholMatrixType &ap) { using std::sqrt; eigen_assert(m_analysisIsOk && "You must first call analyzePattern()"); - eigen_assert(ap.rows()==ap.cols()); - eigen_assert(m_parent.size()==ap.rows()); - eigen_assert(m_nonZerosPerCol.size()==ap.rows()); + eigen_assert(ap.rows() == ap.cols()); + eigen_assert(m_parent.size() == ap.rows()); + eigen_assert(m_nonZerosPerCol.size() == ap.rows()); const StorageIndex size = StorageIndex(ap.rows()); - const StorageIndex* Lp = m_matrix.outerIndexPtr(); - StorageIndex* Li = m_matrix.innerIndexPtr(); - Scalar* Lx = m_matrix.valuePtr(); + const StorageIndex *Lp = m_matrix.outerIndexPtr(); + StorageIndex *Li = m_matrix.innerIndexPtr(); + Scalar *Lx = m_matrix.valuePtr(); ei_declare_aligned_stack_constructed_variable(Scalar, y, size, 0); - ei_declare_aligned_stack_constructed_variable(StorageIndex, pattern, size, 0); - ei_declare_aligned_stack_constructed_variable(StorageIndex, tags, size, 0); + ei_declare_aligned_stack_constructed_variable(StorageIndex, pattern, size, 0); + ei_declare_aligned_stack_constructed_variable(StorageIndex, tags, size, 0); bool ok = true; m_diag.resize(DoLDLT ? size : 0); - for(StorageIndex k = 0; k < size; ++k) - { + for (StorageIndex k = 0; k < size; ++k) { // compute nonzero pattern of kth row of L, in topological order - y[k] = 0.0; // Y(0:k) is now all zero - StorageIndex top = size; // stack for pattern is empty - tags[k] = k; // mark node k as visited - m_nonZerosPerCol[k] = 0; // count of nonzeros in column k of L - for(typename CholMatrixType::InnerIterator it(ap,k); it; ++it) - { + y[k] = 0.0;// Y(0:k) is now all zero + StorageIndex top = size;// stack for pattern is empty + tags[k] = k;// mark node k as visited + m_nonZerosPerCol[k] = 0;// count of nonzeros in column k of L + for (typename CholMatrixType::InnerIterator it(ap, k); it; ++it) { StorageIndex i = it.index(); - if(i <= k) - { - y[i] += numext::conj(it.value()); /* scatter A(i,k) into Y (sum duplicates) */ + if (i <= k) { + y[i] += numext::conj(it.value()); /* scatter A(i,k) into Y (sum duplicates) */ Index len; - for(len = 0; tags[i] != k; i = m_parent[i]) - { - pattern[len++] = i; /* L(k,i) is nonzero */ - tags[i] = k; /* mark i as visited */ + for (len = 0; tags[i] != k; i = m_parent[i]) { + pattern[len++] = i; /* L(k,i) is nonzero */ + tags[i] = k; /* mark i as visited */ } - while(len > 0) - pattern[--top] = pattern[--len]; + while (len > 0) pattern[--top] = pattern[--len]; } } /* compute numerical values kth row of L (a sparse triangular solve) */ - RealScalar d = numext::real(y[k]) * m_shiftScale + m_shiftOffset; // get D(k,k), apply the shift function, and clear Y(k) + RealScalar d = + numext::real(y[k]) * m_shiftScale + m_shiftOffset;// get D(k,k), apply the shift function, and clear Y(k) y[k] = 0.0; - for(; top < size; ++top) - { - Index i = pattern[top]; /* pattern[top:n-1] is pattern of L(:,k) */ - Scalar yi = y[i]; /* get and clear Y(i) */ + for (; top < size; ++top) { + Index i = pattern[top]; /* pattern[top:n-1] is pattern of L(:,k) */ + Scalar yi = y[i]; /* get and clear Y(i) */ y[i] = 0.0; /* the nonzero entry L(k,i) */ Scalar l_ki; - if(DoLDLT) + if (DoLDLT) l_ki = yi / m_diag[i]; else yi = l_ki = yi / Lx[Lp[i]]; Index p2 = Lp[i] + m_nonZerosPerCol[i]; Index p; - for(p = Lp[i] + (DoLDLT ? 0 : 1); p < p2; ++p) - y[Li[p]] -= numext::conj(Lx[p]) * yi; + for (p = Lp[i] + (DoLDLT ? 0 : 1); p < p2; ++p) y[Li[p]] -= numext::conj(Lx[p]) * yi; d -= numext::real(l_ki * numext::conj(yi)); - Li[p] = k; /* store L(k,i) in column form of L */ + Li[p] = k; /* store L(k,i) in column form of L */ Lx[p] = l_ki; - ++m_nonZerosPerCol[i]; /* increment count of nonzeros in col i */ + ++m_nonZerosPerCol[i]; /* increment count of nonzeros in col i */ } - if(DoLDLT) - { + if (DoLDLT) { m_diag[k] = d; - if(d == RealScalar(0)) - { - ok = false; /* failure, D(k,k) is zero */ + if (d == RealScalar(0)) { + ok = false; /* failure, D(k,k) is zero */ break; } - } - else - { + } else { Index p = Lp[k] + m_nonZerosPerCol[k]++; - Li[p] = k ; /* store L(k,k) = sqrt (d) in column k */ - if(d <= RealScalar(0)) { - ok = false; /* failure, matrix is not positive definite */ + Li[p] = k; /* store L(k,k) = sqrt (d) in column k */ + if (d <= RealScalar(0)) { + ok = false; /* failure, matrix is not positive definite */ break; } - Lx[p] = sqrt(d) ; + Lx[p] = sqrt(d); } } @@ -194,6 +178,6 @@ void SimplicialCholeskyBase::factorize_preordered(const CholMatrixType& m_factorizationIsOk = true; } -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_SIMPLICIAL_CHOLESKY_IMPL_H +#endif// EIGEN_SIMPLICIAL_CHOLESKY_IMPL_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/SparseCore/AmbiVector.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/SparseCore/AmbiVector.h index e0295f2a..2a03827e 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/SparseCore/AmbiVector.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/SparseCore/AmbiVector.h @@ -10,18 +10,17 @@ #ifndef EIGEN_AMBIVECTOR_H #define EIGEN_AMBIVECTOR_H -namespace Eigen { +namespace Eigen { namespace internal { -/** \internal - * Hybrid sparse/dense vector class designed for intensive read-write operations. - * - * See BasicSparseLLT and SparseProduct for usage examples. - */ -template -class AmbiVector -{ + /** \internal + * Hybrid sparse/dense vector class designed for intensive read-write operations. + * + * See BasicSparseLLT and SparseProduct for usage examples. + */ + template class AmbiVector + { public: typedef _Scalar Scalar; typedef _StorageIndex StorageIndex; @@ -39,13 +38,17 @@ class AmbiVector Index nonZeros() const; /** Specifies a sub-vector to work on */ - void setBounds(Index start, Index end) { m_start = convert_index(start); m_end = convert_index(end); } + void setBounds(Index start, Index end) + { + m_start = convert_index(start); + m_end = convert_index(end); + } void setZero(); void restart(); - Scalar& coeffRef(Index i); - Scalar& coeff(Index i); + Scalar &coeffRef(Index i); + Scalar &coeff(Index i); class Iterator; @@ -53,33 +56,26 @@ class AmbiVector void resize(Index size) { - if (m_allocatedSize < size) - reallocate(size); + if (m_allocatedSize < size) reallocate(size); m_size = convert_index(size); } StorageIndex size() const { return m_size; } protected: - StorageIndex convert_index(Index idx) - { - return internal::convert_index(idx); - } + StorageIndex convert_index(Index idx) { return internal::convert_index(idx); } void reallocate(Index size) { // if the size of the matrix is not too large, let's allocate a bit more than needed such // that we can handle dense vector even in sparse mode. delete[] m_buffer; - if (size<1000) - { - Index allocSize = (size * sizeof(ListEl) + sizeof(Scalar) - 1)/sizeof(Scalar); - m_allocatedElements = convert_index((allocSize*sizeof(Scalar))/sizeof(ListEl)); + if (size < 1000) { + Index allocSize = (size * sizeof(ListEl) + sizeof(Scalar) - 1) / sizeof(Scalar); + m_allocatedElements = convert_index((allocSize * sizeof(Scalar)) / sizeof(ListEl)); m_buffer = new Scalar[allocSize]; - } - else - { - m_allocatedElements = convert_index((size*sizeof(Scalar))/sizeof(ListEl)); + } else { + m_allocatedElements = convert_index((size * sizeof(Scalar)) / sizeof(ListEl)); m_buffer = new Scalar[size]; } m_size = convert_index(size); @@ -90,11 +86,11 @@ class AmbiVector void reallocateSparse() { Index copyElements = m_allocatedElements; - m_allocatedElements = (std::min)(StorageIndex(m_allocatedElements*1.5),m_size); + m_allocatedElements = (std::min)(StorageIndex(m_allocatedElements * 1.5), m_size); Index allocSize = m_allocatedElements * sizeof(ListEl); - allocSize = (allocSize + sizeof(Scalar) - 1)/sizeof(Scalar); - Scalar* newBuffer = new Scalar[allocSize]; - std::memcpy(newBuffer, m_buffer, copyElements * sizeof(ListEl)); + allocSize = (allocSize + sizeof(Scalar) - 1) / sizeof(Scalar); + Scalar *newBuffer = new Scalar[allocSize]; + std::memcpy(newBuffer, m_buffer, copyElements * sizeof(ListEl)); delete[] m_buffer; m_buffer = newBuffer; } @@ -109,7 +105,7 @@ class AmbiVector }; // used to store data in both mode - Scalar* m_buffer; + Scalar *m_buffer; Scalar m_zero; StorageIndex m_size; StorageIndex m_start; @@ -122,202 +118,168 @@ class AmbiVector StorageIndex m_llStart; StorageIndex m_llCurrent; StorageIndex m_llSize; -}; - -/** \returns the number of non zeros in the current sub vector */ -template -Index AmbiVector<_Scalar,_StorageIndex>::nonZeros() const -{ - if (m_mode==IsSparse) - return m_llSize; - else - return m_end - m_start; -} - -template -void AmbiVector<_Scalar,_StorageIndex>::init(double estimatedDensity) -{ - if (estimatedDensity>0.1) - init(IsDense); - else - init(IsSparse); -} - -template -void AmbiVector<_Scalar,_StorageIndex>::init(int mode) -{ - m_mode = mode; - if (m_mode==IsSparse) + }; + + /** \returns the number of non zeros in the current sub vector */ + template Index AmbiVector<_Scalar, _StorageIndex>::nonZeros() const + { + if (m_mode == IsSparse) + return m_llSize; + else + return m_end - m_start; + } + + template + void AmbiVector<_Scalar, _StorageIndex>::init(double estimatedDensity) { - m_llSize = 0; - m_llStart = -1; + if (estimatedDensity > 0.1) + init(IsDense); + else + init(IsSparse); } -} - -/** Must be called whenever we might perform a write access - * with an index smaller than the previous one. - * - * Don't worry, this function is extremely cheap. - */ -template -void AmbiVector<_Scalar,_StorageIndex>::restart() -{ - m_llCurrent = m_llStart; -} - -/** Set all coefficients of current subvector to zero */ -template -void AmbiVector<_Scalar,_StorageIndex>::setZero() -{ - if (m_mode==IsDense) + + template void AmbiVector<_Scalar, _StorageIndex>::init(int mode) { - for (Index i=m_start; i void AmbiVector<_Scalar, _StorageIndex>::restart() { - eigen_assert(m_mode==IsSparse); - m_llSize = 0; - m_llStart = -1; + m_llCurrent = m_llStart; } -} - -template -_Scalar& AmbiVector<_Scalar,_StorageIndex>::coeffRef(Index i) -{ - if (m_mode==IsDense) - return m_buffer[i]; - else + + /** Set all coefficients of current subvector to zero */ + template void AmbiVector<_Scalar, _StorageIndex>::setZero() { - ListEl* EIGEN_RESTRICT llElements = reinterpret_cast(m_buffer); - // TODO factorize the following code to reduce code generation - eigen_assert(m_mode==IsSparse); - if (m_llSize==0) - { - // this is the first element - m_llStart = 0; - m_llCurrent = 0; - ++m_llSize; - llElements[0].value = Scalar(0); - llElements[0].index = convert_index(i); - llElements[0].next = -1; - return llElements[0].value; - } - else if (i=llElements[m_llCurrent].index && "you must call restart() before inserting an element with lower or equal index"); - while (nextel >= 0 && llElements[nextel].index<=i) - { - m_llCurrent = nextel; - nextel = llElements[nextel].next; - } + } - if (llElements[m_llCurrent].index==i) - { - // the coefficient already exists and we found it ! - return llElements[m_llCurrent].value; - } - else - { - if (m_llSize>=m_allocatedElements) - { - reallocateSparse(); - llElements = reinterpret_cast(m_buffer); - } - eigen_internal_assert(m_llSize _Scalar &AmbiVector<_Scalar, _StorageIndex>::coeffRef(Index i) + { + if (m_mode == IsDense) + return m_buffer[i]; + else { + ListEl *EIGEN_RESTRICT llElements = reinterpret_cast(m_buffer); + // TODO factorize the following code to reduce code generation + eigen_assert(m_mode == IsSparse); + if (m_llSize == 0) { + // this is the first element + m_llStart = 0; + m_llCurrent = 0; + ++m_llSize; + llElements[0].value = Scalar(0); + llElements[0].index = convert_index(i); + llElements[0].next = -1; + return llElements[0].value; + } else if (i < llElements[m_llStart].index) { + // this is going to be the new first element of the list + ListEl &el = llElements[m_llSize]; el.value = Scalar(0); el.index = convert_index(i); - el.next = llElements[m_llCurrent].next; - llElements[m_llCurrent].next = m_llSize; + el.next = m_llStart; + m_llStart = m_llSize; ++m_llSize; + m_llCurrent = m_llStart; return el.value; + } else { + StorageIndex nextel = llElements[m_llCurrent].next; + eigen_assert(i >= llElements[m_llCurrent].index + && "you must call restart() before inserting an element with lower or equal index"); + while (nextel >= 0 && llElements[nextel].index <= i) { + m_llCurrent = nextel; + nextel = llElements[nextel].next; + } + + if (llElements[m_llCurrent].index == i) { + // the coefficient already exists and we found it ! + return llElements[m_llCurrent].value; + } else { + if (m_llSize >= m_allocatedElements) { + reallocateSparse(); + llElements = reinterpret_cast(m_buffer); + } + eigen_internal_assert(m_llSize < m_allocatedElements && "internal error: overflow in sparse mode"); + // let's insert a new coefficient + ListEl &el = llElements[m_llSize]; + el.value = Scalar(0); + el.index = convert_index(i); + el.next = llElements[m_llCurrent].next; + llElements[m_llCurrent].next = m_llSize; + ++m_llSize; + return el.value; + } } } } -} - -template -_Scalar& AmbiVector<_Scalar,_StorageIndex>::coeff(Index i) -{ - if (m_mode==IsDense) - return m_buffer[i]; - else - { - ListEl* EIGEN_RESTRICT llElements = reinterpret_cast(m_buffer); - eigen_assert(m_mode==IsSparse); - if ((m_llSize==0) || (i= 0 && llElements[elid].index _Scalar &AmbiVector<_Scalar, _StorageIndex>::coeff(Index i) + { + if (m_mode == IsDense) + return m_buffer[i]; + else { + ListEl *EIGEN_RESTRICT llElements = reinterpret_cast(m_buffer); + eigen_assert(m_mode == IsSparse); + if ((m_llSize == 0) || (i < llElements[m_llStart].index)) { return m_zero; + } else { + Index elid = m_llStart; + while (elid >= 0 && llElements[elid].index < i) elid = llElements[elid].next; + + if (llElements[elid].index == i) + return llElements[m_llCurrent].value; + else + return m_zero; + } } } -} -/** Iterator over the nonzero coefficients */ -template -class AmbiVector<_Scalar,_StorageIndex>::Iterator -{ + /** Iterator over the nonzero coefficients */ + template class AmbiVector<_Scalar, _StorageIndex>::Iterator + { public: typedef _Scalar Scalar; typedef typename NumTraits::Real RealScalar; /** Default constructor - * \param vec the vector on which we iterate - * \param epsilon the minimal value used to prune zero coefficients. - * In practice, all coefficients having a magnitude smaller than \a epsilon - * are skipped. - */ - explicit Iterator(const AmbiVector& vec, const RealScalar& epsilon = 0) - : m_vector(vec) + * \param vec the vector on which we iterate + * \param epsilon the minimal value used to prune zero coefficients. + * In practice, all coefficients having a magnitude smaller than \a epsilon + * are skipped. + */ + explicit Iterator(const AmbiVector &vec, const RealScalar &epsilon = 0) : m_vector(vec) { using std::abs; m_epsilon = epsilon; - m_isDense = m_vector.m_mode==IsDense; - if (m_isDense) - { - m_currentEl = 0; // this is to avoid a compilation warning - m_cachedValue = 0; // this is to avoid a compilation warning - m_cachedIndex = m_vector.m_start-1; + m_isDense = m_vector.m_mode == IsDense; + if (m_isDense) { + m_currentEl = 0;// this is to avoid a compilation warning + m_cachedValue = 0;// this is to avoid a compilation warning + m_cachedIndex = m_vector.m_start - 1; ++(*this); - } - else - { - ListEl* EIGEN_RESTRICT llElements = reinterpret_cast(m_vector.m_buffer); + } else { + ListEl *EIGEN_RESTRICT llElements = reinterpret_cast(m_vector.m_buffer); m_currentEl = m_vector.m_llStart; - while (m_currentEl>=0 && abs(llElements[m_currentEl].value)<=m_epsilon) + while (m_currentEl >= 0 && abs(llElements[m_currentEl].value) <= m_epsilon) m_currentEl = llElements[m_currentEl].next; - if (m_currentEl<0) - { - m_cachedValue = 0; // this is to avoid a compilation warning + if (m_currentEl < 0) { + m_cachedValue = 0;// this is to avoid a compilation warning m_cachedIndex = -1; - } - else - { + } else { m_cachedIndex = llElements[m_currentEl].index; m_cachedValue = llElements[m_currentEl].value; } @@ -327,33 +289,27 @@ class AmbiVector<_Scalar,_StorageIndex>::Iterator StorageIndex index() const { return m_cachedIndex; } Scalar value() const { return m_cachedValue; } - operator bool() const { return m_cachedIndex>=0; } + operator bool() const { return m_cachedIndex >= 0; } - Iterator& operator++() + Iterator &operator++() { using std::abs; - if (m_isDense) - { + if (m_isDense) { do { ++m_cachedIndex; - } while (m_cachedIndex(m_vector.m_buffer); + m_cachedIndex = -1; + } else { + ListEl *EIGEN_RESTRICT llElements = reinterpret_cast(m_vector.m_buffer); do { m_currentEl = llElements[m_currentEl].next; - } while (m_currentEl>=0 && abs(llElements[m_currentEl].value)<=m_epsilon); - if (m_currentEl<0) - { + } while (m_currentEl >= 0 && abs(llElements[m_currentEl].value) <= m_epsilon); + if (m_currentEl < 0) { m_cachedIndex = -1; - } - else - { + } else { m_cachedIndex = llElements[m_currentEl].index; m_cachedValue = llElements[m_currentEl].value; } @@ -362,16 +318,16 @@ class AmbiVector<_Scalar,_StorageIndex>::Iterator } protected: - const AmbiVector& m_vector; // the target vector - StorageIndex m_currentEl; // the current element in sparse/linked-list mode - RealScalar m_epsilon; // epsilon used to prune zero coefficients - StorageIndex m_cachedIndex; // current coordinate - Scalar m_cachedValue; // current value - bool m_isDense; // mode of the vector -}; + const AmbiVector &m_vector;// the target vector + StorageIndex m_currentEl;// the current element in sparse/linked-list mode + RealScalar m_epsilon;// epsilon used to prune zero coefficients + StorageIndex m_cachedIndex;// current coordinate + Scalar m_cachedValue;// current value + bool m_isDense;// mode of the vector + }; -} // end namespace internal +}// end namespace internal -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_AMBIVECTOR_H +#endif// EIGEN_AMBIVECTOR_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/SparseCore/CompressedStorage.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/SparseCore/CompressedStorage.h index d89fa0da..9fcf88d1 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/SparseCore/CompressedStorage.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/SparseCore/CompressedStorage.h @@ -10,56 +10,44 @@ #ifndef EIGEN_COMPRESSED_STORAGE_H #define EIGEN_COMPRESSED_STORAGE_H -namespace Eigen { +namespace Eigen { namespace internal { -/** \internal - * Stores a sparse set of values as a list of values and a list of indices. - * - */ -template -class CompressedStorage -{ + /** \internal + * Stores a sparse set of values as a list of values and a list of indices. + * + */ + template class CompressedStorage + { public: - typedef _Scalar Scalar; typedef _StorageIndex StorageIndex; protected: - typedef typename NumTraits::Real RealScalar; public: + CompressedStorage() : m_values(0), m_indices(0), m_size(0), m_allocatedSize(0) {} - CompressedStorage() - : m_values(0), m_indices(0), m_size(0), m_allocatedSize(0) - {} - - explicit CompressedStorage(Index size) - : m_values(0), m_indices(0), m_size(0), m_allocatedSize(0) - { - resize(size); - } + explicit CompressedStorage(Index size) : m_values(0), m_indices(0), m_size(0), m_allocatedSize(0) { resize(size); } - CompressedStorage(const CompressedStorage& other) - : m_values(0), m_indices(0), m_size(0), m_allocatedSize(0) + CompressedStorage(const CompressedStorage &other) : m_values(0), m_indices(0), m_size(0), m_allocatedSize(0) { *this = other; } - CompressedStorage& operator=(const CompressedStorage& other) + CompressedStorage &operator=(const CompressedStorage &other) { resize(other.size()); - if(other.size()>0) - { - internal::smart_copy(other.m_values, other.m_values + m_size, m_values); + if (other.size() > 0) { + internal::smart_copy(other.m_values, other.m_values + m_size, m_values); internal::smart_copy(other.m_indices, other.m_indices + m_size, m_indices); } return *this; } - void swap(CompressedStorage& other) + void swap(CompressedStorage &other) { std::swap(m_values, other.m_values); std::swap(m_indices, other.m_indices); @@ -76,32 +64,29 @@ class CompressedStorage void reserve(Index size) { Index newAllocatedSize = m_size + size; - if (newAllocatedSize > m_allocatedSize) - reallocate(newAllocatedSize); + if (newAllocatedSize > m_allocatedSize) reallocate(newAllocatedSize); } void squeeze() { - if (m_allocatedSize>m_size) - reallocate(m_size); + if (m_allocatedSize > m_size) reallocate(m_size); } void resize(Index size, double reserveSizeFactor = 0) { - if (m_allocatedSize)(NumTraits::highest(), size + Index(reserveSizeFactor*double(size))); - if(realloc_size)(NumTraits::highest(), size + Index(reserveSizeFactor * double(size))); + if (realloc_size < size) internal::throw_std_bad_alloc(); reallocate(realloc_size); } m_size = size; } - void append(const Scalar& v, Index i) + void append(const Scalar &v, Index i) { Index id = m_size; - resize(m_size+1, 1); + resize(m_size + 1, 1); m_values[id] = v; m_indices[id] = internal::convert_index(i); } @@ -110,31 +95,43 @@ class CompressedStorage inline Index allocatedSize() const { return m_allocatedSize; } inline void clear() { m_size = 0; } - const Scalar* valuePtr() const { return m_values; } - Scalar* valuePtr() { return m_values; } - const StorageIndex* indexPtr() const { return m_indices; } - StorageIndex* indexPtr() { return m_indices; } + const Scalar *valuePtr() const { return m_values; } + Scalar *valuePtr() { return m_values; } + const StorageIndex *indexPtr() const { return m_indices; } + StorageIndex *indexPtr() { return m_indices; } - inline Scalar& value(Index i) { eigen_internal_assert(m_values!=0); return m_values[i]; } - inline const Scalar& value(Index i) const { eigen_internal_assert(m_values!=0); return m_values[i]; } - - inline StorageIndex& index(Index i) { eigen_internal_assert(m_indices!=0); return m_indices[i]; } - inline const StorageIndex& index(Index i) const { eigen_internal_assert(m_indices!=0); return m_indices[i]; } + inline Scalar &value(Index i) + { + eigen_internal_assert(m_values != 0); + return m_values[i]; + } + inline const Scalar &value(Index i) const + { + eigen_internal_assert(m_values != 0); + return m_values[i]; + } - /** \returns the largest \c k such that for all \c j in [0,k) index[\c j]\<\a key */ - inline Index searchLowerIndex(Index key) const + inline StorageIndex &index(Index i) + { + eigen_internal_assert(m_indices != 0); + return m_indices[i]; + } + inline const StorageIndex &index(Index i) const { - return searchLowerIndex(0, m_size, key); + eigen_internal_assert(m_indices != 0); + return m_indices[i]; } + /** \returns the largest \c k such that for all \c j in [0,k) index[\c j]\<\a key */ + inline Index searchLowerIndex(Index key) const { return searchLowerIndex(0, m_size, key); } + /** \returns the largest \c k in [start,end) such that for all \c j in [start,k) index[\c j]\<\a key */ inline Index searchLowerIndex(Index start, Index end, Index key) const { - while(end>start) - { - Index mid = (end+start)>>1; - if (m_indices[mid] start) { + Index mid = (end + start) >> 1; + if (m_indices[mid] < key) + start = mid + 1; else end = mid; } @@ -142,63 +139,58 @@ class CompressedStorage } /** \returns the stored value at index \a key - * If the value does not exist, then the value \a defaultValue is returned without any insertion. */ - inline Scalar at(Index key, const Scalar& defaultValue = Scalar(0)) const + * If the value does not exist, then the value \a defaultValue is returned without any insertion. */ + inline Scalar at(Index key, const Scalar &defaultValue = Scalar(0)) const { - if (m_size==0) + if (m_size == 0) return defaultValue; - else if (key==m_indices[m_size-1]) - return m_values[m_size-1]; + else if (key == m_indices[m_size - 1]) + return m_values[m_size - 1]; // ^^ optimization: let's first check if it is the last coefficient // (very common in high level algorithms) - const Index id = searchLowerIndex(0,m_size-1,key); - return ((id=end) + if (start >= end) return defaultValue; - else if (end>start && key==m_indices[end-1]) - return m_values[end-1]; + else if (end > start && key == m_indices[end - 1]) + return m_values[end - 1]; // ^^ optimization: let's first check if it is the last coefficient // (very common in high level algorithms) - const Index id = searchLowerIndex(start,end-1,key); - return ((id=m_size || m_indices[id]!=key) - { - if (m_allocatedSize= m_size || m_indices[id] != key) { + if (m_allocatedSize < m_size + 1) { + m_allocatedSize = 2 * (m_size + 1); internal::scoped_array newValues(m_allocatedSize); internal::scoped_array newIndices(m_allocatedSize); // copy first chunk - internal::smart_copy(m_values, m_values +id, newValues.ptr()); - internal::smart_copy(m_indices, m_indices+id, newIndices.ptr()); + internal::smart_copy(m_values, m_values + id, newValues.ptr()); + internal::smart_copy(m_indices, m_indices + id, newIndices.ptr()); // copy the rest - if(m_size>id) - { - internal::smart_copy(m_values +id, m_values +m_size, newValues.ptr() +id+1); - internal::smart_copy(m_indices+id, m_indices+m_size, newIndices.ptr()+id+1); + if (m_size > id) { + internal::smart_copy(m_values + id, m_values + m_size, newValues.ptr() + id + 1); + internal::smart_copy(m_indices + id, m_indices + m_size, newIndices.ptr() + id + 1); } - std::swap(m_values,newValues.ptr()); - std::swap(m_indices,newIndices.ptr()); - } - else if(m_size>id) - { - internal::smart_memmove(m_values +id, m_values +m_size, m_values +id+1); - internal::smart_memmove(m_indices+id, m_indices+m_size, m_indices+id+1); + std::swap(m_values, newValues.ptr()); + std::swap(m_indices, newIndices.ptr()); + } else if (m_size > id) { + internal::smart_memmove(m_values + id, m_values + m_size, m_values + id + 1); + internal::smart_memmove(m_indices + id, m_indices + m_size, m_indices + id + 1); } m_size++; m_indices[id] = internal::convert_index(key); @@ -207,52 +199,48 @@ class CompressedStorage return m_values[id]; } - void prune(const Scalar& reference, const RealScalar& epsilon = NumTraits::dummy_precision()) + void prune(const Scalar &reference, const RealScalar &epsilon = NumTraits::dummy_precision()) { Index k = 0; Index n = size(); - for (Index i=0; i newValues(size); internal::scoped_array newIndices(size); Index copySize = (std::min)(size, m_size); - if (copySize>0) { - internal::smart_copy(m_values, m_values+copySize, newValues.ptr()); - internal::smart_copy(m_indices, m_indices+copySize, newIndices.ptr()); + if (copySize > 0) { + internal::smart_copy(m_values, m_values + copySize, newValues.ptr()); + internal::smart_copy(m_indices, m_indices + copySize, newIndices.ptr()); } - std::swap(m_values,newValues.ptr()); - std::swap(m_indices,newIndices.ptr()); + std::swap(m_values, newValues.ptr()); + std::swap(m_indices, newIndices.ptr()); m_allocatedSize = size; } protected: - Scalar* m_values; - StorageIndex* m_indices; + Scalar *m_values; + StorageIndex *m_indices; Index m_size; Index m_allocatedSize; + }; -}; - -} // end namespace internal +}// end namespace internal -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_COMPRESSED_STORAGE_H +#endif// EIGEN_COMPRESSED_STORAGE_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/SparseCore/ConservativeSparseSparseProduct.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/SparseCore/ConservativeSparseSparseProduct.h index 9db119b6..bffc5c75 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/SparseCore/ConservativeSparseSparseProduct.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/SparseCore/ConservativeSparseSparseProduct.h @@ -10,343 +10,332 @@ #ifndef EIGEN_CONSERVATIVESPARSESPARSEPRODUCT_H #define EIGEN_CONSERVATIVESPARSESPARSEPRODUCT_H -namespace Eigen { +namespace Eigen { namespace internal { -template -static void conservative_sparse_sparse_product_impl(const Lhs& lhs, const Rhs& rhs, ResultType& res, bool sortedInsertion = false) -{ - typedef typename remove_all::type::Scalar LhsScalar; - typedef typename remove_all::type::Scalar RhsScalar; - typedef typename remove_all::type::Scalar ResScalar; - - // make sure to call innerSize/outerSize since we fake the storage order. - Index rows = lhs.innerSize(); - Index cols = rhs.outerSize(); - eigen_assert(lhs.outerSize() == rhs.innerSize()); - - ei_declare_aligned_stack_constructed_variable(bool, mask, rows, 0); - ei_declare_aligned_stack_constructed_variable(ResScalar, values, rows, 0); - ei_declare_aligned_stack_constructed_variable(Index, indices, rows, 0); - - std::memset(mask,0,sizeof(bool)*rows); - - evaluator lhsEval(lhs); - evaluator rhsEval(rhs); - - // estimate the number of non zero entries - // given a rhs column containing Y non zeros, we assume that the respective Y columns - // of the lhs differs in average of one non zeros, thus the number of non zeros for - // the product of a rhs column with the lhs is X+Y where X is the average number of non zero - // per column of the lhs. - // Therefore, we have nnz(lhs*rhs) = nnz(lhs) + nnz(rhs) - Index estimated_nnz_prod = lhsEval.nonZerosEstimate() + rhsEval.nonZerosEstimate(); - - res.setZero(); - res.reserve(Index(estimated_nnz_prod)); - // we compute each column of the result, one after the other - for (Index j=0; j + static void conservative_sparse_sparse_product_impl(const Lhs &lhs, + const Rhs &rhs, + ResultType &res, + bool sortedInsertion = false) { - - res.startVec(j); - Index nnz = 0; - for (typename evaluator::InnerIterator rhsIt(rhsEval, j); rhsIt; ++rhsIt) - { - RhsScalar y = rhsIt.value(); - Index k = rhsIt.index(); - for (typename evaluator::InnerIterator lhsIt(lhsEval, k); lhsIt; ++lhsIt) - { - Index i = lhsIt.index(); - LhsScalar x = lhsIt.value(); - if(!mask[i]) - { - mask[i] = true; - values[i] = x * y; - indices[nnz] = i; - ++nnz; + typedef typename remove_all::type::Scalar LhsScalar; + typedef typename remove_all::type::Scalar RhsScalar; + typedef typename remove_all::type::Scalar ResScalar; + + // make sure to call innerSize/outerSize since we fake the storage order. + Index rows = lhs.innerSize(); + Index cols = rhs.outerSize(); + eigen_assert(lhs.outerSize() == rhs.innerSize()); + + ei_declare_aligned_stack_constructed_variable(bool, mask, rows, 0); + ei_declare_aligned_stack_constructed_variable(ResScalar, values, rows, 0); + ei_declare_aligned_stack_constructed_variable(Index, indices, rows, 0); + + std::memset(mask, 0, sizeof(bool) * rows); + + evaluator lhsEval(lhs); + evaluator rhsEval(rhs); + + // estimate the number of non zero entries + // given a rhs column containing Y non zeros, we assume that the respective Y columns + // of the lhs differs in average of one non zeros, thus the number of non zeros for + // the product of a rhs column with the lhs is X+Y where X is the average number of non zero + // per column of the lhs. + // Therefore, we have nnz(lhs*rhs) = nnz(lhs) + nnz(rhs) + Index estimated_nnz_prod = lhsEval.nonZerosEstimate() + rhsEval.nonZerosEstimate(); + + res.setZero(); + res.reserve(Index(estimated_nnz_prod)); + // we compute each column of the result, one after the other + for (Index j = 0; j < cols; ++j) { + + res.startVec(j); + Index nnz = 0; + for (typename evaluator::InnerIterator rhsIt(rhsEval, j); rhsIt; ++rhsIt) { + RhsScalar y = rhsIt.value(); + Index k = rhsIt.index(); + for (typename evaluator::InnerIterator lhsIt(lhsEval, k); lhsIt; ++lhsIt) { + Index i = lhsIt.index(); + LhsScalar x = lhsIt.value(); + if (!mask[i]) { + mask[i] = true; + values[i] = x * y; + indices[nnz] = i; + ++nnz; + } else + values[i] += x * y; } - else - values[i] += x * y; } - } - if(!sortedInsertion) - { - // unordered insertion - for(Index k=0; k use a quick sort - // otherwise => loop through the entire vector - // In order to avoid to perform an expensive log2 when the - // result is clearly very sparse we use a linear bound up to 200. - if((nnz<200 && nnz1) std::sort(indices,indices+nnz); - for(Index k=0; k use a quick sort + // otherwise => loop through the entire vector + // In order to avoid to perform an expensive log2 when the + // result is clearly very sparse we use a linear bound up to 200. + if ((nnz < 200 && nnz < t200) || nnz * numext::log2(int(nnz)) < t) { + if (nnz > 1) std::sort(indices, indices + nnz); + for (Index k = 0; k < nnz; ++k) { + Index i = indices[k]; + res.insertBackByOuterInner(j, i) = values[i]; mask[i] = false; - res.insertBackByOuterInner(j,i) = values[i]; + } + } else { + // dense path + for (Index i = 0; i < rows; ++i) { + if (mask[i]) { + mask[i] = false; + res.insertBackByOuterInner(j, i) = values[i]; + } } } } } + res.finalize(); } - res.finalize(); -} -} // end namespace internal +}// end namespace internal namespace internal { -template::Flags&RowMajorBit) ? RowMajor : ColMajor, - int RhsStorageOrder = (traits::Flags&RowMajorBit) ? RowMajor : ColMajor, - int ResStorageOrder = (traits::Flags&RowMajorBit) ? RowMajor : ColMajor> -struct conservative_sparse_sparse_product_selector; + template::Flags & RowMajorBit) ? RowMajor : ColMajor, + int RhsStorageOrder = (traits::Flags & RowMajorBit) ? RowMajor : ColMajor, + int ResStorageOrder = (traits::Flags & RowMajorBit) ? RowMajor : ColMajor> + struct conservative_sparse_sparse_product_selector; -template -struct conservative_sparse_sparse_product_selector -{ - typedef typename remove_all::type LhsCleaned; - typedef typename LhsCleaned::Scalar Scalar; - - static void run(const Lhs& lhs, const Rhs& rhs, ResultType& res) + template + struct conservative_sparse_sparse_product_selector { - typedef SparseMatrix RowMajorMatrix; - typedef SparseMatrix ColMajorMatrixAux; - typedef typename sparse_eval::type ColMajorMatrix; - - // If the result is tall and thin (in the extreme case a column vector) - // then it is faster to sort the coefficients inplace instead of transposing twice. - // FIXME, the following heuristic is probably not very good. - if(lhs.rows()>rhs.cols()) - { - ColMajorMatrix resCol(lhs.rows(),rhs.cols()); - // perform sorted insertion - internal::conservative_sparse_sparse_product_impl(lhs, rhs, resCol, true); - res = resCol.markAsRValue(); - } - else + typedef typename remove_all::type LhsCleaned; + typedef typename LhsCleaned::Scalar Scalar; + + static void run(const Lhs &lhs, const Rhs &rhs, ResultType &res) { - ColMajorMatrixAux resCol(lhs.rows(),rhs.cols()); - // ressort to transpose to sort the entries - internal::conservative_sparse_sparse_product_impl(lhs, rhs, resCol, false); - RowMajorMatrix resRow(resCol); - res = resRow.markAsRValue(); + typedef SparseMatrix RowMajorMatrix; + typedef SparseMatrix ColMajorMatrixAux; + typedef typename sparse_eval::type ColMajorMatrix; + + // If the result is tall and thin (in the extreme case a column vector) + // then it is faster to sort the coefficients inplace instead of transposing twice. + // FIXME, the following heuristic is probably not very good. + if (lhs.rows() > rhs.cols()) { + ColMajorMatrix resCol(lhs.rows(), rhs.cols()); + // perform sorted insertion + internal::conservative_sparse_sparse_product_impl(lhs, rhs, resCol, true); + res = resCol.markAsRValue(); + } else { + ColMajorMatrixAux resCol(lhs.rows(), rhs.cols()); + // ressort to transpose to sort the entries + internal::conservative_sparse_sparse_product_impl(lhs, rhs, resCol, false); + RowMajorMatrix resRow(resCol); + res = resRow.markAsRValue(); + } } - } -}; + }; -template -struct conservative_sparse_sparse_product_selector -{ - static void run(const Lhs& lhs, const Rhs& rhs, ResultType& res) + template + struct conservative_sparse_sparse_product_selector { - typedef SparseMatrix RowMajorRhs; - typedef SparseMatrix RowMajorRes; - RowMajorRhs rhsRow = rhs; - RowMajorRes resRow(lhs.rows(), rhs.cols()); - internal::conservative_sparse_sparse_product_impl(rhsRow, lhs, resRow); - res = resRow; - } -}; + static void run(const Lhs &lhs, const Rhs &rhs, ResultType &res) + { + typedef SparseMatrix RowMajorRhs; + typedef SparseMatrix RowMajorRes; + RowMajorRhs rhsRow = rhs; + RowMajorRes resRow(lhs.rows(), rhs.cols()); + internal::conservative_sparse_sparse_product_impl(rhsRow, lhs, resRow); + res = resRow; + } + }; -template -struct conservative_sparse_sparse_product_selector -{ - static void run(const Lhs& lhs, const Rhs& rhs, ResultType& res) + template + struct conservative_sparse_sparse_product_selector { - typedef SparseMatrix RowMajorLhs; - typedef SparseMatrix RowMajorRes; - RowMajorLhs lhsRow = lhs; - RowMajorRes resRow(lhs.rows(), rhs.cols()); - internal::conservative_sparse_sparse_product_impl(rhs, lhsRow, resRow); - res = resRow; - } -}; + static void run(const Lhs &lhs, const Rhs &rhs, ResultType &res) + { + typedef SparseMatrix RowMajorLhs; + typedef SparseMatrix RowMajorRes; + RowMajorLhs lhsRow = lhs; + RowMajorRes resRow(lhs.rows(), rhs.cols()); + internal::conservative_sparse_sparse_product_impl(rhs, lhsRow, resRow); + res = resRow; + } + }; -template -struct conservative_sparse_sparse_product_selector -{ - static void run(const Lhs& lhs, const Rhs& rhs, ResultType& res) + template + struct conservative_sparse_sparse_product_selector { - typedef SparseMatrix RowMajorMatrix; - RowMajorMatrix resRow(lhs.rows(), rhs.cols()); - internal::conservative_sparse_sparse_product_impl(rhs, lhs, resRow); - res = resRow; - } -}; - + static void run(const Lhs &lhs, const Rhs &rhs, ResultType &res) + { + typedef SparseMatrix RowMajorMatrix; + RowMajorMatrix resRow(lhs.rows(), rhs.cols()); + internal::conservative_sparse_sparse_product_impl(rhs, lhs, resRow); + res = resRow; + } + }; -template -struct conservative_sparse_sparse_product_selector -{ - typedef typename traits::type>::Scalar Scalar; - static void run(const Lhs& lhs, const Rhs& rhs, ResultType& res) + template + struct conservative_sparse_sparse_product_selector { - typedef SparseMatrix ColMajorMatrix; - ColMajorMatrix resCol(lhs.rows(), rhs.cols()); - internal::conservative_sparse_sparse_product_impl(lhs, rhs, resCol); - res = resCol; - } -}; + typedef typename traits::type>::Scalar Scalar; + + static void run(const Lhs &lhs, const Rhs &rhs, ResultType &res) + { + typedef SparseMatrix ColMajorMatrix; + ColMajorMatrix resCol(lhs.rows(), rhs.cols()); + internal::conservative_sparse_sparse_product_impl(lhs, rhs, resCol); + res = resCol; + } + }; -template -struct conservative_sparse_sparse_product_selector -{ - static void run(const Lhs& lhs, const Rhs& rhs, ResultType& res) + template + struct conservative_sparse_sparse_product_selector { - typedef SparseMatrix ColMajorLhs; - typedef SparseMatrix ColMajorRes; - ColMajorLhs lhsCol = lhs; - ColMajorRes resCol(lhs.rows(), rhs.cols()); - internal::conservative_sparse_sparse_product_impl(lhsCol, rhs, resCol); - res = resCol; - } -}; + static void run(const Lhs &lhs, const Rhs &rhs, ResultType &res) + { + typedef SparseMatrix ColMajorLhs; + typedef SparseMatrix ColMajorRes; + ColMajorLhs lhsCol = lhs; + ColMajorRes resCol(lhs.rows(), rhs.cols()); + internal::conservative_sparse_sparse_product_impl(lhsCol, rhs, resCol); + res = resCol; + } + }; -template -struct conservative_sparse_sparse_product_selector -{ - static void run(const Lhs& lhs, const Rhs& rhs, ResultType& res) + template + struct conservative_sparse_sparse_product_selector { - typedef SparseMatrix ColMajorRhs; - typedef SparseMatrix ColMajorRes; - ColMajorRhs rhsCol = rhs; - ColMajorRes resCol(lhs.rows(), rhs.cols()); - internal::conservative_sparse_sparse_product_impl(lhs, rhsCol, resCol); - res = resCol; - } -}; + static void run(const Lhs &lhs, const Rhs &rhs, ResultType &res) + { + typedef SparseMatrix ColMajorRhs; + typedef SparseMatrix ColMajorRes; + ColMajorRhs rhsCol = rhs; + ColMajorRes resCol(lhs.rows(), rhs.cols()); + internal::conservative_sparse_sparse_product_impl(lhs, rhsCol, resCol); + res = resCol; + } + }; -template -struct conservative_sparse_sparse_product_selector -{ - static void run(const Lhs& lhs, const Rhs& rhs, ResultType& res) + template + struct conservative_sparse_sparse_product_selector { - typedef SparseMatrix RowMajorMatrix; - typedef SparseMatrix ColMajorMatrix; - RowMajorMatrix resRow(lhs.rows(),rhs.cols()); - internal::conservative_sparse_sparse_product_impl(rhs, lhs, resRow); - // sort the non zeros: - ColMajorMatrix resCol(resRow); - res = resCol; - } -}; + static void run(const Lhs &lhs, const Rhs &rhs, ResultType &res) + { + typedef SparseMatrix RowMajorMatrix; + typedef SparseMatrix ColMajorMatrix; + RowMajorMatrix resRow(lhs.rows(), rhs.cols()); + internal::conservative_sparse_sparse_product_impl(rhs, lhs, resRow); + // sort the non zeros: + ColMajorMatrix resCol(resRow); + res = resCol; + } + }; -} // end namespace internal +}// end namespace internal namespace internal { -template -static void sparse_sparse_to_dense_product_impl(const Lhs& lhs, const Rhs& rhs, ResultType& res) -{ - typedef typename remove_all::type::Scalar LhsScalar; - typedef typename remove_all::type::Scalar RhsScalar; - Index cols = rhs.outerSize(); - eigen_assert(lhs.outerSize() == rhs.innerSize()); - - evaluator lhsEval(lhs); - evaluator rhsEval(rhs); - - for (Index j=0; j + static void sparse_sparse_to_dense_product_impl(const Lhs &lhs, const Rhs &rhs, ResultType &res) { - for (typename evaluator::InnerIterator rhsIt(rhsEval, j); rhsIt; ++rhsIt) - { - RhsScalar y = rhsIt.value(); - Index k = rhsIt.index(); - for (typename evaluator::InnerIterator lhsIt(lhsEval, k); lhsIt; ++lhsIt) - { - Index i = lhsIt.index(); - LhsScalar x = lhsIt.value(); - res.coeffRef(i,j) += x * y; + typedef typename remove_all::type::Scalar LhsScalar; + typedef typename remove_all::type::Scalar RhsScalar; + Index cols = rhs.outerSize(); + eigen_assert(lhs.outerSize() == rhs.innerSize()); + + evaluator lhsEval(lhs); + evaluator rhsEval(rhs); + + for (Index j = 0; j < cols; ++j) { + for (typename evaluator::InnerIterator rhsIt(rhsEval, j); rhsIt; ++rhsIt) { + RhsScalar y = rhsIt.value(); + Index k = rhsIt.index(); + for (typename evaluator::InnerIterator lhsIt(lhsEval, k); lhsIt; ++lhsIt) { + Index i = lhsIt.index(); + LhsScalar x = lhsIt.value(); + res.coeffRef(i, j) += x * y; + } } } } -} -} // end namespace internal +}// end namespace internal namespace internal { -template::Flags&RowMajorBit) ? RowMajor : ColMajor, - int RhsStorageOrder = (traits::Flags&RowMajorBit) ? RowMajor : ColMajor> -struct sparse_sparse_to_dense_product_selector; + template::Flags & RowMajorBit) ? RowMajor : ColMajor, + int RhsStorageOrder = (traits::Flags & RowMajorBit) ? RowMajor : ColMajor> + struct sparse_sparse_to_dense_product_selector; -template -struct sparse_sparse_to_dense_product_selector -{ - static void run(const Lhs& lhs, const Rhs& rhs, ResultType& res) + template + struct sparse_sparse_to_dense_product_selector { - internal::sparse_sparse_to_dense_product_impl(lhs, rhs, res); - } -}; + static void run(const Lhs &lhs, const Rhs &rhs, ResultType &res) + { + internal::sparse_sparse_to_dense_product_impl(lhs, rhs, res); + } + }; -template -struct sparse_sparse_to_dense_product_selector -{ - static void run(const Lhs& lhs, const Rhs& rhs, ResultType& res) + template + struct sparse_sparse_to_dense_product_selector { - typedef SparseMatrix ColMajorLhs; - ColMajorLhs lhsCol(lhs); - internal::sparse_sparse_to_dense_product_impl(lhsCol, rhs, res); - } -}; + static void run(const Lhs &lhs, const Rhs &rhs, ResultType &res) + { + typedef SparseMatrix ColMajorLhs; + ColMajorLhs lhsCol(lhs); + internal::sparse_sparse_to_dense_product_impl(lhsCol, rhs, res); + } + }; -template -struct sparse_sparse_to_dense_product_selector -{ - static void run(const Lhs& lhs, const Rhs& rhs, ResultType& res) + template + struct sparse_sparse_to_dense_product_selector { - typedef SparseMatrix ColMajorRhs; - ColMajorRhs rhsCol(rhs); - internal::sparse_sparse_to_dense_product_impl(lhs, rhsCol, res); - } -}; + static void run(const Lhs &lhs, const Rhs &rhs, ResultType &res) + { + typedef SparseMatrix ColMajorRhs; + ColMajorRhs rhsCol(rhs); + internal::sparse_sparse_to_dense_product_impl(lhs, rhsCol, res); + } + }; -template -struct sparse_sparse_to_dense_product_selector -{ - static void run(const Lhs& lhs, const Rhs& rhs, ResultType& res) + template + struct sparse_sparse_to_dense_product_selector { - Transpose trRes(res); - internal::sparse_sparse_to_dense_product_impl >(rhs, lhs, trRes); - } -}; + static void run(const Lhs &lhs, const Rhs &rhs, ResultType &res) + { + Transpose trRes(res); + internal::sparse_sparse_to_dense_product_impl>(rhs, lhs, trRes); + } + }; -} // end namespace internal +}// end namespace internal -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_CONSERVATIVESPARSESPARSEPRODUCT_H +#endif// EIGEN_CONSERVATIVESPARSESPARSEPRODUCT_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/SparseCore/MappedSparseMatrix.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/SparseCore/MappedSparseMatrix.h index 67718c85..d77c5f71 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/SparseCore/MappedSparseMatrix.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/SparseCore/MappedSparseMatrix.h @@ -13,55 +13,61 @@ namespace Eigen { /** \deprecated Use Map > - * \class MappedSparseMatrix - * - * \brief Sparse matrix - * - * \param _Scalar the scalar type, i.e. the type of the coefficients - * - * See http://www.netlib.org/linalg/html_templates/node91.html for details on the storage scheme. - * - */ + * \class MappedSparseMatrix + * + * \brief Sparse matrix + * + * \param _Scalar the scalar type, i.e. the type of the coefficients + * + * See http://www.netlib.org/linalg/html_templates/node91.html for details on the storage scheme. + * + */ namespace internal { -template -struct traits > : traits > -{}; -} // end namespace internal + template + struct traits> + : traits> + { + }; +}// end namespace internal template -class MappedSparseMatrix - : public Map > +class MappedSparseMatrix : public Map> { - typedef Map > Base; + typedef Map> Base; - public: - - typedef typename Base::StorageIndex StorageIndex; - typedef typename Base::Scalar Scalar; +public: + typedef typename Base::StorageIndex StorageIndex; + typedef typename Base::Scalar Scalar; - inline MappedSparseMatrix(Index rows, Index cols, Index nnz, StorageIndex* outerIndexPtr, StorageIndex* innerIndexPtr, Scalar* valuePtr, StorageIndex* innerNonZeroPtr = 0) - : Base(rows, cols, nnz, outerIndexPtr, innerIndexPtr, valuePtr, innerNonZeroPtr) - {} + inline MappedSparseMatrix(Index rows, + Index cols, + Index nnz, + StorageIndex *outerIndexPtr, + StorageIndex *innerIndexPtr, + Scalar *valuePtr, + StorageIndex *innerNonZeroPtr = 0) + : Base(rows, cols, nnz, outerIndexPtr, innerIndexPtr, valuePtr, innerNonZeroPtr) + {} - /** Empty destructor */ - inline ~MappedSparseMatrix() {} + /** Empty destructor */ + inline ~MappedSparseMatrix() {} }; namespace internal { -template -struct evaluator > - : evaluator > > -{ - typedef MappedSparseMatrix<_Scalar,_Options,_StorageIndex> XprType; - typedef evaluator > Base; - - evaluator() : Base() {} - explicit evaluator(const XprType &mat) : Base(mat) {} -}; + template + struct evaluator> + : evaluator>> + { + typedef MappedSparseMatrix<_Scalar, _Options, _StorageIndex> XprType; + typedef evaluator> Base; + + evaluator() : Base() {} + explicit evaluator(const XprType &mat) : Base(mat) {} + }; -} +}// namespace internal -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_MAPPED_SPARSEMATRIX_H +#endif// EIGEN_MAPPED_SPARSEMATRIX_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/SparseCore/SparseAssign.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/SparseCore/SparseAssign.h index 18352a84..d9566d35 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/SparseCore/SparseAssign.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/SparseCore/SparseAssign.h @@ -10,11 +10,11 @@ #ifndef EIGEN_SPARSEASSIGN_H #define EIGEN_SPARSEASSIGN_H -namespace Eigen { +namespace Eigen { -template +template template -Derived& SparseMatrixBase::operator=(const EigenBase &other) +Derived &SparseMatrixBase::operator=(const EigenBase &other) { internal::call_assignment_no_alias(derived(), other.derived()); return derived(); @@ -22,7 +22,7 @@ Derived& SparseMatrixBase::operator=(const EigenBase &oth template template -Derived& SparseMatrixBase::operator=(const ReturnByValue& other) +Derived &SparseMatrixBase::operator=(const ReturnByValue &other) { // TODO use the evaluator mechanism other.evalTo(derived()); @@ -31,16 +31,15 @@ Derived& SparseMatrixBase::operator=(const ReturnByValue& template template -inline Derived& SparseMatrixBase::operator=(const SparseMatrixBase& other) +inline Derived &SparseMatrixBase::operator=(const SparseMatrixBase &other) { // by default sparse evaluation do not alias, so we can safely bypass the generic call_assignment routine - internal::Assignment > - ::run(derived(), other.derived(), internal::assign_op()); + internal::Assignment>::run( + derived(), other.derived(), internal::assign_op()); return derived(); } -template -inline Derived& SparseMatrixBase::operator=(const Derived& other) +template inline Derived &SparseMatrixBase::operator=(const Derived &other) { internal::call_assignment_no_alias(derived(), other.derived()); return derived(); @@ -48,169 +47,198 @@ inline Derived& SparseMatrixBase::operator=(const Derived& other) namespace internal { -template<> -struct storage_kind_to_evaluator_kind { - typedef IteratorBased Kind; -}; + template<> struct storage_kind_to_evaluator_kind + { + typedef IteratorBased Kind; + }; -template<> -struct storage_kind_to_shape { - typedef SparseShape Shape; -}; + template<> struct storage_kind_to_shape + { + typedef SparseShape Shape; + }; -struct Sparse2Sparse {}; -struct Sparse2Dense {}; + struct Sparse2Sparse + { + }; + struct Sparse2Dense + { + }; -template<> struct AssignmentKind { typedef Sparse2Sparse Kind; }; -template<> struct AssignmentKind { typedef Sparse2Sparse Kind; }; -template<> struct AssignmentKind { typedef Sparse2Dense Kind; }; -template<> struct AssignmentKind { typedef Sparse2Dense Kind; }; + template<> struct AssignmentKind + { + typedef Sparse2Sparse Kind; + }; + template<> struct AssignmentKind + { + typedef Sparse2Sparse Kind; + }; + template<> struct AssignmentKind + { + typedef Sparse2Dense Kind; + }; + template<> struct AssignmentKind + { + typedef Sparse2Dense Kind; + }; -template -void assign_sparse_to_sparse(DstXprType &dst, const SrcXprType &src) -{ - typedef typename DstXprType::Scalar Scalar; - typedef internal::evaluator DstEvaluatorType; - typedef internal::evaluator SrcEvaluatorType; + template + void assign_sparse_to_sparse(DstXprType &dst, const SrcXprType &src) + { + typedef typename DstXprType::Scalar Scalar; + typedef internal::evaluator DstEvaluatorType; + typedef internal::evaluator SrcEvaluatorType; - SrcEvaluatorType srcEvaluator(src); + SrcEvaluatorType srcEvaluator(src); - const bool transpose = (DstEvaluatorType::Flags & RowMajorBit) != (SrcEvaluatorType::Flags & RowMajorBit); - const Index outerEvaluationSize = (SrcEvaluatorType::Flags&RowMajorBit) ? src.rows() : src.cols(); - if ((!transpose) && src.isRValue()) - { - // eval without temporary - dst.resize(src.rows(), src.cols()); - dst.setZero(); - dst.reserve((std::max)(src.rows(),src.cols())*2); - for (Index j=0; j::SupportedAccessPatterns & OuterRandomAccessPattern) + == OuterRandomAccessPattern) + || (!((DstEvaluatorType::Flags & RowMajorBit) != (SrcEvaluatorType::Flags & RowMajorBit)))) + && "the transpose operation is supposed to be handled in SparseMatrix::operator="); + + enum { Flip = (DstEvaluatorType::Flags & RowMajorBit) != (SrcEvaluatorType::Flags & RowMajorBit) }; + + + DstXprType temp(src.rows(), src.cols()); + + temp.reserve((std::max)(src.rows(), src.cols()) * 2); + for (Index j = 0; j < outerEvaluationSize; ++j) { + temp.startVec(j); + for (typename SrcEvaluatorType::InnerIterator it(srcEvaluator, j); it; ++it) { + Scalar v = it.value(); + temp.insertBackByOuterInner(Flip ? it.index() : j, Flip ? j : it.index()) = v; + } + } + temp.finalize(); + + dst = temp.markAsRValue(); } - dst.finalize(); } - else - { - // eval through a temporary - eigen_assert(( ((internal::traits::SupportedAccessPatterns & OuterRandomAccessPattern)==OuterRandomAccessPattern) || - (!((DstEvaluatorType::Flags & RowMajorBit) != (SrcEvaluatorType::Flags & RowMajorBit)))) && - "the transpose operation is supposed to be handled in SparseMatrix::operator="); - - enum { Flip = (DstEvaluatorType::Flags & RowMajorBit) != (SrcEvaluatorType::Flags & RowMajorBit) }; - - DstXprType temp(src.rows(), src.cols()); + // Generic Sparse to Sparse assignment + template + struct Assignment + { + static void run(DstXprType &dst, + const SrcXprType &src, + const internal::assign_op & /*func*/) + { + assign_sparse_to_sparse(dst.derived(), src.derived()); + } + }; - temp.reserve((std::max)(src.rows(),src.cols())*2); - for (Index j=0; j + struct Assignment + { + static void run(DstXprType &dst, const SrcXprType &src, const Functor &func) { - temp.startVec(j); - for (typename SrcEvaluatorType::InnerIterator it(srcEvaluator, j); it; ++it) - { - Scalar v = it.value(); - temp.insertBackByOuterInner(Flip?it.index():j,Flip?j:it.index()) = v; - } + if (internal::is_same>::value) + dst.setZero(); + + internal::evaluator srcEval(src); + resize_if_allowed(dst, src, func); + internal::evaluator dstEval(dst); + + const Index outerEvaluationSize = + (internal::evaluator::Flags & RowMajorBit) ? src.rows() : src.cols(); + for (Index j = 0; j < outerEvaluationSize; ++j) + for (typename internal::evaluator::InnerIterator i(srcEval, j); i; ++i) + func.assignCoeff(dstEval.coeffRef(i.row(), i.col()), i.value()); } - temp.finalize(); + }; - dst = temp.markAsRValue(); - } -} + // Specialization for "dst = dec.solve(rhs)" + // NOTE we need to specialize it for Sparse2Sparse to avoid ambiguous specialization error + template + struct Assignment, internal::assign_op, Sparse2Sparse> + { + typedef Solve SrcXprType; + static void run(DstXprType &dst, const SrcXprType &src, const internal::assign_op &) + { + Index dstRows = src.rows(); + Index dstCols = src.cols(); + if ((dst.rows() != dstRows) || (dst.cols() != dstCols)) dst.resize(dstRows, dstCols); -// Generic Sparse to Sparse assignment -template< typename DstXprType, typename SrcXprType, typename Functor> -struct Assignment -{ - static void run(DstXprType &dst, const SrcXprType &src, const internal::assign_op &/*func*/) + src.dec()._solve_impl(src.rhs(), dst); + } + }; + + struct Diagonal2Sparse { - assign_sparse_to_sparse(dst.derived(), src.derived()); - } -}; + }; -// Generic Sparse to Dense assignment -template< typename DstXprType, typename SrcXprType, typename Functor> -struct Assignment -{ - static void run(DstXprType &dst, const SrcXprType &src, const Functor &func) + template<> struct AssignmentKind { - if(internal::is_same >::value) - dst.setZero(); - - internal::evaluator srcEval(src); - resize_if_allowed(dst, src, func); - internal::evaluator dstEval(dst); - - const Index outerEvaluationSize = (internal::evaluator::Flags&RowMajorBit) ? src.rows() : src.cols(); - for (Index j=0; j::InnerIterator i(srcEval,j); i; ++i) - func.assignCoeff(dstEval.coeffRef(i.row(),i.col()), i.value()); - } -}; + typedef Diagonal2Sparse Kind; + }; -// Specialization for "dst = dec.solve(rhs)" -// NOTE we need to specialize it for Sparse2Sparse to avoid ambiguous specialization error -template -struct Assignment, internal::assign_op, Sparse2Sparse> -{ - typedef Solve SrcXprType; - static void run(DstXprType &dst, const SrcXprType &src, const internal::assign_op &) + template + struct Assignment { - Index dstRows = src.rows(); - Index dstCols = src.cols(); - if((dst.rows()!=dstRows) || (dst.cols()!=dstCols)) - dst.resize(dstRows, dstCols); + typedef typename DstXprType::StorageIndex StorageIndex; + typedef typename DstXprType::Scalar Scalar; + typedef Array ArrayXI; + typedef Array ArrayXS; + template + static void run(SparseMatrix &dst, + const SrcXprType &src, + const internal::assign_op & /*func*/) + { + Index dstRows = src.rows(); + Index dstCols = src.cols(); + if ((dst.rows() != dstRows) || (dst.cols() != dstCols)) dst.resize(dstRows, dstCols); + + Index size = src.diagonal().size(); + dst.makeCompressed(); + dst.resizeNonZeros(size); + Map(dst.innerIndexPtr(), size).setLinSpaced(0, StorageIndex(size) - 1); + Map(dst.outerIndexPtr(), size + 1).setLinSpaced(0, StorageIndex(size)); + Map(dst.valuePtr(), size) = src.diagonal(); + } - src.dec()._solve_impl(src.rhs(), dst); - } -}; + template + static void run(SparseMatrixBase &dst, + const SrcXprType &src, + const internal::assign_op & /*func*/) + { + dst.diagonal() = src.diagonal(); + } + + static void run(DstXprType &dst, + const SrcXprType &src, + const internal::add_assign_op & /*func*/) + { + dst.diagonal() += src.diagonal(); + } -struct Diagonal2Sparse {}; + static void run(DstXprType &dst, + const SrcXprType &src, + const internal::sub_assign_op & /*func*/) + { + dst.diagonal() -= src.diagonal(); + } + }; +}// end namespace internal -template<> struct AssignmentKind { typedef Diagonal2Sparse Kind; }; +}// end namespace Eigen -template< typename DstXprType, typename SrcXprType, typename Functor> -struct Assignment -{ - typedef typename DstXprType::StorageIndex StorageIndex; - typedef typename DstXprType::Scalar Scalar; - typedef Array ArrayXI; - typedef Array ArrayXS; - template - static void run(SparseMatrix &dst, const SrcXprType &src, const internal::assign_op &/*func*/) - { - Index dstRows = src.rows(); - Index dstCols = src.cols(); - if((dst.rows()!=dstRows) || (dst.cols()!=dstCols)) - dst.resize(dstRows, dstCols); - - Index size = src.diagonal().size(); - dst.makeCompressed(); - dst.resizeNonZeros(size); - Map(dst.innerIndexPtr(), size).setLinSpaced(0,StorageIndex(size)-1); - Map(dst.outerIndexPtr(), size+1).setLinSpaced(0,StorageIndex(size)); - Map(dst.valuePtr(), size) = src.diagonal(); - } - - template - static void run(SparseMatrixBase &dst, const SrcXprType &src, const internal::assign_op &/*func*/) - { - dst.diagonal() = src.diagonal(); - } - - static void run(DstXprType &dst, const SrcXprType &src, const internal::add_assign_op &/*func*/) - { dst.diagonal() += src.diagonal(); } - - static void run(DstXprType &dst, const SrcXprType &src, const internal::sub_assign_op &/*func*/) - { dst.diagonal() -= src.diagonal(); } -}; -} // end namespace internal - -} // end namespace Eigen - -#endif // EIGEN_SPARSEASSIGN_H +#endif// EIGEN_SPARSEASSIGN_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/SparseCore/SparseBlock.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/SparseCore/SparseBlock.h index 511e92b2..995bbed1 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/SparseCore/SparseBlock.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/SparseCore/SparseBlock.h @@ -14,196 +14,193 @@ namespace Eigen { // Subset of columns or rows template -class BlockImpl - : public SparseMatrixBase > +class BlockImpl + : public SparseMatrixBase> { - typedef typename internal::remove_all::type _MatrixTypeNested; - typedef Block BlockType; + typedef typename internal::remove_all::type _MatrixTypeNested; + typedef Block BlockType; + public: - enum { IsRowMajor = internal::traits::IsRowMajor }; + enum { IsRowMajor = internal::traits::IsRowMajor }; + protected: - enum { OuterSize = IsRowMajor ? BlockRows : BlockCols }; - typedef SparseMatrixBase Base; - using Base::convert_index; -public: - EIGEN_SPARSE_PUBLIC_INTERFACE(BlockType) + enum { OuterSize = IsRowMajor ? BlockRows : BlockCols }; + typedef SparseMatrixBase Base; + using Base::convert_index; - inline BlockImpl(XprType& xpr, Index i) - : m_matrix(xpr), m_outerStart(convert_index(i)), m_outerSize(OuterSize) - {} +public: + EIGEN_SPARSE_PUBLIC_INTERFACE(BlockType) - inline BlockImpl(XprType& xpr, Index startRow, Index startCol, Index blockRows, Index blockCols) - : m_matrix(xpr), m_outerStart(convert_index(IsRowMajor ? startRow : startCol)), m_outerSize(convert_index(IsRowMajor ? blockRows : blockCols)) - {} + inline BlockImpl(XprType &xpr, Index i) : m_matrix(xpr), m_outerStart(convert_index(i)), m_outerSize(OuterSize) {} - EIGEN_STRONG_INLINE Index rows() const { return IsRowMajor ? m_outerSize.value() : m_matrix.rows(); } - EIGEN_STRONG_INLINE Index cols() const { return IsRowMajor ? m_matrix.cols() : m_outerSize.value(); } + inline BlockImpl(XprType &xpr, Index startRow, Index startCol, Index blockRows, Index blockCols) + : m_matrix(xpr), m_outerStart(convert_index(IsRowMajor ? startRow : startCol)), + m_outerSize(convert_index(IsRowMajor ? blockRows : blockCols)) + {} - Index nonZeros() const - { - typedef internal::evaluator EvaluatorType; - EvaluatorType matEval(m_matrix); - Index nnz = 0; - Index end = m_outerStart + m_outerSize.value(); - for(Index j=m_outerStart; j EvaluatorType; + EvaluatorType matEval(m_matrix); + Index nnz = 0; + Index end = m_outerStart + m_outerSize.value(); + for (Index j = m_outerStart; j < end; ++j) + for (typename EvaluatorType::InnerIterator it(matEval, j); it; ++it) ++nnz; + return nnz; + } - inline const Scalar coeff(Index index) const - { - return m_matrix.coeff(IsRowMajor ? m_outerStart : index, IsRowMajor ? index : m_outerStart); - } + inline const Scalar coeff(Index row, Index col) const + { + return m_matrix.coeff(row + (IsRowMajor ? m_outerStart : 0), col + (IsRowMajor ? 0 : m_outerStart)); + } - inline const XprType& nestedExpression() const { return m_matrix; } - inline XprType& nestedExpression() { return m_matrix; } - Index startRow() const { return IsRowMajor ? m_outerStart : 0; } - Index startCol() const { return IsRowMajor ? 0 : m_outerStart; } - Index blockRows() const { return IsRowMajor ? m_outerSize.value() : m_matrix.rows(); } - Index blockCols() const { return IsRowMajor ? m_matrix.cols() : m_outerSize.value(); } + inline const Scalar coeff(Index index) const + { + return m_matrix.coeff(IsRowMajor ? m_outerStart : index, IsRowMajor ? index : m_outerStart); + } - protected: + inline const XprType &nestedExpression() const { return m_matrix; } + inline XprType &nestedExpression() { return m_matrix; } + Index startRow() const { return IsRowMajor ? m_outerStart : 0; } + Index startCol() const { return IsRowMajor ? 0 : m_outerStart; } + Index blockRows() const { return IsRowMajor ? m_outerSize.value() : m_matrix.rows(); } + Index blockCols() const { return IsRowMajor ? m_matrix.cols() : m_outerSize.value(); } - typename internal::ref_selector::non_const_type m_matrix; - Index m_outerStart; - const internal::variable_if_dynamic m_outerSize; +protected: + typename internal::ref_selector::non_const_type m_matrix; + Index m_outerStart; + const internal::variable_if_dynamic m_outerSize; - protected: - // Disable assignment with clear error message. - // Note that simply removing operator= yields compilation errors with ICC+MSVC - template - BlockImpl& operator=(const T&) - { - EIGEN_STATIC_ASSERT(sizeof(T)==0, THIS_SPARSE_BLOCK_SUBEXPRESSION_IS_READ_ONLY); - return *this; - } +protected: + // Disable assignment with clear error message. + // Note that simply removing operator= yields compilation errors with ICC+MSVC + template BlockImpl &operator=(const T &) + { + EIGEN_STATIC_ASSERT(sizeof(T) == 0, THIS_SPARSE_BLOCK_SUBEXPRESSION_IS_READ_ONLY); + return *this; + } }; /*************************************************************************** -* specialization for SparseMatrix -***************************************************************************/ + * specialization for SparseMatrix + ***************************************************************************/ namespace internal { -template -class sparse_matrix_block_impl - : public SparseCompressedBase > -{ + template + class sparse_matrix_block_impl : public SparseCompressedBase> + { typedef typename internal::remove_all::type _MatrixTypeNested; typedef Block BlockType; - typedef SparseCompressedBase > Base; + typedef SparseCompressedBase> Base; using Base::convert_index; -public: + + public: enum { IsRowMajor = internal::traits::IsRowMajor }; EIGEN_SPARSE_PUBLIC_INTERFACE(BlockType) -protected: + protected: typedef typename Base::IndexVector IndexVector; enum { OuterSize = IsRowMajor ? BlockRows : BlockCols }; -public: - inline sparse_matrix_block_impl(SparseMatrixType& xpr, Index i) + public: + inline sparse_matrix_block_impl(SparseMatrixType &xpr, Index i) : m_matrix(xpr), m_outerStart(convert_index(i)), m_outerSize(OuterSize) {} - inline sparse_matrix_block_impl(SparseMatrixType& xpr, Index startRow, Index startCol, Index blockRows, Index blockCols) - : m_matrix(xpr), m_outerStart(convert_index(IsRowMajor ? startRow : startCol)), m_outerSize(convert_index(IsRowMajor ? blockRows : blockCols)) + inline sparse_matrix_block_impl(SparseMatrixType &xpr, + Index startRow, + Index startCol, + Index blockRows, + Index blockCols) + : m_matrix(xpr), m_outerStart(convert_index(IsRowMajor ? startRow : startCol)), + m_outerSize(convert_index(IsRowMajor ? blockRows : blockCols)) {} - template - inline BlockType& operator=(const SparseMatrixBase& other) + template inline BlockType &operator=(const SparseMatrixBase &other) { typedef typename internal::remove_all::type _NestedMatrixType; - _NestedMatrixType& matrix = m_matrix; + _NestedMatrixType &matrix = m_matrix; // This assignment is slow if this vector set is not empty // and/or it is not at the end of the nonzeros of the underlying matrix. // 1 - eval to a temporary to avoid transposition and/or aliasing issues - Ref > tmp(other.derived()); - eigen_internal_assert(tmp.outerSize()==m_outerSize.value()); + Ref> tmp(other.derived()); + eigen_internal_assert(tmp.outerSize() == m_outerSize.value()); // 2 - let's check whether there is enough allocated memory - Index nnz = tmp.nonZeros(); - Index start = m_outerStart==0 ? 0 : m_matrix.outerIndexPtr()[m_outerStart]; // starting position of the current block - Index end = m_matrix.outerIndexPtr()[m_outerStart+m_outerSize.value()]; // ending position of the current block - Index block_size = end - start; // available room in the current block - Index tail_size = m_matrix.outerIndexPtr()[m_matrix.outerSize()] - end; + Index nnz = tmp.nonZeros(); + Index start = + m_outerStart == 0 ? 0 : m_matrix.outerIndexPtr()[m_outerStart];// starting position of the current block + Index end = m_matrix.outerIndexPtr()[m_outerStart + m_outerSize.value()];// ending position of the current block + Index block_size = end - start;// available room in the current block + Index tail_size = m_matrix.outerIndexPtr()[m_matrix.outerSize()] - end; - Index free_size = m_matrix.isCompressed() - ? Index(matrix.data().allocatedSize()) + block_size - : block_size; + Index free_size = m_matrix.isCompressed() ? Index(matrix.data().allocatedSize()) + block_size : block_size; Index tmp_start = tmp.outerIndexPtr()[0]; bool update_trailing_pointers = false; - if(nnz>free_size) - { + if (nnz > free_size) { // realloc manually to reduce copies typename SparseMatrixType::Storage newdata(m_matrix.data().allocatedSize() - block_size + nnz); - internal::smart_copy(m_matrix.valuePtr(), m_matrix.valuePtr() + start, newdata.valuePtr()); - internal::smart_copy(m_matrix.innerIndexPtr(), m_matrix.innerIndexPtr() + start, newdata.indexPtr()); + internal::smart_copy(m_matrix.valuePtr(), m_matrix.valuePtr() + start, newdata.valuePtr()); + internal::smart_copy(m_matrix.innerIndexPtr(), m_matrix.innerIndexPtr() + start, newdata.indexPtr()); - internal::smart_copy(tmp.valuePtr() + tmp_start, tmp.valuePtr() + tmp_start + nnz, newdata.valuePtr() + start); - internal::smart_copy(tmp.innerIndexPtr() + tmp_start, tmp.innerIndexPtr() + tmp_start + nnz, newdata.indexPtr() + start); + internal::smart_copy(tmp.valuePtr() + tmp_start, tmp.valuePtr() + tmp_start + nnz, newdata.valuePtr() + start); + internal::smart_copy( + tmp.innerIndexPtr() + tmp_start, tmp.innerIndexPtr() + tmp_start + nnz, newdata.indexPtr() + start); - internal::smart_copy(matrix.valuePtr()+end, matrix.valuePtr()+end + tail_size, newdata.valuePtr()+start+nnz); - internal::smart_copy(matrix.innerIndexPtr()+end, matrix.innerIndexPtr()+end + tail_size, newdata.indexPtr()+start+nnz); + internal::smart_copy( + matrix.valuePtr() + end, matrix.valuePtr() + end + tail_size, newdata.valuePtr() + start + nnz); + internal::smart_copy( + matrix.innerIndexPtr() + end, matrix.innerIndexPtr() + end + tail_size, newdata.indexPtr() + start + nnz); newdata.resize(m_matrix.outerIndexPtr()[m_matrix.outerSize()] - block_size + nnz); matrix.data().swap(newdata); update_trailing_pointers = true; - } - else - { - if(m_matrix.isCompressed()) - { + } else { + if (m_matrix.isCompressed()) { // no need to realloc, simply copy the tail at its respective position and insert tmp matrix.data().resize(start + nnz + tail_size); - internal::smart_memmove(matrix.valuePtr()+end, matrix.valuePtr() + end+tail_size, matrix.valuePtr() + start+nnz); - internal::smart_memmove(matrix.innerIndexPtr()+end, matrix.innerIndexPtr() + end+tail_size, matrix.innerIndexPtr() + start+nnz); + internal::smart_memmove( + matrix.valuePtr() + end, matrix.valuePtr() + end + tail_size, matrix.valuePtr() + start + nnz); + internal::smart_memmove(matrix.innerIndexPtr() + end, + matrix.innerIndexPtr() + end + tail_size, + matrix.innerIndexPtr() + start + nnz); update_trailing_pointers = true; } - internal::smart_copy(tmp.valuePtr() + tmp_start, tmp.valuePtr() + tmp_start + nnz, matrix.valuePtr() + start); - internal::smart_copy(tmp.innerIndexPtr() + tmp_start, tmp.innerIndexPtr() + tmp_start + nnz, matrix.innerIndexPtr() + start); + internal::smart_copy(tmp.valuePtr() + tmp_start, tmp.valuePtr() + tmp_start + nnz, matrix.valuePtr() + start); + internal::smart_copy( + tmp.innerIndexPtr() + tmp_start, tmp.innerIndexPtr() + tmp_start + nnz, matrix.innerIndexPtr() + start); } // update outer index pointers and innerNonZeros - if(IsVectorAtCompileTime) - { - if(!m_matrix.isCompressed()) - matrix.innerNonZeroPtr()[m_outerStart] = StorageIndex(nnz); + if (IsVectorAtCompileTime) { + if (!m_matrix.isCompressed()) matrix.innerNonZeroPtr()[m_outerStart] = StorageIndex(nnz); matrix.outerIndexPtr()[m_outerStart] = StorageIndex(start); - } - else - { + } else { StorageIndex p = StorageIndex(start); - for(Index k=0; k(tmp.innerVector(k).nonZeros()); - if(!m_matrix.isCompressed()) - matrix.innerNonZeroPtr()[m_outerStart+k] = nnz_k; - matrix.outerIndexPtr()[m_outerStart+k] = p; + if (!m_matrix.isCompressed()) matrix.innerNonZeroPtr()[m_outerStart + k] = nnz_k; + matrix.outerIndexPtr()[m_outerStart + k] = p; p += nnz_k; } } - if(update_trailing_pointers) - { + if (update_trailing_pointers) { StorageIndex offset = internal::convert_index(nnz - block_size); - for(Index k = m_outerStart + m_outerSize.value(); k<=matrix.outerSize(); ++k) - { + for (Index k = m_outerStart + m_outerSize.value(); k <= matrix.outerSize(); ++k) { matrix.outerIndexPtr()[k] += offset; } } @@ -211,91 +208,80 @@ class sparse_matrix_block_impl return derived(); } - inline BlockType& operator=(const BlockType& other) - { - return operator=(other); - } + inline BlockType &operator=(const BlockType &other) { return operator= (other); } - inline const Scalar* valuePtr() const - { return m_matrix.valuePtr(); } - inline Scalar* valuePtr() - { return m_matrix.valuePtr(); } + inline const Scalar *valuePtr() const { return m_matrix.valuePtr(); } + inline Scalar *valuePtr() { return m_matrix.valuePtr(); } - inline const StorageIndex* innerIndexPtr() const - { return m_matrix.innerIndexPtr(); } - inline StorageIndex* innerIndexPtr() - { return m_matrix.innerIndexPtr(); } + inline const StorageIndex *innerIndexPtr() const { return m_matrix.innerIndexPtr(); } + inline StorageIndex *innerIndexPtr() { return m_matrix.innerIndexPtr(); } - inline const StorageIndex* outerIndexPtr() const - { return m_matrix.outerIndexPtr() + m_outerStart; } - inline StorageIndex* outerIndexPtr() - { return m_matrix.outerIndexPtr() + m_outerStart; } + inline const StorageIndex *outerIndexPtr() const { return m_matrix.outerIndexPtr() + m_outerStart; } + inline StorageIndex *outerIndexPtr() { return m_matrix.outerIndexPtr() + m_outerStart; } - inline const StorageIndex* innerNonZeroPtr() const - { return isCompressed() ? 0 : (m_matrix.innerNonZeroPtr()+m_outerStart); } - inline StorageIndex* innerNonZeroPtr() - { return isCompressed() ? 0 : (m_matrix.innerNonZeroPtr()+m_outerStart); } + inline const StorageIndex *innerNonZeroPtr() const + { + return isCompressed() ? 0 : (m_matrix.innerNonZeroPtr() + m_outerStart); + } + inline StorageIndex *innerNonZeroPtr() { return isCompressed() ? 0 : (m_matrix.innerNonZeroPtr() + m_outerStart); } - bool isCompressed() const { return m_matrix.innerNonZeroPtr()==0; } + bool isCompressed() const { return m_matrix.innerNonZeroPtr() == 0; } - inline Scalar& coeffRef(Index row, Index col) + inline Scalar &coeffRef(Index row, Index col) { - return m_matrix.coeffRef(row + (IsRowMajor ? m_outerStart : 0), col + (IsRowMajor ? 0 : m_outerStart)); + return m_matrix.coeffRef(row + (IsRowMajor ? m_outerStart : 0), col + (IsRowMajor ? 0 : m_outerStart)); } inline const Scalar coeff(Index row, Index col) const { - return m_matrix.coeff(row + (IsRowMajor ? m_outerStart : 0), col + (IsRowMajor ? 0 : m_outerStart)); + return m_matrix.coeff(row + (IsRowMajor ? m_outerStart : 0), col + (IsRowMajor ? 0 : m_outerStart)); } inline const Scalar coeff(Index index) const { - return m_matrix.coeff(IsRowMajor ? m_outerStart : index, IsRowMajor ? index : m_outerStart); + return m_matrix.coeff(IsRowMajor ? m_outerStart : index, IsRowMajor ? index : m_outerStart); } - const Scalar& lastCoeff() const + const Scalar &lastCoeff() const { EIGEN_STATIC_ASSERT_VECTOR_ONLY(sparse_matrix_block_impl); - eigen_assert(Base::nonZeros()>0); - if(m_matrix.isCompressed()) - return m_matrix.valuePtr()[m_matrix.outerIndexPtr()[m_outerStart+1]-1]; + eigen_assert(Base::nonZeros() > 0); + if (m_matrix.isCompressed()) + return m_matrix.valuePtr()[m_matrix.outerIndexPtr()[m_outerStart + 1] - 1]; else - return m_matrix.valuePtr()[m_matrix.outerIndexPtr()[m_outerStart]+m_matrix.innerNonZeroPtr()[m_outerStart]-1]; + return m_matrix + .valuePtr()[m_matrix.outerIndexPtr()[m_outerStart] + m_matrix.innerNonZeroPtr()[m_outerStart] - 1]; } EIGEN_STRONG_INLINE Index rows() const { return IsRowMajor ? m_outerSize.value() : m_matrix.rows(); } EIGEN_STRONG_INLINE Index cols() const { return IsRowMajor ? m_matrix.cols() : m_outerSize.value(); } - inline const SparseMatrixType& nestedExpression() const { return m_matrix; } - inline SparseMatrixType& nestedExpression() { return m_matrix; } + inline const SparseMatrixType &nestedExpression() const { return m_matrix; } + inline SparseMatrixType &nestedExpression() { return m_matrix; } Index startRow() const { return IsRowMajor ? m_outerStart : 0; } Index startCol() const { return IsRowMajor ? 0 : m_outerStart; } Index blockRows() const { return IsRowMajor ? m_outerSize.value() : m_matrix.rows(); } Index blockCols() const { return IsRowMajor ? m_matrix.cols() : m_outerSize.value(); } protected: - typename internal::ref_selector::non_const_type m_matrix; Index m_outerStart; const internal::variable_if_dynamic m_outerSize; + }; -}; - -} // namespace internal +}// namespace internal template -class BlockImpl,BlockRows,BlockCols,true,Sparse> - : public internal::sparse_matrix_block_impl,BlockRows,BlockCols> +class BlockImpl, BlockRows, BlockCols, true, Sparse> + : public internal::sparse_matrix_block_impl, BlockRows, BlockCols> { public: typedef _StorageIndex StorageIndex; typedef SparseMatrix<_Scalar, _Options, _StorageIndex> SparseMatrixType; - typedef internal::sparse_matrix_block_impl Base; - inline BlockImpl(SparseMatrixType& xpr, Index i) - : Base(xpr, i) - {} + typedef internal::sparse_matrix_block_impl Base; + inline BlockImpl(SparseMatrixType &xpr, Index i) : Base(xpr, i) {} - inline BlockImpl(SparseMatrixType& xpr, Index startRow, Index startCol, Index blockRows, Index blockCols) + inline BlockImpl(SparseMatrixType &xpr, Index startRow, Index startCol, Index blockRows, Index blockCols) : Base(xpr, startRow, startCol, blockRows, blockCols) {} @@ -303,192 +289,199 @@ class BlockImpl,BlockRows,BlockCo }; template -class BlockImpl,BlockRows,BlockCols,true,Sparse> - : public internal::sparse_matrix_block_impl,BlockRows,BlockCols> +class BlockImpl, BlockRows, BlockCols, true, Sparse> + : public internal:: + sparse_matrix_block_impl, BlockRows, BlockCols> { public: typedef _StorageIndex StorageIndex; typedef const SparseMatrix<_Scalar, _Options, _StorageIndex> SparseMatrixType; - typedef internal::sparse_matrix_block_impl Base; - inline BlockImpl(SparseMatrixType& xpr, Index i) - : Base(xpr, i) - {} + typedef internal::sparse_matrix_block_impl Base; + inline BlockImpl(SparseMatrixType &xpr, Index i) : Base(xpr, i) {} - inline BlockImpl(SparseMatrixType& xpr, Index startRow, Index startCol, Index blockRows, Index blockCols) + inline BlockImpl(SparseMatrixType &xpr, Index startRow, Index startCol, Index blockRows, Index blockCols) : Base(xpr, startRow, startCol, blockRows, blockCols) {} using Base::operator=; + private: - template BlockImpl(const SparseMatrixBase& xpr, Index i); - template BlockImpl(const SparseMatrixBase& xpr); + template BlockImpl(const SparseMatrixBase &xpr, Index i); + template BlockImpl(const SparseMatrixBase &xpr); }; //---------- /** \returns the \a outer -th column (resp. row) of the matrix \c *this if \c *this - * is col-major (resp. row-major). - */ + * is col-major (resp. row-major). + */ template typename SparseMatrixBase::InnerVectorReturnType SparseMatrixBase::innerVector(Index outer) -{ return InnerVectorReturnType(derived(), outer); } +{ + return InnerVectorReturnType(derived(), outer); +} /** \returns the \a outer -th column (resp. row) of the matrix \c *this if \c *this - * is col-major (resp. row-major). Read-only. - */ + * is col-major (resp. row-major). Read-only. + */ template -const typename SparseMatrixBase::ConstInnerVectorReturnType SparseMatrixBase::innerVector(Index outer) const -{ return ConstInnerVectorReturnType(derived(), outer); } +const typename SparseMatrixBase::ConstInnerVectorReturnType SparseMatrixBase::innerVector( + Index outer) const +{ + return ConstInnerVectorReturnType(derived(), outer); +} /** \returns the \a outer -th column (resp. row) of the matrix \c *this if \c *this - * is col-major (resp. row-major). - */ + * is col-major (resp. row-major). + */ template -typename SparseMatrixBase::InnerVectorsReturnType -SparseMatrixBase::innerVectors(Index outerStart, Index outerSize) +typename SparseMatrixBase::InnerVectorsReturnType SparseMatrixBase::innerVectors(Index outerStart, + Index outerSize) { - return Block(derived(), - IsRowMajor ? outerStart : 0, IsRowMajor ? 0 : outerStart, - IsRowMajor ? outerSize : rows(), IsRowMajor ? cols() : outerSize); - + return Block(derived(), + IsRowMajor ? outerStart : 0, + IsRowMajor ? 0 : outerStart, + IsRowMajor ? outerSize : rows(), + IsRowMajor ? cols() : outerSize); } /** \returns the \a outer -th column (resp. row) of the matrix \c *this if \c *this - * is col-major (resp. row-major). Read-only. - */ + * is col-major (resp. row-major). Read-only. + */ template const typename SparseMatrixBase::ConstInnerVectorsReturnType -SparseMatrixBase::innerVectors(Index outerStart, Index outerSize) const + SparseMatrixBase::innerVectors(Index outerStart, Index outerSize) const { - return Block(derived(), - IsRowMajor ? outerStart : 0, IsRowMajor ? 0 : outerStart, - IsRowMajor ? outerSize : rows(), IsRowMajor ? cols() : outerSize); - + return Block(derived(), + IsRowMajor ? outerStart : 0, + IsRowMajor ? 0 : outerStart, + IsRowMajor ? outerSize : rows(), + IsRowMajor ? cols() : outerSize); } /** Generic implementation of sparse Block expression. - * Real-only. - */ + * Real-only. + */ template -class BlockImpl - : public SparseMatrixBase >, internal::no_assignment_operator +class BlockImpl + : public SparseMatrixBase> + , internal::no_assignment_operator { - typedef Block BlockType; - typedef SparseMatrixBase Base; - using Base::convert_index; -public: - enum { IsRowMajor = internal::traits::IsRowMajor }; - EIGEN_SPARSE_PUBLIC_INTERFACE(BlockType) + typedef Block BlockType; + typedef SparseMatrixBase Base; + using Base::convert_index; - typedef typename internal::remove_all::type _MatrixTypeNested; +public: + enum { IsRowMajor = internal::traits::IsRowMajor }; + EIGEN_SPARSE_PUBLIC_INTERFACE(BlockType) - /** Column or Row constructor - */ - inline BlockImpl(XprType& xpr, Index i) - : m_matrix(xpr), - m_startRow( (BlockRows==1) && (BlockCols==XprType::ColsAtCompileTime) ? convert_index(i) : 0), - m_startCol( (BlockRows==XprType::RowsAtCompileTime) && (BlockCols==1) ? convert_index(i) : 0), - m_blockRows(BlockRows==1 ? 1 : xpr.rows()), - m_blockCols(BlockCols==1 ? 1 : xpr.cols()) - {} + typedef typename internal::remove_all::type _MatrixTypeNested; - /** Dynamic-size constructor - */ - inline BlockImpl(XprType& xpr, Index startRow, Index startCol, Index blockRows, Index blockCols) - : m_matrix(xpr), m_startRow(convert_index(startRow)), m_startCol(convert_index(startCol)), m_blockRows(convert_index(blockRows)), m_blockCols(convert_index(blockCols)) - {} + /** Column or Row constructor + */ + inline BlockImpl(XprType &xpr, Index i) + : m_matrix(xpr), m_startRow((BlockRows == 1) && (BlockCols == XprType::ColsAtCompileTime) ? convert_index(i) : 0), + m_startCol((BlockRows == XprType::RowsAtCompileTime) && (BlockCols == 1) ? convert_index(i) : 0), + m_blockRows(BlockRows == 1 ? 1 : xpr.rows()), m_blockCols(BlockCols == 1 ? 1 : xpr.cols()) + {} - inline Index rows() const { return m_blockRows.value(); } - inline Index cols() const { return m_blockCols.value(); } + /** Dynamic-size constructor + */ + inline BlockImpl(XprType &xpr, Index startRow, Index startCol, Index blockRows, Index blockCols) + : m_matrix(xpr), m_startRow(convert_index(startRow)), m_startCol(convert_index(startCol)), + m_blockRows(convert_index(blockRows)), m_blockCols(convert_index(blockCols)) + {} - inline Scalar& coeffRef(Index row, Index col) - { - return m_matrix.coeffRef(row + m_startRow.value(), col + m_startCol.value()); - } + inline Index rows() const { return m_blockRows.value(); } + inline Index cols() const { return m_blockCols.value(); } - inline const Scalar coeff(Index row, Index col) const - { - return m_matrix.coeff(row + m_startRow.value(), col + m_startCol.value()); - } + inline Scalar &coeffRef(Index row, Index col) + { + return m_matrix.coeffRef(row + m_startRow.value(), col + m_startCol.value()); + } - inline Scalar& coeffRef(Index index) - { - return m_matrix.coeffRef(m_startRow.value() + (RowsAtCompileTime == 1 ? 0 : index), - m_startCol.value() + (RowsAtCompileTime == 1 ? index : 0)); - } + inline const Scalar coeff(Index row, Index col) const + { + return m_matrix.coeff(row + m_startRow.value(), col + m_startCol.value()); + } - inline const Scalar coeff(Index index) const - { - return m_matrix.coeff(m_startRow.value() + (RowsAtCompileTime == 1 ? 0 : index), - m_startCol.value() + (RowsAtCompileTime == 1 ? index : 0)); - } + inline Scalar &coeffRef(Index index) + { + return m_matrix.coeffRef(m_startRow.value() + (RowsAtCompileTime == 1 ? 0 : index), + m_startCol.value() + (RowsAtCompileTime == 1 ? index : 0)); + } - inline const XprType& nestedExpression() const { return m_matrix; } - inline XprType& nestedExpression() { return m_matrix; } - Index startRow() const { return m_startRow.value(); } - Index startCol() const { return m_startCol.value(); } - Index blockRows() const { return m_blockRows.value(); } - Index blockCols() const { return m_blockCols.value(); } + inline const Scalar coeff(Index index) const + { + return m_matrix.coeff(m_startRow.value() + (RowsAtCompileTime == 1 ? 0 : index), + m_startCol.value() + (RowsAtCompileTime == 1 ? index : 0)); + } - protected: -// friend class internal::GenericSparseBlockInnerIteratorImpl; - friend struct internal::unary_evaluator, internal::IteratorBased, Scalar >; + inline const XprType &nestedExpression() const { return m_matrix; } + inline XprType &nestedExpression() { return m_matrix; } + Index startRow() const { return m_startRow.value(); } + Index startCol() const { return m_startCol.value(); } + Index blockRows() const { return m_blockRows.value(); } + Index blockCols() const { return m_blockCols.value(); } - Index nonZeros() const { return Dynamic; } +protected: + // friend class internal::GenericSparseBlockInnerIteratorImpl; + friend struct internal:: + unary_evaluator, internal::IteratorBased, Scalar>; - typename internal::ref_selector::non_const_type m_matrix; - const internal::variable_if_dynamic m_startRow; - const internal::variable_if_dynamic m_startCol; - const internal::variable_if_dynamic m_blockRows; - const internal::variable_if_dynamic m_blockCols; + Index nonZeros() const { return Dynamic; } - protected: - // Disable assignment with clear error message. - // Note that simply removing operator= yields compilation errors with ICC+MSVC - template - BlockImpl& operator=(const T&) - { - EIGEN_STATIC_ASSERT(sizeof(T)==0, THIS_SPARSE_BLOCK_SUBEXPRESSION_IS_READ_ONLY); - return *this; - } + typename internal::ref_selector::non_const_type m_matrix; + const internal::variable_if_dynamic m_startRow; + const internal::variable_if_dynamic m_startCol; + const internal::variable_if_dynamic m_blockRows; + const internal::variable_if_dynamic m_blockCols; +protected: + // Disable assignment with clear error message. + // Note that simply removing operator= yields compilation errors with ICC+MSVC + template BlockImpl &operator=(const T &) + { + EIGEN_STATIC_ASSERT(sizeof(T) == 0, THIS_SPARSE_BLOCK_SUBEXPRESSION_IS_READ_ONLY); + return *this; + } }; namespace internal { -template -struct unary_evaluator, IteratorBased > - : public evaluator_base > -{ + template + struct unary_evaluator, IteratorBased> + : public evaluator_base> + { class InnerVectorInnerIterator; class OuterVectorInnerIterator; + public: - typedef Block XprType; + typedef Block XprType; typedef typename XprType::StorageIndex StorageIndex; typedef typename XprType::Scalar Scalar; enum { IsRowMajor = XprType::IsRowMajor, - OuterVector = (BlockCols==1 && ArgType::IsRowMajor) - | // FIXME | instead of || to please GCC 4.4.0 stupid warning "suggest parentheses around &&". - // revert to || as soon as not needed anymore. - (BlockRows==1 && !ArgType::IsRowMajor), + OuterVector = (BlockCols == 1 && ArgType::IsRowMajor) + |// FIXME | instead of || to please GCC 4.4.0 stupid warning "suggest parentheses around &&". + // revert to || as soon as not needed anymore. + (BlockRows == 1 && !ArgType::IsRowMajor), CoeffReadCost = evaluator::CoeffReadCost, Flags = XprType::Flags }; - typedef typename internal::conditional::type InnerIterator; + typedef typename internal::conditional::type + InnerIterator; - explicit unary_evaluator(const XprType& op) - : m_argImpl(op.nestedExpression()), m_block(op) - {} + explicit unary_evaluator(const XprType &op) : m_argImpl(op.nestedExpression()), m_block(op) {} - inline Index nonZerosEstimate() const { + inline Index nonZerosEstimate() const + { Index nnz = m_block.nonZeros(); - if(nnz<0) - return m_argImpl.nonZerosEstimate() * m_block.size() / m_block.nestedExpression().size(); + if (nnz < 0) return m_argImpl.nonZerosEstimate() * m_block.size() / m_block.nestedExpression().size(); return nnz; } @@ -497,107 +490,119 @@ struct unary_evaluator, IteratorBa evaluator m_argImpl; const XprType &m_block; -}; + }; -template -class unary_evaluator, IteratorBased>::InnerVectorInnerIterator - : public EvalIterator -{ - enum { IsRowMajor = unary_evaluator::IsRowMajor }; - const XprType& m_block; - Index m_end; -public: - - EIGEN_STRONG_INLINE InnerVectorInnerIterator(const unary_evaluator& aEval, Index outer) - : EvalIterator(aEval.m_argImpl, outer + (IsRowMajor ? aEval.m_block.startRow() : aEval.m_block.startCol())), - m_block(aEval.m_block), - m_end(IsRowMajor ? aEval.m_block.startCol()+aEval.m_block.blockCols() : aEval.m_block.startRow()+aEval.m_block.blockRows()) + template + class unary_evaluator, IteratorBased>::InnerVectorInnerIterator + : public EvalIterator { - while( (EvalIterator::operator bool()) && (EvalIterator::index() < (IsRowMajor ? m_block.startCol() : m_block.startRow())) ) - EvalIterator::operator++(); - } + enum { IsRowMajor = unary_evaluator::IsRowMajor }; + const XprType &m_block; + Index m_end; - inline StorageIndex index() const { return EvalIterator::index() - convert_index(IsRowMajor ? m_block.startCol() : m_block.startRow()); } - inline Index outer() const { return EvalIterator::outer() - (IsRowMajor ? m_block.startRow() : m_block.startCol()); } - inline Index row() const { return EvalIterator::row() - m_block.startRow(); } - inline Index col() const { return EvalIterator::col() - m_block.startCol(); } + public: + EIGEN_STRONG_INLINE InnerVectorInnerIterator(const unary_evaluator &aEval, Index outer) + : EvalIterator(aEval.m_argImpl, outer + (IsRowMajor ? aEval.m_block.startRow() : aEval.m_block.startCol())), + m_block(aEval.m_block), m_end(IsRowMajor ? aEval.m_block.startCol() + aEval.m_block.blockCols() + : aEval.m_block.startRow() + aEval.m_block.blockRows()) + { + while ((EvalIterator::operator bool()) + && (EvalIterator::index() < (IsRowMajor ? m_block.startCol() : m_block.startRow()))) + EvalIterator::operator++(); + } - inline operator bool() const { return EvalIterator::operator bool() && EvalIterator::index() < m_end; } -}; + inline StorageIndex index() const + { + return EvalIterator::index() - convert_index(IsRowMajor ? m_block.startCol() : m_block.startRow()); + } + inline Index outer() const + { + return EvalIterator::outer() - (IsRowMajor ? m_block.startRow() : m_block.startCol()); + } + inline Index row() const { return EvalIterator::row() - m_block.startRow(); } + inline Index col() const { return EvalIterator::col() - m_block.startCol(); } -template -class unary_evaluator, IteratorBased>::OuterVectorInnerIterator -{ - enum { IsRowMajor = unary_evaluator::IsRowMajor }; - const unary_evaluator& m_eval; - Index m_outerPos; - const Index m_innerIndex; - Index m_end; - EvalIterator m_it; -public: + inline operator bool() const { return EvalIterator::operator bool() && EvalIterator::index() < m_end; } + }; - EIGEN_STRONG_INLINE OuterVectorInnerIterator(const unary_evaluator& aEval, Index outer) - : m_eval(aEval), - m_outerPos( (IsRowMajor ? aEval.m_block.startCol() : aEval.m_block.startRow()) ), - m_innerIndex(IsRowMajor ? aEval.m_block.startRow() : aEval.m_block.startCol()), - m_end(IsRowMajor ? aEval.m_block.startCol()+aEval.m_block.blockCols() : aEval.m_block.startRow()+aEval.m_block.blockRows()), - m_it(m_eval.m_argImpl, m_outerPos) + template + class unary_evaluator, IteratorBased>::OuterVectorInnerIterator { - EIGEN_UNUSED_VARIABLE(outer); - eigen_assert(outer==0); - - while(m_it && m_it.index() < m_innerIndex) ++m_it; - if((!m_it) || (m_it.index()!=m_innerIndex)) - ++(*this); - } + enum { IsRowMajor = unary_evaluator::IsRowMajor }; + const unary_evaluator &m_eval; + Index m_outerPos; + const Index m_innerIndex; + Index m_end; + EvalIterator m_it; - inline StorageIndex index() const { return convert_index(m_outerPos - (IsRowMajor ? m_eval.m_block.startCol() : m_eval.m_block.startRow())); } - inline Index outer() const { return 0; } - inline Index row() const { return IsRowMajor ? 0 : index(); } - inline Index col() const { return IsRowMajor ? index() : 0; } + public: + EIGEN_STRONG_INLINE OuterVectorInnerIterator(const unary_evaluator &aEval, Index outer) + : m_eval(aEval), m_outerPos((IsRowMajor ? aEval.m_block.startCol() : aEval.m_block.startRow())), + m_innerIndex(IsRowMajor ? aEval.m_block.startRow() : aEval.m_block.startCol()), + m_end(IsRowMajor ? aEval.m_block.startCol() + aEval.m_block.blockCols() + : aEval.m_block.startRow() + aEval.m_block.blockRows()), + m_it(m_eval.m_argImpl, m_outerPos) + { + EIGEN_UNUSED_VARIABLE(outer); + eigen_assert(outer == 0); - inline Scalar value() const { return m_it.value(); } - inline Scalar& valueRef() { return m_it.valueRef(); } + while (m_it && m_it.index() < m_innerIndex) ++m_it; + if ((!m_it) || (m_it.index() != m_innerIndex)) ++(*this); + } - inline OuterVectorInnerIterator& operator++() - { - // search next non-zero entry - while(++m_outerPos( + m_outerPos - (IsRowMajor ? m_eval.m_block.startCol() : m_eval.m_block.startRow())); } - return *this; - } + inline Index outer() const { return 0; } + inline Index row() const { return IsRowMajor ? 0 : index(); } + inline Index col() const { return IsRowMajor ? index() : 0; } - inline operator bool() const { return m_outerPos < m_end; } -}; + inline Scalar value() const { return m_it.value(); } + inline Scalar &valueRef() { return m_it.valueRef(); } -template -struct unary_evaluator,BlockRows,BlockCols,true>, IteratorBased> - : evaluator,BlockRows,BlockCols,true> > > -{ - typedef Block,BlockRows,BlockCols,true> XprType; - typedef evaluator > Base; - explicit unary_evaluator(const XprType &xpr) : Base(xpr) {} -}; + inline OuterVectorInnerIterator &operator++() + { + // search next non-zero entry + while (++m_outerPos < m_end) { + // Restart iterator at the next inner-vector: + m_it.~EvalIterator(); + ::new (&m_it) EvalIterator(m_eval.m_argImpl, m_outerPos); + // search for the key m_innerIndex in the current outer-vector + while (m_it && m_it.index() < m_innerIndex) ++m_it; + if (m_it && m_it.index() == m_innerIndex) break; + } + return *this; + } -template -struct unary_evaluator,BlockRows,BlockCols,true>, IteratorBased> - : evaluator,BlockRows,BlockCols,true> > > -{ - typedef Block,BlockRows,BlockCols,true> XprType; - typedef evaluator > Base; - explicit unary_evaluator(const XprType &xpr) : Base(xpr) {} -}; + inline operator bool() const { return m_outerPos < m_end; } + }; + + template + struct unary_evaluator, BlockRows, BlockCols, true>, + IteratorBased> + : evaluator, BlockRows, BlockCols, true>>> + { + typedef Block, BlockRows, BlockCols, true> XprType; + typedef evaluator> Base; + explicit unary_evaluator(const XprType &xpr) : Base(xpr) {} + }; + + template + struct unary_evaluator, BlockRows, BlockCols, true>, + IteratorBased> + : evaluator< + SparseCompressedBase, BlockRows, BlockCols, true>>> + { + typedef Block, BlockRows, BlockCols, true> XprType; + typedef evaluator> Base; + explicit unary_evaluator(const XprType &xpr) : Base(xpr) {} + }; -} // end namespace internal +}// end namespace internal -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_SPARSE_BLOCK_H +#endif// EIGEN_SPARSE_BLOCK_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/SparseCore/SparseColEtree.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/SparseCore/SparseColEtree.h index ebe02d1a..566630d2 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/SparseCore/SparseColEtree.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/SparseCore/SparseColEtree.h @@ -8,10 +8,10 @@ // with this file, You can obtain one at http://mozilla.org/MPL/2.0/. -/* - - * NOTE: This file is the modified version of sp_coletree.c file in SuperLU - +/* + + * NOTE: This file is the modified version of sp_coletree.c file in SuperLU + * -- SuperLU routine (version 3.1) -- * Univ. of California Berkeley, Xerox Palo Alto Research Center, * and Lawrence Berkeley National Lab. @@ -35,172 +35,168 @@ namespace Eigen { namespace internal { -/** Find the root of the tree/set containing the vertex i : Use Path halving */ -template -Index etree_find (Index i, IndexVector& pp) -{ - Index p = pp(i); // Parent - Index gp = pp(p); // Grand parent - while (gp != p) + /** Find the root of the tree/set containing the vertex i : Use Path halving */ + template Index etree_find(Index i, IndexVector &pp) { - pp(i) = gp; // Parent pointer on find path is changed to former grand parent - i = gp; - p = pp(i); - gp = pp(p); - } - return p; -} - -/** Compute the column elimination tree of a sparse matrix - * \param mat The matrix in column-major format. - * \param parent The elimination tree - * \param firstRowElt The column index of the first element in each row - * \param perm The permutation to apply to the column of \b mat - */ -template -int coletree(const MatrixType& mat, IndexVector& parent, IndexVector& firstRowElt, typename MatrixType::StorageIndex *perm=0) -{ - typedef typename MatrixType::StorageIndex StorageIndex; - StorageIndex nc = convert_index(mat.cols()); // Number of columns - StorageIndex m = convert_index(mat.rows()); - StorageIndex diagSize = (std::min)(nc,m); - IndexVector root(nc); // root of subtree of etree - root.setZero(); - IndexVector pp(nc); // disjoint sets - pp.setZero(); // Initialize disjoint sets - parent.resize(mat.cols()); - //Compute first nonzero column in each row - firstRowElt.resize(m); - firstRowElt.setConstant(nc); - firstRowElt.segment(0, diagSize).setLinSpaced(diagSize, 0, diagSize-1); - bool found_diag; - for (StorageIndex col = 0; col < nc; col++) - { - StorageIndex pcol = col; - if(perm) pcol = perm[col]; - for (typename MatrixType::InnerIterator it(mat, pcol); it; ++it) - { - Index row = it.row(); - firstRowElt(row) = (std::min)(firstRowElt(row), col); + Index p = pp(i);// Parent + Index gp = pp(p);// Grand parent + while (gp != p) { + pp(i) = gp;// Parent pointer on find path is changed to former grand parent + i = gp; + p = pp(i); + gp = pp(p); } + return p; } - /* Compute etree by Liu's algorithm for symmetric matrices, - except use (firstRowElt[r],c) in place of an edge (r,c) of A. - Thus each row clique in A'*A is replaced by a star - centered at its first vertex, which has the same fill. */ - StorageIndex rset, cset, rroot; - for (StorageIndex col = 0; col < nc; col++) + + /** Compute the column elimination tree of a sparse matrix + * \param mat The matrix in column-major format. + * \param parent The elimination tree + * \param firstRowElt The column index of the first element in each row + * \param perm The permutation to apply to the column of \b mat + */ + template + int coletree(const MatrixType &mat, + IndexVector &parent, + IndexVector &firstRowElt, + typename MatrixType::StorageIndex *perm = 0) { - found_diag = col>=m; - pp(col) = col; - cset = col; - root(cset) = col; - parent(col) = nc; - /* The diagonal element is treated here even if it does not exist in the matrix - * hence the loop is executed once more */ - StorageIndex pcol = col; - if(perm) pcol = perm[col]; - for (typename MatrixType::InnerIterator it(mat, pcol); it||!found_diag; ++it) - { // A sequence of interleaved find and union is performed - Index i = col; - if(it) i = it.index(); - if (i == col) found_diag = true; - - StorageIndex row = firstRowElt(i); - if (row >= col) continue; - rset = internal::etree_find(row, pp); // Find the name of the set containing row - rroot = root(rset); - if (rroot != col) - { - parent(rroot) = col; - pp(cset) = rset; - cset = rset; - root(cset) = col; + typedef typename MatrixType::StorageIndex StorageIndex; + StorageIndex nc = convert_index(mat.cols());// Number of columns + StorageIndex m = convert_index(mat.rows()); + StorageIndex diagSize = (std::min)(nc, m); + IndexVector root(nc);// root of subtree of etree + root.setZero(); + IndexVector pp(nc);// disjoint sets + pp.setZero();// Initialize disjoint sets + parent.resize(mat.cols()); + // Compute first nonzero column in each row + firstRowElt.resize(m); + firstRowElt.setConstant(nc); + firstRowElt.segment(0, diagSize).setLinSpaced(diagSize, 0, diagSize - 1); + bool found_diag; + for (StorageIndex col = 0; col < nc; col++) { + StorageIndex pcol = col; + if (perm) pcol = perm[col]; + for (typename MatrixType::InnerIterator it(mat, pcol); it; ++it) { + Index row = it.row(); + firstRowElt(row) = (std::min)(firstRowElt(row), col); } } + /* Compute etree by Liu's algorithm for symmetric matrices, + except use (firstRowElt[r],c) in place of an edge (r,c) of A. + Thus each row clique in A'*A is replaced by a star + centered at its first vertex, which has the same fill. */ + StorageIndex rset, cset, rroot; + for (StorageIndex col = 0; col < nc; col++) { + found_diag = col >= m; + pp(col) = col; + cset = col; + root(cset) = col; + parent(col) = nc; + /* The diagonal element is treated here even if it does not exist in the matrix + * hence the loop is executed once more */ + StorageIndex pcol = col; + if (perm) pcol = perm[col]; + for (typename MatrixType::InnerIterator it(mat, pcol); it || !found_diag; + ++it) {// A sequence of interleaved find and union is performed + Index i = col; + if (it) i = it.index(); + if (i == col) found_diag = true; + + StorageIndex row = firstRowElt(i); + if (row >= col) continue; + rset = internal::etree_find(row, pp);// Find the name of the set containing row + rroot = root(rset); + if (rroot != col) { + parent(rroot) = col; + pp(cset) = rset; + cset = rset; + root(cset) = col; + } + } + } + return 0; } - return 0; -} - -/** - * Depth-first search from vertex n. No recursion. - * This routine was contributed by Cédric Doucet, CEDRAT Group, Meylan, France. -*/ -template -void nr_etdfs (typename IndexVector::Scalar n, IndexVector& parent, IndexVector& first_kid, IndexVector& next_kid, IndexVector& post, typename IndexVector::Scalar postnum) -{ - typedef typename IndexVector::Scalar StorageIndex; - StorageIndex current = n, first, next; - while (postnum != n) + + /** + * Depth-first search from vertex n. No recursion. + * This routine was contributed by Cédric Doucet, CEDRAT Group, Meylan, France. + */ + template + void nr_etdfs(typename IndexVector::Scalar n, + IndexVector &parent, + IndexVector &first_kid, + IndexVector &next_kid, + IndexVector &post, + typename IndexVector::Scalar postnum) { - // No kid for the current node - first = first_kid(current); - - // no kid for the current node - if (first == -1) - { - // Numbering this node because it has no kid - post(current) = postnum++; - - // looking for the next kid - next = next_kid(current); - while (next == -1) - { - // No more kids : back to the parent node - current = parent(current); - // numbering the parent node + typedef typename IndexVector::Scalar StorageIndex; + StorageIndex current = n, first, next; + while (postnum != n) { + // No kid for the current node + first = first_kid(current); + + // no kid for the current node + if (first == -1) { + // Numbering this node because it has no kid post(current) = postnum++; - - // Get the next kid - next = next_kid(current); + + // looking for the next kid + next = next_kid(current); + while (next == -1) { + // No more kids : back to the parent node + current = parent(current); + // numbering the parent node + post(current) = postnum++; + + // Get the next kid + next = next_kid(current); + } + // stopping criterion + if (postnum == n + 1) return; + + // Updating current node + current = next; + } else { + current = first; } - // stopping criterion - if (postnum == n+1) return; - - // Updating current node - current = next; - } - else - { - current = first; } } -} - - -/** - * \brief Post order a tree - * \param n the number of nodes - * \param parent Input tree - * \param post postordered tree - */ -template -void treePostorder(typename IndexVector::Scalar n, IndexVector& parent, IndexVector& post) -{ - typedef typename IndexVector::Scalar StorageIndex; - IndexVector first_kid, next_kid; // Linked list of children - StorageIndex postnum; - // Allocate storage for working arrays and results - first_kid.resize(n+1); - next_kid.setZero(n+1); - post.setZero(n+1); - - // Set up structure describing children - first_kid.setConstant(-1); - for (StorageIndex v = n-1; v >= 0; v--) + + + /** + * \brief Post order a tree + * \param n the number of nodes + * \param parent Input tree + * \param post postordered tree + */ + template + void treePostorder(typename IndexVector::Scalar n, IndexVector &parent, IndexVector &post) { - StorageIndex dad = parent(v); - next_kid(v) = first_kid(dad); - first_kid(dad) = v; + typedef typename IndexVector::Scalar StorageIndex; + IndexVector first_kid, next_kid;// Linked list of children + StorageIndex postnum; + // Allocate storage for working arrays and results + first_kid.resize(n + 1); + next_kid.setZero(n + 1); + post.setZero(n + 1); + + // Set up structure describing children + first_kid.setConstant(-1); + for (StorageIndex v = n - 1; v >= 0; v--) { + StorageIndex dad = parent(v); + next_kid(v) = first_kid(dad); + first_kid(dad) = v; + } + + // Depth-first search from dummy root vertex #n + postnum = 0; + internal::nr_etdfs(n, parent, first_kid, next_kid, post, postnum); } - - // Depth-first search from dummy root vertex #n - postnum = 0; - internal::nr_etdfs(n, parent, first_kid, next_kid, post, postnum); -} -} // end namespace internal +}// end namespace internal -} // end namespace Eigen +}// end namespace Eigen -#endif // SPARSE_COLETREE_H +#endif// SPARSE_COLETREE_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/SparseCore/SparseCompressedBase.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/SparseCore/SparseCompressedBase.h index 5ccb4665..54a4ec82 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/SparseCore/SparseCompressedBase.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/SparseCore/SparseCompressedBase.h @@ -10,332 +10,335 @@ #ifndef EIGEN_SPARSE_COMPRESSED_BASE_H #define EIGEN_SPARSE_COMPRESSED_BASE_H -namespace Eigen { +namespace Eigen { template class SparseCompressedBase; - + namespace internal { -template -struct traits > : traits -{}; + template struct traits> : traits + { + }; -} // end namespace internal +}// end namespace internal /** \ingroup SparseCore_Module - * \class SparseCompressedBase - * \brief Common base class for sparse [compressed]-{row|column}-storage format. - * - * This class defines the common interface for all derived classes implementing the compressed sparse storage format, such as: - * - SparseMatrix - * - Ref - * - Map - * - */ -template -class SparseCompressedBase - : public SparseMatrixBase + * \class SparseCompressedBase + * \brief Common base class for sparse [compressed]-{row|column}-storage format. + * + * This class defines the common interface for all derived classes implementing the compressed sparse storage format, + * such as: + * - SparseMatrix + * - Ref + * - Map + * + */ +template class SparseCompressedBase : public SparseMatrixBase { - public: - typedef SparseMatrixBase Base; - EIGEN_SPARSE_PUBLIC_INTERFACE(SparseCompressedBase) - using Base::operator=; - using Base::IsRowMajor; - - class InnerIterator; - class ReverseInnerIterator; - - protected: - typedef typename Base::IndexVector IndexVector; - Eigen::Map innerNonZeros() { return Eigen::Map(innerNonZeroPtr(), isCompressed()?0:derived().outerSize()); } - const Eigen::Map innerNonZeros() const { return Eigen::Map(innerNonZeroPtr(), isCompressed()?0:derived().outerSize()); } - - public: - - /** \returns the number of non zero coefficients */ - inline Index nonZeros() const - { - if(Derived::IsVectorAtCompileTime && outerIndexPtr()==0) - return derived().nonZeros(); - else if(isCompressed()) - return outerIndexPtr()[derived().outerSize()]-outerIndexPtr()[0]; - else if(derived().outerSize()==0) - return 0; - else - return innerNonZeros().sum(); - } - - /** \returns a const pointer to the array of values. - * This function is aimed at interoperability with other libraries. - * \sa innerIndexPtr(), outerIndexPtr() */ - inline const Scalar* valuePtr() const { return derived().valuePtr(); } - /** \returns a non-const pointer to the array of values. - * This function is aimed at interoperability with other libraries. - * \sa innerIndexPtr(), outerIndexPtr() */ - inline Scalar* valuePtr() { return derived().valuePtr(); } - - /** \returns a const pointer to the array of inner indices. - * This function is aimed at interoperability with other libraries. - * \sa valuePtr(), outerIndexPtr() */ - inline const StorageIndex* innerIndexPtr() const { return derived().innerIndexPtr(); } - /** \returns a non-const pointer to the array of inner indices. - * This function is aimed at interoperability with other libraries. - * \sa valuePtr(), outerIndexPtr() */ - inline StorageIndex* innerIndexPtr() { return derived().innerIndexPtr(); } - - /** \returns a const pointer to the array of the starting positions of the inner vectors. - * This function is aimed at interoperability with other libraries. - * \warning it returns the null pointer 0 for SparseVector - * \sa valuePtr(), innerIndexPtr() */ - inline const StorageIndex* outerIndexPtr() const { return derived().outerIndexPtr(); } - /** \returns a non-const pointer to the array of the starting positions of the inner vectors. - * This function is aimed at interoperability with other libraries. - * \warning it returns the null pointer 0 for SparseVector - * \sa valuePtr(), innerIndexPtr() */ - inline StorageIndex* outerIndexPtr() { return derived().outerIndexPtr(); } - - /** \returns a const pointer to the array of the number of non zeros of the inner vectors. - * This function is aimed at interoperability with other libraries. - * \warning it returns the null pointer 0 in compressed mode */ - inline const StorageIndex* innerNonZeroPtr() const { return derived().innerNonZeroPtr(); } - /** \returns a non-const pointer to the array of the number of non zeros of the inner vectors. - * This function is aimed at interoperability with other libraries. - * \warning it returns the null pointer 0 in compressed mode */ - inline StorageIndex* innerNonZeroPtr() { return derived().innerNonZeroPtr(); } - - /** \returns whether \c *this is in compressed form. */ - inline bool isCompressed() const { return innerNonZeroPtr()==0; } - - /** \returns a read-only view of the stored coefficients as a 1D array expression. - * - * \warning this method is for \b compressed \b storage \b only, and it will trigger an assertion otherwise. - * - * \sa valuePtr(), isCompressed() */ - const Map > coeffs() const { eigen_assert(isCompressed()); return Array::Map(valuePtr(),nonZeros()); } - - /** \returns a read-write view of the stored coefficients as a 1D array expression - * - * \warning this method is for \b compressed \b storage \b only, and it will trigger an assertion otherwise. - * - * Here is an example: - * \include SparseMatrix_coeffs.cpp - * and the output is: - * \include SparseMatrix_coeffs.out - * - * \sa valuePtr(), isCompressed() */ - Map > coeffs() { eigen_assert(isCompressed()); return Array::Map(valuePtr(),nonZeros()); } +public: + typedef SparseMatrixBase Base; + EIGEN_SPARSE_PUBLIC_INTERFACE(SparseCompressedBase) + using Base::operator=; + using Base::IsRowMajor; - protected: - /** Default constructor. Do nothing. */ - SparseCompressedBase() {} - private: - template explicit SparseCompressedBase(const SparseCompressedBase&); + class InnerIterator; + class ReverseInnerIterator; + +protected: + typedef typename Base::IndexVector IndexVector; + Eigen::Map innerNonZeros() + { + return Eigen::Map(innerNonZeroPtr(), isCompressed() ? 0 : derived().outerSize()); + } + const Eigen::Map innerNonZeros() const + { + return Eigen::Map(innerNonZeroPtr(), isCompressed() ? 0 : derived().outerSize()); + } + +public: + /** \returns the number of non zero coefficients */ + inline Index nonZeros() const + { + if (Derived::IsVectorAtCompileTime && outerIndexPtr() == 0) + return derived().nonZeros(); + else if (isCompressed()) + return outerIndexPtr()[derived().outerSize()] - outerIndexPtr()[0]; + else if (derived().outerSize() == 0) + return 0; + else + return innerNonZeros().sum(); + } + + /** \returns a const pointer to the array of values. + * This function is aimed at interoperability with other libraries. + * \sa innerIndexPtr(), outerIndexPtr() */ + inline const Scalar *valuePtr() const { return derived().valuePtr(); } + /** \returns a non-const pointer to the array of values. + * This function is aimed at interoperability with other libraries. + * \sa innerIndexPtr(), outerIndexPtr() */ + inline Scalar *valuePtr() { return derived().valuePtr(); } + + /** \returns a const pointer to the array of inner indices. + * This function is aimed at interoperability with other libraries. + * \sa valuePtr(), outerIndexPtr() */ + inline const StorageIndex *innerIndexPtr() const { return derived().innerIndexPtr(); } + /** \returns a non-const pointer to the array of inner indices. + * This function is aimed at interoperability with other libraries. + * \sa valuePtr(), outerIndexPtr() */ + inline StorageIndex *innerIndexPtr() { return derived().innerIndexPtr(); } + + /** \returns a const pointer to the array of the starting positions of the inner vectors. + * This function is aimed at interoperability with other libraries. + * \warning it returns the null pointer 0 for SparseVector + * \sa valuePtr(), innerIndexPtr() */ + inline const StorageIndex *outerIndexPtr() const { return derived().outerIndexPtr(); } + /** \returns a non-const pointer to the array of the starting positions of the inner vectors. + * This function is aimed at interoperability with other libraries. + * \warning it returns the null pointer 0 for SparseVector + * \sa valuePtr(), innerIndexPtr() */ + inline StorageIndex *outerIndexPtr() { return derived().outerIndexPtr(); } + + /** \returns a const pointer to the array of the number of non zeros of the inner vectors. + * This function is aimed at interoperability with other libraries. + * \warning it returns the null pointer 0 in compressed mode */ + inline const StorageIndex *innerNonZeroPtr() const { return derived().innerNonZeroPtr(); } + /** \returns a non-const pointer to the array of the number of non zeros of the inner vectors. + * This function is aimed at interoperability with other libraries. + * \warning it returns the null pointer 0 in compressed mode */ + inline StorageIndex *innerNonZeroPtr() { return derived().innerNonZeroPtr(); } + + /** \returns whether \c *this is in compressed form. */ + inline bool isCompressed() const { return innerNonZeroPtr() == 0; } + + /** \returns a read-only view of the stored coefficients as a 1D array expression. + * + * \warning this method is for \b compressed \b storage \b only, and it will trigger an assertion otherwise. + * + * \sa valuePtr(), isCompressed() */ + const Map> coeffs() const + { + eigen_assert(isCompressed()); + return Array::Map(valuePtr(), nonZeros()); + } + + /** \returns a read-write view of the stored coefficients as a 1D array expression + * + * \warning this method is for \b compressed \b storage \b only, and it will trigger an assertion otherwise. + * + * Here is an example: + * \include SparseMatrix_coeffs.cpp + * and the output is: + * \include SparseMatrix_coeffs.out + * + * \sa valuePtr(), isCompressed() */ + Map> coeffs() + { + eigen_assert(isCompressed()); + return Array::Map(valuePtr(), nonZeros()); + } + +protected: + /** Default constructor. Do nothing. */ + SparseCompressedBase() {} + +private: + template explicit SparseCompressedBase(const SparseCompressedBase &); }; -template -class SparseCompressedBase::InnerIterator +template class SparseCompressedBase::InnerIterator { - public: - InnerIterator() - : m_values(0), m_indices(0), m_outer(0), m_id(0), m_end(0) - {} +public: + InnerIterator() : m_values(0), m_indices(0), m_outer(0), m_id(0), m_end(0) {} - InnerIterator(const InnerIterator& other) - : m_values(other.m_values), m_indices(other.m_indices), m_outer(other.m_outer), m_id(other.m_id), m_end(other.m_end) - {} + InnerIterator(const InnerIterator &other) + : m_values(other.m_values), m_indices(other.m_indices), m_outer(other.m_outer), m_id(other.m_id), m_end(other.m_end) + {} - InnerIterator& operator=(const InnerIterator& other) - { - m_values = other.m_values; - m_indices = other.m_indices; - const_cast(m_outer).setValue(other.m_outer.value()); - m_id = other.m_id; - m_end = other.m_end; - return *this; - } + InnerIterator &operator=(const InnerIterator &other) + { + m_values = other.m_values; + m_indices = other.m_indices; + const_cast(m_outer).setValue(other.m_outer.value()); + m_id = other.m_id; + m_end = other.m_end; + return *this; + } - InnerIterator(const SparseCompressedBase& mat, Index outer) - : m_values(mat.valuePtr()), m_indices(mat.innerIndexPtr()), m_outer(outer) - { - if(Derived::IsVectorAtCompileTime && mat.outerIndexPtr()==0) - { - m_id = 0; - m_end = mat.nonZeros(); - } + InnerIterator(const SparseCompressedBase &mat, Index outer) + : m_values(mat.valuePtr()), m_indices(mat.innerIndexPtr()), m_outer(outer) + { + if (Derived::IsVectorAtCompileTime && mat.outerIndexPtr() == 0) { + m_id = 0; + m_end = mat.nonZeros(); + } else { + m_id = mat.outerIndexPtr()[outer]; + if (mat.isCompressed()) + m_end = mat.outerIndexPtr()[outer + 1]; else - { - m_id = mat.outerIndexPtr()[outer]; - if(mat.isCompressed()) - m_end = mat.outerIndexPtr()[outer+1]; - else - m_end = m_id + mat.innerNonZeroPtr()[outer]; - } + m_end = m_id + mat.innerNonZeroPtr()[outer]; } + } - explicit InnerIterator(const SparseCompressedBase& mat) - : m_values(mat.valuePtr()), m_indices(mat.innerIndexPtr()), m_outer(0), m_id(0), m_end(mat.nonZeros()) - { - EIGEN_STATIC_ASSERT_VECTOR_ONLY(Derived); - } + explicit InnerIterator(const SparseCompressedBase &mat) + : m_values(mat.valuePtr()), m_indices(mat.innerIndexPtr()), m_outer(0), m_id(0), m_end(mat.nonZeros()) + { + EIGEN_STATIC_ASSERT_VECTOR_ONLY(Derived); + } - explicit InnerIterator(const internal::CompressedStorage& data) - : m_values(data.valuePtr()), m_indices(data.indexPtr()), m_outer(0), m_id(0), m_end(data.size()) - { - EIGEN_STATIC_ASSERT_VECTOR_ONLY(Derived); - } + explicit InnerIterator(const internal::CompressedStorage &data) + : m_values(data.valuePtr()), m_indices(data.indexPtr()), m_outer(0), m_id(0), m_end(data.size()) + { + EIGEN_STATIC_ASSERT_VECTOR_ONLY(Derived); + } - inline InnerIterator& operator++() { m_id++; return *this; } + inline InnerIterator &operator++() + { + m_id++; + return *this; + } - inline const Scalar& value() const { return m_values[m_id]; } - inline Scalar& valueRef() { return const_cast(m_values[m_id]); } + inline const Scalar &value() const { return m_values[m_id]; } + inline Scalar &valueRef() { return const_cast(m_values[m_id]); } - inline StorageIndex index() const { return m_indices[m_id]; } - inline Index outer() const { return m_outer.value(); } - inline Index row() const { return IsRowMajor ? m_outer.value() : index(); } - inline Index col() const { return IsRowMajor ? index() : m_outer.value(); } + inline StorageIndex index() const { return m_indices[m_id]; } + inline Index outer() const { return m_outer.value(); } + inline Index row() const { return IsRowMajor ? m_outer.value() : index(); } + inline Index col() const { return IsRowMajor ? index() : m_outer.value(); } - inline operator bool() const { return (m_id < m_end); } + inline operator bool() const { return (m_id < m_end); } - protected: - const Scalar* m_values; - const StorageIndex* m_indices; - typedef internal::variable_if_dynamic OuterType; - const OuterType m_outer; - Index m_id; - Index m_end; - private: - // If you get here, then you're not using the right InnerIterator type, e.g.: - // SparseMatrix A; - // SparseMatrix::InnerIterator it(A,0); - template InnerIterator(const SparseMatrixBase&, Index outer); +protected: + const Scalar *m_values; + const StorageIndex *m_indices; + typedef internal::variable_if_dynamic OuterType; + const OuterType m_outer; + Index m_id; + Index m_end; + +private: + // If you get here, then you're not using the right InnerIterator type, e.g.: + // SparseMatrix A; + // SparseMatrix::InnerIterator it(A,0); + template InnerIterator(const SparseMatrixBase &, Index outer); }; -template -class SparseCompressedBase::ReverseInnerIterator +template class SparseCompressedBase::ReverseInnerIterator { - public: - ReverseInnerIterator(const SparseCompressedBase& mat, Index outer) - : m_values(mat.valuePtr()), m_indices(mat.innerIndexPtr()), m_outer(outer) - { - if(Derived::IsVectorAtCompileTime && mat.outerIndexPtr()==0) - { - m_start = 0; - m_id = mat.nonZeros(); - } +public: + ReverseInnerIterator(const SparseCompressedBase &mat, Index outer) + : m_values(mat.valuePtr()), m_indices(mat.innerIndexPtr()), m_outer(outer) + { + if (Derived::IsVectorAtCompileTime && mat.outerIndexPtr() == 0) { + m_start = 0; + m_id = mat.nonZeros(); + } else { + m_start = mat.outerIndexPtr()[outer]; + if (mat.isCompressed()) + m_id = mat.outerIndexPtr()[outer + 1]; else - { - m_start = mat.outerIndexPtr()[outer]; - if(mat.isCompressed()) - m_id = mat.outerIndexPtr()[outer+1]; - else - m_id = m_start + mat.innerNonZeroPtr()[outer]; - } + m_id = m_start + mat.innerNonZeroPtr()[outer]; } + } - explicit ReverseInnerIterator(const SparseCompressedBase& mat) - : m_values(mat.valuePtr()), m_indices(mat.innerIndexPtr()), m_outer(0), m_start(0), m_id(mat.nonZeros()) - { - EIGEN_STATIC_ASSERT_VECTOR_ONLY(Derived); - } + explicit ReverseInnerIterator(const SparseCompressedBase &mat) + : m_values(mat.valuePtr()), m_indices(mat.innerIndexPtr()), m_outer(0), m_start(0), m_id(mat.nonZeros()) + { + EIGEN_STATIC_ASSERT_VECTOR_ONLY(Derived); + } - explicit ReverseInnerIterator(const internal::CompressedStorage& data) - : m_values(data.valuePtr()), m_indices(data.indexPtr()), m_outer(0), m_start(0), m_id(data.size()) - { - EIGEN_STATIC_ASSERT_VECTOR_ONLY(Derived); - } + explicit ReverseInnerIterator(const internal::CompressedStorage &data) + : m_values(data.valuePtr()), m_indices(data.indexPtr()), m_outer(0), m_start(0), m_id(data.size()) + { + EIGEN_STATIC_ASSERT_VECTOR_ONLY(Derived); + } - inline ReverseInnerIterator& operator--() { --m_id; return *this; } + inline ReverseInnerIterator &operator--() + { + --m_id; + return *this; + } - inline const Scalar& value() const { return m_values[m_id-1]; } - inline Scalar& valueRef() { return const_cast(m_values[m_id-1]); } + inline const Scalar &value() const { return m_values[m_id - 1]; } + inline Scalar &valueRef() { return const_cast(m_values[m_id - 1]); } - inline StorageIndex index() const { return m_indices[m_id-1]; } - inline Index outer() const { return m_outer.value(); } - inline Index row() const { return IsRowMajor ? m_outer.value() : index(); } - inline Index col() const { return IsRowMajor ? index() : m_outer.value(); } + inline StorageIndex index() const { return m_indices[m_id - 1]; } + inline Index outer() const { return m_outer.value(); } + inline Index row() const { return IsRowMajor ? m_outer.value() : index(); } + inline Index col() const { return IsRowMajor ? index() : m_outer.value(); } - inline operator bool() const { return (m_id > m_start); } + inline operator bool() const { return (m_id > m_start); } - protected: - const Scalar* m_values; - const StorageIndex* m_indices; - typedef internal::variable_if_dynamic OuterType; - const OuterType m_outer; - Index m_start; - Index m_id; +protected: + const Scalar *m_values; + const StorageIndex *m_indices; + typedef internal::variable_if_dynamic OuterType; + const OuterType m_outer; + Index m_start; + Index m_id; }; namespace internal { -template -struct evaluator > - : evaluator_base -{ - typedef typename Derived::Scalar Scalar; - typedef typename Derived::InnerIterator InnerIterator; - - enum { - CoeffReadCost = NumTraits::ReadCost, - Flags = Derived::Flags - }; - - evaluator() : m_matrix(0), m_zero(0) - { - EIGEN_INTERNAL_CHECK_COST_VALUE(CoeffReadCost); - } - explicit evaluator(const Derived &mat) : m_matrix(&mat), m_zero(0) + template struct evaluator> : evaluator_base { - EIGEN_INTERNAL_CHECK_COST_VALUE(CoeffReadCost); - } - - inline Index nonZerosEstimate() const { - return m_matrix->nonZeros(); - } - - operator Derived&() { return m_matrix->const_cast_derived(); } - operator const Derived&() const { return *m_matrix; } - - typedef typename DenseCoeffsBase::CoeffReturnType CoeffReturnType; - const Scalar& coeff(Index row, Index col) const - { - Index p = find(row,col); + typedef typename Derived::Scalar Scalar; + typedef typename Derived::InnerIterator InnerIterator; - if(p==Dynamic) - return m_zero; - else - return m_matrix->const_cast_derived().valuePtr()[p]; - } + enum { CoeffReadCost = NumTraits::ReadCost, Flags = Derived::Flags }; - Scalar& coeffRef(Index row, Index col) - { - Index p = find(row,col); - eigen_assert(p!=Dynamic && "written coefficient does not exist"); - return m_matrix->const_cast_derived().valuePtr()[p]; - } + evaluator() : m_matrix(0), m_zero(0) { EIGEN_INTERNAL_CHECK_COST_VALUE(CoeffReadCost); } + explicit evaluator(const Derived &mat) : m_matrix(&mat), m_zero(0) + { + EIGEN_INTERNAL_CHECK_COST_VALUE(CoeffReadCost); + } -protected: + inline Index nonZerosEstimate() const { return m_matrix->nonZeros(); } - Index find(Index row, Index col) const - { - eigen_internal_assert(row>=0 && rowrows() && col>=0 && colcols()); + operator Derived &() { return m_matrix->const_cast_derived(); } + operator const Derived &() const { return *m_matrix; } - const Index outer = Derived::IsRowMajor ? row : col; - const Index inner = Derived::IsRowMajor ? col : row; + typedef typename DenseCoeffsBase::CoeffReturnType CoeffReturnType; + const Scalar &coeff(Index row, Index col) const + { + Index p = find(row, col); - Index start = m_matrix->outerIndexPtr()[outer]; - Index end = m_matrix->isCompressed() ? m_matrix->outerIndexPtr()[outer+1] : m_matrix->outerIndexPtr()[outer] + m_matrix->innerNonZeroPtr()[outer]; - eigen_assert(end>=start && "you are using a non finalized sparse matrix or written coefficient does not exist"); - const Index p = std::lower_bound(m_matrix->innerIndexPtr()+start, m_matrix->innerIndexPtr()+end,inner) - m_matrix->innerIndexPtr(); + if (p == Dynamic) + return m_zero; + else + return m_matrix->const_cast_derived().valuePtr()[p]; + } - return ((pinnerIndexPtr()[p]==inner)) ? p : Dynamic; - } + Scalar &coeffRef(Index row, Index col) + { + Index p = find(row, col); + eigen_assert(p != Dynamic && "written coefficient does not exist"); + return m_matrix->const_cast_derived().valuePtr()[p]; + } - const Derived *m_matrix; - const Scalar m_zero; -}; + protected: + Index find(Index row, Index col) const + { + eigen_internal_assert(row >= 0 && row < m_matrix->rows() && col >= 0 && col < m_matrix->cols()); + + const Index outer = Derived::IsRowMajor ? row : col; + const Index inner = Derived::IsRowMajor ? col : row; + + Index start = m_matrix->outerIndexPtr()[outer]; + Index end = m_matrix->isCompressed() ? m_matrix->outerIndexPtr()[outer + 1] + : m_matrix->outerIndexPtr()[outer] + m_matrix->innerNonZeroPtr()[outer]; + eigen_assert(end >= start && "you are using a non finalized sparse matrix or written coefficient does not exist"); + const Index p = std::lower_bound(m_matrix->innerIndexPtr() + start, m_matrix->innerIndexPtr() + end, inner) + - m_matrix->innerIndexPtr(); + + return ((p < end) && (m_matrix->innerIndexPtr()[p] == inner)) ? p : Dynamic; + } + + const Derived *m_matrix; + const Scalar m_zero; + }; -} +}// namespace internal -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_SPARSE_COMPRESSED_BASE_H +#endif// EIGEN_SPARSE_COMPRESSED_BASE_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/SparseCore/SparseCwiseBinaryOp.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/SparseCore/SparseCwiseBinaryOp.h index e315e355..6c774513 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/SparseCore/SparseCwiseBinaryOp.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/SparseCore/SparseCwiseBinaryOp.h @@ -10,7 +10,7 @@ #ifndef EIGEN_SPARSE_CWISE_BINARY_OP_H #define EIGEN_SPARSE_CWISE_BINARY_OP_H -namespace Eigen { +namespace Eigen { // Here we have to handle 3 cases: // 1 - sparse op dense @@ -33,694 +33,685 @@ namespace Eigen { // and fallback to cwise-unary evaluator using bind1st_op and bind2nd_op. template -class CwiseBinaryOpImpl - : public SparseMatrixBase > +class CwiseBinaryOpImpl : public SparseMatrixBase> { - public: - typedef CwiseBinaryOp Derived; - typedef SparseMatrixBase Base; - EIGEN_SPARSE_PUBLIC_INTERFACE(Derived) - CwiseBinaryOpImpl() - { - EIGEN_STATIC_ASSERT(( - (!internal::is_same::StorageKind, - typename internal::traits::StorageKind>::value) - || ((internal::evaluator::Flags&RowMajorBit) == (internal::evaluator::Flags&RowMajorBit))), - THE_STORAGE_ORDER_OF_BOTH_SIDES_MUST_MATCH); - } +public: + typedef CwiseBinaryOp Derived; + typedef SparseMatrixBase Base; + EIGEN_SPARSE_PUBLIC_INTERFACE(Derived) + CwiseBinaryOpImpl() + { + EIGEN_STATIC_ASSERT( + ((!internal::is_same::StorageKind, + typename internal::traits::StorageKind>::value) + || ((internal::evaluator::Flags & RowMajorBit) == (internal::evaluator::Flags & RowMajorBit))), + THE_STORAGE_ORDER_OF_BOTH_SIDES_MUST_MATCH); + } }; namespace internal { - -// Generic "sparse OP sparse" -template struct binary_sparse_evaluator; -template -struct binary_evaluator, IteratorBased, IteratorBased> - : evaluator_base > -{ -protected: - typedef typename evaluator::InnerIterator LhsIterator; - typedef typename evaluator::InnerIterator RhsIterator; - typedef CwiseBinaryOp XprType; - typedef typename traits::Scalar Scalar; - typedef typename XprType::StorageIndex StorageIndex; -public: + // Generic "sparse OP sparse" + template struct binary_sparse_evaluator; - class InnerIterator + template + struct binary_evaluator, IteratorBased, IteratorBased> + : evaluator_base> { - public: - - EIGEN_STRONG_INLINE InnerIterator(const binary_evaluator& aEval, Index outer) - : m_lhsIter(aEval.m_lhsImpl,outer), m_rhsIter(aEval.m_rhsImpl,outer), m_functor(aEval.m_functor) - { - this->operator++(); - } + protected: + typedef typename evaluator::InnerIterator LhsIterator; + typedef typename evaluator::InnerIterator RhsIterator; + typedef CwiseBinaryOp XprType; + typedef typename traits::Scalar Scalar; + typedef typename XprType::StorageIndex StorageIndex; - EIGEN_STRONG_INLINE InnerIterator& operator++() + public: + class InnerIterator { - if (m_lhsIter && m_rhsIter && (m_lhsIter.index() == m_rhsIter.index())) + public: + EIGEN_STRONG_INLINE InnerIterator(const binary_evaluator &aEval, Index outer) + : m_lhsIter(aEval.m_lhsImpl, outer), m_rhsIter(aEval.m_rhsImpl, outer), m_functor(aEval.m_functor) { - m_id = m_lhsIter.index(); - m_value = m_functor(m_lhsIter.value(), m_rhsIter.value()); - ++m_lhsIter; - ++m_rhsIter; + this->operator++(); } - else if (m_lhsIter && (!m_rhsIter || (m_lhsIter.index() < m_rhsIter.index()))) - { - m_id = m_lhsIter.index(); - m_value = m_functor(m_lhsIter.value(), Scalar(0)); - ++m_lhsIter; - } - else if (m_rhsIter && (!m_lhsIter || (m_lhsIter.index() > m_rhsIter.index()))) - { - m_id = m_rhsIter.index(); - m_value = m_functor(Scalar(0), m_rhsIter.value()); - ++m_rhsIter; - } - else + + EIGEN_STRONG_INLINE InnerIterator &operator++() { - m_value = 0; // this is to avoid a compilation warning - m_id = -1; + if (m_lhsIter && m_rhsIter && (m_lhsIter.index() == m_rhsIter.index())) { + m_id = m_lhsIter.index(); + m_value = m_functor(m_lhsIter.value(), m_rhsIter.value()); + ++m_lhsIter; + ++m_rhsIter; + } else if (m_lhsIter && (!m_rhsIter || (m_lhsIter.index() < m_rhsIter.index()))) { + m_id = m_lhsIter.index(); + m_value = m_functor(m_lhsIter.value(), Scalar(0)); + ++m_lhsIter; + } else if (m_rhsIter && (!m_lhsIter || (m_lhsIter.index() > m_rhsIter.index()))) { + m_id = m_rhsIter.index(); + m_value = m_functor(Scalar(0), m_rhsIter.value()); + ++m_rhsIter; + } else { + m_value = 0;// this is to avoid a compilation warning + m_id = -1; + } + return *this; } - return *this; - } - EIGEN_STRONG_INLINE Scalar value() const { return m_value; } + EIGEN_STRONG_INLINE Scalar value() const { return m_value; } - EIGEN_STRONG_INLINE StorageIndex index() const { return m_id; } - EIGEN_STRONG_INLINE Index outer() const { return m_lhsIter.outer(); } - EIGEN_STRONG_INLINE Index row() const { return Lhs::IsRowMajor ? m_lhsIter.row() : index(); } - EIGEN_STRONG_INLINE Index col() const { return Lhs::IsRowMajor ? index() : m_lhsIter.col(); } + EIGEN_STRONG_INLINE StorageIndex index() const { return m_id; } + EIGEN_STRONG_INLINE Index outer() const { return m_lhsIter.outer(); } + EIGEN_STRONG_INLINE Index row() const { return Lhs::IsRowMajor ? m_lhsIter.row() : index(); } + EIGEN_STRONG_INLINE Index col() const { return Lhs::IsRowMajor ? index() : m_lhsIter.col(); } - EIGEN_STRONG_INLINE operator bool() const { return m_id>=0; } + EIGEN_STRONG_INLINE operator bool() const { return m_id >= 0; } - protected: - LhsIterator m_lhsIter; - RhsIterator m_rhsIter; - const BinaryOp& m_functor; - Scalar m_value; - StorageIndex m_id; - }; - - - enum { - CoeffReadCost = evaluator::CoeffReadCost + evaluator::CoeffReadCost + functor_traits::Cost, - Flags = XprType::Flags - }; - - explicit binary_evaluator(const XprType& xpr) - : m_functor(xpr.functor()), - m_lhsImpl(xpr.lhs()), - m_rhsImpl(xpr.rhs()) - { - EIGEN_INTERNAL_CHECK_COST_VALUE(functor_traits::Cost); - EIGEN_INTERNAL_CHECK_COST_VALUE(CoeffReadCost); - } - - inline Index nonZerosEstimate() const { - return m_lhsImpl.nonZerosEstimate() + m_rhsImpl.nonZerosEstimate(); - } + protected: + LhsIterator m_lhsIter; + RhsIterator m_rhsIter; + const BinaryOp &m_functor; + Scalar m_value; + StorageIndex m_id; + }; -protected: - const BinaryOp m_functor; - evaluator m_lhsImpl; - evaluator m_rhsImpl; -}; -// dense op sparse -template -struct binary_evaluator, IndexBased, IteratorBased> - : evaluator_base > -{ -protected: - typedef typename evaluator::InnerIterator RhsIterator; - typedef CwiseBinaryOp XprType; - typedef typename traits::Scalar Scalar; - typedef typename XprType::StorageIndex StorageIndex; -public: + enum { + CoeffReadCost = evaluator::CoeffReadCost + evaluator::CoeffReadCost + functor_traits::Cost, + Flags = XprType::Flags + }; - class InnerIterator - { - enum { IsRowMajor = (int(Rhs::Flags)&RowMajorBit)==RowMajorBit }; - public: - - EIGEN_STRONG_INLINE InnerIterator(const binary_evaluator& aEval, Index outer) - : m_lhsEval(aEval.m_lhsImpl), m_rhsIter(aEval.m_rhsImpl,outer), m_functor(aEval.m_functor), m_value(0), m_id(-1), m_innerSize(aEval.m_expr.rhs().innerSize()) + explicit binary_evaluator(const XprType &xpr) : m_functor(xpr.functor()), m_lhsImpl(xpr.lhs()), m_rhsImpl(xpr.rhs()) { - this->operator++(); + EIGEN_INTERNAL_CHECK_COST_VALUE(functor_traits::Cost); + EIGEN_INTERNAL_CHECK_COST_VALUE(CoeffReadCost); } - EIGEN_STRONG_INLINE InnerIterator& operator++() + inline Index nonZerosEstimate() const { return m_lhsImpl.nonZerosEstimate() + m_rhsImpl.nonZerosEstimate(); } + + protected: + const BinaryOp m_functor; + evaluator m_lhsImpl; + evaluator m_rhsImpl; + }; + + // dense op sparse + template + struct binary_evaluator, IndexBased, IteratorBased> + : evaluator_base> + { + protected: + typedef typename evaluator::InnerIterator RhsIterator; + typedef CwiseBinaryOp XprType; + typedef typename traits::Scalar Scalar; + typedef typename XprType::StorageIndex StorageIndex; + + public: + class InnerIterator { - ++m_id; - if(m_idoperator++(); } - return *this; - } + EIGEN_STRONG_INLINE InnerIterator &operator++() + { + ++m_id; + if (m_id < m_innerSize) { + Scalar lhsVal = m_lhsEval.coeff(IsRowMajor ? m_rhsIter.outer() : m_id, IsRowMajor ? m_id : m_rhsIter.outer()); + if (m_rhsIter && m_rhsIter.index() == m_id) { + m_value = m_functor(lhsVal, m_rhsIter.value()); + ++m_rhsIter; + } else + m_value = m_functor(lhsVal, Scalar(0)); + } - EIGEN_STRONG_INLINE Scalar value() const { eigen_internal_assert(m_id &m_lhsEval; - RhsIterator m_rhsIter; - const BinaryOp& m_functor; - Scalar m_value; - StorageIndex m_id; - StorageIndex m_innerSize; - }; + EIGEN_STRONG_INLINE operator bool() const { return m_id < m_innerSize; } + protected: + const evaluator &m_lhsEval; + RhsIterator m_rhsIter; + const BinaryOp &m_functor; + Scalar m_value; + StorageIndex m_id; + StorageIndex m_innerSize; + }; - enum { - CoeffReadCost = evaluator::CoeffReadCost + evaluator::CoeffReadCost + functor_traits::Cost, - // Expose storage order of the sparse expression - Flags = (XprType::Flags & ~RowMajorBit) | (int(Rhs::Flags)&RowMajorBit) - }; - explicit binary_evaluator(const XprType& xpr) - : m_functor(xpr.functor()), - m_lhsImpl(xpr.lhs()), - m_rhsImpl(xpr.rhs()), - m_expr(xpr) - { - EIGEN_INTERNAL_CHECK_COST_VALUE(functor_traits::Cost); - EIGEN_INTERNAL_CHECK_COST_VALUE(CoeffReadCost); - } + enum { + CoeffReadCost = evaluator::CoeffReadCost + evaluator::CoeffReadCost + functor_traits::Cost, + // Expose storage order of the sparse expression + Flags = (XprType::Flags & ~RowMajorBit) | (int(Rhs::Flags) & RowMajorBit) + }; - inline Index nonZerosEstimate() const { - return m_expr.size(); - } + explicit binary_evaluator(const XprType &xpr) + : m_functor(xpr.functor()), m_lhsImpl(xpr.lhs()), m_rhsImpl(xpr.rhs()), m_expr(xpr) + { + EIGEN_INTERNAL_CHECK_COST_VALUE(functor_traits::Cost); + EIGEN_INTERNAL_CHECK_COST_VALUE(CoeffReadCost); + } -protected: - const BinaryOp m_functor; - evaluator m_lhsImpl; - evaluator m_rhsImpl; - const XprType &m_expr; -}; + inline Index nonZerosEstimate() const { return m_expr.size(); } -// sparse op dense -template -struct binary_evaluator, IteratorBased, IndexBased> - : evaluator_base > -{ -protected: - typedef typename evaluator::InnerIterator LhsIterator; - typedef CwiseBinaryOp XprType; - typedef typename traits::Scalar Scalar; - typedef typename XprType::StorageIndex StorageIndex; -public: + protected: + const BinaryOp m_functor; + evaluator m_lhsImpl; + evaluator m_rhsImpl; + const XprType &m_expr; + }; - class InnerIterator + // sparse op dense + template + struct binary_evaluator, IteratorBased, IndexBased> + : evaluator_base> { - enum { IsRowMajor = (int(Lhs::Flags)&RowMajorBit)==RowMajorBit }; - public: + protected: + typedef typename evaluator::InnerIterator LhsIterator; + typedef CwiseBinaryOp XprType; + typedef typename traits::Scalar Scalar; + typedef typename XprType::StorageIndex StorageIndex; - EIGEN_STRONG_INLINE InnerIterator(const binary_evaluator& aEval, Index outer) - : m_lhsIter(aEval.m_lhsImpl,outer), m_rhsEval(aEval.m_rhsImpl), m_functor(aEval.m_functor), m_value(0), m_id(-1), m_innerSize(aEval.m_expr.lhs().innerSize()) + public: + class InnerIterator { - this->operator++(); - } + enum { IsRowMajor = (int(Lhs::Flags) & RowMajorBit) == RowMajorBit }; - EIGEN_STRONG_INLINE InnerIterator& operator++() - { - ++m_id; - if(m_idoperator++(); + } + + EIGEN_STRONG_INLINE InnerIterator &operator++() + { + ++m_id; + if (m_id < m_innerSize) { + Scalar rhsVal = m_rhsEval.coeff(IsRowMajor ? m_lhsIter.outer() : m_id, IsRowMajor ? m_id : m_lhsIter.outer()); + if (m_lhsIter && m_lhsIter.index() == m_id) { + m_value = m_functor(m_lhsIter.value(), rhsVal); + ++m_lhsIter; + } else + m_value = m_functor(Scalar(0), rhsVal); } - else - m_value = m_functor(Scalar(0),rhsVal); + + return *this; } - return *this; - } + EIGEN_STRONG_INLINE Scalar value() const + { + eigen_internal_assert(m_id < m_innerSize); + return m_value; + } + + EIGEN_STRONG_INLINE StorageIndex index() const { return m_id; } + EIGEN_STRONG_INLINE Index outer() const { return m_lhsIter.outer(); } + EIGEN_STRONG_INLINE Index row() const { return IsRowMajor ? m_lhsIter.outer() : m_id; } + EIGEN_STRONG_INLINE Index col() const { return IsRowMajor ? m_id : m_lhsIter.outer(); } - EIGEN_STRONG_INLINE Scalar value() const { eigen_internal_assert(m_id &m_rhsEval; + const BinaryOp &m_functor; + Scalar m_value; + StorageIndex m_id; + StorageIndex m_innerSize; + }; - EIGEN_STRONG_INLINE operator bool() const { return m_id::CoeffReadCost + evaluator::CoeffReadCost + functor_traits::Cost, + // Expose storage order of the sparse expression + Flags = (XprType::Flags & ~RowMajorBit) | (int(Lhs::Flags) & RowMajorBit) + }; + + explicit binary_evaluator(const XprType &xpr) + : m_functor(xpr.functor()), m_lhsImpl(xpr.lhs()), m_rhsImpl(xpr.rhs()), m_expr(xpr) + { + EIGEN_INTERNAL_CHECK_COST_VALUE(functor_traits::Cost); + EIGEN_INTERNAL_CHECK_COST_VALUE(CoeffReadCost); + } + + inline Index nonZerosEstimate() const { return m_expr.size(); } protected: - LhsIterator m_lhsIter; - const evaluator &m_rhsEval; - const BinaryOp& m_functor; - Scalar m_value; - StorageIndex m_id; - StorageIndex m_innerSize; + const BinaryOp m_functor; + evaluator m_lhsImpl; + evaluator m_rhsImpl; + const XprType &m_expr; }; + template::Kind, + typename RhsKind = typename evaluator_traits::Kind, + typename LhsScalar = typename traits::Scalar, + typename RhsScalar = typename traits::Scalar> + struct sparse_conjunction_evaluator; + + // "sparse .* sparse" + template + struct binary_evaluator, Lhs, Rhs>, IteratorBased, IteratorBased> + : sparse_conjunction_evaluator, Lhs, Rhs>> + { + typedef CwiseBinaryOp, Lhs, Rhs> XprType; + typedef sparse_conjunction_evaluator Base; + explicit binary_evaluator(const XprType &xpr) : Base(xpr) {} + }; + // "dense .* sparse" + template + struct binary_evaluator, Lhs, Rhs>, IndexBased, IteratorBased> + : sparse_conjunction_evaluator, Lhs, Rhs>> + { + typedef CwiseBinaryOp, Lhs, Rhs> XprType; + typedef sparse_conjunction_evaluator Base; + explicit binary_evaluator(const XprType &xpr) : Base(xpr) {} + }; + // "sparse .* dense" + template + struct binary_evaluator, Lhs, Rhs>, IteratorBased, IndexBased> + : sparse_conjunction_evaluator, Lhs, Rhs>> + { + typedef CwiseBinaryOp, Lhs, Rhs> XprType; + typedef sparse_conjunction_evaluator Base; + explicit binary_evaluator(const XprType &xpr) : Base(xpr) {} + }; - enum { - CoeffReadCost = evaluator::CoeffReadCost + evaluator::CoeffReadCost + functor_traits::Cost, - // Expose storage order of the sparse expression - Flags = (XprType::Flags & ~RowMajorBit) | (int(Lhs::Flags)&RowMajorBit) + // "sparse ./ dense" + template + struct binary_evaluator, Lhs, Rhs>, IteratorBased, IndexBased> + : sparse_conjunction_evaluator, Lhs, Rhs>> + { + typedef CwiseBinaryOp, Lhs, Rhs> XprType; + typedef sparse_conjunction_evaluator Base; + explicit binary_evaluator(const XprType &xpr) : Base(xpr) {} }; - explicit binary_evaluator(const XprType& xpr) - : m_functor(xpr.functor()), - m_lhsImpl(xpr.lhs()), - m_rhsImpl(xpr.rhs()), - m_expr(xpr) + // "sparse && sparse" + template + struct binary_evaluator, IteratorBased, IteratorBased> + : sparse_conjunction_evaluator> { - EIGEN_INTERNAL_CHECK_COST_VALUE(functor_traits::Cost); - EIGEN_INTERNAL_CHECK_COST_VALUE(CoeffReadCost); - } + typedef CwiseBinaryOp XprType; + typedef sparse_conjunction_evaluator Base; + explicit binary_evaluator(const XprType &xpr) : Base(xpr) {} + }; + // "dense && sparse" + template + struct binary_evaluator, IndexBased, IteratorBased> + : sparse_conjunction_evaluator> + { + typedef CwiseBinaryOp XprType; + typedef sparse_conjunction_evaluator Base; + explicit binary_evaluator(const XprType &xpr) : Base(xpr) {} + }; + // "sparse && dense" + template + struct binary_evaluator, IteratorBased, IndexBased> + : sparse_conjunction_evaluator> + { + typedef CwiseBinaryOp XprType; + typedef sparse_conjunction_evaluator Base; + explicit binary_evaluator(const XprType &xpr) : Base(xpr) {} + }; - inline Index nonZerosEstimate() const { - return m_expr.size(); - } + // "sparse ^ sparse" + template + struct sparse_conjunction_evaluator : evaluator_base + { + protected: + typedef typename XprType::Functor BinaryOp; + typedef typename XprType::Lhs LhsArg; + typedef typename XprType::Rhs RhsArg; + typedef typename evaluator::InnerIterator LhsIterator; + typedef typename evaluator::InnerIterator RhsIterator; + typedef typename XprType::StorageIndex StorageIndex; + typedef typename traits::Scalar Scalar; -protected: - const BinaryOp m_functor; - evaluator m_lhsImpl; - evaluator m_rhsImpl; - const XprType &m_expr; -}; + public: + class InnerIterator + { + public: + EIGEN_STRONG_INLINE InnerIterator(const sparse_conjunction_evaluator &aEval, Index outer) + : m_lhsIter(aEval.m_lhsImpl, outer), m_rhsIter(aEval.m_rhsImpl, outer), m_functor(aEval.m_functor) + { + while (m_lhsIter && m_rhsIter && (m_lhsIter.index() != m_rhsIter.index())) { + if (m_lhsIter.index() < m_rhsIter.index()) + ++m_lhsIter; + else + ++m_rhsIter; + } + } -template::Kind, - typename RhsKind = typename evaluator_traits::Kind, - typename LhsScalar = typename traits::Scalar, - typename RhsScalar = typename traits::Scalar> struct sparse_conjunction_evaluator; + EIGEN_STRONG_INLINE InnerIterator &operator++() + { + ++m_lhsIter; + ++m_rhsIter; + while (m_lhsIter && m_rhsIter && (m_lhsIter.index() != m_rhsIter.index())) { + if (m_lhsIter.index() < m_rhsIter.index()) + ++m_lhsIter; + else + ++m_rhsIter; + } + return *this; + } -// "sparse .* sparse" -template -struct binary_evaluator, Lhs, Rhs>, IteratorBased, IteratorBased> - : sparse_conjunction_evaluator, Lhs, Rhs> > -{ - typedef CwiseBinaryOp, Lhs, Rhs> XprType; - typedef sparse_conjunction_evaluator Base; - explicit binary_evaluator(const XprType& xpr) : Base(xpr) {} -}; -// "dense .* sparse" -template -struct binary_evaluator, Lhs, Rhs>, IndexBased, IteratorBased> - : sparse_conjunction_evaluator, Lhs, Rhs> > -{ - typedef CwiseBinaryOp, Lhs, Rhs> XprType; - typedef sparse_conjunction_evaluator Base; - explicit binary_evaluator(const XprType& xpr) : Base(xpr) {} -}; -// "sparse .* dense" -template -struct binary_evaluator, Lhs, Rhs>, IteratorBased, IndexBased> - : sparse_conjunction_evaluator, Lhs, Rhs> > -{ - typedef CwiseBinaryOp, Lhs, Rhs> XprType; - typedef sparse_conjunction_evaluator Base; - explicit binary_evaluator(const XprType& xpr) : Base(xpr) {} -}; + EIGEN_STRONG_INLINE Scalar value() const { return m_functor(m_lhsIter.value(), m_rhsIter.value()); } -// "sparse ./ dense" -template -struct binary_evaluator, Lhs, Rhs>, IteratorBased, IndexBased> - : sparse_conjunction_evaluator, Lhs, Rhs> > -{ - typedef CwiseBinaryOp, Lhs, Rhs> XprType; - typedef sparse_conjunction_evaluator Base; - explicit binary_evaluator(const XprType& xpr) : Base(xpr) {} -}; + EIGEN_STRONG_INLINE StorageIndex index() const { return m_lhsIter.index(); } + EIGEN_STRONG_INLINE Index outer() const { return m_lhsIter.outer(); } + EIGEN_STRONG_INLINE Index row() const { return m_lhsIter.row(); } + EIGEN_STRONG_INLINE Index col() const { return m_lhsIter.col(); } -// "sparse && sparse" -template -struct binary_evaluator, IteratorBased, IteratorBased> - : sparse_conjunction_evaluator > -{ - typedef CwiseBinaryOp XprType; - typedef sparse_conjunction_evaluator Base; - explicit binary_evaluator(const XprType& xpr) : Base(xpr) {} -}; -// "dense && sparse" -template -struct binary_evaluator, IndexBased, IteratorBased> - : sparse_conjunction_evaluator > -{ - typedef CwiseBinaryOp XprType; - typedef sparse_conjunction_evaluator Base; - explicit binary_evaluator(const XprType& xpr) : Base(xpr) {} -}; -// "sparse && dense" -template -struct binary_evaluator, IteratorBased, IndexBased> - : sparse_conjunction_evaluator > -{ - typedef CwiseBinaryOp XprType; - typedef sparse_conjunction_evaluator Base; - explicit binary_evaluator(const XprType& xpr) : Base(xpr) {} -}; + EIGEN_STRONG_INLINE operator bool() const { return (m_lhsIter && m_rhsIter); } -// "sparse ^ sparse" -template -struct sparse_conjunction_evaluator - : evaluator_base -{ -protected: - typedef typename XprType::Functor BinaryOp; - typedef typename XprType::Lhs LhsArg; - typedef typename XprType::Rhs RhsArg; - typedef typename evaluator::InnerIterator LhsIterator; - typedef typename evaluator::InnerIterator RhsIterator; - typedef typename XprType::StorageIndex StorageIndex; - typedef typename traits::Scalar Scalar; -public: + protected: + LhsIterator m_lhsIter; + RhsIterator m_rhsIter; + const BinaryOp &m_functor; + }; - class InnerIterator - { - public: - - EIGEN_STRONG_INLINE InnerIterator(const sparse_conjunction_evaluator& aEval, Index outer) - : m_lhsIter(aEval.m_lhsImpl,outer), m_rhsIter(aEval.m_rhsImpl,outer), m_functor(aEval.m_functor) + + enum { + CoeffReadCost = + evaluator::CoeffReadCost + evaluator::CoeffReadCost + functor_traits::Cost, + Flags = XprType::Flags + }; + + explicit sparse_conjunction_evaluator(const XprType &xpr) + : m_functor(xpr.functor()), m_lhsImpl(xpr.lhs()), m_rhsImpl(xpr.rhs()) { - while (m_lhsIter && m_rhsIter && (m_lhsIter.index() != m_rhsIter.index())) - { - if (m_lhsIter.index() < m_rhsIter.index()) - ++m_lhsIter; - else - ++m_rhsIter; - } + EIGEN_INTERNAL_CHECK_COST_VALUE(functor_traits::Cost); + EIGEN_INTERNAL_CHECK_COST_VALUE(CoeffReadCost); } - EIGEN_STRONG_INLINE InnerIterator& operator++() + inline Index nonZerosEstimate() const { - ++m_lhsIter; - ++m_rhsIter; - while (m_lhsIter && m_rhsIter && (m_lhsIter.index() != m_rhsIter.index())) - { - if (m_lhsIter.index() < m_rhsIter.index()) - ++m_lhsIter; - else - ++m_rhsIter; - } - return *this; + return (std::min)(m_lhsImpl.nonZerosEstimate(), m_rhsImpl.nonZerosEstimate()); } - - EIGEN_STRONG_INLINE Scalar value() const { return m_functor(m_lhsIter.value(), m_rhsIter.value()); } - - EIGEN_STRONG_INLINE StorageIndex index() const { return m_lhsIter.index(); } - EIGEN_STRONG_INLINE Index outer() const { return m_lhsIter.outer(); } - EIGEN_STRONG_INLINE Index row() const { return m_lhsIter.row(); } - EIGEN_STRONG_INLINE Index col() const { return m_lhsIter.col(); } - - EIGEN_STRONG_INLINE operator bool() const { return (m_lhsIter && m_rhsIter); } protected: - LhsIterator m_lhsIter; - RhsIterator m_rhsIter; - const BinaryOp& m_functor; - }; - - - enum { - CoeffReadCost = evaluator::CoeffReadCost + evaluator::CoeffReadCost + functor_traits::Cost, - Flags = XprType::Flags + const BinaryOp m_functor; + evaluator m_lhsImpl; + evaluator m_rhsImpl; }; - - explicit sparse_conjunction_evaluator(const XprType& xpr) - : m_functor(xpr.functor()), - m_lhsImpl(xpr.lhs()), - m_rhsImpl(xpr.rhs()) + + // "dense ^ sparse" + template + struct sparse_conjunction_evaluator : evaluator_base { - EIGEN_INTERNAL_CHECK_COST_VALUE(functor_traits::Cost); - EIGEN_INTERNAL_CHECK_COST_VALUE(CoeffReadCost); - } - - inline Index nonZerosEstimate() const { - return (std::min)(m_lhsImpl.nonZerosEstimate(), m_rhsImpl.nonZerosEstimate()); - } + protected: + typedef typename XprType::Functor BinaryOp; + typedef typename XprType::Lhs LhsArg; + typedef typename XprType::Rhs RhsArg; + typedef evaluator LhsEvaluator; + typedef typename evaluator::InnerIterator RhsIterator; + typedef typename XprType::StorageIndex StorageIndex; + typedef typename traits::Scalar Scalar; -protected: - const BinaryOp m_functor; - evaluator m_lhsImpl; - evaluator m_rhsImpl; -}; + public: + class InnerIterator + { + enum { IsRowMajor = (int(RhsArg::Flags) & RowMajorBit) == RowMajorBit }; -// "dense ^ sparse" -template -struct sparse_conjunction_evaluator - : evaluator_base -{ -protected: - typedef typename XprType::Functor BinaryOp; - typedef typename XprType::Lhs LhsArg; - typedef typename XprType::Rhs RhsArg; - typedef evaluator LhsEvaluator; - typedef typename evaluator::InnerIterator RhsIterator; - typedef typename XprType::StorageIndex StorageIndex; - typedef typename traits::Scalar Scalar; -public: + public: + EIGEN_STRONG_INLINE InnerIterator(const sparse_conjunction_evaluator &aEval, Index outer) + : m_lhsEval(aEval.m_lhsImpl), m_rhsIter(aEval.m_rhsImpl, outer), m_functor(aEval.m_functor), m_outer(outer) + {} - class InnerIterator - { - enum { IsRowMajor = (int(RhsArg::Flags)&RowMajorBit)==RowMajorBit }; + EIGEN_STRONG_INLINE InnerIterator &operator++() + { + ++m_rhsIter; + return *this; + } - public: - - EIGEN_STRONG_INLINE InnerIterator(const sparse_conjunction_evaluator& aEval, Index outer) - : m_lhsEval(aEval.m_lhsImpl), m_rhsIter(aEval.m_rhsImpl,outer), m_functor(aEval.m_functor), m_outer(outer) - {} + EIGEN_STRONG_INLINE Scalar value() const + { + return m_functor( + m_lhsEval.coeff(IsRowMajor ? m_outer : m_rhsIter.index(), IsRowMajor ? m_rhsIter.index() : m_outer), + m_rhsIter.value()); + } - EIGEN_STRONG_INLINE InnerIterator& operator++() + EIGEN_STRONG_INLINE StorageIndex index() const { return m_rhsIter.index(); } + EIGEN_STRONG_INLINE Index outer() const { return m_rhsIter.outer(); } + EIGEN_STRONG_INLINE Index row() const { return m_rhsIter.row(); } + EIGEN_STRONG_INLINE Index col() const { return m_rhsIter.col(); } + + EIGEN_STRONG_INLINE operator bool() const { return m_rhsIter; } + + protected: + const LhsEvaluator &m_lhsEval; + RhsIterator m_rhsIter; + const BinaryOp &m_functor; + const Index m_outer; + }; + + + enum { + CoeffReadCost = + evaluator::CoeffReadCost + evaluator::CoeffReadCost + functor_traits::Cost, + // Expose storage order of the sparse expression + Flags = (XprType::Flags & ~RowMajorBit) | (int(RhsArg::Flags) & RowMajorBit) + }; + + explicit sparse_conjunction_evaluator(const XprType &xpr) + : m_functor(xpr.functor()), m_lhsImpl(xpr.lhs()), m_rhsImpl(xpr.rhs()) { - ++m_rhsIter; - return *this; + EIGEN_INTERNAL_CHECK_COST_VALUE(functor_traits::Cost); + EIGEN_INTERNAL_CHECK_COST_VALUE(CoeffReadCost); } - EIGEN_STRONG_INLINE Scalar value() const - { return m_functor(m_lhsEval.coeff(IsRowMajor?m_outer:m_rhsIter.index(),IsRowMajor?m_rhsIter.index():m_outer), m_rhsIter.value()); } - - EIGEN_STRONG_INLINE StorageIndex index() const { return m_rhsIter.index(); } - EIGEN_STRONG_INLINE Index outer() const { return m_rhsIter.outer(); } - EIGEN_STRONG_INLINE Index row() const { return m_rhsIter.row(); } - EIGEN_STRONG_INLINE Index col() const { return m_rhsIter.col(); } + inline Index nonZerosEstimate() const { return m_rhsImpl.nonZerosEstimate(); } - EIGEN_STRONG_INLINE operator bool() const { return m_rhsIter; } - protected: - const LhsEvaluator &m_lhsEval; - RhsIterator m_rhsIter; - const BinaryOp& m_functor; - const Index m_outer; + const BinaryOp m_functor; + evaluator m_lhsImpl; + evaluator m_rhsImpl; }; - - - enum { - CoeffReadCost = evaluator::CoeffReadCost + evaluator::CoeffReadCost + functor_traits::Cost, - // Expose storage order of the sparse expression - Flags = (XprType::Flags & ~RowMajorBit) | (int(RhsArg::Flags)&RowMajorBit) - }; - - explicit sparse_conjunction_evaluator(const XprType& xpr) - : m_functor(xpr.functor()), - m_lhsImpl(xpr.lhs()), - m_rhsImpl(xpr.rhs()) + + // "sparse ^ dense" + template + struct sparse_conjunction_evaluator : evaluator_base { - EIGEN_INTERNAL_CHECK_COST_VALUE(functor_traits::Cost); - EIGEN_INTERNAL_CHECK_COST_VALUE(CoeffReadCost); - } - - inline Index nonZerosEstimate() const { - return m_rhsImpl.nonZerosEstimate(); - } + protected: + typedef typename XprType::Functor BinaryOp; + typedef typename XprType::Lhs LhsArg; + typedef typename XprType::Rhs RhsArg; + typedef typename evaluator::InnerIterator LhsIterator; + typedef evaluator RhsEvaluator; + typedef typename XprType::StorageIndex StorageIndex; + typedef typename traits::Scalar Scalar; -protected: - const BinaryOp m_functor; - evaluator m_lhsImpl; - evaluator m_rhsImpl; -}; + public: + class InnerIterator + { + enum { IsRowMajor = (int(LhsArg::Flags) & RowMajorBit) == RowMajorBit }; -// "sparse ^ dense" -template -struct sparse_conjunction_evaluator - : evaluator_base -{ -protected: - typedef typename XprType::Functor BinaryOp; - typedef typename XprType::Lhs LhsArg; - typedef typename XprType::Rhs RhsArg; - typedef typename evaluator::InnerIterator LhsIterator; - typedef evaluator RhsEvaluator; - typedef typename XprType::StorageIndex StorageIndex; - typedef typename traits::Scalar Scalar; -public: + public: + EIGEN_STRONG_INLINE InnerIterator(const sparse_conjunction_evaluator &aEval, Index outer) + : m_lhsIter(aEval.m_lhsImpl, outer), m_rhsEval(aEval.m_rhsImpl), m_functor(aEval.m_functor), m_outer(outer) + {} - class InnerIterator - { - enum { IsRowMajor = (int(LhsArg::Flags)&RowMajorBit)==RowMajorBit }; + EIGEN_STRONG_INLINE InnerIterator &operator++() + { + ++m_lhsIter; + return *this; + } - public: - - EIGEN_STRONG_INLINE InnerIterator(const sparse_conjunction_evaluator& aEval, Index outer) - : m_lhsIter(aEval.m_lhsImpl,outer), m_rhsEval(aEval.m_rhsImpl), m_functor(aEval.m_functor), m_outer(outer) - {} + EIGEN_STRONG_INLINE Scalar value() const + { + return m_functor(m_lhsIter.value(), + m_rhsEval.coeff(IsRowMajor ? m_outer : m_lhsIter.index(), IsRowMajor ? m_lhsIter.index() : m_outer)); + } + + EIGEN_STRONG_INLINE StorageIndex index() const { return m_lhsIter.index(); } + EIGEN_STRONG_INLINE Index outer() const { return m_lhsIter.outer(); } + EIGEN_STRONG_INLINE Index row() const { return m_lhsIter.row(); } + EIGEN_STRONG_INLINE Index col() const { return m_lhsIter.col(); } + + EIGEN_STRONG_INLINE operator bool() const { return m_lhsIter; } - EIGEN_STRONG_INLINE InnerIterator& operator++() + protected: + LhsIterator m_lhsIter; + const evaluator &m_rhsEval; + const BinaryOp &m_functor; + const Index m_outer; + }; + + + enum { + CoeffReadCost = + evaluator::CoeffReadCost + evaluator::CoeffReadCost + functor_traits::Cost, + // Expose storage order of the sparse expression + Flags = (XprType::Flags & ~RowMajorBit) | (int(LhsArg::Flags) & RowMajorBit) + }; + + explicit sparse_conjunction_evaluator(const XprType &xpr) + : m_functor(xpr.functor()), m_lhsImpl(xpr.lhs()), m_rhsImpl(xpr.rhs()) { - ++m_lhsIter; - return *this; + EIGEN_INTERNAL_CHECK_COST_VALUE(functor_traits::Cost); + EIGEN_INTERNAL_CHECK_COST_VALUE(CoeffReadCost); } - EIGEN_STRONG_INLINE Scalar value() const - { return m_functor(m_lhsIter.value(), - m_rhsEval.coeff(IsRowMajor?m_outer:m_lhsIter.index(),IsRowMajor?m_lhsIter.index():m_outer)); } + inline Index nonZerosEstimate() const { return m_lhsImpl.nonZerosEstimate(); } - EIGEN_STRONG_INLINE StorageIndex index() const { return m_lhsIter.index(); } - EIGEN_STRONG_INLINE Index outer() const { return m_lhsIter.outer(); } - EIGEN_STRONG_INLINE Index row() const { return m_lhsIter.row(); } - EIGEN_STRONG_INLINE Index col() const { return m_lhsIter.col(); } - - EIGEN_STRONG_INLINE operator bool() const { return m_lhsIter; } - protected: - LhsIterator m_lhsIter; - const evaluator &m_rhsEval; - const BinaryOp& m_functor; - const Index m_outer; - }; - - - enum { - CoeffReadCost = evaluator::CoeffReadCost + evaluator::CoeffReadCost + functor_traits::Cost, - // Expose storage order of the sparse expression - Flags = (XprType::Flags & ~RowMajorBit) | (int(LhsArg::Flags)&RowMajorBit) + const BinaryOp m_functor; + evaluator m_lhsImpl; + evaluator m_rhsImpl; }; - - explicit sparse_conjunction_evaluator(const XprType& xpr) - : m_functor(xpr.functor()), - m_lhsImpl(xpr.lhs()), - m_rhsImpl(xpr.rhs()) - { - EIGEN_INTERNAL_CHECK_COST_VALUE(functor_traits::Cost); - EIGEN_INTERNAL_CHECK_COST_VALUE(CoeffReadCost); - } - - inline Index nonZerosEstimate() const { - return m_lhsImpl.nonZerosEstimate(); - } - -protected: - const BinaryOp m_functor; - evaluator m_lhsImpl; - evaluator m_rhsImpl; -}; -} +}// namespace internal /*************************************************************************** -* Implementation of SparseMatrixBase and SparseCwise functions/operators -***************************************************************************/ + * Implementation of SparseMatrixBase and SparseCwise functions/operators + ***************************************************************************/ template template -Derived& SparseMatrixBase::operator+=(const EigenBase &other) +Derived &SparseMatrixBase::operator+=(const EigenBase &other) { - call_assignment(derived(), other.derived(), internal::add_assign_op()); + call_assignment(derived(), other.derived(), internal::add_assign_op()); return derived(); } template template -Derived& SparseMatrixBase::operator-=(const EigenBase &other) +Derived &SparseMatrixBase::operator-=(const EigenBase &other) { - call_assignment(derived(), other.derived(), internal::assign_op()); + call_assignment(derived(), other.derived(), internal::assign_op()); return derived(); } template template -EIGEN_STRONG_INLINE Derived & -SparseMatrixBase::operator-=(const SparseMatrixBase &other) +EIGEN_STRONG_INLINE Derived &SparseMatrixBase::operator-=(const SparseMatrixBase &other) { return derived() = derived() - other.derived(); } template template -EIGEN_STRONG_INLINE Derived & -SparseMatrixBase::operator+=(const SparseMatrixBase& other) +EIGEN_STRONG_INLINE Derived &SparseMatrixBase::operator+=(const SparseMatrixBase &other) { return derived() = derived() + other.derived(); } template template -Derived& SparseMatrixBase::operator+=(const DiagonalBase& other) +Derived &SparseMatrixBase::operator+=(const DiagonalBase &other) { - call_assignment_no_alias(derived(), other.derived(), internal::add_assign_op()); + call_assignment_no_alias( + derived(), other.derived(), internal::add_assign_op()); return derived(); } template template -Derived& SparseMatrixBase::operator-=(const DiagonalBase& other) +Derived &SparseMatrixBase::operator-=(const DiagonalBase &other) { - call_assignment_no_alias(derived(), other.derived(), internal::sub_assign_op()); + call_assignment_no_alias( + derived(), other.derived(), internal::sub_assign_op()); return derived(); } - + template template EIGEN_STRONG_INLINE const typename SparseMatrixBase::template CwiseProductDenseReturnType::Type -SparseMatrixBase::cwiseProduct(const MatrixBase &other) const + SparseMatrixBase::cwiseProduct(const MatrixBase &other) const { return typename CwiseProductDenseReturnType::Type(derived(), other.derived()); } template -EIGEN_STRONG_INLINE const CwiseBinaryOp, const DenseDerived, const SparseDerived> -operator+(const MatrixBase &a, const SparseMatrixBase &b) +EIGEN_STRONG_INLINE const + CwiseBinaryOp, + const DenseDerived, + const SparseDerived> + operator+(const MatrixBase &a, const SparseMatrixBase &b) { - return CwiseBinaryOp, const DenseDerived, const SparseDerived>(a.derived(), b.derived()); + return CwiseBinaryOp, + const DenseDerived, + const SparseDerived>(a.derived(), b.derived()); } template -EIGEN_STRONG_INLINE const CwiseBinaryOp, const SparseDerived, const DenseDerived> -operator+(const SparseMatrixBase &a, const MatrixBase &b) +EIGEN_STRONG_INLINE const + CwiseBinaryOp, + const SparseDerived, + const DenseDerived> + operator+(const SparseMatrixBase &a, const MatrixBase &b) { - return CwiseBinaryOp, const SparseDerived, const DenseDerived>(a.derived(), b.derived()); + return CwiseBinaryOp, + const SparseDerived, + const DenseDerived>(a.derived(), b.derived()); } template -EIGEN_STRONG_INLINE const CwiseBinaryOp, const DenseDerived, const SparseDerived> -operator-(const MatrixBase &a, const SparseMatrixBase &b) +EIGEN_STRONG_INLINE const + CwiseBinaryOp, + const DenseDerived, + const SparseDerived> + operator-(const MatrixBase &a, const SparseMatrixBase &b) { - return CwiseBinaryOp, const DenseDerived, const SparseDerived>(a.derived(), b.derived()); + return CwiseBinaryOp, + const DenseDerived, + const SparseDerived>(a.derived(), b.derived()); } template -EIGEN_STRONG_INLINE const CwiseBinaryOp, const SparseDerived, const DenseDerived> -operator-(const SparseMatrixBase &a, const MatrixBase &b) +EIGEN_STRONG_INLINE const + CwiseBinaryOp, + const SparseDerived, + const DenseDerived> + operator-(const SparseMatrixBase &a, const MatrixBase &b) { - return CwiseBinaryOp, const SparseDerived, const DenseDerived>(a.derived(), b.derived()); + return CwiseBinaryOp, + const SparseDerived, + const DenseDerived>(a.derived(), b.derived()); } -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_SPARSE_CWISE_BINARY_OP_H +#endif// EIGEN_SPARSE_CWISE_BINARY_OP_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/SparseCore/SparseCwiseUnaryOp.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/SparseCore/SparseCwiseUnaryOp.h index ea797379..50e64deb 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/SparseCore/SparseCwiseUnaryOp.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/SparseCore/SparseCwiseUnaryOp.h @@ -10,139 +10,132 @@ #ifndef EIGEN_SPARSE_CWISE_UNARY_OP_H #define EIGEN_SPARSE_CWISE_UNARY_OP_H -namespace Eigen { +namespace Eigen { namespace internal { - -template -struct unary_evaluator, IteratorBased> - : public evaluator_base > -{ + + template + struct unary_evaluator, IteratorBased> + : public evaluator_base> + { public: typedef CwiseUnaryOp XprType; class InnerIterator; - - enum { - CoeffReadCost = evaluator::CoeffReadCost + functor_traits::Cost, - Flags = XprType::Flags - }; - - explicit unary_evaluator(const XprType& op) : m_functor(op.functor()), m_argImpl(op.nestedExpression()) + + enum { CoeffReadCost = evaluator::CoeffReadCost + functor_traits::Cost, Flags = XprType::Flags }; + + explicit unary_evaluator(const XprType &op) : m_functor(op.functor()), m_argImpl(op.nestedExpression()) { EIGEN_INTERNAL_CHECK_COST_VALUE(functor_traits::Cost); EIGEN_INTERNAL_CHECK_COST_VALUE(CoeffReadCost); } - - inline Index nonZerosEstimate() const { - return m_argImpl.nonZerosEstimate(); - } + + inline Index nonZerosEstimate() const { return m_argImpl.nonZerosEstimate(); } protected: - typedef typename evaluator::InnerIterator EvalIterator; - + typedef typename evaluator::InnerIterator EvalIterator; + const UnaryOp m_functor; evaluator m_argImpl; -}; + }; -template -class unary_evaluator, IteratorBased>::InnerIterator - : public unary_evaluator, IteratorBased>::EvalIterator -{ + template + class unary_evaluator, IteratorBased>::InnerIterator + : public unary_evaluator, IteratorBased>::EvalIterator + { typedef typename XprType::Scalar Scalar; - typedef typename unary_evaluator, IteratorBased>::EvalIterator Base; - public: + typedef typename unary_evaluator, IteratorBased>::EvalIterator Base; - EIGEN_STRONG_INLINE InnerIterator(const unary_evaluator& unaryOp, Index outer) - : Base(unaryOp.m_argImpl,outer), m_functor(unaryOp.m_functor) + public: + EIGEN_STRONG_INLINE InnerIterator(const unary_evaluator &unaryOp, Index outer) + : Base(unaryOp.m_argImpl, outer), m_functor(unaryOp.m_functor) {} - EIGEN_STRONG_INLINE InnerIterator& operator++() - { Base::operator++(); return *this; } + EIGEN_STRONG_INLINE InnerIterator &operator++() + { + Base::operator++(); + return *this; + } EIGEN_STRONG_INLINE Scalar value() const { return m_functor(Base::value()); } protected: const UnaryOp m_functor; + private: - Scalar& valueRef(); -}; + Scalar &valueRef(); + }; -template -struct unary_evaluator, IteratorBased> - : public evaluator_base > -{ + template + struct unary_evaluator, IteratorBased> + : public evaluator_base> + { public: typedef CwiseUnaryView XprType; class InnerIterator; - - enum { - CoeffReadCost = evaluator::CoeffReadCost + functor_traits::Cost, - Flags = XprType::Flags - }; - - explicit unary_evaluator(const XprType& op) : m_functor(op.functor()), m_argImpl(op.nestedExpression()) + + enum { CoeffReadCost = evaluator::CoeffReadCost + functor_traits::Cost, Flags = XprType::Flags }; + + explicit unary_evaluator(const XprType &op) : m_functor(op.functor()), m_argImpl(op.nestedExpression()) { EIGEN_INTERNAL_CHECK_COST_VALUE(functor_traits::Cost); EIGEN_INTERNAL_CHECK_COST_VALUE(CoeffReadCost); } protected: - typedef typename evaluator::InnerIterator EvalIterator; - + typedef typename evaluator::InnerIterator EvalIterator; + const ViewOp m_functor; evaluator m_argImpl; -}; + }; -template -class unary_evaluator, IteratorBased>::InnerIterator - : public unary_evaluator, IteratorBased>::EvalIterator -{ + template + class unary_evaluator, IteratorBased>::InnerIterator + : public unary_evaluator, IteratorBased>::EvalIterator + { typedef typename XprType::Scalar Scalar; - typedef typename unary_evaluator, IteratorBased>::EvalIterator Base; - public: + typedef typename unary_evaluator, IteratorBased>::EvalIterator Base; - EIGEN_STRONG_INLINE InnerIterator(const unary_evaluator& unaryOp, Index outer) - : Base(unaryOp.m_argImpl,outer), m_functor(unaryOp.m_functor) + public: + EIGEN_STRONG_INLINE InnerIterator(const unary_evaluator &unaryOp, Index outer) + : Base(unaryOp.m_argImpl, outer), m_functor(unaryOp.m_functor) {} - EIGEN_STRONG_INLINE InnerIterator& operator++() - { Base::operator++(); return *this; } + EIGEN_STRONG_INLINE InnerIterator &operator++() + { + Base::operator++(); + return *this; + } EIGEN_STRONG_INLINE Scalar value() const { return m_functor(Base::value()); } - EIGEN_STRONG_INLINE Scalar& valueRef() { return m_functor(Base::valueRef()); } + EIGEN_STRONG_INLINE Scalar &valueRef() { return m_functor(Base::valueRef()); } protected: const ViewOp m_functor; -}; + }; -} // end namespace internal +}// end namespace internal -template -EIGEN_STRONG_INLINE Derived& -SparseMatrixBase::operator*=(const Scalar& other) +template EIGEN_STRONG_INLINE Derived &SparseMatrixBase::operator*=(const Scalar &other) { typedef typename internal::evaluator::InnerIterator EvalIterator; internal::evaluator thisEval(derived()); - for (Index j=0; j -EIGEN_STRONG_INLINE Derived& -SparseMatrixBase::operator/=(const Scalar& other) +template EIGEN_STRONG_INLINE Derived &SparseMatrixBase::operator/=(const Scalar &other) { typedef typename internal::evaluator::InnerIterator EvalIterator; internal::evaluator thisEval(derived()); - for (Index j=0; j struct product_promote_storage_type { typedef Sparse ret; }; -template <> struct product_promote_storage_type { typedef Sparse ret; }; - -template -struct sparse_time_dense_product_impl; - -template -struct sparse_time_dense_product_impl -{ - typedef typename internal::remove_all::type Lhs; - typedef typename internal::remove_all::type Rhs; - typedef typename internal::remove_all::type Res; - typedef typename evaluator::InnerIterator LhsInnerIterator; - typedef evaluator LhsEval; - static void run(const SparseLhsType& lhs, const DenseRhsType& rhs, DenseResType& res, const typename Res::Scalar& alpha) + template<> struct product_promote_storage_type { - LhsEval lhsEval(lhs); - - Index n = lhs.outerSize(); + typedef Sparse ret; + }; + template<> struct product_promote_storage_type + { + typedef Sparse ret; + }; + + template + struct sparse_time_dense_product_impl; + + template + struct sparse_time_dense_product_impl + { + typedef typename internal::remove_all::type Lhs; + typedef typename internal::remove_all::type Rhs; + typedef typename internal::remove_all::type Res; + typedef typename evaluator::InnerIterator LhsInnerIterator; + typedef evaluator LhsEval; + static void + run(const SparseLhsType &lhs, const DenseRhsType &rhs, DenseResType &res, const typename Res::Scalar &alpha) + { + LhsEval lhsEval(lhs); + + Index n = lhs.outerSize(); #ifdef EIGEN_HAS_OPENMP - Eigen::initParallel(); - Index threads = Eigen::nbThreads(); + Eigen::initParallel(); + Index threads = Eigen::nbThreads(); #endif - - for(Index c=0; c1 && lhsEval.nonZerosEstimate() > 20000) - { - #pragma omp parallel for schedule(dynamic,(n+threads*4-1)/(threads*4)) num_threads(threads) - for(Index i=0; i 1 && lhsEval.nonZerosEstimate() > 20000) { +#pragma omp parallel for schedule(dynamic, (n + threads * 4 - 1) / (threads * 4)) num_threads(threads) + for (Index i = 0; i < n; ++i) processRow(lhsEval, rhs, res, alpha, i, c); + } else #endif - { - for(Index i=0; i let's disable it for now as it is conflicting with generic scalar*matrix and matrix*scalar operators -// template -// struct ScalarBinaryOpTraits > -// { -// enum { -// Defined = 1 -// }; -// typedef typename CwiseUnaryOp, T2>::PlainObject ReturnType; -// }; - -template -struct sparse_time_dense_product_impl -{ - typedef typename internal::remove_all::type Lhs; - typedef typename internal::remove_all::type Rhs; - typedef typename internal::remove_all::type Res; - typedef typename evaluator::InnerIterator LhsInnerIterator; - static void run(const SparseLhsType& lhs, const DenseRhsType& rhs, DenseResType& res, const AlphaType& alpha) + + static void processRow(const LhsEval &lhsEval, + const DenseRhsType &rhs, + DenseResType &res, + const typename Res::Scalar &alpha, + Index i, + Index col) + { + typename Res::Scalar tmp(0); + for (LhsInnerIterator it(lhsEval, i); it; ++it) tmp += it.value() * rhs.coeff(it.index(), col); + res.coeffRef(i, col) += alpha * tmp; + } + }; + + // FIXME: what is the purpose of the following specialization? Is it for the BlockedSparse format? + // -> let's disable it for now as it is conflicting with generic scalar*matrix and matrix*scalar operators + // template + // struct ScalarBinaryOpTraits > + // { + // enum { + // Defined = 1 + // }; + // typedef typename CwiseUnaryOp, T2>::PlainObject ReturnType; + // }; + + template + struct sparse_time_dense_product_impl { - evaluator lhsEval(lhs); - for(Index c=0; c::type Lhs; + typedef typename internal::remove_all::type Rhs; + typedef typename internal::remove_all::type Res; + typedef typename evaluator::InnerIterator LhsInnerIterator; + static void run(const SparseLhsType &lhs, const DenseRhsType &rhs, DenseResType &res, const AlphaType &alpha) { - for(Index j=0; j::ReturnType rhs_j(alpha * rhs.coeff(j,c)); - for(LhsInnerIterator it(lhsEval,j); it ;++it) - res.coeffRef(it.index(),c) += it.value() * rhs_j; + evaluator lhsEval(lhs); + for (Index c = 0; c < rhs.cols(); ++c) { + for (Index j = 0; j < lhs.outerSize(); ++j) { + // typename Res::Scalar rhs_j = alpha * rhs.coeff(j,c); + typename ScalarBinaryOpTraits::ReturnType rhs_j(alpha * rhs.coeff(j, c)); + for (LhsInnerIterator it(lhsEval, j); it; ++it) res.coeffRef(it.index(), c) += it.value() * rhs_j; + } } } - } -}; - -template -struct sparse_time_dense_product_impl -{ - typedef typename internal::remove_all::type Lhs; - typedef typename internal::remove_all::type Rhs; - typedef typename internal::remove_all::type Res; - typedef typename evaluator::InnerIterator LhsInnerIterator; - static void run(const SparseLhsType& lhs, const DenseRhsType& rhs, DenseResType& res, const typename Res::Scalar& alpha) + }; + + template + struct sparse_time_dense_product_impl { - evaluator lhsEval(lhs); - for(Index j=0; j::type Lhs; + typedef typename internal::remove_all::type Rhs; + typedef typename internal::remove_all::type Res; + typedef typename evaluator::InnerIterator LhsInnerIterator; + static void + run(const SparseLhsType &lhs, const DenseRhsType &rhs, DenseResType &res, const typename Res::Scalar &alpha) { - typename Res::RowXpr res_j(res.row(j)); - for(LhsInnerIterator it(lhsEval,j); it ;++it) - res_j += (alpha*it.value()) * rhs.row(it.index()); + evaluator lhsEval(lhs); + for (Index j = 0; j < lhs.outerSize(); ++j) { + typename Res::RowXpr res_j(res.row(j)); + for (LhsInnerIterator it(lhsEval, j); it; ++it) res_j += (alpha * it.value()) * rhs.row(it.index()); + } } - } -}; - -template -struct sparse_time_dense_product_impl -{ - typedef typename internal::remove_all::type Lhs; - typedef typename internal::remove_all::type Rhs; - typedef typename internal::remove_all::type Res; - typedef typename evaluator::InnerIterator LhsInnerIterator; - static void run(const SparseLhsType& lhs, const DenseRhsType& rhs, DenseResType& res, const typename Res::Scalar& alpha) + }; + + template + struct sparse_time_dense_product_impl { - evaluator lhsEval(lhs); - for(Index j=0; j::type Lhs; + typedef typename internal::remove_all::type Rhs; + typedef typename internal::remove_all::type Res; + typedef typename evaluator::InnerIterator LhsInnerIterator; + static void + run(const SparseLhsType &lhs, const DenseRhsType &rhs, DenseResType &res, const typename Res::Scalar &alpha) { - typename Rhs::ConstRowXpr rhs_j(rhs.row(j)); - for(LhsInnerIterator it(lhsEval,j); it ;++it) - res.row(it.index()) += (alpha*it.value()) * rhs_j; + evaluator lhsEval(lhs); + for (Index j = 0; j < lhs.outerSize(); ++j) { + typename Rhs::ConstRowXpr rhs_j(rhs.row(j)); + for (LhsInnerIterator it(lhsEval, j); it; ++it) res.row(it.index()) += (alpha * it.value()) * rhs_j; + } } - } -}; + }; -template -inline void sparse_time_dense_product(const SparseLhsType& lhs, const DenseRhsType& rhs, DenseResType& res, const AlphaType& alpha) -{ - sparse_time_dense_product_impl::run(lhs, rhs, res, alpha); -} + template + inline void sparse_time_dense_product(const SparseLhsType &lhs, + const DenseRhsType &rhs, + DenseResType &res, + const AlphaType &alpha) + { + sparse_time_dense_product_impl::run(lhs, rhs, res, alpha); + } -} // end namespace internal +}// end namespace internal namespace internal { -template -struct generic_product_impl - : generic_product_impl_base > -{ - typedef typename Product::Scalar Scalar; - - template - static void scaleAndAddTo(Dest& dst, const Lhs& lhs, const Rhs& rhs, const Scalar& alpha) + template + struct generic_product_impl + : generic_product_impl_base> { - typedef typename nested_eval::type LhsNested; - typedef typename nested_eval::type RhsNested; - LhsNested lhsNested(lhs); - RhsNested rhsNested(rhs); - internal::sparse_time_dense_product(lhsNested, rhsNested, dst, alpha); - } -}; - -template -struct generic_product_impl - : generic_product_impl -{}; - -template -struct generic_product_impl - : generic_product_impl_base > -{ - typedef typename Product::Scalar Scalar; - - template - static void scaleAndAddTo(Dst& dst, const Lhs& lhs, const Rhs& rhs, const Scalar& alpha) + typedef typename Product::Scalar Scalar; + + template static void scaleAndAddTo(Dest &dst, const Lhs &lhs, const Rhs &rhs, const Scalar &alpha) + { + typedef typename nested_eval::type LhsNested; + typedef typename nested_eval::type RhsNested; + LhsNested lhsNested(lhs); + RhsNested rhsNested(rhs); + internal::sparse_time_dense_product(lhsNested, rhsNested, dst, alpha); + } + }; + + template + struct generic_product_impl + : generic_product_impl { - typedef typename nested_eval::type LhsNested; - typedef typename nested_eval::type RhsNested; - LhsNested lhsNested(lhs); - RhsNested rhsNested(rhs); - - // transpose everything - Transpose dstT(dst); - internal::sparse_time_dense_product(rhsNested.transpose(), lhsNested.transpose(), dstT, alpha); - } -}; - -template -struct generic_product_impl - : generic_product_impl -{}; - -template -struct sparse_dense_outer_product_evaluator -{ -protected: - typedef typename conditional::type Lhs1; - typedef typename conditional::type ActualRhs; - typedef Product ProdXprType; - - // if the actual left-hand side is a dense vector, - // then build a sparse-view so that we can seamlessly iterate over it. - typedef typename conditional::StorageKind,Sparse>::value, - Lhs1, SparseView >::type ActualLhs; - typedef typename conditional::StorageKind,Sparse>::value, - Lhs1 const&, SparseView >::type LhsArg; - - typedef evaluator LhsEval; - typedef evaluator RhsEval; - typedef typename evaluator::InnerIterator LhsIterator; - typedef typename ProdXprType::Scalar Scalar; - -public: - enum { - Flags = NeedToTranspose ? RowMajorBit : 0, - CoeffReadCost = HugeCost }; - - class InnerIterator : public LhsIterator + + template + struct generic_product_impl + : generic_product_impl_base> + { + typedef typename Product::Scalar Scalar; + + template static void scaleAndAddTo(Dst &dst, const Lhs &lhs, const Rhs &rhs, const Scalar &alpha) + { + typedef typename nested_eval::type LhsNested; + typedef typename nested_eval::type + RhsNested; + LhsNested lhsNested(lhs); + RhsNested rhsNested(rhs); + + // transpose everything + Transpose dstT(dst); + internal::sparse_time_dense_product(rhsNested.transpose(), lhsNested.transpose(), dstT, alpha); + } + }; + + template + struct generic_product_impl + : generic_product_impl + { + }; + + template struct sparse_dense_outer_product_evaluator { - public: - InnerIterator(const sparse_dense_outer_product_evaluator &xprEval, Index outer) - : LhsIterator(xprEval.m_lhsXprImpl, 0), - m_outer(outer), - m_empty(false), - m_factor(get(xprEval.m_rhsXprImpl, outer, typename internal::traits::StorageKind() )) - {} - - EIGEN_STRONG_INLINE Index outer() const { return m_outer; } - EIGEN_STRONG_INLINE Index row() const { return NeedToTranspose ? m_outer : LhsIterator::index(); } - EIGEN_STRONG_INLINE Index col() const { return NeedToTranspose ? LhsIterator::index() : m_outer; } - - EIGEN_STRONG_INLINE Scalar value() const { return LhsIterator::value() * m_factor; } - EIGEN_STRONG_INLINE operator bool() const { return LhsIterator::operator bool() && (!m_empty); } - protected: - Scalar get(const RhsEval &rhs, Index outer, Dense = Dense()) const + typedef typename conditional::type Lhs1; + typedef typename conditional::type ActualRhs; + typedef Product ProdXprType; + + // if the actual left-hand side is a dense vector, + // then build a sparse-view so that we can seamlessly iterate over it. + typedef typename conditional::StorageKind, Sparse>::value, + Lhs1, + SparseView>::type ActualLhs; + typedef typename conditional::StorageKind, Sparse>::value, + Lhs1 const &, + SparseView>::type LhsArg; + + typedef evaluator LhsEval; + typedef evaluator RhsEval; + typedef typename evaluator::InnerIterator LhsIterator; + typedef typename ProdXprType::Scalar Scalar; + + public: + enum { Flags = NeedToTranspose ? RowMajorBit : 0, CoeffReadCost = HugeCost }; + + class InnerIterator : public LhsIterator + { + public: + InnerIterator(const sparse_dense_outer_product_evaluator &xprEval, Index outer) + : LhsIterator(xprEval.m_lhsXprImpl, 0), m_outer(outer), m_empty(false), + m_factor(get(xprEval.m_rhsXprImpl, outer, typename internal::traits::StorageKind())) + {} + + EIGEN_STRONG_INLINE Index outer() const { return m_outer; } + EIGEN_STRONG_INLINE Index row() const { return NeedToTranspose ? m_outer : LhsIterator::index(); } + EIGEN_STRONG_INLINE Index col() const { return NeedToTranspose ? LhsIterator::index() : m_outer; } + + EIGEN_STRONG_INLINE Scalar value() const { return LhsIterator::value() * m_factor; } + EIGEN_STRONG_INLINE operator bool() const { return LhsIterator::operator bool() && (!m_empty); } + + protected: + Scalar get(const RhsEval &rhs, Index outer, Dense = Dense()) const { return rhs.coeff(outer); } + + Scalar get(const RhsEval &rhs, Index outer, Sparse = Sparse()) + { + typename RhsEval::InnerIterator it(rhs, outer); + if (it && it.index() == 0 && it.value() != Scalar(0)) return it.value(); + m_empty = true; + return Scalar(0); + } + + Index m_outer; + bool m_empty; + Scalar m_factor; + }; + + sparse_dense_outer_product_evaluator(const Lhs1 &lhs, const ActualRhs &rhs) + : m_lhs(lhs), m_lhsXprImpl(m_lhs), m_rhsXprImpl(rhs) { - return rhs.coeff(outer); + EIGEN_INTERNAL_CHECK_COST_VALUE(CoeffReadCost); } - - Scalar get(const RhsEval &rhs, Index outer, Sparse = Sparse()) + + // transpose case + sparse_dense_outer_product_evaluator(const ActualRhs &rhs, const Lhs1 &lhs) + : m_lhs(lhs), m_lhsXprImpl(m_lhs), m_rhsXprImpl(rhs) { - typename RhsEval::InnerIterator it(rhs, outer); - if (it && it.index()==0 && it.value()!=Scalar(0)) - return it.value(); - m_empty = true; - return Scalar(0); + EIGEN_INTERNAL_CHECK_COST_VALUE(CoeffReadCost); } - - Index m_outer; - bool m_empty; - Scalar m_factor; + + protected: + const LhsArg m_lhs; + evaluator m_lhsXprImpl; + evaluator m_rhsXprImpl; }; - - sparse_dense_outer_product_evaluator(const Lhs1 &lhs, const ActualRhs &rhs) - : m_lhs(lhs), m_lhsXprImpl(m_lhs), m_rhsXprImpl(rhs) + + // sparse * dense outer product + template + struct product_evaluator, OuterProduct, SparseShape, DenseShape> + : sparse_dense_outer_product_evaluator { - EIGEN_INTERNAL_CHECK_COST_VALUE(CoeffReadCost); - } - - // transpose case - sparse_dense_outer_product_evaluator(const ActualRhs &rhs, const Lhs1 &lhs) - : m_lhs(lhs), m_lhsXprImpl(m_lhs), m_rhsXprImpl(rhs) + typedef sparse_dense_outer_product_evaluator Base; + + typedef Product XprType; + typedef typename XprType::PlainObject PlainObject; + + explicit product_evaluator(const XprType &xpr) : Base(xpr.lhs(), xpr.rhs()) {} + }; + + template + struct product_evaluator, OuterProduct, DenseShape, SparseShape> + : sparse_dense_outer_product_evaluator { - EIGEN_INTERNAL_CHECK_COST_VALUE(CoeffReadCost); - } - -protected: - const LhsArg m_lhs; - evaluator m_lhsXprImpl; - evaluator m_rhsXprImpl; -}; - -// sparse * dense outer product -template -struct product_evaluator, OuterProduct, SparseShape, DenseShape> - : sparse_dense_outer_product_evaluator -{ - typedef sparse_dense_outer_product_evaluator Base; - - typedef Product XprType; - typedef typename XprType::PlainObject PlainObject; - - explicit product_evaluator(const XprType& xpr) - : Base(xpr.lhs(), xpr.rhs()) - {} - -}; - -template -struct product_evaluator, OuterProduct, DenseShape, SparseShape> - : sparse_dense_outer_product_evaluator -{ - typedef sparse_dense_outer_product_evaluator Base; - - typedef Product XprType; - typedef typename XprType::PlainObject PlainObject; - - explicit product_evaluator(const XprType& xpr) - : Base(xpr.lhs(), xpr.rhs()) - {} - -}; - -} // end namespace internal - -} // end namespace Eigen - -#endif // EIGEN_SPARSEDENSEPRODUCT_H + typedef sparse_dense_outer_product_evaluator Base; + + typedef Product XprType; + typedef typename XprType::PlainObject PlainObject; + + explicit product_evaluator(const XprType &xpr) : Base(xpr.lhs(), xpr.rhs()) {} + }; + +}// end namespace internal + +}// end namespace Eigen + +#endif// EIGEN_SPARSEDENSEPRODUCT_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/SparseCore/SparseDiagonalProduct.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/SparseCore/SparseDiagonalProduct.h index 941c03be..c8a27e34 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/SparseCore/SparseDiagonalProduct.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/SparseCore/SparseDiagonalProduct.h @@ -10,7 +10,7 @@ #ifndef EIGEN_SPARSE_DIAGONAL_PRODUCT_H #define EIGEN_SPARSE_DIAGONAL_PRODUCT_H -namespace Eigen { +namespace Eigen { // The product of a diagonal matrix with a sparse matrix can be easily // implemented using expression template. @@ -26,113 +26,125 @@ namespace Eigen { namespace internal { -enum { - SDP_AsScalarProduct, - SDP_AsCwiseProduct -}; - -template -struct sparse_diagonal_product_evaluator; - -template -struct product_evaluator, ProductTag, DiagonalShape, SparseShape> - : public sparse_diagonal_product_evaluator -{ - typedef Product XprType; - enum { CoeffReadCost = HugeCost, Flags = Rhs::Flags&RowMajorBit, Alignment = 0 }; // FIXME CoeffReadCost & Flags - - typedef sparse_diagonal_product_evaluator Base; - explicit product_evaluator(const XprType& xpr) : Base(xpr.rhs(), xpr.lhs().diagonal()) {} -}; - -template -struct product_evaluator, ProductTag, SparseShape, DiagonalShape> - : public sparse_diagonal_product_evaluator, Lhs::Flags&RowMajorBit?SDP_AsCwiseProduct:SDP_AsScalarProduct> -{ - typedef Product XprType; - enum { CoeffReadCost = HugeCost, Flags = Lhs::Flags&RowMajorBit, Alignment = 0 }; // FIXME CoeffReadCost & Flags - - typedef sparse_diagonal_product_evaluator, Lhs::Flags&RowMajorBit?SDP_AsCwiseProduct:SDP_AsScalarProduct> Base; - explicit product_evaluator(const XprType& xpr) : Base(xpr.lhs(), xpr.rhs().diagonal().transpose()) {} -}; - -template -struct sparse_diagonal_product_evaluator -{ -protected: - typedef typename evaluator::InnerIterator SparseXprInnerIterator; - typedef typename SparseXprType::Scalar Scalar; - -public: - class InnerIterator : public SparseXprInnerIterator + enum { SDP_AsScalarProduct, SDP_AsCwiseProduct }; + + template struct sparse_diagonal_product_evaluator; + + template + struct product_evaluator, ProductTag, DiagonalShape, SparseShape> + : public sparse_diagonal_product_evaluator + { + typedef Product XprType; + enum { CoeffReadCost = HugeCost, Flags = Rhs::Flags & RowMajorBit, Alignment = 0 };// FIXME CoeffReadCost & Flags + + typedef sparse_diagonal_product_evaluator + Base; + explicit product_evaluator(const XprType &xpr) : Base(xpr.rhs(), xpr.lhs().diagonal()) {} + }; + + template + struct product_evaluator, ProductTag, SparseShape, DiagonalShape> + : public sparse_diagonal_product_evaluator, + Lhs::Flags & RowMajorBit ? SDP_AsCwiseProduct : SDP_AsScalarProduct> { + typedef Product XprType; + enum { CoeffReadCost = HugeCost, Flags = Lhs::Flags & RowMajorBit, Alignment = 0 };// FIXME CoeffReadCost & Flags + + typedef sparse_diagonal_product_evaluator, + Lhs::Flags & RowMajorBit ? SDP_AsCwiseProduct : SDP_AsScalarProduct> + Base; + explicit product_evaluator(const XprType &xpr) : Base(xpr.lhs(), xpr.rhs().diagonal().transpose()) {} + }; + + template + struct sparse_diagonal_product_evaluator + { + protected: + typedef typename evaluator::InnerIterator SparseXprInnerIterator; + typedef typename SparseXprType::Scalar Scalar; + public: - InnerIterator(const sparse_diagonal_product_evaluator &xprEval, Index outer) - : SparseXprInnerIterator(xprEval.m_sparseXprImpl, outer), - m_coeff(xprEval.m_diagCoeffImpl.coeff(outer)) + class InnerIterator : public SparseXprInnerIterator + { + public: + InnerIterator(const sparse_diagonal_product_evaluator &xprEval, Index outer) + : SparseXprInnerIterator(xprEval.m_sparseXprImpl, outer), m_coeff(xprEval.m_diagCoeffImpl.coeff(outer)) + {} + + EIGEN_STRONG_INLINE Scalar value() const { return m_coeff * SparseXprInnerIterator::value(); } + + protected: + typename DiagonalCoeffType::Scalar m_coeff; + }; + + sparse_diagonal_product_evaluator(const SparseXprType &sparseXpr, const DiagonalCoeffType &diagCoeff) + : m_sparseXprImpl(sparseXpr), m_diagCoeffImpl(diagCoeff) {} - - EIGEN_STRONG_INLINE Scalar value() const { return m_coeff * SparseXprInnerIterator::value(); } + + Index nonZerosEstimate() const { return m_sparseXprImpl.nonZerosEstimate(); } + protected: - typename DiagonalCoeffType::Scalar m_coeff; + evaluator m_sparseXprImpl; + evaluator m_diagCoeffImpl; }; - - sparse_diagonal_product_evaluator(const SparseXprType &sparseXpr, const DiagonalCoeffType &diagCoeff) - : m_sparseXprImpl(sparseXpr), m_diagCoeffImpl(diagCoeff) - {} - - Index nonZerosEstimate() const { return m_sparseXprImpl.nonZerosEstimate(); } - -protected: - evaluator m_sparseXprImpl; - evaluator m_diagCoeffImpl; -}; - - -template -struct sparse_diagonal_product_evaluator -{ - typedef typename SparseXprType::Scalar Scalar; - typedef typename SparseXprType::StorageIndex StorageIndex; - - typedef typename nested_eval::type DiagCoeffNested; - - class InnerIterator + + + template + struct sparse_diagonal_product_evaluator { - typedef typename evaluator::InnerIterator SparseXprIter; - public: - InnerIterator(const sparse_diagonal_product_evaluator &xprEval, Index outer) - : m_sparseIter(xprEval.m_sparseXprEval, outer), m_diagCoeffNested(xprEval.m_diagCoeffNested) + typedef typename SparseXprType::Scalar Scalar; + typedef typename SparseXprType::StorageIndex StorageIndex; + + typedef typename nested_eval::type + DiagCoeffNested; + + class InnerIterator + { + typedef typename evaluator::InnerIterator SparseXprIter; + + public: + InnerIterator(const sparse_diagonal_product_evaluator &xprEval, Index outer) + : m_sparseIter(xprEval.m_sparseXprEval, outer), m_diagCoeffNested(xprEval.m_diagCoeffNested) + {} + + inline Scalar value() const { return m_sparseIter.value() * m_diagCoeffNested.coeff(index()); } + inline StorageIndex index() const { return m_sparseIter.index(); } + inline Index outer() const { return m_sparseIter.outer(); } + inline Index col() const { return SparseXprType::IsRowMajor ? m_sparseIter.index() : m_sparseIter.outer(); } + inline Index row() const { return SparseXprType::IsRowMajor ? m_sparseIter.outer() : m_sparseIter.index(); } + + EIGEN_STRONG_INLINE InnerIterator &operator++() + { + ++m_sparseIter; + return *this; + } + inline operator bool() const { return m_sparseIter; } + + protected: + SparseXprIter m_sparseIter; + DiagCoeffNested m_diagCoeffNested; + }; + + sparse_diagonal_product_evaluator(const SparseXprType &sparseXpr, const DiagCoeffType &diagCoeff) + : m_sparseXprEval(sparseXpr), m_diagCoeffNested(diagCoeff) {} - - inline Scalar value() const { return m_sparseIter.value() * m_diagCoeffNested.coeff(index()); } - inline StorageIndex index() const { return m_sparseIter.index(); } - inline Index outer() const { return m_sparseIter.outer(); } - inline Index col() const { return SparseXprType::IsRowMajor ? m_sparseIter.index() : m_sparseIter.outer(); } - inline Index row() const { return SparseXprType::IsRowMajor ? m_sparseIter.outer() : m_sparseIter.index(); } - - EIGEN_STRONG_INLINE InnerIterator& operator++() { ++m_sparseIter; return *this; } - inline operator bool() const { return m_sparseIter; } - + + Index nonZerosEstimate() const { return m_sparseXprEval.nonZerosEstimate(); } + protected: - SparseXprIter m_sparseIter; + evaluator m_sparseXprEval; DiagCoeffNested m_diagCoeffNested; }; - - sparse_diagonal_product_evaluator(const SparseXprType &sparseXpr, const DiagCoeffType &diagCoeff) - : m_sparseXprEval(sparseXpr), m_diagCoeffNested(diagCoeff) - {} - - Index nonZerosEstimate() const { return m_sparseXprEval.nonZerosEstimate(); } - -protected: - evaluator m_sparseXprEval; - DiagCoeffNested m_diagCoeffNested; -}; -} // end namespace internal +}// end namespace internal -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_SPARSE_DIAGONAL_PRODUCT_H +#endif// EIGEN_SPARSE_DIAGONAL_PRODUCT_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/SparseCore/SparseDot.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/SparseCore/SparseDot.h index 38bc4aa9..835f4658 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/SparseCore/SparseDot.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/SparseCore/SparseDot.h @@ -10,27 +10,25 @@ #ifndef EIGEN_SPARSE_DOT_H #define EIGEN_SPARSE_DOT_H -namespace Eigen { +namespace Eigen { template template -typename internal::traits::Scalar -SparseMatrixBase::dot(const MatrixBase& other) const +typename internal::traits::Scalar SparseMatrixBase::dot(const MatrixBase &other) const { EIGEN_STATIC_ASSERT_VECTOR_ONLY(Derived) EIGEN_STATIC_ASSERT_VECTOR_ONLY(OtherDerived) - EIGEN_STATIC_ASSERT_SAME_VECTOR_SIZE(Derived,OtherDerived) + EIGEN_STATIC_ASSERT_SAME_VECTOR_SIZE(Derived, OtherDerived) EIGEN_STATIC_ASSERT((internal::is_same::value), YOU_MIXED_DIFFERENT_NUMERIC_TYPES__YOU_NEED_TO_USE_THE_CAST_METHOD_OF_MATRIXBASE_TO_CAST_NUMERIC_TYPES_EXPLICITLY) eigen_assert(size() == other.size()); - eigen_assert(other.size()>0 && "you are using a non initialized vector"); + eigen_assert(other.size() > 0 && "you are using a non initialized vector"); internal::evaluator thisEval(derived()); typename internal::evaluator::InnerIterator i(thisEval, 0); Scalar res(0); - while (i) - { + while (i) { res += numext::conj(i.value()) * other.coeff(i.index()); ++i; } @@ -39,12 +37,12 @@ SparseMatrixBase::dot(const MatrixBase& other) const template template -typename internal::traits::Scalar -SparseMatrixBase::dot(const SparseMatrixBase& other) const +typename internal::traits::Scalar SparseMatrixBase::dot( + const SparseMatrixBase &other) const { EIGEN_STATIC_ASSERT_VECTOR_ONLY(Derived) EIGEN_STATIC_ASSERT_VECTOR_ONLY(OtherDerived) - EIGEN_STATIC_ASSERT_SAME_VECTOR_SIZE(Derived,OtherDerived) + EIGEN_STATIC_ASSERT_SAME_VECTOR_SIZE(Derived, OtherDerived) EIGEN_STATIC_ASSERT((internal::is_same::value), YOU_MIXED_DIFFERENT_NUMERIC_TYPES__YOU_NEED_TO_USE_THE_CAST_METHOD_OF_MATRIXBASE_TO_CAST_NUMERIC_TYPES_EXPLICITLY) @@ -52,19 +50,17 @@ SparseMatrixBase::dot(const SparseMatrixBase& other) cons internal::evaluator thisEval(derived()); typename internal::evaluator::InnerIterator i(thisEval, 0); - - internal::evaluator otherEval(other.derived()); + + internal::evaluator otherEval(other.derived()); typename internal::evaluator::InnerIterator j(otherEval, 0); Scalar res(0); - while (i && j) - { - if (i.index()==j.index()) - { + while (i && j) { + if (i.index() == j.index()) { res += numext::conj(i.value()) * j.value(); - ++i; ++j; - } - else if (i.index()::dot(const SparseMatrixBase& other) cons template inline typename NumTraits::Scalar>::Real -SparseMatrixBase::squaredNorm() const + SparseMatrixBase::squaredNorm() const { return numext::real((*this).cwiseAbs2().sum()); } template -inline typename NumTraits::Scalar>::Real -SparseMatrixBase::norm() const +inline typename NumTraits::Scalar>::Real SparseMatrixBase::norm() const { using std::sqrt; return sqrt(squaredNorm()); } template -inline typename NumTraits::Scalar>::Real -SparseMatrixBase::blueNorm() const +inline typename NumTraits::Scalar>::Real SparseMatrixBase::blueNorm() const { return internal::blueNorm_impl(*this); } -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_SPARSE_DOT_H +#endif// EIGEN_SPARSE_DOT_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/SparseCore/SparseFuzzy.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/SparseCore/SparseFuzzy.h index 7d47eb94..20dfefcf 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/SparseCore/SparseFuzzy.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/SparseCore/SparseFuzzy.h @@ -11,19 +11,19 @@ #define EIGEN_SPARSE_FUZZY_H namespace Eigen { - + template template -bool SparseMatrixBase::isApprox(const SparseMatrixBase& other, const RealScalar &prec) const +bool SparseMatrixBase::isApprox(const SparseMatrixBase &other, const RealScalar &prec) const { - const typename internal::nested_eval::type actualA(derived()); - typename internal::conditional::type, + const typename internal::nested_eval::type actualA(derived()); + typename internal::conditional::type, const PlainObject>::type actualB(other.derived()); return (actualA - actualB).squaredNorm() <= prec * prec * numext::mini(actualA.squaredNorm(), actualB.squaredNorm()); } -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_SPARSE_FUZZY_H +#endif// EIGEN_SPARSE_FUZZY_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/SparseCore/SparseMap.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/SparseCore/SparseMap.h index f99be337..ff1a9718 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/SparseCore/SparseMap.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/SparseCore/SparseMap.h @@ -14,292 +14,303 @@ namespace Eigen { namespace internal { -template -struct traits, Options, StrideType> > - : public traits > -{ - typedef SparseMatrix PlainObjectType; - typedef traits TraitsBase; - enum { - Flags = TraitsBase::Flags & (~NestByRefBit) + template + struct traits, Options, StrideType>> + : public traits> + { + typedef SparseMatrix PlainObjectType; + typedef traits TraitsBase; + enum { Flags = TraitsBase::Flags & (~NestByRefBit) }; }; -}; -template -struct traits, Options, StrideType> > - : public traits > -{ - typedef SparseMatrix PlainObjectType; - typedef traits TraitsBase; - enum { - Flags = TraitsBase::Flags & (~ (NestByRefBit | LvalueBit)) + template + struct traits, Options, StrideType>> + : public traits> + { + typedef SparseMatrix PlainObjectType; + typedef traits TraitsBase; + enum { Flags = TraitsBase::Flags & (~(NestByRefBit | LvalueBit)) }; }; -}; -} // end namespace internal +}// end namespace internal template::has_write_access ? WriteAccessors : ReadOnlyAccessors -> class SparseMapBase; + int Level = internal::accessors_level::has_write_access ? WriteAccessors : ReadOnlyAccessors> +class SparseMapBase; /** \ingroup SparseCore_Module - * class SparseMapBase - * \brief Common base class for Map and Ref instance of sparse matrix and vector. - */ -template -class SparseMapBase - : public SparseCompressedBase + * class SparseMapBase + * \brief Common base class for Map and Ref instance of sparse matrix and vector. + */ +template class SparseMapBase : public SparseCompressedBase { - public: - typedef SparseCompressedBase Base; - typedef typename Base::Scalar Scalar; - typedef typename Base::StorageIndex StorageIndex; - enum { IsRowMajor = Base::IsRowMajor }; - using Base::operator=; - protected: - - typedef typename internal::conditional< - bool(internal::is_lvalue::value), - Scalar *, const Scalar *>::type ScalarPointer; - typedef typename internal::conditional< - bool(internal::is_lvalue::value), - StorageIndex *, const StorageIndex *>::type IndexPointer; - - Index m_outerSize; - Index m_innerSize; - Array m_zero_nnz; - IndexPointer m_outerIndex; - IndexPointer m_innerIndices; - ScalarPointer m_values; - IndexPointer m_innerNonZeros; - - public: - - /** \copydoc SparseMatrixBase::rows() */ - inline Index rows() const { return IsRowMajor ? m_outerSize : m_innerSize; } - /** \copydoc SparseMatrixBase::cols() */ - inline Index cols() const { return IsRowMajor ? m_innerSize : m_outerSize; } - /** \copydoc SparseMatrixBase::innerSize() */ - inline Index innerSize() const { return m_innerSize; } - /** \copydoc SparseMatrixBase::outerSize() */ - inline Index outerSize() const { return m_outerSize; } - /** \copydoc SparseCompressedBase::nonZeros */ - inline Index nonZeros() const { return m_zero_nnz[1]; } - - /** \copydoc SparseCompressedBase::isCompressed */ - bool isCompressed() const { return m_innerNonZeros==0; } - - //---------------------------------------- - // direct access interface - /** \copydoc SparseMatrix::valuePtr */ - inline const Scalar* valuePtr() const { return m_values; } - /** \copydoc SparseMatrix::innerIndexPtr */ - inline const StorageIndex* innerIndexPtr() const { return m_innerIndices; } - /** \copydoc SparseMatrix::outerIndexPtr */ - inline const StorageIndex* outerIndexPtr() const { return m_outerIndex; } - /** \copydoc SparseMatrix::innerNonZeroPtr */ - inline const StorageIndex* innerNonZeroPtr() const { return m_innerNonZeros; } - //---------------------------------------- - - /** \copydoc SparseMatrix::coeff */ - inline Scalar coeff(Index row, Index col) const - { - const Index outer = IsRowMajor ? row : col; - const Index inner = IsRowMajor ? col : row; - - Index start = m_outerIndex[outer]; - Index end = isCompressed() ? m_outerIndex[outer+1] : start + m_innerNonZeros[outer]; - if (start==end) - return Scalar(0); - else if (end>0 && inner==m_innerIndices[end-1]) - return m_values[end-1]; - // ^^ optimization: let's first check if it is the last coefficient - // (very common in high level algorithms) - - const StorageIndex* r = std::lower_bound(&m_innerIndices[start],&m_innerIndices[end-1],inner); - const Index id = r-&m_innerIndices[0]; - return ((*r==inner) && (id(nnz)), m_outerIndex(outerIndexPtr), - m_innerIndices(innerIndexPtr), m_values(valuePtr), m_innerNonZeros(innerNonZerosPtr) - {} - - // for vectors - inline SparseMapBase(Index size, Index nnz, IndexPointer innerIndexPtr, ScalarPointer valuePtr) - : m_outerSize(1), m_innerSize(size), m_zero_nnz(0,internal::convert_index(nnz)), m_outerIndex(m_zero_nnz.data()), - m_innerIndices(innerIndexPtr), m_values(valuePtr), m_innerNonZeros(0) - {} - - /** Empty destructor */ - inline ~SparseMapBase() {} - - protected: - inline SparseMapBase() {} +public: + typedef SparseCompressedBase Base; + typedef typename Base::Scalar Scalar; + typedef typename Base::StorageIndex StorageIndex; + enum { IsRowMajor = Base::IsRowMajor }; + using Base::operator=; + +protected: + typedef typename internal::conditional::value), Scalar *, const Scalar *>::type + ScalarPointer; + typedef typename internal:: + conditional::value), StorageIndex *, const StorageIndex *>::type IndexPointer; + + Index m_outerSize; + Index m_innerSize; + Array m_zero_nnz; + IndexPointer m_outerIndex; + IndexPointer m_innerIndices; + ScalarPointer m_values; + IndexPointer m_innerNonZeros; + +public: + /** \copydoc SparseMatrixBase::rows() */ + inline Index rows() const { return IsRowMajor ? m_outerSize : m_innerSize; } + /** \copydoc SparseMatrixBase::cols() */ + inline Index cols() const { return IsRowMajor ? m_innerSize : m_outerSize; } + /** \copydoc SparseMatrixBase::innerSize() */ + inline Index innerSize() const { return m_innerSize; } + /** \copydoc SparseMatrixBase::outerSize() */ + inline Index outerSize() const { return m_outerSize; } + /** \copydoc SparseCompressedBase::nonZeros */ + inline Index nonZeros() const { return m_zero_nnz[1]; } + + /** \copydoc SparseCompressedBase::isCompressed */ + bool isCompressed() const { return m_innerNonZeros == 0; } + + //---------------------------------------- + // direct access interface + /** \copydoc SparseMatrix::valuePtr */ + inline const Scalar *valuePtr() const { return m_values; } + /** \copydoc SparseMatrix::innerIndexPtr */ + inline const StorageIndex *innerIndexPtr() const { return m_innerIndices; } + /** \copydoc SparseMatrix::outerIndexPtr */ + inline const StorageIndex *outerIndexPtr() const { return m_outerIndex; } + /** \copydoc SparseMatrix::innerNonZeroPtr */ + inline const StorageIndex *innerNonZeroPtr() const { return m_innerNonZeros; } + //---------------------------------------- + + /** \copydoc SparseMatrix::coeff */ + inline Scalar coeff(Index row, Index col) const + { + const Index outer = IsRowMajor ? row : col; + const Index inner = IsRowMajor ? col : row; + + Index start = m_outerIndex[outer]; + Index end = isCompressed() ? m_outerIndex[outer + 1] : start + m_innerNonZeros[outer]; + if (start == end) + return Scalar(0); + else if (end > 0 && inner == m_innerIndices[end - 1]) + return m_values[end - 1]; + // ^^ optimization: let's first check if it is the last coefficient + // (very common in high level algorithms) + + const StorageIndex *r = std::lower_bound(&m_innerIndices[start], &m_innerIndices[end - 1], inner); + const Index id = r - &m_innerIndices[0]; + return ((*r == inner) && (id < end)) ? m_values[id] : Scalar(0); + } + + inline SparseMapBase(Index rows, + Index cols, + Index nnz, + IndexPointer outerIndexPtr, + IndexPointer innerIndexPtr, + ScalarPointer valuePtr, + IndexPointer innerNonZerosPtr = 0) + : m_outerSize(IsRowMajor ? rows : cols), m_innerSize(IsRowMajor ? cols : rows), + m_zero_nnz(0, internal::convert_index(nnz)), m_outerIndex(outerIndexPtr), + m_innerIndices(innerIndexPtr), m_values(valuePtr), m_innerNonZeros(innerNonZerosPtr) + {} + + // for vectors + inline SparseMapBase(Index size, Index nnz, IndexPointer innerIndexPtr, ScalarPointer valuePtr) + : m_outerSize(1), m_innerSize(size), m_zero_nnz(0, internal::convert_index(nnz)), + m_outerIndex(m_zero_nnz.data()), m_innerIndices(innerIndexPtr), m_values(valuePtr), m_innerNonZeros(0) + {} + + /** Empty destructor */ + inline ~SparseMapBase() {} + +protected: + inline SparseMapBase() {} }; /** \ingroup SparseCore_Module - * class SparseMapBase - * \brief Common base class for writable Map and Ref instance of sparse matrix and vector. - */ + * class SparseMapBase + * \brief Common base class for writable Map and Ref instance of sparse matrix and vector. + */ template -class SparseMapBase - : public SparseMapBase +class SparseMapBase : public SparseMapBase { - typedef MapBase ReadOnlyMapBase; - - public: - typedef SparseMapBase Base; - typedef typename Base::Scalar Scalar; - typedef typename Base::StorageIndex StorageIndex; - enum { IsRowMajor = Base::IsRowMajor }; - - using Base::operator=; - - public: - - //---------------------------------------- - // direct access interface - using Base::valuePtr; - using Base::innerIndexPtr; - using Base::outerIndexPtr; - using Base::innerNonZeroPtr; - /** \copydoc SparseMatrix::valuePtr */ - inline Scalar* valuePtr() { return Base::m_values; } - /** \copydoc SparseMatrix::innerIndexPtr */ - inline StorageIndex* innerIndexPtr() { return Base::m_innerIndices; } - /** \copydoc SparseMatrix::outerIndexPtr */ - inline StorageIndex* outerIndexPtr() { return Base::m_outerIndex; } - /** \copydoc SparseMatrix::innerNonZeroPtr */ - inline StorageIndex* innerNonZeroPtr() { return Base::m_innerNonZeros; } - //---------------------------------------- - - /** \copydoc SparseMatrix::coeffRef */ - inline Scalar& coeffRef(Index row, Index col) - { - const Index outer = IsRowMajor ? row : col; - const Index inner = IsRowMajor ? col : row; - - Index start = Base::m_outerIndex[outer]; - Index end = Base::isCompressed() ? Base::m_outerIndex[outer+1] : start + Base::m_innerNonZeros[outer]; - eigen_assert(end>=start && "you probably called coeffRef on a non finalized matrix"); - eigen_assert(end>start && "coeffRef cannot be called on a zero coefficient"); - StorageIndex* r = std::lower_bound(&Base::m_innerIndices[start],&Base::m_innerIndices[end],inner); - const Index id = r - &Base::m_innerIndices[0]; - eigen_assert((*r==inner) && (id(Base::m_values)[id]; - } - - inline SparseMapBase(Index rows, Index cols, Index nnz, StorageIndex* outerIndexPtr, StorageIndex* innerIndexPtr, - Scalar* valuePtr, StorageIndex* innerNonZerosPtr = 0) - : Base(rows, cols, nnz, outerIndexPtr, innerIndexPtr, valuePtr, innerNonZerosPtr) - {} - - // for vectors - inline SparseMapBase(Index size, Index nnz, StorageIndex* innerIndexPtr, Scalar* valuePtr) - : Base(size, nnz, innerIndexPtr, valuePtr) - {} - - /** Empty destructor */ - inline ~SparseMapBase() {} - - protected: - inline SparseMapBase() {} + typedef MapBase ReadOnlyMapBase; + +public: + typedef SparseMapBase Base; + typedef typename Base::Scalar Scalar; + typedef typename Base::StorageIndex StorageIndex; + enum { IsRowMajor = Base::IsRowMajor }; + + using Base::operator=; + +public: + //---------------------------------------- + // direct access interface + using Base::valuePtr; + using Base::innerIndexPtr; + using Base::outerIndexPtr; + using Base::innerNonZeroPtr; + /** \copydoc SparseMatrix::valuePtr */ + inline Scalar *valuePtr() { return Base::m_values; } + /** \copydoc SparseMatrix::innerIndexPtr */ + inline StorageIndex *innerIndexPtr() { return Base::m_innerIndices; } + /** \copydoc SparseMatrix::outerIndexPtr */ + inline StorageIndex *outerIndexPtr() { return Base::m_outerIndex; } + /** \copydoc SparseMatrix::innerNonZeroPtr */ + inline StorageIndex *innerNonZeroPtr() { return Base::m_innerNonZeros; } + //---------------------------------------- + + /** \copydoc SparseMatrix::coeffRef */ + inline Scalar &coeffRef(Index row, Index col) + { + const Index outer = IsRowMajor ? row : col; + const Index inner = IsRowMajor ? col : row; + + Index start = Base::m_outerIndex[outer]; + Index end = Base::isCompressed() ? Base::m_outerIndex[outer + 1] : start + Base::m_innerNonZeros[outer]; + eigen_assert(end >= start && "you probably called coeffRef on a non finalized matrix"); + eigen_assert(end > start && "coeffRef cannot be called on a zero coefficient"); + StorageIndex *r = std::lower_bound(&Base::m_innerIndices[start], &Base::m_innerIndices[end], inner); + const Index id = r - &Base::m_innerIndices[0]; + eigen_assert((*r == inner) && (id < end) && "coeffRef cannot be called on a zero coefficient"); + return const_cast(Base::m_values)[id]; + } + + inline SparseMapBase(Index rows, + Index cols, + Index nnz, + StorageIndex *outerIndexPtr, + StorageIndex *innerIndexPtr, + Scalar *valuePtr, + StorageIndex *innerNonZerosPtr = 0) + : Base(rows, cols, nnz, outerIndexPtr, innerIndexPtr, valuePtr, innerNonZerosPtr) + {} + + // for vectors + inline SparseMapBase(Index size, Index nnz, StorageIndex *innerIndexPtr, Scalar *valuePtr) + : Base(size, nnz, innerIndexPtr, valuePtr) + {} + + /** Empty destructor */ + inline ~SparseMapBase() {} + +protected: + inline SparseMapBase() {} }; /** \ingroup SparseCore_Module - * - * \brief Specialization of class Map for SparseMatrix-like storage. - * - * \tparam SparseMatrixType the equivalent sparse matrix type of the referenced data, it must be a template instance of class SparseMatrix. - * - * \sa class Map, class SparseMatrix, class Ref - */ + * + * \brief Specialization of class Map for SparseMatrix-like storage. + * + * \tparam SparseMatrixType the equivalent sparse matrix type of the referenced data, it must be a template instance of + * class SparseMatrix. + * + * \sa class Map, class SparseMatrix, class Ref + */ #ifndef EIGEN_PARSED_BY_DOXYGEN template -class Map, Options, StrideType> - : public SparseMapBase, Options, StrideType> > +class Map, Options, StrideType> + : public SparseMapBase, Options, StrideType>> #else -template -class Map - : public SparseMapBase +template class Map : public SparseMapBase #endif { - public: - typedef SparseMapBase Base; - EIGEN_SPARSE_PUBLIC_INTERFACE(Map) - enum { IsRowMajor = Base::IsRowMajor }; - - public: - - /** Constructs a read-write Map to a sparse matrix of size \a rows x \a cols, containing \a nnz non-zero coefficients, - * stored as a sparse format as defined by the pointers \a outerIndexPtr, \a innerIndexPtr, and \a valuePtr. - * If the optional parameter \a innerNonZerosPtr is the null pointer, then a standard compressed format is assumed. - * - * This constructor is available only if \c SparseMatrixType is non-const. - * - * More details on the expected storage schemes are given in the \ref TutorialSparse "manual pages". - */ - inline Map(Index rows, Index cols, Index nnz, StorageIndex* outerIndexPtr, - StorageIndex* innerIndexPtr, Scalar* valuePtr, StorageIndex* innerNonZerosPtr = 0) - : Base(rows, cols, nnz, outerIndexPtr, innerIndexPtr, valuePtr, innerNonZerosPtr) - {} +public: + typedef SparseMapBase Base; + EIGEN_SPARSE_PUBLIC_INTERFACE(Map) + enum { IsRowMajor = Base::IsRowMajor }; + +public: + /** Constructs a read-write Map to a sparse matrix of size \a rows x \a cols, containing \a nnz non-zero coefficients, + * stored as a sparse format as defined by the pointers \a outerIndexPtr, \a innerIndexPtr, and \a valuePtr. + * If the optional parameter \a innerNonZerosPtr is the null pointer, then a standard compressed format is assumed. + * + * This constructor is available only if \c SparseMatrixType is non-const. + * + * More details on the expected storage schemes are given in the \ref TutorialSparse "manual pages". + */ + inline Map(Index rows, + Index cols, + Index nnz, + StorageIndex *outerIndexPtr, + StorageIndex *innerIndexPtr, + Scalar *valuePtr, + StorageIndex *innerNonZerosPtr = 0) + : Base(rows, cols, nnz, outerIndexPtr, innerIndexPtr, valuePtr, innerNonZerosPtr) + {} #ifndef EIGEN_PARSED_BY_DOXYGEN - /** Empty destructor */ - inline ~Map() {} + /** Empty destructor */ + inline ~Map() {} }; template -class Map, Options, StrideType> - : public SparseMapBase, Options, StrideType> > +class Map, Options, StrideType> + : public SparseMapBase, Options, StrideType>> { - public: - typedef SparseMapBase Base; - EIGEN_SPARSE_PUBLIC_INTERFACE(Map) - enum { IsRowMajor = Base::IsRowMajor }; +public: + typedef SparseMapBase Base; + EIGEN_SPARSE_PUBLIC_INTERFACE(Map) + enum { IsRowMajor = Base::IsRowMajor }; - public: +public: #endif - /** This is the const version of the above constructor. - * - * This constructor is available only if \c SparseMatrixType is const, e.g.: - * \code Map > \endcode - */ - inline Map(Index rows, Index cols, Index nnz, const StorageIndex* outerIndexPtr, - const StorageIndex* innerIndexPtr, const Scalar* valuePtr, const StorageIndex* innerNonZerosPtr = 0) - : Base(rows, cols, nnz, outerIndexPtr, innerIndexPtr, valuePtr, innerNonZerosPtr) - {} - - /** Empty destructor */ - inline ~Map() {} + /** This is the const version of the above constructor. + * + * This constructor is available only if \c SparseMatrixType is const, e.g.: + * \code Map > \endcode + */ + inline Map(Index rows, + Index cols, + Index nnz, + const StorageIndex *outerIndexPtr, + const StorageIndex *innerIndexPtr, + const Scalar *valuePtr, + const StorageIndex *innerNonZerosPtr = 0) + : Base(rows, cols, nnz, outerIndexPtr, innerIndexPtr, valuePtr, innerNonZerosPtr) + {} + + /** Empty destructor */ + inline ~Map() {} }; namespace internal { -template -struct evaluator, Options, StrideType> > - : evaluator, Options, StrideType> > > -{ - typedef evaluator, Options, StrideType> > > Base; - typedef Map, Options, StrideType> XprType; - evaluator() : Base() {} - explicit evaluator(const XprType &mat) : Base(mat) {} -}; + template + struct evaluator, Options, StrideType>> + : evaluator, Options, StrideType>>> + { + typedef evaluator, Options, StrideType>>> + Base; + typedef Map, Options, StrideType> XprType; + evaluator() : Base() {} + explicit evaluator(const XprType &mat) : Base(mat) {} + }; -template -struct evaluator, Options, StrideType> > - : evaluator, Options, StrideType> > > -{ - typedef evaluator, Options, StrideType> > > Base; - typedef Map, Options, StrideType> XprType; - evaluator() : Base() {} - explicit evaluator(const XprType &mat) : Base(mat) {} -}; + template + struct evaluator, Options, StrideType>> + : evaluator, Options, StrideType>>> + { + typedef evaluator< + SparseCompressedBase, Options, StrideType>>> + Base; + typedef Map, Options, StrideType> XprType; + evaluator() : Base() {} + explicit evaluator(const XprType &mat) : Base(mat) {} + }; -} +}// namespace internal -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_SPARSE_MAP_H +#endif// EIGEN_SPARSE_MAP_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/SparseCore/SparseMatrix.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/SparseCore/SparseMatrix.h index 0a2490bc..290a9910 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/SparseCore/SparseMatrix.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/SparseCore/SparseMatrix.h @@ -10,902 +10,827 @@ #ifndef EIGEN_SPARSEMATRIX_H #define EIGEN_SPARSEMATRIX_H -namespace Eigen { +namespace Eigen { /** \ingroup SparseCore_Module - * - * \class SparseMatrix - * - * \brief A versatible sparse matrix representation - * - * This class implements a more versatile variants of the common \em compressed row/column storage format. - * Each colmun's (resp. row) non zeros are stored as a pair of value with associated row (resp. colmiun) index. - * All the non zeros are stored in a single large buffer. Unlike the \em compressed format, there might be extra - * space inbetween the nonzeros of two successive colmuns (resp. rows) such that insertion of new non-zero - * can be done with limited memory reallocation and copies. - * - * A call to the function makeCompressed() turns the matrix into the standard \em compressed format - * compatible with many library. - * - * More details on this storage sceheme are given in the \ref TutorialSparse "manual pages". - * - * \tparam _Scalar the scalar type, i.e. the type of the coefficients - * \tparam _Options Union of bit flags controlling the storage scheme. Currently the only possibility - * is ColMajor or RowMajor. The default is 0 which means column-major. - * \tparam _StorageIndex the type of the indices. It has to be a \b signed type (e.g., short, int, std::ptrdiff_t). Default is \c int. - * - * \warning In %Eigen 3.2, the undocumented type \c SparseMatrix::Index was improperly defined as the storage index type (e.g., int), - * whereas it is now (starting from %Eigen 3.3) deprecated and always defined as Eigen::Index. - * Codes making use of \c SparseMatrix::Index, might thus likely have to be changed to use \c SparseMatrix::StorageIndex instead. - * - * This class can be extended with the help of the plugin mechanism described on the page - * \ref TopicCustomizing_Plugins by defining the preprocessor symbol \c EIGEN_SPARSEMATRIX_PLUGIN. - */ + * + * \class SparseMatrix + * + * \brief A versatible sparse matrix representation + * + * This class implements a more versatile variants of the common \em compressed row/column storage format. + * Each colmun's (resp. row) non zeros are stored as a pair of value with associated row (resp. colmiun) index. + * All the non zeros are stored in a single large buffer. Unlike the \em compressed format, there might be extra + * space inbetween the nonzeros of two successive colmuns (resp. rows) such that insertion of new non-zero + * can be done with limited memory reallocation and copies. + * + * A call to the function makeCompressed() turns the matrix into the standard \em compressed format + * compatible with many library. + * + * More details on this storage sceheme are given in the \ref TutorialSparse "manual pages". + * + * \tparam _Scalar the scalar type, i.e. the type of the coefficients + * \tparam _Options Union of bit flags controlling the storage scheme. Currently the only possibility + * is ColMajor or RowMajor. The default is 0 which means column-major. + * \tparam _StorageIndex the type of the indices. It has to be a \b signed type (e.g., short, int, std::ptrdiff_t). + * Default is \c int. + * + * \warning In %Eigen 3.2, the undocumented type \c SparseMatrix::Index was improperly defined as the storage index type + * (e.g., int), whereas it is now (starting from %Eigen 3.3) deprecated and always defined as Eigen::Index. Codes making + * use of \c SparseMatrix::Index, might thus likely have to be changed to use \c SparseMatrix::StorageIndex instead. + * + * This class can be extended with the help of the plugin mechanism described on the page + * \ref TopicCustomizing_Plugins by defining the preprocessor symbol \c EIGEN_SPARSEMATRIX_PLUGIN. + */ namespace internal { -template -struct traits > -{ - typedef _Scalar Scalar; - typedef _StorageIndex StorageIndex; - typedef Sparse StorageKind; - typedef MatrixXpr XprKind; - enum { - RowsAtCompileTime = Dynamic, - ColsAtCompileTime = Dynamic, - MaxRowsAtCompileTime = Dynamic, - MaxColsAtCompileTime = Dynamic, - Flags = _Options | NestByRefBit | LvalueBit | CompressedAccessBit, - SupportedAccessPatterns = InnerRandomAccessPattern + template + struct traits> + { + typedef _Scalar Scalar; + typedef _StorageIndex StorageIndex; + typedef Sparse StorageKind; + typedef MatrixXpr XprKind; + enum { + RowsAtCompileTime = Dynamic, + ColsAtCompileTime = Dynamic, + MaxRowsAtCompileTime = Dynamic, + MaxColsAtCompileTime = Dynamic, + Flags = _Options | NestByRefBit | LvalueBit | CompressedAccessBit, + SupportedAccessPatterns = InnerRandomAccessPattern + }; }; -}; -template -struct traits, DiagIndex> > -{ - typedef SparseMatrix<_Scalar, _Options, _StorageIndex> MatrixType; - typedef typename ref_selector::type MatrixTypeNested; - typedef typename remove_reference::type _MatrixTypeNested; - - typedef _Scalar Scalar; - typedef Dense StorageKind; - typedef _StorageIndex StorageIndex; - typedef MatrixXpr XprKind; - - enum { - RowsAtCompileTime = Dynamic, - ColsAtCompileTime = 1, - MaxRowsAtCompileTime = Dynamic, - MaxColsAtCompileTime = 1, - Flags = LvalueBit + template + struct traits, DiagIndex>> + { + typedef SparseMatrix<_Scalar, _Options, _StorageIndex> MatrixType; + typedef typename ref_selector::type MatrixTypeNested; + typedef typename remove_reference::type _MatrixTypeNested; + + typedef _Scalar Scalar; + typedef Dense StorageKind; + typedef _StorageIndex StorageIndex; + typedef MatrixXpr XprKind; + + enum { + RowsAtCompileTime = Dynamic, + ColsAtCompileTime = 1, + MaxRowsAtCompileTime = Dynamic, + MaxColsAtCompileTime = 1, + Flags = LvalueBit + }; }; -}; -template -struct traits, DiagIndex> > - : public traits, DiagIndex> > -{ - enum { - Flags = 0 + template + struct traits, DiagIndex>> + : public traits, DiagIndex>> + { + enum { Flags = 0 }; }; -}; -} // end namespace internal +}// end namespace internal template -class SparseMatrix - : public SparseCompressedBase > +class SparseMatrix : public SparseCompressedBase> { - typedef SparseCompressedBase Base; - using Base::convert_index; - friend class SparseVector<_Scalar,0,_StorageIndex>; - public: - using Base::isCompressed; - using Base::nonZeros; - EIGEN_SPARSE_PUBLIC_INTERFACE(SparseMatrix) - using Base::operator+=; - using Base::operator-=; - - typedef MappedSparseMatrix Map; - typedef Diagonal DiagonalReturnType; - typedef Diagonal ConstDiagonalReturnType; - typedef typename Base::InnerIterator InnerIterator; - typedef typename Base::ReverseInnerIterator ReverseInnerIterator; - - - using Base::IsRowMajor; - typedef internal::CompressedStorage Storage; - enum { - Options = _Options - }; + typedef SparseCompressedBase Base; + using Base::convert_index; + friend class SparseVector<_Scalar, 0, _StorageIndex>; - typedef typename Base::IndexVector IndexVector; - typedef typename Base::ScalarVector ScalarVector; - protected: - typedef SparseMatrix TransposedSparseMatrix; +public: + using Base::isCompressed; + using Base::nonZeros; + EIGEN_SPARSE_PUBLIC_INTERFACE(SparseMatrix) + using Base::operator+=; + using Base::operator-=; - Index m_outerSize; - Index m_innerSize; - StorageIndex* m_outerIndex; - StorageIndex* m_innerNonZeros; // optional, if null then the data is compressed - Storage m_data; + typedef MappedSparseMatrix Map; + typedef Diagonal DiagonalReturnType; + typedef Diagonal ConstDiagonalReturnType; + typedef typename Base::InnerIterator InnerIterator; + typedef typename Base::ReverseInnerIterator ReverseInnerIterator; - public: - - /** \returns the number of rows of the matrix */ - inline Index rows() const { return IsRowMajor ? m_outerSize : m_innerSize; } - /** \returns the number of columns of the matrix */ - inline Index cols() const { return IsRowMajor ? m_innerSize : m_outerSize; } - - /** \returns the number of rows (resp. columns) of the matrix if the storage order column major (resp. row major) */ - inline Index innerSize() const { return m_innerSize; } - /** \returns the number of columns (resp. rows) of the matrix if the storage order column major (resp. row major) */ - inline Index outerSize() const { return m_outerSize; } - - /** \returns a const pointer to the array of values. - * This function is aimed at interoperability with other libraries. - * \sa innerIndexPtr(), outerIndexPtr() */ - inline const Scalar* valuePtr() const { return m_data.valuePtr(); } - /** \returns a non-const pointer to the array of values. - * This function is aimed at interoperability with other libraries. - * \sa innerIndexPtr(), outerIndexPtr() */ - inline Scalar* valuePtr() { return m_data.valuePtr(); } - - /** \returns a const pointer to the array of inner indices. - * This function is aimed at interoperability with other libraries. - * \sa valuePtr(), outerIndexPtr() */ - inline const StorageIndex* innerIndexPtr() const { return m_data.indexPtr(); } - /** \returns a non-const pointer to the array of inner indices. - * This function is aimed at interoperability with other libraries. - * \sa valuePtr(), outerIndexPtr() */ - inline StorageIndex* innerIndexPtr() { return m_data.indexPtr(); } - - /** \returns a const pointer to the array of the starting positions of the inner vectors. - * This function is aimed at interoperability with other libraries. - * \sa valuePtr(), innerIndexPtr() */ - inline const StorageIndex* outerIndexPtr() const { return m_outerIndex; } - /** \returns a non-const pointer to the array of the starting positions of the inner vectors. - * This function is aimed at interoperability with other libraries. - * \sa valuePtr(), innerIndexPtr() */ - inline StorageIndex* outerIndexPtr() { return m_outerIndex; } - - /** \returns a const pointer to the array of the number of non zeros of the inner vectors. - * This function is aimed at interoperability with other libraries. - * \warning it returns the null pointer 0 in compressed mode */ - inline const StorageIndex* innerNonZeroPtr() const { return m_innerNonZeros; } - /** \returns a non-const pointer to the array of the number of non zeros of the inner vectors. - * This function is aimed at interoperability with other libraries. - * \warning it returns the null pointer 0 in compressed mode */ - inline StorageIndex* innerNonZeroPtr() { return m_innerNonZeros; } - - /** \internal */ - inline Storage& data() { return m_data; } - /** \internal */ - inline const Storage& data() const { return m_data; } - - /** \returns the value of the matrix at position \a i, \a j - * This function returns Scalar(0) if the element is an explicit \em zero */ - inline Scalar coeff(Index row, Index col) const - { - eigen_assert(row>=0 && row=0 && col=0 && row=0 && col=start && "you probably called coeffRef on a non finalized matrix"); - if(end<=start) - return insert(row,col); - const Index p = m_data.searchLowerIndex(start,end-1,StorageIndex(inner)); - if((p Storage; + enum { Options = _Options }; - /** \returns a reference to a novel non zero coefficient with coordinates \a row x \a col. - * The non zero coefficient must \b not already exist. - * - * If the matrix \c *this is in compressed mode, then \c *this is turned into uncompressed - * mode while reserving room for 2 x this->innerSize() non zeros if reserve(Index) has not been called earlier. - * In this case, the insertion procedure is optimized for a \e sequential insertion mode where elements are assumed to be - * inserted by increasing outer-indices. - * - * If that's not the case, then it is strongly recommended to either use a triplet-list to assemble the matrix, or to first - * call reserve(const SizesType &) to reserve the appropriate number of non-zero elements per inner vector. - * - * Assuming memory has been appropriately reserved, this function performs a sorted insertion in O(1) - * if the elements of each inner vector are inserted in increasing inner index order, and in O(nnz_j) for a random insertion. - * - */ - Scalar& insert(Index row, Index col); + typedef typename Base::IndexVector IndexVector; + typedef typename Base::ScalarVector ScalarVector; - public: +protected: + typedef SparseMatrix TransposedSparseMatrix; - /** Removes all non zeros but keep allocated memory - * - * This function does not free the currently allocated memory. To release as much as memory as possible, - * call \code mat.data().squeeze(); \endcode after resizing it. - * - * \sa resize(Index,Index), data() - */ - inline void setZero() - { - m_data.clear(); - memset(m_outerIndex, 0, (m_outerSize+1)*sizeof(StorageIndex)); - if(m_innerNonZeros) - memset(m_innerNonZeros, 0, (m_outerSize)*sizeof(StorageIndex)); - } + Index m_outerSize; + Index m_innerSize; + StorageIndex *m_outerIndex; + StorageIndex *m_innerNonZeros;// optional, if null then the data is compressed + Storage m_data; - /** Preallocates \a reserveSize non zeros. - * - * Precondition: the matrix must be in compressed mode. */ - inline void reserve(Index reserveSize) - { - eigen_assert(isCompressed() && "This function does not make sense in non compressed mode."); - m_data.reserve(reserveSize); - } - - #ifdef EIGEN_PARSED_BY_DOXYGEN - /** Preallocates \a reserveSize[\c j] non zeros for each column (resp. row) \c j. - * - * This function turns the matrix in non-compressed mode. - * - * The type \c SizesType must expose the following interface: - \code - typedef value_type; - const value_type& operator[](i) const; - \endcode - * for \c i in the [0,this->outerSize()[ range. - * Typical choices include std::vector, Eigen::VectorXi, Eigen::VectorXi::Constant, etc. - */ - template - inline void reserve(const SizesType& reserveSizes); - #else - template - inline void reserve(const SizesType& reserveSizes, const typename SizesType::value_type& enableif = - #if (!EIGEN_COMP_MSVC) || (EIGEN_COMP_MSVC>=1500) // MSVC 2005 fails to compile with this typename - typename - #endif - SizesType::value_type()) - { - EIGEN_UNUSED_VARIABLE(enableif); - reserveInnerVectors(reserveSizes); - } - #endif // EIGEN_PARSED_BY_DOXYGEN - protected: - template - inline void reserveInnerVectors(const SizesType& reserveSizes) - { - if(isCompressed()) - { - Index totalReserveSize = 0; - // turn the matrix into non-compressed mode - m_innerNonZeros = static_cast(std::malloc(m_outerSize * sizeof(StorageIndex))); - if (!m_innerNonZeros) internal::throw_std_bad_alloc(); - - // temporarily use m_innerSizes to hold the new starting points. - StorageIndex* newOuterIndex = m_innerNonZeros; - - StorageIndex count = 0; - for(Index j=0; j=0; --j) - { - StorageIndex innerNNZ = previousOuterIndex - m_outerIndex[j]; - for(Index i=innerNNZ-1; i>=0; --i) - { - m_data.index(newOuterIndex[j]+i) = m_data.index(m_outerIndex[j]+i); - m_data.value(newOuterIndex[j]+i) = m_data.value(m_outerIndex[j]+i); - } - previousOuterIndex = m_outerIndex[j]; - m_outerIndex[j] = newOuterIndex[j]; - m_innerNonZeros[j] = innerNNZ; - } - m_outerIndex[m_outerSize] = m_outerIndex[m_outerSize-1] + m_innerNonZeros[m_outerSize-1] + reserveSizes[m_outerSize-1]; - - m_data.resize(m_outerIndex[m_outerSize]); +public: + /** \returns the number of rows of the matrix */ + inline Index rows() const { return IsRowMajor ? m_outerSize : m_innerSize; } + /** \returns the number of columns of the matrix */ + inline Index cols() const { return IsRowMajor ? m_innerSize : m_outerSize; } + + /** \returns the number of rows (resp. columns) of the matrix if the storage order column major (resp. row major) */ + inline Index innerSize() const { return m_innerSize; } + /** \returns the number of columns (resp. rows) of the matrix if the storage order column major (resp. row major) */ + inline Index outerSize() const { return m_outerSize; } + + /** \returns a const pointer to the array of values. + * This function is aimed at interoperability with other libraries. + * \sa innerIndexPtr(), outerIndexPtr() */ + inline const Scalar *valuePtr() const { return m_data.valuePtr(); } + /** \returns a non-const pointer to the array of values. + * This function is aimed at interoperability with other libraries. + * \sa innerIndexPtr(), outerIndexPtr() */ + inline Scalar *valuePtr() { return m_data.valuePtr(); } + + /** \returns a const pointer to the array of inner indices. + * This function is aimed at interoperability with other libraries. + * \sa valuePtr(), outerIndexPtr() */ + inline const StorageIndex *innerIndexPtr() const { return m_data.indexPtr(); } + /** \returns a non-const pointer to the array of inner indices. + * This function is aimed at interoperability with other libraries. + * \sa valuePtr(), outerIndexPtr() */ + inline StorageIndex *innerIndexPtr() { return m_data.indexPtr(); } + + /** \returns a const pointer to the array of the starting positions of the inner vectors. + * This function is aimed at interoperability with other libraries. + * \sa valuePtr(), innerIndexPtr() */ + inline const StorageIndex *outerIndexPtr() const { return m_outerIndex; } + /** \returns a non-const pointer to the array of the starting positions of the inner vectors. + * This function is aimed at interoperability with other libraries. + * \sa valuePtr(), innerIndexPtr() */ + inline StorageIndex *outerIndexPtr() { return m_outerIndex; } + + /** \returns a const pointer to the array of the number of non zeros of the inner vectors. + * This function is aimed at interoperability with other libraries. + * \warning it returns the null pointer 0 in compressed mode */ + inline const StorageIndex *innerNonZeroPtr() const { return m_innerNonZeros; } + /** \returns a non-const pointer to the array of the number of non zeros of the inner vectors. + * This function is aimed at interoperability with other libraries. + * \warning it returns the null pointer 0 in compressed mode */ + inline StorageIndex *innerNonZeroPtr() { return m_innerNonZeros; } + + /** \internal */ + inline Storage &data() { return m_data; } + /** \internal */ + inline const Storage &data() const { return m_data; } + + /** \returns the value of the matrix at position \a i, \a j + * This function returns Scalar(0) if the element is an explicit \em zero */ + inline Scalar coeff(Index row, Index col) const + { + eigen_assert(row >= 0 && row < rows() && col >= 0 && col < cols()); + + const Index outer = IsRowMajor ? row : col; + const Index inner = IsRowMajor ? col : row; + Index end = m_innerNonZeros ? m_outerIndex[outer] + m_innerNonZeros[outer] : m_outerIndex[outer + 1]; + return m_data.atInRange(m_outerIndex[outer], end, StorageIndex(inner)); + } + + /** \returns a non-const reference to the value of the matrix at position \a i, \a j + * + * If the element does not exist then it is inserted via the insert(Index,Index) function + * which itself turns the matrix into a non compressed form if that was not the case. + * + * This is a O(log(nnz_j)) operation (binary search) plus the cost of insert(Index,Index) + * function if the element does not already exist. + */ + inline Scalar &coeffRef(Index row, Index col) + { + eigen_assert(row >= 0 && row < rows() && col >= 0 && col < cols()); + + const Index outer = IsRowMajor ? row : col; + const Index inner = IsRowMajor ? col : row; + + Index start = m_outerIndex[outer]; + Index end = m_innerNonZeros ? m_outerIndex[outer] + m_innerNonZeros[outer] : m_outerIndex[outer + 1]; + eigen_assert(end >= start && "you probably called coeffRef on a non finalized matrix"); + if (end <= start) return insert(row, col); + const Index p = m_data.searchLowerIndex(start, end - 1, StorageIndex(inner)); + if ((p < end) && (m_data.index(p) == inner)) + return m_data.value(p); + else + return insert(row, col); + } + + /** \returns a reference to a novel non zero coefficient with coordinates \a row x \a col. + * The non zero coefficient must \b not already exist. + * + * If the matrix \c *this is in compressed mode, then \c *this is turned into uncompressed + * mode while reserving room for 2 x this->innerSize() non zeros if reserve(Index) has not been called earlier. + * In this case, the insertion procedure is optimized for a \e sequential insertion mode where elements are assumed to + * be inserted by increasing outer-indices. + * + * If that's not the case, then it is strongly recommended to either use a triplet-list to assemble the matrix, or to + * first call reserve(const SizesType &) to reserve the appropriate number of non-zero elements per inner vector. + * + * Assuming memory has been appropriately reserved, this function performs a sorted insertion in O(1) + * if the elements of each inner vector are inserted in increasing inner index order, and in O(nnz_j) for a random + * insertion. + * + */ + Scalar &insert(Index row, Index col); + +public: + /** Removes all non zeros but keep allocated memory + * + * This function does not free the currently allocated memory. To release as much as memory as possible, + * call \code mat.data().squeeze(); \endcode after resizing it. + * + * \sa resize(Index,Index), data() + */ + inline void setZero() + { + m_data.clear(); + memset(m_outerIndex, 0, (m_outerSize + 1) * sizeof(StorageIndex)); + if (m_innerNonZeros) memset(m_innerNonZeros, 0, (m_outerSize) * sizeof(StorageIndex)); + } + + /** Preallocates \a reserveSize non zeros. + * + * Precondition: the matrix must be in compressed mode. */ + inline void reserve(Index reserveSize) + { + eigen_assert(isCompressed() && "This function does not make sense in non compressed mode."); + m_data.reserve(reserveSize); + } + +#ifdef EIGEN_PARSED_BY_DOXYGEN + /** Preallocates \a reserveSize[\c j] non zeros for each column (resp. row) \c j. + * + * This function turns the matrix in non-compressed mode. + * + * The type \c SizesType must expose the following interface: + \code + typedef value_type; + const value_type& operator[](i) const; + \endcode + * for \c i in the [0,this->outerSize()[ range. + * Typical choices include std::vector, Eigen::VectorXi, Eigen::VectorXi::Constant, etc. + */ + template inline void reserve(const SizesType &reserveSizes); +#else + template + inline void reserve(const SizesType &reserveSizes, + const typename SizesType::value_type &enableif = +#if (!EIGEN_COMP_MSVC) || (EIGEN_COMP_MSVC >= 1500)// MSVC 2005 fails to compile with this typename + typename +#endif + SizesType::value_type()) + { + EIGEN_UNUSED_VARIABLE(enableif); + reserveInnerVectors(reserveSizes); + } +#endif// EIGEN_PARSED_BY_DOXYGEN +protected: + template inline void reserveInnerVectors(const SizesType &reserveSizes) + { + if (isCompressed()) { + Index totalReserveSize = 0; + // turn the matrix into non-compressed mode + m_innerNonZeros = static_cast(std::malloc(m_outerSize * sizeof(StorageIndex))); + if (!m_innerNonZeros) internal::throw_std_bad_alloc(); + + // temporarily use m_innerSizes to hold the new starting points. + StorageIndex *newOuterIndex = m_innerNonZeros; + + StorageIndex count = 0; + for (Index j = 0; j < m_outerSize; ++j) { + newOuterIndex[j] = count; + count += reserveSizes[j] + (m_outerIndex[j + 1] - m_outerIndex[j]); + totalReserveSize += reserveSizes[j]; } - else - { - StorageIndex* newOuterIndex = static_cast(std::malloc((m_outerSize+1)*sizeof(StorageIndex))); - if (!newOuterIndex) internal::throw_std_bad_alloc(); - - StorageIndex count = 0; - for(Index j=0; j(reserveSizes[j], alreadyReserved); - count += toReserve + m_innerNonZeros[j]; + m_data.reserve(totalReserveSize); + StorageIndex previousOuterIndex = m_outerIndex[m_outerSize]; + for (Index j = m_outerSize - 1; j >= 0; --j) { + StorageIndex innerNNZ = previousOuterIndex - m_outerIndex[j]; + for (Index i = innerNNZ - 1; i >= 0; --i) { + m_data.index(newOuterIndex[j] + i) = m_data.index(m_outerIndex[j] + i); + m_data.value(newOuterIndex[j] + i) = m_data.value(m_outerIndex[j] + i); } - newOuterIndex[m_outerSize] = count; - - m_data.resize(count); - for(Index j=m_outerSize-1; j>=0; --j) - { - Index offset = newOuterIndex[j] - m_outerIndex[j]; - if(offset>0) - { - StorageIndex innerNNZ = m_innerNonZeros[j]; - for(Index i=innerNNZ-1; i>=0; --i) - { - m_data.index(newOuterIndex[j]+i) = m_data.index(m_outerIndex[j]+i); - m_data.value(newOuterIndex[j]+i) = m_data.value(m_outerIndex[j]+i); - } + previousOuterIndex = m_outerIndex[j]; + m_outerIndex[j] = newOuterIndex[j]; + m_innerNonZeros[j] = innerNNZ; + } + m_outerIndex[m_outerSize] = + m_outerIndex[m_outerSize - 1] + m_innerNonZeros[m_outerSize - 1] + reserveSizes[m_outerSize - 1]; + + m_data.resize(m_outerIndex[m_outerSize]); + } else { + StorageIndex *newOuterIndex = static_cast(std::malloc((m_outerSize + 1) * sizeof(StorageIndex))); + if (!newOuterIndex) internal::throw_std_bad_alloc(); + + StorageIndex count = 0; + for (Index j = 0; j < m_outerSize; ++j) { + newOuterIndex[j] = count; + StorageIndex alreadyReserved = (m_outerIndex[j + 1] - m_outerIndex[j]) - m_innerNonZeros[j]; + StorageIndex toReserve = std::max(reserveSizes[j], alreadyReserved); + count += toReserve + m_innerNonZeros[j]; + } + newOuterIndex[m_outerSize] = count; + + m_data.resize(count); + for (Index j = m_outerSize - 1; j >= 0; --j) { + Index offset = newOuterIndex[j] - m_outerIndex[j]; + if (offset > 0) { + StorageIndex innerNNZ = m_innerNonZeros[j]; + for (Index i = innerNNZ - 1; i >= 0; --i) { + m_data.index(newOuterIndex[j] + i) = m_data.index(m_outerIndex[j] + i); + m_data.value(newOuterIndex[j] + i) = m_data.value(m_outerIndex[j] + i); } } - - std::swap(m_outerIndex, newOuterIndex); - std::free(newOuterIndex); } - - } - public: - //--- low level purely coherent filling --- - - /** \internal - * \returns a reference to the non zero coefficient at position \a row, \a col assuming that: - * - the nonzero does not already exist - * - the new coefficient is the last one according to the storage order - * - * Before filling a given inner vector you must call the statVec(Index) function. - * - * After an insertion session, you should call the finalize() function. - * - * \sa insert, insertBackByOuterInner, startVec */ - inline Scalar& insertBack(Index row, Index col) - { - return insertBackByOuterInner(IsRowMajor?row:col, IsRowMajor?col:row); + std::swap(m_outerIndex, newOuterIndex); + std::free(newOuterIndex); } + } - /** \internal - * \sa insertBack, startVec */ - inline Scalar& insertBackByOuterInner(Index outer, Index inner) - { - eigen_assert(Index(m_outerIndex[outer+1]) == m_data.size() && "Invalid ordered insertion (invalid outer index)"); - eigen_assert( (m_outerIndex[outer+1]-m_outerIndex[outer]==0 || m_data.index(m_data.size()-1)(m_data.size()); - Index i = m_outerSize; - // find the last filled column - while (i>=0 && m_outerIndex[i]==0) - --i; + /** \internal + * \sa insertBack, insertBackByOuterInner */ + inline void startVec(Index outer) + { + eigen_assert( + m_outerIndex[outer] == Index(m_data.size()) && "You must call startVec for each inner vector sequentially"); + eigen_assert(m_outerIndex[outer + 1] == 0 && "You must call startVec for each inner vector sequentially"); + m_outerIndex[outer + 1] = m_outerIndex[outer]; + } + + /** \internal + * Must be called after inserting a set of non zero entries using the low level compressed API. + */ + inline void finalize() + { + if (isCompressed()) { + StorageIndex size = internal::convert_index(m_data.size()); + Index i = m_outerSize; + // find the last filled column + while (i >= 0 && m_outerIndex[i] == 0) --i; + ++i; + while (i <= m_outerSize) { + m_outerIndex[i] = size; ++i; - while (i<=m_outerSize) - { - m_outerIndex[i] = size; - ++i; - } } } + } - //--- + //--- - template - void setFromTriplets(const InputIterators& begin, const InputIterators& end); + template void setFromTriplets(const InputIterators &begin, const InputIterators &end); - template - void setFromTriplets(const InputIterators& begin, const InputIterators& end, DupFunctor dup_func); + template + void setFromTriplets(const InputIterators &begin, const InputIterators &end, DupFunctor dup_func); - void sumupDuplicates() { collapseDuplicates(internal::scalar_sum_op()); } + void sumupDuplicates() { collapseDuplicates(internal::scalar_sum_op()); } - template - void collapseDuplicates(DupFunctor dup_func = DupFunctor()); + template void collapseDuplicates(DupFunctor dup_func = DupFunctor()); - //--- - - /** \internal - * same as insert(Index,Index) except that the indices are given relative to the storage order */ - Scalar& insertByOuterInner(Index j, Index i) - { - return insert(IsRowMajor ? j : i, IsRowMajor ? i : j); - } + //--- - /** Turns the matrix into the \em compressed format. - */ - void makeCompressed() - { - if(isCompressed()) - return; - - eigen_internal_assert(m_outerIndex!=0 && m_outerSize>0); - - Index oldStart = m_outerIndex[1]; - m_outerIndex[1] = m_innerNonZeros[0]; - for(Index j=1; j0) - { - for(Index k=0; k 0); + + Index oldStart = m_outerIndex[1]; + m_outerIndex[1] = m_innerNonZeros[0]; + for (Index j = 1; j < m_outerSize; ++j) { + Index nextOldStart = m_outerIndex[j + 1]; + Index offset = oldStart - m_outerIndex[j]; + if (offset > 0) { + for (Index k = 0; k < m_innerNonZeros[j]; ++k) { + m_data.index(m_outerIndex[j] + k) = m_data.index(oldStart + k); + m_data.value(m_outerIndex[j] + k) = m_data.value(oldStart + k); } - m_outerIndex[j+1] = m_outerIndex[j] + m_innerNonZeros[j]; - oldStart = nextOldStart; } - std::free(m_innerNonZeros); - m_innerNonZeros = 0; - m_data.resize(m_outerIndex[m_outerSize]); - m_data.squeeze(); + m_outerIndex[j + 1] = m_outerIndex[j] + m_innerNonZeros[j]; + oldStart = nextOldStart; } + std::free(m_innerNonZeros); + m_innerNonZeros = 0; + m_data.resize(m_outerIndex[m_outerSize]); + m_data.squeeze(); + } - /** Turns the matrix into the uncompressed mode */ - void uncompress() - { - if(m_innerNonZeros != 0) - return; - m_innerNonZeros = static_cast(std::malloc(m_outerSize * sizeof(StorageIndex))); - for (Index i = 0; i < m_outerSize; i++) - { - m_innerNonZeros[i] = m_outerIndex[i+1] - m_outerIndex[i]; - } - } - - /** Suppresses all nonzeros which are \b much \b smaller \b than \a reference under the tolerence \a epsilon */ - void prune(const Scalar& reference, const RealScalar& epsilon = NumTraits::dummy_precision()) - { - prune(default_prunning_func(reference,epsilon)); - } - - /** Turns the matrix into compressed format, and suppresses all nonzeros which do not satisfy the predicate \a keep. - * The functor type \a KeepFunc must implement the following function: - * \code - * bool operator() (const Index& row, const Index& col, const Scalar& value) const; - * \endcode - * \sa prune(Scalar,RealScalar) - */ - template - void prune(const KeepFunc& keep = KeepFunc()) - { - // TODO optimize the uncompressed mode to avoid moving and allocating the data twice - makeCompressed(); - - StorageIndex k = 0; - for(Index j=0; j(std::malloc(m_outerSize * sizeof(StorageIndex))); + for (Index i = 0; i < m_outerSize; i++) { m_innerNonZeros[i] = m_outerIndex[i + 1] - m_outerIndex[i]; } + } + + /** Suppresses all nonzeros which are \b much \b smaller \b than \a reference under the tolerence \a epsilon */ + void prune(const Scalar &reference, const RealScalar &epsilon = NumTraits::dummy_precision()) + { + prune(default_prunning_func(reference, epsilon)); + } + + /** Turns the matrix into compressed format, and suppresses all nonzeros which do not satisfy the predicate \a keep. + * The functor type \a KeepFunc must implement the following function: + * \code + * bool operator() (const Index& row, const Index& col, const Scalar& value) const; + * \endcode + * \sa prune(Scalar,RealScalar) + */ + template void prune(const KeepFunc &keep = KeepFunc()) + { + // TODO optimize the uncompressed mode to avoid moving and allocating the data twice + makeCompressed(); + + StorageIndex k = 0; + for (Index j = 0; j < m_outerSize; ++j) { + Index previousStart = m_outerIndex[j]; + m_outerIndex[j] = k; + Index end = m_outerIndex[j + 1]; + for (Index i = previousStart; i < end; ++i) { + if (keep(IsRowMajor ? j : m_data.index(i), IsRowMajor ? m_data.index(i) : j, m_data.value(i))) { + m_data.value(k) = m_data.value(i); + m_data.index(k) = m_data.index(i); + ++k; } } - m_outerIndex[m_outerSize] = k; - m_data.resize(k,0); } + m_outerIndex[m_outerSize] = k; + m_data.resize(k, 0); + } - /** Resizes the matrix to a \a rows x \a cols matrix leaving old values untouched. - * - * If the sizes of the matrix are decreased, then the matrix is turned to \b uncompressed-mode - * and the storage of the out of bounds coefficients is kept and reserved. - * Call makeCompressed() to pack the entries and squeeze extra memory. - * - * \sa reserve(), setZero(), makeCompressed() - */ - void conservativeResize(Index rows, Index cols) - { - // No change - if (this->rows() == rows && this->cols() == cols) return; - - // If one dimension is null, then there is nothing to be preserved - if(rows==0 || cols==0) return resize(rows,cols); - - Index innerChange = IsRowMajor ? cols - this->cols() : rows - this->rows(); - Index outerChange = IsRowMajor ? rows - this->rows() : cols - this->cols(); - StorageIndex newInnerSize = convert_index(IsRowMajor ? cols : rows); - - // Deals with inner non zeros - if (m_innerNonZeros) - { - // Resize m_innerNonZeros - StorageIndex *newInnerNonZeros = static_cast(std::realloc(m_innerNonZeros, (m_outerSize + outerChange) * sizeof(StorageIndex))); - if (!newInnerNonZeros) internal::throw_std_bad_alloc(); - m_innerNonZeros = newInnerNonZeros; - - for(Index i=m_outerSize; i(std::malloc((m_outerSize+outerChange+1) * sizeof(StorageIndex))); - if (!m_innerNonZeros) internal::throw_std_bad_alloc(); - for(Index i = 0; i < m_outerSize; i++) - m_innerNonZeros[i] = m_outerIndex[i+1] - m_outerIndex[i]; - } - - // Change the m_innerNonZeros in case of a decrease of inner size - if (m_innerNonZeros && innerChange < 0) - { - for(Index i = 0; i < m_outerSize + (std::min)(outerChange, Index(0)); i++) - { - StorageIndex &n = m_innerNonZeros[i]; - StorageIndex start = m_outerIndex[i]; - while (n > 0 && m_data.index(start+n-1) >= newInnerSize) --n; - } - } - - m_innerSize = newInnerSize; - - // Re-allocate outer index structure if necessary - if (outerChange == 0) - return; - - StorageIndex *newOuterIndex = static_cast(std::realloc(m_outerIndex, (m_outerSize + outerChange + 1) * sizeof(StorageIndex))); - if (!newOuterIndex) internal::throw_std_bad_alloc(); - m_outerIndex = newOuterIndex; - if (outerChange > 0) - { - StorageIndex last = m_outerSize == 0 ? 0 : m_outerIndex[m_outerSize]; - for(Index i=m_outerSize; irows() == rows && this->cols() == cols) return; + + // If one dimension is null, then there is nothing to be preserved + if (rows == 0 || cols == 0) return resize(rows, cols); + + Index innerChange = IsRowMajor ? cols - this->cols() : rows - this->rows(); + Index outerChange = IsRowMajor ? rows - this->rows() : cols - this->cols(); + StorageIndex newInnerSize = convert_index(IsRowMajor ? cols : rows); + + // Deals with inner non zeros + if (m_innerNonZeros) { + // Resize m_innerNonZeros + StorageIndex *newInnerNonZeros = + static_cast(std::realloc(m_innerNonZeros, (m_outerSize + outerChange) * sizeof(StorageIndex))); + if (!newInnerNonZeros) internal::throw_std_bad_alloc(); + m_innerNonZeros = newInnerNonZeros; + + for (Index i = m_outerSize; i < m_outerSize + outerChange; i++) m_innerNonZeros[i] = 0; + } else if (innerChange < 0) { + // Inner size decreased: allocate a new m_innerNonZeros + m_innerNonZeros = + static_cast(std::malloc((m_outerSize + outerChange + 1) * sizeof(StorageIndex))); + if (!m_innerNonZeros) internal::throw_std_bad_alloc(); + for (Index i = 0; i < m_outerSize; i++) m_innerNonZeros[i] = m_outerIndex[i + 1] - m_outerIndex[i]; } - - /** Resizes the matrix to a \a rows x \a cols matrix and initializes it to zero. - * - * This function does not free the currently allocated memory. To release as much as memory as possible, - * call \code mat.data().squeeze(); \endcode after resizing it. - * - * \sa reserve(), setZero() - */ - void resize(Index rows, Index cols) - { - const Index outerSize = IsRowMajor ? rows : cols; - m_innerSize = IsRowMajor ? cols : rows; - m_data.clear(); - if (m_outerSize != outerSize || m_outerSize==0) - { - std::free(m_outerIndex); - m_outerIndex = static_cast(std::malloc((outerSize + 1) * sizeof(StorageIndex))); - if (!m_outerIndex) internal::throw_std_bad_alloc(); - - m_outerSize = outerSize; - } - if(m_innerNonZeros) - { - std::free(m_innerNonZeros); - m_innerNonZeros = 0; + + // Change the m_innerNonZeros in case of a decrease of inner size + if (m_innerNonZeros && innerChange < 0) { + for (Index i = 0; i < m_outerSize + (std::min)(outerChange, Index(0)); i++) { + StorageIndex &n = m_innerNonZeros[i]; + StorageIndex start = m_outerIndex[i]; + while (n > 0 && m_data.index(start + n - 1) >= newInnerSize) --n; } - memset(m_outerIndex, 0, (m_outerSize+1)*sizeof(StorageIndex)); } - /** \internal - * Resize the nonzero vector to \a size */ - void resizeNonZeros(Index size) - { - m_data.resize(size); - } + m_innerSize = newInnerSize; - /** \returns a const expression of the diagonal coefficients. */ - const ConstDiagonalReturnType diagonal() const { return ConstDiagonalReturnType(*this); } - - /** \returns a read-write expression of the diagonal coefficients. - * \warning If the diagonal entries are written, then all diagonal - * entries \b must already exist, otherwise an assertion will be raised. - */ - DiagonalReturnType diagonal() { return DiagonalReturnType(*this); } - - /** Default constructor yielding an empty \c 0 \c x \c 0 matrix */ - inline SparseMatrix() - : m_outerSize(-1), m_innerSize(0), m_outerIndex(0), m_innerNonZeros(0) - { - check_template_parameters(); - resize(0, 0); - } + // Re-allocate outer index structure if necessary + if (outerChange == 0) return; - /** Constructs a \a rows \c x \a cols empty matrix */ - inline SparseMatrix(Index rows, Index cols) - : m_outerSize(0), m_innerSize(0), m_outerIndex(0), m_innerNonZeros(0) - { - check_template_parameters(); - resize(rows, cols); + StorageIndex *newOuterIndex = + static_cast(std::realloc(m_outerIndex, (m_outerSize + outerChange + 1) * sizeof(StorageIndex))); + if (!newOuterIndex) internal::throw_std_bad_alloc(); + m_outerIndex = newOuterIndex; + if (outerChange > 0) { + StorageIndex last = m_outerSize == 0 ? 0 : m_outerIndex[m_outerSize]; + for (Index i = m_outerSize; i < m_outerSize + outerChange + 1; i++) m_outerIndex[i] = last; } + m_outerSize += outerChange; + } - /** Constructs a sparse matrix from the sparse expression \a other */ - template - inline SparseMatrix(const SparseMatrixBase& other) - : m_outerSize(0), m_innerSize(0), m_outerIndex(0), m_innerNonZeros(0) - { - EIGEN_STATIC_ASSERT((internal::is_same::value), - YOU_MIXED_DIFFERENT_NUMERIC_TYPES__YOU_NEED_TO_USE_THE_CAST_METHOD_OF_MATRIXBASE_TO_CAST_NUMERIC_TYPES_EXPLICITLY) - check_template_parameters(); - const bool needToTranspose = (Flags & RowMajorBit) != (internal::evaluator::Flags & RowMajorBit); - if (needToTranspose) - *this = other.derived(); - else - { - #ifdef EIGEN_SPARSE_CREATE_TEMPORARY_PLUGIN - EIGEN_SPARSE_CREATE_TEMPORARY_PLUGIN - #endif - internal::call_assignment_no_alias(*this, other.derived()); - } + /** Resizes the matrix to a \a rows x \a cols matrix and initializes it to zero. + * + * This function does not free the currently allocated memory. To release as much as memory as possible, + * call \code mat.data().squeeze(); \endcode after resizing it. + * + * \sa reserve(), setZero() + */ + void resize(Index rows, Index cols) + { + const Index outerSize = IsRowMajor ? rows : cols; + m_innerSize = IsRowMajor ? cols : rows; + m_data.clear(); + if (m_outerSize != outerSize || m_outerSize == 0) { + std::free(m_outerIndex); + m_outerIndex = static_cast(std::malloc((outerSize + 1) * sizeof(StorageIndex))); + if (!m_outerIndex) internal::throw_std_bad_alloc(); + + m_outerSize = outerSize; } - - /** Constructs a sparse matrix from the sparse selfadjoint view \a other */ - template - inline SparseMatrix(const SparseSelfAdjointView& other) - : m_outerSize(0), m_innerSize(0), m_outerIndex(0), m_innerNonZeros(0) - { - check_template_parameters(); - Base::operator=(other); + if (m_innerNonZeros) { + std::free(m_innerNonZeros); + m_innerNonZeros = 0; } + memset(m_outerIndex, 0, (m_outerSize + 1) * sizeof(StorageIndex)); + } - /** Copy constructor (it performs a deep copy) */ - inline SparseMatrix(const SparseMatrix& other) - : Base(), m_outerSize(0), m_innerSize(0), m_outerIndex(0), m_innerNonZeros(0) - { - check_template_parameters(); - *this = other.derived(); - } + /** \internal + * Resize the nonzero vector to \a size */ + void resizeNonZeros(Index size) { m_data.resize(size); } - /** \brief Copy constructor with in-place evaluation */ - template - SparseMatrix(const ReturnByValue& other) - : Base(), m_outerSize(0), m_innerSize(0), m_outerIndex(0), m_innerNonZeros(0) - { - check_template_parameters(); - initAssignment(other); - other.evalTo(*this); - } - - /** \brief Copy constructor with in-place evaluation */ - template - explicit SparseMatrix(const DiagonalBase& other) - : Base(), m_outerSize(0), m_innerSize(0), m_outerIndex(0), m_innerNonZeros(0) - { - check_template_parameters(); + /** \returns a const expression of the diagonal coefficients. */ + const ConstDiagonalReturnType diagonal() const { return ConstDiagonalReturnType(*this); } + + /** \returns a read-write expression of the diagonal coefficients. + * \warning If the diagonal entries are written, then all diagonal + * entries \b must already exist, otherwise an assertion will be raised. + */ + DiagonalReturnType diagonal() { return DiagonalReturnType(*this); } + + /** Default constructor yielding an empty \c 0 \c x \c 0 matrix */ + inline SparseMatrix() : m_outerSize(-1), m_innerSize(0), m_outerIndex(0), m_innerNonZeros(0) + { + check_template_parameters(); + resize(0, 0); + } + + /** Constructs a \a rows \c x \a cols empty matrix */ + inline SparseMatrix(Index rows, Index cols) : m_outerSize(0), m_innerSize(0), m_outerIndex(0), m_innerNonZeros(0) + { + check_template_parameters(); + resize(rows, cols); + } + + /** Constructs a sparse matrix from the sparse expression \a other */ + template + inline SparseMatrix(const SparseMatrixBase &other) + : m_outerSize(0), m_innerSize(0), m_outerIndex(0), m_innerNonZeros(0) + { + EIGEN_STATIC_ASSERT((internal::is_same::value), + YOU_MIXED_DIFFERENT_NUMERIC_TYPES__YOU_NEED_TO_USE_THE_CAST_METHOD_OF_MATRIXBASE_TO_CAST_NUMERIC_TYPES_EXPLICITLY) + check_template_parameters(); + const bool needToTranspose = (Flags & RowMajorBit) != (internal::evaluator::Flags & RowMajorBit); + if (needToTranspose) *this = other.derived(); + else { +#ifdef EIGEN_SPARSE_CREATE_TEMPORARY_PLUGIN + EIGEN_SPARSE_CREATE_TEMPORARY_PLUGIN +#endif + internal::call_assignment_no_alias(*this, other.derived()); } + } - /** Swaps the content of two sparse matrices of the same type. - * This is a fast operation that simply swaps the underlying pointers and parameters. */ - inline void swap(SparseMatrix& other) - { - //EIGEN_DBG_SPARSE(std::cout << "SparseMatrix:: swap\n"); - std::swap(m_outerIndex, other.m_outerIndex); - std::swap(m_innerSize, other.m_innerSize); - std::swap(m_outerSize, other.m_outerSize); - std::swap(m_innerNonZeros, other.m_innerNonZeros); - m_data.swap(other.m_data); - } + /** Constructs a sparse matrix from the sparse selfadjoint view \a other */ + template + inline SparseMatrix(const SparseSelfAdjointView &other) + : m_outerSize(0), m_innerSize(0), m_outerIndex(0), m_innerNonZeros(0) + { + check_template_parameters(); + Base::operator=(other); + } - /** Sets *this to the identity matrix. - * This function also turns the matrix into compressed mode, and drop any reserved memory. */ - inline void setIdentity() - { - eigen_assert(rows() == cols() && "ONLY FOR SQUARED MATRICES"); - this->m_data.resize(rows()); - Eigen::Map(this->m_data.indexPtr(), rows()).setLinSpaced(0, StorageIndex(rows()-1)); - Eigen::Map(this->m_data.valuePtr(), rows()).setOnes(); - Eigen::Map(this->m_outerIndex, rows()+1).setLinSpaced(0, StorageIndex(rows())); - std::free(m_innerNonZeros); - m_innerNonZeros = 0; - } - inline SparseMatrix& operator=(const SparseMatrix& other) - { - if (other.isRValue()) - { - swap(other.const_cast_derived()); - } - else if(this!=&other) - { - #ifdef EIGEN_SPARSE_CREATE_TEMPORARY_PLUGIN - EIGEN_SPARSE_CREATE_TEMPORARY_PLUGIN - #endif - initAssignment(other); - if(other.isCompressed()) - { - internal::smart_copy(other.m_outerIndex, other.m_outerIndex + m_outerSize + 1, m_outerIndex); - m_data = other.m_data; - } - else - { - Base::operator=(other); - } + /** Copy constructor (it performs a deep copy) */ + inline SparseMatrix(const SparseMatrix &other) + : Base(), m_outerSize(0), m_innerSize(0), m_outerIndex(0), m_innerNonZeros(0) + { + check_template_parameters(); + *this = other.derived(); + } + + /** \brief Copy constructor with in-place evaluation */ + template + SparseMatrix(const ReturnByValue &other) + : Base(), m_outerSize(0), m_innerSize(0), m_outerIndex(0), m_innerNonZeros(0) + { + check_template_parameters(); + initAssignment(other); + other.evalTo(*this); + } + + /** \brief Copy constructor with in-place evaluation */ + template + explicit SparseMatrix(const DiagonalBase &other) + : Base(), m_outerSize(0), m_innerSize(0), m_outerIndex(0), m_innerNonZeros(0) + { + check_template_parameters(); + *this = other.derived(); + } + + /** Swaps the content of two sparse matrices of the same type. + * This is a fast operation that simply swaps the underlying pointers and parameters. */ + inline void swap(SparseMatrix &other) + { + // EIGEN_DBG_SPARSE(std::cout << "SparseMatrix:: swap\n"); + std::swap(m_outerIndex, other.m_outerIndex); + std::swap(m_innerSize, other.m_innerSize); + std::swap(m_outerSize, other.m_outerSize); + std::swap(m_innerNonZeros, other.m_innerNonZeros); + m_data.swap(other.m_data); + } + + /** Sets *this to the identity matrix. + * This function also turns the matrix into compressed mode, and drop any reserved memory. */ + inline void setIdentity() + { + eigen_assert(rows() == cols() && "ONLY FOR SQUARED MATRICES"); + this->m_data.resize(rows()); + Eigen::Map(this->m_data.indexPtr(), rows()).setLinSpaced(0, StorageIndex(rows() - 1)); + Eigen::Map(this->m_data.valuePtr(), rows()).setOnes(); + Eigen::Map(this->m_outerIndex, rows() + 1).setLinSpaced(0, StorageIndex(rows())); + std::free(m_innerNonZeros); + m_innerNonZeros = 0; + } + inline SparseMatrix &operator=(const SparseMatrix &other) + { + if (other.isRValue()) { + swap(other.const_cast_derived()); + } else if (this != &other) { +#ifdef EIGEN_SPARSE_CREATE_TEMPORARY_PLUGIN + EIGEN_SPARSE_CREATE_TEMPORARY_PLUGIN +#endif + initAssignment(other); + if (other.isCompressed()) { + internal::smart_copy(other.m_outerIndex, other.m_outerIndex + m_outerSize + 1, m_outerIndex); + m_data = other.m_data; + } else { + Base::operator=(other); } - return *this; } + return *this; + } #ifndef EIGEN_PARSED_BY_DOXYGEN - template - inline SparseMatrix& operator=(const EigenBase& other) - { return Base::operator=(other.derived()); } -#endif // EIGEN_PARSED_BY_DOXYGEN + template inline SparseMatrix &operator=(const EigenBase &other) + { + return Base::operator=(other.derived()); + } +#endif// EIGEN_PARSED_BY_DOXYGEN - template - EIGEN_DONT_INLINE SparseMatrix& operator=(const SparseMatrixBase& other); + template + EIGEN_DONT_INLINE SparseMatrix &operator=(const SparseMatrixBase &other); - friend std::ostream & operator << (std::ostream & s, const SparseMatrix& m) - { - EIGEN_DBG_SPARSE( - s << "Nonzero entries:\n"; - if(m.isCompressed()) - { - for (Index i=0; i&>(m); - return s; - } + } s + << std::endl;); + s << static_cast &>(m); + return s; + } - /** Destructor */ - inline ~SparseMatrix() - { - std::free(m_outerIndex); - std::free(m_innerNonZeros); - } + /** Destructor */ + inline ~SparseMatrix() + { + std::free(m_outerIndex); + std::free(m_innerNonZeros); + } - /** Overloaded for performance */ - Scalar sum() const; - -# ifdef EIGEN_SPARSEMATRIX_PLUGIN -# include EIGEN_SPARSEMATRIX_PLUGIN -# endif + /** Overloaded for performance */ + Scalar sum() const; -protected: +#ifdef EIGEN_SPARSEMATRIX_PLUGIN +#include EIGEN_SPARSEMATRIX_PLUGIN +#endif - template - void initAssignment(const Other& other) - { - resize(other.rows(), other.cols()); - if(m_innerNonZeros) - { - std::free(m_innerNonZeros); - m_innerNonZeros = 0; - } +protected: + template void initAssignment(const Other &other) + { + resize(other.rows(), other.cols()); + if (m_innerNonZeros) { + std::free(m_innerNonZeros); + m_innerNonZeros = 0; } + } - /** \internal - * \sa insert(Index,Index) */ - EIGEN_DONT_INLINE Scalar& insertCompressed(Index row, Index col); + /** \internal + * \sa insert(Index,Index) */ + EIGEN_DONT_INLINE Scalar &insertCompressed(Index row, Index col); - /** \internal - * A vector object that is equal to 0 everywhere but v at the position i */ - class SingletonVector - { - StorageIndex m_index; - StorageIndex m_value; - public: - typedef StorageIndex value_type; - SingletonVector(Index i, Index v) - : m_index(convert_index(i)), m_value(convert_index(v)) - {} - - StorageIndex operator[](Index i) const { return i==m_index ? m_value : 0; } - }; + /** \internal + * A vector object that is equal to 0 everywhere but v at the position i */ + class SingletonVector + { + StorageIndex m_index; + StorageIndex m_value; - /** \internal - * \sa insert(Index,Index) */ - EIGEN_DONT_INLINE Scalar& insertUncompressed(Index row, Index col); + public: + typedef StorageIndex value_type; + SingletonVector(Index i, Index v) : m_index(convert_index(i)), m_value(convert_index(v)) {} + + StorageIndex operator[](Index i) const { return i == m_index ? m_value : 0; } + }; + + /** \internal + * \sa insert(Index,Index) */ + EIGEN_DONT_INLINE Scalar &insertUncompressed(Index row, Index col); public: - /** \internal - * \sa insert(Index,Index) */ - EIGEN_STRONG_INLINE Scalar& insertBackUncompressed(Index row, Index col) - { - const Index outer = IsRowMajor ? row : col; - const Index inner = IsRowMajor ? col : row; + /** \internal + * \sa insert(Index,Index) */ + EIGEN_STRONG_INLINE Scalar &insertBackUncompressed(Index row, Index col) + { + const Index outer = IsRowMajor ? row : col; + const Index inner = IsRowMajor ? col : row; - eigen_assert(!isCompressed()); - eigen_assert(m_innerNonZeros[outer]<=(m_outerIndex[outer+1] - m_outerIndex[outer])); + eigen_assert(!isCompressed()); + eigen_assert(m_innerNonZeros[outer] <= (m_outerIndex[outer + 1] - m_outerIndex[outer])); - Index p = m_outerIndex[outer] + m_innerNonZeros[outer]++; - m_data.index(p) = convert_index(inner); - return (m_data.value(p) = Scalar(0)); - } + Index p = m_outerIndex[outer] + m_innerNonZeros[outer]++; + m_data.index(p) = convert_index(inner); + return (m_data.value(p) = Scalar(0)); + } private: static void check_template_parameters() { - EIGEN_STATIC_ASSERT(NumTraits::IsSigned,THE_INDEX_TYPE_MUST_BE_A_SIGNED_TYPE); - EIGEN_STATIC_ASSERT((Options&(ColMajor|RowMajor))==Options,INVALID_MATRIX_TEMPLATE_PARAMETERS); + EIGEN_STATIC_ASSERT(NumTraits::IsSigned, THE_INDEX_TYPE_MUST_BE_A_SIGNED_TYPE); + EIGEN_STATIC_ASSERT((Options & (ColMajor | RowMajor)) == Options, INVALID_MATRIX_TEMPLATE_PARAMETERS); } - struct default_prunning_func { - default_prunning_func(const Scalar& ref, const RealScalar& eps) : reference(ref), epsilon(eps) {} - inline bool operator() (const Index&, const Index&, const Scalar& value) const + struct default_prunning_func + { + default_prunning_func(const Scalar &ref, const RealScalar &eps) : reference(ref), epsilon(eps) {} + inline bool operator()(const Index &, const Index &, const Scalar &value) const { return !internal::isMuchSmallerThan(value, reference, epsilon); } @@ -916,39 +841,37 @@ class SparseMatrix namespace internal { -template -void set_from_triplets(const InputIterator& begin, const InputIterator& end, SparseMatrixType& mat, DupFunctor dup_func) -{ - enum { IsRowMajor = SparseMatrixType::IsRowMajor }; - typedef typename SparseMatrixType::Scalar Scalar; - typedef typename SparseMatrixType::StorageIndex StorageIndex; - SparseMatrix trMat(mat.rows(),mat.cols()); - - if(begin!=end) + template + void + set_from_triplets(const InputIterator &begin, const InputIterator &end, SparseMatrixType &mat, DupFunctor dup_func) { - // pass 1: count the nnz per inner-vector - typename SparseMatrixType::IndexVector wi(trMat.outerSize()); - wi.setZero(); - for(InputIterator it(begin); it!=end; ++it) - { - eigen_assert(it->row()>=0 && it->row()col()>=0 && it->col()col() : it->row())++; - } + enum { IsRowMajor = SparseMatrixType::IsRowMajor }; + typedef typename SparseMatrixType::Scalar Scalar; + typedef typename SparseMatrixType::StorageIndex StorageIndex; + SparseMatrix trMat(mat.rows(), mat.cols()); + + if (begin != end) { + // pass 1: count the nnz per inner-vector + typename SparseMatrixType::IndexVector wi(trMat.outerSize()); + wi.setZero(); + for (InputIterator it(begin); it != end; ++it) { + eigen_assert(it->row() >= 0 && it->row() < mat.rows() && it->col() >= 0 && it->col() < mat.cols()); + wi(IsRowMajor ? it->col() : it->row())++; + } - // pass 2: insert all the elements into trMat - trMat.reserve(wi); - for(InputIterator it(begin); it!=end; ++it) - trMat.insertBackUncompressed(it->row(),it->col()) = it->value(); + // pass 2: insert all the elements into trMat + trMat.reserve(wi); + for (InputIterator it(begin); it != end; ++it) trMat.insertBackUncompressed(it->row(), it->col()) = it->value(); - // pass 3: - trMat.collapseDuplicates(dup_func); - } + // pass 3: + trMat.collapseDuplicates(dup_func); + } - // pass 4: transposed copy -> implicit sorting - mat = trMat; -} + // pass 4: transposed copy -> implicit sorting + mat = trMat; + } -} +}// namespace internal /** Fill the matrix \c *this with the list of \em triplets defined by the iterator range \a begin - \a end. @@ -990,31 +913,36 @@ void set_from_triplets(const InputIterator& begin, const InputIterator& end, Spa */ template template -void SparseMatrix::setFromTriplets(const InputIterators& begin, const InputIterators& end) +void SparseMatrix::setFromTriplets(const InputIterators &begin, + const InputIterators &end) { - internal::set_from_triplets >(begin, end, *this, internal::scalar_sum_op()); + internal::set_from_triplets>( + begin, end, *this, internal::scalar_sum_op()); } /** The same as setFromTriplets but when duplicates are met the functor \a dup_func is applied: - * \code - * value = dup_func(OldValue, NewValue) - * \endcode - * Here is a C++11 example keeping the latest entry only: - * \code - * mat.setFromTriplets(triplets.begin(), triplets.end(), [] (const Scalar&,const Scalar &b) { return b; }); - * \endcode - */ + * \code + * value = dup_func(OldValue, NewValue) + * \endcode + * Here is a C++11 example keeping the latest entry only: + * \code + * mat.setFromTriplets(triplets.begin(), triplets.end(), [] (const Scalar&,const Scalar &b) { return b; }); + * \endcode + */ template -template -void SparseMatrix::setFromTriplets(const InputIterators& begin, const InputIterators& end, DupFunctor dup_func) +template +void SparseMatrix::setFromTriplets(const InputIterators &begin, + const InputIterators &end, + DupFunctor dup_func) { - internal::set_from_triplets, DupFunctor>(begin, end, *this, dup_func); + internal::set_from_triplets, DupFunctor>( + begin, end, *this, dup_func); } /** \internal */ template template -void SparseMatrix::collapseDuplicates(DupFunctor dup_func) +void SparseMatrix::collapseDuplicates(DupFunctor dup_func) { eigen_assert(!isCompressed()); // TODO, in practice we should be able to use m_innerNonZeros for that task @@ -1022,20 +950,15 @@ void SparseMatrix::collapseDuplicates(DupFunctor wi.fill(-1); StorageIndex count = 0; // for each inner-vector, wi[inner_index] will hold the position of first element into the index/value buffers - for(Index j=0; j=start) - { + if (wi(i) >= start) { // we already meet this entry => accumulate it m_data.value(wi(i)) = dup_func(m_data.value(wi(i)), m_data.value(k)); - } - else - { + } else { m_data.value(count) = m_data.value(k); m_data.index(count) = m_data.index(k); wi(i) = count; @@ -1054,45 +977,45 @@ void SparseMatrix::collapseDuplicates(DupFunctor template template -EIGEN_DONT_INLINE SparseMatrix& SparseMatrix::operator=(const SparseMatrixBase& other) +EIGEN_DONT_INLINE SparseMatrix & + SparseMatrix::operator=(const SparseMatrixBase &other) { EIGEN_STATIC_ASSERT((internal::is_same::value), - YOU_MIXED_DIFFERENT_NUMERIC_TYPES__YOU_NEED_TO_USE_THE_CAST_METHOD_OF_MATRIXBASE_TO_CAST_NUMERIC_TYPES_EXPLICITLY) + YOU_MIXED_DIFFERENT_NUMERIC_TYPES__YOU_NEED_TO_USE_THE_CAST_METHOD_OF_MATRIXBASE_TO_CAST_NUMERIC_TYPES_EXPLICITLY) + +#ifdef EIGEN_SPARSE_CREATE_TEMPORARY_PLUGIN + EIGEN_SPARSE_CREATE_TEMPORARY_PLUGIN +#endif - #ifdef EIGEN_SPARSE_CREATE_TEMPORARY_PLUGIN - EIGEN_SPARSE_CREATE_TEMPORARY_PLUGIN - #endif - const bool needToTranspose = (Flags & RowMajorBit) != (internal::evaluator::Flags & RowMajorBit); - if (needToTranspose) - { - #ifdef EIGEN_SPARSE_TRANSPOSED_COPY_PLUGIN - EIGEN_SPARSE_TRANSPOSED_COPY_PLUGIN - #endif + if (needToTranspose) { +#ifdef EIGEN_SPARSE_TRANSPOSED_COPY_PLUGIN + EIGEN_SPARSE_TRANSPOSED_COPY_PLUGIN +#endif // two passes algorithm: // 1 - compute the number of coeffs per dest inner vector // 2 - do the actual copy/eval // Since each coeff of the rhs has to be evaluated twice, let's evaluate it if needed - typedef typename internal::nested_eval::type >::type OtherCopy; + typedef + typename internal::nested_eval::type>::type + OtherCopy; typedef typename internal::remove_all::type _OtherCopy; typedef internal::evaluator<_OtherCopy> OtherCopyEval; OtherCopy otherCopy(other.derived()); OtherCopyEval otherCopyEval(otherCopy); - SparseMatrix dest(other.rows(),other.cols()); - Eigen::Map (dest.m_outerIndex,dest.outerSize()).setZero(); + SparseMatrix dest(other.rows(), other.cols()); + Eigen::Map(dest.m_outerIndex, dest.outerSize()).setZero(); // pass 1 // FIXME the above copy could be merged with that pass - for (Index j=0; j& SparseMatrix& SparseMatrixswap(dest); return *this; - } - else - { - if(other.isRValue()) - { - initAssignment(other.derived()); - } + } else { + if (other.isRValue()) { initAssignment(other.derived()); } // there is no special optimization return Base::operator=(other.derived()); } } template -typename SparseMatrix<_Scalar,_Options,_StorageIndex>::Scalar& SparseMatrix<_Scalar,_Options,_StorageIndex>::insert(Index row, Index col) +typename SparseMatrix<_Scalar, _Options, _StorageIndex>::Scalar & + SparseMatrix<_Scalar, _Options, _StorageIndex>::insert(Index row, Index col) { - eigen_assert(row>=0 && row=0 && col= 0 && row < rows() && col >= 0 && col < cols()); + const Index outer = IsRowMajor ? row : col; const Index inner = IsRowMajor ? col : row; - - if(isCompressed()) - { - if(nonZeros()==0) - { + + if (isCompressed()) { + if (nonZeros() == 0) { // reserve space if not already done - if(m_data.allocatedSize()==0) - m_data.reserve(2*m_innerSize); - + if (m_data.allocatedSize() == 0) m_data.reserve(2 * m_innerSize); + // turn the matrix into non-compressed mode - m_innerNonZeros = static_cast(std::malloc(m_outerSize * sizeof(StorageIndex))); - if(!m_innerNonZeros) internal::throw_std_bad_alloc(); - - memset(m_innerNonZeros, 0, (m_outerSize)*sizeof(StorageIndex)); - + m_innerNonZeros = static_cast(std::malloc(m_outerSize * sizeof(StorageIndex))); + if (!m_innerNonZeros) internal::throw_std_bad_alloc(); + + memset(m_innerNonZeros, 0, (m_outerSize) * sizeof(StorageIndex)); + // pack all inner-vectors to the end of the pre-allocated space // and allocate the entire free-space to the first inner-vector StorageIndex end = convert_index(m_data.allocatedSize()); - for(Index j=1; j<=m_outerSize; ++j) - m_outerIndex[j] = end; - } - else - { + for (Index j = 1; j <= m_outerSize; ++j) m_outerIndex[j] = end; + } else { // turn the matrix into non-compressed mode - m_innerNonZeros = static_cast(std::malloc(m_outerSize * sizeof(StorageIndex))); - if(!m_innerNonZeros) internal::throw_std_bad_alloc(); - for(Index j=0; j(std::malloc(m_outerSize * sizeof(StorageIndex))); + if (!m_innerNonZeros) internal::throw_std_bad_alloc(); + for (Index j = 0; j < m_outerSize; ++j) m_innerNonZeros[j] = m_outerIndex[j + 1] - m_outerIndex[j]; } } - + // check whether we can do a fast "push back" insertion Index data_end = m_data.allocatedSize(); - + // First case: we are filling a new inner vector which is packed at the end. // We assume that all remaining inner-vectors are also empty and packed to the end. - if(m_outerIndex[outer]==data_end) - { - eigen_internal_assert(m_innerNonZeros[outer]==0); - + if (m_outerIndex[outer] == data_end) { + eigen_internal_assert(m_innerNonZeros[outer] == 0); + // pack previous empty inner-vectors to end of the used-space // and allocate the entire free-space to the current inner-vector. StorageIndex p = convert_index(m_data.size()); Index j = outer; - while(j>=0 && m_innerNonZeros[j]==0) - m_outerIndex[j--] = p; - + while (j >= 0 && m_innerNonZeros[j] == 0) m_outerIndex[j--] = p; + // push back the new element ++m_innerNonZeros[outer]; m_data.append(Scalar(0), inner); - + // check for reallocation - if(data_end != m_data.allocatedSize()) - { + if (data_end != m_data.allocatedSize()) { // m_data has been reallocated // -> move remaining inner-vectors back to the end of the free-space // so that the entire free-space is allocated to the current inner-vector. eigen_internal_assert(data_end < m_data.allocatedSize()); StorageIndex new_end = convert_index(m_data.allocatedSize()); - for(Index k=outer+1; k<=m_outerSize; ++k) - if(m_outerIndex[k]==data_end) - m_outerIndex[k] = new_end; + for (Index k = outer + 1; k <= m_outerSize; ++k) + if (m_outerIndex[k] == data_end) m_outerIndex[k] = new_end; } return m_data.value(p); } - + // Second case: the next inner-vector is packed to the end // and the current inner-vector end match the used-space. - if(m_outerIndex[outer+1]==data_end && m_outerIndex[outer]+m_innerNonZeros[outer]==m_data.size()) - { - eigen_internal_assert(outer+1==m_outerSize || m_innerNonZeros[outer+1]==0); - + if (m_outerIndex[outer + 1] == data_end && m_outerIndex[outer] + m_innerNonZeros[outer] == m_data.size()) { + eigen_internal_assert(outer + 1 == m_outerSize || m_innerNonZeros[outer + 1] == 0); + // add space for the new element ++m_innerNonZeros[outer]; - m_data.resize(m_data.size()+1); - + m_data.resize(m_data.size() + 1); + // check for reallocation - if(data_end != m_data.allocatedSize()) - { + if (data_end != m_data.allocatedSize()) { // m_data has been reallocated // -> move remaining inner-vectors back to the end of the free-space // so that the entire free-space is allocated to the current inner-vector. eigen_internal_assert(data_end < m_data.allocatedSize()); StorageIndex new_end = convert_index(m_data.allocatedSize()); - for(Index k=outer+1; k<=m_outerSize; ++k) - if(m_outerIndex[k]==data_end) - m_outerIndex[k] = new_end; + for (Index k = outer + 1; k <= m_outerSize; ++k) + if (m_outerIndex[k] == data_end) m_outerIndex[k] = new_end; } - + // and insert it at the right position (sorted insertion) Index startId = m_outerIndex[outer]; - Index p = m_outerIndex[outer]+m_innerNonZeros[outer]-1; - while ( (p > startId) && (m_data.index(p-1) > inner) ) - { - m_data.index(p) = m_data.index(p-1); - m_data.value(p) = m_data.value(p-1); + Index p = m_outerIndex[outer] + m_innerNonZeros[outer] - 1; + while ((p > startId) && (m_data.index(p - 1) > inner)) { + m_data.index(p) = m_data.index(p - 1); + m_data.value(p) = m_data.value(p - 1); --p; } - + m_data.index(p) = convert_index(inner); return (m_data.value(p) = 0); } - - if(m_data.size() != m_data.allocatedSize()) - { + + if (m_data.size() != m_data.allocatedSize()) { // make sure the matrix is compatible to random un-compressed insertion: m_data.resize(m_data.allocatedSize()); - this->reserveInnerVectors(Array::Constant(m_outerSize, 2)); + this->reserveInnerVectors(Array::Constant(m_outerSize, 2)); } - - return insertUncompressed(row,col); + + return insertUncompressed(row, col); } - + template -EIGEN_DONT_INLINE typename SparseMatrix<_Scalar,_Options,_StorageIndex>::Scalar& SparseMatrix<_Scalar,_Options,_StorageIndex>::insertUncompressed(Index row, Index col) +EIGEN_DONT_INLINE typename SparseMatrix<_Scalar, _Options, _StorageIndex>::Scalar & + SparseMatrix<_Scalar, _Options, _StorageIndex>::insertUncompressed(Index row, Index col) { eigen_assert(!isCompressed()); const Index outer = IsRowMajor ? row : col; const StorageIndex inner = convert_index(IsRowMajor ? col : row); - Index room = m_outerIndex[outer+1] - m_outerIndex[outer]; + Index room = m_outerIndex[outer + 1] - m_outerIndex[outer]; StorageIndex innerNNZ = m_innerNonZeros[outer]; - if(innerNNZ>=room) - { + if (innerNNZ >= room) { // this inner vector is full, we need to reallocate the whole buffer :( - reserve(SingletonVector(outer,std::max(2,innerNNZ))); + reserve(SingletonVector(outer, std::max(2, innerNNZ))); } Index startId = m_outerIndex[outer]; Index p = startId + m_innerNonZeros[outer]; - while ( (p > startId) && (m_data.index(p-1) > inner) ) - { - m_data.index(p) = m_data.index(p-1); - m_data.value(p) = m_data.value(p-1); + while ((p > startId) && (m_data.index(p - 1) > inner)) { + m_data.index(p) = m_data.index(p - 1); + m_data.value(p) = m_data.value(p - 1); --p; } - eigen_assert((p<=startId || m_data.index(p-1)!=inner) && "you cannot insert an element that already exists, you must call coeffRef to this end"); + eigen_assert((p <= startId || m_data.index(p - 1) != inner) + && "you cannot insert an element that already exists, you must call coeffRef to this end"); m_innerNonZeros[outer]++; @@ -1278,7 +1179,8 @@ EIGEN_DONT_INLINE typename SparseMatrix<_Scalar,_Options,_StorageIndex>::Scalar& } template -EIGEN_DONT_INLINE typename SparseMatrix<_Scalar,_Options,_StorageIndex>::Scalar& SparseMatrix<_Scalar,_Options,_StorageIndex>::insertCompressed(Index row, Index col) +EIGEN_DONT_INLINE typename SparseMatrix<_Scalar, _Options, _StorageIndex>::Scalar & + SparseMatrix<_Scalar, _Options, _StorageIndex>::insertCompressed(Index row, Index col) { eigen_assert(isCompressed()); @@ -1286,97 +1188,80 @@ EIGEN_DONT_INLINE typename SparseMatrix<_Scalar,_Options,_StorageIndex>::Scalar& const Index inner = IsRowMajor ? col : row; Index previousOuter = outer; - if (m_outerIndex[outer+1]==0) - { + if (m_outerIndex[outer + 1] == 0) { // we start a new inner vector - while (previousOuter>=0 && m_outerIndex[previousOuter]==0) - { + while (previousOuter >= 0 && m_outerIndex[previousOuter] == 0) { m_outerIndex[previousOuter] = convert_index(m_data.size()); --previousOuter; } - m_outerIndex[outer+1] = m_outerIndex[outer]; + m_outerIndex[outer + 1] = m_outerIndex[outer]; } // here we have to handle the tricky case where the outerIndex array // starts with: [ 0 0 0 0 0 1 ...] and we are inserted in, e.g., // the 2nd inner vector... - bool isLastVec = (!(previousOuter==-1 && m_data.size()!=0)) - && (std::size_t(m_outerIndex[outer+1]) == m_data.size()); + bool isLastVec = + (!(previousOuter == -1 && m_data.size() != 0)) && (std::size_t(m_outerIndex[outer + 1]) == m_data.size()); std::size_t startId = m_outerIndex[outer]; // FIXME let's make sure sizeof(long int) == sizeof(std::size_t) - std::size_t p = m_outerIndex[outer+1]; - ++m_outerIndex[outer+1]; + std::size_t p = m_outerIndex[outer + 1]; + ++m_outerIndex[outer + 1]; double reallocRatio = 1; - if (m_data.allocatedSize()<=m_data.size()) - { + if (m_data.allocatedSize() <= m_data.size()) { // if there is no preallocated memory, let's reserve a minimum of 32 elements - if (m_data.size()==0) - { + if (m_data.size() == 0) { m_data.reserve(32); - } - else - { + } else { // we need to reallocate the data, to reduce multiple reallocations // we use a smart resize algorithm based on the current filling ratio // in addition, we use double to avoid integers overflows - double nnzEstimate = double(m_outerIndex[outer])*double(m_outerSize)/double(outer+1); - reallocRatio = (nnzEstimate-double(m_data.size()))/double(m_data.size()); + double nnzEstimate = double(m_outerIndex[outer]) * double(m_outerSize) / double(outer + 1); + reallocRatio = (nnzEstimate - double(m_data.size())) / double(m_data.size()); // furthermore we bound the realloc ratio to: // 1) reduce multiple minor realloc when the matrix is almost filled // 2) avoid to allocate too much memory when the matrix is almost empty - reallocRatio = (std::min)((std::max)(reallocRatio,1.5),8.); + reallocRatio = (std::min)((std::max)(reallocRatio, 1.5), 8.); } } - m_data.resize(m_data.size()+1,reallocRatio); + m_data.resize(m_data.size() + 1, reallocRatio); - if (!isLastVec) - { - if (previousOuter==-1) - { + if (!isLastVec) { + if (previousOuter == -1) { // oops wrong guess. // let's correct the outer offsets - for (Index k=0; k<=(outer+1); ++k) - m_outerIndex[k] = 0; - Index k=outer+1; - while(m_outerIndex[k]==0) - m_outerIndex[k++] = 1; - while (k<=m_outerSize && m_outerIndex[k]!=0) - m_outerIndex[k++]++; + for (Index k = 0; k <= (outer + 1); ++k) m_outerIndex[k] = 0; + Index k = outer + 1; + while (m_outerIndex[k] == 0) m_outerIndex[k++] = 1; + while (k <= m_outerSize && m_outerIndex[k] != 0) m_outerIndex[k++]++; p = 0; --k; - k = m_outerIndex[k]-1; - while (k>0) - { - m_data.index(k) = m_data.index(k-1); - m_data.value(k) = m_data.value(k-1); + k = m_outerIndex[k] - 1; + while (k > 0) { + m_data.index(k) = m_data.index(k - 1); + m_data.value(k) = m_data.value(k - 1); k--; } - } - else - { + } else { // we are not inserting into the last inner vec // update outer indices: - Index j = outer+2; - while (j<=m_outerSize && m_outerIndex[j]!=0) - m_outerIndex[j++]++; + Index j = outer + 2; + while (j <= m_outerSize && m_outerIndex[j] != 0) m_outerIndex[j++]++; --j; // shift data of last vecs: - Index k = m_outerIndex[j]-1; - while (k>=Index(p)) - { - m_data.index(k) = m_data.index(k-1); - m_data.value(k) = m_data.value(k-1); + Index k = m_outerIndex[j] - 1; + while (k >= Index(p)) { + m_data.index(k) = m_data.index(k - 1); + m_data.value(k) = m_data.value(k - 1); k--; } } } - while ( (p > startId) && (m_data.index(p-1) > inner) ) - { - m_data.index(p) = m_data.index(p-1); - m_data.value(p) = m_data.value(p-1); + while ((p > startId) && (m_data.index(p - 1) > inner)) { + m_data.index(p) = m_data.index(p - 1); + m_data.value(p) = m_data.value(p - 1); --p; } @@ -1386,18 +1271,18 @@ EIGEN_DONT_INLINE typename SparseMatrix<_Scalar,_Options,_StorageIndex>::Scalar& namespace internal { -template -struct evaluator > - : evaluator > > -{ - typedef evaluator > > Base; - typedef SparseMatrix<_Scalar,_Options,_StorageIndex> SparseMatrixType; - evaluator() : Base() {} - explicit evaluator(const SparseMatrixType &mat) : Base(mat) {} -}; + template + struct evaluator> + : evaluator>> + { + typedef evaluator>> Base; + typedef SparseMatrix<_Scalar, _Options, _StorageIndex> SparseMatrixType; + evaluator() : Base() {} + explicit evaluator(const SparseMatrixType &mat) : Base(mat) {} + }; -} +}// namespace internal -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_SPARSEMATRIX_H +#endif// EIGEN_SPARSEMATRIX_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/SparseCore/SparseMatrixBase.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/SparseCore/SparseMatrixBase.h index c6b548f1..315a846c 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/SparseCore/SparseMatrixBase.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/SparseCore/SparseMatrixBase.h @@ -10,396 +10,385 @@ #ifndef EIGEN_SPARSEMATRIXBASE_H #define EIGEN_SPARSEMATRIXBASE_H -namespace Eigen { +namespace Eigen { /** \ingroup SparseCore_Module - * - * \class SparseMatrixBase - * - * \brief Base class of any sparse matrices or sparse expressions - * - * \tparam Derived is the derived type, e.g. a sparse matrix type, or an expression, etc. - * - * This class can be extended with the help of the plugin mechanism described on the page - * \ref TopicCustomizing_Plugins by defining the preprocessor symbol \c EIGEN_SPARSEMATRIXBASE_PLUGIN. - */ -template class SparseMatrixBase - : public EigenBase + * + * \class SparseMatrixBase + * + * \brief Base class of any sparse matrices or sparse expressions + * + * \tparam Derived is the derived type, e.g. a sparse matrix type, or an expression, etc. + * + * This class can be extended with the help of the plugin mechanism described on the page + * \ref TopicCustomizing_Plugins by defining the preprocessor symbol \c EIGEN_SPARSEMATRIXBASE_PLUGIN. + */ +template class SparseMatrixBase : public EigenBase { - public: - - typedef typename internal::traits::Scalar Scalar; - - /** The numeric type of the expression' coefficients, e.g. float, double, int or std::complex, etc. - * - * It is an alias for the Scalar type */ - typedef Scalar value_type; - - typedef typename internal::packet_traits::type PacketScalar; - typedef typename internal::traits::StorageKind StorageKind; - - /** The integer type used to \b store indices within a SparseMatrix. - * For a \c SparseMatrix it an alias of the third template parameter \c IndexType. */ - typedef typename internal::traits::StorageIndex StorageIndex; - - typedef typename internal::add_const_on_value_type_if_arithmetic< - typename internal::packet_traits::type - >::type PacketReturnType; - - typedef SparseMatrixBase StorageBaseType; - - typedef Matrix IndexVector; - typedef Matrix ScalarVector; - - template - Derived& operator=(const EigenBase &other); - - enum { - - RowsAtCompileTime = internal::traits::RowsAtCompileTime, - /**< The number of rows at compile-time. This is just a copy of the value provided - * by the \a Derived type. If a value is not known at compile-time, - * it is set to the \a Dynamic constant. - * \sa MatrixBase::rows(), MatrixBase::cols(), ColsAtCompileTime, SizeAtCompileTime */ - - ColsAtCompileTime = internal::traits::ColsAtCompileTime, - /**< The number of columns at compile-time. This is just a copy of the value provided - * by the \a Derived type. If a value is not known at compile-time, - * it is set to the \a Dynamic constant. - * \sa MatrixBase::rows(), MatrixBase::cols(), RowsAtCompileTime, SizeAtCompileTime */ - - - SizeAtCompileTime = (internal::size_at_compile_time::RowsAtCompileTime, - internal::traits::ColsAtCompileTime>::ret), - /**< This is equal to the number of coefficients, i.e. the number of - * rows times the number of columns, or to \a Dynamic if this is not - * known at compile-time. \sa RowsAtCompileTime, ColsAtCompileTime */ - - MaxRowsAtCompileTime = RowsAtCompileTime, - MaxColsAtCompileTime = ColsAtCompileTime, - - MaxSizeAtCompileTime = (internal::size_at_compile_time::ret), - - IsVectorAtCompileTime = RowsAtCompileTime == 1 || ColsAtCompileTime == 1, - /**< This is set to true if either the number of rows or the number of - * columns is known at compile-time to be equal to 1. Indeed, in that case, - * we are dealing with a column-vector (if there is only one column) or with - * a row-vector (if there is only one row). */ - - Flags = internal::traits::Flags, - /**< This stores expression \ref flags flags which may or may not be inherited by new expressions - * constructed from this one. See the \ref flags "list of flags". - */ - - IsRowMajor = Flags&RowMajorBit ? 1 : 0, - - InnerSizeAtCompileTime = int(IsVectorAtCompileTime) ? int(SizeAtCompileTime) - : int(IsRowMajor) ? int(ColsAtCompileTime) : int(RowsAtCompileTime), - - #ifndef EIGEN_PARSED_BY_DOXYGEN - _HasDirectAccess = (int(Flags)&DirectAccessBit) ? 1 : 0 // workaround sunCC - #endif - }; - - /** \internal the return type of MatrixBase::adjoint() */ - typedef typename internal::conditional::IsComplex, - CwiseUnaryOp, Eigen::Transpose >, - Transpose - >::type AdjointReturnType; - typedef Transpose TransposeReturnType; - typedef typename internal::add_const >::type ConstTransposeReturnType; - - // FIXME storage order do not match evaluator storage order - typedef SparseMatrix PlainObject; +public: + typedef typename internal::traits::Scalar Scalar; + + /** The numeric type of the expression' coefficients, e.g. float, double, int or std::complex, etc. + * + * It is an alias for the Scalar type */ + typedef Scalar value_type; + + typedef typename internal::packet_traits::type PacketScalar; + typedef typename internal::traits::StorageKind StorageKind; + + /** The integer type used to \b store indices within a SparseMatrix. + * For a \c SparseMatrix it an alias of the third template parameter \c IndexType. */ + typedef typename internal::traits::StorageIndex StorageIndex; + + typedef typename internal::add_const_on_value_type_if_arithmetic::type>::type + PacketReturnType; + + typedef SparseMatrixBase StorageBaseType; + + typedef Matrix IndexVector; + typedef Matrix ScalarVector; + + template Derived &operator=(const EigenBase &other); + + enum { + + RowsAtCompileTime = internal::traits::RowsAtCompileTime, + /**< The number of rows at compile-time. This is just a copy of the value provided + * by the \a Derived type. If a value is not known at compile-time, + * it is set to the \a Dynamic constant. + * \sa MatrixBase::rows(), MatrixBase::cols(), ColsAtCompileTime, SizeAtCompileTime */ + + ColsAtCompileTime = internal::traits::ColsAtCompileTime, + /**< The number of columns at compile-time. This is just a copy of the value provided + * by the \a Derived type. If a value is not known at compile-time, + * it is set to the \a Dynamic constant. + * \sa MatrixBase::rows(), MatrixBase::cols(), RowsAtCompileTime, SizeAtCompileTime */ + + + SizeAtCompileTime = (internal::size_at_compile_time::RowsAtCompileTime, + internal::traits::ColsAtCompileTime>::ret), + /**< This is equal to the number of coefficients, i.e. the number of + * rows times the number of columns, or to \a Dynamic if this is not + * known at compile-time. \sa RowsAtCompileTime, ColsAtCompileTime */ + + MaxRowsAtCompileTime = RowsAtCompileTime, + MaxColsAtCompileTime = ColsAtCompileTime, + + MaxSizeAtCompileTime = (internal::size_at_compile_time::ret), + + IsVectorAtCompileTime = RowsAtCompileTime == 1 || ColsAtCompileTime == 1, + /**< This is set to true if either the number of rows or the number of + * columns is known at compile-time to be equal to 1. Indeed, in that case, + * we are dealing with a column-vector (if there is only one column) or with + * a row-vector (if there is only one row). */ + + Flags = internal::traits::Flags, + /**< This stores expression \ref flags flags which may or may not be inherited by new expressions + * constructed from this one. See the \ref flags "list of flags". + */ + + IsRowMajor = Flags & RowMajorBit ? 1 : 0, + + InnerSizeAtCompileTime = int(IsVectorAtCompileTime) ? int(SizeAtCompileTime) + : int(IsRowMajor) ? int(ColsAtCompileTime) + : int(RowsAtCompileTime), #ifndef EIGEN_PARSED_BY_DOXYGEN - /** This is the "real scalar" type; if the \a Scalar type is already real numbers - * (e.g. int, float or double) then \a RealScalar is just the same as \a Scalar. If - * \a Scalar is \a std::complex then RealScalar is \a T. - * - * \sa class NumTraits - */ - typedef typename NumTraits::Real RealScalar; + _HasDirectAccess = (int(Flags) & DirectAccessBit) ? 1 : 0// workaround sunCC +#endif + }; - /** \internal the return type of coeff() - */ - typedef typename internal::conditional<_HasDirectAccess, const Scalar&, Scalar>::type CoeffReturnType; + /** \internal the return type of MatrixBase::adjoint() */ + typedef typename internal::conditional::IsComplex, + CwiseUnaryOp, Eigen::Transpose>, + Transpose>::type AdjointReturnType; + typedef Transpose TransposeReturnType; + typedef typename internal::add_const>::type ConstTransposeReturnType; - /** \internal Represents a matrix with all coefficients equal to one another*/ - typedef CwiseNullaryOp,Matrix > ConstantReturnType; + // FIXME storage order do not match evaluator storage order + typedef SparseMatrix PlainObject; - /** type of the equivalent dense matrix */ - typedef Matrix DenseMatrixType; - /** type of the equivalent square matrix */ - typedef Matrix SquareMatrixType; +#ifndef EIGEN_PARSED_BY_DOXYGEN + /** This is the "real scalar" type; if the \a Scalar type is already real numbers + * (e.g. int, float or double) then \a RealScalar is just the same as \a Scalar. If + * \a Scalar is \a std::complex then RealScalar is \a T. + * + * \sa class NumTraits + */ + typedef typename NumTraits::Real RealScalar; + + /** \internal the return type of coeff() + */ + typedef typename internal::conditional<_HasDirectAccess, const Scalar &, Scalar>::type CoeffReturnType; + + /** \internal Represents a matrix with all coefficients equal to one another*/ + typedef CwiseNullaryOp, Matrix> ConstantReturnType; - inline const Derived& derived() const { return *static_cast(this); } - inline Derived& derived() { return *static_cast(this); } - inline Derived& const_cast_derived() const - { return *static_cast(const_cast(this)); } + /** type of the equivalent dense matrix */ + typedef Matrix DenseMatrixType; + /** type of the equivalent square matrix */ + typedef Matrix + SquareMatrixType; - typedef EigenBase Base; + inline const Derived &derived() const { return *static_cast(this); } + inline Derived &derived() { return *static_cast(this); } + inline Derived &const_cast_derived() const { return *static_cast(const_cast(this)); } -#endif // not EIGEN_PARSED_BY_DOXYGEN + typedef EigenBase Base; + +#endif// not EIGEN_PARSED_BY_DOXYGEN #define EIGEN_CURRENT_STORAGE_BASE_CLASS Eigen::SparseMatrixBase #ifdef EIGEN_PARSED_BY_DOXYGEN -#define EIGEN_DOC_UNARY_ADDONS(METHOD,OP) /**

This method does not change the sparsity of \c *this: the OP is applied to explicitly stored coefficients only. \sa SparseCompressedBase::coeffs()

*/ -#define EIGEN_DOC_BLOCK_ADDONS_NOT_INNER_PANEL /**

\warning This method returns a read-only expression for any sparse matrices. \sa \ref TutorialSparse_SubMatrices "Sparse block operations"

*/ -#define EIGEN_DOC_BLOCK_ADDONS_INNER_PANEL_IF(COND) /**

\warning This method returns a read-write expression for COND sparse matrices only. Otherwise, the returned expression is read-only. \sa \ref TutorialSparse_SubMatrices "Sparse block operations"

*/ +#define EIGEN_DOC_UNARY_ADDONS( \ + METHOD, OP) /**

This method does not change the sparsity of \c *this: the OP is applied to explicitly stored \ + coefficients only. \sa SparseCompressedBase::coeffs()

*/ +#define EIGEN_DOC_BLOCK_ADDONS_NOT_INNER_PANEL /**

\warning This method returns a read-only expression for any \ + sparse matrices. \sa \ref TutorialSparse_SubMatrices "Sparse block \ + operations"

*/ +#define EIGEN_DOC_BLOCK_ADDONS_INNER_PANEL_IF( \ + COND) /**

\warning This method returns a read-write expression for COND sparse matrices only. Otherwise, the \ + returned expression is read-only. \sa \ref TutorialSparse_SubMatrices "Sparse block operations"

*/ #else -#define EIGEN_DOC_UNARY_ADDONS(X,Y) +#define EIGEN_DOC_UNARY_ADDONS(X, Y) #define EIGEN_DOC_BLOCK_ADDONS_NOT_INNER_PANEL #define EIGEN_DOC_BLOCK_ADDONS_INNER_PANEL_IF(COND) #endif -# include "../plugins/CommonCwiseUnaryOps.h" -# include "../plugins/CommonCwiseBinaryOps.h" -# include "../plugins/MatrixCwiseUnaryOps.h" -# include "../plugins/MatrixCwiseBinaryOps.h" -# include "../plugins/BlockMethods.h" -# ifdef EIGEN_SPARSEMATRIXBASE_PLUGIN -# include EIGEN_SPARSEMATRIXBASE_PLUGIN -# endif +#include "../plugins/BlockMethods.h" +#include "../plugins/CommonCwiseBinaryOps.h" +#include "../plugins/CommonCwiseUnaryOps.h" +#include "../plugins/MatrixCwiseBinaryOps.h" +#include "../plugins/MatrixCwiseUnaryOps.h" +#ifdef EIGEN_SPARSEMATRIXBASE_PLUGIN +#include EIGEN_SPARSEMATRIXBASE_PLUGIN +#endif #undef EIGEN_CURRENT_STORAGE_BASE_CLASS #undef EIGEN_DOC_UNARY_ADDONS #undef EIGEN_DOC_BLOCK_ADDONS_NOT_INNER_PANEL #undef EIGEN_DOC_BLOCK_ADDONS_INNER_PANEL_IF - /** \returns the number of rows. \sa cols() */ - inline Index rows() const { return derived().rows(); } - /** \returns the number of columns. \sa rows() */ - inline Index cols() const { return derived().cols(); } - /** \returns the number of coefficients, which is \a rows()*cols(). - * \sa rows(), cols(). */ - inline Index size() const { return rows() * cols(); } - /** \returns true if either the number of rows or the number of columns is equal to 1. - * In other words, this function returns - * \code rows()==1 || cols()==1 \endcode - * \sa rows(), cols(), IsVectorAtCompileTime. */ - inline bool isVector() const { return rows()==1 || cols()==1; } - /** \returns the size of the storage major dimension, - * i.e., the number of columns for a columns major matrix, and the number of rows otherwise */ - Index outerSize() const { return (int(Flags)&RowMajorBit) ? this->rows() : this->cols(); } - /** \returns the size of the inner dimension according to the storage order, - * i.e., the number of rows for a columns major matrix, and the number of cols otherwise */ - Index innerSize() const { return (int(Flags)&RowMajorBit) ? this->cols() : this->rows(); } - - bool isRValue() const { return m_isRValue; } - Derived& markAsRValue() { m_isRValue = true; return derived(); } - - SparseMatrixBase() : m_isRValue(false) { /* TODO check flags */ } - - - template - Derived& operator=(const ReturnByValue& other); - - template - inline Derived& operator=(const SparseMatrixBase& other); - - inline Derived& operator=(const Derived& other); - - protected: - - template - inline Derived& assign(const OtherDerived& other); - - template - inline void assignGeneric(const OtherDerived& other); - - public: - - friend std::ostream & operator << (std::ostream & s, const SparseMatrixBase& m) - { - typedef typename Derived::Nested Nested; - typedef typename internal::remove_all::type NestedCleaned; - - if (Flags&RowMajorBit) - { - Nested nm(m.derived()); - internal::evaluator thisEval(nm); - for (Index row=0; row::InnerIterator it(thisEval, row); it; ++it) - { - for ( ; colrows() : this->cols(); } + /** \returns the size of the inner dimension according to the storage order, + * i.e., the number of rows for a columns major matrix, and the number of cols otherwise */ + Index innerSize() const { return (int(Flags) & RowMajorBit) ? this->cols() : this->rows(); } + + bool isRValue() const { return m_isRValue; } + Derived &markAsRValue() + { + m_isRValue = true; + return derived(); + } + + SparseMatrixBase() : m_isRValue(false) { /* TODO check flags */ } + + + template Derived &operator=(const ReturnByValue &other); + + template inline Derived &operator=(const SparseMatrixBase &other); + + inline Derived &operator=(const Derived &other); + +protected: + template inline Derived &assign(const OtherDerived &other); + + template inline void assignGeneric(const OtherDerived &other); + +public: + friend std::ostream &operator<<(std::ostream &s, const SparseMatrixBase &m) + { + typedef typename Derived::Nested Nested; + typedef typename internal::remove_all::type NestedCleaned; + + if (Flags & RowMajorBit) { + Nested nm(m.derived()); + internal::evaluator thisEval(nm); + for (Index row = 0; row < nm.outerSize(); ++row) { + Index col = 0; + for (typename internal::evaluator::InnerIterator it(thisEval, row); it; ++it) { + for (; col < it.index(); ++col) s << "0 "; + s << it.value() << " "; + ++col; } + for (; col < m.cols(); ++col) s << "0 "; + s << std::endl; } - else - { - Nested nm(m.derived()); - internal::evaluator thisEval(nm); - if (m.cols() == 1) { - Index row = 0; - for (typename internal::evaluator::InnerIterator it(thisEval, 0); it; ++it) - { - for ( ; row trans = m; - s << static_cast >&>(trans); + } else { + Nested nm(m.derived()); + internal::evaluator thisEval(nm); + if (m.cols() == 1) { + Index row = 0; + for (typename internal::evaluator::InnerIterator it(thisEval, 0); it; ++it) { + for (; row < it.index(); ++row) s << "0" << std::endl; + s << it.value() << std::endl; + ++row; } + for (; row < m.rows(); ++row) s << "0" << std::endl; + } else { + SparseMatrix trans = m; + s << static_cast> &>(trans); } - return s; - } - - template - Derived& operator+=(const SparseMatrixBase& other); - template - Derived& operator-=(const SparseMatrixBase& other); - - template - Derived& operator+=(const DiagonalBase& other); - template - Derived& operator-=(const DiagonalBase& other); - - template - Derived& operator+=(const EigenBase &other); - template - Derived& operator-=(const EigenBase &other); - - Derived& operator*=(const Scalar& other); - Derived& operator/=(const Scalar& other); - - template struct CwiseProductDenseReturnType { - typedef CwiseBinaryOp::Scalar, - typename internal::traits::Scalar - >::ReturnType>, - const Derived, - const OtherDerived - > Type; - }; - - template - EIGEN_STRONG_INLINE const typename CwiseProductDenseReturnType::Type - cwiseProduct(const MatrixBase &other) const; - - // sparse * diagonal - template - const Product - operator*(const DiagonalBase &other) const - { return Product(derived(), other.derived()); } - - // diagonal * sparse - template friend - const Product - operator*(const DiagonalBase &lhs, const SparseMatrixBase& rhs) - { return Product(lhs.derived(), rhs.derived()); } - - // sparse * sparse - template - const Product - operator*(const SparseMatrixBase &other) const; - - // sparse * dense - template - const Product - operator*(const MatrixBase &other) const - { return Product(derived(), other.derived()); } - - // dense * sparse - template friend - const Product - operator*(const MatrixBase &lhs, const SparseMatrixBase& rhs) - { return Product(lhs.derived(), rhs.derived()); } - - /** \returns an expression of P H P^-1 where H is the matrix represented by \c *this */ - SparseSymmetricPermutationProduct twistedBy(const PermutationMatrix& perm) const - { - return SparseSymmetricPermutationProduct(derived(), perm); - } - - template - Derived& operator*=(const SparseMatrixBase& other); - - template - inline const TriangularView triangularView() const; - - template struct SelfAdjointViewReturnType { typedef SparseSelfAdjointView Type; }; - template struct ConstSelfAdjointViewReturnType { typedef const SparseSelfAdjointView Type; }; - - template inline - typename ConstSelfAdjointViewReturnType::Type selfadjointView() const; - template inline - typename SelfAdjointViewReturnType::Type selfadjointView(); - - template Scalar dot(const MatrixBase& other) const; - template Scalar dot(const SparseMatrixBase& other) const; - RealScalar squaredNorm() const; - RealScalar norm() const; - RealScalar blueNorm() const; - - TransposeReturnType transpose() { return TransposeReturnType(derived()); } - const ConstTransposeReturnType transpose() const { return ConstTransposeReturnType(derived()); } - const AdjointReturnType adjoint() const { return AdjointReturnType(transpose()); } - - // inner-vector - typedef Block InnerVectorReturnType; - typedef Block ConstInnerVectorReturnType; - InnerVectorReturnType innerVector(Index outer); - const ConstInnerVectorReturnType innerVector(Index outer) const; - - // set of inner-vectors - typedef Block InnerVectorsReturnType; - typedef Block ConstInnerVectorsReturnType; - InnerVectorsReturnType innerVectors(Index outerStart, Index outerSize); - const ConstInnerVectorsReturnType innerVectors(Index outerStart, Index outerSize) const; - - DenseMatrixType toDense() const - { - return DenseMatrixType(derived()); - } - - template - bool isApprox(const SparseMatrixBase& other, - const RealScalar& prec = NumTraits::dummy_precision()) const; - - template - bool isApprox(const MatrixBase& other, - const RealScalar& prec = NumTraits::dummy_precision()) const - { return toDense().isApprox(other,prec); } - - /** \returns the matrix or vector obtained by evaluating this expression. - * - * Notice that in the case of a plain matrix or vector (not an expression) this function just returns - * a const reference, in order to avoid a useless copy. - */ - inline const typename internal::eval::type eval() const - { return typename internal::eval::type(derived()); } - - Scalar sum() const; - - inline const SparseView - pruned(const Scalar& reference = Scalar(0), const RealScalar& epsilon = NumTraits::dummy_precision()) const; - - protected: - - bool m_isRValue; - - static inline StorageIndex convert_index(const Index idx) { - return internal::convert_index(idx); } - private: - template void evalTo(Dest &) const; + return s; + } + + template Derived &operator+=(const SparseMatrixBase &other); + template Derived &operator-=(const SparseMatrixBase &other); + + template Derived &operator+=(const DiagonalBase &other); + template Derived &operator-=(const DiagonalBase &other); + + template Derived &operator+=(const EigenBase &other); + template Derived &operator-=(const EigenBase &other); + + Derived &operator*=(const Scalar &other); + Derived &operator/=(const Scalar &other); + + template struct CwiseProductDenseReturnType + { + typedef CwiseBinaryOp< + internal::scalar_product_op::Scalar, + typename internal::traits::Scalar>::ReturnType>, + const Derived, + const OtherDerived> + Type; + }; + + template + EIGEN_STRONG_INLINE const typename CwiseProductDenseReturnType::Type cwiseProduct( + const MatrixBase &other) const; + + // sparse * diagonal + template + const Product operator*(const DiagonalBase &other) const + { + return Product(derived(), other.derived()); + } + + // diagonal * sparse + template + friend const Product operator*(const DiagonalBase &lhs, + const SparseMatrixBase &rhs) + { + return Product(lhs.derived(), rhs.derived()); + } + + // sparse * sparse + template + const Product operator*(const SparseMatrixBase &other) const; + + // sparse * dense + template + const Product operator*(const MatrixBase &other) const + { + return Product(derived(), other.derived()); + } + + // dense * sparse + template + friend const Product operator*(const MatrixBase &lhs, + const SparseMatrixBase &rhs) + { + return Product(lhs.derived(), rhs.derived()); + } + + /** \returns an expression of P H P^-1 where H is the matrix represented by \c *this */ + SparseSymmetricPermutationProduct twistedBy( + const PermutationMatrix &perm) const + { + return SparseSymmetricPermutationProduct(derived(), perm); + } + + template Derived &operator*=(const SparseMatrixBase &other); + + template inline const TriangularView triangularView() const; + + template struct SelfAdjointViewReturnType + { + typedef SparseSelfAdjointView Type; + }; + template struct ConstSelfAdjointViewReturnType + { + typedef const SparseSelfAdjointView Type; + }; + + template inline typename ConstSelfAdjointViewReturnType::Type selfadjointView() const; + template inline typename SelfAdjointViewReturnType::Type selfadjointView(); + + template Scalar dot(const MatrixBase &other) const; + template Scalar dot(const SparseMatrixBase &other) const; + RealScalar squaredNorm() const; + RealScalar norm() const; + RealScalar blueNorm() const; + + TransposeReturnType transpose() { return TransposeReturnType(derived()); } + const ConstTransposeReturnType transpose() const { return ConstTransposeReturnType(derived()); } + const AdjointReturnType adjoint() const { return AdjointReturnType(transpose()); } + + // inner-vector + typedef Block InnerVectorReturnType; + typedef Block ConstInnerVectorReturnType; + InnerVectorReturnType innerVector(Index outer); + const ConstInnerVectorReturnType innerVector(Index outer) const; + + // set of inner-vectors + typedef Block InnerVectorsReturnType; + typedef Block ConstInnerVectorsReturnType; + InnerVectorsReturnType innerVectors(Index outerStart, Index outerSize); + const ConstInnerVectorsReturnType innerVectors(Index outerStart, Index outerSize) const; + + DenseMatrixType toDense() const { return DenseMatrixType(derived()); } + + template + bool isApprox(const SparseMatrixBase &other, + const RealScalar &prec = NumTraits::dummy_precision()) const; + + template + bool isApprox(const MatrixBase &other, + const RealScalar &prec = NumTraits::dummy_precision()) const + { + return toDense().isApprox(other, prec); + } + + /** \returns the matrix or vector obtained by evaluating this expression. + * + * Notice that in the case of a plain matrix or vector (not an expression) this function just returns + * a const reference, in order to avoid a useless copy. + */ + inline const typename internal::eval::type eval() const + { + return typename internal::eval::type(derived()); + } + + Scalar sum() const; + + inline const SparseView pruned(const Scalar &reference = Scalar(0), + const RealScalar &epsilon = NumTraits::dummy_precision()) const; + +protected: + bool m_isRValue; + + static inline StorageIndex convert_index(const Index idx) { return internal::convert_index(idx); } + +private: + template void evalTo(Dest &) const; }; -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_SPARSEMATRIXBASE_H +#endif// EIGEN_SPARSEMATRIXBASE_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/SparseCore/SparsePermutation.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/SparseCore/SparsePermutation.h index ef38357a..85196c49 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/SparseCore/SparsePermutation.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/SparseCore/SparsePermutation.h @@ -12,13 +12,13 @@ // This file implements sparse * permutation products -namespace Eigen { +namespace Eigen { namespace internal { -template -struct permutation_matrix_product -{ + template + struct permutation_matrix_product + { typedef typename nested_eval::type MatrixType; typedef typename remove_all::type MatrixTypeCleaned; @@ -26,153 +26,156 @@ struct permutation_matrix_product typedef typename MatrixTypeCleaned::StorageIndex StorageIndex; enum { - SrcStorageOrder = MatrixTypeCleaned::Flags&RowMajorBit ? RowMajor : ColMajor, - MoveOuter = SrcStorageOrder==RowMajor ? Side==OnTheLeft : Side==OnTheRight + SrcStorageOrder = MatrixTypeCleaned::Flags & RowMajorBit ? RowMajor : ColMajor, + MoveOuter = SrcStorageOrder == RowMajor ? Side == OnTheLeft : Side == OnTheRight }; - + typedef typename internal::conditional, - SparseMatrix >::type ReturnType; + SparseMatrix, + SparseMatrix>::type ReturnType; - template - static inline void run(Dest& dst, const PermutationType& perm, const ExpressionType& xpr) + template + static inline void run(Dest &dst, const PermutationType &perm, const ExpressionType &xpr) { MatrixType mat(xpr); - if(MoveOuter) - { - SparseMatrix tmp(mat.rows(), mat.cols()); - Matrix sizes(mat.outerSize()); - for(Index j=0; j tmp(mat.rows(), mat.cols()); + Matrix sizes(mat.outerSize()); + for (Index j = 0; j < mat.outerSize(); ++j) { Index jp = perm.indices().coeff(j); - sizes[((Side==OnTheLeft) ^ Transposed) ? jp : j] = StorageIndex(mat.innerVector(((Side==OnTheRight) ^ Transposed) ? jp : j).nonZeros()); + sizes[((Side == OnTheLeft) ^ Transposed) ? jp : j] = + StorageIndex(mat.innerVector(((Side == OnTheRight) ^ Transposed) ? jp : j).nonZeros()); } tmp.reserve(sizes); - for(Index j=0; j tmp(mat.rows(), mat.cols()); - Matrix sizes(tmp.outerSize()); + } else { + SparseMatrix tmp( + mat.rows(), mat.cols()); + Matrix sizes(tmp.outerSize()); sizes.setZero(); - PermutationMatrix perm_cpy; - if((Side==OnTheLeft) ^ Transposed) + PermutationMatrix perm_cpy; + if ((Side == OnTheLeft) ^ Transposed) perm_cpy = perm; else perm_cpy = perm.transpose(); - for(Index j=0; j struct product_promote_storage_type { typedef Sparse ret; }; -template struct product_promote_storage_type { typedef Sparse ret; }; - -// TODO, the following two overloads are only needed to define the right temporary type through -// typename traits >::ReturnType -// whereas it should be correctly handled by traits >::PlainObject - -template -struct product_evaluator, ProductTag, PermutationShape, SparseShape> - : public evaluator::ReturnType> -{ - typedef Product XprType; - typedef typename permutation_matrix_product::ReturnType PlainObject; - typedef evaluator Base; - - enum { - Flags = Base::Flags | EvalBeforeNestingBit + template struct product_promote_storage_type + { + typedef Sparse ret; + }; + template struct product_promote_storage_type + { + typedef Sparse ret; }; - explicit product_evaluator(const XprType& xpr) - : m_result(xpr.rows(), xpr.cols()) + // TODO, the following two overloads are only needed to define the right temporary type through + // typename traits >::ReturnType + // whereas it should be correctly handled by traits >::PlainObject + + template + struct product_evaluator, ProductTag, PermutationShape, SparseShape> + : public evaluator::ReturnType> { - ::new (static_cast(this)) Base(m_result); - generic_product_impl::evalTo(m_result, xpr.lhs(), xpr.rhs()); - } + typedef Product XprType; + typedef typename permutation_matrix_product::ReturnType PlainObject; + typedef evaluator Base; -protected: - PlainObject m_result; -}; + enum { Flags = Base::Flags | EvalBeforeNestingBit }; -template -struct product_evaluator, ProductTag, SparseShape, PermutationShape > - : public evaluator::ReturnType> -{ - typedef Product XprType; - typedef typename permutation_matrix_product::ReturnType PlainObject; - typedef evaluator Base; + explicit product_evaluator(const XprType &xpr) : m_result(xpr.rows(), xpr.cols()) + { + ::new (static_cast(this)) Base(m_result); + generic_product_impl::evalTo(m_result, xpr.lhs(), xpr.rhs()); + } - enum { - Flags = Base::Flags | EvalBeforeNestingBit + protected: + PlainObject m_result; }; - explicit product_evaluator(const XprType& xpr) - : m_result(xpr.rows(), xpr.cols()) + template + struct product_evaluator, ProductTag, SparseShape, PermutationShape> + : public evaluator::ReturnType> { - ::new (static_cast(this)) Base(m_result); - generic_product_impl::evalTo(m_result, xpr.lhs(), xpr.rhs()); - } + typedef Product XprType; + typedef typename permutation_matrix_product::ReturnType PlainObject; + typedef evaluator Base; -protected: - PlainObject m_result; -}; + enum { Flags = Base::Flags | EvalBeforeNestingBit }; -} // end namespace internal + explicit product_evaluator(const XprType &xpr) : m_result(xpr.rows(), xpr.cols()) + { + ::new (static_cast(this)) Base(m_result); + generic_product_impl::evalTo(m_result, xpr.lhs(), xpr.rhs()); + } + + protected: + PlainObject m_result; + }; + +}// end namespace internal /** \returns the matrix with the permutation applied to the columns - */ + */ template inline const Product -operator*(const SparseMatrixBase& matrix, const PermutationBase& perm) -{ return Product(matrix.derived(), perm.derived()); } + operator*(const SparseMatrixBase &matrix, const PermutationBase &perm) +{ + return Product(matrix.derived(), perm.derived()); +} /** \returns the matrix with the permutation applied to the rows - */ + */ template -inline const Product -operator*( const PermutationBase& perm, const SparseMatrixBase& matrix) -{ return Product(perm.derived(), matrix.derived()); } +inline const Product operator*(const PermutationBase &perm, + const SparseMatrixBase &matrix) +{ + return Product(perm.derived(), matrix.derived()); +} /** \returns the matrix with the inverse permutation applied to the columns. - */ + */ template -inline const Product, AliasFreeProduct> -operator*(const SparseMatrixBase& matrix, const InverseImpl& tperm) +inline const Product, AliasFreeProduct> operator*( + const SparseMatrixBase &matrix, + const InverseImpl &tperm) { return Product, AliasFreeProduct>(matrix.derived(), tperm.derived()); } /** \returns the matrix with the inverse permutation applied to the rows. - */ + */ template -inline const Product, SparseDerived, AliasFreeProduct> -operator*(const InverseImpl& tperm, const SparseMatrixBase& matrix) +inline const Product, SparseDerived, AliasFreeProduct> operator*( + const InverseImpl &tperm, + const SparseMatrixBase &matrix) { return Product, SparseDerived, AliasFreeProduct>(tperm.derived(), matrix.derived()); } -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_SPARSE_SELFADJOINTVIEW_H +#endif// EIGEN_SPARSE_SELFADJOINTVIEW_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/SparseCore/SparseProduct.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/SparseCore/SparseProduct.h index 4cbf6878..ad8a755d 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/SparseCore/SparseProduct.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/SparseCore/SparseProduct.h @@ -10,160 +10,179 @@ #ifndef EIGEN_SPARSEPRODUCT_H #define EIGEN_SPARSEPRODUCT_H -namespace Eigen { +namespace Eigen { /** \returns an expression of the product of two sparse matrices. - * By default a conservative product preserving the symbolic non zeros is performed. - * The automatic pruning of the small values can be achieved by calling the pruned() function - * in which case a totally different product algorithm is employed: - * \code - * C = (A*B).pruned(); // supress numerical zeros (exact) - * C = (A*B).pruned(ref); - * C = (A*B).pruned(ref,epsilon); - * \endcode - * where \c ref is a meaningful non zero reference value. - * */ + * By default a conservative product preserving the symbolic non zeros is performed. + * The automatic pruning of the small values can be achieved by calling the pruned() function + * in which case a totally different product algorithm is employed: + * \code + * C = (A*B).pruned(); // supress numerical zeros (exact) + * C = (A*B).pruned(ref); + * C = (A*B).pruned(ref,epsilon); + * \endcode + * where \c ref is a meaningful non zero reference value. + * */ template template -inline const Product -SparseMatrixBase::operator*(const SparseMatrixBase &other) const +inline const Product SparseMatrixBase::operator*( + const SparseMatrixBase &other) const { - return Product(derived(), other.derived()); + return Product(derived(), other.derived()); } namespace internal { -// sparse * sparse -template -struct generic_product_impl -{ - template - static void evalTo(Dest& dst, const Lhs& lhs, const Rhs& rhs) - { - evalTo(dst, lhs, rhs, typename evaluator_traits::Shape()); - } - - // dense += sparse * sparse - template - static void addTo(Dest& dst, const ActualLhs& lhs, const Rhs& rhs, typename enable_if::Shape,DenseShape>::value,int*>::type* = 0) + // sparse * sparse + template + struct generic_product_impl { - typedef typename nested_eval::type LhsNested; - typedef typename nested_eval::type RhsNested; - LhsNested lhsNested(lhs); - RhsNested rhsNested(rhs); - internal::sparse_sparse_to_dense_product_selector::type, - typename remove_all::type, Dest>::run(lhsNested,rhsNested,dst); - } - - // dense -= sparse * sparse - template - static void subTo(Dest& dst, const Lhs& lhs, const Rhs& rhs, typename enable_if::Shape,DenseShape>::value,int*>::type* = 0) + template static void evalTo(Dest &dst, const Lhs &lhs, const Rhs &rhs) + { + evalTo(dst, lhs, rhs, typename evaluator_traits::Shape()); + } + + // dense += sparse * sparse + template + static void addTo(Dest &dst, + const ActualLhs &lhs, + const Rhs &rhs, + typename enable_if::Shape, DenseShape>::value, int *>::type * = 0) + { + typedef typename nested_eval::type LhsNested; + typedef typename nested_eval::type RhsNested; + LhsNested lhsNested(lhs); + RhsNested rhsNested(rhs); + internal::sparse_sparse_to_dense_product_selector::type, + typename remove_all::type, + Dest>::run(lhsNested, rhsNested, dst); + } + + // dense -= sparse * sparse + template + static void subTo(Dest &dst, + const Lhs &lhs, + const Rhs &rhs, + typename enable_if::Shape, DenseShape>::value, int *>::type * = 0) + { + addTo(dst, -lhs, rhs); + } + + protected: + // sparse = sparse * sparse + template static void evalTo(Dest &dst, const Lhs &lhs, const Rhs &rhs, SparseShape) + { + typedef typename nested_eval::type LhsNested; + typedef typename nested_eval::type RhsNested; + LhsNested lhsNested(lhs); + RhsNested rhsNested(rhs); + internal::conservative_sparse_sparse_product_selector::type, + typename remove_all::type, + Dest>::run(lhsNested, rhsNested, dst); + } + + // dense = sparse * sparse + template static void evalTo(Dest &dst, const Lhs &lhs, const Rhs &rhs, DenseShape) + { + dst.setZero(); + addTo(dst, lhs, rhs); + } + }; + + // sparse * sparse-triangular + template + struct generic_product_impl + : public generic_product_impl { - addTo(dst, -lhs, rhs); - } + }; -protected: - - // sparse = sparse * sparse - template - static void evalTo(Dest& dst, const Lhs& lhs, const Rhs& rhs, SparseShape) + // sparse-triangular * sparse + template + struct generic_product_impl + : public generic_product_impl { - typedef typename nested_eval::type LhsNested; - typedef typename nested_eval::type RhsNested; - LhsNested lhsNested(lhs); - RhsNested rhsNested(rhs); - internal::conservative_sparse_sparse_product_selector::type, - typename remove_all::type, Dest>::run(lhsNested,rhsNested,dst); - } - - // dense = sparse * sparse - template - static void evalTo(Dest& dst, const Lhs& lhs, const Rhs& rhs, DenseShape) + }; + + // dense = sparse-product (can be sparse*sparse, sparse*perm, etc.) + template + struct Assignment, + internal::assign_op::Scalar>, + Sparse2Dense> { - dst.setZero(); - addTo(dst, lhs, rhs); - } -}; - -// sparse * sparse-triangular -template -struct generic_product_impl - : public generic_product_impl -{}; - -// sparse-triangular * sparse -template -struct generic_product_impl - : public generic_product_impl -{}; - -// dense = sparse-product (can be sparse*sparse, sparse*perm, etc.) -template< typename DstXprType, typename Lhs, typename Rhs> -struct Assignment, internal::assign_op::Scalar>, Sparse2Dense> -{ - typedef Product SrcXprType; - static void run(DstXprType &dst, const SrcXprType &src, const internal::assign_op &) + typedef Product SrcXprType; + static void run(DstXprType &dst, + const SrcXprType &src, + const internal::assign_op &) + { + Index dstRows = src.rows(); + Index dstCols = src.cols(); + if ((dst.rows() != dstRows) || (dst.cols() != dstCols)) dst.resize(dstRows, dstCols); + + generic_product_impl::evalTo(dst, src.lhs(), src.rhs()); + } + }; + + // dense += sparse-product (can be sparse*sparse, sparse*perm, etc.) + template + struct Assignment, + internal::add_assign_op::Scalar>, + Sparse2Dense> { - Index dstRows = src.rows(); - Index dstCols = src.cols(); - if((dst.rows()!=dstRows) || (dst.cols()!=dstCols)) - dst.resize(dstRows, dstCols); - - generic_product_impl::evalTo(dst,src.lhs(),src.rhs()); - } -}; - -// dense += sparse-product (can be sparse*sparse, sparse*perm, etc.) -template< typename DstXprType, typename Lhs, typename Rhs> -struct Assignment, internal::add_assign_op::Scalar>, Sparse2Dense> -{ - typedef Product SrcXprType; - static void run(DstXprType &dst, const SrcXprType &src, const internal::add_assign_op &) + typedef Product SrcXprType; + static void run(DstXprType &dst, + const SrcXprType &src, + const internal::add_assign_op &) + { + generic_product_impl::addTo(dst, src.lhs(), src.rhs()); + } + }; + + // dense -= sparse-product (can be sparse*sparse, sparse*perm, etc.) + template + struct Assignment, + internal::sub_assign_op::Scalar>, + Sparse2Dense> { - generic_product_impl::addTo(dst,src.lhs(),src.rhs()); - } -}; - -// dense -= sparse-product (can be sparse*sparse, sparse*perm, etc.) -template< typename DstXprType, typename Lhs, typename Rhs> -struct Assignment, internal::sub_assign_op::Scalar>, Sparse2Dense> -{ - typedef Product SrcXprType; - static void run(DstXprType &dst, const SrcXprType &src, const internal::sub_assign_op &) + typedef Product SrcXprType; + static void run(DstXprType &dst, + const SrcXprType &src, + const internal::sub_assign_op &) + { + generic_product_impl::subTo(dst, src.lhs(), src.rhs()); + } + }; + + template + struct unary_evaluator>, IteratorBased> + : public evaluator::PlainObject> { - generic_product_impl::subTo(dst,src.lhs(),src.rhs()); - } -}; + typedef SparseView> XprType; + typedef typename XprType::PlainObject PlainObject; + typedef evaluator Base; -template -struct unary_evaluator >, IteratorBased> - : public evaluator::PlainObject> -{ - typedef SparseView > XprType; - typedef typename XprType::PlainObject PlainObject; - typedef evaluator Base; - - explicit unary_evaluator(const XprType& xpr) - : m_result(xpr.rows(), xpr.cols()) - { - using std::abs; - ::new (static_cast(this)) Base(m_result); - typedef typename nested_eval::type LhsNested; - typedef typename nested_eval::type RhsNested; - LhsNested lhsNested(xpr.nestedExpression().lhs()); - RhsNested rhsNested(xpr.nestedExpression().rhs()); + explicit unary_evaluator(const XprType &xpr) : m_result(xpr.rows(), xpr.cols()) + { + using std::abs; + ::new (static_cast(this)) Base(m_result); + typedef typename nested_eval::type LhsNested; + typedef typename nested_eval::type RhsNested; + LhsNested lhsNested(xpr.nestedExpression().lhs()); + RhsNested rhsNested(xpr.nestedExpression().rhs()); - internal::sparse_sparse_product_with_pruning_selector::type, - typename remove_all::type, PlainObject>::run(lhsNested,rhsNested,m_result, - abs(xpr.reference())*xpr.epsilon()); - } + internal::sparse_sparse_product_with_pruning_selector::type, + typename remove_all::type, + PlainObject>::run(lhsNested, rhsNested, m_result, abs(xpr.reference()) * xpr.epsilon()); + } -protected: - PlainObject m_result; -}; + protected: + PlainObject m_result; + }; -} // end namespace internal +}// end namespace internal -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_SPARSEPRODUCT_H +#endif// EIGEN_SPARSEPRODUCT_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/SparseCore/SparseRedux.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/SparseCore/SparseRedux.h index 45877496..7c3e4e85 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/SparseCore/SparseRedux.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/SparseCore/SparseRedux.h @@ -10,40 +10,37 @@ #ifndef EIGEN_SPARSEREDUX_H #define EIGEN_SPARSEREDUX_H -namespace Eigen { +namespace Eigen { -template -typename internal::traits::Scalar -SparseMatrixBase::sum() const +template typename internal::traits::Scalar SparseMatrixBase::sum() const { - eigen_assert(rows()>0 && cols()>0 && "you are using a non initialized matrix"); + eigen_assert(rows() > 0 && cols() > 0 && "you are using a non initialized matrix"); Scalar res(0); internal::evaluator thisEval(derived()); - for (Index j=0; j::InnerIterator iter(thisEval,j); iter; ++iter) - res += iter.value(); + for (Index j = 0; j < outerSize(); ++j) + for (typename internal::evaluator::InnerIterator iter(thisEval, j); iter; ++iter) res += iter.value(); return res; } template -typename internal::traits >::Scalar -SparseMatrix<_Scalar,_Options,_Index>::sum() const +typename internal::traits>::Scalar + SparseMatrix<_Scalar, _Options, _Index>::sum() const { - eigen_assert(rows()>0 && cols()>0 && "you are using a non initialized matrix"); - if(this->isCompressed()) - return Matrix::Map(m_data.valuePtr(), m_data.size()).sum(); + eigen_assert(rows() > 0 && cols() > 0 && "you are using a non initialized matrix"); + if (this->isCompressed()) + return Matrix::Map(m_data.valuePtr(), m_data.size()).sum(); else return Base::sum(); } template -typename internal::traits >::Scalar -SparseVector<_Scalar,_Options,_Index>::sum() const +typename internal::traits>::Scalar + SparseVector<_Scalar, _Options, _Index>::sum() const { - eigen_assert(rows()>0 && cols()>0 && "you are using a non initialized matrix"); - return Matrix::Map(m_data.valuePtr(), m_data.size()).sum(); + eigen_assert(rows() > 0 && cols() > 0 && "you are using a non initialized matrix"); + return Matrix::Map(m_data.valuePtr(), m_data.size()).sum(); } -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_SPARSEREDUX_H +#endif// EIGEN_SPARSEREDUX_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/SparseCore/SparseRef.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/SparseCore/SparseRef.h index d91f38f9..d65c3086 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/SparseCore/SparseRef.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/SparseCore/SparseRef.h @@ -13,385 +13,382 @@ namespace Eigen { enum { - StandardCompressedFormat = 2 /**< used by Ref to specify whether the input storage must be in standard compressed form */ + StandardCompressedFormat = + 2 /**< used by Ref to specify whether the input storage must be in standard compressed form */ }; - + namespace internal { -template class SparseRefBase; + template class SparseRefBase; -template -struct traits, _Options, _StrideType> > - : public traits > -{ - typedef SparseMatrix PlainObjectType; - enum { - Options = _Options, - Flags = traits::Flags | CompressedAccessBit | NestByRefBit + template + struct traits, _Options, _StrideType>> + : public traits> + { + typedef SparseMatrix PlainObjectType; + enum { Options = _Options, Flags = traits::Flags | CompressedAccessBit | NestByRefBit }; + + template struct match + { + enum { + StorageOrderMatch = PlainObjectType::IsVectorAtCompileTime || Derived::IsVectorAtCompileTime + || ((PlainObjectType::Flags & RowMajorBit) == (Derived::Flags & RowMajorBit)), + MatchAtCompileTime = (Derived::Flags & CompressedAccessBit) && StorageOrderMatch + }; + typedef typename internal::conditional::type type; + }; }; - template struct match { + template + struct traits, _Options, _StrideType>> + : public traits, _Options, _StrideType>> + { enum { - StorageOrderMatch = PlainObjectType::IsVectorAtCompileTime || Derived::IsVectorAtCompileTime || ((PlainObjectType::Flags&RowMajorBit)==(Derived::Flags&RowMajorBit)), - MatchAtCompileTime = (Derived::Flags&CompressedAccessBit) && StorageOrderMatch + Flags = + (traits>::Flags | CompressedAccessBit | NestByRefBit) & ~LvalueBit }; - typedef typename internal::conditional::type type; }; - -}; -template -struct traits, _Options, _StrideType> > - : public traits, _Options, _StrideType> > -{ - enum { - Flags = (traits >::Flags | CompressedAccessBit | NestByRefBit) & ~LvalueBit - }; -}; + template + struct traits, _Options, _StrideType>> + : public traits> + { + typedef SparseVector PlainObjectType; + enum { Options = _Options, Flags = traits::Flags | CompressedAccessBit | NestByRefBit }; -template -struct traits, _Options, _StrideType> > - : public traits > -{ - typedef SparseVector PlainObjectType; - enum { - Options = _Options, - Flags = traits::Flags | CompressedAccessBit | NestByRefBit + template struct match + { + enum { MatchAtCompileTime = (Derived::Flags & CompressedAccessBit) && Derived::IsVectorAtCompileTime }; + typedef typename internal::conditional::type type; + }; }; - template struct match { + template + struct traits, _Options, _StrideType>> + : public traits, _Options, _StrideType>> + { enum { - MatchAtCompileTime = (Derived::Flags&CompressedAccessBit) && Derived::IsVectorAtCompileTime + Flags = + (traits>::Flags | CompressedAccessBit | NestByRefBit) & ~LvalueBit }; - typedef typename internal::conditional::type type; }; -}; - -template -struct traits, _Options, _StrideType> > - : public traits, _Options, _StrideType> > -{ - enum { - Flags = (traits >::Flags | CompressedAccessBit | NestByRefBit) & ~LvalueBit + template struct traits> : public traits + { }; -}; - -template -struct traits > : public traits {}; - -template class SparseRefBase - : public SparseMapBase -{ -public: - - typedef SparseMapBase Base; - EIGEN_SPARSE_PUBLIC_INTERFACE(SparseRefBase) - SparseRefBase() - : Base(RowsAtCompileTime==Dynamic?0:RowsAtCompileTime,ColsAtCompileTime==Dynamic?0:ColsAtCompileTime, 0, 0, 0, 0, 0) - {} - -protected: - - template - void construct(Expression& expr) + template class SparseRefBase : public SparseMapBase { - if(expr.outerIndexPtr()==0) - ::new (static_cast(this)) Base(expr.size(), expr.nonZeros(), expr.innerIndexPtr(), expr.valuePtr()); - else - ::new (static_cast(this)) Base(expr.rows(), expr.cols(), expr.nonZeros(), expr.outerIndexPtr(), expr.innerIndexPtr(), expr.valuePtr(), expr.innerNonZeroPtr()); - } -}; + public: + typedef SparseMapBase Base; + EIGEN_SPARSE_PUBLIC_INTERFACE(SparseRefBase) + + SparseRefBase() + : Base(RowsAtCompileTime == Dynamic ? 0 : RowsAtCompileTime, + ColsAtCompileTime == Dynamic ? 0 : ColsAtCompileTime, + 0, + 0, + 0, + 0, + 0) + {} -} // namespace internal + protected: + template void construct(Expression &expr) + { + if (expr.outerIndexPtr() == 0) + ::new (static_cast(this)) Base(expr.size(), expr.nonZeros(), expr.innerIndexPtr(), expr.valuePtr()); + else + ::new (static_cast(this)) Base(expr.rows(), + expr.cols(), + expr.nonZeros(), + expr.outerIndexPtr(), + expr.innerIndexPtr(), + expr.valuePtr(), + expr.innerNonZeroPtr()); + } + }; +}// namespace internal -/** - * \ingroup SparseCore_Module - * - * \brief A sparse matrix expression referencing an existing sparse expression - * - * \tparam SparseMatrixType the equivalent sparse matrix type of the referenced data, it must be a template instance of class SparseMatrix. - * \tparam Options specifies whether the a standard compressed format is required \c Options is \c #StandardCompressedFormat, or \c 0. - * The default is \c 0. - * - * \sa class Ref - */ + +/** + * \ingroup SparseCore_Module + * + * \brief A sparse matrix expression referencing an existing sparse expression + * + * \tparam SparseMatrixType the equivalent sparse matrix type of the referenced data, it must be a template instance of + * class SparseMatrix. + * \tparam Options specifies whether the a standard compressed format is required \c Options is \c + * #StandardCompressedFormat, or \c 0. The default is \c 0. + * + * \sa class Ref + */ #ifndef EIGEN_PARSED_BY_DOXYGEN template -class Ref, Options, StrideType > - : public internal::SparseRefBase, Options, StrideType > > +class Ref, Options, StrideType> + : public internal::SparseRefBase, Options, StrideType>> #else template class Ref - : public SparseMapBase // yes, that's weird to use Derived here, but that works! + : public SparseMapBase// yes, that's weird to use Derived here, but that works! #endif { - typedef SparseMatrix PlainObjectType; - typedef internal::traits Traits; - template - inline Ref(const SparseMatrix& expr); - template - inline Ref(const MappedSparseMatrix& expr); - public: + typedef SparseMatrix PlainObjectType; + typedef internal::traits Traits; + template inline Ref(const SparseMatrix &expr); + template inline Ref(const MappedSparseMatrix &expr); - typedef internal::SparseRefBase Base; - EIGEN_SPARSE_PUBLIC_INTERFACE(Ref) +public: + typedef internal::SparseRefBase Base; + EIGEN_SPARSE_PUBLIC_INTERFACE(Ref) - #ifndef EIGEN_PARSED_BY_DOXYGEN - template - inline Ref(SparseMatrix& expr) - { - EIGEN_STATIC_ASSERT(bool(Traits::template match >::MatchAtCompileTime), STORAGE_LAYOUT_DOES_NOT_MATCH); - eigen_assert( ((Options & int(StandardCompressedFormat))==0) || (expr.isCompressed()) ); - Base::construct(expr.derived()); - } - - template - inline Ref(MappedSparseMatrix& expr) - { - EIGEN_STATIC_ASSERT(bool(Traits::template match >::MatchAtCompileTime), STORAGE_LAYOUT_DOES_NOT_MATCH); - eigen_assert( ((Options & int(StandardCompressedFormat))==0) || (expr.isCompressed()) ); - Base::construct(expr.derived()); - } - - template - inline Ref(const SparseCompressedBase& expr) - #else - /** Implicit constructor from any sparse expression (2D matrix or 1D vector) */ - template - inline Ref(SparseCompressedBase& expr) - #endif - { - EIGEN_STATIC_ASSERT(bool(internal::is_lvalue::value), THIS_EXPRESSION_IS_NOT_A_LVALUE__IT_IS_READ_ONLY); - EIGEN_STATIC_ASSERT(bool(Traits::template match::MatchAtCompileTime), STORAGE_LAYOUT_DOES_NOT_MATCH); - eigen_assert( ((Options & int(StandardCompressedFormat))==0) || (expr.isCompressed()) ); - Base::construct(expr.const_cast_derived()); - } +#ifndef EIGEN_PARSED_BY_DOXYGEN + template inline Ref(SparseMatrix &expr) + { + EIGEN_STATIC_ASSERT( + bool(Traits::template match>::MatchAtCompileTime), + STORAGE_LAYOUT_DOES_NOT_MATCH); + eigen_assert(((Options & int(StandardCompressedFormat)) == 0) || (expr.isCompressed())); + Base::construct(expr.derived()); + } + + template inline Ref(MappedSparseMatrix &expr) + { + EIGEN_STATIC_ASSERT( + bool(Traits::template match>::MatchAtCompileTime), + STORAGE_LAYOUT_DOES_NOT_MATCH); + eigen_assert(((Options & int(StandardCompressedFormat)) == 0) || (expr.isCompressed())); + Base::construct(expr.derived()); + } + + template inline Ref(const SparseCompressedBase &expr) +#else + /** Implicit constructor from any sparse expression (2D matrix or 1D vector) */ + template inline Ref(SparseCompressedBase &expr) +#endif + { + EIGEN_STATIC_ASSERT(bool(internal::is_lvalue::value), THIS_EXPRESSION_IS_NOT_A_LVALUE__IT_IS_READ_ONLY); + EIGEN_STATIC_ASSERT(bool(Traits::template match::MatchAtCompileTime), STORAGE_LAYOUT_DOES_NOT_MATCH); + eigen_assert(((Options & int(StandardCompressedFormat)) == 0) || (expr.isCompressed())); + Base::construct(expr.const_cast_derived()); + } }; // this is the const ref version template -class Ref, Options, StrideType> - : public internal::SparseRefBase, Options, StrideType> > +class Ref, Options, StrideType> + : public internal::SparseRefBase, Options, StrideType>> { - typedef SparseMatrix TPlainObjectType; - typedef internal::traits Traits; - public: - - typedef internal::SparseRefBase Base; - EIGEN_SPARSE_PUBLIC_INTERFACE(Ref) - - template - inline Ref(const SparseMatrixBase& expr) : m_hasCopy(false) - { - construct(expr.derived(), typename Traits::template match::type()); - } + typedef SparseMatrix TPlainObjectType; + typedef internal::traits Traits; - inline Ref(const Ref& other) : Base(other), m_hasCopy(false) { - // copy constructor shall not copy the m_object, to avoid unnecessary malloc and copy - } +public: + typedef internal::SparseRefBase Base; + EIGEN_SPARSE_PUBLIC_INTERFACE(Ref) - template - inline Ref(const RefBase& other) : m_hasCopy(false) { - construct(other.derived(), typename Traits::template match::type()); - } + template inline Ref(const SparseMatrixBase &expr) : m_hasCopy(false) + { + construct(expr.derived(), typename Traits::template match::type()); + } - ~Ref() { - if(m_hasCopy) { - TPlainObjectType* obj = reinterpret_cast(m_object_bytes); - obj->~TPlainObjectType(); - } - } + inline Ref(const Ref &other) : Base(other), m_hasCopy(false) + { + // copy constructor shall not copy the m_object, to avoid unnecessary malloc and copy + } - protected: + template inline Ref(const RefBase &other) : m_hasCopy(false) + { + construct(other.derived(), typename Traits::template match::type()); + } - template - void construct(const Expression& expr,internal::true_type) - { - if((Options & int(StandardCompressedFormat)) && (!expr.isCompressed())) - { - TPlainObjectType* obj = reinterpret_cast(m_object_bytes); - ::new (obj) TPlainObjectType(expr); - m_hasCopy = true; - Base::construct(*obj); - } - else - { - Base::construct(expr); - } + ~Ref() + { + if (m_hasCopy) { + TPlainObjectType *obj = reinterpret_cast(m_object_bytes); + obj->~TPlainObjectType(); } + } - template - void construct(const Expression& expr, internal::false_type) - { - TPlainObjectType* obj = reinterpret_cast(m_object_bytes); +protected: + template void construct(const Expression &expr, internal::true_type) + { + if ((Options & int(StandardCompressedFormat)) && (!expr.isCompressed())) { + TPlainObjectType *obj = reinterpret_cast(m_object_bytes); ::new (obj) TPlainObjectType(expr); m_hasCopy = true; Base::construct(*obj); + } else { + Base::construct(expr); } + } - protected: - char m_object_bytes[sizeof(TPlainObjectType)]; - bool m_hasCopy; -}; + template void construct(const Expression &expr, internal::false_type) + { + TPlainObjectType *obj = reinterpret_cast(m_object_bytes); + ::new (obj) TPlainObjectType(expr); + m_hasCopy = true; + Base::construct(*obj); + } +protected: + char m_object_bytes[sizeof(TPlainObjectType)]; + bool m_hasCopy; +}; /** - * \ingroup SparseCore_Module - * - * \brief A sparse vector expression referencing an existing sparse vector expression - * - * \tparam SparseVectorType the equivalent sparse vector type of the referenced data, it must be a template instance of class SparseVector. - * - * \sa class Ref - */ + * \ingroup SparseCore_Module + * + * \brief A sparse vector expression referencing an existing sparse vector expression + * + * \tparam SparseVectorType the equivalent sparse vector type of the referenced data, it must be a template instance of + * class SparseVector. + * + * \sa class Ref + */ #ifndef EIGEN_PARSED_BY_DOXYGEN template -class Ref, Options, StrideType > - : public internal::SparseRefBase, Options, StrideType > > +class Ref, Options, StrideType> + : public internal::SparseRefBase, Options, StrideType>> #else -template -class Ref - : public SparseMapBase +template class Ref : public SparseMapBase #endif { - typedef SparseVector PlainObjectType; - typedef internal::traits Traits; - template - inline Ref(const SparseVector& expr); - public: + typedef SparseVector PlainObjectType; + typedef internal::traits Traits; + template inline Ref(const SparseVector &expr); - typedef internal::SparseRefBase Base; - EIGEN_SPARSE_PUBLIC_INTERFACE(Ref) +public: + typedef internal::SparseRefBase Base; + EIGEN_SPARSE_PUBLIC_INTERFACE(Ref) - #ifndef EIGEN_PARSED_BY_DOXYGEN - template - inline Ref(SparseVector& expr) - { - EIGEN_STATIC_ASSERT(bool(Traits::template match >::MatchAtCompileTime), STORAGE_LAYOUT_DOES_NOT_MATCH); - Base::construct(expr.derived()); - } +#ifndef EIGEN_PARSED_BY_DOXYGEN + template inline Ref(SparseVector &expr) + { + EIGEN_STATIC_ASSERT( + bool(Traits::template match>::MatchAtCompileTime), + STORAGE_LAYOUT_DOES_NOT_MATCH); + Base::construct(expr.derived()); + } - template - inline Ref(const SparseCompressedBase& expr) - #else - /** Implicit constructor from any 1D sparse vector expression */ - template - inline Ref(SparseCompressedBase& expr) - #endif - { - EIGEN_STATIC_ASSERT(bool(internal::is_lvalue::value), THIS_EXPRESSION_IS_NOT_A_LVALUE__IT_IS_READ_ONLY); - EIGEN_STATIC_ASSERT(bool(Traits::template match::MatchAtCompileTime), STORAGE_LAYOUT_DOES_NOT_MATCH); - Base::construct(expr.const_cast_derived()); - } + template inline Ref(const SparseCompressedBase &expr) +#else + /** Implicit constructor from any 1D sparse vector expression */ + template inline Ref(SparseCompressedBase &expr) +#endif + { + EIGEN_STATIC_ASSERT(bool(internal::is_lvalue::value), THIS_EXPRESSION_IS_NOT_A_LVALUE__IT_IS_READ_ONLY); + EIGEN_STATIC_ASSERT(bool(Traits::template match::MatchAtCompileTime), STORAGE_LAYOUT_DOES_NOT_MATCH); + Base::construct(expr.const_cast_derived()); + } }; // this is the const ref version template -class Ref, Options, StrideType> - : public internal::SparseRefBase, Options, StrideType> > +class Ref, Options, StrideType> + : public internal::SparseRefBase, Options, StrideType>> { - typedef SparseVector TPlainObjectType; - typedef internal::traits Traits; - public: + typedef SparseVector TPlainObjectType; + typedef internal::traits Traits; - typedef internal::SparseRefBase Base; - EIGEN_SPARSE_PUBLIC_INTERFACE(Ref) +public: + typedef internal::SparseRefBase Base; + EIGEN_SPARSE_PUBLIC_INTERFACE(Ref) - template - inline Ref(const SparseMatrixBase& expr) : m_hasCopy(false) - { - construct(expr.derived(), typename Traits::template match::type()); - } + template inline Ref(const SparseMatrixBase &expr) : m_hasCopy(false) + { + construct(expr.derived(), typename Traits::template match::type()); + } - inline Ref(const Ref& other) : Base(other), m_hasCopy(false) { - // copy constructor shall not copy the m_object, to avoid unnecessary malloc and copy - } + inline Ref(const Ref &other) : Base(other), m_hasCopy(false) + { + // copy constructor shall not copy the m_object, to avoid unnecessary malloc and copy + } - template - inline Ref(const RefBase& other) : m_hasCopy(false) { - construct(other.derived(), typename Traits::template match::type()); - } + template inline Ref(const RefBase &other) : m_hasCopy(false) + { + construct(other.derived(), typename Traits::template match::type()); + } - ~Ref() { - if(m_hasCopy) { - TPlainObjectType* obj = reinterpret_cast(m_object_bytes); - obj->~TPlainObjectType(); - } + ~Ref() + { + if (m_hasCopy) { + TPlainObjectType *obj = reinterpret_cast(m_object_bytes); + obj->~TPlainObjectType(); } + } - protected: - - template - void construct(const Expression& expr,internal::true_type) - { - Base::construct(expr); - } +protected: + template void construct(const Expression &expr, internal::true_type) { Base::construct(expr); } - template - void construct(const Expression& expr, internal::false_type) - { - TPlainObjectType* obj = reinterpret_cast(m_object_bytes); - ::new (obj) TPlainObjectType(expr); - m_hasCopy = true; - Base::construct(*obj); - } + template void construct(const Expression &expr, internal::false_type) + { + TPlainObjectType *obj = reinterpret_cast(m_object_bytes); + ::new (obj) TPlainObjectType(expr); + m_hasCopy = true; + Base::construct(*obj); + } - protected: - char m_object_bytes[sizeof(TPlainObjectType)]; - bool m_hasCopy; +protected: + char m_object_bytes[sizeof(TPlainObjectType)]; + bool m_hasCopy; }; namespace internal { -// FIXME shall we introduce a general evaluatior_ref that we can specialize for any sparse object once, and thus remove this copy-pasta thing... + // FIXME shall we introduce a general evaluatior_ref that we can specialize for any sparse object once, and thus + // remove this copy-pasta thing... -template -struct evaluator, Options, StrideType> > - : evaluator, Options, StrideType> > > -{ - typedef evaluator, Options, StrideType> > > Base; - typedef Ref, Options, StrideType> XprType; - evaluator() : Base() {} - explicit evaluator(const XprType &mat) : Base(mat) {} -}; + template + struct evaluator, Options, StrideType>> + : evaluator, Options, StrideType>>> + { + typedef evaluator, Options, StrideType>>> + Base; + typedef Ref, Options, StrideType> XprType; + evaluator() : Base() {} + explicit evaluator(const XprType &mat) : Base(mat) {} + }; -template -struct evaluator, Options, StrideType> > - : evaluator, Options, StrideType> > > -{ - typedef evaluator, Options, StrideType> > > Base; - typedef Ref, Options, StrideType> XprType; - evaluator() : Base() {} - explicit evaluator(const XprType &mat) : Base(mat) {} -}; + template + struct evaluator, Options, StrideType>> + : evaluator, Options, StrideType>>> + { + typedef evaluator< + SparseCompressedBase, Options, StrideType>>> + Base; + typedef Ref, Options, StrideType> XprType; + evaluator() : Base() {} + explicit evaluator(const XprType &mat) : Base(mat) {} + }; -template -struct evaluator, Options, StrideType> > - : evaluator, Options, StrideType> > > -{ - typedef evaluator, Options, StrideType> > > Base; - typedef Ref, Options, StrideType> XprType; - evaluator() : Base() {} - explicit evaluator(const XprType &mat) : Base(mat) {} -}; + template + struct evaluator, Options, StrideType>> + : evaluator, Options, StrideType>>> + { + typedef evaluator, Options, StrideType>>> + Base; + typedef Ref, Options, StrideType> XprType; + evaluator() : Base() {} + explicit evaluator(const XprType &mat) : Base(mat) {} + }; -template -struct evaluator, Options, StrideType> > - : evaluator, Options, StrideType> > > -{ - typedef evaluator, Options, StrideType> > > Base; - typedef Ref, Options, StrideType> XprType; - evaluator() : Base() {} - explicit evaluator(const XprType &mat) : Base(mat) {} -}; + template + struct evaluator, Options, StrideType>> + : evaluator, Options, StrideType>>> + { + typedef evaluator< + SparseCompressedBase, Options, StrideType>>> + Base; + typedef Ref, Options, StrideType> XprType; + evaluator() : Base() {} + explicit evaluator(const XprType &mat) : Base(mat) {} + }; -} +}// namespace internal -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_SPARSE_REF_H +#endif// EIGEN_SPARSE_REF_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/SparseCore/SparseSelfAdjointView.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/SparseCore/SparseSelfAdjointView.h index 65611b3d..26831811 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/SparseCore/SparseSelfAdjointView.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/SparseCore/SparseSelfAdjointView.h @@ -10,191 +10,199 @@ #ifndef EIGEN_SPARSE_SELFADJOINTVIEW_H #define EIGEN_SPARSE_SELFADJOINTVIEW_H -namespace Eigen { - +namespace Eigen { + /** \ingroup SparseCore_Module - * \class SparseSelfAdjointView - * - * \brief Pseudo expression to manipulate a triangular sparse matrix as a selfadjoint matrix. - * - * \param MatrixType the type of the dense matrix storing the coefficients - * \param Mode can be either \c #Lower or \c #Upper - * - * This class is an expression of a sefladjoint matrix from a triangular part of a matrix - * with given dense storage of the coefficients. It is the return type of MatrixBase::selfadjointView() - * and most of the time this is the only way that it is used. - * - * \sa SparseMatrixBase::selfadjointView() - */ + * \class SparseSelfAdjointView + * + * \brief Pseudo expression to manipulate a triangular sparse matrix as a selfadjoint matrix. + * + * \param MatrixType the type of the dense matrix storing the coefficients + * \param Mode can be either \c #Lower or \c #Upper + * + * This class is an expression of a sefladjoint matrix from a triangular part of a matrix + * with given dense storage of the coefficients. It is the return type of MatrixBase::selfadjointView() + * and most of the time this is the only way that it is used. + * + * \sa SparseMatrixBase::selfadjointView() + */ namespace internal { - -template -struct traits > : traits { -}; -template -void permute_symm_to_symm(const MatrixType& mat, SparseMatrix& _dest, const typename MatrixType::StorageIndex* perm = 0); + template + struct traits> : traits + { + }; -template -void permute_symm_to_fullsymm(const MatrixType& mat, SparseMatrix& _dest, const typename MatrixType::StorageIndex* perm = 0); + template + void permute_symm_to_symm(const MatrixType &mat, + SparseMatrix &_dest, + const typename MatrixType::StorageIndex *perm = 0); -} + template + void permute_symm_to_fullsymm(const MatrixType &mat, + SparseMatrix &_dest, + const typename MatrixType::StorageIndex *perm = 0); + +}// namespace internal -template class SparseSelfAdjointView - : public EigenBase > +template +class SparseSelfAdjointView : public EigenBase> { - public: - - enum { - Mode = _Mode, - TransposeMode = ((Mode & Upper) ? Lower : 0) | ((Mode & Lower) ? Upper : 0), - RowsAtCompileTime = internal::traits::RowsAtCompileTime, - ColsAtCompileTime = internal::traits::ColsAtCompileTime - }; +public: + enum { + Mode = _Mode, + TransposeMode = ((Mode & Upper) ? Lower : 0) | ((Mode & Lower) ? Upper : 0), + RowsAtCompileTime = internal::traits::RowsAtCompileTime, + ColsAtCompileTime = internal::traits::ColsAtCompileTime + }; - typedef EigenBase Base; - typedef typename MatrixType::Scalar Scalar; - typedef typename MatrixType::StorageIndex StorageIndex; - typedef Matrix VectorI; - typedef typename internal::ref_selector::non_const_type MatrixTypeNested; - typedef typename internal::remove_all::type _MatrixTypeNested; - - explicit inline SparseSelfAdjointView(MatrixType& matrix) : m_matrix(matrix) - { - eigen_assert(rows()==cols() && "SelfAdjointView is only for squared matrices"); - } + typedef EigenBase Base; + typedef typename MatrixType::Scalar Scalar; + typedef typename MatrixType::StorageIndex StorageIndex; + typedef Matrix VectorI; + typedef typename internal::ref_selector::non_const_type MatrixTypeNested; + typedef typename internal::remove_all::type _MatrixTypeNested; - inline Index rows() const { return m_matrix.rows(); } - inline Index cols() const { return m_matrix.cols(); } - - /** \internal \returns a reference to the nested matrix */ - const _MatrixTypeNested& matrix() const { return m_matrix; } - typename internal::remove_reference::type& matrix() { return m_matrix; } - - /** \returns an expression of the matrix product between a sparse self-adjoint matrix \c *this and a sparse matrix \a rhs. - * - * Note that there is no algorithmic advantage of performing such a product compared to a general sparse-sparse matrix product. - * Indeed, the SparseSelfadjointView operand is first copied into a temporary SparseMatrix before computing the product. - */ - template - Product - operator*(const SparseMatrixBase& rhs) const - { - return Product(*this, rhs.derived()); - } + explicit inline SparseSelfAdjointView(MatrixType &matrix) : m_matrix(matrix) + { + eigen_assert(rows() == cols() && "SelfAdjointView is only for squared matrices"); + } - /** \returns an expression of the matrix product between a sparse matrix \a lhs and a sparse self-adjoint matrix \a rhs. - * - * Note that there is no algorithmic advantage of performing such a product compared to a general sparse-sparse matrix product. - * Indeed, the SparseSelfadjointView operand is first copied into a temporary SparseMatrix before computing the product. - */ - template friend - Product - operator*(const SparseMatrixBase& lhs, const SparseSelfAdjointView& rhs) - { - return Product(lhs.derived(), rhs); - } - - /** Efficient sparse self-adjoint matrix times dense vector/matrix product */ - template - Product - operator*(const MatrixBase& rhs) const - { - return Product(*this, rhs.derived()); - } + inline Index rows() const { return m_matrix.rows(); } + inline Index cols() const { return m_matrix.cols(); } + + /** \internal \returns a reference to the nested matrix */ + const _MatrixTypeNested &matrix() const { return m_matrix; } + typename internal::remove_reference::type &matrix() { return m_matrix; } + + /** \returns an expression of the matrix product between a sparse self-adjoint matrix \c *this and a sparse matrix \a + * rhs. + * + * Note that there is no algorithmic advantage of performing such a product compared to a general sparse-sparse matrix + * product. Indeed, the SparseSelfadjointView operand is first copied into a temporary SparseMatrix before computing + * the product. + */ + template + Product operator*(const SparseMatrixBase &rhs) const + { + return Product(*this, rhs.derived()); + } - /** Efficient dense vector/matrix times sparse self-adjoint matrix product */ - template friend - Product - operator*(const MatrixBase& lhs, const SparseSelfAdjointView& rhs) - { - return Product(lhs.derived(), rhs); - } + /** \returns an expression of the matrix product between a sparse matrix \a lhs and a sparse self-adjoint matrix \a + * rhs. + * + * Note that there is no algorithmic advantage of performing such a product compared to a general sparse-sparse matrix + * product. Indeed, the SparseSelfadjointView operand is first copied into a temporary SparseMatrix before computing + * the product. + */ + template + friend Product operator*(const SparseMatrixBase &lhs, + const SparseSelfAdjointView &rhs) + { + return Product(lhs.derived(), rhs); + } - /** Perform a symmetric rank K update of the selfadjoint matrix \c *this: - * \f$ this = this + \alpha ( u u^* ) \f$ where \a u is a vector or matrix. - * - * \returns a reference to \c *this - * - * To perform \f$ this = this + \alpha ( u^* u ) \f$ you can simply - * call this function with u.adjoint(). - */ - template - SparseSelfAdjointView& rankUpdate(const SparseMatrixBase& u, const Scalar& alpha = Scalar(1)); - - /** \returns an expression of P H P^-1 */ - // TODO implement twists in a more evaluator friendly fashion - SparseSymmetricPermutationProduct<_MatrixTypeNested,Mode> twistedBy(const PermutationMatrix& perm) const - { - return SparseSymmetricPermutationProduct<_MatrixTypeNested,Mode>(m_matrix, perm); - } + /** Efficient sparse self-adjoint matrix times dense vector/matrix product */ + template + Product operator*(const MatrixBase &rhs) const + { + return Product(*this, rhs.derived()); + } - template - SparseSelfAdjointView& operator=(const SparseSymmetricPermutationProduct& permutedMatrix) - { - internal::call_assignment_no_alias_no_transpose(*this, permutedMatrix); - return *this; - } + /** Efficient dense vector/matrix times sparse self-adjoint matrix product */ + template + friend Product operator*(const MatrixBase &lhs, + const SparseSelfAdjointView &rhs) + { + return Product(lhs.derived(), rhs); + } - SparseSelfAdjointView& operator=(const SparseSelfAdjointView& src) - { - PermutationMatrix pnull; - return *this = src.twistedBy(pnull); - } + /** Perform a symmetric rank K update of the selfadjoint matrix \c *this: + * \f$ this = this + \alpha ( u u^* ) \f$ where \a u is a vector or matrix. + * + * \returns a reference to \c *this + * + * To perform \f$ this = this + \alpha ( u^* u ) \f$ you can simply + * call this function with u.adjoint(). + */ + template + SparseSelfAdjointView &rankUpdate(const SparseMatrixBase &u, const Scalar &alpha = Scalar(1)); + + /** \returns an expression of P H P^-1 */ + // TODO implement twists in a more evaluator friendly fashion + SparseSymmetricPermutationProduct<_MatrixTypeNested, Mode> twistedBy( + const PermutationMatrix &perm) const + { + return SparseSymmetricPermutationProduct<_MatrixTypeNested, Mode>(m_matrix, perm); + } - template - SparseSelfAdjointView& operator=(const SparseSelfAdjointView& src) - { - PermutationMatrix pnull; - return *this = src.twistedBy(pnull); - } - - void resize(Index rows, Index cols) - { - EIGEN_ONLY_USED_FOR_DEBUG(rows); - EIGEN_ONLY_USED_FOR_DEBUG(cols); - eigen_assert(rows == this->rows() && cols == this->cols() - && "SparseSelfadjointView::resize() does not actually allow to resize."); - } - - protected: + template + SparseSelfAdjointView &operator=(const SparseSymmetricPermutationProduct &permutedMatrix) + { + internal::call_assignment_no_alias_no_transpose(*this, permutedMatrix); + return *this; + } + + SparseSelfAdjointView &operator=(const SparseSelfAdjointView &src) + { + PermutationMatrix pnull; + return *this = src.twistedBy(pnull); + } + + template + SparseSelfAdjointView &operator=(const SparseSelfAdjointView &src) + { + PermutationMatrix pnull; + return *this = src.twistedBy(pnull); + } - MatrixTypeNested m_matrix; - //mutable VectorI m_countPerRow; - //mutable VectorI m_countPerCol; - private: - template void evalTo(Dest &) const; + void resize(Index rows, Index cols) + { + EIGEN_ONLY_USED_FOR_DEBUG(rows); + EIGEN_ONLY_USED_FOR_DEBUG(cols); + eigen_assert(rows == this->rows() && cols == this->cols() + && "SparseSelfadjointView::resize() does not actually allow to resize."); + } + +protected: + MatrixTypeNested m_matrix; + // mutable VectorI m_countPerRow; + // mutable VectorI m_countPerCol; +private: + template void evalTo(Dest &) const; }; /*************************************************************************** -* Implementation of SparseMatrixBase methods -***************************************************************************/ + * Implementation of SparseMatrixBase methods + ***************************************************************************/ template template -typename SparseMatrixBase::template ConstSelfAdjointViewReturnType::Type SparseMatrixBase::selfadjointView() const +typename SparseMatrixBase::template ConstSelfAdjointViewReturnType::Type + SparseMatrixBase::selfadjointView() const { return SparseSelfAdjointView(derived()); } template template -typename SparseMatrixBase::template SelfAdjointViewReturnType::Type SparseMatrixBase::selfadjointView() +typename SparseMatrixBase::template SelfAdjointViewReturnType::Type + SparseMatrixBase::selfadjointView() { return SparseSelfAdjointView(derived()); } /*************************************************************************** -* Implementation of SparseSelfAdjointView methods -***************************************************************************/ + * Implementation of SparseSelfAdjointView methods + ***************************************************************************/ template template -SparseSelfAdjointView& -SparseSelfAdjointView::rankUpdate(const SparseMatrixBase& u, const Scalar& alpha) +SparseSelfAdjointView & + SparseSelfAdjointView::rankUpdate(const SparseMatrixBase &u, const Scalar &alpha) { - SparseMatrix tmp = u * u.adjoint(); - if(alpha==Scalar(0)) + SparseMatrix tmp = u * u.adjoint(); + if (alpha == Scalar(0)) m_matrix = tmp.template triangularView(); else m_matrix += alpha * tmp.template triangularView(); @@ -203,454 +211,457 @@ SparseSelfAdjointView::rankUpdate(const SparseMatrixBase -// in the future selfadjoint-ness should be defined by the expression traits -// such that Transpose > is valid. (currently TriangularBase::transpose() is overloaded to make it work) -template -struct evaluator_traits > -{ - typedef typename storage_kind_to_evaluator_kind::Kind Kind; - typedef SparseSelfAdjointShape Shape; -}; -struct SparseSelfAdjoint2Sparse {}; - -template<> struct AssignmentKind { typedef SparseSelfAdjoint2Sparse Kind; }; -template<> struct AssignmentKind { typedef Sparse2Sparse Kind; }; - -template< typename DstXprType, typename SrcXprType, typename Functor> -struct Assignment -{ - typedef typename DstXprType::StorageIndex StorageIndex; - typedef internal::assign_op AssignOpType; - - template - static void run(SparseMatrix &dst, const SrcXprType &src, const AssignOpType&/*func*/) + // TODO currently a selfadjoint expression has the form SelfAdjointView<.,.> + // in the future selfadjoint-ness should be defined by the expression traits + // such that Transpose > is valid. (currently TriangularBase::transpose() is overloaded to + // make it work) + template struct evaluator_traits> { - internal::permute_symm_to_fullsymm(src.matrix(), dst); - } + typedef typename storage_kind_to_evaluator_kind::Kind Kind; + typedef SparseSelfAdjointShape Shape; + }; - // FIXME: the handling of += and -= in sparse matrices should be cleanup so that next two overloads could be reduced to: - template - static void run(SparseMatrix &dst, const SrcXprType &src, const AssignFunc& func) + struct SparseSelfAdjoint2Sparse { - SparseMatrix tmp(src.rows(),src.cols()); - run(tmp, src, AssignOpType()); - call_assignment_no_alias_no_transpose(dst, tmp, func); - } + }; - template - static void run(SparseMatrix &dst, const SrcXprType &src, - const internal::add_assign_op& /* func */) + template<> struct AssignmentKind { - SparseMatrix tmp(src.rows(),src.cols()); - run(tmp, src, AssignOpType()); - dst += tmp; - } - - template - static void run(SparseMatrix &dst, const SrcXprType &src, - const internal::sub_assign_op& /* func */) + typedef SparseSelfAdjoint2Sparse Kind; + }; + template<> struct AssignmentKind { - SparseMatrix tmp(src.rows(),src.cols()); - run(tmp, src, AssignOpType()); - dst -= tmp; - } - - template - static void run(DynamicSparseMatrix& dst, const SrcXprType &src, const AssignOpType&/*func*/) + typedef Sparse2Sparse Kind; + }; + + template + struct Assignment { - // TODO directly evaluate into dst; - SparseMatrix tmp(dst.rows(),dst.cols()); - internal::permute_symm_to_fullsymm(src.matrix(), tmp); - dst = tmp; - } -}; + typedef typename DstXprType::StorageIndex StorageIndex; + typedef internal::assign_op AssignOpType; -} // end namespace internal + template + static void run(SparseMatrix &dst, + const SrcXprType &src, + const AssignOpType & /*func*/) + { + internal::permute_symm_to_fullsymm(src.matrix(), dst); + } -/*************************************************************************** -* Implementation of sparse self-adjoint time dense matrix -***************************************************************************/ + // FIXME: the handling of += and -= in sparse matrices should be cleanup so that next two overloads could be reduced + // to: + template + static void + run(SparseMatrix &dst, const SrcXprType &src, const AssignFunc &func) + { + SparseMatrix tmp(src.rows(), src.cols()); + run(tmp, src, AssignOpType()); + call_assignment_no_alias_no_transpose(dst, tmp, func); + } -namespace internal { + template + static void run(SparseMatrix &dst, + const SrcXprType &src, + const internal::add_assign_op & /* func */) + { + SparseMatrix tmp(src.rows(), src.cols()); + run(tmp, src, AssignOpType()); + dst += tmp; + } -template -inline void sparse_selfadjoint_time_dense_product(const SparseLhsType& lhs, const DenseRhsType& rhs, DenseResType& res, const AlphaType& alpha) -{ - EIGEN_ONLY_USED_FOR_DEBUG(alpha); - - typedef typename internal::nested_eval::type SparseLhsTypeNested; - typedef typename internal::remove_all::type SparseLhsTypeNestedCleaned; - typedef evaluator LhsEval; - typedef typename LhsEval::InnerIterator LhsIterator; - typedef typename SparseLhsType::Scalar LhsScalar; - - enum { - LhsIsRowMajor = (LhsEval::Flags&RowMajorBit)==RowMajorBit, - ProcessFirstHalf = - ((Mode&(Upper|Lower))==(Upper|Lower)) - || ( (Mode&Upper) && !LhsIsRowMajor) - || ( (Mode&Lower) && LhsIsRowMajor), - ProcessSecondHalf = !ProcessFirstHalf + template + static void run(SparseMatrix &dst, + const SrcXprType &src, + const internal::sub_assign_op & /* func */) + { + SparseMatrix tmp(src.rows(), src.cols()); + run(tmp, src, AssignOpType()); + dst -= tmp; + } + + template + static void run(DynamicSparseMatrix &dst, + const SrcXprType &src, + const AssignOpType & /*func*/) + { + // TODO directly evaluate into dst; + SparseMatrix tmp(dst.rows(), dst.cols()); + internal::permute_symm_to_fullsymm(src.matrix(), tmp); + dst = tmp; + } }; - - SparseLhsTypeNested lhs_nested(lhs); - LhsEval lhsEval(lhs_nested); - // work on one column at once - for (Index k=0; k + inline void sparse_selfadjoint_time_dense_product(const SparseLhsType &lhs, + const DenseRhsType &rhs, + DenseResType &res, + const AlphaType &alpha) { - for (Index j=0; j::type SparseLhsTypeNested; + typedef typename internal::remove_all::type SparseLhsTypeNestedCleaned; + typedef evaluator LhsEval; + typedef typename LhsEval::InnerIterator LhsIterator; + typedef typename SparseLhsType::Scalar LhsScalar; + + enum { + LhsIsRowMajor = (LhsEval::Flags & RowMajorBit) == RowMajorBit, + ProcessFirstHalf = ((Mode & (Upper | Lower)) == (Upper | Lower)) || ((Mode & Upper) && !LhsIsRowMajor) + || ((Mode & Lower) && LhsIsRowMajor), + ProcessSecondHalf = !ProcessFirstHalf + }; + + SparseLhsTypeNested lhs_nested(lhs); + LhsEval lhsEval(lhs_nested); + + // work on one column at once + for (Index k = 0; k < rhs.cols(); ++k) { + for (Index j = 0; j < lhs.outerSize(); ++j) { + LhsIterator i(lhsEval, j); + // handle diagonal coeff + if (ProcessSecondHalf) { + while (i && i.index() < j) ++i; + if (i && i.index() == j) { + res.coeffRef(j, k) += alpha * i.value() * rhs.coeff(j, k); + ++i; + } } - } - // premultiplied rhs for scatters - typename ScalarBinaryOpTraits::ReturnType rhs_j(alpha*rhs(j,k)); - // accumulator for partial scalar product - typename DenseResType::Scalar res_j(0); - for(; (ProcessFirstHalf ? i && i.index() < j : i) ; ++i) - { - LhsScalar lhs_ij = i.value(); - if(!LhsIsRowMajor) lhs_ij = numext::conj(lhs_ij); - res_j += lhs_ij * rhs.coeff(i.index(),k); - res(i.index(),k) += numext::conj(lhs_ij) * rhs_j; - } - res.coeffRef(j,k) += alpha * res_j; + // premultiplied rhs for scatters + typename ScalarBinaryOpTraits::ReturnType rhs_j(alpha * rhs(j, k)); + // accumulator for partial scalar product + typename DenseResType::Scalar res_j(0); + for (; (ProcessFirstHalf ? i && i.index() < j : i); ++i) { + LhsScalar lhs_ij = i.value(); + if (!LhsIsRowMajor) lhs_ij = numext::conj(lhs_ij); + res_j += lhs_ij * rhs.coeff(i.index(), k); + res(i.index(), k) += numext::conj(lhs_ij) * rhs_j; + } + res.coeffRef(j, k) += alpha * res_j; - // handle diagonal coeff - if (ProcessFirstHalf && i && (i.index()==j)) - res.coeffRef(j,k) += alpha * i.value() * rhs.coeff(j,k); + // handle diagonal coeff + if (ProcessFirstHalf && i && (i.index() == j)) res.coeffRef(j, k) += alpha * i.value() * rhs.coeff(j, k); + } } } -} -template -struct generic_product_impl -: generic_product_impl_base > -{ - template - static void scaleAndAddTo(Dest& dst, const LhsView& lhsView, const Rhs& rhs, const typename Dest::Scalar& alpha) + template + struct generic_product_impl + : generic_product_impl_base> { - typedef typename LhsView::_MatrixTypeNested Lhs; - typedef typename nested_eval::type LhsNested; - typedef typename nested_eval::type RhsNested; - LhsNested lhsNested(lhsView.matrix()); - RhsNested rhsNested(rhs); - - internal::sparse_selfadjoint_time_dense_product(lhsNested, rhsNested, dst, alpha); - } -}; + template + static void scaleAndAddTo(Dest &dst, const LhsView &lhsView, const Rhs &rhs, const typename Dest::Scalar &alpha) + { + typedef typename LhsView::_MatrixTypeNested Lhs; + typedef typename nested_eval::type LhsNested; + typedef typename nested_eval::type RhsNested; + LhsNested lhsNested(lhsView.matrix()); + RhsNested rhsNested(rhs); -template -struct generic_product_impl -: generic_product_impl_base > -{ - template - static void scaleAndAddTo(Dest& dst, const Lhs& lhs, const RhsView& rhsView, const typename Dest::Scalar& alpha) - { - typedef typename RhsView::_MatrixTypeNested Rhs; - typedef typename nested_eval::type LhsNested; - typedef typename nested_eval::type RhsNested; - LhsNested lhsNested(lhs); - RhsNested rhsNested(rhsView.matrix()); - - // transpose everything - Transpose dstT(dst); - internal::sparse_selfadjoint_time_dense_product(rhsNested.transpose(), lhsNested.transpose(), dstT, alpha); - } -}; + internal::sparse_selfadjoint_time_dense_product(lhsNested, rhsNested, dst, alpha); + } + }; -// NOTE: these two overloads are needed to evaluate the sparse selfadjoint view into a full sparse matrix -// TODO: maybe the copy could be handled by generic_product_impl so that these overloads would not be needed anymore + template + struct generic_product_impl + : generic_product_impl_base> + { + template + static void scaleAndAddTo(Dest &dst, const Lhs &lhs, const RhsView &rhsView, const typename Dest::Scalar &alpha) + { + typedef typename RhsView::_MatrixTypeNested Rhs; + typedef typename nested_eval::type LhsNested; + typedef typename nested_eval::type RhsNested; + LhsNested lhsNested(lhs); + RhsNested rhsNested(rhsView.matrix()); + + // transpose everything + Transpose dstT(dst); + internal::sparse_selfadjoint_time_dense_product( + rhsNested.transpose(), lhsNested.transpose(), dstT, alpha); + } + }; -template -struct product_evaluator, ProductTag, SparseSelfAdjointShape, SparseShape> - : public evaluator::PlainObject> -{ - typedef Product XprType; - typedef typename XprType::PlainObject PlainObject; - typedef evaluator Base; + // NOTE: these two overloads are needed to evaluate the sparse selfadjoint view into a full sparse matrix + // TODO: maybe the copy could be handled by generic_product_impl so that these overloads would not be needed anymore - product_evaluator(const XprType& xpr) - : m_lhs(xpr.lhs()), m_result(xpr.rows(), xpr.cols()) + template + struct product_evaluator, ProductTag, SparseSelfAdjointShape, SparseShape> + : public evaluator::PlainObject> { - ::new (static_cast(this)) Base(m_result); - generic_product_impl::evalTo(m_result, m_lhs, xpr.rhs()); - } - -protected: - typename Rhs::PlainObject m_lhs; - PlainObject m_result; -}; + typedef Product XprType; + typedef typename XprType::PlainObject PlainObject; + typedef evaluator Base; -template -struct product_evaluator, ProductTag, SparseShape, SparseSelfAdjointShape> - : public evaluator::PlainObject> -{ - typedef Product XprType; - typedef typename XprType::PlainObject PlainObject; - typedef evaluator Base; + product_evaluator(const XprType &xpr) : m_lhs(xpr.lhs()), m_result(xpr.rows(), xpr.cols()) + { + ::new (static_cast(this)) Base(m_result); + generic_product_impl::evalTo( + m_result, m_lhs, xpr.rhs()); + } + + protected: + typename Rhs::PlainObject m_lhs; + PlainObject m_result; + }; - product_evaluator(const XprType& xpr) - : m_rhs(xpr.rhs()), m_result(xpr.rows(), xpr.cols()) + template + struct product_evaluator, ProductTag, SparseShape, SparseSelfAdjointShape> + : public evaluator::PlainObject> { - ::new (static_cast(this)) Base(m_result); - generic_product_impl::evalTo(m_result, xpr.lhs(), m_rhs); - } - -protected: - typename Lhs::PlainObject m_rhs; - PlainObject m_result; -}; + typedef Product XprType; + typedef typename XprType::PlainObject PlainObject; + typedef evaluator Base; -} // namespace internal + product_evaluator(const XprType &xpr) : m_rhs(xpr.rhs()), m_result(xpr.rows(), xpr.cols()) + { + ::new (static_cast(this)) Base(m_result); + generic_product_impl::evalTo( + m_result, xpr.lhs(), m_rhs); + } + + protected: + typename Lhs::PlainObject m_rhs; + PlainObject m_result; + }; + +}// namespace internal /*************************************************************************** -* Implementation of symmetric copies and permutations -***************************************************************************/ + * Implementation of symmetric copies and permutations + ***************************************************************************/ namespace internal { -template -void permute_symm_to_fullsymm(const MatrixType& mat, SparseMatrix& _dest, const typename MatrixType::StorageIndex* perm) -{ - typedef typename MatrixType::StorageIndex StorageIndex; - typedef typename MatrixType::Scalar Scalar; - typedef SparseMatrix Dest; - typedef Matrix VectorI; - typedef evaluator MatEval; - typedef typename evaluator::InnerIterator MatIterator; - - MatEval matEval(mat); - Dest& dest(_dest.derived()); - enum { - StorageOrderMatch = int(Dest::IsRowMajor) == int(MatrixType::IsRowMajor) - }; - - Index size = mat.rows(); - VectorI count; - count.resize(size); - count.setZero(); - dest.resize(size,size); - for(Index j = 0; j + void permute_symm_to_fullsymm(const MatrixType &mat, + SparseMatrix &_dest, + const typename MatrixType::StorageIndex *perm) { - Index jp = perm ? perm[j] : j; - for(MatIterator it(matEval,j); it; ++it) - { - Index i = it.index(); - Index r = it.row(); - Index c = it.col(); - Index ip = perm ? perm[i] : i; - if(Mode==(Upper|Lower)) - count[StorageOrderMatch ? jp : ip]++; - else if(r==c) - count[ip]++; - else if(( Mode==Lower && r>c) || ( Mode==Upper && r Dest; + typedef Matrix VectorI; + typedef evaluator MatEval; + typedef typename evaluator::InnerIterator MatIterator; + + MatEval matEval(mat); + Dest &dest(_dest.derived()); + enum { StorageOrderMatch = int(Dest::IsRowMajor) == int(MatrixType::IsRowMajor) }; + + Index size = mat.rows(); + VectorI count; + count.resize(size); + count.setZero(); + dest.resize(size, size); + for (Index j = 0; j < size; ++j) { + Index jp = perm ? perm[j] : j; + for (MatIterator it(matEval, j); it; ++it) { + Index i = it.index(); + Index r = it.row(); + Index c = it.col(); + Index ip = perm ? perm[i] : i; + if (Mode == (Upper | Lower)) + count[StorageOrderMatch ? jp : ip]++; + else if (r == c) + count[ip]++; + else if ((Mode == Lower && r > c) || (Mode == Upper && r < c)) { + count[ip]++; + count[jp]++; + } + } + } + Index nnz = count.sum(); + + // reserve space + dest.resizeNonZeros(nnz); + dest.outerIndexPtr()[0] = 0; + for (Index j = 0; j < size; ++j) dest.outerIndexPtr()[j + 1] = dest.outerIndexPtr()[j] + count[j]; + for (Index j = 0; j < size; ++j) count[j] = dest.outerIndexPtr()[j]; + + // copy data + for (StorageIndex j = 0; j < size; ++j) { + for (MatIterator it(matEval, j); it; ++it) { + StorageIndex i = internal::convert_index(it.index()); + Index r = it.row(); + Index c = it.col(); + + StorageIndex jp = perm ? perm[j] : j; + StorageIndex ip = perm ? perm[i] : i; + + if (Mode == (Upper | Lower)) { + Index k = count[StorageOrderMatch ? jp : ip]++; + dest.innerIndexPtr()[k] = StorageOrderMatch ? ip : jp; + dest.valuePtr()[k] = it.value(); + } else if (r == c) { + Index k = count[ip]++; + dest.innerIndexPtr()[k] = ip; + dest.valuePtr()[k] = it.value(); + } else if (((Mode & Lower) == Lower && r > c) || ((Mode & Upper) == Upper && r < c)) { + if (!StorageOrderMatch) std::swap(ip, jp); + Index k = count[jp]++; + dest.innerIndexPtr()[k] = ip; + dest.valuePtr()[k] = it.value(); + k = count[ip]++; + dest.innerIndexPtr()[k] = jp; + dest.valuePtr()[k] = numext::conj(it.value()); + } } } } - Index nnz = count.sum(); - - // reserve space - dest.resizeNonZeros(nnz); - dest.outerIndexPtr()[0] = 0; - for(Index j=0; j + void permute_symm_to_symm(const MatrixType &mat, + SparseMatrix &_dest, + const typename MatrixType::StorageIndex *perm) { - for(MatIterator it(matEval,j); it; ++it) - { - StorageIndex i = internal::convert_index(it.index()); - Index r = it.row(); - Index c = it.col(); - + typedef typename MatrixType::StorageIndex StorageIndex; + typedef typename MatrixType::Scalar Scalar; + SparseMatrix &dest(_dest.derived()); + typedef Matrix VectorI; + typedef evaluator MatEval; + typedef typename evaluator::InnerIterator MatIterator; + + enum { + SrcOrder = MatrixType::IsRowMajor ? RowMajor : ColMajor, + StorageOrderMatch = int(SrcOrder) == int(DstOrder), + DstMode = DstOrder == RowMajor ? (_DstMode == Upper ? Lower : Upper) : _DstMode, + SrcMode = SrcOrder == RowMajor ? (_SrcMode == Upper ? Lower : Upper) : _SrcMode + }; + + MatEval matEval(mat); + + Index size = mat.rows(); + VectorI count(size); + count.setZero(); + dest.resize(size, size); + for (StorageIndex j = 0; j < size; ++j) { StorageIndex jp = perm ? perm[j] : j; - StorageIndex ip = perm ? perm[i] : i; - - if(Mode==(Upper|Lower)) - { - Index k = count[StorageOrderMatch ? jp : ip]++; - dest.innerIndexPtr()[k] = StorageOrderMatch ? ip : jp; - dest.valuePtr()[k] = it.value(); - } - else if(r==c) - { - Index k = count[ip]++; - dest.innerIndexPtr()[k] = ip; - dest.valuePtr()[k] = it.value(); - } - else if(( (Mode&Lower)==Lower && r>c) || ( (Mode&Upper)==Upper && r j)) continue; + + StorageIndex ip = perm ? perm[i] : i; + count[int(DstMode) == int(Lower) ? (std::min)(ip, jp) : (std::max)(ip, jp)]++; } } - } -} + dest.outerIndexPtr()[0] = 0; + for (Index j = 0; j < size; ++j) dest.outerIndexPtr()[j + 1] = dest.outerIndexPtr()[j] + count[j]; + dest.resizeNonZeros(dest.outerIndexPtr()[size]); + for (Index j = 0; j < size; ++j) count[j] = dest.outerIndexPtr()[j]; -template -void permute_symm_to_symm(const MatrixType& mat, SparseMatrix& _dest, const typename MatrixType::StorageIndex* perm) -{ - typedef typename MatrixType::StorageIndex StorageIndex; - typedef typename MatrixType::Scalar Scalar; - SparseMatrix& dest(_dest.derived()); - typedef Matrix VectorI; - typedef evaluator MatEval; - typedef typename evaluator::InnerIterator MatIterator; + for (StorageIndex j = 0; j < size; ++j) { - enum { - SrcOrder = MatrixType::IsRowMajor ? RowMajor : ColMajor, - StorageOrderMatch = int(SrcOrder) == int(DstOrder), - DstMode = DstOrder==RowMajor ? (_DstMode==Upper ? Lower : Upper) : _DstMode, - SrcMode = SrcOrder==RowMajor ? (_SrcMode==Upper ? Lower : Upper) : _SrcMode - }; + for (MatIterator it(matEval, j); it; ++it) { + StorageIndex i = it.index(); + if ((int(SrcMode) == int(Lower) && i < j) || (int(SrcMode) == int(Upper) && i > j)) continue; - MatEval matEval(mat); - - Index size = mat.rows(); - VectorI count(size); - count.setZero(); - dest.resize(size,size); - for(StorageIndex j = 0; jj)) - continue; - - StorageIndex ip = perm ? perm[i] : i; - count[int(DstMode)==int(Lower) ? (std::min)(ip,jp) : (std::max)(ip,jp)]++; - } - } - dest.outerIndexPtr()[0] = 0; - for(Index j=0; jj)) - continue; - - StorageIndex jp = perm ? perm[j] : j; - StorageIndex ip = perm? perm[i] : i; - - Index k = count[int(DstMode)==int(Lower) ? (std::min)(ip,jp) : (std::max)(ip,jp)]++; - dest.innerIndexPtr()[k] = int(DstMode)==int(Lower) ? (std::max)(ip,jp) : (std::min)(ip,jp); - - if(!StorageOrderMatch) std::swap(ip,jp); - if( ((int(DstMode)==int(Lower) && ipjp))) - dest.valuePtr()[k] = numext::conj(it.value()); - else - dest.valuePtr()[k] = it.value(); + StorageIndex jp = perm ? perm[j] : j; + StorageIndex ip = perm ? perm[i] : i; + + Index k = count[int(DstMode) == int(Lower) ? (std::min)(ip, jp) : (std::max)(ip, jp)]++; + dest.innerIndexPtr()[k] = int(DstMode) == int(Lower) ? (std::max)(ip, jp) : (std::min)(ip, jp); + + if (!StorageOrderMatch) std::swap(ip, jp); + if (((int(DstMode) == int(Lower) && ip < jp) || (int(DstMode) == int(Upper) && ip > jp))) + dest.valuePtr()[k] = numext::conj(it.value()); + else + dest.valuePtr()[k] = it.value(); + } } } -} -} +}// namespace internal // TODO implement twists in a more evaluator friendly fashion namespace internal { -template -struct traits > : traits { -}; + template + struct traits> : traits + { + }; -} +}// namespace internal -template -class SparseSymmetricPermutationProduct - : public EigenBase > +template +class SparseSymmetricPermutationProduct : public EigenBase> { - public: - typedef typename MatrixType::Scalar Scalar; - typedef typename MatrixType::StorageIndex StorageIndex; - enum { - RowsAtCompileTime = internal::traits::RowsAtCompileTime, - ColsAtCompileTime = internal::traits::ColsAtCompileTime - }; - protected: - typedef PermutationMatrix Perm; - public: - typedef Matrix VectorI; - typedef typename MatrixType::Nested MatrixTypeNested; - typedef typename internal::remove_all::type NestedExpression; - - SparseSymmetricPermutationProduct(const MatrixType& mat, const Perm& perm) - : m_matrix(mat), m_perm(perm) - {} - - inline Index rows() const { return m_matrix.rows(); } - inline Index cols() const { return m_matrix.cols(); } - - const NestedExpression& matrix() const { return m_matrix; } - const Perm& perm() const { return m_perm; } - - protected: - MatrixTypeNested m_matrix; - const Perm& m_perm; +public: + typedef typename MatrixType::Scalar Scalar; + typedef typename MatrixType::StorageIndex StorageIndex; + enum { + RowsAtCompileTime = internal::traits::RowsAtCompileTime, + ColsAtCompileTime = internal::traits::ColsAtCompileTime + }; + +protected: + typedef PermutationMatrix Perm; + +public: + typedef Matrix VectorI; + typedef typename MatrixType::Nested MatrixTypeNested; + typedef typename internal::remove_all::type NestedExpression; + + SparseSymmetricPermutationProduct(const MatrixType &mat, const Perm &perm) : m_matrix(mat), m_perm(perm) {} + + inline Index rows() const { return m_matrix.rows(); } + inline Index cols() const { return m_matrix.cols(); } + + const NestedExpression &matrix() const { return m_matrix; } + const Perm &perm() const { return m_perm; } +protected: + MatrixTypeNested m_matrix; + const Perm &m_perm; }; namespace internal { - -template -struct Assignment, internal::assign_op, Sparse2Sparse> -{ - typedef SparseSymmetricPermutationProduct SrcXprType; - typedef typename DstXprType::StorageIndex DstIndex; - template - static void run(SparseMatrix &dst, const SrcXprType &src, const internal::assign_op &) - { - // internal::permute_symm_to_fullsymm(m_matrix,_dest,m_perm.indices().data()); - SparseMatrix tmp; - internal::permute_symm_to_fullsymm(src.matrix(),tmp,src.perm().indices().data()); - dst = tmp; - } - - template - static void run(SparseSelfAdjointView& dst, const SrcXprType &src, const internal::assign_op &) + + template + struct Assignment, + internal::assign_op, + Sparse2Sparse> { - internal::permute_symm_to_symm(src.matrix(),dst.matrix(),src.perm().indices().data()); - } -}; + typedef SparseSymmetricPermutationProduct SrcXprType; + typedef typename DstXprType::StorageIndex DstIndex; + template + static void run(SparseMatrix &dst, + const SrcXprType &src, + const internal::assign_op &) + { + // internal::permute_symm_to_fullsymm(m_matrix,_dest,m_perm.indices().data()); + SparseMatrix tmp; + internal::permute_symm_to_fullsymm(src.matrix(), tmp, src.perm().indices().data()); + dst = tmp; + } + + template + static void run(SparseSelfAdjointView &dst, + const SrcXprType &src, + const internal::assign_op &) + { + internal::permute_symm_to_symm(src.matrix(), dst.matrix(), src.perm().indices().data()); + } + }; -} // end namespace internal +}// end namespace internal -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_SPARSE_SELFADJOINTVIEW_H +#endif// EIGEN_SPARSE_SELFADJOINTVIEW_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/SparseCore/SparseSolverBase.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/SparseCore/SparseSolverBase.h index b4c9a422..94711b93 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/SparseCore/SparseSolverBase.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/SparseCore/SparseSolverBase.h @@ -10,115 +10,104 @@ #ifndef EIGEN_SPARSESOLVERBASE_H #define EIGEN_SPARSESOLVERBASE_H -namespace Eigen { +namespace Eigen { namespace internal { /** \internal - * Helper functions to solve with a sparse right-hand-side and result. - * The rhs is decomposed into small vertical panels which are solved through dense temporaries. - */ -template -typename enable_if::type -solve_sparse_through_dense_panels(const Decomposition &dec, const Rhs& rhs, Dest &dest) -{ - EIGEN_STATIC_ASSERT((Dest::Flags&RowMajorBit)==0,THIS_METHOD_IS_ONLY_FOR_COLUMN_MAJOR_MATRICES); - typedef typename Dest::Scalar DestScalar; - // we process the sparse rhs per block of NbColsAtOnce columns temporarily stored into a dense matrix. - static const Index NbColsAtOnce = 4; - Index rhsCols = rhs.cols(); - Index size = rhs.rows(); - // the temporary matrices do not need more columns than NbColsAtOnce: - Index tmpCols = (std::min)(rhsCols, NbColsAtOnce); - Eigen::Matrix tmp(size,tmpCols); - Eigen::Matrix tmpX(size,tmpCols); - for(Index k=0; k + typename enable_if::type + solve_sparse_through_dense_panels(const Decomposition &dec, const Rhs &rhs, Dest &dest) { - Index actualCols = std::min(rhsCols-k, NbColsAtOnce); - tmp.leftCols(actualCols) = rhs.middleCols(k,actualCols); - tmpX.leftCols(actualCols) = dec.solve(tmp.leftCols(actualCols)); - dest.middleCols(k,actualCols) = tmpX.leftCols(actualCols).sparseView(); + EIGEN_STATIC_ASSERT((Dest::Flags & RowMajorBit) == 0, THIS_METHOD_IS_ONLY_FOR_COLUMN_MAJOR_MATRICES); + typedef typename Dest::Scalar DestScalar; + // we process the sparse rhs per block of NbColsAtOnce columns temporarily stored into a dense matrix. + static const Index NbColsAtOnce = 4; + Index rhsCols = rhs.cols(); + Index size = rhs.rows(); + // the temporary matrices do not need more columns than NbColsAtOnce: + Index tmpCols = (std::min)(rhsCols, NbColsAtOnce); + Eigen::Matrix tmp(size, tmpCols); + Eigen::Matrix tmpX(size, tmpCols); + for (Index k = 0; k < rhsCols; k += NbColsAtOnce) { + Index actualCols = std::min(rhsCols - k, NbColsAtOnce); + tmp.leftCols(actualCols) = rhs.middleCols(k, actualCols); + tmpX.leftCols(actualCols) = dec.solve(tmp.leftCols(actualCols)); + dest.middleCols(k, actualCols) = tmpX.leftCols(actualCols).sparseView(); + } } -} -// Overload for vector as rhs -template -typename enable_if::type -solve_sparse_through_dense_panels(const Decomposition &dec, const Rhs& rhs, Dest &dest) -{ - typedef typename Dest::Scalar DestScalar; - Index size = rhs.rows(); - Eigen::Matrix rhs_dense(rhs); - Eigen::Matrix dest_dense(size); - dest_dense = dec.solve(rhs_dense); - dest = dest_dense.sparseView(); -} + // Overload for vector as rhs + template + typename enable_if::type + solve_sparse_through_dense_panels(const Decomposition &dec, const Rhs &rhs, Dest &dest) + { + typedef typename Dest::Scalar DestScalar; + Index size = rhs.rows(); + Eigen::Matrix rhs_dense(rhs); + Eigen::Matrix dest_dense(size); + dest_dense = dec.solve(rhs_dense); + dest = dest_dense.sparseView(); + } -} // end namespace internal +}// end namespace internal /** \class SparseSolverBase - * \ingroup SparseCore_Module - * \brief A base class for sparse solvers - * - * \tparam Derived the actual type of the solver. - * - */ -template -class SparseSolverBase : internal::noncopyable + * \ingroup SparseCore_Module + * \brief A base class for sparse solvers + * + * \tparam Derived the actual type of the solver. + * + */ +template class SparseSolverBase : internal::noncopyable { - public: +public: + /** Default constructor */ + SparseSolverBase() : m_isInitialized(false) {} - /** Default constructor */ - SparseSolverBase() - : m_isInitialized(false) - {} + ~SparseSolverBase() {} - ~SparseSolverBase() - {} + Derived &derived() { return *static_cast(this); } + const Derived &derived() const { return *static_cast(this); } - Derived& derived() { return *static_cast(this); } - const Derived& derived() const { return *static_cast(this); } - - /** \returns an expression of the solution x of \f$ A x = b \f$ using the current decomposition of A. - * - * \sa compute() - */ - template - inline const Solve - solve(const MatrixBase& b) const - { - eigen_assert(m_isInitialized && "Solver is not initialized."); - eigen_assert(derived().rows()==b.rows() && "solve(): invalid number of rows of the right hand side matrix b"); - return Solve(derived(), b.derived()); - } - - /** \returns an expression of the solution x of \f$ A x = b \f$ using the current decomposition of A. - * - * \sa compute() - */ - template - inline const Solve - solve(const SparseMatrixBase& b) const - { - eigen_assert(m_isInitialized && "Solver is not initialized."); - eigen_assert(derived().rows()==b.rows() && "solve(): invalid number of rows of the right hand side matrix b"); - return Solve(derived(), b.derived()); - } - - #ifndef EIGEN_PARSED_BY_DOXYGEN - /** \internal default implementation of solving with a sparse rhs */ - template - void _solve_impl(const SparseMatrixBase &b, SparseMatrixBase &dest) const - { - internal::solve_sparse_through_dense_panels(derived(), b.derived(), dest.derived()); - } - #endif // EIGEN_PARSED_BY_DOXYGEN + /** \returns an expression of the solution x of \f$ A x = b \f$ using the current decomposition of A. + * + * \sa compute() + */ + template inline const Solve solve(const MatrixBase &b) const + { + eigen_assert(m_isInitialized && "Solver is not initialized."); + eigen_assert(derived().rows() == b.rows() && "solve(): invalid number of rows of the right hand side matrix b"); + return Solve(derived(), b.derived()); + } + + /** \returns an expression of the solution x of \f$ A x = b \f$ using the current decomposition of A. + * + * \sa compute() + */ + template inline const Solve solve(const SparseMatrixBase &b) const + { + eigen_assert(m_isInitialized && "Solver is not initialized."); + eigen_assert(derived().rows() == b.rows() && "solve(): invalid number of rows of the right hand side matrix b"); + return Solve(derived(), b.derived()); + } + +#ifndef EIGEN_PARSED_BY_DOXYGEN + /** \internal default implementation of solving with a sparse rhs */ + template + void _solve_impl(const SparseMatrixBase &b, SparseMatrixBase &dest) const + { + internal::solve_sparse_through_dense_panels(derived(), b.derived(), dest.derived()); + } +#endif// EIGEN_PARSED_BY_DOXYGEN - protected: - - mutable bool m_isInitialized; +protected: + mutable bool m_isInitialized; }; -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_SPARSESOLVERBASE_H +#endif// EIGEN_SPARSESOLVERBASE_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/SparseCore/SparseSparseProductWithPruning.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/SparseCore/SparseSparseProductWithPruning.h index 88820a48..fe4d965c 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/SparseCore/SparseSparseProductWithPruning.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/SparseCore/SparseSparseProductWithPruning.h @@ -10,189 +10,198 @@ #ifndef EIGEN_SPARSESPARSEPRODUCTWITHPRUNING_H #define EIGEN_SPARSESPARSEPRODUCTWITHPRUNING_H -namespace Eigen { +namespace Eigen { namespace internal { -// perform a pseudo in-place sparse * sparse product assuming all matrices are col major -template -static void sparse_sparse_product_with_pruning_impl(const Lhs& lhs, const Rhs& rhs, ResultType& res, const typename ResultType::RealScalar& tolerance) -{ - // return sparse_sparse_product_with_pruning_impl2(lhs,rhs,res); - - typedef typename remove_all::type::Scalar RhsScalar; - typedef typename remove_all::type::Scalar ResScalar; - typedef typename remove_all::type::StorageIndex StorageIndex; - - // make sure to call innerSize/outerSize since we fake the storage order. - Index rows = lhs.innerSize(); - Index cols = rhs.outerSize(); - //Index size = lhs.outerSize(); - eigen_assert(lhs.outerSize() == rhs.innerSize()); - - // allocate a temporary buffer - AmbiVector tempVector(rows); - - // mimics a resizeByInnerOuter: - if(ResultType::IsRowMajor) - res.resize(cols, rows); - else - res.resize(rows, cols); - - evaluator lhsEval(lhs); - evaluator rhsEval(rhs); - - // estimate the number of non zero entries - // given a rhs column containing Y non zeros, we assume that the respective Y columns - // of the lhs differs in average of one non zeros, thus the number of non zeros for - // the product of a rhs column with the lhs is X+Y where X is the average number of non zero - // per column of the lhs. - // Therefore, we have nnz(lhs*rhs) = nnz(lhs) + nnz(rhs) - Index estimated_nnz_prod = lhsEval.nonZerosEstimate() + rhsEval.nonZerosEstimate(); - - res.reserve(estimated_nnz_prod); - double ratioColRes = double(estimated_nnz_prod)/(double(lhs.rows())*double(rhs.cols())); - for (Index j=0; j + static void sparse_sparse_product_with_pruning_impl(const Lhs &lhs, + const Rhs &rhs, + ResultType &res, + const typename ResultType::RealScalar &tolerance) { - // FIXME: - //double ratioColRes = (double(rhs.innerVector(j).nonZeros()) + double(lhs.nonZeros())/double(lhs.cols()))/double(lhs.rows()); - // let's do a more accurate determination of the nnz ratio for the current column j of res - tempVector.init(ratioColRes); - tempVector.setZero(); - for (typename evaluator::InnerIterator rhsIt(rhsEval, j); rhsIt; ++rhsIt) - { - // FIXME should be written like this: tmp += rhsIt.value() * lhs.col(rhsIt.index()) - tempVector.restart(); - RhsScalar x = rhsIt.value(); - for (typename evaluator::InnerIterator lhsIt(lhsEval, rhsIt.index()); lhsIt; ++lhsIt) - { - tempVector.coeffRef(lhsIt.index()) += lhsIt.value() * x; + // return sparse_sparse_product_with_pruning_impl2(lhs,rhs,res); + + typedef typename remove_all::type::Scalar RhsScalar; + typedef typename remove_all::type::Scalar ResScalar; + typedef typename remove_all::type::StorageIndex StorageIndex; + + // make sure to call innerSize/outerSize since we fake the storage order. + Index rows = lhs.innerSize(); + Index cols = rhs.outerSize(); + // Index size = lhs.outerSize(); + eigen_assert(lhs.outerSize() == rhs.innerSize()); + + // allocate a temporary buffer + AmbiVector tempVector(rows); + + // mimics a resizeByInnerOuter: + if (ResultType::IsRowMajor) + res.resize(cols, rows); + else + res.resize(rows, cols); + + evaluator lhsEval(lhs); + evaluator rhsEval(rhs); + + // estimate the number of non zero entries + // given a rhs column containing Y non zeros, we assume that the respective Y columns + // of the lhs differs in average of one non zeros, thus the number of non zeros for + // the product of a rhs column with the lhs is X+Y where X is the average number of non zero + // per column of the lhs. + // Therefore, we have nnz(lhs*rhs) = nnz(lhs) + nnz(rhs) + Index estimated_nnz_prod = lhsEval.nonZerosEstimate() + rhsEval.nonZerosEstimate(); + + res.reserve(estimated_nnz_prod); + double ratioColRes = double(estimated_nnz_prod) / (double(lhs.rows()) * double(rhs.cols())); + for (Index j = 0; j < cols; ++j) { + // FIXME: + // double ratioColRes = (double(rhs.innerVector(j).nonZeros()) + + // double(lhs.nonZeros())/double(lhs.cols()))/double(lhs.rows()); + // let's do a more accurate determination of the nnz ratio for the current column j of res + tempVector.init(ratioColRes); + tempVector.setZero(); + for (typename evaluator::InnerIterator rhsIt(rhsEval, j); rhsIt; ++rhsIt) { + // FIXME should be written like this: tmp += rhsIt.value() * lhs.col(rhsIt.index()) + tempVector.restart(); + RhsScalar x = rhsIt.value(); + for (typename evaluator::InnerIterator lhsIt(lhsEval, rhsIt.index()); lhsIt; ++lhsIt) { + tempVector.coeffRef(lhsIt.index()) += lhsIt.value() * x; + } } + res.startVec(j); + for (typename AmbiVector::Iterator it(tempVector, tolerance); it; ++it) + res.insertBackByOuterInner(j, it.index()) = it.value(); } - res.startVec(j); - for (typename AmbiVector::Iterator it(tempVector,tolerance); it; ++it) - res.insertBackByOuterInner(j,it.index()) = it.value(); + res.finalize(); } - res.finalize(); -} - -template::Flags&RowMajorBit, - int RhsStorageOrder = traits::Flags&RowMajorBit, - int ResStorageOrder = traits::Flags&RowMajorBit> -struct sparse_sparse_product_with_pruning_selector; -template -struct sparse_sparse_product_with_pruning_selector -{ - typedef typename ResultType::RealScalar RealScalar; + template::Flags & RowMajorBit, + int RhsStorageOrder = traits::Flags & RowMajorBit, + int ResStorageOrder = traits::Flags & RowMajorBit> + struct sparse_sparse_product_with_pruning_selector; - static void run(const Lhs& lhs, const Rhs& rhs, ResultType& res, const RealScalar& tolerance) + template + struct sparse_sparse_product_with_pruning_selector { - typename remove_all::type _res(res.rows(), res.cols()); - internal::sparse_sparse_product_with_pruning_impl(lhs, rhs, _res, tolerance); - res.swap(_res); - } -}; + typedef typename ResultType::RealScalar RealScalar; -template -struct sparse_sparse_product_with_pruning_selector -{ - typedef typename ResultType::RealScalar RealScalar; - static void run(const Lhs& lhs, const Rhs& rhs, ResultType& res, const RealScalar& tolerance) + static void run(const Lhs &lhs, const Rhs &rhs, ResultType &res, const RealScalar &tolerance) + { + typename remove_all::type _res(res.rows(), res.cols()); + internal::sparse_sparse_product_with_pruning_impl(lhs, rhs, _res, tolerance); + res.swap(_res); + } + }; + + template + struct sparse_sparse_product_with_pruning_selector { - // we need a col-major matrix to hold the result - typedef SparseMatrix SparseTemporaryType; - SparseTemporaryType _res(res.rows(), res.cols()); - internal::sparse_sparse_product_with_pruning_impl(lhs, rhs, _res, tolerance); - res = _res; - } -}; + typedef typename ResultType::RealScalar RealScalar; + static void run(const Lhs &lhs, const Rhs &rhs, ResultType &res, const RealScalar &tolerance) + { + // we need a col-major matrix to hold the result + typedef SparseMatrix + SparseTemporaryType; + SparseTemporaryType _res(res.rows(), res.cols()); + internal::sparse_sparse_product_with_pruning_impl(lhs, rhs, _res, tolerance); + res = _res; + } + }; -template -struct sparse_sparse_product_with_pruning_selector -{ - typedef typename ResultType::RealScalar RealScalar; - static void run(const Lhs& lhs, const Rhs& rhs, ResultType& res, const RealScalar& tolerance) + template + struct sparse_sparse_product_with_pruning_selector { - // let's transpose the product to get a column x column product - typename remove_all::type _res(res.rows(), res.cols()); - internal::sparse_sparse_product_with_pruning_impl(rhs, lhs, _res, tolerance); - res.swap(_res); - } -}; + typedef typename ResultType::RealScalar RealScalar; + static void run(const Lhs &lhs, const Rhs &rhs, ResultType &res, const RealScalar &tolerance) + { + // let's transpose the product to get a column x column product + typename remove_all::type _res(res.rows(), res.cols()); + internal::sparse_sparse_product_with_pruning_impl(rhs, lhs, _res, tolerance); + res.swap(_res); + } + }; -template -struct sparse_sparse_product_with_pruning_selector -{ - typedef typename ResultType::RealScalar RealScalar; - static void run(const Lhs& lhs, const Rhs& rhs, ResultType& res, const RealScalar& tolerance) + template + struct sparse_sparse_product_with_pruning_selector { - typedef SparseMatrix ColMajorMatrixLhs; - typedef SparseMatrix ColMajorMatrixRhs; - ColMajorMatrixLhs colLhs(lhs); - ColMajorMatrixRhs colRhs(rhs); - internal::sparse_sparse_product_with_pruning_impl(colLhs, colRhs, res, tolerance); - - // let's transpose the product to get a column x column product -// typedef SparseMatrix SparseTemporaryType; -// SparseTemporaryType _res(res.cols(), res.rows()); -// sparse_sparse_product_with_pruning_impl(rhs, lhs, _res); -// res = _res.transpose(); - } -}; + typedef typename ResultType::RealScalar RealScalar; + static void run(const Lhs &lhs, const Rhs &rhs, ResultType &res, const RealScalar &tolerance) + { + typedef SparseMatrix ColMajorMatrixLhs; + typedef SparseMatrix ColMajorMatrixRhs; + ColMajorMatrixLhs colLhs(lhs); + ColMajorMatrixRhs colRhs(rhs); + internal::sparse_sparse_product_with_pruning_impl( + colLhs, colRhs, res, tolerance); + + // let's transpose the product to get a column x column product + // typedef SparseMatrix SparseTemporaryType; + // SparseTemporaryType _res(res.cols(), res.rows()); + // sparse_sparse_product_with_pruning_impl(rhs, lhs, _res); + // res = _res.transpose(); + } + }; -template -struct sparse_sparse_product_with_pruning_selector -{ - typedef typename ResultType::RealScalar RealScalar; - static void run(const Lhs& lhs, const Rhs& rhs, ResultType& res, const RealScalar& tolerance) + template + struct sparse_sparse_product_with_pruning_selector { - typedef SparseMatrix RowMajorMatrixLhs; - RowMajorMatrixLhs rowLhs(lhs); - sparse_sparse_product_with_pruning_selector(rowLhs,rhs,res,tolerance); - } -}; + typedef typename ResultType::RealScalar RealScalar; + static void run(const Lhs &lhs, const Rhs &rhs, ResultType &res, const RealScalar &tolerance) + { + typedef SparseMatrix RowMajorMatrixLhs; + RowMajorMatrixLhs rowLhs(lhs); + sparse_sparse_product_with_pruning_selector( + rowLhs, rhs, res, tolerance); + } + }; -template -struct sparse_sparse_product_with_pruning_selector -{ - typedef typename ResultType::RealScalar RealScalar; - static void run(const Lhs& lhs, const Rhs& rhs, ResultType& res, const RealScalar& tolerance) + template + struct sparse_sparse_product_with_pruning_selector { - typedef SparseMatrix RowMajorMatrixRhs; - RowMajorMatrixRhs rowRhs(rhs); - sparse_sparse_product_with_pruning_selector(lhs,rowRhs,res,tolerance); - } -}; + typedef typename ResultType::RealScalar RealScalar; + static void run(const Lhs &lhs, const Rhs &rhs, ResultType &res, const RealScalar &tolerance) + { + typedef SparseMatrix RowMajorMatrixRhs; + RowMajorMatrixRhs rowRhs(rhs); + sparse_sparse_product_with_pruning_selector( + lhs, rowRhs, res, tolerance); + } + }; -template -struct sparse_sparse_product_with_pruning_selector -{ - typedef typename ResultType::RealScalar RealScalar; - static void run(const Lhs& lhs, const Rhs& rhs, ResultType& res, const RealScalar& tolerance) + template + struct sparse_sparse_product_with_pruning_selector { - typedef SparseMatrix ColMajorMatrixRhs; - ColMajorMatrixRhs colRhs(rhs); - internal::sparse_sparse_product_with_pruning_impl(lhs, colRhs, res, tolerance); - } -}; + typedef typename ResultType::RealScalar RealScalar; + static void run(const Lhs &lhs, const Rhs &rhs, ResultType &res, const RealScalar &tolerance) + { + typedef SparseMatrix ColMajorMatrixRhs; + ColMajorMatrixRhs colRhs(rhs); + internal::sparse_sparse_product_with_pruning_impl( + lhs, colRhs, res, tolerance); + } + }; -template -struct sparse_sparse_product_with_pruning_selector -{ - typedef typename ResultType::RealScalar RealScalar; - static void run(const Lhs& lhs, const Rhs& rhs, ResultType& res, const RealScalar& tolerance) + template + struct sparse_sparse_product_with_pruning_selector { - typedef SparseMatrix ColMajorMatrixLhs; - ColMajorMatrixLhs colLhs(lhs); - internal::sparse_sparse_product_with_pruning_impl(colLhs, rhs, res, tolerance); - } -}; + typedef typename ResultType::RealScalar RealScalar; + static void run(const Lhs &lhs, const Rhs &rhs, ResultType &res, const RealScalar &tolerance) + { + typedef SparseMatrix ColMajorMatrixLhs; + ColMajorMatrixLhs colLhs(lhs); + internal::sparse_sparse_product_with_pruning_impl( + colLhs, rhs, res, tolerance); + } + }; -} // end namespace internal +}// end namespace internal -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_SPARSESPARSEPRODUCTWITHPRUNING_H +#endif// EIGEN_SPARSESPARSEPRODUCTWITHPRUNING_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/SparseCore/SparseTranspose.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/SparseCore/SparseTranspose.h index 3757d4c6..1ead72dd 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/SparseCore/SparseTranspose.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/SparseCore/SparseTranspose.h @@ -10,83 +10,77 @@ #ifndef EIGEN_SPARSETRANSPOSE_H #define EIGEN_SPARSETRANSPOSE_H -namespace Eigen { +namespace Eigen { namespace internal { - template - class SparseTransposeImpl - : public SparseMatrixBase > - {}; - + template + class SparseTransposeImpl : public SparseMatrixBase> + { + }; + template - class SparseTransposeImpl - : public SparseCompressedBase > + class SparseTransposeImpl : public SparseCompressedBase> { - typedef SparseCompressedBase > Base; + typedef SparseCompressedBase> Base; + public: using Base::derived; typedef typename Base::Scalar Scalar; typedef typename Base::StorageIndex StorageIndex; inline Index nonZeros() const { return derived().nestedExpression().nonZeros(); } - - inline const Scalar* valuePtr() const { return derived().nestedExpression().valuePtr(); } - inline const StorageIndex* innerIndexPtr() const { return derived().nestedExpression().innerIndexPtr(); } - inline const StorageIndex* outerIndexPtr() const { return derived().nestedExpression().outerIndexPtr(); } - inline const StorageIndex* innerNonZeroPtr() const { return derived().nestedExpression().innerNonZeroPtr(); } - - inline Scalar* valuePtr() { return derived().nestedExpression().valuePtr(); } - inline StorageIndex* innerIndexPtr() { return derived().nestedExpression().innerIndexPtr(); } - inline StorageIndex* outerIndexPtr() { return derived().nestedExpression().outerIndexPtr(); } - inline StorageIndex* innerNonZeroPtr() { return derived().nestedExpression().innerNonZeroPtr(); } + + inline const Scalar *valuePtr() const { return derived().nestedExpression().valuePtr(); } + inline const StorageIndex *innerIndexPtr() const { return derived().nestedExpression().innerIndexPtr(); } + inline const StorageIndex *outerIndexPtr() const { return derived().nestedExpression().outerIndexPtr(); } + inline const StorageIndex *innerNonZeroPtr() const { return derived().nestedExpression().innerNonZeroPtr(); } + + inline Scalar *valuePtr() { return derived().nestedExpression().valuePtr(); } + inline StorageIndex *innerIndexPtr() { return derived().nestedExpression().innerIndexPtr(); } + inline StorageIndex *outerIndexPtr() { return derived().nestedExpression().outerIndexPtr(); } + inline StorageIndex *innerNonZeroPtr() { return derived().nestedExpression().innerNonZeroPtr(); } }; -} - -template class TransposeImpl - : public internal::SparseTransposeImpl +}// namespace internal + +template class TransposeImpl : public internal::SparseTransposeImpl { - protected: - typedef internal::SparseTransposeImpl Base; +protected: + typedef internal::SparseTransposeImpl Base; }; namespace internal { - -template -struct unary_evaluator, IteratorBased> - : public evaluator_base > -{ - typedef typename evaluator::InnerIterator EvalIterator; + + template + struct unary_evaluator, IteratorBased> : public evaluator_base> + { + typedef typename evaluator::InnerIterator EvalIterator; + public: typedef Transpose XprType; - - inline Index nonZerosEstimate() const { - return m_argImpl.nonZerosEstimate(); - } + + inline Index nonZerosEstimate() const { return m_argImpl.nonZerosEstimate(); } class InnerIterator : public EvalIterator { public: - EIGEN_STRONG_INLINE InnerIterator(const unary_evaluator& unaryOp, Index outer) - : EvalIterator(unaryOp.m_argImpl,outer) + EIGEN_STRONG_INLINE InnerIterator(const unary_evaluator &unaryOp, Index outer) + : EvalIterator(unaryOp.m_argImpl, outer) {} - + Index row() const { return EvalIterator::col(); } Index col() const { return EvalIterator::row(); } }; - - enum { - CoeffReadCost = evaluator::CoeffReadCost, - Flags = XprType::Flags - }; - - explicit unary_evaluator(const XprType& op) :m_argImpl(op.nestedExpression()) {} + + enum { CoeffReadCost = evaluator::CoeffReadCost, Flags = XprType::Flags }; + + explicit unary_evaluator(const XprType &op) : m_argImpl(op.nestedExpression()) {} protected: evaluator m_argImpl; -}; + }; -} // end namespace internal +}// end namespace internal -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_SPARSETRANSPOSE_H +#endif// EIGEN_SPARSETRANSPOSE_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/SparseCore/SparseTriangularView.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/SparseCore/SparseTriangularView.h index 9ac12026..bfd0f0bb 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/SparseCore/SparseTriangularView.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/SparseCore/SparseTriangularView.h @@ -14,176 +14,167 @@ namespace Eigen { /** \ingroup SparseCore_Module - * - * \brief Base class for a triangular part in a \b sparse matrix - * - * This class is an abstract base class of class TriangularView, and objects of type TriangularViewImpl cannot be instantiated. - * It extends class TriangularView with additional methods which are available for sparse expressions only. - * - * \sa class TriangularView, SparseMatrixBase::triangularView() - */ -template class TriangularViewImpl - : public SparseMatrixBase > + * + * \brief Base class for a triangular part in a \b sparse matrix + * + * This class is an abstract base class of class TriangularView, and objects of type TriangularViewImpl cannot be + * instantiated. It extends class TriangularView with additional methods which are available for sparse expressions + * only. + * + * \sa class TriangularView, SparseMatrixBase::triangularView() + */ +template +class TriangularViewImpl : public SparseMatrixBase> { - enum { SkipFirst = ((Mode&Lower) && !(MatrixType::Flags&RowMajorBit)) - || ((Mode&Upper) && (MatrixType::Flags&RowMajorBit)), - SkipLast = !SkipFirst, - SkipDiag = (Mode&ZeroDiag) ? 1 : 0, - HasUnitDiag = (Mode&UnitDiag) ? 1 : 0 - }; - - typedef TriangularView TriangularViewType; - - protected: - // dummy solve function to make TriangularView happy. - void solve() const; - - typedef SparseMatrixBase Base; - public: - - EIGEN_SPARSE_PUBLIC_INTERFACE(TriangularViewType) - - typedef typename MatrixType::Nested MatrixTypeNested; - typedef typename internal::remove_reference::type MatrixTypeNestedNonRef; - typedef typename internal::remove_all::type MatrixTypeNestedCleaned; - - template - EIGEN_DEVICE_FUNC - EIGEN_STRONG_INLINE void _solve_impl(const RhsType &rhs, DstType &dst) const { - if(!(internal::is_same::value && internal::extract_data(dst) == internal::extract_data(rhs))) - dst = rhs; - this->solveInPlace(dst); - } - - /** Applies the inverse of \c *this to the dense vector or matrix \a other, "in-place" */ - template void solveInPlace(MatrixBase& other) const; - - /** Applies the inverse of \c *this to the sparse vector or matrix \a other, "in-place" */ - template void solveInPlace(SparseMatrixBase& other) const; - -}; + enum { + SkipFirst = + ((Mode & Lower) && !(MatrixType::Flags & RowMajorBit)) || ((Mode & Upper) && (MatrixType::Flags & RowMajorBit)), + SkipLast = !SkipFirst, + SkipDiag = (Mode & ZeroDiag) ? 1 : 0, + HasUnitDiag = (Mode & UnitDiag) ? 1 : 0 + }; -namespace internal { + typedef TriangularView TriangularViewType; -template -struct unary_evaluator, IteratorBased> - : evaluator_base > -{ - typedef TriangularView XprType; - protected: - - typedef typename XprType::Scalar Scalar; - typedef typename XprType::StorageIndex StorageIndex; - typedef typename evaluator::InnerIterator EvalIterator; - - enum { SkipFirst = ((Mode&Lower) && !(ArgType::Flags&RowMajorBit)) - || ((Mode&Upper) && (ArgType::Flags&RowMajorBit)), - SkipLast = !SkipFirst, - SkipDiag = (Mode&ZeroDiag) ? 1 : 0, - HasUnitDiag = (Mode&UnitDiag) ? 1 : 0 - }; - + // dummy solve function to make TriangularView happy. + void solve() const; + + typedef SparseMatrixBase Base; + public: - - enum { - CoeffReadCost = evaluator::CoeffReadCost, - Flags = XprType::Flags - }; - - explicit unary_evaluator(const XprType &xpr) : m_argImpl(xpr.nestedExpression()), m_arg(xpr.nestedExpression()) {} - - inline Index nonZerosEstimate() const { - return m_argImpl.nonZerosEstimate(); + EIGEN_SPARSE_PUBLIC_INTERFACE(TriangularViewType) + + typedef typename MatrixType::Nested MatrixTypeNested; + typedef typename internal::remove_reference::type MatrixTypeNestedNonRef; + typedef typename internal::remove_all::type MatrixTypeNestedCleaned; + + template + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void _solve_impl(const RhsType &rhs, DstType &dst) const + { + if (!(internal::is_same::value && internal::extract_data(dst) == internal::extract_data(rhs))) + dst = rhs; + this->solveInPlace(dst); } - - class InnerIterator : public EvalIterator + + /** Applies the inverse of \c *this to the dense vector or matrix \a other, "in-place" */ + template void solveInPlace(MatrixBase &other) const; + + /** Applies the inverse of \c *this to the sparse vector or matrix \a other, "in-place" */ + template void solveInPlace(SparseMatrixBase &other) const; +}; + +namespace internal { + + template + struct unary_evaluator, IteratorBased> : evaluator_base> { + typedef TriangularView XprType; + + protected: + typedef typename XprType::Scalar Scalar; + typedef typename XprType::StorageIndex StorageIndex; + typedef typename evaluator::InnerIterator EvalIterator; + + enum { + SkipFirst = + ((Mode & Lower) && !(ArgType::Flags & RowMajorBit)) || ((Mode & Upper) && (ArgType::Flags & RowMajorBit)), + SkipLast = !SkipFirst, + SkipDiag = (Mode & ZeroDiag) ? 1 : 0, + HasUnitDiag = (Mode & UnitDiag) ? 1 : 0 + }; + + public: + enum { CoeffReadCost = evaluator::CoeffReadCost, Flags = XprType::Flags }; + + explicit unary_evaluator(const XprType &xpr) : m_argImpl(xpr.nestedExpression()), m_arg(xpr.nestedExpression()) {} + + inline Index nonZerosEstimate() const { return m_argImpl.nonZerosEstimate(); } + + class InnerIterator : public EvalIterator + { typedef EvalIterator Base; - public: - EIGEN_STRONG_INLINE InnerIterator(const unary_evaluator& xprEval, Index outer) - : Base(xprEval.m_argImpl,outer), m_returnOne(false), m_containsDiag(Base::outer()index()<=outer : this->index()=Base::outer())) - { - if((!SkipFirst) && Base::operator bool()) + if (SkipFirst) { + while ((*this) && ((HasUnitDiag || SkipDiag) ? this->index() <= outer : this->index() < outer)) Base::operator++(); + if (HasUnitDiag) m_returnOne = m_containsDiag; + } else if (HasUnitDiag && ((!Base::operator bool()) || Base::index() >= Base::outer())) { + if ((!SkipFirst) && Base::operator bool()) Base::operator++(); m_returnOne = m_containsDiag; } } - EIGEN_STRONG_INLINE InnerIterator& operator++() + EIGEN_STRONG_INLINE InnerIterator &operator++() { - if(HasUnitDiag && m_returnOne) + if (HasUnitDiag && m_returnOne) m_returnOne = false; - else - { + else { Base::operator++(); - if(HasUnitDiag && (!SkipFirst) && ((!Base::operator bool()) || Base::index()>=Base::outer())) - { - if((!SkipFirst) && Base::operator bool()) - Base::operator++(); + if (HasUnitDiag && (!SkipFirst) && ((!Base::operator bool()) || Base::index() >= Base::outer())) { + if ((!SkipFirst) && Base::operator bool()) Base::operator++(); m_returnOne = m_containsDiag; } } return *this; } - + EIGEN_STRONG_INLINE operator bool() const { - if(HasUnitDiag && m_returnOne) - return true; - if(SkipFirst) return Base::operator bool(); - else - { - if (SkipDiag) return (Base::operator bool() && this->index() < this->outer()); - else return (Base::operator bool() && this->index() <= this->outer()); + if (HasUnitDiag && m_returnOne) return true; + if (SkipFirst) + return Base::operator bool(); + else { + if (SkipDiag) + return (Base::operator bool() && this->index() < this->outer()); + else + return (Base::operator bool() && this->index() <= this->outer()); } } -// inline Index row() const { return (ArgType::Flags&RowMajorBit ? Base::outer() : this->index()); } -// inline Index col() const { return (ArgType::Flags&RowMajorBit ? this->index() : Base::outer()); } + // inline Index row() const { return (ArgType::Flags&RowMajorBit ? Base::outer() : this->index()); } + // inline Index col() const { return (ArgType::Flags&RowMajorBit ? this->index() : Base::outer()); } inline StorageIndex index() const { - if(HasUnitDiag && m_returnOne) return internal::convert_index(Base::outer()); - else return Base::index(); + if (HasUnitDiag && m_returnOne) + return internal::convert_index(Base::outer()); + else + return Base::index(); } inline Scalar value() const { - if(HasUnitDiag && m_returnOne) return Scalar(1); - else return Base::value(); + if (HasUnitDiag && m_returnOne) + return Scalar(1); + else + return Base::value(); } protected: bool m_returnOne; bool m_containsDiag; + private: - Scalar& valueRef(); + Scalar &valueRef(); + }; + + protected: + evaluator m_argImpl; + const ArgType &m_arg; }; - -protected: - evaluator m_argImpl; - const ArgType& m_arg; -}; -} // end namespace internal +}// end namespace internal template template -inline const TriangularView -SparseMatrixBase::triangularView() const +inline const TriangularView SparseMatrixBase::triangularView() const { return TriangularView(derived()); } -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_SPARSE_TRIANGULARVIEW_H +#endif// EIGEN_SPARSE_TRIANGULARVIEW_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/SparseCore/SparseUtil.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/SparseCore/SparseUtil.h index 74df0d49..bdda2a70 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/SparseCore/SparseUtil.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/SparseCore/SparseUtil.h @@ -10,7 +10,7 @@ #ifndef EIGEN_SPARSEUTIL_H #define EIGEN_SPARSEUTIL_H -namespace Eigen { +namespace Eigen { #ifdef NDEBUG #define EIGEN_DBG_SPARSE(X) @@ -18,161 +18,182 @@ namespace Eigen { #define EIGEN_DBG_SPARSE(X) X #endif -#define EIGEN_SPARSE_INHERIT_ASSIGNMENT_OPERATOR(Derived, Op) \ -template \ -EIGEN_STRONG_INLINE Derived& operator Op(const Eigen::SparseMatrixBase& other) \ -{ \ - return Base::operator Op(other.derived()); \ -} \ -EIGEN_STRONG_INLINE Derived& operator Op(const Derived& other) \ -{ \ - return Base::operator Op(other); \ -} - -#define EIGEN_SPARSE_INHERIT_SCALAR_ASSIGNMENT_OPERATOR(Derived, Op) \ -template \ -EIGEN_STRONG_INLINE Derived& operator Op(const Other& scalar) \ -{ \ - return Base::operator Op(scalar); \ -} - -#define EIGEN_SPARSE_INHERIT_ASSIGNMENT_OPERATORS(Derived) \ -EIGEN_SPARSE_INHERIT_ASSIGNMENT_OPERATOR(Derived, =) - - -#define EIGEN_SPARSE_PUBLIC_INTERFACE(Derived) \ - EIGEN_GENERIC_PUBLIC_INTERFACE(Derived) - - -const int CoherentAccessPattern = 0x1; -const int InnerRandomAccessPattern = 0x2 | CoherentAccessPattern; -const int OuterRandomAccessPattern = 0x4 | CoherentAccessPattern; -const int RandomAccessPattern = 0x8 | OuterRandomAccessPattern | InnerRandomAccessPattern; - -template class SparseMatrix; -template class DynamicSparseMatrix; -template class SparseVector; -template class MappedSparseMatrix; - -template class SparseSelfAdjointView; -template class SparseDiagonalProduct; +#define EIGEN_SPARSE_INHERIT_ASSIGNMENT_OPERATOR(Derived, Op) \ + template \ + EIGEN_STRONG_INLINE Derived &operator Op(const Eigen::SparseMatrixBase &other) \ + { \ + return Base::operator Op(other.derived()); \ + } \ + EIGEN_STRONG_INLINE Derived &operator Op(const Derived &other) { return Base::operator Op(other); } + +#define EIGEN_SPARSE_INHERIT_SCALAR_ASSIGNMENT_OPERATOR(Derived, Op) \ + template EIGEN_STRONG_INLINE Derived &operator Op(const Other &scalar) \ + { \ + return Base::operator Op(scalar); \ + } + +#define EIGEN_SPARSE_INHERIT_ASSIGNMENT_OPERATORS(Derived) EIGEN_SPARSE_INHERIT_ASSIGNMENT_OPERATOR(Derived, =) + + +#define EIGEN_SPARSE_PUBLIC_INTERFACE(Derived) EIGEN_GENERIC_PUBLIC_INTERFACE(Derived) + + +const int CoherentAccessPattern = 0x1; +const int InnerRandomAccessPattern = 0x2 | CoherentAccessPattern; +const int OuterRandomAccessPattern = 0x4 | CoherentAccessPattern; +const int RandomAccessPattern = 0x8 | OuterRandomAccessPattern | InnerRandomAccessPattern; + +template class SparseMatrix; +template class DynamicSparseMatrix; +template class SparseVector; +template class MappedSparseMatrix; + +template class SparseSelfAdjointView; +template class SparseDiagonalProduct; template class SparseView; -template class SparseSparseProduct; -template class SparseTimeDenseProduct; -template class DenseTimeSparseProduct; +template class SparseSparseProduct; +template class SparseTimeDenseProduct; +template class DenseTimeSparseProduct; template class SparseDenseOuterProduct; template struct SparseSparseProductReturnType; -template::ColsAtCompileTime,internal::traits::RowsAtCompileTime)> struct DenseSparseProductReturnType; - -template::ColsAtCompileTime,internal::traits::RowsAtCompileTime)> struct SparseDenseProductReturnType; -template class SparseSymmetricPermutationProduct; +template::ColsAtCompileTime, internal::traits::RowsAtCompileTime)> +struct DenseSparseProductReturnType; + +template::ColsAtCompileTime, internal::traits::RowsAtCompileTime)> +struct SparseDenseProductReturnType; +template class SparseSymmetricPermutationProduct; namespace internal { -template struct sparse_eval; + template struct sparse_eval; -template struct eval - : sparse_eval::RowsAtCompileTime,traits::ColsAtCompileTime,traits::Flags> -{}; + template + struct eval : sparse_eval::RowsAtCompileTime, traits::ColsAtCompileTime, traits::Flags> + { + }; -template struct sparse_eval { + template struct sparse_eval + { typedef typename traits::Scalar _Scalar; typedef typename traits::StorageIndex _StorageIndex; + public: typedef SparseVector<_Scalar, RowMajor, _StorageIndex> type; -}; + }; -template struct sparse_eval { + template struct sparse_eval + { typedef typename traits::Scalar _Scalar; typedef typename traits::StorageIndex _StorageIndex; + public: typedef SparseVector<_Scalar, ColMajor, _StorageIndex> type; -}; + }; -// TODO this seems almost identical to plain_matrix_type -template struct sparse_eval { + // TODO this seems almost identical to plain_matrix_type + template struct sparse_eval + { typedef typename traits::Scalar _Scalar; typedef typename traits::StorageIndex _StorageIndex; - enum { _Options = ((Flags&RowMajorBit)==RowMajorBit) ? RowMajor : ColMajor }; + enum { _Options = ((Flags & RowMajorBit) == RowMajorBit) ? RowMajor : ColMajor }; + public: typedef SparseMatrix<_Scalar, _Options, _StorageIndex> type; -}; + }; -template struct sparse_eval { + template struct sparse_eval + { typedef typename traits::Scalar _Scalar; + public: typedef Matrix<_Scalar, 1, 1> type; -}; + }; + + template struct plain_matrix_type + { + typedef typename traits::Scalar _Scalar; + typedef typename traits::StorageIndex _StorageIndex; + enum { _Options = ((evaluator::Flags & RowMajorBit) == RowMajorBit) ? RowMajor : ColMajor }; -template struct plain_matrix_type -{ - typedef typename traits::Scalar _Scalar; - typedef typename traits::StorageIndex _StorageIndex; - enum { _Options = ((evaluator::Flags&RowMajorBit)==RowMajorBit) ? RowMajor : ColMajor }; public: typedef SparseMatrix<_Scalar, _Options, _StorageIndex> type; -}; - -template -struct plain_object_eval - : sparse_eval::RowsAtCompileTime,traits::ColsAtCompileTime, evaluator::Flags> -{}; - -template -struct solve_traits -{ - typedef typename sparse_eval::Flags>::type PlainObject; -}; - -template -struct generic_xpr_base -{ - typedef SparseMatrixBase type; -}; - -struct SparseTriangularShape { static std::string debugName() { return "SparseTriangularShape"; } }; -struct SparseSelfAdjointShape { static std::string debugName() { return "SparseSelfAdjointShape"; } }; - -template<> struct glue_shapes { typedef SparseSelfAdjointShape type; }; -template<> struct glue_shapes { typedef SparseTriangularShape type; }; - -} // end namespace internal + }; + + template + struct plain_object_eval + : sparse_eval::RowsAtCompileTime, traits::ColsAtCompileTime, evaluator::Flags> + { + }; + + template struct solve_traits + { + typedef + typename sparse_eval::Flags>:: + type PlainObject; + }; + + template struct generic_xpr_base + { + typedef SparseMatrixBase type; + }; + + struct SparseTriangularShape + { + static std::string debugName() { return "SparseTriangularShape"; } + }; + struct SparseSelfAdjointShape + { + static std::string debugName() { return "SparseSelfAdjointShape"; } + }; + + template<> struct glue_shapes + { + typedef SparseSelfAdjointShape type; + }; + template<> struct glue_shapes + { + typedef SparseTriangularShape type; + }; + +}// end namespace internal /** \ingroup SparseCore_Module - * - * \class Triplet - * - * \brief A small structure to hold a non zero as a triplet (i,j,value). - * - * \sa SparseMatrix::setFromTriplets() - */ -template::StorageIndex > -class Triplet + * + * \class Triplet + * + * \brief A small structure to hold a non zero as a triplet (i,j,value). + * + * \sa SparseMatrix::setFromTriplets() + */ +template::StorageIndex> class Triplet { public: Triplet() : m_row(0), m_col(0), m_value(0) {} - Triplet(const StorageIndex& i, const StorageIndex& j, const Scalar& v = Scalar(0)) - : m_row(i), m_col(j), m_value(v) - {} + Triplet(const StorageIndex &i, const StorageIndex &j, const Scalar &v = Scalar(0)) : m_row(i), m_col(j), m_value(v) {} /** \returns the row index of the element */ - const StorageIndex& row() const { return m_row; } + const StorageIndex &row() const { return m_row; } /** \returns the column index of the element */ - const StorageIndex& col() const { return m_col; } + const StorageIndex &col() const { return m_col; } /** \returns the value of the element */ - const Scalar& value() const { return m_value; } + const Scalar &value() const { return m_value; } + protected: StorageIndex m_row, m_col; Scalar m_value; }; -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_SPARSEUTIL_H +#endif// EIGEN_SPARSEUTIL_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/SparseCore/SparseVector.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/SparseCore/SparseVector.h index 19b0fbc9..5df4c5b0 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/SparseCore/SparseVector.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/SparseCore/SparseVector.h @@ -10,469 +10,451 @@ #ifndef EIGEN_SPARSEVECTOR_H #define EIGEN_SPARSEVECTOR_H -namespace Eigen { +namespace Eigen { /** \ingroup SparseCore_Module - * \class SparseVector - * - * \brief a sparse vector class - * - * \tparam _Scalar the scalar type, i.e. the type of the coefficients - * - * See http://www.netlib.org/linalg/html_templates/node91.html for details on the storage scheme. - * - * This class can be extended with the help of the plugin mechanism described on the page - * \ref TopicCustomizing_Plugins by defining the preprocessor symbol \c EIGEN_SPARSEVECTOR_PLUGIN. - */ + * \class SparseVector + * + * \brief a sparse vector class + * + * \tparam _Scalar the scalar type, i.e. the type of the coefficients + * + * See http://www.netlib.org/linalg/html_templates/node91.html for details on the storage scheme. + * + * This class can be extended with the help of the plugin mechanism described on the page + * \ref TopicCustomizing_Plugins by defining the preprocessor symbol \c EIGEN_SPARSEVECTOR_PLUGIN. + */ namespace internal { -template -struct traits > -{ - typedef _Scalar Scalar; - typedef _StorageIndex StorageIndex; - typedef Sparse StorageKind; - typedef MatrixXpr XprKind; - enum { - IsColVector = (_Options & RowMajorBit) ? 0 : 1, - - RowsAtCompileTime = IsColVector ? Dynamic : 1, - ColsAtCompileTime = IsColVector ? 1 : Dynamic, - MaxRowsAtCompileTime = RowsAtCompileTime, - MaxColsAtCompileTime = ColsAtCompileTime, - Flags = _Options | NestByRefBit | LvalueBit | (IsColVector ? 0 : RowMajorBit) | CompressedAccessBit, - SupportedAccessPatterns = InnerRandomAccessPattern + template + struct traits> + { + typedef _Scalar Scalar; + typedef _StorageIndex StorageIndex; + typedef Sparse StorageKind; + typedef MatrixXpr XprKind; + enum { + IsColVector = (_Options & RowMajorBit) ? 0 : 1, + + RowsAtCompileTime = IsColVector ? Dynamic : 1, + ColsAtCompileTime = IsColVector ? 1 : Dynamic, + MaxRowsAtCompileTime = RowsAtCompileTime, + MaxColsAtCompileTime = ColsAtCompileTime, + Flags = _Options | NestByRefBit | LvalueBit | (IsColVector ? 0 : RowMajorBit) | CompressedAccessBit, + SupportedAccessPatterns = InnerRandomAccessPattern + }; }; -}; -// Sparse-Vector-Assignment kinds: -enum { - SVA_RuntimeSwitch, - SVA_Inner, - SVA_Outer -}; + // Sparse-Vector-Assignment kinds: + enum { SVA_RuntimeSwitch, SVA_Inner, SVA_Outer }; -template< typename Dest, typename Src, - int AssignmentKind = !bool(Src::IsVectorAtCompileTime) ? SVA_RuntimeSwitch - : Src::InnerSizeAtCompileTime==1 ? SVA_Outer - : SVA_Inner> -struct sparse_vector_assign_selector; + template + struct sparse_vector_assign_selector; -} +}// namespace internal template -class SparseVector - : public SparseCompressedBase > +class SparseVector : public SparseCompressedBase> { - typedef SparseCompressedBase Base; - using Base::convert_index; - public: - EIGEN_SPARSE_PUBLIC_INTERFACE(SparseVector) - EIGEN_SPARSE_INHERIT_ASSIGNMENT_OPERATOR(SparseVector, +=) - EIGEN_SPARSE_INHERIT_ASSIGNMENT_OPERATOR(SparseVector, -=) - - typedef internal::CompressedStorage Storage; - enum { IsColVector = internal::traits::IsColVector }; - - enum { - Options = _Options - }; - - EIGEN_STRONG_INLINE Index rows() const { return IsColVector ? m_size : 1; } - EIGEN_STRONG_INLINE Index cols() const { return IsColVector ? 1 : m_size; } - EIGEN_STRONG_INLINE Index innerSize() const { return m_size; } - EIGEN_STRONG_INLINE Index outerSize() const { return 1; } - - EIGEN_STRONG_INLINE const Scalar* valuePtr() const { return m_data.valuePtr(); } - EIGEN_STRONG_INLINE Scalar* valuePtr() { return m_data.valuePtr(); } - - EIGEN_STRONG_INLINE const StorageIndex* innerIndexPtr() const { return m_data.indexPtr(); } - EIGEN_STRONG_INLINE StorageIndex* innerIndexPtr() { return m_data.indexPtr(); } - - inline const StorageIndex* outerIndexPtr() const { return 0; } - inline StorageIndex* outerIndexPtr() { return 0; } - inline const StorageIndex* innerNonZeroPtr() const { return 0; } - inline StorageIndex* innerNonZeroPtr() { return 0; } - - /** \internal */ - inline Storage& data() { return m_data; } - /** \internal */ - inline const Storage& data() const { return m_data; } - - inline Scalar coeff(Index row, Index col) const - { - eigen_assert(IsColVector ? (col==0 && row>=0 && row=0 && col=0 && i Base; + using Base::convert_index; - inline Scalar& coeffRef(Index row, Index col) - { - eigen_assert(IsColVector ? (col==0 && row>=0 && row=0 && col=0 && i Storage; + enum { IsColVector = internal::traits::IsColVector }; - return m_data.atWithInsertion(StorageIndex(i)); - } + enum { Options = _Options }; - public: + EIGEN_STRONG_INLINE Index rows() const { return IsColVector ? m_size : 1; } + EIGEN_STRONG_INLINE Index cols() const { return IsColVector ? 1 : m_size; } + EIGEN_STRONG_INLINE Index innerSize() const { return m_size; } + EIGEN_STRONG_INLINE Index outerSize() const { return 1; } - typedef typename Base::InnerIterator InnerIterator; - typedef typename Base::ReverseInnerIterator ReverseInnerIterator; + EIGEN_STRONG_INLINE const Scalar *valuePtr() const { return m_data.valuePtr(); } + EIGEN_STRONG_INLINE Scalar *valuePtr() { return m_data.valuePtr(); } - inline void setZero() { m_data.clear(); } + EIGEN_STRONG_INLINE const StorageIndex *innerIndexPtr() const { return m_data.indexPtr(); } + EIGEN_STRONG_INLINE StorageIndex *innerIndexPtr() { return m_data.indexPtr(); } - /** \returns the number of non zero coefficients */ - inline Index nonZeros() const { return m_data.size(); } + inline const StorageIndex *outerIndexPtr() const { return 0; } + inline StorageIndex *outerIndexPtr() { return 0; } + inline const StorageIndex *innerNonZeroPtr() const { return 0; } + inline StorageIndex *innerNonZeroPtr() { return 0; } - inline void startVec(Index outer) - { - EIGEN_UNUSED_VARIABLE(outer); - eigen_assert(outer==0); - } + /** \internal */ + inline Storage &data() { return m_data; } + /** \internal */ + inline const Storage &data() const { return m_data; } - inline Scalar& insertBackByOuterInner(Index outer, Index inner) - { - EIGEN_UNUSED_VARIABLE(outer); - eigen_assert(outer==0); - return insertBack(inner); - } - inline Scalar& insertBack(Index i) - { - m_data.append(0, i); - return m_data.value(m_data.size()-1); - } - - Scalar& insertBackByOuterInnerUnordered(Index outer, Index inner) - { - EIGEN_UNUSED_VARIABLE(outer); - eigen_assert(outer==0); - return insertBackUnordered(inner); - } - inline Scalar& insertBackUnordered(Index i) - { - m_data.append(0, i); - return m_data.value(m_data.size()-1); - } + inline Scalar coeff(Index row, Index col) const + { + eigen_assert(IsColVector ? (col == 0 && row >= 0 && row < m_size) : (row == 0 && col >= 0 && col < m_size)); + return coeff(IsColVector ? row : col); + } + inline Scalar coeff(Index i) const + { + eigen_assert(i >= 0 && i < m_size); + return m_data.at(StorageIndex(i)); + } - inline Scalar& insert(Index row, Index col) - { - eigen_assert(IsColVector ? (col==0 && row>=0 && row=0 && col=0 && i= startId) && (m_data.index(p) > i) ) - { - m_data.index(p+1) = m_data.index(p); - m_data.value(p+1) = m_data.value(p); - --p; - } - m_data.index(p+1) = convert_index(i); - m_data.value(p+1) = 0; - return m_data.value(p+1); - } + inline Scalar &coeffRef(Index row, Index col) + { + eigen_assert(IsColVector ? (col == 0 && row >= 0 && row < m_size) : (row == 0 && col >= 0 && col < m_size)); + return coeffRef(IsColVector ? row : col); + } - /** - */ - inline void reserve(Index reserveSize) { m_data.reserve(reserveSize); } + /** \returns a reference to the coefficient value at given index \a i + * This operation involes a log(rho*size) binary search. If the coefficient does not + * exist yet, then a sorted insertion into a sequential buffer is performed. + * + * This insertion might be very costly if the number of nonzeros above \a i is large. + */ + inline Scalar &coeffRef(Index i) + { + eigen_assert(i >= 0 && i < m_size); + return m_data.atWithInsertion(StorageIndex(i)); + } - inline void finalize() {} +public: + typedef typename Base::InnerIterator InnerIterator; + typedef typename Base::ReverseInnerIterator ReverseInnerIterator; - /** \copydoc SparseMatrix::prune(const Scalar&,const RealScalar&) */ - void prune(const Scalar& reference, const RealScalar& epsilon = NumTraits::dummy_precision()) - { - m_data.prune(reference,epsilon); - } + inline void setZero() { m_data.clear(); } - /** Resizes the sparse vector to \a rows x \a cols - * - * This method is provided for compatibility with matrices. - * For a column vector, \a cols must be equal to 1. - * For a row vector, \a rows must be equal to 1. - * - * \sa resize(Index) - */ - void resize(Index rows, Index cols) - { - eigen_assert((IsColVector ? cols : rows)==1 && "Outer dimension must equal 1"); - resize(IsColVector ? rows : cols); - } + /** \returns the number of non zero coefficients */ + inline Index nonZeros() const { return m_data.size(); } - /** Resizes the sparse vector to \a newSize - * This method deletes all entries, thus leaving an empty sparse vector - * - * \sa conservativeResize(), setZero() */ - void resize(Index newSize) - { - m_size = newSize; - m_data.clear(); - } + inline void startVec(Index outer) + { + EIGEN_UNUSED_VARIABLE(outer); + eigen_assert(outer == 0); + } - /** Resizes the sparse vector to \a newSize, while leaving old values untouched. - * - * If the size of the vector is decreased, then the storage of the out-of bounds coefficients is kept and reserved. - * Call .data().squeeze() to free extra memory. - * - * \sa reserve(), setZero() - */ - void conservativeResize(Index newSize) - { - if (newSize < m_size) - { - Index i = 0; - while (i= 0 && row < m_size) : (row == 0 && col >= 0 && col < m_size)); - explicit inline SparseVector(Index size) : m_size(0) { check_template_parameters(); resize(size); } + Index inner = IsColVector ? row : col; + Index outer = IsColVector ? col : row; + EIGEN_ONLY_USED_FOR_DEBUG(outer); + eigen_assert(outer == 0); + return insert(inner); + } + Scalar &insert(Index i) + { + eigen_assert(i >= 0 && i < m_size); - inline SparseVector(Index rows, Index cols) : m_size(0) { check_template_parameters(); resize(rows,cols); } + Index startId = 0; + Index p = Index(m_data.size()) - 1; + // TODO smart realloc + m_data.resize(p + 2, 1); - template - inline SparseVector(const SparseMatrixBase& other) - : m_size(0) - { - #ifdef EIGEN_SPARSE_CREATE_TEMPORARY_PLUGIN - EIGEN_SPARSE_CREATE_TEMPORARY_PLUGIN - #endif - check_template_parameters(); - *this = other.derived(); + while ((p >= startId) && (m_data.index(p) > i)) { + m_data.index(p + 1) = m_data.index(p); + m_data.value(p + 1) = m_data.value(p); + --p; } + m_data.index(p + 1) = convert_index(i); + m_data.value(p + 1) = 0; + return m_data.value(p + 1); + } - inline SparseVector(const SparseVector& other) - : Base(other), m_size(0) - { - check_template_parameters(); - *this = other.derived(); - } + /** + */ + inline void reserve(Index reserveSize) { m_data.reserve(reserveSize); } - /** Swaps the values of \c *this and \a other. - * Overloaded for performance: this version performs a \em shallow swap by swaping pointers and attributes only. - * \sa SparseMatrixBase::swap() - */ - inline void swap(SparseVector& other) - { - std::swap(m_size, other.m_size); - m_data.swap(other.m_data); - } - template - inline void swap(SparseMatrix& other) - { - eigen_assert(other.outerSize()==1); - std::swap(m_size, other.m_innerSize); - m_data.swap(other.m_data); - } + inline void finalize() {} - inline SparseVector& operator=(const SparseVector& other) - { - if (other.isRValue()) - { - swap(other.const_cast_derived()); - } - else - { - resize(other.size()); - m_data = other.m_data; - } - return *this; - } + /** \copydoc SparseMatrix::prune(const Scalar&,const RealScalar&) */ + void prune(const Scalar &reference, const RealScalar &epsilon = NumTraits::dummy_precision()) + { + m_data.prune(reference, epsilon); + } - template - inline SparseVector& operator=(const SparseMatrixBase& other) - { - SparseVector tmp(other.size()); - internal::sparse_vector_assign_selector::run(tmp,other.derived()); - this->swap(tmp); - return *this; - } + /** Resizes the sparse vector to \a rows x \a cols + * + * This method is provided for compatibility with matrices. + * For a column vector, \a cols must be equal to 1. + * For a row vector, \a rows must be equal to 1. + * + * \sa resize(Index) + */ + void resize(Index rows, Index cols) + { + eigen_assert((IsColVector ? cols : rows) == 1 && "Outer dimension must equal 1"); + resize(IsColVector ? rows : cols); + } - #ifndef EIGEN_PARSED_BY_DOXYGEN - template - inline SparseVector& operator=(const SparseSparseProduct& product) - { - return Base::operator=(product); - } - #endif + /** Resizes the sparse vector to \a newSize + * This method deletes all entries, thus leaving an empty sparse vector + * + * \sa conservativeResize(), setZero() */ + void resize(Index newSize) + { + m_size = newSize; + m_data.clear(); + } - friend std::ostream & operator << (std::ostream & s, const SparseVector& m) - { - for (Index i=0; i inline SparseVector(const SparseMatrixBase &other) : m_size(0) + { +#ifdef EIGEN_SPARSE_CREATE_TEMPORARY_PLUGIN + EIGEN_SPARSE_CREATE_TEMPORARY_PLUGIN +#endif + check_template_parameters(); + *this = other.derived(); + } - /** \internal \deprecated use insertBack(Index) */ - EIGEN_DEPRECATED Scalar& fill(Index i) - { - m_data.append(0, i); - return m_data.value(m_data.size()-1); - } + inline SparseVector(const SparseVector &other) : Base(other), m_size(0) + { + check_template_parameters(); + *this = other.derived(); + } - /** \internal \deprecated use insert(Index,Index) */ - EIGEN_DEPRECATED Scalar& fillrand(Index r, Index c) - { - eigen_assert(r==0 || c==0); - return fillrand(IsColVector ? r : c); - } + /** Swaps the values of \c *this and \a other. + * Overloaded for performance: this version performs a \em shallow swap by swaping pointers and attributes only. + * \sa SparseMatrixBase::swap() + */ + inline void swap(SparseVector &other) + { + std::swap(m_size, other.m_size); + m_data.swap(other.m_data); + } - /** \internal \deprecated use insert(Index) */ - EIGEN_DEPRECATED Scalar& fillrand(Index i) - { - return insert(i); + template inline void swap(SparseMatrix &other) + { + eigen_assert(other.outerSize() == 1); + std::swap(m_size, other.m_innerSize); + m_data.swap(other.m_data); + } + + inline SparseVector &operator=(const SparseVector &other) + { + if (other.isRValue()) { + swap(other.const_cast_derived()); + } else { + resize(other.size()); + m_data = other.m_data; } + return *this; + } - /** \internal \deprecated use finalize() */ - EIGEN_DEPRECATED void endFill() {} - - // These two functions were here in the 3.1 release, so let's keep them in case some code rely on them. - /** \internal \deprecated use data() */ - EIGEN_DEPRECATED Storage& _data() { return m_data; } - /** \internal \deprecated use data() */ - EIGEN_DEPRECATED const Storage& _data() const { return m_data; } - -# ifdef EIGEN_SPARSEVECTOR_PLUGIN -# include EIGEN_SPARSEVECTOR_PLUGIN -# endif + template inline SparseVector &operator=(const SparseMatrixBase &other) + { + SparseVector tmp(other.size()); + internal::sparse_vector_assign_selector::run(tmp, other.derived()); + this->swap(tmp); + return *this; + } -protected: - - static void check_template_parameters() - { - EIGEN_STATIC_ASSERT(NumTraits::IsSigned,THE_INDEX_TYPE_MUST_BE_A_SIGNED_TYPE); - EIGEN_STATIC_ASSERT((_Options&(ColMajor|RowMajor))==Options,INVALID_MATRIX_TEMPLATE_PARAMETERS); - } - - Storage m_data; - Index m_size; -}; +#ifndef EIGEN_PARSED_BY_DOXYGEN + template inline SparseVector &operator=(const SparseSparseProduct &product) + { + return Base::operator=(product); + } +#endif -namespace internal { + friend std::ostream &operator<<(std::ostream &s, const SparseVector &m) + { + for (Index i = 0; i < m.nonZeros(); ++i) s << "(" << m.m_data.value(i) << "," << m.m_data.index(i) << ") "; + s << std::endl; + return s; + } -template -struct evaluator > - : evaluator_base > -{ - typedef SparseVector<_Scalar,_Options,_Index> SparseVectorType; - typedef evaluator_base Base; - typedef typename SparseVectorType::InnerIterator InnerIterator; - typedef typename SparseVectorType::ReverseInnerIterator ReverseInnerIterator; - - enum { - CoeffReadCost = NumTraits<_Scalar>::ReadCost, - Flags = SparseVectorType::Flags - }; + /** Destructor */ + inline ~SparseVector() {} + + /** Overloaded for performance */ + Scalar sum() const; + +public: + /** \internal \deprecated use setZero() and reserve() */ + EIGEN_DEPRECATED void startFill(Index reserve) + { + setZero(); + m_data.reserve(reserve); + } - evaluator() : Base() {} - - explicit evaluator(const SparseVectorType &mat) : m_matrix(&mat) + /** \internal \deprecated use insertBack(Index,Index) */ + EIGEN_DEPRECATED Scalar &fill(Index r, Index c) { - EIGEN_INTERNAL_CHECK_COST_VALUE(CoeffReadCost); + eigen_assert(r == 0 || c == 0); + return fill(IsColVector ? r : c); } - - inline Index nonZerosEstimate() const { - return m_matrix->nonZeros(); + + /** \internal \deprecated use insertBack(Index) */ + EIGEN_DEPRECATED Scalar &fill(Index i) + { + m_data.append(0, i); + return m_data.value(m_data.size() - 1); } - - operator SparseVectorType&() { return m_matrix->const_cast_derived(); } - operator const SparseVectorType&() const { return *m_matrix; } - - const SparseVectorType *m_matrix; -}; -template< typename Dest, typename Src> -struct sparse_vector_assign_selector { - static void run(Dest& dst, const Src& src) { - eigen_internal_assert(src.innerSize()==src.size()); - typedef internal::evaluator SrcEvaluatorType; - SrcEvaluatorType srcEval(src); - for(typename SrcEvaluatorType::InnerIterator it(srcEval, 0); it; ++it) - dst.insert(it.index()) = it.value(); + /** \internal \deprecated use insert(Index,Index) */ + EIGEN_DEPRECATED Scalar &fillrand(Index r, Index c) + { + eigen_assert(r == 0 || c == 0); + return fillrand(IsColVector ? r : c); + } + + /** \internal \deprecated use insert(Index) */ + EIGEN_DEPRECATED Scalar &fillrand(Index i) { return insert(i); } + + /** \internal \deprecated use finalize() */ + EIGEN_DEPRECATED void endFill() {} + + // These two functions were here in the 3.1 release, so let's keep them in case some code rely on them. + /** \internal \deprecated use data() */ + EIGEN_DEPRECATED Storage &_data() { return m_data; } + /** \internal \deprecated use data() */ + EIGEN_DEPRECATED const Storage &_data() const { return m_data; } + +#ifdef EIGEN_SPARSEVECTOR_PLUGIN +#include EIGEN_SPARSEVECTOR_PLUGIN +#endif + +protected: + static void check_template_parameters() + { + EIGEN_STATIC_ASSERT(NumTraits::IsSigned, THE_INDEX_TYPE_MUST_BE_A_SIGNED_TYPE); + EIGEN_STATIC_ASSERT((_Options & (ColMajor | RowMajor)) == Options, INVALID_MATRIX_TEMPLATE_PARAMETERS); } + + Storage m_data; + Index m_size; }; -template< typename Dest, typename Src> -struct sparse_vector_assign_selector { - static void run(Dest& dst, const Src& src) { - eigen_internal_assert(src.outerSize()==src.size()); - typedef internal::evaluator SrcEvaluatorType; - SrcEvaluatorType srcEval(src); - for(Index i=0; i + struct evaluator> : evaluator_base> + { + typedef SparseVector<_Scalar, _Options, _Index> SparseVectorType; + typedef evaluator_base Base; + typedef typename SparseVectorType::InnerIterator InnerIterator; + typedef typename SparseVectorType::ReverseInnerIterator ReverseInnerIterator; + + enum { CoeffReadCost = NumTraits<_Scalar>::ReadCost, Flags = SparseVectorType::Flags }; + + evaluator() : Base() {} + + explicit evaluator(const SparseVectorType &mat) : m_matrix(&mat) { EIGEN_INTERNAL_CHECK_COST_VALUE(CoeffReadCost); } + + inline Index nonZerosEstimate() const { return m_matrix->nonZeros(); } + + operator SparseVectorType &() { return m_matrix->const_cast_derived(); } + operator const SparseVectorType &() const { return *m_matrix; } + + const SparseVectorType *m_matrix; + }; + + template struct sparse_vector_assign_selector + { + static void run(Dest &dst, const Src &src) { - typename SrcEvaluatorType::InnerIterator it(srcEval, i); - if(it) - dst.insert(i) = it.value(); + eigen_internal_assert(src.innerSize() == src.size()); + typedef internal::evaluator SrcEvaluatorType; + SrcEvaluatorType srcEval(src); + for (typename SrcEvaluatorType::InnerIterator it(srcEval, 0); it; ++it) dst.insert(it.index()) = it.value(); } - } -}; + }; -template< typename Dest, typename Src> -struct sparse_vector_assign_selector { - static void run(Dest& dst, const Src& src) { - if(src.outerSize()==1) sparse_vector_assign_selector::run(dst, src); - else sparse_vector_assign_selector::run(dst, src); - } -}; + template struct sparse_vector_assign_selector + { + static void run(Dest &dst, const Src &src) + { + eigen_internal_assert(src.outerSize() == src.size()); + typedef internal::evaluator SrcEvaluatorType; + SrcEvaluatorType srcEval(src); + for (Index i = 0; i < src.size(); ++i) { + typename SrcEvaluatorType::InnerIterator it(srcEval, i); + if (it) dst.insert(i) = it.value(); + } + } + }; + + template struct sparse_vector_assign_selector + { + static void run(Dest &dst, const Src &src) + { + if (src.outerSize() == 1) + sparse_vector_assign_selector::run(dst, src); + else + sparse_vector_assign_selector::run(dst, src); + } + }; -} +}// namespace internal -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_SPARSEVECTOR_H +#endif// EIGEN_SPARSEVECTOR_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/SparseCore/SparseView.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/SparseCore/SparseView.h index 7c4aea74..5afd47c3 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/SparseCore/SparseView.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/SparseCore/SparseView.h @@ -11,63 +11,61 @@ #ifndef EIGEN_SPARSEVIEW_H #define EIGEN_SPARSEVIEW_H -namespace Eigen { +namespace Eigen { namespace internal { -template -struct traits > : traits -{ - typedef typename MatrixType::StorageIndex StorageIndex; - typedef Sparse StorageKind; - enum { - Flags = int(traits::Flags) & (RowMajorBit) + template struct traits> : traits + { + typedef typename MatrixType::StorageIndex StorageIndex; + typedef Sparse StorageKind; + enum { Flags = int(traits::Flags) & (RowMajorBit) }; }; -}; -} // end namespace internal +}// end namespace internal /** \ingroup SparseCore_Module - * \class SparseView - * - * \brief Expression of a dense or sparse matrix with zero or too small values removed - * - * \tparam MatrixType the type of the object of which we are removing the small entries - * - * This class represents an expression of a given dense or sparse matrix with - * entries smaller than \c reference * \c epsilon are removed. - * It is the return type of MatrixBase::sparseView() and SparseMatrixBase::pruned() - * and most of the time this is the only way it is used. - * - * \sa MatrixBase::sparseView(), SparseMatrixBase::pruned() - */ -template -class SparseView : public SparseMatrixBase > + * \class SparseView + * + * \brief Expression of a dense or sparse matrix with zero or too small values removed + * + * \tparam MatrixType the type of the object of which we are removing the small entries + * + * This class represents an expression of a given dense or sparse matrix with + * entries smaller than \c reference * \c epsilon are removed. + * It is the return type of MatrixBase::sparseView() and SparseMatrixBase::pruned() + * and most of the time this is the only way it is used. + * + * \sa MatrixBase::sparseView(), SparseMatrixBase::pruned() + */ +template class SparseView : public SparseMatrixBase> { typedef typename MatrixType::Nested MatrixTypeNested; typedef typename internal::remove_all::type _MatrixTypeNested; - typedef SparseMatrixBase Base; + typedef SparseMatrixBase Base; + public: EIGEN_SPARSE_PUBLIC_INTERFACE(SparseView) typedef typename internal::remove_all::type NestedExpression; - explicit SparseView(const MatrixType& mat, const Scalar& reference = Scalar(0), - const RealScalar &epsilon = NumTraits::dummy_precision()) - : m_matrix(mat), m_reference(reference), m_epsilon(epsilon) {} + explicit SparseView(const MatrixType &mat, + const Scalar &reference = Scalar(0), + const RealScalar &epsilon = NumTraits::dummy_precision()) + : m_matrix(mat), m_reference(reference), m_epsilon(epsilon) + {} inline Index rows() const { return m_matrix.rows(); } inline Index cols() const { return m_matrix.cols(); } inline Index innerSize() const { return m_matrix.innerSize(); } inline Index outerSize() const { return m_matrix.outerSize(); } - + /** \returns the nested expression */ - const typename internal::remove_all::type& - nestedExpression() const { return m_matrix; } - + const typename internal::remove_all::type &nestedExpression() const { return m_matrix; } + Scalar reference() const { return m_reference; } RealScalar epsilon() const { return m_epsilon; } - + protected: MatrixTypeNested m_matrix; Scalar m_reference; @@ -76,178 +74,168 @@ class SparseView : public SparseMatrixBase > namespace internal { -// TODO find a way to unify the two following variants -// This is tricky because implementing an inner iterator on top of an IndexBased evaluator is -// not easy because the evaluators do not expose the sizes of the underlying expression. - -template -struct unary_evaluator, IteratorBased> - : public evaluator_base > -{ + // TODO find a way to unify the two following variants + // This is tricky because implementing an inner iterator on top of an IndexBased evaluator is + // not easy because the evaluators do not expose the sizes of the underlying expression. + + template + struct unary_evaluator, IteratorBased> : public evaluator_base> + { typedef typename evaluator::InnerIterator EvalIterator; + public: typedef SparseView XprType; - + class InnerIterator : public EvalIterator { - typedef typename XprType::Scalar Scalar; - public: - - EIGEN_STRONG_INLINE InnerIterator(const unary_evaluator& sve, Index outer) - : EvalIterator(sve.m_argImpl,outer), m_view(sve.m_view) - { - incrementToNonZero(); - } - - EIGEN_STRONG_INLINE InnerIterator& operator++() - { + typedef typename XprType::Scalar Scalar; + + public: + EIGEN_STRONG_INLINE InnerIterator(const unary_evaluator &sve, Index outer) + : EvalIterator(sve.m_argImpl, outer), m_view(sve.m_view) + { + incrementToNonZero(); + } + + EIGEN_STRONG_INLINE InnerIterator &operator++() + { + EvalIterator::operator++(); + incrementToNonZero(); + return *this; + } + + using EvalIterator::value; + + protected: + const XprType &m_view; + + private: + void incrementToNonZero() + { + while ((bool(*this)) && internal::isMuchSmallerThan(value(), m_view.reference(), m_view.epsilon())) { EvalIterator::operator++(); - incrementToNonZero(); - return *this; } + } + }; - using EvalIterator::value; + enum { CoeffReadCost = evaluator::CoeffReadCost, Flags = XprType::Flags }; - protected: - const XprType &m_view; - - private: - void incrementToNonZero() - { - while((bool(*this)) && internal::isMuchSmallerThan(value(), m_view.reference(), m_view.epsilon())) - { - EvalIterator::operator++(); - } - } - }; - - enum { - CoeffReadCost = evaluator::CoeffReadCost, - Flags = XprType::Flags - }; - - explicit unary_evaluator(const XprType& xpr) : m_argImpl(xpr.nestedExpression()), m_view(xpr) {} + explicit unary_evaluator(const XprType &xpr) : m_argImpl(xpr.nestedExpression()), m_view(xpr) {} protected: evaluator m_argImpl; const XprType &m_view; -}; + }; -template -struct unary_evaluator, IndexBased> - : public evaluator_base > -{ + template + struct unary_evaluator, IndexBased> : public evaluator_base> + { public: typedef SparseView XprType; + protected: - enum { IsRowMajor = (XprType::Flags&RowMajorBit)==RowMajorBit }; + enum { IsRowMajor = (XprType::Flags & RowMajorBit) == RowMajorBit }; typedef typename XprType::Scalar Scalar; typedef typename XprType::StorageIndex StorageIndex; + public: - class InnerIterator { - public: - - EIGEN_STRONG_INLINE InnerIterator(const unary_evaluator& sve, Index outer) - : m_sve(sve), m_inner(0), m_outer(outer), m_end(sve.m_view.innerSize()) - { - incrementToNonZero(); - } - - EIGEN_STRONG_INLINE InnerIterator& operator++() - { + public: + EIGEN_STRONG_INLINE InnerIterator(const unary_evaluator &sve, Index outer) + : m_sve(sve), m_inner(0), m_outer(outer), m_end(sve.m_view.innerSize()) + { + incrementToNonZero(); + } + + EIGEN_STRONG_INLINE InnerIterator &operator++() + { + m_inner++; + incrementToNonZero(); + return *this; + } + + EIGEN_STRONG_INLINE Scalar value() const + { + return (IsRowMajor) ? m_sve.m_argImpl.coeff(m_outer, m_inner) : m_sve.m_argImpl.coeff(m_inner, m_outer); + } + + EIGEN_STRONG_INLINE StorageIndex index() const { return m_inner; } + inline Index row() const { return IsRowMajor ? m_outer : index(); } + inline Index col() const { return IsRowMajor ? index() : m_outer; } + + EIGEN_STRONG_INLINE operator bool() const { return m_inner < m_end && m_inner >= 0; } + + protected: + const unary_evaluator &m_sve; + Index m_inner; + const Index m_outer; + const Index m_end; + + private: + void incrementToNonZero() + { + while ( + (bool(*this)) && internal::isMuchSmallerThan(value(), m_sve.m_view.reference(), m_sve.m_view.epsilon())) { m_inner++; - incrementToNonZero(); - return *this; } + } + }; - EIGEN_STRONG_INLINE Scalar value() const - { - return (IsRowMajor) ? m_sve.m_argImpl.coeff(m_outer, m_inner) - : m_sve.m_argImpl.coeff(m_inner, m_outer); - } + enum { CoeffReadCost = evaluator::CoeffReadCost, Flags = XprType::Flags }; - EIGEN_STRONG_INLINE StorageIndex index() const { return m_inner; } - inline Index row() const { return IsRowMajor ? m_outer : index(); } - inline Index col() const { return IsRowMajor ? index() : m_outer; } - - EIGEN_STRONG_INLINE operator bool() const { return m_inner < m_end && m_inner>=0; } - - protected: - const unary_evaluator &m_sve; - Index m_inner; - const Index m_outer; - const Index m_end; - - private: - void incrementToNonZero() - { - while((bool(*this)) && internal::isMuchSmallerThan(value(), m_sve.m_view.reference(), m_sve.m_view.epsilon())) - { - m_inner++; - } - } - }; - - enum { - CoeffReadCost = evaluator::CoeffReadCost, - Flags = XprType::Flags - }; - - explicit unary_evaluator(const XprType& xpr) : m_argImpl(xpr.nestedExpression()), m_view(xpr) {} + explicit unary_evaluator(const XprType &xpr) : m_argImpl(xpr.nestedExpression()), m_view(xpr) {} protected: evaluator m_argImpl; const XprType &m_view; -}; + }; -} // end namespace internal +}// end namespace internal /** \ingroup SparseCore_Module - * - * \returns a sparse expression of the dense expression \c *this with values smaller than - * \a reference * \a epsilon removed. - * - * This method is typically used when prototyping to convert a quickly assembled dense Matrix \c D to a SparseMatrix \c S: - * \code - * MatrixXd D(n,m); - * SparseMatrix S; - * S = D.sparseView(); // suppress numerical zeros (exact) - * S = D.sparseView(reference); - * S = D.sparseView(reference,epsilon); - * \endcode - * where \a reference is a meaningful non zero reference value, - * and \a epsilon is a tolerance factor defaulting to NumTraits::dummy_precision(). - * - * \sa SparseMatrixBase::pruned(), class SparseView */ + * + * \returns a sparse expression of the dense expression \c *this with values smaller than + * \a reference * \a epsilon removed. + * + * This method is typically used when prototyping to convert a quickly assembled dense Matrix \c D to a SparseMatrix \c + * S: + * \code + * MatrixXd D(n,m); + * SparseMatrix S; + * S = D.sparseView(); // suppress numerical zeros (exact) + * S = D.sparseView(reference); + * S = D.sparseView(reference,epsilon); + * \endcode + * where \a reference is a meaningful non zero reference value, + * and \a epsilon is a tolerance factor defaulting to NumTraits::dummy_precision(). + * + * \sa SparseMatrixBase::pruned(), class SparseView */ template -const SparseView MatrixBase::sparseView(const Scalar& reference, - const typename NumTraits::Real& epsilon) const +const SparseView MatrixBase::sparseView(const Scalar &reference, + const typename NumTraits::Real &epsilon) const { return SparseView(derived(), reference, epsilon); } /** \returns an expression of \c *this with values smaller than - * \a reference * \a epsilon removed. - * - * This method is typically used in conjunction with the product of two sparse matrices - * to automatically prune the smallest values as follows: - * \code - * C = (A*B).pruned(); // suppress numerical zeros (exact) - * C = (A*B).pruned(ref); - * C = (A*B).pruned(ref,epsilon); - * \endcode - * where \c ref is a meaningful non zero reference value. - * */ + * \a reference * \a epsilon removed. + * + * This method is typically used in conjunction with the product of two sparse matrices + * to automatically prune the smallest values as follows: + * \code + * C = (A*B).pruned(); // suppress numerical zeros (exact) + * C = (A*B).pruned(ref); + * C = (A*B).pruned(ref,epsilon); + * \endcode + * where \c ref is a meaningful non zero reference value. + * */ template -const SparseView -SparseMatrixBase::pruned(const Scalar& reference, - const RealScalar& epsilon) const +const SparseView SparseMatrixBase::pruned(const Scalar &reference, const RealScalar &epsilon) const { return SparseView(derived(), reference, epsilon); } -} // end namespace Eigen +}// end namespace Eigen #endif diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/SparseCore/TriangularSolver.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/SparseCore/TriangularSolver.h index f9c56ba7..f87365a6 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/SparseCore/TriangularSolver.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/SparseCore/TriangularSolver.h @@ -10,186 +10,166 @@ #ifndef EIGEN_SPARSETRIANGULARSOLVER_H #define EIGEN_SPARSETRIANGULARSOLVER_H -namespace Eigen { +namespace Eigen { namespace internal { -template::Flags) & RowMajorBit> -struct sparse_solve_triangular_selector; - -// forward substitution, row-major -template -struct sparse_solve_triangular_selector -{ - typedef typename Rhs::Scalar Scalar; - typedef evaluator LhsEval; - typedef typename evaluator::InnerIterator LhsIterator; - static void run(const Lhs& lhs, Rhs& other) + template::Flags) & RowMajorBit> + struct sparse_solve_triangular_selector; + + // forward substitution, row-major + template + struct sparse_solve_triangular_selector { - LhsEval lhsEval(lhs); - for(Index col=0 ; col LhsEval; + typedef typename evaluator::InnerIterator LhsIterator; + static void run(const Lhs &lhs, Rhs &other) { - for(Index i=0; i -struct sparse_solve_triangular_selector -{ - typedef typename Rhs::Scalar Scalar; - typedef evaluator LhsEval; - typedef typename evaluator::InnerIterator LhsIterator; - static void run(const Lhs& lhs, Rhs& other) + // backward substitution, row-major + template + struct sparse_solve_triangular_selector { - LhsEval lhsEval(lhs); - for(Index col=0 ; col LhsEval; + typedef typename evaluator::InnerIterator LhsIterator; + static void run(const Lhs &lhs, Rhs &other) { - for(Index i=lhs.rows()-1 ; i>=0 ; --i) - { - Scalar tmp = other.coeff(i,col); - Scalar l_ii(0); - LhsIterator it(lhsEval, i); - while(it && it.index()= 0; --i) { + Scalar tmp = other.coeff(i, col); + Scalar l_ii(0); + LhsIterator it(lhsEval, i); + while (it && it.index() < i) ++it; + if (!(Mode & UnitDiag)) { + eigen_assert(it && it.index() == i); + l_ii = it.value(); + ++it; + } else if (it && it.index() == i) + ++it; + for (; it; ++it) { tmp -= it.value() * other.coeff(it.index(), col); } - if (Mode & UnitDiag) other.coeffRef(i,col) = tmp; - else other.coeffRef(i,col) = tmp/l_ii; + if (Mode & UnitDiag) + other.coeffRef(i, col) = tmp; + else + other.coeffRef(i, col) = tmp / l_ii; + } } } - } -}; + }; -// forward substitution, col-major -template -struct sparse_solve_triangular_selector -{ - typedef typename Rhs::Scalar Scalar; - typedef evaluator LhsEval; - typedef typename evaluator::InnerIterator LhsIterator; - static void run(const Lhs& lhs, Rhs& other) + // forward substitution, col-major + template + struct sparse_solve_triangular_selector { - LhsEval lhsEval(lhs); - for(Index col=0 ; col LhsEval; + typedef typename evaluator::InnerIterator LhsIterator; + static void run(const Lhs &lhs, Rhs &other) { - for(Index i=0; i -struct sparse_solve_triangular_selector -{ - typedef typename Rhs::Scalar Scalar; - typedef evaluator LhsEval; - typedef typename evaluator::InnerIterator LhsIterator; - static void run(const Lhs& lhs, Rhs& other) + // backward substitution, col-major + template + struct sparse_solve_triangular_selector { - LhsEval lhsEval(lhs); - for(Index col=0 ; col LhsEval; + typedef typename evaluator::InnerIterator LhsIterator; + static void run(const Lhs &lhs, Rhs &other) { - for(Index i=lhs.cols()-1; i>=0; --i) - { - Scalar& tmp = other.coeffRef(i,col); - if (tmp!=Scalar(0)) // optimization when other is actually sparse - { - if(!(Mode & UnitDiag)) + LhsEval lhsEval(lhs); + for (Index col = 0; col < other.cols(); ++col) { + for (Index i = lhs.cols() - 1; i >= 0; --i) { + Scalar &tmp = other.coeffRef(i, col); + if (tmp != Scalar(0))// optimization when other is actually sparse { - // TODO replace this by a binary search. make sure the binary search is safe for partially sorted elements + if (!(Mode & UnitDiag)) { + // TODO replace this by a binary search. make sure the binary search is safe for partially sorted elements + LhsIterator it(lhsEval, i); + while (it && it.index() != i) ++it; + eigen_assert(it && it.index() == i); + other.coeffRef(i, col) /= it.value(); + } LhsIterator it(lhsEval, i); - while(it && it.index()!=i) - ++it; - eigen_assert(it && it.index()==i); - other.coeffRef(i,col) /= it.value(); + for (; it && it.index() < i; ++it) other.coeffRef(it.index(), col) -= tmp * it.value(); } - LhsIterator it(lhsEval, i); - for(; it && it.index() +template template -void TriangularViewImpl::solveInPlace(MatrixBase& other) const +void TriangularViewImpl::solveInPlace(MatrixBase &other) const { eigen_assert(derived().cols() == derived().rows() && derived().cols() == other.rows()); - eigen_assert((!(Mode & ZeroDiag)) && bool(Mode & (Upper|Lower))); + eigen_assert((!(Mode & ZeroDiag)) && bool(Mode & (Upper | Lower))); enum { copy = internal::traits::Flags & RowMajorBit }; typedef typename internal::conditional::type, OtherDerived&>::type OtherCopy; + typename internal::plain_matrix_type_column_major::type, + OtherDerived &>::type OtherCopy; OtherCopy otherCopy(other.derived()); - internal::sparse_solve_triangular_selector::type, Mode>::run(derived().nestedExpression(), otherCopy); + internal::sparse_solve_triangular_selector::type, + Mode>::run(derived().nestedExpression(), otherCopy); - if (copy) - other = otherCopy; + if (copy) other = otherCopy; } #endif @@ -197,119 +177,104 @@ void TriangularViewImpl::solveInPlace(MatrixBase -struct sparse_solve_triangular_sparse_selector; - -// forward substitution, col-major -template -struct sparse_solve_triangular_sparse_selector -{ - typedef typename Rhs::Scalar Scalar; - typedef typename promote_index_type::StorageIndex, - typename traits::StorageIndex>::type StorageIndex; - static void run(const Lhs& lhs, Rhs& other) + template + struct sparse_solve_triangular_sparse_selector; + + // forward substitution, col-major + template + struct sparse_solve_triangular_sparse_selector { - const bool IsLower = (UpLo==Lower); - AmbiVector tempVector(other.rows()*2); - tempVector.setBounds(0,other.rows()); - - Rhs res(other.rows(), other.cols()); - res.reserve(other.nonZeros()); - - for(Index col=0 ; col::StorageIndex, typename traits::StorageIndex>::type + StorageIndex; + static void run(const Lhs &lhs, Rhs &other) { - // FIXME estimate number of non zeros - tempVector.init(.99/*float(other.col(col).nonZeros())/float(other.rows())*/); - tempVector.setZero(); - tempVector.restart(); - for (typename Rhs::InnerIterator rhsIt(other, col); rhsIt; ++rhsIt) - { - tempVector.coeffRef(rhsIt.index()) = rhsIt.value(); - } + const bool IsLower = (UpLo == Lower); + AmbiVector tempVector(other.rows() * 2); + tempVector.setBounds(0, other.rows()); - for(Index i=IsLower?0:lhs.cols()-1; - IsLower?i=0; - i+=IsLower?1:-1) - { + Rhs res(other.rows(), other.cols()); + res.reserve(other.nonZeros()); + + for (Index col = 0; col < other.cols(); ++col) { + // FIXME estimate number of non zeros + tempVector.init(.99 /*float(other.col(col).nonZeros())/float(other.rows())*/); + tempVector.setZero(); tempVector.restart(); - Scalar& ci = tempVector.coeffRef(i); - if (ci!=Scalar(0)) - { - // find - typename Lhs::InnerIterator it(lhs, i); - if(!(Mode & UnitDiag)) - { - if (IsLower) - { - eigen_assert(it.index()==i); - ci /= it.value(); - } - else - ci /= lhs.coeff(i,i); - } + for (typename Rhs::InnerIterator rhsIt(other, col); rhsIt; ++rhsIt) { + tempVector.coeffRef(rhsIt.index()) = rhsIt.value(); + } + + for (Index i = IsLower ? 0 : lhs.cols() - 1; IsLower ? i < lhs.cols() : i >= 0; i += IsLower ? 1 : -1) { tempVector.restart(); - if (IsLower) - { - if (it.index()==i) - ++it; - for(; it; ++it) - tempVector.coeffRef(it.index()) -= ci * it.value(); - } - else - { - for(; it && it.index()::Iterator it(tempVector/*,1e-12*/); it; ++it) - { - ++ count; -// std::cerr << "fill " << it.index() << ", " << col << "\n"; -// std::cout << it.value() << " "; - // FIXME use insertBack - res.insert(it.index(), col) = it.value(); + Index count = 0; + // FIXME compute a reference value to filter zeros + for (typename AmbiVector::Iterator it(tempVector /*,1e-12*/); it; ++it) { + ++count; + // std::cerr << "fill " << it.index() << ", " << col << "\n"; + // std::cout << it.value() << " "; + // FIXME use insertBack + res.insert(it.index(), col) = it.value(); + } + // std::cout << "tempVector.nonZeros() == " << int(count) << " / " << (other.rows()) << "\n"; } -// std::cout << "tempVector.nonZeros() == " << int(count) << " / " << (other.rows()) << "\n"; + res.finalize(); + other = res.markAsRValue(); } - res.finalize(); - other = res.markAsRValue(); - } -}; + }; -} // end namespace internal +}// end namespace internal #ifndef EIGEN_PARSED_BY_DOXYGEN -template +template template -void TriangularViewImpl::solveInPlace(SparseMatrixBase& other) const +void TriangularViewImpl::solveInPlace(SparseMatrixBase &other) const { eigen_assert(derived().cols() == derived().rows() && derived().cols() == other.rows()); - eigen_assert( (!(Mode & ZeroDiag)) && bool(Mode & (Upper|Lower))); + eigen_assert((!(Mode & ZeroDiag)) && bool(Mode & (Upper | Lower))); -// enum { copy = internal::traits::Flags & RowMajorBit }; + // enum { copy = internal::traits::Flags & RowMajorBit }; -// typedef typename internal::conditional::type, OtherDerived&>::type OtherCopy; -// OtherCopy otherCopy(other.derived()); + // typedef typename internal::conditional::type, OtherDerived&>::type OtherCopy; + // OtherCopy otherCopy(other.derived()); - internal::sparse_solve_triangular_sparse_selector::run(derived().nestedExpression(), other.derived()); + internal::sparse_solve_triangular_sparse_selector::run( + derived().nestedExpression(), other.derived()); -// if (copy) -// other = otherCopy; + // if (copy) + // other = otherCopy; } #endif -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_SPARSETRIANGULARSOLVER_H +#endif// EIGEN_SPARSETRIANGULARSOLVER_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/SparseLU/SparseLU.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/SparseLU/SparseLU.h index 7104831c..a3003d02 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/SparseLU/SparseLU.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/SparseLU/SparseLU.h @@ -14,760 +14,728 @@ namespace Eigen { -template > class SparseLU; -template struct SparseLUMatrixLReturnType; -template struct SparseLUMatrixUReturnType; +template> +class SparseLU; +template struct SparseLUMatrixLReturnType; +template struct SparseLUMatrixUReturnType; /** \ingroup SparseLU_Module - * \class SparseLU - * - * \brief Sparse supernodal LU factorization for general matrices - * - * This class implements the supernodal LU factorization for general matrices. - * It uses the main techniques from the sequential SuperLU package - * (http://crd-legacy.lbl.gov/~xiaoye/SuperLU/). It handles transparently real - * and complex arithmetics with single and double precision, depending on the - * scalar type of your input matrix. - * The code has been optimized to provide BLAS-3 operations during supernode-panel updates. - * It benefits directly from the built-in high-performant Eigen BLAS routines. - * Moreover, when the size of a supernode is very small, the BLAS calls are avoided to - * enable a better optimization from the compiler. For best performance, - * you should compile it with NDEBUG flag to avoid the numerous bounds checking on vectors. - * - * An important parameter of this class is the ordering method. It is used to reorder the columns - * (and eventually the rows) of the matrix to reduce the number of new elements that are created during - * numerical factorization. The cheapest method available is COLAMD. - * See \link OrderingMethods_Module the OrderingMethods module \endlink for the list of - * built-in and external ordering methods. - * - * Simple example with key steps - * \code - * VectorXd x(n), b(n); - * SparseMatrix A; - * SparseLU, COLAMDOrdering > solver; - * // fill A and b; - * // Compute the ordering permutation vector from the structural pattern of A - * solver.analyzePattern(A); - * // Compute the numerical factorization - * solver.factorize(A); - * //Use the factors to solve the linear system - * x = solver.solve(b); - * \endcode - * - * \warning The input matrix A should be in a \b compressed and \b column-major form. - * Otherwise an expensive copy will be made. You can call the inexpensive makeCompressed() to get a compressed matrix. - * - * \note Unlike the initial SuperLU implementation, there is no step to equilibrate the matrix. - * For badly scaled matrices, this step can be useful to reduce the pivoting during factorization. - * If this is the case for your matrices, you can try the basic scaling method at - * "unsupported/Eigen/src/IterativeSolvers/Scaling.h" - * - * \tparam _MatrixType The type of the sparse matrix. It must be a column-major SparseMatrix<> - * \tparam _OrderingType The ordering method to use, either AMD, COLAMD or METIS. Default is COLMAD - * - * \implsparsesolverconcept - * - * \sa \ref TutorialSparseSolverConcept - * \sa \ref OrderingMethods_Module - */ -template -class SparseLU : public SparseSolverBase >, public internal::SparseLUImpl + * \class SparseLU + * + * \brief Sparse supernodal LU factorization for general matrices + * + * This class implements the supernodal LU factorization for general matrices. + * It uses the main techniques from the sequential SuperLU package + * (http://crd-legacy.lbl.gov/~xiaoye/SuperLU/). It handles transparently real + * and complex arithmetics with single and double precision, depending on the + * scalar type of your input matrix. + * The code has been optimized to provide BLAS-3 operations during supernode-panel updates. + * It benefits directly from the built-in high-performant Eigen BLAS routines. + * Moreover, when the size of a supernode is very small, the BLAS calls are avoided to + * enable a better optimization from the compiler. For best performance, + * you should compile it with NDEBUG flag to avoid the numerous bounds checking on vectors. + * + * An important parameter of this class is the ordering method. It is used to reorder the columns + * (and eventually the rows) of the matrix to reduce the number of new elements that are created during + * numerical factorization. The cheapest method available is COLAMD. + * See \link OrderingMethods_Module the OrderingMethods module \endlink for the list of + * built-in and external ordering methods. + * + * Simple example with key steps + * \code + * VectorXd x(n), b(n); + * SparseMatrix A; + * SparseLU, COLAMDOrdering > solver; + * // fill A and b; + * // Compute the ordering permutation vector from the structural pattern of A + * solver.analyzePattern(A); + * // Compute the numerical factorization + * solver.factorize(A); + * //Use the factors to solve the linear system + * x = solver.solve(b); + * \endcode + * + * \warning The input matrix A should be in a \b compressed and \b column-major form. + * Otherwise an expensive copy will be made. You can call the inexpensive makeCompressed() to get a compressed matrix. + * + * \note Unlike the initial SuperLU implementation, there is no step to equilibrate the matrix. + * For badly scaled matrices, this step can be useful to reduce the pivoting during factorization. + * If this is the case for your matrices, you can try the basic scaling method at + * "unsupported/Eigen/src/IterativeSolvers/Scaling.h" + * + * \tparam _MatrixType The type of the sparse matrix. It must be a column-major SparseMatrix<> + * \tparam _OrderingType The ordering method to use, either AMD, COLAMD or METIS. Default is COLMAD + * + * \implsparsesolverconcept + * + * \sa \ref TutorialSparseSolverConcept + * \sa \ref OrderingMethods_Module + */ +template +class SparseLU + : public SparseSolverBase> + , public internal::SparseLUImpl { - protected: - typedef SparseSolverBase > APIBase; - using APIBase::m_isInitialized; - public: - using APIBase::_solve_impl; - - typedef _MatrixType MatrixType; - typedef _OrderingType OrderingType; - typedef typename MatrixType::Scalar Scalar; - typedef typename MatrixType::RealScalar RealScalar; - typedef typename MatrixType::StorageIndex StorageIndex; - typedef SparseMatrix NCMatrix; - typedef internal::MappedSuperNodalMatrix SCMatrix; - typedef Matrix ScalarVector; - typedef Matrix IndexVector; - typedef PermutationMatrix PermutationType; - typedef internal::SparseLUImpl Base; - - enum { - ColsAtCompileTime = MatrixType::ColsAtCompileTime, - MaxColsAtCompileTime = MatrixType::MaxColsAtCompileTime - }; - - public: - SparseLU():m_lastError(""),m_Ustore(0,0,0,0,0,0),m_symmetricmode(false),m_diagpivotthresh(1.0),m_detPermR(1) - { - initperfvalues(); - } - explicit SparseLU(const MatrixType& matrix) - : m_lastError(""),m_Ustore(0,0,0,0,0,0),m_symmetricmode(false),m_diagpivotthresh(1.0),m_detPermR(1) - { - initperfvalues(); - compute(matrix); - } - - ~SparseLU() - { - // Free all explicit dynamic pointers - } - - void analyzePattern (const MatrixType& matrix); - void factorize (const MatrixType& matrix); - void simplicialfactorize(const MatrixType& matrix); - - /** - * Compute the symbolic and numeric factorization of the input sparse matrix. - * The input matrix should be in column-major storage. - */ - void compute (const MatrixType& matrix) - { - // Analyze - analyzePattern(matrix); - //Factorize - factorize(matrix); - } - - inline Index rows() const { return m_mat.rows(); } - inline Index cols() const { return m_mat.cols(); } - /** Indicate that the pattern of the input matrix is symmetric */ - void isSymmetric(bool sym) - { - m_symmetricmode = sym; - } - - /** \returns an expression of the matrix L, internally stored as supernodes - * The only operation available with this expression is the triangular solve - * \code - * y = b; matrixL().solveInPlace(y); - * \endcode - */ - SparseLUMatrixLReturnType matrixL() const - { - return SparseLUMatrixLReturnType(m_Lstore); - } - /** \returns an expression of the matrix U, - * The only operation available with this expression is the triangular solve - * \code - * y = b; matrixU().solveInPlace(y); - * \endcode - */ - SparseLUMatrixUReturnType > matrixU() const - { - return SparseLUMatrixUReturnType >(m_Lstore, m_Ustore); - } +protected: + typedef SparseSolverBase> APIBase; + using APIBase::m_isInitialized; - /** - * \returns a reference to the row matrix permutation \f$ P_r \f$ such that \f$P_r A P_c^T = L U\f$ - * \sa colsPermutation() - */ - inline const PermutationType& rowsPermutation() const - { - return m_perm_r; - } - /** - * \returns a reference to the column matrix permutation\f$ P_c^T \f$ such that \f$P_r A P_c^T = L U\f$ - * \sa rowsPermutation() - */ - inline const PermutationType& colsPermutation() const - { - return m_perm_c; - } - /** Set the threshold used for a diagonal entry to be an acceptable pivot. */ - void setPivotThreshold(const RealScalar& thresh) - { - m_diagpivotthresh = thresh; - } +public: + using APIBase::_solve_impl; + + typedef _MatrixType MatrixType; + typedef _OrderingType OrderingType; + typedef typename MatrixType::Scalar Scalar; + typedef typename MatrixType::RealScalar RealScalar; + typedef typename MatrixType::StorageIndex StorageIndex; + typedef SparseMatrix NCMatrix; + typedef internal::MappedSuperNodalMatrix SCMatrix; + typedef Matrix ScalarVector; + typedef Matrix IndexVector; + typedef PermutationMatrix PermutationType; + typedef internal::SparseLUImpl Base; + + enum { ColsAtCompileTime = MatrixType::ColsAtCompileTime, MaxColsAtCompileTime = MatrixType::MaxColsAtCompileTime }; + +public: + SparseLU() + : m_lastError(""), m_Ustore(0, 0, 0, 0, 0, 0), m_symmetricmode(false), m_diagpivotthresh(1.0), m_detPermR(1) + { + initperfvalues(); + } + explicit SparseLU(const MatrixType &matrix) + : m_lastError(""), m_Ustore(0, 0, 0, 0, 0, 0), m_symmetricmode(false), m_diagpivotthresh(1.0), m_detPermR(1) + { + initperfvalues(); + compute(matrix); + } + + ~SparseLU() + { + // Free all explicit dynamic pointers + } + + void analyzePattern(const MatrixType &matrix); + void factorize(const MatrixType &matrix); + void simplicialfactorize(const MatrixType &matrix); + + /** + * Compute the symbolic and numeric factorization of the input sparse matrix. + * The input matrix should be in column-major storage. + */ + void compute(const MatrixType &matrix) + { + // Analyze + analyzePattern(matrix); + // Factorize + factorize(matrix); + } + + inline Index rows() const { return m_mat.rows(); } + inline Index cols() const { return m_mat.cols(); } + /** Indicate that the pattern of the input matrix is symmetric */ + void isSymmetric(bool sym) { m_symmetricmode = sym; } + + /** \returns an expression of the matrix L, internally stored as supernodes + * The only operation available with this expression is the triangular solve + * \code + * y = b; matrixL().solveInPlace(y); + * \endcode + */ + SparseLUMatrixLReturnType matrixL() const { return SparseLUMatrixLReturnType(m_Lstore); } + /** \returns an expression of the matrix U, + * The only operation available with this expression is the triangular solve + * \code + * y = b; matrixU().solveInPlace(y); + * \endcode + */ + SparseLUMatrixUReturnType> matrixU() const + { + return SparseLUMatrixUReturnType>(m_Lstore, m_Ustore); + } + + /** + * \returns a reference to the row matrix permutation \f$ P_r \f$ such that \f$P_r A P_c^T = L U\f$ + * \sa colsPermutation() + */ + inline const PermutationType &rowsPermutation() const { return m_perm_r; } + /** + * \returns a reference to the column matrix permutation\f$ P_c^T \f$ such that \f$P_r A P_c^T = L U\f$ + * \sa rowsPermutation() + */ + inline const PermutationType &colsPermutation() const { return m_perm_c; } + /** Set the threshold used for a diagonal entry to be an acceptable pivot. */ + void setPivotThreshold(const RealScalar &thresh) { m_diagpivotthresh = thresh; } #ifdef EIGEN_PARSED_BY_DOXYGEN - /** \returns the solution X of \f$ A X = B \f$ using the current decomposition of A. - * - * \warning the destination matrix X in X = this->solve(B) must be colmun-major. - * - * \sa compute() - */ - template - inline const Solve solve(const MatrixBase& B) const; -#endif // EIGEN_PARSED_BY_DOXYGEN - - /** \brief Reports whether previous computation was successful. - * - * \returns \c Success if computation was succesful, - * \c NumericalIssue if the LU factorization reports a problem, zero diagonal for instance - * \c InvalidInput if the input matrix is invalid - * - * \sa iparm() - */ - ComputationInfo info() const - { - eigen_assert(m_isInitialized && "Decomposition is not initialized."); - return m_info; - } - - /** - * \returns A string describing the type of error - */ - std::string lastErrorMessage() const - { - return m_lastError; - } + /** \returns the solution X of \f$ A X = B \f$ using the current decomposition of A. + * + * \warning the destination matrix X in X = this->solve(B) must be colmun-major. + * + * \sa compute() + */ + template inline const Solve solve(const MatrixBase &B) const; +#endif// EIGEN_PARSED_BY_DOXYGEN - template - bool _solve_impl(const MatrixBase &B, MatrixBase &X_base) const - { - Dest& X(X_base.derived()); - eigen_assert(m_factorizationIsOk && "The matrix should be factorized first"); - EIGEN_STATIC_ASSERT((Dest::Flags&RowMajorBit)==0, - THIS_METHOD_IS_ONLY_FOR_COLUMN_MAJOR_MATRICES); - - // Permute the right hand side to form X = Pr*B - // on return, X is overwritten by the computed solution - X.resize(B.rows(),B.cols()); - - // this ugly const_cast_derived() helps to detect aliasing when applying the permutations - for(Index j = 0; j < B.cols(); ++j) - X.col(j) = rowsPermutation() * B.const_cast_derived().col(j); - - //Forward substitution with L - this->matrixL().solveInPlace(X); - this->matrixU().solveInPlace(X); - - // Permute back the solution - for (Index j = 0; j < B.cols(); ++j) - X.col(j) = colsPermutation().inverse() * X.col(j); - - return true; - } - - /** - * \returns the absolute value of the determinant of the matrix of which - * *this is the QR decomposition. - * - * \warning a determinant can be very big or small, so for matrices - * of large enough dimension, there is a risk of overflow/underflow. - * One way to work around that is to use logAbsDeterminant() instead. - * - * \sa logAbsDeterminant(), signDeterminant() - */ - Scalar absDeterminant() - { - using std::abs; - eigen_assert(m_factorizationIsOk && "The matrix should be factorized first."); - // Initialize with the determinant of the row matrix - Scalar det = Scalar(1.); - // Note that the diagonal blocks of U are stored in supernodes, - // which are available in the L part :) - for (Index j = 0; j < this->cols(); ++j) - { - for (typename SCMatrix::InnerIterator it(m_Lstore, j); it; ++it) - { - if(it.index() == j) - { - det *= abs(it.value()); - break; - } + /** \brief Reports whether previous computation was successful. + * + * \returns \c Success if computation was succesful, + * \c NumericalIssue if the LU factorization reports a problem, zero diagonal for instance + * \c InvalidInput if the input matrix is invalid + * + * \sa iparm() + */ + ComputationInfo info() const + { + eigen_assert(m_isInitialized && "Decomposition is not initialized."); + return m_info; + } + + /** + * \returns A string describing the type of error + */ + std::string lastErrorMessage() const { return m_lastError; } + + template bool _solve_impl(const MatrixBase &B, MatrixBase &X_base) const + { + Dest &X(X_base.derived()); + eigen_assert(m_factorizationIsOk && "The matrix should be factorized first"); + EIGEN_STATIC_ASSERT((Dest::Flags & RowMajorBit) == 0, THIS_METHOD_IS_ONLY_FOR_COLUMN_MAJOR_MATRICES); + + // Permute the right hand side to form X = Pr*B + // on return, X is overwritten by the computed solution + X.resize(B.rows(), B.cols()); + + // this ugly const_cast_derived() helps to detect aliasing when applying the permutations + for (Index j = 0; j < B.cols(); ++j) X.col(j) = rowsPermutation() * B.const_cast_derived().col(j); + + // Forward substitution with L + this->matrixL().solveInPlace(X); + this->matrixU().solveInPlace(X); + + // Permute back the solution + for (Index j = 0; j < B.cols(); ++j) X.col(j) = colsPermutation().inverse() * X.col(j); + + return true; + } + + /** + * \returns the absolute value of the determinant of the matrix of which + * *this is the QR decomposition. + * + * \warning a determinant can be very big or small, so for matrices + * of large enough dimension, there is a risk of overflow/underflow. + * One way to work around that is to use logAbsDeterminant() instead. + * + * \sa logAbsDeterminant(), signDeterminant() + */ + Scalar absDeterminant() + { + using std::abs; + eigen_assert(m_factorizationIsOk && "The matrix should be factorized first."); + // Initialize with the determinant of the row matrix + Scalar det = Scalar(1.); + // Note that the diagonal blocks of U are stored in supernodes, + // which are available in the L part :) + for (Index j = 0; j < this->cols(); ++j) { + for (typename SCMatrix::InnerIterator it(m_Lstore, j); it; ++it) { + if (it.index() == j) { + det *= abs(it.value()); + break; } } - return det; } + return det; + } - /** \returns the natural log of the absolute value of the determinant of the matrix - * of which **this is the QR decomposition - * - * \note This method is useful to work around the risk of overflow/underflow that's - * inherent to the determinant computation. - * - * \sa absDeterminant(), signDeterminant() - */ - Scalar logAbsDeterminant() const - { - using std::log; - using std::abs; - - eigen_assert(m_factorizationIsOk && "The matrix should be factorized first."); - Scalar det = Scalar(0.); - for (Index j = 0; j < this->cols(); ++j) - { - for (typename SCMatrix::InnerIterator it(m_Lstore, j); it; ++it) - { - if(it.row() < j) continue; - if(it.row() == j) - { - det += log(abs(it.value())); - break; - } + /** \returns the natural log of the absolute value of the determinant of the matrix + * of which **this is the QR decomposition + * + * \note This method is useful to work around the risk of overflow/underflow that's + * inherent to the determinant computation. + * + * \sa absDeterminant(), signDeterminant() + */ + Scalar logAbsDeterminant() const + { + using std::log; + using std::abs; + + eigen_assert(m_factorizationIsOk && "The matrix should be factorized first."); + Scalar det = Scalar(0.); + for (Index j = 0; j < this->cols(); ++j) { + for (typename SCMatrix::InnerIterator it(m_Lstore, j); it; ++it) { + if (it.row() < j) continue; + if (it.row() == j) { + det += log(abs(it.value())); + break; } } - return det; } + return det; + } - /** \returns A number representing the sign of the determinant - * - * \sa absDeterminant(), logAbsDeterminant() - */ - Scalar signDeterminant() - { - eigen_assert(m_factorizationIsOk && "The matrix should be factorized first."); - // Initialize with the determinant of the row matrix - Index det = 1; - // Note that the diagonal blocks of U are stored in supernodes, - // which are available in the L part :) - for (Index j = 0; j < this->cols(); ++j) - { - for (typename SCMatrix::InnerIterator it(m_Lstore, j); it; ++it) - { - if(it.index() == j) - { - if(it.value()<0) - det = -det; - else if(it.value()==0) - return 0; - break; - } + /** \returns A number representing the sign of the determinant + * + * \sa absDeterminant(), logAbsDeterminant() + */ + Scalar signDeterminant() + { + eigen_assert(m_factorizationIsOk && "The matrix should be factorized first."); + // Initialize with the determinant of the row matrix + Index det = 1; + // Note that the diagonal blocks of U are stored in supernodes, + // which are available in the L part :) + for (Index j = 0; j < this->cols(); ++j) { + for (typename SCMatrix::InnerIterator it(m_Lstore, j); it; ++it) { + if (it.index() == j) { + if (it.value() < 0) + det = -det; + else if (it.value() == 0) + return 0; + break; } } - return det * m_detPermR * m_detPermC; } - - /** \returns The determinant of the matrix. - * - * \sa absDeterminant(), logAbsDeterminant() - */ - Scalar determinant() - { - eigen_assert(m_factorizationIsOk && "The matrix should be factorized first."); - // Initialize with the determinant of the row matrix - Scalar det = Scalar(1.); - // Note that the diagonal blocks of U are stored in supernodes, - // which are available in the L part :) - for (Index j = 0; j < this->cols(); ++j) - { - for (typename SCMatrix::InnerIterator it(m_Lstore, j); it; ++it) - { - if(it.index() == j) - { - det *= it.value(); - break; - } + return det * m_detPermR * m_detPermC; + } + + /** \returns The determinant of the matrix. + * + * \sa absDeterminant(), logAbsDeterminant() + */ + Scalar determinant() + { + eigen_assert(m_factorizationIsOk && "The matrix should be factorized first."); + // Initialize with the determinant of the row matrix + Scalar det = Scalar(1.); + // Note that the diagonal blocks of U are stored in supernodes, + // which are available in the L part :) + for (Index j = 0; j < this->cols(); ++j) { + for (typename SCMatrix::InnerIterator it(m_Lstore, j); it; ++it) { + if (it.index() == j) { + det *= it.value(); + break; } } - return (m_detPermR * m_detPermC) > 0 ? det : -det; } + return (m_detPermR * m_detPermC) > 0 ? det : -det; + } - protected: - // Functions - void initperfvalues() - { - m_perfv.panel_size = 16; - m_perfv.relax = 1; - m_perfv.maxsuper = 128; - m_perfv.rowblk = 16; - m_perfv.colblk = 8; - m_perfv.fillfactor = 20; - } - - // Variables - mutable ComputationInfo m_info; - bool m_factorizationIsOk; - bool m_analysisIsOk; - std::string m_lastError; - NCMatrix m_mat; // The input (permuted ) matrix - SCMatrix m_Lstore; // The lower triangular matrix (supernodal) - MappedSparseMatrix m_Ustore; // The upper triangular matrix - PermutationType m_perm_c; // Column permutation - PermutationType m_perm_r ; // Row permutation - IndexVector m_etree; // Column elimination tree - - typename Base::GlobalLU_t m_glu; - - // SparseLU options - bool m_symmetricmode; - // values for performance - internal::perfvalues m_perfv; - RealScalar m_diagpivotthresh; // Specifies the threshold used for a diagonal entry to be an acceptable pivot - Index m_nnzL, m_nnzU; // Nonzeros in L and U factors - Index m_detPermR, m_detPermC; // Determinants of the permutation matrices - private: - // Disable copy constructor - SparseLU (const SparseLU& ); - -}; // End class SparseLU +protected: + // Functions + void initperfvalues() + { + m_perfv.panel_size = 16; + m_perfv.relax = 1; + m_perfv.maxsuper = 128; + m_perfv.rowblk = 16; + m_perfv.colblk = 8; + m_perfv.fillfactor = 20; + } + + // Variables + mutable ComputationInfo m_info; + bool m_factorizationIsOk; + bool m_analysisIsOk; + std::string m_lastError; + NCMatrix m_mat;// The input (permuted ) matrix + SCMatrix m_Lstore;// The lower triangular matrix (supernodal) + MappedSparseMatrix m_Ustore;// The upper triangular matrix + PermutationType m_perm_c;// Column permutation + PermutationType m_perm_r;// Row permutation + IndexVector m_etree;// Column elimination tree + + typename Base::GlobalLU_t m_glu; + + // SparseLU options + bool m_symmetricmode; + // values for performance + internal::perfvalues m_perfv; + RealScalar m_diagpivotthresh;// Specifies the threshold used for a diagonal entry to be an acceptable pivot + Index m_nnzL, m_nnzU;// Nonzeros in L and U factors + Index m_detPermR, m_detPermC;// Determinants of the permutation matrices +private: + // Disable copy constructor + SparseLU(const SparseLU &); +};// End class SparseLU // Functions needed by the anaysis phase -/** - * Compute the column permutation to minimize the fill-in - * - * - Apply this permutation to the input matrix - - * - * - Compute the column elimination tree on the permuted matrix - * - * - Postorder the elimination tree and the column permutation - * - */ -template -void SparseLU::analyzePattern(const MatrixType& mat) +/** + * Compute the column permutation to minimize the fill-in + * + * - Apply this permutation to the input matrix - + * + * - Compute the column elimination tree on the permuted matrix + * + * - Postorder the elimination tree and the column permutation + * + */ +template +void SparseLU::analyzePattern(const MatrixType &mat) { - - //TODO It is possible as in SuperLU to compute row and columns scaling vectors to equilibrate the matrix mat. - - // Firstly, copy the whole input matrix. + + // TODO It is possible as in SuperLU to compute row and columns scaling vectors to equilibrate the matrix mat. + + // Firstly, copy the whole input matrix. m_mat = mat; - + // Compute fill-in ordering - OrderingType ord; - ord(m_mat,m_perm_c); - + OrderingType ord; + ord(m_mat, m_perm_c); + // Apply the permutation to the column of the input matrix - if (m_perm_c.size()) - { - m_mat.uncompress(); //NOTE: The effect of this command is only to create the InnerNonzeros pointers. FIXME : This vector is filled but not subsequently used. + if (m_perm_c.size()) { + m_mat.uncompress();// NOTE: The effect of this command is only to create the InnerNonzeros pointers. FIXME : This + // vector is filled but not subsequently used. // Then, permute only the column pointers - ei_declare_aligned_stack_constructed_variable(StorageIndex,outerIndexPtr,mat.cols()+1,mat.isCompressed()?const_cast(mat.outerIndexPtr()):0); - - // If the input matrix 'mat' is uncompressed, then the outer-indices do not match the ones of m_mat, and a copy is thus needed. - if(!mat.isCompressed()) - IndexVector::Map(outerIndexPtr, mat.cols()+1) = IndexVector::Map(m_mat.outerIndexPtr(),mat.cols()+1); - + ei_declare_aligned_stack_constructed_variable(StorageIndex, + outerIndexPtr, + mat.cols() + 1, + mat.isCompressed() ? const_cast(mat.outerIndexPtr()) : 0); + + // If the input matrix 'mat' is uncompressed, then the outer-indices do not match the ones of m_mat, and a copy is + // thus needed. + if (!mat.isCompressed()) + IndexVector::Map(outerIndexPtr, mat.cols() + 1) = IndexVector::Map(m_mat.outerIndexPtr(), mat.cols() + 1); + // Apply the permutation and compute the nnz per column. - for (Index i = 0; i < mat.cols(); i++) - { + for (Index i = 0; i < mat.cols(); i++) { m_mat.outerIndexPtr()[m_perm_c.indices()(i)] = outerIndexPtr[i]; - m_mat.innerNonZeroPtr()[m_perm_c.indices()(i)] = outerIndexPtr[i+1] - outerIndexPtr[i]; + m_mat.innerNonZeroPtr()[m_perm_c.indices()(i)] = outerIndexPtr[i + 1] - outerIndexPtr[i]; } } - - // Compute the column elimination tree of the permuted matrix + + // Compute the column elimination tree of the permuted matrix IndexVector firstRowElt; - internal::coletree(m_mat, m_etree,firstRowElt); - + internal::coletree(m_mat, m_etree, firstRowElt); + // In symmetric mode, do not do postorder here if (!m_symmetricmode) { - IndexVector post, iwork; + IndexVector post, iwork; // Post order etree - internal::treePostorder(StorageIndex(m_mat.cols()), m_etree, post); - - - // Renumber etree in postorder - Index m = m_mat.cols(); - iwork.resize(m+1); + internal::treePostorder(StorageIndex(m_mat.cols()), m_etree, post); + + + // Renumber etree in postorder + Index m = m_mat.cols(); + iwork.resize(m + 1); for (Index i = 0; i < m; ++i) iwork(post(i)) = post(m_etree(i)); m_etree = iwork; - + // Postmultiply A*Pc by post, i.e reorder the matrix according to the postorder of the etree - PermutationType post_perm(m); - for (Index i = 0; i < m; i++) - post_perm.indices()(i) = post(i); - + PermutationType post_perm(m); + for (Index i = 0; i < m; i++) post_perm.indices()(i) = post(i); + // Combine the two permutations : postorder the permutation for future use - if(m_perm_c.size()) { - m_perm_c = post_perm * m_perm_c; - } - - } // end postordering - - m_analysisIsOk = true; + if (m_perm_c.size()) { m_perm_c = post_perm * m_perm_c; } + + }// end postordering + + m_analysisIsOk = true; } // Functions needed by the numerical factorization phase -/** - * - Numerical factorization - * - Interleaved with the symbolic factorization - * On exit, info is - * - * = 0: successful factorization - * - * > 0: if info = i, and i is - * - * <= A->ncol: U(i,i) is exactly zero. The factorization has - * been completed, but the factor U is exactly singular, - * and division by zero will occur if it is used to solve a - * system of equations. - * - * > A->ncol: number of bytes allocated when memory allocation - * failure occurred, plus A->ncol. If lwork = -1, it is - * the estimated amount of space needed, plus A->ncol. - */ -template -void SparseLU::factorize(const MatrixType& matrix) +/** + * - Numerical factorization + * - Interleaved with the symbolic factorization + * On exit, info is + * + * = 0: successful factorization + * + * > 0: if info = i, and i is + * + * <= A->ncol: U(i,i) is exactly zero. The factorization has + * been completed, but the factor U is exactly singular, + * and division by zero will occur if it is used to solve a + * system of equations. + * + * > A->ncol: number of bytes allocated when memory allocation + * failure occurred, plus A->ncol. If lwork = -1, it is + * the estimated amount of space needed, plus A->ncol. + */ +template +void SparseLU::factorize(const MatrixType &matrix) { using internal::emptyIdxLU; - eigen_assert(m_analysisIsOk && "analyzePattern() should be called first"); + eigen_assert(m_analysisIsOk && "analyzePattern() should be called first"); eigen_assert((matrix.rows() == matrix.cols()) && "Only for squared matrices"); - + m_isInitialized = true; - - + + // Apply the column permutation computed in analyzepattern() - // m_mat = matrix * m_perm_c.inverse(); + // m_mat = matrix * m_perm_c.inverse(); m_mat = matrix; - if (m_perm_c.size()) - { - m_mat.uncompress(); //NOTE: The effect of this command is only to create the InnerNonzeros pointers. - //Then, permute only the column pointers - const StorageIndex * outerIndexPtr; - if (matrix.isCompressed()) outerIndexPtr = matrix.outerIndexPtr(); - else - { - StorageIndex* outerIndexPtr_t = new StorageIndex[matrix.cols()+1]; - for(Index i = 0; i <= matrix.cols(); i++) outerIndexPtr_t[i] = m_mat.outerIndexPtr()[i]; + if (m_perm_c.size()) { + m_mat.uncompress();// NOTE: The effect of this command is only to create the InnerNonzeros pointers. + // Then, permute only the column pointers + const StorageIndex *outerIndexPtr; + if (matrix.isCompressed()) + outerIndexPtr = matrix.outerIndexPtr(); + else { + StorageIndex *outerIndexPtr_t = new StorageIndex[matrix.cols() + 1]; + for (Index i = 0; i <= matrix.cols(); i++) outerIndexPtr_t[i] = m_mat.outerIndexPtr()[i]; outerIndexPtr = outerIndexPtr_t; } - for (Index i = 0; i < matrix.cols(); i++) - { + for (Index i = 0; i < matrix.cols(); i++) { m_mat.outerIndexPtr()[m_perm_c.indices()(i)] = outerIndexPtr[i]; - m_mat.innerNonZeroPtr()[m_perm_c.indices()(i)] = outerIndexPtr[i+1] - outerIndexPtr[i]; + m_mat.innerNonZeroPtr()[m_perm_c.indices()(i)] = outerIndexPtr[i + 1] - outerIndexPtr[i]; } - if(!matrix.isCompressed()) delete[] outerIndexPtr; - } - else - { //FIXME This should not be needed if the empty permutation is handled transparently + if (!matrix.isCompressed()) delete[] outerIndexPtr; + } else {// FIXME This should not be needed if the empty permutation is handled transparently m_perm_c.resize(matrix.cols()); - for(StorageIndex i = 0; i < matrix.cols(); ++i) m_perm_c.indices()(i) = i; + for (StorageIndex i = 0; i < matrix.cols(); ++i) m_perm_c.indices()(i) = i; } - + Index m = m_mat.rows(); Index n = m_mat.cols(); Index nnz = m_mat.nonZeros(); Index maxpanel = m_perfv.panel_size * m; // Allocate working storage common to the factor routines Index lwork = 0; - Index info = Base::memInit(m, n, nnz, lwork, m_perfv.fillfactor, m_perfv.panel_size, m_glu); - if (info) - { - m_lastError = "UNABLE TO ALLOCATE WORKING MEMORY\n\n" ; + Index info = Base::memInit(m, n, nnz, lwork, m_perfv.fillfactor, m_perfv.panel_size, m_glu); + if (info) { + m_lastError = "UNABLE TO ALLOCATE WORKING MEMORY\n\n"; m_factorizationIsOk = false; - return ; + return; } - - // Set up pointers for integer working arrays - IndexVector segrep(m); segrep.setZero(); - IndexVector parent(m); parent.setZero(); - IndexVector xplore(m); xplore.setZero(); + + // Set up pointers for integer working arrays + IndexVector segrep(m); + segrep.setZero(); + IndexVector parent(m); + parent.setZero(); + IndexVector xplore(m); + xplore.setZero(); IndexVector repfnz(maxpanel); IndexVector panel_lsub(maxpanel); - IndexVector xprune(n); xprune.setZero(); - IndexVector marker(m*internal::LUNoMarker); marker.setZero(); - - repfnz.setConstant(-1); + IndexVector xprune(n); + xprune.setZero(); + IndexVector marker(m * internal::LUNoMarker); + marker.setZero(); + + repfnz.setConstant(-1); panel_lsub.setConstant(-1); - - // Set up pointers for scalar working arrays - ScalarVector dense; + + // Set up pointers for scalar working arrays + ScalarVector dense; dense.setZero(maxpanel); - ScalarVector tempv; - tempv.setZero(internal::LUnumTempV(m, m_perfv.panel_size, m_perfv.maxsuper, /*m_perfv.rowblk*/m) ); - + ScalarVector tempv; + tempv.setZero(internal::LUnumTempV(m, m_perfv.panel_size, m_perfv.maxsuper, /*m_perfv.rowblk*/ m)); + // Compute the inverse of perm_c - PermutationType iperm_c(m_perm_c.inverse()); - + PermutationType iperm_c(m_perm_c.inverse()); + // Identify initial relaxed snodes IndexVector relax_end(n); - if ( m_symmetricmode == true ) + if (m_symmetricmode == true) Base::heap_relax_snode(n, m_etree, m_perfv.relax, marker, relax_end); else Base::relax_snode(n, m_etree, m_perfv.relax, marker, relax_end); - - - m_perm_r.resize(m); + + + m_perm_r.resize(m); m_perm_r.indices().setConstant(-1); marker.setConstant(-1); - m_detPermR = 1; // Record the determinant of the row permutation - - m_glu.supno(0) = emptyIdxLU; m_glu.xsup.setConstant(0); + m_detPermR = 1;// Record the determinant of the row permutation + + m_glu.supno(0) = emptyIdxLU; + m_glu.xsup.setConstant(0); m_glu.xsup(0) = m_glu.xlsub(0) = m_glu.xusub(0) = m_glu.xlusup(0) = Index(0); - + // Work on one 'panel' at a time. A panel is one of the following : // (a) a relaxed supernode at the bottom of the etree, or // (b) panel_size contiguous columns, defined by the user - Index jcol; + Index jcol; IndexVector panel_histo(n); - Index pivrow; // Pivotal row number in the original row matrix - Index nseg1; // Number of segments in U-column above panel row jcol - Index nseg; // Number of segments in each U-column - Index irep; - Index i, k, jj; - for (jcol = 0; jcol < n; ) - { - // Adjust panel size so that a panel won't overlap with the next relaxed snode. - Index panel_size = m_perfv.panel_size; // upper bound on panel width - for (k = jcol + 1; k < (std::min)(jcol+panel_size, n); k++) - { - if (relax_end(k) != emptyIdxLU) - { - panel_size = k - jcol; - break; + Index pivrow;// Pivotal row number in the original row matrix + Index nseg1;// Number of segments in U-column above panel row jcol + Index nseg;// Number of segments in each U-column + Index irep; + Index i, k, jj; + for (jcol = 0; jcol < n;) { + // Adjust panel size so that a panel won't overlap with the next relaxed snode. + Index panel_size = m_perfv.panel_size;// upper bound on panel width + for (k = jcol + 1; k < (std::min)(jcol + panel_size, n); k++) { + if (relax_end(k) != emptyIdxLU) { + panel_size = k - jcol; + break; } } - if (k == n) - panel_size = n - jcol; - - // Symbolic outer factorization on a panel of columns - Base::panel_dfs(m, panel_size, jcol, m_mat, m_perm_r.indices(), nseg1, dense, panel_lsub, segrep, repfnz, xprune, marker, parent, xplore, m_glu); - - // Numeric sup-panel updates in topological order - Base::panel_bmod(m, panel_size, jcol, nseg1, dense, tempv, segrep, repfnz, m_glu); - - // Sparse LU within the panel, and below the panel diagonal - for ( jj = jcol; jj< jcol + panel_size; jj++) - { - k = (jj - jcol) * m; // Column index for w-wide arrays - - nseg = nseg1; // begin after all the panel segments - //Depth-first-search for the current column + if (k == n) panel_size = n - jcol; + + // Symbolic outer factorization on a panel of columns + Base::panel_dfs(m, + panel_size, + jcol, + m_mat, + m_perm_r.indices(), + nseg1, + dense, + panel_lsub, + segrep, + repfnz, + xprune, + marker, + parent, + xplore, + m_glu); + + // Numeric sup-panel updates in topological order + Base::panel_bmod(m, panel_size, jcol, nseg1, dense, tempv, segrep, repfnz, m_glu); + + // Sparse LU within the panel, and below the panel diagonal + for (jj = jcol; jj < jcol + panel_size; jj++) { + k = (jj - jcol) * m;// Column index for w-wide arrays + + nseg = nseg1;// begin after all the panel segments + // Depth-first-search for the current column VectorBlock panel_lsubk(panel_lsub, k, m); - VectorBlock repfnz_k(repfnz, k, m); - info = Base::column_dfs(m, jj, m_perm_r.indices(), m_perfv.maxsuper, nseg, panel_lsubk, segrep, repfnz_k, xprune, marker, parent, xplore, m_glu); - if ( info ) - { - m_lastError = "UNABLE TO EXPAND MEMORY IN COLUMN_DFS() "; - m_info = NumericalIssue; - m_factorizationIsOk = false; - return; + VectorBlock repfnz_k(repfnz, k, m); + info = Base::column_dfs(m, + jj, + m_perm_r.indices(), + m_perfv.maxsuper, + nseg, + panel_lsubk, + segrep, + repfnz_k, + xprune, + marker, + parent, + xplore, + m_glu); + if (info) { + m_lastError = "UNABLE TO EXPAND MEMORY IN COLUMN_DFS() "; + m_info = NumericalIssue; + m_factorizationIsOk = false; + return; } - // Numeric updates to this column - VectorBlock dense_k(dense, k, m); - VectorBlock segrep_k(segrep, nseg1, m-nseg1); - info = Base::column_bmod(jj, (nseg - nseg1), dense_k, tempv, segrep_k, repfnz_k, jcol, m_glu); - if ( info ) - { + // Numeric updates to this column + VectorBlock dense_k(dense, k, m); + VectorBlock segrep_k(segrep, nseg1, m - nseg1); + info = Base::column_bmod(jj, (nseg - nseg1), dense_k, tempv, segrep_k, repfnz_k, jcol, m_glu); + if (info) { m_lastError = "UNABLE TO EXPAND MEMORY IN COLUMN_BMOD() "; - m_info = NumericalIssue; - m_factorizationIsOk = false; - return; + m_info = NumericalIssue; + m_factorizationIsOk = false; + return; } - + // Copy the U-segments to ucol(*) - info = Base::copy_to_ucol(jj, nseg, segrep, repfnz_k ,m_perm_r.indices(), dense_k, m_glu); - if ( info ) - { + info = Base::copy_to_ucol(jj, nseg, segrep, repfnz_k, m_perm_r.indices(), dense_k, m_glu); + if (info) { m_lastError = "UNABLE TO EXPAND MEMORY IN COPY_TO_UCOL() "; - m_info = NumericalIssue; - m_factorizationIsOk = false; - return; + m_info = NumericalIssue; + m_factorizationIsOk = false; + return; } - - // Form the L-segment + + // Form the L-segment info = Base::pivotL(jj, m_diagpivotthresh, m_perm_r.indices(), iperm_c.indices(), pivrow, m_glu); - if ( info ) - { + if (info) { m_lastError = "THE MATRIX IS STRUCTURALLY SINGULAR ... ZERO COLUMN AT "; std::ostringstream returnInfo; - returnInfo << info; + returnInfo << info; m_lastError += returnInfo.str(); - m_info = NumericalIssue; - m_factorizationIsOk = false; - return; + m_info = NumericalIssue; + m_factorizationIsOk = false; + return; } - + // Update the determinant of the row permutation matrix - // FIXME: the following test is not correct, we should probably take iperm_c into account and pivrow is not directly the row pivot. + // FIXME: the following test is not correct, we should probably take iperm_c into account and pivrow is not + // directly the row pivot. if (pivrow != jj) m_detPermR = -m_detPermR; // Prune columns (0:jj-1) using column jj - Base::pruneL(jj, m_perm_r.indices(), pivrow, nseg, segrep, repfnz_k, xprune, m_glu); - - // Reset repfnz for this column - for (i = 0; i < nseg; i++) - { - irep = segrep(i); - repfnz_k(irep) = emptyIdxLU; + Base::pruneL(jj, m_perm_r.indices(), pivrow, nseg, segrep, repfnz_k, xprune, m_glu); + + // Reset repfnz for this column + for (i = 0; i < nseg; i++) { + irep = segrep(i); + repfnz_k(irep) = emptyIdxLU; } - } // end SparseLU within the panel - jcol += panel_size; // Move to the next panel - } // end for -- end elimination - + }// end SparseLU within the panel + jcol += panel_size;// Move to the next panel + }// end for -- end elimination + m_detPermR = m_perm_r.determinant(); m_detPermC = m_perm_c.determinant(); - - // Count the number of nonzeros in factors - Base::countnz(n, m_nnzL, m_nnzU, m_glu); - // Apply permutation to the L subscripts + + // Count the number of nonzeros in factors + Base::countnz(n, m_nnzL, m_nnzU, m_glu); + // Apply permutation to the L subscripts Base::fixupL(n, m_perm_r.indices(), m_glu); - - // Create supernode matrix L - m_Lstore.setInfos(m, n, m_glu.lusup, m_glu.xlusup, m_glu.lsub, m_glu.xlsub, m_glu.supno, m_glu.xsup); - // Create the column major upper sparse matrix U; - new (&m_Ustore) MappedSparseMatrix ( m, n, m_nnzU, m_glu.xusub.data(), m_glu.usub.data(), m_glu.ucol.data() ); - + + // Create supernode matrix L + m_Lstore.setInfos(m, n, m_glu.lusup, m_glu.xlusup, m_glu.lsub, m_glu.xlsub, m_glu.supno, m_glu.xsup); + // Create the column major upper sparse matrix U; + new (&m_Ustore) MappedSparseMatrix( + m, n, m_nnzU, m_glu.xusub.data(), m_glu.usub.data(), m_glu.ucol.data()); + m_info = Success; m_factorizationIsOk = true; } -template -struct SparseLUMatrixLReturnType : internal::no_assignment_operator +template struct SparseLUMatrixLReturnType : internal::no_assignment_operator { typedef typename MappedSupernodalType::Scalar Scalar; - explicit SparseLUMatrixLReturnType(const MappedSupernodalType& mapL) : m_mapL(mapL) - { } + explicit SparseLUMatrixLReturnType(const MappedSupernodalType &mapL) : m_mapL(mapL) {} Index rows() { return m_mapL.rows(); } Index cols() { return m_mapL.cols(); } - template - void solveInPlace( MatrixBase &X) const - { - m_mapL.solveInPlace(X); - } - const MappedSupernodalType& m_mapL; + template void solveInPlace(MatrixBase &X) const { m_mapL.solveInPlace(X); } + const MappedSupernodalType &m_mapL; }; -template -struct SparseLUMatrixUReturnType : internal::no_assignment_operator +template struct SparseLUMatrixUReturnType : internal::no_assignment_operator { typedef typename MatrixLType::Scalar Scalar; - SparseLUMatrixUReturnType(const MatrixLType& mapL, const MatrixUType& mapU) - : m_mapL(mapL),m_mapU(mapU) - { } + SparseLUMatrixUReturnType(const MatrixLType &mapL, const MatrixUType &mapU) : m_mapL(mapL), m_mapU(mapU) {} Index rows() { return m_mapL.rows(); } Index cols() { return m_mapL.cols(); } - template void solveInPlace(MatrixBase &X) const + template void solveInPlace(MatrixBase &X) const { Index nrhs = X.cols(); - Index n = X.rows(); + Index n = X.rows(); // Backward solve with U - for (Index k = m_mapL.nsuper(); k >= 0; k--) - { + for (Index k = m_mapL.nsuper(); k >= 0; k--) { Index fsupc = m_mapL.supToCol()[k]; - Index lda = m_mapL.colIndexPtr()[fsupc+1] - m_mapL.colIndexPtr()[fsupc]; // leading dimension - Index nsupc = m_mapL.supToCol()[k+1] - fsupc; + Index lda = m_mapL.colIndexPtr()[fsupc + 1] - m_mapL.colIndexPtr()[fsupc];// leading dimension + Index nsupc = m_mapL.supToCol()[k + 1] - fsupc; Index luptr = m_mapL.colIndexPtr()[fsupc]; - if (nsupc == 1) - { - for (Index j = 0; j < nrhs; j++) - { - X(fsupc, j) /= m_mapL.valuePtr()[luptr]; - } - } - else - { - Map, 0, OuterStride<> > A( &(m_mapL.valuePtr()[luptr]), nsupc, nsupc, OuterStride<>(lda) ); - Map< Matrix, 0, OuterStride<> > U (&(X(fsupc,0)), nsupc, nrhs, OuterStride<>(n) ); + if (nsupc == 1) { + for (Index j = 0; j < nrhs; j++) { X(fsupc, j) /= m_mapL.valuePtr()[luptr]; } + } else { + Map, 0, OuterStride<>> A( + &(m_mapL.valuePtr()[luptr]), nsupc, nsupc, OuterStride<>(lda)); + Map, 0, OuterStride<>> U( + &(X(fsupc, 0)), nsupc, nrhs, OuterStride<>(n)); U = A.template triangularView().solve(U); } - for (Index j = 0; j < nrhs; ++j) - { - for (Index jcol = fsupc; jcol < fsupc + nsupc; jcol++) - { + for (Index j = 0; j < nrhs; ++j) { + for (Index jcol = fsupc; jcol < fsupc + nsupc; jcol++) { typename MatrixUType::InnerIterator it(m_mapU, jcol); - for ( ; it; ++it) - { + for (; it; ++it) { Index irow = it.index(); X(irow, j) -= X(jcol, j) * it.value(); } } } - } // End For U-solve + }// End For U-solve } - const MatrixLType& m_mapL; - const MatrixUType& m_mapU; + const MatrixLType &m_mapL; + const MatrixUType &m_mapU; }; -} // End namespace Eigen +}// End namespace Eigen #endif diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/SparseLU/SparseLUImpl.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/SparseLU/SparseLUImpl.h index fc0cfc4d..39a14865 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/SparseLU/SparseLUImpl.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/SparseLU/SparseLUImpl.h @@ -11,56 +11,136 @@ namespace Eigen { namespace internal { - -/** \ingroup SparseLU_Module - * \class SparseLUImpl - * Base class for sparseLU - */ -template -class SparseLUImpl -{ + + /** \ingroup SparseLU_Module + * \class SparseLUImpl + * Base class for sparseLU + */ + template class SparseLUImpl + { public: - typedef Matrix ScalarVector; - typedef Matrix IndexVector; - typedef Matrix ScalarMatrix; - typedef Map > MappedMatrixBlock; - typedef typename ScalarVector::RealScalar RealScalar; - typedef Ref > BlockScalarVector; - typedef Ref > BlockIndexVector; - typedef LU_GlobalLU_t GlobalLU_t; - typedef SparseMatrix MatrixType; - + typedef Matrix ScalarVector; + typedef Matrix IndexVector; + typedef Matrix ScalarMatrix; + typedef Map> MappedMatrixBlock; + typedef typename ScalarVector::RealScalar RealScalar; + typedef Ref> BlockScalarVector; + typedef Ref> BlockIndexVector; + typedef LU_GlobalLU_t GlobalLU_t; + typedef SparseMatrix MatrixType; + protected: - template - Index expand(VectorType& vec, Index& length, Index nbElts, Index keep_prev, Index& num_expansions); - Index memInit(Index m, Index n, Index annz, Index lwork, Index fillratio, Index panel_size, GlobalLU_t& glu); - template - Index memXpand(VectorType& vec, Index& maxlen, Index nbElts, MemType memtype, Index& num_expansions); - void heap_relax_snode (const Index n, IndexVector& et, const Index relax_columns, IndexVector& descendants, IndexVector& relax_end); - void relax_snode (const Index n, IndexVector& et, const Index relax_columns, IndexVector& descendants, IndexVector& relax_end); - Index snode_dfs(const Index jcol, const Index kcol,const MatrixType& mat, IndexVector& xprune, IndexVector& marker, GlobalLU_t& glu); - Index snode_bmod (const Index jcol, const Index fsupc, ScalarVector& dense, GlobalLU_t& glu); - Index pivotL(const Index jcol, const RealScalar& diagpivotthresh, IndexVector& perm_r, IndexVector& iperm_c, Index& pivrow, GlobalLU_t& glu); - template - void dfs_kernel(const StorageIndex jj, IndexVector& perm_r, - Index& nseg, IndexVector& panel_lsub, IndexVector& segrep, - Ref repfnz_col, IndexVector& xprune, Ref marker, IndexVector& parent, - IndexVector& xplore, GlobalLU_t& glu, Index& nextl_col, Index krow, Traits& traits); - void panel_dfs(const Index m, const Index w, const Index jcol, MatrixType& A, IndexVector& perm_r, Index& nseg, ScalarVector& dense, IndexVector& panel_lsub, IndexVector& segrep, IndexVector& repfnz, IndexVector& xprune, IndexVector& marker, IndexVector& parent, IndexVector& xplore, GlobalLU_t& glu); - - void panel_bmod(const Index m, const Index w, const Index jcol, const Index nseg, ScalarVector& dense, ScalarVector& tempv, IndexVector& segrep, IndexVector& repfnz, GlobalLU_t& glu); - Index column_dfs(const Index m, const Index jcol, IndexVector& perm_r, Index maxsuper, Index& nseg, BlockIndexVector lsub_col, IndexVector& segrep, BlockIndexVector repfnz, IndexVector& xprune, IndexVector& marker, IndexVector& parent, IndexVector& xplore, GlobalLU_t& glu); - Index column_bmod(const Index jcol, const Index nseg, BlockScalarVector dense, ScalarVector& tempv, BlockIndexVector segrep, BlockIndexVector repfnz, Index fpanelc, GlobalLU_t& glu); - Index copy_to_ucol(const Index jcol, const Index nseg, IndexVector& segrep, BlockIndexVector repfnz ,IndexVector& perm_r, BlockScalarVector dense, GlobalLU_t& glu); - void pruneL(const Index jcol, const IndexVector& perm_r, const Index pivrow, const Index nseg, const IndexVector& segrep, BlockIndexVector repfnz, IndexVector& xprune, GlobalLU_t& glu); - void countnz(const Index n, Index& nnzL, Index& nnzU, GlobalLU_t& glu); - void fixupL(const Index n, const IndexVector& perm_r, GlobalLU_t& glu); - - template - friend struct column_dfs_traits; -}; + template + Index expand(VectorType &vec, Index &length, Index nbElts, Index keep_prev, Index &num_expansions); + Index memInit(Index m, Index n, Index annz, Index lwork, Index fillratio, Index panel_size, GlobalLU_t &glu); + template + Index memXpand(VectorType &vec, Index &maxlen, Index nbElts, MemType memtype, Index &num_expansions); + void heap_relax_snode(const Index n, + IndexVector &et, + const Index relax_columns, + IndexVector &descendants, + IndexVector &relax_end); + void relax_snode(const Index n, + IndexVector &et, + const Index relax_columns, + IndexVector &descendants, + IndexVector &relax_end); + Index snode_dfs(const Index jcol, + const Index kcol, + const MatrixType &mat, + IndexVector &xprune, + IndexVector &marker, + GlobalLU_t &glu); + Index snode_bmod(const Index jcol, const Index fsupc, ScalarVector &dense, GlobalLU_t &glu); + Index pivotL(const Index jcol, + const RealScalar &diagpivotthresh, + IndexVector &perm_r, + IndexVector &iperm_c, + Index &pivrow, + GlobalLU_t &glu); + template + void dfs_kernel(const StorageIndex jj, + IndexVector &perm_r, + Index &nseg, + IndexVector &panel_lsub, + IndexVector &segrep, + Ref repfnz_col, + IndexVector &xprune, + Ref marker, + IndexVector &parent, + IndexVector &xplore, + GlobalLU_t &glu, + Index &nextl_col, + Index krow, + Traits &traits); + void panel_dfs(const Index m, + const Index w, + const Index jcol, + MatrixType &A, + IndexVector &perm_r, + Index &nseg, + ScalarVector &dense, + IndexVector &panel_lsub, + IndexVector &segrep, + IndexVector &repfnz, + IndexVector &xprune, + IndexVector &marker, + IndexVector &parent, + IndexVector &xplore, + GlobalLU_t &glu); + + void panel_bmod(const Index m, + const Index w, + const Index jcol, + const Index nseg, + ScalarVector &dense, + ScalarVector &tempv, + IndexVector &segrep, + IndexVector &repfnz, + GlobalLU_t &glu); + Index column_dfs(const Index m, + const Index jcol, + IndexVector &perm_r, + Index maxsuper, + Index &nseg, + BlockIndexVector lsub_col, + IndexVector &segrep, + BlockIndexVector repfnz, + IndexVector &xprune, + IndexVector &marker, + IndexVector &parent, + IndexVector &xplore, + GlobalLU_t &glu); + Index column_bmod(const Index jcol, + const Index nseg, + BlockScalarVector dense, + ScalarVector &tempv, + BlockIndexVector segrep, + BlockIndexVector repfnz, + Index fpanelc, + GlobalLU_t &glu); + Index copy_to_ucol(const Index jcol, + const Index nseg, + IndexVector &segrep, + BlockIndexVector repfnz, + IndexVector &perm_r, + BlockScalarVector dense, + GlobalLU_t &glu); + void pruneL(const Index jcol, + const IndexVector &perm_r, + const Index pivrow, + const Index nseg, + const IndexVector &segrep, + BlockIndexVector repfnz, + IndexVector &xprune, + GlobalLU_t &glu); + void countnz(const Index n, Index &nnzL, Index &nnzU, GlobalLU_t &glu); + void fixupL(const Index n, const IndexVector &perm_r, GlobalLU_t &glu); + + template friend struct column_dfs_traits; + }; -} // end namespace internal -} // namespace Eigen +}// end namespace internal +}// namespace Eigen #endif diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/SparseLU/SparseLU_Memory.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/SparseLU/SparseLU_Memory.h index 4dc42e87..97c6638e 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/SparseLU/SparseLU_Memory.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/SparseLU/SparseLU_Memory.h @@ -7,10 +7,10 @@ // Public License v. 2.0. If a copy of the MPL was not distributed // with this file, You can obtain one at http://mozilla.org/MPL/2.0/. -/* - - * NOTE: This file is the modified version of [s,d,c,z]memory.c files in SuperLU - +/* + + * NOTE: This file is the modified version of [s,d,c,z]memory.c files in SuperLU + * -- SuperLU routine (version 3.1) -- * Univ. of California Berkeley, Xerox Palo Alto Research Center, * and Lawrence Berkeley National Lab. @@ -33,194 +33,194 @@ namespace Eigen { namespace internal { - -enum { LUNoMarker = 3 }; -enum {emptyIdxLU = -1}; -inline Index LUnumTempV(Index& m, Index& w, Index& t, Index& b) -{ - return (std::max)(m, (t+b)*w); -} - -template< typename Scalar> -inline Index LUTempSpace(Index&m, Index& w) -{ - return (2*w + 4 + LUNoMarker) * m * sizeof(Index) + (w + 1) * m * sizeof(Scalar); -} - - - - -/** - * Expand the existing storage to accomodate more fill-ins - * \param vec Valid pointer to the vector to allocate or expand - * \param[in,out] length At input, contain the current length of the vector that is to be increased. At output, length of the newly allocated vector - * \param[in] nbElts Current number of elements in the factors - * \param keep_prev 1: use length and do not expand the vector; 0: compute new_len and expand - * \param[in,out] num_expansions Number of times the memory has been expanded - */ -template -template -Index SparseLUImpl::expand(VectorType& vec, Index& length, Index nbElts, Index keep_prev, Index& num_expansions) -{ - - float alpha = 1.5; // Ratio of the memory increase - Index new_len; // New size of the allocated memory - - if(num_expansions == 0 || keep_prev) - new_len = length ; // First time allocate requested - else - new_len = (std::max)(length+1,Index(alpha * length)); - - VectorType old_vec; // Temporary vector to hold the previous values - if (nbElts > 0 ) - old_vec = vec.segment(0,nbElts); - - //Allocate or expand the current vector -#ifdef EIGEN_EXCEPTIONS - try -#endif + + enum { LUNoMarker = 3 }; + enum { emptyIdxLU = -1 }; + inline Index LUnumTempV(Index &m, Index &w, Index &t, Index &b) { return (std::max)(m, (t + b) * w); } + + template inline Index LUTempSpace(Index &m, Index &w) { - vec.resize(new_len); + return (2 * w + 4 + LUNoMarker) * m * sizeof(Index) + (w + 1) * m * sizeof(Scalar); } + + + /** + * Expand the existing storage to accomodate more fill-ins + * \param vec Valid pointer to the vector to allocate or expand + * \param[in,out] length At input, contain the current length of the vector that is to be increased. At output, + * length of the newly allocated vector + * \param[in] nbElts Current number of elements in the factors + * \param keep_prev 1: use length and do not expand the vector; 0: compute new_len and expand + * \param[in,out] num_expansions Number of times the memory has been expanded + */ + template + template + Index SparseLUImpl::expand(VectorType &vec, + Index &length, + Index nbElts, + Index keep_prev, + Index &num_expansions) + { + + float alpha = 1.5;// Ratio of the memory increase + Index new_len;// New size of the allocated memory + + if (num_expansions == 0 || keep_prev) + new_len = length;// First time allocate requested + else + new_len = (std::max)(length + 1, Index(alpha * length)); + + VectorType old_vec;// Temporary vector to hold the previous values + if (nbElts > 0) old_vec = vec.segment(0, nbElts); + + // Allocate or expand the current vector #ifdef EIGEN_EXCEPTIONS - catch(std::bad_alloc& ) -#else - if(!vec.size()) + try #endif - { - if (!num_expansions) { - // First time to allocate from LUMemInit() - // Let LUMemInit() deals with it. - return -1; + vec.resize(new_len); } - if (keep_prev) - { - // In this case, the memory length should not not be reduced - return new_len; - } - else +#ifdef EIGEN_EXCEPTIONS + catch (std::bad_alloc &) +#else + if (!vec.size()) +#endif { - // Reduce the size and increase again - Index tries = 0; // Number of attempts - do - { - alpha = (alpha + 1)/2; - new_len = (std::max)(length+1,Index(alpha * length)); + if (!num_expansions) { + // First time to allocate from LUMemInit() + // Let LUMemInit() deals with it. + return -1; + } + if (keep_prev) { + // In this case, the memory length should not not be reduced + return new_len; + } else { + // Reduce the size and increase again + Index tries = 0;// Number of attempts + do { + alpha = (alpha + 1) / 2; + new_len = (std::max)(length + 1, Index(alpha * length)); #ifdef EIGEN_EXCEPTIONS - try + try #endif - { - vec.resize(new_len); - } + { + vec.resize(new_len); + } #ifdef EIGEN_EXCEPTIONS - catch(std::bad_alloc& ) + catch (std::bad_alloc &) #else - if (!vec.size()) + if (!vec.size()) #endif - { - tries += 1; - if ( tries > 10) return new_len; - } - } while (!vec.size()); + { + tries += 1; + if (tries > 10) return new_len; + } + } while (!vec.size()); + } } + // Copy the previous values to the newly allocated space + if (nbElts > 0) vec.segment(0, nbElts) = old_vec; + + + length = new_len; + if (num_expansions) ++num_expansions; + return 0; } - //Copy the previous values to the newly allocated space - if (nbElts > 0) - vec.segment(0, nbElts) = old_vec; - - - length = new_len; - if(num_expansions) ++num_expansions; - return 0; -} - -/** - * \brief Allocate various working space for the numerical factorization phase. - * \param m number of rows of the input matrix - * \param n number of columns - * \param annz number of initial nonzeros in the matrix - * \param lwork if lwork=-1, this routine returns an estimated size of the required memory - * \param glu persistent data to facilitate multiple factors : will be deleted later ?? - * \param fillratio estimated ratio of fill in the factors - * \param panel_size Size of a panel - * \return an estimated size of the required memory if lwork = -1; otherwise, return the size of actually allocated memory when allocation failed, and 0 on success - * \note Unlike SuperLU, this routine does not support successive factorization with the same pattern and the same row permutation - */ -template -Index SparseLUImpl::memInit(Index m, Index n, Index annz, Index lwork, Index fillratio, Index panel_size, GlobalLU_t& glu) -{ - Index& num_expansions = glu.num_expansions; //No memory expansions so far - num_expansions = 0; - glu.nzumax = glu.nzlumax = (std::min)(fillratio * (annz+1) / n, m) * n; // estimated number of nonzeros in U - glu.nzlmax = (std::max)(Index(4), fillratio) * (annz+1) / 4; // estimated nnz in L factor - // Return the estimated size to the user if necessary - Index tempSpace; - tempSpace = (2*panel_size + 4 + LUNoMarker) * m * sizeof(Index) + (panel_size + 1) * m * sizeof(Scalar); - if (lwork == emptyIdxLU) - { - Index estimated_size; - estimated_size = (5 * n + 5) * sizeof(Index) + tempSpace - + (glu.nzlmax + glu.nzumax) * sizeof(Index) + (glu.nzlumax+glu.nzumax) * sizeof(Scalar) + n; - return estimated_size; - } - - // Setup the required space - - // First allocate Integer pointers for L\U factors - glu.xsup.resize(n+1); - glu.supno.resize(n+1); - glu.xlsub.resize(n+1); - glu.xlusup.resize(n+1); - glu.xusub.resize(n+1); - - // Reserve memory for L/U factors - do + + /** + * \brief Allocate various working space for the numerical factorization phase. + * \param m number of rows of the input matrix + * \param n number of columns + * \param annz number of initial nonzeros in the matrix + * \param lwork if lwork=-1, this routine returns an estimated size of the required memory + * \param glu persistent data to facilitate multiple factors : will be deleted later ?? + * \param fillratio estimated ratio of fill in the factors + * \param panel_size Size of a panel + * \return an estimated size of the required memory if lwork = -1; otherwise, return the size of actually allocated + * memory when allocation failed, and 0 on success + * \note Unlike SuperLU, this routine does not support successive factorization with the same pattern and the same row + * permutation + */ + template + Index SparseLUImpl::memInit(Index m, + Index n, + Index annz, + Index lwork, + Index fillratio, + Index panel_size, + GlobalLU_t &glu) { - if( (expand(glu.lusup, glu.nzlumax, 0, 0, num_expansions)<0) - || (expand(glu.ucol, glu.nzumax, 0, 0, num_expansions)<0) - || (expand (glu.lsub, glu.nzlmax, 0, 0, num_expansions)<0) - || (expand (glu.usub, glu.nzumax, 0, 1, num_expansions)<0) ) - { - //Reduce the estimated size and retry - glu.nzlumax /= 2; - glu.nzumax /= 2; - glu.nzlmax /= 2; - if (glu.nzlumax < annz ) return glu.nzlumax; + Index &num_expansions = glu.num_expansions;// No memory expansions so far + num_expansions = 0; + glu.nzumax = glu.nzlumax = (std::min)(fillratio * (annz + 1) / n, m) * n;// estimated number of nonzeros in U + glu.nzlmax = (std::max)(Index(4), fillratio) * (annz + 1) / 4;// estimated nnz in L factor + // Return the estimated size to the user if necessary + Index tempSpace; + tempSpace = (2 * panel_size + 4 + LUNoMarker) * m * sizeof(Index) + (panel_size + 1) * m * sizeof(Scalar); + if (lwork == emptyIdxLU) { + Index estimated_size; + estimated_size = (5 * n + 5) * sizeof(Index) + tempSpace + (glu.nzlmax + glu.nzumax) * sizeof(Index) + + (glu.nzlumax + glu.nzumax) * sizeof(Scalar) + n; + return estimated_size; } - } while (!glu.lusup.size() || !glu.ucol.size() || !glu.lsub.size() || !glu.usub.size()); - - ++num_expansions; - return 0; - -} // end LuMemInit - -/** - * \brief Expand the existing storage - * \param vec vector to expand - * \param[in,out] maxlen On input, previous size of vec (Number of elements to copy ). on output, new size - * \param nbElts current number of elements in the vector. - * \param memtype Type of the element to expand - * \param num_expansions Number of expansions - * \return 0 on success, > 0 size of the memory allocated so far - */ -template -template -Index SparseLUImpl::memXpand(VectorType& vec, Index& maxlen, Index nbElts, MemType memtype, Index& num_expansions) -{ - Index failed_size; - if (memtype == USUB) - failed_size = this->expand(vec, maxlen, nbElts, 1, num_expansions); - else - failed_size = this->expand(vec, maxlen, nbElts, 0, num_expansions); - - if (failed_size) - return failed_size; - - return 0 ; -} - -} // end namespace internal - -} // end namespace Eigen -#endif // EIGEN_SPARSELU_MEMORY + + // Setup the required space + + // First allocate Integer pointers for L\U factors + glu.xsup.resize(n + 1); + glu.supno.resize(n + 1); + glu.xlsub.resize(n + 1); + glu.xlusup.resize(n + 1); + glu.xusub.resize(n + 1); + + // Reserve memory for L/U factors + do { + if ((expand(glu.lusup, glu.nzlumax, 0, 0, num_expansions) < 0) + || (expand(glu.ucol, glu.nzumax, 0, 0, num_expansions) < 0) + || (expand(glu.lsub, glu.nzlmax, 0, 0, num_expansions) < 0) + || (expand(glu.usub, glu.nzumax, 0, 1, num_expansions) < 0)) { + // Reduce the estimated size and retry + glu.nzlumax /= 2; + glu.nzumax /= 2; + glu.nzlmax /= 2; + if (glu.nzlumax < annz) return glu.nzlumax; + } + } while (!glu.lusup.size() || !glu.ucol.size() || !glu.lsub.size() || !glu.usub.size()); + + ++num_expansions; + return 0; + + }// end LuMemInit + + /** + * \brief Expand the existing storage + * \param vec vector to expand + * \param[in,out] maxlen On input, previous size of vec (Number of elements to copy ). on output, new size + * \param nbElts current number of elements in the vector. + * \param memtype Type of the element to expand + * \param num_expansions Number of expansions + * \return 0 on success, > 0 size of the memory allocated so far + */ + template + template + Index SparseLUImpl::memXpand(VectorType &vec, + Index &maxlen, + Index nbElts, + MemType memtype, + Index &num_expansions) + { + Index failed_size; + if (memtype == USUB) + failed_size = this->expand(vec, maxlen, nbElts, 1, num_expansions); + else + failed_size = this->expand(vec, maxlen, nbElts, 0, num_expansions); + + if (failed_size) return failed_size; + + return 0; + } + +}// end namespace internal + +}// end namespace Eigen +#endif// EIGEN_SPARSELU_MEMORY diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/SparseLU/SparseLU_Structs.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/SparseLU/SparseLU_Structs.h index cf5ec449..4ec9b2a7 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/SparseLU/SparseLU_Structs.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/SparseLU/SparseLU_Structs.h @@ -7,26 +7,26 @@ // Public License v. 2.0. If a copy of the MPL was not distributed // with this file, You can obtain one at http://mozilla.org/MPL/2.0/. -/* +/* * NOTE: This file comes from a partly modified version of files slu_[s,d,c,z]defs.h * -- SuperLU routine (version 4.1) -- * Univ. of California Berkeley, Xerox Palo Alto Research Center, * and Lawrence Berkeley National Lab. * November, 2010 - * + * * Global data structures used in LU factorization - - * + * * nsuper: #supernodes = nsuper + 1, numbered [0, nsuper]. * (xsup,supno): supno[i] is the supernode no to which i belongs; * xsup(s) points to the beginning of the s-th supernode. * e.g. supno 0 1 2 2 3 3 3 4 4 4 4 4 (n=12) * xsup 0 1 2 4 7 12 - * Note: dfs will be performed on supernode rep. relative to the new + * Note: dfs will be performed on supernode rep. relative to the new * row pivoting ordering * * (xlsub,lsub): lsub[*] contains the compressed subscript of * rectangular supernodes; xlsub[j] points to the starting - * location of the j-th column in lsub[*]. Note that xlsub + * location of the j-th column in lsub[*]. Note that xlsub * is indexed by column. * Storage: original row subscripts * @@ -70,41 +70,42 @@ #define EIGEN_LU_STRUCTS namespace Eigen { namespace internal { - -typedef enum {LUSUP, UCOL, LSUB, USUB, LLVL, ULVL} MemType; -template -struct LU_GlobalLU_t { - typedef typename IndexVector::Scalar StorageIndex; - IndexVector xsup; //First supernode column ... xsup(s) points to the beginning of the s-th supernode - IndexVector supno; // Supernode number corresponding to this column (column to supernode mapping) - ScalarVector lusup; // nonzero values of L ordered by columns - IndexVector lsub; // Compressed row indices of L rectangular supernodes. - IndexVector xlusup; // pointers to the beginning of each column in lusup - IndexVector xlsub; // pointers to the beginning of each column in lsub - Index nzlmax; // Current max size of lsub - Index nzlumax; // Current max size of lusup - ScalarVector ucol; // nonzero values of U ordered by columns - IndexVector usub; // row indices of U columns in ucol - IndexVector xusub; // Pointers to the beginning of each column of U in ucol - Index nzumax; // Current max size of ucol - Index n; // Number of columns in the matrix - Index num_expansions; -}; + typedef enum { LUSUP, UCOL, LSUB, USUB, LLVL, ULVL } MemType; + + template struct LU_GlobalLU_t + { + typedef typename IndexVector::Scalar StorageIndex; + IndexVector xsup;// First supernode column ... xsup(s) points to the beginning of the s-th supernode + IndexVector supno;// Supernode number corresponding to this column (column to supernode mapping) + ScalarVector lusup;// nonzero values of L ordered by columns + IndexVector lsub;// Compressed row indices of L rectangular supernodes. + IndexVector xlusup;// pointers to the beginning of each column in lusup + IndexVector xlsub;// pointers to the beginning of each column in lsub + Index nzlmax;// Current max size of lsub + Index nzlumax;// Current max size of lusup + ScalarVector ucol;// nonzero values of U ordered by columns + IndexVector usub;// row indices of U columns in ucol + IndexVector xusub;// Pointers to the beginning of each column of U in ucol + Index nzumax;// Current max size of ucol + Index n;// Number of columns in the matrix + Index num_expansions; + }; -// Values to set for performance -struct perfvalues { - Index panel_size; // a panel consists of at most consecutive columns - Index relax; // To control degree of relaxing supernodes. If the number of nodes (columns) - // in a subtree of the elimination tree is less than relax, this subtree is considered + // Values to set for performance + struct perfvalues + { + Index panel_size;// a panel consists of at most consecutive columns + Index relax;// To control degree of relaxing supernodes. If the number of nodes (columns) + // in a subtree of the elimination tree is less than relax, this subtree is considered // as one supernode regardless of the row structures of those columns - Index maxsuper; // The maximum size for a supernode in complete LU - Index rowblk; // The minimum row dimension for 2-D blocking to be used; - Index colblk; // The minimum column dimension for 2-D blocking to be used; - Index fillfactor; // The estimated fills factors for L and U, compared with A -}; + Index maxsuper;// The maximum size for a supernode in complete LU + Index rowblk;// The minimum row dimension for 2-D blocking to be used; + Index colblk;// The minimum column dimension for 2-D blocking to be used; + Index fillfactor;// The estimated fills factors for L and U, compared with A + }; -} // end namespace internal +}// end namespace internal -} // end namespace Eigen -#endif // EIGEN_LU_STRUCTS +}// end namespace Eigen +#endif// EIGEN_LU_STRUCTS diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/SparseLU/SparseLU_SupernodalMatrix.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/SparseLU/SparseLU_SupernodalMatrix.h index 721e1883..068a8f58 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/SparseLU/SparseLU_SupernodalMatrix.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/SparseLU/SparseLU_SupernodalMatrix.h @@ -14,288 +14,256 @@ namespace Eigen { namespace internal { -/** \ingroup SparseLU_Module - * \brief a class to manipulate the L supernodal factor from the SparseLU factorization - * - * This class contain the data to easily store - * and manipulate the supernodes during the factorization and solution phase of Sparse LU. - * Only the lower triangular matrix has supernodes. - * - * NOTE : This class corresponds to the SCformat structure in SuperLU - * - */ -/* TODO - * InnerIterator as for sparsematrix - * SuperInnerIterator to iterate through all supernodes - * Function for triangular solve - */ -template -class MappedSuperNodalMatrix -{ + /** \ingroup SparseLU_Module + * \brief a class to manipulate the L supernodal factor from the SparseLU factorization + * + * This class contain the data to easily store + * and manipulate the supernodes during the factorization and solution phase of Sparse LU. + * Only the lower triangular matrix has supernodes. + * + * NOTE : This class corresponds to the SCformat structure in SuperLU + * + */ + /* TODO + * InnerIterator as for sparsematrix + * SuperInnerIterator to iterate through all supernodes + * Function for triangular solve + */ + template class MappedSuperNodalMatrix + { public: - typedef _Scalar Scalar; + typedef _Scalar Scalar; typedef _StorageIndex StorageIndex; - typedef Matrix IndexVector; - typedef Matrix ScalarVector; + typedef Matrix IndexVector; + typedef Matrix ScalarVector; + public: - MappedSuperNodalMatrix() - { - - } - MappedSuperNodalMatrix(Index m, Index n, ScalarVector& nzval, IndexVector& nzval_colptr, IndexVector& rowind, - IndexVector& rowind_colptr, IndexVector& col_to_sup, IndexVector& sup_to_col ) + MappedSuperNodalMatrix() {} + MappedSuperNodalMatrix(Index m, + Index n, + ScalarVector &nzval, + IndexVector &nzval_colptr, + IndexVector &rowind, + IndexVector &rowind_colptr, + IndexVector &col_to_sup, + IndexVector &sup_to_col) { setInfos(m, n, nzval, nzval_colptr, rowind, rowind_colptr, col_to_sup, sup_to_col); } - - ~MappedSuperNodalMatrix() - { - - } + + ~MappedSuperNodalMatrix() {} /** * Set appropriate pointers for the lower triangular supernodal matrix * These infos are available at the end of the numerical factorization - * FIXME This class will be modified such that it can be use in the course + * FIXME This class will be modified such that it can be use in the course * of the factorization. */ - void setInfos(Index m, Index n, ScalarVector& nzval, IndexVector& nzval_colptr, IndexVector& rowind, - IndexVector& rowind_colptr, IndexVector& col_to_sup, IndexVector& sup_to_col ) + void setInfos(Index m, + Index n, + ScalarVector &nzval, + IndexVector &nzval_colptr, + IndexVector &rowind, + IndexVector &rowind_colptr, + IndexVector &col_to_sup, + IndexVector &sup_to_col) { m_row = m; - m_col = n; - m_nzval = nzval.data(); - m_nzval_colptr = nzval_colptr.data(); - m_rowind = rowind.data(); - m_rowind_colptr = rowind_colptr.data(); - m_nsuper = col_to_sup(n); - m_col_to_sup = col_to_sup.data(); - m_sup_to_col = sup_to_col.data(); + m_col = n; + m_nzval = nzval.data(); + m_nzval_colptr = nzval_colptr.data(); + m_rowind = rowind.data(); + m_rowind_colptr = rowind_colptr.data(); + m_nsuper = col_to_sup(n); + m_col_to_sup = col_to_sup.data(); + m_sup_to_col = sup_to_col.data(); } - + /** * Number of rows */ Index rows() { return m_row; } - + /** * Number of columns */ Index cols() { return m_col; } - + /** * Return the array of nonzero values packed by column - * + * * The size is nnz */ - Scalar* valuePtr() { return m_nzval; } - - const Scalar* valuePtr() const - { - return m_nzval; - } + Scalar *valuePtr() { return m_nzval; } + + const Scalar *valuePtr() const { return m_nzval; } /** * Return the pointers to the beginning of each column in \ref valuePtr() */ - StorageIndex* colIndexPtr() - { - return m_nzval_colptr; - } - - const StorageIndex* colIndexPtr() const - { - return m_nzval_colptr; - } - + StorageIndex *colIndexPtr() { return m_nzval_colptr; } + + const StorageIndex *colIndexPtr() const { return m_nzval_colptr; } + /** * Return the array of compressed row indices of all supernodes */ - StorageIndex* rowIndex() { return m_rowind; } - - const StorageIndex* rowIndex() const - { - return m_rowind; - } - + StorageIndex *rowIndex() { return m_rowind; } + + const StorageIndex *rowIndex() const { return m_rowind; } + /** * Return the location in \em rowvaluePtr() which starts each column */ - StorageIndex* rowIndexPtr() { return m_rowind_colptr; } - - const StorageIndex* rowIndexPtr() const - { - return m_rowind_colptr; - } - - /** - * Return the array of column-to-supernode mapping + StorageIndex *rowIndexPtr() { return m_rowind_colptr; } + + const StorageIndex *rowIndexPtr() const { return m_rowind_colptr; } + + /** + * Return the array of column-to-supernode mapping */ - StorageIndex* colToSup() { return m_col_to_sup; } - - const StorageIndex* colToSup() const - { - return m_col_to_sup; - } + StorageIndex *colToSup() { return m_col_to_sup; } + + const StorageIndex *colToSup() const { return m_col_to_sup; } /** * Return the array of supernode-to-column mapping */ - StorageIndex* supToCol() { return m_sup_to_col; } - - const StorageIndex* supToCol() const - { - return m_sup_to_col; - } - + StorageIndex *supToCol() { return m_sup_to_col; } + + const StorageIndex *supToCol() const { return m_sup_to_col; } + /** * Return the number of supernodes */ - Index nsuper() const - { - return m_nsuper; - } - - class InnerIterator; - template - void solveInPlace( MatrixBase&X) const; - - - - + Index nsuper() const { return m_nsuper; } + + class InnerIterator; + template void solveInPlace(MatrixBase &X) const; + + protected: - Index m_row; // Number of rows - Index m_col; // Number of columns - Index m_nsuper; // Number of supernodes - Scalar* m_nzval; //array of nonzero values packed by column - StorageIndex* m_nzval_colptr; //nzval_colptr[j] Stores the location in nzval[] which starts column j - StorageIndex* m_rowind; // Array of compressed row indices of rectangular supernodes - StorageIndex* m_rowind_colptr; //rowind_colptr[j] stores the location in rowind[] which starts column j - StorageIndex* m_col_to_sup; // col_to_sup[j] is the supernode number to which column j belongs - StorageIndex* m_sup_to_col; //sup_to_col[s] points to the starting column of the s-th supernode - - private : -}; - -/** - * \brief InnerIterator class to iterate over nonzero values of the current column in the supernodal matrix L - * - */ -template -class MappedSuperNodalMatrix::InnerIterator -{ + Index m_row;// Number of rows + Index m_col;// Number of columns + Index m_nsuper;// Number of supernodes + Scalar *m_nzval;// array of nonzero values packed by column + StorageIndex *m_nzval_colptr;// nzval_colptr[j] Stores the location in nzval[] which starts column j + StorageIndex *m_rowind;// Array of compressed row indices of rectangular supernodes + StorageIndex *m_rowind_colptr;// rowind_colptr[j] stores the location in rowind[] which starts column j + StorageIndex *m_col_to_sup;// col_to_sup[j] is the supernode number to which column j belongs + StorageIndex *m_sup_to_col;// sup_to_col[s] points to the starting column of the s-th supernode + + private: + }; + + /** + * \brief InnerIterator class to iterate over nonzero values of the current column in the supernodal matrix L + * + */ + template class MappedSuperNodalMatrix::InnerIterator + { public: - InnerIterator(const MappedSuperNodalMatrix& mat, Index outer) - : m_matrix(mat), - m_outer(outer), - m_supno(mat.colToSup()[outer]), - m_idval(mat.colIndexPtr()[outer]), - m_startidval(m_idval), - m_endidval(mat.colIndexPtr()[outer+1]), + InnerIterator(const MappedSuperNodalMatrix &mat, Index outer) + : m_matrix(mat), m_outer(outer), m_supno(mat.colToSup()[outer]), m_idval(mat.colIndexPtr()[outer]), + m_startidval(m_idval), m_endidval(mat.colIndexPtr()[outer + 1]), m_idrow(mat.rowIndexPtr()[mat.supToCol()[mat.colToSup()[outer]]]), - m_endidrow(mat.rowIndexPtr()[mat.supToCol()[mat.colToSup()[outer]]+1]) + m_endidrow(mat.rowIndexPtr()[mat.supToCol()[mat.colToSup()[outer]] + 1]) {} - inline InnerIterator& operator++() - { - m_idval++; + inline InnerIterator &operator++() + { + m_idval++; m_idrow++; return *this; } inline Scalar value() const { return m_matrix.valuePtr()[m_idval]; } - - inline Scalar& valueRef() { return const_cast(m_matrix.valuePtr()[m_idval]); } - + + inline Scalar &valueRef() { return const_cast(m_matrix.valuePtr()[m_idval]); } + inline Index index() const { return m_matrix.rowIndex()[m_idrow]; } inline Index row() const { return index(); } inline Index col() const { return m_outer; } - + inline Index supIndex() const { return m_supno; } - - inline operator bool() const - { - return ( (m_idval < m_endidval) && (m_idval >= m_startidval) - && (m_idrow < m_endidrow) ); + + inline operator bool() const + { + return ((m_idval < m_endidval) && (m_idval >= m_startidval) && (m_idrow < m_endidrow)); } - + protected: - const MappedSuperNodalMatrix& m_matrix; // Supernodal lower triangular matrix - const Index m_outer; // Current column - const Index m_supno; // Current SuperNode number - Index m_idval; // Index to browse the values in the current column - const Index m_startidval; // Start of the column value - const Index m_endidval; // End of the column value - Index m_idrow; // Index to browse the row indices - Index m_endidrow; // End index of row indices of the current column -}; - -/** - * \brief Solve with the supernode triangular matrix - * - */ -template -template -void MappedSuperNodalMatrix::solveInPlace( MatrixBase&X) const -{ + const MappedSuperNodalMatrix &m_matrix;// Supernodal lower triangular matrix + const Index m_outer;// Current column + const Index m_supno;// Current SuperNode number + Index m_idval;// Index to browse the values in the current column + const Index m_startidval;// Start of the column value + const Index m_endidval;// End of the column value + Index m_idrow;// Index to browse the row indices + Index m_endidrow;// End index of row indices of the current column + }; + + /** + * \brief Solve with the supernode triangular matrix + * + */ + template + template + void MappedSuperNodalMatrix::solveInPlace(MatrixBase &X) const + { /* Explicit type conversion as the Index type of MatrixBase may be wider than Index */ -// eigen_assert(X.rows() <= NumTraits::highest()); -// eigen_assert(X.cols() <= NumTraits::highest()); - Index n = int(X.rows()); + // eigen_assert(X.rows() <= NumTraits::highest()); + // eigen_assert(X.cols() <= NumTraits::highest()); + Index n = int(X.rows()); Index nrhs = Index(X.cols()); - const Scalar * Lval = valuePtr(); // Nonzero values - Matrix work(n, nrhs); // working vector + const Scalar *Lval = valuePtr();// Nonzero values + Matrix work(n, nrhs);// working vector work.setZero(); - for (Index k = 0; k <= nsuper(); k ++) - { - Index fsupc = supToCol()[k]; // First column of the current supernode - Index istart = rowIndexPtr()[fsupc]; // Pointer index to the subscript of the current column - Index nsupr = rowIndexPtr()[fsupc+1] - istart; // Number of rows in the current supernode - Index nsupc = supToCol()[k+1] - fsupc; // Number of columns in the current supernode - Index nrow = nsupr - nsupc; // Number of rows in the non-diagonal part of the supernode - Index irow; //Current index row - - if (nsupc == 1 ) - { - for (Index j = 0; j < nrhs; j++) - { + for (Index k = 0; k <= nsuper(); k++) { + Index fsupc = supToCol()[k];// First column of the current supernode + Index istart = rowIndexPtr()[fsupc];// Pointer index to the subscript of the current column + Index nsupr = rowIndexPtr()[fsupc + 1] - istart;// Number of rows in the current supernode + Index nsupc = supToCol()[k + 1] - fsupc;// Number of columns in the current supernode + Index nrow = nsupr - nsupc;// Number of rows in the non-diagonal part of the supernode + Index irow;// Current index row + + if (nsupc == 1) { + for (Index j = 0; j < nrhs; j++) { InnerIterator it(*this, fsupc); - ++it; // Skip the diagonal element - for (; it; ++it) - { + ++it;// Skip the diagonal element + for (; it; ++it) { irow = it.row(); X(irow, j) -= X(fsupc, j) * it.value(); } } - } - else - { - // The supernode has more than one column - Index luptr = colIndexPtr()[fsupc]; - Index lda = colIndexPtr()[fsupc+1] - luptr; - - // Triangular solve - Map, 0, OuterStride<> > A( &(Lval[luptr]), nsupc, nsupc, OuterStride<>(lda) ); - Map< Matrix, 0, OuterStride<> > U (&(X(fsupc,0)), nsupc, nrhs, OuterStride<>(n) ); - U = A.template triangularView().solve(U); - - // Matrix-vector product - new (&A) Map, 0, OuterStride<> > ( &(Lval[luptr+nsupc]), nrow, nsupc, OuterStride<>(lda) ); + } else { + // The supernode has more than one column + Index luptr = colIndexPtr()[fsupc]; + Index lda = colIndexPtr()[fsupc + 1] - luptr; + + // Triangular solve + Map, 0, OuterStride<>> A( + &(Lval[luptr]), nsupc, nsupc, OuterStride<>(lda)); + Map, 0, OuterStride<>> U( + &(X(fsupc, 0)), nsupc, nrhs, OuterStride<>(n)); + U = A.template triangularView().solve(U); + + // Matrix-vector product + new (&A) Map, 0, OuterStride<>>( + &(Lval[luptr + nsupc]), nrow, nsupc, OuterStride<>(lda)); work.topRows(nrow).noalias() = A * U; - - //Begin Scatter - for (Index j = 0; j < nrhs; j++) - { - Index iptr = istart + nsupc; - for (Index i = 0; i < nrow; i++) - { - irow = rowIndex()[iptr]; - X(irow, j) -= work(i, j); // Scatter operation - work(i, j) = Scalar(0); + + // Begin Scatter + for (Index j = 0; j < nrhs; j++) { + Index iptr = istart + nsupc; + for (Index i = 0; i < nrow; i++) { + irow = rowIndex()[iptr]; + X(irow, j) -= work(i, j);// Scatter operation + work(i, j) = Scalar(0); iptr++; } } } - } -} + } + } -} // end namespace internal +}// end namespace internal -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_SPARSELU_MATRIX_H +#endif// EIGEN_SPARSELU_MATRIX_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/SparseLU/SparseLU_Utils.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/SparseLU/SparseLU_Utils.h index 9e3dab44..0eec2ae8 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/SparseLU/SparseLU_Utils.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/SparseLU/SparseLU_Utils.h @@ -14,67 +14,62 @@ namespace Eigen { namespace internal { -/** - * \brief Count Nonzero elements in the factors - */ -template -void SparseLUImpl::countnz(const Index n, Index& nnzL, Index& nnzU, GlobalLU_t& glu) -{ - nnzL = 0; - nnzU = (glu.xusub)(n); - Index nsuper = (glu.supno)(n); - Index jlen; - Index i, j, fsupc; - if (n <= 0 ) return; - // For each supernode - for (i = 0; i <= nsuper; i++) - { - fsupc = glu.xsup(i); - jlen = glu.xlsub(fsupc+1) - glu.xlsub(fsupc); - - for (j = fsupc; j < glu.xsup(i+1); j++) - { - nnzL += jlen; - nnzU += j - fsupc + 1; - jlen--; - } - } -} + /** + * \brief Count Nonzero elements in the factors + */ + template + void SparseLUImpl::countnz(const Index n, Index &nnzL, Index &nnzU, GlobalLU_t &glu) + { + nnzL = 0; + nnzU = (glu.xusub)(n); + Index nsuper = (glu.supno)(n); + Index jlen; + Index i, j, fsupc; + if (n <= 0) return; + // For each supernode + for (i = 0; i <= nsuper; i++) { + fsupc = glu.xsup(i); + jlen = glu.xlsub(fsupc + 1) - glu.xlsub(fsupc); + + for (j = fsupc; j < glu.xsup(i + 1); j++) { + nnzL += jlen; + nnzU += j - fsupc + 1; + jlen--; + } + } + } -/** - * \brief Fix up the data storage lsub for L-subscripts. - * - * It removes the subscripts sets for structural pruning, - * and applies permutation to the remaining subscripts - * - */ -template -void SparseLUImpl::fixupL(const Index n, const IndexVector& perm_r, GlobalLU_t& glu) -{ - Index fsupc, i, j, k, jstart; - - StorageIndex nextl = 0; - Index nsuper = (glu.supno)(n); - - // For each supernode - for (i = 0; i <= nsuper; i++) + /** + * \brief Fix up the data storage lsub for L-subscripts. + * + * It removes the subscripts sets for structural pruning, + * and applies permutation to the remaining subscripts + * + */ + template + void SparseLUImpl::fixupL(const Index n, const IndexVector &perm_r, GlobalLU_t &glu) { - fsupc = glu.xsup(i); - jstart = glu.xlsub(fsupc); - glu.xlsub(fsupc) = nextl; - for (j = jstart; j < glu.xlsub(fsupc + 1); j++) - { - glu.lsub(nextl) = perm_r(glu.lsub(j)); // Now indexed into P*A - nextl++; + Index fsupc, i, j, k, jstart; + + StorageIndex nextl = 0; + Index nsuper = (glu.supno)(n); + + // For each supernode + for (i = 0; i <= nsuper; i++) { + fsupc = glu.xsup(i); + jstart = glu.xlsub(fsupc); + glu.xlsub(fsupc) = nextl; + for (j = jstart; j < glu.xlsub(fsupc + 1); j++) { + glu.lsub(nextl) = perm_r(glu.lsub(j));// Now indexed into P*A + nextl++; + } + for (k = fsupc + 1; k < glu.xsup(i + 1); k++) glu.xlsub(k) = nextl;// other columns in supernode i } - for (k = fsupc+1; k < glu.xsup(i+1); k++) - glu.xlsub(k) = nextl; // other columns in supernode i + + glu.xlsub(n) = nextl; } - - glu.xlsub(n) = nextl; -} -} // end namespace internal +}// end namespace internal -} // end namespace Eigen -#endif // EIGEN_SPARSELU_UTILS_H +}// end namespace Eigen +#endif// EIGEN_SPARSELU_UTILS_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/SparseLU/SparseLU_column_bmod.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/SparseLU/SparseLU_column_bmod.h index b57f0680..5a0434e7 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/SparseLU/SparseLU_column_bmod.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/SparseLU/SparseLU_column_bmod.h @@ -8,10 +8,10 @@ // Public License v. 2.0. If a copy of the MPL was not distributed // with this file, You can obtain one at http://mozilla.org/MPL/2.0/. -/* - - * NOTE: This file is the modified version of xcolumn_bmod.c file in SuperLU - +/* + + * NOTE: This file is the modified version of xcolumn_bmod.c file in SuperLU + * -- SuperLU routine (version 3.0) -- * Univ. of California Berkeley, Xerox Palo Alto Research Center, * and Lawrence Berkeley National Lab. @@ -34,148 +34,148 @@ namespace Eigen { namespace internal { -/** - * \brief Performs numeric block updates (sup-col) in topological order - * - * \param jcol current column to update - * \param nseg Number of segments in the U part - * \param dense Store the full representation of the column - * \param tempv working array - * \param segrep segment representative ... - * \param repfnz ??? First nonzero column in each row ??? ... - * \param fpanelc First column in the current panel - * \param glu Global LU data. - * \return 0 - successful return - * > 0 - number of bytes allocated when run out of space - * - */ -template -Index SparseLUImpl::column_bmod(const Index jcol, const Index nseg, BlockScalarVector dense, ScalarVector& tempv, - BlockIndexVector segrep, BlockIndexVector repfnz, Index fpanelc, GlobalLU_t& glu) -{ - Index jsupno, k, ksub, krep, ksupno; - Index lptr, nrow, isub, irow, nextlu, new_next, ufirst; - Index fsupc, nsupc, nsupr, luptr, kfnz, no_zeros; - /* krep = representative of current k-th supernode - * fsupc = first supernodal column - * nsupc = number of columns in a supernode - * nsupr = number of rows in a supernode - * luptr = location of supernodal LU-block in storage - * kfnz = first nonz in the k-th supernodal segment - * no_zeros = no lf leading zeros in a supernodal U-segment - */ - - jsupno = glu.supno(jcol); - // For each nonzero supernode segment of U[*,j] in topological order - k = nseg - 1; - Index d_fsupc; // distance between the first column of the current panel and the - // first column of the current snode - Index fst_col; // First column within small LU update - Index segsize; - for (ksub = 0; ksub < nseg; ksub++) + /** + * \brief Performs numeric block updates (sup-col) in topological order + * + * \param jcol current column to update + * \param nseg Number of segments in the U part + * \param dense Store the full representation of the column + * \param tempv working array + * \param segrep segment representative ... + * \param repfnz ??? First nonzero column in each row ??? ... + * \param fpanelc First column in the current panel + * \param glu Global LU data. + * \return 0 - successful return + * > 0 - number of bytes allocated when run out of space + * + */ + template + Index SparseLUImpl::column_bmod(const Index jcol, + const Index nseg, + BlockScalarVector dense, + ScalarVector &tempv, + BlockIndexVector segrep, + BlockIndexVector repfnz, + Index fpanelc, + GlobalLU_t &glu) { - krep = segrep(k); k--; - ksupno = glu.supno(krep); - if (jsupno != ksupno ) - { - // outside the rectangular supernode - fsupc = glu.xsup(ksupno); - fst_col = (std::max)(fsupc, fpanelc); - - // Distance from the current supernode to the current panel; - // d_fsupc = 0 if fsupc > fpanelc - d_fsupc = fst_col - fsupc; - - luptr = glu.xlusup(fst_col) + d_fsupc; - lptr = glu.xlsub(fsupc) + d_fsupc; - - kfnz = repfnz(krep); - kfnz = (std::max)(kfnz, fpanelc); - - segsize = krep - kfnz + 1; - nsupc = krep - fst_col + 1; - nsupr = glu.xlsub(fsupc+1) - glu.xlsub(fsupc); + Index jsupno, k, ksub, krep, ksupno; + Index lptr, nrow, isub, irow, nextlu, new_next, ufirst; + Index fsupc, nsupc, nsupr, luptr, kfnz, no_zeros; + /* krep = representative of current k-th supernode + * fsupc = first supernodal column + * nsupc = number of columns in a supernode + * nsupr = number of rows in a supernode + * luptr = location of supernodal LU-block in storage + * kfnz = first nonz in the k-th supernodal segment + * no_zeros = no lf leading zeros in a supernodal U-segment + */ + + jsupno = glu.supno(jcol); + // For each nonzero supernode segment of U[*,j] in topological order + k = nseg - 1; + Index d_fsupc;// distance between the first column of the current panel and the + // first column of the current snode + Index fst_col;// First column within small LU update + Index segsize; + for (ksub = 0; ksub < nseg; ksub++) { + krep = segrep(k); + k--; + ksupno = glu.supno(krep); + if (jsupno != ksupno) { + // outside the rectangular supernode + fsupc = glu.xsup(ksupno); + fst_col = (std::max)(fsupc, fpanelc); + + // Distance from the current supernode to the current panel; + // d_fsupc = 0 if fsupc > fpanelc + d_fsupc = fst_col - fsupc; + + luptr = glu.xlusup(fst_col) + d_fsupc; + lptr = glu.xlsub(fsupc) + d_fsupc; + + kfnz = repfnz(krep); + kfnz = (std::max)(kfnz, fpanelc); + + segsize = krep - kfnz + 1; + nsupc = krep - fst_col + 1; + nsupr = glu.xlsub(fsupc + 1) - glu.xlsub(fsupc); + nrow = nsupr - d_fsupc - nsupc; + Index lda = glu.xlusup(fst_col + 1) - glu.xlusup(fst_col); + + + // Perform a triangular solver and block update, + // then scatter the result of sup-col update to dense + no_zeros = kfnz - fst_col; + if (segsize == 1) + LU_kernel_bmod<1>::run(segsize, dense, tempv, glu.lusup, luptr, lda, nrow, glu.lsub, lptr, no_zeros); + else + LU_kernel_bmod::run(segsize, dense, tempv, glu.lusup, luptr, lda, nrow, glu.lsub, lptr, no_zeros); + }// end if jsupno + }// end for each segment + + // Process the supernodal portion of L\U[*,j] + nextlu = glu.xlusup(jcol); + fsupc = glu.xsup(jsupno); + + // copy the SPA dense into L\U[*,j] + Index mem; + new_next = nextlu + glu.xlsub(fsupc + 1) - glu.xlsub(fsupc); + Index offset = internal::first_multiple(new_next, internal::packet_traits::size) - new_next; + if (offset) new_next += offset; + while (new_next > glu.nzlumax) { + mem = memXpand(glu.lusup, glu.nzlumax, nextlu, LUSUP, glu.num_expansions); + if (mem) return mem; + } + + for (isub = glu.xlsub(fsupc); isub < glu.xlsub(fsupc + 1); isub++) { + irow = glu.lsub(isub); + glu.lusup(nextlu) = dense(irow); + dense(irow) = Scalar(0.0); + ++nextlu; + } + + if (offset) { + glu.lusup.segment(nextlu, offset).setZero(); + nextlu += offset; + } + glu.xlusup(jcol + 1) = StorageIndex(nextlu);// close L\U(*,jcol); + + /* For more updates within the panel (also within the current supernode), + * should start from the first column of the panel, or the first column + * of the supernode, whichever is bigger. There are two cases: + * 1) fsupc < fpanelc, then fst_col <-- fpanelc + * 2) fsupc >= fpanelc, then fst_col <-- fsupc + */ + fst_col = (std::max)(fsupc, fpanelc); + + if (fst_col < jcol) { + // Distance between the current supernode and the current panel + // d_fsupc = 0 if fsupc >= fpanelc + d_fsupc = fst_col - fsupc; + + lptr = glu.xlsub(fsupc) + d_fsupc; + luptr = glu.xlusup(fst_col) + d_fsupc; + nsupr = glu.xlsub(fsupc + 1) - glu.xlsub(fsupc);// leading dimension + nsupc = jcol - fst_col;// excluding jcol nrow = nsupr - d_fsupc - nsupc; - Index lda = glu.xlusup(fst_col+1) - glu.xlusup(fst_col); - - - // Perform a triangular solver and block update, - // then scatter the result of sup-col update to dense - no_zeros = kfnz - fst_col; - if(segsize==1) - LU_kernel_bmod<1>::run(segsize, dense, tempv, glu.lusup, luptr, lda, nrow, glu.lsub, lptr, no_zeros); - else - LU_kernel_bmod::run(segsize, dense, tempv, glu.lusup, luptr, lda, nrow, glu.lsub, lptr, no_zeros); - } // end if jsupno - } // end for each segment - - // Process the supernodal portion of L\U[*,j] - nextlu = glu.xlusup(jcol); - fsupc = glu.xsup(jsupno); - - // copy the SPA dense into L\U[*,j] - Index mem; - new_next = nextlu + glu.xlsub(fsupc + 1) - glu.xlsub(fsupc); - Index offset = internal::first_multiple(new_next, internal::packet_traits::size) - new_next; - if(offset) - new_next += offset; - while (new_next > glu.nzlumax ) - { - mem = memXpand(glu.lusup, glu.nzlumax, nextlu, LUSUP, glu.num_expansions); - if (mem) return mem; - } - - for (isub = glu.xlsub(fsupc); isub < glu.xlsub(fsupc+1); isub++) - { - irow = glu.lsub(isub); - glu.lusup(nextlu) = dense(irow); - dense(irow) = Scalar(0.0); - ++nextlu; - } - - if(offset) - { - glu.lusup.segment(nextlu,offset).setZero(); - nextlu += offset; + + // points to the beginning of jcol in snode L\U(jsupno) + ufirst = glu.xlusup(jcol) + d_fsupc; + Index lda = glu.xlusup(jcol + 1) - glu.xlusup(jcol); + MappedMatrixBlock A(&(glu.lusup.data()[luptr]), nsupc, nsupc, OuterStride<>(lda)); + VectorBlock u(glu.lusup, ufirst, nsupc); + u = A.template triangularView().solve(u); + + new (&A) MappedMatrixBlock(&(glu.lusup.data()[luptr + nsupc]), nrow, nsupc, OuterStride<>(lda)); + VectorBlock l(glu.lusup, ufirst + nsupc, nrow); + l.noalias() -= A * u; + + }// End if fst_col + return 0; } - glu.xlusup(jcol + 1) = StorageIndex(nextlu); // close L\U(*,jcol); - - /* For more updates within the panel (also within the current supernode), - * should start from the first column of the panel, or the first column - * of the supernode, whichever is bigger. There are two cases: - * 1) fsupc < fpanelc, then fst_col <-- fpanelc - * 2) fsupc >= fpanelc, then fst_col <-- fsupc - */ - fst_col = (std::max)(fsupc, fpanelc); - - if (fst_col < jcol) - { - // Distance between the current supernode and the current panel - // d_fsupc = 0 if fsupc >= fpanelc - d_fsupc = fst_col - fsupc; - - lptr = glu.xlsub(fsupc) + d_fsupc; - luptr = glu.xlusup(fst_col) + d_fsupc; - nsupr = glu.xlsub(fsupc+1) - glu.xlsub(fsupc); // leading dimension - nsupc = jcol - fst_col; // excluding jcol - nrow = nsupr - d_fsupc - nsupc; - - // points to the beginning of jcol in snode L\U(jsupno) - ufirst = glu.xlusup(jcol) + d_fsupc; - Index lda = glu.xlusup(jcol+1) - glu.xlusup(jcol); - MappedMatrixBlock A( &(glu.lusup.data()[luptr]), nsupc, nsupc, OuterStride<>(lda) ); - VectorBlock u(glu.lusup, ufirst, nsupc); - u = A.template triangularView().solve(u); - - new (&A) MappedMatrixBlock ( &(glu.lusup.data()[luptr+nsupc]), nrow, nsupc, OuterStride<>(lda) ); - VectorBlock l(glu.lusup, ufirst+nsupc, nrow); - l.noalias() -= A * u; - - } // End if fst_col - return 0; -} - -} // end namespace internal -} // end namespace Eigen - -#endif // SPARSELU_COLUMN_BMOD_H + +}// end namespace internal +}// end namespace Eigen + +#endif// SPARSELU_COLUMN_BMOD_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/SparseLU/SparseLU_column_dfs.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/SparseLU/SparseLU_column_dfs.h index c98b30e3..10f05e29 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/SparseLU/SparseLU_column_dfs.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/SparseLU/SparseLU_column_dfs.h @@ -7,10 +7,10 @@ // Public License v. 2.0. If a copy of the MPL was not distributed // with this file, You can obtain one at http://mozilla.org/MPL/2.0/. -/* - - * NOTE: This file is the modified version of [s,d,c,z]column_dfs.c file in SuperLU - +/* + + * NOTE: This file is the modified version of [s,d,c,z]column_dfs.c file in SuperLU + * -- SuperLU routine (version 2.0) -- * Univ. of California Berkeley, Xerox Palo Alto Research Center, * and Lawrence Berkeley National Lab. @@ -30,150 +30,163 @@ #ifndef SPARSELU_COLUMN_DFS_H #define SPARSELU_COLUMN_DFS_H -template class SparseLUImpl; +template class SparseLUImpl; namespace Eigen { namespace internal { -template -struct column_dfs_traits : no_assignment_operator -{ - typedef typename ScalarVector::Scalar Scalar; - typedef typename IndexVector::Scalar StorageIndex; - column_dfs_traits(Index jcol, Index& jsuper, typename SparseLUImpl::GlobalLU_t& glu, SparseLUImpl& luImpl) - : m_jcol(jcol), m_jsuper_ref(jsuper), m_glu(glu), m_luImpl(luImpl) - {} - bool update_segrep(Index /*krep*/, Index /*jj*/) - { - return true; - } - void mem_expand(IndexVector& lsub, Index& nextl, Index chmark) + template struct column_dfs_traits : no_assignment_operator { - if (nextl >= m_glu.nzlmax) - m_luImpl.memXpand(lsub, m_glu.nzlmax, nextl, LSUB, m_glu.num_expansions); - if (chmark != (m_jcol-1)) m_jsuper_ref = emptyIdxLU; - } - enum { ExpandMem = true }; - - Index m_jcol; - Index& m_jsuper_ref; - typename SparseLUImpl::GlobalLU_t& m_glu; - SparseLUImpl& m_luImpl; -}; - - -/** - * \brief Performs a symbolic factorization on column jcol and decide the supernode boundary - * - * A supernode representative is the last column of a supernode. - * The nonzeros in U[*,j] are segments that end at supernodes representatives. - * The routine returns a list of the supernodal representatives - * in topological order of the dfs that generates them. - * The location of the first nonzero in each supernodal segment - * (supernodal entry location) is also returned. - * - * \param m number of rows in the matrix - * \param jcol Current column - * \param perm_r Row permutation - * \param maxsuper Maximum number of column allowed in a supernode - * \param [in,out] nseg Number of segments in current U[*,j] - new segments appended - * \param lsub_col defines the rhs vector to start the dfs - * \param [in,out] segrep Segment representatives - new segments appended - * \param repfnz First nonzero location in each row - * \param xprune - * \param marker marker[i] == jj, if i was visited during dfs of current column jj; - * \param parent - * \param xplore working array - * \param glu global LU data - * \return 0 success - * > 0 number of bytes allocated when run out of space - * - */ -template -Index SparseLUImpl::column_dfs(const Index m, const Index jcol, IndexVector& perm_r, Index maxsuper, Index& nseg, - BlockIndexVector lsub_col, IndexVector& segrep, BlockIndexVector repfnz, IndexVector& xprune, - IndexVector& marker, IndexVector& parent, IndexVector& xplore, GlobalLU_t& glu) -{ - - Index jsuper = glu.supno(jcol); - Index nextl = glu.xlsub(jcol); - VectorBlock marker2(marker, 2*m, m); - - - column_dfs_traits traits(jcol, jsuper, glu, *this); - - // For each nonzero in A(*,jcol) do dfs - for (Index k = 0; ((k < m) ? lsub_col[k] != emptyIdxLU : false) ; k++) + typedef typename ScalarVector::Scalar Scalar; + typedef typename IndexVector::Scalar StorageIndex; + column_dfs_traits(Index jcol, + Index &jsuper, + typename SparseLUImpl::GlobalLU_t &glu, + SparseLUImpl &luImpl) + : m_jcol(jcol), m_jsuper_ref(jsuper), m_glu(glu), m_luImpl(luImpl) + {} + bool update_segrep(Index /*krep*/, Index /*jj*/) { return true; } + void mem_expand(IndexVector &lsub, Index &nextl, Index chmark) + { + if (nextl >= m_glu.nzlmax) m_luImpl.memXpand(lsub, m_glu.nzlmax, nextl, LSUB, m_glu.num_expansions); + if (chmark != (m_jcol - 1)) m_jsuper_ref = emptyIdxLU; + } + enum { ExpandMem = true }; + + Index m_jcol; + Index &m_jsuper_ref; + typename SparseLUImpl::GlobalLU_t &m_glu; + SparseLUImpl &m_luImpl; + }; + + + /** + * \brief Performs a symbolic factorization on column jcol and decide the supernode boundary + * + * A supernode representative is the last column of a supernode. + * The nonzeros in U[*,j] are segments that end at supernodes representatives. + * The routine returns a list of the supernodal representatives + * in topological order of the dfs that generates them. + * The location of the first nonzero in each supernodal segment + * (supernodal entry location) is also returned. + * + * \param m number of rows in the matrix + * \param jcol Current column + * \param perm_r Row permutation + * \param maxsuper Maximum number of column allowed in a supernode + * \param [in,out] nseg Number of segments in current U[*,j] - new segments appended + * \param lsub_col defines the rhs vector to start the dfs + * \param [in,out] segrep Segment representatives - new segments appended + * \param repfnz First nonzero location in each row + * \param xprune + * \param marker marker[i] == jj, if i was visited during dfs of current column jj; + * \param parent + * \param xplore working array + * \param glu global LU data + * \return 0 success + * > 0 number of bytes allocated when run out of space + * + */ + template + Index SparseLUImpl::column_dfs(const Index m, + const Index jcol, + IndexVector &perm_r, + Index maxsuper, + Index &nseg, + BlockIndexVector lsub_col, + IndexVector &segrep, + BlockIndexVector repfnz, + IndexVector &xprune, + IndexVector &marker, + IndexVector &parent, + IndexVector &xplore, + GlobalLU_t &glu) { - Index krow = lsub_col(k); - lsub_col(k) = emptyIdxLU; - Index kmark = marker2(krow); - - // krow was visited before, go to the next nonz; - if (kmark == jcol) continue; - - dfs_kernel(StorageIndex(jcol), perm_r, nseg, glu.lsub, segrep, repfnz, xprune, marker2, parent, - xplore, glu, nextl, krow, traits); - } // for each nonzero ... - - Index fsupc; - StorageIndex nsuper = glu.supno(jcol); - StorageIndex jcolp1 = StorageIndex(jcol) + 1; - Index jcolm1 = jcol - 1; - - // check to see if j belongs in the same supernode as j-1 - if ( jcol == 0 ) - { // Do nothing for column 0 - nsuper = glu.supno(0) = 0 ; + + Index jsuper = glu.supno(jcol); + Index nextl = glu.xlsub(jcol); + VectorBlock marker2(marker, 2 * m, m); + + + column_dfs_traits traits(jcol, jsuper, glu, *this); + + // For each nonzero in A(*,jcol) do dfs + for (Index k = 0; ((k < m) ? lsub_col[k] != emptyIdxLU : false); k++) { + Index krow = lsub_col(k); + lsub_col(k) = emptyIdxLU; + Index kmark = marker2(krow); + + // krow was visited before, go to the next nonz; + if (kmark == jcol) continue; + + dfs_kernel(StorageIndex(jcol), + perm_r, + nseg, + glu.lsub, + segrep, + repfnz, + xprune, + marker2, + parent, + xplore, + glu, + nextl, + krow, + traits); + }// for each nonzero ... + + Index fsupc; + StorageIndex nsuper = glu.supno(jcol); + StorageIndex jcolp1 = StorageIndex(jcol) + 1; + Index jcolm1 = jcol - 1; + + // check to see if j belongs in the same supernode as j-1 + if (jcol == 0) {// Do nothing for column 0 + nsuper = glu.supno(0) = 0; + } else { + fsupc = glu.xsup(nsuper); + StorageIndex jptr = glu.xlsub(jcol);// Not yet compressed + StorageIndex jm1ptr = glu.xlsub(jcolm1); + + // Use supernodes of type T2 : see SuperLU paper + if ((nextl - jptr != jptr - jm1ptr - 1)) jsuper = emptyIdxLU; + + // Make sure the number of columns in a supernode doesn't + // exceed threshold + if ((jcol - fsupc) >= maxsuper) jsuper = emptyIdxLU; + + /* If jcol starts a new supernode, reclaim storage space in + * glu.lsub from previous supernode. Note we only store + * the subscript set of the first and last columns of + * a supernode. (first for num values, last for pruning) + */ + if (jsuper == emptyIdxLU) {// starts a new supernode + if ((fsupc < jcolm1 - 1)) {// >= 3 columns in nsuper + StorageIndex ito = glu.xlsub(fsupc + 1); + glu.xlsub(jcolm1) = ito; + StorageIndex istop = ito + jptr - jm1ptr; + xprune(jcolm1) = istop;// intialize xprune(jcol-1) + glu.xlsub(jcol) = istop; + + for (StorageIndex ifrom = jm1ptr; ifrom < nextl; ++ifrom, ++ito) glu.lsub(ito) = glu.lsub(ifrom); + nextl = ito;// = istop + length(jcol) + } + nsuper++; + glu.supno(jcol) = nsuper; + }// if a new supernode + }// end else: jcol > 0 + + // Tidy up the pointers before exit + glu.xsup(nsuper + 1) = jcolp1; + glu.supno(jcolp1) = nsuper; + xprune(jcol) = StorageIndex(nextl);// Intialize upper bound for pruning + glu.xlsub(jcolp1) = StorageIndex(nextl); + + return 0; } - else - { - fsupc = glu.xsup(nsuper); - StorageIndex jptr = glu.xlsub(jcol); // Not yet compressed - StorageIndex jm1ptr = glu.xlsub(jcolm1); - - // Use supernodes of type T2 : see SuperLU paper - if ( (nextl-jptr != jptr-jm1ptr-1) ) jsuper = emptyIdxLU; - - // Make sure the number of columns in a supernode doesn't - // exceed threshold - if ( (jcol - fsupc) >= maxsuper) jsuper = emptyIdxLU; - - /* If jcol starts a new supernode, reclaim storage space in - * glu.lsub from previous supernode. Note we only store - * the subscript set of the first and last columns of - * a supernode. (first for num values, last for pruning) - */ - if (jsuper == emptyIdxLU) - { // starts a new supernode - if ( (fsupc < jcolm1-1) ) - { // >= 3 columns in nsuper - StorageIndex ito = glu.xlsub(fsupc+1); - glu.xlsub(jcolm1) = ito; - StorageIndex istop = ito + jptr - jm1ptr; - xprune(jcolm1) = istop; // intialize xprune(jcol-1) - glu.xlsub(jcol) = istop; - - for (StorageIndex ifrom = jm1ptr; ifrom < nextl; ++ifrom, ++ito) - glu.lsub(ito) = glu.lsub(ifrom); - nextl = ito; // = istop + length(jcol) - } - nsuper++; - glu.supno(jcol) = nsuper; - } // if a new supernode - } // end else: jcol > 0 - - // Tidy up the pointers before exit - glu.xsup(nsuper+1) = jcolp1; - glu.supno(jcolp1) = nsuper; - xprune(jcol) = StorageIndex(nextl); // Intialize upper bound for pruning - glu.xlsub(jcolp1) = StorageIndex(nextl); - - return 0; -} - -} // end namespace internal - -} // end namespace Eigen + +}// end namespace internal + +}// end namespace Eigen #endif diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/SparseLU/SparseLU_copy_to_ucol.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/SparseLU/SparseLU_copy_to_ucol.h index c32d8d8b..e9e1d763 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/SparseLU/SparseLU_copy_to_ucol.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/SparseLU/SparseLU_copy_to_ucol.h @@ -6,10 +6,10 @@ // This Source Code Form is subject to the terms of the Mozilla // Public License v. 2.0. If a copy of the MPL was not distributed // with this file, You can obtain one at http://mozilla.org/MPL/2.0/. -/* - - * NOTE: This file is the modified version of [s,d,c,z]copy_to_ucol.c file in SuperLU - +/* + + * NOTE: This file is the modified version of [s,d,c,z]copy_to_ucol.c file in SuperLU + * -- SuperLU routine (version 2.0) -- * Univ. of California Berkeley, Xerox Palo Alto Research Center, * and Lawrence Berkeley National Lab. @@ -32,76 +32,77 @@ namespace Eigen { namespace internal { -/** - * \brief Performs numeric block updates (sup-col) in topological order - * - * \param jcol current column to update - * \param nseg Number of segments in the U part - * \param segrep segment representative ... - * \param repfnz First nonzero column in each row ... - * \param perm_r Row permutation - * \param dense Store the full representation of the column - * \param glu Global LU data. - * \return 0 - successful return - * > 0 - number of bytes allocated when run out of space - * - */ -template -Index SparseLUImpl::copy_to_ucol(const Index jcol, const Index nseg, IndexVector& segrep, - BlockIndexVector repfnz ,IndexVector& perm_r, BlockScalarVector dense, GlobalLU_t& glu) -{ - Index ksub, krep, ksupno; - - Index jsupno = glu.supno(jcol); - - // For each nonzero supernode segment of U[*,j] in topological order - Index k = nseg - 1, i; - StorageIndex nextu = glu.xusub(jcol); - Index kfnz, isub, segsize; - Index new_next,irow; - Index fsupc, mem; - for (ksub = 0; ksub < nseg; ksub++) + /** + * \brief Performs numeric block updates (sup-col) in topological order + * + * \param jcol current column to update + * \param nseg Number of segments in the U part + * \param segrep segment representative ... + * \param repfnz First nonzero column in each row ... + * \param perm_r Row permutation + * \param dense Store the full representation of the column + * \param glu Global LU data. + * \return 0 - successful return + * > 0 - number of bytes allocated when run out of space + * + */ + template + Index SparseLUImpl::copy_to_ucol(const Index jcol, + const Index nseg, + IndexVector &segrep, + BlockIndexVector repfnz, + IndexVector &perm_r, + BlockScalarVector dense, + GlobalLU_t &glu) { - krep = segrep(k); k--; - ksupno = glu.supno(krep); - if (jsupno != ksupno ) // should go into ucol(); - { - kfnz = repfnz(krep); - if (kfnz != emptyIdxLU) - { // Nonzero U-segment - fsupc = glu.xsup(ksupno); - isub = glu.xlsub(fsupc) + kfnz - fsupc; - segsize = krep - kfnz + 1; - new_next = nextu + segsize; - while (new_next > glu.nzumax) - { - mem = memXpand(glu.ucol, glu.nzumax, nextu, UCOL, glu.num_expansions); - if (mem) return mem; - mem = memXpand(glu.usub, glu.nzumax, nextu, USUB, glu.num_expansions); - if (mem) return mem; - - } - - for (i = 0; i < segsize; i++) - { - irow = glu.lsub(isub); - glu.usub(nextu) = perm_r(irow); // Unlike the L part, the U part is stored in its final order - glu.ucol(nextu) = dense(irow); - dense(irow) = Scalar(0.0); - nextu++; - isub++; - } - - } // end nonzero U-segment - - } // end if jsupno - - } // end for each segment - glu.xusub(jcol + 1) = nextu; // close U(*,jcol) - return 0; -} + Index ksub, krep, ksupno; + + Index jsupno = glu.supno(jcol); + + // For each nonzero supernode segment of U[*,j] in topological order + Index k = nseg - 1, i; + StorageIndex nextu = glu.xusub(jcol); + Index kfnz, isub, segsize; + Index new_next, irow; + Index fsupc, mem; + for (ksub = 0; ksub < nseg; ksub++) { + krep = segrep(k); + k--; + ksupno = glu.supno(krep); + if (jsupno != ksupno)// should go into ucol(); + { + kfnz = repfnz(krep); + if (kfnz != emptyIdxLU) {// Nonzero U-segment + fsupc = glu.xsup(ksupno); + isub = glu.xlsub(fsupc) + kfnz - fsupc; + segsize = krep - kfnz + 1; + new_next = nextu + segsize; + while (new_next > glu.nzumax) { + mem = memXpand(glu.ucol, glu.nzumax, nextu, UCOL, glu.num_expansions); + if (mem) return mem; + mem = memXpand(glu.usub, glu.nzumax, nextu, USUB, glu.num_expansions); + if (mem) return mem; + } + + for (i = 0; i < segsize; i++) { + irow = glu.lsub(isub); + glu.usub(nextu) = perm_r(irow);// Unlike the L part, the U part is stored in its final order + glu.ucol(nextu) = dense(irow); + dense(irow) = Scalar(0.0); + nextu++; + isub++; + } + + }// end nonzero U-segment + + }// end if jsupno + + }// end for each segment + glu.xusub(jcol + 1) = nextu;// close U(*,jcol) + return 0; + } -} // namespace internal -} // end namespace Eigen +}// namespace internal +}// end namespace Eigen -#endif // SPARSELU_COPY_TO_UCOL_H +#endif// SPARSELU_COPY_TO_UCOL_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/SparseLU/SparseLU_gemm_kernel.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/SparseLU/SparseLU_gemm_kernel.h index 95ba7413..09ce1c8d 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/SparseLU/SparseLU_gemm_kernel.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/SparseLU/SparseLU_gemm_kernel.h @@ -15,266 +15,262 @@ namespace Eigen { namespace internal { -/** \internal - * A general matrix-matrix product kernel optimized for the SparseLU factorization. - * - A, B, and C must be column major - * - lda and ldc must be multiples of the respective packet size - * - C must have the same alignment as A - */ -template -EIGEN_DONT_INLINE -void sparselu_gemm(Index m, Index n, Index d, const Scalar* A, Index lda, const Scalar* B, Index ldb, Scalar* C, Index ldc) -{ - using namespace Eigen::internal; - - typedef typename packet_traits::type Packet; - enum { - NumberOfRegisters = EIGEN_ARCH_DEFAULT_NUMBER_OF_REGISTERS, - PacketSize = packet_traits::size, - PM = 8, // peeling in M - RN = 2, // register blocking - RK = NumberOfRegisters>=16 ? 4 : 2, // register blocking - BM = 4096/sizeof(Scalar), // number of rows of A-C per chunk - SM = PM*PacketSize // step along M - }; - Index d_end = (d/RK)*RK; // number of columns of A (rows of B) suitable for full register blocking - Index n_end = (n/RN)*RN; // number of columns of B-C suitable for processing RN columns at once - Index i0 = internal::first_default_aligned(A,m); - - eigen_internal_assert(((lda%PacketSize)==0) && ((ldc%PacketSize)==0) && (i0==internal::first_default_aligned(C,m))); - - // handle the non aligned rows of A and C without any optimization: - for(Index i=0; i + EIGEN_DONT_INLINE void sparselu_gemm(Index m, + Index n, + Index d, + const Scalar *A, + Index lda, + const Scalar *B, + Index ldb, + Scalar *C, + Index ldc) { - for(Index j=0; j::type Packet; + enum { + NumberOfRegisters = EIGEN_ARCH_DEFAULT_NUMBER_OF_REGISTERS, + PacketSize = packet_traits::size, + PM = 8,// peeling in M + RN = 2,// register blocking + RK = NumberOfRegisters >= 16 ? 4 : 2,// register blocking + BM = 4096 / sizeof(Scalar),// number of rows of A-C per chunk + SM = PM * PacketSize// step along M + }; + Index d_end = (d / RK) * RK;// number of columns of A (rows of B) suitable for full register blocking + Index n_end = (n / RN) * RN;// number of columns of B-C suitable for processing RN columns at once + Index i0 = internal::first_default_aligned(A, m); + + eigen_internal_assert( + ((lda % PacketSize) == 0) && ((ldc % PacketSize) == 0) && (i0 == internal::first_default_aligned(C, m))); + + // handle the non aligned rows of A and C without any optimization: + for (Index i = 0; i < i0; ++i) { + for (Index j = 0; j < n; ++j) { + Scalar c = C[i + j * ldc]; + for (Index k = 0; k < d; ++k) c += B[k + j * ldb] * A[i + k * lda]; + C[i + j * ldc] = c; + } } - } - // process the remaining rows per chunk of BM rows - for(Index ib=i0; ib(BM, m-ib); // actual number of rows - Index actual_b_end1 = (actual_b/SM)*SM; // actual number of rows suitable for peeling - Index actual_b_end2 = (actual_b/PacketSize)*PacketSize; // actual number of rows suitable for vectorization - - // Let's process two columns of B-C at once - for(Index j=0; j(Bc0[0]); } - { b10 = pset1(Bc0[1]); } - if(RK==4) { b20 = pset1(Bc0[2]); } - if(RK==4) { b30 = pset1(Bc0[3]); } - { b01 = pset1(Bc1[0]); } - { b11 = pset1(Bc1[1]); } - if(RK==4) { b21 = pset1(Bc1[2]); } - if(RK==4) { b31 = pset1(Bc1[3]); } - - Packet a0, a1, a2, a3, c0, c1, t0, t1; - - const Scalar* A0 = A+ib+(k+0)*lda; - const Scalar* A1 = A+ib+(k+1)*lda; - const Scalar* A2 = A+ib+(k+2)*lda; - const Scalar* A3 = A+ib+(k+3)*lda; - - Scalar* C0 = C+ib+(j+0)*ldc; - Scalar* C1 = C+ib+(j+1)*ldc; - - a0 = pload(A0); - a1 = pload(A1); - if(RK==4) - { - a2 = pload(A2); - a3 = pload(A3); - } - else - { - // workaround "may be used uninitialized in this function" warning - a2 = a3 = a0; - } - -#define KMADD(c, a, b, tmp) {tmp = b; tmp = pmul(a,tmp); c = padd(c,tmp);} -#define WORK(I) \ - c0 = pload(C0+i+(I)*PacketSize); \ - c1 = pload(C1+i+(I)*PacketSize); \ - KMADD(c0, a0, b00, t0) \ - KMADD(c1, a0, b01, t1) \ - a0 = pload(A0+i+(I+1)*PacketSize); \ - KMADD(c0, a1, b10, t0) \ - KMADD(c1, a1, b11, t1) \ - a1 = pload(A1+i+(I+1)*PacketSize); \ - if(RK==4){ KMADD(c0, a2, b20, t0) }\ - if(RK==4){ KMADD(c1, a2, b21, t1) }\ - if(RK==4){ a2 = pload(A2+i+(I+1)*PacketSize); }\ - if(RK==4){ KMADD(c0, a3, b30, t0) }\ - if(RK==4){ KMADD(c1, a3, b31, t1) }\ - if(RK==4){ a3 = pload(A3+i+(I+1)*PacketSize); }\ - pstore(C0+i+(I)*PacketSize, c0); \ - pstore(C1+i+(I)*PacketSize, c1) - - // process rows of A' - C' with aggressive vectorization and peeling - for(Index i=0; i(BM, m - ib);// actual number of rows + Index actual_b_end1 = (actual_b / SM) * SM;// actual number of rows suitable for peeling + Index actual_b_end2 = (actual_b / PacketSize) * PacketSize;// actual number of rows suitable for vectorization + + // Let's process two columns of B-C at once + for (Index j = 0; j < n_end; j += RN) { + const Scalar *Bc0 = B + (j + 0) * ldb; + const Scalar *Bc1 = B + (j + 1) * ldb; + + for (Index k = 0; k < d_end; k += RK) { + + // load and expand a RN x RK block of B + Packet b00, b10, b20, b30, b01, b11, b21, b31; { - C0[i] += A0[i]*Bc0[0]+A1[i]*Bc0[1]+A2[i]*Bc0[2]+A3[i]*Bc0[3]; - C1[i] += A0[i]*Bc1[0]+A1[i]*Bc1[1]+A2[i]*Bc1[2]+A3[i]*Bc1[3]; + b00 = pset1(Bc0[0]); } - else { - C0[i] += A0[i]*Bc0[0]+A1[i]*Bc0[1]; - C1[i] += A0[i]*Bc1[0]+A1[i]*Bc1[1]; + b10 = pset1(Bc0[1]); } + if (RK == 4) { b20 = pset1(Bc0[2]); } + if (RK == 4) { b30 = pset1(Bc0[3]); } + { + b01 = pset1(Bc1[0]); + } + { + b11 = pset1(Bc1[1]); + } + if (RK == 4) { b21 = pset1(Bc1[2]); } + if (RK == 4) { b31 = pset1(Bc1[3]); } + + Packet a0, a1, a2, a3, c0, c1, t0, t1; + + const Scalar *A0 = A + ib + (k + 0) * lda; + const Scalar *A1 = A + ib + (k + 1) * lda; + const Scalar *A2 = A + ib + (k + 2) * lda; + const Scalar *A3 = A + ib + (k + 3) * lda; + + Scalar *C0 = C + ib + (j + 0) * ldc; + Scalar *C1 = C + ib + (j + 1) * ldc; + + a0 = pload(A0); + a1 = pload(A1); + if (RK == 4) { + a2 = pload(A2); + a3 = pload(A3); + } else { + // workaround "may be used uninitialized in this function" warning + a2 = a3 = a0; + } + +#define KMADD(c, a, b, tmp) \ + { \ + tmp = b; \ + tmp = pmul(a, tmp); \ + c = padd(c, tmp); \ + } +#define WORK(I) \ + c0 = pload(C0 + i + (I) * PacketSize); \ + c1 = pload(C1 + i + (I) * PacketSize); \ + KMADD(c0, a0, b00, t0) \ + KMADD(c1, a0, b01, t1) \ + a0 = pload(A0 + i + (I + 1) * PacketSize); \ + KMADD(c0, a1, b10, t0) \ + KMADD(c1, a1, b11, t1) \ + a1 = pload(A1 + i + (I + 1) * PacketSize); \ + if (RK == 4) { KMADD(c0, a2, b20, t0) } \ + if (RK == 4) { KMADD(c1, a2, b21, t1) } \ + if (RK == 4) { a2 = pload(A2 + i + (I + 1) * PacketSize); } \ + if (RK == 4) { KMADD(c0, a3, b30, t0) } \ + if (RK == 4) { KMADD(c1, a3, b31, t1) } \ + if (RK == 4) { a3 = pload(A3 + i + (I + 1) * PacketSize); } \ + pstore(C0 + i + (I) * PacketSize, c0); \ + pstore(C1 + i + (I) * PacketSize, c1) + + // process rows of A' - C' with aggressive vectorization and peeling + for (Index i = 0; i < actual_b_end1; i += PacketSize * 8) { + EIGEN_ASM_COMMENT("SPARSELU_GEMML_KERNEL1"); + prefetch((A0 + i + (5) * PacketSize)); + prefetch((A1 + i + (5) * PacketSize)); + if (RK == 4) prefetch((A2 + i + (5) * PacketSize)); + if (RK == 4) prefetch((A3 + i + (5) * PacketSize)); + + WORK(0); + WORK(1); + WORK(2); + WORK(3); + WORK(4); + WORK(5); + WORK(6); + WORK(7); + } + // process the remaining rows with vectorization only + for (Index i = actual_b_end1; i < actual_b_end2; i += PacketSize) { WORK(0); } +#undef WORK + // process the remaining rows without vectorization + for (Index i = actual_b_end2; i < actual_b; ++i) { + if (RK == 4) { + C0[i] += A0[i] * Bc0[0] + A1[i] * Bc0[1] + A2[i] * Bc0[2] + A3[i] * Bc0[3]; + C1[i] += A0[i] * Bc1[0] + A1[i] * Bc1[1] + A2[i] * Bc1[2] + A3[i] * Bc1[3]; + } else { + C0[i] += A0[i] * Bc0[0] + A1[i] * Bc0[1]; + C1[i] += A0[i] * Bc1[0] + A1[i] * Bc1[1]; + } + } + + Bc0 += RK; + Bc1 += RK; + }// peeled loop on k + }// peeled loop on the columns j + // process the last column (we now perform a matrix-vector product) + if ((n - n_end) > 0) { + const Scalar *Bc0 = B + (n - 1) * ldb; + + for (Index k = 0; k < d_end; k += RK) { + + // load and expand a 1 x RK block of B + Packet b00, b10, b20, b30; + b00 = pset1(Bc0[0]); + b10 = pset1(Bc0[1]); + if (RK == 4) b20 = pset1(Bc0[2]); + if (RK == 4) b30 = pset1(Bc0[3]); + + Packet a0, a1, a2, a3, c0, t0 /*, t1*/; + + const Scalar *A0 = A + ib + (k + 0) * lda; + const Scalar *A1 = A + ib + (k + 1) * lda; + const Scalar *A2 = A + ib + (k + 2) * lda; + const Scalar *A3 = A + ib + (k + 3) * lda; + + Scalar *C0 = C + ib + (n_end)*ldc; + + a0 = pload(A0); + a1 = pload(A1); + if (RK == 4) { + a2 = pload(A2); + a3 = pload(A3); + } else { + // workaround "may be used uninitialized in this function" warning + a2 = a3 = a0; + } + +#define WORK(I) \ + c0 = pload(C0 + i + (I) * PacketSize); \ + KMADD(c0, a0, b00, t0) \ + a0 = pload(A0 + i + (I + 1) * PacketSize); \ + KMADD(c0, a1, b10, t0) \ + a1 = pload(A1 + i + (I + 1) * PacketSize); \ + if (RK == 4) { KMADD(c0, a2, b20, t0) } \ + if (RK == 4) { a2 = pload(A2 + i + (I + 1) * PacketSize); } \ + if (RK == 4) { KMADD(c0, a3, b30, t0) } \ + if (RK == 4) { a3 = pload(A3 + i + (I + 1) * PacketSize); } \ + pstore(C0 + i + (I) * PacketSize, c0); + + // agressive vectorization and peeling + for (Index i = 0; i < actual_b_end1; i += PacketSize * 8) { + EIGEN_ASM_COMMENT("SPARSELU_GEMML_KERNEL2"); + WORK(0); + WORK(1); + WORK(2); + WORK(3); + WORK(4); + WORK(5); + WORK(6); + WORK(7); + } + // vectorization only + for (Index i = actual_b_end1; i < actual_b_end2; i += PacketSize) { WORK(0); } + // remaining scalars + for (Index i = actual_b_end2; i < actual_b; ++i) { + if (RK == 4) + C0[i] += A0[i] * Bc0[0] + A1[i] * Bc0[1] + A2[i] * Bc0[2] + A3[i] * Bc0[3]; + else + C0[i] += A0[i] * Bc0[0] + A1[i] * Bc0[1]; + } + + Bc0 += RK; +#undef WORK } - - Bc0 += RK; - Bc1 += RK; - } // peeled loop on k - } // peeled loop on the columns j - // process the last column (we now perform a matrix-vector product) - if((n-n_end)>0) - { - const Scalar* Bc0 = B+(n-1)*ldb; - - for(Index k=0; k(Bc0[0]); - b10 = pset1(Bc0[1]); - if(RK==4) b20 = pset1(Bc0[2]); - if(RK==4) b30 = pset1(Bc0[3]); - - Packet a0, a1, a2, a3, c0, t0/*, t1*/; - - const Scalar* A0 = A+ib+(k+0)*lda; - const Scalar* A1 = A+ib+(k+1)*lda; - const Scalar* A2 = A+ib+(k+2)*lda; - const Scalar* A3 = A+ib+(k+3)*lda; - - Scalar* C0 = C+ib+(n_end)*ldc; - - a0 = pload(A0); - a1 = pload(A1); - if(RK==4) - { - a2 = pload(A2); - a3 = pload(A3); - } - else - { - // workaround "may be used uninitialized in this function" warning - a2 = a3 = a0; - } - -#define WORK(I) \ - c0 = pload(C0+i+(I)*PacketSize); \ - KMADD(c0, a0, b00, t0) \ - a0 = pload(A0+i+(I+1)*PacketSize); \ - KMADD(c0, a1, b10, t0) \ - a1 = pload(A1+i+(I+1)*PacketSize); \ - if(RK==4){ KMADD(c0, a2, b20, t0) }\ - if(RK==4){ a2 = pload(A2+i+(I+1)*PacketSize); }\ - if(RK==4){ KMADD(c0, a3, b30, t0) }\ - if(RK==4){ a3 = pload(A3+i+(I+1)*PacketSize); }\ - pstore(C0+i+(I)*PacketSize, c0); - - // agressive vectorization and peeling - for(Index i=0; i 0) { + for (Index j = 0; j < n; ++j) { + enum { Alignment = PacketSize > 1 ? Aligned : 0 }; + typedef Map, Alignment> MapVector; + typedef Map, Alignment> ConstMapVector; + if (rd == 1) + MapVector(C + j * ldc + ib, actual_b) += + B[0 + d_end + j * ldb] * ConstMapVector(A + (d_end + 0) * lda + ib, actual_b); + + else if (rd == 2) + MapVector(C + j * ldc + ib, actual_b) += + B[0 + d_end + j * ldb] * ConstMapVector(A + (d_end + 0) * lda + ib, actual_b) + + B[1 + d_end + j * ldb] * ConstMapVector(A + (d_end + 1) * lda + ib, actual_b); + else - C0[i] += A0[i]*Bc0[0]+A1[i]*Bc0[1]; + MapVector(C + j * ldc + ib, actual_b) += + B[0 + d_end + j * ldb] * ConstMapVector(A + (d_end + 0) * lda + ib, actual_b) + + B[1 + d_end + j * ldb] * ConstMapVector(A + (d_end + 1) * lda + ib, actual_b) + + B[2 + d_end + j * ldb] * ConstMapVector(A + (d_end + 2) * lda + ib, actual_b); } - - Bc0 += RK; -#undef WORK - } - } - - // process the last columns of A, corresponding to the last rows of B - Index rd = d-d_end; - if(rd>0) - { - for(Index j=0; j1 ? Aligned : 0 - }; - typedef Map, Alignment > MapVector; - typedef Map, Alignment > ConstMapVector; - if(rd==1) MapVector(C+j*ldc+ib,actual_b) += B[0+d_end+j*ldb] * ConstMapVector(A+(d_end+0)*lda+ib, actual_b); - - else if(rd==2) MapVector(C+j*ldc+ib,actual_b) += B[0+d_end+j*ldb] * ConstMapVector(A+(d_end+0)*lda+ib, actual_b) - + B[1+d_end+j*ldb] * ConstMapVector(A+(d_end+1)*lda+ib, actual_b); - - else MapVector(C+j*ldc+ib,actual_b) += B[0+d_end+j*ldb] * ConstMapVector(A+(d_end+0)*lda+ib, actual_b) - + B[1+d_end+j*ldb] * ConstMapVector(A+(d_end+1)*lda+ib, actual_b) - + B[2+d_end+j*ldb] * ConstMapVector(A+(d_end+2)*lda+ib, actual_b); } - } - - } // blocking on the rows of A and C -} + + }// blocking on the rows of A and C + } #undef KMADD -} // namespace internal +}// namespace internal -} // namespace Eigen +}// namespace Eigen -#endif // EIGEN_SPARSELU_GEMM_KERNEL_H +#endif// EIGEN_SPARSELU_GEMM_KERNEL_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/SparseLU/SparseLU_heap_relax_snode.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/SparseLU/SparseLU_heap_relax_snode.h index 6f75d500..35340986 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/SparseLU/SparseLU_heap_relax_snode.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/SparseLU/SparseLU_heap_relax_snode.h @@ -31,96 +31,89 @@ namespace Eigen { namespace internal { -/** - * \brief Identify the initial relaxed supernodes - * - * This routine applied to a symmetric elimination tree. - * It assumes that the matrix has been reordered according to the postorder of the etree - * \param n The number of columns - * \param et elimination tree - * \param relax_columns Maximum number of columns allowed in a relaxed snode - * \param descendants Number of descendants of each node in the etree - * \param relax_end last column in a supernode - */ -template -void SparseLUImpl::heap_relax_snode (const Index n, IndexVector& et, const Index relax_columns, IndexVector& descendants, IndexVector& relax_end) -{ - - // The etree may not be postordered, but its heap ordered - IndexVector post; - internal::treePostorder(StorageIndex(n), et, post); // Post order etree - IndexVector inv_post(n+1); - for (StorageIndex i = 0; i < n+1; ++i) inv_post(post(i)) = i; // inv_post = post.inverse()??? - - // Renumber etree in postorder - IndexVector iwork(n); - IndexVector et_save(n+1); - for (Index i = 0; i < n; ++i) - { - iwork(post(i)) = post(et(i)); - } - et_save = et; // Save the original etree - et = iwork; - - // compute the number of descendants of each node in the etree - relax_end.setConstant(emptyIdxLU); - Index j, parent; - descendants.setZero(); - for (j = 0; j < n; j++) + /** + * \brief Identify the initial relaxed supernodes + * + * This routine applied to a symmetric elimination tree. + * It assumes that the matrix has been reordered according to the postorder of the etree + * \param n The number of columns + * \param et elimination tree + * \param relax_columns Maximum number of columns allowed in a relaxed snode + * \param descendants Number of descendants of each node in the etree + * \param relax_end last column in a supernode + */ + template + void SparseLUImpl::heap_relax_snode(const Index n, + IndexVector &et, + const Index relax_columns, + IndexVector &descendants, + IndexVector &relax_end) { - parent = et(j); - if (parent != n) // not the dummy root - descendants(parent) += descendants(j) + 1; - } - // Identify the relaxed supernodes by postorder traversal of the etree - Index snode_start; // beginning of a snode - StorageIndex k; - Index nsuper_et_post = 0; // Number of relaxed snodes in postordered etree - Index nsuper_et = 0; // Number of relaxed snodes in the original etree - StorageIndex l; - for (j = 0; j < n; ) - { - parent = et(j); - snode_start = j; - while ( parent != n && descendants(parent) < relax_columns ) - { - j = parent; + + // The etree may not be postordered, but its heap ordered + IndexVector post; + internal::treePostorder(StorageIndex(n), et, post);// Post order etree + IndexVector inv_post(n + 1); + for (StorageIndex i = 0; i < n + 1; ++i) inv_post(post(i)) = i;// inv_post = post.inverse()??? + + // Renumber etree in postorder + IndexVector iwork(n); + IndexVector et_save(n + 1); + for (Index i = 0; i < n; ++i) { iwork(post(i)) = post(et(i)); } + et_save = et;// Save the original etree + et = iwork; + + // compute the number of descendants of each node in the etree + relax_end.setConstant(emptyIdxLU); + Index j, parent; + descendants.setZero(); + for (j = 0; j < n; j++) { parent = et(j); + if (parent != n)// not the dummy root + descendants(parent) += descendants(j) + 1; } - // Found a supernode in postordered etree, j is the last column - ++nsuper_et_post; - k = StorageIndex(n); - for (Index i = snode_start; i <= j; ++i) - k = (std::min)(k, inv_post(i)); - l = inv_post(j); - if ( (l - k) == (j - snode_start) ) // Same number of columns in the snode - { - // This is also a supernode in the original etree - relax_end(k) = l; // Record last column - ++nsuper_et; - } - else - { - for (Index i = snode_start; i <= j; ++i) + // Identify the relaxed supernodes by postorder traversal of the etree + Index snode_start;// beginning of a snode + StorageIndex k; + Index nsuper_et_post = 0;// Number of relaxed snodes in postordered etree + Index nsuper_et = 0;// Number of relaxed snodes in the original etree + StorageIndex l; + for (j = 0; j < n;) { + parent = et(j); + snode_start = j; + while (parent != n && descendants(parent) < relax_columns) { + j = parent; + parent = et(j); + } + // Found a supernode in postordered etree, j is the last column + ++nsuper_et_post; + k = StorageIndex(n); + for (Index i = snode_start; i <= j; ++i) k = (std::min)(k, inv_post(i)); + l = inv_post(j); + if ((l - k) == (j - snode_start))// Same number of columns in the snode { - l = inv_post(i); - if (descendants(i) == 0) - { - relax_end(l) = l; - ++nsuper_et; + // This is also a supernode in the original etree + relax_end(k) = l;// Record last column + ++nsuper_et; + } else { + for (Index i = snode_start; i <= j; ++i) { + l = inv_post(i); + if (descendants(i) == 0) { + relax_end(l) = l; + ++nsuper_et; + } } } - } - j++; - // Search for a new leaf - while (descendants(j) != 0 && j < n) j++; - } // End postorder traversal of the etree - - // Recover the original etree - et = et_save; -} + j++; + // Search for a new leaf + while (descendants(j) != 0 && j < n) j++; + }// End postorder traversal of the etree + + // Recover the original etree + et = et_save; + } -} // end namespace internal +}// end namespace internal -} // end namespace Eigen -#endif // SPARSELU_HEAP_RELAX_SNODE_H +}// end namespace Eigen +#endif// SPARSELU_HEAP_RELAX_SNODE_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/SparseLU/SparseLU_kernel_bmod.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/SparseLU/SparseLU_kernel_bmod.h index 8c1b3e8b..a792f869 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/SparseLU/SparseLU_kernel_bmod.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/SparseLU/SparseLU_kernel_bmod.h @@ -13,118 +13,149 @@ namespace Eigen { namespace internal { - -template struct LU_kernel_bmod -{ - /** \internal - * \brief Performs numeric block updates from a given supernode to a single column - * - * \param segsize Size of the segment (and blocks ) to use for updates - * \param[in,out] dense Packed values of the original matrix - * \param tempv temporary vector to use for updates - * \param lusup array containing the supernodes - * \param lda Leading dimension in the supernode - * \param nrow Number of rows in the rectangular part of the supernode - * \param lsub compressed row subscripts of supernodes - * \param lptr pointer to the first column of the current supernode in lsub - * \param no_zeros Number of nonzeros elements before the diagonal part of the supernode - */ - template - static EIGEN_DONT_INLINE void run(const Index segsize, BlockScalarVector& dense, ScalarVector& tempv, ScalarVector& lusup, Index& luptr, const Index lda, - const Index nrow, IndexVector& lsub, const Index lptr, const Index no_zeros); -}; -template -template -EIGEN_DONT_INLINE void LU_kernel_bmod::run(const Index segsize, BlockScalarVector& dense, ScalarVector& tempv, ScalarVector& lusup, Index& luptr, const Index lda, - const Index nrow, IndexVector& lsub, const Index lptr, const Index no_zeros) -{ - typedef typename ScalarVector::Scalar Scalar; - // First, copy U[*,j] segment from dense(*) to tempv(*) - // The result of triangular solve is in tempv[*]; - // The result of matric-vector update is in dense[*] - Index isub = lptr + no_zeros; - Index i; - Index irow; - for (i = 0; i < ((SegSizeAtCompileTime==Dynamic)?segsize:SegSizeAtCompileTime); i++) + template struct LU_kernel_bmod { - irow = lsub(isub); - tempv(i) = dense(irow); - ++isub; - } - // Dense triangular solve -- start effective triangle - luptr += lda * no_zeros + no_zeros; - // Form Eigen matrix and vector - Map, 0, OuterStride<> > A( &(lusup.data()[luptr]), segsize, segsize, OuterStride<>(lda) ); - Map > u(tempv.data(), segsize); - - u = A.template triangularView().solve(u); - - // Dense matrix-vector product y <-- B*x - luptr += segsize; - const Index PacketSize = internal::packet_traits::size; - Index ldl = internal::first_multiple(nrow, PacketSize); - Map, 0, OuterStride<> > B( &(lusup.data()[luptr]), nrow, segsize, OuterStride<>(lda) ); - Index aligned_offset = internal::first_default_aligned(tempv.data()+segsize, PacketSize); - Index aligned_with_B_offset = (PacketSize-internal::first_default_aligned(B.data(), PacketSize))%PacketSize; - Map, 0, OuterStride<> > l(tempv.data()+segsize+aligned_offset+aligned_with_B_offset, nrow, OuterStride<>(ldl) ); - - l.setZero(); - internal::sparselu_gemm(l.rows(), l.cols(), B.cols(), B.data(), B.outerStride(), u.data(), u.outerStride(), l.data(), l.outerStride()); - - // Scatter tempv[] into SPA dense[] as a temporary storage - isub = lptr + no_zeros; - for (i = 0; i < ((SegSizeAtCompileTime==Dynamic)?segsize:SegSizeAtCompileTime); i++) + /** \internal + * \brief Performs numeric block updates from a given supernode to a single column + * + * \param segsize Size of the segment (and blocks ) to use for updates + * \param[in,out] dense Packed values of the original matrix + * \param tempv temporary vector to use for updates + * \param lusup array containing the supernodes + * \param lda Leading dimension in the supernode + * \param nrow Number of rows in the rectangular part of the supernode + * \param lsub compressed row subscripts of supernodes + * \param lptr pointer to the first column of the current supernode in lsub + * \param no_zeros Number of nonzeros elements before the diagonal part of the supernode + */ + template + static EIGEN_DONT_INLINE void run(const Index segsize, + BlockScalarVector &dense, + ScalarVector &tempv, + ScalarVector &lusup, + Index &luptr, + const Index lda, + const Index nrow, + IndexVector &lsub, + const Index lptr, + const Index no_zeros); + }; + + template + template + EIGEN_DONT_INLINE void LU_kernel_bmod::run(const Index segsize, + BlockScalarVector &dense, + ScalarVector &tempv, + ScalarVector &lusup, + Index &luptr, + const Index lda, + const Index nrow, + IndexVector &lsub, + const Index lptr, + const Index no_zeros) { - irow = lsub(isub++); - dense(irow) = tempv(i); + typedef typename ScalarVector::Scalar Scalar; + // First, copy U[*,j] segment from dense(*) to tempv(*) + // The result of triangular solve is in tempv[*]; + // The result of matric-vector update is in dense[*] + Index isub = lptr + no_zeros; + Index i; + Index irow; + for (i = 0; i < ((SegSizeAtCompileTime == Dynamic) ? segsize : SegSizeAtCompileTime); i++) { + irow = lsub(isub); + tempv(i) = dense(irow); + ++isub; + } + // Dense triangular solve -- start effective triangle + luptr += lda * no_zeros + no_zeros; + // Form Eigen matrix and vector + Map, 0, OuterStride<>> A( + &(lusup.data()[luptr]), segsize, segsize, OuterStride<>(lda)); + Map> u(tempv.data(), segsize); + + u = A.template triangularView().solve(u); + + // Dense matrix-vector product y <-- B*x + luptr += segsize; + const Index PacketSize = internal::packet_traits::size; + Index ldl = internal::first_multiple(nrow, PacketSize); + Map, 0, OuterStride<>> B( + &(lusup.data()[luptr]), nrow, segsize, OuterStride<>(lda)); + Index aligned_offset = internal::first_default_aligned(tempv.data() + segsize, PacketSize); + Index aligned_with_B_offset = (PacketSize - internal::first_default_aligned(B.data(), PacketSize)) % PacketSize; + Map, 0, OuterStride<>> l( + tempv.data() + segsize + aligned_offset + aligned_with_B_offset, nrow, OuterStride<>(ldl)); + + l.setZero(); + internal::sparselu_gemm( + l.rows(), l.cols(), B.cols(), B.data(), B.outerStride(), u.data(), u.outerStride(), l.data(), l.outerStride()); + + // Scatter tempv[] into SPA dense[] as a temporary storage + isub = lptr + no_zeros; + for (i = 0; i < ((SegSizeAtCompileTime == Dynamic) ? segsize : SegSizeAtCompileTime); i++) { + irow = lsub(isub++); + dense(irow) = tempv(i); + } + + // Scatter l into SPA dense[] + for (i = 0; i < nrow; i++) { + irow = lsub(isub++); + dense(irow) -= l(i); + } } - - // Scatter l into SPA dense[] - for (i = 0; i < nrow; i++) - { - irow = lsub(isub++); - dense(irow) -= l(i); - } -} -template <> struct LU_kernel_bmod<1> -{ - template - static EIGEN_DONT_INLINE void run(const Index /*segsize*/, BlockScalarVector& dense, ScalarVector& /*tempv*/, ScalarVector& lusup, Index& luptr, - const Index lda, const Index nrow, IndexVector& lsub, const Index lptr, const Index no_zeros); -}; + template<> struct LU_kernel_bmod<1> + { + template + static EIGEN_DONT_INLINE void run(const Index /*segsize*/, + BlockScalarVector &dense, + ScalarVector & /*tempv*/, + ScalarVector &lusup, + Index &luptr, + const Index lda, + const Index nrow, + IndexVector &lsub, + const Index lptr, + const Index no_zeros); + }; -template -EIGEN_DONT_INLINE void LU_kernel_bmod<1>::run(const Index /*segsize*/, BlockScalarVector& dense, ScalarVector& /*tempv*/, ScalarVector& lusup, Index& luptr, - const Index lda, const Index nrow, IndexVector& lsub, const Index lptr, const Index no_zeros) -{ - typedef typename ScalarVector::Scalar Scalar; - typedef typename IndexVector::Scalar StorageIndex; - Scalar f = dense(lsub(lptr + no_zeros)); - luptr += lda * no_zeros + no_zeros + 1; - const Scalar* a(lusup.data() + luptr); - const StorageIndex* irow(lsub.data()+lptr + no_zeros + 1); - Index i = 0; - for (; i+1 < nrow; i+=2) + template + EIGEN_DONT_INLINE void LU_kernel_bmod<1>::run(const Index /*segsize*/, + BlockScalarVector &dense, + ScalarVector & /*tempv*/, + ScalarVector &lusup, + Index &luptr, + const Index lda, + const Index nrow, + IndexVector &lsub, + const Index lptr, + const Index no_zeros) { - Index i0 = *(irow++); - Index i1 = *(irow++); - Scalar a0 = *(a++); - Scalar a1 = *(a++); - Scalar d0 = dense.coeff(i0); - Scalar d1 = dense.coeff(i1); - d0 -= f*a0; - d1 -= f*a1; - dense.coeffRef(i0) = d0; - dense.coeffRef(i1) = d1; + typedef typename ScalarVector::Scalar Scalar; + typedef typename IndexVector::Scalar StorageIndex; + Scalar f = dense(lsub(lptr + no_zeros)); + luptr += lda * no_zeros + no_zeros + 1; + const Scalar *a(lusup.data() + luptr); + const StorageIndex *irow(lsub.data() + lptr + no_zeros + 1); + Index i = 0; + for (; i + 1 < nrow; i += 2) { + Index i0 = *(irow++); + Index i1 = *(irow++); + Scalar a0 = *(a++); + Scalar a1 = *(a++); + Scalar d0 = dense.coeff(i0); + Scalar d1 = dense.coeff(i1); + d0 -= f * a0; + d1 -= f * a1; + dense.coeffRef(i0) = d0; + dense.coeffRef(i1) = d1; + } + if (i < nrow) dense.coeffRef(*(irow++)) -= f * *(a++); } - if(i -void SparseLUImpl::panel_bmod(const Index m, const Index w, const Index jcol, - const Index nseg, ScalarVector& dense, ScalarVector& tempv, - IndexVector& segrep, IndexVector& repfnz, GlobalLU_t& glu) -{ - - Index ksub,jj,nextl_col; - Index fsupc, nsupc, nsupr, nrow; - Index krep, kfnz; - Index lptr; // points to the row subscripts of a supernode - Index luptr; // ... - Index segsize,no_zeros ; - // For each nonz supernode segment of U[*,j] in topological order - Index k = nseg - 1; - const Index PacketSize = internal::packet_traits::size; - - for (ksub = 0; ksub < nseg; ksub++) - { // For each updating supernode - /* krep = representative of current k-th supernode - * fsupc = first supernodal column - * nsupc = number of columns in a supernode - * nsupr = number of rows in a supernode - */ - krep = segrep(k); k--; - fsupc = glu.xsup(glu.supno(krep)); - nsupc = krep - fsupc + 1; - nsupr = glu.xlsub(fsupc+1) - glu.xlsub(fsupc); - nrow = nsupr - nsupc; - lptr = glu.xlsub(fsupc); - - // loop over the panel columns to detect the actual number of columns and rows - Index u_rows = 0; - Index u_cols = 0; - for (jj = jcol; jj < jcol + w; jj++) - { - nextl_col = (jj-jcol) * m; - VectorBlock repfnz_col(repfnz, nextl_col, m); // First nonzero column index for each row - - kfnz = repfnz_col(krep); - if ( kfnz == emptyIdxLU ) - continue; // skip any zero segment - - segsize = krep - kfnz + 1; - u_cols++; - u_rows = (std::max)(segsize,u_rows); - } - - if(nsupc >= 2) - { - Index ldu = internal::first_multiple(u_rows, PacketSize); - Map > U(tempv.data(), u_rows, u_cols, OuterStride<>(ldu)); - - // gather U - Index u_col = 0; - for (jj = jcol; jj < jcol + w; jj++) - { - nextl_col = (jj-jcol) * m; - VectorBlock repfnz_col(repfnz, nextl_col, m); // First nonzero column index for each row - VectorBlock dense_col(dense, nextl_col, m); // Scatter/gather entire matrix column from/to here - - kfnz = repfnz_col(krep); - if ( kfnz == emptyIdxLU ) - continue; // skip any zero segment - + /** + * \brief Performs numeric block updates (sup-panel) in topological order. + * + * Before entering this routine, the original nonzeros in the panel + * were already copied i nto the spa[m,w] + * + * \param m number of rows in the matrix + * \param w Panel size + * \param jcol Starting column of the panel + * \param nseg Number of segments in the U part + * \param dense Store the full representation of the panel + * \param tempv working array + * \param segrep segment representative... first row in the segment + * \param repfnz First nonzero rows + * \param glu Global LU data. + * + * + */ + template + void SparseLUImpl::panel_bmod(const Index m, + const Index w, + const Index jcol, + const Index nseg, + ScalarVector &dense, + ScalarVector &tempv, + IndexVector &segrep, + IndexVector &repfnz, + GlobalLU_t &glu) + { + + Index ksub, jj, nextl_col; + Index fsupc, nsupc, nsupr, nrow; + Index krep, kfnz; + Index lptr;// points to the row subscripts of a supernode + Index luptr;// ... + Index segsize, no_zeros; + // For each nonz supernode segment of U[*,j] in topological order + Index k = nseg - 1; + const Index PacketSize = internal::packet_traits::size; + + for (ksub = 0; ksub < nseg; ksub++) {// For each updating supernode + /* krep = representative of current k-th supernode + * fsupc = first supernodal column + * nsupc = number of columns in a supernode + * nsupr = number of rows in a supernode + */ + krep = segrep(k); + k--; + fsupc = glu.xsup(glu.supno(krep)); + nsupc = krep - fsupc + 1; + nsupr = glu.xlsub(fsupc + 1) - glu.xlsub(fsupc); + nrow = nsupr - nsupc; + lptr = glu.xlsub(fsupc); + + // loop over the panel columns to detect the actual number of columns and rows + Index u_rows = 0; + Index u_cols = 0; + for (jj = jcol; jj < jcol + w; jj++) { + nextl_col = (jj - jcol) * m; + VectorBlock repfnz_col(repfnz, nextl_col, m);// First nonzero column index for each row + + kfnz = repfnz_col(krep); + if (kfnz == emptyIdxLU) continue;// skip any zero segment + segsize = krep - kfnz + 1; - luptr = glu.xlusup(fsupc); - no_zeros = kfnz - fsupc; - - Index isub = lptr + no_zeros; - Index off = u_rows-segsize; - for (Index i = 0; i < off; i++) U(i,u_col) = 0; - for (Index i = 0; i < segsize; i++) - { - Index irow = glu.lsub(isub); - U(i+off,u_col) = dense_col(irow); - ++isub; - } - u_col++; + u_cols++; + u_rows = (std::max)(segsize, u_rows); } - // solve U = A^-1 U - luptr = glu.xlusup(fsupc); - Index lda = glu.xlusup(fsupc+1) - glu.xlusup(fsupc); - no_zeros = (krep - u_rows + 1) - fsupc; - luptr += lda * no_zeros + no_zeros; - MappedMatrixBlock A(glu.lusup.data()+luptr, u_rows, u_rows, OuterStride<>(lda) ); - U = A.template triangularView().solve(U); - - // update - luptr += u_rows; - MappedMatrixBlock B(glu.lusup.data()+luptr, nrow, u_rows, OuterStride<>(lda) ); - eigen_assert(tempv.size()>w*ldu + nrow*w + 1); - - Index ldl = internal::first_multiple(nrow, PacketSize); - Index offset = (PacketSize-internal::first_default_aligned(B.data(), PacketSize)) % PacketSize; - MappedMatrixBlock L(tempv.data()+w*ldu+offset, nrow, u_cols, OuterStride<>(ldl)); - - L.setZero(); - internal::sparselu_gemm(L.rows(), L.cols(), B.cols(), B.data(), B.outerStride(), U.data(), U.outerStride(), L.data(), L.outerStride()); - - // scatter U and L - u_col = 0; - for (jj = jcol; jj < jcol + w; jj++) - { - nextl_col = (jj-jcol) * m; - VectorBlock repfnz_col(repfnz, nextl_col, m); // First nonzero column index for each row - VectorBlock dense_col(dense, nextl_col, m); // Scatter/gather entire matrix column from/to here - - kfnz = repfnz_col(krep); - if ( kfnz == emptyIdxLU ) - continue; // skip any zero segment - - segsize = krep - kfnz + 1; - no_zeros = kfnz - fsupc; - Index isub = lptr + no_zeros; - - Index off = u_rows-segsize; - for (Index i = 0; i < segsize; i++) - { - Index irow = glu.lsub(isub++); - dense_col(irow) = U.coeff(i+off,u_col); - U.coeffRef(i+off,u_col) = 0; + + if (nsupc >= 2) { + Index ldu = internal::first_multiple(u_rows, PacketSize); + Map> U(tempv.data(), u_rows, u_cols, OuterStride<>(ldu)); + + // gather U + Index u_col = 0; + for (jj = jcol; jj < jcol + w; jj++) { + nextl_col = (jj - jcol) * m; + VectorBlock repfnz_col(repfnz, nextl_col, m);// First nonzero column index for each row + VectorBlock dense_col(dense, nextl_col, m);// Scatter/gather entire matrix column from/to here + + kfnz = repfnz_col(krep); + if (kfnz == emptyIdxLU) continue;// skip any zero segment + + segsize = krep - kfnz + 1; + luptr = glu.xlusup(fsupc); + no_zeros = kfnz - fsupc; + + Index isub = lptr + no_zeros; + Index off = u_rows - segsize; + for (Index i = 0; i < off; i++) U(i, u_col) = 0; + for (Index i = 0; i < segsize; i++) { + Index irow = glu.lsub(isub); + U(i + off, u_col) = dense_col(irow); + ++isub; + } + u_col++; } - - // Scatter l into SPA dense[] - for (Index i = 0; i < nrow; i++) - { - Index irow = glu.lsub(isub++); - dense_col(irow) -= L.coeff(i,u_col); - L.coeffRef(i,u_col) = 0; + // solve U = A^-1 U + luptr = glu.xlusup(fsupc); + Index lda = glu.xlusup(fsupc + 1) - glu.xlusup(fsupc); + no_zeros = (krep - u_rows + 1) - fsupc; + luptr += lda * no_zeros + no_zeros; + MappedMatrixBlock A(glu.lusup.data() + luptr, u_rows, u_rows, OuterStride<>(lda)); + U = A.template triangularView().solve(U); + + // update + luptr += u_rows; + MappedMatrixBlock B(glu.lusup.data() + luptr, nrow, u_rows, OuterStride<>(lda)); + eigen_assert(tempv.size() > w * ldu + nrow * w + 1); + + Index ldl = internal::first_multiple(nrow, PacketSize); + Index offset = (PacketSize - internal::first_default_aligned(B.data(), PacketSize)) % PacketSize; + MappedMatrixBlock L(tempv.data() + w * ldu + offset, nrow, u_cols, OuterStride<>(ldl)); + + L.setZero(); + internal::sparselu_gemm(L.rows(), + L.cols(), + B.cols(), + B.data(), + B.outerStride(), + U.data(), + U.outerStride(), + L.data(), + L.outerStride()); + + // scatter U and L + u_col = 0; + for (jj = jcol; jj < jcol + w; jj++) { + nextl_col = (jj - jcol) * m; + VectorBlock repfnz_col(repfnz, nextl_col, m);// First nonzero column index for each row + VectorBlock dense_col(dense, nextl_col, m);// Scatter/gather entire matrix column from/to here + + kfnz = repfnz_col(krep); + if (kfnz == emptyIdxLU) continue;// skip any zero segment + + segsize = krep - kfnz + 1; + no_zeros = kfnz - fsupc; + Index isub = lptr + no_zeros; + + Index off = u_rows - segsize; + for (Index i = 0; i < segsize; i++) { + Index irow = glu.lsub(isub++); + dense_col(irow) = U.coeff(i + off, u_col); + U.coeffRef(i + off, u_col) = 0; + } + + // Scatter l into SPA dense[] + for (Index i = 0; i < nrow; i++) { + Index irow = glu.lsub(isub++); + dense_col(irow) -= L.coeff(i, u_col); + L.coeffRef(i, u_col) = 0; + } + u_col++; } - u_col++; - } - } - else // level 2 only - { - // Sequence through each column in the panel - for (jj = jcol; jj < jcol + w; jj++) + } else// level 2 only { - nextl_col = (jj-jcol) * m; - VectorBlock repfnz_col(repfnz, nextl_col, m); // First nonzero column index for each row - VectorBlock dense_col(dense, nextl_col, m); // Scatter/gather entire matrix column from/to here - - kfnz = repfnz_col(krep); - if ( kfnz == emptyIdxLU ) - continue; // skip any zero segment - - segsize = krep - kfnz + 1; - luptr = glu.xlusup(fsupc); - - Index lda = glu.xlusup(fsupc+1)-glu.xlusup(fsupc);// nsupr - - // Perform a trianglar solve and block update, - // then scatter the result of sup-col update to dense[] - no_zeros = kfnz - fsupc; - if(segsize==1) LU_kernel_bmod<1>::run(segsize, dense_col, tempv, glu.lusup, luptr, lda, nrow, glu.lsub, lptr, no_zeros); - else if(segsize==2) LU_kernel_bmod<2>::run(segsize, dense_col, tempv, glu.lusup, luptr, lda, nrow, glu.lsub, lptr, no_zeros); - else if(segsize==3) LU_kernel_bmod<3>::run(segsize, dense_col, tempv, glu.lusup, luptr, lda, nrow, glu.lsub, lptr, no_zeros); - else LU_kernel_bmod::run(segsize, dense_col, tempv, glu.lusup, luptr, lda, nrow, glu.lsub, lptr, no_zeros); - } // End for each column in the panel - } - - } // End for each updating supernode -} // end panel bmod - -} // end namespace internal - -} // end namespace Eigen - -#endif // SPARSELU_PANEL_BMOD_H + // Sequence through each column in the panel + for (jj = jcol; jj < jcol + w; jj++) { + nextl_col = (jj - jcol) * m; + VectorBlock repfnz_col(repfnz, nextl_col, m);// First nonzero column index for each row + VectorBlock dense_col(dense, nextl_col, m);// Scatter/gather entire matrix column from/to here + + kfnz = repfnz_col(krep); + if (kfnz == emptyIdxLU) continue;// skip any zero segment + + segsize = krep - kfnz + 1; + luptr = glu.xlusup(fsupc); + + Index lda = glu.xlusup(fsupc + 1) - glu.xlusup(fsupc);// nsupr + + // Perform a trianglar solve and block update, + // then scatter the result of sup-col update to dense[] + no_zeros = kfnz - fsupc; + if (segsize == 1) + LU_kernel_bmod<1>::run(segsize, dense_col, tempv, glu.lusup, luptr, lda, nrow, glu.lsub, lptr, no_zeros); + else if (segsize == 2) + LU_kernel_bmod<2>::run(segsize, dense_col, tempv, glu.lusup, luptr, lda, nrow, glu.lsub, lptr, no_zeros); + else if (segsize == 3) + LU_kernel_bmod<3>::run(segsize, dense_col, tempv, glu.lusup, luptr, lda, nrow, glu.lsub, lptr, no_zeros); + else + LU_kernel_bmod::run( + segsize, dense_col, tempv, glu.lusup, luptr, lda, nrow, glu.lsub, lptr, no_zeros); + }// End for each column in the panel + } + + }// End for each updating supernode + }// end panel bmod + +}// end namespace internal + +}// end namespace Eigen + +#endif// SPARSELU_PANEL_BMOD_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/SparseLU/SparseLU_panel_dfs.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/SparseLU/SparseLU_panel_dfs.h index 155df733..5c400dd6 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/SparseLU/SparseLU_panel_dfs.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/SparseLU/SparseLU_panel_dfs.h @@ -7,10 +7,10 @@ // Public License v. 2.0. If a copy of the MPL was not distributed // with this file, You can obtain one at http://mozilla.org/MPL/2.0/. -/* - - * NOTE: This file is the modified version of [s,d,c,z]panel_dfs.c file in SuperLU - +/* + + * NOTE: This file is the modified version of [s,d,c,z]panel_dfs.c file in SuperLU + * -- SuperLU routine (version 2.0) -- * Univ. of California Berkeley, Xerox Palo Alto Research Center, * and Lawrence Berkeley National Lab. @@ -33,226 +33,237 @@ namespace Eigen { namespace internal { - -template -struct panel_dfs_traits -{ - typedef typename IndexVector::Scalar StorageIndex; - panel_dfs_traits(Index jcol, StorageIndex* marker) - : m_jcol(jcol), m_marker(marker) - {} - bool update_segrep(Index krep, StorageIndex jj) + + template struct panel_dfs_traits { - if(m_marker[krep] -template -void SparseLUImpl::dfs_kernel(const StorageIndex jj, IndexVector& perm_r, - Index& nseg, IndexVector& panel_lsub, IndexVector& segrep, - Ref repfnz_col, IndexVector& xprune, Ref marker, IndexVector& parent, - IndexVector& xplore, GlobalLU_t& glu, - Index& nextl_col, Index krow, Traits& traits - ) -{ - - StorageIndex kmark = marker(krow); - - // For each unmarked krow of jj - marker(krow) = jj; - StorageIndex kperm = perm_r(krow); - if (kperm == emptyIdxLU ) { - // krow is in L : place it in structure of L(*, jj) - panel_lsub(nextl_col++) = StorageIndex(krow); // krow is indexed into A - - traits.mem_expand(panel_lsub, nextl_col, kmark); - } - else + void mem_expand(IndexVector & /*glu.lsub*/, Index /*nextl*/, Index /*chmark*/) {} + enum { ExpandMem = false }; + Index m_jcol; + StorageIndex *m_marker; + }; + + + template + template + void SparseLUImpl::dfs_kernel(const StorageIndex jj, + IndexVector &perm_r, + Index &nseg, + IndexVector &panel_lsub, + IndexVector &segrep, + Ref repfnz_col, + IndexVector &xprune, + Ref marker, + IndexVector &parent, + IndexVector &xplore, + GlobalLU_t &glu, + Index &nextl_col, + Index krow, + Traits &traits) { - // krow is in U : if its supernode-representative krep - // has been explored, update repfnz(*) - // krep = supernode representative of the current row - StorageIndex krep = glu.xsup(glu.supno(kperm)+1) - 1; - // First nonzero element in the current column: - StorageIndex myfnz = repfnz_col(krep); - - if (myfnz != emptyIdxLU ) - { - // Representative visited before - if (myfnz > kperm ) repfnz_col(krep) = kperm; - - } - else - { - // Otherwise, perform dfs starting at krep - StorageIndex oldrep = emptyIdxLU; - parent(krep) = oldrep; - repfnz_col(krep) = kperm; - StorageIndex xdfs = glu.xlsub(krep); - Index maxdfs = xprune(krep); - - StorageIndex kpar; - do - { - // For each unmarked kchild of krep - while (xdfs < maxdfs) - { - StorageIndex kchild = glu.lsub(xdfs); - xdfs++; - StorageIndex chmark = marker(kchild); - - if (chmark != jj ) + + StorageIndex kmark = marker(krow); + + // For each unmarked krow of jj + marker(krow) = jj; + StorageIndex kperm = perm_r(krow); + if (kperm == emptyIdxLU) { + // krow is in L : place it in structure of L(*, jj) + panel_lsub(nextl_col++) = StorageIndex(krow);// krow is indexed into A + + traits.mem_expand(panel_lsub, nextl_col, kmark); + } else { + // krow is in U : if its supernode-representative krep + // has been explored, update repfnz(*) + // krep = supernode representative of the current row + StorageIndex krep = glu.xsup(glu.supno(kperm) + 1) - 1; + // First nonzero element in the current column: + StorageIndex myfnz = repfnz_col(krep); + + if (myfnz != emptyIdxLU) { + // Representative visited before + if (myfnz > kperm) repfnz_col(krep) = kperm; + + } else { + // Otherwise, perform dfs starting at krep + StorageIndex oldrep = emptyIdxLU; + parent(krep) = oldrep; + repfnz_col(krep) = kperm; + StorageIndex xdfs = glu.xlsub(krep); + Index maxdfs = xprune(krep); + + StorageIndex kpar; + do { + // For each unmarked kchild of krep + while (xdfs < maxdfs) { + StorageIndex kchild = glu.lsub(xdfs); + xdfs++; + StorageIndex chmark = marker(kchild); + + if (chmark != jj) { + marker(kchild) = jj; + StorageIndex chperm = perm_r(kchild); + + if (chperm == emptyIdxLU) { + // case kchild is in L: place it in L(*, j) + panel_lsub(nextl_col++) = kchild; + traits.mem_expand(panel_lsub, nextl_col, chmark); + } else { + // case kchild is in U : + // chrep = its supernode-rep. If its rep has been explored, + // update its repfnz(*) + StorageIndex chrep = glu.xsup(glu.supno(chperm) + 1) - 1; + myfnz = repfnz_col(chrep); + + if (myfnz != emptyIdxLU) {// Visited before + if (myfnz > chperm) repfnz_col(chrep) = chperm; + } else {// Cont. dfs at snode-rep of kchild + xplore(krep) = xdfs; + oldrep = krep; + krep = chrep;// Go deeper down G(L) + parent(krep) = oldrep; + repfnz_col(krep) = chperm; + xdfs = glu.xlsub(krep); + maxdfs = xprune(krep); + + }// end if myfnz != -1 + }// end if chperm == -1 + + }// end if chmark !=jj + }// end while xdfs < maxdfs + + // krow has no more unexplored nbrs : + // Place snode-rep krep in postorder DFS, if this + // segment is seen for the first time. (Note that + // "repfnz(krep)" may change later.) + // Baktrack dfs to its parent + if (traits.update_segrep(krep, jj)) + // if (marker1(krep) < jcol ) { - marker(kchild) = jj; - StorageIndex chperm = perm_r(kchild); - - if (chperm == emptyIdxLU) - { - // case kchild is in L: place it in L(*, j) - panel_lsub(nextl_col++) = kchild; - traits.mem_expand(panel_lsub, nextl_col, chmark); - } - else - { - // case kchild is in U : - // chrep = its supernode-rep. If its rep has been explored, - // update its repfnz(*) - StorageIndex chrep = glu.xsup(glu.supno(chperm)+1) - 1; - myfnz = repfnz_col(chrep); - - if (myfnz != emptyIdxLU) - { // Visited before - if (myfnz > chperm) - repfnz_col(chrep) = chperm; - } - else - { // Cont. dfs at snode-rep of kchild - xplore(krep) = xdfs; - oldrep = krep; - krep = chrep; // Go deeper down G(L) - parent(krep) = oldrep; - repfnz_col(krep) = chperm; - xdfs = glu.xlsub(krep); - maxdfs = xprune(krep); - - } // end if myfnz != -1 - } // end if chperm == -1 - - } // end if chmark !=jj - } // end while xdfs < maxdfs - - // krow has no more unexplored nbrs : - // Place snode-rep krep in postorder DFS, if this - // segment is seen for the first time. (Note that - // "repfnz(krep)" may change later.) - // Baktrack dfs to its parent - if(traits.update_segrep(krep,jj)) - //if (marker1(krep) < jcol ) - { - segrep(nseg) = krep; - ++nseg; - //marker1(krep) = jj; - } - - kpar = parent(krep); // Pop recursion, mimic recursion - if (kpar == emptyIdxLU) - break; // dfs done - krep = kpar; - xdfs = xplore(krep); - maxdfs = xprune(krep); - - } while (kpar != emptyIdxLU); // Do until empty stack - - } // end if (myfnz = -1) - - } // end if (kperm == -1) -} - -/** - * \brief Performs a symbolic factorization on a panel of columns [jcol, jcol+w) - * - * A supernode representative is the last column of a supernode. - * The nonzeros in U[*,j] are segments that end at supernodes representatives - * - * The routine returns a list of the supernodal representatives - * in topological order of the dfs that generates them. This list is - * a superset of the topological order of each individual column within - * the panel. - * The location of the first nonzero in each supernodal segment - * (supernodal entry location) is also returned. Each column has - * a separate list for this purpose. - * - * Two markers arrays are used for dfs : - * marker[i] == jj, if i was visited during dfs of current column jj; - * marker1[i] >= jcol, if i was visited by earlier columns in this panel; - * - * \param[in] m number of rows in the matrix - * \param[in] w Panel size - * \param[in] jcol Starting column of the panel - * \param[in] A Input matrix in column-major storage - * \param[in] perm_r Row permutation - * \param[out] nseg Number of U segments - * \param[out] dense Accumulate the column vectors of the panel - * \param[out] panel_lsub Subscripts of the row in the panel - * \param[out] segrep Segment representative i.e first nonzero row of each segment - * \param[out] repfnz First nonzero location in each row - * \param[out] xprune The pruned elimination tree - * \param[out] marker work vector - * \param parent The elimination tree - * \param xplore work vector - * \param glu The global data structure - * - */ + segrep(nseg) = krep; + ++nseg; + // marker1(krep) = jj; + } + + kpar = parent(krep);// Pop recursion, mimic recursion + if (kpar == emptyIdxLU) break;// dfs done + krep = kpar; + xdfs = xplore(krep); + maxdfs = xprune(krep); -template -void SparseLUImpl::panel_dfs(const Index m, const Index w, const Index jcol, MatrixType& A, IndexVector& perm_r, Index& nseg, ScalarVector& dense, IndexVector& panel_lsub, IndexVector& segrep, IndexVector& repfnz, IndexVector& xprune, IndexVector& marker, IndexVector& parent, IndexVector& xplore, GlobalLU_t& glu) -{ - Index nextl_col; // Next available position in panel_lsub[*,jj] - - // Initialize pointers - VectorBlock marker1(marker, m, m); - nseg = 0; - - panel_dfs_traits traits(jcol, marker1.data()); - - // For each column in the panel - for (StorageIndex jj = StorageIndex(jcol); jj < jcol + w; jj++) + } while (kpar != emptyIdxLU);// Do until empty stack + + }// end if (myfnz = -1) + + }// end if (kperm == -1) + } + + /** + * \brief Performs a symbolic factorization on a panel of columns [jcol, jcol+w) + * + * A supernode representative is the last column of a supernode. + * The nonzeros in U[*,j] are segments that end at supernodes representatives + * + * The routine returns a list of the supernodal representatives + * in topological order of the dfs that generates them. This list is + * a superset of the topological order of each individual column within + * the panel. + * The location of the first nonzero in each supernodal segment + * (supernodal entry location) is also returned. Each column has + * a separate list for this purpose. + * + * Two markers arrays are used for dfs : + * marker[i] == jj, if i was visited during dfs of current column jj; + * marker1[i] >= jcol, if i was visited by earlier columns in this panel; + * + * \param[in] m number of rows in the matrix + * \param[in] w Panel size + * \param[in] jcol Starting column of the panel + * \param[in] A Input matrix in column-major storage + * \param[in] perm_r Row permutation + * \param[out] nseg Number of U segments + * \param[out] dense Accumulate the column vectors of the panel + * \param[out] panel_lsub Subscripts of the row in the panel + * \param[out] segrep Segment representative i.e first nonzero row of each segment + * \param[out] repfnz First nonzero location in each row + * \param[out] xprune The pruned elimination tree + * \param[out] marker work vector + * \param parent The elimination tree + * \param xplore work vector + * \param glu The global data structure + * + */ + + template + void SparseLUImpl::panel_dfs(const Index m, + const Index w, + const Index jcol, + MatrixType &A, + IndexVector &perm_r, + Index &nseg, + ScalarVector &dense, + IndexVector &panel_lsub, + IndexVector &segrep, + IndexVector &repfnz, + IndexVector &xprune, + IndexVector &marker, + IndexVector &parent, + IndexVector &xplore, + GlobalLU_t &glu) { - nextl_col = (jj - jcol) * m; - - VectorBlock repfnz_col(repfnz, nextl_col, m); // First nonzero location in each row - VectorBlock dense_col(dense,nextl_col, m); // Accumulate a column vector here - - - // For each nnz in A[*, jj] do depth first search - for (typename MatrixType::InnerIterator it(A, jj); it; ++it) - { - Index krow = it.row(); - dense_col(krow) = it.value(); - - StorageIndex kmark = marker(krow); - if (kmark == jj) - continue; // krow visited before, go to the next nonzero - - dfs_kernel(jj, perm_r, nseg, panel_lsub, segrep, repfnz_col, xprune, marker, parent, - xplore, glu, nextl_col, krow, traits); - }// end for nonzeros in column jj - - } // end for column jj -} - -} // end namespace internal -} // end namespace Eigen - -#endif // SPARSELU_PANEL_DFS_H + Index nextl_col;// Next available position in panel_lsub[*,jj] + + // Initialize pointers + VectorBlock marker1(marker, m, m); + nseg = 0; + + panel_dfs_traits traits(jcol, marker1.data()); + + // For each column in the panel + for (StorageIndex jj = StorageIndex(jcol); jj < jcol + w; jj++) { + nextl_col = (jj - jcol) * m; + + VectorBlock repfnz_col(repfnz, nextl_col, m);// First nonzero location in each row + VectorBlock dense_col(dense, nextl_col, m);// Accumulate a column vector here + + + // For each nnz in A[*, jj] do depth first search + for (typename MatrixType::InnerIterator it(A, jj); it; ++it) { + Index krow = it.row(); + dense_col(krow) = it.value(); + + StorageIndex kmark = marker(krow); + if (kmark == jj) continue;// krow visited before, go to the next nonzero + + dfs_kernel(jj, + perm_r, + nseg, + panel_lsub, + segrep, + repfnz_col, + xprune, + marker, + parent, + xplore, + glu, + nextl_col, + krow, + traits); + }// end for nonzeros in column jj + + }// end for column jj + } + +}// end namespace internal +}// end namespace Eigen + +#endif// SPARSELU_PANEL_DFS_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/SparseLU/SparseLU_pivotL.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/SparseLU/SparseLU_pivotL.h index a86dac93..0c40a42e 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/SparseLU/SparseLU_pivotL.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/SparseLU/SparseLU_pivotL.h @@ -7,10 +7,10 @@ // Public License v. 2.0. If a copy of the MPL was not distributed // with this file, You can obtain one at http://mozilla.org/MPL/2.0/. -/* - - * NOTE: This file is the modified version of xpivotL.c file in SuperLU - +/* + + * NOTE: This file is the modified version of xpivotL.c file in SuperLU + * -- SuperLU routine (version 3.0) -- * Univ. of California Berkeley, Xerox Palo Alto Research Center, * and Lawrence Berkeley National Lab. @@ -32,106 +32,107 @@ namespace Eigen { namespace internal { - -/** - * \brief Performs the numerical pivotin on the current column of L, and the CDIV operation. - * - * Pivot policy : - * (1) Compute thresh = u * max_(i>=j) abs(A_ij); - * (2) IF user specifies pivot row k and abs(A_kj) >= thresh THEN - * pivot row = k; - * ELSE IF abs(A_jj) >= thresh THEN - * pivot row = j; - * ELSE - * pivot row = m; - * - * Note: If you absolutely want to use a given pivot order, then set u=0.0. - * - * \param jcol The current column of L - * \param diagpivotthresh diagonal pivoting threshold - * \param[in,out] perm_r Row permutation (threshold pivoting) - * \param[in] iperm_c column permutation - used to finf diagonal of Pc*A*Pc' - * \param[out] pivrow The pivot row - * \param glu Global LU data - * \return 0 if success, i > 0 if U(i,i) is exactly zero - * - */ -template -Index SparseLUImpl::pivotL(const Index jcol, const RealScalar& diagpivotthresh, IndexVector& perm_r, IndexVector& iperm_c, Index& pivrow, GlobalLU_t& glu) -{ - - Index fsupc = (glu.xsup)((glu.supno)(jcol)); // First column in the supernode containing the column jcol - Index nsupc = jcol - fsupc; // Number of columns in the supernode portion, excluding jcol; nsupc >=0 - Index lptr = glu.xlsub(fsupc); // pointer to the starting location of the row subscripts for this supernode portion - Index nsupr = glu.xlsub(fsupc+1) - lptr; // Number of rows in the supernode - Index lda = glu.xlusup(fsupc+1) - glu.xlusup(fsupc); // leading dimension - Scalar* lu_sup_ptr = &(glu.lusup.data()[glu.xlusup(fsupc)]); // Start of the current supernode - Scalar* lu_col_ptr = &(glu.lusup.data()[glu.xlusup(jcol)]); // Start of jcol in the supernode - StorageIndex* lsub_ptr = &(glu.lsub.data()[lptr]); // Start of row indices of the supernode - - // Determine the largest abs numerical value for partial pivoting - Index diagind = iperm_c(jcol); // diagonal index - RealScalar pivmax(-1.0); - Index pivptr = nsupc; - Index diag = emptyIdxLU; - RealScalar rtemp; - Index isub, icol, itemp, k; - for (isub = nsupc; isub < nsupr; ++isub) { - using std::abs; - rtemp = abs(lu_col_ptr[isub]); - if (rtemp > pivmax) { - pivmax = rtemp; - pivptr = isub; - } - if (lsub_ptr[isub] == diagind) diag = isub; - } - - // Test for singularity - if ( pivmax <= RealScalar(0.0) ) { - // if pivmax == -1, the column is structurally empty, otherwise it is only numerically zero - pivrow = pivmax < RealScalar(0.0) ? diagind : lsub_ptr[pivptr]; - perm_r(pivrow) = StorageIndex(jcol); - return (jcol+1); - } - - RealScalar thresh = diagpivotthresh * pivmax; - - // Choose appropriate pivotal element - + + /** + * \brief Performs the numerical pivotin on the current column of L, and the CDIV operation. + * + * Pivot policy : + * (1) Compute thresh = u * max_(i>=j) abs(A_ij); + * (2) IF user specifies pivot row k and abs(A_kj) >= thresh THEN + * pivot row = k; + * ELSE IF abs(A_jj) >= thresh THEN + * pivot row = j; + * ELSE + * pivot row = m; + * + * Note: If you absolutely want to use a given pivot order, then set u=0.0. + * + * \param jcol The current column of L + * \param diagpivotthresh diagonal pivoting threshold + * \param[in,out] perm_r Row permutation (threshold pivoting) + * \param[in] iperm_c column permutation - used to finf diagonal of Pc*A*Pc' + * \param[out] pivrow The pivot row + * \param glu Global LU data + * \return 0 if success, i > 0 if U(i,i) is exactly zero + * + */ + template + Index SparseLUImpl::pivotL(const Index jcol, + const RealScalar &diagpivotthresh, + IndexVector &perm_r, + IndexVector &iperm_c, + Index &pivrow, + GlobalLU_t &glu) { - // Test if the diagonal element can be used as a pivot (given the threshold value) - if (diag >= 0 ) - { - // Diagonal element exists + + Index fsupc = (glu.xsup)((glu.supno)(jcol));// First column in the supernode containing the column jcol + Index nsupc = jcol - fsupc;// Number of columns in the supernode portion, excluding jcol; nsupc >=0 + Index lptr = glu.xlsub(fsupc);// pointer to the starting location of the row subscripts for this supernode portion + Index nsupr = glu.xlsub(fsupc + 1) - lptr;// Number of rows in the supernode + Index lda = glu.xlusup(fsupc + 1) - glu.xlusup(fsupc);// leading dimension + Scalar *lu_sup_ptr = &(glu.lusup.data()[glu.xlusup(fsupc)]);// Start of the current supernode + Scalar *lu_col_ptr = &(glu.lusup.data()[glu.xlusup(jcol)]);// Start of jcol in the supernode + StorageIndex *lsub_ptr = &(glu.lsub.data()[lptr]);// Start of row indices of the supernode + + // Determine the largest abs numerical value for partial pivoting + Index diagind = iperm_c(jcol);// diagonal index + RealScalar pivmax(-1.0); + Index pivptr = nsupc; + Index diag = emptyIdxLU; + RealScalar rtemp; + Index isub, icol, itemp, k; + for (isub = nsupc; isub < nsupr; ++isub) { using std::abs; - rtemp = abs(lu_col_ptr[diag]); - if (rtemp != RealScalar(0.0) && rtemp >= thresh) pivptr = diag; + rtemp = abs(lu_col_ptr[isub]); + if (rtemp > pivmax) { + pivmax = rtemp; + pivptr = isub; + } + if (lsub_ptr[isub] == diagind) diag = isub; } - pivrow = lsub_ptr[pivptr]; - } - - // Record pivot row - perm_r(pivrow) = StorageIndex(jcol); - // Interchange row subscripts - if (pivptr != nsupc ) - { - std::swap( lsub_ptr[pivptr], lsub_ptr[nsupc] ); - // Interchange numerical values as well, for the two rows in the whole snode - // such that L is indexed the same way as A - for (icol = 0; icol <= nsupc; icol++) + + // Test for singularity + if (pivmax <= RealScalar(0.0)) { + // if pivmax == -1, the column is structurally empty, otherwise it is only numerically zero + pivrow = pivmax < RealScalar(0.0) ? diagind : lsub_ptr[pivptr]; + perm_r(pivrow) = StorageIndex(jcol); + return (jcol + 1); + } + + RealScalar thresh = diagpivotthresh * pivmax; + + // Choose appropriate pivotal element + { - itemp = pivptr + icol * lda; - std::swap(lu_sup_ptr[itemp], lu_sup_ptr[nsupc + icol * lda]); + // Test if the diagonal element can be used as a pivot (given the threshold value) + if (diag >= 0) { + // Diagonal element exists + using std::abs; + rtemp = abs(lu_col_ptr[diag]); + if (rtemp != RealScalar(0.0) && rtemp >= thresh) pivptr = diag; + } + pivrow = lsub_ptr[pivptr]; + } + + // Record pivot row + perm_r(pivrow) = StorageIndex(jcol); + // Interchange row subscripts + if (pivptr != nsupc) { + std::swap(lsub_ptr[pivptr], lsub_ptr[nsupc]); + // Interchange numerical values as well, for the two rows in the whole snode + // such that L is indexed the same way as A + for (icol = 0; icol <= nsupc; icol++) { + itemp = pivptr + icol * lda; + std::swap(lu_sup_ptr[itemp], lu_sup_ptr[nsupc + icol * lda]); + } } + // cdiv operations + Scalar temp = Scalar(1.0) / lu_col_ptr[nsupc]; + for (k = nsupc + 1; k < nsupr; k++) lu_col_ptr[k] *= temp; + return 0; } - // cdiv operations - Scalar temp = Scalar(1.0) / lu_col_ptr[nsupc]; - for (k = nsupc+1; k < nsupr; k++) - lu_col_ptr[k] *= temp; - return 0; -} -} // end namespace internal -} // end namespace Eigen +}// end namespace internal +}// end namespace Eigen -#endif // SPARSELU_PIVOTL_H +#endif// SPARSELU_PIVOTL_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/SparseLU/SparseLU_pruneL.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/SparseLU/SparseLU_pruneL.h index ad32fed5..0aefa439 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/SparseLU/SparseLU_pruneL.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/SparseLU/SparseLU_pruneL.h @@ -7,10 +7,10 @@ // Public License v. 2.0. If a copy of the MPL was not distributed // with this file, You can obtain one at http://mozilla.org/MPL/2.0/. -/* - - * NOTE: This file is the modified version of [s,d,c,z]pruneL.c file in SuperLU - +/* + + * NOTE: This file is the modified version of [s,d,c,z]pruneL.c file in SuperLU + * -- SuperLU routine (version 2.0) -- * Univ. of California Berkeley, Xerox Palo Alto Research Center, * and Lawrence Berkeley National Lab. @@ -33,104 +33,101 @@ namespace Eigen { namespace internal { -/** - * \brief Prunes the L-structure. - * - * It prunes the L-structure of supernodes whose L-structure contains the current pivot row "pivrow" - * - * - * \param jcol The current column of L - * \param[in] perm_r Row permutation - * \param[out] pivrow The pivot row - * \param nseg Number of segments - * \param segrep - * \param repfnz - * \param[out] xprune - * \param glu Global LU data - * - */ -template -void SparseLUImpl::pruneL(const Index jcol, const IndexVector& perm_r, const Index pivrow, const Index nseg, - const IndexVector& segrep, BlockIndexVector repfnz, IndexVector& xprune, GlobalLU_t& glu) -{ - // For each supernode-rep irep in U(*,j] - Index jsupno = glu.supno(jcol); - Index i,irep,irep1; - bool movnum, do_prune = false; - Index kmin = 0, kmax = 0, minloc, maxloc,krow; - for (i = 0; i < nseg; i++) + /** + * \brief Prunes the L-structure. + * + * It prunes the L-structure of supernodes whose L-structure contains the current pivot row "pivrow" + * + * + * \param jcol The current column of L + * \param[in] perm_r Row permutation + * \param[out] pivrow The pivot row + * \param nseg Number of segments + * \param segrep + * \param repfnz + * \param[out] xprune + * \param glu Global LU data + * + */ + template + void SparseLUImpl::pruneL(const Index jcol, + const IndexVector &perm_r, + const Index pivrow, + const Index nseg, + const IndexVector &segrep, + BlockIndexVector repfnz, + IndexVector &xprune, + GlobalLU_t &glu) { - irep = segrep(i); - irep1 = irep + 1; - do_prune = false; - - // Don't prune with a zero U-segment - if (repfnz(irep) == emptyIdxLU) continue; - - // If a snode overlaps with the next panel, then the U-segment - // is fragmented into two parts -- irep and irep1. We should let - // pruning occur at the rep-column in irep1s snode. - if (glu.supno(irep) == glu.supno(irep1) ) continue; // don't prune - - // If it has not been pruned & it has a nonz in row L(pivrow,i) - if (glu.supno(irep) != jsupno ) - { - if ( xprune (irep) >= glu.xlsub(irep1) ) - { - kmin = glu.xlsub(irep); - kmax = glu.xlsub(irep1) - 1; - for (krow = kmin; krow <= kmax; krow++) - { - if (glu.lsub(krow) == pivrow) - { - do_prune = true; - break; + // For each supernode-rep irep in U(*,j] + Index jsupno = glu.supno(jcol); + Index i, irep, irep1; + bool movnum, do_prune = false; + Index kmin = 0, kmax = 0, minloc, maxloc, krow; + for (i = 0; i < nseg; i++) { + irep = segrep(i); + irep1 = irep + 1; + do_prune = false; + + // Don't prune with a zero U-segment + if (repfnz(irep) == emptyIdxLU) continue; + + // If a snode overlaps with the next panel, then the U-segment + // is fragmented into two parts -- irep and irep1. We should let + // pruning occur at the rep-column in irep1s snode. + if (glu.supno(irep) == glu.supno(irep1)) continue;// don't prune + + // If it has not been pruned & it has a nonz in row L(pivrow,i) + if (glu.supno(irep) != jsupno) { + if (xprune(irep) >= glu.xlsub(irep1)) { + kmin = glu.xlsub(irep); + kmax = glu.xlsub(irep1) - 1; + for (krow = kmin; krow <= kmax; krow++) { + if (glu.lsub(krow) == pivrow) { + do_prune = true; + break; + } } } - } - - if (do_prune) - { - // do a quicksort-type partition - // movnum=true means that the num values have to be exchanged - movnum = false; - if (irep == glu.xsup(glu.supno(irep)) ) // Snode of size 1 - movnum = true; - - while (kmin <= kmax) - { - if (perm_r(glu.lsub(kmax)) == emptyIdxLU) - kmax--; - else if ( perm_r(glu.lsub(kmin)) != emptyIdxLU) - kmin++; - else - { - // kmin below pivrow (not yet pivoted), and kmax - // above pivrow: interchange the two suscripts - std::swap(glu.lsub(kmin), glu.lsub(kmax)); - - // If the supernode has only one column, then we - // only keep one set of subscripts. For any subscript - // intercnahge performed, similar interchange must be - // done on the numerical values. - if (movnum) - { - minloc = glu.xlusup(irep) + ( kmin - glu.xlsub(irep) ); - maxloc = glu.xlusup(irep) + ( kmax - glu.xlsub(irep) ); - std::swap(glu.lusup(minloc), glu.lusup(maxloc)); + + if (do_prune) { + // do a quicksort-type partition + // movnum=true means that the num values have to be exchanged + movnum = false; + if (irep == glu.xsup(glu.supno(irep)))// Snode of size 1 + movnum = true; + + while (kmin <= kmax) { + if (perm_r(glu.lsub(kmax)) == emptyIdxLU) + kmax--; + else if (perm_r(glu.lsub(kmin)) != emptyIdxLU) + kmin++; + else { + // kmin below pivrow (not yet pivoted), and kmax + // above pivrow: interchange the two suscripts + std::swap(glu.lsub(kmin), glu.lsub(kmax)); + + // If the supernode has only one column, then we + // only keep one set of subscripts. For any subscript + // intercnahge performed, similar interchange must be + // done on the numerical values. + if (movnum) { + minloc = glu.xlusup(irep) + (kmin - glu.xlsub(irep)); + maxloc = glu.xlusup(irep) + (kmax - glu.xlsub(irep)); + std::swap(glu.lusup(minloc), glu.lusup(maxloc)); + } + kmin++; + kmax--; } - kmin++; - kmax--; - } - } // end while - - xprune(irep) = StorageIndex(kmin); //Pruning - } // end if do_prune - } // end pruning - } // End for each U-segment -} + }// end while + + xprune(irep) = StorageIndex(kmin);// Pruning + }// end if do_prune + }// end pruning + }// End for each U-segment + } -} // end namespace internal -} // end namespace Eigen +}// end namespace internal +}// end namespace Eigen -#endif // SPARSELU_PRUNEL_H +#endif// SPARSELU_PRUNEL_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/SparseLU/SparseLU_relax_snode.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/SparseLU/SparseLU_relax_snode.h index c408d01b..ddc11479 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/SparseLU/SparseLU_relax_snode.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/SparseLU/SparseLU_relax_snode.h @@ -31,53 +31,53 @@ namespace Eigen { namespace internal { - -/** - * \brief Identify the initial relaxed supernodes - * - * This routine is applied to a column elimination tree. - * It assumes that the matrix has been reordered according to the postorder of the etree - * \param n the number of columns - * \param et elimination tree - * \param relax_columns Maximum number of columns allowed in a relaxed snode - * \param descendants Number of descendants of each node in the etree - * \param relax_end last column in a supernode - */ -template -void SparseLUImpl::relax_snode (const Index n, IndexVector& et, const Index relax_columns, IndexVector& descendants, IndexVector& relax_end) -{ - - // compute the number of descendants of each node in the etree - Index parent; - relax_end.setConstant(emptyIdxLU); - descendants.setZero(); - for (Index j = 0; j < n; j++) - { - parent = et(j); - if (parent != n) // not the dummy root - descendants(parent) += descendants(j) + 1; - } - // Identify the relaxed supernodes by postorder traversal of the etree - Index snode_start; // beginning of a snode - for (Index j = 0; j < n; ) + + /** + * \brief Identify the initial relaxed supernodes + * + * This routine is applied to a column elimination tree. + * It assumes that the matrix has been reordered according to the postorder of the etree + * \param n the number of columns + * \param et elimination tree + * \param relax_columns Maximum number of columns allowed in a relaxed snode + * \param descendants Number of descendants of each node in the etree + * \param relax_end last column in a supernode + */ + template + void SparseLUImpl::relax_snode(const Index n, + IndexVector &et, + const Index relax_columns, + IndexVector &descendants, + IndexVector &relax_end) { - parent = et(j); - snode_start = j; - while ( parent != n && descendants(parent) < relax_columns ) - { - j = parent; + + // compute the number of descendants of each node in the etree + Index parent; + relax_end.setConstant(emptyIdxLU); + descendants.setZero(); + for (Index j = 0; j < n; j++) { parent = et(j); + if (parent != n)// not the dummy root + descendants(parent) += descendants(j) + 1; } - // Found a supernode in postordered etree, j is the last column - relax_end(snode_start) = StorageIndex(j); // Record last column - j++; - // Search for a new leaf - while (descendants(j) != 0 && j < n) j++; - } // End postorder traversal of the etree - -} + // Identify the relaxed supernodes by postorder traversal of the etree + Index snode_start;// beginning of a snode + for (Index j = 0; j < n;) { + parent = et(j); + snode_start = j; + while (parent != n && descendants(parent) < relax_columns) { + j = parent; + parent = et(j); + } + // Found a supernode in postordered etree, j is the last column + relax_end(snode_start) = StorageIndex(j);// Record last column + j++; + // Search for a new leaf + while (descendants(j) != 0 && j < n) j++; + }// End postorder traversal of the etree + } -} // end namespace internal +}// end namespace internal -} // end namespace Eigen +}// end namespace Eigen #endif diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/SparseQR/SparseQR.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/SparseQR/SparseQR.h index 7409fcae..79738c33 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/SparseQR/SparseQR.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/SparseQR/SparseQR.h @@ -18,561 +18,534 @@ template struct SparseQRMatrixQReturnType; template struct SparseQRMatrixQTransposeReturnType; template struct SparseQR_QProduct; namespace internal { - template struct traits > + template struct traits> { typedef typename SparseQRType::MatrixType ReturnType; typedef typename ReturnType::StorageIndex StorageIndex; typedef typename ReturnType::StorageKind StorageKind; - enum { - RowsAtCompileTime = Dynamic, - ColsAtCompileTime = Dynamic - }; + enum { RowsAtCompileTime = Dynamic, ColsAtCompileTime = Dynamic }; }; - template struct traits > + template struct traits> { typedef typename SparseQRType::MatrixType ReturnType; }; - template struct traits > + template struct traits> { typedef typename Derived::PlainObject ReturnType; }; -} // End namespace internal +}// End namespace internal /** - * \ingroup SparseQR_Module - * \class SparseQR - * \brief Sparse left-looking rank-revealing QR factorization - * - * This class implements a left-looking rank-revealing QR decomposition - * of sparse matrices. When a column has a norm less than a given tolerance - * it is implicitly permuted to the end. The QR factorization thus obtained is - * given by A*P = Q*R where R is upper triangular or trapezoidal. - * - * P is the column permutation which is the product of the fill-reducing and the - * rank-revealing permutations. Use colsPermutation() to get it. - * - * Q is the orthogonal matrix represented as products of Householder reflectors. - * Use matrixQ() to get an expression and matrixQ().adjoint() to get the adjoint. - * You can then apply it to a vector. - * - * R is the sparse triangular or trapezoidal matrix. The later occurs when A is rank-deficient. - * matrixR().topLeftCorner(rank(), rank()) always returns a triangular factor of full rank. - * - * \tparam _MatrixType The type of the sparse matrix A, must be a column-major SparseMatrix<> - * \tparam _OrderingType The fill-reducing ordering method. See the \link OrderingMethods_Module - * OrderingMethods \endlink module for the list of built-in and external ordering methods. - * - * \implsparsesolverconcept - * - * \warning The input sparse matrix A must be in compressed mode (see SparseMatrix::makeCompressed()). - * \warning For complex matrices matrixQ().transpose() will actually return the adjoint matrix. - * - */ + * \ingroup SparseQR_Module + * \class SparseQR + * \brief Sparse left-looking rank-revealing QR factorization + * + * This class implements a left-looking rank-revealing QR decomposition + * of sparse matrices. When a column has a norm less than a given tolerance + * it is implicitly permuted to the end. The QR factorization thus obtained is + * given by A*P = Q*R where R is upper triangular or trapezoidal. + * + * P is the column permutation which is the product of the fill-reducing and the + * rank-revealing permutations. Use colsPermutation() to get it. + * + * Q is the orthogonal matrix represented as products of Householder reflectors. + * Use matrixQ() to get an expression and matrixQ().adjoint() to get the adjoint. + * You can then apply it to a vector. + * + * R is the sparse triangular or trapezoidal matrix. The later occurs when A is rank-deficient. + * matrixR().topLeftCorner(rank(), rank()) always returns a triangular factor of full rank. + * + * \tparam _MatrixType The type of the sparse matrix A, must be a column-major SparseMatrix<> + * \tparam _OrderingType The fill-reducing ordering method. See the \link OrderingMethods_Module + * OrderingMethods \endlink module for the list of built-in and external ordering methods. + * + * \implsparsesolverconcept + * + * \warning The input sparse matrix A must be in compressed mode (see SparseMatrix::makeCompressed()). + * \warning For complex matrices matrixQ().transpose() will actually return the adjoint matrix. + * + */ template -class SparseQR : public SparseSolverBase > +class SparseQR : public SparseSolverBase> { - protected: - typedef SparseSolverBase > Base; - using Base::m_isInitialized; - public: - using Base::_solve_impl; - typedef _MatrixType MatrixType; - typedef _OrderingType OrderingType; - typedef typename MatrixType::Scalar Scalar; - typedef typename MatrixType::RealScalar RealScalar; - typedef typename MatrixType::StorageIndex StorageIndex; - typedef SparseMatrix QRMatrixType; - typedef Matrix IndexVector; - typedef Matrix ScalarVector; - typedef PermutationMatrix PermutationType; - - enum { - ColsAtCompileTime = MatrixType::ColsAtCompileTime, - MaxColsAtCompileTime = MatrixType::MaxColsAtCompileTime - }; - - public: - SparseQR () : m_analysisIsok(false), m_lastError(""), m_useDefaultThreshold(true),m_isQSorted(false),m_isEtreeOk(false) - { } - - /** Construct a QR factorization of the matrix \a mat. - * - * \warning The matrix \a mat must be in compressed mode (see SparseMatrix::makeCompressed()). - * - * \sa compute() - */ - explicit SparseQR(const MatrixType& mat) : m_analysisIsok(false), m_lastError(""), m_useDefaultThreshold(true),m_isQSorted(false),m_isEtreeOk(false) - { - compute(mat); - } - - /** Computes the QR factorization of the sparse matrix \a mat. - * - * \warning The matrix \a mat must be in compressed mode (see SparseMatrix::makeCompressed()). - * - * \sa analyzePattern(), factorize() - */ - void compute(const MatrixType& mat) - { - analyzePattern(mat); - factorize(mat); - } - void analyzePattern(const MatrixType& mat); - void factorize(const MatrixType& mat); - - /** \returns the number of rows of the represented matrix. - */ - inline Index rows() const { return m_pmat.rows(); } - - /** \returns the number of columns of the represented matrix. - */ - inline Index cols() const { return m_pmat.cols();} - - /** \returns a const reference to the \b sparse upper triangular matrix R of the QR factorization. - * \warning The entries of the returned matrix are not sorted. This means that using it in algorithms - * expecting sorted entries will fail. This include random coefficient accesses (SpaseMatrix::coeff()), - * and coefficient-wise operations. Matrix products and triangular solves are fine though. - * - * To sort the entries, you can assign it to a row-major matrix, and if a column-major matrix - * is required, you can copy it again: - * \code - * SparseMatrix R = qr.matrixR(); // column-major, not sorted! - * SparseMatrix Rr = qr.matrixR(); // row-major, sorted - * SparseMatrix Rc = Rr; // column-major, sorted - * \endcode - */ - const QRMatrixType& matrixR() const { return m_R; } - - /** \returns the number of non linearly dependent columns as determined by the pivoting threshold. - * - * \sa setPivotThreshold() - */ - Index rank() const - { - eigen_assert(m_isInitialized && "The factorization should be called first, use compute()"); - return m_nonzeropivots; - } - - /** \returns an expression of the matrix Q as products of sparse Householder reflectors. - * The common usage of this function is to apply it to a dense matrix or vector - * \code - * VectorXd B1, B2; - * // Initialize B1 - * B2 = matrixQ() * B1; - * \endcode - * - * To get a plain SparseMatrix representation of Q: - * \code - * SparseMatrix Q; - * Q = SparseQR >(A).matrixQ(); - * \endcode - * Internally, this call simply performs a sparse product between the matrix Q - * and a sparse identity matrix. However, due to the fact that the sparse - * reflectors are stored unsorted, two transpositions are needed to sort - * them before performing the product. - */ - SparseQRMatrixQReturnType matrixQ() const - { return SparseQRMatrixQReturnType(*this); } - - /** \returns a const reference to the column permutation P that was applied to A such that A*P = Q*R - * It is the combination of the fill-in reducing permutation and numerical column pivoting. - */ - const PermutationType& colsPermutation() const - { - eigen_assert(m_isInitialized && "Decomposition is not initialized."); - return m_outputPerm_c; - } - - /** \returns A string describing the type of error. - * This method is provided to ease debugging, not to handle errors. - */ - std::string lastErrorMessage() const { return m_lastError; } - - /** \internal */ - template - bool _solve_impl(const MatrixBase &B, MatrixBase &dest) const - { - eigen_assert(m_isInitialized && "The factorization should be called first, use compute()"); - eigen_assert(this->rows() == B.rows() && "SparseQR::solve() : invalid number of rows in the right hand side matrix"); - - Index rank = this->rank(); - - // Compute Q^* * b; - typename Dest::PlainObject y, b; - y = this->matrixQ().adjoint() * B; - b = y; - - // Solve with the triangular matrix R - y.resize((std::max)(cols(),y.rows()),y.cols()); - y.topRows(rank) = this->matrixR().topLeftCorner(rank, rank).template triangularView().solve(b.topRows(rank)); - y.bottomRows(y.rows()-rank).setZero(); - - // Apply the column permutation - if (m_perm_c.size()) dest = colsPermutation() * y.topRows(cols()); - else dest = y.topRows(cols()); - - m_info = Success; - return true; - } +protected: + typedef SparseSolverBase> Base; + using Base::m_isInitialized; + +public: + using Base::_solve_impl; + typedef _MatrixType MatrixType; + typedef _OrderingType OrderingType; + typedef typename MatrixType::Scalar Scalar; + typedef typename MatrixType::RealScalar RealScalar; + typedef typename MatrixType::StorageIndex StorageIndex; + typedef SparseMatrix QRMatrixType; + typedef Matrix IndexVector; + typedef Matrix ScalarVector; + typedef PermutationMatrix PermutationType; + + enum { ColsAtCompileTime = MatrixType::ColsAtCompileTime, MaxColsAtCompileTime = MatrixType::MaxColsAtCompileTime }; + +public: + SparseQR() + : m_analysisIsok(false), m_lastError(""), m_useDefaultThreshold(true), m_isQSorted(false), m_isEtreeOk(false) + {} + + /** Construct a QR factorization of the matrix \a mat. + * + * \warning The matrix \a mat must be in compressed mode (see SparseMatrix::makeCompressed()). + * + * \sa compute() + */ + explicit SparseQR(const MatrixType &mat) + : m_analysisIsok(false), m_lastError(""), m_useDefaultThreshold(true), m_isQSorted(false), m_isEtreeOk(false) + { + compute(mat); + } - /** Sets the threshold that is used to determine linearly dependent columns during the factorization. - * - * In practice, if during the factorization the norm of the column that has to be eliminated is below - * this threshold, then the entire column is treated as zero, and it is moved at the end. - */ - void setPivotThreshold(const RealScalar& threshold) - { - m_useDefaultThreshold = false; - m_threshold = threshold; - } - - /** \returns the solution X of \f$ A X = B \f$ using the current decomposition of A. - * - * \sa compute() - */ - template - inline const Solve solve(const MatrixBase& B) const - { - eigen_assert(m_isInitialized && "The factorization should be called first, use compute()"); - eigen_assert(this->rows() == B.rows() && "SparseQR::solve() : invalid number of rows in the right hand side matrix"); - return Solve(*this, B.derived()); - } - template - inline const Solve solve(const SparseMatrixBase& B) const - { - eigen_assert(m_isInitialized && "The factorization should be called first, use compute()"); - eigen_assert(this->rows() == B.rows() && "SparseQR::solve() : invalid number of rows in the right hand side matrix"); - return Solve(*this, B.derived()); - } - - /** \brief Reports whether previous computation was successful. - * - * \returns \c Success if computation was successful, - * \c NumericalIssue if the QR factorization reports a numerical problem - * \c InvalidInput if the input matrix is invalid - * - * \sa iparm() - */ - ComputationInfo info() const - { - eigen_assert(m_isInitialized && "Decomposition is not initialized."); - return m_info; - } + /** Computes the QR factorization of the sparse matrix \a mat. + * + * \warning The matrix \a mat must be in compressed mode (see SparseMatrix::makeCompressed()). + * + * \sa analyzePattern(), factorize() + */ + void compute(const MatrixType &mat) + { + analyzePattern(mat); + factorize(mat); + } + void analyzePattern(const MatrixType &mat); + void factorize(const MatrixType &mat); + /** \returns the number of rows of the represented matrix. + */ + inline Index rows() const { return m_pmat.rows(); } - /** \internal */ - inline void _sort_matrix_Q() - { - if(this->m_isQSorted) return; - // The matrix Q is sorted during the transposition - SparseMatrix mQrm(this->m_Q); - this->m_Q = mQrm; - this->m_isQSorted = true; - } + /** \returns the number of columns of the represented matrix. + */ + inline Index cols() const { return m_pmat.cols(); } + + /** \returns a const reference to the \b sparse upper triangular matrix R of the QR factorization. + * \warning The entries of the returned matrix are not sorted. This means that using it in algorithms + * expecting sorted entries will fail. This include random coefficient accesses (SpaseMatrix::coeff()), + * and coefficient-wise operations. Matrix products and triangular solves are fine though. + * + * To sort the entries, you can assign it to a row-major matrix, and if a column-major matrix + * is required, you can copy it again: + * \code + * SparseMatrix R = qr.matrixR(); // column-major, not sorted! + * SparseMatrix Rr = qr.matrixR(); // row-major, sorted + * SparseMatrix Rc = Rr; // column-major, sorted + * \endcode + */ + const QRMatrixType &matrixR() const { return m_R; } + + /** \returns the number of non linearly dependent columns as determined by the pivoting threshold. + * + * \sa setPivotThreshold() + */ + Index rank() const + { + eigen_assert(m_isInitialized && "The factorization should be called first, use compute()"); + return m_nonzeropivots; + } + + /** \returns an expression of the matrix Q as products of sparse Householder reflectors. + * The common usage of this function is to apply it to a dense matrix or vector + * \code + * VectorXd B1, B2; + * // Initialize B1 + * B2 = matrixQ() * B1; + * \endcode + * + * To get a plain SparseMatrix representation of Q: + * \code + * SparseMatrix Q; + * Q = SparseQR >(A).matrixQ(); + * \endcode + * Internally, this call simply performs a sparse product between the matrix Q + * and a sparse identity matrix. However, due to the fact that the sparse + * reflectors are stored unsorted, two transpositions are needed to sort + * them before performing the product. + */ + SparseQRMatrixQReturnType matrixQ() const { return SparseQRMatrixQReturnType(*this); } + + /** \returns a const reference to the column permutation P that was applied to A such that A*P = Q*R + * It is the combination of the fill-in reducing permutation and numerical column pivoting. + */ + const PermutationType &colsPermutation() const + { + eigen_assert(m_isInitialized && "Decomposition is not initialized."); + return m_outputPerm_c; + } + + /** \returns A string describing the type of error. + * This method is provided to ease debugging, not to handle errors. + */ + std::string lastErrorMessage() const { return m_lastError; } + + /** \internal */ + template bool _solve_impl(const MatrixBase &B, MatrixBase &dest) const + { + eigen_assert(m_isInitialized && "The factorization should be called first, use compute()"); + eigen_assert( + this->rows() == B.rows() && "SparseQR::solve() : invalid number of rows in the right hand side matrix"); + + Index rank = this->rank(); + + // Compute Q^* * b; + typename Dest::PlainObject y, b; + y = this->matrixQ().adjoint() * B; + b = y; + + // Solve with the triangular matrix R + y.resize((std::max)(cols(), y.rows()), y.cols()); + y.topRows(rank) = this->matrixR().topLeftCorner(rank, rank).template triangularView().solve(b.topRows(rank)); + y.bottomRows(y.rows() - rank).setZero(); + + // Apply the column permutation + if (m_perm_c.size()) + dest = colsPermutation() * y.topRows(cols()); + else + dest = y.topRows(cols()); + + m_info = Success; + return true; + } + + /** Sets the threshold that is used to determine linearly dependent columns during the factorization. + * + * In practice, if during the factorization the norm of the column that has to be eliminated is below + * this threshold, then the entire column is treated as zero, and it is moved at the end. + */ + void setPivotThreshold(const RealScalar &threshold) + { + m_useDefaultThreshold = false; + m_threshold = threshold; + } - - protected: - bool m_analysisIsok; - bool m_factorizationIsok; - mutable ComputationInfo m_info; - std::string m_lastError; - QRMatrixType m_pmat; // Temporary matrix - QRMatrixType m_R; // The triangular factor matrix - QRMatrixType m_Q; // The orthogonal reflectors - ScalarVector m_hcoeffs; // The Householder coefficients - PermutationType m_perm_c; // Fill-reducing Column permutation - PermutationType m_pivotperm; // The permutation for rank revealing - PermutationType m_outputPerm_c; // The final column permutation - RealScalar m_threshold; // Threshold to determine null Householder reflections - bool m_useDefaultThreshold; // Use default threshold - Index m_nonzeropivots; // Number of non zero pivots found - IndexVector m_etree; // Column elimination tree - IndexVector m_firstRowElt; // First element in each row - bool m_isQSorted; // whether Q is sorted or not - bool m_isEtreeOk; // whether the elimination tree match the initial input matrix - - template friend struct SparseQR_QProduct; - + /** \returns the solution X of \f$ A X = B \f$ using the current decomposition of A. + * + * \sa compute() + */ + template inline const Solve solve(const MatrixBase &B) const + { + eigen_assert(m_isInitialized && "The factorization should be called first, use compute()"); + eigen_assert( + this->rows() == B.rows() && "SparseQR::solve() : invalid number of rows in the right hand side matrix"); + return Solve(*this, B.derived()); + } + template inline const Solve solve(const SparseMatrixBase &B) const + { + eigen_assert(m_isInitialized && "The factorization should be called first, use compute()"); + eigen_assert( + this->rows() == B.rows() && "SparseQR::solve() : invalid number of rows in the right hand side matrix"); + return Solve(*this, B.derived()); + } + + /** \brief Reports whether previous computation was successful. + * + * \returns \c Success if computation was successful, + * \c NumericalIssue if the QR factorization reports a numerical problem + * \c InvalidInput if the input matrix is invalid + * + * \sa iparm() + */ + ComputationInfo info() const + { + eigen_assert(m_isInitialized && "Decomposition is not initialized."); + return m_info; + } + + + /** \internal */ + inline void _sort_matrix_Q() + { + if (this->m_isQSorted) return; + // The matrix Q is sorted during the transposition + SparseMatrix mQrm(this->m_Q); + this->m_Q = mQrm; + this->m_isQSorted = true; + } + + +protected: + bool m_analysisIsok; + bool m_factorizationIsok; + mutable ComputationInfo m_info; + std::string m_lastError; + QRMatrixType m_pmat;// Temporary matrix + QRMatrixType m_R;// The triangular factor matrix + QRMatrixType m_Q;// The orthogonal reflectors + ScalarVector m_hcoeffs;// The Householder coefficients + PermutationType m_perm_c;// Fill-reducing Column permutation + PermutationType m_pivotperm;// The permutation for rank revealing + PermutationType m_outputPerm_c;// The final column permutation + RealScalar m_threshold;// Threshold to determine null Householder reflections + bool m_useDefaultThreshold;// Use default threshold + Index m_nonzeropivots;// Number of non zero pivots found + IndexVector m_etree;// Column elimination tree + IndexVector m_firstRowElt;// First element in each row + bool m_isQSorted;// whether Q is sorted or not + bool m_isEtreeOk;// whether the elimination tree match the initial input matrix + + template friend struct SparseQR_QProduct; }; -/** \brief Preprocessing step of a QR factorization - * - * \warning The matrix \a mat must be in compressed mode (see SparseMatrix::makeCompressed()). - * - * In this step, the fill-reducing permutation is computed and applied to the columns of A - * and the column elimination tree is computed as well. Only the sparsity pattern of \a mat is exploited. - * - * \note In this step it is assumed that there is no empty row in the matrix \a mat. - */ -template -void SparseQR::analyzePattern(const MatrixType& mat) +/** \brief Preprocessing step of a QR factorization + * + * \warning The matrix \a mat must be in compressed mode (see SparseMatrix::makeCompressed()). + * + * In this step, the fill-reducing permutation is computed and applied to the columns of A + * and the column elimination tree is computed as well. Only the sparsity pattern of \a mat is exploited. + * + * \note In this step it is assumed that there is no empty row in the matrix \a mat. + */ +template +void SparseQR::analyzePattern(const MatrixType &mat) { - eigen_assert(mat.isCompressed() && "SparseQR requires a sparse matrix in compressed mode. Call .makeCompressed() before passing it to SparseQR"); + eigen_assert( + mat.isCompressed() + && "SparseQR requires a sparse matrix in compressed mode. Call .makeCompressed() before passing it to SparseQR"); // Copy to a column major matrix if the input is rowmajor - typename internal::conditional::type matCpy(mat); + typename internal::conditional::type matCpy(mat); // Compute the column fill reducing ordering - OrderingType ord; - ord(matCpy, m_perm_c); + OrderingType ord; + ord(matCpy, m_perm_c); Index n = mat.cols(); Index m = mat.rows(); - Index diagSize = (std::min)(m,n); - - if (!m_perm_c.size()) - { + Index diagSize = (std::min)(m, n); + + if (!m_perm_c.size()) { m_perm_c.resize(n); - m_perm_c.indices().setLinSpaced(n, 0,StorageIndex(n-1)); + m_perm_c.indices().setLinSpaced(n, 0, StorageIndex(n - 1)); } - + // Compute the column elimination tree of the permuted matrix m_outputPerm_c = m_perm_c.inverse(); internal::coletree(matCpy, m_etree, m_firstRowElt, m_outputPerm_c.indices().data()); m_isEtreeOk = true; - + m_R.resize(m, n); m_Q.resize(m, diagSize); - + // Allocate space for nonzero elements : rough estimation - m_R.reserve(2*mat.nonZeros()); //FIXME Get a more accurate estimation through symbolic factorization with the etree - m_Q.reserve(2*mat.nonZeros()); + m_R.reserve(2 * mat.nonZeros());// FIXME Get a more accurate estimation through symbolic factorization with the etree + m_Q.reserve(2 * mat.nonZeros()); m_hcoeffs.resize(diagSize); m_analysisIsok = true; } /** \brief Performs the numerical QR factorization of the input matrix - * - * The function SparseQR::analyzePattern(const MatrixType&) must have been called beforehand with - * a matrix having the same sparsity pattern than \a mat. - * - * \param mat The sparse column-major matrix - */ -template -void SparseQR::factorize(const MatrixType& mat) + * + * The function SparseQR::analyzePattern(const MatrixType&) must have been called beforehand with + * a matrix having the same sparsity pattern than \a mat. + * + * \param mat The sparse column-major matrix + */ +template +void SparseQR::factorize(const MatrixType &mat) { using std::abs; - + eigen_assert(m_analysisIsok && "analyzePattern() should be called before this step"); StorageIndex m = StorageIndex(mat.rows()); StorageIndex n = StorageIndex(mat.cols()); - StorageIndex diagSize = (std::min)(m,n); - IndexVector mark((std::max)(m,n)); mark.setConstant(-1); // Record the visited nodes - IndexVector Ridx(n), Qidx(m); // Store temporarily the row indexes for the current column of R and Q - Index nzcolR, nzcolQ; // Number of nonzero for the current column of R and Q - ScalarVector tval(m); // The dense vector used to compute the current column + StorageIndex diagSize = (std::min)(m, n); + IndexVector mark((std::max)(m, n)); + mark.setConstant(-1);// Record the visited nodes + IndexVector Ridx(n), Qidx(m);// Store temporarily the row indexes for the current column of R and Q + Index nzcolR, nzcolQ;// Number of nonzero for the current column of R and Q + ScalarVector tval(m);// The dense vector used to compute the current column RealScalar pivotThreshold = m_threshold; - + m_R.setZero(); m_Q.setZero(); m_pmat = mat; - if(!m_isEtreeOk) - { + if (!m_isEtreeOk) { m_outputPerm_c = m_perm_c.inverse(); internal::coletree(m_pmat, m_etree, m_firstRowElt, m_outputPerm_c.indices().data()); m_isEtreeOk = true; } - m_pmat.uncompress(); // To have the innerNonZeroPtr allocated - + m_pmat.uncompress();// To have the innerNonZeroPtr allocated + // Apply the fill-in reducing permutation lazily: { // If the input is row major, copy the original column indices, // otherwise directly use the input matrix - // + // IndexVector originalOuterIndicesCpy; const StorageIndex *originalOuterIndices = mat.outerIndexPtr(); - if(MatrixType::IsRowMajor) - { - originalOuterIndicesCpy = IndexVector::Map(m_pmat.outerIndexPtr(),n+1); + if (MatrixType::IsRowMajor) { + originalOuterIndicesCpy = IndexVector::Map(m_pmat.outerIndexPtr(), n + 1); originalOuterIndices = originalOuterIndicesCpy.data(); } - - for (int i = 0; i < n; i++) - { + + for (int i = 0; i < n; i++) { Index p = m_perm_c.size() ? m_perm_c.indices()(i) : i; - m_pmat.outerIndexPtr()[p] = originalOuterIndices[i]; - m_pmat.innerNonZeroPtr()[p] = originalOuterIndices[i+1] - originalOuterIndices[i]; + m_pmat.outerIndexPtr()[p] = originalOuterIndices[i]; + m_pmat.innerNonZeroPtr()[p] = originalOuterIndices[i + 1] - originalOuterIndices[i]; } } - + /* Compute the default threshold as in MatLab, see: * Tim Davis, "Algorithm 915, SuiteSparseQR: Multifrontal Multithreaded Rank-Revealing - * Sparse QR Factorization, ACM Trans. on Math. Soft. 38(1), 2011, Page 8:3 + * Sparse QR Factorization, ACM Trans. on Math. Soft. 38(1), 2011, Page 8:3 */ - if(m_useDefaultThreshold) - { + if (m_useDefaultThreshold) { RealScalar max2Norm = 0.0; for (int j = 0; j < n; j++) max2Norm = numext::maxi(max2Norm, m_pmat.col(j).norm()); - if(max2Norm==RealScalar(0)) - max2Norm = RealScalar(1); + if (max2Norm == RealScalar(0)) max2Norm = RealScalar(1); pivotThreshold = 20 * (m + n) * max2Norm * NumTraits::epsilon(); } - + // Initialize the numerical permutation m_pivotperm.setIdentity(n); - - StorageIndex nonzeroCol = 0; // Record the number of valid pivots + + StorageIndex nonzeroCol = 0;// Record the number of valid pivots m_Q.startVec(0); // Left looking rank-revealing QR factorization: compute a column of R and Q at a time - for (StorageIndex col = 0; col < n; ++col) - { + for (StorageIndex col = 0; col < n; ++col) { mark.setConstant(-1); m_R.startVec(col); mark(nonzeroCol) = col; Qidx(0) = nonzeroCol; - nzcolR = 0; nzcolQ = 1; - bool found_diag = nonzeroCol>=m; - tval.setZero(); - + nzcolR = 0; + nzcolQ = 1; + bool found_diag = nonzeroCol >= m; + tval.setZero(); + // Symbolic factorization: find the nonzero locations of the column k of the factors R and Q, i.e., - // all the nodes (with indexes lower than rank) reachable through the column elimination tree (etree) rooted at node k. - // Note: if the diagonal entry does not exist, then its contribution must be explicitly added, - // thus the trick with found_diag that permits to do one more iteration on the diagonal element if this one has not been found. - for (typename QRMatrixType::InnerIterator itp(m_pmat, col); itp || !found_diag; ++itp) - { + // all the nodes (with indexes lower than rank) reachable through the column elimination tree (etree) rooted at node + // k. Note: if the diagonal entry does not exist, then its contribution must be explicitly added, thus the trick + // with found_diag that permits to do one more iteration on the diagonal element if this one has not been found. + for (typename QRMatrixType::InnerIterator itp(m_pmat, col); itp || !found_diag; ++itp) { StorageIndex curIdx = nonzeroCol; - if(itp) curIdx = StorageIndex(itp.row()); - if(curIdx == nonzeroCol) found_diag = true; - + if (itp) curIdx = StorageIndex(itp.row()); + if (curIdx == nonzeroCol) found_diag = true; + // Get the nonzeros indexes of the current column of R - StorageIndex st = m_firstRowElt(curIdx); // The traversal of the etree starts here - if (st < 0 ) - { + StorageIndex st = m_firstRowElt(curIdx);// The traversal of the etree starts here + if (st < 0) { m_lastError = "Empty row found during numerical factorization"; m_info = InvalidInput; return; } - // Traverse the etree + // Traverse the etree Index bi = nzcolR; - for (; mark(st) != col; st = m_etree(st)) - { - Ridx(nzcolR) = st; // Add this row to the list, - mark(st) = col; // and mark this row as visited + for (; mark(st) != col; st = m_etree(st)) { + Ridx(nzcolR) = st;// Add this row to the list, + mark(st) = col;// and mark this row as visited nzcolR++; } // Reverse the list to get the topological ordering - Index nt = nzcolR-bi; - for(Index i = 0; i < nt/2; i++) std::swap(Ridx(bi+i), Ridx(nzcolR-i-1)); - + Index nt = nzcolR - bi; + for (Index i = 0; i < nt / 2; i++) std::swap(Ridx(bi + i), Ridx(nzcolR - i - 1)); + // Copy the current (curIdx,pcol) value of the input matrix - if(itp) tval(curIdx) = itp.value(); - else tval(curIdx) = Scalar(0); - + if (itp) + tval(curIdx) = itp.value(); + else + tval(curIdx) = Scalar(0); + // Compute the pattern of Q(:,k) - if(curIdx > nonzeroCol && mark(curIdx) != col ) - { - Qidx(nzcolQ) = curIdx; // Add this row to the pattern of Q, - mark(curIdx) = col; // and mark it as visited + if (curIdx > nonzeroCol && mark(curIdx) != col) { + Qidx(nzcolQ) = curIdx;// Add this row to the pattern of Q, + mark(curIdx) = col;// and mark it as visited nzcolQ++; } } // Browse all the indexes of R(:,col) in reverse order - for (Index i = nzcolR-1; i >= 0; i--) - { + for (Index i = nzcolR - 1; i >= 0; i--) { Index curIdx = Ridx(i); - + // Apply the curIdx-th householder vector to the current column (temporarily stored into tval) Scalar tdot(0); - + // First compute q' * tval tdot = m_Q.col(curIdx).dot(tval); tdot *= m_hcoeffs(curIdx); - + // Then update tval = tval - q * tau - // FIXME: tval -= tdot * m_Q.col(curIdx) should amount to the same (need to check/add support for efficient "dense ?= sparse") - for (typename QRMatrixType::InnerIterator itq(m_Q, curIdx); itq; ++itq) - tval(itq.row()) -= itq.value() * tdot; + // FIXME: tval -= tdot * m_Q.col(curIdx) should amount to the same (need to check/add support for efficient "dense + // ?= sparse") + for (typename QRMatrixType::InnerIterator itq(m_Q, curIdx); itq; ++itq) tval(itq.row()) -= itq.value() * tdot; // Detect fill-in for the current column of Q - if(m_etree(Ridx(i)) == nonzeroCol) - { - for (typename QRMatrixType::InnerIterator itq(m_Q, curIdx); itq; ++itq) - { + if (m_etree(Ridx(i)) == nonzeroCol) { + for (typename QRMatrixType::InnerIterator itq(m_Q, curIdx); itq; ++itq) { StorageIndex iQ = StorageIndex(itq.row()); - if (mark(iQ) != col) - { - Qidx(nzcolQ++) = iQ; // Add this row to the pattern of Q, - mark(iQ) = col; // and mark it as visited + if (mark(iQ) != col) { + Qidx(nzcolQ++) = iQ;// Add this row to the pattern of Q, + mark(iQ) = col;// and mark it as visited } } } - } // End update current column - + }// End update current column + Scalar tau = RealScalar(0); RealScalar beta = 0; - - if(nonzeroCol < diagSize) - { + + if (nonzeroCol < diagSize) { // Compute the Householder reflection that eliminate the current column // FIXME this step should call the Householder module. Scalar c0 = nzcolQ ? tval(Qidx(0)) : Scalar(0); - + // First, the squared norm of Q((col+1):m, col) RealScalar sqrNorm = 0.; for (Index itq = 1; itq < nzcolQ; ++itq) sqrNorm += numext::abs2(tval(Qidx(itq))); - if(sqrNorm == RealScalar(0) && numext::imag(c0) == RealScalar(0)) - { + if (sqrNorm == RealScalar(0) && numext::imag(c0) == RealScalar(0)) { beta = numext::real(c0); tval(Qidx(0)) = 1; - } - else - { + } else { using std::sqrt; beta = sqrt(numext::abs2(c0) + sqrNorm); - if(numext::real(c0) >= RealScalar(0)) - beta = -beta; + if (numext::real(c0) >= RealScalar(0)) beta = -beta; tval(Qidx(0)) = 1; - for (Index itq = 1; itq < nzcolQ; ++itq) - tval(Qidx(itq)) /= (c0 - beta); - tau = numext::conj((beta-c0) / beta); - + for (Index itq = 1; itq < nzcolQ; ++itq) tval(Qidx(itq)) /= (c0 - beta); + tau = numext::conj((beta - c0) / beta); } } // Insert values in R - for (Index i = nzcolR-1; i >= 0; i--) - { + for (Index i = nzcolR - 1; i >= 0; i--) { Index curIdx = Ridx(i); - if(curIdx < nonzeroCol) - { + if (curIdx < nonzeroCol) { m_R.insertBackByOuterInnerUnordered(col, curIdx) = tval(curIdx); tval(curIdx) = Scalar(0.); } } - if(nonzeroCol < diagSize && abs(beta) >= pivotThreshold) - { + if (nonzeroCol < diagSize && abs(beta) >= pivotThreshold) { m_R.insertBackByOuterInner(col, nonzeroCol) = beta; // The householder coefficient m_hcoeffs(nonzeroCol) = tau; // Record the householder reflections - for (Index itq = 0; itq < nzcolQ; ++itq) - { + for (Index itq = 0; itq < nzcolQ; ++itq) { Index iQ = Qidx(itq); - m_Q.insertBackByOuterInnerUnordered(nonzeroCol,iQ) = tval(iQ); + m_Q.insertBackByOuterInnerUnordered(nonzeroCol, iQ) = tval(iQ); tval(iQ) = Scalar(0.); } nonzeroCol++; - if(nonzeroCol::factorize(const MatrixType& mat) m_isQSorted = false; m_nonzeropivots = nonzeroCol; - - if(nonzeroCol -struct SparseQR_QProduct : ReturnByValue > +template +struct SparseQR_QProduct : ReturnByValue> { typedef typename SparseQRType::QRMatrixType MatrixType; typedef typename SparseQRType::Scalar Scalar; - // Get the references - SparseQR_QProduct(const SparseQRType& qr, const Derived& other, bool transpose) : - m_qr(qr),m_other(other),m_transpose(transpose) {} + // Get the references + SparseQR_QProduct(const SparseQRType &qr, const Derived &other, bool transpose) + : m_qr(qr), m_other(other), m_transpose(transpose) + {} inline Index rows() const { return m_qr.matrixQ().rows(); } inline Index cols() const { return m_other.cols(); } - + // Assign to a vector - template - void evalTo(DesType& res) const + template void evalTo(DesType &res) const { Index m = m_qr.rows(); Index n = m_qr.cols(); - Index diagSize = (std::min)(m,n); + Index diagSize = (std::min)(m, n); res = m_other; - if (m_transpose) - { + if (m_transpose) { eigen_assert(m_qr.m_Q.rows() == m_other.rows() && "Non conforming object sizes"); - //Compute res = Q' * other column by column - for(Index j = 0; j < res.cols(); j++){ - for (Index k = 0; k < diagSize; k++) - { + // Compute res = Q' * other column by column + for (Index j = 0; j < res.cols(); j++) { + for (Index k = 0; k < diagSize; k++) { Scalar tau = Scalar(0); tau = m_qr.m_Q.col(k).dot(res.col(j)); - if(tau==Scalar(0)) continue; + if (tau == Scalar(0)) continue; tau = tau * m_qr.m_hcoeffs(k); res.col(j) -= tau * m_qr.m_Q.col(k); } } - } - else - { + } else { eigen_assert(m_qr.matrixQ().cols() == m_other.rows() && "Non conforming object sizes"); res.conservativeResize(rows(), cols()); // Compute res = Q * other column by column - for(Index j = 0; j < res.cols(); j++) - { - for (Index k = diagSize-1; k >=0; k--) - { + for (Index j = 0; j < res.cols(); j++) { + for (Index k = diagSize - 1; k >= 0; k--) { Scalar tau = Scalar(0); tau = m_qr.m_Q.col(k).dot(res.col(j)); - if(tau==Scalar(0)) continue; + if (tau == Scalar(0)) continue; tau = tau * numext::conj(m_qr.m_hcoeffs(k)); res.col(j) -= tau * m_qr.m_Q.col(k); } } } } - - const SparseQRType& m_qr; - const Derived& m_other; - bool m_transpose; // TODO this actually means adjoint + + const SparseQRType &m_qr; + const Derived &m_other; + bool m_transpose;// TODO this actually means adjoint }; template -struct SparseQRMatrixQReturnType : public EigenBase > -{ +struct SparseQRMatrixQReturnType : public EigenBase> +{ typedef typename SparseQRType::Scalar Scalar; - typedef Matrix DenseMatrix; - enum { - RowsAtCompileTime = Dynamic, - ColsAtCompileTime = Dynamic - }; - explicit SparseQRMatrixQReturnType(const SparseQRType& qr) : m_qr(qr) {} - template - SparseQR_QProduct operator*(const MatrixBase& other) + typedef Matrix DenseMatrix; + enum { RowsAtCompileTime = Dynamic, ColsAtCompileTime = Dynamic }; + explicit SparseQRMatrixQReturnType(const SparseQRType &qr) : m_qr(qr) {} + template SparseQR_QProduct operator*(const MatrixBase &other) { - return SparseQR_QProduct(m_qr,other.derived(),false); + return SparseQR_QProduct(m_qr, other.derived(), false); } // To use for operations with the adjoint of Q SparseQRMatrixQTransposeReturnType adjoint() const @@ -684,62 +646,65 @@ struct SparseQRMatrixQReturnType : public EigenBase(m_qr); } - const SparseQRType& m_qr; + const SparseQRType &m_qr; }; // TODO this actually represents the adjoint of Q -template -struct SparseQRMatrixQTransposeReturnType +template struct SparseQRMatrixQTransposeReturnType { - explicit SparseQRMatrixQTransposeReturnType(const SparseQRType& qr) : m_qr(qr) {} - template - SparseQR_QProduct operator*(const MatrixBase& other) + explicit SparseQRMatrixQTransposeReturnType(const SparseQRType &qr) : m_qr(qr) {} + template SparseQR_QProduct operator*(const MatrixBase &other) { - return SparseQR_QProduct(m_qr,other.derived(), true); + return SparseQR_QProduct(m_qr, other.derived(), true); } - const SparseQRType& m_qr; + const SparseQRType &m_qr; }; namespace internal { - -template -struct evaluator_traits > -{ - typedef typename SparseQRType::MatrixType MatrixType; - typedef typename storage_kind_to_evaluator_kind::Kind Kind; - typedef SparseShape Shape; -}; -template< typename DstXprType, typename SparseQRType> -struct Assignment, internal::assign_op, Sparse2Sparse> -{ - typedef SparseQRMatrixQReturnType SrcXprType; - typedef typename DstXprType::Scalar Scalar; - typedef typename DstXprType::StorageIndex StorageIndex; - static void run(DstXprType &dst, const SrcXprType &src, const internal::assign_op &/*func*/) + template struct evaluator_traits> { - typename DstXprType::PlainObject idMat(src.rows(), src.cols()); - idMat.setIdentity(); - // Sort the sparse householder reflectors if needed - const_cast(&src.m_qr)->_sort_matrix_Q(); - dst = SparseQR_QProduct(src.m_qr, idMat, false); - } -}; + typedef typename SparseQRType::MatrixType MatrixType; + typedef typename storage_kind_to_evaluator_kind::Kind Kind; + typedef SparseShape Shape; + }; -template< typename DstXprType, typename SparseQRType> -struct Assignment, internal::assign_op, Sparse2Dense> -{ - typedef SparseQRMatrixQReturnType SrcXprType; - typedef typename DstXprType::Scalar Scalar; - typedef typename DstXprType::StorageIndex StorageIndex; - static void run(DstXprType &dst, const SrcXprType &src, const internal::assign_op &/*func*/) + template + struct Assignment, + internal::assign_op, + Sparse2Sparse> { - dst = src.m_qr.matrixQ() * DstXprType::Identity(src.m_qr.rows(), src.m_qr.rows()); - } -}; + typedef SparseQRMatrixQReturnType SrcXprType; + typedef typename DstXprType::Scalar Scalar; + typedef typename DstXprType::StorageIndex StorageIndex; + static void run(DstXprType &dst, const SrcXprType &src, const internal::assign_op & /*func*/) + { + typename DstXprType::PlainObject idMat(src.rows(), src.cols()); + idMat.setIdentity(); + // Sort the sparse householder reflectors if needed + const_cast(&src.m_qr)->_sort_matrix_Q(); + dst = SparseQR_QProduct(src.m_qr, idMat, false); + } + }; + + template + struct Assignment, + internal::assign_op, + Sparse2Dense> + { + typedef SparseQRMatrixQReturnType SrcXprType; + typedef typename DstXprType::Scalar Scalar; + typedef typename DstXprType::StorageIndex StorageIndex; + static void run(DstXprType &dst, const SrcXprType &src, const internal::assign_op & /*func*/) + { + dst = src.m_qr.matrixQ() * DstXprType::Identity(src.m_qr.rows(), src.m_qr.rows()); + } + }; -} // end namespace internal +}// end namespace internal -} // end namespace Eigen +}// end namespace Eigen #endif diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/StlSupport/StdDeque.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/StlSupport/StdDeque.h index cf1fedf9..3dc70e09 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/StlSupport/StdDeque.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/StlSupport/StdDeque.h @@ -18,89 +18,96 @@ * std::deque such that for data types with alignment issues the correct allocator * is used automatically. */ -#define EIGEN_DEFINE_STL_DEQUE_SPECIALIZATION(...) \ -namespace std \ -{ \ - template<> \ - class deque<__VA_ARGS__, std::allocator<__VA_ARGS__> > \ - : public deque<__VA_ARGS__, EIGEN_ALIGNED_ALLOCATOR<__VA_ARGS__> > \ - { \ - typedef deque<__VA_ARGS__, EIGEN_ALIGNED_ALLOCATOR<__VA_ARGS__> > deque_base; \ - public: \ - typedef __VA_ARGS__ value_type; \ - typedef deque_base::allocator_type allocator_type; \ - typedef deque_base::size_type size_type; \ - typedef deque_base::iterator iterator; \ - explicit deque(const allocator_type& a = allocator_type()) : deque_base(a) {} \ - template \ - deque(InputIterator first, InputIterator last, const allocator_type& a = allocator_type()) : deque_base(first, last, a) {} \ - deque(const deque& c) : deque_base(c) {} \ - explicit deque(size_type num, const value_type& val = value_type()) : deque_base(num, val) {} \ - deque(iterator start, iterator end) : deque_base(start, end) {} \ - deque& operator=(const deque& x) { \ - deque_base::operator=(x); \ - return *this; \ - } \ - }; \ -} +#define EIGEN_DEFINE_STL_DEQUE_SPECIALIZATION(...) \ + namespace std { \ + template<> \ + class deque<__VA_ARGS__, std::allocator<__VA_ARGS__>> \ + : public deque<__VA_ARGS__, EIGEN_ALIGNED_ALLOCATOR<__VA_ARGS__>> \ + { \ + typedef deque<__VA_ARGS__, EIGEN_ALIGNED_ALLOCATOR<__VA_ARGS__>> deque_base; \ + \ + public: \ + typedef __VA_ARGS__ value_type; \ + typedef deque_base::allocator_type allocator_type; \ + typedef deque_base::size_type size_type; \ + typedef deque_base::iterator iterator; \ + explicit deque(const allocator_type &a = allocator_type()) : deque_base(a) {} \ + template \ + deque(InputIterator first, InputIterator last, const allocator_type &a = allocator_type()) \ + : deque_base(first, last, a) \ + {} \ + deque(const deque &c) : deque_base(c) {} \ + explicit deque(size_type num, const value_type &val = value_type()) : deque_base(num, val) {} \ + deque(iterator start, iterator end) : deque_base(start, end) {} \ + deque &operator=(const deque &x) \ + { \ + deque_base::operator=(x); \ + return *this; \ + } \ + }; \ + } // check whether we really need the std::deque specialization -#if !EIGEN_HAS_CXX11_CONTAINERS && !(defined(_GLIBCXX_DEQUE) && (!EIGEN_GNUC_AT_LEAST(4,1))) /* Note that before gcc-4.1 we already have: std::deque::resize(size_type,const T&). */ +#if !EIGEN_HAS_CXX11_CONTAINERS \ + && !(defined(_GLIBCXX_DEQUE) \ + && (!EIGEN_GNUC_AT_LEAST( \ + 4, 1))) /* Note that before gcc-4.1 we already have: std::deque::resize(size_type,const T&). */ namespace std { -#define EIGEN_STD_DEQUE_SPECIALIZATION_BODY \ - public: \ - typedef T value_type; \ - typedef typename deque_base::allocator_type allocator_type; \ - typedef typename deque_base::size_type size_type; \ - typedef typename deque_base::iterator iterator; \ - typedef typename deque_base::const_iterator const_iterator; \ - explicit deque(const allocator_type& a = allocator_type()) : deque_base(a) {} \ - template \ - deque(InputIterator first, InputIterator last, const allocator_type& a = allocator_type()) \ - : deque_base(first, last, a) {} \ - deque(const deque& c) : deque_base(c) {} \ - explicit deque(size_type num, const value_type& val = value_type()) : deque_base(num, val) {} \ - deque(iterator start, iterator end) : deque_base(start, end) {} \ - deque& operator=(const deque& x) { \ - deque_base::operator=(x); \ - return *this; \ - } +#define EIGEN_STD_DEQUE_SPECIALIZATION_BODY \ +public: \ + typedef T value_type; \ + typedef typename deque_base::allocator_type allocator_type; \ + typedef typename deque_base::size_type size_type; \ + typedef typename deque_base::iterator iterator; \ + typedef typename deque_base::const_iterator const_iterator; \ + explicit deque(const allocator_type &a = allocator_type()) : deque_base(a) {} \ + template \ + deque(InputIterator first, InputIterator last, const allocator_type &a = allocator_type()) \ + : deque_base(first, last, a) \ + {} \ + deque(const deque &c) : deque_base(c) {} \ + explicit deque(size_type num, const value_type &val = value_type()) : deque_base(num, val) {} \ + deque(iterator start, iterator end) : deque_base(start, end) {} \ + deque &operator=(const deque &x) \ + { \ + deque_base::operator=(x); \ + return *this; \ + } - template - class deque > - : public deque > +template +class deque> + : public deque> { typedef deque > deque_base; + Eigen::aligned_allocator_indirection> + deque_base; EIGEN_STD_DEQUE_SPECIALIZATION_BODY - void resize(size_type new_size) - { resize(new_size, T()); } + void resize(size_type new_size) { resize(new_size, T()); } #if defined(_DEQUE_) // workaround MSVC std::deque implementation - void resize(size_type new_size, const value_type& x) + void resize(size_type new_size, const value_type &x) { if (deque_base::size() < new_size) deque_base::_Insert_n(deque_base::end(), new_size - deque_base::size(), x); else if (new_size < deque_base::size()) deque_base::erase(deque_base::begin() + new_size, deque_base::end()); } - void push_back(const value_type& x) - { deque_base::push_back(x); } - void push_front(const value_type& x) - { deque_base::push_front(x); } - using deque_base::insert; - iterator insert(const_iterator position, const value_type& x) - { return deque_base::insert(position,x); } - void insert(const_iterator position, size_type new_size, const value_type& x) - { deque_base::insert(position, new_size, x); } -#elif defined(_GLIBCXX_DEQUE) && EIGEN_GNUC_AT_LEAST(4,2) + void push_back(const value_type &x) { deque_base::push_back(x); } + void push_front(const value_type &x) { deque_base::push_front(x); } + using deque_base::insert; + iterator insert(const_iterator position, const value_type &x) { return deque_base::insert(position, x); } + void insert(const_iterator position, size_type new_size, const value_type &x) + { + deque_base::insert(position, new_size, x); + } +#elif defined(_GLIBCXX_DEQUE) && EIGEN_GNUC_AT_LEAST(4, 2) // workaround GCC std::deque implementation - void resize(size_type new_size, const value_type& x) + void resize(size_type new_size, const value_type &x) { if (new_size < deque_base::size()) deque_base::_M_erase_at_end(this->_M_impl._M_start + new_size); @@ -110,7 +117,7 @@ namespace std { #else // either GCC 4.1 or non-GCC // default implementation which should always work. - void resize(size_type new_size, const value_type& x) + void resize(size_type new_size, const value_type &x) { if (new_size < deque_base::size()) deque_base::erase(deque_base::begin() + new_size, deque_base::end()); @@ -118,9 +125,9 @@ namespace std { deque_base::insert(deque_base::end(), new_size - deque_base::size(), x); } #endif - }; -} +}; +}// namespace std -#endif // check whether specialization is actually required +#endif// check whether specialization is actually required -#endif // EIGEN_STDDEQUE_H +#endif// EIGEN_STDDEQUE_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/StlSupport/StdList.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/StlSupport/StdList.h index e1eba498..fb8d084e 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/StlSupport/StdList.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/StlSupport/StdList.h @@ -17,90 +17,97 @@ * std::list such that for data types with alignment issues the correct allocator * is used automatically. */ -#define EIGEN_DEFINE_STL_LIST_SPECIALIZATION(...) \ -namespace std \ -{ \ - template<> \ - class list<__VA_ARGS__, std::allocator<__VA_ARGS__> > \ - : public list<__VA_ARGS__, EIGEN_ALIGNED_ALLOCATOR<__VA_ARGS__> > \ - { \ - typedef list<__VA_ARGS__, EIGEN_ALIGNED_ALLOCATOR<__VA_ARGS__> > list_base; \ - public: \ - typedef __VA_ARGS__ value_type; \ - typedef list_base::allocator_type allocator_type; \ - typedef list_base::size_type size_type; \ - typedef list_base::iterator iterator; \ - explicit list(const allocator_type& a = allocator_type()) : list_base(a) {} \ - template \ - list(InputIterator first, InputIterator last, const allocator_type& a = allocator_type()) : list_base(first, last, a) {} \ - list(const list& c) : list_base(c) {} \ - explicit list(size_type num, const value_type& val = value_type()) : list_base(num, val) {} \ - list(iterator start, iterator end) : list_base(start, end) {} \ - list& operator=(const list& x) { \ - list_base::operator=(x); \ - return *this; \ - } \ - }; \ -} +#define EIGEN_DEFINE_STL_LIST_SPECIALIZATION(...) \ + namespace std { \ + template<> \ + class list<__VA_ARGS__, std::allocator<__VA_ARGS__>> \ + : public list<__VA_ARGS__, EIGEN_ALIGNED_ALLOCATOR<__VA_ARGS__>> \ + { \ + typedef list<__VA_ARGS__, EIGEN_ALIGNED_ALLOCATOR<__VA_ARGS__>> list_base; \ + \ + public: \ + typedef __VA_ARGS__ value_type; \ + typedef list_base::allocator_type allocator_type; \ + typedef list_base::size_type size_type; \ + typedef list_base::iterator iterator; \ + explicit list(const allocator_type &a = allocator_type()) : list_base(a) {} \ + template \ + list(InputIterator first, InputIterator last, const allocator_type &a = allocator_type()) \ + : list_base(first, last, a) \ + {} \ + list(const list &c) : list_base(c) {} \ + explicit list(size_type num, const value_type &val = value_type()) : list_base(num, val) {} \ + list(iterator start, iterator end) : list_base(start, end) {} \ + list &operator=(const list &x) \ + { \ + list_base::operator=(x); \ + return *this; \ + } \ + }; \ + } // check whether we really need the std::list specialization -#if !EIGEN_HAS_CXX11_CONTAINERS && !(defined(_GLIBCXX_LIST) && (!EIGEN_GNUC_AT_LEAST(4,1))) /* Note that before gcc-4.1 we already have: std::list::resize(size_type,const T&). */ +#if !EIGEN_HAS_CXX11_CONTAINERS \ + && !(defined(_GLIBCXX_LIST) \ + && (!EIGEN_GNUC_AT_LEAST( \ + 4, 1))) /* Note that before gcc-4.1 we already have: std::list::resize(size_type,const T&). */ -namespace std -{ +namespace std { -#define EIGEN_STD_LIST_SPECIALIZATION_BODY \ - public: \ - typedef T value_type; \ - typedef typename list_base::allocator_type allocator_type; \ - typedef typename list_base::size_type size_type; \ - typedef typename list_base::iterator iterator; \ - typedef typename list_base::const_iterator const_iterator; \ - explicit list(const allocator_type& a = allocator_type()) : list_base(a) {} \ - template \ - list(InputIterator first, InputIterator last, const allocator_type& a = allocator_type()) \ - : list_base(first, last, a) {} \ - list(const list& c) : list_base(c) {} \ - explicit list(size_type num, const value_type& val = value_type()) : list_base(num, val) {} \ - list(iterator start, iterator end) : list_base(start, end) {} \ - list& operator=(const list& x) { \ - list_base::operator=(x); \ - return *this; \ +#define EIGEN_STD_LIST_SPECIALIZATION_BODY \ +public: \ + typedef T value_type; \ + typedef typename list_base::allocator_type allocator_type; \ + typedef typename list_base::size_type size_type; \ + typedef typename list_base::iterator iterator; \ + typedef typename list_base::const_iterator const_iterator; \ + explicit list(const allocator_type &a = allocator_type()) : list_base(a) {} \ + template \ + list(InputIterator first, InputIterator last, const allocator_type &a = allocator_type()) \ + : list_base(first, last, a) \ + {} \ + list(const list &c) : list_base(c) {} \ + explicit list(size_type num, const value_type &val = value_type()) : list_base(num, val) {} \ + list(iterator start, iterator end) : list_base(start, end) {} \ + list &operator=(const list &x) \ + { \ + list_base::operator=(x); \ + return *this; \ } - template - class list > - : public list > - { - typedef list > list_base; - EIGEN_STD_LIST_SPECIALIZATION_BODY +template +class list> + : public list> +{ + typedef list> + list_base; + EIGEN_STD_LIST_SPECIALIZATION_BODY - void resize(size_type new_size) - { resize(new_size, T()); } + void resize(size_type new_size) { resize(new_size, T()); } - void resize(size_type new_size, const value_type& x) - { - if (list_base::size() < new_size) - list_base::insert(list_base::end(), new_size - list_base::size(), x); - else - while (new_size < list_base::size()) list_base::pop_back(); - } + void resize(size_type new_size, const value_type &x) + { + if (list_base::size() < new_size) + list_base::insert(list_base::end(), new_size - list_base::size(), x); + else + while (new_size < list_base::size()) list_base::pop_back(); + } #if defined(_LIST_) - // workaround MSVC std::list implementation - void push_back(const value_type& x) - { list_base::push_back(x); } - using list_base::insert; - iterator insert(const_iterator position, const value_type& x) - { return list_base::insert(position,x); } - void insert(const_iterator position, size_type new_size, const value_type& x) - { list_base::insert(position, new_size, x); } + // workaround MSVC std::list implementation + void push_back(const value_type &x) { list_base::push_back(x); } + using list_base::insert; + iterator insert(const_iterator position, const value_type &x) { return list_base::insert(position, x); } + void insert(const_iterator position, size_type new_size, const value_type &x) + { + list_base::insert(position, new_size, x); + } #endif - }; -} +}; +}// namespace std -#endif // check whether specialization is actually required +#endif// check whether specialization is actually required -#endif // EIGEN_STDLIST_H +#endif// EIGEN_STDLIST_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/StlSupport/StdVector.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/StlSupport/StdVector.h index ec22821d..d84dbe34 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/StlSupport/StdVector.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/StlSupport/StdVector.h @@ -18,94 +18,97 @@ * std::vector such that for data types with alignment issues the correct allocator * is used automatically. */ -#define EIGEN_DEFINE_STL_VECTOR_SPECIALIZATION(...) \ -namespace std \ -{ \ - template<> \ - class vector<__VA_ARGS__, std::allocator<__VA_ARGS__> > \ - : public vector<__VA_ARGS__, EIGEN_ALIGNED_ALLOCATOR<__VA_ARGS__> > \ - { \ - typedef vector<__VA_ARGS__, EIGEN_ALIGNED_ALLOCATOR<__VA_ARGS__> > vector_base; \ - public: \ - typedef __VA_ARGS__ value_type; \ - typedef vector_base::allocator_type allocator_type; \ - typedef vector_base::size_type size_type; \ - typedef vector_base::iterator iterator; \ - explicit vector(const allocator_type& a = allocator_type()) : vector_base(a) {} \ - template \ - vector(InputIterator first, InputIterator last, const allocator_type& a = allocator_type()) : vector_base(first, last, a) {} \ - vector(const vector& c) : vector_base(c) {} \ - explicit vector(size_type num, const value_type& val = value_type()) : vector_base(num, val) {} \ - vector(iterator start, iterator end) : vector_base(start, end) {} \ - vector& operator=(const vector& x) { \ - vector_base::operator=(x); \ - return *this; \ - } \ - }; \ -} +#define EIGEN_DEFINE_STL_VECTOR_SPECIALIZATION(...) \ + namespace std { \ + template<> \ + class vector<__VA_ARGS__, std::allocator<__VA_ARGS__>> \ + : public vector<__VA_ARGS__, EIGEN_ALIGNED_ALLOCATOR<__VA_ARGS__>> \ + { \ + typedef vector<__VA_ARGS__, EIGEN_ALIGNED_ALLOCATOR<__VA_ARGS__>> vector_base; \ + \ + public: \ + typedef __VA_ARGS__ value_type; \ + typedef vector_base::allocator_type allocator_type; \ + typedef vector_base::size_type size_type; \ + typedef vector_base::iterator iterator; \ + explicit vector(const allocator_type &a = allocator_type()) : vector_base(a) {} \ + template \ + vector(InputIterator first, InputIterator last, const allocator_type &a = allocator_type()) \ + : vector_base(first, last, a) \ + {} \ + vector(const vector &c) : vector_base(c) {} \ + explicit vector(size_type num, const value_type &val = value_type()) : vector_base(num, val) {} \ + vector(iterator start, iterator end) : vector_base(start, end) {} \ + vector &operator=(const vector &x) \ + { \ + vector_base::operator=(x); \ + return *this; \ + } \ + }; \ + } // Don't specialize if containers are implemented according to C++11 #if !EIGEN_HAS_CXX11_CONTAINERS namespace std { -#define EIGEN_STD_VECTOR_SPECIALIZATION_BODY \ - public: \ - typedef T value_type; \ - typedef typename vector_base::allocator_type allocator_type; \ - typedef typename vector_base::size_type size_type; \ - typedef typename vector_base::iterator iterator; \ - typedef typename vector_base::const_iterator const_iterator; \ - explicit vector(const allocator_type& a = allocator_type()) : vector_base(a) {} \ - template \ - vector(InputIterator first, InputIterator last, const allocator_type& a = allocator_type()) \ - : vector_base(first, last, a) {} \ - vector(const vector& c) : vector_base(c) {} \ - explicit vector(size_type num, const value_type& val = value_type()) : vector_base(num, val) {} \ - vector(iterator start, iterator end) : vector_base(start, end) {} \ - vector& operator=(const vector& x) { \ - vector_base::operator=(x); \ - return *this; \ - } +#define EIGEN_STD_VECTOR_SPECIALIZATION_BODY \ +public: \ + typedef T value_type; \ + typedef typename vector_base::allocator_type allocator_type; \ + typedef typename vector_base::size_type size_type; \ + typedef typename vector_base::iterator iterator; \ + typedef typename vector_base::const_iterator const_iterator; \ + explicit vector(const allocator_type &a = allocator_type()) : vector_base(a) {} \ + template \ + vector(InputIterator first, InputIterator last, const allocator_type &a = allocator_type()) \ + : vector_base(first, last, a) \ + {} \ + vector(const vector &c) : vector_base(c) {} \ + explicit vector(size_type num, const value_type &val = value_type()) : vector_base(num, val) {} \ + vector(iterator start, iterator end) : vector_base(start, end) {} \ + vector &operator=(const vector &x) \ + { \ + vector_base::operator=(x); \ + return *this; \ + } - template - class vector > - : public vector > +template +class vector> + : public vector> { typedef vector > vector_base; + Eigen::aligned_allocator_indirection> + vector_base; EIGEN_STD_VECTOR_SPECIALIZATION_BODY - void resize(size_type new_size) - { resize(new_size, T()); } + void resize(size_type new_size) { resize(new_size, T()); } #if defined(_VECTOR_) // workaround MSVC std::vector implementation - void resize(size_type new_size, const value_type& x) + void resize(size_type new_size, const value_type &x) { if (vector_base::size() < new_size) vector_base::_Insert_n(vector_base::end(), new_size - vector_base::size(), x); else if (new_size < vector_base::size()) vector_base::erase(vector_base::begin() + new_size, vector_base::end()); } - void push_back(const value_type& x) - { vector_base::push_back(x); } - using vector_base::insert; - iterator insert(const_iterator position, const value_type& x) - { return vector_base::insert(position,x); } - void insert(const_iterator position, size_type new_size, const value_type& x) - { vector_base::insert(position, new_size, x); } -#elif defined(_GLIBCXX_VECTOR) && (!(EIGEN_GNUC_AT_LEAST(4,1))) - /* Note that before gcc-4.1 we already have: std::vector::resize(size_type,const T&). - * However, this specialization is still needed to make the above EIGEN_DEFINE_STL_VECTOR_SPECIALIZATION trick to work. */ - void resize(size_type new_size, const value_type& x) + void push_back(const value_type &x) { vector_base::push_back(x); } + using vector_base::insert; + iterator insert(const_iterator position, const value_type &x) { return vector_base::insert(position, x); } + void insert(const_iterator position, size_type new_size, const value_type &x) { - vector_base::resize(new_size,x); + vector_base::insert(position, new_size, x); } -#elif defined(_GLIBCXX_VECTOR) && EIGEN_GNUC_AT_LEAST(4,2) +#elif defined(_GLIBCXX_VECTOR) && (!(EIGEN_GNUC_AT_LEAST(4, 1))) + /* Note that before gcc-4.1 we already have: std::vector::resize(size_type,const T&). + * However, this specialization is still needed to make the above EIGEN_DEFINE_STL_VECTOR_SPECIALIZATION trick to + * work. */ + void resize(size_type new_size, const value_type &x) { vector_base::resize(new_size, x); } +#elif defined(_GLIBCXX_VECTOR) && EIGEN_GNUC_AT_LEAST(4, 2) // workaround GCC std::vector implementation - void resize(size_type new_size, const value_type& x) + void resize(size_type new_size, const value_type &x) { if (new_size < vector_base::size()) vector_base::_M_erase_at_end(this->_M_impl._M_start + new_size); @@ -115,7 +118,7 @@ namespace std { #else // either GCC 4.1 or non-GCC // default implementation which should always work. - void resize(size_type new_size, const value_type& x) + void resize(size_type new_size, const value_type &x) { if (new_size < vector_base::size()) vector_base::erase(vector_base::begin() + new_size, vector_base::end()); @@ -123,9 +126,9 @@ namespace std { vector_base::insert(vector_base::end(), new_size - vector_base::size(), x); } #endif - }; -} -#endif // !EIGEN_HAS_CXX11_CONTAINERS +}; +}// namespace std +#endif// !EIGEN_HAS_CXX11_CONTAINERS -#endif // EIGEN_STDVECTOR_H +#endif// EIGEN_STDVECTOR_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/StlSupport/details.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/StlSupport/details.h index 2cfd13e0..5f57ad0e 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/StlSupport/details.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/StlSupport/details.h @@ -12,66 +12,64 @@ #define EIGEN_STL_DETAILS_H #ifndef EIGEN_ALIGNED_ALLOCATOR - #define EIGEN_ALIGNED_ALLOCATOR Eigen::aligned_allocator +#define EIGEN_ALIGNED_ALLOCATOR Eigen::aligned_allocator #endif namespace Eigen { - // This one is needed to prevent reimplementing the whole std::vector. - template - class aligned_allocator_indirection : public EIGEN_ALIGNED_ALLOCATOR - { - public: - typedef std::size_t size_type; - typedef std::ptrdiff_t difference_type; - typedef T* pointer; - typedef const T* const_pointer; - typedef T& reference; - typedef const T& const_reference; - typedef T value_type; - - template - struct rebind - { - typedef aligned_allocator_indirection other; - }; +// This one is needed to prevent reimplementing the whole std::vector. +template class aligned_allocator_indirection : public EIGEN_ALIGNED_ALLOCATOR +{ +public: + typedef std::size_t size_type; + typedef std::ptrdiff_t difference_type; + typedef T *pointer; + typedef const T *const_pointer; + typedef T &reference; + typedef const T &const_reference; + typedef T value_type; - aligned_allocator_indirection() {} - aligned_allocator_indirection(const aligned_allocator_indirection& ) : EIGEN_ALIGNED_ALLOCATOR() {} - aligned_allocator_indirection(const EIGEN_ALIGNED_ALLOCATOR& ) {} - template - aligned_allocator_indirection(const aligned_allocator_indirection& ) {} - template - aligned_allocator_indirection(const EIGEN_ALIGNED_ALLOCATOR& ) {} - ~aligned_allocator_indirection() {} + template struct rebind + { + typedef aligned_allocator_indirection other; }; + aligned_allocator_indirection() {} + aligned_allocator_indirection(const aligned_allocator_indirection &) : EIGEN_ALIGNED_ALLOCATOR() {} + aligned_allocator_indirection(const EIGEN_ALIGNED_ALLOCATOR &) {} + template aligned_allocator_indirection(const aligned_allocator_indirection &) {} + template aligned_allocator_indirection(const EIGEN_ALIGNED_ALLOCATOR &) {} + ~aligned_allocator_indirection() {} +}; + #if EIGEN_COMP_MSVC - // sometimes, MSVC detects, at compile time, that the argument x - // in std::vector::resize(size_t s,T x) won't be aligned and generate an error - // even if this function is never called. Whence this little wrapper. +// sometimes, MSVC detects, at compile time, that the argument x +// in std::vector::resize(size_t s,T x) won't be aligned and generate an error +// even if this function is never called. Whence this little wrapper. #define EIGEN_WORKAROUND_MSVC_STL_SUPPORT(T) \ - typename Eigen::internal::conditional< \ - Eigen::internal::is_arithmetic::value, \ - T, \ - Eigen::internal::workaround_msvc_stl_support \ - >::type + typename Eigen::internal:: \ + conditional::value, T, Eigen::internal::workaround_msvc_stl_support>::type - namespace internal { +namespace internal { template struct workaround_msvc_stl_support : public T { inline workaround_msvc_stl_support() : T() {} - inline workaround_msvc_stl_support(const T& other) : T(other) {} - inline operator T& () { return *static_cast(this); } - inline operator const T& () const { return *static_cast(this); } - template - inline T& operator=(const OtherT& other) - { T::operator=(other); return *this; } - inline workaround_msvc_stl_support& operator=(const workaround_msvc_stl_support& other) - { T::operator=(other); return *this; } + inline workaround_msvc_stl_support(const T &other) : T(other) {} + inline operator T &() { return *static_cast(this); } + inline operator const T &() const { return *static_cast(this); } + template inline T &operator=(const OtherT &other) + { + T::operator=(other); + return *this; + } + inline workaround_msvc_stl_support &operator=(const workaround_msvc_stl_support &other) + { + T::operator=(other); + return *this; + } }; - } +}// namespace internal #else @@ -79,6 +77,6 @@ namespace Eigen { #endif -} +}// namespace Eigen -#endif // EIGEN_STL_DETAILS_H +#endif// EIGEN_STL_DETAILS_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/SuperLUSupport/SuperLUSupport.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/SuperLUSupport/SuperLUSupport.h index 7261c7d0..a15d9c54 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/SuperLUSupport/SuperLUSupport.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/SuperLUSupport/SuperLUSupport.h @@ -13,58 +13,156 @@ namespace Eigen { #if defined(SUPERLU_MAJOR_VERSION) && (SUPERLU_MAJOR_VERSION >= 5) -#define DECL_GSSVX(PREFIX,FLOATTYPE,KEYTYPE) \ - extern "C" { \ - extern void PREFIX##gssvx(superlu_options_t *, SuperMatrix *, int *, int *, int *, \ - char *, FLOATTYPE *, FLOATTYPE *, SuperMatrix *, SuperMatrix *, \ - void *, int, SuperMatrix *, SuperMatrix *, \ - FLOATTYPE *, FLOATTYPE *, FLOATTYPE *, FLOATTYPE *, \ - GlobalLU_t *, mem_usage_t *, SuperLUStat_t *, int *); \ - } \ - inline float SuperLU_gssvx(superlu_options_t *options, SuperMatrix *A, \ - int *perm_c, int *perm_r, int *etree, char *equed, \ - FLOATTYPE *R, FLOATTYPE *C, SuperMatrix *L, \ - SuperMatrix *U, void *work, int lwork, \ - SuperMatrix *B, SuperMatrix *X, \ - FLOATTYPE *recip_pivot_growth, \ - FLOATTYPE *rcond, FLOATTYPE *ferr, FLOATTYPE *berr, \ - SuperLUStat_t *stats, int *info, KEYTYPE) { \ - mem_usage_t mem_usage; \ - GlobalLU_t gLU; \ - PREFIX##gssvx(options, A, perm_c, perm_r, etree, equed, R, C, L, \ - U, work, lwork, B, X, recip_pivot_growth, rcond, \ - ferr, berr, &gLU, &mem_usage, stats, info); \ - return mem_usage.for_lu; /* bytes used by the factor storage */ \ +#define DECL_GSSVX(PREFIX, FLOATTYPE, KEYTYPE) \ + extern "C" { \ + extern void PREFIX##gssvx(superlu_options_t *, \ + SuperMatrix *, \ + int *, \ + int *, \ + int *, \ + char *, \ + FLOATTYPE *, \ + FLOATTYPE *, \ + SuperMatrix *, \ + SuperMatrix *, \ + void *, \ + int, \ + SuperMatrix *, \ + SuperMatrix *, \ + FLOATTYPE *, \ + FLOATTYPE *, \ + FLOATTYPE *, \ + FLOATTYPE *, \ + GlobalLU_t *, \ + mem_usage_t *, \ + SuperLUStat_t *, \ + int *); \ + } \ + inline float SuperLU_gssvx(superlu_options_t *options, \ + SuperMatrix *A, \ + int *perm_c, \ + int *perm_r, \ + int *etree, \ + char *equed, \ + FLOATTYPE *R, \ + FLOATTYPE *C, \ + SuperMatrix *L, \ + SuperMatrix *U, \ + void *work, \ + int lwork, \ + SuperMatrix *B, \ + SuperMatrix *X, \ + FLOATTYPE *recip_pivot_growth, \ + FLOATTYPE *rcond, \ + FLOATTYPE *ferr, \ + FLOATTYPE *berr, \ + SuperLUStat_t *stats, \ + int *info, \ + KEYTYPE) \ + { \ + mem_usage_t mem_usage; \ + GlobalLU_t gLU; \ + PREFIX##gssvx(options, \ + A, \ + perm_c, \ + perm_r, \ + etree, \ + equed, \ + R, \ + C, \ + L, \ + U, \ + work, \ + lwork, \ + B, \ + X, \ + recip_pivot_growth, \ + rcond, \ + ferr, \ + berr, \ + &gLU, \ + &mem_usage, \ + stats, \ + info); \ + return mem_usage.for_lu; /* bytes used by the factor storage */ \ } -#else // version < 5.0 -#define DECL_GSSVX(PREFIX,FLOATTYPE,KEYTYPE) \ - extern "C" { \ - extern void PREFIX##gssvx(superlu_options_t *, SuperMatrix *, int *, int *, int *, \ - char *, FLOATTYPE *, FLOATTYPE *, SuperMatrix *, SuperMatrix *, \ - void *, int, SuperMatrix *, SuperMatrix *, \ - FLOATTYPE *, FLOATTYPE *, FLOATTYPE *, FLOATTYPE *, \ - mem_usage_t *, SuperLUStat_t *, int *); \ - } \ - inline float SuperLU_gssvx(superlu_options_t *options, SuperMatrix *A, \ - int *perm_c, int *perm_r, int *etree, char *equed, \ - FLOATTYPE *R, FLOATTYPE *C, SuperMatrix *L, \ - SuperMatrix *U, void *work, int lwork, \ - SuperMatrix *B, SuperMatrix *X, \ - FLOATTYPE *recip_pivot_growth, \ - FLOATTYPE *rcond, FLOATTYPE *ferr, FLOATTYPE *berr, \ - SuperLUStat_t *stats, int *info, KEYTYPE) { \ - mem_usage_t mem_usage; \ - PREFIX##gssvx(options, A, perm_c, perm_r, etree, equed, R, C, L, \ - U, work, lwork, B, X, recip_pivot_growth, rcond, \ - ferr, berr, &mem_usage, stats, info); \ - return mem_usage.for_lu; /* bytes used by the factor storage */ \ +#else// version < 5.0 +#define DECL_GSSVX(PREFIX, FLOATTYPE, KEYTYPE) \ + extern "C" { \ + extern void PREFIX##gssvx(superlu_options_t *, \ + SuperMatrix *, \ + int *, \ + int *, \ + int *, \ + char *, \ + FLOATTYPE *, \ + FLOATTYPE *, \ + SuperMatrix *, \ + SuperMatrix *, \ + void *, \ + int, \ + SuperMatrix *, \ + SuperMatrix *, \ + FLOATTYPE *, \ + FLOATTYPE *, \ + FLOATTYPE *, \ + FLOATTYPE *, \ + mem_usage_t *, \ + SuperLUStat_t *, \ + int *); \ + } \ + inline float SuperLU_gssvx(superlu_options_t *options, \ + SuperMatrix *A, \ + int *perm_c, \ + int *perm_r, \ + int *etree, \ + char *equed, \ + FLOATTYPE *R, \ + FLOATTYPE *C, \ + SuperMatrix *L, \ + SuperMatrix *U, \ + void *work, \ + int lwork, \ + SuperMatrix *B, \ + SuperMatrix *X, \ + FLOATTYPE *recip_pivot_growth, \ + FLOATTYPE *rcond, \ + FLOATTYPE *ferr, \ + FLOATTYPE *berr, \ + SuperLUStat_t *stats, \ + int *info, \ + KEYTYPE) \ + { \ + mem_usage_t mem_usage; \ + PREFIX##gssvx(options, \ + A, \ + perm_c, \ + perm_r, \ + etree, \ + equed, \ + R, \ + C, \ + L, \ + U, \ + work, \ + lwork, \ + B, \ + X, \ + recip_pivot_growth, \ + rcond, \ + ferr, \ + berr, \ + &mem_usage, \ + stats, \ + info); \ + return mem_usage.for_lu; /* bytes used by the factor storage */ \ } #endif -DECL_GSSVX(s,float,float) -DECL_GSSVX(c,float,std::complex) -DECL_GSSVX(d,double,double) -DECL_GSSVX(z,double,std::complex) +DECL_GSSVX(s, float, float) +DECL_GSSVX(c, float, std::complex) +DECL_GSSVX(d, double, double) +DECL_GSSVX(z, double, std::complex) #ifdef MILU_ALPHA #define EIGEN_SUPERLU_HAS_ILU @@ -73,62 +171,100 @@ DECL_GSSVX(z,double,std::complex) #ifdef EIGEN_SUPERLU_HAS_ILU // similarly for the incomplete factorization using gsisx -#define DECL_GSISX(PREFIX,FLOATTYPE,KEYTYPE) \ - extern "C" { \ - extern void PREFIX##gsisx(superlu_options_t *, SuperMatrix *, int *, int *, int *, \ - char *, FLOATTYPE *, FLOATTYPE *, SuperMatrix *, SuperMatrix *, \ - void *, int, SuperMatrix *, SuperMatrix *, FLOATTYPE *, FLOATTYPE *, \ - mem_usage_t *, SuperLUStat_t *, int *); \ - } \ - inline float SuperLU_gsisx(superlu_options_t *options, SuperMatrix *A, \ - int *perm_c, int *perm_r, int *etree, char *equed, \ - FLOATTYPE *R, FLOATTYPE *C, SuperMatrix *L, \ - SuperMatrix *U, void *work, int lwork, \ - SuperMatrix *B, SuperMatrix *X, \ - FLOATTYPE *recip_pivot_growth, \ - FLOATTYPE *rcond, \ - SuperLUStat_t *stats, int *info, KEYTYPE) { \ - mem_usage_t mem_usage; \ - PREFIX##gsisx(options, A, perm_c, perm_r, etree, equed, R, C, L, \ - U, work, lwork, B, X, recip_pivot_growth, rcond, \ - &mem_usage, stats, info); \ - return mem_usage.for_lu; /* bytes used by the factor storage */ \ +#define DECL_GSISX(PREFIX, FLOATTYPE, KEYTYPE) \ + extern "C" { \ + extern void PREFIX##gsisx(superlu_options_t *, \ + SuperMatrix *, \ + int *, \ + int *, \ + int *, \ + char *, \ + FLOATTYPE *, \ + FLOATTYPE *, \ + SuperMatrix *, \ + SuperMatrix *, \ + void *, \ + int, \ + SuperMatrix *, \ + SuperMatrix *, \ + FLOATTYPE *, \ + FLOATTYPE *, \ + mem_usage_t *, \ + SuperLUStat_t *, \ + int *); \ + } \ + inline float SuperLU_gsisx(superlu_options_t *options, \ + SuperMatrix *A, \ + int *perm_c, \ + int *perm_r, \ + int *etree, \ + char *equed, \ + FLOATTYPE *R, \ + FLOATTYPE *C, \ + SuperMatrix *L, \ + SuperMatrix *U, \ + void *work, \ + int lwork, \ + SuperMatrix *B, \ + SuperMatrix *X, \ + FLOATTYPE *recip_pivot_growth, \ + FLOATTYPE *rcond, \ + SuperLUStat_t *stats, \ + int *info, \ + KEYTYPE) \ + { \ + mem_usage_t mem_usage; \ + PREFIX##gsisx(options, \ + A, \ + perm_c, \ + perm_r, \ + etree, \ + equed, \ + R, \ + C, \ + L, \ + U, \ + work, \ + lwork, \ + B, \ + X, \ + recip_pivot_growth, \ + rcond, \ + &mem_usage, \ + stats, \ + info); \ + return mem_usage.for_lu; /* bytes used by the factor storage */ \ } -DECL_GSISX(s,float,float) -DECL_GSISX(c,float,std::complex) -DECL_GSISX(d,double,double) -DECL_GSISX(z,double,std::complex) +DECL_GSISX(s, float, float) +DECL_GSISX(c, float, std::complex) +DECL_GSISX(d, double, double) +DECL_GSISX(z, double, std::complex) #endif -template -struct SluMatrixMapHelper; +template struct SluMatrixMapHelper; /** \internal - * - * A wrapper class for SuperLU matrices. It supports only compressed sparse matrices - * and dense matrices. Supernodal and other fancy format are not supported by this wrapper. - * - * This wrapper class mainly aims to avoids the need of dynamic allocation of the storage structure. - */ + * + * A wrapper class for SuperLU matrices. It supports only compressed sparse matrices + * and dense matrices. Supernodal and other fancy format are not supported by this wrapper. + * + * This wrapper class mainly aims to avoids the need of dynamic allocation of the storage structure. + */ struct SluMatrix : SuperMatrix { - SluMatrix() - { - Store = &storage; - } + SluMatrix() { Store = &storage; } - SluMatrix(const SluMatrix& other) - : SuperMatrix(other) + SluMatrix(const SluMatrix &other) : SuperMatrix(other) { Store = &storage; storage = other.storage; } - SluMatrix& operator=(const SluMatrix& other) + SluMatrix &operator=(const SluMatrix &other) { - SuperMatrix::operator=(static_cast(other)); + SuperMatrix::operator=(static_cast(other)); Store = &storage; storage = other.storage; return *this; @@ -136,7 +272,10 @@ struct SluMatrix : SuperMatrix struct { - union {int nnz;int lda;}; + union { + int nnz; + int lda; + }; void *values; int *innerInd; int *outerInd; @@ -145,496 +284,474 @@ struct SluMatrix : SuperMatrix void setStorageType(Stype_t t) { Stype = t; - if (t==SLU_NC || t==SLU_NR || t==SLU_DN) + if (t == SLU_NC || t == SLU_NR || t == SLU_DN) Store = &storage; - else - { + else { eigen_assert(false && "storage type not supported"); Store = 0; } } - template - void setScalarType() + template void setScalarType() { - if (internal::is_same::value) + if (internal::is_same::value) Dtype = SLU_S; - else if (internal::is_same::value) + else if (internal::is_same::value) Dtype = SLU_D; - else if (internal::is_same >::value) + else if (internal::is_same>::value) Dtype = SLU_C; - else if (internal::is_same >::value) + else if (internal::is_same>::value) Dtype = SLU_Z; - else - { + else { eigen_assert(false && "Scalar type not supported by SuperLU"); } } - template - static SluMatrix Map(MatrixBase& _mat) + template static SluMatrix Map(MatrixBase &_mat) { - MatrixType& mat(_mat.derived()); - eigen_assert( ((MatrixType::Flags&RowMajorBit)!=RowMajorBit) && "row-major dense matrices are not supported by SuperLU"); + MatrixType &mat(_mat.derived()); + eigen_assert( + ((MatrixType::Flags & RowMajorBit) != RowMajorBit) && "row-major dense matrices are not supported by SuperLU"); SluMatrix res; res.setStorageType(SLU_DN); res.setScalarType(); - res.Mtype = SLU_GE; + res.Mtype = SLU_GE; - res.nrow = internal::convert_index(mat.rows()); - res.ncol = internal::convert_index(mat.cols()); + res.nrow = internal::convert_index(mat.rows()); + res.ncol = internal::convert_index(mat.cols()); - res.storage.lda = internal::convert_index(MatrixType::IsVectorAtCompileTime ? mat.size() : mat.outerStride()); - res.storage.values = (void*)(mat.data()); + res.storage.lda = internal::convert_index(MatrixType::IsVectorAtCompileTime ? mat.size() : mat.outerStride()); + res.storage.values = (void *)(mat.data()); return res; } - template - static SluMatrix Map(SparseMatrixBase& a_mat) + template static SluMatrix Map(SparseMatrixBase &a_mat) { MatrixType &mat(a_mat.derived()); SluMatrix res; - if ((MatrixType::Flags&RowMajorBit)==RowMajorBit) - { + if ((MatrixType::Flags & RowMajorBit) == RowMajorBit) { res.setStorageType(SLU_NR); - res.nrow = internal::convert_index(mat.cols()); - res.ncol = internal::convert_index(mat.rows()); - } - else - { + res.nrow = internal::convert_index(mat.cols()); + res.ncol = internal::convert_index(mat.rows()); + } else { res.setStorageType(SLU_NC); - res.nrow = internal::convert_index(mat.rows()); - res.ncol = internal::convert_index(mat.cols()); + res.nrow = internal::convert_index(mat.rows()); + res.ncol = internal::convert_index(mat.cols()); } - res.Mtype = SLU_GE; + res.Mtype = SLU_GE; - res.storage.nnz = internal::convert_index(mat.nonZeros()); - res.storage.values = mat.valuePtr(); - res.storage.innerInd = mat.innerIndexPtr(); - res.storage.outerInd = mat.outerIndexPtr(); + res.storage.nnz = internal::convert_index(mat.nonZeros()); + res.storage.values = mat.valuePtr(); + res.storage.innerInd = mat.innerIndexPtr(); + res.storage.outerInd = mat.outerIndexPtr(); res.setScalarType(); // FIXME the following is not very accurate - if (MatrixType::Flags & Upper) - res.Mtype = SLU_TRU; - if (MatrixType::Flags & Lower) - res.Mtype = SLU_TRL; + if (MatrixType::Flags & Upper) res.Mtype = SLU_TRU; + if (MatrixType::Flags & Lower) res.Mtype = SLU_TRL; - eigen_assert(((MatrixType::Flags & SelfAdjoint)==0) && "SelfAdjoint matrix shape not supported by SuperLU"); + eigen_assert(((MatrixType::Flags & SelfAdjoint) == 0) && "SelfAdjoint matrix shape not supported by SuperLU"); return res; } }; template -struct SluMatrixMapHelper > +struct SluMatrixMapHelper> { - typedef Matrix MatrixType; - static void run(MatrixType& mat, SluMatrix& res) + typedef Matrix MatrixType; + static void run(MatrixType &mat, SluMatrix &res) { - eigen_assert( ((Options&RowMajor)!=RowMajor) && "row-major dense matrices is not supported by SuperLU"); + eigen_assert(((Options & RowMajor) != RowMajor) && "row-major dense matrices is not supported by SuperLU"); res.setStorageType(SLU_DN); res.setScalarType(); - res.Mtype = SLU_GE; + res.Mtype = SLU_GE; - res.nrow = mat.rows(); - res.ncol = mat.cols(); + res.nrow = mat.rows(); + res.ncol = mat.cols(); - res.storage.lda = mat.outerStride(); - res.storage.values = mat.data(); + res.storage.lda = mat.outerStride(); + res.storage.values = mat.data(); } }; -template -struct SluMatrixMapHelper > +template struct SluMatrixMapHelper> { typedef Derived MatrixType; - static void run(MatrixType& mat, SluMatrix& res) + static void run(MatrixType &mat, SluMatrix &res) { - if ((MatrixType::Flags&RowMajorBit)==RowMajorBit) - { + if ((MatrixType::Flags & RowMajorBit) == RowMajorBit) { res.setStorageType(SLU_NR); - res.nrow = mat.cols(); - res.ncol = mat.rows(); - } - else - { + res.nrow = mat.cols(); + res.ncol = mat.rows(); + } else { res.setStorageType(SLU_NC); - res.nrow = mat.rows(); - res.ncol = mat.cols(); + res.nrow = mat.rows(); + res.ncol = mat.cols(); } - res.Mtype = SLU_GE; + res.Mtype = SLU_GE; - res.storage.nnz = mat.nonZeros(); - res.storage.values = mat.valuePtr(); - res.storage.innerInd = mat.innerIndexPtr(); - res.storage.outerInd = mat.outerIndexPtr(); + res.storage.nnz = mat.nonZeros(); + res.storage.values = mat.valuePtr(); + res.storage.innerInd = mat.innerIndexPtr(); + res.storage.outerInd = mat.outerIndexPtr(); res.setScalarType(); // FIXME the following is not very accurate - if (MatrixType::Flags & Upper) - res.Mtype = SLU_TRU; - if (MatrixType::Flags & Lower) - res.Mtype = SLU_TRL; + if (MatrixType::Flags & Upper) res.Mtype = SLU_TRU; + if (MatrixType::Flags & Lower) res.Mtype = SLU_TRL; - eigen_assert(((MatrixType::Flags & SelfAdjoint)==0) && "SelfAdjoint matrix shape not supported by SuperLU"); + eigen_assert(((MatrixType::Flags & SelfAdjoint) == 0) && "SelfAdjoint matrix shape not supported by SuperLU"); } }; namespace internal { -template -SluMatrix asSluMatrix(MatrixType& mat) -{ - return SluMatrix::Map(mat); -} + template SluMatrix asSluMatrix(MatrixType &mat) { return SluMatrix::Map(mat); } -/** View a Super LU matrix as an Eigen expression */ -template -MappedSparseMatrix map_superlu(SluMatrix& sluMat) -{ - eigen_assert(((Flags&RowMajor)==RowMajor && sluMat.Stype == SLU_NR) - || ((Flags&ColMajor)==ColMajor && sluMat.Stype == SLU_NC)); + /** View a Super LU matrix as an Eigen expression */ + template + MappedSparseMatrix map_superlu(SluMatrix &sluMat) + { + eigen_assert(((Flags & RowMajor) == RowMajor && sluMat.Stype == SLU_NR) + || ((Flags & ColMajor) == ColMajor && sluMat.Stype == SLU_NC)); - Index outerSize = (Flags&RowMajor)==RowMajor ? sluMat.ncol : sluMat.nrow; + Index outerSize = (Flags & RowMajor) == RowMajor ? sluMat.ncol : sluMat.nrow; - return MappedSparseMatrix( - sluMat.nrow, sluMat.ncol, sluMat.storage.outerInd[outerSize], - sluMat.storage.outerInd, sluMat.storage.innerInd, reinterpret_cast(sluMat.storage.values) ); -} + return MappedSparseMatrix(sluMat.nrow, + sluMat.ncol, + sluMat.storage.outerInd[outerSize], + sluMat.storage.outerInd, + sluMat.storage.innerInd, + reinterpret_cast(sluMat.storage.values)); + } -} // end namespace internal +}// end namespace internal /** \ingroup SuperLUSupport_Module - * \class SuperLUBase - * \brief The base class for the direct and incomplete LU factorization of SuperLU - */ -template -class SuperLUBase : public SparseSolverBase + * \class SuperLUBase + * \brief The base class for the direct and incomplete LU factorization of SuperLU + */ +template class SuperLUBase : public SparseSolverBase { - protected: - typedef SparseSolverBase Base; - using Base::derived; - using Base::m_isInitialized; - public: - typedef _MatrixType MatrixType; - typedef typename MatrixType::Scalar Scalar; - typedef typename MatrixType::RealScalar RealScalar; - typedef typename MatrixType::StorageIndex StorageIndex; - typedef Matrix Vector; - typedef Matrix IntRowVectorType; - typedef Matrix IntColVectorType; - typedef Map > PermutationMap; - typedef SparseMatrix LUMatrixType; - enum { - ColsAtCompileTime = MatrixType::ColsAtCompileTime, - MaxColsAtCompileTime = MatrixType::MaxColsAtCompileTime - }; +protected: + typedef SparseSolverBase Base; + using Base::derived; + using Base::m_isInitialized; + +public: + typedef _MatrixType MatrixType; + typedef typename MatrixType::Scalar Scalar; + typedef typename MatrixType::RealScalar RealScalar; + typedef typename MatrixType::StorageIndex StorageIndex; + typedef Matrix Vector; + typedef Matrix IntRowVectorType; + typedef Matrix IntColVectorType; + typedef Map> PermutationMap; + typedef SparseMatrix LUMatrixType; + enum { ColsAtCompileTime = MatrixType::ColsAtCompileTime, MaxColsAtCompileTime = MatrixType::MaxColsAtCompileTime }; + +public: + SuperLUBase() {} + + ~SuperLUBase() { clearFactors(); } + + inline Index rows() const { return m_matrix.rows(); } + inline Index cols() const { return m_matrix.cols(); } + + /** \returns a reference to the Super LU option object to configure the Super LU algorithms. */ + inline superlu_options_t &options() { return m_sluOptions; } + + /** \brief Reports whether previous computation was successful. + * + * \returns \c Success if computation was succesful, + * \c NumericalIssue if the matrix.appears to be negative. + */ + ComputationInfo info() const + { + eigen_assert(m_isInitialized && "Decomposition is not initialized."); + return m_info; + } - public: + /** Computes the sparse Cholesky decomposition of \a matrix */ + void compute(const MatrixType &matrix) + { + derived().analyzePattern(matrix); + derived().factorize(matrix); + } - SuperLUBase() {} + /** Performs a symbolic decomposition on the sparcity of \a matrix. + * + * This function is particularly useful when solving for several problems having the same structure. + * + * \sa factorize() + */ + void analyzePattern(const MatrixType & /*matrix*/) + { + m_isInitialized = true; + m_info = Success; + m_analysisIsOk = true; + m_factorizationIsOk = false; + } - ~SuperLUBase() - { - clearFactors(); - } - - inline Index rows() const { return m_matrix.rows(); } - inline Index cols() const { return m_matrix.cols(); } - - /** \returns a reference to the Super LU option object to configure the Super LU algorithms. */ - inline superlu_options_t& options() { return m_sluOptions; } - - /** \brief Reports whether previous computation was successful. - * - * \returns \c Success if computation was succesful, - * \c NumericalIssue if the matrix.appears to be negative. - */ - ComputationInfo info() const - { - eigen_assert(m_isInitialized && "Decomposition is not initialized."); - return m_info; - } + template void dumpMemory(Stream & /*s*/) {} - /** Computes the sparse Cholesky decomposition of \a matrix */ - void compute(const MatrixType& matrix) - { - derived().analyzePattern(matrix); - derived().factorize(matrix); - } +protected: + void initFactorization(const MatrixType &a) + { + set_default_options(&this->m_sluOptions); + + const Index size = a.rows(); + m_matrix = a; + + m_sluA = internal::asSluMatrix(m_matrix); + clearFactors(); + + m_p.resize(size); + m_q.resize(size); + m_sluRscale.resize(size); + m_sluCscale.resize(size); + m_sluEtree.resize(size); + + // set empty B and X + m_sluB.setStorageType(SLU_DN); + m_sluB.setScalarType(); + m_sluB.Mtype = SLU_GE; + m_sluB.storage.values = 0; + m_sluB.nrow = 0; + m_sluB.ncol = 0; + m_sluB.storage.lda = internal::convert_index(size); + m_sluX = m_sluB; + + m_extractedDataAreDirty = true; + } - /** Performs a symbolic decomposition on the sparcity of \a matrix. - * - * This function is particularly useful when solving for several problems having the same structure. - * - * \sa factorize() - */ - void analyzePattern(const MatrixType& /*matrix*/) - { - m_isInitialized = true; - m_info = Success; - m_analysisIsOk = true; - m_factorizationIsOk = false; - } - - template - void dumpMemory(Stream& /*s*/) - {} - - protected: - - void initFactorization(const MatrixType& a) - { - set_default_options(&this->m_sluOptions); - - const Index size = a.rows(); - m_matrix = a; - - m_sluA = internal::asSluMatrix(m_matrix); - clearFactors(); - - m_p.resize(size); - m_q.resize(size); - m_sluRscale.resize(size); - m_sluCscale.resize(size); - m_sluEtree.resize(size); - - // set empty B and X - m_sluB.setStorageType(SLU_DN); - m_sluB.setScalarType(); - m_sluB.Mtype = SLU_GE; - m_sluB.storage.values = 0; - m_sluB.nrow = 0; - m_sluB.ncol = 0; - m_sluB.storage.lda = internal::convert_index(size); - m_sluX = m_sluB; - - m_extractedDataAreDirty = true; - } - - void init() - { - m_info = InvalidInput; - m_isInitialized = false; - m_sluL.Store = 0; - m_sluU.Store = 0; - } - - void extractData() const; + void init() + { + m_info = InvalidInput; + m_isInitialized = false; + m_sluL.Store = 0; + m_sluU.Store = 0; + } - void clearFactors() - { - if(m_sluL.Store) - Destroy_SuperNode_Matrix(&m_sluL); - if(m_sluU.Store) - Destroy_CompCol_Matrix(&m_sluU); + void extractData() const; - m_sluL.Store = 0; - m_sluU.Store = 0; + void clearFactors() + { + if (m_sluL.Store) Destroy_SuperNode_Matrix(&m_sluL); + if (m_sluU.Store) Destroy_CompCol_Matrix(&m_sluU); - memset(&m_sluL,0,sizeof m_sluL); - memset(&m_sluU,0,sizeof m_sluU); - } + m_sluL.Store = 0; + m_sluU.Store = 0; - // cached data to reduce reallocation, etc. - mutable LUMatrixType m_l; - mutable LUMatrixType m_u; - mutable IntColVectorType m_p; - mutable IntRowVectorType m_q; - - mutable LUMatrixType m_matrix; // copy of the factorized matrix - mutable SluMatrix m_sluA; - mutable SuperMatrix m_sluL, m_sluU; - mutable SluMatrix m_sluB, m_sluX; - mutable SuperLUStat_t m_sluStat; - mutable superlu_options_t m_sluOptions; - mutable std::vector m_sluEtree; - mutable Matrix m_sluRscale, m_sluCscale; - mutable Matrix m_sluFerr, m_sluBerr; - mutable char m_sluEqued; - - mutable ComputationInfo m_info; - int m_factorizationIsOk; - int m_analysisIsOk; - mutable bool m_extractedDataAreDirty; - - private: - SuperLUBase(SuperLUBase& ) { } + memset(&m_sluL, 0, sizeof m_sluL); + memset(&m_sluU, 0, sizeof m_sluU); + } + + // cached data to reduce reallocation, etc. + mutable LUMatrixType m_l; + mutable LUMatrixType m_u; + mutable IntColVectorType m_p; + mutable IntRowVectorType m_q; + + mutable LUMatrixType m_matrix;// copy of the factorized matrix + mutable SluMatrix m_sluA; + mutable SuperMatrix m_sluL, m_sluU; + mutable SluMatrix m_sluB, m_sluX; + mutable SuperLUStat_t m_sluStat; + mutable superlu_options_t m_sluOptions; + mutable std::vector m_sluEtree; + mutable Matrix m_sluRscale, m_sluCscale; + mutable Matrix m_sluFerr, m_sluBerr; + mutable char m_sluEqued; + + mutable ComputationInfo m_info; + int m_factorizationIsOk; + int m_analysisIsOk; + mutable bool m_extractedDataAreDirty; + +private: + SuperLUBase(SuperLUBase &) {} }; /** \ingroup SuperLUSupport_Module - * \class SuperLU - * \brief A sparse direct LU factorization and solver based on the SuperLU library - * - * This class allows to solve for A.X = B sparse linear problems via a direct LU factorization - * using the SuperLU library. The sparse matrix A must be squared and invertible. The vectors or matrices - * X and B can be either dense or sparse. - * - * \tparam _MatrixType the type of the sparse matrix A, it must be a SparseMatrix<> - * - * \warning This class is only for the 4.x versions of SuperLU. The 3.x and 5.x versions are not supported. - * - * \implsparsesolverconcept - * - * \sa \ref TutorialSparseSolverConcept, class SparseLU - */ -template -class SuperLU : public SuperLUBase<_MatrixType,SuperLU<_MatrixType> > + * \class SuperLU + * \brief A sparse direct LU factorization and solver based on the SuperLU library + * + * This class allows to solve for A.X = B sparse linear problems via a direct LU factorization + * using the SuperLU library. The sparse matrix A must be squared and invertible. The vectors or matrices + * X and B can be either dense or sparse. + * + * \tparam _MatrixType the type of the sparse matrix A, it must be a SparseMatrix<> + * + * \warning This class is only for the 4.x versions of SuperLU. The 3.x and 5.x versions are not supported. + * + * \implsparsesolverconcept + * + * \sa \ref TutorialSparseSolverConcept, class SparseLU + */ +template class SuperLU : public SuperLUBase<_MatrixType, SuperLU<_MatrixType>> { - public: - typedef SuperLUBase<_MatrixType,SuperLU> Base; - typedef _MatrixType MatrixType; - typedef typename Base::Scalar Scalar; - typedef typename Base::RealScalar RealScalar; - typedef typename Base::StorageIndex StorageIndex; - typedef typename Base::IntRowVectorType IntRowVectorType; - typedef typename Base::IntColVectorType IntColVectorType; - typedef typename Base::PermutationMap PermutationMap; - typedef typename Base::LUMatrixType LUMatrixType; - typedef TriangularView LMatrixType; - typedef TriangularView UMatrixType; - - public: - using Base::_solve_impl; - - SuperLU() : Base() { init(); } - - explicit SuperLU(const MatrixType& matrix) : Base() - { - init(); - Base::compute(matrix); - } +public: + typedef SuperLUBase<_MatrixType, SuperLU> Base; + typedef _MatrixType MatrixType; + typedef typename Base::Scalar Scalar; + typedef typename Base::RealScalar RealScalar; + typedef typename Base::StorageIndex StorageIndex; + typedef typename Base::IntRowVectorType IntRowVectorType; + typedef typename Base::IntColVectorType IntColVectorType; + typedef typename Base::PermutationMap PermutationMap; + typedef typename Base::LUMatrixType LUMatrixType; + typedef TriangularView LMatrixType; + typedef TriangularView UMatrixType; + +public: + using Base::_solve_impl; + + SuperLU() : Base() { init(); } + + explicit SuperLU(const MatrixType &matrix) : Base() + { + init(); + Base::compute(matrix); + } - ~SuperLU() - { - } - - /** Performs a symbolic decomposition on the sparcity of \a matrix. - * - * This function is particularly useful when solving for several problems having the same structure. - * - * \sa factorize() - */ - void analyzePattern(const MatrixType& matrix) - { - m_info = InvalidInput; - m_isInitialized = false; - Base::analyzePattern(matrix); - } - - /** Performs a numeric decomposition of \a matrix - * - * The given matrix must has the same sparcity than the matrix on which the symbolic decomposition has been performed. - * - * \sa analyzePattern() - */ - void factorize(const MatrixType& matrix); - - /** \internal */ - template - void _solve_impl(const MatrixBase &b, MatrixBase &dest) const; - - inline const LMatrixType& matrixL() const - { - if (m_extractedDataAreDirty) this->extractData(); - return m_l; - } + ~SuperLU() {} - inline const UMatrixType& matrixU() const - { - if (m_extractedDataAreDirty) this->extractData(); - return m_u; - } + /** Performs a symbolic decomposition on the sparcity of \a matrix. + * + * This function is particularly useful when solving for several problems having the same structure. + * + * \sa factorize() + */ + void analyzePattern(const MatrixType &matrix) + { + m_info = InvalidInput; + m_isInitialized = false; + Base::analyzePattern(matrix); + } - inline const IntColVectorType& permutationP() const - { - if (m_extractedDataAreDirty) this->extractData(); - return m_p; - } + /** Performs a numeric decomposition of \a matrix + * + * The given matrix must has the same sparcity than the matrix on which the symbolic decomposition has been performed. + * + * \sa analyzePattern() + */ + void factorize(const MatrixType &matrix); - inline const IntRowVectorType& permutationQ() const - { - if (m_extractedDataAreDirty) this->extractData(); - return m_q; - } - - Scalar determinant() const; - - protected: - - using Base::m_matrix; - using Base::m_sluOptions; - using Base::m_sluA; - using Base::m_sluB; - using Base::m_sluX; - using Base::m_p; - using Base::m_q; - using Base::m_sluEtree; - using Base::m_sluEqued; - using Base::m_sluRscale; - using Base::m_sluCscale; - using Base::m_sluL; - using Base::m_sluU; - using Base::m_sluStat; - using Base::m_sluFerr; - using Base::m_sluBerr; - using Base::m_l; - using Base::m_u; - - using Base::m_analysisIsOk; - using Base::m_factorizationIsOk; - using Base::m_extractedDataAreDirty; - using Base::m_isInitialized; - using Base::m_info; - - void init() - { - Base::init(); - - set_default_options(&this->m_sluOptions); - m_sluOptions.PrintStat = NO; - m_sluOptions.ConditionNumber = NO; - m_sluOptions.Trans = NOTRANS; - m_sluOptions.ColPerm = COLAMD; - } - - - private: - SuperLU(SuperLU& ) { } + /** \internal */ + template void _solve_impl(const MatrixBase &b, MatrixBase &dest) const; + + inline const LMatrixType &matrixL() const + { + if (m_extractedDataAreDirty) this->extractData(); + return m_l; + } + + inline const UMatrixType &matrixU() const + { + if (m_extractedDataAreDirty) this->extractData(); + return m_u; + } + + inline const IntColVectorType &permutationP() const + { + if (m_extractedDataAreDirty) this->extractData(); + return m_p; + } + + inline const IntRowVectorType &permutationQ() const + { + if (m_extractedDataAreDirty) this->extractData(); + return m_q; + } + + Scalar determinant() const; + +protected: + using Base::m_matrix; + using Base::m_sluOptions; + using Base::m_sluA; + using Base::m_sluB; + using Base::m_sluX; + using Base::m_p; + using Base::m_q; + using Base::m_sluEtree; + using Base::m_sluEqued; + using Base::m_sluRscale; + using Base::m_sluCscale; + using Base::m_sluL; + using Base::m_sluU; + using Base::m_sluStat; + using Base::m_sluFerr; + using Base::m_sluBerr; + using Base::m_l; + using Base::m_u; + + using Base::m_analysisIsOk; + using Base::m_factorizationIsOk; + using Base::m_extractedDataAreDirty; + using Base::m_isInitialized; + using Base::m_info; + + void init() + { + Base::init(); + + set_default_options(&this->m_sluOptions); + m_sluOptions.PrintStat = NO; + m_sluOptions.ConditionNumber = NO; + m_sluOptions.Trans = NOTRANS; + m_sluOptions.ColPerm = COLAMD; + } + + +private: + SuperLU(SuperLU &) {} }; -template -void SuperLU::factorize(const MatrixType& a) +template void SuperLU::factorize(const MatrixType &a) { eigen_assert(m_analysisIsOk && "You must first call analyzePattern()"); - if(!m_analysisIsOk) - { + if (!m_analysisIsOk) { m_info = InvalidInput; return; } - + this->initFactorization(a); - + m_sluOptions.ColPerm = COLAMD; int info = 0; RealScalar recip_pivot_growth, rcond; RealScalar ferr, berr; StatInit(&m_sluStat); - SuperLU_gssvx(&m_sluOptions, &m_sluA, m_q.data(), m_p.data(), &m_sluEtree[0], - &m_sluEqued, &m_sluRscale[0], &m_sluCscale[0], - &m_sluL, &m_sluU, - NULL, 0, - &m_sluB, &m_sluX, - &recip_pivot_growth, &rcond, - &ferr, &berr, - &m_sluStat, &info, Scalar()); + SuperLU_gssvx(&m_sluOptions, + &m_sluA, + m_q.data(), + m_p.data(), + &m_sluEtree[0], + &m_sluEqued, + &m_sluRscale[0], + &m_sluCscale[0], + &m_sluL, + &m_sluU, + NULL, + 0, + &m_sluB, + &m_sluX, + &recip_pivot_growth, + &rcond, + &ferr, + &berr, + &m_sluStat, + &info, + Scalar()); StatFree(&m_sluStat); m_extractedDataAreDirty = true; @@ -645,55 +762,64 @@ void SuperLU::factorize(const MatrixType& a) } template -template -void SuperLU::_solve_impl(const MatrixBase &b, MatrixBase& x) const +template +void SuperLU::_solve_impl(const MatrixBase &b, MatrixBase &x) const { eigen_assert(m_factorizationIsOk && "The decomposition is not in a valid state for solving, you must first call either compute() or analyzePattern()/factorize()"); const Index size = m_matrix.rows(); const Index rhsCols = b.cols(); - eigen_assert(size==b.rows()); + eigen_assert(size == b.rows()); m_sluOptions.Trans = NOTRANS; m_sluOptions.Fact = FACTORED; m_sluOptions.IterRefine = NOREFINE; - + m_sluFerr.resize(rhsCols); m_sluBerr.resize(rhsCols); - - Ref > b_ref(b); - Ref > x_ref(x); - + + Ref> b_ref(b); + Ref> x_ref(x); + m_sluB = SluMatrix::Map(b_ref.const_cast_derived()); m_sluX = SluMatrix::Map(x_ref.const_cast_derived()); - + typename Rhs::PlainObject b_cpy; - if(m_sluEqued!='N') - { + if (m_sluEqued != 'N') { b_cpy = b; - m_sluB = SluMatrix::Map(b_cpy.const_cast_derived()); + m_sluB = SluMatrix::Map(b_cpy.const_cast_derived()); } StatInit(&m_sluStat); int info = 0; RealScalar recip_pivot_growth, rcond; - SuperLU_gssvx(&m_sluOptions, &m_sluA, - m_q.data(), m_p.data(), - &m_sluEtree[0], &m_sluEqued, - &m_sluRscale[0], &m_sluCscale[0], - &m_sluL, &m_sluU, - NULL, 0, - &m_sluB, &m_sluX, - &recip_pivot_growth, &rcond, - &m_sluFerr[0], &m_sluBerr[0], - &m_sluStat, &info, Scalar()); + SuperLU_gssvx(&m_sluOptions, + &m_sluA, + m_q.data(), + m_p.data(), + &m_sluEtree[0], + &m_sluEqued, + &m_sluRscale[0], + &m_sluCscale[0], + &m_sluL, + &m_sluU, + NULL, + 0, + &m_sluB, + &m_sluX, + &recip_pivot_growth, + &rcond, + &m_sluFerr[0], + &m_sluBerr[0], + &m_sluStat, + &info, + Scalar()); StatFree(&m_sluStat); - - if(x.derived().data() != x_ref.data()) - x = x_ref; - - m_info = info==0 ? Success : NumericalIssue; + + if (x.derived().data() != x_ref.data()) x = x_ref; + + m_info = info == 0 ? Success : NumericalIssue; } // the code of this extractData() function has been adapted from the SuperLU's Matlab support code, @@ -703,78 +829,68 @@ void SuperLU::_solve_impl(const MatrixBase &b, MatrixBase // THIS MATERIAL IS PROVIDED AS IS, WITH ABSOLUTELY NO WARRANTY // EXPRESSED OR IMPLIED. ANY USE IS AT YOUR OWN RISK. // -template -void SuperLUBase::extractData() const +template void SuperLUBase::extractData() const { eigen_assert(m_factorizationIsOk && "The decomposition is not in a valid state for extracting factors, you must first call either compute() or analyzePattern()/factorize()"); - if (m_extractedDataAreDirty) - { - int upper; - int fsupc, istart, nsupr; - int lastl = 0, lastu = 0; - SCformat *Lstore = static_cast(m_sluL.Store); - NCformat *Ustore = static_cast(m_sluU.Store); - Scalar *SNptr; + if (m_extractedDataAreDirty) { + int upper; + int fsupc, istart, nsupr; + int lastl = 0, lastu = 0; + SCformat *Lstore = static_cast(m_sluL.Store); + NCformat *Ustore = static_cast(m_sluU.Store); + Scalar *SNptr; const Index size = m_matrix.rows(); - m_l.resize(size,size); + m_l.resize(size, size); m_l.resizeNonZeros(Lstore->nnz); - m_u.resize(size,size); + m_u.resize(size, size); m_u.resizeNonZeros(Ustore->nnz); - int* Lcol = m_l.outerIndexPtr(); - int* Lrow = m_l.innerIndexPtr(); - Scalar* Lval = m_l.valuePtr(); + int *Lcol = m_l.outerIndexPtr(); + int *Lrow = m_l.innerIndexPtr(); + Scalar *Lval = m_l.valuePtr(); - int* Ucol = m_u.outerIndexPtr(); - int* Urow = m_u.innerIndexPtr(); - Scalar* Uval = m_u.valuePtr(); + int *Ucol = m_u.outerIndexPtr(); + int *Urow = m_u.innerIndexPtr(); + Scalar *Uval = m_u.valuePtr(); Ucol[0] = 0; Ucol[0] = 0; /* for each supernode */ - for (int k = 0; k <= Lstore->nsuper; ++k) - { - fsupc = L_FST_SUPC(k); - istart = L_SUB_START(fsupc); - nsupr = L_SUB_START(fsupc+1) - istart; - upper = 1; + for (int k = 0; k <= Lstore->nsuper; ++k) { + fsupc = L_FST_SUPC(k); + istart = L_SUB_START(fsupc); + nsupr = L_SUB_START(fsupc + 1) - istart; + upper = 1; /* for each column in the supernode */ - for (int j = fsupc; j < L_FST_SUPC(k+1); ++j) - { - SNptr = &((Scalar*)Lstore->nzval)[L_NZ_START(j)]; + for (int j = fsupc; j < L_FST_SUPC(k + 1); ++j) { + SNptr = &((Scalar *)Lstore->nzval)[L_NZ_START(j)]; /* Extract U */ - for (int i = U_NZ_START(j); i < U_NZ_START(j+1); ++i) - { - Uval[lastu] = ((Scalar*)Ustore->nzval)[i]; + for (int i = U_NZ_START(j); i < U_NZ_START(j + 1); ++i) { + Uval[lastu] = ((Scalar *)Ustore->nzval)[i]; /* Matlab doesn't like explicit zero. */ - if (Uval[lastu] != 0.0) - Urow[lastu++] = U_SUB(i); + if (Uval[lastu] != 0.0) Urow[lastu++] = U_SUB(i); } - for (int i = 0; i < upper; ++i) - { + for (int i = 0; i < upper; ++i) { /* upper triangle in the supernode */ Uval[lastu] = SNptr[i]; /* Matlab doesn't like explicit zero. */ - if (Uval[lastu] != 0.0) - Urow[lastu++] = L_SUB(istart+i); + if (Uval[lastu] != 0.0) Urow[lastu++] = L_SUB(istart + i); } - Ucol[j+1] = lastu; + Ucol[j + 1] = lastu; /* Extract L */ Lval[lastl] = 1.0; /* unit diagonal */ Lrow[lastl++] = L_SUB(istart + upper - 1); - for (int i = upper; i < nsupr; ++i) - { + for (int i = upper; i < nsupr; ++i) { Lval[lastl] = SNptr[i]; /* Matlab doesn't like explicit zero. */ - if (Lval[lastl] != 0.0) - Lrow[lastl++] = L_SUB(istart+i); + if (Lval[lastl] != 0.0) Lrow[lastl++] = L_SUB(istart + i); } - Lcol[j+1] = lastl; + Lcol[j + 1] = lastl; ++upper; } /* for j ... */ @@ -789,29 +905,24 @@ void SuperLUBase::extractData() const } } -template -typename SuperLU::Scalar SuperLU::determinant() const +template typename SuperLU::Scalar SuperLU::determinant() const { eigen_assert(m_factorizationIsOk && "The decomposition is not in a valid state for computing the determinant, you must first call either compute() or analyzePattern()/factorize()"); - - if (m_extractedDataAreDirty) - this->extractData(); + + if (m_extractedDataAreDirty) this->extractData(); Scalar det = Scalar(1); - for (int j=0; j 0) - { - int lastId = m_u.outerIndexPtr()[j+1]-1; - eigen_assert(m_u.innerIndexPtr()[lastId]<=j); - if (m_u.innerIndexPtr()[lastId]==j) - det *= m_u.valuePtr()[lastId]; + for (int j = 0; j < m_u.cols(); ++j) { + if (m_u.outerIndexPtr()[j + 1] - m_u.outerIndexPtr()[j] > 0) { + int lastId = m_u.outerIndexPtr()[j + 1] - 1; + eigen_assert(m_u.innerIndexPtr()[lastId] <= j); + if (m_u.innerIndexPtr()[lastId] == j) det *= m_u.valuePtr()[lastId]; } } - if(PermutationMap(m_p.data(),m_p.size()).determinant()*PermutationMap(m_q.data(),m_q.size()).determinant()<0) + if (PermutationMap(m_p.data(), m_p.size()).determinant() * PermutationMap(m_q.data(), m_q.size()).determinant() < 0) det = -det; - if(m_sluEqued!='N') - return det/m_sluRscale.prod()/m_sluCscale.prod(); + if (m_sluEqued != 'N') + return det / m_sluRscale.prod() / m_sluCscale.prod(); else return det; } @@ -823,143 +934,146 @@ typename SuperLU::Scalar SuperLU::determinant() const #ifdef EIGEN_SUPERLU_HAS_ILU /** \ingroup SuperLUSupport_Module - * \class SuperILU - * \brief A sparse direct \b incomplete LU factorization and solver based on the SuperLU library - * - * This class allows to solve for an approximate solution of A.X = B sparse linear problems via an incomplete LU factorization - * using the SuperLU library. This class is aimed to be used as a preconditioner of the iterative linear solvers. - * - * \warning This class is only for the 4.x versions of SuperLU. The 3.x and 5.x versions are not supported. - * - * \tparam _MatrixType the type of the sparse matrix A, it must be a SparseMatrix<> - * - * \implsparsesolverconcept - * - * \sa \ref TutorialSparseSolverConcept, class IncompleteLUT, class ConjugateGradient, class BiCGSTAB - */ - -template -class SuperILU : public SuperLUBase<_MatrixType,SuperILU<_MatrixType> > + * \class SuperILU + * \brief A sparse direct \b incomplete LU factorization and solver based on the SuperLU library + * + * This class allows to solve for an approximate solution of A.X = B sparse linear problems via an incomplete LU + * factorization using the SuperLU library. This class is aimed to be used as a preconditioner of the iterative linear + * solvers. + * + * \warning This class is only for the 4.x versions of SuperLU. The 3.x and 5.x versions are not supported. + * + * \tparam _MatrixType the type of the sparse matrix A, it must be a SparseMatrix<> + * + * \implsparsesolverconcept + * + * \sa \ref TutorialSparseSolverConcept, class IncompleteLUT, class ConjugateGradient, class BiCGSTAB + */ + +template class SuperILU : public SuperLUBase<_MatrixType, SuperILU<_MatrixType>> { - public: - typedef SuperLUBase<_MatrixType,SuperILU> Base; - typedef _MatrixType MatrixType; - typedef typename Base::Scalar Scalar; - typedef typename Base::RealScalar RealScalar; +public: + typedef SuperLUBase<_MatrixType, SuperILU> Base; + typedef _MatrixType MatrixType; + typedef typename Base::Scalar Scalar; + typedef typename Base::RealScalar RealScalar; - public: - using Base::_solve_impl; +public: + using Base::_solve_impl; - SuperILU() : Base() { init(); } + SuperILU() : Base() { init(); } - SuperILU(const MatrixType& matrix) : Base() - { - init(); - Base::compute(matrix); - } + SuperILU(const MatrixType &matrix) : Base() + { + init(); + Base::compute(matrix); + } - ~SuperILU() - { - } - - /** Performs a symbolic decomposition on the sparcity of \a matrix. - * - * This function is particularly useful when solving for several problems having the same structure. - * - * \sa factorize() - */ - void analyzePattern(const MatrixType& matrix) - { - Base::analyzePattern(matrix); - } - - /** Performs a numeric decomposition of \a matrix - * - * The given matrix must has the same sparcity than the matrix on which the symbolic decomposition has been performed. - * - * \sa analyzePattern() - */ - void factorize(const MatrixType& matrix); - - #ifndef EIGEN_PARSED_BY_DOXYGEN - /** \internal */ - template - void _solve_impl(const MatrixBase &b, MatrixBase &dest) const; - #endif // EIGEN_PARSED_BY_DOXYGEN - - protected: - - using Base::m_matrix; - using Base::m_sluOptions; - using Base::m_sluA; - using Base::m_sluB; - using Base::m_sluX; - using Base::m_p; - using Base::m_q; - using Base::m_sluEtree; - using Base::m_sluEqued; - using Base::m_sluRscale; - using Base::m_sluCscale; - using Base::m_sluL; - using Base::m_sluU; - using Base::m_sluStat; - using Base::m_sluFerr; - using Base::m_sluBerr; - using Base::m_l; - using Base::m_u; - - using Base::m_analysisIsOk; - using Base::m_factorizationIsOk; - using Base::m_extractedDataAreDirty; - using Base::m_isInitialized; - using Base::m_info; - - void init() - { - Base::init(); - - ilu_set_default_options(&m_sluOptions); - m_sluOptions.PrintStat = NO; - m_sluOptions.ConditionNumber = NO; - m_sluOptions.Trans = NOTRANS; - m_sluOptions.ColPerm = MMD_AT_PLUS_A; - - // no attempt to preserve column sum - m_sluOptions.ILU_MILU = SILU; - // only basic ILU(k) support -- no direct control over memory consumption - // better to use ILU_DropRule = DROP_BASIC | DROP_AREA - // and set ILU_FillFactor to max memory growth - m_sluOptions.ILU_DropRule = DROP_BASIC; - m_sluOptions.ILU_DropTol = NumTraits::dummy_precision()*10; - } - - private: - SuperILU(SuperILU& ) { } + ~SuperILU() {} + + /** Performs a symbolic decomposition on the sparcity of \a matrix. + * + * This function is particularly useful when solving for several problems having the same structure. + * + * \sa factorize() + */ + void analyzePattern(const MatrixType &matrix) { Base::analyzePattern(matrix); } + + /** Performs a numeric decomposition of \a matrix + * + * The given matrix must has the same sparcity than the matrix on which the symbolic decomposition has been performed. + * + * \sa analyzePattern() + */ + void factorize(const MatrixType &matrix); + +#ifndef EIGEN_PARSED_BY_DOXYGEN + /** \internal */ + template void _solve_impl(const MatrixBase &b, MatrixBase &dest) const; +#endif// EIGEN_PARSED_BY_DOXYGEN + +protected: + using Base::m_matrix; + using Base::m_sluOptions; + using Base::m_sluA; + using Base::m_sluB; + using Base::m_sluX; + using Base::m_p; + using Base::m_q; + using Base::m_sluEtree; + using Base::m_sluEqued; + using Base::m_sluRscale; + using Base::m_sluCscale; + using Base::m_sluL; + using Base::m_sluU; + using Base::m_sluStat; + using Base::m_sluFerr; + using Base::m_sluBerr; + using Base::m_l; + using Base::m_u; + + using Base::m_analysisIsOk; + using Base::m_factorizationIsOk; + using Base::m_extractedDataAreDirty; + using Base::m_isInitialized; + using Base::m_info; + + void init() + { + Base::init(); + + ilu_set_default_options(&m_sluOptions); + m_sluOptions.PrintStat = NO; + m_sluOptions.ConditionNumber = NO; + m_sluOptions.Trans = NOTRANS; + m_sluOptions.ColPerm = MMD_AT_PLUS_A; + + // no attempt to preserve column sum + m_sluOptions.ILU_MILU = SILU; + // only basic ILU(k) support -- no direct control over memory consumption + // better to use ILU_DropRule = DROP_BASIC | DROP_AREA + // and set ILU_FillFactor to max memory growth + m_sluOptions.ILU_DropRule = DROP_BASIC; + m_sluOptions.ILU_DropTol = NumTraits::dummy_precision() * 10; + } + +private: + SuperILU(SuperILU &) {} }; -template -void SuperILU::factorize(const MatrixType& a) +template void SuperILU::factorize(const MatrixType &a) { eigen_assert(m_analysisIsOk && "You must first call analyzePattern()"); - if(!m_analysisIsOk) - { + if (!m_analysisIsOk) { m_info = InvalidInput; return; } - + this->initFactorization(a); int info = 0; RealScalar recip_pivot_growth, rcond; StatInit(&m_sluStat); - SuperLU_gsisx(&m_sluOptions, &m_sluA, m_q.data(), m_p.data(), &m_sluEtree[0], - &m_sluEqued, &m_sluRscale[0], &m_sluCscale[0], - &m_sluL, &m_sluU, - NULL, 0, - &m_sluB, &m_sluX, - &recip_pivot_growth, &rcond, - &m_sluStat, &info, Scalar()); + SuperLU_gsisx(&m_sluOptions, + &m_sluA, + m_q.data(), + m_p.data(), + &m_sluEtree[0], + &m_sluEqued, + &m_sluRscale[0], + &m_sluCscale[0], + &m_sluL, + &m_sluU, + NULL, + 0, + &m_sluB, + &m_sluX, + &recip_pivot_growth, + &rcond, + &m_sluStat, + &info, + Scalar()); StatFree(&m_sluStat); // FIXME how to better check for errors ??? @@ -969,14 +1083,14 @@ void SuperILU::factorize(const MatrixType& a) #ifndef EIGEN_PARSED_BY_DOXYGEN template -template -void SuperILU::_solve_impl(const MatrixBase &b, MatrixBase& x) const +template +void SuperILU::_solve_impl(const MatrixBase &b, MatrixBase &x) const { eigen_assert(m_factorizationIsOk && "The decomposition is not in a valid state for solving, you must first call either compute() or analyzePattern()/factorize()"); const int size = m_matrix.rows(); const int rhsCols = b.cols(); - eigen_assert(size==b.rows()); + eigen_assert(size == b.rows()); m_sluOptions.Trans = NOTRANS; m_sluOptions.Fact = FACTORED; @@ -984,44 +1098,52 @@ void SuperILU::_solve_impl(const MatrixBase &b, MatrixBase > b_ref(b); - Ref > x_ref(x); - + + Ref> b_ref(b); + Ref> x_ref(x); + m_sluB = SluMatrix::Map(b_ref.const_cast_derived()); m_sluX = SluMatrix::Map(x_ref.const_cast_derived()); typename Rhs::PlainObject b_cpy; - if(m_sluEqued!='N') - { + if (m_sluEqued != 'N') { b_cpy = b; - m_sluB = SluMatrix::Map(b_cpy.const_cast_derived()); + m_sluB = SluMatrix::Map(b_cpy.const_cast_derived()); } - + int info = 0; RealScalar recip_pivot_growth, rcond; StatInit(&m_sluStat); - SuperLU_gsisx(&m_sluOptions, &m_sluA, - m_q.data(), m_p.data(), - &m_sluEtree[0], &m_sluEqued, - &m_sluRscale[0], &m_sluCscale[0], - &m_sluL, &m_sluU, - NULL, 0, - &m_sluB, &m_sluX, - &recip_pivot_growth, &rcond, - &m_sluStat, &info, Scalar()); + SuperLU_gsisx(&m_sluOptions, + &m_sluA, + m_q.data(), + m_p.data(), + &m_sluEtree[0], + &m_sluEqued, + &m_sluRscale[0], + &m_sluCscale[0], + &m_sluL, + &m_sluU, + NULL, + 0, + &m_sluB, + &m_sluX, + &recip_pivot_growth, + &rcond, + &m_sluStat, + &info, + Scalar()); StatFree(&m_sluStat); - - if(x.derived().data() != x_ref.data()) - x = x_ref; - m_info = info==0 ? Success : NumericalIssue; + if (x.derived().data() != x_ref.data()) x = x_ref; + + m_info = info == 0 ? Success : NumericalIssue; } #endif #endif -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_SUPERLUSUPPORT_H +#endif// EIGEN_SUPERLUSUPPORT_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/UmfPackSupport/UmfPackSupport.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/UmfPackSupport/UmfPackSupport.h index 91c09ab1..4eb211de 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/UmfPackSupport/UmfPackSupport.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/UmfPackSupport/UmfPackSupport.h @@ -17,451 +17,524 @@ namespace Eigen { // generic double/complex wrapper functions: -inline void umfpack_defaults(double control[UMFPACK_CONTROL], double) -{ umfpack_di_defaults(control); } +inline void umfpack_defaults(double control[UMFPACK_CONTROL], double) { umfpack_di_defaults(control); } -inline void umfpack_defaults(double control[UMFPACK_CONTROL], std::complex) -{ umfpack_zi_defaults(control); } +inline void umfpack_defaults(double control[UMFPACK_CONTROL], std::complex) { umfpack_zi_defaults(control); } inline void umfpack_report_info(double control[UMFPACK_CONTROL], double info[UMFPACK_INFO], double) -{ umfpack_di_report_info(control, info);} +{ + umfpack_di_report_info(control, info); +} inline void umfpack_report_info(double control[UMFPACK_CONTROL], double info[UMFPACK_INFO], std::complex) -{ umfpack_zi_report_info(control, info);} +{ + umfpack_zi_report_info(control, info); +} inline void umfpack_report_status(double control[UMFPACK_CONTROL], int status, double) -{ umfpack_di_report_status(control, status);} +{ + umfpack_di_report_status(control, status); +} inline void umfpack_report_status(double control[UMFPACK_CONTROL], int status, std::complex) -{ umfpack_zi_report_status(control, status);} +{ + umfpack_zi_report_status(control, status); +} -inline void umfpack_report_control(double control[UMFPACK_CONTROL], double) -{ umfpack_di_report_control(control);} +inline void umfpack_report_control(double control[UMFPACK_CONTROL], double) { umfpack_di_report_control(control); } inline void umfpack_report_control(double control[UMFPACK_CONTROL], std::complex) -{ umfpack_zi_report_control(control);} +{ + umfpack_zi_report_control(control); +} inline void umfpack_free_numeric(void **Numeric, double) -{ umfpack_di_free_numeric(Numeric); *Numeric = 0; } +{ + umfpack_di_free_numeric(Numeric); + *Numeric = 0; +} inline void umfpack_free_numeric(void **Numeric, std::complex) -{ umfpack_zi_free_numeric(Numeric); *Numeric = 0; } +{ + umfpack_zi_free_numeric(Numeric); + *Numeric = 0; +} inline void umfpack_free_symbolic(void **Symbolic, double) -{ umfpack_di_free_symbolic(Symbolic); *Symbolic = 0; } +{ + umfpack_di_free_symbolic(Symbolic); + *Symbolic = 0; +} inline void umfpack_free_symbolic(void **Symbolic, std::complex) -{ umfpack_zi_free_symbolic(Symbolic); *Symbolic = 0; } +{ + umfpack_zi_free_symbolic(Symbolic); + *Symbolic = 0; +} -inline int umfpack_symbolic(int n_row,int n_col, - const int Ap[], const int Ai[], const double Ax[], void **Symbolic, - const double Control [UMFPACK_CONTROL], double Info [UMFPACK_INFO]) +inline int umfpack_symbolic(int n_row, + int n_col, + const int Ap[], + const int Ai[], + const double Ax[], + void **Symbolic, + const double Control[UMFPACK_CONTROL], + double Info[UMFPACK_INFO]) { - return umfpack_di_symbolic(n_row,n_col,Ap,Ai,Ax,Symbolic,Control,Info); + return umfpack_di_symbolic(n_row, n_col, Ap, Ai, Ax, Symbolic, Control, Info); } -inline int umfpack_symbolic(int n_row,int n_col, - const int Ap[], const int Ai[], const std::complex Ax[], void **Symbolic, - const double Control [UMFPACK_CONTROL], double Info [UMFPACK_INFO]) +inline int umfpack_symbolic(int n_row, + int n_col, + const int Ap[], + const int Ai[], + const std::complex Ax[], + void **Symbolic, + const double Control[UMFPACK_CONTROL], + double Info[UMFPACK_INFO]) { - return umfpack_zi_symbolic(n_row,n_col,Ap,Ai,&numext::real_ref(Ax[0]),0,Symbolic,Control,Info); + return umfpack_zi_symbolic(n_row, n_col, Ap, Ai, &numext::real_ref(Ax[0]), 0, Symbolic, Control, Info); } -inline int umfpack_numeric( const int Ap[], const int Ai[], const double Ax[], - void *Symbolic, void **Numeric, - const double Control[UMFPACK_CONTROL],double Info [UMFPACK_INFO]) +inline int umfpack_numeric(const int Ap[], + const int Ai[], + const double Ax[], + void *Symbolic, + void **Numeric, + const double Control[UMFPACK_CONTROL], + double Info[UMFPACK_INFO]) { - return umfpack_di_numeric(Ap,Ai,Ax,Symbolic,Numeric,Control,Info); + return umfpack_di_numeric(Ap, Ai, Ax, Symbolic, Numeric, Control, Info); } -inline int umfpack_numeric( const int Ap[], const int Ai[], const std::complex Ax[], - void *Symbolic, void **Numeric, - const double Control[UMFPACK_CONTROL],double Info [UMFPACK_INFO]) +inline int umfpack_numeric(const int Ap[], + const int Ai[], + const std::complex Ax[], + void *Symbolic, + void **Numeric, + const double Control[UMFPACK_CONTROL], + double Info[UMFPACK_INFO]) { - return umfpack_zi_numeric(Ap,Ai,&numext::real_ref(Ax[0]),0,Symbolic,Numeric,Control,Info); + return umfpack_zi_numeric(Ap, Ai, &numext::real_ref(Ax[0]), 0, Symbolic, Numeric, Control, Info); } -inline int umfpack_solve( int sys, const int Ap[], const int Ai[], const double Ax[], - double X[], const double B[], void *Numeric, - const double Control[UMFPACK_CONTROL], double Info[UMFPACK_INFO]) +inline int umfpack_solve(int sys, + const int Ap[], + const int Ai[], + const double Ax[], + double X[], + const double B[], + void *Numeric, + const double Control[UMFPACK_CONTROL], + double Info[UMFPACK_INFO]) { - return umfpack_di_solve(sys,Ap,Ai,Ax,X,B,Numeric,Control,Info); + return umfpack_di_solve(sys, Ap, Ai, Ax, X, B, Numeric, Control, Info); } -inline int umfpack_solve( int sys, const int Ap[], const int Ai[], const std::complex Ax[], - std::complex X[], const std::complex B[], void *Numeric, - const double Control[UMFPACK_CONTROL], double Info[UMFPACK_INFO]) +inline int umfpack_solve(int sys, + const int Ap[], + const int Ai[], + const std::complex Ax[], + std::complex X[], + const std::complex B[], + void *Numeric, + const double Control[UMFPACK_CONTROL], + double Info[UMFPACK_INFO]) { - return umfpack_zi_solve(sys,Ap,Ai,&numext::real_ref(Ax[0]),0,&numext::real_ref(X[0]),0,&numext::real_ref(B[0]),0,Numeric,Control,Info); + return umfpack_zi_solve(sys, + Ap, + Ai, + &numext::real_ref(Ax[0]), + 0, + &numext::real_ref(X[0]), + 0, + &numext::real_ref(B[0]), + 0, + Numeric, + Control, + Info); } inline int umfpack_get_lunz(int *lnz, int *unz, int *n_row, int *n_col, int *nz_udiag, void *Numeric, double) { - return umfpack_di_get_lunz(lnz,unz,n_row,n_col,nz_udiag,Numeric); + return umfpack_di_get_lunz(lnz, unz, n_row, n_col, nz_udiag, Numeric); } -inline int umfpack_get_lunz(int *lnz, int *unz, int *n_row, int *n_col, int *nz_udiag, void *Numeric, std::complex) +inline int + umfpack_get_lunz(int *lnz, int *unz, int *n_row, int *n_col, int *nz_udiag, void *Numeric, std::complex) { - return umfpack_zi_get_lunz(lnz,unz,n_row,n_col,nz_udiag,Numeric); + return umfpack_zi_get_lunz(lnz, unz, n_row, n_col, nz_udiag, Numeric); } -inline int umfpack_get_numeric(int Lp[], int Lj[], double Lx[], int Up[], int Ui[], double Ux[], - int P[], int Q[], double Dx[], int *do_recip, double Rs[], void *Numeric) +inline int umfpack_get_numeric(int Lp[], + int Lj[], + double Lx[], + int Up[], + int Ui[], + double Ux[], + int P[], + int Q[], + double Dx[], + int *do_recip, + double Rs[], + void *Numeric) { - return umfpack_di_get_numeric(Lp,Lj,Lx,Up,Ui,Ux,P,Q,Dx,do_recip,Rs,Numeric); + return umfpack_di_get_numeric(Lp, Lj, Lx, Up, Ui, Ux, P, Q, Dx, do_recip, Rs, Numeric); } -inline int umfpack_get_numeric(int Lp[], int Lj[], std::complex Lx[], int Up[], int Ui[], std::complex Ux[], - int P[], int Q[], std::complex Dx[], int *do_recip, double Rs[], void *Numeric) +inline int umfpack_get_numeric(int Lp[], + int Lj[], + std::complex Lx[], + int Up[], + int Ui[], + std::complex Ux[], + int P[], + int Q[], + std::complex Dx[], + int *do_recip, + double Rs[], + void *Numeric) { - double& lx0_real = numext::real_ref(Lx[0]); - double& ux0_real = numext::real_ref(Ux[0]); - double& dx0_real = numext::real_ref(Dx[0]); - return umfpack_zi_get_numeric(Lp,Lj,Lx?&lx0_real:0,0,Up,Ui,Ux?&ux0_real:0,0,P,Q, - Dx?&dx0_real:0,0,do_recip,Rs,Numeric); + double &lx0_real = numext::real_ref(Lx[0]); + double &ux0_real = numext::real_ref(Ux[0]); + double &dx0_real = numext::real_ref(Dx[0]); + return umfpack_zi_get_numeric( + Lp, Lj, Lx ? &lx0_real : 0, 0, Up, Ui, Ux ? &ux0_real : 0, 0, P, Q, Dx ? &dx0_real : 0, 0, do_recip, Rs, Numeric); } -inline int umfpack_get_determinant(double *Mx, double *Ex, void *NumericHandle, double User_Info [UMFPACK_INFO]) +inline int umfpack_get_determinant(double *Mx, double *Ex, void *NumericHandle, double User_Info[UMFPACK_INFO]) { - return umfpack_di_get_determinant(Mx,Ex,NumericHandle,User_Info); + return umfpack_di_get_determinant(Mx, Ex, NumericHandle, User_Info); } -inline int umfpack_get_determinant(std::complex *Mx, double *Ex, void *NumericHandle, double User_Info [UMFPACK_INFO]) +inline int + umfpack_get_determinant(std::complex *Mx, double *Ex, void *NumericHandle, double User_Info[UMFPACK_INFO]) { - double& mx_real = numext::real_ref(*Mx); - return umfpack_zi_get_determinant(&mx_real,0,Ex,NumericHandle,User_Info); + double &mx_real = numext::real_ref(*Mx); + return umfpack_zi_get_determinant(&mx_real, 0, Ex, NumericHandle, User_Info); } /** \ingroup UmfPackSupport_Module - * \brief A sparse LU factorization and solver based on UmfPack - * - * This class allows to solve for A.X = B sparse linear problems via a LU factorization - * using the UmfPack library. The sparse matrix A must be squared and full rank. - * The vectors or matrices X and B can be either dense or sparse. - * - * \warning The input matrix A should be in a \b compressed and \b column-major form. - * Otherwise an expensive copy will be made. You can call the inexpensive makeCompressed() to get a compressed matrix. - * \tparam _MatrixType the type of the sparse matrix A, it must be a SparseMatrix<> - * - * \implsparsesolverconcept - * - * \sa \ref TutorialSparseSolverConcept, class SparseLU - */ -template -class UmfPackLU : public SparseSolverBase > + * \brief A sparse LU factorization and solver based on UmfPack + * + * This class allows to solve for A.X = B sparse linear problems via a LU factorization + * using the UmfPack library. The sparse matrix A must be squared and full rank. + * The vectors or matrices X and B can be either dense or sparse. + * + * \warning The input matrix A should be in a \b compressed and \b column-major form. + * Otherwise an expensive copy will be made. You can call the inexpensive makeCompressed() to get a compressed matrix. + * \tparam _MatrixType the type of the sparse matrix A, it must be a SparseMatrix<> + * + * \implsparsesolverconcept + * + * \sa \ref TutorialSparseSolverConcept, class SparseLU + */ +template class UmfPackLU : public SparseSolverBase> { - protected: - typedef SparseSolverBase > Base; - using Base::m_isInitialized; - public: - using Base::_solve_impl; - typedef _MatrixType MatrixType; - typedef typename MatrixType::Scalar Scalar; - typedef typename MatrixType::RealScalar RealScalar; - typedef typename MatrixType::StorageIndex StorageIndex; - typedef Matrix Vector; - typedef Matrix IntRowVectorType; - typedef Matrix IntColVectorType; - typedef SparseMatrix LUMatrixType; - typedef SparseMatrix UmfpackMatrixType; - typedef Ref UmfpackMatrixRef; - enum { - ColsAtCompileTime = MatrixType::ColsAtCompileTime, - MaxColsAtCompileTime = MatrixType::MaxColsAtCompileTime - }; - - public: - - typedef Array UmfpackControl; - typedef Array UmfpackInfo; - - UmfPackLU() - : m_dummy(0,0), mp_matrix(m_dummy) - { - init(); - } +protected: + typedef SparseSolverBase> Base; + using Base::m_isInitialized; + +public: + using Base::_solve_impl; + typedef _MatrixType MatrixType; + typedef typename MatrixType::Scalar Scalar; + typedef typename MatrixType::RealScalar RealScalar; + typedef typename MatrixType::StorageIndex StorageIndex; + typedef Matrix Vector; + typedef Matrix IntRowVectorType; + typedef Matrix IntColVectorType; + typedef SparseMatrix LUMatrixType; + typedef SparseMatrix UmfpackMatrixType; + typedef Ref UmfpackMatrixRef; + enum { ColsAtCompileTime = MatrixType::ColsAtCompileTime, MaxColsAtCompileTime = MatrixType::MaxColsAtCompileTime }; + +public: + typedef Array UmfpackControl; + typedef Array UmfpackInfo; + + UmfPackLU() : m_dummy(0, 0), mp_matrix(m_dummy) { init(); } + + template explicit UmfPackLU(const InputMatrixType &matrix) : mp_matrix(matrix) + { + init(); + compute(matrix); + } - template - explicit UmfPackLU(const InputMatrixType& matrix) - : mp_matrix(matrix) - { - init(); - compute(matrix); - } + ~UmfPackLU() + { + if (m_symbolic) umfpack_free_symbolic(&m_symbolic, Scalar()); + if (m_numeric) umfpack_free_numeric(&m_numeric, Scalar()); + } - ~UmfPackLU() - { - if(m_symbolic) umfpack_free_symbolic(&m_symbolic,Scalar()); - if(m_numeric) umfpack_free_numeric(&m_numeric,Scalar()); - } + inline Index rows() const { return mp_matrix.rows(); } + inline Index cols() const { return mp_matrix.cols(); } - inline Index rows() const { return mp_matrix.rows(); } - inline Index cols() const { return mp_matrix.cols(); } - - /** \brief Reports whether previous computation was successful. - * - * \returns \c Success if computation was succesful, - * \c NumericalIssue if the matrix.appears to be negative. - */ - ComputationInfo info() const - { - eigen_assert(m_isInitialized && "Decomposition is not initialized."); - return m_info; - } + /** \brief Reports whether previous computation was successful. + * + * \returns \c Success if computation was succesful, + * \c NumericalIssue if the matrix.appears to be negative. + */ + ComputationInfo info() const + { + eigen_assert(m_isInitialized && "Decomposition is not initialized."); + return m_info; + } - inline const LUMatrixType& matrixL() const - { - if (m_extractedDataAreDirty) extractData(); - return m_l; - } + inline const LUMatrixType &matrixL() const + { + if (m_extractedDataAreDirty) extractData(); + return m_l; + } - inline const LUMatrixType& matrixU() const - { - if (m_extractedDataAreDirty) extractData(); - return m_u; - } + inline const LUMatrixType &matrixU() const + { + if (m_extractedDataAreDirty) extractData(); + return m_u; + } - inline const IntColVectorType& permutationP() const - { - if (m_extractedDataAreDirty) extractData(); - return m_p; - } + inline const IntColVectorType &permutationP() const + { + if (m_extractedDataAreDirty) extractData(); + return m_p; + } - inline const IntRowVectorType& permutationQ() const - { - if (m_extractedDataAreDirty) extractData(); - return m_q; - } + inline const IntRowVectorType &permutationQ() const + { + if (m_extractedDataAreDirty) extractData(); + return m_q; + } - /** Computes the sparse Cholesky decomposition of \a matrix - * Note that the matrix should be column-major, and in compressed format for best performance. - * \sa SparseMatrix::makeCompressed(). - */ - template - void compute(const InputMatrixType& matrix) - { - if(m_symbolic) umfpack_free_symbolic(&m_symbolic,Scalar()); - if(m_numeric) umfpack_free_numeric(&m_numeric,Scalar()); - grab(matrix.derived()); - analyzePattern_impl(); - factorize_impl(); - } + /** Computes the sparse Cholesky decomposition of \a matrix + * Note that the matrix should be column-major, and in compressed format for best performance. + * \sa SparseMatrix::makeCompressed(). + */ + template void compute(const InputMatrixType &matrix) + { + if (m_symbolic) umfpack_free_symbolic(&m_symbolic, Scalar()); + if (m_numeric) umfpack_free_numeric(&m_numeric, Scalar()); + grab(matrix.derived()); + analyzePattern_impl(); + factorize_impl(); + } - /** Performs a symbolic decomposition on the sparcity of \a matrix. - * - * This function is particularly useful when solving for several problems having the same structure. - * - * \sa factorize(), compute() - */ - template - void analyzePattern(const InputMatrixType& matrix) - { - if(m_symbolic) umfpack_free_symbolic(&m_symbolic,Scalar()); - if(m_numeric) umfpack_free_numeric(&m_numeric,Scalar()); - - grab(matrix.derived()); - - analyzePattern_impl(); - } + /** Performs a symbolic decomposition on the sparcity of \a matrix. + * + * This function is particularly useful when solving for several problems having the same structure. + * + * \sa factorize(), compute() + */ + template void analyzePattern(const InputMatrixType &matrix) + { + if (m_symbolic) umfpack_free_symbolic(&m_symbolic, Scalar()); + if (m_numeric) umfpack_free_numeric(&m_numeric, Scalar()); - /** Provides the return status code returned by UmfPack during the numeric - * factorization. - * - * \sa factorize(), compute() - */ - inline int umfpackFactorizeReturncode() const - { - eigen_assert(m_numeric && "UmfPackLU: you must first call factorize()"); - return m_fact_errorCode; - } + grab(matrix.derived()); - /** Provides access to the control settings array used by UmfPack. - * - * If this array contains NaN's, the default values are used. - * - * See UMFPACK documentation for details. - */ - inline const UmfpackControl& umfpackControl() const - { - return m_control; - } + analyzePattern_impl(); + } - /** Provides access to the control settings array used by UmfPack. - * - * If this array contains NaN's, the default values are used. - * - * See UMFPACK documentation for details. - */ - inline UmfpackControl& umfpackControl() - { - return m_control; - } + /** Provides the return status code returned by UmfPack during the numeric + * factorization. + * + * \sa factorize(), compute() + */ + inline int umfpackFactorizeReturncode() const + { + eigen_assert(m_numeric && "UmfPackLU: you must first call factorize()"); + return m_fact_errorCode; + } - /** Performs a numeric decomposition of \a matrix - * - * The given matrix must has the same sparcity than the matrix on which the pattern anylysis has been performed. - * - * \sa analyzePattern(), compute() - */ - template - void factorize(const InputMatrixType& matrix) - { - eigen_assert(m_analysisIsOk && "UmfPackLU: you must first call analyzePattern()"); - if(m_numeric) - umfpack_free_numeric(&m_numeric,Scalar()); - - grab(matrix.derived()); - - factorize_impl(); - } + /** Provides access to the control settings array used by UmfPack. + * + * If this array contains NaN's, the default values are used. + * + * See UMFPACK documentation for details. + */ + inline const UmfpackControl &umfpackControl() const { return m_control; } + + /** Provides access to the control settings array used by UmfPack. + * + * If this array contains NaN's, the default values are used. + * + * See UMFPACK documentation for details. + */ + inline UmfpackControl &umfpackControl() { return m_control; } + + /** Performs a numeric decomposition of \a matrix + * + * The given matrix must has the same sparcity than the matrix on which the pattern anylysis has been performed. + * + * \sa analyzePattern(), compute() + */ + template void factorize(const InputMatrixType &matrix) + { + eigen_assert(m_analysisIsOk && "UmfPackLU: you must first call analyzePattern()"); + if (m_numeric) umfpack_free_numeric(&m_numeric, Scalar()); - /** Prints the current UmfPack control settings. - * - * \sa umfpackControl() - */ - void umfpackReportControl() - { - umfpack_report_control(m_control.data(), Scalar()); - } + grab(matrix.derived()); - /** Prints statistics collected by UmfPack. - * - * \sa analyzePattern(), compute() - */ - void umfpackReportInfo() - { - eigen_assert(m_analysisIsOk && "UmfPackLU: you must first call analyzePattern()"); - umfpack_report_info(m_control.data(), m_umfpackInfo.data(), Scalar()); - } + factorize_impl(); + } - /** Prints the status of the previous factorization operation performed by UmfPack (symbolic or numerical factorization). - * - * \sa analyzePattern(), compute() - */ - void umfpackReportStatus() { - eigen_assert(m_analysisIsOk && "UmfPackLU: you must first call analyzePattern()"); - umfpack_report_status(m_control.data(), m_fact_errorCode, Scalar()); - } + /** Prints the current UmfPack control settings. + * + * \sa umfpackControl() + */ + void umfpackReportControl() { umfpack_report_control(m_control.data(), Scalar()); } + + /** Prints statistics collected by UmfPack. + * + * \sa analyzePattern(), compute() + */ + void umfpackReportInfo() + { + eigen_assert(m_analysisIsOk && "UmfPackLU: you must first call analyzePattern()"); + umfpack_report_info(m_control.data(), m_umfpackInfo.data(), Scalar()); + } - /** \internal */ - template - bool _solve_impl(const MatrixBase &b, MatrixBase &x) const; + /** Prints the status of the previous factorization operation performed by UmfPack (symbolic or numerical + * factorization). + * + * \sa analyzePattern(), compute() + */ + void umfpackReportStatus() + { + eigen_assert(m_analysisIsOk && "UmfPackLU: you must first call analyzePattern()"); + umfpack_report_status(m_control.data(), m_fact_errorCode, Scalar()); + } - Scalar determinant() const; + /** \internal */ + template + bool _solve_impl(const MatrixBase &b, MatrixBase &x) const; - void extractData() const; + Scalar determinant() const; - protected: + void extractData() const; - void init() - { - m_info = InvalidInput; - m_isInitialized = false; - m_numeric = 0; - m_symbolic = 0; - m_extractedDataAreDirty = true; +protected: + void init() + { + m_info = InvalidInput; + m_isInitialized = false; + m_numeric = 0; + m_symbolic = 0; + m_extractedDataAreDirty = true; - umfpack_defaults(m_control.data(), Scalar()); - } + umfpack_defaults(m_control.data(), Scalar()); + } - void analyzePattern_impl() - { - m_fact_errorCode = umfpack_symbolic(internal::convert_index(mp_matrix.rows()), - internal::convert_index(mp_matrix.cols()), - mp_matrix.outerIndexPtr(), mp_matrix.innerIndexPtr(), mp_matrix.valuePtr(), - &m_symbolic, m_control.data(), m_umfpackInfo.data()); - - m_isInitialized = true; - m_info = m_fact_errorCode ? InvalidInput : Success; - m_analysisIsOk = true; - m_factorizationIsOk = false; - m_extractedDataAreDirty = true; - } + void analyzePattern_impl() + { + m_fact_errorCode = umfpack_symbolic(internal::convert_index(mp_matrix.rows()), + internal::convert_index(mp_matrix.cols()), + mp_matrix.outerIndexPtr(), + mp_matrix.innerIndexPtr(), + mp_matrix.valuePtr(), + &m_symbolic, + m_control.data(), + m_umfpackInfo.data()); + + m_isInitialized = true; + m_info = m_fact_errorCode ? InvalidInput : Success; + m_analysisIsOk = true; + m_factorizationIsOk = false; + m_extractedDataAreDirty = true; + } - void factorize_impl() - { + void factorize_impl() + { - m_fact_errorCode = umfpack_numeric(mp_matrix.outerIndexPtr(), mp_matrix.innerIndexPtr(), mp_matrix.valuePtr(), - m_symbolic, &m_numeric, m_control.data(), m_umfpackInfo.data()); + m_fact_errorCode = umfpack_numeric(mp_matrix.outerIndexPtr(), + mp_matrix.innerIndexPtr(), + mp_matrix.valuePtr(), + m_symbolic, + &m_numeric, + m_control.data(), + m_umfpackInfo.data()); + + m_info = m_fact_errorCode == UMFPACK_OK ? Success : NumericalIssue; + m_factorizationIsOk = true; + m_extractedDataAreDirty = true; + } - m_info = m_fact_errorCode == UMFPACK_OK ? Success : NumericalIssue; - m_factorizationIsOk = true; - m_extractedDataAreDirty = true; - } + template void grab(const EigenBase &A) + { + mp_matrix.~UmfpackMatrixRef(); + ::new (&mp_matrix) UmfpackMatrixRef(A.derived()); + } - template - void grab(const EigenBase &A) - { + void grab(const UmfpackMatrixRef &A) + { + if (&(A.derived()) != &mp_matrix) { mp_matrix.~UmfpackMatrixRef(); - ::new (&mp_matrix) UmfpackMatrixRef(A.derived()); - } - - void grab(const UmfpackMatrixRef &A) - { - if(&(A.derived()) != &mp_matrix) - { - mp_matrix.~UmfpackMatrixRef(); - ::new (&mp_matrix) UmfpackMatrixRef(A); - } + ::new (&mp_matrix) UmfpackMatrixRef(A); } + } - // cached data to reduce reallocation, etc. - mutable LUMatrixType m_l; - int m_fact_errorCode; - UmfpackControl m_control; - mutable UmfpackInfo m_umfpackInfo; + // cached data to reduce reallocation, etc. + mutable LUMatrixType m_l; + int m_fact_errorCode; + UmfpackControl m_control; + mutable UmfpackInfo m_umfpackInfo; - mutable LUMatrixType m_u; - mutable IntColVectorType m_p; - mutable IntRowVectorType m_q; + mutable LUMatrixType m_u; + mutable IntColVectorType m_p; + mutable IntRowVectorType m_q; - UmfpackMatrixType m_dummy; - UmfpackMatrixRef mp_matrix; + UmfpackMatrixType m_dummy; + UmfpackMatrixRef mp_matrix; - void* m_numeric; - void* m_symbolic; + void *m_numeric; + void *m_symbolic; - mutable ComputationInfo m_info; - int m_factorizationIsOk; - int m_analysisIsOk; - mutable bool m_extractedDataAreDirty; + mutable ComputationInfo m_info; + int m_factorizationIsOk; + int m_analysisIsOk; + mutable bool m_extractedDataAreDirty; - private: - UmfPackLU(const UmfPackLU& ) { } +private: + UmfPackLU(const UmfPackLU &) {} }; -template -void UmfPackLU::extractData() const +template void UmfPackLU::extractData() const { - if (m_extractedDataAreDirty) - { + if (m_extractedDataAreDirty) { // get size of the data int lnz, unz, rows, cols, nz_udiag; umfpack_get_lunz(&lnz, &unz, &rows, &cols, &nz_udiag, m_numeric, Scalar()); // allocate data - m_l.resize(rows,(std::min)(rows,cols)); + m_l.resize(rows, (std::min)(rows, cols)); m_l.resizeNonZeros(lnz); - m_u.resize((std::min)(rows,cols),cols); + m_u.resize((std::min)(rows, cols), cols); m_u.resizeNonZeros(unz); m_p.resize(rows); m_q.resize(cols); // extract - umfpack_get_numeric(m_l.outerIndexPtr(), m_l.innerIndexPtr(), m_l.valuePtr(), - m_u.outerIndexPtr(), m_u.innerIndexPtr(), m_u.valuePtr(), - m_p.data(), m_q.data(), 0, 0, 0, m_numeric); + umfpack_get_numeric(m_l.outerIndexPtr(), + m_l.innerIndexPtr(), + m_l.valuePtr(), + m_u.outerIndexPtr(), + m_u.innerIndexPtr(), + m_u.valuePtr(), + m_p.data(), + m_q.data(), + 0, + 0, + 0, + m_numeric); m_extractedDataAreDirty = false; } } -template -typename UmfPackLU::Scalar UmfPackLU::determinant() const +template typename UmfPackLU::Scalar UmfPackLU::determinant() const { Scalar det; umfpack_get_determinant(&det, 0, m_numeric, 0); @@ -469,38 +542,39 @@ typename UmfPackLU::Scalar UmfPackLU::determinant() cons } template -template +template bool UmfPackLU::_solve_impl(const MatrixBase &b, MatrixBase &x) const { Index rhsCols = b.cols(); - eigen_assert((BDerived::Flags&RowMajorBit)==0 && "UmfPackLU backend does not support non col-major rhs yet"); - eigen_assert((XDerived::Flags&RowMajorBit)==0 && "UmfPackLU backend does not support non col-major result yet"); + eigen_assert((BDerived::Flags & RowMajorBit) == 0 && "UmfPackLU backend does not support non col-major rhs yet"); + eigen_assert((XDerived::Flags & RowMajorBit) == 0 && "UmfPackLU backend does not support non col-major result yet"); eigen_assert(b.derived().data() != x.derived().data() && " Umfpack does not support inplace solve"); int errorCode; - Scalar* x_ptr = 0; - Matrix x_tmp; - if(x.innerStride()!=1) - { + Scalar *x_ptr = 0; + Matrix x_tmp; + if (x.innerStride() != 1) { x_tmp.resize(x.rows()); x_ptr = x_tmp.data(); } - for (int j=0; j -struct traits > -{ - typedef typename DecompositionType::MatrixType MatrixType; - typedef Matrix< - typename MatrixType::Scalar, - MatrixType::RowsAtCompileTime, // the image is a subspace of the destination space, whose - // dimension is the number of rows of the original matrix - Dynamic, // we don't know at compile time the dimension of the image (the rank) - MatrixType::Options, - MatrixType::MaxRowsAtCompileTime, // the image matrix will consist of columns from the original matrix, - MatrixType::MaxColsAtCompileTime // so it has the same number of rows and at most as many columns. - > ReturnType; -}; + /** \class image_retval_base + * + */ + template struct traits> + { + typedef typename DecompositionType::MatrixType MatrixType; + typedef Matrix + ReturnType; + }; -template struct image_retval_base - : public ReturnByValue > -{ - typedef _DecompositionType DecompositionType; - typedef typename DecompositionType::MatrixType MatrixType; - typedef ReturnByValue Base; + template + struct image_retval_base : public ReturnByValue> + { + typedef _DecompositionType DecompositionType; + typedef typename DecompositionType::MatrixType MatrixType; + typedef ReturnByValue Base; - image_retval_base(const DecompositionType& dec, const MatrixType& originalMatrix) - : m_dec(dec), m_rank(dec.rank()), - m_cols(m_rank == 0 ? 1 : m_rank), - m_originalMatrix(originalMatrix) - {} + image_retval_base(const DecompositionType &dec, const MatrixType &originalMatrix) + : m_dec(dec), m_rank(dec.rank()), m_cols(m_rank == 0 ? 1 : m_rank), m_originalMatrix(originalMatrix) + {} - inline Index rows() const { return m_dec.rows(); } - inline Index cols() const { return m_cols; } - inline Index rank() const { return m_rank; } - inline const DecompositionType& dec() const { return m_dec; } - inline const MatrixType& originalMatrix() const { return m_originalMatrix; } + inline Index rows() const { return m_dec.rows(); } + inline Index cols() const { return m_cols; } + inline Index rank() const { return m_rank; } + inline const DecompositionType &dec() const { return m_dec; } + inline const MatrixType &originalMatrix() const { return m_originalMatrix; } - template inline void evalTo(Dest& dst) const - { - static_cast*>(this)->evalTo(dst); - } + template inline void evalTo(Dest &dst) const + { + static_cast *>(this)->evalTo(dst); + } protected: - const DecompositionType& m_dec; + const DecompositionType &m_dec; Index m_rank, m_cols; - const MatrixType& m_originalMatrix; -}; + const MatrixType &m_originalMatrix; + }; -} // end namespace internal +}// end namespace internal -#define EIGEN_MAKE_IMAGE_HELPERS(DecompositionType) \ - typedef typename DecompositionType::MatrixType MatrixType; \ - typedef typename MatrixType::Scalar Scalar; \ - typedef typename MatrixType::RealScalar RealScalar; \ +#define EIGEN_MAKE_IMAGE_HELPERS(DecompositionType) \ + typedef typename DecompositionType::MatrixType MatrixType; \ + typedef typename MatrixType::Scalar Scalar; \ + typedef typename MatrixType::RealScalar RealScalar; \ typedef Eigen::internal::image_retval_base Base; \ - using Base::dec; \ - using Base::originalMatrix; \ - using Base::rank; \ - using Base::rows; \ - using Base::cols; \ - image_retval(const DecompositionType& dec, const MatrixType& originalMatrix) \ - : Base(dec, originalMatrix) {} + using Base::dec; \ + using Base::originalMatrix; \ + using Base::rank; \ + using Base::rows; \ + using Base::cols; \ + image_retval(const DecompositionType &dec, const MatrixType &originalMatrix) : Base(dec, originalMatrix) {} -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_MISC_IMAGE_H +#endif// EIGEN_MISC_IMAGE_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/misc/Kernel.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/misc/Kernel.h index bef5d6ff..3afa3ef4 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/misc/Kernel.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/misc/Kernel.h @@ -10,70 +10,67 @@ #ifndef EIGEN_MISC_KERNEL_H #define EIGEN_MISC_KERNEL_H -namespace Eigen { +namespace Eigen { namespace internal { -/** \class kernel_retval_base - * - */ -template -struct traits > -{ - typedef typename DecompositionType::MatrixType MatrixType; - typedef Matrix< - typename MatrixType::Scalar, - MatrixType::ColsAtCompileTime, // the number of rows in the "kernel matrix" - // is the number of cols of the original matrix - // so that the product "matrix * kernel = zero" makes sense - Dynamic, // we don't know at compile-time the dimension of the kernel - MatrixType::Options, - MatrixType::MaxColsAtCompileTime, // see explanation for 2nd template parameter - MatrixType::MaxColsAtCompileTime // the kernel is a subspace of the domain space, - // whose dimension is the number of columns of the original matrix - > ReturnType; -}; + /** \class kernel_retval_base + * + */ + template struct traits> + { + typedef typename DecompositionType::MatrixType MatrixType; + typedef Matrix + ReturnType; + }; -template struct kernel_retval_base - : public ReturnByValue > -{ - typedef _DecompositionType DecompositionType; - typedef ReturnByValue Base; + template + struct kernel_retval_base : public ReturnByValue> + { + typedef _DecompositionType DecompositionType; + typedef ReturnByValue Base; - explicit kernel_retval_base(const DecompositionType& dec) - : m_dec(dec), - m_rank(dec.rank()), - m_cols(m_rank==dec.cols() ? 1 : dec.cols() - m_rank) - {} + explicit kernel_retval_base(const DecompositionType &dec) + : m_dec(dec), m_rank(dec.rank()), m_cols(m_rank == dec.cols() ? 1 : dec.cols() - m_rank) + {} - inline Index rows() const { return m_dec.cols(); } - inline Index cols() const { return m_cols; } - inline Index rank() const { return m_rank; } - inline const DecompositionType& dec() const { return m_dec; } + inline Index rows() const { return m_dec.cols(); } + inline Index cols() const { return m_cols; } + inline Index rank() const { return m_rank; } + inline const DecompositionType &dec() const { return m_dec; } - template inline void evalTo(Dest& dst) const - { - static_cast*>(this)->evalTo(dst); - } + template inline void evalTo(Dest &dst) const + { + static_cast *>(this)->evalTo(dst); + } protected: - const DecompositionType& m_dec; + const DecompositionType &m_dec; Index m_rank, m_cols; -}; + }; -} // end namespace internal +}// end namespace internal -#define EIGEN_MAKE_KERNEL_HELPERS(DecompositionType) \ - typedef typename DecompositionType::MatrixType MatrixType; \ - typedef typename MatrixType::Scalar Scalar; \ - typedef typename MatrixType::RealScalar RealScalar; \ +#define EIGEN_MAKE_KERNEL_HELPERS(DecompositionType) \ + typedef typename DecompositionType::MatrixType MatrixType; \ + typedef typename MatrixType::Scalar Scalar; \ + typedef typename MatrixType::RealScalar RealScalar; \ typedef Eigen::internal::kernel_retval_base Base; \ - using Base::dec; \ - using Base::rank; \ - using Base::rows; \ - using Base::cols; \ - kernel_retval(const DecompositionType& dec) : Base(dec) {} + using Base::dec; \ + using Base::rank; \ + using Base::rows; \ + using Base::cols; \ + kernel_retval(const DecompositionType &dec) : Base(dec) {} -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_MISC_KERNEL_H +#endif// EIGEN_MISC_KERNEL_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/misc/RealSvd2x2.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/misc/RealSvd2x2.h index abb4d3c2..d9b5b442 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/misc/RealSvd2x2.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/misc/RealSvd2x2.h @@ -15,41 +15,40 @@ namespace Eigen { namespace internal { -template -void real_2x2_jacobi_svd(const MatrixType& matrix, Index p, Index q, - JacobiRotation *j_left, - JacobiRotation *j_right) -{ - using std::sqrt; - using std::abs; - Matrix m; - m << numext::real(matrix.coeff(p,p)), numext::real(matrix.coeff(p,q)), - numext::real(matrix.coeff(q,p)), numext::real(matrix.coeff(q,q)); - JacobiRotation rot1; - RealScalar t = m.coeff(0,0) + m.coeff(1,1); - RealScalar d = m.coeff(1,0) - m.coeff(0,1); - - if(abs(d) < (std::numeric_limits::min)()) + template + void real_2x2_jacobi_svd(const MatrixType &matrix, + Index p, + Index q, + JacobiRotation *j_left, + JacobiRotation *j_right) { - rot1.s() = RealScalar(0); - rot1.c() = RealScalar(1); + using std::sqrt; + using std::abs; + Matrix m; + m << numext::real(matrix.coeff(p, p)), numext::real(matrix.coeff(p, q)), numext::real(matrix.coeff(q, p)), + numext::real(matrix.coeff(q, q)); + JacobiRotation rot1; + RealScalar t = m.coeff(0, 0) + m.coeff(1, 1); + RealScalar d = m.coeff(1, 0) - m.coeff(0, 1); + + if (abs(d) < (std::numeric_limits::min)()) { + rot1.s() = RealScalar(0); + rot1.c() = RealScalar(1); + } else { + // If d!=0, then t/d cannot overflow because the magnitude of the + // entries forming d are not too small compared to the ones forming t. + RealScalar u = t / d; + RealScalar tmp = sqrt(RealScalar(1) + numext::abs2(u)); + rot1.s() = RealScalar(1) / tmp; + rot1.c() = u / tmp; + } + m.applyOnTheLeft(0, 1, rot1); + j_right->makeJacobi(m, 0, 1); + *j_left = rot1 * j_right->transpose(); } - else - { - // If d!=0, then t/d cannot overflow because the magnitude of the - // entries forming d are not too small compared to the ones forming t. - RealScalar u = t / d; - RealScalar tmp = sqrt(RealScalar(1) + numext::abs2(u)); - rot1.s() = RealScalar(1) / tmp; - rot1.c() = u / tmp; - } - m.applyOnTheLeft(0,1,rot1); - j_right->makeJacobi(m,0,1); - *j_left = rot1 * j_right->transpose(); -} -} // end namespace internal +}// end namespace internal -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_REALSVD2X2_H +#endif// EIGEN_REALSVD2X2_H diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/misc/blas.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/misc/blas.h index 25215b15..5a0814c9 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/misc/blas.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/misc/blas.h @@ -2,8 +2,7 @@ #define BLAS_H #ifdef __cplusplus -extern "C" -{ +extern "C" { #endif #define BLASFUNC(FUNC) FUNC##_ @@ -16,421 +15,1057 @@ typedef long BLASLONG; typedef unsigned long BLASULONG; #endif -int BLASFUNC(xerbla)(const char *, int *info, int); - -float BLASFUNC(sdot) (int *, float *, int *, float *, int *); -float BLASFUNC(sdsdot)(int *, float *, float *, int *, float *, int *); - -double BLASFUNC(dsdot) (int *, float *, int *, float *, int *); -double BLASFUNC(ddot) (int *, double *, int *, double *, int *); -double BLASFUNC(qdot) (int *, double *, int *, double *, int *); - -int BLASFUNC(cdotuw) (int *, float *, int *, float *, int *, float*); -int BLASFUNC(cdotcw) (int *, float *, int *, float *, int *, float*); -int BLASFUNC(zdotuw) (int *, double *, int *, double *, int *, double*); -int BLASFUNC(zdotcw) (int *, double *, int *, double *, int *, double*); - -int BLASFUNC(saxpy) (const int *, const float *, const float *, const int *, float *, const int *); -int BLASFUNC(daxpy) (const int *, const double *, const double *, const int *, double *, const int *); -int BLASFUNC(qaxpy) (const int *, const double *, const double *, const int *, double *, const int *); -int BLASFUNC(caxpy) (const int *, const float *, const float *, const int *, float *, const int *); -int BLASFUNC(zaxpy) (const int *, const double *, const double *, const int *, double *, const int *); -int BLASFUNC(xaxpy) (const int *, const double *, const double *, const int *, double *, const int *); -int BLASFUNC(caxpyc)(const int *, const float *, const float *, const int *, float *, const int *); -int BLASFUNC(zaxpyc)(const int *, const double *, const double *, const int *, double *, const int *); -int BLASFUNC(xaxpyc)(const int *, const double *, const double *, const int *, double *, const int *); - -int BLASFUNC(scopy) (int *, float *, int *, float *, int *); -int BLASFUNC(dcopy) (int *, double *, int *, double *, int *); -int BLASFUNC(qcopy) (int *, double *, int *, double *, int *); -int BLASFUNC(ccopy) (int *, float *, int *, float *, int *); -int BLASFUNC(zcopy) (int *, double *, int *, double *, int *); -int BLASFUNC(xcopy) (int *, double *, int *, double *, int *); - -int BLASFUNC(sswap) (int *, float *, int *, float *, int *); -int BLASFUNC(dswap) (int *, double *, int *, double *, int *); -int BLASFUNC(qswap) (int *, double *, int *, double *, int *); -int BLASFUNC(cswap) (int *, float *, int *, float *, int *); -int BLASFUNC(zswap) (int *, double *, int *, double *, int *); -int BLASFUNC(xswap) (int *, double *, int *, double *, int *); - -float BLASFUNC(sasum) (int *, float *, int *); -float BLASFUNC(scasum)(int *, float *, int *); -double BLASFUNC(dasum) (int *, double *, int *); -double BLASFUNC(qasum) (int *, double *, int *); +int BLASFUNC(xerbla)(const char *, int *info, int); + +float BLASFUNC(sdot)(int *, float *, int *, float *, int *); +float BLASFUNC(sdsdot)(int *, float *, float *, int *, float *, int *); + +double BLASFUNC(dsdot)(int *, float *, int *, float *, int *); +double BLASFUNC(ddot)(int *, double *, int *, double *, int *); +double BLASFUNC(qdot)(int *, double *, int *, double *, int *); + +int BLASFUNC(cdotuw)(int *, float *, int *, float *, int *, float *); +int BLASFUNC(cdotcw)(int *, float *, int *, float *, int *, float *); +int BLASFUNC(zdotuw)(int *, double *, int *, double *, int *, double *); +int BLASFUNC(zdotcw)(int *, double *, int *, double *, int *, double *); + +int BLASFUNC(saxpy)(const int *, const float *, const float *, const int *, float *, const int *); +int BLASFUNC(daxpy)(const int *, const double *, const double *, const int *, double *, const int *); +int BLASFUNC(qaxpy)(const int *, const double *, const double *, const int *, double *, const int *); +int BLASFUNC(caxpy)(const int *, const float *, const float *, const int *, float *, const int *); +int BLASFUNC(zaxpy)(const int *, const double *, const double *, const int *, double *, const int *); +int BLASFUNC(xaxpy)(const int *, const double *, const double *, const int *, double *, const int *); +int BLASFUNC(caxpyc)(const int *, const float *, const float *, const int *, float *, const int *); +int BLASFUNC(zaxpyc)(const int *, const double *, const double *, const int *, double *, const int *); +int BLASFUNC(xaxpyc)(const int *, const double *, const double *, const int *, double *, const int *); + +int BLASFUNC(scopy)(int *, float *, int *, float *, int *); +int BLASFUNC(dcopy)(int *, double *, int *, double *, int *); +int BLASFUNC(qcopy)(int *, double *, int *, double *, int *); +int BLASFUNC(ccopy)(int *, float *, int *, float *, int *); +int BLASFUNC(zcopy)(int *, double *, int *, double *, int *); +int BLASFUNC(xcopy)(int *, double *, int *, double *, int *); + +int BLASFUNC(sswap)(int *, float *, int *, float *, int *); +int BLASFUNC(dswap)(int *, double *, int *, double *, int *); +int BLASFUNC(qswap)(int *, double *, int *, double *, int *); +int BLASFUNC(cswap)(int *, float *, int *, float *, int *); +int BLASFUNC(zswap)(int *, double *, int *, double *, int *); +int BLASFUNC(xswap)(int *, double *, int *, double *, int *); + +float BLASFUNC(sasum)(int *, float *, int *); +float BLASFUNC(scasum)(int *, float *, int *); +double BLASFUNC(dasum)(int *, double *, int *); +double BLASFUNC(qasum)(int *, double *, int *); double BLASFUNC(dzasum)(int *, double *, int *); double BLASFUNC(qxasum)(int *, double *, int *); -int BLASFUNC(isamax)(int *, float *, int *); -int BLASFUNC(idamax)(int *, double *, int *); -int BLASFUNC(iqamax)(int *, double *, int *); -int BLASFUNC(icamax)(int *, float *, int *); -int BLASFUNC(izamax)(int *, double *, int *); -int BLASFUNC(ixamax)(int *, double *, int *); - -int BLASFUNC(ismax) (int *, float *, int *); -int BLASFUNC(idmax) (int *, double *, int *); -int BLASFUNC(iqmax) (int *, double *, int *); -int BLASFUNC(icmax) (int *, float *, int *); -int BLASFUNC(izmax) (int *, double *, int *); -int BLASFUNC(ixmax) (int *, double *, int *); - -int BLASFUNC(isamin)(int *, float *, int *); -int BLASFUNC(idamin)(int *, double *, int *); -int BLASFUNC(iqamin)(int *, double *, int *); -int BLASFUNC(icamin)(int *, float *, int *); -int BLASFUNC(izamin)(int *, double *, int *); -int BLASFUNC(ixamin)(int *, double *, int *); - -int BLASFUNC(ismin)(int *, float *, int *); -int BLASFUNC(idmin)(int *, double *, int *); -int BLASFUNC(iqmin)(int *, double *, int *); -int BLASFUNC(icmin)(int *, float *, int *); -int BLASFUNC(izmin)(int *, double *, int *); -int BLASFUNC(ixmin)(int *, double *, int *); - -float BLASFUNC(samax) (int *, float *, int *); -double BLASFUNC(damax) (int *, double *, int *); -double BLASFUNC(qamax) (int *, double *, int *); -float BLASFUNC(scamax)(int *, float *, int *); +int BLASFUNC(isamax)(int *, float *, int *); +int BLASFUNC(idamax)(int *, double *, int *); +int BLASFUNC(iqamax)(int *, double *, int *); +int BLASFUNC(icamax)(int *, float *, int *); +int BLASFUNC(izamax)(int *, double *, int *); +int BLASFUNC(ixamax)(int *, double *, int *); + +int BLASFUNC(ismax)(int *, float *, int *); +int BLASFUNC(idmax)(int *, double *, int *); +int BLASFUNC(iqmax)(int *, double *, int *); +int BLASFUNC(icmax)(int *, float *, int *); +int BLASFUNC(izmax)(int *, double *, int *); +int BLASFUNC(ixmax)(int *, double *, int *); + +int BLASFUNC(isamin)(int *, float *, int *); +int BLASFUNC(idamin)(int *, double *, int *); +int BLASFUNC(iqamin)(int *, double *, int *); +int BLASFUNC(icamin)(int *, float *, int *); +int BLASFUNC(izamin)(int *, double *, int *); +int BLASFUNC(ixamin)(int *, double *, int *); + +int BLASFUNC(ismin)(int *, float *, int *); +int BLASFUNC(idmin)(int *, double *, int *); +int BLASFUNC(iqmin)(int *, double *, int *); +int BLASFUNC(icmin)(int *, float *, int *); +int BLASFUNC(izmin)(int *, double *, int *); +int BLASFUNC(ixmin)(int *, double *, int *); + +float BLASFUNC(samax)(int *, float *, int *); +double BLASFUNC(damax)(int *, double *, int *); +double BLASFUNC(qamax)(int *, double *, int *); +float BLASFUNC(scamax)(int *, float *, int *); double BLASFUNC(dzamax)(int *, double *, int *); double BLASFUNC(qxamax)(int *, double *, int *); -float BLASFUNC(samin) (int *, float *, int *); -double BLASFUNC(damin) (int *, double *, int *); -double BLASFUNC(qamin) (int *, double *, int *); -float BLASFUNC(scamin)(int *, float *, int *); +float BLASFUNC(samin)(int *, float *, int *); +double BLASFUNC(damin)(int *, double *, int *); +double BLASFUNC(qamin)(int *, double *, int *); +float BLASFUNC(scamin)(int *, float *, int *); double BLASFUNC(dzamin)(int *, double *, int *); double BLASFUNC(qxamin)(int *, double *, int *); -float BLASFUNC(smax) (int *, float *, int *); -double BLASFUNC(dmax) (int *, double *, int *); -double BLASFUNC(qmax) (int *, double *, int *); -float BLASFUNC(scmax) (int *, float *, int *); -double BLASFUNC(dzmax) (int *, double *, int *); -double BLASFUNC(qxmax) (int *, double *, int *); - -float BLASFUNC(smin) (int *, float *, int *); -double BLASFUNC(dmin) (int *, double *, int *); -double BLASFUNC(qmin) (int *, double *, int *); -float BLASFUNC(scmin) (int *, float *, int *); -double BLASFUNC(dzmin) (int *, double *, int *); -double BLASFUNC(qxmin) (int *, double *, int *); - -int BLASFUNC(sscal) (int *, float *, float *, int *); -int BLASFUNC(dscal) (int *, double *, double *, int *); -int BLASFUNC(qscal) (int *, double *, double *, int *); -int BLASFUNC(cscal) (int *, float *, float *, int *); -int BLASFUNC(zscal) (int *, double *, double *, int *); -int BLASFUNC(xscal) (int *, double *, double *, int *); -int BLASFUNC(csscal)(int *, float *, float *, int *); -int BLASFUNC(zdscal)(int *, double *, double *, int *); -int BLASFUNC(xqscal)(int *, double *, double *, int *); - -float BLASFUNC(snrm2) (int *, float *, int *); -float BLASFUNC(scnrm2)(int *, float *, int *); - -double BLASFUNC(dnrm2) (int *, double *, int *); -double BLASFUNC(qnrm2) (int *, double *, int *); +float BLASFUNC(smax)(int *, float *, int *); +double BLASFUNC(dmax)(int *, double *, int *); +double BLASFUNC(qmax)(int *, double *, int *); +float BLASFUNC(scmax)(int *, float *, int *); +double BLASFUNC(dzmax)(int *, double *, int *); +double BLASFUNC(qxmax)(int *, double *, int *); + +float BLASFUNC(smin)(int *, float *, int *); +double BLASFUNC(dmin)(int *, double *, int *); +double BLASFUNC(qmin)(int *, double *, int *); +float BLASFUNC(scmin)(int *, float *, int *); +double BLASFUNC(dzmin)(int *, double *, int *); +double BLASFUNC(qxmin)(int *, double *, int *); + +int BLASFUNC(sscal)(int *, float *, float *, int *); +int BLASFUNC(dscal)(int *, double *, double *, int *); +int BLASFUNC(qscal)(int *, double *, double *, int *); +int BLASFUNC(cscal)(int *, float *, float *, int *); +int BLASFUNC(zscal)(int *, double *, double *, int *); +int BLASFUNC(xscal)(int *, double *, double *, int *); +int BLASFUNC(csscal)(int *, float *, float *, int *); +int BLASFUNC(zdscal)(int *, double *, double *, int *); +int BLASFUNC(xqscal)(int *, double *, double *, int *); + +float BLASFUNC(snrm2)(int *, float *, int *); +float BLASFUNC(scnrm2)(int *, float *, int *); + +double BLASFUNC(dnrm2)(int *, double *, int *); +double BLASFUNC(qnrm2)(int *, double *, int *); double BLASFUNC(dznrm2)(int *, double *, int *); double BLASFUNC(qxnrm2)(int *, double *, int *); -int BLASFUNC(srot) (int *, float *, int *, float *, int *, float *, float *); -int BLASFUNC(drot) (int *, double *, int *, double *, int *, double *, double *); -int BLASFUNC(qrot) (int *, double *, int *, double *, int *, double *, double *); -int BLASFUNC(csrot) (int *, float *, int *, float *, int *, float *, float *); -int BLASFUNC(zdrot) (int *, double *, int *, double *, int *, double *, double *); -int BLASFUNC(xqrot) (int *, double *, int *, double *, int *, double *, double *); +int BLASFUNC(srot)(int *, float *, int *, float *, int *, float *, float *); +int BLASFUNC(drot)(int *, double *, int *, double *, int *, double *, double *); +int BLASFUNC(qrot)(int *, double *, int *, double *, int *, double *, double *); +int BLASFUNC(csrot)(int *, float *, int *, float *, int *, float *, float *); +int BLASFUNC(zdrot)(int *, double *, int *, double *, int *, double *, double *); +int BLASFUNC(xqrot)(int *, double *, int *, double *, int *, double *, double *); -int BLASFUNC(srotg) (float *, float *, float *, float *); -int BLASFUNC(drotg) (double *, double *, double *, double *); -int BLASFUNC(qrotg) (double *, double *, double *, double *); -int BLASFUNC(crotg) (float *, float *, float *, float *); -int BLASFUNC(zrotg) (double *, double *, double *, double *); -int BLASFUNC(xrotg) (double *, double *, double *, double *); +int BLASFUNC(srotg)(float *, float *, float *, float *); +int BLASFUNC(drotg)(double *, double *, double *, double *); +int BLASFUNC(qrotg)(double *, double *, double *, double *); +int BLASFUNC(crotg)(float *, float *, float *, float *); +int BLASFUNC(zrotg)(double *, double *, double *, double *); +int BLASFUNC(xrotg)(double *, double *, double *, double *); -int BLASFUNC(srotmg)(float *, float *, float *, float *, float *); -int BLASFUNC(drotmg)(double *, double *, double *, double *, double *); +int BLASFUNC(srotmg)(float *, float *, float *, float *, float *); +int BLASFUNC(drotmg)(double *, double *, double *, double *, double *); -int BLASFUNC(srotm) (int *, float *, int *, float *, int *, float *); -int BLASFUNC(drotm) (int *, double *, int *, double *, int *, double *); -int BLASFUNC(qrotm) (int *, double *, int *, double *, int *, double *); +int BLASFUNC(srotm)(int *, float *, int *, float *, int *, float *); +int BLASFUNC(drotm)(int *, double *, int *, double *, int *, double *); +int BLASFUNC(qrotm)(int *, double *, int *, double *, int *, double *); /* Level 2 routines */ -int BLASFUNC(sger)(int *, int *, float *, float *, int *, - float *, int *, float *, int *); -int BLASFUNC(dger)(int *, int *, double *, double *, int *, - double *, int *, double *, int *); -int BLASFUNC(qger)(int *, int *, double *, double *, int *, - double *, int *, double *, int *); -int BLASFUNC(cgeru)(int *, int *, float *, float *, int *, - float *, int *, float *, int *); -int BLASFUNC(cgerc)(int *, int *, float *, float *, int *, - float *, int *, float *, int *); -int BLASFUNC(zgeru)(int *, int *, double *, double *, int *, - double *, int *, double *, int *); -int BLASFUNC(zgerc)(int *, int *, double *, double *, int *, - double *, int *, double *, int *); -int BLASFUNC(xgeru)(int *, int *, double *, double *, int *, - double *, int *, double *, int *); -int BLASFUNC(xgerc)(int *, int *, double *, double *, int *, - double *, int *, double *, int *); - -int BLASFUNC(sgemv)(const char *, const int *, const int *, const float *, const float *, const int *, const float *, const int *, const float *, float *, const int *); -int BLASFUNC(dgemv)(const char *, const int *, const int *, const double *, const double *, const int *, const double *, const int *, const double *, double *, const int *); -int BLASFUNC(qgemv)(const char *, const int *, const int *, const double *, const double *, const int *, const double *, const int *, const double *, double *, const int *); -int BLASFUNC(cgemv)(const char *, const int *, const int *, const float *, const float *, const int *, const float *, const int *, const float *, float *, const int *); -int BLASFUNC(zgemv)(const char *, const int *, const int *, const double *, const double *, const int *, const double *, const int *, const double *, double *, const int *); -int BLASFUNC(xgemv)(const char *, const int *, const int *, const double *, const double *, const int *, const double *, const int *, const double *, double *, const int *); - -int BLASFUNC(strsv) (const char *, const char *, const char *, const int *, const float *, const int *, float *, const int *); -int BLASFUNC(dtrsv) (const char *, const char *, const char *, const int *, const double *, const int *, double *, const int *); -int BLASFUNC(qtrsv) (const char *, const char *, const char *, const int *, const double *, const int *, double *, const int *); -int BLASFUNC(ctrsv) (const char *, const char *, const char *, const int *, const float *, const int *, float *, const int *); -int BLASFUNC(ztrsv) (const char *, const char *, const char *, const int *, const double *, const int *, double *, const int *); -int BLASFUNC(xtrsv) (const char *, const char *, const char *, const int *, const double *, const int *, double *, const int *); - -int BLASFUNC(stpsv) (char *, char *, char *, int *, float *, float *, int *); -int BLASFUNC(dtpsv) (char *, char *, char *, int *, double *, double *, int *); -int BLASFUNC(qtpsv) (char *, char *, char *, int *, double *, double *, int *); -int BLASFUNC(ctpsv) (char *, char *, char *, int *, float *, float *, int *); -int BLASFUNC(ztpsv) (char *, char *, char *, int *, double *, double *, int *); -int BLASFUNC(xtpsv) (char *, char *, char *, int *, double *, double *, int *); - -int BLASFUNC(strmv) (const char *, const char *, const char *, const int *, const float *, const int *, float *, const int *); -int BLASFUNC(dtrmv) (const char *, const char *, const char *, const int *, const double *, const int *, double *, const int *); -int BLASFUNC(qtrmv) (const char *, const char *, const char *, const int *, const double *, const int *, double *, const int *); -int BLASFUNC(ctrmv) (const char *, const char *, const char *, const int *, const float *, const int *, float *, const int *); -int BLASFUNC(ztrmv) (const char *, const char *, const char *, const int *, const double *, const int *, double *, const int *); -int BLASFUNC(xtrmv) (const char *, const char *, const char *, const int *, const double *, const int *, double *, const int *); - -int BLASFUNC(stpmv) (char *, char *, char *, int *, float *, float *, int *); -int BLASFUNC(dtpmv) (char *, char *, char *, int *, double *, double *, int *); -int BLASFUNC(qtpmv) (char *, char *, char *, int *, double *, double *, int *); -int BLASFUNC(ctpmv) (char *, char *, char *, int *, float *, float *, int *); -int BLASFUNC(ztpmv) (char *, char *, char *, int *, double *, double *, int *); -int BLASFUNC(xtpmv) (char *, char *, char *, int *, double *, double *, int *); - -int BLASFUNC(stbmv) (char *, char *, char *, int *, int *, float *, int *, float *, int *); -int BLASFUNC(dtbmv) (char *, char *, char *, int *, int *, double *, int *, double *, int *); -int BLASFUNC(qtbmv) (char *, char *, char *, int *, int *, double *, int *, double *, int *); -int BLASFUNC(ctbmv) (char *, char *, char *, int *, int *, float *, int *, float *, int *); -int BLASFUNC(ztbmv) (char *, char *, char *, int *, int *, double *, int *, double *, int *); -int BLASFUNC(xtbmv) (char *, char *, char *, int *, int *, double *, int *, double *, int *); - -int BLASFUNC(stbsv) (char *, char *, char *, int *, int *, float *, int *, float *, int *); -int BLASFUNC(dtbsv) (char *, char *, char *, int *, int *, double *, int *, double *, int *); -int BLASFUNC(qtbsv) (char *, char *, char *, int *, int *, double *, int *, double *, int *); -int BLASFUNC(ctbsv) (char *, char *, char *, int *, int *, float *, int *, float *, int *); -int BLASFUNC(ztbsv) (char *, char *, char *, int *, int *, double *, int *, double *, int *); -int BLASFUNC(xtbsv) (char *, char *, char *, int *, int *, double *, int *, double *, int *); - -int BLASFUNC(ssymv) (const char *, const int *, const float *, const float *, const int *, const float *, const int *, const float *, float *, const int *); -int BLASFUNC(dsymv) (const char *, const int *, const double *, const double *, const int *, const double *, const int *, const double *, double *, const int *); -int BLASFUNC(qsymv) (const char *, const int *, const double *, const double *, const int *, const double *, const int *, const double *, double *, const int *); - -int BLASFUNC(sspmv) (char *, int *, float *, float *, - float *, int *, float *, float *, int *); -int BLASFUNC(dspmv) (char *, int *, double *, double *, - double *, int *, double *, double *, int *); -int BLASFUNC(qspmv) (char *, int *, double *, double *, - double *, int *, double *, double *, int *); - -int BLASFUNC(ssyr) (const char *, const int *, const float *, const float *, const int *, float *, const int *); -int BLASFUNC(dsyr) (const char *, const int *, const double *, const double *, const int *, double *, const int *); -int BLASFUNC(qsyr) (const char *, const int *, const double *, const double *, const int *, double *, const int *); - -int BLASFUNC(ssyr2) (const char *, const int *, const float *, const float *, const int *, const float *, const int *, float *, const int *); -int BLASFUNC(dsyr2) (const char *, const int *, const double *, const double *, const int *, const double *, const int *, double *, const int *); -int BLASFUNC(qsyr2) (const char *, const int *, const double *, const double *, const int *, const double *, const int *, double *, const int *); -int BLASFUNC(csyr2) (const char *, const int *, const float *, const float *, const int *, const float *, const int *, float *, const int *); -int BLASFUNC(zsyr2) (const char *, const int *, const double *, const double *, const int *, const double *, const int *, double *, const int *); -int BLASFUNC(xsyr2) (const char *, const int *, const double *, const double *, const int *, const double *, const int *, double *, const int *); - -int BLASFUNC(sspr) (char *, int *, float *, float *, int *, - float *); -int BLASFUNC(dspr) (char *, int *, double *, double *, int *, - double *); -int BLASFUNC(qspr) (char *, int *, double *, double *, int *, - double *); - -int BLASFUNC(sspr2) (char *, int *, float *, - float *, int *, float *, int *, float *); -int BLASFUNC(dspr2) (char *, int *, double *, - double *, int *, double *, int *, double *); -int BLASFUNC(qspr2) (char *, int *, double *, - double *, int *, double *, int *, double *); -int BLASFUNC(cspr2) (char *, int *, float *, - float *, int *, float *, int *, float *); -int BLASFUNC(zspr2) (char *, int *, double *, - double *, int *, double *, int *, double *); -int BLASFUNC(xspr2) (char *, int *, double *, - double *, int *, double *, int *, double *); - -int BLASFUNC(cher) (char *, int *, float *, float *, int *, - float *, int *); -int BLASFUNC(zher) (char *, int *, double *, double *, int *, - double *, int *); -int BLASFUNC(xher) (char *, int *, double *, double *, int *, - double *, int *); - -int BLASFUNC(chpr) (char *, int *, float *, float *, int *, float *); -int BLASFUNC(zhpr) (char *, int *, double *, double *, int *, double *); -int BLASFUNC(xhpr) (char *, int *, double *, double *, int *, double *); - -int BLASFUNC(cher2) (char *, int *, float *, - float *, int *, float *, int *, float *, int *); -int BLASFUNC(zher2) (char *, int *, double *, - double *, int *, double *, int *, double *, int *); -int BLASFUNC(xher2) (char *, int *, double *, - double *, int *, double *, int *, double *, int *); - -int BLASFUNC(chpr2) (char *, int *, float *, - float *, int *, float *, int *, float *); -int BLASFUNC(zhpr2) (char *, int *, double *, - double *, int *, double *, int *, double *); -int BLASFUNC(xhpr2) (char *, int *, double *, - double *, int *, double *, int *, double *); - -int BLASFUNC(chemv) (const char *, const int *, const float *, const float *, const int *, const float *, const int *, const float *, float *, const int *); -int BLASFUNC(zhemv) (const char *, const int *, const double *, const double *, const int *, const double *, const int *, const double *, double *, const int *); -int BLASFUNC(xhemv) (const char *, const int *, const double *, const double *, const int *, const double *, const int *, const double *, double *, const int *); - -int BLASFUNC(chpmv) (char *, int *, float *, float *, - float *, int *, float *, float *, int *); -int BLASFUNC(zhpmv) (char *, int *, double *, double *, - double *, int *, double *, double *, int *); -int BLASFUNC(xhpmv) (char *, int *, double *, double *, - double *, int *, double *, double *, int *); - -int BLASFUNC(snorm)(char *, int *, int *, float *, int *); +int BLASFUNC(sger)(int *, int *, float *, float *, int *, float *, int *, float *, int *); +int BLASFUNC(dger)(int *, int *, double *, double *, int *, double *, int *, double *, int *); +int BLASFUNC(qger)(int *, int *, double *, double *, int *, double *, int *, double *, int *); +int BLASFUNC(cgeru)(int *, int *, float *, float *, int *, float *, int *, float *, int *); +int BLASFUNC(cgerc)(int *, int *, float *, float *, int *, float *, int *, float *, int *); +int BLASFUNC(zgeru)(int *, int *, double *, double *, int *, double *, int *, double *, int *); +int BLASFUNC(zgerc)(int *, int *, double *, double *, int *, double *, int *, double *, int *); +int BLASFUNC(xgeru)(int *, int *, double *, double *, int *, double *, int *, double *, int *); +int BLASFUNC(xgerc)(int *, int *, double *, double *, int *, double *, int *, double *, int *); + +int BLASFUNC(sgemv)(const char *, + const int *, + const int *, + const float *, + const float *, + const int *, + const float *, + const int *, + const float *, + float *, + const int *); +int BLASFUNC(dgemv)(const char *, + const int *, + const int *, + const double *, + const double *, + const int *, + const double *, + const int *, + const double *, + double *, + const int *); +int BLASFUNC(qgemv)(const char *, + const int *, + const int *, + const double *, + const double *, + const int *, + const double *, + const int *, + const double *, + double *, + const int *); +int BLASFUNC(cgemv)(const char *, + const int *, + const int *, + const float *, + const float *, + const int *, + const float *, + const int *, + const float *, + float *, + const int *); +int BLASFUNC(zgemv)(const char *, + const int *, + const int *, + const double *, + const double *, + const int *, + const double *, + const int *, + const double *, + double *, + const int *); +int BLASFUNC(xgemv)(const char *, + const int *, + const int *, + const double *, + const double *, + const int *, + const double *, + const int *, + const double *, + double *, + const int *); + +int BLASFUNC( + strsv)(const char *, const char *, const char *, const int *, const float *, const int *, float *, const int *); +int BLASFUNC( + dtrsv)(const char *, const char *, const char *, const int *, const double *, const int *, double *, const int *); +int BLASFUNC( + qtrsv)(const char *, const char *, const char *, const int *, const double *, const int *, double *, const int *); +int BLASFUNC( + ctrsv)(const char *, const char *, const char *, const int *, const float *, const int *, float *, const int *); +int BLASFUNC( + ztrsv)(const char *, const char *, const char *, const int *, const double *, const int *, double *, const int *); +int BLASFUNC( + xtrsv)(const char *, const char *, const char *, const int *, const double *, const int *, double *, const int *); + +int BLASFUNC(stpsv)(char *, char *, char *, int *, float *, float *, int *); +int BLASFUNC(dtpsv)(char *, char *, char *, int *, double *, double *, int *); +int BLASFUNC(qtpsv)(char *, char *, char *, int *, double *, double *, int *); +int BLASFUNC(ctpsv)(char *, char *, char *, int *, float *, float *, int *); +int BLASFUNC(ztpsv)(char *, char *, char *, int *, double *, double *, int *); +int BLASFUNC(xtpsv)(char *, char *, char *, int *, double *, double *, int *); + +int BLASFUNC( + strmv)(const char *, const char *, const char *, const int *, const float *, const int *, float *, const int *); +int BLASFUNC( + dtrmv)(const char *, const char *, const char *, const int *, const double *, const int *, double *, const int *); +int BLASFUNC( + qtrmv)(const char *, const char *, const char *, const int *, const double *, const int *, double *, const int *); +int BLASFUNC( + ctrmv)(const char *, const char *, const char *, const int *, const float *, const int *, float *, const int *); +int BLASFUNC( + ztrmv)(const char *, const char *, const char *, const int *, const double *, const int *, double *, const int *); +int BLASFUNC( + xtrmv)(const char *, const char *, const char *, const int *, const double *, const int *, double *, const int *); + +int BLASFUNC(stpmv)(char *, char *, char *, int *, float *, float *, int *); +int BLASFUNC(dtpmv)(char *, char *, char *, int *, double *, double *, int *); +int BLASFUNC(qtpmv)(char *, char *, char *, int *, double *, double *, int *); +int BLASFUNC(ctpmv)(char *, char *, char *, int *, float *, float *, int *); +int BLASFUNC(ztpmv)(char *, char *, char *, int *, double *, double *, int *); +int BLASFUNC(xtpmv)(char *, char *, char *, int *, double *, double *, int *); + +int BLASFUNC(stbmv)(char *, char *, char *, int *, int *, float *, int *, float *, int *); +int BLASFUNC(dtbmv)(char *, char *, char *, int *, int *, double *, int *, double *, int *); +int BLASFUNC(qtbmv)(char *, char *, char *, int *, int *, double *, int *, double *, int *); +int BLASFUNC(ctbmv)(char *, char *, char *, int *, int *, float *, int *, float *, int *); +int BLASFUNC(ztbmv)(char *, char *, char *, int *, int *, double *, int *, double *, int *); +int BLASFUNC(xtbmv)(char *, char *, char *, int *, int *, double *, int *, double *, int *); + +int BLASFUNC(stbsv)(char *, char *, char *, int *, int *, float *, int *, float *, int *); +int BLASFUNC(dtbsv)(char *, char *, char *, int *, int *, double *, int *, double *, int *); +int BLASFUNC(qtbsv)(char *, char *, char *, int *, int *, double *, int *, double *, int *); +int BLASFUNC(ctbsv)(char *, char *, char *, int *, int *, float *, int *, float *, int *); +int BLASFUNC(ztbsv)(char *, char *, char *, int *, int *, double *, int *, double *, int *); +int BLASFUNC(xtbsv)(char *, char *, char *, int *, int *, double *, int *, double *, int *); + +int BLASFUNC(ssymv)(const char *, + const int *, + const float *, + const float *, + const int *, + const float *, + const int *, + const float *, + float *, + const int *); +int BLASFUNC(dsymv)(const char *, + const int *, + const double *, + const double *, + const int *, + const double *, + const int *, + const double *, + double *, + const int *); +int BLASFUNC(qsymv)(const char *, + const int *, + const double *, + const double *, + const int *, + const double *, + const int *, + const double *, + double *, + const int *); + +int BLASFUNC(sspmv)(char *, int *, float *, float *, float *, int *, float *, float *, int *); +int BLASFUNC(dspmv)(char *, int *, double *, double *, double *, int *, double *, double *, int *); +int BLASFUNC(qspmv)(char *, int *, double *, double *, double *, int *, double *, double *, int *); + +int BLASFUNC(ssyr)(const char *, const int *, const float *, const float *, const int *, float *, const int *); +int BLASFUNC(dsyr)(const char *, const int *, const double *, const double *, const int *, double *, const int *); +int BLASFUNC(qsyr)(const char *, const int *, const double *, const double *, const int *, double *, const int *); + +int BLASFUNC(ssyr2)(const char *, + const int *, + const float *, + const float *, + const int *, + const float *, + const int *, + float *, + const int *); +int BLASFUNC(dsyr2)(const char *, + const int *, + const double *, + const double *, + const int *, + const double *, + const int *, + double *, + const int *); +int BLASFUNC(qsyr2)(const char *, + const int *, + const double *, + const double *, + const int *, + const double *, + const int *, + double *, + const int *); +int BLASFUNC(csyr2)(const char *, + const int *, + const float *, + const float *, + const int *, + const float *, + const int *, + float *, + const int *); +int BLASFUNC(zsyr2)(const char *, + const int *, + const double *, + const double *, + const int *, + const double *, + const int *, + double *, + const int *); +int BLASFUNC(xsyr2)(const char *, + const int *, + const double *, + const double *, + const int *, + const double *, + const int *, + double *, + const int *); + +int BLASFUNC(sspr)(char *, int *, float *, float *, int *, float *); +int BLASFUNC(dspr)(char *, int *, double *, double *, int *, double *); +int BLASFUNC(qspr)(char *, int *, double *, double *, int *, double *); + +int BLASFUNC(sspr2)(char *, int *, float *, float *, int *, float *, int *, float *); +int BLASFUNC(dspr2)(char *, int *, double *, double *, int *, double *, int *, double *); +int BLASFUNC(qspr2)(char *, int *, double *, double *, int *, double *, int *, double *); +int BLASFUNC(cspr2)(char *, int *, float *, float *, int *, float *, int *, float *); +int BLASFUNC(zspr2)(char *, int *, double *, double *, int *, double *, int *, double *); +int BLASFUNC(xspr2)(char *, int *, double *, double *, int *, double *, int *, double *); + +int BLASFUNC(cher)(char *, int *, float *, float *, int *, float *, int *); +int BLASFUNC(zher)(char *, int *, double *, double *, int *, double *, int *); +int BLASFUNC(xher)(char *, int *, double *, double *, int *, double *, int *); + +int BLASFUNC(chpr)(char *, int *, float *, float *, int *, float *); +int BLASFUNC(zhpr)(char *, int *, double *, double *, int *, double *); +int BLASFUNC(xhpr)(char *, int *, double *, double *, int *, double *); + +int BLASFUNC(cher2)(char *, int *, float *, float *, int *, float *, int *, float *, int *); +int BLASFUNC(zher2)(char *, int *, double *, double *, int *, double *, int *, double *, int *); +int BLASFUNC(xher2)(char *, int *, double *, double *, int *, double *, int *, double *, int *); + +int BLASFUNC(chpr2)(char *, int *, float *, float *, int *, float *, int *, float *); +int BLASFUNC(zhpr2)(char *, int *, double *, double *, int *, double *, int *, double *); +int BLASFUNC(xhpr2)(char *, int *, double *, double *, int *, double *, int *, double *); + +int BLASFUNC(chemv)(const char *, + const int *, + const float *, + const float *, + const int *, + const float *, + const int *, + const float *, + float *, + const int *); +int BLASFUNC(zhemv)(const char *, + const int *, + const double *, + const double *, + const int *, + const double *, + const int *, + const double *, + double *, + const int *); +int BLASFUNC(xhemv)(const char *, + const int *, + const double *, + const double *, + const int *, + const double *, + const int *, + const double *, + double *, + const int *); + +int BLASFUNC(chpmv)(char *, int *, float *, float *, float *, int *, float *, float *, int *); +int BLASFUNC(zhpmv)(char *, int *, double *, double *, double *, int *, double *, double *, int *); +int BLASFUNC(xhpmv)(char *, int *, double *, double *, double *, int *, double *, double *, int *); + +int BLASFUNC(snorm)(char *, int *, int *, float *, int *); int BLASFUNC(dnorm)(char *, int *, int *, double *, int *); -int BLASFUNC(cnorm)(char *, int *, int *, float *, int *); +int BLASFUNC(cnorm)(char *, int *, int *, float *, int *); int BLASFUNC(znorm)(char *, int *, int *, double *, int *); -int BLASFUNC(sgbmv)(char *, int *, int *, int *, int *, float *, float *, int *, - float *, int *, float *, float *, int *); -int BLASFUNC(dgbmv)(char *, int *, int *, int *, int *, double *, double *, int *, - double *, int *, double *, double *, int *); -int BLASFUNC(qgbmv)(char *, int *, int *, int *, int *, double *, double *, int *, - double *, int *, double *, double *, int *); -int BLASFUNC(cgbmv)(char *, int *, int *, int *, int *, float *, float *, int *, - float *, int *, float *, float *, int *); -int BLASFUNC(zgbmv)(char *, int *, int *, int *, int *, double *, double *, int *, - double *, int *, double *, double *, int *); -int BLASFUNC(xgbmv)(char *, int *, int *, int *, int *, double *, double *, int *, - double *, int *, double *, double *, int *); - -int BLASFUNC(ssbmv)(char *, int *, int *, float *, float *, int *, - float *, int *, float *, float *, int *); -int BLASFUNC(dsbmv)(char *, int *, int *, double *, double *, int *, - double *, int *, double *, double *, int *); -int BLASFUNC(qsbmv)(char *, int *, int *, double *, double *, int *, - double *, int *, double *, double *, int *); -int BLASFUNC(csbmv)(char *, int *, int *, float *, float *, int *, - float *, int *, float *, float *, int *); -int BLASFUNC(zsbmv)(char *, int *, int *, double *, double *, int *, - double *, int *, double *, double *, int *); -int BLASFUNC(xsbmv)(char *, int *, int *, double *, double *, int *, - double *, int *, double *, double *, int *); - -int BLASFUNC(chbmv)(char *, int *, int *, float *, float *, int *, - float *, int *, float *, float *, int *); -int BLASFUNC(zhbmv)(char *, int *, int *, double *, double *, int *, - double *, int *, double *, double *, int *); -int BLASFUNC(xhbmv)(char *, int *, int *, double *, double *, int *, - double *, int *, double *, double *, int *); +int BLASFUNC( + sgbmv)(char *, int *, int *, int *, int *, float *, float *, int *, float *, int *, float *, float *, int *); +int BLASFUNC( + dgbmv)(char *, int *, int *, int *, int *, double *, double *, int *, double *, int *, double *, double *, int *); +int BLASFUNC( + qgbmv)(char *, int *, int *, int *, int *, double *, double *, int *, double *, int *, double *, double *, int *); +int BLASFUNC( + cgbmv)(char *, int *, int *, int *, int *, float *, float *, int *, float *, int *, float *, float *, int *); +int BLASFUNC( + zgbmv)(char *, int *, int *, int *, int *, double *, double *, int *, double *, int *, double *, double *, int *); +int BLASFUNC( + xgbmv)(char *, int *, int *, int *, int *, double *, double *, int *, double *, int *, double *, double *, int *); + +int BLASFUNC(ssbmv)(char *, int *, int *, float *, float *, int *, float *, int *, float *, float *, int *); +int BLASFUNC(dsbmv)(char *, int *, int *, double *, double *, int *, double *, int *, double *, double *, int *); +int BLASFUNC(qsbmv)(char *, int *, int *, double *, double *, int *, double *, int *, double *, double *, int *); +int BLASFUNC(csbmv)(char *, int *, int *, float *, float *, int *, float *, int *, float *, float *, int *); +int BLASFUNC(zsbmv)(char *, int *, int *, double *, double *, int *, double *, int *, double *, double *, int *); +int BLASFUNC(xsbmv)(char *, int *, int *, double *, double *, int *, double *, int *, double *, double *, int *); + +int BLASFUNC(chbmv)(char *, int *, int *, float *, float *, int *, float *, int *, float *, float *, int *); +int BLASFUNC(zhbmv)(char *, int *, int *, double *, double *, int *, double *, int *, double *, double *, int *); +int BLASFUNC(xhbmv)(char *, int *, int *, double *, double *, int *, double *, int *, double *, double *, int *); /* Level 3 routines */ -int BLASFUNC(sgemm)(const char *, const char *, const int *, const int *, const int *, const float *, const float *, const int *, const float *, const int *, const float *, float *, const int *); -int BLASFUNC(dgemm)(const char *, const char *, const int *, const int *, const int *, const double *, const double *, const int *, const double *, const int *, const double *, double *, const int *); -int BLASFUNC(qgemm)(const char *, const char *, const int *, const int *, const int *, const double *, const double *, const int *, const double *, const int *, const double *, double *, const int *); -int BLASFUNC(cgemm)(const char *, const char *, const int *, const int *, const int *, const float *, const float *, const int *, const float *, const int *, const float *, float *, const int *); -int BLASFUNC(zgemm)(const char *, const char *, const int *, const int *, const int *, const double *, const double *, const int *, const double *, const int *, const double *, double *, const int *); -int BLASFUNC(xgemm)(const char *, const char *, const int *, const int *, const int *, const double *, const double *, const int *, const double *, const int *, const double *, double *, const int *); - -int BLASFUNC(cgemm3m)(char *, char *, int *, int *, int *, float *, - float *, int *, float *, int *, float *, float *, int *); -int BLASFUNC(zgemm3m)(char *, char *, int *, int *, int *, double *, - double *, int *, double *, int *, double *, double *, int *); -int BLASFUNC(xgemm3m)(char *, char *, int *, int *, int *, double *, - double *, int *, double *, int *, double *, double *, int *); - -int BLASFUNC(sge2mm)(char *, char *, char *, int *, int *, - float *, float *, int *, float *, int *, - float *, float *, int *); -int BLASFUNC(dge2mm)(char *, char *, char *, int *, int *, - double *, double *, int *, double *, int *, - double *, double *, int *); -int BLASFUNC(cge2mm)(char *, char *, char *, int *, int *, - float *, float *, int *, float *, int *, - float *, float *, int *); -int BLASFUNC(zge2mm)(char *, char *, char *, int *, int *, - double *, double *, int *, double *, int *, - double *, double *, int *); - -int BLASFUNC(strsm)(const char *, const char *, const char *, const char *, const int *, const int *, const float *, const float *, const int *, float *, const int *); -int BLASFUNC(dtrsm)(const char *, const char *, const char *, const char *, const int *, const int *, const double *, const double *, const int *, double *, const int *); -int BLASFUNC(qtrsm)(const char *, const char *, const char *, const char *, const int *, const int *, const double *, const double *, const int *, double *, const int *); -int BLASFUNC(ctrsm)(const char *, const char *, const char *, const char *, const int *, const int *, const float *, const float *, const int *, float *, const int *); -int BLASFUNC(ztrsm)(const char *, const char *, const char *, const char *, const int *, const int *, const double *, const double *, const int *, double *, const int *); -int BLASFUNC(xtrsm)(const char *, const char *, const char *, const char *, const int *, const int *, const double *, const double *, const int *, double *, const int *); - -int BLASFUNC(strmm)(const char *, const char *, const char *, const char *, const int *, const int *, const float *, const float *, const int *, float *, const int *); -int BLASFUNC(dtrmm)(const char *, const char *, const char *, const char *, const int *, const int *, const double *, const double *, const int *, double *, const int *); -int BLASFUNC(qtrmm)(const char *, const char *, const char *, const char *, const int *, const int *, const double *, const double *, const int *, double *, const int *); -int BLASFUNC(ctrmm)(const char *, const char *, const char *, const char *, const int *, const int *, const float *, const float *, const int *, float *, const int *); -int BLASFUNC(ztrmm)(const char *, const char *, const char *, const char *, const int *, const int *, const double *, const double *, const int *, double *, const int *); -int BLASFUNC(xtrmm)(const char *, const char *, const char *, const char *, const int *, const int *, const double *, const double *, const int *, double *, const int *); - -int BLASFUNC(ssymm)(const char *, const char *, const int *, const int *, const float *, const float *, const int *, const float *, const int *, const float *, float *, const int *); -int BLASFUNC(dsymm)(const char *, const char *, const int *, const int *, const double *, const double *, const int *, const double *, const int *, const double *, double *, const int *); -int BLASFUNC(qsymm)(const char *, const char *, const int *, const int *, const double *, const double *, const int *, const double *, const int *, const double *, double *, const int *); -int BLASFUNC(csymm)(const char *, const char *, const int *, const int *, const float *, const float *, const int *, const float *, const int *, const float *, float *, const int *); -int BLASFUNC(zsymm)(const char *, const char *, const int *, const int *, const double *, const double *, const int *, const double *, const int *, const double *, double *, const int *); -int BLASFUNC(xsymm)(const char *, const char *, const int *, const int *, const double *, const double *, const int *, const double *, const int *, const double *, double *, const int *); - -int BLASFUNC(csymm3m)(char *, char *, int *, int *, float *, float *, int *, float *, int *, float *, float *, int *); -int BLASFUNC(zsymm3m)(char *, char *, int *, int *, double *, double *, int *, double *, int *, double *, double *, int *); -int BLASFUNC(xsymm3m)(char *, char *, int *, int *, double *, double *, int *, double *, int *, double *, double *, int *); - -int BLASFUNC(ssyrk)(const char *, const char *, const int *, const int *, const float *, const float *, const int *, const float *, float *, const int *); -int BLASFUNC(dsyrk)(const char *, const char *, const int *, const int *, const double *, const double *, const int *, const double *, double *, const int *); -int BLASFUNC(qsyrk)(const char *, const char *, const int *, const int *, const double *, const double *, const int *, const double *, double *, const int *); -int BLASFUNC(csyrk)(const char *, const char *, const int *, const int *, const float *, const float *, const int *, const float *, float *, const int *); -int BLASFUNC(zsyrk)(const char *, const char *, const int *, const int *, const double *, const double *, const int *, const double *, double *, const int *); -int BLASFUNC(xsyrk)(const char *, const char *, const int *, const int *, const double *, const double *, const int *, const double *, double *, const int *); - -int BLASFUNC(ssyr2k)(const char *, const char *, const int *, const int *, const float *, const float *, const int *, const float *, const int *, const float *, float *, const int *); -int BLASFUNC(dsyr2k)(const char *, const char *, const int *, const int *, const double *, const double *, const int *, const double*, const int *, const double *, double *, const int *); -int BLASFUNC(qsyr2k)(const char *, const char *, const int *, const int *, const double *, const double *, const int *, const double*, const int *, const double *, double *, const int *); -int BLASFUNC(csyr2k)(const char *, const char *, const int *, const int *, const float *, const float *, const int *, const float *, const int *, const float *, float *, const int *); -int BLASFUNC(zsyr2k)(const char *, const char *, const int *, const int *, const double *, const double *, const int *, const double*, const int *, const double *, double *, const int *); -int BLASFUNC(xsyr2k)(const char *, const char *, const int *, const int *, const double *, const double *, const int *, const double*, const int *, const double *, double *, const int *); - -int BLASFUNC(chemm)(const char *, const char *, const int *, const int *, const float *, const float *, const int *, const float *, const int *, const float *, float *, const int *); -int BLASFUNC(zhemm)(const char *, const char *, const int *, const int *, const double *, const double *, const int *, const double *, const int *, const double *, double *, const int *); -int BLASFUNC(xhemm)(const char *, const char *, const int *, const int *, const double *, const double *, const int *, const double *, const int *, const double *, double *, const int *); - -int BLASFUNC(chemm3m)(char *, char *, int *, int *, float *, float *, int *, - float *, int *, float *, float *, int *); -int BLASFUNC(zhemm3m)(char *, char *, int *, int *, double *, double *, int *, - double *, int *, double *, double *, int *); -int BLASFUNC(xhemm3m)(char *, char *, int *, int *, double *, double *, int *, - double *, int *, double *, double *, int *); - -int BLASFUNC(cherk)(const char *, const char *, const int *, const int *, const float *, const float *, const int *, const float *, float *, const int *); -int BLASFUNC(zherk)(const char *, const char *, const int *, const int *, const double *, const double *, const int *, const double *, double *, const int *); -int BLASFUNC(xherk)(const char *, const char *, const int *, const int *, const double *, const double *, const int *, const double *, double *, const int *); - -int BLASFUNC(cher2k)(const char *, const char *, const int *, const int *, const float *, const float *, const int *, const float *, const int *, const float *, float *, const int *); -int BLASFUNC(zher2k)(const char *, const char *, const int *, const int *, const double *, const double *, const int *, const double *, const int *, const double *, double *, const int *); -int BLASFUNC(xher2k)(const char *, const char *, const int *, const int *, const double *, const double *, const int *, const double *, const int *, const double *, double *, const int *); -int BLASFUNC(cher2m)(const char *, const char *, const char *, const int *, const int *, const float *, const float *, const int *, const float *, const int *, const float *, float *, const int *); -int BLASFUNC(zher2m)(const char *, const char *, const char *, const int *, const int *, const double *, const double *, const int *, const double*, const int *, const double *, double *, const int *); -int BLASFUNC(xher2m)(const char *, const char *, const char *, const int *, const int *, const double *, const double *, const int *, const double*, const int *, const double *, double *, const int *); +int BLASFUNC(sgemm)(const char *, + const char *, + const int *, + const int *, + const int *, + const float *, + const float *, + const int *, + const float *, + const int *, + const float *, + float *, + const int *); +int BLASFUNC(dgemm)(const char *, + const char *, + const int *, + const int *, + const int *, + const double *, + const double *, + const int *, + const double *, + const int *, + const double *, + double *, + const int *); +int BLASFUNC(qgemm)(const char *, + const char *, + const int *, + const int *, + const int *, + const double *, + const double *, + const int *, + const double *, + const int *, + const double *, + double *, + const int *); +int BLASFUNC(cgemm)(const char *, + const char *, + const int *, + const int *, + const int *, + const float *, + const float *, + const int *, + const float *, + const int *, + const float *, + float *, + const int *); +int BLASFUNC(zgemm)(const char *, + const char *, + const int *, + const int *, + const int *, + const double *, + const double *, + const int *, + const double *, + const int *, + const double *, + double *, + const int *); +int BLASFUNC(xgemm)(const char *, + const char *, + const int *, + const int *, + const int *, + const double *, + const double *, + const int *, + const double *, + const int *, + const double *, + double *, + const int *); + +int BLASFUNC( + cgemm3m)(char *, char *, int *, int *, int *, float *, float *, int *, float *, int *, float *, float *, int *); +int BLASFUNC( + zgemm3m)(char *, char *, int *, int *, int *, double *, double *, int *, double *, int *, double *, double *, int *); +int BLASFUNC( + xgemm3m)(char *, char *, int *, int *, int *, double *, double *, int *, double *, int *, double *, double *, int *); + +int BLASFUNC( + sge2mm)(char *, char *, char *, int *, int *, float *, float *, int *, float *, int *, float *, float *, int *); +int BLASFUNC( + dge2mm)(char *, char *, char *, int *, int *, double *, double *, int *, double *, int *, double *, double *, int *); +int BLASFUNC( + cge2mm)(char *, char *, char *, int *, int *, float *, float *, int *, float *, int *, float *, float *, int *); +int BLASFUNC( + zge2mm)(char *, char *, char *, int *, int *, double *, double *, int *, double *, int *, double *, double *, int *); + +int BLASFUNC(strsm)(const char *, + const char *, + const char *, + const char *, + const int *, + const int *, + const float *, + const float *, + const int *, + float *, + const int *); +int BLASFUNC(dtrsm)(const char *, + const char *, + const char *, + const char *, + const int *, + const int *, + const double *, + const double *, + const int *, + double *, + const int *); +int BLASFUNC(qtrsm)(const char *, + const char *, + const char *, + const char *, + const int *, + const int *, + const double *, + const double *, + const int *, + double *, + const int *); +int BLASFUNC(ctrsm)(const char *, + const char *, + const char *, + const char *, + const int *, + const int *, + const float *, + const float *, + const int *, + float *, + const int *); +int BLASFUNC(ztrsm)(const char *, + const char *, + const char *, + const char *, + const int *, + const int *, + const double *, + const double *, + const int *, + double *, + const int *); +int BLASFUNC(xtrsm)(const char *, + const char *, + const char *, + const char *, + const int *, + const int *, + const double *, + const double *, + const int *, + double *, + const int *); + +int BLASFUNC(strmm)(const char *, + const char *, + const char *, + const char *, + const int *, + const int *, + const float *, + const float *, + const int *, + float *, + const int *); +int BLASFUNC(dtrmm)(const char *, + const char *, + const char *, + const char *, + const int *, + const int *, + const double *, + const double *, + const int *, + double *, + const int *); +int BLASFUNC(qtrmm)(const char *, + const char *, + const char *, + const char *, + const int *, + const int *, + const double *, + const double *, + const int *, + double *, + const int *); +int BLASFUNC(ctrmm)(const char *, + const char *, + const char *, + const char *, + const int *, + const int *, + const float *, + const float *, + const int *, + float *, + const int *); +int BLASFUNC(ztrmm)(const char *, + const char *, + const char *, + const char *, + const int *, + const int *, + const double *, + const double *, + const int *, + double *, + const int *); +int BLASFUNC(xtrmm)(const char *, + const char *, + const char *, + const char *, + const int *, + const int *, + const double *, + const double *, + const int *, + double *, + const int *); + +int BLASFUNC(ssymm)(const char *, + const char *, + const int *, + const int *, + const float *, + const float *, + const int *, + const float *, + const int *, + const float *, + float *, + const int *); +int BLASFUNC(dsymm)(const char *, + const char *, + const int *, + const int *, + const double *, + const double *, + const int *, + const double *, + const int *, + const double *, + double *, + const int *); +int BLASFUNC(qsymm)(const char *, + const char *, + const int *, + const int *, + const double *, + const double *, + const int *, + const double *, + const int *, + const double *, + double *, + const int *); +int BLASFUNC(csymm)(const char *, + const char *, + const int *, + const int *, + const float *, + const float *, + const int *, + const float *, + const int *, + const float *, + float *, + const int *); +int BLASFUNC(zsymm)(const char *, + const char *, + const int *, + const int *, + const double *, + const double *, + const int *, + const double *, + const int *, + const double *, + double *, + const int *); +int BLASFUNC(xsymm)(const char *, + const char *, + const int *, + const int *, + const double *, + const double *, + const int *, + const double *, + const int *, + const double *, + double *, + const int *); + +int BLASFUNC(csymm3m)(char *, char *, int *, int *, float *, float *, int *, float *, int *, float *, float *, int *); +int BLASFUNC( + zsymm3m)(char *, char *, int *, int *, double *, double *, int *, double *, int *, double *, double *, int *); +int BLASFUNC( + xsymm3m)(char *, char *, int *, int *, double *, double *, int *, double *, int *, double *, double *, int *); + +int BLASFUNC(ssyrk)(const char *, + const char *, + const int *, + const int *, + const float *, + const float *, + const int *, + const float *, + float *, + const int *); +int BLASFUNC(dsyrk)(const char *, + const char *, + const int *, + const int *, + const double *, + const double *, + const int *, + const double *, + double *, + const int *); +int BLASFUNC(qsyrk)(const char *, + const char *, + const int *, + const int *, + const double *, + const double *, + const int *, + const double *, + double *, + const int *); +int BLASFUNC(csyrk)(const char *, + const char *, + const int *, + const int *, + const float *, + const float *, + const int *, + const float *, + float *, + const int *); +int BLASFUNC(zsyrk)(const char *, + const char *, + const int *, + const int *, + const double *, + const double *, + const int *, + const double *, + double *, + const int *); +int BLASFUNC(xsyrk)(const char *, + const char *, + const int *, + const int *, + const double *, + const double *, + const int *, + const double *, + double *, + const int *); + +int BLASFUNC(ssyr2k)(const char *, + const char *, + const int *, + const int *, + const float *, + const float *, + const int *, + const float *, + const int *, + const float *, + float *, + const int *); +int BLASFUNC(dsyr2k)(const char *, + const char *, + const int *, + const int *, + const double *, + const double *, + const int *, + const double *, + const int *, + const double *, + double *, + const int *); +int BLASFUNC(qsyr2k)(const char *, + const char *, + const int *, + const int *, + const double *, + const double *, + const int *, + const double *, + const int *, + const double *, + double *, + const int *); +int BLASFUNC(csyr2k)(const char *, + const char *, + const int *, + const int *, + const float *, + const float *, + const int *, + const float *, + const int *, + const float *, + float *, + const int *); +int BLASFUNC(zsyr2k)(const char *, + const char *, + const int *, + const int *, + const double *, + const double *, + const int *, + const double *, + const int *, + const double *, + double *, + const int *); +int BLASFUNC(xsyr2k)(const char *, + const char *, + const int *, + const int *, + const double *, + const double *, + const int *, + const double *, + const int *, + const double *, + double *, + const int *); + +int BLASFUNC(chemm)(const char *, + const char *, + const int *, + const int *, + const float *, + const float *, + const int *, + const float *, + const int *, + const float *, + float *, + const int *); +int BLASFUNC(zhemm)(const char *, + const char *, + const int *, + const int *, + const double *, + const double *, + const int *, + const double *, + const int *, + const double *, + double *, + const int *); +int BLASFUNC(xhemm)(const char *, + const char *, + const int *, + const int *, + const double *, + const double *, + const int *, + const double *, + const int *, + const double *, + double *, + const int *); + +int BLASFUNC(chemm3m)(char *, char *, int *, int *, float *, float *, int *, float *, int *, float *, float *, int *); +int BLASFUNC( + zhemm3m)(char *, char *, int *, int *, double *, double *, int *, double *, int *, double *, double *, int *); +int BLASFUNC( + xhemm3m)(char *, char *, int *, int *, double *, double *, int *, double *, int *, double *, double *, int *); + +int BLASFUNC(cherk)(const char *, + const char *, + const int *, + const int *, + const float *, + const float *, + const int *, + const float *, + float *, + const int *); +int BLASFUNC(zherk)(const char *, + const char *, + const int *, + const int *, + const double *, + const double *, + const int *, + const double *, + double *, + const int *); +int BLASFUNC(xherk)(const char *, + const char *, + const int *, + const int *, + const double *, + const double *, + const int *, + const double *, + double *, + const int *); + +int BLASFUNC(cher2k)(const char *, + const char *, + const int *, + const int *, + const float *, + const float *, + const int *, + const float *, + const int *, + const float *, + float *, + const int *); +int BLASFUNC(zher2k)(const char *, + const char *, + const int *, + const int *, + const double *, + const double *, + const int *, + const double *, + const int *, + const double *, + double *, + const int *); +int BLASFUNC(xher2k)(const char *, + const char *, + const int *, + const int *, + const double *, + const double *, + const int *, + const double *, + const int *, + const double *, + double *, + const int *); +int BLASFUNC(cher2m)(const char *, + const char *, + const char *, + const int *, + const int *, + const float *, + const float *, + const int *, + const float *, + const int *, + const float *, + float *, + const int *); +int BLASFUNC(zher2m)(const char *, + const char *, + const char *, + const int *, + const int *, + const double *, + const double *, + const int *, + const double *, + const int *, + const double *, + double *, + const int *); +int BLASFUNC(xher2m)(const char *, + const char *, + const char *, + const int *, + const int *, + const double *, + const double *, + const int *, + const double *, + const int *, + const double *, + double *, + const int *); #ifdef __cplusplus diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/misc/lapack.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/misc/lapack.h index 249f3575..88ff58c7 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/misc/lapack.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/misc/lapack.h @@ -4,144 +4,153 @@ #include "blas.h" #ifdef __cplusplus -extern "C" -{ +extern "C" { #endif -int BLASFUNC(csymv) (const char *, const int *, const float *, const float *, const int *, const float *, const int *, const float *, float *, const int *); -int BLASFUNC(zsymv) (const char *, const int *, const double *, const double *, const int *, const double *, const int *, const double *, double *, const int *); -int BLASFUNC(xsymv) (const char *, const int *, const double *, const double *, const int *, const double *, const int *, const double *, double *, const int *); - - -int BLASFUNC(cspmv) (char *, int *, float *, float *, - float *, int *, float *, float *, int *); -int BLASFUNC(zspmv) (char *, int *, double *, double *, - double *, int *, double *, double *, int *); -int BLASFUNC(xspmv) (char *, int *, double *, double *, - double *, int *, double *, double *, int *); - -int BLASFUNC(csyr) (char *, int *, float *, float *, int *, - float *, int *); -int BLASFUNC(zsyr) (char *, int *, double *, double *, int *, - double *, int *); -int BLASFUNC(xsyr) (char *, int *, double *, double *, int *, - double *, int *); - -int BLASFUNC(cspr) (char *, int *, float *, float *, int *, - float *); -int BLASFUNC(zspr) (char *, int *, double *, double *, int *, - double *); -int BLASFUNC(xspr) (char *, int *, double *, double *, int *, - double *); - -int BLASFUNC(sgemt)(char *, int *, int *, float *, float *, int *, - float *, int *); -int BLASFUNC(dgemt)(char *, int *, int *, double *, double *, int *, - double *, int *); -int BLASFUNC(cgemt)(char *, int *, int *, float *, float *, int *, - float *, int *); -int BLASFUNC(zgemt)(char *, int *, int *, double *, double *, int *, - double *, int *); - -int BLASFUNC(sgema)(char *, char *, int *, int *, float *, - float *, int *, float *, float *, int *, float *, int *); -int BLASFUNC(dgema)(char *, char *, int *, int *, double *, - double *, int *, double*, double *, int *, double*, int *); -int BLASFUNC(cgema)(char *, char *, int *, int *, float *, - float *, int *, float *, float *, int *, float *, int *); -int BLASFUNC(zgema)(char *, char *, int *, int *, double *, - double *, int *, double*, double *, int *, double*, int *); - -int BLASFUNC(sgems)(char *, char *, int *, int *, float *, - float *, int *, float *, float *, int *, float *, int *); -int BLASFUNC(dgems)(char *, char *, int *, int *, double *, - double *, int *, double*, double *, int *, double*, int *); -int BLASFUNC(cgems)(char *, char *, int *, int *, float *, - float *, int *, float *, float *, int *, float *, int *); -int BLASFUNC(zgems)(char *, char *, int *, int *, double *, - double *, int *, double*, double *, int *, double*, int *); - -int BLASFUNC(sgetf2)(int *, int *, float *, int *, int *, int *); +int BLASFUNC(csymv)(const char *, + const int *, + const float *, + const float *, + const int *, + const float *, + const int *, + const float *, + float *, + const int *); +int BLASFUNC(zsymv)(const char *, + const int *, + const double *, + const double *, + const int *, + const double *, + const int *, + const double *, + double *, + const int *); +int BLASFUNC(xsymv)(const char *, + const int *, + const double *, + const double *, + const int *, + const double *, + const int *, + const double *, + double *, + const int *); + + +int BLASFUNC(cspmv)(char *, int *, float *, float *, float *, int *, float *, float *, int *); +int BLASFUNC(zspmv)(char *, int *, double *, double *, double *, int *, double *, double *, int *); +int BLASFUNC(xspmv)(char *, int *, double *, double *, double *, int *, double *, double *, int *); + +int BLASFUNC(csyr)(char *, int *, float *, float *, int *, float *, int *); +int BLASFUNC(zsyr)(char *, int *, double *, double *, int *, double *, int *); +int BLASFUNC(xsyr)(char *, int *, double *, double *, int *, double *, int *); + +int BLASFUNC(cspr)(char *, int *, float *, float *, int *, float *); +int BLASFUNC(zspr)(char *, int *, double *, double *, int *, double *); +int BLASFUNC(xspr)(char *, int *, double *, double *, int *, double *); + +int BLASFUNC(sgemt)(char *, int *, int *, float *, float *, int *, float *, int *); +int BLASFUNC(dgemt)(char *, int *, int *, double *, double *, int *, double *, int *); +int BLASFUNC(cgemt)(char *, int *, int *, float *, float *, int *, float *, int *); +int BLASFUNC(zgemt)(char *, int *, int *, double *, double *, int *, double *, int *); + +int BLASFUNC(sgema)(char *, char *, int *, int *, float *, float *, int *, float *, float *, int *, float *, int *); +int BLASFUNC( + dgema)(char *, char *, int *, int *, double *, double *, int *, double *, double *, int *, double *, int *); +int BLASFUNC(cgema)(char *, char *, int *, int *, float *, float *, int *, float *, float *, int *, float *, int *); +int BLASFUNC( + zgema)(char *, char *, int *, int *, double *, double *, int *, double *, double *, int *, double *, int *); + +int BLASFUNC(sgems)(char *, char *, int *, int *, float *, float *, int *, float *, float *, int *, float *, int *); +int BLASFUNC( + dgems)(char *, char *, int *, int *, double *, double *, int *, double *, double *, int *, double *, int *); +int BLASFUNC(cgems)(char *, char *, int *, int *, float *, float *, int *, float *, float *, int *, float *, int *); +int BLASFUNC( + zgems)(char *, char *, int *, int *, double *, double *, int *, double *, double *, int *, double *, int *); + +int BLASFUNC(sgetf2)(int *, int *, float *, int *, int *, int *); int BLASFUNC(dgetf2)(int *, int *, double *, int *, int *, int *); int BLASFUNC(qgetf2)(int *, int *, double *, int *, int *, int *); -int BLASFUNC(cgetf2)(int *, int *, float *, int *, int *, int *); +int BLASFUNC(cgetf2)(int *, int *, float *, int *, int *, int *); int BLASFUNC(zgetf2)(int *, int *, double *, int *, int *, int *); int BLASFUNC(xgetf2)(int *, int *, double *, int *, int *, int *); -int BLASFUNC(sgetrf)(int *, int *, float *, int *, int *, int *); +int BLASFUNC(sgetrf)(int *, int *, float *, int *, int *, int *); int BLASFUNC(dgetrf)(int *, int *, double *, int *, int *, int *); int BLASFUNC(qgetrf)(int *, int *, double *, int *, int *, int *); -int BLASFUNC(cgetrf)(int *, int *, float *, int *, int *, int *); +int BLASFUNC(cgetrf)(int *, int *, float *, int *, int *, int *); int BLASFUNC(zgetrf)(int *, int *, double *, int *, int *, int *); int BLASFUNC(xgetrf)(int *, int *, double *, int *, int *, int *); -int BLASFUNC(slaswp)(int *, float *, int *, int *, int *, int *, int *); +int BLASFUNC(slaswp)(int *, float *, int *, int *, int *, int *, int *); int BLASFUNC(dlaswp)(int *, double *, int *, int *, int *, int *, int *); int BLASFUNC(qlaswp)(int *, double *, int *, int *, int *, int *, int *); -int BLASFUNC(claswp)(int *, float *, int *, int *, int *, int *, int *); +int BLASFUNC(claswp)(int *, float *, int *, int *, int *, int *, int *); int BLASFUNC(zlaswp)(int *, double *, int *, int *, int *, int *, int *); int BLASFUNC(xlaswp)(int *, double *, int *, int *, int *, int *, int *); -int BLASFUNC(sgetrs)(char *, int *, int *, float *, int *, int *, float *, int *, int *); +int BLASFUNC(sgetrs)(char *, int *, int *, float *, int *, int *, float *, int *, int *); int BLASFUNC(dgetrs)(char *, int *, int *, double *, int *, int *, double *, int *, int *); int BLASFUNC(qgetrs)(char *, int *, int *, double *, int *, int *, double *, int *, int *); -int BLASFUNC(cgetrs)(char *, int *, int *, float *, int *, int *, float *, int *, int *); +int BLASFUNC(cgetrs)(char *, int *, int *, float *, int *, int *, float *, int *, int *); int BLASFUNC(zgetrs)(char *, int *, int *, double *, int *, int *, double *, int *, int *); int BLASFUNC(xgetrs)(char *, int *, int *, double *, int *, int *, double *, int *, int *); -int BLASFUNC(sgesv)(int *, int *, float *, int *, int *, float *, int *, int *); -int BLASFUNC(dgesv)(int *, int *, double *, int *, int *, double*, int *, int *); -int BLASFUNC(qgesv)(int *, int *, double *, int *, int *, double*, int *, int *); -int BLASFUNC(cgesv)(int *, int *, float *, int *, int *, float *, int *, int *); -int BLASFUNC(zgesv)(int *, int *, double *, int *, int *, double*, int *, int *); -int BLASFUNC(xgesv)(int *, int *, double *, int *, int *, double*, int *, int *); +int BLASFUNC(sgesv)(int *, int *, float *, int *, int *, float *, int *, int *); +int BLASFUNC(dgesv)(int *, int *, double *, int *, int *, double *, int *, int *); +int BLASFUNC(qgesv)(int *, int *, double *, int *, int *, double *, int *, int *); +int BLASFUNC(cgesv)(int *, int *, float *, int *, int *, float *, int *, int *); +int BLASFUNC(zgesv)(int *, int *, double *, int *, int *, double *, int *, int *); +int BLASFUNC(xgesv)(int *, int *, double *, int *, int *, double *, int *, int *); -int BLASFUNC(spotf2)(char *, int *, float *, int *, int *); +int BLASFUNC(spotf2)(char *, int *, float *, int *, int *); int BLASFUNC(dpotf2)(char *, int *, double *, int *, int *); int BLASFUNC(qpotf2)(char *, int *, double *, int *, int *); -int BLASFUNC(cpotf2)(char *, int *, float *, int *, int *); +int BLASFUNC(cpotf2)(char *, int *, float *, int *, int *); int BLASFUNC(zpotf2)(char *, int *, double *, int *, int *); int BLASFUNC(xpotf2)(char *, int *, double *, int *, int *); -int BLASFUNC(spotrf)(char *, int *, float *, int *, int *); +int BLASFUNC(spotrf)(char *, int *, float *, int *, int *); int BLASFUNC(dpotrf)(char *, int *, double *, int *, int *); int BLASFUNC(qpotrf)(char *, int *, double *, int *, int *); -int BLASFUNC(cpotrf)(char *, int *, float *, int *, int *); +int BLASFUNC(cpotrf)(char *, int *, float *, int *, int *); int BLASFUNC(zpotrf)(char *, int *, double *, int *, int *); int BLASFUNC(xpotrf)(char *, int *, double *, int *, int *); -int BLASFUNC(slauu2)(char *, int *, float *, int *, int *); +int BLASFUNC(slauu2)(char *, int *, float *, int *, int *); int BLASFUNC(dlauu2)(char *, int *, double *, int *, int *); int BLASFUNC(qlauu2)(char *, int *, double *, int *, int *); -int BLASFUNC(clauu2)(char *, int *, float *, int *, int *); +int BLASFUNC(clauu2)(char *, int *, float *, int *, int *); int BLASFUNC(zlauu2)(char *, int *, double *, int *, int *); int BLASFUNC(xlauu2)(char *, int *, double *, int *, int *); -int BLASFUNC(slauum)(char *, int *, float *, int *, int *); +int BLASFUNC(slauum)(char *, int *, float *, int *, int *); int BLASFUNC(dlauum)(char *, int *, double *, int *, int *); int BLASFUNC(qlauum)(char *, int *, double *, int *, int *); -int BLASFUNC(clauum)(char *, int *, float *, int *, int *); +int BLASFUNC(clauum)(char *, int *, float *, int *, int *); int BLASFUNC(zlauum)(char *, int *, double *, int *, int *); int BLASFUNC(xlauum)(char *, int *, double *, int *, int *); -int BLASFUNC(strti2)(char *, char *, int *, float *, int *, int *); +int BLASFUNC(strti2)(char *, char *, int *, float *, int *, int *); int BLASFUNC(dtrti2)(char *, char *, int *, double *, int *, int *); int BLASFUNC(qtrti2)(char *, char *, int *, double *, int *, int *); -int BLASFUNC(ctrti2)(char *, char *, int *, float *, int *, int *); +int BLASFUNC(ctrti2)(char *, char *, int *, float *, int *, int *); int BLASFUNC(ztrti2)(char *, char *, int *, double *, int *, int *); int BLASFUNC(xtrti2)(char *, char *, int *, double *, int *, int *); -int BLASFUNC(strtri)(char *, char *, int *, float *, int *, int *); +int BLASFUNC(strtri)(char *, char *, int *, float *, int *, int *); int BLASFUNC(dtrtri)(char *, char *, int *, double *, int *, int *); int BLASFUNC(qtrtri)(char *, char *, int *, double *, int *, int *); -int BLASFUNC(ctrtri)(char *, char *, int *, float *, int *, int *); +int BLASFUNC(ctrtri)(char *, char *, int *, float *, int *, int *); int BLASFUNC(ztrtri)(char *, char *, int *, double *, int *, int *); int BLASFUNC(xtrtri)(char *, char *, int *, double *, int *, int *); -int BLASFUNC(spotri)(char *, int *, float *, int *, int *); +int BLASFUNC(spotri)(char *, int *, float *, int *, int *); int BLASFUNC(dpotri)(char *, int *, double *, int *, int *); int BLASFUNC(qpotri)(char *, int *, double *, int *, int *); -int BLASFUNC(cpotri)(char *, int *, float *, int *, int *); +int BLASFUNC(cpotri)(char *, int *, float *, int *, int *); int BLASFUNC(zpotri)(char *, int *, double *, int *, int *); int BLASFUNC(xpotri)(char *, int *, double *, int *, int *); diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/misc/lapacke.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/misc/lapacke.h old mode 100755 new mode 100644 index 8c7e79b0..1208d370 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/misc/lapacke.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/misc/lapacke.h @@ -37,8 +37,8 @@ #define _LAPACKE_H_ /* -* Turn on HAVE_LAPACK_CONFIG_H to redefine C-LAPACK datatypes -*/ + * Turn on HAVE_LAPACK_CONFIG_H to redefine C-LAPACK datatypes + */ #ifdef HAVE_LAPACK_CONFIG_H #include "lapacke_config.h" #endif @@ -50,7 +50,7 @@ extern "C" { #include #ifndef lapack_int -#define lapack_int int +#define lapack_int int #endif #ifndef lapack_logical @@ -58,16229 +58,36331 @@ extern "C" { #endif /* Complex types are structures equivalent to the -* Fortran complex types COMPLEX(4) and COMPLEX(8). -* -* One can also redefine the types with his own types -* for example by including in the code definitions like -* -* #define lapack_complex_float std::complex -* #define lapack_complex_double std::complex -* -* or define these types in the command line: -* -* -Dlapack_complex_float="std::complex" -* -Dlapack_complex_double="std::complex" -*/ + * Fortran complex types COMPLEX(4) and COMPLEX(8). + * + * One can also redefine the types with his own types + * for example by including in the code definitions like + * + * #define lapack_complex_float std::complex + * #define lapack_complex_double std::complex + * + * or define these types in the command line: + * + * -Dlapack_complex_float="std::complex" + * -Dlapack_complex_double="std::complex" + */ #ifndef LAPACK_COMPLEX_CUSTOM /* Complex type (single precision) */ #ifndef lapack_complex_float #include -#define lapack_complex_float float _Complex +#define lapack_complex_float float _Complex #endif #ifndef lapack_complex_float_real -#define lapack_complex_float_real(z) (creal(z)) +#define lapack_complex_float_real(z) (creal(z)) #endif #ifndef lapack_complex_float_imag -#define lapack_complex_float_imag(z) (cimag(z)) +#define lapack_complex_float_imag(z) (cimag(z)) #endif -lapack_complex_float lapack_make_complex_float( float re, float im ); +lapack_complex_float lapack_make_complex_float(float re, float im); /* Complex type (double precision) */ #ifndef lapack_complex_double #include -#define lapack_complex_double double _Complex +#define lapack_complex_double double _Complex #endif #ifndef lapack_complex_double_real -#define lapack_complex_double_real(z) (creal(z)) +#define lapack_complex_double_real(z) (creal(z)) #endif #ifndef lapack_complex_double_imag -#define lapack_complex_double_imag(z) (cimag(z)) +#define lapack_complex_double_imag(z) (cimag(z)) #endif -lapack_complex_double lapack_make_complex_double( double re, double im ); +lapack_complex_double lapack_make_complex_double(double re, double im); #endif #ifndef LAPACKE_malloc -#define LAPACKE_malloc( size ) malloc( size ) +#define LAPACKE_malloc(size) malloc(size) #endif #ifndef LAPACKE_free -#define LAPACKE_free( p ) free( p ) +#define LAPACKE_free(p) free(p) #endif -#define LAPACK_C2INT( x ) (lapack_int)(*((float*)&x )) -#define LAPACK_Z2INT( x ) (lapack_int)(*((double*)&x )) +#define LAPACK_C2INT(x) (lapack_int)(*((float *)&x)) +#define LAPACK_Z2INT(x) (lapack_int)(*((double *)&x)) -#define LAPACK_ROW_MAJOR 101 -#define LAPACK_COL_MAJOR 102 +#define LAPACK_ROW_MAJOR 101 +#define LAPACK_COL_MAJOR 102 -#define LAPACK_WORK_MEMORY_ERROR -1010 -#define LAPACK_TRANSPOSE_MEMORY_ERROR -1011 +#define LAPACK_WORK_MEMORY_ERROR -1010 +#define LAPACK_TRANSPOSE_MEMORY_ERROR -1011 /* Callback logical functions of one, two, or three arguments are used -* to select eigenvalues to sort to the top left of the Schur form. -* The value is selected if function returns TRUE (non-zero). */ - -typedef lapack_logical (*LAPACK_S_SELECT2) ( const float*, const float* ); -typedef lapack_logical (*LAPACK_S_SELECT3) - ( const float*, const float*, const float* ); -typedef lapack_logical (*LAPACK_D_SELECT2) ( const double*, const double* ); -typedef lapack_logical (*LAPACK_D_SELECT3) - ( const double*, const double*, const double* ); - -typedef lapack_logical (*LAPACK_C_SELECT1) ( const lapack_complex_float* ); -typedef lapack_logical (*LAPACK_C_SELECT2) - ( const lapack_complex_float*, const lapack_complex_float* ); -typedef lapack_logical (*LAPACK_Z_SELECT1) ( const lapack_complex_double* ); -typedef lapack_logical (*LAPACK_Z_SELECT2) - ( const lapack_complex_double*, const lapack_complex_double* ); + * to select eigenvalues to sort to the top left of the Schur form. + * The value is selected if function returns TRUE (non-zero). */ + +typedef lapack_logical (*LAPACK_S_SELECT2)(const float *, const float *); +typedef lapack_logical (*LAPACK_S_SELECT3)(const float *, const float *, const float *); +typedef lapack_logical (*LAPACK_D_SELECT2)(const double *, const double *); +typedef lapack_logical (*LAPACK_D_SELECT3)(const double *, const double *, const double *); + +typedef lapack_logical (*LAPACK_C_SELECT1)(const lapack_complex_float *); +typedef lapack_logical (*LAPACK_C_SELECT2)(const lapack_complex_float *, const lapack_complex_float *); +typedef lapack_logical (*LAPACK_Z_SELECT1)(const lapack_complex_double *); +typedef lapack_logical (*LAPACK_Z_SELECT2)(const lapack_complex_double *, const lapack_complex_double *); #include "lapacke_mangling.h" -#define LAPACK_lsame LAPACK_GLOBAL(lsame,LSAME) -lapack_logical LAPACK_lsame( char* ca, char* cb, - lapack_int lca, lapack_int lcb ); +#define LAPACK_lsame LAPACK_GLOBAL(lsame, LSAME) +lapack_logical LAPACK_lsame(char *ca, char *cb, lapack_int lca, lapack_int lcb); /* C-LAPACK function prototypes */ -lapack_int LAPACKE_sbdsdc( int matrix_order, char uplo, char compq, - lapack_int n, float* d, float* e, float* u, - lapack_int ldu, float* vt, lapack_int ldvt, float* q, - lapack_int* iq ); -lapack_int LAPACKE_dbdsdc( int matrix_order, char uplo, char compq, - lapack_int n, double* d, double* e, double* u, - lapack_int ldu, double* vt, lapack_int ldvt, - double* q, lapack_int* iq ); - -lapack_int LAPACKE_sbdsqr( int matrix_order, char uplo, lapack_int n, - lapack_int ncvt, lapack_int nru, lapack_int ncc, - float* d, float* e, float* vt, lapack_int ldvt, - float* u, lapack_int ldu, float* c, lapack_int ldc ); -lapack_int LAPACKE_dbdsqr( int matrix_order, char uplo, lapack_int n, - lapack_int ncvt, lapack_int nru, lapack_int ncc, - double* d, double* e, double* vt, lapack_int ldvt, - double* u, lapack_int ldu, double* c, - lapack_int ldc ); -lapack_int LAPACKE_cbdsqr( int matrix_order, char uplo, lapack_int n, - lapack_int ncvt, lapack_int nru, lapack_int ncc, - float* d, float* e, lapack_complex_float* vt, - lapack_int ldvt, lapack_complex_float* u, - lapack_int ldu, lapack_complex_float* c, - lapack_int ldc ); -lapack_int LAPACKE_zbdsqr( int matrix_order, char uplo, lapack_int n, - lapack_int ncvt, lapack_int nru, lapack_int ncc, - double* d, double* e, lapack_complex_double* vt, - lapack_int ldvt, lapack_complex_double* u, - lapack_int ldu, lapack_complex_double* c, - lapack_int ldc ); - -lapack_int LAPACKE_sdisna( char job, lapack_int m, lapack_int n, const float* d, - float* sep ); -lapack_int LAPACKE_ddisna( char job, lapack_int m, lapack_int n, - const double* d, double* sep ); - -lapack_int LAPACKE_sgbbrd( int matrix_order, char vect, lapack_int m, - lapack_int n, lapack_int ncc, lapack_int kl, - lapack_int ku, float* ab, lapack_int ldab, float* d, - float* e, float* q, lapack_int ldq, float* pt, - lapack_int ldpt, float* c, lapack_int ldc ); -lapack_int LAPACKE_dgbbrd( int matrix_order, char vect, lapack_int m, - lapack_int n, lapack_int ncc, lapack_int kl, - lapack_int ku, double* ab, lapack_int ldab, - double* d, double* e, double* q, lapack_int ldq, - double* pt, lapack_int ldpt, double* c, - lapack_int ldc ); -lapack_int LAPACKE_cgbbrd( int matrix_order, char vect, lapack_int m, - lapack_int n, lapack_int ncc, lapack_int kl, - lapack_int ku, lapack_complex_float* ab, - lapack_int ldab, float* d, float* e, - lapack_complex_float* q, lapack_int ldq, - lapack_complex_float* pt, lapack_int ldpt, - lapack_complex_float* c, lapack_int ldc ); -lapack_int LAPACKE_zgbbrd( int matrix_order, char vect, lapack_int m, - lapack_int n, lapack_int ncc, lapack_int kl, - lapack_int ku, lapack_complex_double* ab, - lapack_int ldab, double* d, double* e, - lapack_complex_double* q, lapack_int ldq, - lapack_complex_double* pt, lapack_int ldpt, - lapack_complex_double* c, lapack_int ldc ); - -lapack_int LAPACKE_sgbcon( int matrix_order, char norm, lapack_int n, - lapack_int kl, lapack_int ku, const float* ab, - lapack_int ldab, const lapack_int* ipiv, float anorm, - float* rcond ); -lapack_int LAPACKE_dgbcon( int matrix_order, char norm, lapack_int n, - lapack_int kl, lapack_int ku, const double* ab, - lapack_int ldab, const lapack_int* ipiv, - double anorm, double* rcond ); -lapack_int LAPACKE_cgbcon( int matrix_order, char norm, lapack_int n, - lapack_int kl, lapack_int ku, - const lapack_complex_float* ab, lapack_int ldab, - const lapack_int* ipiv, float anorm, float* rcond ); -lapack_int LAPACKE_zgbcon( int matrix_order, char norm, lapack_int n, - lapack_int kl, lapack_int ku, - const lapack_complex_double* ab, lapack_int ldab, - const lapack_int* ipiv, double anorm, - double* rcond ); - -lapack_int LAPACKE_sgbequ( int matrix_order, lapack_int m, lapack_int n, - lapack_int kl, lapack_int ku, const float* ab, - lapack_int ldab, float* r, float* c, float* rowcnd, - float* colcnd, float* amax ); -lapack_int LAPACKE_dgbequ( int matrix_order, lapack_int m, lapack_int n, - lapack_int kl, lapack_int ku, const double* ab, - lapack_int ldab, double* r, double* c, - double* rowcnd, double* colcnd, double* amax ); -lapack_int LAPACKE_cgbequ( int matrix_order, lapack_int m, lapack_int n, - lapack_int kl, lapack_int ku, - const lapack_complex_float* ab, lapack_int ldab, - float* r, float* c, float* rowcnd, float* colcnd, - float* amax ); -lapack_int LAPACKE_zgbequ( int matrix_order, lapack_int m, lapack_int n, - lapack_int kl, lapack_int ku, - const lapack_complex_double* ab, lapack_int ldab, - double* r, double* c, double* rowcnd, double* colcnd, - double* amax ); - -lapack_int LAPACKE_sgbequb( int matrix_order, lapack_int m, lapack_int n, - lapack_int kl, lapack_int ku, const float* ab, - lapack_int ldab, float* r, float* c, float* rowcnd, - float* colcnd, float* amax ); -lapack_int LAPACKE_dgbequb( int matrix_order, lapack_int m, lapack_int n, - lapack_int kl, lapack_int ku, const double* ab, - lapack_int ldab, double* r, double* c, - double* rowcnd, double* colcnd, double* amax ); -lapack_int LAPACKE_cgbequb( int matrix_order, lapack_int m, lapack_int n, - lapack_int kl, lapack_int ku, - const lapack_complex_float* ab, lapack_int ldab, - float* r, float* c, float* rowcnd, float* colcnd, - float* amax ); -lapack_int LAPACKE_zgbequb( int matrix_order, lapack_int m, lapack_int n, - lapack_int kl, lapack_int ku, - const lapack_complex_double* ab, lapack_int ldab, - double* r, double* c, double* rowcnd, - double* colcnd, double* amax ); - -lapack_int LAPACKE_sgbrfs( int matrix_order, char trans, lapack_int n, - lapack_int kl, lapack_int ku, lapack_int nrhs, - const float* ab, lapack_int ldab, const float* afb, - lapack_int ldafb, const lapack_int* ipiv, - const float* b, lapack_int ldb, float* x, - lapack_int ldx, float* ferr, float* berr ); -lapack_int LAPACKE_dgbrfs( int matrix_order, char trans, lapack_int n, - lapack_int kl, lapack_int ku, lapack_int nrhs, - const double* ab, lapack_int ldab, const double* afb, - lapack_int ldafb, const lapack_int* ipiv, - const double* b, lapack_int ldb, double* x, - lapack_int ldx, double* ferr, double* berr ); -lapack_int LAPACKE_cgbrfs( int matrix_order, char trans, lapack_int n, - lapack_int kl, lapack_int ku, lapack_int nrhs, - const lapack_complex_float* ab, lapack_int ldab, - const lapack_complex_float* afb, lapack_int ldafb, - const lapack_int* ipiv, - const lapack_complex_float* b, lapack_int ldb, - lapack_complex_float* x, lapack_int ldx, float* ferr, - float* berr ); -lapack_int LAPACKE_zgbrfs( int matrix_order, char trans, lapack_int n, - lapack_int kl, lapack_int ku, lapack_int nrhs, - const lapack_complex_double* ab, lapack_int ldab, - const lapack_complex_double* afb, lapack_int ldafb, - const lapack_int* ipiv, - const lapack_complex_double* b, lapack_int ldb, - lapack_complex_double* x, lapack_int ldx, - double* ferr, double* berr ); - -lapack_int LAPACKE_sgbrfsx( int matrix_order, char trans, char equed, - lapack_int n, lapack_int kl, lapack_int ku, - lapack_int nrhs, const float* ab, lapack_int ldab, - const float* afb, lapack_int ldafb, - const lapack_int* ipiv, const float* r, - const float* c, const float* b, lapack_int ldb, - float* x, lapack_int ldx, float* rcond, float* berr, - lapack_int n_err_bnds, float* err_bnds_norm, - float* err_bnds_comp, lapack_int nparams, - float* params ); -lapack_int LAPACKE_dgbrfsx( int matrix_order, char trans, char equed, - lapack_int n, lapack_int kl, lapack_int ku, - lapack_int nrhs, const double* ab, lapack_int ldab, - const double* afb, lapack_int ldafb, - const lapack_int* ipiv, const double* r, - const double* c, const double* b, lapack_int ldb, - double* x, lapack_int ldx, double* rcond, - double* berr, lapack_int n_err_bnds, - double* err_bnds_norm, double* err_bnds_comp, - lapack_int nparams, double* params ); -lapack_int LAPACKE_cgbrfsx( int matrix_order, char trans, char equed, - lapack_int n, lapack_int kl, lapack_int ku, - lapack_int nrhs, const lapack_complex_float* ab, - lapack_int ldab, const lapack_complex_float* afb, - lapack_int ldafb, const lapack_int* ipiv, - const float* r, const float* c, - const lapack_complex_float* b, lapack_int ldb, - lapack_complex_float* x, lapack_int ldx, - float* rcond, float* berr, lapack_int n_err_bnds, - float* err_bnds_norm, float* err_bnds_comp, - lapack_int nparams, float* params ); -lapack_int LAPACKE_zgbrfsx( int matrix_order, char trans, char equed, - lapack_int n, lapack_int kl, lapack_int ku, - lapack_int nrhs, const lapack_complex_double* ab, - lapack_int ldab, const lapack_complex_double* afb, - lapack_int ldafb, const lapack_int* ipiv, - const double* r, const double* c, - const lapack_complex_double* b, lapack_int ldb, - lapack_complex_double* x, lapack_int ldx, - double* rcond, double* berr, lapack_int n_err_bnds, - double* err_bnds_norm, double* err_bnds_comp, - lapack_int nparams, double* params ); - -lapack_int LAPACKE_sgbsv( int matrix_order, lapack_int n, lapack_int kl, - lapack_int ku, lapack_int nrhs, float* ab, - lapack_int ldab, lapack_int* ipiv, float* b, - lapack_int ldb ); -lapack_int LAPACKE_dgbsv( int matrix_order, lapack_int n, lapack_int kl, - lapack_int ku, lapack_int nrhs, double* ab, - lapack_int ldab, lapack_int* ipiv, double* b, - lapack_int ldb ); -lapack_int LAPACKE_cgbsv( int matrix_order, lapack_int n, lapack_int kl, - lapack_int ku, lapack_int nrhs, - lapack_complex_float* ab, lapack_int ldab, - lapack_int* ipiv, lapack_complex_float* b, - lapack_int ldb ); -lapack_int LAPACKE_zgbsv( int matrix_order, lapack_int n, lapack_int kl, - lapack_int ku, lapack_int nrhs, - lapack_complex_double* ab, lapack_int ldab, - lapack_int* ipiv, lapack_complex_double* b, - lapack_int ldb ); - -lapack_int LAPACKE_sgbsvx( int matrix_order, char fact, char trans, - lapack_int n, lapack_int kl, lapack_int ku, - lapack_int nrhs, float* ab, lapack_int ldab, - float* afb, lapack_int ldafb, lapack_int* ipiv, - char* equed, float* r, float* c, float* b, - lapack_int ldb, float* x, lapack_int ldx, - float* rcond, float* ferr, float* berr, - float* rpivot ); -lapack_int LAPACKE_dgbsvx( int matrix_order, char fact, char trans, - lapack_int n, lapack_int kl, lapack_int ku, - lapack_int nrhs, double* ab, lapack_int ldab, - double* afb, lapack_int ldafb, lapack_int* ipiv, - char* equed, double* r, double* c, double* b, - lapack_int ldb, double* x, lapack_int ldx, - double* rcond, double* ferr, double* berr, - double* rpivot ); -lapack_int LAPACKE_cgbsvx( int matrix_order, char fact, char trans, - lapack_int n, lapack_int kl, lapack_int ku, - lapack_int nrhs, lapack_complex_float* ab, - lapack_int ldab, lapack_complex_float* afb, - lapack_int ldafb, lapack_int* ipiv, char* equed, - float* r, float* c, lapack_complex_float* b, - lapack_int ldb, lapack_complex_float* x, - lapack_int ldx, float* rcond, float* ferr, - float* berr, float* rpivot ); -lapack_int LAPACKE_zgbsvx( int matrix_order, char fact, char trans, - lapack_int n, lapack_int kl, lapack_int ku, - lapack_int nrhs, lapack_complex_double* ab, - lapack_int ldab, lapack_complex_double* afb, - lapack_int ldafb, lapack_int* ipiv, char* equed, - double* r, double* c, lapack_complex_double* b, - lapack_int ldb, lapack_complex_double* x, - lapack_int ldx, double* rcond, double* ferr, - double* berr, double* rpivot ); - -lapack_int LAPACKE_sgbsvxx( int matrix_order, char fact, char trans, - lapack_int n, lapack_int kl, lapack_int ku, - lapack_int nrhs, float* ab, lapack_int ldab, - float* afb, lapack_int ldafb, lapack_int* ipiv, - char* equed, float* r, float* c, float* b, - lapack_int ldb, float* x, lapack_int ldx, - float* rcond, float* rpvgrw, float* berr, - lapack_int n_err_bnds, float* err_bnds_norm, - float* err_bnds_comp, lapack_int nparams, - float* params ); -lapack_int LAPACKE_dgbsvxx( int matrix_order, char fact, char trans, - lapack_int n, lapack_int kl, lapack_int ku, - lapack_int nrhs, double* ab, lapack_int ldab, - double* afb, lapack_int ldafb, lapack_int* ipiv, - char* equed, double* r, double* c, double* b, - lapack_int ldb, double* x, lapack_int ldx, - double* rcond, double* rpvgrw, double* berr, - lapack_int n_err_bnds, double* err_bnds_norm, - double* err_bnds_comp, lapack_int nparams, - double* params ); -lapack_int LAPACKE_cgbsvxx( int matrix_order, char fact, char trans, - lapack_int n, lapack_int kl, lapack_int ku, - lapack_int nrhs, lapack_complex_float* ab, - lapack_int ldab, lapack_complex_float* afb, - lapack_int ldafb, lapack_int* ipiv, char* equed, - float* r, float* c, lapack_complex_float* b, - lapack_int ldb, lapack_complex_float* x, - lapack_int ldx, float* rcond, float* rpvgrw, - float* berr, lapack_int n_err_bnds, - float* err_bnds_norm, float* err_bnds_comp, - lapack_int nparams, float* params ); -lapack_int LAPACKE_zgbsvxx( int matrix_order, char fact, char trans, - lapack_int n, lapack_int kl, lapack_int ku, - lapack_int nrhs, lapack_complex_double* ab, - lapack_int ldab, lapack_complex_double* afb, - lapack_int ldafb, lapack_int* ipiv, char* equed, - double* r, double* c, lapack_complex_double* b, - lapack_int ldb, lapack_complex_double* x, - lapack_int ldx, double* rcond, double* rpvgrw, - double* berr, lapack_int n_err_bnds, - double* err_bnds_norm, double* err_bnds_comp, - lapack_int nparams, double* params ); - -lapack_int LAPACKE_sgbtrf( int matrix_order, lapack_int m, lapack_int n, - lapack_int kl, lapack_int ku, float* ab, - lapack_int ldab, lapack_int* ipiv ); -lapack_int LAPACKE_dgbtrf( int matrix_order, lapack_int m, lapack_int n, - lapack_int kl, lapack_int ku, double* ab, - lapack_int ldab, lapack_int* ipiv ); -lapack_int LAPACKE_cgbtrf( int matrix_order, lapack_int m, lapack_int n, - lapack_int kl, lapack_int ku, - lapack_complex_float* ab, lapack_int ldab, - lapack_int* ipiv ); -lapack_int LAPACKE_zgbtrf( int matrix_order, lapack_int m, lapack_int n, - lapack_int kl, lapack_int ku, - lapack_complex_double* ab, lapack_int ldab, - lapack_int* ipiv ); - -lapack_int LAPACKE_sgbtrs( int matrix_order, char trans, lapack_int n, - lapack_int kl, lapack_int ku, lapack_int nrhs, - const float* ab, lapack_int ldab, - const lapack_int* ipiv, float* b, lapack_int ldb ); -lapack_int LAPACKE_dgbtrs( int matrix_order, char trans, lapack_int n, - lapack_int kl, lapack_int ku, lapack_int nrhs, - const double* ab, lapack_int ldab, - const lapack_int* ipiv, double* b, lapack_int ldb ); -lapack_int LAPACKE_cgbtrs( int matrix_order, char trans, lapack_int n, - lapack_int kl, lapack_int ku, lapack_int nrhs, - const lapack_complex_float* ab, lapack_int ldab, - const lapack_int* ipiv, lapack_complex_float* b, - lapack_int ldb ); -lapack_int LAPACKE_zgbtrs( int matrix_order, char trans, lapack_int n, - lapack_int kl, lapack_int ku, lapack_int nrhs, - const lapack_complex_double* ab, lapack_int ldab, - const lapack_int* ipiv, lapack_complex_double* b, - lapack_int ldb ); - -lapack_int LAPACKE_sgebak( int matrix_order, char job, char side, lapack_int n, - lapack_int ilo, lapack_int ihi, const float* scale, - lapack_int m, float* v, lapack_int ldv ); -lapack_int LAPACKE_dgebak( int matrix_order, char job, char side, lapack_int n, - lapack_int ilo, lapack_int ihi, const double* scale, - lapack_int m, double* v, lapack_int ldv ); -lapack_int LAPACKE_cgebak( int matrix_order, char job, char side, lapack_int n, - lapack_int ilo, lapack_int ihi, const float* scale, - lapack_int m, lapack_complex_float* v, - lapack_int ldv ); -lapack_int LAPACKE_zgebak( int matrix_order, char job, char side, lapack_int n, - lapack_int ilo, lapack_int ihi, const double* scale, - lapack_int m, lapack_complex_double* v, - lapack_int ldv ); - -lapack_int LAPACKE_sgebal( int matrix_order, char job, lapack_int n, float* a, - lapack_int lda, lapack_int* ilo, lapack_int* ihi, - float* scale ); -lapack_int LAPACKE_dgebal( int matrix_order, char job, lapack_int n, double* a, - lapack_int lda, lapack_int* ilo, lapack_int* ihi, - double* scale ); -lapack_int LAPACKE_cgebal( int matrix_order, char job, lapack_int n, - lapack_complex_float* a, lapack_int lda, - lapack_int* ilo, lapack_int* ihi, float* scale ); -lapack_int LAPACKE_zgebal( int matrix_order, char job, lapack_int n, - lapack_complex_double* a, lapack_int lda, - lapack_int* ilo, lapack_int* ihi, double* scale ); - -lapack_int LAPACKE_sgebrd( int matrix_order, lapack_int m, lapack_int n, - float* a, lapack_int lda, float* d, float* e, - float* tauq, float* taup ); -lapack_int LAPACKE_dgebrd( int matrix_order, lapack_int m, lapack_int n, - double* a, lapack_int lda, double* d, double* e, - double* tauq, double* taup ); -lapack_int LAPACKE_cgebrd( int matrix_order, lapack_int m, lapack_int n, - lapack_complex_float* a, lapack_int lda, float* d, - float* e, lapack_complex_float* tauq, - lapack_complex_float* taup ); -lapack_int LAPACKE_zgebrd( int matrix_order, lapack_int m, lapack_int n, - lapack_complex_double* a, lapack_int lda, double* d, - double* e, lapack_complex_double* tauq, - lapack_complex_double* taup ); - -lapack_int LAPACKE_sgecon( int matrix_order, char norm, lapack_int n, - const float* a, lapack_int lda, float anorm, - float* rcond ); -lapack_int LAPACKE_dgecon( int matrix_order, char norm, lapack_int n, - const double* a, lapack_int lda, double anorm, - double* rcond ); -lapack_int LAPACKE_cgecon( int matrix_order, char norm, lapack_int n, - const lapack_complex_float* a, lapack_int lda, - float anorm, float* rcond ); -lapack_int LAPACKE_zgecon( int matrix_order, char norm, lapack_int n, - const lapack_complex_double* a, lapack_int lda, - double anorm, double* rcond ); - -lapack_int LAPACKE_sgeequ( int matrix_order, lapack_int m, lapack_int n, - const float* a, lapack_int lda, float* r, float* c, - float* rowcnd, float* colcnd, float* amax ); -lapack_int LAPACKE_dgeequ( int matrix_order, lapack_int m, lapack_int n, - const double* a, lapack_int lda, double* r, - double* c, double* rowcnd, double* colcnd, - double* amax ); -lapack_int LAPACKE_cgeequ( int matrix_order, lapack_int m, lapack_int n, - const lapack_complex_float* a, lapack_int lda, - float* r, float* c, float* rowcnd, float* colcnd, - float* amax ); -lapack_int LAPACKE_zgeequ( int matrix_order, lapack_int m, lapack_int n, - const lapack_complex_double* a, lapack_int lda, - double* r, double* c, double* rowcnd, double* colcnd, - double* amax ); - -lapack_int LAPACKE_sgeequb( int matrix_order, lapack_int m, lapack_int n, - const float* a, lapack_int lda, float* r, float* c, - float* rowcnd, float* colcnd, float* amax ); -lapack_int LAPACKE_dgeequb( int matrix_order, lapack_int m, lapack_int n, - const double* a, lapack_int lda, double* r, - double* c, double* rowcnd, double* colcnd, - double* amax ); -lapack_int LAPACKE_cgeequb( int matrix_order, lapack_int m, lapack_int n, - const lapack_complex_float* a, lapack_int lda, - float* r, float* c, float* rowcnd, float* colcnd, - float* amax ); -lapack_int LAPACKE_zgeequb( int matrix_order, lapack_int m, lapack_int n, - const lapack_complex_double* a, lapack_int lda, - double* r, double* c, double* rowcnd, - double* colcnd, double* amax ); - -lapack_int LAPACKE_sgees( int matrix_order, char jobvs, char sort, - LAPACK_S_SELECT2 select, lapack_int n, float* a, - lapack_int lda, lapack_int* sdim, float* wr, - float* wi, float* vs, lapack_int ldvs ); -lapack_int LAPACKE_dgees( int matrix_order, char jobvs, char sort, - LAPACK_D_SELECT2 select, lapack_int n, double* a, - lapack_int lda, lapack_int* sdim, double* wr, - double* wi, double* vs, lapack_int ldvs ); -lapack_int LAPACKE_cgees( int matrix_order, char jobvs, char sort, - LAPACK_C_SELECT1 select, lapack_int n, - lapack_complex_float* a, lapack_int lda, - lapack_int* sdim, lapack_complex_float* w, - lapack_complex_float* vs, lapack_int ldvs ); -lapack_int LAPACKE_zgees( int matrix_order, char jobvs, char sort, - LAPACK_Z_SELECT1 select, lapack_int n, - lapack_complex_double* a, lapack_int lda, - lapack_int* sdim, lapack_complex_double* w, - lapack_complex_double* vs, lapack_int ldvs ); - -lapack_int LAPACKE_sgeesx( int matrix_order, char jobvs, char sort, - LAPACK_S_SELECT2 select, char sense, lapack_int n, - float* a, lapack_int lda, lapack_int* sdim, - float* wr, float* wi, float* vs, lapack_int ldvs, - float* rconde, float* rcondv ); -lapack_int LAPACKE_dgeesx( int matrix_order, char jobvs, char sort, - LAPACK_D_SELECT2 select, char sense, lapack_int n, - double* a, lapack_int lda, lapack_int* sdim, - double* wr, double* wi, double* vs, lapack_int ldvs, - double* rconde, double* rcondv ); -lapack_int LAPACKE_cgeesx( int matrix_order, char jobvs, char sort, - LAPACK_C_SELECT1 select, char sense, lapack_int n, - lapack_complex_float* a, lapack_int lda, - lapack_int* sdim, lapack_complex_float* w, - lapack_complex_float* vs, lapack_int ldvs, - float* rconde, float* rcondv ); -lapack_int LAPACKE_zgeesx( int matrix_order, char jobvs, char sort, - LAPACK_Z_SELECT1 select, char sense, lapack_int n, - lapack_complex_double* a, lapack_int lda, - lapack_int* sdim, lapack_complex_double* w, - lapack_complex_double* vs, lapack_int ldvs, - double* rconde, double* rcondv ); - -lapack_int LAPACKE_sgeev( int matrix_order, char jobvl, char jobvr, - lapack_int n, float* a, lapack_int lda, float* wr, - float* wi, float* vl, lapack_int ldvl, float* vr, - lapack_int ldvr ); -lapack_int LAPACKE_dgeev( int matrix_order, char jobvl, char jobvr, - lapack_int n, double* a, lapack_int lda, double* wr, - double* wi, double* vl, lapack_int ldvl, double* vr, - lapack_int ldvr ); -lapack_int LAPACKE_cgeev( int matrix_order, char jobvl, char jobvr, - lapack_int n, lapack_complex_float* a, lapack_int lda, - lapack_complex_float* w, lapack_complex_float* vl, - lapack_int ldvl, lapack_complex_float* vr, - lapack_int ldvr ); -lapack_int LAPACKE_zgeev( int matrix_order, char jobvl, char jobvr, - lapack_int n, lapack_complex_double* a, - lapack_int lda, lapack_complex_double* w, - lapack_complex_double* vl, lapack_int ldvl, - lapack_complex_double* vr, lapack_int ldvr ); - -lapack_int LAPACKE_sgeevx( int matrix_order, char balanc, char jobvl, - char jobvr, char sense, lapack_int n, float* a, - lapack_int lda, float* wr, float* wi, float* vl, - lapack_int ldvl, float* vr, lapack_int ldvr, - lapack_int* ilo, lapack_int* ihi, float* scale, - float* abnrm, float* rconde, float* rcondv ); -lapack_int LAPACKE_dgeevx( int matrix_order, char balanc, char jobvl, - char jobvr, char sense, lapack_int n, double* a, - lapack_int lda, double* wr, double* wi, double* vl, - lapack_int ldvl, double* vr, lapack_int ldvr, - lapack_int* ilo, lapack_int* ihi, double* scale, - double* abnrm, double* rconde, double* rcondv ); -lapack_int LAPACKE_cgeevx( int matrix_order, char balanc, char jobvl, - char jobvr, char sense, lapack_int n, - lapack_complex_float* a, lapack_int lda, - lapack_complex_float* w, lapack_complex_float* vl, - lapack_int ldvl, lapack_complex_float* vr, - lapack_int ldvr, lapack_int* ilo, lapack_int* ihi, - float* scale, float* abnrm, float* rconde, - float* rcondv ); -lapack_int LAPACKE_zgeevx( int matrix_order, char balanc, char jobvl, - char jobvr, char sense, lapack_int n, - lapack_complex_double* a, lapack_int lda, - lapack_complex_double* w, lapack_complex_double* vl, - lapack_int ldvl, lapack_complex_double* vr, - lapack_int ldvr, lapack_int* ilo, lapack_int* ihi, - double* scale, double* abnrm, double* rconde, - double* rcondv ); - -lapack_int LAPACKE_sgehrd( int matrix_order, lapack_int n, lapack_int ilo, - lapack_int ihi, float* a, lapack_int lda, - float* tau ); -lapack_int LAPACKE_dgehrd( int matrix_order, lapack_int n, lapack_int ilo, - lapack_int ihi, double* a, lapack_int lda, - double* tau ); -lapack_int LAPACKE_cgehrd( int matrix_order, lapack_int n, lapack_int ilo, - lapack_int ihi, lapack_complex_float* a, - lapack_int lda, lapack_complex_float* tau ); -lapack_int LAPACKE_zgehrd( int matrix_order, lapack_int n, lapack_int ilo, - lapack_int ihi, lapack_complex_double* a, - lapack_int lda, lapack_complex_double* tau ); - -lapack_int LAPACKE_sgejsv( int matrix_order, char joba, char jobu, char jobv, - char jobr, char jobt, char jobp, lapack_int m, - lapack_int n, float* a, lapack_int lda, float* sva, - float* u, lapack_int ldu, float* v, lapack_int ldv, - float* stat, lapack_int* istat ); -lapack_int LAPACKE_dgejsv( int matrix_order, char joba, char jobu, char jobv, - char jobr, char jobt, char jobp, lapack_int m, - lapack_int n, double* a, lapack_int lda, double* sva, - double* u, lapack_int ldu, double* v, lapack_int ldv, - double* stat, lapack_int* istat ); - -lapack_int LAPACKE_sgelq2( int matrix_order, lapack_int m, lapack_int n, - float* a, lapack_int lda, float* tau ); -lapack_int LAPACKE_dgelq2( int matrix_order, lapack_int m, lapack_int n, - double* a, lapack_int lda, double* tau ); -lapack_int LAPACKE_cgelq2( int matrix_order, lapack_int m, lapack_int n, - lapack_complex_float* a, lapack_int lda, - lapack_complex_float* tau ); -lapack_int LAPACKE_zgelq2( int matrix_order, lapack_int m, lapack_int n, - lapack_complex_double* a, lapack_int lda, - lapack_complex_double* tau ); - -lapack_int LAPACKE_sgelqf( int matrix_order, lapack_int m, lapack_int n, - float* a, lapack_int lda, float* tau ); -lapack_int LAPACKE_dgelqf( int matrix_order, lapack_int m, lapack_int n, - double* a, lapack_int lda, double* tau ); -lapack_int LAPACKE_cgelqf( int matrix_order, lapack_int m, lapack_int n, - lapack_complex_float* a, lapack_int lda, - lapack_complex_float* tau ); -lapack_int LAPACKE_zgelqf( int matrix_order, lapack_int m, lapack_int n, - lapack_complex_double* a, lapack_int lda, - lapack_complex_double* tau ); - -lapack_int LAPACKE_sgels( int matrix_order, char trans, lapack_int m, - lapack_int n, lapack_int nrhs, float* a, - lapack_int lda, float* b, lapack_int ldb ); -lapack_int LAPACKE_dgels( int matrix_order, char trans, lapack_int m, - lapack_int n, lapack_int nrhs, double* a, - lapack_int lda, double* b, lapack_int ldb ); -lapack_int LAPACKE_cgels( int matrix_order, char trans, lapack_int m, - lapack_int n, lapack_int nrhs, - lapack_complex_float* a, lapack_int lda, - lapack_complex_float* b, lapack_int ldb ); -lapack_int LAPACKE_zgels( int matrix_order, char trans, lapack_int m, - lapack_int n, lapack_int nrhs, - lapack_complex_double* a, lapack_int lda, - lapack_complex_double* b, lapack_int ldb ); - -lapack_int LAPACKE_sgelsd( int matrix_order, lapack_int m, lapack_int n, - lapack_int nrhs, float* a, lapack_int lda, float* b, - lapack_int ldb, float* s, float rcond, - lapack_int* rank ); -lapack_int LAPACKE_dgelsd( int matrix_order, lapack_int m, lapack_int n, - lapack_int nrhs, double* a, lapack_int lda, - double* b, lapack_int ldb, double* s, double rcond, - lapack_int* rank ); -lapack_int LAPACKE_cgelsd( int matrix_order, lapack_int m, lapack_int n, - lapack_int nrhs, lapack_complex_float* a, - lapack_int lda, lapack_complex_float* b, - lapack_int ldb, float* s, float rcond, - lapack_int* rank ); -lapack_int LAPACKE_zgelsd( int matrix_order, lapack_int m, lapack_int n, - lapack_int nrhs, lapack_complex_double* a, - lapack_int lda, lapack_complex_double* b, - lapack_int ldb, double* s, double rcond, - lapack_int* rank ); - -lapack_int LAPACKE_sgelss( int matrix_order, lapack_int m, lapack_int n, - lapack_int nrhs, float* a, lapack_int lda, float* b, - lapack_int ldb, float* s, float rcond, - lapack_int* rank ); -lapack_int LAPACKE_dgelss( int matrix_order, lapack_int m, lapack_int n, - lapack_int nrhs, double* a, lapack_int lda, - double* b, lapack_int ldb, double* s, double rcond, - lapack_int* rank ); -lapack_int LAPACKE_cgelss( int matrix_order, lapack_int m, lapack_int n, - lapack_int nrhs, lapack_complex_float* a, - lapack_int lda, lapack_complex_float* b, - lapack_int ldb, float* s, float rcond, - lapack_int* rank ); -lapack_int LAPACKE_zgelss( int matrix_order, lapack_int m, lapack_int n, - lapack_int nrhs, lapack_complex_double* a, - lapack_int lda, lapack_complex_double* b, - lapack_int ldb, double* s, double rcond, - lapack_int* rank ); - -lapack_int LAPACKE_sgelsy( int matrix_order, lapack_int m, lapack_int n, - lapack_int nrhs, float* a, lapack_int lda, float* b, - lapack_int ldb, lapack_int* jpvt, float rcond, - lapack_int* rank ); -lapack_int LAPACKE_dgelsy( int matrix_order, lapack_int m, lapack_int n, - lapack_int nrhs, double* a, lapack_int lda, - double* b, lapack_int ldb, lapack_int* jpvt, - double rcond, lapack_int* rank ); -lapack_int LAPACKE_cgelsy( int matrix_order, lapack_int m, lapack_int n, - lapack_int nrhs, lapack_complex_float* a, - lapack_int lda, lapack_complex_float* b, - lapack_int ldb, lapack_int* jpvt, float rcond, - lapack_int* rank ); -lapack_int LAPACKE_zgelsy( int matrix_order, lapack_int m, lapack_int n, - lapack_int nrhs, lapack_complex_double* a, - lapack_int lda, lapack_complex_double* b, - lapack_int ldb, lapack_int* jpvt, double rcond, - lapack_int* rank ); - -lapack_int LAPACKE_sgeqlf( int matrix_order, lapack_int m, lapack_int n, - float* a, lapack_int lda, float* tau ); -lapack_int LAPACKE_dgeqlf( int matrix_order, lapack_int m, lapack_int n, - double* a, lapack_int lda, double* tau ); -lapack_int LAPACKE_cgeqlf( int matrix_order, lapack_int m, lapack_int n, - lapack_complex_float* a, lapack_int lda, - lapack_complex_float* tau ); -lapack_int LAPACKE_zgeqlf( int matrix_order, lapack_int m, lapack_int n, - lapack_complex_double* a, lapack_int lda, - lapack_complex_double* tau ); - -lapack_int LAPACKE_sgeqp3( int matrix_order, lapack_int m, lapack_int n, - float* a, lapack_int lda, lapack_int* jpvt, - float* tau ); -lapack_int LAPACKE_dgeqp3( int matrix_order, lapack_int m, lapack_int n, - double* a, lapack_int lda, lapack_int* jpvt, - double* tau ); -lapack_int LAPACKE_cgeqp3( int matrix_order, lapack_int m, lapack_int n, - lapack_complex_float* a, lapack_int lda, - lapack_int* jpvt, lapack_complex_float* tau ); -lapack_int LAPACKE_zgeqp3( int matrix_order, lapack_int m, lapack_int n, - lapack_complex_double* a, lapack_int lda, - lapack_int* jpvt, lapack_complex_double* tau ); - -lapack_int LAPACKE_sgeqpf( int matrix_order, lapack_int m, lapack_int n, - float* a, lapack_int lda, lapack_int* jpvt, - float* tau ); -lapack_int LAPACKE_dgeqpf( int matrix_order, lapack_int m, lapack_int n, - double* a, lapack_int lda, lapack_int* jpvt, - double* tau ); -lapack_int LAPACKE_cgeqpf( int matrix_order, lapack_int m, lapack_int n, - lapack_complex_float* a, lapack_int lda, - lapack_int* jpvt, lapack_complex_float* tau ); -lapack_int LAPACKE_zgeqpf( int matrix_order, lapack_int m, lapack_int n, - lapack_complex_double* a, lapack_int lda, - lapack_int* jpvt, lapack_complex_double* tau ); - -lapack_int LAPACKE_sgeqr2( int matrix_order, lapack_int m, lapack_int n, - float* a, lapack_int lda, float* tau ); -lapack_int LAPACKE_dgeqr2( int matrix_order, lapack_int m, lapack_int n, - double* a, lapack_int lda, double* tau ); -lapack_int LAPACKE_cgeqr2( int matrix_order, lapack_int m, lapack_int n, - lapack_complex_float* a, lapack_int lda, - lapack_complex_float* tau ); -lapack_int LAPACKE_zgeqr2( int matrix_order, lapack_int m, lapack_int n, - lapack_complex_double* a, lapack_int lda, - lapack_complex_double* tau ); - -lapack_int LAPACKE_sgeqrf( int matrix_order, lapack_int m, lapack_int n, - float* a, lapack_int lda, float* tau ); -lapack_int LAPACKE_dgeqrf( int matrix_order, lapack_int m, lapack_int n, - double* a, lapack_int lda, double* tau ); -lapack_int LAPACKE_cgeqrf( int matrix_order, lapack_int m, lapack_int n, - lapack_complex_float* a, lapack_int lda, - lapack_complex_float* tau ); -lapack_int LAPACKE_zgeqrf( int matrix_order, lapack_int m, lapack_int n, - lapack_complex_double* a, lapack_int lda, - lapack_complex_double* tau ); - -lapack_int LAPACKE_sgeqrfp( int matrix_order, lapack_int m, lapack_int n, - float* a, lapack_int lda, float* tau ); -lapack_int LAPACKE_dgeqrfp( int matrix_order, lapack_int m, lapack_int n, - double* a, lapack_int lda, double* tau ); -lapack_int LAPACKE_cgeqrfp( int matrix_order, lapack_int m, lapack_int n, - lapack_complex_float* a, lapack_int lda, - lapack_complex_float* tau ); -lapack_int LAPACKE_zgeqrfp( int matrix_order, lapack_int m, lapack_int n, - lapack_complex_double* a, lapack_int lda, - lapack_complex_double* tau ); - -lapack_int LAPACKE_sgerfs( int matrix_order, char trans, lapack_int n, - lapack_int nrhs, const float* a, lapack_int lda, - const float* af, lapack_int ldaf, - const lapack_int* ipiv, const float* b, - lapack_int ldb, float* x, lapack_int ldx, - float* ferr, float* berr ); -lapack_int LAPACKE_dgerfs( int matrix_order, char trans, lapack_int n, - lapack_int nrhs, const double* a, lapack_int lda, - const double* af, lapack_int ldaf, - const lapack_int* ipiv, const double* b, - lapack_int ldb, double* x, lapack_int ldx, - double* ferr, double* berr ); -lapack_int LAPACKE_cgerfs( int matrix_order, char trans, lapack_int n, - lapack_int nrhs, const lapack_complex_float* a, - lapack_int lda, const lapack_complex_float* af, - lapack_int ldaf, const lapack_int* ipiv, - const lapack_complex_float* b, lapack_int ldb, - lapack_complex_float* x, lapack_int ldx, float* ferr, - float* berr ); -lapack_int LAPACKE_zgerfs( int matrix_order, char trans, lapack_int n, - lapack_int nrhs, const lapack_complex_double* a, - lapack_int lda, const lapack_complex_double* af, - lapack_int ldaf, const lapack_int* ipiv, - const lapack_complex_double* b, lapack_int ldb, - lapack_complex_double* x, lapack_int ldx, - double* ferr, double* berr ); - -lapack_int LAPACKE_sgerfsx( int matrix_order, char trans, char equed, - lapack_int n, lapack_int nrhs, const float* a, - lapack_int lda, const float* af, lapack_int ldaf, - const lapack_int* ipiv, const float* r, - const float* c, const float* b, lapack_int ldb, - float* x, lapack_int ldx, float* rcond, float* berr, - lapack_int n_err_bnds, float* err_bnds_norm, - float* err_bnds_comp, lapack_int nparams, - float* params ); -lapack_int LAPACKE_dgerfsx( int matrix_order, char trans, char equed, - lapack_int n, lapack_int nrhs, const double* a, - lapack_int lda, const double* af, lapack_int ldaf, - const lapack_int* ipiv, const double* r, - const double* c, const double* b, lapack_int ldb, - double* x, lapack_int ldx, double* rcond, - double* berr, lapack_int n_err_bnds, - double* err_bnds_norm, double* err_bnds_comp, - lapack_int nparams, double* params ); -lapack_int LAPACKE_cgerfsx( int matrix_order, char trans, char equed, - lapack_int n, lapack_int nrhs, - const lapack_complex_float* a, lapack_int lda, - const lapack_complex_float* af, lapack_int ldaf, - const lapack_int* ipiv, const float* r, - const float* c, const lapack_complex_float* b, - lapack_int ldb, lapack_complex_float* x, - lapack_int ldx, float* rcond, float* berr, - lapack_int n_err_bnds, float* err_bnds_norm, - float* err_bnds_comp, lapack_int nparams, - float* params ); -lapack_int LAPACKE_zgerfsx( int matrix_order, char trans, char equed, - lapack_int n, lapack_int nrhs, - const lapack_complex_double* a, lapack_int lda, - const lapack_complex_double* af, lapack_int ldaf, - const lapack_int* ipiv, const double* r, - const double* c, const lapack_complex_double* b, - lapack_int ldb, lapack_complex_double* x, - lapack_int ldx, double* rcond, double* berr, - lapack_int n_err_bnds, double* err_bnds_norm, - double* err_bnds_comp, lapack_int nparams, - double* params ); - -lapack_int LAPACKE_sgerqf( int matrix_order, lapack_int m, lapack_int n, - float* a, lapack_int lda, float* tau ); -lapack_int LAPACKE_dgerqf( int matrix_order, lapack_int m, lapack_int n, - double* a, lapack_int lda, double* tau ); -lapack_int LAPACKE_cgerqf( int matrix_order, lapack_int m, lapack_int n, - lapack_complex_float* a, lapack_int lda, - lapack_complex_float* tau ); -lapack_int LAPACKE_zgerqf( int matrix_order, lapack_int m, lapack_int n, - lapack_complex_double* a, lapack_int lda, - lapack_complex_double* tau ); - -lapack_int LAPACKE_sgesdd( int matrix_order, char jobz, lapack_int m, - lapack_int n, float* a, lapack_int lda, float* s, - float* u, lapack_int ldu, float* vt, - lapack_int ldvt ); -lapack_int LAPACKE_dgesdd( int matrix_order, char jobz, lapack_int m, - lapack_int n, double* a, lapack_int lda, double* s, - double* u, lapack_int ldu, double* vt, - lapack_int ldvt ); -lapack_int LAPACKE_cgesdd( int matrix_order, char jobz, lapack_int m, - lapack_int n, lapack_complex_float* a, - lapack_int lda, float* s, lapack_complex_float* u, - lapack_int ldu, lapack_complex_float* vt, - lapack_int ldvt ); -lapack_int LAPACKE_zgesdd( int matrix_order, char jobz, lapack_int m, - lapack_int n, lapack_complex_double* a, - lapack_int lda, double* s, lapack_complex_double* u, - lapack_int ldu, lapack_complex_double* vt, - lapack_int ldvt ); - -lapack_int LAPACKE_sgesv( int matrix_order, lapack_int n, lapack_int nrhs, - float* a, lapack_int lda, lapack_int* ipiv, float* b, - lapack_int ldb ); -lapack_int LAPACKE_dgesv( int matrix_order, lapack_int n, lapack_int nrhs, - double* a, lapack_int lda, lapack_int* ipiv, - double* b, lapack_int ldb ); -lapack_int LAPACKE_cgesv( int matrix_order, lapack_int n, lapack_int nrhs, - lapack_complex_float* a, lapack_int lda, - lapack_int* ipiv, lapack_complex_float* b, - lapack_int ldb ); -lapack_int LAPACKE_zgesv( int matrix_order, lapack_int n, lapack_int nrhs, - lapack_complex_double* a, lapack_int lda, - lapack_int* ipiv, lapack_complex_double* b, - lapack_int ldb ); -lapack_int LAPACKE_dsgesv( int matrix_order, lapack_int n, lapack_int nrhs, - double* a, lapack_int lda, lapack_int* ipiv, - double* b, lapack_int ldb, double* x, lapack_int ldx, - lapack_int* iter ); -lapack_int LAPACKE_zcgesv( int matrix_order, lapack_int n, lapack_int nrhs, - lapack_complex_double* a, lapack_int lda, - lapack_int* ipiv, lapack_complex_double* b, - lapack_int ldb, lapack_complex_double* x, - lapack_int ldx, lapack_int* iter ); - -lapack_int LAPACKE_sgesvd( int matrix_order, char jobu, char jobvt, - lapack_int m, lapack_int n, float* a, lapack_int lda, - float* s, float* u, lapack_int ldu, float* vt, - lapack_int ldvt, float* superb ); -lapack_int LAPACKE_dgesvd( int matrix_order, char jobu, char jobvt, - lapack_int m, lapack_int n, double* a, - lapack_int lda, double* s, double* u, lapack_int ldu, - double* vt, lapack_int ldvt, double* superb ); -lapack_int LAPACKE_cgesvd( int matrix_order, char jobu, char jobvt, - lapack_int m, lapack_int n, lapack_complex_float* a, - lapack_int lda, float* s, lapack_complex_float* u, - lapack_int ldu, lapack_complex_float* vt, - lapack_int ldvt, float* superb ); -lapack_int LAPACKE_zgesvd( int matrix_order, char jobu, char jobvt, - lapack_int m, lapack_int n, lapack_complex_double* a, - lapack_int lda, double* s, lapack_complex_double* u, - lapack_int ldu, lapack_complex_double* vt, - lapack_int ldvt, double* superb ); - -lapack_int LAPACKE_sgesvj( int matrix_order, char joba, char jobu, char jobv, - lapack_int m, lapack_int n, float* a, lapack_int lda, - float* sva, lapack_int mv, float* v, lapack_int ldv, - float* stat ); -lapack_int LAPACKE_dgesvj( int matrix_order, char joba, char jobu, char jobv, - lapack_int m, lapack_int n, double* a, - lapack_int lda, double* sva, lapack_int mv, - double* v, lapack_int ldv, double* stat ); - -lapack_int LAPACKE_sgesvx( int matrix_order, char fact, char trans, - lapack_int n, lapack_int nrhs, float* a, - lapack_int lda, float* af, lapack_int ldaf, - lapack_int* ipiv, char* equed, float* r, float* c, - float* b, lapack_int ldb, float* x, lapack_int ldx, - float* rcond, float* ferr, float* berr, - float* rpivot ); -lapack_int LAPACKE_dgesvx( int matrix_order, char fact, char trans, - lapack_int n, lapack_int nrhs, double* a, - lapack_int lda, double* af, lapack_int ldaf, - lapack_int* ipiv, char* equed, double* r, double* c, - double* b, lapack_int ldb, double* x, lapack_int ldx, - double* rcond, double* ferr, double* berr, - double* rpivot ); -lapack_int LAPACKE_cgesvx( int matrix_order, char fact, char trans, - lapack_int n, lapack_int nrhs, - lapack_complex_float* a, lapack_int lda, - lapack_complex_float* af, lapack_int ldaf, - lapack_int* ipiv, char* equed, float* r, float* c, - lapack_complex_float* b, lapack_int ldb, - lapack_complex_float* x, lapack_int ldx, - float* rcond, float* ferr, float* berr, - float* rpivot ); -lapack_int LAPACKE_zgesvx( int matrix_order, char fact, char trans, - lapack_int n, lapack_int nrhs, - lapack_complex_double* a, lapack_int lda, - lapack_complex_double* af, lapack_int ldaf, - lapack_int* ipiv, char* equed, double* r, double* c, - lapack_complex_double* b, lapack_int ldb, - lapack_complex_double* x, lapack_int ldx, - double* rcond, double* ferr, double* berr, - double* rpivot ); - -lapack_int LAPACKE_sgesvxx( int matrix_order, char fact, char trans, - lapack_int n, lapack_int nrhs, float* a, - lapack_int lda, float* af, lapack_int ldaf, - lapack_int* ipiv, char* equed, float* r, float* c, - float* b, lapack_int ldb, float* x, lapack_int ldx, - float* rcond, float* rpvgrw, float* berr, - lapack_int n_err_bnds, float* err_bnds_norm, - float* err_bnds_comp, lapack_int nparams, - float* params ); -lapack_int LAPACKE_dgesvxx( int matrix_order, char fact, char trans, - lapack_int n, lapack_int nrhs, double* a, - lapack_int lda, double* af, lapack_int ldaf, - lapack_int* ipiv, char* equed, double* r, double* c, - double* b, lapack_int ldb, double* x, - lapack_int ldx, double* rcond, double* rpvgrw, - double* berr, lapack_int n_err_bnds, - double* err_bnds_norm, double* err_bnds_comp, - lapack_int nparams, double* params ); -lapack_int LAPACKE_cgesvxx( int matrix_order, char fact, char trans, - lapack_int n, lapack_int nrhs, - lapack_complex_float* a, lapack_int lda, - lapack_complex_float* af, lapack_int ldaf, - lapack_int* ipiv, char* equed, float* r, float* c, - lapack_complex_float* b, lapack_int ldb, - lapack_complex_float* x, lapack_int ldx, - float* rcond, float* rpvgrw, float* berr, - lapack_int n_err_bnds, float* err_bnds_norm, - float* err_bnds_comp, lapack_int nparams, - float* params ); -lapack_int LAPACKE_zgesvxx( int matrix_order, char fact, char trans, - lapack_int n, lapack_int nrhs, - lapack_complex_double* a, lapack_int lda, - lapack_complex_double* af, lapack_int ldaf, - lapack_int* ipiv, char* equed, double* r, double* c, - lapack_complex_double* b, lapack_int ldb, - lapack_complex_double* x, lapack_int ldx, - double* rcond, double* rpvgrw, double* berr, - lapack_int n_err_bnds, double* err_bnds_norm, - double* err_bnds_comp, lapack_int nparams, - double* params ); - -lapack_int LAPACKE_sgetf2( int matrix_order, lapack_int m, lapack_int n, - float* a, lapack_int lda, lapack_int* ipiv ); -lapack_int LAPACKE_dgetf2( int matrix_order, lapack_int m, lapack_int n, - double* a, lapack_int lda, lapack_int* ipiv ); -lapack_int LAPACKE_cgetf2( int matrix_order, lapack_int m, lapack_int n, - lapack_complex_float* a, lapack_int lda, - lapack_int* ipiv ); -lapack_int LAPACKE_zgetf2( int matrix_order, lapack_int m, lapack_int n, - lapack_complex_double* a, lapack_int lda, - lapack_int* ipiv ); - -lapack_int LAPACKE_sgetrf( int matrix_order, lapack_int m, lapack_int n, - float* a, lapack_int lda, lapack_int* ipiv ); -lapack_int LAPACKE_dgetrf( int matrix_order, lapack_int m, lapack_int n, - double* a, lapack_int lda, lapack_int* ipiv ); -lapack_int LAPACKE_cgetrf( int matrix_order, lapack_int m, lapack_int n, - lapack_complex_float* a, lapack_int lda, - lapack_int* ipiv ); -lapack_int LAPACKE_zgetrf( int matrix_order, lapack_int m, lapack_int n, - lapack_complex_double* a, lapack_int lda, - lapack_int* ipiv ); - -lapack_int LAPACKE_sgetri( int matrix_order, lapack_int n, float* a, - lapack_int lda, const lapack_int* ipiv ); -lapack_int LAPACKE_dgetri( int matrix_order, lapack_int n, double* a, - lapack_int lda, const lapack_int* ipiv ); -lapack_int LAPACKE_cgetri( int matrix_order, lapack_int n, - lapack_complex_float* a, lapack_int lda, - const lapack_int* ipiv ); -lapack_int LAPACKE_zgetri( int matrix_order, lapack_int n, - lapack_complex_double* a, lapack_int lda, - const lapack_int* ipiv ); - -lapack_int LAPACKE_sgetrs( int matrix_order, char trans, lapack_int n, - lapack_int nrhs, const float* a, lapack_int lda, - const lapack_int* ipiv, float* b, lapack_int ldb ); -lapack_int LAPACKE_dgetrs( int matrix_order, char trans, lapack_int n, - lapack_int nrhs, const double* a, lapack_int lda, - const lapack_int* ipiv, double* b, lapack_int ldb ); -lapack_int LAPACKE_cgetrs( int matrix_order, char trans, lapack_int n, - lapack_int nrhs, const lapack_complex_float* a, - lapack_int lda, const lapack_int* ipiv, - lapack_complex_float* b, lapack_int ldb ); -lapack_int LAPACKE_zgetrs( int matrix_order, char trans, lapack_int n, - lapack_int nrhs, const lapack_complex_double* a, - lapack_int lda, const lapack_int* ipiv, - lapack_complex_double* b, lapack_int ldb ); - -lapack_int LAPACKE_sggbak( int matrix_order, char job, char side, lapack_int n, - lapack_int ilo, lapack_int ihi, const float* lscale, - const float* rscale, lapack_int m, float* v, - lapack_int ldv ); -lapack_int LAPACKE_dggbak( int matrix_order, char job, char side, lapack_int n, - lapack_int ilo, lapack_int ihi, const double* lscale, - const double* rscale, lapack_int m, double* v, - lapack_int ldv ); -lapack_int LAPACKE_cggbak( int matrix_order, char job, char side, lapack_int n, - lapack_int ilo, lapack_int ihi, const float* lscale, - const float* rscale, lapack_int m, - lapack_complex_float* v, lapack_int ldv ); -lapack_int LAPACKE_zggbak( int matrix_order, char job, char side, lapack_int n, - lapack_int ilo, lapack_int ihi, const double* lscale, - const double* rscale, lapack_int m, - lapack_complex_double* v, lapack_int ldv ); - -lapack_int LAPACKE_sggbal( int matrix_order, char job, lapack_int n, float* a, - lapack_int lda, float* b, lapack_int ldb, - lapack_int* ilo, lapack_int* ihi, float* lscale, - float* rscale ); -lapack_int LAPACKE_dggbal( int matrix_order, char job, lapack_int n, double* a, - lapack_int lda, double* b, lapack_int ldb, - lapack_int* ilo, lapack_int* ihi, double* lscale, - double* rscale ); -lapack_int LAPACKE_cggbal( int matrix_order, char job, lapack_int n, - lapack_complex_float* a, lapack_int lda, - lapack_complex_float* b, lapack_int ldb, - lapack_int* ilo, lapack_int* ihi, float* lscale, - float* rscale ); -lapack_int LAPACKE_zggbal( int matrix_order, char job, lapack_int n, - lapack_complex_double* a, lapack_int lda, - lapack_complex_double* b, lapack_int ldb, - lapack_int* ilo, lapack_int* ihi, double* lscale, - double* rscale ); - -lapack_int LAPACKE_sgges( int matrix_order, char jobvsl, char jobvsr, char sort, - LAPACK_S_SELECT3 selctg, lapack_int n, float* a, - lapack_int lda, float* b, lapack_int ldb, - lapack_int* sdim, float* alphar, float* alphai, - float* beta, float* vsl, lapack_int ldvsl, float* vsr, - lapack_int ldvsr ); -lapack_int LAPACKE_dgges( int matrix_order, char jobvsl, char jobvsr, char sort, - LAPACK_D_SELECT3 selctg, lapack_int n, double* a, - lapack_int lda, double* b, lapack_int ldb, - lapack_int* sdim, double* alphar, double* alphai, - double* beta, double* vsl, lapack_int ldvsl, - double* vsr, lapack_int ldvsr ); -lapack_int LAPACKE_cgges( int matrix_order, char jobvsl, char jobvsr, char sort, - LAPACK_C_SELECT2 selctg, lapack_int n, - lapack_complex_float* a, lapack_int lda, - lapack_complex_float* b, lapack_int ldb, - lapack_int* sdim, lapack_complex_float* alpha, - lapack_complex_float* beta, lapack_complex_float* vsl, - lapack_int ldvsl, lapack_complex_float* vsr, - lapack_int ldvsr ); -lapack_int LAPACKE_zgges( int matrix_order, char jobvsl, char jobvsr, char sort, - LAPACK_Z_SELECT2 selctg, lapack_int n, - lapack_complex_double* a, lapack_int lda, - lapack_complex_double* b, lapack_int ldb, - lapack_int* sdim, lapack_complex_double* alpha, - lapack_complex_double* beta, - lapack_complex_double* vsl, lapack_int ldvsl, - lapack_complex_double* vsr, lapack_int ldvsr ); - -lapack_int LAPACKE_sggesx( int matrix_order, char jobvsl, char jobvsr, - char sort, LAPACK_S_SELECT3 selctg, char sense, - lapack_int n, float* a, lapack_int lda, float* b, - lapack_int ldb, lapack_int* sdim, float* alphar, - float* alphai, float* beta, float* vsl, - lapack_int ldvsl, float* vsr, lapack_int ldvsr, - float* rconde, float* rcondv ); -lapack_int LAPACKE_dggesx( int matrix_order, char jobvsl, char jobvsr, - char sort, LAPACK_D_SELECT3 selctg, char sense, - lapack_int n, double* a, lapack_int lda, double* b, - lapack_int ldb, lapack_int* sdim, double* alphar, - double* alphai, double* beta, double* vsl, - lapack_int ldvsl, double* vsr, lapack_int ldvsr, - double* rconde, double* rcondv ); -lapack_int LAPACKE_cggesx( int matrix_order, char jobvsl, char jobvsr, - char sort, LAPACK_C_SELECT2 selctg, char sense, - lapack_int n, lapack_complex_float* a, - lapack_int lda, lapack_complex_float* b, - lapack_int ldb, lapack_int* sdim, - lapack_complex_float* alpha, - lapack_complex_float* beta, - lapack_complex_float* vsl, lapack_int ldvsl, - lapack_complex_float* vsr, lapack_int ldvsr, - float* rconde, float* rcondv ); -lapack_int LAPACKE_zggesx( int matrix_order, char jobvsl, char jobvsr, - char sort, LAPACK_Z_SELECT2 selctg, char sense, - lapack_int n, lapack_complex_double* a, - lapack_int lda, lapack_complex_double* b, - lapack_int ldb, lapack_int* sdim, - lapack_complex_double* alpha, - lapack_complex_double* beta, - lapack_complex_double* vsl, lapack_int ldvsl, - lapack_complex_double* vsr, lapack_int ldvsr, - double* rconde, double* rcondv ); - -lapack_int LAPACKE_sggev( int matrix_order, char jobvl, char jobvr, - lapack_int n, float* a, lapack_int lda, float* b, - lapack_int ldb, float* alphar, float* alphai, - float* beta, float* vl, lapack_int ldvl, float* vr, - lapack_int ldvr ); -lapack_int LAPACKE_dggev( int matrix_order, char jobvl, char jobvr, - lapack_int n, double* a, lapack_int lda, double* b, - lapack_int ldb, double* alphar, double* alphai, - double* beta, double* vl, lapack_int ldvl, double* vr, - lapack_int ldvr ); -lapack_int LAPACKE_cggev( int matrix_order, char jobvl, char jobvr, - lapack_int n, lapack_complex_float* a, lapack_int lda, - lapack_complex_float* b, lapack_int ldb, - lapack_complex_float* alpha, - lapack_complex_float* beta, lapack_complex_float* vl, - lapack_int ldvl, lapack_complex_float* vr, - lapack_int ldvr ); -lapack_int LAPACKE_zggev( int matrix_order, char jobvl, char jobvr, - lapack_int n, lapack_complex_double* a, - lapack_int lda, lapack_complex_double* b, - lapack_int ldb, lapack_complex_double* alpha, - lapack_complex_double* beta, - lapack_complex_double* vl, lapack_int ldvl, - lapack_complex_double* vr, lapack_int ldvr ); - -lapack_int LAPACKE_sggevx( int matrix_order, char balanc, char jobvl, - char jobvr, char sense, lapack_int n, float* a, - lapack_int lda, float* b, lapack_int ldb, - float* alphar, float* alphai, float* beta, float* vl, - lapack_int ldvl, float* vr, lapack_int ldvr, - lapack_int* ilo, lapack_int* ihi, float* lscale, - float* rscale, float* abnrm, float* bbnrm, - float* rconde, float* rcondv ); -lapack_int LAPACKE_dggevx( int matrix_order, char balanc, char jobvl, - char jobvr, char sense, lapack_int n, double* a, - lapack_int lda, double* b, lapack_int ldb, - double* alphar, double* alphai, double* beta, - double* vl, lapack_int ldvl, double* vr, - lapack_int ldvr, lapack_int* ilo, lapack_int* ihi, - double* lscale, double* rscale, double* abnrm, - double* bbnrm, double* rconde, double* rcondv ); -lapack_int LAPACKE_cggevx( int matrix_order, char balanc, char jobvl, - char jobvr, char sense, lapack_int n, - lapack_complex_float* a, lapack_int lda, - lapack_complex_float* b, lapack_int ldb, - lapack_complex_float* alpha, - lapack_complex_float* beta, lapack_complex_float* vl, - lapack_int ldvl, lapack_complex_float* vr, - lapack_int ldvr, lapack_int* ilo, lapack_int* ihi, - float* lscale, float* rscale, float* abnrm, - float* bbnrm, float* rconde, float* rcondv ); -lapack_int LAPACKE_zggevx( int matrix_order, char balanc, char jobvl, - char jobvr, char sense, lapack_int n, - lapack_complex_double* a, lapack_int lda, - lapack_complex_double* b, lapack_int ldb, - lapack_complex_double* alpha, - lapack_complex_double* beta, - lapack_complex_double* vl, lapack_int ldvl, - lapack_complex_double* vr, lapack_int ldvr, - lapack_int* ilo, lapack_int* ihi, double* lscale, - double* rscale, double* abnrm, double* bbnrm, - double* rconde, double* rcondv ); - -lapack_int LAPACKE_sggglm( int matrix_order, lapack_int n, lapack_int m, - lapack_int p, float* a, lapack_int lda, float* b, - lapack_int ldb, float* d, float* x, float* y ); -lapack_int LAPACKE_dggglm( int matrix_order, lapack_int n, lapack_int m, - lapack_int p, double* a, lapack_int lda, double* b, - lapack_int ldb, double* d, double* x, double* y ); -lapack_int LAPACKE_cggglm( int matrix_order, lapack_int n, lapack_int m, - lapack_int p, lapack_complex_float* a, - lapack_int lda, lapack_complex_float* b, - lapack_int ldb, lapack_complex_float* d, - lapack_complex_float* x, lapack_complex_float* y ); -lapack_int LAPACKE_zggglm( int matrix_order, lapack_int n, lapack_int m, - lapack_int p, lapack_complex_double* a, - lapack_int lda, lapack_complex_double* b, - lapack_int ldb, lapack_complex_double* d, - lapack_complex_double* x, lapack_complex_double* y ); - -lapack_int LAPACKE_sgghrd( int matrix_order, char compq, char compz, - lapack_int n, lapack_int ilo, lapack_int ihi, - float* a, lapack_int lda, float* b, lapack_int ldb, - float* q, lapack_int ldq, float* z, lapack_int ldz ); -lapack_int LAPACKE_dgghrd( int matrix_order, char compq, char compz, - lapack_int n, lapack_int ilo, lapack_int ihi, - double* a, lapack_int lda, double* b, lapack_int ldb, - double* q, lapack_int ldq, double* z, - lapack_int ldz ); -lapack_int LAPACKE_cgghrd( int matrix_order, char compq, char compz, - lapack_int n, lapack_int ilo, lapack_int ihi, - lapack_complex_float* a, lapack_int lda, - lapack_complex_float* b, lapack_int ldb, - lapack_complex_float* q, lapack_int ldq, - lapack_complex_float* z, lapack_int ldz ); -lapack_int LAPACKE_zgghrd( int matrix_order, char compq, char compz, - lapack_int n, lapack_int ilo, lapack_int ihi, - lapack_complex_double* a, lapack_int lda, - lapack_complex_double* b, lapack_int ldb, - lapack_complex_double* q, lapack_int ldq, - lapack_complex_double* z, lapack_int ldz ); - -lapack_int LAPACKE_sgglse( int matrix_order, lapack_int m, lapack_int n, - lapack_int p, float* a, lapack_int lda, float* b, - lapack_int ldb, float* c, float* d, float* x ); -lapack_int LAPACKE_dgglse( int matrix_order, lapack_int m, lapack_int n, - lapack_int p, double* a, lapack_int lda, double* b, - lapack_int ldb, double* c, double* d, double* x ); -lapack_int LAPACKE_cgglse( int matrix_order, lapack_int m, lapack_int n, - lapack_int p, lapack_complex_float* a, - lapack_int lda, lapack_complex_float* b, - lapack_int ldb, lapack_complex_float* c, - lapack_complex_float* d, lapack_complex_float* x ); -lapack_int LAPACKE_zgglse( int matrix_order, lapack_int m, lapack_int n, - lapack_int p, lapack_complex_double* a, - lapack_int lda, lapack_complex_double* b, - lapack_int ldb, lapack_complex_double* c, - lapack_complex_double* d, lapack_complex_double* x ); - -lapack_int LAPACKE_sggqrf( int matrix_order, lapack_int n, lapack_int m, - lapack_int p, float* a, lapack_int lda, float* taua, - float* b, lapack_int ldb, float* taub ); -lapack_int LAPACKE_dggqrf( int matrix_order, lapack_int n, lapack_int m, - lapack_int p, double* a, lapack_int lda, - double* taua, double* b, lapack_int ldb, - double* taub ); -lapack_int LAPACKE_cggqrf( int matrix_order, lapack_int n, lapack_int m, - lapack_int p, lapack_complex_float* a, - lapack_int lda, lapack_complex_float* taua, - lapack_complex_float* b, lapack_int ldb, - lapack_complex_float* taub ); -lapack_int LAPACKE_zggqrf( int matrix_order, lapack_int n, lapack_int m, - lapack_int p, lapack_complex_double* a, - lapack_int lda, lapack_complex_double* taua, - lapack_complex_double* b, lapack_int ldb, - lapack_complex_double* taub ); - -lapack_int LAPACKE_sggrqf( int matrix_order, lapack_int m, lapack_int p, - lapack_int n, float* a, lapack_int lda, float* taua, - float* b, lapack_int ldb, float* taub ); -lapack_int LAPACKE_dggrqf( int matrix_order, lapack_int m, lapack_int p, - lapack_int n, double* a, lapack_int lda, - double* taua, double* b, lapack_int ldb, - double* taub ); -lapack_int LAPACKE_cggrqf( int matrix_order, lapack_int m, lapack_int p, - lapack_int n, lapack_complex_float* a, - lapack_int lda, lapack_complex_float* taua, - lapack_complex_float* b, lapack_int ldb, - lapack_complex_float* taub ); -lapack_int LAPACKE_zggrqf( int matrix_order, lapack_int m, lapack_int p, - lapack_int n, lapack_complex_double* a, - lapack_int lda, lapack_complex_double* taua, - lapack_complex_double* b, lapack_int ldb, - lapack_complex_double* taub ); - -lapack_int LAPACKE_sggsvd( int matrix_order, char jobu, char jobv, char jobq, - lapack_int m, lapack_int n, lapack_int p, - lapack_int* k, lapack_int* l, float* a, - lapack_int lda, float* b, lapack_int ldb, - float* alpha, float* beta, float* u, lapack_int ldu, - float* v, lapack_int ldv, float* q, lapack_int ldq, - lapack_int* iwork ); -lapack_int LAPACKE_dggsvd( int matrix_order, char jobu, char jobv, char jobq, - lapack_int m, lapack_int n, lapack_int p, - lapack_int* k, lapack_int* l, double* a, - lapack_int lda, double* b, lapack_int ldb, - double* alpha, double* beta, double* u, - lapack_int ldu, double* v, lapack_int ldv, double* q, - lapack_int ldq, lapack_int* iwork ); -lapack_int LAPACKE_cggsvd( int matrix_order, char jobu, char jobv, char jobq, - lapack_int m, lapack_int n, lapack_int p, - lapack_int* k, lapack_int* l, - lapack_complex_float* a, lapack_int lda, - lapack_complex_float* b, lapack_int ldb, - float* alpha, float* beta, lapack_complex_float* u, - lapack_int ldu, lapack_complex_float* v, - lapack_int ldv, lapack_complex_float* q, - lapack_int ldq, lapack_int* iwork ); -lapack_int LAPACKE_zggsvd( int matrix_order, char jobu, char jobv, char jobq, - lapack_int m, lapack_int n, lapack_int p, - lapack_int* k, lapack_int* l, - lapack_complex_double* a, lapack_int lda, - lapack_complex_double* b, lapack_int ldb, - double* alpha, double* beta, - lapack_complex_double* u, lapack_int ldu, - lapack_complex_double* v, lapack_int ldv, - lapack_complex_double* q, lapack_int ldq, - lapack_int* iwork ); - -lapack_int LAPACKE_sggsvp( int matrix_order, char jobu, char jobv, char jobq, - lapack_int m, lapack_int p, lapack_int n, float* a, - lapack_int lda, float* b, lapack_int ldb, float tola, - float tolb, lapack_int* k, lapack_int* l, float* u, - lapack_int ldu, float* v, lapack_int ldv, float* q, - lapack_int ldq ); -lapack_int LAPACKE_dggsvp( int matrix_order, char jobu, char jobv, char jobq, - lapack_int m, lapack_int p, lapack_int n, double* a, - lapack_int lda, double* b, lapack_int ldb, - double tola, double tolb, lapack_int* k, - lapack_int* l, double* u, lapack_int ldu, double* v, - lapack_int ldv, double* q, lapack_int ldq ); -lapack_int LAPACKE_cggsvp( int matrix_order, char jobu, char jobv, char jobq, - lapack_int m, lapack_int p, lapack_int n, - lapack_complex_float* a, lapack_int lda, - lapack_complex_float* b, lapack_int ldb, float tola, - float tolb, lapack_int* k, lapack_int* l, - lapack_complex_float* u, lapack_int ldu, - lapack_complex_float* v, lapack_int ldv, - lapack_complex_float* q, lapack_int ldq ); -lapack_int LAPACKE_zggsvp( int matrix_order, char jobu, char jobv, char jobq, - lapack_int m, lapack_int p, lapack_int n, - lapack_complex_double* a, lapack_int lda, - lapack_complex_double* b, lapack_int ldb, - double tola, double tolb, lapack_int* k, - lapack_int* l, lapack_complex_double* u, - lapack_int ldu, lapack_complex_double* v, - lapack_int ldv, lapack_complex_double* q, - lapack_int ldq ); - -lapack_int LAPACKE_sgtcon( char norm, lapack_int n, const float* dl, - const float* d, const float* du, const float* du2, - const lapack_int* ipiv, float anorm, float* rcond ); -lapack_int LAPACKE_dgtcon( char norm, lapack_int n, const double* dl, - const double* d, const double* du, const double* du2, - const lapack_int* ipiv, double anorm, - double* rcond ); -lapack_int LAPACKE_cgtcon( char norm, lapack_int n, - const lapack_complex_float* dl, - const lapack_complex_float* d, - const lapack_complex_float* du, - const lapack_complex_float* du2, - const lapack_int* ipiv, float anorm, float* rcond ); -lapack_int LAPACKE_zgtcon( char norm, lapack_int n, - const lapack_complex_double* dl, - const lapack_complex_double* d, - const lapack_complex_double* du, - const lapack_complex_double* du2, - const lapack_int* ipiv, double anorm, - double* rcond ); - -lapack_int LAPACKE_sgtrfs( int matrix_order, char trans, lapack_int n, - lapack_int nrhs, const float* dl, const float* d, - const float* du, const float* dlf, const float* df, - const float* duf, const float* du2, - const lapack_int* ipiv, const float* b, - lapack_int ldb, float* x, lapack_int ldx, - float* ferr, float* berr ); -lapack_int LAPACKE_dgtrfs( int matrix_order, char trans, lapack_int n, - lapack_int nrhs, const double* dl, const double* d, - const double* du, const double* dlf, - const double* df, const double* duf, - const double* du2, const lapack_int* ipiv, - const double* b, lapack_int ldb, double* x, - lapack_int ldx, double* ferr, double* berr ); -lapack_int LAPACKE_cgtrfs( int matrix_order, char trans, lapack_int n, - lapack_int nrhs, const lapack_complex_float* dl, - const lapack_complex_float* d, - const lapack_complex_float* du, - const lapack_complex_float* dlf, - const lapack_complex_float* df, - const lapack_complex_float* duf, - const lapack_complex_float* du2, - const lapack_int* ipiv, - const lapack_complex_float* b, lapack_int ldb, - lapack_complex_float* x, lapack_int ldx, float* ferr, - float* berr ); -lapack_int LAPACKE_zgtrfs( int matrix_order, char trans, lapack_int n, - lapack_int nrhs, const lapack_complex_double* dl, - const lapack_complex_double* d, - const lapack_complex_double* du, - const lapack_complex_double* dlf, - const lapack_complex_double* df, - const lapack_complex_double* duf, - const lapack_complex_double* du2, - const lapack_int* ipiv, - const lapack_complex_double* b, lapack_int ldb, - lapack_complex_double* x, lapack_int ldx, - double* ferr, double* berr ); - -lapack_int LAPACKE_sgtsv( int matrix_order, lapack_int n, lapack_int nrhs, - float* dl, float* d, float* du, float* b, - lapack_int ldb ); -lapack_int LAPACKE_dgtsv( int matrix_order, lapack_int n, lapack_int nrhs, - double* dl, double* d, double* du, double* b, - lapack_int ldb ); -lapack_int LAPACKE_cgtsv( int matrix_order, lapack_int n, lapack_int nrhs, - lapack_complex_float* dl, lapack_complex_float* d, - lapack_complex_float* du, lapack_complex_float* b, - lapack_int ldb ); -lapack_int LAPACKE_zgtsv( int matrix_order, lapack_int n, lapack_int nrhs, - lapack_complex_double* dl, lapack_complex_double* d, - lapack_complex_double* du, lapack_complex_double* b, - lapack_int ldb ); - -lapack_int LAPACKE_sgtsvx( int matrix_order, char fact, char trans, - lapack_int n, lapack_int nrhs, const float* dl, - const float* d, const float* du, float* dlf, - float* df, float* duf, float* du2, lapack_int* ipiv, - const float* b, lapack_int ldb, float* x, - lapack_int ldx, float* rcond, float* ferr, - float* berr ); -lapack_int LAPACKE_dgtsvx( int matrix_order, char fact, char trans, - lapack_int n, lapack_int nrhs, const double* dl, - const double* d, const double* du, double* dlf, - double* df, double* duf, double* du2, - lapack_int* ipiv, const double* b, lapack_int ldb, - double* x, lapack_int ldx, double* rcond, - double* ferr, double* berr ); -lapack_int LAPACKE_cgtsvx( int matrix_order, char fact, char trans, - lapack_int n, lapack_int nrhs, - const lapack_complex_float* dl, - const lapack_complex_float* d, - const lapack_complex_float* du, - lapack_complex_float* dlf, lapack_complex_float* df, - lapack_complex_float* duf, lapack_complex_float* du2, - lapack_int* ipiv, const lapack_complex_float* b, - lapack_int ldb, lapack_complex_float* x, - lapack_int ldx, float* rcond, float* ferr, - float* berr ); -lapack_int LAPACKE_zgtsvx( int matrix_order, char fact, char trans, - lapack_int n, lapack_int nrhs, - const lapack_complex_double* dl, - const lapack_complex_double* d, - const lapack_complex_double* du, - lapack_complex_double* dlf, - lapack_complex_double* df, - lapack_complex_double* duf, - lapack_complex_double* du2, lapack_int* ipiv, - const lapack_complex_double* b, lapack_int ldb, - lapack_complex_double* x, lapack_int ldx, - double* rcond, double* ferr, double* berr ); - -lapack_int LAPACKE_sgttrf( lapack_int n, float* dl, float* d, float* du, - float* du2, lapack_int* ipiv ); -lapack_int LAPACKE_dgttrf( lapack_int n, double* dl, double* d, double* du, - double* du2, lapack_int* ipiv ); -lapack_int LAPACKE_cgttrf( lapack_int n, lapack_complex_float* dl, - lapack_complex_float* d, lapack_complex_float* du, - lapack_complex_float* du2, lapack_int* ipiv ); -lapack_int LAPACKE_zgttrf( lapack_int n, lapack_complex_double* dl, - lapack_complex_double* d, lapack_complex_double* du, - lapack_complex_double* du2, lapack_int* ipiv ); - -lapack_int LAPACKE_sgttrs( int matrix_order, char trans, lapack_int n, - lapack_int nrhs, const float* dl, const float* d, - const float* du, const float* du2, - const lapack_int* ipiv, float* b, lapack_int ldb ); -lapack_int LAPACKE_dgttrs( int matrix_order, char trans, lapack_int n, - lapack_int nrhs, const double* dl, const double* d, - const double* du, const double* du2, - const lapack_int* ipiv, double* b, lapack_int ldb ); -lapack_int LAPACKE_cgttrs( int matrix_order, char trans, lapack_int n, - lapack_int nrhs, const lapack_complex_float* dl, - const lapack_complex_float* d, - const lapack_complex_float* du, - const lapack_complex_float* du2, - const lapack_int* ipiv, lapack_complex_float* b, - lapack_int ldb ); -lapack_int LAPACKE_zgttrs( int matrix_order, char trans, lapack_int n, - lapack_int nrhs, const lapack_complex_double* dl, - const lapack_complex_double* d, - const lapack_complex_double* du, - const lapack_complex_double* du2, - const lapack_int* ipiv, lapack_complex_double* b, - lapack_int ldb ); - -lapack_int LAPACKE_chbev( int matrix_order, char jobz, char uplo, lapack_int n, - lapack_int kd, lapack_complex_float* ab, - lapack_int ldab, float* w, lapack_complex_float* z, - lapack_int ldz ); -lapack_int LAPACKE_zhbev( int matrix_order, char jobz, char uplo, lapack_int n, - lapack_int kd, lapack_complex_double* ab, - lapack_int ldab, double* w, lapack_complex_double* z, - lapack_int ldz ); - -lapack_int LAPACKE_chbevd( int matrix_order, char jobz, char uplo, lapack_int n, - lapack_int kd, lapack_complex_float* ab, - lapack_int ldab, float* w, lapack_complex_float* z, - lapack_int ldz ); -lapack_int LAPACKE_zhbevd( int matrix_order, char jobz, char uplo, lapack_int n, - lapack_int kd, lapack_complex_double* ab, - lapack_int ldab, double* w, lapack_complex_double* z, - lapack_int ldz ); - -lapack_int LAPACKE_chbevx( int matrix_order, char jobz, char range, char uplo, - lapack_int n, lapack_int kd, - lapack_complex_float* ab, lapack_int ldab, - lapack_complex_float* q, lapack_int ldq, float vl, - float vu, lapack_int il, lapack_int iu, float abstol, - lapack_int* m, float* w, lapack_complex_float* z, - lapack_int ldz, lapack_int* ifail ); -lapack_int LAPACKE_zhbevx( int matrix_order, char jobz, char range, char uplo, - lapack_int n, lapack_int kd, - lapack_complex_double* ab, lapack_int ldab, - lapack_complex_double* q, lapack_int ldq, double vl, - double vu, lapack_int il, lapack_int iu, - double abstol, lapack_int* m, double* w, - lapack_complex_double* z, lapack_int ldz, - lapack_int* ifail ); - -lapack_int LAPACKE_chbgst( int matrix_order, char vect, char uplo, lapack_int n, - lapack_int ka, lapack_int kb, - lapack_complex_float* ab, lapack_int ldab, - const lapack_complex_float* bb, lapack_int ldbb, - lapack_complex_float* x, lapack_int ldx ); -lapack_int LAPACKE_zhbgst( int matrix_order, char vect, char uplo, lapack_int n, - lapack_int ka, lapack_int kb, - lapack_complex_double* ab, lapack_int ldab, - const lapack_complex_double* bb, lapack_int ldbb, - lapack_complex_double* x, lapack_int ldx ); - -lapack_int LAPACKE_chbgv( int matrix_order, char jobz, char uplo, lapack_int n, - lapack_int ka, lapack_int kb, - lapack_complex_float* ab, lapack_int ldab, - lapack_complex_float* bb, lapack_int ldbb, float* w, - lapack_complex_float* z, lapack_int ldz ); -lapack_int LAPACKE_zhbgv( int matrix_order, char jobz, char uplo, lapack_int n, - lapack_int ka, lapack_int kb, - lapack_complex_double* ab, lapack_int ldab, - lapack_complex_double* bb, lapack_int ldbb, double* w, - lapack_complex_double* z, lapack_int ldz ); - -lapack_int LAPACKE_chbgvd( int matrix_order, char jobz, char uplo, lapack_int n, - lapack_int ka, lapack_int kb, - lapack_complex_float* ab, lapack_int ldab, - lapack_complex_float* bb, lapack_int ldbb, float* w, - lapack_complex_float* z, lapack_int ldz ); -lapack_int LAPACKE_zhbgvd( int matrix_order, char jobz, char uplo, lapack_int n, - lapack_int ka, lapack_int kb, - lapack_complex_double* ab, lapack_int ldab, - lapack_complex_double* bb, lapack_int ldbb, - double* w, lapack_complex_double* z, - lapack_int ldz ); - -lapack_int LAPACKE_chbgvx( int matrix_order, char jobz, char range, char uplo, - lapack_int n, lapack_int ka, lapack_int kb, - lapack_complex_float* ab, lapack_int ldab, - lapack_complex_float* bb, lapack_int ldbb, - lapack_complex_float* q, lapack_int ldq, float vl, - float vu, lapack_int il, lapack_int iu, float abstol, - lapack_int* m, float* w, lapack_complex_float* z, - lapack_int ldz, lapack_int* ifail ); -lapack_int LAPACKE_zhbgvx( int matrix_order, char jobz, char range, char uplo, - lapack_int n, lapack_int ka, lapack_int kb, - lapack_complex_double* ab, lapack_int ldab, - lapack_complex_double* bb, lapack_int ldbb, - lapack_complex_double* q, lapack_int ldq, double vl, - double vu, lapack_int il, lapack_int iu, - double abstol, lapack_int* m, double* w, - lapack_complex_double* z, lapack_int ldz, - lapack_int* ifail ); - -lapack_int LAPACKE_chbtrd( int matrix_order, char vect, char uplo, lapack_int n, - lapack_int kd, lapack_complex_float* ab, - lapack_int ldab, float* d, float* e, - lapack_complex_float* q, lapack_int ldq ); -lapack_int LAPACKE_zhbtrd( int matrix_order, char vect, char uplo, lapack_int n, - lapack_int kd, lapack_complex_double* ab, - lapack_int ldab, double* d, double* e, - lapack_complex_double* q, lapack_int ldq ); - -lapack_int LAPACKE_checon( int matrix_order, char uplo, lapack_int n, - const lapack_complex_float* a, lapack_int lda, - const lapack_int* ipiv, float anorm, float* rcond ); -lapack_int LAPACKE_zhecon( int matrix_order, char uplo, lapack_int n, - const lapack_complex_double* a, lapack_int lda, - const lapack_int* ipiv, double anorm, - double* rcond ); - -lapack_int LAPACKE_cheequb( int matrix_order, char uplo, lapack_int n, - const lapack_complex_float* a, lapack_int lda, - float* s, float* scond, float* amax ); -lapack_int LAPACKE_zheequb( int matrix_order, char uplo, lapack_int n, - const lapack_complex_double* a, lapack_int lda, - double* s, double* scond, double* amax ); - -lapack_int LAPACKE_cheev( int matrix_order, char jobz, char uplo, lapack_int n, - lapack_complex_float* a, lapack_int lda, float* w ); -lapack_int LAPACKE_zheev( int matrix_order, char jobz, char uplo, lapack_int n, - lapack_complex_double* a, lapack_int lda, double* w ); - -lapack_int LAPACKE_cheevd( int matrix_order, char jobz, char uplo, lapack_int n, - lapack_complex_float* a, lapack_int lda, float* w ); -lapack_int LAPACKE_zheevd( int matrix_order, char jobz, char uplo, lapack_int n, - lapack_complex_double* a, lapack_int lda, - double* w ); - -lapack_int LAPACKE_cheevr( int matrix_order, char jobz, char range, char uplo, - lapack_int n, lapack_complex_float* a, - lapack_int lda, float vl, float vu, lapack_int il, - lapack_int iu, float abstol, lapack_int* m, float* w, - lapack_complex_float* z, lapack_int ldz, - lapack_int* isuppz ); -lapack_int LAPACKE_zheevr( int matrix_order, char jobz, char range, char uplo, - lapack_int n, lapack_complex_double* a, - lapack_int lda, double vl, double vu, lapack_int il, - lapack_int iu, double abstol, lapack_int* m, - double* w, lapack_complex_double* z, lapack_int ldz, - lapack_int* isuppz ); - -lapack_int LAPACKE_cheevx( int matrix_order, char jobz, char range, char uplo, - lapack_int n, lapack_complex_float* a, - lapack_int lda, float vl, float vu, lapack_int il, - lapack_int iu, float abstol, lapack_int* m, float* w, - lapack_complex_float* z, lapack_int ldz, - lapack_int* ifail ); -lapack_int LAPACKE_zheevx( int matrix_order, char jobz, char range, char uplo, - lapack_int n, lapack_complex_double* a, - lapack_int lda, double vl, double vu, lapack_int il, - lapack_int iu, double abstol, lapack_int* m, - double* w, lapack_complex_double* z, lapack_int ldz, - lapack_int* ifail ); - -lapack_int LAPACKE_chegst( int matrix_order, lapack_int itype, char uplo, - lapack_int n, lapack_complex_float* a, - lapack_int lda, const lapack_complex_float* b, - lapack_int ldb ); -lapack_int LAPACKE_zhegst( int matrix_order, lapack_int itype, char uplo, - lapack_int n, lapack_complex_double* a, - lapack_int lda, const lapack_complex_double* b, - lapack_int ldb ); - -lapack_int LAPACKE_chegv( int matrix_order, lapack_int itype, char jobz, - char uplo, lapack_int n, lapack_complex_float* a, - lapack_int lda, lapack_complex_float* b, - lapack_int ldb, float* w ); -lapack_int LAPACKE_zhegv( int matrix_order, lapack_int itype, char jobz, - char uplo, lapack_int n, lapack_complex_double* a, - lapack_int lda, lapack_complex_double* b, - lapack_int ldb, double* w ); - -lapack_int LAPACKE_chegvd( int matrix_order, lapack_int itype, char jobz, - char uplo, lapack_int n, lapack_complex_float* a, - lapack_int lda, lapack_complex_float* b, - lapack_int ldb, float* w ); -lapack_int LAPACKE_zhegvd( int matrix_order, lapack_int itype, char jobz, - char uplo, lapack_int n, lapack_complex_double* a, - lapack_int lda, lapack_complex_double* b, - lapack_int ldb, double* w ); - -lapack_int LAPACKE_chegvx( int matrix_order, lapack_int itype, char jobz, - char range, char uplo, lapack_int n, - lapack_complex_float* a, lapack_int lda, - lapack_complex_float* b, lapack_int ldb, float vl, - float vu, lapack_int il, lapack_int iu, float abstol, - lapack_int* m, float* w, lapack_complex_float* z, - lapack_int ldz, lapack_int* ifail ); -lapack_int LAPACKE_zhegvx( int matrix_order, lapack_int itype, char jobz, - char range, char uplo, lapack_int n, - lapack_complex_double* a, lapack_int lda, - lapack_complex_double* b, lapack_int ldb, double vl, - double vu, lapack_int il, lapack_int iu, - double abstol, lapack_int* m, double* w, - lapack_complex_double* z, lapack_int ldz, - lapack_int* ifail ); - -lapack_int LAPACKE_cherfs( int matrix_order, char uplo, lapack_int n, - lapack_int nrhs, const lapack_complex_float* a, - lapack_int lda, const lapack_complex_float* af, - lapack_int ldaf, const lapack_int* ipiv, - const lapack_complex_float* b, lapack_int ldb, - lapack_complex_float* x, lapack_int ldx, float* ferr, - float* berr ); -lapack_int LAPACKE_zherfs( int matrix_order, char uplo, lapack_int n, - lapack_int nrhs, const lapack_complex_double* a, - lapack_int lda, const lapack_complex_double* af, - lapack_int ldaf, const lapack_int* ipiv, - const lapack_complex_double* b, lapack_int ldb, - lapack_complex_double* x, lapack_int ldx, - double* ferr, double* berr ); - -lapack_int LAPACKE_cherfsx( int matrix_order, char uplo, char equed, - lapack_int n, lapack_int nrhs, - const lapack_complex_float* a, lapack_int lda, - const lapack_complex_float* af, lapack_int ldaf, - const lapack_int* ipiv, const float* s, - const lapack_complex_float* b, lapack_int ldb, - lapack_complex_float* x, lapack_int ldx, - float* rcond, float* berr, lapack_int n_err_bnds, - float* err_bnds_norm, float* err_bnds_comp, - lapack_int nparams, float* params ); -lapack_int LAPACKE_zherfsx( int matrix_order, char uplo, char equed, - lapack_int n, lapack_int nrhs, - const lapack_complex_double* a, lapack_int lda, - const lapack_complex_double* af, lapack_int ldaf, - const lapack_int* ipiv, const double* s, - const lapack_complex_double* b, lapack_int ldb, - lapack_complex_double* x, lapack_int ldx, - double* rcond, double* berr, lapack_int n_err_bnds, - double* err_bnds_norm, double* err_bnds_comp, - lapack_int nparams, double* params ); - -lapack_int LAPACKE_chesv( int matrix_order, char uplo, lapack_int n, - lapack_int nrhs, lapack_complex_float* a, - lapack_int lda, lapack_int* ipiv, - lapack_complex_float* b, lapack_int ldb ); -lapack_int LAPACKE_zhesv( int matrix_order, char uplo, lapack_int n, - lapack_int nrhs, lapack_complex_double* a, - lapack_int lda, lapack_int* ipiv, - lapack_complex_double* b, lapack_int ldb ); - -lapack_int LAPACKE_chesvx( int matrix_order, char fact, char uplo, lapack_int n, - lapack_int nrhs, const lapack_complex_float* a, - lapack_int lda, lapack_complex_float* af, - lapack_int ldaf, lapack_int* ipiv, - const lapack_complex_float* b, lapack_int ldb, - lapack_complex_float* x, lapack_int ldx, - float* rcond, float* ferr, float* berr ); -lapack_int LAPACKE_zhesvx( int matrix_order, char fact, char uplo, lapack_int n, - lapack_int nrhs, const lapack_complex_double* a, - lapack_int lda, lapack_complex_double* af, - lapack_int ldaf, lapack_int* ipiv, - const lapack_complex_double* b, lapack_int ldb, - lapack_complex_double* x, lapack_int ldx, - double* rcond, double* ferr, double* berr ); - -lapack_int LAPACKE_chesvxx( int matrix_order, char fact, char uplo, - lapack_int n, lapack_int nrhs, - lapack_complex_float* a, lapack_int lda, - lapack_complex_float* af, lapack_int ldaf, - lapack_int* ipiv, char* equed, float* s, - lapack_complex_float* b, lapack_int ldb, - lapack_complex_float* x, lapack_int ldx, - float* rcond, float* rpvgrw, float* berr, - lapack_int n_err_bnds, float* err_bnds_norm, - float* err_bnds_comp, lapack_int nparams, - float* params ); -lapack_int LAPACKE_zhesvxx( int matrix_order, char fact, char uplo, - lapack_int n, lapack_int nrhs, - lapack_complex_double* a, lapack_int lda, - lapack_complex_double* af, lapack_int ldaf, - lapack_int* ipiv, char* equed, double* s, - lapack_complex_double* b, lapack_int ldb, - lapack_complex_double* x, lapack_int ldx, - double* rcond, double* rpvgrw, double* berr, - lapack_int n_err_bnds, double* err_bnds_norm, - double* err_bnds_comp, lapack_int nparams, - double* params ); - -lapack_int LAPACKE_chetrd( int matrix_order, char uplo, lapack_int n, - lapack_complex_float* a, lapack_int lda, float* d, - float* e, lapack_complex_float* tau ); -lapack_int LAPACKE_zhetrd( int matrix_order, char uplo, lapack_int n, - lapack_complex_double* a, lapack_int lda, double* d, - double* e, lapack_complex_double* tau ); - -lapack_int LAPACKE_chetrf( int matrix_order, char uplo, lapack_int n, - lapack_complex_float* a, lapack_int lda, - lapack_int* ipiv ); -lapack_int LAPACKE_zhetrf( int matrix_order, char uplo, lapack_int n, - lapack_complex_double* a, lapack_int lda, - lapack_int* ipiv ); - -lapack_int LAPACKE_chetri( int matrix_order, char uplo, lapack_int n, - lapack_complex_float* a, lapack_int lda, - const lapack_int* ipiv ); -lapack_int LAPACKE_zhetri( int matrix_order, char uplo, lapack_int n, - lapack_complex_double* a, lapack_int lda, - const lapack_int* ipiv ); - -lapack_int LAPACKE_chetrs( int matrix_order, char uplo, lapack_int n, - lapack_int nrhs, const lapack_complex_float* a, - lapack_int lda, const lapack_int* ipiv, - lapack_complex_float* b, lapack_int ldb ); -lapack_int LAPACKE_zhetrs( int matrix_order, char uplo, lapack_int n, - lapack_int nrhs, const lapack_complex_double* a, - lapack_int lda, const lapack_int* ipiv, - lapack_complex_double* b, lapack_int ldb ); - -lapack_int LAPACKE_chfrk( int matrix_order, char transr, char uplo, char trans, - lapack_int n, lapack_int k, float alpha, - const lapack_complex_float* a, lapack_int lda, - float beta, lapack_complex_float* c ); -lapack_int LAPACKE_zhfrk( int matrix_order, char transr, char uplo, char trans, - lapack_int n, lapack_int k, double alpha, - const lapack_complex_double* a, lapack_int lda, - double beta, lapack_complex_double* c ); - -lapack_int LAPACKE_shgeqz( int matrix_order, char job, char compq, char compz, - lapack_int n, lapack_int ilo, lapack_int ihi, - float* h, lapack_int ldh, float* t, lapack_int ldt, - float* alphar, float* alphai, float* beta, float* q, - lapack_int ldq, float* z, lapack_int ldz ); -lapack_int LAPACKE_dhgeqz( int matrix_order, char job, char compq, char compz, - lapack_int n, lapack_int ilo, lapack_int ihi, - double* h, lapack_int ldh, double* t, lapack_int ldt, - double* alphar, double* alphai, double* beta, - double* q, lapack_int ldq, double* z, - lapack_int ldz ); -lapack_int LAPACKE_chgeqz( int matrix_order, char job, char compq, char compz, - lapack_int n, lapack_int ilo, lapack_int ihi, - lapack_complex_float* h, lapack_int ldh, - lapack_complex_float* t, lapack_int ldt, - lapack_complex_float* alpha, - lapack_complex_float* beta, lapack_complex_float* q, - lapack_int ldq, lapack_complex_float* z, - lapack_int ldz ); -lapack_int LAPACKE_zhgeqz( int matrix_order, char job, char compq, char compz, - lapack_int n, lapack_int ilo, lapack_int ihi, - lapack_complex_double* h, lapack_int ldh, - lapack_complex_double* t, lapack_int ldt, - lapack_complex_double* alpha, - lapack_complex_double* beta, - lapack_complex_double* q, lapack_int ldq, - lapack_complex_double* z, lapack_int ldz ); - -lapack_int LAPACKE_chpcon( int matrix_order, char uplo, lapack_int n, - const lapack_complex_float* ap, - const lapack_int* ipiv, float anorm, float* rcond ); -lapack_int LAPACKE_zhpcon( int matrix_order, char uplo, lapack_int n, - const lapack_complex_double* ap, - const lapack_int* ipiv, double anorm, - double* rcond ); - -lapack_int LAPACKE_chpev( int matrix_order, char jobz, char uplo, lapack_int n, - lapack_complex_float* ap, float* w, - lapack_complex_float* z, lapack_int ldz ); -lapack_int LAPACKE_zhpev( int matrix_order, char jobz, char uplo, lapack_int n, - lapack_complex_double* ap, double* w, - lapack_complex_double* z, lapack_int ldz ); - -lapack_int LAPACKE_chpevd( int matrix_order, char jobz, char uplo, lapack_int n, - lapack_complex_float* ap, float* w, - lapack_complex_float* z, lapack_int ldz ); -lapack_int LAPACKE_zhpevd( int matrix_order, char jobz, char uplo, lapack_int n, - lapack_complex_double* ap, double* w, - lapack_complex_double* z, lapack_int ldz ); - -lapack_int LAPACKE_chpevx( int matrix_order, char jobz, char range, char uplo, - lapack_int n, lapack_complex_float* ap, float vl, - float vu, lapack_int il, lapack_int iu, float abstol, - lapack_int* m, float* w, lapack_complex_float* z, - lapack_int ldz, lapack_int* ifail ); -lapack_int LAPACKE_zhpevx( int matrix_order, char jobz, char range, char uplo, - lapack_int n, lapack_complex_double* ap, double vl, - double vu, lapack_int il, lapack_int iu, - double abstol, lapack_int* m, double* w, - lapack_complex_double* z, lapack_int ldz, - lapack_int* ifail ); - -lapack_int LAPACKE_chpgst( int matrix_order, lapack_int itype, char uplo, - lapack_int n, lapack_complex_float* ap, - const lapack_complex_float* bp ); -lapack_int LAPACKE_zhpgst( int matrix_order, lapack_int itype, char uplo, - lapack_int n, lapack_complex_double* ap, - const lapack_complex_double* bp ); - -lapack_int LAPACKE_chpgv( int matrix_order, lapack_int itype, char jobz, - char uplo, lapack_int n, lapack_complex_float* ap, - lapack_complex_float* bp, float* w, - lapack_complex_float* z, lapack_int ldz ); -lapack_int LAPACKE_zhpgv( int matrix_order, lapack_int itype, char jobz, - char uplo, lapack_int n, lapack_complex_double* ap, - lapack_complex_double* bp, double* w, - lapack_complex_double* z, lapack_int ldz ); - -lapack_int LAPACKE_chpgvd( int matrix_order, lapack_int itype, char jobz, - char uplo, lapack_int n, lapack_complex_float* ap, - lapack_complex_float* bp, float* w, - lapack_complex_float* z, lapack_int ldz ); -lapack_int LAPACKE_zhpgvd( int matrix_order, lapack_int itype, char jobz, - char uplo, lapack_int n, lapack_complex_double* ap, - lapack_complex_double* bp, double* w, - lapack_complex_double* z, lapack_int ldz ); - -lapack_int LAPACKE_chpgvx( int matrix_order, lapack_int itype, char jobz, - char range, char uplo, lapack_int n, - lapack_complex_float* ap, lapack_complex_float* bp, - float vl, float vu, lapack_int il, lapack_int iu, - float abstol, lapack_int* m, float* w, - lapack_complex_float* z, lapack_int ldz, - lapack_int* ifail ); -lapack_int LAPACKE_zhpgvx( int matrix_order, lapack_int itype, char jobz, - char range, char uplo, lapack_int n, - lapack_complex_double* ap, lapack_complex_double* bp, - double vl, double vu, lapack_int il, lapack_int iu, - double abstol, lapack_int* m, double* w, - lapack_complex_double* z, lapack_int ldz, - lapack_int* ifail ); - -lapack_int LAPACKE_chprfs( int matrix_order, char uplo, lapack_int n, - lapack_int nrhs, const lapack_complex_float* ap, - const lapack_complex_float* afp, - const lapack_int* ipiv, - const lapack_complex_float* b, lapack_int ldb, - lapack_complex_float* x, lapack_int ldx, float* ferr, - float* berr ); -lapack_int LAPACKE_zhprfs( int matrix_order, char uplo, lapack_int n, - lapack_int nrhs, const lapack_complex_double* ap, - const lapack_complex_double* afp, - const lapack_int* ipiv, - const lapack_complex_double* b, lapack_int ldb, - lapack_complex_double* x, lapack_int ldx, - double* ferr, double* berr ); - -lapack_int LAPACKE_chpsv( int matrix_order, char uplo, lapack_int n, - lapack_int nrhs, lapack_complex_float* ap, - lapack_int* ipiv, lapack_complex_float* b, - lapack_int ldb ); -lapack_int LAPACKE_zhpsv( int matrix_order, char uplo, lapack_int n, - lapack_int nrhs, lapack_complex_double* ap, - lapack_int* ipiv, lapack_complex_double* b, - lapack_int ldb ); - -lapack_int LAPACKE_chpsvx( int matrix_order, char fact, char uplo, lapack_int n, - lapack_int nrhs, const lapack_complex_float* ap, - lapack_complex_float* afp, lapack_int* ipiv, - const lapack_complex_float* b, lapack_int ldb, - lapack_complex_float* x, lapack_int ldx, - float* rcond, float* ferr, float* berr ); -lapack_int LAPACKE_zhpsvx( int matrix_order, char fact, char uplo, lapack_int n, - lapack_int nrhs, const lapack_complex_double* ap, - lapack_complex_double* afp, lapack_int* ipiv, - const lapack_complex_double* b, lapack_int ldb, - lapack_complex_double* x, lapack_int ldx, - double* rcond, double* ferr, double* berr ); - -lapack_int LAPACKE_chptrd( int matrix_order, char uplo, lapack_int n, - lapack_complex_float* ap, float* d, float* e, - lapack_complex_float* tau ); -lapack_int LAPACKE_zhptrd( int matrix_order, char uplo, lapack_int n, - lapack_complex_double* ap, double* d, double* e, - lapack_complex_double* tau ); - -lapack_int LAPACKE_chptrf( int matrix_order, char uplo, lapack_int n, - lapack_complex_float* ap, lapack_int* ipiv ); -lapack_int LAPACKE_zhptrf( int matrix_order, char uplo, lapack_int n, - lapack_complex_double* ap, lapack_int* ipiv ); - -lapack_int LAPACKE_chptri( int matrix_order, char uplo, lapack_int n, - lapack_complex_float* ap, const lapack_int* ipiv ); -lapack_int LAPACKE_zhptri( int matrix_order, char uplo, lapack_int n, - lapack_complex_double* ap, const lapack_int* ipiv ); - -lapack_int LAPACKE_chptrs( int matrix_order, char uplo, lapack_int n, - lapack_int nrhs, const lapack_complex_float* ap, - const lapack_int* ipiv, lapack_complex_float* b, - lapack_int ldb ); -lapack_int LAPACKE_zhptrs( int matrix_order, char uplo, lapack_int n, - lapack_int nrhs, const lapack_complex_double* ap, - const lapack_int* ipiv, lapack_complex_double* b, - lapack_int ldb ); - -lapack_int LAPACKE_shsein( int matrix_order, char job, char eigsrc, char initv, - lapack_logical* select, lapack_int n, const float* h, - lapack_int ldh, float* wr, const float* wi, - float* vl, lapack_int ldvl, float* vr, - lapack_int ldvr, lapack_int mm, lapack_int* m, - lapack_int* ifaill, lapack_int* ifailr ); -lapack_int LAPACKE_dhsein( int matrix_order, char job, char eigsrc, char initv, - lapack_logical* select, lapack_int n, - const double* h, lapack_int ldh, double* wr, - const double* wi, double* vl, lapack_int ldvl, - double* vr, lapack_int ldvr, lapack_int mm, - lapack_int* m, lapack_int* ifaill, - lapack_int* ifailr ); -lapack_int LAPACKE_chsein( int matrix_order, char job, char eigsrc, char initv, - const lapack_logical* select, lapack_int n, - const lapack_complex_float* h, lapack_int ldh, - lapack_complex_float* w, lapack_complex_float* vl, - lapack_int ldvl, lapack_complex_float* vr, - lapack_int ldvr, lapack_int mm, lapack_int* m, - lapack_int* ifaill, lapack_int* ifailr ); -lapack_int LAPACKE_zhsein( int matrix_order, char job, char eigsrc, char initv, - const lapack_logical* select, lapack_int n, - const lapack_complex_double* h, lapack_int ldh, - lapack_complex_double* w, lapack_complex_double* vl, - lapack_int ldvl, lapack_complex_double* vr, - lapack_int ldvr, lapack_int mm, lapack_int* m, - lapack_int* ifaill, lapack_int* ifailr ); - -lapack_int LAPACKE_shseqr( int matrix_order, char job, char compz, lapack_int n, - lapack_int ilo, lapack_int ihi, float* h, - lapack_int ldh, float* wr, float* wi, float* z, - lapack_int ldz ); -lapack_int LAPACKE_dhseqr( int matrix_order, char job, char compz, lapack_int n, - lapack_int ilo, lapack_int ihi, double* h, - lapack_int ldh, double* wr, double* wi, double* z, - lapack_int ldz ); -lapack_int LAPACKE_chseqr( int matrix_order, char job, char compz, lapack_int n, - lapack_int ilo, lapack_int ihi, - lapack_complex_float* h, lapack_int ldh, - lapack_complex_float* w, lapack_complex_float* z, - lapack_int ldz ); -lapack_int LAPACKE_zhseqr( int matrix_order, char job, char compz, lapack_int n, - lapack_int ilo, lapack_int ihi, - lapack_complex_double* h, lapack_int ldh, - lapack_complex_double* w, lapack_complex_double* z, - lapack_int ldz ); - -lapack_int LAPACKE_clacgv( lapack_int n, lapack_complex_float* x, - lapack_int incx ); -lapack_int LAPACKE_zlacgv( lapack_int n, lapack_complex_double* x, - lapack_int incx ); - -lapack_int LAPACKE_slacpy( int matrix_order, char uplo, lapack_int m, - lapack_int n, const float* a, lapack_int lda, float* b, - lapack_int ldb ); -lapack_int LAPACKE_dlacpy( int matrix_order, char uplo, lapack_int m, - lapack_int n, const double* a, lapack_int lda, double* b, - lapack_int ldb ); -lapack_int LAPACKE_clacpy( int matrix_order, char uplo, lapack_int m, - lapack_int n, const lapack_complex_float* a, - lapack_int lda, lapack_complex_float* b, - lapack_int ldb ); -lapack_int LAPACKE_zlacpy( int matrix_order, char uplo, lapack_int m, - lapack_int n, const lapack_complex_double* a, - lapack_int lda, lapack_complex_double* b, - lapack_int ldb ); - -lapack_int LAPACKE_zlag2c( int matrix_order, lapack_int m, lapack_int n, - const lapack_complex_double* a, lapack_int lda, - lapack_complex_float* sa, lapack_int ldsa ); - -lapack_int LAPACKE_slag2d( int matrix_order, lapack_int m, lapack_int n, - const float* sa, lapack_int ldsa, double* a, - lapack_int lda ); - -lapack_int LAPACKE_dlag2s( int matrix_order, lapack_int m, lapack_int n, - const double* a, lapack_int lda, float* sa, - lapack_int ldsa ); - -lapack_int LAPACKE_clag2z( int matrix_order, lapack_int m, lapack_int n, - const lapack_complex_float* sa, lapack_int ldsa, - lapack_complex_double* a, lapack_int lda ); - -lapack_int LAPACKE_slagge( int matrix_order, lapack_int m, lapack_int n, - lapack_int kl, lapack_int ku, const float* d, - float* a, lapack_int lda, lapack_int* iseed ); -lapack_int LAPACKE_dlagge( int matrix_order, lapack_int m, lapack_int n, - lapack_int kl, lapack_int ku, const double* d, - double* a, lapack_int lda, lapack_int* iseed ); -lapack_int LAPACKE_clagge( int matrix_order, lapack_int m, lapack_int n, - lapack_int kl, lapack_int ku, const float* d, - lapack_complex_float* a, lapack_int lda, - lapack_int* iseed ); -lapack_int LAPACKE_zlagge( int matrix_order, lapack_int m, lapack_int n, - lapack_int kl, lapack_int ku, const double* d, - lapack_complex_double* a, lapack_int lda, - lapack_int* iseed ); - -float LAPACKE_slamch( char cmach ); -double LAPACKE_dlamch( char cmach ); - -float LAPACKE_slange( int matrix_order, char norm, lapack_int m, - lapack_int n, const float* a, lapack_int lda ); -double LAPACKE_dlange( int matrix_order, char norm, lapack_int m, - lapack_int n, const double* a, lapack_int lda ); -float LAPACKE_clange( int matrix_order, char norm, lapack_int m, - lapack_int n, const lapack_complex_float* a, - lapack_int lda ); -double LAPACKE_zlange( int matrix_order, char norm, lapack_int m, - lapack_int n, const lapack_complex_double* a, - lapack_int lda ); - -float LAPACKE_clanhe( int matrix_order, char norm, char uplo, lapack_int n, - const lapack_complex_float* a, lapack_int lda ); -double LAPACKE_zlanhe( int matrix_order, char norm, char uplo, lapack_int n, - const lapack_complex_double* a, lapack_int lda ); - -float LAPACKE_slansy( int matrix_order, char norm, char uplo, lapack_int n, - const float* a, lapack_int lda ); -double LAPACKE_dlansy( int matrix_order, char norm, char uplo, lapack_int n, - const double* a, lapack_int lda ); -float LAPACKE_clansy( int matrix_order, char norm, char uplo, lapack_int n, - const lapack_complex_float* a, lapack_int lda ); -double LAPACKE_zlansy( int matrix_order, char norm, char uplo, lapack_int n, - const lapack_complex_double* a, lapack_int lda ); - -float LAPACKE_slantr( int matrix_order, char norm, char uplo, char diag, - lapack_int m, lapack_int n, const float* a, - lapack_int lda ); -double LAPACKE_dlantr( int matrix_order, char norm, char uplo, char diag, - lapack_int m, lapack_int n, const double* a, - lapack_int lda ); -float LAPACKE_clantr( int matrix_order, char norm, char uplo, char diag, - lapack_int m, lapack_int n, const lapack_complex_float* a, - lapack_int lda ); -double LAPACKE_zlantr( int matrix_order, char norm, char uplo, char diag, - lapack_int m, lapack_int n, const lapack_complex_double* a, - lapack_int lda ); - - -lapack_int LAPACKE_slarfb( int matrix_order, char side, char trans, char direct, - char storev, lapack_int m, lapack_int n, - lapack_int k, const float* v, lapack_int ldv, - const float* t, lapack_int ldt, float* c, - lapack_int ldc ); -lapack_int LAPACKE_dlarfb( int matrix_order, char side, char trans, char direct, - char storev, lapack_int m, lapack_int n, - lapack_int k, const double* v, lapack_int ldv, - const double* t, lapack_int ldt, double* c, - lapack_int ldc ); -lapack_int LAPACKE_clarfb( int matrix_order, char side, char trans, char direct, - char storev, lapack_int m, lapack_int n, - lapack_int k, const lapack_complex_float* v, - lapack_int ldv, const lapack_complex_float* t, - lapack_int ldt, lapack_complex_float* c, - lapack_int ldc ); -lapack_int LAPACKE_zlarfb( int matrix_order, char side, char trans, char direct, - char storev, lapack_int m, lapack_int n, - lapack_int k, const lapack_complex_double* v, - lapack_int ldv, const lapack_complex_double* t, - lapack_int ldt, lapack_complex_double* c, - lapack_int ldc ); - -lapack_int LAPACKE_slarfg( lapack_int n, float* alpha, float* x, - lapack_int incx, float* tau ); -lapack_int LAPACKE_dlarfg( lapack_int n, double* alpha, double* x, - lapack_int incx, double* tau ); -lapack_int LAPACKE_clarfg( lapack_int n, lapack_complex_float* alpha, - lapack_complex_float* x, lapack_int incx, - lapack_complex_float* tau ); -lapack_int LAPACKE_zlarfg( lapack_int n, lapack_complex_double* alpha, - lapack_complex_double* x, lapack_int incx, - lapack_complex_double* tau ); - -lapack_int LAPACKE_slarft( int matrix_order, char direct, char storev, - lapack_int n, lapack_int k, const float* v, - lapack_int ldv, const float* tau, float* t, - lapack_int ldt ); -lapack_int LAPACKE_dlarft( int matrix_order, char direct, char storev, - lapack_int n, lapack_int k, const double* v, - lapack_int ldv, const double* tau, double* t, - lapack_int ldt ); -lapack_int LAPACKE_clarft( int matrix_order, char direct, char storev, - lapack_int n, lapack_int k, - const lapack_complex_float* v, lapack_int ldv, - const lapack_complex_float* tau, - lapack_complex_float* t, lapack_int ldt ); -lapack_int LAPACKE_zlarft( int matrix_order, char direct, char storev, - lapack_int n, lapack_int k, - const lapack_complex_double* v, lapack_int ldv, - const lapack_complex_double* tau, - lapack_complex_double* t, lapack_int ldt ); - -lapack_int LAPACKE_slarfx( int matrix_order, char side, lapack_int m, - lapack_int n, const float* v, float tau, float* c, - lapack_int ldc, float* work ); -lapack_int LAPACKE_dlarfx( int matrix_order, char side, lapack_int m, - lapack_int n, const double* v, double tau, double* c, - lapack_int ldc, double* work ); -lapack_int LAPACKE_clarfx( int matrix_order, char side, lapack_int m, - lapack_int n, const lapack_complex_float* v, - lapack_complex_float tau, lapack_complex_float* c, - lapack_int ldc, lapack_complex_float* work ); -lapack_int LAPACKE_zlarfx( int matrix_order, char side, lapack_int m, - lapack_int n, const lapack_complex_double* v, - lapack_complex_double tau, lapack_complex_double* c, - lapack_int ldc, lapack_complex_double* work ); - -lapack_int LAPACKE_slarnv( lapack_int idist, lapack_int* iseed, lapack_int n, - float* x ); -lapack_int LAPACKE_dlarnv( lapack_int idist, lapack_int* iseed, lapack_int n, - double* x ); -lapack_int LAPACKE_clarnv( lapack_int idist, lapack_int* iseed, lapack_int n, - lapack_complex_float* x ); -lapack_int LAPACKE_zlarnv( lapack_int idist, lapack_int* iseed, lapack_int n, - lapack_complex_double* x ); - -lapack_int LAPACKE_slaset( int matrix_order, char uplo, lapack_int m, - lapack_int n, float alpha, float beta, float* a, - lapack_int lda ); -lapack_int LAPACKE_dlaset( int matrix_order, char uplo, lapack_int m, - lapack_int n, double alpha, double beta, double* a, - lapack_int lda ); -lapack_int LAPACKE_claset( int matrix_order, char uplo, lapack_int m, - lapack_int n, lapack_complex_float alpha, - lapack_complex_float beta, lapack_complex_float* a, - lapack_int lda ); -lapack_int LAPACKE_zlaset( int matrix_order, char uplo, lapack_int m, - lapack_int n, lapack_complex_double alpha, - lapack_complex_double beta, lapack_complex_double* a, - lapack_int lda ); - -lapack_int LAPACKE_slasrt( char id, lapack_int n, float* d ); -lapack_int LAPACKE_dlasrt( char id, lapack_int n, double* d ); - -lapack_int LAPACKE_slaswp( int matrix_order, lapack_int n, float* a, - lapack_int lda, lapack_int k1, lapack_int k2, - const lapack_int* ipiv, lapack_int incx ); -lapack_int LAPACKE_dlaswp( int matrix_order, lapack_int n, double* a, - lapack_int lda, lapack_int k1, lapack_int k2, - const lapack_int* ipiv, lapack_int incx ); -lapack_int LAPACKE_claswp( int matrix_order, lapack_int n, - lapack_complex_float* a, lapack_int lda, - lapack_int k1, lapack_int k2, const lapack_int* ipiv, - lapack_int incx ); -lapack_int LAPACKE_zlaswp( int matrix_order, lapack_int n, - lapack_complex_double* a, lapack_int lda, - lapack_int k1, lapack_int k2, const lapack_int* ipiv, - lapack_int incx ); - -lapack_int LAPACKE_slatms( int matrix_order, lapack_int m, lapack_int n, - char dist, lapack_int* iseed, char sym, float* d, - lapack_int mode, float cond, float dmax, - lapack_int kl, lapack_int ku, char pack, float* a, - lapack_int lda ); -lapack_int LAPACKE_dlatms( int matrix_order, lapack_int m, lapack_int n, - char dist, lapack_int* iseed, char sym, double* d, - lapack_int mode, double cond, double dmax, - lapack_int kl, lapack_int ku, char pack, double* a, - lapack_int lda ); -lapack_int LAPACKE_clatms( int matrix_order, lapack_int m, lapack_int n, - char dist, lapack_int* iseed, char sym, float* d, - lapack_int mode, float cond, float dmax, - lapack_int kl, lapack_int ku, char pack, - lapack_complex_float* a, lapack_int lda ); -lapack_int LAPACKE_zlatms( int matrix_order, lapack_int m, lapack_int n, - char dist, lapack_int* iseed, char sym, double* d, - lapack_int mode, double cond, double dmax, - lapack_int kl, lapack_int ku, char pack, - lapack_complex_double* a, lapack_int lda ); - -lapack_int LAPACKE_slauum( int matrix_order, char uplo, lapack_int n, float* a, - lapack_int lda ); -lapack_int LAPACKE_dlauum( int matrix_order, char uplo, lapack_int n, double* a, - lapack_int lda ); -lapack_int LAPACKE_clauum( int matrix_order, char uplo, lapack_int n, - lapack_complex_float* a, lapack_int lda ); -lapack_int LAPACKE_zlauum( int matrix_order, char uplo, lapack_int n, - lapack_complex_double* a, lapack_int lda ); - -lapack_int LAPACKE_sopgtr( int matrix_order, char uplo, lapack_int n, - const float* ap, const float* tau, float* q, - lapack_int ldq ); -lapack_int LAPACKE_dopgtr( int matrix_order, char uplo, lapack_int n, - const double* ap, const double* tau, double* q, - lapack_int ldq ); - -lapack_int LAPACKE_sopmtr( int matrix_order, char side, char uplo, char trans, - lapack_int m, lapack_int n, const float* ap, - const float* tau, float* c, lapack_int ldc ); -lapack_int LAPACKE_dopmtr( int matrix_order, char side, char uplo, char trans, - lapack_int m, lapack_int n, const double* ap, - const double* tau, double* c, lapack_int ldc ); - -lapack_int LAPACKE_sorgbr( int matrix_order, char vect, lapack_int m, - lapack_int n, lapack_int k, float* a, lapack_int lda, - const float* tau ); -lapack_int LAPACKE_dorgbr( int matrix_order, char vect, lapack_int m, - lapack_int n, lapack_int k, double* a, - lapack_int lda, const double* tau ); - -lapack_int LAPACKE_sorghr( int matrix_order, lapack_int n, lapack_int ilo, - lapack_int ihi, float* a, lapack_int lda, - const float* tau ); -lapack_int LAPACKE_dorghr( int matrix_order, lapack_int n, lapack_int ilo, - lapack_int ihi, double* a, lapack_int lda, - const double* tau ); - -lapack_int LAPACKE_sorglq( int matrix_order, lapack_int m, lapack_int n, - lapack_int k, float* a, lapack_int lda, - const float* tau ); -lapack_int LAPACKE_dorglq( int matrix_order, lapack_int m, lapack_int n, - lapack_int k, double* a, lapack_int lda, - const double* tau ); - -lapack_int LAPACKE_sorgql( int matrix_order, lapack_int m, lapack_int n, - lapack_int k, float* a, lapack_int lda, - const float* tau ); -lapack_int LAPACKE_dorgql( int matrix_order, lapack_int m, lapack_int n, - lapack_int k, double* a, lapack_int lda, - const double* tau ); - -lapack_int LAPACKE_sorgqr( int matrix_order, lapack_int m, lapack_int n, - lapack_int k, float* a, lapack_int lda, - const float* tau ); -lapack_int LAPACKE_dorgqr( int matrix_order, lapack_int m, lapack_int n, - lapack_int k, double* a, lapack_int lda, - const double* tau ); - -lapack_int LAPACKE_sorgrq( int matrix_order, lapack_int m, lapack_int n, - lapack_int k, float* a, lapack_int lda, - const float* tau ); -lapack_int LAPACKE_dorgrq( int matrix_order, lapack_int m, lapack_int n, - lapack_int k, double* a, lapack_int lda, - const double* tau ); - -lapack_int LAPACKE_sorgtr( int matrix_order, char uplo, lapack_int n, float* a, - lapack_int lda, const float* tau ); -lapack_int LAPACKE_dorgtr( int matrix_order, char uplo, lapack_int n, double* a, - lapack_int lda, const double* tau ); - -lapack_int LAPACKE_sormbr( int matrix_order, char vect, char side, char trans, - lapack_int m, lapack_int n, lapack_int k, - const float* a, lapack_int lda, const float* tau, - float* c, lapack_int ldc ); -lapack_int LAPACKE_dormbr( int matrix_order, char vect, char side, char trans, - lapack_int m, lapack_int n, lapack_int k, - const double* a, lapack_int lda, const double* tau, - double* c, lapack_int ldc ); - -lapack_int LAPACKE_sormhr( int matrix_order, char side, char trans, - lapack_int m, lapack_int n, lapack_int ilo, - lapack_int ihi, const float* a, lapack_int lda, - const float* tau, float* c, lapack_int ldc ); -lapack_int LAPACKE_dormhr( int matrix_order, char side, char trans, - lapack_int m, lapack_int n, lapack_int ilo, - lapack_int ihi, const double* a, lapack_int lda, - const double* tau, double* c, lapack_int ldc ); - -lapack_int LAPACKE_sormlq( int matrix_order, char side, char trans, - lapack_int m, lapack_int n, lapack_int k, - const float* a, lapack_int lda, const float* tau, - float* c, lapack_int ldc ); -lapack_int LAPACKE_dormlq( int matrix_order, char side, char trans, - lapack_int m, lapack_int n, lapack_int k, - const double* a, lapack_int lda, const double* tau, - double* c, lapack_int ldc ); - -lapack_int LAPACKE_sormql( int matrix_order, char side, char trans, - lapack_int m, lapack_int n, lapack_int k, - const float* a, lapack_int lda, const float* tau, - float* c, lapack_int ldc ); -lapack_int LAPACKE_dormql( int matrix_order, char side, char trans, - lapack_int m, lapack_int n, lapack_int k, - const double* a, lapack_int lda, const double* tau, - double* c, lapack_int ldc ); - -lapack_int LAPACKE_sormqr( int matrix_order, char side, char trans, - lapack_int m, lapack_int n, lapack_int k, - const float* a, lapack_int lda, const float* tau, - float* c, lapack_int ldc ); -lapack_int LAPACKE_dormqr( int matrix_order, char side, char trans, - lapack_int m, lapack_int n, lapack_int k, - const double* a, lapack_int lda, const double* tau, - double* c, lapack_int ldc ); - -lapack_int LAPACKE_sormrq( int matrix_order, char side, char trans, - lapack_int m, lapack_int n, lapack_int k, - const float* a, lapack_int lda, const float* tau, - float* c, lapack_int ldc ); -lapack_int LAPACKE_dormrq( int matrix_order, char side, char trans, - lapack_int m, lapack_int n, lapack_int k, - const double* a, lapack_int lda, const double* tau, - double* c, lapack_int ldc ); - -lapack_int LAPACKE_sormrz( int matrix_order, char side, char trans, - lapack_int m, lapack_int n, lapack_int k, - lapack_int l, const float* a, lapack_int lda, - const float* tau, float* c, lapack_int ldc ); -lapack_int LAPACKE_dormrz( int matrix_order, char side, char trans, - lapack_int m, lapack_int n, lapack_int k, - lapack_int l, const double* a, lapack_int lda, - const double* tau, double* c, lapack_int ldc ); - -lapack_int LAPACKE_sormtr( int matrix_order, char side, char uplo, char trans, - lapack_int m, lapack_int n, const float* a, - lapack_int lda, const float* tau, float* c, - lapack_int ldc ); -lapack_int LAPACKE_dormtr( int matrix_order, char side, char uplo, char trans, - lapack_int m, lapack_int n, const double* a, - lapack_int lda, const double* tau, double* c, - lapack_int ldc ); - -lapack_int LAPACKE_spbcon( int matrix_order, char uplo, lapack_int n, - lapack_int kd, const float* ab, lapack_int ldab, - float anorm, float* rcond ); -lapack_int LAPACKE_dpbcon( int matrix_order, char uplo, lapack_int n, - lapack_int kd, const double* ab, lapack_int ldab, - double anorm, double* rcond ); -lapack_int LAPACKE_cpbcon( int matrix_order, char uplo, lapack_int n, - lapack_int kd, const lapack_complex_float* ab, - lapack_int ldab, float anorm, float* rcond ); -lapack_int LAPACKE_zpbcon( int matrix_order, char uplo, lapack_int n, - lapack_int kd, const lapack_complex_double* ab, - lapack_int ldab, double anorm, double* rcond ); - -lapack_int LAPACKE_spbequ( int matrix_order, char uplo, lapack_int n, - lapack_int kd, const float* ab, lapack_int ldab, - float* s, float* scond, float* amax ); -lapack_int LAPACKE_dpbequ( int matrix_order, char uplo, lapack_int n, - lapack_int kd, const double* ab, lapack_int ldab, - double* s, double* scond, double* amax ); -lapack_int LAPACKE_cpbequ( int matrix_order, char uplo, lapack_int n, - lapack_int kd, const lapack_complex_float* ab, - lapack_int ldab, float* s, float* scond, - float* amax ); -lapack_int LAPACKE_zpbequ( int matrix_order, char uplo, lapack_int n, - lapack_int kd, const lapack_complex_double* ab, - lapack_int ldab, double* s, double* scond, - double* amax ); - -lapack_int LAPACKE_spbrfs( int matrix_order, char uplo, lapack_int n, - lapack_int kd, lapack_int nrhs, const float* ab, - lapack_int ldab, const float* afb, lapack_int ldafb, - const float* b, lapack_int ldb, float* x, - lapack_int ldx, float* ferr, float* berr ); -lapack_int LAPACKE_dpbrfs( int matrix_order, char uplo, lapack_int n, - lapack_int kd, lapack_int nrhs, const double* ab, - lapack_int ldab, const double* afb, lapack_int ldafb, - const double* b, lapack_int ldb, double* x, - lapack_int ldx, double* ferr, double* berr ); -lapack_int LAPACKE_cpbrfs( int matrix_order, char uplo, lapack_int n, - lapack_int kd, lapack_int nrhs, - const lapack_complex_float* ab, lapack_int ldab, - const lapack_complex_float* afb, lapack_int ldafb, - const lapack_complex_float* b, lapack_int ldb, - lapack_complex_float* x, lapack_int ldx, float* ferr, - float* berr ); -lapack_int LAPACKE_zpbrfs( int matrix_order, char uplo, lapack_int n, - lapack_int kd, lapack_int nrhs, - const lapack_complex_double* ab, lapack_int ldab, - const lapack_complex_double* afb, lapack_int ldafb, - const lapack_complex_double* b, lapack_int ldb, - lapack_complex_double* x, lapack_int ldx, - double* ferr, double* berr ); - -lapack_int LAPACKE_spbstf( int matrix_order, char uplo, lapack_int n, - lapack_int kb, float* bb, lapack_int ldbb ); -lapack_int LAPACKE_dpbstf( int matrix_order, char uplo, lapack_int n, - lapack_int kb, double* bb, lapack_int ldbb ); -lapack_int LAPACKE_cpbstf( int matrix_order, char uplo, lapack_int n, - lapack_int kb, lapack_complex_float* bb, - lapack_int ldbb ); -lapack_int LAPACKE_zpbstf( int matrix_order, char uplo, lapack_int n, - lapack_int kb, lapack_complex_double* bb, - lapack_int ldbb ); - -lapack_int LAPACKE_spbsv( int matrix_order, char uplo, lapack_int n, - lapack_int kd, lapack_int nrhs, float* ab, - lapack_int ldab, float* b, lapack_int ldb ); -lapack_int LAPACKE_dpbsv( int matrix_order, char uplo, lapack_int n, - lapack_int kd, lapack_int nrhs, double* ab, - lapack_int ldab, double* b, lapack_int ldb ); -lapack_int LAPACKE_cpbsv( int matrix_order, char uplo, lapack_int n, - lapack_int kd, lapack_int nrhs, - lapack_complex_float* ab, lapack_int ldab, - lapack_complex_float* b, lapack_int ldb ); -lapack_int LAPACKE_zpbsv( int matrix_order, char uplo, lapack_int n, - lapack_int kd, lapack_int nrhs, - lapack_complex_double* ab, lapack_int ldab, - lapack_complex_double* b, lapack_int ldb ); - -lapack_int LAPACKE_spbsvx( int matrix_order, char fact, char uplo, lapack_int n, - lapack_int kd, lapack_int nrhs, float* ab, - lapack_int ldab, float* afb, lapack_int ldafb, - char* equed, float* s, float* b, lapack_int ldb, - float* x, lapack_int ldx, float* rcond, float* ferr, - float* berr ); -lapack_int LAPACKE_dpbsvx( int matrix_order, char fact, char uplo, lapack_int n, - lapack_int kd, lapack_int nrhs, double* ab, - lapack_int ldab, double* afb, lapack_int ldafb, - char* equed, double* s, double* b, lapack_int ldb, - double* x, lapack_int ldx, double* rcond, - double* ferr, double* berr ); -lapack_int LAPACKE_cpbsvx( int matrix_order, char fact, char uplo, lapack_int n, - lapack_int kd, lapack_int nrhs, - lapack_complex_float* ab, lapack_int ldab, - lapack_complex_float* afb, lapack_int ldafb, - char* equed, float* s, lapack_complex_float* b, - lapack_int ldb, lapack_complex_float* x, - lapack_int ldx, float* rcond, float* ferr, - float* berr ); -lapack_int LAPACKE_zpbsvx( int matrix_order, char fact, char uplo, lapack_int n, - lapack_int kd, lapack_int nrhs, - lapack_complex_double* ab, lapack_int ldab, - lapack_complex_double* afb, lapack_int ldafb, - char* equed, double* s, lapack_complex_double* b, - lapack_int ldb, lapack_complex_double* x, - lapack_int ldx, double* rcond, double* ferr, - double* berr ); - -lapack_int LAPACKE_spbtrf( int matrix_order, char uplo, lapack_int n, - lapack_int kd, float* ab, lapack_int ldab ); -lapack_int LAPACKE_dpbtrf( int matrix_order, char uplo, lapack_int n, - lapack_int kd, double* ab, lapack_int ldab ); -lapack_int LAPACKE_cpbtrf( int matrix_order, char uplo, lapack_int n, - lapack_int kd, lapack_complex_float* ab, - lapack_int ldab ); -lapack_int LAPACKE_zpbtrf( int matrix_order, char uplo, lapack_int n, - lapack_int kd, lapack_complex_double* ab, - lapack_int ldab ); - -lapack_int LAPACKE_spbtrs( int matrix_order, char uplo, lapack_int n, - lapack_int kd, lapack_int nrhs, const float* ab, - lapack_int ldab, float* b, lapack_int ldb ); -lapack_int LAPACKE_dpbtrs( int matrix_order, char uplo, lapack_int n, - lapack_int kd, lapack_int nrhs, const double* ab, - lapack_int ldab, double* b, lapack_int ldb ); -lapack_int LAPACKE_cpbtrs( int matrix_order, char uplo, lapack_int n, - lapack_int kd, lapack_int nrhs, - const lapack_complex_float* ab, lapack_int ldab, - lapack_complex_float* b, lapack_int ldb ); -lapack_int LAPACKE_zpbtrs( int matrix_order, char uplo, lapack_int n, - lapack_int kd, lapack_int nrhs, - const lapack_complex_double* ab, lapack_int ldab, - lapack_complex_double* b, lapack_int ldb ); - -lapack_int LAPACKE_spftrf( int matrix_order, char transr, char uplo, - lapack_int n, float* a ); -lapack_int LAPACKE_dpftrf( int matrix_order, char transr, char uplo, - lapack_int n, double* a ); -lapack_int LAPACKE_cpftrf( int matrix_order, char transr, char uplo, - lapack_int n, lapack_complex_float* a ); -lapack_int LAPACKE_zpftrf( int matrix_order, char transr, char uplo, - lapack_int n, lapack_complex_double* a ); - -lapack_int LAPACKE_spftri( int matrix_order, char transr, char uplo, - lapack_int n, float* a ); -lapack_int LAPACKE_dpftri( int matrix_order, char transr, char uplo, - lapack_int n, double* a ); -lapack_int LAPACKE_cpftri( int matrix_order, char transr, char uplo, - lapack_int n, lapack_complex_float* a ); -lapack_int LAPACKE_zpftri( int matrix_order, char transr, char uplo, - lapack_int n, lapack_complex_double* a ); - -lapack_int LAPACKE_spftrs( int matrix_order, char transr, char uplo, - lapack_int n, lapack_int nrhs, const float* a, - float* b, lapack_int ldb ); -lapack_int LAPACKE_dpftrs( int matrix_order, char transr, char uplo, - lapack_int n, lapack_int nrhs, const double* a, - double* b, lapack_int ldb ); -lapack_int LAPACKE_cpftrs( int matrix_order, char transr, char uplo, - lapack_int n, lapack_int nrhs, - const lapack_complex_float* a, - lapack_complex_float* b, lapack_int ldb ); -lapack_int LAPACKE_zpftrs( int matrix_order, char transr, char uplo, - lapack_int n, lapack_int nrhs, - const lapack_complex_double* a, - lapack_complex_double* b, lapack_int ldb ); - -lapack_int LAPACKE_spocon( int matrix_order, char uplo, lapack_int n, - const float* a, lapack_int lda, float anorm, - float* rcond ); -lapack_int LAPACKE_dpocon( int matrix_order, char uplo, lapack_int n, - const double* a, lapack_int lda, double anorm, - double* rcond ); -lapack_int LAPACKE_cpocon( int matrix_order, char uplo, lapack_int n, - const lapack_complex_float* a, lapack_int lda, - float anorm, float* rcond ); -lapack_int LAPACKE_zpocon( int matrix_order, char uplo, lapack_int n, - const lapack_complex_double* a, lapack_int lda, - double anorm, double* rcond ); - -lapack_int LAPACKE_spoequ( int matrix_order, lapack_int n, const float* a, - lapack_int lda, float* s, float* scond, - float* amax ); -lapack_int LAPACKE_dpoequ( int matrix_order, lapack_int n, const double* a, - lapack_int lda, double* s, double* scond, - double* amax ); -lapack_int LAPACKE_cpoequ( int matrix_order, lapack_int n, - const lapack_complex_float* a, lapack_int lda, - float* s, float* scond, float* amax ); -lapack_int LAPACKE_zpoequ( int matrix_order, lapack_int n, - const lapack_complex_double* a, lapack_int lda, - double* s, double* scond, double* amax ); - -lapack_int LAPACKE_spoequb( int matrix_order, lapack_int n, const float* a, - lapack_int lda, float* s, float* scond, - float* amax ); -lapack_int LAPACKE_dpoequb( int matrix_order, lapack_int n, const double* a, - lapack_int lda, double* s, double* scond, - double* amax ); -lapack_int LAPACKE_cpoequb( int matrix_order, lapack_int n, - const lapack_complex_float* a, lapack_int lda, - float* s, float* scond, float* amax ); -lapack_int LAPACKE_zpoequb( int matrix_order, lapack_int n, - const lapack_complex_double* a, lapack_int lda, - double* s, double* scond, double* amax ); - -lapack_int LAPACKE_sporfs( int matrix_order, char uplo, lapack_int n, - lapack_int nrhs, const float* a, lapack_int lda, - const float* af, lapack_int ldaf, const float* b, - lapack_int ldb, float* x, lapack_int ldx, - float* ferr, float* berr ); -lapack_int LAPACKE_dporfs( int matrix_order, char uplo, lapack_int n, - lapack_int nrhs, const double* a, lapack_int lda, - const double* af, lapack_int ldaf, const double* b, - lapack_int ldb, double* x, lapack_int ldx, - double* ferr, double* berr ); -lapack_int LAPACKE_cporfs( int matrix_order, char uplo, lapack_int n, - lapack_int nrhs, const lapack_complex_float* a, - lapack_int lda, const lapack_complex_float* af, - lapack_int ldaf, const lapack_complex_float* b, - lapack_int ldb, lapack_complex_float* x, - lapack_int ldx, float* ferr, float* berr ); -lapack_int LAPACKE_zporfs( int matrix_order, char uplo, lapack_int n, - lapack_int nrhs, const lapack_complex_double* a, - lapack_int lda, const lapack_complex_double* af, - lapack_int ldaf, const lapack_complex_double* b, - lapack_int ldb, lapack_complex_double* x, - lapack_int ldx, double* ferr, double* berr ); - -lapack_int LAPACKE_sporfsx( int matrix_order, char uplo, char equed, - lapack_int n, lapack_int nrhs, const float* a, - lapack_int lda, const float* af, lapack_int ldaf, - const float* s, const float* b, lapack_int ldb, - float* x, lapack_int ldx, float* rcond, float* berr, - lapack_int n_err_bnds, float* err_bnds_norm, - float* err_bnds_comp, lapack_int nparams, - float* params ); -lapack_int LAPACKE_dporfsx( int matrix_order, char uplo, char equed, - lapack_int n, lapack_int nrhs, const double* a, - lapack_int lda, const double* af, lapack_int ldaf, - const double* s, const double* b, lapack_int ldb, - double* x, lapack_int ldx, double* rcond, - double* berr, lapack_int n_err_bnds, - double* err_bnds_norm, double* err_bnds_comp, - lapack_int nparams, double* params ); -lapack_int LAPACKE_cporfsx( int matrix_order, char uplo, char equed, - lapack_int n, lapack_int nrhs, - const lapack_complex_float* a, lapack_int lda, - const lapack_complex_float* af, lapack_int ldaf, - const float* s, const lapack_complex_float* b, - lapack_int ldb, lapack_complex_float* x, - lapack_int ldx, float* rcond, float* berr, - lapack_int n_err_bnds, float* err_bnds_norm, - float* err_bnds_comp, lapack_int nparams, - float* params ); -lapack_int LAPACKE_zporfsx( int matrix_order, char uplo, char equed, - lapack_int n, lapack_int nrhs, - const lapack_complex_double* a, lapack_int lda, - const lapack_complex_double* af, lapack_int ldaf, - const double* s, const lapack_complex_double* b, - lapack_int ldb, lapack_complex_double* x, - lapack_int ldx, double* rcond, double* berr, - lapack_int n_err_bnds, double* err_bnds_norm, - double* err_bnds_comp, lapack_int nparams, - double* params ); - -lapack_int LAPACKE_sposv( int matrix_order, char uplo, lapack_int n, - lapack_int nrhs, float* a, lapack_int lda, float* b, - lapack_int ldb ); -lapack_int LAPACKE_dposv( int matrix_order, char uplo, lapack_int n, - lapack_int nrhs, double* a, lapack_int lda, double* b, - lapack_int ldb ); -lapack_int LAPACKE_cposv( int matrix_order, char uplo, lapack_int n, - lapack_int nrhs, lapack_complex_float* a, - lapack_int lda, lapack_complex_float* b, - lapack_int ldb ); -lapack_int LAPACKE_zposv( int matrix_order, char uplo, lapack_int n, - lapack_int nrhs, lapack_complex_double* a, - lapack_int lda, lapack_complex_double* b, - lapack_int ldb ); -lapack_int LAPACKE_dsposv( int matrix_order, char uplo, lapack_int n, - lapack_int nrhs, double* a, lapack_int lda, - double* b, lapack_int ldb, double* x, lapack_int ldx, - lapack_int* iter ); -lapack_int LAPACKE_zcposv( int matrix_order, char uplo, lapack_int n, - lapack_int nrhs, lapack_complex_double* a, - lapack_int lda, lapack_complex_double* b, - lapack_int ldb, lapack_complex_double* x, - lapack_int ldx, lapack_int* iter ); - -lapack_int LAPACKE_sposvx( int matrix_order, char fact, char uplo, lapack_int n, - lapack_int nrhs, float* a, lapack_int lda, float* af, - lapack_int ldaf, char* equed, float* s, float* b, - lapack_int ldb, float* x, lapack_int ldx, - float* rcond, float* ferr, float* berr ); -lapack_int LAPACKE_dposvx( int matrix_order, char fact, char uplo, lapack_int n, - lapack_int nrhs, double* a, lapack_int lda, - double* af, lapack_int ldaf, char* equed, double* s, - double* b, lapack_int ldb, double* x, lapack_int ldx, - double* rcond, double* ferr, double* berr ); -lapack_int LAPACKE_cposvx( int matrix_order, char fact, char uplo, lapack_int n, - lapack_int nrhs, lapack_complex_float* a, - lapack_int lda, lapack_complex_float* af, - lapack_int ldaf, char* equed, float* s, - lapack_complex_float* b, lapack_int ldb, - lapack_complex_float* x, lapack_int ldx, - float* rcond, float* ferr, float* berr ); -lapack_int LAPACKE_zposvx( int matrix_order, char fact, char uplo, lapack_int n, - lapack_int nrhs, lapack_complex_double* a, - lapack_int lda, lapack_complex_double* af, - lapack_int ldaf, char* equed, double* s, - lapack_complex_double* b, lapack_int ldb, - lapack_complex_double* x, lapack_int ldx, - double* rcond, double* ferr, double* berr ); - -lapack_int LAPACKE_sposvxx( int matrix_order, char fact, char uplo, - lapack_int n, lapack_int nrhs, float* a, - lapack_int lda, float* af, lapack_int ldaf, - char* equed, float* s, float* b, lapack_int ldb, - float* x, lapack_int ldx, float* rcond, - float* rpvgrw, float* berr, lapack_int n_err_bnds, - float* err_bnds_norm, float* err_bnds_comp, - lapack_int nparams, float* params ); -lapack_int LAPACKE_dposvxx( int matrix_order, char fact, char uplo, - lapack_int n, lapack_int nrhs, double* a, - lapack_int lda, double* af, lapack_int ldaf, - char* equed, double* s, double* b, lapack_int ldb, - double* x, lapack_int ldx, double* rcond, - double* rpvgrw, double* berr, lapack_int n_err_bnds, - double* err_bnds_norm, double* err_bnds_comp, - lapack_int nparams, double* params ); -lapack_int LAPACKE_cposvxx( int matrix_order, char fact, char uplo, - lapack_int n, lapack_int nrhs, - lapack_complex_float* a, lapack_int lda, - lapack_complex_float* af, lapack_int ldaf, - char* equed, float* s, lapack_complex_float* b, - lapack_int ldb, lapack_complex_float* x, - lapack_int ldx, float* rcond, float* rpvgrw, - float* berr, lapack_int n_err_bnds, - float* err_bnds_norm, float* err_bnds_comp, - lapack_int nparams, float* params ); -lapack_int LAPACKE_zposvxx( int matrix_order, char fact, char uplo, - lapack_int n, lapack_int nrhs, - lapack_complex_double* a, lapack_int lda, - lapack_complex_double* af, lapack_int ldaf, - char* equed, double* s, lapack_complex_double* b, - lapack_int ldb, lapack_complex_double* x, - lapack_int ldx, double* rcond, double* rpvgrw, - double* berr, lapack_int n_err_bnds, - double* err_bnds_norm, double* err_bnds_comp, - lapack_int nparams, double* params ); - -lapack_int LAPACKE_spotrf( int matrix_order, char uplo, lapack_int n, float* a, - lapack_int lda ); -lapack_int LAPACKE_dpotrf( int matrix_order, char uplo, lapack_int n, double* a, - lapack_int lda ); -lapack_int LAPACKE_cpotrf( int matrix_order, char uplo, lapack_int n, - lapack_complex_float* a, lapack_int lda ); -lapack_int LAPACKE_zpotrf( int matrix_order, char uplo, lapack_int n, - lapack_complex_double* a, lapack_int lda ); - -lapack_int LAPACKE_spotri( int matrix_order, char uplo, lapack_int n, float* a, - lapack_int lda ); -lapack_int LAPACKE_dpotri( int matrix_order, char uplo, lapack_int n, double* a, - lapack_int lda ); -lapack_int LAPACKE_cpotri( int matrix_order, char uplo, lapack_int n, - lapack_complex_float* a, lapack_int lda ); -lapack_int LAPACKE_zpotri( int matrix_order, char uplo, lapack_int n, - lapack_complex_double* a, lapack_int lda ); - -lapack_int LAPACKE_spotrs( int matrix_order, char uplo, lapack_int n, - lapack_int nrhs, const float* a, lapack_int lda, - float* b, lapack_int ldb ); -lapack_int LAPACKE_dpotrs( int matrix_order, char uplo, lapack_int n, - lapack_int nrhs, const double* a, lapack_int lda, - double* b, lapack_int ldb ); -lapack_int LAPACKE_cpotrs( int matrix_order, char uplo, lapack_int n, - lapack_int nrhs, const lapack_complex_float* a, - lapack_int lda, lapack_complex_float* b, - lapack_int ldb ); -lapack_int LAPACKE_zpotrs( int matrix_order, char uplo, lapack_int n, - lapack_int nrhs, const lapack_complex_double* a, - lapack_int lda, lapack_complex_double* b, - lapack_int ldb ); - -lapack_int LAPACKE_sppcon( int matrix_order, char uplo, lapack_int n, - const float* ap, float anorm, float* rcond ); -lapack_int LAPACKE_dppcon( int matrix_order, char uplo, lapack_int n, - const double* ap, double anorm, double* rcond ); -lapack_int LAPACKE_cppcon( int matrix_order, char uplo, lapack_int n, - const lapack_complex_float* ap, float anorm, - float* rcond ); -lapack_int LAPACKE_zppcon( int matrix_order, char uplo, lapack_int n, - const lapack_complex_double* ap, double anorm, - double* rcond ); - -lapack_int LAPACKE_sppequ( int matrix_order, char uplo, lapack_int n, - const float* ap, float* s, float* scond, - float* amax ); -lapack_int LAPACKE_dppequ( int matrix_order, char uplo, lapack_int n, - const double* ap, double* s, double* scond, - double* amax ); -lapack_int LAPACKE_cppequ( int matrix_order, char uplo, lapack_int n, - const lapack_complex_float* ap, float* s, - float* scond, float* amax ); -lapack_int LAPACKE_zppequ( int matrix_order, char uplo, lapack_int n, - const lapack_complex_double* ap, double* s, - double* scond, double* amax ); - -lapack_int LAPACKE_spprfs( int matrix_order, char uplo, lapack_int n, - lapack_int nrhs, const float* ap, const float* afp, - const float* b, lapack_int ldb, float* x, - lapack_int ldx, float* ferr, float* berr ); -lapack_int LAPACKE_dpprfs( int matrix_order, char uplo, lapack_int n, - lapack_int nrhs, const double* ap, const double* afp, - const double* b, lapack_int ldb, double* x, - lapack_int ldx, double* ferr, double* berr ); -lapack_int LAPACKE_cpprfs( int matrix_order, char uplo, lapack_int n, - lapack_int nrhs, const lapack_complex_float* ap, - const lapack_complex_float* afp, - const lapack_complex_float* b, lapack_int ldb, - lapack_complex_float* x, lapack_int ldx, float* ferr, - float* berr ); -lapack_int LAPACKE_zpprfs( int matrix_order, char uplo, lapack_int n, - lapack_int nrhs, const lapack_complex_double* ap, - const lapack_complex_double* afp, - const lapack_complex_double* b, lapack_int ldb, - lapack_complex_double* x, lapack_int ldx, - double* ferr, double* berr ); - -lapack_int LAPACKE_sppsv( int matrix_order, char uplo, lapack_int n, - lapack_int nrhs, float* ap, float* b, - lapack_int ldb ); -lapack_int LAPACKE_dppsv( int matrix_order, char uplo, lapack_int n, - lapack_int nrhs, double* ap, double* b, - lapack_int ldb ); -lapack_int LAPACKE_cppsv( int matrix_order, char uplo, lapack_int n, - lapack_int nrhs, lapack_complex_float* ap, - lapack_complex_float* b, lapack_int ldb ); -lapack_int LAPACKE_zppsv( int matrix_order, char uplo, lapack_int n, - lapack_int nrhs, lapack_complex_double* ap, - lapack_complex_double* b, lapack_int ldb ); - -lapack_int LAPACKE_sppsvx( int matrix_order, char fact, char uplo, lapack_int n, - lapack_int nrhs, float* ap, float* afp, char* equed, - float* s, float* b, lapack_int ldb, float* x, - lapack_int ldx, float* rcond, float* ferr, - float* berr ); -lapack_int LAPACKE_dppsvx( int matrix_order, char fact, char uplo, lapack_int n, - lapack_int nrhs, double* ap, double* afp, - char* equed, double* s, double* b, lapack_int ldb, - double* x, lapack_int ldx, double* rcond, - double* ferr, double* berr ); -lapack_int LAPACKE_cppsvx( int matrix_order, char fact, char uplo, lapack_int n, - lapack_int nrhs, lapack_complex_float* ap, - lapack_complex_float* afp, char* equed, float* s, - lapack_complex_float* b, lapack_int ldb, - lapack_complex_float* x, lapack_int ldx, - float* rcond, float* ferr, float* berr ); -lapack_int LAPACKE_zppsvx( int matrix_order, char fact, char uplo, lapack_int n, - lapack_int nrhs, lapack_complex_double* ap, - lapack_complex_double* afp, char* equed, double* s, - lapack_complex_double* b, lapack_int ldb, - lapack_complex_double* x, lapack_int ldx, - double* rcond, double* ferr, double* berr ); - -lapack_int LAPACKE_spptrf( int matrix_order, char uplo, lapack_int n, - float* ap ); -lapack_int LAPACKE_dpptrf( int matrix_order, char uplo, lapack_int n, - double* ap ); -lapack_int LAPACKE_cpptrf( int matrix_order, char uplo, lapack_int n, - lapack_complex_float* ap ); -lapack_int LAPACKE_zpptrf( int matrix_order, char uplo, lapack_int n, - lapack_complex_double* ap ); - -lapack_int LAPACKE_spptri( int matrix_order, char uplo, lapack_int n, - float* ap ); -lapack_int LAPACKE_dpptri( int matrix_order, char uplo, lapack_int n, - double* ap ); -lapack_int LAPACKE_cpptri( int matrix_order, char uplo, lapack_int n, - lapack_complex_float* ap ); -lapack_int LAPACKE_zpptri( int matrix_order, char uplo, lapack_int n, - lapack_complex_double* ap ); - -lapack_int LAPACKE_spptrs( int matrix_order, char uplo, lapack_int n, - lapack_int nrhs, const float* ap, float* b, - lapack_int ldb ); -lapack_int LAPACKE_dpptrs( int matrix_order, char uplo, lapack_int n, - lapack_int nrhs, const double* ap, double* b, - lapack_int ldb ); -lapack_int LAPACKE_cpptrs( int matrix_order, char uplo, lapack_int n, - lapack_int nrhs, const lapack_complex_float* ap, - lapack_complex_float* b, lapack_int ldb ); -lapack_int LAPACKE_zpptrs( int matrix_order, char uplo, lapack_int n, - lapack_int nrhs, const lapack_complex_double* ap, - lapack_complex_double* b, lapack_int ldb ); - -lapack_int LAPACKE_spstrf( int matrix_order, char uplo, lapack_int n, float* a, - lapack_int lda, lapack_int* piv, lapack_int* rank, - float tol ); -lapack_int LAPACKE_dpstrf( int matrix_order, char uplo, lapack_int n, double* a, - lapack_int lda, lapack_int* piv, lapack_int* rank, - double tol ); -lapack_int LAPACKE_cpstrf( int matrix_order, char uplo, lapack_int n, - lapack_complex_float* a, lapack_int lda, - lapack_int* piv, lapack_int* rank, float tol ); -lapack_int LAPACKE_zpstrf( int matrix_order, char uplo, lapack_int n, - lapack_complex_double* a, lapack_int lda, - lapack_int* piv, lapack_int* rank, double tol ); - -lapack_int LAPACKE_sptcon( lapack_int n, const float* d, const float* e, - float anorm, float* rcond ); -lapack_int LAPACKE_dptcon( lapack_int n, const double* d, const double* e, - double anorm, double* rcond ); -lapack_int LAPACKE_cptcon( lapack_int n, const float* d, - const lapack_complex_float* e, float anorm, - float* rcond ); -lapack_int LAPACKE_zptcon( lapack_int n, const double* d, - const lapack_complex_double* e, double anorm, - double* rcond ); - -lapack_int LAPACKE_spteqr( int matrix_order, char compz, lapack_int n, float* d, - float* e, float* z, lapack_int ldz ); -lapack_int LAPACKE_dpteqr( int matrix_order, char compz, lapack_int n, - double* d, double* e, double* z, lapack_int ldz ); -lapack_int LAPACKE_cpteqr( int matrix_order, char compz, lapack_int n, float* d, - float* e, lapack_complex_float* z, lapack_int ldz ); -lapack_int LAPACKE_zpteqr( int matrix_order, char compz, lapack_int n, - double* d, double* e, lapack_complex_double* z, - lapack_int ldz ); - -lapack_int LAPACKE_sptrfs( int matrix_order, lapack_int n, lapack_int nrhs, - const float* d, const float* e, const float* df, - const float* ef, const float* b, lapack_int ldb, - float* x, lapack_int ldx, float* ferr, float* berr ); -lapack_int LAPACKE_dptrfs( int matrix_order, lapack_int n, lapack_int nrhs, - const double* d, const double* e, const double* df, - const double* ef, const double* b, lapack_int ldb, - double* x, lapack_int ldx, double* ferr, - double* berr ); -lapack_int LAPACKE_cptrfs( int matrix_order, char uplo, lapack_int n, - lapack_int nrhs, const float* d, - const lapack_complex_float* e, const float* df, - const lapack_complex_float* ef, - const lapack_complex_float* b, lapack_int ldb, - lapack_complex_float* x, lapack_int ldx, float* ferr, - float* berr ); -lapack_int LAPACKE_zptrfs( int matrix_order, char uplo, lapack_int n, - lapack_int nrhs, const double* d, - const lapack_complex_double* e, const double* df, - const lapack_complex_double* ef, - const lapack_complex_double* b, lapack_int ldb, - lapack_complex_double* x, lapack_int ldx, - double* ferr, double* berr ); - -lapack_int LAPACKE_sptsv( int matrix_order, lapack_int n, lapack_int nrhs, - float* d, float* e, float* b, lapack_int ldb ); -lapack_int LAPACKE_dptsv( int matrix_order, lapack_int n, lapack_int nrhs, - double* d, double* e, double* b, lapack_int ldb ); -lapack_int LAPACKE_cptsv( int matrix_order, lapack_int n, lapack_int nrhs, - float* d, lapack_complex_float* e, - lapack_complex_float* b, lapack_int ldb ); -lapack_int LAPACKE_zptsv( int matrix_order, lapack_int n, lapack_int nrhs, - double* d, lapack_complex_double* e, - lapack_complex_double* b, lapack_int ldb ); - -lapack_int LAPACKE_sptsvx( int matrix_order, char fact, lapack_int n, - lapack_int nrhs, const float* d, const float* e, - float* df, float* ef, const float* b, lapack_int ldb, - float* x, lapack_int ldx, float* rcond, float* ferr, - float* berr ); -lapack_int LAPACKE_dptsvx( int matrix_order, char fact, lapack_int n, - lapack_int nrhs, const double* d, const double* e, - double* df, double* ef, const double* b, - lapack_int ldb, double* x, lapack_int ldx, - double* rcond, double* ferr, double* berr ); -lapack_int LAPACKE_cptsvx( int matrix_order, char fact, lapack_int n, - lapack_int nrhs, const float* d, - const lapack_complex_float* e, float* df, - lapack_complex_float* ef, - const lapack_complex_float* b, lapack_int ldb, - lapack_complex_float* x, lapack_int ldx, - float* rcond, float* ferr, float* berr ); -lapack_int LAPACKE_zptsvx( int matrix_order, char fact, lapack_int n, - lapack_int nrhs, const double* d, - const lapack_complex_double* e, double* df, - lapack_complex_double* ef, - const lapack_complex_double* b, lapack_int ldb, - lapack_complex_double* x, lapack_int ldx, - double* rcond, double* ferr, double* berr ); - -lapack_int LAPACKE_spttrf( lapack_int n, float* d, float* e ); -lapack_int LAPACKE_dpttrf( lapack_int n, double* d, double* e ); -lapack_int LAPACKE_cpttrf( lapack_int n, float* d, lapack_complex_float* e ); -lapack_int LAPACKE_zpttrf( lapack_int n, double* d, lapack_complex_double* e ); - -lapack_int LAPACKE_spttrs( int matrix_order, lapack_int n, lapack_int nrhs, - const float* d, const float* e, float* b, - lapack_int ldb ); -lapack_int LAPACKE_dpttrs( int matrix_order, lapack_int n, lapack_int nrhs, - const double* d, const double* e, double* b, - lapack_int ldb ); -lapack_int LAPACKE_cpttrs( int matrix_order, char uplo, lapack_int n, - lapack_int nrhs, const float* d, - const lapack_complex_float* e, - lapack_complex_float* b, lapack_int ldb ); -lapack_int LAPACKE_zpttrs( int matrix_order, char uplo, lapack_int n, - lapack_int nrhs, const double* d, - const lapack_complex_double* e, - lapack_complex_double* b, lapack_int ldb ); - -lapack_int LAPACKE_ssbev( int matrix_order, char jobz, char uplo, lapack_int n, - lapack_int kd, float* ab, lapack_int ldab, float* w, - float* z, lapack_int ldz ); -lapack_int LAPACKE_dsbev( int matrix_order, char jobz, char uplo, lapack_int n, - lapack_int kd, double* ab, lapack_int ldab, double* w, - double* z, lapack_int ldz ); - -lapack_int LAPACKE_ssbevd( int matrix_order, char jobz, char uplo, lapack_int n, - lapack_int kd, float* ab, lapack_int ldab, float* w, - float* z, lapack_int ldz ); -lapack_int LAPACKE_dsbevd( int matrix_order, char jobz, char uplo, lapack_int n, - lapack_int kd, double* ab, lapack_int ldab, - double* w, double* z, lapack_int ldz ); - -lapack_int LAPACKE_ssbevx( int matrix_order, char jobz, char range, char uplo, - lapack_int n, lapack_int kd, float* ab, - lapack_int ldab, float* q, lapack_int ldq, float vl, - float vu, lapack_int il, lapack_int iu, float abstol, - lapack_int* m, float* w, float* z, lapack_int ldz, - lapack_int* ifail ); -lapack_int LAPACKE_dsbevx( int matrix_order, char jobz, char range, char uplo, - lapack_int n, lapack_int kd, double* ab, - lapack_int ldab, double* q, lapack_int ldq, - double vl, double vu, lapack_int il, lapack_int iu, - double abstol, lapack_int* m, double* w, double* z, - lapack_int ldz, lapack_int* ifail ); - -lapack_int LAPACKE_ssbgst( int matrix_order, char vect, char uplo, lapack_int n, - lapack_int ka, lapack_int kb, float* ab, - lapack_int ldab, const float* bb, lapack_int ldbb, - float* x, lapack_int ldx ); -lapack_int LAPACKE_dsbgst( int matrix_order, char vect, char uplo, lapack_int n, - lapack_int ka, lapack_int kb, double* ab, - lapack_int ldab, const double* bb, lapack_int ldbb, - double* x, lapack_int ldx ); - -lapack_int LAPACKE_ssbgv( int matrix_order, char jobz, char uplo, lapack_int n, - lapack_int ka, lapack_int kb, float* ab, - lapack_int ldab, float* bb, lapack_int ldbb, float* w, - float* z, lapack_int ldz ); -lapack_int LAPACKE_dsbgv( int matrix_order, char jobz, char uplo, lapack_int n, - lapack_int ka, lapack_int kb, double* ab, - lapack_int ldab, double* bb, lapack_int ldbb, - double* w, double* z, lapack_int ldz ); - -lapack_int LAPACKE_ssbgvd( int matrix_order, char jobz, char uplo, lapack_int n, - lapack_int ka, lapack_int kb, float* ab, - lapack_int ldab, float* bb, lapack_int ldbb, - float* w, float* z, lapack_int ldz ); -lapack_int LAPACKE_dsbgvd( int matrix_order, char jobz, char uplo, lapack_int n, - lapack_int ka, lapack_int kb, double* ab, - lapack_int ldab, double* bb, lapack_int ldbb, - double* w, double* z, lapack_int ldz ); - -lapack_int LAPACKE_ssbgvx( int matrix_order, char jobz, char range, char uplo, - lapack_int n, lapack_int ka, lapack_int kb, - float* ab, lapack_int ldab, float* bb, - lapack_int ldbb, float* q, lapack_int ldq, float vl, - float vu, lapack_int il, lapack_int iu, float abstol, - lapack_int* m, float* w, float* z, lapack_int ldz, - lapack_int* ifail ); -lapack_int LAPACKE_dsbgvx( int matrix_order, char jobz, char range, char uplo, - lapack_int n, lapack_int ka, lapack_int kb, - double* ab, lapack_int ldab, double* bb, - lapack_int ldbb, double* q, lapack_int ldq, - double vl, double vu, lapack_int il, lapack_int iu, - double abstol, lapack_int* m, double* w, double* z, - lapack_int ldz, lapack_int* ifail ); - -lapack_int LAPACKE_ssbtrd( int matrix_order, char vect, char uplo, lapack_int n, - lapack_int kd, float* ab, lapack_int ldab, float* d, - float* e, float* q, lapack_int ldq ); -lapack_int LAPACKE_dsbtrd( int matrix_order, char vect, char uplo, lapack_int n, - lapack_int kd, double* ab, lapack_int ldab, - double* d, double* e, double* q, lapack_int ldq ); - -lapack_int LAPACKE_ssfrk( int matrix_order, char transr, char uplo, char trans, - lapack_int n, lapack_int k, float alpha, - const float* a, lapack_int lda, float beta, - float* c ); -lapack_int LAPACKE_dsfrk( int matrix_order, char transr, char uplo, char trans, - lapack_int n, lapack_int k, double alpha, - const double* a, lapack_int lda, double beta, - double* c ); - -lapack_int LAPACKE_sspcon( int matrix_order, char uplo, lapack_int n, - const float* ap, const lapack_int* ipiv, float anorm, - float* rcond ); -lapack_int LAPACKE_dspcon( int matrix_order, char uplo, lapack_int n, - const double* ap, const lapack_int* ipiv, - double anorm, double* rcond ); -lapack_int LAPACKE_cspcon( int matrix_order, char uplo, lapack_int n, - const lapack_complex_float* ap, - const lapack_int* ipiv, float anorm, float* rcond ); -lapack_int LAPACKE_zspcon( int matrix_order, char uplo, lapack_int n, - const lapack_complex_double* ap, - const lapack_int* ipiv, double anorm, - double* rcond ); - -lapack_int LAPACKE_sspev( int matrix_order, char jobz, char uplo, lapack_int n, - float* ap, float* w, float* z, lapack_int ldz ); -lapack_int LAPACKE_dspev( int matrix_order, char jobz, char uplo, lapack_int n, - double* ap, double* w, double* z, lapack_int ldz ); - -lapack_int LAPACKE_sspevd( int matrix_order, char jobz, char uplo, lapack_int n, - float* ap, float* w, float* z, lapack_int ldz ); -lapack_int LAPACKE_dspevd( int matrix_order, char jobz, char uplo, lapack_int n, - double* ap, double* w, double* z, lapack_int ldz ); - -lapack_int LAPACKE_sspevx( int matrix_order, char jobz, char range, char uplo, - lapack_int n, float* ap, float vl, float vu, - lapack_int il, lapack_int iu, float abstol, - lapack_int* m, float* w, float* z, lapack_int ldz, - lapack_int* ifail ); -lapack_int LAPACKE_dspevx( int matrix_order, char jobz, char range, char uplo, - lapack_int n, double* ap, double vl, double vu, - lapack_int il, lapack_int iu, double abstol, - lapack_int* m, double* w, double* z, lapack_int ldz, - lapack_int* ifail ); - -lapack_int LAPACKE_sspgst( int matrix_order, lapack_int itype, char uplo, - lapack_int n, float* ap, const float* bp ); -lapack_int LAPACKE_dspgst( int matrix_order, lapack_int itype, char uplo, - lapack_int n, double* ap, const double* bp ); - -lapack_int LAPACKE_sspgv( int matrix_order, lapack_int itype, char jobz, - char uplo, lapack_int n, float* ap, float* bp, - float* w, float* z, lapack_int ldz ); -lapack_int LAPACKE_dspgv( int matrix_order, lapack_int itype, char jobz, - char uplo, lapack_int n, double* ap, double* bp, - double* w, double* z, lapack_int ldz ); - -lapack_int LAPACKE_sspgvd( int matrix_order, lapack_int itype, char jobz, - char uplo, lapack_int n, float* ap, float* bp, - float* w, float* z, lapack_int ldz ); -lapack_int LAPACKE_dspgvd( int matrix_order, lapack_int itype, char jobz, - char uplo, lapack_int n, double* ap, double* bp, - double* w, double* z, lapack_int ldz ); - -lapack_int LAPACKE_sspgvx( int matrix_order, lapack_int itype, char jobz, - char range, char uplo, lapack_int n, float* ap, - float* bp, float vl, float vu, lapack_int il, - lapack_int iu, float abstol, lapack_int* m, float* w, - float* z, lapack_int ldz, lapack_int* ifail ); -lapack_int LAPACKE_dspgvx( int matrix_order, lapack_int itype, char jobz, - char range, char uplo, lapack_int n, double* ap, - double* bp, double vl, double vu, lapack_int il, - lapack_int iu, double abstol, lapack_int* m, - double* w, double* z, lapack_int ldz, - lapack_int* ifail ); - -lapack_int LAPACKE_ssprfs( int matrix_order, char uplo, lapack_int n, - lapack_int nrhs, const float* ap, const float* afp, - const lapack_int* ipiv, const float* b, - lapack_int ldb, float* x, lapack_int ldx, - float* ferr, float* berr ); -lapack_int LAPACKE_dsprfs( int matrix_order, char uplo, lapack_int n, - lapack_int nrhs, const double* ap, const double* afp, - const lapack_int* ipiv, const double* b, - lapack_int ldb, double* x, lapack_int ldx, - double* ferr, double* berr ); -lapack_int LAPACKE_csprfs( int matrix_order, char uplo, lapack_int n, - lapack_int nrhs, const lapack_complex_float* ap, - const lapack_complex_float* afp, - const lapack_int* ipiv, - const lapack_complex_float* b, lapack_int ldb, - lapack_complex_float* x, lapack_int ldx, float* ferr, - float* berr ); -lapack_int LAPACKE_zsprfs( int matrix_order, char uplo, lapack_int n, - lapack_int nrhs, const lapack_complex_double* ap, - const lapack_complex_double* afp, - const lapack_int* ipiv, - const lapack_complex_double* b, lapack_int ldb, - lapack_complex_double* x, lapack_int ldx, - double* ferr, double* berr ); - -lapack_int LAPACKE_sspsv( int matrix_order, char uplo, lapack_int n, - lapack_int nrhs, float* ap, lapack_int* ipiv, - float* b, lapack_int ldb ); -lapack_int LAPACKE_dspsv( int matrix_order, char uplo, lapack_int n, - lapack_int nrhs, double* ap, lapack_int* ipiv, - double* b, lapack_int ldb ); -lapack_int LAPACKE_cspsv( int matrix_order, char uplo, lapack_int n, - lapack_int nrhs, lapack_complex_float* ap, - lapack_int* ipiv, lapack_complex_float* b, - lapack_int ldb ); -lapack_int LAPACKE_zspsv( int matrix_order, char uplo, lapack_int n, - lapack_int nrhs, lapack_complex_double* ap, - lapack_int* ipiv, lapack_complex_double* b, - lapack_int ldb ); - -lapack_int LAPACKE_sspsvx( int matrix_order, char fact, char uplo, lapack_int n, - lapack_int nrhs, const float* ap, float* afp, - lapack_int* ipiv, const float* b, lapack_int ldb, - float* x, lapack_int ldx, float* rcond, float* ferr, - float* berr ); -lapack_int LAPACKE_dspsvx( int matrix_order, char fact, char uplo, lapack_int n, - lapack_int nrhs, const double* ap, double* afp, - lapack_int* ipiv, const double* b, lapack_int ldb, - double* x, lapack_int ldx, double* rcond, - double* ferr, double* berr ); -lapack_int LAPACKE_cspsvx( int matrix_order, char fact, char uplo, lapack_int n, - lapack_int nrhs, const lapack_complex_float* ap, - lapack_complex_float* afp, lapack_int* ipiv, - const lapack_complex_float* b, lapack_int ldb, - lapack_complex_float* x, lapack_int ldx, - float* rcond, float* ferr, float* berr ); -lapack_int LAPACKE_zspsvx( int matrix_order, char fact, char uplo, lapack_int n, - lapack_int nrhs, const lapack_complex_double* ap, - lapack_complex_double* afp, lapack_int* ipiv, - const lapack_complex_double* b, lapack_int ldb, - lapack_complex_double* x, lapack_int ldx, - double* rcond, double* ferr, double* berr ); - -lapack_int LAPACKE_ssptrd( int matrix_order, char uplo, lapack_int n, float* ap, - float* d, float* e, float* tau ); -lapack_int LAPACKE_dsptrd( int matrix_order, char uplo, lapack_int n, - double* ap, double* d, double* e, double* tau ); - -lapack_int LAPACKE_ssptrf( int matrix_order, char uplo, lapack_int n, float* ap, - lapack_int* ipiv ); -lapack_int LAPACKE_dsptrf( int matrix_order, char uplo, lapack_int n, - double* ap, lapack_int* ipiv ); -lapack_int LAPACKE_csptrf( int matrix_order, char uplo, lapack_int n, - lapack_complex_float* ap, lapack_int* ipiv ); -lapack_int LAPACKE_zsptrf( int matrix_order, char uplo, lapack_int n, - lapack_complex_double* ap, lapack_int* ipiv ); - -lapack_int LAPACKE_ssptri( int matrix_order, char uplo, lapack_int n, float* ap, - const lapack_int* ipiv ); -lapack_int LAPACKE_dsptri( int matrix_order, char uplo, lapack_int n, - double* ap, const lapack_int* ipiv ); -lapack_int LAPACKE_csptri( int matrix_order, char uplo, lapack_int n, - lapack_complex_float* ap, const lapack_int* ipiv ); -lapack_int LAPACKE_zsptri( int matrix_order, char uplo, lapack_int n, - lapack_complex_double* ap, const lapack_int* ipiv ); - -lapack_int LAPACKE_ssptrs( int matrix_order, char uplo, lapack_int n, - lapack_int nrhs, const float* ap, - const lapack_int* ipiv, float* b, lapack_int ldb ); -lapack_int LAPACKE_dsptrs( int matrix_order, char uplo, lapack_int n, - lapack_int nrhs, const double* ap, - const lapack_int* ipiv, double* b, lapack_int ldb ); -lapack_int LAPACKE_csptrs( int matrix_order, char uplo, lapack_int n, - lapack_int nrhs, const lapack_complex_float* ap, - const lapack_int* ipiv, lapack_complex_float* b, - lapack_int ldb ); -lapack_int LAPACKE_zsptrs( int matrix_order, char uplo, lapack_int n, - lapack_int nrhs, const lapack_complex_double* ap, - const lapack_int* ipiv, lapack_complex_double* b, - lapack_int ldb ); - -lapack_int LAPACKE_sstebz( char range, char order, lapack_int n, float vl, - float vu, lapack_int il, lapack_int iu, float abstol, - const float* d, const float* e, lapack_int* m, - lapack_int* nsplit, float* w, lapack_int* iblock, - lapack_int* isplit ); -lapack_int LAPACKE_dstebz( char range, char order, lapack_int n, double vl, - double vu, lapack_int il, lapack_int iu, - double abstol, const double* d, const double* e, - lapack_int* m, lapack_int* nsplit, double* w, - lapack_int* iblock, lapack_int* isplit ); - -lapack_int LAPACKE_sstedc( int matrix_order, char compz, lapack_int n, float* d, - float* e, float* z, lapack_int ldz ); -lapack_int LAPACKE_dstedc( int matrix_order, char compz, lapack_int n, - double* d, double* e, double* z, lapack_int ldz ); -lapack_int LAPACKE_cstedc( int matrix_order, char compz, lapack_int n, float* d, - float* e, lapack_complex_float* z, lapack_int ldz ); -lapack_int LAPACKE_zstedc( int matrix_order, char compz, lapack_int n, - double* d, double* e, lapack_complex_double* z, - lapack_int ldz ); - -lapack_int LAPACKE_sstegr( int matrix_order, char jobz, char range, - lapack_int n, float* d, float* e, float vl, float vu, - lapack_int il, lapack_int iu, float abstol, - lapack_int* m, float* w, float* z, lapack_int ldz, - lapack_int* isuppz ); -lapack_int LAPACKE_dstegr( int matrix_order, char jobz, char range, - lapack_int n, double* d, double* e, double vl, - double vu, lapack_int il, lapack_int iu, - double abstol, lapack_int* m, double* w, double* z, - lapack_int ldz, lapack_int* isuppz ); -lapack_int LAPACKE_cstegr( int matrix_order, char jobz, char range, - lapack_int n, float* d, float* e, float vl, float vu, - lapack_int il, lapack_int iu, float abstol, - lapack_int* m, float* w, lapack_complex_float* z, - lapack_int ldz, lapack_int* isuppz ); -lapack_int LAPACKE_zstegr( int matrix_order, char jobz, char range, - lapack_int n, double* d, double* e, double vl, - double vu, lapack_int il, lapack_int iu, - double abstol, lapack_int* m, double* w, - lapack_complex_double* z, lapack_int ldz, - lapack_int* isuppz ); - -lapack_int LAPACKE_sstein( int matrix_order, lapack_int n, const float* d, - const float* e, lapack_int m, const float* w, - const lapack_int* iblock, const lapack_int* isplit, - float* z, lapack_int ldz, lapack_int* ifailv ); -lapack_int LAPACKE_dstein( int matrix_order, lapack_int n, const double* d, - const double* e, lapack_int m, const double* w, - const lapack_int* iblock, const lapack_int* isplit, - double* z, lapack_int ldz, lapack_int* ifailv ); -lapack_int LAPACKE_cstein( int matrix_order, lapack_int n, const float* d, - const float* e, lapack_int m, const float* w, - const lapack_int* iblock, const lapack_int* isplit, - lapack_complex_float* z, lapack_int ldz, - lapack_int* ifailv ); -lapack_int LAPACKE_zstein( int matrix_order, lapack_int n, const double* d, - const double* e, lapack_int m, const double* w, - const lapack_int* iblock, const lapack_int* isplit, - lapack_complex_double* z, lapack_int ldz, - lapack_int* ifailv ); - -lapack_int LAPACKE_sstemr( int matrix_order, char jobz, char range, - lapack_int n, float* d, float* e, float vl, float vu, - lapack_int il, lapack_int iu, lapack_int* m, - float* w, float* z, lapack_int ldz, lapack_int nzc, - lapack_int* isuppz, lapack_logical* tryrac ); -lapack_int LAPACKE_dstemr( int matrix_order, char jobz, char range, - lapack_int n, double* d, double* e, double vl, - double vu, lapack_int il, lapack_int iu, - lapack_int* m, double* w, double* z, lapack_int ldz, - lapack_int nzc, lapack_int* isuppz, - lapack_logical* tryrac ); -lapack_int LAPACKE_cstemr( int matrix_order, char jobz, char range, - lapack_int n, float* d, float* e, float vl, float vu, - lapack_int il, lapack_int iu, lapack_int* m, - float* w, lapack_complex_float* z, lapack_int ldz, - lapack_int nzc, lapack_int* isuppz, - lapack_logical* tryrac ); -lapack_int LAPACKE_zstemr( int matrix_order, char jobz, char range, - lapack_int n, double* d, double* e, double vl, - double vu, lapack_int il, lapack_int iu, - lapack_int* m, double* w, lapack_complex_double* z, - lapack_int ldz, lapack_int nzc, lapack_int* isuppz, - lapack_logical* tryrac ); - -lapack_int LAPACKE_ssteqr( int matrix_order, char compz, lapack_int n, float* d, - float* e, float* z, lapack_int ldz ); -lapack_int LAPACKE_dsteqr( int matrix_order, char compz, lapack_int n, - double* d, double* e, double* z, lapack_int ldz ); -lapack_int LAPACKE_csteqr( int matrix_order, char compz, lapack_int n, float* d, - float* e, lapack_complex_float* z, lapack_int ldz ); -lapack_int LAPACKE_zsteqr( int matrix_order, char compz, lapack_int n, - double* d, double* e, lapack_complex_double* z, - lapack_int ldz ); - -lapack_int LAPACKE_ssterf( lapack_int n, float* d, float* e ); -lapack_int LAPACKE_dsterf( lapack_int n, double* d, double* e ); - -lapack_int LAPACKE_sstev( int matrix_order, char jobz, lapack_int n, float* d, - float* e, float* z, lapack_int ldz ); -lapack_int LAPACKE_dstev( int matrix_order, char jobz, lapack_int n, double* d, - double* e, double* z, lapack_int ldz ); - -lapack_int LAPACKE_sstevd( int matrix_order, char jobz, lapack_int n, float* d, - float* e, float* z, lapack_int ldz ); -lapack_int LAPACKE_dstevd( int matrix_order, char jobz, lapack_int n, double* d, - double* e, double* z, lapack_int ldz ); - -lapack_int LAPACKE_sstevr( int matrix_order, char jobz, char range, - lapack_int n, float* d, float* e, float vl, float vu, - lapack_int il, lapack_int iu, float abstol, - lapack_int* m, float* w, float* z, lapack_int ldz, - lapack_int* isuppz ); -lapack_int LAPACKE_dstevr( int matrix_order, char jobz, char range, - lapack_int n, double* d, double* e, double vl, - double vu, lapack_int il, lapack_int iu, - double abstol, lapack_int* m, double* w, double* z, - lapack_int ldz, lapack_int* isuppz ); - -lapack_int LAPACKE_sstevx( int matrix_order, char jobz, char range, - lapack_int n, float* d, float* e, float vl, float vu, - lapack_int il, lapack_int iu, float abstol, - lapack_int* m, float* w, float* z, lapack_int ldz, - lapack_int* ifail ); -lapack_int LAPACKE_dstevx( int matrix_order, char jobz, char range, - lapack_int n, double* d, double* e, double vl, - double vu, lapack_int il, lapack_int iu, - double abstol, lapack_int* m, double* w, double* z, - lapack_int ldz, lapack_int* ifail ); - -lapack_int LAPACKE_ssycon( int matrix_order, char uplo, lapack_int n, - const float* a, lapack_int lda, - const lapack_int* ipiv, float anorm, float* rcond ); -lapack_int LAPACKE_dsycon( int matrix_order, char uplo, lapack_int n, - const double* a, lapack_int lda, - const lapack_int* ipiv, double anorm, - double* rcond ); -lapack_int LAPACKE_csycon( int matrix_order, char uplo, lapack_int n, - const lapack_complex_float* a, lapack_int lda, - const lapack_int* ipiv, float anorm, float* rcond ); -lapack_int LAPACKE_zsycon( int matrix_order, char uplo, lapack_int n, - const lapack_complex_double* a, lapack_int lda, - const lapack_int* ipiv, double anorm, - double* rcond ); - -lapack_int LAPACKE_ssyequb( int matrix_order, char uplo, lapack_int n, - const float* a, lapack_int lda, float* s, - float* scond, float* amax ); -lapack_int LAPACKE_dsyequb( int matrix_order, char uplo, lapack_int n, - const double* a, lapack_int lda, double* s, - double* scond, double* amax ); -lapack_int LAPACKE_csyequb( int matrix_order, char uplo, lapack_int n, - const lapack_complex_float* a, lapack_int lda, - float* s, float* scond, float* amax ); -lapack_int LAPACKE_zsyequb( int matrix_order, char uplo, lapack_int n, - const lapack_complex_double* a, lapack_int lda, - double* s, double* scond, double* amax ); - -lapack_int LAPACKE_ssyev( int matrix_order, char jobz, char uplo, lapack_int n, - float* a, lapack_int lda, float* w ); -lapack_int LAPACKE_dsyev( int matrix_order, char jobz, char uplo, lapack_int n, - double* a, lapack_int lda, double* w ); - -lapack_int LAPACKE_ssyevd( int matrix_order, char jobz, char uplo, lapack_int n, - float* a, lapack_int lda, float* w ); -lapack_int LAPACKE_dsyevd( int matrix_order, char jobz, char uplo, lapack_int n, - double* a, lapack_int lda, double* w ); - -lapack_int LAPACKE_ssyevr( int matrix_order, char jobz, char range, char uplo, - lapack_int n, float* a, lapack_int lda, float vl, - float vu, lapack_int il, lapack_int iu, float abstol, - lapack_int* m, float* w, float* z, lapack_int ldz, - lapack_int* isuppz ); -lapack_int LAPACKE_dsyevr( int matrix_order, char jobz, char range, char uplo, - lapack_int n, double* a, lapack_int lda, double vl, - double vu, lapack_int il, lapack_int iu, - double abstol, lapack_int* m, double* w, double* z, - lapack_int ldz, lapack_int* isuppz ); - -lapack_int LAPACKE_ssyevx( int matrix_order, char jobz, char range, char uplo, - lapack_int n, float* a, lapack_int lda, float vl, - float vu, lapack_int il, lapack_int iu, float abstol, - lapack_int* m, float* w, float* z, lapack_int ldz, - lapack_int* ifail ); -lapack_int LAPACKE_dsyevx( int matrix_order, char jobz, char range, char uplo, - lapack_int n, double* a, lapack_int lda, double vl, - double vu, lapack_int il, lapack_int iu, - double abstol, lapack_int* m, double* w, double* z, - lapack_int ldz, lapack_int* ifail ); - -lapack_int LAPACKE_ssygst( int matrix_order, lapack_int itype, char uplo, - lapack_int n, float* a, lapack_int lda, - const float* b, lapack_int ldb ); -lapack_int LAPACKE_dsygst( int matrix_order, lapack_int itype, char uplo, - lapack_int n, double* a, lapack_int lda, - const double* b, lapack_int ldb ); - -lapack_int LAPACKE_ssygv( int matrix_order, lapack_int itype, char jobz, - char uplo, lapack_int n, float* a, lapack_int lda, - float* b, lapack_int ldb, float* w ); -lapack_int LAPACKE_dsygv( int matrix_order, lapack_int itype, char jobz, - char uplo, lapack_int n, double* a, lapack_int lda, - double* b, lapack_int ldb, double* w ); - -lapack_int LAPACKE_ssygvd( int matrix_order, lapack_int itype, char jobz, - char uplo, lapack_int n, float* a, lapack_int lda, - float* b, lapack_int ldb, float* w ); -lapack_int LAPACKE_dsygvd( int matrix_order, lapack_int itype, char jobz, - char uplo, lapack_int n, double* a, lapack_int lda, - double* b, lapack_int ldb, double* w ); - -lapack_int LAPACKE_ssygvx( int matrix_order, lapack_int itype, char jobz, - char range, char uplo, lapack_int n, float* a, - lapack_int lda, float* b, lapack_int ldb, float vl, - float vu, lapack_int il, lapack_int iu, float abstol, - lapack_int* m, float* w, float* z, lapack_int ldz, - lapack_int* ifail ); -lapack_int LAPACKE_dsygvx( int matrix_order, lapack_int itype, char jobz, - char range, char uplo, lapack_int n, double* a, - lapack_int lda, double* b, lapack_int ldb, double vl, - double vu, lapack_int il, lapack_int iu, - double abstol, lapack_int* m, double* w, double* z, - lapack_int ldz, lapack_int* ifail ); - -lapack_int LAPACKE_ssyrfs( int matrix_order, char uplo, lapack_int n, - lapack_int nrhs, const float* a, lapack_int lda, - const float* af, lapack_int ldaf, - const lapack_int* ipiv, const float* b, - lapack_int ldb, float* x, lapack_int ldx, - float* ferr, float* berr ); -lapack_int LAPACKE_dsyrfs( int matrix_order, char uplo, lapack_int n, - lapack_int nrhs, const double* a, lapack_int lda, - const double* af, lapack_int ldaf, - const lapack_int* ipiv, const double* b, - lapack_int ldb, double* x, lapack_int ldx, - double* ferr, double* berr ); -lapack_int LAPACKE_csyrfs( int matrix_order, char uplo, lapack_int n, - lapack_int nrhs, const lapack_complex_float* a, - lapack_int lda, const lapack_complex_float* af, - lapack_int ldaf, const lapack_int* ipiv, - const lapack_complex_float* b, lapack_int ldb, - lapack_complex_float* x, lapack_int ldx, float* ferr, - float* berr ); -lapack_int LAPACKE_zsyrfs( int matrix_order, char uplo, lapack_int n, - lapack_int nrhs, const lapack_complex_double* a, - lapack_int lda, const lapack_complex_double* af, - lapack_int ldaf, const lapack_int* ipiv, - const lapack_complex_double* b, lapack_int ldb, - lapack_complex_double* x, lapack_int ldx, - double* ferr, double* berr ); - -lapack_int LAPACKE_ssyrfsx( int matrix_order, char uplo, char equed, - lapack_int n, lapack_int nrhs, const float* a, - lapack_int lda, const float* af, lapack_int ldaf, - const lapack_int* ipiv, const float* s, - const float* b, lapack_int ldb, float* x, - lapack_int ldx, float* rcond, float* berr, - lapack_int n_err_bnds, float* err_bnds_norm, - float* err_bnds_comp, lapack_int nparams, - float* params ); -lapack_int LAPACKE_dsyrfsx( int matrix_order, char uplo, char equed, - lapack_int n, lapack_int nrhs, const double* a, - lapack_int lda, const double* af, lapack_int ldaf, - const lapack_int* ipiv, const double* s, - const double* b, lapack_int ldb, double* x, - lapack_int ldx, double* rcond, double* berr, - lapack_int n_err_bnds, double* err_bnds_norm, - double* err_bnds_comp, lapack_int nparams, - double* params ); -lapack_int LAPACKE_csyrfsx( int matrix_order, char uplo, char equed, - lapack_int n, lapack_int nrhs, - const lapack_complex_float* a, lapack_int lda, - const lapack_complex_float* af, lapack_int ldaf, - const lapack_int* ipiv, const float* s, - const lapack_complex_float* b, lapack_int ldb, - lapack_complex_float* x, lapack_int ldx, - float* rcond, float* berr, lapack_int n_err_bnds, - float* err_bnds_norm, float* err_bnds_comp, - lapack_int nparams, float* params ); -lapack_int LAPACKE_zsyrfsx( int matrix_order, char uplo, char equed, - lapack_int n, lapack_int nrhs, - const lapack_complex_double* a, lapack_int lda, - const lapack_complex_double* af, lapack_int ldaf, - const lapack_int* ipiv, const double* s, - const lapack_complex_double* b, lapack_int ldb, - lapack_complex_double* x, lapack_int ldx, - double* rcond, double* berr, lapack_int n_err_bnds, - double* err_bnds_norm, double* err_bnds_comp, - lapack_int nparams, double* params ); - -lapack_int LAPACKE_ssysv( int matrix_order, char uplo, lapack_int n, - lapack_int nrhs, float* a, lapack_int lda, - lapack_int* ipiv, float* b, lapack_int ldb ); -lapack_int LAPACKE_dsysv( int matrix_order, char uplo, lapack_int n, - lapack_int nrhs, double* a, lapack_int lda, - lapack_int* ipiv, double* b, lapack_int ldb ); -lapack_int LAPACKE_csysv( int matrix_order, char uplo, lapack_int n, - lapack_int nrhs, lapack_complex_float* a, - lapack_int lda, lapack_int* ipiv, - lapack_complex_float* b, lapack_int ldb ); -lapack_int LAPACKE_zsysv( int matrix_order, char uplo, lapack_int n, - lapack_int nrhs, lapack_complex_double* a, - lapack_int lda, lapack_int* ipiv, - lapack_complex_double* b, lapack_int ldb ); - -lapack_int LAPACKE_ssysvx( int matrix_order, char fact, char uplo, lapack_int n, - lapack_int nrhs, const float* a, lapack_int lda, - float* af, lapack_int ldaf, lapack_int* ipiv, - const float* b, lapack_int ldb, float* x, - lapack_int ldx, float* rcond, float* ferr, - float* berr ); -lapack_int LAPACKE_dsysvx( int matrix_order, char fact, char uplo, lapack_int n, - lapack_int nrhs, const double* a, lapack_int lda, - double* af, lapack_int ldaf, lapack_int* ipiv, - const double* b, lapack_int ldb, double* x, - lapack_int ldx, double* rcond, double* ferr, - double* berr ); -lapack_int LAPACKE_csysvx( int matrix_order, char fact, char uplo, lapack_int n, - lapack_int nrhs, const lapack_complex_float* a, - lapack_int lda, lapack_complex_float* af, - lapack_int ldaf, lapack_int* ipiv, - const lapack_complex_float* b, lapack_int ldb, - lapack_complex_float* x, lapack_int ldx, - float* rcond, float* ferr, float* berr ); -lapack_int LAPACKE_zsysvx( int matrix_order, char fact, char uplo, lapack_int n, - lapack_int nrhs, const lapack_complex_double* a, - lapack_int lda, lapack_complex_double* af, - lapack_int ldaf, lapack_int* ipiv, - const lapack_complex_double* b, lapack_int ldb, - lapack_complex_double* x, lapack_int ldx, - double* rcond, double* ferr, double* berr ); - -lapack_int LAPACKE_ssysvxx( int matrix_order, char fact, char uplo, - lapack_int n, lapack_int nrhs, float* a, - lapack_int lda, float* af, lapack_int ldaf, - lapack_int* ipiv, char* equed, float* s, float* b, - lapack_int ldb, float* x, lapack_int ldx, - float* rcond, float* rpvgrw, float* berr, - lapack_int n_err_bnds, float* err_bnds_norm, - float* err_bnds_comp, lapack_int nparams, - float* params ); -lapack_int LAPACKE_dsysvxx( int matrix_order, char fact, char uplo, - lapack_int n, lapack_int nrhs, double* a, - lapack_int lda, double* af, lapack_int ldaf, - lapack_int* ipiv, char* equed, double* s, double* b, - lapack_int ldb, double* x, lapack_int ldx, - double* rcond, double* rpvgrw, double* berr, - lapack_int n_err_bnds, double* err_bnds_norm, - double* err_bnds_comp, lapack_int nparams, - double* params ); -lapack_int LAPACKE_csysvxx( int matrix_order, char fact, char uplo, - lapack_int n, lapack_int nrhs, - lapack_complex_float* a, lapack_int lda, - lapack_complex_float* af, lapack_int ldaf, - lapack_int* ipiv, char* equed, float* s, - lapack_complex_float* b, lapack_int ldb, - lapack_complex_float* x, lapack_int ldx, - float* rcond, float* rpvgrw, float* berr, - lapack_int n_err_bnds, float* err_bnds_norm, - float* err_bnds_comp, lapack_int nparams, - float* params ); -lapack_int LAPACKE_zsysvxx( int matrix_order, char fact, char uplo, - lapack_int n, lapack_int nrhs, - lapack_complex_double* a, lapack_int lda, - lapack_complex_double* af, lapack_int ldaf, - lapack_int* ipiv, char* equed, double* s, - lapack_complex_double* b, lapack_int ldb, - lapack_complex_double* x, lapack_int ldx, - double* rcond, double* rpvgrw, double* berr, - lapack_int n_err_bnds, double* err_bnds_norm, - double* err_bnds_comp, lapack_int nparams, - double* params ); - -lapack_int LAPACKE_ssytrd( int matrix_order, char uplo, lapack_int n, float* a, - lapack_int lda, float* d, float* e, float* tau ); -lapack_int LAPACKE_dsytrd( int matrix_order, char uplo, lapack_int n, double* a, - lapack_int lda, double* d, double* e, double* tau ); - -lapack_int LAPACKE_ssytrf( int matrix_order, char uplo, lapack_int n, float* a, - lapack_int lda, lapack_int* ipiv ); -lapack_int LAPACKE_dsytrf( int matrix_order, char uplo, lapack_int n, double* a, - lapack_int lda, lapack_int* ipiv ); -lapack_int LAPACKE_csytrf( int matrix_order, char uplo, lapack_int n, - lapack_complex_float* a, lapack_int lda, - lapack_int* ipiv ); -lapack_int LAPACKE_zsytrf( int matrix_order, char uplo, lapack_int n, - lapack_complex_double* a, lapack_int lda, - lapack_int* ipiv ); - -lapack_int LAPACKE_ssytri( int matrix_order, char uplo, lapack_int n, float* a, - lapack_int lda, const lapack_int* ipiv ); -lapack_int LAPACKE_dsytri( int matrix_order, char uplo, lapack_int n, double* a, - lapack_int lda, const lapack_int* ipiv ); -lapack_int LAPACKE_csytri( int matrix_order, char uplo, lapack_int n, - lapack_complex_float* a, lapack_int lda, - const lapack_int* ipiv ); -lapack_int LAPACKE_zsytri( int matrix_order, char uplo, lapack_int n, - lapack_complex_double* a, lapack_int lda, - const lapack_int* ipiv ); - -lapack_int LAPACKE_ssytrs( int matrix_order, char uplo, lapack_int n, - lapack_int nrhs, const float* a, lapack_int lda, - const lapack_int* ipiv, float* b, lapack_int ldb ); -lapack_int LAPACKE_dsytrs( int matrix_order, char uplo, lapack_int n, - lapack_int nrhs, const double* a, lapack_int lda, - const lapack_int* ipiv, double* b, lapack_int ldb ); -lapack_int LAPACKE_csytrs( int matrix_order, char uplo, lapack_int n, - lapack_int nrhs, const lapack_complex_float* a, - lapack_int lda, const lapack_int* ipiv, - lapack_complex_float* b, lapack_int ldb ); -lapack_int LAPACKE_zsytrs( int matrix_order, char uplo, lapack_int n, - lapack_int nrhs, const lapack_complex_double* a, - lapack_int lda, const lapack_int* ipiv, - lapack_complex_double* b, lapack_int ldb ); - -lapack_int LAPACKE_stbcon( int matrix_order, char norm, char uplo, char diag, - lapack_int n, lapack_int kd, const float* ab, - lapack_int ldab, float* rcond ); -lapack_int LAPACKE_dtbcon( int matrix_order, char norm, char uplo, char diag, - lapack_int n, lapack_int kd, const double* ab, - lapack_int ldab, double* rcond ); -lapack_int LAPACKE_ctbcon( int matrix_order, char norm, char uplo, char diag, - lapack_int n, lapack_int kd, - const lapack_complex_float* ab, lapack_int ldab, - float* rcond ); -lapack_int LAPACKE_ztbcon( int matrix_order, char norm, char uplo, char diag, - lapack_int n, lapack_int kd, - const lapack_complex_double* ab, lapack_int ldab, - double* rcond ); - -lapack_int LAPACKE_stbrfs( int matrix_order, char uplo, char trans, char diag, - lapack_int n, lapack_int kd, lapack_int nrhs, - const float* ab, lapack_int ldab, const float* b, - lapack_int ldb, const float* x, lapack_int ldx, - float* ferr, float* berr ); -lapack_int LAPACKE_dtbrfs( int matrix_order, char uplo, char trans, char diag, - lapack_int n, lapack_int kd, lapack_int nrhs, - const double* ab, lapack_int ldab, const double* b, - lapack_int ldb, const double* x, lapack_int ldx, - double* ferr, double* berr ); -lapack_int LAPACKE_ctbrfs( int matrix_order, char uplo, char trans, char diag, - lapack_int n, lapack_int kd, lapack_int nrhs, - const lapack_complex_float* ab, lapack_int ldab, - const lapack_complex_float* b, lapack_int ldb, - const lapack_complex_float* x, lapack_int ldx, - float* ferr, float* berr ); -lapack_int LAPACKE_ztbrfs( int matrix_order, char uplo, char trans, char diag, - lapack_int n, lapack_int kd, lapack_int nrhs, - const lapack_complex_double* ab, lapack_int ldab, - const lapack_complex_double* b, lapack_int ldb, - const lapack_complex_double* x, lapack_int ldx, - double* ferr, double* berr ); - -lapack_int LAPACKE_stbtrs( int matrix_order, char uplo, char trans, char diag, - lapack_int n, lapack_int kd, lapack_int nrhs, - const float* ab, lapack_int ldab, float* b, - lapack_int ldb ); -lapack_int LAPACKE_dtbtrs( int matrix_order, char uplo, char trans, char diag, - lapack_int n, lapack_int kd, lapack_int nrhs, - const double* ab, lapack_int ldab, double* b, - lapack_int ldb ); -lapack_int LAPACKE_ctbtrs( int matrix_order, char uplo, char trans, char diag, - lapack_int n, lapack_int kd, lapack_int nrhs, - const lapack_complex_float* ab, lapack_int ldab, - lapack_complex_float* b, lapack_int ldb ); -lapack_int LAPACKE_ztbtrs( int matrix_order, char uplo, char trans, char diag, - lapack_int n, lapack_int kd, lapack_int nrhs, - const lapack_complex_double* ab, lapack_int ldab, - lapack_complex_double* b, lapack_int ldb ); - -lapack_int LAPACKE_stfsm( int matrix_order, char transr, char side, char uplo, - char trans, char diag, lapack_int m, lapack_int n, - float alpha, const float* a, float* b, - lapack_int ldb ); -lapack_int LAPACKE_dtfsm( int matrix_order, char transr, char side, char uplo, - char trans, char diag, lapack_int m, lapack_int n, - double alpha, const double* a, double* b, - lapack_int ldb ); -lapack_int LAPACKE_ctfsm( int matrix_order, char transr, char side, char uplo, - char trans, char diag, lapack_int m, lapack_int n, - lapack_complex_float alpha, - const lapack_complex_float* a, - lapack_complex_float* b, lapack_int ldb ); -lapack_int LAPACKE_ztfsm( int matrix_order, char transr, char side, char uplo, - char trans, char diag, lapack_int m, lapack_int n, - lapack_complex_double alpha, - const lapack_complex_double* a, - lapack_complex_double* b, lapack_int ldb ); - -lapack_int LAPACKE_stftri( int matrix_order, char transr, char uplo, char diag, - lapack_int n, float* a ); -lapack_int LAPACKE_dtftri( int matrix_order, char transr, char uplo, char diag, - lapack_int n, double* a ); -lapack_int LAPACKE_ctftri( int matrix_order, char transr, char uplo, char diag, - lapack_int n, lapack_complex_float* a ); -lapack_int LAPACKE_ztftri( int matrix_order, char transr, char uplo, char diag, - lapack_int n, lapack_complex_double* a ); - -lapack_int LAPACKE_stfttp( int matrix_order, char transr, char uplo, - lapack_int n, const float* arf, float* ap ); -lapack_int LAPACKE_dtfttp( int matrix_order, char transr, char uplo, - lapack_int n, const double* arf, double* ap ); -lapack_int LAPACKE_ctfttp( int matrix_order, char transr, char uplo, - lapack_int n, const lapack_complex_float* arf, - lapack_complex_float* ap ); -lapack_int LAPACKE_ztfttp( int matrix_order, char transr, char uplo, - lapack_int n, const lapack_complex_double* arf, - lapack_complex_double* ap ); - -lapack_int LAPACKE_stfttr( int matrix_order, char transr, char uplo, - lapack_int n, const float* arf, float* a, - lapack_int lda ); -lapack_int LAPACKE_dtfttr( int matrix_order, char transr, char uplo, - lapack_int n, const double* arf, double* a, - lapack_int lda ); -lapack_int LAPACKE_ctfttr( int matrix_order, char transr, char uplo, - lapack_int n, const lapack_complex_float* arf, - lapack_complex_float* a, lapack_int lda ); -lapack_int LAPACKE_ztfttr( int matrix_order, char transr, char uplo, - lapack_int n, const lapack_complex_double* arf, - lapack_complex_double* a, lapack_int lda ); - -lapack_int LAPACKE_stgevc( int matrix_order, char side, char howmny, - const lapack_logical* select, lapack_int n, - const float* s, lapack_int lds, const float* p, - lapack_int ldp, float* vl, lapack_int ldvl, - float* vr, lapack_int ldvr, lapack_int mm, - lapack_int* m ); -lapack_int LAPACKE_dtgevc( int matrix_order, char side, char howmny, - const lapack_logical* select, lapack_int n, - const double* s, lapack_int lds, const double* p, - lapack_int ldp, double* vl, lapack_int ldvl, - double* vr, lapack_int ldvr, lapack_int mm, - lapack_int* m ); -lapack_int LAPACKE_ctgevc( int matrix_order, char side, char howmny, - const lapack_logical* select, lapack_int n, - const lapack_complex_float* s, lapack_int lds, - const lapack_complex_float* p, lapack_int ldp, - lapack_complex_float* vl, lapack_int ldvl, - lapack_complex_float* vr, lapack_int ldvr, - lapack_int mm, lapack_int* m ); -lapack_int LAPACKE_ztgevc( int matrix_order, char side, char howmny, - const lapack_logical* select, lapack_int n, - const lapack_complex_double* s, lapack_int lds, - const lapack_complex_double* p, lapack_int ldp, - lapack_complex_double* vl, lapack_int ldvl, - lapack_complex_double* vr, lapack_int ldvr, - lapack_int mm, lapack_int* m ); - -lapack_int LAPACKE_stgexc( int matrix_order, lapack_logical wantq, - lapack_logical wantz, lapack_int n, float* a, - lapack_int lda, float* b, lapack_int ldb, float* q, - lapack_int ldq, float* z, lapack_int ldz, - lapack_int* ifst, lapack_int* ilst ); -lapack_int LAPACKE_dtgexc( int matrix_order, lapack_logical wantq, - lapack_logical wantz, lapack_int n, double* a, - lapack_int lda, double* b, lapack_int ldb, double* q, - lapack_int ldq, double* z, lapack_int ldz, - lapack_int* ifst, lapack_int* ilst ); -lapack_int LAPACKE_ctgexc( int matrix_order, lapack_logical wantq, - lapack_logical wantz, lapack_int n, - lapack_complex_float* a, lapack_int lda, - lapack_complex_float* b, lapack_int ldb, - lapack_complex_float* q, lapack_int ldq, - lapack_complex_float* z, lapack_int ldz, - lapack_int ifst, lapack_int ilst ); -lapack_int LAPACKE_ztgexc( int matrix_order, lapack_logical wantq, - lapack_logical wantz, lapack_int n, - lapack_complex_double* a, lapack_int lda, - lapack_complex_double* b, lapack_int ldb, - lapack_complex_double* q, lapack_int ldq, - lapack_complex_double* z, lapack_int ldz, - lapack_int ifst, lapack_int ilst ); - -lapack_int LAPACKE_stgsen( int matrix_order, lapack_int ijob, - lapack_logical wantq, lapack_logical wantz, - const lapack_logical* select, lapack_int n, float* a, - lapack_int lda, float* b, lapack_int ldb, - float* alphar, float* alphai, float* beta, float* q, - lapack_int ldq, float* z, lapack_int ldz, - lapack_int* m, float* pl, float* pr, float* dif ); -lapack_int LAPACKE_dtgsen( int matrix_order, lapack_int ijob, - lapack_logical wantq, lapack_logical wantz, - const lapack_logical* select, lapack_int n, - double* a, lapack_int lda, double* b, lapack_int ldb, - double* alphar, double* alphai, double* beta, - double* q, lapack_int ldq, double* z, lapack_int ldz, - lapack_int* m, double* pl, double* pr, double* dif ); -lapack_int LAPACKE_ctgsen( int matrix_order, lapack_int ijob, - lapack_logical wantq, lapack_logical wantz, - const lapack_logical* select, lapack_int n, - lapack_complex_float* a, lapack_int lda, - lapack_complex_float* b, lapack_int ldb, - lapack_complex_float* alpha, - lapack_complex_float* beta, lapack_complex_float* q, - lapack_int ldq, lapack_complex_float* z, - lapack_int ldz, lapack_int* m, float* pl, float* pr, - float* dif ); -lapack_int LAPACKE_ztgsen( int matrix_order, lapack_int ijob, - lapack_logical wantq, lapack_logical wantz, - const lapack_logical* select, lapack_int n, - lapack_complex_double* a, lapack_int lda, - lapack_complex_double* b, lapack_int ldb, - lapack_complex_double* alpha, - lapack_complex_double* beta, - lapack_complex_double* q, lapack_int ldq, - lapack_complex_double* z, lapack_int ldz, - lapack_int* m, double* pl, double* pr, double* dif ); - -lapack_int LAPACKE_stgsja( int matrix_order, char jobu, char jobv, char jobq, - lapack_int m, lapack_int p, lapack_int n, - lapack_int k, lapack_int l, float* a, lapack_int lda, - float* b, lapack_int ldb, float tola, float tolb, - float* alpha, float* beta, float* u, lapack_int ldu, - float* v, lapack_int ldv, float* q, lapack_int ldq, - lapack_int* ncycle ); -lapack_int LAPACKE_dtgsja( int matrix_order, char jobu, char jobv, char jobq, - lapack_int m, lapack_int p, lapack_int n, - lapack_int k, lapack_int l, double* a, - lapack_int lda, double* b, lapack_int ldb, - double tola, double tolb, double* alpha, - double* beta, double* u, lapack_int ldu, double* v, - lapack_int ldv, double* q, lapack_int ldq, - lapack_int* ncycle ); -lapack_int LAPACKE_ctgsja( int matrix_order, char jobu, char jobv, char jobq, - lapack_int m, lapack_int p, lapack_int n, - lapack_int k, lapack_int l, lapack_complex_float* a, - lapack_int lda, lapack_complex_float* b, - lapack_int ldb, float tola, float tolb, float* alpha, - float* beta, lapack_complex_float* u, lapack_int ldu, - lapack_complex_float* v, lapack_int ldv, - lapack_complex_float* q, lapack_int ldq, - lapack_int* ncycle ); -lapack_int LAPACKE_ztgsja( int matrix_order, char jobu, char jobv, char jobq, - lapack_int m, lapack_int p, lapack_int n, - lapack_int k, lapack_int l, lapack_complex_double* a, - lapack_int lda, lapack_complex_double* b, - lapack_int ldb, double tola, double tolb, - double* alpha, double* beta, - lapack_complex_double* u, lapack_int ldu, - lapack_complex_double* v, lapack_int ldv, - lapack_complex_double* q, lapack_int ldq, - lapack_int* ncycle ); - -lapack_int LAPACKE_stgsna( int matrix_order, char job, char howmny, - const lapack_logical* select, lapack_int n, - const float* a, lapack_int lda, const float* b, - lapack_int ldb, const float* vl, lapack_int ldvl, - const float* vr, lapack_int ldvr, float* s, - float* dif, lapack_int mm, lapack_int* m ); -lapack_int LAPACKE_dtgsna( int matrix_order, char job, char howmny, - const lapack_logical* select, lapack_int n, - const double* a, lapack_int lda, const double* b, - lapack_int ldb, const double* vl, lapack_int ldvl, - const double* vr, lapack_int ldvr, double* s, - double* dif, lapack_int mm, lapack_int* m ); -lapack_int LAPACKE_ctgsna( int matrix_order, char job, char howmny, - const lapack_logical* select, lapack_int n, - const lapack_complex_float* a, lapack_int lda, - const lapack_complex_float* b, lapack_int ldb, - const lapack_complex_float* vl, lapack_int ldvl, - const lapack_complex_float* vr, lapack_int ldvr, - float* s, float* dif, lapack_int mm, lapack_int* m ); -lapack_int LAPACKE_ztgsna( int matrix_order, char job, char howmny, - const lapack_logical* select, lapack_int n, - const lapack_complex_double* a, lapack_int lda, - const lapack_complex_double* b, lapack_int ldb, - const lapack_complex_double* vl, lapack_int ldvl, - const lapack_complex_double* vr, lapack_int ldvr, - double* s, double* dif, lapack_int mm, - lapack_int* m ); - -lapack_int LAPACKE_stgsyl( int matrix_order, char trans, lapack_int ijob, - lapack_int m, lapack_int n, const float* a, - lapack_int lda, const float* b, lapack_int ldb, - float* c, lapack_int ldc, const float* d, - lapack_int ldd, const float* e, lapack_int lde, - float* f, lapack_int ldf, float* scale, float* dif ); -lapack_int LAPACKE_dtgsyl( int matrix_order, char trans, lapack_int ijob, - lapack_int m, lapack_int n, const double* a, - lapack_int lda, const double* b, lapack_int ldb, - double* c, lapack_int ldc, const double* d, - lapack_int ldd, const double* e, lapack_int lde, - double* f, lapack_int ldf, double* scale, - double* dif ); -lapack_int LAPACKE_ctgsyl( int matrix_order, char trans, lapack_int ijob, - lapack_int m, lapack_int n, - const lapack_complex_float* a, lapack_int lda, - const lapack_complex_float* b, lapack_int ldb, - lapack_complex_float* c, lapack_int ldc, - const lapack_complex_float* d, lapack_int ldd, - const lapack_complex_float* e, lapack_int lde, - lapack_complex_float* f, lapack_int ldf, - float* scale, float* dif ); -lapack_int LAPACKE_ztgsyl( int matrix_order, char trans, lapack_int ijob, - lapack_int m, lapack_int n, - const lapack_complex_double* a, lapack_int lda, - const lapack_complex_double* b, lapack_int ldb, - lapack_complex_double* c, lapack_int ldc, - const lapack_complex_double* d, lapack_int ldd, - const lapack_complex_double* e, lapack_int lde, - lapack_complex_double* f, lapack_int ldf, - double* scale, double* dif ); - -lapack_int LAPACKE_stpcon( int matrix_order, char norm, char uplo, char diag, - lapack_int n, const float* ap, float* rcond ); -lapack_int LAPACKE_dtpcon( int matrix_order, char norm, char uplo, char diag, - lapack_int n, const double* ap, double* rcond ); -lapack_int LAPACKE_ctpcon( int matrix_order, char norm, char uplo, char diag, - lapack_int n, const lapack_complex_float* ap, - float* rcond ); -lapack_int LAPACKE_ztpcon( int matrix_order, char norm, char uplo, char diag, - lapack_int n, const lapack_complex_double* ap, - double* rcond ); - -lapack_int LAPACKE_stprfs( int matrix_order, char uplo, char trans, char diag, - lapack_int n, lapack_int nrhs, const float* ap, - const float* b, lapack_int ldb, const float* x, - lapack_int ldx, float* ferr, float* berr ); -lapack_int LAPACKE_dtprfs( int matrix_order, char uplo, char trans, char diag, - lapack_int n, lapack_int nrhs, const double* ap, - const double* b, lapack_int ldb, const double* x, - lapack_int ldx, double* ferr, double* berr ); -lapack_int LAPACKE_ctprfs( int matrix_order, char uplo, char trans, char diag, - lapack_int n, lapack_int nrhs, - const lapack_complex_float* ap, - const lapack_complex_float* b, lapack_int ldb, - const lapack_complex_float* x, lapack_int ldx, - float* ferr, float* berr ); -lapack_int LAPACKE_ztprfs( int matrix_order, char uplo, char trans, char diag, - lapack_int n, lapack_int nrhs, - const lapack_complex_double* ap, - const lapack_complex_double* b, lapack_int ldb, - const lapack_complex_double* x, lapack_int ldx, - double* ferr, double* berr ); - -lapack_int LAPACKE_stptri( int matrix_order, char uplo, char diag, lapack_int n, - float* ap ); -lapack_int LAPACKE_dtptri( int matrix_order, char uplo, char diag, lapack_int n, - double* ap ); -lapack_int LAPACKE_ctptri( int matrix_order, char uplo, char diag, lapack_int n, - lapack_complex_float* ap ); -lapack_int LAPACKE_ztptri( int matrix_order, char uplo, char diag, lapack_int n, - lapack_complex_double* ap ); - -lapack_int LAPACKE_stptrs( int matrix_order, char uplo, char trans, char diag, - lapack_int n, lapack_int nrhs, const float* ap, - float* b, lapack_int ldb ); -lapack_int LAPACKE_dtptrs( int matrix_order, char uplo, char trans, char diag, - lapack_int n, lapack_int nrhs, const double* ap, - double* b, lapack_int ldb ); -lapack_int LAPACKE_ctptrs( int matrix_order, char uplo, char trans, char diag, - lapack_int n, lapack_int nrhs, - const lapack_complex_float* ap, - lapack_complex_float* b, lapack_int ldb ); -lapack_int LAPACKE_ztptrs( int matrix_order, char uplo, char trans, char diag, - lapack_int n, lapack_int nrhs, - const lapack_complex_double* ap, - lapack_complex_double* b, lapack_int ldb ); - -lapack_int LAPACKE_stpttf( int matrix_order, char transr, char uplo, - lapack_int n, const float* ap, float* arf ); -lapack_int LAPACKE_dtpttf( int matrix_order, char transr, char uplo, - lapack_int n, const double* ap, double* arf ); -lapack_int LAPACKE_ctpttf( int matrix_order, char transr, char uplo, - lapack_int n, const lapack_complex_float* ap, - lapack_complex_float* arf ); -lapack_int LAPACKE_ztpttf( int matrix_order, char transr, char uplo, - lapack_int n, const lapack_complex_double* ap, - lapack_complex_double* arf ); - -lapack_int LAPACKE_stpttr( int matrix_order, char uplo, lapack_int n, - const float* ap, float* a, lapack_int lda ); -lapack_int LAPACKE_dtpttr( int matrix_order, char uplo, lapack_int n, - const double* ap, double* a, lapack_int lda ); -lapack_int LAPACKE_ctpttr( int matrix_order, char uplo, lapack_int n, - const lapack_complex_float* ap, - lapack_complex_float* a, lapack_int lda ); -lapack_int LAPACKE_ztpttr( int matrix_order, char uplo, lapack_int n, - const lapack_complex_double* ap, - lapack_complex_double* a, lapack_int lda ); - -lapack_int LAPACKE_strcon( int matrix_order, char norm, char uplo, char diag, - lapack_int n, const float* a, lapack_int lda, - float* rcond ); -lapack_int LAPACKE_dtrcon( int matrix_order, char norm, char uplo, char diag, - lapack_int n, const double* a, lapack_int lda, - double* rcond ); -lapack_int LAPACKE_ctrcon( int matrix_order, char norm, char uplo, char diag, - lapack_int n, const lapack_complex_float* a, - lapack_int lda, float* rcond ); -lapack_int LAPACKE_ztrcon( int matrix_order, char norm, char uplo, char diag, - lapack_int n, const lapack_complex_double* a, - lapack_int lda, double* rcond ); - -lapack_int LAPACKE_strevc( int matrix_order, char side, char howmny, - lapack_logical* select, lapack_int n, const float* t, - lapack_int ldt, float* vl, lapack_int ldvl, - float* vr, lapack_int ldvr, lapack_int mm, - lapack_int* m ); -lapack_int LAPACKE_dtrevc( int matrix_order, char side, char howmny, - lapack_logical* select, lapack_int n, - const double* t, lapack_int ldt, double* vl, - lapack_int ldvl, double* vr, lapack_int ldvr, - lapack_int mm, lapack_int* m ); -lapack_int LAPACKE_ctrevc( int matrix_order, char side, char howmny, - const lapack_logical* select, lapack_int n, - lapack_complex_float* t, lapack_int ldt, - lapack_complex_float* vl, lapack_int ldvl, - lapack_complex_float* vr, lapack_int ldvr, - lapack_int mm, lapack_int* m ); -lapack_int LAPACKE_ztrevc( int matrix_order, char side, char howmny, - const lapack_logical* select, lapack_int n, - lapack_complex_double* t, lapack_int ldt, - lapack_complex_double* vl, lapack_int ldvl, - lapack_complex_double* vr, lapack_int ldvr, - lapack_int mm, lapack_int* m ); - -lapack_int LAPACKE_strexc( int matrix_order, char compq, lapack_int n, float* t, - lapack_int ldt, float* q, lapack_int ldq, - lapack_int* ifst, lapack_int* ilst ); -lapack_int LAPACKE_dtrexc( int matrix_order, char compq, lapack_int n, - double* t, lapack_int ldt, double* q, lapack_int ldq, - lapack_int* ifst, lapack_int* ilst ); -lapack_int LAPACKE_ctrexc( int matrix_order, char compq, lapack_int n, - lapack_complex_float* t, lapack_int ldt, - lapack_complex_float* q, lapack_int ldq, - lapack_int ifst, lapack_int ilst ); -lapack_int LAPACKE_ztrexc( int matrix_order, char compq, lapack_int n, - lapack_complex_double* t, lapack_int ldt, - lapack_complex_double* q, lapack_int ldq, - lapack_int ifst, lapack_int ilst ); - -lapack_int LAPACKE_strrfs( int matrix_order, char uplo, char trans, char diag, - lapack_int n, lapack_int nrhs, const float* a, - lapack_int lda, const float* b, lapack_int ldb, - const float* x, lapack_int ldx, float* ferr, - float* berr ); -lapack_int LAPACKE_dtrrfs( int matrix_order, char uplo, char trans, char diag, - lapack_int n, lapack_int nrhs, const double* a, - lapack_int lda, const double* b, lapack_int ldb, - const double* x, lapack_int ldx, double* ferr, - double* berr ); -lapack_int LAPACKE_ctrrfs( int matrix_order, char uplo, char trans, char diag, - lapack_int n, lapack_int nrhs, - const lapack_complex_float* a, lapack_int lda, - const lapack_complex_float* b, lapack_int ldb, - const lapack_complex_float* x, lapack_int ldx, - float* ferr, float* berr ); -lapack_int LAPACKE_ztrrfs( int matrix_order, char uplo, char trans, char diag, - lapack_int n, lapack_int nrhs, - const lapack_complex_double* a, lapack_int lda, - const lapack_complex_double* b, lapack_int ldb, - const lapack_complex_double* x, lapack_int ldx, - double* ferr, double* berr ); - -lapack_int LAPACKE_strsen( int matrix_order, char job, char compq, - const lapack_logical* select, lapack_int n, float* t, - lapack_int ldt, float* q, lapack_int ldq, float* wr, - float* wi, lapack_int* m, float* s, float* sep ); -lapack_int LAPACKE_dtrsen( int matrix_order, char job, char compq, - const lapack_logical* select, lapack_int n, - double* t, lapack_int ldt, double* q, lapack_int ldq, - double* wr, double* wi, lapack_int* m, double* s, - double* sep ); -lapack_int LAPACKE_ctrsen( int matrix_order, char job, char compq, - const lapack_logical* select, lapack_int n, - lapack_complex_float* t, lapack_int ldt, - lapack_complex_float* q, lapack_int ldq, - lapack_complex_float* w, lapack_int* m, float* s, - float* sep ); -lapack_int LAPACKE_ztrsen( int matrix_order, char job, char compq, - const lapack_logical* select, lapack_int n, - lapack_complex_double* t, lapack_int ldt, - lapack_complex_double* q, lapack_int ldq, - lapack_complex_double* w, lapack_int* m, double* s, - double* sep ); - -lapack_int LAPACKE_strsna( int matrix_order, char job, char howmny, - const lapack_logical* select, lapack_int n, - const float* t, lapack_int ldt, const float* vl, - lapack_int ldvl, const float* vr, lapack_int ldvr, - float* s, float* sep, lapack_int mm, lapack_int* m ); -lapack_int LAPACKE_dtrsna( int matrix_order, char job, char howmny, - const lapack_logical* select, lapack_int n, - const double* t, lapack_int ldt, const double* vl, - lapack_int ldvl, const double* vr, lapack_int ldvr, - double* s, double* sep, lapack_int mm, - lapack_int* m ); -lapack_int LAPACKE_ctrsna( int matrix_order, char job, char howmny, - const lapack_logical* select, lapack_int n, - const lapack_complex_float* t, lapack_int ldt, - const lapack_complex_float* vl, lapack_int ldvl, - const lapack_complex_float* vr, lapack_int ldvr, - float* s, float* sep, lapack_int mm, lapack_int* m ); -lapack_int LAPACKE_ztrsna( int matrix_order, char job, char howmny, - const lapack_logical* select, lapack_int n, - const lapack_complex_double* t, lapack_int ldt, - const lapack_complex_double* vl, lapack_int ldvl, - const lapack_complex_double* vr, lapack_int ldvr, - double* s, double* sep, lapack_int mm, - lapack_int* m ); - -lapack_int LAPACKE_strsyl( int matrix_order, char trana, char tranb, - lapack_int isgn, lapack_int m, lapack_int n, - const float* a, lapack_int lda, const float* b, - lapack_int ldb, float* c, lapack_int ldc, - float* scale ); -lapack_int LAPACKE_dtrsyl( int matrix_order, char trana, char tranb, - lapack_int isgn, lapack_int m, lapack_int n, - const double* a, lapack_int lda, const double* b, - lapack_int ldb, double* c, lapack_int ldc, - double* scale ); -lapack_int LAPACKE_ctrsyl( int matrix_order, char trana, char tranb, - lapack_int isgn, lapack_int m, lapack_int n, - const lapack_complex_float* a, lapack_int lda, - const lapack_complex_float* b, lapack_int ldb, - lapack_complex_float* c, lapack_int ldc, - float* scale ); -lapack_int LAPACKE_ztrsyl( int matrix_order, char trana, char tranb, - lapack_int isgn, lapack_int m, lapack_int n, - const lapack_complex_double* a, lapack_int lda, - const lapack_complex_double* b, lapack_int ldb, - lapack_complex_double* c, lapack_int ldc, - double* scale ); - -lapack_int LAPACKE_strtri( int matrix_order, char uplo, char diag, lapack_int n, - float* a, lapack_int lda ); -lapack_int LAPACKE_dtrtri( int matrix_order, char uplo, char diag, lapack_int n, - double* a, lapack_int lda ); -lapack_int LAPACKE_ctrtri( int matrix_order, char uplo, char diag, lapack_int n, - lapack_complex_float* a, lapack_int lda ); -lapack_int LAPACKE_ztrtri( int matrix_order, char uplo, char diag, lapack_int n, - lapack_complex_double* a, lapack_int lda ); - -lapack_int LAPACKE_strtrs( int matrix_order, char uplo, char trans, char diag, - lapack_int n, lapack_int nrhs, const float* a, - lapack_int lda, float* b, lapack_int ldb ); -lapack_int LAPACKE_dtrtrs( int matrix_order, char uplo, char trans, char diag, - lapack_int n, lapack_int nrhs, const double* a, - lapack_int lda, double* b, lapack_int ldb ); -lapack_int LAPACKE_ctrtrs( int matrix_order, char uplo, char trans, char diag, - lapack_int n, lapack_int nrhs, - const lapack_complex_float* a, lapack_int lda, - lapack_complex_float* b, lapack_int ldb ); -lapack_int LAPACKE_ztrtrs( int matrix_order, char uplo, char trans, char diag, - lapack_int n, lapack_int nrhs, - const lapack_complex_double* a, lapack_int lda, - lapack_complex_double* b, lapack_int ldb ); - -lapack_int LAPACKE_strttf( int matrix_order, char transr, char uplo, - lapack_int n, const float* a, lapack_int lda, - float* arf ); -lapack_int LAPACKE_dtrttf( int matrix_order, char transr, char uplo, - lapack_int n, const double* a, lapack_int lda, - double* arf ); -lapack_int LAPACKE_ctrttf( int matrix_order, char transr, char uplo, - lapack_int n, const lapack_complex_float* a, - lapack_int lda, lapack_complex_float* arf ); -lapack_int LAPACKE_ztrttf( int matrix_order, char transr, char uplo, - lapack_int n, const lapack_complex_double* a, - lapack_int lda, lapack_complex_double* arf ); - -lapack_int LAPACKE_strttp( int matrix_order, char uplo, lapack_int n, - const float* a, lapack_int lda, float* ap ); -lapack_int LAPACKE_dtrttp( int matrix_order, char uplo, lapack_int n, - const double* a, lapack_int lda, double* ap ); -lapack_int LAPACKE_ctrttp( int matrix_order, char uplo, lapack_int n, - const lapack_complex_float* a, lapack_int lda, - lapack_complex_float* ap ); -lapack_int LAPACKE_ztrttp( int matrix_order, char uplo, lapack_int n, - const lapack_complex_double* a, lapack_int lda, - lapack_complex_double* ap ); - -lapack_int LAPACKE_stzrzf( int matrix_order, lapack_int m, lapack_int n, - float* a, lapack_int lda, float* tau ); -lapack_int LAPACKE_dtzrzf( int matrix_order, lapack_int m, lapack_int n, - double* a, lapack_int lda, double* tau ); -lapack_int LAPACKE_ctzrzf( int matrix_order, lapack_int m, lapack_int n, - lapack_complex_float* a, lapack_int lda, - lapack_complex_float* tau ); -lapack_int LAPACKE_ztzrzf( int matrix_order, lapack_int m, lapack_int n, - lapack_complex_double* a, lapack_int lda, - lapack_complex_double* tau ); - -lapack_int LAPACKE_cungbr( int matrix_order, char vect, lapack_int m, - lapack_int n, lapack_int k, lapack_complex_float* a, - lapack_int lda, const lapack_complex_float* tau ); -lapack_int LAPACKE_zungbr( int matrix_order, char vect, lapack_int m, - lapack_int n, lapack_int k, lapack_complex_double* a, - lapack_int lda, const lapack_complex_double* tau ); - -lapack_int LAPACKE_cunghr( int matrix_order, lapack_int n, lapack_int ilo, - lapack_int ihi, lapack_complex_float* a, - lapack_int lda, const lapack_complex_float* tau ); -lapack_int LAPACKE_zunghr( int matrix_order, lapack_int n, lapack_int ilo, - lapack_int ihi, lapack_complex_double* a, - lapack_int lda, const lapack_complex_double* tau ); - -lapack_int LAPACKE_cunglq( int matrix_order, lapack_int m, lapack_int n, - lapack_int k, lapack_complex_float* a, - lapack_int lda, const lapack_complex_float* tau ); -lapack_int LAPACKE_zunglq( int matrix_order, lapack_int m, lapack_int n, - lapack_int k, lapack_complex_double* a, - lapack_int lda, const lapack_complex_double* tau ); - -lapack_int LAPACKE_cungql( int matrix_order, lapack_int m, lapack_int n, - lapack_int k, lapack_complex_float* a, - lapack_int lda, const lapack_complex_float* tau ); -lapack_int LAPACKE_zungql( int matrix_order, lapack_int m, lapack_int n, - lapack_int k, lapack_complex_double* a, - lapack_int lda, const lapack_complex_double* tau ); - -lapack_int LAPACKE_cungqr( int matrix_order, lapack_int m, lapack_int n, - lapack_int k, lapack_complex_float* a, - lapack_int lda, const lapack_complex_float* tau ); -lapack_int LAPACKE_zungqr( int matrix_order, lapack_int m, lapack_int n, - lapack_int k, lapack_complex_double* a, - lapack_int lda, const lapack_complex_double* tau ); - -lapack_int LAPACKE_cungrq( int matrix_order, lapack_int m, lapack_int n, - lapack_int k, lapack_complex_float* a, - lapack_int lda, const lapack_complex_float* tau ); -lapack_int LAPACKE_zungrq( int matrix_order, lapack_int m, lapack_int n, - lapack_int k, lapack_complex_double* a, - lapack_int lda, const lapack_complex_double* tau ); - -lapack_int LAPACKE_cungtr( int matrix_order, char uplo, lapack_int n, - lapack_complex_float* a, lapack_int lda, - const lapack_complex_float* tau ); -lapack_int LAPACKE_zungtr( int matrix_order, char uplo, lapack_int n, - lapack_complex_double* a, lapack_int lda, - const lapack_complex_double* tau ); - -lapack_int LAPACKE_cunmbr( int matrix_order, char vect, char side, char trans, - lapack_int m, lapack_int n, lapack_int k, - const lapack_complex_float* a, lapack_int lda, - const lapack_complex_float* tau, - lapack_complex_float* c, lapack_int ldc ); -lapack_int LAPACKE_zunmbr( int matrix_order, char vect, char side, char trans, - lapack_int m, lapack_int n, lapack_int k, - const lapack_complex_double* a, lapack_int lda, - const lapack_complex_double* tau, - lapack_complex_double* c, lapack_int ldc ); - -lapack_int LAPACKE_cunmhr( int matrix_order, char side, char trans, - lapack_int m, lapack_int n, lapack_int ilo, - lapack_int ihi, const lapack_complex_float* a, - lapack_int lda, const lapack_complex_float* tau, - lapack_complex_float* c, lapack_int ldc ); -lapack_int LAPACKE_zunmhr( int matrix_order, char side, char trans, - lapack_int m, lapack_int n, lapack_int ilo, - lapack_int ihi, const lapack_complex_double* a, - lapack_int lda, const lapack_complex_double* tau, - lapack_complex_double* c, lapack_int ldc ); - -lapack_int LAPACKE_cunmlq( int matrix_order, char side, char trans, - lapack_int m, lapack_int n, lapack_int k, - const lapack_complex_float* a, lapack_int lda, - const lapack_complex_float* tau, - lapack_complex_float* c, lapack_int ldc ); -lapack_int LAPACKE_zunmlq( int matrix_order, char side, char trans, - lapack_int m, lapack_int n, lapack_int k, - const lapack_complex_double* a, lapack_int lda, - const lapack_complex_double* tau, - lapack_complex_double* c, lapack_int ldc ); - -lapack_int LAPACKE_cunmql( int matrix_order, char side, char trans, - lapack_int m, lapack_int n, lapack_int k, - const lapack_complex_float* a, lapack_int lda, - const lapack_complex_float* tau, - lapack_complex_float* c, lapack_int ldc ); -lapack_int LAPACKE_zunmql( int matrix_order, char side, char trans, - lapack_int m, lapack_int n, lapack_int k, - const lapack_complex_double* a, lapack_int lda, - const lapack_complex_double* tau, - lapack_complex_double* c, lapack_int ldc ); - -lapack_int LAPACKE_cunmqr( int matrix_order, char side, char trans, - lapack_int m, lapack_int n, lapack_int k, - const lapack_complex_float* a, lapack_int lda, - const lapack_complex_float* tau, - lapack_complex_float* c, lapack_int ldc ); -lapack_int LAPACKE_zunmqr( int matrix_order, char side, char trans, - lapack_int m, lapack_int n, lapack_int k, - const lapack_complex_double* a, lapack_int lda, - const lapack_complex_double* tau, - lapack_complex_double* c, lapack_int ldc ); - -lapack_int LAPACKE_cunmrq( int matrix_order, char side, char trans, - lapack_int m, lapack_int n, lapack_int k, - const lapack_complex_float* a, lapack_int lda, - const lapack_complex_float* tau, - lapack_complex_float* c, lapack_int ldc ); -lapack_int LAPACKE_zunmrq( int matrix_order, char side, char trans, - lapack_int m, lapack_int n, lapack_int k, - const lapack_complex_double* a, lapack_int lda, - const lapack_complex_double* tau, - lapack_complex_double* c, lapack_int ldc ); - -lapack_int LAPACKE_cunmrz( int matrix_order, char side, char trans, - lapack_int m, lapack_int n, lapack_int k, - lapack_int l, const lapack_complex_float* a, - lapack_int lda, const lapack_complex_float* tau, - lapack_complex_float* c, lapack_int ldc ); -lapack_int LAPACKE_zunmrz( int matrix_order, char side, char trans, - lapack_int m, lapack_int n, lapack_int k, - lapack_int l, const lapack_complex_double* a, - lapack_int lda, const lapack_complex_double* tau, - lapack_complex_double* c, lapack_int ldc ); - -lapack_int LAPACKE_cunmtr( int matrix_order, char side, char uplo, char trans, - lapack_int m, lapack_int n, - const lapack_complex_float* a, lapack_int lda, - const lapack_complex_float* tau, - lapack_complex_float* c, lapack_int ldc ); -lapack_int LAPACKE_zunmtr( int matrix_order, char side, char uplo, char trans, - lapack_int m, lapack_int n, - const lapack_complex_double* a, lapack_int lda, - const lapack_complex_double* tau, - lapack_complex_double* c, lapack_int ldc ); - -lapack_int LAPACKE_cupgtr( int matrix_order, char uplo, lapack_int n, - const lapack_complex_float* ap, - const lapack_complex_float* tau, - lapack_complex_float* q, lapack_int ldq ); -lapack_int LAPACKE_zupgtr( int matrix_order, char uplo, lapack_int n, - const lapack_complex_double* ap, - const lapack_complex_double* tau, - lapack_complex_double* q, lapack_int ldq ); - -lapack_int LAPACKE_cupmtr( int matrix_order, char side, char uplo, char trans, - lapack_int m, lapack_int n, - const lapack_complex_float* ap, - const lapack_complex_float* tau, - lapack_complex_float* c, lapack_int ldc ); -lapack_int LAPACKE_zupmtr( int matrix_order, char side, char uplo, char trans, - lapack_int m, lapack_int n, - const lapack_complex_double* ap, - const lapack_complex_double* tau, - lapack_complex_double* c, lapack_int ldc ); - -lapack_int LAPACKE_sbdsdc_work( int matrix_order, char uplo, char compq, - lapack_int n, float* d, float* e, float* u, - lapack_int ldu, float* vt, lapack_int ldvt, - float* q, lapack_int* iq, float* work, - lapack_int* iwork ); -lapack_int LAPACKE_dbdsdc_work( int matrix_order, char uplo, char compq, - lapack_int n, double* d, double* e, double* u, - lapack_int ldu, double* vt, lapack_int ldvt, - double* q, lapack_int* iq, double* work, - lapack_int* iwork ); - -lapack_int LAPACKE_sbdsqr_work( int matrix_order, char uplo, lapack_int n, - lapack_int ncvt, lapack_int nru, lapack_int ncc, - float* d, float* e, float* vt, lapack_int ldvt, - float* u, lapack_int ldu, float* c, - lapack_int ldc, float* work ); -lapack_int LAPACKE_dbdsqr_work( int matrix_order, char uplo, lapack_int n, - lapack_int ncvt, lapack_int nru, lapack_int ncc, - double* d, double* e, double* vt, - lapack_int ldvt, double* u, lapack_int ldu, - double* c, lapack_int ldc, double* work ); -lapack_int LAPACKE_cbdsqr_work( int matrix_order, char uplo, lapack_int n, - lapack_int ncvt, lapack_int nru, lapack_int ncc, - float* d, float* e, lapack_complex_float* vt, - lapack_int ldvt, lapack_complex_float* u, - lapack_int ldu, lapack_complex_float* c, - lapack_int ldc, float* work ); -lapack_int LAPACKE_zbdsqr_work( int matrix_order, char uplo, lapack_int n, - lapack_int ncvt, lapack_int nru, lapack_int ncc, - double* d, double* e, lapack_complex_double* vt, - lapack_int ldvt, lapack_complex_double* u, - lapack_int ldu, lapack_complex_double* c, - lapack_int ldc, double* work ); - -lapack_int LAPACKE_sdisna_work( char job, lapack_int m, lapack_int n, - const float* d, float* sep ); -lapack_int LAPACKE_ddisna_work( char job, lapack_int m, lapack_int n, - const double* d, double* sep ); - -lapack_int LAPACKE_sgbbrd_work( int matrix_order, char vect, lapack_int m, - lapack_int n, lapack_int ncc, lapack_int kl, - lapack_int ku, float* ab, lapack_int ldab, - float* d, float* e, float* q, lapack_int ldq, - float* pt, lapack_int ldpt, float* c, - lapack_int ldc, float* work ); -lapack_int LAPACKE_dgbbrd_work( int matrix_order, char vect, lapack_int m, - lapack_int n, lapack_int ncc, lapack_int kl, - lapack_int ku, double* ab, lapack_int ldab, - double* d, double* e, double* q, lapack_int ldq, - double* pt, lapack_int ldpt, double* c, - lapack_int ldc, double* work ); -lapack_int LAPACKE_cgbbrd_work( int matrix_order, char vect, lapack_int m, - lapack_int n, lapack_int ncc, lapack_int kl, - lapack_int ku, lapack_complex_float* ab, - lapack_int ldab, float* d, float* e, - lapack_complex_float* q, lapack_int ldq, - lapack_complex_float* pt, lapack_int ldpt, - lapack_complex_float* c, lapack_int ldc, - lapack_complex_float* work, float* rwork ); -lapack_int LAPACKE_zgbbrd_work( int matrix_order, char vect, lapack_int m, - lapack_int n, lapack_int ncc, lapack_int kl, - lapack_int ku, lapack_complex_double* ab, - lapack_int ldab, double* d, double* e, - lapack_complex_double* q, lapack_int ldq, - lapack_complex_double* pt, lapack_int ldpt, - lapack_complex_double* c, lapack_int ldc, - lapack_complex_double* work, double* rwork ); - -lapack_int LAPACKE_sgbcon_work( int matrix_order, char norm, lapack_int n, - lapack_int kl, lapack_int ku, const float* ab, - lapack_int ldab, const lapack_int* ipiv, - float anorm, float* rcond, float* work, - lapack_int* iwork ); -lapack_int LAPACKE_dgbcon_work( int matrix_order, char norm, lapack_int n, - lapack_int kl, lapack_int ku, const double* ab, - lapack_int ldab, const lapack_int* ipiv, - double anorm, double* rcond, double* work, - lapack_int* iwork ); -lapack_int LAPACKE_cgbcon_work( int matrix_order, char norm, lapack_int n, - lapack_int kl, lapack_int ku, - const lapack_complex_float* ab, lapack_int ldab, - const lapack_int* ipiv, float anorm, - float* rcond, lapack_complex_float* work, - float* rwork ); -lapack_int LAPACKE_zgbcon_work( int matrix_order, char norm, lapack_int n, - lapack_int kl, lapack_int ku, - const lapack_complex_double* ab, - lapack_int ldab, const lapack_int* ipiv, - double anorm, double* rcond, - lapack_complex_double* work, double* rwork ); - -lapack_int LAPACKE_sgbequ_work( int matrix_order, lapack_int m, lapack_int n, - lapack_int kl, lapack_int ku, const float* ab, - lapack_int ldab, float* r, float* c, - float* rowcnd, float* colcnd, float* amax ); -lapack_int LAPACKE_dgbequ_work( int matrix_order, lapack_int m, lapack_int n, - lapack_int kl, lapack_int ku, const double* ab, - lapack_int ldab, double* r, double* c, - double* rowcnd, double* colcnd, double* amax ); -lapack_int LAPACKE_cgbequ_work( int matrix_order, lapack_int m, lapack_int n, - lapack_int kl, lapack_int ku, - const lapack_complex_float* ab, lapack_int ldab, - float* r, float* c, float* rowcnd, - float* colcnd, float* amax ); -lapack_int LAPACKE_zgbequ_work( int matrix_order, lapack_int m, lapack_int n, - lapack_int kl, lapack_int ku, - const lapack_complex_double* ab, - lapack_int ldab, double* r, double* c, - double* rowcnd, double* colcnd, double* amax ); - -lapack_int LAPACKE_sgbequb_work( int matrix_order, lapack_int m, lapack_int n, - lapack_int kl, lapack_int ku, const float* ab, - lapack_int ldab, float* r, float* c, - float* rowcnd, float* colcnd, float* amax ); -lapack_int LAPACKE_dgbequb_work( int matrix_order, lapack_int m, lapack_int n, - lapack_int kl, lapack_int ku, const double* ab, - lapack_int ldab, double* r, double* c, - double* rowcnd, double* colcnd, double* amax ); -lapack_int LAPACKE_cgbequb_work( int matrix_order, lapack_int m, lapack_int n, - lapack_int kl, lapack_int ku, - const lapack_complex_float* ab, - lapack_int ldab, float* r, float* c, - float* rowcnd, float* colcnd, float* amax ); -lapack_int LAPACKE_zgbequb_work( int matrix_order, lapack_int m, lapack_int n, - lapack_int kl, lapack_int ku, - const lapack_complex_double* ab, - lapack_int ldab, double* r, double* c, - double* rowcnd, double* colcnd, double* amax ); - -lapack_int LAPACKE_sgbrfs_work( int matrix_order, char trans, lapack_int n, - lapack_int kl, lapack_int ku, lapack_int nrhs, - const float* ab, lapack_int ldab, - const float* afb, lapack_int ldafb, - const lapack_int* ipiv, const float* b, - lapack_int ldb, float* x, lapack_int ldx, - float* ferr, float* berr, float* work, - lapack_int* iwork ); -lapack_int LAPACKE_dgbrfs_work( int matrix_order, char trans, lapack_int n, - lapack_int kl, lapack_int ku, lapack_int nrhs, - const double* ab, lapack_int ldab, - const double* afb, lapack_int ldafb, - const lapack_int* ipiv, const double* b, - lapack_int ldb, double* x, lapack_int ldx, - double* ferr, double* berr, double* work, - lapack_int* iwork ); -lapack_int LAPACKE_cgbrfs_work( int matrix_order, char trans, lapack_int n, - lapack_int kl, lapack_int ku, lapack_int nrhs, - const lapack_complex_float* ab, lapack_int ldab, - const lapack_complex_float* afb, - lapack_int ldafb, const lapack_int* ipiv, - const lapack_complex_float* b, lapack_int ldb, - lapack_complex_float* x, lapack_int ldx, - float* ferr, float* berr, - lapack_complex_float* work, float* rwork ); -lapack_int LAPACKE_zgbrfs_work( int matrix_order, char trans, lapack_int n, - lapack_int kl, lapack_int ku, lapack_int nrhs, - const lapack_complex_double* ab, - lapack_int ldab, - const lapack_complex_double* afb, - lapack_int ldafb, const lapack_int* ipiv, - const lapack_complex_double* b, lapack_int ldb, - lapack_complex_double* x, lapack_int ldx, - double* ferr, double* berr, - lapack_complex_double* work, double* rwork ); - -lapack_int LAPACKE_sgbrfsx_work( int matrix_order, char trans, char equed, - lapack_int n, lapack_int kl, lapack_int ku, - lapack_int nrhs, const float* ab, - lapack_int ldab, const float* afb, - lapack_int ldafb, const lapack_int* ipiv, - const float* r, const float* c, const float* b, - lapack_int ldb, float* x, lapack_int ldx, - float* rcond, float* berr, - lapack_int n_err_bnds, float* err_bnds_norm, - float* err_bnds_comp, lapack_int nparams, - float* params, float* work, - lapack_int* iwork ); -lapack_int LAPACKE_dgbrfsx_work( int matrix_order, char trans, char equed, - lapack_int n, lapack_int kl, lapack_int ku, - lapack_int nrhs, const double* ab, - lapack_int ldab, const double* afb, - lapack_int ldafb, const lapack_int* ipiv, - const double* r, const double* c, - const double* b, lapack_int ldb, double* x, - lapack_int ldx, double* rcond, double* berr, - lapack_int n_err_bnds, double* err_bnds_norm, - double* err_bnds_comp, lapack_int nparams, - double* params, double* work, - lapack_int* iwork ); -lapack_int LAPACKE_cgbrfsx_work( int matrix_order, char trans, char equed, - lapack_int n, lapack_int kl, lapack_int ku, - lapack_int nrhs, - const lapack_complex_float* ab, - lapack_int ldab, - const lapack_complex_float* afb, - lapack_int ldafb, const lapack_int* ipiv, - const float* r, const float* c, - const lapack_complex_float* b, lapack_int ldb, - lapack_complex_float* x, lapack_int ldx, - float* rcond, float* berr, - lapack_int n_err_bnds, float* err_bnds_norm, - float* err_bnds_comp, lapack_int nparams, - float* params, lapack_complex_float* work, - float* rwork ); -lapack_int LAPACKE_zgbrfsx_work( int matrix_order, char trans, char equed, - lapack_int n, lapack_int kl, lapack_int ku, - lapack_int nrhs, - const lapack_complex_double* ab, - lapack_int ldab, - const lapack_complex_double* afb, - lapack_int ldafb, const lapack_int* ipiv, - const double* r, const double* c, - const lapack_complex_double* b, lapack_int ldb, - lapack_complex_double* x, lapack_int ldx, - double* rcond, double* berr, - lapack_int n_err_bnds, double* err_bnds_norm, - double* err_bnds_comp, lapack_int nparams, - double* params, lapack_complex_double* work, - double* rwork ); - -lapack_int LAPACKE_sgbsv_work( int matrix_order, lapack_int n, lapack_int kl, - lapack_int ku, lapack_int nrhs, float* ab, - lapack_int ldab, lapack_int* ipiv, float* b, - lapack_int ldb ); -lapack_int LAPACKE_dgbsv_work( int matrix_order, lapack_int n, lapack_int kl, - lapack_int ku, lapack_int nrhs, double* ab, - lapack_int ldab, lapack_int* ipiv, double* b, - lapack_int ldb ); -lapack_int LAPACKE_cgbsv_work( int matrix_order, lapack_int n, lapack_int kl, - lapack_int ku, lapack_int nrhs, - lapack_complex_float* ab, lapack_int ldab, - lapack_int* ipiv, lapack_complex_float* b, - lapack_int ldb ); -lapack_int LAPACKE_zgbsv_work( int matrix_order, lapack_int n, lapack_int kl, - lapack_int ku, lapack_int nrhs, - lapack_complex_double* ab, lapack_int ldab, - lapack_int* ipiv, lapack_complex_double* b, - lapack_int ldb ); - -lapack_int LAPACKE_sgbsvx_work( int matrix_order, char fact, char trans, - lapack_int n, lapack_int kl, lapack_int ku, - lapack_int nrhs, float* ab, lapack_int ldab, - float* afb, lapack_int ldafb, lapack_int* ipiv, - char* equed, float* r, float* c, float* b, - lapack_int ldb, float* x, lapack_int ldx, - float* rcond, float* ferr, float* berr, - float* work, lapack_int* iwork ); -lapack_int LAPACKE_dgbsvx_work( int matrix_order, char fact, char trans, - lapack_int n, lapack_int kl, lapack_int ku, - lapack_int nrhs, double* ab, lapack_int ldab, - double* afb, lapack_int ldafb, lapack_int* ipiv, - char* equed, double* r, double* c, double* b, - lapack_int ldb, double* x, lapack_int ldx, - double* rcond, double* ferr, double* berr, - double* work, lapack_int* iwork ); -lapack_int LAPACKE_cgbsvx_work( int matrix_order, char fact, char trans, - lapack_int n, lapack_int kl, lapack_int ku, - lapack_int nrhs, lapack_complex_float* ab, - lapack_int ldab, lapack_complex_float* afb, - lapack_int ldafb, lapack_int* ipiv, char* equed, - float* r, float* c, lapack_complex_float* b, - lapack_int ldb, lapack_complex_float* x, - lapack_int ldx, float* rcond, float* ferr, - float* berr, lapack_complex_float* work, - float* rwork ); -lapack_int LAPACKE_zgbsvx_work( int matrix_order, char fact, char trans, - lapack_int n, lapack_int kl, lapack_int ku, - lapack_int nrhs, lapack_complex_double* ab, - lapack_int ldab, lapack_complex_double* afb, - lapack_int ldafb, lapack_int* ipiv, char* equed, - double* r, double* c, lapack_complex_double* b, - lapack_int ldb, lapack_complex_double* x, - lapack_int ldx, double* rcond, double* ferr, - double* berr, lapack_complex_double* work, - double* rwork ); - -lapack_int LAPACKE_sgbsvxx_work( int matrix_order, char fact, char trans, - lapack_int n, lapack_int kl, lapack_int ku, - lapack_int nrhs, float* ab, lapack_int ldab, - float* afb, lapack_int ldafb, lapack_int* ipiv, - char* equed, float* r, float* c, float* b, - lapack_int ldb, float* x, lapack_int ldx, - float* rcond, float* rpvgrw, float* berr, - lapack_int n_err_bnds, float* err_bnds_norm, - float* err_bnds_comp, lapack_int nparams, - float* params, float* work, - lapack_int* iwork ); -lapack_int LAPACKE_dgbsvxx_work( int matrix_order, char fact, char trans, - lapack_int n, lapack_int kl, lapack_int ku, - lapack_int nrhs, double* ab, lapack_int ldab, - double* afb, lapack_int ldafb, - lapack_int* ipiv, char* equed, double* r, - double* c, double* b, lapack_int ldb, - double* x, lapack_int ldx, double* rcond, - double* rpvgrw, double* berr, - lapack_int n_err_bnds, double* err_bnds_norm, - double* err_bnds_comp, lapack_int nparams, - double* params, double* work, - lapack_int* iwork ); -lapack_int LAPACKE_cgbsvxx_work( int matrix_order, char fact, char trans, - lapack_int n, lapack_int kl, lapack_int ku, - lapack_int nrhs, lapack_complex_float* ab, - lapack_int ldab, lapack_complex_float* afb, - lapack_int ldafb, lapack_int* ipiv, - char* equed, float* r, float* c, - lapack_complex_float* b, lapack_int ldb, - lapack_complex_float* x, lapack_int ldx, - float* rcond, float* rpvgrw, float* berr, - lapack_int n_err_bnds, float* err_bnds_norm, - float* err_bnds_comp, lapack_int nparams, - float* params, lapack_complex_float* work, - float* rwork ); -lapack_int LAPACKE_zgbsvxx_work( int matrix_order, char fact, char trans, - lapack_int n, lapack_int kl, lapack_int ku, - lapack_int nrhs, lapack_complex_double* ab, - lapack_int ldab, lapack_complex_double* afb, - lapack_int ldafb, lapack_int* ipiv, - char* equed, double* r, double* c, - lapack_complex_double* b, lapack_int ldb, - lapack_complex_double* x, lapack_int ldx, - double* rcond, double* rpvgrw, double* berr, - lapack_int n_err_bnds, double* err_bnds_norm, - double* err_bnds_comp, lapack_int nparams, - double* params, lapack_complex_double* work, - double* rwork ); - -lapack_int LAPACKE_sgbtrf_work( int matrix_order, lapack_int m, lapack_int n, - lapack_int kl, lapack_int ku, float* ab, - lapack_int ldab, lapack_int* ipiv ); -lapack_int LAPACKE_dgbtrf_work( int matrix_order, lapack_int m, lapack_int n, - lapack_int kl, lapack_int ku, double* ab, - lapack_int ldab, lapack_int* ipiv ); -lapack_int LAPACKE_cgbtrf_work( int matrix_order, lapack_int m, lapack_int n, - lapack_int kl, lapack_int ku, - lapack_complex_float* ab, lapack_int ldab, - lapack_int* ipiv ); -lapack_int LAPACKE_zgbtrf_work( int matrix_order, lapack_int m, lapack_int n, - lapack_int kl, lapack_int ku, - lapack_complex_double* ab, lapack_int ldab, - lapack_int* ipiv ); - -lapack_int LAPACKE_sgbtrs_work( int matrix_order, char trans, lapack_int n, - lapack_int kl, lapack_int ku, lapack_int nrhs, - const float* ab, lapack_int ldab, - const lapack_int* ipiv, float* b, - lapack_int ldb ); -lapack_int LAPACKE_dgbtrs_work( int matrix_order, char trans, lapack_int n, - lapack_int kl, lapack_int ku, lapack_int nrhs, - const double* ab, lapack_int ldab, - const lapack_int* ipiv, double* b, - lapack_int ldb ); -lapack_int LAPACKE_cgbtrs_work( int matrix_order, char trans, lapack_int n, - lapack_int kl, lapack_int ku, lapack_int nrhs, - const lapack_complex_float* ab, lapack_int ldab, - const lapack_int* ipiv, lapack_complex_float* b, - lapack_int ldb ); -lapack_int LAPACKE_zgbtrs_work( int matrix_order, char trans, lapack_int n, - lapack_int kl, lapack_int ku, lapack_int nrhs, - const lapack_complex_double* ab, - lapack_int ldab, const lapack_int* ipiv, - lapack_complex_double* b, lapack_int ldb ); - -lapack_int LAPACKE_sgebak_work( int matrix_order, char job, char side, - lapack_int n, lapack_int ilo, lapack_int ihi, - const float* scale, lapack_int m, float* v, - lapack_int ldv ); -lapack_int LAPACKE_dgebak_work( int matrix_order, char job, char side, - lapack_int n, lapack_int ilo, lapack_int ihi, - const double* scale, lapack_int m, double* v, - lapack_int ldv ); -lapack_int LAPACKE_cgebak_work( int matrix_order, char job, char side, - lapack_int n, lapack_int ilo, lapack_int ihi, - const float* scale, lapack_int m, - lapack_complex_float* v, lapack_int ldv ); -lapack_int LAPACKE_zgebak_work( int matrix_order, char job, char side, - lapack_int n, lapack_int ilo, lapack_int ihi, - const double* scale, lapack_int m, - lapack_complex_double* v, lapack_int ldv ); - -lapack_int LAPACKE_sgebal_work( int matrix_order, char job, lapack_int n, - float* a, lapack_int lda, lapack_int* ilo, - lapack_int* ihi, float* scale ); -lapack_int LAPACKE_dgebal_work( int matrix_order, char job, lapack_int n, - double* a, lapack_int lda, lapack_int* ilo, - lapack_int* ihi, double* scale ); -lapack_int LAPACKE_cgebal_work( int matrix_order, char job, lapack_int n, - lapack_complex_float* a, lapack_int lda, - lapack_int* ilo, lapack_int* ihi, - float* scale ); -lapack_int LAPACKE_zgebal_work( int matrix_order, char job, lapack_int n, - lapack_complex_double* a, lapack_int lda, - lapack_int* ilo, lapack_int* ihi, - double* scale ); - -lapack_int LAPACKE_sgebrd_work( int matrix_order, lapack_int m, lapack_int n, - float* a, lapack_int lda, float* d, float* e, - float* tauq, float* taup, float* work, - lapack_int lwork ); -lapack_int LAPACKE_dgebrd_work( int matrix_order, lapack_int m, lapack_int n, - double* a, lapack_int lda, double* d, double* e, - double* tauq, double* taup, double* work, - lapack_int lwork ); -lapack_int LAPACKE_cgebrd_work( int matrix_order, lapack_int m, lapack_int n, - lapack_complex_float* a, lapack_int lda, - float* d, float* e, lapack_complex_float* tauq, - lapack_complex_float* taup, - lapack_complex_float* work, lapack_int lwork ); -lapack_int LAPACKE_zgebrd_work( int matrix_order, lapack_int m, lapack_int n, - lapack_complex_double* a, lapack_int lda, - double* d, double* e, - lapack_complex_double* tauq, - lapack_complex_double* taup, - lapack_complex_double* work, lapack_int lwork ); - -lapack_int LAPACKE_sgecon_work( int matrix_order, char norm, lapack_int n, - const float* a, lapack_int lda, float anorm, - float* rcond, float* work, lapack_int* iwork ); -lapack_int LAPACKE_dgecon_work( int matrix_order, char norm, lapack_int n, - const double* a, lapack_int lda, double anorm, - double* rcond, double* work, - lapack_int* iwork ); -lapack_int LAPACKE_cgecon_work( int matrix_order, char norm, lapack_int n, - const lapack_complex_float* a, lapack_int lda, - float anorm, float* rcond, - lapack_complex_float* work, float* rwork ); -lapack_int LAPACKE_zgecon_work( int matrix_order, char norm, lapack_int n, - const lapack_complex_double* a, lapack_int lda, - double anorm, double* rcond, - lapack_complex_double* work, double* rwork ); - -lapack_int LAPACKE_sgeequ_work( int matrix_order, lapack_int m, lapack_int n, - const float* a, lapack_int lda, float* r, - float* c, float* rowcnd, float* colcnd, - float* amax ); -lapack_int LAPACKE_dgeequ_work( int matrix_order, lapack_int m, lapack_int n, - const double* a, lapack_int lda, double* r, - double* c, double* rowcnd, double* colcnd, - double* amax ); -lapack_int LAPACKE_cgeequ_work( int matrix_order, lapack_int m, lapack_int n, - const lapack_complex_float* a, lapack_int lda, - float* r, float* c, float* rowcnd, - float* colcnd, float* amax ); -lapack_int LAPACKE_zgeequ_work( int matrix_order, lapack_int m, lapack_int n, - const lapack_complex_double* a, lapack_int lda, - double* r, double* c, double* rowcnd, - double* colcnd, double* amax ); - -lapack_int LAPACKE_sgeequb_work( int matrix_order, lapack_int m, lapack_int n, - const float* a, lapack_int lda, float* r, - float* c, float* rowcnd, float* colcnd, - float* amax ); -lapack_int LAPACKE_dgeequb_work( int matrix_order, lapack_int m, lapack_int n, - const double* a, lapack_int lda, double* r, - double* c, double* rowcnd, double* colcnd, - double* amax ); -lapack_int LAPACKE_cgeequb_work( int matrix_order, lapack_int m, lapack_int n, - const lapack_complex_float* a, lapack_int lda, - float* r, float* c, float* rowcnd, - float* colcnd, float* amax ); -lapack_int LAPACKE_zgeequb_work( int matrix_order, lapack_int m, lapack_int n, - const lapack_complex_double* a, lapack_int lda, - double* r, double* c, double* rowcnd, - double* colcnd, double* amax ); - -lapack_int LAPACKE_sgees_work( int matrix_order, char jobvs, char sort, - LAPACK_S_SELECT2 select, lapack_int n, float* a, - lapack_int lda, lapack_int* sdim, float* wr, - float* wi, float* vs, lapack_int ldvs, - float* work, lapack_int lwork, - lapack_logical* bwork ); -lapack_int LAPACKE_dgees_work( int matrix_order, char jobvs, char sort, - LAPACK_D_SELECT2 select, lapack_int n, double* a, - lapack_int lda, lapack_int* sdim, double* wr, - double* wi, double* vs, lapack_int ldvs, - double* work, lapack_int lwork, - lapack_logical* bwork ); -lapack_int LAPACKE_cgees_work( int matrix_order, char jobvs, char sort, - LAPACK_C_SELECT1 select, lapack_int n, - lapack_complex_float* a, lapack_int lda, - lapack_int* sdim, lapack_complex_float* w, - lapack_complex_float* vs, lapack_int ldvs, - lapack_complex_float* work, lapack_int lwork, - float* rwork, lapack_logical* bwork ); -lapack_int LAPACKE_zgees_work( int matrix_order, char jobvs, char sort, - LAPACK_Z_SELECT1 select, lapack_int n, - lapack_complex_double* a, lapack_int lda, - lapack_int* sdim, lapack_complex_double* w, - lapack_complex_double* vs, lapack_int ldvs, - lapack_complex_double* work, lapack_int lwork, - double* rwork, lapack_logical* bwork ); - -lapack_int LAPACKE_sgeesx_work( int matrix_order, char jobvs, char sort, - LAPACK_S_SELECT2 select, char sense, - lapack_int n, float* a, lapack_int lda, - lapack_int* sdim, float* wr, float* wi, - float* vs, lapack_int ldvs, float* rconde, - float* rcondv, float* work, lapack_int lwork, - lapack_int* iwork, lapack_int liwork, - lapack_logical* bwork ); -lapack_int LAPACKE_dgeesx_work( int matrix_order, char jobvs, char sort, - LAPACK_D_SELECT2 select, char sense, - lapack_int n, double* a, lapack_int lda, - lapack_int* sdim, double* wr, double* wi, - double* vs, lapack_int ldvs, double* rconde, - double* rcondv, double* work, lapack_int lwork, - lapack_int* iwork, lapack_int liwork, - lapack_logical* bwork ); -lapack_int LAPACKE_cgeesx_work( int matrix_order, char jobvs, char sort, - LAPACK_C_SELECT1 select, char sense, - lapack_int n, lapack_complex_float* a, - lapack_int lda, lapack_int* sdim, - lapack_complex_float* w, - lapack_complex_float* vs, lapack_int ldvs, - float* rconde, float* rcondv, - lapack_complex_float* work, lapack_int lwork, - float* rwork, lapack_logical* bwork ); -lapack_int LAPACKE_zgeesx_work( int matrix_order, char jobvs, char sort, - LAPACK_Z_SELECT1 select, char sense, - lapack_int n, lapack_complex_double* a, - lapack_int lda, lapack_int* sdim, - lapack_complex_double* w, - lapack_complex_double* vs, lapack_int ldvs, - double* rconde, double* rcondv, - lapack_complex_double* work, lapack_int lwork, - double* rwork, lapack_logical* bwork ); - -lapack_int LAPACKE_sgeev_work( int matrix_order, char jobvl, char jobvr, - lapack_int n, float* a, lapack_int lda, - float* wr, float* wi, float* vl, lapack_int ldvl, - float* vr, lapack_int ldvr, float* work, - lapack_int lwork ); -lapack_int LAPACKE_dgeev_work( int matrix_order, char jobvl, char jobvr, - lapack_int n, double* a, lapack_int lda, - double* wr, double* wi, double* vl, - lapack_int ldvl, double* vr, lapack_int ldvr, - double* work, lapack_int lwork ); -lapack_int LAPACKE_cgeev_work( int matrix_order, char jobvl, char jobvr, - lapack_int n, lapack_complex_float* a, - lapack_int lda, lapack_complex_float* w, - lapack_complex_float* vl, lapack_int ldvl, - lapack_complex_float* vr, lapack_int ldvr, - lapack_complex_float* work, lapack_int lwork, - float* rwork ); -lapack_int LAPACKE_zgeev_work( int matrix_order, char jobvl, char jobvr, - lapack_int n, lapack_complex_double* a, - lapack_int lda, lapack_complex_double* w, - lapack_complex_double* vl, lapack_int ldvl, - lapack_complex_double* vr, lapack_int ldvr, - lapack_complex_double* work, lapack_int lwork, - double* rwork ); - -lapack_int LAPACKE_sgeevx_work( int matrix_order, char balanc, char jobvl, - char jobvr, char sense, lapack_int n, float* a, - lapack_int lda, float* wr, float* wi, float* vl, - lapack_int ldvl, float* vr, lapack_int ldvr, - lapack_int* ilo, lapack_int* ihi, float* scale, - float* abnrm, float* rconde, float* rcondv, - float* work, lapack_int lwork, - lapack_int* iwork ); -lapack_int LAPACKE_dgeevx_work( int matrix_order, char balanc, char jobvl, - char jobvr, char sense, lapack_int n, double* a, - lapack_int lda, double* wr, double* wi, - double* vl, lapack_int ldvl, double* vr, - lapack_int ldvr, lapack_int* ilo, - lapack_int* ihi, double* scale, double* abnrm, - double* rconde, double* rcondv, double* work, - lapack_int lwork, lapack_int* iwork ); -lapack_int LAPACKE_cgeevx_work( int matrix_order, char balanc, char jobvl, - char jobvr, char sense, lapack_int n, - lapack_complex_float* a, lapack_int lda, - lapack_complex_float* w, - lapack_complex_float* vl, lapack_int ldvl, - lapack_complex_float* vr, lapack_int ldvr, - lapack_int* ilo, lapack_int* ihi, float* scale, - float* abnrm, float* rconde, float* rcondv, - lapack_complex_float* work, lapack_int lwork, - float* rwork ); -lapack_int LAPACKE_zgeevx_work( int matrix_order, char balanc, char jobvl, - char jobvr, char sense, lapack_int n, - lapack_complex_double* a, lapack_int lda, - lapack_complex_double* w, - lapack_complex_double* vl, lapack_int ldvl, - lapack_complex_double* vr, lapack_int ldvr, - lapack_int* ilo, lapack_int* ihi, double* scale, - double* abnrm, double* rconde, double* rcondv, - lapack_complex_double* work, lapack_int lwork, - double* rwork ); - -lapack_int LAPACKE_sgehrd_work( int matrix_order, lapack_int n, lapack_int ilo, - lapack_int ihi, float* a, lapack_int lda, - float* tau, float* work, lapack_int lwork ); -lapack_int LAPACKE_dgehrd_work( int matrix_order, lapack_int n, lapack_int ilo, - lapack_int ihi, double* a, lapack_int lda, - double* tau, double* work, lapack_int lwork ); -lapack_int LAPACKE_cgehrd_work( int matrix_order, lapack_int n, lapack_int ilo, - lapack_int ihi, lapack_complex_float* a, - lapack_int lda, lapack_complex_float* tau, - lapack_complex_float* work, lapack_int lwork ); -lapack_int LAPACKE_zgehrd_work( int matrix_order, lapack_int n, lapack_int ilo, - lapack_int ihi, lapack_complex_double* a, - lapack_int lda, lapack_complex_double* tau, - lapack_complex_double* work, lapack_int lwork ); - -lapack_int LAPACKE_sgejsv_work( int matrix_order, char joba, char jobu, - char jobv, char jobr, char jobt, char jobp, - lapack_int m, lapack_int n, float* a, - lapack_int lda, float* sva, float* u, - lapack_int ldu, float* v, lapack_int ldv, - float* work, lapack_int lwork, - lapack_int* iwork ); -lapack_int LAPACKE_dgejsv_work( int matrix_order, char joba, char jobu, - char jobv, char jobr, char jobt, char jobp, - lapack_int m, lapack_int n, double* a, - lapack_int lda, double* sva, double* u, - lapack_int ldu, double* v, lapack_int ldv, - double* work, lapack_int lwork, - lapack_int* iwork ); - -lapack_int LAPACKE_sgelq2_work( int matrix_order, lapack_int m, lapack_int n, - float* a, lapack_int lda, float* tau, - float* work ); -lapack_int LAPACKE_dgelq2_work( int matrix_order, lapack_int m, lapack_int n, - double* a, lapack_int lda, double* tau, - double* work ); -lapack_int LAPACKE_cgelq2_work( int matrix_order, lapack_int m, lapack_int n, - lapack_complex_float* a, lapack_int lda, - lapack_complex_float* tau, - lapack_complex_float* work ); -lapack_int LAPACKE_zgelq2_work( int matrix_order, lapack_int m, lapack_int n, - lapack_complex_double* a, lapack_int lda, - lapack_complex_double* tau, - lapack_complex_double* work ); - -lapack_int LAPACKE_sgelqf_work( int matrix_order, lapack_int m, lapack_int n, - float* a, lapack_int lda, float* tau, - float* work, lapack_int lwork ); -lapack_int LAPACKE_dgelqf_work( int matrix_order, lapack_int m, lapack_int n, - double* a, lapack_int lda, double* tau, - double* work, lapack_int lwork ); -lapack_int LAPACKE_cgelqf_work( int matrix_order, lapack_int m, lapack_int n, - lapack_complex_float* a, lapack_int lda, - lapack_complex_float* tau, - lapack_complex_float* work, lapack_int lwork ); -lapack_int LAPACKE_zgelqf_work( int matrix_order, lapack_int m, lapack_int n, - lapack_complex_double* a, lapack_int lda, - lapack_complex_double* tau, - lapack_complex_double* work, lapack_int lwork ); - -lapack_int LAPACKE_sgels_work( int matrix_order, char trans, lapack_int m, - lapack_int n, lapack_int nrhs, float* a, - lapack_int lda, float* b, lapack_int ldb, - float* work, lapack_int lwork ); -lapack_int LAPACKE_dgels_work( int matrix_order, char trans, lapack_int m, - lapack_int n, lapack_int nrhs, double* a, - lapack_int lda, double* b, lapack_int ldb, - double* work, lapack_int lwork ); -lapack_int LAPACKE_cgels_work( int matrix_order, char trans, lapack_int m, - lapack_int n, lapack_int nrhs, - lapack_complex_float* a, lapack_int lda, - lapack_complex_float* b, lapack_int ldb, - lapack_complex_float* work, lapack_int lwork ); -lapack_int LAPACKE_zgels_work( int matrix_order, char trans, lapack_int m, - lapack_int n, lapack_int nrhs, - lapack_complex_double* a, lapack_int lda, - lapack_complex_double* b, lapack_int ldb, - lapack_complex_double* work, lapack_int lwork ); - -lapack_int LAPACKE_sgelsd_work( int matrix_order, lapack_int m, lapack_int n, - lapack_int nrhs, float* a, lapack_int lda, - float* b, lapack_int ldb, float* s, float rcond, - lapack_int* rank, float* work, lapack_int lwork, - lapack_int* iwork ); -lapack_int LAPACKE_dgelsd_work( int matrix_order, lapack_int m, lapack_int n, - lapack_int nrhs, double* a, lapack_int lda, - double* b, lapack_int ldb, double* s, - double rcond, lapack_int* rank, double* work, - lapack_int lwork, lapack_int* iwork ); -lapack_int LAPACKE_cgelsd_work( int matrix_order, lapack_int m, lapack_int n, - lapack_int nrhs, lapack_complex_float* a, - lapack_int lda, lapack_complex_float* b, - lapack_int ldb, float* s, float rcond, - lapack_int* rank, lapack_complex_float* work, - lapack_int lwork, float* rwork, - lapack_int* iwork ); -lapack_int LAPACKE_zgelsd_work( int matrix_order, lapack_int m, lapack_int n, - lapack_int nrhs, lapack_complex_double* a, - lapack_int lda, lapack_complex_double* b, - lapack_int ldb, double* s, double rcond, - lapack_int* rank, lapack_complex_double* work, - lapack_int lwork, double* rwork, - lapack_int* iwork ); - -lapack_int LAPACKE_sgelss_work( int matrix_order, lapack_int m, lapack_int n, - lapack_int nrhs, float* a, lapack_int lda, - float* b, lapack_int ldb, float* s, float rcond, - lapack_int* rank, float* work, - lapack_int lwork ); -lapack_int LAPACKE_dgelss_work( int matrix_order, lapack_int m, lapack_int n, - lapack_int nrhs, double* a, lapack_int lda, - double* b, lapack_int ldb, double* s, - double rcond, lapack_int* rank, double* work, - lapack_int lwork ); -lapack_int LAPACKE_cgelss_work( int matrix_order, lapack_int m, lapack_int n, - lapack_int nrhs, lapack_complex_float* a, - lapack_int lda, lapack_complex_float* b, - lapack_int ldb, float* s, float rcond, - lapack_int* rank, lapack_complex_float* work, - lapack_int lwork, float* rwork ); -lapack_int LAPACKE_zgelss_work( int matrix_order, lapack_int m, lapack_int n, - lapack_int nrhs, lapack_complex_double* a, - lapack_int lda, lapack_complex_double* b, - lapack_int ldb, double* s, double rcond, - lapack_int* rank, lapack_complex_double* work, - lapack_int lwork, double* rwork ); - -lapack_int LAPACKE_sgelsy_work( int matrix_order, lapack_int m, lapack_int n, - lapack_int nrhs, float* a, lapack_int lda, - float* b, lapack_int ldb, lapack_int* jpvt, - float rcond, lapack_int* rank, float* work, - lapack_int lwork ); -lapack_int LAPACKE_dgelsy_work( int matrix_order, lapack_int m, lapack_int n, - lapack_int nrhs, double* a, lapack_int lda, - double* b, lapack_int ldb, lapack_int* jpvt, - double rcond, lapack_int* rank, double* work, - lapack_int lwork ); -lapack_int LAPACKE_cgelsy_work( int matrix_order, lapack_int m, lapack_int n, - lapack_int nrhs, lapack_complex_float* a, - lapack_int lda, lapack_complex_float* b, - lapack_int ldb, lapack_int* jpvt, float rcond, - lapack_int* rank, lapack_complex_float* work, - lapack_int lwork, float* rwork ); -lapack_int LAPACKE_zgelsy_work( int matrix_order, lapack_int m, lapack_int n, - lapack_int nrhs, lapack_complex_double* a, - lapack_int lda, lapack_complex_double* b, - lapack_int ldb, lapack_int* jpvt, double rcond, - lapack_int* rank, lapack_complex_double* work, - lapack_int lwork, double* rwork ); - -lapack_int LAPACKE_sgeqlf_work( int matrix_order, lapack_int m, lapack_int n, - float* a, lapack_int lda, float* tau, - float* work, lapack_int lwork ); -lapack_int LAPACKE_dgeqlf_work( int matrix_order, lapack_int m, lapack_int n, - double* a, lapack_int lda, double* tau, - double* work, lapack_int lwork ); -lapack_int LAPACKE_cgeqlf_work( int matrix_order, lapack_int m, lapack_int n, - lapack_complex_float* a, lapack_int lda, - lapack_complex_float* tau, - lapack_complex_float* work, lapack_int lwork ); -lapack_int LAPACKE_zgeqlf_work( int matrix_order, lapack_int m, lapack_int n, - lapack_complex_double* a, lapack_int lda, - lapack_complex_double* tau, - lapack_complex_double* work, lapack_int lwork ); - -lapack_int LAPACKE_sgeqp3_work( int matrix_order, lapack_int m, lapack_int n, - float* a, lapack_int lda, lapack_int* jpvt, - float* tau, float* work, lapack_int lwork ); -lapack_int LAPACKE_dgeqp3_work( int matrix_order, lapack_int m, lapack_int n, - double* a, lapack_int lda, lapack_int* jpvt, - double* tau, double* work, lapack_int lwork ); -lapack_int LAPACKE_cgeqp3_work( int matrix_order, lapack_int m, lapack_int n, - lapack_complex_float* a, lapack_int lda, - lapack_int* jpvt, lapack_complex_float* tau, - lapack_complex_float* work, lapack_int lwork, - float* rwork ); -lapack_int LAPACKE_zgeqp3_work( int matrix_order, lapack_int m, lapack_int n, - lapack_complex_double* a, lapack_int lda, - lapack_int* jpvt, lapack_complex_double* tau, - lapack_complex_double* work, lapack_int lwork, - double* rwork ); - -lapack_int LAPACKE_sgeqpf_work( int matrix_order, lapack_int m, lapack_int n, - float* a, lapack_int lda, lapack_int* jpvt, - float* tau, float* work ); -lapack_int LAPACKE_dgeqpf_work( int matrix_order, lapack_int m, lapack_int n, - double* a, lapack_int lda, lapack_int* jpvt, - double* tau, double* work ); -lapack_int LAPACKE_cgeqpf_work( int matrix_order, lapack_int m, lapack_int n, - lapack_complex_float* a, lapack_int lda, - lapack_int* jpvt, lapack_complex_float* tau, - lapack_complex_float* work, float* rwork ); -lapack_int LAPACKE_zgeqpf_work( int matrix_order, lapack_int m, lapack_int n, - lapack_complex_double* a, lapack_int lda, - lapack_int* jpvt, lapack_complex_double* tau, - lapack_complex_double* work, double* rwork ); - -lapack_int LAPACKE_sgeqr2_work( int matrix_order, lapack_int m, lapack_int n, - float* a, lapack_int lda, float* tau, - float* work ); -lapack_int LAPACKE_dgeqr2_work( int matrix_order, lapack_int m, lapack_int n, - double* a, lapack_int lda, double* tau, - double* work ); -lapack_int LAPACKE_cgeqr2_work( int matrix_order, lapack_int m, lapack_int n, - lapack_complex_float* a, lapack_int lda, - lapack_complex_float* tau, - lapack_complex_float* work ); -lapack_int LAPACKE_zgeqr2_work( int matrix_order, lapack_int m, lapack_int n, - lapack_complex_double* a, lapack_int lda, - lapack_complex_double* tau, - lapack_complex_double* work ); - -lapack_int LAPACKE_sgeqrf_work( int matrix_order, lapack_int m, lapack_int n, - float* a, lapack_int lda, float* tau, - float* work, lapack_int lwork ); -lapack_int LAPACKE_dgeqrf_work( int matrix_order, lapack_int m, lapack_int n, - double* a, lapack_int lda, double* tau, - double* work, lapack_int lwork ); -lapack_int LAPACKE_cgeqrf_work( int matrix_order, lapack_int m, lapack_int n, - lapack_complex_float* a, lapack_int lda, - lapack_complex_float* tau, - lapack_complex_float* work, lapack_int lwork ); -lapack_int LAPACKE_zgeqrf_work( int matrix_order, lapack_int m, lapack_int n, - lapack_complex_double* a, lapack_int lda, - lapack_complex_double* tau, - lapack_complex_double* work, lapack_int lwork ); - -lapack_int LAPACKE_sgeqrfp_work( int matrix_order, lapack_int m, lapack_int n, - float* a, lapack_int lda, float* tau, - float* work, lapack_int lwork ); -lapack_int LAPACKE_dgeqrfp_work( int matrix_order, lapack_int m, lapack_int n, - double* a, lapack_int lda, double* tau, - double* work, lapack_int lwork ); -lapack_int LAPACKE_cgeqrfp_work( int matrix_order, lapack_int m, lapack_int n, - lapack_complex_float* a, lapack_int lda, - lapack_complex_float* tau, - lapack_complex_float* work, lapack_int lwork ); -lapack_int LAPACKE_zgeqrfp_work( int matrix_order, lapack_int m, lapack_int n, - lapack_complex_double* a, lapack_int lda, - lapack_complex_double* tau, - lapack_complex_double* work, - lapack_int lwork ); - -lapack_int LAPACKE_sgerfs_work( int matrix_order, char trans, lapack_int n, - lapack_int nrhs, const float* a, lapack_int lda, - const float* af, lapack_int ldaf, - const lapack_int* ipiv, const float* b, - lapack_int ldb, float* x, lapack_int ldx, - float* ferr, float* berr, float* work, - lapack_int* iwork ); -lapack_int LAPACKE_dgerfs_work( int matrix_order, char trans, lapack_int n, - lapack_int nrhs, const double* a, - lapack_int lda, const double* af, - lapack_int ldaf, const lapack_int* ipiv, - const double* b, lapack_int ldb, double* x, - lapack_int ldx, double* ferr, double* berr, - double* work, lapack_int* iwork ); -lapack_int LAPACKE_cgerfs_work( int matrix_order, char trans, lapack_int n, - lapack_int nrhs, const lapack_complex_float* a, - lapack_int lda, const lapack_complex_float* af, - lapack_int ldaf, const lapack_int* ipiv, - const lapack_complex_float* b, lapack_int ldb, - lapack_complex_float* x, lapack_int ldx, - float* ferr, float* berr, - lapack_complex_float* work, float* rwork ); -lapack_int LAPACKE_zgerfs_work( int matrix_order, char trans, lapack_int n, - lapack_int nrhs, const lapack_complex_double* a, - lapack_int lda, const lapack_complex_double* af, - lapack_int ldaf, const lapack_int* ipiv, - const lapack_complex_double* b, lapack_int ldb, - lapack_complex_double* x, lapack_int ldx, - double* ferr, double* berr, - lapack_complex_double* work, double* rwork ); - -lapack_int LAPACKE_sgerfsx_work( int matrix_order, char trans, char equed, - lapack_int n, lapack_int nrhs, const float* a, - lapack_int lda, const float* af, - lapack_int ldaf, const lapack_int* ipiv, - const float* r, const float* c, const float* b, - lapack_int ldb, float* x, lapack_int ldx, - float* rcond, float* berr, - lapack_int n_err_bnds, float* err_bnds_norm, - float* err_bnds_comp, lapack_int nparams, - float* params, float* work, - lapack_int* iwork ); -lapack_int LAPACKE_dgerfsx_work( int matrix_order, char trans, char equed, - lapack_int n, lapack_int nrhs, const double* a, - lapack_int lda, const double* af, - lapack_int ldaf, const lapack_int* ipiv, - const double* r, const double* c, - const double* b, lapack_int ldb, double* x, - lapack_int ldx, double* rcond, double* berr, - lapack_int n_err_bnds, double* err_bnds_norm, - double* err_bnds_comp, lapack_int nparams, - double* params, double* work, - lapack_int* iwork ); -lapack_int LAPACKE_cgerfsx_work( int matrix_order, char trans, char equed, - lapack_int n, lapack_int nrhs, - const lapack_complex_float* a, lapack_int lda, - const lapack_complex_float* af, - lapack_int ldaf, const lapack_int* ipiv, - const float* r, const float* c, - const lapack_complex_float* b, lapack_int ldb, - lapack_complex_float* x, lapack_int ldx, - float* rcond, float* berr, - lapack_int n_err_bnds, float* err_bnds_norm, - float* err_bnds_comp, lapack_int nparams, - float* params, lapack_complex_float* work, - float* rwork ); -lapack_int LAPACKE_zgerfsx_work( int matrix_order, char trans, char equed, - lapack_int n, lapack_int nrhs, - const lapack_complex_double* a, lapack_int lda, - const lapack_complex_double* af, - lapack_int ldaf, const lapack_int* ipiv, - const double* r, const double* c, - const lapack_complex_double* b, lapack_int ldb, - lapack_complex_double* x, lapack_int ldx, - double* rcond, double* berr, - lapack_int n_err_bnds, double* err_bnds_norm, - double* err_bnds_comp, lapack_int nparams, - double* params, lapack_complex_double* work, - double* rwork ); - -lapack_int LAPACKE_sgerqf_work( int matrix_order, lapack_int m, lapack_int n, - float* a, lapack_int lda, float* tau, - float* work, lapack_int lwork ); -lapack_int LAPACKE_dgerqf_work( int matrix_order, lapack_int m, lapack_int n, - double* a, lapack_int lda, double* tau, - double* work, lapack_int lwork ); -lapack_int LAPACKE_cgerqf_work( int matrix_order, lapack_int m, lapack_int n, - lapack_complex_float* a, lapack_int lda, - lapack_complex_float* tau, - lapack_complex_float* work, lapack_int lwork ); -lapack_int LAPACKE_zgerqf_work( int matrix_order, lapack_int m, lapack_int n, - lapack_complex_double* a, lapack_int lda, - lapack_complex_double* tau, - lapack_complex_double* work, lapack_int lwork ); - -lapack_int LAPACKE_sgesdd_work( int matrix_order, char jobz, lapack_int m, - lapack_int n, float* a, lapack_int lda, - float* s, float* u, lapack_int ldu, float* vt, - lapack_int ldvt, float* work, lapack_int lwork, - lapack_int* iwork ); -lapack_int LAPACKE_dgesdd_work( int matrix_order, char jobz, lapack_int m, - lapack_int n, double* a, lapack_int lda, - double* s, double* u, lapack_int ldu, - double* vt, lapack_int ldvt, double* work, - lapack_int lwork, lapack_int* iwork ); -lapack_int LAPACKE_cgesdd_work( int matrix_order, char jobz, lapack_int m, - lapack_int n, lapack_complex_float* a, - lapack_int lda, float* s, - lapack_complex_float* u, lapack_int ldu, - lapack_complex_float* vt, lapack_int ldvt, - lapack_complex_float* work, lapack_int lwork, - float* rwork, lapack_int* iwork ); -lapack_int LAPACKE_zgesdd_work( int matrix_order, char jobz, lapack_int m, - lapack_int n, lapack_complex_double* a, - lapack_int lda, double* s, - lapack_complex_double* u, lapack_int ldu, - lapack_complex_double* vt, lapack_int ldvt, - lapack_complex_double* work, lapack_int lwork, - double* rwork, lapack_int* iwork ); - -lapack_int LAPACKE_sgesv_work( int matrix_order, lapack_int n, lapack_int nrhs, - float* a, lapack_int lda, lapack_int* ipiv, - float* b, lapack_int ldb ); -lapack_int LAPACKE_dgesv_work( int matrix_order, lapack_int n, lapack_int nrhs, - double* a, lapack_int lda, lapack_int* ipiv, - double* b, lapack_int ldb ); -lapack_int LAPACKE_cgesv_work( int matrix_order, lapack_int n, lapack_int nrhs, - lapack_complex_float* a, lapack_int lda, - lapack_int* ipiv, lapack_complex_float* b, - lapack_int ldb ); -lapack_int LAPACKE_zgesv_work( int matrix_order, lapack_int n, lapack_int nrhs, - lapack_complex_double* a, lapack_int lda, - lapack_int* ipiv, lapack_complex_double* b, - lapack_int ldb ); -lapack_int LAPACKE_dsgesv_work( int matrix_order, lapack_int n, lapack_int nrhs, - double* a, lapack_int lda, lapack_int* ipiv, - double* b, lapack_int ldb, double* x, - lapack_int ldx, double* work, float* swork, - lapack_int* iter ); -lapack_int LAPACKE_zcgesv_work( int matrix_order, lapack_int n, lapack_int nrhs, - lapack_complex_double* a, lapack_int lda, - lapack_int* ipiv, lapack_complex_double* b, - lapack_int ldb, lapack_complex_double* x, - lapack_int ldx, lapack_complex_double* work, - lapack_complex_float* swork, double* rwork, - lapack_int* iter ); - -lapack_int LAPACKE_sgesvd_work( int matrix_order, char jobu, char jobvt, - lapack_int m, lapack_int n, float* a, - lapack_int lda, float* s, float* u, - lapack_int ldu, float* vt, lapack_int ldvt, - float* work, lapack_int lwork ); -lapack_int LAPACKE_dgesvd_work( int matrix_order, char jobu, char jobvt, - lapack_int m, lapack_int n, double* a, - lapack_int lda, double* s, double* u, - lapack_int ldu, double* vt, lapack_int ldvt, - double* work, lapack_int lwork ); -lapack_int LAPACKE_cgesvd_work( int matrix_order, char jobu, char jobvt, - lapack_int m, lapack_int n, - lapack_complex_float* a, lapack_int lda, - float* s, lapack_complex_float* u, - lapack_int ldu, lapack_complex_float* vt, - lapack_int ldvt, lapack_complex_float* work, - lapack_int lwork, float* rwork ); -lapack_int LAPACKE_zgesvd_work( int matrix_order, char jobu, char jobvt, - lapack_int m, lapack_int n, - lapack_complex_double* a, lapack_int lda, - double* s, lapack_complex_double* u, - lapack_int ldu, lapack_complex_double* vt, - lapack_int ldvt, lapack_complex_double* work, - lapack_int lwork, double* rwork ); - -lapack_int LAPACKE_sgesvj_work( int matrix_order, char joba, char jobu, - char jobv, lapack_int m, lapack_int n, float* a, - lapack_int lda, float* sva, lapack_int mv, - float* v, lapack_int ldv, float* work, - lapack_int lwork ); -lapack_int LAPACKE_dgesvj_work( int matrix_order, char joba, char jobu, - char jobv, lapack_int m, lapack_int n, - double* a, lapack_int lda, double* sva, - lapack_int mv, double* v, lapack_int ldv, - double* work, lapack_int lwork ); - -lapack_int LAPACKE_sgesvx_work( int matrix_order, char fact, char trans, - lapack_int n, lapack_int nrhs, float* a, - lapack_int lda, float* af, lapack_int ldaf, - lapack_int* ipiv, char* equed, float* r, - float* c, float* b, lapack_int ldb, float* x, - lapack_int ldx, float* rcond, float* ferr, - float* berr, float* work, lapack_int* iwork ); -lapack_int LAPACKE_dgesvx_work( int matrix_order, char fact, char trans, - lapack_int n, lapack_int nrhs, double* a, - lapack_int lda, double* af, lapack_int ldaf, - lapack_int* ipiv, char* equed, double* r, - double* c, double* b, lapack_int ldb, double* x, - lapack_int ldx, double* rcond, double* ferr, - double* berr, double* work, lapack_int* iwork ); -lapack_int LAPACKE_cgesvx_work( int matrix_order, char fact, char trans, - lapack_int n, lapack_int nrhs, - lapack_complex_float* a, lapack_int lda, - lapack_complex_float* af, lapack_int ldaf, - lapack_int* ipiv, char* equed, float* r, - float* c, lapack_complex_float* b, - lapack_int ldb, lapack_complex_float* x, - lapack_int ldx, float* rcond, float* ferr, - float* berr, lapack_complex_float* work, - float* rwork ); -lapack_int LAPACKE_zgesvx_work( int matrix_order, char fact, char trans, - lapack_int n, lapack_int nrhs, - lapack_complex_double* a, lapack_int lda, - lapack_complex_double* af, lapack_int ldaf, - lapack_int* ipiv, char* equed, double* r, - double* c, lapack_complex_double* b, - lapack_int ldb, lapack_complex_double* x, - lapack_int ldx, double* rcond, double* ferr, - double* berr, lapack_complex_double* work, - double* rwork ); - -lapack_int LAPACKE_sgesvxx_work( int matrix_order, char fact, char trans, - lapack_int n, lapack_int nrhs, float* a, - lapack_int lda, float* af, lapack_int ldaf, - lapack_int* ipiv, char* equed, float* r, - float* c, float* b, lapack_int ldb, float* x, - lapack_int ldx, float* rcond, float* rpvgrw, - float* berr, lapack_int n_err_bnds, - float* err_bnds_norm, float* err_bnds_comp, - lapack_int nparams, float* params, float* work, - lapack_int* iwork ); -lapack_int LAPACKE_dgesvxx_work( int matrix_order, char fact, char trans, - lapack_int n, lapack_int nrhs, double* a, - lapack_int lda, double* af, lapack_int ldaf, - lapack_int* ipiv, char* equed, double* r, - double* c, double* b, lapack_int ldb, - double* x, lapack_int ldx, double* rcond, - double* rpvgrw, double* berr, - lapack_int n_err_bnds, double* err_bnds_norm, - double* err_bnds_comp, lapack_int nparams, - double* params, double* work, - lapack_int* iwork ); -lapack_int LAPACKE_cgesvxx_work( int matrix_order, char fact, char trans, - lapack_int n, lapack_int nrhs, - lapack_complex_float* a, lapack_int lda, - lapack_complex_float* af, lapack_int ldaf, - lapack_int* ipiv, char* equed, float* r, - float* c, lapack_complex_float* b, - lapack_int ldb, lapack_complex_float* x, - lapack_int ldx, float* rcond, float* rpvgrw, - float* berr, lapack_int n_err_bnds, - float* err_bnds_norm, float* err_bnds_comp, - lapack_int nparams, float* params, - lapack_complex_float* work, float* rwork ); -lapack_int LAPACKE_zgesvxx_work( int matrix_order, char fact, char trans, - lapack_int n, lapack_int nrhs, - lapack_complex_double* a, lapack_int lda, - lapack_complex_double* af, lapack_int ldaf, - lapack_int* ipiv, char* equed, double* r, - double* c, lapack_complex_double* b, - lapack_int ldb, lapack_complex_double* x, - lapack_int ldx, double* rcond, double* rpvgrw, - double* berr, lapack_int n_err_bnds, - double* err_bnds_norm, double* err_bnds_comp, - lapack_int nparams, double* params, - lapack_complex_double* work, double* rwork ); - -lapack_int LAPACKE_sgetf2_work( int matrix_order, lapack_int m, lapack_int n, - float* a, lapack_int lda, lapack_int* ipiv ); -lapack_int LAPACKE_dgetf2_work( int matrix_order, lapack_int m, lapack_int n, - double* a, lapack_int lda, lapack_int* ipiv ); -lapack_int LAPACKE_cgetf2_work( int matrix_order, lapack_int m, lapack_int n, - lapack_complex_float* a, lapack_int lda, - lapack_int* ipiv ); -lapack_int LAPACKE_zgetf2_work( int matrix_order, lapack_int m, lapack_int n, - lapack_complex_double* a, lapack_int lda, - lapack_int* ipiv ); - -lapack_int LAPACKE_sgetrf_work( int matrix_order, lapack_int m, lapack_int n, - float* a, lapack_int lda, lapack_int* ipiv ); -lapack_int LAPACKE_dgetrf_work( int matrix_order, lapack_int m, lapack_int n, - double* a, lapack_int lda, lapack_int* ipiv ); -lapack_int LAPACKE_cgetrf_work( int matrix_order, lapack_int m, lapack_int n, - lapack_complex_float* a, lapack_int lda, - lapack_int* ipiv ); -lapack_int LAPACKE_zgetrf_work( int matrix_order, lapack_int m, lapack_int n, - lapack_complex_double* a, lapack_int lda, - lapack_int* ipiv ); - -lapack_int LAPACKE_sgetri_work( int matrix_order, lapack_int n, float* a, - lapack_int lda, const lapack_int* ipiv, - float* work, lapack_int lwork ); -lapack_int LAPACKE_dgetri_work( int matrix_order, lapack_int n, double* a, - lapack_int lda, const lapack_int* ipiv, - double* work, lapack_int lwork ); -lapack_int LAPACKE_cgetri_work( int matrix_order, lapack_int n, - lapack_complex_float* a, lapack_int lda, - const lapack_int* ipiv, - lapack_complex_float* work, lapack_int lwork ); -lapack_int LAPACKE_zgetri_work( int matrix_order, lapack_int n, - lapack_complex_double* a, lapack_int lda, - const lapack_int* ipiv, - lapack_complex_double* work, lapack_int lwork ); - -lapack_int LAPACKE_sgetrs_work( int matrix_order, char trans, lapack_int n, - lapack_int nrhs, const float* a, lapack_int lda, - const lapack_int* ipiv, float* b, - lapack_int ldb ); -lapack_int LAPACKE_dgetrs_work( int matrix_order, char trans, lapack_int n, - lapack_int nrhs, const double* a, - lapack_int lda, const lapack_int* ipiv, - double* b, lapack_int ldb ); -lapack_int LAPACKE_cgetrs_work( int matrix_order, char trans, lapack_int n, - lapack_int nrhs, const lapack_complex_float* a, - lapack_int lda, const lapack_int* ipiv, - lapack_complex_float* b, lapack_int ldb ); -lapack_int LAPACKE_zgetrs_work( int matrix_order, char trans, lapack_int n, - lapack_int nrhs, const lapack_complex_double* a, - lapack_int lda, const lapack_int* ipiv, - lapack_complex_double* b, lapack_int ldb ); - -lapack_int LAPACKE_sggbak_work( int matrix_order, char job, char side, - lapack_int n, lapack_int ilo, lapack_int ihi, - const float* lscale, const float* rscale, - lapack_int m, float* v, lapack_int ldv ); -lapack_int LAPACKE_dggbak_work( int matrix_order, char job, char side, - lapack_int n, lapack_int ilo, lapack_int ihi, - const double* lscale, const double* rscale, - lapack_int m, double* v, lapack_int ldv ); -lapack_int LAPACKE_cggbak_work( int matrix_order, char job, char side, - lapack_int n, lapack_int ilo, lapack_int ihi, - const float* lscale, const float* rscale, - lapack_int m, lapack_complex_float* v, - lapack_int ldv ); -lapack_int LAPACKE_zggbak_work( int matrix_order, char job, char side, - lapack_int n, lapack_int ilo, lapack_int ihi, - const double* lscale, const double* rscale, - lapack_int m, lapack_complex_double* v, - lapack_int ldv ); - -lapack_int LAPACKE_sggbal_work( int matrix_order, char job, lapack_int n, - float* a, lapack_int lda, float* b, - lapack_int ldb, lapack_int* ilo, - lapack_int* ihi, float* lscale, float* rscale, - float* work ); -lapack_int LAPACKE_dggbal_work( int matrix_order, char job, lapack_int n, - double* a, lapack_int lda, double* b, - lapack_int ldb, lapack_int* ilo, - lapack_int* ihi, double* lscale, double* rscale, - double* work ); -lapack_int LAPACKE_cggbal_work( int matrix_order, char job, lapack_int n, - lapack_complex_float* a, lapack_int lda, - lapack_complex_float* b, lapack_int ldb, - lapack_int* ilo, lapack_int* ihi, float* lscale, - float* rscale, float* work ); -lapack_int LAPACKE_zggbal_work( int matrix_order, char job, lapack_int n, - lapack_complex_double* a, lapack_int lda, - lapack_complex_double* b, lapack_int ldb, - lapack_int* ilo, lapack_int* ihi, - double* lscale, double* rscale, double* work ); - -lapack_int LAPACKE_sgges_work( int matrix_order, char jobvsl, char jobvsr, - char sort, LAPACK_S_SELECT3 selctg, lapack_int n, - float* a, lapack_int lda, float* b, - lapack_int ldb, lapack_int* sdim, float* alphar, - float* alphai, float* beta, float* vsl, - lapack_int ldvsl, float* vsr, lapack_int ldvsr, - float* work, lapack_int lwork, - lapack_logical* bwork ); -lapack_int LAPACKE_dgges_work( int matrix_order, char jobvsl, char jobvsr, - char sort, LAPACK_D_SELECT3 selctg, lapack_int n, - double* a, lapack_int lda, double* b, - lapack_int ldb, lapack_int* sdim, double* alphar, - double* alphai, double* beta, double* vsl, - lapack_int ldvsl, double* vsr, lapack_int ldvsr, - double* work, lapack_int lwork, - lapack_logical* bwork ); -lapack_int LAPACKE_cgges_work( int matrix_order, char jobvsl, char jobvsr, - char sort, LAPACK_C_SELECT2 selctg, lapack_int n, - lapack_complex_float* a, lapack_int lda, - lapack_complex_float* b, lapack_int ldb, - lapack_int* sdim, lapack_complex_float* alpha, - lapack_complex_float* beta, - lapack_complex_float* vsl, lapack_int ldvsl, - lapack_complex_float* vsr, lapack_int ldvsr, - lapack_complex_float* work, lapack_int lwork, - float* rwork, lapack_logical* bwork ); -lapack_int LAPACKE_zgges_work( int matrix_order, char jobvsl, char jobvsr, - char sort, LAPACK_Z_SELECT2 selctg, lapack_int n, - lapack_complex_double* a, lapack_int lda, - lapack_complex_double* b, lapack_int ldb, - lapack_int* sdim, lapack_complex_double* alpha, - lapack_complex_double* beta, - lapack_complex_double* vsl, lapack_int ldvsl, - lapack_complex_double* vsr, lapack_int ldvsr, - lapack_complex_double* work, lapack_int lwork, - double* rwork, lapack_logical* bwork ); - -lapack_int LAPACKE_sggesx_work( int matrix_order, char jobvsl, char jobvsr, - char sort, LAPACK_S_SELECT3 selctg, char sense, - lapack_int n, float* a, lapack_int lda, - float* b, lapack_int ldb, lapack_int* sdim, - float* alphar, float* alphai, float* beta, - float* vsl, lapack_int ldvsl, float* vsr, - lapack_int ldvsr, float* rconde, float* rcondv, - float* work, lapack_int lwork, - lapack_int* iwork, lapack_int liwork, - lapack_logical* bwork ); -lapack_int LAPACKE_dggesx_work( int matrix_order, char jobvsl, char jobvsr, - char sort, LAPACK_D_SELECT3 selctg, char sense, - lapack_int n, double* a, lapack_int lda, - double* b, lapack_int ldb, lapack_int* sdim, - double* alphar, double* alphai, double* beta, - double* vsl, lapack_int ldvsl, double* vsr, - lapack_int ldvsr, double* rconde, - double* rcondv, double* work, lapack_int lwork, - lapack_int* iwork, lapack_int liwork, - lapack_logical* bwork ); -lapack_int LAPACKE_cggesx_work( int matrix_order, char jobvsl, char jobvsr, - char sort, LAPACK_C_SELECT2 selctg, char sense, - lapack_int n, lapack_complex_float* a, - lapack_int lda, lapack_complex_float* b, - lapack_int ldb, lapack_int* sdim, - lapack_complex_float* alpha, - lapack_complex_float* beta, - lapack_complex_float* vsl, lapack_int ldvsl, - lapack_complex_float* vsr, lapack_int ldvsr, - float* rconde, float* rcondv, - lapack_complex_float* work, lapack_int lwork, - float* rwork, lapack_int* iwork, - lapack_int liwork, lapack_logical* bwork ); -lapack_int LAPACKE_zggesx_work( int matrix_order, char jobvsl, char jobvsr, - char sort, LAPACK_Z_SELECT2 selctg, char sense, - lapack_int n, lapack_complex_double* a, - lapack_int lda, lapack_complex_double* b, - lapack_int ldb, lapack_int* sdim, - lapack_complex_double* alpha, - lapack_complex_double* beta, - lapack_complex_double* vsl, lapack_int ldvsl, - lapack_complex_double* vsr, lapack_int ldvsr, - double* rconde, double* rcondv, - lapack_complex_double* work, lapack_int lwork, - double* rwork, lapack_int* iwork, - lapack_int liwork, lapack_logical* bwork ); - -lapack_int LAPACKE_sggev_work( int matrix_order, char jobvl, char jobvr, - lapack_int n, float* a, lapack_int lda, float* b, - lapack_int ldb, float* alphar, float* alphai, - float* beta, float* vl, lapack_int ldvl, - float* vr, lapack_int ldvr, float* work, - lapack_int lwork ); -lapack_int LAPACKE_dggev_work( int matrix_order, char jobvl, char jobvr, - lapack_int n, double* a, lapack_int lda, - double* b, lapack_int ldb, double* alphar, - double* alphai, double* beta, double* vl, - lapack_int ldvl, double* vr, lapack_int ldvr, - double* work, lapack_int lwork ); -lapack_int LAPACKE_cggev_work( int matrix_order, char jobvl, char jobvr, - lapack_int n, lapack_complex_float* a, - lapack_int lda, lapack_complex_float* b, - lapack_int ldb, lapack_complex_float* alpha, - lapack_complex_float* beta, - lapack_complex_float* vl, lapack_int ldvl, - lapack_complex_float* vr, lapack_int ldvr, - lapack_complex_float* work, lapack_int lwork, - float* rwork ); -lapack_int LAPACKE_zggev_work( int matrix_order, char jobvl, char jobvr, - lapack_int n, lapack_complex_double* a, - lapack_int lda, lapack_complex_double* b, - lapack_int ldb, lapack_complex_double* alpha, - lapack_complex_double* beta, - lapack_complex_double* vl, lapack_int ldvl, - lapack_complex_double* vr, lapack_int ldvr, - lapack_complex_double* work, lapack_int lwork, - double* rwork ); - -lapack_int LAPACKE_sggevx_work( int matrix_order, char balanc, char jobvl, - char jobvr, char sense, lapack_int n, float* a, - lapack_int lda, float* b, lapack_int ldb, - float* alphar, float* alphai, float* beta, - float* vl, lapack_int ldvl, float* vr, - lapack_int ldvr, lapack_int* ilo, - lapack_int* ihi, float* lscale, float* rscale, - float* abnrm, float* bbnrm, float* rconde, - float* rcondv, float* work, lapack_int lwork, - lapack_int* iwork, lapack_logical* bwork ); -lapack_int LAPACKE_dggevx_work( int matrix_order, char balanc, char jobvl, - char jobvr, char sense, lapack_int n, double* a, - lapack_int lda, double* b, lapack_int ldb, - double* alphar, double* alphai, double* beta, - double* vl, lapack_int ldvl, double* vr, - lapack_int ldvr, lapack_int* ilo, - lapack_int* ihi, double* lscale, double* rscale, - double* abnrm, double* bbnrm, double* rconde, - double* rcondv, double* work, lapack_int lwork, - lapack_int* iwork, lapack_logical* bwork ); -lapack_int LAPACKE_cggevx_work( int matrix_order, char balanc, char jobvl, - char jobvr, char sense, lapack_int n, - lapack_complex_float* a, lapack_int lda, - lapack_complex_float* b, lapack_int ldb, - lapack_complex_float* alpha, - lapack_complex_float* beta, - lapack_complex_float* vl, lapack_int ldvl, - lapack_complex_float* vr, lapack_int ldvr, - lapack_int* ilo, lapack_int* ihi, float* lscale, - float* rscale, float* abnrm, float* bbnrm, - float* rconde, float* rcondv, - lapack_complex_float* work, lapack_int lwork, - float* rwork, lapack_int* iwork, - lapack_logical* bwork ); -lapack_int LAPACKE_zggevx_work( int matrix_order, char balanc, char jobvl, - char jobvr, char sense, lapack_int n, - lapack_complex_double* a, lapack_int lda, - lapack_complex_double* b, lapack_int ldb, - lapack_complex_double* alpha, - lapack_complex_double* beta, - lapack_complex_double* vl, lapack_int ldvl, - lapack_complex_double* vr, lapack_int ldvr, - lapack_int* ilo, lapack_int* ihi, - double* lscale, double* rscale, double* abnrm, - double* bbnrm, double* rconde, double* rcondv, - lapack_complex_double* work, lapack_int lwork, - double* rwork, lapack_int* iwork, - lapack_logical* bwork ); - -lapack_int LAPACKE_sggglm_work( int matrix_order, lapack_int n, lapack_int m, - lapack_int p, float* a, lapack_int lda, - float* b, lapack_int ldb, float* d, float* x, - float* y, float* work, lapack_int lwork ); -lapack_int LAPACKE_dggglm_work( int matrix_order, lapack_int n, lapack_int m, - lapack_int p, double* a, lapack_int lda, - double* b, lapack_int ldb, double* d, double* x, - double* y, double* work, lapack_int lwork ); -lapack_int LAPACKE_cggglm_work( int matrix_order, lapack_int n, lapack_int m, - lapack_int p, lapack_complex_float* a, - lapack_int lda, lapack_complex_float* b, - lapack_int ldb, lapack_complex_float* d, - lapack_complex_float* x, - lapack_complex_float* y, - lapack_complex_float* work, lapack_int lwork ); -lapack_int LAPACKE_zggglm_work( int matrix_order, lapack_int n, lapack_int m, - lapack_int p, lapack_complex_double* a, - lapack_int lda, lapack_complex_double* b, - lapack_int ldb, lapack_complex_double* d, - lapack_complex_double* x, - lapack_complex_double* y, - lapack_complex_double* work, lapack_int lwork ); - -lapack_int LAPACKE_sgghrd_work( int matrix_order, char compq, char compz, - lapack_int n, lapack_int ilo, lapack_int ihi, - float* a, lapack_int lda, float* b, - lapack_int ldb, float* q, lapack_int ldq, - float* z, lapack_int ldz ); -lapack_int LAPACKE_dgghrd_work( int matrix_order, char compq, char compz, - lapack_int n, lapack_int ilo, lapack_int ihi, - double* a, lapack_int lda, double* b, - lapack_int ldb, double* q, lapack_int ldq, - double* z, lapack_int ldz ); -lapack_int LAPACKE_cgghrd_work( int matrix_order, char compq, char compz, - lapack_int n, lapack_int ilo, lapack_int ihi, - lapack_complex_float* a, lapack_int lda, - lapack_complex_float* b, lapack_int ldb, - lapack_complex_float* q, lapack_int ldq, - lapack_complex_float* z, lapack_int ldz ); -lapack_int LAPACKE_zgghrd_work( int matrix_order, char compq, char compz, - lapack_int n, lapack_int ilo, lapack_int ihi, - lapack_complex_double* a, lapack_int lda, - lapack_complex_double* b, lapack_int ldb, - lapack_complex_double* q, lapack_int ldq, - lapack_complex_double* z, lapack_int ldz ); - -lapack_int LAPACKE_sgglse_work( int matrix_order, lapack_int m, lapack_int n, - lapack_int p, float* a, lapack_int lda, - float* b, lapack_int ldb, float* c, float* d, - float* x, float* work, lapack_int lwork ); -lapack_int LAPACKE_dgglse_work( int matrix_order, lapack_int m, lapack_int n, - lapack_int p, double* a, lapack_int lda, - double* b, lapack_int ldb, double* c, double* d, - double* x, double* work, lapack_int lwork ); -lapack_int LAPACKE_cgglse_work( int matrix_order, lapack_int m, lapack_int n, - lapack_int p, lapack_complex_float* a, - lapack_int lda, lapack_complex_float* b, - lapack_int ldb, lapack_complex_float* c, - lapack_complex_float* d, - lapack_complex_float* x, - lapack_complex_float* work, lapack_int lwork ); -lapack_int LAPACKE_zgglse_work( int matrix_order, lapack_int m, lapack_int n, - lapack_int p, lapack_complex_double* a, - lapack_int lda, lapack_complex_double* b, - lapack_int ldb, lapack_complex_double* c, - lapack_complex_double* d, - lapack_complex_double* x, - lapack_complex_double* work, lapack_int lwork ); - -lapack_int LAPACKE_sggqrf_work( int matrix_order, lapack_int n, lapack_int m, - lapack_int p, float* a, lapack_int lda, - float* taua, float* b, lapack_int ldb, - float* taub, float* work, lapack_int lwork ); -lapack_int LAPACKE_dggqrf_work( int matrix_order, lapack_int n, lapack_int m, - lapack_int p, double* a, lapack_int lda, - double* taua, double* b, lapack_int ldb, - double* taub, double* work, lapack_int lwork ); -lapack_int LAPACKE_cggqrf_work( int matrix_order, lapack_int n, lapack_int m, - lapack_int p, lapack_complex_float* a, - lapack_int lda, lapack_complex_float* taua, - lapack_complex_float* b, lapack_int ldb, - lapack_complex_float* taub, - lapack_complex_float* work, lapack_int lwork ); -lapack_int LAPACKE_zggqrf_work( int matrix_order, lapack_int n, lapack_int m, - lapack_int p, lapack_complex_double* a, - lapack_int lda, lapack_complex_double* taua, - lapack_complex_double* b, lapack_int ldb, - lapack_complex_double* taub, - lapack_complex_double* work, lapack_int lwork ); - -lapack_int LAPACKE_sggrqf_work( int matrix_order, lapack_int m, lapack_int p, - lapack_int n, float* a, lapack_int lda, - float* taua, float* b, lapack_int ldb, - float* taub, float* work, lapack_int lwork ); -lapack_int LAPACKE_dggrqf_work( int matrix_order, lapack_int m, lapack_int p, - lapack_int n, double* a, lapack_int lda, - double* taua, double* b, lapack_int ldb, - double* taub, double* work, lapack_int lwork ); -lapack_int LAPACKE_cggrqf_work( int matrix_order, lapack_int m, lapack_int p, - lapack_int n, lapack_complex_float* a, - lapack_int lda, lapack_complex_float* taua, - lapack_complex_float* b, lapack_int ldb, - lapack_complex_float* taub, - lapack_complex_float* work, lapack_int lwork ); -lapack_int LAPACKE_zggrqf_work( int matrix_order, lapack_int m, lapack_int p, - lapack_int n, lapack_complex_double* a, - lapack_int lda, lapack_complex_double* taua, - lapack_complex_double* b, lapack_int ldb, - lapack_complex_double* taub, - lapack_complex_double* work, lapack_int lwork ); - -lapack_int LAPACKE_sggsvd_work( int matrix_order, char jobu, char jobv, - char jobq, lapack_int m, lapack_int n, - lapack_int p, lapack_int* k, lapack_int* l, - float* a, lapack_int lda, float* b, - lapack_int ldb, float* alpha, float* beta, - float* u, lapack_int ldu, float* v, - lapack_int ldv, float* q, lapack_int ldq, - float* work, lapack_int* iwork ); -lapack_int LAPACKE_dggsvd_work( int matrix_order, char jobu, char jobv, - char jobq, lapack_int m, lapack_int n, - lapack_int p, lapack_int* k, lapack_int* l, - double* a, lapack_int lda, double* b, - lapack_int ldb, double* alpha, double* beta, - double* u, lapack_int ldu, double* v, - lapack_int ldv, double* q, lapack_int ldq, - double* work, lapack_int* iwork ); -lapack_int LAPACKE_cggsvd_work( int matrix_order, char jobu, char jobv, - char jobq, lapack_int m, lapack_int n, - lapack_int p, lapack_int* k, lapack_int* l, - lapack_complex_float* a, lapack_int lda, - lapack_complex_float* b, lapack_int ldb, - float* alpha, float* beta, - lapack_complex_float* u, lapack_int ldu, - lapack_complex_float* v, lapack_int ldv, - lapack_complex_float* q, lapack_int ldq, - lapack_complex_float* work, float* rwork, - lapack_int* iwork ); -lapack_int LAPACKE_zggsvd_work( int matrix_order, char jobu, char jobv, - char jobq, lapack_int m, lapack_int n, - lapack_int p, lapack_int* k, lapack_int* l, - lapack_complex_double* a, lapack_int lda, - lapack_complex_double* b, lapack_int ldb, - double* alpha, double* beta, - lapack_complex_double* u, lapack_int ldu, - lapack_complex_double* v, lapack_int ldv, - lapack_complex_double* q, lapack_int ldq, - lapack_complex_double* work, double* rwork, - lapack_int* iwork ); - -lapack_int LAPACKE_sggsvp_work( int matrix_order, char jobu, char jobv, - char jobq, lapack_int m, lapack_int p, - lapack_int n, float* a, lapack_int lda, - float* b, lapack_int ldb, float tola, - float tolb, lapack_int* k, lapack_int* l, - float* u, lapack_int ldu, float* v, - lapack_int ldv, float* q, lapack_int ldq, - lapack_int* iwork, float* tau, float* work ); -lapack_int LAPACKE_dggsvp_work( int matrix_order, char jobu, char jobv, - char jobq, lapack_int m, lapack_int p, - lapack_int n, double* a, lapack_int lda, - double* b, lapack_int ldb, double tola, - double tolb, lapack_int* k, lapack_int* l, - double* u, lapack_int ldu, double* v, - lapack_int ldv, double* q, lapack_int ldq, - lapack_int* iwork, double* tau, double* work ); -lapack_int LAPACKE_cggsvp_work( int matrix_order, char jobu, char jobv, - char jobq, lapack_int m, lapack_int p, - lapack_int n, lapack_complex_float* a, - lapack_int lda, lapack_complex_float* b, - lapack_int ldb, float tola, float tolb, - lapack_int* k, lapack_int* l, - lapack_complex_float* u, lapack_int ldu, - lapack_complex_float* v, lapack_int ldv, - lapack_complex_float* q, lapack_int ldq, - lapack_int* iwork, float* rwork, - lapack_complex_float* tau, - lapack_complex_float* work ); -lapack_int LAPACKE_zggsvp_work( int matrix_order, char jobu, char jobv, - char jobq, lapack_int m, lapack_int p, - lapack_int n, lapack_complex_double* a, - lapack_int lda, lapack_complex_double* b, - lapack_int ldb, double tola, double tolb, - lapack_int* k, lapack_int* l, - lapack_complex_double* u, lapack_int ldu, - lapack_complex_double* v, lapack_int ldv, - lapack_complex_double* q, lapack_int ldq, - lapack_int* iwork, double* rwork, - lapack_complex_double* tau, - lapack_complex_double* work ); - -lapack_int LAPACKE_sgtcon_work( char norm, lapack_int n, const float* dl, - const float* d, const float* du, - const float* du2, const lapack_int* ipiv, - float anorm, float* rcond, float* work, - lapack_int* iwork ); -lapack_int LAPACKE_dgtcon_work( char norm, lapack_int n, const double* dl, - const double* d, const double* du, - const double* du2, const lapack_int* ipiv, - double anorm, double* rcond, double* work, - lapack_int* iwork ); -lapack_int LAPACKE_cgtcon_work( char norm, lapack_int n, - const lapack_complex_float* dl, - const lapack_complex_float* d, - const lapack_complex_float* du, - const lapack_complex_float* du2, - const lapack_int* ipiv, float anorm, - float* rcond, lapack_complex_float* work ); -lapack_int LAPACKE_zgtcon_work( char norm, lapack_int n, - const lapack_complex_double* dl, - const lapack_complex_double* d, - const lapack_complex_double* du, - const lapack_complex_double* du2, - const lapack_int* ipiv, double anorm, - double* rcond, lapack_complex_double* work ); - -lapack_int LAPACKE_sgtrfs_work( int matrix_order, char trans, lapack_int n, - lapack_int nrhs, const float* dl, - const float* d, const float* du, - const float* dlf, const float* df, - const float* duf, const float* du2, - const lapack_int* ipiv, const float* b, - lapack_int ldb, float* x, lapack_int ldx, - float* ferr, float* berr, float* work, - lapack_int* iwork ); -lapack_int LAPACKE_dgtrfs_work( int matrix_order, char trans, lapack_int n, - lapack_int nrhs, const double* dl, - const double* d, const double* du, - const double* dlf, const double* df, - const double* duf, const double* du2, - const lapack_int* ipiv, const double* b, - lapack_int ldb, double* x, lapack_int ldx, - double* ferr, double* berr, double* work, - lapack_int* iwork ); -lapack_int LAPACKE_cgtrfs_work( int matrix_order, char trans, lapack_int n, - lapack_int nrhs, const lapack_complex_float* dl, - const lapack_complex_float* d, - const lapack_complex_float* du, - const lapack_complex_float* dlf, - const lapack_complex_float* df, - const lapack_complex_float* duf, - const lapack_complex_float* du2, - const lapack_int* ipiv, - const lapack_complex_float* b, lapack_int ldb, - lapack_complex_float* x, lapack_int ldx, - float* ferr, float* berr, - lapack_complex_float* work, float* rwork ); -lapack_int LAPACKE_zgtrfs_work( int matrix_order, char trans, lapack_int n, - lapack_int nrhs, - const lapack_complex_double* dl, - const lapack_complex_double* d, - const lapack_complex_double* du, - const lapack_complex_double* dlf, - const lapack_complex_double* df, - const lapack_complex_double* duf, - const lapack_complex_double* du2, - const lapack_int* ipiv, - const lapack_complex_double* b, lapack_int ldb, - lapack_complex_double* x, lapack_int ldx, - double* ferr, double* berr, - lapack_complex_double* work, double* rwork ); - -lapack_int LAPACKE_sgtsv_work( int matrix_order, lapack_int n, lapack_int nrhs, - float* dl, float* d, float* du, float* b, - lapack_int ldb ); -lapack_int LAPACKE_dgtsv_work( int matrix_order, lapack_int n, lapack_int nrhs, - double* dl, double* d, double* du, double* b, - lapack_int ldb ); -lapack_int LAPACKE_cgtsv_work( int matrix_order, lapack_int n, lapack_int nrhs, - lapack_complex_float* dl, - lapack_complex_float* d, - lapack_complex_float* du, - lapack_complex_float* b, lapack_int ldb ); -lapack_int LAPACKE_zgtsv_work( int matrix_order, lapack_int n, lapack_int nrhs, - lapack_complex_double* dl, - lapack_complex_double* d, - lapack_complex_double* du, - lapack_complex_double* b, lapack_int ldb ); - -lapack_int LAPACKE_sgtsvx_work( int matrix_order, char fact, char trans, - lapack_int n, lapack_int nrhs, const float* dl, - const float* d, const float* du, float* dlf, - float* df, float* duf, float* du2, - lapack_int* ipiv, const float* b, - lapack_int ldb, float* x, lapack_int ldx, - float* rcond, float* ferr, float* berr, - float* work, lapack_int* iwork ); -lapack_int LAPACKE_dgtsvx_work( int matrix_order, char fact, char trans, - lapack_int n, lapack_int nrhs, const double* dl, - const double* d, const double* du, double* dlf, - double* df, double* duf, double* du2, - lapack_int* ipiv, const double* b, - lapack_int ldb, double* x, lapack_int ldx, - double* rcond, double* ferr, double* berr, - double* work, lapack_int* iwork ); -lapack_int LAPACKE_cgtsvx_work( int matrix_order, char fact, char trans, - lapack_int n, lapack_int nrhs, - const lapack_complex_float* dl, - const lapack_complex_float* d, - const lapack_complex_float* du, - lapack_complex_float* dlf, - lapack_complex_float* df, - lapack_complex_float* duf, - lapack_complex_float* du2, lapack_int* ipiv, - const lapack_complex_float* b, lapack_int ldb, - lapack_complex_float* x, lapack_int ldx, - float* rcond, float* ferr, float* berr, - lapack_complex_float* work, float* rwork ); -lapack_int LAPACKE_zgtsvx_work( int matrix_order, char fact, char trans, - lapack_int n, lapack_int nrhs, - const lapack_complex_double* dl, - const lapack_complex_double* d, - const lapack_complex_double* du, - lapack_complex_double* dlf, - lapack_complex_double* df, - lapack_complex_double* duf, - lapack_complex_double* du2, lapack_int* ipiv, - const lapack_complex_double* b, lapack_int ldb, - lapack_complex_double* x, lapack_int ldx, - double* rcond, double* ferr, double* berr, - lapack_complex_double* work, double* rwork ); - -lapack_int LAPACKE_sgttrf_work( lapack_int n, float* dl, float* d, float* du, - float* du2, lapack_int* ipiv ); -lapack_int LAPACKE_dgttrf_work( lapack_int n, double* dl, double* d, double* du, - double* du2, lapack_int* ipiv ); -lapack_int LAPACKE_cgttrf_work( lapack_int n, lapack_complex_float* dl, - lapack_complex_float* d, - lapack_complex_float* du, - lapack_complex_float* du2, lapack_int* ipiv ); -lapack_int LAPACKE_zgttrf_work( lapack_int n, lapack_complex_double* dl, - lapack_complex_double* d, - lapack_complex_double* du, - lapack_complex_double* du2, lapack_int* ipiv ); - -lapack_int LAPACKE_sgttrs_work( int matrix_order, char trans, lapack_int n, - lapack_int nrhs, const float* dl, - const float* d, const float* du, - const float* du2, const lapack_int* ipiv, - float* b, lapack_int ldb ); -lapack_int LAPACKE_dgttrs_work( int matrix_order, char trans, lapack_int n, - lapack_int nrhs, const double* dl, - const double* d, const double* du, - const double* du2, const lapack_int* ipiv, - double* b, lapack_int ldb ); -lapack_int LAPACKE_cgttrs_work( int matrix_order, char trans, lapack_int n, - lapack_int nrhs, const lapack_complex_float* dl, - const lapack_complex_float* d, - const lapack_complex_float* du, - const lapack_complex_float* du2, - const lapack_int* ipiv, lapack_complex_float* b, - lapack_int ldb ); -lapack_int LAPACKE_zgttrs_work( int matrix_order, char trans, lapack_int n, - lapack_int nrhs, - const lapack_complex_double* dl, - const lapack_complex_double* d, - const lapack_complex_double* du, - const lapack_complex_double* du2, - const lapack_int* ipiv, - lapack_complex_double* b, lapack_int ldb ); - -lapack_int LAPACKE_chbev_work( int matrix_order, char jobz, char uplo, - lapack_int n, lapack_int kd, - lapack_complex_float* ab, lapack_int ldab, - float* w, lapack_complex_float* z, - lapack_int ldz, lapack_complex_float* work, - float* rwork ); -lapack_int LAPACKE_zhbev_work( int matrix_order, char jobz, char uplo, - lapack_int n, lapack_int kd, - lapack_complex_double* ab, lapack_int ldab, - double* w, lapack_complex_double* z, - lapack_int ldz, lapack_complex_double* work, - double* rwork ); - -lapack_int LAPACKE_chbevd_work( int matrix_order, char jobz, char uplo, - lapack_int n, lapack_int kd, - lapack_complex_float* ab, lapack_int ldab, - float* w, lapack_complex_float* z, - lapack_int ldz, lapack_complex_float* work, - lapack_int lwork, float* rwork, - lapack_int lrwork, lapack_int* iwork, - lapack_int liwork ); -lapack_int LAPACKE_zhbevd_work( int matrix_order, char jobz, char uplo, - lapack_int n, lapack_int kd, - lapack_complex_double* ab, lapack_int ldab, - double* w, lapack_complex_double* z, - lapack_int ldz, lapack_complex_double* work, - lapack_int lwork, double* rwork, - lapack_int lrwork, lapack_int* iwork, - lapack_int liwork ); - -lapack_int LAPACKE_chbevx_work( int matrix_order, char jobz, char range, - char uplo, lapack_int n, lapack_int kd, - lapack_complex_float* ab, lapack_int ldab, - lapack_complex_float* q, lapack_int ldq, - float vl, float vu, lapack_int il, - lapack_int iu, float abstol, lapack_int* m, - float* w, lapack_complex_float* z, - lapack_int ldz, lapack_complex_float* work, - float* rwork, lapack_int* iwork, - lapack_int* ifail ); -lapack_int LAPACKE_zhbevx_work( int matrix_order, char jobz, char range, - char uplo, lapack_int n, lapack_int kd, - lapack_complex_double* ab, lapack_int ldab, - lapack_complex_double* q, lapack_int ldq, - double vl, double vu, lapack_int il, - lapack_int iu, double abstol, lapack_int* m, - double* w, lapack_complex_double* z, - lapack_int ldz, lapack_complex_double* work, - double* rwork, lapack_int* iwork, - lapack_int* ifail ); - -lapack_int LAPACKE_chbgst_work( int matrix_order, char vect, char uplo, - lapack_int n, lapack_int ka, lapack_int kb, - lapack_complex_float* ab, lapack_int ldab, - const lapack_complex_float* bb, lapack_int ldbb, - lapack_complex_float* x, lapack_int ldx, - lapack_complex_float* work, float* rwork ); -lapack_int LAPACKE_zhbgst_work( int matrix_order, char vect, char uplo, - lapack_int n, lapack_int ka, lapack_int kb, - lapack_complex_double* ab, lapack_int ldab, - const lapack_complex_double* bb, - lapack_int ldbb, lapack_complex_double* x, - lapack_int ldx, lapack_complex_double* work, - double* rwork ); - -lapack_int LAPACKE_chbgv_work( int matrix_order, char jobz, char uplo, - lapack_int n, lapack_int ka, lapack_int kb, - lapack_complex_float* ab, lapack_int ldab, - lapack_complex_float* bb, lapack_int ldbb, - float* w, lapack_complex_float* z, - lapack_int ldz, lapack_complex_float* work, - float* rwork ); -lapack_int LAPACKE_zhbgv_work( int matrix_order, char jobz, char uplo, - lapack_int n, lapack_int ka, lapack_int kb, - lapack_complex_double* ab, lapack_int ldab, - lapack_complex_double* bb, lapack_int ldbb, - double* w, lapack_complex_double* z, - lapack_int ldz, lapack_complex_double* work, - double* rwork ); - -lapack_int LAPACKE_chbgvd_work( int matrix_order, char jobz, char uplo, - lapack_int n, lapack_int ka, lapack_int kb, - lapack_complex_float* ab, lapack_int ldab, - lapack_complex_float* bb, lapack_int ldbb, - float* w, lapack_complex_float* z, - lapack_int ldz, lapack_complex_float* work, - lapack_int lwork, float* rwork, - lapack_int lrwork, lapack_int* iwork, - lapack_int liwork ); -lapack_int LAPACKE_zhbgvd_work( int matrix_order, char jobz, char uplo, - lapack_int n, lapack_int ka, lapack_int kb, - lapack_complex_double* ab, lapack_int ldab, - lapack_complex_double* bb, lapack_int ldbb, - double* w, lapack_complex_double* z, - lapack_int ldz, lapack_complex_double* work, - lapack_int lwork, double* rwork, - lapack_int lrwork, lapack_int* iwork, - lapack_int liwork ); - -lapack_int LAPACKE_chbgvx_work( int matrix_order, char jobz, char range, - char uplo, lapack_int n, lapack_int ka, - lapack_int kb, lapack_complex_float* ab, - lapack_int ldab, lapack_complex_float* bb, - lapack_int ldbb, lapack_complex_float* q, - lapack_int ldq, float vl, float vu, - lapack_int il, lapack_int iu, float abstol, - lapack_int* m, float* w, - lapack_complex_float* z, lapack_int ldz, - lapack_complex_float* work, float* rwork, - lapack_int* iwork, lapack_int* ifail ); -lapack_int LAPACKE_zhbgvx_work( int matrix_order, char jobz, char range, - char uplo, lapack_int n, lapack_int ka, - lapack_int kb, lapack_complex_double* ab, - lapack_int ldab, lapack_complex_double* bb, - lapack_int ldbb, lapack_complex_double* q, - lapack_int ldq, double vl, double vu, - lapack_int il, lapack_int iu, double abstol, - lapack_int* m, double* w, - lapack_complex_double* z, lapack_int ldz, - lapack_complex_double* work, double* rwork, - lapack_int* iwork, lapack_int* ifail ); - -lapack_int LAPACKE_chbtrd_work( int matrix_order, char vect, char uplo, - lapack_int n, lapack_int kd, - lapack_complex_float* ab, lapack_int ldab, - float* d, float* e, lapack_complex_float* q, - lapack_int ldq, lapack_complex_float* work ); -lapack_int LAPACKE_zhbtrd_work( int matrix_order, char vect, char uplo, - lapack_int n, lapack_int kd, - lapack_complex_double* ab, lapack_int ldab, - double* d, double* e, lapack_complex_double* q, - lapack_int ldq, lapack_complex_double* work ); - -lapack_int LAPACKE_checon_work( int matrix_order, char uplo, lapack_int n, - const lapack_complex_float* a, lapack_int lda, - const lapack_int* ipiv, float anorm, - float* rcond, lapack_complex_float* work ); -lapack_int LAPACKE_zhecon_work( int matrix_order, char uplo, lapack_int n, - const lapack_complex_double* a, lapack_int lda, - const lapack_int* ipiv, double anorm, - double* rcond, lapack_complex_double* work ); - -lapack_int LAPACKE_cheequb_work( int matrix_order, char uplo, lapack_int n, - const lapack_complex_float* a, lapack_int lda, - float* s, float* scond, float* amax, - lapack_complex_float* work ); -lapack_int LAPACKE_zheequb_work( int matrix_order, char uplo, lapack_int n, - const lapack_complex_double* a, lapack_int lda, - double* s, double* scond, double* amax, - lapack_complex_double* work ); - -lapack_int LAPACKE_cheev_work( int matrix_order, char jobz, char uplo, - lapack_int n, lapack_complex_float* a, - lapack_int lda, float* w, - lapack_complex_float* work, lapack_int lwork, - float* rwork ); -lapack_int LAPACKE_zheev_work( int matrix_order, char jobz, char uplo, - lapack_int n, lapack_complex_double* a, - lapack_int lda, double* w, - lapack_complex_double* work, lapack_int lwork, - double* rwork ); - -lapack_int LAPACKE_cheevd_work( int matrix_order, char jobz, char uplo, - lapack_int n, lapack_complex_float* a, - lapack_int lda, float* w, - lapack_complex_float* work, lapack_int lwork, - float* rwork, lapack_int lrwork, - lapack_int* iwork, lapack_int liwork ); -lapack_int LAPACKE_zheevd_work( int matrix_order, char jobz, char uplo, - lapack_int n, lapack_complex_double* a, - lapack_int lda, double* w, - lapack_complex_double* work, lapack_int lwork, - double* rwork, lapack_int lrwork, - lapack_int* iwork, lapack_int liwork ); - -lapack_int LAPACKE_cheevr_work( int matrix_order, char jobz, char range, - char uplo, lapack_int n, - lapack_complex_float* a, lapack_int lda, - float vl, float vu, lapack_int il, - lapack_int iu, float abstol, lapack_int* m, - float* w, lapack_complex_float* z, - lapack_int ldz, lapack_int* isuppz, - lapack_complex_float* work, lapack_int lwork, - float* rwork, lapack_int lrwork, - lapack_int* iwork, lapack_int liwork ); -lapack_int LAPACKE_zheevr_work( int matrix_order, char jobz, char range, - char uplo, lapack_int n, - lapack_complex_double* a, lapack_int lda, - double vl, double vu, lapack_int il, - lapack_int iu, double abstol, lapack_int* m, - double* w, lapack_complex_double* z, - lapack_int ldz, lapack_int* isuppz, - lapack_complex_double* work, lapack_int lwork, - double* rwork, lapack_int lrwork, - lapack_int* iwork, lapack_int liwork ); - -lapack_int LAPACKE_cheevx_work( int matrix_order, char jobz, char range, - char uplo, lapack_int n, - lapack_complex_float* a, lapack_int lda, - float vl, float vu, lapack_int il, - lapack_int iu, float abstol, lapack_int* m, - float* w, lapack_complex_float* z, - lapack_int ldz, lapack_complex_float* work, - lapack_int lwork, float* rwork, - lapack_int* iwork, lapack_int* ifail ); -lapack_int LAPACKE_zheevx_work( int matrix_order, char jobz, char range, - char uplo, lapack_int n, - lapack_complex_double* a, lapack_int lda, - double vl, double vu, lapack_int il, - lapack_int iu, double abstol, lapack_int* m, - double* w, lapack_complex_double* z, - lapack_int ldz, lapack_complex_double* work, - lapack_int lwork, double* rwork, - lapack_int* iwork, lapack_int* ifail ); - -lapack_int LAPACKE_chegst_work( int matrix_order, lapack_int itype, char uplo, - lapack_int n, lapack_complex_float* a, - lapack_int lda, const lapack_complex_float* b, - lapack_int ldb ); -lapack_int LAPACKE_zhegst_work( int matrix_order, lapack_int itype, char uplo, - lapack_int n, lapack_complex_double* a, - lapack_int lda, const lapack_complex_double* b, - lapack_int ldb ); - -lapack_int LAPACKE_chegv_work( int matrix_order, lapack_int itype, char jobz, - char uplo, lapack_int n, lapack_complex_float* a, - lapack_int lda, lapack_complex_float* b, - lapack_int ldb, float* w, - lapack_complex_float* work, lapack_int lwork, - float* rwork ); -lapack_int LAPACKE_zhegv_work( int matrix_order, lapack_int itype, char jobz, - char uplo, lapack_int n, - lapack_complex_double* a, lapack_int lda, - lapack_complex_double* b, lapack_int ldb, - double* w, lapack_complex_double* work, - lapack_int lwork, double* rwork ); - -lapack_int LAPACKE_chegvd_work( int matrix_order, lapack_int itype, char jobz, - char uplo, lapack_int n, - lapack_complex_float* a, lapack_int lda, - lapack_complex_float* b, lapack_int ldb, - float* w, lapack_complex_float* work, - lapack_int lwork, float* rwork, - lapack_int lrwork, lapack_int* iwork, - lapack_int liwork ); -lapack_int LAPACKE_zhegvd_work( int matrix_order, lapack_int itype, char jobz, - char uplo, lapack_int n, - lapack_complex_double* a, lapack_int lda, - lapack_complex_double* b, lapack_int ldb, - double* w, lapack_complex_double* work, - lapack_int lwork, double* rwork, - lapack_int lrwork, lapack_int* iwork, - lapack_int liwork ); - -lapack_int LAPACKE_chegvx_work( int matrix_order, lapack_int itype, char jobz, - char range, char uplo, lapack_int n, - lapack_complex_float* a, lapack_int lda, - lapack_complex_float* b, lapack_int ldb, - float vl, float vu, lapack_int il, - lapack_int iu, float abstol, lapack_int* m, - float* w, lapack_complex_float* z, - lapack_int ldz, lapack_complex_float* work, - lapack_int lwork, float* rwork, - lapack_int* iwork, lapack_int* ifail ); -lapack_int LAPACKE_zhegvx_work( int matrix_order, lapack_int itype, char jobz, - char range, char uplo, lapack_int n, - lapack_complex_double* a, lapack_int lda, - lapack_complex_double* b, lapack_int ldb, - double vl, double vu, lapack_int il, - lapack_int iu, double abstol, lapack_int* m, - double* w, lapack_complex_double* z, - lapack_int ldz, lapack_complex_double* work, - lapack_int lwork, double* rwork, - lapack_int* iwork, lapack_int* ifail ); - -lapack_int LAPACKE_cherfs_work( int matrix_order, char uplo, lapack_int n, - lapack_int nrhs, const lapack_complex_float* a, - lapack_int lda, const lapack_complex_float* af, - lapack_int ldaf, const lapack_int* ipiv, - const lapack_complex_float* b, lapack_int ldb, - lapack_complex_float* x, lapack_int ldx, - float* ferr, float* berr, - lapack_complex_float* work, float* rwork ); -lapack_int LAPACKE_zherfs_work( int matrix_order, char uplo, lapack_int n, - lapack_int nrhs, const lapack_complex_double* a, - lapack_int lda, const lapack_complex_double* af, - lapack_int ldaf, const lapack_int* ipiv, - const lapack_complex_double* b, lapack_int ldb, - lapack_complex_double* x, lapack_int ldx, - double* ferr, double* berr, - lapack_complex_double* work, double* rwork ); - -lapack_int LAPACKE_cherfsx_work( int matrix_order, char uplo, char equed, - lapack_int n, lapack_int nrhs, - const lapack_complex_float* a, lapack_int lda, - const lapack_complex_float* af, - lapack_int ldaf, const lapack_int* ipiv, - const float* s, const lapack_complex_float* b, - lapack_int ldb, lapack_complex_float* x, - lapack_int ldx, float* rcond, float* berr, - lapack_int n_err_bnds, float* err_bnds_norm, - float* err_bnds_comp, lapack_int nparams, - float* params, lapack_complex_float* work, - float* rwork ); -lapack_int LAPACKE_zherfsx_work( int matrix_order, char uplo, char equed, - lapack_int n, lapack_int nrhs, - const lapack_complex_double* a, lapack_int lda, - const lapack_complex_double* af, - lapack_int ldaf, const lapack_int* ipiv, - const double* s, - const lapack_complex_double* b, lapack_int ldb, - lapack_complex_double* x, lapack_int ldx, - double* rcond, double* berr, - lapack_int n_err_bnds, double* err_bnds_norm, - double* err_bnds_comp, lapack_int nparams, - double* params, lapack_complex_double* work, - double* rwork ); - -lapack_int LAPACKE_chesv_work( int matrix_order, char uplo, lapack_int n, - lapack_int nrhs, lapack_complex_float* a, - lapack_int lda, lapack_int* ipiv, - lapack_complex_float* b, lapack_int ldb, - lapack_complex_float* work, lapack_int lwork ); -lapack_int LAPACKE_zhesv_work( int matrix_order, char uplo, lapack_int n, - lapack_int nrhs, lapack_complex_double* a, - lapack_int lda, lapack_int* ipiv, - lapack_complex_double* b, lapack_int ldb, - lapack_complex_double* work, lapack_int lwork ); - -lapack_int LAPACKE_chesvx_work( int matrix_order, char fact, char uplo, - lapack_int n, lapack_int nrhs, - const lapack_complex_float* a, lapack_int lda, - lapack_complex_float* af, lapack_int ldaf, - lapack_int* ipiv, const lapack_complex_float* b, - lapack_int ldb, lapack_complex_float* x, - lapack_int ldx, float* rcond, float* ferr, - float* berr, lapack_complex_float* work, - lapack_int lwork, float* rwork ); -lapack_int LAPACKE_zhesvx_work( int matrix_order, char fact, char uplo, - lapack_int n, lapack_int nrhs, - const lapack_complex_double* a, lapack_int lda, - lapack_complex_double* af, lapack_int ldaf, - lapack_int* ipiv, - const lapack_complex_double* b, lapack_int ldb, - lapack_complex_double* x, lapack_int ldx, - double* rcond, double* ferr, double* berr, - lapack_complex_double* work, lapack_int lwork, - double* rwork ); - -lapack_int LAPACKE_chesvxx_work( int matrix_order, char fact, char uplo, - lapack_int n, lapack_int nrhs, - lapack_complex_float* a, lapack_int lda, - lapack_complex_float* af, lapack_int ldaf, - lapack_int* ipiv, char* equed, float* s, - lapack_complex_float* b, lapack_int ldb, - lapack_complex_float* x, lapack_int ldx, - float* rcond, float* rpvgrw, float* berr, - lapack_int n_err_bnds, float* err_bnds_norm, - float* err_bnds_comp, lapack_int nparams, - float* params, lapack_complex_float* work, - float* rwork ); -lapack_int LAPACKE_zhesvxx_work( int matrix_order, char fact, char uplo, - lapack_int n, lapack_int nrhs, - lapack_complex_double* a, lapack_int lda, - lapack_complex_double* af, lapack_int ldaf, - lapack_int* ipiv, char* equed, double* s, - lapack_complex_double* b, lapack_int ldb, - lapack_complex_double* x, lapack_int ldx, - double* rcond, double* rpvgrw, double* berr, - lapack_int n_err_bnds, double* err_bnds_norm, - double* err_bnds_comp, lapack_int nparams, - double* params, lapack_complex_double* work, - double* rwork ); - -lapack_int LAPACKE_chetrd_work( int matrix_order, char uplo, lapack_int n, - lapack_complex_float* a, lapack_int lda, - float* d, float* e, lapack_complex_float* tau, - lapack_complex_float* work, lapack_int lwork ); -lapack_int LAPACKE_zhetrd_work( int matrix_order, char uplo, lapack_int n, - lapack_complex_double* a, lapack_int lda, - double* d, double* e, - lapack_complex_double* tau, - lapack_complex_double* work, lapack_int lwork ); - -lapack_int LAPACKE_chetrf_work( int matrix_order, char uplo, lapack_int n, - lapack_complex_float* a, lapack_int lda, - lapack_int* ipiv, lapack_complex_float* work, - lapack_int lwork ); -lapack_int LAPACKE_zhetrf_work( int matrix_order, char uplo, lapack_int n, - lapack_complex_double* a, lapack_int lda, - lapack_int* ipiv, lapack_complex_double* work, - lapack_int lwork ); - -lapack_int LAPACKE_chetri_work( int matrix_order, char uplo, lapack_int n, - lapack_complex_float* a, lapack_int lda, - const lapack_int* ipiv, - lapack_complex_float* work ); -lapack_int LAPACKE_zhetri_work( int matrix_order, char uplo, lapack_int n, - lapack_complex_double* a, lapack_int lda, - const lapack_int* ipiv, - lapack_complex_double* work ); - -lapack_int LAPACKE_chetrs_work( int matrix_order, char uplo, lapack_int n, - lapack_int nrhs, const lapack_complex_float* a, - lapack_int lda, const lapack_int* ipiv, - lapack_complex_float* b, lapack_int ldb ); -lapack_int LAPACKE_zhetrs_work( int matrix_order, char uplo, lapack_int n, - lapack_int nrhs, const lapack_complex_double* a, - lapack_int lda, const lapack_int* ipiv, - lapack_complex_double* b, lapack_int ldb ); - -lapack_int LAPACKE_chfrk_work( int matrix_order, char transr, char uplo, - char trans, lapack_int n, lapack_int k, - float alpha, const lapack_complex_float* a, - lapack_int lda, float beta, - lapack_complex_float* c ); -lapack_int LAPACKE_zhfrk_work( int matrix_order, char transr, char uplo, - char trans, lapack_int n, lapack_int k, - double alpha, const lapack_complex_double* a, - lapack_int lda, double beta, - lapack_complex_double* c ); - -lapack_int LAPACKE_shgeqz_work( int matrix_order, char job, char compq, - char compz, lapack_int n, lapack_int ilo, - lapack_int ihi, float* h, lapack_int ldh, - float* t, lapack_int ldt, float* alphar, - float* alphai, float* beta, float* q, - lapack_int ldq, float* z, lapack_int ldz, - float* work, lapack_int lwork ); -lapack_int LAPACKE_dhgeqz_work( int matrix_order, char job, char compq, - char compz, lapack_int n, lapack_int ilo, - lapack_int ihi, double* h, lapack_int ldh, - double* t, lapack_int ldt, double* alphar, - double* alphai, double* beta, double* q, - lapack_int ldq, double* z, lapack_int ldz, - double* work, lapack_int lwork ); -lapack_int LAPACKE_chgeqz_work( int matrix_order, char job, char compq, - char compz, lapack_int n, lapack_int ilo, - lapack_int ihi, lapack_complex_float* h, - lapack_int ldh, lapack_complex_float* t, - lapack_int ldt, lapack_complex_float* alpha, - lapack_complex_float* beta, - lapack_complex_float* q, lapack_int ldq, - lapack_complex_float* z, lapack_int ldz, - lapack_complex_float* work, lapack_int lwork, - float* rwork ); -lapack_int LAPACKE_zhgeqz_work( int matrix_order, char job, char compq, - char compz, lapack_int n, lapack_int ilo, - lapack_int ihi, lapack_complex_double* h, - lapack_int ldh, lapack_complex_double* t, - lapack_int ldt, lapack_complex_double* alpha, - lapack_complex_double* beta, - lapack_complex_double* q, lapack_int ldq, - lapack_complex_double* z, lapack_int ldz, - lapack_complex_double* work, lapack_int lwork, - double* rwork ); - -lapack_int LAPACKE_chpcon_work( int matrix_order, char uplo, lapack_int n, - const lapack_complex_float* ap, - const lapack_int* ipiv, float anorm, - float* rcond, lapack_complex_float* work ); -lapack_int LAPACKE_zhpcon_work( int matrix_order, char uplo, lapack_int n, - const lapack_complex_double* ap, - const lapack_int* ipiv, double anorm, - double* rcond, lapack_complex_double* work ); - -lapack_int LAPACKE_chpev_work( int matrix_order, char jobz, char uplo, - lapack_int n, lapack_complex_float* ap, float* w, - lapack_complex_float* z, lapack_int ldz, - lapack_complex_float* work, float* rwork ); -lapack_int LAPACKE_zhpev_work( int matrix_order, char jobz, char uplo, - lapack_int n, lapack_complex_double* ap, - double* w, lapack_complex_double* z, - lapack_int ldz, lapack_complex_double* work, - double* rwork ); - -lapack_int LAPACKE_chpevd_work( int matrix_order, char jobz, char uplo, - lapack_int n, lapack_complex_float* ap, - float* w, lapack_complex_float* z, - lapack_int ldz, lapack_complex_float* work, - lapack_int lwork, float* rwork, - lapack_int lrwork, lapack_int* iwork, - lapack_int liwork ); -lapack_int LAPACKE_zhpevd_work( int matrix_order, char jobz, char uplo, - lapack_int n, lapack_complex_double* ap, - double* w, lapack_complex_double* z, - lapack_int ldz, lapack_complex_double* work, - lapack_int lwork, double* rwork, - lapack_int lrwork, lapack_int* iwork, - lapack_int liwork ); - -lapack_int LAPACKE_chpevx_work( int matrix_order, char jobz, char range, - char uplo, lapack_int n, - lapack_complex_float* ap, float vl, float vu, - lapack_int il, lapack_int iu, float abstol, - lapack_int* m, float* w, - lapack_complex_float* z, lapack_int ldz, - lapack_complex_float* work, float* rwork, - lapack_int* iwork, lapack_int* ifail ); -lapack_int LAPACKE_zhpevx_work( int matrix_order, char jobz, char range, - char uplo, lapack_int n, - lapack_complex_double* ap, double vl, double vu, - lapack_int il, lapack_int iu, double abstol, - lapack_int* m, double* w, - lapack_complex_double* z, lapack_int ldz, - lapack_complex_double* work, double* rwork, - lapack_int* iwork, lapack_int* ifail ); - -lapack_int LAPACKE_chpgst_work( int matrix_order, lapack_int itype, char uplo, - lapack_int n, lapack_complex_float* ap, - const lapack_complex_float* bp ); -lapack_int LAPACKE_zhpgst_work( int matrix_order, lapack_int itype, char uplo, - lapack_int n, lapack_complex_double* ap, - const lapack_complex_double* bp ); - -lapack_int LAPACKE_chpgv_work( int matrix_order, lapack_int itype, char jobz, - char uplo, lapack_int n, - lapack_complex_float* ap, - lapack_complex_float* bp, float* w, - lapack_complex_float* z, lapack_int ldz, - lapack_complex_float* work, float* rwork ); -lapack_int LAPACKE_zhpgv_work( int matrix_order, lapack_int itype, char jobz, - char uplo, lapack_int n, - lapack_complex_double* ap, - lapack_complex_double* bp, double* w, - lapack_complex_double* z, lapack_int ldz, - lapack_complex_double* work, double* rwork ); - -lapack_int LAPACKE_chpgvd_work( int matrix_order, lapack_int itype, char jobz, - char uplo, lapack_int n, - lapack_complex_float* ap, - lapack_complex_float* bp, float* w, - lapack_complex_float* z, lapack_int ldz, - lapack_complex_float* work, lapack_int lwork, - float* rwork, lapack_int lrwork, - lapack_int* iwork, lapack_int liwork ); -lapack_int LAPACKE_zhpgvd_work( int matrix_order, lapack_int itype, char jobz, - char uplo, lapack_int n, - lapack_complex_double* ap, - lapack_complex_double* bp, double* w, - lapack_complex_double* z, lapack_int ldz, - lapack_complex_double* work, lapack_int lwork, - double* rwork, lapack_int lrwork, - lapack_int* iwork, lapack_int liwork ); - -lapack_int LAPACKE_chpgvx_work( int matrix_order, lapack_int itype, char jobz, - char range, char uplo, lapack_int n, - lapack_complex_float* ap, - lapack_complex_float* bp, float vl, float vu, - lapack_int il, lapack_int iu, float abstol, - lapack_int* m, float* w, - lapack_complex_float* z, lapack_int ldz, - lapack_complex_float* work, float* rwork, - lapack_int* iwork, lapack_int* ifail ); -lapack_int LAPACKE_zhpgvx_work( int matrix_order, lapack_int itype, char jobz, - char range, char uplo, lapack_int n, - lapack_complex_double* ap, - lapack_complex_double* bp, double vl, double vu, - lapack_int il, lapack_int iu, double abstol, - lapack_int* m, double* w, - lapack_complex_double* z, lapack_int ldz, - lapack_complex_double* work, double* rwork, - lapack_int* iwork, lapack_int* ifail ); - -lapack_int LAPACKE_chprfs_work( int matrix_order, char uplo, lapack_int n, - lapack_int nrhs, const lapack_complex_float* ap, - const lapack_complex_float* afp, - const lapack_int* ipiv, - const lapack_complex_float* b, lapack_int ldb, - lapack_complex_float* x, lapack_int ldx, - float* ferr, float* berr, - lapack_complex_float* work, float* rwork ); -lapack_int LAPACKE_zhprfs_work( int matrix_order, char uplo, lapack_int n, - lapack_int nrhs, - const lapack_complex_double* ap, - const lapack_complex_double* afp, - const lapack_int* ipiv, - const lapack_complex_double* b, lapack_int ldb, - lapack_complex_double* x, lapack_int ldx, - double* ferr, double* berr, - lapack_complex_double* work, double* rwork ); - -lapack_int LAPACKE_chpsv_work( int matrix_order, char uplo, lapack_int n, - lapack_int nrhs, lapack_complex_float* ap, - lapack_int* ipiv, lapack_complex_float* b, - lapack_int ldb ); -lapack_int LAPACKE_zhpsv_work( int matrix_order, char uplo, lapack_int n, - lapack_int nrhs, lapack_complex_double* ap, - lapack_int* ipiv, lapack_complex_double* b, - lapack_int ldb ); - -lapack_int LAPACKE_chpsvx_work( int matrix_order, char fact, char uplo, - lapack_int n, lapack_int nrhs, - const lapack_complex_float* ap, - lapack_complex_float* afp, lapack_int* ipiv, - const lapack_complex_float* b, lapack_int ldb, - lapack_complex_float* x, lapack_int ldx, - float* rcond, float* ferr, float* berr, - lapack_complex_float* work, float* rwork ); -lapack_int LAPACKE_zhpsvx_work( int matrix_order, char fact, char uplo, - lapack_int n, lapack_int nrhs, - const lapack_complex_double* ap, - lapack_complex_double* afp, lapack_int* ipiv, - const lapack_complex_double* b, lapack_int ldb, - lapack_complex_double* x, lapack_int ldx, - double* rcond, double* ferr, double* berr, - lapack_complex_double* work, double* rwork ); - -lapack_int LAPACKE_chptrd_work( int matrix_order, char uplo, lapack_int n, - lapack_complex_float* ap, float* d, float* e, - lapack_complex_float* tau ); -lapack_int LAPACKE_zhptrd_work( int matrix_order, char uplo, lapack_int n, - lapack_complex_double* ap, double* d, double* e, - lapack_complex_double* tau ); - -lapack_int LAPACKE_chptrf_work( int matrix_order, char uplo, lapack_int n, - lapack_complex_float* ap, lapack_int* ipiv ); -lapack_int LAPACKE_zhptrf_work( int matrix_order, char uplo, lapack_int n, - lapack_complex_double* ap, lapack_int* ipiv ); - -lapack_int LAPACKE_chptri_work( int matrix_order, char uplo, lapack_int n, - lapack_complex_float* ap, - const lapack_int* ipiv, - lapack_complex_float* work ); -lapack_int LAPACKE_zhptri_work( int matrix_order, char uplo, lapack_int n, - lapack_complex_double* ap, - const lapack_int* ipiv, - lapack_complex_double* work ); - -lapack_int LAPACKE_chptrs_work( int matrix_order, char uplo, lapack_int n, - lapack_int nrhs, const lapack_complex_float* ap, - const lapack_int* ipiv, lapack_complex_float* b, - lapack_int ldb ); -lapack_int LAPACKE_zhptrs_work( int matrix_order, char uplo, lapack_int n, - lapack_int nrhs, - const lapack_complex_double* ap, - const lapack_int* ipiv, - lapack_complex_double* b, lapack_int ldb ); - -lapack_int LAPACKE_shsein_work( int matrix_order, char job, char eigsrc, - char initv, lapack_logical* select, - lapack_int n, const float* h, lapack_int ldh, - float* wr, const float* wi, float* vl, - lapack_int ldvl, float* vr, lapack_int ldvr, - lapack_int mm, lapack_int* m, float* work, - lapack_int* ifaill, lapack_int* ifailr ); -lapack_int LAPACKE_dhsein_work( int matrix_order, char job, char eigsrc, - char initv, lapack_logical* select, - lapack_int n, const double* h, lapack_int ldh, - double* wr, const double* wi, double* vl, - lapack_int ldvl, double* vr, lapack_int ldvr, - lapack_int mm, lapack_int* m, double* work, - lapack_int* ifaill, lapack_int* ifailr ); -lapack_int LAPACKE_chsein_work( int matrix_order, char job, char eigsrc, - char initv, const lapack_logical* select, - lapack_int n, const lapack_complex_float* h, - lapack_int ldh, lapack_complex_float* w, - lapack_complex_float* vl, lapack_int ldvl, - lapack_complex_float* vr, lapack_int ldvr, - lapack_int mm, lapack_int* m, - lapack_complex_float* work, float* rwork, - lapack_int* ifaill, lapack_int* ifailr ); -lapack_int LAPACKE_zhsein_work( int matrix_order, char job, char eigsrc, - char initv, const lapack_logical* select, - lapack_int n, const lapack_complex_double* h, - lapack_int ldh, lapack_complex_double* w, - lapack_complex_double* vl, lapack_int ldvl, - lapack_complex_double* vr, lapack_int ldvr, - lapack_int mm, lapack_int* m, - lapack_complex_double* work, double* rwork, - lapack_int* ifaill, lapack_int* ifailr ); - -lapack_int LAPACKE_shseqr_work( int matrix_order, char job, char compz, - lapack_int n, lapack_int ilo, lapack_int ihi, - float* h, lapack_int ldh, float* wr, float* wi, - float* z, lapack_int ldz, float* work, - lapack_int lwork ); -lapack_int LAPACKE_dhseqr_work( int matrix_order, char job, char compz, - lapack_int n, lapack_int ilo, lapack_int ihi, - double* h, lapack_int ldh, double* wr, - double* wi, double* z, lapack_int ldz, - double* work, lapack_int lwork ); -lapack_int LAPACKE_chseqr_work( int matrix_order, char job, char compz, - lapack_int n, lapack_int ilo, lapack_int ihi, - lapack_complex_float* h, lapack_int ldh, - lapack_complex_float* w, - lapack_complex_float* z, lapack_int ldz, - lapack_complex_float* work, lapack_int lwork ); -lapack_int LAPACKE_zhseqr_work( int matrix_order, char job, char compz, - lapack_int n, lapack_int ilo, lapack_int ihi, - lapack_complex_double* h, lapack_int ldh, - lapack_complex_double* w, - lapack_complex_double* z, lapack_int ldz, - lapack_complex_double* work, lapack_int lwork ); - -lapack_int LAPACKE_clacgv_work( lapack_int n, lapack_complex_float* x, - lapack_int incx ); -lapack_int LAPACKE_zlacgv_work( lapack_int n, lapack_complex_double* x, - lapack_int incx ); - -lapack_int LAPACKE_slacpy_work( int matrix_order, char uplo, lapack_int m, - lapack_int n, const float* a, lapack_int lda, - float* b, lapack_int ldb ); -lapack_int LAPACKE_dlacpy_work( int matrix_order, char uplo, lapack_int m, - lapack_int n, const double* a, lapack_int lda, - double* b, lapack_int ldb ); -lapack_int LAPACKE_clacpy_work( int matrix_order, char uplo, lapack_int m, - lapack_int n, const lapack_complex_float* a, - lapack_int lda, lapack_complex_float* b, - lapack_int ldb ); -lapack_int LAPACKE_zlacpy_work( int matrix_order, char uplo, lapack_int m, - lapack_int n, const lapack_complex_double* a, - lapack_int lda, lapack_complex_double* b, - lapack_int ldb ); - -lapack_int LAPACKE_zlag2c_work( int matrix_order, lapack_int m, lapack_int n, - const lapack_complex_double* a, lapack_int lda, - lapack_complex_float* sa, lapack_int ldsa ); - -lapack_int LAPACKE_slag2d_work( int matrix_order, lapack_int m, lapack_int n, - const float* sa, lapack_int ldsa, double* a, - lapack_int lda ); - -lapack_int LAPACKE_dlag2s_work( int matrix_order, lapack_int m, lapack_int n, - const double* a, lapack_int lda, float* sa, - lapack_int ldsa ); - -lapack_int LAPACKE_clag2z_work( int matrix_order, lapack_int m, lapack_int n, - const lapack_complex_float* sa, lapack_int ldsa, - lapack_complex_double* a, lapack_int lda ); - -lapack_int LAPACKE_slagge_work( int matrix_order, lapack_int m, lapack_int n, - lapack_int kl, lapack_int ku, const float* d, - float* a, lapack_int lda, lapack_int* iseed, - float* work ); -lapack_int LAPACKE_dlagge_work( int matrix_order, lapack_int m, lapack_int n, - lapack_int kl, lapack_int ku, const double* d, - double* a, lapack_int lda, lapack_int* iseed, - double* work ); -lapack_int LAPACKE_clagge_work( int matrix_order, lapack_int m, lapack_int n, - lapack_int kl, lapack_int ku, const float* d, - lapack_complex_float* a, lapack_int lda, - lapack_int* iseed, lapack_complex_float* work ); -lapack_int LAPACKE_zlagge_work( int matrix_order, lapack_int m, lapack_int n, - lapack_int kl, lapack_int ku, const double* d, - lapack_complex_double* a, lapack_int lda, - lapack_int* iseed, - lapack_complex_double* work ); - -lapack_int LAPACKE_claghe_work( int matrix_order, lapack_int n, lapack_int k, - const float* d, lapack_complex_float* a, - lapack_int lda, lapack_int* iseed, - lapack_complex_float* work ); -lapack_int LAPACKE_zlaghe_work( int matrix_order, lapack_int n, lapack_int k, - const double* d, lapack_complex_double* a, - lapack_int lda, lapack_int* iseed, - lapack_complex_double* work ); - -lapack_int LAPACKE_slagsy_work( int matrix_order, lapack_int n, lapack_int k, - const float* d, float* a, lapack_int lda, - lapack_int* iseed, float* work ); -lapack_int LAPACKE_dlagsy_work( int matrix_order, lapack_int n, lapack_int k, - const double* d, double* a, lapack_int lda, - lapack_int* iseed, double* work ); -lapack_int LAPACKE_clagsy_work( int matrix_order, lapack_int n, lapack_int k, - const float* d, lapack_complex_float* a, - lapack_int lda, lapack_int* iseed, - lapack_complex_float* work ); -lapack_int LAPACKE_zlagsy_work( int matrix_order, lapack_int n, lapack_int k, - const double* d, lapack_complex_double* a, - lapack_int lda, lapack_int* iseed, - lapack_complex_double* work ); - -lapack_int LAPACKE_slapmr_work( int matrix_order, lapack_logical forwrd, - lapack_int m, lapack_int n, float* x, - lapack_int ldx, lapack_int* k ); -lapack_int LAPACKE_dlapmr_work( int matrix_order, lapack_logical forwrd, - lapack_int m, lapack_int n, double* x, - lapack_int ldx, lapack_int* k ); -lapack_int LAPACKE_clapmr_work( int matrix_order, lapack_logical forwrd, - lapack_int m, lapack_int n, - lapack_complex_float* x, lapack_int ldx, - lapack_int* k ); -lapack_int LAPACKE_zlapmr_work( int matrix_order, lapack_logical forwrd, - lapack_int m, lapack_int n, - lapack_complex_double* x, lapack_int ldx, - lapack_int* k ); - -lapack_int LAPACKE_slartgp_work( float f, float g, float* cs, float* sn, - float* r ); -lapack_int LAPACKE_dlartgp_work( double f, double g, double* cs, double* sn, - double* r ); - -lapack_int LAPACKE_slartgs_work( float x, float y, float sigma, float* cs, - float* sn ); -lapack_int LAPACKE_dlartgs_work( double x, double y, double sigma, double* cs, - double* sn ); - -float LAPACKE_slapy2_work( float x, float y ); -double LAPACKE_dlapy2_work( double x, double y ); - -float LAPACKE_slapy3_work( float x, float y, float z ); -double LAPACKE_dlapy3_work( double x, double y, double z ); - -float LAPACKE_slamch_work( char cmach ); -double LAPACKE_dlamch_work( char cmach ); - -float LAPACKE_slange_work( int matrix_order, char norm, lapack_int m, - lapack_int n, const float* a, lapack_int lda, - float* work ); -double LAPACKE_dlange_work( int matrix_order, char norm, lapack_int m, - lapack_int n, const double* a, lapack_int lda, - double* work ); -float LAPACKE_clange_work( int matrix_order, char norm, lapack_int m, - lapack_int n, const lapack_complex_float* a, - lapack_int lda, float* work ); -double LAPACKE_zlange_work( int matrix_order, char norm, lapack_int m, - lapack_int n, const lapack_complex_double* a, - lapack_int lda, double* work ); - -float LAPACKE_clanhe_work( int matrix_order, char norm, char uplo, - lapack_int n, const lapack_complex_float* a, - lapack_int lda, float* work ); -double LAPACKE_zlanhe_work( int matrix_order, char norm, char uplo, - lapack_int n, const lapack_complex_double* a, - lapack_int lda, double* work ); - -float LAPACKE_slansy_work( int matrix_order, char norm, char uplo, - lapack_int n, const float* a, lapack_int lda, - float* work ); -double LAPACKE_dlansy_work( int matrix_order, char norm, char uplo, - lapack_int n, const double* a, lapack_int lda, - double* work ); -float LAPACKE_clansy_work( int matrix_order, char norm, char uplo, - lapack_int n, const lapack_complex_float* a, - lapack_int lda, float* work ); -double LAPACKE_zlansy_work( int matrix_order, char norm, char uplo, - lapack_int n, const lapack_complex_double* a, - lapack_int lda, double* work ); - -float LAPACKE_slantr_work( int matrix_order, char norm, char uplo, - char diag, lapack_int m, lapack_int n, const float* a, - lapack_int lda, float* work ); -double LAPACKE_dlantr_work( int matrix_order, char norm, char uplo, - char diag, lapack_int m, lapack_int n, - const double* a, lapack_int lda, double* work ); -float LAPACKE_clantr_work( int matrix_order, char norm, char uplo, - char diag, lapack_int m, lapack_int n, - const lapack_complex_float* a, lapack_int lda, - float* work ); -double LAPACKE_zlantr_work( int matrix_order, char norm, char uplo, - char diag, lapack_int m, lapack_int n, - const lapack_complex_double* a, lapack_int lda, - double* work ); - -lapack_int LAPACKE_slarfb_work( int matrix_order, char side, char trans, - char direct, char storev, lapack_int m, - lapack_int n, lapack_int k, const float* v, - lapack_int ldv, const float* t, lapack_int ldt, - float* c, lapack_int ldc, float* work, - lapack_int ldwork ); -lapack_int LAPACKE_dlarfb_work( int matrix_order, char side, char trans, - char direct, char storev, lapack_int m, - lapack_int n, lapack_int k, const double* v, - lapack_int ldv, const double* t, lapack_int ldt, - double* c, lapack_int ldc, double* work, - lapack_int ldwork ); -lapack_int LAPACKE_clarfb_work( int matrix_order, char side, char trans, - char direct, char storev, lapack_int m, - lapack_int n, lapack_int k, - const lapack_complex_float* v, lapack_int ldv, - const lapack_complex_float* t, lapack_int ldt, - lapack_complex_float* c, lapack_int ldc, - lapack_complex_float* work, lapack_int ldwork ); -lapack_int LAPACKE_zlarfb_work( int matrix_order, char side, char trans, - char direct, char storev, lapack_int m, - lapack_int n, lapack_int k, - const lapack_complex_double* v, lapack_int ldv, - const lapack_complex_double* t, lapack_int ldt, - lapack_complex_double* c, lapack_int ldc, - lapack_complex_double* work, - lapack_int ldwork ); - -lapack_int LAPACKE_slarfg_work( lapack_int n, float* alpha, float* x, - lapack_int incx, float* tau ); -lapack_int LAPACKE_dlarfg_work( lapack_int n, double* alpha, double* x, - lapack_int incx, double* tau ); -lapack_int LAPACKE_clarfg_work( lapack_int n, lapack_complex_float* alpha, - lapack_complex_float* x, lapack_int incx, - lapack_complex_float* tau ); -lapack_int LAPACKE_zlarfg_work( lapack_int n, lapack_complex_double* alpha, - lapack_complex_double* x, lapack_int incx, - lapack_complex_double* tau ); - -lapack_int LAPACKE_slarft_work( int matrix_order, char direct, char storev, - lapack_int n, lapack_int k, const float* v, - lapack_int ldv, const float* tau, float* t, - lapack_int ldt ); -lapack_int LAPACKE_dlarft_work( int matrix_order, char direct, char storev, - lapack_int n, lapack_int k, const double* v, - lapack_int ldv, const double* tau, double* t, - lapack_int ldt ); -lapack_int LAPACKE_clarft_work( int matrix_order, char direct, char storev, - lapack_int n, lapack_int k, - const lapack_complex_float* v, lapack_int ldv, - const lapack_complex_float* tau, - lapack_complex_float* t, lapack_int ldt ); -lapack_int LAPACKE_zlarft_work( int matrix_order, char direct, char storev, - lapack_int n, lapack_int k, - const lapack_complex_double* v, lapack_int ldv, - const lapack_complex_double* tau, - lapack_complex_double* t, lapack_int ldt ); - -lapack_int LAPACKE_slarfx_work( int matrix_order, char side, lapack_int m, - lapack_int n, const float* v, float tau, - float* c, lapack_int ldc, float* work ); -lapack_int LAPACKE_dlarfx_work( int matrix_order, char side, lapack_int m, - lapack_int n, const double* v, double tau, - double* c, lapack_int ldc, double* work ); -lapack_int LAPACKE_clarfx_work( int matrix_order, char side, lapack_int m, - lapack_int n, const lapack_complex_float* v, - lapack_complex_float tau, - lapack_complex_float* c, lapack_int ldc, - lapack_complex_float* work ); -lapack_int LAPACKE_zlarfx_work( int matrix_order, char side, lapack_int m, - lapack_int n, const lapack_complex_double* v, - lapack_complex_double tau, - lapack_complex_double* c, lapack_int ldc, - lapack_complex_double* work ); - -lapack_int LAPACKE_slarnv_work( lapack_int idist, lapack_int* iseed, - lapack_int n, float* x ); -lapack_int LAPACKE_dlarnv_work( lapack_int idist, lapack_int* iseed, - lapack_int n, double* x ); -lapack_int LAPACKE_clarnv_work( lapack_int idist, lapack_int* iseed, - lapack_int n, lapack_complex_float* x ); -lapack_int LAPACKE_zlarnv_work( lapack_int idist, lapack_int* iseed, - lapack_int n, lapack_complex_double* x ); - -lapack_int LAPACKE_slaset_work( int matrix_order, char uplo, lapack_int m, - lapack_int n, float alpha, float beta, float* a, - lapack_int lda ); -lapack_int LAPACKE_dlaset_work( int matrix_order, char uplo, lapack_int m, - lapack_int n, double alpha, double beta, - double* a, lapack_int lda ); -lapack_int LAPACKE_claset_work( int matrix_order, char uplo, lapack_int m, - lapack_int n, lapack_complex_float alpha, - lapack_complex_float beta, - lapack_complex_float* a, lapack_int lda ); -lapack_int LAPACKE_zlaset_work( int matrix_order, char uplo, lapack_int m, - lapack_int n, lapack_complex_double alpha, - lapack_complex_double beta, - lapack_complex_double* a, lapack_int lda ); - -lapack_int LAPACKE_slasrt_work( char id, lapack_int n, float* d ); -lapack_int LAPACKE_dlasrt_work( char id, lapack_int n, double* d ); - -lapack_int LAPACKE_slaswp_work( int matrix_order, lapack_int n, float* a, - lapack_int lda, lapack_int k1, lapack_int k2, - const lapack_int* ipiv, lapack_int incx ); -lapack_int LAPACKE_dlaswp_work( int matrix_order, lapack_int n, double* a, - lapack_int lda, lapack_int k1, lapack_int k2, - const lapack_int* ipiv, lapack_int incx ); -lapack_int LAPACKE_claswp_work( int matrix_order, lapack_int n, - lapack_complex_float* a, lapack_int lda, - lapack_int k1, lapack_int k2, - const lapack_int* ipiv, lapack_int incx ); -lapack_int LAPACKE_zlaswp_work( int matrix_order, lapack_int n, - lapack_complex_double* a, lapack_int lda, - lapack_int k1, lapack_int k2, - const lapack_int* ipiv, lapack_int incx ); - -lapack_int LAPACKE_slatms_work( int matrix_order, lapack_int m, lapack_int n, - char dist, lapack_int* iseed, char sym, - float* d, lapack_int mode, float cond, - float dmax, lapack_int kl, lapack_int ku, - char pack, float* a, lapack_int lda, - float* work ); -lapack_int LAPACKE_dlatms_work( int matrix_order, lapack_int m, lapack_int n, - char dist, lapack_int* iseed, char sym, - double* d, lapack_int mode, double cond, - double dmax, lapack_int kl, lapack_int ku, - char pack, double* a, lapack_int lda, - double* work ); -lapack_int LAPACKE_clatms_work( int matrix_order, lapack_int m, lapack_int n, - char dist, lapack_int* iseed, char sym, - float* d, lapack_int mode, float cond, - float dmax, lapack_int kl, lapack_int ku, - char pack, lapack_complex_float* a, - lapack_int lda, lapack_complex_float* work ); -lapack_int LAPACKE_zlatms_work( int matrix_order, lapack_int m, lapack_int n, - char dist, lapack_int* iseed, char sym, - double* d, lapack_int mode, double cond, - double dmax, lapack_int kl, lapack_int ku, - char pack, lapack_complex_double* a, - lapack_int lda, lapack_complex_double* work ); - -lapack_int LAPACKE_slauum_work( int matrix_order, char uplo, lapack_int n, - float* a, lapack_int lda ); -lapack_int LAPACKE_dlauum_work( int matrix_order, char uplo, lapack_int n, - double* a, lapack_int lda ); -lapack_int LAPACKE_clauum_work( int matrix_order, char uplo, lapack_int n, - lapack_complex_float* a, lapack_int lda ); -lapack_int LAPACKE_zlauum_work( int matrix_order, char uplo, lapack_int n, - lapack_complex_double* a, lapack_int lda ); - -lapack_int LAPACKE_sopgtr_work( int matrix_order, char uplo, lapack_int n, - const float* ap, const float* tau, float* q, - lapack_int ldq, float* work ); -lapack_int LAPACKE_dopgtr_work( int matrix_order, char uplo, lapack_int n, - const double* ap, const double* tau, double* q, - lapack_int ldq, double* work ); - -lapack_int LAPACKE_sopmtr_work( int matrix_order, char side, char uplo, - char trans, lapack_int m, lapack_int n, - const float* ap, const float* tau, float* c, - lapack_int ldc, float* work ); -lapack_int LAPACKE_dopmtr_work( int matrix_order, char side, char uplo, - char trans, lapack_int m, lapack_int n, - const double* ap, const double* tau, double* c, - lapack_int ldc, double* work ); - -lapack_int LAPACKE_sorgbr_work( int matrix_order, char vect, lapack_int m, - lapack_int n, lapack_int k, float* a, - lapack_int lda, const float* tau, float* work, - lapack_int lwork ); -lapack_int LAPACKE_dorgbr_work( int matrix_order, char vect, lapack_int m, - lapack_int n, lapack_int k, double* a, - lapack_int lda, const double* tau, double* work, - lapack_int lwork ); - -lapack_int LAPACKE_sorghr_work( int matrix_order, lapack_int n, lapack_int ilo, - lapack_int ihi, float* a, lapack_int lda, - const float* tau, float* work, - lapack_int lwork ); -lapack_int LAPACKE_dorghr_work( int matrix_order, lapack_int n, lapack_int ilo, - lapack_int ihi, double* a, lapack_int lda, - const double* tau, double* work, - lapack_int lwork ); - -lapack_int LAPACKE_sorglq_work( int matrix_order, lapack_int m, lapack_int n, - lapack_int k, float* a, lapack_int lda, - const float* tau, float* work, - lapack_int lwork ); -lapack_int LAPACKE_dorglq_work( int matrix_order, lapack_int m, lapack_int n, - lapack_int k, double* a, lapack_int lda, - const double* tau, double* work, - lapack_int lwork ); - -lapack_int LAPACKE_sorgql_work( int matrix_order, lapack_int m, lapack_int n, - lapack_int k, float* a, lapack_int lda, - const float* tau, float* work, - lapack_int lwork ); -lapack_int LAPACKE_dorgql_work( int matrix_order, lapack_int m, lapack_int n, - lapack_int k, double* a, lapack_int lda, - const double* tau, double* work, - lapack_int lwork ); - -lapack_int LAPACKE_sorgqr_work( int matrix_order, lapack_int m, lapack_int n, - lapack_int k, float* a, lapack_int lda, - const float* tau, float* work, - lapack_int lwork ); -lapack_int LAPACKE_dorgqr_work( int matrix_order, lapack_int m, lapack_int n, - lapack_int k, double* a, lapack_int lda, - const double* tau, double* work, - lapack_int lwork ); - -lapack_int LAPACKE_sorgrq_work( int matrix_order, lapack_int m, lapack_int n, - lapack_int k, float* a, lapack_int lda, - const float* tau, float* work, - lapack_int lwork ); -lapack_int LAPACKE_dorgrq_work( int matrix_order, lapack_int m, lapack_int n, - lapack_int k, double* a, lapack_int lda, - const double* tau, double* work, - lapack_int lwork ); - -lapack_int LAPACKE_sorgtr_work( int matrix_order, char uplo, lapack_int n, - float* a, lapack_int lda, const float* tau, - float* work, lapack_int lwork ); -lapack_int LAPACKE_dorgtr_work( int matrix_order, char uplo, lapack_int n, - double* a, lapack_int lda, const double* tau, - double* work, lapack_int lwork ); - -lapack_int LAPACKE_sormbr_work( int matrix_order, char vect, char side, - char trans, lapack_int m, lapack_int n, - lapack_int k, const float* a, lapack_int lda, - const float* tau, float* c, lapack_int ldc, - float* work, lapack_int lwork ); -lapack_int LAPACKE_dormbr_work( int matrix_order, char vect, char side, - char trans, lapack_int m, lapack_int n, - lapack_int k, const double* a, lapack_int lda, - const double* tau, double* c, lapack_int ldc, - double* work, lapack_int lwork ); - -lapack_int LAPACKE_sormhr_work( int matrix_order, char side, char trans, - lapack_int m, lapack_int n, lapack_int ilo, - lapack_int ihi, const float* a, lapack_int lda, - const float* tau, float* c, lapack_int ldc, - float* work, lapack_int lwork ); -lapack_int LAPACKE_dormhr_work( int matrix_order, char side, char trans, - lapack_int m, lapack_int n, lapack_int ilo, - lapack_int ihi, const double* a, lapack_int lda, - const double* tau, double* c, lapack_int ldc, - double* work, lapack_int lwork ); - -lapack_int LAPACKE_sormlq_work( int matrix_order, char side, char trans, - lapack_int m, lapack_int n, lapack_int k, - const float* a, lapack_int lda, - const float* tau, float* c, lapack_int ldc, - float* work, lapack_int lwork ); -lapack_int LAPACKE_dormlq_work( int matrix_order, char side, char trans, - lapack_int m, lapack_int n, lapack_int k, - const double* a, lapack_int lda, - const double* tau, double* c, lapack_int ldc, - double* work, lapack_int lwork ); - -lapack_int LAPACKE_sormql_work( int matrix_order, char side, char trans, - lapack_int m, lapack_int n, lapack_int k, - const float* a, lapack_int lda, - const float* tau, float* c, lapack_int ldc, - float* work, lapack_int lwork ); -lapack_int LAPACKE_dormql_work( int matrix_order, char side, char trans, - lapack_int m, lapack_int n, lapack_int k, - const double* a, lapack_int lda, - const double* tau, double* c, lapack_int ldc, - double* work, lapack_int lwork ); - -lapack_int LAPACKE_sormqr_work( int matrix_order, char side, char trans, - lapack_int m, lapack_int n, lapack_int k, - const float* a, lapack_int lda, - const float* tau, float* c, lapack_int ldc, - float* work, lapack_int lwork ); -lapack_int LAPACKE_dormqr_work( int matrix_order, char side, char trans, - lapack_int m, lapack_int n, lapack_int k, - const double* a, lapack_int lda, - const double* tau, double* c, lapack_int ldc, - double* work, lapack_int lwork ); - -lapack_int LAPACKE_sormrq_work( int matrix_order, char side, char trans, - lapack_int m, lapack_int n, lapack_int k, - const float* a, lapack_int lda, - const float* tau, float* c, lapack_int ldc, - float* work, lapack_int lwork ); -lapack_int LAPACKE_dormrq_work( int matrix_order, char side, char trans, - lapack_int m, lapack_int n, lapack_int k, - const double* a, lapack_int lda, - const double* tau, double* c, lapack_int ldc, - double* work, lapack_int lwork ); - -lapack_int LAPACKE_sormrz_work( int matrix_order, char side, char trans, - lapack_int m, lapack_int n, lapack_int k, - lapack_int l, const float* a, lapack_int lda, - const float* tau, float* c, lapack_int ldc, - float* work, lapack_int lwork ); -lapack_int LAPACKE_dormrz_work( int matrix_order, char side, char trans, - lapack_int m, lapack_int n, lapack_int k, - lapack_int l, const double* a, lapack_int lda, - const double* tau, double* c, lapack_int ldc, - double* work, lapack_int lwork ); - -lapack_int LAPACKE_sormtr_work( int matrix_order, char side, char uplo, - char trans, lapack_int m, lapack_int n, - const float* a, lapack_int lda, - const float* tau, float* c, lapack_int ldc, - float* work, lapack_int lwork ); -lapack_int LAPACKE_dormtr_work( int matrix_order, char side, char uplo, - char trans, lapack_int m, lapack_int n, - const double* a, lapack_int lda, - const double* tau, double* c, lapack_int ldc, - double* work, lapack_int lwork ); - -lapack_int LAPACKE_spbcon_work( int matrix_order, char uplo, lapack_int n, - lapack_int kd, const float* ab, lapack_int ldab, - float anorm, float* rcond, float* work, - lapack_int* iwork ); -lapack_int LAPACKE_dpbcon_work( int matrix_order, char uplo, lapack_int n, - lapack_int kd, const double* ab, - lapack_int ldab, double anorm, double* rcond, - double* work, lapack_int* iwork ); -lapack_int LAPACKE_cpbcon_work( int matrix_order, char uplo, lapack_int n, - lapack_int kd, const lapack_complex_float* ab, - lapack_int ldab, float anorm, float* rcond, - lapack_complex_float* work, float* rwork ); -lapack_int LAPACKE_zpbcon_work( int matrix_order, char uplo, lapack_int n, - lapack_int kd, const lapack_complex_double* ab, - lapack_int ldab, double anorm, double* rcond, - lapack_complex_double* work, double* rwork ); - -lapack_int LAPACKE_spbequ_work( int matrix_order, char uplo, lapack_int n, - lapack_int kd, const float* ab, lapack_int ldab, - float* s, float* scond, float* amax ); -lapack_int LAPACKE_dpbequ_work( int matrix_order, char uplo, lapack_int n, - lapack_int kd, const double* ab, - lapack_int ldab, double* s, double* scond, - double* amax ); -lapack_int LAPACKE_cpbequ_work( int matrix_order, char uplo, lapack_int n, - lapack_int kd, const lapack_complex_float* ab, - lapack_int ldab, float* s, float* scond, - float* amax ); -lapack_int LAPACKE_zpbequ_work( int matrix_order, char uplo, lapack_int n, - lapack_int kd, const lapack_complex_double* ab, - lapack_int ldab, double* s, double* scond, - double* amax ); - -lapack_int LAPACKE_spbrfs_work( int matrix_order, char uplo, lapack_int n, - lapack_int kd, lapack_int nrhs, const float* ab, - lapack_int ldab, const float* afb, - lapack_int ldafb, const float* b, - lapack_int ldb, float* x, lapack_int ldx, - float* ferr, float* berr, float* work, - lapack_int* iwork ); -lapack_int LAPACKE_dpbrfs_work( int matrix_order, char uplo, lapack_int n, - lapack_int kd, lapack_int nrhs, - const double* ab, lapack_int ldab, - const double* afb, lapack_int ldafb, - const double* b, lapack_int ldb, double* x, - lapack_int ldx, double* ferr, double* berr, - double* work, lapack_int* iwork ); -lapack_int LAPACKE_cpbrfs_work( int matrix_order, char uplo, lapack_int n, - lapack_int kd, lapack_int nrhs, - const lapack_complex_float* ab, lapack_int ldab, - const lapack_complex_float* afb, - lapack_int ldafb, const lapack_complex_float* b, - lapack_int ldb, lapack_complex_float* x, - lapack_int ldx, float* ferr, float* berr, - lapack_complex_float* work, float* rwork ); -lapack_int LAPACKE_zpbrfs_work( int matrix_order, char uplo, lapack_int n, - lapack_int kd, lapack_int nrhs, - const lapack_complex_double* ab, - lapack_int ldab, - const lapack_complex_double* afb, - lapack_int ldafb, - const lapack_complex_double* b, lapack_int ldb, - lapack_complex_double* x, lapack_int ldx, - double* ferr, double* berr, - lapack_complex_double* work, double* rwork ); - -lapack_int LAPACKE_spbstf_work( int matrix_order, char uplo, lapack_int n, - lapack_int kb, float* bb, lapack_int ldbb ); -lapack_int LAPACKE_dpbstf_work( int matrix_order, char uplo, lapack_int n, - lapack_int kb, double* bb, lapack_int ldbb ); -lapack_int LAPACKE_cpbstf_work( int matrix_order, char uplo, lapack_int n, - lapack_int kb, lapack_complex_float* bb, - lapack_int ldbb ); -lapack_int LAPACKE_zpbstf_work( int matrix_order, char uplo, lapack_int n, - lapack_int kb, lapack_complex_double* bb, - lapack_int ldbb ); - -lapack_int LAPACKE_spbsv_work( int matrix_order, char uplo, lapack_int n, - lapack_int kd, lapack_int nrhs, float* ab, - lapack_int ldab, float* b, lapack_int ldb ); -lapack_int LAPACKE_dpbsv_work( int matrix_order, char uplo, lapack_int n, - lapack_int kd, lapack_int nrhs, double* ab, - lapack_int ldab, double* b, lapack_int ldb ); -lapack_int LAPACKE_cpbsv_work( int matrix_order, char uplo, lapack_int n, - lapack_int kd, lapack_int nrhs, - lapack_complex_float* ab, lapack_int ldab, - lapack_complex_float* b, lapack_int ldb ); -lapack_int LAPACKE_zpbsv_work( int matrix_order, char uplo, lapack_int n, - lapack_int kd, lapack_int nrhs, - lapack_complex_double* ab, lapack_int ldab, - lapack_complex_double* b, lapack_int ldb ); - -lapack_int LAPACKE_spbsvx_work( int matrix_order, char fact, char uplo, - lapack_int n, lapack_int kd, lapack_int nrhs, - float* ab, lapack_int ldab, float* afb, - lapack_int ldafb, char* equed, float* s, - float* b, lapack_int ldb, float* x, - lapack_int ldx, float* rcond, float* ferr, - float* berr, float* work, lapack_int* iwork ); -lapack_int LAPACKE_dpbsvx_work( int matrix_order, char fact, char uplo, - lapack_int n, lapack_int kd, lapack_int nrhs, - double* ab, lapack_int ldab, double* afb, - lapack_int ldafb, char* equed, double* s, - double* b, lapack_int ldb, double* x, - lapack_int ldx, double* rcond, double* ferr, - double* berr, double* work, lapack_int* iwork ); -lapack_int LAPACKE_cpbsvx_work( int matrix_order, char fact, char uplo, - lapack_int n, lapack_int kd, lapack_int nrhs, - lapack_complex_float* ab, lapack_int ldab, - lapack_complex_float* afb, lapack_int ldafb, - char* equed, float* s, lapack_complex_float* b, - lapack_int ldb, lapack_complex_float* x, - lapack_int ldx, float* rcond, float* ferr, - float* berr, lapack_complex_float* work, - float* rwork ); -lapack_int LAPACKE_zpbsvx_work( int matrix_order, char fact, char uplo, - lapack_int n, lapack_int kd, lapack_int nrhs, - lapack_complex_double* ab, lapack_int ldab, - lapack_complex_double* afb, lapack_int ldafb, - char* equed, double* s, - lapack_complex_double* b, lapack_int ldb, - lapack_complex_double* x, lapack_int ldx, - double* rcond, double* ferr, double* berr, - lapack_complex_double* work, double* rwork ); - -lapack_int LAPACKE_spbtrf_work( int matrix_order, char uplo, lapack_int n, - lapack_int kd, float* ab, lapack_int ldab ); -lapack_int LAPACKE_dpbtrf_work( int matrix_order, char uplo, lapack_int n, - lapack_int kd, double* ab, lapack_int ldab ); -lapack_int LAPACKE_cpbtrf_work( int matrix_order, char uplo, lapack_int n, - lapack_int kd, lapack_complex_float* ab, - lapack_int ldab ); -lapack_int LAPACKE_zpbtrf_work( int matrix_order, char uplo, lapack_int n, - lapack_int kd, lapack_complex_double* ab, - lapack_int ldab ); - -lapack_int LAPACKE_spbtrs_work( int matrix_order, char uplo, lapack_int n, - lapack_int kd, lapack_int nrhs, const float* ab, - lapack_int ldab, float* b, lapack_int ldb ); -lapack_int LAPACKE_dpbtrs_work( int matrix_order, char uplo, lapack_int n, - lapack_int kd, lapack_int nrhs, - const double* ab, lapack_int ldab, double* b, - lapack_int ldb ); -lapack_int LAPACKE_cpbtrs_work( int matrix_order, char uplo, lapack_int n, - lapack_int kd, lapack_int nrhs, - const lapack_complex_float* ab, lapack_int ldab, - lapack_complex_float* b, lapack_int ldb ); -lapack_int LAPACKE_zpbtrs_work( int matrix_order, char uplo, lapack_int n, - lapack_int kd, lapack_int nrhs, - const lapack_complex_double* ab, - lapack_int ldab, lapack_complex_double* b, - lapack_int ldb ); - -lapack_int LAPACKE_spftrf_work( int matrix_order, char transr, char uplo, - lapack_int n, float* a ); -lapack_int LAPACKE_dpftrf_work( int matrix_order, char transr, char uplo, - lapack_int n, double* a ); -lapack_int LAPACKE_cpftrf_work( int matrix_order, char transr, char uplo, - lapack_int n, lapack_complex_float* a ); -lapack_int LAPACKE_zpftrf_work( int matrix_order, char transr, char uplo, - lapack_int n, lapack_complex_double* a ); - -lapack_int LAPACKE_spftri_work( int matrix_order, char transr, char uplo, - lapack_int n, float* a ); -lapack_int LAPACKE_dpftri_work( int matrix_order, char transr, char uplo, - lapack_int n, double* a ); -lapack_int LAPACKE_cpftri_work( int matrix_order, char transr, char uplo, - lapack_int n, lapack_complex_float* a ); -lapack_int LAPACKE_zpftri_work( int matrix_order, char transr, char uplo, - lapack_int n, lapack_complex_double* a ); - -lapack_int LAPACKE_spftrs_work( int matrix_order, char transr, char uplo, - lapack_int n, lapack_int nrhs, const float* a, - float* b, lapack_int ldb ); -lapack_int LAPACKE_dpftrs_work( int matrix_order, char transr, char uplo, - lapack_int n, lapack_int nrhs, const double* a, - double* b, lapack_int ldb ); -lapack_int LAPACKE_cpftrs_work( int matrix_order, char transr, char uplo, - lapack_int n, lapack_int nrhs, - const lapack_complex_float* a, - lapack_complex_float* b, lapack_int ldb ); -lapack_int LAPACKE_zpftrs_work( int matrix_order, char transr, char uplo, - lapack_int n, lapack_int nrhs, - const lapack_complex_double* a, - lapack_complex_double* b, lapack_int ldb ); - -lapack_int LAPACKE_spocon_work( int matrix_order, char uplo, lapack_int n, - const float* a, lapack_int lda, float anorm, - float* rcond, float* work, lapack_int* iwork ); -lapack_int LAPACKE_dpocon_work( int matrix_order, char uplo, lapack_int n, - const double* a, lapack_int lda, double anorm, - double* rcond, double* work, - lapack_int* iwork ); -lapack_int LAPACKE_cpocon_work( int matrix_order, char uplo, lapack_int n, - const lapack_complex_float* a, lapack_int lda, - float anorm, float* rcond, - lapack_complex_float* work, float* rwork ); -lapack_int LAPACKE_zpocon_work( int matrix_order, char uplo, lapack_int n, - const lapack_complex_double* a, lapack_int lda, - double anorm, double* rcond, - lapack_complex_double* work, double* rwork ); - -lapack_int LAPACKE_spoequ_work( int matrix_order, lapack_int n, const float* a, - lapack_int lda, float* s, float* scond, - float* amax ); -lapack_int LAPACKE_dpoequ_work( int matrix_order, lapack_int n, const double* a, - lapack_int lda, double* s, double* scond, - double* amax ); -lapack_int LAPACKE_cpoequ_work( int matrix_order, lapack_int n, - const lapack_complex_float* a, lapack_int lda, - float* s, float* scond, float* amax ); -lapack_int LAPACKE_zpoequ_work( int matrix_order, lapack_int n, - const lapack_complex_double* a, lapack_int lda, - double* s, double* scond, double* amax ); - -lapack_int LAPACKE_spoequb_work( int matrix_order, lapack_int n, const float* a, - lapack_int lda, float* s, float* scond, - float* amax ); -lapack_int LAPACKE_dpoequb_work( int matrix_order, lapack_int n, - const double* a, lapack_int lda, double* s, - double* scond, double* amax ); -lapack_int LAPACKE_cpoequb_work( int matrix_order, lapack_int n, - const lapack_complex_float* a, lapack_int lda, - float* s, float* scond, float* amax ); -lapack_int LAPACKE_zpoequb_work( int matrix_order, lapack_int n, - const lapack_complex_double* a, lapack_int lda, - double* s, double* scond, double* amax ); - -lapack_int LAPACKE_sporfs_work( int matrix_order, char uplo, lapack_int n, - lapack_int nrhs, const float* a, lapack_int lda, - const float* af, lapack_int ldaf, - const float* b, lapack_int ldb, float* x, - lapack_int ldx, float* ferr, float* berr, - float* work, lapack_int* iwork ); -lapack_int LAPACKE_dporfs_work( int matrix_order, char uplo, lapack_int n, - lapack_int nrhs, const double* a, - lapack_int lda, const double* af, - lapack_int ldaf, const double* b, - lapack_int ldb, double* x, lapack_int ldx, - double* ferr, double* berr, double* work, - lapack_int* iwork ); -lapack_int LAPACKE_cporfs_work( int matrix_order, char uplo, lapack_int n, - lapack_int nrhs, const lapack_complex_float* a, - lapack_int lda, const lapack_complex_float* af, - lapack_int ldaf, const lapack_complex_float* b, - lapack_int ldb, lapack_complex_float* x, - lapack_int ldx, float* ferr, float* berr, - lapack_complex_float* work, float* rwork ); -lapack_int LAPACKE_zporfs_work( int matrix_order, char uplo, lapack_int n, - lapack_int nrhs, const lapack_complex_double* a, - lapack_int lda, const lapack_complex_double* af, - lapack_int ldaf, const lapack_complex_double* b, - lapack_int ldb, lapack_complex_double* x, - lapack_int ldx, double* ferr, double* berr, - lapack_complex_double* work, double* rwork ); - -lapack_int LAPACKE_sporfsx_work( int matrix_order, char uplo, char equed, - lapack_int n, lapack_int nrhs, const float* a, - lapack_int lda, const float* af, - lapack_int ldaf, const float* s, - const float* b, lapack_int ldb, float* x, - lapack_int ldx, float* rcond, float* berr, - lapack_int n_err_bnds, float* err_bnds_norm, - float* err_bnds_comp, lapack_int nparams, - float* params, float* work, - lapack_int* iwork ); -lapack_int LAPACKE_dporfsx_work( int matrix_order, char uplo, char equed, - lapack_int n, lapack_int nrhs, const double* a, - lapack_int lda, const double* af, - lapack_int ldaf, const double* s, - const double* b, lapack_int ldb, double* x, - lapack_int ldx, double* rcond, double* berr, - lapack_int n_err_bnds, double* err_bnds_norm, - double* err_bnds_comp, lapack_int nparams, - double* params, double* work, - lapack_int* iwork ); -lapack_int LAPACKE_cporfsx_work( int matrix_order, char uplo, char equed, - lapack_int n, lapack_int nrhs, - const lapack_complex_float* a, lapack_int lda, - const lapack_complex_float* af, - lapack_int ldaf, const float* s, - const lapack_complex_float* b, lapack_int ldb, - lapack_complex_float* x, lapack_int ldx, - float* rcond, float* berr, - lapack_int n_err_bnds, float* err_bnds_norm, - float* err_bnds_comp, lapack_int nparams, - float* params, lapack_complex_float* work, - float* rwork ); -lapack_int LAPACKE_zporfsx_work( int matrix_order, char uplo, char equed, - lapack_int n, lapack_int nrhs, - const lapack_complex_double* a, lapack_int lda, - const lapack_complex_double* af, - lapack_int ldaf, const double* s, - const lapack_complex_double* b, lapack_int ldb, - lapack_complex_double* x, lapack_int ldx, - double* rcond, double* berr, - lapack_int n_err_bnds, double* err_bnds_norm, - double* err_bnds_comp, lapack_int nparams, - double* params, lapack_complex_double* work, - double* rwork ); - -lapack_int LAPACKE_sposv_work( int matrix_order, char uplo, lapack_int n, - lapack_int nrhs, float* a, lapack_int lda, - float* b, lapack_int ldb ); -lapack_int LAPACKE_dposv_work( int matrix_order, char uplo, lapack_int n, - lapack_int nrhs, double* a, lapack_int lda, - double* b, lapack_int ldb ); -lapack_int LAPACKE_cposv_work( int matrix_order, char uplo, lapack_int n, - lapack_int nrhs, lapack_complex_float* a, - lapack_int lda, lapack_complex_float* b, - lapack_int ldb ); -lapack_int LAPACKE_zposv_work( int matrix_order, char uplo, lapack_int n, - lapack_int nrhs, lapack_complex_double* a, - lapack_int lda, lapack_complex_double* b, - lapack_int ldb ); -lapack_int LAPACKE_dsposv_work( int matrix_order, char uplo, lapack_int n, - lapack_int nrhs, double* a, lapack_int lda, - double* b, lapack_int ldb, double* x, - lapack_int ldx, double* work, float* swork, - lapack_int* iter ); -lapack_int LAPACKE_zcposv_work( int matrix_order, char uplo, lapack_int n, - lapack_int nrhs, lapack_complex_double* a, - lapack_int lda, lapack_complex_double* b, - lapack_int ldb, lapack_complex_double* x, - lapack_int ldx, lapack_complex_double* work, - lapack_complex_float* swork, double* rwork, - lapack_int* iter ); - -lapack_int LAPACKE_sposvx_work( int matrix_order, char fact, char uplo, - lapack_int n, lapack_int nrhs, float* a, - lapack_int lda, float* af, lapack_int ldaf, - char* equed, float* s, float* b, lapack_int ldb, - float* x, lapack_int ldx, float* rcond, - float* ferr, float* berr, float* work, - lapack_int* iwork ); -lapack_int LAPACKE_dposvx_work( int matrix_order, char fact, char uplo, - lapack_int n, lapack_int nrhs, double* a, - lapack_int lda, double* af, lapack_int ldaf, - char* equed, double* s, double* b, - lapack_int ldb, double* x, lapack_int ldx, - double* rcond, double* ferr, double* berr, - double* work, lapack_int* iwork ); -lapack_int LAPACKE_cposvx_work( int matrix_order, char fact, char uplo, - lapack_int n, lapack_int nrhs, - lapack_complex_float* a, lapack_int lda, - lapack_complex_float* af, lapack_int ldaf, - char* equed, float* s, lapack_complex_float* b, - lapack_int ldb, lapack_complex_float* x, - lapack_int ldx, float* rcond, float* ferr, - float* berr, lapack_complex_float* work, - float* rwork ); -lapack_int LAPACKE_zposvx_work( int matrix_order, char fact, char uplo, - lapack_int n, lapack_int nrhs, - lapack_complex_double* a, lapack_int lda, - lapack_complex_double* af, lapack_int ldaf, - char* equed, double* s, - lapack_complex_double* b, lapack_int ldb, - lapack_complex_double* x, lapack_int ldx, - double* rcond, double* ferr, double* berr, - lapack_complex_double* work, double* rwork ); - -lapack_int LAPACKE_sposvxx_work( int matrix_order, char fact, char uplo, - lapack_int n, lapack_int nrhs, float* a, - lapack_int lda, float* af, lapack_int ldaf, - char* equed, float* s, float* b, - lapack_int ldb, float* x, lapack_int ldx, - float* rcond, float* rpvgrw, float* berr, - lapack_int n_err_bnds, float* err_bnds_norm, - float* err_bnds_comp, lapack_int nparams, - float* params, float* work, - lapack_int* iwork ); -lapack_int LAPACKE_dposvxx_work( int matrix_order, char fact, char uplo, - lapack_int n, lapack_int nrhs, double* a, - lapack_int lda, double* af, lapack_int ldaf, - char* equed, double* s, double* b, - lapack_int ldb, double* x, lapack_int ldx, - double* rcond, double* rpvgrw, double* berr, - lapack_int n_err_bnds, double* err_bnds_norm, - double* err_bnds_comp, lapack_int nparams, - double* params, double* work, - lapack_int* iwork ); -lapack_int LAPACKE_cposvxx_work( int matrix_order, char fact, char uplo, - lapack_int n, lapack_int nrhs, - lapack_complex_float* a, lapack_int lda, - lapack_complex_float* af, lapack_int ldaf, - char* equed, float* s, lapack_complex_float* b, - lapack_int ldb, lapack_complex_float* x, - lapack_int ldx, float* rcond, float* rpvgrw, - float* berr, lapack_int n_err_bnds, - float* err_bnds_norm, float* err_bnds_comp, - lapack_int nparams, float* params, - lapack_complex_float* work, float* rwork ); -lapack_int LAPACKE_zposvxx_work( int matrix_order, char fact, char uplo, - lapack_int n, lapack_int nrhs, - lapack_complex_double* a, lapack_int lda, - lapack_complex_double* af, lapack_int ldaf, - char* equed, double* s, - lapack_complex_double* b, lapack_int ldb, - lapack_complex_double* x, lapack_int ldx, - double* rcond, double* rpvgrw, double* berr, - lapack_int n_err_bnds, double* err_bnds_norm, - double* err_bnds_comp, lapack_int nparams, - double* params, lapack_complex_double* work, - double* rwork ); - -lapack_int LAPACKE_spotrf_work( int matrix_order, char uplo, lapack_int n, - float* a, lapack_int lda ); -lapack_int LAPACKE_dpotrf_work( int matrix_order, char uplo, lapack_int n, - double* a, lapack_int lda ); -lapack_int LAPACKE_cpotrf_work( int matrix_order, char uplo, lapack_int n, - lapack_complex_float* a, lapack_int lda ); -lapack_int LAPACKE_zpotrf_work( int matrix_order, char uplo, lapack_int n, - lapack_complex_double* a, lapack_int lda ); - -lapack_int LAPACKE_spotri_work( int matrix_order, char uplo, lapack_int n, - float* a, lapack_int lda ); -lapack_int LAPACKE_dpotri_work( int matrix_order, char uplo, lapack_int n, - double* a, lapack_int lda ); -lapack_int LAPACKE_cpotri_work( int matrix_order, char uplo, lapack_int n, - lapack_complex_float* a, lapack_int lda ); -lapack_int LAPACKE_zpotri_work( int matrix_order, char uplo, lapack_int n, - lapack_complex_double* a, lapack_int lda ); - -lapack_int LAPACKE_spotrs_work( int matrix_order, char uplo, lapack_int n, - lapack_int nrhs, const float* a, lapack_int lda, - float* b, lapack_int ldb ); -lapack_int LAPACKE_dpotrs_work( int matrix_order, char uplo, lapack_int n, - lapack_int nrhs, const double* a, - lapack_int lda, double* b, lapack_int ldb ); -lapack_int LAPACKE_cpotrs_work( int matrix_order, char uplo, lapack_int n, - lapack_int nrhs, const lapack_complex_float* a, - lapack_int lda, lapack_complex_float* b, - lapack_int ldb ); -lapack_int LAPACKE_zpotrs_work( int matrix_order, char uplo, lapack_int n, - lapack_int nrhs, const lapack_complex_double* a, - lapack_int lda, lapack_complex_double* b, - lapack_int ldb ); - -lapack_int LAPACKE_sppcon_work( int matrix_order, char uplo, lapack_int n, - const float* ap, float anorm, float* rcond, - float* work, lapack_int* iwork ); -lapack_int LAPACKE_dppcon_work( int matrix_order, char uplo, lapack_int n, - const double* ap, double anorm, double* rcond, - double* work, lapack_int* iwork ); -lapack_int LAPACKE_cppcon_work( int matrix_order, char uplo, lapack_int n, - const lapack_complex_float* ap, float anorm, - float* rcond, lapack_complex_float* work, - float* rwork ); -lapack_int LAPACKE_zppcon_work( int matrix_order, char uplo, lapack_int n, - const lapack_complex_double* ap, double anorm, - double* rcond, lapack_complex_double* work, - double* rwork ); - -lapack_int LAPACKE_sppequ_work( int matrix_order, char uplo, lapack_int n, - const float* ap, float* s, float* scond, - float* amax ); -lapack_int LAPACKE_dppequ_work( int matrix_order, char uplo, lapack_int n, - const double* ap, double* s, double* scond, - double* amax ); -lapack_int LAPACKE_cppequ_work( int matrix_order, char uplo, lapack_int n, - const lapack_complex_float* ap, float* s, - float* scond, float* amax ); -lapack_int LAPACKE_zppequ_work( int matrix_order, char uplo, lapack_int n, - const lapack_complex_double* ap, double* s, - double* scond, double* amax ); - -lapack_int LAPACKE_spprfs_work( int matrix_order, char uplo, lapack_int n, - lapack_int nrhs, const float* ap, - const float* afp, const float* b, - lapack_int ldb, float* x, lapack_int ldx, - float* ferr, float* berr, float* work, - lapack_int* iwork ); -lapack_int LAPACKE_dpprfs_work( int matrix_order, char uplo, lapack_int n, - lapack_int nrhs, const double* ap, - const double* afp, const double* b, - lapack_int ldb, double* x, lapack_int ldx, - double* ferr, double* berr, double* work, - lapack_int* iwork ); -lapack_int LAPACKE_cpprfs_work( int matrix_order, char uplo, lapack_int n, - lapack_int nrhs, const lapack_complex_float* ap, - const lapack_complex_float* afp, - const lapack_complex_float* b, lapack_int ldb, - lapack_complex_float* x, lapack_int ldx, - float* ferr, float* berr, - lapack_complex_float* work, float* rwork ); -lapack_int LAPACKE_zpprfs_work( int matrix_order, char uplo, lapack_int n, - lapack_int nrhs, - const lapack_complex_double* ap, - const lapack_complex_double* afp, - const lapack_complex_double* b, lapack_int ldb, - lapack_complex_double* x, lapack_int ldx, - double* ferr, double* berr, - lapack_complex_double* work, double* rwork ); - -lapack_int LAPACKE_sppsv_work( int matrix_order, char uplo, lapack_int n, - lapack_int nrhs, float* ap, float* b, - lapack_int ldb ); -lapack_int LAPACKE_dppsv_work( int matrix_order, char uplo, lapack_int n, - lapack_int nrhs, double* ap, double* b, - lapack_int ldb ); -lapack_int LAPACKE_cppsv_work( int matrix_order, char uplo, lapack_int n, - lapack_int nrhs, lapack_complex_float* ap, - lapack_complex_float* b, lapack_int ldb ); -lapack_int LAPACKE_zppsv_work( int matrix_order, char uplo, lapack_int n, - lapack_int nrhs, lapack_complex_double* ap, - lapack_complex_double* b, lapack_int ldb ); - -lapack_int LAPACKE_sppsvx_work( int matrix_order, char fact, char uplo, - lapack_int n, lapack_int nrhs, float* ap, - float* afp, char* equed, float* s, float* b, - lapack_int ldb, float* x, lapack_int ldx, - float* rcond, float* ferr, float* berr, - float* work, lapack_int* iwork ); -lapack_int LAPACKE_dppsvx_work( int matrix_order, char fact, char uplo, - lapack_int n, lapack_int nrhs, double* ap, - double* afp, char* equed, double* s, double* b, - lapack_int ldb, double* x, lapack_int ldx, - double* rcond, double* ferr, double* berr, - double* work, lapack_int* iwork ); -lapack_int LAPACKE_cppsvx_work( int matrix_order, char fact, char uplo, - lapack_int n, lapack_int nrhs, - lapack_complex_float* ap, - lapack_complex_float* afp, char* equed, - float* s, lapack_complex_float* b, - lapack_int ldb, lapack_complex_float* x, - lapack_int ldx, float* rcond, float* ferr, - float* berr, lapack_complex_float* work, - float* rwork ); -lapack_int LAPACKE_zppsvx_work( int matrix_order, char fact, char uplo, - lapack_int n, lapack_int nrhs, - lapack_complex_double* ap, - lapack_complex_double* afp, char* equed, - double* s, lapack_complex_double* b, - lapack_int ldb, lapack_complex_double* x, - lapack_int ldx, double* rcond, double* ferr, - double* berr, lapack_complex_double* work, - double* rwork ); - -lapack_int LAPACKE_spptrf_work( int matrix_order, char uplo, lapack_int n, - float* ap ); -lapack_int LAPACKE_dpptrf_work( int matrix_order, char uplo, lapack_int n, - double* ap ); -lapack_int LAPACKE_cpptrf_work( int matrix_order, char uplo, lapack_int n, - lapack_complex_float* ap ); -lapack_int LAPACKE_zpptrf_work( int matrix_order, char uplo, lapack_int n, - lapack_complex_double* ap ); - -lapack_int LAPACKE_spptri_work( int matrix_order, char uplo, lapack_int n, - float* ap ); -lapack_int LAPACKE_dpptri_work( int matrix_order, char uplo, lapack_int n, - double* ap ); -lapack_int LAPACKE_cpptri_work( int matrix_order, char uplo, lapack_int n, - lapack_complex_float* ap ); -lapack_int LAPACKE_zpptri_work( int matrix_order, char uplo, lapack_int n, - lapack_complex_double* ap ); - -lapack_int LAPACKE_spptrs_work( int matrix_order, char uplo, lapack_int n, - lapack_int nrhs, const float* ap, float* b, - lapack_int ldb ); -lapack_int LAPACKE_dpptrs_work( int matrix_order, char uplo, lapack_int n, - lapack_int nrhs, const double* ap, double* b, - lapack_int ldb ); -lapack_int LAPACKE_cpptrs_work( int matrix_order, char uplo, lapack_int n, - lapack_int nrhs, const lapack_complex_float* ap, - lapack_complex_float* b, lapack_int ldb ); -lapack_int LAPACKE_zpptrs_work( int matrix_order, char uplo, lapack_int n, - lapack_int nrhs, - const lapack_complex_double* ap, - lapack_complex_double* b, lapack_int ldb ); - -lapack_int LAPACKE_spstrf_work( int matrix_order, char uplo, lapack_int n, - float* a, lapack_int lda, lapack_int* piv, - lapack_int* rank, float tol, float* work ); -lapack_int LAPACKE_dpstrf_work( int matrix_order, char uplo, lapack_int n, - double* a, lapack_int lda, lapack_int* piv, - lapack_int* rank, double tol, double* work ); -lapack_int LAPACKE_cpstrf_work( int matrix_order, char uplo, lapack_int n, - lapack_complex_float* a, lapack_int lda, - lapack_int* piv, lapack_int* rank, float tol, - float* work ); -lapack_int LAPACKE_zpstrf_work( int matrix_order, char uplo, lapack_int n, - lapack_complex_double* a, lapack_int lda, - lapack_int* piv, lapack_int* rank, double tol, - double* work ); - -lapack_int LAPACKE_sptcon_work( lapack_int n, const float* d, const float* e, - float anorm, float* rcond, float* work ); -lapack_int LAPACKE_dptcon_work( lapack_int n, const double* d, const double* e, - double anorm, double* rcond, double* work ); -lapack_int LAPACKE_cptcon_work( lapack_int n, const float* d, - const lapack_complex_float* e, float anorm, - float* rcond, float* work ); -lapack_int LAPACKE_zptcon_work( lapack_int n, const double* d, - const lapack_complex_double* e, double anorm, - double* rcond, double* work ); - -lapack_int LAPACKE_spteqr_work( int matrix_order, char compz, lapack_int n, - float* d, float* e, float* z, lapack_int ldz, - float* work ); -lapack_int LAPACKE_dpteqr_work( int matrix_order, char compz, lapack_int n, - double* d, double* e, double* z, lapack_int ldz, - double* work ); -lapack_int LAPACKE_cpteqr_work( int matrix_order, char compz, lapack_int n, - float* d, float* e, lapack_complex_float* z, - lapack_int ldz, float* work ); -lapack_int LAPACKE_zpteqr_work( int matrix_order, char compz, lapack_int n, - double* d, double* e, lapack_complex_double* z, - lapack_int ldz, double* work ); - -lapack_int LAPACKE_sptrfs_work( int matrix_order, lapack_int n, lapack_int nrhs, - const float* d, const float* e, const float* df, - const float* ef, const float* b, lapack_int ldb, - float* x, lapack_int ldx, float* ferr, - float* berr, float* work ); -lapack_int LAPACKE_dptrfs_work( int matrix_order, lapack_int n, lapack_int nrhs, - const double* d, const double* e, - const double* df, const double* ef, - const double* b, lapack_int ldb, double* x, - lapack_int ldx, double* ferr, double* berr, - double* work ); -lapack_int LAPACKE_cptrfs_work( int matrix_order, char uplo, lapack_int n, - lapack_int nrhs, const float* d, - const lapack_complex_float* e, const float* df, - const lapack_complex_float* ef, - const lapack_complex_float* b, lapack_int ldb, - lapack_complex_float* x, lapack_int ldx, - float* ferr, float* berr, - lapack_complex_float* work, float* rwork ); -lapack_int LAPACKE_zptrfs_work( int matrix_order, char uplo, lapack_int n, - lapack_int nrhs, const double* d, - const lapack_complex_double* e, - const double* df, - const lapack_complex_double* ef, - const lapack_complex_double* b, lapack_int ldb, - lapack_complex_double* x, lapack_int ldx, - double* ferr, double* berr, - lapack_complex_double* work, double* rwork ); - -lapack_int LAPACKE_sptsv_work( int matrix_order, lapack_int n, lapack_int nrhs, - float* d, float* e, float* b, lapack_int ldb ); -lapack_int LAPACKE_dptsv_work( int matrix_order, lapack_int n, lapack_int nrhs, - double* d, double* e, double* b, - lapack_int ldb ); -lapack_int LAPACKE_cptsv_work( int matrix_order, lapack_int n, lapack_int nrhs, - float* d, lapack_complex_float* e, - lapack_complex_float* b, lapack_int ldb ); -lapack_int LAPACKE_zptsv_work( int matrix_order, lapack_int n, lapack_int nrhs, - double* d, lapack_complex_double* e, - lapack_complex_double* b, lapack_int ldb ); - -lapack_int LAPACKE_sptsvx_work( int matrix_order, char fact, lapack_int n, - lapack_int nrhs, const float* d, const float* e, - float* df, float* ef, const float* b, - lapack_int ldb, float* x, lapack_int ldx, - float* rcond, float* ferr, float* berr, - float* work ); -lapack_int LAPACKE_dptsvx_work( int matrix_order, char fact, lapack_int n, - lapack_int nrhs, const double* d, - const double* e, double* df, double* ef, - const double* b, lapack_int ldb, double* x, - lapack_int ldx, double* rcond, double* ferr, - double* berr, double* work ); -lapack_int LAPACKE_cptsvx_work( int matrix_order, char fact, lapack_int n, - lapack_int nrhs, const float* d, - const lapack_complex_float* e, float* df, - lapack_complex_float* ef, - const lapack_complex_float* b, lapack_int ldb, - lapack_complex_float* x, lapack_int ldx, - float* rcond, float* ferr, float* berr, - lapack_complex_float* work, float* rwork ); -lapack_int LAPACKE_zptsvx_work( int matrix_order, char fact, lapack_int n, - lapack_int nrhs, const double* d, - const lapack_complex_double* e, double* df, - lapack_complex_double* ef, - const lapack_complex_double* b, lapack_int ldb, - lapack_complex_double* x, lapack_int ldx, - double* rcond, double* ferr, double* berr, - lapack_complex_double* work, double* rwork ); - -lapack_int LAPACKE_spttrf_work( lapack_int n, float* d, float* e ); -lapack_int LAPACKE_dpttrf_work( lapack_int n, double* d, double* e ); -lapack_int LAPACKE_cpttrf_work( lapack_int n, float* d, - lapack_complex_float* e ); -lapack_int LAPACKE_zpttrf_work( lapack_int n, double* d, - lapack_complex_double* e ); - -lapack_int LAPACKE_spttrs_work( int matrix_order, lapack_int n, lapack_int nrhs, - const float* d, const float* e, float* b, - lapack_int ldb ); -lapack_int LAPACKE_dpttrs_work( int matrix_order, lapack_int n, lapack_int nrhs, - const double* d, const double* e, double* b, - lapack_int ldb ); -lapack_int LAPACKE_cpttrs_work( int matrix_order, char uplo, lapack_int n, - lapack_int nrhs, const float* d, - const lapack_complex_float* e, - lapack_complex_float* b, lapack_int ldb ); -lapack_int LAPACKE_zpttrs_work( int matrix_order, char uplo, lapack_int n, - lapack_int nrhs, const double* d, - const lapack_complex_double* e, - lapack_complex_double* b, lapack_int ldb ); - -lapack_int LAPACKE_ssbev_work( int matrix_order, char jobz, char uplo, - lapack_int n, lapack_int kd, float* ab, - lapack_int ldab, float* w, float* z, - lapack_int ldz, float* work ); -lapack_int LAPACKE_dsbev_work( int matrix_order, char jobz, char uplo, - lapack_int n, lapack_int kd, double* ab, - lapack_int ldab, double* w, double* z, - lapack_int ldz, double* work ); - -lapack_int LAPACKE_ssbevd_work( int matrix_order, char jobz, char uplo, - lapack_int n, lapack_int kd, float* ab, - lapack_int ldab, float* w, float* z, - lapack_int ldz, float* work, lapack_int lwork, - lapack_int* iwork, lapack_int liwork ); -lapack_int LAPACKE_dsbevd_work( int matrix_order, char jobz, char uplo, - lapack_int n, lapack_int kd, double* ab, - lapack_int ldab, double* w, double* z, - lapack_int ldz, double* work, lapack_int lwork, - lapack_int* iwork, lapack_int liwork ); - -lapack_int LAPACKE_ssbevx_work( int matrix_order, char jobz, char range, - char uplo, lapack_int n, lapack_int kd, - float* ab, lapack_int ldab, float* q, - lapack_int ldq, float vl, float vu, - lapack_int il, lapack_int iu, float abstol, - lapack_int* m, float* w, float* z, - lapack_int ldz, float* work, lapack_int* iwork, - lapack_int* ifail ); -lapack_int LAPACKE_dsbevx_work( int matrix_order, char jobz, char range, - char uplo, lapack_int n, lapack_int kd, - double* ab, lapack_int ldab, double* q, - lapack_int ldq, double vl, double vu, - lapack_int il, lapack_int iu, double abstol, - lapack_int* m, double* w, double* z, - lapack_int ldz, double* work, lapack_int* iwork, - lapack_int* ifail ); - -lapack_int LAPACKE_ssbgst_work( int matrix_order, char vect, char uplo, - lapack_int n, lapack_int ka, lapack_int kb, - float* ab, lapack_int ldab, const float* bb, - lapack_int ldbb, float* x, lapack_int ldx, - float* work ); -lapack_int LAPACKE_dsbgst_work( int matrix_order, char vect, char uplo, - lapack_int n, lapack_int ka, lapack_int kb, - double* ab, lapack_int ldab, const double* bb, - lapack_int ldbb, double* x, lapack_int ldx, - double* work ); - -lapack_int LAPACKE_ssbgv_work( int matrix_order, char jobz, char uplo, - lapack_int n, lapack_int ka, lapack_int kb, - float* ab, lapack_int ldab, float* bb, - lapack_int ldbb, float* w, float* z, - lapack_int ldz, float* work ); -lapack_int LAPACKE_dsbgv_work( int matrix_order, char jobz, char uplo, - lapack_int n, lapack_int ka, lapack_int kb, - double* ab, lapack_int ldab, double* bb, - lapack_int ldbb, double* w, double* z, - lapack_int ldz, double* work ); - -lapack_int LAPACKE_ssbgvd_work( int matrix_order, char jobz, char uplo, - lapack_int n, lapack_int ka, lapack_int kb, - float* ab, lapack_int ldab, float* bb, - lapack_int ldbb, float* w, float* z, - lapack_int ldz, float* work, lapack_int lwork, - lapack_int* iwork, lapack_int liwork ); -lapack_int LAPACKE_dsbgvd_work( int matrix_order, char jobz, char uplo, - lapack_int n, lapack_int ka, lapack_int kb, - double* ab, lapack_int ldab, double* bb, - lapack_int ldbb, double* w, double* z, - lapack_int ldz, double* work, lapack_int lwork, - lapack_int* iwork, lapack_int liwork ); - -lapack_int LAPACKE_ssbgvx_work( int matrix_order, char jobz, char range, - char uplo, lapack_int n, lapack_int ka, - lapack_int kb, float* ab, lapack_int ldab, - float* bb, lapack_int ldbb, float* q, - lapack_int ldq, float vl, float vu, - lapack_int il, lapack_int iu, float abstol, - lapack_int* m, float* w, float* z, - lapack_int ldz, float* work, lapack_int* iwork, - lapack_int* ifail ); -lapack_int LAPACKE_dsbgvx_work( int matrix_order, char jobz, char range, - char uplo, lapack_int n, lapack_int ka, - lapack_int kb, double* ab, lapack_int ldab, - double* bb, lapack_int ldbb, double* q, - lapack_int ldq, double vl, double vu, - lapack_int il, lapack_int iu, double abstol, - lapack_int* m, double* w, double* z, - lapack_int ldz, double* work, lapack_int* iwork, - lapack_int* ifail ); - -lapack_int LAPACKE_ssbtrd_work( int matrix_order, char vect, char uplo, - lapack_int n, lapack_int kd, float* ab, - lapack_int ldab, float* d, float* e, float* q, - lapack_int ldq, float* work ); -lapack_int LAPACKE_dsbtrd_work( int matrix_order, char vect, char uplo, - lapack_int n, lapack_int kd, double* ab, - lapack_int ldab, double* d, double* e, - double* q, lapack_int ldq, double* work ); - -lapack_int LAPACKE_ssfrk_work( int matrix_order, char transr, char uplo, - char trans, lapack_int n, lapack_int k, - float alpha, const float* a, lapack_int lda, - float beta, float* c ); -lapack_int LAPACKE_dsfrk_work( int matrix_order, char transr, char uplo, - char trans, lapack_int n, lapack_int k, - double alpha, const double* a, lapack_int lda, - double beta, double* c ); - -lapack_int LAPACKE_sspcon_work( int matrix_order, char uplo, lapack_int n, - const float* ap, const lapack_int* ipiv, - float anorm, float* rcond, float* work, - lapack_int* iwork ); -lapack_int LAPACKE_dspcon_work( int matrix_order, char uplo, lapack_int n, - const double* ap, const lapack_int* ipiv, - double anorm, double* rcond, double* work, - lapack_int* iwork ); -lapack_int LAPACKE_cspcon_work( int matrix_order, char uplo, lapack_int n, - const lapack_complex_float* ap, - const lapack_int* ipiv, float anorm, - float* rcond, lapack_complex_float* work ); -lapack_int LAPACKE_zspcon_work( int matrix_order, char uplo, lapack_int n, - const lapack_complex_double* ap, - const lapack_int* ipiv, double anorm, - double* rcond, lapack_complex_double* work ); - -lapack_int LAPACKE_sspev_work( int matrix_order, char jobz, char uplo, - lapack_int n, float* ap, float* w, float* z, - lapack_int ldz, float* work ); -lapack_int LAPACKE_dspev_work( int matrix_order, char jobz, char uplo, - lapack_int n, double* ap, double* w, double* z, - lapack_int ldz, double* work ); - -lapack_int LAPACKE_sspevd_work( int matrix_order, char jobz, char uplo, - lapack_int n, float* ap, float* w, float* z, - lapack_int ldz, float* work, lapack_int lwork, - lapack_int* iwork, lapack_int liwork ); -lapack_int LAPACKE_dspevd_work( int matrix_order, char jobz, char uplo, - lapack_int n, double* ap, double* w, double* z, - lapack_int ldz, double* work, lapack_int lwork, - lapack_int* iwork, lapack_int liwork ); - -lapack_int LAPACKE_sspevx_work( int matrix_order, char jobz, char range, - char uplo, lapack_int n, float* ap, float vl, - float vu, lapack_int il, lapack_int iu, - float abstol, lapack_int* m, float* w, float* z, - lapack_int ldz, float* work, lapack_int* iwork, - lapack_int* ifail ); -lapack_int LAPACKE_dspevx_work( int matrix_order, char jobz, char range, - char uplo, lapack_int n, double* ap, double vl, - double vu, lapack_int il, lapack_int iu, - double abstol, lapack_int* m, double* w, - double* z, lapack_int ldz, double* work, - lapack_int* iwork, lapack_int* ifail ); - -lapack_int LAPACKE_sspgst_work( int matrix_order, lapack_int itype, char uplo, - lapack_int n, float* ap, const float* bp ); -lapack_int LAPACKE_dspgst_work( int matrix_order, lapack_int itype, char uplo, - lapack_int n, double* ap, const double* bp ); - -lapack_int LAPACKE_sspgv_work( int matrix_order, lapack_int itype, char jobz, - char uplo, lapack_int n, float* ap, float* bp, - float* w, float* z, lapack_int ldz, - float* work ); -lapack_int LAPACKE_dspgv_work( int matrix_order, lapack_int itype, char jobz, - char uplo, lapack_int n, double* ap, double* bp, - double* w, double* z, lapack_int ldz, - double* work ); - -lapack_int LAPACKE_sspgvd_work( int matrix_order, lapack_int itype, char jobz, - char uplo, lapack_int n, float* ap, float* bp, - float* w, float* z, lapack_int ldz, float* work, - lapack_int lwork, lapack_int* iwork, - lapack_int liwork ); -lapack_int LAPACKE_dspgvd_work( int matrix_order, lapack_int itype, char jobz, - char uplo, lapack_int n, double* ap, double* bp, - double* w, double* z, lapack_int ldz, - double* work, lapack_int lwork, - lapack_int* iwork, lapack_int liwork ); - -lapack_int LAPACKE_sspgvx_work( int matrix_order, lapack_int itype, char jobz, - char range, char uplo, lapack_int n, float* ap, - float* bp, float vl, float vu, lapack_int il, - lapack_int iu, float abstol, lapack_int* m, - float* w, float* z, lapack_int ldz, float* work, - lapack_int* iwork, lapack_int* ifail ); -lapack_int LAPACKE_dspgvx_work( int matrix_order, lapack_int itype, char jobz, - char range, char uplo, lapack_int n, double* ap, - double* bp, double vl, double vu, lapack_int il, - lapack_int iu, double abstol, lapack_int* m, - double* w, double* z, lapack_int ldz, - double* work, lapack_int* iwork, - lapack_int* ifail ); - -lapack_int LAPACKE_ssprfs_work( int matrix_order, char uplo, lapack_int n, - lapack_int nrhs, const float* ap, - const float* afp, const lapack_int* ipiv, - const float* b, lapack_int ldb, float* x, - lapack_int ldx, float* ferr, float* berr, - float* work, lapack_int* iwork ); -lapack_int LAPACKE_dsprfs_work( int matrix_order, char uplo, lapack_int n, - lapack_int nrhs, const double* ap, - const double* afp, const lapack_int* ipiv, - const double* b, lapack_int ldb, double* x, - lapack_int ldx, double* ferr, double* berr, - double* work, lapack_int* iwork ); -lapack_int LAPACKE_csprfs_work( int matrix_order, char uplo, lapack_int n, - lapack_int nrhs, const lapack_complex_float* ap, - const lapack_complex_float* afp, - const lapack_int* ipiv, - const lapack_complex_float* b, lapack_int ldb, - lapack_complex_float* x, lapack_int ldx, - float* ferr, float* berr, - lapack_complex_float* work, float* rwork ); -lapack_int LAPACKE_zsprfs_work( int matrix_order, char uplo, lapack_int n, - lapack_int nrhs, - const lapack_complex_double* ap, - const lapack_complex_double* afp, - const lapack_int* ipiv, - const lapack_complex_double* b, lapack_int ldb, - lapack_complex_double* x, lapack_int ldx, - double* ferr, double* berr, - lapack_complex_double* work, double* rwork ); - -lapack_int LAPACKE_sspsv_work( int matrix_order, char uplo, lapack_int n, - lapack_int nrhs, float* ap, lapack_int* ipiv, - float* b, lapack_int ldb ); -lapack_int LAPACKE_dspsv_work( int matrix_order, char uplo, lapack_int n, - lapack_int nrhs, double* ap, lapack_int* ipiv, - double* b, lapack_int ldb ); -lapack_int LAPACKE_cspsv_work( int matrix_order, char uplo, lapack_int n, - lapack_int nrhs, lapack_complex_float* ap, - lapack_int* ipiv, lapack_complex_float* b, - lapack_int ldb ); -lapack_int LAPACKE_zspsv_work( int matrix_order, char uplo, lapack_int n, - lapack_int nrhs, lapack_complex_double* ap, - lapack_int* ipiv, lapack_complex_double* b, - lapack_int ldb ); - -lapack_int LAPACKE_sspsvx_work( int matrix_order, char fact, char uplo, - lapack_int n, lapack_int nrhs, const float* ap, - float* afp, lapack_int* ipiv, const float* b, - lapack_int ldb, float* x, lapack_int ldx, - float* rcond, float* ferr, float* berr, - float* work, lapack_int* iwork ); -lapack_int LAPACKE_dspsvx_work( int matrix_order, char fact, char uplo, - lapack_int n, lapack_int nrhs, const double* ap, - double* afp, lapack_int* ipiv, const double* b, - lapack_int ldb, double* x, lapack_int ldx, - double* rcond, double* ferr, double* berr, - double* work, lapack_int* iwork ); -lapack_int LAPACKE_cspsvx_work( int matrix_order, char fact, char uplo, - lapack_int n, lapack_int nrhs, - const lapack_complex_float* ap, - lapack_complex_float* afp, lapack_int* ipiv, - const lapack_complex_float* b, lapack_int ldb, - lapack_complex_float* x, lapack_int ldx, - float* rcond, float* ferr, float* berr, - lapack_complex_float* work, float* rwork ); -lapack_int LAPACKE_zspsvx_work( int matrix_order, char fact, char uplo, - lapack_int n, lapack_int nrhs, - const lapack_complex_double* ap, - lapack_complex_double* afp, lapack_int* ipiv, - const lapack_complex_double* b, lapack_int ldb, - lapack_complex_double* x, lapack_int ldx, - double* rcond, double* ferr, double* berr, - lapack_complex_double* work, double* rwork ); - -lapack_int LAPACKE_ssptrd_work( int matrix_order, char uplo, lapack_int n, - float* ap, float* d, float* e, float* tau ); -lapack_int LAPACKE_dsptrd_work( int matrix_order, char uplo, lapack_int n, - double* ap, double* d, double* e, double* tau ); - -lapack_int LAPACKE_ssptrf_work( int matrix_order, char uplo, lapack_int n, - float* ap, lapack_int* ipiv ); -lapack_int LAPACKE_dsptrf_work( int matrix_order, char uplo, lapack_int n, - double* ap, lapack_int* ipiv ); -lapack_int LAPACKE_csptrf_work( int matrix_order, char uplo, lapack_int n, - lapack_complex_float* ap, lapack_int* ipiv ); -lapack_int LAPACKE_zsptrf_work( int matrix_order, char uplo, lapack_int n, - lapack_complex_double* ap, lapack_int* ipiv ); - -lapack_int LAPACKE_ssptri_work( int matrix_order, char uplo, lapack_int n, - float* ap, const lapack_int* ipiv, - float* work ); -lapack_int LAPACKE_dsptri_work( int matrix_order, char uplo, lapack_int n, - double* ap, const lapack_int* ipiv, - double* work ); -lapack_int LAPACKE_csptri_work( int matrix_order, char uplo, lapack_int n, - lapack_complex_float* ap, - const lapack_int* ipiv, - lapack_complex_float* work ); -lapack_int LAPACKE_zsptri_work( int matrix_order, char uplo, lapack_int n, - lapack_complex_double* ap, - const lapack_int* ipiv, - lapack_complex_double* work ); - -lapack_int LAPACKE_ssptrs_work( int matrix_order, char uplo, lapack_int n, - lapack_int nrhs, const float* ap, - const lapack_int* ipiv, float* b, - lapack_int ldb ); -lapack_int LAPACKE_dsptrs_work( int matrix_order, char uplo, lapack_int n, - lapack_int nrhs, const double* ap, - const lapack_int* ipiv, double* b, - lapack_int ldb ); -lapack_int LAPACKE_csptrs_work( int matrix_order, char uplo, lapack_int n, - lapack_int nrhs, const lapack_complex_float* ap, - const lapack_int* ipiv, lapack_complex_float* b, - lapack_int ldb ); -lapack_int LAPACKE_zsptrs_work( int matrix_order, char uplo, lapack_int n, - lapack_int nrhs, - const lapack_complex_double* ap, - const lapack_int* ipiv, - lapack_complex_double* b, lapack_int ldb ); - -lapack_int LAPACKE_sstebz_work( char range, char order, lapack_int n, float vl, - float vu, lapack_int il, lapack_int iu, - float abstol, const float* d, const float* e, - lapack_int* m, lapack_int* nsplit, float* w, - lapack_int* iblock, lapack_int* isplit, - float* work, lapack_int* iwork ); -lapack_int LAPACKE_dstebz_work( char range, char order, lapack_int n, double vl, - double vu, lapack_int il, lapack_int iu, - double abstol, const double* d, const double* e, - lapack_int* m, lapack_int* nsplit, double* w, - lapack_int* iblock, lapack_int* isplit, - double* work, lapack_int* iwork ); - -lapack_int LAPACKE_sstedc_work( int matrix_order, char compz, lapack_int n, - float* d, float* e, float* z, lapack_int ldz, - float* work, lapack_int lwork, - lapack_int* iwork, lapack_int liwork ); -lapack_int LAPACKE_dstedc_work( int matrix_order, char compz, lapack_int n, - double* d, double* e, double* z, lapack_int ldz, - double* work, lapack_int lwork, - lapack_int* iwork, lapack_int liwork ); -lapack_int LAPACKE_cstedc_work( int matrix_order, char compz, lapack_int n, - float* d, float* e, lapack_complex_float* z, - lapack_int ldz, lapack_complex_float* work, - lapack_int lwork, float* rwork, - lapack_int lrwork, lapack_int* iwork, - lapack_int liwork ); -lapack_int LAPACKE_zstedc_work( int matrix_order, char compz, lapack_int n, - double* d, double* e, lapack_complex_double* z, - lapack_int ldz, lapack_complex_double* work, - lapack_int lwork, double* rwork, - lapack_int lrwork, lapack_int* iwork, - lapack_int liwork ); - -lapack_int LAPACKE_sstegr_work( int matrix_order, char jobz, char range, - lapack_int n, float* d, float* e, float vl, - float vu, lapack_int il, lapack_int iu, - float abstol, lapack_int* m, float* w, float* z, - lapack_int ldz, lapack_int* isuppz, float* work, - lapack_int lwork, lapack_int* iwork, - lapack_int liwork ); -lapack_int LAPACKE_dstegr_work( int matrix_order, char jobz, char range, - lapack_int n, double* d, double* e, double vl, - double vu, lapack_int il, lapack_int iu, - double abstol, lapack_int* m, double* w, - double* z, lapack_int ldz, lapack_int* isuppz, - double* work, lapack_int lwork, - lapack_int* iwork, lapack_int liwork ); -lapack_int LAPACKE_cstegr_work( int matrix_order, char jobz, char range, - lapack_int n, float* d, float* e, float vl, - float vu, lapack_int il, lapack_int iu, - float abstol, lapack_int* m, float* w, - lapack_complex_float* z, lapack_int ldz, - lapack_int* isuppz, float* work, - lapack_int lwork, lapack_int* iwork, - lapack_int liwork ); -lapack_int LAPACKE_zstegr_work( int matrix_order, char jobz, char range, - lapack_int n, double* d, double* e, double vl, - double vu, lapack_int il, lapack_int iu, - double abstol, lapack_int* m, double* w, - lapack_complex_double* z, lapack_int ldz, - lapack_int* isuppz, double* work, - lapack_int lwork, lapack_int* iwork, - lapack_int liwork ); - -lapack_int LAPACKE_sstein_work( int matrix_order, lapack_int n, const float* d, - const float* e, lapack_int m, const float* w, - const lapack_int* iblock, - const lapack_int* isplit, float* z, - lapack_int ldz, float* work, lapack_int* iwork, - lapack_int* ifailv ); -lapack_int LAPACKE_dstein_work( int matrix_order, lapack_int n, const double* d, - const double* e, lapack_int m, const double* w, - const lapack_int* iblock, - const lapack_int* isplit, double* z, - lapack_int ldz, double* work, lapack_int* iwork, - lapack_int* ifailv ); -lapack_int LAPACKE_cstein_work( int matrix_order, lapack_int n, const float* d, - const float* e, lapack_int m, const float* w, - const lapack_int* iblock, - const lapack_int* isplit, - lapack_complex_float* z, lapack_int ldz, - float* work, lapack_int* iwork, - lapack_int* ifailv ); -lapack_int LAPACKE_zstein_work( int matrix_order, lapack_int n, const double* d, - const double* e, lapack_int m, const double* w, - const lapack_int* iblock, - const lapack_int* isplit, - lapack_complex_double* z, lapack_int ldz, - double* work, lapack_int* iwork, - lapack_int* ifailv ); - -lapack_int LAPACKE_sstemr_work( int matrix_order, char jobz, char range, - lapack_int n, float* d, float* e, float vl, - float vu, lapack_int il, lapack_int iu, - lapack_int* m, float* w, float* z, - lapack_int ldz, lapack_int nzc, - lapack_int* isuppz, lapack_logical* tryrac, - float* work, lapack_int lwork, - lapack_int* iwork, lapack_int liwork ); -lapack_int LAPACKE_dstemr_work( int matrix_order, char jobz, char range, - lapack_int n, double* d, double* e, double vl, - double vu, lapack_int il, lapack_int iu, - lapack_int* m, double* w, double* z, - lapack_int ldz, lapack_int nzc, - lapack_int* isuppz, lapack_logical* tryrac, - double* work, lapack_int lwork, - lapack_int* iwork, lapack_int liwork ); -lapack_int LAPACKE_cstemr_work( int matrix_order, char jobz, char range, - lapack_int n, float* d, float* e, float vl, - float vu, lapack_int il, lapack_int iu, - lapack_int* m, float* w, - lapack_complex_float* z, lapack_int ldz, - lapack_int nzc, lapack_int* isuppz, - lapack_logical* tryrac, float* work, - lapack_int lwork, lapack_int* iwork, - lapack_int liwork ); -lapack_int LAPACKE_zstemr_work( int matrix_order, char jobz, char range, - lapack_int n, double* d, double* e, double vl, - double vu, lapack_int il, lapack_int iu, - lapack_int* m, double* w, - lapack_complex_double* z, lapack_int ldz, - lapack_int nzc, lapack_int* isuppz, - lapack_logical* tryrac, double* work, - lapack_int lwork, lapack_int* iwork, - lapack_int liwork ); - -lapack_int LAPACKE_ssteqr_work( int matrix_order, char compz, lapack_int n, - float* d, float* e, float* z, lapack_int ldz, - float* work ); -lapack_int LAPACKE_dsteqr_work( int matrix_order, char compz, lapack_int n, - double* d, double* e, double* z, lapack_int ldz, - double* work ); -lapack_int LAPACKE_csteqr_work( int matrix_order, char compz, lapack_int n, - float* d, float* e, lapack_complex_float* z, - lapack_int ldz, float* work ); -lapack_int LAPACKE_zsteqr_work( int matrix_order, char compz, lapack_int n, - double* d, double* e, lapack_complex_double* z, - lapack_int ldz, double* work ); - -lapack_int LAPACKE_ssterf_work( lapack_int n, float* d, float* e ); -lapack_int LAPACKE_dsterf_work( lapack_int n, double* d, double* e ); - -lapack_int LAPACKE_sstev_work( int matrix_order, char jobz, lapack_int n, - float* d, float* e, float* z, lapack_int ldz, - float* work ); -lapack_int LAPACKE_dstev_work( int matrix_order, char jobz, lapack_int n, - double* d, double* e, double* z, lapack_int ldz, - double* work ); - -lapack_int LAPACKE_sstevd_work( int matrix_order, char jobz, lapack_int n, - float* d, float* e, float* z, lapack_int ldz, - float* work, lapack_int lwork, - lapack_int* iwork, lapack_int liwork ); -lapack_int LAPACKE_dstevd_work( int matrix_order, char jobz, lapack_int n, - double* d, double* e, double* z, lapack_int ldz, - double* work, lapack_int lwork, - lapack_int* iwork, lapack_int liwork ); - -lapack_int LAPACKE_sstevr_work( int matrix_order, char jobz, char range, - lapack_int n, float* d, float* e, float vl, - float vu, lapack_int il, lapack_int iu, - float abstol, lapack_int* m, float* w, float* z, - lapack_int ldz, lapack_int* isuppz, float* work, - lapack_int lwork, lapack_int* iwork, - lapack_int liwork ); -lapack_int LAPACKE_dstevr_work( int matrix_order, char jobz, char range, - lapack_int n, double* d, double* e, double vl, - double vu, lapack_int il, lapack_int iu, - double abstol, lapack_int* m, double* w, - double* z, lapack_int ldz, lapack_int* isuppz, - double* work, lapack_int lwork, - lapack_int* iwork, lapack_int liwork ); - -lapack_int LAPACKE_sstevx_work( int matrix_order, char jobz, char range, - lapack_int n, float* d, float* e, float vl, - float vu, lapack_int il, lapack_int iu, - float abstol, lapack_int* m, float* w, float* z, - lapack_int ldz, float* work, lapack_int* iwork, - lapack_int* ifail ); -lapack_int LAPACKE_dstevx_work( int matrix_order, char jobz, char range, - lapack_int n, double* d, double* e, double vl, - double vu, lapack_int il, lapack_int iu, - double abstol, lapack_int* m, double* w, - double* z, lapack_int ldz, double* work, - lapack_int* iwork, lapack_int* ifail ); - -lapack_int LAPACKE_ssycon_work( int matrix_order, char uplo, lapack_int n, - const float* a, lapack_int lda, - const lapack_int* ipiv, float anorm, - float* rcond, float* work, lapack_int* iwork ); -lapack_int LAPACKE_dsycon_work( int matrix_order, char uplo, lapack_int n, - const double* a, lapack_int lda, - const lapack_int* ipiv, double anorm, - double* rcond, double* work, - lapack_int* iwork ); -lapack_int LAPACKE_csycon_work( int matrix_order, char uplo, lapack_int n, - const lapack_complex_float* a, lapack_int lda, - const lapack_int* ipiv, float anorm, - float* rcond, lapack_complex_float* work ); -lapack_int LAPACKE_zsycon_work( int matrix_order, char uplo, lapack_int n, - const lapack_complex_double* a, lapack_int lda, - const lapack_int* ipiv, double anorm, - double* rcond, lapack_complex_double* work ); - -lapack_int LAPACKE_ssyequb_work( int matrix_order, char uplo, lapack_int n, - const float* a, lapack_int lda, float* s, - float* scond, float* amax, float* work ); -lapack_int LAPACKE_dsyequb_work( int matrix_order, char uplo, lapack_int n, - const double* a, lapack_int lda, double* s, - double* scond, double* amax, double* work ); -lapack_int LAPACKE_csyequb_work( int matrix_order, char uplo, lapack_int n, - const lapack_complex_float* a, lapack_int lda, - float* s, float* scond, float* amax, - lapack_complex_float* work ); -lapack_int LAPACKE_zsyequb_work( int matrix_order, char uplo, lapack_int n, - const lapack_complex_double* a, lapack_int lda, - double* s, double* scond, double* amax, - lapack_complex_double* work ); - -lapack_int LAPACKE_ssyev_work( int matrix_order, char jobz, char uplo, - lapack_int n, float* a, lapack_int lda, float* w, - float* work, lapack_int lwork ); -lapack_int LAPACKE_dsyev_work( int matrix_order, char jobz, char uplo, - lapack_int n, double* a, lapack_int lda, - double* w, double* work, lapack_int lwork ); - -lapack_int LAPACKE_ssyevd_work( int matrix_order, char jobz, char uplo, - lapack_int n, float* a, lapack_int lda, - float* w, float* work, lapack_int lwork, - lapack_int* iwork, lapack_int liwork ); -lapack_int LAPACKE_dsyevd_work( int matrix_order, char jobz, char uplo, - lapack_int n, double* a, lapack_int lda, - double* w, double* work, lapack_int lwork, - lapack_int* iwork, lapack_int liwork ); - -lapack_int LAPACKE_ssyevr_work( int matrix_order, char jobz, char range, - char uplo, lapack_int n, float* a, - lapack_int lda, float vl, float vu, - lapack_int il, lapack_int iu, float abstol, - lapack_int* m, float* w, float* z, - lapack_int ldz, lapack_int* isuppz, float* work, - lapack_int lwork, lapack_int* iwork, - lapack_int liwork ); -lapack_int LAPACKE_dsyevr_work( int matrix_order, char jobz, char range, - char uplo, lapack_int n, double* a, - lapack_int lda, double vl, double vu, - lapack_int il, lapack_int iu, double abstol, - lapack_int* m, double* w, double* z, - lapack_int ldz, lapack_int* isuppz, - double* work, lapack_int lwork, - lapack_int* iwork, lapack_int liwork ); - -lapack_int LAPACKE_ssyevx_work( int matrix_order, char jobz, char range, - char uplo, lapack_int n, float* a, - lapack_int lda, float vl, float vu, - lapack_int il, lapack_int iu, float abstol, - lapack_int* m, float* w, float* z, - lapack_int ldz, float* work, lapack_int lwork, - lapack_int* iwork, lapack_int* ifail ); -lapack_int LAPACKE_dsyevx_work( int matrix_order, char jobz, char range, - char uplo, lapack_int n, double* a, - lapack_int lda, double vl, double vu, - lapack_int il, lapack_int iu, double abstol, - lapack_int* m, double* w, double* z, - lapack_int ldz, double* work, lapack_int lwork, - lapack_int* iwork, lapack_int* ifail ); - -lapack_int LAPACKE_ssygst_work( int matrix_order, lapack_int itype, char uplo, - lapack_int n, float* a, lapack_int lda, - const float* b, lapack_int ldb ); -lapack_int LAPACKE_dsygst_work( int matrix_order, lapack_int itype, char uplo, - lapack_int n, double* a, lapack_int lda, - const double* b, lapack_int ldb ); - -lapack_int LAPACKE_ssygv_work( int matrix_order, lapack_int itype, char jobz, - char uplo, lapack_int n, float* a, - lapack_int lda, float* b, lapack_int ldb, - float* w, float* work, lapack_int lwork ); -lapack_int LAPACKE_dsygv_work( int matrix_order, lapack_int itype, char jobz, - char uplo, lapack_int n, double* a, - lapack_int lda, double* b, lapack_int ldb, - double* w, double* work, lapack_int lwork ); - -lapack_int LAPACKE_ssygvd_work( int matrix_order, lapack_int itype, char jobz, - char uplo, lapack_int n, float* a, - lapack_int lda, float* b, lapack_int ldb, - float* w, float* work, lapack_int lwork, - lapack_int* iwork, lapack_int liwork ); -lapack_int LAPACKE_dsygvd_work( int matrix_order, lapack_int itype, char jobz, - char uplo, lapack_int n, double* a, - lapack_int lda, double* b, lapack_int ldb, - double* w, double* work, lapack_int lwork, - lapack_int* iwork, lapack_int liwork ); - -lapack_int LAPACKE_ssygvx_work( int matrix_order, lapack_int itype, char jobz, - char range, char uplo, lapack_int n, float* a, - lapack_int lda, float* b, lapack_int ldb, - float vl, float vu, lapack_int il, - lapack_int iu, float abstol, lapack_int* m, - float* w, float* z, lapack_int ldz, float* work, - lapack_int lwork, lapack_int* iwork, - lapack_int* ifail ); -lapack_int LAPACKE_dsygvx_work( int matrix_order, lapack_int itype, char jobz, - char range, char uplo, lapack_int n, double* a, - lapack_int lda, double* b, lapack_int ldb, - double vl, double vu, lapack_int il, - lapack_int iu, double abstol, lapack_int* m, - double* w, double* z, lapack_int ldz, - double* work, lapack_int lwork, - lapack_int* iwork, lapack_int* ifail ); - -lapack_int LAPACKE_ssyrfs_work( int matrix_order, char uplo, lapack_int n, - lapack_int nrhs, const float* a, lapack_int lda, - const float* af, lapack_int ldaf, - const lapack_int* ipiv, const float* b, - lapack_int ldb, float* x, lapack_int ldx, - float* ferr, float* berr, float* work, - lapack_int* iwork ); -lapack_int LAPACKE_dsyrfs_work( int matrix_order, char uplo, lapack_int n, - lapack_int nrhs, const double* a, - lapack_int lda, const double* af, - lapack_int ldaf, const lapack_int* ipiv, - const double* b, lapack_int ldb, double* x, - lapack_int ldx, double* ferr, double* berr, - double* work, lapack_int* iwork ); -lapack_int LAPACKE_csyrfs_work( int matrix_order, char uplo, lapack_int n, - lapack_int nrhs, const lapack_complex_float* a, - lapack_int lda, const lapack_complex_float* af, - lapack_int ldaf, const lapack_int* ipiv, - const lapack_complex_float* b, lapack_int ldb, - lapack_complex_float* x, lapack_int ldx, - float* ferr, float* berr, - lapack_complex_float* work, float* rwork ); -lapack_int LAPACKE_zsyrfs_work( int matrix_order, char uplo, lapack_int n, - lapack_int nrhs, const lapack_complex_double* a, - lapack_int lda, const lapack_complex_double* af, - lapack_int ldaf, const lapack_int* ipiv, - const lapack_complex_double* b, lapack_int ldb, - lapack_complex_double* x, lapack_int ldx, - double* ferr, double* berr, - lapack_complex_double* work, double* rwork ); - -lapack_int LAPACKE_ssyrfsx_work( int matrix_order, char uplo, char equed, - lapack_int n, lapack_int nrhs, const float* a, - lapack_int lda, const float* af, - lapack_int ldaf, const lapack_int* ipiv, - const float* s, const float* b, lapack_int ldb, - float* x, lapack_int ldx, float* rcond, - float* berr, lapack_int n_err_bnds, - float* err_bnds_norm, float* err_bnds_comp, - lapack_int nparams, float* params, float* work, - lapack_int* iwork ); -lapack_int LAPACKE_dsyrfsx_work( int matrix_order, char uplo, char equed, - lapack_int n, lapack_int nrhs, const double* a, - lapack_int lda, const double* af, - lapack_int ldaf, const lapack_int* ipiv, - const double* s, const double* b, - lapack_int ldb, double* x, lapack_int ldx, - double* rcond, double* berr, - lapack_int n_err_bnds, double* err_bnds_norm, - double* err_bnds_comp, lapack_int nparams, - double* params, double* work, - lapack_int* iwork ); -lapack_int LAPACKE_csyrfsx_work( int matrix_order, char uplo, char equed, - lapack_int n, lapack_int nrhs, - const lapack_complex_float* a, lapack_int lda, - const lapack_complex_float* af, - lapack_int ldaf, const lapack_int* ipiv, - const float* s, const lapack_complex_float* b, - lapack_int ldb, lapack_complex_float* x, - lapack_int ldx, float* rcond, float* berr, - lapack_int n_err_bnds, float* err_bnds_norm, - float* err_bnds_comp, lapack_int nparams, - float* params, lapack_complex_float* work, - float* rwork ); -lapack_int LAPACKE_zsyrfsx_work( int matrix_order, char uplo, char equed, - lapack_int n, lapack_int nrhs, - const lapack_complex_double* a, lapack_int lda, - const lapack_complex_double* af, - lapack_int ldaf, const lapack_int* ipiv, - const double* s, - const lapack_complex_double* b, lapack_int ldb, - lapack_complex_double* x, lapack_int ldx, - double* rcond, double* berr, - lapack_int n_err_bnds, double* err_bnds_norm, - double* err_bnds_comp, lapack_int nparams, - double* params, lapack_complex_double* work, - double* rwork ); - -lapack_int LAPACKE_ssysv_work( int matrix_order, char uplo, lapack_int n, - lapack_int nrhs, float* a, lapack_int lda, - lapack_int* ipiv, float* b, lapack_int ldb, - float* work, lapack_int lwork ); -lapack_int LAPACKE_dsysv_work( int matrix_order, char uplo, lapack_int n, - lapack_int nrhs, double* a, lapack_int lda, - lapack_int* ipiv, double* b, lapack_int ldb, - double* work, lapack_int lwork ); -lapack_int LAPACKE_csysv_work( int matrix_order, char uplo, lapack_int n, - lapack_int nrhs, lapack_complex_float* a, - lapack_int lda, lapack_int* ipiv, - lapack_complex_float* b, lapack_int ldb, - lapack_complex_float* work, lapack_int lwork ); -lapack_int LAPACKE_zsysv_work( int matrix_order, char uplo, lapack_int n, - lapack_int nrhs, lapack_complex_double* a, - lapack_int lda, lapack_int* ipiv, - lapack_complex_double* b, lapack_int ldb, - lapack_complex_double* work, lapack_int lwork ); - -lapack_int LAPACKE_ssysvx_work( int matrix_order, char fact, char uplo, - lapack_int n, lapack_int nrhs, const float* a, - lapack_int lda, float* af, lapack_int ldaf, - lapack_int* ipiv, const float* b, - lapack_int ldb, float* x, lapack_int ldx, - float* rcond, float* ferr, float* berr, - float* work, lapack_int lwork, - lapack_int* iwork ); -lapack_int LAPACKE_dsysvx_work( int matrix_order, char fact, char uplo, - lapack_int n, lapack_int nrhs, const double* a, - lapack_int lda, double* af, lapack_int ldaf, - lapack_int* ipiv, const double* b, - lapack_int ldb, double* x, lapack_int ldx, - double* rcond, double* ferr, double* berr, - double* work, lapack_int lwork, - lapack_int* iwork ); -lapack_int LAPACKE_csysvx_work( int matrix_order, char fact, char uplo, - lapack_int n, lapack_int nrhs, - const lapack_complex_float* a, lapack_int lda, - lapack_complex_float* af, lapack_int ldaf, - lapack_int* ipiv, const lapack_complex_float* b, - lapack_int ldb, lapack_complex_float* x, - lapack_int ldx, float* rcond, float* ferr, - float* berr, lapack_complex_float* work, - lapack_int lwork, float* rwork ); -lapack_int LAPACKE_zsysvx_work( int matrix_order, char fact, char uplo, - lapack_int n, lapack_int nrhs, - const lapack_complex_double* a, lapack_int lda, - lapack_complex_double* af, lapack_int ldaf, - lapack_int* ipiv, - const lapack_complex_double* b, lapack_int ldb, - lapack_complex_double* x, lapack_int ldx, - double* rcond, double* ferr, double* berr, - lapack_complex_double* work, lapack_int lwork, - double* rwork ); - -lapack_int LAPACKE_ssysvxx_work( int matrix_order, char fact, char uplo, - lapack_int n, lapack_int nrhs, float* a, - lapack_int lda, float* af, lapack_int ldaf, - lapack_int* ipiv, char* equed, float* s, - float* b, lapack_int ldb, float* x, - lapack_int ldx, float* rcond, float* rpvgrw, - float* berr, lapack_int n_err_bnds, - float* err_bnds_norm, float* err_bnds_comp, - lapack_int nparams, float* params, float* work, - lapack_int* iwork ); -lapack_int LAPACKE_dsysvxx_work( int matrix_order, char fact, char uplo, - lapack_int n, lapack_int nrhs, double* a, - lapack_int lda, double* af, lapack_int ldaf, - lapack_int* ipiv, char* equed, double* s, - double* b, lapack_int ldb, double* x, - lapack_int ldx, double* rcond, double* rpvgrw, - double* berr, lapack_int n_err_bnds, - double* err_bnds_norm, double* err_bnds_comp, - lapack_int nparams, double* params, - double* work, lapack_int* iwork ); -lapack_int LAPACKE_csysvxx_work( int matrix_order, char fact, char uplo, - lapack_int n, lapack_int nrhs, - lapack_complex_float* a, lapack_int lda, - lapack_complex_float* af, lapack_int ldaf, - lapack_int* ipiv, char* equed, float* s, - lapack_complex_float* b, lapack_int ldb, - lapack_complex_float* x, lapack_int ldx, - float* rcond, float* rpvgrw, float* berr, - lapack_int n_err_bnds, float* err_bnds_norm, - float* err_bnds_comp, lapack_int nparams, - float* params, lapack_complex_float* work, - float* rwork ); -lapack_int LAPACKE_zsysvxx_work( int matrix_order, char fact, char uplo, - lapack_int n, lapack_int nrhs, - lapack_complex_double* a, lapack_int lda, - lapack_complex_double* af, lapack_int ldaf, - lapack_int* ipiv, char* equed, double* s, - lapack_complex_double* b, lapack_int ldb, - lapack_complex_double* x, lapack_int ldx, - double* rcond, double* rpvgrw, double* berr, - lapack_int n_err_bnds, double* err_bnds_norm, - double* err_bnds_comp, lapack_int nparams, - double* params, lapack_complex_double* work, - double* rwork ); - -lapack_int LAPACKE_ssytrd_work( int matrix_order, char uplo, lapack_int n, - float* a, lapack_int lda, float* d, float* e, - float* tau, float* work, lapack_int lwork ); -lapack_int LAPACKE_dsytrd_work( int matrix_order, char uplo, lapack_int n, - double* a, lapack_int lda, double* d, double* e, - double* tau, double* work, lapack_int lwork ); - -lapack_int LAPACKE_ssytrf_work( int matrix_order, char uplo, lapack_int n, - float* a, lapack_int lda, lapack_int* ipiv, - float* work, lapack_int lwork ); -lapack_int LAPACKE_dsytrf_work( int matrix_order, char uplo, lapack_int n, - double* a, lapack_int lda, lapack_int* ipiv, - double* work, lapack_int lwork ); -lapack_int LAPACKE_csytrf_work( int matrix_order, char uplo, lapack_int n, - lapack_complex_float* a, lapack_int lda, - lapack_int* ipiv, lapack_complex_float* work, - lapack_int lwork ); -lapack_int LAPACKE_zsytrf_work( int matrix_order, char uplo, lapack_int n, - lapack_complex_double* a, lapack_int lda, - lapack_int* ipiv, lapack_complex_double* work, - lapack_int lwork ); - -lapack_int LAPACKE_ssytri_work( int matrix_order, char uplo, lapack_int n, - float* a, lapack_int lda, - const lapack_int* ipiv, float* work ); -lapack_int LAPACKE_dsytri_work( int matrix_order, char uplo, lapack_int n, - double* a, lapack_int lda, - const lapack_int* ipiv, double* work ); -lapack_int LAPACKE_csytri_work( int matrix_order, char uplo, lapack_int n, - lapack_complex_float* a, lapack_int lda, - const lapack_int* ipiv, - lapack_complex_float* work ); -lapack_int LAPACKE_zsytri_work( int matrix_order, char uplo, lapack_int n, - lapack_complex_double* a, lapack_int lda, - const lapack_int* ipiv, - lapack_complex_double* work ); - -lapack_int LAPACKE_ssytrs_work( int matrix_order, char uplo, lapack_int n, - lapack_int nrhs, const float* a, lapack_int lda, - const lapack_int* ipiv, float* b, - lapack_int ldb ); -lapack_int LAPACKE_dsytrs_work( int matrix_order, char uplo, lapack_int n, - lapack_int nrhs, const double* a, - lapack_int lda, const lapack_int* ipiv, - double* b, lapack_int ldb ); -lapack_int LAPACKE_csytrs_work( int matrix_order, char uplo, lapack_int n, - lapack_int nrhs, const lapack_complex_float* a, - lapack_int lda, const lapack_int* ipiv, - lapack_complex_float* b, lapack_int ldb ); -lapack_int LAPACKE_zsytrs_work( int matrix_order, char uplo, lapack_int n, - lapack_int nrhs, const lapack_complex_double* a, - lapack_int lda, const lapack_int* ipiv, - lapack_complex_double* b, lapack_int ldb ); - -lapack_int LAPACKE_stbcon_work( int matrix_order, char norm, char uplo, - char diag, lapack_int n, lapack_int kd, - const float* ab, lapack_int ldab, float* rcond, - float* work, lapack_int* iwork ); -lapack_int LAPACKE_dtbcon_work( int matrix_order, char norm, char uplo, - char diag, lapack_int n, lapack_int kd, - const double* ab, lapack_int ldab, - double* rcond, double* work, - lapack_int* iwork ); -lapack_int LAPACKE_ctbcon_work( int matrix_order, char norm, char uplo, - char diag, lapack_int n, lapack_int kd, - const lapack_complex_float* ab, lapack_int ldab, - float* rcond, lapack_complex_float* work, - float* rwork ); -lapack_int LAPACKE_ztbcon_work( int matrix_order, char norm, char uplo, - char diag, lapack_int n, lapack_int kd, - const lapack_complex_double* ab, - lapack_int ldab, double* rcond, - lapack_complex_double* work, double* rwork ); - -lapack_int LAPACKE_stbrfs_work( int matrix_order, char uplo, char trans, - char diag, lapack_int n, lapack_int kd, - lapack_int nrhs, const float* ab, - lapack_int ldab, const float* b, lapack_int ldb, - const float* x, lapack_int ldx, float* ferr, - float* berr, float* work, lapack_int* iwork ); -lapack_int LAPACKE_dtbrfs_work( int matrix_order, char uplo, char trans, - char diag, lapack_int n, lapack_int kd, - lapack_int nrhs, const double* ab, - lapack_int ldab, const double* b, - lapack_int ldb, const double* x, lapack_int ldx, - double* ferr, double* berr, double* work, - lapack_int* iwork ); -lapack_int LAPACKE_ctbrfs_work( int matrix_order, char uplo, char trans, - char diag, lapack_int n, lapack_int kd, - lapack_int nrhs, const lapack_complex_float* ab, - lapack_int ldab, const lapack_complex_float* b, - lapack_int ldb, const lapack_complex_float* x, - lapack_int ldx, float* ferr, float* berr, - lapack_complex_float* work, float* rwork ); -lapack_int LAPACKE_ztbrfs_work( int matrix_order, char uplo, char trans, - char diag, lapack_int n, lapack_int kd, - lapack_int nrhs, - const lapack_complex_double* ab, - lapack_int ldab, const lapack_complex_double* b, - lapack_int ldb, const lapack_complex_double* x, - lapack_int ldx, double* ferr, double* berr, - lapack_complex_double* work, double* rwork ); - -lapack_int LAPACKE_stbtrs_work( int matrix_order, char uplo, char trans, - char diag, lapack_int n, lapack_int kd, - lapack_int nrhs, const float* ab, - lapack_int ldab, float* b, lapack_int ldb ); -lapack_int LAPACKE_dtbtrs_work( int matrix_order, char uplo, char trans, - char diag, lapack_int n, lapack_int kd, - lapack_int nrhs, const double* ab, - lapack_int ldab, double* b, lapack_int ldb ); -lapack_int LAPACKE_ctbtrs_work( int matrix_order, char uplo, char trans, - char diag, lapack_int n, lapack_int kd, - lapack_int nrhs, const lapack_complex_float* ab, - lapack_int ldab, lapack_complex_float* b, - lapack_int ldb ); -lapack_int LAPACKE_ztbtrs_work( int matrix_order, char uplo, char trans, - char diag, lapack_int n, lapack_int kd, - lapack_int nrhs, - const lapack_complex_double* ab, - lapack_int ldab, lapack_complex_double* b, - lapack_int ldb ); - -lapack_int LAPACKE_stfsm_work( int matrix_order, char transr, char side, - char uplo, char trans, char diag, lapack_int m, - lapack_int n, float alpha, const float* a, - float* b, lapack_int ldb ); -lapack_int LAPACKE_dtfsm_work( int matrix_order, char transr, char side, - char uplo, char trans, char diag, lapack_int m, - lapack_int n, double alpha, const double* a, - double* b, lapack_int ldb ); -lapack_int LAPACKE_ctfsm_work( int matrix_order, char transr, char side, - char uplo, char trans, char diag, lapack_int m, - lapack_int n, lapack_complex_float alpha, - const lapack_complex_float* a, - lapack_complex_float* b, lapack_int ldb ); -lapack_int LAPACKE_ztfsm_work( int matrix_order, char transr, char side, - char uplo, char trans, char diag, lapack_int m, - lapack_int n, lapack_complex_double alpha, - const lapack_complex_double* a, - lapack_complex_double* b, lapack_int ldb ); - -lapack_int LAPACKE_stftri_work( int matrix_order, char transr, char uplo, - char diag, lapack_int n, float* a ); -lapack_int LAPACKE_dtftri_work( int matrix_order, char transr, char uplo, - char diag, lapack_int n, double* a ); -lapack_int LAPACKE_ctftri_work( int matrix_order, char transr, char uplo, - char diag, lapack_int n, - lapack_complex_float* a ); -lapack_int LAPACKE_ztftri_work( int matrix_order, char transr, char uplo, - char diag, lapack_int n, - lapack_complex_double* a ); - -lapack_int LAPACKE_stfttp_work( int matrix_order, char transr, char uplo, - lapack_int n, const float* arf, float* ap ); -lapack_int LAPACKE_dtfttp_work( int matrix_order, char transr, char uplo, - lapack_int n, const double* arf, double* ap ); -lapack_int LAPACKE_ctfttp_work( int matrix_order, char transr, char uplo, - lapack_int n, const lapack_complex_float* arf, - lapack_complex_float* ap ); -lapack_int LAPACKE_ztfttp_work( int matrix_order, char transr, char uplo, - lapack_int n, const lapack_complex_double* arf, - lapack_complex_double* ap ); - -lapack_int LAPACKE_stfttr_work( int matrix_order, char transr, char uplo, - lapack_int n, const float* arf, float* a, - lapack_int lda ); -lapack_int LAPACKE_dtfttr_work( int matrix_order, char transr, char uplo, - lapack_int n, const double* arf, double* a, - lapack_int lda ); -lapack_int LAPACKE_ctfttr_work( int matrix_order, char transr, char uplo, - lapack_int n, const lapack_complex_float* arf, - lapack_complex_float* a, lapack_int lda ); -lapack_int LAPACKE_ztfttr_work( int matrix_order, char transr, char uplo, - lapack_int n, const lapack_complex_double* arf, - lapack_complex_double* a, lapack_int lda ); - -lapack_int LAPACKE_stgevc_work( int matrix_order, char side, char howmny, - const lapack_logical* select, lapack_int n, - const float* s, lapack_int lds, const float* p, - lapack_int ldp, float* vl, lapack_int ldvl, - float* vr, lapack_int ldvr, lapack_int mm, - lapack_int* m, float* work ); -lapack_int LAPACKE_dtgevc_work( int matrix_order, char side, char howmny, - const lapack_logical* select, lapack_int n, - const double* s, lapack_int lds, - const double* p, lapack_int ldp, double* vl, - lapack_int ldvl, double* vr, lapack_int ldvr, - lapack_int mm, lapack_int* m, double* work ); -lapack_int LAPACKE_ctgevc_work( int matrix_order, char side, char howmny, - const lapack_logical* select, lapack_int n, - const lapack_complex_float* s, lapack_int lds, - const lapack_complex_float* p, lapack_int ldp, - lapack_complex_float* vl, lapack_int ldvl, - lapack_complex_float* vr, lapack_int ldvr, - lapack_int mm, lapack_int* m, - lapack_complex_float* work, float* rwork ); -lapack_int LAPACKE_ztgevc_work( int matrix_order, char side, char howmny, - const lapack_logical* select, lapack_int n, - const lapack_complex_double* s, lapack_int lds, - const lapack_complex_double* p, lapack_int ldp, - lapack_complex_double* vl, lapack_int ldvl, - lapack_complex_double* vr, lapack_int ldvr, - lapack_int mm, lapack_int* m, - lapack_complex_double* work, double* rwork ); - -lapack_int LAPACKE_stgexc_work( int matrix_order, lapack_logical wantq, - lapack_logical wantz, lapack_int n, float* a, - lapack_int lda, float* b, lapack_int ldb, - float* q, lapack_int ldq, float* z, - lapack_int ldz, lapack_int* ifst, - lapack_int* ilst, float* work, - lapack_int lwork ); -lapack_int LAPACKE_dtgexc_work( int matrix_order, lapack_logical wantq, - lapack_logical wantz, lapack_int n, double* a, - lapack_int lda, double* b, lapack_int ldb, - double* q, lapack_int ldq, double* z, - lapack_int ldz, lapack_int* ifst, - lapack_int* ilst, double* work, - lapack_int lwork ); -lapack_int LAPACKE_ctgexc_work( int matrix_order, lapack_logical wantq, - lapack_logical wantz, lapack_int n, - lapack_complex_float* a, lapack_int lda, - lapack_complex_float* b, lapack_int ldb, - lapack_complex_float* q, lapack_int ldq, - lapack_complex_float* z, lapack_int ldz, - lapack_int ifst, lapack_int ilst ); -lapack_int LAPACKE_ztgexc_work( int matrix_order, lapack_logical wantq, - lapack_logical wantz, lapack_int n, - lapack_complex_double* a, lapack_int lda, - lapack_complex_double* b, lapack_int ldb, - lapack_complex_double* q, lapack_int ldq, - lapack_complex_double* z, lapack_int ldz, - lapack_int ifst, lapack_int ilst ); - -lapack_int LAPACKE_stgsen_work( int matrix_order, lapack_int ijob, - lapack_logical wantq, lapack_logical wantz, - const lapack_logical* select, lapack_int n, - float* a, lapack_int lda, float* b, - lapack_int ldb, float* alphar, float* alphai, - float* beta, float* q, lapack_int ldq, float* z, - lapack_int ldz, lapack_int* m, float* pl, - float* pr, float* dif, float* work, - lapack_int lwork, lapack_int* iwork, - lapack_int liwork ); -lapack_int LAPACKE_dtgsen_work( int matrix_order, lapack_int ijob, - lapack_logical wantq, lapack_logical wantz, - const lapack_logical* select, lapack_int n, - double* a, lapack_int lda, double* b, - lapack_int ldb, double* alphar, double* alphai, - double* beta, double* q, lapack_int ldq, - double* z, lapack_int ldz, lapack_int* m, - double* pl, double* pr, double* dif, - double* work, lapack_int lwork, - lapack_int* iwork, lapack_int liwork ); -lapack_int LAPACKE_ctgsen_work( int matrix_order, lapack_int ijob, - lapack_logical wantq, lapack_logical wantz, - const lapack_logical* select, lapack_int n, - lapack_complex_float* a, lapack_int lda, - lapack_complex_float* b, lapack_int ldb, - lapack_complex_float* alpha, - lapack_complex_float* beta, - lapack_complex_float* q, lapack_int ldq, - lapack_complex_float* z, lapack_int ldz, - lapack_int* m, float* pl, float* pr, float* dif, - lapack_complex_float* work, lapack_int lwork, - lapack_int* iwork, lapack_int liwork ); -lapack_int LAPACKE_ztgsen_work( int matrix_order, lapack_int ijob, - lapack_logical wantq, lapack_logical wantz, - const lapack_logical* select, lapack_int n, - lapack_complex_double* a, lapack_int lda, - lapack_complex_double* b, lapack_int ldb, - lapack_complex_double* alpha, - lapack_complex_double* beta, - lapack_complex_double* q, lapack_int ldq, - lapack_complex_double* z, lapack_int ldz, - lapack_int* m, double* pl, double* pr, - double* dif, lapack_complex_double* work, - lapack_int lwork, lapack_int* iwork, - lapack_int liwork ); - -lapack_int LAPACKE_stgsja_work( int matrix_order, char jobu, char jobv, - char jobq, lapack_int m, lapack_int p, - lapack_int n, lapack_int k, lapack_int l, - float* a, lapack_int lda, float* b, - lapack_int ldb, float tola, float tolb, - float* alpha, float* beta, float* u, - lapack_int ldu, float* v, lapack_int ldv, - float* q, lapack_int ldq, float* work, - lapack_int* ncycle ); -lapack_int LAPACKE_dtgsja_work( int matrix_order, char jobu, char jobv, - char jobq, lapack_int m, lapack_int p, - lapack_int n, lapack_int k, lapack_int l, - double* a, lapack_int lda, double* b, - lapack_int ldb, double tola, double tolb, - double* alpha, double* beta, double* u, - lapack_int ldu, double* v, lapack_int ldv, - double* q, lapack_int ldq, double* work, - lapack_int* ncycle ); -lapack_int LAPACKE_ctgsja_work( int matrix_order, char jobu, char jobv, - char jobq, lapack_int m, lapack_int p, - lapack_int n, lapack_int k, lapack_int l, - lapack_complex_float* a, lapack_int lda, - lapack_complex_float* b, lapack_int ldb, - float tola, float tolb, float* alpha, - float* beta, lapack_complex_float* u, - lapack_int ldu, lapack_complex_float* v, - lapack_int ldv, lapack_complex_float* q, - lapack_int ldq, lapack_complex_float* work, - lapack_int* ncycle ); -lapack_int LAPACKE_ztgsja_work( int matrix_order, char jobu, char jobv, - char jobq, lapack_int m, lapack_int p, - lapack_int n, lapack_int k, lapack_int l, - lapack_complex_double* a, lapack_int lda, - lapack_complex_double* b, lapack_int ldb, - double tola, double tolb, double* alpha, - double* beta, lapack_complex_double* u, - lapack_int ldu, lapack_complex_double* v, - lapack_int ldv, lapack_complex_double* q, - lapack_int ldq, lapack_complex_double* work, - lapack_int* ncycle ); - -lapack_int LAPACKE_stgsna_work( int matrix_order, char job, char howmny, - const lapack_logical* select, lapack_int n, - const float* a, lapack_int lda, const float* b, - lapack_int ldb, const float* vl, - lapack_int ldvl, const float* vr, - lapack_int ldvr, float* s, float* dif, - lapack_int mm, lapack_int* m, float* work, - lapack_int lwork, lapack_int* iwork ); -lapack_int LAPACKE_dtgsna_work( int matrix_order, char job, char howmny, - const lapack_logical* select, lapack_int n, - const double* a, lapack_int lda, - const double* b, lapack_int ldb, - const double* vl, lapack_int ldvl, - const double* vr, lapack_int ldvr, double* s, - double* dif, lapack_int mm, lapack_int* m, - double* work, lapack_int lwork, - lapack_int* iwork ); -lapack_int LAPACKE_ctgsna_work( int matrix_order, char job, char howmny, - const lapack_logical* select, lapack_int n, - const lapack_complex_float* a, lapack_int lda, - const lapack_complex_float* b, lapack_int ldb, - const lapack_complex_float* vl, lapack_int ldvl, - const lapack_complex_float* vr, lapack_int ldvr, - float* s, float* dif, lapack_int mm, - lapack_int* m, lapack_complex_float* work, - lapack_int lwork, lapack_int* iwork ); -lapack_int LAPACKE_ztgsna_work( int matrix_order, char job, char howmny, - const lapack_logical* select, lapack_int n, - const lapack_complex_double* a, lapack_int lda, - const lapack_complex_double* b, lapack_int ldb, - const lapack_complex_double* vl, - lapack_int ldvl, - const lapack_complex_double* vr, - lapack_int ldvr, double* s, double* dif, - lapack_int mm, lapack_int* m, - lapack_complex_double* work, lapack_int lwork, - lapack_int* iwork ); - -lapack_int LAPACKE_stgsyl_work( int matrix_order, char trans, lapack_int ijob, - lapack_int m, lapack_int n, const float* a, - lapack_int lda, const float* b, lapack_int ldb, - float* c, lapack_int ldc, const float* d, - lapack_int ldd, const float* e, lapack_int lde, - float* f, lapack_int ldf, float* scale, - float* dif, float* work, lapack_int lwork, - lapack_int* iwork ); -lapack_int LAPACKE_dtgsyl_work( int matrix_order, char trans, lapack_int ijob, - lapack_int m, lapack_int n, const double* a, - lapack_int lda, const double* b, lapack_int ldb, - double* c, lapack_int ldc, const double* d, - lapack_int ldd, const double* e, lapack_int lde, - double* f, lapack_int ldf, double* scale, - double* dif, double* work, lapack_int lwork, - lapack_int* iwork ); -lapack_int LAPACKE_ctgsyl_work( int matrix_order, char trans, lapack_int ijob, - lapack_int m, lapack_int n, - const lapack_complex_float* a, lapack_int lda, - const lapack_complex_float* b, lapack_int ldb, - lapack_complex_float* c, lapack_int ldc, - const lapack_complex_float* d, lapack_int ldd, - const lapack_complex_float* e, lapack_int lde, - lapack_complex_float* f, lapack_int ldf, - float* scale, float* dif, - lapack_complex_float* work, lapack_int lwork, - lapack_int* iwork ); -lapack_int LAPACKE_ztgsyl_work( int matrix_order, char trans, lapack_int ijob, - lapack_int m, lapack_int n, - const lapack_complex_double* a, lapack_int lda, - const lapack_complex_double* b, lapack_int ldb, - lapack_complex_double* c, lapack_int ldc, - const lapack_complex_double* d, lapack_int ldd, - const lapack_complex_double* e, lapack_int lde, - lapack_complex_double* f, lapack_int ldf, - double* scale, double* dif, - lapack_complex_double* work, lapack_int lwork, - lapack_int* iwork ); - -lapack_int LAPACKE_stpcon_work( int matrix_order, char norm, char uplo, - char diag, lapack_int n, const float* ap, - float* rcond, float* work, lapack_int* iwork ); -lapack_int LAPACKE_dtpcon_work( int matrix_order, char norm, char uplo, - char diag, lapack_int n, const double* ap, - double* rcond, double* work, - lapack_int* iwork ); -lapack_int LAPACKE_ctpcon_work( int matrix_order, char norm, char uplo, - char diag, lapack_int n, - const lapack_complex_float* ap, float* rcond, - lapack_complex_float* work, float* rwork ); -lapack_int LAPACKE_ztpcon_work( int matrix_order, char norm, char uplo, - char diag, lapack_int n, - const lapack_complex_double* ap, double* rcond, - lapack_complex_double* work, double* rwork ); - -lapack_int LAPACKE_stprfs_work( int matrix_order, char uplo, char trans, - char diag, lapack_int n, lapack_int nrhs, - const float* ap, const float* b, lapack_int ldb, - const float* x, lapack_int ldx, float* ferr, - float* berr, float* work, lapack_int* iwork ); -lapack_int LAPACKE_dtprfs_work( int matrix_order, char uplo, char trans, - char diag, lapack_int n, lapack_int nrhs, - const double* ap, const double* b, - lapack_int ldb, const double* x, lapack_int ldx, - double* ferr, double* berr, double* work, - lapack_int* iwork ); -lapack_int LAPACKE_ctprfs_work( int matrix_order, char uplo, char trans, - char diag, lapack_int n, lapack_int nrhs, - const lapack_complex_float* ap, - const lapack_complex_float* b, lapack_int ldb, - const lapack_complex_float* x, lapack_int ldx, - float* ferr, float* berr, - lapack_complex_float* work, float* rwork ); -lapack_int LAPACKE_ztprfs_work( int matrix_order, char uplo, char trans, - char diag, lapack_int n, lapack_int nrhs, - const lapack_complex_double* ap, - const lapack_complex_double* b, lapack_int ldb, - const lapack_complex_double* x, lapack_int ldx, - double* ferr, double* berr, - lapack_complex_double* work, double* rwork ); - -lapack_int LAPACKE_stptri_work( int matrix_order, char uplo, char diag, - lapack_int n, float* ap ); -lapack_int LAPACKE_dtptri_work( int matrix_order, char uplo, char diag, - lapack_int n, double* ap ); -lapack_int LAPACKE_ctptri_work( int matrix_order, char uplo, char diag, - lapack_int n, lapack_complex_float* ap ); -lapack_int LAPACKE_ztptri_work( int matrix_order, char uplo, char diag, - lapack_int n, lapack_complex_double* ap ); - -lapack_int LAPACKE_stptrs_work( int matrix_order, char uplo, char trans, - char diag, lapack_int n, lapack_int nrhs, - const float* ap, float* b, lapack_int ldb ); -lapack_int LAPACKE_dtptrs_work( int matrix_order, char uplo, char trans, - char diag, lapack_int n, lapack_int nrhs, - const double* ap, double* b, lapack_int ldb ); -lapack_int LAPACKE_ctptrs_work( int matrix_order, char uplo, char trans, - char diag, lapack_int n, lapack_int nrhs, - const lapack_complex_float* ap, - lapack_complex_float* b, lapack_int ldb ); -lapack_int LAPACKE_ztptrs_work( int matrix_order, char uplo, char trans, - char diag, lapack_int n, lapack_int nrhs, - const lapack_complex_double* ap, - lapack_complex_double* b, lapack_int ldb ); - -lapack_int LAPACKE_stpttf_work( int matrix_order, char transr, char uplo, - lapack_int n, const float* ap, float* arf ); -lapack_int LAPACKE_dtpttf_work( int matrix_order, char transr, char uplo, - lapack_int n, const double* ap, double* arf ); -lapack_int LAPACKE_ctpttf_work( int matrix_order, char transr, char uplo, - lapack_int n, const lapack_complex_float* ap, - lapack_complex_float* arf ); -lapack_int LAPACKE_ztpttf_work( int matrix_order, char transr, char uplo, - lapack_int n, const lapack_complex_double* ap, - lapack_complex_double* arf ); - -lapack_int LAPACKE_stpttr_work( int matrix_order, char uplo, lapack_int n, - const float* ap, float* a, lapack_int lda ); -lapack_int LAPACKE_dtpttr_work( int matrix_order, char uplo, lapack_int n, - const double* ap, double* a, lapack_int lda ); -lapack_int LAPACKE_ctpttr_work( int matrix_order, char uplo, lapack_int n, - const lapack_complex_float* ap, - lapack_complex_float* a, lapack_int lda ); -lapack_int LAPACKE_ztpttr_work( int matrix_order, char uplo, lapack_int n, - const lapack_complex_double* ap, - lapack_complex_double* a, lapack_int lda ); - -lapack_int LAPACKE_strcon_work( int matrix_order, char norm, char uplo, - char diag, lapack_int n, const float* a, - lapack_int lda, float* rcond, float* work, - lapack_int* iwork ); -lapack_int LAPACKE_dtrcon_work( int matrix_order, char norm, char uplo, - char diag, lapack_int n, const double* a, - lapack_int lda, double* rcond, double* work, - lapack_int* iwork ); -lapack_int LAPACKE_ctrcon_work( int matrix_order, char norm, char uplo, - char diag, lapack_int n, - const lapack_complex_float* a, lapack_int lda, - float* rcond, lapack_complex_float* work, - float* rwork ); -lapack_int LAPACKE_ztrcon_work( int matrix_order, char norm, char uplo, - char diag, lapack_int n, - const lapack_complex_double* a, lapack_int lda, - double* rcond, lapack_complex_double* work, - double* rwork ); - -lapack_int LAPACKE_strevc_work( int matrix_order, char side, char howmny, - lapack_logical* select, lapack_int n, - const float* t, lapack_int ldt, float* vl, - lapack_int ldvl, float* vr, lapack_int ldvr, - lapack_int mm, lapack_int* m, float* work ); -lapack_int LAPACKE_dtrevc_work( int matrix_order, char side, char howmny, - lapack_logical* select, lapack_int n, - const double* t, lapack_int ldt, double* vl, - lapack_int ldvl, double* vr, lapack_int ldvr, - lapack_int mm, lapack_int* m, double* work ); -lapack_int LAPACKE_ctrevc_work( int matrix_order, char side, char howmny, - const lapack_logical* select, lapack_int n, - lapack_complex_float* t, lapack_int ldt, - lapack_complex_float* vl, lapack_int ldvl, - lapack_complex_float* vr, lapack_int ldvr, - lapack_int mm, lapack_int* m, - lapack_complex_float* work, float* rwork ); -lapack_int LAPACKE_ztrevc_work( int matrix_order, char side, char howmny, - const lapack_logical* select, lapack_int n, - lapack_complex_double* t, lapack_int ldt, - lapack_complex_double* vl, lapack_int ldvl, - lapack_complex_double* vr, lapack_int ldvr, - lapack_int mm, lapack_int* m, - lapack_complex_double* work, double* rwork ); - -lapack_int LAPACKE_strexc_work( int matrix_order, char compq, lapack_int n, - float* t, lapack_int ldt, float* q, - lapack_int ldq, lapack_int* ifst, - lapack_int* ilst, float* work ); -lapack_int LAPACKE_dtrexc_work( int matrix_order, char compq, lapack_int n, - double* t, lapack_int ldt, double* q, - lapack_int ldq, lapack_int* ifst, - lapack_int* ilst, double* work ); -lapack_int LAPACKE_ctrexc_work( int matrix_order, char compq, lapack_int n, - lapack_complex_float* t, lapack_int ldt, - lapack_complex_float* q, lapack_int ldq, - lapack_int ifst, lapack_int ilst ); -lapack_int LAPACKE_ztrexc_work( int matrix_order, char compq, lapack_int n, - lapack_complex_double* t, lapack_int ldt, - lapack_complex_double* q, lapack_int ldq, - lapack_int ifst, lapack_int ilst ); - -lapack_int LAPACKE_strrfs_work( int matrix_order, char uplo, char trans, - char diag, lapack_int n, lapack_int nrhs, - const float* a, lapack_int lda, const float* b, - lapack_int ldb, const float* x, lapack_int ldx, - float* ferr, float* berr, float* work, - lapack_int* iwork ); -lapack_int LAPACKE_dtrrfs_work( int matrix_order, char uplo, char trans, - char diag, lapack_int n, lapack_int nrhs, - const double* a, lapack_int lda, - const double* b, lapack_int ldb, - const double* x, lapack_int ldx, double* ferr, - double* berr, double* work, lapack_int* iwork ); -lapack_int LAPACKE_ctrrfs_work( int matrix_order, char uplo, char trans, - char diag, lapack_int n, lapack_int nrhs, - const lapack_complex_float* a, lapack_int lda, - const lapack_complex_float* b, lapack_int ldb, - const lapack_complex_float* x, lapack_int ldx, - float* ferr, float* berr, - lapack_complex_float* work, float* rwork ); -lapack_int LAPACKE_ztrrfs_work( int matrix_order, char uplo, char trans, - char diag, lapack_int n, lapack_int nrhs, - const lapack_complex_double* a, lapack_int lda, - const lapack_complex_double* b, lapack_int ldb, - const lapack_complex_double* x, lapack_int ldx, - double* ferr, double* berr, - lapack_complex_double* work, double* rwork ); - -lapack_int LAPACKE_strsen_work( int matrix_order, char job, char compq, - const lapack_logical* select, lapack_int n, - float* t, lapack_int ldt, float* q, - lapack_int ldq, float* wr, float* wi, - lapack_int* m, float* s, float* sep, - float* work, lapack_int lwork, - lapack_int* iwork, lapack_int liwork ); -lapack_int LAPACKE_dtrsen_work( int matrix_order, char job, char compq, - const lapack_logical* select, lapack_int n, - double* t, lapack_int ldt, double* q, - lapack_int ldq, double* wr, double* wi, - lapack_int* m, double* s, double* sep, - double* work, lapack_int lwork, - lapack_int* iwork, lapack_int liwork ); -lapack_int LAPACKE_ctrsen_work( int matrix_order, char job, char compq, - const lapack_logical* select, lapack_int n, - lapack_complex_float* t, lapack_int ldt, - lapack_complex_float* q, lapack_int ldq, - lapack_complex_float* w, lapack_int* m, - float* s, float* sep, - lapack_complex_float* work, lapack_int lwork ); -lapack_int LAPACKE_ztrsen_work( int matrix_order, char job, char compq, - const lapack_logical* select, lapack_int n, - lapack_complex_double* t, lapack_int ldt, - lapack_complex_double* q, lapack_int ldq, - lapack_complex_double* w, lapack_int* m, - double* s, double* sep, - lapack_complex_double* work, lapack_int lwork ); - -lapack_int LAPACKE_strsna_work( int matrix_order, char job, char howmny, - const lapack_logical* select, lapack_int n, - const float* t, lapack_int ldt, const float* vl, - lapack_int ldvl, const float* vr, - lapack_int ldvr, float* s, float* sep, - lapack_int mm, lapack_int* m, float* work, - lapack_int ldwork, lapack_int* iwork ); -lapack_int LAPACKE_dtrsna_work( int matrix_order, char job, char howmny, - const lapack_logical* select, lapack_int n, - const double* t, lapack_int ldt, - const double* vl, lapack_int ldvl, - const double* vr, lapack_int ldvr, double* s, - double* sep, lapack_int mm, lapack_int* m, - double* work, lapack_int ldwork, - lapack_int* iwork ); -lapack_int LAPACKE_ctrsna_work( int matrix_order, char job, char howmny, - const lapack_logical* select, lapack_int n, - const lapack_complex_float* t, lapack_int ldt, - const lapack_complex_float* vl, lapack_int ldvl, - const lapack_complex_float* vr, lapack_int ldvr, - float* s, float* sep, lapack_int mm, - lapack_int* m, lapack_complex_float* work, - lapack_int ldwork, float* rwork ); -lapack_int LAPACKE_ztrsna_work( int matrix_order, char job, char howmny, - const lapack_logical* select, lapack_int n, - const lapack_complex_double* t, lapack_int ldt, - const lapack_complex_double* vl, - lapack_int ldvl, - const lapack_complex_double* vr, - lapack_int ldvr, double* s, double* sep, - lapack_int mm, lapack_int* m, - lapack_complex_double* work, lapack_int ldwork, - double* rwork ); - -lapack_int LAPACKE_strsyl_work( int matrix_order, char trana, char tranb, - lapack_int isgn, lapack_int m, lapack_int n, - const float* a, lapack_int lda, const float* b, - lapack_int ldb, float* c, lapack_int ldc, - float* scale ); -lapack_int LAPACKE_dtrsyl_work( int matrix_order, char trana, char tranb, - lapack_int isgn, lapack_int m, lapack_int n, - const double* a, lapack_int lda, - const double* b, lapack_int ldb, double* c, - lapack_int ldc, double* scale ); -lapack_int LAPACKE_ctrsyl_work( int matrix_order, char trana, char tranb, - lapack_int isgn, lapack_int m, lapack_int n, - const lapack_complex_float* a, lapack_int lda, - const lapack_complex_float* b, lapack_int ldb, - lapack_complex_float* c, lapack_int ldc, - float* scale ); -lapack_int LAPACKE_ztrsyl_work( int matrix_order, char trana, char tranb, - lapack_int isgn, lapack_int m, lapack_int n, - const lapack_complex_double* a, lapack_int lda, - const lapack_complex_double* b, lapack_int ldb, - lapack_complex_double* c, lapack_int ldc, - double* scale ); - -lapack_int LAPACKE_strtri_work( int matrix_order, char uplo, char diag, - lapack_int n, float* a, lapack_int lda ); -lapack_int LAPACKE_dtrtri_work( int matrix_order, char uplo, char diag, - lapack_int n, double* a, lapack_int lda ); -lapack_int LAPACKE_ctrtri_work( int matrix_order, char uplo, char diag, - lapack_int n, lapack_complex_float* a, - lapack_int lda ); -lapack_int LAPACKE_ztrtri_work( int matrix_order, char uplo, char diag, - lapack_int n, lapack_complex_double* a, - lapack_int lda ); - -lapack_int LAPACKE_strtrs_work( int matrix_order, char uplo, char trans, - char diag, lapack_int n, lapack_int nrhs, - const float* a, lapack_int lda, float* b, - lapack_int ldb ); -lapack_int LAPACKE_dtrtrs_work( int matrix_order, char uplo, char trans, - char diag, lapack_int n, lapack_int nrhs, - const double* a, lapack_int lda, double* b, - lapack_int ldb ); -lapack_int LAPACKE_ctrtrs_work( int matrix_order, char uplo, char trans, - char diag, lapack_int n, lapack_int nrhs, - const lapack_complex_float* a, lapack_int lda, - lapack_complex_float* b, lapack_int ldb ); -lapack_int LAPACKE_ztrtrs_work( int matrix_order, char uplo, char trans, - char diag, lapack_int n, lapack_int nrhs, - const lapack_complex_double* a, lapack_int lda, - lapack_complex_double* b, lapack_int ldb ); - -lapack_int LAPACKE_strttf_work( int matrix_order, char transr, char uplo, - lapack_int n, const float* a, lapack_int lda, - float* arf ); -lapack_int LAPACKE_dtrttf_work( int matrix_order, char transr, char uplo, - lapack_int n, const double* a, lapack_int lda, - double* arf ); -lapack_int LAPACKE_ctrttf_work( int matrix_order, char transr, char uplo, - lapack_int n, const lapack_complex_float* a, - lapack_int lda, lapack_complex_float* arf ); -lapack_int LAPACKE_ztrttf_work( int matrix_order, char transr, char uplo, - lapack_int n, const lapack_complex_double* a, - lapack_int lda, lapack_complex_double* arf ); - -lapack_int LAPACKE_strttp_work( int matrix_order, char uplo, lapack_int n, - const float* a, lapack_int lda, float* ap ); -lapack_int LAPACKE_dtrttp_work( int matrix_order, char uplo, lapack_int n, - const double* a, lapack_int lda, double* ap ); -lapack_int LAPACKE_ctrttp_work( int matrix_order, char uplo, lapack_int n, - const lapack_complex_float* a, lapack_int lda, - lapack_complex_float* ap ); -lapack_int LAPACKE_ztrttp_work( int matrix_order, char uplo, lapack_int n, - const lapack_complex_double* a, lapack_int lda, - lapack_complex_double* ap ); - -lapack_int LAPACKE_stzrzf_work( int matrix_order, lapack_int m, lapack_int n, - float* a, lapack_int lda, float* tau, - float* work, lapack_int lwork ); -lapack_int LAPACKE_dtzrzf_work( int matrix_order, lapack_int m, lapack_int n, - double* a, lapack_int lda, double* tau, - double* work, lapack_int lwork ); -lapack_int LAPACKE_ctzrzf_work( int matrix_order, lapack_int m, lapack_int n, - lapack_complex_float* a, lapack_int lda, - lapack_complex_float* tau, - lapack_complex_float* work, lapack_int lwork ); -lapack_int LAPACKE_ztzrzf_work( int matrix_order, lapack_int m, lapack_int n, - lapack_complex_double* a, lapack_int lda, - lapack_complex_double* tau, - lapack_complex_double* work, lapack_int lwork ); - -lapack_int LAPACKE_cungbr_work( int matrix_order, char vect, lapack_int m, - lapack_int n, lapack_int k, - lapack_complex_float* a, lapack_int lda, - const lapack_complex_float* tau, - lapack_complex_float* work, lapack_int lwork ); -lapack_int LAPACKE_zungbr_work( int matrix_order, char vect, lapack_int m, - lapack_int n, lapack_int k, - lapack_complex_double* a, lapack_int lda, - const lapack_complex_double* tau, - lapack_complex_double* work, lapack_int lwork ); - -lapack_int LAPACKE_cunghr_work( int matrix_order, lapack_int n, lapack_int ilo, - lapack_int ihi, lapack_complex_float* a, - lapack_int lda, const lapack_complex_float* tau, - lapack_complex_float* work, lapack_int lwork ); -lapack_int LAPACKE_zunghr_work( int matrix_order, lapack_int n, lapack_int ilo, - lapack_int ihi, lapack_complex_double* a, - lapack_int lda, - const lapack_complex_double* tau, - lapack_complex_double* work, lapack_int lwork ); - -lapack_int LAPACKE_cunglq_work( int matrix_order, lapack_int m, lapack_int n, - lapack_int k, lapack_complex_float* a, - lapack_int lda, const lapack_complex_float* tau, - lapack_complex_float* work, lapack_int lwork ); -lapack_int LAPACKE_zunglq_work( int matrix_order, lapack_int m, lapack_int n, - lapack_int k, lapack_complex_double* a, - lapack_int lda, - const lapack_complex_double* tau, - lapack_complex_double* work, lapack_int lwork ); - -lapack_int LAPACKE_cungql_work( int matrix_order, lapack_int m, lapack_int n, - lapack_int k, lapack_complex_float* a, - lapack_int lda, const lapack_complex_float* tau, - lapack_complex_float* work, lapack_int lwork ); -lapack_int LAPACKE_zungql_work( int matrix_order, lapack_int m, lapack_int n, - lapack_int k, lapack_complex_double* a, - lapack_int lda, - const lapack_complex_double* tau, - lapack_complex_double* work, lapack_int lwork ); - -lapack_int LAPACKE_cungqr_work( int matrix_order, lapack_int m, lapack_int n, - lapack_int k, lapack_complex_float* a, - lapack_int lda, const lapack_complex_float* tau, - lapack_complex_float* work, lapack_int lwork ); -lapack_int LAPACKE_zungqr_work( int matrix_order, lapack_int m, lapack_int n, - lapack_int k, lapack_complex_double* a, - lapack_int lda, - const lapack_complex_double* tau, - lapack_complex_double* work, lapack_int lwork ); - -lapack_int LAPACKE_cungrq_work( int matrix_order, lapack_int m, lapack_int n, - lapack_int k, lapack_complex_float* a, - lapack_int lda, const lapack_complex_float* tau, - lapack_complex_float* work, lapack_int lwork ); -lapack_int LAPACKE_zungrq_work( int matrix_order, lapack_int m, lapack_int n, - lapack_int k, lapack_complex_double* a, - lapack_int lda, - const lapack_complex_double* tau, - lapack_complex_double* work, lapack_int lwork ); - -lapack_int LAPACKE_cungtr_work( int matrix_order, char uplo, lapack_int n, - lapack_complex_float* a, lapack_int lda, - const lapack_complex_float* tau, - lapack_complex_float* work, lapack_int lwork ); -lapack_int LAPACKE_zungtr_work( int matrix_order, char uplo, lapack_int n, - lapack_complex_double* a, lapack_int lda, - const lapack_complex_double* tau, - lapack_complex_double* work, lapack_int lwork ); - -lapack_int LAPACKE_cunmbr_work( int matrix_order, char vect, char side, - char trans, lapack_int m, lapack_int n, - lapack_int k, const lapack_complex_float* a, - lapack_int lda, const lapack_complex_float* tau, - lapack_complex_float* c, lapack_int ldc, - lapack_complex_float* work, lapack_int lwork ); -lapack_int LAPACKE_zunmbr_work( int matrix_order, char vect, char side, - char trans, lapack_int m, lapack_int n, - lapack_int k, const lapack_complex_double* a, - lapack_int lda, - const lapack_complex_double* tau, - lapack_complex_double* c, lapack_int ldc, - lapack_complex_double* work, lapack_int lwork ); - -lapack_int LAPACKE_cunmhr_work( int matrix_order, char side, char trans, - lapack_int m, lapack_int n, lapack_int ilo, - lapack_int ihi, const lapack_complex_float* a, - lapack_int lda, const lapack_complex_float* tau, - lapack_complex_float* c, lapack_int ldc, - lapack_complex_float* work, lapack_int lwork ); -lapack_int LAPACKE_zunmhr_work( int matrix_order, char side, char trans, - lapack_int m, lapack_int n, lapack_int ilo, - lapack_int ihi, const lapack_complex_double* a, - lapack_int lda, - const lapack_complex_double* tau, - lapack_complex_double* c, lapack_int ldc, - lapack_complex_double* work, lapack_int lwork ); - -lapack_int LAPACKE_cunmlq_work( int matrix_order, char side, char trans, - lapack_int m, lapack_int n, lapack_int k, - const lapack_complex_float* a, lapack_int lda, - const lapack_complex_float* tau, - lapack_complex_float* c, lapack_int ldc, - lapack_complex_float* work, lapack_int lwork ); -lapack_int LAPACKE_zunmlq_work( int matrix_order, char side, char trans, - lapack_int m, lapack_int n, lapack_int k, - const lapack_complex_double* a, lapack_int lda, - const lapack_complex_double* tau, - lapack_complex_double* c, lapack_int ldc, - lapack_complex_double* work, lapack_int lwork ); - -lapack_int LAPACKE_cunmql_work( int matrix_order, char side, char trans, - lapack_int m, lapack_int n, lapack_int k, - const lapack_complex_float* a, lapack_int lda, - const lapack_complex_float* tau, - lapack_complex_float* c, lapack_int ldc, - lapack_complex_float* work, lapack_int lwork ); -lapack_int LAPACKE_zunmql_work( int matrix_order, char side, char trans, - lapack_int m, lapack_int n, lapack_int k, - const lapack_complex_double* a, lapack_int lda, - const lapack_complex_double* tau, - lapack_complex_double* c, lapack_int ldc, - lapack_complex_double* work, lapack_int lwork ); - -lapack_int LAPACKE_cunmqr_work( int matrix_order, char side, char trans, - lapack_int m, lapack_int n, lapack_int k, - const lapack_complex_float* a, lapack_int lda, - const lapack_complex_float* tau, - lapack_complex_float* c, lapack_int ldc, - lapack_complex_float* work, lapack_int lwork ); -lapack_int LAPACKE_zunmqr_work( int matrix_order, char side, char trans, - lapack_int m, lapack_int n, lapack_int k, - const lapack_complex_double* a, lapack_int lda, - const lapack_complex_double* tau, - lapack_complex_double* c, lapack_int ldc, - lapack_complex_double* work, lapack_int lwork ); - -lapack_int LAPACKE_cunmrq_work( int matrix_order, char side, char trans, - lapack_int m, lapack_int n, lapack_int k, - const lapack_complex_float* a, lapack_int lda, - const lapack_complex_float* tau, - lapack_complex_float* c, lapack_int ldc, - lapack_complex_float* work, lapack_int lwork ); -lapack_int LAPACKE_zunmrq_work( int matrix_order, char side, char trans, - lapack_int m, lapack_int n, lapack_int k, - const lapack_complex_double* a, lapack_int lda, - const lapack_complex_double* tau, - lapack_complex_double* c, lapack_int ldc, - lapack_complex_double* work, lapack_int lwork ); - -lapack_int LAPACKE_cunmrz_work( int matrix_order, char side, char trans, - lapack_int m, lapack_int n, lapack_int k, - lapack_int l, const lapack_complex_float* a, - lapack_int lda, const lapack_complex_float* tau, - lapack_complex_float* c, lapack_int ldc, - lapack_complex_float* work, lapack_int lwork ); -lapack_int LAPACKE_zunmrz_work( int matrix_order, char side, char trans, - lapack_int m, lapack_int n, lapack_int k, - lapack_int l, const lapack_complex_double* a, - lapack_int lda, - const lapack_complex_double* tau, - lapack_complex_double* c, lapack_int ldc, - lapack_complex_double* work, lapack_int lwork ); - -lapack_int LAPACKE_cunmtr_work( int matrix_order, char side, char uplo, - char trans, lapack_int m, lapack_int n, - const lapack_complex_float* a, lapack_int lda, - const lapack_complex_float* tau, - lapack_complex_float* c, lapack_int ldc, - lapack_complex_float* work, lapack_int lwork ); -lapack_int LAPACKE_zunmtr_work( int matrix_order, char side, char uplo, - char trans, lapack_int m, lapack_int n, - const lapack_complex_double* a, lapack_int lda, - const lapack_complex_double* tau, - lapack_complex_double* c, lapack_int ldc, - lapack_complex_double* work, lapack_int lwork ); - -lapack_int LAPACKE_cupgtr_work( int matrix_order, char uplo, lapack_int n, - const lapack_complex_float* ap, - const lapack_complex_float* tau, - lapack_complex_float* q, lapack_int ldq, - lapack_complex_float* work ); -lapack_int LAPACKE_zupgtr_work( int matrix_order, char uplo, lapack_int n, - const lapack_complex_double* ap, - const lapack_complex_double* tau, - lapack_complex_double* q, lapack_int ldq, - lapack_complex_double* work ); - -lapack_int LAPACKE_cupmtr_work( int matrix_order, char side, char uplo, - char trans, lapack_int m, lapack_int n, - const lapack_complex_float* ap, - const lapack_complex_float* tau, - lapack_complex_float* c, lapack_int ldc, - lapack_complex_float* work ); -lapack_int LAPACKE_zupmtr_work( int matrix_order, char side, char uplo, - char trans, lapack_int m, lapack_int n, - const lapack_complex_double* ap, - const lapack_complex_double* tau, - lapack_complex_double* c, lapack_int ldc, - lapack_complex_double* work ); - -lapack_int LAPACKE_claghe( int matrix_order, lapack_int n, lapack_int k, - const float* d, lapack_complex_float* a, - lapack_int lda, lapack_int* iseed ); -lapack_int LAPACKE_zlaghe( int matrix_order, lapack_int n, lapack_int k, - const double* d, lapack_complex_double* a, - lapack_int lda, lapack_int* iseed ); - -lapack_int LAPACKE_slagsy( int matrix_order, lapack_int n, lapack_int k, - const float* d, float* a, lapack_int lda, - lapack_int* iseed ); -lapack_int LAPACKE_dlagsy( int matrix_order, lapack_int n, lapack_int k, - const double* d, double* a, lapack_int lda, - lapack_int* iseed ); -lapack_int LAPACKE_clagsy( int matrix_order, lapack_int n, lapack_int k, - const float* d, lapack_complex_float* a, - lapack_int lda, lapack_int* iseed ); -lapack_int LAPACKE_zlagsy( int matrix_order, lapack_int n, lapack_int k, - const double* d, lapack_complex_double* a, - lapack_int lda, lapack_int* iseed ); - -lapack_int LAPACKE_slapmr( int matrix_order, lapack_logical forwrd, - lapack_int m, lapack_int n, float* x, lapack_int ldx, - lapack_int* k ); -lapack_int LAPACKE_dlapmr( int matrix_order, lapack_logical forwrd, - lapack_int m, lapack_int n, double* x, - lapack_int ldx, lapack_int* k ); -lapack_int LAPACKE_clapmr( int matrix_order, lapack_logical forwrd, - lapack_int m, lapack_int n, lapack_complex_float* x, - lapack_int ldx, lapack_int* k ); -lapack_int LAPACKE_zlapmr( int matrix_order, lapack_logical forwrd, - lapack_int m, lapack_int n, lapack_complex_double* x, - lapack_int ldx, lapack_int* k ); - - -float LAPACKE_slapy2( float x, float y ); -double LAPACKE_dlapy2( double x, double y ); - -float LAPACKE_slapy3( float x, float y, float z ); -double LAPACKE_dlapy3( double x, double y, double z ); - -lapack_int LAPACKE_slartgp( float f, float g, float* cs, float* sn, float* r ); -lapack_int LAPACKE_dlartgp( double f, double g, double* cs, double* sn, - double* r ); - -lapack_int LAPACKE_slartgs( float x, float y, float sigma, float* cs, - float* sn ); -lapack_int LAPACKE_dlartgs( double x, double y, double sigma, double* cs, - double* sn ); - - -//LAPACK 3.3.0 -lapack_int LAPACKE_cbbcsd( int matrix_order, char jobu1, char jobu2, - char jobv1t, char jobv2t, char trans, lapack_int m, - lapack_int p, lapack_int q, float* theta, float* phi, - lapack_complex_float* u1, lapack_int ldu1, - lapack_complex_float* u2, lapack_int ldu2, - lapack_complex_float* v1t, lapack_int ldv1t, - lapack_complex_float* v2t, lapack_int ldv2t, - float* b11d, float* b11e, float* b12d, float* b12e, - float* b21d, float* b21e, float* b22d, float* b22e ); -lapack_int LAPACKE_cbbcsd_work( int matrix_order, char jobu1, char jobu2, - char jobv1t, char jobv2t, char trans, - lapack_int m, lapack_int p, lapack_int q, - float* theta, float* phi, - lapack_complex_float* u1, lapack_int ldu1, - lapack_complex_float* u2, lapack_int ldu2, - lapack_complex_float* v1t, lapack_int ldv1t, - lapack_complex_float* v2t, lapack_int ldv2t, - float* b11d, float* b11e, float* b12d, - float* b12e, float* b21d, float* b21e, - float* b22d, float* b22e, float* rwork, - lapack_int lrwork ); -lapack_int LAPACKE_cheswapr( int matrix_order, char uplo, lapack_int n, - lapack_complex_float* a, lapack_int i1, - lapack_int i2 ); -lapack_int LAPACKE_cheswapr_work( int matrix_order, char uplo, lapack_int n, - lapack_complex_float* a, lapack_int i1, - lapack_int i2 ); -lapack_int LAPACKE_chetri2( int matrix_order, char uplo, lapack_int n, - lapack_complex_float* a, lapack_int lda, - const lapack_int* ipiv ); -lapack_int LAPACKE_chetri2_work( int matrix_order, char uplo, lapack_int n, - lapack_complex_float* a, lapack_int lda, - const lapack_int* ipiv, - lapack_complex_float* work, lapack_int lwork ); -lapack_int LAPACKE_chetri2x( int matrix_order, char uplo, lapack_int n, - lapack_complex_float* a, lapack_int lda, - const lapack_int* ipiv, lapack_int nb ); -lapack_int LAPACKE_chetri2x_work( int matrix_order, char uplo, lapack_int n, - lapack_complex_float* a, lapack_int lda, - const lapack_int* ipiv, - lapack_complex_float* work, lapack_int nb ); -lapack_int LAPACKE_chetrs2( int matrix_order, char uplo, lapack_int n, - lapack_int nrhs, const lapack_complex_float* a, - lapack_int lda, const lapack_int* ipiv, - lapack_complex_float* b, lapack_int ldb ); -lapack_int LAPACKE_chetrs2_work( int matrix_order, char uplo, lapack_int n, - lapack_int nrhs, const lapack_complex_float* a, - lapack_int lda, const lapack_int* ipiv, - lapack_complex_float* b, lapack_int ldb, - lapack_complex_float* work ); -lapack_int LAPACKE_csyconv( int matrix_order, char uplo, char way, lapack_int n, - lapack_complex_float* a, lapack_int lda, - const lapack_int* ipiv ); -lapack_int LAPACKE_csyconv_work( int matrix_order, char uplo, char way, - lapack_int n, lapack_complex_float* a, - lapack_int lda, const lapack_int* ipiv, - lapack_complex_float* work ); -lapack_int LAPACKE_csyswapr( int matrix_order, char uplo, lapack_int n, - lapack_complex_float* a, lapack_int i1, - lapack_int i2 ); -lapack_int LAPACKE_csyswapr_work( int matrix_order, char uplo, lapack_int n, - lapack_complex_float* a, lapack_int i1, - lapack_int i2 ); -lapack_int LAPACKE_csytri2( int matrix_order, char uplo, lapack_int n, - lapack_complex_float* a, lapack_int lda, - const lapack_int* ipiv ); -lapack_int LAPACKE_csytri2_work( int matrix_order, char uplo, lapack_int n, - lapack_complex_float* a, lapack_int lda, - const lapack_int* ipiv, - lapack_complex_float* work, lapack_int lwork ); -lapack_int LAPACKE_csytri2x( int matrix_order, char uplo, lapack_int n, - lapack_complex_float* a, lapack_int lda, - const lapack_int* ipiv, lapack_int nb ); -lapack_int LAPACKE_csytri2x_work( int matrix_order, char uplo, lapack_int n, - lapack_complex_float* a, lapack_int lda, - const lapack_int* ipiv, - lapack_complex_float* work, lapack_int nb ); -lapack_int LAPACKE_csytrs2( int matrix_order, char uplo, lapack_int n, - lapack_int nrhs, const lapack_complex_float* a, - lapack_int lda, const lapack_int* ipiv, - lapack_complex_float* b, lapack_int ldb ); -lapack_int LAPACKE_csytrs2_work( int matrix_order, char uplo, lapack_int n, - lapack_int nrhs, const lapack_complex_float* a, - lapack_int lda, const lapack_int* ipiv, - lapack_complex_float* b, lapack_int ldb, - lapack_complex_float* work ); -lapack_int LAPACKE_cunbdb( int matrix_order, char trans, char signs, - lapack_int m, lapack_int p, lapack_int q, - lapack_complex_float* x11, lapack_int ldx11, - lapack_complex_float* x12, lapack_int ldx12, - lapack_complex_float* x21, lapack_int ldx21, - lapack_complex_float* x22, lapack_int ldx22, - float* theta, float* phi, - lapack_complex_float* taup1, - lapack_complex_float* taup2, - lapack_complex_float* tauq1, - lapack_complex_float* tauq2 ); -lapack_int LAPACKE_cunbdb_work( int matrix_order, char trans, char signs, - lapack_int m, lapack_int p, lapack_int q, - lapack_complex_float* x11, lapack_int ldx11, - lapack_complex_float* x12, lapack_int ldx12, - lapack_complex_float* x21, lapack_int ldx21, - lapack_complex_float* x22, lapack_int ldx22, - float* theta, float* phi, - lapack_complex_float* taup1, - lapack_complex_float* taup2, - lapack_complex_float* tauq1, - lapack_complex_float* tauq2, - lapack_complex_float* work, lapack_int lwork ); -lapack_int LAPACKE_cuncsd( int matrix_order, char jobu1, char jobu2, - char jobv1t, char jobv2t, char trans, char signs, - lapack_int m, lapack_int p, lapack_int q, - lapack_complex_float* x11, lapack_int ldx11, - lapack_complex_float* x12, lapack_int ldx12, - lapack_complex_float* x21, lapack_int ldx21, - lapack_complex_float* x22, lapack_int ldx22, - float* theta, lapack_complex_float* u1, - lapack_int ldu1, lapack_complex_float* u2, - lapack_int ldu2, lapack_complex_float* v1t, - lapack_int ldv1t, lapack_complex_float* v2t, - lapack_int ldv2t ); -lapack_int LAPACKE_cuncsd_work( int matrix_order, char jobu1, char jobu2, - char jobv1t, char jobv2t, char trans, - char signs, lapack_int m, lapack_int p, - lapack_int q, lapack_complex_float* x11, - lapack_int ldx11, lapack_complex_float* x12, - lapack_int ldx12, lapack_complex_float* x21, - lapack_int ldx21, lapack_complex_float* x22, - lapack_int ldx22, float* theta, - lapack_complex_float* u1, lapack_int ldu1, - lapack_complex_float* u2, lapack_int ldu2, - lapack_complex_float* v1t, lapack_int ldv1t, - lapack_complex_float* v2t, lapack_int ldv2t, - lapack_complex_float* work, lapack_int lwork, - float* rwork, lapack_int lrwork, - lapack_int* iwork ); -lapack_int LAPACKE_dbbcsd( int matrix_order, char jobu1, char jobu2, - char jobv1t, char jobv2t, char trans, lapack_int m, - lapack_int p, lapack_int q, double* theta, - double* phi, double* u1, lapack_int ldu1, double* u2, - lapack_int ldu2, double* v1t, lapack_int ldv1t, - double* v2t, lapack_int ldv2t, double* b11d, - double* b11e, double* b12d, double* b12e, - double* b21d, double* b21e, double* b22d, - double* b22e ); -lapack_int LAPACKE_dbbcsd_work( int matrix_order, char jobu1, char jobu2, - char jobv1t, char jobv2t, char trans, - lapack_int m, lapack_int p, lapack_int q, - double* theta, double* phi, double* u1, - lapack_int ldu1, double* u2, lapack_int ldu2, - double* v1t, lapack_int ldv1t, double* v2t, - lapack_int ldv2t, double* b11d, double* b11e, - double* b12d, double* b12e, double* b21d, - double* b21e, double* b22d, double* b22e, - double* work, lapack_int lwork ); -lapack_int LAPACKE_dorbdb( int matrix_order, char trans, char signs, - lapack_int m, lapack_int p, lapack_int q, - double* x11, lapack_int ldx11, double* x12, - lapack_int ldx12, double* x21, lapack_int ldx21, - double* x22, lapack_int ldx22, double* theta, - double* phi, double* taup1, double* taup2, - double* tauq1, double* tauq2 ); -lapack_int LAPACKE_dorbdb_work( int matrix_order, char trans, char signs, - lapack_int m, lapack_int p, lapack_int q, - double* x11, lapack_int ldx11, double* x12, - lapack_int ldx12, double* x21, lapack_int ldx21, - double* x22, lapack_int ldx22, double* theta, - double* phi, double* taup1, double* taup2, - double* tauq1, double* tauq2, double* work, - lapack_int lwork ); -lapack_int LAPACKE_dorcsd( int matrix_order, char jobu1, char jobu2, - char jobv1t, char jobv2t, char trans, char signs, - lapack_int m, lapack_int p, lapack_int q, - double* x11, lapack_int ldx11, double* x12, - lapack_int ldx12, double* x21, lapack_int ldx21, - double* x22, lapack_int ldx22, double* theta, - double* u1, lapack_int ldu1, double* u2, - lapack_int ldu2, double* v1t, lapack_int ldv1t, - double* v2t, lapack_int ldv2t ); -lapack_int LAPACKE_dorcsd_work( int matrix_order, char jobu1, char jobu2, - char jobv1t, char jobv2t, char trans, - char signs, lapack_int m, lapack_int p, - lapack_int q, double* x11, lapack_int ldx11, - double* x12, lapack_int ldx12, double* x21, - lapack_int ldx21, double* x22, lapack_int ldx22, - double* theta, double* u1, lapack_int ldu1, - double* u2, lapack_int ldu2, double* v1t, - lapack_int ldv1t, double* v2t, lapack_int ldv2t, - double* work, lapack_int lwork, - lapack_int* iwork ); -lapack_int LAPACKE_dsyconv( int matrix_order, char uplo, char way, lapack_int n, - double* a, lapack_int lda, const lapack_int* ipiv ); -lapack_int LAPACKE_dsyconv_work( int matrix_order, char uplo, char way, - lapack_int n, double* a, lapack_int lda, - const lapack_int* ipiv, double* work ); -lapack_int LAPACKE_dsyswapr( int matrix_order, char uplo, lapack_int n, - double* a, lapack_int i1, lapack_int i2 ); -lapack_int LAPACKE_dsyswapr_work( int matrix_order, char uplo, lapack_int n, - double* a, lapack_int i1, lapack_int i2 ); -lapack_int LAPACKE_dsytri2( int matrix_order, char uplo, lapack_int n, - double* a, lapack_int lda, const lapack_int* ipiv ); -lapack_int LAPACKE_dsytri2_work( int matrix_order, char uplo, lapack_int n, - double* a, lapack_int lda, - const lapack_int* ipiv, - lapack_complex_double* work, lapack_int lwork ); -lapack_int LAPACKE_dsytri2x( int matrix_order, char uplo, lapack_int n, - double* a, lapack_int lda, const lapack_int* ipiv, - lapack_int nb ); -lapack_int LAPACKE_dsytri2x_work( int matrix_order, char uplo, lapack_int n, - double* a, lapack_int lda, - const lapack_int* ipiv, double* work, - lapack_int nb ); -lapack_int LAPACKE_dsytrs2( int matrix_order, char uplo, lapack_int n, - lapack_int nrhs, const double* a, lapack_int lda, - const lapack_int* ipiv, double* b, lapack_int ldb ); -lapack_int LAPACKE_dsytrs2_work( int matrix_order, char uplo, lapack_int n, - lapack_int nrhs, const double* a, - lapack_int lda, const lapack_int* ipiv, - double* b, lapack_int ldb, double* work ); -lapack_int LAPACKE_sbbcsd( int matrix_order, char jobu1, char jobu2, - char jobv1t, char jobv2t, char trans, lapack_int m, - lapack_int p, lapack_int q, float* theta, float* phi, - float* u1, lapack_int ldu1, float* u2, - lapack_int ldu2, float* v1t, lapack_int ldv1t, - float* v2t, lapack_int ldv2t, float* b11d, - float* b11e, float* b12d, float* b12e, float* b21d, - float* b21e, float* b22d, float* b22e ); -lapack_int LAPACKE_sbbcsd_work( int matrix_order, char jobu1, char jobu2, - char jobv1t, char jobv2t, char trans, - lapack_int m, lapack_int p, lapack_int q, - float* theta, float* phi, float* u1, - lapack_int ldu1, float* u2, lapack_int ldu2, - float* v1t, lapack_int ldv1t, float* v2t, - lapack_int ldv2t, float* b11d, float* b11e, - float* b12d, float* b12e, float* b21d, - float* b21e, float* b22d, float* b22e, - float* work, lapack_int lwork ); -lapack_int LAPACKE_sorbdb( int matrix_order, char trans, char signs, - lapack_int m, lapack_int p, lapack_int q, float* x11, - lapack_int ldx11, float* x12, lapack_int ldx12, - float* x21, lapack_int ldx21, float* x22, - lapack_int ldx22, float* theta, float* phi, - float* taup1, float* taup2, float* tauq1, - float* tauq2 ); -lapack_int LAPACKE_sorbdb_work( int matrix_order, char trans, char signs, - lapack_int m, lapack_int p, lapack_int q, - float* x11, lapack_int ldx11, float* x12, - lapack_int ldx12, float* x21, lapack_int ldx21, - float* x22, lapack_int ldx22, float* theta, - float* phi, float* taup1, float* taup2, - float* tauq1, float* tauq2, float* work, - lapack_int lwork ); -lapack_int LAPACKE_sorcsd( int matrix_order, char jobu1, char jobu2, - char jobv1t, char jobv2t, char trans, char signs, - lapack_int m, lapack_int p, lapack_int q, float* x11, - lapack_int ldx11, float* x12, lapack_int ldx12, - float* x21, lapack_int ldx21, float* x22, - lapack_int ldx22, float* theta, float* u1, - lapack_int ldu1, float* u2, lapack_int ldu2, - float* v1t, lapack_int ldv1t, float* v2t, - lapack_int ldv2t ); -lapack_int LAPACKE_sorcsd_work( int matrix_order, char jobu1, char jobu2, - char jobv1t, char jobv2t, char trans, - char signs, lapack_int m, lapack_int p, - lapack_int q, float* x11, lapack_int ldx11, - float* x12, lapack_int ldx12, float* x21, - lapack_int ldx21, float* x22, lapack_int ldx22, - float* theta, float* u1, lapack_int ldu1, - float* u2, lapack_int ldu2, float* v1t, - lapack_int ldv1t, float* v2t, lapack_int ldv2t, - float* work, lapack_int lwork, - lapack_int* iwork ); -lapack_int LAPACKE_ssyconv( int matrix_order, char uplo, char way, lapack_int n, - float* a, lapack_int lda, const lapack_int* ipiv ); -lapack_int LAPACKE_ssyconv_work( int matrix_order, char uplo, char way, - lapack_int n, float* a, lapack_int lda, - const lapack_int* ipiv, float* work ); -lapack_int LAPACKE_ssyswapr( int matrix_order, char uplo, lapack_int n, - float* a, lapack_int i1, lapack_int i2 ); -lapack_int LAPACKE_ssyswapr_work( int matrix_order, char uplo, lapack_int n, - float* a, lapack_int i1, lapack_int i2 ); -lapack_int LAPACKE_ssytri2( int matrix_order, char uplo, lapack_int n, float* a, - lapack_int lda, const lapack_int* ipiv ); -lapack_int LAPACKE_ssytri2_work( int matrix_order, char uplo, lapack_int n, - float* a, lapack_int lda, - const lapack_int* ipiv, - lapack_complex_float* work, lapack_int lwork ); -lapack_int LAPACKE_ssytri2x( int matrix_order, char uplo, lapack_int n, - float* a, lapack_int lda, const lapack_int* ipiv, - lapack_int nb ); -lapack_int LAPACKE_ssytri2x_work( int matrix_order, char uplo, lapack_int n, - float* a, lapack_int lda, - const lapack_int* ipiv, float* work, - lapack_int nb ); -lapack_int LAPACKE_ssytrs2( int matrix_order, char uplo, lapack_int n, - lapack_int nrhs, const float* a, lapack_int lda, - const lapack_int* ipiv, float* b, lapack_int ldb ); -lapack_int LAPACKE_ssytrs2_work( int matrix_order, char uplo, lapack_int n, - lapack_int nrhs, const float* a, - lapack_int lda, const lapack_int* ipiv, - float* b, lapack_int ldb, float* work ); -lapack_int LAPACKE_zbbcsd( int matrix_order, char jobu1, char jobu2, - char jobv1t, char jobv2t, char trans, lapack_int m, - lapack_int p, lapack_int q, double* theta, - double* phi, lapack_complex_double* u1, - lapack_int ldu1, lapack_complex_double* u2, - lapack_int ldu2, lapack_complex_double* v1t, - lapack_int ldv1t, lapack_complex_double* v2t, - lapack_int ldv2t, double* b11d, double* b11e, - double* b12d, double* b12e, double* b21d, - double* b21e, double* b22d, double* b22e ); -lapack_int LAPACKE_zbbcsd_work( int matrix_order, char jobu1, char jobu2, - char jobv1t, char jobv2t, char trans, - lapack_int m, lapack_int p, lapack_int q, - double* theta, double* phi, - lapack_complex_double* u1, lapack_int ldu1, - lapack_complex_double* u2, lapack_int ldu2, - lapack_complex_double* v1t, lapack_int ldv1t, - lapack_complex_double* v2t, lapack_int ldv2t, - double* b11d, double* b11e, double* b12d, - double* b12e, double* b21d, double* b21e, - double* b22d, double* b22e, double* rwork, - lapack_int lrwork ); -lapack_int LAPACKE_zheswapr( int matrix_order, char uplo, lapack_int n, - lapack_complex_double* a, lapack_int i1, - lapack_int i2 ); -lapack_int LAPACKE_zheswapr_work( int matrix_order, char uplo, lapack_int n, - lapack_complex_double* a, lapack_int i1, - lapack_int i2 ); -lapack_int LAPACKE_zhetri2( int matrix_order, char uplo, lapack_int n, - lapack_complex_double* a, lapack_int lda, - const lapack_int* ipiv ); -lapack_int LAPACKE_zhetri2_work( int matrix_order, char uplo, lapack_int n, - lapack_complex_double* a, lapack_int lda, - const lapack_int* ipiv, - lapack_complex_double* work, lapack_int lwork ); -lapack_int LAPACKE_zhetri2x( int matrix_order, char uplo, lapack_int n, - lapack_complex_double* a, lapack_int lda, - const lapack_int* ipiv, lapack_int nb ); -lapack_int LAPACKE_zhetri2x_work( int matrix_order, char uplo, lapack_int n, - lapack_complex_double* a, lapack_int lda, - const lapack_int* ipiv, - lapack_complex_double* work, lapack_int nb ); -lapack_int LAPACKE_zhetrs2( int matrix_order, char uplo, lapack_int n, - lapack_int nrhs, const lapack_complex_double* a, - lapack_int lda, const lapack_int* ipiv, - lapack_complex_double* b, lapack_int ldb ); -lapack_int LAPACKE_zhetrs2_work( int matrix_order, char uplo, lapack_int n, - lapack_int nrhs, const lapack_complex_double* a, - lapack_int lda, const lapack_int* ipiv, - lapack_complex_double* b, lapack_int ldb, - lapack_complex_double* work ); -lapack_int LAPACKE_zsyconv( int matrix_order, char uplo, char way, lapack_int n, - lapack_complex_double* a, lapack_int lda, - const lapack_int* ipiv ); -lapack_int LAPACKE_zsyconv_work( int matrix_order, char uplo, char way, - lapack_int n, lapack_complex_double* a, - lapack_int lda, const lapack_int* ipiv, - lapack_complex_double* work ); -lapack_int LAPACKE_zsyswapr( int matrix_order, char uplo, lapack_int n, - lapack_complex_double* a, lapack_int i1, - lapack_int i2 ); -lapack_int LAPACKE_zsyswapr_work( int matrix_order, char uplo, lapack_int n, - lapack_complex_double* a, lapack_int i1, - lapack_int i2 ); -lapack_int LAPACKE_zsytri2( int matrix_order, char uplo, lapack_int n, - lapack_complex_double* a, lapack_int lda, - const lapack_int* ipiv ); -lapack_int LAPACKE_zsytri2_work( int matrix_order, char uplo, lapack_int n, - lapack_complex_double* a, lapack_int lda, - const lapack_int* ipiv, - lapack_complex_double* work, lapack_int lwork ); -lapack_int LAPACKE_zsytri2x( int matrix_order, char uplo, lapack_int n, - lapack_complex_double* a, lapack_int lda, - const lapack_int* ipiv, lapack_int nb ); -lapack_int LAPACKE_zsytri2x_work( int matrix_order, char uplo, lapack_int n, - lapack_complex_double* a, lapack_int lda, - const lapack_int* ipiv, - lapack_complex_double* work, lapack_int nb ); -lapack_int LAPACKE_zsytrs2( int matrix_order, char uplo, lapack_int n, - lapack_int nrhs, const lapack_complex_double* a, - lapack_int lda, const lapack_int* ipiv, - lapack_complex_double* b, lapack_int ldb ); -lapack_int LAPACKE_zsytrs2_work( int matrix_order, char uplo, lapack_int n, - lapack_int nrhs, const lapack_complex_double* a, - lapack_int lda, const lapack_int* ipiv, - lapack_complex_double* b, lapack_int ldb, - lapack_complex_double* work ); -lapack_int LAPACKE_zunbdb( int matrix_order, char trans, char signs, - lapack_int m, lapack_int p, lapack_int q, - lapack_complex_double* x11, lapack_int ldx11, - lapack_complex_double* x12, lapack_int ldx12, - lapack_complex_double* x21, lapack_int ldx21, - lapack_complex_double* x22, lapack_int ldx22, - double* theta, double* phi, - lapack_complex_double* taup1, - lapack_complex_double* taup2, - lapack_complex_double* tauq1, - lapack_complex_double* tauq2 ); -lapack_int LAPACKE_zunbdb_work( int matrix_order, char trans, char signs, - lapack_int m, lapack_int p, lapack_int q, - lapack_complex_double* x11, lapack_int ldx11, - lapack_complex_double* x12, lapack_int ldx12, - lapack_complex_double* x21, lapack_int ldx21, - lapack_complex_double* x22, lapack_int ldx22, - double* theta, double* phi, - lapack_complex_double* taup1, - lapack_complex_double* taup2, - lapack_complex_double* tauq1, - lapack_complex_double* tauq2, - lapack_complex_double* work, lapack_int lwork ); -lapack_int LAPACKE_zuncsd( int matrix_order, char jobu1, char jobu2, - char jobv1t, char jobv2t, char trans, char signs, - lapack_int m, lapack_int p, lapack_int q, - lapack_complex_double* x11, lapack_int ldx11, - lapack_complex_double* x12, lapack_int ldx12, - lapack_complex_double* x21, lapack_int ldx21, - lapack_complex_double* x22, lapack_int ldx22, - double* theta, lapack_complex_double* u1, - lapack_int ldu1, lapack_complex_double* u2, - lapack_int ldu2, lapack_complex_double* v1t, - lapack_int ldv1t, lapack_complex_double* v2t, - lapack_int ldv2t ); -lapack_int LAPACKE_zuncsd_work( int matrix_order, char jobu1, char jobu2, - char jobv1t, char jobv2t, char trans, - char signs, lapack_int m, lapack_int p, - lapack_int q, lapack_complex_double* x11, - lapack_int ldx11, lapack_complex_double* x12, - lapack_int ldx12, lapack_complex_double* x21, - lapack_int ldx21, lapack_complex_double* x22, - lapack_int ldx22, double* theta, - lapack_complex_double* u1, lapack_int ldu1, - lapack_complex_double* u2, lapack_int ldu2, - lapack_complex_double* v1t, lapack_int ldv1t, - lapack_complex_double* v2t, lapack_int ldv2t, - lapack_complex_double* work, lapack_int lwork, - double* rwork, lapack_int lrwork, - lapack_int* iwork ); -//LAPACK 3.4.0 -lapack_int LAPACKE_sgemqrt( int matrix_order, char side, char trans, - lapack_int m, lapack_int n, lapack_int k, - lapack_int nb, const float* v, lapack_int ldv, - const float* t, lapack_int ldt, float* c, - lapack_int ldc ); -lapack_int LAPACKE_dgemqrt( int matrix_order, char side, char trans, - lapack_int m, lapack_int n, lapack_int k, - lapack_int nb, const double* v, lapack_int ldv, - const double* t, lapack_int ldt, double* c, - lapack_int ldc ); -lapack_int LAPACKE_cgemqrt( int matrix_order, char side, char trans, - lapack_int m, lapack_int n, lapack_int k, - lapack_int nb, const lapack_complex_float* v, - lapack_int ldv, const lapack_complex_float* t, - lapack_int ldt, lapack_complex_float* c, - lapack_int ldc ); -lapack_int LAPACKE_zgemqrt( int matrix_order, char side, char trans, - lapack_int m, lapack_int n, lapack_int k, - lapack_int nb, const lapack_complex_double* v, - lapack_int ldv, const lapack_complex_double* t, - lapack_int ldt, lapack_complex_double* c, - lapack_int ldc ); - -lapack_int LAPACKE_sgeqrt( int matrix_order, lapack_int m, lapack_int n, - lapack_int nb, float* a, lapack_int lda, float* t, - lapack_int ldt ); -lapack_int LAPACKE_dgeqrt( int matrix_order, lapack_int m, lapack_int n, - lapack_int nb, double* a, lapack_int lda, double* t, - lapack_int ldt ); -lapack_int LAPACKE_cgeqrt( int matrix_order, lapack_int m, lapack_int n, - lapack_int nb, lapack_complex_float* a, - lapack_int lda, lapack_complex_float* t, - lapack_int ldt ); -lapack_int LAPACKE_zgeqrt( int matrix_order, lapack_int m, lapack_int n, - lapack_int nb, lapack_complex_double* a, - lapack_int lda, lapack_complex_double* t, - lapack_int ldt ); - -lapack_int LAPACKE_sgeqrt2( int matrix_order, lapack_int m, lapack_int n, - float* a, lapack_int lda, float* t, - lapack_int ldt ); -lapack_int LAPACKE_dgeqrt2( int matrix_order, lapack_int m, lapack_int n, - double* a, lapack_int lda, double* t, - lapack_int ldt ); -lapack_int LAPACKE_cgeqrt2( int matrix_order, lapack_int m, lapack_int n, - lapack_complex_float* a, lapack_int lda, - lapack_complex_float* t, lapack_int ldt ); -lapack_int LAPACKE_zgeqrt2( int matrix_order, lapack_int m, lapack_int n, - lapack_complex_double* a, lapack_int lda, - lapack_complex_double* t, lapack_int ldt ); - -lapack_int LAPACKE_sgeqrt3( int matrix_order, lapack_int m, lapack_int n, - float* a, lapack_int lda, float* t, - lapack_int ldt ); -lapack_int LAPACKE_dgeqrt3( int matrix_order, lapack_int m, lapack_int n, - double* a, lapack_int lda, double* t, - lapack_int ldt ); -lapack_int LAPACKE_cgeqrt3( int matrix_order, lapack_int m, lapack_int n, - lapack_complex_float* a, lapack_int lda, - lapack_complex_float* t, lapack_int ldt ); -lapack_int LAPACKE_zgeqrt3( int matrix_order, lapack_int m, lapack_int n, - lapack_complex_double* a, lapack_int lda, - lapack_complex_double* t, lapack_int ldt ); - -lapack_int LAPACKE_stpmqrt( int matrix_order, char side, char trans, - lapack_int m, lapack_int n, lapack_int k, - lapack_int l, lapack_int nb, const float* v, - lapack_int ldv, const float* t, lapack_int ldt, - float* a, lapack_int lda, float* b, - lapack_int ldb ); -lapack_int LAPACKE_dtpmqrt( int matrix_order, char side, char trans, - lapack_int m, lapack_int n, lapack_int k, - lapack_int l, lapack_int nb, const double* v, - lapack_int ldv, const double* t, lapack_int ldt, - double* a, lapack_int lda, double* b, - lapack_int ldb ); -lapack_int LAPACKE_ctpmqrt( int matrix_order, char side, char trans, - lapack_int m, lapack_int n, lapack_int k, - lapack_int l, lapack_int nb, - const lapack_complex_float* v, lapack_int ldv, - const lapack_complex_float* t, lapack_int ldt, - lapack_complex_float* a, lapack_int lda, - lapack_complex_float* b, lapack_int ldb ); -lapack_int LAPACKE_ztpmqrt( int matrix_order, char side, char trans, - lapack_int m, lapack_int n, lapack_int k, - lapack_int l, lapack_int nb, - const lapack_complex_double* v, lapack_int ldv, - const lapack_complex_double* t, lapack_int ldt, - lapack_complex_double* a, lapack_int lda, - lapack_complex_double* b, lapack_int ldb ); - -lapack_int LAPACKE_dtpqrt( int matrix_order, lapack_int m, lapack_int n, - lapack_int l, lapack_int nb, double* a, - lapack_int lda, double* b, lapack_int ldb, double* t, - lapack_int ldt ); -lapack_int LAPACKE_ctpqrt( int matrix_order, lapack_int m, lapack_int n, - lapack_int l, lapack_int nb, lapack_complex_float* a, - lapack_int lda, lapack_complex_float* t, - lapack_complex_float* b, lapack_int ldb, - lapack_int ldt ); -lapack_int LAPACKE_ztpqrt( int matrix_order, lapack_int m, lapack_int n, - lapack_int l, lapack_int nb, - lapack_complex_double* a, lapack_int lda, - lapack_complex_double* b, lapack_int ldb, - lapack_complex_double* t, lapack_int ldt ); - -lapack_int LAPACKE_stpqrt2( int matrix_order, lapack_int m, lapack_int n, - float* a, lapack_int lda, float* b, lapack_int ldb, - float* t, lapack_int ldt ); -lapack_int LAPACKE_dtpqrt2( int matrix_order, lapack_int m, lapack_int n, - double* a, lapack_int lda, double* b, - lapack_int ldb, double* t, lapack_int ldt ); -lapack_int LAPACKE_ctpqrt2( int matrix_order, lapack_int m, lapack_int n, - lapack_complex_float* a, lapack_int lda, - lapack_complex_float* b, lapack_int ldb, - lapack_complex_float* t, lapack_int ldt ); -lapack_int LAPACKE_ztpqrt2( int matrix_order, lapack_int m, lapack_int n, - lapack_complex_double* a, lapack_int lda, - lapack_complex_double* b, lapack_int ldb, - lapack_complex_double* t, lapack_int ldt ); - -lapack_int LAPACKE_stprfb( int matrix_order, char side, char trans, char direct, - char storev, lapack_int m, lapack_int n, - lapack_int k, lapack_int l, const float* v, - lapack_int ldv, const float* t, lapack_int ldt, - float* a, lapack_int lda, float* b, lapack_int ldb, - lapack_int myldwork ); -lapack_int LAPACKE_dtprfb( int matrix_order, char side, char trans, char direct, - char storev, lapack_int m, lapack_int n, - lapack_int k, lapack_int l, const double* v, - lapack_int ldv, const double* t, lapack_int ldt, - double* a, lapack_int lda, double* b, lapack_int ldb, - lapack_int myldwork ); -lapack_int LAPACKE_ctprfb( int matrix_order, char side, char trans, char direct, - char storev, lapack_int m, lapack_int n, - lapack_int k, lapack_int l, - const lapack_complex_float* v, lapack_int ldv, - const lapack_complex_float* t, lapack_int ldt, - lapack_complex_float* a, lapack_int lda, - lapack_complex_float* b, lapack_int ldb, - lapack_int myldwork ); -lapack_int LAPACKE_ztprfb( int matrix_order, char side, char trans, char direct, - char storev, lapack_int m, lapack_int n, - lapack_int k, lapack_int l, - const lapack_complex_double* v, lapack_int ldv, - const lapack_complex_double* t, lapack_int ldt, - lapack_complex_double* a, lapack_int lda, - lapack_complex_double* b, lapack_int ldb, - lapack_int myldwork ); - -lapack_int LAPACKE_sgemqrt_work( int matrix_order, char side, char trans, - lapack_int m, lapack_int n, lapack_int k, - lapack_int nb, const float* v, lapack_int ldv, - const float* t, lapack_int ldt, float* c, - lapack_int ldc, float* work ); -lapack_int LAPACKE_dgemqrt_work( int matrix_order, char side, char trans, - lapack_int m, lapack_int n, lapack_int k, - lapack_int nb, const double* v, lapack_int ldv, - const double* t, lapack_int ldt, double* c, - lapack_int ldc, double* work ); -lapack_int LAPACKE_cgemqrt_work( int matrix_order, char side, char trans, - lapack_int m, lapack_int n, lapack_int k, - lapack_int nb, const lapack_complex_float* v, - lapack_int ldv, const lapack_complex_float* t, - lapack_int ldt, lapack_complex_float* c, - lapack_int ldc, lapack_complex_float* work ); -lapack_int LAPACKE_zgemqrt_work( int matrix_order, char side, char trans, - lapack_int m, lapack_int n, lapack_int k, - lapack_int nb, const lapack_complex_double* v, - lapack_int ldv, const lapack_complex_double* t, - lapack_int ldt, lapack_complex_double* c, - lapack_int ldc, lapack_complex_double* work ); - -lapack_int LAPACKE_sgeqrt_work( int matrix_order, lapack_int m, lapack_int n, - lapack_int nb, float* a, lapack_int lda, - float* t, lapack_int ldt, float* work ); -lapack_int LAPACKE_dgeqrt_work( int matrix_order, lapack_int m, lapack_int n, - lapack_int nb, double* a, lapack_int lda, - double* t, lapack_int ldt, double* work ); -lapack_int LAPACKE_cgeqrt_work( int matrix_order, lapack_int m, lapack_int n, - lapack_int nb, lapack_complex_float* a, - lapack_int lda, lapack_complex_float* t, - lapack_int ldt, lapack_complex_float* work ); -lapack_int LAPACKE_zgeqrt_work( int matrix_order, lapack_int m, lapack_int n, - lapack_int nb, lapack_complex_double* a, - lapack_int lda, lapack_complex_double* t, - lapack_int ldt, lapack_complex_double* work ); - -lapack_int LAPACKE_sgeqrt2_work( int matrix_order, lapack_int m, lapack_int n, - float* a, lapack_int lda, float* t, - lapack_int ldt ); -lapack_int LAPACKE_dgeqrt2_work( int matrix_order, lapack_int m, lapack_int n, - double* a, lapack_int lda, double* t, - lapack_int ldt ); -lapack_int LAPACKE_cgeqrt2_work( int matrix_order, lapack_int m, lapack_int n, - lapack_complex_float* a, lapack_int lda, - lapack_complex_float* t, lapack_int ldt ); -lapack_int LAPACKE_zgeqrt2_work( int matrix_order, lapack_int m, lapack_int n, - lapack_complex_double* a, lapack_int lda, - lapack_complex_double* t, lapack_int ldt ); - -lapack_int LAPACKE_sgeqrt3_work( int matrix_order, lapack_int m, lapack_int n, - float* a, lapack_int lda, float* t, - lapack_int ldt ); -lapack_int LAPACKE_dgeqrt3_work( int matrix_order, lapack_int m, lapack_int n, - double* a, lapack_int lda, double* t, - lapack_int ldt ); -lapack_int LAPACKE_cgeqrt3_work( int matrix_order, lapack_int m, lapack_int n, - lapack_complex_float* a, lapack_int lda, - lapack_complex_float* t, lapack_int ldt ); -lapack_int LAPACKE_zgeqrt3_work( int matrix_order, lapack_int m, lapack_int n, - lapack_complex_double* a, lapack_int lda, - lapack_complex_double* t, lapack_int ldt ); - -lapack_int LAPACKE_stpmqrt_work( int matrix_order, char side, char trans, - lapack_int m, lapack_int n, lapack_int k, - lapack_int l, lapack_int nb, const float* v, - lapack_int ldv, const float* t, lapack_int ldt, - float* a, lapack_int lda, float* b, - lapack_int ldb, float* work ); -lapack_int LAPACKE_dtpmqrt_work( int matrix_order, char side, char trans, - lapack_int m, lapack_int n, lapack_int k, - lapack_int l, lapack_int nb, const double* v, - lapack_int ldv, const double* t, - lapack_int ldt, double* a, lapack_int lda, - double* b, lapack_int ldb, double* work ); -lapack_int LAPACKE_ctpmqrt_work( int matrix_order, char side, char trans, - lapack_int m, lapack_int n, lapack_int k, - lapack_int l, lapack_int nb, - const lapack_complex_float* v, lapack_int ldv, - const lapack_complex_float* t, lapack_int ldt, - lapack_complex_float* a, lapack_int lda, - lapack_complex_float* b, lapack_int ldb, - lapack_complex_float* work ); -lapack_int LAPACKE_ztpmqrt_work( int matrix_order, char side, char trans, - lapack_int m, lapack_int n, lapack_int k, - lapack_int l, lapack_int nb, - const lapack_complex_double* v, lapack_int ldv, - const lapack_complex_double* t, lapack_int ldt, - lapack_complex_double* a, lapack_int lda, - lapack_complex_double* b, lapack_int ldb, - lapack_complex_double* work ); - -lapack_int LAPACKE_dtpqrt_work( int matrix_order, lapack_int m, lapack_int n, - lapack_int l, lapack_int nb, double* a, - lapack_int lda, double* b, lapack_int ldb, - double* t, lapack_int ldt, double* work ); -lapack_int LAPACKE_ctpqrt_work( int matrix_order, lapack_int m, lapack_int n, - lapack_int l, lapack_int nb, - lapack_complex_float* a, lapack_int lda, - lapack_complex_float* t, - lapack_complex_float* b, lapack_int ldb, - lapack_int ldt, lapack_complex_float* work ); -lapack_int LAPACKE_ztpqrt_work( int matrix_order, lapack_int m, lapack_int n, - lapack_int l, lapack_int nb, - lapack_complex_double* a, lapack_int lda, - lapack_complex_double* b, lapack_int ldb, - lapack_complex_double* t, lapack_int ldt, - lapack_complex_double* work ); - -lapack_int LAPACKE_stpqrt2_work( int matrix_order, lapack_int m, lapack_int n, - float* a, lapack_int lda, float* b, - lapack_int ldb, float* t, lapack_int ldt ); -lapack_int LAPACKE_dtpqrt2_work( int matrix_order, lapack_int m, lapack_int n, - double* a, lapack_int lda, double* b, - lapack_int ldb, double* t, lapack_int ldt ); -lapack_int LAPACKE_ctpqrt2_work( int matrix_order, lapack_int m, lapack_int n, - lapack_complex_float* a, lapack_int lda, - lapack_complex_float* b, lapack_int ldb, - lapack_complex_float* t, lapack_int ldt ); -lapack_int LAPACKE_ztpqrt2_work( int matrix_order, lapack_int m, lapack_int n, - lapack_complex_double* a, lapack_int lda, - lapack_complex_double* b, lapack_int ldb, - lapack_complex_double* t, lapack_int ldt ); - -lapack_int LAPACKE_stprfb_work( int matrix_order, char side, char trans, - char direct, char storev, lapack_int m, - lapack_int n, lapack_int k, lapack_int l, - const float* v, lapack_int ldv, const float* t, - lapack_int ldt, float* a, lapack_int lda, - float* b, lapack_int ldb, const float* mywork, - lapack_int myldwork ); -lapack_int LAPACKE_dtprfb_work( int matrix_order, char side, char trans, - char direct, char storev, lapack_int m, - lapack_int n, lapack_int k, lapack_int l, - const double* v, lapack_int ldv, - const double* t, lapack_int ldt, double* a, - lapack_int lda, double* b, lapack_int ldb, - const double* mywork, lapack_int myldwork ); -lapack_int LAPACKE_ctprfb_work( int matrix_order, char side, char trans, - char direct, char storev, lapack_int m, - lapack_int n, lapack_int k, lapack_int l, - const lapack_complex_float* v, lapack_int ldv, - const lapack_complex_float* t, lapack_int ldt, - lapack_complex_float* a, lapack_int lda, - lapack_complex_float* b, lapack_int ldb, - const float* mywork, lapack_int myldwork ); -lapack_int LAPACKE_ztprfb_work( int matrix_order, char side, char trans, - char direct, char storev, lapack_int m, - lapack_int n, lapack_int k, lapack_int l, - const lapack_complex_double* v, lapack_int ldv, - const lapack_complex_double* t, lapack_int ldt, - lapack_complex_double* a, lapack_int lda, - lapack_complex_double* b, lapack_int ldb, - const double* mywork, lapack_int myldwork ); -//LAPACK 3.X.X -lapack_int LAPACKE_csyr( int matrix_order, char uplo, lapack_int n, - lapack_complex_float alpha, - const lapack_complex_float* x, lapack_int incx, - lapack_complex_float* a, lapack_int lda ); -lapack_int LAPACKE_zsyr( int matrix_order, char uplo, lapack_int n, - lapack_complex_double alpha, - const lapack_complex_double* x, lapack_int incx, - lapack_complex_double* a, lapack_int lda ); - -lapack_int LAPACKE_csyr_work( int matrix_order, char uplo, lapack_int n, - lapack_complex_float alpha, - const lapack_complex_float* x, - lapack_int incx, lapack_complex_float* a, - lapack_int lda ); -lapack_int LAPACKE_zsyr_work( int matrix_order, char uplo, lapack_int n, - lapack_complex_double alpha, - const lapack_complex_double* x, - lapack_int incx, lapack_complex_double* a, - lapack_int lda ); - - - -#define LAPACK_sgetrf LAPACK_GLOBAL(sgetrf,SGETRF) -#define LAPACK_dgetrf LAPACK_GLOBAL(dgetrf,DGETRF) -#define LAPACK_cgetrf LAPACK_GLOBAL(cgetrf,CGETRF) -#define LAPACK_zgetrf LAPACK_GLOBAL(zgetrf,ZGETRF) -#define LAPACK_sgbtrf LAPACK_GLOBAL(sgbtrf,SGBTRF) -#define LAPACK_dgbtrf LAPACK_GLOBAL(dgbtrf,DGBTRF) -#define LAPACK_cgbtrf LAPACK_GLOBAL(cgbtrf,CGBTRF) -#define LAPACK_zgbtrf LAPACK_GLOBAL(zgbtrf,ZGBTRF) -#define LAPACK_sgttrf LAPACK_GLOBAL(sgttrf,SGTTRF) -#define LAPACK_dgttrf LAPACK_GLOBAL(dgttrf,DGTTRF) -#define LAPACK_cgttrf LAPACK_GLOBAL(cgttrf,CGTTRF) -#define LAPACK_zgttrf LAPACK_GLOBAL(zgttrf,ZGTTRF) -#define LAPACK_spotrf LAPACK_GLOBAL(spotrf,SPOTRF) -#define LAPACK_dpotrf LAPACK_GLOBAL(dpotrf,DPOTRF) -#define LAPACK_cpotrf LAPACK_GLOBAL(cpotrf,CPOTRF) -#define LAPACK_zpotrf LAPACK_GLOBAL(zpotrf,ZPOTRF) -#define LAPACK_dpstrf LAPACK_GLOBAL(dpstrf,DPSTRF) -#define LAPACK_spstrf LAPACK_GLOBAL(spstrf,SPSTRF) -#define LAPACK_zpstrf LAPACK_GLOBAL(zpstrf,ZPSTRF) -#define LAPACK_cpstrf LAPACK_GLOBAL(cpstrf,CPSTRF) -#define LAPACK_dpftrf LAPACK_GLOBAL(dpftrf,DPFTRF) -#define LAPACK_spftrf LAPACK_GLOBAL(spftrf,SPFTRF) -#define LAPACK_zpftrf LAPACK_GLOBAL(zpftrf,ZPFTRF) -#define LAPACK_cpftrf LAPACK_GLOBAL(cpftrf,CPFTRF) -#define LAPACK_spptrf LAPACK_GLOBAL(spptrf,SPPTRF) -#define LAPACK_dpptrf LAPACK_GLOBAL(dpptrf,DPPTRF) -#define LAPACK_cpptrf LAPACK_GLOBAL(cpptrf,CPPTRF) -#define LAPACK_zpptrf LAPACK_GLOBAL(zpptrf,ZPPTRF) -#define LAPACK_spbtrf LAPACK_GLOBAL(spbtrf,SPBTRF) -#define LAPACK_dpbtrf LAPACK_GLOBAL(dpbtrf,DPBTRF) -#define LAPACK_cpbtrf LAPACK_GLOBAL(cpbtrf,CPBTRF) -#define LAPACK_zpbtrf LAPACK_GLOBAL(zpbtrf,ZPBTRF) -#define LAPACK_spttrf LAPACK_GLOBAL(spttrf,SPTTRF) -#define LAPACK_dpttrf LAPACK_GLOBAL(dpttrf,DPTTRF) -#define LAPACK_cpttrf LAPACK_GLOBAL(cpttrf,CPTTRF) -#define LAPACK_zpttrf LAPACK_GLOBAL(zpttrf,ZPTTRF) -#define LAPACK_ssytrf LAPACK_GLOBAL(ssytrf,SSYTRF) -#define LAPACK_dsytrf LAPACK_GLOBAL(dsytrf,DSYTRF) -#define LAPACK_csytrf LAPACK_GLOBAL(csytrf,CSYTRF) -#define LAPACK_zsytrf LAPACK_GLOBAL(zsytrf,ZSYTRF) -#define LAPACK_chetrf LAPACK_GLOBAL(chetrf,CHETRF) -#define LAPACK_zhetrf LAPACK_GLOBAL(zhetrf,ZHETRF) -#define LAPACK_ssptrf LAPACK_GLOBAL(ssptrf,SSPTRF) -#define LAPACK_dsptrf LAPACK_GLOBAL(dsptrf,DSPTRF) -#define LAPACK_csptrf LAPACK_GLOBAL(csptrf,CSPTRF) -#define LAPACK_zsptrf LAPACK_GLOBAL(zsptrf,ZSPTRF) -#define LAPACK_chptrf LAPACK_GLOBAL(chptrf,CHPTRF) -#define LAPACK_zhptrf LAPACK_GLOBAL(zhptrf,ZHPTRF) -#define LAPACK_sgetrs LAPACK_GLOBAL(sgetrs,SGETRS) -#define LAPACK_dgetrs LAPACK_GLOBAL(dgetrs,DGETRS) -#define LAPACK_cgetrs LAPACK_GLOBAL(cgetrs,CGETRS) -#define LAPACK_zgetrs LAPACK_GLOBAL(zgetrs,ZGETRS) -#define LAPACK_sgbtrs LAPACK_GLOBAL(sgbtrs,SGBTRS) -#define LAPACK_dgbtrs LAPACK_GLOBAL(dgbtrs,DGBTRS) -#define LAPACK_cgbtrs LAPACK_GLOBAL(cgbtrs,CGBTRS) -#define LAPACK_zgbtrs LAPACK_GLOBAL(zgbtrs,ZGBTRS) -#define LAPACK_sgttrs LAPACK_GLOBAL(sgttrs,SGTTRS) -#define LAPACK_dgttrs LAPACK_GLOBAL(dgttrs,DGTTRS) -#define LAPACK_cgttrs LAPACK_GLOBAL(cgttrs,CGTTRS) -#define LAPACK_zgttrs LAPACK_GLOBAL(zgttrs,ZGTTRS) -#define LAPACK_spotrs LAPACK_GLOBAL(spotrs,SPOTRS) -#define LAPACK_dpotrs LAPACK_GLOBAL(dpotrs,DPOTRS) -#define LAPACK_cpotrs LAPACK_GLOBAL(cpotrs,CPOTRS) -#define LAPACK_zpotrs LAPACK_GLOBAL(zpotrs,ZPOTRS) -#define LAPACK_dpftrs LAPACK_GLOBAL(dpftrs,DPFTRS) -#define LAPACK_spftrs LAPACK_GLOBAL(spftrs,SPFTRS) -#define LAPACK_zpftrs LAPACK_GLOBAL(zpftrs,ZPFTRS) -#define LAPACK_cpftrs LAPACK_GLOBAL(cpftrs,CPFTRS) -#define LAPACK_spptrs LAPACK_GLOBAL(spptrs,SPPTRS) -#define LAPACK_dpptrs LAPACK_GLOBAL(dpptrs,DPPTRS) -#define LAPACK_cpptrs LAPACK_GLOBAL(cpptrs,CPPTRS) -#define LAPACK_zpptrs LAPACK_GLOBAL(zpptrs,ZPPTRS) -#define LAPACK_spbtrs LAPACK_GLOBAL(spbtrs,SPBTRS) -#define LAPACK_dpbtrs LAPACK_GLOBAL(dpbtrs,DPBTRS) -#define LAPACK_cpbtrs LAPACK_GLOBAL(cpbtrs,CPBTRS) -#define LAPACK_zpbtrs LAPACK_GLOBAL(zpbtrs,ZPBTRS) -#define LAPACK_spttrs LAPACK_GLOBAL(spttrs,SPTTRS) -#define LAPACK_dpttrs LAPACK_GLOBAL(dpttrs,DPTTRS) -#define LAPACK_cpttrs LAPACK_GLOBAL(cpttrs,CPTTRS) -#define LAPACK_zpttrs LAPACK_GLOBAL(zpttrs,ZPTTRS) -#define LAPACK_ssytrs LAPACK_GLOBAL(ssytrs,SSYTRS) -#define LAPACK_dsytrs LAPACK_GLOBAL(dsytrs,DSYTRS) -#define LAPACK_csytrs LAPACK_GLOBAL(csytrs,CSYTRS) -#define LAPACK_zsytrs LAPACK_GLOBAL(zsytrs,ZSYTRS) -#define LAPACK_chetrs LAPACK_GLOBAL(chetrs,CHETRS) -#define LAPACK_zhetrs LAPACK_GLOBAL(zhetrs,ZHETRS) -#define LAPACK_ssptrs LAPACK_GLOBAL(ssptrs,SSPTRS) -#define LAPACK_dsptrs LAPACK_GLOBAL(dsptrs,DSPTRS) -#define LAPACK_csptrs LAPACK_GLOBAL(csptrs,CSPTRS) -#define LAPACK_zsptrs LAPACK_GLOBAL(zsptrs,ZSPTRS) -#define LAPACK_chptrs LAPACK_GLOBAL(chptrs,CHPTRS) -#define LAPACK_zhptrs LAPACK_GLOBAL(zhptrs,ZHPTRS) -#define LAPACK_strtrs LAPACK_GLOBAL(strtrs,STRTRS) -#define LAPACK_dtrtrs LAPACK_GLOBAL(dtrtrs,DTRTRS) -#define LAPACK_ctrtrs LAPACK_GLOBAL(ctrtrs,CTRTRS) -#define LAPACK_ztrtrs LAPACK_GLOBAL(ztrtrs,ZTRTRS) -#define LAPACK_stptrs LAPACK_GLOBAL(stptrs,STPTRS) -#define LAPACK_dtptrs LAPACK_GLOBAL(dtptrs,DTPTRS) -#define LAPACK_ctptrs LAPACK_GLOBAL(ctptrs,CTPTRS) -#define LAPACK_ztptrs LAPACK_GLOBAL(ztptrs,ZTPTRS) -#define LAPACK_stbtrs LAPACK_GLOBAL(stbtrs,STBTRS) -#define LAPACK_dtbtrs LAPACK_GLOBAL(dtbtrs,DTBTRS) -#define LAPACK_ctbtrs LAPACK_GLOBAL(ctbtrs,CTBTRS) -#define LAPACK_ztbtrs LAPACK_GLOBAL(ztbtrs,ZTBTRS) -#define LAPACK_sgecon LAPACK_GLOBAL(sgecon,SGECON) -#define LAPACK_dgecon LAPACK_GLOBAL(dgecon,DGECON) -#define LAPACK_cgecon LAPACK_GLOBAL(cgecon,CGECON) -#define LAPACK_zgecon LAPACK_GLOBAL(zgecon,ZGECON) -#define LAPACK_sgbcon LAPACK_GLOBAL(sgbcon,SGBCON) -#define LAPACK_dgbcon LAPACK_GLOBAL(dgbcon,DGBCON) -#define LAPACK_cgbcon LAPACK_GLOBAL(cgbcon,CGBCON) -#define LAPACK_zgbcon LAPACK_GLOBAL(zgbcon,ZGBCON) -#define LAPACK_sgtcon LAPACK_GLOBAL(sgtcon,SGTCON) -#define LAPACK_dgtcon LAPACK_GLOBAL(dgtcon,DGTCON) -#define LAPACK_cgtcon LAPACK_GLOBAL(cgtcon,CGTCON) -#define LAPACK_zgtcon LAPACK_GLOBAL(zgtcon,ZGTCON) -#define LAPACK_spocon LAPACK_GLOBAL(spocon,SPOCON) -#define LAPACK_dpocon LAPACK_GLOBAL(dpocon,DPOCON) -#define LAPACK_cpocon LAPACK_GLOBAL(cpocon,CPOCON) -#define LAPACK_zpocon LAPACK_GLOBAL(zpocon,ZPOCON) -#define LAPACK_sppcon LAPACK_GLOBAL(sppcon,SPPCON) -#define LAPACK_dppcon LAPACK_GLOBAL(dppcon,DPPCON) -#define LAPACK_cppcon LAPACK_GLOBAL(cppcon,CPPCON) -#define LAPACK_zppcon LAPACK_GLOBAL(zppcon,ZPPCON) -#define LAPACK_spbcon LAPACK_GLOBAL(spbcon,SPBCON) -#define LAPACK_dpbcon LAPACK_GLOBAL(dpbcon,DPBCON) -#define LAPACK_cpbcon LAPACK_GLOBAL(cpbcon,CPBCON) -#define LAPACK_zpbcon LAPACK_GLOBAL(zpbcon,ZPBCON) -#define LAPACK_sptcon LAPACK_GLOBAL(sptcon,SPTCON) -#define LAPACK_dptcon LAPACK_GLOBAL(dptcon,DPTCON) -#define LAPACK_cptcon LAPACK_GLOBAL(cptcon,CPTCON) -#define LAPACK_zptcon LAPACK_GLOBAL(zptcon,ZPTCON) -#define LAPACK_ssycon LAPACK_GLOBAL(ssycon,SSYCON) -#define LAPACK_dsycon LAPACK_GLOBAL(dsycon,DSYCON) -#define LAPACK_csycon LAPACK_GLOBAL(csycon,CSYCON) -#define LAPACK_zsycon LAPACK_GLOBAL(zsycon,ZSYCON) -#define LAPACK_checon LAPACK_GLOBAL(checon,CHECON) -#define LAPACK_zhecon LAPACK_GLOBAL(zhecon,ZHECON) -#define LAPACK_sspcon LAPACK_GLOBAL(sspcon,SSPCON) -#define LAPACK_dspcon LAPACK_GLOBAL(dspcon,DSPCON) -#define LAPACK_cspcon LAPACK_GLOBAL(cspcon,CSPCON) -#define LAPACK_zspcon LAPACK_GLOBAL(zspcon,ZSPCON) -#define LAPACK_chpcon LAPACK_GLOBAL(chpcon,CHPCON) -#define LAPACK_zhpcon LAPACK_GLOBAL(zhpcon,ZHPCON) -#define LAPACK_strcon LAPACK_GLOBAL(strcon,STRCON) -#define LAPACK_dtrcon LAPACK_GLOBAL(dtrcon,DTRCON) -#define LAPACK_ctrcon LAPACK_GLOBAL(ctrcon,CTRCON) -#define LAPACK_ztrcon LAPACK_GLOBAL(ztrcon,ZTRCON) -#define LAPACK_stpcon LAPACK_GLOBAL(stpcon,STPCON) -#define LAPACK_dtpcon LAPACK_GLOBAL(dtpcon,DTPCON) -#define LAPACK_ctpcon LAPACK_GLOBAL(ctpcon,CTPCON) -#define LAPACK_ztpcon LAPACK_GLOBAL(ztpcon,ZTPCON) -#define LAPACK_stbcon LAPACK_GLOBAL(stbcon,STBCON) -#define LAPACK_dtbcon LAPACK_GLOBAL(dtbcon,DTBCON) -#define LAPACK_ctbcon LAPACK_GLOBAL(ctbcon,CTBCON) -#define LAPACK_ztbcon LAPACK_GLOBAL(ztbcon,ZTBCON) -#define LAPACK_sgerfs LAPACK_GLOBAL(sgerfs,SGERFS) -#define LAPACK_dgerfs LAPACK_GLOBAL(dgerfs,DGERFS) -#define LAPACK_cgerfs LAPACK_GLOBAL(cgerfs,CGERFS) -#define LAPACK_zgerfs LAPACK_GLOBAL(zgerfs,ZGERFS) -#define LAPACK_dgerfsx LAPACK_GLOBAL(dgerfsx,DGERFSX) -#define LAPACK_sgerfsx LAPACK_GLOBAL(sgerfsx,SGERFSX) -#define LAPACK_zgerfsx LAPACK_GLOBAL(zgerfsx,ZGERFSX) -#define LAPACK_cgerfsx LAPACK_GLOBAL(cgerfsx,CGERFSX) -#define LAPACK_sgbrfs LAPACK_GLOBAL(sgbrfs,SGBRFS) -#define LAPACK_dgbrfs LAPACK_GLOBAL(dgbrfs,DGBRFS) -#define LAPACK_cgbrfs LAPACK_GLOBAL(cgbrfs,CGBRFS) -#define LAPACK_zgbrfs LAPACK_GLOBAL(zgbrfs,ZGBRFS) -#define LAPACK_dgbrfsx LAPACK_GLOBAL(dgbrfsx,DGBRFSX) -#define LAPACK_sgbrfsx LAPACK_GLOBAL(sgbrfsx,SGBRFSX) -#define LAPACK_zgbrfsx LAPACK_GLOBAL(zgbrfsx,ZGBRFSX) -#define LAPACK_cgbrfsx LAPACK_GLOBAL(cgbrfsx,CGBRFSX) -#define LAPACK_sgtrfs LAPACK_GLOBAL(sgtrfs,SGTRFS) -#define LAPACK_dgtrfs LAPACK_GLOBAL(dgtrfs,DGTRFS) -#define LAPACK_cgtrfs LAPACK_GLOBAL(cgtrfs,CGTRFS) -#define LAPACK_zgtrfs LAPACK_GLOBAL(zgtrfs,ZGTRFS) -#define LAPACK_sporfs LAPACK_GLOBAL(sporfs,SPORFS) -#define LAPACK_dporfs LAPACK_GLOBAL(dporfs,DPORFS) -#define LAPACK_cporfs LAPACK_GLOBAL(cporfs,CPORFS) -#define LAPACK_zporfs LAPACK_GLOBAL(zporfs,ZPORFS) -#define LAPACK_dporfsx LAPACK_GLOBAL(dporfsx,DPORFSX) -#define LAPACK_sporfsx LAPACK_GLOBAL(sporfsx,SPORFSX) -#define LAPACK_zporfsx LAPACK_GLOBAL(zporfsx,ZPORFSX) -#define LAPACK_cporfsx LAPACK_GLOBAL(cporfsx,CPORFSX) -#define LAPACK_spprfs LAPACK_GLOBAL(spprfs,SPPRFS) -#define LAPACK_dpprfs LAPACK_GLOBAL(dpprfs,DPPRFS) -#define LAPACK_cpprfs LAPACK_GLOBAL(cpprfs,CPPRFS) -#define LAPACK_zpprfs LAPACK_GLOBAL(zpprfs,ZPPRFS) -#define LAPACK_spbrfs LAPACK_GLOBAL(spbrfs,SPBRFS) -#define LAPACK_dpbrfs LAPACK_GLOBAL(dpbrfs,DPBRFS) -#define LAPACK_cpbrfs LAPACK_GLOBAL(cpbrfs,CPBRFS) -#define LAPACK_zpbrfs LAPACK_GLOBAL(zpbrfs,ZPBRFS) -#define LAPACK_sptrfs LAPACK_GLOBAL(sptrfs,SPTRFS) -#define LAPACK_dptrfs LAPACK_GLOBAL(dptrfs,DPTRFS) -#define LAPACK_cptrfs LAPACK_GLOBAL(cptrfs,CPTRFS) -#define LAPACK_zptrfs LAPACK_GLOBAL(zptrfs,ZPTRFS) -#define LAPACK_ssyrfs LAPACK_GLOBAL(ssyrfs,SSYRFS) -#define LAPACK_dsyrfs LAPACK_GLOBAL(dsyrfs,DSYRFS) -#define LAPACK_csyrfs LAPACK_GLOBAL(csyrfs,CSYRFS) -#define LAPACK_zsyrfs LAPACK_GLOBAL(zsyrfs,ZSYRFS) -#define LAPACK_dsyrfsx LAPACK_GLOBAL(dsyrfsx,DSYRFSX) -#define LAPACK_ssyrfsx LAPACK_GLOBAL(ssyrfsx,SSYRFSX) -#define LAPACK_zsyrfsx LAPACK_GLOBAL(zsyrfsx,ZSYRFSX) -#define LAPACK_csyrfsx LAPACK_GLOBAL(csyrfsx,CSYRFSX) -#define LAPACK_cherfs LAPACK_GLOBAL(cherfs,CHERFS) -#define LAPACK_zherfs LAPACK_GLOBAL(zherfs,ZHERFS) -#define LAPACK_zherfsx LAPACK_GLOBAL(zherfsx,ZHERFSX) -#define LAPACK_cherfsx LAPACK_GLOBAL(cherfsx,CHERFSX) -#define LAPACK_ssprfs LAPACK_GLOBAL(ssprfs,SSPRFS) -#define LAPACK_dsprfs LAPACK_GLOBAL(dsprfs,DSPRFS) -#define LAPACK_csprfs LAPACK_GLOBAL(csprfs,CSPRFS) -#define LAPACK_zsprfs LAPACK_GLOBAL(zsprfs,ZSPRFS) -#define LAPACK_chprfs LAPACK_GLOBAL(chprfs,CHPRFS) -#define LAPACK_zhprfs LAPACK_GLOBAL(zhprfs,ZHPRFS) -#define LAPACK_strrfs LAPACK_GLOBAL(strrfs,STRRFS) -#define LAPACK_dtrrfs LAPACK_GLOBAL(dtrrfs,DTRRFS) -#define LAPACK_ctrrfs LAPACK_GLOBAL(ctrrfs,CTRRFS) -#define LAPACK_ztrrfs LAPACK_GLOBAL(ztrrfs,ZTRRFS) -#define LAPACK_stprfs LAPACK_GLOBAL(stprfs,STPRFS) -#define LAPACK_dtprfs LAPACK_GLOBAL(dtprfs,DTPRFS) -#define LAPACK_ctprfs LAPACK_GLOBAL(ctprfs,CTPRFS) -#define LAPACK_ztprfs LAPACK_GLOBAL(ztprfs,ZTPRFS) -#define LAPACK_stbrfs LAPACK_GLOBAL(stbrfs,STBRFS) -#define LAPACK_dtbrfs LAPACK_GLOBAL(dtbrfs,DTBRFS) -#define LAPACK_ctbrfs LAPACK_GLOBAL(ctbrfs,CTBRFS) -#define LAPACK_ztbrfs LAPACK_GLOBAL(ztbrfs,ZTBRFS) -#define LAPACK_sgetri LAPACK_GLOBAL(sgetri,SGETRI) -#define LAPACK_dgetri LAPACK_GLOBAL(dgetri,DGETRI) -#define LAPACK_cgetri LAPACK_GLOBAL(cgetri,CGETRI) -#define LAPACK_zgetri LAPACK_GLOBAL(zgetri,ZGETRI) -#define LAPACK_spotri LAPACK_GLOBAL(spotri,SPOTRI) -#define LAPACK_dpotri LAPACK_GLOBAL(dpotri,DPOTRI) -#define LAPACK_cpotri LAPACK_GLOBAL(cpotri,CPOTRI) -#define LAPACK_zpotri LAPACK_GLOBAL(zpotri,ZPOTRI) -#define LAPACK_dpftri LAPACK_GLOBAL(dpftri,DPFTRI) -#define LAPACK_spftri LAPACK_GLOBAL(spftri,SPFTRI) -#define LAPACK_zpftri LAPACK_GLOBAL(zpftri,ZPFTRI) -#define LAPACK_cpftri LAPACK_GLOBAL(cpftri,CPFTRI) -#define LAPACK_spptri LAPACK_GLOBAL(spptri,SPPTRI) -#define LAPACK_dpptri LAPACK_GLOBAL(dpptri,DPPTRI) -#define LAPACK_cpptri LAPACK_GLOBAL(cpptri,CPPTRI) -#define LAPACK_zpptri LAPACK_GLOBAL(zpptri,ZPPTRI) -#define LAPACK_ssytri LAPACK_GLOBAL(ssytri,SSYTRI) -#define LAPACK_dsytri LAPACK_GLOBAL(dsytri,DSYTRI) -#define LAPACK_csytri LAPACK_GLOBAL(csytri,CSYTRI) -#define LAPACK_zsytri LAPACK_GLOBAL(zsytri,ZSYTRI) -#define LAPACK_chetri LAPACK_GLOBAL(chetri,CHETRI) -#define LAPACK_zhetri LAPACK_GLOBAL(zhetri,ZHETRI) -#define LAPACK_ssptri LAPACK_GLOBAL(ssptri,SSPTRI) -#define LAPACK_dsptri LAPACK_GLOBAL(dsptri,DSPTRI) -#define LAPACK_csptri LAPACK_GLOBAL(csptri,CSPTRI) -#define LAPACK_zsptri LAPACK_GLOBAL(zsptri,ZSPTRI) -#define LAPACK_chptri LAPACK_GLOBAL(chptri,CHPTRI) -#define LAPACK_zhptri LAPACK_GLOBAL(zhptri,ZHPTRI) -#define LAPACK_strtri LAPACK_GLOBAL(strtri,STRTRI) -#define LAPACK_dtrtri LAPACK_GLOBAL(dtrtri,DTRTRI) -#define LAPACK_ctrtri LAPACK_GLOBAL(ctrtri,CTRTRI) -#define LAPACK_ztrtri LAPACK_GLOBAL(ztrtri,ZTRTRI) -#define LAPACK_dtftri LAPACK_GLOBAL(dtftri,DTFTRI) -#define LAPACK_stftri LAPACK_GLOBAL(stftri,STFTRI) -#define LAPACK_ztftri LAPACK_GLOBAL(ztftri,ZTFTRI) -#define LAPACK_ctftri LAPACK_GLOBAL(ctftri,CTFTRI) -#define LAPACK_stptri LAPACK_GLOBAL(stptri,STPTRI) -#define LAPACK_dtptri LAPACK_GLOBAL(dtptri,DTPTRI) -#define LAPACK_ctptri LAPACK_GLOBAL(ctptri,CTPTRI) -#define LAPACK_ztptri LAPACK_GLOBAL(ztptri,ZTPTRI) -#define LAPACK_sgeequ LAPACK_GLOBAL(sgeequ,SGEEQU) -#define LAPACK_dgeequ LAPACK_GLOBAL(dgeequ,DGEEQU) -#define LAPACK_cgeequ LAPACK_GLOBAL(cgeequ,CGEEQU) -#define LAPACK_zgeequ LAPACK_GLOBAL(zgeequ,ZGEEQU) -#define LAPACK_dgeequb LAPACK_GLOBAL(dgeequb,DGEEQUB) -#define LAPACK_sgeequb LAPACK_GLOBAL(sgeequb,SGEEQUB) -#define LAPACK_zgeequb LAPACK_GLOBAL(zgeequb,ZGEEQUB) -#define LAPACK_cgeequb LAPACK_GLOBAL(cgeequb,CGEEQUB) -#define LAPACK_sgbequ LAPACK_GLOBAL(sgbequ,SGBEQU) -#define LAPACK_dgbequ LAPACK_GLOBAL(dgbequ,DGBEQU) -#define LAPACK_cgbequ LAPACK_GLOBAL(cgbequ,CGBEQU) -#define LAPACK_zgbequ LAPACK_GLOBAL(zgbequ,ZGBEQU) -#define LAPACK_dgbequb LAPACK_GLOBAL(dgbequb,DGBEQUB) -#define LAPACK_sgbequb LAPACK_GLOBAL(sgbequb,SGBEQUB) -#define LAPACK_zgbequb LAPACK_GLOBAL(zgbequb,ZGBEQUB) -#define LAPACK_cgbequb LAPACK_GLOBAL(cgbequb,CGBEQUB) -#define LAPACK_spoequ LAPACK_GLOBAL(spoequ,SPOEQU) -#define LAPACK_dpoequ LAPACK_GLOBAL(dpoequ,DPOEQU) -#define LAPACK_cpoequ LAPACK_GLOBAL(cpoequ,CPOEQU) -#define LAPACK_zpoequ LAPACK_GLOBAL(zpoequ,ZPOEQU) -#define LAPACK_dpoequb LAPACK_GLOBAL(dpoequb,DPOEQUB) -#define LAPACK_spoequb LAPACK_GLOBAL(spoequb,SPOEQUB) -#define LAPACK_zpoequb LAPACK_GLOBAL(zpoequb,ZPOEQUB) -#define LAPACK_cpoequb LAPACK_GLOBAL(cpoequb,CPOEQUB) -#define LAPACK_sppequ LAPACK_GLOBAL(sppequ,SPPEQU) -#define LAPACK_dppequ LAPACK_GLOBAL(dppequ,DPPEQU) -#define LAPACK_cppequ LAPACK_GLOBAL(cppequ,CPPEQU) -#define LAPACK_zppequ LAPACK_GLOBAL(zppequ,ZPPEQU) -#define LAPACK_spbequ LAPACK_GLOBAL(spbequ,SPBEQU) -#define LAPACK_dpbequ LAPACK_GLOBAL(dpbequ,DPBEQU) -#define LAPACK_cpbequ LAPACK_GLOBAL(cpbequ,CPBEQU) -#define LAPACK_zpbequ LAPACK_GLOBAL(zpbequ,ZPBEQU) -#define LAPACK_dsyequb LAPACK_GLOBAL(dsyequb,DSYEQUB) -#define LAPACK_ssyequb LAPACK_GLOBAL(ssyequb,SSYEQUB) -#define LAPACK_zsyequb LAPACK_GLOBAL(zsyequb,ZSYEQUB) -#define LAPACK_csyequb LAPACK_GLOBAL(csyequb,CSYEQUB) -#define LAPACK_zheequb LAPACK_GLOBAL(zheequb,ZHEEQUB) -#define LAPACK_cheequb LAPACK_GLOBAL(cheequb,CHEEQUB) -#define LAPACK_sgesv LAPACK_GLOBAL(sgesv,SGESV) -#define LAPACK_dgesv LAPACK_GLOBAL(dgesv,DGESV) -#define LAPACK_cgesv LAPACK_GLOBAL(cgesv,CGESV) -#define LAPACK_zgesv LAPACK_GLOBAL(zgesv,ZGESV) -#define LAPACK_dsgesv LAPACK_GLOBAL(dsgesv,DSGESV) -#define LAPACK_zcgesv LAPACK_GLOBAL(zcgesv,ZCGESV) -#define LAPACK_sgesvx LAPACK_GLOBAL(sgesvx,SGESVX) -#define LAPACK_dgesvx LAPACK_GLOBAL(dgesvx,DGESVX) -#define LAPACK_cgesvx LAPACK_GLOBAL(cgesvx,CGESVX) -#define LAPACK_zgesvx LAPACK_GLOBAL(zgesvx,ZGESVX) -#define LAPACK_dgesvxx LAPACK_GLOBAL(dgesvxx,DGESVXX) -#define LAPACK_sgesvxx LAPACK_GLOBAL(sgesvxx,SGESVXX) -#define LAPACK_zgesvxx LAPACK_GLOBAL(zgesvxx,ZGESVXX) -#define LAPACK_cgesvxx LAPACK_GLOBAL(cgesvxx,CGESVXX) -#define LAPACK_sgbsv LAPACK_GLOBAL(sgbsv,SGBSV) -#define LAPACK_dgbsv LAPACK_GLOBAL(dgbsv,DGBSV) -#define LAPACK_cgbsv LAPACK_GLOBAL(cgbsv,CGBSV) -#define LAPACK_zgbsv LAPACK_GLOBAL(zgbsv,ZGBSV) -#define LAPACK_sgbsvx LAPACK_GLOBAL(sgbsvx,SGBSVX) -#define LAPACK_dgbsvx LAPACK_GLOBAL(dgbsvx,DGBSVX) -#define LAPACK_cgbsvx LAPACK_GLOBAL(cgbsvx,CGBSVX) -#define LAPACK_zgbsvx LAPACK_GLOBAL(zgbsvx,ZGBSVX) -#define LAPACK_dgbsvxx LAPACK_GLOBAL(dgbsvxx,DGBSVXX) -#define LAPACK_sgbsvxx LAPACK_GLOBAL(sgbsvxx,SGBSVXX) -#define LAPACK_zgbsvxx LAPACK_GLOBAL(zgbsvxx,ZGBSVXX) -#define LAPACK_cgbsvxx LAPACK_GLOBAL(cgbsvxx,CGBSVXX) -#define LAPACK_sgtsv LAPACK_GLOBAL(sgtsv,SGTSV) -#define LAPACK_dgtsv LAPACK_GLOBAL(dgtsv,DGTSV) -#define LAPACK_cgtsv LAPACK_GLOBAL(cgtsv,CGTSV) -#define LAPACK_zgtsv LAPACK_GLOBAL(zgtsv,ZGTSV) -#define LAPACK_sgtsvx LAPACK_GLOBAL(sgtsvx,SGTSVX) -#define LAPACK_dgtsvx LAPACK_GLOBAL(dgtsvx,DGTSVX) -#define LAPACK_cgtsvx LAPACK_GLOBAL(cgtsvx,CGTSVX) -#define LAPACK_zgtsvx LAPACK_GLOBAL(zgtsvx,ZGTSVX) -#define LAPACK_sposv LAPACK_GLOBAL(sposv,SPOSV) -#define LAPACK_dposv LAPACK_GLOBAL(dposv,DPOSV) -#define LAPACK_cposv LAPACK_GLOBAL(cposv,CPOSV) -#define LAPACK_zposv LAPACK_GLOBAL(zposv,ZPOSV) -#define LAPACK_dsposv LAPACK_GLOBAL(dsposv,DSPOSV) -#define LAPACK_zcposv LAPACK_GLOBAL(zcposv,ZCPOSV) -#define LAPACK_sposvx LAPACK_GLOBAL(sposvx,SPOSVX) -#define LAPACK_dposvx LAPACK_GLOBAL(dposvx,DPOSVX) -#define LAPACK_cposvx LAPACK_GLOBAL(cposvx,CPOSVX) -#define LAPACK_zposvx LAPACK_GLOBAL(zposvx,ZPOSVX) -#define LAPACK_dposvxx LAPACK_GLOBAL(dposvxx,DPOSVXX) -#define LAPACK_sposvxx LAPACK_GLOBAL(sposvxx,SPOSVXX) -#define LAPACK_zposvxx LAPACK_GLOBAL(zposvxx,ZPOSVXX) -#define LAPACK_cposvxx LAPACK_GLOBAL(cposvxx,CPOSVXX) -#define LAPACK_sppsv LAPACK_GLOBAL(sppsv,SPPSV) -#define LAPACK_dppsv LAPACK_GLOBAL(dppsv,DPPSV) -#define LAPACK_cppsv LAPACK_GLOBAL(cppsv,CPPSV) -#define LAPACK_zppsv LAPACK_GLOBAL(zppsv,ZPPSV) -#define LAPACK_sppsvx LAPACK_GLOBAL(sppsvx,SPPSVX) -#define LAPACK_dppsvx LAPACK_GLOBAL(dppsvx,DPPSVX) -#define LAPACK_cppsvx LAPACK_GLOBAL(cppsvx,CPPSVX) -#define LAPACK_zppsvx LAPACK_GLOBAL(zppsvx,ZPPSVX) -#define LAPACK_spbsv LAPACK_GLOBAL(spbsv,SPBSV) -#define LAPACK_dpbsv LAPACK_GLOBAL(dpbsv,DPBSV) -#define LAPACK_cpbsv LAPACK_GLOBAL(cpbsv,CPBSV) -#define LAPACK_zpbsv LAPACK_GLOBAL(zpbsv,ZPBSV) -#define LAPACK_spbsvx LAPACK_GLOBAL(spbsvx,SPBSVX) -#define LAPACK_dpbsvx LAPACK_GLOBAL(dpbsvx,DPBSVX) -#define LAPACK_cpbsvx LAPACK_GLOBAL(cpbsvx,CPBSVX) -#define LAPACK_zpbsvx LAPACK_GLOBAL(zpbsvx,ZPBSVX) -#define LAPACK_sptsv LAPACK_GLOBAL(sptsv,SPTSV) -#define LAPACK_dptsv LAPACK_GLOBAL(dptsv,DPTSV) -#define LAPACK_cptsv LAPACK_GLOBAL(cptsv,CPTSV) -#define LAPACK_zptsv LAPACK_GLOBAL(zptsv,ZPTSV) -#define LAPACK_sptsvx LAPACK_GLOBAL(sptsvx,SPTSVX) -#define LAPACK_dptsvx LAPACK_GLOBAL(dptsvx,DPTSVX) -#define LAPACK_cptsvx LAPACK_GLOBAL(cptsvx,CPTSVX) -#define LAPACK_zptsvx LAPACK_GLOBAL(zptsvx,ZPTSVX) -#define LAPACK_ssysv LAPACK_GLOBAL(ssysv,SSYSV) -#define LAPACK_dsysv LAPACK_GLOBAL(dsysv,DSYSV) -#define LAPACK_csysv LAPACK_GLOBAL(csysv,CSYSV) -#define LAPACK_zsysv LAPACK_GLOBAL(zsysv,ZSYSV) -#define LAPACK_ssysvx LAPACK_GLOBAL(ssysvx,SSYSVX) -#define LAPACK_dsysvx LAPACK_GLOBAL(dsysvx,DSYSVX) -#define LAPACK_csysvx LAPACK_GLOBAL(csysvx,CSYSVX) -#define LAPACK_zsysvx LAPACK_GLOBAL(zsysvx,ZSYSVX) -#define LAPACK_dsysvxx LAPACK_GLOBAL(dsysvxx,DSYSVXX) -#define LAPACK_ssysvxx LAPACK_GLOBAL(ssysvxx,SSYSVXX) -#define LAPACK_zsysvxx LAPACK_GLOBAL(zsysvxx,ZSYSVXX) -#define LAPACK_csysvxx LAPACK_GLOBAL(csysvxx,CSYSVXX) -#define LAPACK_chesv LAPACK_GLOBAL(chesv,CHESV) -#define LAPACK_zhesv LAPACK_GLOBAL(zhesv,ZHESV) -#define LAPACK_chesvx LAPACK_GLOBAL(chesvx,CHESVX) -#define LAPACK_zhesvx LAPACK_GLOBAL(zhesvx,ZHESVX) -#define LAPACK_zhesvxx LAPACK_GLOBAL(zhesvxx,ZHESVXX) -#define LAPACK_chesvxx LAPACK_GLOBAL(chesvxx,CHESVXX) -#define LAPACK_sspsv LAPACK_GLOBAL(sspsv,SSPSV) -#define LAPACK_dspsv LAPACK_GLOBAL(dspsv,DSPSV) -#define LAPACK_cspsv LAPACK_GLOBAL(cspsv,CSPSV) -#define LAPACK_zspsv LAPACK_GLOBAL(zspsv,ZSPSV) -#define LAPACK_sspsvx LAPACK_GLOBAL(sspsvx,SSPSVX) -#define LAPACK_dspsvx LAPACK_GLOBAL(dspsvx,DSPSVX) -#define LAPACK_cspsvx LAPACK_GLOBAL(cspsvx,CSPSVX) -#define LAPACK_zspsvx LAPACK_GLOBAL(zspsvx,ZSPSVX) -#define LAPACK_chpsv LAPACK_GLOBAL(chpsv,CHPSV) -#define LAPACK_zhpsv LAPACK_GLOBAL(zhpsv,ZHPSV) -#define LAPACK_chpsvx LAPACK_GLOBAL(chpsvx,CHPSVX) -#define LAPACK_zhpsvx LAPACK_GLOBAL(zhpsvx,ZHPSVX) -#define LAPACK_sgeqrf LAPACK_GLOBAL(sgeqrf,SGEQRF) -#define LAPACK_dgeqrf LAPACK_GLOBAL(dgeqrf,DGEQRF) -#define LAPACK_cgeqrf LAPACK_GLOBAL(cgeqrf,CGEQRF) -#define LAPACK_zgeqrf LAPACK_GLOBAL(zgeqrf,ZGEQRF) -#define LAPACK_sgeqpf LAPACK_GLOBAL(sgeqpf,SGEQPF) -#define LAPACK_dgeqpf LAPACK_GLOBAL(dgeqpf,DGEQPF) -#define LAPACK_cgeqpf LAPACK_GLOBAL(cgeqpf,CGEQPF) -#define LAPACK_zgeqpf LAPACK_GLOBAL(zgeqpf,ZGEQPF) -#define LAPACK_sgeqp3 LAPACK_GLOBAL(sgeqp3,SGEQP3) -#define LAPACK_dgeqp3 LAPACK_GLOBAL(dgeqp3,DGEQP3) -#define LAPACK_cgeqp3 LAPACK_GLOBAL(cgeqp3,CGEQP3) -#define LAPACK_zgeqp3 LAPACK_GLOBAL(zgeqp3,ZGEQP3) -#define LAPACK_sorgqr LAPACK_GLOBAL(sorgqr,SORGQR) -#define LAPACK_dorgqr LAPACK_GLOBAL(dorgqr,DORGQR) -#define LAPACK_sormqr LAPACK_GLOBAL(sormqr,SORMQR) -#define LAPACK_dormqr LAPACK_GLOBAL(dormqr,DORMQR) -#define LAPACK_cungqr LAPACK_GLOBAL(cungqr,CUNGQR) -#define LAPACK_zungqr LAPACK_GLOBAL(zungqr,ZUNGQR) -#define LAPACK_cunmqr LAPACK_GLOBAL(cunmqr,CUNMQR) -#define LAPACK_zunmqr LAPACK_GLOBAL(zunmqr,ZUNMQR) -#define LAPACK_sgelqf LAPACK_GLOBAL(sgelqf,SGELQF) -#define LAPACK_dgelqf LAPACK_GLOBAL(dgelqf,DGELQF) -#define LAPACK_cgelqf LAPACK_GLOBAL(cgelqf,CGELQF) -#define LAPACK_zgelqf LAPACK_GLOBAL(zgelqf,ZGELQF) -#define LAPACK_sorglq LAPACK_GLOBAL(sorglq,SORGLQ) -#define LAPACK_dorglq LAPACK_GLOBAL(dorglq,DORGLQ) -#define LAPACK_sormlq LAPACK_GLOBAL(sormlq,SORMLQ) -#define LAPACK_dormlq LAPACK_GLOBAL(dormlq,DORMLQ) -#define LAPACK_cunglq LAPACK_GLOBAL(cunglq,CUNGLQ) -#define LAPACK_zunglq LAPACK_GLOBAL(zunglq,ZUNGLQ) -#define LAPACK_cunmlq LAPACK_GLOBAL(cunmlq,CUNMLQ) -#define LAPACK_zunmlq LAPACK_GLOBAL(zunmlq,ZUNMLQ) -#define LAPACK_sgeqlf LAPACK_GLOBAL(sgeqlf,SGEQLF) -#define LAPACK_dgeqlf LAPACK_GLOBAL(dgeqlf,DGEQLF) -#define LAPACK_cgeqlf LAPACK_GLOBAL(cgeqlf,CGEQLF) -#define LAPACK_zgeqlf LAPACK_GLOBAL(zgeqlf,ZGEQLF) -#define LAPACK_sorgql LAPACK_GLOBAL(sorgql,SORGQL) -#define LAPACK_dorgql LAPACK_GLOBAL(dorgql,DORGQL) -#define LAPACK_cungql LAPACK_GLOBAL(cungql,CUNGQL) -#define LAPACK_zungql LAPACK_GLOBAL(zungql,ZUNGQL) -#define LAPACK_sormql LAPACK_GLOBAL(sormql,SORMQL) -#define LAPACK_dormql LAPACK_GLOBAL(dormql,DORMQL) -#define LAPACK_cunmql LAPACK_GLOBAL(cunmql,CUNMQL) -#define LAPACK_zunmql LAPACK_GLOBAL(zunmql,ZUNMQL) -#define LAPACK_sgerqf LAPACK_GLOBAL(sgerqf,SGERQF) -#define LAPACK_dgerqf LAPACK_GLOBAL(dgerqf,DGERQF) -#define LAPACK_cgerqf LAPACK_GLOBAL(cgerqf,CGERQF) -#define LAPACK_zgerqf LAPACK_GLOBAL(zgerqf,ZGERQF) -#define LAPACK_sorgrq LAPACK_GLOBAL(sorgrq,SORGRQ) -#define LAPACK_dorgrq LAPACK_GLOBAL(dorgrq,DORGRQ) -#define LAPACK_cungrq LAPACK_GLOBAL(cungrq,CUNGRQ) -#define LAPACK_zungrq LAPACK_GLOBAL(zungrq,ZUNGRQ) -#define LAPACK_sormrq LAPACK_GLOBAL(sormrq,SORMRQ) -#define LAPACK_dormrq LAPACK_GLOBAL(dormrq,DORMRQ) -#define LAPACK_cunmrq LAPACK_GLOBAL(cunmrq,CUNMRQ) -#define LAPACK_zunmrq LAPACK_GLOBAL(zunmrq,ZUNMRQ) -#define LAPACK_stzrzf LAPACK_GLOBAL(stzrzf,STZRZF) -#define LAPACK_dtzrzf LAPACK_GLOBAL(dtzrzf,DTZRZF) -#define LAPACK_ctzrzf LAPACK_GLOBAL(ctzrzf,CTZRZF) -#define LAPACK_ztzrzf LAPACK_GLOBAL(ztzrzf,ZTZRZF) -#define LAPACK_sormrz LAPACK_GLOBAL(sormrz,SORMRZ) -#define LAPACK_dormrz LAPACK_GLOBAL(dormrz,DORMRZ) -#define LAPACK_cunmrz LAPACK_GLOBAL(cunmrz,CUNMRZ) -#define LAPACK_zunmrz LAPACK_GLOBAL(zunmrz,ZUNMRZ) -#define LAPACK_sggqrf LAPACK_GLOBAL(sggqrf,SGGQRF) -#define LAPACK_dggqrf LAPACK_GLOBAL(dggqrf,DGGQRF) -#define LAPACK_cggqrf LAPACK_GLOBAL(cggqrf,CGGQRF) -#define LAPACK_zggqrf LAPACK_GLOBAL(zggqrf,ZGGQRF) -#define LAPACK_sggrqf LAPACK_GLOBAL(sggrqf,SGGRQF) -#define LAPACK_dggrqf LAPACK_GLOBAL(dggrqf,DGGRQF) -#define LAPACK_cggrqf LAPACK_GLOBAL(cggrqf,CGGRQF) -#define LAPACK_zggrqf LAPACK_GLOBAL(zggrqf,ZGGRQF) -#define LAPACK_sgebrd LAPACK_GLOBAL(sgebrd,SGEBRD) -#define LAPACK_dgebrd LAPACK_GLOBAL(dgebrd,DGEBRD) -#define LAPACK_cgebrd LAPACK_GLOBAL(cgebrd,CGEBRD) -#define LAPACK_zgebrd LAPACK_GLOBAL(zgebrd,ZGEBRD) -#define LAPACK_sgbbrd LAPACK_GLOBAL(sgbbrd,SGBBRD) -#define LAPACK_dgbbrd LAPACK_GLOBAL(dgbbrd,DGBBRD) -#define LAPACK_cgbbrd LAPACK_GLOBAL(cgbbrd,CGBBRD) -#define LAPACK_zgbbrd LAPACK_GLOBAL(zgbbrd,ZGBBRD) -#define LAPACK_sorgbr LAPACK_GLOBAL(sorgbr,SORGBR) -#define LAPACK_dorgbr LAPACK_GLOBAL(dorgbr,DORGBR) -#define LAPACK_sormbr LAPACK_GLOBAL(sormbr,SORMBR) -#define LAPACK_dormbr LAPACK_GLOBAL(dormbr,DORMBR) -#define LAPACK_cungbr LAPACK_GLOBAL(cungbr,CUNGBR) -#define LAPACK_zungbr LAPACK_GLOBAL(zungbr,ZUNGBR) -#define LAPACK_cunmbr LAPACK_GLOBAL(cunmbr,CUNMBR) -#define LAPACK_zunmbr LAPACK_GLOBAL(zunmbr,ZUNMBR) -#define LAPACK_sbdsqr LAPACK_GLOBAL(sbdsqr,SBDSQR) -#define LAPACK_dbdsqr LAPACK_GLOBAL(dbdsqr,DBDSQR) -#define LAPACK_cbdsqr LAPACK_GLOBAL(cbdsqr,CBDSQR) -#define LAPACK_zbdsqr LAPACK_GLOBAL(zbdsqr,ZBDSQR) -#define LAPACK_sbdsdc LAPACK_GLOBAL(sbdsdc,SBDSDC) -#define LAPACK_dbdsdc LAPACK_GLOBAL(dbdsdc,DBDSDC) -#define LAPACK_ssytrd LAPACK_GLOBAL(ssytrd,SSYTRD) -#define LAPACK_dsytrd LAPACK_GLOBAL(dsytrd,DSYTRD) -#define LAPACK_sorgtr LAPACK_GLOBAL(sorgtr,SORGTR) -#define LAPACK_dorgtr LAPACK_GLOBAL(dorgtr,DORGTR) -#define LAPACK_sormtr LAPACK_GLOBAL(sormtr,SORMTR) -#define LAPACK_dormtr LAPACK_GLOBAL(dormtr,DORMTR) -#define LAPACK_chetrd LAPACK_GLOBAL(chetrd,CHETRD) -#define LAPACK_zhetrd LAPACK_GLOBAL(zhetrd,ZHETRD) -#define LAPACK_cungtr LAPACK_GLOBAL(cungtr,CUNGTR) -#define LAPACK_zungtr LAPACK_GLOBAL(zungtr,ZUNGTR) -#define LAPACK_cunmtr LAPACK_GLOBAL(cunmtr,CUNMTR) -#define LAPACK_zunmtr LAPACK_GLOBAL(zunmtr,ZUNMTR) -#define LAPACK_ssptrd LAPACK_GLOBAL(ssptrd,SSPTRD) -#define LAPACK_dsptrd LAPACK_GLOBAL(dsptrd,DSPTRD) -#define LAPACK_sopgtr LAPACK_GLOBAL(sopgtr,SOPGTR) -#define LAPACK_dopgtr LAPACK_GLOBAL(dopgtr,DOPGTR) -#define LAPACK_sopmtr LAPACK_GLOBAL(sopmtr,SOPMTR) -#define LAPACK_dopmtr LAPACK_GLOBAL(dopmtr,DOPMTR) -#define LAPACK_chptrd LAPACK_GLOBAL(chptrd,CHPTRD) -#define LAPACK_zhptrd LAPACK_GLOBAL(zhptrd,ZHPTRD) -#define LAPACK_cupgtr LAPACK_GLOBAL(cupgtr,CUPGTR) -#define LAPACK_zupgtr LAPACK_GLOBAL(zupgtr,ZUPGTR) -#define LAPACK_cupmtr LAPACK_GLOBAL(cupmtr,CUPMTR) -#define LAPACK_zupmtr LAPACK_GLOBAL(zupmtr,ZUPMTR) -#define LAPACK_ssbtrd LAPACK_GLOBAL(ssbtrd,SSBTRD) -#define LAPACK_dsbtrd LAPACK_GLOBAL(dsbtrd,DSBTRD) -#define LAPACK_chbtrd LAPACK_GLOBAL(chbtrd,CHBTRD) -#define LAPACK_zhbtrd LAPACK_GLOBAL(zhbtrd,ZHBTRD) -#define LAPACK_ssterf LAPACK_GLOBAL(ssterf,SSTERF) -#define LAPACK_dsterf LAPACK_GLOBAL(dsterf,DSTERF) -#define LAPACK_ssteqr LAPACK_GLOBAL(ssteqr,SSTEQR) -#define LAPACK_dsteqr LAPACK_GLOBAL(dsteqr,DSTEQR) -#define LAPACK_csteqr LAPACK_GLOBAL(csteqr,CSTEQR) -#define LAPACK_zsteqr LAPACK_GLOBAL(zsteqr,ZSTEQR) -#define LAPACK_sstemr LAPACK_GLOBAL(sstemr,SSTEMR) -#define LAPACK_dstemr LAPACK_GLOBAL(dstemr,DSTEMR) -#define LAPACK_cstemr LAPACK_GLOBAL(cstemr,CSTEMR) -#define LAPACK_zstemr LAPACK_GLOBAL(zstemr,ZSTEMR) -#define LAPACK_sstedc LAPACK_GLOBAL(sstedc,SSTEDC) -#define LAPACK_dstedc LAPACK_GLOBAL(dstedc,DSTEDC) -#define LAPACK_cstedc LAPACK_GLOBAL(cstedc,CSTEDC) -#define LAPACK_zstedc LAPACK_GLOBAL(zstedc,ZSTEDC) -#define LAPACK_sstegr LAPACK_GLOBAL(sstegr,SSTEGR) -#define LAPACK_dstegr LAPACK_GLOBAL(dstegr,DSTEGR) -#define LAPACK_cstegr LAPACK_GLOBAL(cstegr,CSTEGR) -#define LAPACK_zstegr LAPACK_GLOBAL(zstegr,ZSTEGR) -#define LAPACK_spteqr LAPACK_GLOBAL(spteqr,SPTEQR) -#define LAPACK_dpteqr LAPACK_GLOBAL(dpteqr,DPTEQR) -#define LAPACK_cpteqr LAPACK_GLOBAL(cpteqr,CPTEQR) -#define LAPACK_zpteqr LAPACK_GLOBAL(zpteqr,ZPTEQR) -#define LAPACK_sstebz LAPACK_GLOBAL(sstebz,SSTEBZ) -#define LAPACK_dstebz LAPACK_GLOBAL(dstebz,DSTEBZ) -#define LAPACK_sstein LAPACK_GLOBAL(sstein,SSTEIN) -#define LAPACK_dstein LAPACK_GLOBAL(dstein,DSTEIN) -#define LAPACK_cstein LAPACK_GLOBAL(cstein,CSTEIN) -#define LAPACK_zstein LAPACK_GLOBAL(zstein,ZSTEIN) -#define LAPACK_sdisna LAPACK_GLOBAL(sdisna,SDISNA) -#define LAPACK_ddisna LAPACK_GLOBAL(ddisna,DDISNA) -#define LAPACK_ssygst LAPACK_GLOBAL(ssygst,SSYGST) -#define LAPACK_dsygst LAPACK_GLOBAL(dsygst,DSYGST) -#define LAPACK_chegst LAPACK_GLOBAL(chegst,CHEGST) -#define LAPACK_zhegst LAPACK_GLOBAL(zhegst,ZHEGST) -#define LAPACK_sspgst LAPACK_GLOBAL(sspgst,SSPGST) -#define LAPACK_dspgst LAPACK_GLOBAL(dspgst,DSPGST) -#define LAPACK_chpgst LAPACK_GLOBAL(chpgst,CHPGST) -#define LAPACK_zhpgst LAPACK_GLOBAL(zhpgst,ZHPGST) -#define LAPACK_ssbgst LAPACK_GLOBAL(ssbgst,SSBGST) -#define LAPACK_dsbgst LAPACK_GLOBAL(dsbgst,DSBGST) -#define LAPACK_chbgst LAPACK_GLOBAL(chbgst,CHBGST) -#define LAPACK_zhbgst LAPACK_GLOBAL(zhbgst,ZHBGST) -#define LAPACK_spbstf LAPACK_GLOBAL(spbstf,SPBSTF) -#define LAPACK_dpbstf LAPACK_GLOBAL(dpbstf,DPBSTF) -#define LAPACK_cpbstf LAPACK_GLOBAL(cpbstf,CPBSTF) -#define LAPACK_zpbstf LAPACK_GLOBAL(zpbstf,ZPBSTF) -#define LAPACK_sgehrd LAPACK_GLOBAL(sgehrd,SGEHRD) -#define LAPACK_dgehrd LAPACK_GLOBAL(dgehrd,DGEHRD) -#define LAPACK_cgehrd LAPACK_GLOBAL(cgehrd,CGEHRD) -#define LAPACK_zgehrd LAPACK_GLOBAL(zgehrd,ZGEHRD) -#define LAPACK_sorghr LAPACK_GLOBAL(sorghr,SORGHR) -#define LAPACK_dorghr LAPACK_GLOBAL(dorghr,DORGHR) -#define LAPACK_sormhr LAPACK_GLOBAL(sormhr,SORMHR) -#define LAPACK_dormhr LAPACK_GLOBAL(dormhr,DORMHR) -#define LAPACK_cunghr LAPACK_GLOBAL(cunghr,CUNGHR) -#define LAPACK_zunghr LAPACK_GLOBAL(zunghr,ZUNGHR) -#define LAPACK_cunmhr LAPACK_GLOBAL(cunmhr,CUNMHR) -#define LAPACK_zunmhr LAPACK_GLOBAL(zunmhr,ZUNMHR) -#define LAPACK_sgebal LAPACK_GLOBAL(sgebal,SGEBAL) -#define LAPACK_dgebal LAPACK_GLOBAL(dgebal,DGEBAL) -#define LAPACK_cgebal LAPACK_GLOBAL(cgebal,CGEBAL) -#define LAPACK_zgebal LAPACK_GLOBAL(zgebal,ZGEBAL) -#define LAPACK_sgebak LAPACK_GLOBAL(sgebak,SGEBAK) -#define LAPACK_dgebak LAPACK_GLOBAL(dgebak,DGEBAK) -#define LAPACK_cgebak LAPACK_GLOBAL(cgebak,CGEBAK) -#define LAPACK_zgebak LAPACK_GLOBAL(zgebak,ZGEBAK) -#define LAPACK_shseqr LAPACK_GLOBAL(shseqr,SHSEQR) -#define LAPACK_dhseqr LAPACK_GLOBAL(dhseqr,DHSEQR) -#define LAPACK_chseqr LAPACK_GLOBAL(chseqr,CHSEQR) -#define LAPACK_zhseqr LAPACK_GLOBAL(zhseqr,ZHSEQR) -#define LAPACK_shsein LAPACK_GLOBAL(shsein,SHSEIN) -#define LAPACK_dhsein LAPACK_GLOBAL(dhsein,DHSEIN) -#define LAPACK_chsein LAPACK_GLOBAL(chsein,CHSEIN) -#define LAPACK_zhsein LAPACK_GLOBAL(zhsein,ZHSEIN) -#define LAPACK_strevc LAPACK_GLOBAL(strevc,STREVC) -#define LAPACK_dtrevc LAPACK_GLOBAL(dtrevc,DTREVC) -#define LAPACK_ctrevc LAPACK_GLOBAL(ctrevc,CTREVC) -#define LAPACK_ztrevc LAPACK_GLOBAL(ztrevc,ZTREVC) -#define LAPACK_strsna LAPACK_GLOBAL(strsna,STRSNA) -#define LAPACK_dtrsna LAPACK_GLOBAL(dtrsna,DTRSNA) -#define LAPACK_ctrsna LAPACK_GLOBAL(ctrsna,CTRSNA) -#define LAPACK_ztrsna LAPACK_GLOBAL(ztrsna,ZTRSNA) -#define LAPACK_strexc LAPACK_GLOBAL(strexc,STREXC) -#define LAPACK_dtrexc LAPACK_GLOBAL(dtrexc,DTREXC) -#define LAPACK_ctrexc LAPACK_GLOBAL(ctrexc,CTREXC) -#define LAPACK_ztrexc LAPACK_GLOBAL(ztrexc,ZTREXC) -#define LAPACK_strsen LAPACK_GLOBAL(strsen,STRSEN) -#define LAPACK_dtrsen LAPACK_GLOBAL(dtrsen,DTRSEN) -#define LAPACK_ctrsen LAPACK_GLOBAL(ctrsen,CTRSEN) -#define LAPACK_ztrsen LAPACK_GLOBAL(ztrsen,ZTRSEN) -#define LAPACK_strsyl LAPACK_GLOBAL(strsyl,STRSYL) -#define LAPACK_dtrsyl LAPACK_GLOBAL(dtrsyl,DTRSYL) -#define LAPACK_ctrsyl LAPACK_GLOBAL(ctrsyl,CTRSYL) -#define LAPACK_ztrsyl LAPACK_GLOBAL(ztrsyl,ZTRSYL) -#define LAPACK_sgghrd LAPACK_GLOBAL(sgghrd,SGGHRD) -#define LAPACK_dgghrd LAPACK_GLOBAL(dgghrd,DGGHRD) -#define LAPACK_cgghrd LAPACK_GLOBAL(cgghrd,CGGHRD) -#define LAPACK_zgghrd LAPACK_GLOBAL(zgghrd,ZGGHRD) -#define LAPACK_sggbal LAPACK_GLOBAL(sggbal,SGGBAL) -#define LAPACK_dggbal LAPACK_GLOBAL(dggbal,DGGBAL) -#define LAPACK_cggbal LAPACK_GLOBAL(cggbal,CGGBAL) -#define LAPACK_zggbal LAPACK_GLOBAL(zggbal,ZGGBAL) -#define LAPACK_sggbak LAPACK_GLOBAL(sggbak,SGGBAK) -#define LAPACK_dggbak LAPACK_GLOBAL(dggbak,DGGBAK) -#define LAPACK_cggbak LAPACK_GLOBAL(cggbak,CGGBAK) -#define LAPACK_zggbak LAPACK_GLOBAL(zggbak,ZGGBAK) -#define LAPACK_shgeqz LAPACK_GLOBAL(shgeqz,SHGEQZ) -#define LAPACK_dhgeqz LAPACK_GLOBAL(dhgeqz,DHGEQZ) -#define LAPACK_chgeqz LAPACK_GLOBAL(chgeqz,CHGEQZ) -#define LAPACK_zhgeqz LAPACK_GLOBAL(zhgeqz,ZHGEQZ) -#define LAPACK_stgevc LAPACK_GLOBAL(stgevc,STGEVC) -#define LAPACK_dtgevc LAPACK_GLOBAL(dtgevc,DTGEVC) -#define LAPACK_ctgevc LAPACK_GLOBAL(ctgevc,CTGEVC) -#define LAPACK_ztgevc LAPACK_GLOBAL(ztgevc,ZTGEVC) -#define LAPACK_stgexc LAPACK_GLOBAL(stgexc,STGEXC) -#define LAPACK_dtgexc LAPACK_GLOBAL(dtgexc,DTGEXC) -#define LAPACK_ctgexc LAPACK_GLOBAL(ctgexc,CTGEXC) -#define LAPACK_ztgexc LAPACK_GLOBAL(ztgexc,ZTGEXC) -#define LAPACK_stgsen LAPACK_GLOBAL(stgsen,STGSEN) -#define LAPACK_dtgsen LAPACK_GLOBAL(dtgsen,DTGSEN) -#define LAPACK_ctgsen LAPACK_GLOBAL(ctgsen,CTGSEN) -#define LAPACK_ztgsen LAPACK_GLOBAL(ztgsen,ZTGSEN) -#define LAPACK_stgsyl LAPACK_GLOBAL(stgsyl,STGSYL) -#define LAPACK_dtgsyl LAPACK_GLOBAL(dtgsyl,DTGSYL) -#define LAPACK_ctgsyl LAPACK_GLOBAL(ctgsyl,CTGSYL) -#define LAPACK_ztgsyl LAPACK_GLOBAL(ztgsyl,ZTGSYL) -#define LAPACK_stgsna LAPACK_GLOBAL(stgsna,STGSNA) -#define LAPACK_dtgsna LAPACK_GLOBAL(dtgsna,DTGSNA) -#define LAPACK_ctgsna LAPACK_GLOBAL(ctgsna,CTGSNA) -#define LAPACK_ztgsna LAPACK_GLOBAL(ztgsna,ZTGSNA) -#define LAPACK_sggsvp LAPACK_GLOBAL(sggsvp,SGGSVP) -#define LAPACK_dggsvp LAPACK_GLOBAL(dggsvp,DGGSVP) -#define LAPACK_cggsvp LAPACK_GLOBAL(cggsvp,CGGSVP) -#define LAPACK_zggsvp LAPACK_GLOBAL(zggsvp,ZGGSVP) -#define LAPACK_stgsja LAPACK_GLOBAL(stgsja,STGSJA) -#define LAPACK_dtgsja LAPACK_GLOBAL(dtgsja,DTGSJA) -#define LAPACK_ctgsja LAPACK_GLOBAL(ctgsja,CTGSJA) -#define LAPACK_ztgsja LAPACK_GLOBAL(ztgsja,ZTGSJA) -#define LAPACK_sgels LAPACK_GLOBAL(sgels,SGELS) -#define LAPACK_dgels LAPACK_GLOBAL(dgels,DGELS) -#define LAPACK_cgels LAPACK_GLOBAL(cgels,CGELS) -#define LAPACK_zgels LAPACK_GLOBAL(zgels,ZGELS) -#define LAPACK_sgelsy LAPACK_GLOBAL(sgelsy,SGELSY) -#define LAPACK_dgelsy LAPACK_GLOBAL(dgelsy,DGELSY) -#define LAPACK_cgelsy LAPACK_GLOBAL(cgelsy,CGELSY) -#define LAPACK_zgelsy LAPACK_GLOBAL(zgelsy,ZGELSY) -#define LAPACK_sgelss LAPACK_GLOBAL(sgelss,SGELSS) -#define LAPACK_dgelss LAPACK_GLOBAL(dgelss,DGELSS) -#define LAPACK_cgelss LAPACK_GLOBAL(cgelss,CGELSS) -#define LAPACK_zgelss LAPACK_GLOBAL(zgelss,ZGELSS) -#define LAPACK_sgelsd LAPACK_GLOBAL(sgelsd,SGELSD) -#define LAPACK_dgelsd LAPACK_GLOBAL(dgelsd,DGELSD) -#define LAPACK_cgelsd LAPACK_GLOBAL(cgelsd,CGELSD) -#define LAPACK_zgelsd LAPACK_GLOBAL(zgelsd,ZGELSD) -#define LAPACK_sgglse LAPACK_GLOBAL(sgglse,SGGLSE) -#define LAPACK_dgglse LAPACK_GLOBAL(dgglse,DGGLSE) -#define LAPACK_cgglse LAPACK_GLOBAL(cgglse,CGGLSE) -#define LAPACK_zgglse LAPACK_GLOBAL(zgglse,ZGGLSE) -#define LAPACK_sggglm LAPACK_GLOBAL(sggglm,SGGGLM) -#define LAPACK_dggglm LAPACK_GLOBAL(dggglm,DGGGLM) -#define LAPACK_cggglm LAPACK_GLOBAL(cggglm,CGGGLM) -#define LAPACK_zggglm LAPACK_GLOBAL(zggglm,ZGGGLM) -#define LAPACK_ssyev LAPACK_GLOBAL(ssyev,SSYEV) -#define LAPACK_dsyev LAPACK_GLOBAL(dsyev,DSYEV) -#define LAPACK_cheev LAPACK_GLOBAL(cheev,CHEEV) -#define LAPACK_zheev LAPACK_GLOBAL(zheev,ZHEEV) -#define LAPACK_ssyevd LAPACK_GLOBAL(ssyevd,SSYEVD) -#define LAPACK_dsyevd LAPACK_GLOBAL(dsyevd,DSYEVD) -#define LAPACK_cheevd LAPACK_GLOBAL(cheevd,CHEEVD) -#define LAPACK_zheevd LAPACK_GLOBAL(zheevd,ZHEEVD) -#define LAPACK_ssyevx LAPACK_GLOBAL(ssyevx,SSYEVX) -#define LAPACK_dsyevx LAPACK_GLOBAL(dsyevx,DSYEVX) -#define LAPACK_cheevx LAPACK_GLOBAL(cheevx,CHEEVX) -#define LAPACK_zheevx LAPACK_GLOBAL(zheevx,ZHEEVX) -#define LAPACK_ssyevr LAPACK_GLOBAL(ssyevr,SSYEVR) -#define LAPACK_dsyevr LAPACK_GLOBAL(dsyevr,DSYEVR) -#define LAPACK_cheevr LAPACK_GLOBAL(cheevr,CHEEVR) -#define LAPACK_zheevr LAPACK_GLOBAL(zheevr,ZHEEVR) -#define LAPACK_sspev LAPACK_GLOBAL(sspev,SSPEV) -#define LAPACK_dspev LAPACK_GLOBAL(dspev,DSPEV) -#define LAPACK_chpev LAPACK_GLOBAL(chpev,CHPEV) -#define LAPACK_zhpev LAPACK_GLOBAL(zhpev,ZHPEV) -#define LAPACK_sspevd LAPACK_GLOBAL(sspevd,SSPEVD) -#define LAPACK_dspevd LAPACK_GLOBAL(dspevd,DSPEVD) -#define LAPACK_chpevd LAPACK_GLOBAL(chpevd,CHPEVD) -#define LAPACK_zhpevd LAPACK_GLOBAL(zhpevd,ZHPEVD) -#define LAPACK_sspevx LAPACK_GLOBAL(sspevx,SSPEVX) -#define LAPACK_dspevx LAPACK_GLOBAL(dspevx,DSPEVX) -#define LAPACK_chpevx LAPACK_GLOBAL(chpevx,CHPEVX) -#define LAPACK_zhpevx LAPACK_GLOBAL(zhpevx,ZHPEVX) -#define LAPACK_ssbev LAPACK_GLOBAL(ssbev,SSBEV) -#define LAPACK_dsbev LAPACK_GLOBAL(dsbev,DSBEV) -#define LAPACK_chbev LAPACK_GLOBAL(chbev,CHBEV) -#define LAPACK_zhbev LAPACK_GLOBAL(zhbev,ZHBEV) -#define LAPACK_ssbevd LAPACK_GLOBAL(ssbevd,SSBEVD) -#define LAPACK_dsbevd LAPACK_GLOBAL(dsbevd,DSBEVD) -#define LAPACK_chbevd LAPACK_GLOBAL(chbevd,CHBEVD) -#define LAPACK_zhbevd LAPACK_GLOBAL(zhbevd,ZHBEVD) -#define LAPACK_ssbevx LAPACK_GLOBAL(ssbevx,SSBEVX) -#define LAPACK_dsbevx LAPACK_GLOBAL(dsbevx,DSBEVX) -#define LAPACK_chbevx LAPACK_GLOBAL(chbevx,CHBEVX) -#define LAPACK_zhbevx LAPACK_GLOBAL(zhbevx,ZHBEVX) -#define LAPACK_sstev LAPACK_GLOBAL(sstev,SSTEV) -#define LAPACK_dstev LAPACK_GLOBAL(dstev,DSTEV) -#define LAPACK_sstevd LAPACK_GLOBAL(sstevd,SSTEVD) -#define LAPACK_dstevd LAPACK_GLOBAL(dstevd,DSTEVD) -#define LAPACK_sstevx LAPACK_GLOBAL(sstevx,SSTEVX) -#define LAPACK_dstevx LAPACK_GLOBAL(dstevx,DSTEVX) -#define LAPACK_sstevr LAPACK_GLOBAL(sstevr,SSTEVR) -#define LAPACK_dstevr LAPACK_GLOBAL(dstevr,DSTEVR) -#define LAPACK_sgees LAPACK_GLOBAL(sgees,SGEES) -#define LAPACK_dgees LAPACK_GLOBAL(dgees,DGEES) -#define LAPACK_cgees LAPACK_GLOBAL(cgees,CGEES) -#define LAPACK_zgees LAPACK_GLOBAL(zgees,ZGEES) -#define LAPACK_sgeesx LAPACK_GLOBAL(sgeesx,SGEESX) -#define LAPACK_dgeesx LAPACK_GLOBAL(dgeesx,DGEESX) -#define LAPACK_cgeesx LAPACK_GLOBAL(cgeesx,CGEESX) -#define LAPACK_zgeesx LAPACK_GLOBAL(zgeesx,ZGEESX) -#define LAPACK_sgeev LAPACK_GLOBAL(sgeev,SGEEV) -#define LAPACK_dgeev LAPACK_GLOBAL(dgeev,DGEEV) -#define LAPACK_cgeev LAPACK_GLOBAL(cgeev,CGEEV) -#define LAPACK_zgeev LAPACK_GLOBAL(zgeev,ZGEEV) -#define LAPACK_sgeevx LAPACK_GLOBAL(sgeevx,SGEEVX) -#define LAPACK_dgeevx LAPACK_GLOBAL(dgeevx,DGEEVX) -#define LAPACK_cgeevx LAPACK_GLOBAL(cgeevx,CGEEVX) -#define LAPACK_zgeevx LAPACK_GLOBAL(zgeevx,ZGEEVX) -#define LAPACK_sgesvd LAPACK_GLOBAL(sgesvd,SGESVD) -#define LAPACK_dgesvd LAPACK_GLOBAL(dgesvd,DGESVD) -#define LAPACK_cgesvd LAPACK_GLOBAL(cgesvd,CGESVD) -#define LAPACK_zgesvd LAPACK_GLOBAL(zgesvd,ZGESVD) -#define LAPACK_sgesdd LAPACK_GLOBAL(sgesdd,SGESDD) -#define LAPACK_dgesdd LAPACK_GLOBAL(dgesdd,DGESDD) -#define LAPACK_cgesdd LAPACK_GLOBAL(cgesdd,CGESDD) -#define LAPACK_zgesdd LAPACK_GLOBAL(zgesdd,ZGESDD) -#define LAPACK_dgejsv LAPACK_GLOBAL(dgejsv,DGEJSV) -#define LAPACK_sgejsv LAPACK_GLOBAL(sgejsv,SGEJSV) -#define LAPACK_dgesvj LAPACK_GLOBAL(dgesvj,DGESVJ) -#define LAPACK_sgesvj LAPACK_GLOBAL(sgesvj,SGESVJ) -#define LAPACK_sggsvd LAPACK_GLOBAL(sggsvd,SGGSVD) -#define LAPACK_dggsvd LAPACK_GLOBAL(dggsvd,DGGSVD) -#define LAPACK_cggsvd LAPACK_GLOBAL(cggsvd,CGGSVD) -#define LAPACK_zggsvd LAPACK_GLOBAL(zggsvd,ZGGSVD) -#define LAPACK_ssygv LAPACK_GLOBAL(ssygv,SSYGV) -#define LAPACK_dsygv LAPACK_GLOBAL(dsygv,DSYGV) -#define LAPACK_chegv LAPACK_GLOBAL(chegv,CHEGV) -#define LAPACK_zhegv LAPACK_GLOBAL(zhegv,ZHEGV) -#define LAPACK_ssygvd LAPACK_GLOBAL(ssygvd,SSYGVD) -#define LAPACK_dsygvd LAPACK_GLOBAL(dsygvd,DSYGVD) -#define LAPACK_chegvd LAPACK_GLOBAL(chegvd,CHEGVD) -#define LAPACK_zhegvd LAPACK_GLOBAL(zhegvd,ZHEGVD) -#define LAPACK_ssygvx LAPACK_GLOBAL(ssygvx,SSYGVX) -#define LAPACK_dsygvx LAPACK_GLOBAL(dsygvx,DSYGVX) -#define LAPACK_chegvx LAPACK_GLOBAL(chegvx,CHEGVX) -#define LAPACK_zhegvx LAPACK_GLOBAL(zhegvx,ZHEGVX) -#define LAPACK_sspgv LAPACK_GLOBAL(sspgv,SSPGV) -#define LAPACK_dspgv LAPACK_GLOBAL(dspgv,DSPGV) -#define LAPACK_chpgv LAPACK_GLOBAL(chpgv,CHPGV) -#define LAPACK_zhpgv LAPACK_GLOBAL(zhpgv,ZHPGV) -#define LAPACK_sspgvd LAPACK_GLOBAL(sspgvd,SSPGVD) -#define LAPACK_dspgvd LAPACK_GLOBAL(dspgvd,DSPGVD) -#define LAPACK_chpgvd LAPACK_GLOBAL(chpgvd,CHPGVD) -#define LAPACK_zhpgvd LAPACK_GLOBAL(zhpgvd,ZHPGVD) -#define LAPACK_sspgvx LAPACK_GLOBAL(sspgvx,SSPGVX) -#define LAPACK_dspgvx LAPACK_GLOBAL(dspgvx,DSPGVX) -#define LAPACK_chpgvx LAPACK_GLOBAL(chpgvx,CHPGVX) -#define LAPACK_zhpgvx LAPACK_GLOBAL(zhpgvx,ZHPGVX) -#define LAPACK_ssbgv LAPACK_GLOBAL(ssbgv,SSBGV) -#define LAPACK_dsbgv LAPACK_GLOBAL(dsbgv,DSBGV) -#define LAPACK_chbgv LAPACK_GLOBAL(chbgv,CHBGV) -#define LAPACK_zhbgv LAPACK_GLOBAL(zhbgv,ZHBGV) -#define LAPACK_ssbgvd LAPACK_GLOBAL(ssbgvd,SSBGVD) -#define LAPACK_dsbgvd LAPACK_GLOBAL(dsbgvd,DSBGVD) -#define LAPACK_chbgvd LAPACK_GLOBAL(chbgvd,CHBGVD) -#define LAPACK_zhbgvd LAPACK_GLOBAL(zhbgvd,ZHBGVD) -#define LAPACK_ssbgvx LAPACK_GLOBAL(ssbgvx,SSBGVX) -#define LAPACK_dsbgvx LAPACK_GLOBAL(dsbgvx,DSBGVX) -#define LAPACK_chbgvx LAPACK_GLOBAL(chbgvx,CHBGVX) -#define LAPACK_zhbgvx LAPACK_GLOBAL(zhbgvx,ZHBGVX) -#define LAPACK_sgges LAPACK_GLOBAL(sgges,SGGES) -#define LAPACK_dgges LAPACK_GLOBAL(dgges,DGGES) -#define LAPACK_cgges LAPACK_GLOBAL(cgges,CGGES) -#define LAPACK_zgges LAPACK_GLOBAL(zgges,ZGGES) -#define LAPACK_sggesx LAPACK_GLOBAL(sggesx,SGGESX) -#define LAPACK_dggesx LAPACK_GLOBAL(dggesx,DGGESX) -#define LAPACK_cggesx LAPACK_GLOBAL(cggesx,CGGESX) -#define LAPACK_zggesx LAPACK_GLOBAL(zggesx,ZGGESX) -#define LAPACK_sggev LAPACK_GLOBAL(sggev,SGGEV) -#define LAPACK_dggev LAPACK_GLOBAL(dggev,DGGEV) -#define LAPACK_cggev LAPACK_GLOBAL(cggev,CGGEV) -#define LAPACK_zggev LAPACK_GLOBAL(zggev,ZGGEV) -#define LAPACK_sggevx LAPACK_GLOBAL(sggevx,SGGEVX) -#define LAPACK_dggevx LAPACK_GLOBAL(dggevx,DGGEVX) -#define LAPACK_cggevx LAPACK_GLOBAL(cggevx,CGGEVX) -#define LAPACK_zggevx LAPACK_GLOBAL(zggevx,ZGGEVX) -#define LAPACK_dsfrk LAPACK_GLOBAL(dsfrk,DSFRK) -#define LAPACK_ssfrk LAPACK_GLOBAL(ssfrk,SSFRK) -#define LAPACK_zhfrk LAPACK_GLOBAL(zhfrk,ZHFRK) -#define LAPACK_chfrk LAPACK_GLOBAL(chfrk,CHFRK) -#define LAPACK_dtfsm LAPACK_GLOBAL(dtfsm,DTFSM) -#define LAPACK_stfsm LAPACK_GLOBAL(stfsm,STFSM) -#define LAPACK_ztfsm LAPACK_GLOBAL(ztfsm,ZTFSM) -#define LAPACK_ctfsm LAPACK_GLOBAL(ctfsm,CTFSM) -#define LAPACK_dtfttp LAPACK_GLOBAL(dtfttp,DTFTTP) -#define LAPACK_stfttp LAPACK_GLOBAL(stfttp,STFTTP) -#define LAPACK_ztfttp LAPACK_GLOBAL(ztfttp,ZTFTTP) -#define LAPACK_ctfttp LAPACK_GLOBAL(ctfttp,CTFTTP) -#define LAPACK_dtfttr LAPACK_GLOBAL(dtfttr,DTFTTR) -#define LAPACK_stfttr LAPACK_GLOBAL(stfttr,STFTTR) -#define LAPACK_ztfttr LAPACK_GLOBAL(ztfttr,ZTFTTR) -#define LAPACK_ctfttr LAPACK_GLOBAL(ctfttr,CTFTTR) -#define LAPACK_dtpttf LAPACK_GLOBAL(dtpttf,DTPTTF) -#define LAPACK_stpttf LAPACK_GLOBAL(stpttf,STPTTF) -#define LAPACK_ztpttf LAPACK_GLOBAL(ztpttf,ZTPTTF) -#define LAPACK_ctpttf LAPACK_GLOBAL(ctpttf,CTPTTF) -#define LAPACK_dtpttr LAPACK_GLOBAL(dtpttr,DTPTTR) -#define LAPACK_stpttr LAPACK_GLOBAL(stpttr,STPTTR) -#define LAPACK_ztpttr LAPACK_GLOBAL(ztpttr,ZTPTTR) -#define LAPACK_ctpttr LAPACK_GLOBAL(ctpttr,CTPTTR) -#define LAPACK_dtrttf LAPACK_GLOBAL(dtrttf,DTRTTF) -#define LAPACK_strttf LAPACK_GLOBAL(strttf,STRTTF) -#define LAPACK_ztrttf LAPACK_GLOBAL(ztrttf,ZTRTTF) -#define LAPACK_ctrttf LAPACK_GLOBAL(ctrttf,CTRTTF) -#define LAPACK_dtrttp LAPACK_GLOBAL(dtrttp,DTRTTP) -#define LAPACK_strttp LAPACK_GLOBAL(strttp,STRTTP) -#define LAPACK_ztrttp LAPACK_GLOBAL(ztrttp,ZTRTTP) -#define LAPACK_ctrttp LAPACK_GLOBAL(ctrttp,CTRTTP) -#define LAPACK_sgeqrfp LAPACK_GLOBAL(sgeqrfp,SGEQRFP) -#define LAPACK_dgeqrfp LAPACK_GLOBAL(dgeqrfp,DGEQRFP) -#define LAPACK_cgeqrfp LAPACK_GLOBAL(cgeqrfp,CGEQRFP) -#define LAPACK_zgeqrfp LAPACK_GLOBAL(zgeqrfp,ZGEQRFP) -#define LAPACK_clacgv LAPACK_GLOBAL(clacgv,CLACGV) -#define LAPACK_zlacgv LAPACK_GLOBAL(zlacgv,ZLACGV) -#define LAPACK_slarnv LAPACK_GLOBAL(slarnv,SLARNV) -#define LAPACK_dlarnv LAPACK_GLOBAL(dlarnv,DLARNV) -#define LAPACK_clarnv LAPACK_GLOBAL(clarnv,CLARNV) -#define LAPACK_zlarnv LAPACK_GLOBAL(zlarnv,ZLARNV) -#define LAPACK_sgeqr2 LAPACK_GLOBAL(sgeqr2,SGEQR2) -#define LAPACK_dgeqr2 LAPACK_GLOBAL(dgeqr2,DGEQR2) -#define LAPACK_cgeqr2 LAPACK_GLOBAL(cgeqr2,CGEQR2) -#define LAPACK_zgeqr2 LAPACK_GLOBAL(zgeqr2,ZGEQR2) -#define LAPACK_slacpy LAPACK_GLOBAL(slacpy,SLACPY) -#define LAPACK_dlacpy LAPACK_GLOBAL(dlacpy,DLACPY) -#define LAPACK_clacpy LAPACK_GLOBAL(clacpy,CLACPY) -#define LAPACK_zlacpy LAPACK_GLOBAL(zlacpy,ZLACPY) -#define LAPACK_sgetf2 LAPACK_GLOBAL(sgetf2,SGETF2) -#define LAPACK_dgetf2 LAPACK_GLOBAL(dgetf2,DGETF2) -#define LAPACK_cgetf2 LAPACK_GLOBAL(cgetf2,CGETF2) -#define LAPACK_zgetf2 LAPACK_GLOBAL(zgetf2,ZGETF2) -#define LAPACK_slaswp LAPACK_GLOBAL(slaswp,SLASWP) -#define LAPACK_dlaswp LAPACK_GLOBAL(dlaswp,DLASWP) -#define LAPACK_claswp LAPACK_GLOBAL(claswp,CLASWP) -#define LAPACK_zlaswp LAPACK_GLOBAL(zlaswp,ZLASWP) -#define LAPACK_slange LAPACK_GLOBAL(slange,SLANGE) -#define LAPACK_dlange LAPACK_GLOBAL(dlange,DLANGE) -#define LAPACK_clange LAPACK_GLOBAL(clange,CLANGE) -#define LAPACK_zlange LAPACK_GLOBAL(zlange,ZLANGE) -#define LAPACK_clanhe LAPACK_GLOBAL(clanhe,CLANHE) -#define LAPACK_zlanhe LAPACK_GLOBAL(zlanhe,ZLANHE) -#define LAPACK_slansy LAPACK_GLOBAL(slansy,SLANSY) -#define LAPACK_dlansy LAPACK_GLOBAL(dlansy,DLANSY) -#define LAPACK_clansy LAPACK_GLOBAL(clansy,CLANSY) -#define LAPACK_zlansy LAPACK_GLOBAL(zlansy,ZLANSY) -#define LAPACK_slantr LAPACK_GLOBAL(slantr,SLANTR) -#define LAPACK_dlantr LAPACK_GLOBAL(dlantr,DLANTR) -#define LAPACK_clantr LAPACK_GLOBAL(clantr,CLANTR) -#define LAPACK_zlantr LAPACK_GLOBAL(zlantr,ZLANTR) -#define LAPACK_slamch LAPACK_GLOBAL(slamch,SLAMCH) -#define LAPACK_dlamch LAPACK_GLOBAL(dlamch,DLAMCH) -#define LAPACK_sgelq2 LAPACK_GLOBAL(sgelq2,SGELQ2) -#define LAPACK_dgelq2 LAPACK_GLOBAL(dgelq2,DGELQ2) -#define LAPACK_cgelq2 LAPACK_GLOBAL(cgelq2,CGELQ2) -#define LAPACK_zgelq2 LAPACK_GLOBAL(zgelq2,ZGELQ2) -#define LAPACK_slarfb LAPACK_GLOBAL(slarfb,SLARFB) -#define LAPACK_dlarfb LAPACK_GLOBAL(dlarfb,DLARFB) -#define LAPACK_clarfb LAPACK_GLOBAL(clarfb,CLARFB) -#define LAPACK_zlarfb LAPACK_GLOBAL(zlarfb,ZLARFB) -#define LAPACK_slarfg LAPACK_GLOBAL(slarfg,SLARFG) -#define LAPACK_dlarfg LAPACK_GLOBAL(dlarfg,DLARFG) -#define LAPACK_clarfg LAPACK_GLOBAL(clarfg,CLARFG) -#define LAPACK_zlarfg LAPACK_GLOBAL(zlarfg,ZLARFG) -#define LAPACK_slarft LAPACK_GLOBAL(slarft,SLARFT) -#define LAPACK_dlarft LAPACK_GLOBAL(dlarft,DLARFT) -#define LAPACK_clarft LAPACK_GLOBAL(clarft,CLARFT) -#define LAPACK_zlarft LAPACK_GLOBAL(zlarft,ZLARFT) -#define LAPACK_slarfx LAPACK_GLOBAL(slarfx,SLARFX) -#define LAPACK_dlarfx LAPACK_GLOBAL(dlarfx,DLARFX) -#define LAPACK_clarfx LAPACK_GLOBAL(clarfx,CLARFX) -#define LAPACK_zlarfx LAPACK_GLOBAL(zlarfx,ZLARFX) -#define LAPACK_slatms LAPACK_GLOBAL(slatms,SLATMS) -#define LAPACK_dlatms LAPACK_GLOBAL(dlatms,DLATMS) -#define LAPACK_clatms LAPACK_GLOBAL(clatms,CLATMS) -#define LAPACK_zlatms LAPACK_GLOBAL(zlatms,ZLATMS) -#define LAPACK_slag2d LAPACK_GLOBAL(slag2d,SLAG2D) -#define LAPACK_dlag2s LAPACK_GLOBAL(dlag2s,DLAG2S) -#define LAPACK_clag2z LAPACK_GLOBAL(clag2z,CLAG2Z) -#define LAPACK_zlag2c LAPACK_GLOBAL(zlag2c,ZLAG2C) -#define LAPACK_slauum LAPACK_GLOBAL(slauum,SLAUUM) -#define LAPACK_dlauum LAPACK_GLOBAL(dlauum,DLAUUM) -#define LAPACK_clauum LAPACK_GLOBAL(clauum,CLAUUM) -#define LAPACK_zlauum LAPACK_GLOBAL(zlauum,ZLAUUM) -#define LAPACK_slagge LAPACK_GLOBAL(slagge,SLAGGE) -#define LAPACK_dlagge LAPACK_GLOBAL(dlagge,DLAGGE) -#define LAPACK_clagge LAPACK_GLOBAL(clagge,CLAGGE) -#define LAPACK_zlagge LAPACK_GLOBAL(zlagge,ZLAGGE) -#define LAPACK_slaset LAPACK_GLOBAL(slaset,SLASET) -#define LAPACK_dlaset LAPACK_GLOBAL(dlaset,DLASET) -#define LAPACK_claset LAPACK_GLOBAL(claset,CLASET) -#define LAPACK_zlaset LAPACK_GLOBAL(zlaset,ZLASET) -#define LAPACK_slasrt LAPACK_GLOBAL(slasrt,SLASRT) -#define LAPACK_dlasrt LAPACK_GLOBAL(dlasrt,DLASRT) -#define LAPACK_slagsy LAPACK_GLOBAL(slagsy,SLAGSY) -#define LAPACK_dlagsy LAPACK_GLOBAL(dlagsy,DLAGSY) -#define LAPACK_clagsy LAPACK_GLOBAL(clagsy,CLAGSY) -#define LAPACK_zlagsy LAPACK_GLOBAL(zlagsy,ZLAGSY) -#define LAPACK_claghe LAPACK_GLOBAL(claghe,CLAGHE) -#define LAPACK_zlaghe LAPACK_GLOBAL(zlaghe,ZLAGHE) -#define LAPACK_slapmr LAPACK_GLOBAL(slapmr,SLAPMR) -#define LAPACK_dlapmr LAPACK_GLOBAL(dlapmr,DLAPMR) -#define LAPACK_clapmr LAPACK_GLOBAL(clapmr,CLAPMR) -#define LAPACK_zlapmr LAPACK_GLOBAL(zlapmr,ZLAPMR) -#define LAPACK_slapy2 LAPACK_GLOBAL(slapy2,SLAPY2) -#define LAPACK_dlapy2 LAPACK_GLOBAL(dlapy2,DLAPY2) -#define LAPACK_slapy3 LAPACK_GLOBAL(slapy3,SLAPY3) -#define LAPACK_dlapy3 LAPACK_GLOBAL(dlapy3,DLAPY3) -#define LAPACK_slartgp LAPACK_GLOBAL(slartgp,SLARTGP) -#define LAPACK_dlartgp LAPACK_GLOBAL(dlartgp,DLARTGP) -#define LAPACK_slartgs LAPACK_GLOBAL(slartgs,SLARTGS) -#define LAPACK_dlartgs LAPACK_GLOBAL(dlartgs,DLARTGS) +lapack_int LAPACKE_sbdsdc(int matrix_order, + char uplo, + char compq, + lapack_int n, + float *d, + float *e, + float *u, + lapack_int ldu, + float *vt, + lapack_int ldvt, + float *q, + lapack_int *iq); +lapack_int LAPACKE_dbdsdc(int matrix_order, + char uplo, + char compq, + lapack_int n, + double *d, + double *e, + double *u, + lapack_int ldu, + double *vt, + lapack_int ldvt, + double *q, + lapack_int *iq); + +lapack_int LAPACKE_sbdsqr(int matrix_order, + char uplo, + lapack_int n, + lapack_int ncvt, + lapack_int nru, + lapack_int ncc, + float *d, + float *e, + float *vt, + lapack_int ldvt, + float *u, + lapack_int ldu, + float *c, + lapack_int ldc); +lapack_int LAPACKE_dbdsqr(int matrix_order, + char uplo, + lapack_int n, + lapack_int ncvt, + lapack_int nru, + lapack_int ncc, + double *d, + double *e, + double *vt, + lapack_int ldvt, + double *u, + lapack_int ldu, + double *c, + lapack_int ldc); +lapack_int LAPACKE_cbdsqr(int matrix_order, + char uplo, + lapack_int n, + lapack_int ncvt, + lapack_int nru, + lapack_int ncc, + float *d, + float *e, + lapack_complex_float *vt, + lapack_int ldvt, + lapack_complex_float *u, + lapack_int ldu, + lapack_complex_float *c, + lapack_int ldc); +lapack_int LAPACKE_zbdsqr(int matrix_order, + char uplo, + lapack_int n, + lapack_int ncvt, + lapack_int nru, + lapack_int ncc, + double *d, + double *e, + lapack_complex_double *vt, + lapack_int ldvt, + lapack_complex_double *u, + lapack_int ldu, + lapack_complex_double *c, + lapack_int ldc); + +lapack_int LAPACKE_sdisna(char job, lapack_int m, lapack_int n, const float *d, float *sep); +lapack_int LAPACKE_ddisna(char job, lapack_int m, lapack_int n, const double *d, double *sep); + +lapack_int LAPACKE_sgbbrd(int matrix_order, + char vect, + lapack_int m, + lapack_int n, + lapack_int ncc, + lapack_int kl, + lapack_int ku, + float *ab, + lapack_int ldab, + float *d, + float *e, + float *q, + lapack_int ldq, + float *pt, + lapack_int ldpt, + float *c, + lapack_int ldc); +lapack_int LAPACKE_dgbbrd(int matrix_order, + char vect, + lapack_int m, + lapack_int n, + lapack_int ncc, + lapack_int kl, + lapack_int ku, + double *ab, + lapack_int ldab, + double *d, + double *e, + double *q, + lapack_int ldq, + double *pt, + lapack_int ldpt, + double *c, + lapack_int ldc); +lapack_int LAPACKE_cgbbrd(int matrix_order, + char vect, + lapack_int m, + lapack_int n, + lapack_int ncc, + lapack_int kl, + lapack_int ku, + lapack_complex_float *ab, + lapack_int ldab, + float *d, + float *e, + lapack_complex_float *q, + lapack_int ldq, + lapack_complex_float *pt, + lapack_int ldpt, + lapack_complex_float *c, + lapack_int ldc); +lapack_int LAPACKE_zgbbrd(int matrix_order, + char vect, + lapack_int m, + lapack_int n, + lapack_int ncc, + lapack_int kl, + lapack_int ku, + lapack_complex_double *ab, + lapack_int ldab, + double *d, + double *e, + lapack_complex_double *q, + lapack_int ldq, + lapack_complex_double *pt, + lapack_int ldpt, + lapack_complex_double *c, + lapack_int ldc); + +lapack_int LAPACKE_sgbcon(int matrix_order, + char norm, + lapack_int n, + lapack_int kl, + lapack_int ku, + const float *ab, + lapack_int ldab, + const lapack_int *ipiv, + float anorm, + float *rcond); +lapack_int LAPACKE_dgbcon(int matrix_order, + char norm, + lapack_int n, + lapack_int kl, + lapack_int ku, + const double *ab, + lapack_int ldab, + const lapack_int *ipiv, + double anorm, + double *rcond); +lapack_int LAPACKE_cgbcon(int matrix_order, + char norm, + lapack_int n, + lapack_int kl, + lapack_int ku, + const lapack_complex_float *ab, + lapack_int ldab, + const lapack_int *ipiv, + float anorm, + float *rcond); +lapack_int LAPACKE_zgbcon(int matrix_order, + char norm, + lapack_int n, + lapack_int kl, + lapack_int ku, + const lapack_complex_double *ab, + lapack_int ldab, + const lapack_int *ipiv, + double anorm, + double *rcond); + +lapack_int LAPACKE_sgbequ(int matrix_order, + lapack_int m, + lapack_int n, + lapack_int kl, + lapack_int ku, + const float *ab, + lapack_int ldab, + float *r, + float *c, + float *rowcnd, + float *colcnd, + float *amax); +lapack_int LAPACKE_dgbequ(int matrix_order, + lapack_int m, + lapack_int n, + lapack_int kl, + lapack_int ku, + const double *ab, + lapack_int ldab, + double *r, + double *c, + double *rowcnd, + double *colcnd, + double *amax); +lapack_int LAPACKE_cgbequ(int matrix_order, + lapack_int m, + lapack_int n, + lapack_int kl, + lapack_int ku, + const lapack_complex_float *ab, + lapack_int ldab, + float *r, + float *c, + float *rowcnd, + float *colcnd, + float *amax); +lapack_int LAPACKE_zgbequ(int matrix_order, + lapack_int m, + lapack_int n, + lapack_int kl, + lapack_int ku, + const lapack_complex_double *ab, + lapack_int ldab, + double *r, + double *c, + double *rowcnd, + double *colcnd, + double *amax); + +lapack_int LAPACKE_sgbequb(int matrix_order, + lapack_int m, + lapack_int n, + lapack_int kl, + lapack_int ku, + const float *ab, + lapack_int ldab, + float *r, + float *c, + float *rowcnd, + float *colcnd, + float *amax); +lapack_int LAPACKE_dgbequb(int matrix_order, + lapack_int m, + lapack_int n, + lapack_int kl, + lapack_int ku, + const double *ab, + lapack_int ldab, + double *r, + double *c, + double *rowcnd, + double *colcnd, + double *amax); +lapack_int LAPACKE_cgbequb(int matrix_order, + lapack_int m, + lapack_int n, + lapack_int kl, + lapack_int ku, + const lapack_complex_float *ab, + lapack_int ldab, + float *r, + float *c, + float *rowcnd, + float *colcnd, + float *amax); +lapack_int LAPACKE_zgbequb(int matrix_order, + lapack_int m, + lapack_int n, + lapack_int kl, + lapack_int ku, + const lapack_complex_double *ab, + lapack_int ldab, + double *r, + double *c, + double *rowcnd, + double *colcnd, + double *amax); + +lapack_int LAPACKE_sgbrfs(int matrix_order, + char trans, + lapack_int n, + lapack_int kl, + lapack_int ku, + lapack_int nrhs, + const float *ab, + lapack_int ldab, + const float *afb, + lapack_int ldafb, + const lapack_int *ipiv, + const float *b, + lapack_int ldb, + float *x, + lapack_int ldx, + float *ferr, + float *berr); +lapack_int LAPACKE_dgbrfs(int matrix_order, + char trans, + lapack_int n, + lapack_int kl, + lapack_int ku, + lapack_int nrhs, + const double *ab, + lapack_int ldab, + const double *afb, + lapack_int ldafb, + const lapack_int *ipiv, + const double *b, + lapack_int ldb, + double *x, + lapack_int ldx, + double *ferr, + double *berr); +lapack_int LAPACKE_cgbrfs(int matrix_order, + char trans, + lapack_int n, + lapack_int kl, + lapack_int ku, + lapack_int nrhs, + const lapack_complex_float *ab, + lapack_int ldab, + const lapack_complex_float *afb, + lapack_int ldafb, + const lapack_int *ipiv, + const lapack_complex_float *b, + lapack_int ldb, + lapack_complex_float *x, + lapack_int ldx, + float *ferr, + float *berr); +lapack_int LAPACKE_zgbrfs(int matrix_order, + char trans, + lapack_int n, + lapack_int kl, + lapack_int ku, + lapack_int nrhs, + const lapack_complex_double *ab, + lapack_int ldab, + const lapack_complex_double *afb, + lapack_int ldafb, + const lapack_int *ipiv, + const lapack_complex_double *b, + lapack_int ldb, + lapack_complex_double *x, + lapack_int ldx, + double *ferr, + double *berr); + +lapack_int LAPACKE_sgbrfsx(int matrix_order, + char trans, + char equed, + lapack_int n, + lapack_int kl, + lapack_int ku, + lapack_int nrhs, + const float *ab, + lapack_int ldab, + const float *afb, + lapack_int ldafb, + const lapack_int *ipiv, + const float *r, + const float *c, + const float *b, + lapack_int ldb, + float *x, + lapack_int ldx, + float *rcond, + float *berr, + lapack_int n_err_bnds, + float *err_bnds_norm, + float *err_bnds_comp, + lapack_int nparams, + float *params); +lapack_int LAPACKE_dgbrfsx(int matrix_order, + char trans, + char equed, + lapack_int n, + lapack_int kl, + lapack_int ku, + lapack_int nrhs, + const double *ab, + lapack_int ldab, + const double *afb, + lapack_int ldafb, + const lapack_int *ipiv, + const double *r, + const double *c, + const double *b, + lapack_int ldb, + double *x, + lapack_int ldx, + double *rcond, + double *berr, + lapack_int n_err_bnds, + double *err_bnds_norm, + double *err_bnds_comp, + lapack_int nparams, + double *params); +lapack_int LAPACKE_cgbrfsx(int matrix_order, + char trans, + char equed, + lapack_int n, + lapack_int kl, + lapack_int ku, + lapack_int nrhs, + const lapack_complex_float *ab, + lapack_int ldab, + const lapack_complex_float *afb, + lapack_int ldafb, + const lapack_int *ipiv, + const float *r, + const float *c, + const lapack_complex_float *b, + lapack_int ldb, + lapack_complex_float *x, + lapack_int ldx, + float *rcond, + float *berr, + lapack_int n_err_bnds, + float *err_bnds_norm, + float *err_bnds_comp, + lapack_int nparams, + float *params); +lapack_int LAPACKE_zgbrfsx(int matrix_order, + char trans, + char equed, + lapack_int n, + lapack_int kl, + lapack_int ku, + lapack_int nrhs, + const lapack_complex_double *ab, + lapack_int ldab, + const lapack_complex_double *afb, + lapack_int ldafb, + const lapack_int *ipiv, + const double *r, + const double *c, + const lapack_complex_double *b, + lapack_int ldb, + lapack_complex_double *x, + lapack_int ldx, + double *rcond, + double *berr, + lapack_int n_err_bnds, + double *err_bnds_norm, + double *err_bnds_comp, + lapack_int nparams, + double *params); + +lapack_int LAPACKE_sgbsv(int matrix_order, + lapack_int n, + lapack_int kl, + lapack_int ku, + lapack_int nrhs, + float *ab, + lapack_int ldab, + lapack_int *ipiv, + float *b, + lapack_int ldb); +lapack_int LAPACKE_dgbsv(int matrix_order, + lapack_int n, + lapack_int kl, + lapack_int ku, + lapack_int nrhs, + double *ab, + lapack_int ldab, + lapack_int *ipiv, + double *b, + lapack_int ldb); +lapack_int LAPACKE_cgbsv(int matrix_order, + lapack_int n, + lapack_int kl, + lapack_int ku, + lapack_int nrhs, + lapack_complex_float *ab, + lapack_int ldab, + lapack_int *ipiv, + lapack_complex_float *b, + lapack_int ldb); +lapack_int LAPACKE_zgbsv(int matrix_order, + lapack_int n, + lapack_int kl, + lapack_int ku, + lapack_int nrhs, + lapack_complex_double *ab, + lapack_int ldab, + lapack_int *ipiv, + lapack_complex_double *b, + lapack_int ldb); + +lapack_int LAPACKE_sgbsvx(int matrix_order, + char fact, + char trans, + lapack_int n, + lapack_int kl, + lapack_int ku, + lapack_int nrhs, + float *ab, + lapack_int ldab, + float *afb, + lapack_int ldafb, + lapack_int *ipiv, + char *equed, + float *r, + float *c, + float *b, + lapack_int ldb, + float *x, + lapack_int ldx, + float *rcond, + float *ferr, + float *berr, + float *rpivot); +lapack_int LAPACKE_dgbsvx(int matrix_order, + char fact, + char trans, + lapack_int n, + lapack_int kl, + lapack_int ku, + lapack_int nrhs, + double *ab, + lapack_int ldab, + double *afb, + lapack_int ldafb, + lapack_int *ipiv, + char *equed, + double *r, + double *c, + double *b, + lapack_int ldb, + double *x, + lapack_int ldx, + double *rcond, + double *ferr, + double *berr, + double *rpivot); +lapack_int LAPACKE_cgbsvx(int matrix_order, + char fact, + char trans, + lapack_int n, + lapack_int kl, + lapack_int ku, + lapack_int nrhs, + lapack_complex_float *ab, + lapack_int ldab, + lapack_complex_float *afb, + lapack_int ldafb, + lapack_int *ipiv, + char *equed, + float *r, + float *c, + lapack_complex_float *b, + lapack_int ldb, + lapack_complex_float *x, + lapack_int ldx, + float *rcond, + float *ferr, + float *berr, + float *rpivot); +lapack_int LAPACKE_zgbsvx(int matrix_order, + char fact, + char trans, + lapack_int n, + lapack_int kl, + lapack_int ku, + lapack_int nrhs, + lapack_complex_double *ab, + lapack_int ldab, + lapack_complex_double *afb, + lapack_int ldafb, + lapack_int *ipiv, + char *equed, + double *r, + double *c, + lapack_complex_double *b, + lapack_int ldb, + lapack_complex_double *x, + lapack_int ldx, + double *rcond, + double *ferr, + double *berr, + double *rpivot); + +lapack_int LAPACKE_sgbsvxx(int matrix_order, + char fact, + char trans, + lapack_int n, + lapack_int kl, + lapack_int ku, + lapack_int nrhs, + float *ab, + lapack_int ldab, + float *afb, + lapack_int ldafb, + lapack_int *ipiv, + char *equed, + float *r, + float *c, + float *b, + lapack_int ldb, + float *x, + lapack_int ldx, + float *rcond, + float *rpvgrw, + float *berr, + lapack_int n_err_bnds, + float *err_bnds_norm, + float *err_bnds_comp, + lapack_int nparams, + float *params); +lapack_int LAPACKE_dgbsvxx(int matrix_order, + char fact, + char trans, + lapack_int n, + lapack_int kl, + lapack_int ku, + lapack_int nrhs, + double *ab, + lapack_int ldab, + double *afb, + lapack_int ldafb, + lapack_int *ipiv, + char *equed, + double *r, + double *c, + double *b, + lapack_int ldb, + double *x, + lapack_int ldx, + double *rcond, + double *rpvgrw, + double *berr, + lapack_int n_err_bnds, + double *err_bnds_norm, + double *err_bnds_comp, + lapack_int nparams, + double *params); +lapack_int LAPACKE_cgbsvxx(int matrix_order, + char fact, + char trans, + lapack_int n, + lapack_int kl, + lapack_int ku, + lapack_int nrhs, + lapack_complex_float *ab, + lapack_int ldab, + lapack_complex_float *afb, + lapack_int ldafb, + lapack_int *ipiv, + char *equed, + float *r, + float *c, + lapack_complex_float *b, + lapack_int ldb, + lapack_complex_float *x, + lapack_int ldx, + float *rcond, + float *rpvgrw, + float *berr, + lapack_int n_err_bnds, + float *err_bnds_norm, + float *err_bnds_comp, + lapack_int nparams, + float *params); +lapack_int LAPACKE_zgbsvxx(int matrix_order, + char fact, + char trans, + lapack_int n, + lapack_int kl, + lapack_int ku, + lapack_int nrhs, + lapack_complex_double *ab, + lapack_int ldab, + lapack_complex_double *afb, + lapack_int ldafb, + lapack_int *ipiv, + char *equed, + double *r, + double *c, + lapack_complex_double *b, + lapack_int ldb, + lapack_complex_double *x, + lapack_int ldx, + double *rcond, + double *rpvgrw, + double *berr, + lapack_int n_err_bnds, + double *err_bnds_norm, + double *err_bnds_comp, + lapack_int nparams, + double *params); + +lapack_int LAPACKE_sgbtrf(int matrix_order, + lapack_int m, + lapack_int n, + lapack_int kl, + lapack_int ku, + float *ab, + lapack_int ldab, + lapack_int *ipiv); +lapack_int LAPACKE_dgbtrf(int matrix_order, + lapack_int m, + lapack_int n, + lapack_int kl, + lapack_int ku, + double *ab, + lapack_int ldab, + lapack_int *ipiv); +lapack_int LAPACKE_cgbtrf(int matrix_order, + lapack_int m, + lapack_int n, + lapack_int kl, + lapack_int ku, + lapack_complex_float *ab, + lapack_int ldab, + lapack_int *ipiv); +lapack_int LAPACKE_zgbtrf(int matrix_order, + lapack_int m, + lapack_int n, + lapack_int kl, + lapack_int ku, + lapack_complex_double *ab, + lapack_int ldab, + lapack_int *ipiv); + +lapack_int LAPACKE_sgbtrs(int matrix_order, + char trans, + lapack_int n, + lapack_int kl, + lapack_int ku, + lapack_int nrhs, + const float *ab, + lapack_int ldab, + const lapack_int *ipiv, + float *b, + lapack_int ldb); +lapack_int LAPACKE_dgbtrs(int matrix_order, + char trans, + lapack_int n, + lapack_int kl, + lapack_int ku, + lapack_int nrhs, + const double *ab, + lapack_int ldab, + const lapack_int *ipiv, + double *b, + lapack_int ldb); +lapack_int LAPACKE_cgbtrs(int matrix_order, + char trans, + lapack_int n, + lapack_int kl, + lapack_int ku, + lapack_int nrhs, + const lapack_complex_float *ab, + lapack_int ldab, + const lapack_int *ipiv, + lapack_complex_float *b, + lapack_int ldb); +lapack_int LAPACKE_zgbtrs(int matrix_order, + char trans, + lapack_int n, + lapack_int kl, + lapack_int ku, + lapack_int nrhs, + const lapack_complex_double *ab, + lapack_int ldab, + const lapack_int *ipiv, + lapack_complex_double *b, + lapack_int ldb); + +lapack_int LAPACKE_sgebak(int matrix_order, + char job, + char side, + lapack_int n, + lapack_int ilo, + lapack_int ihi, + const float *scale, + lapack_int m, + float *v, + lapack_int ldv); +lapack_int LAPACKE_dgebak(int matrix_order, + char job, + char side, + lapack_int n, + lapack_int ilo, + lapack_int ihi, + const double *scale, + lapack_int m, + double *v, + lapack_int ldv); +lapack_int LAPACKE_cgebak(int matrix_order, + char job, + char side, + lapack_int n, + lapack_int ilo, + lapack_int ihi, + const float *scale, + lapack_int m, + lapack_complex_float *v, + lapack_int ldv); +lapack_int LAPACKE_zgebak(int matrix_order, + char job, + char side, + lapack_int n, + lapack_int ilo, + lapack_int ihi, + const double *scale, + lapack_int m, + lapack_complex_double *v, + lapack_int ldv); + +lapack_int LAPACKE_sgebal(int matrix_order, + char job, + lapack_int n, + float *a, + lapack_int lda, + lapack_int *ilo, + lapack_int *ihi, + float *scale); +lapack_int LAPACKE_dgebal(int matrix_order, + char job, + lapack_int n, + double *a, + lapack_int lda, + lapack_int *ilo, + lapack_int *ihi, + double *scale); +lapack_int LAPACKE_cgebal(int matrix_order, + char job, + lapack_int n, + lapack_complex_float *a, + lapack_int lda, + lapack_int *ilo, + lapack_int *ihi, + float *scale); +lapack_int LAPACKE_zgebal(int matrix_order, + char job, + lapack_int n, + lapack_complex_double *a, + lapack_int lda, + lapack_int *ilo, + lapack_int *ihi, + double *scale); + +lapack_int LAPACKE_sgebrd(int matrix_order, + lapack_int m, + lapack_int n, + float *a, + lapack_int lda, + float *d, + float *e, + float *tauq, + float *taup); +lapack_int LAPACKE_dgebrd(int matrix_order, + lapack_int m, + lapack_int n, + double *a, + lapack_int lda, + double *d, + double *e, + double *tauq, + double *taup); +lapack_int LAPACKE_cgebrd(int matrix_order, + lapack_int m, + lapack_int n, + lapack_complex_float *a, + lapack_int lda, + float *d, + float *e, + lapack_complex_float *tauq, + lapack_complex_float *taup); +lapack_int LAPACKE_zgebrd(int matrix_order, + lapack_int m, + lapack_int n, + lapack_complex_double *a, + lapack_int lda, + double *d, + double *e, + lapack_complex_double *tauq, + lapack_complex_double *taup); + +lapack_int + LAPACKE_sgecon(int matrix_order, char norm, lapack_int n, const float *a, lapack_int lda, float anorm, float *rcond); +lapack_int LAPACKE_dgecon(int matrix_order, + char norm, + lapack_int n, + const double *a, + lapack_int lda, + double anorm, + double *rcond); +lapack_int LAPACKE_cgecon(int matrix_order, + char norm, + lapack_int n, + const lapack_complex_float *a, + lapack_int lda, + float anorm, + float *rcond); +lapack_int LAPACKE_zgecon(int matrix_order, + char norm, + lapack_int n, + const lapack_complex_double *a, + lapack_int lda, + double anorm, + double *rcond); + +lapack_int LAPACKE_sgeequ(int matrix_order, + lapack_int m, + lapack_int n, + const float *a, + lapack_int lda, + float *r, + float *c, + float *rowcnd, + float *colcnd, + float *amax); +lapack_int LAPACKE_dgeequ(int matrix_order, + lapack_int m, + lapack_int n, + const double *a, + lapack_int lda, + double *r, + double *c, + double *rowcnd, + double *colcnd, + double *amax); +lapack_int LAPACKE_cgeequ(int matrix_order, + lapack_int m, + lapack_int n, + const lapack_complex_float *a, + lapack_int lda, + float *r, + float *c, + float *rowcnd, + float *colcnd, + float *amax); +lapack_int LAPACKE_zgeequ(int matrix_order, + lapack_int m, + lapack_int n, + const lapack_complex_double *a, + lapack_int lda, + double *r, + double *c, + double *rowcnd, + double *colcnd, + double *amax); + +lapack_int LAPACKE_sgeequb(int matrix_order, + lapack_int m, + lapack_int n, + const float *a, + lapack_int lda, + float *r, + float *c, + float *rowcnd, + float *colcnd, + float *amax); +lapack_int LAPACKE_dgeequb(int matrix_order, + lapack_int m, + lapack_int n, + const double *a, + lapack_int lda, + double *r, + double *c, + double *rowcnd, + double *colcnd, + double *amax); +lapack_int LAPACKE_cgeequb(int matrix_order, + lapack_int m, + lapack_int n, + const lapack_complex_float *a, + lapack_int lda, + float *r, + float *c, + float *rowcnd, + float *colcnd, + float *amax); +lapack_int LAPACKE_zgeequb(int matrix_order, + lapack_int m, + lapack_int n, + const lapack_complex_double *a, + lapack_int lda, + double *r, + double *c, + double *rowcnd, + double *colcnd, + double *amax); + +lapack_int LAPACKE_sgees(int matrix_order, + char jobvs, + char sort, + LAPACK_S_SELECT2 select, + lapack_int n, + float *a, + lapack_int lda, + lapack_int *sdim, + float *wr, + float *wi, + float *vs, + lapack_int ldvs); +lapack_int LAPACKE_dgees(int matrix_order, + char jobvs, + char sort, + LAPACK_D_SELECT2 select, + lapack_int n, + double *a, + lapack_int lda, + lapack_int *sdim, + double *wr, + double *wi, + double *vs, + lapack_int ldvs); +lapack_int LAPACKE_cgees(int matrix_order, + char jobvs, + char sort, + LAPACK_C_SELECT1 select, + lapack_int n, + lapack_complex_float *a, + lapack_int lda, + lapack_int *sdim, + lapack_complex_float *w, + lapack_complex_float *vs, + lapack_int ldvs); +lapack_int LAPACKE_zgees(int matrix_order, + char jobvs, + char sort, + LAPACK_Z_SELECT1 select, + lapack_int n, + lapack_complex_double *a, + lapack_int lda, + lapack_int *sdim, + lapack_complex_double *w, + lapack_complex_double *vs, + lapack_int ldvs); + +lapack_int LAPACKE_sgeesx(int matrix_order, + char jobvs, + char sort, + LAPACK_S_SELECT2 select, + char sense, + lapack_int n, + float *a, + lapack_int lda, + lapack_int *sdim, + float *wr, + float *wi, + float *vs, + lapack_int ldvs, + float *rconde, + float *rcondv); +lapack_int LAPACKE_dgeesx(int matrix_order, + char jobvs, + char sort, + LAPACK_D_SELECT2 select, + char sense, + lapack_int n, + double *a, + lapack_int lda, + lapack_int *sdim, + double *wr, + double *wi, + double *vs, + lapack_int ldvs, + double *rconde, + double *rcondv); +lapack_int LAPACKE_cgeesx(int matrix_order, + char jobvs, + char sort, + LAPACK_C_SELECT1 select, + char sense, + lapack_int n, + lapack_complex_float *a, + lapack_int lda, + lapack_int *sdim, + lapack_complex_float *w, + lapack_complex_float *vs, + lapack_int ldvs, + float *rconde, + float *rcondv); +lapack_int LAPACKE_zgeesx(int matrix_order, + char jobvs, + char sort, + LAPACK_Z_SELECT1 select, + char sense, + lapack_int n, + lapack_complex_double *a, + lapack_int lda, + lapack_int *sdim, + lapack_complex_double *w, + lapack_complex_double *vs, + lapack_int ldvs, + double *rconde, + double *rcondv); + +lapack_int LAPACKE_sgeev(int matrix_order, + char jobvl, + char jobvr, + lapack_int n, + float *a, + lapack_int lda, + float *wr, + float *wi, + float *vl, + lapack_int ldvl, + float *vr, + lapack_int ldvr); +lapack_int LAPACKE_dgeev(int matrix_order, + char jobvl, + char jobvr, + lapack_int n, + double *a, + lapack_int lda, + double *wr, + double *wi, + double *vl, + lapack_int ldvl, + double *vr, + lapack_int ldvr); +lapack_int LAPACKE_cgeev(int matrix_order, + char jobvl, + char jobvr, + lapack_int n, + lapack_complex_float *a, + lapack_int lda, + lapack_complex_float *w, + lapack_complex_float *vl, + lapack_int ldvl, + lapack_complex_float *vr, + lapack_int ldvr); +lapack_int LAPACKE_zgeev(int matrix_order, + char jobvl, + char jobvr, + lapack_int n, + lapack_complex_double *a, + lapack_int lda, + lapack_complex_double *w, + lapack_complex_double *vl, + lapack_int ldvl, + lapack_complex_double *vr, + lapack_int ldvr); + +lapack_int LAPACKE_sgeevx(int matrix_order, + char balanc, + char jobvl, + char jobvr, + char sense, + lapack_int n, + float *a, + lapack_int lda, + float *wr, + float *wi, + float *vl, + lapack_int ldvl, + float *vr, + lapack_int ldvr, + lapack_int *ilo, + lapack_int *ihi, + float *scale, + float *abnrm, + float *rconde, + float *rcondv); +lapack_int LAPACKE_dgeevx(int matrix_order, + char balanc, + char jobvl, + char jobvr, + char sense, + lapack_int n, + double *a, + lapack_int lda, + double *wr, + double *wi, + double *vl, + lapack_int ldvl, + double *vr, + lapack_int ldvr, + lapack_int *ilo, + lapack_int *ihi, + double *scale, + double *abnrm, + double *rconde, + double *rcondv); +lapack_int LAPACKE_cgeevx(int matrix_order, + char balanc, + char jobvl, + char jobvr, + char sense, + lapack_int n, + lapack_complex_float *a, + lapack_int lda, + lapack_complex_float *w, + lapack_complex_float *vl, + lapack_int ldvl, + lapack_complex_float *vr, + lapack_int ldvr, + lapack_int *ilo, + lapack_int *ihi, + float *scale, + float *abnrm, + float *rconde, + float *rcondv); +lapack_int LAPACKE_zgeevx(int matrix_order, + char balanc, + char jobvl, + char jobvr, + char sense, + lapack_int n, + lapack_complex_double *a, + lapack_int lda, + lapack_complex_double *w, + lapack_complex_double *vl, + lapack_int ldvl, + lapack_complex_double *vr, + lapack_int ldvr, + lapack_int *ilo, + lapack_int *ihi, + double *scale, + double *abnrm, + double *rconde, + double *rcondv); + +lapack_int + LAPACKE_sgehrd(int matrix_order, lapack_int n, lapack_int ilo, lapack_int ihi, float *a, lapack_int lda, float *tau); +lapack_int LAPACKE_dgehrd(int matrix_order, + lapack_int n, + lapack_int ilo, + lapack_int ihi, + double *a, + lapack_int lda, + double *tau); +lapack_int LAPACKE_cgehrd(int matrix_order, + lapack_int n, + lapack_int ilo, + lapack_int ihi, + lapack_complex_float *a, + lapack_int lda, + lapack_complex_float *tau); +lapack_int LAPACKE_zgehrd(int matrix_order, + lapack_int n, + lapack_int ilo, + lapack_int ihi, + lapack_complex_double *a, + lapack_int lda, + lapack_complex_double *tau); + +lapack_int LAPACKE_sgejsv(int matrix_order, + char joba, + char jobu, + char jobv, + char jobr, + char jobt, + char jobp, + lapack_int m, + lapack_int n, + float *a, + lapack_int lda, + float *sva, + float *u, + lapack_int ldu, + float *v, + lapack_int ldv, + float *stat, + lapack_int *istat); +lapack_int LAPACKE_dgejsv(int matrix_order, + char joba, + char jobu, + char jobv, + char jobr, + char jobt, + char jobp, + lapack_int m, + lapack_int n, + double *a, + lapack_int lda, + double *sva, + double *u, + lapack_int ldu, + double *v, + lapack_int ldv, + double *stat, + lapack_int *istat); + +lapack_int LAPACKE_sgelq2(int matrix_order, lapack_int m, lapack_int n, float *a, lapack_int lda, float *tau); +lapack_int LAPACKE_dgelq2(int matrix_order, lapack_int m, lapack_int n, double *a, lapack_int lda, double *tau); +lapack_int LAPACKE_cgelq2(int matrix_order, + lapack_int m, + lapack_int n, + lapack_complex_float *a, + lapack_int lda, + lapack_complex_float *tau); +lapack_int LAPACKE_zgelq2(int matrix_order, + lapack_int m, + lapack_int n, + lapack_complex_double *a, + lapack_int lda, + lapack_complex_double *tau); + +lapack_int LAPACKE_sgelqf(int matrix_order, lapack_int m, lapack_int n, float *a, lapack_int lda, float *tau); +lapack_int LAPACKE_dgelqf(int matrix_order, lapack_int m, lapack_int n, double *a, lapack_int lda, double *tau); +lapack_int LAPACKE_cgelqf(int matrix_order, + lapack_int m, + lapack_int n, + lapack_complex_float *a, + lapack_int lda, + lapack_complex_float *tau); +lapack_int LAPACKE_zgelqf(int matrix_order, + lapack_int m, + lapack_int n, + lapack_complex_double *a, + lapack_int lda, + lapack_complex_double *tau); + +lapack_int LAPACKE_sgels(int matrix_order, + char trans, + lapack_int m, + lapack_int n, + lapack_int nrhs, + float *a, + lapack_int lda, + float *b, + lapack_int ldb); +lapack_int LAPACKE_dgels(int matrix_order, + char trans, + lapack_int m, + lapack_int n, + lapack_int nrhs, + double *a, + lapack_int lda, + double *b, + lapack_int ldb); +lapack_int LAPACKE_cgels(int matrix_order, + char trans, + lapack_int m, + lapack_int n, + lapack_int nrhs, + lapack_complex_float *a, + lapack_int lda, + lapack_complex_float *b, + lapack_int ldb); +lapack_int LAPACKE_zgels(int matrix_order, + char trans, + lapack_int m, + lapack_int n, + lapack_int nrhs, + lapack_complex_double *a, + lapack_int lda, + lapack_complex_double *b, + lapack_int ldb); + +lapack_int LAPACKE_sgelsd(int matrix_order, + lapack_int m, + lapack_int n, + lapack_int nrhs, + float *a, + lapack_int lda, + float *b, + lapack_int ldb, + float *s, + float rcond, + lapack_int *rank); +lapack_int LAPACKE_dgelsd(int matrix_order, + lapack_int m, + lapack_int n, + lapack_int nrhs, + double *a, + lapack_int lda, + double *b, + lapack_int ldb, + double *s, + double rcond, + lapack_int *rank); +lapack_int LAPACKE_cgelsd(int matrix_order, + lapack_int m, + lapack_int n, + lapack_int nrhs, + lapack_complex_float *a, + lapack_int lda, + lapack_complex_float *b, + lapack_int ldb, + float *s, + float rcond, + lapack_int *rank); +lapack_int LAPACKE_zgelsd(int matrix_order, + lapack_int m, + lapack_int n, + lapack_int nrhs, + lapack_complex_double *a, + lapack_int lda, + lapack_complex_double *b, + lapack_int ldb, + double *s, + double rcond, + lapack_int *rank); + +lapack_int LAPACKE_sgelss(int matrix_order, + lapack_int m, + lapack_int n, + lapack_int nrhs, + float *a, + lapack_int lda, + float *b, + lapack_int ldb, + float *s, + float rcond, + lapack_int *rank); +lapack_int LAPACKE_dgelss(int matrix_order, + lapack_int m, + lapack_int n, + lapack_int nrhs, + double *a, + lapack_int lda, + double *b, + lapack_int ldb, + double *s, + double rcond, + lapack_int *rank); +lapack_int LAPACKE_cgelss(int matrix_order, + lapack_int m, + lapack_int n, + lapack_int nrhs, + lapack_complex_float *a, + lapack_int lda, + lapack_complex_float *b, + lapack_int ldb, + float *s, + float rcond, + lapack_int *rank); +lapack_int LAPACKE_zgelss(int matrix_order, + lapack_int m, + lapack_int n, + lapack_int nrhs, + lapack_complex_double *a, + lapack_int lda, + lapack_complex_double *b, + lapack_int ldb, + double *s, + double rcond, + lapack_int *rank); + +lapack_int LAPACKE_sgelsy(int matrix_order, + lapack_int m, + lapack_int n, + lapack_int nrhs, + float *a, + lapack_int lda, + float *b, + lapack_int ldb, + lapack_int *jpvt, + float rcond, + lapack_int *rank); +lapack_int LAPACKE_dgelsy(int matrix_order, + lapack_int m, + lapack_int n, + lapack_int nrhs, + double *a, + lapack_int lda, + double *b, + lapack_int ldb, + lapack_int *jpvt, + double rcond, + lapack_int *rank); +lapack_int LAPACKE_cgelsy(int matrix_order, + lapack_int m, + lapack_int n, + lapack_int nrhs, + lapack_complex_float *a, + lapack_int lda, + lapack_complex_float *b, + lapack_int ldb, + lapack_int *jpvt, + float rcond, + lapack_int *rank); +lapack_int LAPACKE_zgelsy(int matrix_order, + lapack_int m, + lapack_int n, + lapack_int nrhs, + lapack_complex_double *a, + lapack_int lda, + lapack_complex_double *b, + lapack_int ldb, + lapack_int *jpvt, + double rcond, + lapack_int *rank); + +lapack_int LAPACKE_sgeqlf(int matrix_order, lapack_int m, lapack_int n, float *a, lapack_int lda, float *tau); +lapack_int LAPACKE_dgeqlf(int matrix_order, lapack_int m, lapack_int n, double *a, lapack_int lda, double *tau); +lapack_int LAPACKE_cgeqlf(int matrix_order, + lapack_int m, + lapack_int n, + lapack_complex_float *a, + lapack_int lda, + lapack_complex_float *tau); +lapack_int LAPACKE_zgeqlf(int matrix_order, + lapack_int m, + lapack_int n, + lapack_complex_double *a, + lapack_int lda, + lapack_complex_double *tau); + +lapack_int + LAPACKE_sgeqp3(int matrix_order, lapack_int m, lapack_int n, float *a, lapack_int lda, lapack_int *jpvt, float *tau); +lapack_int LAPACKE_dgeqp3(int matrix_order, + lapack_int m, + lapack_int n, + double *a, + lapack_int lda, + lapack_int *jpvt, + double *tau); +lapack_int LAPACKE_cgeqp3(int matrix_order, + lapack_int m, + lapack_int n, + lapack_complex_float *a, + lapack_int lda, + lapack_int *jpvt, + lapack_complex_float *tau); +lapack_int LAPACKE_zgeqp3(int matrix_order, + lapack_int m, + lapack_int n, + lapack_complex_double *a, + lapack_int lda, + lapack_int *jpvt, + lapack_complex_double *tau); + +lapack_int + LAPACKE_sgeqpf(int matrix_order, lapack_int m, lapack_int n, float *a, lapack_int lda, lapack_int *jpvt, float *tau); +lapack_int LAPACKE_dgeqpf(int matrix_order, + lapack_int m, + lapack_int n, + double *a, + lapack_int lda, + lapack_int *jpvt, + double *tau); +lapack_int LAPACKE_cgeqpf(int matrix_order, + lapack_int m, + lapack_int n, + lapack_complex_float *a, + lapack_int lda, + lapack_int *jpvt, + lapack_complex_float *tau); +lapack_int LAPACKE_zgeqpf(int matrix_order, + lapack_int m, + lapack_int n, + lapack_complex_double *a, + lapack_int lda, + lapack_int *jpvt, + lapack_complex_double *tau); + +lapack_int LAPACKE_sgeqr2(int matrix_order, lapack_int m, lapack_int n, float *a, lapack_int lda, float *tau); +lapack_int LAPACKE_dgeqr2(int matrix_order, lapack_int m, lapack_int n, double *a, lapack_int lda, double *tau); +lapack_int LAPACKE_cgeqr2(int matrix_order, + lapack_int m, + lapack_int n, + lapack_complex_float *a, + lapack_int lda, + lapack_complex_float *tau); +lapack_int LAPACKE_zgeqr2(int matrix_order, + lapack_int m, + lapack_int n, + lapack_complex_double *a, + lapack_int lda, + lapack_complex_double *tau); + +lapack_int LAPACKE_sgeqrf(int matrix_order, lapack_int m, lapack_int n, float *a, lapack_int lda, float *tau); +lapack_int LAPACKE_dgeqrf(int matrix_order, lapack_int m, lapack_int n, double *a, lapack_int lda, double *tau); +lapack_int LAPACKE_cgeqrf(int matrix_order, + lapack_int m, + lapack_int n, + lapack_complex_float *a, + lapack_int lda, + lapack_complex_float *tau); +lapack_int LAPACKE_zgeqrf(int matrix_order, + lapack_int m, + lapack_int n, + lapack_complex_double *a, + lapack_int lda, + lapack_complex_double *tau); + +lapack_int LAPACKE_sgeqrfp(int matrix_order, lapack_int m, lapack_int n, float *a, lapack_int lda, float *tau); +lapack_int LAPACKE_dgeqrfp(int matrix_order, lapack_int m, lapack_int n, double *a, lapack_int lda, double *tau); +lapack_int LAPACKE_cgeqrfp(int matrix_order, + lapack_int m, + lapack_int n, + lapack_complex_float *a, + lapack_int lda, + lapack_complex_float *tau); +lapack_int LAPACKE_zgeqrfp(int matrix_order, + lapack_int m, + lapack_int n, + lapack_complex_double *a, + lapack_int lda, + lapack_complex_double *tau); + +lapack_int LAPACKE_sgerfs(int matrix_order, + char trans, + lapack_int n, + lapack_int nrhs, + const float *a, + lapack_int lda, + const float *af, + lapack_int ldaf, + const lapack_int *ipiv, + const float *b, + lapack_int ldb, + float *x, + lapack_int ldx, + float *ferr, + float *berr); +lapack_int LAPACKE_dgerfs(int matrix_order, + char trans, + lapack_int n, + lapack_int nrhs, + const double *a, + lapack_int lda, + const double *af, + lapack_int ldaf, + const lapack_int *ipiv, + const double *b, + lapack_int ldb, + double *x, + lapack_int ldx, + double *ferr, + double *berr); +lapack_int LAPACKE_cgerfs(int matrix_order, + char trans, + lapack_int n, + lapack_int nrhs, + const lapack_complex_float *a, + lapack_int lda, + const lapack_complex_float *af, + lapack_int ldaf, + const lapack_int *ipiv, + const lapack_complex_float *b, + lapack_int ldb, + lapack_complex_float *x, + lapack_int ldx, + float *ferr, + float *berr); +lapack_int LAPACKE_zgerfs(int matrix_order, + char trans, + lapack_int n, + lapack_int nrhs, + const lapack_complex_double *a, + lapack_int lda, + const lapack_complex_double *af, + lapack_int ldaf, + const lapack_int *ipiv, + const lapack_complex_double *b, + lapack_int ldb, + lapack_complex_double *x, + lapack_int ldx, + double *ferr, + double *berr); + +lapack_int LAPACKE_sgerfsx(int matrix_order, + char trans, + char equed, + lapack_int n, + lapack_int nrhs, + const float *a, + lapack_int lda, + const float *af, + lapack_int ldaf, + const lapack_int *ipiv, + const float *r, + const float *c, + const float *b, + lapack_int ldb, + float *x, + lapack_int ldx, + float *rcond, + float *berr, + lapack_int n_err_bnds, + float *err_bnds_norm, + float *err_bnds_comp, + lapack_int nparams, + float *params); +lapack_int LAPACKE_dgerfsx(int matrix_order, + char trans, + char equed, + lapack_int n, + lapack_int nrhs, + const double *a, + lapack_int lda, + const double *af, + lapack_int ldaf, + const lapack_int *ipiv, + const double *r, + const double *c, + const double *b, + lapack_int ldb, + double *x, + lapack_int ldx, + double *rcond, + double *berr, + lapack_int n_err_bnds, + double *err_bnds_norm, + double *err_bnds_comp, + lapack_int nparams, + double *params); +lapack_int LAPACKE_cgerfsx(int matrix_order, + char trans, + char equed, + lapack_int n, + lapack_int nrhs, + const lapack_complex_float *a, + lapack_int lda, + const lapack_complex_float *af, + lapack_int ldaf, + const lapack_int *ipiv, + const float *r, + const float *c, + const lapack_complex_float *b, + lapack_int ldb, + lapack_complex_float *x, + lapack_int ldx, + float *rcond, + float *berr, + lapack_int n_err_bnds, + float *err_bnds_norm, + float *err_bnds_comp, + lapack_int nparams, + float *params); +lapack_int LAPACKE_zgerfsx(int matrix_order, + char trans, + char equed, + lapack_int n, + lapack_int nrhs, + const lapack_complex_double *a, + lapack_int lda, + const lapack_complex_double *af, + lapack_int ldaf, + const lapack_int *ipiv, + const double *r, + const double *c, + const lapack_complex_double *b, + lapack_int ldb, + lapack_complex_double *x, + lapack_int ldx, + double *rcond, + double *berr, + lapack_int n_err_bnds, + double *err_bnds_norm, + double *err_bnds_comp, + lapack_int nparams, + double *params); + +lapack_int LAPACKE_sgerqf(int matrix_order, lapack_int m, lapack_int n, float *a, lapack_int lda, float *tau); +lapack_int LAPACKE_dgerqf(int matrix_order, lapack_int m, lapack_int n, double *a, lapack_int lda, double *tau); +lapack_int LAPACKE_cgerqf(int matrix_order, + lapack_int m, + lapack_int n, + lapack_complex_float *a, + lapack_int lda, + lapack_complex_float *tau); +lapack_int LAPACKE_zgerqf(int matrix_order, + lapack_int m, + lapack_int n, + lapack_complex_double *a, + lapack_int lda, + lapack_complex_double *tau); + +lapack_int LAPACKE_sgesdd(int matrix_order, + char jobz, + lapack_int m, + lapack_int n, + float *a, + lapack_int lda, + float *s, + float *u, + lapack_int ldu, + float *vt, + lapack_int ldvt); +lapack_int LAPACKE_dgesdd(int matrix_order, + char jobz, + lapack_int m, + lapack_int n, + double *a, + lapack_int lda, + double *s, + double *u, + lapack_int ldu, + double *vt, + lapack_int ldvt); +lapack_int LAPACKE_cgesdd(int matrix_order, + char jobz, + lapack_int m, + lapack_int n, + lapack_complex_float *a, + lapack_int lda, + float *s, + lapack_complex_float *u, + lapack_int ldu, + lapack_complex_float *vt, + lapack_int ldvt); +lapack_int LAPACKE_zgesdd(int matrix_order, + char jobz, + lapack_int m, + lapack_int n, + lapack_complex_double *a, + lapack_int lda, + double *s, + lapack_complex_double *u, + lapack_int ldu, + lapack_complex_double *vt, + lapack_int ldvt); + +lapack_int LAPACKE_sgesv(int matrix_order, + lapack_int n, + lapack_int nrhs, + float *a, + lapack_int lda, + lapack_int *ipiv, + float *b, + lapack_int ldb); +lapack_int LAPACKE_dgesv(int matrix_order, + lapack_int n, + lapack_int nrhs, + double *a, + lapack_int lda, + lapack_int *ipiv, + double *b, + lapack_int ldb); +lapack_int LAPACKE_cgesv(int matrix_order, + lapack_int n, + lapack_int nrhs, + lapack_complex_float *a, + lapack_int lda, + lapack_int *ipiv, + lapack_complex_float *b, + lapack_int ldb); +lapack_int LAPACKE_zgesv(int matrix_order, + lapack_int n, + lapack_int nrhs, + lapack_complex_double *a, + lapack_int lda, + lapack_int *ipiv, + lapack_complex_double *b, + lapack_int ldb); +lapack_int LAPACKE_dsgesv(int matrix_order, + lapack_int n, + lapack_int nrhs, + double *a, + lapack_int lda, + lapack_int *ipiv, + double *b, + lapack_int ldb, + double *x, + lapack_int ldx, + lapack_int *iter); +lapack_int LAPACKE_zcgesv(int matrix_order, + lapack_int n, + lapack_int nrhs, + lapack_complex_double *a, + lapack_int lda, + lapack_int *ipiv, + lapack_complex_double *b, + lapack_int ldb, + lapack_complex_double *x, + lapack_int ldx, + lapack_int *iter); + +lapack_int LAPACKE_sgesvd(int matrix_order, + char jobu, + char jobvt, + lapack_int m, + lapack_int n, + float *a, + lapack_int lda, + float *s, + float *u, + lapack_int ldu, + float *vt, + lapack_int ldvt, + float *superb); +lapack_int LAPACKE_dgesvd(int matrix_order, + char jobu, + char jobvt, + lapack_int m, + lapack_int n, + double *a, + lapack_int lda, + double *s, + double *u, + lapack_int ldu, + double *vt, + lapack_int ldvt, + double *superb); +lapack_int LAPACKE_cgesvd(int matrix_order, + char jobu, + char jobvt, + lapack_int m, + lapack_int n, + lapack_complex_float *a, + lapack_int lda, + float *s, + lapack_complex_float *u, + lapack_int ldu, + lapack_complex_float *vt, + lapack_int ldvt, + float *superb); +lapack_int LAPACKE_zgesvd(int matrix_order, + char jobu, + char jobvt, + lapack_int m, + lapack_int n, + lapack_complex_double *a, + lapack_int lda, + double *s, + lapack_complex_double *u, + lapack_int ldu, + lapack_complex_double *vt, + lapack_int ldvt, + double *superb); + +lapack_int LAPACKE_sgesvj(int matrix_order, + char joba, + char jobu, + char jobv, + lapack_int m, + lapack_int n, + float *a, + lapack_int lda, + float *sva, + lapack_int mv, + float *v, + lapack_int ldv, + float *stat); +lapack_int LAPACKE_dgesvj(int matrix_order, + char joba, + char jobu, + char jobv, + lapack_int m, + lapack_int n, + double *a, + lapack_int lda, + double *sva, + lapack_int mv, + double *v, + lapack_int ldv, + double *stat); + +lapack_int LAPACKE_sgesvx(int matrix_order, + char fact, + char trans, + lapack_int n, + lapack_int nrhs, + float *a, + lapack_int lda, + float *af, + lapack_int ldaf, + lapack_int *ipiv, + char *equed, + float *r, + float *c, + float *b, + lapack_int ldb, + float *x, + lapack_int ldx, + float *rcond, + float *ferr, + float *berr, + float *rpivot); +lapack_int LAPACKE_dgesvx(int matrix_order, + char fact, + char trans, + lapack_int n, + lapack_int nrhs, + double *a, + lapack_int lda, + double *af, + lapack_int ldaf, + lapack_int *ipiv, + char *equed, + double *r, + double *c, + double *b, + lapack_int ldb, + double *x, + lapack_int ldx, + double *rcond, + double *ferr, + double *berr, + double *rpivot); +lapack_int LAPACKE_cgesvx(int matrix_order, + char fact, + char trans, + lapack_int n, + lapack_int nrhs, + lapack_complex_float *a, + lapack_int lda, + lapack_complex_float *af, + lapack_int ldaf, + lapack_int *ipiv, + char *equed, + float *r, + float *c, + lapack_complex_float *b, + lapack_int ldb, + lapack_complex_float *x, + lapack_int ldx, + float *rcond, + float *ferr, + float *berr, + float *rpivot); +lapack_int LAPACKE_zgesvx(int matrix_order, + char fact, + char trans, + lapack_int n, + lapack_int nrhs, + lapack_complex_double *a, + lapack_int lda, + lapack_complex_double *af, + lapack_int ldaf, + lapack_int *ipiv, + char *equed, + double *r, + double *c, + lapack_complex_double *b, + lapack_int ldb, + lapack_complex_double *x, + lapack_int ldx, + double *rcond, + double *ferr, + double *berr, + double *rpivot); + +lapack_int LAPACKE_sgesvxx(int matrix_order, + char fact, + char trans, + lapack_int n, + lapack_int nrhs, + float *a, + lapack_int lda, + float *af, + lapack_int ldaf, + lapack_int *ipiv, + char *equed, + float *r, + float *c, + float *b, + lapack_int ldb, + float *x, + lapack_int ldx, + float *rcond, + float *rpvgrw, + float *berr, + lapack_int n_err_bnds, + float *err_bnds_norm, + float *err_bnds_comp, + lapack_int nparams, + float *params); +lapack_int LAPACKE_dgesvxx(int matrix_order, + char fact, + char trans, + lapack_int n, + lapack_int nrhs, + double *a, + lapack_int lda, + double *af, + lapack_int ldaf, + lapack_int *ipiv, + char *equed, + double *r, + double *c, + double *b, + lapack_int ldb, + double *x, + lapack_int ldx, + double *rcond, + double *rpvgrw, + double *berr, + lapack_int n_err_bnds, + double *err_bnds_norm, + double *err_bnds_comp, + lapack_int nparams, + double *params); +lapack_int LAPACKE_cgesvxx(int matrix_order, + char fact, + char trans, + lapack_int n, + lapack_int nrhs, + lapack_complex_float *a, + lapack_int lda, + lapack_complex_float *af, + lapack_int ldaf, + lapack_int *ipiv, + char *equed, + float *r, + float *c, + lapack_complex_float *b, + lapack_int ldb, + lapack_complex_float *x, + lapack_int ldx, + float *rcond, + float *rpvgrw, + float *berr, + lapack_int n_err_bnds, + float *err_bnds_norm, + float *err_bnds_comp, + lapack_int nparams, + float *params); +lapack_int LAPACKE_zgesvxx(int matrix_order, + char fact, + char trans, + lapack_int n, + lapack_int nrhs, + lapack_complex_double *a, + lapack_int lda, + lapack_complex_double *af, + lapack_int ldaf, + lapack_int *ipiv, + char *equed, + double *r, + double *c, + lapack_complex_double *b, + lapack_int ldb, + lapack_complex_double *x, + lapack_int ldx, + double *rcond, + double *rpvgrw, + double *berr, + lapack_int n_err_bnds, + double *err_bnds_norm, + double *err_bnds_comp, + lapack_int nparams, + double *params); + +lapack_int LAPACKE_sgetf2(int matrix_order, lapack_int m, lapack_int n, float *a, lapack_int lda, lapack_int *ipiv); +lapack_int LAPACKE_dgetf2(int matrix_order, lapack_int m, lapack_int n, double *a, lapack_int lda, lapack_int *ipiv); +lapack_int LAPACKE_cgetf2(int matrix_order, + lapack_int m, + lapack_int n, + lapack_complex_float *a, + lapack_int lda, + lapack_int *ipiv); +lapack_int LAPACKE_zgetf2(int matrix_order, + lapack_int m, + lapack_int n, + lapack_complex_double *a, + lapack_int lda, + lapack_int *ipiv); + +lapack_int LAPACKE_sgetrf(int matrix_order, lapack_int m, lapack_int n, float *a, lapack_int lda, lapack_int *ipiv); +lapack_int LAPACKE_dgetrf(int matrix_order, lapack_int m, lapack_int n, double *a, lapack_int lda, lapack_int *ipiv); +lapack_int LAPACKE_cgetrf(int matrix_order, + lapack_int m, + lapack_int n, + lapack_complex_float *a, + lapack_int lda, + lapack_int *ipiv); +lapack_int LAPACKE_zgetrf(int matrix_order, + lapack_int m, + lapack_int n, + lapack_complex_double *a, + lapack_int lda, + lapack_int *ipiv); + +lapack_int LAPACKE_sgetri(int matrix_order, lapack_int n, float *a, lapack_int lda, const lapack_int *ipiv); +lapack_int LAPACKE_dgetri(int matrix_order, lapack_int n, double *a, lapack_int lda, const lapack_int *ipiv); +lapack_int + LAPACKE_cgetri(int matrix_order, lapack_int n, lapack_complex_float *a, lapack_int lda, const lapack_int *ipiv); +lapack_int + LAPACKE_zgetri(int matrix_order, lapack_int n, lapack_complex_double *a, lapack_int lda, const lapack_int *ipiv); + +lapack_int LAPACKE_sgetrs(int matrix_order, + char trans, + lapack_int n, + lapack_int nrhs, + const float *a, + lapack_int lda, + const lapack_int *ipiv, + float *b, + lapack_int ldb); +lapack_int LAPACKE_dgetrs(int matrix_order, + char trans, + lapack_int n, + lapack_int nrhs, + const double *a, + lapack_int lda, + const lapack_int *ipiv, + double *b, + lapack_int ldb); +lapack_int LAPACKE_cgetrs(int matrix_order, + char trans, + lapack_int n, + lapack_int nrhs, + const lapack_complex_float *a, + lapack_int lda, + const lapack_int *ipiv, + lapack_complex_float *b, + lapack_int ldb); +lapack_int LAPACKE_zgetrs(int matrix_order, + char trans, + lapack_int n, + lapack_int nrhs, + const lapack_complex_double *a, + lapack_int lda, + const lapack_int *ipiv, + lapack_complex_double *b, + lapack_int ldb); + +lapack_int LAPACKE_sggbak(int matrix_order, + char job, + char side, + lapack_int n, + lapack_int ilo, + lapack_int ihi, + const float *lscale, + const float *rscale, + lapack_int m, + float *v, + lapack_int ldv); +lapack_int LAPACKE_dggbak(int matrix_order, + char job, + char side, + lapack_int n, + lapack_int ilo, + lapack_int ihi, + const double *lscale, + const double *rscale, + lapack_int m, + double *v, + lapack_int ldv); +lapack_int LAPACKE_cggbak(int matrix_order, + char job, + char side, + lapack_int n, + lapack_int ilo, + lapack_int ihi, + const float *lscale, + const float *rscale, + lapack_int m, + lapack_complex_float *v, + lapack_int ldv); +lapack_int LAPACKE_zggbak(int matrix_order, + char job, + char side, + lapack_int n, + lapack_int ilo, + lapack_int ihi, + const double *lscale, + const double *rscale, + lapack_int m, + lapack_complex_double *v, + lapack_int ldv); + +lapack_int LAPACKE_sggbal(int matrix_order, + char job, + lapack_int n, + float *a, + lapack_int lda, + float *b, + lapack_int ldb, + lapack_int *ilo, + lapack_int *ihi, + float *lscale, + float *rscale); +lapack_int LAPACKE_dggbal(int matrix_order, + char job, + lapack_int n, + double *a, + lapack_int lda, + double *b, + lapack_int ldb, + lapack_int *ilo, + lapack_int *ihi, + double *lscale, + double *rscale); +lapack_int LAPACKE_cggbal(int matrix_order, + char job, + lapack_int n, + lapack_complex_float *a, + lapack_int lda, + lapack_complex_float *b, + lapack_int ldb, + lapack_int *ilo, + lapack_int *ihi, + float *lscale, + float *rscale); +lapack_int LAPACKE_zggbal(int matrix_order, + char job, + lapack_int n, + lapack_complex_double *a, + lapack_int lda, + lapack_complex_double *b, + lapack_int ldb, + lapack_int *ilo, + lapack_int *ihi, + double *lscale, + double *rscale); + +lapack_int LAPACKE_sgges(int matrix_order, + char jobvsl, + char jobvsr, + char sort, + LAPACK_S_SELECT3 selctg, + lapack_int n, + float *a, + lapack_int lda, + float *b, + lapack_int ldb, + lapack_int *sdim, + float *alphar, + float *alphai, + float *beta, + float *vsl, + lapack_int ldvsl, + float *vsr, + lapack_int ldvsr); +lapack_int LAPACKE_dgges(int matrix_order, + char jobvsl, + char jobvsr, + char sort, + LAPACK_D_SELECT3 selctg, + lapack_int n, + double *a, + lapack_int lda, + double *b, + lapack_int ldb, + lapack_int *sdim, + double *alphar, + double *alphai, + double *beta, + double *vsl, + lapack_int ldvsl, + double *vsr, + lapack_int ldvsr); +lapack_int LAPACKE_cgges(int matrix_order, + char jobvsl, + char jobvsr, + char sort, + LAPACK_C_SELECT2 selctg, + lapack_int n, + lapack_complex_float *a, + lapack_int lda, + lapack_complex_float *b, + lapack_int ldb, + lapack_int *sdim, + lapack_complex_float *alpha, + lapack_complex_float *beta, + lapack_complex_float *vsl, + lapack_int ldvsl, + lapack_complex_float *vsr, + lapack_int ldvsr); +lapack_int LAPACKE_zgges(int matrix_order, + char jobvsl, + char jobvsr, + char sort, + LAPACK_Z_SELECT2 selctg, + lapack_int n, + lapack_complex_double *a, + lapack_int lda, + lapack_complex_double *b, + lapack_int ldb, + lapack_int *sdim, + lapack_complex_double *alpha, + lapack_complex_double *beta, + lapack_complex_double *vsl, + lapack_int ldvsl, + lapack_complex_double *vsr, + lapack_int ldvsr); + +lapack_int LAPACKE_sggesx(int matrix_order, + char jobvsl, + char jobvsr, + char sort, + LAPACK_S_SELECT3 selctg, + char sense, + lapack_int n, + float *a, + lapack_int lda, + float *b, + lapack_int ldb, + lapack_int *sdim, + float *alphar, + float *alphai, + float *beta, + float *vsl, + lapack_int ldvsl, + float *vsr, + lapack_int ldvsr, + float *rconde, + float *rcondv); +lapack_int LAPACKE_dggesx(int matrix_order, + char jobvsl, + char jobvsr, + char sort, + LAPACK_D_SELECT3 selctg, + char sense, + lapack_int n, + double *a, + lapack_int lda, + double *b, + lapack_int ldb, + lapack_int *sdim, + double *alphar, + double *alphai, + double *beta, + double *vsl, + lapack_int ldvsl, + double *vsr, + lapack_int ldvsr, + double *rconde, + double *rcondv); +lapack_int LAPACKE_cggesx(int matrix_order, + char jobvsl, + char jobvsr, + char sort, + LAPACK_C_SELECT2 selctg, + char sense, + lapack_int n, + lapack_complex_float *a, + lapack_int lda, + lapack_complex_float *b, + lapack_int ldb, + lapack_int *sdim, + lapack_complex_float *alpha, + lapack_complex_float *beta, + lapack_complex_float *vsl, + lapack_int ldvsl, + lapack_complex_float *vsr, + lapack_int ldvsr, + float *rconde, + float *rcondv); +lapack_int LAPACKE_zggesx(int matrix_order, + char jobvsl, + char jobvsr, + char sort, + LAPACK_Z_SELECT2 selctg, + char sense, + lapack_int n, + lapack_complex_double *a, + lapack_int lda, + lapack_complex_double *b, + lapack_int ldb, + lapack_int *sdim, + lapack_complex_double *alpha, + lapack_complex_double *beta, + lapack_complex_double *vsl, + lapack_int ldvsl, + lapack_complex_double *vsr, + lapack_int ldvsr, + double *rconde, + double *rcondv); + +lapack_int LAPACKE_sggev(int matrix_order, + char jobvl, + char jobvr, + lapack_int n, + float *a, + lapack_int lda, + float *b, + lapack_int ldb, + float *alphar, + float *alphai, + float *beta, + float *vl, + lapack_int ldvl, + float *vr, + lapack_int ldvr); +lapack_int LAPACKE_dggev(int matrix_order, + char jobvl, + char jobvr, + lapack_int n, + double *a, + lapack_int lda, + double *b, + lapack_int ldb, + double *alphar, + double *alphai, + double *beta, + double *vl, + lapack_int ldvl, + double *vr, + lapack_int ldvr); +lapack_int LAPACKE_cggev(int matrix_order, + char jobvl, + char jobvr, + lapack_int n, + lapack_complex_float *a, + lapack_int lda, + lapack_complex_float *b, + lapack_int ldb, + lapack_complex_float *alpha, + lapack_complex_float *beta, + lapack_complex_float *vl, + lapack_int ldvl, + lapack_complex_float *vr, + lapack_int ldvr); +lapack_int LAPACKE_zggev(int matrix_order, + char jobvl, + char jobvr, + lapack_int n, + lapack_complex_double *a, + lapack_int lda, + lapack_complex_double *b, + lapack_int ldb, + lapack_complex_double *alpha, + lapack_complex_double *beta, + lapack_complex_double *vl, + lapack_int ldvl, + lapack_complex_double *vr, + lapack_int ldvr); + +lapack_int LAPACKE_sggevx(int matrix_order, + char balanc, + char jobvl, + char jobvr, + char sense, + lapack_int n, + float *a, + lapack_int lda, + float *b, + lapack_int ldb, + float *alphar, + float *alphai, + float *beta, + float *vl, + lapack_int ldvl, + float *vr, + lapack_int ldvr, + lapack_int *ilo, + lapack_int *ihi, + float *lscale, + float *rscale, + float *abnrm, + float *bbnrm, + float *rconde, + float *rcondv); +lapack_int LAPACKE_dggevx(int matrix_order, + char balanc, + char jobvl, + char jobvr, + char sense, + lapack_int n, + double *a, + lapack_int lda, + double *b, + lapack_int ldb, + double *alphar, + double *alphai, + double *beta, + double *vl, + lapack_int ldvl, + double *vr, + lapack_int ldvr, + lapack_int *ilo, + lapack_int *ihi, + double *lscale, + double *rscale, + double *abnrm, + double *bbnrm, + double *rconde, + double *rcondv); +lapack_int LAPACKE_cggevx(int matrix_order, + char balanc, + char jobvl, + char jobvr, + char sense, + lapack_int n, + lapack_complex_float *a, + lapack_int lda, + lapack_complex_float *b, + lapack_int ldb, + lapack_complex_float *alpha, + lapack_complex_float *beta, + lapack_complex_float *vl, + lapack_int ldvl, + lapack_complex_float *vr, + lapack_int ldvr, + lapack_int *ilo, + lapack_int *ihi, + float *lscale, + float *rscale, + float *abnrm, + float *bbnrm, + float *rconde, + float *rcondv); +lapack_int LAPACKE_zggevx(int matrix_order, + char balanc, + char jobvl, + char jobvr, + char sense, + lapack_int n, + lapack_complex_double *a, + lapack_int lda, + lapack_complex_double *b, + lapack_int ldb, + lapack_complex_double *alpha, + lapack_complex_double *beta, + lapack_complex_double *vl, + lapack_int ldvl, + lapack_complex_double *vr, + lapack_int ldvr, + lapack_int *ilo, + lapack_int *ihi, + double *lscale, + double *rscale, + double *abnrm, + double *bbnrm, + double *rconde, + double *rcondv); + +lapack_int LAPACKE_sggglm(int matrix_order, + lapack_int n, + lapack_int m, + lapack_int p, + float *a, + lapack_int lda, + float *b, + lapack_int ldb, + float *d, + float *x, + float *y); +lapack_int LAPACKE_dggglm(int matrix_order, + lapack_int n, + lapack_int m, + lapack_int p, + double *a, + lapack_int lda, + double *b, + lapack_int ldb, + double *d, + double *x, + double *y); +lapack_int LAPACKE_cggglm(int matrix_order, + lapack_int n, + lapack_int m, + lapack_int p, + lapack_complex_float *a, + lapack_int lda, + lapack_complex_float *b, + lapack_int ldb, + lapack_complex_float *d, + lapack_complex_float *x, + lapack_complex_float *y); +lapack_int LAPACKE_zggglm(int matrix_order, + lapack_int n, + lapack_int m, + lapack_int p, + lapack_complex_double *a, + lapack_int lda, + lapack_complex_double *b, + lapack_int ldb, + lapack_complex_double *d, + lapack_complex_double *x, + lapack_complex_double *y); + +lapack_int LAPACKE_sgghrd(int matrix_order, + char compq, + char compz, + lapack_int n, + lapack_int ilo, + lapack_int ihi, + float *a, + lapack_int lda, + float *b, + lapack_int ldb, + float *q, + lapack_int ldq, + float *z, + lapack_int ldz); +lapack_int LAPACKE_dgghrd(int matrix_order, + char compq, + char compz, + lapack_int n, + lapack_int ilo, + lapack_int ihi, + double *a, + lapack_int lda, + double *b, + lapack_int ldb, + double *q, + lapack_int ldq, + double *z, + lapack_int ldz); +lapack_int LAPACKE_cgghrd(int matrix_order, + char compq, + char compz, + lapack_int n, + lapack_int ilo, + lapack_int ihi, + lapack_complex_float *a, + lapack_int lda, + lapack_complex_float *b, + lapack_int ldb, + lapack_complex_float *q, + lapack_int ldq, + lapack_complex_float *z, + lapack_int ldz); +lapack_int LAPACKE_zgghrd(int matrix_order, + char compq, + char compz, + lapack_int n, + lapack_int ilo, + lapack_int ihi, + lapack_complex_double *a, + lapack_int lda, + lapack_complex_double *b, + lapack_int ldb, + lapack_complex_double *q, + lapack_int ldq, + lapack_complex_double *z, + lapack_int ldz); + +lapack_int LAPACKE_sgglse(int matrix_order, + lapack_int m, + lapack_int n, + lapack_int p, + float *a, + lapack_int lda, + float *b, + lapack_int ldb, + float *c, + float *d, + float *x); +lapack_int LAPACKE_dgglse(int matrix_order, + lapack_int m, + lapack_int n, + lapack_int p, + double *a, + lapack_int lda, + double *b, + lapack_int ldb, + double *c, + double *d, + double *x); +lapack_int LAPACKE_cgglse(int matrix_order, + lapack_int m, + lapack_int n, + lapack_int p, + lapack_complex_float *a, + lapack_int lda, + lapack_complex_float *b, + lapack_int ldb, + lapack_complex_float *c, + lapack_complex_float *d, + lapack_complex_float *x); +lapack_int LAPACKE_zgglse(int matrix_order, + lapack_int m, + lapack_int n, + lapack_int p, + lapack_complex_double *a, + lapack_int lda, + lapack_complex_double *b, + lapack_int ldb, + lapack_complex_double *c, + lapack_complex_double *d, + lapack_complex_double *x); + +lapack_int LAPACKE_sggqrf(int matrix_order, + lapack_int n, + lapack_int m, + lapack_int p, + float *a, + lapack_int lda, + float *taua, + float *b, + lapack_int ldb, + float *taub); +lapack_int LAPACKE_dggqrf(int matrix_order, + lapack_int n, + lapack_int m, + lapack_int p, + double *a, + lapack_int lda, + double *taua, + double *b, + lapack_int ldb, + double *taub); +lapack_int LAPACKE_cggqrf(int matrix_order, + lapack_int n, + lapack_int m, + lapack_int p, + lapack_complex_float *a, + lapack_int lda, + lapack_complex_float *taua, + lapack_complex_float *b, + lapack_int ldb, + lapack_complex_float *taub); +lapack_int LAPACKE_zggqrf(int matrix_order, + lapack_int n, + lapack_int m, + lapack_int p, + lapack_complex_double *a, + lapack_int lda, + lapack_complex_double *taua, + lapack_complex_double *b, + lapack_int ldb, + lapack_complex_double *taub); + +lapack_int LAPACKE_sggrqf(int matrix_order, + lapack_int m, + lapack_int p, + lapack_int n, + float *a, + lapack_int lda, + float *taua, + float *b, + lapack_int ldb, + float *taub); +lapack_int LAPACKE_dggrqf(int matrix_order, + lapack_int m, + lapack_int p, + lapack_int n, + double *a, + lapack_int lda, + double *taua, + double *b, + lapack_int ldb, + double *taub); +lapack_int LAPACKE_cggrqf(int matrix_order, + lapack_int m, + lapack_int p, + lapack_int n, + lapack_complex_float *a, + lapack_int lda, + lapack_complex_float *taua, + lapack_complex_float *b, + lapack_int ldb, + lapack_complex_float *taub); +lapack_int LAPACKE_zggrqf(int matrix_order, + lapack_int m, + lapack_int p, + lapack_int n, + lapack_complex_double *a, + lapack_int lda, + lapack_complex_double *taua, + lapack_complex_double *b, + lapack_int ldb, + lapack_complex_double *taub); + +lapack_int LAPACKE_sggsvd(int matrix_order, + char jobu, + char jobv, + char jobq, + lapack_int m, + lapack_int n, + lapack_int p, + lapack_int *k, + lapack_int *l, + float *a, + lapack_int lda, + float *b, + lapack_int ldb, + float *alpha, + float *beta, + float *u, + lapack_int ldu, + float *v, + lapack_int ldv, + float *q, + lapack_int ldq, + lapack_int *iwork); +lapack_int LAPACKE_dggsvd(int matrix_order, + char jobu, + char jobv, + char jobq, + lapack_int m, + lapack_int n, + lapack_int p, + lapack_int *k, + lapack_int *l, + double *a, + lapack_int lda, + double *b, + lapack_int ldb, + double *alpha, + double *beta, + double *u, + lapack_int ldu, + double *v, + lapack_int ldv, + double *q, + lapack_int ldq, + lapack_int *iwork); +lapack_int LAPACKE_cggsvd(int matrix_order, + char jobu, + char jobv, + char jobq, + lapack_int m, + lapack_int n, + lapack_int p, + lapack_int *k, + lapack_int *l, + lapack_complex_float *a, + lapack_int lda, + lapack_complex_float *b, + lapack_int ldb, + float *alpha, + float *beta, + lapack_complex_float *u, + lapack_int ldu, + lapack_complex_float *v, + lapack_int ldv, + lapack_complex_float *q, + lapack_int ldq, + lapack_int *iwork); +lapack_int LAPACKE_zggsvd(int matrix_order, + char jobu, + char jobv, + char jobq, + lapack_int m, + lapack_int n, + lapack_int p, + lapack_int *k, + lapack_int *l, + lapack_complex_double *a, + lapack_int lda, + lapack_complex_double *b, + lapack_int ldb, + double *alpha, + double *beta, + lapack_complex_double *u, + lapack_int ldu, + lapack_complex_double *v, + lapack_int ldv, + lapack_complex_double *q, + lapack_int ldq, + lapack_int *iwork); + +lapack_int LAPACKE_sggsvp(int matrix_order, + char jobu, + char jobv, + char jobq, + lapack_int m, + lapack_int p, + lapack_int n, + float *a, + lapack_int lda, + float *b, + lapack_int ldb, + float tola, + float tolb, + lapack_int *k, + lapack_int *l, + float *u, + lapack_int ldu, + float *v, + lapack_int ldv, + float *q, + lapack_int ldq); +lapack_int LAPACKE_dggsvp(int matrix_order, + char jobu, + char jobv, + char jobq, + lapack_int m, + lapack_int p, + lapack_int n, + double *a, + lapack_int lda, + double *b, + lapack_int ldb, + double tola, + double tolb, + lapack_int *k, + lapack_int *l, + double *u, + lapack_int ldu, + double *v, + lapack_int ldv, + double *q, + lapack_int ldq); +lapack_int LAPACKE_cggsvp(int matrix_order, + char jobu, + char jobv, + char jobq, + lapack_int m, + lapack_int p, + lapack_int n, + lapack_complex_float *a, + lapack_int lda, + lapack_complex_float *b, + lapack_int ldb, + float tola, + float tolb, + lapack_int *k, + lapack_int *l, + lapack_complex_float *u, + lapack_int ldu, + lapack_complex_float *v, + lapack_int ldv, + lapack_complex_float *q, + lapack_int ldq); +lapack_int LAPACKE_zggsvp(int matrix_order, + char jobu, + char jobv, + char jobq, + lapack_int m, + lapack_int p, + lapack_int n, + lapack_complex_double *a, + lapack_int lda, + lapack_complex_double *b, + lapack_int ldb, + double tola, + double tolb, + lapack_int *k, + lapack_int *l, + lapack_complex_double *u, + lapack_int ldu, + lapack_complex_double *v, + lapack_int ldv, + lapack_complex_double *q, + lapack_int ldq); + +lapack_int LAPACKE_sgtcon(char norm, + lapack_int n, + const float *dl, + const float *d, + const float *du, + const float *du2, + const lapack_int *ipiv, + float anorm, + float *rcond); +lapack_int LAPACKE_dgtcon(char norm, + lapack_int n, + const double *dl, + const double *d, + const double *du, + const double *du2, + const lapack_int *ipiv, + double anorm, + double *rcond); +lapack_int LAPACKE_cgtcon(char norm, + lapack_int n, + const lapack_complex_float *dl, + const lapack_complex_float *d, + const lapack_complex_float *du, + const lapack_complex_float *du2, + const lapack_int *ipiv, + float anorm, + float *rcond); +lapack_int LAPACKE_zgtcon(char norm, + lapack_int n, + const lapack_complex_double *dl, + const lapack_complex_double *d, + const lapack_complex_double *du, + const lapack_complex_double *du2, + const lapack_int *ipiv, + double anorm, + double *rcond); + +lapack_int LAPACKE_sgtrfs(int matrix_order, + char trans, + lapack_int n, + lapack_int nrhs, + const float *dl, + const float *d, + const float *du, + const float *dlf, + const float *df, + const float *duf, + const float *du2, + const lapack_int *ipiv, + const float *b, + lapack_int ldb, + float *x, + lapack_int ldx, + float *ferr, + float *berr); +lapack_int LAPACKE_dgtrfs(int matrix_order, + char trans, + lapack_int n, + lapack_int nrhs, + const double *dl, + const double *d, + const double *du, + const double *dlf, + const double *df, + const double *duf, + const double *du2, + const lapack_int *ipiv, + const double *b, + lapack_int ldb, + double *x, + lapack_int ldx, + double *ferr, + double *berr); +lapack_int LAPACKE_cgtrfs(int matrix_order, + char trans, + lapack_int n, + lapack_int nrhs, + const lapack_complex_float *dl, + const lapack_complex_float *d, + const lapack_complex_float *du, + const lapack_complex_float *dlf, + const lapack_complex_float *df, + const lapack_complex_float *duf, + const lapack_complex_float *du2, + const lapack_int *ipiv, + const lapack_complex_float *b, + lapack_int ldb, + lapack_complex_float *x, + lapack_int ldx, + float *ferr, + float *berr); +lapack_int LAPACKE_zgtrfs(int matrix_order, + char trans, + lapack_int n, + lapack_int nrhs, + const lapack_complex_double *dl, + const lapack_complex_double *d, + const lapack_complex_double *du, + const lapack_complex_double *dlf, + const lapack_complex_double *df, + const lapack_complex_double *duf, + const lapack_complex_double *du2, + const lapack_int *ipiv, + const lapack_complex_double *b, + lapack_int ldb, + lapack_complex_double *x, + lapack_int ldx, + double *ferr, + double *berr); + +lapack_int LAPACKE_sgtsv(int matrix_order, + lapack_int n, + lapack_int nrhs, + float *dl, + float *d, + float *du, + float *b, + lapack_int ldb); +lapack_int LAPACKE_dgtsv(int matrix_order, + lapack_int n, + lapack_int nrhs, + double *dl, + double *d, + double *du, + double *b, + lapack_int ldb); +lapack_int LAPACKE_cgtsv(int matrix_order, + lapack_int n, + lapack_int nrhs, + lapack_complex_float *dl, + lapack_complex_float *d, + lapack_complex_float *du, + lapack_complex_float *b, + lapack_int ldb); +lapack_int LAPACKE_zgtsv(int matrix_order, + lapack_int n, + lapack_int nrhs, + lapack_complex_double *dl, + lapack_complex_double *d, + lapack_complex_double *du, + lapack_complex_double *b, + lapack_int ldb); + +lapack_int LAPACKE_sgtsvx(int matrix_order, + char fact, + char trans, + lapack_int n, + lapack_int nrhs, + const float *dl, + const float *d, + const float *du, + float *dlf, + float *df, + float *duf, + float *du2, + lapack_int *ipiv, + const float *b, + lapack_int ldb, + float *x, + lapack_int ldx, + float *rcond, + float *ferr, + float *berr); +lapack_int LAPACKE_dgtsvx(int matrix_order, + char fact, + char trans, + lapack_int n, + lapack_int nrhs, + const double *dl, + const double *d, + const double *du, + double *dlf, + double *df, + double *duf, + double *du2, + lapack_int *ipiv, + const double *b, + lapack_int ldb, + double *x, + lapack_int ldx, + double *rcond, + double *ferr, + double *berr); +lapack_int LAPACKE_cgtsvx(int matrix_order, + char fact, + char trans, + lapack_int n, + lapack_int nrhs, + const lapack_complex_float *dl, + const lapack_complex_float *d, + const lapack_complex_float *du, + lapack_complex_float *dlf, + lapack_complex_float *df, + lapack_complex_float *duf, + lapack_complex_float *du2, + lapack_int *ipiv, + const lapack_complex_float *b, + lapack_int ldb, + lapack_complex_float *x, + lapack_int ldx, + float *rcond, + float *ferr, + float *berr); +lapack_int LAPACKE_zgtsvx(int matrix_order, + char fact, + char trans, + lapack_int n, + lapack_int nrhs, + const lapack_complex_double *dl, + const lapack_complex_double *d, + const lapack_complex_double *du, + lapack_complex_double *dlf, + lapack_complex_double *df, + lapack_complex_double *duf, + lapack_complex_double *du2, + lapack_int *ipiv, + const lapack_complex_double *b, + lapack_int ldb, + lapack_complex_double *x, + lapack_int ldx, + double *rcond, + double *ferr, + double *berr); + +lapack_int LAPACKE_sgttrf(lapack_int n, float *dl, float *d, float *du, float *du2, lapack_int *ipiv); +lapack_int LAPACKE_dgttrf(lapack_int n, double *dl, double *d, double *du, double *du2, lapack_int *ipiv); +lapack_int LAPACKE_cgttrf(lapack_int n, + lapack_complex_float *dl, + lapack_complex_float *d, + lapack_complex_float *du, + lapack_complex_float *du2, + lapack_int *ipiv); +lapack_int LAPACKE_zgttrf(lapack_int n, + lapack_complex_double *dl, + lapack_complex_double *d, + lapack_complex_double *du, + lapack_complex_double *du2, + lapack_int *ipiv); + +lapack_int LAPACKE_sgttrs(int matrix_order, + char trans, + lapack_int n, + lapack_int nrhs, + const float *dl, + const float *d, + const float *du, + const float *du2, + const lapack_int *ipiv, + float *b, + lapack_int ldb); +lapack_int LAPACKE_dgttrs(int matrix_order, + char trans, + lapack_int n, + lapack_int nrhs, + const double *dl, + const double *d, + const double *du, + const double *du2, + const lapack_int *ipiv, + double *b, + lapack_int ldb); +lapack_int LAPACKE_cgttrs(int matrix_order, + char trans, + lapack_int n, + lapack_int nrhs, + const lapack_complex_float *dl, + const lapack_complex_float *d, + const lapack_complex_float *du, + const lapack_complex_float *du2, + const lapack_int *ipiv, + lapack_complex_float *b, + lapack_int ldb); +lapack_int LAPACKE_zgttrs(int matrix_order, + char trans, + lapack_int n, + lapack_int nrhs, + const lapack_complex_double *dl, + const lapack_complex_double *d, + const lapack_complex_double *du, + const lapack_complex_double *du2, + const lapack_int *ipiv, + lapack_complex_double *b, + lapack_int ldb); + +lapack_int LAPACKE_chbev(int matrix_order, + char jobz, + char uplo, + lapack_int n, + lapack_int kd, + lapack_complex_float *ab, + lapack_int ldab, + float *w, + lapack_complex_float *z, + lapack_int ldz); +lapack_int LAPACKE_zhbev(int matrix_order, + char jobz, + char uplo, + lapack_int n, + lapack_int kd, + lapack_complex_double *ab, + lapack_int ldab, + double *w, + lapack_complex_double *z, + lapack_int ldz); + +lapack_int LAPACKE_chbevd(int matrix_order, + char jobz, + char uplo, + lapack_int n, + lapack_int kd, + lapack_complex_float *ab, + lapack_int ldab, + float *w, + lapack_complex_float *z, + lapack_int ldz); +lapack_int LAPACKE_zhbevd(int matrix_order, + char jobz, + char uplo, + lapack_int n, + lapack_int kd, + lapack_complex_double *ab, + lapack_int ldab, + double *w, + lapack_complex_double *z, + lapack_int ldz); + +lapack_int LAPACKE_chbevx(int matrix_order, + char jobz, + char range, + char uplo, + lapack_int n, + lapack_int kd, + lapack_complex_float *ab, + lapack_int ldab, + lapack_complex_float *q, + lapack_int ldq, + float vl, + float vu, + lapack_int il, + lapack_int iu, + float abstol, + lapack_int *m, + float *w, + lapack_complex_float *z, + lapack_int ldz, + lapack_int *ifail); +lapack_int LAPACKE_zhbevx(int matrix_order, + char jobz, + char range, + char uplo, + lapack_int n, + lapack_int kd, + lapack_complex_double *ab, + lapack_int ldab, + lapack_complex_double *q, + lapack_int ldq, + double vl, + double vu, + lapack_int il, + lapack_int iu, + double abstol, + lapack_int *m, + double *w, + lapack_complex_double *z, + lapack_int ldz, + lapack_int *ifail); + +lapack_int LAPACKE_chbgst(int matrix_order, + char vect, + char uplo, + lapack_int n, + lapack_int ka, + lapack_int kb, + lapack_complex_float *ab, + lapack_int ldab, + const lapack_complex_float *bb, + lapack_int ldbb, + lapack_complex_float *x, + lapack_int ldx); +lapack_int LAPACKE_zhbgst(int matrix_order, + char vect, + char uplo, + lapack_int n, + lapack_int ka, + lapack_int kb, + lapack_complex_double *ab, + lapack_int ldab, + const lapack_complex_double *bb, + lapack_int ldbb, + lapack_complex_double *x, + lapack_int ldx); + +lapack_int LAPACKE_chbgv(int matrix_order, + char jobz, + char uplo, + lapack_int n, + lapack_int ka, + lapack_int kb, + lapack_complex_float *ab, + lapack_int ldab, + lapack_complex_float *bb, + lapack_int ldbb, + float *w, + lapack_complex_float *z, + lapack_int ldz); +lapack_int LAPACKE_zhbgv(int matrix_order, + char jobz, + char uplo, + lapack_int n, + lapack_int ka, + lapack_int kb, + lapack_complex_double *ab, + lapack_int ldab, + lapack_complex_double *bb, + lapack_int ldbb, + double *w, + lapack_complex_double *z, + lapack_int ldz); + +lapack_int LAPACKE_chbgvd(int matrix_order, + char jobz, + char uplo, + lapack_int n, + lapack_int ka, + lapack_int kb, + lapack_complex_float *ab, + lapack_int ldab, + lapack_complex_float *bb, + lapack_int ldbb, + float *w, + lapack_complex_float *z, + lapack_int ldz); +lapack_int LAPACKE_zhbgvd(int matrix_order, + char jobz, + char uplo, + lapack_int n, + lapack_int ka, + lapack_int kb, + lapack_complex_double *ab, + lapack_int ldab, + lapack_complex_double *bb, + lapack_int ldbb, + double *w, + lapack_complex_double *z, + lapack_int ldz); + +lapack_int LAPACKE_chbgvx(int matrix_order, + char jobz, + char range, + char uplo, + lapack_int n, + lapack_int ka, + lapack_int kb, + lapack_complex_float *ab, + lapack_int ldab, + lapack_complex_float *bb, + lapack_int ldbb, + lapack_complex_float *q, + lapack_int ldq, + float vl, + float vu, + lapack_int il, + lapack_int iu, + float abstol, + lapack_int *m, + float *w, + lapack_complex_float *z, + lapack_int ldz, + lapack_int *ifail); +lapack_int LAPACKE_zhbgvx(int matrix_order, + char jobz, + char range, + char uplo, + lapack_int n, + lapack_int ka, + lapack_int kb, + lapack_complex_double *ab, + lapack_int ldab, + lapack_complex_double *bb, + lapack_int ldbb, + lapack_complex_double *q, + lapack_int ldq, + double vl, + double vu, + lapack_int il, + lapack_int iu, + double abstol, + lapack_int *m, + double *w, + lapack_complex_double *z, + lapack_int ldz, + lapack_int *ifail); + +lapack_int LAPACKE_chbtrd(int matrix_order, + char vect, + char uplo, + lapack_int n, + lapack_int kd, + lapack_complex_float *ab, + lapack_int ldab, + float *d, + float *e, + lapack_complex_float *q, + lapack_int ldq); +lapack_int LAPACKE_zhbtrd(int matrix_order, + char vect, + char uplo, + lapack_int n, + lapack_int kd, + lapack_complex_double *ab, + lapack_int ldab, + double *d, + double *e, + lapack_complex_double *q, + lapack_int ldq); + +lapack_int LAPACKE_checon(int matrix_order, + char uplo, + lapack_int n, + const lapack_complex_float *a, + lapack_int lda, + const lapack_int *ipiv, + float anorm, + float *rcond); +lapack_int LAPACKE_zhecon(int matrix_order, + char uplo, + lapack_int n, + const lapack_complex_double *a, + lapack_int lda, + const lapack_int *ipiv, + double anorm, + double *rcond); + +lapack_int LAPACKE_cheequb(int matrix_order, + char uplo, + lapack_int n, + const lapack_complex_float *a, + lapack_int lda, + float *s, + float *scond, + float *amax); +lapack_int LAPACKE_zheequb(int matrix_order, + char uplo, + lapack_int n, + const lapack_complex_double *a, + lapack_int lda, + double *s, + double *scond, + double *amax); + +lapack_int LAPACKE_cheev(int matrix_order, + char jobz, + char uplo, + lapack_int n, + lapack_complex_float *a, + lapack_int lda, + float *w); +lapack_int LAPACKE_zheev(int matrix_order, + char jobz, + char uplo, + lapack_int n, + lapack_complex_double *a, + lapack_int lda, + double *w); + +lapack_int LAPACKE_cheevd(int matrix_order, + char jobz, + char uplo, + lapack_int n, + lapack_complex_float *a, + lapack_int lda, + float *w); +lapack_int LAPACKE_zheevd(int matrix_order, + char jobz, + char uplo, + lapack_int n, + lapack_complex_double *a, + lapack_int lda, + double *w); + +lapack_int LAPACKE_cheevr(int matrix_order, + char jobz, + char range, + char uplo, + lapack_int n, + lapack_complex_float *a, + lapack_int lda, + float vl, + float vu, + lapack_int il, + lapack_int iu, + float abstol, + lapack_int *m, + float *w, + lapack_complex_float *z, + lapack_int ldz, + lapack_int *isuppz); +lapack_int LAPACKE_zheevr(int matrix_order, + char jobz, + char range, + char uplo, + lapack_int n, + lapack_complex_double *a, + lapack_int lda, + double vl, + double vu, + lapack_int il, + lapack_int iu, + double abstol, + lapack_int *m, + double *w, + lapack_complex_double *z, + lapack_int ldz, + lapack_int *isuppz); + +lapack_int LAPACKE_cheevx(int matrix_order, + char jobz, + char range, + char uplo, + lapack_int n, + lapack_complex_float *a, + lapack_int lda, + float vl, + float vu, + lapack_int il, + lapack_int iu, + float abstol, + lapack_int *m, + float *w, + lapack_complex_float *z, + lapack_int ldz, + lapack_int *ifail); +lapack_int LAPACKE_zheevx(int matrix_order, + char jobz, + char range, + char uplo, + lapack_int n, + lapack_complex_double *a, + lapack_int lda, + double vl, + double vu, + lapack_int il, + lapack_int iu, + double abstol, + lapack_int *m, + double *w, + lapack_complex_double *z, + lapack_int ldz, + lapack_int *ifail); + +lapack_int LAPACKE_chegst(int matrix_order, + lapack_int itype, + char uplo, + lapack_int n, + lapack_complex_float *a, + lapack_int lda, + const lapack_complex_float *b, + lapack_int ldb); +lapack_int LAPACKE_zhegst(int matrix_order, + lapack_int itype, + char uplo, + lapack_int n, + lapack_complex_double *a, + lapack_int lda, + const lapack_complex_double *b, + lapack_int ldb); + +lapack_int LAPACKE_chegv(int matrix_order, + lapack_int itype, + char jobz, + char uplo, + lapack_int n, + lapack_complex_float *a, + lapack_int lda, + lapack_complex_float *b, + lapack_int ldb, + float *w); +lapack_int LAPACKE_zhegv(int matrix_order, + lapack_int itype, + char jobz, + char uplo, + lapack_int n, + lapack_complex_double *a, + lapack_int lda, + lapack_complex_double *b, + lapack_int ldb, + double *w); + +lapack_int LAPACKE_chegvd(int matrix_order, + lapack_int itype, + char jobz, + char uplo, + lapack_int n, + lapack_complex_float *a, + lapack_int lda, + lapack_complex_float *b, + lapack_int ldb, + float *w); +lapack_int LAPACKE_zhegvd(int matrix_order, + lapack_int itype, + char jobz, + char uplo, + lapack_int n, + lapack_complex_double *a, + lapack_int lda, + lapack_complex_double *b, + lapack_int ldb, + double *w); + +lapack_int LAPACKE_chegvx(int matrix_order, + lapack_int itype, + char jobz, + char range, + char uplo, + lapack_int n, + lapack_complex_float *a, + lapack_int lda, + lapack_complex_float *b, + lapack_int ldb, + float vl, + float vu, + lapack_int il, + lapack_int iu, + float abstol, + lapack_int *m, + float *w, + lapack_complex_float *z, + lapack_int ldz, + lapack_int *ifail); +lapack_int LAPACKE_zhegvx(int matrix_order, + lapack_int itype, + char jobz, + char range, + char uplo, + lapack_int n, + lapack_complex_double *a, + lapack_int lda, + lapack_complex_double *b, + lapack_int ldb, + double vl, + double vu, + lapack_int il, + lapack_int iu, + double abstol, + lapack_int *m, + double *w, + lapack_complex_double *z, + lapack_int ldz, + lapack_int *ifail); + +lapack_int LAPACKE_cherfs(int matrix_order, + char uplo, + lapack_int n, + lapack_int nrhs, + const lapack_complex_float *a, + lapack_int lda, + const lapack_complex_float *af, + lapack_int ldaf, + const lapack_int *ipiv, + const lapack_complex_float *b, + lapack_int ldb, + lapack_complex_float *x, + lapack_int ldx, + float *ferr, + float *berr); +lapack_int LAPACKE_zherfs(int matrix_order, + char uplo, + lapack_int n, + lapack_int nrhs, + const lapack_complex_double *a, + lapack_int lda, + const lapack_complex_double *af, + lapack_int ldaf, + const lapack_int *ipiv, + const lapack_complex_double *b, + lapack_int ldb, + lapack_complex_double *x, + lapack_int ldx, + double *ferr, + double *berr); + +lapack_int LAPACKE_cherfsx(int matrix_order, + char uplo, + char equed, + lapack_int n, + lapack_int nrhs, + const lapack_complex_float *a, + lapack_int lda, + const lapack_complex_float *af, + lapack_int ldaf, + const lapack_int *ipiv, + const float *s, + const lapack_complex_float *b, + lapack_int ldb, + lapack_complex_float *x, + lapack_int ldx, + float *rcond, + float *berr, + lapack_int n_err_bnds, + float *err_bnds_norm, + float *err_bnds_comp, + lapack_int nparams, + float *params); +lapack_int LAPACKE_zherfsx(int matrix_order, + char uplo, + char equed, + lapack_int n, + lapack_int nrhs, + const lapack_complex_double *a, + lapack_int lda, + const lapack_complex_double *af, + lapack_int ldaf, + const lapack_int *ipiv, + const double *s, + const lapack_complex_double *b, + lapack_int ldb, + lapack_complex_double *x, + lapack_int ldx, + double *rcond, + double *berr, + lapack_int n_err_bnds, + double *err_bnds_norm, + double *err_bnds_comp, + lapack_int nparams, + double *params); + +lapack_int LAPACKE_chesv(int matrix_order, + char uplo, + lapack_int n, + lapack_int nrhs, + lapack_complex_float *a, + lapack_int lda, + lapack_int *ipiv, + lapack_complex_float *b, + lapack_int ldb); +lapack_int LAPACKE_zhesv(int matrix_order, + char uplo, + lapack_int n, + lapack_int nrhs, + lapack_complex_double *a, + lapack_int lda, + lapack_int *ipiv, + lapack_complex_double *b, + lapack_int ldb); + +lapack_int LAPACKE_chesvx(int matrix_order, + char fact, + char uplo, + lapack_int n, + lapack_int nrhs, + const lapack_complex_float *a, + lapack_int lda, + lapack_complex_float *af, + lapack_int ldaf, + lapack_int *ipiv, + const lapack_complex_float *b, + lapack_int ldb, + lapack_complex_float *x, + lapack_int ldx, + float *rcond, + float *ferr, + float *berr); +lapack_int LAPACKE_zhesvx(int matrix_order, + char fact, + char uplo, + lapack_int n, + lapack_int nrhs, + const lapack_complex_double *a, + lapack_int lda, + lapack_complex_double *af, + lapack_int ldaf, + lapack_int *ipiv, + const lapack_complex_double *b, + lapack_int ldb, + lapack_complex_double *x, + lapack_int ldx, + double *rcond, + double *ferr, + double *berr); + +lapack_int LAPACKE_chesvxx(int matrix_order, + char fact, + char uplo, + lapack_int n, + lapack_int nrhs, + lapack_complex_float *a, + lapack_int lda, + lapack_complex_float *af, + lapack_int ldaf, + lapack_int *ipiv, + char *equed, + float *s, + lapack_complex_float *b, + lapack_int ldb, + lapack_complex_float *x, + lapack_int ldx, + float *rcond, + float *rpvgrw, + float *berr, + lapack_int n_err_bnds, + float *err_bnds_norm, + float *err_bnds_comp, + lapack_int nparams, + float *params); +lapack_int LAPACKE_zhesvxx(int matrix_order, + char fact, + char uplo, + lapack_int n, + lapack_int nrhs, + lapack_complex_double *a, + lapack_int lda, + lapack_complex_double *af, + lapack_int ldaf, + lapack_int *ipiv, + char *equed, + double *s, + lapack_complex_double *b, + lapack_int ldb, + lapack_complex_double *x, + lapack_int ldx, + double *rcond, + double *rpvgrw, + double *berr, + lapack_int n_err_bnds, + double *err_bnds_norm, + double *err_bnds_comp, + lapack_int nparams, + double *params); + +lapack_int LAPACKE_chetrd(int matrix_order, + char uplo, + lapack_int n, + lapack_complex_float *a, + lapack_int lda, + float *d, + float *e, + lapack_complex_float *tau); +lapack_int LAPACKE_zhetrd(int matrix_order, + char uplo, + lapack_int n, + lapack_complex_double *a, + lapack_int lda, + double *d, + double *e, + lapack_complex_double *tau); + +lapack_int + LAPACKE_chetrf(int matrix_order, char uplo, lapack_int n, lapack_complex_float *a, lapack_int lda, lapack_int *ipiv); +lapack_int + LAPACKE_zhetrf(int matrix_order, char uplo, lapack_int n, lapack_complex_double *a, lapack_int lda, lapack_int *ipiv); + +lapack_int LAPACKE_chetri(int matrix_order, + char uplo, + lapack_int n, + lapack_complex_float *a, + lapack_int lda, + const lapack_int *ipiv); +lapack_int LAPACKE_zhetri(int matrix_order, + char uplo, + lapack_int n, + lapack_complex_double *a, + lapack_int lda, + const lapack_int *ipiv); + +lapack_int LAPACKE_chetrs(int matrix_order, + char uplo, + lapack_int n, + lapack_int nrhs, + const lapack_complex_float *a, + lapack_int lda, + const lapack_int *ipiv, + lapack_complex_float *b, + lapack_int ldb); +lapack_int LAPACKE_zhetrs(int matrix_order, + char uplo, + lapack_int n, + lapack_int nrhs, + const lapack_complex_double *a, + lapack_int lda, + const lapack_int *ipiv, + lapack_complex_double *b, + lapack_int ldb); + +lapack_int LAPACKE_chfrk(int matrix_order, + char transr, + char uplo, + char trans, + lapack_int n, + lapack_int k, + float alpha, + const lapack_complex_float *a, + lapack_int lda, + float beta, + lapack_complex_float *c); +lapack_int LAPACKE_zhfrk(int matrix_order, + char transr, + char uplo, + char trans, + lapack_int n, + lapack_int k, + double alpha, + const lapack_complex_double *a, + lapack_int lda, + double beta, + lapack_complex_double *c); + +lapack_int LAPACKE_shgeqz(int matrix_order, + char job, + char compq, + char compz, + lapack_int n, + lapack_int ilo, + lapack_int ihi, + float *h, + lapack_int ldh, + float *t, + lapack_int ldt, + float *alphar, + float *alphai, + float *beta, + float *q, + lapack_int ldq, + float *z, + lapack_int ldz); +lapack_int LAPACKE_dhgeqz(int matrix_order, + char job, + char compq, + char compz, + lapack_int n, + lapack_int ilo, + lapack_int ihi, + double *h, + lapack_int ldh, + double *t, + lapack_int ldt, + double *alphar, + double *alphai, + double *beta, + double *q, + lapack_int ldq, + double *z, + lapack_int ldz); +lapack_int LAPACKE_chgeqz(int matrix_order, + char job, + char compq, + char compz, + lapack_int n, + lapack_int ilo, + lapack_int ihi, + lapack_complex_float *h, + lapack_int ldh, + lapack_complex_float *t, + lapack_int ldt, + lapack_complex_float *alpha, + lapack_complex_float *beta, + lapack_complex_float *q, + lapack_int ldq, + lapack_complex_float *z, + lapack_int ldz); +lapack_int LAPACKE_zhgeqz(int matrix_order, + char job, + char compq, + char compz, + lapack_int n, + lapack_int ilo, + lapack_int ihi, + lapack_complex_double *h, + lapack_int ldh, + lapack_complex_double *t, + lapack_int ldt, + lapack_complex_double *alpha, + lapack_complex_double *beta, + lapack_complex_double *q, + lapack_int ldq, + lapack_complex_double *z, + lapack_int ldz); + +lapack_int LAPACKE_chpcon(int matrix_order, + char uplo, + lapack_int n, + const lapack_complex_float *ap, + const lapack_int *ipiv, + float anorm, + float *rcond); +lapack_int LAPACKE_zhpcon(int matrix_order, + char uplo, + lapack_int n, + const lapack_complex_double *ap, + const lapack_int *ipiv, + double anorm, + double *rcond); + +lapack_int LAPACKE_chpev(int matrix_order, + char jobz, + char uplo, + lapack_int n, + lapack_complex_float *ap, + float *w, + lapack_complex_float *z, + lapack_int ldz); +lapack_int LAPACKE_zhpev(int matrix_order, + char jobz, + char uplo, + lapack_int n, + lapack_complex_double *ap, + double *w, + lapack_complex_double *z, + lapack_int ldz); + +lapack_int LAPACKE_chpevd(int matrix_order, + char jobz, + char uplo, + lapack_int n, + lapack_complex_float *ap, + float *w, + lapack_complex_float *z, + lapack_int ldz); +lapack_int LAPACKE_zhpevd(int matrix_order, + char jobz, + char uplo, + lapack_int n, + lapack_complex_double *ap, + double *w, + lapack_complex_double *z, + lapack_int ldz); + +lapack_int LAPACKE_chpevx(int matrix_order, + char jobz, + char range, + char uplo, + lapack_int n, + lapack_complex_float *ap, + float vl, + float vu, + lapack_int il, + lapack_int iu, + float abstol, + lapack_int *m, + float *w, + lapack_complex_float *z, + lapack_int ldz, + lapack_int *ifail); +lapack_int LAPACKE_zhpevx(int matrix_order, + char jobz, + char range, + char uplo, + lapack_int n, + lapack_complex_double *ap, + double vl, + double vu, + lapack_int il, + lapack_int iu, + double abstol, + lapack_int *m, + double *w, + lapack_complex_double *z, + lapack_int ldz, + lapack_int *ifail); + +lapack_int LAPACKE_chpgst(int matrix_order, + lapack_int itype, + char uplo, + lapack_int n, + lapack_complex_float *ap, + const lapack_complex_float *bp); +lapack_int LAPACKE_zhpgst(int matrix_order, + lapack_int itype, + char uplo, + lapack_int n, + lapack_complex_double *ap, + const lapack_complex_double *bp); + +lapack_int LAPACKE_chpgv(int matrix_order, + lapack_int itype, + char jobz, + char uplo, + lapack_int n, + lapack_complex_float *ap, + lapack_complex_float *bp, + float *w, + lapack_complex_float *z, + lapack_int ldz); +lapack_int LAPACKE_zhpgv(int matrix_order, + lapack_int itype, + char jobz, + char uplo, + lapack_int n, + lapack_complex_double *ap, + lapack_complex_double *bp, + double *w, + lapack_complex_double *z, + lapack_int ldz); + +lapack_int LAPACKE_chpgvd(int matrix_order, + lapack_int itype, + char jobz, + char uplo, + lapack_int n, + lapack_complex_float *ap, + lapack_complex_float *bp, + float *w, + lapack_complex_float *z, + lapack_int ldz); +lapack_int LAPACKE_zhpgvd(int matrix_order, + lapack_int itype, + char jobz, + char uplo, + lapack_int n, + lapack_complex_double *ap, + lapack_complex_double *bp, + double *w, + lapack_complex_double *z, + lapack_int ldz); + +lapack_int LAPACKE_chpgvx(int matrix_order, + lapack_int itype, + char jobz, + char range, + char uplo, + lapack_int n, + lapack_complex_float *ap, + lapack_complex_float *bp, + float vl, + float vu, + lapack_int il, + lapack_int iu, + float abstol, + lapack_int *m, + float *w, + lapack_complex_float *z, + lapack_int ldz, + lapack_int *ifail); +lapack_int LAPACKE_zhpgvx(int matrix_order, + lapack_int itype, + char jobz, + char range, + char uplo, + lapack_int n, + lapack_complex_double *ap, + lapack_complex_double *bp, + double vl, + double vu, + lapack_int il, + lapack_int iu, + double abstol, + lapack_int *m, + double *w, + lapack_complex_double *z, + lapack_int ldz, + lapack_int *ifail); + +lapack_int LAPACKE_chprfs(int matrix_order, + char uplo, + lapack_int n, + lapack_int nrhs, + const lapack_complex_float *ap, + const lapack_complex_float *afp, + const lapack_int *ipiv, + const lapack_complex_float *b, + lapack_int ldb, + lapack_complex_float *x, + lapack_int ldx, + float *ferr, + float *berr); +lapack_int LAPACKE_zhprfs(int matrix_order, + char uplo, + lapack_int n, + lapack_int nrhs, + const lapack_complex_double *ap, + const lapack_complex_double *afp, + const lapack_int *ipiv, + const lapack_complex_double *b, + lapack_int ldb, + lapack_complex_double *x, + lapack_int ldx, + double *ferr, + double *berr); + +lapack_int LAPACKE_chpsv(int matrix_order, + char uplo, + lapack_int n, + lapack_int nrhs, + lapack_complex_float *ap, + lapack_int *ipiv, + lapack_complex_float *b, + lapack_int ldb); +lapack_int LAPACKE_zhpsv(int matrix_order, + char uplo, + lapack_int n, + lapack_int nrhs, + lapack_complex_double *ap, + lapack_int *ipiv, + lapack_complex_double *b, + lapack_int ldb); + +lapack_int LAPACKE_chpsvx(int matrix_order, + char fact, + char uplo, + lapack_int n, + lapack_int nrhs, + const lapack_complex_float *ap, + lapack_complex_float *afp, + lapack_int *ipiv, + const lapack_complex_float *b, + lapack_int ldb, + lapack_complex_float *x, + lapack_int ldx, + float *rcond, + float *ferr, + float *berr); +lapack_int LAPACKE_zhpsvx(int matrix_order, + char fact, + char uplo, + lapack_int n, + lapack_int nrhs, + const lapack_complex_double *ap, + lapack_complex_double *afp, + lapack_int *ipiv, + const lapack_complex_double *b, + lapack_int ldb, + lapack_complex_double *x, + lapack_int ldx, + double *rcond, + double *ferr, + double *berr); + +lapack_int LAPACKE_chptrd(int matrix_order, + char uplo, + lapack_int n, + lapack_complex_float *ap, + float *d, + float *e, + lapack_complex_float *tau); +lapack_int LAPACKE_zhptrd(int matrix_order, + char uplo, + lapack_int n, + lapack_complex_double *ap, + double *d, + double *e, + lapack_complex_double *tau); + +lapack_int LAPACKE_chptrf(int matrix_order, char uplo, lapack_int n, lapack_complex_float *ap, lapack_int *ipiv); +lapack_int LAPACKE_zhptrf(int matrix_order, char uplo, lapack_int n, lapack_complex_double *ap, lapack_int *ipiv); + +lapack_int LAPACKE_chptri(int matrix_order, char uplo, lapack_int n, lapack_complex_float *ap, const lapack_int *ipiv); +lapack_int LAPACKE_zhptri(int matrix_order, char uplo, lapack_int n, lapack_complex_double *ap, const lapack_int *ipiv); + +lapack_int LAPACKE_chptrs(int matrix_order, + char uplo, + lapack_int n, + lapack_int nrhs, + const lapack_complex_float *ap, + const lapack_int *ipiv, + lapack_complex_float *b, + lapack_int ldb); +lapack_int LAPACKE_zhptrs(int matrix_order, + char uplo, + lapack_int n, + lapack_int nrhs, + const lapack_complex_double *ap, + const lapack_int *ipiv, + lapack_complex_double *b, + lapack_int ldb); + +lapack_int LAPACKE_shsein(int matrix_order, + char job, + char eigsrc, + char initv, + lapack_logical *select, + lapack_int n, + const float *h, + lapack_int ldh, + float *wr, + const float *wi, + float *vl, + lapack_int ldvl, + float *vr, + lapack_int ldvr, + lapack_int mm, + lapack_int *m, + lapack_int *ifaill, + lapack_int *ifailr); +lapack_int LAPACKE_dhsein(int matrix_order, + char job, + char eigsrc, + char initv, + lapack_logical *select, + lapack_int n, + const double *h, + lapack_int ldh, + double *wr, + const double *wi, + double *vl, + lapack_int ldvl, + double *vr, + lapack_int ldvr, + lapack_int mm, + lapack_int *m, + lapack_int *ifaill, + lapack_int *ifailr); +lapack_int LAPACKE_chsein(int matrix_order, + char job, + char eigsrc, + char initv, + const lapack_logical *select, + lapack_int n, + const lapack_complex_float *h, + lapack_int ldh, + lapack_complex_float *w, + lapack_complex_float *vl, + lapack_int ldvl, + lapack_complex_float *vr, + lapack_int ldvr, + lapack_int mm, + lapack_int *m, + lapack_int *ifaill, + lapack_int *ifailr); +lapack_int LAPACKE_zhsein(int matrix_order, + char job, + char eigsrc, + char initv, + const lapack_logical *select, + lapack_int n, + const lapack_complex_double *h, + lapack_int ldh, + lapack_complex_double *w, + lapack_complex_double *vl, + lapack_int ldvl, + lapack_complex_double *vr, + lapack_int ldvr, + lapack_int mm, + lapack_int *m, + lapack_int *ifaill, + lapack_int *ifailr); + +lapack_int LAPACKE_shseqr(int matrix_order, + char job, + char compz, + lapack_int n, + lapack_int ilo, + lapack_int ihi, + float *h, + lapack_int ldh, + float *wr, + float *wi, + float *z, + lapack_int ldz); +lapack_int LAPACKE_dhseqr(int matrix_order, + char job, + char compz, + lapack_int n, + lapack_int ilo, + lapack_int ihi, + double *h, + lapack_int ldh, + double *wr, + double *wi, + double *z, + lapack_int ldz); +lapack_int LAPACKE_chseqr(int matrix_order, + char job, + char compz, + lapack_int n, + lapack_int ilo, + lapack_int ihi, + lapack_complex_float *h, + lapack_int ldh, + lapack_complex_float *w, + lapack_complex_float *z, + lapack_int ldz); +lapack_int LAPACKE_zhseqr(int matrix_order, + char job, + char compz, + lapack_int n, + lapack_int ilo, + lapack_int ihi, + lapack_complex_double *h, + lapack_int ldh, + lapack_complex_double *w, + lapack_complex_double *z, + lapack_int ldz); + +lapack_int LAPACKE_clacgv(lapack_int n, lapack_complex_float *x, lapack_int incx); +lapack_int LAPACKE_zlacgv(lapack_int n, lapack_complex_double *x, lapack_int incx); + +lapack_int LAPACKE_slacpy(int matrix_order, + char uplo, + lapack_int m, + lapack_int n, + const float *a, + lapack_int lda, + float *b, + lapack_int ldb); +lapack_int LAPACKE_dlacpy(int matrix_order, + char uplo, + lapack_int m, + lapack_int n, + const double *a, + lapack_int lda, + double *b, + lapack_int ldb); +lapack_int LAPACKE_clacpy(int matrix_order, + char uplo, + lapack_int m, + lapack_int n, + const lapack_complex_float *a, + lapack_int lda, + lapack_complex_float *b, + lapack_int ldb); +lapack_int LAPACKE_zlacpy(int matrix_order, + char uplo, + lapack_int m, + lapack_int n, + const lapack_complex_double *a, + lapack_int lda, + lapack_complex_double *b, + lapack_int ldb); + +lapack_int LAPACKE_zlag2c(int matrix_order, + lapack_int m, + lapack_int n, + const lapack_complex_double *a, + lapack_int lda, + lapack_complex_float *sa, + lapack_int ldsa); + +lapack_int LAPACKE_slag2d(int matrix_order, + lapack_int m, + lapack_int n, + const float *sa, + lapack_int ldsa, + double *a, + lapack_int lda); + +lapack_int LAPACKE_dlag2s(int matrix_order, + lapack_int m, + lapack_int n, + const double *a, + lapack_int lda, + float *sa, + lapack_int ldsa); + +lapack_int LAPACKE_clag2z(int matrix_order, + lapack_int m, + lapack_int n, + const lapack_complex_float *sa, + lapack_int ldsa, + lapack_complex_double *a, + lapack_int lda); + +lapack_int LAPACKE_slagge(int matrix_order, + lapack_int m, + lapack_int n, + lapack_int kl, + lapack_int ku, + const float *d, + float *a, + lapack_int lda, + lapack_int *iseed); +lapack_int LAPACKE_dlagge(int matrix_order, + lapack_int m, + lapack_int n, + lapack_int kl, + lapack_int ku, + const double *d, + double *a, + lapack_int lda, + lapack_int *iseed); +lapack_int LAPACKE_clagge(int matrix_order, + lapack_int m, + lapack_int n, + lapack_int kl, + lapack_int ku, + const float *d, + lapack_complex_float *a, + lapack_int lda, + lapack_int *iseed); +lapack_int LAPACKE_zlagge(int matrix_order, + lapack_int m, + lapack_int n, + lapack_int kl, + lapack_int ku, + const double *d, + lapack_complex_double *a, + lapack_int lda, + lapack_int *iseed); + +float LAPACKE_slamch(char cmach); +double LAPACKE_dlamch(char cmach); + +float LAPACKE_slange(int matrix_order, char norm, lapack_int m, lapack_int n, const float *a, lapack_int lda); +double LAPACKE_dlange(int matrix_order, char norm, lapack_int m, lapack_int n, const double *a, lapack_int lda); +float LAPACKE_clange(int matrix_order, + char norm, + lapack_int m, + lapack_int n, + const lapack_complex_float *a, + lapack_int lda); +double LAPACKE_zlange(int matrix_order, + char norm, + lapack_int m, + lapack_int n, + const lapack_complex_double *a, + lapack_int lda); + +float LAPACKE_clanhe(int matrix_order, + char norm, + char uplo, + lapack_int n, + const lapack_complex_float *a, + lapack_int lda); +double + LAPACKE_zlanhe(int matrix_order, char norm, char uplo, lapack_int n, const lapack_complex_double *a, lapack_int lda); + +float LAPACKE_slansy(int matrix_order, char norm, char uplo, lapack_int n, const float *a, lapack_int lda); +double LAPACKE_dlansy(int matrix_order, char norm, char uplo, lapack_int n, const double *a, lapack_int lda); +float LAPACKE_clansy(int matrix_order, + char norm, + char uplo, + lapack_int n, + const lapack_complex_float *a, + lapack_int lda); +double + LAPACKE_zlansy(int matrix_order, char norm, char uplo, lapack_int n, const lapack_complex_double *a, lapack_int lda); + +float LAPACKE_slantr(int matrix_order, + char norm, + char uplo, + char diag, + lapack_int m, + lapack_int n, + const float *a, + lapack_int lda); +double LAPACKE_dlantr(int matrix_order, + char norm, + char uplo, + char diag, + lapack_int m, + lapack_int n, + const double *a, + lapack_int lda); +float LAPACKE_clantr(int matrix_order, + char norm, + char uplo, + char diag, + lapack_int m, + lapack_int n, + const lapack_complex_float *a, + lapack_int lda); +double LAPACKE_zlantr(int matrix_order, + char norm, + char uplo, + char diag, + lapack_int m, + lapack_int n, + const lapack_complex_double *a, + lapack_int lda); + + +lapack_int LAPACKE_slarfb(int matrix_order, + char side, + char trans, + char direct, + char storev, + lapack_int m, + lapack_int n, + lapack_int k, + const float *v, + lapack_int ldv, + const float *t, + lapack_int ldt, + float *c, + lapack_int ldc); +lapack_int LAPACKE_dlarfb(int matrix_order, + char side, + char trans, + char direct, + char storev, + lapack_int m, + lapack_int n, + lapack_int k, + const double *v, + lapack_int ldv, + const double *t, + lapack_int ldt, + double *c, + lapack_int ldc); +lapack_int LAPACKE_clarfb(int matrix_order, + char side, + char trans, + char direct, + char storev, + lapack_int m, + lapack_int n, + lapack_int k, + const lapack_complex_float *v, + lapack_int ldv, + const lapack_complex_float *t, + lapack_int ldt, + lapack_complex_float *c, + lapack_int ldc); +lapack_int LAPACKE_zlarfb(int matrix_order, + char side, + char trans, + char direct, + char storev, + lapack_int m, + lapack_int n, + lapack_int k, + const lapack_complex_double *v, + lapack_int ldv, + const lapack_complex_double *t, + lapack_int ldt, + lapack_complex_double *c, + lapack_int ldc); + +lapack_int LAPACKE_slarfg(lapack_int n, float *alpha, float *x, lapack_int incx, float *tau); +lapack_int LAPACKE_dlarfg(lapack_int n, double *alpha, double *x, lapack_int incx, double *tau); +lapack_int LAPACKE_clarfg(lapack_int n, + lapack_complex_float *alpha, + lapack_complex_float *x, + lapack_int incx, + lapack_complex_float *tau); +lapack_int LAPACKE_zlarfg(lapack_int n, + lapack_complex_double *alpha, + lapack_complex_double *x, + lapack_int incx, + lapack_complex_double *tau); + +lapack_int LAPACKE_slarft(int matrix_order, + char direct, + char storev, + lapack_int n, + lapack_int k, + const float *v, + lapack_int ldv, + const float *tau, + float *t, + lapack_int ldt); +lapack_int LAPACKE_dlarft(int matrix_order, + char direct, + char storev, + lapack_int n, + lapack_int k, + const double *v, + lapack_int ldv, + const double *tau, + double *t, + lapack_int ldt); +lapack_int LAPACKE_clarft(int matrix_order, + char direct, + char storev, + lapack_int n, + lapack_int k, + const lapack_complex_float *v, + lapack_int ldv, + const lapack_complex_float *tau, + lapack_complex_float *t, + lapack_int ldt); +lapack_int LAPACKE_zlarft(int matrix_order, + char direct, + char storev, + lapack_int n, + lapack_int k, + const lapack_complex_double *v, + lapack_int ldv, + const lapack_complex_double *tau, + lapack_complex_double *t, + lapack_int ldt); + +lapack_int LAPACKE_slarfx(int matrix_order, + char side, + lapack_int m, + lapack_int n, + const float *v, + float tau, + float *c, + lapack_int ldc, + float *work); +lapack_int LAPACKE_dlarfx(int matrix_order, + char side, + lapack_int m, + lapack_int n, + const double *v, + double tau, + double *c, + lapack_int ldc, + double *work); +lapack_int LAPACKE_clarfx(int matrix_order, + char side, + lapack_int m, + lapack_int n, + const lapack_complex_float *v, + lapack_complex_float tau, + lapack_complex_float *c, + lapack_int ldc, + lapack_complex_float *work); +lapack_int LAPACKE_zlarfx(int matrix_order, + char side, + lapack_int m, + lapack_int n, + const lapack_complex_double *v, + lapack_complex_double tau, + lapack_complex_double *c, + lapack_int ldc, + lapack_complex_double *work); + +lapack_int LAPACKE_slarnv(lapack_int idist, lapack_int *iseed, lapack_int n, float *x); +lapack_int LAPACKE_dlarnv(lapack_int idist, lapack_int *iseed, lapack_int n, double *x); +lapack_int LAPACKE_clarnv(lapack_int idist, lapack_int *iseed, lapack_int n, lapack_complex_float *x); +lapack_int LAPACKE_zlarnv(lapack_int idist, lapack_int *iseed, lapack_int n, lapack_complex_double *x); + +lapack_int LAPACKE_slaset(int matrix_order, + char uplo, + lapack_int m, + lapack_int n, + float alpha, + float beta, + float *a, + lapack_int lda); +lapack_int LAPACKE_dlaset(int matrix_order, + char uplo, + lapack_int m, + lapack_int n, + double alpha, + double beta, + double *a, + lapack_int lda); +lapack_int LAPACKE_claset(int matrix_order, + char uplo, + lapack_int m, + lapack_int n, + lapack_complex_float alpha, + lapack_complex_float beta, + lapack_complex_float *a, + lapack_int lda); +lapack_int LAPACKE_zlaset(int matrix_order, + char uplo, + lapack_int m, + lapack_int n, + lapack_complex_double alpha, + lapack_complex_double beta, + lapack_complex_double *a, + lapack_int lda); + +lapack_int LAPACKE_slasrt(char id, lapack_int n, float *d); +lapack_int LAPACKE_dlasrt(char id, lapack_int n, double *d); + +lapack_int LAPACKE_slaswp(int matrix_order, + lapack_int n, + float *a, + lapack_int lda, + lapack_int k1, + lapack_int k2, + const lapack_int *ipiv, + lapack_int incx); +lapack_int LAPACKE_dlaswp(int matrix_order, + lapack_int n, + double *a, + lapack_int lda, + lapack_int k1, + lapack_int k2, + const lapack_int *ipiv, + lapack_int incx); +lapack_int LAPACKE_claswp(int matrix_order, + lapack_int n, + lapack_complex_float *a, + lapack_int lda, + lapack_int k1, + lapack_int k2, + const lapack_int *ipiv, + lapack_int incx); +lapack_int LAPACKE_zlaswp(int matrix_order, + lapack_int n, + lapack_complex_double *a, + lapack_int lda, + lapack_int k1, + lapack_int k2, + const lapack_int *ipiv, + lapack_int incx); + +lapack_int LAPACKE_slatms(int matrix_order, + lapack_int m, + lapack_int n, + char dist, + lapack_int *iseed, + char sym, + float *d, + lapack_int mode, + float cond, + float dmax, + lapack_int kl, + lapack_int ku, + char pack, + float *a, + lapack_int lda); +lapack_int LAPACKE_dlatms(int matrix_order, + lapack_int m, + lapack_int n, + char dist, + lapack_int *iseed, + char sym, + double *d, + lapack_int mode, + double cond, + double dmax, + lapack_int kl, + lapack_int ku, + char pack, + double *a, + lapack_int lda); +lapack_int LAPACKE_clatms(int matrix_order, + lapack_int m, + lapack_int n, + char dist, + lapack_int *iseed, + char sym, + float *d, + lapack_int mode, + float cond, + float dmax, + lapack_int kl, + lapack_int ku, + char pack, + lapack_complex_float *a, + lapack_int lda); +lapack_int LAPACKE_zlatms(int matrix_order, + lapack_int m, + lapack_int n, + char dist, + lapack_int *iseed, + char sym, + double *d, + lapack_int mode, + double cond, + double dmax, + lapack_int kl, + lapack_int ku, + char pack, + lapack_complex_double *a, + lapack_int lda); + +lapack_int LAPACKE_slauum(int matrix_order, char uplo, lapack_int n, float *a, lapack_int lda); +lapack_int LAPACKE_dlauum(int matrix_order, char uplo, lapack_int n, double *a, lapack_int lda); +lapack_int LAPACKE_clauum(int matrix_order, char uplo, lapack_int n, lapack_complex_float *a, lapack_int lda); +lapack_int LAPACKE_zlauum(int matrix_order, char uplo, lapack_int n, lapack_complex_double *a, lapack_int lda); + +lapack_int LAPACKE_sopgtr(int matrix_order, + char uplo, + lapack_int n, + const float *ap, + const float *tau, + float *q, + lapack_int ldq); +lapack_int LAPACKE_dopgtr(int matrix_order, + char uplo, + lapack_int n, + const double *ap, + const double *tau, + double *q, + lapack_int ldq); + +lapack_int LAPACKE_sopmtr(int matrix_order, + char side, + char uplo, + char trans, + lapack_int m, + lapack_int n, + const float *ap, + const float *tau, + float *c, + lapack_int ldc); +lapack_int LAPACKE_dopmtr(int matrix_order, + char side, + char uplo, + char trans, + lapack_int m, + lapack_int n, + const double *ap, + const double *tau, + double *c, + lapack_int ldc); + +lapack_int LAPACKE_sorgbr(int matrix_order, + char vect, + lapack_int m, + lapack_int n, + lapack_int k, + float *a, + lapack_int lda, + const float *tau); +lapack_int LAPACKE_dorgbr(int matrix_order, + char vect, + lapack_int m, + lapack_int n, + lapack_int k, + double *a, + lapack_int lda, + const double *tau); + +lapack_int LAPACKE_sorghr(int matrix_order, + lapack_int n, + lapack_int ilo, + lapack_int ihi, + float *a, + lapack_int lda, + const float *tau); +lapack_int LAPACKE_dorghr(int matrix_order, + lapack_int n, + lapack_int ilo, + lapack_int ihi, + double *a, + lapack_int lda, + const double *tau); + +lapack_int LAPACKE_sorglq(int matrix_order, + lapack_int m, + lapack_int n, + lapack_int k, + float *a, + lapack_int lda, + const float *tau); +lapack_int LAPACKE_dorglq(int matrix_order, + lapack_int m, + lapack_int n, + lapack_int k, + double *a, + lapack_int lda, + const double *tau); + +lapack_int LAPACKE_sorgql(int matrix_order, + lapack_int m, + lapack_int n, + lapack_int k, + float *a, + lapack_int lda, + const float *tau); +lapack_int LAPACKE_dorgql(int matrix_order, + lapack_int m, + lapack_int n, + lapack_int k, + double *a, + lapack_int lda, + const double *tau); + +lapack_int LAPACKE_sorgqr(int matrix_order, + lapack_int m, + lapack_int n, + lapack_int k, + float *a, + lapack_int lda, + const float *tau); +lapack_int LAPACKE_dorgqr(int matrix_order, + lapack_int m, + lapack_int n, + lapack_int k, + double *a, + lapack_int lda, + const double *tau); + +lapack_int LAPACKE_sorgrq(int matrix_order, + lapack_int m, + lapack_int n, + lapack_int k, + float *a, + lapack_int lda, + const float *tau); +lapack_int LAPACKE_dorgrq(int matrix_order, + lapack_int m, + lapack_int n, + lapack_int k, + double *a, + lapack_int lda, + const double *tau); + +lapack_int LAPACKE_sorgtr(int matrix_order, char uplo, lapack_int n, float *a, lapack_int lda, const float *tau); +lapack_int LAPACKE_dorgtr(int matrix_order, char uplo, lapack_int n, double *a, lapack_int lda, const double *tau); + +lapack_int LAPACKE_sormbr(int matrix_order, + char vect, + char side, + char trans, + lapack_int m, + lapack_int n, + lapack_int k, + const float *a, + lapack_int lda, + const float *tau, + float *c, + lapack_int ldc); +lapack_int LAPACKE_dormbr(int matrix_order, + char vect, + char side, + char trans, + lapack_int m, + lapack_int n, + lapack_int k, + const double *a, + lapack_int lda, + const double *tau, + double *c, + lapack_int ldc); + +lapack_int LAPACKE_sormhr(int matrix_order, + char side, + char trans, + lapack_int m, + lapack_int n, + lapack_int ilo, + lapack_int ihi, + const float *a, + lapack_int lda, + const float *tau, + float *c, + lapack_int ldc); +lapack_int LAPACKE_dormhr(int matrix_order, + char side, + char trans, + lapack_int m, + lapack_int n, + lapack_int ilo, + lapack_int ihi, + const double *a, + lapack_int lda, + const double *tau, + double *c, + lapack_int ldc); + +lapack_int LAPACKE_sormlq(int matrix_order, + char side, + char trans, + lapack_int m, + lapack_int n, + lapack_int k, + const float *a, + lapack_int lda, + const float *tau, + float *c, + lapack_int ldc); +lapack_int LAPACKE_dormlq(int matrix_order, + char side, + char trans, + lapack_int m, + lapack_int n, + lapack_int k, + const double *a, + lapack_int lda, + const double *tau, + double *c, + lapack_int ldc); + +lapack_int LAPACKE_sormql(int matrix_order, + char side, + char trans, + lapack_int m, + lapack_int n, + lapack_int k, + const float *a, + lapack_int lda, + const float *tau, + float *c, + lapack_int ldc); +lapack_int LAPACKE_dormql(int matrix_order, + char side, + char trans, + lapack_int m, + lapack_int n, + lapack_int k, + const double *a, + lapack_int lda, + const double *tau, + double *c, + lapack_int ldc); + +lapack_int LAPACKE_sormqr(int matrix_order, + char side, + char trans, + lapack_int m, + lapack_int n, + lapack_int k, + const float *a, + lapack_int lda, + const float *tau, + float *c, + lapack_int ldc); +lapack_int LAPACKE_dormqr(int matrix_order, + char side, + char trans, + lapack_int m, + lapack_int n, + lapack_int k, + const double *a, + lapack_int lda, + const double *tau, + double *c, + lapack_int ldc); + +lapack_int LAPACKE_sormrq(int matrix_order, + char side, + char trans, + lapack_int m, + lapack_int n, + lapack_int k, + const float *a, + lapack_int lda, + const float *tau, + float *c, + lapack_int ldc); +lapack_int LAPACKE_dormrq(int matrix_order, + char side, + char trans, + lapack_int m, + lapack_int n, + lapack_int k, + const double *a, + lapack_int lda, + const double *tau, + double *c, + lapack_int ldc); + +lapack_int LAPACKE_sormrz(int matrix_order, + char side, + char trans, + lapack_int m, + lapack_int n, + lapack_int k, + lapack_int l, + const float *a, + lapack_int lda, + const float *tau, + float *c, + lapack_int ldc); +lapack_int LAPACKE_dormrz(int matrix_order, + char side, + char trans, + lapack_int m, + lapack_int n, + lapack_int k, + lapack_int l, + const double *a, + lapack_int lda, + const double *tau, + double *c, + lapack_int ldc); + +lapack_int LAPACKE_sormtr(int matrix_order, + char side, + char uplo, + char trans, + lapack_int m, + lapack_int n, + const float *a, + lapack_int lda, + const float *tau, + float *c, + lapack_int ldc); +lapack_int LAPACKE_dormtr(int matrix_order, + char side, + char uplo, + char trans, + lapack_int m, + lapack_int n, + const double *a, + lapack_int lda, + const double *tau, + double *c, + lapack_int ldc); + +lapack_int LAPACKE_spbcon(int matrix_order, + char uplo, + lapack_int n, + lapack_int kd, + const float *ab, + lapack_int ldab, + float anorm, + float *rcond); +lapack_int LAPACKE_dpbcon(int matrix_order, + char uplo, + lapack_int n, + lapack_int kd, + const double *ab, + lapack_int ldab, + double anorm, + double *rcond); +lapack_int LAPACKE_cpbcon(int matrix_order, + char uplo, + lapack_int n, + lapack_int kd, + const lapack_complex_float *ab, + lapack_int ldab, + float anorm, + float *rcond); +lapack_int LAPACKE_zpbcon(int matrix_order, + char uplo, + lapack_int n, + lapack_int kd, + const lapack_complex_double *ab, + lapack_int ldab, + double anorm, + double *rcond); + +lapack_int LAPACKE_spbequ(int matrix_order, + char uplo, + lapack_int n, + lapack_int kd, + const float *ab, + lapack_int ldab, + float *s, + float *scond, + float *amax); +lapack_int LAPACKE_dpbequ(int matrix_order, + char uplo, + lapack_int n, + lapack_int kd, + const double *ab, + lapack_int ldab, + double *s, + double *scond, + double *amax); +lapack_int LAPACKE_cpbequ(int matrix_order, + char uplo, + lapack_int n, + lapack_int kd, + const lapack_complex_float *ab, + lapack_int ldab, + float *s, + float *scond, + float *amax); +lapack_int LAPACKE_zpbequ(int matrix_order, + char uplo, + lapack_int n, + lapack_int kd, + const lapack_complex_double *ab, + lapack_int ldab, + double *s, + double *scond, + double *amax); + +lapack_int LAPACKE_spbrfs(int matrix_order, + char uplo, + lapack_int n, + lapack_int kd, + lapack_int nrhs, + const float *ab, + lapack_int ldab, + const float *afb, + lapack_int ldafb, + const float *b, + lapack_int ldb, + float *x, + lapack_int ldx, + float *ferr, + float *berr); +lapack_int LAPACKE_dpbrfs(int matrix_order, + char uplo, + lapack_int n, + lapack_int kd, + lapack_int nrhs, + const double *ab, + lapack_int ldab, + const double *afb, + lapack_int ldafb, + const double *b, + lapack_int ldb, + double *x, + lapack_int ldx, + double *ferr, + double *berr); +lapack_int LAPACKE_cpbrfs(int matrix_order, + char uplo, + lapack_int n, + lapack_int kd, + lapack_int nrhs, + const lapack_complex_float *ab, + lapack_int ldab, + const lapack_complex_float *afb, + lapack_int ldafb, + const lapack_complex_float *b, + lapack_int ldb, + lapack_complex_float *x, + lapack_int ldx, + float *ferr, + float *berr); +lapack_int LAPACKE_zpbrfs(int matrix_order, + char uplo, + lapack_int n, + lapack_int kd, + lapack_int nrhs, + const lapack_complex_double *ab, + lapack_int ldab, + const lapack_complex_double *afb, + lapack_int ldafb, + const lapack_complex_double *b, + lapack_int ldb, + lapack_complex_double *x, + lapack_int ldx, + double *ferr, + double *berr); + +lapack_int LAPACKE_spbstf(int matrix_order, char uplo, lapack_int n, lapack_int kb, float *bb, lapack_int ldbb); +lapack_int LAPACKE_dpbstf(int matrix_order, char uplo, lapack_int n, lapack_int kb, double *bb, lapack_int ldbb); +lapack_int + LAPACKE_cpbstf(int matrix_order, char uplo, lapack_int n, lapack_int kb, lapack_complex_float *bb, lapack_int ldbb); +lapack_int + LAPACKE_zpbstf(int matrix_order, char uplo, lapack_int n, lapack_int kb, lapack_complex_double *bb, lapack_int ldbb); + +lapack_int LAPACKE_spbsv(int matrix_order, + char uplo, + lapack_int n, + lapack_int kd, + lapack_int nrhs, + float *ab, + lapack_int ldab, + float *b, + lapack_int ldb); +lapack_int LAPACKE_dpbsv(int matrix_order, + char uplo, + lapack_int n, + lapack_int kd, + lapack_int nrhs, + double *ab, + lapack_int ldab, + double *b, + lapack_int ldb); +lapack_int LAPACKE_cpbsv(int matrix_order, + char uplo, + lapack_int n, + lapack_int kd, + lapack_int nrhs, + lapack_complex_float *ab, + lapack_int ldab, + lapack_complex_float *b, + lapack_int ldb); +lapack_int LAPACKE_zpbsv(int matrix_order, + char uplo, + lapack_int n, + lapack_int kd, + lapack_int nrhs, + lapack_complex_double *ab, + lapack_int ldab, + lapack_complex_double *b, + lapack_int ldb); + +lapack_int LAPACKE_spbsvx(int matrix_order, + char fact, + char uplo, + lapack_int n, + lapack_int kd, + lapack_int nrhs, + float *ab, + lapack_int ldab, + float *afb, + lapack_int ldafb, + char *equed, + float *s, + float *b, + lapack_int ldb, + float *x, + lapack_int ldx, + float *rcond, + float *ferr, + float *berr); +lapack_int LAPACKE_dpbsvx(int matrix_order, + char fact, + char uplo, + lapack_int n, + lapack_int kd, + lapack_int nrhs, + double *ab, + lapack_int ldab, + double *afb, + lapack_int ldafb, + char *equed, + double *s, + double *b, + lapack_int ldb, + double *x, + lapack_int ldx, + double *rcond, + double *ferr, + double *berr); +lapack_int LAPACKE_cpbsvx(int matrix_order, + char fact, + char uplo, + lapack_int n, + lapack_int kd, + lapack_int nrhs, + lapack_complex_float *ab, + lapack_int ldab, + lapack_complex_float *afb, + lapack_int ldafb, + char *equed, + float *s, + lapack_complex_float *b, + lapack_int ldb, + lapack_complex_float *x, + lapack_int ldx, + float *rcond, + float *ferr, + float *berr); +lapack_int LAPACKE_zpbsvx(int matrix_order, + char fact, + char uplo, + lapack_int n, + lapack_int kd, + lapack_int nrhs, + lapack_complex_double *ab, + lapack_int ldab, + lapack_complex_double *afb, + lapack_int ldafb, + char *equed, + double *s, + lapack_complex_double *b, + lapack_int ldb, + lapack_complex_double *x, + lapack_int ldx, + double *rcond, + double *ferr, + double *berr); + +lapack_int LAPACKE_spbtrf(int matrix_order, char uplo, lapack_int n, lapack_int kd, float *ab, lapack_int ldab); +lapack_int LAPACKE_dpbtrf(int matrix_order, char uplo, lapack_int n, lapack_int kd, double *ab, lapack_int ldab); +lapack_int + LAPACKE_cpbtrf(int matrix_order, char uplo, lapack_int n, lapack_int kd, lapack_complex_float *ab, lapack_int ldab); +lapack_int + LAPACKE_zpbtrf(int matrix_order, char uplo, lapack_int n, lapack_int kd, lapack_complex_double *ab, lapack_int ldab); + +lapack_int LAPACKE_spbtrs(int matrix_order, + char uplo, + lapack_int n, + lapack_int kd, + lapack_int nrhs, + const float *ab, + lapack_int ldab, + float *b, + lapack_int ldb); +lapack_int LAPACKE_dpbtrs(int matrix_order, + char uplo, + lapack_int n, + lapack_int kd, + lapack_int nrhs, + const double *ab, + lapack_int ldab, + double *b, + lapack_int ldb); +lapack_int LAPACKE_cpbtrs(int matrix_order, + char uplo, + lapack_int n, + lapack_int kd, + lapack_int nrhs, + const lapack_complex_float *ab, + lapack_int ldab, + lapack_complex_float *b, + lapack_int ldb); +lapack_int LAPACKE_zpbtrs(int matrix_order, + char uplo, + lapack_int n, + lapack_int kd, + lapack_int nrhs, + const lapack_complex_double *ab, + lapack_int ldab, + lapack_complex_double *b, + lapack_int ldb); + +lapack_int LAPACKE_spftrf(int matrix_order, char transr, char uplo, lapack_int n, float *a); +lapack_int LAPACKE_dpftrf(int matrix_order, char transr, char uplo, lapack_int n, double *a); +lapack_int LAPACKE_cpftrf(int matrix_order, char transr, char uplo, lapack_int n, lapack_complex_float *a); +lapack_int LAPACKE_zpftrf(int matrix_order, char transr, char uplo, lapack_int n, lapack_complex_double *a); + +lapack_int LAPACKE_spftri(int matrix_order, char transr, char uplo, lapack_int n, float *a); +lapack_int LAPACKE_dpftri(int matrix_order, char transr, char uplo, lapack_int n, double *a); +lapack_int LAPACKE_cpftri(int matrix_order, char transr, char uplo, lapack_int n, lapack_complex_float *a); +lapack_int LAPACKE_zpftri(int matrix_order, char transr, char uplo, lapack_int n, lapack_complex_double *a); + +lapack_int LAPACKE_spftrs(int matrix_order, + char transr, + char uplo, + lapack_int n, + lapack_int nrhs, + const float *a, + float *b, + lapack_int ldb); +lapack_int LAPACKE_dpftrs(int matrix_order, + char transr, + char uplo, + lapack_int n, + lapack_int nrhs, + const double *a, + double *b, + lapack_int ldb); +lapack_int LAPACKE_cpftrs(int matrix_order, + char transr, + char uplo, + lapack_int n, + lapack_int nrhs, + const lapack_complex_float *a, + lapack_complex_float *b, + lapack_int ldb); +lapack_int LAPACKE_zpftrs(int matrix_order, + char transr, + char uplo, + lapack_int n, + lapack_int nrhs, + const lapack_complex_double *a, + lapack_complex_double *b, + lapack_int ldb); + +lapack_int + LAPACKE_spocon(int matrix_order, char uplo, lapack_int n, const float *a, lapack_int lda, float anorm, float *rcond); +lapack_int LAPACKE_dpocon(int matrix_order, + char uplo, + lapack_int n, + const double *a, + lapack_int lda, + double anorm, + double *rcond); +lapack_int LAPACKE_cpocon(int matrix_order, + char uplo, + lapack_int n, + const lapack_complex_float *a, + lapack_int lda, + float anorm, + float *rcond); +lapack_int LAPACKE_zpocon(int matrix_order, + char uplo, + lapack_int n, + const lapack_complex_double *a, + lapack_int lda, + double anorm, + double *rcond); + +lapack_int + LAPACKE_spoequ(int matrix_order, lapack_int n, const float *a, lapack_int lda, float *s, float *scond, float *amax); +lapack_int LAPACKE_dpoequ(int matrix_order, + lapack_int n, + const double *a, + lapack_int lda, + double *s, + double *scond, + double *amax); +lapack_int LAPACKE_cpoequ(int matrix_order, + lapack_int n, + const lapack_complex_float *a, + lapack_int lda, + float *s, + float *scond, + float *amax); +lapack_int LAPACKE_zpoequ(int matrix_order, + lapack_int n, + const lapack_complex_double *a, + lapack_int lda, + double *s, + double *scond, + double *amax); + +lapack_int + LAPACKE_spoequb(int matrix_order, lapack_int n, const float *a, lapack_int lda, float *s, float *scond, float *amax); +lapack_int LAPACKE_dpoequb(int matrix_order, + lapack_int n, + const double *a, + lapack_int lda, + double *s, + double *scond, + double *amax); +lapack_int LAPACKE_cpoequb(int matrix_order, + lapack_int n, + const lapack_complex_float *a, + lapack_int lda, + float *s, + float *scond, + float *amax); +lapack_int LAPACKE_zpoequb(int matrix_order, + lapack_int n, + const lapack_complex_double *a, + lapack_int lda, + double *s, + double *scond, + double *amax); + +lapack_int LAPACKE_sporfs(int matrix_order, + char uplo, + lapack_int n, + lapack_int nrhs, + const float *a, + lapack_int lda, + const float *af, + lapack_int ldaf, + const float *b, + lapack_int ldb, + float *x, + lapack_int ldx, + float *ferr, + float *berr); +lapack_int LAPACKE_dporfs(int matrix_order, + char uplo, + lapack_int n, + lapack_int nrhs, + const double *a, + lapack_int lda, + const double *af, + lapack_int ldaf, + const double *b, + lapack_int ldb, + double *x, + lapack_int ldx, + double *ferr, + double *berr); +lapack_int LAPACKE_cporfs(int matrix_order, + char uplo, + lapack_int n, + lapack_int nrhs, + const lapack_complex_float *a, + lapack_int lda, + const lapack_complex_float *af, + lapack_int ldaf, + const lapack_complex_float *b, + lapack_int ldb, + lapack_complex_float *x, + lapack_int ldx, + float *ferr, + float *berr); +lapack_int LAPACKE_zporfs(int matrix_order, + char uplo, + lapack_int n, + lapack_int nrhs, + const lapack_complex_double *a, + lapack_int lda, + const lapack_complex_double *af, + lapack_int ldaf, + const lapack_complex_double *b, + lapack_int ldb, + lapack_complex_double *x, + lapack_int ldx, + double *ferr, + double *berr); + +lapack_int LAPACKE_sporfsx(int matrix_order, + char uplo, + char equed, + lapack_int n, + lapack_int nrhs, + const float *a, + lapack_int lda, + const float *af, + lapack_int ldaf, + const float *s, + const float *b, + lapack_int ldb, + float *x, + lapack_int ldx, + float *rcond, + float *berr, + lapack_int n_err_bnds, + float *err_bnds_norm, + float *err_bnds_comp, + lapack_int nparams, + float *params); +lapack_int LAPACKE_dporfsx(int matrix_order, + char uplo, + char equed, + lapack_int n, + lapack_int nrhs, + const double *a, + lapack_int lda, + const double *af, + lapack_int ldaf, + const double *s, + const double *b, + lapack_int ldb, + double *x, + lapack_int ldx, + double *rcond, + double *berr, + lapack_int n_err_bnds, + double *err_bnds_norm, + double *err_bnds_comp, + lapack_int nparams, + double *params); +lapack_int LAPACKE_cporfsx(int matrix_order, + char uplo, + char equed, + lapack_int n, + lapack_int nrhs, + const lapack_complex_float *a, + lapack_int lda, + const lapack_complex_float *af, + lapack_int ldaf, + const float *s, + const lapack_complex_float *b, + lapack_int ldb, + lapack_complex_float *x, + lapack_int ldx, + float *rcond, + float *berr, + lapack_int n_err_bnds, + float *err_bnds_norm, + float *err_bnds_comp, + lapack_int nparams, + float *params); +lapack_int LAPACKE_zporfsx(int matrix_order, + char uplo, + char equed, + lapack_int n, + lapack_int nrhs, + const lapack_complex_double *a, + lapack_int lda, + const lapack_complex_double *af, + lapack_int ldaf, + const double *s, + const lapack_complex_double *b, + lapack_int ldb, + lapack_complex_double *x, + lapack_int ldx, + double *rcond, + double *berr, + lapack_int n_err_bnds, + double *err_bnds_norm, + double *err_bnds_comp, + lapack_int nparams, + double *params); + +lapack_int LAPACKE_sposv(int matrix_order, + char uplo, + lapack_int n, + lapack_int nrhs, + float *a, + lapack_int lda, + float *b, + lapack_int ldb); +lapack_int LAPACKE_dposv(int matrix_order, + char uplo, + lapack_int n, + lapack_int nrhs, + double *a, + lapack_int lda, + double *b, + lapack_int ldb); +lapack_int LAPACKE_cposv(int matrix_order, + char uplo, + lapack_int n, + lapack_int nrhs, + lapack_complex_float *a, + lapack_int lda, + lapack_complex_float *b, + lapack_int ldb); +lapack_int LAPACKE_zposv(int matrix_order, + char uplo, + lapack_int n, + lapack_int nrhs, + lapack_complex_double *a, + lapack_int lda, + lapack_complex_double *b, + lapack_int ldb); +lapack_int LAPACKE_dsposv(int matrix_order, + char uplo, + lapack_int n, + lapack_int nrhs, + double *a, + lapack_int lda, + double *b, + lapack_int ldb, + double *x, + lapack_int ldx, + lapack_int *iter); +lapack_int LAPACKE_zcposv(int matrix_order, + char uplo, + lapack_int n, + lapack_int nrhs, + lapack_complex_double *a, + lapack_int lda, + lapack_complex_double *b, + lapack_int ldb, + lapack_complex_double *x, + lapack_int ldx, + lapack_int *iter); + +lapack_int LAPACKE_sposvx(int matrix_order, + char fact, + char uplo, + lapack_int n, + lapack_int nrhs, + float *a, + lapack_int lda, + float *af, + lapack_int ldaf, + char *equed, + float *s, + float *b, + lapack_int ldb, + float *x, + lapack_int ldx, + float *rcond, + float *ferr, + float *berr); +lapack_int LAPACKE_dposvx(int matrix_order, + char fact, + char uplo, + lapack_int n, + lapack_int nrhs, + double *a, + lapack_int lda, + double *af, + lapack_int ldaf, + char *equed, + double *s, + double *b, + lapack_int ldb, + double *x, + lapack_int ldx, + double *rcond, + double *ferr, + double *berr); +lapack_int LAPACKE_cposvx(int matrix_order, + char fact, + char uplo, + lapack_int n, + lapack_int nrhs, + lapack_complex_float *a, + lapack_int lda, + lapack_complex_float *af, + lapack_int ldaf, + char *equed, + float *s, + lapack_complex_float *b, + lapack_int ldb, + lapack_complex_float *x, + lapack_int ldx, + float *rcond, + float *ferr, + float *berr); +lapack_int LAPACKE_zposvx(int matrix_order, + char fact, + char uplo, + lapack_int n, + lapack_int nrhs, + lapack_complex_double *a, + lapack_int lda, + lapack_complex_double *af, + lapack_int ldaf, + char *equed, + double *s, + lapack_complex_double *b, + lapack_int ldb, + lapack_complex_double *x, + lapack_int ldx, + double *rcond, + double *ferr, + double *berr); + +lapack_int LAPACKE_sposvxx(int matrix_order, + char fact, + char uplo, + lapack_int n, + lapack_int nrhs, + float *a, + lapack_int lda, + float *af, + lapack_int ldaf, + char *equed, + float *s, + float *b, + lapack_int ldb, + float *x, + lapack_int ldx, + float *rcond, + float *rpvgrw, + float *berr, + lapack_int n_err_bnds, + float *err_bnds_norm, + float *err_bnds_comp, + lapack_int nparams, + float *params); +lapack_int LAPACKE_dposvxx(int matrix_order, + char fact, + char uplo, + lapack_int n, + lapack_int nrhs, + double *a, + lapack_int lda, + double *af, + lapack_int ldaf, + char *equed, + double *s, + double *b, + lapack_int ldb, + double *x, + lapack_int ldx, + double *rcond, + double *rpvgrw, + double *berr, + lapack_int n_err_bnds, + double *err_bnds_norm, + double *err_bnds_comp, + lapack_int nparams, + double *params); +lapack_int LAPACKE_cposvxx(int matrix_order, + char fact, + char uplo, + lapack_int n, + lapack_int nrhs, + lapack_complex_float *a, + lapack_int lda, + lapack_complex_float *af, + lapack_int ldaf, + char *equed, + float *s, + lapack_complex_float *b, + lapack_int ldb, + lapack_complex_float *x, + lapack_int ldx, + float *rcond, + float *rpvgrw, + float *berr, + lapack_int n_err_bnds, + float *err_bnds_norm, + float *err_bnds_comp, + lapack_int nparams, + float *params); +lapack_int LAPACKE_zposvxx(int matrix_order, + char fact, + char uplo, + lapack_int n, + lapack_int nrhs, + lapack_complex_double *a, + lapack_int lda, + lapack_complex_double *af, + lapack_int ldaf, + char *equed, + double *s, + lapack_complex_double *b, + lapack_int ldb, + lapack_complex_double *x, + lapack_int ldx, + double *rcond, + double *rpvgrw, + double *berr, + lapack_int n_err_bnds, + double *err_bnds_norm, + double *err_bnds_comp, + lapack_int nparams, + double *params); + +lapack_int LAPACKE_spotrf(int matrix_order, char uplo, lapack_int n, float *a, lapack_int lda); +lapack_int LAPACKE_dpotrf(int matrix_order, char uplo, lapack_int n, double *a, lapack_int lda); +lapack_int LAPACKE_cpotrf(int matrix_order, char uplo, lapack_int n, lapack_complex_float *a, lapack_int lda); +lapack_int LAPACKE_zpotrf(int matrix_order, char uplo, lapack_int n, lapack_complex_double *a, lapack_int lda); + +lapack_int LAPACKE_spotri(int matrix_order, char uplo, lapack_int n, float *a, lapack_int lda); +lapack_int LAPACKE_dpotri(int matrix_order, char uplo, lapack_int n, double *a, lapack_int lda); +lapack_int LAPACKE_cpotri(int matrix_order, char uplo, lapack_int n, lapack_complex_float *a, lapack_int lda); +lapack_int LAPACKE_zpotri(int matrix_order, char uplo, lapack_int n, lapack_complex_double *a, lapack_int lda); + +lapack_int LAPACKE_spotrs(int matrix_order, + char uplo, + lapack_int n, + lapack_int nrhs, + const float *a, + lapack_int lda, + float *b, + lapack_int ldb); +lapack_int LAPACKE_dpotrs(int matrix_order, + char uplo, + lapack_int n, + lapack_int nrhs, + const double *a, + lapack_int lda, + double *b, + lapack_int ldb); +lapack_int LAPACKE_cpotrs(int matrix_order, + char uplo, + lapack_int n, + lapack_int nrhs, + const lapack_complex_float *a, + lapack_int lda, + lapack_complex_float *b, + lapack_int ldb); +lapack_int LAPACKE_zpotrs(int matrix_order, + char uplo, + lapack_int n, + lapack_int nrhs, + const lapack_complex_double *a, + lapack_int lda, + lapack_complex_double *b, + lapack_int ldb); + +lapack_int LAPACKE_sppcon(int matrix_order, char uplo, lapack_int n, const float *ap, float anorm, float *rcond); +lapack_int LAPACKE_dppcon(int matrix_order, char uplo, lapack_int n, const double *ap, double anorm, double *rcond); +lapack_int + LAPACKE_cppcon(int matrix_order, char uplo, lapack_int n, const lapack_complex_float *ap, float anorm, float *rcond); +lapack_int LAPACKE_zppcon(int matrix_order, + char uplo, + lapack_int n, + const lapack_complex_double *ap, + double anorm, + double *rcond); + +lapack_int + LAPACKE_sppequ(int matrix_order, char uplo, lapack_int n, const float *ap, float *s, float *scond, float *amax); +lapack_int + LAPACKE_dppequ(int matrix_order, char uplo, lapack_int n, const double *ap, double *s, double *scond, double *amax); +lapack_int LAPACKE_cppequ(int matrix_order, + char uplo, + lapack_int n, + const lapack_complex_float *ap, + float *s, + float *scond, + float *amax); +lapack_int LAPACKE_zppequ(int matrix_order, + char uplo, + lapack_int n, + const lapack_complex_double *ap, + double *s, + double *scond, + double *amax); + +lapack_int LAPACKE_spprfs(int matrix_order, + char uplo, + lapack_int n, + lapack_int nrhs, + const float *ap, + const float *afp, + const float *b, + lapack_int ldb, + float *x, + lapack_int ldx, + float *ferr, + float *berr); +lapack_int LAPACKE_dpprfs(int matrix_order, + char uplo, + lapack_int n, + lapack_int nrhs, + const double *ap, + const double *afp, + const double *b, + lapack_int ldb, + double *x, + lapack_int ldx, + double *ferr, + double *berr); +lapack_int LAPACKE_cpprfs(int matrix_order, + char uplo, + lapack_int n, + lapack_int nrhs, + const lapack_complex_float *ap, + const lapack_complex_float *afp, + const lapack_complex_float *b, + lapack_int ldb, + lapack_complex_float *x, + lapack_int ldx, + float *ferr, + float *berr); +lapack_int LAPACKE_zpprfs(int matrix_order, + char uplo, + lapack_int n, + lapack_int nrhs, + const lapack_complex_double *ap, + const lapack_complex_double *afp, + const lapack_complex_double *b, + lapack_int ldb, + lapack_complex_double *x, + lapack_int ldx, + double *ferr, + double *berr); + +lapack_int + LAPACKE_sppsv(int matrix_order, char uplo, lapack_int n, lapack_int nrhs, float *ap, float *b, lapack_int ldb); +lapack_int + LAPACKE_dppsv(int matrix_order, char uplo, lapack_int n, lapack_int nrhs, double *ap, double *b, lapack_int ldb); +lapack_int LAPACKE_cppsv(int matrix_order, + char uplo, + lapack_int n, + lapack_int nrhs, + lapack_complex_float *ap, + lapack_complex_float *b, + lapack_int ldb); +lapack_int LAPACKE_zppsv(int matrix_order, + char uplo, + lapack_int n, + lapack_int nrhs, + lapack_complex_double *ap, + lapack_complex_double *b, + lapack_int ldb); + +lapack_int LAPACKE_sppsvx(int matrix_order, + char fact, + char uplo, + lapack_int n, + lapack_int nrhs, + float *ap, + float *afp, + char *equed, + float *s, + float *b, + lapack_int ldb, + float *x, + lapack_int ldx, + float *rcond, + float *ferr, + float *berr); +lapack_int LAPACKE_dppsvx(int matrix_order, + char fact, + char uplo, + lapack_int n, + lapack_int nrhs, + double *ap, + double *afp, + char *equed, + double *s, + double *b, + lapack_int ldb, + double *x, + lapack_int ldx, + double *rcond, + double *ferr, + double *berr); +lapack_int LAPACKE_cppsvx(int matrix_order, + char fact, + char uplo, + lapack_int n, + lapack_int nrhs, + lapack_complex_float *ap, + lapack_complex_float *afp, + char *equed, + float *s, + lapack_complex_float *b, + lapack_int ldb, + lapack_complex_float *x, + lapack_int ldx, + float *rcond, + float *ferr, + float *berr); +lapack_int LAPACKE_zppsvx(int matrix_order, + char fact, + char uplo, + lapack_int n, + lapack_int nrhs, + lapack_complex_double *ap, + lapack_complex_double *afp, + char *equed, + double *s, + lapack_complex_double *b, + lapack_int ldb, + lapack_complex_double *x, + lapack_int ldx, + double *rcond, + double *ferr, + double *berr); + +lapack_int LAPACKE_spptrf(int matrix_order, char uplo, lapack_int n, float *ap); +lapack_int LAPACKE_dpptrf(int matrix_order, char uplo, lapack_int n, double *ap); +lapack_int LAPACKE_cpptrf(int matrix_order, char uplo, lapack_int n, lapack_complex_float *ap); +lapack_int LAPACKE_zpptrf(int matrix_order, char uplo, lapack_int n, lapack_complex_double *ap); + +lapack_int LAPACKE_spptri(int matrix_order, char uplo, lapack_int n, float *ap); +lapack_int LAPACKE_dpptri(int matrix_order, char uplo, lapack_int n, double *ap); +lapack_int LAPACKE_cpptri(int matrix_order, char uplo, lapack_int n, lapack_complex_float *ap); +lapack_int LAPACKE_zpptri(int matrix_order, char uplo, lapack_int n, lapack_complex_double *ap); + +lapack_int + LAPACKE_spptrs(int matrix_order, char uplo, lapack_int n, lapack_int nrhs, const float *ap, float *b, lapack_int ldb); +lapack_int LAPACKE_dpptrs(int matrix_order, + char uplo, + lapack_int n, + lapack_int nrhs, + const double *ap, + double *b, + lapack_int ldb); +lapack_int LAPACKE_cpptrs(int matrix_order, + char uplo, + lapack_int n, + lapack_int nrhs, + const lapack_complex_float *ap, + lapack_complex_float *b, + lapack_int ldb); +lapack_int LAPACKE_zpptrs(int matrix_order, + char uplo, + lapack_int n, + lapack_int nrhs, + const lapack_complex_double *ap, + lapack_complex_double *b, + lapack_int ldb); + +lapack_int LAPACKE_spstrf(int matrix_order, + char uplo, + lapack_int n, + float *a, + lapack_int lda, + lapack_int *piv, + lapack_int *rank, + float tol); +lapack_int LAPACKE_dpstrf(int matrix_order, + char uplo, + lapack_int n, + double *a, + lapack_int lda, + lapack_int *piv, + lapack_int *rank, + double tol); +lapack_int LAPACKE_cpstrf(int matrix_order, + char uplo, + lapack_int n, + lapack_complex_float *a, + lapack_int lda, + lapack_int *piv, + lapack_int *rank, + float tol); +lapack_int LAPACKE_zpstrf(int matrix_order, + char uplo, + lapack_int n, + lapack_complex_double *a, + lapack_int lda, + lapack_int *piv, + lapack_int *rank, + double tol); + +lapack_int LAPACKE_sptcon(lapack_int n, const float *d, const float *e, float anorm, float *rcond); +lapack_int LAPACKE_dptcon(lapack_int n, const double *d, const double *e, double anorm, double *rcond); +lapack_int LAPACKE_cptcon(lapack_int n, const float *d, const lapack_complex_float *e, float anorm, float *rcond); +lapack_int LAPACKE_zptcon(lapack_int n, const double *d, const lapack_complex_double *e, double anorm, double *rcond); + +lapack_int LAPACKE_spteqr(int matrix_order, char compz, lapack_int n, float *d, float *e, float *z, lapack_int ldz); +lapack_int LAPACKE_dpteqr(int matrix_order, char compz, lapack_int n, double *d, double *e, double *z, lapack_int ldz); +lapack_int LAPACKE_cpteqr(int matrix_order, + char compz, + lapack_int n, + float *d, + float *e, + lapack_complex_float *z, + lapack_int ldz); +lapack_int LAPACKE_zpteqr(int matrix_order, + char compz, + lapack_int n, + double *d, + double *e, + lapack_complex_double *z, + lapack_int ldz); + +lapack_int LAPACKE_sptrfs(int matrix_order, + lapack_int n, + lapack_int nrhs, + const float *d, + const float *e, + const float *df, + const float *ef, + const float *b, + lapack_int ldb, + float *x, + lapack_int ldx, + float *ferr, + float *berr); +lapack_int LAPACKE_dptrfs(int matrix_order, + lapack_int n, + lapack_int nrhs, + const double *d, + const double *e, + const double *df, + const double *ef, + const double *b, + lapack_int ldb, + double *x, + lapack_int ldx, + double *ferr, + double *berr); +lapack_int LAPACKE_cptrfs(int matrix_order, + char uplo, + lapack_int n, + lapack_int nrhs, + const float *d, + const lapack_complex_float *e, + const float *df, + const lapack_complex_float *ef, + const lapack_complex_float *b, + lapack_int ldb, + lapack_complex_float *x, + lapack_int ldx, + float *ferr, + float *berr); +lapack_int LAPACKE_zptrfs(int matrix_order, + char uplo, + lapack_int n, + lapack_int nrhs, + const double *d, + const lapack_complex_double *e, + const double *df, + const lapack_complex_double *ef, + const lapack_complex_double *b, + lapack_int ldb, + lapack_complex_double *x, + lapack_int ldx, + double *ferr, + double *berr); + +lapack_int LAPACKE_sptsv(int matrix_order, lapack_int n, lapack_int nrhs, float *d, float *e, float *b, lapack_int ldb); +lapack_int + LAPACKE_dptsv(int matrix_order, lapack_int n, lapack_int nrhs, double *d, double *e, double *b, lapack_int ldb); +lapack_int LAPACKE_cptsv(int matrix_order, + lapack_int n, + lapack_int nrhs, + float *d, + lapack_complex_float *e, + lapack_complex_float *b, + lapack_int ldb); +lapack_int LAPACKE_zptsv(int matrix_order, + lapack_int n, + lapack_int nrhs, + double *d, + lapack_complex_double *e, + lapack_complex_double *b, + lapack_int ldb); + +lapack_int LAPACKE_sptsvx(int matrix_order, + char fact, + lapack_int n, + lapack_int nrhs, + const float *d, + const float *e, + float *df, + float *ef, + const float *b, + lapack_int ldb, + float *x, + lapack_int ldx, + float *rcond, + float *ferr, + float *berr); +lapack_int LAPACKE_dptsvx(int matrix_order, + char fact, + lapack_int n, + lapack_int nrhs, + const double *d, + const double *e, + double *df, + double *ef, + const double *b, + lapack_int ldb, + double *x, + lapack_int ldx, + double *rcond, + double *ferr, + double *berr); +lapack_int LAPACKE_cptsvx(int matrix_order, + char fact, + lapack_int n, + lapack_int nrhs, + const float *d, + const lapack_complex_float *e, + float *df, + lapack_complex_float *ef, + const lapack_complex_float *b, + lapack_int ldb, + lapack_complex_float *x, + lapack_int ldx, + float *rcond, + float *ferr, + float *berr); +lapack_int LAPACKE_zptsvx(int matrix_order, + char fact, + lapack_int n, + lapack_int nrhs, + const double *d, + const lapack_complex_double *e, + double *df, + lapack_complex_double *ef, + const lapack_complex_double *b, + lapack_int ldb, + lapack_complex_double *x, + lapack_int ldx, + double *rcond, + double *ferr, + double *berr); + +lapack_int LAPACKE_spttrf(lapack_int n, float *d, float *e); +lapack_int LAPACKE_dpttrf(lapack_int n, double *d, double *e); +lapack_int LAPACKE_cpttrf(lapack_int n, float *d, lapack_complex_float *e); +lapack_int LAPACKE_zpttrf(lapack_int n, double *d, lapack_complex_double *e); + +lapack_int LAPACKE_spttrs(int matrix_order, + lapack_int n, + lapack_int nrhs, + const float *d, + const float *e, + float *b, + lapack_int ldb); +lapack_int LAPACKE_dpttrs(int matrix_order, + lapack_int n, + lapack_int nrhs, + const double *d, + const double *e, + double *b, + lapack_int ldb); +lapack_int LAPACKE_cpttrs(int matrix_order, + char uplo, + lapack_int n, + lapack_int nrhs, + const float *d, + const lapack_complex_float *e, + lapack_complex_float *b, + lapack_int ldb); +lapack_int LAPACKE_zpttrs(int matrix_order, + char uplo, + lapack_int n, + lapack_int nrhs, + const double *d, + const lapack_complex_double *e, + lapack_complex_double *b, + lapack_int ldb); + +lapack_int LAPACKE_ssbev(int matrix_order, + char jobz, + char uplo, + lapack_int n, + lapack_int kd, + float *ab, + lapack_int ldab, + float *w, + float *z, + lapack_int ldz); +lapack_int LAPACKE_dsbev(int matrix_order, + char jobz, + char uplo, + lapack_int n, + lapack_int kd, + double *ab, + lapack_int ldab, + double *w, + double *z, + lapack_int ldz); + +lapack_int LAPACKE_ssbevd(int matrix_order, + char jobz, + char uplo, + lapack_int n, + lapack_int kd, + float *ab, + lapack_int ldab, + float *w, + float *z, + lapack_int ldz); +lapack_int LAPACKE_dsbevd(int matrix_order, + char jobz, + char uplo, + lapack_int n, + lapack_int kd, + double *ab, + lapack_int ldab, + double *w, + double *z, + lapack_int ldz); + +lapack_int LAPACKE_ssbevx(int matrix_order, + char jobz, + char range, + char uplo, + lapack_int n, + lapack_int kd, + float *ab, + lapack_int ldab, + float *q, + lapack_int ldq, + float vl, + float vu, + lapack_int il, + lapack_int iu, + float abstol, + lapack_int *m, + float *w, + float *z, + lapack_int ldz, + lapack_int *ifail); +lapack_int LAPACKE_dsbevx(int matrix_order, + char jobz, + char range, + char uplo, + lapack_int n, + lapack_int kd, + double *ab, + lapack_int ldab, + double *q, + lapack_int ldq, + double vl, + double vu, + lapack_int il, + lapack_int iu, + double abstol, + lapack_int *m, + double *w, + double *z, + lapack_int ldz, + lapack_int *ifail); + +lapack_int LAPACKE_ssbgst(int matrix_order, + char vect, + char uplo, + lapack_int n, + lapack_int ka, + lapack_int kb, + float *ab, + lapack_int ldab, + const float *bb, + lapack_int ldbb, + float *x, + lapack_int ldx); +lapack_int LAPACKE_dsbgst(int matrix_order, + char vect, + char uplo, + lapack_int n, + lapack_int ka, + lapack_int kb, + double *ab, + lapack_int ldab, + const double *bb, + lapack_int ldbb, + double *x, + lapack_int ldx); + +lapack_int LAPACKE_ssbgv(int matrix_order, + char jobz, + char uplo, + lapack_int n, + lapack_int ka, + lapack_int kb, + float *ab, + lapack_int ldab, + float *bb, + lapack_int ldbb, + float *w, + float *z, + lapack_int ldz); +lapack_int LAPACKE_dsbgv(int matrix_order, + char jobz, + char uplo, + lapack_int n, + lapack_int ka, + lapack_int kb, + double *ab, + lapack_int ldab, + double *bb, + lapack_int ldbb, + double *w, + double *z, + lapack_int ldz); + +lapack_int LAPACKE_ssbgvd(int matrix_order, + char jobz, + char uplo, + lapack_int n, + lapack_int ka, + lapack_int kb, + float *ab, + lapack_int ldab, + float *bb, + lapack_int ldbb, + float *w, + float *z, + lapack_int ldz); +lapack_int LAPACKE_dsbgvd(int matrix_order, + char jobz, + char uplo, + lapack_int n, + lapack_int ka, + lapack_int kb, + double *ab, + lapack_int ldab, + double *bb, + lapack_int ldbb, + double *w, + double *z, + lapack_int ldz); + +lapack_int LAPACKE_ssbgvx(int matrix_order, + char jobz, + char range, + char uplo, + lapack_int n, + lapack_int ka, + lapack_int kb, + float *ab, + lapack_int ldab, + float *bb, + lapack_int ldbb, + float *q, + lapack_int ldq, + float vl, + float vu, + lapack_int il, + lapack_int iu, + float abstol, + lapack_int *m, + float *w, + float *z, + lapack_int ldz, + lapack_int *ifail); +lapack_int LAPACKE_dsbgvx(int matrix_order, + char jobz, + char range, + char uplo, + lapack_int n, + lapack_int ka, + lapack_int kb, + double *ab, + lapack_int ldab, + double *bb, + lapack_int ldbb, + double *q, + lapack_int ldq, + double vl, + double vu, + lapack_int il, + lapack_int iu, + double abstol, + lapack_int *m, + double *w, + double *z, + lapack_int ldz, + lapack_int *ifail); + +lapack_int LAPACKE_ssbtrd(int matrix_order, + char vect, + char uplo, + lapack_int n, + lapack_int kd, + float *ab, + lapack_int ldab, + float *d, + float *e, + float *q, + lapack_int ldq); +lapack_int LAPACKE_dsbtrd(int matrix_order, + char vect, + char uplo, + lapack_int n, + lapack_int kd, + double *ab, + lapack_int ldab, + double *d, + double *e, + double *q, + lapack_int ldq); + +lapack_int LAPACKE_ssfrk(int matrix_order, + char transr, + char uplo, + char trans, + lapack_int n, + lapack_int k, + float alpha, + const float *a, + lapack_int lda, + float beta, + float *c); +lapack_int LAPACKE_dsfrk(int matrix_order, + char transr, + char uplo, + char trans, + lapack_int n, + lapack_int k, + double alpha, + const double *a, + lapack_int lda, + double beta, + double *c); + +lapack_int LAPACKE_sspcon(int matrix_order, + char uplo, + lapack_int n, + const float *ap, + const lapack_int *ipiv, + float anorm, + float *rcond); +lapack_int LAPACKE_dspcon(int matrix_order, + char uplo, + lapack_int n, + const double *ap, + const lapack_int *ipiv, + double anorm, + double *rcond); +lapack_int LAPACKE_cspcon(int matrix_order, + char uplo, + lapack_int n, + const lapack_complex_float *ap, + const lapack_int *ipiv, + float anorm, + float *rcond); +lapack_int LAPACKE_zspcon(int matrix_order, + char uplo, + lapack_int n, + const lapack_complex_double *ap, + const lapack_int *ipiv, + double anorm, + double *rcond); + +lapack_int + LAPACKE_sspev(int matrix_order, char jobz, char uplo, lapack_int n, float *ap, float *w, float *z, lapack_int ldz); +lapack_int + LAPACKE_dspev(int matrix_order, char jobz, char uplo, lapack_int n, double *ap, double *w, double *z, lapack_int ldz); + +lapack_int + LAPACKE_sspevd(int matrix_order, char jobz, char uplo, lapack_int n, float *ap, float *w, float *z, lapack_int ldz); +lapack_int LAPACKE_dspevd(int matrix_order, + char jobz, + char uplo, + lapack_int n, + double *ap, + double *w, + double *z, + lapack_int ldz); + +lapack_int LAPACKE_sspevx(int matrix_order, + char jobz, + char range, + char uplo, + lapack_int n, + float *ap, + float vl, + float vu, + lapack_int il, + lapack_int iu, + float abstol, + lapack_int *m, + float *w, + float *z, + lapack_int ldz, + lapack_int *ifail); +lapack_int LAPACKE_dspevx(int matrix_order, + char jobz, + char range, + char uplo, + lapack_int n, + double *ap, + double vl, + double vu, + lapack_int il, + lapack_int iu, + double abstol, + lapack_int *m, + double *w, + double *z, + lapack_int ldz, + lapack_int *ifail); + +lapack_int LAPACKE_sspgst(int matrix_order, lapack_int itype, char uplo, lapack_int n, float *ap, const float *bp); +lapack_int LAPACKE_dspgst(int matrix_order, lapack_int itype, char uplo, lapack_int n, double *ap, const double *bp); + +lapack_int LAPACKE_sspgv(int matrix_order, + lapack_int itype, + char jobz, + char uplo, + lapack_int n, + float *ap, + float *bp, + float *w, + float *z, + lapack_int ldz); +lapack_int LAPACKE_dspgv(int matrix_order, + lapack_int itype, + char jobz, + char uplo, + lapack_int n, + double *ap, + double *bp, + double *w, + double *z, + lapack_int ldz); + +lapack_int LAPACKE_sspgvd(int matrix_order, + lapack_int itype, + char jobz, + char uplo, + lapack_int n, + float *ap, + float *bp, + float *w, + float *z, + lapack_int ldz); +lapack_int LAPACKE_dspgvd(int matrix_order, + lapack_int itype, + char jobz, + char uplo, + lapack_int n, + double *ap, + double *bp, + double *w, + double *z, + lapack_int ldz); + +lapack_int LAPACKE_sspgvx(int matrix_order, + lapack_int itype, + char jobz, + char range, + char uplo, + lapack_int n, + float *ap, + float *bp, + float vl, + float vu, + lapack_int il, + lapack_int iu, + float abstol, + lapack_int *m, + float *w, + float *z, + lapack_int ldz, + lapack_int *ifail); +lapack_int LAPACKE_dspgvx(int matrix_order, + lapack_int itype, + char jobz, + char range, + char uplo, + lapack_int n, + double *ap, + double *bp, + double vl, + double vu, + lapack_int il, + lapack_int iu, + double abstol, + lapack_int *m, + double *w, + double *z, + lapack_int ldz, + lapack_int *ifail); + +lapack_int LAPACKE_ssprfs(int matrix_order, + char uplo, + lapack_int n, + lapack_int nrhs, + const float *ap, + const float *afp, + const lapack_int *ipiv, + const float *b, + lapack_int ldb, + float *x, + lapack_int ldx, + float *ferr, + float *berr); +lapack_int LAPACKE_dsprfs(int matrix_order, + char uplo, + lapack_int n, + lapack_int nrhs, + const double *ap, + const double *afp, + const lapack_int *ipiv, + const double *b, + lapack_int ldb, + double *x, + lapack_int ldx, + double *ferr, + double *berr); +lapack_int LAPACKE_csprfs(int matrix_order, + char uplo, + lapack_int n, + lapack_int nrhs, + const lapack_complex_float *ap, + const lapack_complex_float *afp, + const lapack_int *ipiv, + const lapack_complex_float *b, + lapack_int ldb, + lapack_complex_float *x, + lapack_int ldx, + float *ferr, + float *berr); +lapack_int LAPACKE_zsprfs(int matrix_order, + char uplo, + lapack_int n, + lapack_int nrhs, + const lapack_complex_double *ap, + const lapack_complex_double *afp, + const lapack_int *ipiv, + const lapack_complex_double *b, + lapack_int ldb, + lapack_complex_double *x, + lapack_int ldx, + double *ferr, + double *berr); + +lapack_int LAPACKE_sspsv(int matrix_order, + char uplo, + lapack_int n, + lapack_int nrhs, + float *ap, + lapack_int *ipiv, + float *b, + lapack_int ldb); +lapack_int LAPACKE_dspsv(int matrix_order, + char uplo, + lapack_int n, + lapack_int nrhs, + double *ap, + lapack_int *ipiv, + double *b, + lapack_int ldb); +lapack_int LAPACKE_cspsv(int matrix_order, + char uplo, + lapack_int n, + lapack_int nrhs, + lapack_complex_float *ap, + lapack_int *ipiv, + lapack_complex_float *b, + lapack_int ldb); +lapack_int LAPACKE_zspsv(int matrix_order, + char uplo, + lapack_int n, + lapack_int nrhs, + lapack_complex_double *ap, + lapack_int *ipiv, + lapack_complex_double *b, + lapack_int ldb); + +lapack_int LAPACKE_sspsvx(int matrix_order, + char fact, + char uplo, + lapack_int n, + lapack_int nrhs, + const float *ap, + float *afp, + lapack_int *ipiv, + const float *b, + lapack_int ldb, + float *x, + lapack_int ldx, + float *rcond, + float *ferr, + float *berr); +lapack_int LAPACKE_dspsvx(int matrix_order, + char fact, + char uplo, + lapack_int n, + lapack_int nrhs, + const double *ap, + double *afp, + lapack_int *ipiv, + const double *b, + lapack_int ldb, + double *x, + lapack_int ldx, + double *rcond, + double *ferr, + double *berr); +lapack_int LAPACKE_cspsvx(int matrix_order, + char fact, + char uplo, + lapack_int n, + lapack_int nrhs, + const lapack_complex_float *ap, + lapack_complex_float *afp, + lapack_int *ipiv, + const lapack_complex_float *b, + lapack_int ldb, + lapack_complex_float *x, + lapack_int ldx, + float *rcond, + float *ferr, + float *berr); +lapack_int LAPACKE_zspsvx(int matrix_order, + char fact, + char uplo, + lapack_int n, + lapack_int nrhs, + const lapack_complex_double *ap, + lapack_complex_double *afp, + lapack_int *ipiv, + const lapack_complex_double *b, + lapack_int ldb, + lapack_complex_double *x, + lapack_int ldx, + double *rcond, + double *ferr, + double *berr); + +lapack_int LAPACKE_ssptrd(int matrix_order, char uplo, lapack_int n, float *ap, float *d, float *e, float *tau); +lapack_int LAPACKE_dsptrd(int matrix_order, char uplo, lapack_int n, double *ap, double *d, double *e, double *tau); + +lapack_int LAPACKE_ssptrf(int matrix_order, char uplo, lapack_int n, float *ap, lapack_int *ipiv); +lapack_int LAPACKE_dsptrf(int matrix_order, char uplo, lapack_int n, double *ap, lapack_int *ipiv); +lapack_int LAPACKE_csptrf(int matrix_order, char uplo, lapack_int n, lapack_complex_float *ap, lapack_int *ipiv); +lapack_int LAPACKE_zsptrf(int matrix_order, char uplo, lapack_int n, lapack_complex_double *ap, lapack_int *ipiv); + +lapack_int LAPACKE_ssptri(int matrix_order, char uplo, lapack_int n, float *ap, const lapack_int *ipiv); +lapack_int LAPACKE_dsptri(int matrix_order, char uplo, lapack_int n, double *ap, const lapack_int *ipiv); +lapack_int LAPACKE_csptri(int matrix_order, char uplo, lapack_int n, lapack_complex_float *ap, const lapack_int *ipiv); +lapack_int LAPACKE_zsptri(int matrix_order, char uplo, lapack_int n, lapack_complex_double *ap, const lapack_int *ipiv); + +lapack_int LAPACKE_ssptrs(int matrix_order, + char uplo, + lapack_int n, + lapack_int nrhs, + const float *ap, + const lapack_int *ipiv, + float *b, + lapack_int ldb); +lapack_int LAPACKE_dsptrs(int matrix_order, + char uplo, + lapack_int n, + lapack_int nrhs, + const double *ap, + const lapack_int *ipiv, + double *b, + lapack_int ldb); +lapack_int LAPACKE_csptrs(int matrix_order, + char uplo, + lapack_int n, + lapack_int nrhs, + const lapack_complex_float *ap, + const lapack_int *ipiv, + lapack_complex_float *b, + lapack_int ldb); +lapack_int LAPACKE_zsptrs(int matrix_order, + char uplo, + lapack_int n, + lapack_int nrhs, + const lapack_complex_double *ap, + const lapack_int *ipiv, + lapack_complex_double *b, + lapack_int ldb); + +lapack_int LAPACKE_sstebz(char range, + char order, + lapack_int n, + float vl, + float vu, + lapack_int il, + lapack_int iu, + float abstol, + const float *d, + const float *e, + lapack_int *m, + lapack_int *nsplit, + float *w, + lapack_int *iblock, + lapack_int *isplit); +lapack_int LAPACKE_dstebz(char range, + char order, + lapack_int n, + double vl, + double vu, + lapack_int il, + lapack_int iu, + double abstol, + const double *d, + const double *e, + lapack_int *m, + lapack_int *nsplit, + double *w, + lapack_int *iblock, + lapack_int *isplit); + +lapack_int LAPACKE_sstedc(int matrix_order, char compz, lapack_int n, float *d, float *e, float *z, lapack_int ldz); +lapack_int LAPACKE_dstedc(int matrix_order, char compz, lapack_int n, double *d, double *e, double *z, lapack_int ldz); +lapack_int LAPACKE_cstedc(int matrix_order, + char compz, + lapack_int n, + float *d, + float *e, + lapack_complex_float *z, + lapack_int ldz); +lapack_int LAPACKE_zstedc(int matrix_order, + char compz, + lapack_int n, + double *d, + double *e, + lapack_complex_double *z, + lapack_int ldz); + +lapack_int LAPACKE_sstegr(int matrix_order, + char jobz, + char range, + lapack_int n, + float *d, + float *e, + float vl, + float vu, + lapack_int il, + lapack_int iu, + float abstol, + lapack_int *m, + float *w, + float *z, + lapack_int ldz, + lapack_int *isuppz); +lapack_int LAPACKE_dstegr(int matrix_order, + char jobz, + char range, + lapack_int n, + double *d, + double *e, + double vl, + double vu, + lapack_int il, + lapack_int iu, + double abstol, + lapack_int *m, + double *w, + double *z, + lapack_int ldz, + lapack_int *isuppz); +lapack_int LAPACKE_cstegr(int matrix_order, + char jobz, + char range, + lapack_int n, + float *d, + float *e, + float vl, + float vu, + lapack_int il, + lapack_int iu, + float abstol, + lapack_int *m, + float *w, + lapack_complex_float *z, + lapack_int ldz, + lapack_int *isuppz); +lapack_int LAPACKE_zstegr(int matrix_order, + char jobz, + char range, + lapack_int n, + double *d, + double *e, + double vl, + double vu, + lapack_int il, + lapack_int iu, + double abstol, + lapack_int *m, + double *w, + lapack_complex_double *z, + lapack_int ldz, + lapack_int *isuppz); + +lapack_int LAPACKE_sstein(int matrix_order, + lapack_int n, + const float *d, + const float *e, + lapack_int m, + const float *w, + const lapack_int *iblock, + const lapack_int *isplit, + float *z, + lapack_int ldz, + lapack_int *ifailv); +lapack_int LAPACKE_dstein(int matrix_order, + lapack_int n, + const double *d, + const double *e, + lapack_int m, + const double *w, + const lapack_int *iblock, + const lapack_int *isplit, + double *z, + lapack_int ldz, + lapack_int *ifailv); +lapack_int LAPACKE_cstein(int matrix_order, + lapack_int n, + const float *d, + const float *e, + lapack_int m, + const float *w, + const lapack_int *iblock, + const lapack_int *isplit, + lapack_complex_float *z, + lapack_int ldz, + lapack_int *ifailv); +lapack_int LAPACKE_zstein(int matrix_order, + lapack_int n, + const double *d, + const double *e, + lapack_int m, + const double *w, + const lapack_int *iblock, + const lapack_int *isplit, + lapack_complex_double *z, + lapack_int ldz, + lapack_int *ifailv); + +lapack_int LAPACKE_sstemr(int matrix_order, + char jobz, + char range, + lapack_int n, + float *d, + float *e, + float vl, + float vu, + lapack_int il, + lapack_int iu, + lapack_int *m, + float *w, + float *z, + lapack_int ldz, + lapack_int nzc, + lapack_int *isuppz, + lapack_logical *tryrac); +lapack_int LAPACKE_dstemr(int matrix_order, + char jobz, + char range, + lapack_int n, + double *d, + double *e, + double vl, + double vu, + lapack_int il, + lapack_int iu, + lapack_int *m, + double *w, + double *z, + lapack_int ldz, + lapack_int nzc, + lapack_int *isuppz, + lapack_logical *tryrac); +lapack_int LAPACKE_cstemr(int matrix_order, + char jobz, + char range, + lapack_int n, + float *d, + float *e, + float vl, + float vu, + lapack_int il, + lapack_int iu, + lapack_int *m, + float *w, + lapack_complex_float *z, + lapack_int ldz, + lapack_int nzc, + lapack_int *isuppz, + lapack_logical *tryrac); +lapack_int LAPACKE_zstemr(int matrix_order, + char jobz, + char range, + lapack_int n, + double *d, + double *e, + double vl, + double vu, + lapack_int il, + lapack_int iu, + lapack_int *m, + double *w, + lapack_complex_double *z, + lapack_int ldz, + lapack_int nzc, + lapack_int *isuppz, + lapack_logical *tryrac); + +lapack_int LAPACKE_ssteqr(int matrix_order, char compz, lapack_int n, float *d, float *e, float *z, lapack_int ldz); +lapack_int LAPACKE_dsteqr(int matrix_order, char compz, lapack_int n, double *d, double *e, double *z, lapack_int ldz); +lapack_int LAPACKE_csteqr(int matrix_order, + char compz, + lapack_int n, + float *d, + float *e, + lapack_complex_float *z, + lapack_int ldz); +lapack_int LAPACKE_zsteqr(int matrix_order, + char compz, + lapack_int n, + double *d, + double *e, + lapack_complex_double *z, + lapack_int ldz); + +lapack_int LAPACKE_ssterf(lapack_int n, float *d, float *e); +lapack_int LAPACKE_dsterf(lapack_int n, double *d, double *e); + +lapack_int LAPACKE_sstev(int matrix_order, char jobz, lapack_int n, float *d, float *e, float *z, lapack_int ldz); +lapack_int LAPACKE_dstev(int matrix_order, char jobz, lapack_int n, double *d, double *e, double *z, lapack_int ldz); + +lapack_int LAPACKE_sstevd(int matrix_order, char jobz, lapack_int n, float *d, float *e, float *z, lapack_int ldz); +lapack_int LAPACKE_dstevd(int matrix_order, char jobz, lapack_int n, double *d, double *e, double *z, lapack_int ldz); + +lapack_int LAPACKE_sstevr(int matrix_order, + char jobz, + char range, + lapack_int n, + float *d, + float *e, + float vl, + float vu, + lapack_int il, + lapack_int iu, + float abstol, + lapack_int *m, + float *w, + float *z, + lapack_int ldz, + lapack_int *isuppz); +lapack_int LAPACKE_dstevr(int matrix_order, + char jobz, + char range, + lapack_int n, + double *d, + double *e, + double vl, + double vu, + lapack_int il, + lapack_int iu, + double abstol, + lapack_int *m, + double *w, + double *z, + lapack_int ldz, + lapack_int *isuppz); + +lapack_int LAPACKE_sstevx(int matrix_order, + char jobz, + char range, + lapack_int n, + float *d, + float *e, + float vl, + float vu, + lapack_int il, + lapack_int iu, + float abstol, + lapack_int *m, + float *w, + float *z, + lapack_int ldz, + lapack_int *ifail); +lapack_int LAPACKE_dstevx(int matrix_order, + char jobz, + char range, + lapack_int n, + double *d, + double *e, + double vl, + double vu, + lapack_int il, + lapack_int iu, + double abstol, + lapack_int *m, + double *w, + double *z, + lapack_int ldz, + lapack_int *ifail); + +lapack_int LAPACKE_ssycon(int matrix_order, + char uplo, + lapack_int n, + const float *a, + lapack_int lda, + const lapack_int *ipiv, + float anorm, + float *rcond); +lapack_int LAPACKE_dsycon(int matrix_order, + char uplo, + lapack_int n, + const double *a, + lapack_int lda, + const lapack_int *ipiv, + double anorm, + double *rcond); +lapack_int LAPACKE_csycon(int matrix_order, + char uplo, + lapack_int n, + const lapack_complex_float *a, + lapack_int lda, + const lapack_int *ipiv, + float anorm, + float *rcond); +lapack_int LAPACKE_zsycon(int matrix_order, + char uplo, + lapack_int n, + const lapack_complex_double *a, + lapack_int lda, + const lapack_int *ipiv, + double anorm, + double *rcond); + +lapack_int LAPACKE_ssyequb(int matrix_order, + char uplo, + lapack_int n, + const float *a, + lapack_int lda, + float *s, + float *scond, + float *amax); +lapack_int LAPACKE_dsyequb(int matrix_order, + char uplo, + lapack_int n, + const double *a, + lapack_int lda, + double *s, + double *scond, + double *amax); +lapack_int LAPACKE_csyequb(int matrix_order, + char uplo, + lapack_int n, + const lapack_complex_float *a, + lapack_int lda, + float *s, + float *scond, + float *amax); +lapack_int LAPACKE_zsyequb(int matrix_order, + char uplo, + lapack_int n, + const lapack_complex_double *a, + lapack_int lda, + double *s, + double *scond, + double *amax); + +lapack_int LAPACKE_ssyev(int matrix_order, char jobz, char uplo, lapack_int n, float *a, lapack_int lda, float *w); +lapack_int LAPACKE_dsyev(int matrix_order, char jobz, char uplo, lapack_int n, double *a, lapack_int lda, double *w); + +lapack_int LAPACKE_ssyevd(int matrix_order, char jobz, char uplo, lapack_int n, float *a, lapack_int lda, float *w); +lapack_int LAPACKE_dsyevd(int matrix_order, char jobz, char uplo, lapack_int n, double *a, lapack_int lda, double *w); + +lapack_int LAPACKE_ssyevr(int matrix_order, + char jobz, + char range, + char uplo, + lapack_int n, + float *a, + lapack_int lda, + float vl, + float vu, + lapack_int il, + lapack_int iu, + float abstol, + lapack_int *m, + float *w, + float *z, + lapack_int ldz, + lapack_int *isuppz); +lapack_int LAPACKE_dsyevr(int matrix_order, + char jobz, + char range, + char uplo, + lapack_int n, + double *a, + lapack_int lda, + double vl, + double vu, + lapack_int il, + lapack_int iu, + double abstol, + lapack_int *m, + double *w, + double *z, + lapack_int ldz, + lapack_int *isuppz); + +lapack_int LAPACKE_ssyevx(int matrix_order, + char jobz, + char range, + char uplo, + lapack_int n, + float *a, + lapack_int lda, + float vl, + float vu, + lapack_int il, + lapack_int iu, + float abstol, + lapack_int *m, + float *w, + float *z, + lapack_int ldz, + lapack_int *ifail); +lapack_int LAPACKE_dsyevx(int matrix_order, + char jobz, + char range, + char uplo, + lapack_int n, + double *a, + lapack_int lda, + double vl, + double vu, + lapack_int il, + lapack_int iu, + double abstol, + lapack_int *m, + double *w, + double *z, + lapack_int ldz, + lapack_int *ifail); + +lapack_int LAPACKE_ssygst(int matrix_order, + lapack_int itype, + char uplo, + lapack_int n, + float *a, + lapack_int lda, + const float *b, + lapack_int ldb); +lapack_int LAPACKE_dsygst(int matrix_order, + lapack_int itype, + char uplo, + lapack_int n, + double *a, + lapack_int lda, + const double *b, + lapack_int ldb); + +lapack_int LAPACKE_ssygv(int matrix_order, + lapack_int itype, + char jobz, + char uplo, + lapack_int n, + float *a, + lapack_int lda, + float *b, + lapack_int ldb, + float *w); +lapack_int LAPACKE_dsygv(int matrix_order, + lapack_int itype, + char jobz, + char uplo, + lapack_int n, + double *a, + lapack_int lda, + double *b, + lapack_int ldb, + double *w); + +lapack_int LAPACKE_ssygvd(int matrix_order, + lapack_int itype, + char jobz, + char uplo, + lapack_int n, + float *a, + lapack_int lda, + float *b, + lapack_int ldb, + float *w); +lapack_int LAPACKE_dsygvd(int matrix_order, + lapack_int itype, + char jobz, + char uplo, + lapack_int n, + double *a, + lapack_int lda, + double *b, + lapack_int ldb, + double *w); + +lapack_int LAPACKE_ssygvx(int matrix_order, + lapack_int itype, + char jobz, + char range, + char uplo, + lapack_int n, + float *a, + lapack_int lda, + float *b, + lapack_int ldb, + float vl, + float vu, + lapack_int il, + lapack_int iu, + float abstol, + lapack_int *m, + float *w, + float *z, + lapack_int ldz, + lapack_int *ifail); +lapack_int LAPACKE_dsygvx(int matrix_order, + lapack_int itype, + char jobz, + char range, + char uplo, + lapack_int n, + double *a, + lapack_int lda, + double *b, + lapack_int ldb, + double vl, + double vu, + lapack_int il, + lapack_int iu, + double abstol, + lapack_int *m, + double *w, + double *z, + lapack_int ldz, + lapack_int *ifail); + +lapack_int LAPACKE_ssyrfs(int matrix_order, + char uplo, + lapack_int n, + lapack_int nrhs, + const float *a, + lapack_int lda, + const float *af, + lapack_int ldaf, + const lapack_int *ipiv, + const float *b, + lapack_int ldb, + float *x, + lapack_int ldx, + float *ferr, + float *berr); +lapack_int LAPACKE_dsyrfs(int matrix_order, + char uplo, + lapack_int n, + lapack_int nrhs, + const double *a, + lapack_int lda, + const double *af, + lapack_int ldaf, + const lapack_int *ipiv, + const double *b, + lapack_int ldb, + double *x, + lapack_int ldx, + double *ferr, + double *berr); +lapack_int LAPACKE_csyrfs(int matrix_order, + char uplo, + lapack_int n, + lapack_int nrhs, + const lapack_complex_float *a, + lapack_int lda, + const lapack_complex_float *af, + lapack_int ldaf, + const lapack_int *ipiv, + const lapack_complex_float *b, + lapack_int ldb, + lapack_complex_float *x, + lapack_int ldx, + float *ferr, + float *berr); +lapack_int LAPACKE_zsyrfs(int matrix_order, + char uplo, + lapack_int n, + lapack_int nrhs, + const lapack_complex_double *a, + lapack_int lda, + const lapack_complex_double *af, + lapack_int ldaf, + const lapack_int *ipiv, + const lapack_complex_double *b, + lapack_int ldb, + lapack_complex_double *x, + lapack_int ldx, + double *ferr, + double *berr); + +lapack_int LAPACKE_ssyrfsx(int matrix_order, + char uplo, + char equed, + lapack_int n, + lapack_int nrhs, + const float *a, + lapack_int lda, + const float *af, + lapack_int ldaf, + const lapack_int *ipiv, + const float *s, + const float *b, + lapack_int ldb, + float *x, + lapack_int ldx, + float *rcond, + float *berr, + lapack_int n_err_bnds, + float *err_bnds_norm, + float *err_bnds_comp, + lapack_int nparams, + float *params); +lapack_int LAPACKE_dsyrfsx(int matrix_order, + char uplo, + char equed, + lapack_int n, + lapack_int nrhs, + const double *a, + lapack_int lda, + const double *af, + lapack_int ldaf, + const lapack_int *ipiv, + const double *s, + const double *b, + lapack_int ldb, + double *x, + lapack_int ldx, + double *rcond, + double *berr, + lapack_int n_err_bnds, + double *err_bnds_norm, + double *err_bnds_comp, + lapack_int nparams, + double *params); +lapack_int LAPACKE_csyrfsx(int matrix_order, + char uplo, + char equed, + lapack_int n, + lapack_int nrhs, + const lapack_complex_float *a, + lapack_int lda, + const lapack_complex_float *af, + lapack_int ldaf, + const lapack_int *ipiv, + const float *s, + const lapack_complex_float *b, + lapack_int ldb, + lapack_complex_float *x, + lapack_int ldx, + float *rcond, + float *berr, + lapack_int n_err_bnds, + float *err_bnds_norm, + float *err_bnds_comp, + lapack_int nparams, + float *params); +lapack_int LAPACKE_zsyrfsx(int matrix_order, + char uplo, + char equed, + lapack_int n, + lapack_int nrhs, + const lapack_complex_double *a, + lapack_int lda, + const lapack_complex_double *af, + lapack_int ldaf, + const lapack_int *ipiv, + const double *s, + const lapack_complex_double *b, + lapack_int ldb, + lapack_complex_double *x, + lapack_int ldx, + double *rcond, + double *berr, + lapack_int n_err_bnds, + double *err_bnds_norm, + double *err_bnds_comp, + lapack_int nparams, + double *params); + +lapack_int LAPACKE_ssysv(int matrix_order, + char uplo, + lapack_int n, + lapack_int nrhs, + float *a, + lapack_int lda, + lapack_int *ipiv, + float *b, + lapack_int ldb); +lapack_int LAPACKE_dsysv(int matrix_order, + char uplo, + lapack_int n, + lapack_int nrhs, + double *a, + lapack_int lda, + lapack_int *ipiv, + double *b, + lapack_int ldb); +lapack_int LAPACKE_csysv(int matrix_order, + char uplo, + lapack_int n, + lapack_int nrhs, + lapack_complex_float *a, + lapack_int lda, + lapack_int *ipiv, + lapack_complex_float *b, + lapack_int ldb); +lapack_int LAPACKE_zsysv(int matrix_order, + char uplo, + lapack_int n, + lapack_int nrhs, + lapack_complex_double *a, + lapack_int lda, + lapack_int *ipiv, + lapack_complex_double *b, + lapack_int ldb); + +lapack_int LAPACKE_ssysvx(int matrix_order, + char fact, + char uplo, + lapack_int n, + lapack_int nrhs, + const float *a, + lapack_int lda, + float *af, + lapack_int ldaf, + lapack_int *ipiv, + const float *b, + lapack_int ldb, + float *x, + lapack_int ldx, + float *rcond, + float *ferr, + float *berr); +lapack_int LAPACKE_dsysvx(int matrix_order, + char fact, + char uplo, + lapack_int n, + lapack_int nrhs, + const double *a, + lapack_int lda, + double *af, + lapack_int ldaf, + lapack_int *ipiv, + const double *b, + lapack_int ldb, + double *x, + lapack_int ldx, + double *rcond, + double *ferr, + double *berr); +lapack_int LAPACKE_csysvx(int matrix_order, + char fact, + char uplo, + lapack_int n, + lapack_int nrhs, + const lapack_complex_float *a, + lapack_int lda, + lapack_complex_float *af, + lapack_int ldaf, + lapack_int *ipiv, + const lapack_complex_float *b, + lapack_int ldb, + lapack_complex_float *x, + lapack_int ldx, + float *rcond, + float *ferr, + float *berr); +lapack_int LAPACKE_zsysvx(int matrix_order, + char fact, + char uplo, + lapack_int n, + lapack_int nrhs, + const lapack_complex_double *a, + lapack_int lda, + lapack_complex_double *af, + lapack_int ldaf, + lapack_int *ipiv, + const lapack_complex_double *b, + lapack_int ldb, + lapack_complex_double *x, + lapack_int ldx, + double *rcond, + double *ferr, + double *berr); + +lapack_int LAPACKE_ssysvxx(int matrix_order, + char fact, + char uplo, + lapack_int n, + lapack_int nrhs, + float *a, + lapack_int lda, + float *af, + lapack_int ldaf, + lapack_int *ipiv, + char *equed, + float *s, + float *b, + lapack_int ldb, + float *x, + lapack_int ldx, + float *rcond, + float *rpvgrw, + float *berr, + lapack_int n_err_bnds, + float *err_bnds_norm, + float *err_bnds_comp, + lapack_int nparams, + float *params); +lapack_int LAPACKE_dsysvxx(int matrix_order, + char fact, + char uplo, + lapack_int n, + lapack_int nrhs, + double *a, + lapack_int lda, + double *af, + lapack_int ldaf, + lapack_int *ipiv, + char *equed, + double *s, + double *b, + lapack_int ldb, + double *x, + lapack_int ldx, + double *rcond, + double *rpvgrw, + double *berr, + lapack_int n_err_bnds, + double *err_bnds_norm, + double *err_bnds_comp, + lapack_int nparams, + double *params); +lapack_int LAPACKE_csysvxx(int matrix_order, + char fact, + char uplo, + lapack_int n, + lapack_int nrhs, + lapack_complex_float *a, + lapack_int lda, + lapack_complex_float *af, + lapack_int ldaf, + lapack_int *ipiv, + char *equed, + float *s, + lapack_complex_float *b, + lapack_int ldb, + lapack_complex_float *x, + lapack_int ldx, + float *rcond, + float *rpvgrw, + float *berr, + lapack_int n_err_bnds, + float *err_bnds_norm, + float *err_bnds_comp, + lapack_int nparams, + float *params); +lapack_int LAPACKE_zsysvxx(int matrix_order, + char fact, + char uplo, + lapack_int n, + lapack_int nrhs, + lapack_complex_double *a, + lapack_int lda, + lapack_complex_double *af, + lapack_int ldaf, + lapack_int *ipiv, + char *equed, + double *s, + lapack_complex_double *b, + lapack_int ldb, + lapack_complex_double *x, + lapack_int ldx, + double *rcond, + double *rpvgrw, + double *berr, + lapack_int n_err_bnds, + double *err_bnds_norm, + double *err_bnds_comp, + lapack_int nparams, + double *params); + +lapack_int + LAPACKE_ssytrd(int matrix_order, char uplo, lapack_int n, float *a, lapack_int lda, float *d, float *e, float *tau); +lapack_int LAPACKE_dsytrd(int matrix_order, + char uplo, + lapack_int n, + double *a, + lapack_int lda, + double *d, + double *e, + double *tau); + +lapack_int LAPACKE_ssytrf(int matrix_order, char uplo, lapack_int n, float *a, lapack_int lda, lapack_int *ipiv); +lapack_int LAPACKE_dsytrf(int matrix_order, char uplo, lapack_int n, double *a, lapack_int lda, lapack_int *ipiv); +lapack_int + LAPACKE_csytrf(int matrix_order, char uplo, lapack_int n, lapack_complex_float *a, lapack_int lda, lapack_int *ipiv); +lapack_int + LAPACKE_zsytrf(int matrix_order, char uplo, lapack_int n, lapack_complex_double *a, lapack_int lda, lapack_int *ipiv); + +lapack_int LAPACKE_ssytri(int matrix_order, char uplo, lapack_int n, float *a, lapack_int lda, const lapack_int *ipiv); +lapack_int LAPACKE_dsytri(int matrix_order, char uplo, lapack_int n, double *a, lapack_int lda, const lapack_int *ipiv); +lapack_int LAPACKE_csytri(int matrix_order, + char uplo, + lapack_int n, + lapack_complex_float *a, + lapack_int lda, + const lapack_int *ipiv); +lapack_int LAPACKE_zsytri(int matrix_order, + char uplo, + lapack_int n, + lapack_complex_double *a, + lapack_int lda, + const lapack_int *ipiv); + +lapack_int LAPACKE_ssytrs(int matrix_order, + char uplo, + lapack_int n, + lapack_int nrhs, + const float *a, + lapack_int lda, + const lapack_int *ipiv, + float *b, + lapack_int ldb); +lapack_int LAPACKE_dsytrs(int matrix_order, + char uplo, + lapack_int n, + lapack_int nrhs, + const double *a, + lapack_int lda, + const lapack_int *ipiv, + double *b, + lapack_int ldb); +lapack_int LAPACKE_csytrs(int matrix_order, + char uplo, + lapack_int n, + lapack_int nrhs, + const lapack_complex_float *a, + lapack_int lda, + const lapack_int *ipiv, + lapack_complex_float *b, + lapack_int ldb); +lapack_int LAPACKE_zsytrs(int matrix_order, + char uplo, + lapack_int n, + lapack_int nrhs, + const lapack_complex_double *a, + lapack_int lda, + const lapack_int *ipiv, + lapack_complex_double *b, + lapack_int ldb); + +lapack_int LAPACKE_stbcon(int matrix_order, + char norm, + char uplo, + char diag, + lapack_int n, + lapack_int kd, + const float *ab, + lapack_int ldab, + float *rcond); +lapack_int LAPACKE_dtbcon(int matrix_order, + char norm, + char uplo, + char diag, + lapack_int n, + lapack_int kd, + const double *ab, + lapack_int ldab, + double *rcond); +lapack_int LAPACKE_ctbcon(int matrix_order, + char norm, + char uplo, + char diag, + lapack_int n, + lapack_int kd, + const lapack_complex_float *ab, + lapack_int ldab, + float *rcond); +lapack_int LAPACKE_ztbcon(int matrix_order, + char norm, + char uplo, + char diag, + lapack_int n, + lapack_int kd, + const lapack_complex_double *ab, + lapack_int ldab, + double *rcond); + +lapack_int LAPACKE_stbrfs(int matrix_order, + char uplo, + char trans, + char diag, + lapack_int n, + lapack_int kd, + lapack_int nrhs, + const float *ab, + lapack_int ldab, + const float *b, + lapack_int ldb, + const float *x, + lapack_int ldx, + float *ferr, + float *berr); +lapack_int LAPACKE_dtbrfs(int matrix_order, + char uplo, + char trans, + char diag, + lapack_int n, + lapack_int kd, + lapack_int nrhs, + const double *ab, + lapack_int ldab, + const double *b, + lapack_int ldb, + const double *x, + lapack_int ldx, + double *ferr, + double *berr); +lapack_int LAPACKE_ctbrfs(int matrix_order, + char uplo, + char trans, + char diag, + lapack_int n, + lapack_int kd, + lapack_int nrhs, + const lapack_complex_float *ab, + lapack_int ldab, + const lapack_complex_float *b, + lapack_int ldb, + const lapack_complex_float *x, + lapack_int ldx, + float *ferr, + float *berr); +lapack_int LAPACKE_ztbrfs(int matrix_order, + char uplo, + char trans, + char diag, + lapack_int n, + lapack_int kd, + lapack_int nrhs, + const lapack_complex_double *ab, + lapack_int ldab, + const lapack_complex_double *b, + lapack_int ldb, + const lapack_complex_double *x, + lapack_int ldx, + double *ferr, + double *berr); + +lapack_int LAPACKE_stbtrs(int matrix_order, + char uplo, + char trans, + char diag, + lapack_int n, + lapack_int kd, + lapack_int nrhs, + const float *ab, + lapack_int ldab, + float *b, + lapack_int ldb); +lapack_int LAPACKE_dtbtrs(int matrix_order, + char uplo, + char trans, + char diag, + lapack_int n, + lapack_int kd, + lapack_int nrhs, + const double *ab, + lapack_int ldab, + double *b, + lapack_int ldb); +lapack_int LAPACKE_ctbtrs(int matrix_order, + char uplo, + char trans, + char diag, + lapack_int n, + lapack_int kd, + lapack_int nrhs, + const lapack_complex_float *ab, + lapack_int ldab, + lapack_complex_float *b, + lapack_int ldb); +lapack_int LAPACKE_ztbtrs(int matrix_order, + char uplo, + char trans, + char diag, + lapack_int n, + lapack_int kd, + lapack_int nrhs, + const lapack_complex_double *ab, + lapack_int ldab, + lapack_complex_double *b, + lapack_int ldb); + +lapack_int LAPACKE_stfsm(int matrix_order, + char transr, + char side, + char uplo, + char trans, + char diag, + lapack_int m, + lapack_int n, + float alpha, + const float *a, + float *b, + lapack_int ldb); +lapack_int LAPACKE_dtfsm(int matrix_order, + char transr, + char side, + char uplo, + char trans, + char diag, + lapack_int m, + lapack_int n, + double alpha, + const double *a, + double *b, + lapack_int ldb); +lapack_int LAPACKE_ctfsm(int matrix_order, + char transr, + char side, + char uplo, + char trans, + char diag, + lapack_int m, + lapack_int n, + lapack_complex_float alpha, + const lapack_complex_float *a, + lapack_complex_float *b, + lapack_int ldb); +lapack_int LAPACKE_ztfsm(int matrix_order, + char transr, + char side, + char uplo, + char trans, + char diag, + lapack_int m, + lapack_int n, + lapack_complex_double alpha, + const lapack_complex_double *a, + lapack_complex_double *b, + lapack_int ldb); + +lapack_int LAPACKE_stftri(int matrix_order, char transr, char uplo, char diag, lapack_int n, float *a); +lapack_int LAPACKE_dtftri(int matrix_order, char transr, char uplo, char diag, lapack_int n, double *a); +lapack_int LAPACKE_ctftri(int matrix_order, char transr, char uplo, char diag, lapack_int n, lapack_complex_float *a); +lapack_int LAPACKE_ztftri(int matrix_order, char transr, char uplo, char diag, lapack_int n, lapack_complex_double *a); + +lapack_int LAPACKE_stfttp(int matrix_order, char transr, char uplo, lapack_int n, const float *arf, float *ap); +lapack_int LAPACKE_dtfttp(int matrix_order, char transr, char uplo, lapack_int n, const double *arf, double *ap); +lapack_int LAPACKE_ctfttp(int matrix_order, + char transr, + char uplo, + lapack_int n, + const lapack_complex_float *arf, + lapack_complex_float *ap); +lapack_int LAPACKE_ztfttp(int matrix_order, + char transr, + char uplo, + lapack_int n, + const lapack_complex_double *arf, + lapack_complex_double *ap); + +lapack_int + LAPACKE_stfttr(int matrix_order, char transr, char uplo, lapack_int n, const float *arf, float *a, lapack_int lda); +lapack_int + LAPACKE_dtfttr(int matrix_order, char transr, char uplo, lapack_int n, const double *arf, double *a, lapack_int lda); +lapack_int LAPACKE_ctfttr(int matrix_order, + char transr, + char uplo, + lapack_int n, + const lapack_complex_float *arf, + lapack_complex_float *a, + lapack_int lda); +lapack_int LAPACKE_ztfttr(int matrix_order, + char transr, + char uplo, + lapack_int n, + const lapack_complex_double *arf, + lapack_complex_double *a, + lapack_int lda); + +lapack_int LAPACKE_stgevc(int matrix_order, + char side, + char howmny, + const lapack_logical *select, + lapack_int n, + const float *s, + lapack_int lds, + const float *p, + lapack_int ldp, + float *vl, + lapack_int ldvl, + float *vr, + lapack_int ldvr, + lapack_int mm, + lapack_int *m); +lapack_int LAPACKE_dtgevc(int matrix_order, + char side, + char howmny, + const lapack_logical *select, + lapack_int n, + const double *s, + lapack_int lds, + const double *p, + lapack_int ldp, + double *vl, + lapack_int ldvl, + double *vr, + lapack_int ldvr, + lapack_int mm, + lapack_int *m); +lapack_int LAPACKE_ctgevc(int matrix_order, + char side, + char howmny, + const lapack_logical *select, + lapack_int n, + const lapack_complex_float *s, + lapack_int lds, + const lapack_complex_float *p, + lapack_int ldp, + lapack_complex_float *vl, + lapack_int ldvl, + lapack_complex_float *vr, + lapack_int ldvr, + lapack_int mm, + lapack_int *m); +lapack_int LAPACKE_ztgevc(int matrix_order, + char side, + char howmny, + const lapack_logical *select, + lapack_int n, + const lapack_complex_double *s, + lapack_int lds, + const lapack_complex_double *p, + lapack_int ldp, + lapack_complex_double *vl, + lapack_int ldvl, + lapack_complex_double *vr, + lapack_int ldvr, + lapack_int mm, + lapack_int *m); + +lapack_int LAPACKE_stgexc(int matrix_order, + lapack_logical wantq, + lapack_logical wantz, + lapack_int n, + float *a, + lapack_int lda, + float *b, + lapack_int ldb, + float *q, + lapack_int ldq, + float *z, + lapack_int ldz, + lapack_int *ifst, + lapack_int *ilst); +lapack_int LAPACKE_dtgexc(int matrix_order, + lapack_logical wantq, + lapack_logical wantz, + lapack_int n, + double *a, + lapack_int lda, + double *b, + lapack_int ldb, + double *q, + lapack_int ldq, + double *z, + lapack_int ldz, + lapack_int *ifst, + lapack_int *ilst); +lapack_int LAPACKE_ctgexc(int matrix_order, + lapack_logical wantq, + lapack_logical wantz, + lapack_int n, + lapack_complex_float *a, + lapack_int lda, + lapack_complex_float *b, + lapack_int ldb, + lapack_complex_float *q, + lapack_int ldq, + lapack_complex_float *z, + lapack_int ldz, + lapack_int ifst, + lapack_int ilst); +lapack_int LAPACKE_ztgexc(int matrix_order, + lapack_logical wantq, + lapack_logical wantz, + lapack_int n, + lapack_complex_double *a, + lapack_int lda, + lapack_complex_double *b, + lapack_int ldb, + lapack_complex_double *q, + lapack_int ldq, + lapack_complex_double *z, + lapack_int ldz, + lapack_int ifst, + lapack_int ilst); + +lapack_int LAPACKE_stgsen(int matrix_order, + lapack_int ijob, + lapack_logical wantq, + lapack_logical wantz, + const lapack_logical *select, + lapack_int n, + float *a, + lapack_int lda, + float *b, + lapack_int ldb, + float *alphar, + float *alphai, + float *beta, + float *q, + lapack_int ldq, + float *z, + lapack_int ldz, + lapack_int *m, + float *pl, + float *pr, + float *dif); +lapack_int LAPACKE_dtgsen(int matrix_order, + lapack_int ijob, + lapack_logical wantq, + lapack_logical wantz, + const lapack_logical *select, + lapack_int n, + double *a, + lapack_int lda, + double *b, + lapack_int ldb, + double *alphar, + double *alphai, + double *beta, + double *q, + lapack_int ldq, + double *z, + lapack_int ldz, + lapack_int *m, + double *pl, + double *pr, + double *dif); +lapack_int LAPACKE_ctgsen(int matrix_order, + lapack_int ijob, + lapack_logical wantq, + lapack_logical wantz, + const lapack_logical *select, + lapack_int n, + lapack_complex_float *a, + lapack_int lda, + lapack_complex_float *b, + lapack_int ldb, + lapack_complex_float *alpha, + lapack_complex_float *beta, + lapack_complex_float *q, + lapack_int ldq, + lapack_complex_float *z, + lapack_int ldz, + lapack_int *m, + float *pl, + float *pr, + float *dif); +lapack_int LAPACKE_ztgsen(int matrix_order, + lapack_int ijob, + lapack_logical wantq, + lapack_logical wantz, + const lapack_logical *select, + lapack_int n, + lapack_complex_double *a, + lapack_int lda, + lapack_complex_double *b, + lapack_int ldb, + lapack_complex_double *alpha, + lapack_complex_double *beta, + lapack_complex_double *q, + lapack_int ldq, + lapack_complex_double *z, + lapack_int ldz, + lapack_int *m, + double *pl, + double *pr, + double *dif); + +lapack_int LAPACKE_stgsja(int matrix_order, + char jobu, + char jobv, + char jobq, + lapack_int m, + lapack_int p, + lapack_int n, + lapack_int k, + lapack_int l, + float *a, + lapack_int lda, + float *b, + lapack_int ldb, + float tola, + float tolb, + float *alpha, + float *beta, + float *u, + lapack_int ldu, + float *v, + lapack_int ldv, + float *q, + lapack_int ldq, + lapack_int *ncycle); +lapack_int LAPACKE_dtgsja(int matrix_order, + char jobu, + char jobv, + char jobq, + lapack_int m, + lapack_int p, + lapack_int n, + lapack_int k, + lapack_int l, + double *a, + lapack_int lda, + double *b, + lapack_int ldb, + double tola, + double tolb, + double *alpha, + double *beta, + double *u, + lapack_int ldu, + double *v, + lapack_int ldv, + double *q, + lapack_int ldq, + lapack_int *ncycle); +lapack_int LAPACKE_ctgsja(int matrix_order, + char jobu, + char jobv, + char jobq, + lapack_int m, + lapack_int p, + lapack_int n, + lapack_int k, + lapack_int l, + lapack_complex_float *a, + lapack_int lda, + lapack_complex_float *b, + lapack_int ldb, + float tola, + float tolb, + float *alpha, + float *beta, + lapack_complex_float *u, + lapack_int ldu, + lapack_complex_float *v, + lapack_int ldv, + lapack_complex_float *q, + lapack_int ldq, + lapack_int *ncycle); +lapack_int LAPACKE_ztgsja(int matrix_order, + char jobu, + char jobv, + char jobq, + lapack_int m, + lapack_int p, + lapack_int n, + lapack_int k, + lapack_int l, + lapack_complex_double *a, + lapack_int lda, + lapack_complex_double *b, + lapack_int ldb, + double tola, + double tolb, + double *alpha, + double *beta, + lapack_complex_double *u, + lapack_int ldu, + lapack_complex_double *v, + lapack_int ldv, + lapack_complex_double *q, + lapack_int ldq, + lapack_int *ncycle); + +lapack_int LAPACKE_stgsna(int matrix_order, + char job, + char howmny, + const lapack_logical *select, + lapack_int n, + const float *a, + lapack_int lda, + const float *b, + lapack_int ldb, + const float *vl, + lapack_int ldvl, + const float *vr, + lapack_int ldvr, + float *s, + float *dif, + lapack_int mm, + lapack_int *m); +lapack_int LAPACKE_dtgsna(int matrix_order, + char job, + char howmny, + const lapack_logical *select, + lapack_int n, + const double *a, + lapack_int lda, + const double *b, + lapack_int ldb, + const double *vl, + lapack_int ldvl, + const double *vr, + lapack_int ldvr, + double *s, + double *dif, + lapack_int mm, + lapack_int *m); +lapack_int LAPACKE_ctgsna(int matrix_order, + char job, + char howmny, + const lapack_logical *select, + lapack_int n, + const lapack_complex_float *a, + lapack_int lda, + const lapack_complex_float *b, + lapack_int ldb, + const lapack_complex_float *vl, + lapack_int ldvl, + const lapack_complex_float *vr, + lapack_int ldvr, + float *s, + float *dif, + lapack_int mm, + lapack_int *m); +lapack_int LAPACKE_ztgsna(int matrix_order, + char job, + char howmny, + const lapack_logical *select, + lapack_int n, + const lapack_complex_double *a, + lapack_int lda, + const lapack_complex_double *b, + lapack_int ldb, + const lapack_complex_double *vl, + lapack_int ldvl, + const lapack_complex_double *vr, + lapack_int ldvr, + double *s, + double *dif, + lapack_int mm, + lapack_int *m); + +lapack_int LAPACKE_stgsyl(int matrix_order, + char trans, + lapack_int ijob, + lapack_int m, + lapack_int n, + const float *a, + lapack_int lda, + const float *b, + lapack_int ldb, + float *c, + lapack_int ldc, + const float *d, + lapack_int ldd, + const float *e, + lapack_int lde, + float *f, + lapack_int ldf, + float *scale, + float *dif); +lapack_int LAPACKE_dtgsyl(int matrix_order, + char trans, + lapack_int ijob, + lapack_int m, + lapack_int n, + const double *a, + lapack_int lda, + const double *b, + lapack_int ldb, + double *c, + lapack_int ldc, + const double *d, + lapack_int ldd, + const double *e, + lapack_int lde, + double *f, + lapack_int ldf, + double *scale, + double *dif); +lapack_int LAPACKE_ctgsyl(int matrix_order, + char trans, + lapack_int ijob, + lapack_int m, + lapack_int n, + const lapack_complex_float *a, + lapack_int lda, + const lapack_complex_float *b, + lapack_int ldb, + lapack_complex_float *c, + lapack_int ldc, + const lapack_complex_float *d, + lapack_int ldd, + const lapack_complex_float *e, + lapack_int lde, + lapack_complex_float *f, + lapack_int ldf, + float *scale, + float *dif); +lapack_int LAPACKE_ztgsyl(int matrix_order, + char trans, + lapack_int ijob, + lapack_int m, + lapack_int n, + const lapack_complex_double *a, + lapack_int lda, + const lapack_complex_double *b, + lapack_int ldb, + lapack_complex_double *c, + lapack_int ldc, + const lapack_complex_double *d, + lapack_int ldd, + const lapack_complex_double *e, + lapack_int lde, + lapack_complex_double *f, + lapack_int ldf, + double *scale, + double *dif); + +lapack_int + LAPACKE_stpcon(int matrix_order, char norm, char uplo, char diag, lapack_int n, const float *ap, float *rcond); +lapack_int + LAPACKE_dtpcon(int matrix_order, char norm, char uplo, char diag, lapack_int n, const double *ap, double *rcond); +lapack_int LAPACKE_ctpcon(int matrix_order, + char norm, + char uplo, + char diag, + lapack_int n, + const lapack_complex_float *ap, + float *rcond); +lapack_int LAPACKE_ztpcon(int matrix_order, + char norm, + char uplo, + char diag, + lapack_int n, + const lapack_complex_double *ap, + double *rcond); + +lapack_int LAPACKE_stprfs(int matrix_order, + char uplo, + char trans, + char diag, + lapack_int n, + lapack_int nrhs, + const float *ap, + const float *b, + lapack_int ldb, + const float *x, + lapack_int ldx, + float *ferr, + float *berr); +lapack_int LAPACKE_dtprfs(int matrix_order, + char uplo, + char trans, + char diag, + lapack_int n, + lapack_int nrhs, + const double *ap, + const double *b, + lapack_int ldb, + const double *x, + lapack_int ldx, + double *ferr, + double *berr); +lapack_int LAPACKE_ctprfs(int matrix_order, + char uplo, + char trans, + char diag, + lapack_int n, + lapack_int nrhs, + const lapack_complex_float *ap, + const lapack_complex_float *b, + lapack_int ldb, + const lapack_complex_float *x, + lapack_int ldx, + float *ferr, + float *berr); +lapack_int LAPACKE_ztprfs(int matrix_order, + char uplo, + char trans, + char diag, + lapack_int n, + lapack_int nrhs, + const lapack_complex_double *ap, + const lapack_complex_double *b, + lapack_int ldb, + const lapack_complex_double *x, + lapack_int ldx, + double *ferr, + double *berr); + +lapack_int LAPACKE_stptri(int matrix_order, char uplo, char diag, lapack_int n, float *ap); +lapack_int LAPACKE_dtptri(int matrix_order, char uplo, char diag, lapack_int n, double *ap); +lapack_int LAPACKE_ctptri(int matrix_order, char uplo, char diag, lapack_int n, lapack_complex_float *ap); +lapack_int LAPACKE_ztptri(int matrix_order, char uplo, char diag, lapack_int n, lapack_complex_double *ap); + +lapack_int LAPACKE_stptrs(int matrix_order, + char uplo, + char trans, + char diag, + lapack_int n, + lapack_int nrhs, + const float *ap, + float *b, + lapack_int ldb); +lapack_int LAPACKE_dtptrs(int matrix_order, + char uplo, + char trans, + char diag, + lapack_int n, + lapack_int nrhs, + const double *ap, + double *b, + lapack_int ldb); +lapack_int LAPACKE_ctptrs(int matrix_order, + char uplo, + char trans, + char diag, + lapack_int n, + lapack_int nrhs, + const lapack_complex_float *ap, + lapack_complex_float *b, + lapack_int ldb); +lapack_int LAPACKE_ztptrs(int matrix_order, + char uplo, + char trans, + char diag, + lapack_int n, + lapack_int nrhs, + const lapack_complex_double *ap, + lapack_complex_double *b, + lapack_int ldb); + +lapack_int LAPACKE_stpttf(int matrix_order, char transr, char uplo, lapack_int n, const float *ap, float *arf); +lapack_int LAPACKE_dtpttf(int matrix_order, char transr, char uplo, lapack_int n, const double *ap, double *arf); +lapack_int LAPACKE_ctpttf(int matrix_order, + char transr, + char uplo, + lapack_int n, + const lapack_complex_float *ap, + lapack_complex_float *arf); +lapack_int LAPACKE_ztpttf(int matrix_order, + char transr, + char uplo, + lapack_int n, + const lapack_complex_double *ap, + lapack_complex_double *arf); + +lapack_int LAPACKE_stpttr(int matrix_order, char uplo, lapack_int n, const float *ap, float *a, lapack_int lda); +lapack_int LAPACKE_dtpttr(int matrix_order, char uplo, lapack_int n, const double *ap, double *a, lapack_int lda); +lapack_int LAPACKE_ctpttr(int matrix_order, + char uplo, + lapack_int n, + const lapack_complex_float *ap, + lapack_complex_float *a, + lapack_int lda); +lapack_int LAPACKE_ztpttr(int matrix_order, + char uplo, + lapack_int n, + const lapack_complex_double *ap, + lapack_complex_double *a, + lapack_int lda); + +lapack_int LAPACKE_strcon(int matrix_order, + char norm, + char uplo, + char diag, + lapack_int n, + const float *a, + lapack_int lda, + float *rcond); +lapack_int LAPACKE_dtrcon(int matrix_order, + char norm, + char uplo, + char diag, + lapack_int n, + const double *a, + lapack_int lda, + double *rcond); +lapack_int LAPACKE_ctrcon(int matrix_order, + char norm, + char uplo, + char diag, + lapack_int n, + const lapack_complex_float *a, + lapack_int lda, + float *rcond); +lapack_int LAPACKE_ztrcon(int matrix_order, + char norm, + char uplo, + char diag, + lapack_int n, + const lapack_complex_double *a, + lapack_int lda, + double *rcond); + +lapack_int LAPACKE_strevc(int matrix_order, + char side, + char howmny, + lapack_logical *select, + lapack_int n, + const float *t, + lapack_int ldt, + float *vl, + lapack_int ldvl, + float *vr, + lapack_int ldvr, + lapack_int mm, + lapack_int *m); +lapack_int LAPACKE_dtrevc(int matrix_order, + char side, + char howmny, + lapack_logical *select, + lapack_int n, + const double *t, + lapack_int ldt, + double *vl, + lapack_int ldvl, + double *vr, + lapack_int ldvr, + lapack_int mm, + lapack_int *m); +lapack_int LAPACKE_ctrevc(int matrix_order, + char side, + char howmny, + const lapack_logical *select, + lapack_int n, + lapack_complex_float *t, + lapack_int ldt, + lapack_complex_float *vl, + lapack_int ldvl, + lapack_complex_float *vr, + lapack_int ldvr, + lapack_int mm, + lapack_int *m); +lapack_int LAPACKE_ztrevc(int matrix_order, + char side, + char howmny, + const lapack_logical *select, + lapack_int n, + lapack_complex_double *t, + lapack_int ldt, + lapack_complex_double *vl, + lapack_int ldvl, + lapack_complex_double *vr, + lapack_int ldvr, + lapack_int mm, + lapack_int *m); + +lapack_int LAPACKE_strexc(int matrix_order, + char compq, + lapack_int n, + float *t, + lapack_int ldt, + float *q, + lapack_int ldq, + lapack_int *ifst, + lapack_int *ilst); +lapack_int LAPACKE_dtrexc(int matrix_order, + char compq, + lapack_int n, + double *t, + lapack_int ldt, + double *q, + lapack_int ldq, + lapack_int *ifst, + lapack_int *ilst); +lapack_int LAPACKE_ctrexc(int matrix_order, + char compq, + lapack_int n, + lapack_complex_float *t, + lapack_int ldt, + lapack_complex_float *q, + lapack_int ldq, + lapack_int ifst, + lapack_int ilst); +lapack_int LAPACKE_ztrexc(int matrix_order, + char compq, + lapack_int n, + lapack_complex_double *t, + lapack_int ldt, + lapack_complex_double *q, + lapack_int ldq, + lapack_int ifst, + lapack_int ilst); + +lapack_int LAPACKE_strrfs(int matrix_order, + char uplo, + char trans, + char diag, + lapack_int n, + lapack_int nrhs, + const float *a, + lapack_int lda, + const float *b, + lapack_int ldb, + const float *x, + lapack_int ldx, + float *ferr, + float *berr); +lapack_int LAPACKE_dtrrfs(int matrix_order, + char uplo, + char trans, + char diag, + lapack_int n, + lapack_int nrhs, + const double *a, + lapack_int lda, + const double *b, + lapack_int ldb, + const double *x, + lapack_int ldx, + double *ferr, + double *berr); +lapack_int LAPACKE_ctrrfs(int matrix_order, + char uplo, + char trans, + char diag, + lapack_int n, + lapack_int nrhs, + const lapack_complex_float *a, + lapack_int lda, + const lapack_complex_float *b, + lapack_int ldb, + const lapack_complex_float *x, + lapack_int ldx, + float *ferr, + float *berr); +lapack_int LAPACKE_ztrrfs(int matrix_order, + char uplo, + char trans, + char diag, + lapack_int n, + lapack_int nrhs, + const lapack_complex_double *a, + lapack_int lda, + const lapack_complex_double *b, + lapack_int ldb, + const lapack_complex_double *x, + lapack_int ldx, + double *ferr, + double *berr); + +lapack_int LAPACKE_strsen(int matrix_order, + char job, + char compq, + const lapack_logical *select, + lapack_int n, + float *t, + lapack_int ldt, + float *q, + lapack_int ldq, + float *wr, + float *wi, + lapack_int *m, + float *s, + float *sep); +lapack_int LAPACKE_dtrsen(int matrix_order, + char job, + char compq, + const lapack_logical *select, + lapack_int n, + double *t, + lapack_int ldt, + double *q, + lapack_int ldq, + double *wr, + double *wi, + lapack_int *m, + double *s, + double *sep); +lapack_int LAPACKE_ctrsen(int matrix_order, + char job, + char compq, + const lapack_logical *select, + lapack_int n, + lapack_complex_float *t, + lapack_int ldt, + lapack_complex_float *q, + lapack_int ldq, + lapack_complex_float *w, + lapack_int *m, + float *s, + float *sep); +lapack_int LAPACKE_ztrsen(int matrix_order, + char job, + char compq, + const lapack_logical *select, + lapack_int n, + lapack_complex_double *t, + lapack_int ldt, + lapack_complex_double *q, + lapack_int ldq, + lapack_complex_double *w, + lapack_int *m, + double *s, + double *sep); + +lapack_int LAPACKE_strsna(int matrix_order, + char job, + char howmny, + const lapack_logical *select, + lapack_int n, + const float *t, + lapack_int ldt, + const float *vl, + lapack_int ldvl, + const float *vr, + lapack_int ldvr, + float *s, + float *sep, + lapack_int mm, + lapack_int *m); +lapack_int LAPACKE_dtrsna(int matrix_order, + char job, + char howmny, + const lapack_logical *select, + lapack_int n, + const double *t, + lapack_int ldt, + const double *vl, + lapack_int ldvl, + const double *vr, + lapack_int ldvr, + double *s, + double *sep, + lapack_int mm, + lapack_int *m); +lapack_int LAPACKE_ctrsna(int matrix_order, + char job, + char howmny, + const lapack_logical *select, + lapack_int n, + const lapack_complex_float *t, + lapack_int ldt, + const lapack_complex_float *vl, + lapack_int ldvl, + const lapack_complex_float *vr, + lapack_int ldvr, + float *s, + float *sep, + lapack_int mm, + lapack_int *m); +lapack_int LAPACKE_ztrsna(int matrix_order, + char job, + char howmny, + const lapack_logical *select, + lapack_int n, + const lapack_complex_double *t, + lapack_int ldt, + const lapack_complex_double *vl, + lapack_int ldvl, + const lapack_complex_double *vr, + lapack_int ldvr, + double *s, + double *sep, + lapack_int mm, + lapack_int *m); + +lapack_int LAPACKE_strsyl(int matrix_order, + char trana, + char tranb, + lapack_int isgn, + lapack_int m, + lapack_int n, + const float *a, + lapack_int lda, + const float *b, + lapack_int ldb, + float *c, + lapack_int ldc, + float *scale); +lapack_int LAPACKE_dtrsyl(int matrix_order, + char trana, + char tranb, + lapack_int isgn, + lapack_int m, + lapack_int n, + const double *a, + lapack_int lda, + const double *b, + lapack_int ldb, + double *c, + lapack_int ldc, + double *scale); +lapack_int LAPACKE_ctrsyl(int matrix_order, + char trana, + char tranb, + lapack_int isgn, + lapack_int m, + lapack_int n, + const lapack_complex_float *a, + lapack_int lda, + const lapack_complex_float *b, + lapack_int ldb, + lapack_complex_float *c, + lapack_int ldc, + float *scale); +lapack_int LAPACKE_ztrsyl(int matrix_order, + char trana, + char tranb, + lapack_int isgn, + lapack_int m, + lapack_int n, + const lapack_complex_double *a, + lapack_int lda, + const lapack_complex_double *b, + lapack_int ldb, + lapack_complex_double *c, + lapack_int ldc, + double *scale); + +lapack_int LAPACKE_strtri(int matrix_order, char uplo, char diag, lapack_int n, float *a, lapack_int lda); +lapack_int LAPACKE_dtrtri(int matrix_order, char uplo, char diag, lapack_int n, double *a, lapack_int lda); +lapack_int + LAPACKE_ctrtri(int matrix_order, char uplo, char diag, lapack_int n, lapack_complex_float *a, lapack_int lda); +lapack_int + LAPACKE_ztrtri(int matrix_order, char uplo, char diag, lapack_int n, lapack_complex_double *a, lapack_int lda); + +lapack_int LAPACKE_strtrs(int matrix_order, + char uplo, + char trans, + char diag, + lapack_int n, + lapack_int nrhs, + const float *a, + lapack_int lda, + float *b, + lapack_int ldb); +lapack_int LAPACKE_dtrtrs(int matrix_order, + char uplo, + char trans, + char diag, + lapack_int n, + lapack_int nrhs, + const double *a, + lapack_int lda, + double *b, + lapack_int ldb); +lapack_int LAPACKE_ctrtrs(int matrix_order, + char uplo, + char trans, + char diag, + lapack_int n, + lapack_int nrhs, + const lapack_complex_float *a, + lapack_int lda, + lapack_complex_float *b, + lapack_int ldb); +lapack_int LAPACKE_ztrtrs(int matrix_order, + char uplo, + char trans, + char diag, + lapack_int n, + lapack_int nrhs, + const lapack_complex_double *a, + lapack_int lda, + lapack_complex_double *b, + lapack_int ldb); + +lapack_int + LAPACKE_strttf(int matrix_order, char transr, char uplo, lapack_int n, const float *a, lapack_int lda, float *arf); +lapack_int + LAPACKE_dtrttf(int matrix_order, char transr, char uplo, lapack_int n, const double *a, lapack_int lda, double *arf); +lapack_int LAPACKE_ctrttf(int matrix_order, + char transr, + char uplo, + lapack_int n, + const lapack_complex_float *a, + lapack_int lda, + lapack_complex_float *arf); +lapack_int LAPACKE_ztrttf(int matrix_order, + char transr, + char uplo, + lapack_int n, + const lapack_complex_double *a, + lapack_int lda, + lapack_complex_double *arf); + +lapack_int LAPACKE_strttp(int matrix_order, char uplo, lapack_int n, const float *a, lapack_int lda, float *ap); +lapack_int LAPACKE_dtrttp(int matrix_order, char uplo, lapack_int n, const double *a, lapack_int lda, double *ap); +lapack_int LAPACKE_ctrttp(int matrix_order, + char uplo, + lapack_int n, + const lapack_complex_float *a, + lapack_int lda, + lapack_complex_float *ap); +lapack_int LAPACKE_ztrttp(int matrix_order, + char uplo, + lapack_int n, + const lapack_complex_double *a, + lapack_int lda, + lapack_complex_double *ap); + +lapack_int LAPACKE_stzrzf(int matrix_order, lapack_int m, lapack_int n, float *a, lapack_int lda, float *tau); +lapack_int LAPACKE_dtzrzf(int matrix_order, lapack_int m, lapack_int n, double *a, lapack_int lda, double *tau); +lapack_int LAPACKE_ctzrzf(int matrix_order, + lapack_int m, + lapack_int n, + lapack_complex_float *a, + lapack_int lda, + lapack_complex_float *tau); +lapack_int LAPACKE_ztzrzf(int matrix_order, + lapack_int m, + lapack_int n, + lapack_complex_double *a, + lapack_int lda, + lapack_complex_double *tau); + +lapack_int LAPACKE_cungbr(int matrix_order, + char vect, + lapack_int m, + lapack_int n, + lapack_int k, + lapack_complex_float *a, + lapack_int lda, + const lapack_complex_float *tau); +lapack_int LAPACKE_zungbr(int matrix_order, + char vect, + lapack_int m, + lapack_int n, + lapack_int k, + lapack_complex_double *a, + lapack_int lda, + const lapack_complex_double *tau); + +lapack_int LAPACKE_cunghr(int matrix_order, + lapack_int n, + lapack_int ilo, + lapack_int ihi, + lapack_complex_float *a, + lapack_int lda, + const lapack_complex_float *tau); +lapack_int LAPACKE_zunghr(int matrix_order, + lapack_int n, + lapack_int ilo, + lapack_int ihi, + lapack_complex_double *a, + lapack_int lda, + const lapack_complex_double *tau); + +lapack_int LAPACKE_cunglq(int matrix_order, + lapack_int m, + lapack_int n, + lapack_int k, + lapack_complex_float *a, + lapack_int lda, + const lapack_complex_float *tau); +lapack_int LAPACKE_zunglq(int matrix_order, + lapack_int m, + lapack_int n, + lapack_int k, + lapack_complex_double *a, + lapack_int lda, + const lapack_complex_double *tau); + +lapack_int LAPACKE_cungql(int matrix_order, + lapack_int m, + lapack_int n, + lapack_int k, + lapack_complex_float *a, + lapack_int lda, + const lapack_complex_float *tau); +lapack_int LAPACKE_zungql(int matrix_order, + lapack_int m, + lapack_int n, + lapack_int k, + lapack_complex_double *a, + lapack_int lda, + const lapack_complex_double *tau); + +lapack_int LAPACKE_cungqr(int matrix_order, + lapack_int m, + lapack_int n, + lapack_int k, + lapack_complex_float *a, + lapack_int lda, + const lapack_complex_float *tau); +lapack_int LAPACKE_zungqr(int matrix_order, + lapack_int m, + lapack_int n, + lapack_int k, + lapack_complex_double *a, + lapack_int lda, + const lapack_complex_double *tau); + +lapack_int LAPACKE_cungrq(int matrix_order, + lapack_int m, + lapack_int n, + lapack_int k, + lapack_complex_float *a, + lapack_int lda, + const lapack_complex_float *tau); +lapack_int LAPACKE_zungrq(int matrix_order, + lapack_int m, + lapack_int n, + lapack_int k, + lapack_complex_double *a, + lapack_int lda, + const lapack_complex_double *tau); + +lapack_int LAPACKE_cungtr(int matrix_order, + char uplo, + lapack_int n, + lapack_complex_float *a, + lapack_int lda, + const lapack_complex_float *tau); +lapack_int LAPACKE_zungtr(int matrix_order, + char uplo, + lapack_int n, + lapack_complex_double *a, + lapack_int lda, + const lapack_complex_double *tau); + +lapack_int LAPACKE_cunmbr(int matrix_order, + char vect, + char side, + char trans, + lapack_int m, + lapack_int n, + lapack_int k, + const lapack_complex_float *a, + lapack_int lda, + const lapack_complex_float *tau, + lapack_complex_float *c, + lapack_int ldc); +lapack_int LAPACKE_zunmbr(int matrix_order, + char vect, + char side, + char trans, + lapack_int m, + lapack_int n, + lapack_int k, + const lapack_complex_double *a, + lapack_int lda, + const lapack_complex_double *tau, + lapack_complex_double *c, + lapack_int ldc); + +lapack_int LAPACKE_cunmhr(int matrix_order, + char side, + char trans, + lapack_int m, + lapack_int n, + lapack_int ilo, + lapack_int ihi, + const lapack_complex_float *a, + lapack_int lda, + const lapack_complex_float *tau, + lapack_complex_float *c, + lapack_int ldc); +lapack_int LAPACKE_zunmhr(int matrix_order, + char side, + char trans, + lapack_int m, + lapack_int n, + lapack_int ilo, + lapack_int ihi, + const lapack_complex_double *a, + lapack_int lda, + const lapack_complex_double *tau, + lapack_complex_double *c, + lapack_int ldc); + +lapack_int LAPACKE_cunmlq(int matrix_order, + char side, + char trans, + lapack_int m, + lapack_int n, + lapack_int k, + const lapack_complex_float *a, + lapack_int lda, + const lapack_complex_float *tau, + lapack_complex_float *c, + lapack_int ldc); +lapack_int LAPACKE_zunmlq(int matrix_order, + char side, + char trans, + lapack_int m, + lapack_int n, + lapack_int k, + const lapack_complex_double *a, + lapack_int lda, + const lapack_complex_double *tau, + lapack_complex_double *c, + lapack_int ldc); + +lapack_int LAPACKE_cunmql(int matrix_order, + char side, + char trans, + lapack_int m, + lapack_int n, + lapack_int k, + const lapack_complex_float *a, + lapack_int lda, + const lapack_complex_float *tau, + lapack_complex_float *c, + lapack_int ldc); +lapack_int LAPACKE_zunmql(int matrix_order, + char side, + char trans, + lapack_int m, + lapack_int n, + lapack_int k, + const lapack_complex_double *a, + lapack_int lda, + const lapack_complex_double *tau, + lapack_complex_double *c, + lapack_int ldc); + +lapack_int LAPACKE_cunmqr(int matrix_order, + char side, + char trans, + lapack_int m, + lapack_int n, + lapack_int k, + const lapack_complex_float *a, + lapack_int lda, + const lapack_complex_float *tau, + lapack_complex_float *c, + lapack_int ldc); +lapack_int LAPACKE_zunmqr(int matrix_order, + char side, + char trans, + lapack_int m, + lapack_int n, + lapack_int k, + const lapack_complex_double *a, + lapack_int lda, + const lapack_complex_double *tau, + lapack_complex_double *c, + lapack_int ldc); + +lapack_int LAPACKE_cunmrq(int matrix_order, + char side, + char trans, + lapack_int m, + lapack_int n, + lapack_int k, + const lapack_complex_float *a, + lapack_int lda, + const lapack_complex_float *tau, + lapack_complex_float *c, + lapack_int ldc); +lapack_int LAPACKE_zunmrq(int matrix_order, + char side, + char trans, + lapack_int m, + lapack_int n, + lapack_int k, + const lapack_complex_double *a, + lapack_int lda, + const lapack_complex_double *tau, + lapack_complex_double *c, + lapack_int ldc); + +lapack_int LAPACKE_cunmrz(int matrix_order, + char side, + char trans, + lapack_int m, + lapack_int n, + lapack_int k, + lapack_int l, + const lapack_complex_float *a, + lapack_int lda, + const lapack_complex_float *tau, + lapack_complex_float *c, + lapack_int ldc); +lapack_int LAPACKE_zunmrz(int matrix_order, + char side, + char trans, + lapack_int m, + lapack_int n, + lapack_int k, + lapack_int l, + const lapack_complex_double *a, + lapack_int lda, + const lapack_complex_double *tau, + lapack_complex_double *c, + lapack_int ldc); + +lapack_int LAPACKE_cunmtr(int matrix_order, + char side, + char uplo, + char trans, + lapack_int m, + lapack_int n, + const lapack_complex_float *a, + lapack_int lda, + const lapack_complex_float *tau, + lapack_complex_float *c, + lapack_int ldc); +lapack_int LAPACKE_zunmtr(int matrix_order, + char side, + char uplo, + char trans, + lapack_int m, + lapack_int n, + const lapack_complex_double *a, + lapack_int lda, + const lapack_complex_double *tau, + lapack_complex_double *c, + lapack_int ldc); + +lapack_int LAPACKE_cupgtr(int matrix_order, + char uplo, + lapack_int n, + const lapack_complex_float *ap, + const lapack_complex_float *tau, + lapack_complex_float *q, + lapack_int ldq); +lapack_int LAPACKE_zupgtr(int matrix_order, + char uplo, + lapack_int n, + const lapack_complex_double *ap, + const lapack_complex_double *tau, + lapack_complex_double *q, + lapack_int ldq); + +lapack_int LAPACKE_cupmtr(int matrix_order, + char side, + char uplo, + char trans, + lapack_int m, + lapack_int n, + const lapack_complex_float *ap, + const lapack_complex_float *tau, + lapack_complex_float *c, + lapack_int ldc); +lapack_int LAPACKE_zupmtr(int matrix_order, + char side, + char uplo, + char trans, + lapack_int m, + lapack_int n, + const lapack_complex_double *ap, + const lapack_complex_double *tau, + lapack_complex_double *c, + lapack_int ldc); + +lapack_int LAPACKE_sbdsdc_work(int matrix_order, + char uplo, + char compq, + lapack_int n, + float *d, + float *e, + float *u, + lapack_int ldu, + float *vt, + lapack_int ldvt, + float *q, + lapack_int *iq, + float *work, + lapack_int *iwork); +lapack_int LAPACKE_dbdsdc_work(int matrix_order, + char uplo, + char compq, + lapack_int n, + double *d, + double *e, + double *u, + lapack_int ldu, + double *vt, + lapack_int ldvt, + double *q, + lapack_int *iq, + double *work, + lapack_int *iwork); + +lapack_int LAPACKE_sbdsqr_work(int matrix_order, + char uplo, + lapack_int n, + lapack_int ncvt, + lapack_int nru, + lapack_int ncc, + float *d, + float *e, + float *vt, + lapack_int ldvt, + float *u, + lapack_int ldu, + float *c, + lapack_int ldc, + float *work); +lapack_int LAPACKE_dbdsqr_work(int matrix_order, + char uplo, + lapack_int n, + lapack_int ncvt, + lapack_int nru, + lapack_int ncc, + double *d, + double *e, + double *vt, + lapack_int ldvt, + double *u, + lapack_int ldu, + double *c, + lapack_int ldc, + double *work); +lapack_int LAPACKE_cbdsqr_work(int matrix_order, + char uplo, + lapack_int n, + lapack_int ncvt, + lapack_int nru, + lapack_int ncc, + float *d, + float *e, + lapack_complex_float *vt, + lapack_int ldvt, + lapack_complex_float *u, + lapack_int ldu, + lapack_complex_float *c, + lapack_int ldc, + float *work); +lapack_int LAPACKE_zbdsqr_work(int matrix_order, + char uplo, + lapack_int n, + lapack_int ncvt, + lapack_int nru, + lapack_int ncc, + double *d, + double *e, + lapack_complex_double *vt, + lapack_int ldvt, + lapack_complex_double *u, + lapack_int ldu, + lapack_complex_double *c, + lapack_int ldc, + double *work); + +lapack_int LAPACKE_sdisna_work(char job, lapack_int m, lapack_int n, const float *d, float *sep); +lapack_int LAPACKE_ddisna_work(char job, lapack_int m, lapack_int n, const double *d, double *sep); + +lapack_int LAPACKE_sgbbrd_work(int matrix_order, + char vect, + lapack_int m, + lapack_int n, + lapack_int ncc, + lapack_int kl, + lapack_int ku, + float *ab, + lapack_int ldab, + float *d, + float *e, + float *q, + lapack_int ldq, + float *pt, + lapack_int ldpt, + float *c, + lapack_int ldc, + float *work); +lapack_int LAPACKE_dgbbrd_work(int matrix_order, + char vect, + lapack_int m, + lapack_int n, + lapack_int ncc, + lapack_int kl, + lapack_int ku, + double *ab, + lapack_int ldab, + double *d, + double *e, + double *q, + lapack_int ldq, + double *pt, + lapack_int ldpt, + double *c, + lapack_int ldc, + double *work); +lapack_int LAPACKE_cgbbrd_work(int matrix_order, + char vect, + lapack_int m, + lapack_int n, + lapack_int ncc, + lapack_int kl, + lapack_int ku, + lapack_complex_float *ab, + lapack_int ldab, + float *d, + float *e, + lapack_complex_float *q, + lapack_int ldq, + lapack_complex_float *pt, + lapack_int ldpt, + lapack_complex_float *c, + lapack_int ldc, + lapack_complex_float *work, + float *rwork); +lapack_int LAPACKE_zgbbrd_work(int matrix_order, + char vect, + lapack_int m, + lapack_int n, + lapack_int ncc, + lapack_int kl, + lapack_int ku, + lapack_complex_double *ab, + lapack_int ldab, + double *d, + double *e, + lapack_complex_double *q, + lapack_int ldq, + lapack_complex_double *pt, + lapack_int ldpt, + lapack_complex_double *c, + lapack_int ldc, + lapack_complex_double *work, + double *rwork); + +lapack_int LAPACKE_sgbcon_work(int matrix_order, + char norm, + lapack_int n, + lapack_int kl, + lapack_int ku, + const float *ab, + lapack_int ldab, + const lapack_int *ipiv, + float anorm, + float *rcond, + float *work, + lapack_int *iwork); +lapack_int LAPACKE_dgbcon_work(int matrix_order, + char norm, + lapack_int n, + lapack_int kl, + lapack_int ku, + const double *ab, + lapack_int ldab, + const lapack_int *ipiv, + double anorm, + double *rcond, + double *work, + lapack_int *iwork); +lapack_int LAPACKE_cgbcon_work(int matrix_order, + char norm, + lapack_int n, + lapack_int kl, + lapack_int ku, + const lapack_complex_float *ab, + lapack_int ldab, + const lapack_int *ipiv, + float anorm, + float *rcond, + lapack_complex_float *work, + float *rwork); +lapack_int LAPACKE_zgbcon_work(int matrix_order, + char norm, + lapack_int n, + lapack_int kl, + lapack_int ku, + const lapack_complex_double *ab, + lapack_int ldab, + const lapack_int *ipiv, + double anorm, + double *rcond, + lapack_complex_double *work, + double *rwork); + +lapack_int LAPACKE_sgbequ_work(int matrix_order, + lapack_int m, + lapack_int n, + lapack_int kl, + lapack_int ku, + const float *ab, + lapack_int ldab, + float *r, + float *c, + float *rowcnd, + float *colcnd, + float *amax); +lapack_int LAPACKE_dgbequ_work(int matrix_order, + lapack_int m, + lapack_int n, + lapack_int kl, + lapack_int ku, + const double *ab, + lapack_int ldab, + double *r, + double *c, + double *rowcnd, + double *colcnd, + double *amax); +lapack_int LAPACKE_cgbequ_work(int matrix_order, + lapack_int m, + lapack_int n, + lapack_int kl, + lapack_int ku, + const lapack_complex_float *ab, + lapack_int ldab, + float *r, + float *c, + float *rowcnd, + float *colcnd, + float *amax); +lapack_int LAPACKE_zgbequ_work(int matrix_order, + lapack_int m, + lapack_int n, + lapack_int kl, + lapack_int ku, + const lapack_complex_double *ab, + lapack_int ldab, + double *r, + double *c, + double *rowcnd, + double *colcnd, + double *amax); + +lapack_int LAPACKE_sgbequb_work(int matrix_order, + lapack_int m, + lapack_int n, + lapack_int kl, + lapack_int ku, + const float *ab, + lapack_int ldab, + float *r, + float *c, + float *rowcnd, + float *colcnd, + float *amax); +lapack_int LAPACKE_dgbequb_work(int matrix_order, + lapack_int m, + lapack_int n, + lapack_int kl, + lapack_int ku, + const double *ab, + lapack_int ldab, + double *r, + double *c, + double *rowcnd, + double *colcnd, + double *amax); +lapack_int LAPACKE_cgbequb_work(int matrix_order, + lapack_int m, + lapack_int n, + lapack_int kl, + lapack_int ku, + const lapack_complex_float *ab, + lapack_int ldab, + float *r, + float *c, + float *rowcnd, + float *colcnd, + float *amax); +lapack_int LAPACKE_zgbequb_work(int matrix_order, + lapack_int m, + lapack_int n, + lapack_int kl, + lapack_int ku, + const lapack_complex_double *ab, + lapack_int ldab, + double *r, + double *c, + double *rowcnd, + double *colcnd, + double *amax); + +lapack_int LAPACKE_sgbrfs_work(int matrix_order, + char trans, + lapack_int n, + lapack_int kl, + lapack_int ku, + lapack_int nrhs, + const float *ab, + lapack_int ldab, + const float *afb, + lapack_int ldafb, + const lapack_int *ipiv, + const float *b, + lapack_int ldb, + float *x, + lapack_int ldx, + float *ferr, + float *berr, + float *work, + lapack_int *iwork); +lapack_int LAPACKE_dgbrfs_work(int matrix_order, + char trans, + lapack_int n, + lapack_int kl, + lapack_int ku, + lapack_int nrhs, + const double *ab, + lapack_int ldab, + const double *afb, + lapack_int ldafb, + const lapack_int *ipiv, + const double *b, + lapack_int ldb, + double *x, + lapack_int ldx, + double *ferr, + double *berr, + double *work, + lapack_int *iwork); +lapack_int LAPACKE_cgbrfs_work(int matrix_order, + char trans, + lapack_int n, + lapack_int kl, + lapack_int ku, + lapack_int nrhs, + const lapack_complex_float *ab, + lapack_int ldab, + const lapack_complex_float *afb, + lapack_int ldafb, + const lapack_int *ipiv, + const lapack_complex_float *b, + lapack_int ldb, + lapack_complex_float *x, + lapack_int ldx, + float *ferr, + float *berr, + lapack_complex_float *work, + float *rwork); +lapack_int LAPACKE_zgbrfs_work(int matrix_order, + char trans, + lapack_int n, + lapack_int kl, + lapack_int ku, + lapack_int nrhs, + const lapack_complex_double *ab, + lapack_int ldab, + const lapack_complex_double *afb, + lapack_int ldafb, + const lapack_int *ipiv, + const lapack_complex_double *b, + lapack_int ldb, + lapack_complex_double *x, + lapack_int ldx, + double *ferr, + double *berr, + lapack_complex_double *work, + double *rwork); + +lapack_int LAPACKE_sgbrfsx_work(int matrix_order, + char trans, + char equed, + lapack_int n, + lapack_int kl, + lapack_int ku, + lapack_int nrhs, + const float *ab, + lapack_int ldab, + const float *afb, + lapack_int ldafb, + const lapack_int *ipiv, + const float *r, + const float *c, + const float *b, + lapack_int ldb, + float *x, + lapack_int ldx, + float *rcond, + float *berr, + lapack_int n_err_bnds, + float *err_bnds_norm, + float *err_bnds_comp, + lapack_int nparams, + float *params, + float *work, + lapack_int *iwork); +lapack_int LAPACKE_dgbrfsx_work(int matrix_order, + char trans, + char equed, + lapack_int n, + lapack_int kl, + lapack_int ku, + lapack_int nrhs, + const double *ab, + lapack_int ldab, + const double *afb, + lapack_int ldafb, + const lapack_int *ipiv, + const double *r, + const double *c, + const double *b, + lapack_int ldb, + double *x, + lapack_int ldx, + double *rcond, + double *berr, + lapack_int n_err_bnds, + double *err_bnds_norm, + double *err_bnds_comp, + lapack_int nparams, + double *params, + double *work, + lapack_int *iwork); +lapack_int LAPACKE_cgbrfsx_work(int matrix_order, + char trans, + char equed, + lapack_int n, + lapack_int kl, + lapack_int ku, + lapack_int nrhs, + const lapack_complex_float *ab, + lapack_int ldab, + const lapack_complex_float *afb, + lapack_int ldafb, + const lapack_int *ipiv, + const float *r, + const float *c, + const lapack_complex_float *b, + lapack_int ldb, + lapack_complex_float *x, + lapack_int ldx, + float *rcond, + float *berr, + lapack_int n_err_bnds, + float *err_bnds_norm, + float *err_bnds_comp, + lapack_int nparams, + float *params, + lapack_complex_float *work, + float *rwork); +lapack_int LAPACKE_zgbrfsx_work(int matrix_order, + char trans, + char equed, + lapack_int n, + lapack_int kl, + lapack_int ku, + lapack_int nrhs, + const lapack_complex_double *ab, + lapack_int ldab, + const lapack_complex_double *afb, + lapack_int ldafb, + const lapack_int *ipiv, + const double *r, + const double *c, + const lapack_complex_double *b, + lapack_int ldb, + lapack_complex_double *x, + lapack_int ldx, + double *rcond, + double *berr, + lapack_int n_err_bnds, + double *err_bnds_norm, + double *err_bnds_comp, + lapack_int nparams, + double *params, + lapack_complex_double *work, + double *rwork); + +lapack_int LAPACKE_sgbsv_work(int matrix_order, + lapack_int n, + lapack_int kl, + lapack_int ku, + lapack_int nrhs, + float *ab, + lapack_int ldab, + lapack_int *ipiv, + float *b, + lapack_int ldb); +lapack_int LAPACKE_dgbsv_work(int matrix_order, + lapack_int n, + lapack_int kl, + lapack_int ku, + lapack_int nrhs, + double *ab, + lapack_int ldab, + lapack_int *ipiv, + double *b, + lapack_int ldb); +lapack_int LAPACKE_cgbsv_work(int matrix_order, + lapack_int n, + lapack_int kl, + lapack_int ku, + lapack_int nrhs, + lapack_complex_float *ab, + lapack_int ldab, + lapack_int *ipiv, + lapack_complex_float *b, + lapack_int ldb); +lapack_int LAPACKE_zgbsv_work(int matrix_order, + lapack_int n, + lapack_int kl, + lapack_int ku, + lapack_int nrhs, + lapack_complex_double *ab, + lapack_int ldab, + lapack_int *ipiv, + lapack_complex_double *b, + lapack_int ldb); + +lapack_int LAPACKE_sgbsvx_work(int matrix_order, + char fact, + char trans, + lapack_int n, + lapack_int kl, + lapack_int ku, + lapack_int nrhs, + float *ab, + lapack_int ldab, + float *afb, + lapack_int ldafb, + lapack_int *ipiv, + char *equed, + float *r, + float *c, + float *b, + lapack_int ldb, + float *x, + lapack_int ldx, + float *rcond, + float *ferr, + float *berr, + float *work, + lapack_int *iwork); +lapack_int LAPACKE_dgbsvx_work(int matrix_order, + char fact, + char trans, + lapack_int n, + lapack_int kl, + lapack_int ku, + lapack_int nrhs, + double *ab, + lapack_int ldab, + double *afb, + lapack_int ldafb, + lapack_int *ipiv, + char *equed, + double *r, + double *c, + double *b, + lapack_int ldb, + double *x, + lapack_int ldx, + double *rcond, + double *ferr, + double *berr, + double *work, + lapack_int *iwork); +lapack_int LAPACKE_cgbsvx_work(int matrix_order, + char fact, + char trans, + lapack_int n, + lapack_int kl, + lapack_int ku, + lapack_int nrhs, + lapack_complex_float *ab, + lapack_int ldab, + lapack_complex_float *afb, + lapack_int ldafb, + lapack_int *ipiv, + char *equed, + float *r, + float *c, + lapack_complex_float *b, + lapack_int ldb, + lapack_complex_float *x, + lapack_int ldx, + float *rcond, + float *ferr, + float *berr, + lapack_complex_float *work, + float *rwork); +lapack_int LAPACKE_zgbsvx_work(int matrix_order, + char fact, + char trans, + lapack_int n, + lapack_int kl, + lapack_int ku, + lapack_int nrhs, + lapack_complex_double *ab, + lapack_int ldab, + lapack_complex_double *afb, + lapack_int ldafb, + lapack_int *ipiv, + char *equed, + double *r, + double *c, + lapack_complex_double *b, + lapack_int ldb, + lapack_complex_double *x, + lapack_int ldx, + double *rcond, + double *ferr, + double *berr, + lapack_complex_double *work, + double *rwork); + +lapack_int LAPACKE_sgbsvxx_work(int matrix_order, + char fact, + char trans, + lapack_int n, + lapack_int kl, + lapack_int ku, + lapack_int nrhs, + float *ab, + lapack_int ldab, + float *afb, + lapack_int ldafb, + lapack_int *ipiv, + char *equed, + float *r, + float *c, + float *b, + lapack_int ldb, + float *x, + lapack_int ldx, + float *rcond, + float *rpvgrw, + float *berr, + lapack_int n_err_bnds, + float *err_bnds_norm, + float *err_bnds_comp, + lapack_int nparams, + float *params, + float *work, + lapack_int *iwork); +lapack_int LAPACKE_dgbsvxx_work(int matrix_order, + char fact, + char trans, + lapack_int n, + lapack_int kl, + lapack_int ku, + lapack_int nrhs, + double *ab, + lapack_int ldab, + double *afb, + lapack_int ldafb, + lapack_int *ipiv, + char *equed, + double *r, + double *c, + double *b, + lapack_int ldb, + double *x, + lapack_int ldx, + double *rcond, + double *rpvgrw, + double *berr, + lapack_int n_err_bnds, + double *err_bnds_norm, + double *err_bnds_comp, + lapack_int nparams, + double *params, + double *work, + lapack_int *iwork); +lapack_int LAPACKE_cgbsvxx_work(int matrix_order, + char fact, + char trans, + lapack_int n, + lapack_int kl, + lapack_int ku, + lapack_int nrhs, + lapack_complex_float *ab, + lapack_int ldab, + lapack_complex_float *afb, + lapack_int ldafb, + lapack_int *ipiv, + char *equed, + float *r, + float *c, + lapack_complex_float *b, + lapack_int ldb, + lapack_complex_float *x, + lapack_int ldx, + float *rcond, + float *rpvgrw, + float *berr, + lapack_int n_err_bnds, + float *err_bnds_norm, + float *err_bnds_comp, + lapack_int nparams, + float *params, + lapack_complex_float *work, + float *rwork); +lapack_int LAPACKE_zgbsvxx_work(int matrix_order, + char fact, + char trans, + lapack_int n, + lapack_int kl, + lapack_int ku, + lapack_int nrhs, + lapack_complex_double *ab, + lapack_int ldab, + lapack_complex_double *afb, + lapack_int ldafb, + lapack_int *ipiv, + char *equed, + double *r, + double *c, + lapack_complex_double *b, + lapack_int ldb, + lapack_complex_double *x, + lapack_int ldx, + double *rcond, + double *rpvgrw, + double *berr, + lapack_int n_err_bnds, + double *err_bnds_norm, + double *err_bnds_comp, + lapack_int nparams, + double *params, + lapack_complex_double *work, + double *rwork); + +lapack_int LAPACKE_sgbtrf_work(int matrix_order, + lapack_int m, + lapack_int n, + lapack_int kl, + lapack_int ku, + float *ab, + lapack_int ldab, + lapack_int *ipiv); +lapack_int LAPACKE_dgbtrf_work(int matrix_order, + lapack_int m, + lapack_int n, + lapack_int kl, + lapack_int ku, + double *ab, + lapack_int ldab, + lapack_int *ipiv); +lapack_int LAPACKE_cgbtrf_work(int matrix_order, + lapack_int m, + lapack_int n, + lapack_int kl, + lapack_int ku, + lapack_complex_float *ab, + lapack_int ldab, + lapack_int *ipiv); +lapack_int LAPACKE_zgbtrf_work(int matrix_order, + lapack_int m, + lapack_int n, + lapack_int kl, + lapack_int ku, + lapack_complex_double *ab, + lapack_int ldab, + lapack_int *ipiv); + +lapack_int LAPACKE_sgbtrs_work(int matrix_order, + char trans, + lapack_int n, + lapack_int kl, + lapack_int ku, + lapack_int nrhs, + const float *ab, + lapack_int ldab, + const lapack_int *ipiv, + float *b, + lapack_int ldb); +lapack_int LAPACKE_dgbtrs_work(int matrix_order, + char trans, + lapack_int n, + lapack_int kl, + lapack_int ku, + lapack_int nrhs, + const double *ab, + lapack_int ldab, + const lapack_int *ipiv, + double *b, + lapack_int ldb); +lapack_int LAPACKE_cgbtrs_work(int matrix_order, + char trans, + lapack_int n, + lapack_int kl, + lapack_int ku, + lapack_int nrhs, + const lapack_complex_float *ab, + lapack_int ldab, + const lapack_int *ipiv, + lapack_complex_float *b, + lapack_int ldb); +lapack_int LAPACKE_zgbtrs_work(int matrix_order, + char trans, + lapack_int n, + lapack_int kl, + lapack_int ku, + lapack_int nrhs, + const lapack_complex_double *ab, + lapack_int ldab, + const lapack_int *ipiv, + lapack_complex_double *b, + lapack_int ldb); + +lapack_int LAPACKE_sgebak_work(int matrix_order, + char job, + char side, + lapack_int n, + lapack_int ilo, + lapack_int ihi, + const float *scale, + lapack_int m, + float *v, + lapack_int ldv); +lapack_int LAPACKE_dgebak_work(int matrix_order, + char job, + char side, + lapack_int n, + lapack_int ilo, + lapack_int ihi, + const double *scale, + lapack_int m, + double *v, + lapack_int ldv); +lapack_int LAPACKE_cgebak_work(int matrix_order, + char job, + char side, + lapack_int n, + lapack_int ilo, + lapack_int ihi, + const float *scale, + lapack_int m, + lapack_complex_float *v, + lapack_int ldv); +lapack_int LAPACKE_zgebak_work(int matrix_order, + char job, + char side, + lapack_int n, + lapack_int ilo, + lapack_int ihi, + const double *scale, + lapack_int m, + lapack_complex_double *v, + lapack_int ldv); + +lapack_int LAPACKE_sgebal_work(int matrix_order, + char job, + lapack_int n, + float *a, + lapack_int lda, + lapack_int *ilo, + lapack_int *ihi, + float *scale); +lapack_int LAPACKE_dgebal_work(int matrix_order, + char job, + lapack_int n, + double *a, + lapack_int lda, + lapack_int *ilo, + lapack_int *ihi, + double *scale); +lapack_int LAPACKE_cgebal_work(int matrix_order, + char job, + lapack_int n, + lapack_complex_float *a, + lapack_int lda, + lapack_int *ilo, + lapack_int *ihi, + float *scale); +lapack_int LAPACKE_zgebal_work(int matrix_order, + char job, + lapack_int n, + lapack_complex_double *a, + lapack_int lda, + lapack_int *ilo, + lapack_int *ihi, + double *scale); + +lapack_int LAPACKE_sgebrd_work(int matrix_order, + lapack_int m, + lapack_int n, + float *a, + lapack_int lda, + float *d, + float *e, + float *tauq, + float *taup, + float *work, + lapack_int lwork); +lapack_int LAPACKE_dgebrd_work(int matrix_order, + lapack_int m, + lapack_int n, + double *a, + lapack_int lda, + double *d, + double *e, + double *tauq, + double *taup, + double *work, + lapack_int lwork); +lapack_int LAPACKE_cgebrd_work(int matrix_order, + lapack_int m, + lapack_int n, + lapack_complex_float *a, + lapack_int lda, + float *d, + float *e, + lapack_complex_float *tauq, + lapack_complex_float *taup, + lapack_complex_float *work, + lapack_int lwork); +lapack_int LAPACKE_zgebrd_work(int matrix_order, + lapack_int m, + lapack_int n, + lapack_complex_double *a, + lapack_int lda, + double *d, + double *e, + lapack_complex_double *tauq, + lapack_complex_double *taup, + lapack_complex_double *work, + lapack_int lwork); + +lapack_int LAPACKE_sgecon_work(int matrix_order, + char norm, + lapack_int n, + const float *a, + lapack_int lda, + float anorm, + float *rcond, + float *work, + lapack_int *iwork); +lapack_int LAPACKE_dgecon_work(int matrix_order, + char norm, + lapack_int n, + const double *a, + lapack_int lda, + double anorm, + double *rcond, + double *work, + lapack_int *iwork); +lapack_int LAPACKE_cgecon_work(int matrix_order, + char norm, + lapack_int n, + const lapack_complex_float *a, + lapack_int lda, + float anorm, + float *rcond, + lapack_complex_float *work, + float *rwork); +lapack_int LAPACKE_zgecon_work(int matrix_order, + char norm, + lapack_int n, + const lapack_complex_double *a, + lapack_int lda, + double anorm, + double *rcond, + lapack_complex_double *work, + double *rwork); + +lapack_int LAPACKE_sgeequ_work(int matrix_order, + lapack_int m, + lapack_int n, + const float *a, + lapack_int lda, + float *r, + float *c, + float *rowcnd, + float *colcnd, + float *amax); +lapack_int LAPACKE_dgeequ_work(int matrix_order, + lapack_int m, + lapack_int n, + const double *a, + lapack_int lda, + double *r, + double *c, + double *rowcnd, + double *colcnd, + double *amax); +lapack_int LAPACKE_cgeequ_work(int matrix_order, + lapack_int m, + lapack_int n, + const lapack_complex_float *a, + lapack_int lda, + float *r, + float *c, + float *rowcnd, + float *colcnd, + float *amax); +lapack_int LAPACKE_zgeequ_work(int matrix_order, + lapack_int m, + lapack_int n, + const lapack_complex_double *a, + lapack_int lda, + double *r, + double *c, + double *rowcnd, + double *colcnd, + double *amax); + +lapack_int LAPACKE_sgeequb_work(int matrix_order, + lapack_int m, + lapack_int n, + const float *a, + lapack_int lda, + float *r, + float *c, + float *rowcnd, + float *colcnd, + float *amax); +lapack_int LAPACKE_dgeequb_work(int matrix_order, + lapack_int m, + lapack_int n, + const double *a, + lapack_int lda, + double *r, + double *c, + double *rowcnd, + double *colcnd, + double *amax); +lapack_int LAPACKE_cgeequb_work(int matrix_order, + lapack_int m, + lapack_int n, + const lapack_complex_float *a, + lapack_int lda, + float *r, + float *c, + float *rowcnd, + float *colcnd, + float *amax); +lapack_int LAPACKE_zgeequb_work(int matrix_order, + lapack_int m, + lapack_int n, + const lapack_complex_double *a, + lapack_int lda, + double *r, + double *c, + double *rowcnd, + double *colcnd, + double *amax); + +lapack_int LAPACKE_sgees_work(int matrix_order, + char jobvs, + char sort, + LAPACK_S_SELECT2 select, + lapack_int n, + float *a, + lapack_int lda, + lapack_int *sdim, + float *wr, + float *wi, + float *vs, + lapack_int ldvs, + float *work, + lapack_int lwork, + lapack_logical *bwork); +lapack_int LAPACKE_dgees_work(int matrix_order, + char jobvs, + char sort, + LAPACK_D_SELECT2 select, + lapack_int n, + double *a, + lapack_int lda, + lapack_int *sdim, + double *wr, + double *wi, + double *vs, + lapack_int ldvs, + double *work, + lapack_int lwork, + lapack_logical *bwork); +lapack_int LAPACKE_cgees_work(int matrix_order, + char jobvs, + char sort, + LAPACK_C_SELECT1 select, + lapack_int n, + lapack_complex_float *a, + lapack_int lda, + lapack_int *sdim, + lapack_complex_float *w, + lapack_complex_float *vs, + lapack_int ldvs, + lapack_complex_float *work, + lapack_int lwork, + float *rwork, + lapack_logical *bwork); +lapack_int LAPACKE_zgees_work(int matrix_order, + char jobvs, + char sort, + LAPACK_Z_SELECT1 select, + lapack_int n, + lapack_complex_double *a, + lapack_int lda, + lapack_int *sdim, + lapack_complex_double *w, + lapack_complex_double *vs, + lapack_int ldvs, + lapack_complex_double *work, + lapack_int lwork, + double *rwork, + lapack_logical *bwork); + +lapack_int LAPACKE_sgeesx_work(int matrix_order, + char jobvs, + char sort, + LAPACK_S_SELECT2 select, + char sense, + lapack_int n, + float *a, + lapack_int lda, + lapack_int *sdim, + float *wr, + float *wi, + float *vs, + lapack_int ldvs, + float *rconde, + float *rcondv, + float *work, + lapack_int lwork, + lapack_int *iwork, + lapack_int liwork, + lapack_logical *bwork); +lapack_int LAPACKE_dgeesx_work(int matrix_order, + char jobvs, + char sort, + LAPACK_D_SELECT2 select, + char sense, + lapack_int n, + double *a, + lapack_int lda, + lapack_int *sdim, + double *wr, + double *wi, + double *vs, + lapack_int ldvs, + double *rconde, + double *rcondv, + double *work, + lapack_int lwork, + lapack_int *iwork, + lapack_int liwork, + lapack_logical *bwork); +lapack_int LAPACKE_cgeesx_work(int matrix_order, + char jobvs, + char sort, + LAPACK_C_SELECT1 select, + char sense, + lapack_int n, + lapack_complex_float *a, + lapack_int lda, + lapack_int *sdim, + lapack_complex_float *w, + lapack_complex_float *vs, + lapack_int ldvs, + float *rconde, + float *rcondv, + lapack_complex_float *work, + lapack_int lwork, + float *rwork, + lapack_logical *bwork); +lapack_int LAPACKE_zgeesx_work(int matrix_order, + char jobvs, + char sort, + LAPACK_Z_SELECT1 select, + char sense, + lapack_int n, + lapack_complex_double *a, + lapack_int lda, + lapack_int *sdim, + lapack_complex_double *w, + lapack_complex_double *vs, + lapack_int ldvs, + double *rconde, + double *rcondv, + lapack_complex_double *work, + lapack_int lwork, + double *rwork, + lapack_logical *bwork); + +lapack_int LAPACKE_sgeev_work(int matrix_order, + char jobvl, + char jobvr, + lapack_int n, + float *a, + lapack_int lda, + float *wr, + float *wi, + float *vl, + lapack_int ldvl, + float *vr, + lapack_int ldvr, + float *work, + lapack_int lwork); +lapack_int LAPACKE_dgeev_work(int matrix_order, + char jobvl, + char jobvr, + lapack_int n, + double *a, + lapack_int lda, + double *wr, + double *wi, + double *vl, + lapack_int ldvl, + double *vr, + lapack_int ldvr, + double *work, + lapack_int lwork); +lapack_int LAPACKE_cgeev_work(int matrix_order, + char jobvl, + char jobvr, + lapack_int n, + lapack_complex_float *a, + lapack_int lda, + lapack_complex_float *w, + lapack_complex_float *vl, + lapack_int ldvl, + lapack_complex_float *vr, + lapack_int ldvr, + lapack_complex_float *work, + lapack_int lwork, + float *rwork); +lapack_int LAPACKE_zgeev_work(int matrix_order, + char jobvl, + char jobvr, + lapack_int n, + lapack_complex_double *a, + lapack_int lda, + lapack_complex_double *w, + lapack_complex_double *vl, + lapack_int ldvl, + lapack_complex_double *vr, + lapack_int ldvr, + lapack_complex_double *work, + lapack_int lwork, + double *rwork); + +lapack_int LAPACKE_sgeevx_work(int matrix_order, + char balanc, + char jobvl, + char jobvr, + char sense, + lapack_int n, + float *a, + lapack_int lda, + float *wr, + float *wi, + float *vl, + lapack_int ldvl, + float *vr, + lapack_int ldvr, + lapack_int *ilo, + lapack_int *ihi, + float *scale, + float *abnrm, + float *rconde, + float *rcondv, + float *work, + lapack_int lwork, + lapack_int *iwork); +lapack_int LAPACKE_dgeevx_work(int matrix_order, + char balanc, + char jobvl, + char jobvr, + char sense, + lapack_int n, + double *a, + lapack_int lda, + double *wr, + double *wi, + double *vl, + lapack_int ldvl, + double *vr, + lapack_int ldvr, + lapack_int *ilo, + lapack_int *ihi, + double *scale, + double *abnrm, + double *rconde, + double *rcondv, + double *work, + lapack_int lwork, + lapack_int *iwork); +lapack_int LAPACKE_cgeevx_work(int matrix_order, + char balanc, + char jobvl, + char jobvr, + char sense, + lapack_int n, + lapack_complex_float *a, + lapack_int lda, + lapack_complex_float *w, + lapack_complex_float *vl, + lapack_int ldvl, + lapack_complex_float *vr, + lapack_int ldvr, + lapack_int *ilo, + lapack_int *ihi, + float *scale, + float *abnrm, + float *rconde, + float *rcondv, + lapack_complex_float *work, + lapack_int lwork, + float *rwork); +lapack_int LAPACKE_zgeevx_work(int matrix_order, + char balanc, + char jobvl, + char jobvr, + char sense, + lapack_int n, + lapack_complex_double *a, + lapack_int lda, + lapack_complex_double *w, + lapack_complex_double *vl, + lapack_int ldvl, + lapack_complex_double *vr, + lapack_int ldvr, + lapack_int *ilo, + lapack_int *ihi, + double *scale, + double *abnrm, + double *rconde, + double *rcondv, + lapack_complex_double *work, + lapack_int lwork, + double *rwork); + +lapack_int LAPACKE_sgehrd_work(int matrix_order, + lapack_int n, + lapack_int ilo, + lapack_int ihi, + float *a, + lapack_int lda, + float *tau, + float *work, + lapack_int lwork); +lapack_int LAPACKE_dgehrd_work(int matrix_order, + lapack_int n, + lapack_int ilo, + lapack_int ihi, + double *a, + lapack_int lda, + double *tau, + double *work, + lapack_int lwork); +lapack_int LAPACKE_cgehrd_work(int matrix_order, + lapack_int n, + lapack_int ilo, + lapack_int ihi, + lapack_complex_float *a, + lapack_int lda, + lapack_complex_float *tau, + lapack_complex_float *work, + lapack_int lwork); +lapack_int LAPACKE_zgehrd_work(int matrix_order, + lapack_int n, + lapack_int ilo, + lapack_int ihi, + lapack_complex_double *a, + lapack_int lda, + lapack_complex_double *tau, + lapack_complex_double *work, + lapack_int lwork); + +lapack_int LAPACKE_sgejsv_work(int matrix_order, + char joba, + char jobu, + char jobv, + char jobr, + char jobt, + char jobp, + lapack_int m, + lapack_int n, + float *a, + lapack_int lda, + float *sva, + float *u, + lapack_int ldu, + float *v, + lapack_int ldv, + float *work, + lapack_int lwork, + lapack_int *iwork); +lapack_int LAPACKE_dgejsv_work(int matrix_order, + char joba, + char jobu, + char jobv, + char jobr, + char jobt, + char jobp, + lapack_int m, + lapack_int n, + double *a, + lapack_int lda, + double *sva, + double *u, + lapack_int ldu, + double *v, + lapack_int ldv, + double *work, + lapack_int lwork, + lapack_int *iwork); + +lapack_int + LAPACKE_sgelq2_work(int matrix_order, lapack_int m, lapack_int n, float *a, lapack_int lda, float *tau, float *work); +lapack_int LAPACKE_dgelq2_work(int matrix_order, + lapack_int m, + lapack_int n, + double *a, + lapack_int lda, + double *tau, + double *work); +lapack_int LAPACKE_cgelq2_work(int matrix_order, + lapack_int m, + lapack_int n, + lapack_complex_float *a, + lapack_int lda, + lapack_complex_float *tau, + lapack_complex_float *work); +lapack_int LAPACKE_zgelq2_work(int matrix_order, + lapack_int m, + lapack_int n, + lapack_complex_double *a, + lapack_int lda, + lapack_complex_double *tau, + lapack_complex_double *work); + +lapack_int LAPACKE_sgelqf_work(int matrix_order, + lapack_int m, + lapack_int n, + float *a, + lapack_int lda, + float *tau, + float *work, + lapack_int lwork); +lapack_int LAPACKE_dgelqf_work(int matrix_order, + lapack_int m, + lapack_int n, + double *a, + lapack_int lda, + double *tau, + double *work, + lapack_int lwork); +lapack_int LAPACKE_cgelqf_work(int matrix_order, + lapack_int m, + lapack_int n, + lapack_complex_float *a, + lapack_int lda, + lapack_complex_float *tau, + lapack_complex_float *work, + lapack_int lwork); +lapack_int LAPACKE_zgelqf_work(int matrix_order, + lapack_int m, + lapack_int n, + lapack_complex_double *a, + lapack_int lda, + lapack_complex_double *tau, + lapack_complex_double *work, + lapack_int lwork); + +lapack_int LAPACKE_sgels_work(int matrix_order, + char trans, + lapack_int m, + lapack_int n, + lapack_int nrhs, + float *a, + lapack_int lda, + float *b, + lapack_int ldb, + float *work, + lapack_int lwork); +lapack_int LAPACKE_dgels_work(int matrix_order, + char trans, + lapack_int m, + lapack_int n, + lapack_int nrhs, + double *a, + lapack_int lda, + double *b, + lapack_int ldb, + double *work, + lapack_int lwork); +lapack_int LAPACKE_cgels_work(int matrix_order, + char trans, + lapack_int m, + lapack_int n, + lapack_int nrhs, + lapack_complex_float *a, + lapack_int lda, + lapack_complex_float *b, + lapack_int ldb, + lapack_complex_float *work, + lapack_int lwork); +lapack_int LAPACKE_zgels_work(int matrix_order, + char trans, + lapack_int m, + lapack_int n, + lapack_int nrhs, + lapack_complex_double *a, + lapack_int lda, + lapack_complex_double *b, + lapack_int ldb, + lapack_complex_double *work, + lapack_int lwork); + +lapack_int LAPACKE_sgelsd_work(int matrix_order, + lapack_int m, + lapack_int n, + lapack_int nrhs, + float *a, + lapack_int lda, + float *b, + lapack_int ldb, + float *s, + float rcond, + lapack_int *rank, + float *work, + lapack_int lwork, + lapack_int *iwork); +lapack_int LAPACKE_dgelsd_work(int matrix_order, + lapack_int m, + lapack_int n, + lapack_int nrhs, + double *a, + lapack_int lda, + double *b, + lapack_int ldb, + double *s, + double rcond, + lapack_int *rank, + double *work, + lapack_int lwork, + lapack_int *iwork); +lapack_int LAPACKE_cgelsd_work(int matrix_order, + lapack_int m, + lapack_int n, + lapack_int nrhs, + lapack_complex_float *a, + lapack_int lda, + lapack_complex_float *b, + lapack_int ldb, + float *s, + float rcond, + lapack_int *rank, + lapack_complex_float *work, + lapack_int lwork, + float *rwork, + lapack_int *iwork); +lapack_int LAPACKE_zgelsd_work(int matrix_order, + lapack_int m, + lapack_int n, + lapack_int nrhs, + lapack_complex_double *a, + lapack_int lda, + lapack_complex_double *b, + lapack_int ldb, + double *s, + double rcond, + lapack_int *rank, + lapack_complex_double *work, + lapack_int lwork, + double *rwork, + lapack_int *iwork); + +lapack_int LAPACKE_sgelss_work(int matrix_order, + lapack_int m, + lapack_int n, + lapack_int nrhs, + float *a, + lapack_int lda, + float *b, + lapack_int ldb, + float *s, + float rcond, + lapack_int *rank, + float *work, + lapack_int lwork); +lapack_int LAPACKE_dgelss_work(int matrix_order, + lapack_int m, + lapack_int n, + lapack_int nrhs, + double *a, + lapack_int lda, + double *b, + lapack_int ldb, + double *s, + double rcond, + lapack_int *rank, + double *work, + lapack_int lwork); +lapack_int LAPACKE_cgelss_work(int matrix_order, + lapack_int m, + lapack_int n, + lapack_int nrhs, + lapack_complex_float *a, + lapack_int lda, + lapack_complex_float *b, + lapack_int ldb, + float *s, + float rcond, + lapack_int *rank, + lapack_complex_float *work, + lapack_int lwork, + float *rwork); +lapack_int LAPACKE_zgelss_work(int matrix_order, + lapack_int m, + lapack_int n, + lapack_int nrhs, + lapack_complex_double *a, + lapack_int lda, + lapack_complex_double *b, + lapack_int ldb, + double *s, + double rcond, + lapack_int *rank, + lapack_complex_double *work, + lapack_int lwork, + double *rwork); + +lapack_int LAPACKE_sgelsy_work(int matrix_order, + lapack_int m, + lapack_int n, + lapack_int nrhs, + float *a, + lapack_int lda, + float *b, + lapack_int ldb, + lapack_int *jpvt, + float rcond, + lapack_int *rank, + float *work, + lapack_int lwork); +lapack_int LAPACKE_dgelsy_work(int matrix_order, + lapack_int m, + lapack_int n, + lapack_int nrhs, + double *a, + lapack_int lda, + double *b, + lapack_int ldb, + lapack_int *jpvt, + double rcond, + lapack_int *rank, + double *work, + lapack_int lwork); +lapack_int LAPACKE_cgelsy_work(int matrix_order, + lapack_int m, + lapack_int n, + lapack_int nrhs, + lapack_complex_float *a, + lapack_int lda, + lapack_complex_float *b, + lapack_int ldb, + lapack_int *jpvt, + float rcond, + lapack_int *rank, + lapack_complex_float *work, + lapack_int lwork, + float *rwork); +lapack_int LAPACKE_zgelsy_work(int matrix_order, + lapack_int m, + lapack_int n, + lapack_int nrhs, + lapack_complex_double *a, + lapack_int lda, + lapack_complex_double *b, + lapack_int ldb, + lapack_int *jpvt, + double rcond, + lapack_int *rank, + lapack_complex_double *work, + lapack_int lwork, + double *rwork); + +lapack_int LAPACKE_sgeqlf_work(int matrix_order, + lapack_int m, + lapack_int n, + float *a, + lapack_int lda, + float *tau, + float *work, + lapack_int lwork); +lapack_int LAPACKE_dgeqlf_work(int matrix_order, + lapack_int m, + lapack_int n, + double *a, + lapack_int lda, + double *tau, + double *work, + lapack_int lwork); +lapack_int LAPACKE_cgeqlf_work(int matrix_order, + lapack_int m, + lapack_int n, + lapack_complex_float *a, + lapack_int lda, + lapack_complex_float *tau, + lapack_complex_float *work, + lapack_int lwork); +lapack_int LAPACKE_zgeqlf_work(int matrix_order, + lapack_int m, + lapack_int n, + lapack_complex_double *a, + lapack_int lda, + lapack_complex_double *tau, + lapack_complex_double *work, + lapack_int lwork); + +lapack_int LAPACKE_sgeqp3_work(int matrix_order, + lapack_int m, + lapack_int n, + float *a, + lapack_int lda, + lapack_int *jpvt, + float *tau, + float *work, + lapack_int lwork); +lapack_int LAPACKE_dgeqp3_work(int matrix_order, + lapack_int m, + lapack_int n, + double *a, + lapack_int lda, + lapack_int *jpvt, + double *tau, + double *work, + lapack_int lwork); +lapack_int LAPACKE_cgeqp3_work(int matrix_order, + lapack_int m, + lapack_int n, + lapack_complex_float *a, + lapack_int lda, + lapack_int *jpvt, + lapack_complex_float *tau, + lapack_complex_float *work, + lapack_int lwork, + float *rwork); +lapack_int LAPACKE_zgeqp3_work(int matrix_order, + lapack_int m, + lapack_int n, + lapack_complex_double *a, + lapack_int lda, + lapack_int *jpvt, + lapack_complex_double *tau, + lapack_complex_double *work, + lapack_int lwork, + double *rwork); + +lapack_int LAPACKE_sgeqpf_work(int matrix_order, + lapack_int m, + lapack_int n, + float *a, + lapack_int lda, + lapack_int *jpvt, + float *tau, + float *work); +lapack_int LAPACKE_dgeqpf_work(int matrix_order, + lapack_int m, + lapack_int n, + double *a, + lapack_int lda, + lapack_int *jpvt, + double *tau, + double *work); +lapack_int LAPACKE_cgeqpf_work(int matrix_order, + lapack_int m, + lapack_int n, + lapack_complex_float *a, + lapack_int lda, + lapack_int *jpvt, + lapack_complex_float *tau, + lapack_complex_float *work, + float *rwork); +lapack_int LAPACKE_zgeqpf_work(int matrix_order, + lapack_int m, + lapack_int n, + lapack_complex_double *a, + lapack_int lda, + lapack_int *jpvt, + lapack_complex_double *tau, + lapack_complex_double *work, + double *rwork); + +lapack_int + LAPACKE_sgeqr2_work(int matrix_order, lapack_int m, lapack_int n, float *a, lapack_int lda, float *tau, float *work); +lapack_int LAPACKE_dgeqr2_work(int matrix_order, + lapack_int m, + lapack_int n, + double *a, + lapack_int lda, + double *tau, + double *work); +lapack_int LAPACKE_cgeqr2_work(int matrix_order, + lapack_int m, + lapack_int n, + lapack_complex_float *a, + lapack_int lda, + lapack_complex_float *tau, + lapack_complex_float *work); +lapack_int LAPACKE_zgeqr2_work(int matrix_order, + lapack_int m, + lapack_int n, + lapack_complex_double *a, + lapack_int lda, + lapack_complex_double *tau, + lapack_complex_double *work); + +lapack_int LAPACKE_sgeqrf_work(int matrix_order, + lapack_int m, + lapack_int n, + float *a, + lapack_int lda, + float *tau, + float *work, + lapack_int lwork); +lapack_int LAPACKE_dgeqrf_work(int matrix_order, + lapack_int m, + lapack_int n, + double *a, + lapack_int lda, + double *tau, + double *work, + lapack_int lwork); +lapack_int LAPACKE_cgeqrf_work(int matrix_order, + lapack_int m, + lapack_int n, + lapack_complex_float *a, + lapack_int lda, + lapack_complex_float *tau, + lapack_complex_float *work, + lapack_int lwork); +lapack_int LAPACKE_zgeqrf_work(int matrix_order, + lapack_int m, + lapack_int n, + lapack_complex_double *a, + lapack_int lda, + lapack_complex_double *tau, + lapack_complex_double *work, + lapack_int lwork); + +lapack_int LAPACKE_sgeqrfp_work(int matrix_order, + lapack_int m, + lapack_int n, + float *a, + lapack_int lda, + float *tau, + float *work, + lapack_int lwork); +lapack_int LAPACKE_dgeqrfp_work(int matrix_order, + lapack_int m, + lapack_int n, + double *a, + lapack_int lda, + double *tau, + double *work, + lapack_int lwork); +lapack_int LAPACKE_cgeqrfp_work(int matrix_order, + lapack_int m, + lapack_int n, + lapack_complex_float *a, + lapack_int lda, + lapack_complex_float *tau, + lapack_complex_float *work, + lapack_int lwork); +lapack_int LAPACKE_zgeqrfp_work(int matrix_order, + lapack_int m, + lapack_int n, + lapack_complex_double *a, + lapack_int lda, + lapack_complex_double *tau, + lapack_complex_double *work, + lapack_int lwork); + +lapack_int LAPACKE_sgerfs_work(int matrix_order, + char trans, + lapack_int n, + lapack_int nrhs, + const float *a, + lapack_int lda, + const float *af, + lapack_int ldaf, + const lapack_int *ipiv, + const float *b, + lapack_int ldb, + float *x, + lapack_int ldx, + float *ferr, + float *berr, + float *work, + lapack_int *iwork); +lapack_int LAPACKE_dgerfs_work(int matrix_order, + char trans, + lapack_int n, + lapack_int nrhs, + const double *a, + lapack_int lda, + const double *af, + lapack_int ldaf, + const lapack_int *ipiv, + const double *b, + lapack_int ldb, + double *x, + lapack_int ldx, + double *ferr, + double *berr, + double *work, + lapack_int *iwork); +lapack_int LAPACKE_cgerfs_work(int matrix_order, + char trans, + lapack_int n, + lapack_int nrhs, + const lapack_complex_float *a, + lapack_int lda, + const lapack_complex_float *af, + lapack_int ldaf, + const lapack_int *ipiv, + const lapack_complex_float *b, + lapack_int ldb, + lapack_complex_float *x, + lapack_int ldx, + float *ferr, + float *berr, + lapack_complex_float *work, + float *rwork); +lapack_int LAPACKE_zgerfs_work(int matrix_order, + char trans, + lapack_int n, + lapack_int nrhs, + const lapack_complex_double *a, + lapack_int lda, + const lapack_complex_double *af, + lapack_int ldaf, + const lapack_int *ipiv, + const lapack_complex_double *b, + lapack_int ldb, + lapack_complex_double *x, + lapack_int ldx, + double *ferr, + double *berr, + lapack_complex_double *work, + double *rwork); + +lapack_int LAPACKE_sgerfsx_work(int matrix_order, + char trans, + char equed, + lapack_int n, + lapack_int nrhs, + const float *a, + lapack_int lda, + const float *af, + lapack_int ldaf, + const lapack_int *ipiv, + const float *r, + const float *c, + const float *b, + lapack_int ldb, + float *x, + lapack_int ldx, + float *rcond, + float *berr, + lapack_int n_err_bnds, + float *err_bnds_norm, + float *err_bnds_comp, + lapack_int nparams, + float *params, + float *work, + lapack_int *iwork); +lapack_int LAPACKE_dgerfsx_work(int matrix_order, + char trans, + char equed, + lapack_int n, + lapack_int nrhs, + const double *a, + lapack_int lda, + const double *af, + lapack_int ldaf, + const lapack_int *ipiv, + const double *r, + const double *c, + const double *b, + lapack_int ldb, + double *x, + lapack_int ldx, + double *rcond, + double *berr, + lapack_int n_err_bnds, + double *err_bnds_norm, + double *err_bnds_comp, + lapack_int nparams, + double *params, + double *work, + lapack_int *iwork); +lapack_int LAPACKE_cgerfsx_work(int matrix_order, + char trans, + char equed, + lapack_int n, + lapack_int nrhs, + const lapack_complex_float *a, + lapack_int lda, + const lapack_complex_float *af, + lapack_int ldaf, + const lapack_int *ipiv, + const float *r, + const float *c, + const lapack_complex_float *b, + lapack_int ldb, + lapack_complex_float *x, + lapack_int ldx, + float *rcond, + float *berr, + lapack_int n_err_bnds, + float *err_bnds_norm, + float *err_bnds_comp, + lapack_int nparams, + float *params, + lapack_complex_float *work, + float *rwork); +lapack_int LAPACKE_zgerfsx_work(int matrix_order, + char trans, + char equed, + lapack_int n, + lapack_int nrhs, + const lapack_complex_double *a, + lapack_int lda, + const lapack_complex_double *af, + lapack_int ldaf, + const lapack_int *ipiv, + const double *r, + const double *c, + const lapack_complex_double *b, + lapack_int ldb, + lapack_complex_double *x, + lapack_int ldx, + double *rcond, + double *berr, + lapack_int n_err_bnds, + double *err_bnds_norm, + double *err_bnds_comp, + lapack_int nparams, + double *params, + lapack_complex_double *work, + double *rwork); + +lapack_int LAPACKE_sgerqf_work(int matrix_order, + lapack_int m, + lapack_int n, + float *a, + lapack_int lda, + float *tau, + float *work, + lapack_int lwork); +lapack_int LAPACKE_dgerqf_work(int matrix_order, + lapack_int m, + lapack_int n, + double *a, + lapack_int lda, + double *tau, + double *work, + lapack_int lwork); +lapack_int LAPACKE_cgerqf_work(int matrix_order, + lapack_int m, + lapack_int n, + lapack_complex_float *a, + lapack_int lda, + lapack_complex_float *tau, + lapack_complex_float *work, + lapack_int lwork); +lapack_int LAPACKE_zgerqf_work(int matrix_order, + lapack_int m, + lapack_int n, + lapack_complex_double *a, + lapack_int lda, + lapack_complex_double *tau, + lapack_complex_double *work, + lapack_int lwork); + +lapack_int LAPACKE_sgesdd_work(int matrix_order, + char jobz, + lapack_int m, + lapack_int n, + float *a, + lapack_int lda, + float *s, + float *u, + lapack_int ldu, + float *vt, + lapack_int ldvt, + float *work, + lapack_int lwork, + lapack_int *iwork); +lapack_int LAPACKE_dgesdd_work(int matrix_order, + char jobz, + lapack_int m, + lapack_int n, + double *a, + lapack_int lda, + double *s, + double *u, + lapack_int ldu, + double *vt, + lapack_int ldvt, + double *work, + lapack_int lwork, + lapack_int *iwork); +lapack_int LAPACKE_cgesdd_work(int matrix_order, + char jobz, + lapack_int m, + lapack_int n, + lapack_complex_float *a, + lapack_int lda, + float *s, + lapack_complex_float *u, + lapack_int ldu, + lapack_complex_float *vt, + lapack_int ldvt, + lapack_complex_float *work, + lapack_int lwork, + float *rwork, + lapack_int *iwork); +lapack_int LAPACKE_zgesdd_work(int matrix_order, + char jobz, + lapack_int m, + lapack_int n, + lapack_complex_double *a, + lapack_int lda, + double *s, + lapack_complex_double *u, + lapack_int ldu, + lapack_complex_double *vt, + lapack_int ldvt, + lapack_complex_double *work, + lapack_int lwork, + double *rwork, + lapack_int *iwork); + +lapack_int LAPACKE_sgesv_work(int matrix_order, + lapack_int n, + lapack_int nrhs, + float *a, + lapack_int lda, + lapack_int *ipiv, + float *b, + lapack_int ldb); +lapack_int LAPACKE_dgesv_work(int matrix_order, + lapack_int n, + lapack_int nrhs, + double *a, + lapack_int lda, + lapack_int *ipiv, + double *b, + lapack_int ldb); +lapack_int LAPACKE_cgesv_work(int matrix_order, + lapack_int n, + lapack_int nrhs, + lapack_complex_float *a, + lapack_int lda, + lapack_int *ipiv, + lapack_complex_float *b, + lapack_int ldb); +lapack_int LAPACKE_zgesv_work(int matrix_order, + lapack_int n, + lapack_int nrhs, + lapack_complex_double *a, + lapack_int lda, + lapack_int *ipiv, + lapack_complex_double *b, + lapack_int ldb); +lapack_int LAPACKE_dsgesv_work(int matrix_order, + lapack_int n, + lapack_int nrhs, + double *a, + lapack_int lda, + lapack_int *ipiv, + double *b, + lapack_int ldb, + double *x, + lapack_int ldx, + double *work, + float *swork, + lapack_int *iter); +lapack_int LAPACKE_zcgesv_work(int matrix_order, + lapack_int n, + lapack_int nrhs, + lapack_complex_double *a, + lapack_int lda, + lapack_int *ipiv, + lapack_complex_double *b, + lapack_int ldb, + lapack_complex_double *x, + lapack_int ldx, + lapack_complex_double *work, + lapack_complex_float *swork, + double *rwork, + lapack_int *iter); + +lapack_int LAPACKE_sgesvd_work(int matrix_order, + char jobu, + char jobvt, + lapack_int m, + lapack_int n, + float *a, + lapack_int lda, + float *s, + float *u, + lapack_int ldu, + float *vt, + lapack_int ldvt, + float *work, + lapack_int lwork); +lapack_int LAPACKE_dgesvd_work(int matrix_order, + char jobu, + char jobvt, + lapack_int m, + lapack_int n, + double *a, + lapack_int lda, + double *s, + double *u, + lapack_int ldu, + double *vt, + lapack_int ldvt, + double *work, + lapack_int lwork); +lapack_int LAPACKE_cgesvd_work(int matrix_order, + char jobu, + char jobvt, + lapack_int m, + lapack_int n, + lapack_complex_float *a, + lapack_int lda, + float *s, + lapack_complex_float *u, + lapack_int ldu, + lapack_complex_float *vt, + lapack_int ldvt, + lapack_complex_float *work, + lapack_int lwork, + float *rwork); +lapack_int LAPACKE_zgesvd_work(int matrix_order, + char jobu, + char jobvt, + lapack_int m, + lapack_int n, + lapack_complex_double *a, + lapack_int lda, + double *s, + lapack_complex_double *u, + lapack_int ldu, + lapack_complex_double *vt, + lapack_int ldvt, + lapack_complex_double *work, + lapack_int lwork, + double *rwork); + +lapack_int LAPACKE_sgesvj_work(int matrix_order, + char joba, + char jobu, + char jobv, + lapack_int m, + lapack_int n, + float *a, + lapack_int lda, + float *sva, + lapack_int mv, + float *v, + lapack_int ldv, + float *work, + lapack_int lwork); +lapack_int LAPACKE_dgesvj_work(int matrix_order, + char joba, + char jobu, + char jobv, + lapack_int m, + lapack_int n, + double *a, + lapack_int lda, + double *sva, + lapack_int mv, + double *v, + lapack_int ldv, + double *work, + lapack_int lwork); + +lapack_int LAPACKE_sgesvx_work(int matrix_order, + char fact, + char trans, + lapack_int n, + lapack_int nrhs, + float *a, + lapack_int lda, + float *af, + lapack_int ldaf, + lapack_int *ipiv, + char *equed, + float *r, + float *c, + float *b, + lapack_int ldb, + float *x, + lapack_int ldx, + float *rcond, + float *ferr, + float *berr, + float *work, + lapack_int *iwork); +lapack_int LAPACKE_dgesvx_work(int matrix_order, + char fact, + char trans, + lapack_int n, + lapack_int nrhs, + double *a, + lapack_int lda, + double *af, + lapack_int ldaf, + lapack_int *ipiv, + char *equed, + double *r, + double *c, + double *b, + lapack_int ldb, + double *x, + lapack_int ldx, + double *rcond, + double *ferr, + double *berr, + double *work, + lapack_int *iwork); +lapack_int LAPACKE_cgesvx_work(int matrix_order, + char fact, + char trans, + lapack_int n, + lapack_int nrhs, + lapack_complex_float *a, + lapack_int lda, + lapack_complex_float *af, + lapack_int ldaf, + lapack_int *ipiv, + char *equed, + float *r, + float *c, + lapack_complex_float *b, + lapack_int ldb, + lapack_complex_float *x, + lapack_int ldx, + float *rcond, + float *ferr, + float *berr, + lapack_complex_float *work, + float *rwork); +lapack_int LAPACKE_zgesvx_work(int matrix_order, + char fact, + char trans, + lapack_int n, + lapack_int nrhs, + lapack_complex_double *a, + lapack_int lda, + lapack_complex_double *af, + lapack_int ldaf, + lapack_int *ipiv, + char *equed, + double *r, + double *c, + lapack_complex_double *b, + lapack_int ldb, + lapack_complex_double *x, + lapack_int ldx, + double *rcond, + double *ferr, + double *berr, + lapack_complex_double *work, + double *rwork); + +lapack_int LAPACKE_sgesvxx_work(int matrix_order, + char fact, + char trans, + lapack_int n, + lapack_int nrhs, + float *a, + lapack_int lda, + float *af, + lapack_int ldaf, + lapack_int *ipiv, + char *equed, + float *r, + float *c, + float *b, + lapack_int ldb, + float *x, + lapack_int ldx, + float *rcond, + float *rpvgrw, + float *berr, + lapack_int n_err_bnds, + float *err_bnds_norm, + float *err_bnds_comp, + lapack_int nparams, + float *params, + float *work, + lapack_int *iwork); +lapack_int LAPACKE_dgesvxx_work(int matrix_order, + char fact, + char trans, + lapack_int n, + lapack_int nrhs, + double *a, + lapack_int lda, + double *af, + lapack_int ldaf, + lapack_int *ipiv, + char *equed, + double *r, + double *c, + double *b, + lapack_int ldb, + double *x, + lapack_int ldx, + double *rcond, + double *rpvgrw, + double *berr, + lapack_int n_err_bnds, + double *err_bnds_norm, + double *err_bnds_comp, + lapack_int nparams, + double *params, + double *work, + lapack_int *iwork); +lapack_int LAPACKE_cgesvxx_work(int matrix_order, + char fact, + char trans, + lapack_int n, + lapack_int nrhs, + lapack_complex_float *a, + lapack_int lda, + lapack_complex_float *af, + lapack_int ldaf, + lapack_int *ipiv, + char *equed, + float *r, + float *c, + lapack_complex_float *b, + lapack_int ldb, + lapack_complex_float *x, + lapack_int ldx, + float *rcond, + float *rpvgrw, + float *berr, + lapack_int n_err_bnds, + float *err_bnds_norm, + float *err_bnds_comp, + lapack_int nparams, + float *params, + lapack_complex_float *work, + float *rwork); +lapack_int LAPACKE_zgesvxx_work(int matrix_order, + char fact, + char trans, + lapack_int n, + lapack_int nrhs, + lapack_complex_double *a, + lapack_int lda, + lapack_complex_double *af, + lapack_int ldaf, + lapack_int *ipiv, + char *equed, + double *r, + double *c, + lapack_complex_double *b, + lapack_int ldb, + lapack_complex_double *x, + lapack_int ldx, + double *rcond, + double *rpvgrw, + double *berr, + lapack_int n_err_bnds, + double *err_bnds_norm, + double *err_bnds_comp, + lapack_int nparams, + double *params, + lapack_complex_double *work, + double *rwork); + +lapack_int + LAPACKE_sgetf2_work(int matrix_order, lapack_int m, lapack_int n, float *a, lapack_int lda, lapack_int *ipiv); +lapack_int + LAPACKE_dgetf2_work(int matrix_order, lapack_int m, lapack_int n, double *a, lapack_int lda, lapack_int *ipiv); +lapack_int LAPACKE_cgetf2_work(int matrix_order, + lapack_int m, + lapack_int n, + lapack_complex_float *a, + lapack_int lda, + lapack_int *ipiv); +lapack_int LAPACKE_zgetf2_work(int matrix_order, + lapack_int m, + lapack_int n, + lapack_complex_double *a, + lapack_int lda, + lapack_int *ipiv); + +lapack_int + LAPACKE_sgetrf_work(int matrix_order, lapack_int m, lapack_int n, float *a, lapack_int lda, lapack_int *ipiv); +lapack_int + LAPACKE_dgetrf_work(int matrix_order, lapack_int m, lapack_int n, double *a, lapack_int lda, lapack_int *ipiv); +lapack_int LAPACKE_cgetrf_work(int matrix_order, + lapack_int m, + lapack_int n, + lapack_complex_float *a, + lapack_int lda, + lapack_int *ipiv); +lapack_int LAPACKE_zgetrf_work(int matrix_order, + lapack_int m, + lapack_int n, + lapack_complex_double *a, + lapack_int lda, + lapack_int *ipiv); + +lapack_int LAPACKE_sgetri_work(int matrix_order, + lapack_int n, + float *a, + lapack_int lda, + const lapack_int *ipiv, + float *work, + lapack_int lwork); +lapack_int LAPACKE_dgetri_work(int matrix_order, + lapack_int n, + double *a, + lapack_int lda, + const lapack_int *ipiv, + double *work, + lapack_int lwork); +lapack_int LAPACKE_cgetri_work(int matrix_order, + lapack_int n, + lapack_complex_float *a, + lapack_int lda, + const lapack_int *ipiv, + lapack_complex_float *work, + lapack_int lwork); +lapack_int LAPACKE_zgetri_work(int matrix_order, + lapack_int n, + lapack_complex_double *a, + lapack_int lda, + const lapack_int *ipiv, + lapack_complex_double *work, + lapack_int lwork); + +lapack_int LAPACKE_sgetrs_work(int matrix_order, + char trans, + lapack_int n, + lapack_int nrhs, + const float *a, + lapack_int lda, + const lapack_int *ipiv, + float *b, + lapack_int ldb); +lapack_int LAPACKE_dgetrs_work(int matrix_order, + char trans, + lapack_int n, + lapack_int nrhs, + const double *a, + lapack_int lda, + const lapack_int *ipiv, + double *b, + lapack_int ldb); +lapack_int LAPACKE_cgetrs_work(int matrix_order, + char trans, + lapack_int n, + lapack_int nrhs, + const lapack_complex_float *a, + lapack_int lda, + const lapack_int *ipiv, + lapack_complex_float *b, + lapack_int ldb); +lapack_int LAPACKE_zgetrs_work(int matrix_order, + char trans, + lapack_int n, + lapack_int nrhs, + const lapack_complex_double *a, + lapack_int lda, + const lapack_int *ipiv, + lapack_complex_double *b, + lapack_int ldb); + +lapack_int LAPACKE_sggbak_work(int matrix_order, + char job, + char side, + lapack_int n, + lapack_int ilo, + lapack_int ihi, + const float *lscale, + const float *rscale, + lapack_int m, + float *v, + lapack_int ldv); +lapack_int LAPACKE_dggbak_work(int matrix_order, + char job, + char side, + lapack_int n, + lapack_int ilo, + lapack_int ihi, + const double *lscale, + const double *rscale, + lapack_int m, + double *v, + lapack_int ldv); +lapack_int LAPACKE_cggbak_work(int matrix_order, + char job, + char side, + lapack_int n, + lapack_int ilo, + lapack_int ihi, + const float *lscale, + const float *rscale, + lapack_int m, + lapack_complex_float *v, + lapack_int ldv); +lapack_int LAPACKE_zggbak_work(int matrix_order, + char job, + char side, + lapack_int n, + lapack_int ilo, + lapack_int ihi, + const double *lscale, + const double *rscale, + lapack_int m, + lapack_complex_double *v, + lapack_int ldv); + +lapack_int LAPACKE_sggbal_work(int matrix_order, + char job, + lapack_int n, + float *a, + lapack_int lda, + float *b, + lapack_int ldb, + lapack_int *ilo, + lapack_int *ihi, + float *lscale, + float *rscale, + float *work); +lapack_int LAPACKE_dggbal_work(int matrix_order, + char job, + lapack_int n, + double *a, + lapack_int lda, + double *b, + lapack_int ldb, + lapack_int *ilo, + lapack_int *ihi, + double *lscale, + double *rscale, + double *work); +lapack_int LAPACKE_cggbal_work(int matrix_order, + char job, + lapack_int n, + lapack_complex_float *a, + lapack_int lda, + lapack_complex_float *b, + lapack_int ldb, + lapack_int *ilo, + lapack_int *ihi, + float *lscale, + float *rscale, + float *work); +lapack_int LAPACKE_zggbal_work(int matrix_order, + char job, + lapack_int n, + lapack_complex_double *a, + lapack_int lda, + lapack_complex_double *b, + lapack_int ldb, + lapack_int *ilo, + lapack_int *ihi, + double *lscale, + double *rscale, + double *work); + +lapack_int LAPACKE_sgges_work(int matrix_order, + char jobvsl, + char jobvsr, + char sort, + LAPACK_S_SELECT3 selctg, + lapack_int n, + float *a, + lapack_int lda, + float *b, + lapack_int ldb, + lapack_int *sdim, + float *alphar, + float *alphai, + float *beta, + float *vsl, + lapack_int ldvsl, + float *vsr, + lapack_int ldvsr, + float *work, + lapack_int lwork, + lapack_logical *bwork); +lapack_int LAPACKE_dgges_work(int matrix_order, + char jobvsl, + char jobvsr, + char sort, + LAPACK_D_SELECT3 selctg, + lapack_int n, + double *a, + lapack_int lda, + double *b, + lapack_int ldb, + lapack_int *sdim, + double *alphar, + double *alphai, + double *beta, + double *vsl, + lapack_int ldvsl, + double *vsr, + lapack_int ldvsr, + double *work, + lapack_int lwork, + lapack_logical *bwork); +lapack_int LAPACKE_cgges_work(int matrix_order, + char jobvsl, + char jobvsr, + char sort, + LAPACK_C_SELECT2 selctg, + lapack_int n, + lapack_complex_float *a, + lapack_int lda, + lapack_complex_float *b, + lapack_int ldb, + lapack_int *sdim, + lapack_complex_float *alpha, + lapack_complex_float *beta, + lapack_complex_float *vsl, + lapack_int ldvsl, + lapack_complex_float *vsr, + lapack_int ldvsr, + lapack_complex_float *work, + lapack_int lwork, + float *rwork, + lapack_logical *bwork); +lapack_int LAPACKE_zgges_work(int matrix_order, + char jobvsl, + char jobvsr, + char sort, + LAPACK_Z_SELECT2 selctg, + lapack_int n, + lapack_complex_double *a, + lapack_int lda, + lapack_complex_double *b, + lapack_int ldb, + lapack_int *sdim, + lapack_complex_double *alpha, + lapack_complex_double *beta, + lapack_complex_double *vsl, + lapack_int ldvsl, + lapack_complex_double *vsr, + lapack_int ldvsr, + lapack_complex_double *work, + lapack_int lwork, + double *rwork, + lapack_logical *bwork); + +lapack_int LAPACKE_sggesx_work(int matrix_order, + char jobvsl, + char jobvsr, + char sort, + LAPACK_S_SELECT3 selctg, + char sense, + lapack_int n, + float *a, + lapack_int lda, + float *b, + lapack_int ldb, + lapack_int *sdim, + float *alphar, + float *alphai, + float *beta, + float *vsl, + lapack_int ldvsl, + float *vsr, + lapack_int ldvsr, + float *rconde, + float *rcondv, + float *work, + lapack_int lwork, + lapack_int *iwork, + lapack_int liwork, + lapack_logical *bwork); +lapack_int LAPACKE_dggesx_work(int matrix_order, + char jobvsl, + char jobvsr, + char sort, + LAPACK_D_SELECT3 selctg, + char sense, + lapack_int n, + double *a, + lapack_int lda, + double *b, + lapack_int ldb, + lapack_int *sdim, + double *alphar, + double *alphai, + double *beta, + double *vsl, + lapack_int ldvsl, + double *vsr, + lapack_int ldvsr, + double *rconde, + double *rcondv, + double *work, + lapack_int lwork, + lapack_int *iwork, + lapack_int liwork, + lapack_logical *bwork); +lapack_int LAPACKE_cggesx_work(int matrix_order, + char jobvsl, + char jobvsr, + char sort, + LAPACK_C_SELECT2 selctg, + char sense, + lapack_int n, + lapack_complex_float *a, + lapack_int lda, + lapack_complex_float *b, + lapack_int ldb, + lapack_int *sdim, + lapack_complex_float *alpha, + lapack_complex_float *beta, + lapack_complex_float *vsl, + lapack_int ldvsl, + lapack_complex_float *vsr, + lapack_int ldvsr, + float *rconde, + float *rcondv, + lapack_complex_float *work, + lapack_int lwork, + float *rwork, + lapack_int *iwork, + lapack_int liwork, + lapack_logical *bwork); +lapack_int LAPACKE_zggesx_work(int matrix_order, + char jobvsl, + char jobvsr, + char sort, + LAPACK_Z_SELECT2 selctg, + char sense, + lapack_int n, + lapack_complex_double *a, + lapack_int lda, + lapack_complex_double *b, + lapack_int ldb, + lapack_int *sdim, + lapack_complex_double *alpha, + lapack_complex_double *beta, + lapack_complex_double *vsl, + lapack_int ldvsl, + lapack_complex_double *vsr, + lapack_int ldvsr, + double *rconde, + double *rcondv, + lapack_complex_double *work, + lapack_int lwork, + double *rwork, + lapack_int *iwork, + lapack_int liwork, + lapack_logical *bwork); + +lapack_int LAPACKE_sggev_work(int matrix_order, + char jobvl, + char jobvr, + lapack_int n, + float *a, + lapack_int lda, + float *b, + lapack_int ldb, + float *alphar, + float *alphai, + float *beta, + float *vl, + lapack_int ldvl, + float *vr, + lapack_int ldvr, + float *work, + lapack_int lwork); +lapack_int LAPACKE_dggev_work(int matrix_order, + char jobvl, + char jobvr, + lapack_int n, + double *a, + lapack_int lda, + double *b, + lapack_int ldb, + double *alphar, + double *alphai, + double *beta, + double *vl, + lapack_int ldvl, + double *vr, + lapack_int ldvr, + double *work, + lapack_int lwork); +lapack_int LAPACKE_cggev_work(int matrix_order, + char jobvl, + char jobvr, + lapack_int n, + lapack_complex_float *a, + lapack_int lda, + lapack_complex_float *b, + lapack_int ldb, + lapack_complex_float *alpha, + lapack_complex_float *beta, + lapack_complex_float *vl, + lapack_int ldvl, + lapack_complex_float *vr, + lapack_int ldvr, + lapack_complex_float *work, + lapack_int lwork, + float *rwork); +lapack_int LAPACKE_zggev_work(int matrix_order, + char jobvl, + char jobvr, + lapack_int n, + lapack_complex_double *a, + lapack_int lda, + lapack_complex_double *b, + lapack_int ldb, + lapack_complex_double *alpha, + lapack_complex_double *beta, + lapack_complex_double *vl, + lapack_int ldvl, + lapack_complex_double *vr, + lapack_int ldvr, + lapack_complex_double *work, + lapack_int lwork, + double *rwork); + +lapack_int LAPACKE_sggevx_work(int matrix_order, + char balanc, + char jobvl, + char jobvr, + char sense, + lapack_int n, + float *a, + lapack_int lda, + float *b, + lapack_int ldb, + float *alphar, + float *alphai, + float *beta, + float *vl, + lapack_int ldvl, + float *vr, + lapack_int ldvr, + lapack_int *ilo, + lapack_int *ihi, + float *lscale, + float *rscale, + float *abnrm, + float *bbnrm, + float *rconde, + float *rcondv, + float *work, + lapack_int lwork, + lapack_int *iwork, + lapack_logical *bwork); +lapack_int LAPACKE_dggevx_work(int matrix_order, + char balanc, + char jobvl, + char jobvr, + char sense, + lapack_int n, + double *a, + lapack_int lda, + double *b, + lapack_int ldb, + double *alphar, + double *alphai, + double *beta, + double *vl, + lapack_int ldvl, + double *vr, + lapack_int ldvr, + lapack_int *ilo, + lapack_int *ihi, + double *lscale, + double *rscale, + double *abnrm, + double *bbnrm, + double *rconde, + double *rcondv, + double *work, + lapack_int lwork, + lapack_int *iwork, + lapack_logical *bwork); +lapack_int LAPACKE_cggevx_work(int matrix_order, + char balanc, + char jobvl, + char jobvr, + char sense, + lapack_int n, + lapack_complex_float *a, + lapack_int lda, + lapack_complex_float *b, + lapack_int ldb, + lapack_complex_float *alpha, + lapack_complex_float *beta, + lapack_complex_float *vl, + lapack_int ldvl, + lapack_complex_float *vr, + lapack_int ldvr, + lapack_int *ilo, + lapack_int *ihi, + float *lscale, + float *rscale, + float *abnrm, + float *bbnrm, + float *rconde, + float *rcondv, + lapack_complex_float *work, + lapack_int lwork, + float *rwork, + lapack_int *iwork, + lapack_logical *bwork); +lapack_int LAPACKE_zggevx_work(int matrix_order, + char balanc, + char jobvl, + char jobvr, + char sense, + lapack_int n, + lapack_complex_double *a, + lapack_int lda, + lapack_complex_double *b, + lapack_int ldb, + lapack_complex_double *alpha, + lapack_complex_double *beta, + lapack_complex_double *vl, + lapack_int ldvl, + lapack_complex_double *vr, + lapack_int ldvr, + lapack_int *ilo, + lapack_int *ihi, + double *lscale, + double *rscale, + double *abnrm, + double *bbnrm, + double *rconde, + double *rcondv, + lapack_complex_double *work, + lapack_int lwork, + double *rwork, + lapack_int *iwork, + lapack_logical *bwork); + +lapack_int LAPACKE_sggglm_work(int matrix_order, + lapack_int n, + lapack_int m, + lapack_int p, + float *a, + lapack_int lda, + float *b, + lapack_int ldb, + float *d, + float *x, + float *y, + float *work, + lapack_int lwork); +lapack_int LAPACKE_dggglm_work(int matrix_order, + lapack_int n, + lapack_int m, + lapack_int p, + double *a, + lapack_int lda, + double *b, + lapack_int ldb, + double *d, + double *x, + double *y, + double *work, + lapack_int lwork); +lapack_int LAPACKE_cggglm_work(int matrix_order, + lapack_int n, + lapack_int m, + lapack_int p, + lapack_complex_float *a, + lapack_int lda, + lapack_complex_float *b, + lapack_int ldb, + lapack_complex_float *d, + lapack_complex_float *x, + lapack_complex_float *y, + lapack_complex_float *work, + lapack_int lwork); +lapack_int LAPACKE_zggglm_work(int matrix_order, + lapack_int n, + lapack_int m, + lapack_int p, + lapack_complex_double *a, + lapack_int lda, + lapack_complex_double *b, + lapack_int ldb, + lapack_complex_double *d, + lapack_complex_double *x, + lapack_complex_double *y, + lapack_complex_double *work, + lapack_int lwork); + +lapack_int LAPACKE_sgghrd_work(int matrix_order, + char compq, + char compz, + lapack_int n, + lapack_int ilo, + lapack_int ihi, + float *a, + lapack_int lda, + float *b, + lapack_int ldb, + float *q, + lapack_int ldq, + float *z, + lapack_int ldz); +lapack_int LAPACKE_dgghrd_work(int matrix_order, + char compq, + char compz, + lapack_int n, + lapack_int ilo, + lapack_int ihi, + double *a, + lapack_int lda, + double *b, + lapack_int ldb, + double *q, + lapack_int ldq, + double *z, + lapack_int ldz); +lapack_int LAPACKE_cgghrd_work(int matrix_order, + char compq, + char compz, + lapack_int n, + lapack_int ilo, + lapack_int ihi, + lapack_complex_float *a, + lapack_int lda, + lapack_complex_float *b, + lapack_int ldb, + lapack_complex_float *q, + lapack_int ldq, + lapack_complex_float *z, + lapack_int ldz); +lapack_int LAPACKE_zgghrd_work(int matrix_order, + char compq, + char compz, + lapack_int n, + lapack_int ilo, + lapack_int ihi, + lapack_complex_double *a, + lapack_int lda, + lapack_complex_double *b, + lapack_int ldb, + lapack_complex_double *q, + lapack_int ldq, + lapack_complex_double *z, + lapack_int ldz); + +lapack_int LAPACKE_sgglse_work(int matrix_order, + lapack_int m, + lapack_int n, + lapack_int p, + float *a, + lapack_int lda, + float *b, + lapack_int ldb, + float *c, + float *d, + float *x, + float *work, + lapack_int lwork); +lapack_int LAPACKE_dgglse_work(int matrix_order, + lapack_int m, + lapack_int n, + lapack_int p, + double *a, + lapack_int lda, + double *b, + lapack_int ldb, + double *c, + double *d, + double *x, + double *work, + lapack_int lwork); +lapack_int LAPACKE_cgglse_work(int matrix_order, + lapack_int m, + lapack_int n, + lapack_int p, + lapack_complex_float *a, + lapack_int lda, + lapack_complex_float *b, + lapack_int ldb, + lapack_complex_float *c, + lapack_complex_float *d, + lapack_complex_float *x, + lapack_complex_float *work, + lapack_int lwork); +lapack_int LAPACKE_zgglse_work(int matrix_order, + lapack_int m, + lapack_int n, + lapack_int p, + lapack_complex_double *a, + lapack_int lda, + lapack_complex_double *b, + lapack_int ldb, + lapack_complex_double *c, + lapack_complex_double *d, + lapack_complex_double *x, + lapack_complex_double *work, + lapack_int lwork); + +lapack_int LAPACKE_sggqrf_work(int matrix_order, + lapack_int n, + lapack_int m, + lapack_int p, + float *a, + lapack_int lda, + float *taua, + float *b, + lapack_int ldb, + float *taub, + float *work, + lapack_int lwork); +lapack_int LAPACKE_dggqrf_work(int matrix_order, + lapack_int n, + lapack_int m, + lapack_int p, + double *a, + lapack_int lda, + double *taua, + double *b, + lapack_int ldb, + double *taub, + double *work, + lapack_int lwork); +lapack_int LAPACKE_cggqrf_work(int matrix_order, + lapack_int n, + lapack_int m, + lapack_int p, + lapack_complex_float *a, + lapack_int lda, + lapack_complex_float *taua, + lapack_complex_float *b, + lapack_int ldb, + lapack_complex_float *taub, + lapack_complex_float *work, + lapack_int lwork); +lapack_int LAPACKE_zggqrf_work(int matrix_order, + lapack_int n, + lapack_int m, + lapack_int p, + lapack_complex_double *a, + lapack_int lda, + lapack_complex_double *taua, + lapack_complex_double *b, + lapack_int ldb, + lapack_complex_double *taub, + lapack_complex_double *work, + lapack_int lwork); + +lapack_int LAPACKE_sggrqf_work(int matrix_order, + lapack_int m, + lapack_int p, + lapack_int n, + float *a, + lapack_int lda, + float *taua, + float *b, + lapack_int ldb, + float *taub, + float *work, + lapack_int lwork); +lapack_int LAPACKE_dggrqf_work(int matrix_order, + lapack_int m, + lapack_int p, + lapack_int n, + double *a, + lapack_int lda, + double *taua, + double *b, + lapack_int ldb, + double *taub, + double *work, + lapack_int lwork); +lapack_int LAPACKE_cggrqf_work(int matrix_order, + lapack_int m, + lapack_int p, + lapack_int n, + lapack_complex_float *a, + lapack_int lda, + lapack_complex_float *taua, + lapack_complex_float *b, + lapack_int ldb, + lapack_complex_float *taub, + lapack_complex_float *work, + lapack_int lwork); +lapack_int LAPACKE_zggrqf_work(int matrix_order, + lapack_int m, + lapack_int p, + lapack_int n, + lapack_complex_double *a, + lapack_int lda, + lapack_complex_double *taua, + lapack_complex_double *b, + lapack_int ldb, + lapack_complex_double *taub, + lapack_complex_double *work, + lapack_int lwork); + +lapack_int LAPACKE_sggsvd_work(int matrix_order, + char jobu, + char jobv, + char jobq, + lapack_int m, + lapack_int n, + lapack_int p, + lapack_int *k, + lapack_int *l, + float *a, + lapack_int lda, + float *b, + lapack_int ldb, + float *alpha, + float *beta, + float *u, + lapack_int ldu, + float *v, + lapack_int ldv, + float *q, + lapack_int ldq, + float *work, + lapack_int *iwork); +lapack_int LAPACKE_dggsvd_work(int matrix_order, + char jobu, + char jobv, + char jobq, + lapack_int m, + lapack_int n, + lapack_int p, + lapack_int *k, + lapack_int *l, + double *a, + lapack_int lda, + double *b, + lapack_int ldb, + double *alpha, + double *beta, + double *u, + lapack_int ldu, + double *v, + lapack_int ldv, + double *q, + lapack_int ldq, + double *work, + lapack_int *iwork); +lapack_int LAPACKE_cggsvd_work(int matrix_order, + char jobu, + char jobv, + char jobq, + lapack_int m, + lapack_int n, + lapack_int p, + lapack_int *k, + lapack_int *l, + lapack_complex_float *a, + lapack_int lda, + lapack_complex_float *b, + lapack_int ldb, + float *alpha, + float *beta, + lapack_complex_float *u, + lapack_int ldu, + lapack_complex_float *v, + lapack_int ldv, + lapack_complex_float *q, + lapack_int ldq, + lapack_complex_float *work, + float *rwork, + lapack_int *iwork); +lapack_int LAPACKE_zggsvd_work(int matrix_order, + char jobu, + char jobv, + char jobq, + lapack_int m, + lapack_int n, + lapack_int p, + lapack_int *k, + lapack_int *l, + lapack_complex_double *a, + lapack_int lda, + lapack_complex_double *b, + lapack_int ldb, + double *alpha, + double *beta, + lapack_complex_double *u, + lapack_int ldu, + lapack_complex_double *v, + lapack_int ldv, + lapack_complex_double *q, + lapack_int ldq, + lapack_complex_double *work, + double *rwork, + lapack_int *iwork); + +lapack_int LAPACKE_sggsvp_work(int matrix_order, + char jobu, + char jobv, + char jobq, + lapack_int m, + lapack_int p, + lapack_int n, + float *a, + lapack_int lda, + float *b, + lapack_int ldb, + float tola, + float tolb, + lapack_int *k, + lapack_int *l, + float *u, + lapack_int ldu, + float *v, + lapack_int ldv, + float *q, + lapack_int ldq, + lapack_int *iwork, + float *tau, + float *work); +lapack_int LAPACKE_dggsvp_work(int matrix_order, + char jobu, + char jobv, + char jobq, + lapack_int m, + lapack_int p, + lapack_int n, + double *a, + lapack_int lda, + double *b, + lapack_int ldb, + double tola, + double tolb, + lapack_int *k, + lapack_int *l, + double *u, + lapack_int ldu, + double *v, + lapack_int ldv, + double *q, + lapack_int ldq, + lapack_int *iwork, + double *tau, + double *work); +lapack_int LAPACKE_cggsvp_work(int matrix_order, + char jobu, + char jobv, + char jobq, + lapack_int m, + lapack_int p, + lapack_int n, + lapack_complex_float *a, + lapack_int lda, + lapack_complex_float *b, + lapack_int ldb, + float tola, + float tolb, + lapack_int *k, + lapack_int *l, + lapack_complex_float *u, + lapack_int ldu, + lapack_complex_float *v, + lapack_int ldv, + lapack_complex_float *q, + lapack_int ldq, + lapack_int *iwork, + float *rwork, + lapack_complex_float *tau, + lapack_complex_float *work); +lapack_int LAPACKE_zggsvp_work(int matrix_order, + char jobu, + char jobv, + char jobq, + lapack_int m, + lapack_int p, + lapack_int n, + lapack_complex_double *a, + lapack_int lda, + lapack_complex_double *b, + lapack_int ldb, + double tola, + double tolb, + lapack_int *k, + lapack_int *l, + lapack_complex_double *u, + lapack_int ldu, + lapack_complex_double *v, + lapack_int ldv, + lapack_complex_double *q, + lapack_int ldq, + lapack_int *iwork, + double *rwork, + lapack_complex_double *tau, + lapack_complex_double *work); + +lapack_int LAPACKE_sgtcon_work(char norm, + lapack_int n, + const float *dl, + const float *d, + const float *du, + const float *du2, + const lapack_int *ipiv, + float anorm, + float *rcond, + float *work, + lapack_int *iwork); +lapack_int LAPACKE_dgtcon_work(char norm, + lapack_int n, + const double *dl, + const double *d, + const double *du, + const double *du2, + const lapack_int *ipiv, + double anorm, + double *rcond, + double *work, + lapack_int *iwork); +lapack_int LAPACKE_cgtcon_work(char norm, + lapack_int n, + const lapack_complex_float *dl, + const lapack_complex_float *d, + const lapack_complex_float *du, + const lapack_complex_float *du2, + const lapack_int *ipiv, + float anorm, + float *rcond, + lapack_complex_float *work); +lapack_int LAPACKE_zgtcon_work(char norm, + lapack_int n, + const lapack_complex_double *dl, + const lapack_complex_double *d, + const lapack_complex_double *du, + const lapack_complex_double *du2, + const lapack_int *ipiv, + double anorm, + double *rcond, + lapack_complex_double *work); + +lapack_int LAPACKE_sgtrfs_work(int matrix_order, + char trans, + lapack_int n, + lapack_int nrhs, + const float *dl, + const float *d, + const float *du, + const float *dlf, + const float *df, + const float *duf, + const float *du2, + const lapack_int *ipiv, + const float *b, + lapack_int ldb, + float *x, + lapack_int ldx, + float *ferr, + float *berr, + float *work, + lapack_int *iwork); +lapack_int LAPACKE_dgtrfs_work(int matrix_order, + char trans, + lapack_int n, + lapack_int nrhs, + const double *dl, + const double *d, + const double *du, + const double *dlf, + const double *df, + const double *duf, + const double *du2, + const lapack_int *ipiv, + const double *b, + lapack_int ldb, + double *x, + lapack_int ldx, + double *ferr, + double *berr, + double *work, + lapack_int *iwork); +lapack_int LAPACKE_cgtrfs_work(int matrix_order, + char trans, + lapack_int n, + lapack_int nrhs, + const lapack_complex_float *dl, + const lapack_complex_float *d, + const lapack_complex_float *du, + const lapack_complex_float *dlf, + const lapack_complex_float *df, + const lapack_complex_float *duf, + const lapack_complex_float *du2, + const lapack_int *ipiv, + const lapack_complex_float *b, + lapack_int ldb, + lapack_complex_float *x, + lapack_int ldx, + float *ferr, + float *berr, + lapack_complex_float *work, + float *rwork); +lapack_int LAPACKE_zgtrfs_work(int matrix_order, + char trans, + lapack_int n, + lapack_int nrhs, + const lapack_complex_double *dl, + const lapack_complex_double *d, + const lapack_complex_double *du, + const lapack_complex_double *dlf, + const lapack_complex_double *df, + const lapack_complex_double *duf, + const lapack_complex_double *du2, + const lapack_int *ipiv, + const lapack_complex_double *b, + lapack_int ldb, + lapack_complex_double *x, + lapack_int ldx, + double *ferr, + double *berr, + lapack_complex_double *work, + double *rwork); + +lapack_int LAPACKE_sgtsv_work(int matrix_order, + lapack_int n, + lapack_int nrhs, + float *dl, + float *d, + float *du, + float *b, + lapack_int ldb); +lapack_int LAPACKE_dgtsv_work(int matrix_order, + lapack_int n, + lapack_int nrhs, + double *dl, + double *d, + double *du, + double *b, + lapack_int ldb); +lapack_int LAPACKE_cgtsv_work(int matrix_order, + lapack_int n, + lapack_int nrhs, + lapack_complex_float *dl, + lapack_complex_float *d, + lapack_complex_float *du, + lapack_complex_float *b, + lapack_int ldb); +lapack_int LAPACKE_zgtsv_work(int matrix_order, + lapack_int n, + lapack_int nrhs, + lapack_complex_double *dl, + lapack_complex_double *d, + lapack_complex_double *du, + lapack_complex_double *b, + lapack_int ldb); + +lapack_int LAPACKE_sgtsvx_work(int matrix_order, + char fact, + char trans, + lapack_int n, + lapack_int nrhs, + const float *dl, + const float *d, + const float *du, + float *dlf, + float *df, + float *duf, + float *du2, + lapack_int *ipiv, + const float *b, + lapack_int ldb, + float *x, + lapack_int ldx, + float *rcond, + float *ferr, + float *berr, + float *work, + lapack_int *iwork); +lapack_int LAPACKE_dgtsvx_work(int matrix_order, + char fact, + char trans, + lapack_int n, + lapack_int nrhs, + const double *dl, + const double *d, + const double *du, + double *dlf, + double *df, + double *duf, + double *du2, + lapack_int *ipiv, + const double *b, + lapack_int ldb, + double *x, + lapack_int ldx, + double *rcond, + double *ferr, + double *berr, + double *work, + lapack_int *iwork); +lapack_int LAPACKE_cgtsvx_work(int matrix_order, + char fact, + char trans, + lapack_int n, + lapack_int nrhs, + const lapack_complex_float *dl, + const lapack_complex_float *d, + const lapack_complex_float *du, + lapack_complex_float *dlf, + lapack_complex_float *df, + lapack_complex_float *duf, + lapack_complex_float *du2, + lapack_int *ipiv, + const lapack_complex_float *b, + lapack_int ldb, + lapack_complex_float *x, + lapack_int ldx, + float *rcond, + float *ferr, + float *berr, + lapack_complex_float *work, + float *rwork); +lapack_int LAPACKE_zgtsvx_work(int matrix_order, + char fact, + char trans, + lapack_int n, + lapack_int nrhs, + const lapack_complex_double *dl, + const lapack_complex_double *d, + const lapack_complex_double *du, + lapack_complex_double *dlf, + lapack_complex_double *df, + lapack_complex_double *duf, + lapack_complex_double *du2, + lapack_int *ipiv, + const lapack_complex_double *b, + lapack_int ldb, + lapack_complex_double *x, + lapack_int ldx, + double *rcond, + double *ferr, + double *berr, + lapack_complex_double *work, + double *rwork); + +lapack_int LAPACKE_sgttrf_work(lapack_int n, float *dl, float *d, float *du, float *du2, lapack_int *ipiv); +lapack_int LAPACKE_dgttrf_work(lapack_int n, double *dl, double *d, double *du, double *du2, lapack_int *ipiv); +lapack_int LAPACKE_cgttrf_work(lapack_int n, + lapack_complex_float *dl, + lapack_complex_float *d, + lapack_complex_float *du, + lapack_complex_float *du2, + lapack_int *ipiv); +lapack_int LAPACKE_zgttrf_work(lapack_int n, + lapack_complex_double *dl, + lapack_complex_double *d, + lapack_complex_double *du, + lapack_complex_double *du2, + lapack_int *ipiv); + +lapack_int LAPACKE_sgttrs_work(int matrix_order, + char trans, + lapack_int n, + lapack_int nrhs, + const float *dl, + const float *d, + const float *du, + const float *du2, + const lapack_int *ipiv, + float *b, + lapack_int ldb); +lapack_int LAPACKE_dgttrs_work(int matrix_order, + char trans, + lapack_int n, + lapack_int nrhs, + const double *dl, + const double *d, + const double *du, + const double *du2, + const lapack_int *ipiv, + double *b, + lapack_int ldb); +lapack_int LAPACKE_cgttrs_work(int matrix_order, + char trans, + lapack_int n, + lapack_int nrhs, + const lapack_complex_float *dl, + const lapack_complex_float *d, + const lapack_complex_float *du, + const lapack_complex_float *du2, + const lapack_int *ipiv, + lapack_complex_float *b, + lapack_int ldb); +lapack_int LAPACKE_zgttrs_work(int matrix_order, + char trans, + lapack_int n, + lapack_int nrhs, + const lapack_complex_double *dl, + const lapack_complex_double *d, + const lapack_complex_double *du, + const lapack_complex_double *du2, + const lapack_int *ipiv, + lapack_complex_double *b, + lapack_int ldb); + +lapack_int LAPACKE_chbev_work(int matrix_order, + char jobz, + char uplo, + lapack_int n, + lapack_int kd, + lapack_complex_float *ab, + lapack_int ldab, + float *w, + lapack_complex_float *z, + lapack_int ldz, + lapack_complex_float *work, + float *rwork); +lapack_int LAPACKE_zhbev_work(int matrix_order, + char jobz, + char uplo, + lapack_int n, + lapack_int kd, + lapack_complex_double *ab, + lapack_int ldab, + double *w, + lapack_complex_double *z, + lapack_int ldz, + lapack_complex_double *work, + double *rwork); + +lapack_int LAPACKE_chbevd_work(int matrix_order, + char jobz, + char uplo, + lapack_int n, + lapack_int kd, + lapack_complex_float *ab, + lapack_int ldab, + float *w, + lapack_complex_float *z, + lapack_int ldz, + lapack_complex_float *work, + lapack_int lwork, + float *rwork, + lapack_int lrwork, + lapack_int *iwork, + lapack_int liwork); +lapack_int LAPACKE_zhbevd_work(int matrix_order, + char jobz, + char uplo, + lapack_int n, + lapack_int kd, + lapack_complex_double *ab, + lapack_int ldab, + double *w, + lapack_complex_double *z, + lapack_int ldz, + lapack_complex_double *work, + lapack_int lwork, + double *rwork, + lapack_int lrwork, + lapack_int *iwork, + lapack_int liwork); + +lapack_int LAPACKE_chbevx_work(int matrix_order, + char jobz, + char range, + char uplo, + lapack_int n, + lapack_int kd, + lapack_complex_float *ab, + lapack_int ldab, + lapack_complex_float *q, + lapack_int ldq, + float vl, + float vu, + lapack_int il, + lapack_int iu, + float abstol, + lapack_int *m, + float *w, + lapack_complex_float *z, + lapack_int ldz, + lapack_complex_float *work, + float *rwork, + lapack_int *iwork, + lapack_int *ifail); +lapack_int LAPACKE_zhbevx_work(int matrix_order, + char jobz, + char range, + char uplo, + lapack_int n, + lapack_int kd, + lapack_complex_double *ab, + lapack_int ldab, + lapack_complex_double *q, + lapack_int ldq, + double vl, + double vu, + lapack_int il, + lapack_int iu, + double abstol, + lapack_int *m, + double *w, + lapack_complex_double *z, + lapack_int ldz, + lapack_complex_double *work, + double *rwork, + lapack_int *iwork, + lapack_int *ifail); + +lapack_int LAPACKE_chbgst_work(int matrix_order, + char vect, + char uplo, + lapack_int n, + lapack_int ka, + lapack_int kb, + lapack_complex_float *ab, + lapack_int ldab, + const lapack_complex_float *bb, + lapack_int ldbb, + lapack_complex_float *x, + lapack_int ldx, + lapack_complex_float *work, + float *rwork); +lapack_int LAPACKE_zhbgst_work(int matrix_order, + char vect, + char uplo, + lapack_int n, + lapack_int ka, + lapack_int kb, + lapack_complex_double *ab, + lapack_int ldab, + const lapack_complex_double *bb, + lapack_int ldbb, + lapack_complex_double *x, + lapack_int ldx, + lapack_complex_double *work, + double *rwork); + +lapack_int LAPACKE_chbgv_work(int matrix_order, + char jobz, + char uplo, + lapack_int n, + lapack_int ka, + lapack_int kb, + lapack_complex_float *ab, + lapack_int ldab, + lapack_complex_float *bb, + lapack_int ldbb, + float *w, + lapack_complex_float *z, + lapack_int ldz, + lapack_complex_float *work, + float *rwork); +lapack_int LAPACKE_zhbgv_work(int matrix_order, + char jobz, + char uplo, + lapack_int n, + lapack_int ka, + lapack_int kb, + lapack_complex_double *ab, + lapack_int ldab, + lapack_complex_double *bb, + lapack_int ldbb, + double *w, + lapack_complex_double *z, + lapack_int ldz, + lapack_complex_double *work, + double *rwork); + +lapack_int LAPACKE_chbgvd_work(int matrix_order, + char jobz, + char uplo, + lapack_int n, + lapack_int ka, + lapack_int kb, + lapack_complex_float *ab, + lapack_int ldab, + lapack_complex_float *bb, + lapack_int ldbb, + float *w, + lapack_complex_float *z, + lapack_int ldz, + lapack_complex_float *work, + lapack_int lwork, + float *rwork, + lapack_int lrwork, + lapack_int *iwork, + lapack_int liwork); +lapack_int LAPACKE_zhbgvd_work(int matrix_order, + char jobz, + char uplo, + lapack_int n, + lapack_int ka, + lapack_int kb, + lapack_complex_double *ab, + lapack_int ldab, + lapack_complex_double *bb, + lapack_int ldbb, + double *w, + lapack_complex_double *z, + lapack_int ldz, + lapack_complex_double *work, + lapack_int lwork, + double *rwork, + lapack_int lrwork, + lapack_int *iwork, + lapack_int liwork); + +lapack_int LAPACKE_chbgvx_work(int matrix_order, + char jobz, + char range, + char uplo, + lapack_int n, + lapack_int ka, + lapack_int kb, + lapack_complex_float *ab, + lapack_int ldab, + lapack_complex_float *bb, + lapack_int ldbb, + lapack_complex_float *q, + lapack_int ldq, + float vl, + float vu, + lapack_int il, + lapack_int iu, + float abstol, + lapack_int *m, + float *w, + lapack_complex_float *z, + lapack_int ldz, + lapack_complex_float *work, + float *rwork, + lapack_int *iwork, + lapack_int *ifail); +lapack_int LAPACKE_zhbgvx_work(int matrix_order, + char jobz, + char range, + char uplo, + lapack_int n, + lapack_int ka, + lapack_int kb, + lapack_complex_double *ab, + lapack_int ldab, + lapack_complex_double *bb, + lapack_int ldbb, + lapack_complex_double *q, + lapack_int ldq, + double vl, + double vu, + lapack_int il, + lapack_int iu, + double abstol, + lapack_int *m, + double *w, + lapack_complex_double *z, + lapack_int ldz, + lapack_complex_double *work, + double *rwork, + lapack_int *iwork, + lapack_int *ifail); + +lapack_int LAPACKE_chbtrd_work(int matrix_order, + char vect, + char uplo, + lapack_int n, + lapack_int kd, + lapack_complex_float *ab, + lapack_int ldab, + float *d, + float *e, + lapack_complex_float *q, + lapack_int ldq, + lapack_complex_float *work); +lapack_int LAPACKE_zhbtrd_work(int matrix_order, + char vect, + char uplo, + lapack_int n, + lapack_int kd, + lapack_complex_double *ab, + lapack_int ldab, + double *d, + double *e, + lapack_complex_double *q, + lapack_int ldq, + lapack_complex_double *work); + +lapack_int LAPACKE_checon_work(int matrix_order, + char uplo, + lapack_int n, + const lapack_complex_float *a, + lapack_int lda, + const lapack_int *ipiv, + float anorm, + float *rcond, + lapack_complex_float *work); +lapack_int LAPACKE_zhecon_work(int matrix_order, + char uplo, + lapack_int n, + const lapack_complex_double *a, + lapack_int lda, + const lapack_int *ipiv, + double anorm, + double *rcond, + lapack_complex_double *work); + +lapack_int LAPACKE_cheequb_work(int matrix_order, + char uplo, + lapack_int n, + const lapack_complex_float *a, + lapack_int lda, + float *s, + float *scond, + float *amax, + lapack_complex_float *work); +lapack_int LAPACKE_zheequb_work(int matrix_order, + char uplo, + lapack_int n, + const lapack_complex_double *a, + lapack_int lda, + double *s, + double *scond, + double *amax, + lapack_complex_double *work); + +lapack_int LAPACKE_cheev_work(int matrix_order, + char jobz, + char uplo, + lapack_int n, + lapack_complex_float *a, + lapack_int lda, + float *w, + lapack_complex_float *work, + lapack_int lwork, + float *rwork); +lapack_int LAPACKE_zheev_work(int matrix_order, + char jobz, + char uplo, + lapack_int n, + lapack_complex_double *a, + lapack_int lda, + double *w, + lapack_complex_double *work, + lapack_int lwork, + double *rwork); + +lapack_int LAPACKE_cheevd_work(int matrix_order, + char jobz, + char uplo, + lapack_int n, + lapack_complex_float *a, + lapack_int lda, + float *w, + lapack_complex_float *work, + lapack_int lwork, + float *rwork, + lapack_int lrwork, + lapack_int *iwork, + lapack_int liwork); +lapack_int LAPACKE_zheevd_work(int matrix_order, + char jobz, + char uplo, + lapack_int n, + lapack_complex_double *a, + lapack_int lda, + double *w, + lapack_complex_double *work, + lapack_int lwork, + double *rwork, + lapack_int lrwork, + lapack_int *iwork, + lapack_int liwork); + +lapack_int LAPACKE_cheevr_work(int matrix_order, + char jobz, + char range, + char uplo, + lapack_int n, + lapack_complex_float *a, + lapack_int lda, + float vl, + float vu, + lapack_int il, + lapack_int iu, + float abstol, + lapack_int *m, + float *w, + lapack_complex_float *z, + lapack_int ldz, + lapack_int *isuppz, + lapack_complex_float *work, + lapack_int lwork, + float *rwork, + lapack_int lrwork, + lapack_int *iwork, + lapack_int liwork); +lapack_int LAPACKE_zheevr_work(int matrix_order, + char jobz, + char range, + char uplo, + lapack_int n, + lapack_complex_double *a, + lapack_int lda, + double vl, + double vu, + lapack_int il, + lapack_int iu, + double abstol, + lapack_int *m, + double *w, + lapack_complex_double *z, + lapack_int ldz, + lapack_int *isuppz, + lapack_complex_double *work, + lapack_int lwork, + double *rwork, + lapack_int lrwork, + lapack_int *iwork, + lapack_int liwork); + +lapack_int LAPACKE_cheevx_work(int matrix_order, + char jobz, + char range, + char uplo, + lapack_int n, + lapack_complex_float *a, + lapack_int lda, + float vl, + float vu, + lapack_int il, + lapack_int iu, + float abstol, + lapack_int *m, + float *w, + lapack_complex_float *z, + lapack_int ldz, + lapack_complex_float *work, + lapack_int lwork, + float *rwork, + lapack_int *iwork, + lapack_int *ifail); +lapack_int LAPACKE_zheevx_work(int matrix_order, + char jobz, + char range, + char uplo, + lapack_int n, + lapack_complex_double *a, + lapack_int lda, + double vl, + double vu, + lapack_int il, + lapack_int iu, + double abstol, + lapack_int *m, + double *w, + lapack_complex_double *z, + lapack_int ldz, + lapack_complex_double *work, + lapack_int lwork, + double *rwork, + lapack_int *iwork, + lapack_int *ifail); + +lapack_int LAPACKE_chegst_work(int matrix_order, + lapack_int itype, + char uplo, + lapack_int n, + lapack_complex_float *a, + lapack_int lda, + const lapack_complex_float *b, + lapack_int ldb); +lapack_int LAPACKE_zhegst_work(int matrix_order, + lapack_int itype, + char uplo, + lapack_int n, + lapack_complex_double *a, + lapack_int lda, + const lapack_complex_double *b, + lapack_int ldb); + +lapack_int LAPACKE_chegv_work(int matrix_order, + lapack_int itype, + char jobz, + char uplo, + lapack_int n, + lapack_complex_float *a, + lapack_int lda, + lapack_complex_float *b, + lapack_int ldb, + float *w, + lapack_complex_float *work, + lapack_int lwork, + float *rwork); +lapack_int LAPACKE_zhegv_work(int matrix_order, + lapack_int itype, + char jobz, + char uplo, + lapack_int n, + lapack_complex_double *a, + lapack_int lda, + lapack_complex_double *b, + lapack_int ldb, + double *w, + lapack_complex_double *work, + lapack_int lwork, + double *rwork); + +lapack_int LAPACKE_chegvd_work(int matrix_order, + lapack_int itype, + char jobz, + char uplo, + lapack_int n, + lapack_complex_float *a, + lapack_int lda, + lapack_complex_float *b, + lapack_int ldb, + float *w, + lapack_complex_float *work, + lapack_int lwork, + float *rwork, + lapack_int lrwork, + lapack_int *iwork, + lapack_int liwork); +lapack_int LAPACKE_zhegvd_work(int matrix_order, + lapack_int itype, + char jobz, + char uplo, + lapack_int n, + lapack_complex_double *a, + lapack_int lda, + lapack_complex_double *b, + lapack_int ldb, + double *w, + lapack_complex_double *work, + lapack_int lwork, + double *rwork, + lapack_int lrwork, + lapack_int *iwork, + lapack_int liwork); + +lapack_int LAPACKE_chegvx_work(int matrix_order, + lapack_int itype, + char jobz, + char range, + char uplo, + lapack_int n, + lapack_complex_float *a, + lapack_int lda, + lapack_complex_float *b, + lapack_int ldb, + float vl, + float vu, + lapack_int il, + lapack_int iu, + float abstol, + lapack_int *m, + float *w, + lapack_complex_float *z, + lapack_int ldz, + lapack_complex_float *work, + lapack_int lwork, + float *rwork, + lapack_int *iwork, + lapack_int *ifail); +lapack_int LAPACKE_zhegvx_work(int matrix_order, + lapack_int itype, + char jobz, + char range, + char uplo, + lapack_int n, + lapack_complex_double *a, + lapack_int lda, + lapack_complex_double *b, + lapack_int ldb, + double vl, + double vu, + lapack_int il, + lapack_int iu, + double abstol, + lapack_int *m, + double *w, + lapack_complex_double *z, + lapack_int ldz, + lapack_complex_double *work, + lapack_int lwork, + double *rwork, + lapack_int *iwork, + lapack_int *ifail); + +lapack_int LAPACKE_cherfs_work(int matrix_order, + char uplo, + lapack_int n, + lapack_int nrhs, + const lapack_complex_float *a, + lapack_int lda, + const lapack_complex_float *af, + lapack_int ldaf, + const lapack_int *ipiv, + const lapack_complex_float *b, + lapack_int ldb, + lapack_complex_float *x, + lapack_int ldx, + float *ferr, + float *berr, + lapack_complex_float *work, + float *rwork); +lapack_int LAPACKE_zherfs_work(int matrix_order, + char uplo, + lapack_int n, + lapack_int nrhs, + const lapack_complex_double *a, + lapack_int lda, + const lapack_complex_double *af, + lapack_int ldaf, + const lapack_int *ipiv, + const lapack_complex_double *b, + lapack_int ldb, + lapack_complex_double *x, + lapack_int ldx, + double *ferr, + double *berr, + lapack_complex_double *work, + double *rwork); + +lapack_int LAPACKE_cherfsx_work(int matrix_order, + char uplo, + char equed, + lapack_int n, + lapack_int nrhs, + const lapack_complex_float *a, + lapack_int lda, + const lapack_complex_float *af, + lapack_int ldaf, + const lapack_int *ipiv, + const float *s, + const lapack_complex_float *b, + lapack_int ldb, + lapack_complex_float *x, + lapack_int ldx, + float *rcond, + float *berr, + lapack_int n_err_bnds, + float *err_bnds_norm, + float *err_bnds_comp, + lapack_int nparams, + float *params, + lapack_complex_float *work, + float *rwork); +lapack_int LAPACKE_zherfsx_work(int matrix_order, + char uplo, + char equed, + lapack_int n, + lapack_int nrhs, + const lapack_complex_double *a, + lapack_int lda, + const lapack_complex_double *af, + lapack_int ldaf, + const lapack_int *ipiv, + const double *s, + const lapack_complex_double *b, + lapack_int ldb, + lapack_complex_double *x, + lapack_int ldx, + double *rcond, + double *berr, + lapack_int n_err_bnds, + double *err_bnds_norm, + double *err_bnds_comp, + lapack_int nparams, + double *params, + lapack_complex_double *work, + double *rwork); + +lapack_int LAPACKE_chesv_work(int matrix_order, + char uplo, + lapack_int n, + lapack_int nrhs, + lapack_complex_float *a, + lapack_int lda, + lapack_int *ipiv, + lapack_complex_float *b, + lapack_int ldb, + lapack_complex_float *work, + lapack_int lwork); +lapack_int LAPACKE_zhesv_work(int matrix_order, + char uplo, + lapack_int n, + lapack_int nrhs, + lapack_complex_double *a, + lapack_int lda, + lapack_int *ipiv, + lapack_complex_double *b, + lapack_int ldb, + lapack_complex_double *work, + lapack_int lwork); + +lapack_int LAPACKE_chesvx_work(int matrix_order, + char fact, + char uplo, + lapack_int n, + lapack_int nrhs, + const lapack_complex_float *a, + lapack_int lda, + lapack_complex_float *af, + lapack_int ldaf, + lapack_int *ipiv, + const lapack_complex_float *b, + lapack_int ldb, + lapack_complex_float *x, + lapack_int ldx, + float *rcond, + float *ferr, + float *berr, + lapack_complex_float *work, + lapack_int lwork, + float *rwork); +lapack_int LAPACKE_zhesvx_work(int matrix_order, + char fact, + char uplo, + lapack_int n, + lapack_int nrhs, + const lapack_complex_double *a, + lapack_int lda, + lapack_complex_double *af, + lapack_int ldaf, + lapack_int *ipiv, + const lapack_complex_double *b, + lapack_int ldb, + lapack_complex_double *x, + lapack_int ldx, + double *rcond, + double *ferr, + double *berr, + lapack_complex_double *work, + lapack_int lwork, + double *rwork); + +lapack_int LAPACKE_chesvxx_work(int matrix_order, + char fact, + char uplo, + lapack_int n, + lapack_int nrhs, + lapack_complex_float *a, + lapack_int lda, + lapack_complex_float *af, + lapack_int ldaf, + lapack_int *ipiv, + char *equed, + float *s, + lapack_complex_float *b, + lapack_int ldb, + lapack_complex_float *x, + lapack_int ldx, + float *rcond, + float *rpvgrw, + float *berr, + lapack_int n_err_bnds, + float *err_bnds_norm, + float *err_bnds_comp, + lapack_int nparams, + float *params, + lapack_complex_float *work, + float *rwork); +lapack_int LAPACKE_zhesvxx_work(int matrix_order, + char fact, + char uplo, + lapack_int n, + lapack_int nrhs, + lapack_complex_double *a, + lapack_int lda, + lapack_complex_double *af, + lapack_int ldaf, + lapack_int *ipiv, + char *equed, + double *s, + lapack_complex_double *b, + lapack_int ldb, + lapack_complex_double *x, + lapack_int ldx, + double *rcond, + double *rpvgrw, + double *berr, + lapack_int n_err_bnds, + double *err_bnds_norm, + double *err_bnds_comp, + lapack_int nparams, + double *params, + lapack_complex_double *work, + double *rwork); + +lapack_int LAPACKE_chetrd_work(int matrix_order, + char uplo, + lapack_int n, + lapack_complex_float *a, + lapack_int lda, + float *d, + float *e, + lapack_complex_float *tau, + lapack_complex_float *work, + lapack_int lwork); +lapack_int LAPACKE_zhetrd_work(int matrix_order, + char uplo, + lapack_int n, + lapack_complex_double *a, + lapack_int lda, + double *d, + double *e, + lapack_complex_double *tau, + lapack_complex_double *work, + lapack_int lwork); + +lapack_int LAPACKE_chetrf_work(int matrix_order, + char uplo, + lapack_int n, + lapack_complex_float *a, + lapack_int lda, + lapack_int *ipiv, + lapack_complex_float *work, + lapack_int lwork); +lapack_int LAPACKE_zhetrf_work(int matrix_order, + char uplo, + lapack_int n, + lapack_complex_double *a, + lapack_int lda, + lapack_int *ipiv, + lapack_complex_double *work, + lapack_int lwork); + +lapack_int LAPACKE_chetri_work(int matrix_order, + char uplo, + lapack_int n, + lapack_complex_float *a, + lapack_int lda, + const lapack_int *ipiv, + lapack_complex_float *work); +lapack_int LAPACKE_zhetri_work(int matrix_order, + char uplo, + lapack_int n, + lapack_complex_double *a, + lapack_int lda, + const lapack_int *ipiv, + lapack_complex_double *work); + +lapack_int LAPACKE_chetrs_work(int matrix_order, + char uplo, + lapack_int n, + lapack_int nrhs, + const lapack_complex_float *a, + lapack_int lda, + const lapack_int *ipiv, + lapack_complex_float *b, + lapack_int ldb); +lapack_int LAPACKE_zhetrs_work(int matrix_order, + char uplo, + lapack_int n, + lapack_int nrhs, + const lapack_complex_double *a, + lapack_int lda, + const lapack_int *ipiv, + lapack_complex_double *b, + lapack_int ldb); + +lapack_int LAPACKE_chfrk_work(int matrix_order, + char transr, + char uplo, + char trans, + lapack_int n, + lapack_int k, + float alpha, + const lapack_complex_float *a, + lapack_int lda, + float beta, + lapack_complex_float *c); +lapack_int LAPACKE_zhfrk_work(int matrix_order, + char transr, + char uplo, + char trans, + lapack_int n, + lapack_int k, + double alpha, + const lapack_complex_double *a, + lapack_int lda, + double beta, + lapack_complex_double *c); + +lapack_int LAPACKE_shgeqz_work(int matrix_order, + char job, + char compq, + char compz, + lapack_int n, + lapack_int ilo, + lapack_int ihi, + float *h, + lapack_int ldh, + float *t, + lapack_int ldt, + float *alphar, + float *alphai, + float *beta, + float *q, + lapack_int ldq, + float *z, + lapack_int ldz, + float *work, + lapack_int lwork); +lapack_int LAPACKE_dhgeqz_work(int matrix_order, + char job, + char compq, + char compz, + lapack_int n, + lapack_int ilo, + lapack_int ihi, + double *h, + lapack_int ldh, + double *t, + lapack_int ldt, + double *alphar, + double *alphai, + double *beta, + double *q, + lapack_int ldq, + double *z, + lapack_int ldz, + double *work, + lapack_int lwork); +lapack_int LAPACKE_chgeqz_work(int matrix_order, + char job, + char compq, + char compz, + lapack_int n, + lapack_int ilo, + lapack_int ihi, + lapack_complex_float *h, + lapack_int ldh, + lapack_complex_float *t, + lapack_int ldt, + lapack_complex_float *alpha, + lapack_complex_float *beta, + lapack_complex_float *q, + lapack_int ldq, + lapack_complex_float *z, + lapack_int ldz, + lapack_complex_float *work, + lapack_int lwork, + float *rwork); +lapack_int LAPACKE_zhgeqz_work(int matrix_order, + char job, + char compq, + char compz, + lapack_int n, + lapack_int ilo, + lapack_int ihi, + lapack_complex_double *h, + lapack_int ldh, + lapack_complex_double *t, + lapack_int ldt, + lapack_complex_double *alpha, + lapack_complex_double *beta, + lapack_complex_double *q, + lapack_int ldq, + lapack_complex_double *z, + lapack_int ldz, + lapack_complex_double *work, + lapack_int lwork, + double *rwork); + +lapack_int LAPACKE_chpcon_work(int matrix_order, + char uplo, + lapack_int n, + const lapack_complex_float *ap, + const lapack_int *ipiv, + float anorm, + float *rcond, + lapack_complex_float *work); +lapack_int LAPACKE_zhpcon_work(int matrix_order, + char uplo, + lapack_int n, + const lapack_complex_double *ap, + const lapack_int *ipiv, + double anorm, + double *rcond, + lapack_complex_double *work); + +lapack_int LAPACKE_chpev_work(int matrix_order, + char jobz, + char uplo, + lapack_int n, + lapack_complex_float *ap, + float *w, + lapack_complex_float *z, + lapack_int ldz, + lapack_complex_float *work, + float *rwork); +lapack_int LAPACKE_zhpev_work(int matrix_order, + char jobz, + char uplo, + lapack_int n, + lapack_complex_double *ap, + double *w, + lapack_complex_double *z, + lapack_int ldz, + lapack_complex_double *work, + double *rwork); + +lapack_int LAPACKE_chpevd_work(int matrix_order, + char jobz, + char uplo, + lapack_int n, + lapack_complex_float *ap, + float *w, + lapack_complex_float *z, + lapack_int ldz, + lapack_complex_float *work, + lapack_int lwork, + float *rwork, + lapack_int lrwork, + lapack_int *iwork, + lapack_int liwork); +lapack_int LAPACKE_zhpevd_work(int matrix_order, + char jobz, + char uplo, + lapack_int n, + lapack_complex_double *ap, + double *w, + lapack_complex_double *z, + lapack_int ldz, + lapack_complex_double *work, + lapack_int lwork, + double *rwork, + lapack_int lrwork, + lapack_int *iwork, + lapack_int liwork); + +lapack_int LAPACKE_chpevx_work(int matrix_order, + char jobz, + char range, + char uplo, + lapack_int n, + lapack_complex_float *ap, + float vl, + float vu, + lapack_int il, + lapack_int iu, + float abstol, + lapack_int *m, + float *w, + lapack_complex_float *z, + lapack_int ldz, + lapack_complex_float *work, + float *rwork, + lapack_int *iwork, + lapack_int *ifail); +lapack_int LAPACKE_zhpevx_work(int matrix_order, + char jobz, + char range, + char uplo, + lapack_int n, + lapack_complex_double *ap, + double vl, + double vu, + lapack_int il, + lapack_int iu, + double abstol, + lapack_int *m, + double *w, + lapack_complex_double *z, + lapack_int ldz, + lapack_complex_double *work, + double *rwork, + lapack_int *iwork, + lapack_int *ifail); + +lapack_int LAPACKE_chpgst_work(int matrix_order, + lapack_int itype, + char uplo, + lapack_int n, + lapack_complex_float *ap, + const lapack_complex_float *bp); +lapack_int LAPACKE_zhpgst_work(int matrix_order, + lapack_int itype, + char uplo, + lapack_int n, + lapack_complex_double *ap, + const lapack_complex_double *bp); + +lapack_int LAPACKE_chpgv_work(int matrix_order, + lapack_int itype, + char jobz, + char uplo, + lapack_int n, + lapack_complex_float *ap, + lapack_complex_float *bp, + float *w, + lapack_complex_float *z, + lapack_int ldz, + lapack_complex_float *work, + float *rwork); +lapack_int LAPACKE_zhpgv_work(int matrix_order, + lapack_int itype, + char jobz, + char uplo, + lapack_int n, + lapack_complex_double *ap, + lapack_complex_double *bp, + double *w, + lapack_complex_double *z, + lapack_int ldz, + lapack_complex_double *work, + double *rwork); + +lapack_int LAPACKE_chpgvd_work(int matrix_order, + lapack_int itype, + char jobz, + char uplo, + lapack_int n, + lapack_complex_float *ap, + lapack_complex_float *bp, + float *w, + lapack_complex_float *z, + lapack_int ldz, + lapack_complex_float *work, + lapack_int lwork, + float *rwork, + lapack_int lrwork, + lapack_int *iwork, + lapack_int liwork); +lapack_int LAPACKE_zhpgvd_work(int matrix_order, + lapack_int itype, + char jobz, + char uplo, + lapack_int n, + lapack_complex_double *ap, + lapack_complex_double *bp, + double *w, + lapack_complex_double *z, + lapack_int ldz, + lapack_complex_double *work, + lapack_int lwork, + double *rwork, + lapack_int lrwork, + lapack_int *iwork, + lapack_int liwork); + +lapack_int LAPACKE_chpgvx_work(int matrix_order, + lapack_int itype, + char jobz, + char range, + char uplo, + lapack_int n, + lapack_complex_float *ap, + lapack_complex_float *bp, + float vl, + float vu, + lapack_int il, + lapack_int iu, + float abstol, + lapack_int *m, + float *w, + lapack_complex_float *z, + lapack_int ldz, + lapack_complex_float *work, + float *rwork, + lapack_int *iwork, + lapack_int *ifail); +lapack_int LAPACKE_zhpgvx_work(int matrix_order, + lapack_int itype, + char jobz, + char range, + char uplo, + lapack_int n, + lapack_complex_double *ap, + lapack_complex_double *bp, + double vl, + double vu, + lapack_int il, + lapack_int iu, + double abstol, + lapack_int *m, + double *w, + lapack_complex_double *z, + lapack_int ldz, + lapack_complex_double *work, + double *rwork, + lapack_int *iwork, + lapack_int *ifail); + +lapack_int LAPACKE_chprfs_work(int matrix_order, + char uplo, + lapack_int n, + lapack_int nrhs, + const lapack_complex_float *ap, + const lapack_complex_float *afp, + const lapack_int *ipiv, + const lapack_complex_float *b, + lapack_int ldb, + lapack_complex_float *x, + lapack_int ldx, + float *ferr, + float *berr, + lapack_complex_float *work, + float *rwork); +lapack_int LAPACKE_zhprfs_work(int matrix_order, + char uplo, + lapack_int n, + lapack_int nrhs, + const lapack_complex_double *ap, + const lapack_complex_double *afp, + const lapack_int *ipiv, + const lapack_complex_double *b, + lapack_int ldb, + lapack_complex_double *x, + lapack_int ldx, + double *ferr, + double *berr, + lapack_complex_double *work, + double *rwork); + +lapack_int LAPACKE_chpsv_work(int matrix_order, + char uplo, + lapack_int n, + lapack_int nrhs, + lapack_complex_float *ap, + lapack_int *ipiv, + lapack_complex_float *b, + lapack_int ldb); +lapack_int LAPACKE_zhpsv_work(int matrix_order, + char uplo, + lapack_int n, + lapack_int nrhs, + lapack_complex_double *ap, + lapack_int *ipiv, + lapack_complex_double *b, + lapack_int ldb); + +lapack_int LAPACKE_chpsvx_work(int matrix_order, + char fact, + char uplo, + lapack_int n, + lapack_int nrhs, + const lapack_complex_float *ap, + lapack_complex_float *afp, + lapack_int *ipiv, + const lapack_complex_float *b, + lapack_int ldb, + lapack_complex_float *x, + lapack_int ldx, + float *rcond, + float *ferr, + float *berr, + lapack_complex_float *work, + float *rwork); +lapack_int LAPACKE_zhpsvx_work(int matrix_order, + char fact, + char uplo, + lapack_int n, + lapack_int nrhs, + const lapack_complex_double *ap, + lapack_complex_double *afp, + lapack_int *ipiv, + const lapack_complex_double *b, + lapack_int ldb, + lapack_complex_double *x, + lapack_int ldx, + double *rcond, + double *ferr, + double *berr, + lapack_complex_double *work, + double *rwork); + +lapack_int LAPACKE_chptrd_work(int matrix_order, + char uplo, + lapack_int n, + lapack_complex_float *ap, + float *d, + float *e, + lapack_complex_float *tau); +lapack_int LAPACKE_zhptrd_work(int matrix_order, + char uplo, + lapack_int n, + lapack_complex_double *ap, + double *d, + double *e, + lapack_complex_double *tau); + +lapack_int LAPACKE_chptrf_work(int matrix_order, char uplo, lapack_int n, lapack_complex_float *ap, lapack_int *ipiv); +lapack_int LAPACKE_zhptrf_work(int matrix_order, char uplo, lapack_int n, lapack_complex_double *ap, lapack_int *ipiv); + +lapack_int LAPACKE_chptri_work(int matrix_order, + char uplo, + lapack_int n, + lapack_complex_float *ap, + const lapack_int *ipiv, + lapack_complex_float *work); +lapack_int LAPACKE_zhptri_work(int matrix_order, + char uplo, + lapack_int n, + lapack_complex_double *ap, + const lapack_int *ipiv, + lapack_complex_double *work); + +lapack_int LAPACKE_chptrs_work(int matrix_order, + char uplo, + lapack_int n, + lapack_int nrhs, + const lapack_complex_float *ap, + const lapack_int *ipiv, + lapack_complex_float *b, + lapack_int ldb); +lapack_int LAPACKE_zhptrs_work(int matrix_order, + char uplo, + lapack_int n, + lapack_int nrhs, + const lapack_complex_double *ap, + const lapack_int *ipiv, + lapack_complex_double *b, + lapack_int ldb); + +lapack_int LAPACKE_shsein_work(int matrix_order, + char job, + char eigsrc, + char initv, + lapack_logical *select, + lapack_int n, + const float *h, + lapack_int ldh, + float *wr, + const float *wi, + float *vl, + lapack_int ldvl, + float *vr, + lapack_int ldvr, + lapack_int mm, + lapack_int *m, + float *work, + lapack_int *ifaill, + lapack_int *ifailr); +lapack_int LAPACKE_dhsein_work(int matrix_order, + char job, + char eigsrc, + char initv, + lapack_logical *select, + lapack_int n, + const double *h, + lapack_int ldh, + double *wr, + const double *wi, + double *vl, + lapack_int ldvl, + double *vr, + lapack_int ldvr, + lapack_int mm, + lapack_int *m, + double *work, + lapack_int *ifaill, + lapack_int *ifailr); +lapack_int LAPACKE_chsein_work(int matrix_order, + char job, + char eigsrc, + char initv, + const lapack_logical *select, + lapack_int n, + const lapack_complex_float *h, + lapack_int ldh, + lapack_complex_float *w, + lapack_complex_float *vl, + lapack_int ldvl, + lapack_complex_float *vr, + lapack_int ldvr, + lapack_int mm, + lapack_int *m, + lapack_complex_float *work, + float *rwork, + lapack_int *ifaill, + lapack_int *ifailr); +lapack_int LAPACKE_zhsein_work(int matrix_order, + char job, + char eigsrc, + char initv, + const lapack_logical *select, + lapack_int n, + const lapack_complex_double *h, + lapack_int ldh, + lapack_complex_double *w, + lapack_complex_double *vl, + lapack_int ldvl, + lapack_complex_double *vr, + lapack_int ldvr, + lapack_int mm, + lapack_int *m, + lapack_complex_double *work, + double *rwork, + lapack_int *ifaill, + lapack_int *ifailr); + +lapack_int LAPACKE_shseqr_work(int matrix_order, + char job, + char compz, + lapack_int n, + lapack_int ilo, + lapack_int ihi, + float *h, + lapack_int ldh, + float *wr, + float *wi, + float *z, + lapack_int ldz, + float *work, + lapack_int lwork); +lapack_int LAPACKE_dhseqr_work(int matrix_order, + char job, + char compz, + lapack_int n, + lapack_int ilo, + lapack_int ihi, + double *h, + lapack_int ldh, + double *wr, + double *wi, + double *z, + lapack_int ldz, + double *work, + lapack_int lwork); +lapack_int LAPACKE_chseqr_work(int matrix_order, + char job, + char compz, + lapack_int n, + lapack_int ilo, + lapack_int ihi, + lapack_complex_float *h, + lapack_int ldh, + lapack_complex_float *w, + lapack_complex_float *z, + lapack_int ldz, + lapack_complex_float *work, + lapack_int lwork); +lapack_int LAPACKE_zhseqr_work(int matrix_order, + char job, + char compz, + lapack_int n, + lapack_int ilo, + lapack_int ihi, + lapack_complex_double *h, + lapack_int ldh, + lapack_complex_double *w, + lapack_complex_double *z, + lapack_int ldz, + lapack_complex_double *work, + lapack_int lwork); + +lapack_int LAPACKE_clacgv_work(lapack_int n, lapack_complex_float *x, lapack_int incx); +lapack_int LAPACKE_zlacgv_work(lapack_int n, lapack_complex_double *x, lapack_int incx); + +lapack_int LAPACKE_slacpy_work(int matrix_order, + char uplo, + lapack_int m, + lapack_int n, + const float *a, + lapack_int lda, + float *b, + lapack_int ldb); +lapack_int LAPACKE_dlacpy_work(int matrix_order, + char uplo, + lapack_int m, + lapack_int n, + const double *a, + lapack_int lda, + double *b, + lapack_int ldb); +lapack_int LAPACKE_clacpy_work(int matrix_order, + char uplo, + lapack_int m, + lapack_int n, + const lapack_complex_float *a, + lapack_int lda, + lapack_complex_float *b, + lapack_int ldb); +lapack_int LAPACKE_zlacpy_work(int matrix_order, + char uplo, + lapack_int m, + lapack_int n, + const lapack_complex_double *a, + lapack_int lda, + lapack_complex_double *b, + lapack_int ldb); + +lapack_int LAPACKE_zlag2c_work(int matrix_order, + lapack_int m, + lapack_int n, + const lapack_complex_double *a, + lapack_int lda, + lapack_complex_float *sa, + lapack_int ldsa); + +lapack_int LAPACKE_slag2d_work(int matrix_order, + lapack_int m, + lapack_int n, + const float *sa, + lapack_int ldsa, + double *a, + lapack_int lda); + +lapack_int LAPACKE_dlag2s_work(int matrix_order, + lapack_int m, + lapack_int n, + const double *a, + lapack_int lda, + float *sa, + lapack_int ldsa); + +lapack_int LAPACKE_clag2z_work(int matrix_order, + lapack_int m, + lapack_int n, + const lapack_complex_float *sa, + lapack_int ldsa, + lapack_complex_double *a, + lapack_int lda); + +lapack_int LAPACKE_slagge_work(int matrix_order, + lapack_int m, + lapack_int n, + lapack_int kl, + lapack_int ku, + const float *d, + float *a, + lapack_int lda, + lapack_int *iseed, + float *work); +lapack_int LAPACKE_dlagge_work(int matrix_order, + lapack_int m, + lapack_int n, + lapack_int kl, + lapack_int ku, + const double *d, + double *a, + lapack_int lda, + lapack_int *iseed, + double *work); +lapack_int LAPACKE_clagge_work(int matrix_order, + lapack_int m, + lapack_int n, + lapack_int kl, + lapack_int ku, + const float *d, + lapack_complex_float *a, + lapack_int lda, + lapack_int *iseed, + lapack_complex_float *work); +lapack_int LAPACKE_zlagge_work(int matrix_order, + lapack_int m, + lapack_int n, + lapack_int kl, + lapack_int ku, + const double *d, + lapack_complex_double *a, + lapack_int lda, + lapack_int *iseed, + lapack_complex_double *work); + +lapack_int LAPACKE_claghe_work(int matrix_order, + lapack_int n, + lapack_int k, + const float *d, + lapack_complex_float *a, + lapack_int lda, + lapack_int *iseed, + lapack_complex_float *work); +lapack_int LAPACKE_zlaghe_work(int matrix_order, + lapack_int n, + lapack_int k, + const double *d, + lapack_complex_double *a, + lapack_int lda, + lapack_int *iseed, + lapack_complex_double *work); + +lapack_int LAPACKE_slagsy_work(int matrix_order, + lapack_int n, + lapack_int k, + const float *d, + float *a, + lapack_int lda, + lapack_int *iseed, + float *work); +lapack_int LAPACKE_dlagsy_work(int matrix_order, + lapack_int n, + lapack_int k, + const double *d, + double *a, + lapack_int lda, + lapack_int *iseed, + double *work); +lapack_int LAPACKE_clagsy_work(int matrix_order, + lapack_int n, + lapack_int k, + const float *d, + lapack_complex_float *a, + lapack_int lda, + lapack_int *iseed, + lapack_complex_float *work); +lapack_int LAPACKE_zlagsy_work(int matrix_order, + lapack_int n, + lapack_int k, + const double *d, + lapack_complex_double *a, + lapack_int lda, + lapack_int *iseed, + lapack_complex_double *work); + +lapack_int LAPACKE_slapmr_work(int matrix_order, + lapack_logical forwrd, + lapack_int m, + lapack_int n, + float *x, + lapack_int ldx, + lapack_int *k); +lapack_int LAPACKE_dlapmr_work(int matrix_order, + lapack_logical forwrd, + lapack_int m, + lapack_int n, + double *x, + lapack_int ldx, + lapack_int *k); +lapack_int LAPACKE_clapmr_work(int matrix_order, + lapack_logical forwrd, + lapack_int m, + lapack_int n, + lapack_complex_float *x, + lapack_int ldx, + lapack_int *k); +lapack_int LAPACKE_zlapmr_work(int matrix_order, + lapack_logical forwrd, + lapack_int m, + lapack_int n, + lapack_complex_double *x, + lapack_int ldx, + lapack_int *k); + +lapack_int LAPACKE_slartgp_work(float f, float g, float *cs, float *sn, float *r); +lapack_int LAPACKE_dlartgp_work(double f, double g, double *cs, double *sn, double *r); + +lapack_int LAPACKE_slartgs_work(float x, float y, float sigma, float *cs, float *sn); +lapack_int LAPACKE_dlartgs_work(double x, double y, double sigma, double *cs, double *sn); + +float LAPACKE_slapy2_work(float x, float y); +double LAPACKE_dlapy2_work(double x, double y); + +float LAPACKE_slapy3_work(float x, float y, float z); +double LAPACKE_dlapy3_work(double x, double y, double z); + +float LAPACKE_slamch_work(char cmach); +double LAPACKE_dlamch_work(char cmach); + +float LAPACKE_slange_work(int matrix_order, + char norm, + lapack_int m, + lapack_int n, + const float *a, + lapack_int lda, + float *work); +double LAPACKE_dlange_work(int matrix_order, + char norm, + lapack_int m, + lapack_int n, + const double *a, + lapack_int lda, + double *work); +float LAPACKE_clange_work(int matrix_order, + char norm, + lapack_int m, + lapack_int n, + const lapack_complex_float *a, + lapack_int lda, + float *work); +double LAPACKE_zlange_work(int matrix_order, + char norm, + lapack_int m, + lapack_int n, + const lapack_complex_double *a, + lapack_int lda, + double *work); + +float LAPACKE_clanhe_work(int matrix_order, + char norm, + char uplo, + lapack_int n, + const lapack_complex_float *a, + lapack_int lda, + float *work); +double LAPACKE_zlanhe_work(int matrix_order, + char norm, + char uplo, + lapack_int n, + const lapack_complex_double *a, + lapack_int lda, + double *work); + +float LAPACKE_slansy_work(int matrix_order, + char norm, + char uplo, + lapack_int n, + const float *a, + lapack_int lda, + float *work); +double LAPACKE_dlansy_work(int matrix_order, + char norm, + char uplo, + lapack_int n, + const double *a, + lapack_int lda, + double *work); +float LAPACKE_clansy_work(int matrix_order, + char norm, + char uplo, + lapack_int n, + const lapack_complex_float *a, + lapack_int lda, + float *work); +double LAPACKE_zlansy_work(int matrix_order, + char norm, + char uplo, + lapack_int n, + const lapack_complex_double *a, + lapack_int lda, + double *work); + +float LAPACKE_slantr_work(int matrix_order, + char norm, + char uplo, + char diag, + lapack_int m, + lapack_int n, + const float *a, + lapack_int lda, + float *work); +double LAPACKE_dlantr_work(int matrix_order, + char norm, + char uplo, + char diag, + lapack_int m, + lapack_int n, + const double *a, + lapack_int lda, + double *work); +float LAPACKE_clantr_work(int matrix_order, + char norm, + char uplo, + char diag, + lapack_int m, + lapack_int n, + const lapack_complex_float *a, + lapack_int lda, + float *work); +double LAPACKE_zlantr_work(int matrix_order, + char norm, + char uplo, + char diag, + lapack_int m, + lapack_int n, + const lapack_complex_double *a, + lapack_int lda, + double *work); + +lapack_int LAPACKE_slarfb_work(int matrix_order, + char side, + char trans, + char direct, + char storev, + lapack_int m, + lapack_int n, + lapack_int k, + const float *v, + lapack_int ldv, + const float *t, + lapack_int ldt, + float *c, + lapack_int ldc, + float *work, + lapack_int ldwork); +lapack_int LAPACKE_dlarfb_work(int matrix_order, + char side, + char trans, + char direct, + char storev, + lapack_int m, + lapack_int n, + lapack_int k, + const double *v, + lapack_int ldv, + const double *t, + lapack_int ldt, + double *c, + lapack_int ldc, + double *work, + lapack_int ldwork); +lapack_int LAPACKE_clarfb_work(int matrix_order, + char side, + char trans, + char direct, + char storev, + lapack_int m, + lapack_int n, + lapack_int k, + const lapack_complex_float *v, + lapack_int ldv, + const lapack_complex_float *t, + lapack_int ldt, + lapack_complex_float *c, + lapack_int ldc, + lapack_complex_float *work, + lapack_int ldwork); +lapack_int LAPACKE_zlarfb_work(int matrix_order, + char side, + char trans, + char direct, + char storev, + lapack_int m, + lapack_int n, + lapack_int k, + const lapack_complex_double *v, + lapack_int ldv, + const lapack_complex_double *t, + lapack_int ldt, + lapack_complex_double *c, + lapack_int ldc, + lapack_complex_double *work, + lapack_int ldwork); + +lapack_int LAPACKE_slarfg_work(lapack_int n, float *alpha, float *x, lapack_int incx, float *tau); +lapack_int LAPACKE_dlarfg_work(lapack_int n, double *alpha, double *x, lapack_int incx, double *tau); +lapack_int LAPACKE_clarfg_work(lapack_int n, + lapack_complex_float *alpha, + lapack_complex_float *x, + lapack_int incx, + lapack_complex_float *tau); +lapack_int LAPACKE_zlarfg_work(lapack_int n, + lapack_complex_double *alpha, + lapack_complex_double *x, + lapack_int incx, + lapack_complex_double *tau); + +lapack_int LAPACKE_slarft_work(int matrix_order, + char direct, + char storev, + lapack_int n, + lapack_int k, + const float *v, + lapack_int ldv, + const float *tau, + float *t, + lapack_int ldt); +lapack_int LAPACKE_dlarft_work(int matrix_order, + char direct, + char storev, + lapack_int n, + lapack_int k, + const double *v, + lapack_int ldv, + const double *tau, + double *t, + lapack_int ldt); +lapack_int LAPACKE_clarft_work(int matrix_order, + char direct, + char storev, + lapack_int n, + lapack_int k, + const lapack_complex_float *v, + lapack_int ldv, + const lapack_complex_float *tau, + lapack_complex_float *t, + lapack_int ldt); +lapack_int LAPACKE_zlarft_work(int matrix_order, + char direct, + char storev, + lapack_int n, + lapack_int k, + const lapack_complex_double *v, + lapack_int ldv, + const lapack_complex_double *tau, + lapack_complex_double *t, + lapack_int ldt); + +lapack_int LAPACKE_slarfx_work(int matrix_order, + char side, + lapack_int m, + lapack_int n, + const float *v, + float tau, + float *c, + lapack_int ldc, + float *work); +lapack_int LAPACKE_dlarfx_work(int matrix_order, + char side, + lapack_int m, + lapack_int n, + const double *v, + double tau, + double *c, + lapack_int ldc, + double *work); +lapack_int LAPACKE_clarfx_work(int matrix_order, + char side, + lapack_int m, + lapack_int n, + const lapack_complex_float *v, + lapack_complex_float tau, + lapack_complex_float *c, + lapack_int ldc, + lapack_complex_float *work); +lapack_int LAPACKE_zlarfx_work(int matrix_order, + char side, + lapack_int m, + lapack_int n, + const lapack_complex_double *v, + lapack_complex_double tau, + lapack_complex_double *c, + lapack_int ldc, + lapack_complex_double *work); + +lapack_int LAPACKE_slarnv_work(lapack_int idist, lapack_int *iseed, lapack_int n, float *x); +lapack_int LAPACKE_dlarnv_work(lapack_int idist, lapack_int *iseed, lapack_int n, double *x); +lapack_int LAPACKE_clarnv_work(lapack_int idist, lapack_int *iseed, lapack_int n, lapack_complex_float *x); +lapack_int LAPACKE_zlarnv_work(lapack_int idist, lapack_int *iseed, lapack_int n, lapack_complex_double *x); + +lapack_int LAPACKE_slaset_work(int matrix_order, + char uplo, + lapack_int m, + lapack_int n, + float alpha, + float beta, + float *a, + lapack_int lda); +lapack_int LAPACKE_dlaset_work(int matrix_order, + char uplo, + lapack_int m, + lapack_int n, + double alpha, + double beta, + double *a, + lapack_int lda); +lapack_int LAPACKE_claset_work(int matrix_order, + char uplo, + lapack_int m, + lapack_int n, + lapack_complex_float alpha, + lapack_complex_float beta, + lapack_complex_float *a, + lapack_int lda); +lapack_int LAPACKE_zlaset_work(int matrix_order, + char uplo, + lapack_int m, + lapack_int n, + lapack_complex_double alpha, + lapack_complex_double beta, + lapack_complex_double *a, + lapack_int lda); + +lapack_int LAPACKE_slasrt_work(char id, lapack_int n, float *d); +lapack_int LAPACKE_dlasrt_work(char id, lapack_int n, double *d); + +lapack_int LAPACKE_slaswp_work(int matrix_order, + lapack_int n, + float *a, + lapack_int lda, + lapack_int k1, + lapack_int k2, + const lapack_int *ipiv, + lapack_int incx); +lapack_int LAPACKE_dlaswp_work(int matrix_order, + lapack_int n, + double *a, + lapack_int lda, + lapack_int k1, + lapack_int k2, + const lapack_int *ipiv, + lapack_int incx); +lapack_int LAPACKE_claswp_work(int matrix_order, + lapack_int n, + lapack_complex_float *a, + lapack_int lda, + lapack_int k1, + lapack_int k2, + const lapack_int *ipiv, + lapack_int incx); +lapack_int LAPACKE_zlaswp_work(int matrix_order, + lapack_int n, + lapack_complex_double *a, + lapack_int lda, + lapack_int k1, + lapack_int k2, + const lapack_int *ipiv, + lapack_int incx); + +lapack_int LAPACKE_slatms_work(int matrix_order, + lapack_int m, + lapack_int n, + char dist, + lapack_int *iseed, + char sym, + float *d, + lapack_int mode, + float cond, + float dmax, + lapack_int kl, + lapack_int ku, + char pack, + float *a, + lapack_int lda, + float *work); +lapack_int LAPACKE_dlatms_work(int matrix_order, + lapack_int m, + lapack_int n, + char dist, + lapack_int *iseed, + char sym, + double *d, + lapack_int mode, + double cond, + double dmax, + lapack_int kl, + lapack_int ku, + char pack, + double *a, + lapack_int lda, + double *work); +lapack_int LAPACKE_clatms_work(int matrix_order, + lapack_int m, + lapack_int n, + char dist, + lapack_int *iseed, + char sym, + float *d, + lapack_int mode, + float cond, + float dmax, + lapack_int kl, + lapack_int ku, + char pack, + lapack_complex_float *a, + lapack_int lda, + lapack_complex_float *work); +lapack_int LAPACKE_zlatms_work(int matrix_order, + lapack_int m, + lapack_int n, + char dist, + lapack_int *iseed, + char sym, + double *d, + lapack_int mode, + double cond, + double dmax, + lapack_int kl, + lapack_int ku, + char pack, + lapack_complex_double *a, + lapack_int lda, + lapack_complex_double *work); + +lapack_int LAPACKE_slauum_work(int matrix_order, char uplo, lapack_int n, float *a, lapack_int lda); +lapack_int LAPACKE_dlauum_work(int matrix_order, char uplo, lapack_int n, double *a, lapack_int lda); +lapack_int LAPACKE_clauum_work(int matrix_order, char uplo, lapack_int n, lapack_complex_float *a, lapack_int lda); +lapack_int LAPACKE_zlauum_work(int matrix_order, char uplo, lapack_int n, lapack_complex_double *a, lapack_int lda); + +lapack_int LAPACKE_sopgtr_work(int matrix_order, + char uplo, + lapack_int n, + const float *ap, + const float *tau, + float *q, + lapack_int ldq, + float *work); +lapack_int LAPACKE_dopgtr_work(int matrix_order, + char uplo, + lapack_int n, + const double *ap, + const double *tau, + double *q, + lapack_int ldq, + double *work); + +lapack_int LAPACKE_sopmtr_work(int matrix_order, + char side, + char uplo, + char trans, + lapack_int m, + lapack_int n, + const float *ap, + const float *tau, + float *c, + lapack_int ldc, + float *work); +lapack_int LAPACKE_dopmtr_work(int matrix_order, + char side, + char uplo, + char trans, + lapack_int m, + lapack_int n, + const double *ap, + const double *tau, + double *c, + lapack_int ldc, + double *work); + +lapack_int LAPACKE_sorgbr_work(int matrix_order, + char vect, + lapack_int m, + lapack_int n, + lapack_int k, + float *a, + lapack_int lda, + const float *tau, + float *work, + lapack_int lwork); +lapack_int LAPACKE_dorgbr_work(int matrix_order, + char vect, + lapack_int m, + lapack_int n, + lapack_int k, + double *a, + lapack_int lda, + const double *tau, + double *work, + lapack_int lwork); + +lapack_int LAPACKE_sorghr_work(int matrix_order, + lapack_int n, + lapack_int ilo, + lapack_int ihi, + float *a, + lapack_int lda, + const float *tau, + float *work, + lapack_int lwork); +lapack_int LAPACKE_dorghr_work(int matrix_order, + lapack_int n, + lapack_int ilo, + lapack_int ihi, + double *a, + lapack_int lda, + const double *tau, + double *work, + lapack_int lwork); + +lapack_int LAPACKE_sorglq_work(int matrix_order, + lapack_int m, + lapack_int n, + lapack_int k, + float *a, + lapack_int lda, + const float *tau, + float *work, + lapack_int lwork); +lapack_int LAPACKE_dorglq_work(int matrix_order, + lapack_int m, + lapack_int n, + lapack_int k, + double *a, + lapack_int lda, + const double *tau, + double *work, + lapack_int lwork); + +lapack_int LAPACKE_sorgql_work(int matrix_order, + lapack_int m, + lapack_int n, + lapack_int k, + float *a, + lapack_int lda, + const float *tau, + float *work, + lapack_int lwork); +lapack_int LAPACKE_dorgql_work(int matrix_order, + lapack_int m, + lapack_int n, + lapack_int k, + double *a, + lapack_int lda, + const double *tau, + double *work, + lapack_int lwork); + +lapack_int LAPACKE_sorgqr_work(int matrix_order, + lapack_int m, + lapack_int n, + lapack_int k, + float *a, + lapack_int lda, + const float *tau, + float *work, + lapack_int lwork); +lapack_int LAPACKE_dorgqr_work(int matrix_order, + lapack_int m, + lapack_int n, + lapack_int k, + double *a, + lapack_int lda, + const double *tau, + double *work, + lapack_int lwork); + +lapack_int LAPACKE_sorgrq_work(int matrix_order, + lapack_int m, + lapack_int n, + lapack_int k, + float *a, + lapack_int lda, + const float *tau, + float *work, + lapack_int lwork); +lapack_int LAPACKE_dorgrq_work(int matrix_order, + lapack_int m, + lapack_int n, + lapack_int k, + double *a, + lapack_int lda, + const double *tau, + double *work, + lapack_int lwork); + +lapack_int LAPACKE_sorgtr_work(int matrix_order, + char uplo, + lapack_int n, + float *a, + lapack_int lda, + const float *tau, + float *work, + lapack_int lwork); +lapack_int LAPACKE_dorgtr_work(int matrix_order, + char uplo, + lapack_int n, + double *a, + lapack_int lda, + const double *tau, + double *work, + lapack_int lwork); + +lapack_int LAPACKE_sormbr_work(int matrix_order, + char vect, + char side, + char trans, + lapack_int m, + lapack_int n, + lapack_int k, + const float *a, + lapack_int lda, + const float *tau, + float *c, + lapack_int ldc, + float *work, + lapack_int lwork); +lapack_int LAPACKE_dormbr_work(int matrix_order, + char vect, + char side, + char trans, + lapack_int m, + lapack_int n, + lapack_int k, + const double *a, + lapack_int lda, + const double *tau, + double *c, + lapack_int ldc, + double *work, + lapack_int lwork); + +lapack_int LAPACKE_sormhr_work(int matrix_order, + char side, + char trans, + lapack_int m, + lapack_int n, + lapack_int ilo, + lapack_int ihi, + const float *a, + lapack_int lda, + const float *tau, + float *c, + lapack_int ldc, + float *work, + lapack_int lwork); +lapack_int LAPACKE_dormhr_work(int matrix_order, + char side, + char trans, + lapack_int m, + lapack_int n, + lapack_int ilo, + lapack_int ihi, + const double *a, + lapack_int lda, + const double *tau, + double *c, + lapack_int ldc, + double *work, + lapack_int lwork); + +lapack_int LAPACKE_sormlq_work(int matrix_order, + char side, + char trans, + lapack_int m, + lapack_int n, + lapack_int k, + const float *a, + lapack_int lda, + const float *tau, + float *c, + lapack_int ldc, + float *work, + lapack_int lwork); +lapack_int LAPACKE_dormlq_work(int matrix_order, + char side, + char trans, + lapack_int m, + lapack_int n, + lapack_int k, + const double *a, + lapack_int lda, + const double *tau, + double *c, + lapack_int ldc, + double *work, + lapack_int lwork); + +lapack_int LAPACKE_sormql_work(int matrix_order, + char side, + char trans, + lapack_int m, + lapack_int n, + lapack_int k, + const float *a, + lapack_int lda, + const float *tau, + float *c, + lapack_int ldc, + float *work, + lapack_int lwork); +lapack_int LAPACKE_dormql_work(int matrix_order, + char side, + char trans, + lapack_int m, + lapack_int n, + lapack_int k, + const double *a, + lapack_int lda, + const double *tau, + double *c, + lapack_int ldc, + double *work, + lapack_int lwork); + +lapack_int LAPACKE_sormqr_work(int matrix_order, + char side, + char trans, + lapack_int m, + lapack_int n, + lapack_int k, + const float *a, + lapack_int lda, + const float *tau, + float *c, + lapack_int ldc, + float *work, + lapack_int lwork); +lapack_int LAPACKE_dormqr_work(int matrix_order, + char side, + char trans, + lapack_int m, + lapack_int n, + lapack_int k, + const double *a, + lapack_int lda, + const double *tau, + double *c, + lapack_int ldc, + double *work, + lapack_int lwork); + +lapack_int LAPACKE_sormrq_work(int matrix_order, + char side, + char trans, + lapack_int m, + lapack_int n, + lapack_int k, + const float *a, + lapack_int lda, + const float *tau, + float *c, + lapack_int ldc, + float *work, + lapack_int lwork); +lapack_int LAPACKE_dormrq_work(int matrix_order, + char side, + char trans, + lapack_int m, + lapack_int n, + lapack_int k, + const double *a, + lapack_int lda, + const double *tau, + double *c, + lapack_int ldc, + double *work, + lapack_int lwork); + +lapack_int LAPACKE_sormrz_work(int matrix_order, + char side, + char trans, + lapack_int m, + lapack_int n, + lapack_int k, + lapack_int l, + const float *a, + lapack_int lda, + const float *tau, + float *c, + lapack_int ldc, + float *work, + lapack_int lwork); +lapack_int LAPACKE_dormrz_work(int matrix_order, + char side, + char trans, + lapack_int m, + lapack_int n, + lapack_int k, + lapack_int l, + const double *a, + lapack_int lda, + const double *tau, + double *c, + lapack_int ldc, + double *work, + lapack_int lwork); + +lapack_int LAPACKE_sormtr_work(int matrix_order, + char side, + char uplo, + char trans, + lapack_int m, + lapack_int n, + const float *a, + lapack_int lda, + const float *tau, + float *c, + lapack_int ldc, + float *work, + lapack_int lwork); +lapack_int LAPACKE_dormtr_work(int matrix_order, + char side, + char uplo, + char trans, + lapack_int m, + lapack_int n, + const double *a, + lapack_int lda, + const double *tau, + double *c, + lapack_int ldc, + double *work, + lapack_int lwork); + +lapack_int LAPACKE_spbcon_work(int matrix_order, + char uplo, + lapack_int n, + lapack_int kd, + const float *ab, + lapack_int ldab, + float anorm, + float *rcond, + float *work, + lapack_int *iwork); +lapack_int LAPACKE_dpbcon_work(int matrix_order, + char uplo, + lapack_int n, + lapack_int kd, + const double *ab, + lapack_int ldab, + double anorm, + double *rcond, + double *work, + lapack_int *iwork); +lapack_int LAPACKE_cpbcon_work(int matrix_order, + char uplo, + lapack_int n, + lapack_int kd, + const lapack_complex_float *ab, + lapack_int ldab, + float anorm, + float *rcond, + lapack_complex_float *work, + float *rwork); +lapack_int LAPACKE_zpbcon_work(int matrix_order, + char uplo, + lapack_int n, + lapack_int kd, + const lapack_complex_double *ab, + lapack_int ldab, + double anorm, + double *rcond, + lapack_complex_double *work, + double *rwork); + +lapack_int LAPACKE_spbequ_work(int matrix_order, + char uplo, + lapack_int n, + lapack_int kd, + const float *ab, + lapack_int ldab, + float *s, + float *scond, + float *amax); +lapack_int LAPACKE_dpbequ_work(int matrix_order, + char uplo, + lapack_int n, + lapack_int kd, + const double *ab, + lapack_int ldab, + double *s, + double *scond, + double *amax); +lapack_int LAPACKE_cpbequ_work(int matrix_order, + char uplo, + lapack_int n, + lapack_int kd, + const lapack_complex_float *ab, + lapack_int ldab, + float *s, + float *scond, + float *amax); +lapack_int LAPACKE_zpbequ_work(int matrix_order, + char uplo, + lapack_int n, + lapack_int kd, + const lapack_complex_double *ab, + lapack_int ldab, + double *s, + double *scond, + double *amax); + +lapack_int LAPACKE_spbrfs_work(int matrix_order, + char uplo, + lapack_int n, + lapack_int kd, + lapack_int nrhs, + const float *ab, + lapack_int ldab, + const float *afb, + lapack_int ldafb, + const float *b, + lapack_int ldb, + float *x, + lapack_int ldx, + float *ferr, + float *berr, + float *work, + lapack_int *iwork); +lapack_int LAPACKE_dpbrfs_work(int matrix_order, + char uplo, + lapack_int n, + lapack_int kd, + lapack_int nrhs, + const double *ab, + lapack_int ldab, + const double *afb, + lapack_int ldafb, + const double *b, + lapack_int ldb, + double *x, + lapack_int ldx, + double *ferr, + double *berr, + double *work, + lapack_int *iwork); +lapack_int LAPACKE_cpbrfs_work(int matrix_order, + char uplo, + lapack_int n, + lapack_int kd, + lapack_int nrhs, + const lapack_complex_float *ab, + lapack_int ldab, + const lapack_complex_float *afb, + lapack_int ldafb, + const lapack_complex_float *b, + lapack_int ldb, + lapack_complex_float *x, + lapack_int ldx, + float *ferr, + float *berr, + lapack_complex_float *work, + float *rwork); +lapack_int LAPACKE_zpbrfs_work(int matrix_order, + char uplo, + lapack_int n, + lapack_int kd, + lapack_int nrhs, + const lapack_complex_double *ab, + lapack_int ldab, + const lapack_complex_double *afb, + lapack_int ldafb, + const lapack_complex_double *b, + lapack_int ldb, + lapack_complex_double *x, + lapack_int ldx, + double *ferr, + double *berr, + lapack_complex_double *work, + double *rwork); + +lapack_int LAPACKE_spbstf_work(int matrix_order, char uplo, lapack_int n, lapack_int kb, float *bb, lapack_int ldbb); +lapack_int LAPACKE_dpbstf_work(int matrix_order, char uplo, lapack_int n, lapack_int kb, double *bb, lapack_int ldbb); +lapack_int LAPACKE_cpbstf_work(int matrix_order, + char uplo, + lapack_int n, + lapack_int kb, + lapack_complex_float *bb, + lapack_int ldbb); +lapack_int LAPACKE_zpbstf_work(int matrix_order, + char uplo, + lapack_int n, + lapack_int kb, + lapack_complex_double *bb, + lapack_int ldbb); + +lapack_int LAPACKE_spbsv_work(int matrix_order, + char uplo, + lapack_int n, + lapack_int kd, + lapack_int nrhs, + float *ab, + lapack_int ldab, + float *b, + lapack_int ldb); +lapack_int LAPACKE_dpbsv_work(int matrix_order, + char uplo, + lapack_int n, + lapack_int kd, + lapack_int nrhs, + double *ab, + lapack_int ldab, + double *b, + lapack_int ldb); +lapack_int LAPACKE_cpbsv_work(int matrix_order, + char uplo, + lapack_int n, + lapack_int kd, + lapack_int nrhs, + lapack_complex_float *ab, + lapack_int ldab, + lapack_complex_float *b, + lapack_int ldb); +lapack_int LAPACKE_zpbsv_work(int matrix_order, + char uplo, + lapack_int n, + lapack_int kd, + lapack_int nrhs, + lapack_complex_double *ab, + lapack_int ldab, + lapack_complex_double *b, + lapack_int ldb); + +lapack_int LAPACKE_spbsvx_work(int matrix_order, + char fact, + char uplo, + lapack_int n, + lapack_int kd, + lapack_int nrhs, + float *ab, + lapack_int ldab, + float *afb, + lapack_int ldafb, + char *equed, + float *s, + float *b, + lapack_int ldb, + float *x, + lapack_int ldx, + float *rcond, + float *ferr, + float *berr, + float *work, + lapack_int *iwork); +lapack_int LAPACKE_dpbsvx_work(int matrix_order, + char fact, + char uplo, + lapack_int n, + lapack_int kd, + lapack_int nrhs, + double *ab, + lapack_int ldab, + double *afb, + lapack_int ldafb, + char *equed, + double *s, + double *b, + lapack_int ldb, + double *x, + lapack_int ldx, + double *rcond, + double *ferr, + double *berr, + double *work, + lapack_int *iwork); +lapack_int LAPACKE_cpbsvx_work(int matrix_order, + char fact, + char uplo, + lapack_int n, + lapack_int kd, + lapack_int nrhs, + lapack_complex_float *ab, + lapack_int ldab, + lapack_complex_float *afb, + lapack_int ldafb, + char *equed, + float *s, + lapack_complex_float *b, + lapack_int ldb, + lapack_complex_float *x, + lapack_int ldx, + float *rcond, + float *ferr, + float *berr, + lapack_complex_float *work, + float *rwork); +lapack_int LAPACKE_zpbsvx_work(int matrix_order, + char fact, + char uplo, + lapack_int n, + lapack_int kd, + lapack_int nrhs, + lapack_complex_double *ab, + lapack_int ldab, + lapack_complex_double *afb, + lapack_int ldafb, + char *equed, + double *s, + lapack_complex_double *b, + lapack_int ldb, + lapack_complex_double *x, + lapack_int ldx, + double *rcond, + double *ferr, + double *berr, + lapack_complex_double *work, + double *rwork); + +lapack_int LAPACKE_spbtrf_work(int matrix_order, char uplo, lapack_int n, lapack_int kd, float *ab, lapack_int ldab); +lapack_int LAPACKE_dpbtrf_work(int matrix_order, char uplo, lapack_int n, lapack_int kd, double *ab, lapack_int ldab); +lapack_int LAPACKE_cpbtrf_work(int matrix_order, + char uplo, + lapack_int n, + lapack_int kd, + lapack_complex_float *ab, + lapack_int ldab); +lapack_int LAPACKE_zpbtrf_work(int matrix_order, + char uplo, + lapack_int n, + lapack_int kd, + lapack_complex_double *ab, + lapack_int ldab); + +lapack_int LAPACKE_spbtrs_work(int matrix_order, + char uplo, + lapack_int n, + lapack_int kd, + lapack_int nrhs, + const float *ab, + lapack_int ldab, + float *b, + lapack_int ldb); +lapack_int LAPACKE_dpbtrs_work(int matrix_order, + char uplo, + lapack_int n, + lapack_int kd, + lapack_int nrhs, + const double *ab, + lapack_int ldab, + double *b, + lapack_int ldb); +lapack_int LAPACKE_cpbtrs_work(int matrix_order, + char uplo, + lapack_int n, + lapack_int kd, + lapack_int nrhs, + const lapack_complex_float *ab, + lapack_int ldab, + lapack_complex_float *b, + lapack_int ldb); +lapack_int LAPACKE_zpbtrs_work(int matrix_order, + char uplo, + lapack_int n, + lapack_int kd, + lapack_int nrhs, + const lapack_complex_double *ab, + lapack_int ldab, + lapack_complex_double *b, + lapack_int ldb); + +lapack_int LAPACKE_spftrf_work(int matrix_order, char transr, char uplo, lapack_int n, float *a); +lapack_int LAPACKE_dpftrf_work(int matrix_order, char transr, char uplo, lapack_int n, double *a); +lapack_int LAPACKE_cpftrf_work(int matrix_order, char transr, char uplo, lapack_int n, lapack_complex_float *a); +lapack_int LAPACKE_zpftrf_work(int matrix_order, char transr, char uplo, lapack_int n, lapack_complex_double *a); + +lapack_int LAPACKE_spftri_work(int matrix_order, char transr, char uplo, lapack_int n, float *a); +lapack_int LAPACKE_dpftri_work(int matrix_order, char transr, char uplo, lapack_int n, double *a); +lapack_int LAPACKE_cpftri_work(int matrix_order, char transr, char uplo, lapack_int n, lapack_complex_float *a); +lapack_int LAPACKE_zpftri_work(int matrix_order, char transr, char uplo, lapack_int n, lapack_complex_double *a); + +lapack_int LAPACKE_spftrs_work(int matrix_order, + char transr, + char uplo, + lapack_int n, + lapack_int nrhs, + const float *a, + float *b, + lapack_int ldb); +lapack_int LAPACKE_dpftrs_work(int matrix_order, + char transr, + char uplo, + lapack_int n, + lapack_int nrhs, + const double *a, + double *b, + lapack_int ldb); +lapack_int LAPACKE_cpftrs_work(int matrix_order, + char transr, + char uplo, + lapack_int n, + lapack_int nrhs, + const lapack_complex_float *a, + lapack_complex_float *b, + lapack_int ldb); +lapack_int LAPACKE_zpftrs_work(int matrix_order, + char transr, + char uplo, + lapack_int n, + lapack_int nrhs, + const lapack_complex_double *a, + lapack_complex_double *b, + lapack_int ldb); + +lapack_int LAPACKE_spocon_work(int matrix_order, + char uplo, + lapack_int n, + const float *a, + lapack_int lda, + float anorm, + float *rcond, + float *work, + lapack_int *iwork); +lapack_int LAPACKE_dpocon_work(int matrix_order, + char uplo, + lapack_int n, + const double *a, + lapack_int lda, + double anorm, + double *rcond, + double *work, + lapack_int *iwork); +lapack_int LAPACKE_cpocon_work(int matrix_order, + char uplo, + lapack_int n, + const lapack_complex_float *a, + lapack_int lda, + float anorm, + float *rcond, + lapack_complex_float *work, + float *rwork); +lapack_int LAPACKE_zpocon_work(int matrix_order, + char uplo, + lapack_int n, + const lapack_complex_double *a, + lapack_int lda, + double anorm, + double *rcond, + lapack_complex_double *work, + double *rwork); + +lapack_int LAPACKE_spoequ_work(int matrix_order, + lapack_int n, + const float *a, + lapack_int lda, + float *s, + float *scond, + float *amax); +lapack_int LAPACKE_dpoequ_work(int matrix_order, + lapack_int n, + const double *a, + lapack_int lda, + double *s, + double *scond, + double *amax); +lapack_int LAPACKE_cpoequ_work(int matrix_order, + lapack_int n, + const lapack_complex_float *a, + lapack_int lda, + float *s, + float *scond, + float *amax); +lapack_int LAPACKE_zpoequ_work(int matrix_order, + lapack_int n, + const lapack_complex_double *a, + lapack_int lda, + double *s, + double *scond, + double *amax); + +lapack_int LAPACKE_spoequb_work(int matrix_order, + lapack_int n, + const float *a, + lapack_int lda, + float *s, + float *scond, + float *amax); +lapack_int LAPACKE_dpoequb_work(int matrix_order, + lapack_int n, + const double *a, + lapack_int lda, + double *s, + double *scond, + double *amax); +lapack_int LAPACKE_cpoequb_work(int matrix_order, + lapack_int n, + const lapack_complex_float *a, + lapack_int lda, + float *s, + float *scond, + float *amax); +lapack_int LAPACKE_zpoequb_work(int matrix_order, + lapack_int n, + const lapack_complex_double *a, + lapack_int lda, + double *s, + double *scond, + double *amax); + +lapack_int LAPACKE_sporfs_work(int matrix_order, + char uplo, + lapack_int n, + lapack_int nrhs, + const float *a, + lapack_int lda, + const float *af, + lapack_int ldaf, + const float *b, + lapack_int ldb, + float *x, + lapack_int ldx, + float *ferr, + float *berr, + float *work, + lapack_int *iwork); +lapack_int LAPACKE_dporfs_work(int matrix_order, + char uplo, + lapack_int n, + lapack_int nrhs, + const double *a, + lapack_int lda, + const double *af, + lapack_int ldaf, + const double *b, + lapack_int ldb, + double *x, + lapack_int ldx, + double *ferr, + double *berr, + double *work, + lapack_int *iwork); +lapack_int LAPACKE_cporfs_work(int matrix_order, + char uplo, + lapack_int n, + lapack_int nrhs, + const lapack_complex_float *a, + lapack_int lda, + const lapack_complex_float *af, + lapack_int ldaf, + const lapack_complex_float *b, + lapack_int ldb, + lapack_complex_float *x, + lapack_int ldx, + float *ferr, + float *berr, + lapack_complex_float *work, + float *rwork); +lapack_int LAPACKE_zporfs_work(int matrix_order, + char uplo, + lapack_int n, + lapack_int nrhs, + const lapack_complex_double *a, + lapack_int lda, + const lapack_complex_double *af, + lapack_int ldaf, + const lapack_complex_double *b, + lapack_int ldb, + lapack_complex_double *x, + lapack_int ldx, + double *ferr, + double *berr, + lapack_complex_double *work, + double *rwork); + +lapack_int LAPACKE_sporfsx_work(int matrix_order, + char uplo, + char equed, + lapack_int n, + lapack_int nrhs, + const float *a, + lapack_int lda, + const float *af, + lapack_int ldaf, + const float *s, + const float *b, + lapack_int ldb, + float *x, + lapack_int ldx, + float *rcond, + float *berr, + lapack_int n_err_bnds, + float *err_bnds_norm, + float *err_bnds_comp, + lapack_int nparams, + float *params, + float *work, + lapack_int *iwork); +lapack_int LAPACKE_dporfsx_work(int matrix_order, + char uplo, + char equed, + lapack_int n, + lapack_int nrhs, + const double *a, + lapack_int lda, + const double *af, + lapack_int ldaf, + const double *s, + const double *b, + lapack_int ldb, + double *x, + lapack_int ldx, + double *rcond, + double *berr, + lapack_int n_err_bnds, + double *err_bnds_norm, + double *err_bnds_comp, + lapack_int nparams, + double *params, + double *work, + lapack_int *iwork); +lapack_int LAPACKE_cporfsx_work(int matrix_order, + char uplo, + char equed, + lapack_int n, + lapack_int nrhs, + const lapack_complex_float *a, + lapack_int lda, + const lapack_complex_float *af, + lapack_int ldaf, + const float *s, + const lapack_complex_float *b, + lapack_int ldb, + lapack_complex_float *x, + lapack_int ldx, + float *rcond, + float *berr, + lapack_int n_err_bnds, + float *err_bnds_norm, + float *err_bnds_comp, + lapack_int nparams, + float *params, + lapack_complex_float *work, + float *rwork); +lapack_int LAPACKE_zporfsx_work(int matrix_order, + char uplo, + char equed, + lapack_int n, + lapack_int nrhs, + const lapack_complex_double *a, + lapack_int lda, + const lapack_complex_double *af, + lapack_int ldaf, + const double *s, + const lapack_complex_double *b, + lapack_int ldb, + lapack_complex_double *x, + lapack_int ldx, + double *rcond, + double *berr, + lapack_int n_err_bnds, + double *err_bnds_norm, + double *err_bnds_comp, + lapack_int nparams, + double *params, + lapack_complex_double *work, + double *rwork); + +lapack_int LAPACKE_sposv_work(int matrix_order, + char uplo, + lapack_int n, + lapack_int nrhs, + float *a, + lapack_int lda, + float *b, + lapack_int ldb); +lapack_int LAPACKE_dposv_work(int matrix_order, + char uplo, + lapack_int n, + lapack_int nrhs, + double *a, + lapack_int lda, + double *b, + lapack_int ldb); +lapack_int LAPACKE_cposv_work(int matrix_order, + char uplo, + lapack_int n, + lapack_int nrhs, + lapack_complex_float *a, + lapack_int lda, + lapack_complex_float *b, + lapack_int ldb); +lapack_int LAPACKE_zposv_work(int matrix_order, + char uplo, + lapack_int n, + lapack_int nrhs, + lapack_complex_double *a, + lapack_int lda, + lapack_complex_double *b, + lapack_int ldb); +lapack_int LAPACKE_dsposv_work(int matrix_order, + char uplo, + lapack_int n, + lapack_int nrhs, + double *a, + lapack_int lda, + double *b, + lapack_int ldb, + double *x, + lapack_int ldx, + double *work, + float *swork, + lapack_int *iter); +lapack_int LAPACKE_zcposv_work(int matrix_order, + char uplo, + lapack_int n, + lapack_int nrhs, + lapack_complex_double *a, + lapack_int lda, + lapack_complex_double *b, + lapack_int ldb, + lapack_complex_double *x, + lapack_int ldx, + lapack_complex_double *work, + lapack_complex_float *swork, + double *rwork, + lapack_int *iter); + +lapack_int LAPACKE_sposvx_work(int matrix_order, + char fact, + char uplo, + lapack_int n, + lapack_int nrhs, + float *a, + lapack_int lda, + float *af, + lapack_int ldaf, + char *equed, + float *s, + float *b, + lapack_int ldb, + float *x, + lapack_int ldx, + float *rcond, + float *ferr, + float *berr, + float *work, + lapack_int *iwork); +lapack_int LAPACKE_dposvx_work(int matrix_order, + char fact, + char uplo, + lapack_int n, + lapack_int nrhs, + double *a, + lapack_int lda, + double *af, + lapack_int ldaf, + char *equed, + double *s, + double *b, + lapack_int ldb, + double *x, + lapack_int ldx, + double *rcond, + double *ferr, + double *berr, + double *work, + lapack_int *iwork); +lapack_int LAPACKE_cposvx_work(int matrix_order, + char fact, + char uplo, + lapack_int n, + lapack_int nrhs, + lapack_complex_float *a, + lapack_int lda, + lapack_complex_float *af, + lapack_int ldaf, + char *equed, + float *s, + lapack_complex_float *b, + lapack_int ldb, + lapack_complex_float *x, + lapack_int ldx, + float *rcond, + float *ferr, + float *berr, + lapack_complex_float *work, + float *rwork); +lapack_int LAPACKE_zposvx_work(int matrix_order, + char fact, + char uplo, + lapack_int n, + lapack_int nrhs, + lapack_complex_double *a, + lapack_int lda, + lapack_complex_double *af, + lapack_int ldaf, + char *equed, + double *s, + lapack_complex_double *b, + lapack_int ldb, + lapack_complex_double *x, + lapack_int ldx, + double *rcond, + double *ferr, + double *berr, + lapack_complex_double *work, + double *rwork); + +lapack_int LAPACKE_sposvxx_work(int matrix_order, + char fact, + char uplo, + lapack_int n, + lapack_int nrhs, + float *a, + lapack_int lda, + float *af, + lapack_int ldaf, + char *equed, + float *s, + float *b, + lapack_int ldb, + float *x, + lapack_int ldx, + float *rcond, + float *rpvgrw, + float *berr, + lapack_int n_err_bnds, + float *err_bnds_norm, + float *err_bnds_comp, + lapack_int nparams, + float *params, + float *work, + lapack_int *iwork); +lapack_int LAPACKE_dposvxx_work(int matrix_order, + char fact, + char uplo, + lapack_int n, + lapack_int nrhs, + double *a, + lapack_int lda, + double *af, + lapack_int ldaf, + char *equed, + double *s, + double *b, + lapack_int ldb, + double *x, + lapack_int ldx, + double *rcond, + double *rpvgrw, + double *berr, + lapack_int n_err_bnds, + double *err_bnds_norm, + double *err_bnds_comp, + lapack_int nparams, + double *params, + double *work, + lapack_int *iwork); +lapack_int LAPACKE_cposvxx_work(int matrix_order, + char fact, + char uplo, + lapack_int n, + lapack_int nrhs, + lapack_complex_float *a, + lapack_int lda, + lapack_complex_float *af, + lapack_int ldaf, + char *equed, + float *s, + lapack_complex_float *b, + lapack_int ldb, + lapack_complex_float *x, + lapack_int ldx, + float *rcond, + float *rpvgrw, + float *berr, + lapack_int n_err_bnds, + float *err_bnds_norm, + float *err_bnds_comp, + lapack_int nparams, + float *params, + lapack_complex_float *work, + float *rwork); +lapack_int LAPACKE_zposvxx_work(int matrix_order, + char fact, + char uplo, + lapack_int n, + lapack_int nrhs, + lapack_complex_double *a, + lapack_int lda, + lapack_complex_double *af, + lapack_int ldaf, + char *equed, + double *s, + lapack_complex_double *b, + lapack_int ldb, + lapack_complex_double *x, + lapack_int ldx, + double *rcond, + double *rpvgrw, + double *berr, + lapack_int n_err_bnds, + double *err_bnds_norm, + double *err_bnds_comp, + lapack_int nparams, + double *params, + lapack_complex_double *work, + double *rwork); + +lapack_int LAPACKE_spotrf_work(int matrix_order, char uplo, lapack_int n, float *a, lapack_int lda); +lapack_int LAPACKE_dpotrf_work(int matrix_order, char uplo, lapack_int n, double *a, lapack_int lda); +lapack_int LAPACKE_cpotrf_work(int matrix_order, char uplo, lapack_int n, lapack_complex_float *a, lapack_int lda); +lapack_int LAPACKE_zpotrf_work(int matrix_order, char uplo, lapack_int n, lapack_complex_double *a, lapack_int lda); + +lapack_int LAPACKE_spotri_work(int matrix_order, char uplo, lapack_int n, float *a, lapack_int lda); +lapack_int LAPACKE_dpotri_work(int matrix_order, char uplo, lapack_int n, double *a, lapack_int lda); +lapack_int LAPACKE_cpotri_work(int matrix_order, char uplo, lapack_int n, lapack_complex_float *a, lapack_int lda); +lapack_int LAPACKE_zpotri_work(int matrix_order, char uplo, lapack_int n, lapack_complex_double *a, lapack_int lda); + +lapack_int LAPACKE_spotrs_work(int matrix_order, + char uplo, + lapack_int n, + lapack_int nrhs, + const float *a, + lapack_int lda, + float *b, + lapack_int ldb); +lapack_int LAPACKE_dpotrs_work(int matrix_order, + char uplo, + lapack_int n, + lapack_int nrhs, + const double *a, + lapack_int lda, + double *b, + lapack_int ldb); +lapack_int LAPACKE_cpotrs_work(int matrix_order, + char uplo, + lapack_int n, + lapack_int nrhs, + const lapack_complex_float *a, + lapack_int lda, + lapack_complex_float *b, + lapack_int ldb); +lapack_int LAPACKE_zpotrs_work(int matrix_order, + char uplo, + lapack_int n, + lapack_int nrhs, + const lapack_complex_double *a, + lapack_int lda, + lapack_complex_double *b, + lapack_int ldb); + +lapack_int LAPACKE_sppcon_work(int matrix_order, + char uplo, + lapack_int n, + const float *ap, + float anorm, + float *rcond, + float *work, + lapack_int *iwork); +lapack_int LAPACKE_dppcon_work(int matrix_order, + char uplo, + lapack_int n, + const double *ap, + double anorm, + double *rcond, + double *work, + lapack_int *iwork); +lapack_int LAPACKE_cppcon_work(int matrix_order, + char uplo, + lapack_int n, + const lapack_complex_float *ap, + float anorm, + float *rcond, + lapack_complex_float *work, + float *rwork); +lapack_int LAPACKE_zppcon_work(int matrix_order, + char uplo, + lapack_int n, + const lapack_complex_double *ap, + double anorm, + double *rcond, + lapack_complex_double *work, + double *rwork); + +lapack_int + LAPACKE_sppequ_work(int matrix_order, char uplo, lapack_int n, const float *ap, float *s, float *scond, float *amax); +lapack_int LAPACKE_dppequ_work(int matrix_order, + char uplo, + lapack_int n, + const double *ap, + double *s, + double *scond, + double *amax); +lapack_int LAPACKE_cppequ_work(int matrix_order, + char uplo, + lapack_int n, + const lapack_complex_float *ap, + float *s, + float *scond, + float *amax); +lapack_int LAPACKE_zppequ_work(int matrix_order, + char uplo, + lapack_int n, + const lapack_complex_double *ap, + double *s, + double *scond, + double *amax); + +lapack_int LAPACKE_spprfs_work(int matrix_order, + char uplo, + lapack_int n, + lapack_int nrhs, + const float *ap, + const float *afp, + const float *b, + lapack_int ldb, + float *x, + lapack_int ldx, + float *ferr, + float *berr, + float *work, + lapack_int *iwork); +lapack_int LAPACKE_dpprfs_work(int matrix_order, + char uplo, + lapack_int n, + lapack_int nrhs, + const double *ap, + const double *afp, + const double *b, + lapack_int ldb, + double *x, + lapack_int ldx, + double *ferr, + double *berr, + double *work, + lapack_int *iwork); +lapack_int LAPACKE_cpprfs_work(int matrix_order, + char uplo, + lapack_int n, + lapack_int nrhs, + const lapack_complex_float *ap, + const lapack_complex_float *afp, + const lapack_complex_float *b, + lapack_int ldb, + lapack_complex_float *x, + lapack_int ldx, + float *ferr, + float *berr, + lapack_complex_float *work, + float *rwork); +lapack_int LAPACKE_zpprfs_work(int matrix_order, + char uplo, + lapack_int n, + lapack_int nrhs, + const lapack_complex_double *ap, + const lapack_complex_double *afp, + const lapack_complex_double *b, + lapack_int ldb, + lapack_complex_double *x, + lapack_int ldx, + double *ferr, + double *berr, + lapack_complex_double *work, + double *rwork); + +lapack_int + LAPACKE_sppsv_work(int matrix_order, char uplo, lapack_int n, lapack_int nrhs, float *ap, float *b, lapack_int ldb); +lapack_int + LAPACKE_dppsv_work(int matrix_order, char uplo, lapack_int n, lapack_int nrhs, double *ap, double *b, lapack_int ldb); +lapack_int LAPACKE_cppsv_work(int matrix_order, + char uplo, + lapack_int n, + lapack_int nrhs, + lapack_complex_float *ap, + lapack_complex_float *b, + lapack_int ldb); +lapack_int LAPACKE_zppsv_work(int matrix_order, + char uplo, + lapack_int n, + lapack_int nrhs, + lapack_complex_double *ap, + lapack_complex_double *b, + lapack_int ldb); + +lapack_int LAPACKE_sppsvx_work(int matrix_order, + char fact, + char uplo, + lapack_int n, + lapack_int nrhs, + float *ap, + float *afp, + char *equed, + float *s, + float *b, + lapack_int ldb, + float *x, + lapack_int ldx, + float *rcond, + float *ferr, + float *berr, + float *work, + lapack_int *iwork); +lapack_int LAPACKE_dppsvx_work(int matrix_order, + char fact, + char uplo, + lapack_int n, + lapack_int nrhs, + double *ap, + double *afp, + char *equed, + double *s, + double *b, + lapack_int ldb, + double *x, + lapack_int ldx, + double *rcond, + double *ferr, + double *berr, + double *work, + lapack_int *iwork); +lapack_int LAPACKE_cppsvx_work(int matrix_order, + char fact, + char uplo, + lapack_int n, + lapack_int nrhs, + lapack_complex_float *ap, + lapack_complex_float *afp, + char *equed, + float *s, + lapack_complex_float *b, + lapack_int ldb, + lapack_complex_float *x, + lapack_int ldx, + float *rcond, + float *ferr, + float *berr, + lapack_complex_float *work, + float *rwork); +lapack_int LAPACKE_zppsvx_work(int matrix_order, + char fact, + char uplo, + lapack_int n, + lapack_int nrhs, + lapack_complex_double *ap, + lapack_complex_double *afp, + char *equed, + double *s, + lapack_complex_double *b, + lapack_int ldb, + lapack_complex_double *x, + lapack_int ldx, + double *rcond, + double *ferr, + double *berr, + lapack_complex_double *work, + double *rwork); + +lapack_int LAPACKE_spptrf_work(int matrix_order, char uplo, lapack_int n, float *ap); +lapack_int LAPACKE_dpptrf_work(int matrix_order, char uplo, lapack_int n, double *ap); +lapack_int LAPACKE_cpptrf_work(int matrix_order, char uplo, lapack_int n, lapack_complex_float *ap); +lapack_int LAPACKE_zpptrf_work(int matrix_order, char uplo, lapack_int n, lapack_complex_double *ap); + +lapack_int LAPACKE_spptri_work(int matrix_order, char uplo, lapack_int n, float *ap); +lapack_int LAPACKE_dpptri_work(int matrix_order, char uplo, lapack_int n, double *ap); +lapack_int LAPACKE_cpptri_work(int matrix_order, char uplo, lapack_int n, lapack_complex_float *ap); +lapack_int LAPACKE_zpptri_work(int matrix_order, char uplo, lapack_int n, lapack_complex_double *ap); + +lapack_int LAPACKE_spptrs_work(int matrix_order, + char uplo, + lapack_int n, + lapack_int nrhs, + const float *ap, + float *b, + lapack_int ldb); +lapack_int LAPACKE_dpptrs_work(int matrix_order, + char uplo, + lapack_int n, + lapack_int nrhs, + const double *ap, + double *b, + lapack_int ldb); +lapack_int LAPACKE_cpptrs_work(int matrix_order, + char uplo, + lapack_int n, + lapack_int nrhs, + const lapack_complex_float *ap, + lapack_complex_float *b, + lapack_int ldb); +lapack_int LAPACKE_zpptrs_work(int matrix_order, + char uplo, + lapack_int n, + lapack_int nrhs, + const lapack_complex_double *ap, + lapack_complex_double *b, + lapack_int ldb); + +lapack_int LAPACKE_spstrf_work(int matrix_order, + char uplo, + lapack_int n, + float *a, + lapack_int lda, + lapack_int *piv, + lapack_int *rank, + float tol, + float *work); +lapack_int LAPACKE_dpstrf_work(int matrix_order, + char uplo, + lapack_int n, + double *a, + lapack_int lda, + lapack_int *piv, + lapack_int *rank, + double tol, + double *work); +lapack_int LAPACKE_cpstrf_work(int matrix_order, + char uplo, + lapack_int n, + lapack_complex_float *a, + lapack_int lda, + lapack_int *piv, + lapack_int *rank, + float tol, + float *work); +lapack_int LAPACKE_zpstrf_work(int matrix_order, + char uplo, + lapack_int n, + lapack_complex_double *a, + lapack_int lda, + lapack_int *piv, + lapack_int *rank, + double tol, + double *work); + +lapack_int LAPACKE_sptcon_work(lapack_int n, const float *d, const float *e, float anorm, float *rcond, float *work); +lapack_int + LAPACKE_dptcon_work(lapack_int n, const double *d, const double *e, double anorm, double *rcond, double *work); +lapack_int LAPACKE_cptcon_work(lapack_int n, + const float *d, + const lapack_complex_float *e, + float anorm, + float *rcond, + float *work); +lapack_int LAPACKE_zptcon_work(lapack_int n, + const double *d, + const lapack_complex_double *e, + double anorm, + double *rcond, + double *work); + +lapack_int LAPACKE_spteqr_work(int matrix_order, + char compz, + lapack_int n, + float *d, + float *e, + float *z, + lapack_int ldz, + float *work); +lapack_int LAPACKE_dpteqr_work(int matrix_order, + char compz, + lapack_int n, + double *d, + double *e, + double *z, + lapack_int ldz, + double *work); +lapack_int LAPACKE_cpteqr_work(int matrix_order, + char compz, + lapack_int n, + float *d, + float *e, + lapack_complex_float *z, + lapack_int ldz, + float *work); +lapack_int LAPACKE_zpteqr_work(int matrix_order, + char compz, + lapack_int n, + double *d, + double *e, + lapack_complex_double *z, + lapack_int ldz, + double *work); + +lapack_int LAPACKE_sptrfs_work(int matrix_order, + lapack_int n, + lapack_int nrhs, + const float *d, + const float *e, + const float *df, + const float *ef, + const float *b, + lapack_int ldb, + float *x, + lapack_int ldx, + float *ferr, + float *berr, + float *work); +lapack_int LAPACKE_dptrfs_work(int matrix_order, + lapack_int n, + lapack_int nrhs, + const double *d, + const double *e, + const double *df, + const double *ef, + const double *b, + lapack_int ldb, + double *x, + lapack_int ldx, + double *ferr, + double *berr, + double *work); +lapack_int LAPACKE_cptrfs_work(int matrix_order, + char uplo, + lapack_int n, + lapack_int nrhs, + const float *d, + const lapack_complex_float *e, + const float *df, + const lapack_complex_float *ef, + const lapack_complex_float *b, + lapack_int ldb, + lapack_complex_float *x, + lapack_int ldx, + float *ferr, + float *berr, + lapack_complex_float *work, + float *rwork); +lapack_int LAPACKE_zptrfs_work(int matrix_order, + char uplo, + lapack_int n, + lapack_int nrhs, + const double *d, + const lapack_complex_double *e, + const double *df, + const lapack_complex_double *ef, + const lapack_complex_double *b, + lapack_int ldb, + lapack_complex_double *x, + lapack_int ldx, + double *ferr, + double *berr, + lapack_complex_double *work, + double *rwork); + +lapack_int + LAPACKE_sptsv_work(int matrix_order, lapack_int n, lapack_int nrhs, float *d, float *e, float *b, lapack_int ldb); +lapack_int + LAPACKE_dptsv_work(int matrix_order, lapack_int n, lapack_int nrhs, double *d, double *e, double *b, lapack_int ldb); +lapack_int LAPACKE_cptsv_work(int matrix_order, + lapack_int n, + lapack_int nrhs, + float *d, + lapack_complex_float *e, + lapack_complex_float *b, + lapack_int ldb); +lapack_int LAPACKE_zptsv_work(int matrix_order, + lapack_int n, + lapack_int nrhs, + double *d, + lapack_complex_double *e, + lapack_complex_double *b, + lapack_int ldb); + +lapack_int LAPACKE_sptsvx_work(int matrix_order, + char fact, + lapack_int n, + lapack_int nrhs, + const float *d, + const float *e, + float *df, + float *ef, + const float *b, + lapack_int ldb, + float *x, + lapack_int ldx, + float *rcond, + float *ferr, + float *berr, + float *work); +lapack_int LAPACKE_dptsvx_work(int matrix_order, + char fact, + lapack_int n, + lapack_int nrhs, + const double *d, + const double *e, + double *df, + double *ef, + const double *b, + lapack_int ldb, + double *x, + lapack_int ldx, + double *rcond, + double *ferr, + double *berr, + double *work); +lapack_int LAPACKE_cptsvx_work(int matrix_order, + char fact, + lapack_int n, + lapack_int nrhs, + const float *d, + const lapack_complex_float *e, + float *df, + lapack_complex_float *ef, + const lapack_complex_float *b, + lapack_int ldb, + lapack_complex_float *x, + lapack_int ldx, + float *rcond, + float *ferr, + float *berr, + lapack_complex_float *work, + float *rwork); +lapack_int LAPACKE_zptsvx_work(int matrix_order, + char fact, + lapack_int n, + lapack_int nrhs, + const double *d, + const lapack_complex_double *e, + double *df, + lapack_complex_double *ef, + const lapack_complex_double *b, + lapack_int ldb, + lapack_complex_double *x, + lapack_int ldx, + double *rcond, + double *ferr, + double *berr, + lapack_complex_double *work, + double *rwork); + +lapack_int LAPACKE_spttrf_work(lapack_int n, float *d, float *e); +lapack_int LAPACKE_dpttrf_work(lapack_int n, double *d, double *e); +lapack_int LAPACKE_cpttrf_work(lapack_int n, float *d, lapack_complex_float *e); +lapack_int LAPACKE_zpttrf_work(lapack_int n, double *d, lapack_complex_double *e); + +lapack_int LAPACKE_spttrs_work(int matrix_order, + lapack_int n, + lapack_int nrhs, + const float *d, + const float *e, + float *b, + lapack_int ldb); +lapack_int LAPACKE_dpttrs_work(int matrix_order, + lapack_int n, + lapack_int nrhs, + const double *d, + const double *e, + double *b, + lapack_int ldb); +lapack_int LAPACKE_cpttrs_work(int matrix_order, + char uplo, + lapack_int n, + lapack_int nrhs, + const float *d, + const lapack_complex_float *e, + lapack_complex_float *b, + lapack_int ldb); +lapack_int LAPACKE_zpttrs_work(int matrix_order, + char uplo, + lapack_int n, + lapack_int nrhs, + const double *d, + const lapack_complex_double *e, + lapack_complex_double *b, + lapack_int ldb); + +lapack_int LAPACKE_ssbev_work(int matrix_order, + char jobz, + char uplo, + lapack_int n, + lapack_int kd, + float *ab, + lapack_int ldab, + float *w, + float *z, + lapack_int ldz, + float *work); +lapack_int LAPACKE_dsbev_work(int matrix_order, + char jobz, + char uplo, + lapack_int n, + lapack_int kd, + double *ab, + lapack_int ldab, + double *w, + double *z, + lapack_int ldz, + double *work); + +lapack_int LAPACKE_ssbevd_work(int matrix_order, + char jobz, + char uplo, + lapack_int n, + lapack_int kd, + float *ab, + lapack_int ldab, + float *w, + float *z, + lapack_int ldz, + float *work, + lapack_int lwork, + lapack_int *iwork, + lapack_int liwork); +lapack_int LAPACKE_dsbevd_work(int matrix_order, + char jobz, + char uplo, + lapack_int n, + lapack_int kd, + double *ab, + lapack_int ldab, + double *w, + double *z, + lapack_int ldz, + double *work, + lapack_int lwork, + lapack_int *iwork, + lapack_int liwork); + +lapack_int LAPACKE_ssbevx_work(int matrix_order, + char jobz, + char range, + char uplo, + lapack_int n, + lapack_int kd, + float *ab, + lapack_int ldab, + float *q, + lapack_int ldq, + float vl, + float vu, + lapack_int il, + lapack_int iu, + float abstol, + lapack_int *m, + float *w, + float *z, + lapack_int ldz, + float *work, + lapack_int *iwork, + lapack_int *ifail); +lapack_int LAPACKE_dsbevx_work(int matrix_order, + char jobz, + char range, + char uplo, + lapack_int n, + lapack_int kd, + double *ab, + lapack_int ldab, + double *q, + lapack_int ldq, + double vl, + double vu, + lapack_int il, + lapack_int iu, + double abstol, + lapack_int *m, + double *w, + double *z, + lapack_int ldz, + double *work, + lapack_int *iwork, + lapack_int *ifail); + +lapack_int LAPACKE_ssbgst_work(int matrix_order, + char vect, + char uplo, + lapack_int n, + lapack_int ka, + lapack_int kb, + float *ab, + lapack_int ldab, + const float *bb, + lapack_int ldbb, + float *x, + lapack_int ldx, + float *work); +lapack_int LAPACKE_dsbgst_work(int matrix_order, + char vect, + char uplo, + lapack_int n, + lapack_int ka, + lapack_int kb, + double *ab, + lapack_int ldab, + const double *bb, + lapack_int ldbb, + double *x, + lapack_int ldx, + double *work); + +lapack_int LAPACKE_ssbgv_work(int matrix_order, + char jobz, + char uplo, + lapack_int n, + lapack_int ka, + lapack_int kb, + float *ab, + lapack_int ldab, + float *bb, + lapack_int ldbb, + float *w, + float *z, + lapack_int ldz, + float *work); +lapack_int LAPACKE_dsbgv_work(int matrix_order, + char jobz, + char uplo, + lapack_int n, + lapack_int ka, + lapack_int kb, + double *ab, + lapack_int ldab, + double *bb, + lapack_int ldbb, + double *w, + double *z, + lapack_int ldz, + double *work); + +lapack_int LAPACKE_ssbgvd_work(int matrix_order, + char jobz, + char uplo, + lapack_int n, + lapack_int ka, + lapack_int kb, + float *ab, + lapack_int ldab, + float *bb, + lapack_int ldbb, + float *w, + float *z, + lapack_int ldz, + float *work, + lapack_int lwork, + lapack_int *iwork, + lapack_int liwork); +lapack_int LAPACKE_dsbgvd_work(int matrix_order, + char jobz, + char uplo, + lapack_int n, + lapack_int ka, + lapack_int kb, + double *ab, + lapack_int ldab, + double *bb, + lapack_int ldbb, + double *w, + double *z, + lapack_int ldz, + double *work, + lapack_int lwork, + lapack_int *iwork, + lapack_int liwork); + +lapack_int LAPACKE_ssbgvx_work(int matrix_order, + char jobz, + char range, + char uplo, + lapack_int n, + lapack_int ka, + lapack_int kb, + float *ab, + lapack_int ldab, + float *bb, + lapack_int ldbb, + float *q, + lapack_int ldq, + float vl, + float vu, + lapack_int il, + lapack_int iu, + float abstol, + lapack_int *m, + float *w, + float *z, + lapack_int ldz, + float *work, + lapack_int *iwork, + lapack_int *ifail); +lapack_int LAPACKE_dsbgvx_work(int matrix_order, + char jobz, + char range, + char uplo, + lapack_int n, + lapack_int ka, + lapack_int kb, + double *ab, + lapack_int ldab, + double *bb, + lapack_int ldbb, + double *q, + lapack_int ldq, + double vl, + double vu, + lapack_int il, + lapack_int iu, + double abstol, + lapack_int *m, + double *w, + double *z, + lapack_int ldz, + double *work, + lapack_int *iwork, + lapack_int *ifail); + +lapack_int LAPACKE_ssbtrd_work(int matrix_order, + char vect, + char uplo, + lapack_int n, + lapack_int kd, + float *ab, + lapack_int ldab, + float *d, + float *e, + float *q, + lapack_int ldq, + float *work); +lapack_int LAPACKE_dsbtrd_work(int matrix_order, + char vect, + char uplo, + lapack_int n, + lapack_int kd, + double *ab, + lapack_int ldab, + double *d, + double *e, + double *q, + lapack_int ldq, + double *work); + +lapack_int LAPACKE_ssfrk_work(int matrix_order, + char transr, + char uplo, + char trans, + lapack_int n, + lapack_int k, + float alpha, + const float *a, + lapack_int lda, + float beta, + float *c); +lapack_int LAPACKE_dsfrk_work(int matrix_order, + char transr, + char uplo, + char trans, + lapack_int n, + lapack_int k, + double alpha, + const double *a, + lapack_int lda, + double beta, + double *c); + +lapack_int LAPACKE_sspcon_work(int matrix_order, + char uplo, + lapack_int n, + const float *ap, + const lapack_int *ipiv, + float anorm, + float *rcond, + float *work, + lapack_int *iwork); +lapack_int LAPACKE_dspcon_work(int matrix_order, + char uplo, + lapack_int n, + const double *ap, + const lapack_int *ipiv, + double anorm, + double *rcond, + double *work, + lapack_int *iwork); +lapack_int LAPACKE_cspcon_work(int matrix_order, + char uplo, + lapack_int n, + const lapack_complex_float *ap, + const lapack_int *ipiv, + float anorm, + float *rcond, + lapack_complex_float *work); +lapack_int LAPACKE_zspcon_work(int matrix_order, + char uplo, + lapack_int n, + const lapack_complex_double *ap, + const lapack_int *ipiv, + double anorm, + double *rcond, + lapack_complex_double *work); + +lapack_int LAPACKE_sspev_work(int matrix_order, + char jobz, + char uplo, + lapack_int n, + float *ap, + float *w, + float *z, + lapack_int ldz, + float *work); +lapack_int LAPACKE_dspev_work(int matrix_order, + char jobz, + char uplo, + lapack_int n, + double *ap, + double *w, + double *z, + lapack_int ldz, + double *work); + +lapack_int LAPACKE_sspevd_work(int matrix_order, + char jobz, + char uplo, + lapack_int n, + float *ap, + float *w, + float *z, + lapack_int ldz, + float *work, + lapack_int lwork, + lapack_int *iwork, + lapack_int liwork); +lapack_int LAPACKE_dspevd_work(int matrix_order, + char jobz, + char uplo, + lapack_int n, + double *ap, + double *w, + double *z, + lapack_int ldz, + double *work, + lapack_int lwork, + lapack_int *iwork, + lapack_int liwork); + +lapack_int LAPACKE_sspevx_work(int matrix_order, + char jobz, + char range, + char uplo, + lapack_int n, + float *ap, + float vl, + float vu, + lapack_int il, + lapack_int iu, + float abstol, + lapack_int *m, + float *w, + float *z, + lapack_int ldz, + float *work, + lapack_int *iwork, + lapack_int *ifail); +lapack_int LAPACKE_dspevx_work(int matrix_order, + char jobz, + char range, + char uplo, + lapack_int n, + double *ap, + double vl, + double vu, + lapack_int il, + lapack_int iu, + double abstol, + lapack_int *m, + double *w, + double *z, + lapack_int ldz, + double *work, + lapack_int *iwork, + lapack_int *ifail); + +lapack_int LAPACKE_sspgst_work(int matrix_order, lapack_int itype, char uplo, lapack_int n, float *ap, const float *bp); +lapack_int + LAPACKE_dspgst_work(int matrix_order, lapack_int itype, char uplo, lapack_int n, double *ap, const double *bp); + +lapack_int LAPACKE_sspgv_work(int matrix_order, + lapack_int itype, + char jobz, + char uplo, + lapack_int n, + float *ap, + float *bp, + float *w, + float *z, + lapack_int ldz, + float *work); +lapack_int LAPACKE_dspgv_work(int matrix_order, + lapack_int itype, + char jobz, + char uplo, + lapack_int n, + double *ap, + double *bp, + double *w, + double *z, + lapack_int ldz, + double *work); + +lapack_int LAPACKE_sspgvd_work(int matrix_order, + lapack_int itype, + char jobz, + char uplo, + lapack_int n, + float *ap, + float *bp, + float *w, + float *z, + lapack_int ldz, + float *work, + lapack_int lwork, + lapack_int *iwork, + lapack_int liwork); +lapack_int LAPACKE_dspgvd_work(int matrix_order, + lapack_int itype, + char jobz, + char uplo, + lapack_int n, + double *ap, + double *bp, + double *w, + double *z, + lapack_int ldz, + double *work, + lapack_int lwork, + lapack_int *iwork, + lapack_int liwork); + +lapack_int LAPACKE_sspgvx_work(int matrix_order, + lapack_int itype, + char jobz, + char range, + char uplo, + lapack_int n, + float *ap, + float *bp, + float vl, + float vu, + lapack_int il, + lapack_int iu, + float abstol, + lapack_int *m, + float *w, + float *z, + lapack_int ldz, + float *work, + lapack_int *iwork, + lapack_int *ifail); +lapack_int LAPACKE_dspgvx_work(int matrix_order, + lapack_int itype, + char jobz, + char range, + char uplo, + lapack_int n, + double *ap, + double *bp, + double vl, + double vu, + lapack_int il, + lapack_int iu, + double abstol, + lapack_int *m, + double *w, + double *z, + lapack_int ldz, + double *work, + lapack_int *iwork, + lapack_int *ifail); + +lapack_int LAPACKE_ssprfs_work(int matrix_order, + char uplo, + lapack_int n, + lapack_int nrhs, + const float *ap, + const float *afp, + const lapack_int *ipiv, + const float *b, + lapack_int ldb, + float *x, + lapack_int ldx, + float *ferr, + float *berr, + float *work, + lapack_int *iwork); +lapack_int LAPACKE_dsprfs_work(int matrix_order, + char uplo, + lapack_int n, + lapack_int nrhs, + const double *ap, + const double *afp, + const lapack_int *ipiv, + const double *b, + lapack_int ldb, + double *x, + lapack_int ldx, + double *ferr, + double *berr, + double *work, + lapack_int *iwork); +lapack_int LAPACKE_csprfs_work(int matrix_order, + char uplo, + lapack_int n, + lapack_int nrhs, + const lapack_complex_float *ap, + const lapack_complex_float *afp, + const lapack_int *ipiv, + const lapack_complex_float *b, + lapack_int ldb, + lapack_complex_float *x, + lapack_int ldx, + float *ferr, + float *berr, + lapack_complex_float *work, + float *rwork); +lapack_int LAPACKE_zsprfs_work(int matrix_order, + char uplo, + lapack_int n, + lapack_int nrhs, + const lapack_complex_double *ap, + const lapack_complex_double *afp, + const lapack_int *ipiv, + const lapack_complex_double *b, + lapack_int ldb, + lapack_complex_double *x, + lapack_int ldx, + double *ferr, + double *berr, + lapack_complex_double *work, + double *rwork); + +lapack_int LAPACKE_sspsv_work(int matrix_order, + char uplo, + lapack_int n, + lapack_int nrhs, + float *ap, + lapack_int *ipiv, + float *b, + lapack_int ldb); +lapack_int LAPACKE_dspsv_work(int matrix_order, + char uplo, + lapack_int n, + lapack_int nrhs, + double *ap, + lapack_int *ipiv, + double *b, + lapack_int ldb); +lapack_int LAPACKE_cspsv_work(int matrix_order, + char uplo, + lapack_int n, + lapack_int nrhs, + lapack_complex_float *ap, + lapack_int *ipiv, + lapack_complex_float *b, + lapack_int ldb); +lapack_int LAPACKE_zspsv_work(int matrix_order, + char uplo, + lapack_int n, + lapack_int nrhs, + lapack_complex_double *ap, + lapack_int *ipiv, + lapack_complex_double *b, + lapack_int ldb); + +lapack_int LAPACKE_sspsvx_work(int matrix_order, + char fact, + char uplo, + lapack_int n, + lapack_int nrhs, + const float *ap, + float *afp, + lapack_int *ipiv, + const float *b, + lapack_int ldb, + float *x, + lapack_int ldx, + float *rcond, + float *ferr, + float *berr, + float *work, + lapack_int *iwork); +lapack_int LAPACKE_dspsvx_work(int matrix_order, + char fact, + char uplo, + lapack_int n, + lapack_int nrhs, + const double *ap, + double *afp, + lapack_int *ipiv, + const double *b, + lapack_int ldb, + double *x, + lapack_int ldx, + double *rcond, + double *ferr, + double *berr, + double *work, + lapack_int *iwork); +lapack_int LAPACKE_cspsvx_work(int matrix_order, + char fact, + char uplo, + lapack_int n, + lapack_int nrhs, + const lapack_complex_float *ap, + lapack_complex_float *afp, + lapack_int *ipiv, + const lapack_complex_float *b, + lapack_int ldb, + lapack_complex_float *x, + lapack_int ldx, + float *rcond, + float *ferr, + float *berr, + lapack_complex_float *work, + float *rwork); +lapack_int LAPACKE_zspsvx_work(int matrix_order, + char fact, + char uplo, + lapack_int n, + lapack_int nrhs, + const lapack_complex_double *ap, + lapack_complex_double *afp, + lapack_int *ipiv, + const lapack_complex_double *b, + lapack_int ldb, + lapack_complex_double *x, + lapack_int ldx, + double *rcond, + double *ferr, + double *berr, + lapack_complex_double *work, + double *rwork); + +lapack_int LAPACKE_ssptrd_work(int matrix_order, char uplo, lapack_int n, float *ap, float *d, float *e, float *tau); +lapack_int + LAPACKE_dsptrd_work(int matrix_order, char uplo, lapack_int n, double *ap, double *d, double *e, double *tau); + +lapack_int LAPACKE_ssptrf_work(int matrix_order, char uplo, lapack_int n, float *ap, lapack_int *ipiv); +lapack_int LAPACKE_dsptrf_work(int matrix_order, char uplo, lapack_int n, double *ap, lapack_int *ipiv); +lapack_int LAPACKE_csptrf_work(int matrix_order, char uplo, lapack_int n, lapack_complex_float *ap, lapack_int *ipiv); +lapack_int LAPACKE_zsptrf_work(int matrix_order, char uplo, lapack_int n, lapack_complex_double *ap, lapack_int *ipiv); + +lapack_int + LAPACKE_ssptri_work(int matrix_order, char uplo, lapack_int n, float *ap, const lapack_int *ipiv, float *work); +lapack_int + LAPACKE_dsptri_work(int matrix_order, char uplo, lapack_int n, double *ap, const lapack_int *ipiv, double *work); +lapack_int LAPACKE_csptri_work(int matrix_order, + char uplo, + lapack_int n, + lapack_complex_float *ap, + const lapack_int *ipiv, + lapack_complex_float *work); +lapack_int LAPACKE_zsptri_work(int matrix_order, + char uplo, + lapack_int n, + lapack_complex_double *ap, + const lapack_int *ipiv, + lapack_complex_double *work); + +lapack_int LAPACKE_ssptrs_work(int matrix_order, + char uplo, + lapack_int n, + lapack_int nrhs, + const float *ap, + const lapack_int *ipiv, + float *b, + lapack_int ldb); +lapack_int LAPACKE_dsptrs_work(int matrix_order, + char uplo, + lapack_int n, + lapack_int nrhs, + const double *ap, + const lapack_int *ipiv, + double *b, + lapack_int ldb); +lapack_int LAPACKE_csptrs_work(int matrix_order, + char uplo, + lapack_int n, + lapack_int nrhs, + const lapack_complex_float *ap, + const lapack_int *ipiv, + lapack_complex_float *b, + lapack_int ldb); +lapack_int LAPACKE_zsptrs_work(int matrix_order, + char uplo, + lapack_int n, + lapack_int nrhs, + const lapack_complex_double *ap, + const lapack_int *ipiv, + lapack_complex_double *b, + lapack_int ldb); + +lapack_int LAPACKE_sstebz_work(char range, + char order, + lapack_int n, + float vl, + float vu, + lapack_int il, + lapack_int iu, + float abstol, + const float *d, + const float *e, + lapack_int *m, + lapack_int *nsplit, + float *w, + lapack_int *iblock, + lapack_int *isplit, + float *work, + lapack_int *iwork); +lapack_int LAPACKE_dstebz_work(char range, + char order, + lapack_int n, + double vl, + double vu, + lapack_int il, + lapack_int iu, + double abstol, + const double *d, + const double *e, + lapack_int *m, + lapack_int *nsplit, + double *w, + lapack_int *iblock, + lapack_int *isplit, + double *work, + lapack_int *iwork); + +lapack_int LAPACKE_sstedc_work(int matrix_order, + char compz, + lapack_int n, + float *d, + float *e, + float *z, + lapack_int ldz, + float *work, + lapack_int lwork, + lapack_int *iwork, + lapack_int liwork); +lapack_int LAPACKE_dstedc_work(int matrix_order, + char compz, + lapack_int n, + double *d, + double *e, + double *z, + lapack_int ldz, + double *work, + lapack_int lwork, + lapack_int *iwork, + lapack_int liwork); +lapack_int LAPACKE_cstedc_work(int matrix_order, + char compz, + lapack_int n, + float *d, + float *e, + lapack_complex_float *z, + lapack_int ldz, + lapack_complex_float *work, + lapack_int lwork, + float *rwork, + lapack_int lrwork, + lapack_int *iwork, + lapack_int liwork); +lapack_int LAPACKE_zstedc_work(int matrix_order, + char compz, + lapack_int n, + double *d, + double *e, + lapack_complex_double *z, + lapack_int ldz, + lapack_complex_double *work, + lapack_int lwork, + double *rwork, + lapack_int lrwork, + lapack_int *iwork, + lapack_int liwork); + +lapack_int LAPACKE_sstegr_work(int matrix_order, + char jobz, + char range, + lapack_int n, + float *d, + float *e, + float vl, + float vu, + lapack_int il, + lapack_int iu, + float abstol, + lapack_int *m, + float *w, + float *z, + lapack_int ldz, + lapack_int *isuppz, + float *work, + lapack_int lwork, + lapack_int *iwork, + lapack_int liwork); +lapack_int LAPACKE_dstegr_work(int matrix_order, + char jobz, + char range, + lapack_int n, + double *d, + double *e, + double vl, + double vu, + lapack_int il, + lapack_int iu, + double abstol, + lapack_int *m, + double *w, + double *z, + lapack_int ldz, + lapack_int *isuppz, + double *work, + lapack_int lwork, + lapack_int *iwork, + lapack_int liwork); +lapack_int LAPACKE_cstegr_work(int matrix_order, + char jobz, + char range, + lapack_int n, + float *d, + float *e, + float vl, + float vu, + lapack_int il, + lapack_int iu, + float abstol, + lapack_int *m, + float *w, + lapack_complex_float *z, + lapack_int ldz, + lapack_int *isuppz, + float *work, + lapack_int lwork, + lapack_int *iwork, + lapack_int liwork); +lapack_int LAPACKE_zstegr_work(int matrix_order, + char jobz, + char range, + lapack_int n, + double *d, + double *e, + double vl, + double vu, + lapack_int il, + lapack_int iu, + double abstol, + lapack_int *m, + double *w, + lapack_complex_double *z, + lapack_int ldz, + lapack_int *isuppz, + double *work, + lapack_int lwork, + lapack_int *iwork, + lapack_int liwork); + +lapack_int LAPACKE_sstein_work(int matrix_order, + lapack_int n, + const float *d, + const float *e, + lapack_int m, + const float *w, + const lapack_int *iblock, + const lapack_int *isplit, + float *z, + lapack_int ldz, + float *work, + lapack_int *iwork, + lapack_int *ifailv); +lapack_int LAPACKE_dstein_work(int matrix_order, + lapack_int n, + const double *d, + const double *e, + lapack_int m, + const double *w, + const lapack_int *iblock, + const lapack_int *isplit, + double *z, + lapack_int ldz, + double *work, + lapack_int *iwork, + lapack_int *ifailv); +lapack_int LAPACKE_cstein_work(int matrix_order, + lapack_int n, + const float *d, + const float *e, + lapack_int m, + const float *w, + const lapack_int *iblock, + const lapack_int *isplit, + lapack_complex_float *z, + lapack_int ldz, + float *work, + lapack_int *iwork, + lapack_int *ifailv); +lapack_int LAPACKE_zstein_work(int matrix_order, + lapack_int n, + const double *d, + const double *e, + lapack_int m, + const double *w, + const lapack_int *iblock, + const lapack_int *isplit, + lapack_complex_double *z, + lapack_int ldz, + double *work, + lapack_int *iwork, + lapack_int *ifailv); + +lapack_int LAPACKE_sstemr_work(int matrix_order, + char jobz, + char range, + lapack_int n, + float *d, + float *e, + float vl, + float vu, + lapack_int il, + lapack_int iu, + lapack_int *m, + float *w, + float *z, + lapack_int ldz, + lapack_int nzc, + lapack_int *isuppz, + lapack_logical *tryrac, + float *work, + lapack_int lwork, + lapack_int *iwork, + lapack_int liwork); +lapack_int LAPACKE_dstemr_work(int matrix_order, + char jobz, + char range, + lapack_int n, + double *d, + double *e, + double vl, + double vu, + lapack_int il, + lapack_int iu, + lapack_int *m, + double *w, + double *z, + lapack_int ldz, + lapack_int nzc, + lapack_int *isuppz, + lapack_logical *tryrac, + double *work, + lapack_int lwork, + lapack_int *iwork, + lapack_int liwork); +lapack_int LAPACKE_cstemr_work(int matrix_order, + char jobz, + char range, + lapack_int n, + float *d, + float *e, + float vl, + float vu, + lapack_int il, + lapack_int iu, + lapack_int *m, + float *w, + lapack_complex_float *z, + lapack_int ldz, + lapack_int nzc, + lapack_int *isuppz, + lapack_logical *tryrac, + float *work, + lapack_int lwork, + lapack_int *iwork, + lapack_int liwork); +lapack_int LAPACKE_zstemr_work(int matrix_order, + char jobz, + char range, + lapack_int n, + double *d, + double *e, + double vl, + double vu, + lapack_int il, + lapack_int iu, + lapack_int *m, + double *w, + lapack_complex_double *z, + lapack_int ldz, + lapack_int nzc, + lapack_int *isuppz, + lapack_logical *tryrac, + double *work, + lapack_int lwork, + lapack_int *iwork, + lapack_int liwork); + +lapack_int LAPACKE_ssteqr_work(int matrix_order, + char compz, + lapack_int n, + float *d, + float *e, + float *z, + lapack_int ldz, + float *work); +lapack_int LAPACKE_dsteqr_work(int matrix_order, + char compz, + lapack_int n, + double *d, + double *e, + double *z, + lapack_int ldz, + double *work); +lapack_int LAPACKE_csteqr_work(int matrix_order, + char compz, + lapack_int n, + float *d, + float *e, + lapack_complex_float *z, + lapack_int ldz, + float *work); +lapack_int LAPACKE_zsteqr_work(int matrix_order, + char compz, + lapack_int n, + double *d, + double *e, + lapack_complex_double *z, + lapack_int ldz, + double *work); + +lapack_int LAPACKE_ssterf_work(lapack_int n, float *d, float *e); +lapack_int LAPACKE_dsterf_work(lapack_int n, double *d, double *e); + +lapack_int LAPACKE_sstev_work(int matrix_order, + char jobz, + lapack_int n, + float *d, + float *e, + float *z, + lapack_int ldz, + float *work); +lapack_int LAPACKE_dstev_work(int matrix_order, + char jobz, + lapack_int n, + double *d, + double *e, + double *z, + lapack_int ldz, + double *work); + +lapack_int LAPACKE_sstevd_work(int matrix_order, + char jobz, + lapack_int n, + float *d, + float *e, + float *z, + lapack_int ldz, + float *work, + lapack_int lwork, + lapack_int *iwork, + lapack_int liwork); +lapack_int LAPACKE_dstevd_work(int matrix_order, + char jobz, + lapack_int n, + double *d, + double *e, + double *z, + lapack_int ldz, + double *work, + lapack_int lwork, + lapack_int *iwork, + lapack_int liwork); + +lapack_int LAPACKE_sstevr_work(int matrix_order, + char jobz, + char range, + lapack_int n, + float *d, + float *e, + float vl, + float vu, + lapack_int il, + lapack_int iu, + float abstol, + lapack_int *m, + float *w, + float *z, + lapack_int ldz, + lapack_int *isuppz, + float *work, + lapack_int lwork, + lapack_int *iwork, + lapack_int liwork); +lapack_int LAPACKE_dstevr_work(int matrix_order, + char jobz, + char range, + lapack_int n, + double *d, + double *e, + double vl, + double vu, + lapack_int il, + lapack_int iu, + double abstol, + lapack_int *m, + double *w, + double *z, + lapack_int ldz, + lapack_int *isuppz, + double *work, + lapack_int lwork, + lapack_int *iwork, + lapack_int liwork); + +lapack_int LAPACKE_sstevx_work(int matrix_order, + char jobz, + char range, + lapack_int n, + float *d, + float *e, + float vl, + float vu, + lapack_int il, + lapack_int iu, + float abstol, + lapack_int *m, + float *w, + float *z, + lapack_int ldz, + float *work, + lapack_int *iwork, + lapack_int *ifail); +lapack_int LAPACKE_dstevx_work(int matrix_order, + char jobz, + char range, + lapack_int n, + double *d, + double *e, + double vl, + double vu, + lapack_int il, + lapack_int iu, + double abstol, + lapack_int *m, + double *w, + double *z, + lapack_int ldz, + double *work, + lapack_int *iwork, + lapack_int *ifail); + +lapack_int LAPACKE_ssycon_work(int matrix_order, + char uplo, + lapack_int n, + const float *a, + lapack_int lda, + const lapack_int *ipiv, + float anorm, + float *rcond, + float *work, + lapack_int *iwork); +lapack_int LAPACKE_dsycon_work(int matrix_order, + char uplo, + lapack_int n, + const double *a, + lapack_int lda, + const lapack_int *ipiv, + double anorm, + double *rcond, + double *work, + lapack_int *iwork); +lapack_int LAPACKE_csycon_work(int matrix_order, + char uplo, + lapack_int n, + const lapack_complex_float *a, + lapack_int lda, + const lapack_int *ipiv, + float anorm, + float *rcond, + lapack_complex_float *work); +lapack_int LAPACKE_zsycon_work(int matrix_order, + char uplo, + lapack_int n, + const lapack_complex_double *a, + lapack_int lda, + const lapack_int *ipiv, + double anorm, + double *rcond, + lapack_complex_double *work); + +lapack_int LAPACKE_ssyequb_work(int matrix_order, + char uplo, + lapack_int n, + const float *a, + lapack_int lda, + float *s, + float *scond, + float *amax, + float *work); +lapack_int LAPACKE_dsyequb_work(int matrix_order, + char uplo, + lapack_int n, + const double *a, + lapack_int lda, + double *s, + double *scond, + double *amax, + double *work); +lapack_int LAPACKE_csyequb_work(int matrix_order, + char uplo, + lapack_int n, + const lapack_complex_float *a, + lapack_int lda, + float *s, + float *scond, + float *amax, + lapack_complex_float *work); +lapack_int LAPACKE_zsyequb_work(int matrix_order, + char uplo, + lapack_int n, + const lapack_complex_double *a, + lapack_int lda, + double *s, + double *scond, + double *amax, + lapack_complex_double *work); + +lapack_int LAPACKE_ssyev_work(int matrix_order, + char jobz, + char uplo, + lapack_int n, + float *a, + lapack_int lda, + float *w, + float *work, + lapack_int lwork); +lapack_int LAPACKE_dsyev_work(int matrix_order, + char jobz, + char uplo, + lapack_int n, + double *a, + lapack_int lda, + double *w, + double *work, + lapack_int lwork); + +lapack_int LAPACKE_ssyevd_work(int matrix_order, + char jobz, + char uplo, + lapack_int n, + float *a, + lapack_int lda, + float *w, + float *work, + lapack_int lwork, + lapack_int *iwork, + lapack_int liwork); +lapack_int LAPACKE_dsyevd_work(int matrix_order, + char jobz, + char uplo, + lapack_int n, + double *a, + lapack_int lda, + double *w, + double *work, + lapack_int lwork, + lapack_int *iwork, + lapack_int liwork); + +lapack_int LAPACKE_ssyevr_work(int matrix_order, + char jobz, + char range, + char uplo, + lapack_int n, + float *a, + lapack_int lda, + float vl, + float vu, + lapack_int il, + lapack_int iu, + float abstol, + lapack_int *m, + float *w, + float *z, + lapack_int ldz, + lapack_int *isuppz, + float *work, + lapack_int lwork, + lapack_int *iwork, + lapack_int liwork); +lapack_int LAPACKE_dsyevr_work(int matrix_order, + char jobz, + char range, + char uplo, + lapack_int n, + double *a, + lapack_int lda, + double vl, + double vu, + lapack_int il, + lapack_int iu, + double abstol, + lapack_int *m, + double *w, + double *z, + lapack_int ldz, + lapack_int *isuppz, + double *work, + lapack_int lwork, + lapack_int *iwork, + lapack_int liwork); + +lapack_int LAPACKE_ssyevx_work(int matrix_order, + char jobz, + char range, + char uplo, + lapack_int n, + float *a, + lapack_int lda, + float vl, + float vu, + lapack_int il, + lapack_int iu, + float abstol, + lapack_int *m, + float *w, + float *z, + lapack_int ldz, + float *work, + lapack_int lwork, + lapack_int *iwork, + lapack_int *ifail); +lapack_int LAPACKE_dsyevx_work(int matrix_order, + char jobz, + char range, + char uplo, + lapack_int n, + double *a, + lapack_int lda, + double vl, + double vu, + lapack_int il, + lapack_int iu, + double abstol, + lapack_int *m, + double *w, + double *z, + lapack_int ldz, + double *work, + lapack_int lwork, + lapack_int *iwork, + lapack_int *ifail); + +lapack_int LAPACKE_ssygst_work(int matrix_order, + lapack_int itype, + char uplo, + lapack_int n, + float *a, + lapack_int lda, + const float *b, + lapack_int ldb); +lapack_int LAPACKE_dsygst_work(int matrix_order, + lapack_int itype, + char uplo, + lapack_int n, + double *a, + lapack_int lda, + const double *b, + lapack_int ldb); + +lapack_int LAPACKE_ssygv_work(int matrix_order, + lapack_int itype, + char jobz, + char uplo, + lapack_int n, + float *a, + lapack_int lda, + float *b, + lapack_int ldb, + float *w, + float *work, + lapack_int lwork); +lapack_int LAPACKE_dsygv_work(int matrix_order, + lapack_int itype, + char jobz, + char uplo, + lapack_int n, + double *a, + lapack_int lda, + double *b, + lapack_int ldb, + double *w, + double *work, + lapack_int lwork); + +lapack_int LAPACKE_ssygvd_work(int matrix_order, + lapack_int itype, + char jobz, + char uplo, + lapack_int n, + float *a, + lapack_int lda, + float *b, + lapack_int ldb, + float *w, + float *work, + lapack_int lwork, + lapack_int *iwork, + lapack_int liwork); +lapack_int LAPACKE_dsygvd_work(int matrix_order, + lapack_int itype, + char jobz, + char uplo, + lapack_int n, + double *a, + lapack_int lda, + double *b, + lapack_int ldb, + double *w, + double *work, + lapack_int lwork, + lapack_int *iwork, + lapack_int liwork); + +lapack_int LAPACKE_ssygvx_work(int matrix_order, + lapack_int itype, + char jobz, + char range, + char uplo, + lapack_int n, + float *a, + lapack_int lda, + float *b, + lapack_int ldb, + float vl, + float vu, + lapack_int il, + lapack_int iu, + float abstol, + lapack_int *m, + float *w, + float *z, + lapack_int ldz, + float *work, + lapack_int lwork, + lapack_int *iwork, + lapack_int *ifail); +lapack_int LAPACKE_dsygvx_work(int matrix_order, + lapack_int itype, + char jobz, + char range, + char uplo, + lapack_int n, + double *a, + lapack_int lda, + double *b, + lapack_int ldb, + double vl, + double vu, + lapack_int il, + lapack_int iu, + double abstol, + lapack_int *m, + double *w, + double *z, + lapack_int ldz, + double *work, + lapack_int lwork, + lapack_int *iwork, + lapack_int *ifail); + +lapack_int LAPACKE_ssyrfs_work(int matrix_order, + char uplo, + lapack_int n, + lapack_int nrhs, + const float *a, + lapack_int lda, + const float *af, + lapack_int ldaf, + const lapack_int *ipiv, + const float *b, + lapack_int ldb, + float *x, + lapack_int ldx, + float *ferr, + float *berr, + float *work, + lapack_int *iwork); +lapack_int LAPACKE_dsyrfs_work(int matrix_order, + char uplo, + lapack_int n, + lapack_int nrhs, + const double *a, + lapack_int lda, + const double *af, + lapack_int ldaf, + const lapack_int *ipiv, + const double *b, + lapack_int ldb, + double *x, + lapack_int ldx, + double *ferr, + double *berr, + double *work, + lapack_int *iwork); +lapack_int LAPACKE_csyrfs_work(int matrix_order, + char uplo, + lapack_int n, + lapack_int nrhs, + const lapack_complex_float *a, + lapack_int lda, + const lapack_complex_float *af, + lapack_int ldaf, + const lapack_int *ipiv, + const lapack_complex_float *b, + lapack_int ldb, + lapack_complex_float *x, + lapack_int ldx, + float *ferr, + float *berr, + lapack_complex_float *work, + float *rwork); +lapack_int LAPACKE_zsyrfs_work(int matrix_order, + char uplo, + lapack_int n, + lapack_int nrhs, + const lapack_complex_double *a, + lapack_int lda, + const lapack_complex_double *af, + lapack_int ldaf, + const lapack_int *ipiv, + const lapack_complex_double *b, + lapack_int ldb, + lapack_complex_double *x, + lapack_int ldx, + double *ferr, + double *berr, + lapack_complex_double *work, + double *rwork); + +lapack_int LAPACKE_ssyrfsx_work(int matrix_order, + char uplo, + char equed, + lapack_int n, + lapack_int nrhs, + const float *a, + lapack_int lda, + const float *af, + lapack_int ldaf, + const lapack_int *ipiv, + const float *s, + const float *b, + lapack_int ldb, + float *x, + lapack_int ldx, + float *rcond, + float *berr, + lapack_int n_err_bnds, + float *err_bnds_norm, + float *err_bnds_comp, + lapack_int nparams, + float *params, + float *work, + lapack_int *iwork); +lapack_int LAPACKE_dsyrfsx_work(int matrix_order, + char uplo, + char equed, + lapack_int n, + lapack_int nrhs, + const double *a, + lapack_int lda, + const double *af, + lapack_int ldaf, + const lapack_int *ipiv, + const double *s, + const double *b, + lapack_int ldb, + double *x, + lapack_int ldx, + double *rcond, + double *berr, + lapack_int n_err_bnds, + double *err_bnds_norm, + double *err_bnds_comp, + lapack_int nparams, + double *params, + double *work, + lapack_int *iwork); +lapack_int LAPACKE_csyrfsx_work(int matrix_order, + char uplo, + char equed, + lapack_int n, + lapack_int nrhs, + const lapack_complex_float *a, + lapack_int lda, + const lapack_complex_float *af, + lapack_int ldaf, + const lapack_int *ipiv, + const float *s, + const lapack_complex_float *b, + lapack_int ldb, + lapack_complex_float *x, + lapack_int ldx, + float *rcond, + float *berr, + lapack_int n_err_bnds, + float *err_bnds_norm, + float *err_bnds_comp, + lapack_int nparams, + float *params, + lapack_complex_float *work, + float *rwork); +lapack_int LAPACKE_zsyrfsx_work(int matrix_order, + char uplo, + char equed, + lapack_int n, + lapack_int nrhs, + const lapack_complex_double *a, + lapack_int lda, + const lapack_complex_double *af, + lapack_int ldaf, + const lapack_int *ipiv, + const double *s, + const lapack_complex_double *b, + lapack_int ldb, + lapack_complex_double *x, + lapack_int ldx, + double *rcond, + double *berr, + lapack_int n_err_bnds, + double *err_bnds_norm, + double *err_bnds_comp, + lapack_int nparams, + double *params, + lapack_complex_double *work, + double *rwork); + +lapack_int LAPACKE_ssysv_work(int matrix_order, + char uplo, + lapack_int n, + lapack_int nrhs, + float *a, + lapack_int lda, + lapack_int *ipiv, + float *b, + lapack_int ldb, + float *work, + lapack_int lwork); +lapack_int LAPACKE_dsysv_work(int matrix_order, + char uplo, + lapack_int n, + lapack_int nrhs, + double *a, + lapack_int lda, + lapack_int *ipiv, + double *b, + lapack_int ldb, + double *work, + lapack_int lwork); +lapack_int LAPACKE_csysv_work(int matrix_order, + char uplo, + lapack_int n, + lapack_int nrhs, + lapack_complex_float *a, + lapack_int lda, + lapack_int *ipiv, + lapack_complex_float *b, + lapack_int ldb, + lapack_complex_float *work, + lapack_int lwork); +lapack_int LAPACKE_zsysv_work(int matrix_order, + char uplo, + lapack_int n, + lapack_int nrhs, + lapack_complex_double *a, + lapack_int lda, + lapack_int *ipiv, + lapack_complex_double *b, + lapack_int ldb, + lapack_complex_double *work, + lapack_int lwork); + +lapack_int LAPACKE_ssysvx_work(int matrix_order, + char fact, + char uplo, + lapack_int n, + lapack_int nrhs, + const float *a, + lapack_int lda, + float *af, + lapack_int ldaf, + lapack_int *ipiv, + const float *b, + lapack_int ldb, + float *x, + lapack_int ldx, + float *rcond, + float *ferr, + float *berr, + float *work, + lapack_int lwork, + lapack_int *iwork); +lapack_int LAPACKE_dsysvx_work(int matrix_order, + char fact, + char uplo, + lapack_int n, + lapack_int nrhs, + const double *a, + lapack_int lda, + double *af, + lapack_int ldaf, + lapack_int *ipiv, + const double *b, + lapack_int ldb, + double *x, + lapack_int ldx, + double *rcond, + double *ferr, + double *berr, + double *work, + lapack_int lwork, + lapack_int *iwork); +lapack_int LAPACKE_csysvx_work(int matrix_order, + char fact, + char uplo, + lapack_int n, + lapack_int nrhs, + const lapack_complex_float *a, + lapack_int lda, + lapack_complex_float *af, + lapack_int ldaf, + lapack_int *ipiv, + const lapack_complex_float *b, + lapack_int ldb, + lapack_complex_float *x, + lapack_int ldx, + float *rcond, + float *ferr, + float *berr, + lapack_complex_float *work, + lapack_int lwork, + float *rwork); +lapack_int LAPACKE_zsysvx_work(int matrix_order, + char fact, + char uplo, + lapack_int n, + lapack_int nrhs, + const lapack_complex_double *a, + lapack_int lda, + lapack_complex_double *af, + lapack_int ldaf, + lapack_int *ipiv, + const lapack_complex_double *b, + lapack_int ldb, + lapack_complex_double *x, + lapack_int ldx, + double *rcond, + double *ferr, + double *berr, + lapack_complex_double *work, + lapack_int lwork, + double *rwork); + +lapack_int LAPACKE_ssysvxx_work(int matrix_order, + char fact, + char uplo, + lapack_int n, + lapack_int nrhs, + float *a, + lapack_int lda, + float *af, + lapack_int ldaf, + lapack_int *ipiv, + char *equed, + float *s, + float *b, + lapack_int ldb, + float *x, + lapack_int ldx, + float *rcond, + float *rpvgrw, + float *berr, + lapack_int n_err_bnds, + float *err_bnds_norm, + float *err_bnds_comp, + lapack_int nparams, + float *params, + float *work, + lapack_int *iwork); +lapack_int LAPACKE_dsysvxx_work(int matrix_order, + char fact, + char uplo, + lapack_int n, + lapack_int nrhs, + double *a, + lapack_int lda, + double *af, + lapack_int ldaf, + lapack_int *ipiv, + char *equed, + double *s, + double *b, + lapack_int ldb, + double *x, + lapack_int ldx, + double *rcond, + double *rpvgrw, + double *berr, + lapack_int n_err_bnds, + double *err_bnds_norm, + double *err_bnds_comp, + lapack_int nparams, + double *params, + double *work, + lapack_int *iwork); +lapack_int LAPACKE_csysvxx_work(int matrix_order, + char fact, + char uplo, + lapack_int n, + lapack_int nrhs, + lapack_complex_float *a, + lapack_int lda, + lapack_complex_float *af, + lapack_int ldaf, + lapack_int *ipiv, + char *equed, + float *s, + lapack_complex_float *b, + lapack_int ldb, + lapack_complex_float *x, + lapack_int ldx, + float *rcond, + float *rpvgrw, + float *berr, + lapack_int n_err_bnds, + float *err_bnds_norm, + float *err_bnds_comp, + lapack_int nparams, + float *params, + lapack_complex_float *work, + float *rwork); +lapack_int LAPACKE_zsysvxx_work(int matrix_order, + char fact, + char uplo, + lapack_int n, + lapack_int nrhs, + lapack_complex_double *a, + lapack_int lda, + lapack_complex_double *af, + lapack_int ldaf, + lapack_int *ipiv, + char *equed, + double *s, + lapack_complex_double *b, + lapack_int ldb, + lapack_complex_double *x, + lapack_int ldx, + double *rcond, + double *rpvgrw, + double *berr, + lapack_int n_err_bnds, + double *err_bnds_norm, + double *err_bnds_comp, + lapack_int nparams, + double *params, + lapack_complex_double *work, + double *rwork); + +lapack_int LAPACKE_ssytrd_work(int matrix_order, + char uplo, + lapack_int n, + float *a, + lapack_int lda, + float *d, + float *e, + float *tau, + float *work, + lapack_int lwork); +lapack_int LAPACKE_dsytrd_work(int matrix_order, + char uplo, + lapack_int n, + double *a, + lapack_int lda, + double *d, + double *e, + double *tau, + double *work, + lapack_int lwork); + +lapack_int LAPACKE_ssytrf_work(int matrix_order, + char uplo, + lapack_int n, + float *a, + lapack_int lda, + lapack_int *ipiv, + float *work, + lapack_int lwork); +lapack_int LAPACKE_dsytrf_work(int matrix_order, + char uplo, + lapack_int n, + double *a, + lapack_int lda, + lapack_int *ipiv, + double *work, + lapack_int lwork); +lapack_int LAPACKE_csytrf_work(int matrix_order, + char uplo, + lapack_int n, + lapack_complex_float *a, + lapack_int lda, + lapack_int *ipiv, + lapack_complex_float *work, + lapack_int lwork); +lapack_int LAPACKE_zsytrf_work(int matrix_order, + char uplo, + lapack_int n, + lapack_complex_double *a, + lapack_int lda, + lapack_int *ipiv, + lapack_complex_double *work, + lapack_int lwork); + +lapack_int LAPACKE_ssytri_work(int matrix_order, + char uplo, + lapack_int n, + float *a, + lapack_int lda, + const lapack_int *ipiv, + float *work); +lapack_int LAPACKE_dsytri_work(int matrix_order, + char uplo, + lapack_int n, + double *a, + lapack_int lda, + const lapack_int *ipiv, + double *work); +lapack_int LAPACKE_csytri_work(int matrix_order, + char uplo, + lapack_int n, + lapack_complex_float *a, + lapack_int lda, + const lapack_int *ipiv, + lapack_complex_float *work); +lapack_int LAPACKE_zsytri_work(int matrix_order, + char uplo, + lapack_int n, + lapack_complex_double *a, + lapack_int lda, + const lapack_int *ipiv, + lapack_complex_double *work); + +lapack_int LAPACKE_ssytrs_work(int matrix_order, + char uplo, + lapack_int n, + lapack_int nrhs, + const float *a, + lapack_int lda, + const lapack_int *ipiv, + float *b, + lapack_int ldb); +lapack_int LAPACKE_dsytrs_work(int matrix_order, + char uplo, + lapack_int n, + lapack_int nrhs, + const double *a, + lapack_int lda, + const lapack_int *ipiv, + double *b, + lapack_int ldb); +lapack_int LAPACKE_csytrs_work(int matrix_order, + char uplo, + lapack_int n, + lapack_int nrhs, + const lapack_complex_float *a, + lapack_int lda, + const lapack_int *ipiv, + lapack_complex_float *b, + lapack_int ldb); +lapack_int LAPACKE_zsytrs_work(int matrix_order, + char uplo, + lapack_int n, + lapack_int nrhs, + const lapack_complex_double *a, + lapack_int lda, + const lapack_int *ipiv, + lapack_complex_double *b, + lapack_int ldb); + +lapack_int LAPACKE_stbcon_work(int matrix_order, + char norm, + char uplo, + char diag, + lapack_int n, + lapack_int kd, + const float *ab, + lapack_int ldab, + float *rcond, + float *work, + lapack_int *iwork); +lapack_int LAPACKE_dtbcon_work(int matrix_order, + char norm, + char uplo, + char diag, + lapack_int n, + lapack_int kd, + const double *ab, + lapack_int ldab, + double *rcond, + double *work, + lapack_int *iwork); +lapack_int LAPACKE_ctbcon_work(int matrix_order, + char norm, + char uplo, + char diag, + lapack_int n, + lapack_int kd, + const lapack_complex_float *ab, + lapack_int ldab, + float *rcond, + lapack_complex_float *work, + float *rwork); +lapack_int LAPACKE_ztbcon_work(int matrix_order, + char norm, + char uplo, + char diag, + lapack_int n, + lapack_int kd, + const lapack_complex_double *ab, + lapack_int ldab, + double *rcond, + lapack_complex_double *work, + double *rwork); + +lapack_int LAPACKE_stbrfs_work(int matrix_order, + char uplo, + char trans, + char diag, + lapack_int n, + lapack_int kd, + lapack_int nrhs, + const float *ab, + lapack_int ldab, + const float *b, + lapack_int ldb, + const float *x, + lapack_int ldx, + float *ferr, + float *berr, + float *work, + lapack_int *iwork); +lapack_int LAPACKE_dtbrfs_work(int matrix_order, + char uplo, + char trans, + char diag, + lapack_int n, + lapack_int kd, + lapack_int nrhs, + const double *ab, + lapack_int ldab, + const double *b, + lapack_int ldb, + const double *x, + lapack_int ldx, + double *ferr, + double *berr, + double *work, + lapack_int *iwork); +lapack_int LAPACKE_ctbrfs_work(int matrix_order, + char uplo, + char trans, + char diag, + lapack_int n, + lapack_int kd, + lapack_int nrhs, + const lapack_complex_float *ab, + lapack_int ldab, + const lapack_complex_float *b, + lapack_int ldb, + const lapack_complex_float *x, + lapack_int ldx, + float *ferr, + float *berr, + lapack_complex_float *work, + float *rwork); +lapack_int LAPACKE_ztbrfs_work(int matrix_order, + char uplo, + char trans, + char diag, + lapack_int n, + lapack_int kd, + lapack_int nrhs, + const lapack_complex_double *ab, + lapack_int ldab, + const lapack_complex_double *b, + lapack_int ldb, + const lapack_complex_double *x, + lapack_int ldx, + double *ferr, + double *berr, + lapack_complex_double *work, + double *rwork); + +lapack_int LAPACKE_stbtrs_work(int matrix_order, + char uplo, + char trans, + char diag, + lapack_int n, + lapack_int kd, + lapack_int nrhs, + const float *ab, + lapack_int ldab, + float *b, + lapack_int ldb); +lapack_int LAPACKE_dtbtrs_work(int matrix_order, + char uplo, + char trans, + char diag, + lapack_int n, + lapack_int kd, + lapack_int nrhs, + const double *ab, + lapack_int ldab, + double *b, + lapack_int ldb); +lapack_int LAPACKE_ctbtrs_work(int matrix_order, + char uplo, + char trans, + char diag, + lapack_int n, + lapack_int kd, + lapack_int nrhs, + const lapack_complex_float *ab, + lapack_int ldab, + lapack_complex_float *b, + lapack_int ldb); +lapack_int LAPACKE_ztbtrs_work(int matrix_order, + char uplo, + char trans, + char diag, + lapack_int n, + lapack_int kd, + lapack_int nrhs, + const lapack_complex_double *ab, + lapack_int ldab, + lapack_complex_double *b, + lapack_int ldb); + +lapack_int LAPACKE_stfsm_work(int matrix_order, + char transr, + char side, + char uplo, + char trans, + char diag, + lapack_int m, + lapack_int n, + float alpha, + const float *a, + float *b, + lapack_int ldb); +lapack_int LAPACKE_dtfsm_work(int matrix_order, + char transr, + char side, + char uplo, + char trans, + char diag, + lapack_int m, + lapack_int n, + double alpha, + const double *a, + double *b, + lapack_int ldb); +lapack_int LAPACKE_ctfsm_work(int matrix_order, + char transr, + char side, + char uplo, + char trans, + char diag, + lapack_int m, + lapack_int n, + lapack_complex_float alpha, + const lapack_complex_float *a, + lapack_complex_float *b, + lapack_int ldb); +lapack_int LAPACKE_ztfsm_work(int matrix_order, + char transr, + char side, + char uplo, + char trans, + char diag, + lapack_int m, + lapack_int n, + lapack_complex_double alpha, + const lapack_complex_double *a, + lapack_complex_double *b, + lapack_int ldb); + +lapack_int LAPACKE_stftri_work(int matrix_order, char transr, char uplo, char diag, lapack_int n, float *a); +lapack_int LAPACKE_dtftri_work(int matrix_order, char transr, char uplo, char diag, lapack_int n, double *a); +lapack_int + LAPACKE_ctftri_work(int matrix_order, char transr, char uplo, char diag, lapack_int n, lapack_complex_float *a); +lapack_int + LAPACKE_ztftri_work(int matrix_order, char transr, char uplo, char diag, lapack_int n, lapack_complex_double *a); + +lapack_int LAPACKE_stfttp_work(int matrix_order, char transr, char uplo, lapack_int n, const float *arf, float *ap); +lapack_int LAPACKE_dtfttp_work(int matrix_order, char transr, char uplo, lapack_int n, const double *arf, double *ap); +lapack_int LAPACKE_ctfttp_work(int matrix_order, + char transr, + char uplo, + lapack_int n, + const lapack_complex_float *arf, + lapack_complex_float *ap); +lapack_int LAPACKE_ztfttp_work(int matrix_order, + char transr, + char uplo, + lapack_int n, + const lapack_complex_double *arf, + lapack_complex_double *ap); + +lapack_int LAPACKE_stfttr_work(int matrix_order, + char transr, + char uplo, + lapack_int n, + const float *arf, + float *a, + lapack_int lda); +lapack_int LAPACKE_dtfttr_work(int matrix_order, + char transr, + char uplo, + lapack_int n, + const double *arf, + double *a, + lapack_int lda); +lapack_int LAPACKE_ctfttr_work(int matrix_order, + char transr, + char uplo, + lapack_int n, + const lapack_complex_float *arf, + lapack_complex_float *a, + lapack_int lda); +lapack_int LAPACKE_ztfttr_work(int matrix_order, + char transr, + char uplo, + lapack_int n, + const lapack_complex_double *arf, + lapack_complex_double *a, + lapack_int lda); + +lapack_int LAPACKE_stgevc_work(int matrix_order, + char side, + char howmny, + const lapack_logical *select, + lapack_int n, + const float *s, + lapack_int lds, + const float *p, + lapack_int ldp, + float *vl, + lapack_int ldvl, + float *vr, + lapack_int ldvr, + lapack_int mm, + lapack_int *m, + float *work); +lapack_int LAPACKE_dtgevc_work(int matrix_order, + char side, + char howmny, + const lapack_logical *select, + lapack_int n, + const double *s, + lapack_int lds, + const double *p, + lapack_int ldp, + double *vl, + lapack_int ldvl, + double *vr, + lapack_int ldvr, + lapack_int mm, + lapack_int *m, + double *work); +lapack_int LAPACKE_ctgevc_work(int matrix_order, + char side, + char howmny, + const lapack_logical *select, + lapack_int n, + const lapack_complex_float *s, + lapack_int lds, + const lapack_complex_float *p, + lapack_int ldp, + lapack_complex_float *vl, + lapack_int ldvl, + lapack_complex_float *vr, + lapack_int ldvr, + lapack_int mm, + lapack_int *m, + lapack_complex_float *work, + float *rwork); +lapack_int LAPACKE_ztgevc_work(int matrix_order, + char side, + char howmny, + const lapack_logical *select, + lapack_int n, + const lapack_complex_double *s, + lapack_int lds, + const lapack_complex_double *p, + lapack_int ldp, + lapack_complex_double *vl, + lapack_int ldvl, + lapack_complex_double *vr, + lapack_int ldvr, + lapack_int mm, + lapack_int *m, + lapack_complex_double *work, + double *rwork); + +lapack_int LAPACKE_stgexc_work(int matrix_order, + lapack_logical wantq, + lapack_logical wantz, + lapack_int n, + float *a, + lapack_int lda, + float *b, + lapack_int ldb, + float *q, + lapack_int ldq, + float *z, + lapack_int ldz, + lapack_int *ifst, + lapack_int *ilst, + float *work, + lapack_int lwork); +lapack_int LAPACKE_dtgexc_work(int matrix_order, + lapack_logical wantq, + lapack_logical wantz, + lapack_int n, + double *a, + lapack_int lda, + double *b, + lapack_int ldb, + double *q, + lapack_int ldq, + double *z, + lapack_int ldz, + lapack_int *ifst, + lapack_int *ilst, + double *work, + lapack_int lwork); +lapack_int LAPACKE_ctgexc_work(int matrix_order, + lapack_logical wantq, + lapack_logical wantz, + lapack_int n, + lapack_complex_float *a, + lapack_int lda, + lapack_complex_float *b, + lapack_int ldb, + lapack_complex_float *q, + lapack_int ldq, + lapack_complex_float *z, + lapack_int ldz, + lapack_int ifst, + lapack_int ilst); +lapack_int LAPACKE_ztgexc_work(int matrix_order, + lapack_logical wantq, + lapack_logical wantz, + lapack_int n, + lapack_complex_double *a, + lapack_int lda, + lapack_complex_double *b, + lapack_int ldb, + lapack_complex_double *q, + lapack_int ldq, + lapack_complex_double *z, + lapack_int ldz, + lapack_int ifst, + lapack_int ilst); + +lapack_int LAPACKE_stgsen_work(int matrix_order, + lapack_int ijob, + lapack_logical wantq, + lapack_logical wantz, + const lapack_logical *select, + lapack_int n, + float *a, + lapack_int lda, + float *b, + lapack_int ldb, + float *alphar, + float *alphai, + float *beta, + float *q, + lapack_int ldq, + float *z, + lapack_int ldz, + lapack_int *m, + float *pl, + float *pr, + float *dif, + float *work, + lapack_int lwork, + lapack_int *iwork, + lapack_int liwork); +lapack_int LAPACKE_dtgsen_work(int matrix_order, + lapack_int ijob, + lapack_logical wantq, + lapack_logical wantz, + const lapack_logical *select, + lapack_int n, + double *a, + lapack_int lda, + double *b, + lapack_int ldb, + double *alphar, + double *alphai, + double *beta, + double *q, + lapack_int ldq, + double *z, + lapack_int ldz, + lapack_int *m, + double *pl, + double *pr, + double *dif, + double *work, + lapack_int lwork, + lapack_int *iwork, + lapack_int liwork); +lapack_int LAPACKE_ctgsen_work(int matrix_order, + lapack_int ijob, + lapack_logical wantq, + lapack_logical wantz, + const lapack_logical *select, + lapack_int n, + lapack_complex_float *a, + lapack_int lda, + lapack_complex_float *b, + lapack_int ldb, + lapack_complex_float *alpha, + lapack_complex_float *beta, + lapack_complex_float *q, + lapack_int ldq, + lapack_complex_float *z, + lapack_int ldz, + lapack_int *m, + float *pl, + float *pr, + float *dif, + lapack_complex_float *work, + lapack_int lwork, + lapack_int *iwork, + lapack_int liwork); +lapack_int LAPACKE_ztgsen_work(int matrix_order, + lapack_int ijob, + lapack_logical wantq, + lapack_logical wantz, + const lapack_logical *select, + lapack_int n, + lapack_complex_double *a, + lapack_int lda, + lapack_complex_double *b, + lapack_int ldb, + lapack_complex_double *alpha, + lapack_complex_double *beta, + lapack_complex_double *q, + lapack_int ldq, + lapack_complex_double *z, + lapack_int ldz, + lapack_int *m, + double *pl, + double *pr, + double *dif, + lapack_complex_double *work, + lapack_int lwork, + lapack_int *iwork, + lapack_int liwork); + +lapack_int LAPACKE_stgsja_work(int matrix_order, + char jobu, + char jobv, + char jobq, + lapack_int m, + lapack_int p, + lapack_int n, + lapack_int k, + lapack_int l, + float *a, + lapack_int lda, + float *b, + lapack_int ldb, + float tola, + float tolb, + float *alpha, + float *beta, + float *u, + lapack_int ldu, + float *v, + lapack_int ldv, + float *q, + lapack_int ldq, + float *work, + lapack_int *ncycle); +lapack_int LAPACKE_dtgsja_work(int matrix_order, + char jobu, + char jobv, + char jobq, + lapack_int m, + lapack_int p, + lapack_int n, + lapack_int k, + lapack_int l, + double *a, + lapack_int lda, + double *b, + lapack_int ldb, + double tola, + double tolb, + double *alpha, + double *beta, + double *u, + lapack_int ldu, + double *v, + lapack_int ldv, + double *q, + lapack_int ldq, + double *work, + lapack_int *ncycle); +lapack_int LAPACKE_ctgsja_work(int matrix_order, + char jobu, + char jobv, + char jobq, + lapack_int m, + lapack_int p, + lapack_int n, + lapack_int k, + lapack_int l, + lapack_complex_float *a, + lapack_int lda, + lapack_complex_float *b, + lapack_int ldb, + float tola, + float tolb, + float *alpha, + float *beta, + lapack_complex_float *u, + lapack_int ldu, + lapack_complex_float *v, + lapack_int ldv, + lapack_complex_float *q, + lapack_int ldq, + lapack_complex_float *work, + lapack_int *ncycle); +lapack_int LAPACKE_ztgsja_work(int matrix_order, + char jobu, + char jobv, + char jobq, + lapack_int m, + lapack_int p, + lapack_int n, + lapack_int k, + lapack_int l, + lapack_complex_double *a, + lapack_int lda, + lapack_complex_double *b, + lapack_int ldb, + double tola, + double tolb, + double *alpha, + double *beta, + lapack_complex_double *u, + lapack_int ldu, + lapack_complex_double *v, + lapack_int ldv, + lapack_complex_double *q, + lapack_int ldq, + lapack_complex_double *work, + lapack_int *ncycle); + +lapack_int LAPACKE_stgsna_work(int matrix_order, + char job, + char howmny, + const lapack_logical *select, + lapack_int n, + const float *a, + lapack_int lda, + const float *b, + lapack_int ldb, + const float *vl, + lapack_int ldvl, + const float *vr, + lapack_int ldvr, + float *s, + float *dif, + lapack_int mm, + lapack_int *m, + float *work, + lapack_int lwork, + lapack_int *iwork); +lapack_int LAPACKE_dtgsna_work(int matrix_order, + char job, + char howmny, + const lapack_logical *select, + lapack_int n, + const double *a, + lapack_int lda, + const double *b, + lapack_int ldb, + const double *vl, + lapack_int ldvl, + const double *vr, + lapack_int ldvr, + double *s, + double *dif, + lapack_int mm, + lapack_int *m, + double *work, + lapack_int lwork, + lapack_int *iwork); +lapack_int LAPACKE_ctgsna_work(int matrix_order, + char job, + char howmny, + const lapack_logical *select, + lapack_int n, + const lapack_complex_float *a, + lapack_int lda, + const lapack_complex_float *b, + lapack_int ldb, + const lapack_complex_float *vl, + lapack_int ldvl, + const lapack_complex_float *vr, + lapack_int ldvr, + float *s, + float *dif, + lapack_int mm, + lapack_int *m, + lapack_complex_float *work, + lapack_int lwork, + lapack_int *iwork); +lapack_int LAPACKE_ztgsna_work(int matrix_order, + char job, + char howmny, + const lapack_logical *select, + lapack_int n, + const lapack_complex_double *a, + lapack_int lda, + const lapack_complex_double *b, + lapack_int ldb, + const lapack_complex_double *vl, + lapack_int ldvl, + const lapack_complex_double *vr, + lapack_int ldvr, + double *s, + double *dif, + lapack_int mm, + lapack_int *m, + lapack_complex_double *work, + lapack_int lwork, + lapack_int *iwork); + +lapack_int LAPACKE_stgsyl_work(int matrix_order, + char trans, + lapack_int ijob, + lapack_int m, + lapack_int n, + const float *a, + lapack_int lda, + const float *b, + lapack_int ldb, + float *c, + lapack_int ldc, + const float *d, + lapack_int ldd, + const float *e, + lapack_int lde, + float *f, + lapack_int ldf, + float *scale, + float *dif, + float *work, + lapack_int lwork, + lapack_int *iwork); +lapack_int LAPACKE_dtgsyl_work(int matrix_order, + char trans, + lapack_int ijob, + lapack_int m, + lapack_int n, + const double *a, + lapack_int lda, + const double *b, + lapack_int ldb, + double *c, + lapack_int ldc, + const double *d, + lapack_int ldd, + const double *e, + lapack_int lde, + double *f, + lapack_int ldf, + double *scale, + double *dif, + double *work, + lapack_int lwork, + lapack_int *iwork); +lapack_int LAPACKE_ctgsyl_work(int matrix_order, + char trans, + lapack_int ijob, + lapack_int m, + lapack_int n, + const lapack_complex_float *a, + lapack_int lda, + const lapack_complex_float *b, + lapack_int ldb, + lapack_complex_float *c, + lapack_int ldc, + const lapack_complex_float *d, + lapack_int ldd, + const lapack_complex_float *e, + lapack_int lde, + lapack_complex_float *f, + lapack_int ldf, + float *scale, + float *dif, + lapack_complex_float *work, + lapack_int lwork, + lapack_int *iwork); +lapack_int LAPACKE_ztgsyl_work(int matrix_order, + char trans, + lapack_int ijob, + lapack_int m, + lapack_int n, + const lapack_complex_double *a, + lapack_int lda, + const lapack_complex_double *b, + lapack_int ldb, + lapack_complex_double *c, + lapack_int ldc, + const lapack_complex_double *d, + lapack_int ldd, + const lapack_complex_double *e, + lapack_int lde, + lapack_complex_double *f, + lapack_int ldf, + double *scale, + double *dif, + lapack_complex_double *work, + lapack_int lwork, + lapack_int *iwork); + +lapack_int LAPACKE_stpcon_work(int matrix_order, + char norm, + char uplo, + char diag, + lapack_int n, + const float *ap, + float *rcond, + float *work, + lapack_int *iwork); +lapack_int LAPACKE_dtpcon_work(int matrix_order, + char norm, + char uplo, + char diag, + lapack_int n, + const double *ap, + double *rcond, + double *work, + lapack_int *iwork); +lapack_int LAPACKE_ctpcon_work(int matrix_order, + char norm, + char uplo, + char diag, + lapack_int n, + const lapack_complex_float *ap, + float *rcond, + lapack_complex_float *work, + float *rwork); +lapack_int LAPACKE_ztpcon_work(int matrix_order, + char norm, + char uplo, + char diag, + lapack_int n, + const lapack_complex_double *ap, + double *rcond, + lapack_complex_double *work, + double *rwork); + +lapack_int LAPACKE_stprfs_work(int matrix_order, + char uplo, + char trans, + char diag, + lapack_int n, + lapack_int nrhs, + const float *ap, + const float *b, + lapack_int ldb, + const float *x, + lapack_int ldx, + float *ferr, + float *berr, + float *work, + lapack_int *iwork); +lapack_int LAPACKE_dtprfs_work(int matrix_order, + char uplo, + char trans, + char diag, + lapack_int n, + lapack_int nrhs, + const double *ap, + const double *b, + lapack_int ldb, + const double *x, + lapack_int ldx, + double *ferr, + double *berr, + double *work, + lapack_int *iwork); +lapack_int LAPACKE_ctprfs_work(int matrix_order, + char uplo, + char trans, + char diag, + lapack_int n, + lapack_int nrhs, + const lapack_complex_float *ap, + const lapack_complex_float *b, + lapack_int ldb, + const lapack_complex_float *x, + lapack_int ldx, + float *ferr, + float *berr, + lapack_complex_float *work, + float *rwork); +lapack_int LAPACKE_ztprfs_work(int matrix_order, + char uplo, + char trans, + char diag, + lapack_int n, + lapack_int nrhs, + const lapack_complex_double *ap, + const lapack_complex_double *b, + lapack_int ldb, + const lapack_complex_double *x, + lapack_int ldx, + double *ferr, + double *berr, + lapack_complex_double *work, + double *rwork); + +lapack_int LAPACKE_stptri_work(int matrix_order, char uplo, char diag, lapack_int n, float *ap); +lapack_int LAPACKE_dtptri_work(int matrix_order, char uplo, char diag, lapack_int n, double *ap); +lapack_int LAPACKE_ctptri_work(int matrix_order, char uplo, char diag, lapack_int n, lapack_complex_float *ap); +lapack_int LAPACKE_ztptri_work(int matrix_order, char uplo, char diag, lapack_int n, lapack_complex_double *ap); + +lapack_int LAPACKE_stptrs_work(int matrix_order, + char uplo, + char trans, + char diag, + lapack_int n, + lapack_int nrhs, + const float *ap, + float *b, + lapack_int ldb); +lapack_int LAPACKE_dtptrs_work(int matrix_order, + char uplo, + char trans, + char diag, + lapack_int n, + lapack_int nrhs, + const double *ap, + double *b, + lapack_int ldb); +lapack_int LAPACKE_ctptrs_work(int matrix_order, + char uplo, + char trans, + char diag, + lapack_int n, + lapack_int nrhs, + const lapack_complex_float *ap, + lapack_complex_float *b, + lapack_int ldb); +lapack_int LAPACKE_ztptrs_work(int matrix_order, + char uplo, + char trans, + char diag, + lapack_int n, + lapack_int nrhs, + const lapack_complex_double *ap, + lapack_complex_double *b, + lapack_int ldb); + +lapack_int LAPACKE_stpttf_work(int matrix_order, char transr, char uplo, lapack_int n, const float *ap, float *arf); +lapack_int LAPACKE_dtpttf_work(int matrix_order, char transr, char uplo, lapack_int n, const double *ap, double *arf); +lapack_int LAPACKE_ctpttf_work(int matrix_order, + char transr, + char uplo, + lapack_int n, + const lapack_complex_float *ap, + lapack_complex_float *arf); +lapack_int LAPACKE_ztpttf_work(int matrix_order, + char transr, + char uplo, + lapack_int n, + const lapack_complex_double *ap, + lapack_complex_double *arf); + +lapack_int LAPACKE_stpttr_work(int matrix_order, char uplo, lapack_int n, const float *ap, float *a, lapack_int lda); +lapack_int LAPACKE_dtpttr_work(int matrix_order, char uplo, lapack_int n, const double *ap, double *a, lapack_int lda); +lapack_int LAPACKE_ctpttr_work(int matrix_order, + char uplo, + lapack_int n, + const lapack_complex_float *ap, + lapack_complex_float *a, + lapack_int lda); +lapack_int LAPACKE_ztpttr_work(int matrix_order, + char uplo, + lapack_int n, + const lapack_complex_double *ap, + lapack_complex_double *a, + lapack_int lda); + +lapack_int LAPACKE_strcon_work(int matrix_order, + char norm, + char uplo, + char diag, + lapack_int n, + const float *a, + lapack_int lda, + float *rcond, + float *work, + lapack_int *iwork); +lapack_int LAPACKE_dtrcon_work(int matrix_order, + char norm, + char uplo, + char diag, + lapack_int n, + const double *a, + lapack_int lda, + double *rcond, + double *work, + lapack_int *iwork); +lapack_int LAPACKE_ctrcon_work(int matrix_order, + char norm, + char uplo, + char diag, + lapack_int n, + const lapack_complex_float *a, + lapack_int lda, + float *rcond, + lapack_complex_float *work, + float *rwork); +lapack_int LAPACKE_ztrcon_work(int matrix_order, + char norm, + char uplo, + char diag, + lapack_int n, + const lapack_complex_double *a, + lapack_int lda, + double *rcond, + lapack_complex_double *work, + double *rwork); + +lapack_int LAPACKE_strevc_work(int matrix_order, + char side, + char howmny, + lapack_logical *select, + lapack_int n, + const float *t, + lapack_int ldt, + float *vl, + lapack_int ldvl, + float *vr, + lapack_int ldvr, + lapack_int mm, + lapack_int *m, + float *work); +lapack_int LAPACKE_dtrevc_work(int matrix_order, + char side, + char howmny, + lapack_logical *select, + lapack_int n, + const double *t, + lapack_int ldt, + double *vl, + lapack_int ldvl, + double *vr, + lapack_int ldvr, + lapack_int mm, + lapack_int *m, + double *work); +lapack_int LAPACKE_ctrevc_work(int matrix_order, + char side, + char howmny, + const lapack_logical *select, + lapack_int n, + lapack_complex_float *t, + lapack_int ldt, + lapack_complex_float *vl, + lapack_int ldvl, + lapack_complex_float *vr, + lapack_int ldvr, + lapack_int mm, + lapack_int *m, + lapack_complex_float *work, + float *rwork); +lapack_int LAPACKE_ztrevc_work(int matrix_order, + char side, + char howmny, + const lapack_logical *select, + lapack_int n, + lapack_complex_double *t, + lapack_int ldt, + lapack_complex_double *vl, + lapack_int ldvl, + lapack_complex_double *vr, + lapack_int ldvr, + lapack_int mm, + lapack_int *m, + lapack_complex_double *work, + double *rwork); + +lapack_int LAPACKE_strexc_work(int matrix_order, + char compq, + lapack_int n, + float *t, + lapack_int ldt, + float *q, + lapack_int ldq, + lapack_int *ifst, + lapack_int *ilst, + float *work); +lapack_int LAPACKE_dtrexc_work(int matrix_order, + char compq, + lapack_int n, + double *t, + lapack_int ldt, + double *q, + lapack_int ldq, + lapack_int *ifst, + lapack_int *ilst, + double *work); +lapack_int LAPACKE_ctrexc_work(int matrix_order, + char compq, + lapack_int n, + lapack_complex_float *t, + lapack_int ldt, + lapack_complex_float *q, + lapack_int ldq, + lapack_int ifst, + lapack_int ilst); +lapack_int LAPACKE_ztrexc_work(int matrix_order, + char compq, + lapack_int n, + lapack_complex_double *t, + lapack_int ldt, + lapack_complex_double *q, + lapack_int ldq, + lapack_int ifst, + lapack_int ilst); + +lapack_int LAPACKE_strrfs_work(int matrix_order, + char uplo, + char trans, + char diag, + lapack_int n, + lapack_int nrhs, + const float *a, + lapack_int lda, + const float *b, + lapack_int ldb, + const float *x, + lapack_int ldx, + float *ferr, + float *berr, + float *work, + lapack_int *iwork); +lapack_int LAPACKE_dtrrfs_work(int matrix_order, + char uplo, + char trans, + char diag, + lapack_int n, + lapack_int nrhs, + const double *a, + lapack_int lda, + const double *b, + lapack_int ldb, + const double *x, + lapack_int ldx, + double *ferr, + double *berr, + double *work, + lapack_int *iwork); +lapack_int LAPACKE_ctrrfs_work(int matrix_order, + char uplo, + char trans, + char diag, + lapack_int n, + lapack_int nrhs, + const lapack_complex_float *a, + lapack_int lda, + const lapack_complex_float *b, + lapack_int ldb, + const lapack_complex_float *x, + lapack_int ldx, + float *ferr, + float *berr, + lapack_complex_float *work, + float *rwork); +lapack_int LAPACKE_ztrrfs_work(int matrix_order, + char uplo, + char trans, + char diag, + lapack_int n, + lapack_int nrhs, + const lapack_complex_double *a, + lapack_int lda, + const lapack_complex_double *b, + lapack_int ldb, + const lapack_complex_double *x, + lapack_int ldx, + double *ferr, + double *berr, + lapack_complex_double *work, + double *rwork); + +lapack_int LAPACKE_strsen_work(int matrix_order, + char job, + char compq, + const lapack_logical *select, + lapack_int n, + float *t, + lapack_int ldt, + float *q, + lapack_int ldq, + float *wr, + float *wi, + lapack_int *m, + float *s, + float *sep, + float *work, + lapack_int lwork, + lapack_int *iwork, + lapack_int liwork); +lapack_int LAPACKE_dtrsen_work(int matrix_order, + char job, + char compq, + const lapack_logical *select, + lapack_int n, + double *t, + lapack_int ldt, + double *q, + lapack_int ldq, + double *wr, + double *wi, + lapack_int *m, + double *s, + double *sep, + double *work, + lapack_int lwork, + lapack_int *iwork, + lapack_int liwork); +lapack_int LAPACKE_ctrsen_work(int matrix_order, + char job, + char compq, + const lapack_logical *select, + lapack_int n, + lapack_complex_float *t, + lapack_int ldt, + lapack_complex_float *q, + lapack_int ldq, + lapack_complex_float *w, + lapack_int *m, + float *s, + float *sep, + lapack_complex_float *work, + lapack_int lwork); +lapack_int LAPACKE_ztrsen_work(int matrix_order, + char job, + char compq, + const lapack_logical *select, + lapack_int n, + lapack_complex_double *t, + lapack_int ldt, + lapack_complex_double *q, + lapack_int ldq, + lapack_complex_double *w, + lapack_int *m, + double *s, + double *sep, + lapack_complex_double *work, + lapack_int lwork); + +lapack_int LAPACKE_strsna_work(int matrix_order, + char job, + char howmny, + const lapack_logical *select, + lapack_int n, + const float *t, + lapack_int ldt, + const float *vl, + lapack_int ldvl, + const float *vr, + lapack_int ldvr, + float *s, + float *sep, + lapack_int mm, + lapack_int *m, + float *work, + lapack_int ldwork, + lapack_int *iwork); +lapack_int LAPACKE_dtrsna_work(int matrix_order, + char job, + char howmny, + const lapack_logical *select, + lapack_int n, + const double *t, + lapack_int ldt, + const double *vl, + lapack_int ldvl, + const double *vr, + lapack_int ldvr, + double *s, + double *sep, + lapack_int mm, + lapack_int *m, + double *work, + lapack_int ldwork, + lapack_int *iwork); +lapack_int LAPACKE_ctrsna_work(int matrix_order, + char job, + char howmny, + const lapack_logical *select, + lapack_int n, + const lapack_complex_float *t, + lapack_int ldt, + const lapack_complex_float *vl, + lapack_int ldvl, + const lapack_complex_float *vr, + lapack_int ldvr, + float *s, + float *sep, + lapack_int mm, + lapack_int *m, + lapack_complex_float *work, + lapack_int ldwork, + float *rwork); +lapack_int LAPACKE_ztrsna_work(int matrix_order, + char job, + char howmny, + const lapack_logical *select, + lapack_int n, + const lapack_complex_double *t, + lapack_int ldt, + const lapack_complex_double *vl, + lapack_int ldvl, + const lapack_complex_double *vr, + lapack_int ldvr, + double *s, + double *sep, + lapack_int mm, + lapack_int *m, + lapack_complex_double *work, + lapack_int ldwork, + double *rwork); + +lapack_int LAPACKE_strsyl_work(int matrix_order, + char trana, + char tranb, + lapack_int isgn, + lapack_int m, + lapack_int n, + const float *a, + lapack_int lda, + const float *b, + lapack_int ldb, + float *c, + lapack_int ldc, + float *scale); +lapack_int LAPACKE_dtrsyl_work(int matrix_order, + char trana, + char tranb, + lapack_int isgn, + lapack_int m, + lapack_int n, + const double *a, + lapack_int lda, + const double *b, + lapack_int ldb, + double *c, + lapack_int ldc, + double *scale); +lapack_int LAPACKE_ctrsyl_work(int matrix_order, + char trana, + char tranb, + lapack_int isgn, + lapack_int m, + lapack_int n, + const lapack_complex_float *a, + lapack_int lda, + const lapack_complex_float *b, + lapack_int ldb, + lapack_complex_float *c, + lapack_int ldc, + float *scale); +lapack_int LAPACKE_ztrsyl_work(int matrix_order, + char trana, + char tranb, + lapack_int isgn, + lapack_int m, + lapack_int n, + const lapack_complex_double *a, + lapack_int lda, + const lapack_complex_double *b, + lapack_int ldb, + lapack_complex_double *c, + lapack_int ldc, + double *scale); + +lapack_int LAPACKE_strtri_work(int matrix_order, char uplo, char diag, lapack_int n, float *a, lapack_int lda); +lapack_int LAPACKE_dtrtri_work(int matrix_order, char uplo, char diag, lapack_int n, double *a, lapack_int lda); +lapack_int + LAPACKE_ctrtri_work(int matrix_order, char uplo, char diag, lapack_int n, lapack_complex_float *a, lapack_int lda); +lapack_int + LAPACKE_ztrtri_work(int matrix_order, char uplo, char diag, lapack_int n, lapack_complex_double *a, lapack_int lda); + +lapack_int LAPACKE_strtrs_work(int matrix_order, + char uplo, + char trans, + char diag, + lapack_int n, + lapack_int nrhs, + const float *a, + lapack_int lda, + float *b, + lapack_int ldb); +lapack_int LAPACKE_dtrtrs_work(int matrix_order, + char uplo, + char trans, + char diag, + lapack_int n, + lapack_int nrhs, + const double *a, + lapack_int lda, + double *b, + lapack_int ldb); +lapack_int LAPACKE_ctrtrs_work(int matrix_order, + char uplo, + char trans, + char diag, + lapack_int n, + lapack_int nrhs, + const lapack_complex_float *a, + lapack_int lda, + lapack_complex_float *b, + lapack_int ldb); +lapack_int LAPACKE_ztrtrs_work(int matrix_order, + char uplo, + char trans, + char diag, + lapack_int n, + lapack_int nrhs, + const lapack_complex_double *a, + lapack_int lda, + lapack_complex_double *b, + lapack_int ldb); + +lapack_int LAPACKE_strttf_work(int matrix_order, + char transr, + char uplo, + lapack_int n, + const float *a, + lapack_int lda, + float *arf); +lapack_int LAPACKE_dtrttf_work(int matrix_order, + char transr, + char uplo, + lapack_int n, + const double *a, + lapack_int lda, + double *arf); +lapack_int LAPACKE_ctrttf_work(int matrix_order, + char transr, + char uplo, + lapack_int n, + const lapack_complex_float *a, + lapack_int lda, + lapack_complex_float *arf); +lapack_int LAPACKE_ztrttf_work(int matrix_order, + char transr, + char uplo, + lapack_int n, + const lapack_complex_double *a, + lapack_int lda, + lapack_complex_double *arf); + +lapack_int LAPACKE_strttp_work(int matrix_order, char uplo, lapack_int n, const float *a, lapack_int lda, float *ap); +lapack_int LAPACKE_dtrttp_work(int matrix_order, char uplo, lapack_int n, const double *a, lapack_int lda, double *ap); +lapack_int LAPACKE_ctrttp_work(int matrix_order, + char uplo, + lapack_int n, + const lapack_complex_float *a, + lapack_int lda, + lapack_complex_float *ap); +lapack_int LAPACKE_ztrttp_work(int matrix_order, + char uplo, + lapack_int n, + const lapack_complex_double *a, + lapack_int lda, + lapack_complex_double *ap); + +lapack_int LAPACKE_stzrzf_work(int matrix_order, + lapack_int m, + lapack_int n, + float *a, + lapack_int lda, + float *tau, + float *work, + lapack_int lwork); +lapack_int LAPACKE_dtzrzf_work(int matrix_order, + lapack_int m, + lapack_int n, + double *a, + lapack_int lda, + double *tau, + double *work, + lapack_int lwork); +lapack_int LAPACKE_ctzrzf_work(int matrix_order, + lapack_int m, + lapack_int n, + lapack_complex_float *a, + lapack_int lda, + lapack_complex_float *tau, + lapack_complex_float *work, + lapack_int lwork); +lapack_int LAPACKE_ztzrzf_work(int matrix_order, + lapack_int m, + lapack_int n, + lapack_complex_double *a, + lapack_int lda, + lapack_complex_double *tau, + lapack_complex_double *work, + lapack_int lwork); + +lapack_int LAPACKE_cungbr_work(int matrix_order, + char vect, + lapack_int m, + lapack_int n, + lapack_int k, + lapack_complex_float *a, + lapack_int lda, + const lapack_complex_float *tau, + lapack_complex_float *work, + lapack_int lwork); +lapack_int LAPACKE_zungbr_work(int matrix_order, + char vect, + lapack_int m, + lapack_int n, + lapack_int k, + lapack_complex_double *a, + lapack_int lda, + const lapack_complex_double *tau, + lapack_complex_double *work, + lapack_int lwork); + +lapack_int LAPACKE_cunghr_work(int matrix_order, + lapack_int n, + lapack_int ilo, + lapack_int ihi, + lapack_complex_float *a, + lapack_int lda, + const lapack_complex_float *tau, + lapack_complex_float *work, + lapack_int lwork); +lapack_int LAPACKE_zunghr_work(int matrix_order, + lapack_int n, + lapack_int ilo, + lapack_int ihi, + lapack_complex_double *a, + lapack_int lda, + const lapack_complex_double *tau, + lapack_complex_double *work, + lapack_int lwork); + +lapack_int LAPACKE_cunglq_work(int matrix_order, + lapack_int m, + lapack_int n, + lapack_int k, + lapack_complex_float *a, + lapack_int lda, + const lapack_complex_float *tau, + lapack_complex_float *work, + lapack_int lwork); +lapack_int LAPACKE_zunglq_work(int matrix_order, + lapack_int m, + lapack_int n, + lapack_int k, + lapack_complex_double *a, + lapack_int lda, + const lapack_complex_double *tau, + lapack_complex_double *work, + lapack_int lwork); + +lapack_int LAPACKE_cungql_work(int matrix_order, + lapack_int m, + lapack_int n, + lapack_int k, + lapack_complex_float *a, + lapack_int lda, + const lapack_complex_float *tau, + lapack_complex_float *work, + lapack_int lwork); +lapack_int LAPACKE_zungql_work(int matrix_order, + lapack_int m, + lapack_int n, + lapack_int k, + lapack_complex_double *a, + lapack_int lda, + const lapack_complex_double *tau, + lapack_complex_double *work, + lapack_int lwork); + +lapack_int LAPACKE_cungqr_work(int matrix_order, + lapack_int m, + lapack_int n, + lapack_int k, + lapack_complex_float *a, + lapack_int lda, + const lapack_complex_float *tau, + lapack_complex_float *work, + lapack_int lwork); +lapack_int LAPACKE_zungqr_work(int matrix_order, + lapack_int m, + lapack_int n, + lapack_int k, + lapack_complex_double *a, + lapack_int lda, + const lapack_complex_double *tau, + lapack_complex_double *work, + lapack_int lwork); + +lapack_int LAPACKE_cungrq_work(int matrix_order, + lapack_int m, + lapack_int n, + lapack_int k, + lapack_complex_float *a, + lapack_int lda, + const lapack_complex_float *tau, + lapack_complex_float *work, + lapack_int lwork); +lapack_int LAPACKE_zungrq_work(int matrix_order, + lapack_int m, + lapack_int n, + lapack_int k, + lapack_complex_double *a, + lapack_int lda, + const lapack_complex_double *tau, + lapack_complex_double *work, + lapack_int lwork); + +lapack_int LAPACKE_cungtr_work(int matrix_order, + char uplo, + lapack_int n, + lapack_complex_float *a, + lapack_int lda, + const lapack_complex_float *tau, + lapack_complex_float *work, + lapack_int lwork); +lapack_int LAPACKE_zungtr_work(int matrix_order, + char uplo, + lapack_int n, + lapack_complex_double *a, + lapack_int lda, + const lapack_complex_double *tau, + lapack_complex_double *work, + lapack_int lwork); + +lapack_int LAPACKE_cunmbr_work(int matrix_order, + char vect, + char side, + char trans, + lapack_int m, + lapack_int n, + lapack_int k, + const lapack_complex_float *a, + lapack_int lda, + const lapack_complex_float *tau, + lapack_complex_float *c, + lapack_int ldc, + lapack_complex_float *work, + lapack_int lwork); +lapack_int LAPACKE_zunmbr_work(int matrix_order, + char vect, + char side, + char trans, + lapack_int m, + lapack_int n, + lapack_int k, + const lapack_complex_double *a, + lapack_int lda, + const lapack_complex_double *tau, + lapack_complex_double *c, + lapack_int ldc, + lapack_complex_double *work, + lapack_int lwork); + +lapack_int LAPACKE_cunmhr_work(int matrix_order, + char side, + char trans, + lapack_int m, + lapack_int n, + lapack_int ilo, + lapack_int ihi, + const lapack_complex_float *a, + lapack_int lda, + const lapack_complex_float *tau, + lapack_complex_float *c, + lapack_int ldc, + lapack_complex_float *work, + lapack_int lwork); +lapack_int LAPACKE_zunmhr_work(int matrix_order, + char side, + char trans, + lapack_int m, + lapack_int n, + lapack_int ilo, + lapack_int ihi, + const lapack_complex_double *a, + lapack_int lda, + const lapack_complex_double *tau, + lapack_complex_double *c, + lapack_int ldc, + lapack_complex_double *work, + lapack_int lwork); + +lapack_int LAPACKE_cunmlq_work(int matrix_order, + char side, + char trans, + lapack_int m, + lapack_int n, + lapack_int k, + const lapack_complex_float *a, + lapack_int lda, + const lapack_complex_float *tau, + lapack_complex_float *c, + lapack_int ldc, + lapack_complex_float *work, + lapack_int lwork); +lapack_int LAPACKE_zunmlq_work(int matrix_order, + char side, + char trans, + lapack_int m, + lapack_int n, + lapack_int k, + const lapack_complex_double *a, + lapack_int lda, + const lapack_complex_double *tau, + lapack_complex_double *c, + lapack_int ldc, + lapack_complex_double *work, + lapack_int lwork); + +lapack_int LAPACKE_cunmql_work(int matrix_order, + char side, + char trans, + lapack_int m, + lapack_int n, + lapack_int k, + const lapack_complex_float *a, + lapack_int lda, + const lapack_complex_float *tau, + lapack_complex_float *c, + lapack_int ldc, + lapack_complex_float *work, + lapack_int lwork); +lapack_int LAPACKE_zunmql_work(int matrix_order, + char side, + char trans, + lapack_int m, + lapack_int n, + lapack_int k, + const lapack_complex_double *a, + lapack_int lda, + const lapack_complex_double *tau, + lapack_complex_double *c, + lapack_int ldc, + lapack_complex_double *work, + lapack_int lwork); + +lapack_int LAPACKE_cunmqr_work(int matrix_order, + char side, + char trans, + lapack_int m, + lapack_int n, + lapack_int k, + const lapack_complex_float *a, + lapack_int lda, + const lapack_complex_float *tau, + lapack_complex_float *c, + lapack_int ldc, + lapack_complex_float *work, + lapack_int lwork); +lapack_int LAPACKE_zunmqr_work(int matrix_order, + char side, + char trans, + lapack_int m, + lapack_int n, + lapack_int k, + const lapack_complex_double *a, + lapack_int lda, + const lapack_complex_double *tau, + lapack_complex_double *c, + lapack_int ldc, + lapack_complex_double *work, + lapack_int lwork); + +lapack_int LAPACKE_cunmrq_work(int matrix_order, + char side, + char trans, + lapack_int m, + lapack_int n, + lapack_int k, + const lapack_complex_float *a, + lapack_int lda, + const lapack_complex_float *tau, + lapack_complex_float *c, + lapack_int ldc, + lapack_complex_float *work, + lapack_int lwork); +lapack_int LAPACKE_zunmrq_work(int matrix_order, + char side, + char trans, + lapack_int m, + lapack_int n, + lapack_int k, + const lapack_complex_double *a, + lapack_int lda, + const lapack_complex_double *tau, + lapack_complex_double *c, + lapack_int ldc, + lapack_complex_double *work, + lapack_int lwork); + +lapack_int LAPACKE_cunmrz_work(int matrix_order, + char side, + char trans, + lapack_int m, + lapack_int n, + lapack_int k, + lapack_int l, + const lapack_complex_float *a, + lapack_int lda, + const lapack_complex_float *tau, + lapack_complex_float *c, + lapack_int ldc, + lapack_complex_float *work, + lapack_int lwork); +lapack_int LAPACKE_zunmrz_work(int matrix_order, + char side, + char trans, + lapack_int m, + lapack_int n, + lapack_int k, + lapack_int l, + const lapack_complex_double *a, + lapack_int lda, + const lapack_complex_double *tau, + lapack_complex_double *c, + lapack_int ldc, + lapack_complex_double *work, + lapack_int lwork); + +lapack_int LAPACKE_cunmtr_work(int matrix_order, + char side, + char uplo, + char trans, + lapack_int m, + lapack_int n, + const lapack_complex_float *a, + lapack_int lda, + const lapack_complex_float *tau, + lapack_complex_float *c, + lapack_int ldc, + lapack_complex_float *work, + lapack_int lwork); +lapack_int LAPACKE_zunmtr_work(int matrix_order, + char side, + char uplo, + char trans, + lapack_int m, + lapack_int n, + const lapack_complex_double *a, + lapack_int lda, + const lapack_complex_double *tau, + lapack_complex_double *c, + lapack_int ldc, + lapack_complex_double *work, + lapack_int lwork); + +lapack_int LAPACKE_cupgtr_work(int matrix_order, + char uplo, + lapack_int n, + const lapack_complex_float *ap, + const lapack_complex_float *tau, + lapack_complex_float *q, + lapack_int ldq, + lapack_complex_float *work); +lapack_int LAPACKE_zupgtr_work(int matrix_order, + char uplo, + lapack_int n, + const lapack_complex_double *ap, + const lapack_complex_double *tau, + lapack_complex_double *q, + lapack_int ldq, + lapack_complex_double *work); + +lapack_int LAPACKE_cupmtr_work(int matrix_order, + char side, + char uplo, + char trans, + lapack_int m, + lapack_int n, + const lapack_complex_float *ap, + const lapack_complex_float *tau, + lapack_complex_float *c, + lapack_int ldc, + lapack_complex_float *work); +lapack_int LAPACKE_zupmtr_work(int matrix_order, + char side, + char uplo, + char trans, + lapack_int m, + lapack_int n, + const lapack_complex_double *ap, + const lapack_complex_double *tau, + lapack_complex_double *c, + lapack_int ldc, + lapack_complex_double *work); + +lapack_int LAPACKE_claghe(int matrix_order, + lapack_int n, + lapack_int k, + const float *d, + lapack_complex_float *a, + lapack_int lda, + lapack_int *iseed); +lapack_int LAPACKE_zlaghe(int matrix_order, + lapack_int n, + lapack_int k, + const double *d, + lapack_complex_double *a, + lapack_int lda, + lapack_int *iseed); + +lapack_int LAPACKE_slagsy(int matrix_order, + lapack_int n, + lapack_int k, + const float *d, + float *a, + lapack_int lda, + lapack_int *iseed); +lapack_int LAPACKE_dlagsy(int matrix_order, + lapack_int n, + lapack_int k, + const double *d, + double *a, + lapack_int lda, + lapack_int *iseed); +lapack_int LAPACKE_clagsy(int matrix_order, + lapack_int n, + lapack_int k, + const float *d, + lapack_complex_float *a, + lapack_int lda, + lapack_int *iseed); +lapack_int LAPACKE_zlagsy(int matrix_order, + lapack_int n, + lapack_int k, + const double *d, + lapack_complex_double *a, + lapack_int lda, + lapack_int *iseed); + +lapack_int LAPACKE_slapmr(int matrix_order, + lapack_logical forwrd, + lapack_int m, + lapack_int n, + float *x, + lapack_int ldx, + lapack_int *k); +lapack_int LAPACKE_dlapmr(int matrix_order, + lapack_logical forwrd, + lapack_int m, + lapack_int n, + double *x, + lapack_int ldx, + lapack_int *k); +lapack_int LAPACKE_clapmr(int matrix_order, + lapack_logical forwrd, + lapack_int m, + lapack_int n, + lapack_complex_float *x, + lapack_int ldx, + lapack_int *k); +lapack_int LAPACKE_zlapmr(int matrix_order, + lapack_logical forwrd, + lapack_int m, + lapack_int n, + lapack_complex_double *x, + lapack_int ldx, + lapack_int *k); + + +float LAPACKE_slapy2(float x, float y); +double LAPACKE_dlapy2(double x, double y); + +float LAPACKE_slapy3(float x, float y, float z); +double LAPACKE_dlapy3(double x, double y, double z); + +lapack_int LAPACKE_slartgp(float f, float g, float *cs, float *sn, float *r); +lapack_int LAPACKE_dlartgp(double f, double g, double *cs, double *sn, double *r); + +lapack_int LAPACKE_slartgs(float x, float y, float sigma, float *cs, float *sn); +lapack_int LAPACKE_dlartgs(double x, double y, double sigma, double *cs, double *sn); + + +// LAPACK 3.3.0 +lapack_int LAPACKE_cbbcsd(int matrix_order, + char jobu1, + char jobu2, + char jobv1t, + char jobv2t, + char trans, + lapack_int m, + lapack_int p, + lapack_int q, + float *theta, + float *phi, + lapack_complex_float *u1, + lapack_int ldu1, + lapack_complex_float *u2, + lapack_int ldu2, + lapack_complex_float *v1t, + lapack_int ldv1t, + lapack_complex_float *v2t, + lapack_int ldv2t, + float *b11d, + float *b11e, + float *b12d, + float *b12e, + float *b21d, + float *b21e, + float *b22d, + float *b22e); +lapack_int LAPACKE_cbbcsd_work(int matrix_order, + char jobu1, + char jobu2, + char jobv1t, + char jobv2t, + char trans, + lapack_int m, + lapack_int p, + lapack_int q, + float *theta, + float *phi, + lapack_complex_float *u1, + lapack_int ldu1, + lapack_complex_float *u2, + lapack_int ldu2, + lapack_complex_float *v1t, + lapack_int ldv1t, + lapack_complex_float *v2t, + lapack_int ldv2t, + float *b11d, + float *b11e, + float *b12d, + float *b12e, + float *b21d, + float *b21e, + float *b22d, + float *b22e, + float *rwork, + lapack_int lrwork); +lapack_int + LAPACKE_cheswapr(int matrix_order, char uplo, lapack_int n, lapack_complex_float *a, lapack_int i1, lapack_int i2); +lapack_int LAPACKE_cheswapr_work(int matrix_order, + char uplo, + lapack_int n, + lapack_complex_float *a, + lapack_int i1, + lapack_int i2); +lapack_int LAPACKE_chetri2(int matrix_order, + char uplo, + lapack_int n, + lapack_complex_float *a, + lapack_int lda, + const lapack_int *ipiv); +lapack_int LAPACKE_chetri2_work(int matrix_order, + char uplo, + lapack_int n, + lapack_complex_float *a, + lapack_int lda, + const lapack_int *ipiv, + lapack_complex_float *work, + lapack_int lwork); +lapack_int LAPACKE_chetri2x(int matrix_order, + char uplo, + lapack_int n, + lapack_complex_float *a, + lapack_int lda, + const lapack_int *ipiv, + lapack_int nb); +lapack_int LAPACKE_chetri2x_work(int matrix_order, + char uplo, + lapack_int n, + lapack_complex_float *a, + lapack_int lda, + const lapack_int *ipiv, + lapack_complex_float *work, + lapack_int nb); +lapack_int LAPACKE_chetrs2(int matrix_order, + char uplo, + lapack_int n, + lapack_int nrhs, + const lapack_complex_float *a, + lapack_int lda, + const lapack_int *ipiv, + lapack_complex_float *b, + lapack_int ldb); +lapack_int LAPACKE_chetrs2_work(int matrix_order, + char uplo, + lapack_int n, + lapack_int nrhs, + const lapack_complex_float *a, + lapack_int lda, + const lapack_int *ipiv, + lapack_complex_float *b, + lapack_int ldb, + lapack_complex_float *work); +lapack_int LAPACKE_csyconv(int matrix_order, + char uplo, + char way, + lapack_int n, + lapack_complex_float *a, + lapack_int lda, + const lapack_int *ipiv); +lapack_int LAPACKE_csyconv_work(int matrix_order, + char uplo, + char way, + lapack_int n, + lapack_complex_float *a, + lapack_int lda, + const lapack_int *ipiv, + lapack_complex_float *work); +lapack_int + LAPACKE_csyswapr(int matrix_order, char uplo, lapack_int n, lapack_complex_float *a, lapack_int i1, lapack_int i2); +lapack_int LAPACKE_csyswapr_work(int matrix_order, + char uplo, + lapack_int n, + lapack_complex_float *a, + lapack_int i1, + lapack_int i2); +lapack_int LAPACKE_csytri2(int matrix_order, + char uplo, + lapack_int n, + lapack_complex_float *a, + lapack_int lda, + const lapack_int *ipiv); +lapack_int LAPACKE_csytri2_work(int matrix_order, + char uplo, + lapack_int n, + lapack_complex_float *a, + lapack_int lda, + const lapack_int *ipiv, + lapack_complex_float *work, + lapack_int lwork); +lapack_int LAPACKE_csytri2x(int matrix_order, + char uplo, + lapack_int n, + lapack_complex_float *a, + lapack_int lda, + const lapack_int *ipiv, + lapack_int nb); +lapack_int LAPACKE_csytri2x_work(int matrix_order, + char uplo, + lapack_int n, + lapack_complex_float *a, + lapack_int lda, + const lapack_int *ipiv, + lapack_complex_float *work, + lapack_int nb); +lapack_int LAPACKE_csytrs2(int matrix_order, + char uplo, + lapack_int n, + lapack_int nrhs, + const lapack_complex_float *a, + lapack_int lda, + const lapack_int *ipiv, + lapack_complex_float *b, + lapack_int ldb); +lapack_int LAPACKE_csytrs2_work(int matrix_order, + char uplo, + lapack_int n, + lapack_int nrhs, + const lapack_complex_float *a, + lapack_int lda, + const lapack_int *ipiv, + lapack_complex_float *b, + lapack_int ldb, + lapack_complex_float *work); +lapack_int LAPACKE_cunbdb(int matrix_order, + char trans, + char signs, + lapack_int m, + lapack_int p, + lapack_int q, + lapack_complex_float *x11, + lapack_int ldx11, + lapack_complex_float *x12, + lapack_int ldx12, + lapack_complex_float *x21, + lapack_int ldx21, + lapack_complex_float *x22, + lapack_int ldx22, + float *theta, + float *phi, + lapack_complex_float *taup1, + lapack_complex_float *taup2, + lapack_complex_float *tauq1, + lapack_complex_float *tauq2); +lapack_int LAPACKE_cunbdb_work(int matrix_order, + char trans, + char signs, + lapack_int m, + lapack_int p, + lapack_int q, + lapack_complex_float *x11, + lapack_int ldx11, + lapack_complex_float *x12, + lapack_int ldx12, + lapack_complex_float *x21, + lapack_int ldx21, + lapack_complex_float *x22, + lapack_int ldx22, + float *theta, + float *phi, + lapack_complex_float *taup1, + lapack_complex_float *taup2, + lapack_complex_float *tauq1, + lapack_complex_float *tauq2, + lapack_complex_float *work, + lapack_int lwork); +lapack_int LAPACKE_cuncsd(int matrix_order, + char jobu1, + char jobu2, + char jobv1t, + char jobv2t, + char trans, + char signs, + lapack_int m, + lapack_int p, + lapack_int q, + lapack_complex_float *x11, + lapack_int ldx11, + lapack_complex_float *x12, + lapack_int ldx12, + lapack_complex_float *x21, + lapack_int ldx21, + lapack_complex_float *x22, + lapack_int ldx22, + float *theta, + lapack_complex_float *u1, + lapack_int ldu1, + lapack_complex_float *u2, + lapack_int ldu2, + lapack_complex_float *v1t, + lapack_int ldv1t, + lapack_complex_float *v2t, + lapack_int ldv2t); +lapack_int LAPACKE_cuncsd_work(int matrix_order, + char jobu1, + char jobu2, + char jobv1t, + char jobv2t, + char trans, + char signs, + lapack_int m, + lapack_int p, + lapack_int q, + lapack_complex_float *x11, + lapack_int ldx11, + lapack_complex_float *x12, + lapack_int ldx12, + lapack_complex_float *x21, + lapack_int ldx21, + lapack_complex_float *x22, + lapack_int ldx22, + float *theta, + lapack_complex_float *u1, + lapack_int ldu1, + lapack_complex_float *u2, + lapack_int ldu2, + lapack_complex_float *v1t, + lapack_int ldv1t, + lapack_complex_float *v2t, + lapack_int ldv2t, + lapack_complex_float *work, + lapack_int lwork, + float *rwork, + lapack_int lrwork, + lapack_int *iwork); +lapack_int LAPACKE_dbbcsd(int matrix_order, + char jobu1, + char jobu2, + char jobv1t, + char jobv2t, + char trans, + lapack_int m, + lapack_int p, + lapack_int q, + double *theta, + double *phi, + double *u1, + lapack_int ldu1, + double *u2, + lapack_int ldu2, + double *v1t, + lapack_int ldv1t, + double *v2t, + lapack_int ldv2t, + double *b11d, + double *b11e, + double *b12d, + double *b12e, + double *b21d, + double *b21e, + double *b22d, + double *b22e); +lapack_int LAPACKE_dbbcsd_work(int matrix_order, + char jobu1, + char jobu2, + char jobv1t, + char jobv2t, + char trans, + lapack_int m, + lapack_int p, + lapack_int q, + double *theta, + double *phi, + double *u1, + lapack_int ldu1, + double *u2, + lapack_int ldu2, + double *v1t, + lapack_int ldv1t, + double *v2t, + lapack_int ldv2t, + double *b11d, + double *b11e, + double *b12d, + double *b12e, + double *b21d, + double *b21e, + double *b22d, + double *b22e, + double *work, + lapack_int lwork); +lapack_int LAPACKE_dorbdb(int matrix_order, + char trans, + char signs, + lapack_int m, + lapack_int p, + lapack_int q, + double *x11, + lapack_int ldx11, + double *x12, + lapack_int ldx12, + double *x21, + lapack_int ldx21, + double *x22, + lapack_int ldx22, + double *theta, + double *phi, + double *taup1, + double *taup2, + double *tauq1, + double *tauq2); +lapack_int LAPACKE_dorbdb_work(int matrix_order, + char trans, + char signs, + lapack_int m, + lapack_int p, + lapack_int q, + double *x11, + lapack_int ldx11, + double *x12, + lapack_int ldx12, + double *x21, + lapack_int ldx21, + double *x22, + lapack_int ldx22, + double *theta, + double *phi, + double *taup1, + double *taup2, + double *tauq1, + double *tauq2, + double *work, + lapack_int lwork); +lapack_int LAPACKE_dorcsd(int matrix_order, + char jobu1, + char jobu2, + char jobv1t, + char jobv2t, + char trans, + char signs, + lapack_int m, + lapack_int p, + lapack_int q, + double *x11, + lapack_int ldx11, + double *x12, + lapack_int ldx12, + double *x21, + lapack_int ldx21, + double *x22, + lapack_int ldx22, + double *theta, + double *u1, + lapack_int ldu1, + double *u2, + lapack_int ldu2, + double *v1t, + lapack_int ldv1t, + double *v2t, + lapack_int ldv2t); +lapack_int LAPACKE_dorcsd_work(int matrix_order, + char jobu1, + char jobu2, + char jobv1t, + char jobv2t, + char trans, + char signs, + lapack_int m, + lapack_int p, + lapack_int q, + double *x11, + lapack_int ldx11, + double *x12, + lapack_int ldx12, + double *x21, + lapack_int ldx21, + double *x22, + lapack_int ldx22, + double *theta, + double *u1, + lapack_int ldu1, + double *u2, + lapack_int ldu2, + double *v1t, + lapack_int ldv1t, + double *v2t, + lapack_int ldv2t, + double *work, + lapack_int lwork, + lapack_int *iwork); +lapack_int LAPACKE_dsyconv(int matrix_order, + char uplo, + char way, + lapack_int n, + double *a, + lapack_int lda, + const lapack_int *ipiv); +lapack_int LAPACKE_dsyconv_work(int matrix_order, + char uplo, + char way, + lapack_int n, + double *a, + lapack_int lda, + const lapack_int *ipiv, + double *work); +lapack_int LAPACKE_dsyswapr(int matrix_order, char uplo, lapack_int n, double *a, lapack_int i1, lapack_int i2); +lapack_int LAPACKE_dsyswapr_work(int matrix_order, char uplo, lapack_int n, double *a, lapack_int i1, lapack_int i2); +lapack_int + LAPACKE_dsytri2(int matrix_order, char uplo, lapack_int n, double *a, lapack_int lda, const lapack_int *ipiv); +lapack_int LAPACKE_dsytri2_work(int matrix_order, + char uplo, + lapack_int n, + double *a, + lapack_int lda, + const lapack_int *ipiv, + lapack_complex_double *work, + lapack_int lwork); +lapack_int LAPACKE_dsytri2x(int matrix_order, + char uplo, + lapack_int n, + double *a, + lapack_int lda, + const lapack_int *ipiv, + lapack_int nb); +lapack_int LAPACKE_dsytri2x_work(int matrix_order, + char uplo, + lapack_int n, + double *a, + lapack_int lda, + const lapack_int *ipiv, + double *work, + lapack_int nb); +lapack_int LAPACKE_dsytrs2(int matrix_order, + char uplo, + lapack_int n, + lapack_int nrhs, + const double *a, + lapack_int lda, + const lapack_int *ipiv, + double *b, + lapack_int ldb); +lapack_int LAPACKE_dsytrs2_work(int matrix_order, + char uplo, + lapack_int n, + lapack_int nrhs, + const double *a, + lapack_int lda, + const lapack_int *ipiv, + double *b, + lapack_int ldb, + double *work); +lapack_int LAPACKE_sbbcsd(int matrix_order, + char jobu1, + char jobu2, + char jobv1t, + char jobv2t, + char trans, + lapack_int m, + lapack_int p, + lapack_int q, + float *theta, + float *phi, + float *u1, + lapack_int ldu1, + float *u2, + lapack_int ldu2, + float *v1t, + lapack_int ldv1t, + float *v2t, + lapack_int ldv2t, + float *b11d, + float *b11e, + float *b12d, + float *b12e, + float *b21d, + float *b21e, + float *b22d, + float *b22e); +lapack_int LAPACKE_sbbcsd_work(int matrix_order, + char jobu1, + char jobu2, + char jobv1t, + char jobv2t, + char trans, + lapack_int m, + lapack_int p, + lapack_int q, + float *theta, + float *phi, + float *u1, + lapack_int ldu1, + float *u2, + lapack_int ldu2, + float *v1t, + lapack_int ldv1t, + float *v2t, + lapack_int ldv2t, + float *b11d, + float *b11e, + float *b12d, + float *b12e, + float *b21d, + float *b21e, + float *b22d, + float *b22e, + float *work, + lapack_int lwork); +lapack_int LAPACKE_sorbdb(int matrix_order, + char trans, + char signs, + lapack_int m, + lapack_int p, + lapack_int q, + float *x11, + lapack_int ldx11, + float *x12, + lapack_int ldx12, + float *x21, + lapack_int ldx21, + float *x22, + lapack_int ldx22, + float *theta, + float *phi, + float *taup1, + float *taup2, + float *tauq1, + float *tauq2); +lapack_int LAPACKE_sorbdb_work(int matrix_order, + char trans, + char signs, + lapack_int m, + lapack_int p, + lapack_int q, + float *x11, + lapack_int ldx11, + float *x12, + lapack_int ldx12, + float *x21, + lapack_int ldx21, + float *x22, + lapack_int ldx22, + float *theta, + float *phi, + float *taup1, + float *taup2, + float *tauq1, + float *tauq2, + float *work, + lapack_int lwork); +lapack_int LAPACKE_sorcsd(int matrix_order, + char jobu1, + char jobu2, + char jobv1t, + char jobv2t, + char trans, + char signs, + lapack_int m, + lapack_int p, + lapack_int q, + float *x11, + lapack_int ldx11, + float *x12, + lapack_int ldx12, + float *x21, + lapack_int ldx21, + float *x22, + lapack_int ldx22, + float *theta, + float *u1, + lapack_int ldu1, + float *u2, + lapack_int ldu2, + float *v1t, + lapack_int ldv1t, + float *v2t, + lapack_int ldv2t); +lapack_int LAPACKE_sorcsd_work(int matrix_order, + char jobu1, + char jobu2, + char jobv1t, + char jobv2t, + char trans, + char signs, + lapack_int m, + lapack_int p, + lapack_int q, + float *x11, + lapack_int ldx11, + float *x12, + lapack_int ldx12, + float *x21, + lapack_int ldx21, + float *x22, + lapack_int ldx22, + float *theta, + float *u1, + lapack_int ldu1, + float *u2, + lapack_int ldu2, + float *v1t, + lapack_int ldv1t, + float *v2t, + lapack_int ldv2t, + float *work, + lapack_int lwork, + lapack_int *iwork); +lapack_int LAPACKE_ssyconv(int matrix_order, + char uplo, + char way, + lapack_int n, + float *a, + lapack_int lda, + const lapack_int *ipiv); +lapack_int LAPACKE_ssyconv_work(int matrix_order, + char uplo, + char way, + lapack_int n, + float *a, + lapack_int lda, + const lapack_int *ipiv, + float *work); +lapack_int LAPACKE_ssyswapr(int matrix_order, char uplo, lapack_int n, float *a, lapack_int i1, lapack_int i2); +lapack_int LAPACKE_ssyswapr_work(int matrix_order, char uplo, lapack_int n, float *a, lapack_int i1, lapack_int i2); +lapack_int LAPACKE_ssytri2(int matrix_order, char uplo, lapack_int n, float *a, lapack_int lda, const lapack_int *ipiv); +lapack_int LAPACKE_ssytri2_work(int matrix_order, + char uplo, + lapack_int n, + float *a, + lapack_int lda, + const lapack_int *ipiv, + lapack_complex_float *work, + lapack_int lwork); +lapack_int LAPACKE_ssytri2x(int matrix_order, + char uplo, + lapack_int n, + float *a, + lapack_int lda, + const lapack_int *ipiv, + lapack_int nb); +lapack_int LAPACKE_ssytri2x_work(int matrix_order, + char uplo, + lapack_int n, + float *a, + lapack_int lda, + const lapack_int *ipiv, + float *work, + lapack_int nb); +lapack_int LAPACKE_ssytrs2(int matrix_order, + char uplo, + lapack_int n, + lapack_int nrhs, + const float *a, + lapack_int lda, + const lapack_int *ipiv, + float *b, + lapack_int ldb); +lapack_int LAPACKE_ssytrs2_work(int matrix_order, + char uplo, + lapack_int n, + lapack_int nrhs, + const float *a, + lapack_int lda, + const lapack_int *ipiv, + float *b, + lapack_int ldb, + float *work); +lapack_int LAPACKE_zbbcsd(int matrix_order, + char jobu1, + char jobu2, + char jobv1t, + char jobv2t, + char trans, + lapack_int m, + lapack_int p, + lapack_int q, + double *theta, + double *phi, + lapack_complex_double *u1, + lapack_int ldu1, + lapack_complex_double *u2, + lapack_int ldu2, + lapack_complex_double *v1t, + lapack_int ldv1t, + lapack_complex_double *v2t, + lapack_int ldv2t, + double *b11d, + double *b11e, + double *b12d, + double *b12e, + double *b21d, + double *b21e, + double *b22d, + double *b22e); +lapack_int LAPACKE_zbbcsd_work(int matrix_order, + char jobu1, + char jobu2, + char jobv1t, + char jobv2t, + char trans, + lapack_int m, + lapack_int p, + lapack_int q, + double *theta, + double *phi, + lapack_complex_double *u1, + lapack_int ldu1, + lapack_complex_double *u2, + lapack_int ldu2, + lapack_complex_double *v1t, + lapack_int ldv1t, + lapack_complex_double *v2t, + lapack_int ldv2t, + double *b11d, + double *b11e, + double *b12d, + double *b12e, + double *b21d, + double *b21e, + double *b22d, + double *b22e, + double *rwork, + lapack_int lrwork); +lapack_int + LAPACKE_zheswapr(int matrix_order, char uplo, lapack_int n, lapack_complex_double *a, lapack_int i1, lapack_int i2); +lapack_int LAPACKE_zheswapr_work(int matrix_order, + char uplo, + lapack_int n, + lapack_complex_double *a, + lapack_int i1, + lapack_int i2); +lapack_int LAPACKE_zhetri2(int matrix_order, + char uplo, + lapack_int n, + lapack_complex_double *a, + lapack_int lda, + const lapack_int *ipiv); +lapack_int LAPACKE_zhetri2_work(int matrix_order, + char uplo, + lapack_int n, + lapack_complex_double *a, + lapack_int lda, + const lapack_int *ipiv, + lapack_complex_double *work, + lapack_int lwork); +lapack_int LAPACKE_zhetri2x(int matrix_order, + char uplo, + lapack_int n, + lapack_complex_double *a, + lapack_int lda, + const lapack_int *ipiv, + lapack_int nb); +lapack_int LAPACKE_zhetri2x_work(int matrix_order, + char uplo, + lapack_int n, + lapack_complex_double *a, + lapack_int lda, + const lapack_int *ipiv, + lapack_complex_double *work, + lapack_int nb); +lapack_int LAPACKE_zhetrs2(int matrix_order, + char uplo, + lapack_int n, + lapack_int nrhs, + const lapack_complex_double *a, + lapack_int lda, + const lapack_int *ipiv, + lapack_complex_double *b, + lapack_int ldb); +lapack_int LAPACKE_zhetrs2_work(int matrix_order, + char uplo, + lapack_int n, + lapack_int nrhs, + const lapack_complex_double *a, + lapack_int lda, + const lapack_int *ipiv, + lapack_complex_double *b, + lapack_int ldb, + lapack_complex_double *work); +lapack_int LAPACKE_zsyconv(int matrix_order, + char uplo, + char way, + lapack_int n, + lapack_complex_double *a, + lapack_int lda, + const lapack_int *ipiv); +lapack_int LAPACKE_zsyconv_work(int matrix_order, + char uplo, + char way, + lapack_int n, + lapack_complex_double *a, + lapack_int lda, + const lapack_int *ipiv, + lapack_complex_double *work); +lapack_int + LAPACKE_zsyswapr(int matrix_order, char uplo, lapack_int n, lapack_complex_double *a, lapack_int i1, lapack_int i2); +lapack_int LAPACKE_zsyswapr_work(int matrix_order, + char uplo, + lapack_int n, + lapack_complex_double *a, + lapack_int i1, + lapack_int i2); +lapack_int LAPACKE_zsytri2(int matrix_order, + char uplo, + lapack_int n, + lapack_complex_double *a, + lapack_int lda, + const lapack_int *ipiv); +lapack_int LAPACKE_zsytri2_work(int matrix_order, + char uplo, + lapack_int n, + lapack_complex_double *a, + lapack_int lda, + const lapack_int *ipiv, + lapack_complex_double *work, + lapack_int lwork); +lapack_int LAPACKE_zsytri2x(int matrix_order, + char uplo, + lapack_int n, + lapack_complex_double *a, + lapack_int lda, + const lapack_int *ipiv, + lapack_int nb); +lapack_int LAPACKE_zsytri2x_work(int matrix_order, + char uplo, + lapack_int n, + lapack_complex_double *a, + lapack_int lda, + const lapack_int *ipiv, + lapack_complex_double *work, + lapack_int nb); +lapack_int LAPACKE_zsytrs2(int matrix_order, + char uplo, + lapack_int n, + lapack_int nrhs, + const lapack_complex_double *a, + lapack_int lda, + const lapack_int *ipiv, + lapack_complex_double *b, + lapack_int ldb); +lapack_int LAPACKE_zsytrs2_work(int matrix_order, + char uplo, + lapack_int n, + lapack_int nrhs, + const lapack_complex_double *a, + lapack_int lda, + const lapack_int *ipiv, + lapack_complex_double *b, + lapack_int ldb, + lapack_complex_double *work); +lapack_int LAPACKE_zunbdb(int matrix_order, + char trans, + char signs, + lapack_int m, + lapack_int p, + lapack_int q, + lapack_complex_double *x11, + lapack_int ldx11, + lapack_complex_double *x12, + lapack_int ldx12, + lapack_complex_double *x21, + lapack_int ldx21, + lapack_complex_double *x22, + lapack_int ldx22, + double *theta, + double *phi, + lapack_complex_double *taup1, + lapack_complex_double *taup2, + lapack_complex_double *tauq1, + lapack_complex_double *tauq2); +lapack_int LAPACKE_zunbdb_work(int matrix_order, + char trans, + char signs, + lapack_int m, + lapack_int p, + lapack_int q, + lapack_complex_double *x11, + lapack_int ldx11, + lapack_complex_double *x12, + lapack_int ldx12, + lapack_complex_double *x21, + lapack_int ldx21, + lapack_complex_double *x22, + lapack_int ldx22, + double *theta, + double *phi, + lapack_complex_double *taup1, + lapack_complex_double *taup2, + lapack_complex_double *tauq1, + lapack_complex_double *tauq2, + lapack_complex_double *work, + lapack_int lwork); +lapack_int LAPACKE_zuncsd(int matrix_order, + char jobu1, + char jobu2, + char jobv1t, + char jobv2t, + char trans, + char signs, + lapack_int m, + lapack_int p, + lapack_int q, + lapack_complex_double *x11, + lapack_int ldx11, + lapack_complex_double *x12, + lapack_int ldx12, + lapack_complex_double *x21, + lapack_int ldx21, + lapack_complex_double *x22, + lapack_int ldx22, + double *theta, + lapack_complex_double *u1, + lapack_int ldu1, + lapack_complex_double *u2, + lapack_int ldu2, + lapack_complex_double *v1t, + lapack_int ldv1t, + lapack_complex_double *v2t, + lapack_int ldv2t); +lapack_int LAPACKE_zuncsd_work(int matrix_order, + char jobu1, + char jobu2, + char jobv1t, + char jobv2t, + char trans, + char signs, + lapack_int m, + lapack_int p, + lapack_int q, + lapack_complex_double *x11, + lapack_int ldx11, + lapack_complex_double *x12, + lapack_int ldx12, + lapack_complex_double *x21, + lapack_int ldx21, + lapack_complex_double *x22, + lapack_int ldx22, + double *theta, + lapack_complex_double *u1, + lapack_int ldu1, + lapack_complex_double *u2, + lapack_int ldu2, + lapack_complex_double *v1t, + lapack_int ldv1t, + lapack_complex_double *v2t, + lapack_int ldv2t, + lapack_complex_double *work, + lapack_int lwork, + double *rwork, + lapack_int lrwork, + lapack_int *iwork); +// LAPACK 3.4.0 +lapack_int LAPACKE_sgemqrt(int matrix_order, + char side, + char trans, + lapack_int m, + lapack_int n, + lapack_int k, + lapack_int nb, + const float *v, + lapack_int ldv, + const float *t, + lapack_int ldt, + float *c, + lapack_int ldc); +lapack_int LAPACKE_dgemqrt(int matrix_order, + char side, + char trans, + lapack_int m, + lapack_int n, + lapack_int k, + lapack_int nb, + const double *v, + lapack_int ldv, + const double *t, + lapack_int ldt, + double *c, + lapack_int ldc); +lapack_int LAPACKE_cgemqrt(int matrix_order, + char side, + char trans, + lapack_int m, + lapack_int n, + lapack_int k, + lapack_int nb, + const lapack_complex_float *v, + lapack_int ldv, + const lapack_complex_float *t, + lapack_int ldt, + lapack_complex_float *c, + lapack_int ldc); +lapack_int LAPACKE_zgemqrt(int matrix_order, + char side, + char trans, + lapack_int m, + lapack_int n, + lapack_int k, + lapack_int nb, + const lapack_complex_double *v, + lapack_int ldv, + const lapack_complex_double *t, + lapack_int ldt, + lapack_complex_double *c, + lapack_int ldc); + +lapack_int LAPACKE_sgeqrt(int matrix_order, + lapack_int m, + lapack_int n, + lapack_int nb, + float *a, + lapack_int lda, + float *t, + lapack_int ldt); +lapack_int LAPACKE_dgeqrt(int matrix_order, + lapack_int m, + lapack_int n, + lapack_int nb, + double *a, + lapack_int lda, + double *t, + lapack_int ldt); +lapack_int LAPACKE_cgeqrt(int matrix_order, + lapack_int m, + lapack_int n, + lapack_int nb, + lapack_complex_float *a, + lapack_int lda, + lapack_complex_float *t, + lapack_int ldt); +lapack_int LAPACKE_zgeqrt(int matrix_order, + lapack_int m, + lapack_int n, + lapack_int nb, + lapack_complex_double *a, + lapack_int lda, + lapack_complex_double *t, + lapack_int ldt); + +lapack_int + LAPACKE_sgeqrt2(int matrix_order, lapack_int m, lapack_int n, float *a, lapack_int lda, float *t, lapack_int ldt); +lapack_int + LAPACKE_dgeqrt2(int matrix_order, lapack_int m, lapack_int n, double *a, lapack_int lda, double *t, lapack_int ldt); +lapack_int LAPACKE_cgeqrt2(int matrix_order, + lapack_int m, + lapack_int n, + lapack_complex_float *a, + lapack_int lda, + lapack_complex_float *t, + lapack_int ldt); +lapack_int LAPACKE_zgeqrt2(int matrix_order, + lapack_int m, + lapack_int n, + lapack_complex_double *a, + lapack_int lda, + lapack_complex_double *t, + lapack_int ldt); + +lapack_int + LAPACKE_sgeqrt3(int matrix_order, lapack_int m, lapack_int n, float *a, lapack_int lda, float *t, lapack_int ldt); +lapack_int + LAPACKE_dgeqrt3(int matrix_order, lapack_int m, lapack_int n, double *a, lapack_int lda, double *t, lapack_int ldt); +lapack_int LAPACKE_cgeqrt3(int matrix_order, + lapack_int m, + lapack_int n, + lapack_complex_float *a, + lapack_int lda, + lapack_complex_float *t, + lapack_int ldt); +lapack_int LAPACKE_zgeqrt3(int matrix_order, + lapack_int m, + lapack_int n, + lapack_complex_double *a, + lapack_int lda, + lapack_complex_double *t, + lapack_int ldt); + +lapack_int LAPACKE_stpmqrt(int matrix_order, + char side, + char trans, + lapack_int m, + lapack_int n, + lapack_int k, + lapack_int l, + lapack_int nb, + const float *v, + lapack_int ldv, + const float *t, + lapack_int ldt, + float *a, + lapack_int lda, + float *b, + lapack_int ldb); +lapack_int LAPACKE_dtpmqrt(int matrix_order, + char side, + char trans, + lapack_int m, + lapack_int n, + lapack_int k, + lapack_int l, + lapack_int nb, + const double *v, + lapack_int ldv, + const double *t, + lapack_int ldt, + double *a, + lapack_int lda, + double *b, + lapack_int ldb); +lapack_int LAPACKE_ctpmqrt(int matrix_order, + char side, + char trans, + lapack_int m, + lapack_int n, + lapack_int k, + lapack_int l, + lapack_int nb, + const lapack_complex_float *v, + lapack_int ldv, + const lapack_complex_float *t, + lapack_int ldt, + lapack_complex_float *a, + lapack_int lda, + lapack_complex_float *b, + lapack_int ldb); +lapack_int LAPACKE_ztpmqrt(int matrix_order, + char side, + char trans, + lapack_int m, + lapack_int n, + lapack_int k, + lapack_int l, + lapack_int nb, + const lapack_complex_double *v, + lapack_int ldv, + const lapack_complex_double *t, + lapack_int ldt, + lapack_complex_double *a, + lapack_int lda, + lapack_complex_double *b, + lapack_int ldb); + +lapack_int LAPACKE_dtpqrt(int matrix_order, + lapack_int m, + lapack_int n, + lapack_int l, + lapack_int nb, + double *a, + lapack_int lda, + double *b, + lapack_int ldb, + double *t, + lapack_int ldt); +lapack_int LAPACKE_ctpqrt(int matrix_order, + lapack_int m, + lapack_int n, + lapack_int l, + lapack_int nb, + lapack_complex_float *a, + lapack_int lda, + lapack_complex_float *t, + lapack_complex_float *b, + lapack_int ldb, + lapack_int ldt); +lapack_int LAPACKE_ztpqrt(int matrix_order, + lapack_int m, + lapack_int n, + lapack_int l, + lapack_int nb, + lapack_complex_double *a, + lapack_int lda, + lapack_complex_double *b, + lapack_int ldb, + lapack_complex_double *t, + lapack_int ldt); + +lapack_int LAPACKE_stpqrt2(int matrix_order, + lapack_int m, + lapack_int n, + float *a, + lapack_int lda, + float *b, + lapack_int ldb, + float *t, + lapack_int ldt); +lapack_int LAPACKE_dtpqrt2(int matrix_order, + lapack_int m, + lapack_int n, + double *a, + lapack_int lda, + double *b, + lapack_int ldb, + double *t, + lapack_int ldt); +lapack_int LAPACKE_ctpqrt2(int matrix_order, + lapack_int m, + lapack_int n, + lapack_complex_float *a, + lapack_int lda, + lapack_complex_float *b, + lapack_int ldb, + lapack_complex_float *t, + lapack_int ldt); +lapack_int LAPACKE_ztpqrt2(int matrix_order, + lapack_int m, + lapack_int n, + lapack_complex_double *a, + lapack_int lda, + lapack_complex_double *b, + lapack_int ldb, + lapack_complex_double *t, + lapack_int ldt); + +lapack_int LAPACKE_stprfb(int matrix_order, + char side, + char trans, + char direct, + char storev, + lapack_int m, + lapack_int n, + lapack_int k, + lapack_int l, + const float *v, + lapack_int ldv, + const float *t, + lapack_int ldt, + float *a, + lapack_int lda, + float *b, + lapack_int ldb, + lapack_int myldwork); +lapack_int LAPACKE_dtprfb(int matrix_order, + char side, + char trans, + char direct, + char storev, + lapack_int m, + lapack_int n, + lapack_int k, + lapack_int l, + const double *v, + lapack_int ldv, + const double *t, + lapack_int ldt, + double *a, + lapack_int lda, + double *b, + lapack_int ldb, + lapack_int myldwork); +lapack_int LAPACKE_ctprfb(int matrix_order, + char side, + char trans, + char direct, + char storev, + lapack_int m, + lapack_int n, + lapack_int k, + lapack_int l, + const lapack_complex_float *v, + lapack_int ldv, + const lapack_complex_float *t, + lapack_int ldt, + lapack_complex_float *a, + lapack_int lda, + lapack_complex_float *b, + lapack_int ldb, + lapack_int myldwork); +lapack_int LAPACKE_ztprfb(int matrix_order, + char side, + char trans, + char direct, + char storev, + lapack_int m, + lapack_int n, + lapack_int k, + lapack_int l, + const lapack_complex_double *v, + lapack_int ldv, + const lapack_complex_double *t, + lapack_int ldt, + lapack_complex_double *a, + lapack_int lda, + lapack_complex_double *b, + lapack_int ldb, + lapack_int myldwork); + +lapack_int LAPACKE_sgemqrt_work(int matrix_order, + char side, + char trans, + lapack_int m, + lapack_int n, + lapack_int k, + lapack_int nb, + const float *v, + lapack_int ldv, + const float *t, + lapack_int ldt, + float *c, + lapack_int ldc, + float *work); +lapack_int LAPACKE_dgemqrt_work(int matrix_order, + char side, + char trans, + lapack_int m, + lapack_int n, + lapack_int k, + lapack_int nb, + const double *v, + lapack_int ldv, + const double *t, + lapack_int ldt, + double *c, + lapack_int ldc, + double *work); +lapack_int LAPACKE_cgemqrt_work(int matrix_order, + char side, + char trans, + lapack_int m, + lapack_int n, + lapack_int k, + lapack_int nb, + const lapack_complex_float *v, + lapack_int ldv, + const lapack_complex_float *t, + lapack_int ldt, + lapack_complex_float *c, + lapack_int ldc, + lapack_complex_float *work); +lapack_int LAPACKE_zgemqrt_work(int matrix_order, + char side, + char trans, + lapack_int m, + lapack_int n, + lapack_int k, + lapack_int nb, + const lapack_complex_double *v, + lapack_int ldv, + const lapack_complex_double *t, + lapack_int ldt, + lapack_complex_double *c, + lapack_int ldc, + lapack_complex_double *work); + +lapack_int LAPACKE_sgeqrt_work(int matrix_order, + lapack_int m, + lapack_int n, + lapack_int nb, + float *a, + lapack_int lda, + float *t, + lapack_int ldt, + float *work); +lapack_int LAPACKE_dgeqrt_work(int matrix_order, + lapack_int m, + lapack_int n, + lapack_int nb, + double *a, + lapack_int lda, + double *t, + lapack_int ldt, + double *work); +lapack_int LAPACKE_cgeqrt_work(int matrix_order, + lapack_int m, + lapack_int n, + lapack_int nb, + lapack_complex_float *a, + lapack_int lda, + lapack_complex_float *t, + lapack_int ldt, + lapack_complex_float *work); +lapack_int LAPACKE_zgeqrt_work(int matrix_order, + lapack_int m, + lapack_int n, + lapack_int nb, + lapack_complex_double *a, + lapack_int lda, + lapack_complex_double *t, + lapack_int ldt, + lapack_complex_double *work); + +lapack_int LAPACKE_sgeqrt2_work(int matrix_order, + lapack_int m, + lapack_int n, + float *a, + lapack_int lda, + float *t, + lapack_int ldt); +lapack_int LAPACKE_dgeqrt2_work(int matrix_order, + lapack_int m, + lapack_int n, + double *a, + lapack_int lda, + double *t, + lapack_int ldt); +lapack_int LAPACKE_cgeqrt2_work(int matrix_order, + lapack_int m, + lapack_int n, + lapack_complex_float *a, + lapack_int lda, + lapack_complex_float *t, + lapack_int ldt); +lapack_int LAPACKE_zgeqrt2_work(int matrix_order, + lapack_int m, + lapack_int n, + lapack_complex_double *a, + lapack_int lda, + lapack_complex_double *t, + lapack_int ldt); + +lapack_int LAPACKE_sgeqrt3_work(int matrix_order, + lapack_int m, + lapack_int n, + float *a, + lapack_int lda, + float *t, + lapack_int ldt); +lapack_int LAPACKE_dgeqrt3_work(int matrix_order, + lapack_int m, + lapack_int n, + double *a, + lapack_int lda, + double *t, + lapack_int ldt); +lapack_int LAPACKE_cgeqrt3_work(int matrix_order, + lapack_int m, + lapack_int n, + lapack_complex_float *a, + lapack_int lda, + lapack_complex_float *t, + lapack_int ldt); +lapack_int LAPACKE_zgeqrt3_work(int matrix_order, + lapack_int m, + lapack_int n, + lapack_complex_double *a, + lapack_int lda, + lapack_complex_double *t, + lapack_int ldt); + +lapack_int LAPACKE_stpmqrt_work(int matrix_order, + char side, + char trans, + lapack_int m, + lapack_int n, + lapack_int k, + lapack_int l, + lapack_int nb, + const float *v, + lapack_int ldv, + const float *t, + lapack_int ldt, + float *a, + lapack_int lda, + float *b, + lapack_int ldb, + float *work); +lapack_int LAPACKE_dtpmqrt_work(int matrix_order, + char side, + char trans, + lapack_int m, + lapack_int n, + lapack_int k, + lapack_int l, + lapack_int nb, + const double *v, + lapack_int ldv, + const double *t, + lapack_int ldt, + double *a, + lapack_int lda, + double *b, + lapack_int ldb, + double *work); +lapack_int LAPACKE_ctpmqrt_work(int matrix_order, + char side, + char trans, + lapack_int m, + lapack_int n, + lapack_int k, + lapack_int l, + lapack_int nb, + const lapack_complex_float *v, + lapack_int ldv, + const lapack_complex_float *t, + lapack_int ldt, + lapack_complex_float *a, + lapack_int lda, + lapack_complex_float *b, + lapack_int ldb, + lapack_complex_float *work); +lapack_int LAPACKE_ztpmqrt_work(int matrix_order, + char side, + char trans, + lapack_int m, + lapack_int n, + lapack_int k, + lapack_int l, + lapack_int nb, + const lapack_complex_double *v, + lapack_int ldv, + const lapack_complex_double *t, + lapack_int ldt, + lapack_complex_double *a, + lapack_int lda, + lapack_complex_double *b, + lapack_int ldb, + lapack_complex_double *work); + +lapack_int LAPACKE_dtpqrt_work(int matrix_order, + lapack_int m, + lapack_int n, + lapack_int l, + lapack_int nb, + double *a, + lapack_int lda, + double *b, + lapack_int ldb, + double *t, + lapack_int ldt, + double *work); +lapack_int LAPACKE_ctpqrt_work(int matrix_order, + lapack_int m, + lapack_int n, + lapack_int l, + lapack_int nb, + lapack_complex_float *a, + lapack_int lda, + lapack_complex_float *t, + lapack_complex_float *b, + lapack_int ldb, + lapack_int ldt, + lapack_complex_float *work); +lapack_int LAPACKE_ztpqrt_work(int matrix_order, + lapack_int m, + lapack_int n, + lapack_int l, + lapack_int nb, + lapack_complex_double *a, + lapack_int lda, + lapack_complex_double *b, + lapack_int ldb, + lapack_complex_double *t, + lapack_int ldt, + lapack_complex_double *work); + +lapack_int LAPACKE_stpqrt2_work(int matrix_order, + lapack_int m, + lapack_int n, + float *a, + lapack_int lda, + float *b, + lapack_int ldb, + float *t, + lapack_int ldt); +lapack_int LAPACKE_dtpqrt2_work(int matrix_order, + lapack_int m, + lapack_int n, + double *a, + lapack_int lda, + double *b, + lapack_int ldb, + double *t, + lapack_int ldt); +lapack_int LAPACKE_ctpqrt2_work(int matrix_order, + lapack_int m, + lapack_int n, + lapack_complex_float *a, + lapack_int lda, + lapack_complex_float *b, + lapack_int ldb, + lapack_complex_float *t, + lapack_int ldt); +lapack_int LAPACKE_ztpqrt2_work(int matrix_order, + lapack_int m, + lapack_int n, + lapack_complex_double *a, + lapack_int lda, + lapack_complex_double *b, + lapack_int ldb, + lapack_complex_double *t, + lapack_int ldt); + +lapack_int LAPACKE_stprfb_work(int matrix_order, + char side, + char trans, + char direct, + char storev, + lapack_int m, + lapack_int n, + lapack_int k, + lapack_int l, + const float *v, + lapack_int ldv, + const float *t, + lapack_int ldt, + float *a, + lapack_int lda, + float *b, + lapack_int ldb, + const float *mywork, + lapack_int myldwork); +lapack_int LAPACKE_dtprfb_work(int matrix_order, + char side, + char trans, + char direct, + char storev, + lapack_int m, + lapack_int n, + lapack_int k, + lapack_int l, + const double *v, + lapack_int ldv, + const double *t, + lapack_int ldt, + double *a, + lapack_int lda, + double *b, + lapack_int ldb, + const double *mywork, + lapack_int myldwork); +lapack_int LAPACKE_ctprfb_work(int matrix_order, + char side, + char trans, + char direct, + char storev, + lapack_int m, + lapack_int n, + lapack_int k, + lapack_int l, + const lapack_complex_float *v, + lapack_int ldv, + const lapack_complex_float *t, + lapack_int ldt, + lapack_complex_float *a, + lapack_int lda, + lapack_complex_float *b, + lapack_int ldb, + const float *mywork, + lapack_int myldwork); +lapack_int LAPACKE_ztprfb_work(int matrix_order, + char side, + char trans, + char direct, + char storev, + lapack_int m, + lapack_int n, + lapack_int k, + lapack_int l, + const lapack_complex_double *v, + lapack_int ldv, + const lapack_complex_double *t, + lapack_int ldt, + lapack_complex_double *a, + lapack_int lda, + lapack_complex_double *b, + lapack_int ldb, + const double *mywork, + lapack_int myldwork); +// LAPACK 3.X.X +lapack_int LAPACKE_csyr(int matrix_order, + char uplo, + lapack_int n, + lapack_complex_float alpha, + const lapack_complex_float *x, + lapack_int incx, + lapack_complex_float *a, + lapack_int lda); +lapack_int LAPACKE_zsyr(int matrix_order, + char uplo, + lapack_int n, + lapack_complex_double alpha, + const lapack_complex_double *x, + lapack_int incx, + lapack_complex_double *a, + lapack_int lda); + +lapack_int LAPACKE_csyr_work(int matrix_order, + char uplo, + lapack_int n, + lapack_complex_float alpha, + const lapack_complex_float *x, + lapack_int incx, + lapack_complex_float *a, + lapack_int lda); +lapack_int LAPACKE_zsyr_work(int matrix_order, + char uplo, + lapack_int n, + lapack_complex_double alpha, + const lapack_complex_double *x, + lapack_int incx, + lapack_complex_double *a, + lapack_int lda); + + +#define LAPACK_sgetrf LAPACK_GLOBAL(sgetrf, SGETRF) +#define LAPACK_dgetrf LAPACK_GLOBAL(dgetrf, DGETRF) +#define LAPACK_cgetrf LAPACK_GLOBAL(cgetrf, CGETRF) +#define LAPACK_zgetrf LAPACK_GLOBAL(zgetrf, ZGETRF) +#define LAPACK_sgbtrf LAPACK_GLOBAL(sgbtrf, SGBTRF) +#define LAPACK_dgbtrf LAPACK_GLOBAL(dgbtrf, DGBTRF) +#define LAPACK_cgbtrf LAPACK_GLOBAL(cgbtrf, CGBTRF) +#define LAPACK_zgbtrf LAPACK_GLOBAL(zgbtrf, ZGBTRF) +#define LAPACK_sgttrf LAPACK_GLOBAL(sgttrf, SGTTRF) +#define LAPACK_dgttrf LAPACK_GLOBAL(dgttrf, DGTTRF) +#define LAPACK_cgttrf LAPACK_GLOBAL(cgttrf, CGTTRF) +#define LAPACK_zgttrf LAPACK_GLOBAL(zgttrf, ZGTTRF) +#define LAPACK_spotrf LAPACK_GLOBAL(spotrf, SPOTRF) +#define LAPACK_dpotrf LAPACK_GLOBAL(dpotrf, DPOTRF) +#define LAPACK_cpotrf LAPACK_GLOBAL(cpotrf, CPOTRF) +#define LAPACK_zpotrf LAPACK_GLOBAL(zpotrf, ZPOTRF) +#define LAPACK_dpstrf LAPACK_GLOBAL(dpstrf, DPSTRF) +#define LAPACK_spstrf LAPACK_GLOBAL(spstrf, SPSTRF) +#define LAPACK_zpstrf LAPACK_GLOBAL(zpstrf, ZPSTRF) +#define LAPACK_cpstrf LAPACK_GLOBAL(cpstrf, CPSTRF) +#define LAPACK_dpftrf LAPACK_GLOBAL(dpftrf, DPFTRF) +#define LAPACK_spftrf LAPACK_GLOBAL(spftrf, SPFTRF) +#define LAPACK_zpftrf LAPACK_GLOBAL(zpftrf, ZPFTRF) +#define LAPACK_cpftrf LAPACK_GLOBAL(cpftrf, CPFTRF) +#define LAPACK_spptrf LAPACK_GLOBAL(spptrf, SPPTRF) +#define LAPACK_dpptrf LAPACK_GLOBAL(dpptrf, DPPTRF) +#define LAPACK_cpptrf LAPACK_GLOBAL(cpptrf, CPPTRF) +#define LAPACK_zpptrf LAPACK_GLOBAL(zpptrf, ZPPTRF) +#define LAPACK_spbtrf LAPACK_GLOBAL(spbtrf, SPBTRF) +#define LAPACK_dpbtrf LAPACK_GLOBAL(dpbtrf, DPBTRF) +#define LAPACK_cpbtrf LAPACK_GLOBAL(cpbtrf, CPBTRF) +#define LAPACK_zpbtrf LAPACK_GLOBAL(zpbtrf, ZPBTRF) +#define LAPACK_spttrf LAPACK_GLOBAL(spttrf, SPTTRF) +#define LAPACK_dpttrf LAPACK_GLOBAL(dpttrf, DPTTRF) +#define LAPACK_cpttrf LAPACK_GLOBAL(cpttrf, CPTTRF) +#define LAPACK_zpttrf LAPACK_GLOBAL(zpttrf, ZPTTRF) +#define LAPACK_ssytrf LAPACK_GLOBAL(ssytrf, SSYTRF) +#define LAPACK_dsytrf LAPACK_GLOBAL(dsytrf, DSYTRF) +#define LAPACK_csytrf LAPACK_GLOBAL(csytrf, CSYTRF) +#define LAPACK_zsytrf LAPACK_GLOBAL(zsytrf, ZSYTRF) +#define LAPACK_chetrf LAPACK_GLOBAL(chetrf, CHETRF) +#define LAPACK_zhetrf LAPACK_GLOBAL(zhetrf, ZHETRF) +#define LAPACK_ssptrf LAPACK_GLOBAL(ssptrf, SSPTRF) +#define LAPACK_dsptrf LAPACK_GLOBAL(dsptrf, DSPTRF) +#define LAPACK_csptrf LAPACK_GLOBAL(csptrf, CSPTRF) +#define LAPACK_zsptrf LAPACK_GLOBAL(zsptrf, ZSPTRF) +#define LAPACK_chptrf LAPACK_GLOBAL(chptrf, CHPTRF) +#define LAPACK_zhptrf LAPACK_GLOBAL(zhptrf, ZHPTRF) +#define LAPACK_sgetrs LAPACK_GLOBAL(sgetrs, SGETRS) +#define LAPACK_dgetrs LAPACK_GLOBAL(dgetrs, DGETRS) +#define LAPACK_cgetrs LAPACK_GLOBAL(cgetrs, CGETRS) +#define LAPACK_zgetrs LAPACK_GLOBAL(zgetrs, ZGETRS) +#define LAPACK_sgbtrs LAPACK_GLOBAL(sgbtrs, SGBTRS) +#define LAPACK_dgbtrs LAPACK_GLOBAL(dgbtrs, DGBTRS) +#define LAPACK_cgbtrs LAPACK_GLOBAL(cgbtrs, CGBTRS) +#define LAPACK_zgbtrs LAPACK_GLOBAL(zgbtrs, ZGBTRS) +#define LAPACK_sgttrs LAPACK_GLOBAL(sgttrs, SGTTRS) +#define LAPACK_dgttrs LAPACK_GLOBAL(dgttrs, DGTTRS) +#define LAPACK_cgttrs LAPACK_GLOBAL(cgttrs, CGTTRS) +#define LAPACK_zgttrs LAPACK_GLOBAL(zgttrs, ZGTTRS) +#define LAPACK_spotrs LAPACK_GLOBAL(spotrs, SPOTRS) +#define LAPACK_dpotrs LAPACK_GLOBAL(dpotrs, DPOTRS) +#define LAPACK_cpotrs LAPACK_GLOBAL(cpotrs, CPOTRS) +#define LAPACK_zpotrs LAPACK_GLOBAL(zpotrs, ZPOTRS) +#define LAPACK_dpftrs LAPACK_GLOBAL(dpftrs, DPFTRS) +#define LAPACK_spftrs LAPACK_GLOBAL(spftrs, SPFTRS) +#define LAPACK_zpftrs LAPACK_GLOBAL(zpftrs, ZPFTRS) +#define LAPACK_cpftrs LAPACK_GLOBAL(cpftrs, CPFTRS) +#define LAPACK_spptrs LAPACK_GLOBAL(spptrs, SPPTRS) +#define LAPACK_dpptrs LAPACK_GLOBAL(dpptrs, DPPTRS) +#define LAPACK_cpptrs LAPACK_GLOBAL(cpptrs, CPPTRS) +#define LAPACK_zpptrs LAPACK_GLOBAL(zpptrs, ZPPTRS) +#define LAPACK_spbtrs LAPACK_GLOBAL(spbtrs, SPBTRS) +#define LAPACK_dpbtrs LAPACK_GLOBAL(dpbtrs, DPBTRS) +#define LAPACK_cpbtrs LAPACK_GLOBAL(cpbtrs, CPBTRS) +#define LAPACK_zpbtrs LAPACK_GLOBAL(zpbtrs, ZPBTRS) +#define LAPACK_spttrs LAPACK_GLOBAL(spttrs, SPTTRS) +#define LAPACK_dpttrs LAPACK_GLOBAL(dpttrs, DPTTRS) +#define LAPACK_cpttrs LAPACK_GLOBAL(cpttrs, CPTTRS) +#define LAPACK_zpttrs LAPACK_GLOBAL(zpttrs, ZPTTRS) +#define LAPACK_ssytrs LAPACK_GLOBAL(ssytrs, SSYTRS) +#define LAPACK_dsytrs LAPACK_GLOBAL(dsytrs, DSYTRS) +#define LAPACK_csytrs LAPACK_GLOBAL(csytrs, CSYTRS) +#define LAPACK_zsytrs LAPACK_GLOBAL(zsytrs, ZSYTRS) +#define LAPACK_chetrs LAPACK_GLOBAL(chetrs, CHETRS) +#define LAPACK_zhetrs LAPACK_GLOBAL(zhetrs, ZHETRS) +#define LAPACK_ssptrs LAPACK_GLOBAL(ssptrs, SSPTRS) +#define LAPACK_dsptrs LAPACK_GLOBAL(dsptrs, DSPTRS) +#define LAPACK_csptrs LAPACK_GLOBAL(csptrs, CSPTRS) +#define LAPACK_zsptrs LAPACK_GLOBAL(zsptrs, ZSPTRS) +#define LAPACK_chptrs LAPACK_GLOBAL(chptrs, CHPTRS) +#define LAPACK_zhptrs LAPACK_GLOBAL(zhptrs, ZHPTRS) +#define LAPACK_strtrs LAPACK_GLOBAL(strtrs, STRTRS) +#define LAPACK_dtrtrs LAPACK_GLOBAL(dtrtrs, DTRTRS) +#define LAPACK_ctrtrs LAPACK_GLOBAL(ctrtrs, CTRTRS) +#define LAPACK_ztrtrs LAPACK_GLOBAL(ztrtrs, ZTRTRS) +#define LAPACK_stptrs LAPACK_GLOBAL(stptrs, STPTRS) +#define LAPACK_dtptrs LAPACK_GLOBAL(dtptrs, DTPTRS) +#define LAPACK_ctptrs LAPACK_GLOBAL(ctptrs, CTPTRS) +#define LAPACK_ztptrs LAPACK_GLOBAL(ztptrs, ZTPTRS) +#define LAPACK_stbtrs LAPACK_GLOBAL(stbtrs, STBTRS) +#define LAPACK_dtbtrs LAPACK_GLOBAL(dtbtrs, DTBTRS) +#define LAPACK_ctbtrs LAPACK_GLOBAL(ctbtrs, CTBTRS) +#define LAPACK_ztbtrs LAPACK_GLOBAL(ztbtrs, ZTBTRS) +#define LAPACK_sgecon LAPACK_GLOBAL(sgecon, SGECON) +#define LAPACK_dgecon LAPACK_GLOBAL(dgecon, DGECON) +#define LAPACK_cgecon LAPACK_GLOBAL(cgecon, CGECON) +#define LAPACK_zgecon LAPACK_GLOBAL(zgecon, ZGECON) +#define LAPACK_sgbcon LAPACK_GLOBAL(sgbcon, SGBCON) +#define LAPACK_dgbcon LAPACK_GLOBAL(dgbcon, DGBCON) +#define LAPACK_cgbcon LAPACK_GLOBAL(cgbcon, CGBCON) +#define LAPACK_zgbcon LAPACK_GLOBAL(zgbcon, ZGBCON) +#define LAPACK_sgtcon LAPACK_GLOBAL(sgtcon, SGTCON) +#define LAPACK_dgtcon LAPACK_GLOBAL(dgtcon, DGTCON) +#define LAPACK_cgtcon LAPACK_GLOBAL(cgtcon, CGTCON) +#define LAPACK_zgtcon LAPACK_GLOBAL(zgtcon, ZGTCON) +#define LAPACK_spocon LAPACK_GLOBAL(spocon, SPOCON) +#define LAPACK_dpocon LAPACK_GLOBAL(dpocon, DPOCON) +#define LAPACK_cpocon LAPACK_GLOBAL(cpocon, CPOCON) +#define LAPACK_zpocon LAPACK_GLOBAL(zpocon, ZPOCON) +#define LAPACK_sppcon LAPACK_GLOBAL(sppcon, SPPCON) +#define LAPACK_dppcon LAPACK_GLOBAL(dppcon, DPPCON) +#define LAPACK_cppcon LAPACK_GLOBAL(cppcon, CPPCON) +#define LAPACK_zppcon LAPACK_GLOBAL(zppcon, ZPPCON) +#define LAPACK_spbcon LAPACK_GLOBAL(spbcon, SPBCON) +#define LAPACK_dpbcon LAPACK_GLOBAL(dpbcon, DPBCON) +#define LAPACK_cpbcon LAPACK_GLOBAL(cpbcon, CPBCON) +#define LAPACK_zpbcon LAPACK_GLOBAL(zpbcon, ZPBCON) +#define LAPACK_sptcon LAPACK_GLOBAL(sptcon, SPTCON) +#define LAPACK_dptcon LAPACK_GLOBAL(dptcon, DPTCON) +#define LAPACK_cptcon LAPACK_GLOBAL(cptcon, CPTCON) +#define LAPACK_zptcon LAPACK_GLOBAL(zptcon, ZPTCON) +#define LAPACK_ssycon LAPACK_GLOBAL(ssycon, SSYCON) +#define LAPACK_dsycon LAPACK_GLOBAL(dsycon, DSYCON) +#define LAPACK_csycon LAPACK_GLOBAL(csycon, CSYCON) +#define LAPACK_zsycon LAPACK_GLOBAL(zsycon, ZSYCON) +#define LAPACK_checon LAPACK_GLOBAL(checon, CHECON) +#define LAPACK_zhecon LAPACK_GLOBAL(zhecon, ZHECON) +#define LAPACK_sspcon LAPACK_GLOBAL(sspcon, SSPCON) +#define LAPACK_dspcon LAPACK_GLOBAL(dspcon, DSPCON) +#define LAPACK_cspcon LAPACK_GLOBAL(cspcon, CSPCON) +#define LAPACK_zspcon LAPACK_GLOBAL(zspcon, ZSPCON) +#define LAPACK_chpcon LAPACK_GLOBAL(chpcon, CHPCON) +#define LAPACK_zhpcon LAPACK_GLOBAL(zhpcon, ZHPCON) +#define LAPACK_strcon LAPACK_GLOBAL(strcon, STRCON) +#define LAPACK_dtrcon LAPACK_GLOBAL(dtrcon, DTRCON) +#define LAPACK_ctrcon LAPACK_GLOBAL(ctrcon, CTRCON) +#define LAPACK_ztrcon LAPACK_GLOBAL(ztrcon, ZTRCON) +#define LAPACK_stpcon LAPACK_GLOBAL(stpcon, STPCON) +#define LAPACK_dtpcon LAPACK_GLOBAL(dtpcon, DTPCON) +#define LAPACK_ctpcon LAPACK_GLOBAL(ctpcon, CTPCON) +#define LAPACK_ztpcon LAPACK_GLOBAL(ztpcon, ZTPCON) +#define LAPACK_stbcon LAPACK_GLOBAL(stbcon, STBCON) +#define LAPACK_dtbcon LAPACK_GLOBAL(dtbcon, DTBCON) +#define LAPACK_ctbcon LAPACK_GLOBAL(ctbcon, CTBCON) +#define LAPACK_ztbcon LAPACK_GLOBAL(ztbcon, ZTBCON) +#define LAPACK_sgerfs LAPACK_GLOBAL(sgerfs, SGERFS) +#define LAPACK_dgerfs LAPACK_GLOBAL(dgerfs, DGERFS) +#define LAPACK_cgerfs LAPACK_GLOBAL(cgerfs, CGERFS) +#define LAPACK_zgerfs LAPACK_GLOBAL(zgerfs, ZGERFS) +#define LAPACK_dgerfsx LAPACK_GLOBAL(dgerfsx, DGERFSX) +#define LAPACK_sgerfsx LAPACK_GLOBAL(sgerfsx, SGERFSX) +#define LAPACK_zgerfsx LAPACK_GLOBAL(zgerfsx, ZGERFSX) +#define LAPACK_cgerfsx LAPACK_GLOBAL(cgerfsx, CGERFSX) +#define LAPACK_sgbrfs LAPACK_GLOBAL(sgbrfs, SGBRFS) +#define LAPACK_dgbrfs LAPACK_GLOBAL(dgbrfs, DGBRFS) +#define LAPACK_cgbrfs LAPACK_GLOBAL(cgbrfs, CGBRFS) +#define LAPACK_zgbrfs LAPACK_GLOBAL(zgbrfs, ZGBRFS) +#define LAPACK_dgbrfsx LAPACK_GLOBAL(dgbrfsx, DGBRFSX) +#define LAPACK_sgbrfsx LAPACK_GLOBAL(sgbrfsx, SGBRFSX) +#define LAPACK_zgbrfsx LAPACK_GLOBAL(zgbrfsx, ZGBRFSX) +#define LAPACK_cgbrfsx LAPACK_GLOBAL(cgbrfsx, CGBRFSX) +#define LAPACK_sgtrfs LAPACK_GLOBAL(sgtrfs, SGTRFS) +#define LAPACK_dgtrfs LAPACK_GLOBAL(dgtrfs, DGTRFS) +#define LAPACK_cgtrfs LAPACK_GLOBAL(cgtrfs, CGTRFS) +#define LAPACK_zgtrfs LAPACK_GLOBAL(zgtrfs, ZGTRFS) +#define LAPACK_sporfs LAPACK_GLOBAL(sporfs, SPORFS) +#define LAPACK_dporfs LAPACK_GLOBAL(dporfs, DPORFS) +#define LAPACK_cporfs LAPACK_GLOBAL(cporfs, CPORFS) +#define LAPACK_zporfs LAPACK_GLOBAL(zporfs, ZPORFS) +#define LAPACK_dporfsx LAPACK_GLOBAL(dporfsx, DPORFSX) +#define LAPACK_sporfsx LAPACK_GLOBAL(sporfsx, SPORFSX) +#define LAPACK_zporfsx LAPACK_GLOBAL(zporfsx, ZPORFSX) +#define LAPACK_cporfsx LAPACK_GLOBAL(cporfsx, CPORFSX) +#define LAPACK_spprfs LAPACK_GLOBAL(spprfs, SPPRFS) +#define LAPACK_dpprfs LAPACK_GLOBAL(dpprfs, DPPRFS) +#define LAPACK_cpprfs LAPACK_GLOBAL(cpprfs, CPPRFS) +#define LAPACK_zpprfs LAPACK_GLOBAL(zpprfs, ZPPRFS) +#define LAPACK_spbrfs LAPACK_GLOBAL(spbrfs, SPBRFS) +#define LAPACK_dpbrfs LAPACK_GLOBAL(dpbrfs, DPBRFS) +#define LAPACK_cpbrfs LAPACK_GLOBAL(cpbrfs, CPBRFS) +#define LAPACK_zpbrfs LAPACK_GLOBAL(zpbrfs, ZPBRFS) +#define LAPACK_sptrfs LAPACK_GLOBAL(sptrfs, SPTRFS) +#define LAPACK_dptrfs LAPACK_GLOBAL(dptrfs, DPTRFS) +#define LAPACK_cptrfs LAPACK_GLOBAL(cptrfs, CPTRFS) +#define LAPACK_zptrfs LAPACK_GLOBAL(zptrfs, ZPTRFS) +#define LAPACK_ssyrfs LAPACK_GLOBAL(ssyrfs, SSYRFS) +#define LAPACK_dsyrfs LAPACK_GLOBAL(dsyrfs, DSYRFS) +#define LAPACK_csyrfs LAPACK_GLOBAL(csyrfs, CSYRFS) +#define LAPACK_zsyrfs LAPACK_GLOBAL(zsyrfs, ZSYRFS) +#define LAPACK_dsyrfsx LAPACK_GLOBAL(dsyrfsx, DSYRFSX) +#define LAPACK_ssyrfsx LAPACK_GLOBAL(ssyrfsx, SSYRFSX) +#define LAPACK_zsyrfsx LAPACK_GLOBAL(zsyrfsx, ZSYRFSX) +#define LAPACK_csyrfsx LAPACK_GLOBAL(csyrfsx, CSYRFSX) +#define LAPACK_cherfs LAPACK_GLOBAL(cherfs, CHERFS) +#define LAPACK_zherfs LAPACK_GLOBAL(zherfs, ZHERFS) +#define LAPACK_zherfsx LAPACK_GLOBAL(zherfsx, ZHERFSX) +#define LAPACK_cherfsx LAPACK_GLOBAL(cherfsx, CHERFSX) +#define LAPACK_ssprfs LAPACK_GLOBAL(ssprfs, SSPRFS) +#define LAPACK_dsprfs LAPACK_GLOBAL(dsprfs, DSPRFS) +#define LAPACK_csprfs LAPACK_GLOBAL(csprfs, CSPRFS) +#define LAPACK_zsprfs LAPACK_GLOBAL(zsprfs, ZSPRFS) +#define LAPACK_chprfs LAPACK_GLOBAL(chprfs, CHPRFS) +#define LAPACK_zhprfs LAPACK_GLOBAL(zhprfs, ZHPRFS) +#define LAPACK_strrfs LAPACK_GLOBAL(strrfs, STRRFS) +#define LAPACK_dtrrfs LAPACK_GLOBAL(dtrrfs, DTRRFS) +#define LAPACK_ctrrfs LAPACK_GLOBAL(ctrrfs, CTRRFS) +#define LAPACK_ztrrfs LAPACK_GLOBAL(ztrrfs, ZTRRFS) +#define LAPACK_stprfs LAPACK_GLOBAL(stprfs, STPRFS) +#define LAPACK_dtprfs LAPACK_GLOBAL(dtprfs, DTPRFS) +#define LAPACK_ctprfs LAPACK_GLOBAL(ctprfs, CTPRFS) +#define LAPACK_ztprfs LAPACK_GLOBAL(ztprfs, ZTPRFS) +#define LAPACK_stbrfs LAPACK_GLOBAL(stbrfs, STBRFS) +#define LAPACK_dtbrfs LAPACK_GLOBAL(dtbrfs, DTBRFS) +#define LAPACK_ctbrfs LAPACK_GLOBAL(ctbrfs, CTBRFS) +#define LAPACK_ztbrfs LAPACK_GLOBAL(ztbrfs, ZTBRFS) +#define LAPACK_sgetri LAPACK_GLOBAL(sgetri, SGETRI) +#define LAPACK_dgetri LAPACK_GLOBAL(dgetri, DGETRI) +#define LAPACK_cgetri LAPACK_GLOBAL(cgetri, CGETRI) +#define LAPACK_zgetri LAPACK_GLOBAL(zgetri, ZGETRI) +#define LAPACK_spotri LAPACK_GLOBAL(spotri, SPOTRI) +#define LAPACK_dpotri LAPACK_GLOBAL(dpotri, DPOTRI) +#define LAPACK_cpotri LAPACK_GLOBAL(cpotri, CPOTRI) +#define LAPACK_zpotri LAPACK_GLOBAL(zpotri, ZPOTRI) +#define LAPACK_dpftri LAPACK_GLOBAL(dpftri, DPFTRI) +#define LAPACK_spftri LAPACK_GLOBAL(spftri, SPFTRI) +#define LAPACK_zpftri LAPACK_GLOBAL(zpftri, ZPFTRI) +#define LAPACK_cpftri LAPACK_GLOBAL(cpftri, CPFTRI) +#define LAPACK_spptri LAPACK_GLOBAL(spptri, SPPTRI) +#define LAPACK_dpptri LAPACK_GLOBAL(dpptri, DPPTRI) +#define LAPACK_cpptri LAPACK_GLOBAL(cpptri, CPPTRI) +#define LAPACK_zpptri LAPACK_GLOBAL(zpptri, ZPPTRI) +#define LAPACK_ssytri LAPACK_GLOBAL(ssytri, SSYTRI) +#define LAPACK_dsytri LAPACK_GLOBAL(dsytri, DSYTRI) +#define LAPACK_csytri LAPACK_GLOBAL(csytri, CSYTRI) +#define LAPACK_zsytri LAPACK_GLOBAL(zsytri, ZSYTRI) +#define LAPACK_chetri LAPACK_GLOBAL(chetri, CHETRI) +#define LAPACK_zhetri LAPACK_GLOBAL(zhetri, ZHETRI) +#define LAPACK_ssptri LAPACK_GLOBAL(ssptri, SSPTRI) +#define LAPACK_dsptri LAPACK_GLOBAL(dsptri, DSPTRI) +#define LAPACK_csptri LAPACK_GLOBAL(csptri, CSPTRI) +#define LAPACK_zsptri LAPACK_GLOBAL(zsptri, ZSPTRI) +#define LAPACK_chptri LAPACK_GLOBAL(chptri, CHPTRI) +#define LAPACK_zhptri LAPACK_GLOBAL(zhptri, ZHPTRI) +#define LAPACK_strtri LAPACK_GLOBAL(strtri, STRTRI) +#define LAPACK_dtrtri LAPACK_GLOBAL(dtrtri, DTRTRI) +#define LAPACK_ctrtri LAPACK_GLOBAL(ctrtri, CTRTRI) +#define LAPACK_ztrtri LAPACK_GLOBAL(ztrtri, ZTRTRI) +#define LAPACK_dtftri LAPACK_GLOBAL(dtftri, DTFTRI) +#define LAPACK_stftri LAPACK_GLOBAL(stftri, STFTRI) +#define LAPACK_ztftri LAPACK_GLOBAL(ztftri, ZTFTRI) +#define LAPACK_ctftri LAPACK_GLOBAL(ctftri, CTFTRI) +#define LAPACK_stptri LAPACK_GLOBAL(stptri, STPTRI) +#define LAPACK_dtptri LAPACK_GLOBAL(dtptri, DTPTRI) +#define LAPACK_ctptri LAPACK_GLOBAL(ctptri, CTPTRI) +#define LAPACK_ztptri LAPACK_GLOBAL(ztptri, ZTPTRI) +#define LAPACK_sgeequ LAPACK_GLOBAL(sgeequ, SGEEQU) +#define LAPACK_dgeequ LAPACK_GLOBAL(dgeequ, DGEEQU) +#define LAPACK_cgeequ LAPACK_GLOBAL(cgeequ, CGEEQU) +#define LAPACK_zgeequ LAPACK_GLOBAL(zgeequ, ZGEEQU) +#define LAPACK_dgeequb LAPACK_GLOBAL(dgeequb, DGEEQUB) +#define LAPACK_sgeequb LAPACK_GLOBAL(sgeequb, SGEEQUB) +#define LAPACK_zgeequb LAPACK_GLOBAL(zgeequb, ZGEEQUB) +#define LAPACK_cgeequb LAPACK_GLOBAL(cgeequb, CGEEQUB) +#define LAPACK_sgbequ LAPACK_GLOBAL(sgbequ, SGBEQU) +#define LAPACK_dgbequ LAPACK_GLOBAL(dgbequ, DGBEQU) +#define LAPACK_cgbequ LAPACK_GLOBAL(cgbequ, CGBEQU) +#define LAPACK_zgbequ LAPACK_GLOBAL(zgbequ, ZGBEQU) +#define LAPACK_dgbequb LAPACK_GLOBAL(dgbequb, DGBEQUB) +#define LAPACK_sgbequb LAPACK_GLOBAL(sgbequb, SGBEQUB) +#define LAPACK_zgbequb LAPACK_GLOBAL(zgbequb, ZGBEQUB) +#define LAPACK_cgbequb LAPACK_GLOBAL(cgbequb, CGBEQUB) +#define LAPACK_spoequ LAPACK_GLOBAL(spoequ, SPOEQU) +#define LAPACK_dpoequ LAPACK_GLOBAL(dpoequ, DPOEQU) +#define LAPACK_cpoequ LAPACK_GLOBAL(cpoequ, CPOEQU) +#define LAPACK_zpoequ LAPACK_GLOBAL(zpoequ, ZPOEQU) +#define LAPACK_dpoequb LAPACK_GLOBAL(dpoequb, DPOEQUB) +#define LAPACK_spoequb LAPACK_GLOBAL(spoequb, SPOEQUB) +#define LAPACK_zpoequb LAPACK_GLOBAL(zpoequb, ZPOEQUB) +#define LAPACK_cpoequb LAPACK_GLOBAL(cpoequb, CPOEQUB) +#define LAPACK_sppequ LAPACK_GLOBAL(sppequ, SPPEQU) +#define LAPACK_dppequ LAPACK_GLOBAL(dppequ, DPPEQU) +#define LAPACK_cppequ LAPACK_GLOBAL(cppequ, CPPEQU) +#define LAPACK_zppequ LAPACK_GLOBAL(zppequ, ZPPEQU) +#define LAPACK_spbequ LAPACK_GLOBAL(spbequ, SPBEQU) +#define LAPACK_dpbequ LAPACK_GLOBAL(dpbequ, DPBEQU) +#define LAPACK_cpbequ LAPACK_GLOBAL(cpbequ, CPBEQU) +#define LAPACK_zpbequ LAPACK_GLOBAL(zpbequ, ZPBEQU) +#define LAPACK_dsyequb LAPACK_GLOBAL(dsyequb, DSYEQUB) +#define LAPACK_ssyequb LAPACK_GLOBAL(ssyequb, SSYEQUB) +#define LAPACK_zsyequb LAPACK_GLOBAL(zsyequb, ZSYEQUB) +#define LAPACK_csyequb LAPACK_GLOBAL(csyequb, CSYEQUB) +#define LAPACK_zheequb LAPACK_GLOBAL(zheequb, ZHEEQUB) +#define LAPACK_cheequb LAPACK_GLOBAL(cheequb, CHEEQUB) +#define LAPACK_sgesv LAPACK_GLOBAL(sgesv, SGESV) +#define LAPACK_dgesv LAPACK_GLOBAL(dgesv, DGESV) +#define LAPACK_cgesv LAPACK_GLOBAL(cgesv, CGESV) +#define LAPACK_zgesv LAPACK_GLOBAL(zgesv, ZGESV) +#define LAPACK_dsgesv LAPACK_GLOBAL(dsgesv, DSGESV) +#define LAPACK_zcgesv LAPACK_GLOBAL(zcgesv, ZCGESV) +#define LAPACK_sgesvx LAPACK_GLOBAL(sgesvx, SGESVX) +#define LAPACK_dgesvx LAPACK_GLOBAL(dgesvx, DGESVX) +#define LAPACK_cgesvx LAPACK_GLOBAL(cgesvx, CGESVX) +#define LAPACK_zgesvx LAPACK_GLOBAL(zgesvx, ZGESVX) +#define LAPACK_dgesvxx LAPACK_GLOBAL(dgesvxx, DGESVXX) +#define LAPACK_sgesvxx LAPACK_GLOBAL(sgesvxx, SGESVXX) +#define LAPACK_zgesvxx LAPACK_GLOBAL(zgesvxx, ZGESVXX) +#define LAPACK_cgesvxx LAPACK_GLOBAL(cgesvxx, CGESVXX) +#define LAPACK_sgbsv LAPACK_GLOBAL(sgbsv, SGBSV) +#define LAPACK_dgbsv LAPACK_GLOBAL(dgbsv, DGBSV) +#define LAPACK_cgbsv LAPACK_GLOBAL(cgbsv, CGBSV) +#define LAPACK_zgbsv LAPACK_GLOBAL(zgbsv, ZGBSV) +#define LAPACK_sgbsvx LAPACK_GLOBAL(sgbsvx, SGBSVX) +#define LAPACK_dgbsvx LAPACK_GLOBAL(dgbsvx, DGBSVX) +#define LAPACK_cgbsvx LAPACK_GLOBAL(cgbsvx, CGBSVX) +#define LAPACK_zgbsvx LAPACK_GLOBAL(zgbsvx, ZGBSVX) +#define LAPACK_dgbsvxx LAPACK_GLOBAL(dgbsvxx, DGBSVXX) +#define LAPACK_sgbsvxx LAPACK_GLOBAL(sgbsvxx, SGBSVXX) +#define LAPACK_zgbsvxx LAPACK_GLOBAL(zgbsvxx, ZGBSVXX) +#define LAPACK_cgbsvxx LAPACK_GLOBAL(cgbsvxx, CGBSVXX) +#define LAPACK_sgtsv LAPACK_GLOBAL(sgtsv, SGTSV) +#define LAPACK_dgtsv LAPACK_GLOBAL(dgtsv, DGTSV) +#define LAPACK_cgtsv LAPACK_GLOBAL(cgtsv, CGTSV) +#define LAPACK_zgtsv LAPACK_GLOBAL(zgtsv, ZGTSV) +#define LAPACK_sgtsvx LAPACK_GLOBAL(sgtsvx, SGTSVX) +#define LAPACK_dgtsvx LAPACK_GLOBAL(dgtsvx, DGTSVX) +#define LAPACK_cgtsvx LAPACK_GLOBAL(cgtsvx, CGTSVX) +#define LAPACK_zgtsvx LAPACK_GLOBAL(zgtsvx, ZGTSVX) +#define LAPACK_sposv LAPACK_GLOBAL(sposv, SPOSV) +#define LAPACK_dposv LAPACK_GLOBAL(dposv, DPOSV) +#define LAPACK_cposv LAPACK_GLOBAL(cposv, CPOSV) +#define LAPACK_zposv LAPACK_GLOBAL(zposv, ZPOSV) +#define LAPACK_dsposv LAPACK_GLOBAL(dsposv, DSPOSV) +#define LAPACK_zcposv LAPACK_GLOBAL(zcposv, ZCPOSV) +#define LAPACK_sposvx LAPACK_GLOBAL(sposvx, SPOSVX) +#define LAPACK_dposvx LAPACK_GLOBAL(dposvx, DPOSVX) +#define LAPACK_cposvx LAPACK_GLOBAL(cposvx, CPOSVX) +#define LAPACK_zposvx LAPACK_GLOBAL(zposvx, ZPOSVX) +#define LAPACK_dposvxx LAPACK_GLOBAL(dposvxx, DPOSVXX) +#define LAPACK_sposvxx LAPACK_GLOBAL(sposvxx, SPOSVXX) +#define LAPACK_zposvxx LAPACK_GLOBAL(zposvxx, ZPOSVXX) +#define LAPACK_cposvxx LAPACK_GLOBAL(cposvxx, CPOSVXX) +#define LAPACK_sppsv LAPACK_GLOBAL(sppsv, SPPSV) +#define LAPACK_dppsv LAPACK_GLOBAL(dppsv, DPPSV) +#define LAPACK_cppsv LAPACK_GLOBAL(cppsv, CPPSV) +#define LAPACK_zppsv LAPACK_GLOBAL(zppsv, ZPPSV) +#define LAPACK_sppsvx LAPACK_GLOBAL(sppsvx, SPPSVX) +#define LAPACK_dppsvx LAPACK_GLOBAL(dppsvx, DPPSVX) +#define LAPACK_cppsvx LAPACK_GLOBAL(cppsvx, CPPSVX) +#define LAPACK_zppsvx LAPACK_GLOBAL(zppsvx, ZPPSVX) +#define LAPACK_spbsv LAPACK_GLOBAL(spbsv, SPBSV) +#define LAPACK_dpbsv LAPACK_GLOBAL(dpbsv, DPBSV) +#define LAPACK_cpbsv LAPACK_GLOBAL(cpbsv, CPBSV) +#define LAPACK_zpbsv LAPACK_GLOBAL(zpbsv, ZPBSV) +#define LAPACK_spbsvx LAPACK_GLOBAL(spbsvx, SPBSVX) +#define LAPACK_dpbsvx LAPACK_GLOBAL(dpbsvx, DPBSVX) +#define LAPACK_cpbsvx LAPACK_GLOBAL(cpbsvx, CPBSVX) +#define LAPACK_zpbsvx LAPACK_GLOBAL(zpbsvx, ZPBSVX) +#define LAPACK_sptsv LAPACK_GLOBAL(sptsv, SPTSV) +#define LAPACK_dptsv LAPACK_GLOBAL(dptsv, DPTSV) +#define LAPACK_cptsv LAPACK_GLOBAL(cptsv, CPTSV) +#define LAPACK_zptsv LAPACK_GLOBAL(zptsv, ZPTSV) +#define LAPACK_sptsvx LAPACK_GLOBAL(sptsvx, SPTSVX) +#define LAPACK_dptsvx LAPACK_GLOBAL(dptsvx, DPTSVX) +#define LAPACK_cptsvx LAPACK_GLOBAL(cptsvx, CPTSVX) +#define LAPACK_zptsvx LAPACK_GLOBAL(zptsvx, ZPTSVX) +#define LAPACK_ssysv LAPACK_GLOBAL(ssysv, SSYSV) +#define LAPACK_dsysv LAPACK_GLOBAL(dsysv, DSYSV) +#define LAPACK_csysv LAPACK_GLOBAL(csysv, CSYSV) +#define LAPACK_zsysv LAPACK_GLOBAL(zsysv, ZSYSV) +#define LAPACK_ssysvx LAPACK_GLOBAL(ssysvx, SSYSVX) +#define LAPACK_dsysvx LAPACK_GLOBAL(dsysvx, DSYSVX) +#define LAPACK_csysvx LAPACK_GLOBAL(csysvx, CSYSVX) +#define LAPACK_zsysvx LAPACK_GLOBAL(zsysvx, ZSYSVX) +#define LAPACK_dsysvxx LAPACK_GLOBAL(dsysvxx, DSYSVXX) +#define LAPACK_ssysvxx LAPACK_GLOBAL(ssysvxx, SSYSVXX) +#define LAPACK_zsysvxx LAPACK_GLOBAL(zsysvxx, ZSYSVXX) +#define LAPACK_csysvxx LAPACK_GLOBAL(csysvxx, CSYSVXX) +#define LAPACK_chesv LAPACK_GLOBAL(chesv, CHESV) +#define LAPACK_zhesv LAPACK_GLOBAL(zhesv, ZHESV) +#define LAPACK_chesvx LAPACK_GLOBAL(chesvx, CHESVX) +#define LAPACK_zhesvx LAPACK_GLOBAL(zhesvx, ZHESVX) +#define LAPACK_zhesvxx LAPACK_GLOBAL(zhesvxx, ZHESVXX) +#define LAPACK_chesvxx LAPACK_GLOBAL(chesvxx, CHESVXX) +#define LAPACK_sspsv LAPACK_GLOBAL(sspsv, SSPSV) +#define LAPACK_dspsv LAPACK_GLOBAL(dspsv, DSPSV) +#define LAPACK_cspsv LAPACK_GLOBAL(cspsv, CSPSV) +#define LAPACK_zspsv LAPACK_GLOBAL(zspsv, ZSPSV) +#define LAPACK_sspsvx LAPACK_GLOBAL(sspsvx, SSPSVX) +#define LAPACK_dspsvx LAPACK_GLOBAL(dspsvx, DSPSVX) +#define LAPACK_cspsvx LAPACK_GLOBAL(cspsvx, CSPSVX) +#define LAPACK_zspsvx LAPACK_GLOBAL(zspsvx, ZSPSVX) +#define LAPACK_chpsv LAPACK_GLOBAL(chpsv, CHPSV) +#define LAPACK_zhpsv LAPACK_GLOBAL(zhpsv, ZHPSV) +#define LAPACK_chpsvx LAPACK_GLOBAL(chpsvx, CHPSVX) +#define LAPACK_zhpsvx LAPACK_GLOBAL(zhpsvx, ZHPSVX) +#define LAPACK_sgeqrf LAPACK_GLOBAL(sgeqrf, SGEQRF) +#define LAPACK_dgeqrf LAPACK_GLOBAL(dgeqrf, DGEQRF) +#define LAPACK_cgeqrf LAPACK_GLOBAL(cgeqrf, CGEQRF) +#define LAPACK_zgeqrf LAPACK_GLOBAL(zgeqrf, ZGEQRF) +#define LAPACK_sgeqpf LAPACK_GLOBAL(sgeqpf, SGEQPF) +#define LAPACK_dgeqpf LAPACK_GLOBAL(dgeqpf, DGEQPF) +#define LAPACK_cgeqpf LAPACK_GLOBAL(cgeqpf, CGEQPF) +#define LAPACK_zgeqpf LAPACK_GLOBAL(zgeqpf, ZGEQPF) +#define LAPACK_sgeqp3 LAPACK_GLOBAL(sgeqp3, SGEQP3) +#define LAPACK_dgeqp3 LAPACK_GLOBAL(dgeqp3, DGEQP3) +#define LAPACK_cgeqp3 LAPACK_GLOBAL(cgeqp3, CGEQP3) +#define LAPACK_zgeqp3 LAPACK_GLOBAL(zgeqp3, ZGEQP3) +#define LAPACK_sorgqr LAPACK_GLOBAL(sorgqr, SORGQR) +#define LAPACK_dorgqr LAPACK_GLOBAL(dorgqr, DORGQR) +#define LAPACK_sormqr LAPACK_GLOBAL(sormqr, SORMQR) +#define LAPACK_dormqr LAPACK_GLOBAL(dormqr, DORMQR) +#define LAPACK_cungqr LAPACK_GLOBAL(cungqr, CUNGQR) +#define LAPACK_zungqr LAPACK_GLOBAL(zungqr, ZUNGQR) +#define LAPACK_cunmqr LAPACK_GLOBAL(cunmqr, CUNMQR) +#define LAPACK_zunmqr LAPACK_GLOBAL(zunmqr, ZUNMQR) +#define LAPACK_sgelqf LAPACK_GLOBAL(sgelqf, SGELQF) +#define LAPACK_dgelqf LAPACK_GLOBAL(dgelqf, DGELQF) +#define LAPACK_cgelqf LAPACK_GLOBAL(cgelqf, CGELQF) +#define LAPACK_zgelqf LAPACK_GLOBAL(zgelqf, ZGELQF) +#define LAPACK_sorglq LAPACK_GLOBAL(sorglq, SORGLQ) +#define LAPACK_dorglq LAPACK_GLOBAL(dorglq, DORGLQ) +#define LAPACK_sormlq LAPACK_GLOBAL(sormlq, SORMLQ) +#define LAPACK_dormlq LAPACK_GLOBAL(dormlq, DORMLQ) +#define LAPACK_cunglq LAPACK_GLOBAL(cunglq, CUNGLQ) +#define LAPACK_zunglq LAPACK_GLOBAL(zunglq, ZUNGLQ) +#define LAPACK_cunmlq LAPACK_GLOBAL(cunmlq, CUNMLQ) +#define LAPACK_zunmlq LAPACK_GLOBAL(zunmlq, ZUNMLQ) +#define LAPACK_sgeqlf LAPACK_GLOBAL(sgeqlf, SGEQLF) +#define LAPACK_dgeqlf LAPACK_GLOBAL(dgeqlf, DGEQLF) +#define LAPACK_cgeqlf LAPACK_GLOBAL(cgeqlf, CGEQLF) +#define LAPACK_zgeqlf LAPACK_GLOBAL(zgeqlf, ZGEQLF) +#define LAPACK_sorgql LAPACK_GLOBAL(sorgql, SORGQL) +#define LAPACK_dorgql LAPACK_GLOBAL(dorgql, DORGQL) +#define LAPACK_cungql LAPACK_GLOBAL(cungql, CUNGQL) +#define LAPACK_zungql LAPACK_GLOBAL(zungql, ZUNGQL) +#define LAPACK_sormql LAPACK_GLOBAL(sormql, SORMQL) +#define LAPACK_dormql LAPACK_GLOBAL(dormql, DORMQL) +#define LAPACK_cunmql LAPACK_GLOBAL(cunmql, CUNMQL) +#define LAPACK_zunmql LAPACK_GLOBAL(zunmql, ZUNMQL) +#define LAPACK_sgerqf LAPACK_GLOBAL(sgerqf, SGERQF) +#define LAPACK_dgerqf LAPACK_GLOBAL(dgerqf, DGERQF) +#define LAPACK_cgerqf LAPACK_GLOBAL(cgerqf, CGERQF) +#define LAPACK_zgerqf LAPACK_GLOBAL(zgerqf, ZGERQF) +#define LAPACK_sorgrq LAPACK_GLOBAL(sorgrq, SORGRQ) +#define LAPACK_dorgrq LAPACK_GLOBAL(dorgrq, DORGRQ) +#define LAPACK_cungrq LAPACK_GLOBAL(cungrq, CUNGRQ) +#define LAPACK_zungrq LAPACK_GLOBAL(zungrq, ZUNGRQ) +#define LAPACK_sormrq LAPACK_GLOBAL(sormrq, SORMRQ) +#define LAPACK_dormrq LAPACK_GLOBAL(dormrq, DORMRQ) +#define LAPACK_cunmrq LAPACK_GLOBAL(cunmrq, CUNMRQ) +#define LAPACK_zunmrq LAPACK_GLOBAL(zunmrq, ZUNMRQ) +#define LAPACK_stzrzf LAPACK_GLOBAL(stzrzf, STZRZF) +#define LAPACK_dtzrzf LAPACK_GLOBAL(dtzrzf, DTZRZF) +#define LAPACK_ctzrzf LAPACK_GLOBAL(ctzrzf, CTZRZF) +#define LAPACK_ztzrzf LAPACK_GLOBAL(ztzrzf, ZTZRZF) +#define LAPACK_sormrz LAPACK_GLOBAL(sormrz, SORMRZ) +#define LAPACK_dormrz LAPACK_GLOBAL(dormrz, DORMRZ) +#define LAPACK_cunmrz LAPACK_GLOBAL(cunmrz, CUNMRZ) +#define LAPACK_zunmrz LAPACK_GLOBAL(zunmrz, ZUNMRZ) +#define LAPACK_sggqrf LAPACK_GLOBAL(sggqrf, SGGQRF) +#define LAPACK_dggqrf LAPACK_GLOBAL(dggqrf, DGGQRF) +#define LAPACK_cggqrf LAPACK_GLOBAL(cggqrf, CGGQRF) +#define LAPACK_zggqrf LAPACK_GLOBAL(zggqrf, ZGGQRF) +#define LAPACK_sggrqf LAPACK_GLOBAL(sggrqf, SGGRQF) +#define LAPACK_dggrqf LAPACK_GLOBAL(dggrqf, DGGRQF) +#define LAPACK_cggrqf LAPACK_GLOBAL(cggrqf, CGGRQF) +#define LAPACK_zggrqf LAPACK_GLOBAL(zggrqf, ZGGRQF) +#define LAPACK_sgebrd LAPACK_GLOBAL(sgebrd, SGEBRD) +#define LAPACK_dgebrd LAPACK_GLOBAL(dgebrd, DGEBRD) +#define LAPACK_cgebrd LAPACK_GLOBAL(cgebrd, CGEBRD) +#define LAPACK_zgebrd LAPACK_GLOBAL(zgebrd, ZGEBRD) +#define LAPACK_sgbbrd LAPACK_GLOBAL(sgbbrd, SGBBRD) +#define LAPACK_dgbbrd LAPACK_GLOBAL(dgbbrd, DGBBRD) +#define LAPACK_cgbbrd LAPACK_GLOBAL(cgbbrd, CGBBRD) +#define LAPACK_zgbbrd LAPACK_GLOBAL(zgbbrd, ZGBBRD) +#define LAPACK_sorgbr LAPACK_GLOBAL(sorgbr, SORGBR) +#define LAPACK_dorgbr LAPACK_GLOBAL(dorgbr, DORGBR) +#define LAPACK_sormbr LAPACK_GLOBAL(sormbr, SORMBR) +#define LAPACK_dormbr LAPACK_GLOBAL(dormbr, DORMBR) +#define LAPACK_cungbr LAPACK_GLOBAL(cungbr, CUNGBR) +#define LAPACK_zungbr LAPACK_GLOBAL(zungbr, ZUNGBR) +#define LAPACK_cunmbr LAPACK_GLOBAL(cunmbr, CUNMBR) +#define LAPACK_zunmbr LAPACK_GLOBAL(zunmbr, ZUNMBR) +#define LAPACK_sbdsqr LAPACK_GLOBAL(sbdsqr, SBDSQR) +#define LAPACK_dbdsqr LAPACK_GLOBAL(dbdsqr, DBDSQR) +#define LAPACK_cbdsqr LAPACK_GLOBAL(cbdsqr, CBDSQR) +#define LAPACK_zbdsqr LAPACK_GLOBAL(zbdsqr, ZBDSQR) +#define LAPACK_sbdsdc LAPACK_GLOBAL(sbdsdc, SBDSDC) +#define LAPACK_dbdsdc LAPACK_GLOBAL(dbdsdc, DBDSDC) +#define LAPACK_ssytrd LAPACK_GLOBAL(ssytrd, SSYTRD) +#define LAPACK_dsytrd LAPACK_GLOBAL(dsytrd, DSYTRD) +#define LAPACK_sorgtr LAPACK_GLOBAL(sorgtr, SORGTR) +#define LAPACK_dorgtr LAPACK_GLOBAL(dorgtr, DORGTR) +#define LAPACK_sormtr LAPACK_GLOBAL(sormtr, SORMTR) +#define LAPACK_dormtr LAPACK_GLOBAL(dormtr, DORMTR) +#define LAPACK_chetrd LAPACK_GLOBAL(chetrd, CHETRD) +#define LAPACK_zhetrd LAPACK_GLOBAL(zhetrd, ZHETRD) +#define LAPACK_cungtr LAPACK_GLOBAL(cungtr, CUNGTR) +#define LAPACK_zungtr LAPACK_GLOBAL(zungtr, ZUNGTR) +#define LAPACK_cunmtr LAPACK_GLOBAL(cunmtr, CUNMTR) +#define LAPACK_zunmtr LAPACK_GLOBAL(zunmtr, ZUNMTR) +#define LAPACK_ssptrd LAPACK_GLOBAL(ssptrd, SSPTRD) +#define LAPACK_dsptrd LAPACK_GLOBAL(dsptrd, DSPTRD) +#define LAPACK_sopgtr LAPACK_GLOBAL(sopgtr, SOPGTR) +#define LAPACK_dopgtr LAPACK_GLOBAL(dopgtr, DOPGTR) +#define LAPACK_sopmtr LAPACK_GLOBAL(sopmtr, SOPMTR) +#define LAPACK_dopmtr LAPACK_GLOBAL(dopmtr, DOPMTR) +#define LAPACK_chptrd LAPACK_GLOBAL(chptrd, CHPTRD) +#define LAPACK_zhptrd LAPACK_GLOBAL(zhptrd, ZHPTRD) +#define LAPACK_cupgtr LAPACK_GLOBAL(cupgtr, CUPGTR) +#define LAPACK_zupgtr LAPACK_GLOBAL(zupgtr, ZUPGTR) +#define LAPACK_cupmtr LAPACK_GLOBAL(cupmtr, CUPMTR) +#define LAPACK_zupmtr LAPACK_GLOBAL(zupmtr, ZUPMTR) +#define LAPACK_ssbtrd LAPACK_GLOBAL(ssbtrd, SSBTRD) +#define LAPACK_dsbtrd LAPACK_GLOBAL(dsbtrd, DSBTRD) +#define LAPACK_chbtrd LAPACK_GLOBAL(chbtrd, CHBTRD) +#define LAPACK_zhbtrd LAPACK_GLOBAL(zhbtrd, ZHBTRD) +#define LAPACK_ssterf LAPACK_GLOBAL(ssterf, SSTERF) +#define LAPACK_dsterf LAPACK_GLOBAL(dsterf, DSTERF) +#define LAPACK_ssteqr LAPACK_GLOBAL(ssteqr, SSTEQR) +#define LAPACK_dsteqr LAPACK_GLOBAL(dsteqr, DSTEQR) +#define LAPACK_csteqr LAPACK_GLOBAL(csteqr, CSTEQR) +#define LAPACK_zsteqr LAPACK_GLOBAL(zsteqr, ZSTEQR) +#define LAPACK_sstemr LAPACK_GLOBAL(sstemr, SSTEMR) +#define LAPACK_dstemr LAPACK_GLOBAL(dstemr, DSTEMR) +#define LAPACK_cstemr LAPACK_GLOBAL(cstemr, CSTEMR) +#define LAPACK_zstemr LAPACK_GLOBAL(zstemr, ZSTEMR) +#define LAPACK_sstedc LAPACK_GLOBAL(sstedc, SSTEDC) +#define LAPACK_dstedc LAPACK_GLOBAL(dstedc, DSTEDC) +#define LAPACK_cstedc LAPACK_GLOBAL(cstedc, CSTEDC) +#define LAPACK_zstedc LAPACK_GLOBAL(zstedc, ZSTEDC) +#define LAPACK_sstegr LAPACK_GLOBAL(sstegr, SSTEGR) +#define LAPACK_dstegr LAPACK_GLOBAL(dstegr, DSTEGR) +#define LAPACK_cstegr LAPACK_GLOBAL(cstegr, CSTEGR) +#define LAPACK_zstegr LAPACK_GLOBAL(zstegr, ZSTEGR) +#define LAPACK_spteqr LAPACK_GLOBAL(spteqr, SPTEQR) +#define LAPACK_dpteqr LAPACK_GLOBAL(dpteqr, DPTEQR) +#define LAPACK_cpteqr LAPACK_GLOBAL(cpteqr, CPTEQR) +#define LAPACK_zpteqr LAPACK_GLOBAL(zpteqr, ZPTEQR) +#define LAPACK_sstebz LAPACK_GLOBAL(sstebz, SSTEBZ) +#define LAPACK_dstebz LAPACK_GLOBAL(dstebz, DSTEBZ) +#define LAPACK_sstein LAPACK_GLOBAL(sstein, SSTEIN) +#define LAPACK_dstein LAPACK_GLOBAL(dstein, DSTEIN) +#define LAPACK_cstein LAPACK_GLOBAL(cstein, CSTEIN) +#define LAPACK_zstein LAPACK_GLOBAL(zstein, ZSTEIN) +#define LAPACK_sdisna LAPACK_GLOBAL(sdisna, SDISNA) +#define LAPACK_ddisna LAPACK_GLOBAL(ddisna, DDISNA) +#define LAPACK_ssygst LAPACK_GLOBAL(ssygst, SSYGST) +#define LAPACK_dsygst LAPACK_GLOBAL(dsygst, DSYGST) +#define LAPACK_chegst LAPACK_GLOBAL(chegst, CHEGST) +#define LAPACK_zhegst LAPACK_GLOBAL(zhegst, ZHEGST) +#define LAPACK_sspgst LAPACK_GLOBAL(sspgst, SSPGST) +#define LAPACK_dspgst LAPACK_GLOBAL(dspgst, DSPGST) +#define LAPACK_chpgst LAPACK_GLOBAL(chpgst, CHPGST) +#define LAPACK_zhpgst LAPACK_GLOBAL(zhpgst, ZHPGST) +#define LAPACK_ssbgst LAPACK_GLOBAL(ssbgst, SSBGST) +#define LAPACK_dsbgst LAPACK_GLOBAL(dsbgst, DSBGST) +#define LAPACK_chbgst LAPACK_GLOBAL(chbgst, CHBGST) +#define LAPACK_zhbgst LAPACK_GLOBAL(zhbgst, ZHBGST) +#define LAPACK_spbstf LAPACK_GLOBAL(spbstf, SPBSTF) +#define LAPACK_dpbstf LAPACK_GLOBAL(dpbstf, DPBSTF) +#define LAPACK_cpbstf LAPACK_GLOBAL(cpbstf, CPBSTF) +#define LAPACK_zpbstf LAPACK_GLOBAL(zpbstf, ZPBSTF) +#define LAPACK_sgehrd LAPACK_GLOBAL(sgehrd, SGEHRD) +#define LAPACK_dgehrd LAPACK_GLOBAL(dgehrd, DGEHRD) +#define LAPACK_cgehrd LAPACK_GLOBAL(cgehrd, CGEHRD) +#define LAPACK_zgehrd LAPACK_GLOBAL(zgehrd, ZGEHRD) +#define LAPACK_sorghr LAPACK_GLOBAL(sorghr, SORGHR) +#define LAPACK_dorghr LAPACK_GLOBAL(dorghr, DORGHR) +#define LAPACK_sormhr LAPACK_GLOBAL(sormhr, SORMHR) +#define LAPACK_dormhr LAPACK_GLOBAL(dormhr, DORMHR) +#define LAPACK_cunghr LAPACK_GLOBAL(cunghr, CUNGHR) +#define LAPACK_zunghr LAPACK_GLOBAL(zunghr, ZUNGHR) +#define LAPACK_cunmhr LAPACK_GLOBAL(cunmhr, CUNMHR) +#define LAPACK_zunmhr LAPACK_GLOBAL(zunmhr, ZUNMHR) +#define LAPACK_sgebal LAPACK_GLOBAL(sgebal, SGEBAL) +#define LAPACK_dgebal LAPACK_GLOBAL(dgebal, DGEBAL) +#define LAPACK_cgebal LAPACK_GLOBAL(cgebal, CGEBAL) +#define LAPACK_zgebal LAPACK_GLOBAL(zgebal, ZGEBAL) +#define LAPACK_sgebak LAPACK_GLOBAL(sgebak, SGEBAK) +#define LAPACK_dgebak LAPACK_GLOBAL(dgebak, DGEBAK) +#define LAPACK_cgebak LAPACK_GLOBAL(cgebak, CGEBAK) +#define LAPACK_zgebak LAPACK_GLOBAL(zgebak, ZGEBAK) +#define LAPACK_shseqr LAPACK_GLOBAL(shseqr, SHSEQR) +#define LAPACK_dhseqr LAPACK_GLOBAL(dhseqr, DHSEQR) +#define LAPACK_chseqr LAPACK_GLOBAL(chseqr, CHSEQR) +#define LAPACK_zhseqr LAPACK_GLOBAL(zhseqr, ZHSEQR) +#define LAPACK_shsein LAPACK_GLOBAL(shsein, SHSEIN) +#define LAPACK_dhsein LAPACK_GLOBAL(dhsein, DHSEIN) +#define LAPACK_chsein LAPACK_GLOBAL(chsein, CHSEIN) +#define LAPACK_zhsein LAPACK_GLOBAL(zhsein, ZHSEIN) +#define LAPACK_strevc LAPACK_GLOBAL(strevc, STREVC) +#define LAPACK_dtrevc LAPACK_GLOBAL(dtrevc, DTREVC) +#define LAPACK_ctrevc LAPACK_GLOBAL(ctrevc, CTREVC) +#define LAPACK_ztrevc LAPACK_GLOBAL(ztrevc, ZTREVC) +#define LAPACK_strsna LAPACK_GLOBAL(strsna, STRSNA) +#define LAPACK_dtrsna LAPACK_GLOBAL(dtrsna, DTRSNA) +#define LAPACK_ctrsna LAPACK_GLOBAL(ctrsna, CTRSNA) +#define LAPACK_ztrsna LAPACK_GLOBAL(ztrsna, ZTRSNA) +#define LAPACK_strexc LAPACK_GLOBAL(strexc, STREXC) +#define LAPACK_dtrexc LAPACK_GLOBAL(dtrexc, DTREXC) +#define LAPACK_ctrexc LAPACK_GLOBAL(ctrexc, CTREXC) +#define LAPACK_ztrexc LAPACK_GLOBAL(ztrexc, ZTREXC) +#define LAPACK_strsen LAPACK_GLOBAL(strsen, STRSEN) +#define LAPACK_dtrsen LAPACK_GLOBAL(dtrsen, DTRSEN) +#define LAPACK_ctrsen LAPACK_GLOBAL(ctrsen, CTRSEN) +#define LAPACK_ztrsen LAPACK_GLOBAL(ztrsen, ZTRSEN) +#define LAPACK_strsyl LAPACK_GLOBAL(strsyl, STRSYL) +#define LAPACK_dtrsyl LAPACK_GLOBAL(dtrsyl, DTRSYL) +#define LAPACK_ctrsyl LAPACK_GLOBAL(ctrsyl, CTRSYL) +#define LAPACK_ztrsyl LAPACK_GLOBAL(ztrsyl, ZTRSYL) +#define LAPACK_sgghrd LAPACK_GLOBAL(sgghrd, SGGHRD) +#define LAPACK_dgghrd LAPACK_GLOBAL(dgghrd, DGGHRD) +#define LAPACK_cgghrd LAPACK_GLOBAL(cgghrd, CGGHRD) +#define LAPACK_zgghrd LAPACK_GLOBAL(zgghrd, ZGGHRD) +#define LAPACK_sggbal LAPACK_GLOBAL(sggbal, SGGBAL) +#define LAPACK_dggbal LAPACK_GLOBAL(dggbal, DGGBAL) +#define LAPACK_cggbal LAPACK_GLOBAL(cggbal, CGGBAL) +#define LAPACK_zggbal LAPACK_GLOBAL(zggbal, ZGGBAL) +#define LAPACK_sggbak LAPACK_GLOBAL(sggbak, SGGBAK) +#define LAPACK_dggbak LAPACK_GLOBAL(dggbak, DGGBAK) +#define LAPACK_cggbak LAPACK_GLOBAL(cggbak, CGGBAK) +#define LAPACK_zggbak LAPACK_GLOBAL(zggbak, ZGGBAK) +#define LAPACK_shgeqz LAPACK_GLOBAL(shgeqz, SHGEQZ) +#define LAPACK_dhgeqz LAPACK_GLOBAL(dhgeqz, DHGEQZ) +#define LAPACK_chgeqz LAPACK_GLOBAL(chgeqz, CHGEQZ) +#define LAPACK_zhgeqz LAPACK_GLOBAL(zhgeqz, ZHGEQZ) +#define LAPACK_stgevc LAPACK_GLOBAL(stgevc, STGEVC) +#define LAPACK_dtgevc LAPACK_GLOBAL(dtgevc, DTGEVC) +#define LAPACK_ctgevc LAPACK_GLOBAL(ctgevc, CTGEVC) +#define LAPACK_ztgevc LAPACK_GLOBAL(ztgevc, ZTGEVC) +#define LAPACK_stgexc LAPACK_GLOBAL(stgexc, STGEXC) +#define LAPACK_dtgexc LAPACK_GLOBAL(dtgexc, DTGEXC) +#define LAPACK_ctgexc LAPACK_GLOBAL(ctgexc, CTGEXC) +#define LAPACK_ztgexc LAPACK_GLOBAL(ztgexc, ZTGEXC) +#define LAPACK_stgsen LAPACK_GLOBAL(stgsen, STGSEN) +#define LAPACK_dtgsen LAPACK_GLOBAL(dtgsen, DTGSEN) +#define LAPACK_ctgsen LAPACK_GLOBAL(ctgsen, CTGSEN) +#define LAPACK_ztgsen LAPACK_GLOBAL(ztgsen, ZTGSEN) +#define LAPACK_stgsyl LAPACK_GLOBAL(stgsyl, STGSYL) +#define LAPACK_dtgsyl LAPACK_GLOBAL(dtgsyl, DTGSYL) +#define LAPACK_ctgsyl LAPACK_GLOBAL(ctgsyl, CTGSYL) +#define LAPACK_ztgsyl LAPACK_GLOBAL(ztgsyl, ZTGSYL) +#define LAPACK_stgsna LAPACK_GLOBAL(stgsna, STGSNA) +#define LAPACK_dtgsna LAPACK_GLOBAL(dtgsna, DTGSNA) +#define LAPACK_ctgsna LAPACK_GLOBAL(ctgsna, CTGSNA) +#define LAPACK_ztgsna LAPACK_GLOBAL(ztgsna, ZTGSNA) +#define LAPACK_sggsvp LAPACK_GLOBAL(sggsvp, SGGSVP) +#define LAPACK_dggsvp LAPACK_GLOBAL(dggsvp, DGGSVP) +#define LAPACK_cggsvp LAPACK_GLOBAL(cggsvp, CGGSVP) +#define LAPACK_zggsvp LAPACK_GLOBAL(zggsvp, ZGGSVP) +#define LAPACK_stgsja LAPACK_GLOBAL(stgsja, STGSJA) +#define LAPACK_dtgsja LAPACK_GLOBAL(dtgsja, DTGSJA) +#define LAPACK_ctgsja LAPACK_GLOBAL(ctgsja, CTGSJA) +#define LAPACK_ztgsja LAPACK_GLOBAL(ztgsja, ZTGSJA) +#define LAPACK_sgels LAPACK_GLOBAL(sgels, SGELS) +#define LAPACK_dgels LAPACK_GLOBAL(dgels, DGELS) +#define LAPACK_cgels LAPACK_GLOBAL(cgels, CGELS) +#define LAPACK_zgels LAPACK_GLOBAL(zgels, ZGELS) +#define LAPACK_sgelsy LAPACK_GLOBAL(sgelsy, SGELSY) +#define LAPACK_dgelsy LAPACK_GLOBAL(dgelsy, DGELSY) +#define LAPACK_cgelsy LAPACK_GLOBAL(cgelsy, CGELSY) +#define LAPACK_zgelsy LAPACK_GLOBAL(zgelsy, ZGELSY) +#define LAPACK_sgelss LAPACK_GLOBAL(sgelss, SGELSS) +#define LAPACK_dgelss LAPACK_GLOBAL(dgelss, DGELSS) +#define LAPACK_cgelss LAPACK_GLOBAL(cgelss, CGELSS) +#define LAPACK_zgelss LAPACK_GLOBAL(zgelss, ZGELSS) +#define LAPACK_sgelsd LAPACK_GLOBAL(sgelsd, SGELSD) +#define LAPACK_dgelsd LAPACK_GLOBAL(dgelsd, DGELSD) +#define LAPACK_cgelsd LAPACK_GLOBAL(cgelsd, CGELSD) +#define LAPACK_zgelsd LAPACK_GLOBAL(zgelsd, ZGELSD) +#define LAPACK_sgglse LAPACK_GLOBAL(sgglse, SGGLSE) +#define LAPACK_dgglse LAPACK_GLOBAL(dgglse, DGGLSE) +#define LAPACK_cgglse LAPACK_GLOBAL(cgglse, CGGLSE) +#define LAPACK_zgglse LAPACK_GLOBAL(zgglse, ZGGLSE) +#define LAPACK_sggglm LAPACK_GLOBAL(sggglm, SGGGLM) +#define LAPACK_dggglm LAPACK_GLOBAL(dggglm, DGGGLM) +#define LAPACK_cggglm LAPACK_GLOBAL(cggglm, CGGGLM) +#define LAPACK_zggglm LAPACK_GLOBAL(zggglm, ZGGGLM) +#define LAPACK_ssyev LAPACK_GLOBAL(ssyev, SSYEV) +#define LAPACK_dsyev LAPACK_GLOBAL(dsyev, DSYEV) +#define LAPACK_cheev LAPACK_GLOBAL(cheev, CHEEV) +#define LAPACK_zheev LAPACK_GLOBAL(zheev, ZHEEV) +#define LAPACK_ssyevd LAPACK_GLOBAL(ssyevd, SSYEVD) +#define LAPACK_dsyevd LAPACK_GLOBAL(dsyevd, DSYEVD) +#define LAPACK_cheevd LAPACK_GLOBAL(cheevd, CHEEVD) +#define LAPACK_zheevd LAPACK_GLOBAL(zheevd, ZHEEVD) +#define LAPACK_ssyevx LAPACK_GLOBAL(ssyevx, SSYEVX) +#define LAPACK_dsyevx LAPACK_GLOBAL(dsyevx, DSYEVX) +#define LAPACK_cheevx LAPACK_GLOBAL(cheevx, CHEEVX) +#define LAPACK_zheevx LAPACK_GLOBAL(zheevx, ZHEEVX) +#define LAPACK_ssyevr LAPACK_GLOBAL(ssyevr, SSYEVR) +#define LAPACK_dsyevr LAPACK_GLOBAL(dsyevr, DSYEVR) +#define LAPACK_cheevr LAPACK_GLOBAL(cheevr, CHEEVR) +#define LAPACK_zheevr LAPACK_GLOBAL(zheevr, ZHEEVR) +#define LAPACK_sspev LAPACK_GLOBAL(sspev, SSPEV) +#define LAPACK_dspev LAPACK_GLOBAL(dspev, DSPEV) +#define LAPACK_chpev LAPACK_GLOBAL(chpev, CHPEV) +#define LAPACK_zhpev LAPACK_GLOBAL(zhpev, ZHPEV) +#define LAPACK_sspevd LAPACK_GLOBAL(sspevd, SSPEVD) +#define LAPACK_dspevd LAPACK_GLOBAL(dspevd, DSPEVD) +#define LAPACK_chpevd LAPACK_GLOBAL(chpevd, CHPEVD) +#define LAPACK_zhpevd LAPACK_GLOBAL(zhpevd, ZHPEVD) +#define LAPACK_sspevx LAPACK_GLOBAL(sspevx, SSPEVX) +#define LAPACK_dspevx LAPACK_GLOBAL(dspevx, DSPEVX) +#define LAPACK_chpevx LAPACK_GLOBAL(chpevx, CHPEVX) +#define LAPACK_zhpevx LAPACK_GLOBAL(zhpevx, ZHPEVX) +#define LAPACK_ssbev LAPACK_GLOBAL(ssbev, SSBEV) +#define LAPACK_dsbev LAPACK_GLOBAL(dsbev, DSBEV) +#define LAPACK_chbev LAPACK_GLOBAL(chbev, CHBEV) +#define LAPACK_zhbev LAPACK_GLOBAL(zhbev, ZHBEV) +#define LAPACK_ssbevd LAPACK_GLOBAL(ssbevd, SSBEVD) +#define LAPACK_dsbevd LAPACK_GLOBAL(dsbevd, DSBEVD) +#define LAPACK_chbevd LAPACK_GLOBAL(chbevd, CHBEVD) +#define LAPACK_zhbevd LAPACK_GLOBAL(zhbevd, ZHBEVD) +#define LAPACK_ssbevx LAPACK_GLOBAL(ssbevx, SSBEVX) +#define LAPACK_dsbevx LAPACK_GLOBAL(dsbevx, DSBEVX) +#define LAPACK_chbevx LAPACK_GLOBAL(chbevx, CHBEVX) +#define LAPACK_zhbevx LAPACK_GLOBAL(zhbevx, ZHBEVX) +#define LAPACK_sstev LAPACK_GLOBAL(sstev, SSTEV) +#define LAPACK_dstev LAPACK_GLOBAL(dstev, DSTEV) +#define LAPACK_sstevd LAPACK_GLOBAL(sstevd, SSTEVD) +#define LAPACK_dstevd LAPACK_GLOBAL(dstevd, DSTEVD) +#define LAPACK_sstevx LAPACK_GLOBAL(sstevx, SSTEVX) +#define LAPACK_dstevx LAPACK_GLOBAL(dstevx, DSTEVX) +#define LAPACK_sstevr LAPACK_GLOBAL(sstevr, SSTEVR) +#define LAPACK_dstevr LAPACK_GLOBAL(dstevr, DSTEVR) +#define LAPACK_sgees LAPACK_GLOBAL(sgees, SGEES) +#define LAPACK_dgees LAPACK_GLOBAL(dgees, DGEES) +#define LAPACK_cgees LAPACK_GLOBAL(cgees, CGEES) +#define LAPACK_zgees LAPACK_GLOBAL(zgees, ZGEES) +#define LAPACK_sgeesx LAPACK_GLOBAL(sgeesx, SGEESX) +#define LAPACK_dgeesx LAPACK_GLOBAL(dgeesx, DGEESX) +#define LAPACK_cgeesx LAPACK_GLOBAL(cgeesx, CGEESX) +#define LAPACK_zgeesx LAPACK_GLOBAL(zgeesx, ZGEESX) +#define LAPACK_sgeev LAPACK_GLOBAL(sgeev, SGEEV) +#define LAPACK_dgeev LAPACK_GLOBAL(dgeev, DGEEV) +#define LAPACK_cgeev LAPACK_GLOBAL(cgeev, CGEEV) +#define LAPACK_zgeev LAPACK_GLOBAL(zgeev, ZGEEV) +#define LAPACK_sgeevx LAPACK_GLOBAL(sgeevx, SGEEVX) +#define LAPACK_dgeevx LAPACK_GLOBAL(dgeevx, DGEEVX) +#define LAPACK_cgeevx LAPACK_GLOBAL(cgeevx, CGEEVX) +#define LAPACK_zgeevx LAPACK_GLOBAL(zgeevx, ZGEEVX) +#define LAPACK_sgesvd LAPACK_GLOBAL(sgesvd, SGESVD) +#define LAPACK_dgesvd LAPACK_GLOBAL(dgesvd, DGESVD) +#define LAPACK_cgesvd LAPACK_GLOBAL(cgesvd, CGESVD) +#define LAPACK_zgesvd LAPACK_GLOBAL(zgesvd, ZGESVD) +#define LAPACK_sgesdd LAPACK_GLOBAL(sgesdd, SGESDD) +#define LAPACK_dgesdd LAPACK_GLOBAL(dgesdd, DGESDD) +#define LAPACK_cgesdd LAPACK_GLOBAL(cgesdd, CGESDD) +#define LAPACK_zgesdd LAPACK_GLOBAL(zgesdd, ZGESDD) +#define LAPACK_dgejsv LAPACK_GLOBAL(dgejsv, DGEJSV) +#define LAPACK_sgejsv LAPACK_GLOBAL(sgejsv, SGEJSV) +#define LAPACK_dgesvj LAPACK_GLOBAL(dgesvj, DGESVJ) +#define LAPACK_sgesvj LAPACK_GLOBAL(sgesvj, SGESVJ) +#define LAPACK_sggsvd LAPACK_GLOBAL(sggsvd, SGGSVD) +#define LAPACK_dggsvd LAPACK_GLOBAL(dggsvd, DGGSVD) +#define LAPACK_cggsvd LAPACK_GLOBAL(cggsvd, CGGSVD) +#define LAPACK_zggsvd LAPACK_GLOBAL(zggsvd, ZGGSVD) +#define LAPACK_ssygv LAPACK_GLOBAL(ssygv, SSYGV) +#define LAPACK_dsygv LAPACK_GLOBAL(dsygv, DSYGV) +#define LAPACK_chegv LAPACK_GLOBAL(chegv, CHEGV) +#define LAPACK_zhegv LAPACK_GLOBAL(zhegv, ZHEGV) +#define LAPACK_ssygvd LAPACK_GLOBAL(ssygvd, SSYGVD) +#define LAPACK_dsygvd LAPACK_GLOBAL(dsygvd, DSYGVD) +#define LAPACK_chegvd LAPACK_GLOBAL(chegvd, CHEGVD) +#define LAPACK_zhegvd LAPACK_GLOBAL(zhegvd, ZHEGVD) +#define LAPACK_ssygvx LAPACK_GLOBAL(ssygvx, SSYGVX) +#define LAPACK_dsygvx LAPACK_GLOBAL(dsygvx, DSYGVX) +#define LAPACK_chegvx LAPACK_GLOBAL(chegvx, CHEGVX) +#define LAPACK_zhegvx LAPACK_GLOBAL(zhegvx, ZHEGVX) +#define LAPACK_sspgv LAPACK_GLOBAL(sspgv, SSPGV) +#define LAPACK_dspgv LAPACK_GLOBAL(dspgv, DSPGV) +#define LAPACK_chpgv LAPACK_GLOBAL(chpgv, CHPGV) +#define LAPACK_zhpgv LAPACK_GLOBAL(zhpgv, ZHPGV) +#define LAPACK_sspgvd LAPACK_GLOBAL(sspgvd, SSPGVD) +#define LAPACK_dspgvd LAPACK_GLOBAL(dspgvd, DSPGVD) +#define LAPACK_chpgvd LAPACK_GLOBAL(chpgvd, CHPGVD) +#define LAPACK_zhpgvd LAPACK_GLOBAL(zhpgvd, ZHPGVD) +#define LAPACK_sspgvx LAPACK_GLOBAL(sspgvx, SSPGVX) +#define LAPACK_dspgvx LAPACK_GLOBAL(dspgvx, DSPGVX) +#define LAPACK_chpgvx LAPACK_GLOBAL(chpgvx, CHPGVX) +#define LAPACK_zhpgvx LAPACK_GLOBAL(zhpgvx, ZHPGVX) +#define LAPACK_ssbgv LAPACK_GLOBAL(ssbgv, SSBGV) +#define LAPACK_dsbgv LAPACK_GLOBAL(dsbgv, DSBGV) +#define LAPACK_chbgv LAPACK_GLOBAL(chbgv, CHBGV) +#define LAPACK_zhbgv LAPACK_GLOBAL(zhbgv, ZHBGV) +#define LAPACK_ssbgvd LAPACK_GLOBAL(ssbgvd, SSBGVD) +#define LAPACK_dsbgvd LAPACK_GLOBAL(dsbgvd, DSBGVD) +#define LAPACK_chbgvd LAPACK_GLOBAL(chbgvd, CHBGVD) +#define LAPACK_zhbgvd LAPACK_GLOBAL(zhbgvd, ZHBGVD) +#define LAPACK_ssbgvx LAPACK_GLOBAL(ssbgvx, SSBGVX) +#define LAPACK_dsbgvx LAPACK_GLOBAL(dsbgvx, DSBGVX) +#define LAPACK_chbgvx LAPACK_GLOBAL(chbgvx, CHBGVX) +#define LAPACK_zhbgvx LAPACK_GLOBAL(zhbgvx, ZHBGVX) +#define LAPACK_sgges LAPACK_GLOBAL(sgges, SGGES) +#define LAPACK_dgges LAPACK_GLOBAL(dgges, DGGES) +#define LAPACK_cgges LAPACK_GLOBAL(cgges, CGGES) +#define LAPACK_zgges LAPACK_GLOBAL(zgges, ZGGES) +#define LAPACK_sggesx LAPACK_GLOBAL(sggesx, SGGESX) +#define LAPACK_dggesx LAPACK_GLOBAL(dggesx, DGGESX) +#define LAPACK_cggesx LAPACK_GLOBAL(cggesx, CGGESX) +#define LAPACK_zggesx LAPACK_GLOBAL(zggesx, ZGGESX) +#define LAPACK_sggev LAPACK_GLOBAL(sggev, SGGEV) +#define LAPACK_dggev LAPACK_GLOBAL(dggev, DGGEV) +#define LAPACK_cggev LAPACK_GLOBAL(cggev, CGGEV) +#define LAPACK_zggev LAPACK_GLOBAL(zggev, ZGGEV) +#define LAPACK_sggevx LAPACK_GLOBAL(sggevx, SGGEVX) +#define LAPACK_dggevx LAPACK_GLOBAL(dggevx, DGGEVX) +#define LAPACK_cggevx LAPACK_GLOBAL(cggevx, CGGEVX) +#define LAPACK_zggevx LAPACK_GLOBAL(zggevx, ZGGEVX) +#define LAPACK_dsfrk LAPACK_GLOBAL(dsfrk, DSFRK) +#define LAPACK_ssfrk LAPACK_GLOBAL(ssfrk, SSFRK) +#define LAPACK_zhfrk LAPACK_GLOBAL(zhfrk, ZHFRK) +#define LAPACK_chfrk LAPACK_GLOBAL(chfrk, CHFRK) +#define LAPACK_dtfsm LAPACK_GLOBAL(dtfsm, DTFSM) +#define LAPACK_stfsm LAPACK_GLOBAL(stfsm, STFSM) +#define LAPACK_ztfsm LAPACK_GLOBAL(ztfsm, ZTFSM) +#define LAPACK_ctfsm LAPACK_GLOBAL(ctfsm, CTFSM) +#define LAPACK_dtfttp LAPACK_GLOBAL(dtfttp, DTFTTP) +#define LAPACK_stfttp LAPACK_GLOBAL(stfttp, STFTTP) +#define LAPACK_ztfttp LAPACK_GLOBAL(ztfttp, ZTFTTP) +#define LAPACK_ctfttp LAPACK_GLOBAL(ctfttp, CTFTTP) +#define LAPACK_dtfttr LAPACK_GLOBAL(dtfttr, DTFTTR) +#define LAPACK_stfttr LAPACK_GLOBAL(stfttr, STFTTR) +#define LAPACK_ztfttr LAPACK_GLOBAL(ztfttr, ZTFTTR) +#define LAPACK_ctfttr LAPACK_GLOBAL(ctfttr, CTFTTR) +#define LAPACK_dtpttf LAPACK_GLOBAL(dtpttf, DTPTTF) +#define LAPACK_stpttf LAPACK_GLOBAL(stpttf, STPTTF) +#define LAPACK_ztpttf LAPACK_GLOBAL(ztpttf, ZTPTTF) +#define LAPACK_ctpttf LAPACK_GLOBAL(ctpttf, CTPTTF) +#define LAPACK_dtpttr LAPACK_GLOBAL(dtpttr, DTPTTR) +#define LAPACK_stpttr LAPACK_GLOBAL(stpttr, STPTTR) +#define LAPACK_ztpttr LAPACK_GLOBAL(ztpttr, ZTPTTR) +#define LAPACK_ctpttr LAPACK_GLOBAL(ctpttr, CTPTTR) +#define LAPACK_dtrttf LAPACK_GLOBAL(dtrttf, DTRTTF) +#define LAPACK_strttf LAPACK_GLOBAL(strttf, STRTTF) +#define LAPACK_ztrttf LAPACK_GLOBAL(ztrttf, ZTRTTF) +#define LAPACK_ctrttf LAPACK_GLOBAL(ctrttf, CTRTTF) +#define LAPACK_dtrttp LAPACK_GLOBAL(dtrttp, DTRTTP) +#define LAPACK_strttp LAPACK_GLOBAL(strttp, STRTTP) +#define LAPACK_ztrttp LAPACK_GLOBAL(ztrttp, ZTRTTP) +#define LAPACK_ctrttp LAPACK_GLOBAL(ctrttp, CTRTTP) +#define LAPACK_sgeqrfp LAPACK_GLOBAL(sgeqrfp, SGEQRFP) +#define LAPACK_dgeqrfp LAPACK_GLOBAL(dgeqrfp, DGEQRFP) +#define LAPACK_cgeqrfp LAPACK_GLOBAL(cgeqrfp, CGEQRFP) +#define LAPACK_zgeqrfp LAPACK_GLOBAL(zgeqrfp, ZGEQRFP) +#define LAPACK_clacgv LAPACK_GLOBAL(clacgv, CLACGV) +#define LAPACK_zlacgv LAPACK_GLOBAL(zlacgv, ZLACGV) +#define LAPACK_slarnv LAPACK_GLOBAL(slarnv, SLARNV) +#define LAPACK_dlarnv LAPACK_GLOBAL(dlarnv, DLARNV) +#define LAPACK_clarnv LAPACK_GLOBAL(clarnv, CLARNV) +#define LAPACK_zlarnv LAPACK_GLOBAL(zlarnv, ZLARNV) +#define LAPACK_sgeqr2 LAPACK_GLOBAL(sgeqr2, SGEQR2) +#define LAPACK_dgeqr2 LAPACK_GLOBAL(dgeqr2, DGEQR2) +#define LAPACK_cgeqr2 LAPACK_GLOBAL(cgeqr2, CGEQR2) +#define LAPACK_zgeqr2 LAPACK_GLOBAL(zgeqr2, ZGEQR2) +#define LAPACK_slacpy LAPACK_GLOBAL(slacpy, SLACPY) +#define LAPACK_dlacpy LAPACK_GLOBAL(dlacpy, DLACPY) +#define LAPACK_clacpy LAPACK_GLOBAL(clacpy, CLACPY) +#define LAPACK_zlacpy LAPACK_GLOBAL(zlacpy, ZLACPY) +#define LAPACK_sgetf2 LAPACK_GLOBAL(sgetf2, SGETF2) +#define LAPACK_dgetf2 LAPACK_GLOBAL(dgetf2, DGETF2) +#define LAPACK_cgetf2 LAPACK_GLOBAL(cgetf2, CGETF2) +#define LAPACK_zgetf2 LAPACK_GLOBAL(zgetf2, ZGETF2) +#define LAPACK_slaswp LAPACK_GLOBAL(slaswp, SLASWP) +#define LAPACK_dlaswp LAPACK_GLOBAL(dlaswp, DLASWP) +#define LAPACK_claswp LAPACK_GLOBAL(claswp, CLASWP) +#define LAPACK_zlaswp LAPACK_GLOBAL(zlaswp, ZLASWP) +#define LAPACK_slange LAPACK_GLOBAL(slange, SLANGE) +#define LAPACK_dlange LAPACK_GLOBAL(dlange, DLANGE) +#define LAPACK_clange LAPACK_GLOBAL(clange, CLANGE) +#define LAPACK_zlange LAPACK_GLOBAL(zlange, ZLANGE) +#define LAPACK_clanhe LAPACK_GLOBAL(clanhe, CLANHE) +#define LAPACK_zlanhe LAPACK_GLOBAL(zlanhe, ZLANHE) +#define LAPACK_slansy LAPACK_GLOBAL(slansy, SLANSY) +#define LAPACK_dlansy LAPACK_GLOBAL(dlansy, DLANSY) +#define LAPACK_clansy LAPACK_GLOBAL(clansy, CLANSY) +#define LAPACK_zlansy LAPACK_GLOBAL(zlansy, ZLANSY) +#define LAPACK_slantr LAPACK_GLOBAL(slantr, SLANTR) +#define LAPACK_dlantr LAPACK_GLOBAL(dlantr, DLANTR) +#define LAPACK_clantr LAPACK_GLOBAL(clantr, CLANTR) +#define LAPACK_zlantr LAPACK_GLOBAL(zlantr, ZLANTR) +#define LAPACK_slamch LAPACK_GLOBAL(slamch, SLAMCH) +#define LAPACK_dlamch LAPACK_GLOBAL(dlamch, DLAMCH) +#define LAPACK_sgelq2 LAPACK_GLOBAL(sgelq2, SGELQ2) +#define LAPACK_dgelq2 LAPACK_GLOBAL(dgelq2, DGELQ2) +#define LAPACK_cgelq2 LAPACK_GLOBAL(cgelq2, CGELQ2) +#define LAPACK_zgelq2 LAPACK_GLOBAL(zgelq2, ZGELQ2) +#define LAPACK_slarfb LAPACK_GLOBAL(slarfb, SLARFB) +#define LAPACK_dlarfb LAPACK_GLOBAL(dlarfb, DLARFB) +#define LAPACK_clarfb LAPACK_GLOBAL(clarfb, CLARFB) +#define LAPACK_zlarfb LAPACK_GLOBAL(zlarfb, ZLARFB) +#define LAPACK_slarfg LAPACK_GLOBAL(slarfg, SLARFG) +#define LAPACK_dlarfg LAPACK_GLOBAL(dlarfg, DLARFG) +#define LAPACK_clarfg LAPACK_GLOBAL(clarfg, CLARFG) +#define LAPACK_zlarfg LAPACK_GLOBAL(zlarfg, ZLARFG) +#define LAPACK_slarft LAPACK_GLOBAL(slarft, SLARFT) +#define LAPACK_dlarft LAPACK_GLOBAL(dlarft, DLARFT) +#define LAPACK_clarft LAPACK_GLOBAL(clarft, CLARFT) +#define LAPACK_zlarft LAPACK_GLOBAL(zlarft, ZLARFT) +#define LAPACK_slarfx LAPACK_GLOBAL(slarfx, SLARFX) +#define LAPACK_dlarfx LAPACK_GLOBAL(dlarfx, DLARFX) +#define LAPACK_clarfx LAPACK_GLOBAL(clarfx, CLARFX) +#define LAPACK_zlarfx LAPACK_GLOBAL(zlarfx, ZLARFX) +#define LAPACK_slatms LAPACK_GLOBAL(slatms, SLATMS) +#define LAPACK_dlatms LAPACK_GLOBAL(dlatms, DLATMS) +#define LAPACK_clatms LAPACK_GLOBAL(clatms, CLATMS) +#define LAPACK_zlatms LAPACK_GLOBAL(zlatms, ZLATMS) +#define LAPACK_slag2d LAPACK_GLOBAL(slag2d, SLAG2D) +#define LAPACK_dlag2s LAPACK_GLOBAL(dlag2s, DLAG2S) +#define LAPACK_clag2z LAPACK_GLOBAL(clag2z, CLAG2Z) +#define LAPACK_zlag2c LAPACK_GLOBAL(zlag2c, ZLAG2C) +#define LAPACK_slauum LAPACK_GLOBAL(slauum, SLAUUM) +#define LAPACK_dlauum LAPACK_GLOBAL(dlauum, DLAUUM) +#define LAPACK_clauum LAPACK_GLOBAL(clauum, CLAUUM) +#define LAPACK_zlauum LAPACK_GLOBAL(zlauum, ZLAUUM) +#define LAPACK_slagge LAPACK_GLOBAL(slagge, SLAGGE) +#define LAPACK_dlagge LAPACK_GLOBAL(dlagge, DLAGGE) +#define LAPACK_clagge LAPACK_GLOBAL(clagge, CLAGGE) +#define LAPACK_zlagge LAPACK_GLOBAL(zlagge, ZLAGGE) +#define LAPACK_slaset LAPACK_GLOBAL(slaset, SLASET) +#define LAPACK_dlaset LAPACK_GLOBAL(dlaset, DLASET) +#define LAPACK_claset LAPACK_GLOBAL(claset, CLASET) +#define LAPACK_zlaset LAPACK_GLOBAL(zlaset, ZLASET) +#define LAPACK_slasrt LAPACK_GLOBAL(slasrt, SLASRT) +#define LAPACK_dlasrt LAPACK_GLOBAL(dlasrt, DLASRT) +#define LAPACK_slagsy LAPACK_GLOBAL(slagsy, SLAGSY) +#define LAPACK_dlagsy LAPACK_GLOBAL(dlagsy, DLAGSY) +#define LAPACK_clagsy LAPACK_GLOBAL(clagsy, CLAGSY) +#define LAPACK_zlagsy LAPACK_GLOBAL(zlagsy, ZLAGSY) +#define LAPACK_claghe LAPACK_GLOBAL(claghe, CLAGHE) +#define LAPACK_zlaghe LAPACK_GLOBAL(zlaghe, ZLAGHE) +#define LAPACK_slapmr LAPACK_GLOBAL(slapmr, SLAPMR) +#define LAPACK_dlapmr LAPACK_GLOBAL(dlapmr, DLAPMR) +#define LAPACK_clapmr LAPACK_GLOBAL(clapmr, CLAPMR) +#define LAPACK_zlapmr LAPACK_GLOBAL(zlapmr, ZLAPMR) +#define LAPACK_slapy2 LAPACK_GLOBAL(slapy2, SLAPY2) +#define LAPACK_dlapy2 LAPACK_GLOBAL(dlapy2, DLAPY2) +#define LAPACK_slapy3 LAPACK_GLOBAL(slapy3, SLAPY3) +#define LAPACK_dlapy3 LAPACK_GLOBAL(dlapy3, DLAPY3) +#define LAPACK_slartgp LAPACK_GLOBAL(slartgp, SLARTGP) +#define LAPACK_dlartgp LAPACK_GLOBAL(dlartgp, DLARTGP) +#define LAPACK_slartgs LAPACK_GLOBAL(slartgs, SLARTGS) +#define LAPACK_dlartgs LAPACK_GLOBAL(dlartgs, DLARTGS) // LAPACK 3.3.0 -#define LAPACK_cbbcsd LAPACK_GLOBAL(cbbcsd,CBBCSD) -#define LAPACK_cheswapr LAPACK_GLOBAL(cheswapr,CHESWAPR) -#define LAPACK_chetri2 LAPACK_GLOBAL(chetri2,CHETRI2) -#define LAPACK_chetri2x LAPACK_GLOBAL(chetri2x,CHETRI2X) -#define LAPACK_chetrs2 LAPACK_GLOBAL(chetrs2,CHETRS2) -#define LAPACK_csyconv LAPACK_GLOBAL(csyconv,CSYCONV) -#define LAPACK_csyswapr LAPACK_GLOBAL(csyswapr,CSYSWAPR) -#define LAPACK_csytri2 LAPACK_GLOBAL(csytri2,CSYTRI2) -#define LAPACK_csytri2x LAPACK_GLOBAL(csytri2x,CSYTRI2X) -#define LAPACK_csytrs2 LAPACK_GLOBAL(csytrs2,CSYTRS2) -#define LAPACK_cunbdb LAPACK_GLOBAL(cunbdb,CUNBDB) -#define LAPACK_cuncsd LAPACK_GLOBAL(cuncsd,CUNCSD) -#define LAPACK_dbbcsd LAPACK_GLOBAL(dbbcsd,DBBCSD) -#define LAPACK_dorbdb LAPACK_GLOBAL(dorbdb,DORBDB) -#define LAPACK_dorcsd LAPACK_GLOBAL(dorcsd,DORCSD) -#define LAPACK_dsyconv LAPACK_GLOBAL(dsyconv,DSYCONV) -#define LAPACK_dsyswapr LAPACK_GLOBAL(dsyswapr,DSYSWAPR) -#define LAPACK_dsytri2 LAPACK_GLOBAL(dsytri2,DSYTRI2) -#define LAPACK_dsytri2x LAPACK_GLOBAL(dsytri2x,DSYTRI2X) -#define LAPACK_dsytrs2 LAPACK_GLOBAL(dsytrs2,DSYTRS2) -#define LAPACK_sbbcsd LAPACK_GLOBAL(sbbcsd,SBBCSD) -#define LAPACK_sorbdb LAPACK_GLOBAL(sorbdb,SORBDB) -#define LAPACK_sorcsd LAPACK_GLOBAL(sorcsd,SORCSD) -#define LAPACK_ssyconv LAPACK_GLOBAL(ssyconv,SSYCONV) -#define LAPACK_ssyswapr LAPACK_GLOBAL(ssyswapr,SSYSWAPR) -#define LAPACK_ssytri2 LAPACK_GLOBAL(ssytri2,SSYTRI2) -#define LAPACK_ssytri2x LAPACK_GLOBAL(ssytri2x,SSYTRI2X) -#define LAPACK_ssytrs2 LAPACK_GLOBAL(ssytrs2,SSYTRS2) -#define LAPACK_zbbcsd LAPACK_GLOBAL(zbbcsd,ZBBCSD) -#define LAPACK_zheswapr LAPACK_GLOBAL(zheswapr,ZHESWAPR) -#define LAPACK_zhetri2 LAPACK_GLOBAL(zhetri2,ZHETRI2) -#define LAPACK_zhetri2x LAPACK_GLOBAL(zhetri2x,ZHETRI2X) -#define LAPACK_zhetrs2 LAPACK_GLOBAL(zhetrs2,ZHETRS2) -#define LAPACK_zsyconv LAPACK_GLOBAL(zsyconv,ZSYCONV) -#define LAPACK_zsyswapr LAPACK_GLOBAL(zsyswapr,ZSYSWAPR) -#define LAPACK_zsytri2 LAPACK_GLOBAL(zsytri2,ZSYTRI2) -#define LAPACK_zsytri2x LAPACK_GLOBAL(zsytri2x,ZSYTRI2X) -#define LAPACK_zsytrs2 LAPACK_GLOBAL(zsytrs2,ZSYTRS2) -#define LAPACK_zunbdb LAPACK_GLOBAL(zunbdb,ZUNBDB) -#define LAPACK_zuncsd LAPACK_GLOBAL(zuncsd,ZUNCSD) +#define LAPACK_cbbcsd LAPACK_GLOBAL(cbbcsd, CBBCSD) +#define LAPACK_cheswapr LAPACK_GLOBAL(cheswapr, CHESWAPR) +#define LAPACK_chetri2 LAPACK_GLOBAL(chetri2, CHETRI2) +#define LAPACK_chetri2x LAPACK_GLOBAL(chetri2x, CHETRI2X) +#define LAPACK_chetrs2 LAPACK_GLOBAL(chetrs2, CHETRS2) +#define LAPACK_csyconv LAPACK_GLOBAL(csyconv, CSYCONV) +#define LAPACK_csyswapr LAPACK_GLOBAL(csyswapr, CSYSWAPR) +#define LAPACK_csytri2 LAPACK_GLOBAL(csytri2, CSYTRI2) +#define LAPACK_csytri2x LAPACK_GLOBAL(csytri2x, CSYTRI2X) +#define LAPACK_csytrs2 LAPACK_GLOBAL(csytrs2, CSYTRS2) +#define LAPACK_cunbdb LAPACK_GLOBAL(cunbdb, CUNBDB) +#define LAPACK_cuncsd LAPACK_GLOBAL(cuncsd, CUNCSD) +#define LAPACK_dbbcsd LAPACK_GLOBAL(dbbcsd, DBBCSD) +#define LAPACK_dorbdb LAPACK_GLOBAL(dorbdb, DORBDB) +#define LAPACK_dorcsd LAPACK_GLOBAL(dorcsd, DORCSD) +#define LAPACK_dsyconv LAPACK_GLOBAL(dsyconv, DSYCONV) +#define LAPACK_dsyswapr LAPACK_GLOBAL(dsyswapr, DSYSWAPR) +#define LAPACK_dsytri2 LAPACK_GLOBAL(dsytri2, DSYTRI2) +#define LAPACK_dsytri2x LAPACK_GLOBAL(dsytri2x, DSYTRI2X) +#define LAPACK_dsytrs2 LAPACK_GLOBAL(dsytrs2, DSYTRS2) +#define LAPACK_sbbcsd LAPACK_GLOBAL(sbbcsd, SBBCSD) +#define LAPACK_sorbdb LAPACK_GLOBAL(sorbdb, SORBDB) +#define LAPACK_sorcsd LAPACK_GLOBAL(sorcsd, SORCSD) +#define LAPACK_ssyconv LAPACK_GLOBAL(ssyconv, SSYCONV) +#define LAPACK_ssyswapr LAPACK_GLOBAL(ssyswapr, SSYSWAPR) +#define LAPACK_ssytri2 LAPACK_GLOBAL(ssytri2, SSYTRI2) +#define LAPACK_ssytri2x LAPACK_GLOBAL(ssytri2x, SSYTRI2X) +#define LAPACK_ssytrs2 LAPACK_GLOBAL(ssytrs2, SSYTRS2) +#define LAPACK_zbbcsd LAPACK_GLOBAL(zbbcsd, ZBBCSD) +#define LAPACK_zheswapr LAPACK_GLOBAL(zheswapr, ZHESWAPR) +#define LAPACK_zhetri2 LAPACK_GLOBAL(zhetri2, ZHETRI2) +#define LAPACK_zhetri2x LAPACK_GLOBAL(zhetri2x, ZHETRI2X) +#define LAPACK_zhetrs2 LAPACK_GLOBAL(zhetrs2, ZHETRS2) +#define LAPACK_zsyconv LAPACK_GLOBAL(zsyconv, ZSYCONV) +#define LAPACK_zsyswapr LAPACK_GLOBAL(zsyswapr, ZSYSWAPR) +#define LAPACK_zsytri2 LAPACK_GLOBAL(zsytri2, ZSYTRI2) +#define LAPACK_zsytri2x LAPACK_GLOBAL(zsytri2x, ZSYTRI2X) +#define LAPACK_zsytrs2 LAPACK_GLOBAL(zsytrs2, ZSYTRS2) +#define LAPACK_zunbdb LAPACK_GLOBAL(zunbdb, ZUNBDB) +#define LAPACK_zuncsd LAPACK_GLOBAL(zuncsd, ZUNCSD) // LAPACK 3.4.0 -#define LAPACK_sgemqrt LAPACK_GLOBAL(sgemqrt,SGEMQRT) -#define LAPACK_dgemqrt LAPACK_GLOBAL(dgemqrt,DGEMQRT) -#define LAPACK_cgemqrt LAPACK_GLOBAL(cgemqrt,CGEMQRT) -#define LAPACK_zgemqrt LAPACK_GLOBAL(zgemqrt,ZGEMQRT) -#define LAPACK_sgeqrt LAPACK_GLOBAL(sgeqrt,SGEQRT) -#define LAPACK_dgeqrt LAPACK_GLOBAL(dgeqrt,DGEQRT) -#define LAPACK_cgeqrt LAPACK_GLOBAL(cgeqrt,CGEQRT) -#define LAPACK_zgeqrt LAPACK_GLOBAL(zgeqrt,ZGEQRT) -#define LAPACK_sgeqrt2 LAPACK_GLOBAL(sgeqrt2,SGEQRT2) -#define LAPACK_dgeqrt2 LAPACK_GLOBAL(dgeqrt2,DGEQRT2) -#define LAPACK_cgeqrt2 LAPACK_GLOBAL(cgeqrt2,CGEQRT2) -#define LAPACK_zgeqrt2 LAPACK_GLOBAL(zgeqrt2,ZGEQRT2) -#define LAPACK_sgeqrt3 LAPACK_GLOBAL(sgeqrt3,SGEQRT3) -#define LAPACK_dgeqrt3 LAPACK_GLOBAL(dgeqrt3,DGEQRT3) -#define LAPACK_cgeqrt3 LAPACK_GLOBAL(cgeqrt3,CGEQRT3) -#define LAPACK_zgeqrt3 LAPACK_GLOBAL(zgeqrt3,ZGEQRT3) -#define LAPACK_stpmqrt LAPACK_GLOBAL(stpmqrt,STPMQRT) -#define LAPACK_dtpmqrt LAPACK_GLOBAL(dtpmqrt,DTPMQRT) -#define LAPACK_ctpmqrt LAPACK_GLOBAL(ctpmqrt,CTPMQRT) -#define LAPACK_ztpmqrt LAPACK_GLOBAL(ztpmqrt,ZTPMQRT) -#define LAPACK_dtpqrt LAPACK_GLOBAL(dtpqrt,DTPQRT) -#define LAPACK_ctpqrt LAPACK_GLOBAL(ctpqrt,CTPQRT) -#define LAPACK_ztpqrt LAPACK_GLOBAL(ztpqrt,ZTPQRT) -#define LAPACK_stpqrt2 LAPACK_GLOBAL(stpqrt2,STPQRT2) -#define LAPACK_dtpqrt2 LAPACK_GLOBAL(dtpqrt2,DTPQRT2) -#define LAPACK_ctpqrt2 LAPACK_GLOBAL(ctpqrt2,CTPQRT2) -#define LAPACK_ztpqrt2 LAPACK_GLOBAL(ztpqrt2,ZTPQRT2) -#define LAPACK_stprfb LAPACK_GLOBAL(stprfb,STPRFB) -#define LAPACK_dtprfb LAPACK_GLOBAL(dtprfb,DTPRFB) -#define LAPACK_ctprfb LAPACK_GLOBAL(ctprfb,CTPRFB) -#define LAPACK_ztprfb LAPACK_GLOBAL(ztprfb,ZTPRFB) +#define LAPACK_sgemqrt LAPACK_GLOBAL(sgemqrt, SGEMQRT) +#define LAPACK_dgemqrt LAPACK_GLOBAL(dgemqrt, DGEMQRT) +#define LAPACK_cgemqrt LAPACK_GLOBAL(cgemqrt, CGEMQRT) +#define LAPACK_zgemqrt LAPACK_GLOBAL(zgemqrt, ZGEMQRT) +#define LAPACK_sgeqrt LAPACK_GLOBAL(sgeqrt, SGEQRT) +#define LAPACK_dgeqrt LAPACK_GLOBAL(dgeqrt, DGEQRT) +#define LAPACK_cgeqrt LAPACK_GLOBAL(cgeqrt, CGEQRT) +#define LAPACK_zgeqrt LAPACK_GLOBAL(zgeqrt, ZGEQRT) +#define LAPACK_sgeqrt2 LAPACK_GLOBAL(sgeqrt2, SGEQRT2) +#define LAPACK_dgeqrt2 LAPACK_GLOBAL(dgeqrt2, DGEQRT2) +#define LAPACK_cgeqrt2 LAPACK_GLOBAL(cgeqrt2, CGEQRT2) +#define LAPACK_zgeqrt2 LAPACK_GLOBAL(zgeqrt2, ZGEQRT2) +#define LAPACK_sgeqrt3 LAPACK_GLOBAL(sgeqrt3, SGEQRT3) +#define LAPACK_dgeqrt3 LAPACK_GLOBAL(dgeqrt3, DGEQRT3) +#define LAPACK_cgeqrt3 LAPACK_GLOBAL(cgeqrt3, CGEQRT3) +#define LAPACK_zgeqrt3 LAPACK_GLOBAL(zgeqrt3, ZGEQRT3) +#define LAPACK_stpmqrt LAPACK_GLOBAL(stpmqrt, STPMQRT) +#define LAPACK_dtpmqrt LAPACK_GLOBAL(dtpmqrt, DTPMQRT) +#define LAPACK_ctpmqrt LAPACK_GLOBAL(ctpmqrt, CTPMQRT) +#define LAPACK_ztpmqrt LAPACK_GLOBAL(ztpmqrt, ZTPMQRT) +#define LAPACK_dtpqrt LAPACK_GLOBAL(dtpqrt, DTPQRT) +#define LAPACK_ctpqrt LAPACK_GLOBAL(ctpqrt, CTPQRT) +#define LAPACK_ztpqrt LAPACK_GLOBAL(ztpqrt, ZTPQRT) +#define LAPACK_stpqrt2 LAPACK_GLOBAL(stpqrt2, STPQRT2) +#define LAPACK_dtpqrt2 LAPACK_GLOBAL(dtpqrt2, DTPQRT2) +#define LAPACK_ctpqrt2 LAPACK_GLOBAL(ctpqrt2, CTPQRT2) +#define LAPACK_ztpqrt2 LAPACK_GLOBAL(ztpqrt2, ZTPQRT2) +#define LAPACK_stprfb LAPACK_GLOBAL(stprfb, STPRFB) +#define LAPACK_dtprfb LAPACK_GLOBAL(dtprfb, DTPRFB) +#define LAPACK_ctprfb LAPACK_GLOBAL(ctprfb, CTPRFB) +#define LAPACK_ztprfb LAPACK_GLOBAL(ztprfb, ZTPRFB) // LAPACK 3.X.X -#define LAPACK_csyr LAPACK_GLOBAL(csyr,CSYR) -#define LAPACK_zsyr LAPACK_GLOBAL(zsyr,ZSYR) - - -void LAPACK_sgetrf( lapack_int* m, lapack_int* n, float* a, lapack_int* lda, - lapack_int* ipiv, lapack_int *info ); -void LAPACK_dgetrf( lapack_int* m, lapack_int* n, double* a, lapack_int* lda, - lapack_int* ipiv, lapack_int *info ); -void LAPACK_cgetrf( lapack_int* m, lapack_int* n, lapack_complex_float* a, - lapack_int* lda, lapack_int* ipiv, lapack_int *info ); -void LAPACK_zgetrf( lapack_int* m, lapack_int* n, lapack_complex_double* a, - lapack_int* lda, lapack_int* ipiv, lapack_int *info ); -void LAPACK_sgbtrf( lapack_int* m, lapack_int* n, lapack_int* kl, - lapack_int* ku, float* ab, lapack_int* ldab, - lapack_int* ipiv, lapack_int *info ); -void LAPACK_dgbtrf( lapack_int* m, lapack_int* n, lapack_int* kl, - lapack_int* ku, double* ab, lapack_int* ldab, - lapack_int* ipiv, lapack_int *info ); -void LAPACK_cgbtrf( lapack_int* m, lapack_int* n, lapack_int* kl, - lapack_int* ku, lapack_complex_float* ab, lapack_int* ldab, - lapack_int* ipiv, lapack_int *info ); -void LAPACK_zgbtrf( lapack_int* m, lapack_int* n, lapack_int* kl, - lapack_int* ku, lapack_complex_double* ab, lapack_int* ldab, - lapack_int* ipiv, lapack_int *info ); -void LAPACK_sgttrf( lapack_int* n, float* dl, float* d, float* du, float* du2, - lapack_int* ipiv, lapack_int *info ); -void LAPACK_dgttrf( lapack_int* n, double* dl, double* d, double* du, - double* du2, lapack_int* ipiv, lapack_int *info ); -void LAPACK_cgttrf( lapack_int* n, lapack_complex_float* dl, - lapack_complex_float* d, lapack_complex_float* du, - lapack_complex_float* du2, lapack_int* ipiv, - lapack_int *info ); -void LAPACK_zgttrf( lapack_int* n, lapack_complex_double* dl, - lapack_complex_double* d, lapack_complex_double* du, - lapack_complex_double* du2, lapack_int* ipiv, - lapack_int *info ); -void LAPACK_spotrf( char* uplo, lapack_int* n, float* a, lapack_int* lda, - lapack_int *info ); -void LAPACK_dpotrf( char* uplo, lapack_int* n, double* a, lapack_int* lda, - lapack_int *info ); -void LAPACK_cpotrf( char* uplo, lapack_int* n, lapack_complex_float* a, - lapack_int* lda, lapack_int *info ); -void LAPACK_zpotrf( char* uplo, lapack_int* n, lapack_complex_double* a, - lapack_int* lda, lapack_int *info ); -void LAPACK_dpstrf( char* uplo, lapack_int* n, double* a, lapack_int* lda, - lapack_int* piv, lapack_int* rank, double* tol, - double* work, lapack_int *info ); -void LAPACK_spstrf( char* uplo, lapack_int* n, float* a, lapack_int* lda, - lapack_int* piv, lapack_int* rank, float* tol, float* work, - lapack_int *info ); -void LAPACK_zpstrf( char* uplo, lapack_int* n, lapack_complex_double* a, - lapack_int* lda, lapack_int* piv, lapack_int* rank, - double* tol, double* work, lapack_int *info ); -void LAPACK_cpstrf( char* uplo, lapack_int* n, lapack_complex_float* a, - lapack_int* lda, lapack_int* piv, lapack_int* rank, - float* tol, float* work, lapack_int *info ); -void LAPACK_dpftrf( char* transr, char* uplo, lapack_int* n, double* a, - lapack_int *info ); -void LAPACK_spftrf( char* transr, char* uplo, lapack_int* n, float* a, - lapack_int *info ); -void LAPACK_zpftrf( char* transr, char* uplo, lapack_int* n, - lapack_complex_double* a, lapack_int *info ); -void LAPACK_cpftrf( char* transr, char* uplo, lapack_int* n, - lapack_complex_float* a, lapack_int *info ); -void LAPACK_spptrf( char* uplo, lapack_int* n, float* ap, lapack_int *info ); -void LAPACK_dpptrf( char* uplo, lapack_int* n, double* ap, lapack_int *info ); -void LAPACK_cpptrf( char* uplo, lapack_int* n, lapack_complex_float* ap, - lapack_int *info ); -void LAPACK_zpptrf( char* uplo, lapack_int* n, lapack_complex_double* ap, - lapack_int *info ); -void LAPACK_spbtrf( char* uplo, lapack_int* n, lapack_int* kd, float* ab, - lapack_int* ldab, lapack_int *info ); -void LAPACK_dpbtrf( char* uplo, lapack_int* n, lapack_int* kd, double* ab, - lapack_int* ldab, lapack_int *info ); -void LAPACK_cpbtrf( char* uplo, lapack_int* n, lapack_int* kd, - lapack_complex_float* ab, lapack_int* ldab, - lapack_int *info ); -void LAPACK_zpbtrf( char* uplo, lapack_int* n, lapack_int* kd, - lapack_complex_double* ab, lapack_int* ldab, - lapack_int *info ); -void LAPACK_spttrf( lapack_int* n, float* d, float* e, lapack_int *info ); -void LAPACK_dpttrf( lapack_int* n, double* d, double* e, lapack_int *info ); -void LAPACK_cpttrf( lapack_int* n, float* d, lapack_complex_float* e, - lapack_int *info ); -void LAPACK_zpttrf( lapack_int* n, double* d, lapack_complex_double* e, - lapack_int *info ); -void LAPACK_ssytrf( char* uplo, lapack_int* n, float* a, lapack_int* lda, - lapack_int* ipiv, float* work, lapack_int* lwork, - lapack_int *info ); -void LAPACK_dsytrf( char* uplo, lapack_int* n, double* a, lapack_int* lda, - lapack_int* ipiv, double* work, lapack_int* lwork, - lapack_int *info ); -void LAPACK_csytrf( char* uplo, lapack_int* n, lapack_complex_float* a, - lapack_int* lda, lapack_int* ipiv, - lapack_complex_float* work, lapack_int* lwork, - lapack_int *info ); -void LAPACK_zsytrf( char* uplo, lapack_int* n, lapack_complex_double* a, - lapack_int* lda, lapack_int* ipiv, - lapack_complex_double* work, lapack_int* lwork, - lapack_int *info ); -void LAPACK_chetrf( char* uplo, lapack_int* n, lapack_complex_float* a, - lapack_int* lda, lapack_int* ipiv, - lapack_complex_float* work, lapack_int* lwork, - lapack_int *info ); -void LAPACK_zhetrf( char* uplo, lapack_int* n, lapack_complex_double* a, - lapack_int* lda, lapack_int* ipiv, - lapack_complex_double* work, lapack_int* lwork, - lapack_int *info ); -void LAPACK_ssptrf( char* uplo, lapack_int* n, float* ap, lapack_int* ipiv, - lapack_int *info ); -void LAPACK_dsptrf( char* uplo, lapack_int* n, double* ap, lapack_int* ipiv, - lapack_int *info ); -void LAPACK_csptrf( char* uplo, lapack_int* n, lapack_complex_float* ap, - lapack_int* ipiv, lapack_int *info ); -void LAPACK_zsptrf( char* uplo, lapack_int* n, lapack_complex_double* ap, - lapack_int* ipiv, lapack_int *info ); -void LAPACK_chptrf( char* uplo, lapack_int* n, lapack_complex_float* ap, - lapack_int* ipiv, lapack_int *info ); -void LAPACK_zhptrf( char* uplo, lapack_int* n, lapack_complex_double* ap, - lapack_int* ipiv, lapack_int *info ); -void LAPACK_sgetrs( char* trans, lapack_int* n, lapack_int* nrhs, - const float* a, lapack_int* lda, const lapack_int* ipiv, - float* b, lapack_int* ldb, lapack_int *info ); -void LAPACK_dgetrs( char* trans, lapack_int* n, lapack_int* nrhs, - const double* a, lapack_int* lda, const lapack_int* ipiv, - double* b, lapack_int* ldb, lapack_int *info ); -void LAPACK_cgetrs( char* trans, lapack_int* n, lapack_int* nrhs, - const lapack_complex_float* a, lapack_int* lda, - const lapack_int* ipiv, lapack_complex_float* b, - lapack_int* ldb, lapack_int *info ); -void LAPACK_zgetrs( char* trans, lapack_int* n, lapack_int* nrhs, - const lapack_complex_double* a, lapack_int* lda, - const lapack_int* ipiv, lapack_complex_double* b, - lapack_int* ldb, lapack_int *info ); -void LAPACK_sgbtrs( char* trans, lapack_int* n, lapack_int* kl, lapack_int* ku, - lapack_int* nrhs, const float* ab, lapack_int* ldab, - const lapack_int* ipiv, float* b, lapack_int* ldb, - lapack_int *info ); -void LAPACK_dgbtrs( char* trans, lapack_int* n, lapack_int* kl, lapack_int* ku, - lapack_int* nrhs, const double* ab, lapack_int* ldab, - const lapack_int* ipiv, double* b, lapack_int* ldb, - lapack_int *info ); -void LAPACK_cgbtrs( char* trans, lapack_int* n, lapack_int* kl, lapack_int* ku, - lapack_int* nrhs, const lapack_complex_float* ab, - lapack_int* ldab, const lapack_int* ipiv, - lapack_complex_float* b, lapack_int* ldb, - lapack_int *info ); -void LAPACK_zgbtrs( char* trans, lapack_int* n, lapack_int* kl, lapack_int* ku, - lapack_int* nrhs, const lapack_complex_double* ab, - lapack_int* ldab, const lapack_int* ipiv, - lapack_complex_double* b, lapack_int* ldb, - lapack_int *info ); -void LAPACK_sgttrs( char* trans, lapack_int* n, lapack_int* nrhs, - const float* dl, const float* d, const float* du, - const float* du2, const lapack_int* ipiv, float* b, - lapack_int* ldb, lapack_int *info ); -void LAPACK_dgttrs( char* trans, lapack_int* n, lapack_int* nrhs, - const double* dl, const double* d, const double* du, - const double* du2, const lapack_int* ipiv, double* b, - lapack_int* ldb, lapack_int *info ); -void LAPACK_cgttrs( char* trans, lapack_int* n, lapack_int* nrhs, - const lapack_complex_float* dl, - const lapack_complex_float* d, - const lapack_complex_float* du, - const lapack_complex_float* du2, const lapack_int* ipiv, - lapack_complex_float* b, lapack_int* ldb, - lapack_int *info ); -void LAPACK_zgttrs( char* trans, lapack_int* n, lapack_int* nrhs, - const lapack_complex_double* dl, - const lapack_complex_double* d, - const lapack_complex_double* du, - const lapack_complex_double* du2, const lapack_int* ipiv, - lapack_complex_double* b, lapack_int* ldb, - lapack_int *info ); -void LAPACK_spotrs( char* uplo, lapack_int* n, lapack_int* nrhs, const float* a, - lapack_int* lda, float* b, lapack_int* ldb, - lapack_int *info ); -void LAPACK_dpotrs( char* uplo, lapack_int* n, lapack_int* nrhs, - const double* a, lapack_int* lda, double* b, - lapack_int* ldb, lapack_int *info ); -void LAPACK_cpotrs( char* uplo, lapack_int* n, lapack_int* nrhs, - const lapack_complex_float* a, lapack_int* lda, - lapack_complex_float* b, lapack_int* ldb, - lapack_int *info ); -void LAPACK_zpotrs( char* uplo, lapack_int* n, lapack_int* nrhs, - const lapack_complex_double* a, lapack_int* lda, - lapack_complex_double* b, lapack_int* ldb, - lapack_int *info ); -void LAPACK_dpftrs( char* transr, char* uplo, lapack_int* n, lapack_int* nrhs, - const double* a, double* b, lapack_int* ldb, - lapack_int *info ); -void LAPACK_spftrs( char* transr, char* uplo, lapack_int* n, lapack_int* nrhs, - const float* a, float* b, lapack_int* ldb, - lapack_int *info ); -void LAPACK_zpftrs( char* transr, char* uplo, lapack_int* n, lapack_int* nrhs, - const lapack_complex_double* a, lapack_complex_double* b, - lapack_int* ldb, lapack_int *info ); -void LAPACK_cpftrs( char* transr, char* uplo, lapack_int* n, lapack_int* nrhs, - const lapack_complex_float* a, lapack_complex_float* b, - lapack_int* ldb, lapack_int *info ); -void LAPACK_spptrs( char* uplo, lapack_int* n, lapack_int* nrhs, - const float* ap, float* b, lapack_int* ldb, - lapack_int *info ); -void LAPACK_dpptrs( char* uplo, lapack_int* n, lapack_int* nrhs, - const double* ap, double* b, lapack_int* ldb, - lapack_int *info ); -void LAPACK_cpptrs( char* uplo, lapack_int* n, lapack_int* nrhs, - const lapack_complex_float* ap, lapack_complex_float* b, - lapack_int* ldb, lapack_int *info ); -void LAPACK_zpptrs( char* uplo, lapack_int* n, lapack_int* nrhs, - const lapack_complex_double* ap, lapack_complex_double* b, - lapack_int* ldb, lapack_int *info ); -void LAPACK_spbtrs( char* uplo, lapack_int* n, lapack_int* kd, lapack_int* nrhs, - const float* ab, lapack_int* ldab, float* b, - lapack_int* ldb, lapack_int *info ); -void LAPACK_dpbtrs( char* uplo, lapack_int* n, lapack_int* kd, lapack_int* nrhs, - const double* ab, lapack_int* ldab, double* b, - lapack_int* ldb, lapack_int *info ); -void LAPACK_cpbtrs( char* uplo, lapack_int* n, lapack_int* kd, lapack_int* nrhs, - const lapack_complex_float* ab, lapack_int* ldab, - lapack_complex_float* b, lapack_int* ldb, - lapack_int *info ); -void LAPACK_zpbtrs( char* uplo, lapack_int* n, lapack_int* kd, lapack_int* nrhs, - const lapack_complex_double* ab, lapack_int* ldab, - lapack_complex_double* b, lapack_int* ldb, - lapack_int *info ); -void LAPACK_spttrs( lapack_int* n, lapack_int* nrhs, const float* d, - const float* e, float* b, lapack_int* ldb, - lapack_int *info ); -void LAPACK_dpttrs( lapack_int* n, lapack_int* nrhs, const double* d, - const double* e, double* b, lapack_int* ldb, - lapack_int *info ); -void LAPACK_cpttrs( char* uplo, lapack_int* n, lapack_int* nrhs, const float* d, - const lapack_complex_float* e, lapack_complex_float* b, - lapack_int* ldb, lapack_int *info ); -void LAPACK_zpttrs( char* uplo, lapack_int* n, lapack_int* nrhs, - const double* d, const lapack_complex_double* e, - lapack_complex_double* b, lapack_int* ldb, - lapack_int *info ); -void LAPACK_ssytrs( char* uplo, lapack_int* n, lapack_int* nrhs, const float* a, - lapack_int* lda, const lapack_int* ipiv, float* b, - lapack_int* ldb, lapack_int *info ); -void LAPACK_dsytrs( char* uplo, lapack_int* n, lapack_int* nrhs, - const double* a, lapack_int* lda, const lapack_int* ipiv, - double* b, lapack_int* ldb, lapack_int *info ); -void LAPACK_csytrs( char* uplo, lapack_int* n, lapack_int* nrhs, - const lapack_complex_float* a, lapack_int* lda, - const lapack_int* ipiv, lapack_complex_float* b, - lapack_int* ldb, lapack_int *info ); -void LAPACK_zsytrs( char* uplo, lapack_int* n, lapack_int* nrhs, - const lapack_complex_double* a, lapack_int* lda, - const lapack_int* ipiv, lapack_complex_double* b, - lapack_int* ldb, lapack_int *info ); -void LAPACK_chetrs( char* uplo, lapack_int* n, lapack_int* nrhs, - const lapack_complex_float* a, lapack_int* lda, - const lapack_int* ipiv, lapack_complex_float* b, - lapack_int* ldb, lapack_int *info ); -void LAPACK_zhetrs( char* uplo, lapack_int* n, lapack_int* nrhs, - const lapack_complex_double* a, lapack_int* lda, - const lapack_int* ipiv, lapack_complex_double* b, - lapack_int* ldb, lapack_int *info ); -void LAPACK_ssptrs( char* uplo, lapack_int* n, lapack_int* nrhs, - const float* ap, const lapack_int* ipiv, float* b, - lapack_int* ldb, lapack_int *info ); -void LAPACK_dsptrs( char* uplo, lapack_int* n, lapack_int* nrhs, - const double* ap, const lapack_int* ipiv, double* b, - lapack_int* ldb, lapack_int *info ); -void LAPACK_csptrs( char* uplo, lapack_int* n, lapack_int* nrhs, - const lapack_complex_float* ap, const lapack_int* ipiv, - lapack_complex_float* b, lapack_int* ldb, - lapack_int *info ); -void LAPACK_zsptrs( char* uplo, lapack_int* n, lapack_int* nrhs, - const lapack_complex_double* ap, const lapack_int* ipiv, - lapack_complex_double* b, lapack_int* ldb, - lapack_int *info ); -void LAPACK_chptrs( char* uplo, lapack_int* n, lapack_int* nrhs, - const lapack_complex_float* ap, const lapack_int* ipiv, - lapack_complex_float* b, lapack_int* ldb, - lapack_int *info ); -void LAPACK_zhptrs( char* uplo, lapack_int* n, lapack_int* nrhs, - const lapack_complex_double* ap, const lapack_int* ipiv, - lapack_complex_double* b, lapack_int* ldb, - lapack_int *info ); -void LAPACK_strtrs( char* uplo, char* trans, char* diag, lapack_int* n, - lapack_int* nrhs, const float* a, lapack_int* lda, float* b, - lapack_int* ldb, lapack_int *info ); -void LAPACK_dtrtrs( char* uplo, char* trans, char* diag, lapack_int* n, - lapack_int* nrhs, const double* a, lapack_int* lda, - double* b, lapack_int* ldb, lapack_int *info ); -void LAPACK_ctrtrs( char* uplo, char* trans, char* diag, lapack_int* n, - lapack_int* nrhs, const lapack_complex_float* a, - lapack_int* lda, lapack_complex_float* b, lapack_int* ldb, - lapack_int *info ); -void LAPACK_ztrtrs( char* uplo, char* trans, char* diag, lapack_int* n, - lapack_int* nrhs, const lapack_complex_double* a, - lapack_int* lda, lapack_complex_double* b, lapack_int* ldb, - lapack_int *info ); -void LAPACK_stptrs( char* uplo, char* trans, char* diag, lapack_int* n, - lapack_int* nrhs, const float* ap, float* b, - lapack_int* ldb, lapack_int *info ); -void LAPACK_dtptrs( char* uplo, char* trans, char* diag, lapack_int* n, - lapack_int* nrhs, const double* ap, double* b, - lapack_int* ldb, lapack_int *info ); -void LAPACK_ctptrs( char* uplo, char* trans, char* diag, lapack_int* n, - lapack_int* nrhs, const lapack_complex_float* ap, - lapack_complex_float* b, lapack_int* ldb, - lapack_int *info ); -void LAPACK_ztptrs( char* uplo, char* trans, char* diag, lapack_int* n, - lapack_int* nrhs, const lapack_complex_double* ap, - lapack_complex_double* b, lapack_int* ldb, - lapack_int *info ); -void LAPACK_stbtrs( char* uplo, char* trans, char* diag, lapack_int* n, - lapack_int* kd, lapack_int* nrhs, const float* ab, - lapack_int* ldab, float* b, lapack_int* ldb, - lapack_int *info ); -void LAPACK_dtbtrs( char* uplo, char* trans, char* diag, lapack_int* n, - lapack_int* kd, lapack_int* nrhs, const double* ab, - lapack_int* ldab, double* b, lapack_int* ldb, - lapack_int *info ); -void LAPACK_ctbtrs( char* uplo, char* trans, char* diag, lapack_int* n, - lapack_int* kd, lapack_int* nrhs, - const lapack_complex_float* ab, lapack_int* ldab, - lapack_complex_float* b, lapack_int* ldb, - lapack_int *info ); -void LAPACK_ztbtrs( char* uplo, char* trans, char* diag, lapack_int* n, - lapack_int* kd, lapack_int* nrhs, - const lapack_complex_double* ab, lapack_int* ldab, - lapack_complex_double* b, lapack_int* ldb, - lapack_int *info ); -void LAPACK_sgecon( char* norm, lapack_int* n, const float* a, lapack_int* lda, - float* anorm, float* rcond, float* work, lapack_int* iwork, - lapack_int *info ); -void LAPACK_dgecon( char* norm, lapack_int* n, const double* a, lapack_int* lda, - double* anorm, double* rcond, double* work, - lapack_int* iwork, lapack_int *info ); -void LAPACK_cgecon( char* norm, lapack_int* n, const lapack_complex_float* a, - lapack_int* lda, float* anorm, float* rcond, - lapack_complex_float* work, float* rwork, - lapack_int *info ); -void LAPACK_zgecon( char* norm, lapack_int* n, const lapack_complex_double* a, - lapack_int* lda, double* anorm, double* rcond, - lapack_complex_double* work, double* rwork, - lapack_int *info ); -void LAPACK_sgbcon( char* norm, lapack_int* n, lapack_int* kl, lapack_int* ku, - const float* ab, lapack_int* ldab, const lapack_int* ipiv, - float* anorm, float* rcond, float* work, lapack_int* iwork, - lapack_int *info ); -void LAPACK_dgbcon( char* norm, lapack_int* n, lapack_int* kl, lapack_int* ku, - const double* ab, lapack_int* ldab, const lapack_int* ipiv, - double* anorm, double* rcond, double* work, - lapack_int* iwork, lapack_int *info ); -void LAPACK_cgbcon( char* norm, lapack_int* n, lapack_int* kl, lapack_int* ku, - const lapack_complex_float* ab, lapack_int* ldab, - const lapack_int* ipiv, float* anorm, float* rcond, - lapack_complex_float* work, float* rwork, - lapack_int *info ); -void LAPACK_zgbcon( char* norm, lapack_int* n, lapack_int* kl, lapack_int* ku, - const lapack_complex_double* ab, lapack_int* ldab, - const lapack_int* ipiv, double* anorm, double* rcond, - lapack_complex_double* work, double* rwork, - lapack_int *info ); -void LAPACK_sgtcon( char* norm, lapack_int* n, const float* dl, const float* d, - const float* du, const float* du2, const lapack_int* ipiv, - float* anorm, float* rcond, float* work, lapack_int* iwork, - lapack_int *info ); -void LAPACK_dgtcon( char* norm, lapack_int* n, const double* dl, - const double* d, const double* du, const double* du2, - const lapack_int* ipiv, double* anorm, double* rcond, - double* work, lapack_int* iwork, lapack_int *info ); -void LAPACK_cgtcon( char* norm, lapack_int* n, const lapack_complex_float* dl, - const lapack_complex_float* d, - const lapack_complex_float* du, - const lapack_complex_float* du2, const lapack_int* ipiv, - float* anorm, float* rcond, lapack_complex_float* work, - lapack_int *info ); -void LAPACK_zgtcon( char* norm, lapack_int* n, const lapack_complex_double* dl, - const lapack_complex_double* d, - const lapack_complex_double* du, - const lapack_complex_double* du2, const lapack_int* ipiv, - double* anorm, double* rcond, lapack_complex_double* work, - lapack_int *info ); -void LAPACK_spocon( char* uplo, lapack_int* n, const float* a, lapack_int* lda, - float* anorm, float* rcond, float* work, lapack_int* iwork, - lapack_int *info ); -void LAPACK_dpocon( char* uplo, lapack_int* n, const double* a, lapack_int* lda, - double* anorm, double* rcond, double* work, - lapack_int* iwork, lapack_int *info ); -void LAPACK_cpocon( char* uplo, lapack_int* n, const lapack_complex_float* a, - lapack_int* lda, float* anorm, float* rcond, - lapack_complex_float* work, float* rwork, - lapack_int *info ); -void LAPACK_zpocon( char* uplo, lapack_int* n, const lapack_complex_double* a, - lapack_int* lda, double* anorm, double* rcond, - lapack_complex_double* work, double* rwork, - lapack_int *info ); -void LAPACK_sppcon( char* uplo, lapack_int* n, const float* ap, float* anorm, - float* rcond, float* work, lapack_int* iwork, - lapack_int *info ); -void LAPACK_dppcon( char* uplo, lapack_int* n, const double* ap, double* anorm, - double* rcond, double* work, lapack_int* iwork, - lapack_int *info ); -void LAPACK_cppcon( char* uplo, lapack_int* n, const lapack_complex_float* ap, - float* anorm, float* rcond, lapack_complex_float* work, - float* rwork, lapack_int *info ); -void LAPACK_zppcon( char* uplo, lapack_int* n, const lapack_complex_double* ap, - double* anorm, double* rcond, lapack_complex_double* work, - double* rwork, lapack_int *info ); -void LAPACK_spbcon( char* uplo, lapack_int* n, lapack_int* kd, const float* ab, - lapack_int* ldab, float* anorm, float* rcond, float* work, - lapack_int* iwork, lapack_int *info ); -void LAPACK_dpbcon( char* uplo, lapack_int* n, lapack_int* kd, const double* ab, - lapack_int* ldab, double* anorm, double* rcond, - double* work, lapack_int* iwork, lapack_int *info ); -void LAPACK_cpbcon( char* uplo, lapack_int* n, lapack_int* kd, - const lapack_complex_float* ab, lapack_int* ldab, - float* anorm, float* rcond, lapack_complex_float* work, - float* rwork, lapack_int *info ); -void LAPACK_zpbcon( char* uplo, lapack_int* n, lapack_int* kd, - const lapack_complex_double* ab, lapack_int* ldab, - double* anorm, double* rcond, lapack_complex_double* work, - double* rwork, lapack_int *info ); -void LAPACK_sptcon( lapack_int* n, const float* d, const float* e, float* anorm, - float* rcond, float* work, lapack_int *info ); -void LAPACK_dptcon( lapack_int* n, const double* d, const double* e, - double* anorm, double* rcond, double* work, - lapack_int *info ); -void LAPACK_cptcon( lapack_int* n, const float* d, - const lapack_complex_float* e, float* anorm, float* rcond, - float* work, lapack_int *info ); -void LAPACK_zptcon( lapack_int* n, const double* d, - const lapack_complex_double* e, double* anorm, - double* rcond, double* work, lapack_int *info ); -void LAPACK_ssycon( char* uplo, lapack_int* n, const float* a, lapack_int* lda, - const lapack_int* ipiv, float* anorm, float* rcond, - float* work, lapack_int* iwork, lapack_int *info ); -void LAPACK_dsycon( char* uplo, lapack_int* n, const double* a, lapack_int* lda, - const lapack_int* ipiv, double* anorm, double* rcond, - double* work, lapack_int* iwork, lapack_int *info ); -void LAPACK_csycon( char* uplo, lapack_int* n, const lapack_complex_float* a, - lapack_int* lda, const lapack_int* ipiv, float* anorm, - float* rcond, lapack_complex_float* work, - lapack_int *info ); -void LAPACK_zsycon( char* uplo, lapack_int* n, const lapack_complex_double* a, - lapack_int* lda, const lapack_int* ipiv, double* anorm, - double* rcond, lapack_complex_double* work, - lapack_int *info ); -void LAPACK_checon( char* uplo, lapack_int* n, const lapack_complex_float* a, - lapack_int* lda, const lapack_int* ipiv, float* anorm, - float* rcond, lapack_complex_float* work, - lapack_int *info ); -void LAPACK_zhecon( char* uplo, lapack_int* n, const lapack_complex_double* a, - lapack_int* lda, const lapack_int* ipiv, double* anorm, - double* rcond, lapack_complex_double* work, - lapack_int *info ); -void LAPACK_sspcon( char* uplo, lapack_int* n, const float* ap, - const lapack_int* ipiv, float* anorm, float* rcond, - float* work, lapack_int* iwork, lapack_int *info ); -void LAPACK_dspcon( char* uplo, lapack_int* n, const double* ap, - const lapack_int* ipiv, double* anorm, double* rcond, - double* work, lapack_int* iwork, lapack_int *info ); -void LAPACK_cspcon( char* uplo, lapack_int* n, const lapack_complex_float* ap, - const lapack_int* ipiv, float* anorm, float* rcond, - lapack_complex_float* work, lapack_int *info ); -void LAPACK_zspcon( char* uplo, lapack_int* n, const lapack_complex_double* ap, - const lapack_int* ipiv, double* anorm, double* rcond, - lapack_complex_double* work, lapack_int *info ); -void LAPACK_chpcon( char* uplo, lapack_int* n, const lapack_complex_float* ap, - const lapack_int* ipiv, float* anorm, float* rcond, - lapack_complex_float* work, lapack_int *info ); -void LAPACK_zhpcon( char* uplo, lapack_int* n, const lapack_complex_double* ap, - const lapack_int* ipiv, double* anorm, double* rcond, - lapack_complex_double* work, lapack_int *info ); -void LAPACK_strcon( char* norm, char* uplo, char* diag, lapack_int* n, - const float* a, lapack_int* lda, float* rcond, float* work, - lapack_int* iwork, lapack_int *info ); -void LAPACK_dtrcon( char* norm, char* uplo, char* diag, lapack_int* n, - const double* a, lapack_int* lda, double* rcond, - double* work, lapack_int* iwork, lapack_int *info ); -void LAPACK_ctrcon( char* norm, char* uplo, char* diag, lapack_int* n, - const lapack_complex_float* a, lapack_int* lda, - float* rcond, lapack_complex_float* work, float* rwork, - lapack_int *info ); -void LAPACK_ztrcon( char* norm, char* uplo, char* diag, lapack_int* n, - const lapack_complex_double* a, lapack_int* lda, - double* rcond, lapack_complex_double* work, double* rwork, - lapack_int *info ); -void LAPACK_stpcon( char* norm, char* uplo, char* diag, lapack_int* n, - const float* ap, float* rcond, float* work, - lapack_int* iwork, lapack_int *info ); -void LAPACK_dtpcon( char* norm, char* uplo, char* diag, lapack_int* n, - const double* ap, double* rcond, double* work, - lapack_int* iwork, lapack_int *info ); -void LAPACK_ctpcon( char* norm, char* uplo, char* diag, lapack_int* n, - const lapack_complex_float* ap, float* rcond, - lapack_complex_float* work, float* rwork, - lapack_int *info ); -void LAPACK_ztpcon( char* norm, char* uplo, char* diag, lapack_int* n, - const lapack_complex_double* ap, double* rcond, - lapack_complex_double* work, double* rwork, - lapack_int *info ); -void LAPACK_stbcon( char* norm, char* uplo, char* diag, lapack_int* n, - lapack_int* kd, const float* ab, lapack_int* ldab, - float* rcond, float* work, lapack_int* iwork, - lapack_int *info ); -void LAPACK_dtbcon( char* norm, char* uplo, char* diag, lapack_int* n, - lapack_int* kd, const double* ab, lapack_int* ldab, - double* rcond, double* work, lapack_int* iwork, - lapack_int *info ); -void LAPACK_ctbcon( char* norm, char* uplo, char* diag, lapack_int* n, - lapack_int* kd, const lapack_complex_float* ab, - lapack_int* ldab, float* rcond, lapack_complex_float* work, - float* rwork, lapack_int *info ); -void LAPACK_ztbcon( char* norm, char* uplo, char* diag, lapack_int* n, - lapack_int* kd, const lapack_complex_double* ab, - lapack_int* ldab, double* rcond, - lapack_complex_double* work, double* rwork, - lapack_int *info ); -void LAPACK_sgerfs( char* trans, lapack_int* n, lapack_int* nrhs, - const float* a, lapack_int* lda, const float* af, - lapack_int* ldaf, const lapack_int* ipiv, const float* b, - lapack_int* ldb, float* x, lapack_int* ldx, float* ferr, - float* berr, float* work, lapack_int* iwork, - lapack_int *info ); -void LAPACK_dgerfs( char* trans, lapack_int* n, lapack_int* nrhs, - const double* a, lapack_int* lda, const double* af, - lapack_int* ldaf, const lapack_int* ipiv, const double* b, - lapack_int* ldb, double* x, lapack_int* ldx, double* ferr, - double* berr, double* work, lapack_int* iwork, - lapack_int *info ); -void LAPACK_cgerfs( char* trans, lapack_int* n, lapack_int* nrhs, - const lapack_complex_float* a, lapack_int* lda, - const lapack_complex_float* af, lapack_int* ldaf, - const lapack_int* ipiv, const lapack_complex_float* b, - lapack_int* ldb, lapack_complex_float* x, lapack_int* ldx, - float* ferr, float* berr, lapack_complex_float* work, - float* rwork, lapack_int *info ); -void LAPACK_zgerfs( char* trans, lapack_int* n, lapack_int* nrhs, - const lapack_complex_double* a, lapack_int* lda, - const lapack_complex_double* af, lapack_int* ldaf, - const lapack_int* ipiv, const lapack_complex_double* b, - lapack_int* ldb, lapack_complex_double* x, lapack_int* ldx, - double* ferr, double* berr, lapack_complex_double* work, - double* rwork, lapack_int *info ); -void LAPACK_dgerfsx( char* trans, char* equed, lapack_int* n, lapack_int* nrhs, - const double* a, lapack_int* lda, const double* af, - lapack_int* ldaf, const lapack_int* ipiv, const double* r, - const double* c, const double* b, lapack_int* ldb, - double* x, lapack_int* ldx, double* rcond, double* berr, - lapack_int* n_err_bnds, double* err_bnds_norm, - double* err_bnds_comp, lapack_int* nparams, double* params, - double* work, lapack_int* iwork, lapack_int *info ); -void LAPACK_sgerfsx( char* trans, char* equed, lapack_int* n, lapack_int* nrhs, - const float* a, lapack_int* lda, const float* af, - lapack_int* ldaf, const lapack_int* ipiv, const float* r, - const float* c, const float* b, lapack_int* ldb, float* x, - lapack_int* ldx, float* rcond, float* berr, - lapack_int* n_err_bnds, float* err_bnds_norm, - float* err_bnds_comp, lapack_int* nparams, float* params, - float* work, lapack_int* iwork, lapack_int *info ); -void LAPACK_zgerfsx( char* trans, char* equed, lapack_int* n, lapack_int* nrhs, - const lapack_complex_double* a, lapack_int* lda, - const lapack_complex_double* af, lapack_int* ldaf, - const lapack_int* ipiv, const double* r, const double* c, - const lapack_complex_double* b, lapack_int* ldb, - lapack_complex_double* x, lapack_int* ldx, double* rcond, - double* berr, lapack_int* n_err_bnds, - double* err_bnds_norm, double* err_bnds_comp, - lapack_int* nparams, double* params, - lapack_complex_double* work, double* rwork, - lapack_int *info ); -void LAPACK_cgerfsx( char* trans, char* equed, lapack_int* n, lapack_int* nrhs, - const lapack_complex_float* a, lapack_int* lda, - const lapack_complex_float* af, lapack_int* ldaf, - const lapack_int* ipiv, const float* r, const float* c, - const lapack_complex_float* b, lapack_int* ldb, - lapack_complex_float* x, lapack_int* ldx, float* rcond, - float* berr, lapack_int* n_err_bnds, float* err_bnds_norm, - float* err_bnds_comp, lapack_int* nparams, float* params, - lapack_complex_float* work, float* rwork, - lapack_int *info ); -void LAPACK_sgbrfs( char* trans, lapack_int* n, lapack_int* kl, lapack_int* ku, - lapack_int* nrhs, const float* ab, lapack_int* ldab, - const float* afb, lapack_int* ldafb, const lapack_int* ipiv, - const float* b, lapack_int* ldb, float* x, lapack_int* ldx, - float* ferr, float* berr, float* work, lapack_int* iwork, - lapack_int *info ); -void LAPACK_dgbrfs( char* trans, lapack_int* n, lapack_int* kl, lapack_int* ku, - lapack_int* nrhs, const double* ab, lapack_int* ldab, - const double* afb, lapack_int* ldafb, - const lapack_int* ipiv, const double* b, lapack_int* ldb, - double* x, lapack_int* ldx, double* ferr, double* berr, - double* work, lapack_int* iwork, lapack_int *info ); -void LAPACK_cgbrfs( char* trans, lapack_int* n, lapack_int* kl, lapack_int* ku, - lapack_int* nrhs, const lapack_complex_float* ab, - lapack_int* ldab, const lapack_complex_float* afb, - lapack_int* ldafb, const lapack_int* ipiv, - const lapack_complex_float* b, lapack_int* ldb, - lapack_complex_float* x, lapack_int* ldx, float* ferr, - float* berr, lapack_complex_float* work, float* rwork, - lapack_int *info ); -void LAPACK_zgbrfs( char* trans, lapack_int* n, lapack_int* kl, lapack_int* ku, - lapack_int* nrhs, const lapack_complex_double* ab, - lapack_int* ldab, const lapack_complex_double* afb, - lapack_int* ldafb, const lapack_int* ipiv, - const lapack_complex_double* b, lapack_int* ldb, - lapack_complex_double* x, lapack_int* ldx, double* ferr, - double* berr, lapack_complex_double* work, double* rwork, - lapack_int *info ); -void LAPACK_dgbrfsx( char* trans, char* equed, lapack_int* n, lapack_int* kl, - lapack_int* ku, lapack_int* nrhs, const double* ab, - lapack_int* ldab, const double* afb, lapack_int* ldafb, - const lapack_int* ipiv, const double* r, const double* c, - const double* b, lapack_int* ldb, double* x, - lapack_int* ldx, double* rcond, double* berr, - lapack_int* n_err_bnds, double* err_bnds_norm, - double* err_bnds_comp, lapack_int* nparams, double* params, - double* work, lapack_int* iwork, lapack_int *info ); -void LAPACK_sgbrfsx( char* trans, char* equed, lapack_int* n, lapack_int* kl, - lapack_int* ku, lapack_int* nrhs, const float* ab, - lapack_int* ldab, const float* afb, lapack_int* ldafb, - const lapack_int* ipiv, const float* r, const float* c, - const float* b, lapack_int* ldb, float* x, lapack_int* ldx, - float* rcond, float* berr, lapack_int* n_err_bnds, - float* err_bnds_norm, float* err_bnds_comp, - lapack_int* nparams, float* params, float* work, - lapack_int* iwork, lapack_int *info ); -void LAPACK_zgbrfsx( char* trans, char* equed, lapack_int* n, lapack_int* kl, - lapack_int* ku, lapack_int* nrhs, - const lapack_complex_double* ab, lapack_int* ldab, - const lapack_complex_double* afb, lapack_int* ldafb, - const lapack_int* ipiv, const double* r, const double* c, - const lapack_complex_double* b, lapack_int* ldb, - lapack_complex_double* x, lapack_int* ldx, double* rcond, - double* berr, lapack_int* n_err_bnds, - double* err_bnds_norm, double* err_bnds_comp, - lapack_int* nparams, double* params, - lapack_complex_double* work, double* rwork, - lapack_int *info ); -void LAPACK_cgbrfsx( char* trans, char* equed, lapack_int* n, lapack_int* kl, - lapack_int* ku, lapack_int* nrhs, - const lapack_complex_float* ab, lapack_int* ldab, - const lapack_complex_float* afb, lapack_int* ldafb, - const lapack_int* ipiv, const float* r, const float* c, - const lapack_complex_float* b, lapack_int* ldb, - lapack_complex_float* x, lapack_int* ldx, float* rcond, - float* berr, lapack_int* n_err_bnds, float* err_bnds_norm, - float* err_bnds_comp, lapack_int* nparams, float* params, - lapack_complex_float* work, float* rwork, - lapack_int *info ); -void LAPACK_sgtrfs( char* trans, lapack_int* n, lapack_int* nrhs, - const float* dl, const float* d, const float* du, - const float* dlf, const float* df, const float* duf, - const float* du2, const lapack_int* ipiv, const float* b, - lapack_int* ldb, float* x, lapack_int* ldx, float* ferr, - float* berr, float* work, lapack_int* iwork, - lapack_int *info ); -void LAPACK_dgtrfs( char* trans, lapack_int* n, lapack_int* nrhs, - const double* dl, const double* d, const double* du, - const double* dlf, const double* df, const double* duf, - const double* du2, const lapack_int* ipiv, const double* b, - lapack_int* ldb, double* x, lapack_int* ldx, double* ferr, - double* berr, double* work, lapack_int* iwork, - lapack_int *info ); -void LAPACK_cgtrfs( char* trans, lapack_int* n, lapack_int* nrhs, - const lapack_complex_float* dl, - const lapack_complex_float* d, - const lapack_complex_float* du, - const lapack_complex_float* dlf, - const lapack_complex_float* df, - const lapack_complex_float* duf, - const lapack_complex_float* du2, const lapack_int* ipiv, - const lapack_complex_float* b, lapack_int* ldb, - lapack_complex_float* x, lapack_int* ldx, float* ferr, - float* berr, lapack_complex_float* work, float* rwork, - lapack_int *info ); -void LAPACK_zgtrfs( char* trans, lapack_int* n, lapack_int* nrhs, - const lapack_complex_double* dl, - const lapack_complex_double* d, - const lapack_complex_double* du, - const lapack_complex_double* dlf, - const lapack_complex_double* df, - const lapack_complex_double* duf, - const lapack_complex_double* du2, const lapack_int* ipiv, - const lapack_complex_double* b, lapack_int* ldb, - lapack_complex_double* x, lapack_int* ldx, double* ferr, - double* berr, lapack_complex_double* work, double* rwork, - lapack_int *info ); -void LAPACK_sporfs( char* uplo, lapack_int* n, lapack_int* nrhs, const float* a, - lapack_int* lda, const float* af, lapack_int* ldaf, - const float* b, lapack_int* ldb, float* x, lapack_int* ldx, - float* ferr, float* berr, float* work, lapack_int* iwork, - lapack_int *info ); -void LAPACK_dporfs( char* uplo, lapack_int* n, lapack_int* nrhs, - const double* a, lapack_int* lda, const double* af, - lapack_int* ldaf, const double* b, lapack_int* ldb, - double* x, lapack_int* ldx, double* ferr, double* berr, - double* work, lapack_int* iwork, lapack_int *info ); -void LAPACK_cporfs( char* uplo, lapack_int* n, lapack_int* nrhs, - const lapack_complex_float* a, lapack_int* lda, - const lapack_complex_float* af, lapack_int* ldaf, - const lapack_complex_float* b, lapack_int* ldb, - lapack_complex_float* x, lapack_int* ldx, float* ferr, - float* berr, lapack_complex_float* work, float* rwork, - lapack_int *info ); -void LAPACK_zporfs( char* uplo, lapack_int* n, lapack_int* nrhs, - const lapack_complex_double* a, lapack_int* lda, - const lapack_complex_double* af, lapack_int* ldaf, - const lapack_complex_double* b, lapack_int* ldb, - lapack_complex_double* x, lapack_int* ldx, double* ferr, - double* berr, lapack_complex_double* work, double* rwork, - lapack_int *info ); -void LAPACK_dporfsx( char* uplo, char* equed, lapack_int* n, lapack_int* nrhs, - const double* a, lapack_int* lda, const double* af, - lapack_int* ldaf, const double* s, const double* b, - lapack_int* ldb, double* x, lapack_int* ldx, double* rcond, - double* berr, lapack_int* n_err_bnds, - double* err_bnds_norm, double* err_bnds_comp, - lapack_int* nparams, double* params, double* work, - lapack_int* iwork, lapack_int *info ); -void LAPACK_sporfsx( char* uplo, char* equed, lapack_int* n, lapack_int* nrhs, - const float* a, lapack_int* lda, const float* af, - lapack_int* ldaf, const float* s, const float* b, - lapack_int* ldb, float* x, lapack_int* ldx, float* rcond, - float* berr, lapack_int* n_err_bnds, float* err_bnds_norm, - float* err_bnds_comp, lapack_int* nparams, float* params, - float* work, lapack_int* iwork, lapack_int *info ); -void LAPACK_zporfsx( char* uplo, char* equed, lapack_int* n, lapack_int* nrhs, - const lapack_complex_double* a, lapack_int* lda, - const lapack_complex_double* af, lapack_int* ldaf, - const double* s, const lapack_complex_double* b, - lapack_int* ldb, lapack_complex_double* x, lapack_int* ldx, - double* rcond, double* berr, lapack_int* n_err_bnds, - double* err_bnds_norm, double* err_bnds_comp, - lapack_int* nparams, double* params, - lapack_complex_double* work, double* rwork, - lapack_int *info ); -void LAPACK_cporfsx( char* uplo, char* equed, lapack_int* n, lapack_int* nrhs, - const lapack_complex_float* a, lapack_int* lda, - const lapack_complex_float* af, lapack_int* ldaf, - const float* s, const lapack_complex_float* b, - lapack_int* ldb, lapack_complex_float* x, lapack_int* ldx, - float* rcond, float* berr, lapack_int* n_err_bnds, - float* err_bnds_norm, float* err_bnds_comp, - lapack_int* nparams, float* params, - lapack_complex_float* work, float* rwork, - lapack_int *info ); -void LAPACK_spprfs( char* uplo, lapack_int* n, lapack_int* nrhs, - const float* ap, const float* afp, const float* b, - lapack_int* ldb, float* x, lapack_int* ldx, float* ferr, - float* berr, float* work, lapack_int* iwork, - lapack_int *info ); -void LAPACK_dpprfs( char* uplo, lapack_int* n, lapack_int* nrhs, - const double* ap, const double* afp, const double* b, - lapack_int* ldb, double* x, lapack_int* ldx, double* ferr, - double* berr, double* work, lapack_int* iwork, - lapack_int *info ); -void LAPACK_cpprfs( char* uplo, lapack_int* n, lapack_int* nrhs, - const lapack_complex_float* ap, - const lapack_complex_float* afp, - const lapack_complex_float* b, lapack_int* ldb, - lapack_complex_float* x, lapack_int* ldx, float* ferr, - float* berr, lapack_complex_float* work, float* rwork, - lapack_int *info ); -void LAPACK_zpprfs( char* uplo, lapack_int* n, lapack_int* nrhs, - const lapack_complex_double* ap, - const lapack_complex_double* afp, - const lapack_complex_double* b, lapack_int* ldb, - lapack_complex_double* x, lapack_int* ldx, double* ferr, - double* berr, lapack_complex_double* work, double* rwork, - lapack_int *info ); -void LAPACK_spbrfs( char* uplo, lapack_int* n, lapack_int* kd, lapack_int* nrhs, - const float* ab, lapack_int* ldab, const float* afb, - lapack_int* ldafb, const float* b, lapack_int* ldb, - float* x, lapack_int* ldx, float* ferr, float* berr, - float* work, lapack_int* iwork, lapack_int *info ); -void LAPACK_dpbrfs( char* uplo, lapack_int* n, lapack_int* kd, lapack_int* nrhs, - const double* ab, lapack_int* ldab, const double* afb, - lapack_int* ldafb, const double* b, lapack_int* ldb, - double* x, lapack_int* ldx, double* ferr, double* berr, - double* work, lapack_int* iwork, lapack_int *info ); -void LAPACK_cpbrfs( char* uplo, lapack_int* n, lapack_int* kd, lapack_int* nrhs, - const lapack_complex_float* ab, lapack_int* ldab, - const lapack_complex_float* afb, lapack_int* ldafb, - const lapack_complex_float* b, lapack_int* ldb, - lapack_complex_float* x, lapack_int* ldx, float* ferr, - float* berr, lapack_complex_float* work, float* rwork, - lapack_int *info ); -void LAPACK_zpbrfs( char* uplo, lapack_int* n, lapack_int* kd, lapack_int* nrhs, - const lapack_complex_double* ab, lapack_int* ldab, - const lapack_complex_double* afb, lapack_int* ldafb, - const lapack_complex_double* b, lapack_int* ldb, - lapack_complex_double* x, lapack_int* ldx, double* ferr, - double* berr, lapack_complex_double* work, double* rwork, - lapack_int *info ); -void LAPACK_sptrfs( lapack_int* n, lapack_int* nrhs, const float* d, - const float* e, const float* df, const float* ef, - const float* b, lapack_int* ldb, float* x, lapack_int* ldx, - float* ferr, float* berr, float* work, lapack_int *info ); -void LAPACK_dptrfs( lapack_int* n, lapack_int* nrhs, const double* d, - const double* e, const double* df, const double* ef, - const double* b, lapack_int* ldb, double* x, - lapack_int* ldx, double* ferr, double* berr, double* work, - lapack_int *info ); -void LAPACK_cptrfs( char* uplo, lapack_int* n, lapack_int* nrhs, const float* d, - const lapack_complex_float* e, const float* df, - const lapack_complex_float* ef, - const lapack_complex_float* b, lapack_int* ldb, - lapack_complex_float* x, lapack_int* ldx, float* ferr, - float* berr, lapack_complex_float* work, float* rwork, - lapack_int *info ); -void LAPACK_zptrfs( char* uplo, lapack_int* n, lapack_int* nrhs, - const double* d, const lapack_complex_double* e, - const double* df, const lapack_complex_double* ef, - const lapack_complex_double* b, lapack_int* ldb, - lapack_complex_double* x, lapack_int* ldx, double* ferr, - double* berr, lapack_complex_double* work, double* rwork, - lapack_int *info ); -void LAPACK_ssyrfs( char* uplo, lapack_int* n, lapack_int* nrhs, const float* a, - lapack_int* lda, const float* af, lapack_int* ldaf, - const lapack_int* ipiv, const float* b, lapack_int* ldb, - float* x, lapack_int* ldx, float* ferr, float* berr, - float* work, lapack_int* iwork, lapack_int *info ); -void LAPACK_dsyrfs( char* uplo, lapack_int* n, lapack_int* nrhs, - const double* a, lapack_int* lda, const double* af, - lapack_int* ldaf, const lapack_int* ipiv, const double* b, - lapack_int* ldb, double* x, lapack_int* ldx, double* ferr, - double* berr, double* work, lapack_int* iwork, - lapack_int *info ); -void LAPACK_csyrfs( char* uplo, lapack_int* n, lapack_int* nrhs, - const lapack_complex_float* a, lapack_int* lda, - const lapack_complex_float* af, lapack_int* ldaf, - const lapack_int* ipiv, const lapack_complex_float* b, - lapack_int* ldb, lapack_complex_float* x, lapack_int* ldx, - float* ferr, float* berr, lapack_complex_float* work, - float* rwork, lapack_int *info ); -void LAPACK_zsyrfs( char* uplo, lapack_int* n, lapack_int* nrhs, - const lapack_complex_double* a, lapack_int* lda, - const lapack_complex_double* af, lapack_int* ldaf, - const lapack_int* ipiv, const lapack_complex_double* b, - lapack_int* ldb, lapack_complex_double* x, lapack_int* ldx, - double* ferr, double* berr, lapack_complex_double* work, - double* rwork, lapack_int *info ); -void LAPACK_dsyrfsx( char* uplo, char* equed, lapack_int* n, lapack_int* nrhs, - const double* a, lapack_int* lda, const double* af, - lapack_int* ldaf, const lapack_int* ipiv, const double* s, - const double* b, lapack_int* ldb, double* x, - lapack_int* ldx, double* rcond, double* berr, - lapack_int* n_err_bnds, double* err_bnds_norm, - double* err_bnds_comp, lapack_int* nparams, double* params, - double* work, lapack_int* iwork, lapack_int *info ); -void LAPACK_ssyrfsx( char* uplo, char* equed, lapack_int* n, lapack_int* nrhs, - const float* a, lapack_int* lda, const float* af, - lapack_int* ldaf, const lapack_int* ipiv, const float* s, - const float* b, lapack_int* ldb, float* x, lapack_int* ldx, - float* rcond, float* berr, lapack_int* n_err_bnds, - float* err_bnds_norm, float* err_bnds_comp, - lapack_int* nparams, float* params, float* work, - lapack_int* iwork, lapack_int *info ); -void LAPACK_zsyrfsx( char* uplo, char* equed, lapack_int* n, lapack_int* nrhs, - const lapack_complex_double* a, lapack_int* lda, - const lapack_complex_double* af, lapack_int* ldaf, - const lapack_int* ipiv, const double* s, - const lapack_complex_double* b, lapack_int* ldb, - lapack_complex_double* x, lapack_int* ldx, double* rcond, - double* berr, lapack_int* n_err_bnds, - double* err_bnds_norm, double* err_bnds_comp, - lapack_int* nparams, double* params, - lapack_complex_double* work, double* rwork, - lapack_int *info ); -void LAPACK_csyrfsx( char* uplo, char* equed, lapack_int* n, lapack_int* nrhs, - const lapack_complex_float* a, lapack_int* lda, - const lapack_complex_float* af, lapack_int* ldaf, - const lapack_int* ipiv, const float* s, - const lapack_complex_float* b, lapack_int* ldb, - lapack_complex_float* x, lapack_int* ldx, float* rcond, - float* berr, lapack_int* n_err_bnds, float* err_bnds_norm, - float* err_bnds_comp, lapack_int* nparams, float* params, - lapack_complex_float* work, float* rwork, - lapack_int *info ); -void LAPACK_cherfs( char* uplo, lapack_int* n, lapack_int* nrhs, - const lapack_complex_float* a, lapack_int* lda, - const lapack_complex_float* af, lapack_int* ldaf, - const lapack_int* ipiv, const lapack_complex_float* b, - lapack_int* ldb, lapack_complex_float* x, lapack_int* ldx, - float* ferr, float* berr, lapack_complex_float* work, - float* rwork, lapack_int *info ); -void LAPACK_zherfs( char* uplo, lapack_int* n, lapack_int* nrhs, - const lapack_complex_double* a, lapack_int* lda, - const lapack_complex_double* af, lapack_int* ldaf, - const lapack_int* ipiv, const lapack_complex_double* b, - lapack_int* ldb, lapack_complex_double* x, lapack_int* ldx, - double* ferr, double* berr, lapack_complex_double* work, - double* rwork, lapack_int *info ); -void LAPACK_zherfsx( char* uplo, char* equed, lapack_int* n, lapack_int* nrhs, - const lapack_complex_double* a, lapack_int* lda, - const lapack_complex_double* af, lapack_int* ldaf, - const lapack_int* ipiv, const double* s, - const lapack_complex_double* b, lapack_int* ldb, - lapack_complex_double* x, lapack_int* ldx, double* rcond, - double* berr, lapack_int* n_err_bnds, - double* err_bnds_norm, double* err_bnds_comp, - lapack_int* nparams, double* params, - lapack_complex_double* work, double* rwork, - lapack_int *info ); -void LAPACK_cherfsx( char* uplo, char* equed, lapack_int* n, lapack_int* nrhs, - const lapack_complex_float* a, lapack_int* lda, - const lapack_complex_float* af, lapack_int* ldaf, - const lapack_int* ipiv, const float* s, - const lapack_complex_float* b, lapack_int* ldb, - lapack_complex_float* x, lapack_int* ldx, float* rcond, - float* berr, lapack_int* n_err_bnds, float* err_bnds_norm, - float* err_bnds_comp, lapack_int* nparams, float* params, - lapack_complex_float* work, float* rwork, - lapack_int *info ); -void LAPACK_ssprfs( char* uplo, lapack_int* n, lapack_int* nrhs, - const float* ap, const float* afp, const lapack_int* ipiv, - const float* b, lapack_int* ldb, float* x, lapack_int* ldx, - float* ferr, float* berr, float* work, lapack_int* iwork, - lapack_int *info ); -void LAPACK_dsprfs( char* uplo, lapack_int* n, lapack_int* nrhs, - const double* ap, const double* afp, const lapack_int* ipiv, - const double* b, lapack_int* ldb, double* x, - lapack_int* ldx, double* ferr, double* berr, double* work, - lapack_int* iwork, lapack_int *info ); -void LAPACK_csprfs( char* uplo, lapack_int* n, lapack_int* nrhs, - const lapack_complex_float* ap, - const lapack_complex_float* afp, const lapack_int* ipiv, - const lapack_complex_float* b, lapack_int* ldb, - lapack_complex_float* x, lapack_int* ldx, float* ferr, - float* berr, lapack_complex_float* work, float* rwork, - lapack_int *info ); -void LAPACK_zsprfs( char* uplo, lapack_int* n, lapack_int* nrhs, - const lapack_complex_double* ap, - const lapack_complex_double* afp, const lapack_int* ipiv, - const lapack_complex_double* b, lapack_int* ldb, - lapack_complex_double* x, lapack_int* ldx, double* ferr, - double* berr, lapack_complex_double* work, double* rwork, - lapack_int *info ); -void LAPACK_chprfs( char* uplo, lapack_int* n, lapack_int* nrhs, - const lapack_complex_float* ap, - const lapack_complex_float* afp, const lapack_int* ipiv, - const lapack_complex_float* b, lapack_int* ldb, - lapack_complex_float* x, lapack_int* ldx, float* ferr, - float* berr, lapack_complex_float* work, float* rwork, - lapack_int *info ); -void LAPACK_zhprfs( char* uplo, lapack_int* n, lapack_int* nrhs, - const lapack_complex_double* ap, - const lapack_complex_double* afp, const lapack_int* ipiv, - const lapack_complex_double* b, lapack_int* ldb, - lapack_complex_double* x, lapack_int* ldx, double* ferr, - double* berr, lapack_complex_double* work, double* rwork, - lapack_int *info ); -void LAPACK_strrfs( char* uplo, char* trans, char* diag, lapack_int* n, - lapack_int* nrhs, const float* a, lapack_int* lda, - const float* b, lapack_int* ldb, const float* x, - lapack_int* ldx, float* ferr, float* berr, float* work, - lapack_int* iwork, lapack_int *info ); -void LAPACK_dtrrfs( char* uplo, char* trans, char* diag, lapack_int* n, - lapack_int* nrhs, const double* a, lapack_int* lda, - const double* b, lapack_int* ldb, const double* x, - lapack_int* ldx, double* ferr, double* berr, double* work, - lapack_int* iwork, lapack_int *info ); -void LAPACK_ctrrfs( char* uplo, char* trans, char* diag, lapack_int* n, - lapack_int* nrhs, const lapack_complex_float* a, - lapack_int* lda, const lapack_complex_float* b, - lapack_int* ldb, const lapack_complex_float* x, - lapack_int* ldx, float* ferr, float* berr, - lapack_complex_float* work, float* rwork, - lapack_int *info ); -void LAPACK_ztrrfs( char* uplo, char* trans, char* diag, lapack_int* n, - lapack_int* nrhs, const lapack_complex_double* a, - lapack_int* lda, const lapack_complex_double* b, - lapack_int* ldb, const lapack_complex_double* x, - lapack_int* ldx, double* ferr, double* berr, - lapack_complex_double* work, double* rwork, - lapack_int *info ); -void LAPACK_stprfs( char* uplo, char* trans, char* diag, lapack_int* n, - lapack_int* nrhs, const float* ap, const float* b, - lapack_int* ldb, const float* x, lapack_int* ldx, - float* ferr, float* berr, float* work, lapack_int* iwork, - lapack_int *info ); -void LAPACK_dtprfs( char* uplo, char* trans, char* diag, lapack_int* n, - lapack_int* nrhs, const double* ap, const double* b, - lapack_int* ldb, const double* x, lapack_int* ldx, - double* ferr, double* berr, double* work, lapack_int* iwork, - lapack_int *info ); -void LAPACK_ctprfs( char* uplo, char* trans, char* diag, lapack_int* n, - lapack_int* nrhs, const lapack_complex_float* ap, - const lapack_complex_float* b, lapack_int* ldb, - const lapack_complex_float* x, lapack_int* ldx, float* ferr, - float* berr, lapack_complex_float* work, float* rwork, - lapack_int *info ); -void LAPACK_ztprfs( char* uplo, char* trans, char* diag, lapack_int* n, - lapack_int* nrhs, const lapack_complex_double* ap, - const lapack_complex_double* b, lapack_int* ldb, - const lapack_complex_double* x, lapack_int* ldx, - double* ferr, double* berr, lapack_complex_double* work, - double* rwork, lapack_int *info ); -void LAPACK_stbrfs( char* uplo, char* trans, char* diag, lapack_int* n, - lapack_int* kd, lapack_int* nrhs, const float* ab, - lapack_int* ldab, const float* b, lapack_int* ldb, - const float* x, lapack_int* ldx, float* ferr, float* berr, - float* work, lapack_int* iwork, lapack_int *info ); -void LAPACK_dtbrfs( char* uplo, char* trans, char* diag, lapack_int* n, - lapack_int* kd, lapack_int* nrhs, const double* ab, - lapack_int* ldab, const double* b, lapack_int* ldb, - const double* x, lapack_int* ldx, double* ferr, - double* berr, double* work, lapack_int* iwork, - lapack_int *info ); -void LAPACK_ctbrfs( char* uplo, char* trans, char* diag, lapack_int* n, - lapack_int* kd, lapack_int* nrhs, - const lapack_complex_float* ab, lapack_int* ldab, - const lapack_complex_float* b, lapack_int* ldb, - const lapack_complex_float* x, lapack_int* ldx, float* ferr, - float* berr, lapack_complex_float* work, float* rwork, - lapack_int *info ); -void LAPACK_ztbrfs( char* uplo, char* trans, char* diag, lapack_int* n, - lapack_int* kd, lapack_int* nrhs, - const lapack_complex_double* ab, lapack_int* ldab, - const lapack_complex_double* b, lapack_int* ldb, - const lapack_complex_double* x, lapack_int* ldx, - double* ferr, double* berr, lapack_complex_double* work, - double* rwork, lapack_int *info ); -void LAPACK_sgetri( lapack_int* n, float* a, lapack_int* lda, - const lapack_int* ipiv, float* work, lapack_int* lwork, - lapack_int *info ); -void LAPACK_dgetri( lapack_int* n, double* a, lapack_int* lda, - const lapack_int* ipiv, double* work, lapack_int* lwork, - lapack_int *info ); -void LAPACK_cgetri( lapack_int* n, lapack_complex_float* a, lapack_int* lda, - const lapack_int* ipiv, lapack_complex_float* work, - lapack_int* lwork, lapack_int *info ); -void LAPACK_zgetri( lapack_int* n, lapack_complex_double* a, lapack_int* lda, - const lapack_int* ipiv, lapack_complex_double* work, - lapack_int* lwork, lapack_int *info ); -void LAPACK_spotri( char* uplo, lapack_int* n, float* a, lapack_int* lda, - lapack_int *info ); -void LAPACK_dpotri( char* uplo, lapack_int* n, double* a, lapack_int* lda, - lapack_int *info ); -void LAPACK_cpotri( char* uplo, lapack_int* n, lapack_complex_float* a, - lapack_int* lda, lapack_int *info ); -void LAPACK_zpotri( char* uplo, lapack_int* n, lapack_complex_double* a, - lapack_int* lda, lapack_int *info ); -void LAPACK_dpftri( char* transr, char* uplo, lapack_int* n, double* a, - lapack_int *info ); -void LAPACK_spftri( char* transr, char* uplo, lapack_int* n, float* a, - lapack_int *info ); -void LAPACK_zpftri( char* transr, char* uplo, lapack_int* n, - lapack_complex_double* a, lapack_int *info ); -void LAPACK_cpftri( char* transr, char* uplo, lapack_int* n, - lapack_complex_float* a, lapack_int *info ); -void LAPACK_spptri( char* uplo, lapack_int* n, float* ap, lapack_int *info ); -void LAPACK_dpptri( char* uplo, lapack_int* n, double* ap, lapack_int *info ); -void LAPACK_cpptri( char* uplo, lapack_int* n, lapack_complex_float* ap, - lapack_int *info ); -void LAPACK_zpptri( char* uplo, lapack_int* n, lapack_complex_double* ap, - lapack_int *info ); -void LAPACK_ssytri( char* uplo, lapack_int* n, float* a, lapack_int* lda, - const lapack_int* ipiv, float* work, lapack_int *info ); -void LAPACK_dsytri( char* uplo, lapack_int* n, double* a, lapack_int* lda, - const lapack_int* ipiv, double* work, lapack_int *info ); -void LAPACK_csytri( char* uplo, lapack_int* n, lapack_complex_float* a, - lapack_int* lda, const lapack_int* ipiv, - lapack_complex_float* work, lapack_int *info ); -void LAPACK_zsytri( char* uplo, lapack_int* n, lapack_complex_double* a, - lapack_int* lda, const lapack_int* ipiv, - lapack_complex_double* work, lapack_int *info ); -void LAPACK_chetri( char* uplo, lapack_int* n, lapack_complex_float* a, - lapack_int* lda, const lapack_int* ipiv, - lapack_complex_float* work, lapack_int *info ); -void LAPACK_zhetri( char* uplo, lapack_int* n, lapack_complex_double* a, - lapack_int* lda, const lapack_int* ipiv, - lapack_complex_double* work, lapack_int *info ); -void LAPACK_ssptri( char* uplo, lapack_int* n, float* ap, - const lapack_int* ipiv, float* work, lapack_int *info ); -void LAPACK_dsptri( char* uplo, lapack_int* n, double* ap, - const lapack_int* ipiv, double* work, lapack_int *info ); -void LAPACK_csptri( char* uplo, lapack_int* n, lapack_complex_float* ap, - const lapack_int* ipiv, lapack_complex_float* work, - lapack_int *info ); -void LAPACK_zsptri( char* uplo, lapack_int* n, lapack_complex_double* ap, - const lapack_int* ipiv, lapack_complex_double* work, - lapack_int *info ); -void LAPACK_chptri( char* uplo, lapack_int* n, lapack_complex_float* ap, - const lapack_int* ipiv, lapack_complex_float* work, - lapack_int *info ); -void LAPACK_zhptri( char* uplo, lapack_int* n, lapack_complex_double* ap, - const lapack_int* ipiv, lapack_complex_double* work, - lapack_int *info ); -void LAPACK_strtri( char* uplo, char* diag, lapack_int* n, float* a, - lapack_int* lda, lapack_int *info ); -void LAPACK_dtrtri( char* uplo, char* diag, lapack_int* n, double* a, - lapack_int* lda, lapack_int *info ); -void LAPACK_ctrtri( char* uplo, char* diag, lapack_int* n, - lapack_complex_float* a, lapack_int* lda, - lapack_int *info ); -void LAPACK_ztrtri( char* uplo, char* diag, lapack_int* n, - lapack_complex_double* a, lapack_int* lda, - lapack_int *info ); -void LAPACK_dtftri( char* transr, char* uplo, char* diag, lapack_int* n, - double* a, lapack_int *info ); -void LAPACK_stftri( char* transr, char* uplo, char* diag, lapack_int* n, - float* a, lapack_int *info ); -void LAPACK_ztftri( char* transr, char* uplo, char* diag, lapack_int* n, - lapack_complex_double* a, lapack_int *info ); -void LAPACK_ctftri( char* transr, char* uplo, char* diag, lapack_int* n, - lapack_complex_float* a, lapack_int *info ); -void LAPACK_stptri( char* uplo, char* diag, lapack_int* n, float* ap, - lapack_int *info ); -void LAPACK_dtptri( char* uplo, char* diag, lapack_int* n, double* ap, - lapack_int *info ); -void LAPACK_ctptri( char* uplo, char* diag, lapack_int* n, - lapack_complex_float* ap, lapack_int *info ); -void LAPACK_ztptri( char* uplo, char* diag, lapack_int* n, - lapack_complex_double* ap, lapack_int *info ); -void LAPACK_sgeequ( lapack_int* m, lapack_int* n, const float* a, - lapack_int* lda, float* r, float* c, float* rowcnd, - float* colcnd, float* amax, lapack_int *info ); -void LAPACK_dgeequ( lapack_int* m, lapack_int* n, const double* a, - lapack_int* lda, double* r, double* c, double* rowcnd, - double* colcnd, double* amax, lapack_int *info ); -void LAPACK_cgeequ( lapack_int* m, lapack_int* n, const lapack_complex_float* a, - lapack_int* lda, float* r, float* c, float* rowcnd, - float* colcnd, float* amax, lapack_int *info ); -void LAPACK_zgeequ( lapack_int* m, lapack_int* n, - const lapack_complex_double* a, lapack_int* lda, double* r, - double* c, double* rowcnd, double* colcnd, double* amax, - lapack_int *info ); -void LAPACK_dgeequb( lapack_int* m, lapack_int* n, const double* a, - lapack_int* lda, double* r, double* c, double* rowcnd, - double* colcnd, double* amax, lapack_int *info ); -void LAPACK_sgeequb( lapack_int* m, lapack_int* n, const float* a, - lapack_int* lda, float* r, float* c, float* rowcnd, - float* colcnd, float* amax, lapack_int *info ); -void LAPACK_zgeequb( lapack_int* m, lapack_int* n, - const lapack_complex_double* a, lapack_int* lda, double* r, - double* c, double* rowcnd, double* colcnd, double* amax, - lapack_int *info ); -void LAPACK_cgeequb( lapack_int* m, lapack_int* n, - const lapack_complex_float* a, lapack_int* lda, float* r, - float* c, float* rowcnd, float* colcnd, float* amax, - lapack_int *info ); -void LAPACK_sgbequ( lapack_int* m, lapack_int* n, lapack_int* kl, - lapack_int* ku, const float* ab, lapack_int* ldab, float* r, - float* c, float* rowcnd, float* colcnd, float* amax, - lapack_int *info ); -void LAPACK_dgbequ( lapack_int* m, lapack_int* n, lapack_int* kl, - lapack_int* ku, const double* ab, lapack_int* ldab, - double* r, double* c, double* rowcnd, double* colcnd, - double* amax, lapack_int *info ); -void LAPACK_cgbequ( lapack_int* m, lapack_int* n, lapack_int* kl, - lapack_int* ku, const lapack_complex_float* ab, - lapack_int* ldab, float* r, float* c, float* rowcnd, - float* colcnd, float* amax, lapack_int *info ); -void LAPACK_zgbequ( lapack_int* m, lapack_int* n, lapack_int* kl, - lapack_int* ku, const lapack_complex_double* ab, - lapack_int* ldab, double* r, double* c, double* rowcnd, - double* colcnd, double* amax, lapack_int *info ); -void LAPACK_dgbequb( lapack_int* m, lapack_int* n, lapack_int* kl, - lapack_int* ku, const double* ab, lapack_int* ldab, - double* r, double* c, double* rowcnd, double* colcnd, - double* amax, lapack_int *info ); -void LAPACK_sgbequb( lapack_int* m, lapack_int* n, lapack_int* kl, - lapack_int* ku, const float* ab, lapack_int* ldab, - float* r, float* c, float* rowcnd, float* colcnd, - float* amax, lapack_int *info ); -void LAPACK_zgbequb( lapack_int* m, lapack_int* n, lapack_int* kl, - lapack_int* ku, const lapack_complex_double* ab, - lapack_int* ldab, double* r, double* c, double* rowcnd, - double* colcnd, double* amax, lapack_int *info ); -void LAPACK_cgbequb( lapack_int* m, lapack_int* n, lapack_int* kl, - lapack_int* ku, const lapack_complex_float* ab, - lapack_int* ldab, float* r, float* c, float* rowcnd, - float* colcnd, float* amax, lapack_int *info ); -void LAPACK_spoequ( lapack_int* n, const float* a, lapack_int* lda, float* s, - float* scond, float* amax, lapack_int *info ); -void LAPACK_dpoequ( lapack_int* n, const double* a, lapack_int* lda, double* s, - double* scond, double* amax, lapack_int *info ); -void LAPACK_cpoequ( lapack_int* n, const lapack_complex_float* a, - lapack_int* lda, float* s, float* scond, float* amax, - lapack_int *info ); -void LAPACK_zpoequ( lapack_int* n, const lapack_complex_double* a, - lapack_int* lda, double* s, double* scond, double* amax, - lapack_int *info ); -void LAPACK_dpoequb( lapack_int* n, const double* a, lapack_int* lda, double* s, - double* scond, double* amax, lapack_int *info ); -void LAPACK_spoequb( lapack_int* n, const float* a, lapack_int* lda, float* s, - float* scond, float* amax, lapack_int *info ); -void LAPACK_zpoequb( lapack_int* n, const lapack_complex_double* a, - lapack_int* lda, double* s, double* scond, double* amax, - lapack_int *info ); -void LAPACK_cpoequb( lapack_int* n, const lapack_complex_float* a, - lapack_int* lda, float* s, float* scond, float* amax, - lapack_int *info ); -void LAPACK_sppequ( char* uplo, lapack_int* n, const float* ap, float* s, - float* scond, float* amax, lapack_int *info ); -void LAPACK_dppequ( char* uplo, lapack_int* n, const double* ap, double* s, - double* scond, double* amax, lapack_int *info ); -void LAPACK_cppequ( char* uplo, lapack_int* n, const lapack_complex_float* ap, - float* s, float* scond, float* amax, lapack_int *info ); -void LAPACK_zppequ( char* uplo, lapack_int* n, const lapack_complex_double* ap, - double* s, double* scond, double* amax, lapack_int *info ); -void LAPACK_spbequ( char* uplo, lapack_int* n, lapack_int* kd, const float* ab, - lapack_int* ldab, float* s, float* scond, float* amax, - lapack_int *info ); -void LAPACK_dpbequ( char* uplo, lapack_int* n, lapack_int* kd, const double* ab, - lapack_int* ldab, double* s, double* scond, double* amax, - lapack_int *info ); -void LAPACK_cpbequ( char* uplo, lapack_int* n, lapack_int* kd, - const lapack_complex_float* ab, lapack_int* ldab, float* s, - float* scond, float* amax, lapack_int *info ); -void LAPACK_zpbequ( char* uplo, lapack_int* n, lapack_int* kd, - const lapack_complex_double* ab, lapack_int* ldab, - double* s, double* scond, double* amax, lapack_int *info ); -void LAPACK_dsyequb( char* uplo, lapack_int* n, const double* a, - lapack_int* lda, double* s, double* scond, double* amax, - double* work, lapack_int *info ); -void LAPACK_ssyequb( char* uplo, lapack_int* n, const float* a, lapack_int* lda, - float* s, float* scond, float* amax, float* work, - lapack_int *info ); -void LAPACK_zsyequb( char* uplo, lapack_int* n, const lapack_complex_double* a, - lapack_int* lda, double* s, double* scond, double* amax, - lapack_complex_double* work, lapack_int *info ); -void LAPACK_csyequb( char* uplo, lapack_int* n, const lapack_complex_float* a, - lapack_int* lda, float* s, float* scond, float* amax, - lapack_complex_float* work, lapack_int *info ); -void LAPACK_zheequb( char* uplo, lapack_int* n, const lapack_complex_double* a, - lapack_int* lda, double* s, double* scond, double* amax, - lapack_complex_double* work, lapack_int *info ); -void LAPACK_cheequb( char* uplo, lapack_int* n, const lapack_complex_float* a, - lapack_int* lda, float* s, float* scond, float* amax, - lapack_complex_float* work, lapack_int *info ); -void LAPACK_sgesv( lapack_int* n, lapack_int* nrhs, float* a, lapack_int* lda, - lapack_int* ipiv, float* b, lapack_int* ldb, - lapack_int *info ); -void LAPACK_dgesv( lapack_int* n, lapack_int* nrhs, double* a, lapack_int* lda, - lapack_int* ipiv, double* b, lapack_int* ldb, - lapack_int *info ); -void LAPACK_cgesv( lapack_int* n, lapack_int* nrhs, lapack_complex_float* a, - lapack_int* lda, lapack_int* ipiv, lapack_complex_float* b, - lapack_int* ldb, lapack_int *info ); -void LAPACK_zgesv( lapack_int* n, lapack_int* nrhs, lapack_complex_double* a, - lapack_int* lda, lapack_int* ipiv, lapack_complex_double* b, - lapack_int* ldb, lapack_int *info ); -void LAPACK_dsgesv( lapack_int* n, lapack_int* nrhs, double* a, lapack_int* lda, - lapack_int* ipiv, double* b, lapack_int* ldb, double* x, - lapack_int* ldx, double* work, float* swork, - lapack_int* iter, lapack_int *info ); -void LAPACK_zcgesv( lapack_int* n, lapack_int* nrhs, lapack_complex_double* a, - lapack_int* lda, lapack_int* ipiv, lapack_complex_double* b, - lapack_int* ldb, lapack_complex_double* x, lapack_int* ldx, - lapack_complex_double* work, lapack_complex_float* swork, - double* rwork, lapack_int* iter, lapack_int *info ); -void LAPACK_sgesvx( char* fact, char* trans, lapack_int* n, lapack_int* nrhs, - float* a, lapack_int* lda, float* af, lapack_int* ldaf, - lapack_int* ipiv, char* equed, float* r, float* c, float* b, - lapack_int* ldb, float* x, lapack_int* ldx, float* rcond, - float* ferr, float* berr, float* work, lapack_int* iwork, - lapack_int *info ); -void LAPACK_dgesvx( char* fact, char* trans, lapack_int* n, lapack_int* nrhs, - double* a, lapack_int* lda, double* af, lapack_int* ldaf, - lapack_int* ipiv, char* equed, double* r, double* c, - double* b, lapack_int* ldb, double* x, lapack_int* ldx, - double* rcond, double* ferr, double* berr, double* work, - lapack_int* iwork, lapack_int *info ); -void LAPACK_cgesvx( char* fact, char* trans, lapack_int* n, lapack_int* nrhs, - lapack_complex_float* a, lapack_int* lda, - lapack_complex_float* af, lapack_int* ldaf, - lapack_int* ipiv, char* equed, float* r, float* c, - lapack_complex_float* b, lapack_int* ldb, - lapack_complex_float* x, lapack_int* ldx, float* rcond, - float* ferr, float* berr, lapack_complex_float* work, - float* rwork, lapack_int *info ); -void LAPACK_zgesvx( char* fact, char* trans, lapack_int* n, lapack_int* nrhs, - lapack_complex_double* a, lapack_int* lda, - lapack_complex_double* af, lapack_int* ldaf, - lapack_int* ipiv, char* equed, double* r, double* c, - lapack_complex_double* b, lapack_int* ldb, - lapack_complex_double* x, lapack_int* ldx, double* rcond, - double* ferr, double* berr, lapack_complex_double* work, - double* rwork, lapack_int *info ); -void LAPACK_dgesvxx( char* fact, char* trans, lapack_int* n, lapack_int* nrhs, - double* a, lapack_int* lda, double* af, lapack_int* ldaf, - lapack_int* ipiv, char* equed, double* r, double* c, - double* b, lapack_int* ldb, double* x, lapack_int* ldx, - double* rcond, double* rpvgrw, double* berr, - lapack_int* n_err_bnds, double* err_bnds_norm, - double* err_bnds_comp, lapack_int* nparams, double* params, - double* work, lapack_int* iwork, lapack_int *info ); -void LAPACK_sgesvxx( char* fact, char* trans, lapack_int* n, lapack_int* nrhs, - float* a, lapack_int* lda, float* af, lapack_int* ldaf, - lapack_int* ipiv, char* equed, float* r, float* c, - float* b, lapack_int* ldb, float* x, lapack_int* ldx, - float* rcond, float* rpvgrw, float* berr, - lapack_int* n_err_bnds, float* err_bnds_norm, - float* err_bnds_comp, lapack_int* nparams, float* params, - float* work, lapack_int* iwork, lapack_int *info ); -void LAPACK_zgesvxx( char* fact, char* trans, lapack_int* n, lapack_int* nrhs, - lapack_complex_double* a, lapack_int* lda, - lapack_complex_double* af, lapack_int* ldaf, - lapack_int* ipiv, char* equed, double* r, double* c, - lapack_complex_double* b, lapack_int* ldb, - lapack_complex_double* x, lapack_int* ldx, double* rcond, - double* rpvgrw, double* berr, lapack_int* n_err_bnds, - double* err_bnds_norm, double* err_bnds_comp, - lapack_int* nparams, double* params, - lapack_complex_double* work, double* rwork, - lapack_int *info ); -void LAPACK_cgesvxx( char* fact, char* trans, lapack_int* n, lapack_int* nrhs, - lapack_complex_float* a, lapack_int* lda, - lapack_complex_float* af, lapack_int* ldaf, - lapack_int* ipiv, char* equed, float* r, float* c, - lapack_complex_float* b, lapack_int* ldb, - lapack_complex_float* x, lapack_int* ldx, float* rcond, - float* rpvgrw, float* berr, lapack_int* n_err_bnds, - float* err_bnds_norm, float* err_bnds_comp, - lapack_int* nparams, float* params, - lapack_complex_float* work, float* rwork, - lapack_int *info ); -void LAPACK_sgbsv( lapack_int* n, lapack_int* kl, lapack_int* ku, - lapack_int* nrhs, float* ab, lapack_int* ldab, - lapack_int* ipiv, float* b, lapack_int* ldb, - lapack_int *info ); -void LAPACK_dgbsv( lapack_int* n, lapack_int* kl, lapack_int* ku, - lapack_int* nrhs, double* ab, lapack_int* ldab, - lapack_int* ipiv, double* b, lapack_int* ldb, - lapack_int *info ); -void LAPACK_cgbsv( lapack_int* n, lapack_int* kl, lapack_int* ku, - lapack_int* nrhs, lapack_complex_float* ab, lapack_int* ldab, - lapack_int* ipiv, lapack_complex_float* b, lapack_int* ldb, - lapack_int *info ); -void LAPACK_zgbsv( lapack_int* n, lapack_int* kl, lapack_int* ku, - lapack_int* nrhs, lapack_complex_double* ab, - lapack_int* ldab, lapack_int* ipiv, lapack_complex_double* b, - lapack_int* ldb, lapack_int *info ); -void LAPACK_sgbsvx( char* fact, char* trans, lapack_int* n, lapack_int* kl, - lapack_int* ku, lapack_int* nrhs, float* ab, - lapack_int* ldab, float* afb, lapack_int* ldafb, - lapack_int* ipiv, char* equed, float* r, float* c, float* b, - lapack_int* ldb, float* x, lapack_int* ldx, float* rcond, - float* ferr, float* berr, float* work, lapack_int* iwork, - lapack_int *info ); -void LAPACK_dgbsvx( char* fact, char* trans, lapack_int* n, lapack_int* kl, - lapack_int* ku, lapack_int* nrhs, double* ab, - lapack_int* ldab, double* afb, lapack_int* ldafb, - lapack_int* ipiv, char* equed, double* r, double* c, - double* b, lapack_int* ldb, double* x, lapack_int* ldx, - double* rcond, double* ferr, double* berr, double* work, - lapack_int* iwork, lapack_int *info ); -void LAPACK_cgbsvx( char* fact, char* trans, lapack_int* n, lapack_int* kl, - lapack_int* ku, lapack_int* nrhs, lapack_complex_float* ab, - lapack_int* ldab, lapack_complex_float* afb, - lapack_int* ldafb, lapack_int* ipiv, char* equed, float* r, - float* c, lapack_complex_float* b, lapack_int* ldb, - lapack_complex_float* x, lapack_int* ldx, float* rcond, - float* ferr, float* berr, lapack_complex_float* work, - float* rwork, lapack_int *info ); -void LAPACK_zgbsvx( char* fact, char* trans, lapack_int* n, lapack_int* kl, - lapack_int* ku, lapack_int* nrhs, lapack_complex_double* ab, - lapack_int* ldab, lapack_complex_double* afb, - lapack_int* ldafb, lapack_int* ipiv, char* equed, double* r, - double* c, lapack_complex_double* b, lapack_int* ldb, - lapack_complex_double* x, lapack_int* ldx, double* rcond, - double* ferr, double* berr, lapack_complex_double* work, - double* rwork, lapack_int *info ); -void LAPACK_dgbsvxx( char* fact, char* trans, lapack_int* n, lapack_int* kl, - lapack_int* ku, lapack_int* nrhs, double* ab, - lapack_int* ldab, double* afb, lapack_int* ldafb, - lapack_int* ipiv, char* equed, double* r, double* c, - double* b, lapack_int* ldb, double* x, lapack_int* ldx, - double* rcond, double* rpvgrw, double* berr, - lapack_int* n_err_bnds, double* err_bnds_norm, - double* err_bnds_comp, lapack_int* nparams, double* params, - double* work, lapack_int* iwork, lapack_int *info ); -void LAPACK_sgbsvxx( char* fact, char* trans, lapack_int* n, lapack_int* kl, - lapack_int* ku, lapack_int* nrhs, float* ab, - lapack_int* ldab, float* afb, lapack_int* ldafb, - lapack_int* ipiv, char* equed, float* r, float* c, - float* b, lapack_int* ldb, float* x, lapack_int* ldx, - float* rcond, float* rpvgrw, float* berr, - lapack_int* n_err_bnds, float* err_bnds_norm, - float* err_bnds_comp, lapack_int* nparams, float* params, - float* work, lapack_int* iwork, lapack_int *info ); -void LAPACK_zgbsvxx( char* fact, char* trans, lapack_int* n, lapack_int* kl, - lapack_int* ku, lapack_int* nrhs, - lapack_complex_double* ab, lapack_int* ldab, - lapack_complex_double* afb, lapack_int* ldafb, - lapack_int* ipiv, char* equed, double* r, double* c, - lapack_complex_double* b, lapack_int* ldb, - lapack_complex_double* x, lapack_int* ldx, double* rcond, - double* rpvgrw, double* berr, lapack_int* n_err_bnds, - double* err_bnds_norm, double* err_bnds_comp, - lapack_int* nparams, double* params, - lapack_complex_double* work, double* rwork, - lapack_int *info ); -void LAPACK_cgbsvxx( char* fact, char* trans, lapack_int* n, lapack_int* kl, - lapack_int* ku, lapack_int* nrhs, lapack_complex_float* ab, - lapack_int* ldab, lapack_complex_float* afb, - lapack_int* ldafb, lapack_int* ipiv, char* equed, float* r, - float* c, lapack_complex_float* b, lapack_int* ldb, - lapack_complex_float* x, lapack_int* ldx, float* rcond, - float* rpvgrw, float* berr, lapack_int* n_err_bnds, - float* err_bnds_norm, float* err_bnds_comp, - lapack_int* nparams, float* params, - lapack_complex_float* work, float* rwork, - lapack_int *info ); -void LAPACK_sgtsv( lapack_int* n, lapack_int* nrhs, float* dl, float* d, - float* du, float* b, lapack_int* ldb, lapack_int *info ); -void LAPACK_dgtsv( lapack_int* n, lapack_int* nrhs, double* dl, double* d, - double* du, double* b, lapack_int* ldb, lapack_int *info ); -void LAPACK_cgtsv( lapack_int* n, lapack_int* nrhs, lapack_complex_float* dl, - lapack_complex_float* d, lapack_complex_float* du, - lapack_complex_float* b, lapack_int* ldb, lapack_int *info ); -void LAPACK_zgtsv( lapack_int* n, lapack_int* nrhs, lapack_complex_double* dl, - lapack_complex_double* d, lapack_complex_double* du, - lapack_complex_double* b, lapack_int* ldb, - lapack_int *info ); -void LAPACK_sgtsvx( char* fact, char* trans, lapack_int* n, lapack_int* nrhs, - const float* dl, const float* d, const float* du, - float* dlf, float* df, float* duf, float* du2, - lapack_int* ipiv, const float* b, lapack_int* ldb, float* x, - lapack_int* ldx, float* rcond, float* ferr, float* berr, - float* work, lapack_int* iwork, lapack_int *info ); -void LAPACK_dgtsvx( char* fact, char* trans, lapack_int* n, lapack_int* nrhs, - const double* dl, const double* d, const double* du, - double* dlf, double* df, double* duf, double* du2, - lapack_int* ipiv, const double* b, lapack_int* ldb, - double* x, lapack_int* ldx, double* rcond, double* ferr, - double* berr, double* work, lapack_int* iwork, - lapack_int *info ); -void LAPACK_cgtsvx( char* fact, char* trans, lapack_int* n, lapack_int* nrhs, - const lapack_complex_float* dl, - const lapack_complex_float* d, - const lapack_complex_float* du, lapack_complex_float* dlf, - lapack_complex_float* df, lapack_complex_float* duf, - lapack_complex_float* du2, lapack_int* ipiv, - const lapack_complex_float* b, lapack_int* ldb, - lapack_complex_float* x, lapack_int* ldx, float* rcond, - float* ferr, float* berr, lapack_complex_float* work, - float* rwork, lapack_int *info ); -void LAPACK_zgtsvx( char* fact, char* trans, lapack_int* n, lapack_int* nrhs, - const lapack_complex_double* dl, - const lapack_complex_double* d, - const lapack_complex_double* du, lapack_complex_double* dlf, - lapack_complex_double* df, lapack_complex_double* duf, - lapack_complex_double* du2, lapack_int* ipiv, - const lapack_complex_double* b, lapack_int* ldb, - lapack_complex_double* x, lapack_int* ldx, double* rcond, - double* ferr, double* berr, lapack_complex_double* work, - double* rwork, lapack_int *info ); -void LAPACK_sposv( char* uplo, lapack_int* n, lapack_int* nrhs, float* a, - lapack_int* lda, float* b, lapack_int* ldb, - lapack_int *info ); -void LAPACK_dposv( char* uplo, lapack_int* n, lapack_int* nrhs, double* a, - lapack_int* lda, double* b, lapack_int* ldb, - lapack_int *info ); -void LAPACK_cposv( char* uplo, lapack_int* n, lapack_int* nrhs, - lapack_complex_float* a, lapack_int* lda, - lapack_complex_float* b, lapack_int* ldb, lapack_int *info ); -void LAPACK_zposv( char* uplo, lapack_int* n, lapack_int* nrhs, - lapack_complex_double* a, lapack_int* lda, - lapack_complex_double* b, lapack_int* ldb, - lapack_int *info ); -void LAPACK_dsposv( char* uplo, lapack_int* n, lapack_int* nrhs, double* a, - lapack_int* lda, double* b, lapack_int* ldb, double* x, - lapack_int* ldx, double* work, float* swork, - lapack_int* iter, lapack_int *info ); -void LAPACK_zcposv( char* uplo, lapack_int* n, lapack_int* nrhs, - lapack_complex_double* a, lapack_int* lda, - lapack_complex_double* b, lapack_int* ldb, - lapack_complex_double* x, lapack_int* ldx, - lapack_complex_double* work, lapack_complex_float* swork, - double* rwork, lapack_int* iter, lapack_int *info ); -void LAPACK_sposvx( char* fact, char* uplo, lapack_int* n, lapack_int* nrhs, - float* a, lapack_int* lda, float* af, lapack_int* ldaf, - char* equed, float* s, float* b, lapack_int* ldb, float* x, - lapack_int* ldx, float* rcond, float* ferr, float* berr, - float* work, lapack_int* iwork, lapack_int *info ); -void LAPACK_dposvx( char* fact, char* uplo, lapack_int* n, lapack_int* nrhs, - double* a, lapack_int* lda, double* af, lapack_int* ldaf, - char* equed, double* s, double* b, lapack_int* ldb, - double* x, lapack_int* ldx, double* rcond, double* ferr, - double* berr, double* work, lapack_int* iwork, - lapack_int *info ); -void LAPACK_cposvx( char* fact, char* uplo, lapack_int* n, lapack_int* nrhs, - lapack_complex_float* a, lapack_int* lda, - lapack_complex_float* af, lapack_int* ldaf, char* equed, - float* s, lapack_complex_float* b, lapack_int* ldb, - lapack_complex_float* x, lapack_int* ldx, float* rcond, - float* ferr, float* berr, lapack_complex_float* work, - float* rwork, lapack_int *info ); -void LAPACK_zposvx( char* fact, char* uplo, lapack_int* n, lapack_int* nrhs, - lapack_complex_double* a, lapack_int* lda, - lapack_complex_double* af, lapack_int* ldaf, char* equed, - double* s, lapack_complex_double* b, lapack_int* ldb, - lapack_complex_double* x, lapack_int* ldx, double* rcond, - double* ferr, double* berr, lapack_complex_double* work, - double* rwork, lapack_int *info ); -void LAPACK_dposvxx( char* fact, char* uplo, lapack_int* n, lapack_int* nrhs, - double* a, lapack_int* lda, double* af, lapack_int* ldaf, - char* equed, double* s, double* b, lapack_int* ldb, - double* x, lapack_int* ldx, double* rcond, double* rpvgrw, - double* berr, lapack_int* n_err_bnds, - double* err_bnds_norm, double* err_bnds_comp, - lapack_int* nparams, double* params, double* work, - lapack_int* iwork, lapack_int *info ); -void LAPACK_sposvxx( char* fact, char* uplo, lapack_int* n, lapack_int* nrhs, - float* a, lapack_int* lda, float* af, lapack_int* ldaf, - char* equed, float* s, float* b, lapack_int* ldb, float* x, - lapack_int* ldx, float* rcond, float* rpvgrw, float* berr, - lapack_int* n_err_bnds, float* err_bnds_norm, - float* err_bnds_comp, lapack_int* nparams, float* params, - float* work, lapack_int* iwork, lapack_int *info ); -void LAPACK_zposvxx( char* fact, char* uplo, lapack_int* n, lapack_int* nrhs, - lapack_complex_double* a, lapack_int* lda, - lapack_complex_double* af, lapack_int* ldaf, char* equed, - double* s, lapack_complex_double* b, lapack_int* ldb, - lapack_complex_double* x, lapack_int* ldx, double* rcond, - double* rpvgrw, double* berr, lapack_int* n_err_bnds, - double* err_bnds_norm, double* err_bnds_comp, - lapack_int* nparams, double* params, - lapack_complex_double* work, double* rwork, - lapack_int *info ); -void LAPACK_cposvxx( char* fact, char* uplo, lapack_int* n, lapack_int* nrhs, - lapack_complex_float* a, lapack_int* lda, - lapack_complex_float* af, lapack_int* ldaf, char* equed, - float* s, lapack_complex_float* b, lapack_int* ldb, - lapack_complex_float* x, lapack_int* ldx, float* rcond, - float* rpvgrw, float* berr, lapack_int* n_err_bnds, - float* err_bnds_norm, float* err_bnds_comp, - lapack_int* nparams, float* params, - lapack_complex_float* work, float* rwork, - lapack_int *info ); -void LAPACK_sppsv( char* uplo, lapack_int* n, lapack_int* nrhs, float* ap, - float* b, lapack_int* ldb, lapack_int *info ); -void LAPACK_dppsv( char* uplo, lapack_int* n, lapack_int* nrhs, double* ap, - double* b, lapack_int* ldb, lapack_int *info ); -void LAPACK_cppsv( char* uplo, lapack_int* n, lapack_int* nrhs, - lapack_complex_float* ap, lapack_complex_float* b, - lapack_int* ldb, lapack_int *info ); -void LAPACK_zppsv( char* uplo, lapack_int* n, lapack_int* nrhs, - lapack_complex_double* ap, lapack_complex_double* b, - lapack_int* ldb, lapack_int *info ); -void LAPACK_sppsvx( char* fact, char* uplo, lapack_int* n, lapack_int* nrhs, - float* ap, float* afp, char* equed, float* s, float* b, - lapack_int* ldb, float* x, lapack_int* ldx, float* rcond, - float* ferr, float* berr, float* work, lapack_int* iwork, - lapack_int *info ); -void LAPACK_dppsvx( char* fact, char* uplo, lapack_int* n, lapack_int* nrhs, - double* ap, double* afp, char* equed, double* s, double* b, - lapack_int* ldb, double* x, lapack_int* ldx, double* rcond, - double* ferr, double* berr, double* work, lapack_int* iwork, - lapack_int *info ); -void LAPACK_cppsvx( char* fact, char* uplo, lapack_int* n, lapack_int* nrhs, - lapack_complex_float* ap, lapack_complex_float* afp, - char* equed, float* s, lapack_complex_float* b, - lapack_int* ldb, lapack_complex_float* x, lapack_int* ldx, - float* rcond, float* ferr, float* berr, - lapack_complex_float* work, float* rwork, - lapack_int *info ); -void LAPACK_zppsvx( char* fact, char* uplo, lapack_int* n, lapack_int* nrhs, - lapack_complex_double* ap, lapack_complex_double* afp, - char* equed, double* s, lapack_complex_double* b, - lapack_int* ldb, lapack_complex_double* x, lapack_int* ldx, - double* rcond, double* ferr, double* berr, - lapack_complex_double* work, double* rwork, - lapack_int *info ); -void LAPACK_spbsv( char* uplo, lapack_int* n, lapack_int* kd, lapack_int* nrhs, - float* ab, lapack_int* ldab, float* b, lapack_int* ldb, - lapack_int *info ); -void LAPACK_dpbsv( char* uplo, lapack_int* n, lapack_int* kd, lapack_int* nrhs, - double* ab, lapack_int* ldab, double* b, lapack_int* ldb, - lapack_int *info ); -void LAPACK_cpbsv( char* uplo, lapack_int* n, lapack_int* kd, lapack_int* nrhs, - lapack_complex_float* ab, lapack_int* ldab, - lapack_complex_float* b, lapack_int* ldb, lapack_int *info ); -void LAPACK_zpbsv( char* uplo, lapack_int* n, lapack_int* kd, lapack_int* nrhs, - lapack_complex_double* ab, lapack_int* ldab, - lapack_complex_double* b, lapack_int* ldb, - lapack_int *info ); -void LAPACK_spbsvx( char* fact, char* uplo, lapack_int* n, lapack_int* kd, - lapack_int* nrhs, float* ab, lapack_int* ldab, float* afb, - lapack_int* ldafb, char* equed, float* s, float* b, - lapack_int* ldb, float* x, lapack_int* ldx, float* rcond, - float* ferr, float* berr, float* work, lapack_int* iwork, - lapack_int *info ); -void LAPACK_dpbsvx( char* fact, char* uplo, lapack_int* n, lapack_int* kd, - lapack_int* nrhs, double* ab, lapack_int* ldab, double* afb, - lapack_int* ldafb, char* equed, double* s, double* b, - lapack_int* ldb, double* x, lapack_int* ldx, double* rcond, - double* ferr, double* berr, double* work, lapack_int* iwork, - lapack_int *info ); -void LAPACK_cpbsvx( char* fact, char* uplo, lapack_int* n, lapack_int* kd, - lapack_int* nrhs, lapack_complex_float* ab, - lapack_int* ldab, lapack_complex_float* afb, - lapack_int* ldafb, char* equed, float* s, - lapack_complex_float* b, lapack_int* ldb, - lapack_complex_float* x, lapack_int* ldx, float* rcond, - float* ferr, float* berr, lapack_complex_float* work, - float* rwork, lapack_int *info ); -void LAPACK_zpbsvx( char* fact, char* uplo, lapack_int* n, lapack_int* kd, - lapack_int* nrhs, lapack_complex_double* ab, - lapack_int* ldab, lapack_complex_double* afb, - lapack_int* ldafb, char* equed, double* s, - lapack_complex_double* b, lapack_int* ldb, - lapack_complex_double* x, lapack_int* ldx, double* rcond, - double* ferr, double* berr, lapack_complex_double* work, - double* rwork, lapack_int *info ); -void LAPACK_sptsv( lapack_int* n, lapack_int* nrhs, float* d, float* e, - float* b, lapack_int* ldb, lapack_int *info ); -void LAPACK_dptsv( lapack_int* n, lapack_int* nrhs, double* d, double* e, - double* b, lapack_int* ldb, lapack_int *info ); -void LAPACK_cptsv( lapack_int* n, lapack_int* nrhs, float* d, - lapack_complex_float* e, lapack_complex_float* b, - lapack_int* ldb, lapack_int *info ); -void LAPACK_zptsv( lapack_int* n, lapack_int* nrhs, double* d, - lapack_complex_double* e, lapack_complex_double* b, - lapack_int* ldb, lapack_int *info ); -void LAPACK_sptsvx( char* fact, lapack_int* n, lapack_int* nrhs, const float* d, - const float* e, float* df, float* ef, const float* b, - lapack_int* ldb, float* x, lapack_int* ldx, float* rcond, - float* ferr, float* berr, float* work, lapack_int *info ); -void LAPACK_dptsvx( char* fact, lapack_int* n, lapack_int* nrhs, - const double* d, const double* e, double* df, double* ef, - const double* b, lapack_int* ldb, double* x, - lapack_int* ldx, double* rcond, double* ferr, double* berr, - double* work, lapack_int *info ); -void LAPACK_cptsvx( char* fact, lapack_int* n, lapack_int* nrhs, const float* d, - const lapack_complex_float* e, float* df, - lapack_complex_float* ef, const lapack_complex_float* b, - lapack_int* ldb, lapack_complex_float* x, lapack_int* ldx, - float* rcond, float* ferr, float* berr, - lapack_complex_float* work, float* rwork, - lapack_int *info ); -void LAPACK_zptsvx( char* fact, lapack_int* n, lapack_int* nrhs, - const double* d, const lapack_complex_double* e, double* df, - lapack_complex_double* ef, const lapack_complex_double* b, - lapack_int* ldb, lapack_complex_double* x, lapack_int* ldx, - double* rcond, double* ferr, double* berr, - lapack_complex_double* work, double* rwork, - lapack_int *info ); -void LAPACK_ssysv( char* uplo, lapack_int* n, lapack_int* nrhs, float* a, - lapack_int* lda, lapack_int* ipiv, float* b, lapack_int* ldb, - float* work, lapack_int* lwork, lapack_int *info ); -void LAPACK_dsysv( char* uplo, lapack_int* n, lapack_int* nrhs, double* a, - lapack_int* lda, lapack_int* ipiv, double* b, - lapack_int* ldb, double* work, lapack_int* lwork, - lapack_int *info ); -void LAPACK_csysv( char* uplo, lapack_int* n, lapack_int* nrhs, - lapack_complex_float* a, lapack_int* lda, lapack_int* ipiv, - lapack_complex_float* b, lapack_int* ldb, - lapack_complex_float* work, lapack_int* lwork, - lapack_int *info ); -void LAPACK_zsysv( char* uplo, lapack_int* n, lapack_int* nrhs, - lapack_complex_double* a, lapack_int* lda, lapack_int* ipiv, - lapack_complex_double* b, lapack_int* ldb, - lapack_complex_double* work, lapack_int* lwork, - lapack_int *info ); -void LAPACK_ssysvx( char* fact, char* uplo, lapack_int* n, lapack_int* nrhs, - const float* a, lapack_int* lda, float* af, - lapack_int* ldaf, lapack_int* ipiv, const float* b, - lapack_int* ldb, float* x, lapack_int* ldx, float* rcond, - float* ferr, float* berr, float* work, lapack_int* lwork, - lapack_int* iwork, lapack_int *info ); -void LAPACK_dsysvx( char* fact, char* uplo, lapack_int* n, lapack_int* nrhs, - const double* a, lapack_int* lda, double* af, - lapack_int* ldaf, lapack_int* ipiv, const double* b, - lapack_int* ldb, double* x, lapack_int* ldx, double* rcond, - double* ferr, double* berr, double* work, lapack_int* lwork, - lapack_int* iwork, lapack_int *info ); -void LAPACK_csysvx( char* fact, char* uplo, lapack_int* n, lapack_int* nrhs, - const lapack_complex_float* a, lapack_int* lda, - lapack_complex_float* af, lapack_int* ldaf, - lapack_int* ipiv, const lapack_complex_float* b, - lapack_int* ldb, lapack_complex_float* x, lapack_int* ldx, - float* rcond, float* ferr, float* berr, - lapack_complex_float* work, lapack_int* lwork, float* rwork, - lapack_int *info ); -void LAPACK_zsysvx( char* fact, char* uplo, lapack_int* n, lapack_int* nrhs, - const lapack_complex_double* a, lapack_int* lda, - lapack_complex_double* af, lapack_int* ldaf, - lapack_int* ipiv, const lapack_complex_double* b, - lapack_int* ldb, lapack_complex_double* x, lapack_int* ldx, - double* rcond, double* ferr, double* berr, - lapack_complex_double* work, lapack_int* lwork, - double* rwork, lapack_int *info ); -void LAPACK_dsysvxx( char* fact, char* uplo, lapack_int* n, lapack_int* nrhs, - double* a, lapack_int* lda, double* af, lapack_int* ldaf, - lapack_int* ipiv, char* equed, double* s, double* b, - lapack_int* ldb, double* x, lapack_int* ldx, double* rcond, - double* rpvgrw, double* berr, lapack_int* n_err_bnds, - double* err_bnds_norm, double* err_bnds_comp, - lapack_int* nparams, double* params, double* work, - lapack_int* iwork, lapack_int *info ); -void LAPACK_ssysvxx( char* fact, char* uplo, lapack_int* n, lapack_int* nrhs, - float* a, lapack_int* lda, float* af, lapack_int* ldaf, - lapack_int* ipiv, char* equed, float* s, float* b, - lapack_int* ldb, float* x, lapack_int* ldx, float* rcond, - float* rpvgrw, float* berr, lapack_int* n_err_bnds, - float* err_bnds_norm, float* err_bnds_comp, - lapack_int* nparams, float* params, float* work, - lapack_int* iwork, lapack_int *info ); -void LAPACK_zsysvxx( char* fact, char* uplo, lapack_int* n, lapack_int* nrhs, - lapack_complex_double* a, lapack_int* lda, - lapack_complex_double* af, lapack_int* ldaf, - lapack_int* ipiv, char* equed, double* s, - lapack_complex_double* b, lapack_int* ldb, - lapack_complex_double* x, lapack_int* ldx, double* rcond, - double* rpvgrw, double* berr, lapack_int* n_err_bnds, - double* err_bnds_norm, double* err_bnds_comp, - lapack_int* nparams, double* params, - lapack_complex_double* work, double* rwork, - lapack_int *info ); -void LAPACK_csysvxx( char* fact, char* uplo, lapack_int* n, lapack_int* nrhs, - lapack_complex_float* a, lapack_int* lda, - lapack_complex_float* af, lapack_int* ldaf, - lapack_int* ipiv, char* equed, float* s, - lapack_complex_float* b, lapack_int* ldb, - lapack_complex_float* x, lapack_int* ldx, float* rcond, - float* rpvgrw, float* berr, lapack_int* n_err_bnds, - float* err_bnds_norm, float* err_bnds_comp, - lapack_int* nparams, float* params, - lapack_complex_float* work, float* rwork, - lapack_int *info ); -void LAPACK_chesv( char* uplo, lapack_int* n, lapack_int* nrhs, - lapack_complex_float* a, lapack_int* lda, lapack_int* ipiv, - lapack_complex_float* b, lapack_int* ldb, - lapack_complex_float* work, lapack_int* lwork, - lapack_int *info ); -void LAPACK_zhesv( char* uplo, lapack_int* n, lapack_int* nrhs, - lapack_complex_double* a, lapack_int* lda, lapack_int* ipiv, - lapack_complex_double* b, lapack_int* ldb, - lapack_complex_double* work, lapack_int* lwork, - lapack_int *info ); -void LAPACK_chesvx( char* fact, char* uplo, lapack_int* n, lapack_int* nrhs, - const lapack_complex_float* a, lapack_int* lda, - lapack_complex_float* af, lapack_int* ldaf, - lapack_int* ipiv, const lapack_complex_float* b, - lapack_int* ldb, lapack_complex_float* x, lapack_int* ldx, - float* rcond, float* ferr, float* berr, - lapack_complex_float* work, lapack_int* lwork, float* rwork, - lapack_int *info ); -void LAPACK_zhesvx( char* fact, char* uplo, lapack_int* n, lapack_int* nrhs, - const lapack_complex_double* a, lapack_int* lda, - lapack_complex_double* af, lapack_int* ldaf, - lapack_int* ipiv, const lapack_complex_double* b, - lapack_int* ldb, lapack_complex_double* x, lapack_int* ldx, - double* rcond, double* ferr, double* berr, - lapack_complex_double* work, lapack_int* lwork, - double* rwork, lapack_int *info ); -void LAPACK_zhesvxx( char* fact, char* uplo, lapack_int* n, lapack_int* nrhs, - lapack_complex_double* a, lapack_int* lda, - lapack_complex_double* af, lapack_int* ldaf, - lapack_int* ipiv, char* equed, double* s, - lapack_complex_double* b, lapack_int* ldb, - lapack_complex_double* x, lapack_int* ldx, double* rcond, - double* rpvgrw, double* berr, lapack_int* n_err_bnds, - double* err_bnds_norm, double* err_bnds_comp, - lapack_int* nparams, double* params, - lapack_complex_double* work, double* rwork, - lapack_int *info ); -void LAPACK_chesvxx( char* fact, char* uplo, lapack_int* n, lapack_int* nrhs, - lapack_complex_float* a, lapack_int* lda, - lapack_complex_float* af, lapack_int* ldaf, - lapack_int* ipiv, char* equed, float* s, - lapack_complex_float* b, lapack_int* ldb, - lapack_complex_float* x, lapack_int* ldx, float* rcond, - float* rpvgrw, float* berr, lapack_int* n_err_bnds, - float* err_bnds_norm, float* err_bnds_comp, - lapack_int* nparams, float* params, - lapack_complex_float* work, float* rwork, - lapack_int *info ); -void LAPACK_sspsv( char* uplo, lapack_int* n, lapack_int* nrhs, float* ap, - lapack_int* ipiv, float* b, lapack_int* ldb, - lapack_int *info ); -void LAPACK_dspsv( char* uplo, lapack_int* n, lapack_int* nrhs, double* ap, - lapack_int* ipiv, double* b, lapack_int* ldb, - lapack_int *info ); -void LAPACK_cspsv( char* uplo, lapack_int* n, lapack_int* nrhs, - lapack_complex_float* ap, lapack_int* ipiv, - lapack_complex_float* b, lapack_int* ldb, lapack_int *info ); -void LAPACK_zspsv( char* uplo, lapack_int* n, lapack_int* nrhs, - lapack_complex_double* ap, lapack_int* ipiv, - lapack_complex_double* b, lapack_int* ldb, - lapack_int *info ); -void LAPACK_sspsvx( char* fact, char* uplo, lapack_int* n, lapack_int* nrhs, - const float* ap, float* afp, lapack_int* ipiv, - const float* b, lapack_int* ldb, float* x, lapack_int* ldx, - float* rcond, float* ferr, float* berr, float* work, - lapack_int* iwork, lapack_int *info ); -void LAPACK_dspsvx( char* fact, char* uplo, lapack_int* n, lapack_int* nrhs, - const double* ap, double* afp, lapack_int* ipiv, - const double* b, lapack_int* ldb, double* x, - lapack_int* ldx, double* rcond, double* ferr, double* berr, - double* work, lapack_int* iwork, lapack_int *info ); -void LAPACK_cspsvx( char* fact, char* uplo, lapack_int* n, lapack_int* nrhs, - const lapack_complex_float* ap, lapack_complex_float* afp, - lapack_int* ipiv, const lapack_complex_float* b, - lapack_int* ldb, lapack_complex_float* x, lapack_int* ldx, - float* rcond, float* ferr, float* berr, - lapack_complex_float* work, float* rwork, - lapack_int *info ); -void LAPACK_zspsvx( char* fact, char* uplo, lapack_int* n, lapack_int* nrhs, - const lapack_complex_double* ap, lapack_complex_double* afp, - lapack_int* ipiv, const lapack_complex_double* b, - lapack_int* ldb, lapack_complex_double* x, lapack_int* ldx, - double* rcond, double* ferr, double* berr, - lapack_complex_double* work, double* rwork, - lapack_int *info ); -void LAPACK_chpsv( char* uplo, lapack_int* n, lapack_int* nrhs, - lapack_complex_float* ap, lapack_int* ipiv, - lapack_complex_float* b, lapack_int* ldb, lapack_int *info ); -void LAPACK_zhpsv( char* uplo, lapack_int* n, lapack_int* nrhs, - lapack_complex_double* ap, lapack_int* ipiv, - lapack_complex_double* b, lapack_int* ldb, - lapack_int *info ); -void LAPACK_chpsvx( char* fact, char* uplo, lapack_int* n, lapack_int* nrhs, - const lapack_complex_float* ap, lapack_complex_float* afp, - lapack_int* ipiv, const lapack_complex_float* b, - lapack_int* ldb, lapack_complex_float* x, lapack_int* ldx, - float* rcond, float* ferr, float* berr, - lapack_complex_float* work, float* rwork, - lapack_int *info ); -void LAPACK_zhpsvx( char* fact, char* uplo, lapack_int* n, lapack_int* nrhs, - const lapack_complex_double* ap, lapack_complex_double* afp, - lapack_int* ipiv, const lapack_complex_double* b, - lapack_int* ldb, lapack_complex_double* x, lapack_int* ldx, - double* rcond, double* ferr, double* berr, - lapack_complex_double* work, double* rwork, - lapack_int *info ); -void LAPACK_sgeqrf( lapack_int* m, lapack_int* n, float* a, lapack_int* lda, - float* tau, float* work, lapack_int* lwork, - lapack_int *info ); -void LAPACK_dgeqrf( lapack_int* m, lapack_int* n, double* a, lapack_int* lda, - double* tau, double* work, lapack_int* lwork, - lapack_int *info ); -void LAPACK_cgeqrf( lapack_int* m, lapack_int* n, lapack_complex_float* a, - lapack_int* lda, lapack_complex_float* tau, - lapack_complex_float* work, lapack_int* lwork, - lapack_int *info ); -void LAPACK_zgeqrf( lapack_int* m, lapack_int* n, lapack_complex_double* a, - lapack_int* lda, lapack_complex_double* tau, - lapack_complex_double* work, lapack_int* lwork, - lapack_int *info ); -void LAPACK_sgeqpf( lapack_int* m, lapack_int* n, float* a, lapack_int* lda, - lapack_int* jpvt, float* tau, float* work, - lapack_int *info ); -void LAPACK_dgeqpf( lapack_int* m, lapack_int* n, double* a, lapack_int* lda, - lapack_int* jpvt, double* tau, double* work, - lapack_int *info ); -void LAPACK_cgeqpf( lapack_int* m, lapack_int* n, lapack_complex_float* a, - lapack_int* lda, lapack_int* jpvt, - lapack_complex_float* tau, lapack_complex_float* work, - float* rwork, lapack_int *info ); -void LAPACK_zgeqpf( lapack_int* m, lapack_int* n, lapack_complex_double* a, - lapack_int* lda, lapack_int* jpvt, - lapack_complex_double* tau, lapack_complex_double* work, - double* rwork, lapack_int *info ); -void LAPACK_sgeqp3( lapack_int* m, lapack_int* n, float* a, lapack_int* lda, - lapack_int* jpvt, float* tau, float* work, - lapack_int* lwork, lapack_int *info ); -void LAPACK_dgeqp3( lapack_int* m, lapack_int* n, double* a, lapack_int* lda, - lapack_int* jpvt, double* tau, double* work, - lapack_int* lwork, lapack_int *info ); -void LAPACK_cgeqp3( lapack_int* m, lapack_int* n, lapack_complex_float* a, - lapack_int* lda, lapack_int* jpvt, - lapack_complex_float* tau, lapack_complex_float* work, - lapack_int* lwork, float* rwork, lapack_int *info ); -void LAPACK_zgeqp3( lapack_int* m, lapack_int* n, lapack_complex_double* a, - lapack_int* lda, lapack_int* jpvt, - lapack_complex_double* tau, lapack_complex_double* work, - lapack_int* lwork, double* rwork, lapack_int *info ); -void LAPACK_sorgqr( lapack_int* m, lapack_int* n, lapack_int* k, float* a, - lapack_int* lda, const float* tau, float* work, - lapack_int* lwork, lapack_int *info ); -void LAPACK_dorgqr( lapack_int* m, lapack_int* n, lapack_int* k, double* a, - lapack_int* lda, const double* tau, double* work, - lapack_int* lwork, lapack_int *info ); -void LAPACK_sormqr( char* side, char* trans, lapack_int* m, lapack_int* n, - lapack_int* k, const float* a, lapack_int* lda, - const float* tau, float* c, lapack_int* ldc, float* work, - lapack_int* lwork, lapack_int *info ); -void LAPACK_dormqr( char* side, char* trans, lapack_int* m, lapack_int* n, - lapack_int* k, const double* a, lapack_int* lda, - const double* tau, double* c, lapack_int* ldc, double* work, - lapack_int* lwork, lapack_int *info ); -void LAPACK_cungqr( lapack_int* m, lapack_int* n, lapack_int* k, - lapack_complex_float* a, lapack_int* lda, - const lapack_complex_float* tau, lapack_complex_float* work, - lapack_int* lwork, lapack_int *info ); -void LAPACK_zungqr( lapack_int* m, lapack_int* n, lapack_int* k, - lapack_complex_double* a, lapack_int* lda, - const lapack_complex_double* tau, - lapack_complex_double* work, lapack_int* lwork, - lapack_int *info ); -void LAPACK_cunmqr( char* side, char* trans, lapack_int* m, lapack_int* n, - lapack_int* k, const lapack_complex_float* a, - lapack_int* lda, const lapack_complex_float* tau, - lapack_complex_float* c, lapack_int* ldc, - lapack_complex_float* work, lapack_int* lwork, - lapack_int *info ); -void LAPACK_zunmqr( char* side, char* trans, lapack_int* m, lapack_int* n, - lapack_int* k, const lapack_complex_double* a, - lapack_int* lda, const lapack_complex_double* tau, - lapack_complex_double* c, lapack_int* ldc, - lapack_complex_double* work, lapack_int* lwork, - lapack_int *info ); -void LAPACK_sgelqf( lapack_int* m, lapack_int* n, float* a, lapack_int* lda, - float* tau, float* work, lapack_int* lwork, - lapack_int *info ); -void LAPACK_dgelqf( lapack_int* m, lapack_int* n, double* a, lapack_int* lda, - double* tau, double* work, lapack_int* lwork, - lapack_int *info ); -void LAPACK_cgelqf( lapack_int* m, lapack_int* n, lapack_complex_float* a, - lapack_int* lda, lapack_complex_float* tau, - lapack_complex_float* work, lapack_int* lwork, - lapack_int *info ); -void LAPACK_zgelqf( lapack_int* m, lapack_int* n, lapack_complex_double* a, - lapack_int* lda, lapack_complex_double* tau, - lapack_complex_double* work, lapack_int* lwork, - lapack_int *info ); -void LAPACK_sorglq( lapack_int* m, lapack_int* n, lapack_int* k, float* a, - lapack_int* lda, const float* tau, float* work, - lapack_int* lwork, lapack_int *info ); -void LAPACK_dorglq( lapack_int* m, lapack_int* n, lapack_int* k, double* a, - lapack_int* lda, const double* tau, double* work, - lapack_int* lwork, lapack_int *info ); -void LAPACK_sormlq( char* side, char* trans, lapack_int* m, lapack_int* n, - lapack_int* k, const float* a, lapack_int* lda, - const float* tau, float* c, lapack_int* ldc, float* work, - lapack_int* lwork, lapack_int *info ); -void LAPACK_dormlq( char* side, char* trans, lapack_int* m, lapack_int* n, - lapack_int* k, const double* a, lapack_int* lda, - const double* tau, double* c, lapack_int* ldc, double* work, - lapack_int* lwork, lapack_int *info ); -void LAPACK_cunglq( lapack_int* m, lapack_int* n, lapack_int* k, - lapack_complex_float* a, lapack_int* lda, - const lapack_complex_float* tau, lapack_complex_float* work, - lapack_int* lwork, lapack_int *info ); -void LAPACK_zunglq( lapack_int* m, lapack_int* n, lapack_int* k, - lapack_complex_double* a, lapack_int* lda, - const lapack_complex_double* tau, - lapack_complex_double* work, lapack_int* lwork, - lapack_int *info ); -void LAPACK_cunmlq( char* side, char* trans, lapack_int* m, lapack_int* n, - lapack_int* k, const lapack_complex_float* a, - lapack_int* lda, const lapack_complex_float* tau, - lapack_complex_float* c, lapack_int* ldc, - lapack_complex_float* work, lapack_int* lwork, - lapack_int *info ); -void LAPACK_zunmlq( char* side, char* trans, lapack_int* m, lapack_int* n, - lapack_int* k, const lapack_complex_double* a, - lapack_int* lda, const lapack_complex_double* tau, - lapack_complex_double* c, lapack_int* ldc, - lapack_complex_double* work, lapack_int* lwork, - lapack_int *info ); -void LAPACK_sgeqlf( lapack_int* m, lapack_int* n, float* a, lapack_int* lda, - float* tau, float* work, lapack_int* lwork, - lapack_int *info ); -void LAPACK_dgeqlf( lapack_int* m, lapack_int* n, double* a, lapack_int* lda, - double* tau, double* work, lapack_int* lwork, - lapack_int *info ); -void LAPACK_cgeqlf( lapack_int* m, lapack_int* n, lapack_complex_float* a, - lapack_int* lda, lapack_complex_float* tau, - lapack_complex_float* work, lapack_int* lwork, - lapack_int *info ); -void LAPACK_zgeqlf( lapack_int* m, lapack_int* n, lapack_complex_double* a, - lapack_int* lda, lapack_complex_double* tau, - lapack_complex_double* work, lapack_int* lwork, - lapack_int *info ); -void LAPACK_sorgql( lapack_int* m, lapack_int* n, lapack_int* k, float* a, - lapack_int* lda, const float* tau, float* work, - lapack_int* lwork, lapack_int *info ); -void LAPACK_dorgql( lapack_int* m, lapack_int* n, lapack_int* k, double* a, - lapack_int* lda, const double* tau, double* work, - lapack_int* lwork, lapack_int *info ); -void LAPACK_cungql( lapack_int* m, lapack_int* n, lapack_int* k, - lapack_complex_float* a, lapack_int* lda, - const lapack_complex_float* tau, lapack_complex_float* work, - lapack_int* lwork, lapack_int *info ); -void LAPACK_zungql( lapack_int* m, lapack_int* n, lapack_int* k, - lapack_complex_double* a, lapack_int* lda, - const lapack_complex_double* tau, - lapack_complex_double* work, lapack_int* lwork, - lapack_int *info ); -void LAPACK_sormql( char* side, char* trans, lapack_int* m, lapack_int* n, - lapack_int* k, const float* a, lapack_int* lda, - const float* tau, float* c, lapack_int* ldc, float* work, - lapack_int* lwork, lapack_int *info ); -void LAPACK_dormql( char* side, char* trans, lapack_int* m, lapack_int* n, - lapack_int* k, const double* a, lapack_int* lda, - const double* tau, double* c, lapack_int* ldc, double* work, - lapack_int* lwork, lapack_int *info ); -void LAPACK_cunmql( char* side, char* trans, lapack_int* m, lapack_int* n, - lapack_int* k, const lapack_complex_float* a, - lapack_int* lda, const lapack_complex_float* tau, - lapack_complex_float* c, lapack_int* ldc, - lapack_complex_float* work, lapack_int* lwork, - lapack_int *info ); -void LAPACK_zunmql( char* side, char* trans, lapack_int* m, lapack_int* n, - lapack_int* k, const lapack_complex_double* a, - lapack_int* lda, const lapack_complex_double* tau, - lapack_complex_double* c, lapack_int* ldc, - lapack_complex_double* work, lapack_int* lwork, - lapack_int *info ); -void LAPACK_sgerqf( lapack_int* m, lapack_int* n, float* a, lapack_int* lda, - float* tau, float* work, lapack_int* lwork, - lapack_int *info ); -void LAPACK_dgerqf( lapack_int* m, lapack_int* n, double* a, lapack_int* lda, - double* tau, double* work, lapack_int* lwork, - lapack_int *info ); -void LAPACK_cgerqf( lapack_int* m, lapack_int* n, lapack_complex_float* a, - lapack_int* lda, lapack_complex_float* tau, - lapack_complex_float* work, lapack_int* lwork, - lapack_int *info ); -void LAPACK_zgerqf( lapack_int* m, lapack_int* n, lapack_complex_double* a, - lapack_int* lda, lapack_complex_double* tau, - lapack_complex_double* work, lapack_int* lwork, - lapack_int *info ); -void LAPACK_sorgrq( lapack_int* m, lapack_int* n, lapack_int* k, float* a, - lapack_int* lda, const float* tau, float* work, - lapack_int* lwork, lapack_int *info ); -void LAPACK_dorgrq( lapack_int* m, lapack_int* n, lapack_int* k, double* a, - lapack_int* lda, const double* tau, double* work, - lapack_int* lwork, lapack_int *info ); -void LAPACK_cungrq( lapack_int* m, lapack_int* n, lapack_int* k, - lapack_complex_float* a, lapack_int* lda, - const lapack_complex_float* tau, lapack_complex_float* work, - lapack_int* lwork, lapack_int *info ); -void LAPACK_zungrq( lapack_int* m, lapack_int* n, lapack_int* k, - lapack_complex_double* a, lapack_int* lda, - const lapack_complex_double* tau, - lapack_complex_double* work, lapack_int* lwork, - lapack_int *info ); -void LAPACK_sormrq( char* side, char* trans, lapack_int* m, lapack_int* n, - lapack_int* k, const float* a, lapack_int* lda, - const float* tau, float* c, lapack_int* ldc, float* work, - lapack_int* lwork, lapack_int *info ); -void LAPACK_dormrq( char* side, char* trans, lapack_int* m, lapack_int* n, - lapack_int* k, const double* a, lapack_int* lda, - const double* tau, double* c, lapack_int* ldc, double* work, - lapack_int* lwork, lapack_int *info ); -void LAPACK_cunmrq( char* side, char* trans, lapack_int* m, lapack_int* n, - lapack_int* k, const lapack_complex_float* a, - lapack_int* lda, const lapack_complex_float* tau, - lapack_complex_float* c, lapack_int* ldc, - lapack_complex_float* work, lapack_int* lwork, - lapack_int *info ); -void LAPACK_zunmrq( char* side, char* trans, lapack_int* m, lapack_int* n, - lapack_int* k, const lapack_complex_double* a, - lapack_int* lda, const lapack_complex_double* tau, - lapack_complex_double* c, lapack_int* ldc, - lapack_complex_double* work, lapack_int* lwork, - lapack_int *info ); -void LAPACK_stzrzf( lapack_int* m, lapack_int* n, float* a, lapack_int* lda, - float* tau, float* work, lapack_int* lwork, - lapack_int *info ); -void LAPACK_dtzrzf( lapack_int* m, lapack_int* n, double* a, lapack_int* lda, - double* tau, double* work, lapack_int* lwork, - lapack_int *info ); -void LAPACK_ctzrzf( lapack_int* m, lapack_int* n, lapack_complex_float* a, - lapack_int* lda, lapack_complex_float* tau, - lapack_complex_float* work, lapack_int* lwork, - lapack_int *info ); -void LAPACK_ztzrzf( lapack_int* m, lapack_int* n, lapack_complex_double* a, - lapack_int* lda, lapack_complex_double* tau, - lapack_complex_double* work, lapack_int* lwork, - lapack_int *info ); -void LAPACK_sormrz( char* side, char* trans, lapack_int* m, lapack_int* n, - lapack_int* k, lapack_int* l, const float* a, - lapack_int* lda, const float* tau, float* c, - lapack_int* ldc, float* work, lapack_int* lwork, - lapack_int *info ); -void LAPACK_dormrz( char* side, char* trans, lapack_int* m, lapack_int* n, - lapack_int* k, lapack_int* l, const double* a, - lapack_int* lda, const double* tau, double* c, - lapack_int* ldc, double* work, lapack_int* lwork, - lapack_int *info ); -void LAPACK_cunmrz( char* side, char* trans, lapack_int* m, lapack_int* n, - lapack_int* k, lapack_int* l, const lapack_complex_float* a, - lapack_int* lda, const lapack_complex_float* tau, - lapack_complex_float* c, lapack_int* ldc, - lapack_complex_float* work, lapack_int* lwork, - lapack_int *info ); -void LAPACK_zunmrz( char* side, char* trans, lapack_int* m, lapack_int* n, - lapack_int* k, lapack_int* l, - const lapack_complex_double* a, lapack_int* lda, - const lapack_complex_double* tau, lapack_complex_double* c, - lapack_int* ldc, lapack_complex_double* work, - lapack_int* lwork, lapack_int *info ); -void LAPACK_sggqrf( lapack_int* n, lapack_int* m, lapack_int* p, float* a, - lapack_int* lda, float* taua, float* b, lapack_int* ldb, - float* taub, float* work, lapack_int* lwork, - lapack_int *info ); -void LAPACK_dggqrf( lapack_int* n, lapack_int* m, lapack_int* p, double* a, - lapack_int* lda, double* taua, double* b, lapack_int* ldb, - double* taub, double* work, lapack_int* lwork, - lapack_int *info ); -void LAPACK_cggqrf( lapack_int* n, lapack_int* m, lapack_int* p, - lapack_complex_float* a, lapack_int* lda, - lapack_complex_float* taua, lapack_complex_float* b, - lapack_int* ldb, lapack_complex_float* taub, - lapack_complex_float* work, lapack_int* lwork, - lapack_int *info ); -void LAPACK_zggqrf( lapack_int* n, lapack_int* m, lapack_int* p, - lapack_complex_double* a, lapack_int* lda, - lapack_complex_double* taua, lapack_complex_double* b, - lapack_int* ldb, lapack_complex_double* taub, - lapack_complex_double* work, lapack_int* lwork, - lapack_int *info ); -void LAPACK_sggrqf( lapack_int* m, lapack_int* p, lapack_int* n, float* a, - lapack_int* lda, float* taua, float* b, lapack_int* ldb, - float* taub, float* work, lapack_int* lwork, - lapack_int *info ); -void LAPACK_dggrqf( lapack_int* m, lapack_int* p, lapack_int* n, double* a, - lapack_int* lda, double* taua, double* b, lapack_int* ldb, - double* taub, double* work, lapack_int* lwork, - lapack_int *info ); -void LAPACK_cggrqf( lapack_int* m, lapack_int* p, lapack_int* n, - lapack_complex_float* a, lapack_int* lda, - lapack_complex_float* taua, lapack_complex_float* b, - lapack_int* ldb, lapack_complex_float* taub, - lapack_complex_float* work, lapack_int* lwork, - lapack_int *info ); -void LAPACK_zggrqf( lapack_int* m, lapack_int* p, lapack_int* n, - lapack_complex_double* a, lapack_int* lda, - lapack_complex_double* taua, lapack_complex_double* b, - lapack_int* ldb, lapack_complex_double* taub, - lapack_complex_double* work, lapack_int* lwork, - lapack_int *info ); -void LAPACK_sgebrd( lapack_int* m, lapack_int* n, float* a, lapack_int* lda, - float* d, float* e, float* tauq, float* taup, float* work, - lapack_int* lwork, lapack_int *info ); -void LAPACK_dgebrd( lapack_int* m, lapack_int* n, double* a, lapack_int* lda, - double* d, double* e, double* tauq, double* taup, - double* work, lapack_int* lwork, lapack_int *info ); -void LAPACK_cgebrd( lapack_int* m, lapack_int* n, lapack_complex_float* a, - lapack_int* lda, float* d, float* e, - lapack_complex_float* tauq, lapack_complex_float* taup, - lapack_complex_float* work, lapack_int* lwork, - lapack_int *info ); -void LAPACK_zgebrd( lapack_int* m, lapack_int* n, lapack_complex_double* a, - lapack_int* lda, double* d, double* e, - lapack_complex_double* tauq, lapack_complex_double* taup, - lapack_complex_double* work, lapack_int* lwork, - lapack_int *info ); -void LAPACK_sgbbrd( char* vect, lapack_int* m, lapack_int* n, lapack_int* ncc, - lapack_int* kl, lapack_int* ku, float* ab, lapack_int* ldab, - float* d, float* e, float* q, lapack_int* ldq, float* pt, - lapack_int* ldpt, float* c, lapack_int* ldc, float* work, - lapack_int *info ); -void LAPACK_dgbbrd( char* vect, lapack_int* m, lapack_int* n, lapack_int* ncc, - lapack_int* kl, lapack_int* ku, double* ab, - lapack_int* ldab, double* d, double* e, double* q, - lapack_int* ldq, double* pt, lapack_int* ldpt, double* c, - lapack_int* ldc, double* work, lapack_int *info ); -void LAPACK_cgbbrd( char* vect, lapack_int* m, lapack_int* n, lapack_int* ncc, - lapack_int* kl, lapack_int* ku, lapack_complex_float* ab, - lapack_int* ldab, float* d, float* e, - lapack_complex_float* q, lapack_int* ldq, - lapack_complex_float* pt, lapack_int* ldpt, - lapack_complex_float* c, lapack_int* ldc, - lapack_complex_float* work, float* rwork, - lapack_int *info ); -void LAPACK_zgbbrd( char* vect, lapack_int* m, lapack_int* n, lapack_int* ncc, - lapack_int* kl, lapack_int* ku, lapack_complex_double* ab, - lapack_int* ldab, double* d, double* e, - lapack_complex_double* q, lapack_int* ldq, - lapack_complex_double* pt, lapack_int* ldpt, - lapack_complex_double* c, lapack_int* ldc, - lapack_complex_double* work, double* rwork, - lapack_int *info ); -void LAPACK_sorgbr( char* vect, lapack_int* m, lapack_int* n, lapack_int* k, - float* a, lapack_int* lda, const float* tau, float* work, - lapack_int* lwork, lapack_int *info ); -void LAPACK_dorgbr( char* vect, lapack_int* m, lapack_int* n, lapack_int* k, - double* a, lapack_int* lda, const double* tau, double* work, - lapack_int* lwork, lapack_int *info ); -void LAPACK_sormbr( char* vect, char* side, char* trans, lapack_int* m, - lapack_int* n, lapack_int* k, const float* a, - lapack_int* lda, const float* tau, float* c, - lapack_int* ldc, float* work, lapack_int* lwork, - lapack_int *info ); -void LAPACK_dormbr( char* vect, char* side, char* trans, lapack_int* m, - lapack_int* n, lapack_int* k, const double* a, - lapack_int* lda, const double* tau, double* c, - lapack_int* ldc, double* work, lapack_int* lwork, - lapack_int *info ); -void LAPACK_cungbr( char* vect, lapack_int* m, lapack_int* n, lapack_int* k, - lapack_complex_float* a, lapack_int* lda, - const lapack_complex_float* tau, lapack_complex_float* work, - lapack_int* lwork, lapack_int *info ); -void LAPACK_zungbr( char* vect, lapack_int* m, lapack_int* n, lapack_int* k, - lapack_complex_double* a, lapack_int* lda, - const lapack_complex_double* tau, - lapack_complex_double* work, lapack_int* lwork, - lapack_int *info ); -void LAPACK_cunmbr( char* vect, char* side, char* trans, lapack_int* m, - lapack_int* n, lapack_int* k, const lapack_complex_float* a, - lapack_int* lda, const lapack_complex_float* tau, - lapack_complex_float* c, lapack_int* ldc, - lapack_complex_float* work, lapack_int* lwork, - lapack_int *info ); -void LAPACK_zunmbr( char* vect, char* side, char* trans, lapack_int* m, - lapack_int* n, lapack_int* k, - const lapack_complex_double* a, lapack_int* lda, - const lapack_complex_double* tau, lapack_complex_double* c, - lapack_int* ldc, lapack_complex_double* work, - lapack_int* lwork, lapack_int *info ); -void LAPACK_sbdsqr( char* uplo, lapack_int* n, lapack_int* ncvt, - lapack_int* nru, lapack_int* ncc, float* d, float* e, - float* vt, lapack_int* ldvt, float* u, lapack_int* ldu, - float* c, lapack_int* ldc, float* work, lapack_int *info ); -void LAPACK_dbdsqr( char* uplo, lapack_int* n, lapack_int* ncvt, - lapack_int* nru, lapack_int* ncc, double* d, double* e, - double* vt, lapack_int* ldvt, double* u, lapack_int* ldu, - double* c, lapack_int* ldc, double* work, - lapack_int *info ); -void LAPACK_cbdsqr( char* uplo, lapack_int* n, lapack_int* ncvt, - lapack_int* nru, lapack_int* ncc, float* d, float* e, - lapack_complex_float* vt, lapack_int* ldvt, - lapack_complex_float* u, lapack_int* ldu, - lapack_complex_float* c, lapack_int* ldc, float* work, - lapack_int *info ); -void LAPACK_zbdsqr( char* uplo, lapack_int* n, lapack_int* ncvt, - lapack_int* nru, lapack_int* ncc, double* d, double* e, - lapack_complex_double* vt, lapack_int* ldvt, - lapack_complex_double* u, lapack_int* ldu, - lapack_complex_double* c, lapack_int* ldc, double* work, - lapack_int *info ); -void LAPACK_sbdsdc( char* uplo, char* compq, lapack_int* n, float* d, float* e, - float* u, lapack_int* ldu, float* vt, lapack_int* ldvt, - float* q, lapack_int* iq, float* work, lapack_int* iwork, - lapack_int *info ); -void LAPACK_dbdsdc( char* uplo, char* compq, lapack_int* n, double* d, - double* e, double* u, lapack_int* ldu, double* vt, - lapack_int* ldvt, double* q, lapack_int* iq, double* work, - lapack_int* iwork, lapack_int *info ); -void LAPACK_ssytrd( char* uplo, lapack_int* n, float* a, lapack_int* lda, - float* d, float* e, float* tau, float* work, - lapack_int* lwork, lapack_int *info ); -void LAPACK_dsytrd( char* uplo, lapack_int* n, double* a, lapack_int* lda, - double* d, double* e, double* tau, double* work, - lapack_int* lwork, lapack_int *info ); -void LAPACK_sorgtr( char* uplo, lapack_int* n, float* a, lapack_int* lda, - const float* tau, float* work, lapack_int* lwork, - lapack_int *info ); -void LAPACK_dorgtr( char* uplo, lapack_int* n, double* a, lapack_int* lda, - const double* tau, double* work, lapack_int* lwork, - lapack_int *info ); -void LAPACK_sormtr( char* side, char* uplo, char* trans, lapack_int* m, - lapack_int* n, const float* a, lapack_int* lda, - const float* tau, float* c, lapack_int* ldc, float* work, - lapack_int* lwork, lapack_int *info ); -void LAPACK_dormtr( char* side, char* uplo, char* trans, lapack_int* m, - lapack_int* n, const double* a, lapack_int* lda, - const double* tau, double* c, lapack_int* ldc, double* work, - lapack_int* lwork, lapack_int *info ); -void LAPACK_chetrd( char* uplo, lapack_int* n, lapack_complex_float* a, - lapack_int* lda, float* d, float* e, - lapack_complex_float* tau, lapack_complex_float* work, - lapack_int* lwork, lapack_int *info ); -void LAPACK_zhetrd( char* uplo, lapack_int* n, lapack_complex_double* a, - lapack_int* lda, double* d, double* e, - lapack_complex_double* tau, lapack_complex_double* work, - lapack_int* lwork, lapack_int *info ); -void LAPACK_cungtr( char* uplo, lapack_int* n, lapack_complex_float* a, - lapack_int* lda, const lapack_complex_float* tau, - lapack_complex_float* work, lapack_int* lwork, - lapack_int *info ); -void LAPACK_zungtr( char* uplo, lapack_int* n, lapack_complex_double* a, - lapack_int* lda, const lapack_complex_double* tau, - lapack_complex_double* work, lapack_int* lwork, - lapack_int *info ); -void LAPACK_cunmtr( char* side, char* uplo, char* trans, lapack_int* m, - lapack_int* n, const lapack_complex_float* a, - lapack_int* lda, const lapack_complex_float* tau, - lapack_complex_float* c, lapack_int* ldc, - lapack_complex_float* work, lapack_int* lwork, - lapack_int *info ); -void LAPACK_zunmtr( char* side, char* uplo, char* trans, lapack_int* m, - lapack_int* n, const lapack_complex_double* a, - lapack_int* lda, const lapack_complex_double* tau, - lapack_complex_double* c, lapack_int* ldc, - lapack_complex_double* work, lapack_int* lwork, - lapack_int *info ); -void LAPACK_ssptrd( char* uplo, lapack_int* n, float* ap, float* d, float* e, - float* tau, lapack_int *info ); -void LAPACK_dsptrd( char* uplo, lapack_int* n, double* ap, double* d, double* e, - double* tau, lapack_int *info ); -void LAPACK_sopgtr( char* uplo, lapack_int* n, const float* ap, - const float* tau, float* q, lapack_int* ldq, float* work, - lapack_int *info ); -void LAPACK_dopgtr( char* uplo, lapack_int* n, const double* ap, - const double* tau, double* q, lapack_int* ldq, double* work, - lapack_int *info ); -void LAPACK_sopmtr( char* side, char* uplo, char* trans, lapack_int* m, - lapack_int* n, const float* ap, const float* tau, float* c, - lapack_int* ldc, float* work, lapack_int *info ); -void LAPACK_dopmtr( char* side, char* uplo, char* trans, lapack_int* m, - lapack_int* n, const double* ap, const double* tau, - double* c, lapack_int* ldc, double* work, - lapack_int *info ); -void LAPACK_chptrd( char* uplo, lapack_int* n, lapack_complex_float* ap, - float* d, float* e, lapack_complex_float* tau, - lapack_int *info ); -void LAPACK_zhptrd( char* uplo, lapack_int* n, lapack_complex_double* ap, - double* d, double* e, lapack_complex_double* tau, - lapack_int *info ); -void LAPACK_cupgtr( char* uplo, lapack_int* n, const lapack_complex_float* ap, - const lapack_complex_float* tau, lapack_complex_float* q, - lapack_int* ldq, lapack_complex_float* work, - lapack_int *info ); -void LAPACK_zupgtr( char* uplo, lapack_int* n, const lapack_complex_double* ap, - const lapack_complex_double* tau, lapack_complex_double* q, - lapack_int* ldq, lapack_complex_double* work, - lapack_int *info ); -void LAPACK_cupmtr( char* side, char* uplo, char* trans, lapack_int* m, - lapack_int* n, const lapack_complex_float* ap, - const lapack_complex_float* tau, lapack_complex_float* c, - lapack_int* ldc, lapack_complex_float* work, - lapack_int *info ); -void LAPACK_zupmtr( char* side, char* uplo, char* trans, lapack_int* m, - lapack_int* n, const lapack_complex_double* ap, - const lapack_complex_double* tau, lapack_complex_double* c, - lapack_int* ldc, lapack_complex_double* work, - lapack_int *info ); -void LAPACK_ssbtrd( char* vect, char* uplo, lapack_int* n, lapack_int* kd, - float* ab, lapack_int* ldab, float* d, float* e, float* q, - lapack_int* ldq, float* work, lapack_int *info ); -void LAPACK_dsbtrd( char* vect, char* uplo, lapack_int* n, lapack_int* kd, - double* ab, lapack_int* ldab, double* d, double* e, - double* q, lapack_int* ldq, double* work, - lapack_int *info ); -void LAPACK_chbtrd( char* vect, char* uplo, lapack_int* n, lapack_int* kd, - lapack_complex_float* ab, lapack_int* ldab, float* d, - float* e, lapack_complex_float* q, lapack_int* ldq, - lapack_complex_float* work, lapack_int *info ); -void LAPACK_zhbtrd( char* vect, char* uplo, lapack_int* n, lapack_int* kd, - lapack_complex_double* ab, lapack_int* ldab, double* d, - double* e, lapack_complex_double* q, lapack_int* ldq, - lapack_complex_double* work, lapack_int *info ); -void LAPACK_ssterf( lapack_int* n, float* d, float* e, lapack_int *info ); -void LAPACK_dsterf( lapack_int* n, double* d, double* e, lapack_int *info ); -void LAPACK_ssteqr( char* compz, lapack_int* n, float* d, float* e, float* z, - lapack_int* ldz, float* work, lapack_int *info ); -void LAPACK_dsteqr( char* compz, lapack_int* n, double* d, double* e, double* z, - lapack_int* ldz, double* work, lapack_int *info ); -void LAPACK_csteqr( char* compz, lapack_int* n, float* d, float* e, - lapack_complex_float* z, lapack_int* ldz, float* work, - lapack_int *info ); -void LAPACK_zsteqr( char* compz, lapack_int* n, double* d, double* e, - lapack_complex_double* z, lapack_int* ldz, double* work, - lapack_int *info ); -void LAPACK_sstemr( char* jobz, char* range, lapack_int* n, float* d, float* e, - float* vl, float* vu, lapack_int* il, lapack_int* iu, - lapack_int* m, float* w, float* z, lapack_int* ldz, - lapack_int* nzc, lapack_int* isuppz, lapack_logical* tryrac, - float* work, lapack_int* lwork, lapack_int* iwork, - lapack_int* liwork, lapack_int *info ); -void LAPACK_dstemr( char* jobz, char* range, lapack_int* n, double* d, - double* e, double* vl, double* vu, lapack_int* il, - lapack_int* iu, lapack_int* m, double* w, double* z, - lapack_int* ldz, lapack_int* nzc, lapack_int* isuppz, - lapack_logical* tryrac, double* work, lapack_int* lwork, - lapack_int* iwork, lapack_int* liwork, lapack_int *info ); -void LAPACK_cstemr( char* jobz, char* range, lapack_int* n, float* d, float* e, - float* vl, float* vu, lapack_int* il, lapack_int* iu, - lapack_int* m, float* w, lapack_complex_float* z, - lapack_int* ldz, lapack_int* nzc, lapack_int* isuppz, - lapack_logical* tryrac, float* work, lapack_int* lwork, - lapack_int* iwork, lapack_int* liwork, lapack_int *info ); -void LAPACK_zstemr( char* jobz, char* range, lapack_int* n, double* d, - double* e, double* vl, double* vu, lapack_int* il, - lapack_int* iu, lapack_int* m, double* w, - lapack_complex_double* z, lapack_int* ldz, lapack_int* nzc, - lapack_int* isuppz, lapack_logical* tryrac, double* work, - lapack_int* lwork, lapack_int* iwork, lapack_int* liwork, - lapack_int *info ); -void LAPACK_sstedc( char* compz, lapack_int* n, float* d, float* e, float* z, - lapack_int* ldz, float* work, lapack_int* lwork, - lapack_int* iwork, lapack_int* liwork, lapack_int *info ); -void LAPACK_dstedc( char* compz, lapack_int* n, double* d, double* e, double* z, - lapack_int* ldz, double* work, lapack_int* lwork, - lapack_int* iwork, lapack_int* liwork, lapack_int *info ); -void LAPACK_cstedc( char* compz, lapack_int* n, float* d, float* e, - lapack_complex_float* z, lapack_int* ldz, - lapack_complex_float* work, lapack_int* lwork, float* rwork, - lapack_int* lrwork, lapack_int* iwork, lapack_int* liwork, - lapack_int *info ); -void LAPACK_zstedc( char* compz, lapack_int* n, double* d, double* e, - lapack_complex_double* z, lapack_int* ldz, - lapack_complex_double* work, lapack_int* lwork, - double* rwork, lapack_int* lrwork, lapack_int* iwork, - lapack_int* liwork, lapack_int *info ); -void LAPACK_sstegr( char* jobz, char* range, lapack_int* n, float* d, float* e, - float* vl, float* vu, lapack_int* il, lapack_int* iu, - float* abstol, lapack_int* m, float* w, float* z, - lapack_int* ldz, lapack_int* isuppz, float* work, - lapack_int* lwork, lapack_int* iwork, lapack_int* liwork, - lapack_int *info ); -void LAPACK_dstegr( char* jobz, char* range, lapack_int* n, double* d, - double* e, double* vl, double* vu, lapack_int* il, - lapack_int* iu, double* abstol, lapack_int* m, double* w, - double* z, lapack_int* ldz, lapack_int* isuppz, - double* work, lapack_int* lwork, lapack_int* iwork, - lapack_int* liwork, lapack_int *info ); -void LAPACK_cstegr( char* jobz, char* range, lapack_int* n, float* d, float* e, - float* vl, float* vu, lapack_int* il, lapack_int* iu, - float* abstol, lapack_int* m, float* w, - lapack_complex_float* z, lapack_int* ldz, - lapack_int* isuppz, float* work, lapack_int* lwork, - lapack_int* iwork, lapack_int* liwork, lapack_int *info ); -void LAPACK_zstegr( char* jobz, char* range, lapack_int* n, double* d, - double* e, double* vl, double* vu, lapack_int* il, - lapack_int* iu, double* abstol, lapack_int* m, double* w, - lapack_complex_double* z, lapack_int* ldz, - lapack_int* isuppz, double* work, lapack_int* lwork, - lapack_int* iwork, lapack_int* liwork, lapack_int *info ); -void LAPACK_spteqr( char* compz, lapack_int* n, float* d, float* e, float* z, - lapack_int* ldz, float* work, lapack_int *info ); -void LAPACK_dpteqr( char* compz, lapack_int* n, double* d, double* e, double* z, - lapack_int* ldz, double* work, lapack_int *info ); -void LAPACK_cpteqr( char* compz, lapack_int* n, float* d, float* e, - lapack_complex_float* z, lapack_int* ldz, float* work, - lapack_int *info ); -void LAPACK_zpteqr( char* compz, lapack_int* n, double* d, double* e, - lapack_complex_double* z, lapack_int* ldz, double* work, - lapack_int *info ); -void LAPACK_sstebz( char* range, char* order, lapack_int* n, float* vl, - float* vu, lapack_int* il, lapack_int* iu, float* abstol, - const float* d, const float* e, lapack_int* m, - lapack_int* nsplit, float* w, lapack_int* iblock, - lapack_int* isplit, float* work, lapack_int* iwork, - lapack_int *info ); -void LAPACK_dstebz( char* range, char* order, lapack_int* n, double* vl, - double* vu, lapack_int* il, lapack_int* iu, double* abstol, - const double* d, const double* e, lapack_int* m, - lapack_int* nsplit, double* w, lapack_int* iblock, - lapack_int* isplit, double* work, lapack_int* iwork, - lapack_int *info ); -void LAPACK_sstein( lapack_int* n, const float* d, const float* e, - lapack_int* m, const float* w, const lapack_int* iblock, - const lapack_int* isplit, float* z, lapack_int* ldz, - float* work, lapack_int* iwork, lapack_int* ifailv, - lapack_int *info ); -void LAPACK_dstein( lapack_int* n, const double* d, const double* e, - lapack_int* m, const double* w, const lapack_int* iblock, - const lapack_int* isplit, double* z, lapack_int* ldz, - double* work, lapack_int* iwork, lapack_int* ifailv, - lapack_int *info ); -void LAPACK_cstein( lapack_int* n, const float* d, const float* e, - lapack_int* m, const float* w, const lapack_int* iblock, - const lapack_int* isplit, lapack_complex_float* z, - lapack_int* ldz, float* work, lapack_int* iwork, - lapack_int* ifailv, lapack_int *info ); -void LAPACK_zstein( lapack_int* n, const double* d, const double* e, - lapack_int* m, const double* w, const lapack_int* iblock, - const lapack_int* isplit, lapack_complex_double* z, - lapack_int* ldz, double* work, lapack_int* iwork, - lapack_int* ifailv, lapack_int *info ); -void LAPACK_sdisna( char* job, lapack_int* m, lapack_int* n, const float* d, - float* sep, lapack_int *info ); -void LAPACK_ddisna( char* job, lapack_int* m, lapack_int* n, const double* d, - double* sep, lapack_int *info ); -void LAPACK_ssygst( lapack_int* itype, char* uplo, lapack_int* n, float* a, - lapack_int* lda, const float* b, lapack_int* ldb, - lapack_int *info ); -void LAPACK_dsygst( lapack_int* itype, char* uplo, lapack_int* n, double* a, - lapack_int* lda, const double* b, lapack_int* ldb, - lapack_int *info ); -void LAPACK_chegst( lapack_int* itype, char* uplo, lapack_int* n, - lapack_complex_float* a, lapack_int* lda, - const lapack_complex_float* b, lapack_int* ldb, - lapack_int *info ); -void LAPACK_zhegst( lapack_int* itype, char* uplo, lapack_int* n, - lapack_complex_double* a, lapack_int* lda, - const lapack_complex_double* b, lapack_int* ldb, - lapack_int *info ); -void LAPACK_sspgst( lapack_int* itype, char* uplo, lapack_int* n, float* ap, - const float* bp, lapack_int *info ); -void LAPACK_dspgst( lapack_int* itype, char* uplo, lapack_int* n, double* ap, - const double* bp, lapack_int *info ); -void LAPACK_chpgst( lapack_int* itype, char* uplo, lapack_int* n, - lapack_complex_float* ap, const lapack_complex_float* bp, - lapack_int *info ); -void LAPACK_zhpgst( lapack_int* itype, char* uplo, lapack_int* n, - lapack_complex_double* ap, const lapack_complex_double* bp, - lapack_int *info ); -void LAPACK_ssbgst( char* vect, char* uplo, lapack_int* n, lapack_int* ka, - lapack_int* kb, float* ab, lapack_int* ldab, - const float* bb, lapack_int* ldbb, float* x, - lapack_int* ldx, float* work, lapack_int *info ); -void LAPACK_dsbgst( char* vect, char* uplo, lapack_int* n, lapack_int* ka, - lapack_int* kb, double* ab, lapack_int* ldab, - const double* bb, lapack_int* ldbb, double* x, - lapack_int* ldx, double* work, lapack_int *info ); -void LAPACK_chbgst( char* vect, char* uplo, lapack_int* n, lapack_int* ka, - lapack_int* kb, lapack_complex_float* ab, lapack_int* ldab, - const lapack_complex_float* bb, lapack_int* ldbb, - lapack_complex_float* x, lapack_int* ldx, - lapack_complex_float* work, float* rwork, - lapack_int *info ); -void LAPACK_zhbgst( char* vect, char* uplo, lapack_int* n, lapack_int* ka, - lapack_int* kb, lapack_complex_double* ab, lapack_int* ldab, - const lapack_complex_double* bb, lapack_int* ldbb, - lapack_complex_double* x, lapack_int* ldx, - lapack_complex_double* work, double* rwork, - lapack_int *info ); -void LAPACK_spbstf( char* uplo, lapack_int* n, lapack_int* kb, float* bb, - lapack_int* ldbb, lapack_int *info ); -void LAPACK_dpbstf( char* uplo, lapack_int* n, lapack_int* kb, double* bb, - lapack_int* ldbb, lapack_int *info ); -void LAPACK_cpbstf( char* uplo, lapack_int* n, lapack_int* kb, - lapack_complex_float* bb, lapack_int* ldbb, - lapack_int *info ); -void LAPACK_zpbstf( char* uplo, lapack_int* n, lapack_int* kb, - lapack_complex_double* bb, lapack_int* ldbb, - lapack_int *info ); -void LAPACK_sgehrd( lapack_int* n, lapack_int* ilo, lapack_int* ihi, float* a, - lapack_int* lda, float* tau, float* work, lapack_int* lwork, - lapack_int *info ); -void LAPACK_dgehrd( lapack_int* n, lapack_int* ilo, lapack_int* ihi, double* a, - lapack_int* lda, double* tau, double* work, - lapack_int* lwork, lapack_int *info ); -void LAPACK_cgehrd( lapack_int* n, lapack_int* ilo, lapack_int* ihi, - lapack_complex_float* a, lapack_int* lda, - lapack_complex_float* tau, lapack_complex_float* work, - lapack_int* lwork, lapack_int *info ); -void LAPACK_zgehrd( lapack_int* n, lapack_int* ilo, lapack_int* ihi, - lapack_complex_double* a, lapack_int* lda, - lapack_complex_double* tau, lapack_complex_double* work, - lapack_int* lwork, lapack_int *info ); -void LAPACK_sorghr( lapack_int* n, lapack_int* ilo, lapack_int* ihi, float* a, - lapack_int* lda, const float* tau, float* work, - lapack_int* lwork, lapack_int *info ); -void LAPACK_dorghr( lapack_int* n, lapack_int* ilo, lapack_int* ihi, double* a, - lapack_int* lda, const double* tau, double* work, - lapack_int* lwork, lapack_int *info ); -void LAPACK_sormhr( char* side, char* trans, lapack_int* m, lapack_int* n, - lapack_int* ilo, lapack_int* ihi, const float* a, - lapack_int* lda, const float* tau, float* c, - lapack_int* ldc, float* work, lapack_int* lwork, - lapack_int *info ); -void LAPACK_dormhr( char* side, char* trans, lapack_int* m, lapack_int* n, - lapack_int* ilo, lapack_int* ihi, const double* a, - lapack_int* lda, const double* tau, double* c, - lapack_int* ldc, double* work, lapack_int* lwork, - lapack_int *info ); -void LAPACK_cunghr( lapack_int* n, lapack_int* ilo, lapack_int* ihi, - lapack_complex_float* a, lapack_int* lda, - const lapack_complex_float* tau, lapack_complex_float* work, - lapack_int* lwork, lapack_int *info ); -void LAPACK_zunghr( lapack_int* n, lapack_int* ilo, lapack_int* ihi, - lapack_complex_double* a, lapack_int* lda, - const lapack_complex_double* tau, - lapack_complex_double* work, lapack_int* lwork, - lapack_int *info ); -void LAPACK_cunmhr( char* side, char* trans, lapack_int* m, lapack_int* n, - lapack_int* ilo, lapack_int* ihi, - const lapack_complex_float* a, lapack_int* lda, - const lapack_complex_float* tau, lapack_complex_float* c, - lapack_int* ldc, lapack_complex_float* work, - lapack_int* lwork, lapack_int *info ); -void LAPACK_zunmhr( char* side, char* trans, lapack_int* m, lapack_int* n, - lapack_int* ilo, lapack_int* ihi, - const lapack_complex_double* a, lapack_int* lda, - const lapack_complex_double* tau, lapack_complex_double* c, - lapack_int* ldc, lapack_complex_double* work, - lapack_int* lwork, lapack_int *info ); -void LAPACK_sgebal( char* job, lapack_int* n, float* a, lapack_int* lda, - lapack_int* ilo, lapack_int* ihi, float* scale, - lapack_int *info ); -void LAPACK_dgebal( char* job, lapack_int* n, double* a, lapack_int* lda, - lapack_int* ilo, lapack_int* ihi, double* scale, - lapack_int *info ); -void LAPACK_cgebal( char* job, lapack_int* n, lapack_complex_float* a, - lapack_int* lda, lapack_int* ilo, lapack_int* ihi, - float* scale, lapack_int *info ); -void LAPACK_zgebal( char* job, lapack_int* n, lapack_complex_double* a, - lapack_int* lda, lapack_int* ilo, lapack_int* ihi, - double* scale, lapack_int *info ); -void LAPACK_sgebak( char* job, char* side, lapack_int* n, lapack_int* ilo, - lapack_int* ihi, const float* scale, lapack_int* m, - float* v, lapack_int* ldv, lapack_int *info ); -void LAPACK_dgebak( char* job, char* side, lapack_int* n, lapack_int* ilo, - lapack_int* ihi, const double* scale, lapack_int* m, - double* v, lapack_int* ldv, lapack_int *info ); -void LAPACK_cgebak( char* job, char* side, lapack_int* n, lapack_int* ilo, - lapack_int* ihi, const float* scale, lapack_int* m, - lapack_complex_float* v, lapack_int* ldv, - lapack_int *info ); -void LAPACK_zgebak( char* job, char* side, lapack_int* n, lapack_int* ilo, - lapack_int* ihi, const double* scale, lapack_int* m, - lapack_complex_double* v, lapack_int* ldv, - lapack_int *info ); -void LAPACK_shseqr( char* job, char* compz, lapack_int* n, lapack_int* ilo, - lapack_int* ihi, float* h, lapack_int* ldh, float* wr, - float* wi, float* z, lapack_int* ldz, float* work, - lapack_int* lwork, lapack_int *info ); -void LAPACK_dhseqr( char* job, char* compz, lapack_int* n, lapack_int* ilo, - lapack_int* ihi, double* h, lapack_int* ldh, double* wr, - double* wi, double* z, lapack_int* ldz, double* work, - lapack_int* lwork, lapack_int *info ); -void LAPACK_chseqr( char* job, char* compz, lapack_int* n, lapack_int* ilo, - lapack_int* ihi, lapack_complex_float* h, lapack_int* ldh, - lapack_complex_float* w, lapack_complex_float* z, - lapack_int* ldz, lapack_complex_float* work, - lapack_int* lwork, lapack_int *info ); -void LAPACK_zhseqr( char* job, char* compz, lapack_int* n, lapack_int* ilo, - lapack_int* ihi, lapack_complex_double* h, lapack_int* ldh, - lapack_complex_double* w, lapack_complex_double* z, - lapack_int* ldz, lapack_complex_double* work, - lapack_int* lwork, lapack_int *info ); -void LAPACK_shsein( char* job, char* eigsrc, char* initv, - lapack_logical* select, lapack_int* n, const float* h, - lapack_int* ldh, float* wr, const float* wi, float* vl, - lapack_int* ldvl, float* vr, lapack_int* ldvr, - lapack_int* mm, lapack_int* m, float* work, - lapack_int* ifaill, lapack_int* ifailr, lapack_int *info ); -void LAPACK_dhsein( char* job, char* eigsrc, char* initv, - lapack_logical* select, lapack_int* n, const double* h, - lapack_int* ldh, double* wr, const double* wi, double* vl, - lapack_int* ldvl, double* vr, lapack_int* ldvr, - lapack_int* mm, lapack_int* m, double* work, - lapack_int* ifaill, lapack_int* ifailr, lapack_int *info ); -void LAPACK_chsein( char* job, char* eigsrc, char* initv, - const lapack_logical* select, lapack_int* n, - const lapack_complex_float* h, lapack_int* ldh, - lapack_complex_float* w, lapack_complex_float* vl, - lapack_int* ldvl, lapack_complex_float* vr, - lapack_int* ldvr, lapack_int* mm, lapack_int* m, - lapack_complex_float* work, float* rwork, - lapack_int* ifaill, lapack_int* ifailr, lapack_int *info ); -void LAPACK_zhsein( char* job, char* eigsrc, char* initv, - const lapack_logical* select, lapack_int* n, - const lapack_complex_double* h, lapack_int* ldh, - lapack_complex_double* w, lapack_complex_double* vl, - lapack_int* ldvl, lapack_complex_double* vr, - lapack_int* ldvr, lapack_int* mm, lapack_int* m, - lapack_complex_double* work, double* rwork, - lapack_int* ifaill, lapack_int* ifailr, lapack_int *info ); -void LAPACK_strevc( char* side, char* howmny, lapack_logical* select, - lapack_int* n, const float* t, lapack_int* ldt, float* vl, - lapack_int* ldvl, float* vr, lapack_int* ldvr, - lapack_int* mm, lapack_int* m, float* work, - lapack_int *info ); -void LAPACK_dtrevc( char* side, char* howmny, lapack_logical* select, - lapack_int* n, const double* t, lapack_int* ldt, double* vl, - lapack_int* ldvl, double* vr, lapack_int* ldvr, - lapack_int* mm, lapack_int* m, double* work, - lapack_int *info ); -void LAPACK_ctrevc( char* side, char* howmny, const lapack_logical* select, - lapack_int* n, lapack_complex_float* t, lapack_int* ldt, - lapack_complex_float* vl, lapack_int* ldvl, - lapack_complex_float* vr, lapack_int* ldvr, lapack_int* mm, - lapack_int* m, lapack_complex_float* work, float* rwork, - lapack_int *info ); -void LAPACK_ztrevc( char* side, char* howmny, const lapack_logical* select, - lapack_int* n, lapack_complex_double* t, lapack_int* ldt, - lapack_complex_double* vl, lapack_int* ldvl, - lapack_complex_double* vr, lapack_int* ldvr, lapack_int* mm, - lapack_int* m, lapack_complex_double* work, double* rwork, - lapack_int *info ); -void LAPACK_strsna( char* job, char* howmny, const lapack_logical* select, - lapack_int* n, const float* t, lapack_int* ldt, - const float* vl, lapack_int* ldvl, const float* vr, - lapack_int* ldvr, float* s, float* sep, lapack_int* mm, - lapack_int* m, float* work, lapack_int* ldwork, - lapack_int* iwork, lapack_int *info ); -void LAPACK_dtrsna( char* job, char* howmny, const lapack_logical* select, - lapack_int* n, const double* t, lapack_int* ldt, - const double* vl, lapack_int* ldvl, const double* vr, - lapack_int* ldvr, double* s, double* sep, lapack_int* mm, - lapack_int* m, double* work, lapack_int* ldwork, - lapack_int* iwork, lapack_int *info ); -void LAPACK_ctrsna( char* job, char* howmny, const lapack_logical* select, - lapack_int* n, const lapack_complex_float* t, - lapack_int* ldt, const lapack_complex_float* vl, - lapack_int* ldvl, const lapack_complex_float* vr, - lapack_int* ldvr, float* s, float* sep, lapack_int* mm, - lapack_int* m, lapack_complex_float* work, - lapack_int* ldwork, float* rwork, lapack_int *info ); -void LAPACK_ztrsna( char* job, char* howmny, const lapack_logical* select, - lapack_int* n, const lapack_complex_double* t, - lapack_int* ldt, const lapack_complex_double* vl, - lapack_int* ldvl, const lapack_complex_double* vr, - lapack_int* ldvr, double* s, double* sep, lapack_int* mm, - lapack_int* m, lapack_complex_double* work, - lapack_int* ldwork, double* rwork, lapack_int *info ); -void LAPACK_strexc( char* compq, lapack_int* n, float* t, lapack_int* ldt, - float* q, lapack_int* ldq, lapack_int* ifst, - lapack_int* ilst, float* work, lapack_int *info ); -void LAPACK_dtrexc( char* compq, lapack_int* n, double* t, lapack_int* ldt, - double* q, lapack_int* ldq, lapack_int* ifst, - lapack_int* ilst, double* work, lapack_int *info ); -void LAPACK_ctrexc( char* compq, lapack_int* n, lapack_complex_float* t, - lapack_int* ldt, lapack_complex_float* q, lapack_int* ldq, - lapack_int* ifst, lapack_int* ilst, lapack_int *info ); -void LAPACK_ztrexc( char* compq, lapack_int* n, lapack_complex_double* t, - lapack_int* ldt, lapack_complex_double* q, lapack_int* ldq, - lapack_int* ifst, lapack_int* ilst, lapack_int *info ); -void LAPACK_strsen( char* job, char* compq, const lapack_logical* select, - lapack_int* n, float* t, lapack_int* ldt, float* q, - lapack_int* ldq, float* wr, float* wi, lapack_int* m, - float* s, float* sep, float* work, lapack_int* lwork, - lapack_int* iwork, lapack_int* liwork, lapack_int *info ); -void LAPACK_dtrsen( char* job, char* compq, const lapack_logical* select, - lapack_int* n, double* t, lapack_int* ldt, double* q, - lapack_int* ldq, double* wr, double* wi, lapack_int* m, - double* s, double* sep, double* work, lapack_int* lwork, - lapack_int* iwork, lapack_int* liwork, lapack_int *info ); -void LAPACK_ctrsen( char* job, char* compq, const lapack_logical* select, - lapack_int* n, lapack_complex_float* t, lapack_int* ldt, - lapack_complex_float* q, lapack_int* ldq, - lapack_complex_float* w, lapack_int* m, float* s, - float* sep, lapack_complex_float* work, lapack_int* lwork, - lapack_int *info ); -void LAPACK_ztrsen( char* job, char* compq, const lapack_logical* select, - lapack_int* n, lapack_complex_double* t, lapack_int* ldt, - lapack_complex_double* q, lapack_int* ldq, - lapack_complex_double* w, lapack_int* m, double* s, - double* sep, lapack_complex_double* work, lapack_int* lwork, - lapack_int *info ); -void LAPACK_strsyl( char* trana, char* tranb, lapack_int* isgn, lapack_int* m, - lapack_int* n, const float* a, lapack_int* lda, - const float* b, lapack_int* ldb, float* c, lapack_int* ldc, - float* scale, lapack_int *info ); -void LAPACK_dtrsyl( char* trana, char* tranb, lapack_int* isgn, lapack_int* m, - lapack_int* n, const double* a, lapack_int* lda, - const double* b, lapack_int* ldb, double* c, - lapack_int* ldc, double* scale, lapack_int *info ); -void LAPACK_ctrsyl( char* trana, char* tranb, lapack_int* isgn, lapack_int* m, - lapack_int* n, const lapack_complex_float* a, - lapack_int* lda, const lapack_complex_float* b, - lapack_int* ldb, lapack_complex_float* c, lapack_int* ldc, - float* scale, lapack_int *info ); -void LAPACK_ztrsyl( char* trana, char* tranb, lapack_int* isgn, lapack_int* m, - lapack_int* n, const lapack_complex_double* a, - lapack_int* lda, const lapack_complex_double* b, - lapack_int* ldb, lapack_complex_double* c, lapack_int* ldc, - double* scale, lapack_int *info ); -void LAPACK_sgghrd( char* compq, char* compz, lapack_int* n, lapack_int* ilo, - lapack_int* ihi, float* a, lapack_int* lda, float* b, - lapack_int* ldb, float* q, lapack_int* ldq, float* z, - lapack_int* ldz, lapack_int *info ); -void LAPACK_dgghrd( char* compq, char* compz, lapack_int* n, lapack_int* ilo, - lapack_int* ihi, double* a, lapack_int* lda, double* b, - lapack_int* ldb, double* q, lapack_int* ldq, double* z, - lapack_int* ldz, lapack_int *info ); -void LAPACK_cgghrd( char* compq, char* compz, lapack_int* n, lapack_int* ilo, - lapack_int* ihi, lapack_complex_float* a, lapack_int* lda, - lapack_complex_float* b, lapack_int* ldb, - lapack_complex_float* q, lapack_int* ldq, - lapack_complex_float* z, lapack_int* ldz, - lapack_int *info ); -void LAPACK_zgghrd( char* compq, char* compz, lapack_int* n, lapack_int* ilo, - lapack_int* ihi, lapack_complex_double* a, lapack_int* lda, - lapack_complex_double* b, lapack_int* ldb, - lapack_complex_double* q, lapack_int* ldq, - lapack_complex_double* z, lapack_int* ldz, - lapack_int *info ); -void LAPACK_sggbal( char* job, lapack_int* n, float* a, lapack_int* lda, - float* b, lapack_int* ldb, lapack_int* ilo, lapack_int* ihi, - float* lscale, float* rscale, float* work, - lapack_int *info ); -void LAPACK_dggbal( char* job, lapack_int* n, double* a, lapack_int* lda, - double* b, lapack_int* ldb, lapack_int* ilo, - lapack_int* ihi, double* lscale, double* rscale, - double* work, lapack_int *info ); -void LAPACK_cggbal( char* job, lapack_int* n, lapack_complex_float* a, - lapack_int* lda, lapack_complex_float* b, lapack_int* ldb, - lapack_int* ilo, lapack_int* ihi, float* lscale, - float* rscale, float* work, lapack_int *info ); -void LAPACK_zggbal( char* job, lapack_int* n, lapack_complex_double* a, - lapack_int* lda, lapack_complex_double* b, lapack_int* ldb, - lapack_int* ilo, lapack_int* ihi, double* lscale, - double* rscale, double* work, lapack_int *info ); -void LAPACK_sggbak( char* job, char* side, lapack_int* n, lapack_int* ilo, - lapack_int* ihi, const float* lscale, const float* rscale, - lapack_int* m, float* v, lapack_int* ldv, - lapack_int *info ); -void LAPACK_dggbak( char* job, char* side, lapack_int* n, lapack_int* ilo, - lapack_int* ihi, const double* lscale, const double* rscale, - lapack_int* m, double* v, lapack_int* ldv, - lapack_int *info ); -void LAPACK_cggbak( char* job, char* side, lapack_int* n, lapack_int* ilo, - lapack_int* ihi, const float* lscale, const float* rscale, - lapack_int* m, lapack_complex_float* v, lapack_int* ldv, - lapack_int *info ); -void LAPACK_zggbak( char* job, char* side, lapack_int* n, lapack_int* ilo, - lapack_int* ihi, const double* lscale, const double* rscale, - lapack_int* m, lapack_complex_double* v, lapack_int* ldv, - lapack_int *info ); -void LAPACK_shgeqz( char* job, char* compq, char* compz, lapack_int* n, - lapack_int* ilo, lapack_int* ihi, float* h, lapack_int* ldh, - float* t, lapack_int* ldt, float* alphar, float* alphai, - float* beta, float* q, lapack_int* ldq, float* z, - lapack_int* ldz, float* work, lapack_int* lwork, - lapack_int *info ); -void LAPACK_dhgeqz( char* job, char* compq, char* compz, lapack_int* n, - lapack_int* ilo, lapack_int* ihi, double* h, - lapack_int* ldh, double* t, lapack_int* ldt, double* alphar, - double* alphai, double* beta, double* q, lapack_int* ldq, - double* z, lapack_int* ldz, double* work, lapack_int* lwork, - lapack_int *info ); -void LAPACK_chgeqz( char* job, char* compq, char* compz, lapack_int* n, - lapack_int* ilo, lapack_int* ihi, lapack_complex_float* h, - lapack_int* ldh, lapack_complex_float* t, lapack_int* ldt, - lapack_complex_float* alpha, lapack_complex_float* beta, - lapack_complex_float* q, lapack_int* ldq, - lapack_complex_float* z, lapack_int* ldz, - lapack_complex_float* work, lapack_int* lwork, float* rwork, - lapack_int *info ); -void LAPACK_zhgeqz( char* job, char* compq, char* compz, lapack_int* n, - lapack_int* ilo, lapack_int* ihi, lapack_complex_double* h, - lapack_int* ldh, lapack_complex_double* t, lapack_int* ldt, - lapack_complex_double* alpha, lapack_complex_double* beta, - lapack_complex_double* q, lapack_int* ldq, - lapack_complex_double* z, lapack_int* ldz, - lapack_complex_double* work, lapack_int* lwork, - double* rwork, lapack_int *info ); -void LAPACK_stgevc( char* side, char* howmny, const lapack_logical* select, - lapack_int* n, const float* s, lapack_int* lds, - const float* p, lapack_int* ldp, float* vl, - lapack_int* ldvl, float* vr, lapack_int* ldvr, - lapack_int* mm, lapack_int* m, float* work, - lapack_int *info ); -void LAPACK_dtgevc( char* side, char* howmny, const lapack_logical* select, - lapack_int* n, const double* s, lapack_int* lds, - const double* p, lapack_int* ldp, double* vl, - lapack_int* ldvl, double* vr, lapack_int* ldvr, - lapack_int* mm, lapack_int* m, double* work, - lapack_int *info ); -void LAPACK_ctgevc( char* side, char* howmny, const lapack_logical* select, - lapack_int* n, const lapack_complex_float* s, - lapack_int* lds, const lapack_complex_float* p, - lapack_int* ldp, lapack_complex_float* vl, lapack_int* ldvl, - lapack_complex_float* vr, lapack_int* ldvr, lapack_int* mm, - lapack_int* m, lapack_complex_float* work, float* rwork, - lapack_int *info ); -void LAPACK_ztgevc( char* side, char* howmny, const lapack_logical* select, - lapack_int* n, const lapack_complex_double* s, - lapack_int* lds, const lapack_complex_double* p, - lapack_int* ldp, lapack_complex_double* vl, - lapack_int* ldvl, lapack_complex_double* vr, - lapack_int* ldvr, lapack_int* mm, lapack_int* m, - lapack_complex_double* work, double* rwork, - lapack_int *info ); -void LAPACK_stgexc( lapack_logical* wantq, lapack_logical* wantz, lapack_int* n, - float* a, lapack_int* lda, float* b, lapack_int* ldb, - float* q, lapack_int* ldq, float* z, lapack_int* ldz, - lapack_int* ifst, lapack_int* ilst, float* work, - lapack_int* lwork, lapack_int *info ); -void LAPACK_dtgexc( lapack_logical* wantq, lapack_logical* wantz, lapack_int* n, - double* a, lapack_int* lda, double* b, lapack_int* ldb, - double* q, lapack_int* ldq, double* z, lapack_int* ldz, - lapack_int* ifst, lapack_int* ilst, double* work, - lapack_int* lwork, lapack_int *info ); -void LAPACK_ctgexc( lapack_logical* wantq, lapack_logical* wantz, lapack_int* n, - lapack_complex_float* a, lapack_int* lda, - lapack_complex_float* b, lapack_int* ldb, - lapack_complex_float* q, lapack_int* ldq, - lapack_complex_float* z, lapack_int* ldz, lapack_int* ifst, - lapack_int* ilst, lapack_int *info ); -void LAPACK_ztgexc( lapack_logical* wantq, lapack_logical* wantz, lapack_int* n, - lapack_complex_double* a, lapack_int* lda, - lapack_complex_double* b, lapack_int* ldb, - lapack_complex_double* q, lapack_int* ldq, - lapack_complex_double* z, lapack_int* ldz, lapack_int* ifst, - lapack_int* ilst, lapack_int *info ); -void LAPACK_stgsen( lapack_int* ijob, lapack_logical* wantq, - lapack_logical* wantz, const lapack_logical* select, - lapack_int* n, float* a, lapack_int* lda, float* b, - lapack_int* ldb, float* alphar, float* alphai, float* beta, - float* q, lapack_int* ldq, float* z, lapack_int* ldz, - lapack_int* m, float* pl, float* pr, float* dif, - float* work, lapack_int* lwork, lapack_int* iwork, - lapack_int* liwork, lapack_int *info ); -void LAPACK_dtgsen( lapack_int* ijob, lapack_logical* wantq, - lapack_logical* wantz, const lapack_logical* select, - lapack_int* n, double* a, lapack_int* lda, double* b, - lapack_int* ldb, double* alphar, double* alphai, - double* beta, double* q, lapack_int* ldq, double* z, - lapack_int* ldz, lapack_int* m, double* pl, double* pr, - double* dif, double* work, lapack_int* lwork, - lapack_int* iwork, lapack_int* liwork, lapack_int *info ); -void LAPACK_ctgsen( lapack_int* ijob, lapack_logical* wantq, - lapack_logical* wantz, const lapack_logical* select, - lapack_int* n, lapack_complex_float* a, lapack_int* lda, - lapack_complex_float* b, lapack_int* ldb, - lapack_complex_float* alpha, lapack_complex_float* beta, - lapack_complex_float* q, lapack_int* ldq, - lapack_complex_float* z, lapack_int* ldz, lapack_int* m, - float* pl, float* pr, float* dif, - lapack_complex_float* work, lapack_int* lwork, - lapack_int* iwork, lapack_int* liwork, lapack_int *info ); -void LAPACK_ztgsen( lapack_int* ijob, lapack_logical* wantq, - lapack_logical* wantz, const lapack_logical* select, - lapack_int* n, lapack_complex_double* a, lapack_int* lda, - lapack_complex_double* b, lapack_int* ldb, - lapack_complex_double* alpha, lapack_complex_double* beta, - lapack_complex_double* q, lapack_int* ldq, - lapack_complex_double* z, lapack_int* ldz, lapack_int* m, - double* pl, double* pr, double* dif, - lapack_complex_double* work, lapack_int* lwork, - lapack_int* iwork, lapack_int* liwork, lapack_int *info ); -void LAPACK_stgsyl( char* trans, lapack_int* ijob, lapack_int* m, lapack_int* n, - const float* a, lapack_int* lda, const float* b, - lapack_int* ldb, float* c, lapack_int* ldc, const float* d, - lapack_int* ldd, const float* e, lapack_int* lde, float* f, - lapack_int* ldf, float* scale, float* dif, float* work, - lapack_int* lwork, lapack_int* iwork, lapack_int *info ); -void LAPACK_dtgsyl( char* trans, lapack_int* ijob, lapack_int* m, lapack_int* n, - const double* a, lapack_int* lda, const double* b, - lapack_int* ldb, double* c, lapack_int* ldc, - const double* d, lapack_int* ldd, const double* e, - lapack_int* lde, double* f, lapack_int* ldf, double* scale, - double* dif, double* work, lapack_int* lwork, - lapack_int* iwork, lapack_int *info ); -void LAPACK_ctgsyl( char* trans, lapack_int* ijob, lapack_int* m, lapack_int* n, - const lapack_complex_float* a, lapack_int* lda, - const lapack_complex_float* b, lapack_int* ldb, - lapack_complex_float* c, lapack_int* ldc, - const lapack_complex_float* d, lapack_int* ldd, - const lapack_complex_float* e, lapack_int* lde, - lapack_complex_float* f, lapack_int* ldf, float* scale, - float* dif, lapack_complex_float* work, lapack_int* lwork, - lapack_int* iwork, lapack_int *info ); -void LAPACK_ztgsyl( char* trans, lapack_int* ijob, lapack_int* m, lapack_int* n, - const lapack_complex_double* a, lapack_int* lda, - const lapack_complex_double* b, lapack_int* ldb, - lapack_complex_double* c, lapack_int* ldc, - const lapack_complex_double* d, lapack_int* ldd, - const lapack_complex_double* e, lapack_int* lde, - lapack_complex_double* f, lapack_int* ldf, double* scale, - double* dif, lapack_complex_double* work, lapack_int* lwork, - lapack_int* iwork, lapack_int *info ); -void LAPACK_stgsna( char* job, char* howmny, const lapack_logical* select, - lapack_int* n, const float* a, lapack_int* lda, - const float* b, lapack_int* ldb, const float* vl, - lapack_int* ldvl, const float* vr, lapack_int* ldvr, - float* s, float* dif, lapack_int* mm, lapack_int* m, - float* work, lapack_int* lwork, lapack_int* iwork, - lapack_int *info ); -void LAPACK_dtgsna( char* job, char* howmny, const lapack_logical* select, - lapack_int* n, const double* a, lapack_int* lda, - const double* b, lapack_int* ldb, const double* vl, - lapack_int* ldvl, const double* vr, lapack_int* ldvr, - double* s, double* dif, lapack_int* mm, lapack_int* m, - double* work, lapack_int* lwork, lapack_int* iwork, - lapack_int *info ); -void LAPACK_ctgsna( char* job, char* howmny, const lapack_logical* select, - lapack_int* n, const lapack_complex_float* a, - lapack_int* lda, const lapack_complex_float* b, - lapack_int* ldb, const lapack_complex_float* vl, - lapack_int* ldvl, const lapack_complex_float* vr, - lapack_int* ldvr, float* s, float* dif, lapack_int* mm, - lapack_int* m, lapack_complex_float* work, - lapack_int* lwork, lapack_int* iwork, lapack_int *info ); -void LAPACK_ztgsna( char* job, char* howmny, const lapack_logical* select, - lapack_int* n, const lapack_complex_double* a, - lapack_int* lda, const lapack_complex_double* b, - lapack_int* ldb, const lapack_complex_double* vl, - lapack_int* ldvl, const lapack_complex_double* vr, - lapack_int* ldvr, double* s, double* dif, lapack_int* mm, - lapack_int* m, lapack_complex_double* work, - lapack_int* lwork, lapack_int* iwork, lapack_int *info ); -void LAPACK_sggsvp( char* jobu, char* jobv, char* jobq, lapack_int* m, - lapack_int* p, lapack_int* n, float* a, lapack_int* lda, - float* b, lapack_int* ldb, float* tola, float* tolb, - lapack_int* k, lapack_int* l, float* u, lapack_int* ldu, - float* v, lapack_int* ldv, float* q, lapack_int* ldq, - lapack_int* iwork, float* tau, float* work, - lapack_int *info ); -void LAPACK_dggsvp( char* jobu, char* jobv, char* jobq, lapack_int* m, - lapack_int* p, lapack_int* n, double* a, lapack_int* lda, - double* b, lapack_int* ldb, double* tola, double* tolb, - lapack_int* k, lapack_int* l, double* u, lapack_int* ldu, - double* v, lapack_int* ldv, double* q, lapack_int* ldq, - lapack_int* iwork, double* tau, double* work, - lapack_int *info ); -void LAPACK_cggsvp( char* jobu, char* jobv, char* jobq, lapack_int* m, - lapack_int* p, lapack_int* n, lapack_complex_float* a, - lapack_int* lda, lapack_complex_float* b, lapack_int* ldb, - float* tola, float* tolb, lapack_int* k, lapack_int* l, - lapack_complex_float* u, lapack_int* ldu, - lapack_complex_float* v, lapack_int* ldv, - lapack_complex_float* q, lapack_int* ldq, lapack_int* iwork, - float* rwork, lapack_complex_float* tau, - lapack_complex_float* work, lapack_int *info ); -void LAPACK_zggsvp( char* jobu, char* jobv, char* jobq, lapack_int* m, - lapack_int* p, lapack_int* n, lapack_complex_double* a, - lapack_int* lda, lapack_complex_double* b, lapack_int* ldb, - double* tola, double* tolb, lapack_int* k, lapack_int* l, - lapack_complex_double* u, lapack_int* ldu, - lapack_complex_double* v, lapack_int* ldv, - lapack_complex_double* q, lapack_int* ldq, - lapack_int* iwork, double* rwork, - lapack_complex_double* tau, lapack_complex_double* work, - lapack_int *info ); -void LAPACK_stgsja( char* jobu, char* jobv, char* jobq, lapack_int* m, - lapack_int* p, lapack_int* n, lapack_int* k, lapack_int* l, - float* a, lapack_int* lda, float* b, lapack_int* ldb, - float* tola, float* tolb, float* alpha, float* beta, - float* u, lapack_int* ldu, float* v, lapack_int* ldv, - float* q, lapack_int* ldq, float* work, lapack_int* ncycle, - lapack_int *info ); -void LAPACK_dtgsja( char* jobu, char* jobv, char* jobq, lapack_int* m, - lapack_int* p, lapack_int* n, lapack_int* k, lapack_int* l, - double* a, lapack_int* lda, double* b, lapack_int* ldb, - double* tola, double* tolb, double* alpha, double* beta, - double* u, lapack_int* ldu, double* v, lapack_int* ldv, - double* q, lapack_int* ldq, double* work, - lapack_int* ncycle, lapack_int *info ); -void LAPACK_ctgsja( char* jobu, char* jobv, char* jobq, lapack_int* m, - lapack_int* p, lapack_int* n, lapack_int* k, lapack_int* l, - lapack_complex_float* a, lapack_int* lda, - lapack_complex_float* b, lapack_int* ldb, float* tola, - float* tolb, float* alpha, float* beta, - lapack_complex_float* u, lapack_int* ldu, - lapack_complex_float* v, lapack_int* ldv, - lapack_complex_float* q, lapack_int* ldq, - lapack_complex_float* work, lapack_int* ncycle, - lapack_int *info ); -void LAPACK_ztgsja( char* jobu, char* jobv, char* jobq, lapack_int* m, - lapack_int* p, lapack_int* n, lapack_int* k, lapack_int* l, - lapack_complex_double* a, lapack_int* lda, - lapack_complex_double* b, lapack_int* ldb, double* tola, - double* tolb, double* alpha, double* beta, - lapack_complex_double* u, lapack_int* ldu, - lapack_complex_double* v, lapack_int* ldv, - lapack_complex_double* q, lapack_int* ldq, - lapack_complex_double* work, lapack_int* ncycle, - lapack_int *info ); -void LAPACK_sgels( char* trans, lapack_int* m, lapack_int* n, lapack_int* nrhs, - float* a, lapack_int* lda, float* b, lapack_int* ldb, - float* work, lapack_int* lwork, lapack_int *info ); -void LAPACK_dgels( char* trans, lapack_int* m, lapack_int* n, lapack_int* nrhs, - double* a, lapack_int* lda, double* b, lapack_int* ldb, - double* work, lapack_int* lwork, lapack_int *info ); -void LAPACK_cgels( char* trans, lapack_int* m, lapack_int* n, lapack_int* nrhs, - lapack_complex_float* a, lapack_int* lda, - lapack_complex_float* b, lapack_int* ldb, - lapack_complex_float* work, lapack_int* lwork, - lapack_int *info ); -void LAPACK_zgels( char* trans, lapack_int* m, lapack_int* n, lapack_int* nrhs, - lapack_complex_double* a, lapack_int* lda, - lapack_complex_double* b, lapack_int* ldb, - lapack_complex_double* work, lapack_int* lwork, - lapack_int *info ); -void LAPACK_sgelsy( lapack_int* m, lapack_int* n, lapack_int* nrhs, float* a, - lapack_int* lda, float* b, lapack_int* ldb, - lapack_int* jpvt, float* rcond, lapack_int* rank, - float* work, lapack_int* lwork, lapack_int *info ); -void LAPACK_dgelsy( lapack_int* m, lapack_int* n, lapack_int* nrhs, double* a, - lapack_int* lda, double* b, lapack_int* ldb, - lapack_int* jpvt, double* rcond, lapack_int* rank, - double* work, lapack_int* lwork, lapack_int *info ); -void LAPACK_cgelsy( lapack_int* m, lapack_int* n, lapack_int* nrhs, - lapack_complex_float* a, lapack_int* lda, - lapack_complex_float* b, lapack_int* ldb, lapack_int* jpvt, - float* rcond, lapack_int* rank, lapack_complex_float* work, - lapack_int* lwork, float* rwork, lapack_int *info ); -void LAPACK_zgelsy( lapack_int* m, lapack_int* n, lapack_int* nrhs, - lapack_complex_double* a, lapack_int* lda, - lapack_complex_double* b, lapack_int* ldb, lapack_int* jpvt, - double* rcond, lapack_int* rank, - lapack_complex_double* work, lapack_int* lwork, - double* rwork, lapack_int *info ); -void LAPACK_sgelss( lapack_int* m, lapack_int* n, lapack_int* nrhs, float* a, - lapack_int* lda, float* b, lapack_int* ldb, float* s, - float* rcond, lapack_int* rank, float* work, - lapack_int* lwork, lapack_int *info ); -void LAPACK_dgelss( lapack_int* m, lapack_int* n, lapack_int* nrhs, double* a, - lapack_int* lda, double* b, lapack_int* ldb, double* s, - double* rcond, lapack_int* rank, double* work, - lapack_int* lwork, lapack_int *info ); -void LAPACK_cgelss( lapack_int* m, lapack_int* n, lapack_int* nrhs, - lapack_complex_float* a, lapack_int* lda, - lapack_complex_float* b, lapack_int* ldb, float* s, - float* rcond, lapack_int* rank, lapack_complex_float* work, - lapack_int* lwork, float* rwork, lapack_int *info ); -void LAPACK_zgelss( lapack_int* m, lapack_int* n, lapack_int* nrhs, - lapack_complex_double* a, lapack_int* lda, - lapack_complex_double* b, lapack_int* ldb, double* s, - double* rcond, lapack_int* rank, - lapack_complex_double* work, lapack_int* lwork, - double* rwork, lapack_int *info ); -void LAPACK_sgelsd( lapack_int* m, lapack_int* n, lapack_int* nrhs, float* a, - lapack_int* lda, float* b, lapack_int* ldb, float* s, - float* rcond, lapack_int* rank, float* work, - lapack_int* lwork, lapack_int* iwork, lapack_int *info ); -void LAPACK_dgelsd( lapack_int* m, lapack_int* n, lapack_int* nrhs, double* a, - lapack_int* lda, double* b, lapack_int* ldb, double* s, - double* rcond, lapack_int* rank, double* work, - lapack_int* lwork, lapack_int* iwork, lapack_int *info ); -void LAPACK_cgelsd( lapack_int* m, lapack_int* n, lapack_int* nrhs, - lapack_complex_float* a, lapack_int* lda, - lapack_complex_float* b, lapack_int* ldb, float* s, - float* rcond, lapack_int* rank, lapack_complex_float* work, - lapack_int* lwork, float* rwork, lapack_int* iwork, - lapack_int *info ); -void LAPACK_zgelsd( lapack_int* m, lapack_int* n, lapack_int* nrhs, - lapack_complex_double* a, lapack_int* lda, - lapack_complex_double* b, lapack_int* ldb, double* s, - double* rcond, lapack_int* rank, - lapack_complex_double* work, lapack_int* lwork, - double* rwork, lapack_int* iwork, lapack_int *info ); -void LAPACK_sgglse( lapack_int* m, lapack_int* n, lapack_int* p, float* a, - lapack_int* lda, float* b, lapack_int* ldb, float* c, - float* d, float* x, float* work, lapack_int* lwork, - lapack_int *info ); -void LAPACK_dgglse( lapack_int* m, lapack_int* n, lapack_int* p, double* a, - lapack_int* lda, double* b, lapack_int* ldb, double* c, - double* d, double* x, double* work, lapack_int* lwork, - lapack_int *info ); -void LAPACK_cgglse( lapack_int* m, lapack_int* n, lapack_int* p, - lapack_complex_float* a, lapack_int* lda, - lapack_complex_float* b, lapack_int* ldb, - lapack_complex_float* c, lapack_complex_float* d, - lapack_complex_float* x, lapack_complex_float* work, - lapack_int* lwork, lapack_int *info ); -void LAPACK_zgglse( lapack_int* m, lapack_int* n, lapack_int* p, - lapack_complex_double* a, lapack_int* lda, - lapack_complex_double* b, lapack_int* ldb, - lapack_complex_double* c, lapack_complex_double* d, - lapack_complex_double* x, lapack_complex_double* work, - lapack_int* lwork, lapack_int *info ); -void LAPACK_sggglm( lapack_int* n, lapack_int* m, lapack_int* p, float* a, - lapack_int* lda, float* b, lapack_int* ldb, float* d, - float* x, float* y, float* work, lapack_int* lwork, - lapack_int *info ); -void LAPACK_dggglm( lapack_int* n, lapack_int* m, lapack_int* p, double* a, - lapack_int* lda, double* b, lapack_int* ldb, double* d, - double* x, double* y, double* work, lapack_int* lwork, - lapack_int *info ); -void LAPACK_cggglm( lapack_int* n, lapack_int* m, lapack_int* p, - lapack_complex_float* a, lapack_int* lda, - lapack_complex_float* b, lapack_int* ldb, - lapack_complex_float* d, lapack_complex_float* x, - lapack_complex_float* y, lapack_complex_float* work, - lapack_int* lwork, lapack_int *info ); -void LAPACK_zggglm( lapack_int* n, lapack_int* m, lapack_int* p, - lapack_complex_double* a, lapack_int* lda, - lapack_complex_double* b, lapack_int* ldb, - lapack_complex_double* d, lapack_complex_double* x, - lapack_complex_double* y, lapack_complex_double* work, - lapack_int* lwork, lapack_int *info ); -void LAPACK_ssyev( char* jobz, char* uplo, lapack_int* n, float* a, - lapack_int* lda, float* w, float* work, lapack_int* lwork, - lapack_int *info ); -void LAPACK_dsyev( char* jobz, char* uplo, lapack_int* n, double* a, - lapack_int* lda, double* w, double* work, lapack_int* lwork, - lapack_int *info ); -void LAPACK_cheev( char* jobz, char* uplo, lapack_int* n, - lapack_complex_float* a, lapack_int* lda, float* w, - lapack_complex_float* work, lapack_int* lwork, float* rwork, - lapack_int *info ); -void LAPACK_zheev( char* jobz, char* uplo, lapack_int* n, - lapack_complex_double* a, lapack_int* lda, double* w, - lapack_complex_double* work, lapack_int* lwork, - double* rwork, lapack_int *info ); -void LAPACK_ssyevd( char* jobz, char* uplo, lapack_int* n, float* a, - lapack_int* lda, float* w, float* work, lapack_int* lwork, - lapack_int* iwork, lapack_int* liwork, lapack_int *info ); -void LAPACK_dsyevd( char* jobz, char* uplo, lapack_int* n, double* a, - lapack_int* lda, double* w, double* work, lapack_int* lwork, - lapack_int* iwork, lapack_int* liwork, lapack_int *info ); -void LAPACK_cheevd( char* jobz, char* uplo, lapack_int* n, - lapack_complex_float* a, lapack_int* lda, float* w, - lapack_complex_float* work, lapack_int* lwork, float* rwork, - lapack_int* lrwork, lapack_int* iwork, lapack_int* liwork, - lapack_int *info ); -void LAPACK_zheevd( char* jobz, char* uplo, lapack_int* n, - lapack_complex_double* a, lapack_int* lda, double* w, - lapack_complex_double* work, lapack_int* lwork, - double* rwork, lapack_int* lrwork, lapack_int* iwork, - lapack_int* liwork, lapack_int *info ); -void LAPACK_ssyevx( char* jobz, char* range, char* uplo, lapack_int* n, - float* a, lapack_int* lda, float* vl, float* vu, - lapack_int* il, lapack_int* iu, float* abstol, - lapack_int* m, float* w, float* z, lapack_int* ldz, - float* work, lapack_int* lwork, lapack_int* iwork, - lapack_int* ifail, lapack_int *info ); -void LAPACK_dsyevx( char* jobz, char* range, char* uplo, lapack_int* n, - double* a, lapack_int* lda, double* vl, double* vu, - lapack_int* il, lapack_int* iu, double* abstol, - lapack_int* m, double* w, double* z, lapack_int* ldz, - double* work, lapack_int* lwork, lapack_int* iwork, - lapack_int* ifail, lapack_int *info ); -void LAPACK_cheevx( char* jobz, char* range, char* uplo, lapack_int* n, - lapack_complex_float* a, lapack_int* lda, float* vl, - float* vu, lapack_int* il, lapack_int* iu, float* abstol, - lapack_int* m, float* w, lapack_complex_float* z, - lapack_int* ldz, lapack_complex_float* work, - lapack_int* lwork, float* rwork, lapack_int* iwork, - lapack_int* ifail, lapack_int *info ); -void LAPACK_zheevx( char* jobz, char* range, char* uplo, lapack_int* n, - lapack_complex_double* a, lapack_int* lda, double* vl, - double* vu, lapack_int* il, lapack_int* iu, double* abstol, - lapack_int* m, double* w, lapack_complex_double* z, - lapack_int* ldz, lapack_complex_double* work, - lapack_int* lwork, double* rwork, lapack_int* iwork, - lapack_int* ifail, lapack_int *info ); -void LAPACK_ssyevr( char* jobz, char* range, char* uplo, lapack_int* n, - float* a, lapack_int* lda, float* vl, float* vu, - lapack_int* il, lapack_int* iu, float* abstol, - lapack_int* m, float* w, float* z, lapack_int* ldz, - lapack_int* isuppz, float* work, lapack_int* lwork, - lapack_int* iwork, lapack_int* liwork, lapack_int *info ); -void LAPACK_dsyevr( char* jobz, char* range, char* uplo, lapack_int* n, - double* a, lapack_int* lda, double* vl, double* vu, - lapack_int* il, lapack_int* iu, double* abstol, - lapack_int* m, double* w, double* z, lapack_int* ldz, - lapack_int* isuppz, double* work, lapack_int* lwork, - lapack_int* iwork, lapack_int* liwork, lapack_int *info ); -void LAPACK_cheevr( char* jobz, char* range, char* uplo, lapack_int* n, - lapack_complex_float* a, lapack_int* lda, float* vl, - float* vu, lapack_int* il, lapack_int* iu, float* abstol, - lapack_int* m, float* w, lapack_complex_float* z, - lapack_int* ldz, lapack_int* isuppz, - lapack_complex_float* work, lapack_int* lwork, float* rwork, - lapack_int* lrwork, lapack_int* iwork, lapack_int* liwork, - lapack_int *info ); -void LAPACK_zheevr( char* jobz, char* range, char* uplo, lapack_int* n, - lapack_complex_double* a, lapack_int* lda, double* vl, - double* vu, lapack_int* il, lapack_int* iu, double* abstol, - lapack_int* m, double* w, lapack_complex_double* z, - lapack_int* ldz, lapack_int* isuppz, - lapack_complex_double* work, lapack_int* lwork, - double* rwork, lapack_int* lrwork, lapack_int* iwork, - lapack_int* liwork, lapack_int *info ); -void LAPACK_sspev( char* jobz, char* uplo, lapack_int* n, float* ap, float* w, - float* z, lapack_int* ldz, float* work, lapack_int *info ); -void LAPACK_dspev( char* jobz, char* uplo, lapack_int* n, double* ap, double* w, - double* z, lapack_int* ldz, double* work, lapack_int *info ); -void LAPACK_chpev( char* jobz, char* uplo, lapack_int* n, - lapack_complex_float* ap, float* w, lapack_complex_float* z, - lapack_int* ldz, lapack_complex_float* work, float* rwork, - lapack_int *info ); -void LAPACK_zhpev( char* jobz, char* uplo, lapack_int* n, - lapack_complex_double* ap, double* w, - lapack_complex_double* z, lapack_int* ldz, - lapack_complex_double* work, double* rwork, - lapack_int *info ); -void LAPACK_sspevd( char* jobz, char* uplo, lapack_int* n, float* ap, float* w, - float* z, lapack_int* ldz, float* work, lapack_int* lwork, - lapack_int* iwork, lapack_int* liwork, lapack_int *info ); -void LAPACK_dspevd( char* jobz, char* uplo, lapack_int* n, double* ap, - double* w, double* z, lapack_int* ldz, double* work, - lapack_int* lwork, lapack_int* iwork, lapack_int* liwork, - lapack_int *info ); -void LAPACK_chpevd( char* jobz, char* uplo, lapack_int* n, - lapack_complex_float* ap, float* w, lapack_complex_float* z, - lapack_int* ldz, lapack_complex_float* work, - lapack_int* lwork, float* rwork, lapack_int* lrwork, - lapack_int* iwork, lapack_int* liwork, lapack_int *info ); -void LAPACK_zhpevd( char* jobz, char* uplo, lapack_int* n, - lapack_complex_double* ap, double* w, - lapack_complex_double* z, lapack_int* ldz, - lapack_complex_double* work, lapack_int* lwork, - double* rwork, lapack_int* lrwork, lapack_int* iwork, - lapack_int* liwork, lapack_int *info ); -void LAPACK_sspevx( char* jobz, char* range, char* uplo, lapack_int* n, - float* ap, float* vl, float* vu, lapack_int* il, - lapack_int* iu, float* abstol, lapack_int* m, float* w, - float* z, lapack_int* ldz, float* work, lapack_int* iwork, - lapack_int* ifail, lapack_int *info ); -void LAPACK_dspevx( char* jobz, char* range, char* uplo, lapack_int* n, - double* ap, double* vl, double* vu, lapack_int* il, - lapack_int* iu, double* abstol, lapack_int* m, double* w, - double* z, lapack_int* ldz, double* work, lapack_int* iwork, - lapack_int* ifail, lapack_int *info ); -void LAPACK_chpevx( char* jobz, char* range, char* uplo, lapack_int* n, - lapack_complex_float* ap, float* vl, float* vu, - lapack_int* il, lapack_int* iu, float* abstol, - lapack_int* m, float* w, lapack_complex_float* z, - lapack_int* ldz, lapack_complex_float* work, float* rwork, - lapack_int* iwork, lapack_int* ifail, lapack_int *info ); -void LAPACK_zhpevx( char* jobz, char* range, char* uplo, lapack_int* n, - lapack_complex_double* ap, double* vl, double* vu, - lapack_int* il, lapack_int* iu, double* abstol, - lapack_int* m, double* w, lapack_complex_double* z, - lapack_int* ldz, lapack_complex_double* work, double* rwork, - lapack_int* iwork, lapack_int* ifail, lapack_int *info ); -void LAPACK_ssbev( char* jobz, char* uplo, lapack_int* n, lapack_int* kd, - float* ab, lapack_int* ldab, float* w, float* z, - lapack_int* ldz, float* work, lapack_int *info ); -void LAPACK_dsbev( char* jobz, char* uplo, lapack_int* n, lapack_int* kd, - double* ab, lapack_int* ldab, double* w, double* z, - lapack_int* ldz, double* work, lapack_int *info ); -void LAPACK_chbev( char* jobz, char* uplo, lapack_int* n, lapack_int* kd, - lapack_complex_float* ab, lapack_int* ldab, float* w, - lapack_complex_float* z, lapack_int* ldz, - lapack_complex_float* work, float* rwork, lapack_int *info ); -void LAPACK_zhbev( char* jobz, char* uplo, lapack_int* n, lapack_int* kd, - lapack_complex_double* ab, lapack_int* ldab, double* w, - lapack_complex_double* z, lapack_int* ldz, - lapack_complex_double* work, double* rwork, - lapack_int *info ); -void LAPACK_ssbevd( char* jobz, char* uplo, lapack_int* n, lapack_int* kd, - float* ab, lapack_int* ldab, float* w, float* z, - lapack_int* ldz, float* work, lapack_int* lwork, - lapack_int* iwork, lapack_int* liwork, lapack_int *info ); -void LAPACK_dsbevd( char* jobz, char* uplo, lapack_int* n, lapack_int* kd, - double* ab, lapack_int* ldab, double* w, double* z, - lapack_int* ldz, double* work, lapack_int* lwork, - lapack_int* iwork, lapack_int* liwork, lapack_int *info ); -void LAPACK_chbevd( char* jobz, char* uplo, lapack_int* n, lapack_int* kd, - lapack_complex_float* ab, lapack_int* ldab, float* w, - lapack_complex_float* z, lapack_int* ldz, - lapack_complex_float* work, lapack_int* lwork, float* rwork, - lapack_int* lrwork, lapack_int* iwork, lapack_int* liwork, - lapack_int *info ); -void LAPACK_zhbevd( char* jobz, char* uplo, lapack_int* n, lapack_int* kd, - lapack_complex_double* ab, lapack_int* ldab, double* w, - lapack_complex_double* z, lapack_int* ldz, - lapack_complex_double* work, lapack_int* lwork, - double* rwork, lapack_int* lrwork, lapack_int* iwork, - lapack_int* liwork, lapack_int *info ); -void LAPACK_ssbevx( char* jobz, char* range, char* uplo, lapack_int* n, - lapack_int* kd, float* ab, lapack_int* ldab, float* q, - lapack_int* ldq, float* vl, float* vu, lapack_int* il, - lapack_int* iu, float* abstol, lapack_int* m, float* w, - float* z, lapack_int* ldz, float* work, lapack_int* iwork, - lapack_int* ifail, lapack_int *info ); -void LAPACK_dsbevx( char* jobz, char* range, char* uplo, lapack_int* n, - lapack_int* kd, double* ab, lapack_int* ldab, double* q, - lapack_int* ldq, double* vl, double* vu, lapack_int* il, - lapack_int* iu, double* abstol, lapack_int* m, double* w, - double* z, lapack_int* ldz, double* work, lapack_int* iwork, - lapack_int* ifail, lapack_int *info ); -void LAPACK_chbevx( char* jobz, char* range, char* uplo, lapack_int* n, - lapack_int* kd, lapack_complex_float* ab, lapack_int* ldab, - lapack_complex_float* q, lapack_int* ldq, float* vl, - float* vu, lapack_int* il, lapack_int* iu, float* abstol, - lapack_int* m, float* w, lapack_complex_float* z, - lapack_int* ldz, lapack_complex_float* work, float* rwork, - lapack_int* iwork, lapack_int* ifail, lapack_int *info ); -void LAPACK_zhbevx( char* jobz, char* range, char* uplo, lapack_int* n, - lapack_int* kd, lapack_complex_double* ab, lapack_int* ldab, - lapack_complex_double* q, lapack_int* ldq, double* vl, - double* vu, lapack_int* il, lapack_int* iu, double* abstol, - lapack_int* m, double* w, lapack_complex_double* z, - lapack_int* ldz, lapack_complex_double* work, double* rwork, - lapack_int* iwork, lapack_int* ifail, lapack_int *info ); -void LAPACK_sstev( char* jobz, lapack_int* n, float* d, float* e, float* z, - lapack_int* ldz, float* work, lapack_int *info ); -void LAPACK_dstev( char* jobz, lapack_int* n, double* d, double* e, double* z, - lapack_int* ldz, double* work, lapack_int *info ); -void LAPACK_sstevd( char* jobz, lapack_int* n, float* d, float* e, float* z, - lapack_int* ldz, float* work, lapack_int* lwork, - lapack_int* iwork, lapack_int* liwork, lapack_int *info ); -void LAPACK_dstevd( char* jobz, lapack_int* n, double* d, double* e, double* z, - lapack_int* ldz, double* work, lapack_int* lwork, - lapack_int* iwork, lapack_int* liwork, lapack_int *info ); -void LAPACK_sstevx( char* jobz, char* range, lapack_int* n, float* d, float* e, - float* vl, float* vu, lapack_int* il, lapack_int* iu, - float* abstol, lapack_int* m, float* w, float* z, - lapack_int* ldz, float* work, lapack_int* iwork, - lapack_int* ifail, lapack_int *info ); -void LAPACK_dstevx( char* jobz, char* range, lapack_int* n, double* d, - double* e, double* vl, double* vu, lapack_int* il, - lapack_int* iu, double* abstol, lapack_int* m, double* w, - double* z, lapack_int* ldz, double* work, lapack_int* iwork, - lapack_int* ifail, lapack_int *info ); -void LAPACK_sstevr( char* jobz, char* range, lapack_int* n, float* d, float* e, - float* vl, float* vu, lapack_int* il, lapack_int* iu, - float* abstol, lapack_int* m, float* w, float* z, - lapack_int* ldz, lapack_int* isuppz, float* work, - lapack_int* lwork, lapack_int* iwork, lapack_int* liwork, - lapack_int *info ); -void LAPACK_dstevr( char* jobz, char* range, lapack_int* n, double* d, - double* e, double* vl, double* vu, lapack_int* il, - lapack_int* iu, double* abstol, lapack_int* m, double* w, - double* z, lapack_int* ldz, lapack_int* isuppz, - double* work, lapack_int* lwork, lapack_int* iwork, - lapack_int* liwork, lapack_int *info ); -void LAPACK_sgees( char* jobvs, char* sort, LAPACK_S_SELECT2 select, - lapack_int* n, float* a, lapack_int* lda, lapack_int* sdim, - float* wr, float* wi, float* vs, lapack_int* ldvs, - float* work, lapack_int* lwork, lapack_logical* bwork, - lapack_int *info ); -void LAPACK_dgees( char* jobvs, char* sort, LAPACK_D_SELECT2 select, - lapack_int* n, double* a, lapack_int* lda, lapack_int* sdim, - double* wr, double* wi, double* vs, lapack_int* ldvs, - double* work, lapack_int* lwork, lapack_logical* bwork, - lapack_int *info ); -void LAPACK_cgees( char* jobvs, char* sort, LAPACK_C_SELECT1 select, - lapack_int* n, lapack_complex_float* a, lapack_int* lda, - lapack_int* sdim, lapack_complex_float* w, - lapack_complex_float* vs, lapack_int* ldvs, - lapack_complex_float* work, lapack_int* lwork, float* rwork, - lapack_logical* bwork, lapack_int *info ); -void LAPACK_zgees( char* jobvs, char* sort, LAPACK_Z_SELECT1 select, - lapack_int* n, lapack_complex_double* a, lapack_int* lda, - lapack_int* sdim, lapack_complex_double* w, - lapack_complex_double* vs, lapack_int* ldvs, - lapack_complex_double* work, lapack_int* lwork, - double* rwork, lapack_logical* bwork, lapack_int *info ); -void LAPACK_sgeesx( char* jobvs, char* sort, LAPACK_S_SELECT2 select, - char* sense, lapack_int* n, float* a, lapack_int* lda, - lapack_int* sdim, float* wr, float* wi, float* vs, - lapack_int* ldvs, float* rconde, float* rcondv, float* work, - lapack_int* lwork, lapack_int* iwork, lapack_int* liwork, - lapack_logical* bwork, lapack_int *info ); -void LAPACK_dgeesx( char* jobvs, char* sort, LAPACK_D_SELECT2 select, - char* sense, lapack_int* n, double* a, lapack_int* lda, - lapack_int* sdim, double* wr, double* wi, double* vs, - lapack_int* ldvs, double* rconde, double* rcondv, - double* work, lapack_int* lwork, lapack_int* iwork, - lapack_int* liwork, lapack_logical* bwork, - lapack_int *info ); -void LAPACK_cgeesx( char* jobvs, char* sort, LAPACK_C_SELECT1 select, - char* sense, lapack_int* n, lapack_complex_float* a, - lapack_int* lda, lapack_int* sdim, lapack_complex_float* w, - lapack_complex_float* vs, lapack_int* ldvs, float* rconde, - float* rcondv, lapack_complex_float* work, - lapack_int* lwork, float* rwork, lapack_logical* bwork, - lapack_int *info ); -void LAPACK_zgeesx( char* jobvs, char* sort, LAPACK_Z_SELECT1 select, - char* sense, lapack_int* n, lapack_complex_double* a, - lapack_int* lda, lapack_int* sdim, lapack_complex_double* w, - lapack_complex_double* vs, lapack_int* ldvs, double* rconde, - double* rcondv, lapack_complex_double* work, - lapack_int* lwork, double* rwork, lapack_logical* bwork, - lapack_int *info ); -void LAPACK_sgeev( char* jobvl, char* jobvr, lapack_int* n, float* a, - lapack_int* lda, float* wr, float* wi, float* vl, - lapack_int* ldvl, float* vr, lapack_int* ldvr, float* work, - lapack_int* lwork, lapack_int *info ); -void LAPACK_dgeev( char* jobvl, char* jobvr, lapack_int* n, double* a, - lapack_int* lda, double* wr, double* wi, double* vl, - lapack_int* ldvl, double* vr, lapack_int* ldvr, double* work, - lapack_int* lwork, lapack_int *info ); -void LAPACK_cgeev( char* jobvl, char* jobvr, lapack_int* n, - lapack_complex_float* a, lapack_int* lda, - lapack_complex_float* w, lapack_complex_float* vl, - lapack_int* ldvl, lapack_complex_float* vr, lapack_int* ldvr, - lapack_complex_float* work, lapack_int* lwork, float* rwork, - lapack_int *info ); -void LAPACK_zgeev( char* jobvl, char* jobvr, lapack_int* n, - lapack_complex_double* a, lapack_int* lda, - lapack_complex_double* w, lapack_complex_double* vl, - lapack_int* ldvl, lapack_complex_double* vr, - lapack_int* ldvr, lapack_complex_double* work, - lapack_int* lwork, double* rwork, lapack_int *info ); -void LAPACK_sgeevx( char* balanc, char* jobvl, char* jobvr, char* sense, - lapack_int* n, float* a, lapack_int* lda, float* wr, - float* wi, float* vl, lapack_int* ldvl, float* vr, - lapack_int* ldvr, lapack_int* ilo, lapack_int* ihi, - float* scale, float* abnrm, float* rconde, float* rcondv, - float* work, lapack_int* lwork, lapack_int* iwork, - lapack_int *info ); -void LAPACK_dgeevx( char* balanc, char* jobvl, char* jobvr, char* sense, - lapack_int* n, double* a, lapack_int* lda, double* wr, - double* wi, double* vl, lapack_int* ldvl, double* vr, - lapack_int* ldvr, lapack_int* ilo, lapack_int* ihi, - double* scale, double* abnrm, double* rconde, - double* rcondv, double* work, lapack_int* lwork, - lapack_int* iwork, lapack_int *info ); -void LAPACK_cgeevx( char* balanc, char* jobvl, char* jobvr, char* sense, - lapack_int* n, lapack_complex_float* a, lapack_int* lda, - lapack_complex_float* w, lapack_complex_float* vl, - lapack_int* ldvl, lapack_complex_float* vr, - lapack_int* ldvr, lapack_int* ilo, lapack_int* ihi, - float* scale, float* abnrm, float* rconde, float* rcondv, - lapack_complex_float* work, lapack_int* lwork, float* rwork, - lapack_int *info ); -void LAPACK_zgeevx( char* balanc, char* jobvl, char* jobvr, char* sense, - lapack_int* n, lapack_complex_double* a, lapack_int* lda, - lapack_complex_double* w, lapack_complex_double* vl, - lapack_int* ldvl, lapack_complex_double* vr, - lapack_int* ldvr, lapack_int* ilo, lapack_int* ihi, - double* scale, double* abnrm, double* rconde, - double* rcondv, lapack_complex_double* work, - lapack_int* lwork, double* rwork, lapack_int *info ); -void LAPACK_sgesvd( char* jobu, char* jobvt, lapack_int* m, lapack_int* n, - float* a, lapack_int* lda, float* s, float* u, - lapack_int* ldu, float* vt, lapack_int* ldvt, float* work, - lapack_int* lwork, lapack_int *info ); -void LAPACK_dgesvd( char* jobu, char* jobvt, lapack_int* m, lapack_int* n, - double* a, lapack_int* lda, double* s, double* u, - lapack_int* ldu, double* vt, lapack_int* ldvt, double* work, - lapack_int* lwork, lapack_int *info ); -void LAPACK_cgesvd( char* jobu, char* jobvt, lapack_int* m, lapack_int* n, - lapack_complex_float* a, lapack_int* lda, float* s, - lapack_complex_float* u, lapack_int* ldu, - lapack_complex_float* vt, lapack_int* ldvt, - lapack_complex_float* work, lapack_int* lwork, float* rwork, - lapack_int *info ); -void LAPACK_zgesvd( char* jobu, char* jobvt, lapack_int* m, lapack_int* n, - lapack_complex_double* a, lapack_int* lda, double* s, - lapack_complex_double* u, lapack_int* ldu, - lapack_complex_double* vt, lapack_int* ldvt, - lapack_complex_double* work, lapack_int* lwork, - double* rwork, lapack_int *info ); -void LAPACK_sgesdd( char* jobz, lapack_int* m, lapack_int* n, float* a, - lapack_int* lda, float* s, float* u, lapack_int* ldu, - float* vt, lapack_int* ldvt, float* work, lapack_int* lwork, - lapack_int* iwork, lapack_int *info ); -void LAPACK_dgesdd( char* jobz, lapack_int* m, lapack_int* n, double* a, - lapack_int* lda, double* s, double* u, lapack_int* ldu, - double* vt, lapack_int* ldvt, double* work, - lapack_int* lwork, lapack_int* iwork, lapack_int *info ); -void LAPACK_cgesdd( char* jobz, lapack_int* m, lapack_int* n, - lapack_complex_float* a, lapack_int* lda, float* s, - lapack_complex_float* u, lapack_int* ldu, - lapack_complex_float* vt, lapack_int* ldvt, - lapack_complex_float* work, lapack_int* lwork, float* rwork, - lapack_int* iwork, lapack_int *info ); -void LAPACK_zgesdd( char* jobz, lapack_int* m, lapack_int* n, - lapack_complex_double* a, lapack_int* lda, double* s, - lapack_complex_double* u, lapack_int* ldu, - lapack_complex_double* vt, lapack_int* ldvt, - lapack_complex_double* work, lapack_int* lwork, - double* rwork, lapack_int* iwork, lapack_int *info ); -void LAPACK_dgejsv( char* joba, char* jobu, char* jobv, char* jobr, char* jobt, - char* jobp, lapack_int* m, lapack_int* n, double* a, - lapack_int* lda, double* sva, double* u, lapack_int* ldu, - double* v, lapack_int* ldv, double* work, lapack_int* lwork, - lapack_int* iwork, lapack_int *info ); -void LAPACK_sgejsv( char* joba, char* jobu, char* jobv, char* jobr, char* jobt, - char* jobp, lapack_int* m, lapack_int* n, float* a, - lapack_int* lda, float* sva, float* u, lapack_int* ldu, - float* v, lapack_int* ldv, float* work, lapack_int* lwork, - lapack_int* iwork, lapack_int *info ); -void LAPACK_dgesvj( char* joba, char* jobu, char* jobv, lapack_int* m, - lapack_int* n, double* a, lapack_int* lda, double* sva, - lapack_int* mv, double* v, lapack_int* ldv, double* work, - lapack_int* lwork, lapack_int *info ); -void LAPACK_sgesvj( char* joba, char* jobu, char* jobv, lapack_int* m, - lapack_int* n, float* a, lapack_int* lda, float* sva, - lapack_int* mv, float* v, lapack_int* ldv, float* work, - lapack_int* lwork, lapack_int *info ); -void LAPACK_sggsvd( char* jobu, char* jobv, char* jobq, lapack_int* m, - lapack_int* n, lapack_int* p, lapack_int* k, lapack_int* l, - float* a, lapack_int* lda, float* b, lapack_int* ldb, - float* alpha, float* beta, float* u, lapack_int* ldu, - float* v, lapack_int* ldv, float* q, lapack_int* ldq, - float* work, lapack_int* iwork, lapack_int *info ); -void LAPACK_dggsvd( char* jobu, char* jobv, char* jobq, lapack_int* m, - lapack_int* n, lapack_int* p, lapack_int* k, lapack_int* l, - double* a, lapack_int* lda, double* b, lapack_int* ldb, - double* alpha, double* beta, double* u, lapack_int* ldu, - double* v, lapack_int* ldv, double* q, lapack_int* ldq, - double* work, lapack_int* iwork, lapack_int *info ); -void LAPACK_cggsvd( char* jobu, char* jobv, char* jobq, lapack_int* m, - lapack_int* n, lapack_int* p, lapack_int* k, lapack_int* l, - lapack_complex_float* a, lapack_int* lda, - lapack_complex_float* b, lapack_int* ldb, float* alpha, - float* beta, lapack_complex_float* u, lapack_int* ldu, - lapack_complex_float* v, lapack_int* ldv, - lapack_complex_float* q, lapack_int* ldq, - lapack_complex_float* work, float* rwork, lapack_int* iwork, - lapack_int *info ); -void LAPACK_zggsvd( char* jobu, char* jobv, char* jobq, lapack_int* m, - lapack_int* n, lapack_int* p, lapack_int* k, lapack_int* l, - lapack_complex_double* a, lapack_int* lda, - lapack_complex_double* b, lapack_int* ldb, double* alpha, - double* beta, lapack_complex_double* u, lapack_int* ldu, - lapack_complex_double* v, lapack_int* ldv, - lapack_complex_double* q, lapack_int* ldq, - lapack_complex_double* work, double* rwork, - lapack_int* iwork, lapack_int *info ); -void LAPACK_ssygv( lapack_int* itype, char* jobz, char* uplo, lapack_int* n, - float* a, lapack_int* lda, float* b, lapack_int* ldb, - float* w, float* work, lapack_int* lwork, lapack_int *info ); -void LAPACK_dsygv( lapack_int* itype, char* jobz, char* uplo, lapack_int* n, - double* a, lapack_int* lda, double* b, lapack_int* ldb, - double* w, double* work, lapack_int* lwork, - lapack_int *info ); -void LAPACK_chegv( lapack_int* itype, char* jobz, char* uplo, lapack_int* n, - lapack_complex_float* a, lapack_int* lda, - lapack_complex_float* b, lapack_int* ldb, float* w, - lapack_complex_float* work, lapack_int* lwork, float* rwork, - lapack_int *info ); -void LAPACK_zhegv( lapack_int* itype, char* jobz, char* uplo, lapack_int* n, - lapack_complex_double* a, lapack_int* lda, - lapack_complex_double* b, lapack_int* ldb, double* w, - lapack_complex_double* work, lapack_int* lwork, - double* rwork, lapack_int *info ); -void LAPACK_ssygvd( lapack_int* itype, char* jobz, char* uplo, lapack_int* n, - float* a, lapack_int* lda, float* b, lapack_int* ldb, - float* w, float* work, lapack_int* lwork, lapack_int* iwork, - lapack_int* liwork, lapack_int *info ); -void LAPACK_dsygvd( lapack_int* itype, char* jobz, char* uplo, lapack_int* n, - double* a, lapack_int* lda, double* b, lapack_int* ldb, - double* w, double* work, lapack_int* lwork, - lapack_int* iwork, lapack_int* liwork, lapack_int *info ); -void LAPACK_chegvd( lapack_int* itype, char* jobz, char* uplo, lapack_int* n, - lapack_complex_float* a, lapack_int* lda, - lapack_complex_float* b, lapack_int* ldb, float* w, - lapack_complex_float* work, lapack_int* lwork, float* rwork, - lapack_int* lrwork, lapack_int* iwork, lapack_int* liwork, - lapack_int *info ); -void LAPACK_zhegvd( lapack_int* itype, char* jobz, char* uplo, lapack_int* n, - lapack_complex_double* a, lapack_int* lda, - lapack_complex_double* b, lapack_int* ldb, double* w, - lapack_complex_double* work, lapack_int* lwork, - double* rwork, lapack_int* lrwork, lapack_int* iwork, - lapack_int* liwork, lapack_int *info ); -void LAPACK_ssygvx( lapack_int* itype, char* jobz, char* range, char* uplo, - lapack_int* n, float* a, lapack_int* lda, float* b, - lapack_int* ldb, float* vl, float* vu, lapack_int* il, - lapack_int* iu, float* abstol, lapack_int* m, float* w, - float* z, lapack_int* ldz, float* work, lapack_int* lwork, - lapack_int* iwork, lapack_int* ifail, lapack_int *info ); -void LAPACK_dsygvx( lapack_int* itype, char* jobz, char* range, char* uplo, - lapack_int* n, double* a, lapack_int* lda, double* b, - lapack_int* ldb, double* vl, double* vu, lapack_int* il, - lapack_int* iu, double* abstol, lapack_int* m, double* w, - double* z, lapack_int* ldz, double* work, lapack_int* lwork, - lapack_int* iwork, lapack_int* ifail, lapack_int *info ); -void LAPACK_chegvx( lapack_int* itype, char* jobz, char* range, char* uplo, - lapack_int* n, lapack_complex_float* a, lapack_int* lda, - lapack_complex_float* b, lapack_int* ldb, float* vl, - float* vu, lapack_int* il, lapack_int* iu, float* abstol, - lapack_int* m, float* w, lapack_complex_float* z, - lapack_int* ldz, lapack_complex_float* work, - lapack_int* lwork, float* rwork, lapack_int* iwork, - lapack_int* ifail, lapack_int *info ); -void LAPACK_zhegvx( lapack_int* itype, char* jobz, char* range, char* uplo, - lapack_int* n, lapack_complex_double* a, lapack_int* lda, - lapack_complex_double* b, lapack_int* ldb, double* vl, - double* vu, lapack_int* il, lapack_int* iu, double* abstol, - lapack_int* m, double* w, lapack_complex_double* z, - lapack_int* ldz, lapack_complex_double* work, - lapack_int* lwork, double* rwork, lapack_int* iwork, - lapack_int* ifail, lapack_int *info ); -void LAPACK_sspgv( lapack_int* itype, char* jobz, char* uplo, lapack_int* n, - float* ap, float* bp, float* w, float* z, lapack_int* ldz, - float* work, lapack_int *info ); -void LAPACK_dspgv( lapack_int* itype, char* jobz, char* uplo, lapack_int* n, - double* ap, double* bp, double* w, double* z, - lapack_int* ldz, double* work, lapack_int *info ); -void LAPACK_chpgv( lapack_int* itype, char* jobz, char* uplo, lapack_int* n, - lapack_complex_float* ap, lapack_complex_float* bp, float* w, - lapack_complex_float* z, lapack_int* ldz, - lapack_complex_float* work, float* rwork, lapack_int *info ); -void LAPACK_zhpgv( lapack_int* itype, char* jobz, char* uplo, lapack_int* n, - lapack_complex_double* ap, lapack_complex_double* bp, - double* w, lapack_complex_double* z, lapack_int* ldz, - lapack_complex_double* work, double* rwork, - lapack_int *info ); -void LAPACK_sspgvd( lapack_int* itype, char* jobz, char* uplo, lapack_int* n, - float* ap, float* bp, float* w, float* z, lapack_int* ldz, - float* work, lapack_int* lwork, lapack_int* iwork, - lapack_int* liwork, lapack_int *info ); -void LAPACK_dspgvd( lapack_int* itype, char* jobz, char* uplo, lapack_int* n, - double* ap, double* bp, double* w, double* z, - lapack_int* ldz, double* work, lapack_int* lwork, - lapack_int* iwork, lapack_int* liwork, lapack_int *info ); -void LAPACK_chpgvd( lapack_int* itype, char* jobz, char* uplo, lapack_int* n, - lapack_complex_float* ap, lapack_complex_float* bp, - float* w, lapack_complex_float* z, lapack_int* ldz, - lapack_complex_float* work, lapack_int* lwork, float* rwork, - lapack_int* lrwork, lapack_int* iwork, lapack_int* liwork, - lapack_int *info ); -void LAPACK_zhpgvd( lapack_int* itype, char* jobz, char* uplo, lapack_int* n, - lapack_complex_double* ap, lapack_complex_double* bp, - double* w, lapack_complex_double* z, lapack_int* ldz, - lapack_complex_double* work, lapack_int* lwork, - double* rwork, lapack_int* lrwork, lapack_int* iwork, - lapack_int* liwork, lapack_int *info ); -void LAPACK_sspgvx( lapack_int* itype, char* jobz, char* range, char* uplo, - lapack_int* n, float* ap, float* bp, float* vl, float* vu, - lapack_int* il, lapack_int* iu, float* abstol, - lapack_int* m, float* w, float* z, lapack_int* ldz, - float* work, lapack_int* iwork, lapack_int* ifail, - lapack_int *info ); -void LAPACK_dspgvx( lapack_int* itype, char* jobz, char* range, char* uplo, - lapack_int* n, double* ap, double* bp, double* vl, - double* vu, lapack_int* il, lapack_int* iu, double* abstol, - lapack_int* m, double* w, double* z, lapack_int* ldz, - double* work, lapack_int* iwork, lapack_int* ifail, - lapack_int *info ); -void LAPACK_chpgvx( lapack_int* itype, char* jobz, char* range, char* uplo, - lapack_int* n, lapack_complex_float* ap, - lapack_complex_float* bp, float* vl, float* vu, - lapack_int* il, lapack_int* iu, float* abstol, - lapack_int* m, float* w, lapack_complex_float* z, - lapack_int* ldz, lapack_complex_float* work, float* rwork, - lapack_int* iwork, lapack_int* ifail, lapack_int *info ); -void LAPACK_zhpgvx( lapack_int* itype, char* jobz, char* range, char* uplo, - lapack_int* n, lapack_complex_double* ap, - lapack_complex_double* bp, double* vl, double* vu, - lapack_int* il, lapack_int* iu, double* abstol, - lapack_int* m, double* w, lapack_complex_double* z, - lapack_int* ldz, lapack_complex_double* work, double* rwork, - lapack_int* iwork, lapack_int* ifail, lapack_int *info ); -void LAPACK_ssbgv( char* jobz, char* uplo, lapack_int* n, lapack_int* ka, - lapack_int* kb, float* ab, lapack_int* ldab, float* bb, - lapack_int* ldbb, float* w, float* z, lapack_int* ldz, - float* work, lapack_int *info ); -void LAPACK_dsbgv( char* jobz, char* uplo, lapack_int* n, lapack_int* ka, - lapack_int* kb, double* ab, lapack_int* ldab, double* bb, - lapack_int* ldbb, double* w, double* z, lapack_int* ldz, - double* work, lapack_int *info ); -void LAPACK_chbgv( char* jobz, char* uplo, lapack_int* n, lapack_int* ka, - lapack_int* kb, lapack_complex_float* ab, lapack_int* ldab, - lapack_complex_float* bb, lapack_int* ldbb, float* w, - lapack_complex_float* z, lapack_int* ldz, - lapack_complex_float* work, float* rwork, lapack_int *info ); -void LAPACK_zhbgv( char* jobz, char* uplo, lapack_int* n, lapack_int* ka, - lapack_int* kb, lapack_complex_double* ab, lapack_int* ldab, - lapack_complex_double* bb, lapack_int* ldbb, double* w, - lapack_complex_double* z, lapack_int* ldz, - lapack_complex_double* work, double* rwork, - lapack_int *info ); -void LAPACK_ssbgvd( char* jobz, char* uplo, lapack_int* n, lapack_int* ka, - lapack_int* kb, float* ab, lapack_int* ldab, float* bb, - lapack_int* ldbb, float* w, float* z, lapack_int* ldz, - float* work, lapack_int* lwork, lapack_int* iwork, - lapack_int* liwork, lapack_int *info ); -void LAPACK_dsbgvd( char* jobz, char* uplo, lapack_int* n, lapack_int* ka, - lapack_int* kb, double* ab, lapack_int* ldab, double* bb, - lapack_int* ldbb, double* w, double* z, lapack_int* ldz, - double* work, lapack_int* lwork, lapack_int* iwork, - lapack_int* liwork, lapack_int *info ); -void LAPACK_chbgvd( char* jobz, char* uplo, lapack_int* n, lapack_int* ka, - lapack_int* kb, lapack_complex_float* ab, lapack_int* ldab, - lapack_complex_float* bb, lapack_int* ldbb, float* w, - lapack_complex_float* z, lapack_int* ldz, - lapack_complex_float* work, lapack_int* lwork, float* rwork, - lapack_int* lrwork, lapack_int* iwork, lapack_int* liwork, - lapack_int *info ); -void LAPACK_zhbgvd( char* jobz, char* uplo, lapack_int* n, lapack_int* ka, - lapack_int* kb, lapack_complex_double* ab, lapack_int* ldab, - lapack_complex_double* bb, lapack_int* ldbb, double* w, - lapack_complex_double* z, lapack_int* ldz, - lapack_complex_double* work, lapack_int* lwork, - double* rwork, lapack_int* lrwork, lapack_int* iwork, - lapack_int* liwork, lapack_int *info ); -void LAPACK_ssbgvx( char* jobz, char* range, char* uplo, lapack_int* n, - lapack_int* ka, lapack_int* kb, float* ab, lapack_int* ldab, - float* bb, lapack_int* ldbb, float* q, lapack_int* ldq, - float* vl, float* vu, lapack_int* il, lapack_int* iu, - float* abstol, lapack_int* m, float* w, float* z, - lapack_int* ldz, float* work, lapack_int* iwork, - lapack_int* ifail, lapack_int *info ); -void LAPACK_dsbgvx( char* jobz, char* range, char* uplo, lapack_int* n, - lapack_int* ka, lapack_int* kb, double* ab, - lapack_int* ldab, double* bb, lapack_int* ldbb, double* q, - lapack_int* ldq, double* vl, double* vu, lapack_int* il, - lapack_int* iu, double* abstol, lapack_int* m, double* w, - double* z, lapack_int* ldz, double* work, lapack_int* iwork, - lapack_int* ifail, lapack_int *info ); -void LAPACK_chbgvx( char* jobz, char* range, char* uplo, lapack_int* n, - lapack_int* ka, lapack_int* kb, lapack_complex_float* ab, - lapack_int* ldab, lapack_complex_float* bb, - lapack_int* ldbb, lapack_complex_float* q, lapack_int* ldq, - float* vl, float* vu, lapack_int* il, lapack_int* iu, - float* abstol, lapack_int* m, float* w, - lapack_complex_float* z, lapack_int* ldz, - lapack_complex_float* work, float* rwork, lapack_int* iwork, - lapack_int* ifail, lapack_int *info ); -void LAPACK_zhbgvx( char* jobz, char* range, char* uplo, lapack_int* n, - lapack_int* ka, lapack_int* kb, lapack_complex_double* ab, - lapack_int* ldab, lapack_complex_double* bb, - lapack_int* ldbb, lapack_complex_double* q, lapack_int* ldq, - double* vl, double* vu, lapack_int* il, lapack_int* iu, - double* abstol, lapack_int* m, double* w, - lapack_complex_double* z, lapack_int* ldz, - lapack_complex_double* work, double* rwork, - lapack_int* iwork, lapack_int* ifail, lapack_int *info ); -void LAPACK_sgges( char* jobvsl, char* jobvsr, char* sort, - LAPACK_S_SELECT3 selctg, lapack_int* n, float* a, - lapack_int* lda, float* b, lapack_int* ldb, lapack_int* sdim, - float* alphar, float* alphai, float* beta, float* vsl, - lapack_int* ldvsl, float* vsr, lapack_int* ldvsr, - float* work, lapack_int* lwork, lapack_logical* bwork, - lapack_int *info ); -void LAPACK_dgges( char* jobvsl, char* jobvsr, char* sort, - LAPACK_D_SELECT3 selctg, lapack_int* n, double* a, - lapack_int* lda, double* b, lapack_int* ldb, - lapack_int* sdim, double* alphar, double* alphai, - double* beta, double* vsl, lapack_int* ldvsl, double* vsr, - lapack_int* ldvsr, double* work, lapack_int* lwork, - lapack_logical* bwork, lapack_int *info ); -void LAPACK_cgges( char* jobvsl, char* jobvsr, char* sort, - LAPACK_C_SELECT2 selctg, lapack_int* n, - lapack_complex_float* a, lapack_int* lda, - lapack_complex_float* b, lapack_int* ldb, lapack_int* sdim, - lapack_complex_float* alpha, lapack_complex_float* beta, - lapack_complex_float* vsl, lapack_int* ldvsl, - lapack_complex_float* vsr, lapack_int* ldvsr, - lapack_complex_float* work, lapack_int* lwork, float* rwork, - lapack_logical* bwork, lapack_int *info ); -void LAPACK_zgges( char* jobvsl, char* jobvsr, char* sort, - LAPACK_Z_SELECT2 selctg, lapack_int* n, - lapack_complex_double* a, lapack_int* lda, - lapack_complex_double* b, lapack_int* ldb, lapack_int* sdim, - lapack_complex_double* alpha, lapack_complex_double* beta, - lapack_complex_double* vsl, lapack_int* ldvsl, - lapack_complex_double* vsr, lapack_int* ldvsr, - lapack_complex_double* work, lapack_int* lwork, - double* rwork, lapack_logical* bwork, lapack_int *info ); -void LAPACK_sggesx( char* jobvsl, char* jobvsr, char* sort, - LAPACK_S_SELECT3 selctg, char* sense, lapack_int* n, - float* a, lapack_int* lda, float* b, lapack_int* ldb, - lapack_int* sdim, float* alphar, float* alphai, float* beta, - float* vsl, lapack_int* ldvsl, float* vsr, - lapack_int* ldvsr, float* rconde, float* rcondv, - float* work, lapack_int* lwork, lapack_int* iwork, - lapack_int* liwork, lapack_logical* bwork, - lapack_int *info ); -void LAPACK_dggesx( char* jobvsl, char* jobvsr, char* sort, - LAPACK_D_SELECT3 selctg, char* sense, lapack_int* n, - double* a, lapack_int* lda, double* b, lapack_int* ldb, - lapack_int* sdim, double* alphar, double* alphai, - double* beta, double* vsl, lapack_int* ldvsl, double* vsr, - lapack_int* ldvsr, double* rconde, double* rcondv, - double* work, lapack_int* lwork, lapack_int* iwork, - lapack_int* liwork, lapack_logical* bwork, - lapack_int *info ); -void LAPACK_cggesx( char* jobvsl, char* jobvsr, char* sort, - LAPACK_C_SELECT2 selctg, char* sense, lapack_int* n, - lapack_complex_float* a, lapack_int* lda, - lapack_complex_float* b, lapack_int* ldb, lapack_int* sdim, - lapack_complex_float* alpha, lapack_complex_float* beta, - lapack_complex_float* vsl, lapack_int* ldvsl, - lapack_complex_float* vsr, lapack_int* ldvsr, float* rconde, - float* rcondv, lapack_complex_float* work, - lapack_int* lwork, float* rwork, lapack_int* iwork, - lapack_int* liwork, lapack_logical* bwork, - lapack_int *info ); -void LAPACK_zggesx( char* jobvsl, char* jobvsr, char* sort, - LAPACK_Z_SELECT2 selctg, char* sense, lapack_int* n, - lapack_complex_double* a, lapack_int* lda, - lapack_complex_double* b, lapack_int* ldb, lapack_int* sdim, - lapack_complex_double* alpha, lapack_complex_double* beta, - lapack_complex_double* vsl, lapack_int* ldvsl, - lapack_complex_double* vsr, lapack_int* ldvsr, - double* rconde, double* rcondv, lapack_complex_double* work, - lapack_int* lwork, double* rwork, lapack_int* iwork, - lapack_int* liwork, lapack_logical* bwork, - lapack_int *info ); -void LAPACK_sggev( char* jobvl, char* jobvr, lapack_int* n, float* a, - lapack_int* lda, float* b, lapack_int* ldb, float* alphar, - float* alphai, float* beta, float* vl, lapack_int* ldvl, - float* vr, lapack_int* ldvr, float* work, lapack_int* lwork, - lapack_int *info ); -void LAPACK_dggev( char* jobvl, char* jobvr, lapack_int* n, double* a, - lapack_int* lda, double* b, lapack_int* ldb, double* alphar, - double* alphai, double* beta, double* vl, lapack_int* ldvl, - double* vr, lapack_int* ldvr, double* work, - lapack_int* lwork, lapack_int *info ); -void LAPACK_cggev( char* jobvl, char* jobvr, lapack_int* n, - lapack_complex_float* a, lapack_int* lda, - lapack_complex_float* b, lapack_int* ldb, - lapack_complex_float* alpha, lapack_complex_float* beta, - lapack_complex_float* vl, lapack_int* ldvl, - lapack_complex_float* vr, lapack_int* ldvr, - lapack_complex_float* work, lapack_int* lwork, float* rwork, - lapack_int *info ); -void LAPACK_zggev( char* jobvl, char* jobvr, lapack_int* n, - lapack_complex_double* a, lapack_int* lda, - lapack_complex_double* b, lapack_int* ldb, - lapack_complex_double* alpha, lapack_complex_double* beta, - lapack_complex_double* vl, lapack_int* ldvl, - lapack_complex_double* vr, lapack_int* ldvr, - lapack_complex_double* work, lapack_int* lwork, - double* rwork, lapack_int *info ); -void LAPACK_sggevx( char* balanc, char* jobvl, char* jobvr, char* sense, - lapack_int* n, float* a, lapack_int* lda, float* b, - lapack_int* ldb, float* alphar, float* alphai, float* beta, - float* vl, lapack_int* ldvl, float* vr, lapack_int* ldvr, - lapack_int* ilo, lapack_int* ihi, float* lscale, - float* rscale, float* abnrm, float* bbnrm, float* rconde, - float* rcondv, float* work, lapack_int* lwork, - lapack_int* iwork, lapack_logical* bwork, - lapack_int *info ); -void LAPACK_dggevx( char* balanc, char* jobvl, char* jobvr, char* sense, - lapack_int* n, double* a, lapack_int* lda, double* b, - lapack_int* ldb, double* alphar, double* alphai, - double* beta, double* vl, lapack_int* ldvl, double* vr, - lapack_int* ldvr, lapack_int* ilo, lapack_int* ihi, - double* lscale, double* rscale, double* abnrm, - double* bbnrm, double* rconde, double* rcondv, double* work, - lapack_int* lwork, lapack_int* iwork, lapack_logical* bwork, - lapack_int *info ); -void LAPACK_cggevx( char* balanc, char* jobvl, char* jobvr, char* sense, - lapack_int* n, lapack_complex_float* a, lapack_int* lda, - lapack_complex_float* b, lapack_int* ldb, - lapack_complex_float* alpha, lapack_complex_float* beta, - lapack_complex_float* vl, lapack_int* ldvl, - lapack_complex_float* vr, lapack_int* ldvr, lapack_int* ilo, - lapack_int* ihi, float* lscale, float* rscale, float* abnrm, - float* bbnrm, float* rconde, float* rcondv, - lapack_complex_float* work, lapack_int* lwork, float* rwork, - lapack_int* iwork, lapack_logical* bwork, - lapack_int *info ); -void LAPACK_zggevx( char* balanc, char* jobvl, char* jobvr, char* sense, - lapack_int* n, lapack_complex_double* a, lapack_int* lda, - lapack_complex_double* b, lapack_int* ldb, - lapack_complex_double* alpha, lapack_complex_double* beta, - lapack_complex_double* vl, lapack_int* ldvl, - lapack_complex_double* vr, lapack_int* ldvr, - lapack_int* ilo, lapack_int* ihi, double* lscale, - double* rscale, double* abnrm, double* bbnrm, - double* rconde, double* rcondv, lapack_complex_double* work, - lapack_int* lwork, double* rwork, lapack_int* iwork, - lapack_logical* bwork, lapack_int *info ); -void LAPACK_dsfrk( char* transr, char* uplo, char* trans, lapack_int* n, - lapack_int* k, double* alpha, const double* a, - lapack_int* lda, double* beta, double* c ); -void LAPACK_ssfrk( char* transr, char* uplo, char* trans, lapack_int* n, - lapack_int* k, float* alpha, const float* a, lapack_int* lda, - float* beta, float* c ); -void LAPACK_zhfrk( char* transr, char* uplo, char* trans, lapack_int* n, - lapack_int* k, double* alpha, const lapack_complex_double* a, - lapack_int* lda, double* beta, lapack_complex_double* c ); -void LAPACK_chfrk( char* transr, char* uplo, char* trans, lapack_int* n, - lapack_int* k, float* alpha, const lapack_complex_float* a, - lapack_int* lda, float* beta, lapack_complex_float* c ); -void LAPACK_dtfsm( char* transr, char* side, char* uplo, char* trans, - char* diag, lapack_int* m, lapack_int* n, double* alpha, - const double* a, double* b, lapack_int* ldb ); -void LAPACK_stfsm( char* transr, char* side, char* uplo, char* trans, - char* diag, lapack_int* m, lapack_int* n, float* alpha, - const float* a, float* b, lapack_int* ldb ); -void LAPACK_ztfsm( char* transr, char* side, char* uplo, char* trans, - char* diag, lapack_int* m, lapack_int* n, - lapack_complex_double* alpha, const lapack_complex_double* a, - lapack_complex_double* b, lapack_int* ldb ); -void LAPACK_ctfsm( char* transr, char* side, char* uplo, char* trans, - char* diag, lapack_int* m, lapack_int* n, - lapack_complex_float* alpha, const lapack_complex_float* a, - lapack_complex_float* b, lapack_int* ldb ); -void LAPACK_dtfttp( char* transr, char* uplo, lapack_int* n, const double* arf, - double* ap, lapack_int *info ); -void LAPACK_stfttp( char* transr, char* uplo, lapack_int* n, const float* arf, - float* ap, lapack_int *info ); -void LAPACK_ztfttp( char* transr, char* uplo, lapack_int* n, - const lapack_complex_double* arf, lapack_complex_double* ap, - lapack_int *info ); -void LAPACK_ctfttp( char* transr, char* uplo, lapack_int* n, - const lapack_complex_float* arf, lapack_complex_float* ap, - lapack_int *info ); -void LAPACK_dtfttr( char* transr, char* uplo, lapack_int* n, const double* arf, - double* a, lapack_int* lda, lapack_int *info ); -void LAPACK_stfttr( char* transr, char* uplo, lapack_int* n, const float* arf, - float* a, lapack_int* lda, lapack_int *info ); -void LAPACK_ztfttr( char* transr, char* uplo, lapack_int* n, - const lapack_complex_double* arf, lapack_complex_double* a, - lapack_int* lda, lapack_int *info ); -void LAPACK_ctfttr( char* transr, char* uplo, lapack_int* n, - const lapack_complex_float* arf, lapack_complex_float* a, - lapack_int* lda, lapack_int *info ); -void LAPACK_dtpttf( char* transr, char* uplo, lapack_int* n, const double* ap, - double* arf, lapack_int *info ); -void LAPACK_stpttf( char* transr, char* uplo, lapack_int* n, const float* ap, - float* arf, lapack_int *info ); -void LAPACK_ztpttf( char* transr, char* uplo, lapack_int* n, - const lapack_complex_double* ap, lapack_complex_double* arf, - lapack_int *info ); -void LAPACK_ctpttf( char* transr, char* uplo, lapack_int* n, - const lapack_complex_float* ap, lapack_complex_float* arf, - lapack_int *info ); -void LAPACK_dtpttr( char* uplo, lapack_int* n, const double* ap, double* a, - lapack_int* lda, lapack_int *info ); -void LAPACK_stpttr( char* uplo, lapack_int* n, const float* ap, float* a, - lapack_int* lda, lapack_int *info ); -void LAPACK_ztpttr( char* uplo, lapack_int* n, const lapack_complex_double* ap, - lapack_complex_double* a, lapack_int* lda, - lapack_int *info ); -void LAPACK_ctpttr( char* uplo, lapack_int* n, const lapack_complex_float* ap, - lapack_complex_float* a, lapack_int* lda, - lapack_int *info ); -void LAPACK_dtrttf( char* transr, char* uplo, lapack_int* n, const double* a, - lapack_int* lda, double* arf, lapack_int *info ); -void LAPACK_strttf( char* transr, char* uplo, lapack_int* n, const float* a, - lapack_int* lda, float* arf, lapack_int *info ); -void LAPACK_ztrttf( char* transr, char* uplo, lapack_int* n, - const lapack_complex_double* a, lapack_int* lda, - lapack_complex_double* arf, lapack_int *info ); -void LAPACK_ctrttf( char* transr, char* uplo, lapack_int* n, - const lapack_complex_float* a, lapack_int* lda, - lapack_complex_float* arf, lapack_int *info ); -void LAPACK_dtrttp( char* uplo, lapack_int* n, const double* a, lapack_int* lda, - double* ap, lapack_int *info ); -void LAPACK_strttp( char* uplo, lapack_int* n, const float* a, lapack_int* lda, - float* ap, lapack_int *info ); -void LAPACK_ztrttp( char* uplo, lapack_int* n, const lapack_complex_double* a, - lapack_int* lda, lapack_complex_double* ap, - lapack_int *info ); -void LAPACK_ctrttp( char* uplo, lapack_int* n, const lapack_complex_float* a, - lapack_int* lda, lapack_complex_float* ap, - lapack_int *info ); -void LAPACK_sgeqrfp( lapack_int* m, lapack_int* n, float* a, lapack_int* lda, - float* tau, float* work, lapack_int* lwork, - lapack_int *info ); -void LAPACK_dgeqrfp( lapack_int* m, lapack_int* n, double* a, lapack_int* lda, - double* tau, double* work, lapack_int* lwork, - lapack_int *info ); -void LAPACK_cgeqrfp( lapack_int* m, lapack_int* n, lapack_complex_float* a, - lapack_int* lda, lapack_complex_float* tau, - lapack_complex_float* work, lapack_int* lwork, - lapack_int *info ); -void LAPACK_zgeqrfp( lapack_int* m, lapack_int* n, lapack_complex_double* a, - lapack_int* lda, lapack_complex_double* tau, - lapack_complex_double* work, lapack_int* lwork, - lapack_int *info ); -void LAPACK_clacgv( lapack_int* n, lapack_complex_float* x, lapack_int* incx ); -void LAPACK_zlacgv( lapack_int* n, lapack_complex_double* x, lapack_int* incx ); -void LAPACK_slarnv( lapack_int* idist, lapack_int* iseed, lapack_int* n, - float* x ); -void LAPACK_dlarnv( lapack_int* idist, lapack_int* iseed, lapack_int* n, - double* x ); -void LAPACK_clarnv( lapack_int* idist, lapack_int* iseed, lapack_int* n, - lapack_complex_float* x ); -void LAPACK_zlarnv( lapack_int* idist, lapack_int* iseed, lapack_int* n, - lapack_complex_double* x ); -void LAPACK_sgeqr2( lapack_int* m, lapack_int* n, float* a, lapack_int* lda, - float* tau, float* work, lapack_int *info ); -void LAPACK_dgeqr2( lapack_int* m, lapack_int* n, double* a, lapack_int* lda, - double* tau, double* work, lapack_int *info ); -void LAPACK_cgeqr2( lapack_int* m, lapack_int* n, lapack_complex_float* a, - lapack_int* lda, lapack_complex_float* tau, - lapack_complex_float* work, lapack_int *info ); -void LAPACK_zgeqr2( lapack_int* m, lapack_int* n, lapack_complex_double* a, - lapack_int* lda, lapack_complex_double* tau, - lapack_complex_double* work, lapack_int *info ); -void LAPACK_slacpy( char* uplo, lapack_int* m, lapack_int* n, const float* a, - lapack_int* lda, float* b, lapack_int* ldb ); -void LAPACK_dlacpy( char* uplo, lapack_int* m, lapack_int* n, const double* a, - lapack_int* lda, double* b, lapack_int* ldb ); -void LAPACK_clacpy( char* uplo, lapack_int* m, lapack_int* n, - const lapack_complex_float* a, lapack_int* lda, - lapack_complex_float* b, lapack_int* ldb ); -void LAPACK_zlacpy( char* uplo, lapack_int* m, lapack_int* n, - const lapack_complex_double* a, lapack_int* lda, - lapack_complex_double* b, lapack_int* ldb ); -void LAPACK_sgetf2( lapack_int* m, lapack_int* n, float* a, lapack_int* lda, - lapack_int* ipiv, lapack_int *info ); -void LAPACK_dgetf2( lapack_int* m, lapack_int* n, double* a, lapack_int* lda, - lapack_int* ipiv, lapack_int *info ); -void LAPACK_cgetf2( lapack_int* m, lapack_int* n, lapack_complex_float* a, - lapack_int* lda, lapack_int* ipiv, lapack_int *info ); -void LAPACK_zgetf2( lapack_int* m, lapack_int* n, lapack_complex_double* a, - lapack_int* lda, lapack_int* ipiv, lapack_int *info ); -void LAPACK_slaswp( lapack_int* n, float* a, lapack_int* lda, lapack_int* k1, - lapack_int* k2, const lapack_int* ipiv, lapack_int* incx ); -void LAPACK_dlaswp( lapack_int* n, double* a, lapack_int* lda, lapack_int* k1, - lapack_int* k2, const lapack_int* ipiv, lapack_int* incx ); -void LAPACK_claswp( lapack_int* n, lapack_complex_float* a, lapack_int* lda, - lapack_int* k1, lapack_int* k2, const lapack_int* ipiv, - lapack_int* incx ); -void LAPACK_zlaswp( lapack_int* n, lapack_complex_double* a, lapack_int* lda, - lapack_int* k1, lapack_int* k2, const lapack_int* ipiv, - lapack_int* incx ); -float LAPACK_slange( char* norm, lapack_int* m, lapack_int* n, const float* a, - lapack_int* lda, float* work ); -double LAPACK_dlange( char* norm, lapack_int* m, lapack_int* n, const double* a, - lapack_int* lda, double* work ); -float LAPACK_clange( char* norm, lapack_int* m, lapack_int* n, - const lapack_complex_float* a, lapack_int* lda, float* work ); -double LAPACK_zlange( char* norm, lapack_int* m, lapack_int* n, - const lapack_complex_double* a, lapack_int* lda, double* work ); -float LAPACK_clanhe( char* norm, char* uplo, lapack_int* n, - const lapack_complex_float* a, lapack_int* lda, float* work ); -double LAPACK_zlanhe( char* norm, char* uplo, lapack_int* n, - const lapack_complex_double* a, lapack_int* lda, double* work ); -float LAPACK_slansy( char* norm, char* uplo, lapack_int* n, const float* a, - lapack_int* lda, float* work ); -double LAPACK_dlansy( char* norm, char* uplo, lapack_int* n, const double* a, - lapack_int* lda, double* work ); -float LAPACK_clansy( char* norm, char* uplo, lapack_int* n, - const lapack_complex_float* a, lapack_int* lda, float* work ); -double LAPACK_zlansy( char* norm, char* uplo, lapack_int* n, - const lapack_complex_double* a, lapack_int* lda, double* work ); -float LAPACK_slantr( char* norm, char* uplo, char* diag, lapack_int* m, - lapack_int* n, const float* a, lapack_int* lda, float* work ); -double LAPACK_dlantr( char* norm, char* uplo, char* diag, lapack_int* m, - lapack_int* n, const double* a, lapack_int* lda, double* work ); -float LAPACK_clantr( char* norm, char* uplo, char* diag, lapack_int* m, - lapack_int* n, const lapack_complex_float* a, lapack_int* lda, - float* work ); -double LAPACK_zlantr( char* norm, char* uplo, char* diag, lapack_int* m, - lapack_int* n, const lapack_complex_double* a, lapack_int* lda, - double* work ); -float LAPACK_slamch( char* cmach ); -double LAPACK_dlamch( char* cmach ); -void LAPACK_sgelq2( lapack_int* m, lapack_int* n, float* a, lapack_int* lda, - float* tau, float* work, lapack_int *info ); -void LAPACK_dgelq2( lapack_int* m, lapack_int* n, double* a, lapack_int* lda, - double* tau, double* work, lapack_int *info ); -void LAPACK_cgelq2( lapack_int* m, lapack_int* n, lapack_complex_float* a, - lapack_int* lda, lapack_complex_float* tau, - lapack_complex_float* work, lapack_int *info ); -void LAPACK_zgelq2( lapack_int* m, lapack_int* n, lapack_complex_double* a, - lapack_int* lda, lapack_complex_double* tau, - lapack_complex_double* work, lapack_int *info ); -void LAPACK_slarfb( char* side, char* trans, char* direct, char* storev, - lapack_int* m, lapack_int* n, lapack_int* k, const float* v, - lapack_int* ldv, const float* t, lapack_int* ldt, float* c, - lapack_int* ldc, float* work, lapack_int* ldwork ); -void LAPACK_dlarfb( char* side, char* trans, char* direct, char* storev, - lapack_int* m, lapack_int* n, lapack_int* k, - const double* v, lapack_int* ldv, const double* t, - lapack_int* ldt, double* c, lapack_int* ldc, double* work, - lapack_int* ldwork ); -void LAPACK_clarfb( char* side, char* trans, char* direct, char* storev, - lapack_int* m, lapack_int* n, lapack_int* k, - const lapack_complex_float* v, lapack_int* ldv, - const lapack_complex_float* t, lapack_int* ldt, - lapack_complex_float* c, lapack_int* ldc, - lapack_complex_float* work, lapack_int* ldwork ); -void LAPACK_zlarfb( char* side, char* trans, char* direct, char* storev, - lapack_int* m, lapack_int* n, lapack_int* k, - const lapack_complex_double* v, lapack_int* ldv, - const lapack_complex_double* t, lapack_int* ldt, - lapack_complex_double* c, lapack_int* ldc, - lapack_complex_double* work, lapack_int* ldwork ); -void LAPACK_slarfg( lapack_int* n, float* alpha, float* x, lapack_int* incx, - float* tau ); -void LAPACK_dlarfg( lapack_int* n, double* alpha, double* x, lapack_int* incx, - double* tau ); -void LAPACK_clarfg( lapack_int* n, lapack_complex_float* alpha, - lapack_complex_float* x, lapack_int* incx, - lapack_complex_float* tau ); -void LAPACK_zlarfg( lapack_int* n, lapack_complex_double* alpha, - lapack_complex_double* x, lapack_int* incx, - lapack_complex_double* tau ); -void LAPACK_slarft( char* direct, char* storev, lapack_int* n, lapack_int* k, - const float* v, lapack_int* ldv, const float* tau, float* t, - lapack_int* ldt ); -void LAPACK_dlarft( char* direct, char* storev, lapack_int* n, lapack_int* k, - const double* v, lapack_int* ldv, const double* tau, - double* t, lapack_int* ldt ); -void LAPACK_clarft( char* direct, char* storev, lapack_int* n, lapack_int* k, - const lapack_complex_float* v, lapack_int* ldv, - const lapack_complex_float* tau, lapack_complex_float* t, - lapack_int* ldt ); -void LAPACK_zlarft( char* direct, char* storev, lapack_int* n, lapack_int* k, - const lapack_complex_double* v, lapack_int* ldv, - const lapack_complex_double* tau, lapack_complex_double* t, - lapack_int* ldt ); -void LAPACK_slarfx( char* side, lapack_int* m, lapack_int* n, const float* v, - float* tau, float* c, lapack_int* ldc, float* work ); -void LAPACK_dlarfx( char* side, lapack_int* m, lapack_int* n, const double* v, - double* tau, double* c, lapack_int* ldc, double* work ); -void LAPACK_clarfx( char* side, lapack_int* m, lapack_int* n, - const lapack_complex_float* v, lapack_complex_float* tau, - lapack_complex_float* c, lapack_int* ldc, - lapack_complex_float* work ); -void LAPACK_zlarfx( char* side, lapack_int* m, lapack_int* n, - const lapack_complex_double* v, lapack_complex_double* tau, - lapack_complex_double* c, lapack_int* ldc, - lapack_complex_double* work ); -void LAPACK_slatms( lapack_int* m, lapack_int* n, char* dist, lapack_int* iseed, - char* sym, float* d, lapack_int* mode, float* cond, - float* dmax, lapack_int* kl, lapack_int* ku, char* pack, - float* a, lapack_int* lda, float* work, lapack_int *info ); -void LAPACK_dlatms( lapack_int* m, lapack_int* n, char* dist, lapack_int* iseed, - char* sym, double* d, lapack_int* mode, double* cond, - double* dmax, lapack_int* kl, lapack_int* ku, char* pack, - double* a, lapack_int* lda, double* work, - lapack_int *info ); -void LAPACK_clatms( lapack_int* m, lapack_int* n, char* dist, lapack_int* iseed, - char* sym, float* d, lapack_int* mode, float* cond, - float* dmax, lapack_int* kl, lapack_int* ku, char* pack, - lapack_complex_float* a, lapack_int* lda, - lapack_complex_float* work, lapack_int *info ); -void LAPACK_zlatms( lapack_int* m, lapack_int* n, char* dist, lapack_int* iseed, - char* sym, double* d, lapack_int* mode, double* cond, - double* dmax, lapack_int* kl, lapack_int* ku, char* pack, - lapack_complex_double* a, lapack_int* lda, - lapack_complex_double* work, lapack_int *info ); -void LAPACK_slag2d( lapack_int* m, lapack_int* n, const float* sa, - lapack_int* ldsa, double* a, lapack_int* lda, - lapack_int *info ); -void LAPACK_dlag2s( lapack_int* m, lapack_int* n, const double* a, - lapack_int* lda, float* sa, lapack_int* ldsa, - lapack_int *info ); -void LAPACK_clag2z( lapack_int* m, lapack_int* n, - const lapack_complex_float* sa, lapack_int* ldsa, - lapack_complex_double* a, lapack_int* lda, - lapack_int *info ); -void LAPACK_zlag2c( lapack_int* m, lapack_int* n, - const lapack_complex_double* a, lapack_int* lda, - lapack_complex_float* sa, lapack_int* ldsa, - lapack_int *info ); -void LAPACK_slauum( char* uplo, lapack_int* n, float* a, lapack_int* lda, - lapack_int *info ); -void LAPACK_dlauum( char* uplo, lapack_int* n, double* a, lapack_int* lda, - lapack_int *info ); -void LAPACK_clauum( char* uplo, lapack_int* n, lapack_complex_float* a, - lapack_int* lda, lapack_int *info ); -void LAPACK_zlauum( char* uplo, lapack_int* n, lapack_complex_double* a, - lapack_int* lda, lapack_int *info ); -void LAPACK_slagge( lapack_int* m, lapack_int* n, lapack_int* kl, - lapack_int* ku, const float* d, float* a, lapack_int* lda, - lapack_int* iseed, float* work, lapack_int *info ); -void LAPACK_dlagge( lapack_int* m, lapack_int* n, lapack_int* kl, - lapack_int* ku, const double* d, double* a, lapack_int* lda, - lapack_int* iseed, double* work, lapack_int *info ); -void LAPACK_clagge( lapack_int* m, lapack_int* n, lapack_int* kl, - lapack_int* ku, const float* d, lapack_complex_float* a, - lapack_int* lda, lapack_int* iseed, - lapack_complex_float* work, lapack_int *info ); -void LAPACK_zlagge( lapack_int* m, lapack_int* n, lapack_int* kl, - lapack_int* ku, const double* d, lapack_complex_double* a, - lapack_int* lda, lapack_int* iseed, - lapack_complex_double* work, lapack_int *info ); -void LAPACK_slaset( char* uplo, lapack_int* m, lapack_int* n, float* alpha, - float* beta, float* a, lapack_int* lda ); -void LAPACK_dlaset( char* uplo, lapack_int* m, lapack_int* n, double* alpha, - double* beta, double* a, lapack_int* lda ); -void LAPACK_claset( char* uplo, lapack_int* m, lapack_int* n, - lapack_complex_float* alpha, lapack_complex_float* beta, - lapack_complex_float* a, lapack_int* lda ); -void LAPACK_zlaset( char* uplo, lapack_int* m, lapack_int* n, - lapack_complex_double* alpha, lapack_complex_double* beta, - lapack_complex_double* a, lapack_int* lda ); -void LAPACK_slasrt( char* id, lapack_int* n, float* d, lapack_int *info ); -void LAPACK_dlasrt( char* id, lapack_int* n, double* d, lapack_int *info ); -void LAPACK_claghe( lapack_int* n, lapack_int* k, const float* d, - lapack_complex_float* a, lapack_int* lda, lapack_int* iseed, - lapack_complex_float* work, lapack_int *info ); -void LAPACK_zlaghe( lapack_int* n, lapack_int* k, const double* d, - lapack_complex_double* a, lapack_int* lda, - lapack_int* iseed, lapack_complex_double* work, - lapack_int *info ); -void LAPACK_slagsy( lapack_int* n, lapack_int* k, const float* d, float* a, - lapack_int* lda, lapack_int* iseed, float* work, - lapack_int *info ); -void LAPACK_dlagsy( lapack_int* n, lapack_int* k, const double* d, double* a, - lapack_int* lda, lapack_int* iseed, double* work, - lapack_int *info ); -void LAPACK_clagsy( lapack_int* n, lapack_int* k, const float* d, - lapack_complex_float* a, lapack_int* lda, lapack_int* iseed, - lapack_complex_float* work, lapack_int *info ); -void LAPACK_zlagsy( lapack_int* n, lapack_int* k, const double* d, - lapack_complex_double* a, lapack_int* lda, - lapack_int* iseed, lapack_complex_double* work, - lapack_int *info ); -void LAPACK_slapmr( lapack_logical* forwrd, lapack_int* m, lapack_int* n, - float* x, lapack_int* ldx, lapack_int* k ); -void LAPACK_dlapmr( lapack_logical* forwrd, lapack_int* m, lapack_int* n, - double* x, lapack_int* ldx, lapack_int* k ); -void LAPACK_clapmr( lapack_logical* forwrd, lapack_int* m, lapack_int* n, - lapack_complex_float* x, lapack_int* ldx, lapack_int* k ); -void LAPACK_zlapmr( lapack_logical* forwrd, lapack_int* m, lapack_int* n, - lapack_complex_double* x, lapack_int* ldx, lapack_int* k ); -float LAPACK_slapy2( float* x, float* y ); -double LAPACK_dlapy2( double* x, double* y ); -float LAPACK_slapy3( float* x, float* y, float* z ); -double LAPACK_dlapy3( double* x, double* y, double* z ); -void LAPACK_slartgp( float* f, float* g, float* cs, float* sn, float* r ); -void LAPACK_dlartgp( double* f, double* g, double* cs, double* sn, double* r ); -void LAPACK_slartgs( float* x, float* y, float* sigma, float* cs, float* sn ); -void LAPACK_dlartgs( double* x, double* y, double* sigma, double* cs, - double* sn ); +#define LAPACK_csyr LAPACK_GLOBAL(csyr, CSYR) +#define LAPACK_zsyr LAPACK_GLOBAL(zsyr, ZSYR) + + +void LAPACK_sgetrf(lapack_int *m, lapack_int *n, float *a, lapack_int *lda, lapack_int *ipiv, lapack_int *info); +void LAPACK_dgetrf(lapack_int *m, lapack_int *n, double *a, lapack_int *lda, lapack_int *ipiv, lapack_int *info); +void LAPACK_cgetrf(lapack_int *m, + lapack_int *n, + lapack_complex_float *a, + lapack_int *lda, + lapack_int *ipiv, + lapack_int *info); +void LAPACK_zgetrf(lapack_int *m, + lapack_int *n, + lapack_complex_double *a, + lapack_int *lda, + lapack_int *ipiv, + lapack_int *info); +void LAPACK_sgbtrf(lapack_int *m, + lapack_int *n, + lapack_int *kl, + lapack_int *ku, + float *ab, + lapack_int *ldab, + lapack_int *ipiv, + lapack_int *info); +void LAPACK_dgbtrf(lapack_int *m, + lapack_int *n, + lapack_int *kl, + lapack_int *ku, + double *ab, + lapack_int *ldab, + lapack_int *ipiv, + lapack_int *info); +void LAPACK_cgbtrf(lapack_int *m, + lapack_int *n, + lapack_int *kl, + lapack_int *ku, + lapack_complex_float *ab, + lapack_int *ldab, + lapack_int *ipiv, + lapack_int *info); +void LAPACK_zgbtrf(lapack_int *m, + lapack_int *n, + lapack_int *kl, + lapack_int *ku, + lapack_complex_double *ab, + lapack_int *ldab, + lapack_int *ipiv, + lapack_int *info); +void LAPACK_sgttrf(lapack_int *n, float *dl, float *d, float *du, float *du2, lapack_int *ipiv, lapack_int *info); +void LAPACK_dgttrf(lapack_int *n, double *dl, double *d, double *du, double *du2, lapack_int *ipiv, lapack_int *info); +void LAPACK_cgttrf(lapack_int *n, + lapack_complex_float *dl, + lapack_complex_float *d, + lapack_complex_float *du, + lapack_complex_float *du2, + lapack_int *ipiv, + lapack_int *info); +void LAPACK_zgttrf(lapack_int *n, + lapack_complex_double *dl, + lapack_complex_double *d, + lapack_complex_double *du, + lapack_complex_double *du2, + lapack_int *ipiv, + lapack_int *info); +void LAPACK_spotrf(char *uplo, lapack_int *n, float *a, lapack_int *lda, lapack_int *info); +void LAPACK_dpotrf(char *uplo, lapack_int *n, double *a, lapack_int *lda, lapack_int *info); +void LAPACK_cpotrf(char *uplo, lapack_int *n, lapack_complex_float *a, lapack_int *lda, lapack_int *info); +void LAPACK_zpotrf(char *uplo, lapack_int *n, lapack_complex_double *a, lapack_int *lda, lapack_int *info); +void LAPACK_dpstrf(char *uplo, + lapack_int *n, + double *a, + lapack_int *lda, + lapack_int *piv, + lapack_int *rank, + double *tol, + double *work, + lapack_int *info); +void LAPACK_spstrf(char *uplo, + lapack_int *n, + float *a, + lapack_int *lda, + lapack_int *piv, + lapack_int *rank, + float *tol, + float *work, + lapack_int *info); +void LAPACK_zpstrf(char *uplo, + lapack_int *n, + lapack_complex_double *a, + lapack_int *lda, + lapack_int *piv, + lapack_int *rank, + double *tol, + double *work, + lapack_int *info); +void LAPACK_cpstrf(char *uplo, + lapack_int *n, + lapack_complex_float *a, + lapack_int *lda, + lapack_int *piv, + lapack_int *rank, + float *tol, + float *work, + lapack_int *info); +void LAPACK_dpftrf(char *transr, char *uplo, lapack_int *n, double *a, lapack_int *info); +void LAPACK_spftrf(char *transr, char *uplo, lapack_int *n, float *a, lapack_int *info); +void LAPACK_zpftrf(char *transr, char *uplo, lapack_int *n, lapack_complex_double *a, lapack_int *info); +void LAPACK_cpftrf(char *transr, char *uplo, lapack_int *n, lapack_complex_float *a, lapack_int *info); +void LAPACK_spptrf(char *uplo, lapack_int *n, float *ap, lapack_int *info); +void LAPACK_dpptrf(char *uplo, lapack_int *n, double *ap, lapack_int *info); +void LAPACK_cpptrf(char *uplo, lapack_int *n, lapack_complex_float *ap, lapack_int *info); +void LAPACK_zpptrf(char *uplo, lapack_int *n, lapack_complex_double *ap, lapack_int *info); +void LAPACK_spbtrf(char *uplo, lapack_int *n, lapack_int *kd, float *ab, lapack_int *ldab, lapack_int *info); +void LAPACK_dpbtrf(char *uplo, lapack_int *n, lapack_int *kd, double *ab, lapack_int *ldab, lapack_int *info); +void LAPACK_cpbtrf(char *uplo, + lapack_int *n, + lapack_int *kd, + lapack_complex_float *ab, + lapack_int *ldab, + lapack_int *info); +void LAPACK_zpbtrf(char *uplo, + lapack_int *n, + lapack_int *kd, + lapack_complex_double *ab, + lapack_int *ldab, + lapack_int *info); +void LAPACK_spttrf(lapack_int *n, float *d, float *e, lapack_int *info); +void LAPACK_dpttrf(lapack_int *n, double *d, double *e, lapack_int *info); +void LAPACK_cpttrf(lapack_int *n, float *d, lapack_complex_float *e, lapack_int *info); +void LAPACK_zpttrf(lapack_int *n, double *d, lapack_complex_double *e, lapack_int *info); +void LAPACK_ssytrf(char *uplo, + lapack_int *n, + float *a, + lapack_int *lda, + lapack_int *ipiv, + float *work, + lapack_int *lwork, + lapack_int *info); +void LAPACK_dsytrf(char *uplo, + lapack_int *n, + double *a, + lapack_int *lda, + lapack_int *ipiv, + double *work, + lapack_int *lwork, + lapack_int *info); +void LAPACK_csytrf(char *uplo, + lapack_int *n, + lapack_complex_float *a, + lapack_int *lda, + lapack_int *ipiv, + lapack_complex_float *work, + lapack_int *lwork, + lapack_int *info); +void LAPACK_zsytrf(char *uplo, + lapack_int *n, + lapack_complex_double *a, + lapack_int *lda, + lapack_int *ipiv, + lapack_complex_double *work, + lapack_int *lwork, + lapack_int *info); +void LAPACK_chetrf(char *uplo, + lapack_int *n, + lapack_complex_float *a, + lapack_int *lda, + lapack_int *ipiv, + lapack_complex_float *work, + lapack_int *lwork, + lapack_int *info); +void LAPACK_zhetrf(char *uplo, + lapack_int *n, + lapack_complex_double *a, + lapack_int *lda, + lapack_int *ipiv, + lapack_complex_double *work, + lapack_int *lwork, + lapack_int *info); +void LAPACK_ssptrf(char *uplo, lapack_int *n, float *ap, lapack_int *ipiv, lapack_int *info); +void LAPACK_dsptrf(char *uplo, lapack_int *n, double *ap, lapack_int *ipiv, lapack_int *info); +void LAPACK_csptrf(char *uplo, lapack_int *n, lapack_complex_float *ap, lapack_int *ipiv, lapack_int *info); +void LAPACK_zsptrf(char *uplo, lapack_int *n, lapack_complex_double *ap, lapack_int *ipiv, lapack_int *info); +void LAPACK_chptrf(char *uplo, lapack_int *n, lapack_complex_float *ap, lapack_int *ipiv, lapack_int *info); +void LAPACK_zhptrf(char *uplo, lapack_int *n, lapack_complex_double *ap, lapack_int *ipiv, lapack_int *info); +void LAPACK_sgetrs(char *trans, + lapack_int *n, + lapack_int *nrhs, + const float *a, + lapack_int *lda, + const lapack_int *ipiv, + float *b, + lapack_int *ldb, + lapack_int *info); +void LAPACK_dgetrs(char *trans, + lapack_int *n, + lapack_int *nrhs, + const double *a, + lapack_int *lda, + const lapack_int *ipiv, + double *b, + lapack_int *ldb, + lapack_int *info); +void LAPACK_cgetrs(char *trans, + lapack_int *n, + lapack_int *nrhs, + const lapack_complex_float *a, + lapack_int *lda, + const lapack_int *ipiv, + lapack_complex_float *b, + lapack_int *ldb, + lapack_int *info); +void LAPACK_zgetrs(char *trans, + lapack_int *n, + lapack_int *nrhs, + const lapack_complex_double *a, + lapack_int *lda, + const lapack_int *ipiv, + lapack_complex_double *b, + lapack_int *ldb, + lapack_int *info); +void LAPACK_sgbtrs(char *trans, + lapack_int *n, + lapack_int *kl, + lapack_int *ku, + lapack_int *nrhs, + const float *ab, + lapack_int *ldab, + const lapack_int *ipiv, + float *b, + lapack_int *ldb, + lapack_int *info); +void LAPACK_dgbtrs(char *trans, + lapack_int *n, + lapack_int *kl, + lapack_int *ku, + lapack_int *nrhs, + const double *ab, + lapack_int *ldab, + const lapack_int *ipiv, + double *b, + lapack_int *ldb, + lapack_int *info); +void LAPACK_cgbtrs(char *trans, + lapack_int *n, + lapack_int *kl, + lapack_int *ku, + lapack_int *nrhs, + const lapack_complex_float *ab, + lapack_int *ldab, + const lapack_int *ipiv, + lapack_complex_float *b, + lapack_int *ldb, + lapack_int *info); +void LAPACK_zgbtrs(char *trans, + lapack_int *n, + lapack_int *kl, + lapack_int *ku, + lapack_int *nrhs, + const lapack_complex_double *ab, + lapack_int *ldab, + const lapack_int *ipiv, + lapack_complex_double *b, + lapack_int *ldb, + lapack_int *info); +void LAPACK_sgttrs(char *trans, + lapack_int *n, + lapack_int *nrhs, + const float *dl, + const float *d, + const float *du, + const float *du2, + const lapack_int *ipiv, + float *b, + lapack_int *ldb, + lapack_int *info); +void LAPACK_dgttrs(char *trans, + lapack_int *n, + lapack_int *nrhs, + const double *dl, + const double *d, + const double *du, + const double *du2, + const lapack_int *ipiv, + double *b, + lapack_int *ldb, + lapack_int *info); +void LAPACK_cgttrs(char *trans, + lapack_int *n, + lapack_int *nrhs, + const lapack_complex_float *dl, + const lapack_complex_float *d, + const lapack_complex_float *du, + const lapack_complex_float *du2, + const lapack_int *ipiv, + lapack_complex_float *b, + lapack_int *ldb, + lapack_int *info); +void LAPACK_zgttrs(char *trans, + lapack_int *n, + lapack_int *nrhs, + const lapack_complex_double *dl, + const lapack_complex_double *d, + const lapack_complex_double *du, + const lapack_complex_double *du2, + const lapack_int *ipiv, + lapack_complex_double *b, + lapack_int *ldb, + lapack_int *info); +void LAPACK_spotrs(char *uplo, + lapack_int *n, + lapack_int *nrhs, + const float *a, + lapack_int *lda, + float *b, + lapack_int *ldb, + lapack_int *info); +void LAPACK_dpotrs(char *uplo, + lapack_int *n, + lapack_int *nrhs, + const double *a, + lapack_int *lda, + double *b, + lapack_int *ldb, + lapack_int *info); +void LAPACK_cpotrs(char *uplo, + lapack_int *n, + lapack_int *nrhs, + const lapack_complex_float *a, + lapack_int *lda, + lapack_complex_float *b, + lapack_int *ldb, + lapack_int *info); +void LAPACK_zpotrs(char *uplo, + lapack_int *n, + lapack_int *nrhs, + const lapack_complex_double *a, + lapack_int *lda, + lapack_complex_double *b, + lapack_int *ldb, + lapack_int *info); +void LAPACK_dpftrs(char *transr, + char *uplo, + lapack_int *n, + lapack_int *nrhs, + const double *a, + double *b, + lapack_int *ldb, + lapack_int *info); +void LAPACK_spftrs(char *transr, + char *uplo, + lapack_int *n, + lapack_int *nrhs, + const float *a, + float *b, + lapack_int *ldb, + lapack_int *info); +void LAPACK_zpftrs(char *transr, + char *uplo, + lapack_int *n, + lapack_int *nrhs, + const lapack_complex_double *a, + lapack_complex_double *b, + lapack_int *ldb, + lapack_int *info); +void LAPACK_cpftrs(char *transr, + char *uplo, + lapack_int *n, + lapack_int *nrhs, + const lapack_complex_float *a, + lapack_complex_float *b, + lapack_int *ldb, + lapack_int *info); +void LAPACK_spptrs(char *uplo, + lapack_int *n, + lapack_int *nrhs, + const float *ap, + float *b, + lapack_int *ldb, + lapack_int *info); +void LAPACK_dpptrs(char *uplo, + lapack_int *n, + lapack_int *nrhs, + const double *ap, + double *b, + lapack_int *ldb, + lapack_int *info); +void LAPACK_cpptrs(char *uplo, + lapack_int *n, + lapack_int *nrhs, + const lapack_complex_float *ap, + lapack_complex_float *b, + lapack_int *ldb, + lapack_int *info); +void LAPACK_zpptrs(char *uplo, + lapack_int *n, + lapack_int *nrhs, + const lapack_complex_double *ap, + lapack_complex_double *b, + lapack_int *ldb, + lapack_int *info); +void LAPACK_spbtrs(char *uplo, + lapack_int *n, + lapack_int *kd, + lapack_int *nrhs, + const float *ab, + lapack_int *ldab, + float *b, + lapack_int *ldb, + lapack_int *info); +void LAPACK_dpbtrs(char *uplo, + lapack_int *n, + lapack_int *kd, + lapack_int *nrhs, + const double *ab, + lapack_int *ldab, + double *b, + lapack_int *ldb, + lapack_int *info); +void LAPACK_cpbtrs(char *uplo, + lapack_int *n, + lapack_int *kd, + lapack_int *nrhs, + const lapack_complex_float *ab, + lapack_int *ldab, + lapack_complex_float *b, + lapack_int *ldb, + lapack_int *info); +void LAPACK_zpbtrs(char *uplo, + lapack_int *n, + lapack_int *kd, + lapack_int *nrhs, + const lapack_complex_double *ab, + lapack_int *ldab, + lapack_complex_double *b, + lapack_int *ldb, + lapack_int *info); +void LAPACK_spttrs(lapack_int *n, + lapack_int *nrhs, + const float *d, + const float *e, + float *b, + lapack_int *ldb, + lapack_int *info); +void LAPACK_dpttrs(lapack_int *n, + lapack_int *nrhs, + const double *d, + const double *e, + double *b, + lapack_int *ldb, + lapack_int *info); +void LAPACK_cpttrs(char *uplo, + lapack_int *n, + lapack_int *nrhs, + const float *d, + const lapack_complex_float *e, + lapack_complex_float *b, + lapack_int *ldb, + lapack_int *info); +void LAPACK_zpttrs(char *uplo, + lapack_int *n, + lapack_int *nrhs, + const double *d, + const lapack_complex_double *e, + lapack_complex_double *b, + lapack_int *ldb, + lapack_int *info); +void LAPACK_ssytrs(char *uplo, + lapack_int *n, + lapack_int *nrhs, + const float *a, + lapack_int *lda, + const lapack_int *ipiv, + float *b, + lapack_int *ldb, + lapack_int *info); +void LAPACK_dsytrs(char *uplo, + lapack_int *n, + lapack_int *nrhs, + const double *a, + lapack_int *lda, + const lapack_int *ipiv, + double *b, + lapack_int *ldb, + lapack_int *info); +void LAPACK_csytrs(char *uplo, + lapack_int *n, + lapack_int *nrhs, + const lapack_complex_float *a, + lapack_int *lda, + const lapack_int *ipiv, + lapack_complex_float *b, + lapack_int *ldb, + lapack_int *info); +void LAPACK_zsytrs(char *uplo, + lapack_int *n, + lapack_int *nrhs, + const lapack_complex_double *a, + lapack_int *lda, + const lapack_int *ipiv, + lapack_complex_double *b, + lapack_int *ldb, + lapack_int *info); +void LAPACK_chetrs(char *uplo, + lapack_int *n, + lapack_int *nrhs, + const lapack_complex_float *a, + lapack_int *lda, + const lapack_int *ipiv, + lapack_complex_float *b, + lapack_int *ldb, + lapack_int *info); +void LAPACK_zhetrs(char *uplo, + lapack_int *n, + lapack_int *nrhs, + const lapack_complex_double *a, + lapack_int *lda, + const lapack_int *ipiv, + lapack_complex_double *b, + lapack_int *ldb, + lapack_int *info); +void LAPACK_ssptrs(char *uplo, + lapack_int *n, + lapack_int *nrhs, + const float *ap, + const lapack_int *ipiv, + float *b, + lapack_int *ldb, + lapack_int *info); +void LAPACK_dsptrs(char *uplo, + lapack_int *n, + lapack_int *nrhs, + const double *ap, + const lapack_int *ipiv, + double *b, + lapack_int *ldb, + lapack_int *info); +void LAPACK_csptrs(char *uplo, + lapack_int *n, + lapack_int *nrhs, + const lapack_complex_float *ap, + const lapack_int *ipiv, + lapack_complex_float *b, + lapack_int *ldb, + lapack_int *info); +void LAPACK_zsptrs(char *uplo, + lapack_int *n, + lapack_int *nrhs, + const lapack_complex_double *ap, + const lapack_int *ipiv, + lapack_complex_double *b, + lapack_int *ldb, + lapack_int *info); +void LAPACK_chptrs(char *uplo, + lapack_int *n, + lapack_int *nrhs, + const lapack_complex_float *ap, + const lapack_int *ipiv, + lapack_complex_float *b, + lapack_int *ldb, + lapack_int *info); +void LAPACK_zhptrs(char *uplo, + lapack_int *n, + lapack_int *nrhs, + const lapack_complex_double *ap, + const lapack_int *ipiv, + lapack_complex_double *b, + lapack_int *ldb, + lapack_int *info); +void LAPACK_strtrs(char *uplo, + char *trans, + char *diag, + lapack_int *n, + lapack_int *nrhs, + const float *a, + lapack_int *lda, + float *b, + lapack_int *ldb, + lapack_int *info); +void LAPACK_dtrtrs(char *uplo, + char *trans, + char *diag, + lapack_int *n, + lapack_int *nrhs, + const double *a, + lapack_int *lda, + double *b, + lapack_int *ldb, + lapack_int *info); +void LAPACK_ctrtrs(char *uplo, + char *trans, + char *diag, + lapack_int *n, + lapack_int *nrhs, + const lapack_complex_float *a, + lapack_int *lda, + lapack_complex_float *b, + lapack_int *ldb, + lapack_int *info); +void LAPACK_ztrtrs(char *uplo, + char *trans, + char *diag, + lapack_int *n, + lapack_int *nrhs, + const lapack_complex_double *a, + lapack_int *lda, + lapack_complex_double *b, + lapack_int *ldb, + lapack_int *info); +void LAPACK_stptrs(char *uplo, + char *trans, + char *diag, + lapack_int *n, + lapack_int *nrhs, + const float *ap, + float *b, + lapack_int *ldb, + lapack_int *info); +void LAPACK_dtptrs(char *uplo, + char *trans, + char *diag, + lapack_int *n, + lapack_int *nrhs, + const double *ap, + double *b, + lapack_int *ldb, + lapack_int *info); +void LAPACK_ctptrs(char *uplo, + char *trans, + char *diag, + lapack_int *n, + lapack_int *nrhs, + const lapack_complex_float *ap, + lapack_complex_float *b, + lapack_int *ldb, + lapack_int *info); +void LAPACK_ztptrs(char *uplo, + char *trans, + char *diag, + lapack_int *n, + lapack_int *nrhs, + const lapack_complex_double *ap, + lapack_complex_double *b, + lapack_int *ldb, + lapack_int *info); +void LAPACK_stbtrs(char *uplo, + char *trans, + char *diag, + lapack_int *n, + lapack_int *kd, + lapack_int *nrhs, + const float *ab, + lapack_int *ldab, + float *b, + lapack_int *ldb, + lapack_int *info); +void LAPACK_dtbtrs(char *uplo, + char *trans, + char *diag, + lapack_int *n, + lapack_int *kd, + lapack_int *nrhs, + const double *ab, + lapack_int *ldab, + double *b, + lapack_int *ldb, + lapack_int *info); +void LAPACK_ctbtrs(char *uplo, + char *trans, + char *diag, + lapack_int *n, + lapack_int *kd, + lapack_int *nrhs, + const lapack_complex_float *ab, + lapack_int *ldab, + lapack_complex_float *b, + lapack_int *ldb, + lapack_int *info); +void LAPACK_ztbtrs(char *uplo, + char *trans, + char *diag, + lapack_int *n, + lapack_int *kd, + lapack_int *nrhs, + const lapack_complex_double *ab, + lapack_int *ldab, + lapack_complex_double *b, + lapack_int *ldb, + lapack_int *info); +void LAPACK_sgecon(char *norm, + lapack_int *n, + const float *a, + lapack_int *lda, + float *anorm, + float *rcond, + float *work, + lapack_int *iwork, + lapack_int *info); +void LAPACK_dgecon(char *norm, + lapack_int *n, + const double *a, + lapack_int *lda, + double *anorm, + double *rcond, + double *work, + lapack_int *iwork, + lapack_int *info); +void LAPACK_cgecon(char *norm, + lapack_int *n, + const lapack_complex_float *a, + lapack_int *lda, + float *anorm, + float *rcond, + lapack_complex_float *work, + float *rwork, + lapack_int *info); +void LAPACK_zgecon(char *norm, + lapack_int *n, + const lapack_complex_double *a, + lapack_int *lda, + double *anorm, + double *rcond, + lapack_complex_double *work, + double *rwork, + lapack_int *info); +void LAPACK_sgbcon(char *norm, + lapack_int *n, + lapack_int *kl, + lapack_int *ku, + const float *ab, + lapack_int *ldab, + const lapack_int *ipiv, + float *anorm, + float *rcond, + float *work, + lapack_int *iwork, + lapack_int *info); +void LAPACK_dgbcon(char *norm, + lapack_int *n, + lapack_int *kl, + lapack_int *ku, + const double *ab, + lapack_int *ldab, + const lapack_int *ipiv, + double *anorm, + double *rcond, + double *work, + lapack_int *iwork, + lapack_int *info); +void LAPACK_cgbcon(char *norm, + lapack_int *n, + lapack_int *kl, + lapack_int *ku, + const lapack_complex_float *ab, + lapack_int *ldab, + const lapack_int *ipiv, + float *anorm, + float *rcond, + lapack_complex_float *work, + float *rwork, + lapack_int *info); +void LAPACK_zgbcon(char *norm, + lapack_int *n, + lapack_int *kl, + lapack_int *ku, + const lapack_complex_double *ab, + lapack_int *ldab, + const lapack_int *ipiv, + double *anorm, + double *rcond, + lapack_complex_double *work, + double *rwork, + lapack_int *info); +void LAPACK_sgtcon(char *norm, + lapack_int *n, + const float *dl, + const float *d, + const float *du, + const float *du2, + const lapack_int *ipiv, + float *anorm, + float *rcond, + float *work, + lapack_int *iwork, + lapack_int *info); +void LAPACK_dgtcon(char *norm, + lapack_int *n, + const double *dl, + const double *d, + const double *du, + const double *du2, + const lapack_int *ipiv, + double *anorm, + double *rcond, + double *work, + lapack_int *iwork, + lapack_int *info); +void LAPACK_cgtcon(char *norm, + lapack_int *n, + const lapack_complex_float *dl, + const lapack_complex_float *d, + const lapack_complex_float *du, + const lapack_complex_float *du2, + const lapack_int *ipiv, + float *anorm, + float *rcond, + lapack_complex_float *work, + lapack_int *info); +void LAPACK_zgtcon(char *norm, + lapack_int *n, + const lapack_complex_double *dl, + const lapack_complex_double *d, + const lapack_complex_double *du, + const lapack_complex_double *du2, + const lapack_int *ipiv, + double *anorm, + double *rcond, + lapack_complex_double *work, + lapack_int *info); +void LAPACK_spocon(char *uplo, + lapack_int *n, + const float *a, + lapack_int *lda, + float *anorm, + float *rcond, + float *work, + lapack_int *iwork, + lapack_int *info); +void LAPACK_dpocon(char *uplo, + lapack_int *n, + const double *a, + lapack_int *lda, + double *anorm, + double *rcond, + double *work, + lapack_int *iwork, + lapack_int *info); +void LAPACK_cpocon(char *uplo, + lapack_int *n, + const lapack_complex_float *a, + lapack_int *lda, + float *anorm, + float *rcond, + lapack_complex_float *work, + float *rwork, + lapack_int *info); +void LAPACK_zpocon(char *uplo, + lapack_int *n, + const lapack_complex_double *a, + lapack_int *lda, + double *anorm, + double *rcond, + lapack_complex_double *work, + double *rwork, + lapack_int *info); +void LAPACK_sppcon(char *uplo, + lapack_int *n, + const float *ap, + float *anorm, + float *rcond, + float *work, + lapack_int *iwork, + lapack_int *info); +void LAPACK_dppcon(char *uplo, + lapack_int *n, + const double *ap, + double *anorm, + double *rcond, + double *work, + lapack_int *iwork, + lapack_int *info); +void LAPACK_cppcon(char *uplo, + lapack_int *n, + const lapack_complex_float *ap, + float *anorm, + float *rcond, + lapack_complex_float *work, + float *rwork, + lapack_int *info); +void LAPACK_zppcon(char *uplo, + lapack_int *n, + const lapack_complex_double *ap, + double *anorm, + double *rcond, + lapack_complex_double *work, + double *rwork, + lapack_int *info); +void LAPACK_spbcon(char *uplo, + lapack_int *n, + lapack_int *kd, + const float *ab, + lapack_int *ldab, + float *anorm, + float *rcond, + float *work, + lapack_int *iwork, + lapack_int *info); +void LAPACK_dpbcon(char *uplo, + lapack_int *n, + lapack_int *kd, + const double *ab, + lapack_int *ldab, + double *anorm, + double *rcond, + double *work, + lapack_int *iwork, + lapack_int *info); +void LAPACK_cpbcon(char *uplo, + lapack_int *n, + lapack_int *kd, + const lapack_complex_float *ab, + lapack_int *ldab, + float *anorm, + float *rcond, + lapack_complex_float *work, + float *rwork, + lapack_int *info); +void LAPACK_zpbcon(char *uplo, + lapack_int *n, + lapack_int *kd, + const lapack_complex_double *ab, + lapack_int *ldab, + double *anorm, + double *rcond, + lapack_complex_double *work, + double *rwork, + lapack_int *info); +void LAPACK_sptcon(lapack_int *n, + const float *d, + const float *e, + float *anorm, + float *rcond, + float *work, + lapack_int *info); +void LAPACK_dptcon(lapack_int *n, + const double *d, + const double *e, + double *anorm, + double *rcond, + double *work, + lapack_int *info); +void LAPACK_cptcon(lapack_int *n, + const float *d, + const lapack_complex_float *e, + float *anorm, + float *rcond, + float *work, + lapack_int *info); +void LAPACK_zptcon(lapack_int *n, + const double *d, + const lapack_complex_double *e, + double *anorm, + double *rcond, + double *work, + lapack_int *info); +void LAPACK_ssycon(char *uplo, + lapack_int *n, + const float *a, + lapack_int *lda, + const lapack_int *ipiv, + float *anorm, + float *rcond, + float *work, + lapack_int *iwork, + lapack_int *info); +void LAPACK_dsycon(char *uplo, + lapack_int *n, + const double *a, + lapack_int *lda, + const lapack_int *ipiv, + double *anorm, + double *rcond, + double *work, + lapack_int *iwork, + lapack_int *info); +void LAPACK_csycon(char *uplo, + lapack_int *n, + const lapack_complex_float *a, + lapack_int *lda, + const lapack_int *ipiv, + float *anorm, + float *rcond, + lapack_complex_float *work, + lapack_int *info); +void LAPACK_zsycon(char *uplo, + lapack_int *n, + const lapack_complex_double *a, + lapack_int *lda, + const lapack_int *ipiv, + double *anorm, + double *rcond, + lapack_complex_double *work, + lapack_int *info); +void LAPACK_checon(char *uplo, + lapack_int *n, + const lapack_complex_float *a, + lapack_int *lda, + const lapack_int *ipiv, + float *anorm, + float *rcond, + lapack_complex_float *work, + lapack_int *info); +void LAPACK_zhecon(char *uplo, + lapack_int *n, + const lapack_complex_double *a, + lapack_int *lda, + const lapack_int *ipiv, + double *anorm, + double *rcond, + lapack_complex_double *work, + lapack_int *info); +void LAPACK_sspcon(char *uplo, + lapack_int *n, + const float *ap, + const lapack_int *ipiv, + float *anorm, + float *rcond, + float *work, + lapack_int *iwork, + lapack_int *info); +void LAPACK_dspcon(char *uplo, + lapack_int *n, + const double *ap, + const lapack_int *ipiv, + double *anorm, + double *rcond, + double *work, + lapack_int *iwork, + lapack_int *info); +void LAPACK_cspcon(char *uplo, + lapack_int *n, + const lapack_complex_float *ap, + const lapack_int *ipiv, + float *anorm, + float *rcond, + lapack_complex_float *work, + lapack_int *info); +void LAPACK_zspcon(char *uplo, + lapack_int *n, + const lapack_complex_double *ap, + const lapack_int *ipiv, + double *anorm, + double *rcond, + lapack_complex_double *work, + lapack_int *info); +void LAPACK_chpcon(char *uplo, + lapack_int *n, + const lapack_complex_float *ap, + const lapack_int *ipiv, + float *anorm, + float *rcond, + lapack_complex_float *work, + lapack_int *info); +void LAPACK_zhpcon(char *uplo, + lapack_int *n, + const lapack_complex_double *ap, + const lapack_int *ipiv, + double *anorm, + double *rcond, + lapack_complex_double *work, + lapack_int *info); +void LAPACK_strcon(char *norm, + char *uplo, + char *diag, + lapack_int *n, + const float *a, + lapack_int *lda, + float *rcond, + float *work, + lapack_int *iwork, + lapack_int *info); +void LAPACK_dtrcon(char *norm, + char *uplo, + char *diag, + lapack_int *n, + const double *a, + lapack_int *lda, + double *rcond, + double *work, + lapack_int *iwork, + lapack_int *info); +void LAPACK_ctrcon(char *norm, + char *uplo, + char *diag, + lapack_int *n, + const lapack_complex_float *a, + lapack_int *lda, + float *rcond, + lapack_complex_float *work, + float *rwork, + lapack_int *info); +void LAPACK_ztrcon(char *norm, + char *uplo, + char *diag, + lapack_int *n, + const lapack_complex_double *a, + lapack_int *lda, + double *rcond, + lapack_complex_double *work, + double *rwork, + lapack_int *info); +void LAPACK_stpcon(char *norm, + char *uplo, + char *diag, + lapack_int *n, + const float *ap, + float *rcond, + float *work, + lapack_int *iwork, + lapack_int *info); +void LAPACK_dtpcon(char *norm, + char *uplo, + char *diag, + lapack_int *n, + const double *ap, + double *rcond, + double *work, + lapack_int *iwork, + lapack_int *info); +void LAPACK_ctpcon(char *norm, + char *uplo, + char *diag, + lapack_int *n, + const lapack_complex_float *ap, + float *rcond, + lapack_complex_float *work, + float *rwork, + lapack_int *info); +void LAPACK_ztpcon(char *norm, + char *uplo, + char *diag, + lapack_int *n, + const lapack_complex_double *ap, + double *rcond, + lapack_complex_double *work, + double *rwork, + lapack_int *info); +void LAPACK_stbcon(char *norm, + char *uplo, + char *diag, + lapack_int *n, + lapack_int *kd, + const float *ab, + lapack_int *ldab, + float *rcond, + float *work, + lapack_int *iwork, + lapack_int *info); +void LAPACK_dtbcon(char *norm, + char *uplo, + char *diag, + lapack_int *n, + lapack_int *kd, + const double *ab, + lapack_int *ldab, + double *rcond, + double *work, + lapack_int *iwork, + lapack_int *info); +void LAPACK_ctbcon(char *norm, + char *uplo, + char *diag, + lapack_int *n, + lapack_int *kd, + const lapack_complex_float *ab, + lapack_int *ldab, + float *rcond, + lapack_complex_float *work, + float *rwork, + lapack_int *info); +void LAPACK_ztbcon(char *norm, + char *uplo, + char *diag, + lapack_int *n, + lapack_int *kd, + const lapack_complex_double *ab, + lapack_int *ldab, + double *rcond, + lapack_complex_double *work, + double *rwork, + lapack_int *info); +void LAPACK_sgerfs(char *trans, + lapack_int *n, + lapack_int *nrhs, + const float *a, + lapack_int *lda, + const float *af, + lapack_int *ldaf, + const lapack_int *ipiv, + const float *b, + lapack_int *ldb, + float *x, + lapack_int *ldx, + float *ferr, + float *berr, + float *work, + lapack_int *iwork, + lapack_int *info); +void LAPACK_dgerfs(char *trans, + lapack_int *n, + lapack_int *nrhs, + const double *a, + lapack_int *lda, + const double *af, + lapack_int *ldaf, + const lapack_int *ipiv, + const double *b, + lapack_int *ldb, + double *x, + lapack_int *ldx, + double *ferr, + double *berr, + double *work, + lapack_int *iwork, + lapack_int *info); +void LAPACK_cgerfs(char *trans, + lapack_int *n, + lapack_int *nrhs, + const lapack_complex_float *a, + lapack_int *lda, + const lapack_complex_float *af, + lapack_int *ldaf, + const lapack_int *ipiv, + const lapack_complex_float *b, + lapack_int *ldb, + lapack_complex_float *x, + lapack_int *ldx, + float *ferr, + float *berr, + lapack_complex_float *work, + float *rwork, + lapack_int *info); +void LAPACK_zgerfs(char *trans, + lapack_int *n, + lapack_int *nrhs, + const lapack_complex_double *a, + lapack_int *lda, + const lapack_complex_double *af, + lapack_int *ldaf, + const lapack_int *ipiv, + const lapack_complex_double *b, + lapack_int *ldb, + lapack_complex_double *x, + lapack_int *ldx, + double *ferr, + double *berr, + lapack_complex_double *work, + double *rwork, + lapack_int *info); +void LAPACK_dgerfsx(char *trans, + char *equed, + lapack_int *n, + lapack_int *nrhs, + const double *a, + lapack_int *lda, + const double *af, + lapack_int *ldaf, + const lapack_int *ipiv, + const double *r, + const double *c, + const double *b, + lapack_int *ldb, + double *x, + lapack_int *ldx, + double *rcond, + double *berr, + lapack_int *n_err_bnds, + double *err_bnds_norm, + double *err_bnds_comp, + lapack_int *nparams, + double *params, + double *work, + lapack_int *iwork, + lapack_int *info); +void LAPACK_sgerfsx(char *trans, + char *equed, + lapack_int *n, + lapack_int *nrhs, + const float *a, + lapack_int *lda, + const float *af, + lapack_int *ldaf, + const lapack_int *ipiv, + const float *r, + const float *c, + const float *b, + lapack_int *ldb, + float *x, + lapack_int *ldx, + float *rcond, + float *berr, + lapack_int *n_err_bnds, + float *err_bnds_norm, + float *err_bnds_comp, + lapack_int *nparams, + float *params, + float *work, + lapack_int *iwork, + lapack_int *info); +void LAPACK_zgerfsx(char *trans, + char *equed, + lapack_int *n, + lapack_int *nrhs, + const lapack_complex_double *a, + lapack_int *lda, + const lapack_complex_double *af, + lapack_int *ldaf, + const lapack_int *ipiv, + const double *r, + const double *c, + const lapack_complex_double *b, + lapack_int *ldb, + lapack_complex_double *x, + lapack_int *ldx, + double *rcond, + double *berr, + lapack_int *n_err_bnds, + double *err_bnds_norm, + double *err_bnds_comp, + lapack_int *nparams, + double *params, + lapack_complex_double *work, + double *rwork, + lapack_int *info); +void LAPACK_cgerfsx(char *trans, + char *equed, + lapack_int *n, + lapack_int *nrhs, + const lapack_complex_float *a, + lapack_int *lda, + const lapack_complex_float *af, + lapack_int *ldaf, + const lapack_int *ipiv, + const float *r, + const float *c, + const lapack_complex_float *b, + lapack_int *ldb, + lapack_complex_float *x, + lapack_int *ldx, + float *rcond, + float *berr, + lapack_int *n_err_bnds, + float *err_bnds_norm, + float *err_bnds_comp, + lapack_int *nparams, + float *params, + lapack_complex_float *work, + float *rwork, + lapack_int *info); +void LAPACK_sgbrfs(char *trans, + lapack_int *n, + lapack_int *kl, + lapack_int *ku, + lapack_int *nrhs, + const float *ab, + lapack_int *ldab, + const float *afb, + lapack_int *ldafb, + const lapack_int *ipiv, + const float *b, + lapack_int *ldb, + float *x, + lapack_int *ldx, + float *ferr, + float *berr, + float *work, + lapack_int *iwork, + lapack_int *info); +void LAPACK_dgbrfs(char *trans, + lapack_int *n, + lapack_int *kl, + lapack_int *ku, + lapack_int *nrhs, + const double *ab, + lapack_int *ldab, + const double *afb, + lapack_int *ldafb, + const lapack_int *ipiv, + const double *b, + lapack_int *ldb, + double *x, + lapack_int *ldx, + double *ferr, + double *berr, + double *work, + lapack_int *iwork, + lapack_int *info); +void LAPACK_cgbrfs(char *trans, + lapack_int *n, + lapack_int *kl, + lapack_int *ku, + lapack_int *nrhs, + const lapack_complex_float *ab, + lapack_int *ldab, + const lapack_complex_float *afb, + lapack_int *ldafb, + const lapack_int *ipiv, + const lapack_complex_float *b, + lapack_int *ldb, + lapack_complex_float *x, + lapack_int *ldx, + float *ferr, + float *berr, + lapack_complex_float *work, + float *rwork, + lapack_int *info); +void LAPACK_zgbrfs(char *trans, + lapack_int *n, + lapack_int *kl, + lapack_int *ku, + lapack_int *nrhs, + const lapack_complex_double *ab, + lapack_int *ldab, + const lapack_complex_double *afb, + lapack_int *ldafb, + const lapack_int *ipiv, + const lapack_complex_double *b, + lapack_int *ldb, + lapack_complex_double *x, + lapack_int *ldx, + double *ferr, + double *berr, + lapack_complex_double *work, + double *rwork, + lapack_int *info); +void LAPACK_dgbrfsx(char *trans, + char *equed, + lapack_int *n, + lapack_int *kl, + lapack_int *ku, + lapack_int *nrhs, + const double *ab, + lapack_int *ldab, + const double *afb, + lapack_int *ldafb, + const lapack_int *ipiv, + const double *r, + const double *c, + const double *b, + lapack_int *ldb, + double *x, + lapack_int *ldx, + double *rcond, + double *berr, + lapack_int *n_err_bnds, + double *err_bnds_norm, + double *err_bnds_comp, + lapack_int *nparams, + double *params, + double *work, + lapack_int *iwork, + lapack_int *info); +void LAPACK_sgbrfsx(char *trans, + char *equed, + lapack_int *n, + lapack_int *kl, + lapack_int *ku, + lapack_int *nrhs, + const float *ab, + lapack_int *ldab, + const float *afb, + lapack_int *ldafb, + const lapack_int *ipiv, + const float *r, + const float *c, + const float *b, + lapack_int *ldb, + float *x, + lapack_int *ldx, + float *rcond, + float *berr, + lapack_int *n_err_bnds, + float *err_bnds_norm, + float *err_bnds_comp, + lapack_int *nparams, + float *params, + float *work, + lapack_int *iwork, + lapack_int *info); +void LAPACK_zgbrfsx(char *trans, + char *equed, + lapack_int *n, + lapack_int *kl, + lapack_int *ku, + lapack_int *nrhs, + const lapack_complex_double *ab, + lapack_int *ldab, + const lapack_complex_double *afb, + lapack_int *ldafb, + const lapack_int *ipiv, + const double *r, + const double *c, + const lapack_complex_double *b, + lapack_int *ldb, + lapack_complex_double *x, + lapack_int *ldx, + double *rcond, + double *berr, + lapack_int *n_err_bnds, + double *err_bnds_norm, + double *err_bnds_comp, + lapack_int *nparams, + double *params, + lapack_complex_double *work, + double *rwork, + lapack_int *info); +void LAPACK_cgbrfsx(char *trans, + char *equed, + lapack_int *n, + lapack_int *kl, + lapack_int *ku, + lapack_int *nrhs, + const lapack_complex_float *ab, + lapack_int *ldab, + const lapack_complex_float *afb, + lapack_int *ldafb, + const lapack_int *ipiv, + const float *r, + const float *c, + const lapack_complex_float *b, + lapack_int *ldb, + lapack_complex_float *x, + lapack_int *ldx, + float *rcond, + float *berr, + lapack_int *n_err_bnds, + float *err_bnds_norm, + float *err_bnds_comp, + lapack_int *nparams, + float *params, + lapack_complex_float *work, + float *rwork, + lapack_int *info); +void LAPACK_sgtrfs(char *trans, + lapack_int *n, + lapack_int *nrhs, + const float *dl, + const float *d, + const float *du, + const float *dlf, + const float *df, + const float *duf, + const float *du2, + const lapack_int *ipiv, + const float *b, + lapack_int *ldb, + float *x, + lapack_int *ldx, + float *ferr, + float *berr, + float *work, + lapack_int *iwork, + lapack_int *info); +void LAPACK_dgtrfs(char *trans, + lapack_int *n, + lapack_int *nrhs, + const double *dl, + const double *d, + const double *du, + const double *dlf, + const double *df, + const double *duf, + const double *du2, + const lapack_int *ipiv, + const double *b, + lapack_int *ldb, + double *x, + lapack_int *ldx, + double *ferr, + double *berr, + double *work, + lapack_int *iwork, + lapack_int *info); +void LAPACK_cgtrfs(char *trans, + lapack_int *n, + lapack_int *nrhs, + const lapack_complex_float *dl, + const lapack_complex_float *d, + const lapack_complex_float *du, + const lapack_complex_float *dlf, + const lapack_complex_float *df, + const lapack_complex_float *duf, + const lapack_complex_float *du2, + const lapack_int *ipiv, + const lapack_complex_float *b, + lapack_int *ldb, + lapack_complex_float *x, + lapack_int *ldx, + float *ferr, + float *berr, + lapack_complex_float *work, + float *rwork, + lapack_int *info); +void LAPACK_zgtrfs(char *trans, + lapack_int *n, + lapack_int *nrhs, + const lapack_complex_double *dl, + const lapack_complex_double *d, + const lapack_complex_double *du, + const lapack_complex_double *dlf, + const lapack_complex_double *df, + const lapack_complex_double *duf, + const lapack_complex_double *du2, + const lapack_int *ipiv, + const lapack_complex_double *b, + lapack_int *ldb, + lapack_complex_double *x, + lapack_int *ldx, + double *ferr, + double *berr, + lapack_complex_double *work, + double *rwork, + lapack_int *info); +void LAPACK_sporfs(char *uplo, + lapack_int *n, + lapack_int *nrhs, + const float *a, + lapack_int *lda, + const float *af, + lapack_int *ldaf, + const float *b, + lapack_int *ldb, + float *x, + lapack_int *ldx, + float *ferr, + float *berr, + float *work, + lapack_int *iwork, + lapack_int *info); +void LAPACK_dporfs(char *uplo, + lapack_int *n, + lapack_int *nrhs, + const double *a, + lapack_int *lda, + const double *af, + lapack_int *ldaf, + const double *b, + lapack_int *ldb, + double *x, + lapack_int *ldx, + double *ferr, + double *berr, + double *work, + lapack_int *iwork, + lapack_int *info); +void LAPACK_cporfs(char *uplo, + lapack_int *n, + lapack_int *nrhs, + const lapack_complex_float *a, + lapack_int *lda, + const lapack_complex_float *af, + lapack_int *ldaf, + const lapack_complex_float *b, + lapack_int *ldb, + lapack_complex_float *x, + lapack_int *ldx, + float *ferr, + float *berr, + lapack_complex_float *work, + float *rwork, + lapack_int *info); +void LAPACK_zporfs(char *uplo, + lapack_int *n, + lapack_int *nrhs, + const lapack_complex_double *a, + lapack_int *lda, + const lapack_complex_double *af, + lapack_int *ldaf, + const lapack_complex_double *b, + lapack_int *ldb, + lapack_complex_double *x, + lapack_int *ldx, + double *ferr, + double *berr, + lapack_complex_double *work, + double *rwork, + lapack_int *info); +void LAPACK_dporfsx(char *uplo, + char *equed, + lapack_int *n, + lapack_int *nrhs, + const double *a, + lapack_int *lda, + const double *af, + lapack_int *ldaf, + const double *s, + const double *b, + lapack_int *ldb, + double *x, + lapack_int *ldx, + double *rcond, + double *berr, + lapack_int *n_err_bnds, + double *err_bnds_norm, + double *err_bnds_comp, + lapack_int *nparams, + double *params, + double *work, + lapack_int *iwork, + lapack_int *info); +void LAPACK_sporfsx(char *uplo, + char *equed, + lapack_int *n, + lapack_int *nrhs, + const float *a, + lapack_int *lda, + const float *af, + lapack_int *ldaf, + const float *s, + const float *b, + lapack_int *ldb, + float *x, + lapack_int *ldx, + float *rcond, + float *berr, + lapack_int *n_err_bnds, + float *err_bnds_norm, + float *err_bnds_comp, + lapack_int *nparams, + float *params, + float *work, + lapack_int *iwork, + lapack_int *info); +void LAPACK_zporfsx(char *uplo, + char *equed, + lapack_int *n, + lapack_int *nrhs, + const lapack_complex_double *a, + lapack_int *lda, + const lapack_complex_double *af, + lapack_int *ldaf, + const double *s, + const lapack_complex_double *b, + lapack_int *ldb, + lapack_complex_double *x, + lapack_int *ldx, + double *rcond, + double *berr, + lapack_int *n_err_bnds, + double *err_bnds_norm, + double *err_bnds_comp, + lapack_int *nparams, + double *params, + lapack_complex_double *work, + double *rwork, + lapack_int *info); +void LAPACK_cporfsx(char *uplo, + char *equed, + lapack_int *n, + lapack_int *nrhs, + const lapack_complex_float *a, + lapack_int *lda, + const lapack_complex_float *af, + lapack_int *ldaf, + const float *s, + const lapack_complex_float *b, + lapack_int *ldb, + lapack_complex_float *x, + lapack_int *ldx, + float *rcond, + float *berr, + lapack_int *n_err_bnds, + float *err_bnds_norm, + float *err_bnds_comp, + lapack_int *nparams, + float *params, + lapack_complex_float *work, + float *rwork, + lapack_int *info); +void LAPACK_spprfs(char *uplo, + lapack_int *n, + lapack_int *nrhs, + const float *ap, + const float *afp, + const float *b, + lapack_int *ldb, + float *x, + lapack_int *ldx, + float *ferr, + float *berr, + float *work, + lapack_int *iwork, + lapack_int *info); +void LAPACK_dpprfs(char *uplo, + lapack_int *n, + lapack_int *nrhs, + const double *ap, + const double *afp, + const double *b, + lapack_int *ldb, + double *x, + lapack_int *ldx, + double *ferr, + double *berr, + double *work, + lapack_int *iwork, + lapack_int *info); +void LAPACK_cpprfs(char *uplo, + lapack_int *n, + lapack_int *nrhs, + const lapack_complex_float *ap, + const lapack_complex_float *afp, + const lapack_complex_float *b, + lapack_int *ldb, + lapack_complex_float *x, + lapack_int *ldx, + float *ferr, + float *berr, + lapack_complex_float *work, + float *rwork, + lapack_int *info); +void LAPACK_zpprfs(char *uplo, + lapack_int *n, + lapack_int *nrhs, + const lapack_complex_double *ap, + const lapack_complex_double *afp, + const lapack_complex_double *b, + lapack_int *ldb, + lapack_complex_double *x, + lapack_int *ldx, + double *ferr, + double *berr, + lapack_complex_double *work, + double *rwork, + lapack_int *info); +void LAPACK_spbrfs(char *uplo, + lapack_int *n, + lapack_int *kd, + lapack_int *nrhs, + const float *ab, + lapack_int *ldab, + const float *afb, + lapack_int *ldafb, + const float *b, + lapack_int *ldb, + float *x, + lapack_int *ldx, + float *ferr, + float *berr, + float *work, + lapack_int *iwork, + lapack_int *info); +void LAPACK_dpbrfs(char *uplo, + lapack_int *n, + lapack_int *kd, + lapack_int *nrhs, + const double *ab, + lapack_int *ldab, + const double *afb, + lapack_int *ldafb, + const double *b, + lapack_int *ldb, + double *x, + lapack_int *ldx, + double *ferr, + double *berr, + double *work, + lapack_int *iwork, + lapack_int *info); +void LAPACK_cpbrfs(char *uplo, + lapack_int *n, + lapack_int *kd, + lapack_int *nrhs, + const lapack_complex_float *ab, + lapack_int *ldab, + const lapack_complex_float *afb, + lapack_int *ldafb, + const lapack_complex_float *b, + lapack_int *ldb, + lapack_complex_float *x, + lapack_int *ldx, + float *ferr, + float *berr, + lapack_complex_float *work, + float *rwork, + lapack_int *info); +void LAPACK_zpbrfs(char *uplo, + lapack_int *n, + lapack_int *kd, + lapack_int *nrhs, + const lapack_complex_double *ab, + lapack_int *ldab, + const lapack_complex_double *afb, + lapack_int *ldafb, + const lapack_complex_double *b, + lapack_int *ldb, + lapack_complex_double *x, + lapack_int *ldx, + double *ferr, + double *berr, + lapack_complex_double *work, + double *rwork, + lapack_int *info); +void LAPACK_sptrfs(lapack_int *n, + lapack_int *nrhs, + const float *d, + const float *e, + const float *df, + const float *ef, + const float *b, + lapack_int *ldb, + float *x, + lapack_int *ldx, + float *ferr, + float *berr, + float *work, + lapack_int *info); +void LAPACK_dptrfs(lapack_int *n, + lapack_int *nrhs, + const double *d, + const double *e, + const double *df, + const double *ef, + const double *b, + lapack_int *ldb, + double *x, + lapack_int *ldx, + double *ferr, + double *berr, + double *work, + lapack_int *info); +void LAPACK_cptrfs(char *uplo, + lapack_int *n, + lapack_int *nrhs, + const float *d, + const lapack_complex_float *e, + const float *df, + const lapack_complex_float *ef, + const lapack_complex_float *b, + lapack_int *ldb, + lapack_complex_float *x, + lapack_int *ldx, + float *ferr, + float *berr, + lapack_complex_float *work, + float *rwork, + lapack_int *info); +void LAPACK_zptrfs(char *uplo, + lapack_int *n, + lapack_int *nrhs, + const double *d, + const lapack_complex_double *e, + const double *df, + const lapack_complex_double *ef, + const lapack_complex_double *b, + lapack_int *ldb, + lapack_complex_double *x, + lapack_int *ldx, + double *ferr, + double *berr, + lapack_complex_double *work, + double *rwork, + lapack_int *info); +void LAPACK_ssyrfs(char *uplo, + lapack_int *n, + lapack_int *nrhs, + const float *a, + lapack_int *lda, + const float *af, + lapack_int *ldaf, + const lapack_int *ipiv, + const float *b, + lapack_int *ldb, + float *x, + lapack_int *ldx, + float *ferr, + float *berr, + float *work, + lapack_int *iwork, + lapack_int *info); +void LAPACK_dsyrfs(char *uplo, + lapack_int *n, + lapack_int *nrhs, + const double *a, + lapack_int *lda, + const double *af, + lapack_int *ldaf, + const lapack_int *ipiv, + const double *b, + lapack_int *ldb, + double *x, + lapack_int *ldx, + double *ferr, + double *berr, + double *work, + lapack_int *iwork, + lapack_int *info); +void LAPACK_csyrfs(char *uplo, + lapack_int *n, + lapack_int *nrhs, + const lapack_complex_float *a, + lapack_int *lda, + const lapack_complex_float *af, + lapack_int *ldaf, + const lapack_int *ipiv, + const lapack_complex_float *b, + lapack_int *ldb, + lapack_complex_float *x, + lapack_int *ldx, + float *ferr, + float *berr, + lapack_complex_float *work, + float *rwork, + lapack_int *info); +void LAPACK_zsyrfs(char *uplo, + lapack_int *n, + lapack_int *nrhs, + const lapack_complex_double *a, + lapack_int *lda, + const lapack_complex_double *af, + lapack_int *ldaf, + const lapack_int *ipiv, + const lapack_complex_double *b, + lapack_int *ldb, + lapack_complex_double *x, + lapack_int *ldx, + double *ferr, + double *berr, + lapack_complex_double *work, + double *rwork, + lapack_int *info); +void LAPACK_dsyrfsx(char *uplo, + char *equed, + lapack_int *n, + lapack_int *nrhs, + const double *a, + lapack_int *lda, + const double *af, + lapack_int *ldaf, + const lapack_int *ipiv, + const double *s, + const double *b, + lapack_int *ldb, + double *x, + lapack_int *ldx, + double *rcond, + double *berr, + lapack_int *n_err_bnds, + double *err_bnds_norm, + double *err_bnds_comp, + lapack_int *nparams, + double *params, + double *work, + lapack_int *iwork, + lapack_int *info); +void LAPACK_ssyrfsx(char *uplo, + char *equed, + lapack_int *n, + lapack_int *nrhs, + const float *a, + lapack_int *lda, + const float *af, + lapack_int *ldaf, + const lapack_int *ipiv, + const float *s, + const float *b, + lapack_int *ldb, + float *x, + lapack_int *ldx, + float *rcond, + float *berr, + lapack_int *n_err_bnds, + float *err_bnds_norm, + float *err_bnds_comp, + lapack_int *nparams, + float *params, + float *work, + lapack_int *iwork, + lapack_int *info); +void LAPACK_zsyrfsx(char *uplo, + char *equed, + lapack_int *n, + lapack_int *nrhs, + const lapack_complex_double *a, + lapack_int *lda, + const lapack_complex_double *af, + lapack_int *ldaf, + const lapack_int *ipiv, + const double *s, + const lapack_complex_double *b, + lapack_int *ldb, + lapack_complex_double *x, + lapack_int *ldx, + double *rcond, + double *berr, + lapack_int *n_err_bnds, + double *err_bnds_norm, + double *err_bnds_comp, + lapack_int *nparams, + double *params, + lapack_complex_double *work, + double *rwork, + lapack_int *info); +void LAPACK_csyrfsx(char *uplo, + char *equed, + lapack_int *n, + lapack_int *nrhs, + const lapack_complex_float *a, + lapack_int *lda, + const lapack_complex_float *af, + lapack_int *ldaf, + const lapack_int *ipiv, + const float *s, + const lapack_complex_float *b, + lapack_int *ldb, + lapack_complex_float *x, + lapack_int *ldx, + float *rcond, + float *berr, + lapack_int *n_err_bnds, + float *err_bnds_norm, + float *err_bnds_comp, + lapack_int *nparams, + float *params, + lapack_complex_float *work, + float *rwork, + lapack_int *info); +void LAPACK_cherfs(char *uplo, + lapack_int *n, + lapack_int *nrhs, + const lapack_complex_float *a, + lapack_int *lda, + const lapack_complex_float *af, + lapack_int *ldaf, + const lapack_int *ipiv, + const lapack_complex_float *b, + lapack_int *ldb, + lapack_complex_float *x, + lapack_int *ldx, + float *ferr, + float *berr, + lapack_complex_float *work, + float *rwork, + lapack_int *info); +void LAPACK_zherfs(char *uplo, + lapack_int *n, + lapack_int *nrhs, + const lapack_complex_double *a, + lapack_int *lda, + const lapack_complex_double *af, + lapack_int *ldaf, + const lapack_int *ipiv, + const lapack_complex_double *b, + lapack_int *ldb, + lapack_complex_double *x, + lapack_int *ldx, + double *ferr, + double *berr, + lapack_complex_double *work, + double *rwork, + lapack_int *info); +void LAPACK_zherfsx(char *uplo, + char *equed, + lapack_int *n, + lapack_int *nrhs, + const lapack_complex_double *a, + lapack_int *lda, + const lapack_complex_double *af, + lapack_int *ldaf, + const lapack_int *ipiv, + const double *s, + const lapack_complex_double *b, + lapack_int *ldb, + lapack_complex_double *x, + lapack_int *ldx, + double *rcond, + double *berr, + lapack_int *n_err_bnds, + double *err_bnds_norm, + double *err_bnds_comp, + lapack_int *nparams, + double *params, + lapack_complex_double *work, + double *rwork, + lapack_int *info); +void LAPACK_cherfsx(char *uplo, + char *equed, + lapack_int *n, + lapack_int *nrhs, + const lapack_complex_float *a, + lapack_int *lda, + const lapack_complex_float *af, + lapack_int *ldaf, + const lapack_int *ipiv, + const float *s, + const lapack_complex_float *b, + lapack_int *ldb, + lapack_complex_float *x, + lapack_int *ldx, + float *rcond, + float *berr, + lapack_int *n_err_bnds, + float *err_bnds_norm, + float *err_bnds_comp, + lapack_int *nparams, + float *params, + lapack_complex_float *work, + float *rwork, + lapack_int *info); +void LAPACK_ssprfs(char *uplo, + lapack_int *n, + lapack_int *nrhs, + const float *ap, + const float *afp, + const lapack_int *ipiv, + const float *b, + lapack_int *ldb, + float *x, + lapack_int *ldx, + float *ferr, + float *berr, + float *work, + lapack_int *iwork, + lapack_int *info); +void LAPACK_dsprfs(char *uplo, + lapack_int *n, + lapack_int *nrhs, + const double *ap, + const double *afp, + const lapack_int *ipiv, + const double *b, + lapack_int *ldb, + double *x, + lapack_int *ldx, + double *ferr, + double *berr, + double *work, + lapack_int *iwork, + lapack_int *info); +void LAPACK_csprfs(char *uplo, + lapack_int *n, + lapack_int *nrhs, + const lapack_complex_float *ap, + const lapack_complex_float *afp, + const lapack_int *ipiv, + const lapack_complex_float *b, + lapack_int *ldb, + lapack_complex_float *x, + lapack_int *ldx, + float *ferr, + float *berr, + lapack_complex_float *work, + float *rwork, + lapack_int *info); +void LAPACK_zsprfs(char *uplo, + lapack_int *n, + lapack_int *nrhs, + const lapack_complex_double *ap, + const lapack_complex_double *afp, + const lapack_int *ipiv, + const lapack_complex_double *b, + lapack_int *ldb, + lapack_complex_double *x, + lapack_int *ldx, + double *ferr, + double *berr, + lapack_complex_double *work, + double *rwork, + lapack_int *info); +void LAPACK_chprfs(char *uplo, + lapack_int *n, + lapack_int *nrhs, + const lapack_complex_float *ap, + const lapack_complex_float *afp, + const lapack_int *ipiv, + const lapack_complex_float *b, + lapack_int *ldb, + lapack_complex_float *x, + lapack_int *ldx, + float *ferr, + float *berr, + lapack_complex_float *work, + float *rwork, + lapack_int *info); +void LAPACK_zhprfs(char *uplo, + lapack_int *n, + lapack_int *nrhs, + const lapack_complex_double *ap, + const lapack_complex_double *afp, + const lapack_int *ipiv, + const lapack_complex_double *b, + lapack_int *ldb, + lapack_complex_double *x, + lapack_int *ldx, + double *ferr, + double *berr, + lapack_complex_double *work, + double *rwork, + lapack_int *info); +void LAPACK_strrfs(char *uplo, + char *trans, + char *diag, + lapack_int *n, + lapack_int *nrhs, + const float *a, + lapack_int *lda, + const float *b, + lapack_int *ldb, + const float *x, + lapack_int *ldx, + float *ferr, + float *berr, + float *work, + lapack_int *iwork, + lapack_int *info); +void LAPACK_dtrrfs(char *uplo, + char *trans, + char *diag, + lapack_int *n, + lapack_int *nrhs, + const double *a, + lapack_int *lda, + const double *b, + lapack_int *ldb, + const double *x, + lapack_int *ldx, + double *ferr, + double *berr, + double *work, + lapack_int *iwork, + lapack_int *info); +void LAPACK_ctrrfs(char *uplo, + char *trans, + char *diag, + lapack_int *n, + lapack_int *nrhs, + const lapack_complex_float *a, + lapack_int *lda, + const lapack_complex_float *b, + lapack_int *ldb, + const lapack_complex_float *x, + lapack_int *ldx, + float *ferr, + float *berr, + lapack_complex_float *work, + float *rwork, + lapack_int *info); +void LAPACK_ztrrfs(char *uplo, + char *trans, + char *diag, + lapack_int *n, + lapack_int *nrhs, + const lapack_complex_double *a, + lapack_int *lda, + const lapack_complex_double *b, + lapack_int *ldb, + const lapack_complex_double *x, + lapack_int *ldx, + double *ferr, + double *berr, + lapack_complex_double *work, + double *rwork, + lapack_int *info); +void LAPACK_stprfs(char *uplo, + char *trans, + char *diag, + lapack_int *n, + lapack_int *nrhs, + const float *ap, + const float *b, + lapack_int *ldb, + const float *x, + lapack_int *ldx, + float *ferr, + float *berr, + float *work, + lapack_int *iwork, + lapack_int *info); +void LAPACK_dtprfs(char *uplo, + char *trans, + char *diag, + lapack_int *n, + lapack_int *nrhs, + const double *ap, + const double *b, + lapack_int *ldb, + const double *x, + lapack_int *ldx, + double *ferr, + double *berr, + double *work, + lapack_int *iwork, + lapack_int *info); +void LAPACK_ctprfs(char *uplo, + char *trans, + char *diag, + lapack_int *n, + lapack_int *nrhs, + const lapack_complex_float *ap, + const lapack_complex_float *b, + lapack_int *ldb, + const lapack_complex_float *x, + lapack_int *ldx, + float *ferr, + float *berr, + lapack_complex_float *work, + float *rwork, + lapack_int *info); +void LAPACK_ztprfs(char *uplo, + char *trans, + char *diag, + lapack_int *n, + lapack_int *nrhs, + const lapack_complex_double *ap, + const lapack_complex_double *b, + lapack_int *ldb, + const lapack_complex_double *x, + lapack_int *ldx, + double *ferr, + double *berr, + lapack_complex_double *work, + double *rwork, + lapack_int *info); +void LAPACK_stbrfs(char *uplo, + char *trans, + char *diag, + lapack_int *n, + lapack_int *kd, + lapack_int *nrhs, + const float *ab, + lapack_int *ldab, + const float *b, + lapack_int *ldb, + const float *x, + lapack_int *ldx, + float *ferr, + float *berr, + float *work, + lapack_int *iwork, + lapack_int *info); +void LAPACK_dtbrfs(char *uplo, + char *trans, + char *diag, + lapack_int *n, + lapack_int *kd, + lapack_int *nrhs, + const double *ab, + lapack_int *ldab, + const double *b, + lapack_int *ldb, + const double *x, + lapack_int *ldx, + double *ferr, + double *berr, + double *work, + lapack_int *iwork, + lapack_int *info); +void LAPACK_ctbrfs(char *uplo, + char *trans, + char *diag, + lapack_int *n, + lapack_int *kd, + lapack_int *nrhs, + const lapack_complex_float *ab, + lapack_int *ldab, + const lapack_complex_float *b, + lapack_int *ldb, + const lapack_complex_float *x, + lapack_int *ldx, + float *ferr, + float *berr, + lapack_complex_float *work, + float *rwork, + lapack_int *info); +void LAPACK_ztbrfs(char *uplo, + char *trans, + char *diag, + lapack_int *n, + lapack_int *kd, + lapack_int *nrhs, + const lapack_complex_double *ab, + lapack_int *ldab, + const lapack_complex_double *b, + lapack_int *ldb, + const lapack_complex_double *x, + lapack_int *ldx, + double *ferr, + double *berr, + lapack_complex_double *work, + double *rwork, + lapack_int *info); +void LAPACK_sgetri(lapack_int *n, + float *a, + lapack_int *lda, + const lapack_int *ipiv, + float *work, + lapack_int *lwork, + lapack_int *info); +void LAPACK_dgetri(lapack_int *n, + double *a, + lapack_int *lda, + const lapack_int *ipiv, + double *work, + lapack_int *lwork, + lapack_int *info); +void LAPACK_cgetri(lapack_int *n, + lapack_complex_float *a, + lapack_int *lda, + const lapack_int *ipiv, + lapack_complex_float *work, + lapack_int *lwork, + lapack_int *info); +void LAPACK_zgetri(lapack_int *n, + lapack_complex_double *a, + lapack_int *lda, + const lapack_int *ipiv, + lapack_complex_double *work, + lapack_int *lwork, + lapack_int *info); +void LAPACK_spotri(char *uplo, lapack_int *n, float *a, lapack_int *lda, lapack_int *info); +void LAPACK_dpotri(char *uplo, lapack_int *n, double *a, lapack_int *lda, lapack_int *info); +void LAPACK_cpotri(char *uplo, lapack_int *n, lapack_complex_float *a, lapack_int *lda, lapack_int *info); +void LAPACK_zpotri(char *uplo, lapack_int *n, lapack_complex_double *a, lapack_int *lda, lapack_int *info); +void LAPACK_dpftri(char *transr, char *uplo, lapack_int *n, double *a, lapack_int *info); +void LAPACK_spftri(char *transr, char *uplo, lapack_int *n, float *a, lapack_int *info); +void LAPACK_zpftri(char *transr, char *uplo, lapack_int *n, lapack_complex_double *a, lapack_int *info); +void LAPACK_cpftri(char *transr, char *uplo, lapack_int *n, lapack_complex_float *a, lapack_int *info); +void LAPACK_spptri(char *uplo, lapack_int *n, float *ap, lapack_int *info); +void LAPACK_dpptri(char *uplo, lapack_int *n, double *ap, lapack_int *info); +void LAPACK_cpptri(char *uplo, lapack_int *n, lapack_complex_float *ap, lapack_int *info); +void LAPACK_zpptri(char *uplo, lapack_int *n, lapack_complex_double *ap, lapack_int *info); +void LAPACK_ssytri(char *uplo, + lapack_int *n, + float *a, + lapack_int *lda, + const lapack_int *ipiv, + float *work, + lapack_int *info); +void LAPACK_dsytri(char *uplo, + lapack_int *n, + double *a, + lapack_int *lda, + const lapack_int *ipiv, + double *work, + lapack_int *info); +void LAPACK_csytri(char *uplo, + lapack_int *n, + lapack_complex_float *a, + lapack_int *lda, + const lapack_int *ipiv, + lapack_complex_float *work, + lapack_int *info); +void LAPACK_zsytri(char *uplo, + lapack_int *n, + lapack_complex_double *a, + lapack_int *lda, + const lapack_int *ipiv, + lapack_complex_double *work, + lapack_int *info); +void LAPACK_chetri(char *uplo, + lapack_int *n, + lapack_complex_float *a, + lapack_int *lda, + const lapack_int *ipiv, + lapack_complex_float *work, + lapack_int *info); +void LAPACK_zhetri(char *uplo, + lapack_int *n, + lapack_complex_double *a, + lapack_int *lda, + const lapack_int *ipiv, + lapack_complex_double *work, + lapack_int *info); +void LAPACK_ssptri(char *uplo, lapack_int *n, float *ap, const lapack_int *ipiv, float *work, lapack_int *info); +void LAPACK_dsptri(char *uplo, lapack_int *n, double *ap, const lapack_int *ipiv, double *work, lapack_int *info); +void LAPACK_csptri(char *uplo, + lapack_int *n, + lapack_complex_float *ap, + const lapack_int *ipiv, + lapack_complex_float *work, + lapack_int *info); +void LAPACK_zsptri(char *uplo, + lapack_int *n, + lapack_complex_double *ap, + const lapack_int *ipiv, + lapack_complex_double *work, + lapack_int *info); +void LAPACK_chptri(char *uplo, + lapack_int *n, + lapack_complex_float *ap, + const lapack_int *ipiv, + lapack_complex_float *work, + lapack_int *info); +void LAPACK_zhptri(char *uplo, + lapack_int *n, + lapack_complex_double *ap, + const lapack_int *ipiv, + lapack_complex_double *work, + lapack_int *info); +void LAPACK_strtri(char *uplo, char *diag, lapack_int *n, float *a, lapack_int *lda, lapack_int *info); +void LAPACK_dtrtri(char *uplo, char *diag, lapack_int *n, double *a, lapack_int *lda, lapack_int *info); +void LAPACK_ctrtri(char *uplo, char *diag, lapack_int *n, lapack_complex_float *a, lapack_int *lda, lapack_int *info); +void LAPACK_ztrtri(char *uplo, char *diag, lapack_int *n, lapack_complex_double *a, lapack_int *lda, lapack_int *info); +void LAPACK_dtftri(char *transr, char *uplo, char *diag, lapack_int *n, double *a, lapack_int *info); +void LAPACK_stftri(char *transr, char *uplo, char *diag, lapack_int *n, float *a, lapack_int *info); +void LAPACK_ztftri(char *transr, char *uplo, char *diag, lapack_int *n, lapack_complex_double *a, lapack_int *info); +void LAPACK_ctftri(char *transr, char *uplo, char *diag, lapack_int *n, lapack_complex_float *a, lapack_int *info); +void LAPACK_stptri(char *uplo, char *diag, lapack_int *n, float *ap, lapack_int *info); +void LAPACK_dtptri(char *uplo, char *diag, lapack_int *n, double *ap, lapack_int *info); +void LAPACK_ctptri(char *uplo, char *diag, lapack_int *n, lapack_complex_float *ap, lapack_int *info); +void LAPACK_ztptri(char *uplo, char *diag, lapack_int *n, lapack_complex_double *ap, lapack_int *info); +void LAPACK_sgeequ(lapack_int *m, + lapack_int *n, + const float *a, + lapack_int *lda, + float *r, + float *c, + float *rowcnd, + float *colcnd, + float *amax, + lapack_int *info); +void LAPACK_dgeequ(lapack_int *m, + lapack_int *n, + const double *a, + lapack_int *lda, + double *r, + double *c, + double *rowcnd, + double *colcnd, + double *amax, + lapack_int *info); +void LAPACK_cgeequ(lapack_int *m, + lapack_int *n, + const lapack_complex_float *a, + lapack_int *lda, + float *r, + float *c, + float *rowcnd, + float *colcnd, + float *amax, + lapack_int *info); +void LAPACK_zgeequ(lapack_int *m, + lapack_int *n, + const lapack_complex_double *a, + lapack_int *lda, + double *r, + double *c, + double *rowcnd, + double *colcnd, + double *amax, + lapack_int *info); +void LAPACK_dgeequb(lapack_int *m, + lapack_int *n, + const double *a, + lapack_int *lda, + double *r, + double *c, + double *rowcnd, + double *colcnd, + double *amax, + lapack_int *info); +void LAPACK_sgeequb(lapack_int *m, + lapack_int *n, + const float *a, + lapack_int *lda, + float *r, + float *c, + float *rowcnd, + float *colcnd, + float *amax, + lapack_int *info); +void LAPACK_zgeequb(lapack_int *m, + lapack_int *n, + const lapack_complex_double *a, + lapack_int *lda, + double *r, + double *c, + double *rowcnd, + double *colcnd, + double *amax, + lapack_int *info); +void LAPACK_cgeequb(lapack_int *m, + lapack_int *n, + const lapack_complex_float *a, + lapack_int *lda, + float *r, + float *c, + float *rowcnd, + float *colcnd, + float *amax, + lapack_int *info); +void LAPACK_sgbequ(lapack_int *m, + lapack_int *n, + lapack_int *kl, + lapack_int *ku, + const float *ab, + lapack_int *ldab, + float *r, + float *c, + float *rowcnd, + float *colcnd, + float *amax, + lapack_int *info); +void LAPACK_dgbequ(lapack_int *m, + lapack_int *n, + lapack_int *kl, + lapack_int *ku, + const double *ab, + lapack_int *ldab, + double *r, + double *c, + double *rowcnd, + double *colcnd, + double *amax, + lapack_int *info); +void LAPACK_cgbequ(lapack_int *m, + lapack_int *n, + lapack_int *kl, + lapack_int *ku, + const lapack_complex_float *ab, + lapack_int *ldab, + float *r, + float *c, + float *rowcnd, + float *colcnd, + float *amax, + lapack_int *info); +void LAPACK_zgbequ(lapack_int *m, + lapack_int *n, + lapack_int *kl, + lapack_int *ku, + const lapack_complex_double *ab, + lapack_int *ldab, + double *r, + double *c, + double *rowcnd, + double *colcnd, + double *amax, + lapack_int *info); +void LAPACK_dgbequb(lapack_int *m, + lapack_int *n, + lapack_int *kl, + lapack_int *ku, + const double *ab, + lapack_int *ldab, + double *r, + double *c, + double *rowcnd, + double *colcnd, + double *amax, + lapack_int *info); +void LAPACK_sgbequb(lapack_int *m, + lapack_int *n, + lapack_int *kl, + lapack_int *ku, + const float *ab, + lapack_int *ldab, + float *r, + float *c, + float *rowcnd, + float *colcnd, + float *amax, + lapack_int *info); +void LAPACK_zgbequb(lapack_int *m, + lapack_int *n, + lapack_int *kl, + lapack_int *ku, + const lapack_complex_double *ab, + lapack_int *ldab, + double *r, + double *c, + double *rowcnd, + double *colcnd, + double *amax, + lapack_int *info); +void LAPACK_cgbequb(lapack_int *m, + lapack_int *n, + lapack_int *kl, + lapack_int *ku, + const lapack_complex_float *ab, + lapack_int *ldab, + float *r, + float *c, + float *rowcnd, + float *colcnd, + float *amax, + lapack_int *info); +void LAPACK_spoequ(lapack_int *n, + const float *a, + lapack_int *lda, + float *s, + float *scond, + float *amax, + lapack_int *info); +void LAPACK_dpoequ(lapack_int *n, + const double *a, + lapack_int *lda, + double *s, + double *scond, + double *amax, + lapack_int *info); +void LAPACK_cpoequ(lapack_int *n, + const lapack_complex_float *a, + lapack_int *lda, + float *s, + float *scond, + float *amax, + lapack_int *info); +void LAPACK_zpoequ(lapack_int *n, + const lapack_complex_double *a, + lapack_int *lda, + double *s, + double *scond, + double *amax, + lapack_int *info); +void LAPACK_dpoequb(lapack_int *n, + const double *a, + lapack_int *lda, + double *s, + double *scond, + double *amax, + lapack_int *info); +void LAPACK_spoequb(lapack_int *n, + const float *a, + lapack_int *lda, + float *s, + float *scond, + float *amax, + lapack_int *info); +void LAPACK_zpoequb(lapack_int *n, + const lapack_complex_double *a, + lapack_int *lda, + double *s, + double *scond, + double *amax, + lapack_int *info); +void LAPACK_cpoequb(lapack_int *n, + const lapack_complex_float *a, + lapack_int *lda, + float *s, + float *scond, + float *amax, + lapack_int *info); +void LAPACK_sppequ(char *uplo, lapack_int *n, const float *ap, float *s, float *scond, float *amax, lapack_int *info); +void LAPACK_dppequ(char *uplo, + lapack_int *n, + const double *ap, + double *s, + double *scond, + double *amax, + lapack_int *info); +void LAPACK_cppequ(char *uplo, + lapack_int *n, + const lapack_complex_float *ap, + float *s, + float *scond, + float *amax, + lapack_int *info); +void LAPACK_zppequ(char *uplo, + lapack_int *n, + const lapack_complex_double *ap, + double *s, + double *scond, + double *amax, + lapack_int *info); +void LAPACK_spbequ(char *uplo, + lapack_int *n, + lapack_int *kd, + const float *ab, + lapack_int *ldab, + float *s, + float *scond, + float *amax, + lapack_int *info); +void LAPACK_dpbequ(char *uplo, + lapack_int *n, + lapack_int *kd, + const double *ab, + lapack_int *ldab, + double *s, + double *scond, + double *amax, + lapack_int *info); +void LAPACK_cpbequ(char *uplo, + lapack_int *n, + lapack_int *kd, + const lapack_complex_float *ab, + lapack_int *ldab, + float *s, + float *scond, + float *amax, + lapack_int *info); +void LAPACK_zpbequ(char *uplo, + lapack_int *n, + lapack_int *kd, + const lapack_complex_double *ab, + lapack_int *ldab, + double *s, + double *scond, + double *amax, + lapack_int *info); +void LAPACK_dsyequb(char *uplo, + lapack_int *n, + const double *a, + lapack_int *lda, + double *s, + double *scond, + double *amax, + double *work, + lapack_int *info); +void LAPACK_ssyequb(char *uplo, + lapack_int *n, + const float *a, + lapack_int *lda, + float *s, + float *scond, + float *amax, + float *work, + lapack_int *info); +void LAPACK_zsyequb(char *uplo, + lapack_int *n, + const lapack_complex_double *a, + lapack_int *lda, + double *s, + double *scond, + double *amax, + lapack_complex_double *work, + lapack_int *info); +void LAPACK_csyequb(char *uplo, + lapack_int *n, + const lapack_complex_float *a, + lapack_int *lda, + float *s, + float *scond, + float *amax, + lapack_complex_float *work, + lapack_int *info); +void LAPACK_zheequb(char *uplo, + lapack_int *n, + const lapack_complex_double *a, + lapack_int *lda, + double *s, + double *scond, + double *amax, + lapack_complex_double *work, + lapack_int *info); +void LAPACK_cheequb(char *uplo, + lapack_int *n, + const lapack_complex_float *a, + lapack_int *lda, + float *s, + float *scond, + float *amax, + lapack_complex_float *work, + lapack_int *info); +void LAPACK_sgesv(lapack_int *n, + lapack_int *nrhs, + float *a, + lapack_int *lda, + lapack_int *ipiv, + float *b, + lapack_int *ldb, + lapack_int *info); +void LAPACK_dgesv(lapack_int *n, + lapack_int *nrhs, + double *a, + lapack_int *lda, + lapack_int *ipiv, + double *b, + lapack_int *ldb, + lapack_int *info); +void LAPACK_cgesv(lapack_int *n, + lapack_int *nrhs, + lapack_complex_float *a, + lapack_int *lda, + lapack_int *ipiv, + lapack_complex_float *b, + lapack_int *ldb, + lapack_int *info); +void LAPACK_zgesv(lapack_int *n, + lapack_int *nrhs, + lapack_complex_double *a, + lapack_int *lda, + lapack_int *ipiv, + lapack_complex_double *b, + lapack_int *ldb, + lapack_int *info); +void LAPACK_dsgesv(lapack_int *n, + lapack_int *nrhs, + double *a, + lapack_int *lda, + lapack_int *ipiv, + double *b, + lapack_int *ldb, + double *x, + lapack_int *ldx, + double *work, + float *swork, + lapack_int *iter, + lapack_int *info); +void LAPACK_zcgesv(lapack_int *n, + lapack_int *nrhs, + lapack_complex_double *a, + lapack_int *lda, + lapack_int *ipiv, + lapack_complex_double *b, + lapack_int *ldb, + lapack_complex_double *x, + lapack_int *ldx, + lapack_complex_double *work, + lapack_complex_float *swork, + double *rwork, + lapack_int *iter, + lapack_int *info); +void LAPACK_sgesvx(char *fact, + char *trans, + lapack_int *n, + lapack_int *nrhs, + float *a, + lapack_int *lda, + float *af, + lapack_int *ldaf, + lapack_int *ipiv, + char *equed, + float *r, + float *c, + float *b, + lapack_int *ldb, + float *x, + lapack_int *ldx, + float *rcond, + float *ferr, + float *berr, + float *work, + lapack_int *iwork, + lapack_int *info); +void LAPACK_dgesvx(char *fact, + char *trans, + lapack_int *n, + lapack_int *nrhs, + double *a, + lapack_int *lda, + double *af, + lapack_int *ldaf, + lapack_int *ipiv, + char *equed, + double *r, + double *c, + double *b, + lapack_int *ldb, + double *x, + lapack_int *ldx, + double *rcond, + double *ferr, + double *berr, + double *work, + lapack_int *iwork, + lapack_int *info); +void LAPACK_cgesvx(char *fact, + char *trans, + lapack_int *n, + lapack_int *nrhs, + lapack_complex_float *a, + lapack_int *lda, + lapack_complex_float *af, + lapack_int *ldaf, + lapack_int *ipiv, + char *equed, + float *r, + float *c, + lapack_complex_float *b, + lapack_int *ldb, + lapack_complex_float *x, + lapack_int *ldx, + float *rcond, + float *ferr, + float *berr, + lapack_complex_float *work, + float *rwork, + lapack_int *info); +void LAPACK_zgesvx(char *fact, + char *trans, + lapack_int *n, + lapack_int *nrhs, + lapack_complex_double *a, + lapack_int *lda, + lapack_complex_double *af, + lapack_int *ldaf, + lapack_int *ipiv, + char *equed, + double *r, + double *c, + lapack_complex_double *b, + lapack_int *ldb, + lapack_complex_double *x, + lapack_int *ldx, + double *rcond, + double *ferr, + double *berr, + lapack_complex_double *work, + double *rwork, + lapack_int *info); +void LAPACK_dgesvxx(char *fact, + char *trans, + lapack_int *n, + lapack_int *nrhs, + double *a, + lapack_int *lda, + double *af, + lapack_int *ldaf, + lapack_int *ipiv, + char *equed, + double *r, + double *c, + double *b, + lapack_int *ldb, + double *x, + lapack_int *ldx, + double *rcond, + double *rpvgrw, + double *berr, + lapack_int *n_err_bnds, + double *err_bnds_norm, + double *err_bnds_comp, + lapack_int *nparams, + double *params, + double *work, + lapack_int *iwork, + lapack_int *info); +void LAPACK_sgesvxx(char *fact, + char *trans, + lapack_int *n, + lapack_int *nrhs, + float *a, + lapack_int *lda, + float *af, + lapack_int *ldaf, + lapack_int *ipiv, + char *equed, + float *r, + float *c, + float *b, + lapack_int *ldb, + float *x, + lapack_int *ldx, + float *rcond, + float *rpvgrw, + float *berr, + lapack_int *n_err_bnds, + float *err_bnds_norm, + float *err_bnds_comp, + lapack_int *nparams, + float *params, + float *work, + lapack_int *iwork, + lapack_int *info); +void LAPACK_zgesvxx(char *fact, + char *trans, + lapack_int *n, + lapack_int *nrhs, + lapack_complex_double *a, + lapack_int *lda, + lapack_complex_double *af, + lapack_int *ldaf, + lapack_int *ipiv, + char *equed, + double *r, + double *c, + lapack_complex_double *b, + lapack_int *ldb, + lapack_complex_double *x, + lapack_int *ldx, + double *rcond, + double *rpvgrw, + double *berr, + lapack_int *n_err_bnds, + double *err_bnds_norm, + double *err_bnds_comp, + lapack_int *nparams, + double *params, + lapack_complex_double *work, + double *rwork, + lapack_int *info); +void LAPACK_cgesvxx(char *fact, + char *trans, + lapack_int *n, + lapack_int *nrhs, + lapack_complex_float *a, + lapack_int *lda, + lapack_complex_float *af, + lapack_int *ldaf, + lapack_int *ipiv, + char *equed, + float *r, + float *c, + lapack_complex_float *b, + lapack_int *ldb, + lapack_complex_float *x, + lapack_int *ldx, + float *rcond, + float *rpvgrw, + float *berr, + lapack_int *n_err_bnds, + float *err_bnds_norm, + float *err_bnds_comp, + lapack_int *nparams, + float *params, + lapack_complex_float *work, + float *rwork, + lapack_int *info); +void LAPACK_sgbsv(lapack_int *n, + lapack_int *kl, + lapack_int *ku, + lapack_int *nrhs, + float *ab, + lapack_int *ldab, + lapack_int *ipiv, + float *b, + lapack_int *ldb, + lapack_int *info); +void LAPACK_dgbsv(lapack_int *n, + lapack_int *kl, + lapack_int *ku, + lapack_int *nrhs, + double *ab, + lapack_int *ldab, + lapack_int *ipiv, + double *b, + lapack_int *ldb, + lapack_int *info); +void LAPACK_cgbsv(lapack_int *n, + lapack_int *kl, + lapack_int *ku, + lapack_int *nrhs, + lapack_complex_float *ab, + lapack_int *ldab, + lapack_int *ipiv, + lapack_complex_float *b, + lapack_int *ldb, + lapack_int *info); +void LAPACK_zgbsv(lapack_int *n, + lapack_int *kl, + lapack_int *ku, + lapack_int *nrhs, + lapack_complex_double *ab, + lapack_int *ldab, + lapack_int *ipiv, + lapack_complex_double *b, + lapack_int *ldb, + lapack_int *info); +void LAPACK_sgbsvx(char *fact, + char *trans, + lapack_int *n, + lapack_int *kl, + lapack_int *ku, + lapack_int *nrhs, + float *ab, + lapack_int *ldab, + float *afb, + lapack_int *ldafb, + lapack_int *ipiv, + char *equed, + float *r, + float *c, + float *b, + lapack_int *ldb, + float *x, + lapack_int *ldx, + float *rcond, + float *ferr, + float *berr, + float *work, + lapack_int *iwork, + lapack_int *info); +void LAPACK_dgbsvx(char *fact, + char *trans, + lapack_int *n, + lapack_int *kl, + lapack_int *ku, + lapack_int *nrhs, + double *ab, + lapack_int *ldab, + double *afb, + lapack_int *ldafb, + lapack_int *ipiv, + char *equed, + double *r, + double *c, + double *b, + lapack_int *ldb, + double *x, + lapack_int *ldx, + double *rcond, + double *ferr, + double *berr, + double *work, + lapack_int *iwork, + lapack_int *info); +void LAPACK_cgbsvx(char *fact, + char *trans, + lapack_int *n, + lapack_int *kl, + lapack_int *ku, + lapack_int *nrhs, + lapack_complex_float *ab, + lapack_int *ldab, + lapack_complex_float *afb, + lapack_int *ldafb, + lapack_int *ipiv, + char *equed, + float *r, + float *c, + lapack_complex_float *b, + lapack_int *ldb, + lapack_complex_float *x, + lapack_int *ldx, + float *rcond, + float *ferr, + float *berr, + lapack_complex_float *work, + float *rwork, + lapack_int *info); +void LAPACK_zgbsvx(char *fact, + char *trans, + lapack_int *n, + lapack_int *kl, + lapack_int *ku, + lapack_int *nrhs, + lapack_complex_double *ab, + lapack_int *ldab, + lapack_complex_double *afb, + lapack_int *ldafb, + lapack_int *ipiv, + char *equed, + double *r, + double *c, + lapack_complex_double *b, + lapack_int *ldb, + lapack_complex_double *x, + lapack_int *ldx, + double *rcond, + double *ferr, + double *berr, + lapack_complex_double *work, + double *rwork, + lapack_int *info); +void LAPACK_dgbsvxx(char *fact, + char *trans, + lapack_int *n, + lapack_int *kl, + lapack_int *ku, + lapack_int *nrhs, + double *ab, + lapack_int *ldab, + double *afb, + lapack_int *ldafb, + lapack_int *ipiv, + char *equed, + double *r, + double *c, + double *b, + lapack_int *ldb, + double *x, + lapack_int *ldx, + double *rcond, + double *rpvgrw, + double *berr, + lapack_int *n_err_bnds, + double *err_bnds_norm, + double *err_bnds_comp, + lapack_int *nparams, + double *params, + double *work, + lapack_int *iwork, + lapack_int *info); +void LAPACK_sgbsvxx(char *fact, + char *trans, + lapack_int *n, + lapack_int *kl, + lapack_int *ku, + lapack_int *nrhs, + float *ab, + lapack_int *ldab, + float *afb, + lapack_int *ldafb, + lapack_int *ipiv, + char *equed, + float *r, + float *c, + float *b, + lapack_int *ldb, + float *x, + lapack_int *ldx, + float *rcond, + float *rpvgrw, + float *berr, + lapack_int *n_err_bnds, + float *err_bnds_norm, + float *err_bnds_comp, + lapack_int *nparams, + float *params, + float *work, + lapack_int *iwork, + lapack_int *info); +void LAPACK_zgbsvxx(char *fact, + char *trans, + lapack_int *n, + lapack_int *kl, + lapack_int *ku, + lapack_int *nrhs, + lapack_complex_double *ab, + lapack_int *ldab, + lapack_complex_double *afb, + lapack_int *ldafb, + lapack_int *ipiv, + char *equed, + double *r, + double *c, + lapack_complex_double *b, + lapack_int *ldb, + lapack_complex_double *x, + lapack_int *ldx, + double *rcond, + double *rpvgrw, + double *berr, + lapack_int *n_err_bnds, + double *err_bnds_norm, + double *err_bnds_comp, + lapack_int *nparams, + double *params, + lapack_complex_double *work, + double *rwork, + lapack_int *info); +void LAPACK_cgbsvxx(char *fact, + char *trans, + lapack_int *n, + lapack_int *kl, + lapack_int *ku, + lapack_int *nrhs, + lapack_complex_float *ab, + lapack_int *ldab, + lapack_complex_float *afb, + lapack_int *ldafb, + lapack_int *ipiv, + char *equed, + float *r, + float *c, + lapack_complex_float *b, + lapack_int *ldb, + lapack_complex_float *x, + lapack_int *ldx, + float *rcond, + float *rpvgrw, + float *berr, + lapack_int *n_err_bnds, + float *err_bnds_norm, + float *err_bnds_comp, + lapack_int *nparams, + float *params, + lapack_complex_float *work, + float *rwork, + lapack_int *info); +void LAPACK_sgtsv(lapack_int *n, + lapack_int *nrhs, + float *dl, + float *d, + float *du, + float *b, + lapack_int *ldb, + lapack_int *info); +void LAPACK_dgtsv(lapack_int *n, + lapack_int *nrhs, + double *dl, + double *d, + double *du, + double *b, + lapack_int *ldb, + lapack_int *info); +void LAPACK_cgtsv(lapack_int *n, + lapack_int *nrhs, + lapack_complex_float *dl, + lapack_complex_float *d, + lapack_complex_float *du, + lapack_complex_float *b, + lapack_int *ldb, + lapack_int *info); +void LAPACK_zgtsv(lapack_int *n, + lapack_int *nrhs, + lapack_complex_double *dl, + lapack_complex_double *d, + lapack_complex_double *du, + lapack_complex_double *b, + lapack_int *ldb, + lapack_int *info); +void LAPACK_sgtsvx(char *fact, + char *trans, + lapack_int *n, + lapack_int *nrhs, + const float *dl, + const float *d, + const float *du, + float *dlf, + float *df, + float *duf, + float *du2, + lapack_int *ipiv, + const float *b, + lapack_int *ldb, + float *x, + lapack_int *ldx, + float *rcond, + float *ferr, + float *berr, + float *work, + lapack_int *iwork, + lapack_int *info); +void LAPACK_dgtsvx(char *fact, + char *trans, + lapack_int *n, + lapack_int *nrhs, + const double *dl, + const double *d, + const double *du, + double *dlf, + double *df, + double *duf, + double *du2, + lapack_int *ipiv, + const double *b, + lapack_int *ldb, + double *x, + lapack_int *ldx, + double *rcond, + double *ferr, + double *berr, + double *work, + lapack_int *iwork, + lapack_int *info); +void LAPACK_cgtsvx(char *fact, + char *trans, + lapack_int *n, + lapack_int *nrhs, + const lapack_complex_float *dl, + const lapack_complex_float *d, + const lapack_complex_float *du, + lapack_complex_float *dlf, + lapack_complex_float *df, + lapack_complex_float *duf, + lapack_complex_float *du2, + lapack_int *ipiv, + const lapack_complex_float *b, + lapack_int *ldb, + lapack_complex_float *x, + lapack_int *ldx, + float *rcond, + float *ferr, + float *berr, + lapack_complex_float *work, + float *rwork, + lapack_int *info); +void LAPACK_zgtsvx(char *fact, + char *trans, + lapack_int *n, + lapack_int *nrhs, + const lapack_complex_double *dl, + const lapack_complex_double *d, + const lapack_complex_double *du, + lapack_complex_double *dlf, + lapack_complex_double *df, + lapack_complex_double *duf, + lapack_complex_double *du2, + lapack_int *ipiv, + const lapack_complex_double *b, + lapack_int *ldb, + lapack_complex_double *x, + lapack_int *ldx, + double *rcond, + double *ferr, + double *berr, + lapack_complex_double *work, + double *rwork, + lapack_int *info); +void LAPACK_sposv(char *uplo, + lapack_int *n, + lapack_int *nrhs, + float *a, + lapack_int *lda, + float *b, + lapack_int *ldb, + lapack_int *info); +void LAPACK_dposv(char *uplo, + lapack_int *n, + lapack_int *nrhs, + double *a, + lapack_int *lda, + double *b, + lapack_int *ldb, + lapack_int *info); +void LAPACK_cposv(char *uplo, + lapack_int *n, + lapack_int *nrhs, + lapack_complex_float *a, + lapack_int *lda, + lapack_complex_float *b, + lapack_int *ldb, + lapack_int *info); +void LAPACK_zposv(char *uplo, + lapack_int *n, + lapack_int *nrhs, + lapack_complex_double *a, + lapack_int *lda, + lapack_complex_double *b, + lapack_int *ldb, + lapack_int *info); +void LAPACK_dsposv(char *uplo, + lapack_int *n, + lapack_int *nrhs, + double *a, + lapack_int *lda, + double *b, + lapack_int *ldb, + double *x, + lapack_int *ldx, + double *work, + float *swork, + lapack_int *iter, + lapack_int *info); +void LAPACK_zcposv(char *uplo, + lapack_int *n, + lapack_int *nrhs, + lapack_complex_double *a, + lapack_int *lda, + lapack_complex_double *b, + lapack_int *ldb, + lapack_complex_double *x, + lapack_int *ldx, + lapack_complex_double *work, + lapack_complex_float *swork, + double *rwork, + lapack_int *iter, + lapack_int *info); +void LAPACK_sposvx(char *fact, + char *uplo, + lapack_int *n, + lapack_int *nrhs, + float *a, + lapack_int *lda, + float *af, + lapack_int *ldaf, + char *equed, + float *s, + float *b, + lapack_int *ldb, + float *x, + lapack_int *ldx, + float *rcond, + float *ferr, + float *berr, + float *work, + lapack_int *iwork, + lapack_int *info); +void LAPACK_dposvx(char *fact, + char *uplo, + lapack_int *n, + lapack_int *nrhs, + double *a, + lapack_int *lda, + double *af, + lapack_int *ldaf, + char *equed, + double *s, + double *b, + lapack_int *ldb, + double *x, + lapack_int *ldx, + double *rcond, + double *ferr, + double *berr, + double *work, + lapack_int *iwork, + lapack_int *info); +void LAPACK_cposvx(char *fact, + char *uplo, + lapack_int *n, + lapack_int *nrhs, + lapack_complex_float *a, + lapack_int *lda, + lapack_complex_float *af, + lapack_int *ldaf, + char *equed, + float *s, + lapack_complex_float *b, + lapack_int *ldb, + lapack_complex_float *x, + lapack_int *ldx, + float *rcond, + float *ferr, + float *berr, + lapack_complex_float *work, + float *rwork, + lapack_int *info); +void LAPACK_zposvx(char *fact, + char *uplo, + lapack_int *n, + lapack_int *nrhs, + lapack_complex_double *a, + lapack_int *lda, + lapack_complex_double *af, + lapack_int *ldaf, + char *equed, + double *s, + lapack_complex_double *b, + lapack_int *ldb, + lapack_complex_double *x, + lapack_int *ldx, + double *rcond, + double *ferr, + double *berr, + lapack_complex_double *work, + double *rwork, + lapack_int *info); +void LAPACK_dposvxx(char *fact, + char *uplo, + lapack_int *n, + lapack_int *nrhs, + double *a, + lapack_int *lda, + double *af, + lapack_int *ldaf, + char *equed, + double *s, + double *b, + lapack_int *ldb, + double *x, + lapack_int *ldx, + double *rcond, + double *rpvgrw, + double *berr, + lapack_int *n_err_bnds, + double *err_bnds_norm, + double *err_bnds_comp, + lapack_int *nparams, + double *params, + double *work, + lapack_int *iwork, + lapack_int *info); +void LAPACK_sposvxx(char *fact, + char *uplo, + lapack_int *n, + lapack_int *nrhs, + float *a, + lapack_int *lda, + float *af, + lapack_int *ldaf, + char *equed, + float *s, + float *b, + lapack_int *ldb, + float *x, + lapack_int *ldx, + float *rcond, + float *rpvgrw, + float *berr, + lapack_int *n_err_bnds, + float *err_bnds_norm, + float *err_bnds_comp, + lapack_int *nparams, + float *params, + float *work, + lapack_int *iwork, + lapack_int *info); +void LAPACK_zposvxx(char *fact, + char *uplo, + lapack_int *n, + lapack_int *nrhs, + lapack_complex_double *a, + lapack_int *lda, + lapack_complex_double *af, + lapack_int *ldaf, + char *equed, + double *s, + lapack_complex_double *b, + lapack_int *ldb, + lapack_complex_double *x, + lapack_int *ldx, + double *rcond, + double *rpvgrw, + double *berr, + lapack_int *n_err_bnds, + double *err_bnds_norm, + double *err_bnds_comp, + lapack_int *nparams, + double *params, + lapack_complex_double *work, + double *rwork, + lapack_int *info); +void LAPACK_cposvxx(char *fact, + char *uplo, + lapack_int *n, + lapack_int *nrhs, + lapack_complex_float *a, + lapack_int *lda, + lapack_complex_float *af, + lapack_int *ldaf, + char *equed, + float *s, + lapack_complex_float *b, + lapack_int *ldb, + lapack_complex_float *x, + lapack_int *ldx, + float *rcond, + float *rpvgrw, + float *berr, + lapack_int *n_err_bnds, + float *err_bnds_norm, + float *err_bnds_comp, + lapack_int *nparams, + float *params, + lapack_complex_float *work, + float *rwork, + lapack_int *info); +void LAPACK_sppsv(char *uplo, lapack_int *n, lapack_int *nrhs, float *ap, float *b, lapack_int *ldb, lapack_int *info); +void LAPACK_dppsv(char *uplo, + lapack_int *n, + lapack_int *nrhs, + double *ap, + double *b, + lapack_int *ldb, + lapack_int *info); +void LAPACK_cppsv(char *uplo, + lapack_int *n, + lapack_int *nrhs, + lapack_complex_float *ap, + lapack_complex_float *b, + lapack_int *ldb, + lapack_int *info); +void LAPACK_zppsv(char *uplo, + lapack_int *n, + lapack_int *nrhs, + lapack_complex_double *ap, + lapack_complex_double *b, + lapack_int *ldb, + lapack_int *info); +void LAPACK_sppsvx(char *fact, + char *uplo, + lapack_int *n, + lapack_int *nrhs, + float *ap, + float *afp, + char *equed, + float *s, + float *b, + lapack_int *ldb, + float *x, + lapack_int *ldx, + float *rcond, + float *ferr, + float *berr, + float *work, + lapack_int *iwork, + lapack_int *info); +void LAPACK_dppsvx(char *fact, + char *uplo, + lapack_int *n, + lapack_int *nrhs, + double *ap, + double *afp, + char *equed, + double *s, + double *b, + lapack_int *ldb, + double *x, + lapack_int *ldx, + double *rcond, + double *ferr, + double *berr, + double *work, + lapack_int *iwork, + lapack_int *info); +void LAPACK_cppsvx(char *fact, + char *uplo, + lapack_int *n, + lapack_int *nrhs, + lapack_complex_float *ap, + lapack_complex_float *afp, + char *equed, + float *s, + lapack_complex_float *b, + lapack_int *ldb, + lapack_complex_float *x, + lapack_int *ldx, + float *rcond, + float *ferr, + float *berr, + lapack_complex_float *work, + float *rwork, + lapack_int *info); +void LAPACK_zppsvx(char *fact, + char *uplo, + lapack_int *n, + lapack_int *nrhs, + lapack_complex_double *ap, + lapack_complex_double *afp, + char *equed, + double *s, + lapack_complex_double *b, + lapack_int *ldb, + lapack_complex_double *x, + lapack_int *ldx, + double *rcond, + double *ferr, + double *berr, + lapack_complex_double *work, + double *rwork, + lapack_int *info); +void LAPACK_spbsv(char *uplo, + lapack_int *n, + lapack_int *kd, + lapack_int *nrhs, + float *ab, + lapack_int *ldab, + float *b, + lapack_int *ldb, + lapack_int *info); +void LAPACK_dpbsv(char *uplo, + lapack_int *n, + lapack_int *kd, + lapack_int *nrhs, + double *ab, + lapack_int *ldab, + double *b, + lapack_int *ldb, + lapack_int *info); +void LAPACK_cpbsv(char *uplo, + lapack_int *n, + lapack_int *kd, + lapack_int *nrhs, + lapack_complex_float *ab, + lapack_int *ldab, + lapack_complex_float *b, + lapack_int *ldb, + lapack_int *info); +void LAPACK_zpbsv(char *uplo, + lapack_int *n, + lapack_int *kd, + lapack_int *nrhs, + lapack_complex_double *ab, + lapack_int *ldab, + lapack_complex_double *b, + lapack_int *ldb, + lapack_int *info); +void LAPACK_spbsvx(char *fact, + char *uplo, + lapack_int *n, + lapack_int *kd, + lapack_int *nrhs, + float *ab, + lapack_int *ldab, + float *afb, + lapack_int *ldafb, + char *equed, + float *s, + float *b, + lapack_int *ldb, + float *x, + lapack_int *ldx, + float *rcond, + float *ferr, + float *berr, + float *work, + lapack_int *iwork, + lapack_int *info); +void LAPACK_dpbsvx(char *fact, + char *uplo, + lapack_int *n, + lapack_int *kd, + lapack_int *nrhs, + double *ab, + lapack_int *ldab, + double *afb, + lapack_int *ldafb, + char *equed, + double *s, + double *b, + lapack_int *ldb, + double *x, + lapack_int *ldx, + double *rcond, + double *ferr, + double *berr, + double *work, + lapack_int *iwork, + lapack_int *info); +void LAPACK_cpbsvx(char *fact, + char *uplo, + lapack_int *n, + lapack_int *kd, + lapack_int *nrhs, + lapack_complex_float *ab, + lapack_int *ldab, + lapack_complex_float *afb, + lapack_int *ldafb, + char *equed, + float *s, + lapack_complex_float *b, + lapack_int *ldb, + lapack_complex_float *x, + lapack_int *ldx, + float *rcond, + float *ferr, + float *berr, + lapack_complex_float *work, + float *rwork, + lapack_int *info); +void LAPACK_zpbsvx(char *fact, + char *uplo, + lapack_int *n, + lapack_int *kd, + lapack_int *nrhs, + lapack_complex_double *ab, + lapack_int *ldab, + lapack_complex_double *afb, + lapack_int *ldafb, + char *equed, + double *s, + lapack_complex_double *b, + lapack_int *ldb, + lapack_complex_double *x, + lapack_int *ldx, + double *rcond, + double *ferr, + double *berr, + lapack_complex_double *work, + double *rwork, + lapack_int *info); +void LAPACK_sptsv(lapack_int *n, lapack_int *nrhs, float *d, float *e, float *b, lapack_int *ldb, lapack_int *info); +void LAPACK_dptsv(lapack_int *n, lapack_int *nrhs, double *d, double *e, double *b, lapack_int *ldb, lapack_int *info); +void LAPACK_cptsv(lapack_int *n, + lapack_int *nrhs, + float *d, + lapack_complex_float *e, + lapack_complex_float *b, + lapack_int *ldb, + lapack_int *info); +void LAPACK_zptsv(lapack_int *n, + lapack_int *nrhs, + double *d, + lapack_complex_double *e, + lapack_complex_double *b, + lapack_int *ldb, + lapack_int *info); +void LAPACK_sptsvx(char *fact, + lapack_int *n, + lapack_int *nrhs, + const float *d, + const float *e, + float *df, + float *ef, + const float *b, + lapack_int *ldb, + float *x, + lapack_int *ldx, + float *rcond, + float *ferr, + float *berr, + float *work, + lapack_int *info); +void LAPACK_dptsvx(char *fact, + lapack_int *n, + lapack_int *nrhs, + const double *d, + const double *e, + double *df, + double *ef, + const double *b, + lapack_int *ldb, + double *x, + lapack_int *ldx, + double *rcond, + double *ferr, + double *berr, + double *work, + lapack_int *info); +void LAPACK_cptsvx(char *fact, + lapack_int *n, + lapack_int *nrhs, + const float *d, + const lapack_complex_float *e, + float *df, + lapack_complex_float *ef, + const lapack_complex_float *b, + lapack_int *ldb, + lapack_complex_float *x, + lapack_int *ldx, + float *rcond, + float *ferr, + float *berr, + lapack_complex_float *work, + float *rwork, + lapack_int *info); +void LAPACK_zptsvx(char *fact, + lapack_int *n, + lapack_int *nrhs, + const double *d, + const lapack_complex_double *e, + double *df, + lapack_complex_double *ef, + const lapack_complex_double *b, + lapack_int *ldb, + lapack_complex_double *x, + lapack_int *ldx, + double *rcond, + double *ferr, + double *berr, + lapack_complex_double *work, + double *rwork, + lapack_int *info); +void LAPACK_ssysv(char *uplo, + lapack_int *n, + lapack_int *nrhs, + float *a, + lapack_int *lda, + lapack_int *ipiv, + float *b, + lapack_int *ldb, + float *work, + lapack_int *lwork, + lapack_int *info); +void LAPACK_dsysv(char *uplo, + lapack_int *n, + lapack_int *nrhs, + double *a, + lapack_int *lda, + lapack_int *ipiv, + double *b, + lapack_int *ldb, + double *work, + lapack_int *lwork, + lapack_int *info); +void LAPACK_csysv(char *uplo, + lapack_int *n, + lapack_int *nrhs, + lapack_complex_float *a, + lapack_int *lda, + lapack_int *ipiv, + lapack_complex_float *b, + lapack_int *ldb, + lapack_complex_float *work, + lapack_int *lwork, + lapack_int *info); +void LAPACK_zsysv(char *uplo, + lapack_int *n, + lapack_int *nrhs, + lapack_complex_double *a, + lapack_int *lda, + lapack_int *ipiv, + lapack_complex_double *b, + lapack_int *ldb, + lapack_complex_double *work, + lapack_int *lwork, + lapack_int *info); +void LAPACK_ssysvx(char *fact, + char *uplo, + lapack_int *n, + lapack_int *nrhs, + const float *a, + lapack_int *lda, + float *af, + lapack_int *ldaf, + lapack_int *ipiv, + const float *b, + lapack_int *ldb, + float *x, + lapack_int *ldx, + float *rcond, + float *ferr, + float *berr, + float *work, + lapack_int *lwork, + lapack_int *iwork, + lapack_int *info); +void LAPACK_dsysvx(char *fact, + char *uplo, + lapack_int *n, + lapack_int *nrhs, + const double *a, + lapack_int *lda, + double *af, + lapack_int *ldaf, + lapack_int *ipiv, + const double *b, + lapack_int *ldb, + double *x, + lapack_int *ldx, + double *rcond, + double *ferr, + double *berr, + double *work, + lapack_int *lwork, + lapack_int *iwork, + lapack_int *info); +void LAPACK_csysvx(char *fact, + char *uplo, + lapack_int *n, + lapack_int *nrhs, + const lapack_complex_float *a, + lapack_int *lda, + lapack_complex_float *af, + lapack_int *ldaf, + lapack_int *ipiv, + const lapack_complex_float *b, + lapack_int *ldb, + lapack_complex_float *x, + lapack_int *ldx, + float *rcond, + float *ferr, + float *berr, + lapack_complex_float *work, + lapack_int *lwork, + float *rwork, + lapack_int *info); +void LAPACK_zsysvx(char *fact, + char *uplo, + lapack_int *n, + lapack_int *nrhs, + const lapack_complex_double *a, + lapack_int *lda, + lapack_complex_double *af, + lapack_int *ldaf, + lapack_int *ipiv, + const lapack_complex_double *b, + lapack_int *ldb, + lapack_complex_double *x, + lapack_int *ldx, + double *rcond, + double *ferr, + double *berr, + lapack_complex_double *work, + lapack_int *lwork, + double *rwork, + lapack_int *info); +void LAPACK_dsysvxx(char *fact, + char *uplo, + lapack_int *n, + lapack_int *nrhs, + double *a, + lapack_int *lda, + double *af, + lapack_int *ldaf, + lapack_int *ipiv, + char *equed, + double *s, + double *b, + lapack_int *ldb, + double *x, + lapack_int *ldx, + double *rcond, + double *rpvgrw, + double *berr, + lapack_int *n_err_bnds, + double *err_bnds_norm, + double *err_bnds_comp, + lapack_int *nparams, + double *params, + double *work, + lapack_int *iwork, + lapack_int *info); +void LAPACK_ssysvxx(char *fact, + char *uplo, + lapack_int *n, + lapack_int *nrhs, + float *a, + lapack_int *lda, + float *af, + lapack_int *ldaf, + lapack_int *ipiv, + char *equed, + float *s, + float *b, + lapack_int *ldb, + float *x, + lapack_int *ldx, + float *rcond, + float *rpvgrw, + float *berr, + lapack_int *n_err_bnds, + float *err_bnds_norm, + float *err_bnds_comp, + lapack_int *nparams, + float *params, + float *work, + lapack_int *iwork, + lapack_int *info); +void LAPACK_zsysvxx(char *fact, + char *uplo, + lapack_int *n, + lapack_int *nrhs, + lapack_complex_double *a, + lapack_int *lda, + lapack_complex_double *af, + lapack_int *ldaf, + lapack_int *ipiv, + char *equed, + double *s, + lapack_complex_double *b, + lapack_int *ldb, + lapack_complex_double *x, + lapack_int *ldx, + double *rcond, + double *rpvgrw, + double *berr, + lapack_int *n_err_bnds, + double *err_bnds_norm, + double *err_bnds_comp, + lapack_int *nparams, + double *params, + lapack_complex_double *work, + double *rwork, + lapack_int *info); +void LAPACK_csysvxx(char *fact, + char *uplo, + lapack_int *n, + lapack_int *nrhs, + lapack_complex_float *a, + lapack_int *lda, + lapack_complex_float *af, + lapack_int *ldaf, + lapack_int *ipiv, + char *equed, + float *s, + lapack_complex_float *b, + lapack_int *ldb, + lapack_complex_float *x, + lapack_int *ldx, + float *rcond, + float *rpvgrw, + float *berr, + lapack_int *n_err_bnds, + float *err_bnds_norm, + float *err_bnds_comp, + lapack_int *nparams, + float *params, + lapack_complex_float *work, + float *rwork, + lapack_int *info); +void LAPACK_chesv(char *uplo, + lapack_int *n, + lapack_int *nrhs, + lapack_complex_float *a, + lapack_int *lda, + lapack_int *ipiv, + lapack_complex_float *b, + lapack_int *ldb, + lapack_complex_float *work, + lapack_int *lwork, + lapack_int *info); +void LAPACK_zhesv(char *uplo, + lapack_int *n, + lapack_int *nrhs, + lapack_complex_double *a, + lapack_int *lda, + lapack_int *ipiv, + lapack_complex_double *b, + lapack_int *ldb, + lapack_complex_double *work, + lapack_int *lwork, + lapack_int *info); +void LAPACK_chesvx(char *fact, + char *uplo, + lapack_int *n, + lapack_int *nrhs, + const lapack_complex_float *a, + lapack_int *lda, + lapack_complex_float *af, + lapack_int *ldaf, + lapack_int *ipiv, + const lapack_complex_float *b, + lapack_int *ldb, + lapack_complex_float *x, + lapack_int *ldx, + float *rcond, + float *ferr, + float *berr, + lapack_complex_float *work, + lapack_int *lwork, + float *rwork, + lapack_int *info); +void LAPACK_zhesvx(char *fact, + char *uplo, + lapack_int *n, + lapack_int *nrhs, + const lapack_complex_double *a, + lapack_int *lda, + lapack_complex_double *af, + lapack_int *ldaf, + lapack_int *ipiv, + const lapack_complex_double *b, + lapack_int *ldb, + lapack_complex_double *x, + lapack_int *ldx, + double *rcond, + double *ferr, + double *berr, + lapack_complex_double *work, + lapack_int *lwork, + double *rwork, + lapack_int *info); +void LAPACK_zhesvxx(char *fact, + char *uplo, + lapack_int *n, + lapack_int *nrhs, + lapack_complex_double *a, + lapack_int *lda, + lapack_complex_double *af, + lapack_int *ldaf, + lapack_int *ipiv, + char *equed, + double *s, + lapack_complex_double *b, + lapack_int *ldb, + lapack_complex_double *x, + lapack_int *ldx, + double *rcond, + double *rpvgrw, + double *berr, + lapack_int *n_err_bnds, + double *err_bnds_norm, + double *err_bnds_comp, + lapack_int *nparams, + double *params, + lapack_complex_double *work, + double *rwork, + lapack_int *info); +void LAPACK_chesvxx(char *fact, + char *uplo, + lapack_int *n, + lapack_int *nrhs, + lapack_complex_float *a, + lapack_int *lda, + lapack_complex_float *af, + lapack_int *ldaf, + lapack_int *ipiv, + char *equed, + float *s, + lapack_complex_float *b, + lapack_int *ldb, + lapack_complex_float *x, + lapack_int *ldx, + float *rcond, + float *rpvgrw, + float *berr, + lapack_int *n_err_bnds, + float *err_bnds_norm, + float *err_bnds_comp, + lapack_int *nparams, + float *params, + lapack_complex_float *work, + float *rwork, + lapack_int *info); +void LAPACK_sspsv(char *uplo, + lapack_int *n, + lapack_int *nrhs, + float *ap, + lapack_int *ipiv, + float *b, + lapack_int *ldb, + lapack_int *info); +void LAPACK_dspsv(char *uplo, + lapack_int *n, + lapack_int *nrhs, + double *ap, + lapack_int *ipiv, + double *b, + lapack_int *ldb, + lapack_int *info); +void LAPACK_cspsv(char *uplo, + lapack_int *n, + lapack_int *nrhs, + lapack_complex_float *ap, + lapack_int *ipiv, + lapack_complex_float *b, + lapack_int *ldb, + lapack_int *info); +void LAPACK_zspsv(char *uplo, + lapack_int *n, + lapack_int *nrhs, + lapack_complex_double *ap, + lapack_int *ipiv, + lapack_complex_double *b, + lapack_int *ldb, + lapack_int *info); +void LAPACK_sspsvx(char *fact, + char *uplo, + lapack_int *n, + lapack_int *nrhs, + const float *ap, + float *afp, + lapack_int *ipiv, + const float *b, + lapack_int *ldb, + float *x, + lapack_int *ldx, + float *rcond, + float *ferr, + float *berr, + float *work, + lapack_int *iwork, + lapack_int *info); +void LAPACK_dspsvx(char *fact, + char *uplo, + lapack_int *n, + lapack_int *nrhs, + const double *ap, + double *afp, + lapack_int *ipiv, + const double *b, + lapack_int *ldb, + double *x, + lapack_int *ldx, + double *rcond, + double *ferr, + double *berr, + double *work, + lapack_int *iwork, + lapack_int *info); +void LAPACK_cspsvx(char *fact, + char *uplo, + lapack_int *n, + lapack_int *nrhs, + const lapack_complex_float *ap, + lapack_complex_float *afp, + lapack_int *ipiv, + const lapack_complex_float *b, + lapack_int *ldb, + lapack_complex_float *x, + lapack_int *ldx, + float *rcond, + float *ferr, + float *berr, + lapack_complex_float *work, + float *rwork, + lapack_int *info); +void LAPACK_zspsvx(char *fact, + char *uplo, + lapack_int *n, + lapack_int *nrhs, + const lapack_complex_double *ap, + lapack_complex_double *afp, + lapack_int *ipiv, + const lapack_complex_double *b, + lapack_int *ldb, + lapack_complex_double *x, + lapack_int *ldx, + double *rcond, + double *ferr, + double *berr, + lapack_complex_double *work, + double *rwork, + lapack_int *info); +void LAPACK_chpsv(char *uplo, + lapack_int *n, + lapack_int *nrhs, + lapack_complex_float *ap, + lapack_int *ipiv, + lapack_complex_float *b, + lapack_int *ldb, + lapack_int *info); +void LAPACK_zhpsv(char *uplo, + lapack_int *n, + lapack_int *nrhs, + lapack_complex_double *ap, + lapack_int *ipiv, + lapack_complex_double *b, + lapack_int *ldb, + lapack_int *info); +void LAPACK_chpsvx(char *fact, + char *uplo, + lapack_int *n, + lapack_int *nrhs, + const lapack_complex_float *ap, + lapack_complex_float *afp, + lapack_int *ipiv, + const lapack_complex_float *b, + lapack_int *ldb, + lapack_complex_float *x, + lapack_int *ldx, + float *rcond, + float *ferr, + float *berr, + lapack_complex_float *work, + float *rwork, + lapack_int *info); +void LAPACK_zhpsvx(char *fact, + char *uplo, + lapack_int *n, + lapack_int *nrhs, + const lapack_complex_double *ap, + lapack_complex_double *afp, + lapack_int *ipiv, + const lapack_complex_double *b, + lapack_int *ldb, + lapack_complex_double *x, + lapack_int *ldx, + double *rcond, + double *ferr, + double *berr, + lapack_complex_double *work, + double *rwork, + lapack_int *info); +void LAPACK_sgeqrf(lapack_int *m, + lapack_int *n, + float *a, + lapack_int *lda, + float *tau, + float *work, + lapack_int *lwork, + lapack_int *info); +void LAPACK_dgeqrf(lapack_int *m, + lapack_int *n, + double *a, + lapack_int *lda, + double *tau, + double *work, + lapack_int *lwork, + lapack_int *info); +void LAPACK_cgeqrf(lapack_int *m, + lapack_int *n, + lapack_complex_float *a, + lapack_int *lda, + lapack_complex_float *tau, + lapack_complex_float *work, + lapack_int *lwork, + lapack_int *info); +void LAPACK_zgeqrf(lapack_int *m, + lapack_int *n, + lapack_complex_double *a, + lapack_int *lda, + lapack_complex_double *tau, + lapack_complex_double *work, + lapack_int *lwork, + lapack_int *info); +void LAPACK_sgeqpf(lapack_int *m, + lapack_int *n, + float *a, + lapack_int *lda, + lapack_int *jpvt, + float *tau, + float *work, + lapack_int *info); +void LAPACK_dgeqpf(lapack_int *m, + lapack_int *n, + double *a, + lapack_int *lda, + lapack_int *jpvt, + double *tau, + double *work, + lapack_int *info); +void LAPACK_cgeqpf(lapack_int *m, + lapack_int *n, + lapack_complex_float *a, + lapack_int *lda, + lapack_int *jpvt, + lapack_complex_float *tau, + lapack_complex_float *work, + float *rwork, + lapack_int *info); +void LAPACK_zgeqpf(lapack_int *m, + lapack_int *n, + lapack_complex_double *a, + lapack_int *lda, + lapack_int *jpvt, + lapack_complex_double *tau, + lapack_complex_double *work, + double *rwork, + lapack_int *info); +void LAPACK_sgeqp3(lapack_int *m, + lapack_int *n, + float *a, + lapack_int *lda, + lapack_int *jpvt, + float *tau, + float *work, + lapack_int *lwork, + lapack_int *info); +void LAPACK_dgeqp3(lapack_int *m, + lapack_int *n, + double *a, + lapack_int *lda, + lapack_int *jpvt, + double *tau, + double *work, + lapack_int *lwork, + lapack_int *info); +void LAPACK_cgeqp3(lapack_int *m, + lapack_int *n, + lapack_complex_float *a, + lapack_int *lda, + lapack_int *jpvt, + lapack_complex_float *tau, + lapack_complex_float *work, + lapack_int *lwork, + float *rwork, + lapack_int *info); +void LAPACK_zgeqp3(lapack_int *m, + lapack_int *n, + lapack_complex_double *a, + lapack_int *lda, + lapack_int *jpvt, + lapack_complex_double *tau, + lapack_complex_double *work, + lapack_int *lwork, + double *rwork, + lapack_int *info); +void LAPACK_sorgqr(lapack_int *m, + lapack_int *n, + lapack_int *k, + float *a, + lapack_int *lda, + const float *tau, + float *work, + lapack_int *lwork, + lapack_int *info); +void LAPACK_dorgqr(lapack_int *m, + lapack_int *n, + lapack_int *k, + double *a, + lapack_int *lda, + const double *tau, + double *work, + lapack_int *lwork, + lapack_int *info); +void LAPACK_sormqr(char *side, + char *trans, + lapack_int *m, + lapack_int *n, + lapack_int *k, + const float *a, + lapack_int *lda, + const float *tau, + float *c, + lapack_int *ldc, + float *work, + lapack_int *lwork, + lapack_int *info); +void LAPACK_dormqr(char *side, + char *trans, + lapack_int *m, + lapack_int *n, + lapack_int *k, + const double *a, + lapack_int *lda, + const double *tau, + double *c, + lapack_int *ldc, + double *work, + lapack_int *lwork, + lapack_int *info); +void LAPACK_cungqr(lapack_int *m, + lapack_int *n, + lapack_int *k, + lapack_complex_float *a, + lapack_int *lda, + const lapack_complex_float *tau, + lapack_complex_float *work, + lapack_int *lwork, + lapack_int *info); +void LAPACK_zungqr(lapack_int *m, + lapack_int *n, + lapack_int *k, + lapack_complex_double *a, + lapack_int *lda, + const lapack_complex_double *tau, + lapack_complex_double *work, + lapack_int *lwork, + lapack_int *info); +void LAPACK_cunmqr(char *side, + char *trans, + lapack_int *m, + lapack_int *n, + lapack_int *k, + const lapack_complex_float *a, + lapack_int *lda, + const lapack_complex_float *tau, + lapack_complex_float *c, + lapack_int *ldc, + lapack_complex_float *work, + lapack_int *lwork, + lapack_int *info); +void LAPACK_zunmqr(char *side, + char *trans, + lapack_int *m, + lapack_int *n, + lapack_int *k, + const lapack_complex_double *a, + lapack_int *lda, + const lapack_complex_double *tau, + lapack_complex_double *c, + lapack_int *ldc, + lapack_complex_double *work, + lapack_int *lwork, + lapack_int *info); +void LAPACK_sgelqf(lapack_int *m, + lapack_int *n, + float *a, + lapack_int *lda, + float *tau, + float *work, + lapack_int *lwork, + lapack_int *info); +void LAPACK_dgelqf(lapack_int *m, + lapack_int *n, + double *a, + lapack_int *lda, + double *tau, + double *work, + lapack_int *lwork, + lapack_int *info); +void LAPACK_cgelqf(lapack_int *m, + lapack_int *n, + lapack_complex_float *a, + lapack_int *lda, + lapack_complex_float *tau, + lapack_complex_float *work, + lapack_int *lwork, + lapack_int *info); +void LAPACK_zgelqf(lapack_int *m, + lapack_int *n, + lapack_complex_double *a, + lapack_int *lda, + lapack_complex_double *tau, + lapack_complex_double *work, + lapack_int *lwork, + lapack_int *info); +void LAPACK_sorglq(lapack_int *m, + lapack_int *n, + lapack_int *k, + float *a, + lapack_int *lda, + const float *tau, + float *work, + lapack_int *lwork, + lapack_int *info); +void LAPACK_dorglq(lapack_int *m, + lapack_int *n, + lapack_int *k, + double *a, + lapack_int *lda, + const double *tau, + double *work, + lapack_int *lwork, + lapack_int *info); +void LAPACK_sormlq(char *side, + char *trans, + lapack_int *m, + lapack_int *n, + lapack_int *k, + const float *a, + lapack_int *lda, + const float *tau, + float *c, + lapack_int *ldc, + float *work, + lapack_int *lwork, + lapack_int *info); +void LAPACK_dormlq(char *side, + char *trans, + lapack_int *m, + lapack_int *n, + lapack_int *k, + const double *a, + lapack_int *lda, + const double *tau, + double *c, + lapack_int *ldc, + double *work, + lapack_int *lwork, + lapack_int *info); +void LAPACK_cunglq(lapack_int *m, + lapack_int *n, + lapack_int *k, + lapack_complex_float *a, + lapack_int *lda, + const lapack_complex_float *tau, + lapack_complex_float *work, + lapack_int *lwork, + lapack_int *info); +void LAPACK_zunglq(lapack_int *m, + lapack_int *n, + lapack_int *k, + lapack_complex_double *a, + lapack_int *lda, + const lapack_complex_double *tau, + lapack_complex_double *work, + lapack_int *lwork, + lapack_int *info); +void LAPACK_cunmlq(char *side, + char *trans, + lapack_int *m, + lapack_int *n, + lapack_int *k, + const lapack_complex_float *a, + lapack_int *lda, + const lapack_complex_float *tau, + lapack_complex_float *c, + lapack_int *ldc, + lapack_complex_float *work, + lapack_int *lwork, + lapack_int *info); +void LAPACK_zunmlq(char *side, + char *trans, + lapack_int *m, + lapack_int *n, + lapack_int *k, + const lapack_complex_double *a, + lapack_int *lda, + const lapack_complex_double *tau, + lapack_complex_double *c, + lapack_int *ldc, + lapack_complex_double *work, + lapack_int *lwork, + lapack_int *info); +void LAPACK_sgeqlf(lapack_int *m, + lapack_int *n, + float *a, + lapack_int *lda, + float *tau, + float *work, + lapack_int *lwork, + lapack_int *info); +void LAPACK_dgeqlf(lapack_int *m, + lapack_int *n, + double *a, + lapack_int *lda, + double *tau, + double *work, + lapack_int *lwork, + lapack_int *info); +void LAPACK_cgeqlf(lapack_int *m, + lapack_int *n, + lapack_complex_float *a, + lapack_int *lda, + lapack_complex_float *tau, + lapack_complex_float *work, + lapack_int *lwork, + lapack_int *info); +void LAPACK_zgeqlf(lapack_int *m, + lapack_int *n, + lapack_complex_double *a, + lapack_int *lda, + lapack_complex_double *tau, + lapack_complex_double *work, + lapack_int *lwork, + lapack_int *info); +void LAPACK_sorgql(lapack_int *m, + lapack_int *n, + lapack_int *k, + float *a, + lapack_int *lda, + const float *tau, + float *work, + lapack_int *lwork, + lapack_int *info); +void LAPACK_dorgql(lapack_int *m, + lapack_int *n, + lapack_int *k, + double *a, + lapack_int *lda, + const double *tau, + double *work, + lapack_int *lwork, + lapack_int *info); +void LAPACK_cungql(lapack_int *m, + lapack_int *n, + lapack_int *k, + lapack_complex_float *a, + lapack_int *lda, + const lapack_complex_float *tau, + lapack_complex_float *work, + lapack_int *lwork, + lapack_int *info); +void LAPACK_zungql(lapack_int *m, + lapack_int *n, + lapack_int *k, + lapack_complex_double *a, + lapack_int *lda, + const lapack_complex_double *tau, + lapack_complex_double *work, + lapack_int *lwork, + lapack_int *info); +void LAPACK_sormql(char *side, + char *trans, + lapack_int *m, + lapack_int *n, + lapack_int *k, + const float *a, + lapack_int *lda, + const float *tau, + float *c, + lapack_int *ldc, + float *work, + lapack_int *lwork, + lapack_int *info); +void LAPACK_dormql(char *side, + char *trans, + lapack_int *m, + lapack_int *n, + lapack_int *k, + const double *a, + lapack_int *lda, + const double *tau, + double *c, + lapack_int *ldc, + double *work, + lapack_int *lwork, + lapack_int *info); +void LAPACK_cunmql(char *side, + char *trans, + lapack_int *m, + lapack_int *n, + lapack_int *k, + const lapack_complex_float *a, + lapack_int *lda, + const lapack_complex_float *tau, + lapack_complex_float *c, + lapack_int *ldc, + lapack_complex_float *work, + lapack_int *lwork, + lapack_int *info); +void LAPACK_zunmql(char *side, + char *trans, + lapack_int *m, + lapack_int *n, + lapack_int *k, + const lapack_complex_double *a, + lapack_int *lda, + const lapack_complex_double *tau, + lapack_complex_double *c, + lapack_int *ldc, + lapack_complex_double *work, + lapack_int *lwork, + lapack_int *info); +void LAPACK_sgerqf(lapack_int *m, + lapack_int *n, + float *a, + lapack_int *lda, + float *tau, + float *work, + lapack_int *lwork, + lapack_int *info); +void LAPACK_dgerqf(lapack_int *m, + lapack_int *n, + double *a, + lapack_int *lda, + double *tau, + double *work, + lapack_int *lwork, + lapack_int *info); +void LAPACK_cgerqf(lapack_int *m, + lapack_int *n, + lapack_complex_float *a, + lapack_int *lda, + lapack_complex_float *tau, + lapack_complex_float *work, + lapack_int *lwork, + lapack_int *info); +void LAPACK_zgerqf(lapack_int *m, + lapack_int *n, + lapack_complex_double *a, + lapack_int *lda, + lapack_complex_double *tau, + lapack_complex_double *work, + lapack_int *lwork, + lapack_int *info); +void LAPACK_sorgrq(lapack_int *m, + lapack_int *n, + lapack_int *k, + float *a, + lapack_int *lda, + const float *tau, + float *work, + lapack_int *lwork, + lapack_int *info); +void LAPACK_dorgrq(lapack_int *m, + lapack_int *n, + lapack_int *k, + double *a, + lapack_int *lda, + const double *tau, + double *work, + lapack_int *lwork, + lapack_int *info); +void LAPACK_cungrq(lapack_int *m, + lapack_int *n, + lapack_int *k, + lapack_complex_float *a, + lapack_int *lda, + const lapack_complex_float *tau, + lapack_complex_float *work, + lapack_int *lwork, + lapack_int *info); +void LAPACK_zungrq(lapack_int *m, + lapack_int *n, + lapack_int *k, + lapack_complex_double *a, + lapack_int *lda, + const lapack_complex_double *tau, + lapack_complex_double *work, + lapack_int *lwork, + lapack_int *info); +void LAPACK_sormrq(char *side, + char *trans, + lapack_int *m, + lapack_int *n, + lapack_int *k, + const float *a, + lapack_int *lda, + const float *tau, + float *c, + lapack_int *ldc, + float *work, + lapack_int *lwork, + lapack_int *info); +void LAPACK_dormrq(char *side, + char *trans, + lapack_int *m, + lapack_int *n, + lapack_int *k, + const double *a, + lapack_int *lda, + const double *tau, + double *c, + lapack_int *ldc, + double *work, + lapack_int *lwork, + lapack_int *info); +void LAPACK_cunmrq(char *side, + char *trans, + lapack_int *m, + lapack_int *n, + lapack_int *k, + const lapack_complex_float *a, + lapack_int *lda, + const lapack_complex_float *tau, + lapack_complex_float *c, + lapack_int *ldc, + lapack_complex_float *work, + lapack_int *lwork, + lapack_int *info); +void LAPACK_zunmrq(char *side, + char *trans, + lapack_int *m, + lapack_int *n, + lapack_int *k, + const lapack_complex_double *a, + lapack_int *lda, + const lapack_complex_double *tau, + lapack_complex_double *c, + lapack_int *ldc, + lapack_complex_double *work, + lapack_int *lwork, + lapack_int *info); +void LAPACK_stzrzf(lapack_int *m, + lapack_int *n, + float *a, + lapack_int *lda, + float *tau, + float *work, + lapack_int *lwork, + lapack_int *info); +void LAPACK_dtzrzf(lapack_int *m, + lapack_int *n, + double *a, + lapack_int *lda, + double *tau, + double *work, + lapack_int *lwork, + lapack_int *info); +void LAPACK_ctzrzf(lapack_int *m, + lapack_int *n, + lapack_complex_float *a, + lapack_int *lda, + lapack_complex_float *tau, + lapack_complex_float *work, + lapack_int *lwork, + lapack_int *info); +void LAPACK_ztzrzf(lapack_int *m, + lapack_int *n, + lapack_complex_double *a, + lapack_int *lda, + lapack_complex_double *tau, + lapack_complex_double *work, + lapack_int *lwork, + lapack_int *info); +void LAPACK_sormrz(char *side, + char *trans, + lapack_int *m, + lapack_int *n, + lapack_int *k, + lapack_int *l, + const float *a, + lapack_int *lda, + const float *tau, + float *c, + lapack_int *ldc, + float *work, + lapack_int *lwork, + lapack_int *info); +void LAPACK_dormrz(char *side, + char *trans, + lapack_int *m, + lapack_int *n, + lapack_int *k, + lapack_int *l, + const double *a, + lapack_int *lda, + const double *tau, + double *c, + lapack_int *ldc, + double *work, + lapack_int *lwork, + lapack_int *info); +void LAPACK_cunmrz(char *side, + char *trans, + lapack_int *m, + lapack_int *n, + lapack_int *k, + lapack_int *l, + const lapack_complex_float *a, + lapack_int *lda, + const lapack_complex_float *tau, + lapack_complex_float *c, + lapack_int *ldc, + lapack_complex_float *work, + lapack_int *lwork, + lapack_int *info); +void LAPACK_zunmrz(char *side, + char *trans, + lapack_int *m, + lapack_int *n, + lapack_int *k, + lapack_int *l, + const lapack_complex_double *a, + lapack_int *lda, + const lapack_complex_double *tau, + lapack_complex_double *c, + lapack_int *ldc, + lapack_complex_double *work, + lapack_int *lwork, + lapack_int *info); +void LAPACK_sggqrf(lapack_int *n, + lapack_int *m, + lapack_int *p, + float *a, + lapack_int *lda, + float *taua, + float *b, + lapack_int *ldb, + float *taub, + float *work, + lapack_int *lwork, + lapack_int *info); +void LAPACK_dggqrf(lapack_int *n, + lapack_int *m, + lapack_int *p, + double *a, + lapack_int *lda, + double *taua, + double *b, + lapack_int *ldb, + double *taub, + double *work, + lapack_int *lwork, + lapack_int *info); +void LAPACK_cggqrf(lapack_int *n, + lapack_int *m, + lapack_int *p, + lapack_complex_float *a, + lapack_int *lda, + lapack_complex_float *taua, + lapack_complex_float *b, + lapack_int *ldb, + lapack_complex_float *taub, + lapack_complex_float *work, + lapack_int *lwork, + lapack_int *info); +void LAPACK_zggqrf(lapack_int *n, + lapack_int *m, + lapack_int *p, + lapack_complex_double *a, + lapack_int *lda, + lapack_complex_double *taua, + lapack_complex_double *b, + lapack_int *ldb, + lapack_complex_double *taub, + lapack_complex_double *work, + lapack_int *lwork, + lapack_int *info); +void LAPACK_sggrqf(lapack_int *m, + lapack_int *p, + lapack_int *n, + float *a, + lapack_int *lda, + float *taua, + float *b, + lapack_int *ldb, + float *taub, + float *work, + lapack_int *lwork, + lapack_int *info); +void LAPACK_dggrqf(lapack_int *m, + lapack_int *p, + lapack_int *n, + double *a, + lapack_int *lda, + double *taua, + double *b, + lapack_int *ldb, + double *taub, + double *work, + lapack_int *lwork, + lapack_int *info); +void LAPACK_cggrqf(lapack_int *m, + lapack_int *p, + lapack_int *n, + lapack_complex_float *a, + lapack_int *lda, + lapack_complex_float *taua, + lapack_complex_float *b, + lapack_int *ldb, + lapack_complex_float *taub, + lapack_complex_float *work, + lapack_int *lwork, + lapack_int *info); +void LAPACK_zggrqf(lapack_int *m, + lapack_int *p, + lapack_int *n, + lapack_complex_double *a, + lapack_int *lda, + lapack_complex_double *taua, + lapack_complex_double *b, + lapack_int *ldb, + lapack_complex_double *taub, + lapack_complex_double *work, + lapack_int *lwork, + lapack_int *info); +void LAPACK_sgebrd(lapack_int *m, + lapack_int *n, + float *a, + lapack_int *lda, + float *d, + float *e, + float *tauq, + float *taup, + float *work, + lapack_int *lwork, + lapack_int *info); +void LAPACK_dgebrd(lapack_int *m, + lapack_int *n, + double *a, + lapack_int *lda, + double *d, + double *e, + double *tauq, + double *taup, + double *work, + lapack_int *lwork, + lapack_int *info); +void LAPACK_cgebrd(lapack_int *m, + lapack_int *n, + lapack_complex_float *a, + lapack_int *lda, + float *d, + float *e, + lapack_complex_float *tauq, + lapack_complex_float *taup, + lapack_complex_float *work, + lapack_int *lwork, + lapack_int *info); +void LAPACK_zgebrd(lapack_int *m, + lapack_int *n, + lapack_complex_double *a, + lapack_int *lda, + double *d, + double *e, + lapack_complex_double *tauq, + lapack_complex_double *taup, + lapack_complex_double *work, + lapack_int *lwork, + lapack_int *info); +void LAPACK_sgbbrd(char *vect, + lapack_int *m, + lapack_int *n, + lapack_int *ncc, + lapack_int *kl, + lapack_int *ku, + float *ab, + lapack_int *ldab, + float *d, + float *e, + float *q, + lapack_int *ldq, + float *pt, + lapack_int *ldpt, + float *c, + lapack_int *ldc, + float *work, + lapack_int *info); +void LAPACK_dgbbrd(char *vect, + lapack_int *m, + lapack_int *n, + lapack_int *ncc, + lapack_int *kl, + lapack_int *ku, + double *ab, + lapack_int *ldab, + double *d, + double *e, + double *q, + lapack_int *ldq, + double *pt, + lapack_int *ldpt, + double *c, + lapack_int *ldc, + double *work, + lapack_int *info); +void LAPACK_cgbbrd(char *vect, + lapack_int *m, + lapack_int *n, + lapack_int *ncc, + lapack_int *kl, + lapack_int *ku, + lapack_complex_float *ab, + lapack_int *ldab, + float *d, + float *e, + lapack_complex_float *q, + lapack_int *ldq, + lapack_complex_float *pt, + lapack_int *ldpt, + lapack_complex_float *c, + lapack_int *ldc, + lapack_complex_float *work, + float *rwork, + lapack_int *info); +void LAPACK_zgbbrd(char *vect, + lapack_int *m, + lapack_int *n, + lapack_int *ncc, + lapack_int *kl, + lapack_int *ku, + lapack_complex_double *ab, + lapack_int *ldab, + double *d, + double *e, + lapack_complex_double *q, + lapack_int *ldq, + lapack_complex_double *pt, + lapack_int *ldpt, + lapack_complex_double *c, + lapack_int *ldc, + lapack_complex_double *work, + double *rwork, + lapack_int *info); +void LAPACK_sorgbr(char *vect, + lapack_int *m, + lapack_int *n, + lapack_int *k, + float *a, + lapack_int *lda, + const float *tau, + float *work, + lapack_int *lwork, + lapack_int *info); +void LAPACK_dorgbr(char *vect, + lapack_int *m, + lapack_int *n, + lapack_int *k, + double *a, + lapack_int *lda, + const double *tau, + double *work, + lapack_int *lwork, + lapack_int *info); +void LAPACK_sormbr(char *vect, + char *side, + char *trans, + lapack_int *m, + lapack_int *n, + lapack_int *k, + const float *a, + lapack_int *lda, + const float *tau, + float *c, + lapack_int *ldc, + float *work, + lapack_int *lwork, + lapack_int *info); +void LAPACK_dormbr(char *vect, + char *side, + char *trans, + lapack_int *m, + lapack_int *n, + lapack_int *k, + const double *a, + lapack_int *lda, + const double *tau, + double *c, + lapack_int *ldc, + double *work, + lapack_int *lwork, + lapack_int *info); +void LAPACK_cungbr(char *vect, + lapack_int *m, + lapack_int *n, + lapack_int *k, + lapack_complex_float *a, + lapack_int *lda, + const lapack_complex_float *tau, + lapack_complex_float *work, + lapack_int *lwork, + lapack_int *info); +void LAPACK_zungbr(char *vect, + lapack_int *m, + lapack_int *n, + lapack_int *k, + lapack_complex_double *a, + lapack_int *lda, + const lapack_complex_double *tau, + lapack_complex_double *work, + lapack_int *lwork, + lapack_int *info); +void LAPACK_cunmbr(char *vect, + char *side, + char *trans, + lapack_int *m, + lapack_int *n, + lapack_int *k, + const lapack_complex_float *a, + lapack_int *lda, + const lapack_complex_float *tau, + lapack_complex_float *c, + lapack_int *ldc, + lapack_complex_float *work, + lapack_int *lwork, + lapack_int *info); +void LAPACK_zunmbr(char *vect, + char *side, + char *trans, + lapack_int *m, + lapack_int *n, + lapack_int *k, + const lapack_complex_double *a, + lapack_int *lda, + const lapack_complex_double *tau, + lapack_complex_double *c, + lapack_int *ldc, + lapack_complex_double *work, + lapack_int *lwork, + lapack_int *info); +void LAPACK_sbdsqr(char *uplo, + lapack_int *n, + lapack_int *ncvt, + lapack_int *nru, + lapack_int *ncc, + float *d, + float *e, + float *vt, + lapack_int *ldvt, + float *u, + lapack_int *ldu, + float *c, + lapack_int *ldc, + float *work, + lapack_int *info); +void LAPACK_dbdsqr(char *uplo, + lapack_int *n, + lapack_int *ncvt, + lapack_int *nru, + lapack_int *ncc, + double *d, + double *e, + double *vt, + lapack_int *ldvt, + double *u, + lapack_int *ldu, + double *c, + lapack_int *ldc, + double *work, + lapack_int *info); +void LAPACK_cbdsqr(char *uplo, + lapack_int *n, + lapack_int *ncvt, + lapack_int *nru, + lapack_int *ncc, + float *d, + float *e, + lapack_complex_float *vt, + lapack_int *ldvt, + lapack_complex_float *u, + lapack_int *ldu, + lapack_complex_float *c, + lapack_int *ldc, + float *work, + lapack_int *info); +void LAPACK_zbdsqr(char *uplo, + lapack_int *n, + lapack_int *ncvt, + lapack_int *nru, + lapack_int *ncc, + double *d, + double *e, + lapack_complex_double *vt, + lapack_int *ldvt, + lapack_complex_double *u, + lapack_int *ldu, + lapack_complex_double *c, + lapack_int *ldc, + double *work, + lapack_int *info); +void LAPACK_sbdsdc(char *uplo, + char *compq, + lapack_int *n, + float *d, + float *e, + float *u, + lapack_int *ldu, + float *vt, + lapack_int *ldvt, + float *q, + lapack_int *iq, + float *work, + lapack_int *iwork, + lapack_int *info); +void LAPACK_dbdsdc(char *uplo, + char *compq, + lapack_int *n, + double *d, + double *e, + double *u, + lapack_int *ldu, + double *vt, + lapack_int *ldvt, + double *q, + lapack_int *iq, + double *work, + lapack_int *iwork, + lapack_int *info); +void LAPACK_ssytrd(char *uplo, + lapack_int *n, + float *a, + lapack_int *lda, + float *d, + float *e, + float *tau, + float *work, + lapack_int *lwork, + lapack_int *info); +void LAPACK_dsytrd(char *uplo, + lapack_int *n, + double *a, + lapack_int *lda, + double *d, + double *e, + double *tau, + double *work, + lapack_int *lwork, + lapack_int *info); +void LAPACK_sorgtr(char *uplo, + lapack_int *n, + float *a, + lapack_int *lda, + const float *tau, + float *work, + lapack_int *lwork, + lapack_int *info); +void LAPACK_dorgtr(char *uplo, + lapack_int *n, + double *a, + lapack_int *lda, + const double *tau, + double *work, + lapack_int *lwork, + lapack_int *info); +void LAPACK_sormtr(char *side, + char *uplo, + char *trans, + lapack_int *m, + lapack_int *n, + const float *a, + lapack_int *lda, + const float *tau, + float *c, + lapack_int *ldc, + float *work, + lapack_int *lwork, + lapack_int *info); +void LAPACK_dormtr(char *side, + char *uplo, + char *trans, + lapack_int *m, + lapack_int *n, + const double *a, + lapack_int *lda, + const double *tau, + double *c, + lapack_int *ldc, + double *work, + lapack_int *lwork, + lapack_int *info); +void LAPACK_chetrd(char *uplo, + lapack_int *n, + lapack_complex_float *a, + lapack_int *lda, + float *d, + float *e, + lapack_complex_float *tau, + lapack_complex_float *work, + lapack_int *lwork, + lapack_int *info); +void LAPACK_zhetrd(char *uplo, + lapack_int *n, + lapack_complex_double *a, + lapack_int *lda, + double *d, + double *e, + lapack_complex_double *tau, + lapack_complex_double *work, + lapack_int *lwork, + lapack_int *info); +void LAPACK_cungtr(char *uplo, + lapack_int *n, + lapack_complex_float *a, + lapack_int *lda, + const lapack_complex_float *tau, + lapack_complex_float *work, + lapack_int *lwork, + lapack_int *info); +void LAPACK_zungtr(char *uplo, + lapack_int *n, + lapack_complex_double *a, + lapack_int *lda, + const lapack_complex_double *tau, + lapack_complex_double *work, + lapack_int *lwork, + lapack_int *info); +void LAPACK_cunmtr(char *side, + char *uplo, + char *trans, + lapack_int *m, + lapack_int *n, + const lapack_complex_float *a, + lapack_int *lda, + const lapack_complex_float *tau, + lapack_complex_float *c, + lapack_int *ldc, + lapack_complex_float *work, + lapack_int *lwork, + lapack_int *info); +void LAPACK_zunmtr(char *side, + char *uplo, + char *trans, + lapack_int *m, + lapack_int *n, + const lapack_complex_double *a, + lapack_int *lda, + const lapack_complex_double *tau, + lapack_complex_double *c, + lapack_int *ldc, + lapack_complex_double *work, + lapack_int *lwork, + lapack_int *info); +void LAPACK_ssptrd(char *uplo, lapack_int *n, float *ap, float *d, float *e, float *tau, lapack_int *info); +void LAPACK_dsptrd(char *uplo, lapack_int *n, double *ap, double *d, double *e, double *tau, lapack_int *info); +void LAPACK_sopgtr(char *uplo, + lapack_int *n, + const float *ap, + const float *tau, + float *q, + lapack_int *ldq, + float *work, + lapack_int *info); +void LAPACK_dopgtr(char *uplo, + lapack_int *n, + const double *ap, + const double *tau, + double *q, + lapack_int *ldq, + double *work, + lapack_int *info); +void LAPACK_sopmtr(char *side, + char *uplo, + char *trans, + lapack_int *m, + lapack_int *n, + const float *ap, + const float *tau, + float *c, + lapack_int *ldc, + float *work, + lapack_int *info); +void LAPACK_dopmtr(char *side, + char *uplo, + char *trans, + lapack_int *m, + lapack_int *n, + const double *ap, + const double *tau, + double *c, + lapack_int *ldc, + double *work, + lapack_int *info); +void LAPACK_chptrd(char *uplo, + lapack_int *n, + lapack_complex_float *ap, + float *d, + float *e, + lapack_complex_float *tau, + lapack_int *info); +void LAPACK_zhptrd(char *uplo, + lapack_int *n, + lapack_complex_double *ap, + double *d, + double *e, + lapack_complex_double *tau, + lapack_int *info); +void LAPACK_cupgtr(char *uplo, + lapack_int *n, + const lapack_complex_float *ap, + const lapack_complex_float *tau, + lapack_complex_float *q, + lapack_int *ldq, + lapack_complex_float *work, + lapack_int *info); +void LAPACK_zupgtr(char *uplo, + lapack_int *n, + const lapack_complex_double *ap, + const lapack_complex_double *tau, + lapack_complex_double *q, + lapack_int *ldq, + lapack_complex_double *work, + lapack_int *info); +void LAPACK_cupmtr(char *side, + char *uplo, + char *trans, + lapack_int *m, + lapack_int *n, + const lapack_complex_float *ap, + const lapack_complex_float *tau, + lapack_complex_float *c, + lapack_int *ldc, + lapack_complex_float *work, + lapack_int *info); +void LAPACK_zupmtr(char *side, + char *uplo, + char *trans, + lapack_int *m, + lapack_int *n, + const lapack_complex_double *ap, + const lapack_complex_double *tau, + lapack_complex_double *c, + lapack_int *ldc, + lapack_complex_double *work, + lapack_int *info); +void LAPACK_ssbtrd(char *vect, + char *uplo, + lapack_int *n, + lapack_int *kd, + float *ab, + lapack_int *ldab, + float *d, + float *e, + float *q, + lapack_int *ldq, + float *work, + lapack_int *info); +void LAPACK_dsbtrd(char *vect, + char *uplo, + lapack_int *n, + lapack_int *kd, + double *ab, + lapack_int *ldab, + double *d, + double *e, + double *q, + lapack_int *ldq, + double *work, + lapack_int *info); +void LAPACK_chbtrd(char *vect, + char *uplo, + lapack_int *n, + lapack_int *kd, + lapack_complex_float *ab, + lapack_int *ldab, + float *d, + float *e, + lapack_complex_float *q, + lapack_int *ldq, + lapack_complex_float *work, + lapack_int *info); +void LAPACK_zhbtrd(char *vect, + char *uplo, + lapack_int *n, + lapack_int *kd, + lapack_complex_double *ab, + lapack_int *ldab, + double *d, + double *e, + lapack_complex_double *q, + lapack_int *ldq, + lapack_complex_double *work, + lapack_int *info); +void LAPACK_ssterf(lapack_int *n, float *d, float *e, lapack_int *info); +void LAPACK_dsterf(lapack_int *n, double *d, double *e, lapack_int *info); +void LAPACK_ssteqr(char *compz, + lapack_int *n, + float *d, + float *e, + float *z, + lapack_int *ldz, + float *work, + lapack_int *info); +void LAPACK_dsteqr(char *compz, + lapack_int *n, + double *d, + double *e, + double *z, + lapack_int *ldz, + double *work, + lapack_int *info); +void LAPACK_csteqr(char *compz, + lapack_int *n, + float *d, + float *e, + lapack_complex_float *z, + lapack_int *ldz, + float *work, + lapack_int *info); +void LAPACK_zsteqr(char *compz, + lapack_int *n, + double *d, + double *e, + lapack_complex_double *z, + lapack_int *ldz, + double *work, + lapack_int *info); +void LAPACK_sstemr(char *jobz, + char *range, + lapack_int *n, + float *d, + float *e, + float *vl, + float *vu, + lapack_int *il, + lapack_int *iu, + lapack_int *m, + float *w, + float *z, + lapack_int *ldz, + lapack_int *nzc, + lapack_int *isuppz, + lapack_logical *tryrac, + float *work, + lapack_int *lwork, + lapack_int *iwork, + lapack_int *liwork, + lapack_int *info); +void LAPACK_dstemr(char *jobz, + char *range, + lapack_int *n, + double *d, + double *e, + double *vl, + double *vu, + lapack_int *il, + lapack_int *iu, + lapack_int *m, + double *w, + double *z, + lapack_int *ldz, + lapack_int *nzc, + lapack_int *isuppz, + lapack_logical *tryrac, + double *work, + lapack_int *lwork, + lapack_int *iwork, + lapack_int *liwork, + lapack_int *info); +void LAPACK_cstemr(char *jobz, + char *range, + lapack_int *n, + float *d, + float *e, + float *vl, + float *vu, + lapack_int *il, + lapack_int *iu, + lapack_int *m, + float *w, + lapack_complex_float *z, + lapack_int *ldz, + lapack_int *nzc, + lapack_int *isuppz, + lapack_logical *tryrac, + float *work, + lapack_int *lwork, + lapack_int *iwork, + lapack_int *liwork, + lapack_int *info); +void LAPACK_zstemr(char *jobz, + char *range, + lapack_int *n, + double *d, + double *e, + double *vl, + double *vu, + lapack_int *il, + lapack_int *iu, + lapack_int *m, + double *w, + lapack_complex_double *z, + lapack_int *ldz, + lapack_int *nzc, + lapack_int *isuppz, + lapack_logical *tryrac, + double *work, + lapack_int *lwork, + lapack_int *iwork, + lapack_int *liwork, + lapack_int *info); +void LAPACK_sstedc(char *compz, + lapack_int *n, + float *d, + float *e, + float *z, + lapack_int *ldz, + float *work, + lapack_int *lwork, + lapack_int *iwork, + lapack_int *liwork, + lapack_int *info); +void LAPACK_dstedc(char *compz, + lapack_int *n, + double *d, + double *e, + double *z, + lapack_int *ldz, + double *work, + lapack_int *lwork, + lapack_int *iwork, + lapack_int *liwork, + lapack_int *info); +void LAPACK_cstedc(char *compz, + lapack_int *n, + float *d, + float *e, + lapack_complex_float *z, + lapack_int *ldz, + lapack_complex_float *work, + lapack_int *lwork, + float *rwork, + lapack_int *lrwork, + lapack_int *iwork, + lapack_int *liwork, + lapack_int *info); +void LAPACK_zstedc(char *compz, + lapack_int *n, + double *d, + double *e, + lapack_complex_double *z, + lapack_int *ldz, + lapack_complex_double *work, + lapack_int *lwork, + double *rwork, + lapack_int *lrwork, + lapack_int *iwork, + lapack_int *liwork, + lapack_int *info); +void LAPACK_sstegr(char *jobz, + char *range, + lapack_int *n, + float *d, + float *e, + float *vl, + float *vu, + lapack_int *il, + lapack_int *iu, + float *abstol, + lapack_int *m, + float *w, + float *z, + lapack_int *ldz, + lapack_int *isuppz, + float *work, + lapack_int *lwork, + lapack_int *iwork, + lapack_int *liwork, + lapack_int *info); +void LAPACK_dstegr(char *jobz, + char *range, + lapack_int *n, + double *d, + double *e, + double *vl, + double *vu, + lapack_int *il, + lapack_int *iu, + double *abstol, + lapack_int *m, + double *w, + double *z, + lapack_int *ldz, + lapack_int *isuppz, + double *work, + lapack_int *lwork, + lapack_int *iwork, + lapack_int *liwork, + lapack_int *info); +void LAPACK_cstegr(char *jobz, + char *range, + lapack_int *n, + float *d, + float *e, + float *vl, + float *vu, + lapack_int *il, + lapack_int *iu, + float *abstol, + lapack_int *m, + float *w, + lapack_complex_float *z, + lapack_int *ldz, + lapack_int *isuppz, + float *work, + lapack_int *lwork, + lapack_int *iwork, + lapack_int *liwork, + lapack_int *info); +void LAPACK_zstegr(char *jobz, + char *range, + lapack_int *n, + double *d, + double *e, + double *vl, + double *vu, + lapack_int *il, + lapack_int *iu, + double *abstol, + lapack_int *m, + double *w, + lapack_complex_double *z, + lapack_int *ldz, + lapack_int *isuppz, + double *work, + lapack_int *lwork, + lapack_int *iwork, + lapack_int *liwork, + lapack_int *info); +void LAPACK_spteqr(char *compz, + lapack_int *n, + float *d, + float *e, + float *z, + lapack_int *ldz, + float *work, + lapack_int *info); +void LAPACK_dpteqr(char *compz, + lapack_int *n, + double *d, + double *e, + double *z, + lapack_int *ldz, + double *work, + lapack_int *info); +void LAPACK_cpteqr(char *compz, + lapack_int *n, + float *d, + float *e, + lapack_complex_float *z, + lapack_int *ldz, + float *work, + lapack_int *info); +void LAPACK_zpteqr(char *compz, + lapack_int *n, + double *d, + double *e, + lapack_complex_double *z, + lapack_int *ldz, + double *work, + lapack_int *info); +void LAPACK_sstebz(char *range, + char *order, + lapack_int *n, + float *vl, + float *vu, + lapack_int *il, + lapack_int *iu, + float *abstol, + const float *d, + const float *e, + lapack_int *m, + lapack_int *nsplit, + float *w, + lapack_int *iblock, + lapack_int *isplit, + float *work, + lapack_int *iwork, + lapack_int *info); +void LAPACK_dstebz(char *range, + char *order, + lapack_int *n, + double *vl, + double *vu, + lapack_int *il, + lapack_int *iu, + double *abstol, + const double *d, + const double *e, + lapack_int *m, + lapack_int *nsplit, + double *w, + lapack_int *iblock, + lapack_int *isplit, + double *work, + lapack_int *iwork, + lapack_int *info); +void LAPACK_sstein(lapack_int *n, + const float *d, + const float *e, + lapack_int *m, + const float *w, + const lapack_int *iblock, + const lapack_int *isplit, + float *z, + lapack_int *ldz, + float *work, + lapack_int *iwork, + lapack_int *ifailv, + lapack_int *info); +void LAPACK_dstein(lapack_int *n, + const double *d, + const double *e, + lapack_int *m, + const double *w, + const lapack_int *iblock, + const lapack_int *isplit, + double *z, + lapack_int *ldz, + double *work, + lapack_int *iwork, + lapack_int *ifailv, + lapack_int *info); +void LAPACK_cstein(lapack_int *n, + const float *d, + const float *e, + lapack_int *m, + const float *w, + const lapack_int *iblock, + const lapack_int *isplit, + lapack_complex_float *z, + lapack_int *ldz, + float *work, + lapack_int *iwork, + lapack_int *ifailv, + lapack_int *info); +void LAPACK_zstein(lapack_int *n, + const double *d, + const double *e, + lapack_int *m, + const double *w, + const lapack_int *iblock, + const lapack_int *isplit, + lapack_complex_double *z, + lapack_int *ldz, + double *work, + lapack_int *iwork, + lapack_int *ifailv, + lapack_int *info); +void LAPACK_sdisna(char *job, lapack_int *m, lapack_int *n, const float *d, float *sep, lapack_int *info); +void LAPACK_ddisna(char *job, lapack_int *m, lapack_int *n, const double *d, double *sep, lapack_int *info); +void LAPACK_ssygst(lapack_int *itype, + char *uplo, + lapack_int *n, + float *a, + lapack_int *lda, + const float *b, + lapack_int *ldb, + lapack_int *info); +void LAPACK_dsygst(lapack_int *itype, + char *uplo, + lapack_int *n, + double *a, + lapack_int *lda, + const double *b, + lapack_int *ldb, + lapack_int *info); +void LAPACK_chegst(lapack_int *itype, + char *uplo, + lapack_int *n, + lapack_complex_float *a, + lapack_int *lda, + const lapack_complex_float *b, + lapack_int *ldb, + lapack_int *info); +void LAPACK_zhegst(lapack_int *itype, + char *uplo, + lapack_int *n, + lapack_complex_double *a, + lapack_int *lda, + const lapack_complex_double *b, + lapack_int *ldb, + lapack_int *info); +void LAPACK_sspgst(lapack_int *itype, char *uplo, lapack_int *n, float *ap, const float *bp, lapack_int *info); +void LAPACK_dspgst(lapack_int *itype, char *uplo, lapack_int *n, double *ap, const double *bp, lapack_int *info); +void LAPACK_chpgst(lapack_int *itype, + char *uplo, + lapack_int *n, + lapack_complex_float *ap, + const lapack_complex_float *bp, + lapack_int *info); +void LAPACK_zhpgst(lapack_int *itype, + char *uplo, + lapack_int *n, + lapack_complex_double *ap, + const lapack_complex_double *bp, + lapack_int *info); +void LAPACK_ssbgst(char *vect, + char *uplo, + lapack_int *n, + lapack_int *ka, + lapack_int *kb, + float *ab, + lapack_int *ldab, + const float *bb, + lapack_int *ldbb, + float *x, + lapack_int *ldx, + float *work, + lapack_int *info); +void LAPACK_dsbgst(char *vect, + char *uplo, + lapack_int *n, + lapack_int *ka, + lapack_int *kb, + double *ab, + lapack_int *ldab, + const double *bb, + lapack_int *ldbb, + double *x, + lapack_int *ldx, + double *work, + lapack_int *info); +void LAPACK_chbgst(char *vect, + char *uplo, + lapack_int *n, + lapack_int *ka, + lapack_int *kb, + lapack_complex_float *ab, + lapack_int *ldab, + const lapack_complex_float *bb, + lapack_int *ldbb, + lapack_complex_float *x, + lapack_int *ldx, + lapack_complex_float *work, + float *rwork, + lapack_int *info); +void LAPACK_zhbgst(char *vect, + char *uplo, + lapack_int *n, + lapack_int *ka, + lapack_int *kb, + lapack_complex_double *ab, + lapack_int *ldab, + const lapack_complex_double *bb, + lapack_int *ldbb, + lapack_complex_double *x, + lapack_int *ldx, + lapack_complex_double *work, + double *rwork, + lapack_int *info); +void LAPACK_spbstf(char *uplo, lapack_int *n, lapack_int *kb, float *bb, lapack_int *ldbb, lapack_int *info); +void LAPACK_dpbstf(char *uplo, lapack_int *n, lapack_int *kb, double *bb, lapack_int *ldbb, lapack_int *info); +void LAPACK_cpbstf(char *uplo, + lapack_int *n, + lapack_int *kb, + lapack_complex_float *bb, + lapack_int *ldbb, + lapack_int *info); +void LAPACK_zpbstf(char *uplo, + lapack_int *n, + lapack_int *kb, + lapack_complex_double *bb, + lapack_int *ldbb, + lapack_int *info); +void LAPACK_sgehrd(lapack_int *n, + lapack_int *ilo, + lapack_int *ihi, + float *a, + lapack_int *lda, + float *tau, + float *work, + lapack_int *lwork, + lapack_int *info); +void LAPACK_dgehrd(lapack_int *n, + lapack_int *ilo, + lapack_int *ihi, + double *a, + lapack_int *lda, + double *tau, + double *work, + lapack_int *lwork, + lapack_int *info); +void LAPACK_cgehrd(lapack_int *n, + lapack_int *ilo, + lapack_int *ihi, + lapack_complex_float *a, + lapack_int *lda, + lapack_complex_float *tau, + lapack_complex_float *work, + lapack_int *lwork, + lapack_int *info); +void LAPACK_zgehrd(lapack_int *n, + lapack_int *ilo, + lapack_int *ihi, + lapack_complex_double *a, + lapack_int *lda, + lapack_complex_double *tau, + lapack_complex_double *work, + lapack_int *lwork, + lapack_int *info); +void LAPACK_sorghr(lapack_int *n, + lapack_int *ilo, + lapack_int *ihi, + float *a, + lapack_int *lda, + const float *tau, + float *work, + lapack_int *lwork, + lapack_int *info); +void LAPACK_dorghr(lapack_int *n, + lapack_int *ilo, + lapack_int *ihi, + double *a, + lapack_int *lda, + const double *tau, + double *work, + lapack_int *lwork, + lapack_int *info); +void LAPACK_sormhr(char *side, + char *trans, + lapack_int *m, + lapack_int *n, + lapack_int *ilo, + lapack_int *ihi, + const float *a, + lapack_int *lda, + const float *tau, + float *c, + lapack_int *ldc, + float *work, + lapack_int *lwork, + lapack_int *info); +void LAPACK_dormhr(char *side, + char *trans, + lapack_int *m, + lapack_int *n, + lapack_int *ilo, + lapack_int *ihi, + const double *a, + lapack_int *lda, + const double *tau, + double *c, + lapack_int *ldc, + double *work, + lapack_int *lwork, + lapack_int *info); +void LAPACK_cunghr(lapack_int *n, + lapack_int *ilo, + lapack_int *ihi, + lapack_complex_float *a, + lapack_int *lda, + const lapack_complex_float *tau, + lapack_complex_float *work, + lapack_int *lwork, + lapack_int *info); +void LAPACK_zunghr(lapack_int *n, + lapack_int *ilo, + lapack_int *ihi, + lapack_complex_double *a, + lapack_int *lda, + const lapack_complex_double *tau, + lapack_complex_double *work, + lapack_int *lwork, + lapack_int *info); +void LAPACK_cunmhr(char *side, + char *trans, + lapack_int *m, + lapack_int *n, + lapack_int *ilo, + lapack_int *ihi, + const lapack_complex_float *a, + lapack_int *lda, + const lapack_complex_float *tau, + lapack_complex_float *c, + lapack_int *ldc, + lapack_complex_float *work, + lapack_int *lwork, + lapack_int *info); +void LAPACK_zunmhr(char *side, + char *trans, + lapack_int *m, + lapack_int *n, + lapack_int *ilo, + lapack_int *ihi, + const lapack_complex_double *a, + lapack_int *lda, + const lapack_complex_double *tau, + lapack_complex_double *c, + lapack_int *ldc, + lapack_complex_double *work, + lapack_int *lwork, + lapack_int *info); +void LAPACK_sgebal(char *job, + lapack_int *n, + float *a, + lapack_int *lda, + lapack_int *ilo, + lapack_int *ihi, + float *scale, + lapack_int *info); +void LAPACK_dgebal(char *job, + lapack_int *n, + double *a, + lapack_int *lda, + lapack_int *ilo, + lapack_int *ihi, + double *scale, + lapack_int *info); +void LAPACK_cgebal(char *job, + lapack_int *n, + lapack_complex_float *a, + lapack_int *lda, + lapack_int *ilo, + lapack_int *ihi, + float *scale, + lapack_int *info); +void LAPACK_zgebal(char *job, + lapack_int *n, + lapack_complex_double *a, + lapack_int *lda, + lapack_int *ilo, + lapack_int *ihi, + double *scale, + lapack_int *info); +void LAPACK_sgebak(char *job, + char *side, + lapack_int *n, + lapack_int *ilo, + lapack_int *ihi, + const float *scale, + lapack_int *m, + float *v, + lapack_int *ldv, + lapack_int *info); +void LAPACK_dgebak(char *job, + char *side, + lapack_int *n, + lapack_int *ilo, + lapack_int *ihi, + const double *scale, + lapack_int *m, + double *v, + lapack_int *ldv, + lapack_int *info); +void LAPACK_cgebak(char *job, + char *side, + lapack_int *n, + lapack_int *ilo, + lapack_int *ihi, + const float *scale, + lapack_int *m, + lapack_complex_float *v, + lapack_int *ldv, + lapack_int *info); +void LAPACK_zgebak(char *job, + char *side, + lapack_int *n, + lapack_int *ilo, + lapack_int *ihi, + const double *scale, + lapack_int *m, + lapack_complex_double *v, + lapack_int *ldv, + lapack_int *info); +void LAPACK_shseqr(char *job, + char *compz, + lapack_int *n, + lapack_int *ilo, + lapack_int *ihi, + float *h, + lapack_int *ldh, + float *wr, + float *wi, + float *z, + lapack_int *ldz, + float *work, + lapack_int *lwork, + lapack_int *info); +void LAPACK_dhseqr(char *job, + char *compz, + lapack_int *n, + lapack_int *ilo, + lapack_int *ihi, + double *h, + lapack_int *ldh, + double *wr, + double *wi, + double *z, + lapack_int *ldz, + double *work, + lapack_int *lwork, + lapack_int *info); +void LAPACK_chseqr(char *job, + char *compz, + lapack_int *n, + lapack_int *ilo, + lapack_int *ihi, + lapack_complex_float *h, + lapack_int *ldh, + lapack_complex_float *w, + lapack_complex_float *z, + lapack_int *ldz, + lapack_complex_float *work, + lapack_int *lwork, + lapack_int *info); +void LAPACK_zhseqr(char *job, + char *compz, + lapack_int *n, + lapack_int *ilo, + lapack_int *ihi, + lapack_complex_double *h, + lapack_int *ldh, + lapack_complex_double *w, + lapack_complex_double *z, + lapack_int *ldz, + lapack_complex_double *work, + lapack_int *lwork, + lapack_int *info); +void LAPACK_shsein(char *job, + char *eigsrc, + char *initv, + lapack_logical *select, + lapack_int *n, + const float *h, + lapack_int *ldh, + float *wr, + const float *wi, + float *vl, + lapack_int *ldvl, + float *vr, + lapack_int *ldvr, + lapack_int *mm, + lapack_int *m, + float *work, + lapack_int *ifaill, + lapack_int *ifailr, + lapack_int *info); +void LAPACK_dhsein(char *job, + char *eigsrc, + char *initv, + lapack_logical *select, + lapack_int *n, + const double *h, + lapack_int *ldh, + double *wr, + const double *wi, + double *vl, + lapack_int *ldvl, + double *vr, + lapack_int *ldvr, + lapack_int *mm, + lapack_int *m, + double *work, + lapack_int *ifaill, + lapack_int *ifailr, + lapack_int *info); +void LAPACK_chsein(char *job, + char *eigsrc, + char *initv, + const lapack_logical *select, + lapack_int *n, + const lapack_complex_float *h, + lapack_int *ldh, + lapack_complex_float *w, + lapack_complex_float *vl, + lapack_int *ldvl, + lapack_complex_float *vr, + lapack_int *ldvr, + lapack_int *mm, + lapack_int *m, + lapack_complex_float *work, + float *rwork, + lapack_int *ifaill, + lapack_int *ifailr, + lapack_int *info); +void LAPACK_zhsein(char *job, + char *eigsrc, + char *initv, + const lapack_logical *select, + lapack_int *n, + const lapack_complex_double *h, + lapack_int *ldh, + lapack_complex_double *w, + lapack_complex_double *vl, + lapack_int *ldvl, + lapack_complex_double *vr, + lapack_int *ldvr, + lapack_int *mm, + lapack_int *m, + lapack_complex_double *work, + double *rwork, + lapack_int *ifaill, + lapack_int *ifailr, + lapack_int *info); +void LAPACK_strevc(char *side, + char *howmny, + lapack_logical *select, + lapack_int *n, + const float *t, + lapack_int *ldt, + float *vl, + lapack_int *ldvl, + float *vr, + lapack_int *ldvr, + lapack_int *mm, + lapack_int *m, + float *work, + lapack_int *info); +void LAPACK_dtrevc(char *side, + char *howmny, + lapack_logical *select, + lapack_int *n, + const double *t, + lapack_int *ldt, + double *vl, + lapack_int *ldvl, + double *vr, + lapack_int *ldvr, + lapack_int *mm, + lapack_int *m, + double *work, + lapack_int *info); +void LAPACK_ctrevc(char *side, + char *howmny, + const lapack_logical *select, + lapack_int *n, + lapack_complex_float *t, + lapack_int *ldt, + lapack_complex_float *vl, + lapack_int *ldvl, + lapack_complex_float *vr, + lapack_int *ldvr, + lapack_int *mm, + lapack_int *m, + lapack_complex_float *work, + float *rwork, + lapack_int *info); +void LAPACK_ztrevc(char *side, + char *howmny, + const lapack_logical *select, + lapack_int *n, + lapack_complex_double *t, + lapack_int *ldt, + lapack_complex_double *vl, + lapack_int *ldvl, + lapack_complex_double *vr, + lapack_int *ldvr, + lapack_int *mm, + lapack_int *m, + lapack_complex_double *work, + double *rwork, + lapack_int *info); +void LAPACK_strsna(char *job, + char *howmny, + const lapack_logical *select, + lapack_int *n, + const float *t, + lapack_int *ldt, + const float *vl, + lapack_int *ldvl, + const float *vr, + lapack_int *ldvr, + float *s, + float *sep, + lapack_int *mm, + lapack_int *m, + float *work, + lapack_int *ldwork, + lapack_int *iwork, + lapack_int *info); +void LAPACK_dtrsna(char *job, + char *howmny, + const lapack_logical *select, + lapack_int *n, + const double *t, + lapack_int *ldt, + const double *vl, + lapack_int *ldvl, + const double *vr, + lapack_int *ldvr, + double *s, + double *sep, + lapack_int *mm, + lapack_int *m, + double *work, + lapack_int *ldwork, + lapack_int *iwork, + lapack_int *info); +void LAPACK_ctrsna(char *job, + char *howmny, + const lapack_logical *select, + lapack_int *n, + const lapack_complex_float *t, + lapack_int *ldt, + const lapack_complex_float *vl, + lapack_int *ldvl, + const lapack_complex_float *vr, + lapack_int *ldvr, + float *s, + float *sep, + lapack_int *mm, + lapack_int *m, + lapack_complex_float *work, + lapack_int *ldwork, + float *rwork, + lapack_int *info); +void LAPACK_ztrsna(char *job, + char *howmny, + const lapack_logical *select, + lapack_int *n, + const lapack_complex_double *t, + lapack_int *ldt, + const lapack_complex_double *vl, + lapack_int *ldvl, + const lapack_complex_double *vr, + lapack_int *ldvr, + double *s, + double *sep, + lapack_int *mm, + lapack_int *m, + lapack_complex_double *work, + lapack_int *ldwork, + double *rwork, + lapack_int *info); +void LAPACK_strexc(char *compq, + lapack_int *n, + float *t, + lapack_int *ldt, + float *q, + lapack_int *ldq, + lapack_int *ifst, + lapack_int *ilst, + float *work, + lapack_int *info); +void LAPACK_dtrexc(char *compq, + lapack_int *n, + double *t, + lapack_int *ldt, + double *q, + lapack_int *ldq, + lapack_int *ifst, + lapack_int *ilst, + double *work, + lapack_int *info); +void LAPACK_ctrexc(char *compq, + lapack_int *n, + lapack_complex_float *t, + lapack_int *ldt, + lapack_complex_float *q, + lapack_int *ldq, + lapack_int *ifst, + lapack_int *ilst, + lapack_int *info); +void LAPACK_ztrexc(char *compq, + lapack_int *n, + lapack_complex_double *t, + lapack_int *ldt, + lapack_complex_double *q, + lapack_int *ldq, + lapack_int *ifst, + lapack_int *ilst, + lapack_int *info); +void LAPACK_strsen(char *job, + char *compq, + const lapack_logical *select, + lapack_int *n, + float *t, + lapack_int *ldt, + float *q, + lapack_int *ldq, + float *wr, + float *wi, + lapack_int *m, + float *s, + float *sep, + float *work, + lapack_int *lwork, + lapack_int *iwork, + lapack_int *liwork, + lapack_int *info); +void LAPACK_dtrsen(char *job, + char *compq, + const lapack_logical *select, + lapack_int *n, + double *t, + lapack_int *ldt, + double *q, + lapack_int *ldq, + double *wr, + double *wi, + lapack_int *m, + double *s, + double *sep, + double *work, + lapack_int *lwork, + lapack_int *iwork, + lapack_int *liwork, + lapack_int *info); +void LAPACK_ctrsen(char *job, + char *compq, + const lapack_logical *select, + lapack_int *n, + lapack_complex_float *t, + lapack_int *ldt, + lapack_complex_float *q, + lapack_int *ldq, + lapack_complex_float *w, + lapack_int *m, + float *s, + float *sep, + lapack_complex_float *work, + lapack_int *lwork, + lapack_int *info); +void LAPACK_ztrsen(char *job, + char *compq, + const lapack_logical *select, + lapack_int *n, + lapack_complex_double *t, + lapack_int *ldt, + lapack_complex_double *q, + lapack_int *ldq, + lapack_complex_double *w, + lapack_int *m, + double *s, + double *sep, + lapack_complex_double *work, + lapack_int *lwork, + lapack_int *info); +void LAPACK_strsyl(char *trana, + char *tranb, + lapack_int *isgn, + lapack_int *m, + lapack_int *n, + const float *a, + lapack_int *lda, + const float *b, + lapack_int *ldb, + float *c, + lapack_int *ldc, + float *scale, + lapack_int *info); +void LAPACK_dtrsyl(char *trana, + char *tranb, + lapack_int *isgn, + lapack_int *m, + lapack_int *n, + const double *a, + lapack_int *lda, + const double *b, + lapack_int *ldb, + double *c, + lapack_int *ldc, + double *scale, + lapack_int *info); +void LAPACK_ctrsyl(char *trana, + char *tranb, + lapack_int *isgn, + lapack_int *m, + lapack_int *n, + const lapack_complex_float *a, + lapack_int *lda, + const lapack_complex_float *b, + lapack_int *ldb, + lapack_complex_float *c, + lapack_int *ldc, + float *scale, + lapack_int *info); +void LAPACK_ztrsyl(char *trana, + char *tranb, + lapack_int *isgn, + lapack_int *m, + lapack_int *n, + const lapack_complex_double *a, + lapack_int *lda, + const lapack_complex_double *b, + lapack_int *ldb, + lapack_complex_double *c, + lapack_int *ldc, + double *scale, + lapack_int *info); +void LAPACK_sgghrd(char *compq, + char *compz, + lapack_int *n, + lapack_int *ilo, + lapack_int *ihi, + float *a, + lapack_int *lda, + float *b, + lapack_int *ldb, + float *q, + lapack_int *ldq, + float *z, + lapack_int *ldz, + lapack_int *info); +void LAPACK_dgghrd(char *compq, + char *compz, + lapack_int *n, + lapack_int *ilo, + lapack_int *ihi, + double *a, + lapack_int *lda, + double *b, + lapack_int *ldb, + double *q, + lapack_int *ldq, + double *z, + lapack_int *ldz, + lapack_int *info); +void LAPACK_cgghrd(char *compq, + char *compz, + lapack_int *n, + lapack_int *ilo, + lapack_int *ihi, + lapack_complex_float *a, + lapack_int *lda, + lapack_complex_float *b, + lapack_int *ldb, + lapack_complex_float *q, + lapack_int *ldq, + lapack_complex_float *z, + lapack_int *ldz, + lapack_int *info); +void LAPACK_zgghrd(char *compq, + char *compz, + lapack_int *n, + lapack_int *ilo, + lapack_int *ihi, + lapack_complex_double *a, + lapack_int *lda, + lapack_complex_double *b, + lapack_int *ldb, + lapack_complex_double *q, + lapack_int *ldq, + lapack_complex_double *z, + lapack_int *ldz, + lapack_int *info); +void LAPACK_sggbal(char *job, + lapack_int *n, + float *a, + lapack_int *lda, + float *b, + lapack_int *ldb, + lapack_int *ilo, + lapack_int *ihi, + float *lscale, + float *rscale, + float *work, + lapack_int *info); +void LAPACK_dggbal(char *job, + lapack_int *n, + double *a, + lapack_int *lda, + double *b, + lapack_int *ldb, + lapack_int *ilo, + lapack_int *ihi, + double *lscale, + double *rscale, + double *work, + lapack_int *info); +void LAPACK_cggbal(char *job, + lapack_int *n, + lapack_complex_float *a, + lapack_int *lda, + lapack_complex_float *b, + lapack_int *ldb, + lapack_int *ilo, + lapack_int *ihi, + float *lscale, + float *rscale, + float *work, + lapack_int *info); +void LAPACK_zggbal(char *job, + lapack_int *n, + lapack_complex_double *a, + lapack_int *lda, + lapack_complex_double *b, + lapack_int *ldb, + lapack_int *ilo, + lapack_int *ihi, + double *lscale, + double *rscale, + double *work, + lapack_int *info); +void LAPACK_sggbak(char *job, + char *side, + lapack_int *n, + lapack_int *ilo, + lapack_int *ihi, + const float *lscale, + const float *rscale, + lapack_int *m, + float *v, + lapack_int *ldv, + lapack_int *info); +void LAPACK_dggbak(char *job, + char *side, + lapack_int *n, + lapack_int *ilo, + lapack_int *ihi, + const double *lscale, + const double *rscale, + lapack_int *m, + double *v, + lapack_int *ldv, + lapack_int *info); +void LAPACK_cggbak(char *job, + char *side, + lapack_int *n, + lapack_int *ilo, + lapack_int *ihi, + const float *lscale, + const float *rscale, + lapack_int *m, + lapack_complex_float *v, + lapack_int *ldv, + lapack_int *info); +void LAPACK_zggbak(char *job, + char *side, + lapack_int *n, + lapack_int *ilo, + lapack_int *ihi, + const double *lscale, + const double *rscale, + lapack_int *m, + lapack_complex_double *v, + lapack_int *ldv, + lapack_int *info); +void LAPACK_shgeqz(char *job, + char *compq, + char *compz, + lapack_int *n, + lapack_int *ilo, + lapack_int *ihi, + float *h, + lapack_int *ldh, + float *t, + lapack_int *ldt, + float *alphar, + float *alphai, + float *beta, + float *q, + lapack_int *ldq, + float *z, + lapack_int *ldz, + float *work, + lapack_int *lwork, + lapack_int *info); +void LAPACK_dhgeqz(char *job, + char *compq, + char *compz, + lapack_int *n, + lapack_int *ilo, + lapack_int *ihi, + double *h, + lapack_int *ldh, + double *t, + lapack_int *ldt, + double *alphar, + double *alphai, + double *beta, + double *q, + lapack_int *ldq, + double *z, + lapack_int *ldz, + double *work, + lapack_int *lwork, + lapack_int *info); +void LAPACK_chgeqz(char *job, + char *compq, + char *compz, + lapack_int *n, + lapack_int *ilo, + lapack_int *ihi, + lapack_complex_float *h, + lapack_int *ldh, + lapack_complex_float *t, + lapack_int *ldt, + lapack_complex_float *alpha, + lapack_complex_float *beta, + lapack_complex_float *q, + lapack_int *ldq, + lapack_complex_float *z, + lapack_int *ldz, + lapack_complex_float *work, + lapack_int *lwork, + float *rwork, + lapack_int *info); +void LAPACK_zhgeqz(char *job, + char *compq, + char *compz, + lapack_int *n, + lapack_int *ilo, + lapack_int *ihi, + lapack_complex_double *h, + lapack_int *ldh, + lapack_complex_double *t, + lapack_int *ldt, + lapack_complex_double *alpha, + lapack_complex_double *beta, + lapack_complex_double *q, + lapack_int *ldq, + lapack_complex_double *z, + lapack_int *ldz, + lapack_complex_double *work, + lapack_int *lwork, + double *rwork, + lapack_int *info); +void LAPACK_stgevc(char *side, + char *howmny, + const lapack_logical *select, + lapack_int *n, + const float *s, + lapack_int *lds, + const float *p, + lapack_int *ldp, + float *vl, + lapack_int *ldvl, + float *vr, + lapack_int *ldvr, + lapack_int *mm, + lapack_int *m, + float *work, + lapack_int *info); +void LAPACK_dtgevc(char *side, + char *howmny, + const lapack_logical *select, + lapack_int *n, + const double *s, + lapack_int *lds, + const double *p, + lapack_int *ldp, + double *vl, + lapack_int *ldvl, + double *vr, + lapack_int *ldvr, + lapack_int *mm, + lapack_int *m, + double *work, + lapack_int *info); +void LAPACK_ctgevc(char *side, + char *howmny, + const lapack_logical *select, + lapack_int *n, + const lapack_complex_float *s, + lapack_int *lds, + const lapack_complex_float *p, + lapack_int *ldp, + lapack_complex_float *vl, + lapack_int *ldvl, + lapack_complex_float *vr, + lapack_int *ldvr, + lapack_int *mm, + lapack_int *m, + lapack_complex_float *work, + float *rwork, + lapack_int *info); +void LAPACK_ztgevc(char *side, + char *howmny, + const lapack_logical *select, + lapack_int *n, + const lapack_complex_double *s, + lapack_int *lds, + const lapack_complex_double *p, + lapack_int *ldp, + lapack_complex_double *vl, + lapack_int *ldvl, + lapack_complex_double *vr, + lapack_int *ldvr, + lapack_int *mm, + lapack_int *m, + lapack_complex_double *work, + double *rwork, + lapack_int *info); +void LAPACK_stgexc(lapack_logical *wantq, + lapack_logical *wantz, + lapack_int *n, + float *a, + lapack_int *lda, + float *b, + lapack_int *ldb, + float *q, + lapack_int *ldq, + float *z, + lapack_int *ldz, + lapack_int *ifst, + lapack_int *ilst, + float *work, + lapack_int *lwork, + lapack_int *info); +void LAPACK_dtgexc(lapack_logical *wantq, + lapack_logical *wantz, + lapack_int *n, + double *a, + lapack_int *lda, + double *b, + lapack_int *ldb, + double *q, + lapack_int *ldq, + double *z, + lapack_int *ldz, + lapack_int *ifst, + lapack_int *ilst, + double *work, + lapack_int *lwork, + lapack_int *info); +void LAPACK_ctgexc(lapack_logical *wantq, + lapack_logical *wantz, + lapack_int *n, + lapack_complex_float *a, + lapack_int *lda, + lapack_complex_float *b, + lapack_int *ldb, + lapack_complex_float *q, + lapack_int *ldq, + lapack_complex_float *z, + lapack_int *ldz, + lapack_int *ifst, + lapack_int *ilst, + lapack_int *info); +void LAPACK_ztgexc(lapack_logical *wantq, + lapack_logical *wantz, + lapack_int *n, + lapack_complex_double *a, + lapack_int *lda, + lapack_complex_double *b, + lapack_int *ldb, + lapack_complex_double *q, + lapack_int *ldq, + lapack_complex_double *z, + lapack_int *ldz, + lapack_int *ifst, + lapack_int *ilst, + lapack_int *info); +void LAPACK_stgsen(lapack_int *ijob, + lapack_logical *wantq, + lapack_logical *wantz, + const lapack_logical *select, + lapack_int *n, + float *a, + lapack_int *lda, + float *b, + lapack_int *ldb, + float *alphar, + float *alphai, + float *beta, + float *q, + lapack_int *ldq, + float *z, + lapack_int *ldz, + lapack_int *m, + float *pl, + float *pr, + float *dif, + float *work, + lapack_int *lwork, + lapack_int *iwork, + lapack_int *liwork, + lapack_int *info); +void LAPACK_dtgsen(lapack_int *ijob, + lapack_logical *wantq, + lapack_logical *wantz, + const lapack_logical *select, + lapack_int *n, + double *a, + lapack_int *lda, + double *b, + lapack_int *ldb, + double *alphar, + double *alphai, + double *beta, + double *q, + lapack_int *ldq, + double *z, + lapack_int *ldz, + lapack_int *m, + double *pl, + double *pr, + double *dif, + double *work, + lapack_int *lwork, + lapack_int *iwork, + lapack_int *liwork, + lapack_int *info); +void LAPACK_ctgsen(lapack_int *ijob, + lapack_logical *wantq, + lapack_logical *wantz, + const lapack_logical *select, + lapack_int *n, + lapack_complex_float *a, + lapack_int *lda, + lapack_complex_float *b, + lapack_int *ldb, + lapack_complex_float *alpha, + lapack_complex_float *beta, + lapack_complex_float *q, + lapack_int *ldq, + lapack_complex_float *z, + lapack_int *ldz, + lapack_int *m, + float *pl, + float *pr, + float *dif, + lapack_complex_float *work, + lapack_int *lwork, + lapack_int *iwork, + lapack_int *liwork, + lapack_int *info); +void LAPACK_ztgsen(lapack_int *ijob, + lapack_logical *wantq, + lapack_logical *wantz, + const lapack_logical *select, + lapack_int *n, + lapack_complex_double *a, + lapack_int *lda, + lapack_complex_double *b, + lapack_int *ldb, + lapack_complex_double *alpha, + lapack_complex_double *beta, + lapack_complex_double *q, + lapack_int *ldq, + lapack_complex_double *z, + lapack_int *ldz, + lapack_int *m, + double *pl, + double *pr, + double *dif, + lapack_complex_double *work, + lapack_int *lwork, + lapack_int *iwork, + lapack_int *liwork, + lapack_int *info); +void LAPACK_stgsyl(char *trans, + lapack_int *ijob, + lapack_int *m, + lapack_int *n, + const float *a, + lapack_int *lda, + const float *b, + lapack_int *ldb, + float *c, + lapack_int *ldc, + const float *d, + lapack_int *ldd, + const float *e, + lapack_int *lde, + float *f, + lapack_int *ldf, + float *scale, + float *dif, + float *work, + lapack_int *lwork, + lapack_int *iwork, + lapack_int *info); +void LAPACK_dtgsyl(char *trans, + lapack_int *ijob, + lapack_int *m, + lapack_int *n, + const double *a, + lapack_int *lda, + const double *b, + lapack_int *ldb, + double *c, + lapack_int *ldc, + const double *d, + lapack_int *ldd, + const double *e, + lapack_int *lde, + double *f, + lapack_int *ldf, + double *scale, + double *dif, + double *work, + lapack_int *lwork, + lapack_int *iwork, + lapack_int *info); +void LAPACK_ctgsyl(char *trans, + lapack_int *ijob, + lapack_int *m, + lapack_int *n, + const lapack_complex_float *a, + lapack_int *lda, + const lapack_complex_float *b, + lapack_int *ldb, + lapack_complex_float *c, + lapack_int *ldc, + const lapack_complex_float *d, + lapack_int *ldd, + const lapack_complex_float *e, + lapack_int *lde, + lapack_complex_float *f, + lapack_int *ldf, + float *scale, + float *dif, + lapack_complex_float *work, + lapack_int *lwork, + lapack_int *iwork, + lapack_int *info); +void LAPACK_ztgsyl(char *trans, + lapack_int *ijob, + lapack_int *m, + lapack_int *n, + const lapack_complex_double *a, + lapack_int *lda, + const lapack_complex_double *b, + lapack_int *ldb, + lapack_complex_double *c, + lapack_int *ldc, + const lapack_complex_double *d, + lapack_int *ldd, + const lapack_complex_double *e, + lapack_int *lde, + lapack_complex_double *f, + lapack_int *ldf, + double *scale, + double *dif, + lapack_complex_double *work, + lapack_int *lwork, + lapack_int *iwork, + lapack_int *info); +void LAPACK_stgsna(char *job, + char *howmny, + const lapack_logical *select, + lapack_int *n, + const float *a, + lapack_int *lda, + const float *b, + lapack_int *ldb, + const float *vl, + lapack_int *ldvl, + const float *vr, + lapack_int *ldvr, + float *s, + float *dif, + lapack_int *mm, + lapack_int *m, + float *work, + lapack_int *lwork, + lapack_int *iwork, + lapack_int *info); +void LAPACK_dtgsna(char *job, + char *howmny, + const lapack_logical *select, + lapack_int *n, + const double *a, + lapack_int *lda, + const double *b, + lapack_int *ldb, + const double *vl, + lapack_int *ldvl, + const double *vr, + lapack_int *ldvr, + double *s, + double *dif, + lapack_int *mm, + lapack_int *m, + double *work, + lapack_int *lwork, + lapack_int *iwork, + lapack_int *info); +void LAPACK_ctgsna(char *job, + char *howmny, + const lapack_logical *select, + lapack_int *n, + const lapack_complex_float *a, + lapack_int *lda, + const lapack_complex_float *b, + lapack_int *ldb, + const lapack_complex_float *vl, + lapack_int *ldvl, + const lapack_complex_float *vr, + lapack_int *ldvr, + float *s, + float *dif, + lapack_int *mm, + lapack_int *m, + lapack_complex_float *work, + lapack_int *lwork, + lapack_int *iwork, + lapack_int *info); +void LAPACK_ztgsna(char *job, + char *howmny, + const lapack_logical *select, + lapack_int *n, + const lapack_complex_double *a, + lapack_int *lda, + const lapack_complex_double *b, + lapack_int *ldb, + const lapack_complex_double *vl, + lapack_int *ldvl, + const lapack_complex_double *vr, + lapack_int *ldvr, + double *s, + double *dif, + lapack_int *mm, + lapack_int *m, + lapack_complex_double *work, + lapack_int *lwork, + lapack_int *iwork, + lapack_int *info); +void LAPACK_sggsvp(char *jobu, + char *jobv, + char *jobq, + lapack_int *m, + lapack_int *p, + lapack_int *n, + float *a, + lapack_int *lda, + float *b, + lapack_int *ldb, + float *tola, + float *tolb, + lapack_int *k, + lapack_int *l, + float *u, + lapack_int *ldu, + float *v, + lapack_int *ldv, + float *q, + lapack_int *ldq, + lapack_int *iwork, + float *tau, + float *work, + lapack_int *info); +void LAPACK_dggsvp(char *jobu, + char *jobv, + char *jobq, + lapack_int *m, + lapack_int *p, + lapack_int *n, + double *a, + lapack_int *lda, + double *b, + lapack_int *ldb, + double *tola, + double *tolb, + lapack_int *k, + lapack_int *l, + double *u, + lapack_int *ldu, + double *v, + lapack_int *ldv, + double *q, + lapack_int *ldq, + lapack_int *iwork, + double *tau, + double *work, + lapack_int *info); +void LAPACK_cggsvp(char *jobu, + char *jobv, + char *jobq, + lapack_int *m, + lapack_int *p, + lapack_int *n, + lapack_complex_float *a, + lapack_int *lda, + lapack_complex_float *b, + lapack_int *ldb, + float *tola, + float *tolb, + lapack_int *k, + lapack_int *l, + lapack_complex_float *u, + lapack_int *ldu, + lapack_complex_float *v, + lapack_int *ldv, + lapack_complex_float *q, + lapack_int *ldq, + lapack_int *iwork, + float *rwork, + lapack_complex_float *tau, + lapack_complex_float *work, + lapack_int *info); +void LAPACK_zggsvp(char *jobu, + char *jobv, + char *jobq, + lapack_int *m, + lapack_int *p, + lapack_int *n, + lapack_complex_double *a, + lapack_int *lda, + lapack_complex_double *b, + lapack_int *ldb, + double *tola, + double *tolb, + lapack_int *k, + lapack_int *l, + lapack_complex_double *u, + lapack_int *ldu, + lapack_complex_double *v, + lapack_int *ldv, + lapack_complex_double *q, + lapack_int *ldq, + lapack_int *iwork, + double *rwork, + lapack_complex_double *tau, + lapack_complex_double *work, + lapack_int *info); +void LAPACK_stgsja(char *jobu, + char *jobv, + char *jobq, + lapack_int *m, + lapack_int *p, + lapack_int *n, + lapack_int *k, + lapack_int *l, + float *a, + lapack_int *lda, + float *b, + lapack_int *ldb, + float *tola, + float *tolb, + float *alpha, + float *beta, + float *u, + lapack_int *ldu, + float *v, + lapack_int *ldv, + float *q, + lapack_int *ldq, + float *work, + lapack_int *ncycle, + lapack_int *info); +void LAPACK_dtgsja(char *jobu, + char *jobv, + char *jobq, + lapack_int *m, + lapack_int *p, + lapack_int *n, + lapack_int *k, + lapack_int *l, + double *a, + lapack_int *lda, + double *b, + lapack_int *ldb, + double *tola, + double *tolb, + double *alpha, + double *beta, + double *u, + lapack_int *ldu, + double *v, + lapack_int *ldv, + double *q, + lapack_int *ldq, + double *work, + lapack_int *ncycle, + lapack_int *info); +void LAPACK_ctgsja(char *jobu, + char *jobv, + char *jobq, + lapack_int *m, + lapack_int *p, + lapack_int *n, + lapack_int *k, + lapack_int *l, + lapack_complex_float *a, + lapack_int *lda, + lapack_complex_float *b, + lapack_int *ldb, + float *tola, + float *tolb, + float *alpha, + float *beta, + lapack_complex_float *u, + lapack_int *ldu, + lapack_complex_float *v, + lapack_int *ldv, + lapack_complex_float *q, + lapack_int *ldq, + lapack_complex_float *work, + lapack_int *ncycle, + lapack_int *info); +void LAPACK_ztgsja(char *jobu, + char *jobv, + char *jobq, + lapack_int *m, + lapack_int *p, + lapack_int *n, + lapack_int *k, + lapack_int *l, + lapack_complex_double *a, + lapack_int *lda, + lapack_complex_double *b, + lapack_int *ldb, + double *tola, + double *tolb, + double *alpha, + double *beta, + lapack_complex_double *u, + lapack_int *ldu, + lapack_complex_double *v, + lapack_int *ldv, + lapack_complex_double *q, + lapack_int *ldq, + lapack_complex_double *work, + lapack_int *ncycle, + lapack_int *info); +void LAPACK_sgels(char *trans, + lapack_int *m, + lapack_int *n, + lapack_int *nrhs, + float *a, + lapack_int *lda, + float *b, + lapack_int *ldb, + float *work, + lapack_int *lwork, + lapack_int *info); +void LAPACK_dgels(char *trans, + lapack_int *m, + lapack_int *n, + lapack_int *nrhs, + double *a, + lapack_int *lda, + double *b, + lapack_int *ldb, + double *work, + lapack_int *lwork, + lapack_int *info); +void LAPACK_cgels(char *trans, + lapack_int *m, + lapack_int *n, + lapack_int *nrhs, + lapack_complex_float *a, + lapack_int *lda, + lapack_complex_float *b, + lapack_int *ldb, + lapack_complex_float *work, + lapack_int *lwork, + lapack_int *info); +void LAPACK_zgels(char *trans, + lapack_int *m, + lapack_int *n, + lapack_int *nrhs, + lapack_complex_double *a, + lapack_int *lda, + lapack_complex_double *b, + lapack_int *ldb, + lapack_complex_double *work, + lapack_int *lwork, + lapack_int *info); +void LAPACK_sgelsy(lapack_int *m, + lapack_int *n, + lapack_int *nrhs, + float *a, + lapack_int *lda, + float *b, + lapack_int *ldb, + lapack_int *jpvt, + float *rcond, + lapack_int *rank, + float *work, + lapack_int *lwork, + lapack_int *info); +void LAPACK_dgelsy(lapack_int *m, + lapack_int *n, + lapack_int *nrhs, + double *a, + lapack_int *lda, + double *b, + lapack_int *ldb, + lapack_int *jpvt, + double *rcond, + lapack_int *rank, + double *work, + lapack_int *lwork, + lapack_int *info); +void LAPACK_cgelsy(lapack_int *m, + lapack_int *n, + lapack_int *nrhs, + lapack_complex_float *a, + lapack_int *lda, + lapack_complex_float *b, + lapack_int *ldb, + lapack_int *jpvt, + float *rcond, + lapack_int *rank, + lapack_complex_float *work, + lapack_int *lwork, + float *rwork, + lapack_int *info); +void LAPACK_zgelsy(lapack_int *m, + lapack_int *n, + lapack_int *nrhs, + lapack_complex_double *a, + lapack_int *lda, + lapack_complex_double *b, + lapack_int *ldb, + lapack_int *jpvt, + double *rcond, + lapack_int *rank, + lapack_complex_double *work, + lapack_int *lwork, + double *rwork, + lapack_int *info); +void LAPACK_sgelss(lapack_int *m, + lapack_int *n, + lapack_int *nrhs, + float *a, + lapack_int *lda, + float *b, + lapack_int *ldb, + float *s, + float *rcond, + lapack_int *rank, + float *work, + lapack_int *lwork, + lapack_int *info); +void LAPACK_dgelss(lapack_int *m, + lapack_int *n, + lapack_int *nrhs, + double *a, + lapack_int *lda, + double *b, + lapack_int *ldb, + double *s, + double *rcond, + lapack_int *rank, + double *work, + lapack_int *lwork, + lapack_int *info); +void LAPACK_cgelss(lapack_int *m, + lapack_int *n, + lapack_int *nrhs, + lapack_complex_float *a, + lapack_int *lda, + lapack_complex_float *b, + lapack_int *ldb, + float *s, + float *rcond, + lapack_int *rank, + lapack_complex_float *work, + lapack_int *lwork, + float *rwork, + lapack_int *info); +void LAPACK_zgelss(lapack_int *m, + lapack_int *n, + lapack_int *nrhs, + lapack_complex_double *a, + lapack_int *lda, + lapack_complex_double *b, + lapack_int *ldb, + double *s, + double *rcond, + lapack_int *rank, + lapack_complex_double *work, + lapack_int *lwork, + double *rwork, + lapack_int *info); +void LAPACK_sgelsd(lapack_int *m, + lapack_int *n, + lapack_int *nrhs, + float *a, + lapack_int *lda, + float *b, + lapack_int *ldb, + float *s, + float *rcond, + lapack_int *rank, + float *work, + lapack_int *lwork, + lapack_int *iwork, + lapack_int *info); +void LAPACK_dgelsd(lapack_int *m, + lapack_int *n, + lapack_int *nrhs, + double *a, + lapack_int *lda, + double *b, + lapack_int *ldb, + double *s, + double *rcond, + lapack_int *rank, + double *work, + lapack_int *lwork, + lapack_int *iwork, + lapack_int *info); +void LAPACK_cgelsd(lapack_int *m, + lapack_int *n, + lapack_int *nrhs, + lapack_complex_float *a, + lapack_int *lda, + lapack_complex_float *b, + lapack_int *ldb, + float *s, + float *rcond, + lapack_int *rank, + lapack_complex_float *work, + lapack_int *lwork, + float *rwork, + lapack_int *iwork, + lapack_int *info); +void LAPACK_zgelsd(lapack_int *m, + lapack_int *n, + lapack_int *nrhs, + lapack_complex_double *a, + lapack_int *lda, + lapack_complex_double *b, + lapack_int *ldb, + double *s, + double *rcond, + lapack_int *rank, + lapack_complex_double *work, + lapack_int *lwork, + double *rwork, + lapack_int *iwork, + lapack_int *info); +void LAPACK_sgglse(lapack_int *m, + lapack_int *n, + lapack_int *p, + float *a, + lapack_int *lda, + float *b, + lapack_int *ldb, + float *c, + float *d, + float *x, + float *work, + lapack_int *lwork, + lapack_int *info); +void LAPACK_dgglse(lapack_int *m, + lapack_int *n, + lapack_int *p, + double *a, + lapack_int *lda, + double *b, + lapack_int *ldb, + double *c, + double *d, + double *x, + double *work, + lapack_int *lwork, + lapack_int *info); +void LAPACK_cgglse(lapack_int *m, + lapack_int *n, + lapack_int *p, + lapack_complex_float *a, + lapack_int *lda, + lapack_complex_float *b, + lapack_int *ldb, + lapack_complex_float *c, + lapack_complex_float *d, + lapack_complex_float *x, + lapack_complex_float *work, + lapack_int *lwork, + lapack_int *info); +void LAPACK_zgglse(lapack_int *m, + lapack_int *n, + lapack_int *p, + lapack_complex_double *a, + lapack_int *lda, + lapack_complex_double *b, + lapack_int *ldb, + lapack_complex_double *c, + lapack_complex_double *d, + lapack_complex_double *x, + lapack_complex_double *work, + lapack_int *lwork, + lapack_int *info); +void LAPACK_sggglm(lapack_int *n, + lapack_int *m, + lapack_int *p, + float *a, + lapack_int *lda, + float *b, + lapack_int *ldb, + float *d, + float *x, + float *y, + float *work, + lapack_int *lwork, + lapack_int *info); +void LAPACK_dggglm(lapack_int *n, + lapack_int *m, + lapack_int *p, + double *a, + lapack_int *lda, + double *b, + lapack_int *ldb, + double *d, + double *x, + double *y, + double *work, + lapack_int *lwork, + lapack_int *info); +void LAPACK_cggglm(lapack_int *n, + lapack_int *m, + lapack_int *p, + lapack_complex_float *a, + lapack_int *lda, + lapack_complex_float *b, + lapack_int *ldb, + lapack_complex_float *d, + lapack_complex_float *x, + lapack_complex_float *y, + lapack_complex_float *work, + lapack_int *lwork, + lapack_int *info); +void LAPACK_zggglm(lapack_int *n, + lapack_int *m, + lapack_int *p, + lapack_complex_double *a, + lapack_int *lda, + lapack_complex_double *b, + lapack_int *ldb, + lapack_complex_double *d, + lapack_complex_double *x, + lapack_complex_double *y, + lapack_complex_double *work, + lapack_int *lwork, + lapack_int *info); +void LAPACK_ssyev(char *jobz, + char *uplo, + lapack_int *n, + float *a, + lapack_int *lda, + float *w, + float *work, + lapack_int *lwork, + lapack_int *info); +void LAPACK_dsyev(char *jobz, + char *uplo, + lapack_int *n, + double *a, + lapack_int *lda, + double *w, + double *work, + lapack_int *lwork, + lapack_int *info); +void LAPACK_cheev(char *jobz, + char *uplo, + lapack_int *n, + lapack_complex_float *a, + lapack_int *lda, + float *w, + lapack_complex_float *work, + lapack_int *lwork, + float *rwork, + lapack_int *info); +void LAPACK_zheev(char *jobz, + char *uplo, + lapack_int *n, + lapack_complex_double *a, + lapack_int *lda, + double *w, + lapack_complex_double *work, + lapack_int *lwork, + double *rwork, + lapack_int *info); +void LAPACK_ssyevd(char *jobz, + char *uplo, + lapack_int *n, + float *a, + lapack_int *lda, + float *w, + float *work, + lapack_int *lwork, + lapack_int *iwork, + lapack_int *liwork, + lapack_int *info); +void LAPACK_dsyevd(char *jobz, + char *uplo, + lapack_int *n, + double *a, + lapack_int *lda, + double *w, + double *work, + lapack_int *lwork, + lapack_int *iwork, + lapack_int *liwork, + lapack_int *info); +void LAPACK_cheevd(char *jobz, + char *uplo, + lapack_int *n, + lapack_complex_float *a, + lapack_int *lda, + float *w, + lapack_complex_float *work, + lapack_int *lwork, + float *rwork, + lapack_int *lrwork, + lapack_int *iwork, + lapack_int *liwork, + lapack_int *info); +void LAPACK_zheevd(char *jobz, + char *uplo, + lapack_int *n, + lapack_complex_double *a, + lapack_int *lda, + double *w, + lapack_complex_double *work, + lapack_int *lwork, + double *rwork, + lapack_int *lrwork, + lapack_int *iwork, + lapack_int *liwork, + lapack_int *info); +void LAPACK_ssyevx(char *jobz, + char *range, + char *uplo, + lapack_int *n, + float *a, + lapack_int *lda, + float *vl, + float *vu, + lapack_int *il, + lapack_int *iu, + float *abstol, + lapack_int *m, + float *w, + float *z, + lapack_int *ldz, + float *work, + lapack_int *lwork, + lapack_int *iwork, + lapack_int *ifail, + lapack_int *info); +void LAPACK_dsyevx(char *jobz, + char *range, + char *uplo, + lapack_int *n, + double *a, + lapack_int *lda, + double *vl, + double *vu, + lapack_int *il, + lapack_int *iu, + double *abstol, + lapack_int *m, + double *w, + double *z, + lapack_int *ldz, + double *work, + lapack_int *lwork, + lapack_int *iwork, + lapack_int *ifail, + lapack_int *info); +void LAPACK_cheevx(char *jobz, + char *range, + char *uplo, + lapack_int *n, + lapack_complex_float *a, + lapack_int *lda, + float *vl, + float *vu, + lapack_int *il, + lapack_int *iu, + float *abstol, + lapack_int *m, + float *w, + lapack_complex_float *z, + lapack_int *ldz, + lapack_complex_float *work, + lapack_int *lwork, + float *rwork, + lapack_int *iwork, + lapack_int *ifail, + lapack_int *info); +void LAPACK_zheevx(char *jobz, + char *range, + char *uplo, + lapack_int *n, + lapack_complex_double *a, + lapack_int *lda, + double *vl, + double *vu, + lapack_int *il, + lapack_int *iu, + double *abstol, + lapack_int *m, + double *w, + lapack_complex_double *z, + lapack_int *ldz, + lapack_complex_double *work, + lapack_int *lwork, + double *rwork, + lapack_int *iwork, + lapack_int *ifail, + lapack_int *info); +void LAPACK_ssyevr(char *jobz, + char *range, + char *uplo, + lapack_int *n, + float *a, + lapack_int *lda, + float *vl, + float *vu, + lapack_int *il, + lapack_int *iu, + float *abstol, + lapack_int *m, + float *w, + float *z, + lapack_int *ldz, + lapack_int *isuppz, + float *work, + lapack_int *lwork, + lapack_int *iwork, + lapack_int *liwork, + lapack_int *info); +void LAPACK_dsyevr(char *jobz, + char *range, + char *uplo, + lapack_int *n, + double *a, + lapack_int *lda, + double *vl, + double *vu, + lapack_int *il, + lapack_int *iu, + double *abstol, + lapack_int *m, + double *w, + double *z, + lapack_int *ldz, + lapack_int *isuppz, + double *work, + lapack_int *lwork, + lapack_int *iwork, + lapack_int *liwork, + lapack_int *info); +void LAPACK_cheevr(char *jobz, + char *range, + char *uplo, + lapack_int *n, + lapack_complex_float *a, + lapack_int *lda, + float *vl, + float *vu, + lapack_int *il, + lapack_int *iu, + float *abstol, + lapack_int *m, + float *w, + lapack_complex_float *z, + lapack_int *ldz, + lapack_int *isuppz, + lapack_complex_float *work, + lapack_int *lwork, + float *rwork, + lapack_int *lrwork, + lapack_int *iwork, + lapack_int *liwork, + lapack_int *info); +void LAPACK_zheevr(char *jobz, + char *range, + char *uplo, + lapack_int *n, + lapack_complex_double *a, + lapack_int *lda, + double *vl, + double *vu, + lapack_int *il, + lapack_int *iu, + double *abstol, + lapack_int *m, + double *w, + lapack_complex_double *z, + lapack_int *ldz, + lapack_int *isuppz, + lapack_complex_double *work, + lapack_int *lwork, + double *rwork, + lapack_int *lrwork, + lapack_int *iwork, + lapack_int *liwork, + lapack_int *info); +void LAPACK_sspev(char *jobz, + char *uplo, + lapack_int *n, + float *ap, + float *w, + float *z, + lapack_int *ldz, + float *work, + lapack_int *info); +void LAPACK_dspev(char *jobz, + char *uplo, + lapack_int *n, + double *ap, + double *w, + double *z, + lapack_int *ldz, + double *work, + lapack_int *info); +void LAPACK_chpev(char *jobz, + char *uplo, + lapack_int *n, + lapack_complex_float *ap, + float *w, + lapack_complex_float *z, + lapack_int *ldz, + lapack_complex_float *work, + float *rwork, + lapack_int *info); +void LAPACK_zhpev(char *jobz, + char *uplo, + lapack_int *n, + lapack_complex_double *ap, + double *w, + lapack_complex_double *z, + lapack_int *ldz, + lapack_complex_double *work, + double *rwork, + lapack_int *info); +void LAPACK_sspevd(char *jobz, + char *uplo, + lapack_int *n, + float *ap, + float *w, + float *z, + lapack_int *ldz, + float *work, + lapack_int *lwork, + lapack_int *iwork, + lapack_int *liwork, + lapack_int *info); +void LAPACK_dspevd(char *jobz, + char *uplo, + lapack_int *n, + double *ap, + double *w, + double *z, + lapack_int *ldz, + double *work, + lapack_int *lwork, + lapack_int *iwork, + lapack_int *liwork, + lapack_int *info); +void LAPACK_chpevd(char *jobz, + char *uplo, + lapack_int *n, + lapack_complex_float *ap, + float *w, + lapack_complex_float *z, + lapack_int *ldz, + lapack_complex_float *work, + lapack_int *lwork, + float *rwork, + lapack_int *lrwork, + lapack_int *iwork, + lapack_int *liwork, + lapack_int *info); +void LAPACK_zhpevd(char *jobz, + char *uplo, + lapack_int *n, + lapack_complex_double *ap, + double *w, + lapack_complex_double *z, + lapack_int *ldz, + lapack_complex_double *work, + lapack_int *lwork, + double *rwork, + lapack_int *lrwork, + lapack_int *iwork, + lapack_int *liwork, + lapack_int *info); +void LAPACK_sspevx(char *jobz, + char *range, + char *uplo, + lapack_int *n, + float *ap, + float *vl, + float *vu, + lapack_int *il, + lapack_int *iu, + float *abstol, + lapack_int *m, + float *w, + float *z, + lapack_int *ldz, + float *work, + lapack_int *iwork, + lapack_int *ifail, + lapack_int *info); +void LAPACK_dspevx(char *jobz, + char *range, + char *uplo, + lapack_int *n, + double *ap, + double *vl, + double *vu, + lapack_int *il, + lapack_int *iu, + double *abstol, + lapack_int *m, + double *w, + double *z, + lapack_int *ldz, + double *work, + lapack_int *iwork, + lapack_int *ifail, + lapack_int *info); +void LAPACK_chpevx(char *jobz, + char *range, + char *uplo, + lapack_int *n, + lapack_complex_float *ap, + float *vl, + float *vu, + lapack_int *il, + lapack_int *iu, + float *abstol, + lapack_int *m, + float *w, + lapack_complex_float *z, + lapack_int *ldz, + lapack_complex_float *work, + float *rwork, + lapack_int *iwork, + lapack_int *ifail, + lapack_int *info); +void LAPACK_zhpevx(char *jobz, + char *range, + char *uplo, + lapack_int *n, + lapack_complex_double *ap, + double *vl, + double *vu, + lapack_int *il, + lapack_int *iu, + double *abstol, + lapack_int *m, + double *w, + lapack_complex_double *z, + lapack_int *ldz, + lapack_complex_double *work, + double *rwork, + lapack_int *iwork, + lapack_int *ifail, + lapack_int *info); +void LAPACK_ssbev(char *jobz, + char *uplo, + lapack_int *n, + lapack_int *kd, + float *ab, + lapack_int *ldab, + float *w, + float *z, + lapack_int *ldz, + float *work, + lapack_int *info); +void LAPACK_dsbev(char *jobz, + char *uplo, + lapack_int *n, + lapack_int *kd, + double *ab, + lapack_int *ldab, + double *w, + double *z, + lapack_int *ldz, + double *work, + lapack_int *info); +void LAPACK_chbev(char *jobz, + char *uplo, + lapack_int *n, + lapack_int *kd, + lapack_complex_float *ab, + lapack_int *ldab, + float *w, + lapack_complex_float *z, + lapack_int *ldz, + lapack_complex_float *work, + float *rwork, + lapack_int *info); +void LAPACK_zhbev(char *jobz, + char *uplo, + lapack_int *n, + lapack_int *kd, + lapack_complex_double *ab, + lapack_int *ldab, + double *w, + lapack_complex_double *z, + lapack_int *ldz, + lapack_complex_double *work, + double *rwork, + lapack_int *info); +void LAPACK_ssbevd(char *jobz, + char *uplo, + lapack_int *n, + lapack_int *kd, + float *ab, + lapack_int *ldab, + float *w, + float *z, + lapack_int *ldz, + float *work, + lapack_int *lwork, + lapack_int *iwork, + lapack_int *liwork, + lapack_int *info); +void LAPACK_dsbevd(char *jobz, + char *uplo, + lapack_int *n, + lapack_int *kd, + double *ab, + lapack_int *ldab, + double *w, + double *z, + lapack_int *ldz, + double *work, + lapack_int *lwork, + lapack_int *iwork, + lapack_int *liwork, + lapack_int *info); +void LAPACK_chbevd(char *jobz, + char *uplo, + lapack_int *n, + lapack_int *kd, + lapack_complex_float *ab, + lapack_int *ldab, + float *w, + lapack_complex_float *z, + lapack_int *ldz, + lapack_complex_float *work, + lapack_int *lwork, + float *rwork, + lapack_int *lrwork, + lapack_int *iwork, + lapack_int *liwork, + lapack_int *info); +void LAPACK_zhbevd(char *jobz, + char *uplo, + lapack_int *n, + lapack_int *kd, + lapack_complex_double *ab, + lapack_int *ldab, + double *w, + lapack_complex_double *z, + lapack_int *ldz, + lapack_complex_double *work, + lapack_int *lwork, + double *rwork, + lapack_int *lrwork, + lapack_int *iwork, + lapack_int *liwork, + lapack_int *info); +void LAPACK_ssbevx(char *jobz, + char *range, + char *uplo, + lapack_int *n, + lapack_int *kd, + float *ab, + lapack_int *ldab, + float *q, + lapack_int *ldq, + float *vl, + float *vu, + lapack_int *il, + lapack_int *iu, + float *abstol, + lapack_int *m, + float *w, + float *z, + lapack_int *ldz, + float *work, + lapack_int *iwork, + lapack_int *ifail, + lapack_int *info); +void LAPACK_dsbevx(char *jobz, + char *range, + char *uplo, + lapack_int *n, + lapack_int *kd, + double *ab, + lapack_int *ldab, + double *q, + lapack_int *ldq, + double *vl, + double *vu, + lapack_int *il, + lapack_int *iu, + double *abstol, + lapack_int *m, + double *w, + double *z, + lapack_int *ldz, + double *work, + lapack_int *iwork, + lapack_int *ifail, + lapack_int *info); +void LAPACK_chbevx(char *jobz, + char *range, + char *uplo, + lapack_int *n, + lapack_int *kd, + lapack_complex_float *ab, + lapack_int *ldab, + lapack_complex_float *q, + lapack_int *ldq, + float *vl, + float *vu, + lapack_int *il, + lapack_int *iu, + float *abstol, + lapack_int *m, + float *w, + lapack_complex_float *z, + lapack_int *ldz, + lapack_complex_float *work, + float *rwork, + lapack_int *iwork, + lapack_int *ifail, + lapack_int *info); +void LAPACK_zhbevx(char *jobz, + char *range, + char *uplo, + lapack_int *n, + lapack_int *kd, + lapack_complex_double *ab, + lapack_int *ldab, + lapack_complex_double *q, + lapack_int *ldq, + double *vl, + double *vu, + lapack_int *il, + lapack_int *iu, + double *abstol, + lapack_int *m, + double *w, + lapack_complex_double *z, + lapack_int *ldz, + lapack_complex_double *work, + double *rwork, + lapack_int *iwork, + lapack_int *ifail, + lapack_int *info); +void LAPACK_sstev(char *jobz, + lapack_int *n, + float *d, + float *e, + float *z, + lapack_int *ldz, + float *work, + lapack_int *info); +void LAPACK_dstev(char *jobz, + lapack_int *n, + double *d, + double *e, + double *z, + lapack_int *ldz, + double *work, + lapack_int *info); +void LAPACK_sstevd(char *jobz, + lapack_int *n, + float *d, + float *e, + float *z, + lapack_int *ldz, + float *work, + lapack_int *lwork, + lapack_int *iwork, + lapack_int *liwork, + lapack_int *info); +void LAPACK_dstevd(char *jobz, + lapack_int *n, + double *d, + double *e, + double *z, + lapack_int *ldz, + double *work, + lapack_int *lwork, + lapack_int *iwork, + lapack_int *liwork, + lapack_int *info); +void LAPACK_sstevx(char *jobz, + char *range, + lapack_int *n, + float *d, + float *e, + float *vl, + float *vu, + lapack_int *il, + lapack_int *iu, + float *abstol, + lapack_int *m, + float *w, + float *z, + lapack_int *ldz, + float *work, + lapack_int *iwork, + lapack_int *ifail, + lapack_int *info); +void LAPACK_dstevx(char *jobz, + char *range, + lapack_int *n, + double *d, + double *e, + double *vl, + double *vu, + lapack_int *il, + lapack_int *iu, + double *abstol, + lapack_int *m, + double *w, + double *z, + lapack_int *ldz, + double *work, + lapack_int *iwork, + lapack_int *ifail, + lapack_int *info); +void LAPACK_sstevr(char *jobz, + char *range, + lapack_int *n, + float *d, + float *e, + float *vl, + float *vu, + lapack_int *il, + lapack_int *iu, + float *abstol, + lapack_int *m, + float *w, + float *z, + lapack_int *ldz, + lapack_int *isuppz, + float *work, + lapack_int *lwork, + lapack_int *iwork, + lapack_int *liwork, + lapack_int *info); +void LAPACK_dstevr(char *jobz, + char *range, + lapack_int *n, + double *d, + double *e, + double *vl, + double *vu, + lapack_int *il, + lapack_int *iu, + double *abstol, + lapack_int *m, + double *w, + double *z, + lapack_int *ldz, + lapack_int *isuppz, + double *work, + lapack_int *lwork, + lapack_int *iwork, + lapack_int *liwork, + lapack_int *info); +void LAPACK_sgees(char *jobvs, + char *sort, + LAPACK_S_SELECT2 select, + lapack_int *n, + float *a, + lapack_int *lda, + lapack_int *sdim, + float *wr, + float *wi, + float *vs, + lapack_int *ldvs, + float *work, + lapack_int *lwork, + lapack_logical *bwork, + lapack_int *info); +void LAPACK_dgees(char *jobvs, + char *sort, + LAPACK_D_SELECT2 select, + lapack_int *n, + double *a, + lapack_int *lda, + lapack_int *sdim, + double *wr, + double *wi, + double *vs, + lapack_int *ldvs, + double *work, + lapack_int *lwork, + lapack_logical *bwork, + lapack_int *info); +void LAPACK_cgees(char *jobvs, + char *sort, + LAPACK_C_SELECT1 select, + lapack_int *n, + lapack_complex_float *a, + lapack_int *lda, + lapack_int *sdim, + lapack_complex_float *w, + lapack_complex_float *vs, + lapack_int *ldvs, + lapack_complex_float *work, + lapack_int *lwork, + float *rwork, + lapack_logical *bwork, + lapack_int *info); +void LAPACK_zgees(char *jobvs, + char *sort, + LAPACK_Z_SELECT1 select, + lapack_int *n, + lapack_complex_double *a, + lapack_int *lda, + lapack_int *sdim, + lapack_complex_double *w, + lapack_complex_double *vs, + lapack_int *ldvs, + lapack_complex_double *work, + lapack_int *lwork, + double *rwork, + lapack_logical *bwork, + lapack_int *info); +void LAPACK_sgeesx(char *jobvs, + char *sort, + LAPACK_S_SELECT2 select, + char *sense, + lapack_int *n, + float *a, + lapack_int *lda, + lapack_int *sdim, + float *wr, + float *wi, + float *vs, + lapack_int *ldvs, + float *rconde, + float *rcondv, + float *work, + lapack_int *lwork, + lapack_int *iwork, + lapack_int *liwork, + lapack_logical *bwork, + lapack_int *info); +void LAPACK_dgeesx(char *jobvs, + char *sort, + LAPACK_D_SELECT2 select, + char *sense, + lapack_int *n, + double *a, + lapack_int *lda, + lapack_int *sdim, + double *wr, + double *wi, + double *vs, + lapack_int *ldvs, + double *rconde, + double *rcondv, + double *work, + lapack_int *lwork, + lapack_int *iwork, + lapack_int *liwork, + lapack_logical *bwork, + lapack_int *info); +void LAPACK_cgeesx(char *jobvs, + char *sort, + LAPACK_C_SELECT1 select, + char *sense, + lapack_int *n, + lapack_complex_float *a, + lapack_int *lda, + lapack_int *sdim, + lapack_complex_float *w, + lapack_complex_float *vs, + lapack_int *ldvs, + float *rconde, + float *rcondv, + lapack_complex_float *work, + lapack_int *lwork, + float *rwork, + lapack_logical *bwork, + lapack_int *info); +void LAPACK_zgeesx(char *jobvs, + char *sort, + LAPACK_Z_SELECT1 select, + char *sense, + lapack_int *n, + lapack_complex_double *a, + lapack_int *lda, + lapack_int *sdim, + lapack_complex_double *w, + lapack_complex_double *vs, + lapack_int *ldvs, + double *rconde, + double *rcondv, + lapack_complex_double *work, + lapack_int *lwork, + double *rwork, + lapack_logical *bwork, + lapack_int *info); +void LAPACK_sgeev(char *jobvl, + char *jobvr, + lapack_int *n, + float *a, + lapack_int *lda, + float *wr, + float *wi, + float *vl, + lapack_int *ldvl, + float *vr, + lapack_int *ldvr, + float *work, + lapack_int *lwork, + lapack_int *info); +void LAPACK_dgeev(char *jobvl, + char *jobvr, + lapack_int *n, + double *a, + lapack_int *lda, + double *wr, + double *wi, + double *vl, + lapack_int *ldvl, + double *vr, + lapack_int *ldvr, + double *work, + lapack_int *lwork, + lapack_int *info); +void LAPACK_cgeev(char *jobvl, + char *jobvr, + lapack_int *n, + lapack_complex_float *a, + lapack_int *lda, + lapack_complex_float *w, + lapack_complex_float *vl, + lapack_int *ldvl, + lapack_complex_float *vr, + lapack_int *ldvr, + lapack_complex_float *work, + lapack_int *lwork, + float *rwork, + lapack_int *info); +void LAPACK_zgeev(char *jobvl, + char *jobvr, + lapack_int *n, + lapack_complex_double *a, + lapack_int *lda, + lapack_complex_double *w, + lapack_complex_double *vl, + lapack_int *ldvl, + lapack_complex_double *vr, + lapack_int *ldvr, + lapack_complex_double *work, + lapack_int *lwork, + double *rwork, + lapack_int *info); +void LAPACK_sgeevx(char *balanc, + char *jobvl, + char *jobvr, + char *sense, + lapack_int *n, + float *a, + lapack_int *lda, + float *wr, + float *wi, + float *vl, + lapack_int *ldvl, + float *vr, + lapack_int *ldvr, + lapack_int *ilo, + lapack_int *ihi, + float *scale, + float *abnrm, + float *rconde, + float *rcondv, + float *work, + lapack_int *lwork, + lapack_int *iwork, + lapack_int *info); +void LAPACK_dgeevx(char *balanc, + char *jobvl, + char *jobvr, + char *sense, + lapack_int *n, + double *a, + lapack_int *lda, + double *wr, + double *wi, + double *vl, + lapack_int *ldvl, + double *vr, + lapack_int *ldvr, + lapack_int *ilo, + lapack_int *ihi, + double *scale, + double *abnrm, + double *rconde, + double *rcondv, + double *work, + lapack_int *lwork, + lapack_int *iwork, + lapack_int *info); +void LAPACK_cgeevx(char *balanc, + char *jobvl, + char *jobvr, + char *sense, + lapack_int *n, + lapack_complex_float *a, + lapack_int *lda, + lapack_complex_float *w, + lapack_complex_float *vl, + lapack_int *ldvl, + lapack_complex_float *vr, + lapack_int *ldvr, + lapack_int *ilo, + lapack_int *ihi, + float *scale, + float *abnrm, + float *rconde, + float *rcondv, + lapack_complex_float *work, + lapack_int *lwork, + float *rwork, + lapack_int *info); +void LAPACK_zgeevx(char *balanc, + char *jobvl, + char *jobvr, + char *sense, + lapack_int *n, + lapack_complex_double *a, + lapack_int *lda, + lapack_complex_double *w, + lapack_complex_double *vl, + lapack_int *ldvl, + lapack_complex_double *vr, + lapack_int *ldvr, + lapack_int *ilo, + lapack_int *ihi, + double *scale, + double *abnrm, + double *rconde, + double *rcondv, + lapack_complex_double *work, + lapack_int *lwork, + double *rwork, + lapack_int *info); +void LAPACK_sgesvd(char *jobu, + char *jobvt, + lapack_int *m, + lapack_int *n, + float *a, + lapack_int *lda, + float *s, + float *u, + lapack_int *ldu, + float *vt, + lapack_int *ldvt, + float *work, + lapack_int *lwork, + lapack_int *info); +void LAPACK_dgesvd(char *jobu, + char *jobvt, + lapack_int *m, + lapack_int *n, + double *a, + lapack_int *lda, + double *s, + double *u, + lapack_int *ldu, + double *vt, + lapack_int *ldvt, + double *work, + lapack_int *lwork, + lapack_int *info); +void LAPACK_cgesvd(char *jobu, + char *jobvt, + lapack_int *m, + lapack_int *n, + lapack_complex_float *a, + lapack_int *lda, + float *s, + lapack_complex_float *u, + lapack_int *ldu, + lapack_complex_float *vt, + lapack_int *ldvt, + lapack_complex_float *work, + lapack_int *lwork, + float *rwork, + lapack_int *info); +void LAPACK_zgesvd(char *jobu, + char *jobvt, + lapack_int *m, + lapack_int *n, + lapack_complex_double *a, + lapack_int *lda, + double *s, + lapack_complex_double *u, + lapack_int *ldu, + lapack_complex_double *vt, + lapack_int *ldvt, + lapack_complex_double *work, + lapack_int *lwork, + double *rwork, + lapack_int *info); +void LAPACK_sgesdd(char *jobz, + lapack_int *m, + lapack_int *n, + float *a, + lapack_int *lda, + float *s, + float *u, + lapack_int *ldu, + float *vt, + lapack_int *ldvt, + float *work, + lapack_int *lwork, + lapack_int *iwork, + lapack_int *info); +void LAPACK_dgesdd(char *jobz, + lapack_int *m, + lapack_int *n, + double *a, + lapack_int *lda, + double *s, + double *u, + lapack_int *ldu, + double *vt, + lapack_int *ldvt, + double *work, + lapack_int *lwork, + lapack_int *iwork, + lapack_int *info); +void LAPACK_cgesdd(char *jobz, + lapack_int *m, + lapack_int *n, + lapack_complex_float *a, + lapack_int *lda, + float *s, + lapack_complex_float *u, + lapack_int *ldu, + lapack_complex_float *vt, + lapack_int *ldvt, + lapack_complex_float *work, + lapack_int *lwork, + float *rwork, + lapack_int *iwork, + lapack_int *info); +void LAPACK_zgesdd(char *jobz, + lapack_int *m, + lapack_int *n, + lapack_complex_double *a, + lapack_int *lda, + double *s, + lapack_complex_double *u, + lapack_int *ldu, + lapack_complex_double *vt, + lapack_int *ldvt, + lapack_complex_double *work, + lapack_int *lwork, + double *rwork, + lapack_int *iwork, + lapack_int *info); +void LAPACK_dgejsv(char *joba, + char *jobu, + char *jobv, + char *jobr, + char *jobt, + char *jobp, + lapack_int *m, + lapack_int *n, + double *a, + lapack_int *lda, + double *sva, + double *u, + lapack_int *ldu, + double *v, + lapack_int *ldv, + double *work, + lapack_int *lwork, + lapack_int *iwork, + lapack_int *info); +void LAPACK_sgejsv(char *joba, + char *jobu, + char *jobv, + char *jobr, + char *jobt, + char *jobp, + lapack_int *m, + lapack_int *n, + float *a, + lapack_int *lda, + float *sva, + float *u, + lapack_int *ldu, + float *v, + lapack_int *ldv, + float *work, + lapack_int *lwork, + lapack_int *iwork, + lapack_int *info); +void LAPACK_dgesvj(char *joba, + char *jobu, + char *jobv, + lapack_int *m, + lapack_int *n, + double *a, + lapack_int *lda, + double *sva, + lapack_int *mv, + double *v, + lapack_int *ldv, + double *work, + lapack_int *lwork, + lapack_int *info); +void LAPACK_sgesvj(char *joba, + char *jobu, + char *jobv, + lapack_int *m, + lapack_int *n, + float *a, + lapack_int *lda, + float *sva, + lapack_int *mv, + float *v, + lapack_int *ldv, + float *work, + lapack_int *lwork, + lapack_int *info); +void LAPACK_sggsvd(char *jobu, + char *jobv, + char *jobq, + lapack_int *m, + lapack_int *n, + lapack_int *p, + lapack_int *k, + lapack_int *l, + float *a, + lapack_int *lda, + float *b, + lapack_int *ldb, + float *alpha, + float *beta, + float *u, + lapack_int *ldu, + float *v, + lapack_int *ldv, + float *q, + lapack_int *ldq, + float *work, + lapack_int *iwork, + lapack_int *info); +void LAPACK_dggsvd(char *jobu, + char *jobv, + char *jobq, + lapack_int *m, + lapack_int *n, + lapack_int *p, + lapack_int *k, + lapack_int *l, + double *a, + lapack_int *lda, + double *b, + lapack_int *ldb, + double *alpha, + double *beta, + double *u, + lapack_int *ldu, + double *v, + lapack_int *ldv, + double *q, + lapack_int *ldq, + double *work, + lapack_int *iwork, + lapack_int *info); +void LAPACK_cggsvd(char *jobu, + char *jobv, + char *jobq, + lapack_int *m, + lapack_int *n, + lapack_int *p, + lapack_int *k, + lapack_int *l, + lapack_complex_float *a, + lapack_int *lda, + lapack_complex_float *b, + lapack_int *ldb, + float *alpha, + float *beta, + lapack_complex_float *u, + lapack_int *ldu, + lapack_complex_float *v, + lapack_int *ldv, + lapack_complex_float *q, + lapack_int *ldq, + lapack_complex_float *work, + float *rwork, + lapack_int *iwork, + lapack_int *info); +void LAPACK_zggsvd(char *jobu, + char *jobv, + char *jobq, + lapack_int *m, + lapack_int *n, + lapack_int *p, + lapack_int *k, + lapack_int *l, + lapack_complex_double *a, + lapack_int *lda, + lapack_complex_double *b, + lapack_int *ldb, + double *alpha, + double *beta, + lapack_complex_double *u, + lapack_int *ldu, + lapack_complex_double *v, + lapack_int *ldv, + lapack_complex_double *q, + lapack_int *ldq, + lapack_complex_double *work, + double *rwork, + lapack_int *iwork, + lapack_int *info); +void LAPACK_ssygv(lapack_int *itype, + char *jobz, + char *uplo, + lapack_int *n, + float *a, + lapack_int *lda, + float *b, + lapack_int *ldb, + float *w, + float *work, + lapack_int *lwork, + lapack_int *info); +void LAPACK_dsygv(lapack_int *itype, + char *jobz, + char *uplo, + lapack_int *n, + double *a, + lapack_int *lda, + double *b, + lapack_int *ldb, + double *w, + double *work, + lapack_int *lwork, + lapack_int *info); +void LAPACK_chegv(lapack_int *itype, + char *jobz, + char *uplo, + lapack_int *n, + lapack_complex_float *a, + lapack_int *lda, + lapack_complex_float *b, + lapack_int *ldb, + float *w, + lapack_complex_float *work, + lapack_int *lwork, + float *rwork, + lapack_int *info); +void LAPACK_zhegv(lapack_int *itype, + char *jobz, + char *uplo, + lapack_int *n, + lapack_complex_double *a, + lapack_int *lda, + lapack_complex_double *b, + lapack_int *ldb, + double *w, + lapack_complex_double *work, + lapack_int *lwork, + double *rwork, + lapack_int *info); +void LAPACK_ssygvd(lapack_int *itype, + char *jobz, + char *uplo, + lapack_int *n, + float *a, + lapack_int *lda, + float *b, + lapack_int *ldb, + float *w, + float *work, + lapack_int *lwork, + lapack_int *iwork, + lapack_int *liwork, + lapack_int *info); +void LAPACK_dsygvd(lapack_int *itype, + char *jobz, + char *uplo, + lapack_int *n, + double *a, + lapack_int *lda, + double *b, + lapack_int *ldb, + double *w, + double *work, + lapack_int *lwork, + lapack_int *iwork, + lapack_int *liwork, + lapack_int *info); +void LAPACK_chegvd(lapack_int *itype, + char *jobz, + char *uplo, + lapack_int *n, + lapack_complex_float *a, + lapack_int *lda, + lapack_complex_float *b, + lapack_int *ldb, + float *w, + lapack_complex_float *work, + lapack_int *lwork, + float *rwork, + lapack_int *lrwork, + lapack_int *iwork, + lapack_int *liwork, + lapack_int *info); +void LAPACK_zhegvd(lapack_int *itype, + char *jobz, + char *uplo, + lapack_int *n, + lapack_complex_double *a, + lapack_int *lda, + lapack_complex_double *b, + lapack_int *ldb, + double *w, + lapack_complex_double *work, + lapack_int *lwork, + double *rwork, + lapack_int *lrwork, + lapack_int *iwork, + lapack_int *liwork, + lapack_int *info); +void LAPACK_ssygvx(lapack_int *itype, + char *jobz, + char *range, + char *uplo, + lapack_int *n, + float *a, + lapack_int *lda, + float *b, + lapack_int *ldb, + float *vl, + float *vu, + lapack_int *il, + lapack_int *iu, + float *abstol, + lapack_int *m, + float *w, + float *z, + lapack_int *ldz, + float *work, + lapack_int *lwork, + lapack_int *iwork, + lapack_int *ifail, + lapack_int *info); +void LAPACK_dsygvx(lapack_int *itype, + char *jobz, + char *range, + char *uplo, + lapack_int *n, + double *a, + lapack_int *lda, + double *b, + lapack_int *ldb, + double *vl, + double *vu, + lapack_int *il, + lapack_int *iu, + double *abstol, + lapack_int *m, + double *w, + double *z, + lapack_int *ldz, + double *work, + lapack_int *lwork, + lapack_int *iwork, + lapack_int *ifail, + lapack_int *info); +void LAPACK_chegvx(lapack_int *itype, + char *jobz, + char *range, + char *uplo, + lapack_int *n, + lapack_complex_float *a, + lapack_int *lda, + lapack_complex_float *b, + lapack_int *ldb, + float *vl, + float *vu, + lapack_int *il, + lapack_int *iu, + float *abstol, + lapack_int *m, + float *w, + lapack_complex_float *z, + lapack_int *ldz, + lapack_complex_float *work, + lapack_int *lwork, + float *rwork, + lapack_int *iwork, + lapack_int *ifail, + lapack_int *info); +void LAPACK_zhegvx(lapack_int *itype, + char *jobz, + char *range, + char *uplo, + lapack_int *n, + lapack_complex_double *a, + lapack_int *lda, + lapack_complex_double *b, + lapack_int *ldb, + double *vl, + double *vu, + lapack_int *il, + lapack_int *iu, + double *abstol, + lapack_int *m, + double *w, + lapack_complex_double *z, + lapack_int *ldz, + lapack_complex_double *work, + lapack_int *lwork, + double *rwork, + lapack_int *iwork, + lapack_int *ifail, + lapack_int *info); +void LAPACK_sspgv(lapack_int *itype, + char *jobz, + char *uplo, + lapack_int *n, + float *ap, + float *bp, + float *w, + float *z, + lapack_int *ldz, + float *work, + lapack_int *info); +void LAPACK_dspgv(lapack_int *itype, + char *jobz, + char *uplo, + lapack_int *n, + double *ap, + double *bp, + double *w, + double *z, + lapack_int *ldz, + double *work, + lapack_int *info); +void LAPACK_chpgv(lapack_int *itype, + char *jobz, + char *uplo, + lapack_int *n, + lapack_complex_float *ap, + lapack_complex_float *bp, + float *w, + lapack_complex_float *z, + lapack_int *ldz, + lapack_complex_float *work, + float *rwork, + lapack_int *info); +void LAPACK_zhpgv(lapack_int *itype, + char *jobz, + char *uplo, + lapack_int *n, + lapack_complex_double *ap, + lapack_complex_double *bp, + double *w, + lapack_complex_double *z, + lapack_int *ldz, + lapack_complex_double *work, + double *rwork, + lapack_int *info); +void LAPACK_sspgvd(lapack_int *itype, + char *jobz, + char *uplo, + lapack_int *n, + float *ap, + float *bp, + float *w, + float *z, + lapack_int *ldz, + float *work, + lapack_int *lwork, + lapack_int *iwork, + lapack_int *liwork, + lapack_int *info); +void LAPACK_dspgvd(lapack_int *itype, + char *jobz, + char *uplo, + lapack_int *n, + double *ap, + double *bp, + double *w, + double *z, + lapack_int *ldz, + double *work, + lapack_int *lwork, + lapack_int *iwork, + lapack_int *liwork, + lapack_int *info); +void LAPACK_chpgvd(lapack_int *itype, + char *jobz, + char *uplo, + lapack_int *n, + lapack_complex_float *ap, + lapack_complex_float *bp, + float *w, + lapack_complex_float *z, + lapack_int *ldz, + lapack_complex_float *work, + lapack_int *lwork, + float *rwork, + lapack_int *lrwork, + lapack_int *iwork, + lapack_int *liwork, + lapack_int *info); +void LAPACK_zhpgvd(lapack_int *itype, + char *jobz, + char *uplo, + lapack_int *n, + lapack_complex_double *ap, + lapack_complex_double *bp, + double *w, + lapack_complex_double *z, + lapack_int *ldz, + lapack_complex_double *work, + lapack_int *lwork, + double *rwork, + lapack_int *lrwork, + lapack_int *iwork, + lapack_int *liwork, + lapack_int *info); +void LAPACK_sspgvx(lapack_int *itype, + char *jobz, + char *range, + char *uplo, + lapack_int *n, + float *ap, + float *bp, + float *vl, + float *vu, + lapack_int *il, + lapack_int *iu, + float *abstol, + lapack_int *m, + float *w, + float *z, + lapack_int *ldz, + float *work, + lapack_int *iwork, + lapack_int *ifail, + lapack_int *info); +void LAPACK_dspgvx(lapack_int *itype, + char *jobz, + char *range, + char *uplo, + lapack_int *n, + double *ap, + double *bp, + double *vl, + double *vu, + lapack_int *il, + lapack_int *iu, + double *abstol, + lapack_int *m, + double *w, + double *z, + lapack_int *ldz, + double *work, + lapack_int *iwork, + lapack_int *ifail, + lapack_int *info); +void LAPACK_chpgvx(lapack_int *itype, + char *jobz, + char *range, + char *uplo, + lapack_int *n, + lapack_complex_float *ap, + lapack_complex_float *bp, + float *vl, + float *vu, + lapack_int *il, + lapack_int *iu, + float *abstol, + lapack_int *m, + float *w, + lapack_complex_float *z, + lapack_int *ldz, + lapack_complex_float *work, + float *rwork, + lapack_int *iwork, + lapack_int *ifail, + lapack_int *info); +void LAPACK_zhpgvx(lapack_int *itype, + char *jobz, + char *range, + char *uplo, + lapack_int *n, + lapack_complex_double *ap, + lapack_complex_double *bp, + double *vl, + double *vu, + lapack_int *il, + lapack_int *iu, + double *abstol, + lapack_int *m, + double *w, + lapack_complex_double *z, + lapack_int *ldz, + lapack_complex_double *work, + double *rwork, + lapack_int *iwork, + lapack_int *ifail, + lapack_int *info); +void LAPACK_ssbgv(char *jobz, + char *uplo, + lapack_int *n, + lapack_int *ka, + lapack_int *kb, + float *ab, + lapack_int *ldab, + float *bb, + lapack_int *ldbb, + float *w, + float *z, + lapack_int *ldz, + float *work, + lapack_int *info); +void LAPACK_dsbgv(char *jobz, + char *uplo, + lapack_int *n, + lapack_int *ka, + lapack_int *kb, + double *ab, + lapack_int *ldab, + double *bb, + lapack_int *ldbb, + double *w, + double *z, + lapack_int *ldz, + double *work, + lapack_int *info); +void LAPACK_chbgv(char *jobz, + char *uplo, + lapack_int *n, + lapack_int *ka, + lapack_int *kb, + lapack_complex_float *ab, + lapack_int *ldab, + lapack_complex_float *bb, + lapack_int *ldbb, + float *w, + lapack_complex_float *z, + lapack_int *ldz, + lapack_complex_float *work, + float *rwork, + lapack_int *info); +void LAPACK_zhbgv(char *jobz, + char *uplo, + lapack_int *n, + lapack_int *ka, + lapack_int *kb, + lapack_complex_double *ab, + lapack_int *ldab, + lapack_complex_double *bb, + lapack_int *ldbb, + double *w, + lapack_complex_double *z, + lapack_int *ldz, + lapack_complex_double *work, + double *rwork, + lapack_int *info); +void LAPACK_ssbgvd(char *jobz, + char *uplo, + lapack_int *n, + lapack_int *ka, + lapack_int *kb, + float *ab, + lapack_int *ldab, + float *bb, + lapack_int *ldbb, + float *w, + float *z, + lapack_int *ldz, + float *work, + lapack_int *lwork, + lapack_int *iwork, + lapack_int *liwork, + lapack_int *info); +void LAPACK_dsbgvd(char *jobz, + char *uplo, + lapack_int *n, + lapack_int *ka, + lapack_int *kb, + double *ab, + lapack_int *ldab, + double *bb, + lapack_int *ldbb, + double *w, + double *z, + lapack_int *ldz, + double *work, + lapack_int *lwork, + lapack_int *iwork, + lapack_int *liwork, + lapack_int *info); +void LAPACK_chbgvd(char *jobz, + char *uplo, + lapack_int *n, + lapack_int *ka, + lapack_int *kb, + lapack_complex_float *ab, + lapack_int *ldab, + lapack_complex_float *bb, + lapack_int *ldbb, + float *w, + lapack_complex_float *z, + lapack_int *ldz, + lapack_complex_float *work, + lapack_int *lwork, + float *rwork, + lapack_int *lrwork, + lapack_int *iwork, + lapack_int *liwork, + lapack_int *info); +void LAPACK_zhbgvd(char *jobz, + char *uplo, + lapack_int *n, + lapack_int *ka, + lapack_int *kb, + lapack_complex_double *ab, + lapack_int *ldab, + lapack_complex_double *bb, + lapack_int *ldbb, + double *w, + lapack_complex_double *z, + lapack_int *ldz, + lapack_complex_double *work, + lapack_int *lwork, + double *rwork, + lapack_int *lrwork, + lapack_int *iwork, + lapack_int *liwork, + lapack_int *info); +void LAPACK_ssbgvx(char *jobz, + char *range, + char *uplo, + lapack_int *n, + lapack_int *ka, + lapack_int *kb, + float *ab, + lapack_int *ldab, + float *bb, + lapack_int *ldbb, + float *q, + lapack_int *ldq, + float *vl, + float *vu, + lapack_int *il, + lapack_int *iu, + float *abstol, + lapack_int *m, + float *w, + float *z, + lapack_int *ldz, + float *work, + lapack_int *iwork, + lapack_int *ifail, + lapack_int *info); +void LAPACK_dsbgvx(char *jobz, + char *range, + char *uplo, + lapack_int *n, + lapack_int *ka, + lapack_int *kb, + double *ab, + lapack_int *ldab, + double *bb, + lapack_int *ldbb, + double *q, + lapack_int *ldq, + double *vl, + double *vu, + lapack_int *il, + lapack_int *iu, + double *abstol, + lapack_int *m, + double *w, + double *z, + lapack_int *ldz, + double *work, + lapack_int *iwork, + lapack_int *ifail, + lapack_int *info); +void LAPACK_chbgvx(char *jobz, + char *range, + char *uplo, + lapack_int *n, + lapack_int *ka, + lapack_int *kb, + lapack_complex_float *ab, + lapack_int *ldab, + lapack_complex_float *bb, + lapack_int *ldbb, + lapack_complex_float *q, + lapack_int *ldq, + float *vl, + float *vu, + lapack_int *il, + lapack_int *iu, + float *abstol, + lapack_int *m, + float *w, + lapack_complex_float *z, + lapack_int *ldz, + lapack_complex_float *work, + float *rwork, + lapack_int *iwork, + lapack_int *ifail, + lapack_int *info); +void LAPACK_zhbgvx(char *jobz, + char *range, + char *uplo, + lapack_int *n, + lapack_int *ka, + lapack_int *kb, + lapack_complex_double *ab, + lapack_int *ldab, + lapack_complex_double *bb, + lapack_int *ldbb, + lapack_complex_double *q, + lapack_int *ldq, + double *vl, + double *vu, + lapack_int *il, + lapack_int *iu, + double *abstol, + lapack_int *m, + double *w, + lapack_complex_double *z, + lapack_int *ldz, + lapack_complex_double *work, + double *rwork, + lapack_int *iwork, + lapack_int *ifail, + lapack_int *info); +void LAPACK_sgges(char *jobvsl, + char *jobvsr, + char *sort, + LAPACK_S_SELECT3 selctg, + lapack_int *n, + float *a, + lapack_int *lda, + float *b, + lapack_int *ldb, + lapack_int *sdim, + float *alphar, + float *alphai, + float *beta, + float *vsl, + lapack_int *ldvsl, + float *vsr, + lapack_int *ldvsr, + float *work, + lapack_int *lwork, + lapack_logical *bwork, + lapack_int *info); +void LAPACK_dgges(char *jobvsl, + char *jobvsr, + char *sort, + LAPACK_D_SELECT3 selctg, + lapack_int *n, + double *a, + lapack_int *lda, + double *b, + lapack_int *ldb, + lapack_int *sdim, + double *alphar, + double *alphai, + double *beta, + double *vsl, + lapack_int *ldvsl, + double *vsr, + lapack_int *ldvsr, + double *work, + lapack_int *lwork, + lapack_logical *bwork, + lapack_int *info); +void LAPACK_cgges(char *jobvsl, + char *jobvsr, + char *sort, + LAPACK_C_SELECT2 selctg, + lapack_int *n, + lapack_complex_float *a, + lapack_int *lda, + lapack_complex_float *b, + lapack_int *ldb, + lapack_int *sdim, + lapack_complex_float *alpha, + lapack_complex_float *beta, + lapack_complex_float *vsl, + lapack_int *ldvsl, + lapack_complex_float *vsr, + lapack_int *ldvsr, + lapack_complex_float *work, + lapack_int *lwork, + float *rwork, + lapack_logical *bwork, + lapack_int *info); +void LAPACK_zgges(char *jobvsl, + char *jobvsr, + char *sort, + LAPACK_Z_SELECT2 selctg, + lapack_int *n, + lapack_complex_double *a, + lapack_int *lda, + lapack_complex_double *b, + lapack_int *ldb, + lapack_int *sdim, + lapack_complex_double *alpha, + lapack_complex_double *beta, + lapack_complex_double *vsl, + lapack_int *ldvsl, + lapack_complex_double *vsr, + lapack_int *ldvsr, + lapack_complex_double *work, + lapack_int *lwork, + double *rwork, + lapack_logical *bwork, + lapack_int *info); +void LAPACK_sggesx(char *jobvsl, + char *jobvsr, + char *sort, + LAPACK_S_SELECT3 selctg, + char *sense, + lapack_int *n, + float *a, + lapack_int *lda, + float *b, + lapack_int *ldb, + lapack_int *sdim, + float *alphar, + float *alphai, + float *beta, + float *vsl, + lapack_int *ldvsl, + float *vsr, + lapack_int *ldvsr, + float *rconde, + float *rcondv, + float *work, + lapack_int *lwork, + lapack_int *iwork, + lapack_int *liwork, + lapack_logical *bwork, + lapack_int *info); +void LAPACK_dggesx(char *jobvsl, + char *jobvsr, + char *sort, + LAPACK_D_SELECT3 selctg, + char *sense, + lapack_int *n, + double *a, + lapack_int *lda, + double *b, + lapack_int *ldb, + lapack_int *sdim, + double *alphar, + double *alphai, + double *beta, + double *vsl, + lapack_int *ldvsl, + double *vsr, + lapack_int *ldvsr, + double *rconde, + double *rcondv, + double *work, + lapack_int *lwork, + lapack_int *iwork, + lapack_int *liwork, + lapack_logical *bwork, + lapack_int *info); +void LAPACK_cggesx(char *jobvsl, + char *jobvsr, + char *sort, + LAPACK_C_SELECT2 selctg, + char *sense, + lapack_int *n, + lapack_complex_float *a, + lapack_int *lda, + lapack_complex_float *b, + lapack_int *ldb, + lapack_int *sdim, + lapack_complex_float *alpha, + lapack_complex_float *beta, + lapack_complex_float *vsl, + lapack_int *ldvsl, + lapack_complex_float *vsr, + lapack_int *ldvsr, + float *rconde, + float *rcondv, + lapack_complex_float *work, + lapack_int *lwork, + float *rwork, + lapack_int *iwork, + lapack_int *liwork, + lapack_logical *bwork, + lapack_int *info); +void LAPACK_zggesx(char *jobvsl, + char *jobvsr, + char *sort, + LAPACK_Z_SELECT2 selctg, + char *sense, + lapack_int *n, + lapack_complex_double *a, + lapack_int *lda, + lapack_complex_double *b, + lapack_int *ldb, + lapack_int *sdim, + lapack_complex_double *alpha, + lapack_complex_double *beta, + lapack_complex_double *vsl, + lapack_int *ldvsl, + lapack_complex_double *vsr, + lapack_int *ldvsr, + double *rconde, + double *rcondv, + lapack_complex_double *work, + lapack_int *lwork, + double *rwork, + lapack_int *iwork, + lapack_int *liwork, + lapack_logical *bwork, + lapack_int *info); +void LAPACK_sggev(char *jobvl, + char *jobvr, + lapack_int *n, + float *a, + lapack_int *lda, + float *b, + lapack_int *ldb, + float *alphar, + float *alphai, + float *beta, + float *vl, + lapack_int *ldvl, + float *vr, + lapack_int *ldvr, + float *work, + lapack_int *lwork, + lapack_int *info); +void LAPACK_dggev(char *jobvl, + char *jobvr, + lapack_int *n, + double *a, + lapack_int *lda, + double *b, + lapack_int *ldb, + double *alphar, + double *alphai, + double *beta, + double *vl, + lapack_int *ldvl, + double *vr, + lapack_int *ldvr, + double *work, + lapack_int *lwork, + lapack_int *info); +void LAPACK_cggev(char *jobvl, + char *jobvr, + lapack_int *n, + lapack_complex_float *a, + lapack_int *lda, + lapack_complex_float *b, + lapack_int *ldb, + lapack_complex_float *alpha, + lapack_complex_float *beta, + lapack_complex_float *vl, + lapack_int *ldvl, + lapack_complex_float *vr, + lapack_int *ldvr, + lapack_complex_float *work, + lapack_int *lwork, + float *rwork, + lapack_int *info); +void LAPACK_zggev(char *jobvl, + char *jobvr, + lapack_int *n, + lapack_complex_double *a, + lapack_int *lda, + lapack_complex_double *b, + lapack_int *ldb, + lapack_complex_double *alpha, + lapack_complex_double *beta, + lapack_complex_double *vl, + lapack_int *ldvl, + lapack_complex_double *vr, + lapack_int *ldvr, + lapack_complex_double *work, + lapack_int *lwork, + double *rwork, + lapack_int *info); +void LAPACK_sggevx(char *balanc, + char *jobvl, + char *jobvr, + char *sense, + lapack_int *n, + float *a, + lapack_int *lda, + float *b, + lapack_int *ldb, + float *alphar, + float *alphai, + float *beta, + float *vl, + lapack_int *ldvl, + float *vr, + lapack_int *ldvr, + lapack_int *ilo, + lapack_int *ihi, + float *lscale, + float *rscale, + float *abnrm, + float *bbnrm, + float *rconde, + float *rcondv, + float *work, + lapack_int *lwork, + lapack_int *iwork, + lapack_logical *bwork, + lapack_int *info); +void LAPACK_dggevx(char *balanc, + char *jobvl, + char *jobvr, + char *sense, + lapack_int *n, + double *a, + lapack_int *lda, + double *b, + lapack_int *ldb, + double *alphar, + double *alphai, + double *beta, + double *vl, + lapack_int *ldvl, + double *vr, + lapack_int *ldvr, + lapack_int *ilo, + lapack_int *ihi, + double *lscale, + double *rscale, + double *abnrm, + double *bbnrm, + double *rconde, + double *rcondv, + double *work, + lapack_int *lwork, + lapack_int *iwork, + lapack_logical *bwork, + lapack_int *info); +void LAPACK_cggevx(char *balanc, + char *jobvl, + char *jobvr, + char *sense, + lapack_int *n, + lapack_complex_float *a, + lapack_int *lda, + lapack_complex_float *b, + lapack_int *ldb, + lapack_complex_float *alpha, + lapack_complex_float *beta, + lapack_complex_float *vl, + lapack_int *ldvl, + lapack_complex_float *vr, + lapack_int *ldvr, + lapack_int *ilo, + lapack_int *ihi, + float *lscale, + float *rscale, + float *abnrm, + float *bbnrm, + float *rconde, + float *rcondv, + lapack_complex_float *work, + lapack_int *lwork, + float *rwork, + lapack_int *iwork, + lapack_logical *bwork, + lapack_int *info); +void LAPACK_zggevx(char *balanc, + char *jobvl, + char *jobvr, + char *sense, + lapack_int *n, + lapack_complex_double *a, + lapack_int *lda, + lapack_complex_double *b, + lapack_int *ldb, + lapack_complex_double *alpha, + lapack_complex_double *beta, + lapack_complex_double *vl, + lapack_int *ldvl, + lapack_complex_double *vr, + lapack_int *ldvr, + lapack_int *ilo, + lapack_int *ihi, + double *lscale, + double *rscale, + double *abnrm, + double *bbnrm, + double *rconde, + double *rcondv, + lapack_complex_double *work, + lapack_int *lwork, + double *rwork, + lapack_int *iwork, + lapack_logical *bwork, + lapack_int *info); +void LAPACK_dsfrk(char *transr, + char *uplo, + char *trans, + lapack_int *n, + lapack_int *k, + double *alpha, + const double *a, + lapack_int *lda, + double *beta, + double *c); +void LAPACK_ssfrk(char *transr, + char *uplo, + char *trans, + lapack_int *n, + lapack_int *k, + float *alpha, + const float *a, + lapack_int *lda, + float *beta, + float *c); +void LAPACK_zhfrk(char *transr, + char *uplo, + char *trans, + lapack_int *n, + lapack_int *k, + double *alpha, + const lapack_complex_double *a, + lapack_int *lda, + double *beta, + lapack_complex_double *c); +void LAPACK_chfrk(char *transr, + char *uplo, + char *trans, + lapack_int *n, + lapack_int *k, + float *alpha, + const lapack_complex_float *a, + lapack_int *lda, + float *beta, + lapack_complex_float *c); +void LAPACK_dtfsm(char *transr, + char *side, + char *uplo, + char *trans, + char *diag, + lapack_int *m, + lapack_int *n, + double *alpha, + const double *a, + double *b, + lapack_int *ldb); +void LAPACK_stfsm(char *transr, + char *side, + char *uplo, + char *trans, + char *diag, + lapack_int *m, + lapack_int *n, + float *alpha, + const float *a, + float *b, + lapack_int *ldb); +void LAPACK_ztfsm(char *transr, + char *side, + char *uplo, + char *trans, + char *diag, + lapack_int *m, + lapack_int *n, + lapack_complex_double *alpha, + const lapack_complex_double *a, + lapack_complex_double *b, + lapack_int *ldb); +void LAPACK_ctfsm(char *transr, + char *side, + char *uplo, + char *trans, + char *diag, + lapack_int *m, + lapack_int *n, + lapack_complex_float *alpha, + const lapack_complex_float *a, + lapack_complex_float *b, + lapack_int *ldb); +void LAPACK_dtfttp(char *transr, char *uplo, lapack_int *n, const double *arf, double *ap, lapack_int *info); +void LAPACK_stfttp(char *transr, char *uplo, lapack_int *n, const float *arf, float *ap, lapack_int *info); +void LAPACK_ztfttp(char *transr, + char *uplo, + lapack_int *n, + const lapack_complex_double *arf, + lapack_complex_double *ap, + lapack_int *info); +void LAPACK_ctfttp(char *transr, + char *uplo, + lapack_int *n, + const lapack_complex_float *arf, + lapack_complex_float *ap, + lapack_int *info); +void LAPACK_dtfttr(char *transr, + char *uplo, + lapack_int *n, + const double *arf, + double *a, + lapack_int *lda, + lapack_int *info); +void LAPACK_stfttr(char *transr, + char *uplo, + lapack_int *n, + const float *arf, + float *a, + lapack_int *lda, + lapack_int *info); +void LAPACK_ztfttr(char *transr, + char *uplo, + lapack_int *n, + const lapack_complex_double *arf, + lapack_complex_double *a, + lapack_int *lda, + lapack_int *info); +void LAPACK_ctfttr(char *transr, + char *uplo, + lapack_int *n, + const lapack_complex_float *arf, + lapack_complex_float *a, + lapack_int *lda, + lapack_int *info); +void LAPACK_dtpttf(char *transr, char *uplo, lapack_int *n, const double *ap, double *arf, lapack_int *info); +void LAPACK_stpttf(char *transr, char *uplo, lapack_int *n, const float *ap, float *arf, lapack_int *info); +void LAPACK_ztpttf(char *transr, + char *uplo, + lapack_int *n, + const lapack_complex_double *ap, + lapack_complex_double *arf, + lapack_int *info); +void LAPACK_ctpttf(char *transr, + char *uplo, + lapack_int *n, + const lapack_complex_float *ap, + lapack_complex_float *arf, + lapack_int *info); +void LAPACK_dtpttr(char *uplo, lapack_int *n, const double *ap, double *a, lapack_int *lda, lapack_int *info); +void LAPACK_stpttr(char *uplo, lapack_int *n, const float *ap, float *a, lapack_int *lda, lapack_int *info); +void LAPACK_ztpttr(char *uplo, + lapack_int *n, + const lapack_complex_double *ap, + lapack_complex_double *a, + lapack_int *lda, + lapack_int *info); +void LAPACK_ctpttr(char *uplo, + lapack_int *n, + const lapack_complex_float *ap, + lapack_complex_float *a, + lapack_int *lda, + lapack_int *info); +void LAPACK_dtrttf(char *transr, + char *uplo, + lapack_int *n, + const double *a, + lapack_int *lda, + double *arf, + lapack_int *info); +void LAPACK_strttf(char *transr, + char *uplo, + lapack_int *n, + const float *a, + lapack_int *lda, + float *arf, + lapack_int *info); +void LAPACK_ztrttf(char *transr, + char *uplo, + lapack_int *n, + const lapack_complex_double *a, + lapack_int *lda, + lapack_complex_double *arf, + lapack_int *info); +void LAPACK_ctrttf(char *transr, + char *uplo, + lapack_int *n, + const lapack_complex_float *a, + lapack_int *lda, + lapack_complex_float *arf, + lapack_int *info); +void LAPACK_dtrttp(char *uplo, lapack_int *n, const double *a, lapack_int *lda, double *ap, lapack_int *info); +void LAPACK_strttp(char *uplo, lapack_int *n, const float *a, lapack_int *lda, float *ap, lapack_int *info); +void LAPACK_ztrttp(char *uplo, + lapack_int *n, + const lapack_complex_double *a, + lapack_int *lda, + lapack_complex_double *ap, + lapack_int *info); +void LAPACK_ctrttp(char *uplo, + lapack_int *n, + const lapack_complex_float *a, + lapack_int *lda, + lapack_complex_float *ap, + lapack_int *info); +void LAPACK_sgeqrfp(lapack_int *m, + lapack_int *n, + float *a, + lapack_int *lda, + float *tau, + float *work, + lapack_int *lwork, + lapack_int *info); +void LAPACK_dgeqrfp(lapack_int *m, + lapack_int *n, + double *a, + lapack_int *lda, + double *tau, + double *work, + lapack_int *lwork, + lapack_int *info); +void LAPACK_cgeqrfp(lapack_int *m, + lapack_int *n, + lapack_complex_float *a, + lapack_int *lda, + lapack_complex_float *tau, + lapack_complex_float *work, + lapack_int *lwork, + lapack_int *info); +void LAPACK_zgeqrfp(lapack_int *m, + lapack_int *n, + lapack_complex_double *a, + lapack_int *lda, + lapack_complex_double *tau, + lapack_complex_double *work, + lapack_int *lwork, + lapack_int *info); +void LAPACK_clacgv(lapack_int *n, lapack_complex_float *x, lapack_int *incx); +void LAPACK_zlacgv(lapack_int *n, lapack_complex_double *x, lapack_int *incx); +void LAPACK_slarnv(lapack_int *idist, lapack_int *iseed, lapack_int *n, float *x); +void LAPACK_dlarnv(lapack_int *idist, lapack_int *iseed, lapack_int *n, double *x); +void LAPACK_clarnv(lapack_int *idist, lapack_int *iseed, lapack_int *n, lapack_complex_float *x); +void LAPACK_zlarnv(lapack_int *idist, lapack_int *iseed, lapack_int *n, lapack_complex_double *x); +void LAPACK_sgeqr2(lapack_int *m, lapack_int *n, float *a, lapack_int *lda, float *tau, float *work, lapack_int *info); +void LAPACK_dgeqr2(lapack_int *m, + lapack_int *n, + double *a, + lapack_int *lda, + double *tau, + double *work, + lapack_int *info); +void LAPACK_cgeqr2(lapack_int *m, + lapack_int *n, + lapack_complex_float *a, + lapack_int *lda, + lapack_complex_float *tau, + lapack_complex_float *work, + lapack_int *info); +void LAPACK_zgeqr2(lapack_int *m, + lapack_int *n, + lapack_complex_double *a, + lapack_int *lda, + lapack_complex_double *tau, + lapack_complex_double *work, + lapack_int *info); +void LAPACK_slacpy(char *uplo, + lapack_int *m, + lapack_int *n, + const float *a, + lapack_int *lda, + float *b, + lapack_int *ldb); +void LAPACK_dlacpy(char *uplo, + lapack_int *m, + lapack_int *n, + const double *a, + lapack_int *lda, + double *b, + lapack_int *ldb); +void LAPACK_clacpy(char *uplo, + lapack_int *m, + lapack_int *n, + const lapack_complex_float *a, + lapack_int *lda, + lapack_complex_float *b, + lapack_int *ldb); +void LAPACK_zlacpy(char *uplo, + lapack_int *m, + lapack_int *n, + const lapack_complex_double *a, + lapack_int *lda, + lapack_complex_double *b, + lapack_int *ldb); +void LAPACK_sgetf2(lapack_int *m, lapack_int *n, float *a, lapack_int *lda, lapack_int *ipiv, lapack_int *info); +void LAPACK_dgetf2(lapack_int *m, lapack_int *n, double *a, lapack_int *lda, lapack_int *ipiv, lapack_int *info); +void LAPACK_cgetf2(lapack_int *m, + lapack_int *n, + lapack_complex_float *a, + lapack_int *lda, + lapack_int *ipiv, + lapack_int *info); +void LAPACK_zgetf2(lapack_int *m, + lapack_int *n, + lapack_complex_double *a, + lapack_int *lda, + lapack_int *ipiv, + lapack_int *info); +void LAPACK_slaswp(lapack_int *n, + float *a, + lapack_int *lda, + lapack_int *k1, + lapack_int *k2, + const lapack_int *ipiv, + lapack_int *incx); +void LAPACK_dlaswp(lapack_int *n, + double *a, + lapack_int *lda, + lapack_int *k1, + lapack_int *k2, + const lapack_int *ipiv, + lapack_int *incx); +void LAPACK_claswp(lapack_int *n, + lapack_complex_float *a, + lapack_int *lda, + lapack_int *k1, + lapack_int *k2, + const lapack_int *ipiv, + lapack_int *incx); +void LAPACK_zlaswp(lapack_int *n, + lapack_complex_double *a, + lapack_int *lda, + lapack_int *k1, + lapack_int *k2, + const lapack_int *ipiv, + lapack_int *incx); +float LAPACK_slange(char *norm, lapack_int *m, lapack_int *n, const float *a, lapack_int *lda, float *work); +double LAPACK_dlange(char *norm, lapack_int *m, lapack_int *n, const double *a, lapack_int *lda, double *work); +float LAPACK_clange(char *norm, + lapack_int *m, + lapack_int *n, + const lapack_complex_float *a, + lapack_int *lda, + float *work); +double LAPACK_zlange(char *norm, + lapack_int *m, + lapack_int *n, + const lapack_complex_double *a, + lapack_int *lda, + double *work); +float LAPACK_clanhe(char *norm, char *uplo, lapack_int *n, const lapack_complex_float *a, lapack_int *lda, float *work); +double + LAPACK_zlanhe(char *norm, char *uplo, lapack_int *n, const lapack_complex_double *a, lapack_int *lda, double *work); +float LAPACK_slansy(char *norm, char *uplo, lapack_int *n, const float *a, lapack_int *lda, float *work); +double LAPACK_dlansy(char *norm, char *uplo, lapack_int *n, const double *a, lapack_int *lda, double *work); +float LAPACK_clansy(char *norm, char *uplo, lapack_int *n, const lapack_complex_float *a, lapack_int *lda, float *work); +double + LAPACK_zlansy(char *norm, char *uplo, lapack_int *n, const lapack_complex_double *a, lapack_int *lda, double *work); +float LAPACK_slantr(char *norm, + char *uplo, + char *diag, + lapack_int *m, + lapack_int *n, + const float *a, + lapack_int *lda, + float *work); +double LAPACK_dlantr(char *norm, + char *uplo, + char *diag, + lapack_int *m, + lapack_int *n, + const double *a, + lapack_int *lda, + double *work); +float LAPACK_clantr(char *norm, + char *uplo, + char *diag, + lapack_int *m, + lapack_int *n, + const lapack_complex_float *a, + lapack_int *lda, + float *work); +double LAPACK_zlantr(char *norm, + char *uplo, + char *diag, + lapack_int *m, + lapack_int *n, + const lapack_complex_double *a, + lapack_int *lda, + double *work); +float LAPACK_slamch(char *cmach); +double LAPACK_dlamch(char *cmach); +void LAPACK_sgelq2(lapack_int *m, lapack_int *n, float *a, lapack_int *lda, float *tau, float *work, lapack_int *info); +void LAPACK_dgelq2(lapack_int *m, + lapack_int *n, + double *a, + lapack_int *lda, + double *tau, + double *work, + lapack_int *info); +void LAPACK_cgelq2(lapack_int *m, + lapack_int *n, + lapack_complex_float *a, + lapack_int *lda, + lapack_complex_float *tau, + lapack_complex_float *work, + lapack_int *info); +void LAPACK_zgelq2(lapack_int *m, + lapack_int *n, + lapack_complex_double *a, + lapack_int *lda, + lapack_complex_double *tau, + lapack_complex_double *work, + lapack_int *info); +void LAPACK_slarfb(char *side, + char *trans, + char *direct, + char *storev, + lapack_int *m, + lapack_int *n, + lapack_int *k, + const float *v, + lapack_int *ldv, + const float *t, + lapack_int *ldt, + float *c, + lapack_int *ldc, + float *work, + lapack_int *ldwork); +void LAPACK_dlarfb(char *side, + char *trans, + char *direct, + char *storev, + lapack_int *m, + lapack_int *n, + lapack_int *k, + const double *v, + lapack_int *ldv, + const double *t, + lapack_int *ldt, + double *c, + lapack_int *ldc, + double *work, + lapack_int *ldwork); +void LAPACK_clarfb(char *side, + char *trans, + char *direct, + char *storev, + lapack_int *m, + lapack_int *n, + lapack_int *k, + const lapack_complex_float *v, + lapack_int *ldv, + const lapack_complex_float *t, + lapack_int *ldt, + lapack_complex_float *c, + lapack_int *ldc, + lapack_complex_float *work, + lapack_int *ldwork); +void LAPACK_zlarfb(char *side, + char *trans, + char *direct, + char *storev, + lapack_int *m, + lapack_int *n, + lapack_int *k, + const lapack_complex_double *v, + lapack_int *ldv, + const lapack_complex_double *t, + lapack_int *ldt, + lapack_complex_double *c, + lapack_int *ldc, + lapack_complex_double *work, + lapack_int *ldwork); +void LAPACK_slarfg(lapack_int *n, float *alpha, float *x, lapack_int *incx, float *tau); +void LAPACK_dlarfg(lapack_int *n, double *alpha, double *x, lapack_int *incx, double *tau); +void LAPACK_clarfg(lapack_int *n, + lapack_complex_float *alpha, + lapack_complex_float *x, + lapack_int *incx, + lapack_complex_float *tau); +void LAPACK_zlarfg(lapack_int *n, + lapack_complex_double *alpha, + lapack_complex_double *x, + lapack_int *incx, + lapack_complex_double *tau); +void LAPACK_slarft(char *direct, + char *storev, + lapack_int *n, + lapack_int *k, + const float *v, + lapack_int *ldv, + const float *tau, + float *t, + lapack_int *ldt); +void LAPACK_dlarft(char *direct, + char *storev, + lapack_int *n, + lapack_int *k, + const double *v, + lapack_int *ldv, + const double *tau, + double *t, + lapack_int *ldt); +void LAPACK_clarft(char *direct, + char *storev, + lapack_int *n, + lapack_int *k, + const lapack_complex_float *v, + lapack_int *ldv, + const lapack_complex_float *tau, + lapack_complex_float *t, + lapack_int *ldt); +void LAPACK_zlarft(char *direct, + char *storev, + lapack_int *n, + lapack_int *k, + const lapack_complex_double *v, + lapack_int *ldv, + const lapack_complex_double *tau, + lapack_complex_double *t, + lapack_int *ldt); +void LAPACK_slarfx(char *side, + lapack_int *m, + lapack_int *n, + const float *v, + float *tau, + float *c, + lapack_int *ldc, + float *work); +void LAPACK_dlarfx(char *side, + lapack_int *m, + lapack_int *n, + const double *v, + double *tau, + double *c, + lapack_int *ldc, + double *work); +void LAPACK_clarfx(char *side, + lapack_int *m, + lapack_int *n, + const lapack_complex_float *v, + lapack_complex_float *tau, + lapack_complex_float *c, + lapack_int *ldc, + lapack_complex_float *work); +void LAPACK_zlarfx(char *side, + lapack_int *m, + lapack_int *n, + const lapack_complex_double *v, + lapack_complex_double *tau, + lapack_complex_double *c, + lapack_int *ldc, + lapack_complex_double *work); +void LAPACK_slatms(lapack_int *m, + lapack_int *n, + char *dist, + lapack_int *iseed, + char *sym, + float *d, + lapack_int *mode, + float *cond, + float *dmax, + lapack_int *kl, + lapack_int *ku, + char *pack, + float *a, + lapack_int *lda, + float *work, + lapack_int *info); +void LAPACK_dlatms(lapack_int *m, + lapack_int *n, + char *dist, + lapack_int *iseed, + char *sym, + double *d, + lapack_int *mode, + double *cond, + double *dmax, + lapack_int *kl, + lapack_int *ku, + char *pack, + double *a, + lapack_int *lda, + double *work, + lapack_int *info); +void LAPACK_clatms(lapack_int *m, + lapack_int *n, + char *dist, + lapack_int *iseed, + char *sym, + float *d, + lapack_int *mode, + float *cond, + float *dmax, + lapack_int *kl, + lapack_int *ku, + char *pack, + lapack_complex_float *a, + lapack_int *lda, + lapack_complex_float *work, + lapack_int *info); +void LAPACK_zlatms(lapack_int *m, + lapack_int *n, + char *dist, + lapack_int *iseed, + char *sym, + double *d, + lapack_int *mode, + double *cond, + double *dmax, + lapack_int *kl, + lapack_int *ku, + char *pack, + lapack_complex_double *a, + lapack_int *lda, + lapack_complex_double *work, + lapack_int *info); +void LAPACK_slag2d(lapack_int *m, + lapack_int *n, + const float *sa, + lapack_int *ldsa, + double *a, + lapack_int *lda, + lapack_int *info); +void LAPACK_dlag2s(lapack_int *m, + lapack_int *n, + const double *a, + lapack_int *lda, + float *sa, + lapack_int *ldsa, + lapack_int *info); +void LAPACK_clag2z(lapack_int *m, + lapack_int *n, + const lapack_complex_float *sa, + lapack_int *ldsa, + lapack_complex_double *a, + lapack_int *lda, + lapack_int *info); +void LAPACK_zlag2c(lapack_int *m, + lapack_int *n, + const lapack_complex_double *a, + lapack_int *lda, + lapack_complex_float *sa, + lapack_int *ldsa, + lapack_int *info); +void LAPACK_slauum(char *uplo, lapack_int *n, float *a, lapack_int *lda, lapack_int *info); +void LAPACK_dlauum(char *uplo, lapack_int *n, double *a, lapack_int *lda, lapack_int *info); +void LAPACK_clauum(char *uplo, lapack_int *n, lapack_complex_float *a, lapack_int *lda, lapack_int *info); +void LAPACK_zlauum(char *uplo, lapack_int *n, lapack_complex_double *a, lapack_int *lda, lapack_int *info); +void LAPACK_slagge(lapack_int *m, + lapack_int *n, + lapack_int *kl, + lapack_int *ku, + const float *d, + float *a, + lapack_int *lda, + lapack_int *iseed, + float *work, + lapack_int *info); +void LAPACK_dlagge(lapack_int *m, + lapack_int *n, + lapack_int *kl, + lapack_int *ku, + const double *d, + double *a, + lapack_int *lda, + lapack_int *iseed, + double *work, + lapack_int *info); +void LAPACK_clagge(lapack_int *m, + lapack_int *n, + lapack_int *kl, + lapack_int *ku, + const float *d, + lapack_complex_float *a, + lapack_int *lda, + lapack_int *iseed, + lapack_complex_float *work, + lapack_int *info); +void LAPACK_zlagge(lapack_int *m, + lapack_int *n, + lapack_int *kl, + lapack_int *ku, + const double *d, + lapack_complex_double *a, + lapack_int *lda, + lapack_int *iseed, + lapack_complex_double *work, + lapack_int *info); +void LAPACK_slaset(char *uplo, lapack_int *m, lapack_int *n, float *alpha, float *beta, float *a, lapack_int *lda); +void LAPACK_dlaset(char *uplo, lapack_int *m, lapack_int *n, double *alpha, double *beta, double *a, lapack_int *lda); +void LAPACK_claset(char *uplo, + lapack_int *m, + lapack_int *n, + lapack_complex_float *alpha, + lapack_complex_float *beta, + lapack_complex_float *a, + lapack_int *lda); +void LAPACK_zlaset(char *uplo, + lapack_int *m, + lapack_int *n, + lapack_complex_double *alpha, + lapack_complex_double *beta, + lapack_complex_double *a, + lapack_int *lda); +void LAPACK_slasrt(char *id, lapack_int *n, float *d, lapack_int *info); +void LAPACK_dlasrt(char *id, lapack_int *n, double *d, lapack_int *info); +void LAPACK_claghe(lapack_int *n, + lapack_int *k, + const float *d, + lapack_complex_float *a, + lapack_int *lda, + lapack_int *iseed, + lapack_complex_float *work, + lapack_int *info); +void LAPACK_zlaghe(lapack_int *n, + lapack_int *k, + const double *d, + lapack_complex_double *a, + lapack_int *lda, + lapack_int *iseed, + lapack_complex_double *work, + lapack_int *info); +void LAPACK_slagsy(lapack_int *n, + lapack_int *k, + const float *d, + float *a, + lapack_int *lda, + lapack_int *iseed, + float *work, + lapack_int *info); +void LAPACK_dlagsy(lapack_int *n, + lapack_int *k, + const double *d, + double *a, + lapack_int *lda, + lapack_int *iseed, + double *work, + lapack_int *info); +void LAPACK_clagsy(lapack_int *n, + lapack_int *k, + const float *d, + lapack_complex_float *a, + lapack_int *lda, + lapack_int *iseed, + lapack_complex_float *work, + lapack_int *info); +void LAPACK_zlagsy(lapack_int *n, + lapack_int *k, + const double *d, + lapack_complex_double *a, + lapack_int *lda, + lapack_int *iseed, + lapack_complex_double *work, + lapack_int *info); +void LAPACK_slapmr(lapack_logical *forwrd, lapack_int *m, lapack_int *n, float *x, lapack_int *ldx, lapack_int *k); +void LAPACK_dlapmr(lapack_logical *forwrd, lapack_int *m, lapack_int *n, double *x, lapack_int *ldx, lapack_int *k); +void LAPACK_clapmr(lapack_logical *forwrd, + lapack_int *m, + lapack_int *n, + lapack_complex_float *x, + lapack_int *ldx, + lapack_int *k); +void LAPACK_zlapmr(lapack_logical *forwrd, + lapack_int *m, + lapack_int *n, + lapack_complex_double *x, + lapack_int *ldx, + lapack_int *k); +float LAPACK_slapy2(float *x, float *y); +double LAPACK_dlapy2(double *x, double *y); +float LAPACK_slapy3(float *x, float *y, float *z); +double LAPACK_dlapy3(double *x, double *y, double *z); +void LAPACK_slartgp(float *f, float *g, float *cs, float *sn, float *r); +void LAPACK_dlartgp(double *f, double *g, double *cs, double *sn, double *r); +void LAPACK_slartgs(float *x, float *y, float *sigma, float *cs, float *sn); +void LAPACK_dlartgs(double *x, double *y, double *sigma, double *cs, double *sn); // LAPACK 3.3.0 -void LAPACK_cbbcsd( char* jobu1, char* jobu2, - char* jobv1t, char* jobv2t, char* trans, - lapack_int* m, lapack_int* p, lapack_int* q, - float* theta, float* phi, - lapack_complex_float* u1, lapack_int* ldu1, - lapack_complex_float* u2, lapack_int* ldu2, - lapack_complex_float* v1t, lapack_int* ldv1t, - lapack_complex_float* v2t, lapack_int* ldv2t, - float* b11d, float* b11e, float* b12d, - float* b12e, float* b21d, float* b21e, - float* b22d, float* b22e, float* rwork, - lapack_int* lrwork , lapack_int *info ); -void LAPACK_cheswapr( char* uplo, lapack_int* n, - lapack_complex_float* a, lapack_int* i1, - lapack_int* i2 ); -void LAPACK_chetri2( char* uplo, lapack_int* n, - lapack_complex_float* a, lapack_int* lda, - const lapack_int* ipiv, - lapack_complex_float* work, lapack_int* lwork , lapack_int *info ); -void LAPACK_chetri2x( char* uplo, lapack_int* n, - lapack_complex_float* a, lapack_int* lda, - const lapack_int* ipiv, - lapack_complex_float* work, lapack_int* nb , lapack_int *info ); -void LAPACK_chetrs2( char* uplo, lapack_int* n, - lapack_int* nrhs, const lapack_complex_float* a, - lapack_int* lda, const lapack_int* ipiv, - lapack_complex_float* b, lapack_int* ldb, - lapack_complex_float* work , lapack_int *info ); -void LAPACK_csyconv( char* uplo, char* way, - lapack_int* n, lapack_complex_float* a, - lapack_int* lda, const lapack_int* ipiv, - lapack_complex_float* work , lapack_int *info ); -void LAPACK_csyswapr( char* uplo, lapack_int* n, - lapack_complex_float* a, lapack_int* i1, - lapack_int* i2 ); -void LAPACK_csytri2( char* uplo, lapack_int* n, - lapack_complex_float* a, lapack_int* lda, - const lapack_int* ipiv, - lapack_complex_float* work, lapack_int* lwork , lapack_int *info ); -void LAPACK_csytri2x( char* uplo, lapack_int* n, - lapack_complex_float* a, lapack_int* lda, - const lapack_int* ipiv, - lapack_complex_float* work, lapack_int* nb , lapack_int *info ); -void LAPACK_csytrs2( char* uplo, lapack_int* n, - lapack_int* nrhs, const lapack_complex_float* a, - lapack_int* lda, const lapack_int* ipiv, - lapack_complex_float* b, lapack_int* ldb, - lapack_complex_float* work , lapack_int *info ); -void LAPACK_cunbdb( char* trans, char* signs, - lapack_int* m, lapack_int* p, lapack_int* q, - lapack_complex_float* x11, lapack_int* ldx11, - lapack_complex_float* x12, lapack_int* ldx12, - lapack_complex_float* x21, lapack_int* ldx21, - lapack_complex_float* x22, lapack_int* ldx22, - float* theta, float* phi, - lapack_complex_float* taup1, - lapack_complex_float* taup2, - lapack_complex_float* tauq1, - lapack_complex_float* tauq2, - lapack_complex_float* work, lapack_int* lwork , lapack_int *info ); -void LAPACK_cuncsd( char* jobu1, char* jobu2, - char* jobv1t, char* jobv2t, char* trans, - char* signs, lapack_int* m, lapack_int* p, - lapack_int* q, lapack_complex_float* x11, - lapack_int* ldx11, lapack_complex_float* x12, - lapack_int* ldx12, lapack_complex_float* x21, - lapack_int* ldx21, lapack_complex_float* x22, - lapack_int* ldx22, float* theta, - lapack_complex_float* u1, lapack_int* ldu1, - lapack_complex_float* u2, lapack_int* ldu2, - lapack_complex_float* v1t, lapack_int* ldv1t, - lapack_complex_float* v2t, lapack_int* ldv2t, - lapack_complex_float* work, lapack_int* lwork, - float* rwork, lapack_int* lrwork, - lapack_int* iwork , lapack_int *info ); -void LAPACK_dbbcsd( char* jobu1, char* jobu2, - char* jobv1t, char* jobv2t, char* trans, - lapack_int* m, lapack_int* p, lapack_int* q, - double* theta, double* phi, double* u1, - lapack_int* ldu1, double* u2, lapack_int* ldu2, - double* v1t, lapack_int* ldv1t, double* v2t, - lapack_int* ldv2t, double* b11d, double* b11e, - double* b12d, double* b12e, double* b21d, - double* b21e, double* b22d, double* b22e, - double* work, lapack_int* lwork , lapack_int *info ); -void LAPACK_dorbdb( char* trans, char* signs, - lapack_int* m, lapack_int* p, lapack_int* q, - double* x11, lapack_int* ldx11, double* x12, - lapack_int* ldx12, double* x21, lapack_int* ldx21, - double* x22, lapack_int* ldx22, double* theta, - double* phi, double* taup1, double* taup2, - double* tauq1, double* tauq2, double* work, - lapack_int* lwork , lapack_int *info ); -void LAPACK_dorcsd( char* jobu1, char* jobu2, - char* jobv1t, char* jobv2t, char* trans, - char* signs, lapack_int* m, lapack_int* p, - lapack_int* q, double* x11, lapack_int* ldx11, - double* x12, lapack_int* ldx12, double* x21, - lapack_int* ldx21, double* x22, lapack_int* ldx22, - double* theta, double* u1, lapack_int* ldu1, - double* u2, lapack_int* ldu2, double* v1t, - lapack_int* ldv1t, double* v2t, lapack_int* ldv2t, - double* work, lapack_int* lwork, - lapack_int* iwork , lapack_int *info ); -void LAPACK_dsyconv( char* uplo, char* way, - lapack_int* n, double* a, lapack_int* lda, - const lapack_int* ipiv, double* work , lapack_int *info ); -void LAPACK_dsyswapr( char* uplo, lapack_int* n, - double* a, lapack_int* i1, lapack_int* i2 ); -void LAPACK_dsytri2( char* uplo, lapack_int* n, - double* a, lapack_int* lda, - const lapack_int* ipiv, - lapack_complex_double* work, lapack_int* lwork , lapack_int *info ); -void LAPACK_dsytri2x( char* uplo, lapack_int* n, - double* a, lapack_int* lda, - const lapack_int* ipiv, double* work, - lapack_int* nb , lapack_int *info ); -void LAPACK_dsytrs2( char* uplo, lapack_int* n, - lapack_int* nrhs, const double* a, - lapack_int* lda, const lapack_int* ipiv, - double* b, lapack_int* ldb, double* work , lapack_int *info ); -void LAPACK_sbbcsd( char* jobu1, char* jobu2, - char* jobv1t, char* jobv2t, char* trans, - lapack_int* m, lapack_int* p, lapack_int* q, - float* theta, float* phi, float* u1, - lapack_int* ldu1, float* u2, lapack_int* ldu2, - float* v1t, lapack_int* ldv1t, float* v2t, - lapack_int* ldv2t, float* b11d, float* b11e, - float* b12d, float* b12e, float* b21d, - float* b21e, float* b22d, float* b22e, - float* work, lapack_int* lwork , lapack_int *info ); -void LAPACK_sorbdb( char* trans, char* signs, - lapack_int* m, lapack_int* p, lapack_int* q, - float* x11, lapack_int* ldx11, float* x12, - lapack_int* ldx12, float* x21, lapack_int* ldx21, - float* x22, lapack_int* ldx22, float* theta, - float* phi, float* taup1, float* taup2, - float* tauq1, float* tauq2, float* work, - lapack_int* lwork , lapack_int *info ); -void LAPACK_sorcsd( char* jobu1, char* jobu2, - char* jobv1t, char* jobv2t, char* trans, - char* signs, lapack_int* m, lapack_int* p, - lapack_int* q, float* x11, lapack_int* ldx11, - float* x12, lapack_int* ldx12, float* x21, - lapack_int* ldx21, float* x22, lapack_int* ldx22, - float* theta, float* u1, lapack_int* ldu1, - float* u2, lapack_int* ldu2, float* v1t, - lapack_int* ldv1t, float* v2t, lapack_int* ldv2t, - float* work, lapack_int* lwork, - lapack_int* iwork , lapack_int *info ); -void LAPACK_ssyconv( char* uplo, char* way, - lapack_int* n, float* a, lapack_int* lda, - const lapack_int* ipiv, float* work , lapack_int *info ); -void LAPACK_ssyswapr( char* uplo, lapack_int* n, - float* a, lapack_int* i1, lapack_int* i2 ); -void LAPACK_ssytri2( char* uplo, lapack_int* n, - float* a, lapack_int* lda, - const lapack_int* ipiv, - lapack_complex_float* work, lapack_int* lwork , lapack_int *info ); -void LAPACK_ssytri2x( char* uplo, lapack_int* n, - float* a, lapack_int* lda, - const lapack_int* ipiv, float* work, - lapack_int* nb , lapack_int *info ); -void LAPACK_ssytrs2( char* uplo, lapack_int* n, - lapack_int* nrhs, const float* a, - lapack_int* lda, const lapack_int* ipiv, - float* b, lapack_int* ldb, float* work , lapack_int *info ); -void LAPACK_zbbcsd( char* jobu1, char* jobu2, - char* jobv1t, char* jobv2t, char* trans, - lapack_int* m, lapack_int* p, lapack_int* q, - double* theta, double* phi, - lapack_complex_double* u1, lapack_int* ldu1, - lapack_complex_double* u2, lapack_int* ldu2, - lapack_complex_double* v1t, lapack_int* ldv1t, - lapack_complex_double* v2t, lapack_int* ldv2t, - double* b11d, double* b11e, double* b12d, - double* b12e, double* b21d, double* b21e, - double* b22d, double* b22e, double* rwork, - lapack_int* lrwork , lapack_int *info ); -void LAPACK_zheswapr( char* uplo, lapack_int* n, - lapack_complex_double* a, lapack_int* i1, - lapack_int* i2 ); -void LAPACK_zhetri2( char* uplo, lapack_int* n, - lapack_complex_double* a, lapack_int* lda, - const lapack_int* ipiv, - lapack_complex_double* work, lapack_int* lwork , lapack_int *info ); -void LAPACK_zhetri2x( char* uplo, lapack_int* n, - lapack_complex_double* a, lapack_int* lda, - const lapack_int* ipiv, - lapack_complex_double* work, lapack_int* nb , lapack_int *info ); -void LAPACK_zhetrs2( char* uplo, lapack_int* n, - lapack_int* nrhs, - const lapack_complex_double* a, lapack_int* lda, - const lapack_int* ipiv, - lapack_complex_double* b, lapack_int* ldb, - lapack_complex_double* work , lapack_int *info ); -void LAPACK_zsyconv( char* uplo, char* way, - lapack_int* n, lapack_complex_double* a, - lapack_int* lda, const lapack_int* ipiv, - lapack_complex_double* work , lapack_int *info ); -void LAPACK_zsyswapr( char* uplo, lapack_int* n, - lapack_complex_double* a, lapack_int* i1, - lapack_int* i2 ); -void LAPACK_zsytri2( char* uplo, lapack_int* n, - lapack_complex_double* a, lapack_int* lda, - const lapack_int* ipiv, - lapack_complex_double* work, lapack_int* lwork , lapack_int *info ); -void LAPACK_zsytri2x( char* uplo, lapack_int* n, - lapack_complex_double* a, lapack_int* lda, - const lapack_int* ipiv, - lapack_complex_double* work, lapack_int* nb , lapack_int *info ); -void LAPACK_zsytrs2( char* uplo, lapack_int* n, - lapack_int* nrhs, - const lapack_complex_double* a, lapack_int* lda, - const lapack_int* ipiv, - lapack_complex_double* b, lapack_int* ldb, - lapack_complex_double* work , lapack_int *info ); -void LAPACK_zunbdb( char* trans, char* signs, - lapack_int* m, lapack_int* p, lapack_int* q, - lapack_complex_double* x11, lapack_int* ldx11, - lapack_complex_double* x12, lapack_int* ldx12, - lapack_complex_double* x21, lapack_int* ldx21, - lapack_complex_double* x22, lapack_int* ldx22, - double* theta, double* phi, - lapack_complex_double* taup1, - lapack_complex_double* taup2, - lapack_complex_double* tauq1, - lapack_complex_double* tauq2, - lapack_complex_double* work, lapack_int* lwork , lapack_int *info ); -void LAPACK_zuncsd( char* jobu1, char* jobu2, - char* jobv1t, char* jobv2t, char* trans, - char* signs, lapack_int* m, lapack_int* p, - lapack_int* q, lapack_complex_double* x11, - lapack_int* ldx11, lapack_complex_double* x12, - lapack_int* ldx12, lapack_complex_double* x21, - lapack_int* ldx21, lapack_complex_double* x22, - lapack_int* ldx22, double* theta, - lapack_complex_double* u1, lapack_int* ldu1, - lapack_complex_double* u2, lapack_int* ldu2, - lapack_complex_double* v1t, lapack_int* ldv1t, - lapack_complex_double* v2t, lapack_int* ldv2t, - lapack_complex_double* work, lapack_int* lwork, - double* rwork, lapack_int* lrwork, - lapack_int* iwork , lapack_int *info ); +void LAPACK_cbbcsd(char *jobu1, + char *jobu2, + char *jobv1t, + char *jobv2t, + char *trans, + lapack_int *m, + lapack_int *p, + lapack_int *q, + float *theta, + float *phi, + lapack_complex_float *u1, + lapack_int *ldu1, + lapack_complex_float *u2, + lapack_int *ldu2, + lapack_complex_float *v1t, + lapack_int *ldv1t, + lapack_complex_float *v2t, + lapack_int *ldv2t, + float *b11d, + float *b11e, + float *b12d, + float *b12e, + float *b21d, + float *b21e, + float *b22d, + float *b22e, + float *rwork, + lapack_int *lrwork, + lapack_int *info); +void LAPACK_cheswapr(char *uplo, lapack_int *n, lapack_complex_float *a, lapack_int *i1, lapack_int *i2); +void LAPACK_chetri2(char *uplo, + lapack_int *n, + lapack_complex_float *a, + lapack_int *lda, + const lapack_int *ipiv, + lapack_complex_float *work, + lapack_int *lwork, + lapack_int *info); +void LAPACK_chetri2x(char *uplo, + lapack_int *n, + lapack_complex_float *a, + lapack_int *lda, + const lapack_int *ipiv, + lapack_complex_float *work, + lapack_int *nb, + lapack_int *info); +void LAPACK_chetrs2(char *uplo, + lapack_int *n, + lapack_int *nrhs, + const lapack_complex_float *a, + lapack_int *lda, + const lapack_int *ipiv, + lapack_complex_float *b, + lapack_int *ldb, + lapack_complex_float *work, + lapack_int *info); +void LAPACK_csyconv(char *uplo, + char *way, + lapack_int *n, + lapack_complex_float *a, + lapack_int *lda, + const lapack_int *ipiv, + lapack_complex_float *work, + lapack_int *info); +void LAPACK_csyswapr(char *uplo, lapack_int *n, lapack_complex_float *a, lapack_int *i1, lapack_int *i2); +void LAPACK_csytri2(char *uplo, + lapack_int *n, + lapack_complex_float *a, + lapack_int *lda, + const lapack_int *ipiv, + lapack_complex_float *work, + lapack_int *lwork, + lapack_int *info); +void LAPACK_csytri2x(char *uplo, + lapack_int *n, + lapack_complex_float *a, + lapack_int *lda, + const lapack_int *ipiv, + lapack_complex_float *work, + lapack_int *nb, + lapack_int *info); +void LAPACK_csytrs2(char *uplo, + lapack_int *n, + lapack_int *nrhs, + const lapack_complex_float *a, + lapack_int *lda, + const lapack_int *ipiv, + lapack_complex_float *b, + lapack_int *ldb, + lapack_complex_float *work, + lapack_int *info); +void LAPACK_cunbdb(char *trans, + char *signs, + lapack_int *m, + lapack_int *p, + lapack_int *q, + lapack_complex_float *x11, + lapack_int *ldx11, + lapack_complex_float *x12, + lapack_int *ldx12, + lapack_complex_float *x21, + lapack_int *ldx21, + lapack_complex_float *x22, + lapack_int *ldx22, + float *theta, + float *phi, + lapack_complex_float *taup1, + lapack_complex_float *taup2, + lapack_complex_float *tauq1, + lapack_complex_float *tauq2, + lapack_complex_float *work, + lapack_int *lwork, + lapack_int *info); +void LAPACK_cuncsd(char *jobu1, + char *jobu2, + char *jobv1t, + char *jobv2t, + char *trans, + char *signs, + lapack_int *m, + lapack_int *p, + lapack_int *q, + lapack_complex_float *x11, + lapack_int *ldx11, + lapack_complex_float *x12, + lapack_int *ldx12, + lapack_complex_float *x21, + lapack_int *ldx21, + lapack_complex_float *x22, + lapack_int *ldx22, + float *theta, + lapack_complex_float *u1, + lapack_int *ldu1, + lapack_complex_float *u2, + lapack_int *ldu2, + lapack_complex_float *v1t, + lapack_int *ldv1t, + lapack_complex_float *v2t, + lapack_int *ldv2t, + lapack_complex_float *work, + lapack_int *lwork, + float *rwork, + lapack_int *lrwork, + lapack_int *iwork, + lapack_int *info); +void LAPACK_dbbcsd(char *jobu1, + char *jobu2, + char *jobv1t, + char *jobv2t, + char *trans, + lapack_int *m, + lapack_int *p, + lapack_int *q, + double *theta, + double *phi, + double *u1, + lapack_int *ldu1, + double *u2, + lapack_int *ldu2, + double *v1t, + lapack_int *ldv1t, + double *v2t, + lapack_int *ldv2t, + double *b11d, + double *b11e, + double *b12d, + double *b12e, + double *b21d, + double *b21e, + double *b22d, + double *b22e, + double *work, + lapack_int *lwork, + lapack_int *info); +void LAPACK_dorbdb(char *trans, + char *signs, + lapack_int *m, + lapack_int *p, + lapack_int *q, + double *x11, + lapack_int *ldx11, + double *x12, + lapack_int *ldx12, + double *x21, + lapack_int *ldx21, + double *x22, + lapack_int *ldx22, + double *theta, + double *phi, + double *taup1, + double *taup2, + double *tauq1, + double *tauq2, + double *work, + lapack_int *lwork, + lapack_int *info); +void LAPACK_dorcsd(char *jobu1, + char *jobu2, + char *jobv1t, + char *jobv2t, + char *trans, + char *signs, + lapack_int *m, + lapack_int *p, + lapack_int *q, + double *x11, + lapack_int *ldx11, + double *x12, + lapack_int *ldx12, + double *x21, + lapack_int *ldx21, + double *x22, + lapack_int *ldx22, + double *theta, + double *u1, + lapack_int *ldu1, + double *u2, + lapack_int *ldu2, + double *v1t, + lapack_int *ldv1t, + double *v2t, + lapack_int *ldv2t, + double *work, + lapack_int *lwork, + lapack_int *iwork, + lapack_int *info); +void LAPACK_dsyconv(char *uplo, + char *way, + lapack_int *n, + double *a, + lapack_int *lda, + const lapack_int *ipiv, + double *work, + lapack_int *info); +void LAPACK_dsyswapr(char *uplo, lapack_int *n, double *a, lapack_int *i1, lapack_int *i2); +void LAPACK_dsytri2(char *uplo, + lapack_int *n, + double *a, + lapack_int *lda, + const lapack_int *ipiv, + lapack_complex_double *work, + lapack_int *lwork, + lapack_int *info); +void LAPACK_dsytri2x(char *uplo, + lapack_int *n, + double *a, + lapack_int *lda, + const lapack_int *ipiv, + double *work, + lapack_int *nb, + lapack_int *info); +void LAPACK_dsytrs2(char *uplo, + lapack_int *n, + lapack_int *nrhs, + const double *a, + lapack_int *lda, + const lapack_int *ipiv, + double *b, + lapack_int *ldb, + double *work, + lapack_int *info); +void LAPACK_sbbcsd(char *jobu1, + char *jobu2, + char *jobv1t, + char *jobv2t, + char *trans, + lapack_int *m, + lapack_int *p, + lapack_int *q, + float *theta, + float *phi, + float *u1, + lapack_int *ldu1, + float *u2, + lapack_int *ldu2, + float *v1t, + lapack_int *ldv1t, + float *v2t, + lapack_int *ldv2t, + float *b11d, + float *b11e, + float *b12d, + float *b12e, + float *b21d, + float *b21e, + float *b22d, + float *b22e, + float *work, + lapack_int *lwork, + lapack_int *info); +void LAPACK_sorbdb(char *trans, + char *signs, + lapack_int *m, + lapack_int *p, + lapack_int *q, + float *x11, + lapack_int *ldx11, + float *x12, + lapack_int *ldx12, + float *x21, + lapack_int *ldx21, + float *x22, + lapack_int *ldx22, + float *theta, + float *phi, + float *taup1, + float *taup2, + float *tauq1, + float *tauq2, + float *work, + lapack_int *lwork, + lapack_int *info); +void LAPACK_sorcsd(char *jobu1, + char *jobu2, + char *jobv1t, + char *jobv2t, + char *trans, + char *signs, + lapack_int *m, + lapack_int *p, + lapack_int *q, + float *x11, + lapack_int *ldx11, + float *x12, + lapack_int *ldx12, + float *x21, + lapack_int *ldx21, + float *x22, + lapack_int *ldx22, + float *theta, + float *u1, + lapack_int *ldu1, + float *u2, + lapack_int *ldu2, + float *v1t, + lapack_int *ldv1t, + float *v2t, + lapack_int *ldv2t, + float *work, + lapack_int *lwork, + lapack_int *iwork, + lapack_int *info); +void LAPACK_ssyconv(char *uplo, + char *way, + lapack_int *n, + float *a, + lapack_int *lda, + const lapack_int *ipiv, + float *work, + lapack_int *info); +void LAPACK_ssyswapr(char *uplo, lapack_int *n, float *a, lapack_int *i1, lapack_int *i2); +void LAPACK_ssytri2(char *uplo, + lapack_int *n, + float *a, + lapack_int *lda, + const lapack_int *ipiv, + lapack_complex_float *work, + lapack_int *lwork, + lapack_int *info); +void LAPACK_ssytri2x(char *uplo, + lapack_int *n, + float *a, + lapack_int *lda, + const lapack_int *ipiv, + float *work, + lapack_int *nb, + lapack_int *info); +void LAPACK_ssytrs2(char *uplo, + lapack_int *n, + lapack_int *nrhs, + const float *a, + lapack_int *lda, + const lapack_int *ipiv, + float *b, + lapack_int *ldb, + float *work, + lapack_int *info); +void LAPACK_zbbcsd(char *jobu1, + char *jobu2, + char *jobv1t, + char *jobv2t, + char *trans, + lapack_int *m, + lapack_int *p, + lapack_int *q, + double *theta, + double *phi, + lapack_complex_double *u1, + lapack_int *ldu1, + lapack_complex_double *u2, + lapack_int *ldu2, + lapack_complex_double *v1t, + lapack_int *ldv1t, + lapack_complex_double *v2t, + lapack_int *ldv2t, + double *b11d, + double *b11e, + double *b12d, + double *b12e, + double *b21d, + double *b21e, + double *b22d, + double *b22e, + double *rwork, + lapack_int *lrwork, + lapack_int *info); +void LAPACK_zheswapr(char *uplo, lapack_int *n, lapack_complex_double *a, lapack_int *i1, lapack_int *i2); +void LAPACK_zhetri2(char *uplo, + lapack_int *n, + lapack_complex_double *a, + lapack_int *lda, + const lapack_int *ipiv, + lapack_complex_double *work, + lapack_int *lwork, + lapack_int *info); +void LAPACK_zhetri2x(char *uplo, + lapack_int *n, + lapack_complex_double *a, + lapack_int *lda, + const lapack_int *ipiv, + lapack_complex_double *work, + lapack_int *nb, + lapack_int *info); +void LAPACK_zhetrs2(char *uplo, + lapack_int *n, + lapack_int *nrhs, + const lapack_complex_double *a, + lapack_int *lda, + const lapack_int *ipiv, + lapack_complex_double *b, + lapack_int *ldb, + lapack_complex_double *work, + lapack_int *info); +void LAPACK_zsyconv(char *uplo, + char *way, + lapack_int *n, + lapack_complex_double *a, + lapack_int *lda, + const lapack_int *ipiv, + lapack_complex_double *work, + lapack_int *info); +void LAPACK_zsyswapr(char *uplo, lapack_int *n, lapack_complex_double *a, lapack_int *i1, lapack_int *i2); +void LAPACK_zsytri2(char *uplo, + lapack_int *n, + lapack_complex_double *a, + lapack_int *lda, + const lapack_int *ipiv, + lapack_complex_double *work, + lapack_int *lwork, + lapack_int *info); +void LAPACK_zsytri2x(char *uplo, + lapack_int *n, + lapack_complex_double *a, + lapack_int *lda, + const lapack_int *ipiv, + lapack_complex_double *work, + lapack_int *nb, + lapack_int *info); +void LAPACK_zsytrs2(char *uplo, + lapack_int *n, + lapack_int *nrhs, + const lapack_complex_double *a, + lapack_int *lda, + const lapack_int *ipiv, + lapack_complex_double *b, + lapack_int *ldb, + lapack_complex_double *work, + lapack_int *info); +void LAPACK_zunbdb(char *trans, + char *signs, + lapack_int *m, + lapack_int *p, + lapack_int *q, + lapack_complex_double *x11, + lapack_int *ldx11, + lapack_complex_double *x12, + lapack_int *ldx12, + lapack_complex_double *x21, + lapack_int *ldx21, + lapack_complex_double *x22, + lapack_int *ldx22, + double *theta, + double *phi, + lapack_complex_double *taup1, + lapack_complex_double *taup2, + lapack_complex_double *tauq1, + lapack_complex_double *tauq2, + lapack_complex_double *work, + lapack_int *lwork, + lapack_int *info); +void LAPACK_zuncsd(char *jobu1, + char *jobu2, + char *jobv1t, + char *jobv2t, + char *trans, + char *signs, + lapack_int *m, + lapack_int *p, + lapack_int *q, + lapack_complex_double *x11, + lapack_int *ldx11, + lapack_complex_double *x12, + lapack_int *ldx12, + lapack_complex_double *x21, + lapack_int *ldx21, + lapack_complex_double *x22, + lapack_int *ldx22, + double *theta, + lapack_complex_double *u1, + lapack_int *ldu1, + lapack_complex_double *u2, + lapack_int *ldu2, + lapack_complex_double *v1t, + lapack_int *ldv1t, + lapack_complex_double *v2t, + lapack_int *ldv2t, + lapack_complex_double *work, + lapack_int *lwork, + double *rwork, + lapack_int *lrwork, + lapack_int *iwork, + lapack_int *info); // LAPACK 3.4.0 -void LAPACK_sgemqrt( char* side, char* trans, lapack_int* m, lapack_int* n, - lapack_int* k, lapack_int* nb, const float* v, - lapack_int* ldv, const float* t, lapack_int* ldt, float* c, - lapack_int* ldc, float* work, lapack_int *info ); -void LAPACK_dgemqrt( char* side, char* trans, lapack_int* m, lapack_int* n, - lapack_int* k, lapack_int* nb, const double* v, - lapack_int* ldv, const double* t, lapack_int* ldt, - double* c, lapack_int* ldc, double* work, - lapack_int *info ); -void LAPACK_cgemqrt( char* side, char* trans, lapack_int* m, lapack_int* n, - lapack_int* k, lapack_int* nb, - const lapack_complex_float* v, lapack_int* ldv, - const lapack_complex_float* t, lapack_int* ldt, - lapack_complex_float* c, lapack_int* ldc, - lapack_complex_float* work, lapack_int *info ); -void LAPACK_zgemqrt( char* side, char* trans, lapack_int* m, lapack_int* n, - lapack_int* k, lapack_int* nb, - const lapack_complex_double* v, lapack_int* ldv, - const lapack_complex_double* t, lapack_int* ldt, - lapack_complex_double* c, lapack_int* ldc, - lapack_complex_double* work, lapack_int *info ); -void LAPACK_sgeqrt( lapack_int* m, lapack_int* n, lapack_int* nb, float* a, - lapack_int* lda, float* t, lapack_int* ldt, float* work, - lapack_int *info ); -void LAPACK_dgeqrt( lapack_int* m, lapack_int* n, lapack_int* nb, double* a, - lapack_int* lda, double* t, lapack_int* ldt, double* work, - lapack_int *info ); -void LAPACK_cgeqrt( lapack_int* m, lapack_int* n, lapack_int* nb, - lapack_complex_float* a, lapack_int* lda, - lapack_complex_float* t, lapack_int* ldt, - lapack_complex_float* work, lapack_int *info ); -void LAPACK_zgeqrt( lapack_int* m, lapack_int* n, lapack_int* nb, - lapack_complex_double* a, lapack_int* lda, - lapack_complex_double* t, lapack_int* ldt, - lapack_complex_double* work, lapack_int *info ); -void LAPACK_sgeqrt2( lapack_int* m, lapack_int* n, float* a, lapack_int* lda, - float* t, lapack_int* ldt, lapack_int *info ); -void LAPACK_dgeqrt2( lapack_int* m, lapack_int* n, double* a, lapack_int* lda, - double* t, lapack_int* ldt, lapack_int *info ); -void LAPACK_cgeqrt2( lapack_int* m, lapack_int* n, lapack_complex_float* a, - lapack_int* lda, lapack_complex_float* t, lapack_int* ldt, - lapack_int *info ); -void LAPACK_zgeqrt2( lapack_int* m, lapack_int* n, lapack_complex_double* a, - lapack_int* lda, lapack_complex_double* t, lapack_int* ldt, - lapack_int *info ); -void LAPACK_sgeqrt3( lapack_int* m, lapack_int* n, float* a, lapack_int* lda, - float* t, lapack_int* ldt, lapack_int *info ); -void LAPACK_dgeqrt3( lapack_int* m, lapack_int* n, double* a, lapack_int* lda, - double* t, lapack_int* ldt, lapack_int *info ); -void LAPACK_cgeqrt3( lapack_int* m, lapack_int* n, lapack_complex_float* a, - lapack_int* lda, lapack_complex_float* t, lapack_int* ldt, - lapack_int *info ); -void LAPACK_zgeqrt3( lapack_int* m, lapack_int* n, lapack_complex_double* a, - lapack_int* lda, lapack_complex_double* t, lapack_int* ldt, - lapack_int *info ); -void LAPACK_stpmqrt( char* side, char* trans, lapack_int* m, lapack_int* n, - lapack_int* k, lapack_int* l, lapack_int* nb, - const float* v, lapack_int* ldv, const float* t, - lapack_int* ldt, float* a, lapack_int* lda, float* b, - lapack_int* ldb, float* work, lapack_int *info ); -void LAPACK_dtpmqrt( char* side, char* trans, lapack_int* m, lapack_int* n, - lapack_int* k, lapack_int* l, lapack_int* nb, - const double* v, lapack_int* ldv, const double* t, - lapack_int* ldt, double* a, lapack_int* lda, double* b, - lapack_int* ldb, double* work, lapack_int *info ); -void LAPACK_ctpmqrt( char* side, char* trans, lapack_int* m, lapack_int* n, - lapack_int* k, lapack_int* l, lapack_int* nb, - const lapack_complex_float* v, lapack_int* ldv, - const lapack_complex_float* t, lapack_int* ldt, - lapack_complex_float* a, lapack_int* lda, - lapack_complex_float* b, lapack_int* ldb, - lapack_complex_float* work, lapack_int *info ); -void LAPACK_ztpmqrt( char* side, char* trans, lapack_int* m, lapack_int* n, - lapack_int* k, lapack_int* l, lapack_int* nb, - const lapack_complex_double* v, lapack_int* ldv, - const lapack_complex_double* t, lapack_int* ldt, - lapack_complex_double* a, lapack_int* lda, - lapack_complex_double* b, lapack_int* ldb, - lapack_complex_double* work, lapack_int *info ); -void LAPACK_dtpqrt( lapack_int* m, lapack_int* n, lapack_int* l, lapack_int* nb, - double* a, lapack_int* lda, double* b, lapack_int* ldb, - double* t, lapack_int* ldt, double* work, - lapack_int *info ); -void LAPACK_ctpqrt( lapack_int* m, lapack_int* n, lapack_int* l, lapack_int* nb, - lapack_complex_float* a, lapack_int* lda, - lapack_complex_float* t, lapack_complex_float* b, - lapack_int* ldb, lapack_int* ldt, - lapack_complex_float* work, lapack_int *info ); -void LAPACK_ztpqrt( lapack_int* m, lapack_int* n, lapack_int* l, lapack_int* nb, - lapack_complex_double* a, lapack_int* lda, - lapack_complex_double* b, lapack_int* ldb, - lapack_complex_double* t, lapack_int* ldt, - lapack_complex_double* work, lapack_int *info ); -void LAPACK_stpqrt2( lapack_int* m, lapack_int* n, float* a, lapack_int* lda, - float* b, lapack_int* ldb, float* t, lapack_int* ldt, - lapack_int *info ); -void LAPACK_dtpqrt2( lapack_int* m, lapack_int* n, double* a, lapack_int* lda, - double* b, lapack_int* ldb, double* t, lapack_int* ldt, - lapack_int *info ); -void LAPACK_ctpqrt2( lapack_int* m, lapack_int* n, lapack_complex_float* a, - lapack_int* lda, lapack_complex_float* b, lapack_int* ldb, - lapack_complex_float* t, lapack_int* ldt, - lapack_int *info ); -void LAPACK_ztpqrt2( lapack_int* m, lapack_int* n, lapack_complex_double* a, - lapack_int* lda, lapack_complex_double* b, lapack_int* ldb, - lapack_complex_double* t, lapack_int* ldt, - lapack_int *info ); -void LAPACK_stprfb( char* side, char* trans, char* direct, char* storev, - lapack_int* m, lapack_int* n, lapack_int* k, lapack_int* l, - const float* v, lapack_int* ldv, const float* t, - lapack_int* ldt, float* a, lapack_int* lda, float* b, - lapack_int* ldb, const float* mywork, - lapack_int* myldwork ); -void LAPACK_dtprfb( char* side, char* trans, char* direct, char* storev, - lapack_int* m, lapack_int* n, lapack_int* k, lapack_int* l, - const double* v, lapack_int* ldv, const double* t, - lapack_int* ldt, double* a, lapack_int* lda, double* b, - lapack_int* ldb, const double* mywork, - lapack_int* myldwork ); -void LAPACK_ctprfb( char* side, char* trans, char* direct, char* storev, - lapack_int* m, lapack_int* n, lapack_int* k, lapack_int* l, - const lapack_complex_float* v, lapack_int* ldv, - const lapack_complex_float* t, lapack_int* ldt, - lapack_complex_float* a, lapack_int* lda, - lapack_complex_float* b, lapack_int* ldb, - const float* mywork, lapack_int* myldwork ); -void LAPACK_ztprfb( char* side, char* trans, char* direct, char* storev, - lapack_int* m, lapack_int* n, lapack_int* k, lapack_int* l, - const lapack_complex_double* v, lapack_int* ldv, - const lapack_complex_double* t, lapack_int* ldt, - lapack_complex_double* a, lapack_int* lda, - lapack_complex_double* b, lapack_int* ldb, - const double* mywork, lapack_int* myldwork ); +void LAPACK_sgemqrt(char *side, + char *trans, + lapack_int *m, + lapack_int *n, + lapack_int *k, + lapack_int *nb, + const float *v, + lapack_int *ldv, + const float *t, + lapack_int *ldt, + float *c, + lapack_int *ldc, + float *work, + lapack_int *info); +void LAPACK_dgemqrt(char *side, + char *trans, + lapack_int *m, + lapack_int *n, + lapack_int *k, + lapack_int *nb, + const double *v, + lapack_int *ldv, + const double *t, + lapack_int *ldt, + double *c, + lapack_int *ldc, + double *work, + lapack_int *info); +void LAPACK_cgemqrt(char *side, + char *trans, + lapack_int *m, + lapack_int *n, + lapack_int *k, + lapack_int *nb, + const lapack_complex_float *v, + lapack_int *ldv, + const lapack_complex_float *t, + lapack_int *ldt, + lapack_complex_float *c, + lapack_int *ldc, + lapack_complex_float *work, + lapack_int *info); +void LAPACK_zgemqrt(char *side, + char *trans, + lapack_int *m, + lapack_int *n, + lapack_int *k, + lapack_int *nb, + const lapack_complex_double *v, + lapack_int *ldv, + const lapack_complex_double *t, + lapack_int *ldt, + lapack_complex_double *c, + lapack_int *ldc, + lapack_complex_double *work, + lapack_int *info); +void LAPACK_sgeqrt(lapack_int *m, + lapack_int *n, + lapack_int *nb, + float *a, + lapack_int *lda, + float *t, + lapack_int *ldt, + float *work, + lapack_int *info); +void LAPACK_dgeqrt(lapack_int *m, + lapack_int *n, + lapack_int *nb, + double *a, + lapack_int *lda, + double *t, + lapack_int *ldt, + double *work, + lapack_int *info); +void LAPACK_cgeqrt(lapack_int *m, + lapack_int *n, + lapack_int *nb, + lapack_complex_float *a, + lapack_int *lda, + lapack_complex_float *t, + lapack_int *ldt, + lapack_complex_float *work, + lapack_int *info); +void LAPACK_zgeqrt(lapack_int *m, + lapack_int *n, + lapack_int *nb, + lapack_complex_double *a, + lapack_int *lda, + lapack_complex_double *t, + lapack_int *ldt, + lapack_complex_double *work, + lapack_int *info); +void LAPACK_sgeqrt2(lapack_int *m, + lapack_int *n, + float *a, + lapack_int *lda, + float *t, + lapack_int *ldt, + lapack_int *info); +void LAPACK_dgeqrt2(lapack_int *m, + lapack_int *n, + double *a, + lapack_int *lda, + double *t, + lapack_int *ldt, + lapack_int *info); +void LAPACK_cgeqrt2(lapack_int *m, + lapack_int *n, + lapack_complex_float *a, + lapack_int *lda, + lapack_complex_float *t, + lapack_int *ldt, + lapack_int *info); +void LAPACK_zgeqrt2(lapack_int *m, + lapack_int *n, + lapack_complex_double *a, + lapack_int *lda, + lapack_complex_double *t, + lapack_int *ldt, + lapack_int *info); +void LAPACK_sgeqrt3(lapack_int *m, + lapack_int *n, + float *a, + lapack_int *lda, + float *t, + lapack_int *ldt, + lapack_int *info); +void LAPACK_dgeqrt3(lapack_int *m, + lapack_int *n, + double *a, + lapack_int *lda, + double *t, + lapack_int *ldt, + lapack_int *info); +void LAPACK_cgeqrt3(lapack_int *m, + lapack_int *n, + lapack_complex_float *a, + lapack_int *lda, + lapack_complex_float *t, + lapack_int *ldt, + lapack_int *info); +void LAPACK_zgeqrt3(lapack_int *m, + lapack_int *n, + lapack_complex_double *a, + lapack_int *lda, + lapack_complex_double *t, + lapack_int *ldt, + lapack_int *info); +void LAPACK_stpmqrt(char *side, + char *trans, + lapack_int *m, + lapack_int *n, + lapack_int *k, + lapack_int *l, + lapack_int *nb, + const float *v, + lapack_int *ldv, + const float *t, + lapack_int *ldt, + float *a, + lapack_int *lda, + float *b, + lapack_int *ldb, + float *work, + lapack_int *info); +void LAPACK_dtpmqrt(char *side, + char *trans, + lapack_int *m, + lapack_int *n, + lapack_int *k, + lapack_int *l, + lapack_int *nb, + const double *v, + lapack_int *ldv, + const double *t, + lapack_int *ldt, + double *a, + lapack_int *lda, + double *b, + lapack_int *ldb, + double *work, + lapack_int *info); +void LAPACK_ctpmqrt(char *side, + char *trans, + lapack_int *m, + lapack_int *n, + lapack_int *k, + lapack_int *l, + lapack_int *nb, + const lapack_complex_float *v, + lapack_int *ldv, + const lapack_complex_float *t, + lapack_int *ldt, + lapack_complex_float *a, + lapack_int *lda, + lapack_complex_float *b, + lapack_int *ldb, + lapack_complex_float *work, + lapack_int *info); +void LAPACK_ztpmqrt(char *side, + char *trans, + lapack_int *m, + lapack_int *n, + lapack_int *k, + lapack_int *l, + lapack_int *nb, + const lapack_complex_double *v, + lapack_int *ldv, + const lapack_complex_double *t, + lapack_int *ldt, + lapack_complex_double *a, + lapack_int *lda, + lapack_complex_double *b, + lapack_int *ldb, + lapack_complex_double *work, + lapack_int *info); +void LAPACK_dtpqrt(lapack_int *m, + lapack_int *n, + lapack_int *l, + lapack_int *nb, + double *a, + lapack_int *lda, + double *b, + lapack_int *ldb, + double *t, + lapack_int *ldt, + double *work, + lapack_int *info); +void LAPACK_ctpqrt(lapack_int *m, + lapack_int *n, + lapack_int *l, + lapack_int *nb, + lapack_complex_float *a, + lapack_int *lda, + lapack_complex_float *t, + lapack_complex_float *b, + lapack_int *ldb, + lapack_int *ldt, + lapack_complex_float *work, + lapack_int *info); +void LAPACK_ztpqrt(lapack_int *m, + lapack_int *n, + lapack_int *l, + lapack_int *nb, + lapack_complex_double *a, + lapack_int *lda, + lapack_complex_double *b, + lapack_int *ldb, + lapack_complex_double *t, + lapack_int *ldt, + lapack_complex_double *work, + lapack_int *info); +void LAPACK_stpqrt2(lapack_int *m, + lapack_int *n, + float *a, + lapack_int *lda, + float *b, + lapack_int *ldb, + float *t, + lapack_int *ldt, + lapack_int *info); +void LAPACK_dtpqrt2(lapack_int *m, + lapack_int *n, + double *a, + lapack_int *lda, + double *b, + lapack_int *ldb, + double *t, + lapack_int *ldt, + lapack_int *info); +void LAPACK_ctpqrt2(lapack_int *m, + lapack_int *n, + lapack_complex_float *a, + lapack_int *lda, + lapack_complex_float *b, + lapack_int *ldb, + lapack_complex_float *t, + lapack_int *ldt, + lapack_int *info); +void LAPACK_ztpqrt2(lapack_int *m, + lapack_int *n, + lapack_complex_double *a, + lapack_int *lda, + lapack_complex_double *b, + lapack_int *ldb, + lapack_complex_double *t, + lapack_int *ldt, + lapack_int *info); +void LAPACK_stprfb(char *side, + char *trans, + char *direct, + char *storev, + lapack_int *m, + lapack_int *n, + lapack_int *k, + lapack_int *l, + const float *v, + lapack_int *ldv, + const float *t, + lapack_int *ldt, + float *a, + lapack_int *lda, + float *b, + lapack_int *ldb, + const float *mywork, + lapack_int *myldwork); +void LAPACK_dtprfb(char *side, + char *trans, + char *direct, + char *storev, + lapack_int *m, + lapack_int *n, + lapack_int *k, + lapack_int *l, + const double *v, + lapack_int *ldv, + const double *t, + lapack_int *ldt, + double *a, + lapack_int *lda, + double *b, + lapack_int *ldb, + const double *mywork, + lapack_int *myldwork); +void LAPACK_ctprfb(char *side, + char *trans, + char *direct, + char *storev, + lapack_int *m, + lapack_int *n, + lapack_int *k, + lapack_int *l, + const lapack_complex_float *v, + lapack_int *ldv, + const lapack_complex_float *t, + lapack_int *ldt, + lapack_complex_float *a, + lapack_int *lda, + lapack_complex_float *b, + lapack_int *ldb, + const float *mywork, + lapack_int *myldwork); +void LAPACK_ztprfb(char *side, + char *trans, + char *direct, + char *storev, + lapack_int *m, + lapack_int *n, + lapack_int *k, + lapack_int *l, + const lapack_complex_double *v, + lapack_int *ldv, + const lapack_complex_double *t, + lapack_int *ldt, + lapack_complex_double *a, + lapack_int *lda, + lapack_complex_double *b, + lapack_int *ldb, + const double *mywork, + lapack_int *myldwork); // LAPACK 3.X.X -void LAPACK_csyr( char* uplo, lapack_int* n, lapack_complex_float* alpha, - const lapack_complex_float* x, lapack_int* incx, - lapack_complex_float* a, lapack_int* lda ); -void LAPACK_zsyr( char* uplo, lapack_int* n, lapack_complex_double* alpha, - const lapack_complex_double* x, lapack_int* incx, - lapack_complex_double* a, lapack_int* lda ); +void LAPACK_csyr(char *uplo, + lapack_int *n, + lapack_complex_float *alpha, + const lapack_complex_float *x, + lapack_int *incx, + lapack_complex_float *a, + lapack_int *lda); +void LAPACK_zsyr(char *uplo, + lapack_int *n, + lapack_complex_double *alpha, + const lapack_complex_double *x, + lapack_int *incx, + lapack_complex_double *a, + lapack_int *lda); #ifdef __cplusplus } diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/misc/lapacke_mangling.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/misc/lapacke_mangling.h index 6211fd14..d852de7a 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/misc/lapacke_mangling.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/misc/lapacke_mangling.h @@ -3,15 +3,14 @@ #ifndef LAPACK_GLOBAL #if defined(LAPACK_GLOBAL_PATTERN_LC) || defined(ADD_) -#define LAPACK_GLOBAL(lcname,UCNAME) lcname##_ +#define LAPACK_GLOBAL(lcname, UCNAME) lcname##_ #elif defined(LAPACK_GLOBAL_PATTERN_UC) || defined(UPPER) -#define LAPACK_GLOBAL(lcname,UCNAME) UCNAME +#define LAPACK_GLOBAL(lcname, UCNAME) UCNAME #elif defined(LAPACK_GLOBAL_PATTERN_MC) || defined(NOCHANGE) -#define LAPACK_GLOBAL(lcname,UCNAME) lcname +#define LAPACK_GLOBAL(lcname, UCNAME) lcname #else -#define LAPACK_GLOBAL(lcname,UCNAME) lcname##_ +#define LAPACK_GLOBAL(lcname, UCNAME) lcname##_ #endif #endif #endif - diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/plugins/ArrayCwiseBinaryOps.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/plugins/ArrayCwiseBinaryOps.h index 1f8a531a..9d0c2913 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/plugins/ArrayCwiseBinaryOps.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/plugins/ArrayCwiseBinaryOps.h @@ -1,209 +1,228 @@ /** \returns an expression of the coefficient wise product of \c *this and \a other - * - * \sa MatrixBase::cwiseProduct - */ + * + * \sa MatrixBase::cwiseProduct + */ template -EIGEN_DEVICE_FUNC -EIGEN_STRONG_INLINE const EIGEN_CWISE_BINARY_RETURN_TYPE(Derived,OtherDerived,product) -operator*(const EIGEN_CURRENT_STORAGE_BASE_CLASS &other) const +EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const EIGEN_CWISE_BINARY_RETURN_TYPE(Derived, OtherDerived, product) operator*( + const EIGEN_CURRENT_STORAGE_BASE_CLASS &other) const { - return EIGEN_CWISE_BINARY_RETURN_TYPE(Derived,OtherDerived,product)(derived(), other.derived()); + return EIGEN_CWISE_BINARY_RETURN_TYPE(Derived, OtherDerived, product)(derived(), other.derived()); } /** \returns an expression of the coefficient wise quotient of \c *this and \a other - * - * \sa MatrixBase::cwiseQuotient - */ + * + * \sa MatrixBase::cwiseQuotient + */ template -EIGEN_DEVICE_FUNC -EIGEN_STRONG_INLINE const CwiseBinaryOp, const Derived, const OtherDerived> -operator/(const EIGEN_CURRENT_STORAGE_BASE_CLASS &other) const +EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const + CwiseBinaryOp, const Derived, const OtherDerived> + operator/(const EIGEN_CURRENT_STORAGE_BASE_CLASS &other) const { - return CwiseBinaryOp, const Derived, const OtherDerived>(derived(), other.derived()); + return CwiseBinaryOp, + const Derived, + const OtherDerived>(derived(), other.derived()); } /** \returns an expression of the coefficient-wise min of \c *this and \a other - * - * Example: \include Cwise_min.cpp - * Output: \verbinclude Cwise_min.out - * - * \sa max() - */ -EIGEN_MAKE_CWISE_BINARY_OP(min,min) + * + * Example: \include Cwise_min.cpp + * Output: \verbinclude Cwise_min.out + * + * \sa max() + */ +EIGEN_MAKE_CWISE_BINARY_OP(min, min) /** \returns an expression of the coefficient-wise min of \c *this and scalar \a other - * - * \sa max() - */ + * + * \sa max() + */ EIGEN_DEVICE_FUNC -EIGEN_STRONG_INLINE const CwiseBinaryOp, const Derived, - const CwiseNullaryOp, PlainObject> > +EIGEN_STRONG_INLINE const CwiseBinaryOp, + const Derived, + const CwiseNullaryOp, PlainObject>> #ifdef EIGEN_PARSED_BY_DOXYGEN -min + min #else -(min) + (min) #endif -(const Scalar &other) const + (const Scalar &other) const { return (min)(Derived::PlainObject::Constant(rows(), cols(), other)); } /** \returns an expression of the coefficient-wise max of \c *this and \a other - * - * Example: \include Cwise_max.cpp - * Output: \verbinclude Cwise_max.out - * - * \sa min() - */ -EIGEN_MAKE_CWISE_BINARY_OP(max,max) + * + * Example: \include Cwise_max.cpp + * Output: \verbinclude Cwise_max.out + * + * \sa min() + */ +EIGEN_MAKE_CWISE_BINARY_OP(max, max) /** \returns an expression of the coefficient-wise max of \c *this and scalar \a other - * - * \sa min() - */ + * + * \sa min() + */ EIGEN_DEVICE_FUNC -EIGEN_STRONG_INLINE const CwiseBinaryOp, const Derived, - const CwiseNullaryOp, PlainObject> > +EIGEN_STRONG_INLINE const CwiseBinaryOp, + const Derived, + const CwiseNullaryOp, PlainObject>> #ifdef EIGEN_PARSED_BY_DOXYGEN -max + max #else -(max) + (max) #endif -(const Scalar &other) const + (const Scalar &other) const { return (max)(Derived::PlainObject::Constant(rows(), cols(), other)); } /** \returns an expression of the coefficient-wise power of \c *this to the given array of \a exponents. - * - * This function computes the coefficient-wise power. - * - * Example: \include Cwise_array_power_array.cpp - * Output: \verbinclude Cwise_array_power_array.out - */ -EIGEN_MAKE_CWISE_BINARY_OP(pow,pow) + * + * This function computes the coefficient-wise power. + * + * Example: \include Cwise_array_power_array.cpp + * Output: \verbinclude Cwise_array_power_array.out + */ +EIGEN_MAKE_CWISE_BINARY_OP(pow, pow) #ifndef EIGEN_PARSED_BY_DOXYGEN -EIGEN_MAKE_SCALAR_BINARY_OP_ONTHERIGHT(pow,pow) +EIGEN_MAKE_SCALAR_BINARY_OP_ONTHERIGHT(pow, pow) #else /** \returns an expression of the coefficients of \c *this rasied to the constant power \a exponent - * - * \tparam T is the scalar type of \a exponent. It must be compatible with the scalar type of the given expression. - * - * This function computes the coefficient-wise power. The function MatrixBase::pow() in the - * unsupported module MatrixFunctions computes the matrix power. - * - * Example: \include Cwise_pow.cpp - * Output: \verbinclude Cwise_pow.out - * - * \sa ArrayBase::pow(ArrayBase), square(), cube(), exp(), log() - */ + * + * \tparam T is the scalar type of \a exponent. It must be compatible with the scalar type of the given expression. + * + * This function computes the coefficient-wise power. The function MatrixBase::pow() in the + * unsupported module MatrixFunctions computes the matrix power. + * + * Example: \include Cwise_pow.cpp + * Output: \verbinclude Cwise_pow.out + * + * \sa ArrayBase::pow(ArrayBase), square(), cube(), exp(), log() + */ template -const CwiseBinaryOp,Derived,Constant > pow(const T& exponent) const; +const CwiseBinaryOp, Derived, Constant> pow(const T &exponent) const; #endif // TODO code generating macros could be moved to Macros.h and could include generation of documentation -#define EIGEN_MAKE_CWISE_COMP_OP(OP, COMPARATOR) \ -template \ -EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const CwiseBinaryOp, const Derived, const OtherDerived> \ -OP(const EIGEN_CURRENT_STORAGE_BASE_CLASS &other) const \ -{ \ - return CwiseBinaryOp, const Derived, const OtherDerived>(derived(), other.derived()); \ -}\ -typedef CwiseBinaryOp, const Derived, const CwiseNullaryOp, PlainObject> > Cmp ## COMPARATOR ## ReturnType; \ -typedef CwiseBinaryOp, const CwiseNullaryOp, PlainObject>, const Derived > RCmp ## COMPARATOR ## ReturnType; \ -EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const Cmp ## COMPARATOR ## ReturnType \ -OP(const Scalar& s) const { \ - return this->OP(Derived::PlainObject::Constant(rows(), cols(), s)); \ -} \ -EIGEN_DEVICE_FUNC friend EIGEN_STRONG_INLINE const RCmp ## COMPARATOR ## ReturnType \ -OP(const Scalar& s, const Derived& d) { \ - return Derived::PlainObject::Constant(d.rows(), d.cols(), s).OP(d); \ -} - -#define EIGEN_MAKE_CWISE_COMP_R_OP(OP, R_OP, RCOMPARATOR) \ -template \ -EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const CwiseBinaryOp, const OtherDerived, const Derived> \ -OP(const EIGEN_CURRENT_STORAGE_BASE_CLASS &other) const \ -{ \ - return CwiseBinaryOp, const OtherDerived, const Derived>(other.derived(), derived()); \ -} \ -EIGEN_DEVICE_FUNC \ -inline const RCmp ## RCOMPARATOR ## ReturnType \ -OP(const Scalar& s) const { \ - return Derived::PlainObject::Constant(rows(), cols(), s).R_OP(*this); \ -} \ -friend inline const Cmp ## RCOMPARATOR ## ReturnType \ -OP(const Scalar& s, const Derived& d) { \ - return d.R_OP(Derived::PlainObject::Constant(d.rows(), d.cols(), s)); \ -} +#define EIGEN_MAKE_CWISE_COMP_OP(OP, COMPARATOR) \ + template \ + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const \ + CwiseBinaryOp, \ + const Derived, \ + const OtherDerived> \ + OP(const EIGEN_CURRENT_STORAGE_BASE_CLASS &other) const \ + { \ + return CwiseBinaryOp, \ + const Derived, \ + const OtherDerived>(derived(), other.derived()); \ + } \ + typedef CwiseBinaryOp, \ + const Derived, \ + const CwiseNullaryOp, PlainObject>> \ + Cmp##COMPARATOR##ReturnType; \ + typedef CwiseBinaryOp, \ + const CwiseNullaryOp, PlainObject>, \ + const Derived> \ + RCmp##COMPARATOR##ReturnType; \ + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const Cmp##COMPARATOR##ReturnType OP(const Scalar &s) const \ + { \ + return this->OP(Derived::PlainObject::Constant(rows(), cols(), s)); \ + } \ + EIGEN_DEVICE_FUNC friend EIGEN_STRONG_INLINE const RCmp##COMPARATOR##ReturnType OP( \ + const Scalar &s, const Derived &d) \ + { \ + return Derived::PlainObject::Constant(d.rows(), d.cols(), s).OP(d); \ + } +#define EIGEN_MAKE_CWISE_COMP_R_OP(OP, R_OP, RCOMPARATOR) \ + template \ + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const \ + CwiseBinaryOp, \ + const OtherDerived, \ + const Derived> \ + OP(const EIGEN_CURRENT_STORAGE_BASE_CLASS &other) const \ + { \ + return CwiseBinaryOp, \ + const OtherDerived, \ + const Derived>(other.derived(), derived()); \ + } \ + EIGEN_DEVICE_FUNC \ + inline const RCmp##RCOMPARATOR##ReturnType OP(const Scalar &s) const \ + { \ + return Derived::PlainObject::Constant(rows(), cols(), s).R_OP(*this); \ + } \ + friend inline const Cmp##RCOMPARATOR##ReturnType OP(const Scalar &s, const Derived &d) \ + { \ + return d.R_OP(Derived::PlainObject::Constant(d.rows(), d.cols(), s)); \ + } /** \returns an expression of the coefficient-wise \< operator of *this and \a other - * - * Example: \include Cwise_less.cpp - * Output: \verbinclude Cwise_less.out - * - * \sa all(), any(), operator>(), operator<=() - */ + * + * Example: \include Cwise_less.cpp + * Output: \verbinclude Cwise_less.out + * + * \sa all(), any(), operator>(), operator<=() + */ EIGEN_MAKE_CWISE_COMP_OP(operator<, LT) /** \returns an expression of the coefficient-wise \<= operator of *this and \a other - * - * Example: \include Cwise_less_equal.cpp - * Output: \verbinclude Cwise_less_equal.out - * - * \sa all(), any(), operator>=(), operator<() - */ + * + * Example: \include Cwise_less_equal.cpp + * Output: \verbinclude Cwise_less_equal.out + * + * \sa all(), any(), operator>=(), operator<() + */ EIGEN_MAKE_CWISE_COMP_OP(operator<=, LE) /** \returns an expression of the coefficient-wise \> operator of *this and \a other - * - * Example: \include Cwise_greater.cpp - * Output: \verbinclude Cwise_greater.out - * - * \sa all(), any(), operator>=(), operator<() - */ + * + * Example: \include Cwise_greater.cpp + * Output: \verbinclude Cwise_greater.out + * + * \sa all(), any(), operator>=(), operator<() + */ EIGEN_MAKE_CWISE_COMP_R_OP(operator>, operator<, LT) /** \returns an expression of the coefficient-wise \>= operator of *this and \a other - * - * Example: \include Cwise_greater_equal.cpp - * Output: \verbinclude Cwise_greater_equal.out - * - * \sa all(), any(), operator>(), operator<=() - */ + * + * Example: \include Cwise_greater_equal.cpp + * Output: \verbinclude Cwise_greater_equal.out + * + * \sa all(), any(), operator>(), operator<=() + */ EIGEN_MAKE_CWISE_COMP_R_OP(operator>=, operator<=, LE) /** \returns an expression of the coefficient-wise == operator of *this and \a other - * - * \warning this performs an exact comparison, which is generally a bad idea with floating-point types. - * In order to check for equality between two vectors or matrices with floating-point coefficients, it is - * generally a far better idea to use a fuzzy comparison as provided by isApprox() and - * isMuchSmallerThan(). - * - * Example: \include Cwise_equal_equal.cpp - * Output: \verbinclude Cwise_equal_equal.out - * - * \sa all(), any(), isApprox(), isMuchSmallerThan() - */ + * + * \warning this performs an exact comparison, which is generally a bad idea with floating-point types. + * In order to check for equality between two vectors or matrices with floating-point coefficients, it is + * generally a far better idea to use a fuzzy comparison as provided by isApprox() and + * isMuchSmallerThan(). + * + * Example: \include Cwise_equal_equal.cpp + * Output: \verbinclude Cwise_equal_equal.out + * + * \sa all(), any(), isApprox(), isMuchSmallerThan() + */ EIGEN_MAKE_CWISE_COMP_OP(operator==, EQ) /** \returns an expression of the coefficient-wise != operator of *this and \a other - * - * \warning this performs an exact comparison, which is generally a bad idea with floating-point types. - * In order to check for equality between two vectors or matrices with floating-point coefficients, it is - * generally a far better idea to use a fuzzy comparison as provided by isApprox() and - * isMuchSmallerThan(). - * - * Example: \include Cwise_not_equal.cpp - * Output: \verbinclude Cwise_not_equal.out - * - * \sa all(), any(), isApprox(), isMuchSmallerThan() - */ + * + * \warning this performs an exact comparison, which is generally a bad idea with floating-point types. + * In order to check for equality between two vectors or matrices with floating-point coefficients, it is + * generally a far better idea to use a fuzzy comparison as provided by isApprox() and + * isMuchSmallerThan(). + * + * Example: \include Cwise_not_equal.cpp + * Output: \verbinclude Cwise_not_equal.out + * + * \sa all(), any(), isApprox(), isMuchSmallerThan() + */ EIGEN_MAKE_CWISE_COMP_OP(operator!=, NEQ) @@ -212,61 +231,64 @@ EIGEN_MAKE_CWISE_COMP_OP(operator!=, NEQ) // scalar addition #ifndef EIGEN_PARSED_BY_DOXYGEN -EIGEN_MAKE_SCALAR_BINARY_OP(operator+,sum) +EIGEN_MAKE_SCALAR_BINARY_OP(operator+, sum) #else /** \returns an expression of \c *this with each coeff incremented by the constant \a scalar - * - * \tparam T is the scalar type of \a scalar. It must be compatible with the scalar type of the given expression. - * - * Example: \include Cwise_plus.cpp - * Output: \verbinclude Cwise_plus.out - * - * \sa operator+=(), operator-() - */ + * + * \tparam T is the scalar type of \a scalar. It must be compatible with the scalar type of the given expression. + * + * Example: \include Cwise_plus.cpp + * Output: \verbinclude Cwise_plus.out + * + * \sa operator+=(), operator-() + */ template -const CwiseBinaryOp,Derived,Constant > operator+(const T& scalar) const; +const CwiseBinaryOp, Derived, Constant> operator+(const T &scalar) const; /** \returns an expression of \a expr with each coeff incremented by the constant \a scalar - * - * \tparam T is the scalar type of \a scalar. It must be compatible with the scalar type of the given expression. - */ -template friend -const CwiseBinaryOp,Constant,Derived> operator+(const T& scalar, const StorageBaseType& expr); + * + * \tparam T is the scalar type of \a scalar. It must be compatible with the scalar type of the given expression. + */ +template +friend const CwiseBinaryOp, Constant, Derived> operator+(const T &scalar, + const StorageBaseType &expr); #endif #ifndef EIGEN_PARSED_BY_DOXYGEN -EIGEN_MAKE_SCALAR_BINARY_OP(operator-,difference) +EIGEN_MAKE_SCALAR_BINARY_OP(operator-, difference) #else /** \returns an expression of \c *this with each coeff decremented by the constant \a scalar - * - * \tparam T is the scalar type of \a scalar. It must be compatible with the scalar type of the given expression. - * - * Example: \include Cwise_minus.cpp - * Output: \verbinclude Cwise_minus.out - * - * \sa operator+=(), operator-() - */ + * + * \tparam T is the scalar type of \a scalar. It must be compatible with the scalar type of the given expression. + * + * Example: \include Cwise_minus.cpp + * Output: \verbinclude Cwise_minus.out + * + * \sa operator+=(), operator-() + */ template -const CwiseBinaryOp,Derived,Constant > operator-(const T& scalar) const; +const CwiseBinaryOp, Derived, Constant> operator-(const T &scalar) const; /** \returns an expression of the constant matrix of value \a scalar decremented by the coefficients of \a expr - * - * \tparam T is the scalar type of \a scalar. It must be compatible with the scalar type of the given expression. - */ -template friend -const CwiseBinaryOp,Constant,Derived> operator-(const T& scalar, const StorageBaseType& expr); + * + * \tparam T is the scalar type of \a scalar. It must be compatible with the scalar type of the given expression. + */ +template +friend const CwiseBinaryOp, Constant, Derived> operator-(const T &scalar, + const StorageBaseType &expr); #endif #ifndef EIGEN_PARSED_BY_DOXYGEN - EIGEN_MAKE_SCALAR_BINARY_OP_ONTHELEFT(operator/,quotient) +EIGEN_MAKE_SCALAR_BINARY_OP_ONTHELEFT(operator/, quotient) #else - /** - * \brief Component-wise division of the scalar \a s by array elements of \a a. - * - * \tparam Scalar is the scalar type of \a x. It must be compatible with the scalar type of the given array expression (\c Derived::Scalar). - */ - template friend - inline const CwiseBinaryOp,Constant,Derived> - operator/(const T& s,const StorageBaseType& a); +/** + * \brief Component-wise division of the scalar \a s by array elements of \a a. + * + * \tparam Scalar is the scalar type of \a x. It must be compatible with the scalar type of the given array expression + * (\c Derived::Scalar). + */ +template +friend inline const CwiseBinaryOp, Constant, Derived> operator/(const T &s, + const StorageBaseType &a); #endif /** \returns an expression of the coefficient-wise ^ operator of *this and \a other @@ -279,13 +301,13 @@ const CwiseBinaryOp,Constant,Derived * \sa operator&&(), select() */ template -EIGEN_DEVICE_FUNC -inline const CwiseBinaryOp -operator^(const EIGEN_CURRENT_STORAGE_BASE_CLASS &other) const +EIGEN_DEVICE_FUNC inline const CwiseBinaryOp + operator^(const EIGEN_CURRENT_STORAGE_BASE_CLASS &other) const { - EIGEN_STATIC_ASSERT((internal::is_same::value && internal::is_same::value), - THIS_METHOD_IS_ONLY_FOR_EXPRESSIONS_OF_BOOL); - return CwiseBinaryOp(derived(),other.derived()); + EIGEN_STATIC_ASSERT( + (internal::is_same::value && internal::is_same::value), + THIS_METHOD_IS_ONLY_FOR_EXPRESSIONS_OF_BOOL); + return CwiseBinaryOp(derived(), other.derived()); } // NOTE disabled until we agree on argument order @@ -309,24 +331,24 @@ polygamma(const EIGEN_CURRENT_STORAGE_BASE_CLASS &n) const #endif /** \returns an expression of the coefficient-wise zeta function. - * - * \specialfunctions_module - * - * It returns the Riemann zeta function of two arguments \c *this and \a q: - * - * \param *this is the exposent, it must be > 1 - * \param q is the shift, it must be > 0 - * - * \note This function supports only float and double scalar types. To support other scalar types, the user has - * to provide implementations of zeta(T,T) for any scalar type T to be supported. - * - * This method is an alias for zeta(*this,q); - * - * \sa Eigen::zeta() - */ + * + * \specialfunctions_module + * + * It returns the Riemann zeta function of two arguments \c *this and \a q: + * + * \param *this is the exposent, it must be > 1 + * \param q is the shift, it must be > 0 + * + * \note This function supports only float and double scalar types. To support other scalar types, the user has + * to provide implementations of zeta(T,T) for any scalar type T to be supported. + * + * This method is an alias for zeta(*this,q); + * + * \sa Eigen::zeta() + */ template -inline const CwiseBinaryOp, const Derived, const DerivedQ> -zeta(const EIGEN_CURRENT_STORAGE_BASE_CLASS &q) const +inline const CwiseBinaryOp, const Derived, const DerivedQ> zeta( + const EIGEN_CURRENT_STORAGE_BASE_CLASS &q) const { return CwiseBinaryOp, const Derived, const DerivedQ>(this->derived(), q.derived()); } diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/plugins/ArrayCwiseUnaryOps.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/plugins/ArrayCwiseUnaryOps.h index ebaa3f19..38d51f44 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/plugins/ArrayCwiseUnaryOps.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/plugins/ArrayCwiseUnaryOps.h @@ -32,435 +32,321 @@ typedef CwiseUnaryOp, const Derived> IsInfRetu typedef CwiseUnaryOp, const Derived> IsFiniteReturnType; /** \returns an expression of the coefficient-wise absolute value of \c *this - * - * Example: \include Cwise_abs.cpp - * Output: \verbinclude Cwise_abs.out - * - * \sa Math functions, abs2() - */ -EIGEN_DEVICE_FUNC -EIGEN_STRONG_INLINE const AbsReturnType -abs() const -{ - return AbsReturnType(derived()); -} + * + * Example: \include Cwise_abs.cpp + * Output: \verbinclude Cwise_abs.out + * + * \sa Math functions, abs2() + */ +EIGEN_DEVICE_FUNC +EIGEN_STRONG_INLINE const AbsReturnType abs() const { return AbsReturnType(derived()); } /** \returns an expression of the coefficient-wise phase angle of \c *this - * - * Example: \include Cwise_arg.cpp - * Output: \verbinclude Cwise_arg.out - * - * \sa abs() - */ -EIGEN_DEVICE_FUNC -EIGEN_STRONG_INLINE const ArgReturnType -arg() const -{ - return ArgReturnType(derived()); -} + * + * Example: \include Cwise_arg.cpp + * Output: \verbinclude Cwise_arg.out + * + * \sa abs() + */ +EIGEN_DEVICE_FUNC +EIGEN_STRONG_INLINE const ArgReturnType arg() const { return ArgReturnType(derived()); } /** \returns an expression of the coefficient-wise squared absolute value of \c *this - * - * Example: \include Cwise_abs2.cpp - * Output: \verbinclude Cwise_abs2.out - * - * \sa Math functions, abs(), square() - */ -EIGEN_DEVICE_FUNC -EIGEN_STRONG_INLINE const Abs2ReturnType -abs2() const -{ - return Abs2ReturnType(derived()); -} + * + * Example: \include Cwise_abs2.cpp + * Output: \verbinclude Cwise_abs2.out + * + * \sa Math functions, abs(), square() + */ +EIGEN_DEVICE_FUNC +EIGEN_STRONG_INLINE const Abs2ReturnType abs2() const { return Abs2ReturnType(derived()); } /** \returns an expression of the coefficient-wise exponential of *this. - * - * This function computes the coefficient-wise exponential. The function MatrixBase::exp() in the - * unsupported module MatrixFunctions computes the matrix exponential. - * - * Example: \include Cwise_exp.cpp - * Output: \verbinclude Cwise_exp.out - * - * \sa Math functions, pow(), log(), sin(), cos() - */ -EIGEN_DEVICE_FUNC -inline const ExpReturnType -exp() const -{ - return ExpReturnType(derived()); -} + * + * This function computes the coefficient-wise exponential. The function MatrixBase::exp() in the + * unsupported module MatrixFunctions computes the matrix exponential. + * + * Example: \include Cwise_exp.cpp + * Output: \verbinclude Cwise_exp.out + * + * \sa Math functions, pow(), log(), sin(), cos() + */ +EIGEN_DEVICE_FUNC +inline const ExpReturnType exp() const { return ExpReturnType(derived()); } /** \returns an expression of the coefficient-wise logarithm of *this. - * - * This function computes the coefficient-wise logarithm. The function MatrixBase::log() in the - * unsupported module MatrixFunctions computes the matrix logarithm. - * - * Example: \include Cwise_log.cpp - * Output: \verbinclude Cwise_log.out - * - * \sa Math functions, exp() - */ -EIGEN_DEVICE_FUNC -inline const LogReturnType -log() const -{ - return LogReturnType(derived()); -} + * + * This function computes the coefficient-wise logarithm. The function MatrixBase::log() in the + * unsupported module MatrixFunctions computes the matrix logarithm. + * + * Example: \include Cwise_log.cpp + * Output: \verbinclude Cwise_log.out + * + * \sa Math functions, exp() + */ +EIGEN_DEVICE_FUNC +inline const LogReturnType log() const { return LogReturnType(derived()); } /** \returns an expression of the coefficient-wise logarithm of 1 plus \c *this. - * - * In exact arithmetic, \c x.log() is equivalent to \c (x+1).log(), - * however, with finite precision, this function is much more accurate when \c x is close to zero. - * - * \sa Math functions, log() - */ -EIGEN_DEVICE_FUNC -inline const Log1pReturnType -log1p() const -{ - return Log1pReturnType(derived()); -} + * + * In exact arithmetic, \c x.log() is equivalent to \c (x+1).log(), + * however, with finite precision, this function is much more accurate when \c x is close to zero. + * + * \sa Math functions, log() + */ +EIGEN_DEVICE_FUNC +inline const Log1pReturnType log1p() const { return Log1pReturnType(derived()); } /** \returns an expression of the coefficient-wise base-10 logarithm of *this. - * - * This function computes the coefficient-wise base-10 logarithm. - * - * Example: \include Cwise_log10.cpp - * Output: \verbinclude Cwise_log10.out - * - * \sa Math functions, log() - */ -EIGEN_DEVICE_FUNC -inline const Log10ReturnType -log10() const -{ - return Log10ReturnType(derived()); -} + * + * This function computes the coefficient-wise base-10 logarithm. + * + * Example: \include Cwise_log10.cpp + * Output: \verbinclude Cwise_log10.out + * + * \sa Math functions, log() + */ +EIGEN_DEVICE_FUNC +inline const Log10ReturnType log10() const { return Log10ReturnType(derived()); } /** \returns an expression of the coefficient-wise square root of *this. - * - * This function computes the coefficient-wise square root. The function MatrixBase::sqrt() in the - * unsupported module MatrixFunctions computes the matrix square root. - * - * Example: \include Cwise_sqrt.cpp - * Output: \verbinclude Cwise_sqrt.out - * - * \sa Math functions, pow(), square() - */ -EIGEN_DEVICE_FUNC -inline const SqrtReturnType -sqrt() const -{ - return SqrtReturnType(derived()); -} + * + * This function computes the coefficient-wise square root. The function MatrixBase::sqrt() in the + * unsupported module MatrixFunctions computes the matrix square root. + * + * Example: \include Cwise_sqrt.cpp + * Output: \verbinclude Cwise_sqrt.out + * + * \sa Math functions, pow(), square() + */ +EIGEN_DEVICE_FUNC +inline const SqrtReturnType sqrt() const { return SqrtReturnType(derived()); } /** \returns an expression of the coefficient-wise inverse square root of *this. - * - * This function computes the coefficient-wise inverse square root. - * - * Example: \include Cwise_sqrt.cpp - * Output: \verbinclude Cwise_sqrt.out - * - * \sa pow(), square() - */ -EIGEN_DEVICE_FUNC -inline const RsqrtReturnType -rsqrt() const -{ - return RsqrtReturnType(derived()); -} + * + * This function computes the coefficient-wise inverse square root. + * + * Example: \include Cwise_sqrt.cpp + * Output: \verbinclude Cwise_sqrt.out + * + * \sa pow(), square() + */ +EIGEN_DEVICE_FUNC +inline const RsqrtReturnType rsqrt() const { return RsqrtReturnType(derived()); } /** \returns an expression of the coefficient-wise signum of *this. - * - * This function computes the coefficient-wise signum. - * - * Example: \include Cwise_sign.cpp - * Output: \verbinclude Cwise_sign.out - * - * \sa pow(), square() - */ -EIGEN_DEVICE_FUNC -inline const SignReturnType -sign() const -{ - return SignReturnType(derived()); -} + * + * This function computes the coefficient-wise signum. + * + * Example: \include Cwise_sign.cpp + * Output: \verbinclude Cwise_sign.out + * + * \sa pow(), square() + */ +EIGEN_DEVICE_FUNC +inline const SignReturnType sign() const { return SignReturnType(derived()); } /** \returns an expression of the coefficient-wise cosine of *this. - * - * This function computes the coefficient-wise cosine. The function MatrixBase::cos() in the - * unsupported module MatrixFunctions computes the matrix cosine. - * - * Example: \include Cwise_cos.cpp - * Output: \verbinclude Cwise_cos.out - * - * \sa Math functions, sin(), acos() - */ -EIGEN_DEVICE_FUNC -inline const CosReturnType -cos() const -{ - return CosReturnType(derived()); -} + * + * This function computes the coefficient-wise cosine. The function MatrixBase::cos() in the + * unsupported module MatrixFunctions computes the matrix cosine. + * + * Example: \include Cwise_cos.cpp + * Output: \verbinclude Cwise_cos.out + * + * \sa Math functions, sin(), acos() + */ +EIGEN_DEVICE_FUNC +inline const CosReturnType cos() const { return CosReturnType(derived()); } /** \returns an expression of the coefficient-wise sine of *this. - * - * This function computes the coefficient-wise sine. The function MatrixBase::sin() in the - * unsupported module MatrixFunctions computes the matrix sine. - * - * Example: \include Cwise_sin.cpp - * Output: \verbinclude Cwise_sin.out - * - * \sa Math functions, cos(), asin() - */ -EIGEN_DEVICE_FUNC -inline const SinReturnType -sin() const -{ - return SinReturnType(derived()); -} + * + * This function computes the coefficient-wise sine. The function MatrixBase::sin() in the + * unsupported module MatrixFunctions computes the matrix sine. + * + * Example: \include Cwise_sin.cpp + * Output: \verbinclude Cwise_sin.out + * + * \sa Math functions, cos(), asin() + */ +EIGEN_DEVICE_FUNC +inline const SinReturnType sin() const { return SinReturnType(derived()); } /** \returns an expression of the coefficient-wise tan of *this. - * - * Example: \include Cwise_tan.cpp - * Output: \verbinclude Cwise_tan.out - * - * \sa Math functions, cos(), sin() - */ -EIGEN_DEVICE_FUNC -inline const TanReturnType -tan() const -{ - return TanReturnType(derived()); -} + * + * Example: \include Cwise_tan.cpp + * Output: \verbinclude Cwise_tan.out + * + * \sa Math functions, cos(), sin() + */ +EIGEN_DEVICE_FUNC +inline const TanReturnType tan() const { return TanReturnType(derived()); } /** \returns an expression of the coefficient-wise arc tan of *this. - * - * Example: \include Cwise_atan.cpp - * Output: \verbinclude Cwise_atan.out - * - * \sa Math functions, tan(), asin(), acos() - */ -EIGEN_DEVICE_FUNC -inline const AtanReturnType -atan() const -{ - return AtanReturnType(derived()); -} + * + * Example: \include Cwise_atan.cpp + * Output: \verbinclude Cwise_atan.out + * + * \sa Math functions, tan(), asin(), acos() + */ +EIGEN_DEVICE_FUNC +inline const AtanReturnType atan() const { return AtanReturnType(derived()); } /** \returns an expression of the coefficient-wise arc cosine of *this. - * - * Example: \include Cwise_acos.cpp - * Output: \verbinclude Cwise_acos.out - * - * \sa Math functions, cos(), asin() - */ -EIGEN_DEVICE_FUNC -inline const AcosReturnType -acos() const -{ - return AcosReturnType(derived()); -} + * + * Example: \include Cwise_acos.cpp + * Output: \verbinclude Cwise_acos.out + * + * \sa Math functions, cos(), asin() + */ +EIGEN_DEVICE_FUNC +inline const AcosReturnType acos() const { return AcosReturnType(derived()); } /** \returns an expression of the coefficient-wise arc sine of *this. - * - * Example: \include Cwise_asin.cpp - * Output: \verbinclude Cwise_asin.out - * - * \sa Math functions, sin(), acos() - */ -EIGEN_DEVICE_FUNC -inline const AsinReturnType -asin() const -{ - return AsinReturnType(derived()); -} + * + * Example: \include Cwise_asin.cpp + * Output: \verbinclude Cwise_asin.out + * + * \sa Math functions, sin(), acos() + */ +EIGEN_DEVICE_FUNC +inline const AsinReturnType asin() const { return AsinReturnType(derived()); } /** \returns an expression of the coefficient-wise hyperbolic tan of *this. - * - * Example: \include Cwise_tanh.cpp - * Output: \verbinclude Cwise_tanh.out - * - * \sa Math functions, tan(), sinh(), cosh() - */ -EIGEN_DEVICE_FUNC -inline const TanhReturnType -tanh() const -{ - return TanhReturnType(derived()); -} + * + * Example: \include Cwise_tanh.cpp + * Output: \verbinclude Cwise_tanh.out + * + * \sa Math functions, tan(), sinh(), cosh() + */ +EIGEN_DEVICE_FUNC +inline const TanhReturnType tanh() const { return TanhReturnType(derived()); } /** \returns an expression of the coefficient-wise hyperbolic sin of *this. - * - * Example: \include Cwise_sinh.cpp - * Output: \verbinclude Cwise_sinh.out - * - * \sa Math functions, sin(), tanh(), cosh() - */ -EIGEN_DEVICE_FUNC -inline const SinhReturnType -sinh() const -{ - return SinhReturnType(derived()); -} + * + * Example: \include Cwise_sinh.cpp + * Output: \verbinclude Cwise_sinh.out + * + * \sa Math functions, sin(), tanh(), cosh() + */ +EIGEN_DEVICE_FUNC +inline const SinhReturnType sinh() const { return SinhReturnType(derived()); } /** \returns an expression of the coefficient-wise hyperbolic cos of *this. - * - * Example: \include Cwise_cosh.cpp - * Output: \verbinclude Cwise_cosh.out - * - * \sa Math functions, tan(), sinh(), cosh() - */ -EIGEN_DEVICE_FUNC -inline const CoshReturnType -cosh() const -{ - return CoshReturnType(derived()); -} + * + * Example: \include Cwise_cosh.cpp + * Output: \verbinclude Cwise_cosh.out + * + * \sa Math functions, tan(), sinh(), cosh() + */ +EIGEN_DEVICE_FUNC +inline const CoshReturnType cosh() const { return CoshReturnType(derived()); } /** \returns an expression of the coefficient-wise inverse of *this. - * - * Example: \include Cwise_inverse.cpp - * Output: \verbinclude Cwise_inverse.out - * - * \sa operator/(), operator*() - */ -EIGEN_DEVICE_FUNC -inline const InverseReturnType -inverse() const -{ - return InverseReturnType(derived()); -} + * + * Example: \include Cwise_inverse.cpp + * Output: \verbinclude Cwise_inverse.out + * + * \sa operator/(), operator*() + */ +EIGEN_DEVICE_FUNC +inline const InverseReturnType inverse() const { return InverseReturnType(derived()); } /** \returns an expression of the coefficient-wise square of *this. - * - * Example: \include Cwise_square.cpp - * Output: \verbinclude Cwise_square.out - * - * \sa Math functions, abs2(), cube(), pow() - */ -EIGEN_DEVICE_FUNC -inline const SquareReturnType -square() const -{ - return SquareReturnType(derived()); -} + * + * Example: \include Cwise_square.cpp + * Output: \verbinclude Cwise_square.out + * + * \sa Math functions, abs2(), cube(), pow() + */ +EIGEN_DEVICE_FUNC +inline const SquareReturnType square() const { return SquareReturnType(derived()); } /** \returns an expression of the coefficient-wise cube of *this. - * - * Example: \include Cwise_cube.cpp - * Output: \verbinclude Cwise_cube.out - * - * \sa Math functions, square(), pow() - */ -EIGEN_DEVICE_FUNC -inline const CubeReturnType -cube() const -{ - return CubeReturnType(derived()); -} + * + * Example: \include Cwise_cube.cpp + * Output: \verbinclude Cwise_cube.out + * + * \sa Math functions, square(), pow() + */ +EIGEN_DEVICE_FUNC +inline const CubeReturnType cube() const { return CubeReturnType(derived()); } /** \returns an expression of the coefficient-wise round of *this. - * - * Example: \include Cwise_round.cpp - * Output: \verbinclude Cwise_round.out - * - * \sa Math functions, ceil(), floor() - */ -EIGEN_DEVICE_FUNC -inline const RoundReturnType -round() const -{ - return RoundReturnType(derived()); -} + * + * Example: \include Cwise_round.cpp + * Output: \verbinclude Cwise_round.out + * + * \sa Math functions, ceil(), floor() + */ +EIGEN_DEVICE_FUNC +inline const RoundReturnType round() const { return RoundReturnType(derived()); } /** \returns an expression of the coefficient-wise floor of *this. - * - * Example: \include Cwise_floor.cpp - * Output: \verbinclude Cwise_floor.out - * - * \sa Math functions, ceil(), round() - */ -EIGEN_DEVICE_FUNC -inline const FloorReturnType -floor() const -{ - return FloorReturnType(derived()); -} + * + * Example: \include Cwise_floor.cpp + * Output: \verbinclude Cwise_floor.out + * + * \sa Math functions, ceil(), round() + */ +EIGEN_DEVICE_FUNC +inline const FloorReturnType floor() const { return FloorReturnType(derived()); } /** \returns an expression of the coefficient-wise ceil of *this. - * - * Example: \include Cwise_ceil.cpp - * Output: \verbinclude Cwise_ceil.out - * - * \sa Math functions, floor(), round() - */ -EIGEN_DEVICE_FUNC -inline const CeilReturnType -ceil() const -{ - return CeilReturnType(derived()); -} + * + * Example: \include Cwise_ceil.cpp + * Output: \verbinclude Cwise_ceil.out + * + * \sa Math functions, floor(), round() + */ +EIGEN_DEVICE_FUNC +inline const CeilReturnType ceil() const { return CeilReturnType(derived()); } /** \returns an expression of the coefficient-wise isnan of *this. - * - * Example: \include Cwise_isNaN.cpp - * Output: \verbinclude Cwise_isNaN.out - * - * \sa isfinite(), isinf() - */ -EIGEN_DEVICE_FUNC -inline const IsNaNReturnType -isNaN() const -{ - return IsNaNReturnType(derived()); -} + * + * Example: \include Cwise_isNaN.cpp + * Output: \verbinclude Cwise_isNaN.out + * + * \sa isfinite(), isinf() + */ +EIGEN_DEVICE_FUNC +inline const IsNaNReturnType isNaN() const { return IsNaNReturnType(derived()); } /** \returns an expression of the coefficient-wise isinf of *this. - * - * Example: \include Cwise_isInf.cpp - * Output: \verbinclude Cwise_isInf.out - * - * \sa isnan(), isfinite() - */ -EIGEN_DEVICE_FUNC -inline const IsInfReturnType -isInf() const -{ - return IsInfReturnType(derived()); -} + * + * Example: \include Cwise_isInf.cpp + * Output: \verbinclude Cwise_isInf.out + * + * \sa isnan(), isfinite() + */ +EIGEN_DEVICE_FUNC +inline const IsInfReturnType isInf() const { return IsInfReturnType(derived()); } /** \returns an expression of the coefficient-wise isfinite of *this. - * - * Example: \include Cwise_isFinite.cpp - * Output: \verbinclude Cwise_isFinite.out - * - * \sa isnan(), isinf() - */ -EIGEN_DEVICE_FUNC -inline const IsFiniteReturnType -isFinite() const -{ - return IsFiniteReturnType(derived()); -} + * + * Example: \include Cwise_isFinite.cpp + * Output: \verbinclude Cwise_isFinite.out + * + * \sa isnan(), isinf() + */ +EIGEN_DEVICE_FUNC +inline const IsFiniteReturnType isFinite() const { return IsFiniteReturnType(derived()); } /** \returns an expression of the coefficient-wise ! operator of *this - * - * \warning this operator is for expression of bool only. - * - * Example: \include Cwise_boolean_not.cpp - * Output: \verbinclude Cwise_boolean_not.out - * - * \sa operator!=() - */ -EIGEN_DEVICE_FUNC -inline const BooleanNotReturnType -operator!() const -{ - EIGEN_STATIC_ASSERT((internal::is_same::value), - THIS_METHOD_IS_ONLY_FOR_EXPRESSIONS_OF_BOOL); + * + * \warning this operator is for expression of bool only. + * + * Example: \include Cwise_boolean_not.cpp + * Output: \verbinclude Cwise_boolean_not.out + * + * \sa operator!=() + */ +EIGEN_DEVICE_FUNC +inline const BooleanNotReturnType operator!() const +{ + EIGEN_STATIC_ASSERT((internal::is_same::value), THIS_METHOD_IS_ONLY_FOR_EXPRESSIONS_OF_BOOL); return BooleanNotReturnType(derived()); } @@ -473,80 +359,65 @@ typedef CwiseUnaryOp, const Derived> ErfReturnTy typedef CwiseUnaryOp, const Derived> ErfcReturnType; /** \cpp11 \returns an expression of the coefficient-wise ln(|gamma(*this)|). - * - * \specialfunctions_module - * - * Example: \include Cwise_lgamma.cpp - * Output: \verbinclude Cwise_lgamma.out - * - * \note This function supports only float and double scalar types in c++11 mode. To support other scalar types, - * or float/double in non c++11 mode, the user has to provide implementations of lgamma(T) for any scalar - * type T to be supported. - * - * \sa Math functions, digamma() - */ -EIGEN_DEVICE_FUNC -inline const LgammaReturnType -lgamma() const -{ - return LgammaReturnType(derived()); -} + * + * \specialfunctions_module + * + * Example: \include Cwise_lgamma.cpp + * Output: \verbinclude Cwise_lgamma.out + * + * \note This function supports only float and double scalar types in c++11 mode. To support other scalar types, + * or float/double in non c++11 mode, the user has to provide implementations of lgamma(T) for any scalar + * type T to be supported. + * + * \sa Math functions, digamma() + */ +EIGEN_DEVICE_FUNC +inline const LgammaReturnType lgamma() const { return LgammaReturnType(derived()); } /** \returns an expression of the coefficient-wise digamma (psi, derivative of lgamma). - * - * \specialfunctions_module - * - * \note This function supports only float and double scalar types. To support other scalar types, - * the user has to provide implementations of digamma(T) for any scalar - * type T to be supported. - * - * \sa Math functions, Eigen::digamma(), Eigen::polygamma(), lgamma() - */ -EIGEN_DEVICE_FUNC -inline const DigammaReturnType -digamma() const -{ - return DigammaReturnType(derived()); -} + * + * \specialfunctions_module + * + * \note This function supports only float and double scalar types. To support other scalar types, + * the user has to provide implementations of digamma(T) for any scalar + * type T to be supported. + * + * \sa Math functions, Eigen::digamma(), + * Eigen::polygamma(), lgamma() + */ +EIGEN_DEVICE_FUNC +inline const DigammaReturnType digamma() const { return DigammaReturnType(derived()); } /** \cpp11 \returns an expression of the coefficient-wise Gauss error - * function of *this. - * - * \specialfunctions_module - * - * Example: \include Cwise_erf.cpp - * Output: \verbinclude Cwise_erf.out - * - * \note This function supports only float and double scalar types in c++11 mode. To support other scalar types, - * or float/double in non c++11 mode, the user has to provide implementations of erf(T) for any scalar - * type T to be supported. - * - * \sa Math functions, erfc() - */ -EIGEN_DEVICE_FUNC -inline const ErfReturnType -erf() const -{ - return ErfReturnType(derived()); -} + * function of *this. + * + * \specialfunctions_module + * + * Example: \include Cwise_erf.cpp + * Output: \verbinclude Cwise_erf.out + * + * \note This function supports only float and double scalar types in c++11 mode. To support other scalar types, + * or float/double in non c++11 mode, the user has to provide implementations of erf(T) for any scalar + * type T to be supported. + * + * \sa Math functions, erfc() + */ +EIGEN_DEVICE_FUNC +inline const ErfReturnType erf() const { return ErfReturnType(derived()); } /** \cpp11 \returns an expression of the coefficient-wise Complementary error - * function of *this. - * - * \specialfunctions_module - * - * Example: \include Cwise_erfc.cpp - * Output: \verbinclude Cwise_erfc.out - * - * \note This function supports only float and double scalar types in c++11 mode. To support other scalar types, - * or float/double in non c++11 mode, the user has to provide implementations of erfc(T) for any scalar - * type T to be supported. - * - * \sa Math functions, erf() - */ -EIGEN_DEVICE_FUNC -inline const ErfcReturnType -erfc() const -{ - return ErfcReturnType(derived()); -} + * function of *this. + * + * \specialfunctions_module + * + * Example: \include Cwise_erfc.cpp + * Output: \verbinclude Cwise_erfc.out + * + * \note This function supports only float and double scalar types in c++11 mode. To support other scalar types, + * or float/double in non c++11 mode, the user has to provide implementations of erfc(T) for any scalar + * type T to be supported. + * + * \sa Math functions, erf() + */ +EIGEN_DEVICE_FUNC +inline const ErfcReturnType erfc() const { return ErfcReturnType(derived()); } diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/plugins/BlockMethods.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/plugins/BlockMethods.h index ac35a008..9bfcae9d 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/plugins/BlockMethods.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/plugins/BlockMethods.h @@ -18,29 +18,54 @@ typedef Block::ColsAtCompileTime, IsRowMaj typedef const Block::ColsAtCompileTime, IsRowMajor> ConstRowXpr; /// \internal expression type of a block of whole columns */ typedef Block::RowsAtCompileTime, Dynamic, !IsRowMajor> ColsBlockXpr; -typedef const Block::RowsAtCompileTime, Dynamic, !IsRowMajor> ConstColsBlockXpr; +typedef const Block::RowsAtCompileTime, Dynamic, !IsRowMajor> + ConstColsBlockXpr; /// \internal expression type of a block of whole rows */ typedef Block::ColsAtCompileTime, IsRowMajor> RowsBlockXpr; typedef const Block::ColsAtCompileTime, IsRowMajor> ConstRowsBlockXpr; /// \internal expression type of a block of whole columns */ -template struct NColsBlockXpr { typedef Block::RowsAtCompileTime, N, !IsRowMajor> Type; }; -template struct ConstNColsBlockXpr { typedef const Block::RowsAtCompileTime, N, !IsRowMajor> Type; }; +template struct NColsBlockXpr +{ + typedef Block::RowsAtCompileTime, N, !IsRowMajor> Type; +}; +template struct ConstNColsBlockXpr +{ + typedef const Block::RowsAtCompileTime, N, !IsRowMajor> Type; +}; /// \internal expression type of a block of whole rows */ -template struct NRowsBlockXpr { typedef Block::ColsAtCompileTime, IsRowMajor> Type; }; -template struct ConstNRowsBlockXpr { typedef const Block::ColsAtCompileTime, IsRowMajor> Type; }; +template struct NRowsBlockXpr +{ + typedef Block::ColsAtCompileTime, IsRowMajor> Type; +}; +template struct ConstNRowsBlockXpr +{ + typedef const Block::ColsAtCompileTime, IsRowMajor> Type; +}; /// \internal expression of a block */ typedef Block BlockXpr; typedef const Block ConstBlockXpr; /// \internal expression of a block of fixed sizes */ -template struct FixedBlockXpr { typedef Block Type; }; -template struct ConstFixedBlockXpr { typedef Block Type; }; +template struct FixedBlockXpr +{ + typedef Block Type; +}; +template struct ConstFixedBlockXpr +{ + typedef Block Type; +}; typedef VectorBlock SegmentReturnType; typedef const VectorBlock ConstSegmentReturnType; -template struct FixedSegmentReturnType { typedef VectorBlock Type; }; -template struct ConstFixedSegmentReturnType { typedef const VectorBlock Type; }; +template struct FixedSegmentReturnType +{ + typedef VectorBlock Type; +}; +template struct ConstFixedSegmentReturnType +{ + typedef const VectorBlock Type; +}; -#endif // not EIGEN_PARSED_BY_DOXYGEN +#endif// not EIGEN_PARSED_BY_DOXYGEN /// \returns a dynamic-size expression of a block in *this. /// @@ -74,8 +99,6 @@ inline const ConstBlockXpr block(Index startRow, Index startCol, Index blockRows } - - /// \returns a dynamic-size expression of a top-right corner of *this. /// /// \param cRows the number of rows in the corner @@ -113,19 +136,16 @@ EIGEN_DOC_BLOCK_ADDONS_NOT_INNER_PANEL /// /// \sa class Block, block(Index,Index) /// -template -EIGEN_DEVICE_FUNC -inline typename FixedBlockXpr::Type topRightCorner() +template EIGEN_DEVICE_FUNC inline typename FixedBlockXpr::Type topRightCorner() { - return typename FixedBlockXpr::Type(derived(), 0, cols() - CCols); + return typename FixedBlockXpr::Type(derived(), 0, cols() - CCols); } /// This is the const version of topRightCorner(). template -EIGEN_DEVICE_FUNC -inline const typename ConstFixedBlockXpr::Type topRightCorner() const +EIGEN_DEVICE_FUNC inline const typename ConstFixedBlockXpr::Type topRightCorner() const { - return typename ConstFixedBlockXpr::Type(derived(), 0, cols() - CCols); + return typename ConstFixedBlockXpr::Type(derived(), 0, cols() - CCols); } /// \returns an expression of a top-right corner of *this. @@ -148,20 +168,19 @@ EIGEN_DOC_BLOCK_ADDONS_NOT_INNER_PANEL /// \sa class Block /// template -inline typename FixedBlockXpr::Type topRightCorner(Index cRows, Index cCols) +inline typename FixedBlockXpr::Type topRightCorner(Index cRows, Index cCols) { - return typename FixedBlockXpr::Type(derived(), 0, cols() - cCols, cRows, cCols); + return typename FixedBlockXpr::Type(derived(), 0, cols() - cCols, cRows, cCols); } /// This is the const version of topRightCorner(Index, Index). template -inline const typename ConstFixedBlockXpr::Type topRightCorner(Index cRows, Index cCols) const +inline const typename ConstFixedBlockXpr::Type topRightCorner(Index cRows, Index cCols) const { - return typename ConstFixedBlockXpr::Type(derived(), 0, cols() - cCols, cRows, cCols); + return typename ConstFixedBlockXpr::Type(derived(), 0, cols() - cCols, cRows, cCols); } - /// \returns a dynamic-size expression of a top-left corner of *this. /// /// \param cRows the number of rows in the corner @@ -175,10 +194,7 @@ EIGEN_DOC_BLOCK_ADDONS_NOT_INNER_PANEL /// \sa class Block, block(Index,Index,Index,Index) /// EIGEN_DEVICE_FUNC -inline BlockXpr topLeftCorner(Index cRows, Index cCols) -{ - return BlockXpr(derived(), 0, 0, cRows, cCols); -} +inline BlockXpr topLeftCorner(Index cRows, Index cCols) { return BlockXpr(derived(), 0, 0, cRows, cCols); } /// This is the const version of topLeftCorner(Index, Index). EIGEN_DEVICE_FUNC @@ -198,19 +214,16 @@ EIGEN_DOC_BLOCK_ADDONS_NOT_INNER_PANEL /// /// \sa class Block, block(Index,Index,Index,Index) /// -template -EIGEN_DEVICE_FUNC -inline typename FixedBlockXpr::Type topLeftCorner() +template EIGEN_DEVICE_FUNC inline typename FixedBlockXpr::Type topLeftCorner() { - return typename FixedBlockXpr::Type(derived(), 0, 0); + return typename FixedBlockXpr::Type(derived(), 0, 0); } /// This is the const version of topLeftCorner(). template -EIGEN_DEVICE_FUNC -inline const typename ConstFixedBlockXpr::Type topLeftCorner() const +EIGEN_DEVICE_FUNC inline const typename ConstFixedBlockXpr::Type topLeftCorner() const { - return typename ConstFixedBlockXpr::Type(derived(), 0, 0); + return typename ConstFixedBlockXpr::Type(derived(), 0, 0); } /// \returns an expression of a top-left corner of *this. @@ -232,21 +245,19 @@ EIGEN_DOC_BLOCK_ADDONS_NOT_INNER_PANEL /// /// \sa class Block /// -template -inline typename FixedBlockXpr::Type topLeftCorner(Index cRows, Index cCols) +template inline typename FixedBlockXpr::Type topLeftCorner(Index cRows, Index cCols) { - return typename FixedBlockXpr::Type(derived(), 0, 0, cRows, cCols); + return typename FixedBlockXpr::Type(derived(), 0, 0, cRows, cCols); } /// This is the const version of topLeftCorner(Index, Index). template -inline const typename ConstFixedBlockXpr::Type topLeftCorner(Index cRows, Index cCols) const +inline const typename ConstFixedBlockXpr::Type topLeftCorner(Index cRows, Index cCols) const { - return typename ConstFixedBlockXpr::Type(derived(), 0, 0, cRows, cCols); + return typename ConstFixedBlockXpr::Type(derived(), 0, 0, cRows, cCols); } - /// \returns a dynamic-size expression of a bottom-right corner of *this. /// /// \param cRows the number of rows in the corner @@ -283,19 +294,16 @@ EIGEN_DOC_BLOCK_ADDONS_NOT_INNER_PANEL /// /// \sa class Block, block(Index,Index,Index,Index) /// -template -EIGEN_DEVICE_FUNC -inline typename FixedBlockXpr::Type bottomRightCorner() +template EIGEN_DEVICE_FUNC inline typename FixedBlockXpr::Type bottomRightCorner() { - return typename FixedBlockXpr::Type(derived(), rows() - CRows, cols() - CCols); + return typename FixedBlockXpr::Type(derived(), rows() - CRows, cols() - CCols); } /// This is the const version of bottomRightCorner(). template -EIGEN_DEVICE_FUNC -inline const typename ConstFixedBlockXpr::Type bottomRightCorner() const +EIGEN_DEVICE_FUNC inline const typename ConstFixedBlockXpr::Type bottomRightCorner() const { - return typename ConstFixedBlockXpr::Type(derived(), rows() - CRows, cols() - CCols); + return typename ConstFixedBlockXpr::Type(derived(), rows() - CRows, cols() - CCols); } /// \returns an expression of a bottom-right corner of *this. @@ -318,20 +326,19 @@ EIGEN_DOC_BLOCK_ADDONS_NOT_INNER_PANEL /// \sa class Block /// template -inline typename FixedBlockXpr::Type bottomRightCorner(Index cRows, Index cCols) +inline typename FixedBlockXpr::Type bottomRightCorner(Index cRows, Index cCols) { - return typename FixedBlockXpr::Type(derived(), rows() - cRows, cols() - cCols, cRows, cCols); + return typename FixedBlockXpr::Type(derived(), rows() - cRows, cols() - cCols, cRows, cCols); } /// This is the const version of bottomRightCorner(Index, Index). template -inline const typename ConstFixedBlockXpr::Type bottomRightCorner(Index cRows, Index cCols) const +inline const typename ConstFixedBlockXpr::Type bottomRightCorner(Index cRows, Index cCols) const { - return typename ConstFixedBlockXpr::Type(derived(), rows() - cRows, cols() - cCols, cRows, cCols); + return typename ConstFixedBlockXpr::Type(derived(), rows() - cRows, cols() - cCols, cRows, cCols); } - /// \returns a dynamic-size expression of a bottom-left corner of *this. /// /// \param cRows the number of rows in the corner @@ -368,19 +375,16 @@ EIGEN_DOC_BLOCK_ADDONS_NOT_INNER_PANEL /// /// \sa class Block, block(Index,Index,Index,Index) /// -template -EIGEN_DEVICE_FUNC -inline typename FixedBlockXpr::Type bottomLeftCorner() +template EIGEN_DEVICE_FUNC inline typename FixedBlockXpr::Type bottomLeftCorner() { - return typename FixedBlockXpr::Type(derived(), rows() - CRows, 0); + return typename FixedBlockXpr::Type(derived(), rows() - CRows, 0); } /// This is the const version of bottomLeftCorner(). template -EIGEN_DEVICE_FUNC -inline const typename ConstFixedBlockXpr::Type bottomLeftCorner() const +EIGEN_DEVICE_FUNC inline const typename ConstFixedBlockXpr::Type bottomLeftCorner() const { - return typename ConstFixedBlockXpr::Type(derived(), rows() - CRows, 0); + return typename ConstFixedBlockXpr::Type(derived(), rows() - CRows, 0); } /// \returns an expression of a bottom-left corner of *this. @@ -403,20 +407,19 @@ EIGEN_DOC_BLOCK_ADDONS_NOT_INNER_PANEL /// \sa class Block /// template -inline typename FixedBlockXpr::Type bottomLeftCorner(Index cRows, Index cCols) +inline typename FixedBlockXpr::Type bottomLeftCorner(Index cRows, Index cCols) { - return typename FixedBlockXpr::Type(derived(), rows() - cRows, 0, cRows, cCols); + return typename FixedBlockXpr::Type(derived(), rows() - cRows, 0, cRows, cCols); } /// This is the const version of bottomLeftCorner(Index, Index). template -inline const typename ConstFixedBlockXpr::Type bottomLeftCorner(Index cRows, Index cCols) const +inline const typename ConstFixedBlockXpr::Type bottomLeftCorner(Index cRows, Index cCols) const { - return typename ConstFixedBlockXpr::Type(derived(), rows() - cRows, 0, cRows, cCols); + return typename ConstFixedBlockXpr::Type(derived(), rows() - cRows, 0, cRows, cCols); } - /// \returns a block consisting of the top rows of *this. /// /// \param n the number of rows in the block @@ -424,22 +427,16 @@ inline const typename ConstFixedBlockXpr::Type bottomLeftCorner(Ind /// Example: \include MatrixBase_topRows_int.cpp /// Output: \verbinclude MatrixBase_topRows_int.out /// -EIGEN_DOC_BLOCK_ADDONS_INNER_PANEL_IF(row-major) +EIGEN_DOC_BLOCK_ADDONS_INNER_PANEL_IF(row - major) /// /// \sa class Block, block(Index,Index,Index,Index) /// EIGEN_DEVICE_FUNC -inline RowsBlockXpr topRows(Index n) -{ - return RowsBlockXpr(derived(), 0, 0, n, cols()); -} +inline RowsBlockXpr topRows(Index n) { return RowsBlockXpr(derived(), 0, 0, n, cols()); } /// This is the const version of topRows(Index). EIGEN_DEVICE_FUNC -inline ConstRowsBlockXpr topRows(Index n) const -{ - return ConstRowsBlockXpr(derived(), 0, 0, n, cols()); -} +inline ConstRowsBlockXpr topRows(Index n) const { return ConstRowsBlockXpr(derived(), 0, 0, n, cols()); } /// \returns a block consisting of the top rows of *this. /// @@ -452,27 +449,22 @@ inline ConstRowsBlockXpr topRows(Index n) const /// Example: \include MatrixBase_template_int_topRows.cpp /// Output: \verbinclude MatrixBase_template_int_topRows.out /// -EIGEN_DOC_BLOCK_ADDONS_INNER_PANEL_IF(row-major) +EIGEN_DOC_BLOCK_ADDONS_INNER_PANEL_IF(row - major) /// /// \sa class Block, block(Index,Index,Index,Index) /// -template -EIGEN_DEVICE_FUNC -inline typename NRowsBlockXpr::Type topRows(Index n = N) +template EIGEN_DEVICE_FUNC inline typename NRowsBlockXpr::Type topRows(Index n = N) { return typename NRowsBlockXpr::Type(derived(), 0, 0, n, cols()); } /// This is the const version of topRows(). -template -EIGEN_DEVICE_FUNC -inline typename ConstNRowsBlockXpr::Type topRows(Index n = N) const +template EIGEN_DEVICE_FUNC inline typename ConstNRowsBlockXpr::Type topRows(Index n = N) const { return typename ConstNRowsBlockXpr::Type(derived(), 0, 0, n, cols()); } - /// \returns a block consisting of the bottom rows of *this. /// /// \param n the number of rows in the block @@ -480,22 +472,16 @@ inline typename ConstNRowsBlockXpr::Type topRows(Index n = N) const /// Example: \include MatrixBase_bottomRows_int.cpp /// Output: \verbinclude MatrixBase_bottomRows_int.out /// -EIGEN_DOC_BLOCK_ADDONS_INNER_PANEL_IF(row-major) +EIGEN_DOC_BLOCK_ADDONS_INNER_PANEL_IF(row - major) /// /// \sa class Block, block(Index,Index,Index,Index) /// EIGEN_DEVICE_FUNC -inline RowsBlockXpr bottomRows(Index n) -{ - return RowsBlockXpr(derived(), rows() - n, 0, n, cols()); -} +inline RowsBlockXpr bottomRows(Index n) { return RowsBlockXpr(derived(), rows() - n, 0, n, cols()); } /// This is the const version of bottomRows(Index). EIGEN_DEVICE_FUNC -inline ConstRowsBlockXpr bottomRows(Index n) const -{ - return ConstRowsBlockXpr(derived(), rows() - n, 0, n, cols()); -} +inline ConstRowsBlockXpr bottomRows(Index n) const { return ConstRowsBlockXpr(derived(), rows() - n, 0, n, cols()); } /// \returns a block consisting of the bottom rows of *this. /// @@ -508,27 +494,22 @@ inline ConstRowsBlockXpr bottomRows(Index n) const /// Example: \include MatrixBase_template_int_bottomRows.cpp /// Output: \verbinclude MatrixBase_template_int_bottomRows.out /// -EIGEN_DOC_BLOCK_ADDONS_INNER_PANEL_IF(row-major) +EIGEN_DOC_BLOCK_ADDONS_INNER_PANEL_IF(row - major) /// /// \sa class Block, block(Index,Index,Index,Index) /// -template -EIGEN_DEVICE_FUNC -inline typename NRowsBlockXpr::Type bottomRows(Index n = N) +template EIGEN_DEVICE_FUNC inline typename NRowsBlockXpr::Type bottomRows(Index n = N) { return typename NRowsBlockXpr::Type(derived(), rows() - n, 0, n, cols()); } /// This is the const version of bottomRows(). -template -EIGEN_DEVICE_FUNC -inline typename ConstNRowsBlockXpr::Type bottomRows(Index n = N) const +template EIGEN_DEVICE_FUNC inline typename ConstNRowsBlockXpr::Type bottomRows(Index n = N) const { return typename ConstNRowsBlockXpr::Type(derived(), rows() - n, 0, n, cols()); } - /// \returns a block consisting of a range of rows of *this. /// /// \param startRow the index of the first row in the block @@ -537,15 +518,12 @@ inline typename ConstNRowsBlockXpr::Type bottomRows(Index n = N) const /// Example: \include DenseBase_middleRows_int.cpp /// Output: \verbinclude DenseBase_middleRows_int.out /// -EIGEN_DOC_BLOCK_ADDONS_INNER_PANEL_IF(row-major) +EIGEN_DOC_BLOCK_ADDONS_INNER_PANEL_IF(row - major) /// /// \sa class Block, block(Index,Index,Index,Index) /// EIGEN_DEVICE_FUNC -inline RowsBlockXpr middleRows(Index startRow, Index n) -{ - return RowsBlockXpr(derived(), startRow, 0, n, cols()); -} +inline RowsBlockXpr middleRows(Index startRow, Index n) { return RowsBlockXpr(derived(), startRow, 0, n, cols()); } /// This is the const version of middleRows(Index,Index). EIGEN_DEVICE_FUNC @@ -566,27 +544,23 @@ inline ConstRowsBlockXpr middleRows(Index startRow, Index n) const /// Example: \include DenseBase_template_int_middleRows.cpp /// Output: \verbinclude DenseBase_template_int_middleRows.out /// -EIGEN_DOC_BLOCK_ADDONS_INNER_PANEL_IF(row-major) +EIGEN_DOC_BLOCK_ADDONS_INNER_PANEL_IF(row - major) /// /// \sa class Block, block(Index,Index,Index,Index) /// -template -EIGEN_DEVICE_FUNC -inline typename NRowsBlockXpr::Type middleRows(Index startRow, Index n = N) +template EIGEN_DEVICE_FUNC inline typename NRowsBlockXpr::Type middleRows(Index startRow, Index n = N) { return typename NRowsBlockXpr::Type(derived(), startRow, 0, n, cols()); } /// This is the const version of middleRows(). template -EIGEN_DEVICE_FUNC -inline typename ConstNRowsBlockXpr::Type middleRows(Index startRow, Index n = N) const +EIGEN_DEVICE_FUNC inline typename ConstNRowsBlockXpr::Type middleRows(Index startRow, Index n = N) const { return typename ConstNRowsBlockXpr::Type(derived(), startRow, 0, n, cols()); } - /// \returns a block consisting of the left columns of *this. /// /// \param n the number of columns in the block @@ -594,22 +568,16 @@ inline typename ConstNRowsBlockXpr::Type middleRows(Index startRow, Index n = /// Example: \include MatrixBase_leftCols_int.cpp /// Output: \verbinclude MatrixBase_leftCols_int.out /// -EIGEN_DOC_BLOCK_ADDONS_INNER_PANEL_IF(column-major) +EIGEN_DOC_BLOCK_ADDONS_INNER_PANEL_IF(column - major) /// /// \sa class Block, block(Index,Index,Index,Index) /// EIGEN_DEVICE_FUNC -inline ColsBlockXpr leftCols(Index n) -{ - return ColsBlockXpr(derived(), 0, 0, rows(), n); -} +inline ColsBlockXpr leftCols(Index n) { return ColsBlockXpr(derived(), 0, 0, rows(), n); } /// This is the const version of leftCols(Index). EIGEN_DEVICE_FUNC -inline ConstColsBlockXpr leftCols(Index n) const -{ - return ConstColsBlockXpr(derived(), 0, 0, rows(), n); -} +inline ConstColsBlockXpr leftCols(Index n) const { return ConstColsBlockXpr(derived(), 0, 0, rows(), n); } /// \returns a block consisting of the left columns of *this. /// @@ -622,27 +590,22 @@ inline ConstColsBlockXpr leftCols(Index n) const /// Example: \include MatrixBase_template_int_leftCols.cpp /// Output: \verbinclude MatrixBase_template_int_leftCols.out /// -EIGEN_DOC_BLOCK_ADDONS_INNER_PANEL_IF(column-major) +EIGEN_DOC_BLOCK_ADDONS_INNER_PANEL_IF(column - major) /// /// \sa class Block, block(Index,Index,Index,Index) /// -template -EIGEN_DEVICE_FUNC -inline typename NColsBlockXpr::Type leftCols(Index n = N) +template EIGEN_DEVICE_FUNC inline typename NColsBlockXpr::Type leftCols(Index n = N) { return typename NColsBlockXpr::Type(derived(), 0, 0, rows(), n); } /// This is the const version of leftCols(). -template -EIGEN_DEVICE_FUNC -inline typename ConstNColsBlockXpr::Type leftCols(Index n = N) const +template EIGEN_DEVICE_FUNC inline typename ConstNColsBlockXpr::Type leftCols(Index n = N) const { return typename ConstNColsBlockXpr::Type(derived(), 0, 0, rows(), n); } - /// \returns a block consisting of the right columns of *this. /// /// \param n the number of columns in the block @@ -650,22 +613,16 @@ inline typename ConstNColsBlockXpr::Type leftCols(Index n = N) const /// Example: \include MatrixBase_rightCols_int.cpp /// Output: \verbinclude MatrixBase_rightCols_int.out /// -EIGEN_DOC_BLOCK_ADDONS_INNER_PANEL_IF(column-major) +EIGEN_DOC_BLOCK_ADDONS_INNER_PANEL_IF(column - major) /// /// \sa class Block, block(Index,Index,Index,Index) /// EIGEN_DEVICE_FUNC -inline ColsBlockXpr rightCols(Index n) -{ - return ColsBlockXpr(derived(), 0, cols() - n, rows(), n); -} +inline ColsBlockXpr rightCols(Index n) { return ColsBlockXpr(derived(), 0, cols() - n, rows(), n); } /// This is the const version of rightCols(Index). EIGEN_DEVICE_FUNC -inline ConstColsBlockXpr rightCols(Index n) const -{ - return ConstColsBlockXpr(derived(), 0, cols() - n, rows(), n); -} +inline ConstColsBlockXpr rightCols(Index n) const { return ConstColsBlockXpr(derived(), 0, cols() - n, rows(), n); } /// \returns a block consisting of the right columns of *this. /// @@ -678,27 +635,22 @@ inline ConstColsBlockXpr rightCols(Index n) const /// Example: \include MatrixBase_template_int_rightCols.cpp /// Output: \verbinclude MatrixBase_template_int_rightCols.out /// -EIGEN_DOC_BLOCK_ADDONS_INNER_PANEL_IF(column-major) +EIGEN_DOC_BLOCK_ADDONS_INNER_PANEL_IF(column - major) /// /// \sa class Block, block(Index,Index,Index,Index) /// -template -EIGEN_DEVICE_FUNC -inline typename NColsBlockXpr::Type rightCols(Index n = N) +template EIGEN_DEVICE_FUNC inline typename NColsBlockXpr::Type rightCols(Index n = N) { return typename NColsBlockXpr::Type(derived(), 0, cols() - n, rows(), n); } /// This is the const version of rightCols(). -template -EIGEN_DEVICE_FUNC -inline typename ConstNColsBlockXpr::Type rightCols(Index n = N) const +template EIGEN_DEVICE_FUNC inline typename ConstNColsBlockXpr::Type rightCols(Index n = N) const { return typename ConstNColsBlockXpr::Type(derived(), 0, cols() - n, rows(), n); } - /// \returns a block consisting of a range of columns of *this. /// /// \param startCol the index of the first column in the block @@ -707,7 +659,7 @@ inline typename ConstNColsBlockXpr::Type rightCols(Index n = N) const /// Example: \include DenseBase_middleCols_int.cpp /// Output: \verbinclude DenseBase_middleCols_int.out /// -EIGEN_DOC_BLOCK_ADDONS_INNER_PANEL_IF(column-major) +EIGEN_DOC_BLOCK_ADDONS_INNER_PANEL_IF(column - major) /// /// \sa class Block, block(Index,Index,Index,Index) /// @@ -736,27 +688,23 @@ inline ConstColsBlockXpr middleCols(Index startCol, Index numCols) const /// Example: \include DenseBase_template_int_middleCols.cpp /// Output: \verbinclude DenseBase_template_int_middleCols.out /// -EIGEN_DOC_BLOCK_ADDONS_INNER_PANEL_IF(column-major) +EIGEN_DOC_BLOCK_ADDONS_INNER_PANEL_IF(column - major) /// /// \sa class Block, block(Index,Index,Index,Index) /// -template -EIGEN_DEVICE_FUNC -inline typename NColsBlockXpr::Type middleCols(Index startCol, Index n = N) +template EIGEN_DEVICE_FUNC inline typename NColsBlockXpr::Type middleCols(Index startCol, Index n = N) { return typename NColsBlockXpr::Type(derived(), 0, startCol, rows(), n); } /// This is the const version of middleCols(). template -EIGEN_DEVICE_FUNC -inline typename ConstNColsBlockXpr::Type middleCols(Index startCol, Index n = N) const +EIGEN_DEVICE_FUNC inline typename ConstNColsBlockXpr::Type middleCols(Index startCol, Index n = N) const { return typename ConstNColsBlockXpr::Type(derived(), 0, startCol, rows(), n); } - /// \returns a fixed-size expression of a block in *this. /// /// The template parameters \a NRows and \a NCols are the number of @@ -776,18 +724,17 @@ EIGEN_DOC_BLOCK_ADDONS_NOT_INNER_PANEL /// \sa class Block, block(Index,Index,Index,Index) /// template -EIGEN_DEVICE_FUNC -inline typename FixedBlockXpr::Type block(Index startRow, Index startCol) +EIGEN_DEVICE_FUNC inline typename FixedBlockXpr::Type block(Index startRow, Index startCol) { - return typename FixedBlockXpr::Type(derived(), startRow, startCol); + return typename FixedBlockXpr::Type(derived(), startRow, startCol); } /// This is the const version of block<>(Index, Index). */ template -EIGEN_DEVICE_FUNC -inline const typename ConstFixedBlockXpr::Type block(Index startRow, Index startCol) const +EIGEN_DEVICE_FUNC inline const typename ConstFixedBlockXpr::Type block(Index startRow, + Index startCol) const { - return typename ConstFixedBlockXpr::Type(derived(), startRow, startCol); + return typename ConstFixedBlockXpr::Type(derived(), startRow, startCol); } /// \returns an expression of a block in *this. @@ -812,18 +759,18 @@ EIGEN_DOC_BLOCK_ADDONS_NOT_INNER_PANEL /// \sa class Block, block(Index,Index,Index,Index) /// template -inline typename FixedBlockXpr::Type block(Index startRow, Index startCol, - Index blockRows, Index blockCols) +inline typename FixedBlockXpr::Type + block(Index startRow, Index startCol, Index blockRows, Index blockCols) { - return typename FixedBlockXpr::Type(derived(), startRow, startCol, blockRows, blockCols); + return typename FixedBlockXpr::Type(derived(), startRow, startCol, blockRows, blockCols); } /// This is the const version of block<>(Index, Index, Index, Index). template -inline const typename ConstFixedBlockXpr::Type block(Index startRow, Index startCol, - Index blockRows, Index blockCols) const +inline const typename ConstFixedBlockXpr::Type + block(Index startRow, Index startCol, Index blockRows, Index blockCols) const { - return typename ConstFixedBlockXpr::Type(derived(), startRow, startCol, blockRows, blockCols); + return typename ConstFixedBlockXpr::Type(derived(), startRow, startCol, blockRows, blockCols); } /// \returns an expression of the \a i-th column of *this. Note that the numbering starts at 0. @@ -831,42 +778,30 @@ inline const typename ConstFixedBlockXpr::Type block(Index startRow /// Example: \include MatrixBase_col.cpp /// Output: \verbinclude MatrixBase_col.out /// -EIGEN_DOC_BLOCK_ADDONS_INNER_PANEL_IF(column-major) +EIGEN_DOC_BLOCK_ADDONS_INNER_PANEL_IF(column - major) /** - * \sa row(), class Block */ + * \sa row(), class Block */ EIGEN_DEVICE_FUNC -inline ColXpr col(Index i) -{ - return ColXpr(derived(), i); -} +inline ColXpr col(Index i) { return ColXpr(derived(), i); } /// This is the const version of col(). EIGEN_DEVICE_FUNC -inline ConstColXpr col(Index i) const -{ - return ConstColXpr(derived(), i); -} +inline ConstColXpr col(Index i) const { return ConstColXpr(derived(), i); } /// \returns an expression of the \a i-th row of *this. Note that the numbering starts at 0. /// /// Example: \include MatrixBase_row.cpp /// Output: \verbinclude MatrixBase_row.out /// -EIGEN_DOC_BLOCK_ADDONS_INNER_PANEL_IF(row-major) +EIGEN_DOC_BLOCK_ADDONS_INNER_PANEL_IF(row - major) /** - * \sa col(), class Block */ + * \sa col(), class Block */ EIGEN_DEVICE_FUNC -inline RowXpr row(Index i) -{ - return RowXpr(derived(), i); -} +inline RowXpr row(Index i) { return RowXpr(derived(), i); } /// This is the const version of row(). */ EIGEN_DEVICE_FUNC -inline ConstRowXpr row(Index i) const -{ - return ConstRowXpr(derived(), i); -} +inline ConstRowXpr row(Index i) const { return ConstRowXpr(derived(), i); } /// \returns a dynamic-size expression of a segment (i.e. a vector block) in *this. /// @@ -976,9 +911,7 @@ inline ConstSegmentReturnType tail(Index n) const /// /// \sa class Block /// -template -EIGEN_DEVICE_FUNC -inline typename FixedSegmentReturnType::Type segment(Index start, Index n = N) +template EIGEN_DEVICE_FUNC inline typename FixedSegmentReturnType::Type segment(Index start, Index n = N) { EIGEN_STATIC_ASSERT_VECTOR_ONLY(Derived) return typename FixedSegmentReturnType::Type(derived(), start, n); @@ -986,8 +919,7 @@ inline typename FixedSegmentReturnType::Type segment(Index start, Index n = N /// This is the const version of segment(Index). template -EIGEN_DEVICE_FUNC -inline typename ConstFixedSegmentReturnType::Type segment(Index start, Index n = N) const +EIGEN_DEVICE_FUNC inline typename ConstFixedSegmentReturnType::Type segment(Index start, Index n = N) const { EIGEN_STATIC_ASSERT_VECTOR_ONLY(Derived) return typename ConstFixedSegmentReturnType::Type(derived(), start, n); @@ -1008,18 +940,14 @@ inline typename ConstFixedSegmentReturnType::Type segment(Index start, Index /// /// \sa class Block /// -template -EIGEN_DEVICE_FUNC -inline typename FixedSegmentReturnType::Type head(Index n = N) +template EIGEN_DEVICE_FUNC inline typename FixedSegmentReturnType::Type head(Index n = N) { EIGEN_STATIC_ASSERT_VECTOR_ONLY(Derived) return typename FixedSegmentReturnType::Type(derived(), 0, n); } /// This is the const version of head(). -template -EIGEN_DEVICE_FUNC -inline typename ConstFixedSegmentReturnType::Type head(Index n = N) const +template EIGEN_DEVICE_FUNC inline typename ConstFixedSegmentReturnType::Type head(Index n = N) const { EIGEN_STATIC_ASSERT_VECTOR_ONLY(Derived) return typename ConstFixedSegmentReturnType::Type(derived(), 0, n); @@ -1040,18 +968,14 @@ inline typename ConstFixedSegmentReturnType::Type head(Index n = N) const /// /// \sa class Block /// -template -EIGEN_DEVICE_FUNC -inline typename FixedSegmentReturnType::Type tail(Index n = N) +template EIGEN_DEVICE_FUNC inline typename FixedSegmentReturnType::Type tail(Index n = N) { EIGEN_STATIC_ASSERT_VECTOR_ONLY(Derived) return typename FixedSegmentReturnType::Type(derived(), size() - n); } /// This is the const version of tail. -template -EIGEN_DEVICE_FUNC -inline typename ConstFixedSegmentReturnType::Type tail(Index n = N) const +template EIGEN_DEVICE_FUNC inline typename ConstFixedSegmentReturnType::Type tail(Index n = N) const { EIGEN_STATIC_ASSERT_VECTOR_ONLY(Derived) return typename ConstFixedSegmentReturnType::Type(derived(), size() - n); diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/plugins/CommonCwiseBinaryOps.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/plugins/CommonCwiseBinaryOps.h index 8b6730ed..4d79dc87 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/plugins/CommonCwiseBinaryOps.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/plugins/CommonCwiseBinaryOps.h @@ -11,105 +11,105 @@ // This file is a base class plugin containing common coefficient wise functions. /** \returns an expression of the difference of \c *this and \a other - * - * \note If you want to substract a given scalar from all coefficients, see Cwise::operator-(). - * - * \sa class CwiseBinaryOp, operator-=() - */ -EIGEN_MAKE_CWISE_BINARY_OP(operator-,difference) + * + * \note If you want to substract a given scalar from all coefficients, see Cwise::operator-(). + * + * \sa class CwiseBinaryOp, operator-=() + */ +EIGEN_MAKE_CWISE_BINARY_OP(operator-, difference) /** \returns an expression of the sum of \c *this and \a other - * - * \note If you want to add a given scalar to all coefficients, see Cwise::operator+(). - * - * \sa class CwiseBinaryOp, operator+=() - */ -EIGEN_MAKE_CWISE_BINARY_OP(operator+,sum) + * + * \note If you want to add a given scalar to all coefficients, see Cwise::operator+(). + * + * \sa class CwiseBinaryOp, operator+=() + */ +EIGEN_MAKE_CWISE_BINARY_OP(operator+, sum) /** \returns an expression of a custom coefficient-wise operator \a func of *this and \a other - * - * The template parameter \a CustomBinaryOp is the type of the functor - * of the custom operator (see class CwiseBinaryOp for an example) - * - * Here is an example illustrating the use of custom functors: - * \include class_CwiseBinaryOp.cpp - * Output: \verbinclude class_CwiseBinaryOp.out - * - * \sa class CwiseBinaryOp, operator+(), operator-(), cwiseProduct() - */ + * + * The template parameter \a CustomBinaryOp is the type of the functor + * of the custom operator (see class CwiseBinaryOp for an example) + * + * Here is an example illustrating the use of custom functors: + * \include class_CwiseBinaryOp.cpp + * Output: \verbinclude class_CwiseBinaryOp.out + * + * \sa class CwiseBinaryOp, operator+(), operator-(), cwiseProduct() + */ template -EIGEN_DEVICE_FUNC -EIGEN_STRONG_INLINE const CwiseBinaryOp -binaryExpr(const EIGEN_CURRENT_STORAGE_BASE_CLASS &other, const CustomBinaryOp& func = CustomBinaryOp()) const +EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const CwiseBinaryOp binaryExpr( + const EIGEN_CURRENT_STORAGE_BASE_CLASS &other, + const CustomBinaryOp &func = CustomBinaryOp()) const { return CwiseBinaryOp(derived(), other.derived(), func); } #ifndef EIGEN_PARSED_BY_DOXYGEN -EIGEN_MAKE_SCALAR_BINARY_OP(operator*,product) +EIGEN_MAKE_SCALAR_BINARY_OP(operator*, product) #else /** \returns an expression of \c *this scaled by the scalar factor \a scalar - * - * \tparam T is the scalar type of \a scalar. It must be compatible with the scalar type of the given expression. - */ + * + * \tparam T is the scalar type of \a scalar. It must be compatible with the scalar type of the given expression. + */ template -const CwiseBinaryOp,Derived,Constant > operator*(const T& scalar) const; +const CwiseBinaryOp, Derived, Constant> operator*(const T &scalar) const; /** \returns an expression of \a expr scaled by the scalar factor \a scalar - * - * \tparam T is the scalar type of \a scalar. It must be compatible with the scalar type of the given expression. - */ -template friend -const CwiseBinaryOp,Constant,Derived> operator*(const T& scalar, const StorageBaseType& expr); + * + * \tparam T is the scalar type of \a scalar. It must be compatible with the scalar type of the given expression. + */ +template +friend const CwiseBinaryOp, Constant, Derived> operator*(const T &scalar, + const StorageBaseType &expr); #endif - #ifndef EIGEN_PARSED_BY_DOXYGEN -EIGEN_MAKE_SCALAR_BINARY_OP_ONTHERIGHT(operator/,quotient) +EIGEN_MAKE_SCALAR_BINARY_OP_ONTHERIGHT(operator/, quotient) #else /** \returns an expression of \c *this divided by the scalar value \a scalar - * - * \tparam T is the scalar type of \a scalar. It must be compatible with the scalar type of the given expression. - */ + * + * \tparam T is the scalar type of \a scalar. It must be compatible with the scalar type of the given expression. + */ template -const CwiseBinaryOp,Derived,Constant > operator/(const T& scalar) const; +const CwiseBinaryOp, Derived, Constant> operator/(const T &scalar) const; #endif /** \returns an expression of the coefficient-wise boolean \b and operator of \c *this and \a other - * - * \warning this operator is for expression of bool only. - * - * Example: \include Cwise_boolean_and.cpp - * Output: \verbinclude Cwise_boolean_and.out - * - * \sa operator||(), select() - */ + * + * \warning this operator is for expression of bool only. + * + * Example: \include Cwise_boolean_and.cpp + * Output: \verbinclude Cwise_boolean_and.out + * + * \sa operator||(), select() + */ template -EIGEN_DEVICE_FUNC -inline const CwiseBinaryOp -operator&&(const EIGEN_CURRENT_STORAGE_BASE_CLASS &other) const +EIGEN_DEVICE_FUNC inline const CwiseBinaryOp + operator&&(const EIGEN_CURRENT_STORAGE_BASE_CLASS &other) const { - EIGEN_STATIC_ASSERT((internal::is_same::value && internal::is_same::value), - THIS_METHOD_IS_ONLY_FOR_EXPRESSIONS_OF_BOOL); - return CwiseBinaryOp(derived(),other.derived()); + EIGEN_STATIC_ASSERT( + (internal::is_same::value && internal::is_same::value), + THIS_METHOD_IS_ONLY_FOR_EXPRESSIONS_OF_BOOL); + return CwiseBinaryOp(derived(), other.derived()); } /** \returns an expression of the coefficient-wise boolean \b or operator of \c *this and \a other - * - * \warning this operator is for expression of bool only. - * - * Example: \include Cwise_boolean_or.cpp - * Output: \verbinclude Cwise_boolean_or.out - * - * \sa operator&&(), select() - */ + * + * \warning this operator is for expression of bool only. + * + * Example: \include Cwise_boolean_or.cpp + * Output: \verbinclude Cwise_boolean_or.out + * + * \sa operator&&(), select() + */ template -EIGEN_DEVICE_FUNC -inline const CwiseBinaryOp -operator||(const EIGEN_CURRENT_STORAGE_BASE_CLASS &other) const +EIGEN_DEVICE_FUNC inline const CwiseBinaryOp + operator||(const EIGEN_CURRENT_STORAGE_BASE_CLASS &other) const { - EIGEN_STATIC_ASSERT((internal::is_same::value && internal::is_same::value), - THIS_METHOD_IS_ONLY_FOR_EXPRESSIONS_OF_BOOL); - return CwiseBinaryOp(derived(),other.derived()); + EIGEN_STATIC_ASSERT( + (internal::is_same::value && internal::is_same::value), + THIS_METHOD_IS_ONLY_FOR_EXPRESSIONS_OF_BOOL); + return CwiseBinaryOp(derived(), other.derived()); } diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/plugins/CommonCwiseUnaryOps.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/plugins/CommonCwiseUnaryOps.h index 89f4faaa..b811ec6a 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/plugins/CommonCwiseUnaryOps.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/plugins/CommonCwiseUnaryOps.h @@ -14,19 +14,16 @@ /** \internal the return type of conjugate() */ typedef typename internal::conditional::IsComplex, - const CwiseUnaryOp, const Derived>, - const Derived& - >::type ConjugateReturnType; + const CwiseUnaryOp, const Derived>, + const Derived &>::type ConjugateReturnType; /** \internal the return type of real() const */ typedef typename internal::conditional::IsComplex, - const CwiseUnaryOp, const Derived>, - const Derived& - >::type RealReturnType; + const CwiseUnaryOp, const Derived>, + const Derived &>::type RealReturnType; /** \internal the return type of real() */ typedef typename internal::conditional::IsComplex, - CwiseUnaryView, Derived>, - Derived& - >::type NonConstRealReturnType; + CwiseUnaryView, Derived>, + Derived &>::type NonConstRealReturnType; /** \internal the return type of imag() const */ typedef CwiseUnaryOp, const Derived> ImagReturnType; /** \internal the return type of imag() */ @@ -34,65 +31,59 @@ typedef CwiseUnaryView, Derived> NonConstIm typedef CwiseUnaryOp, const Derived> NegativeReturnType; -#endif // not EIGEN_PARSED_BY_DOXYGEN +#endif// not EIGEN_PARSED_BY_DOXYGEN /// \returns an expression of the opposite of \c *this /// -EIGEN_DOC_UNARY_ADDONS(operator-,opposite) +EIGEN_DOC_UNARY_ADDONS(operator-, opposite) /// EIGEN_DEVICE_FUNC -inline const NegativeReturnType -operator-() const { return NegativeReturnType(derived()); } +inline const NegativeReturnType operator-() const { return NegativeReturnType(derived()); } -template struct CastXpr { typedef typename internal::cast_return_type, const Derived> >::type Type; }; +template struct CastXpr +{ + typedef typename internal::cast_return_type, const Derived>>::type Type; +}; /// \returns an expression of \c *this with the \a Scalar type casted to /// \a NewScalar. /// /// The template parameter \a NewScalar is the type we are casting the scalars to. /// -EIGEN_DOC_UNARY_ADDONS(cast,conversion function) +EIGEN_DOC_UNARY_ADDONS(cast, conversion function) /// /// \sa class CwiseUnaryOp /// -template -EIGEN_DEVICE_FUNC -typename CastXpr::Type -cast() const +template EIGEN_DEVICE_FUNC typename CastXpr::Type cast() const { return typename CastXpr::Type(derived()); } /// \returns an expression of the complex conjugate of \c *this. /// -EIGEN_DOC_UNARY_ADDONS(conjugate,complex conjugate) +EIGEN_DOC_UNARY_ADDONS(conjugate, complex conjugate) /// /// \sa Math functions, MatrixBase::adjoint() EIGEN_DEVICE_FUNC -inline ConjugateReturnType -conjugate() const -{ - return ConjugateReturnType(derived()); -} +inline ConjugateReturnType conjugate() const { return ConjugateReturnType(derived()); } /// \returns a read-only expression of the real part of \c *this. /// -EIGEN_DOC_UNARY_ADDONS(real,real part function) +EIGEN_DOC_UNARY_ADDONS(real, real part function) /// /// \sa imag() EIGEN_DEVICE_FUNC -inline RealReturnType -real() const { return RealReturnType(derived()); } +inline RealReturnType real() const { return RealReturnType(derived()); } /// \returns an read-only expression of the imaginary part of \c *this. /// -EIGEN_DOC_UNARY_ADDONS(imag,imaginary part function) +EIGEN_DOC_UNARY_ADDONS(imag, imaginary part function) /// /// \sa real() EIGEN_DEVICE_FUNC -inline const ImagReturnType -imag() const { return ImagReturnType(derived()); } +inline const ImagReturnType imag() const { return ImagReturnType(derived()); } /// \brief Apply a unary operator coefficient-wise /// \param[in] func Functor implementing the unary operator @@ -111,14 +102,13 @@ imag() const { return ImagReturnType(derived()); } /// \include class_CwiseUnaryOp.cpp /// Output: \verbinclude class_CwiseUnaryOp.out /// -EIGEN_DOC_UNARY_ADDONS(unaryExpr,unary function) +EIGEN_DOC_UNARY_ADDONS(unaryExpr, unary function) /// /// \sa unaryViewExpr, binaryExpr, class CwiseUnaryOp /// template -EIGEN_DEVICE_FUNC -inline const CwiseUnaryOp -unaryExpr(const CustomUnaryOp& func = CustomUnaryOp()) const +EIGEN_DEVICE_FUNC inline const CwiseUnaryOp unaryExpr( + const CustomUnaryOp &func = CustomUnaryOp()) const { return CwiseUnaryOp(derived(), func); } @@ -132,32 +122,29 @@ unaryExpr(const CustomUnaryOp& func = CustomUnaryOp()) const /// \include class_CwiseUnaryOp.cpp /// Output: \verbinclude class_CwiseUnaryOp.out /// -EIGEN_DOC_UNARY_ADDONS(unaryViewExpr,unary function) +EIGEN_DOC_UNARY_ADDONS(unaryViewExpr, unary function) /// /// \sa unaryExpr, binaryExpr class CwiseUnaryOp /// template -EIGEN_DEVICE_FUNC -inline const CwiseUnaryView -unaryViewExpr(const CustomViewOp& func = CustomViewOp()) const +EIGEN_DEVICE_FUNC inline const CwiseUnaryView unaryViewExpr( + const CustomViewOp &func = CustomViewOp()) const { return CwiseUnaryView(derived(), func); } /// \returns a non const expression of the real part of \c *this. /// -EIGEN_DOC_UNARY_ADDONS(real,real part function) +EIGEN_DOC_UNARY_ADDONS(real, real part function) /// /// \sa imag() EIGEN_DEVICE_FUNC -inline NonConstRealReturnType -real() { return NonConstRealReturnType(derived()); } +inline NonConstRealReturnType real() { return NonConstRealReturnType(derived()); } /// \returns a non const expression of the imaginary part of \c *this. /// -EIGEN_DOC_UNARY_ADDONS(imag,imaginary part function) +EIGEN_DOC_UNARY_ADDONS(imag, imaginary part function) /// /// \sa real() EIGEN_DEVICE_FUNC -inline NonConstImagReturnType -imag() { return NonConstImagReturnType(derived()); } +inline NonConstImagReturnType imag() { return NonConstImagReturnType(derived()); } diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/plugins/MatrixCwiseBinaryOps.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/plugins/MatrixCwiseBinaryOps.h index f1084abe..97570842 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/plugins/MatrixCwiseBinaryOps.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/plugins/MatrixCwiseBinaryOps.h @@ -11,142 +11,147 @@ // This file is a base class plugin containing matrix specifics coefficient wise functions. /** \returns an expression of the Schur product (coefficient wise product) of *this and \a other - * - * Example: \include MatrixBase_cwiseProduct.cpp - * Output: \verbinclude MatrixBase_cwiseProduct.out - * - * \sa class CwiseBinaryOp, cwiseAbs2 - */ + * + * Example: \include MatrixBase_cwiseProduct.cpp + * Output: \verbinclude MatrixBase_cwiseProduct.out + * + * \sa class CwiseBinaryOp, cwiseAbs2 + */ template -EIGEN_DEVICE_FUNC -EIGEN_STRONG_INLINE const EIGEN_CWISE_BINARY_RETURN_TYPE(Derived,OtherDerived,product) -cwiseProduct(const EIGEN_CURRENT_STORAGE_BASE_CLASS &other) const +EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const EIGEN_CWISE_BINARY_RETURN_TYPE(Derived, OtherDerived, product) + cwiseProduct(const EIGEN_CURRENT_STORAGE_BASE_CLASS &other) const { - return EIGEN_CWISE_BINARY_RETURN_TYPE(Derived,OtherDerived,product)(derived(), other.derived()); + return EIGEN_CWISE_BINARY_RETURN_TYPE(Derived, OtherDerived, product)(derived(), other.derived()); } /** \returns an expression of the coefficient-wise == operator of *this and \a other - * - * \warning this performs an exact comparison, which is generally a bad idea with floating-point types. - * In order to check for equality between two vectors or matrices with floating-point coefficients, it is - * generally a far better idea to use a fuzzy comparison as provided by isApprox() and - * isMuchSmallerThan(). - * - * Example: \include MatrixBase_cwiseEqual.cpp - * Output: \verbinclude MatrixBase_cwiseEqual.out - * - * \sa cwiseNotEqual(), isApprox(), isMuchSmallerThan() - */ + * + * \warning this performs an exact comparison, which is generally a bad idea with floating-point types. + * In order to check for equality between two vectors or matrices with floating-point coefficients, it is + * generally a far better idea to use a fuzzy comparison as provided by isApprox() and + * isMuchSmallerThan(). + * + * Example: \include MatrixBase_cwiseEqual.cpp + * Output: \verbinclude MatrixBase_cwiseEqual.out + * + * \sa cwiseNotEqual(), isApprox(), isMuchSmallerThan() + */ template -EIGEN_DEVICE_FUNC -inline const CwiseBinaryOp, const Derived, const OtherDerived> -cwiseEqual(const EIGEN_CURRENT_STORAGE_BASE_CLASS &other) const +EIGEN_DEVICE_FUNC inline const CwiseBinaryOp, const Derived, const OtherDerived> cwiseEqual( + const EIGEN_CURRENT_STORAGE_BASE_CLASS &other) const { return CwiseBinaryOp, const Derived, const OtherDerived>(derived(), other.derived()); } /** \returns an expression of the coefficient-wise != operator of *this and \a other - * - * \warning this performs an exact comparison, which is generally a bad idea with floating-point types. - * In order to check for equality between two vectors or matrices with floating-point coefficients, it is - * generally a far better idea to use a fuzzy comparison as provided by isApprox() and - * isMuchSmallerThan(). - * - * Example: \include MatrixBase_cwiseNotEqual.cpp - * Output: \verbinclude MatrixBase_cwiseNotEqual.out - * - * \sa cwiseEqual(), isApprox(), isMuchSmallerThan() - */ + * + * \warning this performs an exact comparison, which is generally a bad idea with floating-point types. + * In order to check for equality between two vectors or matrices with floating-point coefficients, it is + * generally a far better idea to use a fuzzy comparison as provided by isApprox() and + * isMuchSmallerThan(). + * + * Example: \include MatrixBase_cwiseNotEqual.cpp + * Output: \verbinclude MatrixBase_cwiseNotEqual.out + * + * \sa cwiseEqual(), isApprox(), isMuchSmallerThan() + */ template -EIGEN_DEVICE_FUNC -inline const CwiseBinaryOp, const Derived, const OtherDerived> -cwiseNotEqual(const EIGEN_CURRENT_STORAGE_BASE_CLASS &other) const +EIGEN_DEVICE_FUNC inline const CwiseBinaryOp, const Derived, const OtherDerived> + cwiseNotEqual(const EIGEN_CURRENT_STORAGE_BASE_CLASS &other) const { return CwiseBinaryOp, const Derived, const OtherDerived>(derived(), other.derived()); } /** \returns an expression of the coefficient-wise min of *this and \a other - * - * Example: \include MatrixBase_cwiseMin.cpp - * Output: \verbinclude MatrixBase_cwiseMin.out - * - * \sa class CwiseBinaryOp, max() - */ + * + * Example: \include MatrixBase_cwiseMin.cpp + * Output: \verbinclude MatrixBase_cwiseMin.out + * + * \sa class CwiseBinaryOp, max() + */ template -EIGEN_DEVICE_FUNC -EIGEN_STRONG_INLINE const CwiseBinaryOp, const Derived, const OtherDerived> -cwiseMin(const EIGEN_CURRENT_STORAGE_BASE_CLASS &other) const +EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const + CwiseBinaryOp, const Derived, const OtherDerived> + cwiseMin(const EIGEN_CURRENT_STORAGE_BASE_CLASS &other) const { - return CwiseBinaryOp, const Derived, const OtherDerived>(derived(), other.derived()); + return CwiseBinaryOp, const Derived, const OtherDerived>( + derived(), other.derived()); } /** \returns an expression of the coefficient-wise min of *this and scalar \a other - * - * \sa class CwiseBinaryOp, min() - */ + * + * \sa class CwiseBinaryOp, min() + */ EIGEN_DEVICE_FUNC -EIGEN_STRONG_INLINE const CwiseBinaryOp, const Derived, const ConstantReturnType> -cwiseMin(const Scalar &other) const +EIGEN_STRONG_INLINE const + CwiseBinaryOp, const Derived, const ConstantReturnType> + cwiseMin(const Scalar &other) const { return cwiseMin(Derived::Constant(rows(), cols(), other)); } /** \returns an expression of the coefficient-wise max of *this and \a other - * - * Example: \include MatrixBase_cwiseMax.cpp - * Output: \verbinclude MatrixBase_cwiseMax.out - * - * \sa class CwiseBinaryOp, min() - */ + * + * Example: \include MatrixBase_cwiseMax.cpp + * Output: \verbinclude MatrixBase_cwiseMax.out + * + * \sa class CwiseBinaryOp, min() + */ template -EIGEN_DEVICE_FUNC -EIGEN_STRONG_INLINE const CwiseBinaryOp, const Derived, const OtherDerived> -cwiseMax(const EIGEN_CURRENT_STORAGE_BASE_CLASS &other) const +EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const + CwiseBinaryOp, const Derived, const OtherDerived> + cwiseMax(const EIGEN_CURRENT_STORAGE_BASE_CLASS &other) const { - return CwiseBinaryOp, const Derived, const OtherDerived>(derived(), other.derived()); + return CwiseBinaryOp, const Derived, const OtherDerived>( + derived(), other.derived()); } /** \returns an expression of the coefficient-wise max of *this and scalar \a other - * - * \sa class CwiseBinaryOp, min() - */ + * + * \sa class CwiseBinaryOp, min() + */ EIGEN_DEVICE_FUNC -EIGEN_STRONG_INLINE const CwiseBinaryOp, const Derived, const ConstantReturnType> -cwiseMax(const Scalar &other) const +EIGEN_STRONG_INLINE const + CwiseBinaryOp, const Derived, const ConstantReturnType> + cwiseMax(const Scalar &other) const { return cwiseMax(Derived::Constant(rows(), cols(), other)); } /** \returns an expression of the coefficient-wise quotient of *this and \a other - * - * Example: \include MatrixBase_cwiseQuotient.cpp - * Output: \verbinclude MatrixBase_cwiseQuotient.out - * - * \sa class CwiseBinaryOp, cwiseProduct(), cwiseInverse() - */ + * + * Example: \include MatrixBase_cwiseQuotient.cpp + * Output: \verbinclude MatrixBase_cwiseQuotient.out + * + * \sa class CwiseBinaryOp, cwiseProduct(), cwiseInverse() + */ template -EIGEN_DEVICE_FUNC -EIGEN_STRONG_INLINE const CwiseBinaryOp, const Derived, const OtherDerived> -cwiseQuotient(const EIGEN_CURRENT_STORAGE_BASE_CLASS &other) const +EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const + CwiseBinaryOp, const Derived, const OtherDerived> + cwiseQuotient(const EIGEN_CURRENT_STORAGE_BASE_CLASS &other) const { - return CwiseBinaryOp, const Derived, const OtherDerived>(derived(), other.derived()); + return CwiseBinaryOp, const Derived, const OtherDerived>( + derived(), other.derived()); } -typedef CwiseBinaryOp, const Derived, const ConstantReturnType> CwiseScalarEqualReturnType; +typedef CwiseBinaryOp, + const Derived, + const ConstantReturnType> + CwiseScalarEqualReturnType; /** \returns an expression of the coefficient-wise == operator of \c *this and a scalar \a s - * - * \warning this performs an exact comparison, which is generally a bad idea with floating-point types. - * In order to check for equality between two vectors or matrices with floating-point coefficients, it is - * generally a far better idea to use a fuzzy comparison as provided by isApprox() and - * isMuchSmallerThan(). - * - * \sa cwiseEqual(const MatrixBase &) const - */ + * + * \warning this performs an exact comparison, which is generally a bad idea with floating-point types. + * In order to check for equality between two vectors or matrices with floating-point coefficients, it is + * generally a far better idea to use a fuzzy comparison as provided by isApprox() and + * isMuchSmallerThan(). + * + * \sa cwiseEqual(const MatrixBase &) const + */ EIGEN_DEVICE_FUNC -inline const CwiseScalarEqualReturnType -cwiseEqual(const Scalar& s) const +inline const CwiseScalarEqualReturnType cwiseEqual(const Scalar &s) const { - return CwiseScalarEqualReturnType(derived(), Derived::Constant(rows(), cols(), s), internal::scalar_cmp_op()); + return CwiseScalarEqualReturnType( + derived(), Derived::Constant(rows(), cols(), s), internal::scalar_cmp_op()); } diff --git a/filmulator-gui/core/nlmeans/eigen/Eigen/src/plugins/MatrixCwiseUnaryOps.h b/filmulator-gui/core/nlmeans/eigen/Eigen/src/plugins/MatrixCwiseUnaryOps.h index b1be3d56..8d6ab840 100644 --- a/filmulator-gui/core/nlmeans/eigen/Eigen/src/plugins/MatrixCwiseUnaryOps.h +++ b/filmulator-gui/core/nlmeans/eigen/Eigen/src/plugins/MatrixCwiseUnaryOps.h @@ -23,50 +23,46 @@ typedef CwiseUnaryOp, const Derived> CwiseIn /// Example: \include MatrixBase_cwiseAbs.cpp /// Output: \verbinclude MatrixBase_cwiseAbs.out /// -EIGEN_DOC_UNARY_ADDONS(cwiseAbs,absolute value) +EIGEN_DOC_UNARY_ADDONS(cwiseAbs, absolute value) /// /// \sa cwiseAbs2() /// EIGEN_DEVICE_FUNC -EIGEN_STRONG_INLINE const CwiseAbsReturnType -cwiseAbs() const { return CwiseAbsReturnType(derived()); } +EIGEN_STRONG_INLINE const CwiseAbsReturnType cwiseAbs() const { return CwiseAbsReturnType(derived()); } /// \returns an expression of the coefficient-wise squared absolute value of \c *this /// /// Example: \include MatrixBase_cwiseAbs2.cpp /// Output: \verbinclude MatrixBase_cwiseAbs2.out /// -EIGEN_DOC_UNARY_ADDONS(cwiseAbs2,squared absolute value) +EIGEN_DOC_UNARY_ADDONS(cwiseAbs2, squared absolute value) /// /// \sa cwiseAbs() /// EIGEN_DEVICE_FUNC -EIGEN_STRONG_INLINE const CwiseAbs2ReturnType -cwiseAbs2() const { return CwiseAbs2ReturnType(derived()); } +EIGEN_STRONG_INLINE const CwiseAbs2ReturnType cwiseAbs2() const { return CwiseAbs2ReturnType(derived()); } /// \returns an expression of the coefficient-wise square root of *this. /// /// Example: \include MatrixBase_cwiseSqrt.cpp /// Output: \verbinclude MatrixBase_cwiseSqrt.out /// -EIGEN_DOC_UNARY_ADDONS(cwiseSqrt,square-root) +EIGEN_DOC_UNARY_ADDONS(cwiseSqrt, square - root) /// /// \sa cwisePow(), cwiseSquare() /// EIGEN_DEVICE_FUNC -inline const CwiseSqrtReturnType -cwiseSqrt() const { return CwiseSqrtReturnType(derived()); } +inline const CwiseSqrtReturnType cwiseSqrt() const { return CwiseSqrtReturnType(derived()); } /// \returns an expression of the coefficient-wise signum of *this. /// /// Example: \include MatrixBase_cwiseSign.cpp /// Output: \verbinclude MatrixBase_cwiseSign.out /// -EIGEN_DOC_UNARY_ADDONS(cwiseSign,sign function) +EIGEN_DOC_UNARY_ADDONS(cwiseSign, sign function) /// EIGEN_DEVICE_FUNC -inline const CwiseSignReturnType -cwiseSign() const { return CwiseSignReturnType(derived()); } +inline const CwiseSignReturnType cwiseSign() const { return CwiseSignReturnType(derived()); } /// \returns an expression of the coefficient-wise inverse of *this. @@ -74,12 +70,9 @@ cwiseSign() const { return CwiseSignReturnType(derived()); } /// Example: \include MatrixBase_cwiseInverse.cpp /// Output: \verbinclude MatrixBase_cwiseInverse.out /// -EIGEN_DOC_UNARY_ADDONS(cwiseInverse,inverse) +EIGEN_DOC_UNARY_ADDONS(cwiseInverse, inverse) /// /// \sa cwiseProduct() /// EIGEN_DEVICE_FUNC -inline const CwiseInverseReturnType -cwiseInverse() const { return CwiseInverseReturnType(derived()); } - - +inline const CwiseInverseReturnType cwiseInverse() const { return CwiseInverseReturnType(derived()); } diff --git a/filmulator-gui/core/nlmeans/eigen/bench/BenchSparseUtil.h b/filmulator-gui/core/nlmeans/eigen/bench/BenchSparseUtil.h index 13981f6b..498108dd 100644 --- a/filmulator-gui/core/nlmeans/eigen/bench/BenchSparseUtil.h +++ b/filmulator-gui/core/nlmeans/eigen/bench/BenchSparseUtil.h @@ -20,77 +20,68 @@ using namespace Eigen; #endif typedef SCALAR Scalar; -typedef Matrix DenseMatrix; -typedef Matrix DenseVector; +typedef Matrix DenseMatrix; +typedef Matrix DenseVector; typedef SparseMatrix EigenSparseMatrix; -void fillMatrix(float density, int rows, int cols, EigenSparseMatrix& dst) +void fillMatrix(float density, int rows, int cols, EigenSparseMatrix &dst) { - dst.reserve(double(rows)*cols*density); - for(int j = 0; j < cols; j++) - { - for(int i = 0; i < rows; i++) - { - Scalar v = (internal::random(0,1) < density) ? internal::random() : 0; - if (v!=0) - dst.insert(i,j) = v; + dst.reserve(double(rows) * cols * density); + for (int j = 0; j < cols; j++) { + for (int i = 0; i < rows; i++) { + Scalar v = (internal::random(0, 1) < density) ? internal::random() : 0; + if (v != 0) dst.insert(i, j) = v; } } dst.finalize(); } -void fillMatrix2(int nnzPerCol, int rows, int cols, EigenSparseMatrix& dst) +void fillMatrix2(int nnzPerCol, int rows, int cols, EigenSparseMatrix &dst) { -// std::cout << "alloc " << nnzPerCol*cols << "\n"; - dst.reserve(nnzPerCol*cols); - for(int j = 0; j < cols; j++) - { + // std::cout << "alloc " << nnzPerCol*cols << "\n"; + dst.reserve(nnzPerCol * cols); + for (int j = 0; j < cols; j++) { std::set aux; - for(int i = 0; i < nnzPerCol; i++) - { - int k = internal::random(0,rows-1); - while (aux.find(k)!=aux.end()) - k = internal::random(0,rows-1); + for (int i = 0; i < nnzPerCol; i++) { + int k = internal::random(0, rows - 1); + while (aux.find(k) != aux.end()) k = internal::random(0, rows - 1); aux.insert(k); - dst.insert(k,j) = internal::random(); + dst.insert(k, j) = internal::random(); } } dst.finalize(); } -void eiToDense(const EigenSparseMatrix& src, DenseMatrix& dst) +void eiToDense(const EigenSparseMatrix &src, DenseMatrix &dst) { dst.setZero(); - for (int j=0; j GmmSparse; -typedef gmm::col_matrix< gmm::wsvector > GmmDynSparse; -void eiToGmm(const EigenSparseMatrix& src, GmmSparse& dst) +typedef gmm::col_matrix> GmmDynSparse; +void eiToGmm(const EigenSparseMatrix &src, GmmSparse &dst) { GmmDynSparse tmp(src.rows(), src.cols()); - for (int j=0; j -typedef mtl::compressed2D > MtlSparse; -typedef mtl::compressed2D > MtlSparseRowMajor; -void eiToMtl(const EigenSparseMatrix& src, MtlSparse& dst) +typedef mtl::compressed2D> MtlSparse; +typedef mtl::compressed2D> MtlSparseRowMajor; +void eiToMtl(const EigenSparseMatrix &src, MtlSparse &dst) { mtl::matrix::inserter ins(dst); - for (int j=0; j -#include #include -#include -#include +#include #include -#include #include +#include +#include +#include +#include -typedef boost::numeric::ublas::compressed_matrix UBlasSparse; +typedef boost::numeric::ublas::compressed_matrix UBlasSparse; -void eiToUblas(const EigenSparseMatrix& src, UBlasSparse& dst) +void eiToUblas(const EigenSparseMatrix &src, UBlasSparse &dst) { dst.resize(src.rows(), src.cols(), false); - for (int j=0; j -void eiToUblasVec(const EigenType& src, UblasType& dst) +template void eiToUblasVec(const EigenType &src, UblasType &dst) { dst.resize(src.size()); - for (int j=0; j +#ifndef NOMINMAX +#define NOMINMAX +#define EIGEN_BT_UNDEF_NOMINMAX +#endif +#ifndef WIN32_LEAN_AND_MEAN +#define WIN32_LEAN_AND_MEAN +#define EIGEN_BT_UNDEF_WIN32_LEAN_AND_MEAN +#endif +#include #elif defined(__APPLE__) #include #else -# include +#include #endif -static void escape(void *p) { - asm volatile("" : : "g"(p) : "memory"); -} +static void escape(void *p) { asm volatile("" : : "g"(p) : "memory"); } -static void clobber() { - asm volatile("" : : : "memory"); -} +static void clobber() { asm volatile("" : : : "memory"); } #include -namespace Eigen -{ +namespace Eigen { -enum { - CPU_TIMER = 0, - REAL_TIMER = 1 -}; +enum { CPU_TIMER = 0, REAL_TIMER = 1 }; /** Elapsed time timer keeping the best try. - * - * On POSIX platforms we use clock_gettime with CLOCK_PROCESS_CPUTIME_ID. - * On Windows we use QueryPerformanceCounter - * - * Important: on linux, you must link with -lrt - */ + * + * On POSIX platforms we use clock_gettime with CLOCK_PROCESS_CPUTIME_ID. + * On Windows we use QueryPerformanceCounter + * + * Important: on linux, you must link with -lrt + */ class BenchTimer { public: - BenchTimer() { #if defined(_WIN32) || defined(__CYGWIN__) @@ -76,61 +67,49 @@ class BenchTimer } inline void start() { - m_starts[CPU_TIMER] = getCpuTime(); + m_starts[CPU_TIMER] = getCpuTime(); m_starts[REAL_TIMER] = getRealTime(); } inline void stop() { m_times[CPU_TIMER] = getCpuTime() - m_starts[CPU_TIMER]; m_times[REAL_TIMER] = getRealTime() - m_starts[REAL_TIMER]; - #if EIGEN_VERSION_AT_LEAST(2,90,0) +#if EIGEN_VERSION_AT_LEAST(2, 90, 0) m_bests = m_bests.cwiseMin(m_times); m_worsts = m_worsts.cwiseMax(m_times); - #else - m_bests(0) = std::min(m_bests(0),m_times(0)); - m_bests(1) = std::min(m_bests(1),m_times(1)); - m_worsts(0) = std::max(m_worsts(0),m_times(0)); - m_worsts(1) = std::max(m_worsts(1),m_times(1)); - #endif +#else + m_bests(0) = std::min(m_bests(0), m_times(0)); + m_bests(1) = std::min(m_bests(1), m_times(1)); + m_worsts(0) = std::max(m_worsts(0), m_times(0)); + m_worsts(1) = std::max(m_worsts(1), m_times(1)); +#endif m_totals += m_times; } /** Return the elapsed time in seconds between the last start/stop pair - */ - inline double value(int TIMER = CPU_TIMER) const - { - return m_times[TIMER]; - } + */ + inline double value(int TIMER = CPU_TIMER) const { return m_times[TIMER]; } /** Return the best elapsed time in seconds - */ - inline double best(int TIMER = CPU_TIMER) const - { - return m_bests[TIMER]; - } + */ + inline double best(int TIMER = CPU_TIMER) const { return m_bests[TIMER]; } /** Return the worst elapsed time in seconds - */ - inline double worst(int TIMER = CPU_TIMER) const - { - return m_worsts[TIMER]; - } + */ + inline double worst(int TIMER = CPU_TIMER) const { return m_worsts[TIMER]; } /** Return the total elapsed time in seconds. - */ - inline double total(int TIMER = CPU_TIMER) const - { - return m_totals[TIMER]; - } + */ + inline double total(int TIMER = CPU_TIMER) const { return m_totals[TIMER]; } inline double getCpuTime() const { #ifdef _WIN32 LARGE_INTEGER query_ticks; QueryPerformanceCounter(&query_ticks); - return query_ticks.QuadPart/m_frequency; + return query_ticks.QuadPart / m_frequency; #elif __APPLE__ - return double(mach_absolute_time())*1e-9; + return double(mach_absolute_time()) * 1e-9; #else timespec ts; clock_gettime(CLOCK_PROCESS_CPUTIME_ID, &ts); @@ -145,7 +124,7 @@ class BenchTimer GetSystemTime(&st); return (double)st.wSecond + 1.e-3 * (double)st.wMilliseconds; #elif __APPLE__ - return double(mach_absolute_time())*1e-9; + return double(mach_absolute_time()) * 1e-9; #else timespec ts; clock_gettime(CLOCK_REALTIME, &ts); @@ -167,29 +146,28 @@ class BenchTimer EIGEN_MAKE_ALIGNED_OPERATOR_NEW }; -#define BENCH(TIMER,TRIES,REP,CODE) { \ - TIMER.reset(); \ - for(int uglyvarname1=0; uglyvarname1 #include "BenchTimer.h" +#include using namespace std; using namespace Eigen; -#include -#include -#include -#include #include +#include #include #include #include +#include +#include +#include #include -template void initMatrix_random(MatrixType& mat) __attribute__((noinline)); -template void initMatrix_random(MatrixType& mat) +template void initMatrix_random(MatrixType &mat) __attribute__((noinline)); +template void initMatrix_random(MatrixType &mat) { mat.setRandom();// = MatrixType::random(mat.rows(), mat.cols()); } -template void initMatrix_identity(MatrixType& mat) __attribute__((noinline)); -template void initMatrix_identity(MatrixType& mat) -{ - mat.setIdentity(); -} +template void initMatrix_identity(MatrixType &mat) __attribute__((noinline)); +template void initMatrix_identity(MatrixType &mat) { mat.setIdentity(); } #ifndef __INTEL_COMPILER -#define DISABLE_SSE_EXCEPTIONS() { \ - int aux; \ - asm( \ - "stmxcsr %[aux] \n\t" \ - "orl $32832, %[aux] \n\t" \ - "ldmxcsr %[aux] \n\t" \ - : : [aux] "m" (aux)); \ -} +#define DISABLE_SSE_EXCEPTIONS() \ + { \ + int aux; \ + asm( \ + "stmxcsr %[aux] \n\t" \ + "orl $32832, %[aux] \n\t" \ + "ldmxcsr %[aux] \n\t" \ + : \ + : [aux] "m"(aux)); \ + } #else -#define DISABLE_SSE_EXCEPTIONS() +#define DISABLE_SSE_EXCEPTIONS() #endif #ifdef BENCH_GMM #include -template -void eiToGmm(const EigenMatrixType& src, GmmMatrixType& dst) +template void eiToGmm(const EigenMatrixType &src, GmmMatrixType &dst) { - dst.resize(src.rows(),src.cols()); - for (int j=0; j -#include #include -template -void eiToGsl(const EigenMatrixType& src, gsl_matrix** dst) +#include +#include +template void eiToGsl(const EigenMatrixType &src, gsl_matrix **dst) { - for (int j=0; j #include -template -void eiToUblas(const EigenMatrixType& src, UblasMatrixType& dst) +template +void eiToUblas(const EigenMatrixType &src, UblasMatrixType &dst) { - dst.resize(src.rows(),src.cols()); - for (int j=0; j -void eiToUblasVec(const EigenType& src, UblasType& dst) +template void eiToUblasVec(const EigenType &src, UblasType &dst) { dst.resize(src.size()); - for (int j=0; j -#include -#include -#include #include -#include -#include -#include #include +#include +#include +#include #include +#include +#include #include +#include +#include #include @@ -31,7 +31,8 @@ bool only_cubic_sizes = false; // see --dump-tables bool dump_tables = false; -uint8_t log2_pot(size_t x) { +uint8_t log2_pot(size_t x) +{ size_t l = 0; while (x >>= 1) l++; return l; @@ -48,7 +49,7 @@ struct size_triple_t uint16_t k, m, n; size_triple_t() : k(0), m(0), n(0) {} size_triple_t(size_t _k, size_t _m, size_t _n) : k(_k), m(_m), n(_n) {} - size_triple_t(const size_triple_t& o) : k(o.k), m(o.m), n(o.n) {} + size_triple_t(const size_triple_t &o) : k(o.k), m(o.m), n(o.n) {} size_triple_t(uint16_t compact) { k = 1 << ((compact & 0xf00) >> 8); @@ -58,10 +59,7 @@ struct size_triple_t bool is_cubic() const { return k == m && m == n; } }; -ostream& operator<<(ostream& s, const size_triple_t& t) -{ - return s << "(" << t.k << ", " << t.m << ", " << t.n << ")"; -} +ostream &operator<<(ostream &s, const size_triple_t &t) { return s << "(" << t.k << ", " << t.m << ", " << t.n << ")"; } struct inputfile_entry_t { @@ -73,19 +71,13 @@ struct inputfile_entry_t struct inputfile_t { - enum class type_t { - unknown, - all_pot_sizes, - default_sizes - }; + enum class type_t { unknown, all_pot_sizes, default_sizes }; string filename; vector entries; type_t type; - inputfile_t(const string& fname) - : filename(fname) - , type(type_t::unknown) + inputfile_t(const string &fname) : filename(fname), type(type_t::unknown) { ifstream stream(filename); if (!stream.is_open()) { @@ -111,73 +103,50 @@ struct inputfile_t type = type_t::default_sizes; continue; } - - if (type == type_t::unknown) { - continue; - } - switch(type) { - case type_t::all_pot_sizes: { - unsigned int product_size, block_size; - float gflops; - int sscanf_result = - sscanf(line.c_str(), "%x %x %f", - &product_size, - &block_size, - &gflops); - if (3 != sscanf_result || - !product_size || - product_size > 0xfff || - !block_size || - block_size > 0xfff || - !isfinite(gflops)) - { - cerr << "ill-formed input file: " << filename << endl; - cerr << "offending line:" << endl << line << endl; - exit(1); - } - if (only_cubic_sizes && !size_triple_t(product_size).is_cubic()) { - continue; - } - inputfile_entry_t entry; - entry.product_size = uint16_t(product_size); - entry.pot_block_size = uint16_t(block_size); - entry.gflops = gflops; - entries.push_back(entry); - break; + + if (type == type_t::unknown) { continue; } + switch (type) { + case type_t::all_pot_sizes: { + unsigned int product_size, block_size; + float gflops; + int sscanf_result = sscanf(line.c_str(), "%x %x %f", &product_size, &block_size, &gflops); + if (3 != sscanf_result || !product_size || product_size > 0xfff || !block_size || block_size > 0xfff + || !isfinite(gflops)) { + cerr << "ill-formed input file: " << filename << endl; + cerr << "offending line:" << endl << line << endl; + exit(1); } - case type_t::default_sizes: { - unsigned int product_size; - float gflops; - int bk, bm, bn; - int sscanf_result = - sscanf(line.c_str(), "%x default(%d, %d, %d) %f", - &product_size, - &bk, &bm, &bn, - &gflops); - if (5 != sscanf_result || - !product_size || - product_size > 0xfff || - !isfinite(gflops)) - { - cerr << "ill-formed input file: " << filename << endl; - cerr << "offending line:" << endl << line << endl; - exit(1); - } - if (only_cubic_sizes && !size_triple_t(product_size).is_cubic()) { - continue; - } - inputfile_entry_t entry; - entry.product_size = uint16_t(product_size); - entry.pot_block_size = 0; - entry.nonpot_block_size = size_triple_t(bk, bm, bn); - entry.gflops = gflops; - entries.push_back(entry); - break; + if (only_cubic_sizes && !size_triple_t(product_size).is_cubic()) { continue; } + inputfile_entry_t entry; + entry.product_size = uint16_t(product_size); + entry.pot_block_size = uint16_t(block_size); + entry.gflops = gflops; + entries.push_back(entry); + break; + } + case type_t::default_sizes: { + unsigned int product_size; + float gflops; + int bk, bm, bn; + int sscanf_result = sscanf(line.c_str(), "%x default(%d, %d, %d) %f", &product_size, &bk, &bm, &bn, &gflops); + if (5 != sscanf_result || !product_size || product_size > 0xfff || !isfinite(gflops)) { + cerr << "ill-formed input file: " << filename << endl; + cerr << "offending line:" << endl << line << endl; + exit(1); } - - default: - break; + if (only_cubic_sizes && !size_triple_t(product_size).is_cubic()) { continue; } + inputfile_entry_t entry; + entry.product_size = uint16_t(product_size); + entry.pot_block_size = 0; + entry.nonpot_block_size = size_triple_t(bk, bm, bn); + entry.gflops = gflops; + entries.push_back(entry); + break; + } + + default: + break; } } stream.close(); @@ -200,7 +169,7 @@ struct preprocessed_inputfile_entry_t float efficiency; }; -bool lower_efficiency(const preprocessed_inputfile_entry_t& e1, const preprocessed_inputfile_entry_t& e2) +bool lower_efficiency(const preprocessed_inputfile_entry_t &e1, const preprocessed_inputfile_entry_t &e2) { return e1.efficiency < e2.efficiency; } @@ -210,19 +179,14 @@ struct preprocessed_inputfile_t string filename; vector entries; - preprocessed_inputfile_t(const inputfile_t& inputfile) - : filename(inputfile.filename) + preprocessed_inputfile_t(const inputfile_t &inputfile) : filename(inputfile.filename) { - if (inputfile.type != inputfile_t::type_t::all_pot_sizes) { - abort(); - } + if (inputfile.type != inputfile_t::type_t::all_pot_sizes) { abort(); } auto it = inputfile.entries.begin(); auto it_first_with_given_product_size = it; while (it != inputfile.entries.end()) { ++it; - if (it == inputfile.entries.end() || - it->product_size != it_first_with_given_product_size->product_size) - { + if (it == inputfile.entries.end() || it->product_size != it_first_with_given_product_size->product_size) { import_input_file_range_one_product_size(it_first_with_given_product_size, it); it_first_with_given_product_size = it; } @@ -230,9 +194,8 @@ struct preprocessed_inputfile_t } private: - void import_input_file_range_one_product_size( - const vector::const_iterator& begin, - const vector::const_iterator& end) + void import_input_file_range_one_product_size(const vector::const_iterator &begin, + const vector::const_iterator &end) { uint16_t product_size = begin->product_size; float max_gflops = 0.0f; @@ -254,23 +217,17 @@ struct preprocessed_inputfile_t } }; -void check_all_files_in_same_exact_order( - const vector& preprocessed_inputfiles) +void check_all_files_in_same_exact_order(const vector &preprocessed_inputfiles) { - if (preprocessed_inputfiles.empty()) { - return; - } + if (preprocessed_inputfiles.empty()) { return; } - const preprocessed_inputfile_t& first_file = preprocessed_inputfiles[0]; + const preprocessed_inputfile_t &first_file = preprocessed_inputfiles[0]; const size_t num_entries = first_file.entries.size(); for (size_t i = 0; i < preprocessed_inputfiles.size(); i++) { if (preprocessed_inputfiles[i].entries.size() != num_entries) { - cerr << "these files have different number of entries: " - << preprocessed_inputfiles[i].filename - << " and " - << first_file.filename - << endl; + cerr << "these files have different number of entries: " << preprocessed_inputfiles[i].filename << " and " + << first_file.filename << endl; exit(1); } } @@ -279,14 +236,10 @@ void check_all_files_in_same_exact_order( const uint16_t entry_product_size = first_file.entries[entry_index].product_size; const uint16_t entry_block_size = first_file.entries[entry_index].block_size; for (size_t file_index = 0; file_index < preprocessed_inputfiles.size(); file_index++) { - const preprocessed_inputfile_t& cur_file = preprocessed_inputfiles[file_index]; - if (cur_file.entries[entry_index].product_size != entry_product_size || - cur_file.entries[entry_index].block_size != entry_block_size) - { - cerr << "entries not in same order between these files: " - << first_file.filename - << " and " - << cur_file.filename + const preprocessed_inputfile_t &cur_file = preprocessed_inputfiles[file_index]; + if (cur_file.entries[entry_index].product_size != entry_product_size + || cur_file.entries[entry_index].block_size != entry_block_size) { + cerr << "entries not in same order between these files: " << first_file.filename << " and " << cur_file.filename << endl; exit(1); } @@ -294,14 +247,11 @@ void check_all_files_in_same_exact_order( } } -float efficiency_of_subset( - const vector& preprocessed_inputfiles, - const vector& subset) +float efficiency_of_subset(const vector &preprocessed_inputfiles, + const vector &subset) { - if (subset.size() <= 1) { - return 1.0f; - } - const preprocessed_inputfile_t& first_file = preprocessed_inputfiles[subset[0]]; + if (subset.size() <= 1) { return 1.0f; } + const preprocessed_inputfile_t &first_file = preprocessed_inputfiles[subset[0]]; const size_t num_entries = first_file.entries.size(); float efficiency = 1.0f; size_t entry_index = 0; @@ -309,9 +259,7 @@ float efficiency_of_subset( uint16_t product_size = first_file.entries[0].product_size; while (entry_index < num_entries) { ++entry_index; - if (entry_index == num_entries || - first_file.entries[entry_index].product_size != product_size) - { + if (entry_index == num_entries || first_file.entries[entry_index].product_size != product_size) { float efficiency_this_product_size = 0.0f; for (size_t e = first_entry_index_with_this_product_size; e < entry_index; e++) { float efficiency_this_entry = 1.0f; @@ -331,11 +279,10 @@ float efficiency_of_subset( return efficiency; } -void dump_table_for_subset( - const vector& preprocessed_inputfiles, - const vector& subset) +void dump_table_for_subset(const vector &preprocessed_inputfiles, + const vector &subset) { - const preprocessed_inputfile_t& first_file = preprocessed_inputfiles[subset[0]]; + const preprocessed_inputfile_t &first_file = preprocessed_inputfiles[subset[0]]; const size_t num_entries = first_file.entries.size(); size_t entry_index = 0; size_t first_entry_index_with_this_product_size = 0; @@ -343,9 +290,7 @@ void dump_table_for_subset( size_t i = 0; size_triple_t min_product_size(first_file.entries.front().product_size); size_triple_t max_product_size(first_file.entries.back().product_size); - if (!min_product_size.is_cubic() || !max_product_size.is_cubic()) { - abort(); - } + if (!min_product_size.is_cubic() || !max_product_size.is_cubic()) { abort(); } if (only_cubic_sizes) { cerr << "Can't generate tables with --only-cubic-sizes." << endl; abort(); @@ -359,9 +304,7 @@ void dump_table_for_subset( cout << " static const unsigned short data[" << TableSize << "] = {"; while (entry_index < num_entries) { ++entry_index; - if (entry_index == num_entries || - first_file.entries[entry_index].product_size != product_size) - { + if (entry_index == num_entries || first_file.entries[entry_index].product_size != product_size) { float best_efficiency_this_product_size = 0.0f; uint16_t best_block_size_this_product_size = 0; for (size_t e = first_entry_index_with_this_product_size; e < entry_index; e++) { @@ -397,9 +340,8 @@ void dump_table_for_subset( cout << "};" << endl; } -float efficiency_of_partition( - const vector& preprocessed_inputfiles, - const vector>& partition) +float efficiency_of_partition(const vector &preprocessed_inputfiles, + const vector> &partition) { float efficiency = 1.0f; for (auto s = partition.begin(); s != partition.end(); ++s) { @@ -408,21 +350,16 @@ float efficiency_of_partition( return efficiency; } -void make_first_subset(size_t subset_size, vector& out_subset, size_t set_size) +void make_first_subset(size_t subset_size, vector &out_subset, size_t set_size) { assert(subset_size >= 1 && subset_size <= set_size); out_subset.resize(subset_size); - for (size_t i = 0; i < subset_size; i++) { - out_subset[i] = i; - } + for (size_t i = 0; i < subset_size; i++) { out_subset[i] = i; } } -bool is_last_subset(const vector& subset, size_t set_size) -{ - return subset[0] == set_size - subset.size(); -} +bool is_last_subset(const vector &subset, size_t set_size) { return subset[0] == set_size - subset.size(); } -void next_subset(vector& inout_subset, size_t set_size) +void next_subset(vector &inout_subset, size_t set_size) { if (is_last_subset(inout_subset, set_size)) { cerr << "iterating past the last subset" << endl; @@ -436,24 +373,20 @@ void next_subset(vector& inout_subset, size_t set_size) size_t first_index_to_change = inout_subset.size() - i; inout_subset[first_index_to_change]++; size_t p = inout_subset[first_index_to_change]; - for (size_t j = first_index_to_change + 1; j < inout_subset.size(); j++) { - inout_subset[j] = ++p; - } + for (size_t j = first_index_to_change + 1; j < inout_subset.size(); j++) { inout_subset[j] = ++p; } } const size_t number_of_subsets_limit = 100; const size_t always_search_subsets_of_size_at_least = 2; bool is_number_of_subsets_feasible(size_t n, size_t p) -{ - assert(n>0 && p>0 && p<=n); +{ + assert(n > 0 && p > 0 && p <= n); uint64_t numerator = 1, denominator = 1; for (size_t i = 0; i < p; i++) { numerator *= n - i; denominator *= i + 1; - if (numerator > denominator * number_of_subsets_limit) { - return false; - } + if (numerator > denominator * number_of_subsets_limit) { return false; } } return true; } @@ -461,20 +394,17 @@ bool is_number_of_subsets_feasible(size_t n, size_t p) size_t max_feasible_subset_size(size_t n) { assert(n > 0); - const size_t minresult = min(n-1, always_search_subsets_of_size_at_least); + const size_t minresult = min(n - 1, always_search_subsets_of_size_at_least); for (size_t p = 1; p <= n - 1; p++) { - if (!is_number_of_subsets_feasible(n, p+1)) { - return max(p, minresult); - } + if (!is_number_of_subsets_feasible(n, p + 1)) { return max(p, minresult); } } return n - 1; } -void find_subset_with_efficiency_higher_than( - const vector& preprocessed_inputfiles, - float required_efficiency_to_beat, - vector& inout_remainder, - vector& out_subset) +void find_subset_with_efficiency_higher_than(const vector &preprocessed_inputfiles, + float required_efficiency_to_beat, + vector &inout_remainder, + vector &out_subset) { out_subset.resize(0); @@ -486,38 +416,31 @@ void find_subset_with_efficiency_higher_than( while (!inout_remainder.empty()) { vector candidate_indices(inout_remainder.size()); - for (size_t i = 0; i < candidate_indices.size(); i++) { - candidate_indices[i] = i; - } + for (size_t i = 0; i < candidate_indices.size(); i++) { candidate_indices[i] = i; } size_t candidate_indices_subset_size = max_feasible_subset_size(candidate_indices.size()); while (candidate_indices_subset_size >= 1) { vector candidate_indices_subset; - make_first_subset(candidate_indices_subset_size, - candidate_indices_subset, - candidate_indices.size()); + make_first_subset(candidate_indices_subset_size, candidate_indices_subset, candidate_indices.size()); vector best_candidate_indices_subset; float best_efficiency = 0.0f; vector trial_subset = out_subset; trial_subset.resize(out_subset.size() + candidate_indices_subset_size); - while (true) - { + while (true) { for (size_t i = 0; i < candidate_indices_subset_size; i++) { trial_subset[out_subset.size() + i] = inout_remainder[candidate_indices_subset[i]]; } - + float trial_efficiency = efficiency_of_subset(preprocessed_inputfiles, trial_subset); if (trial_efficiency > best_efficiency) { best_efficiency = trial_efficiency; best_candidate_indices_subset = candidate_indices_subset; } - if (is_last_subset(candidate_indices_subset, candidate_indices.size())) { - break; - } + if (is_last_subset(candidate_indices_subset, candidate_indices.size())) { break; } next_subset(candidate_indices_subset, candidate_indices.size()); } - + if (best_efficiency > required_efficiency_to_beat) { for (size_t i = 0; i < best_candidate_indices_subset.size(); i++) { candidate_indices[i] = candidate_indices[best_candidate_indices_subset[i]]; @@ -526,7 +449,7 @@ void find_subset_with_efficiency_higher_than( } candidate_indices_subset_size--; } - + size_t candidate_index = candidate_indices[0]; auto candidate_iterator = inout_remainder.begin() + candidate_index; vector trial_subset = out_subset; @@ -542,39 +465,31 @@ void find_subset_with_efficiency_higher_than( } } -void find_partition_with_efficiency_higher_than( - const vector& preprocessed_inputfiles, - float required_efficiency_to_beat, - vector>& out_partition) +void find_partition_with_efficiency_higher_than(const vector &preprocessed_inputfiles, + float required_efficiency_to_beat, + vector> &out_partition) { out_partition.resize(0); vector remainder; - for (size_t i = 0; i < preprocessed_inputfiles.size(); i++) { - remainder.push_back(i); - } + for (size_t i = 0; i < preprocessed_inputfiles.size(); i++) { remainder.push_back(i); } while (!remainder.empty()) { vector new_subset; find_subset_with_efficiency_higher_than( - preprocessed_inputfiles, - required_efficiency_to_beat, - remainder, - new_subset); + preprocessed_inputfiles, required_efficiency_to_beat, remainder, new_subset); out_partition.push_back(new_subset); } } -void print_partition( - const vector& preprocessed_inputfiles, - const vector>& partition) +void print_partition(const vector &preprocessed_inputfiles, + const vector> &partition) { float efficiency = efficiency_of_partition(preprocessed_inputfiles, partition); - cout << "Partition into " << partition.size() << " subsets for " << efficiency * 100.0f << "% efficiency" << endl; + cout << "Partition into " << partition.size() << " subsets for " << efficiency * 100.0f << "% efficiency" << endl; for (auto subset = partition.begin(); subset != partition.end(); ++subset) { - cout << " Subset " << (subset - partition.begin()) - << ", efficiency " << efficiency_of_subset(preprocessed_inputfiles, *subset) * 100.0f << "%:" - << endl; + cout << " Subset " << (subset - partition.begin()) << ", efficiency " + << efficiency_of_subset(preprocessed_inputfiles, *subset) * 100.0f << "%:" << endl; for (auto file = subset->begin(); file != subset->end(); ++file) { cout << " " << preprocessed_inputfiles[*file].filename << endl; } @@ -588,15 +503,19 @@ void print_partition( struct action_t { - virtual const char* invokation_name() const { abort(); return nullptr; } - virtual void run(const vector&) const { abort(); } + virtual const char *invokation_name() const + { + abort(); + return nullptr; + } + virtual void run(const vector &) const { abort(); } virtual ~action_t() {} }; struct partition_action_t : action_t { - virtual const char* invokation_name() const override { return "partition"; } - virtual void run(const vector& input_filenames) const override + virtual const char *invokation_name() const override { return "partition"; } + virtual void run(const vector &input_filenames) const override { vector preprocessed_inputfiles; @@ -608,17 +527,17 @@ struct partition_action_t : action_t for (auto it = input_filenames.begin(); it != input_filenames.end(); ++it) { inputfile_t inputfile(*it); switch (inputfile.type) { - case inputfile_t::type_t::all_pot_sizes: - preprocessed_inputfiles.emplace_back(inputfile); - break; - case inputfile_t::type_t::default_sizes: - cerr << "The " << invokation_name() << " action only uses measurements for all pot sizes, and " - << "has no use for " << *it << " which contains measurements for default sizes." << endl; - exit(1); - break; - default: - cerr << "Unrecognized input file: " << *it << endl; - exit(1); + case inputfile_t::type_t::all_pot_sizes: + preprocessed_inputfiles.emplace_back(inputfile); + break; + case inputfile_t::type_t::default_sizes: + cerr << "The " << invokation_name() << " action only uses measurements for all pot sizes, and " + << "has no use for " << *it << " which contains measurements for default sizes." << endl; + exit(1); + break; + default: + cerr << "Unrecognized input file: " << *it << endl; + exit(1); } } @@ -627,47 +546,37 @@ struct partition_action_t : action_t float required_efficiency_to_beat = 0.0f; vector>> partitions; cerr << "searching for partitions...\r" << flush; - while (true) - { + while (true) { vector> partition; - find_partition_with_efficiency_higher_than( - preprocessed_inputfiles, - required_efficiency_to_beat, - partition); + find_partition_with_efficiency_higher_than(preprocessed_inputfiles, required_efficiency_to_beat, partition); float actual_efficiency = efficiency_of_partition(preprocessed_inputfiles, partition); - cerr << "partition " << preprocessed_inputfiles.size() << " files into " << partition.size() - << " subsets for " << 100.0f * actual_efficiency - << " % efficiency" + cerr << "partition " << preprocessed_inputfiles.size() << " files into " << partition.size() << " subsets for " + << 100.0f * actual_efficiency << " % efficiency" << " \r" << flush; partitions.push_back(partition); - if (partition.size() == preprocessed_inputfiles.size() || actual_efficiency == 1.0f) { - break; - } + if (partition.size() == preprocessed_inputfiles.size() || actual_efficiency == 1.0f) { break; } required_efficiency_to_beat = actual_efficiency; } cerr << " " << endl; while (true) { bool repeat = false; for (size_t i = 0; i < partitions.size() - 1; i++) { - if (partitions[i].size() >= partitions[i+1].size()) { + if (partitions[i].size() >= partitions[i + 1].size()) { partitions.erase(partitions.begin() + i); repeat = true; break; } } - if (!repeat) { - break; - } - } - for (auto it = partitions.begin(); it != partitions.end(); ++it) { - print_partition(preprocessed_inputfiles, *it); + if (!repeat) { break; } } + for (auto it = partitions.begin(); it != partitions.end(); ++it) { print_partition(preprocessed_inputfiles, *it); } } }; struct evaluate_defaults_action_t : action_t { - struct results_entry_t { + struct results_entry_t + { uint16_t product_size; size_triple_t default_block_size; uint16_t best_pot_block_size; @@ -675,21 +584,19 @@ struct evaluate_defaults_action_t : action_t float best_pot_gflops; float default_efficiency; }; - friend ostream& operator<<(ostream& s, const results_entry_t& entry) + friend ostream &operator<<(ostream &s, const results_entry_t &entry) { - return s - << "Product size " << size_triple_t(entry.product_size) - << ": default block size " << entry.default_block_size - << " -> " << entry.default_gflops - << " GFlop/s = " << entry.default_efficiency * 100.0f << " %" - << " of best POT block size " << size_triple_t(entry.best_pot_block_size) - << " -> " << entry.best_pot_gflops - << " GFlop/s" << dec; + return s << "Product size " << size_triple_t(entry.product_size) << ": default block size " + << entry.default_block_size << " -> " << entry.default_gflops + << " GFlop/s = " << entry.default_efficiency * 100.0f << " %" + << " of best POT block size " << size_triple_t(entry.best_pot_block_size) << " -> " + << entry.best_pot_gflops << " GFlop/s" << dec; } - static bool lower_efficiency(const results_entry_t& e1, const results_entry_t& e2) { + static bool lower_efficiency(const results_entry_t &e1, const results_entry_t &e2) + { return e1.default_efficiency < e2.default_efficiency; } - virtual const char* invokation_name() const override { return "evaluate-defaults"; } + virtual const char *invokation_name() const override { return "evaluate-defaults"; } void show_usage_and_exit() const { cerr << "usage: " << invokation_name() << " default-sizes-data all-pot-sizes-data" << endl; @@ -697,11 +604,9 @@ struct evaluate_defaults_action_t : action_t << "performance measured over all POT sizes." << endl; exit(1); } - virtual void run(const vector& input_filenames) const override + virtual void run(const vector &input_filenames) const override { - if (input_filenames.size() != 2) { - show_usage_and_exit(); - } + if (input_filenames.size() != 2) { show_usage_and_exit(); } inputfile_t inputfile_default_sizes(input_filenames[0]); inputfile_t inputfile_all_pot_sizes(input_filenames[1]); if (inputfile_default_sizes.type != inputfile_t::type_t::default_sizes) { @@ -714,31 +619,23 @@ struct evaluate_defaults_action_t : action_t } vector results; vector cubic_results; - + uint16_t product_size = 0; auto it_all_pot_sizes = inputfile_all_pot_sizes.entries.begin(); for (auto it_default_sizes = inputfile_default_sizes.entries.begin(); - it_default_sizes != inputfile_default_sizes.entries.end(); - ++it_default_sizes) - { - if (it_default_sizes->product_size == product_size) { - continue; - } + it_default_sizes != inputfile_default_sizes.entries.end(); + ++it_default_sizes) { + if (it_default_sizes->product_size == product_size) { continue; } product_size = it_default_sizes->product_size; - while (it_all_pot_sizes != inputfile_all_pot_sizes.entries.end() && - it_all_pot_sizes->product_size != product_size) - { + while ( + it_all_pot_sizes != inputfile_all_pot_sizes.entries.end() && it_all_pot_sizes->product_size != product_size) { ++it_all_pot_sizes; } - if (it_all_pot_sizes == inputfile_all_pot_sizes.entries.end()) { - break; - } + if (it_all_pot_sizes == inputfile_all_pot_sizes.entries.end()) { break; } uint16_t best_pot_block_size = 0; float best_pot_gflops = 0; - for (auto it = it_all_pot_sizes; - it != inputfile_all_pot_sizes.entries.end() && it->product_size == product_size; - ++it) - { + for (auto it = it_all_pot_sizes; it != inputfile_all_pot_sizes.entries.end() && it->product_size == product_size; + ++it) { if (it->gflops > best_pot_gflops) { best_pot_gflops = it->gflops; best_pot_block_size = it->pot_block_size; @@ -754,60 +651,51 @@ struct evaluate_defaults_action_t : action_t results.push_back(entry); size_triple_t t(product_size); - if (t.k == t.m && t.m == t.n) { - cubic_results.push_back(entry); - } + if (t.k == t.m && t.m == t.n) { cubic_results.push_back(entry); } } cout << "All results:" << endl; - for (auto it = results.begin(); it != results.end(); ++it) { - cout << *it << endl; - } + for (auto it = results.begin(); it != results.end(); ++it) { cout << *it << endl; } cout << endl; sort(results.begin(), results.end(), lower_efficiency); - + const size_t n = min(20, results.size()); cout << n << " worst results:" << endl; - for (size_t i = 0; i < n; i++) { - cout << results[i] << endl; - } + for (size_t i = 0; i < n; i++) { cout << results[i] << endl; } cout << endl; cout << "cubic results:" << endl; - for (auto it = cubic_results.begin(); it != cubic_results.end(); ++it) { - cout << *it << endl; - } + for (auto it = cubic_results.begin(); it != cubic_results.end(); ++it) { cout << *it << endl; } cout << endl; sort(cubic_results.begin(), cubic_results.end(), lower_efficiency); - + cout.precision(2); - vector a = {0.5f, 0.20f, 0.10f, 0.05f, 0.02f, 0.01f}; + vector a = { 0.5f, 0.20f, 0.10f, 0.05f, 0.02f, 0.01f }; for (auto it = a.begin(); it != a.end(); ++it) { size_t n = min(results.size() - 1, size_t(*it * results.size())); cout << (100.0f * n / (results.size() - 1)) - << " % of product sizes have default efficiency <= " - << 100.0f * results[n].default_efficiency << " %" << endl; + << " % of product sizes have default efficiency <= " << 100.0f * results[n].default_efficiency << " %" + << endl; } cout.precision(default_precision); } }; -void show_usage_and_exit(int argc, char* argv[], - const vector>& available_actions) +void show_usage_and_exit(int argc, char *argv[], const vector> &available_actions) { cerr << "usage: " << argv[0] << " [options...] " << endl; cerr << "available actions:" << endl; for (auto it = available_actions.begin(); it != available_actions.end(); ++it) { cerr << " " << (*it)->invokation_name() << endl; - } + } cerr << "the input files should each contain an output of benchmark-blocking-sizes" << endl; exit(1); } -int main(int argc, char* argv[]) +int main(int argc, char *argv[]) { cout.precision(default_precision); cerr.precision(default_precision); @@ -818,11 +706,9 @@ int main(int argc, char* argv[]) vector input_filenames; - action_t* action = nullptr; + action_t *action = nullptr; - if (argc < 2) { - show_usage_and_exit(argc, argv, available_actions); - } + if (argc < 2) { show_usage_and_exit(argc, argv, available_actions); } for (int i = 1; i < argc; i++) { bool arg_handled = false; // Step 1. Try to match action invokation names. @@ -838,9 +724,7 @@ int main(int argc, char* argv[]) } } } - if (arg_handled) { - continue; - } + if (arg_handled) { continue; } // Step 2. Try to match option names. if (argv[i][0] == '-') { if (!strcmp(argv[i], "--only-cubic-sizes")) { @@ -856,9 +740,7 @@ int main(int argc, char* argv[]) show_usage_and_exit(argc, argv, available_actions); } } - if (arg_handled) { - continue; - } + if (arg_handled) { continue; } // Step 3. Default to interpreting args as input filenames. input_filenames.emplace_back(argv[i]); } @@ -868,9 +750,7 @@ int main(int argc, char* argv[]) show_usage_and_exit(argc, argv, available_actions); } - if (!action) { - show_usage_and_exit(argc, argv, available_actions); - } + if (!action) { show_usage_and_exit(argc, argv, available_actions); } action->run(input_filenames); } diff --git a/filmulator-gui/core/nlmeans/eigen/bench/basicbenchmark.cpp b/filmulator-gui/core/nlmeans/eigen/bench/basicbenchmark.cpp index a26ea853..b432cb13 100644 --- a/filmulator-gui/core/nlmeans/eigen/bench/basicbenchmark.cpp +++ b/filmulator-gui/core/nlmeans/eigen/bench/basicbenchmark.cpp @@ -1,34 +1,34 @@ -#include -#include "BenchUtil.h" #include "basicbenchmark.h" +#include "BenchUtil.h" +#include int main(int argc, char *argv[]) { DISABLE_SSE_EXCEPTIONS(); - // this is the list of matrix type and size we want to bench: - // ((suffix) (matrix size) (number of iterations)) - #define MODES ((3d)(3)(4000000)) ((4d)(4)(1000000)) ((Xd)(4)(1000000)) ((Xd)(20)(10000)) -// #define MODES ((Xd)(20)(10000)) +// this is the list of matrix type and size we want to bench: +// ((suffix) (matrix size) (number of iterations)) +#define MODES ((3d)(3)(4000000))((4d)(4)(1000000))((Xd)(4)(1000000))((Xd)(20)(10000)) + // #define MODES ((Xd)(20)(10000)) - #define _GENERATE_HEADER(R,ARG,EL) << BOOST_PP_STRINGIZE(BOOST_PP_SEQ_HEAD(EL)) << "-" \ - << BOOST_PP_STRINGIZE(BOOST_PP_SEQ_ELEM(1,EL)) << "x" \ - << BOOST_PP_STRINGIZE(BOOST_PP_SEQ_ELEM(1,EL)) << " / " +#define _GENERATE_HEADER(R, ARG, EL) \ + << BOOST_PP_STRINGIZE(BOOST_PP_SEQ_HEAD(EL)) \ + << "-" \ + << BOOST_PP_STRINGIZE(BOOST_PP_SEQ_ELEM(1,EL)) << "x" << BOOST_PP_STRINGIZE(BOOST_PP_SEQ_ELEM(1,EL)) << " / " - std::cout BOOST_PP_SEQ_FOR_EACH(_GENERATE_HEADER, ~, MODES ) << endl; + std::cout BOOST_PP_SEQ_FOR_EACH(_GENERATE_HEADER, ~, MODES) << endl; const int tries = 10; - #define _RUN_BENCH(R,ARG,EL) \ - std::cout << ARG( \ - BOOST_PP_CAT(Matrix, BOOST_PP_SEQ_HEAD(EL)) (\ - BOOST_PP_SEQ_ELEM(1,EL),BOOST_PP_SEQ_ELEM(1,EL)), BOOST_PP_SEQ_ELEM(2,EL), tries) \ - << " "; +#define _RUN_BENCH(R, ARG, EL) \ + std::cout << ARG(BOOST_PP_CAT(Matrix, BOOST_PP_SEQ_HEAD(EL))(BOOST_PP_SEQ_ELEM(1, EL), BOOST_PP_SEQ_ELEM(1, EL)), \ + BOOST_PP_SEQ_ELEM(2, EL), \ + tries) << " "; - BOOST_PP_SEQ_FOR_EACH(_RUN_BENCH, benchBasic, MODES ); + BOOST_PP_SEQ_FOR_EACH(_RUN_BENCH, benchBasic, MODES); std::cout << endl; - BOOST_PP_SEQ_FOR_EACH(_RUN_BENCH, benchBasic, MODES ); + BOOST_PP_SEQ_FOR_EACH(_RUN_BENCH, benchBasic, MODES); std::cout << endl; return 0; diff --git a/filmulator-gui/core/nlmeans/eigen/bench/basicbenchmark.h b/filmulator-gui/core/nlmeans/eigen/bench/basicbenchmark.h index 3fdc3573..3e0bb3be 100644 --- a/filmulator-gui/core/nlmeans/eigen/bench/basicbenchmark.h +++ b/filmulator-gui/core/nlmeans/eigen/bench/basicbenchmark.h @@ -2,32 +2,25 @@ #ifndef EIGEN_BENCH_BASICBENCH_H #define EIGEN_BENCH_BASICBENCH_H -enum {LazyEval, EarlyEval, OmpEval}; +enum { LazyEval, EarlyEval, OmpEval }; template -void benchBasic_loop(const MatrixType& I, MatrixType& m, int iterations) __attribute__((noinline)); +void benchBasic_loop(const MatrixType &I, MatrixType &m, int iterations) __attribute__((noinline)); -template -void benchBasic_loop(const MatrixType& I, MatrixType& m, int iterations) +template void benchBasic_loop(const MatrixType &I, MatrixType &m, int iterations) { - for(int a = 0; a < iterations; a++) - { - if (Mode==LazyEval) - { + for (int a = 0; a < iterations; a++) { + if (Mode == LazyEval) { asm("#begin_bench_loop LazyEval"); - if (MatrixType::SizeAtCompileTime!=Eigen::Dynamic) asm("#fixedsize"); + if (MatrixType::SizeAtCompileTime != Eigen::Dynamic) asm("#fixedsize"); m = (I + 0.00005 * (m + m.lazy() * m)).eval(); - } - else if (Mode==OmpEval) - { + } else if (Mode == OmpEval) { asm("#begin_bench_loop OmpEval"); - if (MatrixType::SizeAtCompileTime!=Eigen::Dynamic) asm("#fixedsize"); + if (MatrixType::SizeAtCompileTime != Eigen::Dynamic) asm("#fixedsize"); m = (I + 0.00005 * (m + m.lazy() * m)).evalOMP(); - } - else - { + } else { asm("#begin_bench_loop EarlyEval"); - if (MatrixType::SizeAtCompileTime!=Eigen::Dynamic) asm("#fixedsize"); + if (MatrixType::SizeAtCompileTime != Eigen::Dynamic) asm("#fixedsize"); m = I + 0.00005 * (m + m * m); } asm("#end_bench_loop"); @@ -35,22 +28,20 @@ void benchBasic_loop(const MatrixType& I, MatrixType& m, int iterations) } template -double benchBasic(const MatrixType& mat, int size, int tries) __attribute__((noinline)); +double benchBasic(const MatrixType &mat, int size, int tries) __attribute__((noinline)); -template -double benchBasic(const MatrixType& mat, int iterations, int tries) +template double benchBasic(const MatrixType &mat, int iterations, int tries) { const int rows = mat.rows(); const int cols = mat.cols(); - MatrixType I(rows,cols); - MatrixType m(rows,cols); + MatrixType I(rows, cols); + MatrixType m(rows, cols); initMatrix_identity(I); Eigen::BenchTimer timer; - for(uint t=0; t(I, m, iterations); @@ -60,4 +51,4 @@ double benchBasic(const MatrixType& mat, int iterations, int tries) return timer.value(); }; -#endif // EIGEN_BENCH_BASICBENCH_H +#endif// EIGEN_BENCH_BASICBENCH_H diff --git a/filmulator-gui/core/nlmeans/eigen/bench/benchBlasGemm.cpp b/filmulator-gui/core/nlmeans/eigen/bench/benchBlasGemm.cpp index cb086a55..2e43fc26 100644 --- a/filmulator-gui/core/nlmeans/eigen/bench/benchBlasGemm.cpp +++ b/filmulator-gui/core/nlmeans/eigen/bench/benchBlasGemm.cpp @@ -8,8 +8,8 @@ #include -#include #include "BenchTimer.h" +#include // include the BLAS headers extern "C" { @@ -26,58 +26,49 @@ typedef double Scalar; #endif -typedef Eigen::Matrix MyMatrix; -void bench_eigengemm(MyMatrix& mc, const MyMatrix& ma, const MyMatrix& mb, int nbloops); +typedef Eigen::Matrix MyMatrix; +void bench_eigengemm(MyMatrix &mc, const MyMatrix &ma, const MyMatrix &mb, int nbloops); void check_product(int M, int N, int K); void check_product(void); int main(int argc, char *argv[]) { - // disable SSE exceptions - #ifdef __GNUC__ +// disable SSE exceptions +#ifdef __GNUC__ { int aux; asm( - "stmxcsr %[aux] \n\t" - "orl $32832, %[aux] \n\t" - "ldmxcsr %[aux] \n\t" - : : [aux] "m" (aux)); + "stmxcsr %[aux] \n\t" + "orl $32832, %[aux] \n\t" + "ldmxcsr %[aux] \n\t" + : + : [aux] "m"(aux)); } - #endif +#endif - int nbtries=1, nbloops=1, M, N, K; + int nbtries = 1, nbloops = 1, M, N, K; - if (argc==2) - { - if (std::string(argv[1])=="check") + if (argc == 2) { + if (std::string(argv[1]) == "check") check_product(); else M = N = K = atoi(argv[1]); - } - else if ((argc==3) && (std::string(argv[1])=="auto")) - { + } else if ((argc == 3) && (std::string(argv[1]) == "auto")) { M = N = K = atoi(argv[2]); - nbloops = 1000000000/(M*M*M); - if (nbloops<1) - nbloops = 1; + nbloops = 1000000000 / (M * M * M); + if (nbloops < 1) nbloops = 1; nbtries = 6; - } - else if (argc==4) - { + } else if (argc == 4) { M = N = K = atoi(argv[1]); nbloops = atoi(argv[2]); nbtries = atoi(argv[3]); - } - else if (argc==6) - { + } else if (argc == 6) { M = atoi(argv[1]); N = atoi(argv[2]); K = atoi(argv[3]); nbloops = atoi(argv[4]); nbtries = atoi(argv[5]); - } - else - { + } else { std::cout << "Usage: " << argv[0] << " size \n"; std::cout << "Usage: " << argv[0] << " auto size\n"; std::cout << "Usage: " << argv[0] << " size nbloops nbtries\n"; @@ -95,14 +86,13 @@ int main(int argc, char *argv[]) double nbmad = double(M) * double(N) * double(K) * double(nbloops); - if (!(std::string(argv[1])=="auto")) - std::cout << M << " x " << N << " x " << K << "\n"; + if (!(std::string(argv[1]) == "auto")) std::cout << M << " x " << N << " x " << K << "\n"; Scalar alpha, beta; - MyMatrix ma(M,K), mb(K,N), mc(M,N); - ma = MyMatrix::Random(M,K); - mb = MyMatrix::Random(K,N); - mc = MyMatrix::Random(M,N); + MyMatrix ma(M, K), mb(K, N), mc(M, N); + ma = MyMatrix::Random(M, K); + mb = MyMatrix::Random(K, N); + mc = MyMatrix::Random(M, N); Eigen::BenchTimer timer; @@ -112,108 +102,103 @@ int main(int argc, char *argv[]) // bench cblas // ROWS_A, COLS_B, COLS_A, 1.0, A, COLS_A, B, COLS_B, 0.0, C, COLS_B); - if (!(std::string(argv[1])=="auto")) - { + if (!(std::string(argv[1]) == "auto")) { timer.reset(); - for (uint k=0 ; k(1,64); - N = internal::random(1,768); - K = internal::random(1,768); + for (uint i = 0; i < 1000; ++i) { + M = internal::random(1, 64); + N = internal::random(1, 768); + K = internal::random(1, 768); M = (0 + M) * 1; std::cout << M << " x " << N << " x " << K << "\n"; check_product(M, N, K); } } - diff --git a/filmulator-gui/core/nlmeans/eigen/bench/benchCholesky.cpp b/filmulator-gui/core/nlmeans/eigen/bench/benchCholesky.cpp index 9a8e7cf6..f0faf438 100644 --- a/filmulator-gui/core/nlmeans/eigen/bench/benchCholesky.cpp +++ b/filmulator-gui/core/nlmeans/eigen/bench/benchCholesky.cpp @@ -10,8 +10,8 @@ #include -#include #include +#include #include using namespace Eigen; @@ -25,118 +25,103 @@ using namespace Eigen; typedef float Scalar; -template -__attribute__ ((noinline)) void benchLLT(const MatrixType& m) +template __attribute__((noinline)) void benchLLT(const MatrixType &m) { int rows = m.rows(); int cols = m.cols(); double cost = 0; - for (int j=0; j SquareMatrixType; - MatrixType a = MatrixType::Random(rows,cols); - SquareMatrixType covMat = a * a.adjoint(); + MatrixType a = MatrixType::Random(rows, cols); + SquareMatrixType covMat = a * a.adjoint(); BenchTimer timerNoSqrt, timerSqrt; Scalar acc = 0; - int r = internal::random(0,covMat.rows()-1); - int c = internal::random(0,covMat.cols()-1); - for (int t=0; t(0, covMat.rows() - 1); + int c = internal::random(0, covMat.cols() - 1); + for (int t = 0; t < TRIES; ++t) { timerNoSqrt.start(); - for (int k=0; k cholnosqrt(covMat); - acc += cholnosqrt.matrixL().coeff(r,c); + acc += cholnosqrt.matrixL().coeff(r, c); } timerNoSqrt.stop(); } - for (int t=0; t chol(covMat); - acc += chol.matrixL().coeff(r,c); + acc += chol.matrixL().coeff(r, c); } timerSqrt.stop(); } - if (MatrixType::RowsAtCompileTime==Dynamic) + if (MatrixType::RowsAtCompileTime == Dynamic) std::cout << "dyn "; else std::cout << "fixed "; - std::cout << covMat.rows() << " \t" - << (timerNoSqrt.best()) / repeats << "s " - << "(" << 1e-9 * cost*repeats/timerNoSqrt.best() << " GFLOPS)\t" - << (timerSqrt.best()) / repeats << "s " - << "(" << 1e-9 * cost*repeats/timerSqrt.best() << " GFLOPS)\n"; + std::cout << covMat.rows() << " \t" << (timerNoSqrt.best()) / repeats << "s " + << "(" << 1e-9 * cost * repeats / timerNoSqrt.best() << " GFLOPS)\t" << (timerSqrt.best()) / repeats << "s " + << "(" << 1e-9 * cost * repeats / timerSqrt.best() << " GFLOPS)\n"; - #ifdef BENCH_GSL - if (MatrixType::RowsAtCompileTime==Dynamic) - { +#ifdef BENCH_GSL + if (MatrixType::RowsAtCompileTime == Dynamic) { timerSqrt.reset(); - gsl_matrix* gslCovMat = gsl_matrix_alloc(covMat.rows(),covMat.cols()); - gsl_matrix* gslCopy = gsl_matrix_alloc(covMat.rows(),covMat.cols()); + gsl_matrix *gslCovMat = gsl_matrix_alloc(covMat.rows(), covMat.cols()); + gsl_matrix *gslCopy = gsl_matrix_alloc(covMat.rows(), covMat.cols()); eiToGsl(covMat, &gslCovMat); - for (int t=0; t0; ++i) - benchLLT(Matrix(dynsizes[i],dynsizes[i])); - - benchLLT(Matrix()); - benchLLT(Matrix()); - benchLLT(Matrix()); - benchLLT(Matrix()); - benchLLT(Matrix()); - benchLLT(Matrix()); - benchLLT(Matrix()); - benchLLT(Matrix()); - benchLLT(Matrix()); + for (int i = 0; dynsizes[i] > 0; ++i) benchLLT(Matrix(dynsizes[i], dynsizes[i])); + + benchLLT(Matrix()); + benchLLT(Matrix()); + benchLLT(Matrix()); + benchLLT(Matrix()); + benchLLT(Matrix()); + benchLLT(Matrix()); + benchLLT(Matrix()); + benchLLT(Matrix()); + benchLLT(Matrix()); return 0; } - diff --git a/filmulator-gui/core/nlmeans/eigen/bench/benchEigenSolver.cpp b/filmulator-gui/core/nlmeans/eigen/bench/benchEigenSolver.cpp index dd78c7e0..344bab6a 100644 --- a/filmulator-gui/core/nlmeans/eigen/bench/benchEigenSolver.cpp +++ b/filmulator-gui/core/nlmeans/eigen/bench/benchEigenSolver.cpp @@ -30,35 +30,32 @@ using namespace Eigen; typedef SCALAR Scalar; -template -__attribute__ ((noinline)) void benchEigenSolver(const MatrixType& m) +template __attribute__((noinline)) void benchEigenSolver(const MatrixType &m) { int rows = m.rows(); int cols = m.cols(); - int stdRepeats = std::max(1,int((REPEAT*1000)/(rows*rows*sqrt(rows)))); + int stdRepeats = std::max(1, int((REPEAT * 1000) / (rows * rows * sqrt(rows)))); int saRepeats = stdRepeats * 4; typedef typename MatrixType::Scalar Scalar; typedef Matrix SquareMatrixType; - MatrixType a = MatrixType::Random(rows,cols); - SquareMatrixType covMat = a * a.adjoint(); + MatrixType a = MatrixType::Random(rows, cols); + SquareMatrixType covMat = a * a.adjoint(); BenchTimer timerSa, timerStd; Scalar acc = 0; - int r = internal::random(0,covMat.rows()-1); - int c = internal::random(0,covMat.cols()-1); + int r = internal::random(0, covMat.rows() - 1); + int c = internal::random(0, covMat.cols() - 1); { SelfAdjointEigenSolver ei(covMat); - for (int t=0; t ei(covMat); - for (int t=0; t gmmCovMat(covMat.rows(),covMat.cols()); - gmm::dense_matrix eigvect(covMat.rows(),covMat.cols()); + gmm::dense_matrix gmmCovMat(covMat.rows(), covMat.cols()); + gmm::dense_matrix eigvect(covMat.rows(), covMat.cols()); std::vector eigval(covMat.rows()); eiToGmm(covMat, gmmCovMat); - for (int t=0; t0; ++i) - benchEigenSolver(Matrix(dynsizes[i],dynsizes[i])); - - benchEigenSolver(Matrix()); - benchEigenSolver(Matrix()); - benchEigenSolver(Matrix()); - benchEigenSolver(Matrix()); - benchEigenSolver(Matrix()); - benchEigenSolver(Matrix()); - benchEigenSolver(Matrix()); + for (uint i = 0; dynsizes[i] > 0; ++i) benchEigenSolver(Matrix(dynsizes[i], dynsizes[i])); + + benchEigenSolver(Matrix()); + benchEigenSolver(Matrix()); + benchEigenSolver(Matrix()); + benchEigenSolver(Matrix()); + benchEigenSolver(Matrix()); + benchEigenSolver(Matrix()); + benchEigenSolver(Matrix()); return 0; } - diff --git a/filmulator-gui/core/nlmeans/eigen/bench/benchFFT.cpp b/filmulator-gui/core/nlmeans/eigen/bench/benchFFT.cpp index 3eb1a1ac..0cb0f882 100644 --- a/filmulator-gui/core/nlmeans/eigen/bench/benchFFT.cpp +++ b/filmulator-gui/core/nlmeans/eigen/bench/benchFFT.cpp @@ -9,10 +9,10 @@ #include +#include #include #include #include -#include #include @@ -20,12 +20,11 @@ using namespace Eigen; using namespace std; -template -string nameof(); +template string nameof(); -template <> string nameof() {return "float";} -template <> string nameof() {return "double";} -template <> string nameof() {return "long double";} +template<> string nameof() { return "float"; } +template<> string nameof() { return "double"; } +template<> string nameof() { return "long double"; } #ifndef TYPE #define TYPE float @@ -40,76 +39,73 @@ template <> string nameof() {return "long double";} using namespace Eigen; -template -void bench(int nfft,bool fwd,bool unscaled=false, bool halfspec=false) +template void bench(int nfft, bool fwd, bool unscaled = false, bool halfspec = false) { - typedef typename NumTraits::Real Scalar; - typedef typename std::complex Complex; - int nits = NDATA/nfft; - vector inbuf(nfft); - vector outbuf(nfft); - FFT< Scalar > fft; - - if (unscaled) { - fft.SetFlag(fft.Unscaled); - cout << "unscaled "; - } - if (halfspec) { - fft.SetFlag(fft.HalfSpectrum); - cout << "halfspec "; - } - - - std::fill(inbuf.begin(),inbuf.end(),0); - fft.fwd( outbuf , inbuf); - - BenchTimer timer; - timer.reset(); - for (int k=0;k<8;++k) { - timer.start(); - if (fwd) - for(int i = 0; i < nits; i++) - fft.fwd( outbuf , inbuf); - else - for(int i = 0; i < nits; i++) - fft.inv(inbuf,outbuf); - timer.stop(); - } - - cout << nameof() << " "; - double mflops = 5.*nfft*log2((double)nfft) / (1e6 * timer.value() / (double)nits ); - if ( NumTraits::IsComplex ) { - cout << "complex"; - }else{ - cout << "real "; - mflops /= 2; - } - - + typedef typename NumTraits::Real Scalar; + typedef typename std::complex Complex; + int nits = NDATA / nfft; + vector inbuf(nfft); + vector outbuf(nfft); + FFT fft; + + if (unscaled) { + fft.SetFlag(fft.Unscaled); + cout << "unscaled "; + } + if (halfspec) { + fft.SetFlag(fft.HalfSpectrum); + cout << "halfspec "; + } + + + std::fill(inbuf.begin(), inbuf.end(), 0); + fft.fwd(outbuf, inbuf); + + BenchTimer timer; + timer.reset(); + for (int k = 0; k < 8; ++k) { + timer.start(); if (fwd) - cout << " fwd"; + for (int i = 0; i < nits; i++) fft.fwd(outbuf, inbuf); else - cout << " inv"; - - cout << " NFFT=" << nfft << " " << (double(1e-6*nfft*nits)/timer.value()) << " MS/s " << mflops << "MFLOPS\n"; + for (int i = 0; i < nits; i++) fft.inv(inbuf, outbuf); + timer.stop(); + } + + cout << nameof() << " "; + double mflops = 5. * nfft * log2((double)nfft) / (1e6 * timer.value() / (double)nits); + if (NumTraits::IsComplex) { + cout << "complex"; + } else { + cout << "real "; + mflops /= 2; + } + + + if (fwd) + cout << " fwd"; + else + cout << " inv"; + + cout << " NFFT=" << nfft << " " << (double(1e-6 * nfft * nits) / timer.value()) << " MS/s " << mflops << "MFLOPS\n"; } -int main(int argc,char ** argv) +int main(int argc, char **argv) { - bench >(NFFT,true); - bench >(NFFT,false); - bench(NFFT,true); - bench(NFFT,false); - bench(NFFT,false,true); - bench(NFFT,false,true,true); - - bench >(NFFT,true); - bench >(NFFT,false); - bench(NFFT,true); - bench(NFFT,false); - bench >(NFFT,true); - bench >(NFFT,false); - bench(NFFT,true); - bench(NFFT,false); - return 0; + bench>(NFFT, true); + bench>(NFFT, false); + bench(NFFT, true); + bench(NFFT, false); + bench(NFFT, false, true); + bench(NFFT, false, true, true); + + bench>(NFFT, true); + bench>(NFFT, false); + bench(NFFT, true); + bench(NFFT, false); + bench>(NFFT, true); + bench>(NFFT, false); + bench(NFFT, true); + bench(NFFT, false); + return 0; } diff --git a/filmulator-gui/core/nlmeans/eigen/bench/benchGeometry.cpp b/filmulator-gui/core/nlmeans/eigen/bench/benchGeometry.cpp index 6e16c033..97433c9e 100644 --- a/filmulator-gui/core/nlmeans/eigen/bench/benchGeometry.cpp +++ b/filmulator-gui/core/nlmeans/eigen/bench/benchGeometry.cpp @@ -1,8 +1,8 @@ -#include -#include #include #include #include +#include +#include using namespace Eigen; using namespace std; @@ -11,124 +11,116 @@ using namespace std; #define REPEAT 1000000 #endif -enum func_opt -{ - TV, - TMATV, - TMATVMAT, +enum func_opt { + TV, + TMATV, + TMATVMAT, }; -template -struct func; +template struct func; -template -struct func +template struct func { - static EIGEN_DONT_INLINE res run( arg1& a1, arg2& a2 ) - { - asm (""); - return a1 * a2; - } + static EIGEN_DONT_INLINE res run(arg1 &a1, arg2 &a2) + { + asm(""); + return a1 * a2; + } }; -template -struct func +template struct func { - static EIGEN_DONT_INLINE res run( arg1& a1, arg2& a2 ) - { - asm (""); - return a1.matrix() * a2; - } + static EIGEN_DONT_INLINE res run(arg1 &a1, arg2 &a2) + { + asm(""); + return a1.matrix() * a2; + } }; -template -struct func +template struct func { - static EIGEN_DONT_INLINE res run( arg1& a1, arg2& a2 ) - { - asm (""); - return res(a1.matrix() * a2.matrix()); - } + static EIGEN_DONT_INLINE res run(arg1 &a1, arg2 &a2) + { + asm(""); + return res(a1.matrix() * a2.matrix()); + } }; -template -struct test_transform +template struct test_transform { - static void run() - { - arg1 a1; - a1.setIdentity(); - arg2 a2; - a2.setIdentity(); - - BenchTimer timer; - timer.reset(); - for (int k=0; k<10; ++k) - { - timer.start(); - for (int k=0; k Trans;\ - typedef Matrix Vec;\ - typedef func Func;\ - test_transform< Func, Trans, Vec >::run();\ - } - -#define run_trans( op, scalar, mode, option ) \ - std::cout << #scalar << "\t " << #mode << "\t " << #option << " "; \ - {\ - typedef Transform Trans;\ - typedef func Func;\ - test_transform< Func, Trans, Trans >::run();\ - } - -int main(int argc, char* argv[]) +#define run_vec(op, scalar, mode, option, vsize) \ + std::cout << #scalar << "\t " << #mode << "\t " << #option << " " << #vsize " "; \ + { \ + typedef Transform Trans; \ + typedef Matrix Vec; \ + typedef func Func; \ + test_transform::run(); \ + } + +#define run_trans(op, scalar, mode, option) \ + std::cout << #scalar << "\t " << #mode << "\t " << #option << " "; \ + { \ + typedef Transform Trans; \ + typedef func Func; \ + test_transform::run(); \ + } + +int main(int argc, char *argv[]) { - cout << "vec = trans * vec" << endl; - run_vec(TV, float, Isometry, AutoAlign, 3); - run_vec(TV, float, Isometry, DontAlign, 3); - run_vec(TV, float, Isometry, AutoAlign, 4); - run_vec(TV, float, Isometry, DontAlign, 4); - run_vec(TV, float, Projective, AutoAlign, 4); - run_vec(TV, float, Projective, DontAlign, 4); - run_vec(TV, double, Isometry, AutoAlign, 3); - run_vec(TV, double, Isometry, DontAlign, 3); - run_vec(TV, double, Isometry, AutoAlign, 4); - run_vec(TV, double, Isometry, DontAlign, 4); - run_vec(TV, double, Projective, AutoAlign, 4); - run_vec(TV, double, Projective, DontAlign, 4); - - cout << "vec = trans.matrix() * vec" << endl; - run_vec(TMATV, float, Isometry, AutoAlign, 4); - run_vec(TMATV, float, Isometry, DontAlign, 4); - run_vec(TMATV, double, Isometry, AutoAlign, 4); - run_vec(TMATV, double, Isometry, DontAlign, 4); - - cout << "trans = trans1 * trans" << endl; - run_trans(TV, float, Isometry, AutoAlign); - run_trans(TV, float, Isometry, DontAlign); - run_trans(TV, double, Isometry, AutoAlign); - run_trans(TV, double, Isometry, DontAlign); - run_trans(TV, float, Projective, AutoAlign); - run_trans(TV, float, Projective, DontAlign); - run_trans(TV, double, Projective, AutoAlign); - run_trans(TV, double, Projective, DontAlign); - - cout << "trans = trans1.matrix() * trans.matrix()" << endl; - run_trans(TMATVMAT, float, Isometry, AutoAlign); - run_trans(TMATVMAT, float, Isometry, DontAlign); - run_trans(TMATVMAT, double, Isometry, AutoAlign); - run_trans(TMATVMAT, double, Isometry, DontAlign); + cout << "vec = trans * vec" << endl; + run_vec(TV, float, Isometry, AutoAlign, 3); + run_vec(TV, float, Isometry, DontAlign, 3); + run_vec(TV, float, Isometry, AutoAlign, 4); + run_vec(TV, float, Isometry, DontAlign, 4); + run_vec(TV, float, Projective, AutoAlign, 4); + run_vec(TV, float, Projective, DontAlign, 4); + run_vec(TV, double, Isometry, AutoAlign, 3); + run_vec(TV, double, Isometry, DontAlign, 3); + run_vec(TV, double, Isometry, AutoAlign, 4); + run_vec(TV, double, Isometry, DontAlign, 4); + run_vec(TV, double, Projective, AutoAlign, 4); + run_vec(TV, double, Projective, DontAlign, 4); + + cout << "vec = trans.matrix() * vec" << endl; + run_vec(TMATV, float, Isometry, AutoAlign, 4); + run_vec(TMATV, float, Isometry, DontAlign, 4); + run_vec(TMATV, double, Isometry, AutoAlign, 4); + run_vec(TMATV, double, Isometry, DontAlign, 4); + + cout << "trans = trans1 * trans" << endl; + run_trans(TV, float, Isometry, AutoAlign); + run_trans(TV, float, Isometry, DontAlign); + run_trans(TV, double, Isometry, AutoAlign); + run_trans(TV, double, Isometry, DontAlign); + run_trans(TV, float, Projective, AutoAlign); + run_trans(TV, float, Projective, DontAlign); + run_trans(TV, double, Projective, AutoAlign); + run_trans(TV, double, Projective, DontAlign); + + cout << "trans = trans1.matrix() * trans.matrix()" << endl; + run_trans(TMATVMAT, float, Isometry, AutoAlign); + run_trans(TMATVMAT, float, Isometry, DontAlign); + run_trans(TMATVMAT, double, Isometry, AutoAlign); + run_trans(TMATVMAT, double, Isometry, DontAlign); } - diff --git a/filmulator-gui/core/nlmeans/eigen/bench/benchVecAdd.cpp b/filmulator-gui/core/nlmeans/eigen/bench/benchVecAdd.cpp index ce8e1e91..c0b10be7 100644 --- a/filmulator-gui/core/nlmeans/eigen/bench/benchVecAdd.cpp +++ b/filmulator-gui/core/nlmeans/eigen/bench/benchVecAdd.cpp @@ -1,7 +1,7 @@ -#include #include #include +#include using namespace Eigen; #ifndef SIZE @@ -14,122 +14,120 @@ using namespace Eigen; typedef float Scalar; -__attribute__ ((noinline)) void benchVec(Scalar* a, Scalar* b, Scalar* c, int size); -__attribute__ ((noinline)) void benchVec(MatrixXf& a, MatrixXf& b, MatrixXf& c); -__attribute__ ((noinline)) void benchVec(VectorXf& a, VectorXf& b, VectorXf& c); +__attribute__((noinline)) void benchVec(Scalar *a, Scalar *b, Scalar *c, int size); +__attribute__((noinline)) void benchVec(MatrixXf &a, MatrixXf &b, MatrixXf &c); +__attribute__((noinline)) void benchVec(VectorXf &a, VectorXf &b, VectorXf &c); -int main(int argc, char* argv[]) +int main(int argc, char *argv[]) { - int size = SIZE * 8; - int size2 = size * size; - Scalar* a = internal::aligned_new(size2); - Scalar* b = internal::aligned_new(size2+4)+1; - Scalar* c = internal::aligned_new(size2); - - for (int i=0; i2 ; --innersize) - { - if (size2%innersize==0) - { - int outersize = size2/innersize; - MatrixXf ma = Map(a, innersize, outersize ); - MatrixXf mb = Map(b, innersize, outersize ); - MatrixXf mc = Map(c, innersize, outersize ); - timer.reset(); - for (int k=0; k<3; ++k) - { - timer.start(); - benchVec(ma, mb, mc); - timer.stop(); - } - std::cout << innersize << " x " << outersize << " " << timer.value() << "s " << (double(size2*REPEAT)/timer.value())/(1024.*1024.*1024.) << " GFlops\n"; - } - } - - VectorXf va = Map(a, size2); - VectorXf vb = Map(b, size2); - VectorXf vc = Map(c, size2); - timer.reset(); - for (int k=0; k<3; ++k) - { + int size = SIZE * 8; + int size2 = size * size; + Scalar *a = internal::aligned_new(size2); + Scalar *b = internal::aligned_new(size2 + 4) + 1; + Scalar *c = internal::aligned_new(size2); + + for (int i = 0; i < size; ++i) { a[i] = b[i] = c[i] = 0; } + + BenchTimer timer; + + timer.reset(); + for (int k = 0; k < 10; ++k) { + timer.start(); + benchVec(a, b, c, size2); + timer.stop(); + } + std::cout << timer.value() << "s " << (double(size2 * REPEAT) / timer.value()) / (1024. * 1024. * 1024.) + << " GFlops\n"; + return 0; + for (int innersize = size; innersize > 2; --innersize) { + if (size2 % innersize == 0) { + int outersize = size2 / innersize; + MatrixXf ma = Map(a, innersize, outersize); + MatrixXf mb = Map(b, innersize, outersize); + MatrixXf mc = Map(c, innersize, outersize); + timer.reset(); + for (int k = 0; k < 3; ++k) { timer.start(); - benchVec(va, vb, vc); + benchVec(ma, mb, mc); timer.stop(); + } + std::cout << innersize << " x " << outersize << " " << timer.value() << "s " + << (double(size2 * REPEAT) / timer.value()) / (1024. * 1024. * 1024.) << " GFlops\n"; } - std::cout << timer.value() << "s " << (double(size2*REPEAT)/timer.value())/(1024.*1024.*1024.) << " GFlops\n"; + } - return 0; + VectorXf va = Map(a, size2); + VectorXf vb = Map(b, size2); + VectorXf vc = Map(c, size2); + timer.reset(); + for (int k = 0; k < 3; ++k) { + timer.start(); + benchVec(va, vb, vc); + timer.stop(); + } + std::cout << timer.value() << "s " << (double(size2 * REPEAT) / timer.value()) / (1024. * 1024. * 1024.) + << " GFlops\n"; + + return 0; } -void benchVec(MatrixXf& a, MatrixXf& b, MatrixXf& c) +void benchVec(MatrixXf &a, MatrixXf &b, MatrixXf &c) { - for (int k=0; k::type PacketScalar; - const int PacketSize = internal::packet_traits::size; - PacketScalar a0, a1, a2, a3, b0, b1, b2, b3; - for (int k=0; k::type PacketScalar; + const int PacketSize = internal::packet_traits::size; + PacketScalar a0, a1, a2, a3, b0, b1, b2, b3; + for (int k = 0; k < REPEAT; ++k) + for (int i = 0; i < size; i += PacketSize * 8) { + // a0 = internal::pload(&a[i]); + // b0 = internal::pload(&b[i]); + // a1 = internal::pload(&a[i+1*PacketSize]); + // b1 = internal::pload(&b[i+1*PacketSize]); + // a2 = internal::pload(&a[i+2*PacketSize]); + // b2 = internal::pload(&b[i+2*PacketSize]); + // a3 = internal::pload(&a[i+3*PacketSize]); + // b3 = internal::pload(&b[i+3*PacketSize]); + // internal::pstore(&a[i], internal::padd(a0, b0)); + // a0 = internal::pload(&a[i+4*PacketSize]); + // b0 = internal::pload(&b[i+4*PacketSize]); + // + // internal::pstore(&a[i+1*PacketSize], internal::padd(a1, b1)); + // a1 = internal::pload(&a[i+5*PacketSize]); + // b1 = internal::pload(&b[i+5*PacketSize]); + // + // internal::pstore(&a[i+2*PacketSize], internal::padd(a2, b2)); + // a2 = internal::pload(&a[i+6*PacketSize]); + // b2 = internal::pload(&b[i+6*PacketSize]); + // + // internal::pstore(&a[i+3*PacketSize], internal::padd(a3, b3)); + // a3 = internal::pload(&a[i+7*PacketSize]); + // b3 = internal::pload(&b[i+7*PacketSize]); + // + // internal::pstore(&a[i+4*PacketSize], internal::padd(a0, b0)); + // internal::pstore(&a[i+5*PacketSize], internal::padd(a1, b1)); + // internal::pstore(&a[i+6*PacketSize], internal::padd(a2, b2)); + // internal::pstore(&a[i+7*PacketSize], internal::padd(a3, b3)); + + internal::pstore(&a[i + 2 * PacketSize], + internal::padd(internal::ploadu(&a[i + 2 * PacketSize]), internal::ploadu(&b[i + 2 * PacketSize]))); + internal::pstore(&a[i + 3 * PacketSize], + internal::padd(internal::ploadu(&a[i + 3 * PacketSize]), internal::ploadu(&b[i + 3 * PacketSize]))); + internal::pstore(&a[i + 4 * PacketSize], + internal::padd(internal::ploadu(&a[i + 4 * PacketSize]), internal::ploadu(&b[i + 4 * PacketSize]))); + internal::pstore(&a[i + 5 * PacketSize], + internal::padd(internal::ploadu(&a[i + 5 * PacketSize]), internal::ploadu(&b[i + 5 * PacketSize]))); + internal::pstore(&a[i + 6 * PacketSize], + internal::padd(internal::ploadu(&a[i + 6 * PacketSize]), internal::ploadu(&b[i + 6 * PacketSize]))); + internal::pstore(&a[i + 7 * PacketSize], + internal::padd(internal::ploadu(&a[i + 7 * PacketSize]), internal::ploadu(&b[i + 7 * PacketSize]))); + } } diff --git a/filmulator-gui/core/nlmeans/eigen/bench/bench_gemm.cpp b/filmulator-gui/core/nlmeans/eigen/bench/bench_gemm.cpp index 8528c558..fcc18c4b 100644 --- a/filmulator-gui/core/nlmeans/eigen/bench/bench_gemm.cpp +++ b/filmulator-gui/core/nlmeans/eigen/bench/bench_gemm.cpp @@ -3,16 +3,16 @@ // icpc bench_gemm.cpp -I .. -O3 -DNDEBUG -lrt -openmp && OMP_NUM_THREADS=2 ./a.out // Compilation options: -// +// // -DSCALAR=std::complex // -DSCALARA=double or -DSCALARB=double // -DHAVE_BLAS // -DDECOUPLED // -#include #include #include +#include using namespace std; using namespace Eigen; @@ -32,15 +32,15 @@ using namespace Eigen; typedef SCALAR Scalar; typedef NumTraits::Real RealScalar; -typedef Matrix A; -typedef Matrix B; -typedef Matrix C; -typedef Matrix M; +typedef Matrix A; +typedef Matrix B; +typedef Matrix C; +typedef Matrix M; #ifdef HAVE_BLAS extern "C" { - #include +#include } static float fone = 1; @@ -52,61 +52,112 @@ static std::complex cfzero = 0; static std::complex cdone = 1; static std::complex cdzero = 0; static char notrans = 'N'; -static char trans = 'T'; +static char trans = 'T'; static char nonunit = 'N'; static char lower = 'L'; static char right = 'R'; static int intone = 1; -void blas_gemm(const MatrixXf& a, const MatrixXf& b, MatrixXf& c) +void blas_gemm(const MatrixXf &a, const MatrixXf &b, MatrixXf &c) { - int M = c.rows(); int N = c.cols(); int K = a.cols(); - int lda = a.rows(); int ldb = b.rows(); int ldc = c.rows(); - - sgemm_(¬rans,¬rans,&M,&N,&K,&fone, - const_cast(a.data()),&lda, - const_cast(b.data()),&ldb,&fone, - c.data(),&ldc); + int M = c.rows(); + int N = c.cols(); + int K = a.cols(); + int lda = a.rows(); + int ldb = b.rows(); + int ldc = c.rows(); + + sgemm_(¬rans, + ¬rans, + &M, + &N, + &K, + &fone, + const_cast(a.data()), + &lda, + const_cast(b.data()), + &ldb, + &fone, + c.data(), + &ldc); } -EIGEN_DONT_INLINE void blas_gemm(const MatrixXd& a, const MatrixXd& b, MatrixXd& c) +EIGEN_DONT_INLINE void blas_gemm(const MatrixXd &a, const MatrixXd &b, MatrixXd &c) { - int M = c.rows(); int N = c.cols(); int K = a.cols(); - int lda = a.rows(); int ldb = b.rows(); int ldc = c.rows(); - - dgemm_(¬rans,¬rans,&M,&N,&K,&done, - const_cast(a.data()),&lda, - const_cast(b.data()),&ldb,&done, - c.data(),&ldc); + int M = c.rows(); + int N = c.cols(); + int K = a.cols(); + int lda = a.rows(); + int ldb = b.rows(); + int ldc = c.rows(); + + dgemm_(¬rans, + ¬rans, + &M, + &N, + &K, + &done, + const_cast(a.data()), + &lda, + const_cast(b.data()), + &ldb, + &done, + c.data(), + &ldc); } -void blas_gemm(const MatrixXcf& a, const MatrixXcf& b, MatrixXcf& c) +void blas_gemm(const MatrixXcf &a, const MatrixXcf &b, MatrixXcf &c) { - int M = c.rows(); int N = c.cols(); int K = a.cols(); - int lda = a.rows(); int ldb = b.rows(); int ldc = c.rows(); - - cgemm_(¬rans,¬rans,&M,&N,&K,(float*)&cfone, - const_cast((const float*)a.data()),&lda, - const_cast((const float*)b.data()),&ldb,(float*)&cfone, - (float*)c.data(),&ldc); + int M = c.rows(); + int N = c.cols(); + int K = a.cols(); + int lda = a.rows(); + int ldb = b.rows(); + int ldc = c.rows(); + + cgemm_(¬rans, + ¬rans, + &M, + &N, + &K, + (float *)&cfone, + const_cast((const float *)a.data()), + &lda, + const_cast((const float *)b.data()), + &ldb, + (float *)&cfone, + (float *)c.data(), + &ldc); } -void blas_gemm(const MatrixXcd& a, const MatrixXcd& b, MatrixXcd& c) +void blas_gemm(const MatrixXcd &a, const MatrixXcd &b, MatrixXcd &c) { - int M = c.rows(); int N = c.cols(); int K = a.cols(); - int lda = a.rows(); int ldb = b.rows(); int ldc = c.rows(); - - zgemm_(¬rans,¬rans,&M,&N,&K,(double*)&cdone, - const_cast((const double*)a.data()),&lda, - const_cast((const double*)b.data()),&ldb,(double*)&cdone, - (double*)c.data(),&ldc); + int M = c.rows(); + int N = c.cols(); + int K = a.cols(); + int lda = a.rows(); + int ldb = b.rows(); + int ldc = c.rows(); + + zgemm_(¬rans, + ¬rans, + &M, + &N, + &K, + (double *)&cdone, + const_cast((const double *)a.data()), + &lda, + const_cast((const double *)b.data()), + &ldb, + (double *)&cdone, + (double *)c.data(), + &ldc); } - #endif -void matlab_cplx_cplx(const M& ar, const M& ai, const M& br, const M& bi, M& cr, M& ci) +void matlab_cplx_cplx(const M &ar, const M &ai, const M &br, const M &bi, M &cr, M &ci) { cr.noalias() += ar * br; cr.noalias() -= ai * bi; @@ -114,227 +165,250 @@ void matlab_cplx_cplx(const M& ar, const M& ai, const M& br, const M& bi, M& cr, ci.noalias() += ai * br; } -void matlab_real_cplx(const M& a, const M& br, const M& bi, M& cr, M& ci) +void matlab_real_cplx(const M &a, const M &br, const M &bi, M &cr, M &ci) { cr.noalias() += a * br; ci.noalias() += a * bi; } -void matlab_cplx_real(const M& ar, const M& ai, const M& b, M& cr, M& ci) +void matlab_cplx_real(const M &ar, const M &ai, const M &b, M &cr, M &ci) { cr.noalias() += ar * b; ci.noalias() += ai * b; } -template -EIGEN_DONT_INLINE void gemm(const A& a, const B& b, C& c) +template EIGEN_DONT_INLINE void gemm(const A &a, const B &b, C &c) { - c.noalias() += a * b; + c.noalias() += a * b; } -int main(int argc, char ** argv) +int main(int argc, char **argv) { std::ptrdiff_t l1 = internal::queryL1CacheSize(); std::ptrdiff_t l2 = internal::queryTopLevelCacheSize(); - std::cout << "L1 cache size = " << (l1>0 ? l1/1024 : -1) << " KB\n"; - std::cout << "L2/L3 cache size = " << (l2>0 ? l2/1024 : -1) << " KB\n"; - typedef internal::gebp_traits Traits; + std::cout << "L1 cache size = " << (l1 > 0 ? l1 / 1024 : -1) << " KB\n"; + std::cout << "L2/L3 cache size = " << (l2 > 0 ? l2 / 1024 : -1) << " KB\n"; + typedef internal::gebp_traits Traits; std::cout << "Register blocking = " << Traits::mr << " x " << Traits::nr << "\n"; - int rep = 1; // number of repetitions per try - int tries = 2; // number of tries, we keep the best + int rep = 1;// number of repetitions per try + int tries = 2;// number of tries, we keep the best int s = 2048; int m = s; int n = s; int p = s; - int cache_size1=-1, cache_size2=l2, cache_size3 = 0; + int cache_size1 = -1, cache_size2 = l2, cache_size3 = 0; bool need_help = false; - for (int i=1; i -c -t -p \n"; std::cout << " : size\n"; std::cout << " : rows columns depth\n"; return 1; } -#if EIGEN_VERSION_AT_LEAST(3,2,90) - if(cache_size1>0) - setCpuCacheSizes(cache_size1,cache_size2,cache_size3); +#if EIGEN_VERSION_AT_LEAST(3, 2, 90) + if (cache_size1 > 0) setCpuCacheSizes(cache_size1, cache_size2, cache_size3); #endif - - A a(m,p); a.setRandom(); - B b(p,n); b.setRandom(); - C c(m,n); c.setOnes(); + + A a(m, p); + a.setRandom(); + B b(p, n); + b.setRandom(); + C c(m, n); + c.setOnes(); C rc = c; std::cout << "Matrix sizes = " << m << "x" << p << " * " << p << "x" << n << "\n"; std::ptrdiff_t mc(m), nc(n), kc(p); - internal::computeProductBlockingSizes(kc, mc, nc); + internal::computeProductBlockingSizes(kc, mc, nc); std::cout << "blocking size (mc x kc) = " << mc << " x " << kc << "\n"; C r = c; - // check the parallel product is correct - #if defined EIGEN_HAS_OPENMP +// check the parallel product is correct +#if defined EIGEN_HAS_OPENMP Eigen::initParallel(); int procs = omp_get_max_threads(); - if(procs>1) - { - #ifdef HAVE_BLAS - blas_gemm(a,b,r); - #else + if (procs > 1) { +#ifdef HAVE_BLAS + blas_gemm(a, b, r); +#else omp_set_num_threads(1); r.noalias() += a * b; omp_set_num_threads(procs); - #endif +#endif c.noalias() += a * b; - if(!r.isApprox(c)) std::cerr << "Warning, your parallel product is crap!\n\n"; + if (!r.isApprox(c)) std::cerr << "Warning, your parallel product is crap!\n\n"; } - #elif defined HAVE_BLAS - blas_gemm(a,b,r); - c.noalias() += a * b; - if(!r.isApprox(c)) { - std::cout << r - c << "\n"; +#elif defined HAVE_BLAS + blas_gemm(a, b, r); + c.noalias() += a * b; + if (!r.isApprox(c)) { + std::cout << r - c << "\n"; + std::cerr << "Warning, your product is crap!\n\n"; + } +#else + if (1. * m * n * p < 2000. * 2000 * 2000) { + gemm(a, b, c); + r.noalias() += a.cast().lazyProduct(b.cast()); + if (!r.isApprox(c)) { + std::cout << r - c << "\n"; std::cerr << "Warning, your product is crap!\n\n"; } - #else - if(1.*m*n*p<2000.*2000*2000) - { - gemm(a,b,c); - r.noalias() += a.cast() .lazyProduct( b.cast() ); - if(!r.isApprox(c)) { - std::cout << r - c << "\n"; - std::cerr << "Warning, your product is crap!\n\n"; - } - } - #endif + } +#endif - #ifdef HAVE_BLAS +#ifdef HAVE_BLAS BenchTimer tblas; c = rc; - BENCH(tblas, tries, rep, blas_gemm(a,b,c)); - std::cout << "blas cpu " << tblas.best(CPU_TIMER)/rep << "s \t" << (double(m)*n*p*rep*2/tblas.best(CPU_TIMER))*1e-9 << " GFLOPS \t(" << tblas.total(CPU_TIMER) << "s)\n"; - std::cout << "blas real " << tblas.best(REAL_TIMER)/rep << "s \t" << (double(m)*n*p*rep*2/tblas.best(REAL_TIMER))*1e-9 << " GFLOPS \t(" << tblas.total(REAL_TIMER) << "s)\n"; - #endif + BENCH(tblas, tries, rep, blas_gemm(a, b, c)); + std::cout << "blas cpu " << tblas.best(CPU_TIMER) / rep << "s \t" + << (double(m) * n * p * rep * 2 / tblas.best(CPU_TIMER)) * 1e-9 << " GFLOPS \t(" << tblas.total(CPU_TIMER) + << "s)\n"; + std::cout << "blas real " << tblas.best(REAL_TIMER) / rep << "s \t" + << (double(m) * n * p * rep * 2 / tblas.best(REAL_TIMER)) * 1e-9 << " GFLOPS \t(" << tblas.total(REAL_TIMER) + << "s)\n"; +#endif BenchTimer tmt; c = rc; - BENCH(tmt, tries, rep, gemm(a,b,c)); - std::cout << "eigen cpu " << tmt.best(CPU_TIMER)/rep << "s \t" << (double(m)*n*p*rep*2/tmt.best(CPU_TIMER))*1e-9 << " GFLOPS \t(" << tmt.total(CPU_TIMER) << "s)\n"; - std::cout << "eigen real " << tmt.best(REAL_TIMER)/rep << "s \t" << (double(m)*n*p*rep*2/tmt.best(REAL_TIMER))*1e-9 << " GFLOPS \t(" << tmt.total(REAL_TIMER) << "s)\n"; - - #ifdef EIGEN_HAS_OPENMP - if(procs>1) - { + BENCH(tmt, tries, rep, gemm(a, b, c)); + std::cout << "eigen cpu " << tmt.best(CPU_TIMER) / rep << "s \t" + << (double(m) * n * p * rep * 2 / tmt.best(CPU_TIMER)) * 1e-9 << " GFLOPS \t(" << tmt.total(CPU_TIMER) + << "s)\n"; + std::cout << "eigen real " << tmt.best(REAL_TIMER) / rep << "s \t" + << (double(m) * n * p * rep * 2 / tmt.best(REAL_TIMER)) * 1e-9 << " GFLOPS \t(" << tmt.total(REAL_TIMER) + << "s)\n"; + +#ifdef EIGEN_HAS_OPENMP + if (procs > 1) { BenchTimer tmono; omp_set_num_threads(1); Eigen::setNbThreads(1); c = rc; - BENCH(tmono, tries, rep, gemm(a,b,c)); - std::cout << "eigen mono cpu " << tmono.best(CPU_TIMER)/rep << "s \t" << (double(m)*n*p*rep*2/tmono.best(CPU_TIMER))*1e-9 << " GFLOPS \t(" << tmono.total(CPU_TIMER) << "s)\n"; - std::cout << "eigen mono real " << tmono.best(REAL_TIMER)/rep << "s \t" << (double(m)*n*p*rep*2/tmono.best(REAL_TIMER))*1e-9 << " GFLOPS \t(" << tmono.total(REAL_TIMER) << "s)\n"; - std::cout << "mt speed up x" << tmono.best(CPU_TIMER) / tmt.best(REAL_TIMER) << " => " << (100.0*tmono.best(CPU_TIMER) / tmt.best(REAL_TIMER))/procs << "%\n"; + BENCH(tmono, tries, rep, gemm(a, b, c)); + std::cout << "eigen mono cpu " << tmono.best(CPU_TIMER) / rep << "s \t" + << (double(m) * n * p * rep * 2 / tmono.best(CPU_TIMER)) * 1e-9 << " GFLOPS \t(" << tmono.total(CPU_TIMER) + << "s)\n"; + std::cout << "eigen mono real " << tmono.best(REAL_TIMER) / rep << "s \t" + << (double(m) * n * p * rep * 2 / tmono.best(REAL_TIMER)) * 1e-9 << " GFLOPS \t(" + << tmono.total(REAL_TIMER) << "s)\n"; + std::cout << "mt speed up x" << tmono.best(CPU_TIMER) / tmt.best(REAL_TIMER) << " => " + << (100.0 * tmono.best(CPU_TIMER) / tmt.best(REAL_TIMER)) / procs << "%\n"; } - #endif - - if(1.*m*n*p<30*30*30) - { - BenchTimer tmt; - c = rc; - BENCH(tmt, tries, rep, c.noalias()+=a.lazyProduct(b)); - std::cout << "lazy cpu " << tmt.best(CPU_TIMER)/rep << "s \t" << (double(m)*n*p*rep*2/tmt.best(CPU_TIMER))*1e-9 << " GFLOPS \t(" << tmt.total(CPU_TIMER) << "s)\n"; - std::cout << "lazy real " << tmt.best(REAL_TIMER)/rep << "s \t" << (double(m)*n*p*rep*2/tmt.best(REAL_TIMER))*1e-9 << " GFLOPS \t(" << tmt.total(REAL_TIMER) << "s)\n"; +#endif + + if (1. * m * n * p < 30 * 30 * 30) { + BenchTimer tmt; + c = rc; + BENCH(tmt, tries, rep, c.noalias() += a.lazyProduct(b)); + std::cout << "lazy cpu " << tmt.best(CPU_TIMER) / rep << "s \t" + << (double(m) * n * p * rep * 2 / tmt.best(CPU_TIMER)) * 1e-9 << " GFLOPS \t(" << tmt.total(CPU_TIMER) + << "s)\n"; + std::cout << "lazy real " << tmt.best(REAL_TIMER) / rep << "s \t" + << (double(m) * n * p * rep * 2 / tmt.best(REAL_TIMER)) * 1e-9 << " GFLOPS \t(" << tmt.total(REAL_TIMER) + << "s)\n"; } - - #ifdef DECOUPLED - if((NumTraits::IsComplex) && (NumTraits::IsComplex)) - { - M ar(m,p); ar.setRandom(); - M ai(m,p); ai.setRandom(); - M br(p,n); br.setRandom(); - M bi(p,n); bi.setRandom(); - M cr(m,n); cr.setRandom(); - M ci(m,n); ci.setRandom(); - + +#ifdef DECOUPLED + if ((NumTraits::IsComplex) && (NumTraits::IsComplex)) { + M ar(m, p); + ar.setRandom(); + M ai(m, p); + ai.setRandom(); + M br(p, n); + br.setRandom(); + M bi(p, n); + bi.setRandom(); + M cr(m, n); + cr.setRandom(); + M ci(m, n); + ci.setRandom(); + BenchTimer t; - BENCH(t, tries, rep, matlab_cplx_cplx(ar,ai,br,bi,cr,ci)); - std::cout << "\"matlab\" cpu " << t.best(CPU_TIMER)/rep << "s \t" << (double(m)*n*p*rep*2/t.best(CPU_TIMER))*1e-9 << " GFLOPS \t(" << t.total(CPU_TIMER) << "s)\n"; - std::cout << "\"matlab\" real " << t.best(REAL_TIMER)/rep << "s \t" << (double(m)*n*p*rep*2/t.best(REAL_TIMER))*1e-9 << " GFLOPS \t(" << t.total(REAL_TIMER) << "s)\n"; + BENCH(t, tries, rep, matlab_cplx_cplx(ar, ai, br, bi, cr, ci)); + std::cout << "\"matlab\" cpu " << t.best(CPU_TIMER) / rep << "s \t" + << (double(m) * n * p * rep * 2 / t.best(CPU_TIMER)) * 1e-9 << " GFLOPS \t(" << t.total(CPU_TIMER) + << "s)\n"; + std::cout << "\"matlab\" real " << t.best(REAL_TIMER) / rep << "s \t" + << (double(m) * n * p * rep * 2 / t.best(REAL_TIMER)) * 1e-9 << " GFLOPS \t(" << t.total(REAL_TIMER) + << "s)\n"; } - if((!NumTraits::IsComplex) && (NumTraits::IsComplex)) - { - M a(m,p); a.setRandom(); - M br(p,n); br.setRandom(); - M bi(p,n); bi.setRandom(); - M cr(m,n); cr.setRandom(); - M ci(m,n); ci.setRandom(); - + if ((!NumTraits::IsComplex) && (NumTraits::IsComplex)) { + M a(m, p); + a.setRandom(); + M br(p, n); + br.setRandom(); + M bi(p, n); + bi.setRandom(); + M cr(m, n); + cr.setRandom(); + M ci(m, n); + ci.setRandom(); + BenchTimer t; - BENCH(t, tries, rep, matlab_real_cplx(a,br,bi,cr,ci)); - std::cout << "\"matlab\" cpu " << t.best(CPU_TIMER)/rep << "s \t" << (double(m)*n*p*rep*2/t.best(CPU_TIMER))*1e-9 << " GFLOPS \t(" << t.total(CPU_TIMER) << "s)\n"; - std::cout << "\"matlab\" real " << t.best(REAL_TIMER)/rep << "s \t" << (double(m)*n*p*rep*2/t.best(REAL_TIMER))*1e-9 << " GFLOPS \t(" << t.total(REAL_TIMER) << "s)\n"; + BENCH(t, tries, rep, matlab_real_cplx(a, br, bi, cr, ci)); + std::cout << "\"matlab\" cpu " << t.best(CPU_TIMER) / rep << "s \t" + << (double(m) * n * p * rep * 2 / t.best(CPU_TIMER)) * 1e-9 << " GFLOPS \t(" << t.total(CPU_TIMER) + << "s)\n"; + std::cout << "\"matlab\" real " << t.best(REAL_TIMER) / rep << "s \t" + << (double(m) * n * p * rep * 2 / t.best(REAL_TIMER)) * 1e-9 << " GFLOPS \t(" << t.total(REAL_TIMER) + << "s)\n"; } - if((NumTraits::IsComplex) && (!NumTraits::IsComplex)) - { - M ar(m,p); ar.setRandom(); - M ai(m,p); ai.setRandom(); - M b(p,n); b.setRandom(); - M cr(m,n); cr.setRandom(); - M ci(m,n); ci.setRandom(); - + if ((NumTraits::IsComplex) && (!NumTraits::IsComplex)) { + M ar(m, p); + ar.setRandom(); + M ai(m, p); + ai.setRandom(); + M b(p, n); + b.setRandom(); + M cr(m, n); + cr.setRandom(); + M ci(m, n); + ci.setRandom(); + BenchTimer t; - BENCH(t, tries, rep, matlab_cplx_real(ar,ai,b,cr,ci)); - std::cout << "\"matlab\" cpu " << t.best(CPU_TIMER)/rep << "s \t" << (double(m)*n*p*rep*2/t.best(CPU_TIMER))*1e-9 << " GFLOPS \t(" << t.total(CPU_TIMER) << "s)\n"; - std::cout << "\"matlab\" real " << t.best(REAL_TIMER)/rep << "s \t" << (double(m)*n*p*rep*2/t.best(REAL_TIMER))*1e-9 << " GFLOPS \t(" << t.total(REAL_TIMER) << "s)\n"; + BENCH(t, tries, rep, matlab_cplx_real(ar, ai, b, cr, ci)); + std::cout << "\"matlab\" cpu " << t.best(CPU_TIMER) / rep << "s \t" + << (double(m) * n * p * rep * 2 / t.best(CPU_TIMER)) * 1e-9 << " GFLOPS \t(" << t.total(CPU_TIMER) + << "s)\n"; + std::cout << "\"matlab\" real " << t.best(REAL_TIMER) / rep << "s \t" + << (double(m) * n * p * rep * 2 / t.best(REAL_TIMER)) * 1e-9 << " GFLOPS \t(" << t.total(REAL_TIMER) + << "s)\n"; } - #endif +#endif return 0; } - diff --git a/filmulator-gui/core/nlmeans/eigen/bench/bench_norm.cpp b/filmulator-gui/core/nlmeans/eigen/bench/bench_norm.cpp index 129afcfb..9fd602c6 100644 --- a/filmulator-gui/core/nlmeans/eigen/bench/bench_norm.cpp +++ b/filmulator-gui/core/nlmeans/eigen/bench/bench_norm.cpp @@ -1,83 +1,53 @@ -#include -#include -#include #include "BenchTimer.h" +#include +#include +#include using namespace Eigen; using namespace std; -template -EIGEN_DONT_INLINE typename T::Scalar sqsumNorm(T& v) -{ - return v.norm(); -} +template EIGEN_DONT_INLINE typename T::Scalar sqsumNorm(T &v) { return v.norm(); } -template -EIGEN_DONT_INLINE typename T::Scalar stableNorm(T& v) -{ - return v.stableNorm(); -} +template EIGEN_DONT_INLINE typename T::Scalar stableNorm(T &v) { return v.stableNorm(); } -template -EIGEN_DONT_INLINE typename T::Scalar hypotNorm(T& v) -{ - return v.hypotNorm(); -} +template EIGEN_DONT_INLINE typename T::Scalar hypotNorm(T &v) { return v.hypotNorm(); } -template -EIGEN_DONT_INLINE typename T::Scalar blueNorm(T& v) -{ - return v.blueNorm(); -} +template EIGEN_DONT_INLINE typename T::Scalar blueNorm(T &v) { return v.blueNorm(); } -template -EIGEN_DONT_INLINE typename T::Scalar lapackNorm(T& v) +template EIGEN_DONT_INLINE typename T::Scalar lapackNorm(T &v) { typedef typename T::Scalar Scalar; int n = v.size(); Scalar scale = 0; Scalar ssq = 1; - for (int i=0;i= ax) - { - ssq += numext::abs2(ax/scale); - } - else - { - ssq = Scalar(1) + ssq * numext::abs2(scale/ax); + if (scale >= ax) { + ssq += numext::abs2(ax / scale); + } else { + ssq = Scalar(1) + ssq * numext::abs2(scale / ax); scale = ax; } } return scale * std::sqrt(ssq); } -template -EIGEN_DONT_INLINE typename T::Scalar twopassNorm(T& v) +template EIGEN_DONT_INLINE typename T::Scalar twopassNorm(T &v) { typedef typename T::Scalar Scalar; Scalar s = v.array().abs().maxCoeff(); - return s*(v/s).norm(); + return s * (v / s).norm(); } -template -EIGEN_DONT_INLINE typename T::Scalar bl2passNorm(T& v) -{ - return v.stableNorm(); -} +template EIGEN_DONT_INLINE typename T::Scalar bl2passNorm(T &v) { return v.stableNorm(); } -template -EIGEN_DONT_INLINE typename T::Scalar divacNorm(T& v) +template EIGEN_DONT_INLINE typename T::Scalar divacNorm(T &v) { - int n =v.size() / 2; - for (int i=0;i0) - { - for (int i=0;i 0) { + for (int i = 0; i < n; ++i) v(i) = v(2 * i) + v(2 * i + 1); + n = n / 2; } return std::sqrt(v(0)); } @@ -85,61 +55,61 @@ EIGEN_DONT_INLINE typename T::Scalar divacNorm(T& v) namespace Eigen { namespace internal { #ifdef EIGEN_VECTORIZE -Packet4f plt(const Packet4f& a, Packet4f& b) { return _mm_cmplt_ps(a,b); } -Packet2d plt(const Packet2d& a, Packet2d& b) { return _mm_cmplt_pd(a,b); } + Packet4f plt(const Packet4f &a, Packet4f &b) { return _mm_cmplt_ps(a, b); } + Packet2d plt(const Packet2d &a, Packet2d &b) { return _mm_cmplt_pd(a, b); } -Packet4f pandnot(const Packet4f& a, Packet4f& b) { return _mm_andnot_ps(a,b); } -Packet2d pandnot(const Packet2d& a, Packet2d& b) { return _mm_andnot_pd(a,b); } + Packet4f pandnot(const Packet4f &a, Packet4f &b) { return _mm_andnot_ps(a, b); } + Packet2d pandnot(const Packet2d &a, Packet2d &b) { return _mm_andnot_pd(a, b); } #endif -} -} +}// namespace internal +}// namespace Eigen -template -EIGEN_DONT_INLINE typename T::Scalar pblueNorm(const T& v) +template EIGEN_DONT_INLINE typename T::Scalar pblueNorm(const T &v) { - #ifndef EIGEN_VECTORIZE +#ifndef EIGEN_VECTORIZE return v.blueNorm(); - #else +#else typedef typename T::Scalar Scalar; static int nmax = 0; static Scalar b1, b2, s1m, s2m, overfl, rbig, relerr; int n; - if(nmax <= 0) - { + if (nmax <= 0) { int nbig, ibeta, it, iemin, iemax, iexp; Scalar abig, eps; - nbig = std::numeric_limits::max(); // largest integer - ibeta = std::numeric_limits::radix; //NumTraits::Base; // base for floating-point numbers - it = std::numeric_limits::digits; //NumTraits::Mantissa; // number of base-beta digits in mantissa - iemin = std::numeric_limits::min_exponent; // minimum exponent - iemax = std::numeric_limits::max_exponent; // maximum exponent - rbig = std::numeric_limits::max(); // largest floating-point number + nbig = std::numeric_limits::max();// largest integer + ibeta = std::numeric_limits::radix;// NumTraits::Base; // base for floating-point + // numbers + it = std::numeric_limits::digits;// NumTraits::Mantissa; // number of base-beta + // digits in mantissa + iemin = std::numeric_limits::min_exponent;// minimum exponent + iemax = std::numeric_limits::max_exponent;// maximum exponent + rbig = std::numeric_limits::max();// largest floating-point number // Check the basic machine-dependent constants. - if(iemin > 1 - 2*it || 1+it>iemax || (it==2 && ibeta<5) - || (it<=4 && ibeta <= 3 ) || it<2) - { + if (iemin > 1 - 2 * it || 1 + it > iemax || (it == 2 && ibeta < 5) || (it <= 4 && ibeta <= 3) || it < 2) { eigen_assert(false && "the algorithm cannot be guaranteed on this computer"); } - iexp = -((1-iemin)/2); - b1 = std::pow(ibeta, iexp); // lower boundary of midrange - iexp = (iemax + 1 - it)/2; - b2 = std::pow(ibeta,iexp); // upper boundary of midrange - - iexp = (2-iemin)/2; - s1m = std::pow(ibeta,iexp); // scaling factor for lower range - iexp = - ((iemax+it)/2); - s2m = std::pow(ibeta,iexp); // scaling factor for upper range - - overfl = rbig*s2m; // overfow boundary for abig - eps = std::pow(ibeta, 1-it); - relerr = std::sqrt(eps); // tolerance for neglecting asml - abig = 1.0/eps - 1.0; - if (Scalar(nbig)>abig) nmax = abig; // largest safe n - else nmax = nbig; + iexp = -((1 - iemin) / 2); + b1 = std::pow(ibeta, iexp);// lower boundary of midrange + iexp = (iemax + 1 - it) / 2; + b2 = std::pow(ibeta, iexp);// upper boundary of midrange + + iexp = (2 - iemin) / 2; + s1m = std::pow(ibeta, iexp);// scaling factor for lower range + iexp = -((iemax + it) / 2); + s2m = std::pow(ibeta, iexp);// scaling factor for upper range + + overfl = rbig * s2m;// overfow boundary for abig + eps = std::pow(ibeta, 1 - it); + relerr = std::sqrt(eps);// tolerance for neglecting asml + abig = 1.0 / eps - 1.0; + if (Scalar(nbig) > abig) + nmax = abig;// largest safe n + else + nmax = nbig; } typedef typename internal::packet_traits::type Packet; @@ -149,99 +119,92 @@ EIGEN_DONT_INLINE typename T::Scalar pblueNorm(const T& v) Packet pabig = internal::pset1(Scalar(0)); Packet ps2m = internal::pset1(s2m); Packet ps1m = internal::pset1(s1m); - Packet pb2 = internal::pset1(b2); - Packet pb1 = internal::pset1(b1); - for(int j=0; j(b2); + Packet pb1 = internal::pset1(b1); + for (int j = 0; j < v.size(); j += ps) { Packet ax = internal::pabs(v.template packet(j)); - Packet ax_s2m = internal::pmul(ax,ps2m); - Packet ax_s1m = internal::pmul(ax,ps1m); - Packet maskBig = internal::plt(pb2,ax); - Packet maskSml = internal::plt(ax,pb1); - -// Packet maskMed = internal::pand(maskSml,maskBig); -// Packet scale = internal::pset1(Scalar(0)); -// scale = internal::por(scale, internal::pand(maskBig,ps2m)); -// scale = internal::por(scale, internal::pand(maskSml,ps1m)); -// scale = internal::por(scale, internal::pandnot(internal::pset1(Scalar(1)),maskMed)); -// ax = internal::pmul(ax,scale); -// ax = internal::pmul(ax,ax); -// pabig = internal::padd(pabig, internal::pand(maskBig, ax)); -// pasml = internal::padd(pasml, internal::pand(maskSml, ax)); -// pamed = internal::padd(pamed, internal::pandnot(ax,maskMed)); - - - pabig = internal::padd(pabig, internal::pand(maskBig, internal::pmul(ax_s2m,ax_s2m))); - pasml = internal::padd(pasml, internal::pand(maskSml, internal::pmul(ax_s1m,ax_s1m))); - pamed = internal::padd(pamed, internal::pandnot(internal::pmul(ax,ax),internal::pand(maskSml,maskBig))); + Packet ax_s2m = internal::pmul(ax, ps2m); + Packet ax_s1m = internal::pmul(ax, ps1m); + Packet maskBig = internal::plt(pb2, ax); + Packet maskSml = internal::plt(ax, pb1); + + // Packet maskMed = internal::pand(maskSml,maskBig); + // Packet scale = internal::pset1(Scalar(0)); + // scale = internal::por(scale, internal::pand(maskBig,ps2m)); + // scale = internal::por(scale, internal::pand(maskSml,ps1m)); + // scale = internal::por(scale, internal::pandnot(internal::pset1(Scalar(1)),maskMed)); + // ax = internal::pmul(ax,scale); + // ax = internal::pmul(ax,ax); + // pabig = internal::padd(pabig, internal::pand(maskBig, ax)); + // pasml = internal::padd(pasml, internal::pand(maskSml, ax)); + // pamed = internal::padd(pamed, internal::pandnot(ax,maskMed)); + + + pabig = internal::padd(pabig, internal::pand(maskBig, internal::pmul(ax_s2m, ax_s2m))); + pasml = internal::padd(pasml, internal::pand(maskSml, internal::pmul(ax_s1m, ax_s1m))); + pamed = internal::padd(pamed, internal::pandnot(internal::pmul(ax, ax), internal::pand(maskSml, maskBig))); } Scalar abig = internal::predux(pabig); Scalar asml = internal::predux(pasml); Scalar amed = internal::predux(pamed); - if(abig > Scalar(0)) - { + if (abig > Scalar(0)) { abig = std::sqrt(abig); - if(abig > overfl) - { + if (abig > overfl) { eigen_assert(false && "overflow"); return rbig; } - if(amed > Scalar(0)) - { - abig = abig/s2m; + if (amed > Scalar(0)) { + abig = abig / s2m; amed = std::sqrt(amed); - } - else - { - return abig/s2m; + } else { + return abig / s2m; } - } - else if(asml > Scalar(0)) - { - if (amed > Scalar(0)) - { + } else if (asml > Scalar(0)) { + if (amed > Scalar(0)) { abig = std::sqrt(amed); amed = std::sqrt(asml) / s1m; + } else { + return std::sqrt(asml) / s1m; } - else - { - return std::sqrt(asml)/s1m; - } - } - else - { + } else { return std::sqrt(amed); } asml = std::min(abig, amed); abig = std::max(abig, amed); - if(asml <= abig*relerr) + if (asml <= abig * relerr) return abig; else - return abig * std::sqrt(Scalar(1) + numext::abs2(asml/abig)); - #endif + return abig * std::sqrt(Scalar(1) + numext::abs2(asml / abig)); +#endif } -#define BENCH_PERF(NRM) { \ - float af = 0; double ad = 0; std::complex ac = 0; \ - Eigen::BenchTimer tf, td, tcf; tf.reset(); td.reset(); tcf.reset();\ - for (int k=0; k ac = 0; \ + Eigen::BenchTimer tf, td, tcf; \ + tf.reset(); \ + td.reset(); \ + tcf.reset(); \ + for (int k = 0; k < tries; ++k) { \ + tf.start(); \ + for (int i = 0; i < iters; ++i) { af += NRM(vf); } \ + tf.stop(); \ + } \ + for (int k = 0; k < tries; ++k) { \ + td.start(); \ + for (int i = 0; i < iters; ++i) { ad += NRM(vd); } \ + td.stop(); \ + } \ + /*for (int k=0; k()) * std::pow(double(10), internal::random(ef0,ef1)); - vd[i] = std::abs(internal::random()) * std::pow(double(10), internal::random(ed0,ed1)); + for (int i = 0; i < s; ++i) { + vf[i] = std::abs(internal::random()) * std::pow(double(10), internal::random(ef0, ef1)); + vd[i] = std::abs(internal::random()) * std::pow(double(10), internal::random(ed0, ed1)); } - //std::cout << "reference\t" << internal::sqrt(double(s))*yf << "\t" << internal::sqrt(double(s))*yd << "\n"; - std::cout << "sqsumNorm\t" << sqsumNorm(vf) << "\t" << sqsumNorm(vd) << "\t" << sqsumNorm(vf.cast()) << "\t" << sqsumNorm(vd.cast()) << "\n"; - std::cout << "hypotNorm\t" << hypotNorm(vf) << "\t" << hypotNorm(vd) << "\t" << hypotNorm(vf.cast()) << "\t" << hypotNorm(vd.cast()) << "\n"; - std::cout << "blueNorm\t" << blueNorm(vf) << "\t" << blueNorm(vd) << "\t" << blueNorm(vf.cast()) << "\t" << blueNorm(vd.cast()) << "\n"; - std::cout << "pblueNorm\t" << pblueNorm(vf) << "\t" << pblueNorm(vd) << "\t" << blueNorm(vf.cast()) << "\t" << blueNorm(vd.cast()) << "\n"; - std::cout << "lapackNorm\t" << lapackNorm(vf) << "\t" << lapackNorm(vd) << "\t" << lapackNorm(vf.cast()) << "\t" << lapackNorm(vd.cast()) << "\n"; - std::cout << "twopassNorm\t" << twopassNorm(vf) << "\t" << twopassNorm(vd) << "\t" << twopassNorm(vf.cast()) << "\t" << twopassNorm(vd.cast()) << "\n"; -// std::cout << "bl2passNorm\t" << bl2passNorm(vf) << "\t" << bl2passNorm(vd) << "\t" << bl2passNorm(vf.cast()) << "\t" << bl2passNorm(vd.cast()) << "\n"; + // std::cout << "reference\t" << internal::sqrt(double(s))*yf << "\t" << internal::sqrt(double(s))*yd << "\n"; + std::cout << "sqsumNorm\t" << sqsumNorm(vf) << "\t" << sqsumNorm(vd) << "\t" << sqsumNorm(vf.cast()) + << "\t" << sqsumNorm(vd.cast()) << "\n"; + std::cout << "hypotNorm\t" << hypotNorm(vf) << "\t" << hypotNorm(vd) << "\t" << hypotNorm(vf.cast()) + << "\t" << hypotNorm(vd.cast()) << "\n"; + std::cout << "blueNorm\t" << blueNorm(vf) << "\t" << blueNorm(vd) << "\t" << blueNorm(vf.cast()) << "\t" + << blueNorm(vd.cast()) << "\n"; + std::cout << "pblueNorm\t" << pblueNorm(vf) << "\t" << pblueNorm(vd) << "\t" << blueNorm(vf.cast()) + << "\t" << blueNorm(vd.cast()) << "\n"; + std::cout << "lapackNorm\t" << lapackNorm(vf) << "\t" << lapackNorm(vd) << "\t" << lapackNorm(vf.cast()) + << "\t" << lapackNorm(vd.cast()) << "\n"; + std::cout << "twopassNorm\t" << twopassNorm(vf) << "\t" << twopassNorm(vd) << "\t" + << twopassNorm(vf.cast()) << "\t" << twopassNorm(vd.cast()) << "\n"; + // std::cout << "bl2passNorm\t" << bl2passNorm(vf) << "\t" << bl2passNorm(vd) << "\t" << bl2passNorm(vf.cast()) << "\t" << bl2passNorm(vd.cast()) << "\n"; } -int main(int argc, char** argv) +int main(int argc, char **argv) { int tries = 10; int iters = 100000; double y = 1.1345743233455785456788e12 * internal::random(); VectorXf v = VectorXf::Ones(1024) * y; -// return 0; + // return 0; int s = 10000; double basef_ok = 1.1345743233455785456788e15; double based_ok = 1.1345743233455785456788e95; @@ -310,22 +279,20 @@ int main(int argc, char** argv) check_accuracy(basef_over, based_over, s); std::cerr << "\nVarying (over):\n"; - for (int k=0; k<1; ++k) - { - check_accuracy_var(20,27,190,302,s); + for (int k = 0; k < 1; ++k) { + check_accuracy_var(20, 27, 190, 302, s); std::cout << "\n"; } std::cerr << "\nVarying (under):\n"; - for (int k=0; k<1; ++k) - { - check_accuracy_var(-27,20,-302,-190,s); + for (int k = 0; k < 1; ++k) { + check_accuracy_var(-27, 20, -302, -190, s); std::cout << "\n"; } y = 1; std::cout.precision(4); - int s1 = 1024*1024*32; + int s1 = 1024 * 1024 * 32; std::cerr << "Performance (out of cache, " << s1 << "):\n"; { int iters = 1; diff --git a/filmulator-gui/core/nlmeans/eigen/bench/bench_reverse.cpp b/filmulator-gui/core/nlmeans/eigen/bench/bench_reverse.cpp index 1e69ca1b..b6aa7827 100644 --- a/filmulator-gui/core/nlmeans/eigen/bench/bench_reverse.cpp +++ b/filmulator-gui/core/nlmeans/eigen/bench/bench_reverse.cpp @@ -1,7 +1,7 @@ -#include #include #include +#include using namespace Eigen; #ifndef REPEAT @@ -14,71 +14,64 @@ using namespace Eigen; typedef double Scalar; -template -__attribute__ ((noinline)) void bench_reverse(const MatrixType& m) +template __attribute__((noinline)) void bench_reverse(const MatrixType &m) { int rows = m.rows(); int cols = m.cols(); int size = m.size(); - int repeats = (REPEAT*1000)/size; - MatrixType a = MatrixType::Random(rows,cols); - MatrixType b = MatrixType::Random(rows,cols); + int repeats = (REPEAT * 1000) / size; + MatrixType a = MatrixType::Random(rows, cols); + MatrixType b = MatrixType::Random(rows, cols); BenchTimer timerB, timerH, timerV; Scalar acc = 0; - int r = internal::random(0,rows-1); - int c = internal::random(0,cols-1); - for (int t=0; t(0, rows - 1); + int c = internal::random(0, cols - 1); + for (int t = 0; t < TRIES; ++t) { timerB.start(); - for (int k=0; k0; ++i) - { - bench_reverse(Matrix(dynsizes[i],dynsizes[i])); - bench_reverse(Matrix(dynsizes[i]*dynsizes[i])); + for (uint i = 0; dynsizes[i] > 0; ++i) { + bench_reverse(Matrix(dynsizes[i], dynsizes[i])); + bench_reverse(Matrix(dynsizes[i] * dynsizes[i])); } -// bench_reverse(Matrix()); -// bench_reverse(Matrix()); -// bench_reverse(Matrix()); -// bench_reverse(Matrix()); -// bench_reverse(Matrix()); -// bench_reverse(Matrix()); -// bench_reverse(Matrix()); -// bench_reverse(Matrix()); -// bench_reverse(Matrix()); + // bench_reverse(Matrix()); + // bench_reverse(Matrix()); + // bench_reverse(Matrix()); + // bench_reverse(Matrix()); + // bench_reverse(Matrix()); + // bench_reverse(Matrix()); + // bench_reverse(Matrix()); + // bench_reverse(Matrix()); + // bench_reverse(Matrix()); return 0; } - diff --git a/filmulator-gui/core/nlmeans/eigen/bench/bench_sum.cpp b/filmulator-gui/core/nlmeans/eigen/bench/bench_sum.cpp index a3d925e4..0bcff6c9 100644 --- a/filmulator-gui/core/nlmeans/eigen/bench/bench_sum.cpp +++ b/filmulator-gui/core/nlmeans/eigen/bench/bench_sum.cpp @@ -1,18 +1,15 @@ -#include #include +#include using namespace Eigen; using namespace std; -int main() +int main() { - typedef Matrix Vec; + typedef Matrix Vec; Vec v(SIZE); v.setZero(); v[0] = 1; v[1] = 2; - for(int i = 0; i < 1000000; i++) - { - v.coeffRef(0) += v.sum() * SCALAR(1e-20); - } + for (int i = 0; i < 1000000; i++) { v.coeffRef(0) += v.sum() * SCALAR(1e-20); } cout << v.sum() << endl; } diff --git a/filmulator-gui/core/nlmeans/eigen/bench/benchmark-blocking-sizes.cpp b/filmulator-gui/core/nlmeans/eigen/bench/benchmark-blocking-sizes.cpp index 827be288..4b4097aa 100644 --- a/filmulator-gui/core/nlmeans/eigen/bench/benchmark-blocking-sizes.cpp +++ b/filmulator-gui/core/nlmeans/eigen/bench/benchmark-blocking-sizes.cpp @@ -7,13 +7,13 @@ // Public License v. 2.0. If a copy of the MPL was not distributed // with this file, You can obtain one at http://mozilla.org/MPL/2.0/. -#include #include +#include #include -#include #include +#include #include -#include +#include bool eigen_use_specific_block_size; int eigen_block_size_k, eigen_block_size_m, eigen_block_size_n; @@ -64,7 +64,7 @@ struct size_triple_t size_t k, m, n; size_triple_t() : k(0), m(0), n(0) {} size_triple_t(size_t _k, size_t _m, size_t _n) : k(_k), m(_m), n(_n) {} - size_triple_t(const size_triple_t& o) : k(o.k), m(o.m), n(o.n) {} + size_triple_t(const size_triple_t &o) : k(o.k), m(o.m), n(o.n) {} size_triple_t(uint16_t compact) { k = 1 << ((compact & 0xf00) >> 8); @@ -73,7 +73,8 @@ struct size_triple_t } }; -uint8_t log2_pot(size_t x) { +uint8_t log2_pot(size_t x) +{ size_t l = 0; while (x >>= 1) l++; return l; @@ -87,10 +88,7 @@ uint16_t compact_size_triple(size_t k, size_t m, size_t n) return (log2_pot(k) << 8) | (log2_pot(m) << 4) | log2_pot(n); } -uint16_t compact_size_triple(const size_triple_t& t) -{ - return compact_size_triple(t.k, t.m, t.n); -} +uint16_t compact_size_triple(const size_triple_t &t) { return compact_size_triple(t.k, t.m, t.n); } // A single benchmark. Initially only contains benchmark params. // Then call run(), which stores the result in the gflops field. @@ -100,31 +98,20 @@ struct benchmark_t uint16_t compact_block_size; bool use_default_block_size; float gflops; - benchmark_t() - : compact_product_size(0) - , compact_block_size(0) - , use_default_block_size(false) - , gflops(0) - { - } - benchmark_t(size_t pk, size_t pm, size_t pn, - size_t bk, size_t bm, size_t bn) - : compact_product_size(compact_size_triple(pk, pm, pn)) - , compact_block_size(compact_size_triple(bk, bm, bn)) - , use_default_block_size(false) - , gflops(0) + benchmark_t() : compact_product_size(0), compact_block_size(0), use_default_block_size(false), gflops(0) {} + benchmark_t(size_t pk, size_t pm, size_t pn, size_t bk, size_t bm, size_t bn) + : compact_product_size(compact_size_triple(pk, pm, pn)), compact_block_size(compact_size_triple(bk, bm, bn)), + use_default_block_size(false), gflops(0) {} benchmark_t(size_t pk, size_t pm, size_t pn) - : compact_product_size(compact_size_triple(pk, pm, pn)) - , compact_block_size(0) - , use_default_block_size(true) - , gflops(0) + : compact_product_size(compact_size_triple(pk, pm, pn)), compact_block_size(0), use_default_block_size(true), + gflops(0) {} void run(); }; -ostream& operator<<(ostream& s, const benchmark_t& b) +ostream &operator<<(ostream &s, const benchmark_t &b) { s << hex << b.compact_product_size << dec; if (b.use_default_block_size) { @@ -141,13 +128,12 @@ ostream& operator<<(ostream& s, const benchmark_t& b) // We sort first by increasing benchmark parameters, // then by decreasing performance. -bool operator<(const benchmark_t& b1, const benchmark_t& b2) -{ - return b1.compact_product_size < b2.compact_product_size || - (b1.compact_product_size == b2.compact_product_size && ( - (b1.compact_block_size < b2.compact_block_size || ( - b1.compact_block_size == b2.compact_block_size && - b1.gflops > b2.gflops)))); +bool operator<(const benchmark_t &b1, const benchmark_t &b2) +{ + return b1.compact_product_size < b2.compact_product_size + || (b1.compact_product_size == b2.compact_product_size + && ((b1.compact_block_size < b2.compact_block_size + || (b1.compact_block_size == b2.compact_block_size && b1.gflops > b2.gflops)))); } void benchmark_t::run() @@ -168,26 +154,22 @@ void benchmark_t::run() // set up the matrix pool const size_t combined_three_matrices_sizes = - sizeof(Scalar) * - (productsizes.k * productsizes.m + - productsizes.k * productsizes.n + - productsizes.m * productsizes.n); + sizeof(Scalar) + * (productsizes.k * productsizes.m + productsizes.k * productsizes.n + productsizes.m * productsizes.n); // 64 M is large enough that nobody has a cache bigger than that, // while still being small enough that everybody has this much RAM, // so conveniently we don't need to special-case platforms here. const size_t unlikely_large_cache_size = 64 << 20; - const size_t working_set_size = - min_working_set_size ? min_working_set_size : unlikely_large_cache_size; + const size_t working_set_size = min_working_set_size ? min_working_set_size : unlikely_large_cache_size; - const size_t matrix_pool_size = - 1 + working_set_size / combined_three_matrices_sizes; + const size_t matrix_pool_size = 1 + working_set_size / combined_three_matrices_sizes; MatrixType *lhs = new MatrixType[matrix_pool_size]; MatrixType *rhs = new MatrixType[matrix_pool_size]; MatrixType *dst = new MatrixType[matrix_pool_size]; - + for (size_t i = 0; i < matrix_pool_size; i++) { lhs[i] = MatrixType::Zero(productsizes.m, productsizes.k); rhs[i] = MatrixType::Zero(productsizes.k, productsizes.n); @@ -205,9 +187,7 @@ void benchmark_t::run() for (int i = 0; i < iters_at_a_time; i++) { dst[matrix_index].noalias() = lhs[matrix_index] * rhs[matrix_index]; matrix_index++; - if (matrix_index == matrix_pool_size) { - matrix_index = 0; - } + if (matrix_index == matrix_pool_size) { matrix_index = 0; } } double endtime = timer.getCpuTime(); @@ -235,9 +215,7 @@ void print_cpuinfo() string line; ifstream cpuinfo("/proc/cpuinfo"); if (cpuinfo.is_open()) { - while (getline(cpuinfo, line)) { - cout << line << endl; - } + while (getline(cpuinfo, line)) { cout << line << endl; } cpuinfo.close(); } cout << endl; @@ -248,33 +226,24 @@ void print_cpuinfo() #endif } -template -string type_name() -{ - return "unknown"; -} +template string type_name() { return "unknown"; } -template<> -string type_name() -{ - return "float"; -} +template<> string type_name() { return "float"; } -template<> -string type_name() -{ - return "double"; -} +template<> string type_name() { return "double"; } struct action_t { - virtual const char* invokation_name() const { abort(); return nullptr; } + virtual const char *invokation_name() const + { + abort(); + return nullptr; + } virtual void run() const { abort(); } virtual ~action_t() {} }; -void show_usage_and_exit(int /*argc*/, char* argv[], - const vector>& available_actions) +void show_usage_and_exit(int /*argc*/, char *argv[], const vector> &available_actions) { cerr << "usage: " << argv[0] << " [options...]" << endl << endl; cerr << "available actions:" << endl << endl; @@ -293,11 +262,11 @@ void show_usage_and_exit(int /*argc*/, char* argv[], cerr << " avoid warm caches." << endl; exit(1); } - + float measure_clock_speed() { cerr << "Measuring clock speed... \r" << flush; - + vector all_gflops; for (int i = 0; i < 8; i++) { benchmark_t b(1024, 1024, 1024); @@ -321,7 +290,7 @@ struct human_duration_t human_duration_t(int s) : seconds(s) {} }; -ostream& operator<<(ostream& s, const human_duration_t& d) +ostream &operator<<(ostream &s, const human_duration_t &d) { int remainder = d.seconds; if (remainder > 3600) { @@ -334,17 +303,15 @@ ostream& operator<<(ostream& s, const human_duration_t& d) s << minutes << " min "; remainder -= minutes * 60; } - if (d.seconds < 600) { - s << remainder << " s"; - } + if (d.seconds < 600) { s << remainder << " s"; } return s; } const char session_filename[] = "/data/local/tmp/benchmark-blocking-sizes-session.data"; -void serialize_benchmarks(const char* filename, const vector& benchmarks, size_t first_benchmark_to_run) +void serialize_benchmarks(const char *filename, const vector &benchmarks, size_t first_benchmark_to_run) { - FILE* file = fopen(filename, "w"); + FILE *file = fopen(filename, "w"); if (!file) { cerr << "Could not open file " << filename << " for writing." << endl; cerr << "Do you have write permissions on the current working directory?" << endl; @@ -358,38 +325,23 @@ void serialize_benchmarks(const char* filename, const vector& bench fclose(file); } -bool deserialize_benchmarks(const char* filename, vector& benchmarks, size_t& first_benchmark_to_run) +bool deserialize_benchmarks(const char *filename, vector &benchmarks, size_t &first_benchmark_to_run) { - FILE* file = fopen(filename, "r"); - if (!file) { - return false; - } - if (1 != fread(&max_clock_speed, sizeof(max_clock_speed), 1, file)) { - return false; - } + FILE *file = fopen(filename, "r"); + if (!file) { return false; } + if (1 != fread(&max_clock_speed, sizeof(max_clock_speed), 1, file)) { return false; } size_t benchmarks_vector_size = 0; - if (1 != fread(&benchmarks_vector_size, sizeof(benchmarks_vector_size), 1, file)) { - return false; - } - if (1 != fread(&first_benchmark_to_run, sizeof(first_benchmark_to_run), 1, file)) { - return false; - } + if (1 != fread(&benchmarks_vector_size, sizeof(benchmarks_vector_size), 1, file)) { return false; } + if (1 != fread(&first_benchmark_to_run, sizeof(first_benchmark_to_run), 1, file)) { return false; } benchmarks.resize(benchmarks_vector_size); - if (benchmarks.size() != fread(benchmarks.data(), sizeof(benchmark_t), benchmarks.size(), file)) { - return false; - } + if (benchmarks.size() != fread(benchmarks.data(), sizeof(benchmark_t), benchmarks.size(), file)) { return false; } unlink(filename); return true; } -void try_run_some_benchmarks( - vector& benchmarks, - double time_start, - size_t& first_benchmark_to_run) +void try_run_some_benchmarks(vector &benchmarks, double time_start, size_t &first_benchmark_to_run) { - if (first_benchmark_to_run == benchmarks.size()) { - return; - } + if (first_benchmark_to_run == benchmarks.size()) { return; } double time_last_progress_update = 0; double time_last_clock_speed_measurement = 0; @@ -402,9 +354,7 @@ void try_run_some_benchmarks( time_now = timer.getRealTime(); // We check clock speed every minute and at the end. - if (benchmark_index == benchmarks.size() || - time_now > time_last_clock_speed_measurement + 60.0f) - { + if (benchmark_index == benchmarks.size() || time_now > time_last_clock_speed_measurement + 60.0f) { time_last_clock_speed_measurement = time_now; // Ensure that clock speed is as expected @@ -425,8 +375,7 @@ void try_run_some_benchmarks( // which invalidates all benchmark results collected so far. // Either way, we better restart all over again now. if (benchmark_index) { - cerr << "Restarting at " << 100.0f * ratio_done - << " % because clock speed increased. " << endl; + cerr << "Restarting at " << 100.0f * ratio_done << " % because clock speed increased. " << endl; } max_clock_speed = current_clock_speed; first_benchmark_to_run = 0; @@ -436,12 +385,9 @@ void try_run_some_benchmarks( bool rerun_last_tests = false; if (current_clock_speed < (1 - clock_speed_tolerance) * max_clock_speed) { - cerr << "Measurements completed so far: " - << 100.0f * ratio_done - << " % " << endl; - cerr << "Clock speed seems to be only " - << current_clock_speed/max_clock_speed - << " times what it used to be." << endl; + cerr << "Measurements completed so far: " << 100.0f * ratio_done << " % " << endl; + cerr << "Clock speed seems to be only " << current_clock_speed / max_clock_speed << " times what it used to be." + << endl; unsigned int seconds_to_sleep_if_lower_clock_speed = 1; @@ -454,9 +400,8 @@ void try_run_some_benchmarks( exit(2); } rerun_last_tests = true; - cerr << "Sleeping " - << seconds_to_sleep_if_lower_clock_speed - << " s... \r" << endl; + cerr << "Sleeping " << seconds_to_sleep_if_lower_clock_speed << " s... \r" + << endl; sleep(seconds_to_sleep_if_lower_clock_speed); current_clock_speed = measure_clock_speed(); seconds_to_sleep_if_lower_clock_speed *= 2; @@ -464,8 +409,7 @@ void try_run_some_benchmarks( } if (rerun_last_tests) { - cerr << "Redoing the last " - << 100.0f * float(benchmark_index - first_benchmark_to_run) / benchmarks.size() + cerr << "Redoing the last " << 100.0f * float(benchmark_index - first_benchmark_to_run) / benchmarks.size() << " % because clock speed had been low. " << endl; return; } @@ -486,8 +430,7 @@ void try_run_some_benchmarks( // Display progress info on stderr if (time_now > time_last_progress_update + 1.0f) { time_last_progress_update = time_now; - cerr << "Measurements... " << 100.0f * ratio_done - << " %, ETA " + cerr << "Measurements... " << 100.0f * ratio_done << " %, ETA " << human_duration_t(float(time_now - time_start) * (1.0f - ratio_done) / ratio_done) << " \r" << flush; } @@ -498,19 +441,16 @@ void try_run_some_benchmarks( } } -void run_benchmarks(vector& benchmarks) +void run_benchmarks(vector &benchmarks) { size_t first_benchmark_to_run; vector deserialized_benchmarks; bool use_deserialized_benchmarks = false; if (deserialize_benchmarks(session_filename, deserialized_benchmarks, first_benchmark_to_run)) { - cerr << "Found serialized session with " - << 100.0f * first_benchmark_to_run / deserialized_benchmarks.size() + cerr << "Found serialized session with " << 100.0f * first_benchmark_to_run / deserialized_benchmarks.size() << " % already done" << endl; - if (deserialized_benchmarks.size() == benchmarks.size() && - first_benchmark_to_run > 0 && - first_benchmark_to_run < benchmarks.size()) - { + if (deserialized_benchmarks.size() == benchmarks.size() && first_benchmark_to_run > 0 + && first_benchmark_to_run < benchmarks.size()) { use_deserialized_benchmarks = true; } } @@ -528,18 +468,12 @@ void run_benchmarks(vector& benchmarks) random_shuffle(benchmarks.begin(), benchmarks.end()); } - for (int i = 0; i < 4; i++) { - max_clock_speed = max(max_clock_speed, measure_clock_speed()); - } - + for (int i = 0; i < 4; i++) { max_clock_speed = max(max_clock_speed, measure_clock_speed()); } + double time_start = 0.0; while (first_benchmark_to_run < benchmarks.size()) { - if (first_benchmark_to_run == 0) { - time_start = timer.getRealTime(); - } - try_run_some_benchmarks(benchmarks, - time_start, - first_benchmark_to_run); + if (first_benchmark_to_run == 0) { time_start = timer.getRealTime(); } + try_run_some_benchmarks(benchmarks, time_start, first_benchmark_to_run); } // Sort timings by increasing benchmark parameters, and decreasing gflops. @@ -550,10 +484,8 @@ void run_benchmarks(vector& benchmarks) // Collect best (i.e. now first) results for each parameter values. vector best_benchmarks; for (auto it = benchmarks.begin(); it != benchmarks.end(); ++it) { - if (best_benchmarks.empty() || - best_benchmarks.back().compact_product_size != it->compact_product_size || - best_benchmarks.back().compact_block_size != it->compact_block_size) - { + if (best_benchmarks.empty() || best_benchmarks.back().compact_product_size != it->compact_product_size + || best_benchmarks.back().compact_block_size != it->compact_block_size) { best_benchmarks.push_back(*it); } } @@ -564,7 +496,7 @@ void run_benchmarks(vector& benchmarks) struct measure_all_pot_sizes_action_t : action_t { - virtual const char* invokation_name() const { return "all-pot-sizes"; } + virtual const char *invokation_name() const { return "all-pot-sizes"; } virtual void run() const { vector benchmarks; @@ -587,24 +519,20 @@ struct measure_all_pot_sizes_action_t : action_t run_benchmarks(benchmarks); cout << "BEGIN MEASUREMENTS ALL POT SIZES" << endl; - for (auto it = benchmarks.begin(); it != benchmarks.end(); ++it) { - cout << *it << endl; - } + for (auto it = benchmarks.begin(); it != benchmarks.end(); ++it) { cout << *it << endl; } } }; struct measure_default_sizes_action_t : action_t { - virtual const char* invokation_name() const { return "default-sizes"; } + virtual const char *invokation_name() const { return "default-sizes"; } virtual void run() const { vector benchmarks; for (int repetition = 0; repetition < measurement_repetitions; repetition++) { for (size_t ksize = minsize; ksize <= maxsize; ksize *= 2) { for (size_t msize = minsize; msize <= maxsize; msize *= 2) { - for (size_t nsize = minsize; nsize <= maxsize; nsize *= 2) { - benchmarks.emplace_back(ksize, msize, nsize); - } + for (size_t nsize = minsize; nsize <= maxsize; nsize *= 2) { benchmarks.emplace_back(ksize, msize, nsize); } } } } @@ -612,13 +540,11 @@ struct measure_default_sizes_action_t : action_t run_benchmarks(benchmarks); cout << "BEGIN MEASUREMENTS DEFAULT SIZES" << endl; - for (auto it = benchmarks.begin(); it != benchmarks.end(); ++it) { - cout << *it << endl; - } + for (auto it = benchmarks.begin(); it != benchmarks.end(); ++it) { cout << *it << endl; } } }; -int main(int argc, char* argv[]) +int main(int argc, char *argv[]) { double time_start = timer.getRealTime(); cout.precision(4); @@ -630,9 +556,7 @@ int main(int argc, char* argv[]) auto action = available_actions.end(); - if (argc <= 1) { - show_usage_and_exit(argc, argv, available_actions); - } + if (argc <= 1) { show_usage_and_exit(argc, argv, available_actions); } for (auto it = available_actions.begin(); it != available_actions.end(); ++it) { if (!strcmp(argv[1], (*it)->invokation_name())) { action = it; @@ -640,14 +564,12 @@ int main(int argc, char* argv[]) } } - if (action == available_actions.end()) { - show_usage_and_exit(argc, argv, available_actions); - } + if (action == available_actions.end()) { show_usage_and_exit(argc, argv, available_actions); } for (int i = 2; i < argc; i++) { if (argv[i] == strstr(argv[i], "--min-working-set-size=")) { - const char* equals_sign = strchr(argv[i], '='); - min_working_set_size = strtoul(equals_sign+1, nullptr, 10); + const char *equals_sign = strchr(argv[i], '='); + min_working_set_size = strtoul(equals_sign + 1, nullptr, 10); } else { cerr << "unrecognized option: " << argv[i] << endl << endl; show_usage_and_exit(argc, argv, available_actions); @@ -657,7 +579,7 @@ int main(int argc, char* argv[]) print_cpuinfo(); cout << "benchmark parameters:" << endl; - cout << "pointer size: " << 8*sizeof(void*) << " bits" << endl; + cout << "pointer size: " << 8 * sizeof(void *) << " bits" << endl; cout << "scalar type: " << type_name() << endl; cout << "packet size: " << internal::packet_traits::size << endl; cout << "minsize = " << minsize << endl; @@ -665,9 +587,7 @@ int main(int argc, char* argv[]) cout << "measurement_repetitions = " << measurement_repetitions << endl; cout << "min_accurate_time = " << min_accurate_time << endl; cout << "min_working_set_size = " << min_working_set_size; - if (min_working_set_size == 0) { - cout << " (try to outsize caches)"; - } + if (min_working_set_size == 0) { cout << " (try to outsize caches)"; } cout << endl << endl; (*action)->run(); diff --git a/filmulator-gui/core/nlmeans/eigen/bench/benchmark.cpp b/filmulator-gui/core/nlmeans/eigen/bench/benchmark.cpp index c721b908..a609ec84 100644 --- a/filmulator-gui/core/nlmeans/eigen/bench/benchmark.cpp +++ b/filmulator-gui/core/nlmeans/eigen/bench/benchmark.cpp @@ -21,19 +21,13 @@ using namespace Eigen; int main(int argc, char *argv[]) { - Matrix I = Matrix::Ones(); - Matrix m; - for(int i = 0; i < MATSIZE; i++) - for(int j = 0; j < MATSIZE; j++) - { - m(i,j) = (i+MATSIZE*j); - } - asm("#begin"); - for(int a = 0; a < REPEAT; a++) - { - m = Matrix::Ones() + 0.00005 * (m + (m*m)); - } - asm("#end"); - cout << m << endl; - return 0; + Matrix I = Matrix::Ones(); + Matrix m; + for (int i = 0; i < MATSIZE; i++) + for (int j = 0; j < MATSIZE; j++) { m(i, j) = (i + MATSIZE * j); } + asm("#begin"); + for (int a = 0; a < REPEAT; a++) { m = Matrix::Ones() + 0.00005 * (m + (m * m)); } + asm("#end"); + cout << m << endl; + return 0; } diff --git a/filmulator-gui/core/nlmeans/eigen/bench/benchmarkSlice.cpp b/filmulator-gui/core/nlmeans/eigen/bench/benchmarkSlice.cpp index c5b89c54..5fcbdf05 100644 --- a/filmulator-gui/core/nlmeans/eigen/bench/benchmarkSlice.cpp +++ b/filmulator-gui/core/nlmeans/eigen/bench/benchmarkSlice.cpp @@ -21,17 +21,16 @@ int main(int argc, char *argv[]) Mat m(100, 100); m.setRandom(); - for(int a = 0; a < REPEAT; a++) - { + for (int a = 0; a < REPEAT; a++) { int r, c, nr, nc; - r = Eigen::internal::random(0,10); - c = Eigen::internal::random(0,10); - nr = Eigen::internal::random(50,80); - nc = Eigen::internal::random(50,80); - m.block(r,c,nr,nc) += Mat::Ones(nr,nc); - m.block(r,c,nr,nc) *= SCALAR(10); - m.block(r,c,nr,nc) -= Mat::constant(nr,nc,10); - m.block(r,c,nr,nc) /= SCALAR(10); + r = Eigen::internal::random(0, 10); + c = Eigen::internal::random(0, 10); + nr = Eigen::internal::random(50, 80); + nc = Eigen::internal::random(50, 80); + m.block(r, c, nr, nc) += Mat::Ones(nr, nc); + m.block(r, c, nr, nc) *= SCALAR(10); + m.block(r, c, nr, nc) -= Mat::constant(nr, nc, 10); + m.block(r, c, nr, nc) /= SCALAR(10); } cout << m[0] << endl; return 0; diff --git a/filmulator-gui/core/nlmeans/eigen/bench/benchmarkX.cpp b/filmulator-gui/core/nlmeans/eigen/bench/benchmarkX.cpp index 8e4b60c2..1efc4b26 100644 --- a/filmulator-gui/core/nlmeans/eigen/bench/benchmarkX.cpp +++ b/filmulator-gui/core/nlmeans/eigen/bench/benchmarkX.cpp @@ -21,16 +21,11 @@ using namespace Eigen; int main(int argc, char *argv[]) { - MATTYPE I = MATTYPE::Ones(MATSIZE,MATSIZE); - MATTYPE m(MATSIZE,MATSIZE); - for(int i = 0; i < MATSIZE; i++) for(int j = 0; j < MATSIZE; j++) - { - m(i,j) = (i+j+1)/(MATSIZE*MATSIZE); - } - for(int a = 0; a < REPEAT; a++) - { - m = I + 0.0001 * (m + m*m); - } - cout << m(0,0) << endl; - return 0; + MATTYPE I = MATTYPE::Ones(MATSIZE, MATSIZE); + MATTYPE m(MATSIZE, MATSIZE); + for (int i = 0; i < MATSIZE; i++) + for (int j = 0; j < MATSIZE; j++) { m(i, j) = (i + j + 1) / (MATSIZE * MATSIZE); } + for (int a = 0; a < REPEAT; a++) { m = I + 0.0001 * (m + m * m); } + cout << m(0, 0) << endl; + return 0; } diff --git a/filmulator-gui/core/nlmeans/eigen/bench/benchmarkXcwise.cpp b/filmulator-gui/core/nlmeans/eigen/bench/benchmarkXcwise.cpp index 62437435..d41e7959 100644 --- a/filmulator-gui/core/nlmeans/eigen/bench/benchmarkXcwise.cpp +++ b/filmulator-gui/core/nlmeans/eigen/bench/benchmarkXcwise.cpp @@ -1,7 +1,7 @@ // g++ -O3 -DNDEBUG benchmarkX.cpp -o benchmarkX && time ./benchmarkX -#include #include +#include using namespace std; using namespace Eigen; @@ -20,16 +20,10 @@ using namespace Eigen; int main(int argc, char *argv[]) { - VECTYPE I = VECTYPE::Ones(VECSIZE); - VECTYPE m(VECSIZE,1); - for(int i = 0; i < VECSIZE; i++) - { - m[i] = 0.1 * i/VECSIZE; - } - for(int a = 0; a < REPEAT; a++) - { - m = VECTYPE::Ones(VECSIZE) + 0.00005 * (m.cwise().square() + m/4); - } - cout << m[0] << endl; - return 0; + VECTYPE I = VECTYPE::Ones(VECSIZE); + VECTYPE m(VECSIZE, 1); + for (int i = 0; i < VECSIZE; i++) { m[i] = 0.1 * i / VECSIZE; } + for (int a = 0; a < REPEAT; a++) { m = VECTYPE::Ones(VECSIZE) + 0.00005 * (m.cwise().square() + m / 4); } + cout << m[0] << endl; + return 0; } diff --git a/filmulator-gui/core/nlmeans/eigen/bench/btl/generic_bench/utils/utilities.h b/filmulator-gui/core/nlmeans/eigen/bench/btl/generic_bench/utils/utilities.h index d2330d06..55bc827e 100644 --- a/filmulator-gui/core/nlmeans/eigen/bench/btl/generic_bench/utils/utilities.h +++ b/filmulator-gui/core/nlmeans/eigen/bench/btl/generic_bench/utils/utilities.h @@ -9,82 +9,122 @@ /* --- Definition macros file to print information if _DEBUG_ is defined --- */ -# ifndef UTILITIES_H -# define UTILITIES_H +#ifndef UTILITIES_H +#define UTILITIES_H -# include -//# include ok for gcc3.01 -# include +#include +// # include ok for gcc3.01 +#include /* --- INFOS is always defined (without _DEBUG_): to be used for warnings, with release version --- */ -# define HEREWEARE cout< BLASFUNC(cdotu) (int *, float *, int *, float *, int *); -std::complex BLASFUNC(cdotc) (int *, float *, int *, float *, int *); -std::complex BLASFUNC(zdotu) (int *, double *, int *, double *, int *); -std::complex BLASFUNC(zdotc) (int *, double *, int *, double *, int *); -double BLASFUNC(xdotu) (int *, double *, int *, double *, int *); -double BLASFUNC(xdotc) (int *, double *, int *, double *, int *); +std::complex BLASFUNC(cdotu)(int *, float *, int *, float *, int *); +std::complex BLASFUNC(cdotc)(int *, float *, int *, float *, int *); +std::complex BLASFUNC(zdotu)(int *, double *, int *, double *, int *); +std::complex BLASFUNC(zdotc)(int *, double *, int *, double *, int *); +double BLASFUNC(xdotu)(int *, double *, int *, double *, int *); +double BLASFUNC(xdotc)(int *, double *, int *, double *, int *); #endif -int BLASFUNC(cdotuw) (int *, float *, int *, float *, int *, float*); -int BLASFUNC(cdotcw) (int *, float *, int *, float *, int *, float*); -int BLASFUNC(zdotuw) (int *, double *, int *, double *, int *, double*); -int BLASFUNC(zdotcw) (int *, double *, int *, double *, int *, double*); - -int BLASFUNC(saxpy) (int *, float *, float *, int *, float *, int *); -int BLASFUNC(daxpy) (int *, double *, double *, int *, double *, int *); -int BLASFUNC(qaxpy) (int *, double *, double *, int *, double *, int *); -int BLASFUNC(caxpy) (int *, float *, float *, int *, float *, int *); -int BLASFUNC(zaxpy) (int *, double *, double *, int *, double *, int *); -int BLASFUNC(xaxpy) (int *, double *, double *, int *, double *, int *); -int BLASFUNC(caxpyc)(int *, float *, float *, int *, float *, int *); -int BLASFUNC(zaxpyc)(int *, double *, double *, int *, double *, int *); -int BLASFUNC(xaxpyc)(int *, double *, double *, int *, double *, int *); - -int BLASFUNC(scopy) (int *, float *, int *, float *, int *); -int BLASFUNC(dcopy) (int *, double *, int *, double *, int *); -int BLASFUNC(qcopy) (int *, double *, int *, double *, int *); -int BLASFUNC(ccopy) (int *, float *, int *, float *, int *); -int BLASFUNC(zcopy) (int *, double *, int *, double *, int *); -int BLASFUNC(xcopy) (int *, double *, int *, double *, int *); - -int BLASFUNC(sswap) (int *, float *, int *, float *, int *); -int BLASFUNC(dswap) (int *, double *, int *, double *, int *); -int BLASFUNC(qswap) (int *, double *, int *, double *, int *); -int BLASFUNC(cswap) (int *, float *, int *, float *, int *); -int BLASFUNC(zswap) (int *, double *, int *, double *, int *); -int BLASFUNC(xswap) (int *, double *, int *, double *, int *); - -float BLASFUNC(sasum) (int *, float *, int *); -float BLASFUNC(scasum)(int *, float *, int *); -double BLASFUNC(dasum) (int *, double *, int *); -double BLASFUNC(qasum) (int *, double *, int *); +int BLASFUNC(cdotuw)(int *, float *, int *, float *, int *, float *); +int BLASFUNC(cdotcw)(int *, float *, int *, float *, int *, float *); +int BLASFUNC(zdotuw)(int *, double *, int *, double *, int *, double *); +int BLASFUNC(zdotcw)(int *, double *, int *, double *, int *, double *); + +int BLASFUNC(saxpy)(int *, float *, float *, int *, float *, int *); +int BLASFUNC(daxpy)(int *, double *, double *, int *, double *, int *); +int BLASFUNC(qaxpy)(int *, double *, double *, int *, double *, int *); +int BLASFUNC(caxpy)(int *, float *, float *, int *, float *, int *); +int BLASFUNC(zaxpy)(int *, double *, double *, int *, double *, int *); +int BLASFUNC(xaxpy)(int *, double *, double *, int *, double *, int *); +int BLASFUNC(caxpyc)(int *, float *, float *, int *, float *, int *); +int BLASFUNC(zaxpyc)(int *, double *, double *, int *, double *, int *); +int BLASFUNC(xaxpyc)(int *, double *, double *, int *, double *, int *); + +int BLASFUNC(scopy)(int *, float *, int *, float *, int *); +int BLASFUNC(dcopy)(int *, double *, int *, double *, int *); +int BLASFUNC(qcopy)(int *, double *, int *, double *, int *); +int BLASFUNC(ccopy)(int *, float *, int *, float *, int *); +int BLASFUNC(zcopy)(int *, double *, int *, double *, int *); +int BLASFUNC(xcopy)(int *, double *, int *, double *, int *); + +int BLASFUNC(sswap)(int *, float *, int *, float *, int *); +int BLASFUNC(dswap)(int *, double *, int *, double *, int *); +int BLASFUNC(qswap)(int *, double *, int *, double *, int *); +int BLASFUNC(cswap)(int *, float *, int *, float *, int *); +int BLASFUNC(zswap)(int *, double *, int *, double *, int *); +int BLASFUNC(xswap)(int *, double *, int *, double *, int *); + +float BLASFUNC(sasum)(int *, float *, int *); +float BLASFUNC(scasum)(int *, float *, int *); +double BLASFUNC(dasum)(int *, double *, int *); +double BLASFUNC(qasum)(int *, double *, int *); double BLASFUNC(dzasum)(int *, double *, int *); double BLASFUNC(qxasum)(int *, double *, int *); -int BLASFUNC(isamax)(int *, float *, int *); -int BLASFUNC(idamax)(int *, double *, int *); -int BLASFUNC(iqamax)(int *, double *, int *); -int BLASFUNC(icamax)(int *, float *, int *); -int BLASFUNC(izamax)(int *, double *, int *); -int BLASFUNC(ixamax)(int *, double *, int *); - -int BLASFUNC(ismax) (int *, float *, int *); -int BLASFUNC(idmax) (int *, double *, int *); -int BLASFUNC(iqmax) (int *, double *, int *); -int BLASFUNC(icmax) (int *, float *, int *); -int BLASFUNC(izmax) (int *, double *, int *); -int BLASFUNC(ixmax) (int *, double *, int *); - -int BLASFUNC(isamin)(int *, float *, int *); -int BLASFUNC(idamin)(int *, double *, int *); -int BLASFUNC(iqamin)(int *, double *, int *); -int BLASFUNC(icamin)(int *, float *, int *); -int BLASFUNC(izamin)(int *, double *, int *); -int BLASFUNC(ixamin)(int *, double *, int *); - -int BLASFUNC(ismin)(int *, float *, int *); -int BLASFUNC(idmin)(int *, double *, int *); -int BLASFUNC(iqmin)(int *, double *, int *); -int BLASFUNC(icmin)(int *, float *, int *); -int BLASFUNC(izmin)(int *, double *, int *); -int BLASFUNC(ixmin)(int *, double *, int *); - -float BLASFUNC(samax) (int *, float *, int *); -double BLASFUNC(damax) (int *, double *, int *); -double BLASFUNC(qamax) (int *, double *, int *); -float BLASFUNC(scamax)(int *, float *, int *); +int BLASFUNC(isamax)(int *, float *, int *); +int BLASFUNC(idamax)(int *, double *, int *); +int BLASFUNC(iqamax)(int *, double *, int *); +int BLASFUNC(icamax)(int *, float *, int *); +int BLASFUNC(izamax)(int *, double *, int *); +int BLASFUNC(ixamax)(int *, double *, int *); + +int BLASFUNC(ismax)(int *, float *, int *); +int BLASFUNC(idmax)(int *, double *, int *); +int BLASFUNC(iqmax)(int *, double *, int *); +int BLASFUNC(icmax)(int *, float *, int *); +int BLASFUNC(izmax)(int *, double *, int *); +int BLASFUNC(ixmax)(int *, double *, int *); + +int BLASFUNC(isamin)(int *, float *, int *); +int BLASFUNC(idamin)(int *, double *, int *); +int BLASFUNC(iqamin)(int *, double *, int *); +int BLASFUNC(icamin)(int *, float *, int *); +int BLASFUNC(izamin)(int *, double *, int *); +int BLASFUNC(ixamin)(int *, double *, int *); + +int BLASFUNC(ismin)(int *, float *, int *); +int BLASFUNC(idmin)(int *, double *, int *); +int BLASFUNC(iqmin)(int *, double *, int *); +int BLASFUNC(icmin)(int *, float *, int *); +int BLASFUNC(izmin)(int *, double *, int *); +int BLASFUNC(ixmin)(int *, double *, int *); + +float BLASFUNC(samax)(int *, float *, int *); +double BLASFUNC(damax)(int *, double *, int *); +double BLASFUNC(qamax)(int *, double *, int *); +float BLASFUNC(scamax)(int *, float *, int *); double BLASFUNC(dzamax)(int *, double *, int *); double BLASFUNC(qxamax)(int *, double *, int *); -float BLASFUNC(samin) (int *, float *, int *); -double BLASFUNC(damin) (int *, double *, int *); -double BLASFUNC(qamin) (int *, double *, int *); -float BLASFUNC(scamin)(int *, float *, int *); +float BLASFUNC(samin)(int *, float *, int *); +double BLASFUNC(damin)(int *, double *, int *); +double BLASFUNC(qamin)(int *, double *, int *); +float BLASFUNC(scamin)(int *, float *, int *); double BLASFUNC(dzamin)(int *, double *, int *); double BLASFUNC(qxamin)(int *, double *, int *); -float BLASFUNC(smax) (int *, float *, int *); -double BLASFUNC(dmax) (int *, double *, int *); -double BLASFUNC(qmax) (int *, double *, int *); -float BLASFUNC(scmax) (int *, float *, int *); -double BLASFUNC(dzmax) (int *, double *, int *); -double BLASFUNC(qxmax) (int *, double *, int *); - -float BLASFUNC(smin) (int *, float *, int *); -double BLASFUNC(dmin) (int *, double *, int *); -double BLASFUNC(qmin) (int *, double *, int *); -float BLASFUNC(scmin) (int *, float *, int *); -double BLASFUNC(dzmin) (int *, double *, int *); -double BLASFUNC(qxmin) (int *, double *, int *); - -int BLASFUNC(sscal) (int *, float *, float *, int *); -int BLASFUNC(dscal) (int *, double *, double *, int *); -int BLASFUNC(qscal) (int *, double *, double *, int *); -int BLASFUNC(cscal) (int *, float *, float *, int *); -int BLASFUNC(zscal) (int *, double *, double *, int *); -int BLASFUNC(xscal) (int *, double *, double *, int *); -int BLASFUNC(csscal)(int *, float *, float *, int *); -int BLASFUNC(zdscal)(int *, double *, double *, int *); -int BLASFUNC(xqscal)(int *, double *, double *, int *); - -float BLASFUNC(snrm2) (int *, float *, int *); -float BLASFUNC(scnrm2)(int *, float *, int *); - -double BLASFUNC(dnrm2) (int *, double *, int *); -double BLASFUNC(qnrm2) (int *, double *, int *); +float BLASFUNC(smax)(int *, float *, int *); +double BLASFUNC(dmax)(int *, double *, int *); +double BLASFUNC(qmax)(int *, double *, int *); +float BLASFUNC(scmax)(int *, float *, int *); +double BLASFUNC(dzmax)(int *, double *, int *); +double BLASFUNC(qxmax)(int *, double *, int *); + +float BLASFUNC(smin)(int *, float *, int *); +double BLASFUNC(dmin)(int *, double *, int *); +double BLASFUNC(qmin)(int *, double *, int *); +float BLASFUNC(scmin)(int *, float *, int *); +double BLASFUNC(dzmin)(int *, double *, int *); +double BLASFUNC(qxmin)(int *, double *, int *); + +int BLASFUNC(sscal)(int *, float *, float *, int *); +int BLASFUNC(dscal)(int *, double *, double *, int *); +int BLASFUNC(qscal)(int *, double *, double *, int *); +int BLASFUNC(cscal)(int *, float *, float *, int *); +int BLASFUNC(zscal)(int *, double *, double *, int *); +int BLASFUNC(xscal)(int *, double *, double *, int *); +int BLASFUNC(csscal)(int *, float *, float *, int *); +int BLASFUNC(zdscal)(int *, double *, double *, int *); +int BLASFUNC(xqscal)(int *, double *, double *, int *); + +float BLASFUNC(snrm2)(int *, float *, int *); +float BLASFUNC(scnrm2)(int *, float *, int *); + +double BLASFUNC(dnrm2)(int *, double *, int *); +double BLASFUNC(qnrm2)(int *, double *, int *); double BLASFUNC(dznrm2)(int *, double *, int *); double BLASFUNC(qxnrm2)(int *, double *, int *); -int BLASFUNC(srot) (int *, float *, int *, float *, int *, float *, float *); -int BLASFUNC(drot) (int *, double *, int *, double *, int *, double *, double *); -int BLASFUNC(qrot) (int *, double *, int *, double *, int *, double *, double *); -int BLASFUNC(csrot) (int *, float *, int *, float *, int *, float *, float *); -int BLASFUNC(zdrot) (int *, double *, int *, double *, int *, double *, double *); -int BLASFUNC(xqrot) (int *, double *, int *, double *, int *, double *, double *); +int BLASFUNC(srot)(int *, float *, int *, float *, int *, float *, float *); +int BLASFUNC(drot)(int *, double *, int *, double *, int *, double *, double *); +int BLASFUNC(qrot)(int *, double *, int *, double *, int *, double *, double *); +int BLASFUNC(csrot)(int *, float *, int *, float *, int *, float *, float *); +int BLASFUNC(zdrot)(int *, double *, int *, double *, int *, double *, double *); +int BLASFUNC(xqrot)(int *, double *, int *, double *, int *, double *, double *); -int BLASFUNC(srotg) (float *, float *, float *, float *); -int BLASFUNC(drotg) (double *, double *, double *, double *); -int BLASFUNC(qrotg) (double *, double *, double *, double *); -int BLASFUNC(crotg) (float *, float *, float *, float *); -int BLASFUNC(zrotg) (double *, double *, double *, double *); -int BLASFUNC(xrotg) (double *, double *, double *, double *); +int BLASFUNC(srotg)(float *, float *, float *, float *); +int BLASFUNC(drotg)(double *, double *, double *, double *); +int BLASFUNC(qrotg)(double *, double *, double *, double *); +int BLASFUNC(crotg)(float *, float *, float *, float *); +int BLASFUNC(zrotg)(double *, double *, double *, double *); +int BLASFUNC(xrotg)(double *, double *, double *, double *); -int BLASFUNC(srotmg)(float *, float *, float *, float *, float *); -int BLASFUNC(drotmg)(double *, double *, double *, double *, double *); +int BLASFUNC(srotmg)(float *, float *, float *, float *, float *); +int BLASFUNC(drotmg)(double *, double *, double *, double *, double *); -int BLASFUNC(srotm) (int *, float *, int *, float *, int *, float *); -int BLASFUNC(drotm) (int *, double *, int *, double *, int *, double *); -int BLASFUNC(qrotm) (int *, double *, int *, double *, int *, double *); +int BLASFUNC(srotm)(int *, float *, int *, float *, int *, float *); +int BLASFUNC(drotm)(int *, double *, int *, double *, int *, double *); +int BLASFUNC(qrotm)(int *, double *, int *, double *, int *, double *); /* Level 2 routines */ -int BLASFUNC(sger)(int *, int *, float *, float *, int *, - float *, int *, float *, int *); -int BLASFUNC(dger)(int *, int *, double *, double *, int *, - double *, int *, double *, int *); -int BLASFUNC(qger)(int *, int *, double *, double *, int *, - double *, int *, double *, int *); -int BLASFUNC(cgeru)(int *, int *, float *, float *, int *, - float *, int *, float *, int *); -int BLASFUNC(cgerc)(int *, int *, float *, float *, int *, - float *, int *, float *, int *); -int BLASFUNC(zgeru)(int *, int *, double *, double *, int *, - double *, int *, double *, int *); -int BLASFUNC(zgerc)(int *, int *, double *, double *, int *, - double *, int *, double *, int *); -int BLASFUNC(xgeru)(int *, int *, double *, double *, int *, - double *, int *, double *, int *); -int BLASFUNC(xgerc)(int *, int *, double *, double *, int *, - double *, int *, double *, int *); - -int BLASFUNC(sgemv)(char *, int *, int *, float *, float *, int *, - float *, int *, float *, float *, int *); -int BLASFUNC(dgemv)(char *, int *, int *, double *, double *, int *, - double *, int *, double *, double *, int *); -int BLASFUNC(qgemv)(char *, int *, int *, double *, double *, int *, - double *, int *, double *, double *, int *); -int BLASFUNC(cgemv)(char *, int *, int *, float *, float *, int *, - float *, int *, float *, float *, int *); -int BLASFUNC(zgemv)(char *, int *, int *, double *, double *, int *, - double *, int *, double *, double *, int *); -int BLASFUNC(xgemv)(char *, int *, int *, double *, double *, int *, - double *, int *, double *, double *, int *); - -int BLASFUNC(strsv) (char *, char *, char *, int *, float *, int *, - float *, int *); -int BLASFUNC(dtrsv) (char *, char *, char *, int *, double *, int *, - double *, int *); -int BLASFUNC(qtrsv) (char *, char *, char *, int *, double *, int *, - double *, int *); -int BLASFUNC(ctrsv) (char *, char *, char *, int *, float *, int *, - float *, int *); -int BLASFUNC(ztrsv) (char *, char *, char *, int *, double *, int *, - double *, int *); -int BLASFUNC(xtrsv) (char *, char *, char *, int *, double *, int *, - double *, int *); - -int BLASFUNC(stpsv) (char *, char *, char *, int *, float *, float *, int *); -int BLASFUNC(dtpsv) (char *, char *, char *, int *, double *, double *, int *); -int BLASFUNC(qtpsv) (char *, char *, char *, int *, double *, double *, int *); -int BLASFUNC(ctpsv) (char *, char *, char *, int *, float *, float *, int *); -int BLASFUNC(ztpsv) (char *, char *, char *, int *, double *, double *, int *); -int BLASFUNC(xtpsv) (char *, char *, char *, int *, double *, double *, int *); - -int BLASFUNC(strmv) (char *, char *, char *, int *, float *, int *, - float *, int *); -int BLASFUNC(dtrmv) (char *, char *, char *, int *, double *, int *, - double *, int *); -int BLASFUNC(qtrmv) (char *, char *, char *, int *, double *, int *, - double *, int *); -int BLASFUNC(ctrmv) (char *, char *, char *, int *, float *, int *, - float *, int *); -int BLASFUNC(ztrmv) (char *, char *, char *, int *, double *, int *, - double *, int *); -int BLASFUNC(xtrmv) (char *, char *, char *, int *, double *, int *, - double *, int *); - -int BLASFUNC(stpmv) (char *, char *, char *, int *, float *, float *, int *); -int BLASFUNC(dtpmv) (char *, char *, char *, int *, double *, double *, int *); -int BLASFUNC(qtpmv) (char *, char *, char *, int *, double *, double *, int *); -int BLASFUNC(ctpmv) (char *, char *, char *, int *, float *, float *, int *); -int BLASFUNC(ztpmv) (char *, char *, char *, int *, double *, double *, int *); -int BLASFUNC(xtpmv) (char *, char *, char *, int *, double *, double *, int *); - -int BLASFUNC(stbmv) (char *, char *, char *, int *, int *, float *, int *, float *, int *); -int BLASFUNC(dtbmv) (char *, char *, char *, int *, int *, double *, int *, double *, int *); -int BLASFUNC(qtbmv) (char *, char *, char *, int *, int *, double *, int *, double *, int *); -int BLASFUNC(ctbmv) (char *, char *, char *, int *, int *, float *, int *, float *, int *); -int BLASFUNC(ztbmv) (char *, char *, char *, int *, int *, double *, int *, double *, int *); -int BLASFUNC(xtbmv) (char *, char *, char *, int *, int *, double *, int *, double *, int *); - -int BLASFUNC(stbsv) (char *, char *, char *, int *, int *, float *, int *, float *, int *); -int BLASFUNC(dtbsv) (char *, char *, char *, int *, int *, double *, int *, double *, int *); -int BLASFUNC(qtbsv) (char *, char *, char *, int *, int *, double *, int *, double *, int *); -int BLASFUNC(ctbsv) (char *, char *, char *, int *, int *, float *, int *, float *, int *); -int BLASFUNC(ztbsv) (char *, char *, char *, int *, int *, double *, int *, double *, int *); -int BLASFUNC(xtbsv) (char *, char *, char *, int *, int *, double *, int *, double *, int *); - -int BLASFUNC(ssymv) (char *, int *, float *, float *, int *, - float *, int *, float *, float *, int *); -int BLASFUNC(dsymv) (char *, int *, double *, double *, int *, - double *, int *, double *, double *, int *); -int BLASFUNC(qsymv) (char *, int *, double *, double *, int *, - double *, int *, double *, double *, int *); -int BLASFUNC(csymv) (char *, int *, float *, float *, int *, - float *, int *, float *, float *, int *); -int BLASFUNC(zsymv) (char *, int *, double *, double *, int *, - double *, int *, double *, double *, int *); -int BLASFUNC(xsymv) (char *, int *, double *, double *, int *, - double *, int *, double *, double *, int *); - -int BLASFUNC(sspmv) (char *, int *, float *, float *, - float *, int *, float *, float *, int *); -int BLASFUNC(dspmv) (char *, int *, double *, double *, - double *, int *, double *, double *, int *); -int BLASFUNC(qspmv) (char *, int *, double *, double *, - double *, int *, double *, double *, int *); -int BLASFUNC(cspmv) (char *, int *, float *, float *, - float *, int *, float *, float *, int *); -int BLASFUNC(zspmv) (char *, int *, double *, double *, - double *, int *, double *, double *, int *); -int BLASFUNC(xspmv) (char *, int *, double *, double *, - double *, int *, double *, double *, int *); - -int BLASFUNC(ssyr) (char *, int *, float *, float *, int *, - float *, int *); -int BLASFUNC(dsyr) (char *, int *, double *, double *, int *, - double *, int *); -int BLASFUNC(qsyr) (char *, int *, double *, double *, int *, - double *, int *); -int BLASFUNC(csyr) (char *, int *, float *, float *, int *, - float *, int *); -int BLASFUNC(zsyr) (char *, int *, double *, double *, int *, - double *, int *); -int BLASFUNC(xsyr) (char *, int *, double *, double *, int *, - double *, int *); - -int BLASFUNC(ssyr2) (char *, int *, float *, - float *, int *, float *, int *, float *, int *); -int BLASFUNC(dsyr2) (char *, int *, double *, - double *, int *, double *, int *, double *, int *); -int BLASFUNC(qsyr2) (char *, int *, double *, - double *, int *, double *, int *, double *, int *); -int BLASFUNC(csyr2) (char *, int *, float *, - float *, int *, float *, int *, float *, int *); -int BLASFUNC(zsyr2) (char *, int *, double *, - double *, int *, double *, int *, double *, int *); -int BLASFUNC(xsyr2) (char *, int *, double *, - double *, int *, double *, int *, double *, int *); - -int BLASFUNC(sspr) (char *, int *, float *, float *, int *, - float *); -int BLASFUNC(dspr) (char *, int *, double *, double *, int *, - double *); -int BLASFUNC(qspr) (char *, int *, double *, double *, int *, - double *); -int BLASFUNC(cspr) (char *, int *, float *, float *, int *, - float *); -int BLASFUNC(zspr) (char *, int *, double *, double *, int *, - double *); -int BLASFUNC(xspr) (char *, int *, double *, double *, int *, - double *); - -int BLASFUNC(sspr2) (char *, int *, float *, - float *, int *, float *, int *, float *); -int BLASFUNC(dspr2) (char *, int *, double *, - double *, int *, double *, int *, double *); -int BLASFUNC(qspr2) (char *, int *, double *, - double *, int *, double *, int *, double *); -int BLASFUNC(cspr2) (char *, int *, float *, - float *, int *, float *, int *, float *); -int BLASFUNC(zspr2) (char *, int *, double *, - double *, int *, double *, int *, double *); -int BLASFUNC(xspr2) (char *, int *, double *, - double *, int *, double *, int *, double *); - -int BLASFUNC(cher) (char *, int *, float *, float *, int *, - float *, int *); -int BLASFUNC(zher) (char *, int *, double *, double *, int *, - double *, int *); -int BLASFUNC(xher) (char *, int *, double *, double *, int *, - double *, int *); - -int BLASFUNC(chpr) (char *, int *, float *, float *, int *, float *); -int BLASFUNC(zhpr) (char *, int *, double *, double *, int *, double *); -int BLASFUNC(xhpr) (char *, int *, double *, double *, int *, double *); - -int BLASFUNC(cher2) (char *, int *, float *, - float *, int *, float *, int *, float *, int *); -int BLASFUNC(zher2) (char *, int *, double *, - double *, int *, double *, int *, double *, int *); -int BLASFUNC(xher2) (char *, int *, double *, - double *, int *, double *, int *, double *, int *); - -int BLASFUNC(chpr2) (char *, int *, float *, - float *, int *, float *, int *, float *); -int BLASFUNC(zhpr2) (char *, int *, double *, - double *, int *, double *, int *, double *); -int BLASFUNC(xhpr2) (char *, int *, double *, - double *, int *, double *, int *, double *); - -int BLASFUNC(chemv) (char *, int *, float *, float *, int *, - float *, int *, float *, float *, int *); -int BLASFUNC(zhemv) (char *, int *, double *, double *, int *, - double *, int *, double *, double *, int *); -int BLASFUNC(xhemv) (char *, int *, double *, double *, int *, - double *, int *, double *, double *, int *); - -int BLASFUNC(chpmv) (char *, int *, float *, float *, - float *, int *, float *, float *, int *); -int BLASFUNC(zhpmv) (char *, int *, double *, double *, - double *, int *, double *, double *, int *); -int BLASFUNC(xhpmv) (char *, int *, double *, double *, - double *, int *, double *, double *, int *); - -int BLASFUNC(snorm)(char *, int *, int *, float *, int *); +int BLASFUNC(sger)(int *, int *, float *, float *, int *, float *, int *, float *, int *); +int BLASFUNC(dger)(int *, int *, double *, double *, int *, double *, int *, double *, int *); +int BLASFUNC(qger)(int *, int *, double *, double *, int *, double *, int *, double *, int *); +int BLASFUNC(cgeru)(int *, int *, float *, float *, int *, float *, int *, float *, int *); +int BLASFUNC(cgerc)(int *, int *, float *, float *, int *, float *, int *, float *, int *); +int BLASFUNC(zgeru)(int *, int *, double *, double *, int *, double *, int *, double *, int *); +int BLASFUNC(zgerc)(int *, int *, double *, double *, int *, double *, int *, double *, int *); +int BLASFUNC(xgeru)(int *, int *, double *, double *, int *, double *, int *, double *, int *); +int BLASFUNC(xgerc)(int *, int *, double *, double *, int *, double *, int *, double *, int *); + +int BLASFUNC(sgemv)(char *, int *, int *, float *, float *, int *, float *, int *, float *, float *, int *); +int BLASFUNC(dgemv)(char *, int *, int *, double *, double *, int *, double *, int *, double *, double *, int *); +int BLASFUNC(qgemv)(char *, int *, int *, double *, double *, int *, double *, int *, double *, double *, int *); +int BLASFUNC(cgemv)(char *, int *, int *, float *, float *, int *, float *, int *, float *, float *, int *); +int BLASFUNC(zgemv)(char *, int *, int *, double *, double *, int *, double *, int *, double *, double *, int *); +int BLASFUNC(xgemv)(char *, int *, int *, double *, double *, int *, double *, int *, double *, double *, int *); + +int BLASFUNC(strsv)(char *, char *, char *, int *, float *, int *, float *, int *); +int BLASFUNC(dtrsv)(char *, char *, char *, int *, double *, int *, double *, int *); +int BLASFUNC(qtrsv)(char *, char *, char *, int *, double *, int *, double *, int *); +int BLASFUNC(ctrsv)(char *, char *, char *, int *, float *, int *, float *, int *); +int BLASFUNC(ztrsv)(char *, char *, char *, int *, double *, int *, double *, int *); +int BLASFUNC(xtrsv)(char *, char *, char *, int *, double *, int *, double *, int *); + +int BLASFUNC(stpsv)(char *, char *, char *, int *, float *, float *, int *); +int BLASFUNC(dtpsv)(char *, char *, char *, int *, double *, double *, int *); +int BLASFUNC(qtpsv)(char *, char *, char *, int *, double *, double *, int *); +int BLASFUNC(ctpsv)(char *, char *, char *, int *, float *, float *, int *); +int BLASFUNC(ztpsv)(char *, char *, char *, int *, double *, double *, int *); +int BLASFUNC(xtpsv)(char *, char *, char *, int *, double *, double *, int *); + +int BLASFUNC(strmv)(char *, char *, char *, int *, float *, int *, float *, int *); +int BLASFUNC(dtrmv)(char *, char *, char *, int *, double *, int *, double *, int *); +int BLASFUNC(qtrmv)(char *, char *, char *, int *, double *, int *, double *, int *); +int BLASFUNC(ctrmv)(char *, char *, char *, int *, float *, int *, float *, int *); +int BLASFUNC(ztrmv)(char *, char *, char *, int *, double *, int *, double *, int *); +int BLASFUNC(xtrmv)(char *, char *, char *, int *, double *, int *, double *, int *); + +int BLASFUNC(stpmv)(char *, char *, char *, int *, float *, float *, int *); +int BLASFUNC(dtpmv)(char *, char *, char *, int *, double *, double *, int *); +int BLASFUNC(qtpmv)(char *, char *, char *, int *, double *, double *, int *); +int BLASFUNC(ctpmv)(char *, char *, char *, int *, float *, float *, int *); +int BLASFUNC(ztpmv)(char *, char *, char *, int *, double *, double *, int *); +int BLASFUNC(xtpmv)(char *, char *, char *, int *, double *, double *, int *); + +int BLASFUNC(stbmv)(char *, char *, char *, int *, int *, float *, int *, float *, int *); +int BLASFUNC(dtbmv)(char *, char *, char *, int *, int *, double *, int *, double *, int *); +int BLASFUNC(qtbmv)(char *, char *, char *, int *, int *, double *, int *, double *, int *); +int BLASFUNC(ctbmv)(char *, char *, char *, int *, int *, float *, int *, float *, int *); +int BLASFUNC(ztbmv)(char *, char *, char *, int *, int *, double *, int *, double *, int *); +int BLASFUNC(xtbmv)(char *, char *, char *, int *, int *, double *, int *, double *, int *); + +int BLASFUNC(stbsv)(char *, char *, char *, int *, int *, float *, int *, float *, int *); +int BLASFUNC(dtbsv)(char *, char *, char *, int *, int *, double *, int *, double *, int *); +int BLASFUNC(qtbsv)(char *, char *, char *, int *, int *, double *, int *, double *, int *); +int BLASFUNC(ctbsv)(char *, char *, char *, int *, int *, float *, int *, float *, int *); +int BLASFUNC(ztbsv)(char *, char *, char *, int *, int *, double *, int *, double *, int *); +int BLASFUNC(xtbsv)(char *, char *, char *, int *, int *, double *, int *, double *, int *); + +int BLASFUNC(ssymv)(char *, int *, float *, float *, int *, float *, int *, float *, float *, int *); +int BLASFUNC(dsymv)(char *, int *, double *, double *, int *, double *, int *, double *, double *, int *); +int BLASFUNC(qsymv)(char *, int *, double *, double *, int *, double *, int *, double *, double *, int *); +int BLASFUNC(csymv)(char *, int *, float *, float *, int *, float *, int *, float *, float *, int *); +int BLASFUNC(zsymv)(char *, int *, double *, double *, int *, double *, int *, double *, double *, int *); +int BLASFUNC(xsymv)(char *, int *, double *, double *, int *, double *, int *, double *, double *, int *); + +int BLASFUNC(sspmv)(char *, int *, float *, float *, float *, int *, float *, float *, int *); +int BLASFUNC(dspmv)(char *, int *, double *, double *, double *, int *, double *, double *, int *); +int BLASFUNC(qspmv)(char *, int *, double *, double *, double *, int *, double *, double *, int *); +int BLASFUNC(cspmv)(char *, int *, float *, float *, float *, int *, float *, float *, int *); +int BLASFUNC(zspmv)(char *, int *, double *, double *, double *, int *, double *, double *, int *); +int BLASFUNC(xspmv)(char *, int *, double *, double *, double *, int *, double *, double *, int *); + +int BLASFUNC(ssyr)(char *, int *, float *, float *, int *, float *, int *); +int BLASFUNC(dsyr)(char *, int *, double *, double *, int *, double *, int *); +int BLASFUNC(qsyr)(char *, int *, double *, double *, int *, double *, int *); +int BLASFUNC(csyr)(char *, int *, float *, float *, int *, float *, int *); +int BLASFUNC(zsyr)(char *, int *, double *, double *, int *, double *, int *); +int BLASFUNC(xsyr)(char *, int *, double *, double *, int *, double *, int *); + +int BLASFUNC(ssyr2)(char *, int *, float *, float *, int *, float *, int *, float *, int *); +int BLASFUNC(dsyr2)(char *, int *, double *, double *, int *, double *, int *, double *, int *); +int BLASFUNC(qsyr2)(char *, int *, double *, double *, int *, double *, int *, double *, int *); +int BLASFUNC(csyr2)(char *, int *, float *, float *, int *, float *, int *, float *, int *); +int BLASFUNC(zsyr2)(char *, int *, double *, double *, int *, double *, int *, double *, int *); +int BLASFUNC(xsyr2)(char *, int *, double *, double *, int *, double *, int *, double *, int *); + +int BLASFUNC(sspr)(char *, int *, float *, float *, int *, float *); +int BLASFUNC(dspr)(char *, int *, double *, double *, int *, double *); +int BLASFUNC(qspr)(char *, int *, double *, double *, int *, double *); +int BLASFUNC(cspr)(char *, int *, float *, float *, int *, float *); +int BLASFUNC(zspr)(char *, int *, double *, double *, int *, double *); +int BLASFUNC(xspr)(char *, int *, double *, double *, int *, double *); + +int BLASFUNC(sspr2)(char *, int *, float *, float *, int *, float *, int *, float *); +int BLASFUNC(dspr2)(char *, int *, double *, double *, int *, double *, int *, double *); +int BLASFUNC(qspr2)(char *, int *, double *, double *, int *, double *, int *, double *); +int BLASFUNC(cspr2)(char *, int *, float *, float *, int *, float *, int *, float *); +int BLASFUNC(zspr2)(char *, int *, double *, double *, int *, double *, int *, double *); +int BLASFUNC(xspr2)(char *, int *, double *, double *, int *, double *, int *, double *); + +int BLASFUNC(cher)(char *, int *, float *, float *, int *, float *, int *); +int BLASFUNC(zher)(char *, int *, double *, double *, int *, double *, int *); +int BLASFUNC(xher)(char *, int *, double *, double *, int *, double *, int *); + +int BLASFUNC(chpr)(char *, int *, float *, float *, int *, float *); +int BLASFUNC(zhpr)(char *, int *, double *, double *, int *, double *); +int BLASFUNC(xhpr)(char *, int *, double *, double *, int *, double *); + +int BLASFUNC(cher2)(char *, int *, float *, float *, int *, float *, int *, float *, int *); +int BLASFUNC(zher2)(char *, int *, double *, double *, int *, double *, int *, double *, int *); +int BLASFUNC(xher2)(char *, int *, double *, double *, int *, double *, int *, double *, int *); + +int BLASFUNC(chpr2)(char *, int *, float *, float *, int *, float *, int *, float *); +int BLASFUNC(zhpr2)(char *, int *, double *, double *, int *, double *, int *, double *); +int BLASFUNC(xhpr2)(char *, int *, double *, double *, int *, double *, int *, double *); + +int BLASFUNC(chemv)(char *, int *, float *, float *, int *, float *, int *, float *, float *, int *); +int BLASFUNC(zhemv)(char *, int *, double *, double *, int *, double *, int *, double *, double *, int *); +int BLASFUNC(xhemv)(char *, int *, double *, double *, int *, double *, int *, double *, double *, int *); + +int BLASFUNC(chpmv)(char *, int *, float *, float *, float *, int *, float *, float *, int *); +int BLASFUNC(zhpmv)(char *, int *, double *, double *, double *, int *, double *, double *, int *); +int BLASFUNC(xhpmv)(char *, int *, double *, double *, double *, int *, double *, double *, int *); + +int BLASFUNC(snorm)(char *, int *, int *, float *, int *); int BLASFUNC(dnorm)(char *, int *, int *, double *, int *); -int BLASFUNC(cnorm)(char *, int *, int *, float *, int *); +int BLASFUNC(cnorm)(char *, int *, int *, float *, int *); int BLASFUNC(znorm)(char *, int *, int *, double *, int *); -int BLASFUNC(sgbmv)(char *, int *, int *, int *, int *, float *, float *, int *, - float *, int *, float *, float *, int *); -int BLASFUNC(dgbmv)(char *, int *, int *, int *, int *, double *, double *, int *, - double *, int *, double *, double *, int *); -int BLASFUNC(qgbmv)(char *, int *, int *, int *, int *, double *, double *, int *, - double *, int *, double *, double *, int *); -int BLASFUNC(cgbmv)(char *, int *, int *, int *, int *, float *, float *, int *, - float *, int *, float *, float *, int *); -int BLASFUNC(zgbmv)(char *, int *, int *, int *, int *, double *, double *, int *, - double *, int *, double *, double *, int *); -int BLASFUNC(xgbmv)(char *, int *, int *, int *, int *, double *, double *, int *, - double *, int *, double *, double *, int *); - -int BLASFUNC(ssbmv)(char *, int *, int *, float *, float *, int *, - float *, int *, float *, float *, int *); -int BLASFUNC(dsbmv)(char *, int *, int *, double *, double *, int *, - double *, int *, double *, double *, int *); -int BLASFUNC(qsbmv)(char *, int *, int *, double *, double *, int *, - double *, int *, double *, double *, int *); -int BLASFUNC(csbmv)(char *, int *, int *, float *, float *, int *, - float *, int *, float *, float *, int *); -int BLASFUNC(zsbmv)(char *, int *, int *, double *, double *, int *, - double *, int *, double *, double *, int *); -int BLASFUNC(xsbmv)(char *, int *, int *, double *, double *, int *, - double *, int *, double *, double *, int *); - -int BLASFUNC(chbmv)(char *, int *, int *, float *, float *, int *, - float *, int *, float *, float *, int *); -int BLASFUNC(zhbmv)(char *, int *, int *, double *, double *, int *, - double *, int *, double *, double *, int *); -int BLASFUNC(xhbmv)(char *, int *, int *, double *, double *, int *, - double *, int *, double *, double *, int *); +int BLASFUNC( + sgbmv)(char *, int *, int *, int *, int *, float *, float *, int *, float *, int *, float *, float *, int *); +int BLASFUNC( + dgbmv)(char *, int *, int *, int *, int *, double *, double *, int *, double *, int *, double *, double *, int *); +int BLASFUNC( + qgbmv)(char *, int *, int *, int *, int *, double *, double *, int *, double *, int *, double *, double *, int *); +int BLASFUNC( + cgbmv)(char *, int *, int *, int *, int *, float *, float *, int *, float *, int *, float *, float *, int *); +int BLASFUNC( + zgbmv)(char *, int *, int *, int *, int *, double *, double *, int *, double *, int *, double *, double *, int *); +int BLASFUNC( + xgbmv)(char *, int *, int *, int *, int *, double *, double *, int *, double *, int *, double *, double *, int *); + +int BLASFUNC(ssbmv)(char *, int *, int *, float *, float *, int *, float *, int *, float *, float *, int *); +int BLASFUNC(dsbmv)(char *, int *, int *, double *, double *, int *, double *, int *, double *, double *, int *); +int BLASFUNC(qsbmv)(char *, int *, int *, double *, double *, int *, double *, int *, double *, double *, int *); +int BLASFUNC(csbmv)(char *, int *, int *, float *, float *, int *, float *, int *, float *, float *, int *); +int BLASFUNC(zsbmv)(char *, int *, int *, double *, double *, int *, double *, int *, double *, double *, int *); +int BLASFUNC(xsbmv)(char *, int *, int *, double *, double *, int *, double *, int *, double *, double *, int *); + +int BLASFUNC(chbmv)(char *, int *, int *, float *, float *, int *, float *, int *, float *, float *, int *); +int BLASFUNC(zhbmv)(char *, int *, int *, double *, double *, int *, double *, int *, double *, double *, int *); +int BLASFUNC(xhbmv)(char *, int *, int *, double *, double *, int *, double *, int *, double *, double *, int *); /* Level 3 routines */ -int BLASFUNC(sgemm)(char *, char *, int *, int *, int *, float *, - float *, int *, float *, int *, float *, float *, int *); -int BLASFUNC(dgemm)(char *, char *, int *, int *, int *, double *, - double *, int *, double *, int *, double *, double *, int *); -int BLASFUNC(qgemm)(char *, char *, int *, int *, int *, double *, - double *, int *, double *, int *, double *, double *, int *); -int BLASFUNC(cgemm)(char *, char *, int *, int *, int *, float *, - float *, int *, float *, int *, float *, float *, int *); -int BLASFUNC(zgemm)(char *, char *, int *, int *, int *, double *, - double *, int *, double *, int *, double *, double *, int *); -int BLASFUNC(xgemm)(char *, char *, int *, int *, int *, double *, - double *, int *, double *, int *, double *, double *, int *); - -int BLASFUNC(cgemm3m)(char *, char *, int *, int *, int *, float *, - float *, int *, float *, int *, float *, float *, int *); -int BLASFUNC(zgemm3m)(char *, char *, int *, int *, int *, double *, - double *, int *, double *, int *, double *, double *, int *); -int BLASFUNC(xgemm3m)(char *, char *, int *, int *, int *, double *, - double *, int *, double *, int *, double *, double *, int *); - -int BLASFUNC(sge2mm)(char *, char *, char *, int *, int *, - float *, float *, int *, float *, int *, - float *, float *, int *); -int BLASFUNC(dge2mm)(char *, char *, char *, int *, int *, - double *, double *, int *, double *, int *, - double *, double *, int *); -int BLASFUNC(cge2mm)(char *, char *, char *, int *, int *, - float *, float *, int *, float *, int *, - float *, float *, int *); -int BLASFUNC(zge2mm)(char *, char *, char *, int *, int *, - double *, double *, int *, double *, int *, - double *, double *, int *); - -int BLASFUNC(strsm)(char *, char *, char *, char *, int *, int *, - float *, float *, int *, float *, int *); -int BLASFUNC(dtrsm)(char *, char *, char *, char *, int *, int *, - double *, double *, int *, double *, int *); -int BLASFUNC(qtrsm)(char *, char *, char *, char *, int *, int *, - double *, double *, int *, double *, int *); -int BLASFUNC(ctrsm)(char *, char *, char *, char *, int *, int *, - float *, float *, int *, float *, int *); -int BLASFUNC(ztrsm)(char *, char *, char *, char *, int *, int *, - double *, double *, int *, double *, int *); -int BLASFUNC(xtrsm)(char *, char *, char *, char *, int *, int *, - double *, double *, int *, double *, int *); - -int BLASFUNC(strmm)(char *, char *, char *, char *, int *, int *, - float *, float *, int *, float *, int *); -int BLASFUNC(dtrmm)(char *, char *, char *, char *, int *, int *, - double *, double *, int *, double *, int *); -int BLASFUNC(qtrmm)(char *, char *, char *, char *, int *, int *, - double *, double *, int *, double *, int *); -int BLASFUNC(ctrmm)(char *, char *, char *, char *, int *, int *, - float *, float *, int *, float *, int *); -int BLASFUNC(ztrmm)(char *, char *, char *, char *, int *, int *, - double *, double *, int *, double *, int *); -int BLASFUNC(xtrmm)(char *, char *, char *, char *, int *, int *, - double *, double *, int *, double *, int *); - -int BLASFUNC(ssymm)(char *, char *, int *, int *, float *, float *, int *, - float *, int *, float *, float *, int *); -int BLASFUNC(dsymm)(char *, char *, int *, int *, double *, double *, int *, - double *, int *, double *, double *, int *); -int BLASFUNC(qsymm)(char *, char *, int *, int *, double *, double *, int *, - double *, int *, double *, double *, int *); -int BLASFUNC(csymm)(char *, char *, int *, int *, float *, float *, int *, - float *, int *, float *, float *, int *); -int BLASFUNC(zsymm)(char *, char *, int *, int *, double *, double *, int *, - double *, int *, double *, double *, int *); -int BLASFUNC(xsymm)(char *, char *, int *, int *, double *, double *, int *, - double *, int *, double *, double *, int *); - -int BLASFUNC(csymm3m)(char *, char *, int *, int *, float *, float *, int *, - float *, int *, float *, float *, int *); -int BLASFUNC(zsymm3m)(char *, char *, int *, int *, double *, double *, int *, - double *, int *, double *, double *, int *); -int BLASFUNC(xsymm3m)(char *, char *, int *, int *, double *, double *, int *, - double *, int *, double *, double *, int *); - -int BLASFUNC(ssyrk)(char *, char *, int *, int *, float *, float *, int *, - float *, float *, int *); -int BLASFUNC(dsyrk)(char *, char *, int *, int *, double *, double *, int *, - double *, double *, int *); -int BLASFUNC(qsyrk)(char *, char *, int *, int *, double *, double *, int *, - double *, double *, int *); -int BLASFUNC(csyrk)(char *, char *, int *, int *, float *, float *, int *, - float *, float *, int *); -int BLASFUNC(zsyrk)(char *, char *, int *, int *, double *, double *, int *, - double *, double *, int *); -int BLASFUNC(xsyrk)(char *, char *, int *, int *, double *, double *, int *, - double *, double *, int *); - -int BLASFUNC(ssyr2k)(char *, char *, int *, int *, float *, float *, int *, - float *, int *, float *, float *, int *); -int BLASFUNC(dsyr2k)(char *, char *, int *, int *, double *, double *, int *, - double*, int *, double *, double *, int *); -int BLASFUNC(qsyr2k)(char *, char *, int *, int *, double *, double *, int *, - double*, int *, double *, double *, int *); -int BLASFUNC(csyr2k)(char *, char *, int *, int *, float *, float *, int *, - float *, int *, float *, float *, int *); -int BLASFUNC(zsyr2k)(char *, char *, int *, int *, double *, double *, int *, - double*, int *, double *, double *, int *); -int BLASFUNC(xsyr2k)(char *, char *, int *, int *, double *, double *, int *, - double*, int *, double *, double *, int *); - -int BLASFUNC(chemm)(char *, char *, int *, int *, float *, float *, int *, - float *, int *, float *, float *, int *); -int BLASFUNC(zhemm)(char *, char *, int *, int *, double *, double *, int *, - double *, int *, double *, double *, int *); -int BLASFUNC(xhemm)(char *, char *, int *, int *, double *, double *, int *, - double *, int *, double *, double *, int *); - -int BLASFUNC(chemm3m)(char *, char *, int *, int *, float *, float *, int *, - float *, int *, float *, float *, int *); -int BLASFUNC(zhemm3m)(char *, char *, int *, int *, double *, double *, int *, - double *, int *, double *, double *, int *); -int BLASFUNC(xhemm3m)(char *, char *, int *, int *, double *, double *, int *, - double *, int *, double *, double *, int *); - -int BLASFUNC(cherk)(char *, char *, int *, int *, float *, float *, int *, - float *, float *, int *); -int BLASFUNC(zherk)(char *, char *, int *, int *, double *, double *, int *, - double *, double *, int *); -int BLASFUNC(xherk)(char *, char *, int *, int *, double *, double *, int *, - double *, double *, int *); - -int BLASFUNC(cher2k)(char *, char *, int *, int *, float *, float *, int *, - float *, int *, float *, float *, int *); -int BLASFUNC(zher2k)(char *, char *, int *, int *, double *, double *, int *, - double*, int *, double *, double *, int *); -int BLASFUNC(xher2k)(char *, char *, int *, int *, double *, double *, int *, - double*, int *, double *, double *, int *); -int BLASFUNC(cher2m)(char *, char *, char *, int *, int *, float *, float *, int *, - float *, int *, float *, float *, int *); -int BLASFUNC(zher2m)(char *, char *, char *, int *, int *, double *, double *, int *, - double*, int *, double *, double *, int *); -int BLASFUNC(xher2m)(char *, char *, char *, int *, int *, double *, double *, int *, - double*, int *, double *, double *, int *); - -int BLASFUNC(sgemt)(char *, int *, int *, float *, float *, int *, - float *, int *); -int BLASFUNC(dgemt)(char *, int *, int *, double *, double *, int *, - double *, int *); -int BLASFUNC(cgemt)(char *, int *, int *, float *, float *, int *, - float *, int *); -int BLASFUNC(zgemt)(char *, int *, int *, double *, double *, int *, - double *, int *); - -int BLASFUNC(sgema)(char *, char *, int *, int *, float *, - float *, int *, float *, float *, int *, float *, int *); -int BLASFUNC(dgema)(char *, char *, int *, int *, double *, - double *, int *, double*, double *, int *, double*, int *); -int BLASFUNC(cgema)(char *, char *, int *, int *, float *, - float *, int *, float *, float *, int *, float *, int *); -int BLASFUNC(zgema)(char *, char *, int *, int *, double *, - double *, int *, double*, double *, int *, double*, int *); - -int BLASFUNC(sgems)(char *, char *, int *, int *, float *, - float *, int *, float *, float *, int *, float *, int *); -int BLASFUNC(dgems)(char *, char *, int *, int *, double *, - double *, int *, double*, double *, int *, double*, int *); -int BLASFUNC(cgems)(char *, char *, int *, int *, float *, - float *, int *, float *, float *, int *, float *, int *); -int BLASFUNC(zgems)(char *, char *, int *, int *, double *, - double *, int *, double*, double *, int *, double*, int *); - -int BLASFUNC(sgetf2)(int *, int *, float *, int *, int *, int *); +int BLASFUNC( + sgemm)(char *, char *, int *, int *, int *, float *, float *, int *, float *, int *, float *, float *, int *); +int BLASFUNC( + dgemm)(char *, char *, int *, int *, int *, double *, double *, int *, double *, int *, double *, double *, int *); +int BLASFUNC( + qgemm)(char *, char *, int *, int *, int *, double *, double *, int *, double *, int *, double *, double *, int *); +int BLASFUNC( + cgemm)(char *, char *, int *, int *, int *, float *, float *, int *, float *, int *, float *, float *, int *); +int BLASFUNC( + zgemm)(char *, char *, int *, int *, int *, double *, double *, int *, double *, int *, double *, double *, int *); +int BLASFUNC( + xgemm)(char *, char *, int *, int *, int *, double *, double *, int *, double *, int *, double *, double *, int *); + +int BLASFUNC( + cgemm3m)(char *, char *, int *, int *, int *, float *, float *, int *, float *, int *, float *, float *, int *); +int BLASFUNC( + zgemm3m)(char *, char *, int *, int *, int *, double *, double *, int *, double *, int *, double *, double *, int *); +int BLASFUNC( + xgemm3m)(char *, char *, int *, int *, int *, double *, double *, int *, double *, int *, double *, double *, int *); + +int BLASFUNC( + sge2mm)(char *, char *, char *, int *, int *, float *, float *, int *, float *, int *, float *, float *, int *); +int BLASFUNC( + dge2mm)(char *, char *, char *, int *, int *, double *, double *, int *, double *, int *, double *, double *, int *); +int BLASFUNC( + cge2mm)(char *, char *, char *, int *, int *, float *, float *, int *, float *, int *, float *, float *, int *); +int BLASFUNC( + zge2mm)(char *, char *, char *, int *, int *, double *, double *, int *, double *, int *, double *, double *, int *); + +int BLASFUNC(strsm)(char *, char *, char *, char *, int *, int *, float *, float *, int *, float *, int *); +int BLASFUNC(dtrsm)(char *, char *, char *, char *, int *, int *, double *, double *, int *, double *, int *); +int BLASFUNC(qtrsm)(char *, char *, char *, char *, int *, int *, double *, double *, int *, double *, int *); +int BLASFUNC(ctrsm)(char *, char *, char *, char *, int *, int *, float *, float *, int *, float *, int *); +int BLASFUNC(ztrsm)(char *, char *, char *, char *, int *, int *, double *, double *, int *, double *, int *); +int BLASFUNC(xtrsm)(char *, char *, char *, char *, int *, int *, double *, double *, int *, double *, int *); + +int BLASFUNC(strmm)(char *, char *, char *, char *, int *, int *, float *, float *, int *, float *, int *); +int BLASFUNC(dtrmm)(char *, char *, char *, char *, int *, int *, double *, double *, int *, double *, int *); +int BLASFUNC(qtrmm)(char *, char *, char *, char *, int *, int *, double *, double *, int *, double *, int *); +int BLASFUNC(ctrmm)(char *, char *, char *, char *, int *, int *, float *, float *, int *, float *, int *); +int BLASFUNC(ztrmm)(char *, char *, char *, char *, int *, int *, double *, double *, int *, double *, int *); +int BLASFUNC(xtrmm)(char *, char *, char *, char *, int *, int *, double *, double *, int *, double *, int *); + +int BLASFUNC(ssymm)(char *, char *, int *, int *, float *, float *, int *, float *, int *, float *, float *, int *); +int BLASFUNC( + dsymm)(char *, char *, int *, int *, double *, double *, int *, double *, int *, double *, double *, int *); +int BLASFUNC( + qsymm)(char *, char *, int *, int *, double *, double *, int *, double *, int *, double *, double *, int *); +int BLASFUNC(csymm)(char *, char *, int *, int *, float *, float *, int *, float *, int *, float *, float *, int *); +int BLASFUNC( + zsymm)(char *, char *, int *, int *, double *, double *, int *, double *, int *, double *, double *, int *); +int BLASFUNC( + xsymm)(char *, char *, int *, int *, double *, double *, int *, double *, int *, double *, double *, int *); + +int BLASFUNC(csymm3m)(char *, char *, int *, int *, float *, float *, int *, float *, int *, float *, float *, int *); +int BLASFUNC( + zsymm3m)(char *, char *, int *, int *, double *, double *, int *, double *, int *, double *, double *, int *); +int BLASFUNC( + xsymm3m)(char *, char *, int *, int *, double *, double *, int *, double *, int *, double *, double *, int *); + +int BLASFUNC(ssyrk)(char *, char *, int *, int *, float *, float *, int *, float *, float *, int *); +int BLASFUNC(dsyrk)(char *, char *, int *, int *, double *, double *, int *, double *, double *, int *); +int BLASFUNC(qsyrk)(char *, char *, int *, int *, double *, double *, int *, double *, double *, int *); +int BLASFUNC(csyrk)(char *, char *, int *, int *, float *, float *, int *, float *, float *, int *); +int BLASFUNC(zsyrk)(char *, char *, int *, int *, double *, double *, int *, double *, double *, int *); +int BLASFUNC(xsyrk)(char *, char *, int *, int *, double *, double *, int *, double *, double *, int *); + +int BLASFUNC(ssyr2k)(char *, char *, int *, int *, float *, float *, int *, float *, int *, float *, float *, int *); +int BLASFUNC( + dsyr2k)(char *, char *, int *, int *, double *, double *, int *, double *, int *, double *, double *, int *); +int BLASFUNC( + qsyr2k)(char *, char *, int *, int *, double *, double *, int *, double *, int *, double *, double *, int *); +int BLASFUNC(csyr2k)(char *, char *, int *, int *, float *, float *, int *, float *, int *, float *, float *, int *); +int BLASFUNC( + zsyr2k)(char *, char *, int *, int *, double *, double *, int *, double *, int *, double *, double *, int *); +int BLASFUNC( + xsyr2k)(char *, char *, int *, int *, double *, double *, int *, double *, int *, double *, double *, int *); + +int BLASFUNC(chemm)(char *, char *, int *, int *, float *, float *, int *, float *, int *, float *, float *, int *); +int BLASFUNC( + zhemm)(char *, char *, int *, int *, double *, double *, int *, double *, int *, double *, double *, int *); +int BLASFUNC( + xhemm)(char *, char *, int *, int *, double *, double *, int *, double *, int *, double *, double *, int *); + +int BLASFUNC(chemm3m)(char *, char *, int *, int *, float *, float *, int *, float *, int *, float *, float *, int *); +int BLASFUNC( + zhemm3m)(char *, char *, int *, int *, double *, double *, int *, double *, int *, double *, double *, int *); +int BLASFUNC( + xhemm3m)(char *, char *, int *, int *, double *, double *, int *, double *, int *, double *, double *, int *); + +int BLASFUNC(cherk)(char *, char *, int *, int *, float *, float *, int *, float *, float *, int *); +int BLASFUNC(zherk)(char *, char *, int *, int *, double *, double *, int *, double *, double *, int *); +int BLASFUNC(xherk)(char *, char *, int *, int *, double *, double *, int *, double *, double *, int *); + +int BLASFUNC(cher2k)(char *, char *, int *, int *, float *, float *, int *, float *, int *, float *, float *, int *); +int BLASFUNC( + zher2k)(char *, char *, int *, int *, double *, double *, int *, double *, int *, double *, double *, int *); +int BLASFUNC( + xher2k)(char *, char *, int *, int *, double *, double *, int *, double *, int *, double *, double *, int *); +int BLASFUNC( + cher2m)(char *, char *, char *, int *, int *, float *, float *, int *, float *, int *, float *, float *, int *); +int BLASFUNC( + zher2m)(char *, char *, char *, int *, int *, double *, double *, int *, double *, int *, double *, double *, int *); +int BLASFUNC( + xher2m)(char *, char *, char *, int *, int *, double *, double *, int *, double *, int *, double *, double *, int *); + +int BLASFUNC(sgemt)(char *, int *, int *, float *, float *, int *, float *, int *); +int BLASFUNC(dgemt)(char *, int *, int *, double *, double *, int *, double *, int *); +int BLASFUNC(cgemt)(char *, int *, int *, float *, float *, int *, float *, int *); +int BLASFUNC(zgemt)(char *, int *, int *, double *, double *, int *, double *, int *); + +int BLASFUNC(sgema)(char *, char *, int *, int *, float *, float *, int *, float *, float *, int *, float *, int *); +int BLASFUNC( + dgema)(char *, char *, int *, int *, double *, double *, int *, double *, double *, int *, double *, int *); +int BLASFUNC(cgema)(char *, char *, int *, int *, float *, float *, int *, float *, float *, int *, float *, int *); +int BLASFUNC( + zgema)(char *, char *, int *, int *, double *, double *, int *, double *, double *, int *, double *, int *); + +int BLASFUNC(sgems)(char *, char *, int *, int *, float *, float *, int *, float *, float *, int *, float *, int *); +int BLASFUNC( + dgems)(char *, char *, int *, int *, double *, double *, int *, double *, double *, int *, double *, int *); +int BLASFUNC(cgems)(char *, char *, int *, int *, float *, float *, int *, float *, float *, int *, float *, int *); +int BLASFUNC( + zgems)(char *, char *, int *, int *, double *, double *, int *, double *, double *, int *, double *, int *); + +int BLASFUNC(sgetf2)(int *, int *, float *, int *, int *, int *); int BLASFUNC(dgetf2)(int *, int *, double *, int *, int *, int *); int BLASFUNC(qgetf2)(int *, int *, double *, int *, int *, int *); -int BLASFUNC(cgetf2)(int *, int *, float *, int *, int *, int *); +int BLASFUNC(cgetf2)(int *, int *, float *, int *, int *, int *); int BLASFUNC(zgetf2)(int *, int *, double *, int *, int *, int *); int BLASFUNC(xgetf2)(int *, int *, double *, int *, int *, int *); -int BLASFUNC(sgetrf)(int *, int *, float *, int *, int *, int *); +int BLASFUNC(sgetrf)(int *, int *, float *, int *, int *, int *); int BLASFUNC(dgetrf)(int *, int *, double *, int *, int *, int *); int BLASFUNC(qgetrf)(int *, int *, double *, int *, int *, int *); -int BLASFUNC(cgetrf)(int *, int *, float *, int *, int *, int *); +int BLASFUNC(cgetrf)(int *, int *, float *, int *, int *, int *); int BLASFUNC(zgetrf)(int *, int *, double *, int *, int *, int *); int BLASFUNC(xgetrf)(int *, int *, double *, int *, int *, int *); -int BLASFUNC(slaswp)(int *, float *, int *, int *, int *, int *, int *); +int BLASFUNC(slaswp)(int *, float *, int *, int *, int *, int *, int *); int BLASFUNC(dlaswp)(int *, double *, int *, int *, int *, int *, int *); int BLASFUNC(qlaswp)(int *, double *, int *, int *, int *, int *, int *); -int BLASFUNC(claswp)(int *, float *, int *, int *, int *, int *, int *); +int BLASFUNC(claswp)(int *, float *, int *, int *, int *, int *, int *); int BLASFUNC(zlaswp)(int *, double *, int *, int *, int *, int *, int *); int BLASFUNC(xlaswp)(int *, double *, int *, int *, int *, int *, int *); -int BLASFUNC(sgetrs)(char *, int *, int *, float *, int *, int *, float *, int *, int *); +int BLASFUNC(sgetrs)(char *, int *, int *, float *, int *, int *, float *, int *, int *); int BLASFUNC(dgetrs)(char *, int *, int *, double *, int *, int *, double *, int *, int *); int BLASFUNC(qgetrs)(char *, int *, int *, double *, int *, int *, double *, int *, int *); -int BLASFUNC(cgetrs)(char *, int *, int *, float *, int *, int *, float *, int *, int *); +int BLASFUNC(cgetrs)(char *, int *, int *, float *, int *, int *, float *, int *, int *); int BLASFUNC(zgetrs)(char *, int *, int *, double *, int *, int *, double *, int *, int *); int BLASFUNC(xgetrs)(char *, int *, int *, double *, int *, int *, double *, int *, int *); -int BLASFUNC(sgesv)(int *, int *, float *, int *, int *, float *, int *, int *); -int BLASFUNC(dgesv)(int *, int *, double *, int *, int *, double*, int *, int *); -int BLASFUNC(qgesv)(int *, int *, double *, int *, int *, double*, int *, int *); -int BLASFUNC(cgesv)(int *, int *, float *, int *, int *, float *, int *, int *); -int BLASFUNC(zgesv)(int *, int *, double *, int *, int *, double*, int *, int *); -int BLASFUNC(xgesv)(int *, int *, double *, int *, int *, double*, int *, int *); +int BLASFUNC(sgesv)(int *, int *, float *, int *, int *, float *, int *, int *); +int BLASFUNC(dgesv)(int *, int *, double *, int *, int *, double *, int *, int *); +int BLASFUNC(qgesv)(int *, int *, double *, int *, int *, double *, int *, int *); +int BLASFUNC(cgesv)(int *, int *, float *, int *, int *, float *, int *, int *); +int BLASFUNC(zgesv)(int *, int *, double *, int *, int *, double *, int *, int *); +int BLASFUNC(xgesv)(int *, int *, double *, int *, int *, double *, int *, int *); -int BLASFUNC(spotf2)(char *, int *, float *, int *, int *); +int BLASFUNC(spotf2)(char *, int *, float *, int *, int *); int BLASFUNC(dpotf2)(char *, int *, double *, int *, int *); int BLASFUNC(qpotf2)(char *, int *, double *, int *, int *); -int BLASFUNC(cpotf2)(char *, int *, float *, int *, int *); +int BLASFUNC(cpotf2)(char *, int *, float *, int *, int *); int BLASFUNC(zpotf2)(char *, int *, double *, int *, int *); int BLASFUNC(xpotf2)(char *, int *, double *, int *, int *); -int BLASFUNC(spotrf)(char *, int *, float *, int *, int *); +int BLASFUNC(spotrf)(char *, int *, float *, int *, int *); int BLASFUNC(dpotrf)(char *, int *, double *, int *, int *); int BLASFUNC(qpotrf)(char *, int *, double *, int *, int *); -int BLASFUNC(cpotrf)(char *, int *, float *, int *, int *); +int BLASFUNC(cpotrf)(char *, int *, float *, int *, int *); int BLASFUNC(zpotrf)(char *, int *, double *, int *, int *); int BLASFUNC(xpotrf)(char *, int *, double *, int *, int *); -int BLASFUNC(slauu2)(char *, int *, float *, int *, int *); +int BLASFUNC(slauu2)(char *, int *, float *, int *, int *); int BLASFUNC(dlauu2)(char *, int *, double *, int *, int *); int BLASFUNC(qlauu2)(char *, int *, double *, int *, int *); -int BLASFUNC(clauu2)(char *, int *, float *, int *, int *); +int BLASFUNC(clauu2)(char *, int *, float *, int *, int *); int BLASFUNC(zlauu2)(char *, int *, double *, int *, int *); int BLASFUNC(xlauu2)(char *, int *, double *, int *, int *); -int BLASFUNC(slauum)(char *, int *, float *, int *, int *); +int BLASFUNC(slauum)(char *, int *, float *, int *, int *); int BLASFUNC(dlauum)(char *, int *, double *, int *, int *); int BLASFUNC(qlauum)(char *, int *, double *, int *, int *); -int BLASFUNC(clauum)(char *, int *, float *, int *, int *); +int BLASFUNC(clauum)(char *, int *, float *, int *, int *); int BLASFUNC(zlauum)(char *, int *, double *, int *, int *); int BLASFUNC(xlauum)(char *, int *, double *, int *, int *); -int BLASFUNC(strti2)(char *, char *, int *, float *, int *, int *); +int BLASFUNC(strti2)(char *, char *, int *, float *, int *, int *); int BLASFUNC(dtrti2)(char *, char *, int *, double *, int *, int *); int BLASFUNC(qtrti2)(char *, char *, int *, double *, int *, int *); -int BLASFUNC(ctrti2)(char *, char *, int *, float *, int *, int *); +int BLASFUNC(ctrti2)(char *, char *, int *, float *, int *, int *); int BLASFUNC(ztrti2)(char *, char *, int *, double *, int *, int *); int BLASFUNC(xtrti2)(char *, char *, int *, double *, int *, int *); -int BLASFUNC(strtri)(char *, char *, int *, float *, int *, int *); +int BLASFUNC(strtri)(char *, char *, int *, float *, int *, int *); int BLASFUNC(dtrtri)(char *, char *, int *, double *, int *, int *); int BLASFUNC(qtrtri)(char *, char *, int *, double *, int *, int *); -int BLASFUNC(ctrtri)(char *, char *, int *, float *, int *, int *); +int BLASFUNC(ctrtri)(char *, char *, int *, float *, int *, int *); int BLASFUNC(ztrtri)(char *, char *, int *, double *, int *, int *); int BLASFUNC(xtrtri)(char *, char *, int *, double *, int *, int *); -int BLASFUNC(spotri)(char *, int *, float *, int *, int *); +int BLASFUNC(spotri)(char *, int *, float *, int *, int *); int BLASFUNC(dpotri)(char *, int *, double *, int *, int *); int BLASFUNC(qpotri)(char *, int *, double *, int *, int *); -int BLASFUNC(cpotri)(char *, int *, float *, int *, int *); +int BLASFUNC(cpotri)(char *, int *, float *, int *, int *); int BLASFUNC(zpotri)(char *, int *, double *, int *, int *); int BLASFUNC(xpotri)(char *, int *, double *, int *, int *); diff --git a/filmulator-gui/core/nlmeans/eigen/bench/btl/libs/BLAS/c_interface_base.h b/filmulator-gui/core/nlmeans/eigen/bench/btl/libs/BLAS/c_interface_base.h index de613803..abc44f66 100644 --- a/filmulator-gui/core/nlmeans/eigen/bench/btl/libs/BLAS/c_interface_base.h +++ b/filmulator-gui/core/nlmeans/eigen/bench/btl/libs/BLAS/c_interface_base.h @@ -9,65 +9,58 @@ template class c_interface_base { public: + typedef real real_type; + typedef std::vector stl_vector; + typedef std::vector stl_matrix; - typedef real real_type; - typedef std::vector stl_vector; - typedef std::vector stl_matrix; + typedef real *gene_matrix; + typedef real *gene_vector; - typedef real* gene_matrix; - typedef real* gene_vector; + static void free_matrix(gene_matrix &A, int /*N*/) { delete[] A; } - static void free_matrix(gene_matrix & A, int /*N*/){ - delete[] A; - } - - static void free_vector(gene_vector & B){ - delete[] B; - } + static void free_vector(gene_vector &B) { delete[] B; } - static inline void matrix_from_stl(gene_matrix & A, stl_matrix & A_stl){ + static inline void matrix_from_stl(gene_matrix &A, stl_matrix &A_stl) + { int N = A_stl.size(); - A = new real[N*N]; - for (int j=0;j > >(MIN_AXPY,MAX_AXPY,NB_POINT); - bench > >(MIN_AXPY,MAX_AXPY,NB_POINT); + bench>>(MIN_AXPY, MAX_AXPY, NB_POINT); + bench>>(MIN_AXPY, MAX_AXPY, NB_POINT); - bench > >(MIN_MV,MAX_MV,NB_POINT); - bench > >(MIN_MV,MAX_MV,NB_POINT); - bench > >(MIN_MV,MAX_MV,NB_POINT); - bench > >(MIN_MV,MAX_MV,NB_POINT); + bench>>(MIN_MV, MAX_MV, NB_POINT); + bench>>(MIN_MV, MAX_MV, NB_POINT); + bench>>(MIN_MV, MAX_MV, NB_POINT); + bench>>(MIN_MV, MAX_MV, NB_POINT); - bench > >(MIN_MV,MAX_MV,NB_POINT); - bench > >(MIN_AXPY,MAX_AXPY,NB_POINT); + bench>>(MIN_MV, MAX_MV, NB_POINT); + bench>>(MIN_AXPY, MAX_AXPY, NB_POINT); - bench > >(MIN_MM,MAX_MM,NB_POINT); -// bench > >(MIN_MM,MAX_MM,NB_POINT); - bench > >(MIN_MM,MAX_MM,NB_POINT); + bench>>(MIN_MM, MAX_MM, NB_POINT); + // bench > >(MIN_MM,MAX_MM,NB_POINT); + bench>>(MIN_MM, MAX_MM, NB_POINT); - bench > >(MIN_MM,MAX_MM,NB_POINT); - bench > >(MIN_MM,MAX_MM,NB_POINT); + bench>>(MIN_MM, MAX_MM, NB_POINT); + bench>>(MIN_MM, MAX_MM, NB_POINT); - bench > >(MIN_MM,MAX_MM,NB_POINT); + bench>>(MIN_MM, MAX_MM, NB_POINT); - bench > >(MIN_LU,MAX_LU,NB_POINT); - bench > >(MIN_LU,MAX_LU,NB_POINT); + bench>>(MIN_LU, MAX_LU, NB_POINT); + bench>>(MIN_LU, MAX_LU, NB_POINT); - #ifdef HAS_LAPACK -// bench > >(MIN_LU,MAX_LU,NB_POINT); - bench > >(MIN_LU,MAX_LU,NB_POINT); - bench > >(MIN_LU,MAX_LU,NB_POINT); - #endif +#ifdef HAS_LAPACK + // bench > >(MIN_LU,MAX_LU,NB_POINT); + bench>>(MIN_LU, MAX_LU, NB_POINT); + bench>>(MIN_LU, MAX_LU, NB_POINT); +#endif - //bench > >(MIN_LU,MAX_LU,NB_POINT); + // bench > >(MIN_LU,MAX_LU,NB_POINT); return 0; } - - diff --git a/filmulator-gui/core/nlmeans/eigen/bench/btl/libs/STL/main.cpp b/filmulator-gui/core/nlmeans/eigen/bench/btl/libs/STL/main.cpp index 4e73328e..476b7da9 100644 --- a/filmulator-gui/core/nlmeans/eigen/bench/btl/libs/STL/main.cpp +++ b/filmulator-gui/core/nlmeans/eigen/bench/btl/libs/STL/main.cpp @@ -17,26 +17,24 @@ // along with this program; if not, write to the Free Software // Foundation, Inc., 59 Temple Place - Suite 330, Boston, MA 02111-1307, USA. // -#include "utilities.h" #include "STL_interface.hh" -#include "bench.hh" #include "basic_actions.hh" +#include "bench.hh" +#include "utilities.h" BTL_MAIN; int main() { - bench > >(MIN_AXPY,MAX_AXPY,NB_POINT); - bench > >(MIN_AXPY,MAX_AXPY,NB_POINT); - bench > >(MIN_MV,MAX_MV,NB_POINT); - bench > >(MIN_MV,MAX_MV,NB_POINT); - bench > >(MIN_MV,MAX_MV,NB_POINT); - bench > >(MIN_MV,MAX_MV,NB_POINT); - bench > >(MIN_MM,MAX_MM,NB_POINT); - bench > >(MIN_MM,MAX_MM,NB_POINT); - bench > >(MIN_MM,MAX_MM,NB_POINT); + bench>>(MIN_AXPY, MAX_AXPY, NB_POINT); + bench>>(MIN_AXPY, MAX_AXPY, NB_POINT); + bench>>(MIN_MV, MAX_MV, NB_POINT); + bench>>(MIN_MV, MAX_MV, NB_POINT); + bench>>(MIN_MV, MAX_MV, NB_POINT); + bench>>(MIN_MV, MAX_MV, NB_POINT); + bench>>(MIN_MM, MAX_MM, NB_POINT); + bench>>(MIN_MM, MAX_MM, NB_POINT); + bench>>(MIN_MM, MAX_MM, NB_POINT); return 0; } - - diff --git a/filmulator-gui/core/nlmeans/eigen/bench/btl/libs/blaze/main.cpp b/filmulator-gui/core/nlmeans/eigen/bench/btl/libs/blaze/main.cpp index 80e8f4ea..a0e60329 100644 --- a/filmulator-gui/core/nlmeans/eigen/bench/btl/libs/blaze/main.cpp +++ b/filmulator-gui/core/nlmeans/eigen/bench/btl/libs/blaze/main.cpp @@ -15,26 +15,24 @@ // along with this program; if not, write to the Free Software // Foundation, Inc., 59 Temple Place - Suite 330, Boston, MA 02111-1307, USA. // -#include "utilities.h" -#include "blaze_interface.hh" -#include "bench.hh" #include "basic_actions.hh" +#include "bench.hh" +#include "blaze_interface.hh" +#include "utilities.h" BTL_MAIN; int main() { - bench > >(MIN_AXPY,MAX_AXPY,NB_POINT); - bench > >(MIN_AXPY,MAX_AXPY,NB_POINT); + bench>>(MIN_AXPY, MAX_AXPY, NB_POINT); + bench>>(MIN_AXPY, MAX_AXPY, NB_POINT); - bench > >(MIN_MV,MAX_MV,NB_POINT); - bench > >(MIN_MV,MAX_MV,NB_POINT); -// bench > >(MIN_MM,MAX_MM,NB_POINT); -// bench > >(MIN_MM,MAX_MM,NB_POINT); -// bench > >(MIN_MM,MAX_MM,NB_POINT); + bench>>(MIN_MV, MAX_MV, NB_POINT); + bench>>(MIN_MV, MAX_MV, NB_POINT); + // bench > >(MIN_MM,MAX_MM,NB_POINT); + // bench > >(MIN_MM,MAX_MM,NB_POINT); + // bench > >(MIN_MM,MAX_MM,NB_POINT); return 0; } - - diff --git a/filmulator-gui/core/nlmeans/eigen/bench/btl/libs/blitz/btl_blitz.cpp b/filmulator-gui/core/nlmeans/eigen/bench/btl/libs/blitz/btl_blitz.cpp index 16d2b595..ab4d1a30 100644 --- a/filmulator-gui/core/nlmeans/eigen/bench/btl/libs/blitz/btl_blitz.cpp +++ b/filmulator-gui/core/nlmeans/eigen/bench/btl/libs/blitz/btl_blitz.cpp @@ -17,35 +17,33 @@ // along with this program; if not, write to the Free Software // Foundation, Inc., 59 Temple Place - Suite 330, Boston, MA 02111-1307, USA. // -#include "utilities.h" -#include "blitz_interface.hh" -#include "blitz_LU_solve_interface.hh" -#include "bench.hh" -#include "action_matrix_vector_product.hh" -#include "action_matrix_matrix_product.hh" -#include "action_axpy.hh" -#include "action_lu_solve.hh" -#include "action_ata_product.hh" #include "action_aat_product.hh" +#include "action_ata_product.hh" #include "action_atv_product.hh" +#include "action_axpy.hh" +#include "action_lu_solve.hh" +#include "action_matrix_matrix_product.hh" +#include "action_matrix_vector_product.hh" +#include "bench.hh" +#include "blitz_LU_solve_interface.hh" +#include "blitz_interface.hh" +#include "utilities.h" BTL_MAIN; int main() { - bench > >(MIN_MV,MAX_MV,NB_POINT); - bench > >(MIN_MV,MAX_MV,NB_POINT); + bench>>(MIN_MV, MAX_MV, NB_POINT); + bench>>(MIN_MV, MAX_MV, NB_POINT); - bench > >(MIN_MM,MAX_MM,NB_POINT); - bench > >(MIN_MM,MAX_MM,NB_POINT); - bench > >(MIN_MM,MAX_MM,NB_POINT); + bench>>(MIN_MM, MAX_MM, NB_POINT); + bench>>(MIN_MM, MAX_MM, NB_POINT); + bench>>(MIN_MM, MAX_MM, NB_POINT); - bench > >(MIN_AXPY,MAX_AXPY,NB_POINT); + bench>>(MIN_AXPY, MAX_AXPY, NB_POINT); - //bench > >(MIN_LU,MAX_LU,NB_POINT); + // bench > >(MIN_LU,MAX_LU,NB_POINT); return 0; } - - diff --git a/filmulator-gui/core/nlmeans/eigen/bench/btl/libs/blitz/btl_tiny_blitz.cpp b/filmulator-gui/core/nlmeans/eigen/bench/btl/libs/blitz/btl_tiny_blitz.cpp index 9fddde75..9fed4d95 100644 --- a/filmulator-gui/core/nlmeans/eigen/bench/btl/libs/blitz/btl_tiny_blitz.cpp +++ b/filmulator-gui/core/nlmeans/eigen/bench/btl/libs/blitz/btl_tiny_blitz.cpp @@ -17,22 +17,20 @@ // along with this program; if not, write to the Free Software // Foundation, Inc., 59 Temple Place - Suite 330, Boston, MA 02111-1307, USA. // -#include "utilities.h" -#include "tiny_blitz_interface.hh" -#include "static/bench_static.hh" -#include "action_matrix_vector_product.hh" -#include "action_matrix_matrix_product.hh" #include "action_axpy.hh" +#include "action_matrix_matrix_product.hh" +#include "action_matrix_vector_product.hh" +#include "static/bench_static.hh" +#include "tiny_blitz_interface.hh" +#include "utilities.h" BTL_MAIN; int main() { - bench_static(); - bench_static(); - bench_static(); + bench_static(); + bench_static(); + bench_static(); return 0; } - - diff --git a/filmulator-gui/core/nlmeans/eigen/bench/btl/libs/eigen2/btl_tiny_eigen2.cpp b/filmulator-gui/core/nlmeans/eigen/bench/btl/libs/eigen2/btl_tiny_eigen2.cpp index d1515be8..4d3d60db 100644 --- a/filmulator-gui/core/nlmeans/eigen/bench/btl/libs/eigen2/btl_tiny_eigen2.cpp +++ b/filmulator-gui/core/nlmeans/eigen/bench/btl/libs/eigen2/btl_tiny_eigen2.cpp @@ -15,32 +15,30 @@ // along with this program; if not, write to the Free Software // Foundation, Inc., 59 Temple Place - Suite 330, Boston, MA 02111-1307, USA. // -#include "utilities.h" -#include "eigen3_interface.hh" -#include "static/bench_static.hh" -#include "action_matrix_vector_product.hh" -#include "action_matrix_matrix_product.hh" -#include "action_axpy.hh" -#include "action_lu_solve.hh" -#include "action_ata_product.hh" #include "action_aat_product.hh" +#include "action_ata_product.hh" #include "action_atv_product.hh" +#include "action_axpy.hh" #include "action_cholesky.hh" +#include "action_lu_solve.hh" +#include "action_matrix_matrix_product.hh" +#include "action_matrix_vector_product.hh" #include "action_trisolve.hh" +#include "eigen3_interface.hh" +#include "static/bench_static.hh" +#include "utilities.h" BTL_MAIN; int main() { - bench_static(); - bench_static(); - bench_static(); - bench_static(); - bench_static(); - bench_static(); + bench_static(); + bench_static(); + bench_static(); + bench_static(); + bench_static(); + bench_static(); return 0; } - - diff --git a/filmulator-gui/core/nlmeans/eigen/bench/btl/libs/eigen2/main_adv.cpp b/filmulator-gui/core/nlmeans/eigen/bench/btl/libs/eigen2/main_adv.cpp index fe336892..67a9ef85 100644 --- a/filmulator-gui/core/nlmeans/eigen/bench/btl/libs/eigen2/main_adv.cpp +++ b/filmulator-gui/core/nlmeans/eigen/bench/btl/libs/eigen2/main_adv.cpp @@ -15,30 +15,28 @@ // along with this program; if not, write to the Free Software // Foundation, Inc., 59 Temple Place - Suite 330, Boston, MA 02111-1307, USA. // -#include "utilities.h" -#include "eigen2_interface.hh" -#include "bench.hh" -#include "action_trisolve.hh" -#include "action_trisolve_matrix.hh" #include "action_cholesky.hh" #include "action_hessenberg.hh" #include "action_lu_decomp.hh" +#include "action_trisolve.hh" +#include "action_trisolve_matrix.hh" +#include "bench.hh" +#include "eigen2_interface.hh" +#include "utilities.h" // #include "action_partial_lu.hh" BTL_MAIN; int main() { - bench > >(MIN_MM,MAX_MM,NB_POINT); - bench > >(MIN_MM,MAX_MM,NB_POINT); - bench > >(MIN_MM,MAX_MM,NB_POINT); - bench > >(MIN_MM,MAX_MM,NB_POINT); -// bench > >(MIN_MM,MAX_MM,NB_POINT); + bench>>(MIN_MM, MAX_MM, NB_POINT); + bench>>(MIN_MM, MAX_MM, NB_POINT); + bench>>(MIN_MM, MAX_MM, NB_POINT); + bench>>(MIN_MM, MAX_MM, NB_POINT); + // bench > >(MIN_MM,MAX_MM,NB_POINT); - bench > >(MIN_MM,MAX_MM,NB_POINT); - bench > >(MIN_MM,MAX_MM,NB_POINT); + bench>>(MIN_MM, MAX_MM, NB_POINT); + bench>>(MIN_MM, MAX_MM, NB_POINT); return 0; } - - diff --git a/filmulator-gui/core/nlmeans/eigen/bench/btl/libs/eigen2/main_linear.cpp b/filmulator-gui/core/nlmeans/eigen/bench/btl/libs/eigen2/main_linear.cpp index c17d16c0..596deb1c 100644 --- a/filmulator-gui/core/nlmeans/eigen/bench/btl/libs/eigen2/main_linear.cpp +++ b/filmulator-gui/core/nlmeans/eigen/bench/btl/libs/eigen2/main_linear.cpp @@ -15,20 +15,18 @@ // along with this program; if not, write to the Free Software // Foundation, Inc., 59 Temple Place - Suite 330, Boston, MA 02111-1307, USA. // -#include "utilities.h" -#include "eigen2_interface.hh" -#include "bench.hh" #include "basic_actions.hh" +#include "bench.hh" +#include "eigen2_interface.hh" +#include "utilities.h" BTL_MAIN; int main() { - bench > >(MIN_AXPY,MAX_AXPY,NB_POINT); - bench > >(MIN_AXPY,MAX_AXPY,NB_POINT); - + bench>>(MIN_AXPY, MAX_AXPY, NB_POINT); + bench>>(MIN_AXPY, MAX_AXPY, NB_POINT); + return 0; } - - diff --git a/filmulator-gui/core/nlmeans/eigen/bench/btl/libs/eigen2/main_matmat.cpp b/filmulator-gui/core/nlmeans/eigen/bench/btl/libs/eigen2/main_matmat.cpp index cd9dc9cb..f52e85bc 100644 --- a/filmulator-gui/core/nlmeans/eigen/bench/btl/libs/eigen2/main_matmat.cpp +++ b/filmulator-gui/core/nlmeans/eigen/bench/btl/libs/eigen2/main_matmat.cpp @@ -15,21 +15,19 @@ // along with this program; if not, write to the Free Software // Foundation, Inc., 59 Temple Place - Suite 330, Boston, MA 02111-1307, USA. // -#include "utilities.h" -#include "eigen2_interface.hh" -#include "bench.hh" #include "basic_actions.hh" +#include "bench.hh" +#include "eigen2_interface.hh" +#include "utilities.h" BTL_MAIN; int main() { - bench > >(MIN_MM,MAX_MM,NB_POINT); -// bench > >(MIN_MM,MAX_MM,NB_POINT); - bench > >(MIN_MM,MAX_MM,NB_POINT); -// bench > >(MIN_MM,MAX_MM,NB_POINT); + bench>>(MIN_MM, MAX_MM, NB_POINT); + // bench > >(MIN_MM,MAX_MM,NB_POINT); + bench>>(MIN_MM, MAX_MM, NB_POINT); + // bench > >(MIN_MM,MAX_MM,NB_POINT); return 0; } - - diff --git a/filmulator-gui/core/nlmeans/eigen/bench/btl/libs/eigen2/main_vecmat.cpp b/filmulator-gui/core/nlmeans/eigen/bench/btl/libs/eigen2/main_vecmat.cpp index 8b66cd2d..92c91869 100644 --- a/filmulator-gui/core/nlmeans/eigen/bench/btl/libs/eigen2/main_vecmat.cpp +++ b/filmulator-gui/core/nlmeans/eigen/bench/btl/libs/eigen2/main_vecmat.cpp @@ -15,22 +15,20 @@ // along with this program; if not, write to the Free Software // Foundation, Inc., 59 Temple Place - Suite 330, Boston, MA 02111-1307, USA. // -#include "utilities.h" -#include "eigen2_interface.hh" -#include "bench.hh" #include "basic_actions.hh" +#include "bench.hh" +#include "eigen2_interface.hh" +#include "utilities.h" BTL_MAIN; int main() { - bench > >(MIN_MV,MAX_MV,NB_POINT); - bench > >(MIN_MV,MAX_MV,NB_POINT); -// bench > >(MIN_MV,MAX_MV,NB_POINT); -// bench > >(MIN_MV,MAX_MV,NB_POINT); -// bench > >(MIN_MV,MAX_MV,NB_POINT); + bench>>(MIN_MV, MAX_MV, NB_POINT); + bench>>(MIN_MV, MAX_MV, NB_POINT); + // bench > >(MIN_MV,MAX_MV,NB_POINT); + // bench > >(MIN_MV,MAX_MV,NB_POINT); + // bench > >(MIN_MV,MAX_MV,NB_POINT); return 0; } - - diff --git a/filmulator-gui/core/nlmeans/eigen/bench/btl/libs/eigen3/btl_tiny_eigen3.cpp b/filmulator-gui/core/nlmeans/eigen/bench/btl/libs/eigen3/btl_tiny_eigen3.cpp index d1515be8..4d3d60db 100644 --- a/filmulator-gui/core/nlmeans/eigen/bench/btl/libs/eigen3/btl_tiny_eigen3.cpp +++ b/filmulator-gui/core/nlmeans/eigen/bench/btl/libs/eigen3/btl_tiny_eigen3.cpp @@ -15,32 +15,30 @@ // along with this program; if not, write to the Free Software // Foundation, Inc., 59 Temple Place - Suite 330, Boston, MA 02111-1307, USA. // -#include "utilities.h" -#include "eigen3_interface.hh" -#include "static/bench_static.hh" -#include "action_matrix_vector_product.hh" -#include "action_matrix_matrix_product.hh" -#include "action_axpy.hh" -#include "action_lu_solve.hh" -#include "action_ata_product.hh" #include "action_aat_product.hh" +#include "action_ata_product.hh" #include "action_atv_product.hh" +#include "action_axpy.hh" #include "action_cholesky.hh" +#include "action_lu_solve.hh" +#include "action_matrix_matrix_product.hh" +#include "action_matrix_vector_product.hh" #include "action_trisolve.hh" +#include "eigen3_interface.hh" +#include "static/bench_static.hh" +#include "utilities.h" BTL_MAIN; int main() { - bench_static(); - bench_static(); - bench_static(); - bench_static(); - bench_static(); - bench_static(); + bench_static(); + bench_static(); + bench_static(); + bench_static(); + bench_static(); + bench_static(); return 0; } - - diff --git a/filmulator-gui/core/nlmeans/eigen/bench/btl/libs/eigen3/main_adv.cpp b/filmulator-gui/core/nlmeans/eigen/bench/btl/libs/eigen3/main_adv.cpp index 95865357..41f76814 100644 --- a/filmulator-gui/core/nlmeans/eigen/bench/btl/libs/eigen3/main_adv.cpp +++ b/filmulator-gui/core/nlmeans/eigen/bench/btl/libs/eigen3/main_adv.cpp @@ -15,30 +15,28 @@ // along with this program; if not, write to the Free Software // Foundation, Inc., 59 Temple Place - Suite 330, Boston, MA 02111-1307, USA. // -#include "utilities.h" -#include "eigen3_interface.hh" -#include "bench.hh" -#include "action_trisolve.hh" -#include "action_trisolve_matrix.hh" #include "action_cholesky.hh" #include "action_hessenberg.hh" #include "action_lu_decomp.hh" #include "action_partial_lu.hh" +#include "action_trisolve.hh" +#include "action_trisolve_matrix.hh" +#include "bench.hh" +#include "eigen3_interface.hh" +#include "utilities.h" BTL_MAIN; int main() { - bench > >(MIN_LU,MAX_LU,NB_POINT); - bench > >(MIN_LU,MAX_LU,NB_POINT); - bench > >(MIN_LU,MAX_LU,NB_POINT); -// bench > >(MIN_LU,MAX_LU,NB_POINT); - bench > >(MIN_LU,MAX_LU,NB_POINT); + bench>>(MIN_LU, MAX_LU, NB_POINT); + bench>>(MIN_LU, MAX_LU, NB_POINT); + bench>>(MIN_LU, MAX_LU, NB_POINT); + // bench > >(MIN_LU,MAX_LU,NB_POINT); + bench>>(MIN_LU, MAX_LU, NB_POINT); -// bench > >(MIN_LU,MAX_LU,NB_POINT); - bench > >(MIN_LU,MAX_LU,NB_POINT); + // bench > >(MIN_LU,MAX_LU,NB_POINT); + bench>>(MIN_LU, MAX_LU, NB_POINT); return 0; } - - diff --git a/filmulator-gui/core/nlmeans/eigen/bench/btl/libs/eigen3/main_linear.cpp b/filmulator-gui/core/nlmeans/eigen/bench/btl/libs/eigen3/main_linear.cpp index e8538b7d..e0b76ece 100644 --- a/filmulator-gui/core/nlmeans/eigen/bench/btl/libs/eigen3/main_linear.cpp +++ b/filmulator-gui/core/nlmeans/eigen/bench/btl/libs/eigen3/main_linear.cpp @@ -15,21 +15,19 @@ // along with this program; if not, write to the Free Software // Foundation, Inc., 59 Temple Place - Suite 330, Boston, MA 02111-1307, USA. // -#include "utilities.h" -#include "eigen3_interface.hh" -#include "bench.hh" #include "basic_actions.hh" +#include "bench.hh" +#include "eigen3_interface.hh" +#include "utilities.h" BTL_MAIN; int main() { - bench > >(MIN_AXPY,MAX_AXPY,NB_POINT); - bench > >(MIN_AXPY,MAX_AXPY,NB_POINT); - bench > >(MIN_AXPY,MAX_AXPY,NB_POINT); - + bench>>(MIN_AXPY, MAX_AXPY, NB_POINT); + bench>>(MIN_AXPY, MAX_AXPY, NB_POINT); + bench>>(MIN_AXPY, MAX_AXPY, NB_POINT); + return 0; } - - diff --git a/filmulator-gui/core/nlmeans/eigen/bench/btl/libs/eigen3/main_matmat.cpp b/filmulator-gui/core/nlmeans/eigen/bench/btl/libs/eigen3/main_matmat.cpp index 926fa2b0..e5a935c3 100644 --- a/filmulator-gui/core/nlmeans/eigen/bench/btl/libs/eigen3/main_matmat.cpp +++ b/filmulator-gui/core/nlmeans/eigen/bench/btl/libs/eigen3/main_matmat.cpp @@ -15,21 +15,19 @@ // along with this program; if not, write to the Free Software // Foundation, Inc., 59 Temple Place - Suite 330, Boston, MA 02111-1307, USA. // -#include "utilities.h" -#include "eigen3_interface.hh" -#include "bench.hh" #include "basic_actions.hh" +#include "bench.hh" +#include "eigen3_interface.hh" +#include "utilities.h" BTL_MAIN; int main() { - bench > >(MIN_MM,MAX_MM,NB_POINT); -// bench > >(MIN_MM,MAX_MM,NB_POINT); - bench > >(MIN_MM,MAX_MM,NB_POINT); - bench > >(MIN_MM,MAX_MM,NB_POINT); + bench>>(MIN_MM, MAX_MM, NB_POINT); + // bench > >(MIN_MM,MAX_MM,NB_POINT); + bench>>(MIN_MM, MAX_MM, NB_POINT); + bench>>(MIN_MM, MAX_MM, NB_POINT); return 0; } - - diff --git a/filmulator-gui/core/nlmeans/eigen/bench/btl/libs/eigen3/main_vecmat.cpp b/filmulator-gui/core/nlmeans/eigen/bench/btl/libs/eigen3/main_vecmat.cpp index 0dda444c..39333d4c 100644 --- a/filmulator-gui/core/nlmeans/eigen/bench/btl/libs/eigen3/main_vecmat.cpp +++ b/filmulator-gui/core/nlmeans/eigen/bench/btl/libs/eigen3/main_vecmat.cpp @@ -15,22 +15,20 @@ // along with this program; if not, write to the Free Software // Foundation, Inc., 59 Temple Place - Suite 330, Boston, MA 02111-1307, USA. // -#include "utilities.h" -#include "eigen3_interface.hh" -#include "bench.hh" #include "basic_actions.hh" +#include "bench.hh" +#include "eigen3_interface.hh" +#include "utilities.h" BTL_MAIN; int main() { - bench > >(MIN_MV,MAX_MV,NB_POINT); - bench > >(MIN_MV,MAX_MV,NB_POINT); - bench > >(MIN_MV,MAX_MV,NB_POINT); - bench > >(MIN_MV,MAX_MV,NB_POINT); - bench > >(MIN_MV,MAX_MV,NB_POINT); + bench>>(MIN_MV, MAX_MV, NB_POINT); + bench>>(MIN_MV, MAX_MV, NB_POINT); + bench>>(MIN_MV, MAX_MV, NB_POINT); + bench>>(MIN_MV, MAX_MV, NB_POINT); + bench>>(MIN_MV, MAX_MV, NB_POINT); return 0; } - - diff --git a/filmulator-gui/core/nlmeans/eigen/bench/btl/libs/gmm/main.cpp b/filmulator-gui/core/nlmeans/eigen/bench/btl/libs/gmm/main.cpp index 1f0c051e..e0d93556 100644 --- a/filmulator-gui/core/nlmeans/eigen/bench/btl/libs/gmm/main.cpp +++ b/filmulator-gui/core/nlmeans/eigen/bench/btl/libs/gmm/main.cpp @@ -15,37 +15,35 @@ // along with this program; if not, write to the Free Software // Foundation, Inc., 59 Temple Place - Suite 330, Boston, MA 02111-1307, USA. // -#include "utilities.h" -#include "gmm_interface.hh" -#include "bench.hh" -#include "basic_actions.hh" #include "action_hessenberg.hh" #include "action_partial_lu.hh" +#include "basic_actions.hh" +#include "bench.hh" +#include "gmm_interface.hh" +#include "utilities.h" BTL_MAIN; int main() { - bench > >(MIN_AXPY,MAX_AXPY,NB_POINT); - bench > >(MIN_AXPY,MAX_AXPY,NB_POINT); + bench>>(MIN_AXPY, MAX_AXPY, NB_POINT); + bench>>(MIN_AXPY, MAX_AXPY, NB_POINT); + + bench>>(MIN_MV, MAX_MV, NB_POINT); + bench>>(MIN_MV, MAX_MV, NB_POINT); - bench > >(MIN_MV,MAX_MV,NB_POINT); - bench > >(MIN_MV,MAX_MV,NB_POINT); + bench>>(MIN_MM, MAX_MM, NB_POINT); + // bench > >(MIN_MM,MAX_MM,NB_POINT); + // bench > >(MIN_MM,MAX_MM,NB_POINT); - bench > >(MIN_MM,MAX_MM,NB_POINT); -// bench > >(MIN_MM,MAX_MM,NB_POINT); -// bench > >(MIN_MM,MAX_MM,NB_POINT); + bench>>(MIN_MM, MAX_MM, NB_POINT); + // bench > >(MIN_LU,MAX_LU,NB_POINT); - bench > >(MIN_MM,MAX_MM,NB_POINT); - //bench > >(MIN_LU,MAX_LU,NB_POINT); + bench>>(MIN_MM, MAX_MM, NB_POINT); - bench > >(MIN_MM,MAX_MM,NB_POINT); - - bench > >(MIN_MM,MAX_MM,NB_POINT); - bench > >(MIN_MM,MAX_MM,NB_POINT); + bench>>(MIN_MM, MAX_MM, NB_POINT); + bench>>(MIN_MM, MAX_MM, NB_POINT); return 0; } - - diff --git a/filmulator-gui/core/nlmeans/eigen/bench/btl/libs/mtl4/main.cpp b/filmulator-gui/core/nlmeans/eigen/bench/btl/libs/mtl4/main.cpp index 96fcfb9c..087311ef 100644 --- a/filmulator-gui/core/nlmeans/eigen/bench/btl/libs/mtl4/main.cpp +++ b/filmulator-gui/core/nlmeans/eigen/bench/btl/libs/mtl4/main.cpp @@ -15,11 +15,11 @@ // along with this program; if not, write to the Free Software // Foundation, Inc., 59 Temple Place - Suite 330, Boston, MA 02111-1307, USA. // -#include "utilities.h" -#include "mtl4_interface.hh" -#include "bench.hh" -#include "basic_actions.hh" #include "action_cholesky.hh" +#include "basic_actions.hh" +#include "bench.hh" +#include "mtl4_interface.hh" +#include "utilities.h" // #include "action_lu_decomp.hh" BTL_MAIN; @@ -27,20 +27,18 @@ BTL_MAIN; int main() { - bench > >(MIN_AXPY,MAX_AXPY,NB_POINT); - bench > >(MIN_AXPY,MAX_AXPY,NB_POINT); + bench>>(MIN_AXPY, MAX_AXPY, NB_POINT); + bench>>(MIN_AXPY, MAX_AXPY, NB_POINT); - bench > >(MIN_MV,MAX_MV,NB_POINT); - bench > >(MIN_MV,MAX_MV,NB_POINT); - bench > >(MIN_MM,MAX_MM,NB_POINT); -// bench > >(MIN_MM,MAX_MM,NB_POINT); -// bench > >(MIN_MM,MAX_MM,NB_POINT); + bench>>(MIN_MV, MAX_MV, NB_POINT); + bench>>(MIN_MV, MAX_MV, NB_POINT); + bench>>(MIN_MM, MAX_MM, NB_POINT); + // bench > >(MIN_MM,MAX_MM,NB_POINT); + // bench > >(MIN_MM,MAX_MM,NB_POINT); - bench > >(MIN_MM,MAX_MM,NB_POINT); -// bench > >(MIN_MM,MAX_MM,NB_POINT); -// bench > >(MIN_MM,MAX_MM,NB_POINT); + bench>>(MIN_MM, MAX_MM, NB_POINT); + // bench > >(MIN_MM,MAX_MM,NB_POINT); + // bench > >(MIN_MM,MAX_MM,NB_POINT); return 0; } - - diff --git a/filmulator-gui/core/nlmeans/eigen/bench/btl/libs/tensors/main_linear.cpp b/filmulator-gui/core/nlmeans/eigen/bench/btl/libs/tensors/main_linear.cpp index e257f1e7..466e639c 100644 --- a/filmulator-gui/core/nlmeans/eigen/bench/btl/libs/tensors/main_linear.cpp +++ b/filmulator-gui/core/nlmeans/eigen/bench/btl/libs/tensors/main_linear.cpp @@ -7,17 +7,17 @@ // Public License v. 2.0. If a copy of the MPL was not distributed // with this file, You can obtain one at http://mozilla.org/MPL/2.0/. -#include "utilities.h" -#include "tensor_interface.hh" -#include "bench.hh" #include "basic_actions.hh" +#include "bench.hh" +#include "tensor_interface.hh" +#include "utilities.h" BTL_MAIN; int main() { - bench > >(MIN_AXPY,MAX_AXPY,NB_POINT); - bench > >(MIN_AXPY,MAX_AXPY,NB_POINT); + bench>>(MIN_AXPY, MAX_AXPY, NB_POINT); + bench>>(MIN_AXPY, MAX_AXPY, NB_POINT); return 0; } diff --git a/filmulator-gui/core/nlmeans/eigen/bench/btl/libs/tensors/main_matmat.cpp b/filmulator-gui/core/nlmeans/eigen/bench/btl/libs/tensors/main_matmat.cpp index 675fcfc6..ca353add 100644 --- a/filmulator-gui/core/nlmeans/eigen/bench/btl/libs/tensors/main_matmat.cpp +++ b/filmulator-gui/core/nlmeans/eigen/bench/btl/libs/tensors/main_matmat.cpp @@ -6,16 +6,16 @@ // Public License v. 2.0. If a copy of the MPL was not distributed // with this file, You can obtain one at http://mozilla.org/MPL/2.0/. // -#include "utilities.h" -#include "tensor_interface.hh" -#include "bench.hh" #include "basic_actions.hh" +#include "bench.hh" +#include "tensor_interface.hh" +#include "utilities.h" BTL_MAIN; int main() { - bench > >(MIN_MM,MAX_MM,NB_POINT); + bench>>(MIN_MM, MAX_MM, NB_POINT); return 0; } diff --git a/filmulator-gui/core/nlmeans/eigen/bench/btl/libs/tensors/main_vecmat.cpp b/filmulator-gui/core/nlmeans/eigen/bench/btl/libs/tensors/main_vecmat.cpp index 1af00c81..0f65b800 100644 --- a/filmulator-gui/core/nlmeans/eigen/bench/btl/libs/tensors/main_vecmat.cpp +++ b/filmulator-gui/core/nlmeans/eigen/bench/btl/libs/tensors/main_vecmat.cpp @@ -6,16 +6,16 @@ // Public License v. 2.0. If a copy of the MPL was not distributed // with this file, You can obtain one at http://mozilla.org/MPL/2.0/. // -#include "utilities.h" -#include "tensor_interface.hh" -#include "bench.hh" #include "basic_actions.hh" +#include "bench.hh" +#include "tensor_interface.hh" +#include "utilities.h" BTL_MAIN; int main() { - bench > >(MIN_MV,MAX_MV,NB_POINT); + bench>>(MIN_MV, MAX_MV, NB_POINT); return 0; } diff --git a/filmulator-gui/core/nlmeans/eigen/bench/btl/libs/tvmet/main.cpp b/filmulator-gui/core/nlmeans/eigen/bench/btl/libs/tvmet/main.cpp index 633215c4..d707dcae 100644 --- a/filmulator-gui/core/nlmeans/eigen/bench/btl/libs/tvmet/main.cpp +++ b/filmulator-gui/core/nlmeans/eigen/bench/btl/libs/tvmet/main.cpp @@ -17,24 +17,22 @@ // along with this program; if not, write to the Free Software // Foundation, Inc., 59 Temple Place - Suite 330, Boston, MA 02111-1307, USA. // -#include "utilities.h" -#include "tvmet_interface.hh" -#include "static/bench_static.hh" -#include "action_matrix_vector_product.hh" -#include "action_matrix_matrix_product.hh" #include "action_atv_product.hh" #include "action_axpy.hh" +#include "action_matrix_matrix_product.hh" +#include "action_matrix_vector_product.hh" +#include "static/bench_static.hh" +#include "tvmet_interface.hh" +#include "utilities.h" BTL_MAIN; int main() { - bench_static(); - bench_static(); - bench_static(); - bench_static(); + bench_static(); + bench_static(); + bench_static(); + bench_static(); return 0; } - - diff --git a/filmulator-gui/core/nlmeans/eigen/bench/btl/libs/ublas/main.cpp b/filmulator-gui/core/nlmeans/eigen/bench/btl/libs/ublas/main.cpp index e2e77ee1..d10eb795 100644 --- a/filmulator-gui/core/nlmeans/eigen/bench/btl/libs/ublas/main.cpp +++ b/filmulator-gui/core/nlmeans/eigen/bench/btl/libs/ublas/main.cpp @@ -17,28 +17,26 @@ // along with this program; if not, write to the Free Software // Foundation, Inc., 59 Temple Place - Suite 330, Boston, MA 02111-1307, USA. // -#include "utilities.h" -#include "ublas_interface.hh" -#include "bench.hh" #include "basic_actions.hh" +#include "bench.hh" +#include "ublas_interface.hh" +#include "utilities.h" BTL_MAIN; int main() { - bench > >(MIN_AXPY,MAX_AXPY,NB_POINT); - bench > >(MIN_AXPY,MAX_AXPY,NB_POINT); + bench>>(MIN_AXPY, MAX_AXPY, NB_POINT); + bench>>(MIN_AXPY, MAX_AXPY, NB_POINT); - bench > >(MIN_MV,MAX_MV,NB_POINT); - bench > >(MIN_MV,MAX_MV,NB_POINT); + bench>>(MIN_MV, MAX_MV, NB_POINT); + bench>>(MIN_MV, MAX_MV, NB_POINT); - bench > >(MIN_MM,MAX_MM,NB_POINT); -// bench > >(MIN_MM,MAX_MM,NB_POINT); -// bench > >(MIN_MM,MAX_MM,NB_POINT); + bench>>(MIN_MM, MAX_MM, NB_POINT); + // bench > >(MIN_MM,MAX_MM,NB_POINT); + // bench > >(MIN_MM,MAX_MM,NB_POINT); - bench > >(MIN_MM,MAX_MM,NB_POINT); + bench>>(MIN_MM, MAX_MM, NB_POINT); return 0; } - - diff --git a/filmulator-gui/core/nlmeans/eigen/bench/check_cache_queries.cpp b/filmulator-gui/core/nlmeans/eigen/bench/check_cache_queries.cpp index 029d44cf..0be189a3 100644 --- a/filmulator-gui/core/nlmeans/eigen/bench/check_cache_queries.cpp +++ b/filmulator-gui/core/nlmeans/eigen/bench/check_cache_queries.cpp @@ -1,20 +1,20 @@ #define EIGEN_INTERNAL_DEBUG_CACHE_QUERY -#include #include "../Eigen/Core" +#include using namespace Eigen; using namespace std; -#define DUMP_CPUID(CODE) {\ - int abcd[4]; \ - abcd[0] = abcd[1] = abcd[2] = abcd[3] = 0;\ - EIGEN_CPUID(abcd, CODE, 0); \ - std::cout << "The code " << CODE << " gives " \ - << (int*)(abcd[0]) << " " << (int*)(abcd[1]) << " " \ - << (int*)(abcd[2]) << " " << (int*)(abcd[3]) << " " << std::endl; \ +#define DUMP_CPUID(CODE) \ + { \ + int abcd[4]; \ + abcd[0] = abcd[1] = abcd[2] = abcd[3] = 0; \ + EIGEN_CPUID(abcd, CODE, 0); \ + std::cout << "The code " << CODE << " gives " << (int *)(abcd[0]) << " " << (int *)(abcd[1]) << " " \ + << (int *)(abcd[2]) << " " << (int *)(abcd[3]) << " " << std::endl; \ } - + int main() { cout << "Eigen's L1 = " << internal::queryL1CacheSize() << endl; @@ -22,15 +22,15 @@ int main() int l1, l2, l3; internal::queryCacheSizes(l1, l2, l3); cout << "Eigen's L1, L2, L3 = " << l1 << " " << l2 << " " << l3 << endl; - - #ifdef EIGEN_CPUID + +#ifdef EIGEN_CPUID int abcd[4]; int string[8]; - char* string_char = (char*)(string); + char *string_char = (char *)(string); // vendor ID - EIGEN_CPUID(abcd,0x0,0); + EIGEN_CPUID(abcd, 0x0, 0); string[0] = abcd[1]; string[1] = abcd[3]; string[2] = abcd[2]; @@ -42,32 +42,30 @@ int main() internal::queryCacheSizes_intel_codes(l1, l2, l3); cout << "Eigen's intel codes L1, L2, L3 = " << l1 << " " << l2 << " " << l3 << endl; - if(max_funcs>=4) - { + if (max_funcs >= 4) { internal::queryCacheSizes_intel_direct(l1, l2, l3); cout << "Eigen's intel direct L1, L2, L3 = " << l1 << " " << l2 << " " << l3 << endl; } internal::queryCacheSizes_amd(l1, l2, l3); cout << "Eigen's amd L1, L2, L3 = " << l1 << " " << l2 << " " << l3 << endl; cout << endl; - + // dump Intel direct method - if(max_funcs>=4) - { + if (max_funcs >= 4) { l1 = l2 = l3 = 0; int cache_id = 0; int cache_type = 0; do { abcd[0] = abcd[1] = abcd[2] = abcd[3] = 0; - EIGEN_CPUID(abcd,0x4,cache_id); - cache_type = (abcd[0] & 0x0F) >> 0; - int cache_level = (abcd[0] & 0xE0) >> 5; // A[7:5] - int ways = (abcd[1] & 0xFFC00000) >> 22; // B[31:22] - int partitions = (abcd[1] & 0x003FF000) >> 12; // B[21:12] - int line_size = (abcd[1] & 0x00000FFF) >> 0; // B[11:0] - int sets = (abcd[2]); // C[31:0] - int cache_size = (ways+1) * (partitions+1) * (line_size+1) * (sets+1); - + EIGEN_CPUID(abcd, 0x4, cache_id); + cache_type = (abcd[0] & 0x0F) >> 0; + int cache_level = (abcd[0] & 0xE0) >> 5;// A[7:5] + int ways = (abcd[1] & 0xFFC00000) >> 22;// B[31:22] + int partitions = (abcd[1] & 0x003FF000) >> 12;// B[21:12] + int line_size = (abcd[1] & 0x00000FFF) >> 0;// B[11:0] + int sets = (abcd[2]);// C[31:0] + int cache_size = (ways + 1) * (partitions + 1) * (line_size + 1) * (sets + 1); + cout << "cache[" << cache_id << "].type = " << cache_type << "\n"; cout << "cache[" << cache_id << "].level = " << cache_level << "\n"; cout << "cache[" << cache_id << "].ways = " << ways << "\n"; @@ -75,15 +73,14 @@ int main() cout << "cache[" << cache_id << "].line_size = " << line_size << "\n"; cout << "cache[" << cache_id << "].sets = " << sets << "\n"; cout << "cache[" << cache_id << "].size = " << cache_size << "\n"; - + cache_id++; - } while(cache_type>0 && cache_id<16); + } while (cache_type > 0 && cache_id < 16); } - + // dump everything - std::cout << endl <<"Raw dump:" << endl; - for(int i=0; i #include "BenchTimer.h" #include +#include #include -#include -#include #include +#include +#include using namespace Eigen; -std::map > results; +std::map> results; std::vector labels; std::vector sizes; -template -EIGEN_DONT_INLINE -void compute_norm_equation(Solver &solver, const MatrixType &A) { - if(A.rows()!=A.cols()) - solver.compute(A.transpose()*A); +template +EIGEN_DONT_INLINE void compute_norm_equation(Solver &solver, const MatrixType &A) +{ + if (A.rows() != A.cols()) + solver.compute(A.transpose() * A); else solver.compute(A); } -template -EIGEN_DONT_INLINE -void compute(Solver &solver, const MatrixType &A) { +template EIGEN_DONT_INLINE void compute(Solver &solver, const MatrixType &A) +{ solver.compute(A); } -template -void bench(int id, int rows, int size = Size) +template void bench(int id, int rows, int size = Size) { - typedef Matrix Mat; - typedef Matrix MatDyn; - typedef Matrix MatSquare; - Mat A(rows,size); + typedef Matrix Mat; + typedef Matrix MatDyn; + typedef Matrix MatSquare; + Mat A(rows, size); A.setRandom(); - if(rows==size) - A = A*A.adjoint(); + if (rows == size) A = A * A.adjoint(); BenchTimer t_llt, t_ldlt, t_lu, t_fplu, t_qr, t_cpqr, t_cod, t_fpqr, t_jsvd, t_bdcsvd; - int svd_opt = ComputeThinU|ComputeThinV; - + int svd_opt = ComputeThinU | ComputeThinV; + int tries = 5; - int rep = 1000/size; - if(rep==0) rep = 1; -// rep = rep*rep; - + int rep = 1000 / size; + if (rep == 0) rep = 1; + // rep = rep*rep; + LLT llt(size); LDLT ldlt(size); PartialPivLU lu(size); - FullPivLU fplu(size,size); - HouseholderQR qr(A.rows(),A.cols()); - ColPivHouseholderQR cpqr(A.rows(),A.cols()); - CompleteOrthogonalDecomposition cod(A.rows(),A.cols()); - FullPivHouseholderQR fpqr(A.rows(),A.cols()); - JacobiSVD jsvd(A.rows(),A.cols()); - BDCSVD bdcsvd(A.rows(),A.cols()); - - BENCH(t_llt, tries, rep, compute_norm_equation(llt,A)); - BENCH(t_ldlt, tries, rep, compute_norm_equation(ldlt,A)); - BENCH(t_lu, tries, rep, compute_norm_equation(lu,A)); - if(size<=1000) - BENCH(t_fplu, tries, rep, compute_norm_equation(fplu,A)); - BENCH(t_qr, tries, rep, compute(qr,A)); - BENCH(t_cpqr, tries, rep, compute(cpqr,A)); - BENCH(t_cod, tries, rep, compute(cod,A)); - if(size*rows<=10000000) - BENCH(t_fpqr, tries, rep, compute(fpqr,A)); - if(size<500) // JacobiSVD is really too slow for too large matrices - BENCH(t_jsvd, tries, rep, jsvd.compute(A,svd_opt)); -// if(size*rows<=20000000) - BENCH(t_bdcsvd, tries, rep, bdcsvd.compute(A,svd_opt)); - + FullPivLU fplu(size, size); + HouseholderQR qr(A.rows(), A.cols()); + ColPivHouseholderQR cpqr(A.rows(), A.cols()); + CompleteOrthogonalDecomposition cod(A.rows(), A.cols()); + FullPivHouseholderQR fpqr(A.rows(), A.cols()); + JacobiSVD jsvd(A.rows(), A.cols()); + BDCSVD bdcsvd(A.rows(), A.cols()); + + BENCH(t_llt, tries, rep, compute_norm_equation(llt, A)); + BENCH(t_ldlt, tries, rep, compute_norm_equation(ldlt, A)); + BENCH(t_lu, tries, rep, compute_norm_equation(lu, A)); + if (size <= 1000) BENCH(t_fplu, tries, rep, compute_norm_equation(fplu, A)); + BENCH(t_qr, tries, rep, compute(qr, A)); + BENCH(t_cpqr, tries, rep, compute(cpqr, A)); + BENCH(t_cod, tries, rep, compute(cod, A)); + if (size * rows <= 10000000) BENCH(t_fpqr, tries, rep, compute(fpqr, A)); + if (size < 500)// JacobiSVD is really too slow for too large matrices + BENCH(t_jsvd, tries, rep, jsvd.compute(A, svd_opt)); + // if(size*rows<=20000000) + BENCH(t_bdcsvd, tries, rep, bdcsvd.compute(A, svd_opt)); + results["LLT"][id] = t_llt.best(); results["LDLT"][id] = t_ldlt.best(); results["PartialPivLU"][id] = t_lu.best(); @@ -97,48 +92,49 @@ int main() labels.push_back("JacobiSVD"); labels.push_back("BDCSVD"); - for(int i=0; i(k,sizes[k](0),sizes[k](1)); + bench(k, sizes[k](0), sizes[k](1)); } cout.width(32); cout << "solver/size"; cout << " "; - for(int k=0; k=1e6) cout << "-"; - else cout << r(k); + if (r(k) >= 1e6) + cout << "-"; + else + cout << r(k); cout << " "; } cout << endl; @@ -147,25 +143,20 @@ int main() // HTML output cout << "" << endl; cout << "" << endl; - for(int k=0; k" << sizes[k](0) << "x" << sizes[k](1) << ""; + for (int k = 0; k < sizes.size(); ++k) cout << " "; cout << "" << endl; - for(int i=0; i"; - ArrayXf r = (results[labels[i]]*100000.f).floor()/100.f; - for(int k=0; k=1e6) cout << ""; - else - { + ArrayXf r = (results[labels[i]] * 100000.f).floor() / 100.f; + for (int k = 0; k < sizes.size(); ++k) { + if (r(k) >= 1e6) + cout << ""; + else { cout << ""; } } @@ -173,14 +164,15 @@ int main() } cout << "
solver/size" << sizes[k](0) << "x" << sizes[k](1) << "
" << labels[i] << "--" << r(k); - if(i>0) - cout << " (x" << numext::round(10.f*results[labels[i]](k)/results["LLT"](k))/10.f << ")"; - if(i<4 && sizes[k](0)!=sizes[k](1)) - cout << " *"; + if (i > 0) cout << " (x" << numext::round(10.f * results[labels[i]](k) / results["LLT"](k)) / 10.f << ")"; + if (i < 4 && sizes[k](0) != sizes[k](1)) cout << " *"; cout << "
" << endl; -// cout << "LLT (ms) " << (results["LLT"]*1000.).format(fmt) << "\n"; -// cout << "LDLT (%) " << (results["LDLT"]/results["LLT"]).format(fmt) << "\n"; -// cout << "PartialPivLU (%) " << (results["PartialPivLU"]/results["LLT"]).format(fmt) << "\n"; -// cout << "FullPivLU (%) " << (results["FullPivLU"]/results["LLT"]).format(fmt) << "\n"; -// cout << "HouseholderQR (%) " << (results["HouseholderQR"]/results["LLT"]).format(fmt) << "\n"; -// cout << "ColPivHouseholderQR (%) " << (results["ColPivHouseholderQR"]/results["LLT"]).format(fmt) << "\n"; -// cout << "CompleteOrthogonalDecomposition (%) " << (results["CompleteOrthogonalDecomposition"]/results["LLT"]).format(fmt) << "\n"; -// cout << "FullPivHouseholderQR (%) " << (results["FullPivHouseholderQR"]/results["LLT"]).format(fmt) << "\n"; -// cout << "JacobiSVD (%) " << (results["JacobiSVD"]/results["LLT"]).format(fmt) << "\n"; -// cout << "BDCSVD (%) " << (results["BDCSVD"]/results["LLT"]).format(fmt) << "\n"; + // cout << "LLT (ms) " << (results["LLT"]*1000.).format(fmt) << "\n"; + // cout << "LDLT (%) " << (results["LDLT"]/results["LLT"]).format(fmt) << "\n"; + // cout << "PartialPivLU (%) " << (results["PartialPivLU"]/results["LLT"]).format(fmt) << "\n"; + // cout << "FullPivLU (%) " << (results["FullPivLU"]/results["LLT"]).format(fmt) << "\n"; + // cout << "HouseholderQR (%) " << (results["HouseholderQR"]/results["LLT"]).format(fmt) << + // "\n"; cout << "ColPivHouseholderQR (%) " << + // (results["ColPivHouseholderQR"]/results["LLT"]).format(fmt) << "\n"; cout << "CompleteOrthogonalDecomposition (%) + // " << (results["CompleteOrthogonalDecomposition"]/results["LLT"]).format(fmt) << "\n"; cout << + // "FullPivHouseholderQR (%) " << (results["FullPivHouseholderQR"]/results["LLT"]).format(fmt) << "\n"; + // cout << "JacobiSVD (%) " << (results["JacobiSVD"]/results["LLT"]).format(fmt) << "\n"; + // cout << "BDCSVD (%) " << (results["BDCSVD"]/results["LLT"]).format(fmt) << "\n"; } diff --git a/filmulator-gui/core/nlmeans/eigen/bench/eig33.cpp b/filmulator-gui/core/nlmeans/eigen/bench/eig33.cpp index 47947a9b..b59feaa8 100644 --- a/filmulator-gui/core/nlmeans/eigen/bench/eig33.cpp +++ b/filmulator-gui/core/nlmeans/eigen/bench/eig33.cpp @@ -9,25 +9,25 @@ // The computeRoots function included in this is based on materials // covered by the following copyright and license: -// +// // Geometric Tools, LLC // Copyright (c) 1998-2010 // Distributed under the Boost Software License, Version 1.0. -// +// // Permission is hereby granted, free of charge, to any person or organization // obtaining a copy of the software and accompanying documentation covered by // this license (the "Software") to use, reproduce, display, distribute, // execute, and transmit the Software, and to prepare derivative works of the // Software, and to permit third-parties to whom the Software is furnished to // do so, all subject to the following: -// +// // The copyright notices in the Software and this entire statement, including // the above license grant, this restriction and the following disclaimer, // must be included in all copies of the Software, in whole or in part, and // all derivative works of the Software, unless such copies or derivative // works are solely in the form of machine-executable object code generated by // a source language processor. -// +// // THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR // IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, // FITNESS FOR A PARTICULAR PURPOSE, TITLE AND NON-INFRINGEMENT. IN NO EVENT @@ -36,130 +36,125 @@ // ARISING FROM, OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER // DEALINGS IN THE SOFTWARE. -#include #include #include #include #include +#include using namespace Eigen; using namespace std; -template -inline void computeRoots(const Matrix& m, Roots& roots) +template inline void computeRoots(const Matrix &m, Roots &roots) { typedef typename Matrix::Scalar Scalar; - const Scalar s_inv3 = 1.0/3.0; + const Scalar s_inv3 = 1.0 / 3.0; const Scalar s_sqrt3 = std::sqrt(Scalar(3.0)); // The characteristic equation is x^3 - c2*x^2 + c1*x - c0 = 0. The // eigenvalues are the roots to this equation, all guaranteed to be // real-valued, because the matrix is symmetric. - Scalar c0 = m(0,0)*m(1,1)*m(2,2) + Scalar(2)*m(0,1)*m(0,2)*m(1,2) - m(0,0)*m(1,2)*m(1,2) - m(1,1)*m(0,2)*m(0,2) - m(2,2)*m(0,1)*m(0,1); - Scalar c1 = m(0,0)*m(1,1) - m(0,1)*m(0,1) + m(0,0)*m(2,2) - m(0,2)*m(0,2) + m(1,1)*m(2,2) - m(1,2)*m(1,2); - Scalar c2 = m(0,0) + m(1,1) + m(2,2); + Scalar c0 = m(0, 0) * m(1, 1) * m(2, 2) + Scalar(2) * m(0, 1) * m(0, 2) * m(1, 2) - m(0, 0) * m(1, 2) * m(1, 2) + - m(1, 1) * m(0, 2) * m(0, 2) - m(2, 2) * m(0, 1) * m(0, 1); + Scalar c1 = m(0, 0) * m(1, 1) - m(0, 1) * m(0, 1) + m(0, 0) * m(2, 2) - m(0, 2) * m(0, 2) + m(1, 1) * m(2, 2) + - m(1, 2) * m(1, 2); + Scalar c2 = m(0, 0) + m(1, 1) + m(2, 2); // Construct the parameters used in classifying the roots of the equation // and in solving the equation for the roots in closed form. - Scalar c2_over_3 = c2*s_inv3; - Scalar a_over_3 = (c1 - c2*c2_over_3)*s_inv3; - if (a_over_3 > Scalar(0)) - a_over_3 = Scalar(0); + Scalar c2_over_3 = c2 * s_inv3; + Scalar a_over_3 = (c1 - c2 * c2_over_3) * s_inv3; + if (a_over_3 > Scalar(0)) a_over_3 = Scalar(0); - Scalar half_b = Scalar(0.5)*(c0 + c2_over_3*(Scalar(2)*c2_over_3*c2_over_3 - c1)); + Scalar half_b = Scalar(0.5) * (c0 + c2_over_3 * (Scalar(2) * c2_over_3 * c2_over_3 - c1)); - Scalar q = half_b*half_b + a_over_3*a_over_3*a_over_3; - if (q > Scalar(0)) - q = Scalar(0); + Scalar q = half_b * half_b + a_over_3 * a_over_3 * a_over_3; + if (q > Scalar(0)) q = Scalar(0); // Compute the eigenvalues by solving for the roots of the polynomial. Scalar rho = std::sqrt(-a_over_3); - Scalar theta = std::atan2(std::sqrt(-q),half_b)*s_inv3; + Scalar theta = std::atan2(std::sqrt(-q), half_b) * s_inv3; Scalar cos_theta = std::cos(theta); Scalar sin_theta = std::sin(theta); - roots(2) = c2_over_3 + Scalar(2)*rho*cos_theta; - roots(0) = c2_over_3 - rho*(cos_theta + s_sqrt3*sin_theta); - roots(1) = c2_over_3 - rho*(cos_theta - s_sqrt3*sin_theta); + roots(2) = c2_over_3 + Scalar(2) * rho * cos_theta; + roots(0) = c2_over_3 - rho * (cos_theta + s_sqrt3 * sin_theta); + roots(1) = c2_over_3 - rho * (cos_theta - s_sqrt3 * sin_theta); } -template -void eigen33(const Matrix& mat, Matrix& evecs, Vector& evals) +template void eigen33(const Matrix &mat, Matrix &evecs, Vector &evals) { typedef typename Matrix::Scalar Scalar; // Scale the matrix so its entries are in [-1,1]. The scaling is applied // only when at least one matrix entry has magnitude larger than 1. - Scalar shift = mat.trace()/3; + Scalar shift = mat.trace() / 3; Matrix scaledMat = mat; scaledMat.diagonal().array() -= shift; - Scalar scale = scaledMat.cwiseAbs()/*.template triangularView()*/.maxCoeff(); - scale = std::max(scale,Scalar(1)); - scaledMat/=scale; + Scalar scale = scaledMat.cwiseAbs() /*.template triangularView()*/.maxCoeff(); + scale = std::max(scale, Scalar(1)); + scaledMat /= scale; // Compute the eigenvalues -// scaledMat.setZero(); - computeRoots(scaledMat,evals); + // scaledMat.setZero(); + computeRoots(scaledMat, evals); // compute the eigen vectors // **here we assume 3 differents eigenvalues** // "optimized version" which appears to be slower with gcc! -// Vector base; -// Scalar alpha, beta; -// base << scaledMat(1,0) * scaledMat(2,1), -// scaledMat(1,0) * scaledMat(2,0), -// -scaledMat(1,0) * scaledMat(1,0); -// for(int k=0; k<2; ++k) -// { -// alpha = scaledMat(0,0) - evals(k); -// beta = scaledMat(1,1) - evals(k); -// evecs.col(k) = (base + Vector(-beta*scaledMat(2,0), -alpha*scaledMat(2,1), alpha*beta)).normalized(); -// } -// evecs.col(2) = evecs.col(0).cross(evecs.col(1)).normalized(); - -// // naive version -// Matrix tmp; -// tmp = scaledMat; -// tmp.diagonal().array() -= evals(0); -// evecs.col(0) = tmp.row(0).cross(tmp.row(1)).normalized(); -// -// tmp = scaledMat; -// tmp.diagonal().array() -= evals(1); -// evecs.col(1) = tmp.row(0).cross(tmp.row(1)).normalized(); -// -// tmp = scaledMat; -// tmp.diagonal().array() -= evals(2); -// evecs.col(2) = tmp.row(0).cross(tmp.row(1)).normalized(); - + // Vector base; + // Scalar alpha, beta; + // base << scaledMat(1,0) * scaledMat(2,1), + // scaledMat(1,0) * scaledMat(2,0), + // -scaledMat(1,0) * scaledMat(1,0); + // for(int k=0; k<2; ++k) + // { + // alpha = scaledMat(0,0) - evals(k); + // beta = scaledMat(1,1) - evals(k); + // evecs.col(k) = (base + Vector(-beta*scaledMat(2,0), -alpha*scaledMat(2,1), alpha*beta)).normalized(); + // } + // evecs.col(2) = evecs.col(0).cross(evecs.col(1)).normalized(); + + // // naive version + // Matrix tmp; + // tmp = scaledMat; + // tmp.diagonal().array() -= evals(0); + // evecs.col(0) = tmp.row(0).cross(tmp.row(1)).normalized(); + // + // tmp = scaledMat; + // tmp.diagonal().array() -= evals(1); + // evecs.col(1) = tmp.row(0).cross(tmp.row(1)).normalized(); + // + // tmp = scaledMat; + // tmp.diagonal().array() -= evals(2); + // evecs.col(2) = tmp.row(0).cross(tmp.row(1)).normalized(); + // a more stable version: - if((evals(2)-evals(0))<=Eigen::NumTraits::epsilon()) - { + if ((evals(2) - evals(0)) <= Eigen::NumTraits::epsilon()) { evecs.setIdentity(); - } - else - { + } else { Matrix tmp; tmp = scaledMat; - tmp.diagonal ().array () -= evals (2); - evecs.col (2) = tmp.row (0).cross (tmp.row (1)).normalized (); - + tmp.diagonal().array() -= evals(2); + evecs.col(2) = tmp.row(0).cross(tmp.row(1)).normalized(); + tmp = scaledMat; - tmp.diagonal ().array () -= evals (1); - evecs.col(1) = tmp.row (0).cross(tmp.row (1)); + tmp.diagonal().array() -= evals(1); + evecs.col(1) = tmp.row(0).cross(tmp.row(1)); Scalar n1 = evecs.col(1).norm(); - if(n1<=Eigen::NumTraits::epsilon()) + if (n1 <= Eigen::NumTraits::epsilon()) evecs.col(1) = evecs.col(2).unitOrthogonal(); else evecs.col(1) /= n1; - + // make sure that evecs[1] is orthogonal to evecs[2] evecs.col(1) = evecs.col(2).cross(evecs.col(1).cross(evecs.col(2))).normalized(); evecs.col(0) = evecs.col(2).cross(evecs.col(1)); } - + // Rescale back to the original size. evals *= scale; - evals.array()+=shift; + evals.array() += shift; } int main() @@ -169,27 +164,27 @@ int main() int rep = 400000; typedef Matrix3d Mat; typedef Vector3d Vec; - Mat A = Mat::Random(3,3); + Mat A = Mat::Random(3, 3); A = A.adjoint() * A; -// Mat Q = A.householderQr().householderQ(); -// A = Q * Vec(2.2424567,2.2424566,7.454353).asDiagonal() * Q.transpose(); + // Mat Q = A.householderQr().householderQ(); + // A = Q * Vec(2.2424567,2.2424566,7.454353).asDiagonal() * Q.transpose(); SelfAdjointEigenSolver eig(A); BENCH(t, tries, rep, eig.compute(A)); std::cout << "Eigen iterative: " << t.best() << "s\n"; - + BENCH(t, tries, rep, eig.computeDirect(A)); std::cout << "Eigen direct : " << t.best() << "s\n"; Mat evecs; Vec evals; - BENCH(t, tries, rep, eigen33(A,evecs,evals)); + BENCH(t, tries, rep, eigen33(A, evecs, evals)); std::cout << "Direct: " << t.best() << "s\n\n"; -// std::cerr << "Eigenvalue/eigenvector diffs:\n"; -// std::cerr << (evals - eig.eigenvalues()).transpose() << "\n"; -// for(int k=0;k<3;++k) -// if(evecs.col(k).dot(eig.eigenvectors().col(k))<0) -// evecs.col(k) = -evecs.col(k); -// std::cerr << evecs - eig.eigenvectors() << "\n\n"; + // std::cerr << "Eigenvalue/eigenvector diffs:\n"; + // std::cerr << (evals - eig.eigenvalues()).transpose() << "\n"; + // for(int k=0;k<3;++k) + // if(evecs.col(k).dot(eig.eigenvectors().col(k))<0) + // evecs.col(k) = -evecs.col(k); + // std::cerr << evecs - eig.eigenvectors() << "\n\n"; } diff --git a/filmulator-gui/core/nlmeans/eigen/bench/geometry.cpp b/filmulator-gui/core/nlmeans/eigen/bench/geometry.cpp index b187a515..a5fd1ac4 100644 --- a/filmulator-gui/core/nlmeans/eigen/bench/geometry.cpp +++ b/filmulator-gui/core/nlmeans/eigen/bench/geometry.cpp @@ -1,7 +1,7 @@ -#include #include #include +#include using namespace std; using namespace Eigen; @@ -16,38 +16,35 @@ using namespace Eigen; typedef SCALAR Scalar; typedef NumTraits::Real RealScalar; -typedef Matrix A; -typedef Matrix B; -typedef Matrix C; -typedef Matrix M; +typedef Matrix A; +typedef Matrix B; +typedef Matrix C; +typedef Matrix M; -template -EIGEN_DONT_INLINE void transform(const Transformation& t, Data& data) +template EIGEN_DONT_INLINE void transform(const Transformation &t, Data &data) { EIGEN_ASM_COMMENT("begin"); data = t * data; EIGEN_ASM_COMMENT("end"); } -template -EIGEN_DONT_INLINE void transform(const Quaternion& t, Data& data) +template EIGEN_DONT_INLINE void transform(const Quaternion &t, Data &data) { EIGEN_ASM_COMMENT("begin quat"); - for(int i=0;i struct ToRotationMatrixWrapper { - enum {Dim = T::Dim}; + enum { Dim = T::Dim }; typedef typename T::Scalar Scalar; - ToRotationMatrixWrapper(const T& o) : object(o) {} + ToRotationMatrixWrapper(const T &o) : object(o) {} T object; }; template -EIGEN_DONT_INLINE void transform(const ToRotationMatrixWrapper& t, Data& data) +EIGEN_DONT_INLINE void transform(const ToRotationMatrixWrapper &t, Data &data) { EIGEN_ASM_COMMENT("begin quat via mat"); data = t.object.toRotationMatrix() * data; @@ -55,66 +52,69 @@ EIGEN_DONT_INLINE void transform(const ToRotationMatrixWrapper& t, Data& } template -EIGEN_DONT_INLINE void transform(const Transform& t, Data& data) +EIGEN_DONT_INLINE void transform(const Transform &t, Data &data) { - data = (t * data.colwise().homogeneous()).template block(0,0); + data = (t * data.colwise().homogeneous()).template block(0, 0); } -template struct get_dim { enum { Dim = T::Dim }; }; -template -struct get_dim > { enum { Dim = R }; }; +template struct get_dim +{ + enum { Dim = T::Dim }; +}; +template struct get_dim> +{ + enum { Dim = R }; +}; -template -struct bench_impl +template struct bench_impl { - static EIGEN_DONT_INLINE void run(const Transformation& t) + static EIGEN_DONT_INLINE void run(const Transformation &t) { - Matrix::Dim,N> data; + Matrix::Dim, N> data; data.setRandom(); - bench_impl::run(t); + bench_impl::run(t); BenchTimer timer; - BENCH(timer,10,100000,transform(t,data)); + BENCH(timer, 10, 100000, transform(t, data)); cout.width(9); cout << timer.best() << " "; } }; -template -struct bench_impl +template struct bench_impl { - static EIGEN_DONT_INLINE void run(const Transformation&) {} + static EIGEN_DONT_INLINE void run(const Transformation &) {} }; -template -EIGEN_DONT_INLINE void bench(const std::string& msg, const Transformation& t) +template EIGEN_DONT_INLINE void bench(const std::string &msg, const Transformation &t) { cout << msg << " "; - bench_impl::run(t); + bench_impl::run(t); std::cout << "\n"; } -int main(int argc, char ** argv) +int main(int argc, char **argv) { - Matrix mat34; mat34.setRandom(); - Transform iso3(mat34); - Transform aff3(mat34); - Transform caff3(mat34); - Transform proj3(mat34); - Quaternion quat;quat.setIdentity(); - ToRotationMatrixWrapper > quatmat(quat); - Matrix mat33; mat33.setRandom(); - + Matrix mat34; + mat34.setRandom(); + Transform iso3(mat34); + Transform aff3(mat34); + Transform caff3(mat34); + Transform proj3(mat34); + Quaternion quat; + quat.setIdentity(); + ToRotationMatrixWrapper> quatmat(quat); + Matrix mat33; + mat33.setRandom(); + cout.precision(4); - std::cout - << "N "; - for(int i=0;i +#include "../../BenchTimer.h" +#include #include +#include #include -#include -#include "../../BenchTimer.h" using namespace Eigen; #ifndef SCALAR @@ -11,57 +11,52 @@ using namespace Eigen; typedef SCALAR Scalar; -typedef Matrix Mat; +typedef Matrix Mat; EIGEN_DONT_INLINE -void gemm(const Mat &A, const Mat &B, Mat &C) -{ - C.noalias() += A * B; -} +void gemm(const Mat &A, const Mat &B, Mat &C) { C.noalias() += A * B; } EIGEN_DONT_INLINE double bench(long m, long n, long k) { - Mat A(m,k); - Mat B(k,n); - Mat C(m,n); + Mat A(m, k); + Mat B(k, n); + Mat C(m, n); A.setRandom(); B.setRandom(); C.setZero(); - + BenchTimer t; - - double up = 1e8*4/sizeof(Scalar); + + double up = 1e8 * 4 / sizeof(Scalar); double tm0 = 4, tm1 = 10; - if(NumTraits::IsComplex) - { + if (NumTraits::IsComplex) { up /= 4; tm0 = 2; tm1 = 4; } - + double flops = 2. * m * n * k; - long rep = std::max(1., std::min(100., up/flops) ); - long tries = std::max(tm0, std::min(tm1, up/flops) ); - - BENCH(t, tries, rep, gemm(A,B,C)); - + long rep = std::max(1., std::min(100., up / flops)); + long tries = std::max(tm0, std::min(tm1, up / flops)); + + BENCH(t, tries, rep, gemm(A, B, C)); + return 1e-9 * rep * flops / t.best(); } int main(int argc, char **argv) { std::vector results; - + std::ifstream settings("gemm_settings.txt"); long m, n, k; - while(settings >> m >> n >> k) - { - //std::cerr << " Testing " << m << " " << n << " " << k << std::endl; - results.push_back( bench(m, n, k) ); + while (settings >> m >> n >> k) { + // std::cerr << " Testing " << m << " " << n << " " << k << std::endl; + results.push_back(bench(m, n, k)); } - + std::cout << RowVectorXd::Map(results.data(), results.size()); - + return 0; } diff --git a/filmulator-gui/core/nlmeans/eigen/bench/perf_monitoring/gemm/lazy_gemm.cpp b/filmulator-gui/core/nlmeans/eigen/bench/perf_monitoring/gemm/lazy_gemm.cpp index 6dc37015..2d16c221 100644 --- a/filmulator-gui/core/nlmeans/eigen/bench/perf_monitoring/gemm/lazy_gemm.cpp +++ b/filmulator-gui/core/nlmeans/eigen/bench/perf_monitoring/gemm/lazy_gemm.cpp @@ -1,8 +1,8 @@ -#include +#include "../../BenchTimer.h" +#include #include +#include #include -#include -#include "../../BenchTimer.h" using namespace Eigen; #ifndef SCALAR @@ -12,70 +12,90 @@ using namespace Eigen; typedef SCALAR Scalar; template -EIGEN_DONT_INLINE -void lazy_gemm(const MatA &A, const MatB &B, MatC &C) +EIGEN_DONT_INLINE void lazy_gemm(const MatA &A, const MatB &B, MatC &C) { -// escape((void*)A.data()); -// escape((void*)B.data()); + // escape((void*)A.data()); + // escape((void*)B.data()); C.noalias() += A.lazyProduct(B); -// escape((void*)C.data()); + // escape((void*)C.data()); } -template -EIGEN_DONT_INLINE -double bench() +template EIGEN_DONT_INLINE double bench() { - typedef Matrix MatA; - typedef Matrix MatB; - typedef Matrix MatC; + typedef Matrix MatA; + typedef Matrix MatB; + typedef Matrix MatC; - MatA A(m,k); - MatB B(k,n); - MatC C(m,n); + MatA A(m, k); + MatB B(k, n); + MatC C(m, n); A.setRandom(); B.setRandom(); C.setZero(); BenchTimer t; - double up = 1e7*4/sizeof(Scalar); + double up = 1e7 * 4 / sizeof(Scalar); double tm0 = 10, tm1 = 20; double flops = 2. * m * n * k; - long rep = std::max(10., std::min(10000., up/flops) ); - long tries = std::max(tm0, std::min(tm1, up/flops) ); + long rep = std::max(10., std::min(10000., up / flops)); + long tries = std::max(tm0, std::min(tm1, up / flops)); - BENCH(t, tries, rep, lazy_gemm(A,B,C)); + BENCH(t, tries, rep, lazy_gemm(A, B, C)); return 1e-9 * rep * flops / t.best(); } -template -double bench_t(int t) +template double bench_t(int t) { - if(t) - return bench(); + if (t) + return bench(); else - return bench(); + return bench(); } EIGEN_DONT_INLINE double bench_mnk(int m, int n, int k, int t) { - int id = m*10000 + n*100 + k; - switch(id) { - case 10101 : return bench_t< 1, 1, 1>(t); break; - case 20202 : return bench_t< 2, 2, 2>(t); break; - case 30303 : return bench_t< 3, 3, 3>(t); break; - case 40404 : return bench_t< 4, 4, 4>(t); break; - case 50505 : return bench_t< 5, 5, 5>(t); break; - case 60606 : return bench_t< 6, 6, 6>(t); break; - case 70707 : return bench_t< 7, 7, 7>(t); break; - case 80808 : return bench_t< 8, 8, 8>(t); break; - case 90909 : return bench_t< 9, 9, 9>(t); break; - case 101010 : return bench_t<10,10,10>(t); break; - case 111111 : return bench_t<11,11,11>(t); break; - case 121212 : return bench_t<12,12,12>(t); break; + int id = m * 10000 + n * 100 + k; + switch (id) { + case 10101: + return bench_t<1, 1, 1>(t); + break; + case 20202: + return bench_t<2, 2, 2>(t); + break; + case 30303: + return bench_t<3, 3, 3>(t); + break; + case 40404: + return bench_t<4, 4, 4>(t); + break; + case 50505: + return bench_t<5, 5, 5>(t); + break; + case 60606: + return bench_t<6, 6, 6>(t); + break; + case 70707: + return bench_t<7, 7, 7>(t); + break; + case 80808: + return bench_t<8, 8, 8>(t); + break; + case 90909: + return bench_t<9, 9, 9>(t); + break; + case 101010: + return bench_t<10, 10, 10>(t); + break; + case 111111: + return bench_t<11, 11, 11>(t); + break; + case 121212: + return bench_t<12, 12, 12>(t); + break; } return 0; } @@ -83,16 +103,15 @@ double bench_mnk(int m, int n, int k, int t) int main(int argc, char **argv) { std::vector results; - + std::ifstream settings("lazy_gemm_settings.txt"); long m, n, k, t; - while(settings >> m >> n >> k >> t) - { - //std::cerr << " Testing " << m << " " << n << " " << k << std::endl; - results.push_back( bench_mnk(m, n, k, t) ); + while (settings >> m >> n >> k >> t) { + // std::cerr << " Testing " << m << " " << n << " " << k << std::endl; + results.push_back(bench_mnk(m, n, k, t)); } - + std::cout << RowVectorXd::Map(results.data(), results.size()); - + return 0; } diff --git a/filmulator-gui/core/nlmeans/eigen/bench/product_threshold.cpp b/filmulator-gui/core/nlmeans/eigen/bench/product_threshold.cpp index dd6d15a0..847a22b8 100644 --- a/filmulator-gui/core/nlmeans/eigen/bench/product_threshold.cpp +++ b/filmulator-gui/core/nlmeans/eigen/bench/product_threshold.cpp @@ -1,77 +1,96 @@ -#include #include #include +#include using namespace Eigen; using namespace std; #define END 9 -template struct map_size { enum { ret = S }; }; -template<> struct map_size<10> { enum { ret = 20 }; }; -template<> struct map_size<11> { enum { ret = 50 }; }; -template<> struct map_size<12> { enum { ret = 100 }; }; -template<> struct map_size<13> { enum { ret = 300 }; }; +template struct map_size +{ + enum { ret = S }; +}; +template<> struct map_size<10> +{ + enum { ret = 20 }; +}; +template<> struct map_size<11> +{ + enum { ret = 50 }; +}; +template<> struct map_size<12> +{ + enum { ret = 100 }; +}; +template<> struct map_size<13> +{ + enum { ret = 300 }; +}; -template struct alt_prod +template struct alt_prod { enum { - ret = M==1 && N==1 ? InnerProduct - : K==1 ? OuterProduct - : M==1 ? GemvProduct - : N==1 ? GemvProduct - : GemmProduct + ret = M == 1 && N == 1 ? InnerProduct + : K == 1 ? OuterProduct + : M == 1 ? GemvProduct + : N == 1 ? GemvProduct + : GemmProduct }; }; - + void print_mode(int mode) { - if(mode==InnerProduct) std::cout << "i"; - if(mode==OuterProduct) std::cout << "o"; - if(mode==CoeffBasedProductMode) std::cout << "c"; - if(mode==LazyCoeffBasedProductMode) std::cout << "l"; - if(mode==GemvProduct) std::cout << "v"; - if(mode==GemmProduct) std::cout << "m"; + if (mode == InnerProduct) std::cout << "i"; + if (mode == OuterProduct) std::cout << "o"; + if (mode == CoeffBasedProductMode) std::cout << "c"; + if (mode == LazyCoeffBasedProductMode) std::cout << "l"; + if (mode == GemvProduct) std::cout << "v"; + if (mode == GemmProduct) std::cout << "m"; } template -EIGEN_DONT_INLINE void prod(const Lhs& a, const Rhs& b, Res& c) +EIGEN_DONT_INLINE void prod(const Lhs &a, const Rhs &b, Res &c) { - c.noalias() += typename ProductReturnType::Type(a,b); + c.noalias() += typename ProductReturnType::Type(a, b); } -template -EIGEN_DONT_INLINE void bench_prod() +template EIGEN_DONT_INLINE void bench_prod() { - typedef Matrix Lhs; Lhs a; a.setRandom(); - typedef Matrix Rhs; Rhs b; b.setRandom(); - typedef Matrix Res; Res c; c.setRandom(); + typedef Matrix Lhs; + Lhs a; + a.setRandom(); + typedef Matrix Rhs; + Rhs b; + b.setRandom(); + typedef Matrix Res; + Res c; + c.setRandom(); BenchTimer t; - double n = 2.*double(M)*double(N)*double(K); - int rep = 100000./n; + double n = 2. * double(M) * double(N) * double(K); + int rep = 100000. / n; rep /= 2; - if(rep<1) rep = 1; + if (rep < 1) rep = 1; do { rep *= 2; t.reset(); - BENCH(t,1,rep,prod(a,b,c)); - } while(t.best()<0.1); - + BENCH(t, 1, rep, prod(a, b, c)); + } while (t.best() < 0.1); + t.reset(); - BENCH(t,5,rep,prod(a,b,c)); + BENCH(t, 5, rep, prod(a, b, c)); print_mode(Mode); - std::cout << int(1e-6*n*rep/t.best()) << "\t"; + std::cout << int(1e-6 * n * rep / t.best()) << "\t"; } template struct print_n; template struct loop_on_m; template struct loop_on_n; -template -struct loop_on_k +template struct loop_on_k { static void run() { @@ -79,65 +98,72 @@ struct loop_on_k print_n::run(); std::cout << "\n"; - loop_on_m::run(); + loop_on_m::run(); std::cout << "\n\n"; - loop_on_k::run(); + loop_on_k::run(); } }; -template -struct loop_on_k { static void run(){} }; +template struct loop_on_k +{ + static void run() {} +}; -template -struct loop_on_m +template struct loop_on_m { static void run() { std::cout << M << "f\t"; - loop_on_n::run(); + loop_on_n::run(); std::cout << "\n"; - + std::cout << M << "f\t"; - loop_on_n::run(); + loop_on_n::run(); std::cout << "\n"; - loop_on_m::run(); + loop_on_m::run(); } }; -template -struct loop_on_m { static void run(){} }; +template struct loop_on_m +{ + static void run() {} +}; -template -struct loop_on_n +template struct loop_on_n { static void run() { - bench_prod::ret : Mode>(); - - loop_on_n::run(); + bench_prod::ret : Mode>(); + + loop_on_n::run(); } }; -template -struct loop_on_n { static void run(){} }; +template struct loop_on_n +{ + static void run() {} +}; template struct print_n { static void run() { std::cout << map_size::ret << "\t"; - print_n::run(); + print_n::run(); } }; -template<> struct print_n { static void run(){} }; +template<> struct print_n +{ + static void run() {} +}; int main() { - loop_on_k<1,1,1>::run(); - - return 0; + loop_on_k<1, 1, 1>::run(); + + return 0; } diff --git a/filmulator-gui/core/nlmeans/eigen/bench/quat_slerp.cpp b/filmulator-gui/core/nlmeans/eigen/bench/quat_slerp.cpp index bffb3bf1..e01c4323 100644 --- a/filmulator-gui/core/nlmeans/eigen/bench/quat_slerp.cpp +++ b/filmulator-gui/core/nlmeans/eigen/bench/quat_slerp.cpp @@ -1,141 +1,122 @@ -#include #include #include +#include using namespace Eigen; using namespace std; - -template -EIGEN_DONT_INLINE Q nlerp(const Q& a, const Q& b, typename Q::Scalar t) +template EIGEN_DONT_INLINE Q nlerp(const Q &a, const Q &b, typename Q::Scalar t) { - return Q((a.coeffs() * (1.0-t) + b.coeffs() * t).normalized()); + return Q((a.coeffs() * (1.0 - t) + b.coeffs() * t).normalized()); } -template -EIGEN_DONT_INLINE Q slerp_eigen(const Q& a, const Q& b, typename Q::Scalar t) +template EIGEN_DONT_INLINE Q slerp_eigen(const Q &a, const Q &b, typename Q::Scalar t) { - return a.slerp(t,b); + return a.slerp(t, b); } -template -EIGEN_DONT_INLINE Q slerp_legacy(const Q& a, const Q& b, typename Q::Scalar t) +template EIGEN_DONT_INLINE Q slerp_legacy(const Q &a, const Q &b, typename Q::Scalar t) { typedef typename Q::Scalar Scalar; static const Scalar one = Scalar(1) - dummy_precision(); Scalar d = a.dot(b); Scalar absD = internal::abs(d); - if (absD>=one) - return a; + if (absD >= one) return a; // theta is the angle between the 2 quaternions Scalar theta = std::acos(absD); Scalar sinTheta = internal::sin(theta); - Scalar scale0 = internal::sin( ( Scalar(1) - t ) * theta) / sinTheta; - Scalar scale1 = internal::sin( ( t * theta) ) / sinTheta; - if (d<0) - scale1 = -scale1; + Scalar scale0 = internal::sin((Scalar(1) - t) * theta) / sinTheta; + Scalar scale1 = internal::sin((t * theta)) / sinTheta; + if (d < 0) scale1 = -scale1; return Q(scale0 * a.coeffs() + scale1 * b.coeffs()); } -template -EIGEN_DONT_INLINE Q slerp_legacy_nlerp(const Q& a, const Q& b, typename Q::Scalar t) +template EIGEN_DONT_INLINE Q slerp_legacy_nlerp(const Q &a, const Q &b, typename Q::Scalar t) { typedef typename Q::Scalar Scalar; static const Scalar one = Scalar(1) - epsilon(); Scalar d = a.dot(b); Scalar absD = internal::abs(d); - + Scalar scale0; Scalar scale1; - - if (absD>=one) - { + + if (absD >= one) { scale0 = Scalar(1) - t; scale1 = t; - } - else - { + } else { // theta is the angle between the 2 quaternions Scalar theta = std::acos(absD); Scalar sinTheta = internal::sin(theta); - scale0 = internal::sin( ( Scalar(1) - t ) * theta) / sinTheta; - scale1 = internal::sin( ( t * theta) ) / sinTheta; - if (d<0) - scale1 = -scale1; + scale0 = internal::sin((Scalar(1) - t) * theta) / sinTheta; + scale1 = internal::sin((t * theta)) / sinTheta; + if (d < 0) scale1 = -scale1; } return Q(scale0 * a.coeffs() + scale1 * b.coeffs()); } -template -inline T sin_over_x(T x) +template inline T sin_over_x(T x) { - if (T(1) + x*x == T(1)) + if (T(1) + x * x == T(1)) return T(1); else - return std::sin(x)/x; + return std::sin(x) / x; } -template -EIGEN_DONT_INLINE Q slerp_rw(const Q& a, const Q& b, typename Q::Scalar t) +template EIGEN_DONT_INLINE Q slerp_rw(const Q &a, const Q &b, typename Q::Scalar t) { typedef typename Q::Scalar Scalar; - + Scalar d = a.dot(b); Scalar theta; - if (d<0.0) - theta = /*M_PI -*/ Scalar(2)*std::asin( (a.coeffs()+b.coeffs()).norm()/2 ); + if (d < 0.0) + theta = /*M_PI -*/ Scalar(2) * std::asin((a.coeffs() + b.coeffs()).norm() / 2); else - theta = Scalar(2)*std::asin( (a.coeffs()-b.coeffs()).norm()/2 ); - + theta = Scalar(2) * std::asin((a.coeffs() - b.coeffs()).norm() / 2); + // theta is the angle between the 2 quaternions -// Scalar theta = std::acos(absD); + // Scalar theta = std::acos(absD); Scalar sinOverTheta = sin_over_x(theta); - Scalar scale0 = (Scalar(1)-t)*sin_over_x( ( Scalar(1) - t ) * theta) / sinOverTheta; - Scalar scale1 = t * sin_over_x( ( t * theta) ) / sinOverTheta; - if (d<0) - scale1 = -scale1; + Scalar scale0 = (Scalar(1) - t) * sin_over_x((Scalar(1) - t) * theta) / sinOverTheta; + Scalar scale1 = t * sin_over_x((t * theta)) / sinOverTheta; + if (d < 0) scale1 = -scale1; return Quaternion(scale0 * a.coeffs() + scale1 * b.coeffs()); } -template -EIGEN_DONT_INLINE Q slerp_gael(const Q& a, const Q& b, typename Q::Scalar t) +template EIGEN_DONT_INLINE Q slerp_gael(const Q &a, const Q &b, typename Q::Scalar t) { typedef typename Q::Scalar Scalar; - + Scalar d = a.dot(b); Scalar theta; -// theta = Scalar(2) * atan2((a.coeffs()-b.coeffs()).norm(),(a.coeffs()+b.coeffs()).norm()); -// if (d<0.0) -// theta = M_PI-theta; - - if (d<0.0) - theta = /*M_PI -*/ Scalar(2)*std::asin( (-a.coeffs()-b.coeffs()).norm()/2 ); + // theta = Scalar(2) * atan2((a.coeffs()-b.coeffs()).norm(),(a.coeffs()+b.coeffs()).norm()); + // if (d<0.0) + // theta = M_PI-theta; + + if (d < 0.0) + theta = /*M_PI -*/ Scalar(2) * std::asin((-a.coeffs() - b.coeffs()).norm() / 2); else - theta = Scalar(2)*std::asin( (a.coeffs()-b.coeffs()).norm()/2 ); - - + theta = Scalar(2) * std::asin((a.coeffs() - b.coeffs()).norm() / 2); + + Scalar scale0; Scalar scale1; - if(theta*theta-Scalar(6)==-Scalar(6)) - { + if (theta * theta - Scalar(6) == -Scalar(6)) { scale0 = Scalar(1) - t; scale1 = t; - } - else - { + } else { Scalar sinTheta = std::sin(theta); - scale0 = internal::sin( ( Scalar(1) - t ) * theta) / sinTheta; - scale1 = internal::sin( ( t * theta) ) / sinTheta; - if (d<0) - scale1 = -scale1; + scale0 = internal::sin((Scalar(1) - t) * theta) / sinTheta; + scale1 = internal::sin((t * theta)) / sinTheta; + if (d < 0) scale1 = -scale1; } return Quaternion(scale0 * a.coeffs() + scale1 * b.coeffs()); @@ -145,97 +126,94 @@ int main() { typedef double RefScalar; typedef float TestScalar; - - typedef Quaternion Qd; + + typedef Quaternion Qd; typedef Quaternion Qf; - - unsigned int g_seed = (unsigned int) time(NULL); + + unsigned int g_seed = (unsigned int)time(NULL); std::cout << g_seed << "\n"; -// g_seed = 1259932496; + // g_seed = 1259932496; srand(g_seed); - - Matrix maxerr(7); + + Matrix maxerr(7); maxerr.setZero(); - - Matrix avgerr(7); + + Matrix avgerr(7); avgerr.setZero(); - - cout << "double=>float=>double nlerp eigen legacy(snap) legacy(nlerp) rightway gael's criteria\n"; - + + cout << "double=>float=>double nlerp eigen legacy(snap) legacy(nlerp) rightway " + " gael's criteria\n"; + int rep = 100; int iters = 40; - for (int w=0; w()); Qd br(b.cast()); Qd cr; - - - + + cout.precision(8); cout << std::scientific; - for (int i=0; i(); - c[0] = nlerp(a,b,t); - c[1] = slerp_eigen(a,b,t); - c[2] = slerp_legacy(a,b,t); - c[3] = slerp_legacy_nlerp(a,b,t); - c[4] = slerp_rw(a,b,t); - c[5] = slerp_gael(a,b,t); - + c[0] = nlerp(a, b, t); + c[1] = slerp_eigen(a, b, t); + c[2] = slerp_legacy(a, b, t); + c[3] = slerp_legacy_nlerp(a, b, t); + c[4] = slerp_rw(a, b, t); + c[5] = slerp_gael(a, b, t); + VectorXd err(7); - err[0] = (cr.coeffs()-refc.cast().coeffs()).norm(); -// std::cout << err[0] << " "; - for (int k=0; k<6; ++k) - { - err[k+1] = (c[k].coeffs()-refc.coeffs()).norm(); -// std::cout << err[k+1] << " "; + err[0] = (cr.coeffs() - refc.cast().coeffs()).norm(); + // std::cout << err[0] << " "; + for (int k = 0; k < 6; ++k) { + err[k + 1] = (c[k].coeffs() - refc.coeffs()).norm(); + // std::cout << err[k+1] << " "; } maxerr = maxerr.cwise().max(err); avgerr += err; -// std::cout << "\n"; + // std::cout << "\n"; b = cr.cast(); br = cr; } -// std::cout << "\n"; + // std::cout << "\n"; } - avgerr /= RefScalar(rep*iters); + avgerr /= RefScalar(rep * iters); cout << "\n\nAccuracy:\n" << " max: " << maxerr.transpose() << "\n"; cout << " avg: " << avgerr.transpose() << "\n"; - + // perf bench - Quaternionf a,b; + Quaternionf a, b; a.coeffs().setRandom(); a.normalize(); b.coeffs().setRandom(); b.normalize(); - //b = a; + // b = a; float s = 0.65; - - #define BENCH(FUNC) {\ - BenchTimer t; \ - for(int k=0; k<2; ++k) {\ - t.start(); \ - for(int i=0; i<1000000; ++i) \ - FUNC(a,b,s); \ - t.stop(); \ - } \ + +#define BENCH(FUNC) \ + { \ + BenchTimer t; \ + for (int k = 0; k < 2; ++k) { \ + t.start(); \ + for (int i = 0; i < 1000000; ++i) FUNC(a, b, s); \ + t.stop(); \ + } \ cout << " " << #FUNC << " => \t " << t.value() << "s\n"; \ } - + cout << "\nSpeed:\n" << std::fixed; BENCH(nlerp); BENCH(slerp_eigen); @@ -244,4 +222,3 @@ int main() BENCH(slerp_rw); BENCH(slerp_gael); } - diff --git a/filmulator-gui/core/nlmeans/eigen/bench/quatmul.cpp b/filmulator-gui/core/nlmeans/eigen/bench/quatmul.cpp index 8d9d7922..e15f8241 100644 --- a/filmulator-gui/core/nlmeans/eigen/bench/quatmul.cpp +++ b/filmulator-gui/core/nlmeans/eigen/bench/quatmul.cpp @@ -1,39 +1,36 @@ -#include #include #include #include +#include -using namespace Eigen; +using namespace Eigen; -template -EIGEN_DONT_INLINE void quatmul_default(const Quat& a, const Quat& b, Quat& c) -{ - c = a * b; -} +template EIGEN_DONT_INLINE void quatmul_default(const Quat &a, const Quat &b, Quat &c) { c = a * b; } -template -EIGEN_DONT_INLINE void quatmul_novec(const Quat& a, const Quat& b, Quat& c) +template EIGEN_DONT_INLINE void quatmul_novec(const Quat &a, const Quat &b, Quat &c) { - c = internal::quat_product<0, Quat, Quat, typename Quat::Scalar, Aligned>::run(a,b); + c = internal::quat_product<0, Quat, Quat, typename Quat::Scalar, Aligned>::run(a, b); } -template void bench(const std::string& label) +template void bench(const std::string &label) { int tries = 10; int rep = 1000000; BenchTimer t; - + Quat a(4, 1, 2, 3); Quat b(2, 3, 4, 5); Quat c; - + std::cout.precision(3); - - BENCH(t, tries, rep, quatmul_default(a,b,c)); - std::cout << label << " default " << 1e3*t.best(CPU_TIMER) << "ms \t" << 1e-6*double(rep)/(t.best(CPU_TIMER)) << " M mul/s\n"; - - BENCH(t, tries, rep, quatmul_novec(a,b,c)); - std::cout << label << " novec " << 1e3*t.best(CPU_TIMER) << "ms \t" << 1e-6*double(rep)/(t.best(CPU_TIMER)) << " M mul/s\n"; + + BENCH(t, tries, rep, quatmul_default(a, b, c)); + std::cout << label << " default " << 1e3 * t.best(CPU_TIMER) << "ms \t" << 1e-6 * double(rep) / (t.best(CPU_TIMER)) + << " M mul/s\n"; + + BENCH(t, tries, rep, quatmul_novec(a, b, c)); + std::cout << label << " novec " << 1e3 * t.best(CPU_TIMER) << "ms \t" << 1e-6 * double(rep) / (t.best(CPU_TIMER)) + << " M mul/s\n"; } int main() @@ -42,6 +39,4 @@ int main() bench("double"); return 0; - } - diff --git a/filmulator-gui/core/nlmeans/eigen/bench/sparse_cholesky.cpp b/filmulator-gui/core/nlmeans/eigen/bench/sparse_cholesky.cpp index ecb22678..c3ca8501 100644 --- a/filmulator-gui/core/nlmeans/eigen/bench/sparse_cholesky.cpp +++ b/filmulator-gui/core/nlmeans/eigen/bench/sparse_cholesky.cpp @@ -1,9 +1,19 @@ // #define EIGEN_TAUCS_SUPPORT // #define EIGEN_CHOLMOD_SUPPORT -#include #include +#include -// g++ -DSIZE=10000 -DDENSITY=0.001 sparse_cholesky.cpp -I.. -DDENSEMATRI -O3 -g0 -DNDEBUG -DNBTRIES=1 -I /home/gael/Coding/LinearAlgebra/taucs_full/src/ -I/home/gael/Coding/LinearAlgebra/taucs_full/build/linux/ -L/home/gael/Coding/LinearAlgebra/taucs_full/lib/linux/ -ltaucs /home/gael/Coding/LinearAlgebra/GotoBLAS/libgoto.a -lpthread -I /home/gael/Coding/LinearAlgebra/SuiteSparse/CHOLMOD/Include/ $CHOLLIB -I /home/gael/Coding/LinearAlgebra/SuiteSparse/UFconfig/ /home/gael/Coding/LinearAlgebra/SuiteSparse/CCOLAMD/Lib/libccolamd.a /home/gael/Coding/LinearAlgebra/SuiteSparse/CHOLMOD/Lib/libcholmod.a -lmetis /home/gael/Coding/LinearAlgebra/SuiteSparse/AMD/Lib/libamd.a /home/gael/Coding/LinearAlgebra/SuiteSparse/CAMD/Lib/libcamd.a /home/gael/Coding/LinearAlgebra/SuiteSparse/CCOLAMD/Lib/libccolamd.a /home/gael/Coding/LinearAlgebra/SuiteSparse/COLAMD/Lib/libcolamd.a -llapack && ./a.out +// g++ -DSIZE=10000 -DDENSITY=0.001 sparse_cholesky.cpp -I.. -DDENSEMATRI -O3 -g0 -DNDEBUG -DNBTRIES=1 -I +// /home/gael/Coding/LinearAlgebra/taucs_full/src/ -I/home/gael/Coding/LinearAlgebra/taucs_full/build/linux/ +// -L/home/gael/Coding/LinearAlgebra/taucs_full/lib/linux/ -ltaucs /home/gael/Coding/LinearAlgebra/GotoBLAS/libgoto.a +// -lpthread -I /home/gael/Coding/LinearAlgebra/SuiteSparse/CHOLMOD/Include/ $CHOLLIB -I +// /home/gael/Coding/LinearAlgebra/SuiteSparse/UFconfig/ +// /home/gael/Coding/LinearAlgebra/SuiteSparse/CCOLAMD/Lib/libccolamd.a +// /home/gael/Coding/LinearAlgebra/SuiteSparse/CHOLMOD/Lib/libcholmod.a -lmetis +// /home/gael/Coding/LinearAlgebra/SuiteSparse/AMD/Lib/libamd.a +// /home/gael/Coding/LinearAlgebra/SuiteSparse/CAMD/Lib/libcamd.a +// /home/gael/Coding/LinearAlgebra/SuiteSparse/CCOLAMD/Lib/libccolamd.a +// /home/gael/Coding/LinearAlgebra/SuiteSparse/COLAMD/Lib/libcolamd.a -llapack && ./a.out #define NOGMM #define NOMTL @@ -30,48 +40,43 @@ #define NBTRIES 10 #endif -#define BENCH(X) \ - timer.reset(); \ - for (int _j=0; _j EigenSparseTriMatrix; -typedef SparseMatrix EigenSparseSelfAdjointMatrix; +typedef SparseMatrix EigenSparseSelfAdjointMatrix; -void fillSpdMatrix(float density, int rows, int cols, EigenSparseSelfAdjointMatrix& dst) +void fillSpdMatrix(float density, int rows, int cols, EigenSparseSelfAdjointMatrix &dst) { - dst.startFill(rows*cols*density); - for(int j = 0; j < cols; j++) - { - dst.fill(j,j) = internal::random(10,20); - for(int i = j+1; i < rows; i++) - { - Scalar v = (internal::random(0,1) < density) ? internal::random() : 0; - if (v!=0) - dst.fill(i,j) = v; + dst.startFill(rows * cols * density); + for (int j = 0; j < cols; j++) { + dst.fill(j, j) = internal::random(10, 20); + for (int i = j + 1; i < rows; i++) { + Scalar v = (internal::random(0, 1) < density) ? internal::random() : 0; + if (v != 0) dst.fill(i, j) = v; } - } dst.endFill(); } #include -template -void doEigen(const char* name, const EigenSparseSelfAdjointMatrix& sm1, int flags = 0) +template void doEigen(const char *name, const EigenSparseSelfAdjointMatrix &sm1, int flags = 0) { std::cout << name << "..." << std::flush; BenchTimer timer; timer.start(); - SparseLLT chol(sm1, flags); + SparseLLT chol(sm1, flags); timer.stop(); std::cout << ":\t" << timer.value() << endl; std::cout << " nnz: " << sm1.nonZeros() << " => " << chol.matrixL().nonZeros() << "\n"; -// std::cout << "sparse\n" << chol.matrixL() << "%\n"; + // std::cout << "sparse\n" << chol.matrixL() << "%\n"; } int main(int argc, char *argv[]) @@ -86,27 +91,26 @@ int main(int argc, char *argv[]) bool densedone = false; - //for (float density = DENSITY; density>=MINDENSITY; density*=0.5) -// float density = 0.5; + // for (float density = DENSITY; density>=MINDENSITY; density*=0.5) + // float density = 0.5; { EigenSparseSelfAdjointMatrix sm1(rows, cols); std::cout << "Generate sparse matrix (might take a while)...\n"; fillSpdMatrix(density, rows, cols, sm1); std::cout << "DONE\n\n"; - // dense matrices - #ifdef DENSEMATRIX - if (!densedone) - { +// dense matrices +#ifdef DENSEMATRIX + if (!densedone) { densedone = true; - std::cout << "Eigen Dense\t" << density*100 << "%\n"; - DenseMatrix m1(rows,cols); + std::cout << "Eigen Dense\t" << density * 100 << "%\n"; + DenseMatrix m1(rows, cols); eiToDense(sm1, m1); m1 = (m1 + m1.transpose()).eval(); m1.diagonal() *= 0.5; -// BENCH(LLT chol(m1);) -// std::cout << "dense:\t" << timer.value() << endl; + // BENCH(LLT chol(m1);) + // std::cout << "dense:\t" << timer.value() << endl; BenchTimer timer; timer.start(); @@ -114,27 +118,26 @@ int main(int argc, char *argv[]) timer.stop(); std::cout << "dense:\t" << timer.value() << endl; int count = 0; - for (int j=0; j("Eigen/Sparse", sm1, Eigen::IncompleteFactorization); - #ifdef EIGEN_CHOLMOD_SUPPORT +#ifdef EIGEN_CHOLMOD_SUPPORT doEigen("Eigen/Cholmod", sm1, Eigen::IncompleteFactorization); - #endif +#endif - #ifdef EIGEN_TAUCS_SUPPORT +#ifdef EIGEN_TAUCS_SUPPORT doEigen("Eigen/Taucs", sm1, Eigen::IncompleteFactorization); - #endif +#endif - #if 0 +#if 0 // TAUCS { taucs_ccs_matrix A = sm1.asTaucsMatrix(); @@ -153,7 +156,7 @@ int main(int argc, char *argv[]) } // CHOLMOD - #ifdef EIGEN_CHOLMOD_SUPPORT +#ifdef EIGEN_CHOLMOD_SUPPORT { cholmod_common c; cholmod_start (&c); @@ -202,15 +205,11 @@ int main(int argc, char *argv[]) // std::cout << chol->values.s[i] << " "; // } } - #endif - - #endif - - +#endif +#endif } return 0; } - diff --git a/filmulator-gui/core/nlmeans/eigen/bench/sparse_dense_product.cpp b/filmulator-gui/core/nlmeans/eigen/bench/sparse_dense_product.cpp index f3f51940..9d9c7e6f 100644 --- a/filmulator-gui/core/nlmeans/eigen/bench/sparse_dense_product.cpp +++ b/filmulator-gui/core/nlmeans/eigen/bench/sparse_dense_product.cpp @@ -1,8 +1,9 @@ -//g++ -O3 -g0 -DNDEBUG sparse_product.cpp -I.. -I/home/gael/Coding/LinearAlgebra/mtl4/ -DDENSITY=0.005 -DSIZE=10000 && ./a.out -//g++ -O3 -g0 -DNDEBUG sparse_product.cpp -I.. -I/home/gael/Coding/LinearAlgebra/mtl4/ -DDENSITY=0.05 -DSIZE=2000 && ./a.out -// -DNOGMM -DNOMTL -DCSPARSE -// -I /home/gael/Coding/LinearAlgebra/CSparse/Include/ /home/gael/Coding/LinearAlgebra/CSparse/Lib/libcsparse.a +// g++ -O3 -g0 -DNDEBUG sparse_product.cpp -I.. -I/home/gael/Coding/LinearAlgebra/mtl4/ -DDENSITY=0.005 -DSIZE=10000 && +// ./a.out g++ -O3 -g0 -DNDEBUG sparse_product.cpp -I.. -I/home/gael/Coding/LinearAlgebra/mtl4/ -DDENSITY=0.05 +// -DSIZE=2000 && ./a.out +// -DNOGMM -DNOMTL -DCSPARSE +// -I /home/gael/Coding/LinearAlgebra/CSparse/Include/ /home/gael/Coding/LinearAlgebra/CSparse/Lib/libcsparse.a #ifndef SIZE #define SIZE 650000 #endif @@ -25,26 +26,26 @@ #define NBTRIES 10 #endif -#define BENCH(X) \ - timer.reset(); \ - for (int _j=0; _j=MINDENSITY; density*=0.5) - { - //fillMatrix(density, rows, cols, sm1); + for (float density = DENSITY; density >= MINDENSITY; density *= 0.5) { + // fillMatrix(density, rows, cols, sm1); fillMatrix2(7, rows, cols, sm1); - // dense matrices - #ifdef DENSEMATRIX +// dense matrices +#ifdef DENSEMATRIX { - std::cout << "Eigen Dense\t" << density*100 << "%\n"; - DenseMatrix m1(rows,cols); + std::cout << "Eigen Dense\t" << density * 100 << "%\n"; + DenseMatrix m1(rows, cols); eiToDense(sm1, m1); timer.reset(); timer.start(); - for (int k=0; k m1(sm1); -// std::cout << "Eigen dyn-sparse\t" << m1.nonZeros()/float(m1.rows()*m1.cols())*100 << "%\n"; -// -// BENCH(for (int k=0; k m1(sm1); + // std::cout << "Eigen dyn-sparse\t" << m1.nonZeros()/float(m1.rows()*m1.cols())*100 << "%\n"; + // + // BENCH(for (int k=0; k gmmV1(cols), gmmV2(cols); - Map >(&gmmV1[0], cols) = v1; - Map >(&gmmV2[0], cols) = v2; + Map>(&gmmV1[0], cols) = v1; + Map>(&gmmV2[0], cols) = v2; - BENCH( asm("#myx"); gmm::mult(m1, gmmV1, gmmV2); asm("#myy"); ) + BENCH(asm("#myx"); gmm::mult(m1, gmmV1, gmmV2); asm("#myy");) std::cout << " a * v:\t" << timer.value() << endl; - BENCH( gmm::mult(gmm::transposed(m1), gmmV1, gmmV2); ) + BENCH(gmm::mult(gmm::transposed(m1), gmmV1, gmmV2);) std::cout << " a' * v:\t" << timer.value() << endl; } - #endif - - #ifndef NOUBLAS +#endif + +#ifndef NOUBLAS { - std::cout << "ublas sparse\t" << density*100 << "%\n"; - UBlasSparse m1(rows,cols); + std::cout << "ublas sparse\t" << density * 100 << "%\n"; + UBlasSparse m1(rows, cols); eiToUblas(sm1, m1); - + boost::numeric::ublas::vector uv1, uv2; - eiToUblasVec(v1,uv1); - eiToUblasVec(v2,uv2); + eiToUblasVec(v1, uv1); + eiToUblasVec(v2, uv2); -// std::vector gmmV1(cols), gmmV2(cols); -// Map >(&gmmV1[0], cols) = v1; -// Map >(&gmmV2[0], cols) = v2; + // std::vector gmmV1(cols), gmmV2(cols); + // Map >(&gmmV1[0], cols) = v1; + // Map >(&gmmV2[0], cols) = v2; - BENCH( uv2 = boost::numeric::ublas::prod(m1, uv1); ) + BENCH(uv2 = boost::numeric::ublas::prod(m1, uv1);) std::cout << " a * v:\t" << timer.value() << endl; -// BENCH( boost::ublas::prod(gmm::transposed(m1), gmmV1, gmmV2); ) -// std::cout << " a' * v:\t" << timer.value() << endl; + // BENCH( boost::ublas::prod(gmm::transposed(m1), gmmV1, gmmV2); ) + // std::cout << " a' * v:\t" << timer.value() << endl; } - #endif +#endif - // MTL4 - #ifndef NOMTL +// MTL4 +#ifndef NOMTL { - std::cout << "MTL4\t" << density*100 << "%\n"; - MtlSparse m1(rows,cols); + std::cout << "MTL4\t" << density * 100 << "%\n"; + MtlSparse m1(rows, cols); eiToMtl(sm1, m1); mtl::dense_vector mtlV1(cols, 1.0); mtl::dense_vector mtlV2(cols, 1.0); timer.reset(); timer.start(); - for (int k=0; k VectorX; +typedef Matrix VectorX; #include template -void doEigen(const char* name, const EigenSparseMatrix& sm1, const VectorX& b, VectorX& x, int flags = 0) +void doEigen(const char *name, const EigenSparseMatrix &sm1, const VectorX &b, VectorX &x, int flags = 0) { std::cout << name << "..." << std::flush; - BenchTimer timer; timer.start(); - SparseLU lu(sm1, flags); + BenchTimer timer; + timer.start(); + SparseLU lu(sm1, flags); timer.stop(); if (lu.succeeded()) std::cout << ":\t" << timer.value() << endl; - else - { + else { std::cout << ":\t FAILED" << endl; return; } bool ok; - timer.reset(); timer.start(); - ok = lu.solve(b,&x); + timer.reset(); + timer.start(); + ok = lu.solve(b, &x); timer.stop(); if (ok) std::cout << " solve:\t" << timer.value() << endl; else std::cout << " solve:\t" << " FAILED" << endl; - //std::cout << x.transpose() << "\n"; + // std::cout << x.transpose() << "\n"; } int main(int argc, char *argv[]) @@ -81,19 +82,18 @@ int main(int argc, char *argv[]) bool densedone = false; - //for (float density = DENSITY; density>=MINDENSITY; density*=0.5) -// float density = 0.5; + // for (float density = DENSITY; density>=MINDENSITY; density*=0.5) + // float density = 0.5; { EigenSparseMatrix sm1(rows, cols); fillMatrix(density, rows, cols, sm1); - // dense matrices - #ifdef DENSEMATRIX - if (!densedone) - { +// dense matrices +#ifdef DENSEMATRIX + if (!densedone) { densedone = true; - std::cout << "Eigen Dense\t" << density*100 << "%\n"; - DenseMatrix m1(rows,cols); + std::cout << "Eigen Dense\t" << density * 100 << "%\n"; + DenseMatrix m1(rows, cols); eiToDense(sm1, m1); BenchTimer timer; @@ -104,29 +104,27 @@ int main(int argc, char *argv[]) timer.reset(); timer.start(); - lu.solve(b,&x); + lu.solve(b, &x); timer.stop(); std::cout << " solve:\t" << timer.value() << endl; -// std::cout << b.transpose() << "\n"; -// std::cout << x.transpose() << "\n"; + // std::cout << b.transpose() << "\n"; + // std::cout << x.transpose() << "\n"; } - #endif +#endif - #ifdef EIGEN_UMFPACK_SUPPORT +#ifdef EIGEN_UMFPACK_SUPPORT x.setZero(); doEigen("Eigen/UmfPack (auto)", sm1, b, x, 0); - #endif +#endif - #ifdef EIGEN_SUPERLU_SUPPORT +#ifdef EIGEN_SUPERLU_SUPPORT x.setZero(); doEigen("Eigen/SuperLU (nat)", sm1, b, x, Eigen::NaturalOrdering); -// doEigen("Eigen/SuperLU (MD AT+A)", sm1, b, x, Eigen::MinimumDegree_AT_PLUS_A); -// doEigen("Eigen/SuperLU (MD ATA)", sm1, b, x, Eigen::MinimumDegree_ATA); + // doEigen("Eigen/SuperLU (MD AT+A)", sm1, b, x, Eigen::MinimumDegree_AT_PLUS_A); + // doEigen("Eigen/SuperLU (MD ATA)", sm1, b, x, Eigen::MinimumDegree_ATA); doEigen("Eigen/SuperLU (COLAMD)", sm1, b, x, Eigen::ColApproxMinimumDegree); - #endif - +#endif } return 0; } - diff --git a/filmulator-gui/core/nlmeans/eigen/bench/sparse_product.cpp b/filmulator-gui/core/nlmeans/eigen/bench/sparse_product.cpp index d2fc44f0..dd6a12c6 100644 --- a/filmulator-gui/core/nlmeans/eigen/bench/sparse_product.cpp +++ b/filmulator-gui/core/nlmeans/eigen/bench/sparse_product.cpp @@ -1,8 +1,9 @@ -//g++ -O3 -g0 -DNDEBUG sparse_product.cpp -I.. -I/home/gael/Coding/LinearAlgebra/mtl4/ -DDENSITY=0.005 -DSIZE=10000 && ./a.out -//g++ -O3 -g0 -DNDEBUG sparse_product.cpp -I.. -I/home/gael/Coding/LinearAlgebra/mtl4/ -DDENSITY=0.05 -DSIZE=2000 && ./a.out -// -DNOGMM -DNOMTL -DCSPARSE -// -I /home/gael/Coding/LinearAlgebra/CSparse/Include/ /home/gael/Coding/LinearAlgebra/CSparse/Lib/libcsparse.a +// g++ -O3 -g0 -DNDEBUG sparse_product.cpp -I.. -I/home/gael/Coding/LinearAlgebra/mtl4/ -DDENSITY=0.005 -DSIZE=10000 && +// ./a.out g++ -O3 -g0 -DNDEBUG sparse_product.cpp -I.. -I/home/gael/Coding/LinearAlgebra/mtl4/ -DDENSITY=0.05 +// -DSIZE=2000 && ./a.out +// -DNOGMM -DNOMTL -DCSPARSE +// -I /home/gael/Coding/LinearAlgebra/CSparse/Include/ /home/gael/Coding/LinearAlgebra/CSparse/Lib/libcsparse.a #include @@ -18,22 +19,22 @@ #define REPEAT 1 #endif -#include +#include "BenchSparseUtil.h" #include "BenchTimer.h" #include "BenchUtil.h" -#include "BenchSparseUtil.h" +#include #ifndef NBTRIES #define NBTRIES 1 #endif -#define BENCH(X) \ - timer.reset(); \ - for (int _j=0; _j1; nnzPerCol/=1.1) - { + for (int nnzPerCol = NNZPERCOL; nnzPerCol > 1; nnzPerCol /= 1.1) { sm1.setZero(); sm2.setZero(); fillMatrix2(nnzPerCol, rows, cols, sm1); fillMatrix2(nnzPerCol, rows, cols, sm2); -// std::cerr << "filling OK\n"; + // std::cerr << "filling OK\n"; - // dense matrices - #ifdef DENSEMATRIX +// dense matrices +#ifdef DENSEMATRIX { std::cout << "Eigen Dense\t" << nnzPerCol << "%\n"; - DenseMatrix m1(rows,cols), m2(rows,cols), m3(rows,cols); + DenseMatrix m1(rows, cols), m2(rows, cols), m3(rows, cols); eiToDense(sm1, m1); eiToDense(sm2, m2); timer.reset(); timer.start(); - for (int k=0; k m1(sm1), m2(sm2), m3(sm3); - std::cout << "Eigen dyn-sparse\t" << m1.nonZeros()/(float(m1.rows())*float(m1.cols()))*100 << "% * " - << m2.nonZeros()/(float(m2.rows())*float(m2.cols()))*100 << "%\n"; +// eigen dyn-sparse matrices +/*{ + DynamicSparseMatrix m1(sm1), m2(sm2), m3(sm3); + std::cout << "Eigen dyn-sparse\t" << m1.nonZeros()/(float(m1.rows())*float(m1.cols()))*100 << "% * " + << m2.nonZeros()/(float(m2.rows())*float(m2.cols()))*100 << "%\n"; // timer.reset(); // timer.start(); - BENCH(for (int k=0; k #include #include #include +#include #ifndef SIZE #define SIZE 10000 @@ -29,97 +29,97 @@ #define NBTRIES 10 #endif -#define BENCH(X) \ - timer.reset(); \ - for (int _j=0; _j -void dostuff(const char* name, EigenSparseMatrix& sm1) +template void dostuff(const char *name, EigenSparseMatrix &sm1) { int rows = sm1.rows(); int cols = sm1.cols(); sm1.setZero(); BenchTimer t; - SetterType* set1 = new SetterType(sm1); - t.reset(); t.start(); - for (int k=0; k(0,rows-1),internal::random(0,cols-1)) += 1; + SetterType *set1 = new SetterType(sm1); + t.reset(); + t.start(); + for (int k = 0; k < nentries; ++k) + (*set1)(internal::random(0, rows - 1), internal::random(0, cols - 1)) += 1; t.stop(); - std::cout << "std::map => \t" << t.value()-rtime - << " nnz=" << set1->nonZeros() << std::flush; + std::cout << "std::map => \t" << t.value() - rtime << " nnz=" << set1->nonZeros() << std::flush; // getchar(); - t.reset(); t.start(); delete set1; t.stop(); + t.reset(); + t.start(); + delete set1; + t.stop(); std::cout << " back: \t" << t.value() << "\n"; } - + int main(int argc, char *argv[]) { int rows = SIZE; int cols = SIZE; float density = DENSITY; - EigenSparseMatrix sm1(rows,cols), sm2(rows,cols); + EigenSparseMatrix sm1(rows, cols), sm2(rows, cols); - nentries = rows*cols*density; + nentries = rows * cols * density; std::cout << "n = " << nentries << "\n"; int dummy; BenchTimer t; - t.reset(); t.start(); - for (int k=0; k(0,rows-1) + internal::random(0,cols-1); + t.reset(); + t.start(); + for (int k = 0; k < nentries; ++k) dummy = internal::random(0, rows - 1) + internal::random(0, cols - 1); t.stop(); rtime = t.value(); std::cout << "rtime = " << rtime << " (" << dummy << ")\n\n"; const int Bits = 6; - for (;;) - { - dostuff >("std::map ", sm1); - dostuff >("gnu::hash_map", sm1); - dostuff >("google::dense", sm1); - dostuff >("google::sparse", sm1); - -// { -// RandomSetter set1(sm1); -// t.reset(); t.start(); -// for (int k=0; k(0,rows-1),internal::random(0,cols-1)) += 1; -// t.stop(); -// std::cout << "gnu::hash_map => \t" << t.value()-rtime -// << " nnz=" << set1.nonZeros() << "\n";getchar(); -// } -// { -// RandomSetter set1(sm1); -// t.reset(); t.start(); -// for (int k=0; k(0,rows-1),internal::random(0,cols-1)) += 1; -// t.stop(); -// std::cout << "google::dense => \t" << t.value()-rtime -// << " nnz=" << set1.nonZeros() << "\n";getchar(); -// } -// { -// RandomSetter set1(sm1); -// t.reset(); t.start(); -// for (int k=0; k(0,rows-1),internal::random(0,cols-1)) += 1; -// t.stop(); -// std::cout << "google::sparse => \t" << t.value()-rtime -// << " nnz=" << set1.nonZeros() << "\n";getchar(); -// } + for (;;) { + dostuff>("std::map ", sm1); + dostuff>("gnu::hash_map", sm1); + dostuff>("google::dense", sm1); + dostuff>("google::sparse", sm1); + + // { + // RandomSetter set1(sm1); + // t.reset(); t.start(); + // for (int k=0; k(0,rows-1),internal::random(0,cols-1)) += 1; + // t.stop(); + // std::cout << "gnu::hash_map => \t" << t.value()-rtime + // << " nnz=" << set1.nonZeros() << "\n";getchar(); + // } + // { + // RandomSetter set1(sm1); + // t.reset(); t.start(); + // for (int k=0; k(0,rows-1),internal::random(0,cols-1)) += 1; + // t.stop(); + // std::cout << "google::dense => \t" << t.value()-rtime + // << " nnz=" << set1.nonZeros() << "\n";getchar(); + // } + // { + // RandomSetter set1(sm1); + // t.reset(); t.start(); + // for (int k=0; k(0,rows-1),internal::random(0,cols-1)) += 1; + // t.stop(); + // std::cout << "google::sparse => \t" << t.value()-rtime + // << " nnz=" << set1.nonZeros() << "\n";getchar(); + // } std::cout << "\n\n"; } return 0; } - diff --git a/filmulator-gui/core/nlmeans/eigen/bench/sparse_setter.cpp b/filmulator-gui/core/nlmeans/eigen/bench/sparse_setter.cpp index a9f0b11c..2b495245 100644 --- a/filmulator-gui/core/nlmeans/eigen/bench/sparse_setter.cpp +++ b/filmulator-gui/core/nlmeans/eigen/bench/sparse_setter.cpp @@ -1,8 +1,9 @@ -//g++ -O3 -g0 -DNDEBUG sparse_product.cpp -I.. -I/home/gael/Coding/LinearAlgebra/mtl4/ -DDENSITY=0.005 -DSIZE=10000 && ./a.out -//g++ -O3 -g0 -DNDEBUG sparse_product.cpp -I.. -I/home/gael/Coding/LinearAlgebra/mtl4/ -DDENSITY=0.05 -DSIZE=2000 && ./a.out -// -DNOGMM -DNOMTL -DCSPARSE -// -I /home/gael/Coding/LinearAlgebra/CSparse/Include/ /home/gael/Coding/LinearAlgebra/CSparse/Lib/libcsparse.a +// g++ -O3 -g0 -DNDEBUG sparse_product.cpp -I.. -I/home/gael/Coding/LinearAlgebra/mtl4/ -DDENSITY=0.005 -DSIZE=10000 && +// ./a.out g++ -O3 -g0 -DNDEBUG sparse_product.cpp -I.. -I/home/gael/Coding/LinearAlgebra/mtl4/ -DDENSITY=0.05 +// -DSIZE=2000 && ./a.out +// -DNOGMM -DNOMTL -DCSPARSE +// -I /home/gael/Coding/LinearAlgebra/CSparse/Include/ /home/gael/Coding/LinearAlgebra/CSparse/Lib/libcsparse.a #ifndef SIZE #define SIZE 100000 #endif @@ -33,30 +34,30 @@ #define CHECK_MEM // #define CHECK_MEM std/**/::cout << "check mem\n"; getchar(); -#define BENCH(X) \ - timer.reset(); \ - for (int _j=0; _j Coordinates; typedef std::vector Values; -EIGEN_DONT_INLINE Scalar* setinnerrand_eigen(const Coordinates& coords, const Values& vals); -EIGEN_DONT_INLINE Scalar* setrand_eigen_dynamic(const Coordinates& coords, const Values& vals); -EIGEN_DONT_INLINE Scalar* setrand_eigen_compact(const Coordinates& coords, const Values& vals); -EIGEN_DONT_INLINE Scalar* setrand_eigen_sumeq(const Coordinates& coords, const Values& vals); -EIGEN_DONT_INLINE Scalar* setrand_eigen_gnu_hash(const Coordinates& coords, const Values& vals); -EIGEN_DONT_INLINE Scalar* setrand_eigen_google_dense(const Coordinates& coords, const Values& vals); -EIGEN_DONT_INLINE Scalar* setrand_eigen_google_sparse(const Coordinates& coords, const Values& vals); -EIGEN_DONT_INLINE Scalar* setrand_scipy(const Coordinates& coords, const Values& vals); -EIGEN_DONT_INLINE Scalar* setrand_ublas_mapped(const Coordinates& coords, const Values& vals); -EIGEN_DONT_INLINE Scalar* setrand_ublas_coord(const Coordinates& coords, const Values& vals); -EIGEN_DONT_INLINE Scalar* setrand_ublas_compressed(const Coordinates& coords, const Values& vals); -EIGEN_DONT_INLINE Scalar* setrand_ublas_genvec(const Coordinates& coords, const Values& vals); -EIGEN_DONT_INLINE Scalar* setrand_mtl(const Coordinates& coords, const Values& vals); +EIGEN_DONT_INLINE Scalar *setinnerrand_eigen(const Coordinates &coords, const Values &vals); +EIGEN_DONT_INLINE Scalar *setrand_eigen_dynamic(const Coordinates &coords, const Values &vals); +EIGEN_DONT_INLINE Scalar *setrand_eigen_compact(const Coordinates &coords, const Values &vals); +EIGEN_DONT_INLINE Scalar *setrand_eigen_sumeq(const Coordinates &coords, const Values &vals); +EIGEN_DONT_INLINE Scalar *setrand_eigen_gnu_hash(const Coordinates &coords, const Values &vals); +EIGEN_DONT_INLINE Scalar *setrand_eigen_google_dense(const Coordinates &coords, const Values &vals); +EIGEN_DONT_INLINE Scalar *setrand_eigen_google_sparse(const Coordinates &coords, const Values &vals); +EIGEN_DONT_INLINE Scalar *setrand_scipy(const Coordinates &coords, const Values &vals); +EIGEN_DONT_INLINE Scalar *setrand_ublas_mapped(const Coordinates &coords, const Values &vals); +EIGEN_DONT_INLINE Scalar *setrand_ublas_coord(const Coordinates &coords, const Values &vals); +EIGEN_DONT_INLINE Scalar *setrand_ublas_compressed(const Coordinates &coords, const Values &vals); +EIGEN_DONT_INLINE Scalar *setrand_ublas_genvec(const Coordinates &coords, const Values &vals); +EIGEN_DONT_INLINE Scalar *setrand_mtl(const Coordinates &coords, const Values &vals); int main(int argc, char *argv[]) { @@ -67,228 +68,203 @@ int main(int argc, char *argv[]) BenchTimer timer; Coordinates coords; Values values; - if(fullyrand) - { + if (fullyrand) { Coordinates pool; - pool.reserve(cols*NBPERROW); + pool.reserve(cols * NBPERROW); std::cerr << "fill pool" << "\n"; - for (int i=0; i stencil(SIZE,SIZE); - Vector2i ij(internal::random(0,rows-1),internal::random(0,cols-1)); -// if(stencil.coeffRef(ij.x(), ij.y())==0) + for (int i = 0; i < cols * NBPERROW;) { + // DynamicSparseMatrix stencil(SIZE,SIZE); + Vector2i ij(internal::random(0, rows - 1), internal::random(0, cols - 1)); + // if(stencil.coeffRef(ij.x(), ij.y())==0) { -// stencil.coeffRef(ij.x(), ij.y()) = 1; + // stencil.coeffRef(ij.x(), ij.y()) = 1; pool.push_back(ij); - } ++i; } std::cerr << "pool ok" << "\n"; - int n = cols*NBPERROW*KK; + int n = cols * NBPERROW * KK; coords.reserve(n); values.reserve(n); - for (int i=0; i(0,pool.size()); + for (int i = 0; i < n; ++i) { + int i = internal::random(0, pool.size()); coords.push_back(pool[i]); values.push_back(internal::random()); } + } else { + for (int j = 0; j < cols; ++j) + for (int i = 0; i < NBPERROW; ++i) { + coords.push_back(Vector2i(internal::random(0, rows - 1), j)); + values.push_back(internal::random()); + } } - else + std::cout << "nnz = " << coords.size() << "\n"; + CHECK_MEM + +// dense matrices +#ifdef DENSEMATRIX { - for (int j=0; j(0,rows-1),j)); - values.push_back(internal::random()); - } + BENCH(setrand_eigen_dense(coords, values);) + std::cout << "Eigen Dense\t" << timer.value() << "\n"; } - std::cout << "nnz = " << coords.size() << "\n"; - CHECK_MEM +#endif - // dense matrices - #ifdef DENSEMATRIX - { - BENCH(setrand_eigen_dense(coords,values);) - std::cout << "Eigen Dense\t" << timer.value() << "\n"; - } - #endif - - // eigen sparse matrices -// if (!fullyrand) -// { -// BENCH(setinnerrand_eigen(coords,values);) -// std::cout << "Eigen fillrand\t" << timer.value() << "\n"; -// } - { - BENCH(setrand_eigen_dynamic(coords,values);) - std::cout << "Eigen dynamic\t" << timer.value() << "\n"; - } -// { -// BENCH(setrand_eigen_compact(coords,values);) -// std::cout << "Eigen compact\t" << timer.value() << "\n"; -// } - { - BENCH(setrand_eigen_sumeq(coords,values);) - std::cout << "Eigen sumeq\t" << timer.value() << "\n"; - } - { -// BENCH(setrand_eigen_gnu_hash(coords,values);) -// std::cout << "Eigen std::map\t" << timer.value() << "\n"; - } - { - BENCH(setrand_scipy(coords,values);) - std::cout << "scipy\t" << timer.value() << "\n"; - } - #ifndef NOGOOGLE - { - BENCH(setrand_eigen_google_dense(coords,values);) - std::cout << "Eigen google dense\t" << timer.value() << "\n"; - } - { - BENCH(setrand_eigen_google_sparse(coords,values);) - std::cout << "Eigen google sparse\t" << timer.value() << "\n"; - } - #endif + // eigen sparse matrices + // if (!fullyrand) + // { + // BENCH(setinnerrand_eigen(coords,values);) + // std::cout << "Eigen fillrand\t" << timer.value() << "\n"; + // } + { + BENCH(setrand_eigen_dynamic(coords, values);) + std::cout << "Eigen dynamic\t" << timer.value() << "\n"; + } + // { + // BENCH(setrand_eigen_compact(coords,values);) + // std::cout << "Eigen compact\t" << timer.value() << "\n"; + // } + { + BENCH(setrand_eigen_sumeq(coords, values);) + std::cout << "Eigen sumeq\t" << timer.value() << "\n"; + } + { + // BENCH(setrand_eigen_gnu_hash(coords,values);) + // std::cout << "Eigen std::map\t" << timer.value() << "\n"; + } + { + BENCH(setrand_scipy(coords, values);) + std::cout << "scipy\t" << timer.value() << "\n"; + } +#ifndef NOGOOGLE + { + BENCH(setrand_eigen_google_dense(coords, values);) + std::cout << "Eigen google dense\t" << timer.value() << "\n"; + } + { + BENCH(setrand_eigen_google_sparse(coords, values);) + std::cout << "Eigen google sparse\t" << timer.value() << "\n"; + } +#endif - #ifndef NOUBLAS - { -// BENCH(setrand_ublas_mapped(coords,values);) -// std::cout << "ublas mapped\t" << timer.value() << "\n"; - } - { - BENCH(setrand_ublas_genvec(coords,values);) - std::cout << "ublas vecofvec\t" << timer.value() << "\n"; - } - /*{ - timer.reset(); - timer.start(); - for (int k=0; k mat(SIZE,SIZE); - //mat.startFill(2000000/*coords.size()*/); - for (int i=0; i mat(SIZE, SIZE); + // mat.startFill(2000000/*coords.size()*/); + for (int i = 0; i < coords.size(); ++i) { mat.insert(coords[i].x(), coords[i].y()) = vals[i]; } mat.finalize(); CHECK_MEM; return 0; } -EIGEN_DONT_INLINE Scalar* setrand_eigen_dynamic(const Coordinates& coords, const Values& vals) +EIGEN_DONT_INLINE Scalar *setrand_eigen_dynamic(const Coordinates &coords, const Values &vals) { using namespace Eigen; - DynamicSparseMatrix mat(SIZE,SIZE); - mat.reserve(coords.size()/10); - for (int i=0; i mat(SIZE, SIZE); + mat.reserve(coords.size() / 10); + for (int i = 0; i < coords.size(); ++i) { mat.coeffRef(coords[i].x(), coords[i].y()) += vals[i]; } mat.finalize(); CHECK_MEM; return &mat.coeffRef(coords[0].x(), coords[0].y()); } -EIGEN_DONT_INLINE Scalar* setrand_eigen_sumeq(const Coordinates& coords, const Values& vals) +EIGEN_DONT_INLINE Scalar *setrand_eigen_sumeq(const Coordinates &coords, const Values &vals) { using namespace Eigen; - int n = coords.size()/KK; - DynamicSparseMatrix mat(SIZE,SIZE); - for (int j=0; j aux(SIZE,SIZE); + int n = coords.size() / KK; + DynamicSparseMatrix mat(SIZE, SIZE); + for (int j = 0; j < KK; ++j) { + DynamicSparseMatrix aux(SIZE, SIZE); mat.reserve(n); - for (int i=j*n; i<(j+1)*n; ++i) - { - aux.insert(coords[i].x(), coords[i].y()) += vals[i]; - } + for (int i = j * n; i < (j + 1) * n; ++i) { aux.insert(coords[i].x(), coords[i].y()) += vals[i]; } aux.finalize(); mat += aux; } return &mat.coeffRef(coords[0].x(), coords[0].y()); } -EIGEN_DONT_INLINE Scalar* setrand_eigen_compact(const Coordinates& coords, const Values& vals) +EIGEN_DONT_INLINE Scalar *setrand_eigen_compact(const Coordinates &coords, const Values &vals) { using namespace Eigen; - DynamicSparseMatrix setter(SIZE,SIZE); - setter.reserve(coords.size()/10); - for (int i=0; i setter(SIZE, SIZE); + setter.reserve(coords.size() / 10); + for (int i = 0; i < coords.size(); ++i) { setter.coeffRef(coords[i].x(), coords[i].y()) += vals[i]; } SparseMatrix mat = setter; CHECK_MEM; return &mat.coeffRef(coords[0].x(), coords[0].y()); } -EIGEN_DONT_INLINE Scalar* setrand_eigen_gnu_hash(const Coordinates& coords, const Values& vals) +EIGEN_DONT_INLINE Scalar *setrand_eigen_gnu_hash(const Coordinates &coords, const Values &vals) { using namespace Eigen; - SparseMatrix mat(SIZE,SIZE); + SparseMatrix mat(SIZE, SIZE); { - RandomSetter, StdMapTraits > setter(mat); - for (int i=0; i, StdMapTraits> setter(mat); + for (int i = 0; i < coords.size(); ++i) { setter(coords[i].x(), coords[i].y()) += vals[i]; } CHECK_MEM; } return &mat.coeffRef(coords[0].x(), coords[0].y()); } #ifndef NOGOOGLE -EIGEN_DONT_INLINE Scalar* setrand_eigen_google_dense(const Coordinates& coords, const Values& vals) +EIGEN_DONT_INLINE Scalar *setrand_eigen_google_dense(const Coordinates &coords, const Values &vals) { using namespace Eigen; - SparseMatrix mat(SIZE,SIZE); + SparseMatrix mat(SIZE, SIZE); { RandomSetter, GoogleDenseHashMapTraits> setter(mat); - for (int i=0; i mat(SIZE,SIZE); + SparseMatrix mat(SIZE, SIZE); { RandomSetter, GoogleSparseHashMapTraits> setter(mat); - for (int i=0; i +template void coo_tocsr(const int n_row, - const int n_col, - const int nnz, - const Coordinates Aij, - const Values Ax, - int Bp[], - int Bj[], - T Bx[]) + const int n_col, + const int nnz, + const Coordinates Aij, + const Values Ax, + int Bp[], + int Bj[], + T Bx[]) { - //compute number of non-zero entries per row of A coo_tocsr - std::fill(Bp, Bp + n_row, 0); + // compute number of non-zero entries per row of A coo_tocsr + std::fill(Bp, Bp + n_row, 0); - for (int n = 0; n < nnz; n++){ - Bp[Aij[n].x()]++; - } + for (int n = 0; n < nnz; n++) { Bp[Aij[n].x()]++; } - //cumsum the nnz per row to get Bp[] - for(int i = 0, cumsum = 0; i < n_row; i++){ - int temp = Bp[i]; - Bp[i] = cumsum; - cumsum += temp; - } - Bp[n_row] = nnz; + // cumsum the nnz per row to get Bp[] + for (int i = 0, cumsum = 0; i < n_row; i++) { + int temp = Bp[i]; + Bp[i] = cumsum; + cumsum += temp; + } + Bp[n_row] = nnz; - //write Aj,Ax into Bj,Bx - for(int n = 0; n < nnz; n++){ - int row = Aij[n].x(); - int dest = Bp[row]; + // write Aj,Ax into Bj,Bx + for (int n = 0; n < nnz; n++) { + int row = Aij[n].x(); + int dest = Bp[row]; - Bj[dest] = Aij[n].y(); - Bx[dest] = Ax[n]; + Bj[dest] = Aij[n].y(); + Bx[dest] = Ax[n]; - Bp[row]++; - } + Bp[row]++; + } - for(int i = 0, last = 0; i <= n_row; i++){ - int temp = Bp[i]; - Bp[i] = last; - last = temp; - } + for (int i = 0, last = 0; i <= n_row; i++) { + int temp = Bp[i]; + Bp[i] = last; + last = temp; + } - //now Bp,Bj,Bx form a CSR representation (with possible duplicates) + // now Bp,Bj,Bx form a CSR representation (with possible duplicates) } -template< class T1, class T2 > -bool kv_pair_less(const std::pair& x, const std::pair& y){ - return x.first < y.first; +template bool kv_pair_less(const std::pair &x, const std::pair &y) +{ + return x.first < y.first; } -template -void csr_sort_indices(const I n_row, - const I Ap[], - I Aj[], - T Ax[]) +template void csr_sort_indices(const I n_row, const I Ap[], I Aj[], T Ax[]) { - std::vector< std::pair > temp; + std::vector> temp; - for(I i = 0; i < n_row; i++){ - I row_start = Ap[i]; - I row_end = Ap[i+1]; + for (I i = 0; i < n_row; i++) { + I row_start = Ap[i]; + I row_end = Ap[i + 1]; - temp.clear(); + temp.clear(); - for(I jj = row_start; jj < row_end; jj++){ - temp.push_back(std::make_pair(Aj[jj],Ax[jj])); - } + for (I jj = row_start; jj < row_end; jj++) { temp.push_back(std::make_pair(Aj[jj], Ax[jj])); } - std::sort(temp.begin(),temp.end(),kv_pair_less); + std::sort(temp.begin(), temp.end(), kv_pair_less); - for(I jj = row_start, n = 0; jj < row_end; jj++, n++){ - Aj[jj] = temp[n].first; - Ax[jj] = temp[n].second; - } + for (I jj = row_start, n = 0; jj < row_end; jj++, n++) { + Aj[jj] = temp[n].first; + Ax[jj] = temp[n].second; } + } } -template -void csr_sum_duplicates(const I n_row, - const I n_col, - I Ap[], - I Aj[], - T Ax[]) +template void csr_sum_duplicates(const I n_row, const I n_col, I Ap[], I Aj[], T Ax[]) { - I nnz = 0; - I row_end = 0; - for(I i = 0; i < n_row; i++){ - I jj = row_end; - row_end = Ap[i+1]; - while( jj < row_end ){ - I j = Aj[jj]; - T x = Ax[jj]; - jj++; - while( jj < row_end && Aj[jj] == j ){ - x += Ax[jj]; - jj++; - } - Aj[nnz] = j; - Ax[nnz] = x; - nnz++; - } - Ap[i+1] = nnz; + I nnz = 0; + I row_end = 0; + for (I i = 0; i < n_row; i++) { + I jj = row_end; + row_end = Ap[i + 1]; + while (jj < row_end) { + I j = Aj[jj]; + T x = Ax[jj]; + jj++; + while (jj < row_end && Aj[jj] == j) { + x += Ax[jj]; + jj++; + } + Aj[nnz] = j; + Ax[nnz] = x; + nnz++; } + Ap[i + 1] = nnz; + } } -EIGEN_DONT_INLINE Scalar* setrand_scipy(const Coordinates& coords, const Values& vals) +EIGEN_DONT_INLINE Scalar *setrand_scipy(const Coordinates &coords, const Values &vals) { using namespace Eigen; - SparseMatrix mat(SIZE,SIZE); + SparseMatrix mat(SIZE, SIZE); mat.resizeNonZeros(coords.size()); -// std::cerr << "setrand_scipy...\n"; - coo_tocsr(SIZE,SIZE, coords.size(), coords, vals, mat._outerIndexPtr(), mat._innerIndexPtr(), mat._valuePtr()); -// std::cerr << "coo_tocsr ok\n"; + // std::cerr << "setrand_scipy...\n"; + coo_tocsr( + SIZE, SIZE, coords.size(), coords, vals, mat._outerIndexPtr(), mat._innerIndexPtr(), mat._valuePtr()); + // std::cerr << "coo_tocsr ok\n"; csr_sort_indices(SIZE, mat._outerIndexPtr(), mat._innerIndexPtr(), mat._valuePtr()); @@ -422,16 +386,13 @@ EIGEN_DONT_INLINE Scalar* setrand_scipy(const Coordinates& coords, const Values& #ifndef NOUBLAS -EIGEN_DONT_INLINE Scalar* setrand_ublas_mapped(const Coordinates& coords, const Values& vals) +EIGEN_DONT_INLINE Scalar *setrand_ublas_mapped(const Coordinates &coords, const Values &vals) { using namespace boost; using namespace boost::numeric; using namespace boost::numeric::ublas; - mapped_matrix aux(SIZE,SIZE); - for (int i=0; i aux(SIZE, SIZE); + for (int i = 0; i < coords.size(); ++i) { aux(coords[i].x(), coords[i].y()) += vals[i]; } CHECK_MEM; compressed_matrix mat(aux); return 0;// &mat(coords[0].x(), coords[0].y()); @@ -461,25 +422,21 @@ EIGEN_DONT_INLINE Scalar* setrand_ublas_compressed(const Coordinates& coords, co } return 0;//&mat(coords[0].x(), coords[0].y()); }*/ -EIGEN_DONT_INLINE Scalar* setrand_ublas_genvec(const Coordinates& coords, const Values& vals) +EIGEN_DONT_INLINE Scalar *setrand_ublas_genvec(const Coordinates &coords, const Values &vals) { using namespace boost; using namespace boost::numeric; using namespace boost::numeric::ublas; -// ublas::vector > foo; - generalized_vector_of_vector > > aux(SIZE,SIZE); - for (int i=0; i > foo; + generalized_vector_of_vector>> aux(SIZE, SIZE); + for (int i = 0; i < coords.size(); ++i) { aux(coords[i].x(), coords[i].y()) += vals[i]; } CHECK_MEM; - compressed_matrix mat(aux); + compressed_matrix mat(aux); return 0;//&mat(coords[0].x(), coords[0].y()); } #endif #ifndef NOMTL -EIGEN_DONT_INLINE void setrand_mtl(const Coordinates& coords, const Values& vals); +EIGEN_DONT_INLINE void setrand_mtl(const Coordinates &coords, const Values &vals); #endif - diff --git a/filmulator-gui/core/nlmeans/eigen/bench/sparse_transpose.cpp b/filmulator-gui/core/nlmeans/eigen/bench/sparse_transpose.cpp index c9aacf5f..f2796c1d 100644 --- a/filmulator-gui/core/nlmeans/eigen/bench/sparse_transpose.cpp +++ b/filmulator-gui/core/nlmeans/eigen/bench/sparse_transpose.cpp @@ -1,7 +1,9 @@ -//g++ -O3 -g0 -DNDEBUG sparse_transpose.cpp -I.. -I/home/gael/Coding/LinearAlgebra/mtl4/ -DDENSITY=0.005 -DSIZE=10000 && ./a.out -// -DNOGMM -DNOMTL -// -DCSPARSE -I /home/gael/Coding/LinearAlgebra/CSparse/Include/ /home/gael/Coding/LinearAlgebra/CSparse/Lib/libcsparse.a +// g++ -O3 -g0 -DNDEBUG sparse_transpose.cpp -I.. -I/home/gael/Coding/LinearAlgebra/mtl4/ -DDENSITY=0.005 -DSIZE=10000 +// && ./a.out +// -DNOGMM -DNOMTL +// -DCSPARSE -I /home/gael/Coding/LinearAlgebra/CSparse/Include/ +// /home/gael/Coding/LinearAlgebra/CSparse/Lib/libcsparse.a #ifndef SIZE #define SIZE 10000 @@ -25,13 +27,13 @@ #define NBTRIES 10 #endif -#define BENCH(X) \ - timer.reset(); \ - for (int _j=0; _j=MINDENSITY; density*=0.5) - { + for (float density = DENSITY; density >= MINDENSITY; density *= 0.5) { fillMatrix(density, rows, cols, sm1); - // dense matrices - #ifdef DENSEMATRIX +// dense matrices +#ifdef DENSEMATRIX { - DenseMatrix m1(rows,cols), m3(rows,cols); + DenseMatrix m1(rows, cols), m3(rows, cols); eiToDense(sm1, m1); - BENCH(for (int k=0; k EigenSparseTriMatrix; -typedef SparseMatrix EigenSparseTriMatrixRow; +typedef SparseMatrix EigenSparseTriMatrix; +typedef SparseMatrix EigenSparseTriMatrixRow; -void fillMatrix(float density, int rows, int cols, EigenSparseTriMatrix& dst) +void fillMatrix(float density, int rows, int cols, EigenSparseTriMatrix &dst) { - dst.startFill(rows*cols*density); - for(int j = 0; j < cols; j++) - { - for(int i = 0; i < j; i++) - { - Scalar v = (internal::random(0,1) < density) ? internal::random() : 0; - if (v!=0) - dst.fill(i,j) = v; + dst.startFill(rows * cols * density); + for (int j = 0; j < cols; j++) { + for (int i = 0; i < j; i++) { + Scalar v = (internal::random(0, 1) < density) ? internal::random() : 0; + if (v != 0) dst.fill(i, j) = v; } - dst.fill(j,j) = internal::random(); + dst.fill(j, j) = internal::random(); } dst.endFill(); } @@ -59,130 +57,130 @@ int main(int argc, char *argv[]) int cols = SIZE; float density = DENSITY; BenchTimer timer; - #if 1 - EigenSparseTriMatrix sm1(rows,cols); - typedef Matrix DenseVector; +#if 1 + EigenSparseTriMatrix sm1(rows, cols); + typedef Matrix DenseVector; DenseVector b = DenseVector::Random(cols); DenseVector x = DenseVector::Random(cols); bool densedone = false; - for (float density = DENSITY; density>=MINDENSITY; density*=0.5) - { + for (float density = DENSITY; density >= MINDENSITY; density *= 0.5) { EigenSparseTriMatrix sm1(rows, cols); fillMatrix(density, rows, cols, sm1); - // dense matrices - #ifdef DENSEMATRIX - if (!densedone) - { +// dense matrices +#ifdef DENSEMATRIX + if (!densedone) { densedone = true; - std::cout << "Eigen Dense\t" << density*100 << "%\n"; - DenseMatrix m1(rows,cols); - Matrix m2(rows,cols); + std::cout << "Eigen Dense\t" << density * 100 << "%\n"; + DenseMatrix m1(rows, cols); + Matrix m2(rows, cols); eiToDense(sm1, m1); m2 = m1; BENCH(x = m1.marked().solveTriangular(b);) std::cout << " colmajor^-1 * b:\t" << timer.value() << endl; -// std::cerr << x.transpose() << "\n"; + // std::cerr << x.transpose() << "\n"; BENCH(x = m2.marked().solveTriangular(b);) std::cout << " rowmajor^-1 * b:\t" << timer.value() << endl; -// std::cerr << x.transpose() << "\n"; + // std::cerr << x.transpose() << "\n"; } - #endif +#endif // eigen sparse matrices { - std::cout << "Eigen sparse\t" << density*100 << "%\n"; + std::cout << "Eigen sparse\t" << density * 100 << "%\n"; EigenSparseTriMatrixRow sm2 = sm1; BENCH(x = sm1.solveTriangular(b);) std::cout << " colmajor^-1 * b:\t" << timer.value() << endl; -// std::cerr << x.transpose() << "\n"; + // std::cerr << x.transpose() << "\n"; BENCH(x = sm2.solveTriangular(b);) std::cout << " rowmajor^-1 * b:\t" << timer.value() << endl; -// std::cerr << x.transpose() << "\n"; - -// x = b; -// BENCH(sm1.inverseProductInPlace(x);) -// std::cout << " colmajor^-1 * b:\t" << timer.value() << " (inplace)" << endl; -// std::cerr << x.transpose() << "\n"; -// -// x = b; -// BENCH(sm2.inverseProductInPlace(x);) -// std::cout << " rowmajor^-1 * b:\t" << timer.value() << " (inplace)" << endl; -// std::cerr << x.transpose() << "\n"; + // std::cerr << x.transpose() << "\n"; + + // x = b; + // BENCH(sm1.inverseProductInPlace(x);) + // std::cout << " colmajor^-1 * b:\t" << timer.value() << " (inplace)" << endl; + // std::cerr << x.transpose() << "\n"; + // + // x = b; + // BENCH(sm2.inverseProductInPlace(x);) + // std::cout << " rowmajor^-1 * b:\t" << timer.value() << " (inplace)" << endl; + // std::cerr << x.transpose() << "\n"; } - - // CSparse - #ifdef CSPARSE +// CSparse +#ifdef CSPARSE { - std::cout << "CSparse \t" << density*100 << "%\n"; + std::cout << "CSparse \t" << density * 100 << "%\n"; cs *m1; eiToCSparse(sm1, m1); - BENCH(x = b; if (!cs_lsolve (m1, x.data())){std::cerr << "cs_lsolve failed\n"; break;}; ) + BENCH(x = b; if (!cs_lsolve(m1, x.data())) { + std::cerr << "cs_lsolve failed\n"; + break; + };) std::cout << " colmajor^-1 * b:\t" << timer.value() << endl; } - #endif +#endif - // GMM++ - #ifndef NOGMM +// GMM++ +#ifndef NOGMM { - std::cout << "GMM++ sparse\t" << density*100 << "%\n"; - GmmSparse m1(rows,cols); + std::cout << "GMM++ sparse\t" << density * 100 << "%\n"; + GmmSparse m1(rows, cols); gmm::csr_matrix m2; eiToGmm(sm1, m1); - gmm::copy(m1,m2); + gmm::copy(m1, m2); std::vector gmmX(cols), gmmB(cols); - Map >(&gmmX[0], cols) = x; - Map >(&gmmB[0], cols) = b; + Map>(&gmmX[0], cols) = x; + Map>(&gmmB[0], cols) = b; gmmX = gmmB; BENCH(gmm::upper_tri_solve(m1, gmmX, false);) std::cout << " colmajor^-1 * b:\t" << timer.value() << endl; -// std::cerr << Map >(&gmmX[0], cols).transpose() << "\n"; + // std::cerr << Map >(&gmmX[0], cols).transpose() << "\n"; gmmX = gmmB; BENCH(gmm::upper_tri_solve(m2, gmmX, false);) timer.stop(); std::cout << " rowmajor^-1 * b:\t" << timer.value() << endl; -// std::cerr << Map >(&gmmX[0], cols).transpose() << "\n"; + // std::cerr << Map >(&gmmX[0], cols).transpose() << "\n"; } - #endif +#endif - // MTL4 - #ifndef NOMTL +// MTL4 +#ifndef NOMTL { - std::cout << "MTL4\t" << density*100 << "%\n"; - MtlSparse m1(rows,cols); - MtlSparseRowMajor m2(rows,cols); + std::cout << "MTL4\t" << density * 100 << "%\n"; + MtlSparse m1(rows, cols); + MtlSparseRowMajor m2(rows, cols); eiToMtl(sm1, m1); m2 = m1; mtl::dense_vector x(rows, 1.0); mtl::dense_vector b(rows, 1.0); - BENCH(x = mtl::upper_trisolve(m1,b);) + BENCH(x = mtl::upper_trisolve(m1, b);) std::cout << " colmajor^-1 * b:\t" << timer.value() << endl; -// std::cerr << x << "\n"; + // std::cerr << x << "\n"; - BENCH(x = mtl::upper_trisolve(m2,b);) + BENCH(x = mtl::upper_trisolve(m2, b);) std::cout << " rowmajor^-1 * b:\t" << timer.value() << endl; -// std::cerr << x << "\n"; + // std::cerr << x << "\n"; } - #endif +#endif std::cout << "\n\n"; } - #endif +#endif - #if 0 +#if 0 // bench small matrices (in-place versus return bye value) { timer.reset(); @@ -213,8 +211,7 @@ int main(int argc, char *argv[]) } std::cout << "4x4 IP :\t" << timer.value() << endl; } - #endif +#endif return 0; } - diff --git a/filmulator-gui/core/nlmeans/eigen/bench/spbench/sp_solver.cpp b/filmulator-gui/core/nlmeans/eigen/bench/spbench/sp_solver.cpp index a1f4bac8..c0b44460 100644 --- a/filmulator-gui/core/nlmeans/eigen/bench/spbench/sp_solver.cpp +++ b/filmulator-gui/core/nlmeans/eigen/bench/spbench/sp_solver.cpp @@ -1,15 +1,15 @@ // Small bench routine for Eigen available in Eigen // (C) Desire NUENTSA WAKAM, INRIA -#include -#include -#include -#include #include #include +#include #include +#include +#include +#include #include -//#include +// #include #include // #include #include @@ -19,107 +19,112 @@ using namespace Eigen; int main(int argc, char **args) { - SparseMatrix A; + SparseMatrix A; typedef SparseMatrix::Index Index; typedef Matrix DenseMatrix; typedef Matrix DenseRhs; VectorXd b, x, tmp; - BenchTimer timer,totaltime; - //SparseLU > solver; -// SuperLU > solver; - ConjugateGradient, Lower,IncompleteCholesky > solver; - ifstream matrix_file; + BenchTimer timer, totaltime; + // SparseLU > solver; + // SuperLU > solver; + ConjugateGradient, Lower, IncompleteCholesky> solver; + ifstream matrix_file; string line; - int n; + int n; // Set parameters -// solver.iparm(IPARM_THREAD_NBR) = 4; + // solver.iparm(IPARM_THREAD_NBR) = 4; /* Fill the matrix with sparse matrix stored in Matrix-Market coordinate column-oriented format */ if (argc < 2) assert(false && "please, give the matrix market file "); - + timer.start(); totaltime.start(); loadMarket(A, args[1]); cout << "End charging matrix " << endl; - bool iscomplex=false, isvector=false; + bool iscomplex = false, isvector = false; int sym; getMarketHeader(args[1], sym, iscomplex, isvector); - if (iscomplex) { cout<< " Not for complex matrices \n"; return -1; } - if (isvector) { cout << "The provided file is not a matrix file\n"; return -1;} - if (sym != 0) { // symmetric matrices, only the lower part is stored - SparseMatrix temp; + if (iscomplex) { + cout << " Not for complex matrices \n"; + return -1; + } + if (isvector) { + cout << "The provided file is not a matrix file\n"; + return -1; + } + if (sym != 0) {// symmetric matrices, only the lower part is stored + SparseMatrix temp; temp = A; A = temp.selfadjointView(); } timer.stop(); - + n = A.cols(); // ====== TESTS FOR SPARSE TUTORIAL ====== -// cout<< "OuterSize " << A.outerSize() << " inner " << A.innerSize() << endl; -// SparseMatrix mat1(A); -// SparseMatrix mat2; -// cout << " norm of A " << mat1.norm() << endl; ; -// PermutationMatrix perm(n); -// perm.resize(n,1); -// perm.indices().setLinSpaced(n, 0, n-1); -// mat2 = perm * mat1; -// mat.subrows(); -// mat2.resize(n,n); -// mat2.reserve(10); -// mat2.setConstant(); -// std::cout<< "NORM " << mat1.squaredNorm()<< endl; + // cout<< "OuterSize " << A.outerSize() << " inner " << A.innerSize() << endl; + // SparseMatrix mat1(A); + // SparseMatrix mat2; + // cout << " norm of A " << mat1.norm() << endl; ; + // PermutationMatrix perm(n); + // perm.resize(n,1); + // perm.indices().setLinSpaced(n, 0, n-1); + // mat2 = perm * mat1; + // mat.subrows(); + // mat2.resize(n,n); + // mat2.reserve(10); + // mat2.setConstant(); + // std::cout<< "NORM " << mat1.squaredNorm()<< endl; - cout<< "Time to load the matrix " << timer.value() < 2) loadMarketVector(b, args[2]); - else - { + else { b.resize(n); tmp.resize(n); -// tmp.setRandom(); - for (int i = 0; i < n; i++) tmp(i) = i; - b = A * tmp ; + // tmp.setRandom(); + for (int i = 0; i < n; i++) tmp(i) = i; + b = A * tmp; } -// Scaling > scal; -// scal.computeRef(A); -// b = scal.LeftScaling().cwiseProduct(b); + // Scaling > scal; + // scal.computeRef(A); + // b = scal.LeftScaling().cwiseProduct(b); /* Compute the factorization */ - cout<< "Starting the factorization "<< endl; + cout << "Starting the factorization " << endl; timer.reset(); - timer.start(); - cout<< "Size of Input Matrix "<< b.size()<<"\n\n"; - cout<< "Rows and columns "<< A.rows() <<" " < Sets the relative tolerance for iterative solvers (default 1e-08) \n\n"; - cout<< " --maxits Sets the maximum number of iterations (default 1000) \n\n"; - + cout << " \nbenchsolver : performs a benchmark of all the solvers available in Eigen \n\n"; + cout << " MATRIX FOLDER : \n"; + cout << " The matrices for the benchmark should be collected in a folder specified with an environment variable " + "EIGEN_MATRIXDIR \n"; + cout << " The matrices are stored using the matrix market coordinate format \n"; + cout << " The matrix and associated right-hand side (rhs) files are named respectively \n"; + cout << " as MatrixName.mtx and MatrixName_b.mtx. If the rhs does not exist, a random one is generated. \n"; + cout << " If a matrix is SPD, the matrix should be named as MatrixName_SPD.mtx \n"; + cout << " If a true solution exists, it should be named as MatrixName_x.mtx; \n"; + cout << " it will be used to compute the norm of the error relative to the computed solutions\n\n"; + cout << " OPTIONS : \n"; + cout << " -h or --help \n print this help and return\n\n"; + cout << " -d matrixdir \n Use matrixdir as the matrix folder instead of the one specified in the environment " + "variable EIGEN_MATRIXDIR\n\n"; + cout << " -o outputfile.xml \n Output the statistics to a xml file \n\n"; + cout << " --eps Sets the relative tolerance for iterative solvers (default 1e-08) \n\n"; + cout << " --maxits Sets the maximum number of iterations (default 1000) \n\n"; } -int main(int argc, char ** args) +int main(int argc, char **args) { - - bool help = ( get_options(argc, args, "-h") || get_options(argc, args, "--help") ); - if(help) { + + bool help = (get_options(argc, args, "-h") || get_options(argc, args, "--help")); + if (help) { bench_printhelp(); return 0; } // Get the location of the test matrices string matrix_dir; - if (!get_options(argc, args, "-d", &matrix_dir)) - { - if(getenv("EIGEN_MATRIXDIR") == NULL){ - std::cerr << "Please, specify the location of the matrices with -d mat_folder or the environment variable EIGEN_MATRIXDIR \n"; + if (!get_options(argc, args, "-d", &matrix_dir)) { + if (getenv("EIGEN_MATRIXDIR") == NULL) { + std::cerr << "Please, specify the location of the matrices with -d mat_folder or the environment variable " + "EIGEN_MATRIXDIR \n"; std::cerr << " Run with --help to see the list of all the available options \n"; return -1; } matrix_dir = getenv("EIGEN_MATRIXDIR"); } - + std::ofstream statbuf; - string statFile ; - + string statFile; + // Get the file to write the statistics bool statFileExists = get_options(argc, args, "-o", &statFile); - if(statFileExists) - { + if (statFileExists) { statbuf.open(statFile.c_str(), std::ios::out); - if(statbuf.good()){ - statFileExists = true; + if (statbuf.good()) { + statFileExists = true; printStatheader(statbuf); statbuf.close(); - } - else + } else std::cerr << "Unable to open the provided file for writting... \n"; - } - + } + // Get the maximum number of iterations and the tolerance - int maxiters = 1000; - double tol = 1e-08; - string inval; - if (get_options(argc, args, "--eps", &inval)) - tol = atof(inval.c_str()); - if(get_options(argc, args, "--maxits", &inval)) - maxiters = atoi(inval.c_str()); - - string current_dir; + int maxiters = 1000; + double tol = 1e-08; + string inval; + if (get_options(argc, args, "--eps", &inval)) tol = atof(inval.c_str()); + if (get_options(argc, args, "--maxits", &inval)) maxiters = atoi(inval.c_str()); + + string current_dir; // Test the real-arithmetics matrices - Browse_Matrices(matrix_dir, statFileExists, statFile,maxiters, tol); - + Browse_Matrices(matrix_dir, statFileExists, statFile, maxiters, tol); + // Test the complex-arithmetics matrices - Browse_Matrices >(matrix_dir, statFileExists, statFile, maxiters, tol); - - if(statFileExists) - { - statbuf.open(statFile.c_str(), std::ios::app); + Browse_Matrices>(matrix_dir, statFileExists, statFile, maxiters, tol); + + if (statFileExists) { + statbuf.open(statFile.c_str(), std::ios::app); statbuf << " \n"; cout << "\n Output written in " << statFile << " ...\n"; statbuf.close(); @@ -83,5 +79,3 @@ int main(int argc, char ** args) return 0; } - - diff --git a/filmulator-gui/core/nlmeans/eigen/bench/spbench/spbenchsolver.h b/filmulator-gui/core/nlmeans/eigen/bench/spbench/spbenchsolver.h index 19c719c0..ffe3d956 100644 --- a/filmulator-gui/core/nlmeans/eigen/bench/spbench/spbenchsolver.h +++ b/filmulator-gui/core/nlmeans/eigen/bench/spbench/spbenchsolver.h @@ -8,20 +8,20 @@ // with this file, You can obtain one at http://mozilla.org/MPL/2.0/. -#include -#include +#include +#include +#include +#include +#include #include +#include #include #include +#include +#include #include -#include -#include -#include -#include #include -#include #include -#include #include "spbenchstyle.h" @@ -50,225 +50,225 @@ #endif // CONSTANTS -#define EIGEN_UMFPACK 10 -#define EIGEN_SUPERLU 20 -#define EIGEN_PASTIX 30 -#define EIGEN_PARDISO 40 +#define EIGEN_UMFPACK 10 +#define EIGEN_SUPERLU 20 +#define EIGEN_PASTIX 30 +#define EIGEN_PARDISO 40 #define EIGEN_SPARSELU_COLAMD 50 #define EIGEN_SPARSELU_METIS 51 -#define EIGEN_BICGSTAB 60 -#define EIGEN_BICGSTAB_ILUT 61 +#define EIGEN_BICGSTAB 60 +#define EIGEN_BICGSTAB_ILUT 61 #define EIGEN_GMRES 70 #define EIGEN_GMRES_ILUT 71 -#define EIGEN_SIMPLICIAL_LDLT 80 -#define EIGEN_CHOLMOD_LDLT 90 -#define EIGEN_PASTIX_LDLT 100 -#define EIGEN_PARDISO_LDLT 110 -#define EIGEN_SIMPLICIAL_LLT 120 -#define EIGEN_CHOLMOD_SUPERNODAL_LLT 130 -#define EIGEN_CHOLMOD_SIMPLICIAL_LLT 140 -#define EIGEN_PASTIX_LLT 150 -#define EIGEN_PARDISO_LLT 160 -#define EIGEN_CG 170 -#define EIGEN_CG_PRECOND 180 +#define EIGEN_SIMPLICIAL_LDLT 80 +#define EIGEN_CHOLMOD_LDLT 90 +#define EIGEN_PASTIX_LDLT 100 +#define EIGEN_PARDISO_LDLT 110 +#define EIGEN_SIMPLICIAL_LLT 120 +#define EIGEN_CHOLMOD_SUPERNODAL_LLT 130 +#define EIGEN_CHOLMOD_SIMPLICIAL_LLT 140 +#define EIGEN_PASTIX_LLT 150 +#define EIGEN_PARDISO_LLT 160 +#define EIGEN_CG 170 +#define EIGEN_CG_PRECOND 180 using namespace Eigen; -using namespace std; +using namespace std; // Global variables for input parameters -int MaximumIters; // Maximum number of iterations -double RelErr; // Relative error of the computed solution -double best_time_val; // Current best time overall solvers -int best_time_id; // id of the best solver for the current system +int MaximumIters;// Maximum number of iterations +double RelErr;// Relative error of the computed solution +double best_time_val;// Current best time overall solvers +int best_time_id;// id of the best solver for the current system template inline typename NumTraits::Real test_precision() { return NumTraits::dummy_precision(); } -template<> inline float test_precision() { return 1e-3f; } -template<> inline double test_precision() { return 1e-6; } -template<> inline float test_precision >() { return test_precision(); } -template<> inline double test_precision >() { return test_precision(); } +template<> inline float test_precision() { return 1e-3f; } +template<> inline double test_precision() { return 1e-6; } +template<> inline float test_precision>() { return test_precision(); } +template<> inline double test_precision>() { return test_precision(); } -void printStatheader(std::ofstream& out) +void printStatheader(std::ofstream &out) { // Print XML header - // NOTE It would have been much easier to write these XML documents using external libraries like tinyXML or Xerces-C++. - + // NOTE It would have been much easier to write these XML documents using external libraries like tinyXML or + // Xerces-C++. + out << " \n"; - out << " \n"; + out << " \n"; out << "\n]>"; - out << "\n\n\n"; - - out << "\n \n" ; //root XML element + out << "\n\n\n"; + + out << "\n \n";// root XML element // Print the xsl style section - printBenchStyle(out); - // List all available solvers + printBenchStyle(out); + // List all available solvers out << " \n"; #ifdef EIGEN_UMFPACK_SUPPORT - out <<" \n"; + out << " \n"; out << " LU \n"; - out << " UMFPACK \n"; - out << " \n"; + out << " UMFPACK \n"; + out << " \n"; #endif #ifdef EIGEN_SUPERLU_SUPPORT - out <<" \n"; + out << " \n"; out << " LU \n"; - out << " SUPERLU \n"; - out << " \n"; + out << " SUPERLU \n"; + out << " \n"; #endif #ifdef EIGEN_CHOLMOD_SUPPORT - out <<" \n"; + out << " \n"; out << " LLT SP \n"; out << " CHOLMOD \n"; - out << " \n"; - - out <<" \n"; + out << " \n"; + + out << " \n"; out << " LLT \n"; out << " CHOLMOD \n"; out << " \n"; - - out <<" \n"; + + out << " \n"; out << " LDLT \n"; - out << " CHOLMOD \n"; - out << " \n"; + out << " CHOLMOD \n"; + out << " \n"; #endif #ifdef EIGEN_PARDISO_SUPPORT - out <<" \n"; + out << " \n"; out << " LU \n"; - out << " PARDISO \n"; - out << " \n"; - - out <<" \n"; + out << " PARDISO \n"; + out << " \n"; + + out << " \n"; out << " LLT \n"; - out << " PARDISO \n"; - out << " \n"; - - out <<" \n"; + out << " PARDISO \n"; + out << " \n"; + + out << " \n"; out << " LDLT \n"; - out << " PARDISO \n"; - out << " \n"; + out << " PARDISO \n"; + out << " \n"; #endif #ifdef EIGEN_PASTIX_SUPPORT - out <<" \n"; + out << " \n"; out << " LU \n"; - out << " PASTIX \n"; - out << " \n"; - - out <<" \n"; + out << " PASTIX \n"; + out << " \n"; + + out << " \n"; out << " LLT \n"; - out << " PASTIX \n"; - out << " \n"; - - out <<" \n"; + out << " PASTIX \n"; + out << " \n"; + + out << " \n"; out << " LDLT \n"; - out << " PASTIX \n"; - out << " \n"; + out << " PASTIX \n"; + out << " \n"; #endif - - out <<" \n"; + + out << " \n"; out << " BICGSTAB \n"; - out << " EIGEN \n"; - out << " \n"; - - out <<" \n"; + out << " EIGEN \n"; + out << " \n"; + + out << " \n"; out << " BICGSTAB_ILUT \n"; - out << " EIGEN \n"; - out << " \n"; - - out <<" \n"; + out << " EIGEN \n"; + out << " \n"; + + out << " \n"; out << " GMRES_ILUT \n"; - out << " EIGEN \n"; - out << " \n"; - - out <<" \n"; + out << " EIGEN \n"; + out << " \n"; + + out << " \n"; out << " LDLT \n"; - out << " EIGEN \n"; - out << " \n"; - - out <<" \n"; + out << " EIGEN \n"; + out << " \n"; + + out << " \n"; out << " LLT \n"; - out << " EIGEN \n"; - out << " \n"; - - out <<" \n"; + out << " EIGEN \n"; + out << " \n"; + + out << " \n"; out << " CG \n"; - out << " EIGEN \n"; - out << " \n"; - - out <<" \n"; + out << " EIGEN \n"; + out << " \n"; + + out << " \n"; out << " LU_COLAMD \n"; - out << " EIGEN \n"; - out << " \n"; - + out << " EIGEN \n"; + out << " \n"; + #ifdef EIGEN_METIS_SUPPORT - out <<" \n"; + out << " \n"; out << " LU_METIS \n"; - out << " EIGEN \n"; - out << " \n"; + out << " EIGEN \n"; + out << " \n"; #endif - out << " \n"; - + out << " \n"; } template -void call_solver(Solver &solver, const int solver_id, const typename Solver::MatrixType& A, const Matrix& b, const Matrix& refX,std::ofstream& statbuf) +void call_solver(Solver &solver, + const int solver_id, + const typename Solver::MatrixType &A, + const Matrix &b, + const Matrix &refX, + std::ofstream &statbuf) { - + double total_time; double compute_time; - double solve_time; + double solve_time; double rel_error; - Matrix x; - BenchTimer timer; + Matrix x; + BenchTimer timer; timer.reset(); timer.start(); - solver.compute(A); - if (solver.info() != Success) - { + solver.compute(A); + if (solver.info() != Success) { std::cerr << "Solver failed ... \n"; return; } timer.stop(); compute_time = timer.value(); - statbuf << "
\n"; - std::cout<< "SOLVE TIME : " << timer.value() < " << timer.value() << "
\n"; + std::cout << "SOLVE TIME : " << timer.value() << std::endl; + total_time = solve_time + compute_time; - statbuf << " " << total_time << "\n"; - std::cout<< "TOTAL TIME : " << total_time <\n"; - + statbuf << " " << total_time << "\n"; + std::cout << "TOTAL TIME : " << total_time << std::endl; + statbuf << " \n"; + // Verify the relative error - if(refX.size() != 0) - rel_error = (refX - x).norm()/refX.norm(); - else - { + if (refX.size() != 0) + rel_error = (refX - x).norm() / refX.norm(); + else { // Compute the relative residual norm - Matrix temp; - temp = A * x; - rel_error = (b-temp).norm()/b.norm(); + Matrix temp; + temp = A * x; + rel_error = (b - temp).norm() / b.norm(); } - statbuf << " " << rel_error << "\n"; - std::cout<< "REL. ERROR : " << rel_error << "\n\n" ; - if ( rel_error <= RelErr ) - { + statbuf << " " << rel_error << "\n"; + std::cout << "REL. ERROR : " << rel_error << "\n\n"; + if (rel_error <= RelErr) { // check the best time if convergence - if(!best_time_val || (best_time_val > total_time)) - { + if (!best_time_val || (best_time_val > total_time)) { best_time_val = total_time; best_time_id = solver_id; } @@ -276,279 +276,283 @@ void call_solver(Solver &solver, const int solver_id, const typename Solver::Mat } template -void call_directsolver(Solver& solver, const int solver_id, const typename Solver::MatrixType& A, const Matrix& b, const Matrix& refX, std::string& statFile) +void call_directsolver(Solver &solver, + const int solver_id, + const typename Solver::MatrixType &A, + const Matrix &b, + const Matrix &refX, + std::string &statFile) { - std::ofstream statbuf(statFile.c_str(), std::ios::app); - statbuf << " \n"; - call_solver(solver, solver_id, A, b, refX,statbuf); - statbuf << " \n"; - statbuf.close(); + std::ofstream statbuf(statFile.c_str(), std::ios::app); + statbuf << " \n"; + call_solver(solver, solver_id, A, b, refX, statbuf); + statbuf << " \n"; + statbuf.close(); } template -void call_itersolver(Solver &solver, const int solver_id, const typename Solver::MatrixType& A, const Matrix& b, const Matrix& refX, std::string& statFile) +void call_itersolver(Solver &solver, + const int solver_id, + const typename Solver::MatrixType &A, + const Matrix &b, + const Matrix &refX, + std::string &statFile) { - solver.setTolerance(RelErr); + solver.setTolerance(RelErr); solver.setMaxIterations(MaximumIters); - + std::ofstream statbuf(statFile.c_str(), std::ios::app); - statbuf << " \n"; - call_solver(solver, solver_id, A, b, refX,statbuf); - statbuf << " "<< solver.iterations() << "\n"; + statbuf << " \n"; + call_solver(solver, solver_id, A, b, refX, statbuf); + statbuf << " " << solver.iterations() << "\n"; statbuf << " \n"; - std::cout << "ITERATIONS : " << solver.iterations() <<"\n\n\n"; - + std::cout << "ITERATIONS : " << solver.iterations() << "\n\n\n"; } -template -void SelectSolvers(const SparseMatrix&A, unsigned int sym, Matrix& b, const Matrix& refX, std::string& statFile) +template +void SelectSolvers(const SparseMatrix &A, + unsigned int sym, + Matrix &b, + const Matrix &refX, + std::string &statFile) { - typedef SparseMatrix SpMat; + typedef SparseMatrix SpMat; // First, deal with Nonsymmetric and symmetric matrices - best_time_id = 0; + best_time_id = 0; best_time_val = 0.0; - //UMFPACK - #ifdef EIGEN_UMFPACK_SUPPORT +// UMFPACK +#ifdef EIGEN_UMFPACK_SUPPORT { - cout << "Solving with UMFPACK LU ... \n"; - UmfPackLU solver; - call_directsolver(solver, EIGEN_UMFPACK, A, b, refX,statFile); + cout << "Solving with UMFPACK LU ... \n"; + UmfPackLU solver; + call_directsolver(solver, EIGEN_UMFPACK, A, b, refX, statFile); } - #endif - //SuperLU - #ifdef EIGEN_SUPERLU_SUPPORT +#endif + // SuperLU +#ifdef EIGEN_SUPERLU_SUPPORT { - cout << "\nSolving with SUPERLU ... \n"; + cout << "\nSolving with SUPERLU ... \n"; SuperLU solver; - call_directsolver(solver, EIGEN_SUPERLU, A, b, refX,statFile); + call_directsolver(solver, EIGEN_SUPERLU, A, b, refX, statFile); } - #endif - - // PaStix LU - #ifdef EIGEN_PASTIX_SUPPORT +#endif + + // PaStix LU +#ifdef EIGEN_PASTIX_SUPPORT { - cout << "\nSolving with PASTIX LU ... \n"; - PastixLU solver; - call_directsolver(solver, EIGEN_PASTIX, A, b, refX,statFile) ; + cout << "\nSolving with PASTIX LU ... \n"; + PastixLU solver; + call_directsolver(solver, EIGEN_PASTIX, A, b, refX, statFile); } - #endif +#endif - //PARDISO LU - #ifdef EIGEN_PARDISO_SUPPORT + // PARDISO LU +#ifdef EIGEN_PARDISO_SUPPORT { - cout << "\nSolving with PARDISO LU ... \n"; - PardisoLU solver; - call_directsolver(solver, EIGEN_PARDISO, A, b, refX,statFile); + cout << "\nSolving with PARDISO LU ... \n"; + PardisoLU solver; + call_directsolver(solver, EIGEN_PARDISO, A, b, refX, statFile); } - #endif - +#endif + // Eigen SparseLU METIS cout << "\n Solving with Sparse LU AND COLAMD ... \n"; - SparseLU > solver; - call_directsolver(solver, EIGEN_SPARSELU_COLAMD, A, b, refX, statFile); - // Eigen SparseLU METIS - #ifdef EIGEN_METIS_SUPPORT + SparseLU> solver; + call_directsolver(solver, EIGEN_SPARSELU_COLAMD, A, b, refX, statFile); +// Eigen SparseLU METIS +#ifdef EIGEN_METIS_SUPPORT { cout << "\n Solving with Sparse LU AND METIS ... \n"; - SparseLU > solver; - call_directsolver(solver, EIGEN_SPARSELU_METIS, A, b, refX, statFile); + SparseLU> solver; + call_directsolver(solver, EIGEN_SPARSELU_METIS, A, b, refX, statFile); } - #endif - - //BiCGSTAB +#endif + + // BiCGSTAB { - cout << "\nSolving with BiCGSTAB ... \n"; - BiCGSTAB solver; - call_itersolver(solver, EIGEN_BICGSTAB, A, b, refX,statFile); + cout << "\nSolving with BiCGSTAB ... \n"; + BiCGSTAB solver; + call_itersolver(solver, EIGEN_BICGSTAB, A, b, refX, statFile); } - //BiCGSTAB+ILUT + // BiCGSTAB+ILUT { - cout << "\nSolving with BiCGSTAB and ILUT ... \n"; - BiCGSTAB > solver; - call_itersolver(solver, EIGEN_BICGSTAB_ILUT, A, b, refX,statFile); + cout << "\nSolving with BiCGSTAB and ILUT ... \n"; + BiCGSTAB> solver; + call_itersolver(solver, EIGEN_BICGSTAB_ILUT, A, b, refX, statFile); } - - - //GMRES -// { -// cout << "\nSolving with GMRES ... \n"; -// GMRES solver; -// call_itersolver(solver, EIGEN_GMRES, A, b, refX,statFile); -// } - //GMRES+ILUT + + + // GMRES + // { + // cout << "\nSolving with GMRES ... \n"; + // GMRES solver; + // call_itersolver(solver, EIGEN_GMRES, A, b, refX,statFile); + // } + // GMRES+ILUT { - cout << "\nSolving with GMRES and ILUT ... \n"; - GMRES > solver; - call_itersolver(solver, EIGEN_GMRES_ILUT, A, b, refX,statFile); + cout << "\nSolving with GMRES and ILUT ... \n"; + GMRES> solver; + call_itersolver(solver, EIGEN_GMRES_ILUT, A, b, refX, statFile); } - + // Hermitian and not necessarily positive-definites - if (sym != NonSymmetric) - { + if (sym != NonSymmetric) { // Internal Cholesky { - cout << "\nSolving with Simplicial LDLT ... \n"; + cout << "\nSolving with Simplicial LDLT ... \n"; SimplicialLDLT solver; - call_directsolver(solver, EIGEN_SIMPLICIAL_LDLT, A, b, refX,statFile); + call_directsolver(solver, EIGEN_SIMPLICIAL_LDLT, A, b, refX, statFile); } - - // CHOLMOD - #ifdef EIGEN_CHOLMOD_SUPPORT + +// CHOLMOD +#ifdef EIGEN_CHOLMOD_SUPPORT { - cout << "\nSolving with CHOLMOD LDLT ... \n"; + cout << "\nSolving with CHOLMOD LDLT ... \n"; CholmodDecomposition solver; solver.setMode(CholmodLDLt); - call_directsolver(solver,EIGEN_CHOLMOD_LDLT, A, b, refX,statFile); + call_directsolver(solver, EIGEN_CHOLMOD_LDLT, A, b, refX, statFile); } - #endif - - //PASTIX LLT - #ifdef EIGEN_PASTIX_SUPPORT +#endif + +// PASTIX LLT +#ifdef EIGEN_PASTIX_SUPPORT { - cout << "\nSolving with PASTIX LDLT ... \n"; - PastixLDLT solver; - call_directsolver(solver,EIGEN_PASTIX_LDLT, A, b, refX,statFile); + cout << "\nSolving with PASTIX LDLT ... \n"; + PastixLDLT solver; + call_directsolver(solver, EIGEN_PASTIX_LDLT, A, b, refX, statFile); } - #endif - - //PARDISO LLT - #ifdef EIGEN_PARDISO_SUPPORT +#endif + +// PARDISO LLT +#ifdef EIGEN_PARDISO_SUPPORT { - cout << "\nSolving with PARDISO LDLT ... \n"; - PardisoLDLT solver; - call_directsolver(solver,EIGEN_PARDISO_LDLT, A, b, refX,statFile); + cout << "\nSolving with PARDISO LDLT ... \n"; + PardisoLDLT solver; + call_directsolver(solver, EIGEN_PARDISO_LDLT, A, b, refX, statFile); } - #endif +#endif } - // Now, symmetric POSITIVE DEFINITE matrices - if (sym == SPD) - { - - //Internal Sparse Cholesky + // Now, symmetric POSITIVE DEFINITE matrices + if (sym == SPD) { + + // Internal Sparse Cholesky { - cout << "\nSolving with SIMPLICIAL LLT ... \n"; - SimplicialLLT solver; - call_directsolver(solver,EIGEN_SIMPLICIAL_LLT, A, b, refX,statFile); + cout << "\nSolving with SIMPLICIAL LLT ... \n"; + SimplicialLLT solver; + call_directsolver(solver, EIGEN_SIMPLICIAL_LLT, A, b, refX, statFile); } - - // CHOLMOD - #ifdef EIGEN_CHOLMOD_SUPPORT + +// CHOLMOD +#ifdef EIGEN_CHOLMOD_SUPPORT { // CholMOD SuperNodal LLT - cout << "\nSolving with CHOLMOD LLT (Supernodal)... \n"; + cout << "\nSolving with CHOLMOD LLT (Supernodal)... \n"; CholmodDecomposition solver; solver.setMode(CholmodSupernodalLLt); - call_directsolver(solver,EIGEN_CHOLMOD_SUPERNODAL_LLT, A, b, refX,statFile); + call_directsolver(solver, EIGEN_CHOLMOD_SUPERNODAL_LLT, A, b, refX, statFile); // CholMod Simplicial LLT - cout << "\nSolving with CHOLMOD LLT (Simplicial) ... \n"; + cout << "\nSolving with CHOLMOD LLT (Simplicial) ... \n"; solver.setMode(CholmodSimplicialLLt); - call_directsolver(solver,EIGEN_CHOLMOD_SIMPLICIAL_LLT, A, b, refX,statFile); + call_directsolver(solver, EIGEN_CHOLMOD_SIMPLICIAL_LLT, A, b, refX, statFile); } - #endif - - //PASTIX LLT - #ifdef EIGEN_PASTIX_SUPPORT +#endif + +// PASTIX LLT +#ifdef EIGEN_PASTIX_SUPPORT { - cout << "\nSolving with PASTIX LLT ... \n"; - PastixLLT solver; - call_directsolver(solver,EIGEN_PASTIX_LLT, A, b, refX,statFile); + cout << "\nSolving with PASTIX LLT ... \n"; + PastixLLT solver; + call_directsolver(solver, EIGEN_PASTIX_LLT, A, b, refX, statFile); } - #endif - - //PARDISO LLT - #ifdef EIGEN_PARDISO_SUPPORT +#endif + +// PARDISO LLT +#ifdef EIGEN_PARDISO_SUPPORT { - cout << "\nSolving with PARDISO LLT ... \n"; - PardisoLLT solver; - call_directsolver(solver,EIGEN_PARDISO_LLT, A, b, refX,statFile); + cout << "\nSolving with PARDISO LLT ... \n"; + PardisoLLT solver; + call_directsolver(solver, EIGEN_PARDISO_LLT, A, b, refX, statFile); } - #endif - +#endif + // Internal CG { - cout << "\nSolving with CG ... \n"; - ConjugateGradient solver; - call_itersolver(solver,EIGEN_CG, A, b, refX,statFile); + cout << "\nSolving with CG ... \n"; + ConjugateGradient solver; + call_itersolver(solver, EIGEN_CG, A, b, refX, statFile); } - //CG+IdentityPreconditioner -// { -// cout << "\nSolving with CG and IdentityPreconditioner ... \n"; -// ConjugateGradient solver; -// call_itersolver(solver,EIGEN_CG_PRECOND, A, b, refX,statFile); -// } - } // End SPD matrices + // CG+IdentityPreconditioner + // { + // cout << "\nSolving with CG and IdentityPreconditioner ... \n"; + // ConjugateGradient solver; + // call_itersolver(solver,EIGEN_CG_PRECOND, A, b, refX,statFile); + // } + }// End SPD matrices } -/* Browse all the matrices available in the specified folder +/* Browse all the matrices available in the specified folder * and solve the associated linear system. * The results of each solve are printed in the standard output * and optionally in the provided html file */ -template -void Browse_Matrices(const string folder, bool statFileExists, std::string& statFile, int maxiters, double tol) +template +void Browse_Matrices(const string folder, bool statFileExists, std::string &statFile, int maxiters, double tol) { - MaximumIters = maxiters; // Maximum number of iterations, global variable - RelErr = tol; //Relative residual error as stopping criterion for iterative solvers + MaximumIters = maxiters;// Maximum number of iterations, global variable + RelErr = tol;// Relative residual error as stopping criterion for iterative solvers MatrixMarketIterator it(folder); - for ( ; it; ++it) - { - //print the infos for this linear system - if(statFileExists) - { + for (; it; ++it) { + // print the infos for this linear system + if (statFileExists) { std::ofstream statbuf(statFile.c_str(), std::ios::app); statbuf << " \n"; statbuf << " \n"; - statbuf << " " << it.matname() << " \n"; - statbuf << " " << it.matrix().rows() << " \n"; + statbuf << " " << it.matname() << " \n"; + statbuf << " " << it.matrix().rows() << " \n"; statbuf << " " << it.matrix().nonZeros() << "\n"; - if (it.sym()!=NonSymmetric) - { - statbuf << " Symmetric \n" ; - if (it.sym() == SPD) - statbuf << " YES \n"; - else - statbuf << " NO \n"; - - } - else - { - statbuf << " NonSymmetric \n" ; - statbuf << " NO \n"; + if (it.sym() != NonSymmetric) { + statbuf << " Symmetric \n"; + if (it.sym() == SPD) + statbuf << " YES \n"; + else + statbuf << " NO \n"; + + } else { + statbuf << " NonSymmetric \n"; + statbuf << " NO \n"; } statbuf << " \n"; statbuf.close(); } - - cout<< "\n\n===================================================== \n"; - cout<< " ====== SOLVING WITH MATRIX " << it.matname() << " ====\n"; - cout<< " =================================================== \n\n"; + + cout << "\n\n===================================================== \n"; + cout << " ====== SOLVING WITH MATRIX " << it.matname() << " ====\n"; + cout << " =================================================== \n\n"; Matrix refX; - if(it.hasrefX()) refX = it.refX(); - // Call all suitable solvers for this linear system + if (it.hasrefX()) refX = it.refX(); + // Call all suitable solvers for this linear system SelectSolvers(it.matrix(), it.sym(), it.rhs(), refX, statFile); - - if(statFileExists) - { + + if (statFileExists) { std::ofstream statbuf(statFile.c_str(), std::ios::app); - statbuf << " \n"; - statbuf << " \n"; + statbuf << " \n"; + statbuf << " \n"; statbuf.close(); } - } -} + } +} -bool get_options(int argc, char **args, string option, string* value=0) +bool get_options(int argc, char **args, string option, string *value = 0) { - int idx = 1, found=false; - while (idx\n \ @@ -23,7 +23,7 @@ void printBenchStyle(std::ofstream& out) \n \ \n \ "; - out<<"\n \ + out << "
\n \ \n \ \n \ \n \ @@ -36,8 +36,8 @@ void printBenchStyle(std::ofstream& out) \n \ \n \ "; - - out<<" \n \ + + out << " \n \ \n \ \n \ \n \ @@ -50,7 +50,7 @@ void printBenchStyle(std::ofstream& out) \n \ \n \ "; - out<<" \n \ + out << " \n \ \n \ \n \ \n \ @@ -71,7 +71,7 @@ void printBenchStyle(std::ofstream& out) \n \ \n \ "; - out<<" \n \ + out << " \n \ \n \ \n \ \n \ @@ -89,7 +89,6 @@ void printBenchStyle(std::ofstream& out) \n \ \n \ \n\n"; - } #endif diff --git a/filmulator-gui/core/nlmeans/eigen/bench/spbench/test_sparseLU.cpp b/filmulator-gui/core/nlmeans/eigen/bench/spbench/test_sparseLU.cpp index f8ecbe69..e52f4c5a 100644 --- a/filmulator-gui/core/nlmeans/eigen/bench/spbench/test_sparseLU.cpp +++ b/filmulator-gui/core/nlmeans/eigen/bench/spbench/test_sparseLU.cpp @@ -1,12 +1,12 @@ // Small bench routine for Eigen available in Eigen // (C) Desire NUENTSA WAKAM, INRIA -#include +#include +#include #include #include +#include #include -#include -#include #ifdef EIGEN_METIS_SUPPORT #include #endif @@ -16,39 +16,42 @@ using namespace Eigen; int main(int argc, char **args) { -// typedef complex scalar; - typedef double scalar; - SparseMatrix A; + // typedef complex scalar; + typedef double scalar; + SparseMatrix A; typedef SparseMatrix::Index Index; typedef Matrix DenseMatrix; typedef Matrix DenseRhs; Matrix b, x, tmp; -// SparseLU, AMDOrdering > solver; -// #ifdef EIGEN_METIS_SUPPORT -// SparseLU, MetisOrdering > solver; -// std::cout<< "ORDERING : METIS\n"; -// #else - SparseLU, COLAMDOrdering > solver; - std::cout<< "ORDERING : COLAMD\n"; -// #endif - - ifstream matrix_file; + // SparseLU, AMDOrdering > solver; + // #ifdef EIGEN_METIS_SUPPORT + // SparseLU, MetisOrdering > solver; + // std::cout<< "ORDERING : METIS\n"; + // #else + SparseLU, COLAMDOrdering> solver; + std::cout << "ORDERING : COLAMD\n"; + // #endif + + ifstream matrix_file; string line; - int n; - BenchTimer timer; - + int n; + BenchTimer timer; + // Set parameters /* Fill the matrix with sparse matrix stored in Matrix-Market coordinate column-oriented format */ if (argc < 2) assert(false && "please, give the matrix market file "); loadMarket(A, args[1]); cout << "End charging matrix " << endl; - bool iscomplex=false, isvector=false; + bool iscomplex = false, isvector = false; int sym; getMarketHeader(args[1], sym, iscomplex, isvector); -// if (iscomplex) { cout<< " Not for complex matrices \n"; return -1; } - if (isvector) { cout << "The provided file is not a matrix file\n"; return -1;} - if (sym != 0) { // symmetric matrices, only the lower part is stored - SparseMatrix temp; + // if (iscomplex) { cout<< " Not for complex matrices \n"; return -1; } + if (isvector) { + cout << "The provided file is not a matrix file\n"; + return -1; + } + if (sym != 0) {// symmetric matrices, only the lower part is stored + SparseMatrix temp; temp = A; A = temp.selfadjointView(); } @@ -57,37 +60,36 @@ int main(int argc, char **args) if (argc > 2) loadMarketVector(b, args[2]); - else - { + else { b.resize(n); tmp.resize(n); -// tmp.setRandom(); - for (int i = 0; i < n; i++) tmp(i) = i; - b = A * tmp ; + // tmp.setRandom(); + for (int i = 0; i < n; i++) tmp(i) = i; + b = A * tmp; } /* Compute the factorization */ -// solver.isSymmetric(true); - timer.start(); -// solver.compute(A); - solver.analyzePattern(A); - timer.stop(); + // solver.isSymmetric(true); + timer.start(); + // solver.compute(A); + solver.analyzePattern(A); + timer.stop(); cout << "Time to analyze " << timer.value() << std::endl; - timer.reset(); - timer.start(); - solver.factorize(A); - timer.stop(); + timer.reset(); + timer.start(); + solver.factorize(A); + timer.stop(); cout << "Factorize Time " << timer.value() << std::endl; - timer.reset(); - timer.start(); + timer.reset(); + timer.start(); x = solver.solve(b); timer.stop(); - cout << "solve time " << timer.value() << std::endl; + cout << "solve time " << timer.value() << std::endl; /* Check the accuracy */ - Matrix tmp2 = b - A*x; - scalar tempNorm = tmp2.norm()/b.norm(); - cout << "Relative norm of the computed solution : " << tempNorm <<"\n"; - cout << "Number of nonzeros in the factor : " << solver.nnzL() + solver.nnzU() << std::endl; - + Matrix tmp2 = b - A * x; + scalar tempNorm = tmp2.norm() / b.norm(); + cout << "Relative norm of the computed solution : " << tempNorm << "\n"; + cout << "Number of nonzeros in the factor : " << solver.nnzL() + solver.nnzU() << std::endl; + return 0; } \ No newline at end of file diff --git a/filmulator-gui/core/nlmeans/eigen/bench/spmv.cpp b/filmulator-gui/core/nlmeans/eigen/bench/spmv.cpp index 959bab09..47a96591 100644 --- a/filmulator-gui/core/nlmeans/eigen/bench/spmv.cpp +++ b/filmulator-gui/core/nlmeans/eigen/bench/spmv.cpp @@ -1,14 +1,16 @@ -//g++-4.4 -DNOMTL -Wl,-rpath /usr/local/lib/oski -L /usr/local/lib/oski/ -l oski -l oski_util -l oski_util_Tid -DOSKI -I ~/Coding/LinearAlgebra/mtl4/ spmv.cpp -I .. -O2 -DNDEBUG -lrt -lm -l oski_mat_CSC_Tid -loskilt && ./a.out r200000 c200000 n100 t1 p1 +// g++-4.4 -DNOMTL -Wl,-rpath /usr/local/lib/oski -L /usr/local/lib/oski/ -l oski -l oski_util -l oski_util_Tid -DOSKI +// -I ~/Coding/LinearAlgebra/mtl4/ spmv.cpp -I .. -O2 -DNDEBUG -lrt -lm -l oski_mat_CSC_Tid -loskilt && ./a.out +// r200000 c200000 n100 t1 p1 #define SCALAR double -#include -#include -#include "BenchTimer.h" #include "BenchSparseUtil.h" +#include "BenchTimer.h" +#include +#include -#define SPMV_BENCH(CODE) BENCH(t,tries,repeats,CODE); +#define SPMV_BENCH(CODE) BENCH(t, tries, repeats, CODE); // #ifdef MKL // @@ -44,106 +46,93 @@ int main(int argc, char *argv[]) int repeats = 2; bool need_help = false; - for(int i = 1; i < argc; i++) - { - if(argv[i][0] == 'r') - { - rows = atoi(argv[i]+1); - } - else if(argv[i][0] == 'c') - { - cols = atoi(argv[i]+1); - } - else if(argv[i][0] == 'n') - { - nnzPerCol = atoi(argv[i]+1); - } - else if(argv[i][0] == 't') - { - tries = atoi(argv[i]+1); - } - else if(argv[i][0] == 'p') - { - repeats = atoi(argv[i]+1); - } - else - { + for (int i = 1; i < argc; i++) { + if (argv[i][0] == 'r') { + rows = atoi(argv[i] + 1); + } else if (argv[i][0] == 'c') { + cols = atoi(argv[i] + 1); + } else if (argv[i][0] == 'n') { + nnzPerCol = atoi(argv[i] + 1); + } else if (argv[i][0] == 't') { + tries = atoi(argv[i] + 1); + } else if (argv[i][0] == 'p') { + repeats = atoi(argv[i] + 1); + } else { need_help = true; } } - if(need_help) - { + if (need_help) { std::cout << argv[0] << " r c n t p\n"; return 1; } - std::cout << "SpMV " << rows << " x " << cols << " with " << nnzPerCol << " non zeros per column. (" << repeats << " repeats, and " << tries << " tries)\n\n"; + std::cout << "SpMV " << rows << " x " << cols << " with " << nnzPerCol << " non zeros per column. (" << repeats + << " repeats, and " << tries << " tries)\n\n"; - EigenSparseMatrix sm(rows,cols); + EigenSparseMatrix sm(rows, cols); DenseVector dv(cols), res(rows); dv.setRandom(); BenchTimer t; - while (nnzPerCol>=4) - { + while (nnzPerCol >= 4) { std::cout << "nnz: " << nnzPerCol << "\n"; sm.setZero(); fillMatrix2(nnzPerCol, rows, cols, sm); - // dense matrices - #ifdef DENSEMATRIX +// dense matrices +#ifdef DENSEMATRIX { - DenseMatrix dm(rows,cols), (rows,cols); + DenseMatrix dm(rows, cols), (rows, cols); eiToDense(sm, dm); SPMV_BENCH(res = dm * sm); - std::cout << "Dense " << t.value()/repeats << "\t"; + std::cout << "Dense " << t.value() / repeats << "\t"; SPMV_BENCH(res = dm.transpose() * sm); - std::cout << t.value()/repeats << endl; + std::cout << t.value() / repeats << endl; } - #endif +#endif // eigen sparse matrices { - SPMV_BENCH(res.noalias() += sm * dv; ) - std::cout << "Eigen " << t.value()/repeats << "\t"; + SPMV_BENCH(res.noalias() += sm * dv;) + std::cout << "Eigen " << t.value() / repeats << "\t"; - SPMV_BENCH(res.noalias() += sm.transpose() * dv; ) - std::cout << t.value()/repeats << endl; + SPMV_BENCH(res.noalias() += sm.transpose() * dv;) + std::cout << t.value() / repeats << endl; } - // CSparse - #ifdef CSPARSE +// CSparse +#ifdef CSPARSE { std::cout << "CSparse \n"; cs *csm; eiToCSparse(sm, csm); -// BENCH(); -// timer.stop(); -// std::cout << " a * b:\t" << timer.value() << endl; + // BENCH(); + // timer.stop(); + // std::cout << " a * b:\t" << timer.value() << endl; -// BENCH( { m3 = cs_sorted_multiply2(m1, m2); cs_spfree(m3); } ); -// std::cout << " a * b:\t" << timer.value() << endl; + // BENCH( { m3 = cs_sorted_multiply2(m1, m2); cs_spfree(m3); } ); + // std::cout << " a * b:\t" << timer.value() << endl; } - #endif +#endif - #ifdef OSKI +#ifdef OSKI { oski_matrix_t om; oski_vecview_t ov, ores; oski_Init(); - om = oski_CreateMatCSC(sm._outerIndexPtr(), sm._innerIndexPtr(), sm._valuePtr(), rows, cols, - SHARE_INPUTMAT, 1, INDEX_ZERO_BASED); + om = oski_CreateMatCSC( + sm._outerIndexPtr(), sm._innerIndexPtr(), sm._valuePtr(), rows, cols, SHARE_INPUTMAT, 1, INDEX_ZERO_BASED); ov = oski_CreateVecView(dv.data(), cols, STRIDE_UNIT); ores = oski_CreateVecView(res.data(), rows, STRIDE_UNIT); - SPMV_BENCH( oski_MatMult(om, OP_NORMAL, 1, ov, 0, ores) ); - std::cout << "OSKI " << t.value()/repeats << "\t"; + SPMV_BENCH(oski_MatMult(om, OP_NORMAL, 1, ov, 0, ores)); + std::cout << "OSKI " << t.value() / repeats << "\t"; - SPMV_BENCH( oski_MatMult(om, OP_TRANS, 1, ov, 0, ores) ); - std::cout << t.value()/repeats << "\n"; + SPMV_BENCH(oski_MatMult(om, OP_TRANS, 1, ov, 0, ores)); + std::cout << t.value() / repeats << "\n"; // tune t.reset(); @@ -153,11 +142,11 @@ int main(int argc, char *argv[]) t.stop(); double tuning = t.value(); - SPMV_BENCH( oski_MatMult(om, OP_NORMAL, 1, ov, 0, ores) ); - std::cout << "OSKI tuned " << t.value()/repeats << "\t"; + SPMV_BENCH(oski_MatMult(om, OP_NORMAL, 1, ov, 0, ores)); + std::cout << "OSKI tuned " << t.value() / repeats << "\t"; - SPMV_BENCH( oski_MatMult(om, OP_TRANS, 1, ov, 0, ores) ); - std::cout << t.value()/repeats << "\t(" << tuning << ")\n"; + SPMV_BENCH(oski_MatMult(om, OP_TRANS, 1, ov, 0, ores)); + std::cout << t.value() / repeats << "\t(" << tuning << ")\n"; oski_DestroyMat(om); @@ -165,69 +154,65 @@ int main(int argc, char *argv[]) oski_DestroyVecView(ores); oski_Close(); } - #endif +#endif - #ifndef NOUBLAS +#ifndef NOUBLAS { using namespace boost::numeric; - UblasMatrix um(rows,cols); + UblasMatrix um(rows, cols); eiToUblas(sm, um); boost::numeric::ublas::vector uv(cols), ures(rows); - Map >(&uv[0], cols) = dv; - Map >(&ures[0], rows) = res; + Map>(&uv[0], cols) = dv; + Map>(&ures[0], rows) = res; SPMV_BENCH(ublas::axpy_prod(um, uv, ures, true)); - std::cout << "ublas " << t.value()/repeats << "\t"; + std::cout << "ublas " << t.value() / repeats << "\t"; SPMV_BENCH(ublas::axpy_prod(boost::numeric::ublas::trans(um), uv, ures, true)); - std::cout << t.value()/repeats << endl; + std::cout << t.value() / repeats << endl; } - #endif +#endif - // GMM++ - #ifndef NOGMM +// GMM++ +#ifndef NOGMM { - GmmSparse gm(rows,cols); + GmmSparse gm(rows, cols); eiToGmm(sm, gm); std::vector gv(cols), gres(rows); - Map >(&gv[0], cols) = dv; - Map >(&gres[0], rows) = res; + Map>(&gv[0], cols) = dv; + Map>(&gres[0], rows) = res; SPMV_BENCH(gmm::mult(gm, gv, gres)); - std::cout << "GMM++ " << t.value()/repeats << "\t"; + std::cout << "GMM++ " << t.value() / repeats << "\t"; SPMV_BENCH(gmm::mult(gmm::transposed(gm), gv, gres)); - std::cout << t.value()/repeats << endl; + std::cout << t.value() / repeats << endl; } - #endif +#endif - // MTL4 - #ifndef NOMTL +// MTL4 +#ifndef NOMTL { - MtlSparse mm(rows,cols); + MtlSparse mm(rows, cols); eiToMtl(sm, mm); mtl::dense_vector mv(cols, 1.0); mtl::dense_vector mres(rows, 1.0); SPMV_BENCH(mres = mm * mv); - std::cout << "MTL4 " << t.value()/repeats << "\t"; + std::cout << "MTL4 " << t.value() / repeats << "\t"; SPMV_BENCH(mres = trans(mm) * mv); - std::cout << t.value()/repeats << endl; + std::cout << t.value() / repeats << endl; } - #endif +#endif std::cout << "\n"; - if(nnzPerCol==1) - break; - nnzPerCol -= nnzPerCol/2; + if (nnzPerCol == 1) break; + nnzPerCol -= nnzPerCol / 2; } return 0; } - - - diff --git a/filmulator-gui/core/nlmeans/eigen/bench/tensors/benchmark.h b/filmulator-gui/core/nlmeans/eigen/bench/tensors/benchmark.h index f115b54a..25d20907 100644 --- a/filmulator-gui/core/nlmeans/eigen/bench/tensors/benchmark.h +++ b/filmulator-gui/core/nlmeans/eigen/bench/tensors/benchmark.h @@ -18,32 +18,29 @@ #include namespace testing { -class Benchmark { - public: - Benchmark(const char* name, void (*fn)(int)) { - Register(name, fn, NULL); - } - Benchmark(const char* name, void (*fn_range)(int, int)) { - Register(name, NULL, fn_range); - } - Benchmark* Arg(int x); - Benchmark* Range(int lo, int hi); - const char* Name(); - bool ShouldRun(int argc, char* argv[]); +class Benchmark +{ +public: + Benchmark(const char *name, void (*fn)(int)) { Register(name, fn, NULL); } + Benchmark(const char *name, void (*fn_range)(int, int)) { Register(name, NULL, fn_range); } + Benchmark *Arg(int x); + Benchmark *Range(int lo, int hi); + const char *Name(); + bool ShouldRun(int argc, char *argv[]); void Run(); - private: - const char* name_; + +private: + const char *name_; void (*fn_)(int); void (*fn_range_)(int, int); std::vector args_; - void Register(const char* name, void (*fn)(int), void (*fn_range)(int, int)); + void Register(const char *name, void (*fn)(int), void (*fn_range)(int, int)); void RunRepeatedlyWithArg(int iterations, int arg); void RunWithArg(int arg); }; -} // namespace testing +}// namespace testing void SetBenchmarkFlopsProcessed(int64_t); void StopBenchmarkTiming(); void StartBenchmarkTiming(); #define BENCHMARK(f) \ - static ::testing::Benchmark* _benchmark_##f __attribute__((unused)) = \ - (new ::testing::Benchmark(#f, f)) + static ::testing::Benchmark *_benchmark_##f __attribute__((unused)) = (new ::testing::Benchmark(#f, f)) diff --git a/filmulator-gui/core/nlmeans/eigen/bench/tensors/tensor_benchmarks.h b/filmulator-gui/core/nlmeans/eigen/bench/tensors/tensor_benchmarks.h index c2fb3ded..7ae3c1a3 100644 --- a/filmulator-gui/core/nlmeans/eigen/bench/tensors/tensor_benchmarks.h +++ b/filmulator-gui/core/nlmeans/eigen/bench/tensors/tensor_benchmarks.h @@ -4,46 +4,44 @@ typedef int TensorIndex; #define EIGEN_DEFAULT_DENSE_INDEX_TYPE int -#include "unsupported/Eigen/CXX11/Tensor" #include "benchmark.h" +#include "unsupported/Eigen/CXX11/Tensor" -#define BENCHMARK_RANGE(bench, lo, hi) \ - BENCHMARK(bench)->Range(lo, hi) +#define BENCHMARK_RANGE(bench, lo, hi) BENCHMARK(bench)->Range(lo, hi) using Eigen::Tensor; using Eigen::TensorMap; // TODO(bsteiner): also templatize on the input type since we have users // for int8 as well as floats. -template class BenchmarkSuite { - public: - BenchmarkSuite(const Device& device, size_t m, size_t k, size_t n) - : m_(m), k_(k), n_(n), device_(device) { +template class BenchmarkSuite +{ +public: + BenchmarkSuite(const Device &device, size_t m, size_t k, size_t n) : m_(m), k_(k), n_(n), device_(device) + { initialize(); } - BenchmarkSuite(const Device& device, size_t m) - : m_(m), k_(m), n_(m), device_(device) { - initialize(); - } + BenchmarkSuite(const Device &device, size_t m) : m_(m), k_(m), n_(m), device_(device) { initialize(); } - ~BenchmarkSuite() { + ~BenchmarkSuite() + { device_.deallocate(a_); device_.deallocate(b_); device_.deallocate(c_); } - void memcpy(int num_iters) { + void memcpy(int num_iters) + { eigen_assert(m_ == k_ && k_ == n_); StartBenchmarkTiming(); - for (int iter = 0; iter < num_iters; ++iter) { - device_.memcpy(c_, a_, m_ * m_ * sizeof(T)); - } + for (int iter = 0; iter < num_iters; ++iter) { device_.memcpy(c_, a_, m_ * m_ * sizeof(T)); } // Record the number of values copied per second finalizeBenchmark(static_cast(m_) * m_ * num_iters); } - void typeCasting(int num_iters) { + void typeCasting(int num_iters) + { eigen_assert(m_ == n_); Eigen::array sizes; if (sizeof(T) >= sizeof(int)) { @@ -53,18 +51,17 @@ template class BenchmarkSuite { sizes[0] = m_ * sizeof(T) / sizeof(int); sizes[1] = k_ * sizeof(T) / sizeof(int); } - const TensorMap, Eigen::Aligned> A((int*)a_, sizes); + const TensorMap, Eigen::Aligned> A((int *)a_, sizes); TensorMap, Eigen::Aligned> B(b_, sizes); StartBenchmarkTiming(); - for (int iter = 0; iter < num_iters; ++iter) { - B.device(device_) = A.template cast(); - } + for (int iter = 0; iter < num_iters; ++iter) { B.device(device_) = A.template cast(); } // Record the number of values copied per second finalizeBenchmark(static_cast(m_) * k_ * num_iters); } - void random(int num_iters) { + void random(int num_iters) + { eigen_assert(m_ == k_ && k_ == n_); Eigen::array sizes; sizes[0] = m_; @@ -72,14 +69,13 @@ template class BenchmarkSuite { TensorMap, Eigen::Aligned> C(c_, sizes); StartBenchmarkTiming(); - for (int iter = 0; iter < num_iters; ++iter) { - C.device(device_) = C.random(); - } + for (int iter = 0; iter < num_iters; ++iter) { C.device(device_) = C.random(); } // Record the number of random numbers generated per second finalizeBenchmark(static_cast(m_) * m_ * num_iters); } - void slicing(int num_iters) { + void slicing(int num_iters) + { eigen_assert(m_ == k_ && k_ == n_); Eigen::array sizes; sizes[0] = m_; @@ -88,29 +84,26 @@ template class BenchmarkSuite { const TensorMap, Eigen::Aligned> B(b_, sizes); TensorMap, Eigen::Aligned> C(c_, sizes); - const Eigen::DSizes quarter_sizes(m_/2, m_/2); + const Eigen::DSizes quarter_sizes(m_ / 2, m_ / 2); const Eigen::DSizes first_quadrant(0, 0); - const Eigen::DSizes second_quadrant(0, m_/2); - const Eigen::DSizes third_quadrant(m_/2, 0); - const Eigen::DSizes fourth_quadrant(m_/2, m_/2); + const Eigen::DSizes second_quadrant(0, m_ / 2); + const Eigen::DSizes third_quadrant(m_ / 2, 0); + const Eigen::DSizes fourth_quadrant(m_ / 2, m_ / 2); StartBenchmarkTiming(); for (int iter = 0; iter < num_iters; ++iter) { - C.slice(first_quadrant, quarter_sizes).device(device_) = - A.slice(first_quadrant, quarter_sizes); - C.slice(second_quadrant, quarter_sizes).device(device_) = - B.slice(second_quadrant, quarter_sizes); - C.slice(third_quadrant, quarter_sizes).device(device_) = - A.slice(third_quadrant, quarter_sizes); - C.slice(fourth_quadrant, quarter_sizes).device(device_) = - B.slice(fourth_quadrant, quarter_sizes); + C.slice(first_quadrant, quarter_sizes).device(device_) = A.slice(first_quadrant, quarter_sizes); + C.slice(second_quadrant, quarter_sizes).device(device_) = B.slice(second_quadrant, quarter_sizes); + C.slice(third_quadrant, quarter_sizes).device(device_) = A.slice(third_quadrant, quarter_sizes); + C.slice(fourth_quadrant, quarter_sizes).device(device_) = B.slice(fourth_quadrant, quarter_sizes); } // Record the number of values copied from the rhs slice to the lhs slice // each second finalizeBenchmark(static_cast(m_) * m_ * num_iters); } - void rowChip(int num_iters) { + void rowChip(int num_iters) + { Eigen::array input_size; input_size[0] = k_; input_size[1] = n_; @@ -120,14 +113,13 @@ template class BenchmarkSuite { TensorMap, Eigen::Aligned> C(c_, output_size); StartBenchmarkTiming(); - for (int iter = 0; iter < num_iters; ++iter) { - C.device(device_) = B.chip(iter % k_, 0); - } + for (int iter = 0; iter < num_iters; ++iter) { C.device(device_) = B.chip(iter % k_, 0); } // Record the number of values copied from the rhs chip to the lhs. finalizeBenchmark(static_cast(n_) * num_iters); } - void colChip(int num_iters) { + void colChip(int num_iters) + { Eigen::array input_size; input_size[0] = k_; input_size[1] = n_; @@ -137,14 +129,13 @@ template class BenchmarkSuite { TensorMap, Eigen::Aligned> C(c_, output_size); StartBenchmarkTiming(); - for (int iter = 0; iter < num_iters; ++iter) { - C.device(device_) = B.chip(iter % n_, 1); - } + for (int iter = 0; iter < num_iters; ++iter) { C.device(device_) = B.chip(iter % n_, 1); } // Record the number of values copied from the rhs chip to the lhs. finalizeBenchmark(static_cast(n_) * num_iters); } - void shuffling(int num_iters) { + void shuffling(int num_iters) + { eigen_assert(m_ == n_); Eigen::array size_a; size_a[0] = m_; @@ -160,18 +151,17 @@ template class BenchmarkSuite { shuffle[1] = 0; StartBenchmarkTiming(); - for (int iter = 0; iter < num_iters; ++iter) { - B.device(device_) = A.shuffle(shuffle); - } + for (int iter = 0; iter < num_iters; ++iter) { B.device(device_) = A.shuffle(shuffle); } // Record the number of values shuffled from A and copied to B each second finalizeBenchmark(static_cast(m_) * k_ * num_iters); } - void padding(int num_iters) { + void padding(int num_iters) + { eigen_assert(m_ == k_); Eigen::array size_a; size_a[0] = m_; - size_a[1] = k_-3; + size_a[1] = k_ - 3; const TensorMap, Eigen::Aligned> A(a_, size_a); Eigen::array size_b; size_b[0] = k_; @@ -179,8 +169,7 @@ template class BenchmarkSuite { TensorMap, Eigen::Aligned> B(b_, size_b); #if defined(EIGEN_HAS_INDEX_LIST) - Eigen::IndexPairList, - Eigen::type2indexpair<2, 1> > paddings; + Eigen::IndexPairList, Eigen::type2indexpair<2, 1>> paddings; #else Eigen::array, 2> paddings; paddings[0] = Eigen::IndexPair(0, 0); @@ -188,14 +177,13 @@ template class BenchmarkSuite { #endif StartBenchmarkTiming(); - for (int iter = 0; iter < num_iters; ++iter) { - B.device(device_) = A.pad(paddings); - } + for (int iter = 0; iter < num_iters; ++iter) { B.device(device_) = A.pad(paddings); } // Record the number of values copied from the padded tensor A each second finalizeBenchmark(static_cast(m_) * k_ * num_iters); } - void striding(int num_iters) { + void striding(int num_iters) + { eigen_assert(m_ == k_); Eigen::array size_a; size_a[0] = m_; @@ -203,7 +191,7 @@ template class BenchmarkSuite { const TensorMap, Eigen::Aligned> A(a_, size_a); Eigen::array size_b; size_b[0] = m_; - size_b[1] = k_/2; + size_b[1] = k_ / 2; TensorMap, Eigen::Aligned> B(b_, size_b); #ifndef EIGEN_HAS_INDEX_LIST @@ -213,18 +201,17 @@ template class BenchmarkSuite { #else // Take advantage of cxx11 to give the compiler information it can use to // optimize the code. - Eigen::IndexList, Eigen::type2index<2> > strides; + Eigen::IndexList, Eigen::type2index<2>> strides; #endif StartBenchmarkTiming(); - for (int iter = 0; iter < num_iters; ++iter) { - B.device(device_) = A.stride(strides); - } + for (int iter = 0; iter < num_iters; ++iter) { B.device(device_) = A.stride(strides); } // Record the number of values copied from the padded tensor A each second finalizeBenchmark(static_cast(m_) * k_ * num_iters); } - void broadcasting(int num_iters) { + void broadcasting(int num_iters) + { Eigen::array size_a; size_a[0] = m_; size_a[1] = 1; @@ -246,14 +233,13 @@ template class BenchmarkSuite { #endif StartBenchmarkTiming(); - for (int iter = 0; iter < num_iters; ++iter) { - C.device(device_) = A.broadcast(broadcast); - } + for (int iter = 0; iter < num_iters; ++iter) { C.device(device_) = A.broadcast(broadcast); } // Record the number of values broadcasted from A and copied to C each second finalizeBenchmark(static_cast(m_) * n_ * num_iters); } - void coeffWiseOp(int num_iters) { + void coeffWiseOp(int num_iters) + { eigen_assert(m_ == k_ && k_ == n_); Eigen::array sizes; sizes[0] = m_; @@ -271,7 +257,8 @@ template class BenchmarkSuite { finalizeBenchmark(static_cast(3) * m_ * m_ * num_iters); } - void algebraicFunc(int num_iters) { + void algebraicFunc(int num_iters) + { eigen_assert(m_ == k_ && k_ == n_); Eigen::array sizes; sizes[0] = m_; @@ -281,15 +268,14 @@ template class BenchmarkSuite { TensorMap, Eigen::Aligned> C(c_, sizes); StartBenchmarkTiming(); - for (int iter = 0; iter < num_iters; ++iter) { - C.device(device_) = A.rsqrt() + B.sqrt() * B.square(); - } + for (int iter = 0; iter < num_iters; ++iter) { C.device(device_) = A.rsqrt() + B.sqrt() * B.square(); } // Record the number of FLOP executed per second (assuming one operation // per value) finalizeBenchmark(static_cast(m_) * m_ * num_iters); } - void transcendentalFunc(int num_iters) { + void transcendentalFunc(int num_iters) + { eigen_assert(m_ == k_ && k_ == n_); Eigen::array sizes; sizes[0] = m_; @@ -299,16 +285,15 @@ template class BenchmarkSuite { TensorMap, Eigen::Aligned> C(c_, sizes); StartBenchmarkTiming(); - for (int iter = 0; iter < num_iters; ++iter) { - C.device(device_) = A.exp() + B.log(); - } + for (int iter = 0; iter < num_iters; ++iter) { C.device(device_) = A.exp() + B.log(); } // Record the number of FLOP executed per second (assuming one operation // per value) finalizeBenchmark(static_cast(m_) * m_ * num_iters); } - // Row reduction - void rowReduction(int num_iters) { + // Row reduction + void rowReduction(int num_iters) + { Eigen::array input_size; input_size[0] = k_; input_size[1] = n_; @@ -327,25 +312,22 @@ template class BenchmarkSuite { #endif StartBenchmarkTiming(); - for (int iter = 0; iter < num_iters; ++iter) { - C.device(device_) = B.sum(sum_along_dim); - } + for (int iter = 0; iter < num_iters; ++iter) { C.device(device_) = B.sum(sum_along_dim); } // Record the number of FLOP executed per second (assuming one operation // per value) finalizeBenchmark(static_cast(k_) * n_ * num_iters); } // Column reduction - void colReduction(int num_iters) { + void colReduction(int num_iters) + { Eigen::array input_size; input_size[0] = k_; input_size[1] = n_; - const TensorMap, Eigen::Aligned> B( - b_, input_size); + const TensorMap, Eigen::Aligned> B(b_, input_size); Eigen::array output_size; output_size[0] = k_; - TensorMap, Eigen::Aligned> C( - c_, output_size); + TensorMap, Eigen::Aligned> C(c_, output_size); #ifndef EIGEN_HAS_INDEX_LIST Eigen::array sum_along_dim; @@ -357,36 +339,32 @@ template class BenchmarkSuite { #endif StartBenchmarkTiming(); - for (int iter = 0; iter < num_iters; ++iter) { - C.device(device_) = B.sum(sum_along_dim); - } + for (int iter = 0; iter < num_iters; ++iter) { C.device(device_) = B.sum(sum_along_dim); } // Record the number of FLOP executed per second (assuming one operation // per value) finalizeBenchmark(static_cast(k_) * n_ * num_iters); } // Full reduction - void fullReduction(int num_iters) { + void fullReduction(int num_iters) + { Eigen::array input_size; input_size[0] = k_; input_size[1] = n_; - const TensorMap, Eigen::Aligned> B( - b_, input_size); + const TensorMap, Eigen::Aligned> B(b_, input_size); Eigen::array output_size; - TensorMap, Eigen::Aligned> C( - c_, output_size); + TensorMap, Eigen::Aligned> C(c_, output_size); StartBenchmarkTiming(); - for (int iter = 0; iter < num_iters; ++iter) { - C.device(device_) = B.sum(); - } + for (int iter = 0; iter < num_iters; ++iter) { C.device(device_) = B.sum(); } // Record the number of FLOP executed per second (assuming one operation // per value) finalizeBenchmark(static_cast(k_) * n_ * num_iters); } // do a contraction which is equivalent to a matrix multiplication - void contraction(int num_iters) { + void contraction(int num_iters) + { Eigen::array sizeA; sizeA[0] = m_; sizeA[1] = k_; @@ -406,15 +384,14 @@ template class BenchmarkSuite { dims[0] = DimPair(1, 0); StartBenchmarkTiming(); - for (int iter = 0; iter < num_iters; ++iter) { - C.device(device_) = A.contract(B, dims); - } + for (int iter = 0; iter < num_iters; ++iter) { C.device(device_) = A.contract(B, dims); } // Record the number of FLOP executed per second (size_ multiplications and // additions for each value in the resulting tensor) finalizeBenchmark(static_cast(2) * m_ * n_ * k_ * num_iters); } - void convolution(int num_iters, int kernel_x, int kernel_y) { + void convolution(int num_iters, int kernel_x, int kernel_y) + { Eigen::array input_sizes; input_sizes[0] = m_; input_sizes[1] = n_; @@ -432,20 +409,19 @@ template class BenchmarkSuite { dims[1] = 1; StartBenchmarkTiming(); - for (int iter = 0; iter < num_iters; ++iter) { - C.device(device_) = A.convolve(B, dims); - } + for (int iter = 0; iter < num_iters; ++iter) { C.device(device_) = A.convolve(B, dims); } // Record the number of FLOP executed per second (kernel_size // multiplications and additions for each value in the resulting tensor) - finalizeBenchmark(static_cast(2) * - (m_ - kernel_x + 1) * (n_ - kernel_y + 1) * kernel_x * kernel_y * num_iters); + finalizeBenchmark( + static_cast(2) * (m_ - kernel_x + 1) * (n_ - kernel_y + 1) * kernel_x * kernel_y * num_iters); } - private: - void initialize() { - a_ = (T *) device_.allocate(m_ * k_ * sizeof(T)); - b_ = (T *) device_.allocate(k_ * n_ * sizeof(T)); - c_ = (T *) device_.allocate(m_ * n_ * sizeof(T)); +private: + void initialize() + { + a_ = (T *)device_.allocate(m_ * k_ * sizeof(T)); + b_ = (T *)device_.allocate(k_ * n_ * sizeof(T)); + c_ = (T *)device_.allocate(m_ * n_ * sizeof(T)); // Initialize the content of the memory pools to prevent asan from // complaining. @@ -453,14 +429,13 @@ template class BenchmarkSuite { device_.memset(b_, 23, k_ * n_ * sizeof(T)); device_.memset(c_, 31, m_ * n_ * sizeof(T)); - //BenchmarkUseRealTime(); + // BenchmarkUseRealTime(); } - inline void finalizeBenchmark(int64_t num_items) { + inline void finalizeBenchmark(int64_t num_items) + { #if defined(EIGEN_USE_GPU) && defined(__CUDACC__) - if (Eigen::internal::is_same::value) { - device_.synchronize(); - } + if (Eigen::internal::is_same::value) { device_.synchronize(); } #endif StopBenchmarkTiming(); SetBenchmarkFlopsProcessed(num_items); @@ -470,9 +445,9 @@ template class BenchmarkSuite { TensorIndex m_; TensorIndex k_; TensorIndex n_; - T* a_; - T* b_; - T* c_; + T *a_; + T *b_; + T *c_; Device device_; }; -#endif // THIRD_PARTY_EIGEN3_TENSOR_BENCHMARKS_H_ +#endif// THIRD_PARTY_EIGEN3_TENSOR_BENCHMARKS_H_ diff --git a/filmulator-gui/core/nlmeans/eigen/bench/vdw_new.cpp b/filmulator-gui/core/nlmeans/eigen/bench/vdw_new.cpp index d2604049..ed1d390f 100644 --- a/filmulator-gui/core/nlmeans/eigen/bench/vdw_new.cpp +++ b/filmulator-gui/core/nlmeans/eigen/bench/vdw_new.cpp @@ -1,5 +1,5 @@ -#include #include +#include using namespace Eigen; @@ -21,14 +21,10 @@ using namespace std; SCALAR E_VDW(const Vec &interactions1, const Vec &interactions2) { - return (interactions2.cwise()/interactions1) - .cwise().cube() - .cwise().square() - .cwise().square() - .sum(); + return (interactions2.cwise() / interactions1).cwise().cube().cwise().square().cwise().square().sum(); } -int main() +int main() { // // 1 2 3 4 ... (interactions) @@ -40,17 +36,17 @@ int main() // for // interaction) // - Vec interactions1(SIZE), interactions2(SIZE); // SIZE is the number of vdw interactions in our system + Vec interactions1(SIZE), interactions2(SIZE);// SIZE is the number of vdw interactions in our system // SetupCalculations() - SCALAR rab = 1.0; + SCALAR rab = 1.0; interactions1.setConstant(2.4); interactions2.setConstant(rab); - + // Energy() SCALAR energy = 0.0; - for (unsigned int i = 0; i::Real RealScalar; typedef std::complex Complex; -enum -{ - IsComplex = Eigen::NumTraits::IsComplex, - Conj = IsComplex -}; +enum { IsComplex = Eigen::NumTraits::IsComplex, Conj = IsComplex }; -typedef Matrix PlainMatrixType; -typedef Map, 0, OuterStride<> > MatrixType; -typedef Map, 0, OuterStride<> > ConstMatrixType; -typedef Map, 0, InnerStride > StridedVectorType; -typedef Map > CompactVectorType; +typedef Matrix PlainMatrixType; +typedef Map, 0, OuterStride<>> MatrixType; +typedef Map, 0, OuterStride<>> ConstMatrixType; +typedef Map, 0, InnerStride> StridedVectorType; +typedef Map> CompactVectorType; template -Map, 0, OuterStride<> > -matrix(T* data, int rows, int cols, int stride) +Map, 0, OuterStride<>> matrix(T *data, int rows, int cols, int stride) { - return Map, 0, OuterStride<> >(data, rows, cols, OuterStride<>(stride)); + return Map, 0, OuterStride<>>(data, rows, cols, OuterStride<>(stride)); } template -Map, 0, OuterStride<> > -matrix(const T* data, int rows, int cols, int stride) +Map, 0, OuterStride<>> matrix(const T *data, int rows, int cols, int stride) { - return Map, 0, OuterStride<> >(data, rows, cols, OuterStride<>(stride)); + return Map, 0, OuterStride<>>(data, rows, cols, OuterStride<>(stride)); } -template -Map, 0, InnerStride > make_vector(T* data, int size, int incr) +template Map, 0, InnerStride> make_vector(T *data, int size, int incr) { - return Map, 0, InnerStride >(data, size, InnerStride(incr)); + return Map, 0, InnerStride>(data, size, InnerStride(incr)); } template -Map, 0, InnerStride > make_vector(const T* data, int size, int incr) +Map, 0, InnerStride> make_vector(const T *data, int size, int incr) { - return Map, 0, InnerStride >(data, size, InnerStride(incr)); + return Map, 0, InnerStride>(data, size, InnerStride(incr)); } -template -Map > make_vector(T* data, int size) +template Map> make_vector(T *data, int size) { - return Map >(data, size); + return Map>(data, size); } -template -Map > make_vector(const T* data, int size) +template Map> make_vector(const T *data, int size) { - return Map >(data, size); + return Map>(data, size); } -template -T* get_compact_vector(T* x, int n, int incx) +template T *get_compact_vector(T *x, int n, int incx) { - if(incx==1) - return x; + if (incx == 1) return x; - typename Eigen::internal::remove_const::type* ret = new Scalar[n]; - if(incx<0) make_vector(ret,n) = make_vector(x,n,-incx).reverse(); - else make_vector(ret,n) = make_vector(x,n, incx); + typename Eigen::internal::remove_const::type *ret = new Scalar[n]; + if (incx < 0) + make_vector(ret, n) = make_vector(x, n, -incx).reverse(); + else + make_vector(ret, n) = make_vector(x, n, incx); return ret; } -template -T* copy_back(T* x_cpy, T* x, int n, int incx) +template T *copy_back(T *x_cpy, T *x, int n, int incx) { - if(x_cpy==x) - return 0; + if (x_cpy == x) return 0; - if(incx<0) make_vector(x,n,-incx).reverse() = make_vector(x_cpy,n); - else make_vector(x,n, incx) = make_vector(x_cpy,n); + if (incx < 0) + make_vector(x, n, -incx).reverse() = make_vector(x_cpy, n); + else + make_vector(x, n, incx) = make_vector(x_cpy, n); return x_cpy; } -#define EIGEN_BLAS_FUNC(X) EIGEN_CAT(SCALAR_SUFFIX,X##_) +#define EIGEN_BLAS_FUNC(X) EIGEN_CAT(SCALAR_SUFFIX, X##_) -#endif // EIGEN_BLAS_COMMON_H +#endif// EIGEN_BLAS_COMMON_H diff --git a/filmulator-gui/core/nlmeans/eigen/blas/complex_double.cpp b/filmulator-gui/core/nlmeans/eigen/blas/complex_double.cpp index 648c6d4c..985c9e9e 100644 --- a/filmulator-gui/core/nlmeans/eigen/blas/complex_double.cpp +++ b/filmulator-gui/core/nlmeans/eigen/blas/complex_double.cpp @@ -7,14 +7,14 @@ // Public License v. 2.0. If a copy of the MPL was not distributed // with this file, You can obtain one at http://mozilla.org/MPL/2.0/. -#define SCALAR std::complex +#define SCALAR std::complex #define SCALAR_SUFFIX z #define SCALAR_SUFFIX_UP "Z" #define REAL_SCALAR_SUFFIX d -#define ISCOMPLEX 1 +#define ISCOMPLEX 1 -#include "level1_impl.h" #include "level1_cplx_impl.h" -#include "level2_impl.h" +#include "level1_impl.h" #include "level2_cplx_impl.h" +#include "level2_impl.h" #include "level3_impl.h" diff --git a/filmulator-gui/core/nlmeans/eigen/blas/complex_single.cpp b/filmulator-gui/core/nlmeans/eigen/blas/complex_single.cpp index 77865194..e248a666 100644 --- a/filmulator-gui/core/nlmeans/eigen/blas/complex_single.cpp +++ b/filmulator-gui/core/nlmeans/eigen/blas/complex_single.cpp @@ -7,14 +7,14 @@ // Public License v. 2.0. If a copy of the MPL was not distributed // with this file, You can obtain one at http://mozilla.org/MPL/2.0/. -#define SCALAR std::complex +#define SCALAR std::complex #define SCALAR_SUFFIX c #define SCALAR_SUFFIX_UP "C" #define REAL_SCALAR_SUFFIX s -#define ISCOMPLEX 1 +#define ISCOMPLEX 1 -#include "level1_impl.h" #include "level1_cplx_impl.h" -#include "level2_impl.h" +#include "level1_impl.h" #include "level2_cplx_impl.h" +#include "level2_impl.h" #include "level3_impl.h" diff --git a/filmulator-gui/core/nlmeans/eigen/blas/double.cpp b/filmulator-gui/core/nlmeans/eigen/blas/double.cpp index 295b1d1f..04d02c4b 100644 --- a/filmulator-gui/core/nlmeans/eigen/blas/double.cpp +++ b/filmulator-gui/core/nlmeans/eigen/blas/double.cpp @@ -8,10 +8,10 @@ // Public License v. 2.0. If a copy of the MPL was not distributed // with this file, You can obtain one at http://mozilla.org/MPL/2.0/. -#define SCALAR double +#define SCALAR double #define SCALAR_SUFFIX d #define SCALAR_SUFFIX_UP "D" -#define ISCOMPLEX 0 +#define ISCOMPLEX 0 #include "level1_impl.h" #include "level1_real_impl.h" @@ -19,14 +19,26 @@ #include "level2_real_impl.h" #include "level3_impl.h" -double BLASFUNC(dsdot)(int* n, float* x, int* incx, float* y, int* incy) +double BLASFUNC(dsdot)(int *n, float *x, int *incx, float *y, int *incy) { - if(*n<=0) return 0; + if (*n <= 0) return 0; - if(*incx==1 && *incy==1) return (make_vector(x,*n).cast().cwiseProduct(make_vector(y,*n).cast())).sum(); - else if(*incx>0 && *incy>0) return (make_vector(x,*n,*incx).cast().cwiseProduct(make_vector(y,*n,*incy).cast())).sum(); - else if(*incx<0 && *incy>0) return (make_vector(x,*n,-*incx).reverse().cast().cwiseProduct(make_vector(y,*n,*incy).cast())).sum(); - else if(*incx>0 && *incy<0) return (make_vector(x,*n,*incx).cast().cwiseProduct(make_vector(y,*n,-*incy).reverse().cast())).sum(); - else if(*incx<0 && *incy<0) return (make_vector(x,*n,-*incx).reverse().cast().cwiseProduct(make_vector(y,*n,-*incy).reverse().cast())).sum(); - else return 0; + if (*incx == 1 && *incy == 1) + return (make_vector(x, *n).cast().cwiseProduct(make_vector(y, *n).cast())).sum(); + else if (*incx > 0 && *incy > 0) + return (make_vector(x, *n, *incx).cast().cwiseProduct(make_vector(y, *n, *incy).cast())).sum(); + else if (*incx < 0 && *incy > 0) + return (make_vector(x, *n, -*incx).reverse().cast().cwiseProduct(make_vector(y, *n, *incy).cast())) + .sum(); + else if (*incx > 0 && *incy < 0) + return (make_vector(x, *n, *incx).cast().cwiseProduct(make_vector(y, *n, -*incy).reverse().cast())) + .sum(); + else if (*incx < 0 && *incy < 0) + return (make_vector(x, *n, -*incx) + .reverse() + .cast() + .cwiseProduct(make_vector(y, *n, -*incy).reverse().cast())) + .sum(); + else + return 0; } diff --git a/filmulator-gui/core/nlmeans/eigen/blas/f2c/datatypes.h b/filmulator-gui/core/nlmeans/eigen/blas/f2c/datatypes.h index 63232b24..cd36d420 100644 --- a/filmulator-gui/core/nlmeans/eigen/blas/f2c/datatypes.h +++ b/filmulator-gui/core/nlmeans/eigen/blas/f2c/datatypes.h @@ -9,16 +9,22 @@ typedef int integer; typedef unsigned int uinteger; typedef float real; typedef double doublereal; -typedef struct { real r, i; } complex; -typedef struct { doublereal r, i; } doublecomplex; +typedef struct +{ + real r, i; +} complex; +typedef struct +{ + doublereal r, i; +} doublecomplex; typedef int ftnlen; typedef int logical; #define abs(x) ((x) >= 0 ? (x) : -(x)) -#define dabs(x) (doublereal)abs(x) -#define min(a,b) ((a) <= (b) ? (a) : (b)) -#define max(a,b) ((a) >= (b) ? (a) : (b)) -#define dmin(a,b) (doublereal)min(a,b) -#define dmax(a,b) (doublereal)max(a,b) +#define dabs(x) (doublereal) abs(x) +#define min(a, b) ((a) <= (b) ? (a) : (b)) +#define max(a, b) ((a) >= (b) ? (a) : (b)) +#define dmin(a, b) (doublereal) min(a, b) +#define dmax(a, b) (doublereal) max(a, b) #endif diff --git a/filmulator-gui/core/nlmeans/eigen/blas/level1_cplx_impl.h b/filmulator-gui/core/nlmeans/eigen/blas/level1_cplx_impl.h index 719f5bac..e4bd4956 100644 --- a/filmulator-gui/core/nlmeans/eigen/blas/level1_cplx_impl.h +++ b/filmulator-gui/core/nlmeans/eigen/blas/level1_cplx_impl.h @@ -9,125 +9,142 @@ #include "common.h" -struct scalar_norm1_op { +struct scalar_norm1_op +{ typedef RealScalar result_type; EIGEN_EMPTY_STRUCT_CTOR(scalar_norm1_op) - inline RealScalar operator() (const Scalar& a) const { return numext::norm1(a); } + inline RealScalar operator()(const Scalar &a) const { return numext::norm1(a); } }; namespace Eigen { - namespace internal { - template<> struct functor_traits - { - enum { Cost = 3 * NumTraits::AddCost, PacketAccess = 0 }; - }; - } -} +namespace internal { + template<> struct functor_traits + { + enum { Cost = 3 * NumTraits::AddCost, PacketAccess = 0 }; + }; +}// namespace internal +}// namespace Eigen // computes the sum of magnitudes of all vector elements or, for a complex vector x, the sum // res = |Rex1| + |Imx1| + |Rex2| + |Imx2| + ... + |Rexn| + |Imxn|, where x is a vector of order n -RealScalar EIGEN_CAT(EIGEN_CAT(REAL_SCALAR_SUFFIX,SCALAR_SUFFIX),asum_)(int *n, RealScalar *px, int *incx) +RealScalar EIGEN_CAT(EIGEN_CAT(REAL_SCALAR_SUFFIX, SCALAR_SUFFIX), asum_)(int *n, RealScalar *px, int *incx) { -// std::cerr << "__asum " << *n << " " << *incx << "\n"; - Complex* x = reinterpret_cast(px); + // std::cerr << "__asum " << *n << " " << *incx << "\n"; + Complex *x = reinterpret_cast(px); - if(*n<=0) return 0; + if (*n <= 0) return 0; - if(*incx==1) return make_vector(x,*n).unaryExpr().sum(); - else return make_vector(x,*n,std::abs(*incx)).unaryExpr().sum(); + if (*incx == 1) + return make_vector(x, *n).unaryExpr().sum(); + else + return make_vector(x, *n, std::abs(*incx)).unaryExpr().sum(); } // computes a dot product of a conjugated vector with another vector. -int EIGEN_BLAS_FUNC(dotcw)(int *n, RealScalar *px, int *incx, RealScalar *py, int *incy, RealScalar* pres) +int EIGEN_BLAS_FUNC(dotcw)(int *n, RealScalar *px, int *incx, RealScalar *py, int *incy, RealScalar *pres) { -// std::cerr << "_dotc " << *n << " " << *incx << " " << *incy << "\n"; - Scalar* res = reinterpret_cast(pres); + // std::cerr << "_dotc " << *n << " " << *incx << " " << *incy << "\n"; + Scalar *res = reinterpret_cast(pres); - if(*n<=0) - { + if (*n <= 0) { *res = Scalar(0); return 0; } - Scalar* x = reinterpret_cast(px); - Scalar* y = reinterpret_cast(py); - - if(*incx==1 && *incy==1) *res = (make_vector(x,*n).dot(make_vector(y,*n))); - else if(*incx>0 && *incy>0) *res = (make_vector(x,*n,*incx).dot(make_vector(y,*n,*incy))); - else if(*incx<0 && *incy>0) *res = (make_vector(x,*n,-*incx).reverse().dot(make_vector(y,*n,*incy))); - else if(*incx>0 && *incy<0) *res = (make_vector(x,*n,*incx).dot(make_vector(y,*n,-*incy).reverse())); - else if(*incx<0 && *incy<0) *res = (make_vector(x,*n,-*incx).reverse().dot(make_vector(y,*n,-*incy).reverse())); + Scalar *x = reinterpret_cast(px); + Scalar *y = reinterpret_cast(py); + + if (*incx == 1 && *incy == 1) + *res = (make_vector(x, *n).dot(make_vector(y, *n))); + else if (*incx > 0 && *incy > 0) + *res = (make_vector(x, *n, *incx).dot(make_vector(y, *n, *incy))); + else if (*incx < 0 && *incy > 0) + *res = (make_vector(x, *n, -*incx).reverse().dot(make_vector(y, *n, *incy))); + else if (*incx > 0 && *incy < 0) + *res = (make_vector(x, *n, *incx).dot(make_vector(y, *n, -*incy).reverse())); + else if (*incx < 0 && *incy < 0) + *res = (make_vector(x, *n, -*incx).reverse().dot(make_vector(y, *n, -*incy).reverse())); return 0; } // computes a vector-vector dot product without complex conjugation. -int EIGEN_BLAS_FUNC(dotuw)(int *n, RealScalar *px, int *incx, RealScalar *py, int *incy, RealScalar* pres) +int EIGEN_BLAS_FUNC(dotuw)(int *n, RealScalar *px, int *incx, RealScalar *py, int *incy, RealScalar *pres) { - Scalar* res = reinterpret_cast(pres); + Scalar *res = reinterpret_cast(pres); - if(*n<=0) - { + if (*n <= 0) { *res = Scalar(0); return 0; } - Scalar* x = reinterpret_cast(px); - Scalar* y = reinterpret_cast(py); - - if(*incx==1 && *incy==1) *res = (make_vector(x,*n).cwiseProduct(make_vector(y,*n))).sum(); - else if(*incx>0 && *incy>0) *res = (make_vector(x,*n,*incx).cwiseProduct(make_vector(y,*n,*incy))).sum(); - else if(*incx<0 && *incy>0) *res = (make_vector(x,*n,-*incx).reverse().cwiseProduct(make_vector(y,*n,*incy))).sum(); - else if(*incx>0 && *incy<0) *res = (make_vector(x,*n,*incx).cwiseProduct(make_vector(y,*n,-*incy).reverse())).sum(); - else if(*incx<0 && *incy<0) *res = (make_vector(x,*n,-*incx).reverse().cwiseProduct(make_vector(y,*n,-*incy).reverse())).sum(); + Scalar *x = reinterpret_cast(px); + Scalar *y = reinterpret_cast(py); + + if (*incx == 1 && *incy == 1) + *res = (make_vector(x, *n).cwiseProduct(make_vector(y, *n))).sum(); + else if (*incx > 0 && *incy > 0) + *res = (make_vector(x, *n, *incx).cwiseProduct(make_vector(y, *n, *incy))).sum(); + else if (*incx < 0 && *incy > 0) + *res = (make_vector(x, *n, -*incx).reverse().cwiseProduct(make_vector(y, *n, *incy))).sum(); + else if (*incx > 0 && *incy < 0) + *res = (make_vector(x, *n, *incx).cwiseProduct(make_vector(y, *n, -*incy).reverse())).sum(); + else if (*incx < 0 && *incy < 0) + *res = (make_vector(x, *n, -*incx).reverse().cwiseProduct(make_vector(y, *n, -*incy).reverse())).sum(); return 0; } -RealScalar EIGEN_CAT(EIGEN_CAT(REAL_SCALAR_SUFFIX,SCALAR_SUFFIX),nrm2_)(int *n, RealScalar *px, int *incx) +RealScalar EIGEN_CAT(EIGEN_CAT(REAL_SCALAR_SUFFIX, SCALAR_SUFFIX), nrm2_)(int *n, RealScalar *px, int *incx) { -// std::cerr << "__nrm2 " << *n << " " << *incx << "\n"; - if(*n<=0) return 0; + // std::cerr << "__nrm2 " << *n << " " << *incx << "\n"; + if (*n <= 0) return 0; - Scalar* x = reinterpret_cast(px); + Scalar *x = reinterpret_cast(px); - if(*incx==1) - return make_vector(x,*n).stableNorm(); + if (*incx == 1) return make_vector(x, *n).stableNorm(); - return make_vector(x,*n,*incx).stableNorm(); + return make_vector(x, *n, *incx).stableNorm(); } -int EIGEN_CAT(EIGEN_CAT(SCALAR_SUFFIX,REAL_SCALAR_SUFFIX),rot_)(int *n, RealScalar *px, int *incx, RealScalar *py, int *incy, RealScalar *pc, RealScalar *ps) +int EIGEN_CAT(EIGEN_CAT(SCALAR_SUFFIX, REAL_SCALAR_SUFFIX), + rot_)(int *n, RealScalar *px, int *incx, RealScalar *py, int *incy, RealScalar *pc, RealScalar *ps) { - if(*n<=0) return 0; + if (*n <= 0) return 0; - Scalar* x = reinterpret_cast(px); - Scalar* y = reinterpret_cast(py); + Scalar *x = reinterpret_cast(px); + Scalar *y = reinterpret_cast(py); RealScalar c = *pc; RealScalar s = *ps; - StridedVectorType vx(make_vector(x,*n,std::abs(*incx))); - StridedVectorType vy(make_vector(y,*n,std::abs(*incy))); + StridedVectorType vx(make_vector(x, *n, std::abs(*incx))); + StridedVectorType vy(make_vector(y, *n, std::abs(*incy))); Reverse rvx(vx); Reverse rvy(vy); // TODO implement mixed real-scalar rotations - if(*incx<0 && *incy>0) internal::apply_rotation_in_the_plane(rvx, vy, JacobiRotation(c,s)); - else if(*incx>0 && *incy<0) internal::apply_rotation_in_the_plane(vx, rvy, JacobiRotation(c,s)); - else internal::apply_rotation_in_the_plane(vx, vy, JacobiRotation(c,s)); + if (*incx < 0 && *incy > 0) + internal::apply_rotation_in_the_plane(rvx, vy, JacobiRotation(c, s)); + else if (*incx > 0 && *incy < 0) + internal::apply_rotation_in_the_plane(vx, rvy, JacobiRotation(c, s)); + else + internal::apply_rotation_in_the_plane(vx, vy, JacobiRotation(c, s)); return 0; } -int EIGEN_CAT(EIGEN_CAT(SCALAR_SUFFIX,REAL_SCALAR_SUFFIX),scal_)(int *n, RealScalar *palpha, RealScalar *px, int *incx) +int EIGEN_CAT(EIGEN_CAT(SCALAR_SUFFIX, REAL_SCALAR_SUFFIX), + scal_)(int *n, RealScalar *palpha, RealScalar *px, int *incx) { - if(*n<=0) return 0; + if (*n <= 0) return 0; - Scalar* x = reinterpret_cast(px); + Scalar *x = reinterpret_cast(px); RealScalar alpha = *palpha; -// std::cerr << "__scal " << *n << " " << alpha << " " << *incx << "\n"; + // std::cerr << "__scal " << *n << " " << alpha << " " << *incx << "\n"; - if(*incx==1) make_vector(x,*n) *= alpha; - else make_vector(x,*n,std::abs(*incx)) *= alpha; + if (*incx == 1) + make_vector(x, *n) *= alpha; + else + make_vector(x, *n, std::abs(*incx)) *= alpha; return 0; } diff --git a/filmulator-gui/core/nlmeans/eigen/blas/level1_impl.h b/filmulator-gui/core/nlmeans/eigen/blas/level1_impl.h index f857bfa2..79424c50 100644 --- a/filmulator-gui/core/nlmeans/eigen/blas/level1_impl.h +++ b/filmulator-gui/core/nlmeans/eigen/blas/level1_impl.h @@ -9,39 +9,43 @@ #include "common.h" -int EIGEN_BLAS_FUNC(axpy)(const int *n, const RealScalar *palpha, const RealScalar *px, const int *incx, RealScalar *py, const int *incy) +int EIGEN_BLAS_FUNC( + axpy)(const int *n, const RealScalar *palpha, const RealScalar *px, const int *incx, RealScalar *py, const int *incy) { - const Scalar* x = reinterpret_cast(px); - Scalar* y = reinterpret_cast(py); - Scalar alpha = *reinterpret_cast(palpha); - - if(*n<=0) return 0; - - if(*incx==1 && *incy==1) make_vector(y,*n) += alpha * make_vector(x,*n); - else if(*incx>0 && *incy>0) make_vector(y,*n,*incy) += alpha * make_vector(x,*n,*incx); - else if(*incx>0 && *incy<0) make_vector(y,*n,-*incy).reverse() += alpha * make_vector(x,*n,*incx); - else if(*incx<0 && *incy>0) make_vector(y,*n,*incy) += alpha * make_vector(x,*n,-*incx).reverse(); - else if(*incx<0 && *incy<0) make_vector(y,*n,-*incy).reverse() += alpha * make_vector(x,*n,-*incx).reverse(); + const Scalar *x = reinterpret_cast(px); + Scalar *y = reinterpret_cast(py); + Scalar alpha = *reinterpret_cast(palpha); + + if (*n <= 0) return 0; + + if (*incx == 1 && *incy == 1) + make_vector(y, *n) += alpha * make_vector(x, *n); + else if (*incx > 0 && *incy > 0) + make_vector(y, *n, *incy) += alpha * make_vector(x, *n, *incx); + else if (*incx > 0 && *incy < 0) + make_vector(y, *n, -*incy).reverse() += alpha * make_vector(x, *n, *incx); + else if (*incx < 0 && *incy > 0) + make_vector(y, *n, *incy) += alpha * make_vector(x, *n, -*incx).reverse(); + else if (*incx < 0 && *incy < 0) + make_vector(y, *n, -*incy).reverse() += alpha * make_vector(x, *n, -*incx).reverse(); return 0; } int EIGEN_BLAS_FUNC(copy)(int *n, RealScalar *px, int *incx, RealScalar *py, int *incy) { - if(*n<=0) return 0; + if (*n <= 0) return 0; - Scalar* x = reinterpret_cast(px); - Scalar* y = reinterpret_cast(py); + Scalar *x = reinterpret_cast(px); + Scalar *y = reinterpret_cast(py); // be carefull, *incx==0 is allowed !! - if(*incx==1 && *incy==1) - make_vector(y,*n) = make_vector(x,*n); - else - { - if(*incx<0) x = x - (*n-1)*(*incx); - if(*incy<0) y = y - (*n-1)*(*incy); - for(int i=0;i<*n;++i) - { + if (*incx == 1 && *incy == 1) + make_vector(y, *n) = make_vector(x, *n); + else { + if (*incx < 0) x = x - (*n - 1) * (*incx); + if (*incy < 0) y = y - (*n - 1) * (*incy); + for (int i = 0; i < *n; ++i) { *y = *x; x += *incx; y += *incy; @@ -51,26 +55,30 @@ int EIGEN_BLAS_FUNC(copy)(int *n, RealScalar *px, int *incx, RealScalar *py, int return 0; } -int EIGEN_CAT(EIGEN_CAT(i,SCALAR_SUFFIX),amax_)(int *n, RealScalar *px, int *incx) +int EIGEN_CAT(EIGEN_CAT(i, SCALAR_SUFFIX), amax_)(int *n, RealScalar *px, int *incx) { - if(*n<=0) return 0; - Scalar* x = reinterpret_cast(px); + if (*n <= 0) return 0; + Scalar *x = reinterpret_cast(px); DenseIndex ret; - if(*incx==1) make_vector(x,*n).cwiseAbs().maxCoeff(&ret); - else make_vector(x,*n,std::abs(*incx)).cwiseAbs().maxCoeff(&ret); - return int(ret)+1; + if (*incx == 1) + make_vector(x, *n).cwiseAbs().maxCoeff(&ret); + else + make_vector(x, *n, std::abs(*incx)).cwiseAbs().maxCoeff(&ret); + return int(ret) + 1; } -int EIGEN_CAT(EIGEN_CAT(i,SCALAR_SUFFIX),amin_)(int *n, RealScalar *px, int *incx) +int EIGEN_CAT(EIGEN_CAT(i, SCALAR_SUFFIX), amin_)(int *n, RealScalar *px, int *incx) { - if(*n<=0) return 0; - Scalar* x = reinterpret_cast(px); + if (*n <= 0) return 0; + Scalar *x = reinterpret_cast(px); DenseIndex ret; - if(*incx==1) make_vector(x,*n).cwiseAbs().minCoeff(&ret); - else make_vector(x,*n,std::abs(*incx)).cwiseAbs().minCoeff(&ret); - return int(ret)+1; + if (*incx == 1) + make_vector(x, *n).cwiseAbs().minCoeff(&ret); + else + make_vector(x, *n, std::abs(*incx)).cwiseAbs().minCoeff(&ret); + return int(ret) + 1; } int EIGEN_BLAS_FUNC(rotg)(RealScalar *pa, RealScalar *pb, RealScalar *pc, RealScalar *ps) @@ -78,89 +86,89 @@ int EIGEN_BLAS_FUNC(rotg)(RealScalar *pa, RealScalar *pb, RealScalar *pc, RealSc using std::sqrt; using std::abs; - Scalar& a = *reinterpret_cast(pa); - Scalar& b = *reinterpret_cast(pb); - RealScalar* c = pc; - Scalar* s = reinterpret_cast(ps); + Scalar &a = *reinterpret_cast(pa); + Scalar &b = *reinterpret_cast(pb); + RealScalar *c = pc; + Scalar *s = reinterpret_cast(ps); - #if !ISCOMPLEX - Scalar r,z; +#if !ISCOMPLEX + Scalar r, z; Scalar aa = abs(a); Scalar ab = abs(b); - if((aa+ab)==Scalar(0)) - { + if ((aa + ab) == Scalar(0)) { *c = 1; *s = 0; r = 0; z = 0; - } - else - { - r = sqrt(a*a + b*b); - Scalar amax = aa>ab ? a : b; - r = amax>0 ? r : -r; - *c = a/r; - *s = b/r; + } else { + r = sqrt(a * a + b * b); + Scalar amax = aa > ab ? a : b; + r = amax > 0 ? r : -r; + *c = a / r; + *s = b / r; z = 1; if (aa > ab) z = *s; - if (ab > aa && *c!=RealScalar(0)) - z = Scalar(1)/ *c; + if (ab > aa && *c != RealScalar(0)) z = Scalar(1) / *c; } *pa = r; *pb = z; - #else +#else Scalar alpha; - RealScalar norm,scale; - if(abs(a)==RealScalar(0)) - { + RealScalar norm, scale; + if (abs(a) == RealScalar(0)) { *c = RealScalar(0); *s = Scalar(1); a = b; - } - else - { + } else { scale = abs(a) + abs(b); - norm = scale*sqrt((numext::abs2(a/scale)) + (numext::abs2(b/scale))); - alpha = a/abs(a); - *c = abs(a)/norm; - *s = alpha*numext::conj(b)/norm; - a = alpha*norm; + norm = scale * sqrt((numext::abs2(a / scale)) + (numext::abs2(b / scale))); + alpha = a / abs(a); + *c = abs(a) / norm; + *s = alpha * numext::conj(b) / norm; + a = alpha * norm; } - #endif +#endif -// JacobiRotation r; -// r.makeGivens(a,b); -// *c = r.c(); -// *s = r.s(); + // JacobiRotation r; + // r.makeGivens(a,b); + // *c = r.c(); + // *s = r.s(); return 0; } int EIGEN_BLAS_FUNC(scal)(int *n, RealScalar *palpha, RealScalar *px, int *incx) { - if(*n<=0) return 0; + if (*n <= 0) return 0; - Scalar* x = reinterpret_cast(px); - Scalar alpha = *reinterpret_cast(palpha); + Scalar *x = reinterpret_cast(px); + Scalar alpha = *reinterpret_cast(palpha); - if(*incx==1) make_vector(x,*n) *= alpha; - else make_vector(x,*n,std::abs(*incx)) *= alpha; + if (*incx == 1) + make_vector(x, *n) *= alpha; + else + make_vector(x, *n, std::abs(*incx)) *= alpha; return 0; } int EIGEN_BLAS_FUNC(swap)(int *n, RealScalar *px, int *incx, RealScalar *py, int *incy) { - if(*n<=0) return 0; - - Scalar* x = reinterpret_cast(px); - Scalar* y = reinterpret_cast(py); - - if(*incx==1 && *incy==1) make_vector(y,*n).swap(make_vector(x,*n)); - else if(*incx>0 && *incy>0) make_vector(y,*n,*incy).swap(make_vector(x,*n,*incx)); - else if(*incx>0 && *incy<0) make_vector(y,*n,-*incy).reverse().swap(make_vector(x,*n,*incx)); - else if(*incx<0 && *incy>0) make_vector(y,*n,*incy).swap(make_vector(x,*n,-*incx).reverse()); - else if(*incx<0 && *incy<0) make_vector(y,*n,-*incy).reverse().swap(make_vector(x,*n,-*incx).reverse()); + if (*n <= 0) return 0; + + Scalar *x = reinterpret_cast(px); + Scalar *y = reinterpret_cast(py); + + if (*incx == 1 && *incy == 1) + make_vector(y, *n).swap(make_vector(x, *n)); + else if (*incx > 0 && *incy > 0) + make_vector(y, *n, *incy).swap(make_vector(x, *n, *incx)); + else if (*incx > 0 && *incy < 0) + make_vector(y, *n, -*incy).reverse().swap(make_vector(x, *n, *incx)); + else if (*incx < 0 && *incy > 0) + make_vector(y, *n, *incy).swap(make_vector(x, *n, -*incx).reverse()); + else if (*incx < 0 && *incy < 0) + make_vector(y, *n, -*incy).reverse().swap(make_vector(x, *n, -*incx).reverse()); return 1; } diff --git a/filmulator-gui/core/nlmeans/eigen/blas/level1_real_impl.h b/filmulator-gui/core/nlmeans/eigen/blas/level1_real_impl.h index 02586d51..9a026a27 100644 --- a/filmulator-gui/core/nlmeans/eigen/blas/level1_real_impl.h +++ b/filmulator-gui/core/nlmeans/eigen/blas/level1_real_impl.h @@ -13,66 +13,79 @@ // res = |Rex1| + |Imx1| + |Rex2| + |Imx2| + ... + |Rexn| + |Imxn|, where x is a vector of order n RealScalar EIGEN_BLAS_FUNC(asum)(int *n, RealScalar *px, int *incx) { -// std::cerr << "_asum " << *n << " " << *incx << "\n"; + // std::cerr << "_asum " << *n << " " << *incx << "\n"; - Scalar* x = reinterpret_cast(px); + Scalar *x = reinterpret_cast(px); - if(*n<=0) return 0; + if (*n <= 0) return 0; - if(*incx==1) return make_vector(x,*n).cwiseAbs().sum(); - else return make_vector(x,*n,std::abs(*incx)).cwiseAbs().sum(); + if (*incx == 1) + return make_vector(x, *n).cwiseAbs().sum(); + else + return make_vector(x, *n, std::abs(*incx)).cwiseAbs().sum(); } // computes a vector-vector dot product. Scalar EIGEN_BLAS_FUNC(dot)(int *n, RealScalar *px, int *incx, RealScalar *py, int *incy) { -// std::cerr << "_dot " << *n << " " << *incx << " " << *incy << "\n"; - - if(*n<=0) return 0; - - Scalar* x = reinterpret_cast(px); - Scalar* y = reinterpret_cast(py); - - if(*incx==1 && *incy==1) return (make_vector(x,*n).cwiseProduct(make_vector(y,*n))).sum(); - else if(*incx>0 && *incy>0) return (make_vector(x,*n,*incx).cwiseProduct(make_vector(y,*n,*incy))).sum(); - else if(*incx<0 && *incy>0) return (make_vector(x,*n,-*incx).reverse().cwiseProduct(make_vector(y,*n,*incy))).sum(); - else if(*incx>0 && *incy<0) return (make_vector(x,*n,*incx).cwiseProduct(make_vector(y,*n,-*incy).reverse())).sum(); - else if(*incx<0 && *incy<0) return (make_vector(x,*n,-*incx).reverse().cwiseProduct(make_vector(y,*n,-*incy).reverse())).sum(); - else return 0; + // std::cerr << "_dot " << *n << " " << *incx << " " << *incy << "\n"; + + if (*n <= 0) return 0; + + Scalar *x = reinterpret_cast(px); + Scalar *y = reinterpret_cast(py); + + if (*incx == 1 && *incy == 1) + return (make_vector(x, *n).cwiseProduct(make_vector(y, *n))).sum(); + else if (*incx > 0 && *incy > 0) + return (make_vector(x, *n, *incx).cwiseProduct(make_vector(y, *n, *incy))).sum(); + else if (*incx < 0 && *incy > 0) + return (make_vector(x, *n, -*incx).reverse().cwiseProduct(make_vector(y, *n, *incy))).sum(); + else if (*incx > 0 && *incy < 0) + return (make_vector(x, *n, *incx).cwiseProduct(make_vector(y, *n, -*incy).reverse())).sum(); + else if (*incx < 0 && *incy < 0) + return (make_vector(x, *n, -*incx).reverse().cwiseProduct(make_vector(y, *n, -*incy).reverse())).sum(); + else + return 0; } // computes the Euclidean norm of a vector. // FIXME Scalar EIGEN_BLAS_FUNC(nrm2)(int *n, RealScalar *px, int *incx) { -// std::cerr << "_nrm2 " << *n << " " << *incx << "\n"; - if(*n<=0) return 0; + // std::cerr << "_nrm2 " << *n << " " << *incx << "\n"; + if (*n <= 0) return 0; - Scalar* x = reinterpret_cast(px); + Scalar *x = reinterpret_cast(px); - if(*incx==1) return make_vector(x,*n).stableNorm(); - else return make_vector(x,*n,std::abs(*incx)).stableNorm(); + if (*incx == 1) + return make_vector(x, *n).stableNorm(); + else + return make_vector(x, *n, std::abs(*incx)).stableNorm(); } int EIGEN_BLAS_FUNC(rot)(int *n, RealScalar *px, int *incx, RealScalar *py, int *incy, RealScalar *pc, RealScalar *ps) { -// std::cerr << "_rot " << *n << " " << *incx << " " << *incy << "\n"; - if(*n<=0) return 0; + // std::cerr << "_rot " << *n << " " << *incx << " " << *incy << "\n"; + if (*n <= 0) return 0; - Scalar* x = reinterpret_cast(px); - Scalar* y = reinterpret_cast(py); - Scalar c = *reinterpret_cast(pc); - Scalar s = *reinterpret_cast(ps); + Scalar *x = reinterpret_cast(px); + Scalar *y = reinterpret_cast(py); + Scalar c = *reinterpret_cast(pc); + Scalar s = *reinterpret_cast(ps); - StridedVectorType vx(make_vector(x,*n,std::abs(*incx))); - StridedVectorType vy(make_vector(y,*n,std::abs(*incy))); + StridedVectorType vx(make_vector(x, *n, std::abs(*incx))); + StridedVectorType vy(make_vector(y, *n, std::abs(*incy))); Reverse rvx(vx); Reverse rvy(vy); - if(*incx<0 && *incy>0) internal::apply_rotation_in_the_plane(rvx, vy, JacobiRotation(c,s)); - else if(*incx>0 && *incy<0) internal::apply_rotation_in_the_plane(vx, rvy, JacobiRotation(c,s)); - else internal::apply_rotation_in_the_plane(vx, vy, JacobiRotation(c,s)); + if (*incx < 0 && *incy > 0) + internal::apply_rotation_in_the_plane(rvx, vy, JacobiRotation(c, s)); + else if (*incx > 0 && *incy < 0) + internal::apply_rotation_in_the_plane(vx, rvy, JacobiRotation(c, s)); + else + internal::apply_rotation_in_the_plane(vx, vy, JacobiRotation(c, s)); return 0; diff --git a/filmulator-gui/core/nlmeans/eigen/blas/level2_cplx_impl.h b/filmulator-gui/core/nlmeans/eigen/blas/level2_cplx_impl.h index e3ce6143..fee5e074 100644 --- a/filmulator-gui/core/nlmeans/eigen/blas/level2_cplx_impl.h +++ b/filmulator-gui/core/nlmeans/eigen/blas/level2_cplx_impl.h @@ -10,73 +10,83 @@ #include "common.h" /** ZHEMV performs the matrix-vector operation - * - * y := alpha*A*x + beta*y, - * - * where alpha and beta are scalars, x and y are n element vectors and - * A is an n by n hermitian matrix. - */ -int EIGEN_BLAS_FUNC(hemv)(const char *uplo, const int *n, const RealScalar *palpha, const RealScalar *pa, const int *lda, - const RealScalar *px, const int *incx, const RealScalar *pbeta, RealScalar *py, const int *incy) + * + * y := alpha*A*x + beta*y, + * + * where alpha and beta are scalars, x and y are n element vectors and + * A is an n by n hermitian matrix. + */ +int EIGEN_BLAS_FUNC(hemv)(const char *uplo, + const int *n, + const RealScalar *palpha, + const RealScalar *pa, + const int *lda, + const RealScalar *px, + const int *incx, + const RealScalar *pbeta, + RealScalar *py, + const int *incy) { - typedef void (*functype)(int, const Scalar*, int, const Scalar*, Scalar*, Scalar); + typedef void (*functype)(int, const Scalar *, int, const Scalar *, Scalar *, Scalar); static const functype func[2] = { // array index: UP - (internal::selfadjoint_matrix_vector_product::run), + (internal::selfadjoint_matrix_vector_product::run), // array index: LO - (internal::selfadjoint_matrix_vector_product::run), + (internal::selfadjoint_matrix_vector_product::run), }; - const Scalar* a = reinterpret_cast(pa); - const Scalar* x = reinterpret_cast(px); - Scalar* y = reinterpret_cast(py); - Scalar alpha = *reinterpret_cast(palpha); - Scalar beta = *reinterpret_cast(pbeta); + const Scalar *a = reinterpret_cast(pa); + const Scalar *x = reinterpret_cast(px); + Scalar *y = reinterpret_cast(py); + Scalar alpha = *reinterpret_cast(palpha); + Scalar beta = *reinterpret_cast(pbeta); // check arguments int info = 0; - if(UPLO(*uplo)==INVALID) info = 1; - else if(*n<0) info = 2; - else if(*lda=2 || func[code]==0) - return 0; + if (code >= 2 || func[code] == 0) return 0; func[code](*n, a, *lda, actual_x, actual_y, alpha); } - if(actual_x!=x) delete[] actual_x; - if(actual_y!=y) delete[] copy_back(actual_y,y,*n,*incy); + if (actual_x != x) delete[] actual_x; + if (actual_y != y) delete[] copy_back(actual_y, y, *n, *incy); return 1; } /** ZHBMV performs the matrix-vector operation - * - * y := alpha*A*x + beta*y, - * - * where alpha and beta are scalars, x and y are n element vectors and - * A is an n by n hermitian band matrix, with k super-diagonals. - */ + * + * y := alpha*A*x + beta*y, + * + * where alpha and beta are scalars, x and y are n element vectors and + * A is an n by n hermitian band matrix, with k super-diagonals. + */ // int EIGEN_BLAS_FUNC(hbmv)(char *uplo, int *n, int *k, RealScalar *alpha, RealScalar *a, int *lda, // RealScalar *x, int *incx, RealScalar *beta, RealScalar *y, int *incy) // { @@ -84,277 +94,313 @@ int EIGEN_BLAS_FUNC(hemv)(const char *uplo, const int *n, const RealScalar *palp // } /** ZHPMV performs the matrix-vector operation - * - * y := alpha*A*x + beta*y, - * - * where alpha and beta are scalars, x and y are n element vectors and - * A is an n by n hermitian matrix, supplied in packed form. - */ -// int EIGEN_BLAS_FUNC(hpmv)(char *uplo, int *n, RealScalar *alpha, RealScalar *ap, RealScalar *x, int *incx, RealScalar *beta, RealScalar *y, int *incy) + * + * y := alpha*A*x + beta*y, + * + * where alpha and beta are scalars, x and y are n element vectors and + * A is an n by n hermitian matrix, supplied in packed form. + */ +// int EIGEN_BLAS_FUNC(hpmv)(char *uplo, int *n, RealScalar *alpha, RealScalar *ap, RealScalar *x, int *incx, RealScalar +// *beta, RealScalar *y, int *incy) // { // return 1; // } /** ZHPR performs the hermitian rank 1 operation - * - * A := alpha*x*conjg( x' ) + A, - * - * where alpha is a real scalar, x is an n element vector and A is an - * n by n hermitian matrix, supplied in packed form. - */ + * + * A := alpha*x*conjg( x' ) + A, + * + * where alpha is a real scalar, x is an n element vector and A is an + * n by n hermitian matrix, supplied in packed form. + */ int EIGEN_BLAS_FUNC(hpr)(char *uplo, int *n, RealScalar *palpha, RealScalar *px, int *incx, RealScalar *pap) { - typedef void (*functype)(int, Scalar*, const Scalar*, RealScalar); + typedef void (*functype)(int, Scalar *, const Scalar *, RealScalar); static const functype func[2] = { // array index: UP - (internal::selfadjoint_packed_rank1_update::run), + (internal::selfadjoint_packed_rank1_update::run), // array index: LO - (internal::selfadjoint_packed_rank1_update::run), + (internal::selfadjoint_packed_rank1_update::run), }; - Scalar* x = reinterpret_cast(px); - Scalar* ap = reinterpret_cast(pap); + Scalar *x = reinterpret_cast(px); + Scalar *ap = reinterpret_cast(pap); RealScalar alpha = *palpha; int info = 0; - if(UPLO(*uplo)==INVALID) info = 1; - else if(*n<0) info = 2; - else if(*incx==0) info = 5; - if(info) - return xerbla_(SCALAR_SUFFIX_UP"HPR ",&info,6); + if (UPLO(*uplo) == INVALID) + info = 1; + else if (*n < 0) + info = 2; + else if (*incx == 0) + info = 5; + if (info) return xerbla_(SCALAR_SUFFIX_UP "HPR ", &info, 6); - if(alpha==Scalar(0)) - return 1; + if (alpha == Scalar(0)) return 1; - Scalar* x_cpy = get_compact_vector(x, *n, *incx); + Scalar *x_cpy = get_compact_vector(x, *n, *incx); int code = UPLO(*uplo); - if(code>=2 || func[code]==0) - return 0; + if (code >= 2 || func[code] == 0) return 0; func[code](*n, ap, x_cpy, alpha); - if(x_cpy!=x) delete[] x_cpy; + if (x_cpy != x) delete[] x_cpy; return 1; } /** ZHPR2 performs the hermitian rank 2 operation - * - * A := alpha*x*conjg( y' ) + conjg( alpha )*y*conjg( x' ) + A, - * - * where alpha is a scalar, x and y are n element vectors and A is an - * n by n hermitian matrix, supplied in packed form. - */ -int EIGEN_BLAS_FUNC(hpr2)(char *uplo, int *n, RealScalar *palpha, RealScalar *px, int *incx, RealScalar *py, int *incy, RealScalar *pap) + * + * A := alpha*x*conjg( y' ) + conjg( alpha )*y*conjg( x' ) + A, + * + * where alpha is a scalar, x and y are n element vectors and A is an + * n by n hermitian matrix, supplied in packed form. + */ +int EIGEN_BLAS_FUNC( + hpr2)(char *uplo, int *n, RealScalar *palpha, RealScalar *px, int *incx, RealScalar *py, int *incy, RealScalar *pap) { - typedef void (*functype)(int, Scalar*, const Scalar*, const Scalar*, Scalar); + typedef void (*functype)(int, Scalar *, const Scalar *, const Scalar *, Scalar); static const functype func[2] = { // array index: UP - (internal::packed_rank2_update_selector::run), + (internal::packed_rank2_update_selector::run), // array index: LO - (internal::packed_rank2_update_selector::run), + (internal::packed_rank2_update_selector::run), }; - Scalar* x = reinterpret_cast(px); - Scalar* y = reinterpret_cast(py); - Scalar* ap = reinterpret_cast(pap); - Scalar alpha = *reinterpret_cast(palpha); + Scalar *x = reinterpret_cast(px); + Scalar *y = reinterpret_cast(py); + Scalar *ap = reinterpret_cast(pap); + Scalar alpha = *reinterpret_cast(palpha); int info = 0; - if(UPLO(*uplo)==INVALID) info = 1; - else if(*n<0) info = 2; - else if(*incx==0) info = 5; - else if(*incy==0) info = 7; - if(info) - return xerbla_(SCALAR_SUFFIX_UP"HPR2 ",&info,6); + if (UPLO(*uplo) == INVALID) + info = 1; + else if (*n < 0) + info = 2; + else if (*incx == 0) + info = 5; + else if (*incy == 0) + info = 7; + if (info) return xerbla_(SCALAR_SUFFIX_UP "HPR2 ", &info, 6); - if(alpha==Scalar(0)) - return 1; + if (alpha == Scalar(0)) return 1; - Scalar* x_cpy = get_compact_vector(x, *n, *incx); - Scalar* y_cpy = get_compact_vector(y, *n, *incy); + Scalar *x_cpy = get_compact_vector(x, *n, *incx); + Scalar *y_cpy = get_compact_vector(y, *n, *incy); int code = UPLO(*uplo); - if(code>=2 || func[code]==0) - return 0; + if (code >= 2 || func[code] == 0) return 0; func[code](*n, ap, x_cpy, y_cpy, alpha); - if(x_cpy!=x) delete[] x_cpy; - if(y_cpy!=y) delete[] y_cpy; + if (x_cpy != x) delete[] x_cpy; + if (y_cpy != y) delete[] y_cpy; return 1; } /** ZHER performs the hermitian rank 1 operation - * - * A := alpha*x*conjg( x' ) + A, - * - * where alpha is a real scalar, x is an n element vector and A is an - * n by n hermitian matrix. - */ + * + * A := alpha*x*conjg( x' ) + A, + * + * where alpha is a real scalar, x is an n element vector and A is an + * n by n hermitian matrix. + */ int EIGEN_BLAS_FUNC(her)(char *uplo, int *n, RealScalar *palpha, RealScalar *px, int *incx, RealScalar *pa, int *lda) { - typedef void (*functype)(int, Scalar*, int, const Scalar*, const Scalar*, const Scalar&); + typedef void (*functype)(int, Scalar *, int, const Scalar *, const Scalar *, const Scalar &); static const functype func[2] = { // array index: UP - (selfadjoint_rank1_update::run), + (selfadjoint_rank1_update::run), // array index: LO - (selfadjoint_rank1_update::run), + (selfadjoint_rank1_update::run), }; - Scalar* x = reinterpret_cast(px); - Scalar* a = reinterpret_cast(pa); - RealScalar alpha = *reinterpret_cast(palpha); + Scalar *x = reinterpret_cast(px); + Scalar *a = reinterpret_cast(pa); + RealScalar alpha = *reinterpret_cast(palpha); int info = 0; - if(UPLO(*uplo)==INVALID) info = 1; - else if(*n<0) info = 2; - else if(*incx==0) info = 5; - else if(*lda=2 || func[code]==0) - return 0; + if (code >= 2 || func[code] == 0) return 0; func[code](*n, a, *lda, x_cpy, x_cpy, alpha); - matrix(a,*n,*n,*lda).diagonal().imag().setZero(); + matrix(a, *n, *n, *lda).diagonal().imag().setZero(); - if(x_cpy!=x) delete[] x_cpy; + if (x_cpy != x) delete[] x_cpy; return 1; } /** ZHER2 performs the hermitian rank 2 operation - * - * A := alpha*x*conjg( y' ) + conjg( alpha )*y*conjg( x' ) + A, - * - * where alpha is a scalar, x and y are n element vectors and A is an n - * by n hermitian matrix. - */ -int EIGEN_BLAS_FUNC(her2)(char *uplo, int *n, RealScalar *palpha, RealScalar *px, int *incx, RealScalar *py, int *incy, RealScalar *pa, int *lda) + * + * A := alpha*x*conjg( y' ) + conjg( alpha )*y*conjg( x' ) + A, + * + * where alpha is a scalar, x and y are n element vectors and A is an n + * by n hermitian matrix. + */ +int EIGEN_BLAS_FUNC(her2)(char *uplo, + int *n, + RealScalar *palpha, + RealScalar *px, + int *incx, + RealScalar *py, + int *incy, + RealScalar *pa, + int *lda) { - typedef void (*functype)(int, Scalar*, int, const Scalar*, const Scalar*, Scalar); + typedef void (*functype)(int, Scalar *, int, const Scalar *, const Scalar *, Scalar); static const functype func[2] = { // array index: UP - (internal::rank2_update_selector::run), + (internal::rank2_update_selector::run), // array index: LO - (internal::rank2_update_selector::run), + (internal::rank2_update_selector::run), }; - Scalar* x = reinterpret_cast(px); - Scalar* y = reinterpret_cast(py); - Scalar* a = reinterpret_cast(pa); - Scalar alpha = *reinterpret_cast(palpha); + Scalar *x = reinterpret_cast(px); + Scalar *y = reinterpret_cast(py); + Scalar *a = reinterpret_cast(pa); + Scalar alpha = *reinterpret_cast(palpha); int info = 0; - if(UPLO(*uplo)==INVALID) info = 1; - else if(*n<0) info = 2; - else if(*incx==0) info = 5; - else if(*incy==0) info = 7; - else if(*lda=2 || func[code]==0) - return 0; + if (code >= 2 || func[code] == 0) return 0; func[code](*n, a, *lda, x_cpy, y_cpy, alpha); - matrix(a,*n,*n,*lda).diagonal().imag().setZero(); + matrix(a, *n, *n, *lda).diagonal().imag().setZero(); - if(x_cpy!=x) delete[] x_cpy; - if(y_cpy!=y) delete[] y_cpy; + if (x_cpy != x) delete[] x_cpy; + if (y_cpy != y) delete[] y_cpy; return 1; } /** ZGERU performs the rank 1 operation - * - * A := alpha*x*y' + A, - * - * where alpha is a scalar, x is an m element vector, y is an n element - * vector and A is an m by n matrix. - */ -int EIGEN_BLAS_FUNC(geru)(int *m, int *n, RealScalar *palpha, RealScalar *px, int *incx, RealScalar *py, int *incy, RealScalar *pa, int *lda) + * + * A := alpha*x*y' + A, + * + * where alpha is a scalar, x is an m element vector, y is an n element + * vector and A is an m by n matrix. + */ +int EIGEN_BLAS_FUNC(geru)(int *m, + int *n, + RealScalar *palpha, + RealScalar *px, + int *incx, + RealScalar *py, + int *incy, + RealScalar *pa, + int *lda) { - Scalar* x = reinterpret_cast(px); - Scalar* y = reinterpret_cast(py); - Scalar* a = reinterpret_cast(pa); - Scalar alpha = *reinterpret_cast(palpha); + Scalar *x = reinterpret_cast(px); + Scalar *y = reinterpret_cast(py); + Scalar *a = reinterpret_cast(pa); + Scalar alpha = *reinterpret_cast(palpha); int info = 0; - if(*m<0) info = 1; - else if(*n<0) info = 2; - else if(*incx==0) info = 5; - else if(*incy==0) info = 7; - else if(*lda::run(*m, *n, a, *lda, x_cpy, y_cpy, alpha); + internal::general_rank1_update::run(*m, *n, a, *lda, x_cpy, y_cpy, alpha); - if(x_cpy!=x) delete[] x_cpy; - if(y_cpy!=y) delete[] y_cpy; + if (x_cpy != x) delete[] x_cpy; + if (y_cpy != y) delete[] y_cpy; return 1; } /** ZGERC performs the rank 1 operation - * - * A := alpha*x*conjg( y' ) + A, - * - * where alpha is a scalar, x is an m element vector, y is an n element - * vector and A is an m by n matrix. - */ -int EIGEN_BLAS_FUNC(gerc)(int *m, int *n, RealScalar *palpha, RealScalar *px, int *incx, RealScalar *py, int *incy, RealScalar *pa, int *lda) + * + * A := alpha*x*conjg( y' ) + A, + * + * where alpha is a scalar, x is an m element vector, y is an n element + * vector and A is an m by n matrix. + */ +int EIGEN_BLAS_FUNC(gerc)(int *m, + int *n, + RealScalar *palpha, + RealScalar *px, + int *incx, + RealScalar *py, + int *incy, + RealScalar *pa, + int *lda) { - Scalar* x = reinterpret_cast(px); - Scalar* y = reinterpret_cast(py); - Scalar* a = reinterpret_cast(pa); - Scalar alpha = *reinterpret_cast(palpha); + Scalar *x = reinterpret_cast(px); + Scalar *y = reinterpret_cast(py); + Scalar *a = reinterpret_cast(pa); + Scalar alpha = *reinterpret_cast(palpha); int info = 0; - if(*m<0) info = 1; - else if(*n<0) info = 2; - else if(*incx==0) info = 5; - else if(*incy==0) info = 7; - else if(*lda::run(*m, *n, a, *lda, x_cpy, y_cpy, alpha); - - if(x_cpy!=x) delete[] x_cpy; - if(y_cpy!=y) delete[] y_cpy; + if (*m < 0) + info = 1; + else if (*n < 0) + info = 2; + else if (*incx == 0) + info = 5; + else if (*incy == 0) + info = 7; + else if (*lda < std::max(1, *m)) + info = 9; + if (info) return xerbla_(SCALAR_SUFFIX_UP "GERC ", &info, 6); + + if (alpha == Scalar(0)) return 1; + + Scalar *x_cpy = get_compact_vector(x, *m, *incx); + Scalar *y_cpy = get_compact_vector(y, *n, *incy); + + internal::general_rank1_update::run(*m, *n, a, *lda, x_cpy, y_cpy, alpha); + + if (x_cpy != x) delete[] x_cpy; + if (y_cpy != y) delete[] y_cpy; return 1; } diff --git a/filmulator-gui/core/nlmeans/eigen/blas/level2_impl.h b/filmulator-gui/core/nlmeans/eigen/blas/level2_impl.h index 173f40b4..bb78a708 100644 --- a/filmulator-gui/core/nlmeans/eigen/blas/level2_impl.h +++ b/filmulator-gui/core/nlmeans/eigen/blas/level2_impl.h @@ -12,267 +12,328 @@ template struct general_matrix_vector_product_wrapper { - static void run(Index rows, Index cols,const Scalar *lhs, Index lhsStride, const Scalar *rhs, Index rhsIncr, Scalar* res, Index resIncr, Scalar alpha) + static void run(Index rows, + Index cols, + const Scalar *lhs, + Index lhsStride, + const Scalar *rhs, + Index rhsIncr, + Scalar *res, + Index resIncr, + Scalar alpha) { - typedef internal::const_blas_data_mapper LhsMapper; - typedef internal::const_blas_data_mapper RhsMapper; - - internal::general_matrix_vector_product - ::run( - rows, cols, LhsMapper(lhs, lhsStride), RhsMapper(rhs, rhsIncr), res, resIncr, alpha); + typedef internal::const_blas_data_mapper LhsMapper; + typedef internal::const_blas_data_mapper RhsMapper; + + internal::general_matrix_vector_product::run(rows, cols, LhsMapper(lhs, lhsStride), RhsMapper(rhs, rhsIncr), res, resIncr, alpha); } }; -int EIGEN_BLAS_FUNC(gemv)(const char *opa, const int *m, const int *n, const RealScalar *palpha, - const RealScalar *pa, const int *lda, const RealScalar *pb, const int *incb, const RealScalar *pbeta, RealScalar *pc, const int *incc) +int EIGEN_BLAS_FUNC(gemv)(const char *opa, + const int *m, + const int *n, + const RealScalar *palpha, + const RealScalar *pa, + const int *lda, + const RealScalar *pb, + const int *incb, + const RealScalar *pbeta, + RealScalar *pc, + const int *incc) { - typedef void (*functype)(int, int, const Scalar *, int, const Scalar *, int , Scalar *, int, Scalar); - static const functype func[4] = { - // array index: NOTR - (general_matrix_vector_product_wrapper::run), - // array index: TR - (general_matrix_vector_product_wrapper::run), - // array index: ADJ - (general_matrix_vector_product_wrapper::run), + typedef void (*functype)(int, int, const Scalar *, int, const Scalar *, int, Scalar *, int, Scalar); + static const functype func[4] = { // array index: NOTR + (general_matrix_vector_product_wrapper::run), + // array index: TR + (general_matrix_vector_product_wrapper::run), + // array index: ADJ + (general_matrix_vector_product_wrapper::run), 0 }; - const Scalar* a = reinterpret_cast(pa); - const Scalar* b = reinterpret_cast(pb); - Scalar* c = reinterpret_cast(pc); - Scalar alpha = *reinterpret_cast(palpha); - Scalar beta = *reinterpret_cast(pbeta); + const Scalar *a = reinterpret_cast(pa); + const Scalar *b = reinterpret_cast(pb); + Scalar *c = reinterpret_cast(pc); + Scalar alpha = *reinterpret_cast(palpha); + Scalar beta = *reinterpret_cast(pbeta); // check arguments int info = 0; - if(OP(*opa)==INVALID) info = 1; - else if(*m<0) info = 2; - else if(*n<0) info = 3; - else if(*lda=4 || func[code]==0) - return 0; + if (code >= 4 || func[code] == 0) return 0; func[code](actual_m, actual_n, a, *lda, actual_b, 1, actual_c, 1, alpha); - if(actual_b!=b) delete[] actual_b; - if(actual_c!=c) delete[] copy_back(actual_c,c,actual_m,*incc); + if (actual_b != b) delete[] actual_b; + if (actual_c != c) delete[] copy_back(actual_c, c, actual_m, *incc); return 1; } -int EIGEN_BLAS_FUNC(trsv)(const char *uplo, const char *opa, const char *diag, const int *n, const RealScalar *pa, const int *lda, RealScalar *pb, const int *incb) +int EIGEN_BLAS_FUNC(trsv)(const char *uplo, + const char *opa, + const char *diag, + const int *n, + const RealScalar *pa, + const int *lda, + RealScalar *pb, + const int *incb) { typedef void (*functype)(int, const Scalar *, int, Scalar *); - static const functype func[16] = { - // array index: NOTR | (UP << 2) | (NUNIT << 3) - (internal::triangular_solve_vector::run), + static const functype func[16] = { // array index: NOTR | (UP << 2) | (NUNIT << 3) + (internal::triangular_solve_vector::run), // array index: TR | (UP << 2) | (NUNIT << 3) - (internal::triangular_solve_vector::run), + (internal::triangular_solve_vector::run), // array index: ADJ | (UP << 2) | (NUNIT << 3) - (internal::triangular_solve_vector::run), + (internal::triangular_solve_vector::run), 0, // array index: NOTR | (LO << 2) | (NUNIT << 3) - (internal::triangular_solve_vector::run), + (internal::triangular_solve_vector::run), // array index: TR | (LO << 2) | (NUNIT << 3) - (internal::triangular_solve_vector::run), + (internal::triangular_solve_vector::run), // array index: ADJ | (LO << 2) | (NUNIT << 3) - (internal::triangular_solve_vector::run), + (internal::triangular_solve_vector::run), 0, // array index: NOTR | (UP << 2) | (UNIT << 3) - (internal::triangular_solve_vector::run), + (internal::triangular_solve_vector::run), // array index: TR | (UP << 2) | (UNIT << 3) - (internal::triangular_solve_vector::run), + (internal::triangular_solve_vector::run), // array index: ADJ | (UP << 2) | (UNIT << 3) - (internal::triangular_solve_vector::run), + (internal::triangular_solve_vector::run), 0, // array index: NOTR | (LO << 2) | (UNIT << 3) - (internal::triangular_solve_vector::run), + (internal::triangular_solve_vector::run), // array index: TR | (LO << 2) | (UNIT << 3) - (internal::triangular_solve_vector::run), + (internal::triangular_solve_vector::run), // array index: ADJ | (LO << 2) | (UNIT << 3) - (internal::triangular_solve_vector::run), + (internal::triangular_solve_vector::run), 0 }; - const Scalar* a = reinterpret_cast(pa); - Scalar* b = reinterpret_cast(pb); + const Scalar *a = reinterpret_cast(pa); + Scalar *b = reinterpret_cast(pb); int info = 0; - if(UPLO(*uplo)==INVALID) info = 1; - else if(OP(*opa)==INVALID) info = 2; - else if(DIAG(*diag)==INVALID) info = 3; - else if(*n<0) info = 4; - else if(*lda::run), + typedef void (*functype)(int, int, const Scalar *, int, const Scalar *, int, Scalar *, int, const Scalar &); + static const functype func[16] = { // array index: NOTR | (UP << 2) | (NUNIT << 3) + (internal::triangular_matrix_vector_product::run), // array index: TR | (UP << 2) | (NUNIT << 3) - (internal::triangular_matrix_vector_product::run), + (internal::triangular_matrix_vector_product::run), // array index: ADJ | (UP << 2) | (NUNIT << 3) - (internal::triangular_matrix_vector_product::run), + (internal::triangular_matrix_vector_product::run), 0, // array index: NOTR | (LO << 2) | (NUNIT << 3) - (internal::triangular_matrix_vector_product::run), + (internal::triangular_matrix_vector_product::run), // array index: TR | (LO << 2) | (NUNIT << 3) - (internal::triangular_matrix_vector_product::run), + (internal::triangular_matrix_vector_product::run), // array index: ADJ | (LO << 2) | (NUNIT << 3) - (internal::triangular_matrix_vector_product::run), + (internal::triangular_matrix_vector_product::run), 0, // array index: NOTR | (UP << 2) | (UNIT << 3) - (internal::triangular_matrix_vector_product::run), + (internal::triangular_matrix_vector_product::run), // array index: TR | (UP << 2) | (UNIT << 3) - (internal::triangular_matrix_vector_product::run), + (internal::triangular_matrix_vector_product::run), // array index: ADJ | (UP << 2) | (UNIT << 3) - (internal::triangular_matrix_vector_product::run), + (internal::triangular_matrix_vector_product::run), 0, // array index: NOTR | (LO << 2) | (UNIT << 3) - (internal::triangular_matrix_vector_product::run), + (internal::triangular_matrix_vector_product::run), // array index: TR | (LO << 2) | (UNIT << 3) - (internal::triangular_matrix_vector_product::run), + (internal::triangular_matrix_vector_product::run), // array index: ADJ | (LO << 2) | (UNIT << 3) - (internal::triangular_matrix_vector_product::run), + (internal::triangular_matrix_vector_product::run), 0 }; - const Scalar* a = reinterpret_cast(pa); - Scalar* b = reinterpret_cast(pb); + const Scalar *a = reinterpret_cast(pa); + Scalar *b = reinterpret_cast(pb); int info = 0; - if(UPLO(*uplo)==INVALID) info = 1; - else if(OP(*opa)==INVALID) info = 2; - else if(DIAG(*diag)==INVALID) info = 3; - else if(*n<0) info = 4; - else if(*lda res(*n); + if (UPLO(*uplo) == INVALID) + info = 1; + else if (OP(*opa) == INVALID) + info = 2; + else if (DIAG(*diag) == INVALID) + info = 3; + else if (*n < 0) + info = 4; + else if (*lda < std::max(1, *n)) + info = 6; + else if (*incb == 0) + info = 8; + if (info) return xerbla_(SCALAR_SUFFIX_UP "TRMV ", &info, 6); + + if (*n == 0) return 1; + + Scalar *actual_b = get_compact_vector(b, *n, *incb); + Matrix res(*n); res.setZero(); int code = OP(*opa) | (UPLO(*uplo) << 2) | (DIAG(*diag) << 3); - if(code>=16 || func[code]==0) - return 0; + if (code >= 16 || func[code] == 0) return 0; func[code](*n, *n, a, *lda, actual_b, 1, res.data(), 1, Scalar(1)); - copy_back(res.data(),b,*n,*incb); - if(actual_b!=b) delete[] actual_b; + copy_back(res.data(), b, *n, *incb); + if (actual_b != b) delete[] actual_b; return 1; } /** GBMV performs one of the matrix-vector operations - * - * y := alpha*A*x + beta*y, or y := alpha*A'*x + beta*y, - * - * where alpha and beta are scalars, x and y are vectors and A is an - * m by n band matrix, with kl sub-diagonals and ku super-diagonals. - */ -int EIGEN_BLAS_FUNC(gbmv)(char *trans, int *m, int *n, int *kl, int *ku, RealScalar *palpha, RealScalar *pa, int *lda, - RealScalar *px, int *incx, RealScalar *pbeta, RealScalar *py, int *incy) + * + * y := alpha*A*x + beta*y, or y := alpha*A'*x + beta*y, + * + * where alpha and beta are scalars, x and y are vectors and A is an + * m by n band matrix, with kl sub-diagonals and ku super-diagonals. + */ +int EIGEN_BLAS_FUNC(gbmv)(char *trans, + int *m, + int *n, + int *kl, + int *ku, + RealScalar *palpha, + RealScalar *pa, + int *lda, + RealScalar *px, + int *incx, + RealScalar *pbeta, + RealScalar *py, + int *incy) { - const Scalar* a = reinterpret_cast(pa); - const Scalar* x = reinterpret_cast(px); - Scalar* y = reinterpret_cast(py); - Scalar alpha = *reinterpret_cast(palpha); - Scalar beta = *reinterpret_cast(pbeta); - int coeff_rows = *kl+*ku+1; + const Scalar *a = reinterpret_cast(pa); + const Scalar *x = reinterpret_cast(px); + Scalar *y = reinterpret_cast(py); + Scalar alpha = *reinterpret_cast(palpha); + Scalar beta = *reinterpret_cast(pbeta); + int coeff_rows = *kl + *ku + 1; int info = 0; - if(OP(*trans)==INVALID) info = 1; - else if(*m<0) info = 2; - else if(*n<0) info = 3; - else if(*kl<0) info = 4; - else if(*ku<0) info = 5; - else if(*lda::run), + (internal::band_solve_triangular_selector::run), // array index: TR | (UP << 2) | (NUNIT << 3) - (internal::band_solve_triangular_selector::run), + (internal::band_solve_triangular_selector::run), // array index: ADJ | (UP << 2) | (NUNIT << 3) - (internal::band_solve_triangular_selector::run), + (internal::band_solve_triangular_selector::run), 0, // array index: NOTR | (LO << 2) | (NUNIT << 3) - (internal::band_solve_triangular_selector::run), + (internal::band_solve_triangular_selector::run), // array index: TR | (LO << 2) | (NUNIT << 3) - (internal::band_solve_triangular_selector::run), + (internal::band_solve_triangular_selector::run), // array index: ADJ | (LO << 2) | (NUNIT << 3) - (internal::band_solve_triangular_selector::run), + (internal::band_solve_triangular_selector::run), 0, // array index: NOTR | (UP << 2) | (UNIT << 3) - (internal::band_solve_triangular_selector::run), + (internal::band_solve_triangular_selector::run), // array index: TR | (UP << 2) | (UNIT << 3) - (internal::band_solve_triangular_selector::run), + (internal::band_solve_triangular_selector::run), // array index: ADJ | (UP << 2) | (UNIT << 3) - (internal::band_solve_triangular_selector::run), + (internal::band_solve_triangular_selector::run), 0, // array index: NOTR | (LO << 2) | (UNIT << 3) - (internal::band_solve_triangular_selector::run), + (internal::band_solve_triangular_selector::run), // array index: TR | (LO << 2) | (UNIT << 3) - (internal::band_solve_triangular_selector::run), + (internal::band_solve_triangular_selector::run), // array index: ADJ | (LO << 2) | (UNIT << 3) - (internal::band_solve_triangular_selector::run), + (internal::band_solve_triangular_selector::run), 0, }; - Scalar* a = reinterpret_cast(pa); - Scalar* x = reinterpret_cast(px); - int coeff_rows = *k+1; + Scalar *a = reinterpret_cast(pa); + Scalar *x = reinterpret_cast(px); + int coeff_rows = *k + 1; int info = 0; - if(UPLO(*uplo)==INVALID) info = 1; - else if(OP(*op)==INVALID) info = 2; - else if(DIAG(*diag)==INVALID) info = 3; - else if(*n<0) info = 4; - else if(*k<0) info = 5; - else if(*lda=16 || func[code]==0) - return 0; + if (code >= 16 || func[code] == 0) return 0; func[code](*n, *k, a, *lda, actual_x); - if(actual_x!=x) delete[] copy_back(actual_x,x,actual_n,*incx); + if (actual_x != x) delete[] copy_back(actual_x, x, actual_n, *incx); return 0; } /** DTPMV performs one of the matrix-vector operations - * - * x := A*x, or x := A'*x, - * - * where x is an n element vector and A is an n by n unit, or non-unit, - * upper or lower triangular matrix, supplied in packed form. - */ + * + * x := A*x, or x := A'*x, + * + * where x is an n element vector and A is an n by n unit, or non-unit, + * upper or lower triangular matrix, supplied in packed form. + */ int EIGEN_BLAS_FUNC(tpmv)(char *uplo, char *opa, char *diag, int *n, RealScalar *pap, RealScalar *px, int *incx) { - typedef void (*functype)(int, const Scalar*, const Scalar*, Scalar*, Scalar); - static const functype func[16] = { - // array index: NOTR | (UP << 2) | (NUNIT << 3) - (internal::packed_triangular_matrix_vector_product::run), + typedef void (*functype)(int, const Scalar *, const Scalar *, Scalar *, Scalar); + static const functype func[16] = { // array index: NOTR | (UP << 2) | (NUNIT << 3) + (internal::packed_triangular_matrix_vector_product::run), // array index: TR | (UP << 2) | (NUNIT << 3) - (internal::packed_triangular_matrix_vector_product::run), + (internal::packed_triangular_matrix_vector_product::run), // array index: ADJ | (UP << 2) | (NUNIT << 3) - (internal::packed_triangular_matrix_vector_product::run), + (internal::packed_triangular_matrix_vector_product::run), 0, // array index: NOTR | (LO << 2) | (NUNIT << 3) - (internal::packed_triangular_matrix_vector_product::run), + (internal::packed_triangular_matrix_vector_product::run), // array index: TR | (LO << 2) | (NUNIT << 3) - (internal::packed_triangular_matrix_vector_product::run), + (internal::packed_triangular_matrix_vector_product::run), // array index: ADJ | (LO << 2) | (NUNIT << 3) - (internal::packed_triangular_matrix_vector_product::run), + (internal::packed_triangular_matrix_vector_product::run), 0, // array index: NOTR | (UP << 2) | (UNIT << 3) - (internal::packed_triangular_matrix_vector_product::run), + (internal::packed_triangular_matrix_vector_product:: + run), // array index: TR | (UP << 2) | (UNIT << 3) - (internal::packed_triangular_matrix_vector_product::run), + (internal::packed_triangular_matrix_vector_product:: + run), // array index: ADJ | (UP << 2) | (UNIT << 3) - (internal::packed_triangular_matrix_vector_product::run), + (internal::packed_triangular_matrix_vector_product:: + run), 0, // array index: NOTR | (LO << 2) | (UNIT << 3) - (internal::packed_triangular_matrix_vector_product::run), + (internal::packed_triangular_matrix_vector_product:: + run), // array index: TR | (LO << 2) | (UNIT << 3) - (internal::packed_triangular_matrix_vector_product::run), + (internal::packed_triangular_matrix_vector_product:: + run), // array index: ADJ | (LO << 2) | (UNIT << 3) - (internal::packed_triangular_matrix_vector_product::run), + (internal::packed_triangular_matrix_vector_product:: + run), 0 }; - Scalar* ap = reinterpret_cast(pap); - Scalar* x = reinterpret_cast(px); + Scalar *ap = reinterpret_cast(pap); + Scalar *x = reinterpret_cast(px); int info = 0; - if(UPLO(*uplo)==INVALID) info = 1; - else if(OP(*opa)==INVALID) info = 2; - else if(DIAG(*diag)==INVALID) info = 3; - else if(*n<0) info = 4; - else if(*incx==0) info = 7; - if(info) - return xerbla_(SCALAR_SUFFIX_UP"TPMV ",&info,6); - - if(*n==0) - return 1; - - Scalar* actual_x = get_compact_vector(x,*n,*incx); - Matrix res(*n); + if (UPLO(*uplo) == INVALID) + info = 1; + else if (OP(*opa) == INVALID) + info = 2; + else if (DIAG(*diag) == INVALID) + info = 3; + else if (*n < 0) + info = 4; + else if (*incx == 0) + info = 7; + if (info) return xerbla_(SCALAR_SUFFIX_UP "TPMV ", &info, 6); + + if (*n == 0) return 1; + + Scalar *actual_x = get_compact_vector(x, *n, *incx); + Matrix res(*n); res.setZero(); int code = OP(*opa) | (UPLO(*uplo) << 2) | (DIAG(*diag) << 3); - if(code>=16 || func[code]==0) - return 0; + if (code >= 16 || func[code] == 0) return 0; func[code](*n, ap, actual_x, res.data(), Scalar(1)); - copy_back(res.data(),x,*n,*incx); - if(actual_x!=x) delete[] actual_x; + copy_back(res.data(), x, *n, *incx); + if (actual_x != x) delete[] actual_x; return 1; } /** DTPSV solves one of the systems of equations - * - * A*x = b, or A'*x = b, - * - * where b and x are n element vectors and A is an n by n unit, or - * non-unit, upper or lower triangular matrix, supplied in packed form. - * - * No test for singularity or near-singularity is included in this - * routine. Such tests must be performed before calling this routine. - */ + * + * A*x = b, or A'*x = b, + * + * where b and x are n element vectors and A is an n by n unit, or + * non-unit, upper or lower triangular matrix, supplied in packed form. + * + * No test for singularity or near-singularity is included in this + * routine. Such tests must be performed before calling this routine. + */ int EIGEN_BLAS_FUNC(tpsv)(char *uplo, char *opa, char *diag, int *n, RealScalar *pap, RealScalar *px, int *incx) { - typedef void (*functype)(int, const Scalar*, Scalar*); - static const functype func[16] = { - // array index: NOTR | (UP << 2) | (NUNIT << 3) - (internal::packed_triangular_solve_vector::run), + typedef void (*functype)(int, const Scalar *, Scalar *); + static const functype func[16] = { // array index: NOTR | (UP << 2) | (NUNIT << 3) + (internal::packed_triangular_solve_vector::run), // array index: TR | (UP << 2) | (NUNIT << 3) - (internal::packed_triangular_solve_vector::run), + (internal::packed_triangular_solve_vector::run), // array index: ADJ | (UP << 2) | (NUNIT << 3) - (internal::packed_triangular_solve_vector::run), + (internal::packed_triangular_solve_vector::run), 0, // array index: NOTR | (LO << 2) | (NUNIT << 3) - (internal::packed_triangular_solve_vector::run), + (internal::packed_triangular_solve_vector::run), // array index: TR | (LO << 2) | (NUNIT << 3) - (internal::packed_triangular_solve_vector::run), + (internal::packed_triangular_solve_vector::run), // array index: ADJ | (LO << 2) | (NUNIT << 3) - (internal::packed_triangular_solve_vector::run), + (internal::packed_triangular_solve_vector::run), 0, // array index: NOTR | (UP << 2) | (UNIT << 3) - (internal::packed_triangular_solve_vector::run), + (internal::packed_triangular_solve_vector::run), // array index: TR | (UP << 2) | (UNIT << 3) - (internal::packed_triangular_solve_vector::run), + (internal::packed_triangular_solve_vector::run), // array index: ADJ | (UP << 2) | (UNIT << 3) - (internal::packed_triangular_solve_vector::run), + (internal::packed_triangular_solve_vector::run), 0, // array index: NOTR | (LO << 2) | (UNIT << 3) - (internal::packed_triangular_solve_vector::run), + (internal::packed_triangular_solve_vector::run), // array index: TR | (LO << 2) | (UNIT << 3) - (internal::packed_triangular_solve_vector::run), + (internal::packed_triangular_solve_vector::run), // array index: ADJ | (LO << 2) | (UNIT << 3) - (internal::packed_triangular_solve_vector::run), + (internal::packed_triangular_solve_vector::run), 0 }; - Scalar* ap = reinterpret_cast(pap); - Scalar* x = reinterpret_cast(px); + Scalar *ap = reinterpret_cast(pap); + Scalar *x = reinterpret_cast(px); int info = 0; - if(UPLO(*uplo)==INVALID) info = 1; - else if(OP(*opa)==INVALID) info = 2; - else if(DIAG(*diag)==INVALID) info = 3; - else if(*n<0) info = 4; - else if(*incx==0) info = 7; - if(info) - return xerbla_(SCALAR_SUFFIX_UP"TPSV ",&info,6); - - Scalar* actual_x = get_compact_vector(x,*n,*incx); + if (UPLO(*uplo) == INVALID) + info = 1; + else if (OP(*opa) == INVALID) + info = 2; + else if (DIAG(*diag) == INVALID) + info = 3; + else if (*n < 0) + info = 4; + else if (*incx == 0) + info = 7; + if (info) return xerbla_(SCALAR_SUFFIX_UP "TPSV ", &info, 6); + + Scalar *actual_x = get_compact_vector(x, *n, *incx); int code = OP(*opa) | (UPLO(*uplo) << 2) | (DIAG(*diag) << 3); func[code](*n, ap, actual_x); - if(actual_x!=x) delete[] copy_back(actual_x,x,*n,*incx); + if (actual_x != x) delete[] copy_back(actual_x, x, *n, *incx); return 1; } diff --git a/filmulator-gui/core/nlmeans/eigen/blas/level2_real_impl.h b/filmulator-gui/core/nlmeans/eigen/blas/level2_real_impl.h index 7620f0a3..9a01287f 100644 --- a/filmulator-gui/core/nlmeans/eigen/blas/level2_real_impl.h +++ b/filmulator-gui/core/nlmeans/eigen/blas/level2_real_impl.h @@ -10,152 +10,181 @@ #include "common.h" // y = alpha*A*x + beta*y -int EIGEN_BLAS_FUNC(symv) (const char *uplo, const int *n, const RealScalar *palpha, const RealScalar *pa, const int *lda, - const RealScalar *px, const int *incx, const RealScalar *pbeta, RealScalar *py, const int *incy) +int EIGEN_BLAS_FUNC(symv)(const char *uplo, + const int *n, + const RealScalar *palpha, + const RealScalar *pa, + const int *lda, + const RealScalar *px, + const int *incx, + const RealScalar *pbeta, + RealScalar *py, + const int *incy) { - typedef void (*functype)(int, const Scalar*, int, const Scalar*, Scalar*, Scalar); + typedef void (*functype)(int, const Scalar *, int, const Scalar *, Scalar *, Scalar); static const functype func[2] = { // array index: UP - (internal::selfadjoint_matrix_vector_product::run), + (internal::selfadjoint_matrix_vector_product::run), // array index: LO - (internal::selfadjoint_matrix_vector_product::run), + (internal::selfadjoint_matrix_vector_product::run), }; - const Scalar* a = reinterpret_cast(pa); - const Scalar* x = reinterpret_cast(px); - Scalar* y = reinterpret_cast(py); - Scalar alpha = *reinterpret_cast(palpha); - Scalar beta = *reinterpret_cast(pbeta); + const Scalar *a = reinterpret_cast(pa); + const Scalar *x = reinterpret_cast(px); + Scalar *y = reinterpret_cast(py); + Scalar alpha = *reinterpret_cast(palpha); + Scalar beta = *reinterpret_cast(pbeta); // check arguments int info = 0; - if(UPLO(*uplo)==INVALID) info = 1; - else if(*n<0) info = 2; - else if(*lda=2 || func[code]==0) - return 0; + if (code >= 2 || func[code] == 0) return 0; func[code](*n, a, *lda, actual_x, actual_y, alpha); - if(actual_x!=x) delete[] actual_x; - if(actual_y!=y) delete[] copy_back(actual_y,y,*n,*incy); + if (actual_x != x) delete[] actual_x; + if (actual_y != y) delete[] copy_back(actual_y, y, *n, *incy); return 1; } // C := alpha*x*x' + C -int EIGEN_BLAS_FUNC(syr)(const char *uplo, const int *n, const RealScalar *palpha, const RealScalar *px, const int *incx, RealScalar *pc, const int *ldc) +int EIGEN_BLAS_FUNC(syr)(const char *uplo, + const int *n, + const RealScalar *palpha, + const RealScalar *px, + const int *incx, + RealScalar *pc, + const int *ldc) { - typedef void (*functype)(int, Scalar*, int, const Scalar*, const Scalar*, const Scalar&); + typedef void (*functype)(int, Scalar *, int, const Scalar *, const Scalar *, const Scalar &); static const functype func[2] = { // array index: UP - (selfadjoint_rank1_update::run), + (selfadjoint_rank1_update::run), // array index: LO - (selfadjoint_rank1_update::run), + (selfadjoint_rank1_update::run), }; - const Scalar* x = reinterpret_cast(px); - Scalar* c = reinterpret_cast(pc); - Scalar alpha = *reinterpret_cast(palpha); + const Scalar *x = reinterpret_cast(px); + Scalar *c = reinterpret_cast(pc); + Scalar alpha = *reinterpret_cast(palpha); int info = 0; - if(UPLO(*uplo)==INVALID) info = 1; - else if(*n<0) info = 2; - else if(*incx==0) info = 5; - else if(*ldc=2 || func[code]==0) - return 0; + if (code >= 2 || func[code] == 0) return 0; func[code](*n, c, *ldc, x_cpy, x_cpy, alpha); - if(x_cpy!=x) delete[] x_cpy; + if (x_cpy != x) delete[] x_cpy; return 1; } // C := alpha*x*y' + alpha*y*x' + C -int EIGEN_BLAS_FUNC(syr2)(const char *uplo, const int *n, const RealScalar *palpha, const RealScalar *px, const int *incx, const RealScalar *py, const int *incy, RealScalar *pc, const int *ldc) +int EIGEN_BLAS_FUNC(syr2)(const char *uplo, + const int *n, + const RealScalar *palpha, + const RealScalar *px, + const int *incx, + const RealScalar *py, + const int *incy, + RealScalar *pc, + const int *ldc) { - typedef void (*functype)(int, Scalar*, int, const Scalar*, const Scalar*, Scalar); + typedef void (*functype)(int, Scalar *, int, const Scalar *, const Scalar *, Scalar); static const functype func[2] = { // array index: UP - (internal::rank2_update_selector::run), + (internal::rank2_update_selector::run), // array index: LO - (internal::rank2_update_selector::run), + (internal::rank2_update_selector::run), }; - const Scalar* x = reinterpret_cast(px); - const Scalar* y = reinterpret_cast(py); - Scalar* c = reinterpret_cast(pc); - Scalar alpha = *reinterpret_cast(palpha); + const Scalar *x = reinterpret_cast(px); + const Scalar *y = reinterpret_cast(py); + Scalar *c = reinterpret_cast(pc); + Scalar alpha = *reinterpret_cast(palpha); int info = 0; - if(UPLO(*uplo)==INVALID) info = 1; - else if(*n<0) info = 2; - else if(*incx==0) info = 5; - else if(*incy==0) info = 7; - else if(*ldc=2 || func[code]==0) - return 0; + if (code >= 2 || func[code] == 0) return 0; func[code](*n, c, *ldc, x_cpy, y_cpy, alpha); - if(x_cpy!=x) delete[] x_cpy; - if(y_cpy!=y) delete[] y_cpy; + if (x_cpy != x) delete[] x_cpy; + if (y_cpy != y) delete[] y_cpy; -// int code = UPLO(*uplo); -// if(code>=2 || func[code]==0) -// return 0; + // int code = UPLO(*uplo); + // if(code>=2 || func[code]==0) + // return 0; -// func[code](*n, a, *inca, b, *incb, c, *ldc, alpha); + // func[code](*n, a, *inca, b, *incb, c, *ldc, alpha); return 1; } /** DSBMV performs the matrix-vector operation - * - * y := alpha*A*x + beta*y, - * - * where alpha and beta are scalars, x and y are n element vectors and - * A is an n by n symmetric band matrix, with k super-diagonals. - */ + * + * y := alpha*A*x + beta*y, + * + * where alpha and beta are scalars, x and y are n element vectors and + * A is an n by n symmetric band matrix, with k super-diagonals. + */ // int EIGEN_BLAS_FUNC(sbmv)( char *uplo, int *n, int *k, RealScalar *alpha, RealScalar *a, int *lda, // RealScalar *x, int *incx, RealScalar *beta, RealScalar *y, int *incy) // { @@ -164,143 +193,150 @@ int EIGEN_BLAS_FUNC(syr2)(const char *uplo, const int *n, const RealScalar *palp /** DSPMV performs the matrix-vector operation - * - * y := alpha*A*x + beta*y, - * - * where alpha and beta are scalars, x and y are n element vectors and - * A is an n by n symmetric matrix, supplied in packed form. - * - */ -// int EIGEN_BLAS_FUNC(spmv)(char *uplo, int *n, RealScalar *alpha, RealScalar *ap, RealScalar *x, int *incx, RealScalar *beta, RealScalar *y, int *incy) + * + * y := alpha*A*x + beta*y, + * + * where alpha and beta are scalars, x and y are n element vectors and + * A is an n by n symmetric matrix, supplied in packed form. + * + */ +// int EIGEN_BLAS_FUNC(spmv)(char *uplo, int *n, RealScalar *alpha, RealScalar *ap, RealScalar *x, int *incx, RealScalar +// *beta, RealScalar *y, int *incy) // { // return 1; // } /** DSPR performs the symmetric rank 1 operation - * - * A := alpha*x*x' + A, - * - * where alpha is a real scalar, x is an n element vector and A is an - * n by n symmetric matrix, supplied in packed form. - */ + * + * A := alpha*x*x' + A, + * + * where alpha is a real scalar, x is an n element vector and A is an + * n by n symmetric matrix, supplied in packed form. + */ int EIGEN_BLAS_FUNC(spr)(char *uplo, int *n, Scalar *palpha, Scalar *px, int *incx, Scalar *pap) { - typedef void (*functype)(int, Scalar*, const Scalar*, Scalar); + typedef void (*functype)(int, Scalar *, const Scalar *, Scalar); static const functype func[2] = { // array index: UP - (internal::selfadjoint_packed_rank1_update::run), + (internal::selfadjoint_packed_rank1_update::run), // array index: LO - (internal::selfadjoint_packed_rank1_update::run), + (internal::selfadjoint_packed_rank1_update::run), }; - Scalar* x = reinterpret_cast(px); - Scalar* ap = reinterpret_cast(pap); - Scalar alpha = *reinterpret_cast(palpha); + Scalar *x = reinterpret_cast(px); + Scalar *ap = reinterpret_cast(pap); + Scalar alpha = *reinterpret_cast(palpha); int info = 0; - if(UPLO(*uplo)==INVALID) info = 1; - else if(*n<0) info = 2; - else if(*incx==0) info = 5; - if(info) - return xerbla_(SCALAR_SUFFIX_UP"SPR ",&info,6); + if (UPLO(*uplo) == INVALID) + info = 1; + else if (*n < 0) + info = 2; + else if (*incx == 0) + info = 5; + if (info) return xerbla_(SCALAR_SUFFIX_UP "SPR ", &info, 6); - if(alpha==Scalar(0)) - return 1; + if (alpha == Scalar(0)) return 1; - Scalar* x_cpy = get_compact_vector(x, *n, *incx); + Scalar *x_cpy = get_compact_vector(x, *n, *incx); int code = UPLO(*uplo); - if(code>=2 || func[code]==0) - return 0; + if (code >= 2 || func[code] == 0) return 0; func[code](*n, ap, x_cpy, alpha); - if(x_cpy!=x) delete[] x_cpy; + if (x_cpy != x) delete[] x_cpy; return 1; } /** DSPR2 performs the symmetric rank 2 operation - * - * A := alpha*x*y' + alpha*y*x' + A, - * - * where alpha is a scalar, x and y are n element vectors and A is an - * n by n symmetric matrix, supplied in packed form. - */ -int EIGEN_BLAS_FUNC(spr2)(char *uplo, int *n, RealScalar *palpha, RealScalar *px, int *incx, RealScalar *py, int *incy, RealScalar *pap) + * + * A := alpha*x*y' + alpha*y*x' + A, + * + * where alpha is a scalar, x and y are n element vectors and A is an + * n by n symmetric matrix, supplied in packed form. + */ +int EIGEN_BLAS_FUNC( + spr2)(char *uplo, int *n, RealScalar *palpha, RealScalar *px, int *incx, RealScalar *py, int *incy, RealScalar *pap) { - typedef void (*functype)(int, Scalar*, const Scalar*, const Scalar*, Scalar); + typedef void (*functype)(int, Scalar *, const Scalar *, const Scalar *, Scalar); static const functype func[2] = { // array index: UP - (internal::packed_rank2_update_selector::run), + (internal::packed_rank2_update_selector::run), // array index: LO - (internal::packed_rank2_update_selector::run), + (internal::packed_rank2_update_selector::run), }; - Scalar* x = reinterpret_cast(px); - Scalar* y = reinterpret_cast(py); - Scalar* ap = reinterpret_cast(pap); - Scalar alpha = *reinterpret_cast(palpha); + Scalar *x = reinterpret_cast(px); + Scalar *y = reinterpret_cast(py); + Scalar *ap = reinterpret_cast(pap); + Scalar alpha = *reinterpret_cast(palpha); int info = 0; - if(UPLO(*uplo)==INVALID) info = 1; - else if(*n<0) info = 2; - else if(*incx==0) info = 5; - else if(*incy==0) info = 7; - if(info) - return xerbla_(SCALAR_SUFFIX_UP"SPR2 ",&info,6); + if (UPLO(*uplo) == INVALID) + info = 1; + else if (*n < 0) + info = 2; + else if (*incx == 0) + info = 5; + else if (*incy == 0) + info = 7; + if (info) return xerbla_(SCALAR_SUFFIX_UP "SPR2 ", &info, 6); - if(alpha==Scalar(0)) - return 1; + if (alpha == Scalar(0)) return 1; - Scalar* x_cpy = get_compact_vector(x, *n, *incx); - Scalar* y_cpy = get_compact_vector(y, *n, *incy); + Scalar *x_cpy = get_compact_vector(x, *n, *incx); + Scalar *y_cpy = get_compact_vector(y, *n, *incy); int code = UPLO(*uplo); - if(code>=2 || func[code]==0) - return 0; + if (code >= 2 || func[code] == 0) return 0; func[code](*n, ap, x_cpy, y_cpy, alpha); - if(x_cpy!=x) delete[] x_cpy; - if(y_cpy!=y) delete[] y_cpy; + if (x_cpy != x) delete[] x_cpy; + if (y_cpy != y) delete[] y_cpy; return 1; } /** DGER performs the rank 1 operation - * - * A := alpha*x*y' + A, - * - * where alpha is a scalar, x is an m element vector, y is an n element - * vector and A is an m by n matrix. - */ -int EIGEN_BLAS_FUNC(ger)(int *m, int *n, Scalar *palpha, Scalar *px, int *incx, Scalar *py, int *incy, Scalar *pa, int *lda) + * + * A := alpha*x*y' + A, + * + * where alpha is a scalar, x is an m element vector, y is an n element + * vector and A is an m by n matrix. + */ +int EIGEN_BLAS_FUNC( + ger)(int *m, int *n, Scalar *palpha, Scalar *px, int *incx, Scalar *py, int *incy, Scalar *pa, int *lda) { - Scalar* x = reinterpret_cast(px); - Scalar* y = reinterpret_cast(py); - Scalar* a = reinterpret_cast(pa); - Scalar alpha = *reinterpret_cast(palpha); + Scalar *x = reinterpret_cast(px); + Scalar *y = reinterpret_cast(py); + Scalar *a = reinterpret_cast(pa); + Scalar alpha = *reinterpret_cast(palpha); int info = 0; - if(*m<0) info = 1; - else if(*n<0) info = 2; - else if(*incx==0) info = 5; - else if(*incy==0) info = 7; - else if(*lda::run(*m, *n, a, *lda, x_cpy, y_cpy, alpha); - - if(x_cpy!=x) delete[] x_cpy; - if(y_cpy!=y) delete[] y_cpy; + if (*m < 0) + info = 1; + else if (*n < 0) + info = 2; + else if (*incx == 0) + info = 5; + else if (*incy == 0) + info = 7; + else if (*lda < std::max(1, *m)) + info = 9; + if (info) return xerbla_(SCALAR_SUFFIX_UP "GER ", &info, 6); + + if (alpha == Scalar(0)) return 1; + + Scalar *x_cpy = get_compact_vector(x, *m, *incx); + Scalar *y_cpy = get_compact_vector(y, *n, *incy); + + internal::general_rank1_update::run(*m, *n, a, *lda, x_cpy, y_cpy, alpha); + + if (x_cpy != x) delete[] x_cpy; + if (y_cpy != y) delete[] y_cpy; return 1; } diff --git a/filmulator-gui/core/nlmeans/eigen/blas/level3_impl.h b/filmulator-gui/core/nlmeans/eigen/blas/level3_impl.h index 6c802cd5..d2a768b0 100644 --- a/filmulator-gui/core/nlmeans/eigen/blas/level3_impl.h +++ b/filmulator-gui/core/nlmeans/eigen/blas/level3_impl.h @@ -6,173 +6,239 @@ // This Source Code Form is subject to the terms of the Mozilla // Public License v. 2.0. If a copy of the MPL was not distributed // with this file, You can obtain one at http://mozilla.org/MPL/2.0/. -#include #include "common.h" +#include -int EIGEN_BLAS_FUNC(gemm)(const char *opa, const char *opb, const int *m, const int *n, const int *k, const RealScalar *palpha, - const RealScalar *pa, const int *lda, const RealScalar *pb, const int *ldb, const RealScalar *pbeta, RealScalar *pc, const int *ldc) +int EIGEN_BLAS_FUNC(gemm)(const char *opa, + const char *opb, + const int *m, + const int *n, + const int *k, + const RealScalar *palpha, + const RealScalar *pa, + const int *lda, + const RealScalar *pb, + const int *ldb, + const RealScalar *pbeta, + RealScalar *pc, + const int *ldc) { -// std::cerr << "in gemm " << *opa << " " << *opb << " " << *m << " " << *n << " " << *k << " " << *lda << " " << *ldb << " " << *ldc << " " << *palpha << " " << *pbeta << "\n"; - typedef void (*functype)(DenseIndex, DenseIndex, DenseIndex, const Scalar *, DenseIndex, const Scalar *, DenseIndex, Scalar *, DenseIndex, Scalar, internal::level3_blocking&, Eigen::internal::GemmParallelInfo*); - static const functype func[12] = { - // array index: NOTR | (NOTR << 2) - (internal::general_matrix_matrix_product::run), + // std::cerr << "in gemm " << *opa << " " << *opb << " " << *m << " " << *n << " " << *k << " " << *lda << " " << + // *ldb << " " << *ldc << " " << *palpha << " " << *pbeta << "\n"; + typedef void (*functype)(DenseIndex, + DenseIndex, + DenseIndex, + const Scalar *, + DenseIndex, + const Scalar *, + DenseIndex, + Scalar *, + DenseIndex, + Scalar, + internal::level3_blocking &, + Eigen::internal::GemmParallelInfo *); + static const functype func[12] = { // array index: NOTR | (NOTR << 2) + (internal::general_matrix_matrix_product:: + run), // array index: TR | (NOTR << 2) - (internal::general_matrix_matrix_product::run), + (internal::general_matrix_matrix_product:: + run), // array index: ADJ | (NOTR << 2) - (internal::general_matrix_matrix_product::run), + (internal::general_matrix_matrix_product:: + run), 0, // array index: NOTR | (TR << 2) - (internal::general_matrix_matrix_product::run), + (internal::general_matrix_matrix_product:: + run), // array index: TR | (TR << 2) - (internal::general_matrix_matrix_product::run), + (internal::general_matrix_matrix_product:: + run), // array index: ADJ | (TR << 2) - (internal::general_matrix_matrix_product::run), + (internal::general_matrix_matrix_product:: + run), 0, // array index: NOTR | (ADJ << 2) - (internal::general_matrix_matrix_product::run), + (internal::general_matrix_matrix_product:: + run), // array index: TR | (ADJ << 2) - (internal::general_matrix_matrix_product::run), + (internal::general_matrix_matrix_product:: + run), // array index: ADJ | (ADJ << 2) - (internal::general_matrix_matrix_product::run), + (internal::general_matrix_matrix_product:: + run), 0 }; - const Scalar* a = reinterpret_cast(pa); - const Scalar* b = reinterpret_cast(pb); - Scalar* c = reinterpret_cast(pc); - Scalar alpha = *reinterpret_cast(palpha); - Scalar beta = *reinterpret_cast(pbeta); + const Scalar *a = reinterpret_cast(pa); + const Scalar *b = reinterpret_cast(pb); + Scalar *c = reinterpret_cast(pc); + Scalar alpha = *reinterpret_cast(palpha); + Scalar beta = *reinterpret_cast(pbeta); int info = 0; - if(OP(*opa)==INVALID) info = 1; - else if(OP(*opb)==INVALID) info = 2; - else if(*m<0) info = 3; - else if(*n<0) info = 4; - else if(*k<0) info = 5; - else if(*lda blocking(*m,*n,*k,1,true); + internal::gemm_blocking_space blocking(*m, *n, *k, 1, true); int code = OP(*opa) | (OP(*opb) << 2); func[code](*m, *n, *k, a, *lda, b, *ldb, c, *ldc, alpha, blocking, 0); return 0; } -int EIGEN_BLAS_FUNC(trsm)(const char *side, const char *uplo, const char *opa, const char *diag, const int *m, const int *n, - const RealScalar *palpha, const RealScalar *pa, const int *lda, RealScalar *pb, const int *ldb) +int EIGEN_BLAS_FUNC(trsm)(const char *side, + const char *uplo, + const char *opa, + const char *diag, + const int *m, + const int *n, + const RealScalar *palpha, + const RealScalar *pa, + const int *lda, + RealScalar *pb, + const int *ldb) { -// std::cerr << "in trsm " << *side << " " << *uplo << " " << *opa << " " << *diag << " " << *m << "," << *n << " " << *palpha << " " << *lda << " " << *ldb<< "\n"; - typedef void (*functype)(DenseIndex, DenseIndex, const Scalar *, DenseIndex, Scalar *, DenseIndex, internal::level3_blocking&); - static const functype func[32] = { - // array index: NOTR | (LEFT << 2) | (UP << 3) | (NUNIT << 4) - (internal::triangular_solve_matrix::run), + // std::cerr << "in trsm " << *side << " " << *uplo << " " << *opa << " " << *diag << " " << *m << "," << *n << " " + // << *palpha << " " << *lda << " " << *ldb<< "\n"; + typedef void (*functype)(DenseIndex, + DenseIndex, + const Scalar *, + DenseIndex, + Scalar *, + DenseIndex, + internal::level3_blocking &); + static const functype func[32] = { // array index: NOTR | (LEFT << 2) | (UP << 3) | (NUNIT << 4) + (internal::triangular_solve_matrix::run), // array index: TR | (LEFT << 2) | (UP << 3) | (NUNIT << 4) - (internal::triangular_solve_matrix::run), + (internal::triangular_solve_matrix::run), // array index: ADJ | (LEFT << 2) | (UP << 3) | (NUNIT << 4) - (internal::triangular_solve_matrix::run),\ + (internal::triangular_solve_matrix::run), 0, // array index: NOTR | (RIGHT << 2) | (UP << 3) | (NUNIT << 4) - (internal::triangular_solve_matrix::run), + (internal::triangular_solve_matrix::run), // array index: TR | (RIGHT << 2) | (UP << 3) | (NUNIT << 4) - (internal::triangular_solve_matrix::run), + (internal::triangular_solve_matrix::run), // array index: ADJ | (RIGHT << 2) | (UP << 3) | (NUNIT << 4) - (internal::triangular_solve_matrix::run), + (internal::triangular_solve_matrix::run), 0, // array index: NOTR | (LEFT << 2) | (LO << 3) | (NUNIT << 4) - (internal::triangular_solve_matrix::run), + (internal::triangular_solve_matrix::run), // array index: TR | (LEFT << 2) | (LO << 3) | (NUNIT << 4) - (internal::triangular_solve_matrix::run), + (internal::triangular_solve_matrix::run), // array index: ADJ | (LEFT << 2) | (LO << 3) | (NUNIT << 4) - (internal::triangular_solve_matrix::run), + (internal::triangular_solve_matrix::run), 0, // array index: NOTR | (RIGHT << 2) | (LO << 3) | (NUNIT << 4) - (internal::triangular_solve_matrix::run), + (internal::triangular_solve_matrix::run), // array index: TR | (RIGHT << 2) | (LO << 3) | (NUNIT << 4) - (internal::triangular_solve_matrix::run), + (internal::triangular_solve_matrix::run), // array index: ADJ | (RIGHT << 2) | (LO << 3) | (NUNIT << 4) - (internal::triangular_solve_matrix::run), + (internal::triangular_solve_matrix::run), 0, // array index: NOTR | (LEFT << 2) | (UP << 3) | (UNIT << 4) - (internal::triangular_solve_matrix::run), + (internal::triangular_solve_matrix:: + run), // array index: TR | (LEFT << 2) | (UP << 3) | (UNIT << 4) - (internal::triangular_solve_matrix::run), + (internal::triangular_solve_matrix:: + run), // array index: ADJ | (LEFT << 2) | (UP << 3) | (UNIT << 4) - (internal::triangular_solve_matrix::run), + (internal::triangular_solve_matrix::run), 0, // array index: NOTR | (RIGHT << 2) | (UP << 3) | (UNIT << 4) - (internal::triangular_solve_matrix::run), + (internal::triangular_solve_matrix:: + run), // array index: TR | (RIGHT << 2) | (UP << 3) | (UNIT << 4) - (internal::triangular_solve_matrix::run), + (internal::triangular_solve_matrix:: + run), // array index: ADJ | (RIGHT << 2) | (UP << 3) | (UNIT << 4) - (internal::triangular_solve_matrix::run), + (internal::triangular_solve_matrix:: + run), 0, // array index: NOTR | (LEFT << 2) | (LO << 3) | (UNIT << 4) - (internal::triangular_solve_matrix::run), + (internal::triangular_solve_matrix:: + run), // array index: TR | (LEFT << 2) | (LO << 3) | (UNIT << 4) - (internal::triangular_solve_matrix::run), + (internal::triangular_solve_matrix:: + run), // array index: ADJ | (LEFT << 2) | (LO << 3) | (UNIT << 4) - (internal::triangular_solve_matrix::run), + (internal::triangular_solve_matrix::run), 0, // array index: NOTR | (RIGHT << 2) | (LO << 3) | (UNIT << 4) - (internal::triangular_solve_matrix::run), + (internal::triangular_solve_matrix:: + run), // array index: TR | (RIGHT << 2) | (LO << 3) | (UNIT << 4) - (internal::triangular_solve_matrix::run), + (internal::triangular_solve_matrix:: + run), // array index: ADJ | (RIGHT << 2) | (LO << 3) | (UNIT << 4) - (internal::triangular_solve_matrix::run), + (internal::triangular_solve_matrix:: + run), 0 }; - const Scalar* a = reinterpret_cast(pa); - Scalar* b = reinterpret_cast(pb); - Scalar alpha = *reinterpret_cast(palpha); + const Scalar *a = reinterpret_cast(pa); + Scalar *b = reinterpret_cast(pb); + Scalar alpha = *reinterpret_cast(palpha); int info = 0; - if(SIDE(*side)==INVALID) info = 1; - else if(UPLO(*uplo)==INVALID) info = 2; - else if(OP(*opa)==INVALID) info = 3; - else if(DIAG(*diag)==INVALID) info = 4; - else if(*m<0) info = 5; - else if(*n<0) info = 6; - else if(*lda blocking(*m,*n,*m,1,false); + if (SIDE(*side) == LEFT) { + internal::gemm_blocking_space blocking( + *m, *n, *m, 1, false); func[code](*m, *n, a, *lda, b, *ldb, blocking); - } - else - { - internal::gemm_blocking_space blocking(*m,*n,*n,1,false); + } else { + internal::gemm_blocking_space blocking( + *m, *n, *n, 1, false); func[code](*n, *m, a, *lda, b, *ldb, blocking); } - if(alpha!=Scalar(1)) - matrix(b,*m,*n,*ldb) *= alpha; + if (alpha != Scalar(1)) matrix(b, *m, *n, *ldb) *= alpha; return 0; } @@ -180,103 +246,319 @@ int EIGEN_BLAS_FUNC(trsm)(const char *side, const char *uplo, const char *opa, c // b = alpha*op(a)*b for side = 'L'or'l' // b = alpha*b*op(a) for side = 'R'or'r' -int EIGEN_BLAS_FUNC(trmm)(const char *side, const char *uplo, const char *opa, const char *diag, const int *m, const int *n, - const RealScalar *palpha, const RealScalar *pa, const int *lda, RealScalar *pb, const int *ldb) +int EIGEN_BLAS_FUNC(trmm)(const char *side, + const char *uplo, + const char *opa, + const char *diag, + const int *m, + const int *n, + const RealScalar *palpha, + const RealScalar *pa, + const int *lda, + RealScalar *pb, + const int *ldb) { -// std::cerr << "in trmm " << *side << " " << *uplo << " " << *opa << " " << *diag << " " << *m << " " << *n << " " << *lda << " " << *ldb << " " << *palpha << "\n"; - typedef void (*functype)(DenseIndex, DenseIndex, DenseIndex, const Scalar *, DenseIndex, const Scalar *, DenseIndex, Scalar *, DenseIndex, const Scalar&, internal::level3_blocking&); - static const functype func[32] = { - // array index: NOTR | (LEFT << 2) | (UP << 3) | (NUNIT << 4) - (internal::product_triangular_matrix_matrix::run), + // std::cerr << "in trmm " << *side << " " << *uplo << " " << *opa << " " << *diag << " " << *m << " " << *n << " " + // << *lda << " " << *ldb << " " << *palpha << "\n"; + typedef void (*functype)(DenseIndex, + DenseIndex, + DenseIndex, + const Scalar *, + DenseIndex, + const Scalar *, + DenseIndex, + Scalar *, + DenseIndex, + const Scalar &, + internal::level3_blocking &); + static const functype func[32] = { // array index: NOTR | (LEFT << 2) | (UP << 3) | (NUNIT << 4) + (internal::product_triangular_matrix_matrix::run), // array index: TR | (LEFT << 2) | (UP << 3) | (NUNIT << 4) - (internal::product_triangular_matrix_matrix::run), + (internal::product_triangular_matrix_matrix::run), // array index: ADJ | (LEFT << 2) | (UP << 3) | (NUNIT << 4) - (internal::product_triangular_matrix_matrix::run), + (internal::product_triangular_matrix_matrix::run), 0, // array index: NOTR | (RIGHT << 2) | (UP << 3) | (NUNIT << 4) - (internal::product_triangular_matrix_matrix::run), + (internal::product_triangular_matrix_matrix::run), // array index: TR | (RIGHT << 2) | (UP << 3) | (NUNIT << 4) - (internal::product_triangular_matrix_matrix::run), + (internal::product_triangular_matrix_matrix::run), // array index: ADJ | (RIGHT << 2) | (UP << 3) | (NUNIT << 4) - (internal::product_triangular_matrix_matrix::run), + (internal::product_triangular_matrix_matrix::run), 0, // array index: NOTR | (LEFT << 2) | (LO << 3) | (NUNIT << 4) - (internal::product_triangular_matrix_matrix::run), + (internal::product_triangular_matrix_matrix::run), // array index: TR | (LEFT << 2) | (LO << 3) | (NUNIT << 4) - (internal::product_triangular_matrix_matrix::run), + (internal::product_triangular_matrix_matrix::run), // array index: ADJ | (LEFT << 2) | (LO << 3) | (NUNIT << 4) - (internal::product_triangular_matrix_matrix::run), + (internal::product_triangular_matrix_matrix::run), 0, // array index: NOTR | (RIGHT << 2) | (LO << 3) | (NUNIT << 4) - (internal::product_triangular_matrix_matrix::run), + (internal::product_triangular_matrix_matrix::run), // array index: TR | (RIGHT << 2) | (LO << 3) | (NUNIT << 4) - (internal::product_triangular_matrix_matrix::run), + (internal::product_triangular_matrix_matrix::run), // array index: ADJ | (RIGHT << 2) | (LO << 3) | (NUNIT << 4) - (internal::product_triangular_matrix_matrix::run), + (internal::product_triangular_matrix_matrix::run), 0, // array index: NOTR | (LEFT << 2) | (UP << 3) | (UNIT << 4) - (internal::product_triangular_matrix_matrix::run), + (internal::product_triangular_matrix_matrix::run), // array index: TR | (LEFT << 2) | (UP << 3) | (UNIT << 4) - (internal::product_triangular_matrix_matrix::run), + (internal::product_triangular_matrix_matrix::run), // array index: ADJ | (LEFT << 2) | (UP << 3) | (UNIT << 4) - (internal::product_triangular_matrix_matrix::run), + (internal::product_triangular_matrix_matrix::run), 0, // array index: NOTR | (RIGHT << 2) | (UP << 3) | (UNIT << 4) - (internal::product_triangular_matrix_matrix::run), + (internal::product_triangular_matrix_matrix::run), // array index: TR | (RIGHT << 2) | (UP << 3) | (UNIT << 4) - (internal::product_triangular_matrix_matrix::run), + (internal::product_triangular_matrix_matrix::run), // array index: ADJ | (RIGHT << 2) | (UP << 3) | (UNIT << 4) - (internal::product_triangular_matrix_matrix::run), + (internal::product_triangular_matrix_matrix::run), 0, // array index: NOTR | (LEFT << 2) | (LO << 3) | (UNIT << 4) - (internal::product_triangular_matrix_matrix::run), + (internal::product_triangular_matrix_matrix::run), // array index: TR | (LEFT << 2) | (LO << 3) | (UNIT << 4) - (internal::product_triangular_matrix_matrix::run), + (internal::product_triangular_matrix_matrix::run), // array index: ADJ | (LEFT << 2) | (LO << 3) | (UNIT << 4) - (internal::product_triangular_matrix_matrix::run), + (internal::product_triangular_matrix_matrix::run), 0, // array index: NOTR | (RIGHT << 2) | (LO << 3) | (UNIT << 4) - (internal::product_triangular_matrix_matrix::run), + (internal::product_triangular_matrix_matrix::run), // array index: TR | (RIGHT << 2) | (LO << 3) | (UNIT << 4) - (internal::product_triangular_matrix_matrix::run), + (internal::product_triangular_matrix_matrix::run), // array index: ADJ | (RIGHT << 2) | (LO << 3) | (UNIT << 4) - (internal::product_triangular_matrix_matrix::run), + (internal::product_triangular_matrix_matrix::run), 0 }; - const Scalar* a = reinterpret_cast(pa); - Scalar* b = reinterpret_cast(pb); - Scalar alpha = *reinterpret_cast(palpha); + const Scalar *a = reinterpret_cast(pa); + Scalar *b = reinterpret_cast(pb); + Scalar alpha = *reinterpret_cast(palpha); int info = 0; - if(SIDE(*side)==INVALID) info = 1; - else if(UPLO(*uplo)==INVALID) info = 2; - else if(OP(*opa)==INVALID) info = 3; - else if(DIAG(*diag)==INVALID) info = 4; - else if(*m<0) info = 5; - else if(*n<0) info = 6; - else if(*lda tmp = matrix(b,*m,*n,*ldb); - matrix(b,*m,*n,*ldb).setZero(); + Matrix tmp = matrix(b, *m, *n, *ldb); + matrix(b, *m, *n, *ldb).setZero(); - if(SIDE(*side)==LEFT) - { - internal::gemm_blocking_space blocking(*m,*n,*m,1,false); + if (SIDE(*side) == LEFT) { + internal::gemm_blocking_space blocking( + *m, *n, *m, 1, false); func[code](*m, *n, *m, a, *lda, tmp.data(), tmp.outerStride(), b, *ldb, alpha, blocking); - } - else - { - internal::gemm_blocking_space blocking(*m,*n,*n,1,false); + } else { + internal::gemm_blocking_space blocking( + *m, *n, *n, 1, false); func[code](*m, *n, *n, tmp.data(), tmp.outerStride(), a, *lda, b, *ldb, alpha, blocking); } return 1; @@ -284,214 +566,325 @@ int EIGEN_BLAS_FUNC(trmm)(const char *side, const char *uplo, const char *opa, c // c = alpha*a*b + beta*c for side = 'L'or'l' // c = alpha*b*a + beta*c for side = 'R'or'r -int EIGEN_BLAS_FUNC(symm)(const char *side, const char *uplo, const int *m, const int *n, const RealScalar *palpha, - const RealScalar *pa, const int *lda, const RealScalar *pb, const int *ldb, const RealScalar *pbeta, RealScalar *pc, const int *ldc) +int EIGEN_BLAS_FUNC(symm)(const char *side, + const char *uplo, + const int *m, + const int *n, + const RealScalar *palpha, + const RealScalar *pa, + const int *lda, + const RealScalar *pb, + const int *ldb, + const RealScalar *pbeta, + RealScalar *pc, + const int *ldc) { -// std::cerr << "in symm " << *side << " " << *uplo << " " << *m << "x" << *n << " lda:" << *lda << " ldb:" << *ldb << " ldc:" << *ldc << " alpha:" << *palpha << " beta:" << *pbeta << "\n"; - const Scalar* a = reinterpret_cast(pa); - const Scalar* b = reinterpret_cast(pb); - Scalar* c = reinterpret_cast(pc); - Scalar alpha = *reinterpret_cast(palpha); - Scalar beta = *reinterpret_cast(pbeta); + // std::cerr << "in symm " << *side << " " << *uplo << " " << *m << "x" << *n << " lda:" << *lda << " ldb:" << *ldb + // << " ldc:" << *ldc << " alpha:" << *palpha << " beta:" << *pbeta << "\n"; + const Scalar *a = reinterpret_cast(pa); + const Scalar *b = reinterpret_cast(pb); + Scalar *c = reinterpret_cast(pc); + Scalar alpha = *reinterpret_cast(palpha); + Scalar beta = *reinterpret_cast(pbeta); int info = 0; - if(SIDE(*side)==INVALID) info = 1; - else if(UPLO(*uplo)==INVALID) info = 2; - else if(*m<0) info = 3; - else if(*n<0) info = 4; - else if(*lda matA(size,size); - if(UPLO(*uplo)==UP) - { - matA.triangularView() = matrix(a,size,size,*lda); - matA.triangularView() = matrix(a,size,size,*lda).transpose(); - } - else if(UPLO(*uplo)==LO) - { - matA.triangularView() = matrix(a,size,size,*lda); - matA.triangularView() = matrix(a,size,size,*lda).transpose(); + Matrix matA(size, size); + if (UPLO(*uplo) == UP) { + matA.triangularView() = matrix(a, size, size, *lda); + matA.triangularView() = matrix(a, size, size, *lda).transpose(); + } else if (UPLO(*uplo) == LO) { + matA.triangularView() = matrix(a, size, size, *lda); + matA.triangularView() = matrix(a, size, size, *lda).transpose(); } - if(SIDE(*side)==LEFT) + if (SIDE(*side) == LEFT) matrix(c, *m, *n, *ldc) += alpha * matA * matrix(b, *m, *n, *ldb); - else if(SIDE(*side)==RIGHT) + else if (SIDE(*side) == RIGHT) matrix(c, *m, *n, *ldc) += alpha * matrix(b, *m, *n, *ldb) * matA; - #else - internal::gemm_blocking_space blocking(*m,*n,size,1,false); - - if(SIDE(*side)==LEFT) - if(UPLO(*uplo)==UP) internal::product_selfadjoint_matrix::run(*m, *n, a, *lda, b, *ldb, c, *ldc, alpha, blocking); - else if(UPLO(*uplo)==LO) internal::product_selfadjoint_matrix::run(*m, *n, a, *lda, b, *ldb, c, *ldc, alpha, blocking); - else return 0; - else if(SIDE(*side)==RIGHT) - if(UPLO(*uplo)==UP) internal::product_selfadjoint_matrix::run(*m, *n, b, *ldb, a, *lda, c, *ldc, alpha, blocking); - else if(UPLO(*uplo)==LO) internal::product_selfadjoint_matrix::run(*m, *n, b, *ldb, a, *lda, c, *ldc, alpha, blocking); - else return 0; +#else + internal::gemm_blocking_space blocking(*m, *n, size, 1, false); + + if (SIDE(*side) == LEFT) + if (UPLO(*uplo) == UP) + internal:: + product_selfadjoint_matrix::run( + *m, *n, a, *lda, b, *ldb, c, *ldc, alpha, blocking); + else if (UPLO(*uplo) == LO) + internal:: + product_selfadjoint_matrix::run( + *m, *n, a, *lda, b, *ldb, c, *ldc, alpha, blocking); + else + return 0; + else if (SIDE(*side) == RIGHT) + if (UPLO(*uplo) == UP) + internal:: + product_selfadjoint_matrix::run( + *m, *n, b, *ldb, a, *lda, c, *ldc, alpha, blocking); + else if (UPLO(*uplo) == LO) + internal:: + product_selfadjoint_matrix::run( + *m, *n, b, *ldb, a, *lda, c, *ldc, alpha, blocking); + else + return 0; else return 0; - #endif +#endif return 0; } // c = alpha*a*a' + beta*c for op = 'N'or'n' // c = alpha*a'*a + beta*c for op = 'T'or't','C'or'c' -int EIGEN_BLAS_FUNC(syrk)(const char *uplo, const char *op, const int *n, const int *k, - const RealScalar *palpha, const RealScalar *pa, const int *lda, const RealScalar *pbeta, RealScalar *pc, const int *ldc) +int EIGEN_BLAS_FUNC(syrk)(const char *uplo, + const char *op, + const int *n, + const int *k, + const RealScalar *palpha, + const RealScalar *pa, + const int *lda, + const RealScalar *pbeta, + RealScalar *pc, + const int *ldc) { -// std::cerr << "in syrk " << *uplo << " " << *op << " " << *n << " " << *k << " " << *palpha << " " << *lda << " " << *pbeta << " " << *ldc << "\n"; - #if !ISCOMPLEX - typedef void (*functype)(DenseIndex, DenseIndex, const Scalar *, DenseIndex, const Scalar *, DenseIndex, Scalar *, DenseIndex, const Scalar&, internal::level3_blocking&); - static const functype func[8] = { - // array index: NOTR | (UP << 2) - (internal::general_matrix_matrix_triangular_product::run), + // std::cerr << "in syrk " << *uplo << " " << *op << " " << *n << " " << *k << " " << *palpha << " " << *lda << " " + // << *pbeta << " " << *ldc << "\n"; +#if !ISCOMPLEX + typedef void (*functype)(DenseIndex, + DenseIndex, + const Scalar *, + DenseIndex, + const Scalar *, + DenseIndex, + Scalar *, + DenseIndex, + const Scalar &, + internal::level3_blocking &); + static const functype func[8] = { // array index: NOTR | (UP << 2) + (internal::general_matrix_matrix_triangular_product::run), // array index: TR | (UP << 2) - (internal::general_matrix_matrix_triangular_product::run), + (internal::general_matrix_matrix_triangular_product::run), // array index: ADJ | (UP << 2) - (internal::general_matrix_matrix_triangular_product::run), + (internal::general_matrix_matrix_triangular_product::run), 0, // array index: NOTR | (LO << 2) - (internal::general_matrix_matrix_triangular_product::run), + (internal::general_matrix_matrix_triangular_product::run), // array index: TR | (LO << 2) - (internal::general_matrix_matrix_triangular_product::run), + (internal::general_matrix_matrix_triangular_product::run), // array index: ADJ | (LO << 2) - (internal::general_matrix_matrix_triangular_product::run), + (internal::general_matrix_matrix_triangular_product::run), 0 }; - #endif +#endif - const Scalar* a = reinterpret_cast(pa); - Scalar* c = reinterpret_cast(pc); - Scalar alpha = *reinterpret_cast(palpha); - Scalar beta = *reinterpret_cast(pbeta); + const Scalar *a = reinterpret_cast(pa); + Scalar *c = reinterpret_cast(pc); + Scalar alpha = *reinterpret_cast(palpha); + Scalar beta = *reinterpret_cast(pbeta); int info = 0; - if(UPLO(*uplo)==INVALID) info = 1; - else if(OP(*op)==INVALID || (ISCOMPLEX && OP(*op)==ADJ) ) info = 2; - else if(*n<0) info = 3; - else if(*k<0) info = 4; - else if(*lda().setZero(); - else matrix(c, *n, *n, *ldc).triangularView() *= beta; + if (UPLO(*uplo) == INVALID) + info = 1; + else if (OP(*op) == INVALID || (ISCOMPLEX && OP(*op) == ADJ)) + info = 2; + else if (*n < 0) + info = 3; + else if (*k < 0) + info = 4; + else if (*lda < std::max(1, (OP(*op) == NOTR) ? *n : *k)) + info = 7; + else if (*ldc < std::max(1, *n)) + info = 10; + if (info) return xerbla_(SCALAR_SUFFIX_UP "SYRK ", &info, 6); + + if (beta != Scalar(1)) { + if (UPLO(*uplo) == UP) + if (beta == Scalar(0)) + matrix(c, *n, *n, *ldc).triangularView().setZero(); + else + matrix(c, *n, *n, *ldc).triangularView() *= beta; + else if (beta == Scalar(0)) + matrix(c, *n, *n, *ldc).triangularView().setZero(); else - if(beta==Scalar(0)) matrix(c, *n, *n, *ldc).triangularView().setZero(); - else matrix(c, *n, *n, *ldc).triangularView() *= beta; + matrix(c, *n, *n, *ldc).triangularView() *= beta; } - if(*n==0 || *k==0) - return 0; + if (*n == 0 || *k == 0) return 0; - #if ISCOMPLEX +#if ISCOMPLEX // FIXME add support for symmetric complex matrix - if(UPLO(*uplo)==UP) - { - if(OP(*op)==NOTR) - matrix(c, *n, *n, *ldc).triangularView() += alpha * matrix(a,*n,*k,*lda) * matrix(a,*n,*k,*lda).transpose(); + if (UPLO(*uplo) == UP) { + if (OP(*op) == NOTR) + matrix(c, *n, *n, *ldc).triangularView() += + alpha * matrix(a, *n, *k, *lda) * matrix(a, *n, *k, *lda).transpose(); else - matrix(c, *n, *n, *ldc).triangularView() += alpha * matrix(a,*k,*n,*lda).transpose() * matrix(a,*k,*n,*lda); - } - else - { - if(OP(*op)==NOTR) - matrix(c, *n, *n, *ldc).triangularView() += alpha * matrix(a,*n,*k,*lda) * matrix(a,*n,*k,*lda).transpose(); + matrix(c, *n, *n, *ldc).triangularView() += + alpha * matrix(a, *k, *n, *lda).transpose() * matrix(a, *k, *n, *lda); + } else { + if (OP(*op) == NOTR) + matrix(c, *n, *n, *ldc).triangularView() += + alpha * matrix(a, *n, *k, *lda) * matrix(a, *n, *k, *lda).transpose(); else - matrix(c, *n, *n, *ldc).triangularView() += alpha * matrix(a,*k,*n,*lda).transpose() * matrix(a,*k,*n,*lda); + matrix(c, *n, *n, *ldc).triangularView() += + alpha * matrix(a, *k, *n, *lda).transpose() * matrix(a, *k, *n, *lda); } - #else - internal::gemm_blocking_space blocking(*n,*n,*k,1,false); +#else + internal::gemm_blocking_space blocking(*n, *n, *k, 1, false); int code = OP(*op) | (UPLO(*uplo) << 2); func[code](*n, *k, a, *lda, a, *lda, c, *ldc, alpha, blocking); - #endif +#endif return 0; } // c = alpha*a*b' + alpha*b*a' + beta*c for op = 'N'or'n' // c = alpha*a'*b + alpha*b'*a + beta*c for op = 'T'or't' -int EIGEN_BLAS_FUNC(syr2k)(const char *uplo, const char *op, const int *n, const int *k, const RealScalar *palpha, - const RealScalar *pa, const int *lda, const RealScalar *pb, const int *ldb, const RealScalar *pbeta, RealScalar *pc, const int *ldc) +int EIGEN_BLAS_FUNC(syr2k)(const char *uplo, + const char *op, + const int *n, + const int *k, + const RealScalar *palpha, + const RealScalar *pa, + const int *lda, + const RealScalar *pb, + const int *ldb, + const RealScalar *pbeta, + RealScalar *pc, + const int *ldc) { - const Scalar* a = reinterpret_cast(pa); - const Scalar* b = reinterpret_cast(pb); - Scalar* c = reinterpret_cast(pc); - Scalar alpha = *reinterpret_cast(palpha); - Scalar beta = *reinterpret_cast(pbeta); + const Scalar *a = reinterpret_cast(pa); + const Scalar *b = reinterpret_cast(pb); + Scalar *c = reinterpret_cast(pc); + Scalar alpha = *reinterpret_cast(palpha); + Scalar beta = *reinterpret_cast(pbeta); -// std::cerr << "in syr2k " << *uplo << " " << *op << " " << *n << " " << *k << " " << alpha << " " << *lda << " " << *ldb << " " << beta << " " << *ldc << "\n"; + // std::cerr << "in syr2k " << *uplo << " " << *op << " " << *n << " " << *k << " " << alpha << " " << *lda << " " + // << *ldb << " " << beta << " " << *ldc << "\n"; int info = 0; - if(UPLO(*uplo)==INVALID) info = 1; - else if(OP(*op)==INVALID || (ISCOMPLEX && OP(*op)==ADJ) ) info = 2; - else if(*n<0) info = 3; - else if(*k<0) info = 4; - else if(*lda().setZero(); - else matrix(c, *n, *n, *ldc).triangularView() *= beta; + if (UPLO(*uplo) == INVALID) + info = 1; + else if (OP(*op) == INVALID || (ISCOMPLEX && OP(*op) == ADJ)) + info = 2; + else if (*n < 0) + info = 3; + else if (*k < 0) + info = 4; + else if (*lda < std::max(1, (OP(*op) == NOTR) ? *n : *k)) + info = 7; + else if (*ldb < std::max(1, (OP(*op) == NOTR) ? *n : *k)) + info = 9; + else if (*ldc < std::max(1, *n)) + info = 12; + if (info) return xerbla_(SCALAR_SUFFIX_UP "SYR2K", &info, 6); + + if (beta != Scalar(1)) { + if (UPLO(*uplo) == UP) + if (beta == Scalar(0)) + matrix(c, *n, *n, *ldc).triangularView().setZero(); + else + matrix(c, *n, *n, *ldc).triangularView() *= beta; + else if (beta == Scalar(0)) + matrix(c, *n, *n, *ldc).triangularView().setZero(); else - if(beta==Scalar(0)) matrix(c, *n, *n, *ldc).triangularView().setZero(); - else matrix(c, *n, *n, *ldc).triangularView() *= beta; + matrix(c, *n, *n, *ldc).triangularView() *= beta; } - if(*k==0) - return 1; - - if(OP(*op)==NOTR) - { - if(UPLO(*uplo)==UP) - { - matrix(c, *n, *n, *ldc).triangularView() - += alpha *matrix(a, *n, *k, *lda)*matrix(b, *n, *k, *ldb).transpose() - + alpha*matrix(b, *n, *k, *ldb)*matrix(a, *n, *k, *lda).transpose(); - } - else if(UPLO(*uplo)==LO) - matrix(c, *n, *n, *ldc).triangularView() - += alpha*matrix(a, *n, *k, *lda)*matrix(b, *n, *k, *ldb).transpose() - + alpha*matrix(b, *n, *k, *ldb)*matrix(a, *n, *k, *lda).transpose(); - } - else if(OP(*op)==TR || OP(*op)==ADJ) - { - if(UPLO(*uplo)==UP) - matrix(c, *n, *n, *ldc).triangularView() - += alpha*matrix(a, *k, *n, *lda).transpose()*matrix(b, *k, *n, *ldb) - + alpha*matrix(b, *k, *n, *ldb).transpose()*matrix(a, *k, *n, *lda); - else if(UPLO(*uplo)==LO) - matrix(c, *n, *n, *ldc).triangularView() - += alpha*matrix(a, *k, *n, *lda).transpose()*matrix(b, *k, *n, *ldb) - + alpha*matrix(b, *k, *n, *ldb).transpose()*matrix(a, *k, *n, *lda); + if (*k == 0) return 1; + + if (OP(*op) == NOTR) { + if (UPLO(*uplo) == UP) { + matrix(c, *n, *n, *ldc).triangularView() += + alpha * matrix(a, *n, *k, *lda) * matrix(b, *n, *k, *ldb).transpose() + + alpha * matrix(b, *n, *k, *ldb) * matrix(a, *n, *k, *lda).transpose(); + } else if (UPLO(*uplo) == LO) + matrix(c, *n, *n, *ldc).triangularView() += + alpha * matrix(a, *n, *k, *lda) * matrix(b, *n, *k, *ldb).transpose() + + alpha * matrix(b, *n, *k, *ldb) * matrix(a, *n, *k, *lda).transpose(); + } else if (OP(*op) == TR || OP(*op) == ADJ) { + if (UPLO(*uplo) == UP) + matrix(c, *n, *n, *ldc).triangularView() += + alpha * matrix(a, *k, *n, *lda).transpose() * matrix(b, *k, *n, *ldb) + + alpha * matrix(b, *k, *n, *ldb).transpose() * matrix(a, *k, *n, *lda); + else if (UPLO(*uplo) == LO) + matrix(c, *n, *n, *ldc).triangularView() += + alpha * matrix(a, *k, *n, *lda).transpose() * matrix(b, *k, *n, *ldb) + + alpha * matrix(b, *k, *n, *ldb).transpose() * matrix(a, *k, *n, *lda); } return 0; @@ -502,57 +895,78 @@ int EIGEN_BLAS_FUNC(syr2k)(const char *uplo, const char *op, const int *n, const // c = alpha*a*b + beta*c for side = 'L'or'l' // c = alpha*b*a + beta*c for side = 'R'or'r -int EIGEN_BLAS_FUNC(hemm)(const char *side, const char *uplo, const int *m, const int *n, const RealScalar *palpha, - const RealScalar *pa, const int *lda, const RealScalar *pb, const int *ldb, const RealScalar *pbeta, RealScalar *pc, const int *ldc) +int EIGEN_BLAS_FUNC(hemm)(const char *side, + const char *uplo, + const int *m, + const int *n, + const RealScalar *palpha, + const RealScalar *pa, + const int *lda, + const RealScalar *pb, + const int *ldb, + const RealScalar *pbeta, + RealScalar *pc, + const int *ldc) { - const Scalar* a = reinterpret_cast(pa); - const Scalar* b = reinterpret_cast(pb); - Scalar* c = reinterpret_cast(pc); - Scalar alpha = *reinterpret_cast(palpha); - Scalar beta = *reinterpret_cast(pbeta); + const Scalar *a = reinterpret_cast(pa); + const Scalar *b = reinterpret_cast(pb); + Scalar *c = reinterpret_cast(pc); + Scalar alpha = *reinterpret_cast(palpha); + Scalar beta = *reinterpret_cast(pbeta); -// std::cerr << "in hemm " << *side << " " << *uplo << " " << *m << " " << *n << " " << alpha << " " << *lda << " " << beta << " " << *ldc << "\n"; + // std::cerr << "in hemm " << *side << " " << *uplo << " " << *m << " " << *n << " " << alpha << " " << *lda << " " + // << beta << " " << *ldc << "\n"; int info = 0; - if(SIDE(*side)==INVALID) info = 1; - else if(UPLO(*uplo)==INVALID) info = 2; - else if(*m<0) info = 3; - else if(*n<0) info = 4; - else if(*lda blocking(*m,*n,size,1,false); - - if(SIDE(*side)==LEFT) - { - if(UPLO(*uplo)==UP) internal::product_selfadjoint_matrix - ::run(*m, *n, a, *lda, b, *ldb, c, *ldc, alpha, blocking); - else if(UPLO(*uplo)==LO) internal::product_selfadjoint_matrix - ::run(*m, *n, a, *lda, b, *ldb, c, *ldc, alpha, blocking); - else return 0; - } - else if(SIDE(*side)==RIGHT) - { - if(UPLO(*uplo)==UP) matrix(c,*m,*n,*ldc) += alpha * matrix(b,*m,*n,*ldb) * matrix(a,*n,*n,*lda).selfadjointView();/*internal::product_selfadjoint_matrix - ::run(*m, *n, b, *ldb, a, *lda, c, *ldc, alpha, blocking);*/ - else if(UPLO(*uplo)==LO) internal::product_selfadjoint_matrix - ::run(*m, *n, b, *ldb, a, *lda, c, *ldc, alpha, blocking); - else return 0; - } - else - { + if (SIDE(*side) == INVALID) + info = 1; + else if (UPLO(*uplo) == INVALID) + info = 2; + else if (*m < 0) + info = 3; + else if (*n < 0) + info = 4; + else if (*lda < std::max(1, (SIDE(*side) == LEFT) ? *m : *n)) + info = 7; + else if (*ldb < std::max(1, *m)) + info = 9; + else if (*ldc < std::max(1, *m)) + info = 12; + if (info) return xerbla_(SCALAR_SUFFIX_UP "HEMM ", &info, 6); + + if (beta == Scalar(0)) + matrix(c, *m, *n, *ldc).setZero(); + else if (beta != Scalar(1)) + matrix(c, *m, *n, *ldc) *= beta; + + if (*m == 0 || *n == 0) { return 1; } + + int size = (SIDE(*side) == LEFT) ? (*m) : (*n); + internal::gemm_blocking_space blocking(*m, *n, size, 1, false); + + if (SIDE(*side) == LEFT) { + if (UPLO(*uplo) == UP) + internal::product_selfadjoint_matrix:: + run(*m, *n, a, *lda, b, *ldb, c, *ldc, alpha, blocking); + else if (UPLO(*uplo) == LO) + internal:: + product_selfadjoint_matrix::run( + *m, *n, a, *lda, b, *ldb, c, *ldc, alpha, blocking); + else + return 0; + } else if (SIDE(*side) == RIGHT) { + if (UPLO(*uplo) == UP) + matrix(c, *m, *n, *ldc) += + alpha * matrix(b, *m, *n, *ldb) * matrix(a, *n, *n, *lda).selfadjointView(); /*internal::product_selfadjoint_matrix +::run(*m, *n, b, *ldb, a, *lda, c, *ldc, alpha, blocking);*/ + else if (UPLO(*uplo) == LO) + internal:: + product_selfadjoint_matrix::run( + *m, *n, b, *ldb, a, *lda, c, *ldc, alpha, blocking); + else + return 0; + } else { return 0; } @@ -561,65 +975,120 @@ int EIGEN_BLAS_FUNC(hemm)(const char *side, const char *uplo, const int *m, cons // c = alpha*a*conj(a') + beta*c for op = 'N'or'n' // c = alpha*conj(a')*a + beta*c for op = 'C'or'c' -int EIGEN_BLAS_FUNC(herk)(const char *uplo, const char *op, const int *n, const int *k, - const RealScalar *palpha, const RealScalar *pa, const int *lda, const RealScalar *pbeta, RealScalar *pc, const int *ldc) +int EIGEN_BLAS_FUNC(herk)(const char *uplo, + const char *op, + const int *n, + const int *k, + const RealScalar *palpha, + const RealScalar *pa, + const int *lda, + const RealScalar *pbeta, + RealScalar *pc, + const int *ldc) { -// std::cerr << "in herk " << *uplo << " " << *op << " " << *n << " " << *k << " " << *palpha << " " << *lda << " " << *pbeta << " " << *ldc << "\n"; - - typedef void (*functype)(DenseIndex, DenseIndex, const Scalar *, DenseIndex, const Scalar *, DenseIndex, Scalar *, DenseIndex, const Scalar&, internal::level3_blocking&); - static const functype func[8] = { - // array index: NOTR | (UP << 2) - (internal::general_matrix_matrix_triangular_product::run), + // std::cerr << "in herk " << *uplo << " " << *op << " " << *n << " " << *k << " " << *palpha << " " << *lda << " " + // << *pbeta << " " << *ldc << "\n"; + + typedef void (*functype)(DenseIndex, + DenseIndex, + const Scalar *, + DenseIndex, + const Scalar *, + DenseIndex, + Scalar *, + DenseIndex, + const Scalar &, + internal::level3_blocking &); + static const functype func[8] = { // array index: NOTR | (UP << 2) + (internal::general_matrix_matrix_triangular_product::run), 0, // array index: ADJ | (UP << 2) - (internal::general_matrix_matrix_triangular_product::run), + (internal::general_matrix_matrix_triangular_product::run), 0, // array index: NOTR | (LO << 2) - (internal::general_matrix_matrix_triangular_product::run), + (internal::general_matrix_matrix_triangular_product::run), 0, // array index: ADJ | (LO << 2) - (internal::general_matrix_matrix_triangular_product::run), + (internal::general_matrix_matrix_triangular_product::run), 0 }; - const Scalar* a = reinterpret_cast(pa); - Scalar* c = reinterpret_cast(pc); + const Scalar *a = reinterpret_cast(pa); + Scalar *c = reinterpret_cast(pc); RealScalar alpha = *palpha; - RealScalar beta = *pbeta; + RealScalar beta = *pbeta; -// std::cerr << "in herk " << *uplo << " " << *op << " " << *n << " " << *k << " " << alpha << " " << *lda << " " << beta << " " << *ldc << "\n"; + // std::cerr << "in herk " << *uplo << " " << *op << " " << *n << " " << *k << " " << alpha << " " << *lda << " " << + // beta << " " << *ldc << "\n"; int info = 0; - if(UPLO(*uplo)==INVALID) info = 1; - else if((OP(*op)==INVALID) || (OP(*op)==TR)) info = 2; - else if(*n<0) info = 3; - else if(*k<0) info = 4; - else if(*lda().setZero(); - else matrix(c, *n, *n, *ldc).triangularView() *= beta; + if (beta != RealScalar(1)) { + if (UPLO(*uplo) == UP) + if (beta == Scalar(0)) + matrix(c, *n, *n, *ldc).triangularView().setZero(); + else + matrix(c, *n, *n, *ldc).triangularView() *= beta; + else if (beta == Scalar(0)) + matrix(c, *n, *n, *ldc).triangularView().setZero(); else - if(beta==Scalar(0)) matrix(c, *n, *n, *ldc).triangularView().setZero(); - else matrix(c, *n, *n, *ldc).triangularView() *= beta; + matrix(c, *n, *n, *ldc).triangularView() *= beta; - if(beta!=Scalar(0)) - { + if (beta != Scalar(0)) { matrix(c, *n, *n, *ldc).diagonal().real() *= beta; matrix(c, *n, *n, *ldc).diagonal().imag().setZero(); } } - if(*k>0 && alpha!=RealScalar(0)) - { - internal::gemm_blocking_space blocking(*n,*n,*k,1,false); + if (*k > 0 && alpha != RealScalar(0)) { + internal::gemm_blocking_space blocking(*n, *n, *k, 1, false); func[code](*n, *k, a, *lda, a, *lda, c, *ldc, alpha, blocking); matrix(c, *n, *n, *ldc).diagonal().imag().setZero(); } @@ -628,75 +1097,86 @@ int EIGEN_BLAS_FUNC(herk)(const char *uplo, const char *op, const int *n, const // c = alpha*a*conj(b') + conj(alpha)*b*conj(a') + beta*c, for op = 'N'or'n' // c = alpha*conj(a')*b + conj(alpha)*conj(b')*a + beta*c, for op = 'C'or'c' -int EIGEN_BLAS_FUNC(her2k)(const char *uplo, const char *op, const int *n, const int *k, - const RealScalar *palpha, const RealScalar *pa, const int *lda, const RealScalar *pb, const int *ldb, const RealScalar *pbeta, RealScalar *pc, const int *ldc) +int EIGEN_BLAS_FUNC(her2k)(const char *uplo, + const char *op, + const int *n, + const int *k, + const RealScalar *palpha, + const RealScalar *pa, + const int *lda, + const RealScalar *pb, + const int *ldb, + const RealScalar *pbeta, + RealScalar *pc, + const int *ldc) { - const Scalar* a = reinterpret_cast(pa); - const Scalar* b = reinterpret_cast(pb); - Scalar* c = reinterpret_cast(pc); - Scalar alpha = *reinterpret_cast(palpha); - RealScalar beta = *pbeta; + const Scalar *a = reinterpret_cast(pa); + const Scalar *b = reinterpret_cast(pb); + Scalar *c = reinterpret_cast(pc); + Scalar alpha = *reinterpret_cast(palpha); + RealScalar beta = *pbeta; -// std::cerr << "in her2k " << *uplo << " " << *op << " " << *n << " " << *k << " " << alpha << " " << *lda << " " << *ldb << " " << beta << " " << *ldc << "\n"; + // std::cerr << "in her2k " << *uplo << " " << *op << " " << *n << " " << *k << " " << alpha << " " << *lda << " " + // << *ldb << " " << beta << " " << *ldc << "\n"; int info = 0; - if(UPLO(*uplo)==INVALID) info = 1; - else if((OP(*op)==INVALID) || (OP(*op)==TR)) info = 2; - else if(*n<0) info = 3; - else if(*k<0) info = 4; - else if(*lda().setZero(); - else matrix(c, *n, *n, *ldc).triangularView() *= beta; + if (UPLO(*uplo) == INVALID) + info = 1; + else if ((OP(*op) == INVALID) || (OP(*op) == TR)) + info = 2; + else if (*n < 0) + info = 3; + else if (*k < 0) + info = 4; + else if (*lda < std::max(1, (OP(*op) == NOTR) ? *n : *k)) + info = 7; + else if (*ldb < std::max(1, (OP(*op) == NOTR) ? *n : *k)) + info = 9; + else if (*ldc < std::max(1, *n)) + info = 12; + if (info) return xerbla_(SCALAR_SUFFIX_UP "HER2K", &info, 6); + + if (beta != RealScalar(1)) { + if (UPLO(*uplo) == UP) + if (beta == Scalar(0)) + matrix(c, *n, *n, *ldc).triangularView().setZero(); + else + matrix(c, *n, *n, *ldc).triangularView() *= beta; + else if (beta == Scalar(0)) + matrix(c, *n, *n, *ldc).triangularView().setZero(); else - if(beta==Scalar(0)) matrix(c, *n, *n, *ldc).triangularView().setZero(); - else matrix(c, *n, *n, *ldc).triangularView() *= beta; + matrix(c, *n, *n, *ldc).triangularView() *= beta; - if(beta!=Scalar(0)) - { + if (beta != Scalar(0)) { matrix(c, *n, *n, *ldc).diagonal().real() *= beta; matrix(c, *n, *n, *ldc).diagonal().imag().setZero(); } - } - else if(*k>0 && alpha!=Scalar(0)) + } else if (*k > 0 && alpha != Scalar(0)) matrix(c, *n, *n, *ldc).diagonal().imag().setZero(); - if(*k==0) - return 1; - - if(OP(*op)==NOTR) - { - if(UPLO(*uplo)==UP) - { - matrix(c, *n, *n, *ldc).triangularView() - += alpha *matrix(a, *n, *k, *lda)*matrix(b, *n, *k, *ldb).adjoint() - + numext::conj(alpha)*matrix(b, *n, *k, *ldb)*matrix(a, *n, *k, *lda).adjoint(); - } - else if(UPLO(*uplo)==LO) - matrix(c, *n, *n, *ldc).triangularView() - += alpha*matrix(a, *n, *k, *lda)*matrix(b, *n, *k, *ldb).adjoint() - + numext::conj(alpha)*matrix(b, *n, *k, *ldb)*matrix(a, *n, *k, *lda).adjoint(); - } - else if(OP(*op)==ADJ) - { - if(UPLO(*uplo)==UP) - matrix(c, *n, *n, *ldc).triangularView() - += alpha*matrix(a, *k, *n, *lda).adjoint()*matrix(b, *k, *n, *ldb) - + numext::conj(alpha)*matrix(b, *k, *n, *ldb).adjoint()*matrix(a, *k, *n, *lda); - else if(UPLO(*uplo)==LO) - matrix(c, *n, *n, *ldc).triangularView() - += alpha*matrix(a, *k, *n, *lda).adjoint()*matrix(b, *k, *n, *ldb) - + numext::conj(alpha)*matrix(b, *k, *n, *ldb).adjoint()*matrix(a, *k, *n, *lda); + if (*k == 0) return 1; + + if (OP(*op) == NOTR) { + if (UPLO(*uplo) == UP) { + matrix(c, *n, *n, *ldc).triangularView() += + alpha * matrix(a, *n, *k, *lda) * matrix(b, *n, *k, *ldb).adjoint() + + numext::conj(alpha) * matrix(b, *n, *k, *ldb) * matrix(a, *n, *k, *lda).adjoint(); + } else if (UPLO(*uplo) == LO) + matrix(c, *n, *n, *ldc).triangularView() += + alpha * matrix(a, *n, *k, *lda) * matrix(b, *n, *k, *ldb).adjoint() + + numext::conj(alpha) * matrix(b, *n, *k, *ldb) * matrix(a, *n, *k, *lda).adjoint(); + } else if (OP(*op) == ADJ) { + if (UPLO(*uplo) == UP) + matrix(c, *n, *n, *ldc).triangularView() += + alpha * matrix(a, *k, *n, *lda).adjoint() * matrix(b, *k, *n, *ldb) + + numext::conj(alpha) * matrix(b, *k, *n, *ldb).adjoint() * matrix(a, *k, *n, *lda); + else if (UPLO(*uplo) == LO) + matrix(c, *n, *n, *ldc).triangularView() += + alpha * matrix(a, *k, *n, *lda).adjoint() * matrix(b, *k, *n, *ldb) + + numext::conj(alpha) * matrix(b, *k, *n, *ldb).adjoint() * matrix(a, *k, *n, *lda); } return 1; } -#endif // ISCOMPLEX +#endif// ISCOMPLEX diff --git a/filmulator-gui/core/nlmeans/eigen/blas/single.cpp b/filmulator-gui/core/nlmeans/eigen/blas/single.cpp index 20ea57d5..4fbcc668 100644 --- a/filmulator-gui/core/nlmeans/eigen/blas/single.cpp +++ b/filmulator-gui/core/nlmeans/eigen/blas/single.cpp @@ -7,10 +7,10 @@ // Public License v. 2.0. If a copy of the MPL was not distributed // with this file, You can obtain one at http://mozilla.org/MPL/2.0/. -#define SCALAR float +#define SCALAR float #define SCALAR_SUFFIX s #define SCALAR_SUFFIX_UP "S" -#define ISCOMPLEX 0 +#define ISCOMPLEX 0 #include "level1_impl.h" #include "level1_real_impl.h" @@ -18,5 +18,7 @@ #include "level2_real_impl.h" #include "level3_impl.h" -float BLASFUNC(sdsdot)(int* n, float* alpha, float* x, int* incx, float* y, int* incy) -{ return double(*alpha) + BLASFUNC(dsdot)(n, x, incx, y, incy); } +float BLASFUNC(sdsdot)(int *n, float *alpha, float *x, int *incx, float *y, int *incy) +{ + return double(*alpha) + BLASFUNC(dsdot)(n, x, incx, y, incy); +} diff --git a/filmulator-gui/core/nlmeans/eigen/blas/xerbla.cpp b/filmulator-gui/core/nlmeans/eigen/blas/xerbla.cpp index c373e869..9ec56373 100644 --- a/filmulator-gui/core/nlmeans/eigen/blas/xerbla.cpp +++ b/filmulator-gui/core/nlmeans/eigen/blas/xerbla.cpp @@ -2,19 +2,18 @@ #include #if (defined __GNUC__) && (!defined __MINGW32__) && (!defined __CYGWIN__) -#define EIGEN_WEAK_LINKING __attribute__ ((weak)) +#define EIGEN_WEAK_LINKING __attribute__((weak)) #else #define EIGEN_WEAK_LINKING #endif #ifdef __cplusplus -extern "C" -{ +extern "C" { #endif -EIGEN_WEAK_LINKING int xerbla_(const char * msg, int *info, int) +EIGEN_WEAK_LINKING int xerbla_(const char *msg, int *info, int) { - printf("Eigen BLAS ERROR #%i: %s\n", *info, msg ); + printf("Eigen BLAS ERROR #%i: %s\n", *info, msg); return 0; } diff --git a/filmulator-gui/core/nlmeans/eigen/demos/mandelbrot/mandelbrot.cpp b/filmulator-gui/core/nlmeans/eigen/demos/mandelbrot/mandelbrot.cpp index 5d575d5b..c8dd5c33 100644 --- a/filmulator-gui/core/nlmeans/eigen/demos/mandelbrot/mandelbrot.cpp +++ b/filmulator-gui/core/nlmeans/eigen/demos/mandelbrot/mandelbrot.cpp @@ -8,102 +8,102 @@ // with this file, You can obtain one at http://mozilla.org/MPL/2.0/. #include "mandelbrot.h" +#include +#include +#include +#include #include -#include -#include -#include -#include void MandelbrotWidget::resizeEvent(QResizeEvent *) { - if(size < width() * height()) - { + if (size < width() * height()) { std::cout << "reallocate buffer" << std::endl; size = width() * height(); - if(buffer) delete[]buffer; - buffer = new unsigned char[4*size]; + if (buffer) delete[] buffer; + buffer = new unsigned char[4 * size]; } } -template struct iters_before_test { enum { ret = 8 }; }; -template<> struct iters_before_test { enum { ret = 16 }; }; +template struct iters_before_test +{ + enum { ret = 8 }; +}; +template<> struct iters_before_test +{ + enum { ret = 16 }; +}; template void MandelbrotThread::render(int img_width, int img_height) { - enum { packetSize = Eigen::internal::packet_traits::size }; // number of reals in a Packet - typedef Eigen::Array Packet; // wrap a Packet as a vector + enum { packetSize = Eigen::internal::packet_traits::size };// number of reals in a Packet + typedef Eigen::Array Packet;// wrap a Packet as a vector enum { iters_before_test = iters_before_test::ret }; max_iter = (max_iter / iters_before_test) * iters_before_test; - const int alignedWidth = (img_width/packetSize)*packetSize; + const int alignedWidth = (img_width / packetSize) * packetSize; unsigned char *const buffer = widget->buffer; const double xradius = widget->xradius; const double yradius = xradius * img_height / img_width; const int threadcount = widget->threadcount; typedef Eigen::Array Vector2; Vector2 start(widget->center.x() - widget->xradius, widget->center.y() - yradius); - Vector2 step(2*widget->xradius/img_width, 2*yradius/img_height); + Vector2 step(2 * widget->xradius / img_width, 2 * yradius / img_height); total_iter = 0; - for(int y = id; y < img_height; y += threadcount) - { + for (int y = id; y < img_height; y += threadcount) { int pix = y * img_width; - // for each pixel, we're going to do the iteration z := z^2 + c where z and c are complex numbers, + // for each pixel, we're going to do the iteration z := z^2 + c where z and c are complex numbers, // starting with z = c = complex coord of the pixel. pzi and pzr denote the real and imaginary parts of z. // pci and pcr denote the real and imaginary parts of c. Packet pzi_start, pci_start; - for(int i = 0; i < packetSize; i++) pzi_start[i] = pci_start[i] = start.y() + y * step.y(); + for (int i = 0; i < packetSize; i++) pzi_start[i] = pci_start[i] = start.y() + y * step.y(); - for(int x = 0; x < alignedWidth; x += packetSize, pix += packetSize) - { + for (int x = 0; x < alignedWidth; x += packetSize, pix += packetSize) { Packet pcr, pci = pci_start, pzr, pzi = pzi_start, pzr_buf; - for(int i = 0; i < packetSize; i++) pzr[i] = pcr[i] = start.x() + (x+i) * step.x(); + for (int i = 0; i < packetSize; i++) pzr[i] = pcr[i] = start.x() + (x + i) * step.x(); // do the iterations. Every iters_before_test iterations we check for divergence, // in which case we can stop iterating. int j = 0; typedef Eigen::Matrix Packeti; - Packeti pix_iter = Packeti::Zero(), // number of iteration per pixel in the packet - pix_dont_diverge; // whether or not each pixel has already diverged - do - { - for(int i = 0; i < iters_before_test/4; i++) // peel the inner loop by 4 + Packeti pix_iter = Packeti::Zero(),// number of iteration per pixel in the packet + pix_dont_diverge;// whether or not each pixel has already diverged + do { + for (int i = 0; i < iters_before_test / 4; i++)// peel the inner loop by 4 { -# define ITERATE \ - pzr_buf = pzr; \ - pzr = pzr.square(); \ - pzr -= pzi.square(); \ - pzr += pcr; \ - pzi = (2*pzr_buf)*pzi; \ - pzi += pci; +#define ITERATE \ + pzr_buf = pzr; \ + pzr = pzr.square(); \ + pzr -= pzi.square(); \ + pzr += pcr; \ + pzi = (2 * pzr_buf) * pzi; \ + pzi += pci; ITERATE ITERATE ITERATE ITERATE } - pix_dont_diverge = ((pzr.square() + pzi.square()) - .eval() // temporary fix as what follows is not yet vectorized by Eigen - <= Packet::Constant(4)) - // the 4 here is not a magic value, it's a math fact that if - // the square modulus is >4 then divergence is inevitable. - .template cast(); + pix_dont_diverge = + ((pzr.square() + pzi.square()).eval()// temporary fix as what follows is not yet vectorized by Eigen + <= Packet::Constant(4)) + // the 4 here is not a magic value, it's a math fact that if + // the square modulus is >4 then divergence is inevitable. + .template cast(); pix_iter += iters_before_test * pix_dont_diverge; j++; total_iter += iters_before_test * packetSize; - } - while(j < max_iter/iters_before_test && pix_dont_diverge.any()); // any() is not yet vectorized by Eigen + } while (j < max_iter / iters_before_test && pix_dont_diverge.any());// any() is not yet vectorized by Eigen // compute pixel colors - for(int i = 0; i < packetSize; i++) - { - buffer[4*(pix+i)] = 255*pix_iter[i]/max_iter; - buffer[4*(pix+i)+1] = 0; - buffer[4*(pix+i)+2] = 0; + for (int i = 0; i < packetSize; i++) { + buffer[4 * (pix + i)] = 255 * pix_iter[i] / max_iter; + buffer[4 * (pix + i) + 1] = 0; + buffer[4 * (pix + i) + 2] = 0; } } // if the width is not a multiple of packetSize, fill the remainder in black - for(int x = alignedWidth; x < img_width; x++, pix++) - buffer[4*pix] = buffer[4*pix+1] = buffer[4*pix+2] = 0; + for (int x = alignedWidth; x < img_width; x++, pix++) + buffer[4 * pix] = buffer[4 * pix + 1] = buffer[4 * pix + 2] = 0; } return; } @@ -111,14 +111,14 @@ template void MandelbrotThread::render(int img_width, int img_hei void MandelbrotThread::run() { setTerminationEnabled(true); - double resolution = widget->xradius*2/widget->width(); + double resolution = widget->xradius * 2 / widget->width(); max_iter = 128; - if(resolution < 1e-4f) max_iter += 128 * ( - 4 - std::log10(resolution)); - int img_width = widget->width()/widget->draft; - int img_height = widget->height()/widget->draft; + if (resolution < 1e-4f) max_iter += 128 * (-4 - std::log10(resolution)); + int img_width = widget->width() / widget->draft; + int img_height = widget->height() / widget->draft; single_precision = resolution > 1e-7f; - if(single_precision) + if (single_precision) render(img_width, img_height); else render(img_width, img_height); @@ -131,39 +131,32 @@ void MandelbrotWidget::paintEvent(QPaintEvent *) QTime time; time.start(); - for(int th = 0; th < threadcount; th++) - threads[th]->start(QThread::LowPriority); - for(int th = 0; th < threadcount; th++) - { + for (int th = 0; th < threadcount; th++) threads[th]->start(QThread::LowPriority); + for (int th = 0; th < threadcount; th++) { threads[th]->wait(); total_iter += threads[th]->total_iter; } int elapsed = time.elapsed(); - if(draft == 1) - { - float speed = elapsed ? float(total_iter)*1000/elapsed : 0; + if (draft == 1) { + float speed = elapsed ? float(total_iter) * 1000 / elapsed : 0; max_speed = std::max(max_speed, speed); - std::cout << threadcount << " threads, " - << elapsed << " ms, " - << speed << " iters/s (max " << max_speed << ")" << std::endl; - int packetSize = threads[0]->single_precision - ? int(Eigen::internal::packet_traits::size) - : int(Eigen::internal::packet_traits::size); - setWindowTitle(QString("resolution ")+QString::number(xradius*2/width(), 'e', 2) - +QString(", %1 iterations per pixel, ").arg(threads[0]->max_iter) - +(threads[0]->single_precision ? QString("single ") : QString("double ")) - +QString("precision, ") - +(packetSize==1 ? QString("no vectorization") - : QString("vectorized (%1 per packet)").arg(packetSize))); + std::cout << threadcount << " threads, " << elapsed << " ms, " << speed << " iters/s (max " << max_speed << ")" + << std::endl; + int packetSize = threads[0]->single_precision ? int(Eigen::internal::packet_traits::size) + : int(Eigen::internal::packet_traits::size); + setWindowTitle( + QString("resolution ") + QString::number(xradius * 2 / width(), 'e', 2) + + QString(", %1 iterations per pixel, ").arg(threads[0]->max_iter) + + (threads[0]->single_precision ? QString("single ") : QString("double ")) + QString("precision, ") + + (packetSize == 1 ? QString("no vectorization") : QString("vectorized (%1 per packet)").arg(packetSize))); } - - QImage image(buffer, width()/draft, height()/draft, QImage::Format_RGB32); + + QImage image(buffer, width() / draft, height() / draft, QImage::Format_RGB32); QPainter painter(this); painter.drawImage(QPoint(0, 0), image.scaled(width(), height())); - if(draft>1) - { + if (draft > 1) { draft /= 2; setWindowTitle(QString("recomputing at 1/%1 resolution...").arg(draft)); update(); @@ -172,15 +165,13 @@ void MandelbrotWidget::paintEvent(QPaintEvent *) void MandelbrotWidget::mousePressEvent(QMouseEvent *event) { - if( event->buttons() & Qt::LeftButton ) - { + if (event->buttons() & Qt::LeftButton) { lastpos = event->pos(); double yradius = xradius * height() / width(); - center = Eigen::Vector2d(center.x() + (event->pos().x() - width()/2) * xradius * 2 / width(), - center.y() + (event->pos().y() - height()/2) * yradius * 2 / height()); + center = Eigen::Vector2d(center.x() + (event->pos().x() - width() / 2) * xradius * 2 / width(), + center.y() + (event->pos().y() - height() / 2) * yradius * 2 / height()); draft = 16; - for(int th = 0; th < threadcount; th++) - threads[th]->terminate(); + for (int th = 0; th < threadcount; th++) threads[th]->terminate(); update(); } } @@ -189,15 +180,13 @@ void MandelbrotWidget::mouseMoveEvent(QMouseEvent *event) { QPoint delta = event->pos() - lastpos; lastpos = event->pos(); - if( event->buttons() & Qt::LeftButton ) - { + if (event->buttons() & Qt::LeftButton) { double t = 1 + 5 * double(delta.y()) / height(); - if(t < 0.5) t = 0.5; - if(t > 2) t = 2; + if (t < 0.5) t = 0.5; + if (t > 2) t = 2; xradius *= t; draft = 16; - for(int th = 0; th < threadcount; th++) - threads[th]->terminate(); + for (int th = 0; th < threadcount; th++) threads[th]->terminate(); update(); } } diff --git a/filmulator-gui/core/nlmeans/eigen/demos/mandelbrot/mandelbrot.h b/filmulator-gui/core/nlmeans/eigen/demos/mandelbrot/mandelbrot.h index a687fd01..fdf21cac 100644 --- a/filmulator-gui/core/nlmeans/eigen/demos/mandelbrot/mandelbrot.h +++ b/filmulator-gui/core/nlmeans/eigen/demos/mandelbrot/mandelbrot.h @@ -11,61 +11,60 @@ #define MANDELBROT_H #include +#include #include #include -#include class MandelbrotWidget; class MandelbrotThread : public QThread { - friend class MandelbrotWidget; - MandelbrotWidget *widget; - long long total_iter; - int id, max_iter; - bool single_precision; + friend class MandelbrotWidget; + MandelbrotWidget *widget; + long long total_iter; + int id, max_iter; + bool single_precision; - public: - MandelbrotThread(MandelbrotWidget *w, int i) : widget(w), id(i) {} - void run(); - template void render(int img_width, int img_height); +public: + MandelbrotThread(MandelbrotWidget *w, int i) : widget(w), id(i) {} + void run(); + template void render(int img_width, int img_height); }; class MandelbrotWidget : public QWidget { - Q_OBJECT + Q_OBJECT - friend class MandelbrotThread; - Eigen::Vector2d center; - double xradius; - int size; - unsigned char *buffer; - QPoint lastpos; - int draft; - MandelbrotThread **threads; - int threadcount; + friend class MandelbrotThread; + Eigen::Vector2d center; + double xradius; + int size; + unsigned char *buffer; + QPoint lastpos; + int draft; + MandelbrotThread **threads; + int threadcount; - protected: - void resizeEvent(QResizeEvent *); - void paintEvent(QPaintEvent *); - void mousePressEvent(QMouseEvent *event); - void mouseMoveEvent(QMouseEvent *event); +protected: + void resizeEvent(QResizeEvent *); + void paintEvent(QPaintEvent *); + void mousePressEvent(QMouseEvent *event); + void mouseMoveEvent(QMouseEvent *event); - public: - MandelbrotWidget() : QWidget(), center(0,0), xradius(2), - size(0), buffer(0), draft(16) - { - setAutoFillBackground(false); - threadcount = QThread::idealThreadCount(); - threads = new MandelbrotThread*[threadcount]; - for(int th = 0; th < threadcount; th++) threads[th] = new MandelbrotThread(this, th); - } - ~MandelbrotWidget() - { - if(buffer) delete[]buffer; - for(int th = 0; th < threadcount; th++) delete threads[th]; - delete[] threads; - } +public: + MandelbrotWidget() : QWidget(), center(0, 0), xradius(2), size(0), buffer(0), draft(16) + { + setAutoFillBackground(false); + threadcount = QThread::idealThreadCount(); + threads = new MandelbrotThread *[threadcount]; + for (int th = 0; th < threadcount; th++) threads[th] = new MandelbrotThread(this, th); + } + ~MandelbrotWidget() + { + if (buffer) delete[] buffer; + for (int th = 0; th < threadcount; th++) delete threads[th]; + delete[] threads; + } }; -#endif // MANDELBROT_H +#endif// MANDELBROT_H diff --git a/filmulator-gui/core/nlmeans/eigen/demos/mix_eigen_and_c/binary_library.cpp b/filmulator-gui/core/nlmeans/eigen/demos/mix_eigen_and_c/binary_library.cpp index 15a2d03e..a45904f4 100644 --- a/filmulator-gui/core/nlmeans/eigen/demos/mix_eigen_and_c/binary_library.cpp +++ b/filmulator-gui/core/nlmeans/eigen/demos/mix_eigen_and_c/binary_library.cpp @@ -20,46 +20,28 @@ using namespace Eigen; ////// class MatrixXd ////// -inline MatrixXd& c_to_eigen(C_MatrixXd* ptr) -{ - return *reinterpret_cast(ptr); -} +inline MatrixXd &c_to_eigen(C_MatrixXd *ptr) { return *reinterpret_cast(ptr); } -inline const MatrixXd& c_to_eigen(const C_MatrixXd* ptr) -{ - return *reinterpret_cast(ptr); -} +inline const MatrixXd &c_to_eigen(const C_MatrixXd *ptr) { return *reinterpret_cast(ptr); } -inline C_MatrixXd* eigen_to_c(MatrixXd& ref) -{ - return reinterpret_cast(&ref); -} +inline C_MatrixXd *eigen_to_c(MatrixXd &ref) { return reinterpret_cast(&ref); } -inline const C_MatrixXd* eigen_to_c(const MatrixXd& ref) -{ - return reinterpret_cast(&ref); -} +inline const C_MatrixXd *eigen_to_c(const MatrixXd &ref) { return reinterpret_cast(&ref); } ////// class Map ////// -inline Map& c_to_eigen(C_Map_MatrixXd* ptr) -{ - return *reinterpret_cast*>(ptr); -} +inline Map &c_to_eigen(C_Map_MatrixXd *ptr) { return *reinterpret_cast *>(ptr); } -inline const Map& c_to_eigen(const C_Map_MatrixXd* ptr) +inline const Map &c_to_eigen(const C_Map_MatrixXd *ptr) { - return *reinterpret_cast*>(ptr); + return *reinterpret_cast *>(ptr); } -inline C_Map_MatrixXd* eigen_to_c(Map& ref) -{ - return reinterpret_cast(&ref); -} +inline C_Map_MatrixXd *eigen_to_c(Map &ref) { return reinterpret_cast(&ref); } -inline const C_Map_MatrixXd* eigen_to_c(const Map& ref) +inline const C_Map_MatrixXd *eigen_to_c(const Map &ref) { - return reinterpret_cast(&ref); + return reinterpret_cast(&ref); } @@ -69,55 +51,25 @@ inline const C_Map_MatrixXd* eigen_to_c(const Map& ref) ////// class MatrixXd ////// -C_MatrixXd* MatrixXd_new(int rows, int cols) -{ - return eigen_to_c(*new MatrixXd(rows,cols)); -} +C_MatrixXd *MatrixXd_new(int rows, int cols) { return eigen_to_c(*new MatrixXd(rows, cols)); } -void MatrixXd_delete(C_MatrixXd *m) -{ - delete &c_to_eigen(m); -} +void MatrixXd_delete(C_MatrixXd *m) { delete &c_to_eigen(m); } -double* MatrixXd_data(C_MatrixXd *m) -{ - return c_to_eigen(m).data(); -} +double *MatrixXd_data(C_MatrixXd *m) { return c_to_eigen(m).data(); } -void MatrixXd_set_zero(C_MatrixXd *m) -{ - c_to_eigen(m).setZero(); -} +void MatrixXd_set_zero(C_MatrixXd *m) { c_to_eigen(m).setZero(); } -void MatrixXd_resize(C_MatrixXd *m, int rows, int cols) -{ - c_to_eigen(m).resize(rows,cols); -} +void MatrixXd_resize(C_MatrixXd *m, int rows, int cols) { c_to_eigen(m).resize(rows, cols); } -void MatrixXd_copy(C_MatrixXd *dst, const C_MatrixXd *src) -{ - c_to_eigen(dst) = c_to_eigen(src); -} +void MatrixXd_copy(C_MatrixXd *dst, const C_MatrixXd *src) { c_to_eigen(dst) = c_to_eigen(src); } -void MatrixXd_copy_map(C_MatrixXd *dst, const C_Map_MatrixXd *src) -{ - c_to_eigen(dst) = c_to_eigen(src); -} +void MatrixXd_copy_map(C_MatrixXd *dst, const C_Map_MatrixXd *src) { c_to_eigen(dst) = c_to_eigen(src); } -void MatrixXd_set_coeff(C_MatrixXd *m, int i, int j, double coeff) -{ - c_to_eigen(m)(i,j) = coeff; -} +void MatrixXd_set_coeff(C_MatrixXd *m, int i, int j, double coeff) { c_to_eigen(m)(i, j) = coeff; } -double MatrixXd_get_coeff(const C_MatrixXd *m, int i, int j) -{ - return c_to_eigen(m)(i,j); -} +double MatrixXd_get_coeff(const C_MatrixXd *m, int i, int j) { return c_to_eigen(m)(i, j); } -void MatrixXd_print(const C_MatrixXd *m) -{ - std::cout << c_to_eigen(m) << std::endl; -} +void MatrixXd_print(const C_MatrixXd *m) { std::cout << c_to_eigen(m) << std::endl; } void MatrixXd_multiply(const C_MatrixXd *m1, const C_MatrixXd *m2, C_MatrixXd *result) { @@ -130,49 +82,27 @@ void MatrixXd_add(const C_MatrixXd *m1, const C_MatrixXd *m2, C_MatrixXd *result } - ////// class Map_MatrixXd ////// -C_Map_MatrixXd* Map_MatrixXd_new(double *array, int rows, int cols) +C_Map_MatrixXd *Map_MatrixXd_new(double *array, int rows, int cols) { - return eigen_to_c(*new Map(array,rows,cols)); + return eigen_to_c(*new Map(array, rows, cols)); } -void Map_MatrixXd_delete(C_Map_MatrixXd *m) -{ - delete &c_to_eigen(m); -} +void Map_MatrixXd_delete(C_Map_MatrixXd *m) { delete &c_to_eigen(m); } -void Map_MatrixXd_set_zero(C_Map_MatrixXd *m) -{ - c_to_eigen(m).setZero(); -} +void Map_MatrixXd_set_zero(C_Map_MatrixXd *m) { c_to_eigen(m).setZero(); } -void Map_MatrixXd_copy(C_Map_MatrixXd *dst, const C_Map_MatrixXd *src) -{ - c_to_eigen(dst) = c_to_eigen(src); -} +void Map_MatrixXd_copy(C_Map_MatrixXd *dst, const C_Map_MatrixXd *src) { c_to_eigen(dst) = c_to_eigen(src); } -void Map_MatrixXd_copy_matrix(C_Map_MatrixXd *dst, const C_MatrixXd *src) -{ - c_to_eigen(dst) = c_to_eigen(src); -} +void Map_MatrixXd_copy_matrix(C_Map_MatrixXd *dst, const C_MatrixXd *src) { c_to_eigen(dst) = c_to_eigen(src); } -void Map_MatrixXd_set_coeff(C_Map_MatrixXd *m, int i, int j, double coeff) -{ - c_to_eigen(m)(i,j) = coeff; -} +void Map_MatrixXd_set_coeff(C_Map_MatrixXd *m, int i, int j, double coeff) { c_to_eigen(m)(i, j) = coeff; } -double Map_MatrixXd_get_coeff(const C_Map_MatrixXd *m, int i, int j) -{ - return c_to_eigen(m)(i,j); -} +double Map_MatrixXd_get_coeff(const C_Map_MatrixXd *m, int i, int j) { return c_to_eigen(m)(i, j); } -void Map_MatrixXd_print(const C_Map_MatrixXd *m) -{ - std::cout << c_to_eigen(m) << std::endl; -} +void Map_MatrixXd_print(const C_Map_MatrixXd *m) { std::cout << c_to_eigen(m) << std::endl; } void Map_MatrixXd_multiply(const C_Map_MatrixXd *m1, const C_Map_MatrixXd *m2, C_Map_MatrixXd *result) { diff --git a/filmulator-gui/core/nlmeans/eigen/demos/mix_eigen_and_c/binary_library.h b/filmulator-gui/core/nlmeans/eigen/demos/mix_eigen_and_c/binary_library.h index 0b983ad3..6b63c45b 100644 --- a/filmulator-gui/core/nlmeans/eigen/demos/mix_eigen_and_c/binary_library.h +++ b/filmulator-gui/core/nlmeans/eigen/demos/mix_eigen_and_c/binary_library.h @@ -13,59 +13,48 @@ // they will be compiled to C object code. #ifdef __cplusplus -extern "C" -{ +extern "C" { #endif - // just dummy empty structs to give different pointer types, - // instead of using void* which would be type unsafe - struct C_MatrixXd {}; - struct C_Map_MatrixXd {}; +// just dummy empty structs to give different pointer types, +// instead of using void* which would be type unsafe +struct C_MatrixXd +{ +}; +struct C_Map_MatrixXd +{ +}; + +// the C_MatrixXd class, wraps some of the functionality +// of Eigen::MatrixXd. +struct C_MatrixXd *MatrixXd_new(int rows, int cols); +void MatrixXd_delete(struct C_MatrixXd *m); +double *MatrixXd_data(struct C_MatrixXd *m); +void MatrixXd_set_zero(struct C_MatrixXd *m); +void MatrixXd_resize(struct C_MatrixXd *m, int rows, int cols); +void MatrixXd_copy(struct C_MatrixXd *dst, const struct C_MatrixXd *src); +void MatrixXd_copy_map(struct C_MatrixXd *dst, const struct C_Map_MatrixXd *src); +void MatrixXd_set_coeff(struct C_MatrixXd *m, int i, int j, double coeff); +double MatrixXd_get_coeff(const struct C_MatrixXd *m, int i, int j); +void MatrixXd_print(const struct C_MatrixXd *m); +void MatrixXd_add(const struct C_MatrixXd *m1, const struct C_MatrixXd *m2, struct C_MatrixXd *result); +void MatrixXd_multiply(const struct C_MatrixXd *m1, const struct C_MatrixXd *m2, struct C_MatrixXd *result); - // the C_MatrixXd class, wraps some of the functionality - // of Eigen::MatrixXd. - struct C_MatrixXd* MatrixXd_new(int rows, int cols); - void MatrixXd_delete (struct C_MatrixXd *m); - double* MatrixXd_data (struct C_MatrixXd *m); - void MatrixXd_set_zero (struct C_MatrixXd *m); - void MatrixXd_resize (struct C_MatrixXd *m, int rows, int cols); - void MatrixXd_copy (struct C_MatrixXd *dst, - const struct C_MatrixXd *src); - void MatrixXd_copy_map (struct C_MatrixXd *dst, - const struct C_Map_MatrixXd *src); - void MatrixXd_set_coeff (struct C_MatrixXd *m, - int i, int j, double coeff); - double MatrixXd_get_coeff (const struct C_MatrixXd *m, - int i, int j); - void MatrixXd_print (const struct C_MatrixXd *m); - void MatrixXd_add (const struct C_MatrixXd *m1, - const struct C_MatrixXd *m2, - struct C_MatrixXd *result); - void MatrixXd_multiply (const struct C_MatrixXd *m1, - const struct C_MatrixXd *m2, - struct C_MatrixXd *result); - - // the C_Map_MatrixXd class, wraps some of the functionality - // of Eigen::Map - struct C_Map_MatrixXd* Map_MatrixXd_new(double *array, int rows, int cols); - void Map_MatrixXd_delete (struct C_Map_MatrixXd *m); - void Map_MatrixXd_set_zero (struct C_Map_MatrixXd *m); - void Map_MatrixXd_copy (struct C_Map_MatrixXd *dst, - const struct C_Map_MatrixXd *src); - void Map_MatrixXd_copy_matrix(struct C_Map_MatrixXd *dst, - const struct C_MatrixXd *src); - void Map_MatrixXd_set_coeff (struct C_Map_MatrixXd *m, - int i, int j, double coeff); - double Map_MatrixXd_get_coeff (const struct C_Map_MatrixXd *m, - int i, int j); - void Map_MatrixXd_print (const struct C_Map_MatrixXd *m); - void Map_MatrixXd_add (const struct C_Map_MatrixXd *m1, - const struct C_Map_MatrixXd *m2, - struct C_Map_MatrixXd *result); - void Map_MatrixXd_multiply (const struct C_Map_MatrixXd *m1, - const struct C_Map_MatrixXd *m2, - struct C_Map_MatrixXd *result); +// the C_Map_MatrixXd class, wraps some of the functionality +// of Eigen::Map +struct C_Map_MatrixXd *Map_MatrixXd_new(double *array, int rows, int cols); +void Map_MatrixXd_delete(struct C_Map_MatrixXd *m); +void Map_MatrixXd_set_zero(struct C_Map_MatrixXd *m); +void Map_MatrixXd_copy(struct C_Map_MatrixXd *dst, const struct C_Map_MatrixXd *src); +void Map_MatrixXd_copy_matrix(struct C_Map_MatrixXd *dst, const struct C_MatrixXd *src); +void Map_MatrixXd_set_coeff(struct C_Map_MatrixXd *m, int i, int j, double coeff); +double Map_MatrixXd_get_coeff(const struct C_Map_MatrixXd *m, int i, int j); +void Map_MatrixXd_print(const struct C_Map_MatrixXd *m); +void Map_MatrixXd_add(const struct C_Map_MatrixXd *m1, const struct C_Map_MatrixXd *m2, struct C_Map_MatrixXd *result); +void Map_MatrixXd_multiply(const struct C_Map_MatrixXd *m1, + const struct C_Map_MatrixXd *m2, + struct C_Map_MatrixXd *result); #ifdef __cplusplus -} // end extern "C" +}// end extern "C" #endif \ No newline at end of file diff --git a/filmulator-gui/core/nlmeans/eigen/demos/opengl/camera.cpp b/filmulator-gui/core/nlmeans/eigen/demos/opengl/camera.cpp index 8a2344c8..8c200d79 100644 --- a/filmulator-gui/core/nlmeans/eigen/demos/opengl/camera.cpp +++ b/filmulator-gui/core/nlmeans/eigen/demos/opengl/camera.cpp @@ -15,194 +15,173 @@ #include "Eigen/LU" using namespace Eigen; -Camera::Camera() - : mViewIsUptodate(false), mProjIsUptodate(false) +Camera::Camera() : mViewIsUptodate(false), mProjIsUptodate(false) { - mViewMatrix.setIdentity(); - - mFovY = M_PI/3.; - mNearDist = 1.; - mFarDist = 50000.; - - mVpX = 0; - mVpY = 0; - - setPosition(Vector3f::Constant(100.)); - setTarget(Vector3f::Zero()); -} + mViewMatrix.setIdentity(); -Camera& Camera::operator=(const Camera& other) -{ - mViewIsUptodate = false; - mProjIsUptodate = false; - - mVpX = other.mVpX; - mVpY = other.mVpY; - mVpWidth = other.mVpWidth; - mVpHeight = other.mVpHeight; - - mTarget = other.mTarget; - mFovY = other.mFovY; - mNearDist = other.mNearDist; - mFarDist = other.mFarDist; - - mViewMatrix = other.mViewMatrix; - mProjectionMatrix = other.mProjectionMatrix; - - return *this; -} + mFovY = M_PI / 3.; + mNearDist = 1.; + mFarDist = 50000.; -Camera::Camera(const Camera& other) -{ - *this = other; + mVpX = 0; + mVpY = 0; + + setPosition(Vector3f::Constant(100.)); + setTarget(Vector3f::Zero()); } -Camera::~Camera() +Camera &Camera::operator=(const Camera &other) { + mViewIsUptodate = false; + mProjIsUptodate = false; + + mVpX = other.mVpX; + mVpY = other.mVpY; + mVpWidth = other.mVpWidth; + mVpHeight = other.mVpHeight; + + mTarget = other.mTarget; + mFovY = other.mFovY; + mNearDist = other.mNearDist; + mFarDist = other.mFarDist; + + mViewMatrix = other.mViewMatrix; + mProjectionMatrix = other.mProjectionMatrix; + + return *this; } +Camera::Camera(const Camera &other) { *this = other; } + +Camera::~Camera() {} + void Camera::setViewport(uint offsetx, uint offsety, uint width, uint height) { - mVpX = offsetx; - mVpY = offsety; - mVpWidth = width; - mVpHeight = height; - - mProjIsUptodate = false; + mVpX = offsetx; + mVpY = offsety; + mVpWidth = width; + mVpHeight = height; + + mProjIsUptodate = false; } void Camera::setViewport(uint width, uint height) { - mVpWidth = width; - mVpHeight = height; - - mProjIsUptodate = false; + mVpWidth = width; + mVpHeight = height; + + mProjIsUptodate = false; } void Camera::setFovY(float value) { - mFovY = value; - mProjIsUptodate = false; + mFovY = value; + mProjIsUptodate = false; } -Vector3f Camera::direction(void) const -{ - return - (orientation() * Vector3f::UnitZ()); -} -Vector3f Camera::up(void) const -{ - return orientation() * Vector3f::UnitY(); -} -Vector3f Camera::right(void) const -{ - return orientation() * Vector3f::UnitX(); -} +Vector3f Camera::direction(void) const { return -(orientation() * Vector3f::UnitZ()); } +Vector3f Camera::up(void) const { return orientation() * Vector3f::UnitY(); } +Vector3f Camera::right(void) const { return orientation() * Vector3f::UnitX(); } -void Camera::setDirection(const Vector3f& newDirection) +void Camera::setDirection(const Vector3f &newDirection) { - // TODO implement it computing the rotation between newDirection and current dir ? - Vector3f up = this->up(); - - Matrix3f camAxes; - - camAxes.col(2) = (-newDirection).normalized(); - camAxes.col(0) = up.cross( camAxes.col(2) ).normalized(); - camAxes.col(1) = camAxes.col(2).cross( camAxes.col(0) ).normalized(); - setOrientation(Quaternionf(camAxes)); - - mViewIsUptodate = false; + // TODO implement it computing the rotation between newDirection and current dir ? + Vector3f up = this->up(); + + Matrix3f camAxes; + + camAxes.col(2) = (-newDirection).normalized(); + camAxes.col(0) = up.cross(camAxes.col(2)).normalized(); + camAxes.col(1) = camAxes.col(2).cross(camAxes.col(0)).normalized(); + setOrientation(Quaternionf(camAxes)); + + mViewIsUptodate = false; } -void Camera::setTarget(const Vector3f& target) +void Camera::setTarget(const Vector3f &target) { - mTarget = target; - if (!mTarget.isApprox(position())) - { - Vector3f newDirection = mTarget - position(); - setDirection(newDirection.normalized()); - } + mTarget = target; + if (!mTarget.isApprox(position())) { + Vector3f newDirection = mTarget - position(); + setDirection(newDirection.normalized()); + } } -void Camera::setPosition(const Vector3f& p) +void Camera::setPosition(const Vector3f &p) { - mFrame.position = p; - mViewIsUptodate = false; + mFrame.position = p; + mViewIsUptodate = false; } -void Camera::setOrientation(const Quaternionf& q) +void Camera::setOrientation(const Quaternionf &q) { - mFrame.orientation = q; - mViewIsUptodate = false; + mFrame.orientation = q; + mViewIsUptodate = false; } -void Camera::setFrame(const Frame& f) +void Camera::setFrame(const Frame &f) { mFrame = f; mViewIsUptodate = false; } -void Camera::rotateAroundTarget(const Quaternionf& q) +void Camera::rotateAroundTarget(const Quaternionf &q) { - Matrix4f mrot, mt, mtm; - - // update the transform matrix - updateViewMatrix(); - Vector3f t = mViewMatrix * mTarget; - - mViewMatrix = Translation3f(t) - * q - * Translation3f(-t) - * mViewMatrix; - - Quaternionf qa(mViewMatrix.linear()); - qa = qa.conjugate(); - setOrientation(qa); - setPosition(- (qa * mViewMatrix.translation()) ); + Matrix4f mrot, mt, mtm; - mViewIsUptodate = true; + // update the transform matrix + updateViewMatrix(); + Vector3f t = mViewMatrix * mTarget; + + mViewMatrix = Translation3f(t) * q * Translation3f(-t) * mViewMatrix; + + Quaternionf qa(mViewMatrix.linear()); + qa = qa.conjugate(); + setOrientation(qa); + setPosition(-(qa * mViewMatrix.translation())); + + mViewIsUptodate = true; } -void Camera::localRotate(const Quaternionf& q) +void Camera::localRotate(const Quaternionf &q) { - float dist = (position() - mTarget).norm(); - setOrientation(orientation() * q); - mTarget = position() + dist * direction(); - mViewIsUptodate = false; + float dist = (position() - mTarget).norm(); + setOrientation(orientation() * q); + mTarget = position() + dist * direction(); + mViewIsUptodate = false; } void Camera::zoom(float d) { - float dist = (position() - mTarget).norm(); - if(dist > d) - { - setPosition(position() + direction() * d); - mViewIsUptodate = false; - } + float dist = (position() - mTarget).norm(); + if (dist > d) { + setPosition(position() + direction() * d); + mViewIsUptodate = false; + } } -void Camera::localTranslate(const Vector3f& t) +void Camera::localTranslate(const Vector3f &t) { Vector3f trans = orientation() * t; - setPosition( position() + trans ); - setTarget( mTarget + trans ); + setPosition(position() + trans); + setTarget(mTarget + trans); mViewIsUptodate = false; } void Camera::updateViewMatrix(void) const { - if(!mViewIsUptodate) - { - Quaternionf q = orientation().conjugate(); - mViewMatrix.linear() = q.toRotationMatrix(); - mViewMatrix.translation() = - (mViewMatrix.linear() * position()); - - mViewIsUptodate = true; - } + if (!mViewIsUptodate) { + Quaternionf q = orientation().conjugate(); + mViewMatrix.linear() = q.toRotationMatrix(); + mViewMatrix.translation() = -(mViewMatrix.linear() * position()); + + mViewIsUptodate = true; + } } -const Affine3f& Camera::viewMatrix(void) const +const Affine3f &Camera::viewMatrix(void) const { updateViewMatrix(); return mViewMatrix; @@ -210,26 +189,25 @@ const Affine3f& Camera::viewMatrix(void) const void Camera::updateProjectionMatrix(void) const { - if(!mProjIsUptodate) - { + if (!mProjIsUptodate) { mProjectionMatrix.setIdentity(); - float aspect = float(mVpWidth)/float(mVpHeight); - float theta = mFovY*0.5; + float aspect = float(mVpWidth) / float(mVpHeight); + float theta = mFovY * 0.5; float range = mFarDist - mNearDist; - float invtan = 1./tan(theta); - - mProjectionMatrix(0,0) = invtan / aspect; - mProjectionMatrix(1,1) = invtan; - mProjectionMatrix(2,2) = -(mNearDist + mFarDist) / range; - mProjectionMatrix(3,2) = -1; - mProjectionMatrix(2,3) = -2 * mNearDist * mFarDist / range; - mProjectionMatrix(3,3) = 0; - + float invtan = 1. / tan(theta); + + mProjectionMatrix(0, 0) = invtan / aspect; + mProjectionMatrix(1, 1) = invtan; + mProjectionMatrix(2, 2) = -(mNearDist + mFarDist) / range; + mProjectionMatrix(3, 2) = -1; + mProjectionMatrix(2, 3) = -2 * mNearDist * mFarDist / range; + mProjectionMatrix(3, 3) = 0; + mProjIsUptodate = true; } } -const Matrix4f& Camera::projectionMatrix(void) const +const Matrix4f &Camera::projectionMatrix(void) const { updateProjectionMatrix(); return mProjectionMatrix; @@ -238,27 +216,27 @@ const Matrix4f& Camera::projectionMatrix(void) const void Camera::activateGL(void) { glViewport(vpX(), vpY(), vpWidth(), vpHeight()); - gpu.loadMatrix(projectionMatrix(),GL_PROJECTION); - gpu.loadMatrix(viewMatrix().matrix(),GL_MODELVIEW); + gpu.loadMatrix(projectionMatrix(), GL_PROJECTION); + gpu.loadMatrix(viewMatrix().matrix(), GL_MODELVIEW); } -Vector3f Camera::unProject(const Vector2f& uv, float depth) const +Vector3f Camera::unProject(const Vector2f &uv, float depth) const { - Matrix4f inv = mViewMatrix.inverse().matrix(); - return unProject(uv, depth, inv); + Matrix4f inv = mViewMatrix.inverse().matrix(); + return unProject(uv, depth, inv); } -Vector3f Camera::unProject(const Vector2f& uv, float depth, const Matrix4f& invModelview) const +Vector3f Camera::unProject(const Vector2f &uv, float depth, const Matrix4f &invModelview) const { - updateViewMatrix(); - updateProjectionMatrix(); - - Vector3f a(2.*uv.x()/float(mVpWidth)-1., 2.*uv.y()/float(mVpHeight)-1., 1.); - a.x() *= depth/mProjectionMatrix(0,0); - a.y() *= depth/mProjectionMatrix(1,1); - a.z() = -depth; - // FIXME /\/| - Vector4f b = invModelview * Vector4f(a.x(), a.y(), a.z(), 1.); - return Vector3f(b.x(), b.y(), b.z()); + updateViewMatrix(); + updateProjectionMatrix(); + + Vector3f a(2. * uv.x() / float(mVpWidth) - 1., 2. * uv.y() / float(mVpHeight) - 1., 1.); + a.x() *= depth / mProjectionMatrix(0, 0); + a.y() *= depth / mProjectionMatrix(1, 1); + a.z() = -depth; + // FIXME /\/| + Vector4f b = invModelview * Vector4f(a.x(), a.y(), a.z(), 1.); + return Vector3f(b.x(), b.y(), b.z()); } diff --git a/filmulator-gui/core/nlmeans/eigen/demos/opengl/camera.h b/filmulator-gui/core/nlmeans/eigen/demos/opengl/camera.h index 15714d2e..50cf78e3 100644 --- a/filmulator-gui/core/nlmeans/eigen/demos/opengl/camera.h +++ b/filmulator-gui/core/nlmeans/eigen/demos/opengl/camera.h @@ -16,103 +16,100 @@ class Frame { - public: - EIGEN_MAKE_ALIGNED_OPERATOR_NEW - - inline Frame(const Eigen::Vector3f& pos = Eigen::Vector3f::Zero(), - const Eigen::Quaternionf& o = Eigen::Quaternionf()) - : orientation(o), position(pos) - {} - Frame lerp(float alpha, const Frame& other) const - { - return Frame((1.f-alpha)*position + alpha * other.position, - orientation.slerp(alpha,other.orientation)); - } - - Eigen::Quaternionf orientation; - Eigen::Vector3f position; +public: + EIGEN_MAKE_ALIGNED_OPERATOR_NEW + + inline Frame(const Eigen::Vector3f &pos = Eigen::Vector3f::Zero(), const Eigen::Quaternionf &o = Eigen::Quaternionf()) + : orientation(o), position(pos) + {} + Frame lerp(float alpha, const Frame &other) const + { + return Frame((1.f - alpha) * position + alpha * other.position, orientation.slerp(alpha, other.orientation)); + } + + Eigen::Quaternionf orientation; + Eigen::Vector3f position; }; class Camera { - public: - EIGEN_MAKE_ALIGNED_OPERATOR_NEW - - Camera(void); - - Camera(const Camera& other); - - virtual ~Camera(); - - Camera& operator=(const Camera& other); - - void setViewport(uint offsetx, uint offsety, uint width, uint height); - void setViewport(uint width, uint height); - - inline uint vpX(void) const { return mVpX; } - inline uint vpY(void) const { return mVpY; } - inline uint vpWidth(void) const { return mVpWidth; } - inline uint vpHeight(void) const { return mVpHeight; } - - inline float fovY(void) const { return mFovY; } - void setFovY(float value); - - void setPosition(const Eigen::Vector3f& pos); - inline const Eigen::Vector3f& position(void) const { return mFrame.position; } - - void setOrientation(const Eigen::Quaternionf& q); - inline const Eigen::Quaternionf& orientation(void) const { return mFrame.orientation; } - - void setFrame(const Frame& f); - const Frame& frame(void) const { return mFrame; } - - void setDirection(const Eigen::Vector3f& newDirection); - Eigen::Vector3f direction(void) const; - void setUp(const Eigen::Vector3f& vectorUp); - Eigen::Vector3f up(void) const; - Eigen::Vector3f right(void) const; - - void setTarget(const Eigen::Vector3f& target); - inline const Eigen::Vector3f& target(void) { return mTarget; } - - const Eigen::Affine3f& viewMatrix(void) const; - const Eigen::Matrix4f& projectionMatrix(void) const; - - void rotateAroundTarget(const Eigen::Quaternionf& q); - void localRotate(const Eigen::Quaternionf& q); - void zoom(float d); - - void localTranslate(const Eigen::Vector3f& t); - - /** Setup OpenGL matrices and viewport */ - void activateGL(void); - - Eigen::Vector3f unProject(const Eigen::Vector2f& uv, float depth, const Eigen::Matrix4f& invModelview) const; - Eigen::Vector3f unProject(const Eigen::Vector2f& uv, float depth) const; - - protected: - void updateViewMatrix(void) const; - void updateProjectionMatrix(void) const; - - protected: - - uint mVpX, mVpY; - uint mVpWidth, mVpHeight; - - Frame mFrame; - - mutable Eigen::Affine3f mViewMatrix; - mutable Eigen::Matrix4f mProjectionMatrix; - - mutable bool mViewIsUptodate; - mutable bool mProjIsUptodate; - - // used by rotateAroundTarget - Eigen::Vector3f mTarget; - - float mFovY; - float mNearDist; - float mFarDist; +public: + EIGEN_MAKE_ALIGNED_OPERATOR_NEW + + Camera(void); + + Camera(const Camera &other); + + virtual ~Camera(); + + Camera &operator=(const Camera &other); + + void setViewport(uint offsetx, uint offsety, uint width, uint height); + void setViewport(uint width, uint height); + + inline uint vpX(void) const { return mVpX; } + inline uint vpY(void) const { return mVpY; } + inline uint vpWidth(void) const { return mVpWidth; } + inline uint vpHeight(void) const { return mVpHeight; } + + inline float fovY(void) const { return mFovY; } + void setFovY(float value); + + void setPosition(const Eigen::Vector3f &pos); + inline const Eigen::Vector3f &position(void) const { return mFrame.position; } + + void setOrientation(const Eigen::Quaternionf &q); + inline const Eigen::Quaternionf &orientation(void) const { return mFrame.orientation; } + + void setFrame(const Frame &f); + const Frame &frame(void) const { return mFrame; } + + void setDirection(const Eigen::Vector3f &newDirection); + Eigen::Vector3f direction(void) const; + void setUp(const Eigen::Vector3f &vectorUp); + Eigen::Vector3f up(void) const; + Eigen::Vector3f right(void) const; + + void setTarget(const Eigen::Vector3f &target); + inline const Eigen::Vector3f &target(void) { return mTarget; } + + const Eigen::Affine3f &viewMatrix(void) const; + const Eigen::Matrix4f &projectionMatrix(void) const; + + void rotateAroundTarget(const Eigen::Quaternionf &q); + void localRotate(const Eigen::Quaternionf &q); + void zoom(float d); + + void localTranslate(const Eigen::Vector3f &t); + + /** Setup OpenGL matrices and viewport */ + void activateGL(void); + + Eigen::Vector3f unProject(const Eigen::Vector2f &uv, float depth, const Eigen::Matrix4f &invModelview) const; + Eigen::Vector3f unProject(const Eigen::Vector2f &uv, float depth) const; + +protected: + void updateViewMatrix(void) const; + void updateProjectionMatrix(void) const; + +protected: + uint mVpX, mVpY; + uint mVpWidth, mVpHeight; + + Frame mFrame; + + mutable Eigen::Affine3f mViewMatrix; + mutable Eigen::Matrix4f mProjectionMatrix; + + mutable bool mViewIsUptodate; + mutable bool mProjIsUptodate; + + // used by rotateAroundTarget + Eigen::Vector3f mTarget; + + float mFovY; + float mNearDist; + float mFarDist; }; -#endif // EIGEN_CAMERA_H +#endif// EIGEN_CAMERA_H diff --git a/filmulator-gui/core/nlmeans/eigen/demos/opengl/gpuhelper.cpp b/filmulator-gui/core/nlmeans/eigen/demos/opengl/gpuhelper.cpp index fd236b11..7e5daa8d 100644 --- a/filmulator-gui/core/nlmeans/eigen/demos/opengl/gpuhelper.cpp +++ b/filmulator-gui/core/nlmeans/eigen/demos/opengl/gpuhelper.cpp @@ -12,109 +12,125 @@ #include // PLEASE don't look at this old code... ;) -#include #include +#include GpuHelper gpu; GpuHelper::GpuHelper() { - mVpWidth = mVpHeight = 0; - mCurrentMatrixTarget = 0; - mInitialized = false; + mVpWidth = mVpHeight = 0; + mCurrentMatrixTarget = 0; + mInitialized = false; } -GpuHelper::~GpuHelper() -{ -} +GpuHelper::~GpuHelper() {} void GpuHelper::pushProjectionMode2D(ProjectionMode2D pm) { - // switch to 2D projection - pushMatrix(Matrix4f::Identity(),GL_PROJECTION); - - if(pm==PM_Normalized) - { - //glOrtho(-1., 1., -1., 1., 0., 1.); - } - else if(pm==PM_Viewport) - { - GLint vp[4]; - glGetIntegerv(GL_VIEWPORT, vp); - glOrtho(0., vp[2], 0., vp[3], -1., 1.); - } - - pushMatrix(Matrix4f::Identity(),GL_MODELVIEW); + // switch to 2D projection + pushMatrix(Matrix4f::Identity(), GL_PROJECTION); + + if (pm == PM_Normalized) { + // glOrtho(-1., 1., -1., 1., 0., 1.); + } else if (pm == PM_Viewport) { + GLint vp[4]; + glGetIntegerv(GL_VIEWPORT, vp); + glOrtho(0., vp[2], 0., vp[3], -1., 1.); + } + + pushMatrix(Matrix4f::Identity(), GL_MODELVIEW); } void GpuHelper::popProjectionMode2D(void) { - popMatrix(GL_PROJECTION); - popMatrix(GL_MODELVIEW); + popMatrix(GL_PROJECTION); + popMatrix(GL_MODELVIEW); } -void GpuHelper::drawVector(const Vector3f& position, const Vector3f& vec, const Color& color, float aspect /* = 50.*/) +void GpuHelper::drawVector(const Vector3f &position, const Vector3f &vec, const Color &color, float aspect /* = 50.*/) { - static GLUquadricObj *cylindre = gluNewQuadric(); - glColor4fv(color.data()); - float length = vec.norm(); - pushMatrix(GL_MODELVIEW); - glTranslatef(position.x(), position.y(), position.z()); - Vector3f ax = Matrix3f::Identity().col(2).cross(vec); - ax.normalize(); - Vector3f tmp = vec; - tmp.normalize(); - float angle = 180.f/M_PI * acos(tmp.z()); - if (angle>1e-3) - glRotatef(angle, ax.x(), ax.y(), ax.z()); - gluCylinder(cylindre, length/aspect, length/aspect, 0.8*length, 10, 10); - glTranslatef(0.0,0.0,0.8*length); - gluCylinder(cylindre, 2.0*length/aspect, 0.0, 0.2*length, 10, 10); - - popMatrix(GL_MODELVIEW); + static GLUquadricObj *cylindre = gluNewQuadric(); + glColor4fv(color.data()); + float length = vec.norm(); + pushMatrix(GL_MODELVIEW); + glTranslatef(position.x(), position.y(), position.z()); + Vector3f ax = Matrix3f::Identity().col(2).cross(vec); + ax.normalize(); + Vector3f tmp = vec; + tmp.normalize(); + float angle = 180.f / M_PI * acos(tmp.z()); + if (angle > 1e-3) glRotatef(angle, ax.x(), ax.y(), ax.z()); + gluCylinder(cylindre, length / aspect, length / aspect, 0.8 * length, 10, 10); + glTranslatef(0.0, 0.0, 0.8 * length); + gluCylinder(cylindre, 2.0 * length / aspect, 0.0, 0.2 * length, 10, 10); + + popMatrix(GL_MODELVIEW); } -void GpuHelper::drawVectorBox(const Vector3f& position, const Vector3f& vec, const Color& color, float aspect) +void GpuHelper::drawVectorBox(const Vector3f &position, const Vector3f &vec, const Color &color, float aspect) { - static GLUquadricObj *cylindre = gluNewQuadric(); - glColor4fv(color.data()); - float length = vec.norm(); - pushMatrix(GL_MODELVIEW); - glTranslatef(position.x(), position.y(), position.z()); - Vector3f ax = Matrix3f::Identity().col(2).cross(vec); - ax.normalize(); - Vector3f tmp = vec; - tmp.normalize(); - float angle = 180.f/M_PI * acos(tmp.z()); - if (angle>1e-3) - glRotatef(angle, ax.x(), ax.y(), ax.z()); - gluCylinder(cylindre, length/aspect, length/aspect, 0.8*length, 10, 10); - glTranslatef(0.0,0.0,0.8*length); - glScalef(4.0*length/aspect,4.0*length/aspect,4.0*length/aspect); - drawUnitCube(); - popMatrix(GL_MODELVIEW); + static GLUquadricObj *cylindre = gluNewQuadric(); + glColor4fv(color.data()); + float length = vec.norm(); + pushMatrix(GL_MODELVIEW); + glTranslatef(position.x(), position.y(), position.z()); + Vector3f ax = Matrix3f::Identity().col(2).cross(vec); + ax.normalize(); + Vector3f tmp = vec; + tmp.normalize(); + float angle = 180.f / M_PI * acos(tmp.z()); + if (angle > 1e-3) glRotatef(angle, ax.x(), ax.y(), ax.z()); + gluCylinder(cylindre, length / aspect, length / aspect, 0.8 * length, 10, 10); + glTranslatef(0.0, 0.0, 0.8 * length); + glScalef(4.0 * length / aspect, 4.0 * length / aspect, 4.0 * length / aspect); + drawUnitCube(); + popMatrix(GL_MODELVIEW); } void GpuHelper::drawUnitCube(void) { - static float vertices[][3] = { - {-0.5,-0.5,-0.5}, - { 0.5,-0.5,-0.5}, - {-0.5, 0.5,-0.5}, - { 0.5, 0.5,-0.5}, - {-0.5,-0.5, 0.5}, - { 0.5,-0.5, 0.5}, - {-0.5, 0.5, 0.5}, - { 0.5, 0.5, 0.5}}; - - glBegin(GL_QUADS); - glNormal3f(0,0,-1); glVertex3fv(vertices[0]); glVertex3fv(vertices[2]); glVertex3fv(vertices[3]); glVertex3fv(vertices[1]); - glNormal3f(0,0, 1); glVertex3fv(vertices[4]); glVertex3fv(vertices[5]); glVertex3fv(vertices[7]); glVertex3fv(vertices[6]); - glNormal3f(0,-1,0); glVertex3fv(vertices[0]); glVertex3fv(vertices[1]); glVertex3fv(vertices[5]); glVertex3fv(vertices[4]); - glNormal3f(0, 1,0); glVertex3fv(vertices[2]); glVertex3fv(vertices[6]); glVertex3fv(vertices[7]); glVertex3fv(vertices[3]); - glNormal3f(-1,0,0); glVertex3fv(vertices[0]); glVertex3fv(vertices[4]); glVertex3fv(vertices[6]); glVertex3fv(vertices[2]); - glNormal3f( 1,0,0); glVertex3fv(vertices[1]); glVertex3fv(vertices[3]); glVertex3fv(vertices[7]); glVertex3fv(vertices[5]); - glEnd(); + static float vertices[][3] = { { -0.5, -0.5, -0.5 }, + { 0.5, -0.5, -0.5 }, + { -0.5, 0.5, -0.5 }, + { 0.5, 0.5, -0.5 }, + { -0.5, -0.5, 0.5 }, + { 0.5, -0.5, 0.5 }, + { -0.5, 0.5, 0.5 }, + { 0.5, 0.5, 0.5 } }; + + glBegin(GL_QUADS); + glNormal3f(0, 0, -1); + glVertex3fv(vertices[0]); + glVertex3fv(vertices[2]); + glVertex3fv(vertices[3]); + glVertex3fv(vertices[1]); + glNormal3f(0, 0, 1); + glVertex3fv(vertices[4]); + glVertex3fv(vertices[5]); + glVertex3fv(vertices[7]); + glVertex3fv(vertices[6]); + glNormal3f(0, -1, 0); + glVertex3fv(vertices[0]); + glVertex3fv(vertices[1]); + glVertex3fv(vertices[5]); + glVertex3fv(vertices[4]); + glNormal3f(0, 1, 0); + glVertex3fv(vertices[2]); + glVertex3fv(vertices[6]); + glVertex3fv(vertices[7]); + glVertex3fv(vertices[3]); + glNormal3f(-1, 0, 0); + glVertex3fv(vertices[0]); + glVertex3fv(vertices[4]); + glVertex3fv(vertices[6]); + glVertex3fv(vertices[2]); + glNormal3f(1, 0, 0); + glVertex3fv(vertices[1]); + glVertex3fv(vertices[3]); + glVertex3fv(vertices[7]); + glVertex3fv(vertices[5]); + glEnd(); } void GpuHelper::drawUnitSphere(int level) @@ -122,5 +138,3 @@ void GpuHelper::drawUnitSphere(int level) static IcoSphere sphere; sphere.draw(level); } - - diff --git a/filmulator-gui/core/nlmeans/eigen/demos/opengl/gpuhelper.h b/filmulator-gui/core/nlmeans/eigen/demos/opengl/gpuhelper.h index 9ff98e9d..4f1abb19 100644 --- a/filmulator-gui/core/nlmeans/eigen/demos/opengl/gpuhelper.h +++ b/filmulator-gui/core/nlmeans/eigen/demos/opengl/gpuhelper.h @@ -20,188 +20,181 @@ typedef Vector4f Color; class GpuHelper { - public: - - GpuHelper(); - - ~GpuHelper(); - - enum ProjectionMode2D { PM_Normalized = 1, PM_Viewport = 2 }; - void pushProjectionMode2D(ProjectionMode2D pm); - void popProjectionMode2D(); - - /** Multiply the OpenGL matrix \a matrixTarget by the matrix \a mat. - Essentially, this helper function automatically calls glMatrixMode(matrixTarget) if required - and does a proper call to the right glMultMatrix*() function according to the scalar type - and storage order. - \warning glMatrixMode() must never be called directly. If your're unsure, use forceMatrixMode(). - \sa Matrix, loadMatrix(), forceMatrixMode() - */ - template - void multMatrix(const Matrix& mat, GLenum matrixTarget); - - /** Load the matrix \a mat to the OpenGL matrix \a matrixTarget. - Essentially, this helper function automatically calls glMatrixMode(matrixTarget) if required - and does a proper call to the right glLoadMatrix*() or glLoadIdentity() function according to the scalar type - and storage order. - \warning glMatrixMode() must never be called directly. If your're unsure, use forceMatrixMode(). - \sa Matrix, multMatrix(), forceMatrixMode() - */ - template - void loadMatrix(const Eigen::Matrix& mat, GLenum matrixTarget); - - template - void loadMatrix( - const Eigen::CwiseNullaryOp,Derived>&, - GLenum matrixTarget); - - /** Make the matrix \a matrixTarget the current OpenGL matrix target. - Call this function before loadMatrix() or multMatrix() if you cannot guarantee that glMatrixMode() - has never been called after the last loadMatrix() or multMatrix() calls. - \todo provides a debug mode checking the sanity of the cached matrix mode. - */ - inline void forceMatrixTarget(GLenum matrixTarget) {glMatrixMode(mCurrentMatrixTarget=matrixTarget);} - - inline void setMatrixTarget(GLenum matrixTarget); - - /** Push the OpenGL matrix \a matrixTarget and load \a mat. - */ - template - inline void pushMatrix(const Matrix& mat, GLenum matrixTarget); - - template - void pushMatrix( - const Eigen::CwiseNullaryOp,Derived>&, - GLenum matrixTarget); - - /** Push and clone the OpenGL matrix \a matrixTarget - */ - inline void pushMatrix(GLenum matrixTarget); - - /** Pop the OpenGL matrix \a matrixTarget - */ - inline void popMatrix(GLenum matrixTarget); - - void drawVector(const Vector3f& position, const Vector3f& vec, const Color& color, float aspect = 50.); - void drawVectorBox(const Vector3f& position, const Vector3f& vec, const Color& color, float aspect = 50.); - void drawUnitCube(void); - void drawUnitSphere(int level=0); - - /// draw the \a nofElement first elements - inline void draw(GLenum mode, uint nofElement); - - /// draw a range of elements - inline void draw(GLenum mode, uint start, uint end); - - /// draw an indexed subset - inline void draw(GLenum mode, const std::vector* pIndexes); +public: + GpuHelper(); + + ~GpuHelper(); + + enum ProjectionMode2D { PM_Normalized = 1, PM_Viewport = 2 }; + void pushProjectionMode2D(ProjectionMode2D pm); + void popProjectionMode2D(); + + /** Multiply the OpenGL matrix \a matrixTarget by the matrix \a mat. + Essentially, this helper function automatically calls glMatrixMode(matrixTarget) if required + and does a proper call to the right glMultMatrix*() function according to the scalar type + and storage order. + \warning glMatrixMode() must never be called directly. If your're unsure, use forceMatrixMode(). + \sa Matrix, loadMatrix(), forceMatrixMode() + */ + template + void multMatrix(const Matrix &mat, GLenum matrixTarget); + + /** Load the matrix \a mat to the OpenGL matrix \a matrixTarget. + Essentially, this helper function automatically calls glMatrixMode(matrixTarget) if required + and does a proper call to the right glLoadMatrix*() or glLoadIdentity() function according to the scalar type + and storage order. + \warning glMatrixMode() must never be called directly. If your're unsure, use forceMatrixMode(). + \sa Matrix, multMatrix(), forceMatrixMode() + */ + template + void loadMatrix(const Eigen::Matrix &mat, GLenum matrixTarget); + + template + void loadMatrix(const Eigen::CwiseNullaryOp, Derived> &, + GLenum matrixTarget); + + /** Make the matrix \a matrixTarget the current OpenGL matrix target. + Call this function before loadMatrix() or multMatrix() if you cannot guarantee that glMatrixMode() + has never been called after the last loadMatrix() or multMatrix() calls. + \todo provides a debug mode checking the sanity of the cached matrix mode. + */ + inline void forceMatrixTarget(GLenum matrixTarget) { glMatrixMode(mCurrentMatrixTarget = matrixTarget); } + + inline void setMatrixTarget(GLenum matrixTarget); + + /** Push the OpenGL matrix \a matrixTarget and load \a mat. + */ + template + inline void pushMatrix(const Matrix &mat, GLenum matrixTarget); + + template + void pushMatrix(const Eigen::CwiseNullaryOp, Derived> &, + GLenum matrixTarget); + + /** Push and clone the OpenGL matrix \a matrixTarget + */ + inline void pushMatrix(GLenum matrixTarget); + + /** Pop the OpenGL matrix \a matrixTarget + */ + inline void popMatrix(GLenum matrixTarget); + + void drawVector(const Vector3f &position, const Vector3f &vec, const Color &color, float aspect = 50.); + void drawVectorBox(const Vector3f &position, const Vector3f &vec, const Color &color, float aspect = 50.); + void drawUnitCube(void); + void drawUnitSphere(int level = 0); + + /// draw the \a nofElement first elements + inline void draw(GLenum mode, uint nofElement); + + /// draw a range of elements + inline void draw(GLenum mode, uint start, uint end); + + /// draw an indexed subset + inline void draw(GLenum mode, const std::vector *pIndexes); protected: + void update(void); - void update(void); - - GLuint mColorBufferId; - int mVpWidth, mVpHeight; - GLenum mCurrentMatrixTarget; - bool mInitialized; + GLuint mColorBufferId; + int mVpWidth, mVpHeight; + GLenum mCurrentMatrixTarget; + bool mInitialized; }; /** Singleton shortcut -*/ + */ extern GpuHelper gpu; /** \internal -*/ + */ template struct GlMatrixHelper; -template struct GlMatrixHelper +template struct GlMatrixHelper { - static void loadMatrix(const Matrix& mat) { glLoadMatrixf(mat.data()); } - static void loadMatrix(const Matrix& mat) { glLoadMatrixd(mat.data()); } - static void multMatrix(const Matrix& mat) { glMultMatrixf(mat.data()); } - static void multMatrix(const Matrix& mat) { glMultMatrixd(mat.data()); } + static void loadMatrix(const Matrix &mat) { glLoadMatrixf(mat.data()); } + static void loadMatrix(const Matrix &mat) { glLoadMatrixd(mat.data()); } + static void multMatrix(const Matrix &mat) { glMultMatrixf(mat.data()); } + static void multMatrix(const Matrix &mat) { glMultMatrixd(mat.data()); } }; -template struct GlMatrixHelper +template struct GlMatrixHelper { - static void loadMatrix(const Matrix& mat) { glLoadMatrixf(mat.transpose().eval().data()); } - static void loadMatrix(const Matrix& mat) { glLoadMatrixd(mat.transpose().eval().data()); } - static void multMatrix(const Matrix& mat) { glMultMatrixf(mat.transpose().eval().data()); } - static void multMatrix(const Matrix& mat) { glMultMatrixd(mat.transpose().eval().data()); } + static void loadMatrix(const Matrix &mat) { glLoadMatrixf(mat.transpose().eval().data()); } + static void loadMatrix(const Matrix &mat) + { + glLoadMatrixd(mat.transpose().eval().data()); + } + static void multMatrix(const Matrix &mat) { glMultMatrixf(mat.transpose().eval().data()); } + static void multMatrix(const Matrix &mat) + { + glMultMatrixd(mat.transpose().eval().data()); + } }; inline void GpuHelper::setMatrixTarget(GLenum matrixTarget) { - if (matrixTarget != mCurrentMatrixTarget) - glMatrixMode(mCurrentMatrixTarget=matrixTarget); + if (matrixTarget != mCurrentMatrixTarget) glMatrixMode(mCurrentMatrixTarget = matrixTarget); } template -void GpuHelper::multMatrix(const Matrix& mat, GLenum matrixTarget) +void GpuHelper::multMatrix(const Matrix &mat, GLenum matrixTarget) { - setMatrixTarget(matrixTarget); - GlMatrixHelper<_Flags&Eigen::RowMajorBit, _Flags>::multMatrix(mat); + setMatrixTarget(matrixTarget); + GlMatrixHelper<_Flags & Eigen::RowMajorBit, _Flags>::multMatrix(mat); } template -void GpuHelper::loadMatrix( - const Eigen::CwiseNullaryOp,Derived>&, - GLenum matrixTarget) +void GpuHelper::loadMatrix(const Eigen::CwiseNullaryOp, Derived> &, + GLenum matrixTarget) { - setMatrixTarget(matrixTarget); - glLoadIdentity(); + setMatrixTarget(matrixTarget); + glLoadIdentity(); } template -void GpuHelper::loadMatrix(const Eigen::Matrix& mat, GLenum matrixTarget) +void GpuHelper::loadMatrix(const Eigen::Matrix &mat, GLenum matrixTarget) { - setMatrixTarget(matrixTarget); - GlMatrixHelper<(_Flags&Eigen::RowMajorBit)!=0, _Flags>::loadMatrix(mat); + setMatrixTarget(matrixTarget); + GlMatrixHelper<(_Flags & Eigen::RowMajorBit) != 0, _Flags>::loadMatrix(mat); } inline void GpuHelper::pushMatrix(GLenum matrixTarget) { - setMatrixTarget(matrixTarget); - glPushMatrix(); + setMatrixTarget(matrixTarget); + glPushMatrix(); } template -inline void GpuHelper::pushMatrix(const Matrix& mat, GLenum matrixTarget) +inline void GpuHelper::pushMatrix(const Matrix &mat, GLenum matrixTarget) { - pushMatrix(matrixTarget); - GlMatrixHelper<_Flags&Eigen::RowMajorBit,_Flags>::loadMatrix(mat); + pushMatrix(matrixTarget); + GlMatrixHelper<_Flags & Eigen::RowMajorBit, _Flags>::loadMatrix(mat); } template -void GpuHelper::pushMatrix( - const Eigen::CwiseNullaryOp,Derived>&, - GLenum matrixTarget) +void GpuHelper::pushMatrix(const Eigen::CwiseNullaryOp, Derived> &, + GLenum matrixTarget) { - pushMatrix(matrixTarget); - glLoadIdentity(); + pushMatrix(matrixTarget); + glLoadIdentity(); } inline void GpuHelper::popMatrix(GLenum matrixTarget) { - setMatrixTarget(matrixTarget); - glPopMatrix(); + setMatrixTarget(matrixTarget); + glPopMatrix(); } -inline void GpuHelper::draw(GLenum mode, uint nofElement) -{ - glDrawArrays(mode, 0, nofElement); -} +inline void GpuHelper::draw(GLenum mode, uint nofElement) { glDrawArrays(mode, 0, nofElement); } -inline void GpuHelper::draw(GLenum mode, const std::vector* pIndexes) +inline void GpuHelper::draw(GLenum mode, const std::vector *pIndexes) { - glDrawElements(mode, pIndexes->size(), GL_UNSIGNED_INT, &(pIndexes->front())); + glDrawElements(mode, pIndexes->size(), GL_UNSIGNED_INT, &(pIndexes->front())); } -inline void GpuHelper::draw(GLenum mode, uint start, uint end) -{ - glDrawArrays(mode, start, end-start); -} +inline void GpuHelper::draw(GLenum mode, uint start, uint end) { glDrawArrays(mode, start, end - start); } -#endif // EIGEN_GPUHELPER_H +#endif// EIGEN_GPUHELPER_H diff --git a/filmulator-gui/core/nlmeans/eigen/demos/opengl/icosphere.cpp b/filmulator-gui/core/nlmeans/eigen/demos/opengl/icosphere.cpp index 39444cbb..09ad72e8 100644 --- a/filmulator-gui/core/nlmeans/eigen/demos/opengl/icosphere.cpp +++ b/filmulator-gui/core/nlmeans/eigen/demos/opengl/icosphere.cpp @@ -20,101 +20,117 @@ using namespace Eigen; #define X .525731112119133606 #define Z .850650808352039932 -static GLfloat vdata[12][3] = { - {-X, 0.0, Z}, {X, 0.0, Z}, {-X, 0.0, -Z}, {X, 0.0, -Z}, - {0.0, Z, X}, {0.0, Z, -X}, {0.0, -Z, X}, {0.0, -Z, -X}, - {Z, X, 0.0}, {-Z, X, 0.0}, {Z, -X, 0.0}, {-Z, -X, 0.0} -}; +static GLfloat vdata[12][3] = { { -X, 0.0, Z }, + { X, 0.0, Z }, + { -X, 0.0, -Z }, + { X, 0.0, -Z }, + { 0.0, Z, X }, + { 0.0, Z, -X }, + { 0.0, -Z, X }, + { 0.0, -Z, -X }, + { Z, X, 0.0 }, + { -Z, X, 0.0 }, + { Z, -X, 0.0 }, + { -Z, -X, 0.0 } }; -static GLint tindices[20][3] = { - {0,4,1}, {0,9,4}, {9,5,4}, {4,5,8}, {4,8,1}, - {8,10,1}, {8,3,10}, {5,3,8}, {5,2,3}, {2,7,3}, - {7,10,3}, {7,6,10}, {7,11,6}, {11,0,6}, {0,1,6}, - {6,1,10}, {9,0,11}, {9,11,2}, {9,2,5}, {7,2,11} }; +static GLint tindices[20][3] = { { 0, 4, 1 }, + { 0, 9, 4 }, + { 9, 5, 4 }, + { 4, 5, 8 }, + { 4, 8, 1 }, + { 8, 10, 1 }, + { 8, 3, 10 }, + { 5, 3, 8 }, + { 5, 2, 3 }, + { 2, 7, 3 }, + { 7, 10, 3 }, + { 7, 6, 10 }, + { 7, 11, 6 }, + { 11, 0, 6 }, + { 0, 1, 6 }, + { 6, 1, 10 }, + { 9, 0, 11 }, + { 9, 11, 2 }, + { 9, 2, 5 }, + { 7, 2, 11 } }; //-------------------------------------------------------------------------------- IcoSphere::IcoSphere(unsigned int levels) { // init with an icosahedron - for (int i = 0; i < 12; i++) - mVertices.push_back(Map(vdata[i])); + for (int i = 0; i < 12; i++) mVertices.push_back(Map(vdata[i])); mIndices.push_back(new std::vector); - std::vector& indices = *mIndices.back(); - for (int i = 0; i < 20; i++) - { - for (int k = 0; k < 3; k++) - indices.push_back(tindices[i][k]); + std::vector &indices = *mIndices.back(); + for (int i = 0; i < 20; i++) { + for (int k = 0; k < 3; k++) indices.push_back(tindices[i][k]); } mListIds.push_back(0); - while(mIndices.size()& IcoSphere::indices(int level) const +const std::vector &IcoSphere::indices(int level) const { - while (level>=int(mIndices.size())) - const_cast(this)->_subdivide(); + while (level >= int(mIndices.size())) const_cast(this)->_subdivide(); return *mIndices[level]; } void IcoSphere::_subdivide(void) { typedef unsigned long long Key; - std::map edgeMap; - const std::vector& indices = *mIndices.back(); + std::map edgeMap; + const std::vector &indices = *mIndices.back(); mIndices.push_back(new std::vector); - std::vector& refinedIndices = *mIndices.back(); + std::vector &refinedIndices = *mIndices.back(); int end = indices.size(); - for (int i=0; ie0) - std::swap(e0,e1); - Key edgeKey = Key(e0) | (Key(e1)<<32); - std::map::iterator it = edgeMap.find(edgeKey); - if (it==edgeMap.end()) - { + if (e1 > e0) std::swap(e0, e1); + Key edgeKey = Key(e0) | (Key(e1) << 32); + std::map::iterator it = edgeMap.find(edgeKey); + if (it == edgeMap.end()) { ids1[k] = mVertices.size(); edgeMap[edgeKey] = ids1[k]; - mVertices.push_back( (mVertices[e0]+mVertices[e1]).normalized() ); - } - else + mVertices.push_back((mVertices[e0] + mVertices[e1]).normalized()); + } else ids1[k] = it->second; } - refinedIndices.push_back(ids0[0]); refinedIndices.push_back(ids1[0]); refinedIndices.push_back(ids1[2]); - refinedIndices.push_back(ids0[1]); refinedIndices.push_back(ids1[1]); refinedIndices.push_back(ids1[0]); - refinedIndices.push_back(ids0[2]); refinedIndices.push_back(ids1[2]); refinedIndices.push_back(ids1[1]); - refinedIndices.push_back(ids1[0]); refinedIndices.push_back(ids1[1]); refinedIndices.push_back(ids1[2]); + refinedIndices.push_back(ids0[0]); + refinedIndices.push_back(ids1[0]); + refinedIndices.push_back(ids1[2]); + refinedIndices.push_back(ids0[1]); + refinedIndices.push_back(ids1[1]); + refinedIndices.push_back(ids1[0]); + refinedIndices.push_back(ids0[2]); + refinedIndices.push_back(ids1[2]); + refinedIndices.push_back(ids1[1]); + refinedIndices.push_back(ids1[0]); + refinedIndices.push_back(ids1[1]); + refinedIndices.push_back(ids1[2]); } mListIds.push_back(0); } void IcoSphere::draw(int level) { - while (level>=int(mIndices.size())) - const_cast(this)->_subdivide(); - if (mListIds[level]==0) - { + while (level >= int(mIndices.size())) const_cast(this)->_subdivide(); + if (mListIds[level] == 0) { mListIds[level] = glGenLists(1); glNewList(mListIds[level], GL_COMPILE); - glVertexPointer(3, GL_FLOAT, 0, mVertices[0].data()); - glNormalPointer(GL_FLOAT, 0, mVertices[0].data()); - glEnableClientState(GL_VERTEX_ARRAY); - glEnableClientState(GL_NORMAL_ARRAY); - glDrawElements(GL_TRIANGLES, mIndices[level]->size(), GL_UNSIGNED_INT, &(mIndices[level]->at(0))); - glDisableClientState(GL_VERTEX_ARRAY); - glDisableClientState(GL_NORMAL_ARRAY); + glVertexPointer(3, GL_FLOAT, 0, mVertices[0].data()); + glNormalPointer(GL_FLOAT, 0, mVertices[0].data()); + glEnableClientState(GL_VERTEX_ARRAY); + glEnableClientState(GL_NORMAL_ARRAY); + glDrawElements(GL_TRIANGLES, mIndices[level]->size(), GL_UNSIGNED_INT, &(mIndices[level]->at(0))); + glDisableClientState(GL_VERTEX_ARRAY); + glDisableClientState(GL_NORMAL_ARRAY); glEndList(); } glCallList(mListIds[level]); } - - diff --git a/filmulator-gui/core/nlmeans/eigen/demos/opengl/icosphere.h b/filmulator-gui/core/nlmeans/eigen/demos/opengl/icosphere.h index b0210edc..c848509d 100644 --- a/filmulator-gui/core/nlmeans/eigen/demos/opengl/icosphere.h +++ b/filmulator-gui/core/nlmeans/eigen/demos/opengl/icosphere.h @@ -15,16 +15,17 @@ class IcoSphere { - public: - IcoSphere(unsigned int levels=1); - const std::vector& vertices() const { return mVertices; } - const std::vector& indices(int level) const; - void draw(int level); - protected: - void _subdivide(); - std::vector mVertices; - std::vector*> mIndices; - std::vector mListIds; +public: + IcoSphere(unsigned int levels = 1); + const std::vector &vertices() const { return mVertices; } + const std::vector &indices(int level) const; + void draw(int level); + +protected: + void _subdivide(); + std::vector mVertices; + std::vector *> mIndices; + std::vector mListIds; }; -#endif // EIGEN_ICOSPHERE_H +#endif// EIGEN_ICOSPHERE_H diff --git a/filmulator-gui/core/nlmeans/eigen/demos/opengl/quaternion_demo.cpp b/filmulator-gui/core/nlmeans/eigen/demos/opengl/quaternion_demo.cpp index dd323a4c..72d8dd16 100644 --- a/filmulator-gui/core/nlmeans/eigen/demos/opengl/quaternion_demo.cpp +++ b/filmulator-gui/core/nlmeans/eigen/demos/opengl/quaternion_demo.cpp @@ -11,125 +11,111 @@ #include "icosphere.h" #include -#include #include +#include -#include -#include -#include -#include -#include #include -#include #include -#include +#include +#include #include +#include +#include +#include +#include +#include using namespace Eigen; class FancySpheres { - public: - EIGEN_MAKE_ALIGNED_OPERATOR_NEW - - FancySpheres() +public: + EIGEN_MAKE_ALIGNED_OPERATOR_NEW + + FancySpheres() + { + const int levels = 4; + const float scale = 0.33; + float radius = 100; + std::vector parents; + + // leval 0 + mCenters.push_back(Vector3f::Zero()); + parents.push_back(-1); + mRadii.push_back(radius); + + // generate level 1 using icosphere vertices + radius *= 0.45; { - const int levels = 4; - const float scale = 0.33; - float radius = 100; - std::vector parents; - - // leval 0 - mCenters.push_back(Vector3f::Zero()); - parents.push_back(-1); - mRadii.push_back(radius); - - // generate level 1 using icosphere vertices - radius *= 0.45; - { - float dist = mRadii[0]*0.9; - for (int i=0; i<12; ++i) - { - mCenters.push_back(mIcoSphere.vertices()[i] * dist); - mRadii.push_back(radius); - parents.push_back(0); - } + float dist = mRadii[0] * 0.9; + for (int i = 0; i < 12; ++i) { + mCenters.push_back(mIcoSphere.vertices()[i] * dist); + mRadii.push_back(radius); + parents.push_back(0); } + } + + static const float angles[10] = { 0, 0, M_PI, 0. * M_PI, M_PI, 0.5 * M_PI, M_PI, 1. * M_PI, M_PI, 1.5 * M_PI }; - static const float angles [10] = { - 0, 0, - M_PI, 0.*M_PI, - M_PI, 0.5*M_PI, - M_PI, 1.*M_PI, - M_PI, 1.5*M_PI - }; - - // generate other levels - int start = 1; - for (int l=1; l mCenters; - std::vector mRadii; - IcoSphere mIcoSphere; + glDisable(GL_NORMALIZE); + } + +protected: + std::vector mCenters; + std::vector mRadii; + IcoSphere mIcoSphere; }; // generic linear interpolation method -template T lerp(float t, const T& a, const T& b) -{ - return a*(1-t) + b*t; -} +template T lerp(float t, const T &a, const T &b) { return a * (1 - t) + b * t; } // quaternion slerp -template<> Quaternionf lerp(float t, const Quaternionf& a, const Quaternionf& b) -{ return a.slerp(t,b); } +template<> Quaternionf lerp(float t, const Quaternionf &a, const Quaternionf &b) { return a.slerp(t, b); } // linear interpolation of a frame using the type OrientationType // to perform the interpolation of the orientations -template -inline static Frame lerpFrame(float alpha, const Frame& a, const Frame& b) +template inline static Frame lerpFrame(float alpha, const Frame &a, const Frame &b) { - return Frame(lerp(alpha,a.position,b.position), - Quaternionf(lerp(alpha,OrientationType(a.orientation),OrientationType(b.orientation)))); + return Frame(lerp(alpha, a.position, b.position), + Quaternionf(lerp(alpha, OrientationType(a.orientation), OrientationType(b.orientation)))); } template class EulerAngles @@ -137,37 +123,35 @@ template class EulerAngles public: enum { Dim = 3 }; typedef _Scalar Scalar; - typedef Matrix Matrix3; - typedef Matrix Vector3; + typedef Matrix Matrix3; + typedef Matrix Vector3; typedef Quaternion QuaternionType; protected: - Vector3 m_angles; public: - EulerAngles() {} inline EulerAngles(Scalar a0, Scalar a1, Scalar a2) : m_angles(a0, a1, a2) {} - inline EulerAngles(const QuaternionType& q) { *this = q; } + inline EulerAngles(const QuaternionType &q) { *this = q; } - const Vector3& coeffs() const { return m_angles; } - Vector3& coeffs() { return m_angles; } + const Vector3 &coeffs() const { return m_angles; } + Vector3 &coeffs() { return m_angles; } - EulerAngles& operator=(const QuaternionType& q) + EulerAngles &operator=(const QuaternionType &q) { Matrix3 m = q.toRotationMatrix(); return *this = m; } - EulerAngles& operator=(const Matrix3& m) + EulerAngles &operator=(const Matrix3 &m) { // mat = cy*cz -cy*sz sy // cz*sx*sy+cx*sz cx*cz-sx*sy*sz -cy*sx // -cx*cz*sy+sx*sz cz*sx+cx*sy*sz cx*cy - m_angles.coeffRef(1) = std::asin(m.coeff(0,2)); - m_angles.coeffRef(0) = std::atan2(-m.coeff(1,2),m.coeff(2,2)); - m_angles.coeffRef(2) = std::atan2(-m.coeff(0,1),m.coeff(0,0)); + m_angles.coeffRef(1) = std::asin(m.coeff(0, 2)); + m_angles.coeffRef(0) = std::atan2(-m.coeff(1, 2), m.coeff(2, 2)); + m_angles.coeffRef(2) = std::atan2(-m.coeff(0, 1), m.coeff(0, 0)); return *this; } @@ -176,9 +160,9 @@ template class EulerAngles Vector3 c = m_angles.array().cos(); Vector3 s = m_angles.array().sin(); Matrix3 res; - res << c.y()*c.z(), -c.y()*s.z(), s.y(), - c.z()*s.x()*s.y()+c.x()*s.z(), c.x()*c.z()-s.x()*s.y()*s.z(), -c.y()*s.x(), - -c.x()*c.z()*s.y()+s.x()*s.z(), c.z()*s.x()+c.x()*s.y()*s.z(), c.x()*c.y(); + res << c.y() * c.z(), -c.y() * s.z(), s.y(), c.z() * s.x() * s.y() + c.x() * s.z(), + c.x() * c.z() - s.x() * s.y() * s.z(), -c.y() * s.x(), -c.x() * c.z() * s.y() + s.x() * s.z(), + c.z() * s.x() + c.x() * s.y() * s.z(), c.x() * c.y(); return res; } @@ -186,7 +170,7 @@ template class EulerAngles }; // Euler angles slerp -template<> EulerAngles lerp(float t, const EulerAngles& a, const EulerAngles& b) +template<> EulerAngles lerp(float t, const EulerAngles &a, const EulerAngles &b) { EulerAngles res; res.coeffs() = lerp(t, a.coeffs(), b.coeffs()); @@ -209,41 +193,38 @@ RenderingWidget::RenderingWidget() void RenderingWidget::grabFrame(void) { - // ask user for a time - bool ok = false; - double t = 0; - if (!m_timeline.empty()) - t = (--m_timeline.end())->first + 1.; - t = QInputDialog::getDouble(this, "Eigen's RenderingWidget", "time value: ", - t, 0, 1e3, 1, &ok); - if (ok) - { - Frame aux; - aux.orientation = mCamera.viewMatrix().linear(); - aux.position = mCamera.viewMatrix().translation(); - m_timeline[t] = aux; - } + // ask user for a time + bool ok = false; + double t = 0; + if (!m_timeline.empty()) t = (--m_timeline.end())->first + 1.; + t = QInputDialog::getDouble(this, "Eigen's RenderingWidget", "time value: ", t, 0, 1e3, 1, &ok); + if (ok) { + Frame aux; + aux.orientation = mCamera.viewMatrix().linear(); + aux.position = mCamera.viewMatrix().translation(); + m_timeline[t] = aux; + } } void RenderingWidget::drawScene() { static FancySpheres sFancySpheres; float length = 50; - gpu.drawVector(Vector3f::Zero(), length*Vector3f::UnitX(), Color(1,0,0,1)); - gpu.drawVector(Vector3f::Zero(), length*Vector3f::UnitY(), Color(0,1,0,1)); - gpu.drawVector(Vector3f::Zero(), length*Vector3f::UnitZ(), Color(0,0,1,1)); + gpu.drawVector(Vector3f::Zero(), length * Vector3f::UnitX(), Color(1, 0, 0, 1)); + gpu.drawVector(Vector3f::Zero(), length * Vector3f::UnitY(), Color(0, 1, 0, 1)); + gpu.drawVector(Vector3f::Zero(), length * Vector3f::UnitZ(), Color(0, 0, 1, 1)); // draw the fractal object float sqrt3 = std::sqrt(3.); - glLightfv(GL_LIGHT0, GL_AMBIENT, Vector4f(0.5,0.5,0.5,1).data()); - glLightfv(GL_LIGHT0, GL_DIFFUSE, Vector4f(0.5,1,0.5,1).data()); - glLightfv(GL_LIGHT0, GL_SPECULAR, Vector4f(1,1,1,1).data()); - glLightfv(GL_LIGHT0, GL_POSITION, Vector4f(-sqrt3,-sqrt3,sqrt3,0).data()); + glLightfv(GL_LIGHT0, GL_AMBIENT, Vector4f(0.5, 0.5, 0.5, 1).data()); + glLightfv(GL_LIGHT0, GL_DIFFUSE, Vector4f(0.5, 1, 0.5, 1).data()); + glLightfv(GL_LIGHT0, GL_SPECULAR, Vector4f(1, 1, 1, 1).data()); + glLightfv(GL_LIGHT0, GL_POSITION, Vector4f(-sqrt3, -sqrt3, sqrt3, 0).data()); - glLightfv(GL_LIGHT1, GL_AMBIENT, Vector4f(0,0,0,1).data()); - glLightfv(GL_LIGHT1, GL_DIFFUSE, Vector4f(1,0.5,0.5,1).data()); - glLightfv(GL_LIGHT1, GL_SPECULAR, Vector4f(1,1,1,1).data()); - glLightfv(GL_LIGHT1, GL_POSITION, Vector4f(-sqrt3,sqrt3,-sqrt3,0).data()); + glLightfv(GL_LIGHT1, GL_AMBIENT, Vector4f(0, 0, 0, 1).data()); + glLightfv(GL_LIGHT1, GL_DIFFUSE, Vector4f(1, 0.5, 0.5, 1).data()); + glLightfv(GL_LIGHT1, GL_SPECULAR, Vector4f(1, 1, 1, 1).data()); + glLightfv(GL_LIGHT1, GL_POSITION, Vector4f(-sqrt3, sqrt3, -sqrt3, 0).data()); glMaterialfv(GL_FRONT_AND_BACK, GL_AMBIENT, Vector4f(0.7, 0.7, 0.7, 1).data()); glMaterialfv(GL_FRONT_AND_BACK, GL_DIFFUSE, Vector4f(0.8, 0.75, 0.6, 1).data()); @@ -276,26 +257,20 @@ void RenderingWidget::animate() Frame currentFrame; - if(hi==m_timeline.end()) - { + if (hi == m_timeline.end()) { // end currentFrame = lo->second; stopAnimation(); - } - else if(hi==m_timeline.begin()) - { + } else if (hi == m_timeline.begin()) { // start currentFrame = hi->second; - } - else - { - float s = (m_alpha - lo->first)/(hi->first - lo->first); - if (mLerpMode==LerpEulerAngles) - currentFrame = ::lerpFrame >(s, lo->second, hi->second); - else if (mLerpMode==LerpQuaternion) + } else { + float s = (m_alpha - lo->first) / (hi->first - lo->first); + if (mLerpMode == LerpEulerAngles) + currentFrame = ::lerpFrame>(s, lo->second, hi->second); + else if (mLerpMode == LerpQuaternion) currentFrame = ::lerpFrame(s, lo->second, hi->second); - else - { + else { std::cerr << "Invalid rotation interpolation mode (abort)\n"; exit(2); } @@ -303,53 +278,49 @@ void RenderingWidget::animate() } currentFrame.orientation = currentFrame.orientation.inverse(); - currentFrame.position = - (currentFrame.orientation * currentFrame.position); + currentFrame.position = -(currentFrame.orientation * currentFrame.position); mCamera.setFrame(currentFrame); updateGL(); } -void RenderingWidget::keyPressEvent(QKeyEvent * e) +void RenderingWidget::keyPressEvent(QKeyEvent *e) { - switch(e->key()) - { - case Qt::Key_Up: - mCamera.zoom(2); - break; - case Qt::Key_Down: - mCamera.zoom(-2); - break; - // add a frame - case Qt::Key_G: - grabFrame(); - break; - // clear the time line - case Qt::Key_C: - m_timeline.clear(); - break; - // move the camera to initial pos - case Qt::Key_R: - resetCamera(); - break; - // start/stop the animation - case Qt::Key_A: - if (mAnimate) - { - stopAnimation(); - } - else - { - m_alpha = 0; - connect(&m_timer, SIGNAL(timeout()), this, SLOT(animate())); - m_timer.start(1000/30); - mAnimate = true; - } - break; - default: - break; + switch (e->key()) { + case Qt::Key_Up: + mCamera.zoom(2); + break; + case Qt::Key_Down: + mCamera.zoom(-2); + break; + // add a frame + case Qt::Key_G: + grabFrame(); + break; + // clear the time line + case Qt::Key_C: + m_timeline.clear(); + break; + // move the camera to initial pos + case Qt::Key_R: + resetCamera(); + break; + // start/stop the animation + case Qt::Key_A: + if (mAnimate) { + stopAnimation(); + } else { + m_alpha = 0; + connect(&m_timer, SIGNAL(timeout()), this, SLOT(animate())); + m_timer.start(1000 / 30); + mAnimate = true; } + break; + default: + break; + } - updateGL(); + updateGL(); } void RenderingWidget::stopAnimation() @@ -360,105 +331,94 @@ void RenderingWidget::stopAnimation() m_alpha = 0; } -void RenderingWidget::mousePressEvent(QMouseEvent* e) +void RenderingWidget::mousePressEvent(QMouseEvent *e) { mMouseCoords = Vector2i(e->pos().x(), e->pos().y()); - bool fly = (mNavMode==NavFly) || (e->modifiers()&Qt::ControlModifier); - switch(e->button()) - { - case Qt::LeftButton: - if(fly) - { - mCurrentTrackingMode = TM_LOCAL_ROTATE; - mTrackball.start(Trackball::Local); - } - else - { - mCurrentTrackingMode = TM_ROTATE_AROUND; - mTrackball.start(Trackball::Around); - } - mTrackball.track(mMouseCoords); - break; - case Qt::MidButton: - if(fly) - mCurrentTrackingMode = TM_FLY_Z; - else - mCurrentTrackingMode = TM_ZOOM; - break; - case Qt::RightButton: - mCurrentTrackingMode = TM_FLY_PAN; - break; - default: - break; + bool fly = (mNavMode == NavFly) || (e->modifiers() & Qt::ControlModifier); + switch (e->button()) { + case Qt::LeftButton: + if (fly) { + mCurrentTrackingMode = TM_LOCAL_ROTATE; + mTrackball.start(Trackball::Local); + } else { + mCurrentTrackingMode = TM_ROTATE_AROUND; + mTrackball.start(Trackball::Around); + } + mTrackball.track(mMouseCoords); + break; + case Qt::MidButton: + if (fly) + mCurrentTrackingMode = TM_FLY_Z; + else + mCurrentTrackingMode = TM_ZOOM; + break; + case Qt::RightButton: + mCurrentTrackingMode = TM_FLY_PAN; + break; + default: + break; } } -void RenderingWidget::mouseReleaseEvent(QMouseEvent*) +void RenderingWidget::mouseReleaseEvent(QMouseEvent *) { - mCurrentTrackingMode = TM_NO_TRACK; - updateGL(); + mCurrentTrackingMode = TM_NO_TRACK; + updateGL(); } -void RenderingWidget::mouseMoveEvent(QMouseEvent* e) +void RenderingWidget::mouseMoveEvent(QMouseEvent *e) { - // tracking - if(mCurrentTrackingMode != TM_NO_TRACK) - { - float dx = float(e->x() - mMouseCoords.x()) / float(mCamera.vpWidth()); - float dy = - float(e->y() - mMouseCoords.y()) / float(mCamera.vpHeight()); - - // speedup the transformations - if(e->modifiers() & Qt::ShiftModifier) - { - dx *= 10.; - dy *= 10.; - } - - switch(mCurrentTrackingMode) - { - case TM_ROTATE_AROUND: - case TM_LOCAL_ROTATE: - if (mRotationMode==RotationStable) - { - // use the stable trackball implementation mapping - // the 2D coordinates to 3D points on a sphere. - mTrackball.track(Vector2i(e->pos().x(), e->pos().y())); - } - else - { - // standard approach mapping the x and y displacements as rotations - // around the camera's X and Y axes. - Quaternionf q = AngleAxisf( dx*M_PI, Vector3f::UnitY()) - * AngleAxisf(-dy*M_PI, Vector3f::UnitX()); - if (mCurrentTrackingMode==TM_LOCAL_ROTATE) - mCamera.localRotate(q); - else - mCamera.rotateAroundTarget(q); - } - break; - case TM_ZOOM : - mCamera.zoom(dy*100); - break; - case TM_FLY_Z : - mCamera.localTranslate(Vector3f(0, 0, -dy*200)); - break; - case TM_FLY_PAN : - mCamera.localTranslate(Vector3f(dx*200, dy*200, 0)); - break; - default: - break; - } + // tracking + if (mCurrentTrackingMode != TM_NO_TRACK) { + float dx = float(e->x() - mMouseCoords.x()) / float(mCamera.vpWidth()); + float dy = -float(e->y() - mMouseCoords.y()) / float(mCamera.vpHeight()); + + // speedup the transformations + if (e->modifiers() & Qt::ShiftModifier) { + dx *= 10.; + dy *= 10.; + } - updateGL(); + switch (mCurrentTrackingMode) { + case TM_ROTATE_AROUND: + case TM_LOCAL_ROTATE: + if (mRotationMode == RotationStable) { + // use the stable trackball implementation mapping + // the 2D coordinates to 3D points on a sphere. + mTrackball.track(Vector2i(e->pos().x(), e->pos().y())); + } else { + // standard approach mapping the x and y displacements as rotations + // around the camera's X and Y axes. + Quaternionf q = AngleAxisf(dx * M_PI, Vector3f::UnitY()) * AngleAxisf(-dy * M_PI, Vector3f::UnitX()); + if (mCurrentTrackingMode == TM_LOCAL_ROTATE) + mCamera.localRotate(q); + else + mCamera.rotateAroundTarget(q); + } + break; + case TM_ZOOM: + mCamera.zoom(dy * 100); + break; + case TM_FLY_Z: + mCamera.localTranslate(Vector3f(0, 0, -dy * 200)); + break; + case TM_FLY_PAN: + mCamera.localTranslate(Vector3f(dx * 200, dy * 200, 0)); + break; + default: + break; } - mMouseCoords = Vector2i(e->pos().x(), e->pos().y()); + updateGL(); + } + + mMouseCoords = Vector2i(e->pos().x(), e->pos().y()); } void RenderingWidget::paintGL() { glEnable(GL_DEPTH_TEST); glDisable(GL_CULL_FACE); - glPolygonMode(GL_FRONT_AND_BACK,GL_FILL); + glPolygonMode(GL_FRONT_AND_BACK, GL_FILL); glDisable(GL_COLOR_MATERIAL); glDisable(GL_BLEND); glDisable(GL_ALPHA_TEST); @@ -487,30 +447,17 @@ void RenderingWidget::initializeGL() mInitFrame.position = mCamera.viewMatrix().translation(); } -void RenderingWidget::resizeGL(int width, int height) -{ - mCamera.setViewport(width,height); -} +void RenderingWidget::resizeGL(int width, int height) { mCamera.setViewport(width, height); } -void RenderingWidget::setNavMode(int m) -{ - mNavMode = NavMode(m); -} +void RenderingWidget::setNavMode(int m) { mNavMode = NavMode(m); } -void RenderingWidget::setLerpMode(int m) -{ - mLerpMode = LerpMode(m); -} +void RenderingWidget::setLerpMode(int m) { mLerpMode = LerpMode(m); } -void RenderingWidget::setRotationMode(int m) -{ - mRotationMode = RotationMode(m); -} +void RenderingWidget::setRotationMode(int m) { mRotationMode = RotationMode(m); } void RenderingWidget::resetCamera() { - if (mAnimate) - stopAnimation(); + if (mAnimate) stopAnimation(); m_timeline.clear(); Frame aux0 = mCamera.frame(); aux0.orientation = aux0.orientation.inverse(); @@ -525,13 +472,13 @@ void RenderingWidget::resetCamera() aux1.orientation = aux1.orientation.inverse(); aux1.position = mCamera.viewMatrix().translation(); float duration = aux0.orientation.angularDistance(aux1.orientation) * 0.9; - if (duration<0.1) duration = 0.1; + if (duration < 0.1) duration = 0.1; // put the camera at that time step: - aux1 = aux0.lerp(duration/2,mInitFrame); + aux1 = aux0.lerp(duration / 2, mInitFrame); // and make it look at the target again aux1.orientation = aux1.orientation.inverse(); - aux1.position = - (aux1.orientation * aux1.position); + aux1.position = -(aux1.orientation * aux1.position); mCamera.setFrame(aux1); mCamera.setTarget(Vector3f::Zero()); @@ -544,27 +491,27 @@ void RenderingWidget::resetCamera() m_alpha = 0; animate(); connect(&m_timer, SIGNAL(timeout()), this, SLOT(animate())); - m_timer.start(1000/30); + m_timer.start(1000 / 30); mAnimate = true; } -QWidget* RenderingWidget::createNavigationControlWidget() +QWidget *RenderingWidget::createNavigationControlWidget() { - QWidget* panel = new QWidget(); - QVBoxLayout* layout = new QVBoxLayout(); + QWidget *panel = new QWidget(); + QVBoxLayout *layout = new QVBoxLayout(); { - QPushButton* but = new QPushButton("reset"); + QPushButton *but = new QPushButton("reset"); but->setToolTip("move the camera to initial position (with animation)"); layout->addWidget(but); connect(but, SIGNAL(clicked()), this, SLOT(resetCamera())); } { // navigation mode - QGroupBox* box = new QGroupBox("navigation mode"); - QVBoxLayout* boxLayout = new QVBoxLayout; - QButtonGroup* group = new QButtonGroup(panel); - QRadioButton* but; + QGroupBox *box = new QGroupBox("navigation mode"); + QVBoxLayout *boxLayout = new QVBoxLayout; + QButtonGroup *group = new QButtonGroup(panel); + QRadioButton *but; but = new QRadioButton("turn around"); but->setToolTip("look around an object"); group->addButton(but, NavTurnAround); @@ -580,10 +527,10 @@ QWidget* RenderingWidget::createNavigationControlWidget() } { // track ball, rotation mode - QGroupBox* box = new QGroupBox("rotation mode"); - QVBoxLayout* boxLayout = new QVBoxLayout; - QButtonGroup* group = new QButtonGroup(panel); - QRadioButton* but; + QGroupBox *box = new QGroupBox("rotation mode"); + QVBoxLayout *boxLayout = new QVBoxLayout; + QButtonGroup *group = new QButtonGroup(panel); + QRadioButton *but; but = new QRadioButton("stable trackball"); group->addButton(but, RotationStable); boxLayout->addWidget(but); @@ -591,7 +538,8 @@ QWidget* RenderingWidget::createNavigationControlWidget() but = new QRadioButton("standard rotation"); group->addButton(but, RotationStandard); boxLayout->addWidget(but); - but->setToolTip("standard approach mapping the x and y displacements\nas rotations around the camera's X and Y axes"); + but->setToolTip( + "standard approach mapping the x and y displacements\nas rotations around the camera's X and Y axes"); group->button(mRotationMode)->setChecked(true); connect(group, SIGNAL(buttonClicked(int)), this, SLOT(setRotationMode(int))); box->setLayout(boxLayout); @@ -599,10 +547,10 @@ QWidget* RenderingWidget::createNavigationControlWidget() } { // interpolation mode - QGroupBox* box = new QGroupBox("spherical interpolation"); - QVBoxLayout* boxLayout = new QVBoxLayout; - QButtonGroup* group = new QButtonGroup(panel); - QRadioButton* but; + QGroupBox *box = new QGroupBox("spherical interpolation"); + QVBoxLayout *boxLayout = new QVBoxLayout; + QButtonGroup *group = new QButtonGroup(panel); + QRadioButton *but; but = new QRadioButton("quaternion slerp"); group->addButton(but, LerpQuaternion); boxLayout->addWidget(but); @@ -616,7 +564,7 @@ QWidget* RenderingWidget::createNavigationControlWidget() box->setLayout(boxLayout); layout->addWidget(box); } - layout->addItem(new QSpacerItem(0,0,QSizePolicy::Minimum,QSizePolicy::Expanding)); + layout->addItem(new QSpacerItem(0, 0, QSizePolicy::Minimum, QSizePolicy::Expanding)); panel->setLayout(layout); return panel; } @@ -626,7 +574,7 @@ QuaternionDemo::QuaternionDemo() mRenderingWidget = new RenderingWidget(); setCentralWidget(mRenderingWidget); - QDockWidget* panel = new QDockWidget("navigation", this); + QDockWidget *panel = new QDockWidget("navigation", this); panel->setAllowedAreas((QFlags)(Qt::RightDockWidgetArea | Qt::LeftDockWidgetArea)); addDockWidget(Qt::RightDockWidgetArea, panel); panel->setWidget(mRenderingWidget->createNavigationControlWidget()); @@ -647,10 +595,9 @@ int main(int argc, char *argv[]) QApplication app(argc, argv); QuaternionDemo demo; - demo.resize(600,500); + demo.resize(600, 500); demo.show(); return app.exec(); } #include "quaternion_demo.moc" - diff --git a/filmulator-gui/core/nlmeans/eigen/demos/opengl/quaternion_demo.h b/filmulator-gui/core/nlmeans/eigen/demos/opengl/quaternion_demo.h index dbff46c3..ce63c26e 100644 --- a/filmulator-gui/core/nlmeans/eigen/demos/opengl/quaternion_demo.h +++ b/filmulator-gui/core/nlmeans/eigen/demos/opengl/quaternion_demo.h @@ -10,105 +10,93 @@ #ifndef EIGEN_QUATERNION_DEMO_H #define EIGEN_QUATERNION_DEMO_H -#include "gpuhelper.h" #include "camera.h" +#include "gpuhelper.h" #include "trackball.h" -#include #include #include -#include #include +#include +#include class RenderingWidget : public QGLWidget { Q_OBJECT - typedef std::map TimeLine; - TimeLine m_timeline; - Frame lerpFrame(float t); - - Frame mInitFrame; - bool mAnimate; - float m_alpha; - - enum TrackMode { - TM_NO_TRACK=0, TM_ROTATE_AROUND, TM_ZOOM, - TM_LOCAL_ROTATE, TM_FLY_Z, TM_FLY_PAN - }; - - enum NavMode { - NavTurnAround, - NavFly - }; - - enum LerpMode { - LerpQuaternion, - LerpEulerAngles - }; - - enum RotationMode { - RotationStable, - RotationStandard - }; - - Camera mCamera; - TrackMode mCurrentTrackingMode; - NavMode mNavMode; - LerpMode mLerpMode; - RotationMode mRotationMode; - Vector2i mMouseCoords; - Trackball mTrackball; - - QTimer m_timer; - - void setupCamera(); - - std::vector mVertices; - std::vector mNormals; - std::vector mIndices; - - protected slots: - - virtual void animate(void); - virtual void drawScene(void); - - virtual void grabFrame(void); - virtual void stopAnimation(); - - virtual void setNavMode(int); - virtual void setLerpMode(int); - virtual void setRotationMode(int); - virtual void resetCamera(); - - protected: - - virtual void initializeGL(); - virtual void resizeGL(int width, int height); - virtual void paintGL(); - - //-------------------------------------------------------------------------------- - virtual void mousePressEvent(QMouseEvent * e); - virtual void mouseReleaseEvent(QMouseEvent * e); - virtual void mouseMoveEvent(QMouseEvent * e); - virtual void keyPressEvent(QKeyEvent * e); - //-------------------------------------------------------------------------------- - - public: - EIGEN_MAKE_ALIGNED_OPERATOR_NEW - - RenderingWidget(); - ~RenderingWidget() { } - - QWidget* createNavigationControlWidget(); + typedef std::map TimeLine; + TimeLine m_timeline; + Frame lerpFrame(float t); + + Frame mInitFrame; + bool mAnimate; + float m_alpha; + + enum TrackMode { TM_NO_TRACK = 0, TM_ROTATE_AROUND, TM_ZOOM, TM_LOCAL_ROTATE, TM_FLY_Z, TM_FLY_PAN }; + + enum NavMode { NavTurnAround, NavFly }; + + enum LerpMode { LerpQuaternion, LerpEulerAngles }; + + enum RotationMode { RotationStable, RotationStandard }; + + Camera mCamera; + TrackMode mCurrentTrackingMode; + NavMode mNavMode; + LerpMode mLerpMode; + RotationMode mRotationMode; + Vector2i mMouseCoords; + Trackball mTrackball; + + QTimer m_timer; + + void setupCamera(); + + std::vector mVertices; + std::vector mNormals; + std::vector mIndices; + +protected slots: + + virtual void animate(void); + virtual void drawScene(void); + + virtual void grabFrame(void); + virtual void stopAnimation(); + + virtual void setNavMode(int); + virtual void setLerpMode(int); + virtual void setRotationMode(int); + virtual void resetCamera(); + +protected: + virtual void initializeGL(); + virtual void resizeGL(int width, int height); + virtual void paintGL(); + + //-------------------------------------------------------------------------------- + virtual void mousePressEvent(QMouseEvent *e); + virtual void mouseReleaseEvent(QMouseEvent *e); + virtual void mouseMoveEvent(QMouseEvent *e); + virtual void keyPressEvent(QKeyEvent *e); + //-------------------------------------------------------------------------------- + +public: + EIGEN_MAKE_ALIGNED_OPERATOR_NEW + + RenderingWidget(); + ~RenderingWidget() {} + + QWidget *createNavigationControlWidget(); }; class QuaternionDemo : public QMainWindow { Q_OBJECT - public: - QuaternionDemo(); - protected: - RenderingWidget* mRenderingWidget; +public: + QuaternionDemo(); + +protected: + RenderingWidget *mRenderingWidget; }; -#endif // EIGEN_QUATERNION_DEMO_H +#endif// EIGEN_QUATERNION_DEMO_H diff --git a/filmulator-gui/core/nlmeans/eigen/demos/opengl/trackball.cpp b/filmulator-gui/core/nlmeans/eigen/demos/opengl/trackball.cpp index 7c2da8e9..273dec9c 100644 --- a/filmulator-gui/core/nlmeans/eigen/demos/opengl/trackball.cpp +++ b/filmulator-gui/core/nlmeans/eigen/demos/opengl/trackball.cpp @@ -12,21 +12,18 @@ using namespace Eigen; -void Trackball::track(const Vector2i& point2D) +void Trackball::track(const Vector2i &point2D) { - if (mpCamera==0) - return; + if (mpCamera == 0) return; Vector3f newPoint3D; bool newPointOk = mapToSphere(point2D, newPoint3D); - if (mLastPointOk && newPointOk) - { + if (mLastPointOk && newPointOk) { Vector3f axis = mLastPoint3D.cross(newPoint3D).normalized(); float cos_angle = mLastPoint3D.dot(newPoint3D); - if ( std::abs(cos_angle) < 1.0 ) - { + if (std::abs(cos_angle) < 1.0) { float angle = 2. * acos(cos_angle); - if (mMode==Around) + if (mMode == Around) mpCamera->rotateAroundTarget(Quaternionf(AngleAxisf(angle, axis))); else mpCamera->localRotate(Quaternionf(AngleAxisf(-angle, axis))); @@ -37,23 +34,20 @@ void Trackball::track(const Vector2i& point2D) mLastPointOk = newPointOk; } -bool Trackball::mapToSphere(const Vector2i& p2, Vector3f& v3) +bool Trackball::mapToSphere(const Vector2i &p2, Vector3f &v3) { - if ((p2.x() >= 0) && (p2.x() <= int(mpCamera->vpWidth())) && - (p2.y() >= 0) && (p2.y() <= int(mpCamera->vpHeight())) ) - { - double x = (double)(p2.x() - 0.5*mpCamera->vpWidth()) / (double)mpCamera->vpWidth(); - double y = (double)(0.5*mpCamera->vpHeight() - p2.y()) / (double)mpCamera->vpHeight(); - double sinx = sin(M_PI * x * 0.5); - double siny = sin(M_PI * y * 0.5); - double sinx2siny2 = sinx * sinx + siny * siny; + if ((p2.x() >= 0) && (p2.x() <= int(mpCamera->vpWidth())) && (p2.y() >= 0) && (p2.y() <= int(mpCamera->vpHeight()))) { + double x = (double)(p2.x() - 0.5 * mpCamera->vpWidth()) / (double)mpCamera->vpWidth(); + double y = (double)(0.5 * mpCamera->vpHeight() - p2.y()) / (double)mpCamera->vpHeight(); + double sinx = sin(M_PI * x * 0.5); + double siny = sin(M_PI * y * 0.5); + double sinx2siny2 = sinx * sinx + siny * siny; v3.x() = sinx; v3.y() = siny; v3.z() = sinx2siny2 < 1.0 ? sqrt(1.0 - sinx2siny2) : 0.0; return true; - } - else + } else return false; } diff --git a/filmulator-gui/core/nlmeans/eigen/demos/opengl/trackball.h b/filmulator-gui/core/nlmeans/eigen/demos/opengl/trackball.h index 1ea842f1..e1412340 100644 --- a/filmulator-gui/core/nlmeans/eigen/demos/opengl/trackball.h +++ b/filmulator-gui/core/nlmeans/eigen/demos/opengl/trackball.h @@ -16,27 +16,28 @@ class Camera; class Trackball { - public: +public: + enum Mode { Around, Local }; - enum Mode {Around, Local}; + Trackball() : mpCamera(0) {} - Trackball() : mpCamera(0) {} + void start(Mode m = Around) + { + mMode = m; + mLastPointOk = false; + } - void start(Mode m = Around) { mMode = m; mLastPointOk = false; } + void setCamera(Camera *pCam) { mpCamera = pCam; } - void setCamera(Camera* pCam) { mpCamera = pCam; } + void track(const Eigen::Vector2i &newPoint2D); - void track(const Eigen::Vector2i& newPoint2D); - - protected: - - bool mapToSphere( const Eigen::Vector2i& p2, Eigen::Vector3f& v3); - - Camera* mpCamera; - Eigen::Vector3f mLastPoint3D; - Mode mMode; - bool mLastPointOk; +protected: + bool mapToSphere(const Eigen::Vector2i &p2, Eigen::Vector3f &v3); + Camera *mpCamera; + Eigen::Vector3f mLastPoint3D; + Mode mMode; + bool mLastPointOk; }; -#endif // EIGEN_TRACKBALL_H +#endif// EIGEN_TRACKBALL_H diff --git a/filmulator-gui/core/nlmeans/eigen/doc/examples/CustomizingEigen_Inheritance.cpp b/filmulator-gui/core/nlmeans/eigen/doc/examples/CustomizingEigen_Inheritance.cpp index 48df64ee..e099b2e3 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/examples/CustomizingEigen_Inheritance.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/examples/CustomizingEigen_Inheritance.cpp @@ -4,21 +4,17 @@ class MyVectorType : public Eigen::VectorXd { public: - MyVectorType(void):Eigen::VectorXd() {} + MyVectorType(void) : Eigen::VectorXd() {} - // This constructor allows you to construct MyVectorType from Eigen expressions - template - MyVectorType(const Eigen::MatrixBase& other) - : Eigen::VectorXd(other) - { } + // This constructor allows you to construct MyVectorType from Eigen expressions + template MyVectorType(const Eigen::MatrixBase &other) : Eigen::VectorXd(other) {} - // This method allows you to assign Eigen expressions to MyVectorType - template - MyVectorType& operator=(const Eigen::MatrixBase & other) - { - this->Eigen::VectorXd::operator=(other); - return *this; - } + // This method allows you to assign Eigen expressions to MyVectorType + template MyVectorType &operator=(const Eigen::MatrixBase &other) + { + this->Eigen::VectorXd::operator=(other); + return *this; + } }; int main() diff --git a/filmulator-gui/core/nlmeans/eigen/doc/examples/Cwise_erf.cpp b/filmulator-gui/core/nlmeans/eigen/doc/examples/Cwise_erf.cpp index e7cd2c1c..7bc7ebb6 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/examples/Cwise_erf.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/examples/Cwise_erf.cpp @@ -1,9 +1,9 @@ #include -#include #include +#include using namespace Eigen; int main() { - Array4d v(-0.5,2,0,-7); + Array4d v(-0.5, 2, 0, -7); std::cout << v.erf() << std::endl; } diff --git a/filmulator-gui/core/nlmeans/eigen/doc/examples/Cwise_erfc.cpp b/filmulator-gui/core/nlmeans/eigen/doc/examples/Cwise_erfc.cpp index d8bb04c3..2772e2fe 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/examples/Cwise_erfc.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/examples/Cwise_erfc.cpp @@ -1,9 +1,9 @@ #include -#include #include +#include using namespace Eigen; int main() { - Array4d v(-0.5,2,0,-7); + Array4d v(-0.5, 2, 0, -7); std::cout << v.erfc() << std::endl; } diff --git a/filmulator-gui/core/nlmeans/eigen/doc/examples/Cwise_lgamma.cpp b/filmulator-gui/core/nlmeans/eigen/doc/examples/Cwise_lgamma.cpp index 6bfaccbc..bee84121 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/examples/Cwise_lgamma.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/examples/Cwise_lgamma.cpp @@ -1,9 +1,9 @@ #include -#include #include +#include using namespace Eigen; int main() { - Array4d v(0.5,10,0,-1); + Array4d v(0.5, 10, 0, -1); std::cout << v.lgamma() << std::endl; } diff --git a/filmulator-gui/core/nlmeans/eigen/doc/examples/DenseBase_middleCols_int.cpp b/filmulator-gui/core/nlmeans/eigen/doc/examples/DenseBase_middleCols_int.cpp index 0ebd955e..53fdf607 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/examples/DenseBase_middleCols_int.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/examples/DenseBase_middleCols_int.cpp @@ -6,10 +6,10 @@ using namespace std; int main(void) { - int const N = 5; - MatrixXi A(N,N); - A.setRandom(); - cout << "A =\n" << A << '\n' << endl; - cout << "A(1..3,:) =\n" << A.middleCols(1,3) << endl; - return 0; + int const N = 5; + MatrixXi A(N, N); + A.setRandom(); + cout << "A =\n" << A << '\n' << endl; + cout << "A(1..3,:) =\n" << A.middleCols(1, 3) << endl; + return 0; } diff --git a/filmulator-gui/core/nlmeans/eigen/doc/examples/DenseBase_middleRows_int.cpp b/filmulator-gui/core/nlmeans/eigen/doc/examples/DenseBase_middleRows_int.cpp index a6fe9e84..783a9970 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/examples/DenseBase_middleRows_int.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/examples/DenseBase_middleRows_int.cpp @@ -6,10 +6,10 @@ using namespace std; int main(void) { - int const N = 5; - MatrixXi A(N,N); - A.setRandom(); - cout << "A =\n" << A << '\n' << endl; - cout << "A(2..3,:) =\n" << A.middleRows(2,2) << endl; - return 0; + int const N = 5; + MatrixXi A(N, N); + A.setRandom(); + cout << "A =\n" << A << '\n' << endl; + cout << "A(2..3,:) =\n" << A.middleRows(2, 2) << endl; + return 0; } diff --git a/filmulator-gui/core/nlmeans/eigen/doc/examples/DenseBase_template_int_middleCols.cpp b/filmulator-gui/core/nlmeans/eigen/doc/examples/DenseBase_template_int_middleCols.cpp index 6191d79c..30d547be 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/examples/DenseBase_template_int_middleCols.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/examples/DenseBase_template_int_middleCols.cpp @@ -6,10 +6,10 @@ using namespace std; int main(void) { - int const N = 5; - MatrixXi A(N,N); - A.setRandom(); - cout << "A =\n" << A << '\n' << endl; - cout << "A(:,1..3) =\n" << A.middleCols<3>(1) << endl; - return 0; + int const N = 5; + MatrixXi A(N, N); + A.setRandom(); + cout << "A =\n" << A << '\n' << endl; + cout << "A(:,1..3) =\n" << A.middleCols<3>(1) << endl; + return 0; } diff --git a/filmulator-gui/core/nlmeans/eigen/doc/examples/DenseBase_template_int_middleRows.cpp b/filmulator-gui/core/nlmeans/eigen/doc/examples/DenseBase_template_int_middleRows.cpp index 7e8b6573..6c5a851a 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/examples/DenseBase_template_int_middleRows.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/examples/DenseBase_template_int_middleRows.cpp @@ -6,10 +6,10 @@ using namespace std; int main(void) { - int const N = 5; - MatrixXi A(N,N); - A.setRandom(); - cout << "A =\n" << A << '\n' << endl; - cout << "A(1..3,:) =\n" << A.middleRows<3>(1) << endl; - return 0; + int const N = 5; + MatrixXi A(N, N); + A.setRandom(); + cout << "A =\n" << A << '\n' << endl; + cout << "A(1..3,:) =\n" << A.middleRows<3>(1) << endl; + return 0; } diff --git a/filmulator-gui/core/nlmeans/eigen/doc/examples/QuickStart_example.cpp b/filmulator-gui/core/nlmeans/eigen/doc/examples/QuickStart_example.cpp index 7238c0c4..5e4e6538 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/examples/QuickStart_example.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/examples/QuickStart_example.cpp @@ -1,14 +1,14 @@ -#include #include +#include using Eigen::MatrixXd; int main() { - MatrixXd m(2,2); - m(0,0) = 3; - m(1,0) = 2.5; - m(0,1) = -1; - m(1,1) = m(1,0) + m(0,1); + MatrixXd m(2, 2); + m(0, 0) = 3; + m(1, 0) = 2.5; + m(0, 1) = -1; + m(1, 1) = m(1, 0) + m(0, 1); std::cout << m << std::endl; } diff --git a/filmulator-gui/core/nlmeans/eigen/doc/examples/QuickStart_example2_dynamic.cpp b/filmulator-gui/core/nlmeans/eigen/doc/examples/QuickStart_example2_dynamic.cpp index ff6746e2..84553abe 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/examples/QuickStart_example2_dynamic.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/examples/QuickStart_example2_dynamic.cpp @@ -1,13 +1,13 @@ -#include #include +#include using namespace Eigen; using namespace std; int main() { - MatrixXd m = MatrixXd::Random(3,3); - m = (m + MatrixXd::Constant(3,3,1.2)) * 50; + MatrixXd m = MatrixXd::Random(3, 3); + m = (m + MatrixXd::Constant(3, 3, 1.2)) * 50; cout << "m =" << endl << m << endl; VectorXd v(3); v << 1, 2, 3; diff --git a/filmulator-gui/core/nlmeans/eigen/doc/examples/QuickStart_example2_fixed.cpp b/filmulator-gui/core/nlmeans/eigen/doc/examples/QuickStart_example2_fixed.cpp index d9117527..dfc7fb50 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/examples/QuickStart_example2_fixed.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/examples/QuickStart_example2_fixed.cpp @@ -1,5 +1,5 @@ -#include #include +#include using namespace Eigen; using namespace std; @@ -9,7 +9,7 @@ int main() Matrix3d m = Matrix3d::Random(); m = (m + Matrix3d::Constant(1.2)) * 50; cout << "m =" << endl << m << endl; - Vector3d v(1,2,3); - + Vector3d v(1, 2, 3); + cout << "m * v =" << endl << m * v << endl; } diff --git a/filmulator-gui/core/nlmeans/eigen/doc/examples/TemplateKeyword_flexible.cpp b/filmulator-gui/core/nlmeans/eigen/doc/examples/TemplateKeyword_flexible.cpp index 9d85292d..59e509d7 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/examples/TemplateKeyword_flexible.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/examples/TemplateKeyword_flexible.cpp @@ -3,8 +3,8 @@ using namespace Eigen; -template -void copyUpperTriangularPart(MatrixBase& dst, const MatrixBase& src) +template +void copyUpperTriangularPart(MatrixBase &dst, const MatrixBase &src) { /* Note the 'template' keywords in the following line! */ dst.template triangularView() = src.template triangularView(); @@ -12,11 +12,11 @@ void copyUpperTriangularPart(MatrixBase& dst, const MatrixBase() = src.triangularView(); } int main() { - MatrixXf m1 = MatrixXf::Ones(4,4); - MatrixXf m2 = MatrixXf::Random(4,4); + MatrixXf m1 = MatrixXf::Ones(4, 4); + MatrixXf m2 = MatrixXf::Random(4, 4); std::cout << "m2 before copy:" << std::endl; std::cout << m2 << std::endl << std::endl; copyUpperTriangularPart(m2, m1); diff --git a/filmulator-gui/core/nlmeans/eigen/doc/examples/TutorialInplaceLU.cpp b/filmulator-gui/core/nlmeans/eigen/doc/examples/TutorialInplaceLU.cpp index cb9c59b6..93f9c667 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/examples/TutorialInplaceLU.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/examples/TutorialInplaceLU.cpp @@ -1,61 +1,63 @@ #include -struct init { +struct init +{ init() { std::cout << "[" << "init" << "]" << std::endl; } }; init init_obj; // [init] -#include #include +#include using namespace std; using namespace Eigen; int main() { - MatrixXd A(2,2); + MatrixXd A(2, 2); A << 2, -1, 1, 3; cout << "Here is the input matrix A before decomposition:\n" << A << endl; -cout << "[init]" << endl; + cout << "[init]" << endl; -cout << "[declaration]" << endl; - PartialPivLU > lu(A); + cout << "[declaration]" << endl; + PartialPivLU> lu(A); cout << "Here is the input matrix A after decomposition:\n" << A << endl; -cout << "[declaration]" << endl; + cout << "[declaration]" << endl; -cout << "[matrixLU]" << endl; + cout << "[matrixLU]" << endl; cout << "Here is the matrix storing the L and U factors:\n" << lu.matrixLU() << endl; -cout << "[matrixLU]" << endl; + cout << "[matrixLU]" << endl; -cout << "[solve]" << endl; - MatrixXd A0(2,2); A0 << 2, -1, 1, 3; - VectorXd b(2); b << 1, 2; + cout << "[solve]" << endl; + MatrixXd A0(2, 2); + A0 << 2, -1, 1, 3; + VectorXd b(2); + b << 1, 2; VectorXd x = lu.solve(b); cout << "Residual: " << (A0 * x - b).norm() << endl; -cout << "[solve]" << endl; + cout << "[solve]" << endl; -cout << "[modifyA]" << endl; + cout << "[modifyA]" << endl; A << 3, 4, -2, 1; x = lu.solve(b); cout << "Residual: " << (A0 * x - b).norm() << endl; -cout << "[modifyA]" << endl; + cout << "[modifyA]" << endl; -cout << "[recompute]" << endl; - A0 = A; // save A + cout << "[recompute]" << endl; + A0 = A;// save A lu.compute(A); x = lu.solve(b); cout << "Residual: " << (A0 * x - b).norm() << endl; -cout << "[recompute]" << endl; + cout << "[recompute]" << endl; -cout << "[recompute_bis0]" << endl; - MatrixXd A1(2,2); - A1 << 5,-2,3,4; + cout << "[recompute_bis0]" << endl; + MatrixXd A1(2, 2); + A1 << 5, -2, 3, 4; lu.compute(A1); cout << "Here is the input matrix A1 after decomposition:\n" << A1 << endl; -cout << "[recompute_bis0]" << endl; + cout << "[recompute_bis0]" << endl; -cout << "[recompute_bis1]" << endl; + cout << "[recompute_bis1]" << endl; x = lu.solve(b); cout << "Residual: " << (A1 * x - b).norm() << endl; -cout << "[recompute_bis1]" << endl; - + cout << "[recompute_bis1]" << endl; } diff --git a/filmulator-gui/core/nlmeans/eigen/doc/examples/TutorialLinAlgComputeTwice.cpp b/filmulator-gui/core/nlmeans/eigen/doc/examples/TutorialLinAlgComputeTwice.cpp index 06ba6461..83b806fa 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/examples/TutorialLinAlgComputeTwice.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/examples/TutorialLinAlgComputeTwice.cpp @@ -1,23 +1,23 @@ -#include #include +#include using namespace std; using namespace Eigen; int main() { - Matrix2f A, b; - LLT llt; - A << 2, -1, -1, 3; - b << 1, 2, 3, 1; - cout << "Here is the matrix A:\n" << A << endl; - cout << "Here is the right hand side b:\n" << b << endl; - cout << "Computing LLT decomposition..." << endl; - llt.compute(A); - cout << "The solution is:\n" << llt.solve(b) << endl; - A(1,1)++; - cout << "The matrix A is now:\n" << A << endl; - cout << "Computing LLT decomposition..." << endl; - llt.compute(A); - cout << "The solution is now:\n" << llt.solve(b) << endl; + Matrix2f A, b; + LLT llt; + A << 2, -1, -1, 3; + b << 1, 2, 3, 1; + cout << "Here is the matrix A:\n" << A << endl; + cout << "Here is the right hand side b:\n" << b << endl; + cout << "Computing LLT decomposition..." << endl; + llt.compute(A); + cout << "The solution is:\n" << llt.solve(b) << endl; + A(1, 1)++; + cout << "The matrix A is now:\n" << A << endl; + cout << "Computing LLT decomposition..." << endl; + llt.compute(A); + cout << "The solution is now:\n" << llt.solve(b) << endl; } diff --git a/filmulator-gui/core/nlmeans/eigen/doc/examples/TutorialLinAlgExComputeSolveError.cpp b/filmulator-gui/core/nlmeans/eigen/doc/examples/TutorialLinAlgExComputeSolveError.cpp index f362fb71..3c586255 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/examples/TutorialLinAlgExComputeSolveError.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/examples/TutorialLinAlgExComputeSolveError.cpp @@ -1,14 +1,14 @@ -#include #include +#include using namespace std; using namespace Eigen; int main() { - MatrixXd A = MatrixXd::Random(100,100); - MatrixXd b = MatrixXd::Random(100,50); - MatrixXd x = A.fullPivLu().solve(b); - double relative_error = (A*x - b).norm() / b.norm(); // norm() is L2 norm - cout << "The relative error is:\n" << relative_error << endl; + MatrixXd A = MatrixXd::Random(100, 100); + MatrixXd b = MatrixXd::Random(100, 50); + MatrixXd x = A.fullPivLu().solve(b); + double relative_error = (A * x - b).norm() / b.norm();// norm() is L2 norm + cout << "The relative error is:\n" << relative_error << endl; } diff --git a/filmulator-gui/core/nlmeans/eigen/doc/examples/TutorialLinAlgExSolveColPivHouseholderQR.cpp b/filmulator-gui/core/nlmeans/eigen/doc/examples/TutorialLinAlgExSolveColPivHouseholderQR.cpp index 3a99a94d..e19e854d 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/examples/TutorialLinAlgExSolveColPivHouseholderQR.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/examples/TutorialLinAlgExSolveColPivHouseholderQR.cpp @@ -1,17 +1,17 @@ -#include #include +#include using namespace std; using namespace Eigen; int main() { - Matrix3f A; - Vector3f b; - A << 1,2,3, 4,5,6, 7,8,10; - b << 3, 3, 4; - cout << "Here is the matrix A:\n" << A << endl; - cout << "Here is the vector b:\n" << b << endl; - Vector3f x = A.colPivHouseholderQr().solve(b); - cout << "The solution is:\n" << x << endl; + Matrix3f A; + Vector3f b; + A << 1, 2, 3, 4, 5, 6, 7, 8, 10; + b << 3, 3, 4; + cout << "Here is the matrix A:\n" << A << endl; + cout << "Here is the vector b:\n" << b << endl; + Vector3f x = A.colPivHouseholderQr().solve(b); + cout << "The solution is:\n" << x << endl; } diff --git a/filmulator-gui/core/nlmeans/eigen/doc/examples/TutorialLinAlgExSolveLDLT.cpp b/filmulator-gui/core/nlmeans/eigen/doc/examples/TutorialLinAlgExSolveLDLT.cpp index f8beacd2..2f9d4b49 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/examples/TutorialLinAlgExSolveLDLT.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/examples/TutorialLinAlgExSolveLDLT.cpp @@ -1,16 +1,16 @@ -#include #include +#include using namespace std; using namespace Eigen; int main() { - Matrix2f A, b; - A << 2, -1, -1, 3; - b << 1, 2, 3, 1; - cout << "Here is the matrix A:\n" << A << endl; - cout << "Here is the right hand side b:\n" << b << endl; - Matrix2f x = A.ldlt().solve(b); - cout << "The solution is:\n" << x << endl; + Matrix2f A, b; + A << 2, -1, -1, 3; + b << 1, 2, 3, 1; + cout << "Here is the matrix A:\n" << A << endl; + cout << "Here is the right hand side b:\n" << b << endl; + Matrix2f x = A.ldlt().solve(b); + cout << "The solution is:\n" << x << endl; } diff --git a/filmulator-gui/core/nlmeans/eigen/doc/examples/TutorialLinAlgInverseDeterminant.cpp b/filmulator-gui/core/nlmeans/eigen/doc/examples/TutorialLinAlgInverseDeterminant.cpp index 14dde5b3..615f7943 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/examples/TutorialLinAlgInverseDeterminant.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/examples/TutorialLinAlgInverseDeterminant.cpp @@ -1,16 +1,14 @@ -#include #include +#include using namespace std; using namespace Eigen; int main() { - Matrix3f A; - A << 1, 2, 1, - 2, 1, 0, - -1, 1, 2; - cout << "Here is the matrix A:\n" << A << endl; - cout << "The determinant of A is " << A.determinant() << endl; - cout << "The inverse of A is:\n" << A.inverse() << endl; + Matrix3f A; + A << 1, 2, 1, 2, 1, 0, -1, 1, 2; + cout << "Here is the matrix A:\n" << A << endl; + cout << "The determinant of A is " << A.determinant() << endl; + cout << "The inverse of A is:\n" << A.inverse() << endl; } diff --git a/filmulator-gui/core/nlmeans/eigen/doc/examples/TutorialLinAlgRankRevealing.cpp b/filmulator-gui/core/nlmeans/eigen/doc/examples/TutorialLinAlgRankRevealing.cpp index c5165077..db649ce8 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/examples/TutorialLinAlgRankRevealing.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/examples/TutorialLinAlgRankRevealing.cpp @@ -1,20 +1,17 @@ -#include #include +#include using namespace std; using namespace Eigen; int main() { - Matrix3f A; - A << 1, 2, 5, - 2, 1, 4, - 3, 0, 3; - cout << "Here is the matrix A:\n" << A << endl; - FullPivLU lu_decomp(A); - cout << "The rank of A is " << lu_decomp.rank() << endl; - cout << "Here is a matrix whose columns form a basis of the null-space of A:\n" - << lu_decomp.kernel() << endl; - cout << "Here is a matrix whose columns form a basis of the column-space of A:\n" - << lu_decomp.image(A) << endl; // yes, have to pass the original A + Matrix3f A; + A << 1, 2, 5, 2, 1, 4, 3, 0, 3; + cout << "Here is the matrix A:\n" << A << endl; + FullPivLU lu_decomp(A); + cout << "The rank of A is " << lu_decomp.rank() << endl; + cout << "Here is a matrix whose columns form a basis of the null-space of A:\n" << lu_decomp.kernel() << endl; + cout << "Here is a matrix whose columns form a basis of the column-space of A:\n" + << lu_decomp.image(A) << endl;// yes, have to pass the original A } diff --git a/filmulator-gui/core/nlmeans/eigen/doc/examples/TutorialLinAlgSVDSolve.cpp b/filmulator-gui/core/nlmeans/eigen/doc/examples/TutorialLinAlgSVDSolve.cpp index f109f04e..edd6cfcf 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/examples/TutorialLinAlgSVDSolve.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/examples/TutorialLinAlgSVDSolve.cpp @@ -1,15 +1,14 @@ -#include #include +#include using namespace std; using namespace Eigen; int main() { - MatrixXf A = MatrixXf::Random(3, 2); - cout << "Here is the matrix A:\n" << A << endl; - VectorXf b = VectorXf::Random(3); - cout << "Here is the right hand side b:\n" << b << endl; - cout << "The least-squares solution is:\n" - << A.bdcSvd(ComputeThinU | ComputeThinV).solve(b) << endl; + MatrixXf A = MatrixXf::Random(3, 2); + cout << "Here is the matrix A:\n" << A << endl; + VectorXf b = VectorXf::Random(3); + cout << "Here is the right hand side b:\n" << b << endl; + cout << "The least-squares solution is:\n" << A.bdcSvd(ComputeThinU | ComputeThinV).solve(b) << endl; } diff --git a/filmulator-gui/core/nlmeans/eigen/doc/examples/TutorialLinAlgSelfAdjointEigenSolver.cpp b/filmulator-gui/core/nlmeans/eigen/doc/examples/TutorialLinAlgSelfAdjointEigenSolver.cpp index 8d1d1ed6..11ff1d18 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/examples/TutorialLinAlgSelfAdjointEigenSolver.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/examples/TutorialLinAlgSelfAdjointEigenSolver.cpp @@ -1,18 +1,18 @@ -#include #include +#include using namespace std; using namespace Eigen; int main() { - Matrix2f A; - A << 1, 2, 2, 3; - cout << "Here is the matrix A:\n" << A << endl; - SelfAdjointEigenSolver eigensolver(A); - if (eigensolver.info() != Success) abort(); - cout << "The eigenvalues of A are:\n" << eigensolver.eigenvalues() << endl; - cout << "Here's a matrix whose columns are eigenvectors of A \n" - << "corresponding to these eigenvalues:\n" - << eigensolver.eigenvectors() << endl; + Matrix2f A; + A << 1, 2, 2, 3; + cout << "Here is the matrix A:\n" << A << endl; + SelfAdjointEigenSolver eigensolver(A); + if (eigensolver.info() != Success) abort(); + cout << "The eigenvalues of A are:\n" << eigensolver.eigenvalues() << endl; + cout << "Here's a matrix whose columns are eigenvectors of A \n" + << "corresponding to these eigenvalues:\n" + << eigensolver.eigenvectors() << endl; } diff --git a/filmulator-gui/core/nlmeans/eigen/doc/examples/TutorialLinAlgSetThreshold.cpp b/filmulator-gui/core/nlmeans/eigen/doc/examples/TutorialLinAlgSetThreshold.cpp index 3956b13a..5a2e7c9a 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/examples/TutorialLinAlgSetThreshold.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/examples/TutorialLinAlgSetThreshold.cpp @@ -1,16 +1,15 @@ -#include #include +#include using namespace std; using namespace Eigen; int main() { - Matrix2d A; - A << 2, 1, - 2, 0.9999999999; - FullPivLU lu(A); - cout << "By default, the rank of A is found to be " << lu.rank() << endl; - lu.setThreshold(1e-5); - cout << "With threshold 1e-5, the rank of A is found to be " << lu.rank() << endl; + Matrix2d A; + A << 2, 1, 2, 0.9999999999; + FullPivLU lu(A); + cout << "By default, the rank of A is found to be " << lu.rank() << endl; + lu.setThreshold(1e-5); + cout << "With threshold 1e-5, the rank of A is found to be " << lu.rank() << endl; } diff --git a/filmulator-gui/core/nlmeans/eigen/doc/examples/Tutorial_ArrayClass_accessors.cpp b/filmulator-gui/core/nlmeans/eigen/doc/examples/Tutorial_ArrayClass_accessors.cpp index dc720ff5..a84d6826 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/examples/Tutorial_ArrayClass_accessors.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/examples/Tutorial_ArrayClass_accessors.cpp @@ -6,19 +6,20 @@ using namespace std; int main() { - ArrayXXf m(2,2); - + ArrayXXf m(2, 2); + // assign some values coefficient by coefficient - m(0,0) = 1.0; m(0,1) = 2.0; - m(1,0) = 3.0; m(1,1) = m(0,1) + m(1,0); - + m(0, 0) = 1.0; + m(0, 1) = 2.0; + m(1, 0) = 3.0; + m(1, 1) = m(0, 1) + m(1, 0); + // print values to standard output cout << m << endl << endl; - + // using the comma-initializer is also allowed - m << 1.0,2.0, - 3.0,4.0; - + m << 1.0, 2.0, 3.0, 4.0; + // print values to standard output cout << m << endl; } diff --git a/filmulator-gui/core/nlmeans/eigen/doc/examples/Tutorial_ArrayClass_addition.cpp b/filmulator-gui/core/nlmeans/eigen/doc/examples/Tutorial_ArrayClass_addition.cpp index 480ffb00..3378e601 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/examples/Tutorial_ArrayClass_addition.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/examples/Tutorial_ArrayClass_addition.cpp @@ -6,15 +6,11 @@ using namespace std; int main() { - ArrayXXf a(3,3); - ArrayXXf b(3,3); - a << 1,2,3, - 4,5,6, - 7,8,9; - b << 1,2,3, - 1,2,3, - 1,2,3; - + ArrayXXf a(3, 3); + ArrayXXf b(3, 3); + a << 1, 2, 3, 4, 5, 6, 7, 8, 9; + b << 1, 2, 3, 1, 2, 3, 1, 2, 3; + // Adding two arrays cout << "a + b = " << endl << a + b << endl << endl; diff --git a/filmulator-gui/core/nlmeans/eigen/doc/examples/Tutorial_ArrayClass_cwise_other.cpp b/filmulator-gui/core/nlmeans/eigen/doc/examples/Tutorial_ArrayClass_cwise_other.cpp index d9046c63..1f5c9030 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/examples/Tutorial_ArrayClass_cwise_other.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/examples/Tutorial_ArrayClass_cwise_other.cpp @@ -8,12 +8,8 @@ int main() { ArrayXf a = ArrayXf::Random(5); a *= 2; - cout << "a =" << endl - << a << endl; - cout << "a.abs() =" << endl - << a.abs() << endl; - cout << "a.abs().sqrt() =" << endl - << a.abs().sqrt() << endl; - cout << "a.min(a.abs().sqrt()) =" << endl - << a.min(a.abs().sqrt()) << endl; + cout << "a =" << endl << a << endl; + cout << "a.abs() =" << endl << a.abs() << endl; + cout << "a.abs().sqrt() =" << endl << a.abs().sqrt() << endl; + cout << "a.min(a.abs().sqrt()) =" << endl << a.min(a.abs().sqrt()) << endl; } diff --git a/filmulator-gui/core/nlmeans/eigen/doc/examples/Tutorial_ArrayClass_interop.cpp b/filmulator-gui/core/nlmeans/eigen/doc/examples/Tutorial_ArrayClass_interop.cpp index 371f0706..72980bc4 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/examples/Tutorial_ArrayClass_interop.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/examples/Tutorial_ArrayClass_interop.cpp @@ -6,15 +6,13 @@ using namespace std; int main() { - MatrixXf m(2,2); - MatrixXf n(2,2); - MatrixXf result(2,2); + MatrixXf m(2, 2); + MatrixXf n(2, 2); + MatrixXf result(2, 2); + + m << 1, 2, 3, 4; + n << 5, 6, 7, 8; - m << 1,2, - 3,4; - n << 5,6, - 7,8; - result = (m.array() + 4).matrix() * m; cout << "-- Combination 1: --" << endl << result << endl << endl; result = (m.array() * n.array()).matrix() * m; diff --git a/filmulator-gui/core/nlmeans/eigen/doc/examples/Tutorial_ArrayClass_interop_matrix.cpp b/filmulator-gui/core/nlmeans/eigen/doc/examples/Tutorial_ArrayClass_interop_matrix.cpp index 10142751..c8185c54 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/examples/Tutorial_ArrayClass_interop_matrix.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/examples/Tutorial_ArrayClass_interop_matrix.cpp @@ -6,14 +6,12 @@ using namespace std; int main() { - MatrixXf m(2,2); - MatrixXf n(2,2); - MatrixXf result(2,2); + MatrixXf m(2, 2); + MatrixXf n(2, 2); + MatrixXf result(2, 2); - m << 1,2, - 3,4; - n << 5,6, - 7,8; + m << 1, 2, 3, 4; + n << 5, 6, 7, 8; result = m * n; cout << "-- Matrix m*n: --" << endl << result << endl << endl; diff --git a/filmulator-gui/core/nlmeans/eigen/doc/examples/Tutorial_ArrayClass_mult.cpp b/filmulator-gui/core/nlmeans/eigen/doc/examples/Tutorial_ArrayClass_mult.cpp index 6cb439ff..5c46edb6 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/examples/Tutorial_ArrayClass_mult.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/examples/Tutorial_ArrayClass_mult.cpp @@ -6,11 +6,9 @@ using namespace std; int main() { - ArrayXXf a(2,2); - ArrayXXf b(2,2); - a << 1,2, - 3,4; - b << 5,6, - 7,8; + ArrayXXf a(2, 2); + ArrayXXf b(2, 2); + a << 1, 2, 3, 4; + b << 5, 6, 7, 8; cout << "a * b = " << endl << a * b << endl; } diff --git a/filmulator-gui/core/nlmeans/eigen/doc/examples/Tutorial_BlockOperations_block_assignment.cpp b/filmulator-gui/core/nlmeans/eigen/doc/examples/Tutorial_BlockOperations_block_assignment.cpp index 76f49f2f..3ef2f614 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/examples/Tutorial_BlockOperations_block_assignment.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/examples/Tutorial_BlockOperations_block_assignment.cpp @@ -7,12 +7,11 @@ using namespace Eigen; int main() { Array22f m; - m << 1,2, - 3,4; + m << 1, 2, 3, 4; Array44f a = Array44f::Constant(0.6); cout << "Here is the array a:" << endl << a << endl << endl; - a.block<2,2>(1,1) = m; + a.block<2, 2>(1, 1) = m; cout << "Here is now a with m copied into its central 2x2 block:" << endl << a << endl << endl; - a.block(0,0,2,3) = a.block(2,1,2,3); + a.block(0, 0, 2, 3) = a.block(2, 1, 2, 3); cout << "Here is now a with bottom-right 2x3 block copied into top-left 2x2 block:" << endl << a << endl << endl; } diff --git a/filmulator-gui/core/nlmeans/eigen/doc/examples/Tutorial_BlockOperations_colrow.cpp b/filmulator-gui/core/nlmeans/eigen/doc/examples/Tutorial_BlockOperations_colrow.cpp index 2e7eb009..1f0d61f1 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/examples/Tutorial_BlockOperations_colrow.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/examples/Tutorial_BlockOperations_colrow.cpp @@ -5,10 +5,8 @@ using namespace std; int main() { - Eigen::MatrixXf m(3,3); - m << 1,2,3, - 4,5,6, - 7,8,9; + Eigen::MatrixXf m(3, 3); + m << 1, 2, 3, 4, 5, 6, 7, 8, 9; cout << "Here is the matrix m:" << endl << m << endl; cout << "2nd Row: " << m.row(1) << endl; m.col(2) += 3 * m.col(0); diff --git a/filmulator-gui/core/nlmeans/eigen/doc/examples/Tutorial_BlockOperations_corner.cpp b/filmulator-gui/core/nlmeans/eigen/doc/examples/Tutorial_BlockOperations_corner.cpp index 3a31507a..a7d4da88 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/examples/Tutorial_BlockOperations_corner.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/examples/Tutorial_BlockOperations_corner.cpp @@ -6,12 +6,9 @@ using namespace std; int main() { Eigen::Matrix4f m; - m << 1, 2, 3, 4, - 5, 6, 7, 8, - 9, 10,11,12, - 13,14,15,16; + m << 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15, 16; cout << "m.leftCols(2) =" << endl << m.leftCols(2) << endl << endl; cout << "m.bottomRows<2>() =" << endl << m.bottomRows<2>() << endl << endl; - m.topLeftCorner(1,3) = m.bottomRightCorner(3,1).transpose(); + m.topLeftCorner(1, 3) = m.bottomRightCorner(3, 1).transpose(); cout << "After assignment, m = " << endl << m << endl; } diff --git a/filmulator-gui/core/nlmeans/eigen/doc/examples/Tutorial_BlockOperations_print_block.cpp b/filmulator-gui/core/nlmeans/eigen/doc/examples/Tutorial_BlockOperations_print_block.cpp index edea4aef..2d61d0fe 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/examples/Tutorial_BlockOperations_print_block.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/examples/Tutorial_BlockOperations_print_block.cpp @@ -5,16 +5,12 @@ using namespace std; int main() { - Eigen::MatrixXf m(4,4); - m << 1, 2, 3, 4, - 5, 6, 7, 8, - 9,10,11,12, - 13,14,15,16; + Eigen::MatrixXf m(4, 4); + m << 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15, 16; cout << "Block in the middle" << endl; - cout << m.block<2,2>(1,1) << endl << endl; - for (int i = 1; i <= 3; ++i) - { + cout << m.block<2, 2>(1, 1) << endl << endl; + for (int i = 1; i <= 3; ++i) { cout << "Block of size " << i << "x" << i << endl; - cout << m.block(0,0,i,i) << endl << endl; + cout << m.block(0, 0, i, i) << endl << endl; } } diff --git a/filmulator-gui/core/nlmeans/eigen/doc/examples/Tutorial_BlockOperations_vector.cpp b/filmulator-gui/core/nlmeans/eigen/doc/examples/Tutorial_BlockOperations_vector.cpp index 4a0b0234..b610455d 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/examples/Tutorial_BlockOperations_vector.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/examples/Tutorial_BlockOperations_vector.cpp @@ -9,6 +9,6 @@ int main() v << 1, 2, 3, 4, 5, 6; cout << "v.head(3) =" << endl << v.head(3) << endl << endl; cout << "v.tail<3>() = " << endl << v.tail<3>() << endl << endl; - v.segment(1,4) *= 2; + v.segment(1, 4) *= 2; cout << "after 'v.segment(1,4) *= 2', v =" << endl << v << endl; } diff --git a/filmulator-gui/core/nlmeans/eigen/doc/examples/Tutorial_PartialLU_solve.cpp b/filmulator-gui/core/nlmeans/eigen/doc/examples/Tutorial_PartialLU_solve.cpp index a5608792..5a616df3 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/examples/Tutorial_PartialLU_solve.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/examples/Tutorial_PartialLU_solve.cpp @@ -7,12 +7,12 @@ using namespace Eigen; int main() { - Matrix3f A; - Vector3f b; - A << 1,2,3, 4,5,6, 7,8,10; - b << 3, 3, 4; - cout << "Here is the matrix A:" << endl << A << endl; - cout << "Here is the vector b:" << endl << b << endl; - Vector3f x = A.lu().solve(b); - cout << "The solution is:" << endl << x << endl; + Matrix3f A; + Vector3f b; + A << 1, 2, 3, 4, 5, 6, 7, 8, 10; + b << 3, 3, 4; + cout << "Here is the matrix A:" << endl << A << endl; + cout << "Here is the vector b:" << endl << b << endl; + Vector3f x = A.lu().solve(b); + cout << "The solution is:" << endl << x << endl; } diff --git a/filmulator-gui/core/nlmeans/eigen/doc/examples/Tutorial_ReductionsVisitorsBroadcasting_broadcast_1nn.cpp b/filmulator-gui/core/nlmeans/eigen/doc/examples/Tutorial_ReductionsVisitorsBroadcasting_broadcast_1nn.cpp index 334b4d85..89d9b1f5 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/examples/Tutorial_ReductionsVisitorsBroadcasting_broadcast_1nn.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/examples/Tutorial_ReductionsVisitorsBroadcasting_broadcast_1nn.cpp @@ -1,19 +1,17 @@ -#include #include +#include using namespace std; using namespace Eigen; int main() { - Eigen::MatrixXf m(2,4); + Eigen::MatrixXf m(2, 4); Eigen::VectorXf v(2); - - m << 1, 23, 6, 9, - 3, 11, 7, 2; - - v << 2, - 3; + + m << 1, 23, 6, 9, 3, 11, 7, 2; + + v << 2, 3; MatrixXf::Index index; // find nearest neighbour diff --git a/filmulator-gui/core/nlmeans/eigen/doc/examples/Tutorial_ReductionsVisitorsBroadcasting_broadcast_simple.cpp b/filmulator-gui/core/nlmeans/eigen/doc/examples/Tutorial_ReductionsVisitorsBroadcasting_broadcast_simple.cpp index e6c87c6a..6e5c2850 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/examples/Tutorial_ReductionsVisitorsBroadcasting_broadcast_simple.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/examples/Tutorial_ReductionsVisitorsBroadcasting_broadcast_simple.cpp @@ -1,21 +1,19 @@ -#include #include +#include using namespace std; int main() { - Eigen::MatrixXf mat(2,4); + Eigen::MatrixXf mat(2, 4); Eigen::VectorXf v(2); - - mat << 1, 2, 6, 9, - 3, 1, 7, 2; - - v << 0, - 1; - - //add v to each column of m + + mat << 1, 2, 6, 9, 3, 1, 7, 2; + + v << 0, 1; + + // add v to each column of m mat.colwise() += v; - + std::cout << "Broadcasting result: " << std::endl; std::cout << mat << std::endl; } diff --git a/filmulator-gui/core/nlmeans/eigen/doc/examples/Tutorial_ReductionsVisitorsBroadcasting_broadcast_simple_rowwise.cpp b/filmulator-gui/core/nlmeans/eigen/doc/examples/Tutorial_ReductionsVisitorsBroadcasting_broadcast_simple_rowwise.cpp index d87c96ab..123e3b58 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/examples/Tutorial_ReductionsVisitorsBroadcasting_broadcast_simple_rowwise.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/examples/Tutorial_ReductionsVisitorsBroadcasting_broadcast_simple_rowwise.cpp @@ -1,20 +1,19 @@ -#include #include +#include using namespace std; int main() { - Eigen::MatrixXf mat(2,4); + Eigen::MatrixXf mat(2, 4); Eigen::VectorXf v(4); - - mat << 1, 2, 6, 9, - 3, 1, 7, 2; - - v << 0,1,2,3; - - //add v to each row of m + + mat << 1, 2, 6, 9, 3, 1, 7, 2; + + v << 0, 1, 2, 3; + + // add v to each row of m mat.rowwise() += v.transpose(); - + std::cout << "Broadcasting result: " << std::endl; std::cout << mat << std::endl; } diff --git a/filmulator-gui/core/nlmeans/eigen/doc/examples/Tutorial_ReductionsVisitorsBroadcasting_colwise.cpp b/filmulator-gui/core/nlmeans/eigen/doc/examples/Tutorial_ReductionsVisitorsBroadcasting_colwise.cpp index df682566..d27285e6 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/examples/Tutorial_ReductionsVisitorsBroadcasting_colwise.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/examples/Tutorial_ReductionsVisitorsBroadcasting_colwise.cpp @@ -1,13 +1,11 @@ -#include #include +#include using namespace std; int main() { - Eigen::MatrixXf mat(2,4); - mat << 1, 2, 6, 9, - 3, 1, 7, 2; - - std::cout << "Column's maximum: " << std::endl - << mat.colwise().maxCoeff() << std::endl; + Eigen::MatrixXf mat(2, 4); + mat << 1, 2, 6, 9, 3, 1, 7, 2; + + std::cout << "Column's maximum: " << std::endl << mat.colwise().maxCoeff() << std::endl; } diff --git a/filmulator-gui/core/nlmeans/eigen/doc/examples/Tutorial_ReductionsVisitorsBroadcasting_maxnorm.cpp b/filmulator-gui/core/nlmeans/eigen/doc/examples/Tutorial_ReductionsVisitorsBroadcasting_maxnorm.cpp index 049c747b..f0918782 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/examples/Tutorial_ReductionsVisitorsBroadcasting_maxnorm.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/examples/Tutorial_ReductionsVisitorsBroadcasting_maxnorm.cpp @@ -1,20 +1,19 @@ -#include #include +#include using namespace std; using namespace Eigen; int main() { - MatrixXf mat(2,4); - mat << 1, 2, 6, 9, - 3, 1, 7, 2; - - MatrixXf::Index maxIndex; + MatrixXf mat(2, 4); + mat << 1, 2, 6, 9, 3, 1, 7, 2; + + MatrixXf::Index maxIndex; float maxNorm = mat.colwise().sum().maxCoeff(&maxIndex); - + std::cout << "Maximum sum at position " << maxIndex << std::endl; std::cout << "The corresponding vector is: " << std::endl; - std::cout << mat.col( maxIndex ) << std::endl; + std::cout << mat.col(maxIndex) << std::endl; std::cout << "And its sum is is: " << maxNorm << std::endl; } diff --git a/filmulator-gui/core/nlmeans/eigen/doc/examples/Tutorial_ReductionsVisitorsBroadcasting_reductions_bool.cpp b/filmulator-gui/core/nlmeans/eigen/doc/examples/Tutorial_ReductionsVisitorsBroadcasting_reductions_bool.cpp index 0cca37f3..17b470c8 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/examples/Tutorial_ReductionsVisitorsBroadcasting_reductions_bool.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/examples/Tutorial_ReductionsVisitorsBroadcasting_reductions_bool.cpp @@ -6,10 +6,9 @@ using namespace Eigen; int main() { - ArrayXXf a(2,2); - - a << 1,2, - 3,4; + ArrayXXf a(2, 2); + + a << 1, 2, 3, 4; cout << "(a > 0).all() = " << (a > 0).all() << endl; cout << "(a > 0).any() = " << (a > 0).any() << endl; diff --git a/filmulator-gui/core/nlmeans/eigen/doc/examples/Tutorial_ReductionsVisitorsBroadcasting_reductions_norm.cpp b/filmulator-gui/core/nlmeans/eigen/doc/examples/Tutorial_ReductionsVisitorsBroadcasting_reductions_norm.cpp index 740439fb..71f16240 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/examples/Tutorial_ReductionsVisitorsBroadcasting_reductions_norm.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/examples/Tutorial_ReductionsVisitorsBroadcasting_reductions_norm.cpp @@ -7,13 +7,11 @@ using namespace Eigen; int main() { VectorXf v(2); - MatrixXf m(2,2), n(2,2); - - v << -1, - 2; - - m << 1,-2, - -3,4; + MatrixXf m(2, 2), n(2, 2); + + v << -1, 2; + + m << 1, -2, -3, 4; cout << "v.squaredNorm() = " << v.squaredNorm() << endl; cout << "v.norm() = " << v.norm() << endl; diff --git a/filmulator-gui/core/nlmeans/eigen/doc/examples/Tutorial_ReductionsVisitorsBroadcasting_reductions_operatornorm.cpp b/filmulator-gui/core/nlmeans/eigen/doc/examples/Tutorial_ReductionsVisitorsBroadcasting_reductions_operatornorm.cpp index 62e28fc3..e973a3f1 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/examples/Tutorial_ReductionsVisitorsBroadcasting_reductions_operatornorm.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/examples/Tutorial_ReductionsVisitorsBroadcasting_reductions_operatornorm.cpp @@ -6,13 +6,12 @@ using namespace std; int main() { - MatrixXf m(2,2); - m << 1,-2, - -3,4; + MatrixXf m(2, 2); + m << 1, -2, -3, 4; - cout << "1-norm(m) = " << m.cwiseAbs().colwise().sum().maxCoeff() - << " == " << m.colwise().lpNorm<1>().maxCoeff() << endl; + cout << "1-norm(m) = " << m.cwiseAbs().colwise().sum().maxCoeff() << " == " << m.colwise().lpNorm<1>().maxCoeff() + << endl; - cout << "infty-norm(m) = " << m.cwiseAbs().rowwise().sum().maxCoeff() - << " == " << m.rowwise().lpNorm<1>().maxCoeff() << endl; + cout << "infty-norm(m) = " << m.cwiseAbs().rowwise().sum().maxCoeff() << " == " << m.rowwise().lpNorm<1>().maxCoeff() + << endl; } diff --git a/filmulator-gui/core/nlmeans/eigen/doc/examples/Tutorial_ReductionsVisitorsBroadcasting_rowwise.cpp b/filmulator-gui/core/nlmeans/eigen/doc/examples/Tutorial_ReductionsVisitorsBroadcasting_rowwise.cpp index 80427c9f..3b614354 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/examples/Tutorial_ReductionsVisitorsBroadcasting_rowwise.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/examples/Tutorial_ReductionsVisitorsBroadcasting_rowwise.cpp @@ -1,13 +1,11 @@ -#include #include +#include using namespace std; int main() { - Eigen::MatrixXf mat(2,4); - mat << 1, 2, 6, 9, - 3, 1, 7, 2; - - std::cout << "Row's maximum: " << std::endl - << mat.rowwise().maxCoeff() << std::endl; + Eigen::MatrixXf mat(2, 4); + mat << 1, 2, 6, 9, 3, 1, 7, 2; + + std::cout << "Row's maximum: " << std::endl << mat.rowwise().maxCoeff() << std::endl; } diff --git a/filmulator-gui/core/nlmeans/eigen/doc/examples/Tutorial_ReductionsVisitorsBroadcasting_visitors.cpp b/filmulator-gui/core/nlmeans/eigen/doc/examples/Tutorial_ReductionsVisitorsBroadcasting_visitors.cpp index b54e9aa3..a94544ce 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/examples/Tutorial_ReductionsVisitorsBroadcasting_visitors.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/examples/Tutorial_ReductionsVisitorsBroadcasting_visitors.cpp @@ -1,26 +1,23 @@ -#include #include +#include using namespace std; using namespace Eigen; int main() { - Eigen::MatrixXf m(2,2); - - m << 1, 2, - 3, 4; + Eigen::MatrixXf m(2, 2); + + m << 1, 2, 3, 4; - //get location of maximum + // get location of maximum MatrixXf::Index maxRow, maxCol; float max = m.maxCoeff(&maxRow, &maxCol); - //get location of minimum + // get location of minimum MatrixXf::Index minRow, minCol; float min = m.minCoeff(&minRow, &minCol); - cout << "Max: " << max << ", at: " << - maxRow << "," << maxCol << endl; - cout << "Min: " << min << ", at: " << - minRow << "," << minCol << endl; + cout << "Max: " << max << ", at: " << maxRow << "," << maxCol << endl; + cout << "Min: " << min << ", at: " << minRow << "," << minCol << endl; } diff --git a/filmulator-gui/core/nlmeans/eigen/doc/examples/Tutorial_simple_example_dynamic_size.cpp b/filmulator-gui/core/nlmeans/eigen/doc/examples/Tutorial_simple_example_dynamic_size.cpp index defcb1ee..911d4f57 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/examples/Tutorial_simple_example_dynamic_size.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/examples/Tutorial_simple_example_dynamic_size.cpp @@ -5,18 +5,20 @@ using namespace Eigen; int main() { - for (int size=1; size<=4; ++size) - { - MatrixXi m(size,size+1); // a (size)x(size+1)-matrix of int's - for (int j=0; j -Eigen::Block -topLeftCorner(MatrixBase& m, int rows, int cols) +template Eigen::Block topLeftCorner(MatrixBase &m, int rows, int cols) { return Eigen::Block(m.derived(), 0, 0, rows, cols); } template -const Eigen::Block -topLeftCorner(const MatrixBase& m, int rows, int cols) +const Eigen::Block topLeftCorner(const MatrixBase &m, int rows, int cols) { return Eigen::Block(m.derived(), 0, 0, rows, cols); } -int main(int, char**) +int main(int, char **) { Matrix4d m = Matrix4d::Identity(); - cout << topLeftCorner(4*m, 2, 3) << endl; // calls the const version - topLeftCorner(m, 2, 3) *= 5; // calls the non-const version + cout << topLeftCorner(4 * m, 2, 3) << endl;// calls the const version + topLeftCorner(m, 2, 3) *= 5;// calls the non-const version cout << "Now the matrix m is:" << endl << m << endl; return 0; } diff --git a/filmulator-gui/core/nlmeans/eigen/doc/examples/class_CwiseBinaryOp.cpp b/filmulator-gui/core/nlmeans/eigen/doc/examples/class_CwiseBinaryOp.cpp index 682af46d..0bb0d684 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/examples/class_CwiseBinaryOp.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/examples/class_CwiseBinaryOp.cpp @@ -4,13 +4,14 @@ using namespace Eigen; using namespace std; // define a custom template binary functor -template struct MakeComplexOp { +template struct MakeComplexOp +{ EIGEN_EMPTY_STRUCT_CTOR(MakeComplexOp) typedef complex result_type; - complex operator()(const Scalar& a, const Scalar& b) const { return complex(a,b); } + complex operator()(const Scalar &a, const Scalar &b) const { return complex(a, b); } }; -int main(int, char**) +int main(int, char **) { Matrix4d m1 = Matrix4d::Random(), m2 = Matrix4d::Random(); cout << m1.binaryExpr(m2, MakeComplexOp()) << endl; diff --git a/filmulator-gui/core/nlmeans/eigen/doc/examples/class_CwiseUnaryOp.cpp b/filmulator-gui/core/nlmeans/eigen/doc/examples/class_CwiseUnaryOp.cpp index a5fcc153..d73725a4 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/examples/class_CwiseUnaryOp.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/examples/class_CwiseUnaryOp.cpp @@ -4,16 +4,16 @@ using namespace Eigen; using namespace std; // define a custom template unary functor -template -struct CwiseClampOp { - CwiseClampOp(const Scalar& inf, const Scalar& sup) : m_inf(inf), m_sup(sup) {} - const Scalar operator()(const Scalar& x) const { return xm_sup ? m_sup : x); } +template struct CwiseClampOp +{ + CwiseClampOp(const Scalar &inf, const Scalar &sup) : m_inf(inf), m_sup(sup) {} + const Scalar operator()(const Scalar &x) const { return x < m_inf ? m_inf : (x > m_sup ? m_sup : x); } Scalar m_inf, m_sup; }; -int main(int, char**) +int main(int, char **) { Matrix4d m1 = Matrix4d::Random(); - cout << m1 << endl << "becomes: " << endl << m1.unaryExpr(CwiseClampOp(-0.5,0.5)) << endl; + cout << m1 << endl << "becomes: " << endl << m1.unaryExpr(CwiseClampOp(-0.5, 0.5)) << endl; return 0; } diff --git a/filmulator-gui/core/nlmeans/eigen/doc/examples/class_CwiseUnaryOp_ptrfun.cpp b/filmulator-gui/core/nlmeans/eigen/doc/examples/class_CwiseUnaryOp_ptrfun.cpp index 36706d8e..38f2744c 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/examples/class_CwiseUnaryOp_ptrfun.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/examples/class_CwiseUnaryOp_ptrfun.cpp @@ -8,11 +8,11 @@ double ramp(double x) { if (x > 0) return x; - else + else return 0; } -int main(int, char**) +int main(int, char **) { Matrix4d m1 = Matrix4d::Random(); cout << m1 << endl << "becomes: " << endl << m1.unaryExpr(ptr_fun(ramp)) << endl; diff --git a/filmulator-gui/core/nlmeans/eigen/doc/examples/class_FixedBlock.cpp b/filmulator-gui/core/nlmeans/eigen/doc/examples/class_FixedBlock.cpp index 9978b32e..c82b8576 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/examples/class_FixedBlock.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/examples/class_FixedBlock.cpp @@ -3,25 +3,21 @@ using namespace Eigen; using namespace std; -template -Eigen::Block -topLeft2x2Corner(MatrixBase& m) +template Eigen::Block topLeft2x2Corner(MatrixBase &m) { return Eigen::Block(m.derived(), 0, 0); } -template -const Eigen::Block -topLeft2x2Corner(const MatrixBase& m) +template const Eigen::Block topLeft2x2Corner(const MatrixBase &m) { return Eigen::Block(m.derived(), 0, 0); } -int main(int, char**) +int main(int, char **) { Matrix3d m = Matrix3d::Identity(); - cout << topLeft2x2Corner(4*m) << endl; // calls the const version - topLeft2x2Corner(m) *= 2; // calls the non-const version + cout << topLeft2x2Corner(4 * m) << endl;// calls the const version + topLeft2x2Corner(m) *= 2;// calls the non-const version cout << "Now the matrix m is:" << endl << m << endl; return 0; } diff --git a/filmulator-gui/core/nlmeans/eigen/doc/examples/class_FixedVectorBlock.cpp b/filmulator-gui/core/nlmeans/eigen/doc/examples/class_FixedVectorBlock.cpp index c88c9fbf..36f988af 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/examples/class_FixedVectorBlock.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/examples/class_FixedVectorBlock.cpp @@ -3,25 +3,22 @@ using namespace Eigen; using namespace std; -template -Eigen::VectorBlock -firstTwo(MatrixBase& v) +template Eigen::VectorBlock firstTwo(MatrixBase &v) { return Eigen::VectorBlock(v.derived(), 0); } -template -const Eigen::VectorBlock -firstTwo(const MatrixBase& v) +template const Eigen::VectorBlock firstTwo(const MatrixBase &v) { return Eigen::VectorBlock(v.derived(), 0); } -int main(int, char**) +int main(int, char **) { - Matrix v; v << 1,2,3,4,5,6; - cout << firstTwo(4*v) << endl; // calls the const version - firstTwo(v) *= 2; // calls the non-const version + Matrix v; + v << 1, 2, 3, 4, 5, 6; + cout << firstTwo(4 * v) << endl;// calls the const version + firstTwo(v) *= 2;// calls the non-const version cout << "Now the vector v is:" << endl << v << endl; return 0; } diff --git a/filmulator-gui/core/nlmeans/eigen/doc/examples/class_VectorBlock.cpp b/filmulator-gui/core/nlmeans/eigen/doc/examples/class_VectorBlock.cpp index dc213df2..35600597 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/examples/class_VectorBlock.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/examples/class_VectorBlock.cpp @@ -3,25 +3,23 @@ using namespace Eigen; using namespace std; -template -Eigen::VectorBlock -segmentFromRange(MatrixBase& v, int start, int end) +template Eigen::VectorBlock segmentFromRange(MatrixBase &v, int start, int end) { - return Eigen::VectorBlock(v.derived(), start, end-start); + return Eigen::VectorBlock(v.derived(), start, end - start); } template -const Eigen::VectorBlock -segmentFromRange(const MatrixBase& v, int start, int end) +const Eigen::VectorBlock segmentFromRange(const MatrixBase &v, int start, int end) { - return Eigen::VectorBlock(v.derived(), start, end-start); + return Eigen::VectorBlock(v.derived(), start, end - start); } -int main(int, char**) +int main(int, char **) { - Matrix v; v << 1,2,3,4,5,6; - cout << segmentFromRange(2*v, 2, 4) << endl; // calls the const version - segmentFromRange(v, 1, 3) *= 5; // calls the non-const version + Matrix v; + v << 1, 2, 3, 4, 5, 6; + cout << segmentFromRange(2 * v, 2, 4) << endl;// calls the const version + segmentFromRange(v, 1, 3) *= 5;// calls the non-const version cout << "Now the vector v is:" << endl << v << endl; return 0; } diff --git a/filmulator-gui/core/nlmeans/eigen/doc/examples/function_taking_eigenbase.cpp b/filmulator-gui/core/nlmeans/eigen/doc/examples/function_taking_eigenbase.cpp index 49d94b3d..000ec3aa 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/examples/function_taking_eigenbase.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/examples/function_taking_eigenbase.cpp @@ -1,18 +1,16 @@ -#include #include +#include using namespace Eigen; -template -void print_size(const EigenBase& b) +template void print_size(const EigenBase &b) { - std::cout << "size (rows, cols): " << b.size() << " (" << b.rows() - << ", " << b.cols() << ")" << std::endl; + std::cout << "size (rows, cols): " << b.size() << " (" << b.rows() << ", " << b.cols() << ")" << std::endl; } int main() { - Vector3f v; - print_size(v); - // v.asDiagonal() returns a 3x3 diagonal matrix pseudo-expression - print_size(v.asDiagonal()); + Vector3f v; + print_size(v); + // v.asDiagonal() returns a 3x3 diagonal matrix pseudo-expression + print_size(v.asDiagonal()); } diff --git a/filmulator-gui/core/nlmeans/eigen/doc/examples/function_taking_ref.cpp b/filmulator-gui/core/nlmeans/eigen/doc/examples/function_taking_ref.cpp index 162a202e..078b0360 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/examples/function_taking_ref.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/examples/function_taking_ref.cpp @@ -1,19 +1,19 @@ -#include #include +#include using namespace Eigen; using namespace std; -float inv_cond(const Ref& a) +float inv_cond(const Ref &a) { const VectorXf sing_vals = a.jacobiSvd().singularValues(); - return sing_vals(sing_vals.size()-1) / sing_vals(0); + return sing_vals(sing_vals.size() - 1) / sing_vals(0); } int main() { Matrix4f m = Matrix4f::Random(); cout << "matrix m:" << endl << m << endl << endl; - cout << "inv_cond(m): " << inv_cond(m) << endl; - cout << "inv_cond(m(1:3,1:3)): " << inv_cond(m.topLeftCorner(3,3)) << endl; - cout << "inv_cond(m+I): " << inv_cond(m+Matrix4f::Identity()) << endl; + cout << "inv_cond(m): " << inv_cond(m) << endl; + cout << "inv_cond(m(1:3,1:3)): " << inv_cond(m.topLeftCorner(3, 3)) << endl; + cout << "inv_cond(m+I): " << inv_cond(m + Matrix4f::Identity()) << endl; } diff --git a/filmulator-gui/core/nlmeans/eigen/doc/examples/make_circulant.cpp b/filmulator-gui/core/nlmeans/eigen/doc/examples/make_circulant.cpp index 92e6aaa2..7900ec34 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/examples/make_circulant.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/examples/make_circulant.cpp @@ -3,9 +3,9 @@ This program is presented in several fragments in the doc page. Every fragment is in its own file; this file simply combines them. */ -#include "make_circulant.cpp.preamble" -#include "make_circulant.cpp.traits" -#include "make_circulant.cpp.expression" -#include "make_circulant.cpp.evaluator" #include "make_circulant.cpp.entry" +#include "make_circulant.cpp.evaluator" +#include "make_circulant.cpp.expression" #include "make_circulant.cpp.main" +#include "make_circulant.cpp.preamble" +#include "make_circulant.cpp.traits" diff --git a/filmulator-gui/core/nlmeans/eigen/doc/examples/make_circulant2.cpp b/filmulator-gui/core/nlmeans/eigen/doc/examples/make_circulant2.cpp index 95d3dd31..da343a66 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/examples/make_circulant2.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/examples/make_circulant2.cpp @@ -4,13 +4,15 @@ using namespace Eigen; // [circulant_func] -template -class circulant_functor { +template class circulant_functor +{ const ArgType &m_vec; + public: - circulant_functor(const ArgType& arg) : m_vec(arg) {} + circulant_functor(const ArgType &arg) : m_vec(arg) {} - const typename ArgType::Scalar& operator() (Index row, Index col) const { + const typename ArgType::Scalar &operator()(Index row, Index col) const + { Index index = row - col; if (index < 0) index += m_vec.size(); return m_vec(index); @@ -19,21 +21,22 @@ class circulant_functor { // [circulant_func] // [square] -template -struct circulant_helper { +template struct circulant_helper +{ typedef Matrix MatrixType; + ArgType::SizeAtCompileTime, + ArgType::SizeAtCompileTime, + ColMajor, + ArgType::MaxSizeAtCompileTime, + ArgType::MaxSizeAtCompileTime> + MatrixType; }; // [square] // [makeCirculant] -template -CwiseNullaryOp, typename circulant_helper::MatrixType> -makeCirculant(const Eigen::MatrixBase& arg) +template +CwiseNullaryOp, typename circulant_helper::MatrixType> makeCirculant( + const Eigen::MatrixBase &arg) { typedef typename circulant_helper::MatrixType MatrixType; return MatrixType::NullaryExpr(arg.size(), arg.size(), circulant_functor(arg.derived())); diff --git a/filmulator-gui/core/nlmeans/eigen/doc/examples/matrixfree_cg.cpp b/filmulator-gui/core/nlmeans/eigen/doc/examples/matrixfree_cg.cpp index 74699381..9b94d617 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/examples/matrixfree_cg.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/examples/matrixfree_cg.cpp @@ -1,7 +1,7 @@ -#include #include #include #include +#include #include class MatrixReplacement; @@ -10,40 +10,36 @@ using Eigen::SparseMatrix; namespace Eigen { namespace internal { // MatrixReplacement looks-like a SparseMatrix, so let's inherits its traits: - template<> - struct traits : public Eigen::internal::traits > - {}; -} -} + template<> struct traits : public Eigen::internal::traits> + { + }; +}// namespace internal +}// namespace Eigen // Example of a matrix-free wrapper from a user type to Eigen's compatible type // For the sake of simplicity, this example simply wrap a Eigen::SparseMatrix. -class MatrixReplacement : public Eigen::EigenBase { +class MatrixReplacement : public Eigen::EigenBase +{ public: // Required typedefs, constants, and method: typedef double Scalar; typedef double RealScalar; typedef int StorageIndex; - enum { - ColsAtCompileTime = Eigen::Dynamic, - MaxColsAtCompileTime = Eigen::Dynamic, - IsRowMajor = false - }; + enum { ColsAtCompileTime = Eigen::Dynamic, MaxColsAtCompileTime = Eigen::Dynamic, IsRowMajor = false }; Index rows() const { return mp_mat->rows(); } Index cols() const { return mp_mat->cols(); } template - Eigen::Product operator*(const Eigen::MatrixBase& x) const { - return Eigen::Product(*this, x.derived()); + Eigen::Product operator*(const Eigen::MatrixBase &x) const + { + return Eigen::Product(*this, x.derived()); } // Custom API: MatrixReplacement() : mp_mat(0) {} - void attachMyMatrix(const SparseMatrix &mat) { - mp_mat = &mat; - } + void attachMyMatrix(const SparseMatrix &mat) { mp_mat = &mat; } const SparseMatrix my_matrix() const { return *mp_mat; } private: @@ -56,34 +52,34 @@ namespace Eigen { namespace internal { template - struct generic_product_impl // GEMV stands for matrix-vector - : generic_product_impl_base > + struct generic_product_impl// GEMV stands for + // matrix-vector + : generic_product_impl_base> { - typedef typename Product::Scalar Scalar; + typedef typename Product::Scalar Scalar; template - static void scaleAndAddTo(Dest& dst, const MatrixReplacement& lhs, const Rhs& rhs, const Scalar& alpha) + static void scaleAndAddTo(Dest &dst, const MatrixReplacement &lhs, const Rhs &rhs, const Scalar &alpha) { // This method should implement "dst += alpha * lhs * rhs" inplace, // however, for iterative solvers, alpha is always equal to 1, so let's not bother about it. - assert(alpha==Scalar(1) && "scaling is not implemented"); + assert(alpha == Scalar(1) && "scaling is not implemented"); EIGEN_ONLY_USED_FOR_DEBUG(alpha); // Here we could simply call dst.noalias() += lhs.my_matrix() * rhs, // but let's do something fancier (and less efficient): - for(Index i=0; i S = Eigen::MatrixXd::Random(n,n).sparseView(0.5,1); - S = S.transpose()*S; + Eigen::SparseMatrix S = Eigen::MatrixXd::Random(n, n).sparseView(0.5, 1); + S = S.transpose() * S; MatrixReplacement A; A.attachMyMatrix(S); @@ -93,7 +89,7 @@ int main() // Solve Ax = b using various iterative solver with matrix-free version: { - Eigen::ConjugateGradient cg; + Eigen::ConjugateGradient cg; cg.compute(A); x = cg.solve(b); std::cout << "CG: #iterations: " << cg.iterations() << ", estimated error: " << cg.error() << std::endl; @@ -121,9 +117,10 @@ int main() } { - Eigen::MINRES minres; + Eigen::MINRES minres; minres.compute(A); x = minres.solve(b); - std::cout << "MINRES: #iterations: " << minres.iterations() << ", estimated error: " << minres.error() << std::endl; + std::cout << "MINRES: #iterations: " << minres.iterations() << ", estimated error: " << minres.error() + << std::endl; } } diff --git a/filmulator-gui/core/nlmeans/eigen/doc/examples/nullary_indexing.cpp b/filmulator-gui/core/nlmeans/eigen/doc/examples/nullary_indexing.cpp index e27c3585..e627fc91 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/examples/nullary_indexing.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/examples/nullary_indexing.cpp @@ -4,35 +4,39 @@ using namespace Eigen; // [functor] -template -class indexing_functor { +template class indexing_functor +{ const ArgType &m_arg; const RowIndexType &m_rowIndices; const ColIndexType &m_colIndices; + public: typedef Matrix MatrixType; + RowIndexType::SizeAtCompileTime, + ColIndexType::SizeAtCompileTime, + ArgType::Flags & RowMajorBit ? RowMajor : ColMajor, + RowIndexType::MaxSizeAtCompileTime, + ColIndexType::MaxSizeAtCompileTime> + MatrixType; - indexing_functor(const ArgType& arg, const RowIndexType& row_indices, const ColIndexType& col_indices) + indexing_functor(const ArgType &arg, const RowIndexType &row_indices, const ColIndexType &col_indices) : m_arg(arg), m_rowIndices(row_indices), m_colIndices(col_indices) {} - const typename ArgType::Scalar& operator() (Index row, Index col) const { + const typename ArgType::Scalar &operator()(Index row, Index col) const + { return m_arg(m_rowIndices[row], m_colIndices[col]); } }; // [functor] // [function] -template -CwiseNullaryOp, typename indexing_functor::MatrixType> -indexing(const Eigen::MatrixBase& arg, const RowIndexType& row_indices, const ColIndexType& col_indices) +template +CwiseNullaryOp, + typename indexing_functor::MatrixType> + indexing(const Eigen::MatrixBase &arg, const RowIndexType &row_indices, const ColIndexType &col_indices) { - typedef indexing_functor Func; + typedef indexing_functor Func; typedef typename Func::MatrixType MatrixType; return MatrixType::NullaryExpr(row_indices.size(), col_indices.size(), Func(arg.derived(), row_indices, col_indices)); } @@ -42,9 +46,10 @@ indexing(const Eigen::MatrixBase& arg, const RowIndexType& row_indices, int main() { std::cout << "[main1]\n"; - Eigen::MatrixXi A = Eigen::MatrixXi::Random(4,4); - Array3i ri(1,2,1); - ArrayXi ci(6); ci << 3,2,1,0,0,2; + Eigen::MatrixXi A = Eigen::MatrixXi::Random(4, 4); + Array3i ri(1, 2, 1); + ArrayXi ci(6); + ci << 3, 2, 1, 0, 0, 2; Eigen::MatrixXi B = indexing(A, ri, ci); std::cout << "A =" << std::endl; std::cout << A << std::endl << std::endl; @@ -53,14 +58,14 @@ int main() std::cout << "[main1]\n"; std::cout << "[main2]\n"; - B = indexing(A, ri+1, ci); + B = indexing(A, ri + 1, ci); std::cout << "A(ri+1,ci) =" << std::endl; std::cout << B << std::endl << std::endl; #if __cplusplus >= 201103L - B = indexing(A, ArrayXi::LinSpaced(13,0,12).unaryExpr([](int x){return x%4;}), ArrayXi::LinSpaced(4,0,3)); - std::cout << "A(ArrayXi::LinSpaced(13,0,12).unaryExpr([](int x){return x%4;}), ArrayXi::LinSpaced(4,0,3)) =" << std::endl; + B = indexing(A, ArrayXi::LinSpaced(13, 0, 12).unaryExpr([](int x) { return x % 4; }), ArrayXi::LinSpaced(4, 0, 3)); + std::cout << "A(ArrayXi::LinSpaced(13,0,12).unaryExpr([](int x){return x%4;}), ArrayXi::LinSpaced(4,0,3)) =" + << std::endl; std::cout << B << std::endl << std::endl; #endif std::cout << "[main2]\n"; } - diff --git a/filmulator-gui/core/nlmeans/eigen/doc/examples/tut_arithmetic_add_sub.cpp b/filmulator-gui/core/nlmeans/eigen/doc/examples/tut_arithmetic_add_sub.cpp index e97477b6..4411f68a 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/examples/tut_arithmetic_add_sub.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/examples/tut_arithmetic_add_sub.cpp @@ -1,22 +1,20 @@ -#include #include +#include using namespace Eigen; int main() { Matrix2d a; - a << 1, 2, - 3, 4; - MatrixXd b(2,2); - b << 2, 3, - 1, 4; + a << 1, 2, 3, 4; + MatrixXd b(2, 2); + b << 2, 3, 1, 4; std::cout << "a + b =\n" << a + b << std::endl; std::cout << "a - b =\n" << a - b << std::endl; std::cout << "Doing a += b;" << std::endl; a += b; std::cout << "Now a =\n" << a << std::endl; - Vector3d v(1,2,3); - Vector3d w(1,0,0); + Vector3d v(1, 2, 3); + Vector3d w(1, 0, 0); std::cout << "-v + w - v =\n" << -v + w - v << std::endl; } diff --git a/filmulator-gui/core/nlmeans/eigen/doc/examples/tut_arithmetic_dot_cross.cpp b/filmulator-gui/core/nlmeans/eigen/doc/examples/tut_arithmetic_dot_cross.cpp index 631c9a5e..216ba0d6 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/examples/tut_arithmetic_dot_cross.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/examples/tut_arithmetic_dot_cross.cpp @@ -1,15 +1,15 @@ -#include #include +#include using namespace Eigen; using namespace std; int main() { - Vector3d v(1,2,3); - Vector3d w(0,1,2); + Vector3d v(1, 2, 3); + Vector3d w(0, 1, 2); cout << "Dot product: " << v.dot(w) << endl; - double dp = v.adjoint()*w; // automatic conversion of the inner product to a scalar + double dp = v.adjoint() * w;// automatic conversion of the inner product to a scalar cout << "Dot product via a matrix product: " << dp << endl; cout << "Cross product:\n" << v.cross(w) << endl; } diff --git a/filmulator-gui/core/nlmeans/eigen/doc/examples/tut_arithmetic_matrix_mul.cpp b/filmulator-gui/core/nlmeans/eigen/doc/examples/tut_arithmetic_matrix_mul.cpp index f2139024..93bea26d 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/examples/tut_arithmetic_matrix_mul.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/examples/tut_arithmetic_matrix_mul.cpp @@ -1,19 +1,18 @@ -#include #include +#include using namespace Eigen; int main() { Matrix2d mat; - mat << 1, 2, - 3, 4; - Vector2d u(-1,1), v(2,0); - std::cout << "Here is mat*mat:\n" << mat*mat << std::endl; - std::cout << "Here is mat*u:\n" << mat*u << std::endl; - std::cout << "Here is u^T*mat:\n" << u.transpose()*mat << std::endl; - std::cout << "Here is u^T*v:\n" << u.transpose()*v << std::endl; - std::cout << "Here is u*v^T:\n" << u*v.transpose() << std::endl; + mat << 1, 2, 3, 4; + Vector2d u(-1, 1), v(2, 0); + std::cout << "Here is mat*mat:\n" << mat * mat << std::endl; + std::cout << "Here is mat*u:\n" << mat * u << std::endl; + std::cout << "Here is u^T*mat:\n" << u.transpose() * mat << std::endl; + std::cout << "Here is u^T*v:\n" << u.transpose() * v << std::endl; + std::cout << "Here is u*v^T:\n" << u * v.transpose() << std::endl; std::cout << "Let's multiply mat by itself" << std::endl; - mat = mat*mat; + mat = mat * mat; std::cout << "Now mat is mat:\n" << mat << std::endl; } diff --git a/filmulator-gui/core/nlmeans/eigen/doc/examples/tut_arithmetic_redux_basic.cpp b/filmulator-gui/core/nlmeans/eigen/doc/examples/tut_arithmetic_redux_basic.cpp index 5632fb52..ab40200f 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/examples/tut_arithmetic_redux_basic.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/examples/tut_arithmetic_redux_basic.cpp @@ -1,16 +1,15 @@ -#include #include +#include using namespace std; int main() { Eigen::Matrix2d mat; - mat << 1, 2, - 3, 4; - cout << "Here is mat.sum(): " << mat.sum() << endl; - cout << "Here is mat.prod(): " << mat.prod() << endl; - cout << "Here is mat.mean(): " << mat.mean() << endl; - cout << "Here is mat.minCoeff(): " << mat.minCoeff() << endl; - cout << "Here is mat.maxCoeff(): " << mat.maxCoeff() << endl; - cout << "Here is mat.trace(): " << mat.trace() << endl; + mat << 1, 2, 3, 4; + cout << "Here is mat.sum(): " << mat.sum() << endl; + cout << "Here is mat.prod(): " << mat.prod() << endl; + cout << "Here is mat.mean(): " << mat.mean() << endl; + cout << "Here is mat.minCoeff(): " << mat.minCoeff() << endl; + cout << "Here is mat.maxCoeff(): " << mat.maxCoeff() << endl; + cout << "Here is mat.trace(): " << mat.trace() << endl; } diff --git a/filmulator-gui/core/nlmeans/eigen/doc/examples/tut_arithmetic_scalar_mul_div.cpp b/filmulator-gui/core/nlmeans/eigen/doc/examples/tut_arithmetic_scalar_mul_div.cpp index d5f65b53..86f5b89f 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/examples/tut_arithmetic_scalar_mul_div.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/examples/tut_arithmetic_scalar_mul_div.cpp @@ -1,14 +1,13 @@ -#include #include +#include using namespace Eigen; int main() { Matrix2d a; - a << 1, 2, - 3, 4; - Vector3d v(1,2,3); + a << 1, 2, 3, 4; + Vector3d v(1, 2, 3); std::cout << "a * 2.5 =\n" << a * 2.5 << std::endl; std::cout << "0.1 * v =\n" << 0.1 * v << std::endl; std::cout << "Doing v *= 2;" << std::endl; diff --git a/filmulator-gui/core/nlmeans/eigen/doc/examples/tut_matrix_coefficient_accessors.cpp b/filmulator-gui/core/nlmeans/eigen/doc/examples/tut_matrix_coefficient_accessors.cpp index c2da1715..18f61331 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/examples/tut_matrix_coefficient_accessors.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/examples/tut_matrix_coefficient_accessors.cpp @@ -1,15 +1,15 @@ -#include #include +#include using namespace Eigen; int main() { - MatrixXd m(2,2); - m(0,0) = 3; - m(1,0) = 2.5; - m(0,1) = -1; - m(1,1) = m(1,0) + m(0,1); + MatrixXd m(2, 2); + m(0, 0) = 3; + m(1, 0) = 2.5; + m(0, 1) = -1; + m(1, 1) = m(1, 0) + m(0, 1); std::cout << "Here is the matrix m:\n" << m << std::endl; VectorXd v(2); v(0) = 4; diff --git a/filmulator-gui/core/nlmeans/eigen/doc/examples/tut_matrix_resize.cpp b/filmulator-gui/core/nlmeans/eigen/doc/examples/tut_matrix_resize.cpp index 0392c3aa..c1cd1f8e 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/examples/tut_matrix_resize.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/examples/tut_matrix_resize.cpp @@ -1,18 +1,16 @@ -#include #include +#include using namespace Eigen; int main() { - MatrixXd m(2,5); - m.resize(4,3); - std::cout << "The matrix m is of size " - << m.rows() << "x" << m.cols() << std::endl; + MatrixXd m(2, 5); + m.resize(4, 3); + std::cout << "The matrix m is of size " << m.rows() << "x" << m.cols() << std::endl; std::cout << "It has " << m.size() << " coefficients" << std::endl; VectorXd v(2); v.resize(5); std::cout << "The vector v is of size " << v.size() << std::endl; - std::cout << "As a matrix, v is of size " - << v.rows() << "x" << v.cols() << std::endl; + std::cout << "As a matrix, v is of size " << v.rows() << "x" << v.cols() << std::endl; } diff --git a/filmulator-gui/core/nlmeans/eigen/doc/examples/tut_matrix_resize_fixed_size.cpp b/filmulator-gui/core/nlmeans/eigen/doc/examples/tut_matrix_resize_fixed_size.cpp index dcbdfa78..b4868d7d 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/examples/tut_matrix_resize_fixed_size.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/examples/tut_matrix_resize_fixed_size.cpp @@ -1,12 +1,11 @@ -#include #include +#include using namespace Eigen; int main() { Matrix4d m; - m.resize(4,4); // no operation - std::cout << "The matrix m is of size " - << m.rows() << "x" << m.cols() << std::endl; + m.resize(4, 4);// no operation + std::cout << "The matrix m is of size " << m.rows() << "x" << m.cols() << std::endl; } diff --git a/filmulator-gui/core/nlmeans/eigen/doc/snippets/AngleAxis_mimic_euler.cpp b/filmulator-gui/core/nlmeans/eigen/doc/snippets/AngleAxis_mimic_euler.cpp index 456de7f7..00e6b0f9 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/snippets/AngleAxis_mimic_euler.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/snippets/AngleAxis_mimic_euler.cpp @@ -1,5 +1,4 @@ Matrix3f m; -m = AngleAxisf(0.25*M_PI, Vector3f::UnitX()) - * AngleAxisf(0.5*M_PI, Vector3f::UnitY()) - * AngleAxisf(0.33*M_PI, Vector3f::UnitZ()); +m = AngleAxisf(0.25 * M_PI, Vector3f::UnitX()) * AngleAxisf(0.5 * M_PI, Vector3f::UnitY()) + * AngleAxisf(0.33 * M_PI, Vector3f::UnitZ()); cout << m << endl << "is unitary: " << m.isUnitary() << endl; diff --git a/filmulator-gui/core/nlmeans/eigen/doc/snippets/BiCGSTAB_simple.cpp b/filmulator-gui/core/nlmeans/eigen/doc/snippets/BiCGSTAB_simple.cpp index 5520f4f1..1e32dbae 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/snippets/BiCGSTAB_simple.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/snippets/BiCGSTAB_simple.cpp @@ -1,11 +1,11 @@ - int n = 10000; - VectorXd x(n), b(n); - SparseMatrix A(n,n); - /* ... fill A and b ... */ - BiCGSTAB > solver; - solver.compute(A); - x = solver.solve(b); - std::cout << "#iterations: " << solver.iterations() << std::endl; - std::cout << "estimated error: " << solver.error() << std::endl; - /* ... update b ... */ - x = solver.solve(b); // solve again \ No newline at end of file +int n = 10000; +VectorXd x(n), b(n); +SparseMatrix A(n, n); +/* ... fill A and b ... */ +BiCGSTAB> solver; +solver.compute(A); +x = solver.solve(b); +std::cout << "#iterations: " << solver.iterations() << std::endl; +std::cout << "estimated error: " << solver.error() << std::endl; +/* ... update b ... */ +x = solver.solve(b);// solve again \ No newline at end of file diff --git a/filmulator-gui/core/nlmeans/eigen/doc/snippets/BiCGSTAB_step_by_step.cpp b/filmulator-gui/core/nlmeans/eigen/doc/snippets/BiCGSTAB_step_by_step.cpp index 06147bb8..051d2a20 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/snippets/BiCGSTAB_step_by_step.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/snippets/BiCGSTAB_step_by_step.cpp @@ -1,14 +1,14 @@ - int n = 10000; - VectorXd x(n), b(n); - SparseMatrix A(n,n); - /* ... fill A and b ... */ - BiCGSTAB > solver(A); - // start from a random solution - x = VectorXd::Random(n); - solver.setMaxIterations(1); - int i = 0; - do { - x = solver.solveWithGuess(b,x); - std::cout << i << " : " << solver.error() << std::endl; - ++i; - } while (solver.info()!=Success && i<100); \ No newline at end of file +int n = 10000; +VectorXd x(n), b(n); +SparseMatrix A(n, n); +/* ... fill A and b ... */ +BiCGSTAB> solver(A); +// start from a random solution +x = VectorXd::Random(n); +solver.setMaxIterations(1); +int i = 0; +do { + x = solver.solveWithGuess(b, x); + std::cout << i << " : " << solver.error() << std::endl; + ++i; +} while (solver.info() != Success && i < 100); \ No newline at end of file diff --git a/filmulator-gui/core/nlmeans/eigen/doc/snippets/ColPivHouseholderQR_solve.cpp b/filmulator-gui/core/nlmeans/eigen/doc/snippets/ColPivHouseholderQR_solve.cpp index b7b204a1..b5124d09 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/snippets/ColPivHouseholderQR_solve.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/snippets/ColPivHouseholderQR_solve.cpp @@ -4,5 +4,5 @@ cout << "Here is the matrix m:" << endl << m << endl; cout << "Here is the matrix y:" << endl << y << endl; Matrix3f x; x = m.colPivHouseholderQr().solve(y); -assert(y.isApprox(m*x)); +assert(y.isApprox(m *x)); cout << "Here is a solution x to the equation mx=y:" << endl << x << endl; diff --git a/filmulator-gui/core/nlmeans/eigen/doc/snippets/ComplexEigenSolver_compute.cpp b/filmulator-gui/core/nlmeans/eigen/doc/snippets/ComplexEigenSolver_compute.cpp index 11d6bd39..e55409d3 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/snippets/ComplexEigenSolver_compute.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/snippets/ComplexEigenSolver_compute.cpp @@ -1,4 +1,4 @@ -MatrixXcf A = MatrixXcf::Random(4,4); +MatrixXcf A = MatrixXcf::Random(4, 4); cout << "Here is a random 4x4 matrix, A:" << endl << A << endl << endl; ComplexEigenSolver ces; diff --git a/filmulator-gui/core/nlmeans/eigen/doc/snippets/ComplexEigenSolver_eigenvalues.cpp b/filmulator-gui/core/nlmeans/eigen/doc/snippets/ComplexEigenSolver_eigenvalues.cpp index 5509bd89..d3bff519 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/snippets/ComplexEigenSolver_eigenvalues.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/snippets/ComplexEigenSolver_eigenvalues.cpp @@ -1,4 +1,3 @@ -MatrixXcf ones = MatrixXcf::Ones(3,3); +MatrixXcf ones = MatrixXcf::Ones(3, 3); ComplexEigenSolver ces(ones, /* computeEigenvectors = */ false); -cout << "The eigenvalues of the 3x3 matrix of ones are:" - << endl << ces.eigenvalues() << endl; +cout << "The eigenvalues of the 3x3 matrix of ones are:" << endl << ces.eigenvalues() << endl; diff --git a/filmulator-gui/core/nlmeans/eigen/doc/snippets/ComplexEigenSolver_eigenvectors.cpp b/filmulator-gui/core/nlmeans/eigen/doc/snippets/ComplexEigenSolver_eigenvectors.cpp index bb1c2ccf..169ef04e 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/snippets/ComplexEigenSolver_eigenvectors.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/snippets/ComplexEigenSolver_eigenvectors.cpp @@ -1,4 +1,3 @@ -MatrixXcf ones = MatrixXcf::Ones(3,3); +MatrixXcf ones = MatrixXcf::Ones(3, 3); ComplexEigenSolver ces(ones); -cout << "The first eigenvector of the 3x3 matrix of ones is:" - << endl << ces.eigenvectors().col(1) << endl; +cout << "The first eigenvector of the 3x3 matrix of ones is:" << endl << ces.eigenvectors().col(1) << endl; diff --git a/filmulator-gui/core/nlmeans/eigen/doc/snippets/ComplexSchur_compute.cpp b/filmulator-gui/core/nlmeans/eigen/doc/snippets/ComplexSchur_compute.cpp index 3a517010..53b7f8b2 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/snippets/ComplexSchur_compute.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/snippets/ComplexSchur_compute.cpp @@ -1,4 +1,4 @@ -MatrixXcf A = MatrixXcf::Random(4,4); +MatrixXcf A = MatrixXcf::Random(4, 4); ComplexSchur schur(4); schur.compute(A); cout << "The matrix T in the decomposition of A is:" << endl << schur.matrixT() << endl; diff --git a/filmulator-gui/core/nlmeans/eigen/doc/snippets/ComplexSchur_matrixT.cpp b/filmulator-gui/core/nlmeans/eigen/doc/snippets/ComplexSchur_matrixT.cpp index 8380571a..c1ff1a0a 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/snippets/ComplexSchur_matrixT.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/snippets/ComplexSchur_matrixT.cpp @@ -1,4 +1,4 @@ -MatrixXcf A = MatrixXcf::Random(4,4); +MatrixXcf A = MatrixXcf::Random(4, 4); cout << "Here is a random 4x4 matrix, A:" << endl << A << endl << endl; -ComplexSchur schurOfA(A, false); // false means do not compute U +ComplexSchur schurOfA(A, false);// false means do not compute U cout << "The triangular matrix T is:" << endl << schurOfA.matrixT() << endl; diff --git a/filmulator-gui/core/nlmeans/eigen/doc/snippets/ComplexSchur_matrixU.cpp b/filmulator-gui/core/nlmeans/eigen/doc/snippets/ComplexSchur_matrixU.cpp index ba3d9c22..81383947 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/snippets/ComplexSchur_matrixU.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/snippets/ComplexSchur_matrixU.cpp @@ -1,4 +1,4 @@ -MatrixXcf A = MatrixXcf::Random(4,4); +MatrixXcf A = MatrixXcf::Random(4, 4); cout << "Here is a random 4x4 matrix, A:" << endl << A << endl << endl; ComplexSchur schurOfA(A); cout << "The unitary matrix U is:" << endl << schurOfA.matrixU() << endl; diff --git a/filmulator-gui/core/nlmeans/eigen/doc/snippets/Cwise_abs.cpp b/filmulator-gui/core/nlmeans/eigen/doc/snippets/Cwise_abs.cpp index 0aeec3a4..744851a0 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/snippets/Cwise_abs.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/snippets/Cwise_abs.cpp @@ -1,2 +1,2 @@ -Array3d v(1,-2,-3); +Array3d v(1, -2, -3); cout << v.abs() << endl; diff --git a/filmulator-gui/core/nlmeans/eigen/doc/snippets/Cwise_abs2.cpp b/filmulator-gui/core/nlmeans/eigen/doc/snippets/Cwise_abs2.cpp index 2c4f9b34..71e5cb2d 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/snippets/Cwise_abs2.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/snippets/Cwise_abs2.cpp @@ -1,2 +1,2 @@ -Array3d v(1,-2,-3); +Array3d v(1, -2, -3); cout << v.abs2() << endl; diff --git a/filmulator-gui/core/nlmeans/eigen/doc/snippets/Cwise_acos.cpp b/filmulator-gui/core/nlmeans/eigen/doc/snippets/Cwise_acos.cpp index 34432cba..bbbbf52f 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/snippets/Cwise_acos.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/snippets/Cwise_acos.cpp @@ -1,2 +1,2 @@ -Array3d v(0, sqrt(2.)/2, 1); +Array3d v(0, sqrt(2.) / 2, 1); cout << v.acos() << endl; diff --git a/filmulator-gui/core/nlmeans/eigen/doc/snippets/Cwise_array_power_array.cpp b/filmulator-gui/core/nlmeans/eigen/doc/snippets/Cwise_array_power_array.cpp index 432a76ee..6a0974c4 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/snippets/Cwise_array_power_array.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/snippets/Cwise_array_power_array.cpp @@ -1,4 +1,3 @@ -Array x(8,25,3), - e(1./3.,0.5,2.); -cout << "[" << x << "]^[" << e << "] = " << x.pow(e) << endl; // using ArrayBase::pow -cout << "[" << x << "]^[" << e << "] = " << pow(x,e) << endl; // using Eigen::pow +Array x(8, 25, 3), e(1. / 3., 0.5, 2.); +cout << "[" << x << "]^[" << e << "] = " << x.pow(e) << endl;// using ArrayBase::pow +cout << "[" << x << "]^[" << e << "] = " << pow(x, e) << endl;// using Eigen::pow diff --git a/filmulator-gui/core/nlmeans/eigen/doc/snippets/Cwise_asin.cpp b/filmulator-gui/core/nlmeans/eigen/doc/snippets/Cwise_asin.cpp index 8dad838f..3a64671d 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/snippets/Cwise_asin.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/snippets/Cwise_asin.cpp @@ -1,2 +1,2 @@ -Array3d v(0, sqrt(2.)/2, 1); +Array3d v(0, sqrt(2.) / 2, 1); cout << v.asin() << endl; diff --git a/filmulator-gui/core/nlmeans/eigen/doc/snippets/Cwise_atan.cpp b/filmulator-gui/core/nlmeans/eigen/doc/snippets/Cwise_atan.cpp index 44684472..1b60a9a0 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/snippets/Cwise_atan.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/snippets/Cwise_atan.cpp @@ -1,2 +1,2 @@ -ArrayXd v = ArrayXd::LinSpaced(5,0,1); +ArrayXd v = ArrayXd::LinSpaced(5, 0, 1); cout << v.atan() << endl; diff --git a/filmulator-gui/core/nlmeans/eigen/doc/snippets/Cwise_boolean_and.cpp b/filmulator-gui/core/nlmeans/eigen/doc/snippets/Cwise_boolean_and.cpp index df6b60d9..227ac36d 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/snippets/Cwise_boolean_and.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/snippets/Cwise_boolean_and.cpp @@ -1,2 +1,2 @@ -Array3d v(-1,2,1), w(-3,2,3); -cout << ((vw) << endl; +Array3d v(1, 2, 3), w(3, 2, 1); +cout << (v > w) << endl; diff --git a/filmulator-gui/core/nlmeans/eigen/doc/snippets/Cwise_greater_equal.cpp b/filmulator-gui/core/nlmeans/eigen/doc/snippets/Cwise_greater_equal.cpp index 6a08f894..8aa94e85 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/snippets/Cwise_greater_equal.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/snippets/Cwise_greater_equal.cpp @@ -1,2 +1,2 @@ -Array3d v(1,2,3), w(3,2,1); -cout << (v>=w) << endl; +Array3d v(1, 2, 3), w(3, 2, 1); +cout << (v >= w) << endl; diff --git a/filmulator-gui/core/nlmeans/eigen/doc/snippets/Cwise_inverse.cpp b/filmulator-gui/core/nlmeans/eigen/doc/snippets/Cwise_inverse.cpp index 3967a7ec..292daef5 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/snippets/Cwise_inverse.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/snippets/Cwise_inverse.cpp @@ -1,2 +1,2 @@ -Array3d v(2,3,4); +Array3d v(2, 3, 4); cout << v.inverse() << endl; diff --git a/filmulator-gui/core/nlmeans/eigen/doc/snippets/Cwise_isFinite.cpp b/filmulator-gui/core/nlmeans/eigen/doc/snippets/Cwise_isFinite.cpp index 1da55fd1..c3309f26 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/snippets/Cwise_isFinite.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/snippets/Cwise_isFinite.cpp @@ -1,5 +1,5 @@ -Array3d v(1,2,3); -v(1) *= 0.0/0.0; +Array3d v(1, 2, 3); +v(1) *= 0.0 / 0.0; v(2) /= 0.0; cout << v << endl << endl; cout << isfinite(v) << endl; diff --git a/filmulator-gui/core/nlmeans/eigen/doc/snippets/Cwise_isInf.cpp b/filmulator-gui/core/nlmeans/eigen/doc/snippets/Cwise_isInf.cpp index be793081..c97b8a6f 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/snippets/Cwise_isInf.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/snippets/Cwise_isInf.cpp @@ -1,5 +1,5 @@ -Array3d v(1,2,3); -v(1) *= 0.0/0.0; +Array3d v(1, 2, 3); +v(1) *= 0.0 / 0.0; v(2) /= 0.0; cout << v << endl << endl; cout << isinf(v) << endl; diff --git a/filmulator-gui/core/nlmeans/eigen/doc/snippets/Cwise_isNaN.cpp b/filmulator-gui/core/nlmeans/eigen/doc/snippets/Cwise_isNaN.cpp index 7b2a9308..ab2b5285 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/snippets/Cwise_isNaN.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/snippets/Cwise_isNaN.cpp @@ -1,5 +1,5 @@ -Array3d v(1,2,3); -v(1) *= 0.0/0.0; +Array3d v(1, 2, 3); +v(1) *= 0.0 / 0.0; v(2) /= 0.0; cout << v << endl << endl; cout << isnan(v) << endl; diff --git a/filmulator-gui/core/nlmeans/eigen/doc/snippets/Cwise_less.cpp b/filmulator-gui/core/nlmeans/eigen/doc/snippets/Cwise_less.cpp index cafd3b6e..95ccb302 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/snippets/Cwise_less.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/snippets/Cwise_less.cpp @@ -1,2 +1,2 @@ -Array3d v(1,2,3), w(3,2,1); -cout << (v e(2,-3,1./3.); -cout << "10^[" << e << "] = " << pow(10,e) << endl; +Array e(2, -3, 1. / 3.); +cout << "10^[" << e << "] = " << pow(10, e) << endl; diff --git a/filmulator-gui/core/nlmeans/eigen/doc/snippets/Cwise_sign.cpp b/filmulator-gui/core/nlmeans/eigen/doc/snippets/Cwise_sign.cpp index 49920e4f..55d24abf 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/snippets/Cwise_sign.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/snippets/Cwise_sign.cpp @@ -1,2 +1,2 @@ -Array3d v(-3,5,0); +Array3d v(-3, 5, 0); cout << v.sign() << endl; diff --git a/filmulator-gui/core/nlmeans/eigen/doc/snippets/Cwise_sin.cpp b/filmulator-gui/core/nlmeans/eigen/doc/snippets/Cwise_sin.cpp index 46fa908c..43c4a047 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/snippets/Cwise_sin.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/snippets/Cwise_sin.cpp @@ -1,2 +1,2 @@ -Array3d v(M_PI, M_PI/2, M_PI/3); +Array3d v(M_PI, M_PI / 2, M_PI / 3); cout << v.sin() << endl; diff --git a/filmulator-gui/core/nlmeans/eigen/doc/snippets/Cwise_sinh.cpp b/filmulator-gui/core/nlmeans/eigen/doc/snippets/Cwise_sinh.cpp index fac9b19a..aefcd6dd 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/snippets/Cwise_sinh.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/snippets/Cwise_sinh.cpp @@ -1,2 +1,2 @@ -ArrayXd v = ArrayXd::LinSpaced(5,0,1); +ArrayXd v = ArrayXd::LinSpaced(5, 0, 1); cout << sinh(v) << endl; diff --git a/filmulator-gui/core/nlmeans/eigen/doc/snippets/Cwise_slash_equal.cpp b/filmulator-gui/core/nlmeans/eigen/doc/snippets/Cwise_slash_equal.cpp index 2efd32d8..cb776b89 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/snippets/Cwise_slash_equal.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/snippets/Cwise_slash_equal.cpp @@ -1,3 +1,3 @@ -Array3d v(3,2,4), w(5,4,2); +Array3d v(3, 2, 4), w(5, 4, 2); v /= w; cout << v << endl; diff --git a/filmulator-gui/core/nlmeans/eigen/doc/snippets/Cwise_sqrt.cpp b/filmulator-gui/core/nlmeans/eigen/doc/snippets/Cwise_sqrt.cpp index 97bafe8b..e0e5d360 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/snippets/Cwise_sqrt.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/snippets/Cwise_sqrt.cpp @@ -1,2 +1,2 @@ -Array3d v(1,2,4); +Array3d v(1, 2, 4); cout << v.sqrt() << endl; diff --git a/filmulator-gui/core/nlmeans/eigen/doc/snippets/Cwise_square.cpp b/filmulator-gui/core/nlmeans/eigen/doc/snippets/Cwise_square.cpp index f704c5e0..2b3132e4 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/snippets/Cwise_square.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/snippets/Cwise_square.cpp @@ -1,2 +1,2 @@ -Array3d v(2,3,4); +Array3d v(2, 3, 4); cout << v.square() << endl; diff --git a/filmulator-gui/core/nlmeans/eigen/doc/snippets/Cwise_tan.cpp b/filmulator-gui/core/nlmeans/eigen/doc/snippets/Cwise_tan.cpp index b758ef04..abe8d1cf 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/snippets/Cwise_tan.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/snippets/Cwise_tan.cpp @@ -1,2 +1,2 @@ -Array3d v(M_PI, M_PI/2, M_PI/3); +Array3d v(M_PI, M_PI / 2, M_PI / 3); cout << v.tan() << endl; diff --git a/filmulator-gui/core/nlmeans/eigen/doc/snippets/Cwise_tanh.cpp b/filmulator-gui/core/nlmeans/eigen/doc/snippets/Cwise_tanh.cpp index 30cd0450..ae8fea5a 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/snippets/Cwise_tanh.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/snippets/Cwise_tanh.cpp @@ -1,2 +1,2 @@ -ArrayXd v = ArrayXd::LinSpaced(5,0,1); +ArrayXd v = ArrayXd::LinSpaced(5, 0, 1); cout << tanh(v) << endl; diff --git a/filmulator-gui/core/nlmeans/eigen/doc/snippets/Cwise_times_equal.cpp b/filmulator-gui/core/nlmeans/eigen/doc/snippets/Cwise_times_equal.cpp index 147556c7..45151be6 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/snippets/Cwise_times_equal.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/snippets/Cwise_times_equal.cpp @@ -1,3 +1,3 @@ -Array3d v(1,2,3), w(2,3,0); +Array3d v(1, 2, 3), w(2, 3, 0); v *= w; cout << v << endl; diff --git a/filmulator-gui/core/nlmeans/eigen/doc/snippets/DenseBase_LinSpaced.cpp b/filmulator-gui/core/nlmeans/eigen/doc/snippets/DenseBase_LinSpaced.cpp index 8e54b17f..a57053f9 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/snippets/DenseBase_LinSpaced.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/snippets/DenseBase_LinSpaced.cpp @@ -1,2 +1,2 @@ -cout << VectorXi::LinSpaced(4,7,10).transpose() << endl; -cout << VectorXd::LinSpaced(5,0.0,1.0).transpose() << endl; +cout << VectorXi::LinSpaced(4, 7, 10).transpose() << endl; +cout << VectorXd::LinSpaced(5, 0.0, 1.0).transpose() << endl; diff --git a/filmulator-gui/core/nlmeans/eigen/doc/snippets/DenseBase_LinSpacedInt.cpp b/filmulator-gui/core/nlmeans/eigen/doc/snippets/DenseBase_LinSpacedInt.cpp index 0d7ae068..732d70a7 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/snippets/DenseBase_LinSpacedInt.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/snippets/DenseBase_LinSpacedInt.cpp @@ -1,8 +1,8 @@ cout << "Even spacing inputs:" << endl; -cout << VectorXi::LinSpaced(8,1,4).transpose() << endl; -cout << VectorXi::LinSpaced(8,1,8).transpose() << endl; -cout << VectorXi::LinSpaced(8,1,15).transpose() << endl; +cout << VectorXi::LinSpaced(8, 1, 4).transpose() << endl; +cout << VectorXi::LinSpaced(8, 1, 8).transpose() << endl; +cout << VectorXi::LinSpaced(8, 1, 15).transpose() << endl; cout << "Uneven spacing inputs:" << endl; -cout << VectorXi::LinSpaced(8,1,7).transpose() << endl; -cout << VectorXi::LinSpaced(8,1,9).transpose() << endl; -cout << VectorXi::LinSpaced(8,1,16).transpose() << endl; +cout << VectorXi::LinSpaced(8, 1, 7).transpose() << endl; +cout << VectorXi::LinSpaced(8, 1, 9).transpose() << endl; +cout << VectorXi::LinSpaced(8, 1, 16).transpose() << endl; diff --git a/filmulator-gui/core/nlmeans/eigen/doc/snippets/DenseBase_LinSpaced_seq.cpp b/filmulator-gui/core/nlmeans/eigen/doc/snippets/DenseBase_LinSpaced_seq.cpp index f55c5085..3c4a565c 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/snippets/DenseBase_LinSpaced_seq.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/snippets/DenseBase_LinSpaced_seq.cpp @@ -1,2 +1,2 @@ -cout << VectorXi::LinSpaced(Sequential,4,7,10).transpose() << endl; -cout << VectorXd::LinSpaced(Sequential,5,0.0,1.0).transpose() << endl; +cout << VectorXi::LinSpaced(Sequential, 4, 7, 10).transpose() << endl; +cout << VectorXd::LinSpaced(Sequential, 5, 0.0, 1.0).transpose() << endl; diff --git a/filmulator-gui/core/nlmeans/eigen/doc/snippets/DenseBase_setLinSpaced.cpp b/filmulator-gui/core/nlmeans/eigen/doc/snippets/DenseBase_setLinSpaced.cpp index 46054f23..6c1eca67 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/snippets/DenseBase_setLinSpaced.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/snippets/DenseBase_setLinSpaced.cpp @@ -1,3 +1,3 @@ VectorXf v; -v.setLinSpaced(5,0.5f,1.5f); +v.setLinSpaced(5, 0.5f, 1.5f); cout << v << endl; diff --git a/filmulator-gui/core/nlmeans/eigen/doc/snippets/DirectionWise_hnormalized.cpp b/filmulator-gui/core/nlmeans/eigen/doc/snippets/DirectionWise_hnormalized.cpp index 3410790a..0645bd84 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/snippets/DirectionWise_hnormalized.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/snippets/DirectionWise_hnormalized.cpp @@ -1,7 +1,7 @@ -typedef Matrix Matrix4Xd; -Matrix4Xd M = Matrix4Xd::Random(4,5); +typedef Matrix Matrix4Xd; +Matrix4Xd M = Matrix4Xd::Random(4, 5); Projective3d P(Matrix4d::Random()); cout << "The matrix M is:" << endl << M << endl << endl; cout << "M.colwise().hnormalized():" << endl << M.colwise().hnormalized() << endl << endl; -cout << "P*M:" << endl << P*M << endl << endl; -cout << "(P*M).colwise().hnormalized():" << endl << (P*M).colwise().hnormalized() << endl << endl; \ No newline at end of file +cout << "P*M:" << endl << P * M << endl << endl; +cout << "(P*M).colwise().hnormalized():" << endl << (P * M).colwise().hnormalized() << endl << endl; \ No newline at end of file diff --git a/filmulator-gui/core/nlmeans/eigen/doc/snippets/DirectionWise_replicate.cpp b/filmulator-gui/core/nlmeans/eigen/doc/snippets/DirectionWise_replicate.cpp index d92d4a35..c448307e 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/snippets/DirectionWise_replicate.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/snippets/DirectionWise_replicate.cpp @@ -1,4 +1,4 @@ -MatrixXi m = MatrixXi::Random(2,3); +MatrixXi m = MatrixXi::Random(2, 3); cout << "Here is the matrix m:" << endl << m << endl; cout << "m.colwise().replicate<3>() = ..." << endl; cout << m.colwise().replicate<3>() << endl; diff --git a/filmulator-gui/core/nlmeans/eigen/doc/snippets/EigenSolver_EigenSolver_MatrixType.cpp b/filmulator-gui/core/nlmeans/eigen/doc/snippets/EigenSolver_EigenSolver_MatrixType.cpp index c1d9fa87..b77f1842 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/snippets/EigenSolver_EigenSolver_MatrixType.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/snippets/EigenSolver_EigenSolver_MatrixType.cpp @@ -1,4 +1,4 @@ -MatrixXd A = MatrixXd::Random(6,6); +MatrixXd A = MatrixXd::Random(6, 6); cout << "Here is a random 6x6 matrix, A:" << endl << A << endl << endl; EigenSolver es(A); @@ -9,7 +9,7 @@ complex lambda = es.eigenvalues()[0]; cout << "Consider the first eigenvalue, lambda = " << lambda << endl; VectorXcd v = es.eigenvectors().col(0); cout << "If v is the corresponding eigenvector, then lambda * v = " << endl << lambda * v << endl; -cout << "... and A * v = " << endl << A.cast >() * v << endl << endl; +cout << "... and A * v = " << endl << A.cast>() * v << endl << endl; MatrixXcd D = es.eigenvalues().asDiagonal(); MatrixXcd V = es.eigenvectors(); diff --git a/filmulator-gui/core/nlmeans/eigen/doc/snippets/EigenSolver_compute.cpp b/filmulator-gui/core/nlmeans/eigen/doc/snippets/EigenSolver_compute.cpp index a5c96e9b..6ed3b7a4 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/snippets/EigenSolver_compute.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/snippets/EigenSolver_compute.cpp @@ -1,6 +1,6 @@ EigenSolver es; -MatrixXf A = MatrixXf::Random(4,4); +MatrixXf A = MatrixXf::Random(4, 4); es.compute(A, /* computeEigenvectors = */ false); cout << "The eigenvalues of A are: " << es.eigenvalues().transpose() << endl; -es.compute(A + MatrixXf::Identity(4,4), false); // re-use es to compute eigenvalues of A+I +es.compute(A + MatrixXf::Identity(4, 4), false);// re-use es to compute eigenvalues of A+I cout << "The eigenvalues of A+I are: " << es.eigenvalues().transpose() << endl; diff --git a/filmulator-gui/core/nlmeans/eigen/doc/snippets/EigenSolver_eigenvalues.cpp b/filmulator-gui/core/nlmeans/eigen/doc/snippets/EigenSolver_eigenvalues.cpp index ed28869a..866183c8 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/snippets/EigenSolver_eigenvalues.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/snippets/EigenSolver_eigenvalues.cpp @@ -1,4 +1,3 @@ -MatrixXd ones = MatrixXd::Ones(3,3); +MatrixXd ones = MatrixXd::Ones(3, 3); EigenSolver es(ones, false); -cout << "The eigenvalues of the 3x3 matrix of ones are:" - << endl << es.eigenvalues() << endl; +cout << "The eigenvalues of the 3x3 matrix of ones are:" << endl << es.eigenvalues() << endl; diff --git a/filmulator-gui/core/nlmeans/eigen/doc/snippets/EigenSolver_eigenvectors.cpp b/filmulator-gui/core/nlmeans/eigen/doc/snippets/EigenSolver_eigenvectors.cpp index 8355f76c..6d4606dc 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/snippets/EigenSolver_eigenvectors.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/snippets/EigenSolver_eigenvectors.cpp @@ -1,4 +1,3 @@ -MatrixXd ones = MatrixXd::Ones(3,3); +MatrixXd ones = MatrixXd::Ones(3, 3); EigenSolver es(ones); -cout << "The first eigenvector of the 3x3 matrix of ones is:" - << endl << es.eigenvectors().col(0) << endl; +cout << "The first eigenvector of the 3x3 matrix of ones is:" << endl << es.eigenvectors().col(0) << endl; diff --git a/filmulator-gui/core/nlmeans/eigen/doc/snippets/EigenSolver_pseudoEigenvectors.cpp b/filmulator-gui/core/nlmeans/eigen/doc/snippets/EigenSolver_pseudoEigenvectors.cpp index 85e2569d..d9f3698a 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/snippets/EigenSolver_pseudoEigenvectors.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/snippets/EigenSolver_pseudoEigenvectors.cpp @@ -1,4 +1,4 @@ -MatrixXd A = MatrixXd::Random(6,6); +MatrixXd A = MatrixXd::Random(6, 6); cout << "Here is a random 6x6 matrix, A:" << endl << A << endl << endl; EigenSolver es(A); diff --git a/filmulator-gui/core/nlmeans/eigen/doc/snippets/FullPivHouseholderQR_solve.cpp b/filmulator-gui/core/nlmeans/eigen/doc/snippets/FullPivHouseholderQR_solve.cpp index 23bc0749..abb2a698 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/snippets/FullPivHouseholderQR_solve.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/snippets/FullPivHouseholderQR_solve.cpp @@ -4,5 +4,5 @@ cout << "Here is the matrix m:" << endl << m << endl; cout << "Here is the matrix y:" << endl << y << endl; Matrix3f x; x = m.fullPivHouseholderQr().solve(y); -assert(y.isApprox(m*x)); +assert(y.isApprox(m *x)); cout << "Here is a solution x to the equation mx=y:" << endl << x << endl; diff --git a/filmulator-gui/core/nlmeans/eigen/doc/snippets/FullPivLU_image.cpp b/filmulator-gui/core/nlmeans/eigen/doc/snippets/FullPivLU_image.cpp index 817bc1e2..a8f664e1 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/snippets/FullPivLU_image.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/snippets/FullPivLU_image.cpp @@ -1,9 +1,7 @@ Matrix3d m; -m << 1,1,0, - 1,3,2, - 0,1,1; +m << 1, 1, 0, 1, 3, 2, 0, 1, 1; cout << "Here is the matrix m:" << endl << m << endl; cout << "Notice that the middle column is the sum of the two others, so the " << "columns are linearly dependent." << endl; -cout << "Here is a matrix whose columns have the same span but are linearly independent:" - << endl << m.fullPivLu().image(m) << endl; +cout << "Here is a matrix whose columns have the same span but are linearly independent:" << endl + << m.fullPivLu().image(m) << endl; diff --git a/filmulator-gui/core/nlmeans/eigen/doc/snippets/FullPivLU_kernel.cpp b/filmulator-gui/core/nlmeans/eigen/doc/snippets/FullPivLU_kernel.cpp index 7086e01e..448a5157 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/snippets/FullPivLU_kernel.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/snippets/FullPivLU_kernel.cpp @@ -1,7 +1,5 @@ -MatrixXf m = MatrixXf::Random(3,5); +MatrixXf m = MatrixXf::Random(3, 5); cout << "Here is the matrix m:" << endl << m << endl; MatrixXf ker = m.fullPivLu().kernel(); -cout << "Here is a matrix whose columns form a basis of the kernel of m:" - << endl << ker << endl; -cout << "By definition of the kernel, m*ker is zero:" - << endl << m*ker << endl; +cout << "Here is a matrix whose columns form a basis of the kernel of m:" << endl << ker << endl; +cout << "By definition of the kernel, m*ker is zero:" << endl << m * ker << endl; diff --git a/filmulator-gui/core/nlmeans/eigen/doc/snippets/FullPivLU_solve.cpp b/filmulator-gui/core/nlmeans/eigen/doc/snippets/FullPivLU_solve.cpp index c1f88235..a4ecb838 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/snippets/FullPivLU_solve.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/snippets/FullPivLU_solve.cpp @@ -1,11 +1,9 @@ -Matrix m = Matrix::Random(); +Matrix m = Matrix::Random(); Matrix2f y = Matrix2f::Random(); cout << "Here is the matrix m:" << endl << m << endl; cout << "Here is the matrix y:" << endl << y << endl; -Matrix x = m.fullPivLu().solve(y); -if((m*x).isApprox(y)) -{ +Matrix x = m.fullPivLu().solve(y); +if ((m * x).isApprox(y)) { cout << "Here is a solution x to the equation mx=y:" << endl << x << endl; -} -else +} else cout << "The equation mx=y does not have any solution." << endl; diff --git a/filmulator-gui/core/nlmeans/eigen/doc/snippets/GeneralizedEigenSolver.cpp b/filmulator-gui/core/nlmeans/eigen/doc/snippets/GeneralizedEigenSolver.cpp index 2acda45f..f7a72544 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/snippets/GeneralizedEigenSolver.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/snippets/GeneralizedEigenSolver.cpp @@ -1,6 +1,6 @@ GeneralizedEigenSolver ges; -MatrixXf A = MatrixXf::Random(4,4); -MatrixXf B = MatrixXf::Random(4,4); +MatrixXf A = MatrixXf::Random(4, 4); +MatrixXf B = MatrixXf::Random(4, 4); ges.compute(A, B); cout << "The (complex) numerators of the generalzied eigenvalues are: " << ges.alphas().transpose() << endl; cout << "The (real) denominatore of the generalzied eigenvalues are: " << ges.betas().transpose() << endl; diff --git a/filmulator-gui/core/nlmeans/eigen/doc/snippets/HessenbergDecomposition_compute.cpp b/filmulator-gui/core/nlmeans/eigen/doc/snippets/HessenbergDecomposition_compute.cpp index 50e37833..15a2d257 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/snippets/HessenbergDecomposition_compute.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/snippets/HessenbergDecomposition_compute.cpp @@ -1,6 +1,6 @@ -MatrixXcf A = MatrixXcf::Random(4,4); +MatrixXcf A = MatrixXcf::Random(4, 4); HessenbergDecomposition hd(4); hd.compute(A); cout << "The matrix H in the decomposition of A is:" << endl << hd.matrixH() << endl; -hd.compute(2*A); // re-use hd to compute and store decomposition of 2A +hd.compute(2 * A);// re-use hd to compute and store decomposition of 2A cout << "The matrix H in the decomposition of 2A is:" << endl << hd.matrixH() << endl; diff --git a/filmulator-gui/core/nlmeans/eigen/doc/snippets/HessenbergDecomposition_matrixH.cpp b/filmulator-gui/core/nlmeans/eigen/doc/snippets/HessenbergDecomposition_matrixH.cpp index af013666..1386febc 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/snippets/HessenbergDecomposition_matrixH.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/snippets/HessenbergDecomposition_matrixH.cpp @@ -1,4 +1,4 @@ -Matrix4f A = MatrixXf::Random(4,4); +Matrix4f A = MatrixXf::Random(4, 4); cout << "Here is a random 4x4 matrix:" << endl << A << endl; HessenbergDecomposition hessOfA(A); MatrixXf H = hessOfA.matrixH(); diff --git a/filmulator-gui/core/nlmeans/eigen/doc/snippets/HessenbergDecomposition_packedMatrix.cpp b/filmulator-gui/core/nlmeans/eigen/doc/snippets/HessenbergDecomposition_packedMatrix.cpp index 4fa5957e..66c1fd0c 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/snippets/HessenbergDecomposition_packedMatrix.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/snippets/HessenbergDecomposition_packedMatrix.cpp @@ -1,9 +1,8 @@ -Matrix4d A = Matrix4d::Random(4,4); +Matrix4d A = Matrix4d::Random(4, 4); cout << "Here is a random 4x4 matrix:" << endl << A << endl; HessenbergDecomposition hessOfA(A); Matrix4d pm = hessOfA.packedMatrix(); cout << "The packed matrix M is:" << endl << pm << endl; -cout << "The upper Hessenberg part corresponds to the matrix H, which is:" - << endl << hessOfA.matrixH() << endl; +cout << "The upper Hessenberg part corresponds to the matrix H, which is:" << endl << hessOfA.matrixH() << endl; Vector3d hc = hessOfA.householderCoefficients(); cout << "The vector of Householder coefficients is:" << endl << hc << endl; diff --git a/filmulator-gui/core/nlmeans/eigen/doc/snippets/HouseholderQR_householderQ.cpp b/filmulator-gui/core/nlmeans/eigen/doc/snippets/HouseholderQR_householderQ.cpp index e859ce55..6b5cb92e 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/snippets/HouseholderQR_householderQ.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/snippets/HouseholderQR_householderQ.cpp @@ -1,4 +1,4 @@ -MatrixXf A(MatrixXf::Random(5,3)), thinQ(MatrixXf::Identity(5,3)), Q; +MatrixXf A(MatrixXf::Random(5, 3)), thinQ(MatrixXf::Identity(5, 3)), Q; A.setRandom(); HouseholderQR qr(A); Q = qr.householderQ(); diff --git a/filmulator-gui/core/nlmeans/eigen/doc/snippets/HouseholderQR_solve.cpp b/filmulator-gui/core/nlmeans/eigen/doc/snippets/HouseholderQR_solve.cpp index 8cce6ce6..5290e63d 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/snippets/HouseholderQR_solve.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/snippets/HouseholderQR_solve.cpp @@ -1,9 +1,9 @@ -typedef Matrix Matrix3x3; +typedef Matrix Matrix3x3; Matrix3x3 m = Matrix3x3::Random(); Matrix3f y = Matrix3f::Random(); cout << "Here is the matrix m:" << endl << m << endl; cout << "Here is the matrix y:" << endl << y << endl; Matrix3f x; x = m.householderQr().solve(y); -assert(y.isApprox(m*x)); +assert(y.isApprox(m *x)); cout << "Here is a solution x to the equation mx=y:" << endl << x << endl; diff --git a/filmulator-gui/core/nlmeans/eigen/doc/snippets/HouseholderSequence_HouseholderSequence.cpp b/filmulator-gui/core/nlmeans/eigen/doc/snippets/HouseholderSequence_HouseholderSequence.cpp index 2632b83b..ae088e68 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/snippets/HouseholderSequence_HouseholderSequence.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/snippets/HouseholderSequence_HouseholderSequence.cpp @@ -2,10 +2,10 @@ Matrix3d v = Matrix3d::Random(); cout << "The matrix v is:" << endl; cout << v << endl; -Vector3d v0(1, v(1,0), v(2,0)); +Vector3d v0(1, v(1, 0), v(2, 0)); cout << "The first Householder vector is: v_0 = " << v0.transpose() << endl; -Vector3d v1(0, 1, v(2,1)); -cout << "The second Householder vector is: v_1 = " << v1.transpose() << endl; +Vector3d v1(0, 1, v(2, 1)); +cout << "The second Householder vector is: v_1 = " << v1.transpose() << endl; Vector3d v2(0, 0, 1); cout << "The third Householder vector is: v_2 = " << v2.transpose() << endl; diff --git a/filmulator-gui/core/nlmeans/eigen/doc/snippets/JacobiSVD_basic.cpp b/filmulator-gui/core/nlmeans/eigen/doc/snippets/JacobiSVD_basic.cpp index ab24b9bc..5b02c93a 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/snippets/JacobiSVD_basic.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/snippets/JacobiSVD_basic.cpp @@ -1,4 +1,4 @@ -MatrixXf m = MatrixXf::Random(3,2); +MatrixXf m = MatrixXf::Random(3, 2); cout << "Here is the matrix m:" << endl << m << endl; JacobiSVD svd(m, ComputeThinU | ComputeThinV); cout << "Its singular values are:" << endl << svd.singularValues() << endl; diff --git a/filmulator-gui/core/nlmeans/eigen/doc/snippets/LLT_example.cpp b/filmulator-gui/core/nlmeans/eigen/doc/snippets/LLT_example.cpp index 46fb4070..dad16d74 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/snippets/LLT_example.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/snippets/LLT_example.cpp @@ -1,9 +1,9 @@ -MatrixXd A(3,3); -A << 4,-1,2, -1,6,0, 2,0,5; +MatrixXd A(3, 3); +A << 4, -1, 2, -1, 6, 0, 2, 0, 5; cout << "The matrix A is" << endl << A << endl; -LLT lltOfA(A); // compute the Cholesky decomposition of A -MatrixXd L = lltOfA.matrixL(); // retrieve factor L in the decomposition +LLT lltOfA(A);// compute the Cholesky decomposition of A +MatrixXd L = lltOfA.matrixL();// retrieve factor L in the decomposition // The previous two lines can also be written as "L = A.llt().matrixL()" cout << "The Cholesky factor L is" << endl << L << endl; diff --git a/filmulator-gui/core/nlmeans/eigen/doc/snippets/LLT_solve.cpp b/filmulator-gui/core/nlmeans/eigen/doc/snippets/LLT_solve.cpp index 7095d2cc..52659aba 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/snippets/LLT_solve.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/snippets/LLT_solve.cpp @@ -1,8 +1,7 @@ -typedef Matrix DataMatrix; +typedef Matrix DataMatrix; // let's generate some samples on the 3D plane of equation z = 2x+3y (with some noise) -DataMatrix samples = DataMatrix::Random(12,2); -VectorXf elevations = 2*samples.col(0) + 3*samples.col(1) + VectorXf::Random(12)*0.1; +DataMatrix samples = DataMatrix::Random(12, 2); +VectorXf elevations = 2 * samples.col(0) + 3 * samples.col(1) + VectorXf::Random(12) * 0.1; // and let's solve samples * [x y]^T = elevations in least square sense: -Matrix xy - = (samples.adjoint() * samples).llt().solve((samples.adjoint()*elevations)); +Matrix xy = (samples.adjoint() * samples).llt().solve((samples.adjoint() * elevations)); cout << xy << endl; diff --git a/filmulator-gui/core/nlmeans/eigen/doc/snippets/LeastSquaresNormalEquations.cpp b/filmulator-gui/core/nlmeans/eigen/doc/snippets/LeastSquaresNormalEquations.cpp index 997cf171..00434f78 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/snippets/LeastSquaresNormalEquations.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/snippets/LeastSquaresNormalEquations.cpp @@ -1,4 +1,3 @@ MatrixXf A = MatrixXf::Random(3, 2); VectorXf b = VectorXf::Random(3); -cout << "The solution using normal equations is:\n" - << (A.transpose() * A).ldlt().solve(A.transpose() * b) << endl; +cout << "The solution using normal equations is:\n" << (A.transpose() * A).ldlt().solve(A.transpose() * b) << endl; diff --git a/filmulator-gui/core/nlmeans/eigen/doc/snippets/LeastSquaresQR.cpp b/filmulator-gui/core/nlmeans/eigen/doc/snippets/LeastSquaresQR.cpp index 6c970454..9f8f723b 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/snippets/LeastSquaresQR.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/snippets/LeastSquaresQR.cpp @@ -1,4 +1,3 @@ MatrixXf A = MatrixXf::Random(3, 2); VectorXf b = VectorXf::Random(3); -cout << "The solution using the QR decomposition is:\n" - << A.colPivHouseholderQr().solve(b) << endl; +cout << "The solution using the QR decomposition is:\n" << A.colPivHouseholderQr().solve(b) << endl; diff --git a/filmulator-gui/core/nlmeans/eigen/doc/snippets/Map_general_stride.cpp b/filmulator-gui/core/nlmeans/eigen/doc/snippets/Map_general_stride.cpp index 0657e7f8..ddd588de 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/snippets/Map_general_stride.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/snippets/Map_general_stride.cpp @@ -1,5 +1,3 @@ int array[24]; -for(int i = 0; i < 24; ++i) array[i] = i; -cout << Map > - (array, 3, 3, Stride(8, 2)) - << endl; +for (int i = 0; i < 24; ++i) array[i] = i; +cout << Map>(array, 3, 3, Stride(8, 2)) << endl; diff --git a/filmulator-gui/core/nlmeans/eigen/doc/snippets/Map_inner_stride.cpp b/filmulator-gui/core/nlmeans/eigen/doc/snippets/Map_inner_stride.cpp index d95ae9b3..16a4bc8b 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/snippets/Map_inner_stride.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/snippets/Map_inner_stride.cpp @@ -1,5 +1,4 @@ int array[12]; -for(int i = 0; i < 12; ++i) array[i] = i; -cout << Map > - (array, 6) // the inner stride has already been passed as template parameter +for (int i = 0; i < 12; ++i) array[i] = i; +cout << Map>(array, 6)// the inner stride has already been passed as template parameter << endl; diff --git a/filmulator-gui/core/nlmeans/eigen/doc/snippets/Map_outer_stride.cpp b/filmulator-gui/core/nlmeans/eigen/doc/snippets/Map_outer_stride.cpp index 2f6f052c..2406d10e 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/snippets/Map_outer_stride.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/snippets/Map_outer_stride.cpp @@ -1,3 +1,3 @@ int array[12]; -for(int i = 0; i < 12; ++i) array[i] = i; -cout << Map >(array, 3, 3, OuterStride<>(4)) << endl; +for (int i = 0; i < 12; ++i) array[i] = i; +cout << Map>(array, 3, 3, OuterStride<>(4)) << endl; diff --git a/filmulator-gui/core/nlmeans/eigen/doc/snippets/Map_placement_new.cpp b/filmulator-gui/core/nlmeans/eigen/doc/snippets/Map_placement_new.cpp index 2e40eca3..1f4b80f0 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/snippets/Map_placement_new.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/snippets/Map_placement_new.cpp @@ -1,5 +1,5 @@ -int data[] = {1,2,3,4,5,6,7,8,9}; -Map v(data,4); +int data[] = { 1, 2, 3, 4, 5, 6, 7, 8, 9 }; +Map v(data, 4); cout << "The mapped vector v is: " << v << "\n"; -new (&v) Map(data+4,5); +new (&v) Map(data + 4, 5); cout << "Now v is: " << v << "\n"; \ No newline at end of file diff --git a/filmulator-gui/core/nlmeans/eigen/doc/snippets/Map_simple.cpp b/filmulator-gui/core/nlmeans/eigen/doc/snippets/Map_simple.cpp index 423bb52a..3e8fa812 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/snippets/Map_simple.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/snippets/Map_simple.cpp @@ -1,3 +1,3 @@ int array[9]; -for(int i = 0; i < 9; ++i) array[i] = i; +for (int i = 0; i < 9; ++i) array[i] = i; cout << Map(array) << endl; diff --git a/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_all.cpp b/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_all.cpp index 46f26f18..f31d10c1 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_all.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_all.cpp @@ -1,7 +1,7 @@ Vector3f boxMin(Vector3f::Zero()), boxMax(Vector3f::Ones()); Vector3f p0 = Vector3f::Random(), p1 = Vector3f::Random().cwiseAbs(); // let's check if p0 and p1 are inside the axis aligned box defined by the corners boxMin,boxMax: -cout << "Is (" << p0.transpose() << ") inside the box: " - << ((boxMin.array()p0.array()).all()) << endl; -cout << "Is (" << p1.transpose() << ") inside the box: " - << ((boxMin.array()p1.array()).all()) << endl; +cout << "Is (" << p0.transpose() + << ") inside the box: " << ((boxMin.array() < p0.array()).all() && (boxMax.array() > p0.array()).all()) << endl; +cout << "Is (" << p1.transpose() + << ") inside the box: " << ((boxMin.array() < p1.array()).all() && (boxMax.array() > p1.array()).all()) << endl; diff --git a/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_applyOnTheLeft.cpp b/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_applyOnTheLeft.cpp index 6398c873..00676a35 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_applyOnTheLeft.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_applyOnTheLeft.cpp @@ -1,7 +1,5 @@ -Matrix3f A = Matrix3f::Random(3,3), B; -B << 0,1,0, - 0,0,1, - 1,0,0; +Matrix3f A = Matrix3f::Random(3, 3), B; +B << 0, 1, 0, 0, 0, 1, 1, 0, 0; cout << "At start, A = " << endl << A << endl; -A.applyOnTheLeft(B); +A.applyOnTheLeft(B); cout << "After applyOnTheLeft, A = " << endl << A << endl; diff --git a/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_applyOnTheRight.cpp b/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_applyOnTheRight.cpp index e4b71b2d..d3dbcee4 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_applyOnTheRight.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_applyOnTheRight.cpp @@ -1,9 +1,7 @@ -Matrix3f A = Matrix3f::Random(3,3), B; -B << 0,1,0, - 0,0,1, - 1,0,0; +Matrix3f A = Matrix3f::Random(3, 3), B; +B << 0, 1, 0, 0, 0, 1, 1, 0, 0; cout << "At start, A = " << endl << A << endl; A *= B; cout << "After A *= B, A = " << endl << A << endl; -A.applyOnTheRight(B); // equivalent to A *= B +A.applyOnTheRight(B);// equivalent to A *= B cout << "After applyOnTheRight, A = " << endl << A << endl; diff --git a/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_array.cpp b/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_array.cpp index f215086d..456d1d61 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_array.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_array.cpp @@ -1,4 +1,4 @@ -Vector3d v(1,2,3); +Vector3d v(1, 2, 3); v.array() += 3; v.array() -= 2; cout << v << endl; diff --git a/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_array_const.cpp b/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_array_const.cpp index cd3b26a7..e662f6cd 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_array_const.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_array_const.cpp @@ -1,4 +1,4 @@ -Vector3d v(-1,2,-3); +Vector3d v(-1, 2, -3); cout << "the absolute values:" << endl << v.array().abs() << endl; -cout << "the absolute values plus one:" << endl << v.array().abs()+1 << endl; +cout << "the absolute values plus one:" << endl << v.array().abs() + 1 << endl; cout << "sum of the squares: " << v.array().square().sum() << endl; diff --git a/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_asDiagonal.cpp b/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_asDiagonal.cpp index b01082db..637e95b8 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_asDiagonal.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_asDiagonal.cpp @@ -1 +1 @@ -cout << Matrix3i(Vector3i(2,5,6).asDiagonal()) << endl; +cout << Matrix3i(Vector3i(2, 5, 6).asDiagonal()) << endl; diff --git a/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_block_int_int.cpp b/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_block_int_int.cpp index f99b6d4c..ade73ee3 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_block_int_int.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_block_int_int.cpp @@ -1,5 +1,5 @@ Matrix4i m = Matrix4i::Random(); cout << "Here is the matrix m:" << endl << m << endl; -cout << "Here is m.block<2,2>(1,1):" << endl << m.block<2,2>(1,1) << endl; -m.block<2,2>(1,1).setZero(); +cout << "Here is m.block<2,2>(1,1):" << endl << m.block<2, 2>(1, 1) << endl; +m.block<2, 2>(1, 1).setZero(); cout << "Now the matrix m is:" << endl << m << endl; diff --git a/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_col.cpp b/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_col.cpp index 87c91b12..cf0241e9 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_col.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_col.cpp @@ -1,3 +1,3 @@ Matrix3d m = Matrix3d::Identity(); -m.col(1) = Vector3d(4,5,6); +m.col(1) = Vector3d(4, 5, 6); cout << m << endl; diff --git a/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_colwise.cpp b/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_colwise.cpp index a048beff..a31033da 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_colwise.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_colwise.cpp @@ -1,5 +1,4 @@ Matrix3d m = Matrix3d::Random(); cout << "Here is the matrix m:" << endl << m << endl; cout << "Here is the sum of each column:" << endl << m.colwise().sum() << endl; -cout << "Here is the maximum absolute value of each column:" - << endl << m.cwiseAbs().colwise().maxCoeff() << endl; +cout << "Here is the maximum absolute value of each column:" << endl << m.cwiseAbs().colwise().maxCoeff() << endl; diff --git a/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_computeInverseAndDetWithCheck.cpp b/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_computeInverseAndDetWithCheck.cpp index a7b084fd..774d4d4f 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_computeInverseAndDetWithCheck.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_computeInverseAndDetWithCheck.cpp @@ -3,11 +3,10 @@ cout << "Here is the matrix m:" << endl << m << endl; Matrix3d inverse; bool invertible; double determinant; -m.computeInverseAndDetWithCheck(inverse,determinant,invertible); +m.computeInverseAndDetWithCheck(inverse, determinant, invertible); cout << "Its determinant is " << determinant << endl; -if(invertible) { +if (invertible) { cout << "It is invertible, and its inverse is:" << endl << inverse << endl; -} -else { +} else { cout << "It is not invertible." << endl; } diff --git a/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_computeInverseWithCheck.cpp b/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_computeInverseWithCheck.cpp index 873a9f87..7c95100f 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_computeInverseWithCheck.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_computeInverseWithCheck.cpp @@ -2,10 +2,9 @@ Matrix3d m = Matrix3d::Random(); cout << "Here is the matrix m:" << endl << m << endl; Matrix3d inverse; bool invertible; -m.computeInverseWithCheck(inverse,invertible); -if(invertible) { +m.computeInverseWithCheck(inverse, invertible); +if (invertible) { cout << "It is invertible, and its inverse is:" << endl << inverse << endl; -} -else { +} else { cout << "It is not invertible." << endl; } diff --git a/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_cwiseAbs.cpp b/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_cwiseAbs.cpp index 28a31600..e4241907 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_cwiseAbs.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_cwiseAbs.cpp @@ -1,4 +1,3 @@ -MatrixXd m(2,3); -m << 2, -4, 6, - -5, 1, 0; +MatrixXd m(2, 3); +m << 2, -4, 6, -5, 1, 0; cout << m.cwiseAbs() << endl; diff --git a/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_cwiseAbs2.cpp b/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_cwiseAbs2.cpp index 889a2e2b..952ba9d3 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_cwiseAbs2.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_cwiseAbs2.cpp @@ -1,4 +1,3 @@ -MatrixXd m(2,3); -m << 2, -4, 6, - -5, 1, 0; +MatrixXd m(2, 3); +m << 2, -4, 6, -5, 1, 0; cout << m.cwiseAbs2() << endl; diff --git a/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_cwiseEqual.cpp b/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_cwiseEqual.cpp index 469af642..dae553c8 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_cwiseEqual.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_cwiseEqual.cpp @@ -1,7 +1,6 @@ -MatrixXi m(2,2); -m << 1, 0, - 1, 1; +MatrixXi m(2, 2); +m << 1, 0, 1, 1; cout << "Comparing m with identity matrix:" << endl; -cout << m.cwiseEqual(MatrixXi::Identity(2,2)) << endl; -Index count = m.cwiseEqual(MatrixXi::Identity(2,2)).count(); +cout << m.cwiseEqual(MatrixXi::Identity(2, 2)) << endl; +Index count = m.cwiseEqual(MatrixXi::Identity(2, 2)).count(); cout << "Number of coefficients that are equal: " << count << endl; diff --git a/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_cwiseInverse.cpp b/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_cwiseInverse.cpp index 23e08f7b..74551cbb 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_cwiseInverse.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_cwiseInverse.cpp @@ -1,4 +1,3 @@ -MatrixXd m(2,3); -m << 2, 0.5, 1, - 3, 0.25, 1; +MatrixXd m(2, 3); +m << 2, 0.5, 1, 3, 0.25, 1; cout << m.cwiseInverse() << endl; diff --git a/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_cwiseMax.cpp b/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_cwiseMax.cpp index 3c956818..cd613b58 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_cwiseMax.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_cwiseMax.cpp @@ -1,2 +1,2 @@ -Vector3d v(2,3,4), w(4,2,3); +Vector3d v(2, 3, 4), w(4, 2, 3); cout << v.cwiseMax(w) << endl; diff --git a/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_cwiseMin.cpp b/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_cwiseMin.cpp index 82fc761e..6fa93f36 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_cwiseMin.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_cwiseMin.cpp @@ -1,2 +1,2 @@ -Vector3d v(2,3,4), w(4,2,3); +Vector3d v(2, 3, 4), w(4, 2, 3); cout << v.cwiseMin(w) << endl; diff --git a/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_cwiseNotEqual.cpp b/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_cwiseNotEqual.cpp index 7f0a105d..1a3ec73a 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_cwiseNotEqual.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_cwiseNotEqual.cpp @@ -1,7 +1,6 @@ -MatrixXi m(2,2); -m << 1, 0, - 1, 1; +MatrixXi m(2, 2); +m << 1, 0, 1, 1; cout << "Comparing m with identity matrix:" << endl; -cout << m.cwiseNotEqual(MatrixXi::Identity(2,2)) << endl; -Index count = m.cwiseNotEqual(MatrixXi::Identity(2,2)).count(); +cout << m.cwiseNotEqual(MatrixXi::Identity(2, 2)) << endl; +Index count = m.cwiseNotEqual(MatrixXi::Identity(2, 2)).count(); cout << "Number of coefficients that are not equal: " << count << endl; diff --git a/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_cwiseProduct.cpp b/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_cwiseProduct.cpp index 1db3a113..79540a9d 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_cwiseProduct.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_cwiseProduct.cpp @@ -1,4 +1,3 @@ Matrix3i a = Matrix3i::Random(), b = Matrix3i::Random(); Matrix3i c = a.cwiseProduct(b); cout << "a:\n" << a << "\nb:\n" << b << "\nc:\n" << c << endl; - diff --git a/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_cwiseQuotient.cpp b/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_cwiseQuotient.cpp index 96912120..c78110fc 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_cwiseQuotient.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_cwiseQuotient.cpp @@ -1,2 +1,2 @@ -Vector3d v(2,3,4), w(4,2,3); +Vector3d v(2, 3, 4), w(4, 2, 3); cout << v.cwiseQuotient(w) << endl; diff --git a/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_cwiseSign.cpp b/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_cwiseSign.cpp index efd71795..c2ee94d4 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_cwiseSign.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_cwiseSign.cpp @@ -1,4 +1,3 @@ -MatrixXd m(2,3); -m << 2, -4, 6, - -5, 1, 0; +MatrixXd m(2, 3); +m << 2, -4, 6, -5, 1, 0; cout << m.cwiseSign() << endl; diff --git a/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_cwiseSqrt.cpp b/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_cwiseSqrt.cpp index 4bfd75d5..5bfb5f38 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_cwiseSqrt.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_cwiseSqrt.cpp @@ -1,2 +1,2 @@ -Vector3d v(1,2,4); +Vector3d v(1, 2, 4); cout << v.cwiseSqrt() << endl; diff --git a/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_diagonal.cpp b/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_diagonal.cpp index cd63413f..c15dcf10 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_diagonal.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_diagonal.cpp @@ -1,4 +1,3 @@ Matrix3i m = Matrix3i::Random(); cout << "Here is the matrix m:" << endl << m << endl; -cout << "Here are the coefficients on the main diagonal of m:" << endl - << m.diagonal() << endl; +cout << "Here are the coefficients on the main diagonal of m:" << endl << m.diagonal() << endl; diff --git a/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_eigenvalues.cpp b/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_eigenvalues.cpp index 039f8870..010f45dc 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_eigenvalues.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_eigenvalues.cpp @@ -1,3 +1,3 @@ -MatrixXd ones = MatrixXd::Ones(3,3); +MatrixXd ones = MatrixXd::Ones(3, 3); VectorXcd eivals = ones.eigenvalues(); cout << "The eigenvalues of the 3x3 matrix of ones are:" << endl << eivals << endl; diff --git a/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_fixedBlock_int_int.cpp b/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_fixedBlock_int_int.cpp index 32011274..6a3a8cba 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_fixedBlock_int_int.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_fixedBlock_int_int.cpp @@ -1,4 +1,4 @@ -Matrix4d m = Vector4d(1,2,3,4).asDiagonal(); +Matrix4d m = Vector4d(1, 2, 3, 4).asDiagonal(); cout << "Here is the matrix m:" << endl << m << endl; cout << "Here is m.fixed<2, 2>(2, 2):" << endl << m.block<2, 2>(2, 2) << endl; m.block<2, 2>(2, 0) = m.block<2, 2>(2, 2); diff --git a/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_hnormalized.cpp b/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_hnormalized.cpp index 652cd77c..71b7576e 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_hnormalized.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_hnormalized.cpp @@ -2,5 +2,5 @@ Vector4d v = Vector4d::Random(); Projective3d P(Matrix4d::Random()); cout << "v = " << v.transpose() << "]^T" << endl; cout << "v.hnormalized() = " << v.hnormalized().transpose() << "]^T" << endl; -cout << "P*v = " << (P*v).transpose() << "]^T" << endl; -cout << "(P*v).hnormalized() = " << (P*v).hnormalized().transpose() << "]^T" << endl; \ No newline at end of file +cout << "P*v = " << (P * v).transpose() << "]^T" << endl; +cout << "(P*v).hnormalized() = " << (P * v).hnormalized().transpose() << "]^T" << endl; \ No newline at end of file diff --git a/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_homogeneous.cpp b/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_homogeneous.cpp index 457c28f9..be661431 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_homogeneous.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_homogeneous.cpp @@ -3,4 +3,5 @@ Projective3d P(Matrix4d::Random()); cout << "v = [" << v.transpose() << "]^T" << endl; cout << "h.homogeneous() = [" << v.homogeneous().transpose() << "]^T" << endl; cout << "(P * v.homogeneous()) = [" << (P * v.homogeneous()).transpose() << "]^T" << endl; -cout << "(P * v.homogeneous()).hnormalized() = [" << (P * v.homogeneous()).eval().hnormalized().transpose() << "]^T" << endl; \ No newline at end of file +cout << "(P * v.homogeneous()).hnormalized() = [" << (P * v.homogeneous()).eval().hnormalized().transpose() << "]^T" + << endl; \ No newline at end of file diff --git a/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_isDiagonal.cpp b/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_isDiagonal.cpp index 5b1d5997..290654fd 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_isDiagonal.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_isDiagonal.cpp @@ -1,6 +1,5 @@ Matrix3d m = 10000 * Matrix3d::Identity(); -m(0,2) = 1; +m(0, 2) = 1; cout << "Here's the matrix m:" << endl << m << endl; cout << "m.isDiagonal() returns: " << m.isDiagonal() << endl; cout << "m.isDiagonal(1e-3) returns: " << m.isDiagonal(1e-3) << endl; - diff --git a/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_isIdentity.cpp b/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_isIdentity.cpp index 17b756c9..ea535c78 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_isIdentity.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_isIdentity.cpp @@ -1,5 +1,5 @@ Matrix3d m = Matrix3d::Identity(); -m(0,2) = 1e-4; +m(0, 2) = 1e-4; cout << "Here's the matrix m:" << endl << m << endl; cout << "m.isIdentity() returns: " << m.isIdentity() << endl; cout << "m.isIdentity(1e-3) returns: " << m.isIdentity(1e-3) << endl; diff --git a/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_isOnes.cpp b/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_isOnes.cpp index f82f6280..899230ab 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_isOnes.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_isOnes.cpp @@ -1,5 +1,5 @@ Matrix3d m = Matrix3d::Ones(); -m(0,2) += 1e-4; +m(0, 2) += 1e-4; cout << "Here's the matrix m:" << endl << m << endl; cout << "m.isOnes() returns: " << m.isOnes() << endl; cout << "m.isOnes(1e-3) returns: " << m.isOnes(1e-3) << endl; diff --git a/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_isOrthogonal.cpp b/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_isOrthogonal.cpp index b22af066..3e079b4a 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_isOrthogonal.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_isOrthogonal.cpp @@ -1,6 +1,6 @@ -Vector3d v(1,0,0); -Vector3d w(1e-4,0,1); +Vector3d v(1, 0, 0); +Vector3d w(1e-4, 0, 1); cout << "Here's the vector v:" << endl << v << endl; cout << "Here's the vector w:" << endl << w << endl; cout << "v.isOrthogonal(w) returns: " << v.isOrthogonal(w) << endl; -cout << "v.isOrthogonal(w,1e-3) returns: " << v.isOrthogonal(w,1e-3) << endl; +cout << "v.isOrthogonal(w,1e-3) returns: " << v.isOrthogonal(w, 1e-3) << endl; diff --git a/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_isUnitary.cpp b/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_isUnitary.cpp index 3877da34..56f1b964 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_isUnitary.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_isUnitary.cpp @@ -1,5 +1,5 @@ Matrix3d m = Matrix3d::Identity(); -m(0,2) = 1e-4; +m(0, 2) = 1e-4; cout << "Here's the matrix m:" << endl << m << endl; cout << "m.isUnitary() returns: " << m.isUnitary() << endl; cout << "m.isUnitary(1e-3) returns: " << m.isUnitary(1e-3) << endl; diff --git a/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_isZero.cpp b/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_isZero.cpp index c2cfe220..c878f1c4 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_isZero.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_isZero.cpp @@ -1,5 +1,5 @@ Matrix3d m = Matrix3d::Zero(); -m(0,2) = 1e-4; +m(0, 2) = 1e-4; cout << "Here's the matrix m:" << endl << m << endl; cout << "m.isZero() returns: " << m.isZero() << endl; cout << "m.isZero(1e-3) returns: " << m.isZero(1e-3) << endl; diff --git a/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_noalias.cpp b/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_noalias.cpp index 3b54a79a..c918d8f6 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_noalias.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_noalias.cpp @@ -1,3 +1,5 @@ -Matrix2d a, b, c; a << 1,2,3,4; b << 5,6,7,8; -c.noalias() = a * b; // this computes the product directly to c +Matrix2d a, b, c; +a << 1, 2, 3, 4; +b << 5, 6, 7, 8; +c.noalias() = a * b;// this computes the product directly to c cout << c << endl; diff --git a/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_ones_int_int.cpp b/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_ones_int_int.cpp index 60f5a31e..38bd8cf2 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_ones_int_int.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_ones_int_int.cpp @@ -1 +1 @@ -cout << MatrixXi::Ones(2,3) << endl; +cout << MatrixXi::Ones(2, 3) << endl; diff --git a/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_operatorNorm.cpp b/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_operatorNorm.cpp index 355246f0..99ca14d8 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_operatorNorm.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_operatorNorm.cpp @@ -1,3 +1,2 @@ -MatrixXd ones = MatrixXd::Ones(3,3); -cout << "The operator norm of the 3x3 matrix of ones is " - << ones.operatorNorm() << endl; +MatrixXd ones = MatrixXd::Ones(3, 3); +cout << "The operator norm of the 3x3 matrix of ones is " << ones.operatorNorm() << endl; diff --git a/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_random_int_int.cpp b/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_random_int_int.cpp index 3f0f7dd5..92f87c8c 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_random_int_int.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_random_int_int.cpp @@ -1 +1 @@ -cout << MatrixXi::Random(2,3) << endl; +cout << MatrixXi::Random(2, 3) << endl; diff --git a/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_replicate.cpp b/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_replicate.cpp index 3ce52bcd..bffd1021 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_replicate.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_replicate.cpp @@ -1,4 +1,4 @@ -MatrixXi m = MatrixXi::Random(2,3); +MatrixXi m = MatrixXi::Random(2, 3); cout << "Here is the matrix m:" << endl << m << endl; cout << "m.replicate<3,2>() = ..." << endl; -cout << m.replicate<3,2>() << endl; +cout << m.replicate<3, 2>() << endl; diff --git a/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_replicate_int_int.cpp b/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_replicate_int_int.cpp index b1dbc70b..1aee265b 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_replicate_int_int.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_replicate_int_int.cpp @@ -1,4 +1,4 @@ Vector3i v = Vector3i::Random(); cout << "Here is the vector v:" << endl << v << endl; cout << "v.replicate(2,5) = ..." << endl; -cout << v.replicate(2,5) << endl; +cout << v.replicate(2, 5) << endl; diff --git a/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_reverse.cpp b/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_reverse.cpp index f545a283..ca125580 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_reverse.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_reverse.cpp @@ -1,8 +1,7 @@ -MatrixXi m = MatrixXi::Random(3,4); +MatrixXi m = MatrixXi::Random(3, 4); cout << "Here is the matrix m:" << endl << m << endl; cout << "Here is the reverse of m:" << endl << m.reverse() << endl; -cout << "Here is the coefficient (1,0) in the reverse of m:" << endl - << m.reverse()(1,0) << endl; +cout << "Here is the coefficient (1,0) in the reverse of m:" << endl << m.reverse()(1, 0) << endl; cout << "Let us overwrite this coefficient with the value 4." << endl; -m.reverse()(1,0) = 4; +m.reverse()(1, 0) = 4; cout << "Now the matrix m is:" << endl << m << endl; diff --git a/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_row.cpp b/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_row.cpp index b15e6260..481bff84 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_row.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_row.cpp @@ -1,3 +1,3 @@ Matrix3d m = Matrix3d::Identity(); -m.row(1) = Vector3d(4,5,6); +m.row(1) = Vector3d(4, 5, 6); cout << m << endl; diff --git a/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_rowwise.cpp b/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_rowwise.cpp index ae93964e..b869224a 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_rowwise.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_rowwise.cpp @@ -1,5 +1,4 @@ Matrix3d m = Matrix3d::Random(); cout << "Here is the matrix m:" << endl << m << endl; cout << "Here is the sum of each row:" << endl << m.rowwise().sum() << endl; -cout << "Here is the maximum absolute value of each row:" - << endl << m.cwiseAbs().rowwise().maxCoeff() << endl; +cout << "Here is the maximum absolute value of each row:" << endl << m.cwiseAbs().rowwise().maxCoeff() << endl; diff --git a/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_select.cpp b/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_select.cpp index ae5477f0..d14941bd 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_select.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_select.cpp @@ -1,6 +1,4 @@ MatrixXi m(3, 3); -m << 1, 2, 3, - 4, 5, 6, - 7, 8, 9; +m << 1, 2, 3, 4, 5, 6, 7, 8, 9; m = (m.array() >= 5).select(-m, m); cout << m << endl; diff --git a/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_set.cpp b/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_set.cpp index 50ecf5fb..d392b68b 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_set.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_set.cpp @@ -1,13 +1,10 @@ Matrix3i m1; -m1 << 1, 2, 3, - 4, 5, 6, - 7, 8, 9; +m1 << 1, 2, 3, 4, 5, 6, 7, 8, 9; cout << m1 << endl << endl; Matrix3i m2 = Matrix3i::Identity(); -m2.block(0,0, 2,2) << 10, 11, 12, 13; +m2.block(0, 0, 2, 2) << 10, 11, 12, 13; cout << m2 << endl << endl; Vector2i v1; v1 << 14, 15; -m2 << v1.transpose(), 16, - v1, m1.block(1,1,2,2); +m2 << v1.transpose(), 16, v1, m1.block(1, 1, 2, 2); cout << m2 << endl; diff --git a/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_setIdentity.cpp b/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_setIdentity.cpp index 4fd0aa24..916b477c 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_setIdentity.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_setIdentity.cpp @@ -1,3 +1,3 @@ Matrix4i m = Matrix4i::Zero(); -m.block<3,3>(1,0).setIdentity(); +m.block<3, 3>(1, 0).setIdentity(); cout << m << endl; diff --git a/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_template_int_int_bottomLeftCorner.cpp b/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_template_int_int_bottomLeftCorner.cpp index 847892a2..bf9f0ab0 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_template_int_int_bottomLeftCorner.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_template_int_int_bottomLeftCorner.cpp @@ -1,6 +1,6 @@ Matrix4i m = Matrix4i::Random(); cout << "Here is the matrix m:" << endl << m << endl; cout << "Here is m.bottomLeftCorner<2,2>():" << endl; -cout << m.bottomLeftCorner<2,2>() << endl; -m.bottomLeftCorner<2,2>().setZero(); +cout << m.bottomLeftCorner<2, 2>() << endl; +m.bottomLeftCorner<2, 2>().setZero(); cout << "Now the matrix m is:" << endl << m << endl; diff --git a/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_template_int_int_bottomLeftCorner_int_int.cpp b/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_template_int_int_bottomLeftCorner_int_int.cpp index a1edcc80..3d22fc2f 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_template_int_int_bottomLeftCorner_int_int.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_template_int_int_bottomLeftCorner_int_int.cpp @@ -1,6 +1,6 @@ Matrix4i m = Matrix4i::Random(); cout << "Here is the matrix m:" << endl << m << endl; cout << "Here is m.bottomLeftCorner<2,Dynamic>(2,2):" << endl; -cout << m.bottomLeftCorner<2,Dynamic>(2,2) << endl; -m.bottomLeftCorner<2,Dynamic>(2,2).setZero(); +cout << m.bottomLeftCorner<2, Dynamic>(2, 2) << endl; +m.bottomLeftCorner<2, Dynamic>(2, 2).setZero(); cout << "Now the matrix m is:" << endl << m << endl; diff --git a/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_template_int_int_bottomRightCorner.cpp b/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_template_int_int_bottomRightCorner.cpp index abacb014..4a78bed5 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_template_int_int_bottomRightCorner.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_template_int_int_bottomRightCorner.cpp @@ -1,6 +1,6 @@ Matrix4i m = Matrix4i::Random(); cout << "Here is the matrix m:" << endl << m << endl; cout << "Here is m.bottomRightCorner<2,2>():" << endl; -cout << m.bottomRightCorner<2,2>() << endl; -m.bottomRightCorner<2,2>().setZero(); +cout << m.bottomRightCorner<2, 2>() << endl; +m.bottomRightCorner<2, 2>().setZero(); cout << "Now the matrix m is:" << endl << m << endl; diff --git a/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_template_int_int_bottomRightCorner_int_int.cpp b/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_template_int_int_bottomRightCorner_int_int.cpp index a65508fd..75a3e939 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_template_int_int_bottomRightCorner_int_int.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_template_int_int_bottomRightCorner_int_int.cpp @@ -1,6 +1,6 @@ Matrix4i m = Matrix4i::Random(); cout << "Here is the matrix m:" << endl << m << endl; cout << "Here is m.bottomRightCorner<2,Dynamic>(2,2):" << endl; -cout << m.bottomRightCorner<2,Dynamic>(2,2) << endl; -m.bottomRightCorner<2,Dynamic>(2,2).setZero(); +cout << m.bottomRightCorner<2, Dynamic>(2, 2) << endl; +m.bottomRightCorner<2, Dynamic>(2, 2).setZero(); cout << "Now the matrix m is:" << endl << m << endl; diff --git a/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_template_int_int_topLeftCorner.cpp b/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_template_int_int_topLeftCorner.cpp index 1899d902..7c765d43 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_template_int_int_topLeftCorner.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_template_int_int_topLeftCorner.cpp @@ -1,6 +1,6 @@ Matrix4i m = Matrix4i::Random(); cout << "Here is the matrix m:" << endl << m << endl; cout << "Here is m.topLeftCorner<2,2>():" << endl; -cout << m.topLeftCorner<2,2>() << endl; -m.topLeftCorner<2,2>().setZero(); +cout << m.topLeftCorner<2, 2>() << endl; +m.topLeftCorner<2, 2>().setZero(); cout << "Now the matrix m is:" << endl << m << endl; diff --git a/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_template_int_int_topLeftCorner_int_int.cpp b/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_template_int_int_topLeftCorner_int_int.cpp index fac761f6..ae72684b 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_template_int_int_topLeftCorner_int_int.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_template_int_int_topLeftCorner_int_int.cpp @@ -1,6 +1,6 @@ Matrix4i m = Matrix4i::Random(); cout << "Here is the matrix m:" << endl << m << endl; cout << "Here is m.topLeftCorner<2,Dynamic>(2,2):" << endl; -cout << m.topLeftCorner<2,Dynamic>(2,2) << endl; -m.topLeftCorner<2,Dynamic>(2,2).setZero(); +cout << m.topLeftCorner<2, Dynamic>(2, 2) << endl; +m.topLeftCorner<2, Dynamic>(2, 2).setZero(); cout << "Now the matrix m is:" << endl << m << endl; diff --git a/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_template_int_int_topRightCorner.cpp b/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_template_int_int_topRightCorner.cpp index c3a17711..9698521b 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_template_int_int_topRightCorner.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_template_int_int_topRightCorner.cpp @@ -1,6 +1,6 @@ Matrix4i m = Matrix4i::Random(); cout << "Here is the matrix m:" << endl << m << endl; cout << "Here is m.topRightCorner<2,2>():" << endl; -cout << m.topRightCorner<2,2>() << endl; -m.topRightCorner<2,2>().setZero(); +cout << m.topRightCorner<2, 2>() << endl; +m.topRightCorner<2, 2>().setZero(); cout << "Now the matrix m is:" << endl << m << endl; diff --git a/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_template_int_int_topRightCorner_int_int.cpp b/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_template_int_int_topRightCorner_int_int.cpp index a17acc00..4144242d 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_template_int_int_topRightCorner_int_int.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_template_int_int_topRightCorner_int_int.cpp @@ -1,6 +1,6 @@ Matrix4i m = Matrix4i::Random(); cout << "Here is the matrix m:" << endl << m << endl; cout << "Here is m.topRightCorner<2,Dynamic>(2,2):" << endl; -cout << m.topRightCorner<2,Dynamic>(2,2) << endl; -m.topRightCorner<2,Dynamic>(2,2).setZero(); +cout << m.topRightCorner<2, Dynamic>(2, 2) << endl; +m.topRightCorner<2, Dynamic>(2, 2).setZero(); cout << "Now the matrix m is:" << endl << m << endl; diff --git a/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_transpose.cpp b/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_transpose.cpp index 88eea83c..8bf2716e 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_transpose.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_transpose.cpp @@ -1,8 +1,7 @@ Matrix2i m = Matrix2i::Random(); cout << "Here is the matrix m:" << endl << m << endl; cout << "Here is the transpose of m:" << endl << m.transpose() << endl; -cout << "Here is the coefficient (1,0) in the transpose of m:" << endl - << m.transpose()(1,0) << endl; +cout << "Here is the coefficient (1,0) in the transpose of m:" << endl << m.transpose()(1, 0) << endl; cout << "Let us overwrite this coefficient with the value 0." << endl; -m.transpose()(1,0) = 0; +m.transpose()(1, 0) = 0; cout << "Now the matrix m is:" << endl << m << endl; diff --git a/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_zero_int_int.cpp b/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_zero_int_int.cpp index 4099c5d4..a5007244 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_zero_int_int.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/snippets/MatrixBase_zero_int_int.cpp @@ -1 +1 @@ -cout << MatrixXi::Zero(2,3) << endl; +cout << MatrixXi::Zero(2, 3) << endl; diff --git a/filmulator-gui/core/nlmeans/eigen/doc/snippets/Matrix_Map_stride.cpp b/filmulator-gui/core/nlmeans/eigen/doc/snippets/Matrix_Map_stride.cpp index ae42a127..5125b54f 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/snippets/Matrix_Map_stride.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/snippets/Matrix_Map_stride.cpp @@ -1,7 +1,4 @@ Matrix4i A; -A << 1, 2, 3, 4, - 5, 6, 7, 8, - 9, 10, 11, 12, - 13, 14, 15, 16; +A << 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15, 16; -std::cout << Matrix2i::Map(&A(1,1),Stride<8,2>()) << std::endl; +std::cout << Matrix2i::Map(&A(1, 1), Stride<8, 2>()) << std::endl; diff --git a/filmulator-gui/core/nlmeans/eigen/doc/snippets/Matrix_resize_NoChange_int.cpp b/filmulator-gui/core/nlmeans/eigen/doc/snippets/Matrix_resize_NoChange_int.cpp index acdf18c4..c86ee30b 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/snippets/Matrix_resize_NoChange_int.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/snippets/Matrix_resize_NoChange_int.cpp @@ -1,3 +1,3 @@ -MatrixXd m(3,4); +MatrixXd m(3, 4); m.resize(NoChange, 5); cout << "m: " << m.rows() << " rows, " << m.cols() << " cols" << endl; diff --git a/filmulator-gui/core/nlmeans/eigen/doc/snippets/Matrix_resize_int.cpp b/filmulator-gui/core/nlmeans/eigen/doc/snippets/Matrix_resize_int.cpp index 044c7898..e20883a7 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/snippets/Matrix_resize_int.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/snippets/Matrix_resize_int.cpp @@ -1,6 +1,6 @@ VectorXd v(10); v.resize(3); RowVector3d w; -w.resize(3); // this is legal, but has no effect +w.resize(3);// this is legal, but has no effect cout << "v: " << v.rows() << " rows, " << v.cols() << " cols" << endl; cout << "w: " << w.rows() << " rows, " << w.cols() << " cols" << endl; diff --git a/filmulator-gui/core/nlmeans/eigen/doc/snippets/Matrix_resize_int_NoChange.cpp b/filmulator-gui/core/nlmeans/eigen/doc/snippets/Matrix_resize_int_NoChange.cpp index 5c37c906..e30763dd 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/snippets/Matrix_resize_int_NoChange.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/snippets/Matrix_resize_int_NoChange.cpp @@ -1,3 +1,3 @@ -MatrixXd m(3,4); +MatrixXd m(3, 4); m.resize(5, NoChange); cout << "m: " << m.rows() << " rows, " << m.cols() << " cols" << endl; diff --git a/filmulator-gui/core/nlmeans/eigen/doc/snippets/Matrix_resize_int_int.cpp b/filmulator-gui/core/nlmeans/eigen/doc/snippets/Matrix_resize_int_int.cpp index bfd47415..5453dbd1 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/snippets/Matrix_resize_int_int.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/snippets/Matrix_resize_int_int.cpp @@ -1,9 +1,9 @@ -MatrixXd m(2,3); -m << 1,2,3,4,5,6; +MatrixXd m(2, 3); +m << 1, 2, 3, 4, 5, 6; cout << "here's the 2x3 matrix m:" << endl << m << endl; cout << "let's resize m to 3x2. This is a conservative resizing because 2*3==3*2." << endl; -m.resize(3,2); +m.resize(3, 2); cout << "here's the 3x2 matrix m:" << endl << m << endl; cout << "now let's resize m to size 2x2. This is NOT a conservative resizing, so it becomes uninitialized:" << endl; -m.resize(2,2); +m.resize(2, 2); cout << m << endl; diff --git a/filmulator-gui/core/nlmeans/eigen/doc/snippets/PartialPivLU_solve.cpp b/filmulator-gui/core/nlmeans/eigen/doc/snippets/PartialPivLU_solve.cpp index fa3570ab..ebfa56d9 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/snippets/PartialPivLU_solve.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/snippets/PartialPivLU_solve.cpp @@ -1,7 +1,7 @@ -MatrixXd A = MatrixXd::Random(3,3); -MatrixXd B = MatrixXd::Random(3,2); +MatrixXd A = MatrixXd::Random(3, 3); +MatrixXd B = MatrixXd::Random(3, 2); cout << "Here is the invertible matrix A:" << endl << A << endl; cout << "Here is the matrix B:" << endl << B << endl; MatrixXd X = A.lu().solve(B); cout << "Here is the (unique) solution X to the equation AX=B:" << endl << X << endl; -cout << "Relative error: " << (A*X-B).norm() / B.norm() << endl; +cout << "Relative error: " << (A * X - B).norm() / B.norm() << endl; diff --git a/filmulator-gui/core/nlmeans/eigen/doc/snippets/RealQZ_compute.cpp b/filmulator-gui/core/nlmeans/eigen/doc/snippets/RealQZ_compute.cpp index a18da42e..e600a7f5 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/snippets/RealQZ_compute.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/snippets/RealQZ_compute.cpp @@ -1,7 +1,7 @@ -MatrixXf A = MatrixXf::Random(4,4); -MatrixXf B = MatrixXf::Random(4,4); -RealQZ qz(4); // preallocate space for 4x4 matrices -qz.compute(A,B); // A = Q S Z, B = Q T Z +MatrixXf A = MatrixXf::Random(4, 4); +MatrixXf B = MatrixXf::Random(4, 4); +RealQZ qz(4);// preallocate space for 4x4 matrices +qz.compute(A, B);// A = Q S Z, B = Q T Z // print original matrices and result of decomposition cout << "A:\n" << A << "\n" << "B:\n" << B << "\n"; @@ -10,8 +10,7 @@ cout << "Q:\n" << qz.matrixQ() << "\n" << "Z:\n" << qz.matrixZ() << "\n"; // verify precision cout << "\nErrors:" - << "\n|A-QSZ|: " << (A-qz.matrixQ()*qz.matrixS()*qz.matrixZ()).norm() - << ", |B-QTZ|: " << (B-qz.matrixQ()*qz.matrixT()*qz.matrixZ()).norm() - << "\n|QQ* - I|: " << (qz.matrixQ()*qz.matrixQ().adjoint() - MatrixXf::Identity(4,4)).norm() - << ", |ZZ* - I|: " << (qz.matrixZ()*qz.matrixZ().adjoint() - MatrixXf::Identity(4,4)).norm() - << "\n"; + << "\n|A-QSZ|: " << (A - qz.matrixQ() * qz.matrixS() * qz.matrixZ()).norm() + << ", |B-QTZ|: " << (B - qz.matrixQ() * qz.matrixT() * qz.matrixZ()).norm() + << "\n|QQ* - I|: " << (qz.matrixQ() * qz.matrixQ().adjoint() - MatrixXf::Identity(4, 4)).norm() + << ", |ZZ* - I|: " << (qz.matrixZ() * qz.matrixZ().adjoint() - MatrixXf::Identity(4, 4)).norm() << "\n"; diff --git a/filmulator-gui/core/nlmeans/eigen/doc/snippets/RealSchur_RealSchur_MatrixType.cpp b/filmulator-gui/core/nlmeans/eigen/doc/snippets/RealSchur_RealSchur_MatrixType.cpp index a5530dcc..485b0eea 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/snippets/RealSchur_RealSchur_MatrixType.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/snippets/RealSchur_RealSchur_MatrixType.cpp @@ -1,4 +1,4 @@ -MatrixXd A = MatrixXd::Random(6,6); +MatrixXd A = MatrixXd::Random(6, 6); cout << "Here is a random 6x6 matrix, A:" << endl << A << endl << endl; RealSchur schur(A); diff --git a/filmulator-gui/core/nlmeans/eigen/doc/snippets/RealSchur_compute.cpp b/filmulator-gui/core/nlmeans/eigen/doc/snippets/RealSchur_compute.cpp index 20c2611b..78c71c83 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/snippets/RealSchur_compute.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/snippets/RealSchur_compute.cpp @@ -1,4 +1,4 @@ -MatrixXf A = MatrixXf::Random(4,4); +MatrixXf A = MatrixXf::Random(4, 4); RealSchur schur(4); schur.compute(A, /* computeU = */ false); cout << "The matrix T in the decomposition of A is:" << endl << schur.matrixT() << endl; diff --git a/filmulator-gui/core/nlmeans/eigen/doc/snippets/SelfAdjointEigenSolver_SelfAdjointEigenSolver.cpp b/filmulator-gui/core/nlmeans/eigen/doc/snippets/SelfAdjointEigenSolver_SelfAdjointEigenSolver.cpp index 73a7f625..2ae3bf6c 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/snippets/SelfAdjointEigenSolver_SelfAdjointEigenSolver.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/snippets/SelfAdjointEigenSolver_SelfAdjointEigenSolver.cpp @@ -1,7 +1,7 @@ SelfAdjointEigenSolver es; -Matrix4f X = Matrix4f::Random(4,4); +Matrix4f X = Matrix4f::Random(4, 4); Matrix4f A = X + X.transpose(); es.compute(A); cout << "The eigenvalues of A are: " << es.eigenvalues().transpose() << endl; -es.compute(A + Matrix4f::Identity(4,4)); // re-use es to compute eigenvalues of A+I +es.compute(A + Matrix4f::Identity(4, 4));// re-use es to compute eigenvalues of A+I cout << "The eigenvalues of A+I are: " << es.eigenvalues().transpose() << endl; diff --git a/filmulator-gui/core/nlmeans/eigen/doc/snippets/SelfAdjointEigenSolver_SelfAdjointEigenSolver_MatrixType.cpp b/filmulator-gui/core/nlmeans/eigen/doc/snippets/SelfAdjointEigenSolver_SelfAdjointEigenSolver_MatrixType.cpp index 3599b17a..0ed9092c 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/snippets/SelfAdjointEigenSolver_SelfAdjointEigenSolver_MatrixType.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/snippets/SelfAdjointEigenSolver_SelfAdjointEigenSolver_MatrixType.cpp @@ -1,4 +1,4 @@ -MatrixXd X = MatrixXd::Random(5,5); +MatrixXd X = MatrixXd::Random(5, 5); MatrixXd A = X + X.transpose(); cout << "Here is a random symmetric 5x5 matrix, A:" << endl << A << endl << endl; diff --git a/filmulator-gui/core/nlmeans/eigen/doc/snippets/SelfAdjointEigenSolver_SelfAdjointEigenSolver_MatrixType2.cpp b/filmulator-gui/core/nlmeans/eigen/doc/snippets/SelfAdjointEigenSolver_SelfAdjointEigenSolver_MatrixType2.cpp index bbb821e0..d99a3fc1 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/snippets/SelfAdjointEigenSolver_SelfAdjointEigenSolver_MatrixType2.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/snippets/SelfAdjointEigenSolver_SelfAdjointEigenSolver_MatrixType2.cpp @@ -1,11 +1,11 @@ -MatrixXd X = MatrixXd::Random(5,5); +MatrixXd X = MatrixXd::Random(5, 5); MatrixXd A = X + X.transpose(); cout << "Here is a random symmetric matrix, A:" << endl << A << endl; -X = MatrixXd::Random(5,5); +X = MatrixXd::Random(5, 5); MatrixXd B = X * X.transpose(); cout << "and a random postive-definite matrix, B:" << endl << B << endl << endl; -GeneralizedSelfAdjointEigenSolver es(A,B); +GeneralizedSelfAdjointEigenSolver es(A, B); cout << "The eigenvalues of the pencil (A,B) are:" << endl << es.eigenvalues() << endl; cout << "The matrix of eigenvectors, V, is:" << endl << es.eigenvectors() << endl << endl; diff --git a/filmulator-gui/core/nlmeans/eigen/doc/snippets/SelfAdjointEigenSolver_compute_MatrixType.cpp b/filmulator-gui/core/nlmeans/eigen/doc/snippets/SelfAdjointEigenSolver_compute_MatrixType.cpp index 2975cc3f..aff7c7e9 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/snippets/SelfAdjointEigenSolver_compute_MatrixType.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/snippets/SelfAdjointEigenSolver_compute_MatrixType.cpp @@ -1,7 +1,7 @@ SelfAdjointEigenSolver es(4); -MatrixXf X = MatrixXf::Random(4,4); +MatrixXf X = MatrixXf::Random(4, 4); MatrixXf A = X + X.transpose(); es.compute(A); cout << "The eigenvalues of A are: " << es.eigenvalues().transpose() << endl; -es.compute(A + MatrixXf::Identity(4,4)); // re-use es to compute eigenvalues of A+I +es.compute(A + MatrixXf::Identity(4, 4));// re-use es to compute eigenvalues of A+I cout << "The eigenvalues of A+I are: " << es.eigenvalues().transpose() << endl; diff --git a/filmulator-gui/core/nlmeans/eigen/doc/snippets/SelfAdjointEigenSolver_compute_MatrixType2.cpp b/filmulator-gui/core/nlmeans/eigen/doc/snippets/SelfAdjointEigenSolver_compute_MatrixType2.cpp index 07c92a1e..88d44c62 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/snippets/SelfAdjointEigenSolver_compute_MatrixType2.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/snippets/SelfAdjointEigenSolver_compute_MatrixType2.cpp @@ -1,9 +1,9 @@ -MatrixXd X = MatrixXd::Random(5,5); +MatrixXd X = MatrixXd::Random(5, 5); MatrixXd A = X * X.transpose(); -X = MatrixXd::Random(5,5); +X = MatrixXd::Random(5, 5); MatrixXd B = X * X.transpose(); -GeneralizedSelfAdjointEigenSolver es(A,B,EigenvaluesOnly); +GeneralizedSelfAdjointEigenSolver es(A, B, EigenvaluesOnly); cout << "The eigenvalues of the pencil (A,B) are:" << endl << es.eigenvalues() << endl; -es.compute(B,A,false); +es.compute(B, A, false); cout << "The eigenvalues of the pencil (B,A) are:" << endl << es.eigenvalues() << endl; diff --git a/filmulator-gui/core/nlmeans/eigen/doc/snippets/SelfAdjointEigenSolver_eigenvalues.cpp b/filmulator-gui/core/nlmeans/eigen/doc/snippets/SelfAdjointEigenSolver_eigenvalues.cpp index 0ff33c68..d80f90c1 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/snippets/SelfAdjointEigenSolver_eigenvalues.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/snippets/SelfAdjointEigenSolver_eigenvalues.cpp @@ -1,4 +1,3 @@ -MatrixXd ones = MatrixXd::Ones(3,3); +MatrixXd ones = MatrixXd::Ones(3, 3); SelfAdjointEigenSolver es(ones); -cout << "The eigenvalues of the 3x3 matrix of ones are:" - << endl << es.eigenvalues() << endl; +cout << "The eigenvalues of the 3x3 matrix of ones are:" << endl << es.eigenvalues() << endl; diff --git a/filmulator-gui/core/nlmeans/eigen/doc/snippets/SelfAdjointEigenSolver_eigenvectors.cpp b/filmulator-gui/core/nlmeans/eigen/doc/snippets/SelfAdjointEigenSolver_eigenvectors.cpp index cfc8b0d5..7eae8973 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/snippets/SelfAdjointEigenSolver_eigenvectors.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/snippets/SelfAdjointEigenSolver_eigenvectors.cpp @@ -1,4 +1,3 @@ -MatrixXd ones = MatrixXd::Ones(3,3); +MatrixXd ones = MatrixXd::Ones(3, 3); SelfAdjointEigenSolver es(ones); -cout << "The first eigenvector of the 3x3 matrix of ones is:" - << endl << es.eigenvectors().col(1) << endl; +cout << "The first eigenvector of the 3x3 matrix of ones is:" << endl << es.eigenvectors().col(1) << endl; diff --git a/filmulator-gui/core/nlmeans/eigen/doc/snippets/SelfAdjointEigenSolver_operatorInverseSqrt.cpp b/filmulator-gui/core/nlmeans/eigen/doc/snippets/SelfAdjointEigenSolver_operatorInverseSqrt.cpp index 114c65fb..50e83f33 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/snippets/SelfAdjointEigenSolver_operatorInverseSqrt.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/snippets/SelfAdjointEigenSolver_operatorInverseSqrt.cpp @@ -1,4 +1,4 @@ -MatrixXd X = MatrixXd::Random(4,4); +MatrixXd X = MatrixXd::Random(4, 4); MatrixXd A = X * X.transpose(); cout << "Here is a random positive-definite matrix, A:" << endl << A << endl << endl; diff --git a/filmulator-gui/core/nlmeans/eigen/doc/snippets/SelfAdjointEigenSolver_operatorSqrt.cpp b/filmulator-gui/core/nlmeans/eigen/doc/snippets/SelfAdjointEigenSolver_operatorSqrt.cpp index eeacca74..bced567e 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/snippets/SelfAdjointEigenSolver_operatorSqrt.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/snippets/SelfAdjointEigenSolver_operatorSqrt.cpp @@ -1,8 +1,8 @@ -MatrixXd X = MatrixXd::Random(4,4); +MatrixXd X = MatrixXd::Random(4, 4); MatrixXd A = X * X.transpose(); cout << "Here is a random positive-definite matrix, A:" << endl << A << endl << endl; SelfAdjointEigenSolver es(A); MatrixXd sqrtA = es.operatorSqrt(); cout << "The square root of A is: " << endl << sqrtA << endl; -cout << "If we square this, we get: " << endl << sqrtA*sqrtA << endl; +cout << "If we square this, we get: " << endl << sqrtA * sqrtA << endl; diff --git a/filmulator-gui/core/nlmeans/eigen/doc/snippets/SelfAdjointView_eigenvalues.cpp b/filmulator-gui/core/nlmeans/eigen/doc/snippets/SelfAdjointView_eigenvalues.cpp index be198677..8cef6d8a 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/snippets/SelfAdjointView_eigenvalues.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/snippets/SelfAdjointView_eigenvalues.cpp @@ -1,3 +1,3 @@ -MatrixXd ones = MatrixXd::Ones(3,3); +MatrixXd ones = MatrixXd::Ones(3, 3); VectorXd eivals = ones.selfadjointView().eigenvalues(); cout << "The eigenvalues of the 3x3 matrix of ones are:" << endl << eivals << endl; diff --git a/filmulator-gui/core/nlmeans/eigen/doc/snippets/SelfAdjointView_operatorNorm.cpp b/filmulator-gui/core/nlmeans/eigen/doc/snippets/SelfAdjointView_operatorNorm.cpp index f380f559..c229fafe 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/snippets/SelfAdjointView_operatorNorm.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/snippets/SelfAdjointView_operatorNorm.cpp @@ -1,3 +1,2 @@ -MatrixXd ones = MatrixXd::Ones(3,3); -cout << "The operator norm of the 3x3 matrix of ones is " - << ones.selfadjointView().operatorNorm() << endl; +MatrixXd ones = MatrixXd::Ones(3, 3); +cout << "The operator norm of the 3x3 matrix of ones is " << ones.selfadjointView().operatorNorm() << endl; diff --git a/filmulator-gui/core/nlmeans/eigen/doc/snippets/SparseMatrix_coeffs.cpp b/filmulator-gui/core/nlmeans/eigen/doc/snippets/SparseMatrix_coeffs.cpp index f71a69b0..266e32d9 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/snippets/SparseMatrix_coeffs.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/snippets/SparseMatrix_coeffs.cpp @@ -1,7 +1,7 @@ -SparseMatrix A(3,3); -A.insert(1,2) = 0; -A.insert(0,1) = 1; -A.insert(2,0) = 2; +SparseMatrix A(3, 3); +A.insert(1, 2) = 0; +A.insert(0, 1) = 1; +A.insert(2, 0) = 2; A.makeCompressed(); cout << "The matrix A is:" << endl << MatrixXd(A) << endl; cout << "it has " << A.nonZeros() << " stored non zero coefficients that are: " << A.coeffs().transpose() << endl; diff --git a/filmulator-gui/core/nlmeans/eigen/doc/snippets/TopicAliasing_block.cpp b/filmulator-gui/core/nlmeans/eigen/doc/snippets/TopicAliasing_block.cpp index 03282f4f..bbb7c2a0 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/snippets/TopicAliasing_block.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/snippets/TopicAliasing_block.cpp @@ -1,7 +1,7 @@ -MatrixXi mat(3,3); -mat << 1, 2, 3, 4, 5, 6, 7, 8, 9; +MatrixXi mat(3, 3); +mat << 1, 2, 3, 4, 5, 6, 7, 8, 9; cout << "Here is the matrix mat:\n" << mat << endl; // This assignment shows the aliasing problem -mat.bottomRightCorner(2,2) = mat.topLeftCorner(2,2); +mat.bottomRightCorner(2, 2) = mat.topLeftCorner(2, 2); cout << "After the assignment, mat = \n" << mat << endl; diff --git a/filmulator-gui/core/nlmeans/eigen/doc/snippets/TopicAliasing_block_correct.cpp b/filmulator-gui/core/nlmeans/eigen/doc/snippets/TopicAliasing_block_correct.cpp index 6fee5801..6a3002e5 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/snippets/TopicAliasing_block_correct.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/snippets/TopicAliasing_block_correct.cpp @@ -1,7 +1,7 @@ -MatrixXi mat(3,3); -mat << 1, 2, 3, 4, 5, 6, 7, 8, 9; +MatrixXi mat(3, 3); +mat << 1, 2, 3, 4, 5, 6, 7, 8, 9; cout << "Here is the matrix mat:\n" << mat << endl; // The eval() solves the aliasing problem -mat.bottomRightCorner(2,2) = mat.topLeftCorner(2,2).eval(); +mat.bottomRightCorner(2, 2) = mat.topLeftCorner(2, 2).eval(); cout << "After the assignment, mat = \n" << mat << endl; diff --git a/filmulator-gui/core/nlmeans/eigen/doc/snippets/TopicAliasing_cwise.cpp b/filmulator-gui/core/nlmeans/eigen/doc/snippets/TopicAliasing_cwise.cpp index 7049f6c5..5ec2623b 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/snippets/TopicAliasing_cwise.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/snippets/TopicAliasing_cwise.cpp @@ -1,12 +1,12 @@ -MatrixXf mat(2,2); -mat << 1, 2, 4, 7; +MatrixXf mat(2, 2); +mat << 1, 2, 4, 7; cout << "Here is the matrix mat:\n" << mat << endl << endl; mat = 2 * mat; cout << "After 'mat = 2 * mat', mat = \n" << mat << endl << endl; -mat = mat - MatrixXf::Identity(2,2); +mat = mat - MatrixXf::Identity(2, 2); cout << "After the subtraction, it becomes\n" << mat << endl << endl; @@ -15,6 +15,6 @@ arr = arr.square(); cout << "After squaring, it becomes\n" << arr << endl << endl; // Combining all operations in one statement: -mat << 1, 2, 4, 7; -mat = (2 * mat - MatrixXf::Identity(2,2)).array().square(); +mat << 1, 2, 4, 7; +mat = (2 * mat - MatrixXf::Identity(2, 2)).array().square(); cout << "Doing everything at once yields\n" << mat << endl << endl; diff --git a/filmulator-gui/core/nlmeans/eigen/doc/snippets/TopicAliasing_mult1.cpp b/filmulator-gui/core/nlmeans/eigen/doc/snippets/TopicAliasing_mult1.cpp index cd7e9004..4ae2a471 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/snippets/TopicAliasing_mult1.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/snippets/TopicAliasing_mult1.cpp @@ -1,4 +1,4 @@ -MatrixXf matA(2,2); -matA << 2, 0, 0, 2; +MatrixXf matA(2, 2); +matA << 2, 0, 0, 2; matA = matA * matA; cout << matA; diff --git a/filmulator-gui/core/nlmeans/eigen/doc/snippets/TopicAliasing_mult2.cpp b/filmulator-gui/core/nlmeans/eigen/doc/snippets/TopicAliasing_mult2.cpp index a3ff5685..52c99875 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/snippets/TopicAliasing_mult2.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/snippets/TopicAliasing_mult2.cpp @@ -1,5 +1,5 @@ -MatrixXf matA(2,2), matB(2,2); -matA << 2, 0, 0, 2; +MatrixXf matA(2, 2), matB(2, 2); +matA << 2, 0, 0, 2; // Simple but not quite as efficient matB = matA * matA; diff --git a/filmulator-gui/core/nlmeans/eigen/doc/snippets/TopicAliasing_mult3.cpp b/filmulator-gui/core/nlmeans/eigen/doc/snippets/TopicAliasing_mult3.cpp index 1d12a6c6..b2bed6f3 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/snippets/TopicAliasing_mult3.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/snippets/TopicAliasing_mult3.cpp @@ -1,4 +1,4 @@ -MatrixXf matA(2,2); -matA << 2, 0, 0, 2; +MatrixXf matA(2, 2); +matA << 2, 0, 0, 2; matA.noalias() = matA * matA; cout << matA; diff --git a/filmulator-gui/core/nlmeans/eigen/doc/snippets/TopicAliasing_mult4.cpp b/filmulator-gui/core/nlmeans/eigen/doc/snippets/TopicAliasing_mult4.cpp index 8a8992f6..a584c426 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/snippets/TopicAliasing_mult4.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/snippets/TopicAliasing_mult4.cpp @@ -1,5 +1,5 @@ -MatrixXf A(2,2), B(3,2); -B << 2, 0, 0, 3, 1, 1; +MatrixXf A(2, 2), B(3, 2); +B << 2, 0, 0, 3, 1, 1; A << 2, 0, 0, -2; A = (B * A).cwiseAbs(); cout << A; \ No newline at end of file diff --git a/filmulator-gui/core/nlmeans/eigen/doc/snippets/TopicAliasing_mult5.cpp b/filmulator-gui/core/nlmeans/eigen/doc/snippets/TopicAliasing_mult5.cpp index 1a36defd..79a94c86 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/snippets/TopicAliasing_mult5.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/snippets/TopicAliasing_mult5.cpp @@ -1,5 +1,5 @@ -MatrixXf A(2,2), B(3,2); -B << 2, 0, 0, 3, 1, 1; +MatrixXf A(2, 2), B(3, 2); +B << 2, 0, 0, 3, 1, 1; A << 2, 0, 0, -2; A = (B * A).eval().cwiseAbs(); cout << A; diff --git a/filmulator-gui/core/nlmeans/eigen/doc/snippets/TopicStorageOrders_example.cpp b/filmulator-gui/core/nlmeans/eigen/doc/snippets/TopicStorageOrders_example.cpp index 0623ef0c..64e2123d 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/snippets/TopicStorageOrders_example.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/snippets/TopicStorageOrders_example.cpp @@ -1,18 +1,13 @@ Matrix Acolmajor; -Acolmajor << 8, 2, 2, 9, - 9, 1, 4, 4, - 3, 5, 4, 5; +Acolmajor << 8, 2, 2, 9, 9, 1, 4, 4, 3, 5, 4, 5; cout << "The matrix A:" << endl; -cout << Acolmajor << endl << endl; +cout << Acolmajor << endl << endl; cout << "In memory (column-major):" << endl; -for (int i = 0; i < Acolmajor.size(); i++) - cout << *(Acolmajor.data() + i) << " "; +for (int i = 0; i < Acolmajor.size(); i++) cout << *(Acolmajor.data() + i) << " "; cout << endl << endl; Matrix Arowmajor = Acolmajor; cout << "In memory (row-major):" << endl; -for (int i = 0; i < Arowmajor.size(); i++) - cout << *(Arowmajor.data() + i) << " "; +for (int i = 0; i < Arowmajor.size(); i++) cout << *(Arowmajor.data() + i) << " "; cout << endl; - diff --git a/filmulator-gui/core/nlmeans/eigen/doc/snippets/Triangular_solve.cpp b/filmulator-gui/core/nlmeans/eigen/doc/snippets/Triangular_solve.cpp index 54844246..2da806b0 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/snippets/Triangular_solve.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/snippets/Triangular_solve.cpp @@ -7,5 +7,4 @@ cout << "Here is the matrix n:\n" << n << endl; cout << "And now here is m.inverse()*n, taking advantage of the fact that" " m is upper-triangular:\n" << m.triangularView().solve(n) << endl; -cout << "And this is n*m.inverse():\n" - << m.triangularView().solve(n); +cout << "And this is n*m.inverse():\n" << m.triangularView().solve(n); diff --git a/filmulator-gui/core/nlmeans/eigen/doc/snippets/Tridiagonalization_Tridiagonalization_MatrixType.cpp b/filmulator-gui/core/nlmeans/eigen/doc/snippets/Tridiagonalization_Tridiagonalization_MatrixType.cpp index a2601243..087b1ec3 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/snippets/Tridiagonalization_Tridiagonalization_MatrixType.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/snippets/Tridiagonalization_Tridiagonalization_MatrixType.cpp @@ -1,4 +1,4 @@ -MatrixXd X = MatrixXd::Random(5,5); +MatrixXd X = MatrixXd::Random(5, 5); MatrixXd A = X + X.transpose(); cout << "Here is a random symmetric 5x5 matrix:" << endl << A << endl << endl; Tridiagonalization triOfA(A); diff --git a/filmulator-gui/core/nlmeans/eigen/doc/snippets/Tridiagonalization_compute.cpp b/filmulator-gui/core/nlmeans/eigen/doc/snippets/Tridiagonalization_compute.cpp index 0062a99e..077c1db4 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/snippets/Tridiagonalization_compute.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/snippets/Tridiagonalization_compute.cpp @@ -1,9 +1,9 @@ Tridiagonalization tri; -MatrixXf X = MatrixXf::Random(4,4); +MatrixXf X = MatrixXf::Random(4, 4); MatrixXf A = X + X.transpose(); tri.compute(A); cout << "The matrix T in the tridiagonal decomposition of A is: " << endl; cout << tri.matrixT() << endl; -tri.compute(2*A); // re-use tri to compute eigenvalues of 2A +tri.compute(2 * A);// re-use tri to compute eigenvalues of 2A cout << "The matrix T in the tridiagonal decomposition of 2A is: " << endl; cout << tri.matrixT() << endl; diff --git a/filmulator-gui/core/nlmeans/eigen/doc/snippets/Tridiagonalization_decomposeInPlace.cpp b/filmulator-gui/core/nlmeans/eigen/doc/snippets/Tridiagonalization_decomposeInPlace.cpp index 93dcfca1..16be332c 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/snippets/Tridiagonalization_decomposeInPlace.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/snippets/Tridiagonalization_decomposeInPlace.cpp @@ -1,4 +1,4 @@ -MatrixXd X = MatrixXd::Random(5,5); +MatrixXd X = MatrixXd::Random(5, 5); MatrixXd A = X + X.transpose(); cout << "Here is a random symmetric 5x5 matrix:" << endl << A << endl << endl; diff --git a/filmulator-gui/core/nlmeans/eigen/doc/snippets/Tridiagonalization_diagonal.cpp b/filmulator-gui/core/nlmeans/eigen/doc/snippets/Tridiagonalization_diagonal.cpp index 6eec8216..18edc1e7 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/snippets/Tridiagonalization_diagonal.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/snippets/Tridiagonalization_diagonal.cpp @@ -1,4 +1,4 @@ -MatrixXcd X = MatrixXcd::Random(4,4); +MatrixXcd X = MatrixXcd::Random(4, 4); MatrixXcd A = X + X.adjoint(); cout << "Here is a random self-adjoint 4x4 matrix:" << endl << A << endl << endl; @@ -8,6 +8,6 @@ cout << "The tridiagonal matrix T is:" << endl << T << endl << endl; cout << "We can also extract the diagonals of T directly ..." << endl; VectorXd diag = triOfA.diagonal(); -cout << "The diagonal is:" << endl << diag << endl; +cout << "The diagonal is:" << endl << diag << endl; VectorXd subdiag = triOfA.subDiagonal(); cout << "The subdiagonal is:" << endl << subdiag << endl; diff --git a/filmulator-gui/core/nlmeans/eigen/doc/snippets/Tridiagonalization_householderCoefficients.cpp b/filmulator-gui/core/nlmeans/eigen/doc/snippets/Tridiagonalization_householderCoefficients.cpp index e5d87288..f880f1d2 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/snippets/Tridiagonalization_householderCoefficients.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/snippets/Tridiagonalization_householderCoefficients.cpp @@ -1,4 +1,4 @@ -Matrix4d X = Matrix4d::Random(4,4); +Matrix4d X = Matrix4d::Random(4, 4); Matrix4d A = X + X.transpose(); cout << "Here is a random symmetric 4x4 matrix:" << endl << A << endl; Tridiagonalization triOfA(A); diff --git a/filmulator-gui/core/nlmeans/eigen/doc/snippets/Tridiagonalization_packedMatrix.cpp b/filmulator-gui/core/nlmeans/eigen/doc/snippets/Tridiagonalization_packedMatrix.cpp index 0f55d0c2..3b186a8c 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/snippets/Tridiagonalization_packedMatrix.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/snippets/Tridiagonalization_packedMatrix.cpp @@ -1,8 +1,7 @@ -Matrix4d X = Matrix4d::Random(4,4); +Matrix4d X = Matrix4d::Random(4, 4); Matrix4d A = X + X.transpose(); cout << "Here is a random symmetric 4x4 matrix:" << endl << A << endl; Tridiagonalization triOfA(A); Matrix4d pm = triOfA.packedMatrix(); cout << "The packed matrix M is:" << endl << pm << endl; -cout << "The diagonal and subdiagonal corresponds to the matrix T, which is:" - << endl << triOfA.matrixT() << endl; +cout << "The diagonal and subdiagonal corresponds to the matrix T, which is:" << endl << triOfA.matrixT() << endl; diff --git a/filmulator-gui/core/nlmeans/eigen/doc/snippets/Tutorial_AdvancedInitialization_Block.cpp b/filmulator-gui/core/nlmeans/eigen/doc/snippets/Tutorial_AdvancedInitialization_Block.cpp index 96e40acf..752c522a 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/snippets/Tutorial_AdvancedInitialization_Block.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/snippets/Tutorial_AdvancedInitialization_Block.cpp @@ -1,5 +1,5 @@ MatrixXf matA(2, 2); matA << 1, 2, 3, 4; MatrixXf matB(4, 4); -matB << matA, matA/10, matA/10, matA; +matB << matA, matA / 10, matA / 10, matA; std::cout << matB << std::endl; diff --git a/filmulator-gui/core/nlmeans/eigen/doc/snippets/Tutorial_AdvancedInitialization_CommaTemporary.cpp b/filmulator-gui/core/nlmeans/eigen/doc/snippets/Tutorial_AdvancedInitialization_CommaTemporary.cpp index 50cff4cb..8af3d747 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/snippets/Tutorial_AdvancedInitialization_CommaTemporary.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/snippets/Tutorial_AdvancedInitialization_CommaTemporary.cpp @@ -1,4 +1,4 @@ MatrixXf mat = MatrixXf::Random(2, 3); std::cout << mat << std::endl << std::endl; -mat = (MatrixXf(2,2) << 0, 1, 1, 0).finished() * mat; +mat = (MatrixXf(2, 2) << 0, 1, 1, 0).finished() * mat; std::cout << mat << std::endl; diff --git a/filmulator-gui/core/nlmeans/eigen/doc/snippets/Tutorial_AdvancedInitialization_ThreeWays.cpp b/filmulator-gui/core/nlmeans/eigen/doc/snippets/Tutorial_AdvancedInitialization_ThreeWays.cpp index cb745765..65284e15 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/snippets/Tutorial_AdvancedInitialization_ThreeWays.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/snippets/Tutorial_AdvancedInitialization_ThreeWays.cpp @@ -1,20 +1,19 @@ const int size = 6; MatrixXd mat1(size, size); -mat1.topLeftCorner(size/2, size/2) = MatrixXd::Zero(size/2, size/2); -mat1.topRightCorner(size/2, size/2) = MatrixXd::Identity(size/2, size/2); -mat1.bottomLeftCorner(size/2, size/2) = MatrixXd::Identity(size/2, size/2); -mat1.bottomRightCorner(size/2, size/2) = MatrixXd::Zero(size/2, size/2); +mat1.topLeftCorner(size / 2, size / 2) = MatrixXd::Zero(size / 2, size / 2); +mat1.topRightCorner(size / 2, size / 2) = MatrixXd::Identity(size / 2, size / 2); +mat1.bottomLeftCorner(size / 2, size / 2) = MatrixXd::Identity(size / 2, size / 2); +mat1.bottomRightCorner(size / 2, size / 2) = MatrixXd::Zero(size / 2, size / 2); std::cout << mat1 << std::endl << std::endl; MatrixXd mat2(size, size); -mat2.topLeftCorner(size/2, size/2).setZero(); -mat2.topRightCorner(size/2, size/2).setIdentity(); -mat2.bottomLeftCorner(size/2, size/2).setIdentity(); -mat2.bottomRightCorner(size/2, size/2).setZero(); +mat2.topLeftCorner(size / 2, size / 2).setZero(); +mat2.topRightCorner(size / 2, size / 2).setIdentity(); +mat2.bottomLeftCorner(size / 2, size / 2).setIdentity(); +mat2.bottomRightCorner(size / 2, size / 2).setZero(); std::cout << mat2 << std::endl << std::endl; MatrixXd mat3(size, size); -mat3 << MatrixXd::Zero(size/2, size/2), MatrixXd::Identity(size/2, size/2), - MatrixXd::Identity(size/2, size/2), MatrixXd::Zero(size/2, size/2); +mat3 << MatrixXd::Zero(size / 2, size / 2), + MatrixXd::Identity(size / 2, size / 2), MatrixXd::Identity(size / 2, size / 2), MatrixXd::Zero(size / 2, size / 2); std::cout << mat3 << std::endl; - diff --git a/filmulator-gui/core/nlmeans/eigen/doc/snippets/Tutorial_Map_rowmajor.cpp b/filmulator-gui/core/nlmeans/eigen/doc/snippets/Tutorial_Map_rowmajor.cpp index fd45ace0..133591d5 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/snippets/Tutorial_Map_rowmajor.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/snippets/Tutorial_Map_rowmajor.cpp @@ -1,7 +1,5 @@ int array[8]; -for(int i = 0; i < 8; ++i) array[i] = i; -cout << "Column-major:\n" << Map >(array) << endl; -cout << "Row-major:\n" << Map >(array) << endl; -cout << "Row-major using stride:\n" << - Map, Unaligned, Stride<1,4> >(array) << endl; - +for (int i = 0; i < 8; ++i) array[i] = i; +cout << "Column-major:\n" << Map>(array) << endl; +cout << "Row-major:\n" << Map>(array) << endl; +cout << "Row-major using stride:\n" << Map, Unaligned, Stride<1, 4>>(array) << endl; diff --git a/filmulator-gui/core/nlmeans/eigen/doc/snippets/Tutorial_Map_using.cpp b/filmulator-gui/core/nlmeans/eigen/doc/snippets/Tutorial_Map_using.cpp index e5e499f1..b16de19d 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/snippets/Tutorial_Map_using.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/snippets/Tutorial_Map_using.cpp @@ -1,21 +1,20 @@ -typedef Matrix MatrixType; +typedef Matrix MatrixType; typedef Map MapType; -typedef Map MapTypeConst; // a read-only map +typedef Map MapTypeConst;// a read-only map const int n_dims = 5; - + MatrixType m1(n_dims), m2(n_dims); m1.setRandom(); m2.setRandom(); -float *p = &m2(0); // get the address storing the data for m2 -MapType m2map(p,m2.size()); // m2map shares data with m2 -MapTypeConst m2mapconst(p,m2.size()); // a read-only accessor for m2 +float *p = &m2(0);// get the address storing the data for m2 +MapType m2map(p, m2.size());// m2map shares data with m2 +MapTypeConst m2mapconst(p, m2.size());// a read-only accessor for m2 cout << "m1: " << m1 << endl; cout << "m2: " << m2 << endl; -cout << "Squared euclidean distance: " << (m1-m2).squaredNorm() << endl; -cout << "Squared euclidean distance, using map: " << - (m1-m2map).squaredNorm() << endl; -m2map(3) = 7; // this will change m2, since they share the same array +cout << "Squared euclidean distance: " << (m1 - m2).squaredNorm() << endl; +cout << "Squared euclidean distance, using map: " << (m1 - m2map).squaredNorm() << endl; +m2map(3) = 7;// this will change m2, since they share the same array cout << "Updated m2: " << m2 << endl; cout << "m2 coefficient 2, constant accessor: " << m2mapconst(2) << endl; -/* m2mapconst(2) = 5; */ // this yields a compile-time error +/* m2mapconst(2) = 5; */// this yields a compile-time error diff --git a/filmulator-gui/core/nlmeans/eigen/doc/snippets/Tutorial_ReshapeMat2Mat.cpp b/filmulator-gui/core/nlmeans/eigen/doc/snippets/Tutorial_ReshapeMat2Mat.cpp index f84d6e76..a384ab81 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/snippets/Tutorial_ReshapeMat2Mat.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/snippets/Tutorial_ReshapeMat2Mat.cpp @@ -1,6 +1,5 @@ -MatrixXf M1(2,6); // Column-major storage -M1 << 1, 2, 3, 4, 5, 6, - 7, 8, 9, 10, 11, 12; +MatrixXf M1(2, 6);// Column-major storage +M1 << 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12; -Map M2(M1.data(), 6,2); +Map M2(M1.data(), 6, 2); cout << "M2:" << endl << M2 << endl; \ No newline at end of file diff --git a/filmulator-gui/core/nlmeans/eigen/doc/snippets/Tutorial_ReshapeMat2Vec.cpp b/filmulator-gui/core/nlmeans/eigen/doc/snippets/Tutorial_ReshapeMat2Vec.cpp index 95bd4e0e..5f72c5ef 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/snippets/Tutorial_ReshapeMat2Vec.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/snippets/Tutorial_ReshapeMat2Vec.cpp @@ -1,11 +1,9 @@ -MatrixXf M1(3,3); // Column-major storage -M1 << 1, 2, 3, - 4, 5, 6, - 7, 8, 9; +MatrixXf M1(3, 3);// Column-major storage +M1 << 1, 2, 3, 4, 5, 6, 7, 8, 9; Map v1(M1.data(), M1.size()); cout << "v1:" << endl << v1 << endl; -Matrix M2(M1); +Matrix M2(M1); Map v2(M2.data(), M2.size()); cout << "v2:" << endl << v2 << endl; \ No newline at end of file diff --git a/filmulator-gui/core/nlmeans/eigen/doc/snippets/Tutorial_SlicingCol.cpp b/filmulator-gui/core/nlmeans/eigen/doc/snippets/Tutorial_SlicingCol.cpp index f667ff68..ff2e02c5 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/snippets/Tutorial_SlicingCol.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/snippets/Tutorial_SlicingCol.cpp @@ -1,11 +1,11 @@ -MatrixXf M1 = MatrixXf::Random(3,8); +MatrixXf M1 = MatrixXf::Random(3, 8); cout << "Column major input:" << endl << M1 << "\n"; -Map > M2(M1.data(), M1.rows(), (M1.cols()+2)/3, OuterStride<>(M1.outerStride()*3)); +Map> M2(M1.data(), M1.rows(), (M1.cols() + 2) / 3, OuterStride<>(M1.outerStride() * 3)); cout << "1 column over 3:" << endl << M2 << "\n"; -typedef Matrix RowMajorMatrixXf; +typedef Matrix RowMajorMatrixXf; RowMajorMatrixXf M3(M1); cout << "Row major input:" << endl << M3 << "\n"; -Map > M4(M3.data(), M3.rows(), (M3.cols()+2)/3, - Stride(M3.outerStride(),3)); +Map> + M4(M3.data(), M3.rows(), (M3.cols() + 2) / 3, Stride(M3.outerStride(), 3)); cout << "1 column over 3:" << endl << M4 << "\n"; \ No newline at end of file diff --git a/filmulator-gui/core/nlmeans/eigen/doc/snippets/Tutorial_SlicingVec.cpp b/filmulator-gui/core/nlmeans/eigen/doc/snippets/Tutorial_SlicingVec.cpp index 07e10bf6..abf1fe7b 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/snippets/Tutorial_SlicingVec.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/snippets/Tutorial_SlicingVec.cpp @@ -1,4 +1,4 @@ -RowVectorXf v = RowVectorXf::LinSpaced(20,0,19); +RowVectorXf v = RowVectorXf::LinSpaced(20, 0, 19); cout << "Input:" << endl << v << endl; -Map > v2(v.data(), v.size()/2); +Map> v2(v.data(), v.size() / 2); cout << "Even:" << v2 << endl; \ No newline at end of file diff --git a/filmulator-gui/core/nlmeans/eigen/doc/snippets/Tutorial_commainit_01.cpp b/filmulator-gui/core/nlmeans/eigen/doc/snippets/Tutorial_commainit_01.cpp index 47ba31dc..36728a78 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/snippets/Tutorial_commainit_01.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/snippets/Tutorial_commainit_01.cpp @@ -1,5 +1,3 @@ Matrix3f m; -m << 1, 2, 3, - 4, 5, 6, - 7, 8, 9; +m << 1, 2, 3, 4, 5, 6, 7, 8, 9; std::cout << m; diff --git a/filmulator-gui/core/nlmeans/eigen/doc/snippets/Tutorial_commainit_01b.cpp b/filmulator-gui/core/nlmeans/eigen/doc/snippets/Tutorial_commainit_01b.cpp index 2adb2e21..126277bc 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/snippets/Tutorial_commainit_01b.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/snippets/Tutorial_commainit_01b.cpp @@ -1,5 +1,5 @@ Matrix3f m; m.row(0) << 1, 2, 3; -m.block(1,0,2,2) << 4, 5, 7, 8; -m.col(2).tail(2) << 6, 9; +m.block(1, 0, 2, 2) << 4, 5, 7, 8; +m.col(2).tail(2) << 6, 9; std::cout << m; diff --git a/filmulator-gui/core/nlmeans/eigen/doc/snippets/Tutorial_commainit_02.cpp b/filmulator-gui/core/nlmeans/eigen/doc/snippets/Tutorial_commainit_02.cpp index c960d6ab..d412bc38 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/snippets/Tutorial_commainit_02.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/snippets/Tutorial_commainit_02.cpp @@ -1,7 +1,5 @@ -int rows=5, cols=5; -MatrixXf m(rows,cols); +int rows = 5, cols = 5; +MatrixXf m(rows, cols); m << (Matrix3f() << 1, 2, 3, 4, 5, 6, 7, 8, 9).finished(), - MatrixXf::Zero(3,cols-3), - MatrixXf::Zero(rows-3,3), - MatrixXf::Identity(rows-3,cols-3); + MatrixXf::Zero(3, cols - 3), MatrixXf::Zero(rows - 3, 3), MatrixXf::Identity(rows - 3, cols - 3); cout << m; diff --git a/filmulator-gui/core/nlmeans/eigen/doc/snippets/Tutorial_solve_matrix_inverse.cpp b/filmulator-gui/core/nlmeans/eigen/doc/snippets/Tutorial_solve_matrix_inverse.cpp index fff32444..e39696f4 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/snippets/Tutorial_solve_matrix_inverse.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/snippets/Tutorial_solve_matrix_inverse.cpp @@ -1,6 +1,6 @@ Matrix3f A; Vector3f b; -A << 1,2,3, 4,5,6, 7,8,10; +A << 1, 2, 3, 4, 5, 6, 7, 8, 10; b << 3, 3, 4; Vector3f x = A.inverse() * b; cout << "The solution is:" << endl << x << endl; diff --git a/filmulator-gui/core/nlmeans/eigen/doc/snippets/Tutorial_solve_multiple_rhs.cpp b/filmulator-gui/core/nlmeans/eigen/doc/snippets/Tutorial_solve_multiple_rhs.cpp index 5411a44a..b08100a4 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/snippets/Tutorial_solve_multiple_rhs.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/snippets/Tutorial_solve_multiple_rhs.cpp @@ -1,8 +1,8 @@ -Matrix3f A(3,3); -A << 1,2,3, 4,5,6, 7,8,10; -Matrix B; -B << 3,1, 3,1, 4,1; -Matrix X; +Matrix3f A(3, 3); +A << 1, 2, 3, 4, 5, 6, 7, 8, 10; +Matrix B; +B << 3, 1, 3, 1, 4, 1; +Matrix X; X = A.fullPivLu().solve(B); cout << "The solution with right-hand side (3,3,4) is:" << endl; cout << X.col(0) << endl; diff --git a/filmulator-gui/core/nlmeans/eigen/doc/snippets/Tutorial_solve_reuse_decomposition.cpp b/filmulator-gui/core/nlmeans/eigen/doc/snippets/Tutorial_solve_reuse_decomposition.cpp index 3ca06453..d0d44792 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/snippets/Tutorial_solve_reuse_decomposition.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/snippets/Tutorial_solve_reuse_decomposition.cpp @@ -1,13 +1,13 @@ -Matrix3f A(3,3); -A << 1,2,3, 4,5,6, 7,8,10; -PartialPivLU luOfA(A); // compute LU decomposition of A +Matrix3f A(3, 3); +A << 1, 2, 3, 4, 5, 6, 7, 8, 10; +PartialPivLU luOfA(A);// compute LU decomposition of A Vector3f b; -b << 3,3,4; +b << 3, 3, 4; Vector3f x; x = luOfA.solve(b); cout << "The solution with right-hand side (3,3,4) is:" << endl; cout << x << endl; -b << 1,1,1; +b << 1, 1, 1; x = luOfA.solve(b); cout << "The solution with right-hand side (1,1,1) is:" << endl; cout << x << endl; diff --git a/filmulator-gui/core/nlmeans/eigen/doc/snippets/Tutorial_solve_singular.cpp b/filmulator-gui/core/nlmeans/eigen/doc/snippets/Tutorial_solve_singular.cpp index abff1ef7..4d168dfd 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/snippets/Tutorial_solve_singular.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/snippets/Tutorial_solve_singular.cpp @@ -1,6 +1,6 @@ Matrix3f A; Vector3f b; -A << 1,2,3, 4,5,6, 7,8,9; +A << 1, 2, 3, 4, 5, 6, 7, 8, 9; b << 3, 3, 4; cout << "Here is the matrix A:" << endl << A << endl; cout << "Here is the vector b:" << endl << b << endl; diff --git a/filmulator-gui/core/nlmeans/eigen/doc/snippets/Tutorial_solve_triangular.cpp b/filmulator-gui/core/nlmeans/eigen/doc/snippets/Tutorial_solve_triangular.cpp index 9d13f22e..b422f249 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/snippets/Tutorial_solve_triangular.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/snippets/Tutorial_solve_triangular.cpp @@ -1,6 +1,6 @@ Matrix3f A; Vector3f b; -A << 1,2,3, 0,5,6, 0,0,10; +A << 1, 2, 3, 0, 5, 6, 0, 0, 10; b << 3, 3, 4; cout << "Here is the matrix A:" << endl << A << endl; cout << "Here is the vector b:" << endl << b << endl; diff --git a/filmulator-gui/core/nlmeans/eigen/doc/snippets/Tutorial_solve_triangular_inplace.cpp b/filmulator-gui/core/nlmeans/eigen/doc/snippets/Tutorial_solve_triangular_inplace.cpp index 16ae633a..f928e78e 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/snippets/Tutorial_solve_triangular_inplace.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/snippets/Tutorial_solve_triangular_inplace.cpp @@ -1,6 +1,6 @@ Matrix3f A; Vector3f b; -A << 1,2,3, 0,5,6, 0,0,10; +A << 1, 2, 3, 0, 5, 6, 0, 0, 10; b << 3, 3, 4; A.triangularView().solveInPlace(b); cout << "The solution is:" << endl << b << endl; diff --git a/filmulator-gui/core/nlmeans/eigen/doc/snippets/VectorwiseOp_homogeneous.cpp b/filmulator-gui/core/nlmeans/eigen/doc/snippets/VectorwiseOp_homogeneous.cpp index aba4fed0..f387aa3c 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/snippets/VectorwiseOp_homogeneous.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/snippets/VectorwiseOp_homogeneous.cpp @@ -1,7 +1,9 @@ -typedef Matrix Matrix3Xd; -Matrix3Xd M = Matrix3Xd::Random(3,5); +typedef Matrix Matrix3Xd; +Matrix3Xd M = Matrix3Xd::Random(3, 5); Projective3d P(Matrix4d::Random()); cout << "The matrix M is:" << endl << M << endl << endl; cout << "M.colwise().homogeneous():" << endl << M.colwise().homogeneous() << endl << endl; cout << "P * M.colwise().homogeneous():" << endl << P * M.colwise().homogeneous() << endl << endl; -cout << "P * M.colwise().homogeneous().hnormalized(): " << endl << (P * M.colwise().homogeneous()).colwise().hnormalized() << endl << endl; \ No newline at end of file +cout << "P * M.colwise().homogeneous().hnormalized(): " << endl + << (P * M.colwise().homogeneous()).colwise().hnormalized() << endl + << endl; \ No newline at end of file diff --git a/filmulator-gui/core/nlmeans/eigen/doc/snippets/Vectorwise_reverse.cpp b/filmulator-gui/core/nlmeans/eigen/doc/snippets/Vectorwise_reverse.cpp index 2f6a3508..85a00653 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/snippets/Vectorwise_reverse.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/snippets/Vectorwise_reverse.cpp @@ -1,10 +1,9 @@ -MatrixXi m = MatrixXi::Random(3,4); +MatrixXi m = MatrixXi::Random(3, 4); cout << "Here is the matrix m:" << endl << m << endl; cout << "Here is the rowwise reverse of m:" << endl << m.rowwise().reverse() << endl; cout << "Here is the colwise reverse of m:" << endl << m.colwise().reverse() << endl; -cout << "Here is the coefficient (1,0) in the rowise reverse of m:" << endl -<< m.rowwise().reverse()(1,0) << endl; +cout << "Here is the coefficient (1,0) in the rowise reverse of m:" << endl << m.rowwise().reverse()(1, 0) << endl; cout << "Let us overwrite this coefficient with the value 4." << endl; -//m.colwise().reverse()(1,0) = 4; +// m.colwise().reverse()(1,0) = 4; cout << "Now the matrix m is:" << endl << m << endl; diff --git a/filmulator-gui/core/nlmeans/eigen/doc/snippets/class_FullPivLU.cpp b/filmulator-gui/core/nlmeans/eigen/doc/snippets/class_FullPivLU.cpp index fce7fac0..2951c8d2 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/snippets/class_FullPivLU.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/snippets/class_FullPivLU.cpp @@ -3,11 +3,10 @@ typedef Matrix Matrix5x5; Matrix5x3 m = Matrix5x3::Random(); cout << "Here is the matrix m:" << endl << m << endl; Eigen::FullPivLU lu(m); -cout << "Here is, up to permutations, its LU decomposition matrix:" - << endl << lu.matrixLU() << endl; +cout << "Here is, up to permutations, its LU decomposition matrix:" << endl << lu.matrixLU() << endl; cout << "Here is the L part:" << endl; Matrix5x5 l = Matrix5x5::Identity(); -l.block<5,3>(0,0).triangularView() = lu.matrixLU(); +l.block<5, 3>(0, 0).triangularView() = lu.matrixLU(); cout << l << endl; cout << "Here is the U part:" << endl; Matrix5x3 u = lu.matrixLU().triangularView(); diff --git a/filmulator-gui/core/nlmeans/eigen/doc/snippets/tut_arithmetic_redux_minmax.cpp b/filmulator-gui/core/nlmeans/eigen/doc/snippets/tut_arithmetic_redux_minmax.cpp index f4ae7f40..dc433584 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/snippets/tut_arithmetic_redux_minmax.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/snippets/tut_arithmetic_redux_minmax.cpp @@ -1,12 +1,10 @@ - Matrix3f m = Matrix3f::Random(); - std::ptrdiff_t i, j; - float minOfM = m.minCoeff(&i,&j); - cout << "Here is the matrix m:\n" << m << endl; - cout << "Its minimum coefficient (" << minOfM - << ") is at position (" << i << "," << j << ")\n\n"; +Matrix3f m = Matrix3f::Random(); +std::ptrdiff_t i, j; +float minOfM = m.minCoeff(&i, &j); +cout << "Here is the matrix m:\n" << m << endl; +cout << "Its minimum coefficient (" << minOfM << ") is at position (" << i << "," << j << ")\n\n"; - RowVector4i v = RowVector4i::Random(); - int maxOfV = v.maxCoeff(&i); - cout << "Here is the vector v: " << v << endl; - cout << "Its maximum coefficient (" << maxOfV - << ") is at position " << i << endl; +RowVector4i v = RowVector4i::Random(); +int maxOfV = v.maxCoeff(&i); +cout << "Here is the vector v: " << v << endl; +cout << "Its maximum coefficient (" << maxOfV << ") is at position " << i << endl; diff --git a/filmulator-gui/core/nlmeans/eigen/doc/snippets/tut_arithmetic_transpose_aliasing.cpp b/filmulator-gui/core/nlmeans/eigen/doc/snippets/tut_arithmetic_transpose_aliasing.cpp index c8e4746d..f9bdaf6e 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/snippets/tut_arithmetic_transpose_aliasing.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/snippets/tut_arithmetic_transpose_aliasing.cpp @@ -1,5 +1,6 @@ -Matrix2i a; a << 1, 2, 3, 4; +Matrix2i a; +a << 1, 2, 3, 4; cout << "Here is the matrix a:\n" << a << endl; -a = a.transpose(); // !!! do NOT do this !!! +a = a.transpose();// !!! do NOT do this !!! cout << "and the result of the aliasing effect:\n" << a << endl; \ No newline at end of file diff --git a/filmulator-gui/core/nlmeans/eigen/doc/snippets/tut_arithmetic_transpose_conjugate.cpp b/filmulator-gui/core/nlmeans/eigen/doc/snippets/tut_arithmetic_transpose_conjugate.cpp index 88496b22..c13d0be7 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/snippets/tut_arithmetic_transpose_conjugate.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/snippets/tut_arithmetic_transpose_conjugate.cpp @@ -1,4 +1,4 @@ -MatrixXcf a = MatrixXcf::Random(2,2); +MatrixXcf a = MatrixXcf::Random(2, 2); cout << "Here is the matrix a\n" << a << endl; cout << "Here is the matrix a^T\n" << a.transpose() << endl; @@ -8,5 +8,3 @@ cout << "Here is the conjugate of a\n" << a.conjugate() << endl; cout << "Here is the matrix a^*\n" << a.adjoint() << endl; - - diff --git a/filmulator-gui/core/nlmeans/eigen/doc/snippets/tut_arithmetic_transpose_inplace.cpp b/filmulator-gui/core/nlmeans/eigen/doc/snippets/tut_arithmetic_transpose_inplace.cpp index 7a069ff2..21599348 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/snippets/tut_arithmetic_transpose_inplace.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/snippets/tut_arithmetic_transpose_inplace.cpp @@ -1,4 +1,5 @@ -MatrixXf a(2,3); a << 1, 2, 3, 4, 5, 6; +MatrixXf a(2, 3); +a << 1, 2, 3, 4, 5, 6; cout << "Here is the initial matrix a:\n" << a << endl; diff --git a/filmulator-gui/core/nlmeans/eigen/doc/snippets/tut_matrix_assignment_resizing.cpp b/filmulator-gui/core/nlmeans/eigen/doc/snippets/tut_matrix_assignment_resizing.cpp index cf189983..9f5da435 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/snippets/tut_matrix_assignment_resizing.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/snippets/tut_matrix_assignment_resizing.cpp @@ -1,5 +1,5 @@ -MatrixXf a(2,2); +MatrixXf a(2, 2); std::cout << "a is of size " << a.rows() << "x" << a.cols() << std::endl; -MatrixXf b(3,3); +MatrixXf b(3, 3); a = b; std::cout << "a is now of size " << a.rows() << "x" << a.cols() << std::endl; diff --git a/filmulator-gui/core/nlmeans/eigen/doc/special_examples/Tutorial_sparse_example.cpp b/filmulator-gui/core/nlmeans/eigen/doc/special_examples/Tutorial_sparse_example.cpp index c5767a8d..157a9eb8 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/special_examples/Tutorial_sparse_example.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/special_examples/Tutorial_sparse_example.cpp @@ -1,38 +1,37 @@ #include -#include #include +#include -typedef Eigen::SparseMatrix SpMat; // declares a column-major sparse matrix type of double +typedef Eigen::SparseMatrix SpMat;// declares a column-major sparse matrix type of double typedef Eigen::Triplet T; -void buildProblem(std::vector& coefficients, Eigen::VectorXd& b, int n); -void saveAsBitmap(const Eigen::VectorXd& x, int n, const char* filename); +void buildProblem(std::vector &coefficients, Eigen::VectorXd &b, int n); +void saveAsBitmap(const Eigen::VectorXd &x, int n, const char *filename); -int main(int argc, char** argv) +int main(int argc, char **argv) { - if(argc!=2) { + if (argc != 2) { std::cerr << "Error: expected one and only one argument.\n"; return -1; } - - int n = 300; // size of the image - int m = n*n; // number of unknows (=number of pixels) + + int n = 300;// size of the image + int m = n * n;// number of unknows (=number of pixels) // Assembly: - std::vector coefficients; // list of non-zeros coefficients - Eigen::VectorXd b(m); // the right hand side-vector resulting from the constraints + std::vector coefficients;// list of non-zeros coefficients + Eigen::VectorXd b(m);// the right hand side-vector resulting from the constraints buildProblem(coefficients, b, n); - SpMat A(m,m); + SpMat A(m, m); A.setFromTriplets(coefficients.begin(), coefficients.end()); // Solving: - Eigen::SimplicialCholesky chol(A); // performs a Cholesky factorization of A - Eigen::VectorXd x = chol.solve(b); // use the factorization to solve for the given right hand side + Eigen::SimplicialCholesky chol(A);// performs a Cholesky factorization of A + Eigen::VectorXd x = chol.solve(b);// use the factorization to solve for the given right hand side // Export the result to a file: saveAsBitmap(x, n, argv[1]); return 0; } - diff --git a/filmulator-gui/core/nlmeans/eigen/doc/special_examples/Tutorial_sparse_example_details.cpp b/filmulator-gui/core/nlmeans/eigen/doc/special_examples/Tutorial_sparse_example_details.cpp index bc18b018..0d8b091f 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/special_examples/Tutorial_sparse_example_details.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/special_examples/Tutorial_sparse_example_details.cpp @@ -1,44 +1,50 @@ #include -#include #include +#include -typedef Eigen::SparseMatrix SpMat; // declares a column-major sparse matrix type of double +typedef Eigen::SparseMatrix SpMat;// declares a column-major sparse matrix type of double typedef Eigen::Triplet T; -void insertCoefficient(int id, int i, int j, double w, std::vector& coeffs, - Eigen::VectorXd& b, const Eigen::VectorXd& boundary) +void insertCoefficient(int id, + int i, + int j, + double w, + std::vector &coeffs, + Eigen::VectorXd &b, + const Eigen::VectorXd &boundary) { int n = int(boundary.size()); - int id1 = i+j*n; + int id1 = i + j * n; - if(i==-1 || i==n) b(id) -= w * boundary(j); // constrained coefficient - else if(j==-1 || j==n) b(id) -= w * boundary(i); // constrained coefficient - else coeffs.push_back(T(id,id1,w)); // unknown coefficient + if (i == -1 || i == n) + b(id) -= w * boundary(j);// constrained coefficient + else if (j == -1 || j == n) + b(id) -= w * boundary(i);// constrained coefficient + else + coeffs.push_back(T(id, id1, w));// unknown coefficient } -void buildProblem(std::vector& coefficients, Eigen::VectorXd& b, int n) +void buildProblem(std::vector &coefficients, Eigen::VectorXd &b, int n) { b.setZero(); - Eigen::ArrayXd boundary = Eigen::ArrayXd::LinSpaced(n, 0,M_PI).sin().pow(2); - for(int j=0; j bits = (x*255).cast(); - QImage img(bits.data(), n,n,QImage::Format_Indexed8); + Eigen::Array bits = (x * 255).cast(); + QImage img(bits.data(), n, n, QImage::Format_Indexed8); img.setColorCount(256); - for(int i=0;i<256;i++) img.setColor(i,qRgb(i,i,i)); + for (int i = 0; i < 256; i++) img.setColor(i, qRgb(i, i, i)); img.save(filename); } diff --git a/filmulator-gui/core/nlmeans/eigen/doc/special_examples/random_cpp11.cpp b/filmulator-gui/core/nlmeans/eigen/doc/special_examples/random_cpp11.cpp index 33744c05..913ee704 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/special_examples/random_cpp11.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/special_examples/random_cpp11.cpp @@ -4,11 +4,12 @@ using namespace Eigen; -int main() { +int main() +{ std::default_random_engine generator; std::poisson_distribution distribution(4.1); - auto poisson = [&] () {return distribution(generator);}; + auto poisson = [&]() { return distribution(generator); }; - RowVectorXi v = RowVectorXi::NullaryExpr(10, poisson ); + RowVectorXi v = RowVectorXi::NullaryExpr(10, poisson); std::cout << v << "\n"; } diff --git a/filmulator-gui/core/nlmeans/eigen/doc/tutorial.cpp b/filmulator-gui/core/nlmeans/eigen/doc/tutorial.cpp index 62be7c27..57b272b9 100644 --- a/filmulator-gui/core/nlmeans/eigen/doc/tutorial.cpp +++ b/filmulator-gui/core/nlmeans/eigen/doc/tutorial.cpp @@ -13,29 +13,29 @@ int main(int argc, char *argv[]) // demo non-static set... functions m4.setZero(); m3.diagonal().setOnes(); - + std::cout << "*** Step 2 ***\nm3:\n" << m3 << "\nm4:\n" << m4 << std::endl; // demo fixed-size block() expression as lvalue and as rvalue - m4.block<3,3>(0,1) = m3; - m3.row(2) = m4.block<1,3>(2,0); + m4.block<3, 3>(0, 1) = m3; + m3.row(2) = m4.block<1, 3>(2, 0); std::cout << "*** Step 3 ***\nm3:\n" << m3 << "\nm4:\n" << m4 << std::endl; // demo dynamic-size block() { int rows = 3, cols = 3; - m4.block(0,1,3,3).setIdentity(); + m4.block(0, 1, 3, 3).setIdentity(); std::cout << "*** Step 4 ***\nm4:\n" << m4 << std::endl; } // demo vector blocks - m4.diagonal().block(1,2).setOnes(); + m4.diagonal().block(1, 2).setOnes(); std::cout << "*** Step 5 ***\nm4.diagonal():\n" << m4.diagonal() << std::endl; std::cout << "m4.diagonal().start(3)\n" << m4.diagonal().start(3) << std::endl; // demo coeff-wise operations - m4 = m4.cwise()*m4; + m4 = m4.cwise() * m4; m3 = m3.cwise().cos(); std::cout << "*** Step 6 ***\nm3:\n" << m3 << "\nm4:\n" << m4 << std::endl; @@ -46,17 +46,17 @@ int main(int argc, char *argv[]) std::cout << "m4.rowwise().sum():\n" << m4.rowwise().sum() << std::endl; // demo intelligent auto-evaluation - m4 = m4 * m4; // auto-evaluates so no aliasing problem (performance penalty is low) - Eigen::Matrix4f other = (m4 * m4).lazy(); // forces lazy evaluation - m4 = m4 + m4; // here Eigen goes for lazy evaluation, as with most expressions - m4 = -m4 + m4 + 5 * m4; // same here, Eigen chooses lazy evaluation for all that. - m4 = m4 * (m4 + m4); // here Eigen chooses to first evaluate m4 + m4 into a temporary. - // indeed, here it is an optimization to cache this intermediate result. - m3 = m3 * m4.block<3,3>(1,1); // here Eigen chooses NOT to evaluate block() into a temporary - // because accessing coefficients of that block expression is not more costly than accessing - // coefficients of a plain matrix. - m4 = m4 * m4.transpose(); // same here, lazy evaluation of the transpose. - m4 = m4 * m4.transpose().eval(); // forces immediate evaluation of the transpose + m4 = m4 * m4;// auto-evaluates so no aliasing problem (performance penalty is low) + Eigen::Matrix4f other = (m4 * m4).lazy();// forces lazy evaluation + m4 = m4 + m4;// here Eigen goes for lazy evaluation, as with most expressions + m4 = -m4 + m4 + 5 * m4;// same here, Eigen chooses lazy evaluation for all that. + m4 = m4 * (m4 + m4);// here Eigen chooses to first evaluate m4 + m4 into a temporary. + // indeed, here it is an optimization to cache this intermediate result. + m3 = m3 * m4.block<3, 3>(1, 1);// here Eigen chooses NOT to evaluate block() into a temporary + // because accessing coefficients of that block expression is not more costly than + // accessing coefficients of a plain matrix. + m4 = m4 * m4.transpose();// same here, lazy evaluation of the transpose. + m4 = m4 * m4.transpose().eval();// forces immediate evaluation of the transpose std::cout << "*** Step 8 ***\nm3:\n" << m3 << "\nm4:\n" << m4 << std::endl; } diff --git a/filmulator-gui/core/nlmeans/eigen/failtest/bdcsvd_int.cpp b/filmulator-gui/core/nlmeans/eigen/failtest/bdcsvd_int.cpp index 670752cf..6fe2aca0 100644 --- a/filmulator-gui/core/nlmeans/eigen/failtest/bdcsvd_int.cpp +++ b/filmulator-gui/core/nlmeans/eigen/failtest/bdcsvd_int.cpp @@ -8,7 +8,4 @@ using namespace Eigen; -int main() -{ - BDCSVD > qr(Matrix::Random(10,10)); -} +int main() { BDCSVD> qr(Matrix::Random(10, 10)); } diff --git a/filmulator-gui/core/nlmeans/eigen/failtest/block_nonconst_ctor_on_const_xpr_0.cpp b/filmulator-gui/core/nlmeans/eigen/failtest/block_nonconst_ctor_on_const_xpr_0.cpp index 40b82014..b2e134bf 100644 --- a/filmulator-gui/core/nlmeans/eigen/failtest/block_nonconst_ctor_on_const_xpr_0.cpp +++ b/filmulator-gui/core/nlmeans/eigen/failtest/block_nonconst_ctor_on_const_xpr_0.cpp @@ -8,8 +8,6 @@ using namespace Eigen; -void foo(CV_QUALIFIER Matrix3d &m){ - Block b(m,0,0); -} +void foo(CV_QUALIFIER Matrix3d &m) { Block b(m, 0, 0); } int main() {} diff --git a/filmulator-gui/core/nlmeans/eigen/failtest/block_nonconst_ctor_on_const_xpr_1.cpp b/filmulator-gui/core/nlmeans/eigen/failtest/block_nonconst_ctor_on_const_xpr_1.cpp index ef6d5370..c3ddc9af 100644 --- a/filmulator-gui/core/nlmeans/eigen/failtest/block_nonconst_ctor_on_const_xpr_1.cpp +++ b/filmulator-gui/core/nlmeans/eigen/failtest/block_nonconst_ctor_on_const_xpr_1.cpp @@ -8,8 +8,6 @@ using namespace Eigen; -void foo(CV_QUALIFIER Matrix3d &m){ - Block b(m,0,0,3,3); -} +void foo(CV_QUALIFIER Matrix3d &m) { Block b(m, 0, 0, 3, 3); } int main() {} diff --git a/filmulator-gui/core/nlmeans/eigen/failtest/block_nonconst_ctor_on_const_xpr_2.cpp b/filmulator-gui/core/nlmeans/eigen/failtest/block_nonconst_ctor_on_const_xpr_2.cpp index 43f18aec..538ba866 100644 --- a/filmulator-gui/core/nlmeans/eigen/failtest/block_nonconst_ctor_on_const_xpr_2.cpp +++ b/filmulator-gui/core/nlmeans/eigen/failtest/block_nonconst_ctor_on_const_xpr_2.cpp @@ -8,9 +8,10 @@ using namespace Eigen; -void foo(CV_QUALIFIER Matrix3d &m){ - // row/column constructor - Block b(m,0); +void foo(CV_QUALIFIER Matrix3d &m) +{ + // row/column constructor + Block b(m, 0); } int main() {} diff --git a/filmulator-gui/core/nlmeans/eigen/failtest/block_on_const_type_actually_const_0.cpp b/filmulator-gui/core/nlmeans/eigen/failtest/block_on_const_type_actually_const_0.cpp index 009bebec..93ee3019 100644 --- a/filmulator-gui/core/nlmeans/eigen/failtest/block_on_const_type_actually_const_0.cpp +++ b/filmulator-gui/core/nlmeans/eigen/failtest/block_on_const_type_actually_const_0.cpp @@ -8,9 +8,10 @@ using namespace Eigen; -void foo(){ - Matrix3f m; - Block(m, 0, 0, 3, 3).coeffRef(0, 0) = 1.0f; +void foo() +{ + Matrix3f m; + Block(m, 0, 0, 3, 3).coeffRef(0, 0) = 1.0f; } int main() {} diff --git a/filmulator-gui/core/nlmeans/eigen/failtest/block_on_const_type_actually_const_1.cpp b/filmulator-gui/core/nlmeans/eigen/failtest/block_on_const_type_actually_const_1.cpp index 4c3e93ff..bafe5e4c 100644 --- a/filmulator-gui/core/nlmeans/eigen/failtest/block_on_const_type_actually_const_1.cpp +++ b/filmulator-gui/core/nlmeans/eigen/failtest/block_on_const_type_actually_const_1.cpp @@ -8,9 +8,10 @@ using namespace Eigen; -void foo(){ - MatrixXf m; - Block(m, 0, 0).coeffRef(0, 0) = 1.0f; +void foo() +{ + MatrixXf m; + Block(m, 0, 0).coeffRef(0, 0) = 1.0f; } int main() {} diff --git a/filmulator-gui/core/nlmeans/eigen/failtest/colpivqr_int.cpp b/filmulator-gui/core/nlmeans/eigen/failtest/colpivqr_int.cpp index db11910d..23cb9b84 100644 --- a/filmulator-gui/core/nlmeans/eigen/failtest/colpivqr_int.cpp +++ b/filmulator-gui/core/nlmeans/eigen/failtest/colpivqr_int.cpp @@ -10,5 +10,5 @@ using namespace Eigen; int main() { - ColPivHouseholderQR > qr(Matrix::Random(10,10)); + ColPivHouseholderQR> qr(Matrix::Random(10, 10)); } diff --git a/filmulator-gui/core/nlmeans/eigen/failtest/const_qualified_block_method_retval_0.cpp b/filmulator-gui/core/nlmeans/eigen/failtest/const_qualified_block_method_retval_0.cpp index a6bd5fee..08b5d3c7 100644 --- a/filmulator-gui/core/nlmeans/eigen/failtest/const_qualified_block_method_retval_0.cpp +++ b/filmulator-gui/core/nlmeans/eigen/failtest/const_qualified_block_method_retval_0.cpp @@ -8,8 +8,6 @@ using namespace Eigen; -void foo(CV_QUALIFIER Matrix3d &m){ - Block b(m.block<3,3>(0,0)); -} +void foo(CV_QUALIFIER Matrix3d &m) { Block b(m.block<3, 3>(0, 0)); } int main() {} diff --git a/filmulator-gui/core/nlmeans/eigen/failtest/const_qualified_block_method_retval_1.cpp b/filmulator-gui/core/nlmeans/eigen/failtest/const_qualified_block_method_retval_1.cpp index ef40c247..06e12c7f 100644 --- a/filmulator-gui/core/nlmeans/eigen/failtest/const_qualified_block_method_retval_1.cpp +++ b/filmulator-gui/core/nlmeans/eigen/failtest/const_qualified_block_method_retval_1.cpp @@ -8,8 +8,6 @@ using namespace Eigen; -void foo(CV_QUALIFIER Matrix3d &m){ - Block b(m.block(0,0,3,3)); -} +void foo(CV_QUALIFIER Matrix3d &m) { Block b(m.block(0, 0, 3, 3)); } int main() {} diff --git a/filmulator-gui/core/nlmeans/eigen/failtest/const_qualified_diagonal_method_retval.cpp b/filmulator-gui/core/nlmeans/eigen/failtest/const_qualified_diagonal_method_retval.cpp index 809594aa..f3acba62 100644 --- a/filmulator-gui/core/nlmeans/eigen/failtest/const_qualified_diagonal_method_retval.cpp +++ b/filmulator-gui/core/nlmeans/eigen/failtest/const_qualified_diagonal_method_retval.cpp @@ -8,8 +8,6 @@ using namespace Eigen; -void foo(CV_QUALIFIER Matrix3d &m){ - Diagonal b(m.diagonal()); -} +void foo(CV_QUALIFIER Matrix3d &m) { Diagonal b(m.diagonal()); } int main() {} diff --git a/filmulator-gui/core/nlmeans/eigen/failtest/const_qualified_transpose_method_retval.cpp b/filmulator-gui/core/nlmeans/eigen/failtest/const_qualified_transpose_method_retval.cpp index 2d7f19ca..394f64a6 100644 --- a/filmulator-gui/core/nlmeans/eigen/failtest/const_qualified_transpose_method_retval.cpp +++ b/filmulator-gui/core/nlmeans/eigen/failtest/const_qualified_transpose_method_retval.cpp @@ -8,8 +8,6 @@ using namespace Eigen; -void foo(CV_QUALIFIER Matrix3d &m){ - Transpose b(m.transpose()); -} +void foo(CV_QUALIFIER Matrix3d &m) { Transpose b(m.transpose()); } int main() {} diff --git a/filmulator-gui/core/nlmeans/eigen/failtest/cwiseunaryview_nonconst_ctor_on_const_xpr.cpp b/filmulator-gui/core/nlmeans/eigen/failtest/cwiseunaryview_nonconst_ctor_on_const_xpr.cpp index e23cf8fd..643335ac 100644 --- a/filmulator-gui/core/nlmeans/eigen/failtest/cwiseunaryview_nonconst_ctor_on_const_xpr.cpp +++ b/filmulator-gui/core/nlmeans/eigen/failtest/cwiseunaryview_nonconst_ctor_on_const_xpr.cpp @@ -8,8 +8,6 @@ using namespace Eigen; -void foo(CV_QUALIFIER Matrix3d &m){ - CwiseUnaryView,Matrix3d> t(m); -} +void foo(CV_QUALIFIER Matrix3d &m) { CwiseUnaryView, Matrix3d> t(m); } int main() {} diff --git a/filmulator-gui/core/nlmeans/eigen/failtest/cwiseunaryview_on_const_type_actually_const.cpp b/filmulator-gui/core/nlmeans/eigen/failtest/cwiseunaryview_on_const_type_actually_const.cpp index fcd41dfd..5902b1ef 100644 --- a/filmulator-gui/core/nlmeans/eigen/failtest/cwiseunaryview_on_const_type_actually_const.cpp +++ b/filmulator-gui/core/nlmeans/eigen/failtest/cwiseunaryview_on_const_type_actually_const.cpp @@ -8,9 +8,10 @@ using namespace Eigen; -void foo(){ - MatrixXf m; - CwiseUnaryView,CV_QUALIFIER MatrixXf>(m).coeffRef(0, 0) = 1.0f; +void foo() +{ + MatrixXf m; + CwiseUnaryView, CV_QUALIFIER MatrixXf>(m).coeffRef(0, 0) = 1.0f; } int main() {} diff --git a/filmulator-gui/core/nlmeans/eigen/failtest/diagonal_nonconst_ctor_on_const_xpr.cpp b/filmulator-gui/core/nlmeans/eigen/failtest/diagonal_nonconst_ctor_on_const_xpr.cpp index 76398a2c..92412703 100644 --- a/filmulator-gui/core/nlmeans/eigen/failtest/diagonal_nonconst_ctor_on_const_xpr.cpp +++ b/filmulator-gui/core/nlmeans/eigen/failtest/diagonal_nonconst_ctor_on_const_xpr.cpp @@ -8,8 +8,6 @@ using namespace Eigen; -void foo(CV_QUALIFIER Matrix3d &m){ - Diagonal d(m); -} +void foo(CV_QUALIFIER Matrix3d &m) { Diagonal d(m); } int main() {} diff --git a/filmulator-gui/core/nlmeans/eigen/failtest/diagonal_on_const_type_actually_const.cpp b/filmulator-gui/core/nlmeans/eigen/failtest/diagonal_on_const_type_actually_const.cpp index d4b2fd9b..0e883c99 100644 --- a/filmulator-gui/core/nlmeans/eigen/failtest/diagonal_on_const_type_actually_const.cpp +++ b/filmulator-gui/core/nlmeans/eigen/failtest/diagonal_on_const_type_actually_const.cpp @@ -8,9 +8,10 @@ using namespace Eigen; -void foo(){ - MatrixXf m; - Diagonal(m).coeffRef(0) = 1.0f; +void foo() +{ + MatrixXf m; + Diagonal(m).coeffRef(0) = 1.0f; } int main() {} diff --git a/filmulator-gui/core/nlmeans/eigen/failtest/eigensolver_cplx.cpp b/filmulator-gui/core/nlmeans/eigen/failtest/eigensolver_cplx.cpp index c2e21e18..d828d9f6 100644 --- a/filmulator-gui/core/nlmeans/eigen/failtest/eigensolver_cplx.cpp +++ b/filmulator-gui/core/nlmeans/eigen/failtest/eigensolver_cplx.cpp @@ -8,7 +8,4 @@ using namespace Eigen; -int main() -{ - EigenSolver > eig(Matrix::Random(10,10)); -} +int main() { EigenSolver> eig(Matrix::Random(10, 10)); } diff --git a/filmulator-gui/core/nlmeans/eigen/failtest/eigensolver_int.cpp b/filmulator-gui/core/nlmeans/eigen/failtest/eigensolver_int.cpp index eda8dc20..c75cb914 100644 --- a/filmulator-gui/core/nlmeans/eigen/failtest/eigensolver_int.cpp +++ b/filmulator-gui/core/nlmeans/eigen/failtest/eigensolver_int.cpp @@ -8,7 +8,4 @@ using namespace Eigen; -int main() -{ - EigenSolver > eig(Matrix::Random(10,10)); -} +int main() { EigenSolver> eig(Matrix::Random(10, 10)); } diff --git a/filmulator-gui/core/nlmeans/eigen/failtest/fullpivlu_int.cpp b/filmulator-gui/core/nlmeans/eigen/failtest/fullpivlu_int.cpp index e9d2c6eb..16a0756b 100644 --- a/filmulator-gui/core/nlmeans/eigen/failtest/fullpivlu_int.cpp +++ b/filmulator-gui/core/nlmeans/eigen/failtest/fullpivlu_int.cpp @@ -8,7 +8,4 @@ using namespace Eigen; -int main() -{ - FullPivLU > lu(Matrix::Random(10,10)); -} +int main() { FullPivLU> lu(Matrix::Random(10, 10)); } diff --git a/filmulator-gui/core/nlmeans/eigen/failtest/fullpivqr_int.cpp b/filmulator-gui/core/nlmeans/eigen/failtest/fullpivqr_int.cpp index d182a7b6..96e39e9b 100644 --- a/filmulator-gui/core/nlmeans/eigen/failtest/fullpivqr_int.cpp +++ b/filmulator-gui/core/nlmeans/eigen/failtest/fullpivqr_int.cpp @@ -10,5 +10,5 @@ using namespace Eigen; int main() { - FullPivHouseholderQR > qr(Matrix::Random(10,10)); + FullPivHouseholderQR> qr(Matrix::Random(10, 10)); } diff --git a/filmulator-gui/core/nlmeans/eigen/failtest/jacobisvd_int.cpp b/filmulator-gui/core/nlmeans/eigen/failtest/jacobisvd_int.cpp index 12790aef..e647f32c 100644 --- a/filmulator-gui/core/nlmeans/eigen/failtest/jacobisvd_int.cpp +++ b/filmulator-gui/core/nlmeans/eigen/failtest/jacobisvd_int.cpp @@ -8,7 +8,4 @@ using namespace Eigen; -int main() -{ - JacobiSVD > qr(Matrix::Random(10,10)); -} +int main() { JacobiSVD> qr(Matrix::Random(10, 10)); } diff --git a/filmulator-gui/core/nlmeans/eigen/failtest/ldlt_int.cpp b/filmulator-gui/core/nlmeans/eigen/failtest/ldlt_int.cpp index 243e4574..25a20709 100644 --- a/filmulator-gui/core/nlmeans/eigen/failtest/ldlt_int.cpp +++ b/filmulator-gui/core/nlmeans/eigen/failtest/ldlt_int.cpp @@ -8,7 +8,4 @@ using namespace Eigen; -int main() -{ - LDLT > ldlt(Matrix::Random(10,10)); -} +int main() { LDLT> ldlt(Matrix::Random(10, 10)); } diff --git a/filmulator-gui/core/nlmeans/eigen/failtest/llt_int.cpp b/filmulator-gui/core/nlmeans/eigen/failtest/llt_int.cpp index cb020650..37f829a2 100644 --- a/filmulator-gui/core/nlmeans/eigen/failtest/llt_int.cpp +++ b/filmulator-gui/core/nlmeans/eigen/failtest/llt_int.cpp @@ -8,7 +8,4 @@ using namespace Eigen; -int main() -{ - LLT > llt(Matrix::Random(10,10)); -} +int main() { LLT> llt(Matrix::Random(10, 10)); } diff --git a/filmulator-gui/core/nlmeans/eigen/failtest/map_nonconst_ctor_on_const_ptr_0.cpp b/filmulator-gui/core/nlmeans/eigen/failtest/map_nonconst_ctor_on_const_ptr_0.cpp index d75686f5..5a76cdec 100644 --- a/filmulator-gui/core/nlmeans/eigen/failtest/map_nonconst_ctor_on_const_ptr_0.cpp +++ b/filmulator-gui/core/nlmeans/eigen/failtest/map_nonconst_ctor_on_const_ptr_0.cpp @@ -8,8 +8,6 @@ using namespace Eigen; -void foo(CV_QUALIFIER float *ptr){ - Map m(ptr); -} +void foo(CV_QUALIFIER float *ptr) { Map m(ptr); } int main() {} diff --git a/filmulator-gui/core/nlmeans/eigen/failtest/map_nonconst_ctor_on_const_ptr_1.cpp b/filmulator-gui/core/nlmeans/eigen/failtest/map_nonconst_ctor_on_const_ptr_1.cpp index eda134dc..4306fba9 100644 --- a/filmulator-gui/core/nlmeans/eigen/failtest/map_nonconst_ctor_on_const_ptr_1.cpp +++ b/filmulator-gui/core/nlmeans/eigen/failtest/map_nonconst_ctor_on_const_ptr_1.cpp @@ -8,8 +8,6 @@ using namespace Eigen; -void foo(CV_QUALIFIER float *ptr, DenseIndex size){ - Map m(ptr, size); -} +void foo(CV_QUALIFIER float *ptr, DenseIndex size) { Map m(ptr, size); } int main() {} diff --git a/filmulator-gui/core/nlmeans/eigen/failtest/map_nonconst_ctor_on_const_ptr_2.cpp b/filmulator-gui/core/nlmeans/eigen/failtest/map_nonconst_ctor_on_const_ptr_2.cpp index 06b4b627..411d9420 100644 --- a/filmulator-gui/core/nlmeans/eigen/failtest/map_nonconst_ctor_on_const_ptr_2.cpp +++ b/filmulator-gui/core/nlmeans/eigen/failtest/map_nonconst_ctor_on_const_ptr_2.cpp @@ -8,8 +8,6 @@ using namespace Eigen; -void foo(CV_QUALIFIER float *ptr, DenseIndex rows, DenseIndex cols){ - Map m(ptr, rows, cols); -} +void foo(CV_QUALIFIER float *ptr, DenseIndex rows, DenseIndex cols) { Map m(ptr, rows, cols); } int main() {} diff --git a/filmulator-gui/core/nlmeans/eigen/failtest/map_nonconst_ctor_on_const_ptr_3.cpp b/filmulator-gui/core/nlmeans/eigen/failtest/map_nonconst_ctor_on_const_ptr_3.cpp index 830f6f0c..638803c9 100644 --- a/filmulator-gui/core/nlmeans/eigen/failtest/map_nonconst_ctor_on_const_ptr_3.cpp +++ b/filmulator-gui/core/nlmeans/eigen/failtest/map_nonconst_ctor_on_const_ptr_3.cpp @@ -8,8 +8,9 @@ using namespace Eigen; -void foo(CV_QUALIFIER float *ptr, DenseIndex rows, DenseIndex cols){ - Map > m(ptr, rows, cols, InnerStride<2>()); +void foo(CV_QUALIFIER float *ptr, DenseIndex rows, DenseIndex cols) +{ + Map> m(ptr, rows, cols, InnerStride<2>()); } int main() {} diff --git a/filmulator-gui/core/nlmeans/eigen/failtest/map_nonconst_ctor_on_const_ptr_4.cpp b/filmulator-gui/core/nlmeans/eigen/failtest/map_nonconst_ctor_on_const_ptr_4.cpp index c3e8c952..638cc582 100644 --- a/filmulator-gui/core/nlmeans/eigen/failtest/map_nonconst_ctor_on_const_ptr_4.cpp +++ b/filmulator-gui/core/nlmeans/eigen/failtest/map_nonconst_ctor_on_const_ptr_4.cpp @@ -8,8 +8,9 @@ using namespace Eigen; -void foo(const float *ptr, DenseIndex rows, DenseIndex cols){ - Map > m(ptr, rows, cols, OuterStride<>(2)); +void foo(const float *ptr, DenseIndex rows, DenseIndex cols) +{ + Map> m(ptr, rows, cols, OuterStride<>(2)); } int main() {} diff --git a/filmulator-gui/core/nlmeans/eigen/failtest/map_on_const_type_actually_const_0.cpp b/filmulator-gui/core/nlmeans/eigen/failtest/map_on_const_type_actually_const_0.cpp index 8cb6aa0c..9f485dba 100644 --- a/filmulator-gui/core/nlmeans/eigen/failtest/map_on_const_type_actually_const_0.cpp +++ b/filmulator-gui/core/nlmeans/eigen/failtest/map_on_const_type_actually_const_0.cpp @@ -8,8 +8,6 @@ using namespace Eigen; -void foo(float *ptr){ - Map(ptr, 1, 1).coeffRef(0,0) = 1.0f; -} +void foo(float *ptr) { Map(ptr, 1, 1).coeffRef(0, 0) = 1.0f; } int main() {} diff --git a/filmulator-gui/core/nlmeans/eigen/failtest/map_on_const_type_actually_const_1.cpp b/filmulator-gui/core/nlmeans/eigen/failtest/map_on_const_type_actually_const_1.cpp index 04e067c3..c5ff0e6b 100644 --- a/filmulator-gui/core/nlmeans/eigen/failtest/map_on_const_type_actually_const_1.cpp +++ b/filmulator-gui/core/nlmeans/eigen/failtest/map_on_const_type_actually_const_1.cpp @@ -8,8 +8,6 @@ using namespace Eigen; -void foo(float *ptr){ - Map(ptr).coeffRef(0) = 1.0f; -} +void foo(float *ptr) { Map(ptr).coeffRef(0) = 1.0f; } int main() {} diff --git a/filmulator-gui/core/nlmeans/eigen/failtest/partialpivlu_int.cpp b/filmulator-gui/core/nlmeans/eigen/failtest/partialpivlu_int.cpp index 98ef282e..affdb3f1 100644 --- a/filmulator-gui/core/nlmeans/eigen/failtest/partialpivlu_int.cpp +++ b/filmulator-gui/core/nlmeans/eigen/failtest/partialpivlu_int.cpp @@ -8,7 +8,4 @@ using namespace Eigen; -int main() -{ - PartialPivLU > lu(Matrix::Random(10,10)); -} +int main() { PartialPivLU> lu(Matrix::Random(10, 10)); } diff --git a/filmulator-gui/core/nlmeans/eigen/failtest/qr_int.cpp b/filmulator-gui/core/nlmeans/eigen/failtest/qr_int.cpp index ce200e81..e1f381bc 100644 --- a/filmulator-gui/core/nlmeans/eigen/failtest/qr_int.cpp +++ b/filmulator-gui/core/nlmeans/eigen/failtest/qr_int.cpp @@ -8,7 +8,4 @@ using namespace Eigen; -int main() -{ - HouseholderQR > qr(Matrix::Random(10,10)); -} +int main() { HouseholderQR> qr(Matrix::Random(10, 10)); } diff --git a/filmulator-gui/core/nlmeans/eigen/failtest/ref_1.cpp b/filmulator-gui/core/nlmeans/eigen/failtest/ref_1.cpp index 8b798d53..6eee3c11 100644 --- a/filmulator-gui/core/nlmeans/eigen/failtest/ref_1.cpp +++ b/filmulator-gui/core/nlmeans/eigen/failtest/ref_1.cpp @@ -8,11 +8,11 @@ using namespace Eigen; -void call_ref(Ref a) { } +void call_ref(Ref a) {} int main() { VectorXf a(10); - CV_QUALIFIER VectorXf& ac(a); + CV_QUALIFIER VectorXf &ac(a); call_ref(ac); } diff --git a/filmulator-gui/core/nlmeans/eigen/failtest/ref_2.cpp b/filmulator-gui/core/nlmeans/eigen/failtest/ref_2.cpp index 0b779ccf..9902a87d 100644 --- a/filmulator-gui/core/nlmeans/eigen/failtest/ref_2.cpp +++ b/filmulator-gui/core/nlmeans/eigen/failtest/ref_2.cpp @@ -2,11 +2,11 @@ using namespace Eigen; -void call_ref(Ref a) { } +void call_ref(Ref a) {} int main() { - MatrixXf A(10,10); + MatrixXf A(10, 10); #ifdef EIGEN_SHOULD_FAIL_TO_BUILD call_ref(A.row(3)); #else diff --git a/filmulator-gui/core/nlmeans/eigen/failtest/ref_3.cpp b/filmulator-gui/core/nlmeans/eigen/failtest/ref_3.cpp index f46027d4..4e68c379 100644 --- a/filmulator-gui/core/nlmeans/eigen/failtest/ref_3.cpp +++ b/filmulator-gui/core/nlmeans/eigen/failtest/ref_3.cpp @@ -3,13 +3,13 @@ using namespace Eigen; #ifdef EIGEN_SHOULD_FAIL_TO_BUILD -void call_ref(Ref a) { } +void call_ref(Ref a) {} #else -void call_ref(const Ref &a) { } +void call_ref(const Ref &a) {} #endif int main() { VectorXf a(10); - call_ref(a+a); + call_ref(a + a); } diff --git a/filmulator-gui/core/nlmeans/eigen/failtest/ref_4.cpp b/filmulator-gui/core/nlmeans/eigen/failtest/ref_4.cpp index 6c11fa4c..a4167a34 100644 --- a/filmulator-gui/core/nlmeans/eigen/failtest/ref_4.cpp +++ b/filmulator-gui/core/nlmeans/eigen/failtest/ref_4.cpp @@ -2,11 +2,11 @@ using namespace Eigen; -void call_ref(Ref > a) {} +void call_ref(Ref> a) {} int main() { - MatrixXf A(10,10); + MatrixXf A(10, 10); #ifdef EIGEN_SHOULD_FAIL_TO_BUILD call_ref(A.transpose()); #else diff --git a/filmulator-gui/core/nlmeans/eigen/failtest/ref_5.cpp b/filmulator-gui/core/nlmeans/eigen/failtest/ref_5.cpp index 846d5279..579d6a78 100644 --- a/filmulator-gui/core/nlmeans/eigen/failtest/ref_5.cpp +++ b/filmulator-gui/core/nlmeans/eigen/failtest/ref_5.cpp @@ -2,7 +2,7 @@ using namespace Eigen; -void call_ref(Ref a) { } +void call_ref(Ref a) {} int main() { diff --git a/filmulator-gui/core/nlmeans/eigen/failtest/selfadjointview_nonconst_ctor_on_const_xpr.cpp b/filmulator-gui/core/nlmeans/eigen/failtest/selfadjointview_nonconst_ctor_on_const_xpr.cpp index a240f818..cf049571 100644 --- a/filmulator-gui/core/nlmeans/eigen/failtest/selfadjointview_nonconst_ctor_on_const_xpr.cpp +++ b/filmulator-gui/core/nlmeans/eigen/failtest/selfadjointview_nonconst_ctor_on_const_xpr.cpp @@ -8,8 +8,6 @@ using namespace Eigen; -void foo(CV_QUALIFIER Matrix3d &m){ - SelfAdjointView t(m); -} +void foo(CV_QUALIFIER Matrix3d &m) { SelfAdjointView t(m); } int main() {} diff --git a/filmulator-gui/core/nlmeans/eigen/failtest/selfadjointview_on_const_type_actually_const.cpp b/filmulator-gui/core/nlmeans/eigen/failtest/selfadjointview_on_const_type_actually_const.cpp index 19aaad6d..d9388891 100644 --- a/filmulator-gui/core/nlmeans/eigen/failtest/selfadjointview_on_const_type_actually_const.cpp +++ b/filmulator-gui/core/nlmeans/eigen/failtest/selfadjointview_on_const_type_actually_const.cpp @@ -8,9 +8,10 @@ using namespace Eigen; -void foo(){ - MatrixXf m; - SelfAdjointView(m).coeffRef(0, 0) = 1.0f; +void foo() +{ + MatrixXf m; + SelfAdjointView(m).coeffRef(0, 0) = 1.0f; } int main() {} diff --git a/filmulator-gui/core/nlmeans/eigen/failtest/sparse_ref_1.cpp b/filmulator-gui/core/nlmeans/eigen/failtest/sparse_ref_1.cpp index d78d1f9b..dd0beaf0 100644 --- a/filmulator-gui/core/nlmeans/eigen/failtest/sparse_ref_1.cpp +++ b/filmulator-gui/core/nlmeans/eigen/failtest/sparse_ref_1.cpp @@ -8,11 +8,11 @@ using namespace Eigen; -void call_ref(Ref > a) { } +void call_ref(Ref> a) {} int main() { - SparseMatrix a(10,10); - CV_QUALIFIER SparseMatrix& ac(a); + SparseMatrix a(10, 10); + CV_QUALIFIER SparseMatrix &ac(a); call_ref(ac); } diff --git a/filmulator-gui/core/nlmeans/eigen/failtest/sparse_ref_2.cpp b/filmulator-gui/core/nlmeans/eigen/failtest/sparse_ref_2.cpp index 46c9440c..3122f60a 100644 --- a/filmulator-gui/core/nlmeans/eigen/failtest/sparse_ref_2.cpp +++ b/filmulator-gui/core/nlmeans/eigen/failtest/sparse_ref_2.cpp @@ -2,11 +2,11 @@ using namespace Eigen; -void call_ref(Ref > a) { } +void call_ref(Ref> a) {} int main() { - SparseMatrix A(10,10); + SparseMatrix A(10, 10); #ifdef EIGEN_SHOULD_FAIL_TO_BUILD call_ref(A.row(3)); #else diff --git a/filmulator-gui/core/nlmeans/eigen/failtest/sparse_ref_3.cpp b/filmulator-gui/core/nlmeans/eigen/failtest/sparse_ref_3.cpp index a9949b55..75da6675 100644 --- a/filmulator-gui/core/nlmeans/eigen/failtest/sparse_ref_3.cpp +++ b/filmulator-gui/core/nlmeans/eigen/failtest/sparse_ref_3.cpp @@ -3,13 +3,13 @@ using namespace Eigen; #ifdef EIGEN_SHOULD_FAIL_TO_BUILD -void call_ref(Ref > a) { } +void call_ref(Ref> a) {} #else -void call_ref(const Ref > &a) { } +void call_ref(const Ref> &a) {} #endif int main() { - SparseMatrix a(10,10); - call_ref(a+a); + SparseMatrix a(10, 10); + call_ref(a + a); } diff --git a/filmulator-gui/core/nlmeans/eigen/failtest/sparse_ref_4.cpp b/filmulator-gui/core/nlmeans/eigen/failtest/sparse_ref_4.cpp index 57bb6a1f..94379f48 100644 --- a/filmulator-gui/core/nlmeans/eigen/failtest/sparse_ref_4.cpp +++ b/filmulator-gui/core/nlmeans/eigen/failtest/sparse_ref_4.cpp @@ -2,11 +2,11 @@ using namespace Eigen; -void call_ref(Ref > a) {} +void call_ref(Ref> a) {} int main() { - SparseMatrix A(10,10); + SparseMatrix A(10, 10); #ifdef EIGEN_SHOULD_FAIL_TO_BUILD call_ref(A.transpose()); #else diff --git a/filmulator-gui/core/nlmeans/eigen/failtest/sparse_ref_5.cpp b/filmulator-gui/core/nlmeans/eigen/failtest/sparse_ref_5.cpp index 4478f6f2..c78b52b5 100644 --- a/filmulator-gui/core/nlmeans/eigen/failtest/sparse_ref_5.cpp +++ b/filmulator-gui/core/nlmeans/eigen/failtest/sparse_ref_5.cpp @@ -2,12 +2,12 @@ using namespace Eigen; -void call_ref(Ref > a) { } +void call_ref(Ref> a) {} int main() { - SparseMatrix a(10,10); - SparseMatrixBase > &ac(a); + SparseMatrix a(10, 10); + SparseMatrixBase> &ac(a); #ifdef EIGEN_SHOULD_FAIL_TO_BUILD call_ref(ac); #else diff --git a/filmulator-gui/core/nlmeans/eigen/failtest/sparse_storage_mismatch.cpp b/filmulator-gui/core/nlmeans/eigen/failtest/sparse_storage_mismatch.cpp index 51840d41..73bd768a 100644 --- a/filmulator-gui/core/nlmeans/eigen/failtest/sparse_storage_mismatch.cpp +++ b/filmulator-gui/core/nlmeans/eigen/failtest/sparse_storage_mismatch.cpp @@ -1,16 +1,16 @@ #include "../Eigen/Sparse" using namespace Eigen; -typedef SparseMatrix Mat1; +typedef SparseMatrix Mat1; #ifdef EIGEN_SHOULD_FAIL_TO_BUILD -typedef SparseMatrix Mat2; +typedef SparseMatrix Mat2; #else -typedef SparseMatrix Mat2; +typedef SparseMatrix Mat2; #endif int main() { - Mat1 a(10,10); - Mat2 b(10,10); + Mat1 a(10, 10); + Mat2 b(10, 10); a += b; } diff --git a/filmulator-gui/core/nlmeans/eigen/failtest/ternary_1.cpp b/filmulator-gui/core/nlmeans/eigen/failtest/ternary_1.cpp index b40bcb0c..0aad2f0a 100644 --- a/filmulator-gui/core/nlmeans/eigen/failtest/ternary_1.cpp +++ b/filmulator-gui/core/nlmeans/eigen/failtest/ternary_1.cpp @@ -2,12 +2,12 @@ using namespace Eigen; -int main(int argc,char **) +int main(int argc, char **) { VectorXf a(10), b(10); #ifdef EIGEN_SHOULD_FAIL_TO_BUILD - b = argc>1 ? 2*a : -a; + b = argc > 1 ? 2 * a : -a; #else - b = argc>1 ? 2*a : VectorXf(-a); + b = argc > 1 ? 2 * a : VectorXf(-a); #endif } diff --git a/filmulator-gui/core/nlmeans/eigen/failtest/ternary_2.cpp b/filmulator-gui/core/nlmeans/eigen/failtest/ternary_2.cpp index a46b12b2..1d7ce988 100644 --- a/filmulator-gui/core/nlmeans/eigen/failtest/ternary_2.cpp +++ b/filmulator-gui/core/nlmeans/eigen/failtest/ternary_2.cpp @@ -2,12 +2,12 @@ using namespace Eigen; -int main(int argc,char **) +int main(int argc, char **) { VectorXf a(10), b(10); #ifdef EIGEN_SHOULD_FAIL_TO_BUILD - b = argc>1 ? 2*a : a+a; + b = argc > 1 ? 2 * a : a + a; #else - b = argc>1 ? VectorXf(2*a) : VectorXf(a+a); + b = argc > 1 ? VectorXf(2 * a) : VectorXf(a + a); #endif } diff --git a/filmulator-gui/core/nlmeans/eigen/failtest/transpose_nonconst_ctor_on_const_xpr.cpp b/filmulator-gui/core/nlmeans/eigen/failtest/transpose_nonconst_ctor_on_const_xpr.cpp index 4223e7fd..79d39aac 100644 --- a/filmulator-gui/core/nlmeans/eigen/failtest/transpose_nonconst_ctor_on_const_xpr.cpp +++ b/filmulator-gui/core/nlmeans/eigen/failtest/transpose_nonconst_ctor_on_const_xpr.cpp @@ -8,8 +8,6 @@ using namespace Eigen; -void foo(CV_QUALIFIER Matrix3d &m){ - Transpose t(m); -} +void foo(CV_QUALIFIER Matrix3d &m) { Transpose t(m); } int main() {} diff --git a/filmulator-gui/core/nlmeans/eigen/failtest/transpose_on_const_type_actually_const.cpp b/filmulator-gui/core/nlmeans/eigen/failtest/transpose_on_const_type_actually_const.cpp index d0b7d0df..e7a112af 100644 --- a/filmulator-gui/core/nlmeans/eigen/failtest/transpose_on_const_type_actually_const.cpp +++ b/filmulator-gui/core/nlmeans/eigen/failtest/transpose_on_const_type_actually_const.cpp @@ -8,9 +8,10 @@ using namespace Eigen; -void foo(){ - MatrixXf m; - Transpose(m).coeffRef(0, 0) = 1.0f; +void foo() +{ + MatrixXf m; + Transpose(m).coeffRef(0, 0) = 1.0f; } int main() {} diff --git a/filmulator-gui/core/nlmeans/eigen/failtest/triangularview_nonconst_ctor_on_const_xpr.cpp b/filmulator-gui/core/nlmeans/eigen/failtest/triangularview_nonconst_ctor_on_const_xpr.cpp index 807447e4..811ec3a6 100644 --- a/filmulator-gui/core/nlmeans/eigen/failtest/triangularview_nonconst_ctor_on_const_xpr.cpp +++ b/filmulator-gui/core/nlmeans/eigen/failtest/triangularview_nonconst_ctor_on_const_xpr.cpp @@ -8,8 +8,6 @@ using namespace Eigen; -void foo(CV_QUALIFIER Matrix3d &m){ - TriangularView t(m); -} +void foo(CV_QUALIFIER Matrix3d &m) { TriangularView t(m); } int main() {} diff --git a/filmulator-gui/core/nlmeans/eigen/failtest/triangularview_on_const_type_actually_const.cpp b/filmulator-gui/core/nlmeans/eigen/failtest/triangularview_on_const_type_actually_const.cpp index 0a381a61..92a4f192 100644 --- a/filmulator-gui/core/nlmeans/eigen/failtest/triangularview_on_const_type_actually_const.cpp +++ b/filmulator-gui/core/nlmeans/eigen/failtest/triangularview_on_const_type_actually_const.cpp @@ -8,9 +8,10 @@ using namespace Eigen; -void foo(){ - MatrixXf m; - TriangularView(m).coeffRef(0, 0) = 1.0f; +void foo() +{ + MatrixXf m; + TriangularView(m).coeffRef(0, 0) = 1.0f; } int main() {} diff --git a/filmulator-gui/core/nlmeans/eigen/lapack/cholesky.cpp b/filmulator-gui/core/nlmeans/eigen/lapack/cholesky.cpp index ea3bc123..56b938f5 100644 --- a/filmulator-gui/core/nlmeans/eigen/lapack/cholesky.cpp +++ b/filmulator-gui/core/nlmeans/eigen/lapack/cholesky.cpp @@ -11,59 +11,63 @@ #include // POTRF computes the Cholesky factorization of a real symmetric positive definite matrix A. -EIGEN_LAPACK_FUNC(potrf,(char* uplo, int *n, RealScalar *pa, int *lda, int *info)) +EIGEN_LAPACK_FUNC(potrf, (char *uplo, int *n, RealScalar *pa, int *lda, int *info)) { *info = 0; - if(UPLO(*uplo)==INVALID) *info = -1; - else if(*n<0) *info = -2; - else if(*lda(pa); - MatrixType A(a,*n,*n,*lda); + Scalar *a = reinterpret_cast(pa); + MatrixType A(a, *n, *n, *lda); int ret; - if(UPLO(*uplo)==UP) ret = int(internal::llt_inplace::blocked(A)); - else ret = int(internal::llt_inplace::blocked(A)); + if (UPLO(*uplo) == UP) + ret = int(internal::llt_inplace::blocked(A)); + else + ret = int(internal::llt_inplace::blocked(A)); + + if (ret >= 0) *info = ret + 1; - if(ret>=0) - *info = ret+1; - return 0; } // POTRS solves a system of linear equations A*X = B with a symmetric // positive definite matrix A using the Cholesky factorization // A = U**T*U or A = L*L**T computed by DPOTRF. -EIGEN_LAPACK_FUNC(potrs,(char* uplo, int *n, int *nrhs, RealScalar *pa, int *lda, RealScalar *pb, int *ldb, int *info)) +EIGEN_LAPACK_FUNC(potrs, (char *uplo, int *n, int *nrhs, RealScalar *pa, int *lda, RealScalar *pb, int *ldb, int *info)) { *info = 0; - if(UPLO(*uplo)==INVALID) *info = -1; - else if(*n<0) *info = -2; - else if(*nrhs<0) *info = -3; - else if(*lda(pa); - Scalar* b = reinterpret_cast(pb); - MatrixType A(a,*n,*n,*lda); - MatrixType B(b,*n,*nrhs,*ldb); + Scalar *a = reinterpret_cast(pa); + Scalar *b = reinterpret_cast(pb); + MatrixType A(a, *n, *n, *lda); + MatrixType B(b, *n, *nrhs, *ldb); - if(UPLO(*uplo)==UP) - { + if (UPLO(*uplo) == UP) { A.triangularView().adjoint().solveInPlace(B); A.triangularView().solveInPlace(B); - } - else - { + } else { A.triangularView().solveInPlace(B); A.triangularView().adjoint().solveInPlace(B); } diff --git a/filmulator-gui/core/nlmeans/eigen/lapack/complex_double.cpp b/filmulator-gui/core/nlmeans/eigen/lapack/complex_double.cpp index c9c57527..6d34b1c7 100644 --- a/filmulator-gui/core/nlmeans/eigen/lapack/complex_double.cpp +++ b/filmulator-gui/core/nlmeans/eigen/lapack/complex_double.cpp @@ -7,11 +7,11 @@ // Public License v. 2.0. If a copy of the MPL was not distributed // with this file, You can obtain one at http://mozilla.org/MPL/2.0/. -#define SCALAR std::complex +#define SCALAR std::complex #define SCALAR_SUFFIX z #define SCALAR_SUFFIX_UP "Z" #define REAL_SCALAR_SUFFIX d -#define ISCOMPLEX 1 +#define ISCOMPLEX 1 #include "cholesky.cpp" #include "lu.cpp" diff --git a/filmulator-gui/core/nlmeans/eigen/lapack/complex_single.cpp b/filmulator-gui/core/nlmeans/eigen/lapack/complex_single.cpp index 6d11b26c..3a99f73a 100644 --- a/filmulator-gui/core/nlmeans/eigen/lapack/complex_single.cpp +++ b/filmulator-gui/core/nlmeans/eigen/lapack/complex_single.cpp @@ -7,11 +7,11 @@ // Public License v. 2.0. If a copy of the MPL was not distributed // with this file, You can obtain one at http://mozilla.org/MPL/2.0/. -#define SCALAR std::complex +#define SCALAR std::complex #define SCALAR_SUFFIX c #define SCALAR_SUFFIX_UP "C" #define REAL_SCALAR_SUFFIX s -#define ISCOMPLEX 1 +#define ISCOMPLEX 1 #include "cholesky.cpp" #include "lu.cpp" diff --git a/filmulator-gui/core/nlmeans/eigen/lapack/double.cpp b/filmulator-gui/core/nlmeans/eigen/lapack/double.cpp index ea78bb66..ae4286e3 100644 --- a/filmulator-gui/core/nlmeans/eigen/lapack/double.cpp +++ b/filmulator-gui/core/nlmeans/eigen/lapack/double.cpp @@ -7,12 +7,12 @@ // Public License v. 2.0. If a copy of the MPL was not distributed // with this file, You can obtain one at http://mozilla.org/MPL/2.0/. -#define SCALAR double +#define SCALAR double #define SCALAR_SUFFIX d #define SCALAR_SUFFIX_UP "D" -#define ISCOMPLEX 0 +#define ISCOMPLEX 0 #include "cholesky.cpp" -#include "lu.cpp" #include "eigenvalues.cpp" +#include "lu.cpp" #include "svd.cpp" diff --git a/filmulator-gui/core/nlmeans/eigen/lapack/eigenvalues.cpp b/filmulator-gui/core/nlmeans/eigen/lapack/eigenvalues.cpp index 921c5156..fac8fce6 100644 --- a/filmulator-gui/core/nlmeans/eigen/lapack/eigenvalues.cpp +++ b/filmulator-gui/core/nlmeans/eigen/lapack/eigenvalues.cpp @@ -11,52 +11,54 @@ #include // computes eigen values and vectors of a general N-by-N matrix A -EIGEN_LAPACK_FUNC(syev,(char *jobz, char *uplo, int* n, Scalar* a, int *lda, Scalar* w, Scalar* /*work*/, int* lwork, int *info)) +EIGEN_LAPACK_FUNC(syev, + (char *jobz, char *uplo, int *n, Scalar *a, int *lda, Scalar *w, Scalar * /*work*/, int *lwork, int *info)) { // TODO exploit the work buffer - bool query_size = *lwork==-1; - + bool query_size = *lwork == -1; + *info = 0; - if(*jobz!='N' && *jobz!='V') *info = -1; - else if(UPLO(*uplo)==INVALID) *info = -2; - else if(*n<0) *info = -3; - else if(*lda eig(mat,computeVectors?ComputeEigenvectors:EigenvaluesOnly); - - if(eig.info()==NoConvergence) - { - make_vector(w,*n).setZero(); - if(computeVectors) - matrix(a,*n,*n,*lda).setIdentity(); + + if (*n == 0) return 0; + + PlainMatrixType mat(*n, *n); + if (UPLO(*uplo) == UP) + mat = matrix(a, *n, *n, *lda).adjoint(); + else + mat = matrix(a, *n, *n, *lda); + + bool computeVectors = *jobz == 'V' || *jobz == 'v'; + SelfAdjointEigenSolver eig(mat, computeVectors ? ComputeEigenvectors : EigenvaluesOnly); + + if (eig.info() == NoConvergence) { + make_vector(w, *n).setZero(); + if (computeVectors) matrix(a, *n, *n, *lda).setIdentity(); //*info = 1; return 0; } - - make_vector(w,*n) = eig.eigenvalues(); - if(computeVectors) - matrix(a,*n,*n,*lda) = eig.eigenvectors(); - + + make_vector(w, *n) = eig.eigenvalues(); + if (computeVectors) matrix(a, *n, *n, *lda) = eig.eigenvectors(); + return 0; } diff --git a/filmulator-gui/core/nlmeans/eigen/lapack/lapack_common.h b/filmulator-gui/core/nlmeans/eigen/lapack/lapack_common.h index c872a813..e0d65942 100644 --- a/filmulator-gui/core/nlmeans/eigen/lapack/lapack_common.h +++ b/filmulator-gui/core/nlmeans/eigen/lapack/lapack_common.h @@ -10,14 +10,17 @@ #ifndef EIGEN_LAPACK_COMMON_H #define EIGEN_LAPACK_COMMON_H -#include "../blas/common.h" #include "../Eigen/src/misc/lapack.h" +#include "../blas/common.h" -#define EIGEN_LAPACK_FUNC(FUNC,ARGLIST) \ - extern "C" { int EIGEN_BLAS_FUNC(FUNC) ARGLIST; } \ - int EIGEN_BLAS_FUNC(FUNC) ARGLIST +#define EIGEN_LAPACK_FUNC(FUNC, ARGLIST) \ + extern "C" { \ + int EIGEN_BLAS_FUNC(FUNC) ARGLIST; \ + } \ + int EIGEN_BLAS_FUNC(FUNC) \ + ARGLIST -typedef Eigen::Map > PivotsType; +typedef Eigen::Map> PivotsType; #if ISCOMPLEX #define EIGEN_LAPACK_ARG_IF_COMPLEX(X) X, @@ -26,4 +29,4 @@ typedef Eigen::Map > Pi #endif -#endif // EIGEN_LAPACK_COMMON_H +#endif// EIGEN_LAPACK_COMMON_H diff --git a/filmulator-gui/core/nlmeans/eigen/lapack/lu.cpp b/filmulator-gui/core/nlmeans/eigen/lapack/lu.cpp index 90cebe0f..f1efb3e1 100644 --- a/filmulator-gui/core/nlmeans/eigen/lapack/lu.cpp +++ b/filmulator-gui/core/nlmeans/eigen/lapack/lu.cpp @@ -11,79 +11,76 @@ #include // computes an LU factorization of a general M-by-N matrix A using partial pivoting with row interchanges -EIGEN_LAPACK_FUNC(getrf,(int *m, int *n, RealScalar *pa, int *lda, int *ipiv, int *info)) +EIGEN_LAPACK_FUNC(getrf, (int *m, int *n, RealScalar *pa, int *lda, int *ipiv, int *info)) { *info = 0; - if(*m<0) *info = -1; - else if(*n<0) *info = -2; - else if(*lda(pa); + Scalar *a = reinterpret_cast(pa); int nb_transpositions; - int ret = int(Eigen::internal::partial_lu_impl - ::blocked_lu(*m, *n, a, *lda, ipiv, nb_transpositions)); + int ret = + int(Eigen::internal::partial_lu_impl::blocked_lu(*m, *n, a, *lda, ipiv, nb_transpositions)); - for(int i=0; i=0) - *info = ret+1; + if (ret >= 0) *info = ret + 1; return 0; } -//GETRS solves a system of linear equations -// A * X = B or A' * X = B -// with a general N-by-N matrix A using the LU factorization computed by GETRF -EIGEN_LAPACK_FUNC(getrs,(char *trans, int *n, int *nrhs, RealScalar *pa, int *lda, int *ipiv, RealScalar *pb, int *ldb, int *info)) +// GETRS solves a system of linear equations +// A * X = B or A' * X = B +// with a general N-by-N matrix A using the LU factorization computed by GETRF +EIGEN_LAPACK_FUNC(getrs, + (char *trans, int *n, int *nrhs, RealScalar *pa, int *lda, int *ipiv, RealScalar *pb, int *ldb, int *info)) { *info = 0; - if(OP(*trans)==INVALID) *info = -1; - else if(*n<0) *info = -2; - else if(*nrhs<0) *info = -3; - else if(*lda(pa); - Scalar* b = reinterpret_cast(pb); - MatrixType lu(a,*n,*n,*lda); - MatrixType B(b,*n,*nrhs,*ldb); + Scalar *a = reinterpret_cast(pa); + Scalar *b = reinterpret_cast(pb); + MatrixType lu(a, *n, *n, *lda); + MatrixType B(b, *n, *nrhs, *ldb); - for(int i=0; i<*n; ++i) - ipiv[i]--; - if(OP(*trans)==NOTR) - { - B = PivotsType(ipiv,*n) * B; + for (int i = 0; i < *n; ++i) ipiv[i]--; + if (OP(*trans) == NOTR) { + B = PivotsType(ipiv, *n) * B; lu.triangularView().solveInPlace(B); lu.triangularView().solveInPlace(B); - } - else if(OP(*trans)==TR) - { + } else if (OP(*trans) == TR) { lu.triangularView().transpose().solveInPlace(B); lu.triangularView().transpose().solveInPlace(B); - B = PivotsType(ipiv,*n).transpose() * B; - } - else if(OP(*trans)==ADJ) - { + B = PivotsType(ipiv, *n).transpose() * B; + } else if (OP(*trans) == ADJ) { lu.triangularView().adjoint().solveInPlace(B); lu.triangularView().adjoint().solveInPlace(B); - B = PivotsType(ipiv,*n).transpose() * B; + B = PivotsType(ipiv, *n).transpose() * B; } - for(int i=0; i<*n; ++i) - ipiv[i]++; + for (int i = 0; i < *n; ++i) ipiv[i]++; return 0; } diff --git a/filmulator-gui/core/nlmeans/eigen/lapack/single.cpp b/filmulator-gui/core/nlmeans/eigen/lapack/single.cpp index c7da3eff..00c58e3b 100644 --- a/filmulator-gui/core/nlmeans/eigen/lapack/single.cpp +++ b/filmulator-gui/core/nlmeans/eigen/lapack/single.cpp @@ -7,12 +7,12 @@ // Public License v. 2.0. If a copy of the MPL was not distributed // with this file, You can obtain one at http://mozilla.org/MPL/2.0/. -#define SCALAR float +#define SCALAR float #define SCALAR_SUFFIX s #define SCALAR_SUFFIX_UP "S" -#define ISCOMPLEX 0 +#define ISCOMPLEX 0 #include "cholesky.cpp" -#include "lu.cpp" #include "eigenvalues.cpp" +#include "lu.cpp" #include "svd.cpp" diff --git a/filmulator-gui/core/nlmeans/eigen/lapack/svd.cpp b/filmulator-gui/core/nlmeans/eigen/lapack/svd.cpp index 77b302b6..51e1ba04 100644 --- a/filmulator-gui/core/nlmeans/eigen/lapack/svd.cpp +++ b/filmulator-gui/core/nlmeans/eigen/lapack/svd.cpp @@ -11,128 +11,161 @@ #include // computes the singular values/vectors a general M-by-N matrix A using divide-and-conquer -EIGEN_LAPACK_FUNC(gesdd,(char *jobz, int *m, int* n, Scalar* a, int *lda, RealScalar *s, Scalar *u, int *ldu, Scalar *vt, int *ldvt, Scalar* /*work*/, int* lwork, - EIGEN_LAPACK_ARG_IF_COMPLEX(RealScalar */*rwork*/) int * /*iwork*/, int *info)) +EIGEN_LAPACK_FUNC(gesdd, + (char *jobz, + int *m, + int *n, + Scalar *a, + int *lda, + RealScalar *s, + Scalar *u, + int *ldu, + Scalar *vt, + int *ldvt, + Scalar * /*work*/, + int *lwork, + EIGEN_LAPACK_ARG_IF_COMPLEX(RealScalar * /*rwork*/) int * /*iwork*/, + int *info)) { // TODO exploit the work buffer - bool query_size = *lwork==-1; - int diag_size = (std::min)(*m,*n); - + bool query_size = *lwork == -1; + int diag_size = (std::min)(*m, *n); + *info = 0; - if(*jobz!='A' && *jobz!='S' && *jobz!='O' && *jobz!='N') *info = -1; - else if(*m<0) *info = -2; - else if(*n<0) *info = -3; - else if(*lda=*n && *ldvt<*n)) *info = -10; - - if(*info!=0) - { + if (*jobz != 'A' && *jobz != 'S' && *jobz != 'O' && *jobz != 'N') + *info = -1; + else if (*m < 0) + *info = -2; + else if (*n < 0) + *info = -3; + else if (*lda < std::max(1, *m)) + *info = -5; + else if (*lda < std::max(1, *m)) + *info = -8; + else if (*ldu < 1 || (*jobz == 'A' && *ldu < *m) || (*jobz == 'O' && *m < *n && *ldu < *m)) + *info = -8; + else if (*ldvt < 1 || (*jobz == 'A' && *ldvt < *n) || (*jobz == 'S' && *ldvt < diag_size) + || (*jobz == 'O' && *m >= *n && *ldvt < *n)) + *info = -10; + + if (*info != 0) { int e = -*info; - return xerbla_(SCALAR_SUFFIX_UP"GESDD ", &e, 6); + return xerbla_(SCALAR_SUFFIX_UP "GESDD ", &e, 6); } - - if(query_size) - { + + if (query_size) { *lwork = 0; return 0; } - - if(*n==0 || *m==0) - return 0; - - PlainMatrixType mat(*m,*n); - mat = matrix(a,*m,*n,*lda); - - int option = *jobz=='A' ? ComputeFullU|ComputeFullV - : *jobz=='S' ? ComputeThinU|ComputeThinV - : *jobz=='O' ? ComputeThinU|ComputeThinV - : 0; - - BDCSVD svd(mat,option); - - make_vector(s,diag_size) = svd.singularValues().head(diag_size); - - if(*jobz=='A') - { - matrix(u,*m,*m,*ldu) = svd.matrixU(); - matrix(vt,*n,*n,*ldvt) = svd.matrixV().adjoint(); - } - else if(*jobz=='S') - { - matrix(u,*m,diag_size,*ldu) = svd.matrixU(); - matrix(vt,diag_size,*n,*ldvt) = svd.matrixV().adjoint(); - } - else if(*jobz=='O' && *m>=*n) - { - matrix(a,*m,*n,*lda) = svd.matrixU(); - matrix(vt,*n,*n,*ldvt) = svd.matrixV().adjoint(); - } - else if(*jobz=='O') - { - matrix(u,*m,*m,*ldu) = svd.matrixU(); - matrix(a,diag_size,*n,*lda) = svd.matrixV().adjoint(); + + if (*n == 0 || *m == 0) return 0; + + PlainMatrixType mat(*m, *n); + mat = matrix(a, *m, *n, *lda); + + int option = *jobz == 'A' ? ComputeFullU | ComputeFullV + : *jobz == 'S' ? ComputeThinU | ComputeThinV + : *jobz == 'O' ? ComputeThinU | ComputeThinV + : 0; + + BDCSVD svd(mat, option); + + make_vector(s, diag_size) = svd.singularValues().head(diag_size); + + if (*jobz == 'A') { + matrix(u, *m, *m, *ldu) = svd.matrixU(); + matrix(vt, *n, *n, *ldvt) = svd.matrixV().adjoint(); + } else if (*jobz == 'S') { + matrix(u, *m, diag_size, *ldu) = svd.matrixU(); + matrix(vt, diag_size, *n, *ldvt) = svd.matrixV().adjoint(); + } else if (*jobz == 'O' && *m >= *n) { + matrix(a, *m, *n, *lda) = svd.matrixU(); + matrix(vt, *n, *n, *ldvt) = svd.matrixV().adjoint(); + } else if (*jobz == 'O') { + matrix(u, *m, *m, *ldu) = svd.matrixU(); + matrix(a, diag_size, *n, *lda) = svd.matrixV().adjoint(); } - + return 0; } // computes the singular values/vectors a general M-by-N matrix A using two sided jacobi algorithm -EIGEN_LAPACK_FUNC(gesvd,(char *jobu, char *jobv, int *m, int* n, Scalar* a, int *lda, RealScalar *s, Scalar *u, int *ldu, Scalar *vt, int *ldvt, Scalar* /*work*/, int* lwork, - EIGEN_LAPACK_ARG_IF_COMPLEX(RealScalar */*rwork*/) int *info)) +EIGEN_LAPACK_FUNC(gesvd, + (char *jobu, + char *jobv, + int *m, + int *n, + Scalar *a, + int *lda, + RealScalar *s, + Scalar *u, + int *ldu, + Scalar *vt, + int *ldvt, + Scalar * /*work*/, + int *lwork, + EIGEN_LAPACK_ARG_IF_COMPLEX(RealScalar * /*rwork*/) int *info)) { // TODO exploit the work buffer - bool query_size = *lwork==-1; - int diag_size = (std::min)(*m,*n); - + bool query_size = *lwork == -1; + int diag_size = (std::min)(*m, *n); + *info = 0; - if( *jobu!='A' && *jobu!='S' && *jobu!='O' && *jobu!='N') *info = -1; - else if((*jobv!='A' && *jobv!='S' && *jobv!='O' && *jobv!='N') - || (*jobu=='O' && *jobv=='O')) *info = -2; - else if(*m<0) *info = -3; - else if(*n<0) *info = -4; - else if(*lda svd(mat,option); - - make_vector(s,diag_size) = svd.singularValues().head(diag_size); + + if (*n == 0 || *m == 0) return 0; + + PlainMatrixType mat(*m, *n); + mat = matrix(a, *m, *n, *lda); + + int option = (*jobu == 'A' ? ComputeFullU + : *jobu == 'S' || *jobu == 'O' ? ComputeThinU + : 0) + | (*jobv == 'A' ? ComputeFullV + : *jobv == 'S' || *jobv == 'O' ? ComputeThinV + : 0); + + JacobiSVD svd(mat, option); + + make_vector(s, diag_size) = svd.singularValues().head(diag_size); { - if(*jobu=='A') matrix(u,*m,*m,*ldu) = svd.matrixU(); - else if(*jobu=='S') matrix(u,*m,diag_size,*ldu) = svd.matrixU(); - else if(*jobu=='O') matrix(a,*m,diag_size,*lda) = svd.matrixU(); + if (*jobu == 'A') + matrix(u, *m, *m, *ldu) = svd.matrixU(); + else if (*jobu == 'S') + matrix(u, *m, diag_size, *ldu) = svd.matrixU(); + else if (*jobu == 'O') + matrix(a, *m, diag_size, *lda) = svd.matrixU(); } { - if(*jobv=='A') matrix(vt,*n,*n,*ldvt) = svd.matrixV().adjoint(); - else if(*jobv=='S') matrix(vt,diag_size,*n,*ldvt) = svd.matrixV().adjoint(); - else if(*jobv=='O') matrix(a,diag_size,*n,*lda) = svd.matrixV().adjoint(); + if (*jobv == 'A') + matrix(vt, *n, *n, *ldvt) = svd.matrixV().adjoint(); + else if (*jobv == 'S') + matrix(vt, diag_size, *n, *ldvt) = svd.matrixV().adjoint(); + else if (*jobv == 'O') + matrix(a, diag_size, *n, *lda) = svd.matrixV().adjoint(); } return 0; } diff --git a/filmulator-gui/core/nlmeans/eigen/scripts/eigen_gen_credits.cpp b/filmulator-gui/core/nlmeans/eigen/scripts/eigen_gen_credits.cpp index f2e81631..f2484395 100644 --- a/filmulator-gui/core/nlmeans/eigen/scripts/eigen_gen_credits.cpp +++ b/filmulator-gui/core/nlmeans/eigen/scripts/eigen_gen_credits.cpp @@ -1,56 +1,46 @@ -#include -#include -#include #include #include -#include +#include #include +#include +#include +#include using namespace std; // this function takes a line that may contain a name and/or email address, // and returns just the name, while fixing the "bad cases". -std::string contributor_name(const std::string& line) +std::string contributor_name(const std::string &line) { string result; // let's first take care of the case of isolated email addresses, like // "user@localhost.localdomain" entries - if(line.find("markb@localhost.localdomain") != string::npos) - { - return "Mark Borgerding"; - } + if (line.find("markb@localhost.localdomain") != string::npos) { return "Mark Borgerding"; } - if(line.find("kayhman@contact.intra.cea.fr") != string::npos) - { - return "Guillaume Saupin"; - } + if (line.find("kayhman@contact.intra.cea.fr") != string::npos) { return "Guillaume Saupin"; } // from there on we assume that we have a entry of the form // either: // Bla bli Blurp // or: // Bla bli Blurp - + size_t position_of_email_address = line.find_first_of('<'); - if(position_of_email_address != string::npos) - { + if (position_of_email_address != string::npos) { // there is an e-mail address in <...>. - + // Hauke once committed as "John Smith", fix that. - if(line.find("hauke.heibel") != string::npos) + if (line.find("hauke.heibel") != string::npos) result = "Hauke Heibel"; - else - { + else { // just remove the e-mail address result = line.substr(0, position_of_email_address); } - } - else - { + } else { // there is no e-mail address in <...>. - - if(line.find("convert-repo") != string::npos) + + if (line.find("convert-repo") != string::npos) result = ""; else result = line; @@ -58,44 +48,42 @@ std::string contributor_name(const std::string& line) // remove trailing spaces size_t length = result.length(); - while(length >= 1 && result[length-1] == ' ') result.erase(--length); + while (length >= 1 && result[length - 1] == ' ') result.erase(--length); return result; } // parses hg churn output to generate a contributors map. -map contributors_map_from_churn_output(const char *filename) +map contributors_map_from_churn_output(const char *filename) { - map contributors_map; + map contributors_map; string line; ifstream churn_out; churn_out.open(filename, ios::in); - while(!getline(churn_out,line).eof()) - { + while (!getline(churn_out, line).eof()) { // remove the histograms "******" that hg churn may draw at the end of some lines size_t first_star = line.find_first_of('*'); - if(first_star != string::npos) line.erase(first_star); - + if (first_star != string::npos) line.erase(first_star); + // remove trailing spaces size_t length = line.length(); - while(length >= 1 && line[length-1] == ' ') line.erase(--length); + while (length >= 1 && line[length - 1] == ' ') line.erase(--length); // now the last space indicates where the number starts size_t last_space = line.find_last_of(' '); - + // get the number (of changesets or of modified lines for each contributor) int number; - istringstream(line.substr(last_space+1)) >> number; + istringstream(line.substr(last_space + 1)) >> number; // get the name of the contributor - line.erase(last_space); + line.erase(last_space); string name = contributor_name(line); - - map::iterator it = contributors_map.find(name); + + map::iterator it = contributors_map.find(name); // if new contributor, insert - if(it == contributors_map.end()) - contributors_map.insert(pair(name, number)); + if (it == contributors_map.end()) contributors_map.insert(pair(name, number)); // if duplicate, just add the number else it->second += number; @@ -107,11 +95,13 @@ map contributors_map_from_churn_output(const char *filename) // find the last name, i.e. the last word. // for "van den Schbling" types of last names, that's not a problem, that's actually what we want. -string lastname(const string& name) +string lastname(const string &name) { size_t last_space = name.find_last_of(' '); - if(last_space >= name.length()-1) return name; - else return name.substr(last_space+1); + if (last_space >= name.length() - 1) + return name; + else + return name.substr(last_space + 1); } struct contributor @@ -121,61 +111,50 @@ struct contributor int changesets; string url; string misc; - + contributor() : changedlines(0), changesets(0) {} - - bool operator < (const contributor& other) - { - return lastname(name).compare(lastname(other.name)) < 0; - } + + bool operator<(const contributor &other) { return lastname(name).compare(lastname(other.name)) < 0; } }; -void add_online_info_into_contributors_list(list& contributors_list, const char *filename) +void add_online_info_into_contributors_list(list &contributors_list, const char *filename) { string line; ifstream online_info; online_info.open(filename, ios::in); - while(!getline(online_info,line).eof()) - { + while (!getline(online_info, line).eof()) { string hgname, realname, url, misc; - + size_t last_bar = line.find_last_of('|'); - if(last_bar == string::npos) continue; - if(last_bar < line.length()) - misc = line.substr(last_bar+1); + if (last_bar == string::npos) continue; + if (last_bar < line.length()) misc = line.substr(last_bar + 1); line.erase(last_bar); - + last_bar = line.find_last_of('|'); - if(last_bar == string::npos) continue; - if(last_bar < line.length()) - url = line.substr(last_bar+1); + if (last_bar == string::npos) continue; + if (last_bar < line.length()) url = line.substr(last_bar + 1); line.erase(last_bar); last_bar = line.find_last_of('|'); - if(last_bar == string::npos) continue; - if(last_bar < line.length()) - realname = line.substr(last_bar+1); + if (last_bar == string::npos) continue; + if (last_bar < line.length()) realname = line.substr(last_bar + 1); line.erase(last_bar); hgname = line; - + // remove the example line - if(hgname.find("MercurialName") != string::npos) continue; - + if (hgname.find("MercurialName") != string::npos) continue; + list::iterator it; - for(it=contributors_list.begin(); it != contributors_list.end() && it->name != hgname; ++it) - {} - - if(it == contributors_list.end()) - { + for (it = contributors_list.begin(); it != contributors_list.end() && it->name != hgname; ++it) {} + + if (it == contributors_list.end()) { contributor c; c.name = realname; c.url = url; c.misc = misc; contributors_list.push_back(c); - } - else - { + } else { it->name = realname; it->url = url; it->misc = misc; @@ -186,25 +165,24 @@ void add_online_info_into_contributors_list(list& contributors_list int main() { // parse the hg churn output files - map contributors_map_for_changedlines = contributors_map_from_churn_output("churn-changedlines.out"); - //map contributors_map_for_changesets = contributors_map_from_churn_output("churn-changesets.out"); - + map contributors_map_for_changedlines = contributors_map_from_churn_output("churn-changedlines.out"); + // map contributors_map_for_changesets = contributors_map_from_churn_output("churn-changesets.out"); + // merge into the contributors list list contributors_list; - map::iterator it; - for(it=contributors_map_for_changedlines.begin(); it != contributors_map_for_changedlines.end(); ++it) - { + map::iterator it; + for (it = contributors_map_for_changedlines.begin(); it != contributors_map_for_changedlines.end(); ++it) { contributor c; c.name = it->first; c.changedlines = it->second; - c.changesets = 0; //contributors_map_for_changesets.find(it->first)->second; + c.changesets = 0;// contributors_map_for_changesets.find(it->first)->second; contributors_list.push_back(c); } - + add_online_info_into_contributors_list(contributors_list, "online-info.out"); - + contributors_list.sort(); - + cout << "{| cellpadding=\"5\"\n"; cout << "!\n"; cout << "! Lines changed\n"; @@ -212,16 +190,17 @@ int main() list::iterator itc; int i = 0; - for(itc=contributors_list.begin(); itc != contributors_list.end(); ++itc) - { - if(itc->name.length() == 0) continue; - if(i%2) cout << "|-\n"; - else cout << "|- style=\"background:#FFFFD0\"\n"; - if(itc->url.length()) + for (itc = contributors_list.begin(); itc != contributors_list.end(); ++itc) { + if (itc->name.length() == 0) continue; + if (i % 2) + cout << "|-\n"; + else + cout << "|- style=\"background:#FFFFD0\"\n"; + if (itc->url.length()) cout << "| [" << itc->url << " " << itc->name << "]\n"; else cout << "| " << itc->name << "\n"; - if(itc->changedlines) + if (itc->changedlines) cout << "| " << itc->changedlines << "\n"; else cout << "| (no information)\n"; diff --git a/filmulator-gui/core/nlmeans/eigen/test/adjoint.cpp b/filmulator-gui/core/nlmeans/eigen/test/adjoint.cpp index 37032d22..06d5a3f9 100644 --- a/filmulator-gui/core/nlmeans/eigen/test/adjoint.cpp +++ b/filmulator-gui/core/nlmeans/eigen/test/adjoint.cpp @@ -13,28 +13,34 @@ template struct adjoint_specific; -template<> struct adjoint_specific { +template<> struct adjoint_specific +{ template - static void run(const Vec& v1, const Vec& v2, Vec& v3, const Mat& square, Scalar s1, Scalar s2) { - VERIFY(test_isApproxWithRef((s1 * v1 + s2 * v2).dot(v3), numext::conj(s1) * v1.dot(v3) + numext::conj(s2) * v2.dot(v3), 0)); - VERIFY(test_isApproxWithRef(v3.dot(s1 * v1 + s2 * v2), s1*v3.dot(v1)+s2*v3.dot(v2), 0)); - + static void run(const Vec &v1, const Vec &v2, Vec &v3, const Mat &square, Scalar s1, Scalar s2) + { + VERIFY(test_isApproxWithRef( + (s1 * v1 + s2 * v2).dot(v3), numext::conj(s1) * v1.dot(v3) + numext::conj(s2) * v2.dot(v3), 0)); + VERIFY(test_isApproxWithRef(v3.dot(s1 * v1 + s2 * v2), s1 * v3.dot(v1) + s2 * v3.dot(v2), 0)); + // check compatibility of dot and adjoint VERIFY(test_isApproxWithRef(v1.dot(square * v2), (square.adjoint() * v1).dot(v2), 0)); } }; -template<> struct adjoint_specific { +template<> struct adjoint_specific +{ template - static void run(const Vec& v1, const Vec& v2, Vec& v3, const Mat& square, Scalar s1, Scalar s2) { + static void run(const Vec &v1, const Vec &v2, Vec &v3, const Mat &square, Scalar s1, Scalar s2) + { typedef typename NumTraits::Real RealScalar; using std::abs; - - RealScalar ref = NumTraits::IsInteger ? RealScalar(0) : (std::max)((s1 * v1 + s2 * v2).norm(),v3.norm()); - VERIFY(test_isApproxWithRef((s1 * v1 + s2 * v2).dot(v3), numext::conj(s1) * v1.dot(v3) + numext::conj(s2) * v2.dot(v3), ref)); - VERIFY(test_isApproxWithRef(v3.dot(s1 * v1 + s2 * v2), s1*v3.dot(v1)+s2*v3.dot(v2), ref)); - - VERIFY_IS_APPROX(v1.squaredNorm(), v1.norm() * v1.norm()); + + RealScalar ref = NumTraits::IsInteger ? RealScalar(0) : (std::max)((s1 * v1 + s2 * v2).norm(), v3.norm()); + VERIFY(test_isApproxWithRef( + (s1 * v1 + s2 * v2).dot(v3), numext::conj(s1) * v1.dot(v3) + numext::conj(s2) * v2.dot(v3), ref)); + VERIFY(test_isApproxWithRef(v3.dot(s1 * v1 + s2 * v2), s1 * v3.dot(v1) + s2 * v3.dot(v2), ref)); + + VERIFY_IS_APPROX(v1.squaredNorm(), v1.norm() * v1.norm()); // check normalized() and normalize() VERIFY_IS_APPROX(v1, v1.norm() * v1.normalized()); v3 = v1; @@ -44,27 +50,30 @@ template<> struct adjoint_specific { VERIFY_IS_APPROX(v3.norm(), RealScalar(1)); // check null inputs - VERIFY_IS_APPROX((v1*0).normalized(), (v1*0)); + VERIFY_IS_APPROX((v1 * 0).normalized(), (v1 * 0)); #if (!EIGEN_ARCH_i386) || defined(EIGEN_VECTORIZE) RealScalar very_small = (std::numeric_limits::min)(); - VERIFY( (v1*very_small).norm() == 0 ); - VERIFY_IS_APPROX((v1*very_small).normalized(), (v1*very_small)); - v3 = v1*very_small; + VERIFY((v1 * very_small).norm() == 0); + VERIFY_IS_APPROX((v1 * very_small).normalized(), (v1 * very_small)); + v3 = v1 * very_small; v3.normalize(); - VERIFY_IS_APPROX(v3, (v1*very_small)); + VERIFY_IS_APPROX(v3, (v1 * very_small)); #endif - + // check compatibility of dot and adjoint - ref = NumTraits::IsInteger ? 0 : (std::max)((std::max)(v1.norm(),v2.norm()),(std::max)((square * v2).norm(),(square.adjoint() * v1).norm())); - VERIFY(internal::isMuchSmallerThan(abs(v1.dot(square * v2) - (square.adjoint() * v1).dot(v2)), ref, test_precision())); - + ref = NumTraits::IsInteger ? 0 + : (std::max)((std::max)(v1.norm(), v2.norm()), + (std::max)((square * v2).norm(), (square.adjoint() * v1).norm())); + VERIFY(internal::isMuchSmallerThan( + abs(v1.dot(square * v2) - (square.adjoint() * v1).dot(v2)), ref, test_precision())); + // check that Random().normalized() works: tricky as the random xpr must be evaluated by // normalized() in order to produce a consistent result. VERIFY_IS_APPROX(Vec::Random(v1.size()).normalized().norm(), RealScalar(1)); } }; -template void adjoint(const MatrixType& m) +template void adjoint(const MatrixType &m) { /* this test covers the following files: Transpose.h Conjugate.h Dot.h @@ -75,68 +84,62 @@ template void adjoint(const MatrixType& m) typedef Matrix VectorType; typedef Matrix SquareMatrixType; const Index PacketSize = internal::packet_traits::size; - + Index rows = m.rows(); Index cols = m.cols(); - MatrixType m1 = MatrixType::Random(rows, cols), - m2 = MatrixType::Random(rows, cols), - m3(rows, cols), + MatrixType m1 = MatrixType::Random(rows, cols), m2 = MatrixType::Random(rows, cols), m3(rows, cols), square = SquareMatrixType::Random(rows, rows); - VectorType v1 = VectorType::Random(rows), - v2 = VectorType::Random(rows), - v3 = VectorType::Random(rows), + VectorType v1 = VectorType::Random(rows), v2 = VectorType::Random(rows), v3 = VectorType::Random(rows), vzero = VectorType::Zero(rows); - Scalar s1 = internal::random(), - s2 = internal::random(); + Scalar s1 = internal::random(), s2 = internal::random(); // check basic compatibility of adjoint, transpose, conjugate - VERIFY_IS_APPROX(m1.transpose().conjugate().adjoint(), m1); - VERIFY_IS_APPROX(m1.adjoint().conjugate().transpose(), m1); + VERIFY_IS_APPROX(m1.transpose().conjugate().adjoint(), m1); + VERIFY_IS_APPROX(m1.adjoint().conjugate().transpose(), m1); // check multiplicative behavior - VERIFY_IS_APPROX((m1.adjoint() * m2).adjoint(), m2.adjoint() * m1); - VERIFY_IS_APPROX((s1 * m1).adjoint(), numext::conj(s1) * m1.adjoint()); + VERIFY_IS_APPROX((m1.adjoint() * m2).adjoint(), m2.adjoint() * m1); + VERIFY_IS_APPROX((s1 * m1).adjoint(), numext::conj(s1) * m1.adjoint()); // check basic properties of dot, squaredNorm - VERIFY_IS_APPROX(numext::conj(v1.dot(v2)), v2.dot(v1)); - VERIFY_IS_APPROX(numext::real(v1.dot(v1)), v1.squaredNorm()); - + VERIFY_IS_APPROX(numext::conj(v1.dot(v2)), v2.dot(v1)); + VERIFY_IS_APPROX(numext::real(v1.dot(v1)), v1.squaredNorm()); + adjoint_specific::IsInteger>::run(v1, v2, v3, square, s1, s2); - - VERIFY_IS_MUCH_SMALLER_THAN(abs(vzero.dot(v1)), static_cast(1)); - + + VERIFY_IS_MUCH_SMALLER_THAN(abs(vzero.dot(v1)), static_cast(1)); + // like in testBasicStuff, test operator() to check const-qualification - Index r = internal::random(0, rows-1), - c = internal::random(0, cols-1); - VERIFY_IS_APPROX(m1.conjugate()(r,c), numext::conj(m1(r,c))); - VERIFY_IS_APPROX(m1.adjoint()(c,r), numext::conj(m1(r,c))); + Index r = internal::random(0, rows - 1), c = internal::random(0, cols - 1); + VERIFY_IS_APPROX(m1.conjugate()(r, c), numext::conj(m1(r, c))); + VERIFY_IS_APPROX(m1.adjoint()(c, r), numext::conj(m1(r, c))); // check inplace transpose m3 = m1; m3.transposeInPlace(); - VERIFY_IS_APPROX(m3,m1.transpose()); + VERIFY_IS_APPROX(m3, m1.transpose()); m3.transposeInPlace(); - VERIFY_IS_APPROX(m3,m1); - - if(PacketSize(0,m3.rows()-PacketSize); - Index j = internal::random(0,m3.cols()-PacketSize); - m3.template block(i,j).transposeInPlace(); - VERIFY_IS_APPROX( (m3.template block(i,j)), (m1.template block(i,j).transpose()) ); - m3.template block(i,j).transposeInPlace(); - VERIFY_IS_APPROX(m3,m1); + Index i = internal::random(0, m3.rows() - PacketSize); + Index j = internal::random(0, m3.cols() - PacketSize); + m3.template block(i, j).transposeInPlace(); + VERIFY_IS_APPROX( + (m3.template block(i, j)), (m1.template block(i, j).transpose())); + m3.template block(i, j).transposeInPlace(); + VERIFY_IS_APPROX(m3, m1); } // check inplace adjoint m3 = m1; m3.adjointInPlace(); - VERIFY_IS_APPROX(m3,m1.adjoint()); + VERIFY_IS_APPROX(m3, m1.adjoint()); m3.transposeInPlace(); - VERIFY_IS_APPROX(m3,m1.conjugate()); + VERIFY_IS_APPROX(m3, m1.conjugate()); // check mixed dot product typedef Matrix RealVectorType; @@ -147,30 +150,33 @@ template void adjoint(const MatrixType& m) void test_adjoint() { - for(int i = 0; i < g_repeat; i++) { - CALL_SUBTEST_1( adjoint(Matrix()) ); - CALL_SUBTEST_2( adjoint(Matrix3d()) ); - CALL_SUBTEST_3( adjoint(Matrix4f()) ); - - CALL_SUBTEST_4( adjoint(MatrixXcf(internal::random(1,EIGEN_TEST_MAX_SIZE/2), internal::random(1,EIGEN_TEST_MAX_SIZE/2))) ); - CALL_SUBTEST_5( adjoint(MatrixXi(internal::random(1,EIGEN_TEST_MAX_SIZE), internal::random(1,EIGEN_TEST_MAX_SIZE))) ); - CALL_SUBTEST_6( adjoint(MatrixXf(internal::random(1,EIGEN_TEST_MAX_SIZE), internal::random(1,EIGEN_TEST_MAX_SIZE))) ); - + for (int i = 0; i < g_repeat; i++) { + CALL_SUBTEST_1(adjoint(Matrix())); + CALL_SUBTEST_2(adjoint(Matrix3d())); + CALL_SUBTEST_3(adjoint(Matrix4f())); + + CALL_SUBTEST_4(adjoint( + MatrixXcf(internal::random(1, EIGEN_TEST_MAX_SIZE / 2), internal::random(1, EIGEN_TEST_MAX_SIZE / 2)))); + CALL_SUBTEST_5( + adjoint(MatrixXi(internal::random(1, EIGEN_TEST_MAX_SIZE), internal::random(1, EIGEN_TEST_MAX_SIZE)))); + CALL_SUBTEST_6( + adjoint(MatrixXf(internal::random(1, EIGEN_TEST_MAX_SIZE), internal::random(1, EIGEN_TEST_MAX_SIZE)))); + // Complement for 128 bits vectorization: - CALL_SUBTEST_8( adjoint(Matrix2d()) ); - CALL_SUBTEST_9( adjoint(Matrix()) ); - + CALL_SUBTEST_8(adjoint(Matrix2d())); + CALL_SUBTEST_9(adjoint(Matrix())); + // 256 bits vectorization: - CALL_SUBTEST_10( adjoint(Matrix()) ); - CALL_SUBTEST_11( adjoint(Matrix()) ); - CALL_SUBTEST_12( adjoint(Matrix()) ); + CALL_SUBTEST_10(adjoint(Matrix())); + CALL_SUBTEST_11(adjoint(Matrix())); + CALL_SUBTEST_12(adjoint(Matrix())); } // test a large static matrix only once - CALL_SUBTEST_7( adjoint(Matrix()) ); + CALL_SUBTEST_7(adjoint(Matrix())); #ifdef EIGEN_TEST_PART_13 { - MatrixXcf a(10,10), b(10,10); + MatrixXcf a(10, 10), b(10, 10); VERIFY_RAISES_ASSERT(a = a.transpose()); VERIFY_RAISES_ASSERT(a = a.transpose() + b); VERIFY_RAISES_ASSERT(a = b + a.transpose()); @@ -188,12 +194,11 @@ void test_adjoint() a.transpose() += a.adjoint() + b; // regression tests for check_for_aliasing - MatrixXd c(10,10); - c = 1.0 * MatrixXd::Ones(10,10) + c; - c = MatrixXd::Ones(10,10) * 1.0 + c; - c = c + MatrixXd::Ones(10,10) .cwiseProduct( MatrixXd::Zero(10,10) ); - c = MatrixXd::Ones(10,10) * MatrixXd::Zero(10,10); + MatrixXd c(10, 10); + c = 1.0 * MatrixXd::Ones(10, 10) + c; + c = MatrixXd::Ones(10, 10) * 1.0 + c; + c = c + MatrixXd::Ones(10, 10).cwiseProduct(MatrixXd::Zero(10, 10)); + c = MatrixXd::Ones(10, 10) * MatrixXd::Zero(10, 10); } #endif } - diff --git a/filmulator-gui/core/nlmeans/eigen/test/array.cpp b/filmulator-gui/core/nlmeans/eigen/test/array.cpp index 7afd3ed3..9c2eaa93 100644 --- a/filmulator-gui/core/nlmeans/eigen/test/array.cpp +++ b/filmulator-gui/core/nlmeans/eigen/test/array.cpp @@ -9,7 +9,7 @@ #include "main.h" -template void array(const ArrayType& m) +template void array(const ArrayType &m) { typedef typename ArrayType::Scalar Scalar; typedef typename ArrayType::RealScalar RealScalar; @@ -17,51 +17,48 @@ template void array(const ArrayType& m) typedef Array RowVectorType; Index rows = m.rows(); - Index cols = m.cols(); + Index cols = m.cols(); - ArrayType m1 = ArrayType::Random(rows, cols), - m2 = ArrayType::Random(rows, cols), - m3(rows, cols); - ArrayType m4 = m1; // copy constructor + ArrayType m1 = ArrayType::Random(rows, cols), m2 = ArrayType::Random(rows, cols), m3(rows, cols); + ArrayType m4 = m1;// copy constructor VERIFY_IS_APPROX(m1, m4); ColVectorType cv1 = ColVectorType::Random(rows); RowVectorType rv1 = RowVectorType::Random(cols); - Scalar s1 = internal::random(), - s2 = internal::random(); + Scalar s1 = internal::random(), s2 = internal::random(); // scalar addition VERIFY_IS_APPROX(m1 + s1, s1 + m1); - VERIFY_IS_APPROX(m1 + s1, ArrayType::Constant(rows,cols,s1) + m1); - VERIFY_IS_APPROX(s1 - m1, (-m1)+s1 ); - VERIFY_IS_APPROX(m1 - s1, m1 - ArrayType::Constant(rows,cols,s1)); - VERIFY_IS_APPROX(s1 - m1, ArrayType::Constant(rows,cols,s1) - m1); - VERIFY_IS_APPROX((m1*Scalar(2)) - s2, (m1+m1) - ArrayType::Constant(rows,cols,s2) ); + VERIFY_IS_APPROX(m1 + s1, ArrayType::Constant(rows, cols, s1) + m1); + VERIFY_IS_APPROX(s1 - m1, (-m1) + s1); + VERIFY_IS_APPROX(m1 - s1, m1 - ArrayType::Constant(rows, cols, s1)); + VERIFY_IS_APPROX(s1 - m1, ArrayType::Constant(rows, cols, s1) - m1); + VERIFY_IS_APPROX((m1 * Scalar(2)) - s2, (m1 + m1) - ArrayType::Constant(rows, cols, s2)); m3 = m1; m3 += s2; VERIFY_IS_APPROX(m3, m1 + s2); m3 = m1; m3 -= s1; - VERIFY_IS_APPROX(m3, m1 - s1); - + VERIFY_IS_APPROX(m3, m1 - s1); + // scalar operators via Maps m3 = m1; ArrayType::Map(m1.data(), m1.rows(), m1.cols()) -= ArrayType::Map(m2.data(), m2.rows(), m2.cols()); VERIFY_IS_APPROX(m1, m3 - m2); - + m3 = m1; ArrayType::Map(m1.data(), m1.rows(), m1.cols()) += ArrayType::Map(m2.data(), m2.rows(), m2.cols()); VERIFY_IS_APPROX(m1, m3 + m2); - + m3 = m1; ArrayType::Map(m1.data(), m1.rows(), m1.cols()) *= ArrayType::Map(m2.data(), m2.rows(), m2.cols()); VERIFY_IS_APPROX(m1, m3 * m2); - + m3 = m1; - m2 = ArrayType::Random(rows,cols); - m2 = (m2==0).select(1,m2); - ArrayType::Map(m1.data(), m1.rows(), m1.cols()) /= ArrayType::Map(m2.data(), m2.rows(), m2.cols()); + m2 = ArrayType::Random(rows, cols); + m2 = (m2 == 0).select(1, m2); + ArrayType::Map(m1.data(), m1.rows(), m1.cols()) /= ArrayType::Map(m2.data(), m2.rows(), m2.cols()); VERIFY_IS_APPROX(m1, m3 / m2); // reductions @@ -70,9 +67,9 @@ template void array(const ArrayType& m) using std::abs; VERIFY_IS_MUCH_SMALLER_THAN(abs(m1.colwise().sum().sum() - m1.sum()), m1.abs().sum()); VERIFY_IS_MUCH_SMALLER_THAN(abs(m1.rowwise().sum().sum() - m1.sum()), m1.abs().sum()); - if (!internal::isMuchSmallerThan(abs(m1.sum() - (m1+m2).sum()), m1.abs().sum(), test_precision())) - VERIFY_IS_NOT_APPROX(((m1+m2).rowwise().sum()).sum(), m1.sum()); - VERIFY_IS_APPROX(m1.colwise().sum(), m1.colwise().redux(internal::scalar_sum_op())); + if (!internal::isMuchSmallerThan(abs(m1.sum() - (m1 + m2).sum()), m1.abs().sum(), test_precision())) + VERIFY_IS_NOT_APPROX(((m1 + m2).rowwise().sum()).sum(), m1.sum()); + VERIFY_IS_APPROX(m1.colwise().sum(), m1.colwise().redux(internal::scalar_sum_op())); // vector-wise ops m3 = m1; @@ -83,50 +80,51 @@ template void array(const ArrayType& m) VERIFY_IS_APPROX(m3.rowwise() += rv1, m1.rowwise() + rv1); m3 = m1; VERIFY_IS_APPROX(m3.rowwise() -= rv1, m1.rowwise() - rv1); - + // Conversion from scalar - VERIFY_IS_APPROX((m3 = s1), ArrayType::Constant(rows,cols,s1)); - VERIFY_IS_APPROX((m3 = 1), ArrayType::Constant(rows,cols,1)); - VERIFY_IS_APPROX((m3.topLeftCorner(rows,cols) = 1), ArrayType::Constant(rows,cols,1)); + VERIFY_IS_APPROX((m3 = s1), ArrayType::Constant(rows, cols, s1)); + VERIFY_IS_APPROX((m3 = 1), ArrayType::Constant(rows, cols, 1)); + VERIFY_IS_APPROX((m3.topLeftCorner(rows, cols) = 1), ArrayType::Constant(rows, cols, 1)); typedef Array FixedArrayType; + ArrayType::RowsAtCompileTime == Dynamic ? 2 : ArrayType::RowsAtCompileTime, + ArrayType::ColsAtCompileTime == Dynamic ? 2 : ArrayType::ColsAtCompileTime, + ArrayType::Options> + FixedArrayType; FixedArrayType f1(s1); VERIFY_IS_APPROX(f1, FixedArrayType::Constant(s1)); FixedArrayType f2(numext::real(s1)); VERIFY_IS_APPROX(f2, FixedArrayType::Constant(numext::real(s1))); - FixedArrayType f3((int)100*numext::real(s1)); - VERIFY_IS_APPROX(f3, FixedArrayType::Constant((int)100*numext::real(s1))); + FixedArrayType f3((int)100 * numext::real(s1)); + VERIFY_IS_APPROX(f3, FixedArrayType::Constant((int)100 * numext::real(s1))); f1.setRandom(); FixedArrayType f4(f1.data()); VERIFY_IS_APPROX(f4, f1); - + // pow VERIFY_IS_APPROX(m1.pow(2), m1.square()); - VERIFY_IS_APPROX(pow(m1,2), m1.square()); + VERIFY_IS_APPROX(pow(m1, 2), m1.square()); VERIFY_IS_APPROX(m1.pow(3), m1.cube()); - VERIFY_IS_APPROX(pow(m1,3), m1.cube()); + VERIFY_IS_APPROX(pow(m1, 3), m1.cube()); VERIFY_IS_APPROX((-m1).pow(3), -m1.cube()); - VERIFY_IS_APPROX(pow(2*m1,3), 8*m1.cube()); + VERIFY_IS_APPROX(pow(2 * m1, 3), 8 * m1.cube()); ArrayType exponents = ArrayType::Constant(rows, cols, RealScalar(2)); - VERIFY_IS_APPROX(Eigen::pow(m1,exponents), m1.square()); + VERIFY_IS_APPROX(Eigen::pow(m1, exponents), m1.square()); VERIFY_IS_APPROX(m1.pow(exponents), m1.square()); - VERIFY_IS_APPROX(Eigen::pow(2*m1,exponents), 4*m1.square()); - VERIFY_IS_APPROX((2*m1).pow(exponents), 4*m1.square()); - VERIFY_IS_APPROX(Eigen::pow(m1,2*exponents), m1.square().square()); - VERIFY_IS_APPROX(m1.pow(2*exponents), m1.square().square()); - VERIFY_IS_APPROX(Eigen::pow(m1(0,0), exponents), ArrayType::Constant(rows,cols,m1(0,0)*m1(0,0))); + VERIFY_IS_APPROX(Eigen::pow(2 * m1, exponents), 4 * m1.square()); + VERIFY_IS_APPROX((2 * m1).pow(exponents), 4 * m1.square()); + VERIFY_IS_APPROX(Eigen::pow(m1, 2 * exponents), m1.square().square()); + VERIFY_IS_APPROX(m1.pow(2 * exponents), m1.square().square()); + VERIFY_IS_APPROX(Eigen::pow(m1(0, 0), exponents), ArrayType::Constant(rows, cols, m1(0, 0) * m1(0, 0))); // Check possible conflicts with 1D ctor typedef Array OneDArrayType; OneDArrayType o1(rows); - VERIFY(o1.size()==rows); + VERIFY(o1.size() == rows); OneDArrayType o4((int)rows); - VERIFY(o4.size()==rows); + VERIFY(o4.size() == rows); } -template void comparisons(const ArrayType& m) +template void comparisons(const ArrayType &m) { using std::abs; typedef typename ArrayType::Scalar Scalar; @@ -135,74 +133,66 @@ template void comparisons(const ArrayType& m) Index rows = m.rows(); Index cols = m.cols(); - Index r = internal::random(0, rows-1), - c = internal::random(0, cols-1); + Index r = internal::random(0, rows - 1), c = internal::random(0, cols - 1); - ArrayType m1 = ArrayType::Random(rows, cols), - m2 = ArrayType::Random(rows, cols), - m3(rows, cols), - m4 = m1; - - m4 = (m4.abs()==Scalar(0)).select(1,m4); + ArrayType m1 = ArrayType::Random(rows, cols), m2 = ArrayType::Random(rows, cols), m3(rows, cols), m4 = m1; + + m4 = (m4.abs() == Scalar(0)).select(1, m4); VERIFY(((m1 + Scalar(1)) > m1).all()); VERIFY(((m1 - Scalar(1)) < m1).all()); - if (rows*cols>1) - { + if (rows * cols > 1) { m3 = m1; - m3(r,c) += 1; - VERIFY(! (m1 < m3).all() ); - VERIFY(! (m1 > m3).all() ); + m3(r, c) += 1; + VERIFY(!(m1 < m3).all()); + VERIFY(!(m1 > m3).all()); } VERIFY(!(m1 > m2 && m1 < m2).any()); VERIFY((m1 <= m2 || m1 >= m2).all()); // comparisons array to scalar - VERIFY( (m1 != (m1(r,c)+1) ).any() ); - VERIFY( (m1 > (m1(r,c)-1) ).any() ); - VERIFY( (m1 < (m1(r,c)+1) ).any() ); - VERIFY( (m1 == m1(r,c) ).any() ); + VERIFY((m1 != (m1(r, c) + 1)).any()); + VERIFY((m1 > (m1(r, c) - 1)).any()); + VERIFY((m1 < (m1(r, c) + 1)).any()); + VERIFY((m1 == m1(r, c)).any()); // comparisons scalar to array - VERIFY( ( (m1(r,c)+1) != m1).any() ); - VERIFY( ( (m1(r,c)-1) < m1).any() ); - VERIFY( ( (m1(r,c)+1) > m1).any() ); - VERIFY( ( m1(r,c) == m1).any() ); + VERIFY(((m1(r, c) + 1) != m1).any()); + VERIFY(((m1(r, c) - 1) < m1).any()); + VERIFY(((m1(r, c) + 1) > m1).any()); + VERIFY((m1(r, c) == m1).any()); // test Select - VERIFY_IS_APPROX( (m1m2).select(m1,m2), m1.cwiseMax(m2) ); - Scalar mid = (m1.cwiseAbs().minCoeff() + m1.cwiseAbs().maxCoeff())/Scalar(2); - for (int j=0; j m2).select(m1, m2), m1.cwiseMax(m2)); + Scalar mid = (m1.cwiseAbs().minCoeff() + m1.cwiseAbs().maxCoeff()) / Scalar(2); + for (int j = 0; j < cols; ++j) + for (int i = 0; i < rows; ++i) m3(i, j) = abs(m1(i, j)) < mid ? 0 : m1(i, j); + VERIFY_IS_APPROX((m1.abs() < ArrayType::Constant(rows, cols, mid)).select(ArrayType::Zero(rows, cols), m1), m3); // shorter versions: - VERIFY_IS_APPROX( (m1.abs()=ArrayType::Constant(rows,cols,mid)) - .select(m1,0), m3); + VERIFY_IS_APPROX((m1.abs() < ArrayType::Constant(rows, cols, mid)).select(0, m1), m3); + VERIFY_IS_APPROX((m1.abs() >= ArrayType::Constant(rows, cols, mid)).select(m1, 0), m3); // even shorter version: - VERIFY_IS_APPROX( (m1.abs()RealScalar(0.1)).count() == rows*cols); + VERIFY(((m1.abs() + 1) > RealScalar(0.1)).count() == rows * cols); // and/or - VERIFY( (m1RealScalar(0)).count() == 0); - VERIFY( (m1=RealScalar(0)).count() == rows*cols); + VERIFY((m1 < RealScalar(0) && m1 > RealScalar(0)).count() == 0); + VERIFY((m1 < RealScalar(0) || m1 >= RealScalar(0)).count() == rows * cols); RealScalar a = m1.abs().mean(); - VERIFY( (m1<-a || m1>a).count() == (m1.abs()>a).count()); + VERIFY((m1 < -a || m1 > a).count() == (m1.abs() > a).count()); typedef Array ArrayOfIndices; // TODO allows colwise/rowwise for array - VERIFY_IS_APPROX(((m1.abs()+1)>RealScalar(0.1)).colwise().count(), ArrayOfIndices::Constant(cols,rows).transpose()); - VERIFY_IS_APPROX(((m1.abs()+1)>RealScalar(0.1)).rowwise().count(), ArrayOfIndices::Constant(rows, cols)); + VERIFY_IS_APPROX( + ((m1.abs() + 1) > RealScalar(0.1)).colwise().count(), ArrayOfIndices::Constant(cols, rows).transpose()); + VERIFY_IS_APPROX(((m1.abs() + 1) > RealScalar(0.1)).rowwise().count(), ArrayOfIndices::Constant(rows, cols)); } -template void array_real(const ArrayType& m) +template void array_real(const ArrayType &m) { using std::abs; using std::sqrt; @@ -212,14 +202,11 @@ template void array_real(const ArrayType& m) Index rows = m.rows(); Index cols = m.cols(); - ArrayType m1 = ArrayType::Random(rows, cols), - m2 = ArrayType::Random(rows, cols), - m3(rows, cols), - m4 = m1; + ArrayType m1 = ArrayType::Random(rows, cols), m2 = ArrayType::Random(rows, cols), m3(rows, cols), m4 = m1; - m4 = (m4.abs()==Scalar(0)).select(1,m4); + m4 = (m4.abs() == Scalar(0)).select(1, m4); - Scalar s1 = internal::random(); + Scalar s1 = internal::random(); // these tests are mostly to check possible compilation issues with free-functions. VERIFY_IS_APPROX(m1.sin(), sin(m1)); @@ -244,67 +231,66 @@ template void array_real(const ArrayType& m) VERIFY_IS_APPROX(m1.abs2(), abs2(m1)); VERIFY_IS_APPROX(m1.square(), square(m1)); VERIFY_IS_APPROX(m1.cube(), cube(m1)); - VERIFY_IS_APPROX(cos(m1+RealScalar(3)*m2), cos((m1+RealScalar(3)*m2).eval())); + VERIFY_IS_APPROX(cos(m1 + RealScalar(3) * m2), cos((m1 + RealScalar(3) * m2).eval())); VERIFY_IS_APPROX(m1.sign(), sign(m1)); // avoid NaNs with abs() so verification doesn't fail m3 = m1.abs(); VERIFY_IS_APPROX(m3.sqrt(), sqrt(abs(m1))); - VERIFY_IS_APPROX(m3.rsqrt(), Scalar(1)/sqrt(abs(m1))); - VERIFY_IS_APPROX(rsqrt(m3), Scalar(1)/sqrt(abs(m1))); + VERIFY_IS_APPROX(m3.rsqrt(), Scalar(1) / sqrt(abs(m1))); + VERIFY_IS_APPROX(rsqrt(m3), Scalar(1) / sqrt(abs(m1))); VERIFY_IS_APPROX(m3.log(), log(m3)); VERIFY_IS_APPROX(m3.log1p(), log1p(m3)); VERIFY_IS_APPROX(m3.log10(), log10(m3)); - VERIFY((!(m1>m2) == (m1<=m2)).all()); + VERIFY((!(m1 > m2) == (m1 <= m2)).all()); VERIFY_IS_APPROX(sin(m1.asin()), m1); VERIFY_IS_APPROX(cos(m1.acos()), m1); VERIFY_IS_APPROX(tan(m1.atan()), m1); - VERIFY_IS_APPROX(sinh(m1), 0.5*(exp(m1)-exp(-m1))); - VERIFY_IS_APPROX(cosh(m1), 0.5*(exp(m1)+exp(-m1))); - VERIFY_IS_APPROX(tanh(m1), (0.5*(exp(m1)-exp(-m1)))/(0.5*(exp(m1)+exp(-m1)))); - VERIFY_IS_APPROX(arg(m1), ((m1<0).template cast())*std::acos(-1.0)); + VERIFY_IS_APPROX(sinh(m1), 0.5 * (exp(m1) - exp(-m1))); + VERIFY_IS_APPROX(cosh(m1), 0.5 * (exp(m1) + exp(-m1))); + VERIFY_IS_APPROX(tanh(m1), (0.5 * (exp(m1) - exp(-m1))) / (0.5 * (exp(m1) + exp(-m1)))); + VERIFY_IS_APPROX(arg(m1), ((m1 < 0).template cast()) * std::acos(-1.0)); VERIFY((round(m1) <= ceil(m1) && round(m1) >= floor(m1)).all()); - VERIFY((Eigen::isnan)((m1*0.0)/0.0).all()); - VERIFY((Eigen::isinf)(m4/0.0).all()); - VERIFY(((Eigen::isfinite)(m1) && (!(Eigen::isfinite)(m1*0.0/0.0)) && (!(Eigen::isfinite)(m4/0.0))).all()); - VERIFY_IS_APPROX(inverse(inverse(m1)),m1); + VERIFY((Eigen::isnan)((m1 * 0.0) / 0.0).all()); + VERIFY((Eigen::isinf)(m4 / 0.0).all()); + VERIFY(((Eigen::isfinite)(m1) && (!(Eigen::isfinite)(m1 * 0.0 / 0.0)) && (!(Eigen::isfinite)(m4 / 0.0))).all()); + VERIFY_IS_APPROX(inverse(inverse(m1)), m1); VERIFY((abs(m1) == m1 || abs(m1) == -m1).all()); VERIFY_IS_APPROX(m3, sqrt(abs2(m1))); - VERIFY_IS_APPROX( m1.sign(), -(-m1).sign() ); - VERIFY_IS_APPROX( m1*m1.sign(),m1.abs()); + VERIFY_IS_APPROX(m1.sign(), -(-m1).sign()); + VERIFY_IS_APPROX(m1 * m1.sign(), m1.abs()); VERIFY_IS_APPROX(m1.sign() * m1.abs(), m1); VERIFY_IS_APPROX(numext::abs2(numext::real(m1)) + numext::abs2(numext::imag(m1)), numext::abs2(m1)); VERIFY_IS_APPROX(numext::abs2(real(m1)) + numext::abs2(imag(m1)), numext::abs2(m1)); - if(!NumTraits::IsComplex) - VERIFY_IS_APPROX(numext::real(m1), m1); + if (!NumTraits::IsComplex) VERIFY_IS_APPROX(numext::real(m1), m1); // shift argument of logarithm so that it is not zero Scalar smallNumber = NumTraits::dummy_precision(); - VERIFY_IS_APPROX((m3 + smallNumber).log() , log(abs(m1) + smallNumber)); - VERIFY_IS_APPROX((m3 + smallNumber + 1).log() , log1p(abs(m1) + smallNumber)); + VERIFY_IS_APPROX((m3 + smallNumber).log(), log(abs(m1) + smallNumber)); + VERIFY_IS_APPROX((m3 + smallNumber + 1).log(), log1p(abs(m1) + smallNumber)); - VERIFY_IS_APPROX(m1.exp() * m2.exp(), exp(m1+m2)); + VERIFY_IS_APPROX(m1.exp() * m2.exp(), exp(m1 + m2)); VERIFY_IS_APPROX(m1.exp(), exp(m1)); - VERIFY_IS_APPROX(m1.exp() / m2.exp(),(m1-m2).exp()); + VERIFY_IS_APPROX(m1.exp() / m2.exp(), (m1 - m2).exp()); VERIFY_IS_APPROX(m3.pow(RealScalar(0.5)), m3.sqrt()); - VERIFY_IS_APPROX(pow(m3,RealScalar(0.5)), m3.sqrt()); + VERIFY_IS_APPROX(pow(m3, RealScalar(0.5)), m3.sqrt()); VERIFY_IS_APPROX(m3.pow(RealScalar(-0.5)), m3.rsqrt()); - VERIFY_IS_APPROX(pow(m3,RealScalar(-0.5)), m3.rsqrt()); + VERIFY_IS_APPROX(pow(m3, RealScalar(-0.5)), m3.rsqrt()); - VERIFY_IS_APPROX(log10(m3), log(m3)/log(10)); + VERIFY_IS_APPROX(log10(m3), log(m3) / log(10)); // scalar by array division const RealScalar tiny = sqrt(std::numeric_limits::epsilon()); s1 += Scalar(tiny); - m1 += ArrayType::Constant(rows,cols,Scalar(tiny)); - VERIFY_IS_APPROX(s1/m1, s1 * m1.inverse()); + m1 += ArrayType::Constant(rows, cols, Scalar(tiny)); + VERIFY_IS_APPROX(s1 / m1, s1 * m1.inverse()); // check inplace transpose m3 = m1; @@ -314,7 +300,7 @@ template void array_real(const ArrayType& m) VERIFY_IS_APPROX(m3, m1); } -template void array_complex(const ArrayType& m) +template void array_complex(const ArrayType &m) { typedef typename ArrayType::Scalar Scalar; typedef typename NumTraits::Real RealScalar; @@ -322,18 +308,15 @@ template void array_complex(const ArrayType& m) Index rows = m.rows(); Index cols = m.cols(); - ArrayType m1 = ArrayType::Random(rows, cols), - m2(rows, cols), - m4 = m1; - - m4.real() = (m4.real().abs()==RealScalar(0)).select(RealScalar(1),m4.real()); - m4.imag() = (m4.imag().abs()==RealScalar(0)).select(RealScalar(1),m4.imag()); + ArrayType m1 = ArrayType::Random(rows, cols), m2(rows, cols), m4 = m1; + + m4.real() = (m4.real().abs() == RealScalar(0)).select(RealScalar(1), m4.real()); + m4.imag() = (m4.imag().abs() == RealScalar(0)).select(RealScalar(1), m4.imag()); Array m3(rows, cols); for (Index i = 0; i < m.rows(); ++i) - for (Index j = 0; j < m.cols(); ++j) - m2(i,j) = sqrt(m1(i,j)); + for (Index j = 0; j < m.cols(); ++j) m2(i, j) = sqrt(m1(i, j)); // these tests are mostly to check possible compilation issues with free-functions. VERIFY_IS_APPROX(m1.sin(), sin(m1)); @@ -354,60 +337,57 @@ template void array_complex(const ArrayType& m) VERIFY_IS_APPROX(m1.sqrt(), sqrt(m1)); VERIFY_IS_APPROX(m1.square(), square(m1)); VERIFY_IS_APPROX(m1.cube(), cube(m1)); - VERIFY_IS_APPROX(cos(m1+RealScalar(3)*m2), cos((m1+RealScalar(3)*m2).eval())); + VERIFY_IS_APPROX(cos(m1 + RealScalar(3) * m2), cos((m1 + RealScalar(3) * m2).eval())); VERIFY_IS_APPROX(m1.sign(), sign(m1)); - VERIFY_IS_APPROX(m1.exp() * m2.exp(), exp(m1+m2)); + VERIFY_IS_APPROX(m1.exp() * m2.exp(), exp(m1 + m2)); VERIFY_IS_APPROX(m1.exp(), exp(m1)); - VERIFY_IS_APPROX(m1.exp() / m2.exp(),(m1-m2).exp()); + VERIFY_IS_APPROX(m1.exp() / m2.exp(), (m1 - m2).exp()); - VERIFY_IS_APPROX(sinh(m1), 0.5*(exp(m1)-exp(-m1))); - VERIFY_IS_APPROX(cosh(m1), 0.5*(exp(m1)+exp(-m1))); - VERIFY_IS_APPROX(tanh(m1), (0.5*(exp(m1)-exp(-m1)))/(0.5*(exp(m1)+exp(-m1)))); + VERIFY_IS_APPROX(sinh(m1), 0.5 * (exp(m1) - exp(-m1))); + VERIFY_IS_APPROX(cosh(m1), 0.5 * (exp(m1) + exp(-m1))); + VERIFY_IS_APPROX(tanh(m1), (0.5 * (exp(m1) - exp(-m1))) / (0.5 * (exp(m1) + exp(-m1)))); for (Index i = 0; i < m.rows(); ++i) - for (Index j = 0; j < m.cols(); ++j) - m3(i,j) = std::atan2(imag(m1(i,j)), real(m1(i,j))); + for (Index j = 0; j < m.cols(); ++j) m3(i, j) = std::atan2(imag(m1(i, j)), real(m1(i, j))); VERIFY_IS_APPROX(arg(m1), m3); - std::complex zero(0.0,0.0); - VERIFY((Eigen::isnan)(m1*zero/zero).all()); + std::complex zero(0.0, 0.0); + VERIFY((Eigen::isnan)(m1 * zero / zero).all()); #if EIGEN_COMP_MSVC // msvc complex division is not robust - VERIFY((Eigen::isinf)(m4/RealScalar(0)).all()); + VERIFY((Eigen::isinf)(m4 / RealScalar(0)).all()); #else #if EIGEN_COMP_CLANG // clang's complex division is notoriously broken too - if((numext::isinf)(m4(0,0)/RealScalar(0))) { + if ((numext::isinf)(m4(0, 0) / RealScalar(0))) { #endif - VERIFY((Eigen::isinf)(m4/zero).all()); + VERIFY((Eigen::isinf)(m4 / zero).all()); #if EIGEN_COMP_CLANG - } - else - { - VERIFY((Eigen::isinf)(m4.real()/zero.real()).all()); + } else { + VERIFY((Eigen::isinf)(m4.real() / zero.real()).all()); } #endif -#endif // MSVC +#endif// MSVC - VERIFY(((Eigen::isfinite)(m1) && (!(Eigen::isfinite)(m1*zero/zero)) && (!(Eigen::isfinite)(m1/zero))).all()); + VERIFY(((Eigen::isfinite)(m1) && (!(Eigen::isfinite)(m1 * zero / zero)) && (!(Eigen::isfinite)(m1 / zero))).all()); - VERIFY_IS_APPROX(inverse(inverse(m1)),m1); + VERIFY_IS_APPROX(inverse(inverse(m1)), m1); VERIFY_IS_APPROX(conj(m1.conjugate()), m1); - VERIFY_IS_APPROX(abs(m1), sqrt(square(real(m1))+square(imag(m1)))); + VERIFY_IS_APPROX(abs(m1), sqrt(square(real(m1)) + square(imag(m1)))); VERIFY_IS_APPROX(abs(m1), sqrt(abs2(m1))); - VERIFY_IS_APPROX(log10(m1), log(m1)/log(10)); + VERIFY_IS_APPROX(log10(m1), log(m1) / log(10)); - VERIFY_IS_APPROX( m1.sign(), -(-m1).sign() ); - VERIFY_IS_APPROX( m1.sign() * m1.abs(), m1); + VERIFY_IS_APPROX(m1.sign(), -(-m1).sign()); + VERIFY_IS_APPROX(m1.sign() * m1.abs(), m1); // scalar by array division - Scalar s1 = internal::random(); + Scalar s1 = internal::random(); const RealScalar tiny = std::sqrt(std::numeric_limits::epsilon()); s1 += Scalar(tiny); - m1 += ArrayType::Constant(rows,cols,Scalar(tiny)); - VERIFY_IS_APPROX(s1/m1, s1 * m1.inverse()); + m1 += ArrayType::Constant(rows, cols, Scalar(tiny)); + VERIFY_IS_APPROX(s1 / m1, s1 * m1.inverse()); // check inplace transpose m2 = m1; @@ -415,10 +395,9 @@ template void array_complex(const ArrayType& m) VERIFY_IS_APPROX(m2, m1.transpose()); m2.transposeInPlace(); VERIFY_IS_APPROX(m2, m1); - } -template void min_max(const ArrayType& m) +template void min_max(const ArrayType &m) { typedef typename ArrayType::Scalar Scalar; @@ -431,60 +410,66 @@ template void min_max(const ArrayType& m) Scalar maxM1 = m1.maxCoeff(); Scalar minM1 = m1.minCoeff(); - VERIFY_IS_APPROX(ArrayType::Constant(rows,cols, minM1), (m1.min)(ArrayType::Constant(rows,cols, minM1))); - VERIFY_IS_APPROX(m1, (m1.min)(ArrayType::Constant(rows,cols, maxM1))); + VERIFY_IS_APPROX(ArrayType::Constant(rows, cols, minM1), (m1.min)(ArrayType::Constant(rows, cols, minM1))); + VERIFY_IS_APPROX(m1, (m1.min)(ArrayType::Constant(rows, cols, maxM1))); - VERIFY_IS_APPROX(ArrayType::Constant(rows,cols, maxM1), (m1.max)(ArrayType::Constant(rows,cols, maxM1))); - VERIFY_IS_APPROX(m1, (m1.max)(ArrayType::Constant(rows,cols, minM1))); + VERIFY_IS_APPROX(ArrayType::Constant(rows, cols, maxM1), (m1.max)(ArrayType::Constant(rows, cols, maxM1))); + VERIFY_IS_APPROX(m1, (m1.max)(ArrayType::Constant(rows, cols, minM1))); // min/max with scalar input - VERIFY_IS_APPROX(ArrayType::Constant(rows,cols, minM1), (m1.min)( minM1)); - VERIFY_IS_APPROX(m1, (m1.min)( maxM1)); - - VERIFY_IS_APPROX(ArrayType::Constant(rows,cols, maxM1), (m1.max)( maxM1)); - VERIFY_IS_APPROX(m1, (m1.max)( minM1)); + VERIFY_IS_APPROX(ArrayType::Constant(rows, cols, minM1), (m1.min)(minM1)); + VERIFY_IS_APPROX(m1, (m1.min)(maxM1)); + VERIFY_IS_APPROX(ArrayType::Constant(rows, cols, maxM1), (m1.max)(maxM1)); + VERIFY_IS_APPROX(m1, (m1.max)(minM1)); } void test_array() { - for(int i = 0; i < g_repeat; i++) { - CALL_SUBTEST_1( array(Array()) ); - CALL_SUBTEST_2( array(Array22f()) ); - CALL_SUBTEST_3( array(Array44d()) ); - CALL_SUBTEST_4( array(ArrayXXcf(internal::random(1,EIGEN_TEST_MAX_SIZE), internal::random(1,EIGEN_TEST_MAX_SIZE))) ); - CALL_SUBTEST_5( array(ArrayXXf(internal::random(1,EIGEN_TEST_MAX_SIZE), internal::random(1,EIGEN_TEST_MAX_SIZE))) ); - CALL_SUBTEST_6( array(ArrayXXi(internal::random(1,EIGEN_TEST_MAX_SIZE), internal::random(1,EIGEN_TEST_MAX_SIZE))) ); + for (int i = 0; i < g_repeat; i++) { + CALL_SUBTEST_1(array(Array())); + CALL_SUBTEST_2(array(Array22f())); + CALL_SUBTEST_3(array(Array44d())); + CALL_SUBTEST_4( + array(ArrayXXcf(internal::random(1, EIGEN_TEST_MAX_SIZE), internal::random(1, EIGEN_TEST_MAX_SIZE)))); + CALL_SUBTEST_5( + array(ArrayXXf(internal::random(1, EIGEN_TEST_MAX_SIZE), internal::random(1, EIGEN_TEST_MAX_SIZE)))); + CALL_SUBTEST_6( + array(ArrayXXi(internal::random(1, EIGEN_TEST_MAX_SIZE), internal::random(1, EIGEN_TEST_MAX_SIZE)))); } - for(int i = 0; i < g_repeat; i++) { - CALL_SUBTEST_1( comparisons(Array()) ); - CALL_SUBTEST_2( comparisons(Array22f()) ); - CALL_SUBTEST_3( comparisons(Array44d()) ); - CALL_SUBTEST_5( comparisons(ArrayXXf(internal::random(1,EIGEN_TEST_MAX_SIZE), internal::random(1,EIGEN_TEST_MAX_SIZE))) ); - CALL_SUBTEST_6( comparisons(ArrayXXi(internal::random(1,EIGEN_TEST_MAX_SIZE), internal::random(1,EIGEN_TEST_MAX_SIZE))) ); + for (int i = 0; i < g_repeat; i++) { + CALL_SUBTEST_1(comparisons(Array())); + CALL_SUBTEST_2(comparisons(Array22f())); + CALL_SUBTEST_3(comparisons(Array44d())); + CALL_SUBTEST_5(comparisons( + ArrayXXf(internal::random(1, EIGEN_TEST_MAX_SIZE), internal::random(1, EIGEN_TEST_MAX_SIZE)))); + CALL_SUBTEST_6(comparisons( + ArrayXXi(internal::random(1, EIGEN_TEST_MAX_SIZE), internal::random(1, EIGEN_TEST_MAX_SIZE)))); } - for(int i = 0; i < g_repeat; i++) { - CALL_SUBTEST_1( min_max(Array()) ); - CALL_SUBTEST_2( min_max(Array22f()) ); - CALL_SUBTEST_3( min_max(Array44d()) ); - CALL_SUBTEST_5( min_max(ArrayXXf(internal::random(1,EIGEN_TEST_MAX_SIZE), internal::random(1,EIGEN_TEST_MAX_SIZE))) ); - CALL_SUBTEST_6( min_max(ArrayXXi(internal::random(1,EIGEN_TEST_MAX_SIZE), internal::random(1,EIGEN_TEST_MAX_SIZE))) ); + for (int i = 0; i < g_repeat; i++) { + CALL_SUBTEST_1(min_max(Array())); + CALL_SUBTEST_2(min_max(Array22f())); + CALL_SUBTEST_3(min_max(Array44d())); + CALL_SUBTEST_5( + min_max(ArrayXXf(internal::random(1, EIGEN_TEST_MAX_SIZE), internal::random(1, EIGEN_TEST_MAX_SIZE)))); + CALL_SUBTEST_6( + min_max(ArrayXXi(internal::random(1, EIGEN_TEST_MAX_SIZE), internal::random(1, EIGEN_TEST_MAX_SIZE)))); } - for(int i = 0; i < g_repeat; i++) { - CALL_SUBTEST_1( array_real(Array()) ); - CALL_SUBTEST_2( array_real(Array22f()) ); - CALL_SUBTEST_3( array_real(Array44d()) ); - CALL_SUBTEST_5( array_real(ArrayXXf(internal::random(1,EIGEN_TEST_MAX_SIZE), internal::random(1,EIGEN_TEST_MAX_SIZE))) ); + for (int i = 0; i < g_repeat; i++) { + CALL_SUBTEST_1(array_real(Array())); + CALL_SUBTEST_2(array_real(Array22f())); + CALL_SUBTEST_3(array_real(Array44d())); + CALL_SUBTEST_5(array_real( + ArrayXXf(internal::random(1, EIGEN_TEST_MAX_SIZE), internal::random(1, EIGEN_TEST_MAX_SIZE)))); } - for(int i = 0; i < g_repeat; i++) { - CALL_SUBTEST_4( array_complex(ArrayXXcf(internal::random(1,EIGEN_TEST_MAX_SIZE), internal::random(1,EIGEN_TEST_MAX_SIZE))) ); + for (int i = 0; i < g_repeat; i++) { + CALL_SUBTEST_4(array_complex( + ArrayXXcf(internal::random(1, EIGEN_TEST_MAX_SIZE), internal::random(1, EIGEN_TEST_MAX_SIZE)))); } - VERIFY((internal::is_same< internal::global_math_functions_filtering_base::type, int >::value)); - VERIFY((internal::is_same< internal::global_math_functions_filtering_base::type, float >::value)); - VERIFY((internal::is_same< internal::global_math_functions_filtering_base::type, ArrayBase >::value)); - typedef CwiseUnaryOp, ArrayXd > Xpr; - VERIFY((internal::is_same< internal::global_math_functions_filtering_base::type, - ArrayBase - >::value)); + VERIFY((internal::is_same::type, int>::value)); + VERIFY((internal::is_same::type, float>::value)); + VERIFY((internal::is_same::type, ArrayBase>::value)); + typedef CwiseUnaryOp, ArrayXd> Xpr; + VERIFY((internal::is_same::type, ArrayBase>::value)); } diff --git a/filmulator-gui/core/nlmeans/eigen/test/array_for_matrix.cpp b/filmulator-gui/core/nlmeans/eigen/test/array_for_matrix.cpp index a05bba19..7e8abb1c 100644 --- a/filmulator-gui/core/nlmeans/eigen/test/array_for_matrix.cpp +++ b/filmulator-gui/core/nlmeans/eigen/test/array_for_matrix.cpp @@ -9,29 +9,26 @@ #include "main.h" -template void array_for_matrix(const MatrixType& m) +template void array_for_matrix(const MatrixType &m) { typedef typename MatrixType::Scalar Scalar; typedef Matrix ColVectorType; - typedef Matrix RowVectorType; + typedef Matrix RowVectorType; Index rows = m.rows(); Index cols = m.cols(); - MatrixType m1 = MatrixType::Random(rows, cols), - m2 = MatrixType::Random(rows, cols), - m3(rows, cols); + MatrixType m1 = MatrixType::Random(rows, cols), m2 = MatrixType::Random(rows, cols), m3(rows, cols); ColVectorType cv1 = ColVectorType::Random(rows); RowVectorType rv1 = RowVectorType::Random(cols); - - Scalar s1 = internal::random(), - s2 = internal::random(); - + + Scalar s1 = internal::random(), s2 = internal::random(); + // scalar addition VERIFY_IS_APPROX(m1.array() + s1, s1 + m1.array()); - VERIFY_IS_APPROX((m1.array() + s1).matrix(), MatrixType::Constant(rows,cols,s1) + m1); - VERIFY_IS_APPROX(((m1*Scalar(2)).array() - s2).matrix(), (m1+m1) - MatrixType::Constant(rows,cols,s2) ); + VERIFY_IS_APPROX((m1.array() + s1).matrix(), MatrixType::Constant(rows, cols, s1) + m1); + VERIFY_IS_APPROX(((m1 * Scalar(2)).array() - s2).matrix(), (m1 + m1) - MatrixType::Constant(rows, cols, s2)); m3 = m1; m3.array() += s2; VERIFY_IS_APPROX(m3, (m1.array() + s2).matrix()); @@ -42,9 +39,11 @@ template void array_for_matrix(const MatrixType& m) // reductions VERIFY_IS_MUCH_SMALLER_THAN(m1.colwise().sum().sum() - m1.sum(), m1.squaredNorm()); VERIFY_IS_MUCH_SMALLER_THAN(m1.rowwise().sum().sum() - m1.sum(), m1.squaredNorm()); - VERIFY_IS_MUCH_SMALLER_THAN(m1.colwise().sum() + m2.colwise().sum() - (m1+m2).colwise().sum(), (m1+m2).squaredNorm()); - VERIFY_IS_MUCH_SMALLER_THAN(m1.rowwise().sum() - m2.rowwise().sum() - (m1-m2).rowwise().sum(), (m1-m2).squaredNorm()); - VERIFY_IS_APPROX(m1.colwise().sum(), m1.colwise().redux(internal::scalar_sum_op())); + VERIFY_IS_MUCH_SMALLER_THAN( + m1.colwise().sum() + m2.colwise().sum() - (m1 + m2).colwise().sum(), (m1 + m2).squaredNorm()); + VERIFY_IS_MUCH_SMALLER_THAN( + m1.rowwise().sum() - m2.rowwise().sum() - (m1 - m2).rowwise().sum(), (m1 - m2).squaredNorm()); + VERIFY_IS_APPROX(m1.colwise().sum(), m1.colwise().redux(internal::scalar_sum_op())); // vector-wise ops m3 = m1; @@ -55,31 +54,31 @@ template void array_for_matrix(const MatrixType& m) VERIFY_IS_APPROX(m3.rowwise() += rv1, m1.rowwise() + rv1); m3 = m1; VERIFY_IS_APPROX(m3.rowwise() -= rv1, m1.rowwise() - rv1); - + // empty objects - VERIFY_IS_APPROX(m1.block(0,0,0,cols).colwise().sum(), RowVectorType::Zero(cols)); - VERIFY_IS_APPROX(m1.block(0,0,rows,0).rowwise().prod(), ColVectorType::Ones(rows)); - + VERIFY_IS_APPROX(m1.block(0, 0, 0, cols).colwise().sum(), RowVectorType::Zero(cols)); + VERIFY_IS_APPROX(m1.block(0, 0, rows, 0).rowwise().prod(), ColVectorType::Ones(rows)); + // verify the const accessors exist - const Scalar& ref_m1 = m.matrix().array().coeffRef(0); - const Scalar& ref_m2 = m.matrix().array().coeffRef(0,0); - const Scalar& ref_a1 = m.array().matrix().coeffRef(0); - const Scalar& ref_a2 = m.array().matrix().coeffRef(0,0); + const Scalar &ref_m1 = m.matrix().array().coeffRef(0); + const Scalar &ref_m2 = m.matrix().array().coeffRef(0, 0); + const Scalar &ref_a1 = m.array().matrix().coeffRef(0); + const Scalar &ref_a2 = m.array().matrix().coeffRef(0, 0); VERIFY(&ref_a1 == &ref_m1); VERIFY(&ref_a2 == &ref_m2); // Check write accessors: - m1.array().coeffRef(0,0) = 1; - VERIFY_IS_APPROX(m1(0,0),Scalar(1)); - m1.array()(0,0) = 2; - VERIFY_IS_APPROX(m1(0,0),Scalar(2)); - m1.array().matrix().coeffRef(0,0) = 3; - VERIFY_IS_APPROX(m1(0,0),Scalar(3)); - m1.array().matrix()(0,0) = 4; - VERIFY_IS_APPROX(m1(0,0),Scalar(4)); + m1.array().coeffRef(0, 0) = 1; + VERIFY_IS_APPROX(m1(0, 0), Scalar(1)); + m1.array()(0, 0) = 2; + VERIFY_IS_APPROX(m1(0, 0), Scalar(2)); + m1.array().matrix().coeffRef(0, 0) = 3; + VERIFY_IS_APPROX(m1(0, 0), Scalar(3)); + m1.array().matrix()(0, 0) = 4; + VERIFY_IS_APPROX(m1(0, 0), Scalar(4)); } -template void comparisons(const MatrixType& m) +template void comparisons(const MatrixType &m) { using std::abs; typedef typename MatrixType::Scalar Scalar; @@ -88,87 +87,80 @@ template void comparisons(const MatrixType& m) Index rows = m.rows(); Index cols = m.cols(); - Index r = internal::random(0, rows-1), - c = internal::random(0, cols-1); + Index r = internal::random(0, rows - 1), c = internal::random(0, cols - 1); - MatrixType m1 = MatrixType::Random(rows, cols), - m2 = MatrixType::Random(rows, cols), - m3(rows, cols); + MatrixType m1 = MatrixType::Random(rows, cols), m2 = MatrixType::Random(rows, cols), m3(rows, cols); VERIFY(((m1.array() + Scalar(1)) > m1.array()).all()); VERIFY(((m1.array() - Scalar(1)) < m1.array()).all()); - if (rows*cols>1) - { + if (rows * cols > 1) { m3 = m1; - m3(r,c) += 1; - VERIFY(! (m1.array() < m3.array()).all() ); - VERIFY(! (m1.array() > m3.array()).all() ); + m3(r, c) += 1; + VERIFY(!(m1.array() < m3.array()).all()); + VERIFY(!(m1.array() > m3.array()).all()); } // comparisons to scalar - VERIFY( (m1.array() != (m1(r,c)+1) ).any() ); - VERIFY( (m1.array() > (m1(r,c)-1) ).any() ); - VERIFY( (m1.array() < (m1(r,c)+1) ).any() ); - VERIFY( (m1.array() == m1(r,c) ).any() ); - VERIFY( m1.cwiseEqual(m1(r,c)).any() ); + VERIFY((m1.array() != (m1(r, c) + 1)).any()); + VERIFY((m1.array() > (m1(r, c) - 1)).any()); + VERIFY((m1.array() < (m1(r, c) + 1)).any()); + VERIFY((m1.array() == m1(r, c)).any()); + VERIFY(m1.cwiseEqual(m1(r, c)).any()); // test Select - VERIFY_IS_APPROX( (m1.array()m2.array()).select(m1,m2), m1.cwiseMax(m2) ); - Scalar mid = (m1.cwiseAbs().minCoeff() + m1.cwiseAbs().maxCoeff())/Scalar(2); - for (int j=0; j m2.array()).select(m1, m2), m1.cwiseMax(m2)); + Scalar mid = (m1.cwiseAbs().minCoeff() + m1.cwiseAbs().maxCoeff()) / Scalar(2); + for (int j = 0; j < cols; ++j) + for (int i = 0; i < rows; ++i) m3(i, j) = abs(m1(i, j)) < mid ? 0 : m1(i, j); + VERIFY_IS_APPROX( + (m1.array().abs() < MatrixType::Constant(rows, cols, mid).array()).select(MatrixType::Zero(rows, cols), m1), m3); // shorter versions: - VERIFY_IS_APPROX( (m1.array().abs()=MatrixType::Constant(rows,cols,mid).array()) - .select(m1,0), m3); + VERIFY_IS_APPROX((m1.array().abs() < MatrixType::Constant(rows, cols, mid).array()).select(0, m1), m3); + VERIFY_IS_APPROX((m1.array().abs() >= MatrixType::Constant(rows, cols, mid).array()).select(m1, 0), m3); // even shorter version: - VERIFY_IS_APPROX( (m1.array().abs()RealScalar(0.1)).count() == rows*cols); + VERIFY(((m1.array().abs() + 1) > RealScalar(0.1)).count() == rows * cols); // and/or - VERIFY( ((m1.array()RealScalar(0)).matrix()).count() == 0); - VERIFY( ((m1.array()=RealScalar(0)).matrix()).count() == rows*cols); + VERIFY(((m1.array() < RealScalar(0)).matrix() && (m1.array() > RealScalar(0)).matrix()).count() == 0); + VERIFY(((m1.array() < RealScalar(0)).matrix() || (m1.array() >= RealScalar(0)).matrix()).count() == rows * cols); RealScalar a = m1.cwiseAbs().mean(); - VERIFY( ((m1.array()<-a).matrix() || (m1.array()>a).matrix()).count() == (m1.cwiseAbs().array()>a).count()); + VERIFY(((m1.array() < -a).matrix() || (m1.array() > a).matrix()).count() == (m1.cwiseAbs().array() > a).count()); typedef Matrix VectorOfIndices; // TODO allows colwise/rowwise for array - VERIFY_IS_APPROX(((m1.array().abs()+1)>RealScalar(0.1)).matrix().colwise().count(), VectorOfIndices::Constant(cols,rows).transpose()); - VERIFY_IS_APPROX(((m1.array().abs()+1)>RealScalar(0.1)).matrix().rowwise().count(), VectorOfIndices::Constant(rows, cols)); + VERIFY_IS_APPROX(((m1.array().abs() + 1) > RealScalar(0.1)).matrix().colwise().count(), + VectorOfIndices::Constant(cols, rows).transpose()); + VERIFY_IS_APPROX( + ((m1.array().abs() + 1) > RealScalar(0.1)).matrix().rowwise().count(), VectorOfIndices::Constant(rows, cols)); } -template void lpNorm(const VectorType& v) +template void lpNorm(const VectorType &v) { using std::sqrt; typedef typename VectorType::RealScalar RealScalar; VectorType u = VectorType::Random(v.size()); - if(v.size()==0) - { + if (v.size() == 0) { VERIFY_IS_APPROX(u.template lpNorm(), RealScalar(0)); VERIFY_IS_APPROX(u.template lpNorm<1>(), RealScalar(0)); VERIFY_IS_APPROX(u.template lpNorm<2>(), RealScalar(0)); VERIFY_IS_APPROX(u.template lpNorm<5>(), RealScalar(0)); - } - else - { + } else { VERIFY_IS_APPROX(u.template lpNorm(), u.cwiseAbs().maxCoeff()); } VERIFY_IS_APPROX(u.template lpNorm<1>(), u.cwiseAbs().sum()); VERIFY_IS_APPROX(u.template lpNorm<2>(), sqrt(u.array().abs().square().sum())); - VERIFY_IS_APPROX(numext::pow(u.template lpNorm<5>(), typename VectorType::RealScalar(5)), u.array().abs().pow(5).sum()); + VERIFY_IS_APPROX( + numext::pow(u.template lpNorm<5>(), typename VectorType::RealScalar(5)), u.array().abs().pow(5).sum()); } -template void cwise_min_max(const MatrixType& m) +template void cwise_min_max(const MatrixType &m) { typedef typename MatrixType::Scalar Scalar; @@ -181,66 +173,63 @@ template void cwise_min_max(const MatrixType& m) Scalar maxM1 = m1.maxCoeff(); Scalar minM1 = m1.minCoeff(); - VERIFY_IS_APPROX(MatrixType::Constant(rows,cols, minM1), m1.cwiseMin(MatrixType::Constant(rows,cols, minM1))); - VERIFY_IS_APPROX(m1, m1.cwiseMin(MatrixType::Constant(rows,cols, maxM1))); + VERIFY_IS_APPROX(MatrixType::Constant(rows, cols, minM1), m1.cwiseMin(MatrixType::Constant(rows, cols, minM1))); + VERIFY_IS_APPROX(m1, m1.cwiseMin(MatrixType::Constant(rows, cols, maxM1))); - VERIFY_IS_APPROX(MatrixType::Constant(rows,cols, maxM1), m1.cwiseMax(MatrixType::Constant(rows,cols, maxM1))); - VERIFY_IS_APPROX(m1, m1.cwiseMax(MatrixType::Constant(rows,cols, minM1))); + VERIFY_IS_APPROX(MatrixType::Constant(rows, cols, maxM1), m1.cwiseMax(MatrixType::Constant(rows, cols, maxM1))); + VERIFY_IS_APPROX(m1, m1.cwiseMax(MatrixType::Constant(rows, cols, minM1))); // min/max with scalar input - VERIFY_IS_APPROX(MatrixType::Constant(rows,cols, minM1), m1.cwiseMin( minM1)); + VERIFY_IS_APPROX(MatrixType::Constant(rows, cols, minM1), m1.cwiseMin(minM1)); VERIFY_IS_APPROX(m1, m1.cwiseMin(maxM1)); VERIFY_IS_APPROX(-m1, (-m1).cwiseMin(-minM1)); - VERIFY_IS_APPROX(-m1.array(), ((-m1).array().min)( -minM1)); + VERIFY_IS_APPROX(-m1.array(), ((-m1).array().min)(-minM1)); - VERIFY_IS_APPROX(MatrixType::Constant(rows,cols, maxM1), m1.cwiseMax( maxM1)); + VERIFY_IS_APPROX(MatrixType::Constant(rows, cols, maxM1), m1.cwiseMax(maxM1)); VERIFY_IS_APPROX(m1, m1.cwiseMax(minM1)); VERIFY_IS_APPROX(-m1, (-m1).cwiseMax(-maxM1)); VERIFY_IS_APPROX(-m1.array(), ((-m1).array().max)(-maxM1)); - VERIFY_IS_APPROX(MatrixType::Constant(rows,cols, minM1).array(), (m1.array().min)( minM1)); - VERIFY_IS_APPROX(m1.array(), (m1.array().min)( maxM1)); - - VERIFY_IS_APPROX(MatrixType::Constant(rows,cols, maxM1).array(), (m1.array().max)( maxM1)); - VERIFY_IS_APPROX(m1.array(), (m1.array().max)( minM1)); + VERIFY_IS_APPROX(MatrixType::Constant(rows, cols, minM1).array(), (m1.array().min)(minM1)); + VERIFY_IS_APPROX(m1.array(), (m1.array().min)(maxM1)); + VERIFY_IS_APPROX(MatrixType::Constant(rows, cols, maxM1).array(), (m1.array().max)(maxM1)); + VERIFY_IS_APPROX(m1.array(), (m1.array().max)(minM1)); } -template void resize(const MatrixTraits& t) +template void resize(const MatrixTraits &t) { typedef typename MatrixTraits::Scalar Scalar; - typedef Matrix MatrixType; - typedef Array Array2DType; - typedef Matrix VectorType; - typedef Array Array1DType; + typedef Matrix MatrixType; + typedef Array Array2DType; + typedef Matrix VectorType; + typedef Array Array1DType; Index rows = t.rows(), cols = t.cols(); - MatrixType m(rows,cols); + MatrixType m(rows, cols); VectorType v(rows); - Array2DType a2(rows,cols); + Array2DType a2(rows, cols); Array1DType a1(rows); - m.array().resize(rows+1,cols+1); - VERIFY(m.rows()==rows+1 && m.cols()==cols+1); - a2.matrix().resize(rows+1,cols+1); - VERIFY(a2.rows()==rows+1 && a2.cols()==cols+1); + m.array().resize(rows + 1, cols + 1); + VERIFY(m.rows() == rows + 1 && m.cols() == cols + 1); + a2.matrix().resize(rows + 1, cols + 1); + VERIFY(a2.rows() == rows + 1 && a2.cols() == cols + 1); v.array().resize(cols); - VERIFY(v.size()==cols); + VERIFY(v.size() == cols); a1.matrix().resize(cols); - VERIFY(a1.size()==cols); + VERIFY(a1.size() == cols); } -template -void regression_bug_654() +template void regression_bug_654() { ArrayXf a = RowVectorXf(3); - VectorXf v = Array(3); + VectorXf v = Array(3); } // Check propagation of LvalueBit through Array/Matrix-Wrapper -template -void regrrssion_bug_1410() +template void regrrssion_bug_1410() { const Matrix4i M; const Array4i A; @@ -249,52 +238,62 @@ void regrrssion_bug_1410() MatrixWrapper AM = A.matrix(); AM.row(0); - VERIFY((internal::traits >::Flags&LvalueBit)==0); - VERIFY((internal::traits >::Flags&LvalueBit)==0); + VERIFY((internal::traits>::Flags & LvalueBit) == 0); + VERIFY((internal::traits>::Flags & LvalueBit) == 0); - VERIFY((internal::traits >::Flags&LvalueBit)==LvalueBit); - VERIFY((internal::traits >::Flags&LvalueBit)==LvalueBit); + VERIFY((internal::traits>::Flags & LvalueBit) == LvalueBit); + VERIFY((internal::traits>::Flags & LvalueBit) == LvalueBit); } void test_array_for_matrix() { - for(int i = 0; i < g_repeat; i++) { - CALL_SUBTEST_1( array_for_matrix(Matrix()) ); - CALL_SUBTEST_2( array_for_matrix(Matrix2f()) ); - CALL_SUBTEST_3( array_for_matrix(Matrix4d()) ); - CALL_SUBTEST_4( array_for_matrix(MatrixXcf(internal::random(1,EIGEN_TEST_MAX_SIZE), internal::random(1,EIGEN_TEST_MAX_SIZE))) ); - CALL_SUBTEST_5( array_for_matrix(MatrixXf(internal::random(1,EIGEN_TEST_MAX_SIZE), internal::random(1,EIGEN_TEST_MAX_SIZE))) ); - CALL_SUBTEST_6( array_for_matrix(MatrixXi(internal::random(1,EIGEN_TEST_MAX_SIZE), internal::random(1,EIGEN_TEST_MAX_SIZE))) ); + for (int i = 0; i < g_repeat; i++) { + CALL_SUBTEST_1(array_for_matrix(Matrix())); + CALL_SUBTEST_2(array_for_matrix(Matrix2f())); + CALL_SUBTEST_3(array_for_matrix(Matrix4d())); + CALL_SUBTEST_4(array_for_matrix( + MatrixXcf(internal::random(1, EIGEN_TEST_MAX_SIZE), internal::random(1, EIGEN_TEST_MAX_SIZE)))); + CALL_SUBTEST_5(array_for_matrix( + MatrixXf(internal::random(1, EIGEN_TEST_MAX_SIZE), internal::random(1, EIGEN_TEST_MAX_SIZE)))); + CALL_SUBTEST_6(array_for_matrix( + MatrixXi(internal::random(1, EIGEN_TEST_MAX_SIZE), internal::random(1, EIGEN_TEST_MAX_SIZE)))); } - for(int i = 0; i < g_repeat; i++) { - CALL_SUBTEST_1( comparisons(Matrix()) ); - CALL_SUBTEST_2( comparisons(Matrix2f()) ); - CALL_SUBTEST_3( comparisons(Matrix4d()) ); - CALL_SUBTEST_5( comparisons(MatrixXf(internal::random(1,EIGEN_TEST_MAX_SIZE), internal::random(1,EIGEN_TEST_MAX_SIZE))) ); - CALL_SUBTEST_6( comparisons(MatrixXi(internal::random(1,EIGEN_TEST_MAX_SIZE), internal::random(1,EIGEN_TEST_MAX_SIZE))) ); + for (int i = 0; i < g_repeat; i++) { + CALL_SUBTEST_1(comparisons(Matrix())); + CALL_SUBTEST_2(comparisons(Matrix2f())); + CALL_SUBTEST_3(comparisons(Matrix4d())); + CALL_SUBTEST_5(comparisons( + MatrixXf(internal::random(1, EIGEN_TEST_MAX_SIZE), internal::random(1, EIGEN_TEST_MAX_SIZE)))); + CALL_SUBTEST_6(comparisons( + MatrixXi(internal::random(1, EIGEN_TEST_MAX_SIZE), internal::random(1, EIGEN_TEST_MAX_SIZE)))); } - for(int i = 0; i < g_repeat; i++) { - CALL_SUBTEST_1( cwise_min_max(Matrix()) ); - CALL_SUBTEST_2( cwise_min_max(Matrix2f()) ); - CALL_SUBTEST_3( cwise_min_max(Matrix4d()) ); - CALL_SUBTEST_5( cwise_min_max(MatrixXf(internal::random(1,EIGEN_TEST_MAX_SIZE), internal::random(1,EIGEN_TEST_MAX_SIZE))) ); - CALL_SUBTEST_6( cwise_min_max(MatrixXi(internal::random(1,EIGEN_TEST_MAX_SIZE), internal::random(1,EIGEN_TEST_MAX_SIZE))) ); + for (int i = 0; i < g_repeat; i++) { + CALL_SUBTEST_1(cwise_min_max(Matrix())); + CALL_SUBTEST_2(cwise_min_max(Matrix2f())); + CALL_SUBTEST_3(cwise_min_max(Matrix4d())); + CALL_SUBTEST_5(cwise_min_max( + MatrixXf(internal::random(1, EIGEN_TEST_MAX_SIZE), internal::random(1, EIGEN_TEST_MAX_SIZE)))); + CALL_SUBTEST_6(cwise_min_max( + MatrixXi(internal::random(1, EIGEN_TEST_MAX_SIZE), internal::random(1, EIGEN_TEST_MAX_SIZE)))); } - for(int i = 0; i < g_repeat; i++) { - CALL_SUBTEST_1( lpNorm(Matrix()) ); - CALL_SUBTEST_2( lpNorm(Vector2f()) ); - CALL_SUBTEST_7( lpNorm(Vector3d()) ); - CALL_SUBTEST_8( lpNorm(Vector4f()) ); - CALL_SUBTEST_5( lpNorm(VectorXf(internal::random(1,EIGEN_TEST_MAX_SIZE))) ); - CALL_SUBTEST_4( lpNorm(VectorXcf(internal::random(1,EIGEN_TEST_MAX_SIZE))) ); + for (int i = 0; i < g_repeat; i++) { + CALL_SUBTEST_1(lpNorm(Matrix())); + CALL_SUBTEST_2(lpNorm(Vector2f())); + CALL_SUBTEST_7(lpNorm(Vector3d())); + CALL_SUBTEST_8(lpNorm(Vector4f())); + CALL_SUBTEST_5(lpNorm(VectorXf(internal::random(1, EIGEN_TEST_MAX_SIZE)))); + CALL_SUBTEST_4(lpNorm(VectorXcf(internal::random(1, EIGEN_TEST_MAX_SIZE)))); } - CALL_SUBTEST_5( lpNorm(VectorXf(0)) ); - CALL_SUBTEST_4( lpNorm(VectorXcf(0)) ); - for(int i = 0; i < g_repeat; i++) { - CALL_SUBTEST_4( resize(MatrixXcf(internal::random(1,EIGEN_TEST_MAX_SIZE), internal::random(1,EIGEN_TEST_MAX_SIZE))) ); - CALL_SUBTEST_5( resize(MatrixXf(internal::random(1,EIGEN_TEST_MAX_SIZE), internal::random(1,EIGEN_TEST_MAX_SIZE))) ); - CALL_SUBTEST_6( resize(MatrixXi(internal::random(1,EIGEN_TEST_MAX_SIZE), internal::random(1,EIGEN_TEST_MAX_SIZE))) ); + CALL_SUBTEST_5(lpNorm(VectorXf(0))); + CALL_SUBTEST_4(lpNorm(VectorXcf(0))); + for (int i = 0; i < g_repeat; i++) { + CALL_SUBTEST_4( + resize(MatrixXcf(internal::random(1, EIGEN_TEST_MAX_SIZE), internal::random(1, EIGEN_TEST_MAX_SIZE)))); + CALL_SUBTEST_5( + resize(MatrixXf(internal::random(1, EIGEN_TEST_MAX_SIZE), internal::random(1, EIGEN_TEST_MAX_SIZE)))); + CALL_SUBTEST_6( + resize(MatrixXi(internal::random(1, EIGEN_TEST_MAX_SIZE), internal::random(1, EIGEN_TEST_MAX_SIZE)))); } - CALL_SUBTEST_6( regression_bug_654<0>() ); - CALL_SUBTEST_6( regrrssion_bug_1410<0>() ); + CALL_SUBTEST_6(regression_bug_654<0>()); + CALL_SUBTEST_6(regrrssion_bug_1410<0>()); } diff --git a/filmulator-gui/core/nlmeans/eigen/test/array_of_string.cpp b/filmulator-gui/core/nlmeans/eigen/test/array_of_string.cpp index e23b7c59..a3d7af91 100644 --- a/filmulator-gui/core/nlmeans/eigen/test/array_of_string.cpp +++ b/filmulator-gui/core/nlmeans/eigen/test/array_of_string.cpp @@ -11,7 +11,7 @@ void test_array_of_string() { - typedef Array ArrayXs; + typedef Array ArrayXs; ArrayXs a1(3), a2(3), a3(3), a3ref(3); a1 << "one", "two", "three"; a2 << "1", "2", "3"; @@ -20,13 +20,13 @@ void test_array_of_string() s1 << a1; VERIFY_IS_EQUAL(s1.str(), std::string(" one two three")); a3 = a1 + std::string(" (") + a2 + std::string(")"); - VERIFY((a3==a3ref).all()); + VERIFY((a3 == a3ref).all()); a3 = a1; a3 += std::string(" (") + a2 + std::string(")"); - VERIFY((a3==a3ref).all()); + VERIFY((a3 == a3ref).all()); a1.swap(a3); - VERIFY((a1==a3ref).all()); - VERIFY((a3!=a3ref).all()); + VERIFY((a1 == a3ref).all()); + VERIFY((a3 != a3ref).all()); } diff --git a/filmulator-gui/core/nlmeans/eigen/test/array_replicate.cpp b/filmulator-gui/core/nlmeans/eigen/test/array_replicate.cpp index 0dad5bac..73da0722 100644 --- a/filmulator-gui/core/nlmeans/eigen/test/array_replicate.cpp +++ b/filmulator-gui/core/nlmeans/eigen/test/array_replicate.cpp @@ -9,7 +9,7 @@ #include "main.h" -template void replicate(const MatrixType& m) +template void replicate(const MatrixType &m) { /* this test covers the following files: Replicate.cpp @@ -22,60 +22,53 @@ template void replicate(const MatrixType& m) Index rows = m.rows(); Index cols = m.cols(); - MatrixType m1 = MatrixType::Random(rows, cols), - m2 = MatrixType::Random(rows, cols); + MatrixType m1 = MatrixType::Random(rows, cols), m2 = MatrixType::Random(rows, cols); VectorType v1 = VectorType::Random(rows); MatrixX x1, x2; VectorX vx1; - int f1 = internal::random(1,10), - f2 = internal::random(1,10); - - x1.resize(rows*f1,cols*f2); - for(int j=0; j())); - - x2.resize(rows,3*cols); + int f1 = internal::random(1, 10), f2 = internal::random(1, 10); + + x1.resize(rows * f1, cols * f2); + for (int j = 0; j < f2; j++) + for (int i = 0; i < f1; i++) x1.block(i * rows, j * cols, rows, cols) = m1; + VERIFY_IS_APPROX(x1, m1.replicate(f1, f2)); + + x2.resize(2 * rows, 3 * cols); + x2 << m2, m2, m2, m2, m2, m2; + VERIFY_IS_APPROX(x2, (m2.template replicate<2, 3>())); + + x2.resize(rows, 3 * cols); x2 << m2, m2, m2; - VERIFY_IS_APPROX(x2, (m2.template replicate<1,3>())); - - vx1.resize(3*rows,cols); + VERIFY_IS_APPROX(x2, (m2.template replicate<1, 3>())); + + vx1.resize(3 * rows, cols); vx1 << m2, m2, m2; - VERIFY_IS_APPROX(vx1+vx1, vx1+(m2.template replicate<3,1>())); - - vx1=m2+(m2.colwise().replicate(1)); - - if(m2.cols()==1) - VERIFY_IS_APPROX(m2.coeff(0), (m2.template replicate<3,1>().coeff(m2.rows()))); - - x2.resize(rows,f1); - for (int j=0; j())); + + vx1 = m2 + (m2.colwise().replicate(1)); + + if (m2.cols() == 1) VERIFY_IS_APPROX(m2.coeff(0), (m2.template replicate<3, 1>().coeff(m2.rows()))); + + x2.resize(rows, f1); + for (int j = 0; j < f1; ++j) x2.col(j) = v1; VERIFY_IS_APPROX(x2, v1.rowwise().replicate(f1)); - vx1.resize(rows*f2); - for (int j=0; j()) ); - CALL_SUBTEST_2( replicate(Vector2f()) ); - CALL_SUBTEST_3( replicate(Vector3d()) ); - CALL_SUBTEST_4( replicate(Vector4f()) ); - CALL_SUBTEST_5( replicate(VectorXf(16)) ); - CALL_SUBTEST_6( replicate(VectorXcd(10)) ); + for (int i = 0; i < g_repeat; i++) { + CALL_SUBTEST_1(replicate(Matrix())); + CALL_SUBTEST_2(replicate(Vector2f())); + CALL_SUBTEST_3(replicate(Vector3d())); + CALL_SUBTEST_4(replicate(Vector4f())); + CALL_SUBTEST_5(replicate(VectorXf(16))); + CALL_SUBTEST_6(replicate(VectorXcd(10))); } } diff --git a/filmulator-gui/core/nlmeans/eigen/test/array_reverse.cpp b/filmulator-gui/core/nlmeans/eigen/test/array_reverse.cpp index 9d5b9a66..b682080b 100644 --- a/filmulator-gui/core/nlmeans/eigen/test/array_reverse.cpp +++ b/filmulator-gui/core/nlmeans/eigen/test/array_reverse.cpp @@ -13,7 +13,7 @@ using namespace std; -template void reverse(const MatrixType& m) +template void reverse(const MatrixType &m) { typedef typename MatrixType::Scalar Scalar; typedef Matrix VectorType; @@ -28,93 +28,76 @@ template void reverse(const MatrixType& m) MatrixType m1_r = m1.reverse(); // Verify that MatrixBase::reverse() works - for ( int i = 0; i < rows; i++ ) { - for ( int j = 0; j < cols; j++ ) { - VERIFY_IS_APPROX(m1_r(i, j), m1(rows - 1 - i, cols - 1 - j)); - } + for (int i = 0; i < rows; i++) { + for (int j = 0; j < cols; j++) { VERIFY_IS_APPROX(m1_r(i, j), m1(rows - 1 - i, cols - 1 - j)); } } Reverse m1_rd(m1); // Verify that a Reverse default (in both directions) of an expression works - for ( int i = 0; i < rows; i++ ) { - for ( int j = 0; j < cols; j++ ) { - VERIFY_IS_APPROX(m1_rd(i, j), m1(rows - 1 - i, cols - 1 - j)); - } + for (int i = 0; i < rows; i++) { + for (int j = 0; j < cols; j++) { VERIFY_IS_APPROX(m1_rd(i, j), m1(rows - 1 - i, cols - 1 - j)); } } Reverse m1_rb(m1); // Verify that a Reverse in both directions of an expression works - for ( int i = 0; i < rows; i++ ) { - for ( int j = 0; j < cols; j++ ) { - VERIFY_IS_APPROX(m1_rb(i, j), m1(rows - 1 - i, cols - 1 - j)); - } + for (int i = 0; i < rows; i++) { + for (int j = 0; j < cols; j++) { VERIFY_IS_APPROX(m1_rb(i, j), m1(rows - 1 - i, cols - 1 - j)); } } Reverse m1_rv(m1); // Verify that a Reverse in the vertical directions of an expression works - for ( int i = 0; i < rows; i++ ) { - for ( int j = 0; j < cols; j++ ) { - VERIFY_IS_APPROX(m1_rv(i, j), m1(rows - 1 - i, j)); - } + for (int i = 0; i < rows; i++) { + for (int j = 0; j < cols; j++) { VERIFY_IS_APPROX(m1_rv(i, j), m1(rows - 1 - i, j)); } } Reverse m1_rh(m1); // Verify that a Reverse in the horizontal directions of an expression works - for ( int i = 0; i < rows; i++ ) { - for ( int j = 0; j < cols; j++ ) { - VERIFY_IS_APPROX(m1_rh(i, j), m1(i, cols - 1 - j)); - } + for (int i = 0; i < rows; i++) { + for (int j = 0; j < cols; j++) { VERIFY_IS_APPROX(m1_rh(i, j), m1(i, cols - 1 - j)); } } VectorType v1_r = v1.reverse(); // Verify that a VectorType::reverse() of an expression works - for ( int i = 0; i < rows; i++ ) { - VERIFY_IS_APPROX(v1_r(i), v1(rows - 1 - i)); - } + for (int i = 0; i < rows; i++) { VERIFY_IS_APPROX(v1_r(i), v1(rows - 1 - i)); } MatrixType m1_cr = m1.colwise().reverse(); // Verify that PartialRedux::reverse() works (for colwise()) - for ( int i = 0; i < rows; i++ ) { - for ( int j = 0; j < cols; j++ ) { - VERIFY_IS_APPROX(m1_cr(i, j), m1(rows - 1 - i, j)); - } + for (int i = 0; i < rows; i++) { + for (int j = 0; j < cols; j++) { VERIFY_IS_APPROX(m1_cr(i, j), m1(rows - 1 - i, j)); } } MatrixType m1_rr = m1.rowwise().reverse(); // Verify that PartialRedux::reverse() works (for rowwise()) - for ( int i = 0; i < rows; i++ ) { - for ( int j = 0; j < cols; j++ ) { - VERIFY_IS_APPROX(m1_rr(i, j), m1(i, cols - 1 - j)); - } + for (int i = 0; i < rows; i++) { + for (int j = 0; j < cols; j++) { VERIFY_IS_APPROX(m1_rr(i, j), m1(i, cols - 1 - j)); } } Scalar x = internal::random(); - Index r = internal::random(0, rows-1), - c = internal::random(0, cols-1); + Index r = internal::random(0, rows - 1), c = internal::random(0, cols - 1); m1.reverse()(r, c) = x; VERIFY_IS_APPROX(x, m1(rows - 1 - r, cols - 1 - c)); - + m2 = m1; m2.reverseInPlace(); - VERIFY_IS_APPROX(m2,m1.reverse().eval()); - + VERIFY_IS_APPROX(m2, m1.reverse().eval()); + m2 = m1; m2.col(0).reverseInPlace(); - VERIFY_IS_APPROX(m2.col(0),m1.col(0).reverse().eval()); - + VERIFY_IS_APPROX(m2.col(0), m1.col(0).reverse().eval()); + m2 = m1; m2.row(0).reverseInPlace(); - VERIFY_IS_APPROX(m2.row(0),m1.row(0).reverse().eval()); - + VERIFY_IS_APPROX(m2.row(0), m1.row(0).reverse().eval()); + m2 = m1; m2.rowwise().reverseInPlace(); - VERIFY_IS_APPROX(m2,m1.rowwise().reverse().eval()); - + VERIFY_IS_APPROX(m2, m1.rowwise().reverse().eval()); + m2 = m1; m2.colwise().reverseInPlace(); - VERIFY_IS_APPROX(m2,m1.colwise().reverse().eval()); + VERIFY_IS_APPROX(m2, m1.colwise().reverse().eval()); m1.colwise().reverse()(r, c) = x; VERIFY_IS_APPROX(x, m1(rows - 1 - r, c)); @@ -125,20 +108,26 @@ template void reverse(const MatrixType& m) void test_array_reverse() { - for(int i = 0; i < g_repeat; i++) { - CALL_SUBTEST_1( reverse(Matrix()) ); - CALL_SUBTEST_2( reverse(Matrix2f()) ); - CALL_SUBTEST_3( reverse(Matrix4f()) ); - CALL_SUBTEST_4( reverse(Matrix4d()) ); - CALL_SUBTEST_5( reverse(MatrixXcf(internal::random(1,EIGEN_TEST_MAX_SIZE), internal::random(1,EIGEN_TEST_MAX_SIZE))) ); - CALL_SUBTEST_6( reverse(MatrixXi(internal::random(1,EIGEN_TEST_MAX_SIZE), internal::random(1,EIGEN_TEST_MAX_SIZE))) ); - CALL_SUBTEST_7( reverse(MatrixXcd(internal::random(1,EIGEN_TEST_MAX_SIZE), internal::random(1,EIGEN_TEST_MAX_SIZE))) ); - CALL_SUBTEST_8( reverse(Matrix()) ); - CALL_SUBTEST_9( reverse(Matrix(internal::random(1,EIGEN_TEST_MAX_SIZE), internal::random(1,EIGEN_TEST_MAX_SIZE))) ); + for (int i = 0; i < g_repeat; i++) { + CALL_SUBTEST_1(reverse(Matrix())); + CALL_SUBTEST_2(reverse(Matrix2f())); + CALL_SUBTEST_3(reverse(Matrix4f())); + CALL_SUBTEST_4(reverse(Matrix4d())); + CALL_SUBTEST_5( + reverse(MatrixXcf(internal::random(1, EIGEN_TEST_MAX_SIZE), internal::random(1, EIGEN_TEST_MAX_SIZE)))); + CALL_SUBTEST_6( + reverse(MatrixXi(internal::random(1, EIGEN_TEST_MAX_SIZE), internal::random(1, EIGEN_TEST_MAX_SIZE)))); + CALL_SUBTEST_7( + reverse(MatrixXcd(internal::random(1, EIGEN_TEST_MAX_SIZE), internal::random(1, EIGEN_TEST_MAX_SIZE)))); + CALL_SUBTEST_8(reverse(Matrix())); + CALL_SUBTEST_9(reverse(Matrix( + internal::random(1, EIGEN_TEST_MAX_SIZE), internal::random(1, EIGEN_TEST_MAX_SIZE)))); } #ifdef EIGEN_TEST_PART_3 - Vector4f x; x << 1, 2, 3, 4; - Vector4f y; y << 4, 3, 2, 1; + Vector4f x; + x << 1, 2, 3, 4; + Vector4f y; + y << 4, 3, 2, 1; VERIFY(x.reverse()[1] == 3); VERIFY(x.reverse() == y); #endif diff --git a/filmulator-gui/core/nlmeans/eigen/test/bandmatrix.cpp b/filmulator-gui/core/nlmeans/eigen/test/bandmatrix.cpp index f8c38f7c..ba7ebedd 100644 --- a/filmulator-gui/core/nlmeans/eigen/test/bandmatrix.cpp +++ b/filmulator-gui/core/nlmeans/eigen/test/bandmatrix.cpp @@ -9,63 +9,59 @@ #include "main.h" -template void bandmatrix(const MatrixType& _m) +template void bandmatrix(const MatrixType &_m) { typedef typename MatrixType::Scalar Scalar; typedef typename NumTraits::Real RealScalar; - typedef Matrix DenseMatrixType; + typedef Matrix DenseMatrixType; Index rows = _m.rows(); Index cols = _m.cols(); Index supers = _m.supers(); Index subs = _m.subs(); - MatrixType m(rows,cols,supers,subs); + MatrixType m(rows, cols, supers, subs); - DenseMatrixType dm1(rows,cols); + DenseMatrixType dm1(rows, cols); dm1.setZero(); m.diagonal().setConstant(123); dm1.diagonal().setConstant(123); - for (int i=1; i<=m.supers();++i) - { + for (int i = 1; i <= m.supers(); ++i) { m.diagonal(i).setConstant(static_cast(i)); dm1.diagonal(i).setConstant(static_cast(i)); } - for (int i=1; i<=m.subs();++i) - { + for (int i = 1; i <= m.subs(); ++i) { m.diagonal(-i).setConstant(-static_cast(i)); dm1.diagonal(-i).setConstant(-static_cast(i)); } - //std::cerr << m.m_data << "\n\n" << m.toDense() << "\n\n" << dm1 << "\n\n\n\n"; - VERIFY_IS_APPROX(dm1,m.toDenseMatrix()); + // std::cerr << m.m_data << "\n\n" << m.toDense() << "\n\n" << dm1 << "\n\n\n\n"; + VERIFY_IS_APPROX(dm1, m.toDenseMatrix()); - for (int i=0; i(i+1)); - dm1.col(i).setConstant(static_cast(i+1)); + for (int i = 0; i < cols; ++i) { + m.col(i).setConstant(static_cast(i + 1)); + dm1.col(i).setConstant(static_cast(i + 1)); } - Index d = (std::min)(rows,cols); - Index a = std::max(0,cols-d-supers); - Index b = std::max(0,rows-d-subs); - if(a>0) dm1.block(0,d+supers,rows,a).setZero(); - dm1.block(0,supers+1,cols-supers-1-a,cols-supers-1-a).template triangularView().setZero(); - dm1.block(subs+1,0,rows-subs-1-b,rows-subs-1-b).template triangularView().setZero(); - if(b>0) dm1.block(d+subs,0,b,cols).setZero(); - //std::cerr << m.m_data << "\n\n" << m.toDense() << "\n\n" << dm1 << "\n\n"; - VERIFY_IS_APPROX(dm1,m.toDenseMatrix()); - + Index d = (std::min)(rows, cols); + Index a = std::max(0, cols - d - supers); + Index b = std::max(0, rows - d - subs); + if (a > 0) dm1.block(0, d + supers, rows, a).setZero(); + dm1.block(0, supers + 1, cols - supers - 1 - a, cols - supers - 1 - a).template triangularView().setZero(); + dm1.block(subs + 1, 0, rows - subs - 1 - b, rows - subs - 1 - b).template triangularView().setZero(); + if (b > 0) dm1.block(d + subs, 0, b, cols).setZero(); + // std::cerr << m.m_data << "\n\n" << m.toDense() << "\n\n" << dm1 << "\n\n"; + VERIFY_IS_APPROX(dm1, m.toDenseMatrix()); } using Eigen::internal::BandMatrix; void test_bandmatrix() { - for(int i = 0; i < 10*g_repeat ; i++) { - Index rows = internal::random(1,10); - Index cols = internal::random(1,10); - Index sups = internal::random(0,cols-1); - Index subs = internal::random(0,rows-1); - CALL_SUBTEST(bandmatrix(BandMatrix(rows,cols,sups,subs)) ); + for (int i = 0; i < 10 * g_repeat; i++) { + Index rows = internal::random(1, 10); + Index cols = internal::random(1, 10); + Index sups = internal::random(0, cols - 1); + Index subs = internal::random(0, rows - 1); + CALL_SUBTEST(bandmatrix(BandMatrix(rows, cols, sups, subs))); } } diff --git a/filmulator-gui/core/nlmeans/eigen/test/basicstuff.cpp b/filmulator-gui/core/nlmeans/eigen/test/basicstuff.cpp index 2e532f7a..ea04eadb 100644 --- a/filmulator-gui/core/nlmeans/eigen/test/basicstuff.cpp +++ b/filmulator-gui/core/nlmeans/eigen/test/basicstuff.cpp @@ -11,7 +11,7 @@ #include "main.h" -template void basicStuff(const MatrixType& m) +template void basicStuff(const MatrixType &m) { typedef typename MatrixType::Scalar Scalar; typedef Matrix VectorType; @@ -22,25 +22,21 @@ template void basicStuff(const MatrixType& m) // this test relies a lot on Random.h, and there's not much more that we can do // to test it, hence I consider that we will have tested Random.h - MatrixType m1 = MatrixType::Random(rows, cols), - m2 = MatrixType::Random(rows, cols), - m3(rows, cols), + MatrixType m1 = MatrixType::Random(rows, cols), m2 = MatrixType::Random(rows, cols), m3(rows, cols), mzero = MatrixType::Zero(rows, cols), square = Matrix::Random(rows, rows); - VectorType v1 = VectorType::Random(rows), - vzero = VectorType::Zero(rows); - SquareMatrixType sm1 = SquareMatrixType::Random(rows,rows), sm2(rows,rows); + VectorType v1 = VectorType::Random(rows), vzero = VectorType::Zero(rows); + SquareMatrixType sm1 = SquareMatrixType::Random(rows, rows), sm2(rows, rows); Scalar x = 0; - while(x == Scalar(0)) x = internal::random(); + while (x == Scalar(0)) x = internal::random(); - Index r = internal::random(0, rows-1), - c = internal::random(0, cols-1); + Index r = internal::random(0, rows - 1), c = internal::random(0, cols - 1); - m1.coeffRef(r,c) = x; - VERIFY_IS_APPROX(x, m1.coeff(r,c)); - m1(r,c) = x; - VERIFY_IS_APPROX(x, m1(r,c)); + m1.coeffRef(r, c) = x; + VERIFY_IS_APPROX(x, m1.coeff(r, c)); + m1(r, c) = x; + VERIFY_IS_APPROX(x, m1(r, c)); v1.coeffRef(r) = x; VERIFY_IS_APPROX(x, v1.coeff(r)); v1(r) = x; @@ -48,17 +44,17 @@ template void basicStuff(const MatrixType& m) v1[r] = x; VERIFY_IS_APPROX(x, v1[r]); - VERIFY_IS_APPROX( v1, v1); - VERIFY_IS_NOT_APPROX( v1, 2*v1); - VERIFY_IS_MUCH_SMALLER_THAN( vzero, v1); - VERIFY_IS_MUCH_SMALLER_THAN( vzero, v1.squaredNorm()); - VERIFY_IS_NOT_MUCH_SMALLER_THAN(v1, v1); - VERIFY_IS_APPROX( vzero, v1-v1); - VERIFY_IS_APPROX( m1, m1); - VERIFY_IS_NOT_APPROX( m1, 2*m1); - VERIFY_IS_MUCH_SMALLER_THAN( mzero, m1); - VERIFY_IS_NOT_MUCH_SMALLER_THAN(m1, m1); - VERIFY_IS_APPROX( mzero, m1-m1); + VERIFY_IS_APPROX(v1, v1); + VERIFY_IS_NOT_APPROX(v1, 2 * v1); + VERIFY_IS_MUCH_SMALLER_THAN(vzero, v1); + VERIFY_IS_MUCH_SMALLER_THAN(vzero, v1.squaredNorm()); + VERIFY_IS_NOT_MUCH_SMALLER_THAN(v1, v1); + VERIFY_IS_APPROX(vzero, v1 - v1); + VERIFY_IS_APPROX(m1, m1); + VERIFY_IS_NOT_APPROX(m1, 2 * m1); + VERIFY_IS_MUCH_SMALLER_THAN(mzero, m1); + VERIFY_IS_NOT_MUCH_SMALLER_THAN(m1, m1); + VERIFY_IS_APPROX(mzero, m1 - m1); // always test operator() on each read-only expression class, // in order to check const-qualifiers. @@ -66,7 +62,7 @@ template void basicStuff(const MatrixType& m) // hence has no _write() method, the corresponding MatrixBase method (here zero()) // should return a const-qualified object so that it is the const-qualified // operator() that gets called, which in turn calls _read(). - VERIFY_IS_MUCH_SMALLER_THAN(MatrixType::Zero(rows,cols)(r,c), static_cast(1)); + VERIFY_IS_MUCH_SMALLER_THAN(MatrixType::Zero(rows, cols)(r, c), static_cast(1)); // now test copying a row-vector into a (column-)vector and conversely. square.col(r) = square.row(r).eval(); @@ -74,74 +70,74 @@ template void basicStuff(const MatrixType& m) Matrix cv(rows); rv = square.row(r); cv = square.col(r); - + VERIFY_IS_APPROX(rv, cv.transpose()); - if(cols!=1 && rows!=1 && MatrixType::SizeAtCompileTime!=Dynamic) - { - VERIFY_RAISES_ASSERT(m1 = (m2.block(0,0, rows-1, cols-1))); + if (cols != 1 && rows != 1 && MatrixType::SizeAtCompileTime != Dynamic) { + VERIFY_RAISES_ASSERT(m1 = (m2.block(0, 0, rows - 1, cols - 1))); } - if(cols!=1 && rows!=1) - { + if (cols != 1 && rows != 1) { VERIFY_RAISES_ASSERT(m1[0]); - VERIFY_RAISES_ASSERT((m1+m1)[0]); + VERIFY_RAISES_ASSERT((m1 + m1)[0]); } - VERIFY_IS_APPROX(m3 = m1,m1); + VERIFY_IS_APPROX(m3 = m1, m1); MatrixType m4; - VERIFY_IS_APPROX(m4 = m1,m1); + VERIFY_IS_APPROX(m4 = m1, m1); m3.real() = m1.real(); - VERIFY_IS_APPROX(static_cast(m3).real(), static_cast(m1).real()); - VERIFY_IS_APPROX(static_cast(m3).real(), m1.real()); + VERIFY_IS_APPROX(static_cast(m3).real(), static_cast(m1).real()); + VERIFY_IS_APPROX(static_cast(m3).real(), m1.real()); // check == / != operators - VERIFY(m1==m1); - VERIFY(m1!=m2); - VERIFY(!(m1==m2)); - VERIFY(!(m1!=m1)); + VERIFY(m1 == m1); + VERIFY(m1 != m2); + VERIFY(!(m1 == m2)); + VERIFY(!(m1 != m1)); m1 = m2; - VERIFY(m1==m2); - VERIFY(!(m1!=m2)); - + VERIFY(m1 == m2); + VERIFY(!(m1 != m2)); + // check automatic transposition sm2.setZero(); - for(typename MatrixType::Index i=0;i(0,10)>5; + bool b = internal::random(0, 10) > 5; m3 = b ? m1 : m2; - if(b) VERIFY_IS_APPROX(m3,m1); - else VERIFY_IS_APPROX(m3,m2); + if (b) + VERIFY_IS_APPROX(m3, m1); + else + VERIFY_IS_APPROX(m3, m2); m3 = b ? -m1 : m2; - if(b) VERIFY_IS_APPROX(m3,-m1); - else VERIFY_IS_APPROX(m3,m2); + if (b) + VERIFY_IS_APPROX(m3, -m1); + else + VERIFY_IS_APPROX(m3, m2); m3 = b ? m1 : -m2; - if(b) VERIFY_IS_APPROX(m3,m1); - else VERIFY_IS_APPROX(m3,-m2); + if (b) + VERIFY_IS_APPROX(m3, m1); + else + VERIFY_IS_APPROX(m3, -m2); } } -template void basicStuffComplex(const MatrixType& m) +template void basicStuffComplex(const MatrixType &m) { typedef typename MatrixType::Scalar Scalar; typedef typename NumTraits::Real RealScalar; @@ -150,32 +146,30 @@ template void basicStuffComplex(const MatrixType& m) Index rows = m.rows(); Index cols = m.cols(); - Scalar s1 = internal::random(), - s2 = internal::random(); + Scalar s1 = internal::random(), s2 = internal::random(); - VERIFY(numext::real(s1)==numext::real_ref(s1)); - VERIFY(numext::imag(s1)==numext::imag_ref(s1)); + VERIFY(numext::real(s1) == numext::real_ref(s1)); + VERIFY(numext::imag(s1) == numext::imag_ref(s1)); numext::real_ref(s1) = numext::real(s2); numext::imag_ref(s1) = numext::imag(s2); VERIFY(internal::isApprox(s1, s2, NumTraits::epsilon())); // extended precision in Intel FPUs means that s1 == s2 in the line above is not guaranteed. - RealMatrixType rm1 = RealMatrixType::Random(rows,cols), - rm2 = RealMatrixType::Random(rows,cols); - MatrixType cm(rows,cols); + RealMatrixType rm1 = RealMatrixType::Random(rows, cols), rm2 = RealMatrixType::Random(rows, cols); + MatrixType cm(rows, cols); cm.real() = rm1; cm.imag() = rm2; - VERIFY_IS_APPROX(static_cast(cm).real(), rm1); - VERIFY_IS_APPROX(static_cast(cm).imag(), rm2); + VERIFY_IS_APPROX(static_cast(cm).real(), rm1); + VERIFY_IS_APPROX(static_cast(cm).imag(), rm2); rm1.setZero(); rm2.setZero(); rm1 = cm.real(); rm2 = cm.imag(); - VERIFY_IS_APPROX(static_cast(cm).real(), rm1); - VERIFY_IS_APPROX(static_cast(cm).imag(), rm2); + VERIFY_IS_APPROX(static_cast(cm).real(), rm1); + VERIFY_IS_APPROX(static_cast(cm).imag(), rm2); cm.real().setZero(); - VERIFY(static_cast(cm).real().isZero()); - VERIFY(!static_cast(cm).imag().isZero()); + VERIFY(static_cast(cm).real().isZero()); + VERIFY(!static_cast(cm).imag().isZero()); } #ifdef EIGEN_TEST_PART_2 @@ -184,62 +178,58 @@ void casting() Matrix4f m = Matrix4f::Random(), m2; Matrix4d n = m.cast(); VERIFY(m.isApprox(n.cast())); - m2 = m.cast(); // check the specialization when NewType == Type + m2 = m.cast();// check the specialization when NewType == Type VERIFY(m.isApprox(m2)); } #endif -template -void fixedSizeMatrixConstruction() +template void fixedSizeMatrixConstruction() { Scalar raw[4]; - for(int k=0; k<4; ++k) - raw[k] = internal::random(); - + for (int k = 0; k < 4; ++k) raw[k] = internal::random(); + { - Matrix m(raw); - Array a(raw); - for(int k=0; k<4; ++k) VERIFY(m(k) == raw[k]); - for(int k=0; k<4; ++k) VERIFY(a(k) == raw[k]); - VERIFY_IS_EQUAL(m,(Matrix(raw[0],raw[1],raw[2],raw[3]))); - VERIFY((a==(Array(raw[0],raw[1],raw[2],raw[3]))).all()); + Matrix m(raw); + Array a(raw); + for (int k = 0; k < 4; ++k) VERIFY(m(k) == raw[k]); + for (int k = 0; k < 4; ++k) VERIFY(a(k) == raw[k]); + VERIFY_IS_EQUAL(m, (Matrix(raw[0], raw[1], raw[2], raw[3]))); + VERIFY((a == (Array(raw[0], raw[1], raw[2], raw[3]))).all()); } { - Matrix m(raw); - Array a(raw); - for(int k=0; k<3; ++k) VERIFY(m(k) == raw[k]); - for(int k=0; k<3; ++k) VERIFY(a(k) == raw[k]); - VERIFY_IS_EQUAL(m,(Matrix(raw[0],raw[1],raw[2]))); - VERIFY((a==Array(raw[0],raw[1],raw[2])).all()); + Matrix m(raw); + Array a(raw); + for (int k = 0; k < 3; ++k) VERIFY(m(k) == raw[k]); + for (int k = 0; k < 3; ++k) VERIFY(a(k) == raw[k]); + VERIFY_IS_EQUAL(m, (Matrix(raw[0], raw[1], raw[2]))); + VERIFY((a == Array(raw[0], raw[1], raw[2])).all()); } { - Matrix m(raw), m2( (DenseIndex(raw[0])), (DenseIndex(raw[1])) ); - Array a(raw), a2( (DenseIndex(raw[0])), (DenseIndex(raw[1])) ); - for(int k=0; k<2; ++k) VERIFY(m(k) == raw[k]); - for(int k=0; k<2; ++k) VERIFY(a(k) == raw[k]); - VERIFY_IS_EQUAL(m,(Matrix(raw[0],raw[1]))); - VERIFY((a==Array(raw[0],raw[1])).all()); - for(int k=0; k<2; ++k) VERIFY(m2(k) == DenseIndex(raw[k])); - for(int k=0; k<2; ++k) VERIFY(a2(k) == DenseIndex(raw[k])); + Matrix m(raw), m2((DenseIndex(raw[0])), (DenseIndex(raw[1]))); + Array a(raw), a2((DenseIndex(raw[0])), (DenseIndex(raw[1]))); + for (int k = 0; k < 2; ++k) VERIFY(m(k) == raw[k]); + for (int k = 0; k < 2; ++k) VERIFY(a(k) == raw[k]); + VERIFY_IS_EQUAL(m, (Matrix(raw[0], raw[1]))); + VERIFY((a == Array(raw[0], raw[1])).all()); + for (int k = 0; k < 2; ++k) VERIFY(m2(k) == DenseIndex(raw[k])); + for (int k = 0; k < 2; ++k) VERIFY(a2(k) == DenseIndex(raw[k])); } { - Matrix m(raw), - m2( (DenseIndex(raw[0])), (DenseIndex(raw[1])) ), - m3( (int(raw[0])), (int(raw[1])) ), - m4( (float(raw[0])), (float(raw[1])) ); - Array a(raw), a2( (DenseIndex(raw[0])), (DenseIndex(raw[1])) ); - for(int k=0; k<2; ++k) VERIFY(m(k) == raw[k]); - for(int k=0; k<2; ++k) VERIFY(a(k) == raw[k]); - VERIFY_IS_EQUAL(m,(Matrix(raw[0],raw[1]))); - VERIFY((a==Array(raw[0],raw[1])).all()); - for(int k=0; k<2; ++k) VERIFY(m2(k) == DenseIndex(raw[k])); - for(int k=0; k<2; ++k) VERIFY(a2(k) == DenseIndex(raw[k])); - for(int k=0; k<2; ++k) VERIFY(m3(k) == int(raw[k])); - for(int k=0; k<2; ++k) VERIFY((m4(k)) == Scalar(float(raw[k]))); + Matrix m(raw), m2((DenseIndex(raw[0])), (DenseIndex(raw[1]))), m3((int(raw[0])), (int(raw[1]))), + m4((float(raw[0])), (float(raw[1]))); + Array a(raw), a2((DenseIndex(raw[0])), (DenseIndex(raw[1]))); + for (int k = 0; k < 2; ++k) VERIFY(m(k) == raw[k]); + for (int k = 0; k < 2; ++k) VERIFY(a(k) == raw[k]); + VERIFY_IS_EQUAL(m, (Matrix(raw[0], raw[1]))); + VERIFY((a == Array(raw[0], raw[1])).all()); + for (int k = 0; k < 2; ++k) VERIFY(m2(k) == DenseIndex(raw[k])); + for (int k = 0; k < 2; ++k) VERIFY(a2(k) == DenseIndex(raw[k])); + for (int k = 0; k < 2; ++k) VERIFY(m3(k) == int(raw[k])); + for (int k = 0; k < 2; ++k) VERIFY((m4(k)) == Scalar(float(raw[k]))); } { - Matrix m(raw), m1(raw[0]), m2( (DenseIndex(raw[0])) ), m3( (int(raw[0])) ); - Array a(raw), a1(raw[0]), a2( (DenseIndex(raw[0])) ); + Matrix m(raw), m1(raw[0]), m2((DenseIndex(raw[0]))), m3((int(raw[0]))); + Array a(raw), a1(raw[0]), a2((DenseIndex(raw[0]))); VERIFY(m(0) == raw[0]); VERIFY(a(0) == raw[0]); VERIFY(m1(0) == raw[0]); @@ -247,24 +237,30 @@ void fixedSizeMatrixConstruction() VERIFY(m2(0) == DenseIndex(raw[0])); VERIFY(a2(0) == DenseIndex(raw[0])); VERIFY(m3(0) == int(raw[0])); - VERIFY_IS_EQUAL(m,(Matrix(raw[0]))); - VERIFY((a==Array(raw[0])).all()); + VERIFY_IS_EQUAL(m, (Matrix(raw[0]))); + VERIFY((a == Array(raw[0])).all()); } } void test_basicstuff() { - for(int i = 0; i < g_repeat; i++) { - CALL_SUBTEST_1( basicStuff(Matrix()) ); - CALL_SUBTEST_2( basicStuff(Matrix4d()) ); - CALL_SUBTEST_3( basicStuff(MatrixXcf(internal::random(1,EIGEN_TEST_MAX_SIZE), internal::random(1,EIGEN_TEST_MAX_SIZE))) ); - CALL_SUBTEST_4( basicStuff(MatrixXi(internal::random(1,EIGEN_TEST_MAX_SIZE), internal::random(1,EIGEN_TEST_MAX_SIZE))) ); - CALL_SUBTEST_5( basicStuff(MatrixXcd(internal::random(1,EIGEN_TEST_MAX_SIZE), internal::random(1,EIGEN_TEST_MAX_SIZE))) ); - CALL_SUBTEST_6( basicStuff(Matrix()) ); - CALL_SUBTEST_7( basicStuff(Matrix(internal::random(1,EIGEN_TEST_MAX_SIZE),internal::random(1,EIGEN_TEST_MAX_SIZE))) ); - - CALL_SUBTEST_3( basicStuffComplex(MatrixXcf(internal::random(1,EIGEN_TEST_MAX_SIZE), internal::random(1,EIGEN_TEST_MAX_SIZE))) ); - CALL_SUBTEST_5( basicStuffComplex(MatrixXcd(internal::random(1,EIGEN_TEST_MAX_SIZE), internal::random(1,EIGEN_TEST_MAX_SIZE))) ); + for (int i = 0; i < g_repeat; i++) { + CALL_SUBTEST_1(basicStuff(Matrix())); + CALL_SUBTEST_2(basicStuff(Matrix4d())); + CALL_SUBTEST_3(basicStuff( + MatrixXcf(internal::random(1, EIGEN_TEST_MAX_SIZE), internal::random(1, EIGEN_TEST_MAX_SIZE)))); + CALL_SUBTEST_4(basicStuff( + MatrixXi(internal::random(1, EIGEN_TEST_MAX_SIZE), internal::random(1, EIGEN_TEST_MAX_SIZE)))); + CALL_SUBTEST_5(basicStuff( + MatrixXcd(internal::random(1, EIGEN_TEST_MAX_SIZE), internal::random(1, EIGEN_TEST_MAX_SIZE)))); + CALL_SUBTEST_6(basicStuff(Matrix())); + CALL_SUBTEST_7(basicStuff(Matrix( + internal::random(1, EIGEN_TEST_MAX_SIZE), internal::random(1, EIGEN_TEST_MAX_SIZE)))); + + CALL_SUBTEST_3(basicStuffComplex( + MatrixXcf(internal::random(1, EIGEN_TEST_MAX_SIZE), internal::random(1, EIGEN_TEST_MAX_SIZE)))); + CALL_SUBTEST_5(basicStuffComplex( + MatrixXcd(internal::random(1, EIGEN_TEST_MAX_SIZE), internal::random(1, EIGEN_TEST_MAX_SIZE)))); } CALL_SUBTEST_1(fixedSizeMatrixConstruction()); diff --git a/filmulator-gui/core/nlmeans/eigen/test/bdcsvd.cpp b/filmulator-gui/core/nlmeans/eigen/test/bdcsvd.cpp index 6c7b0969..276fa56c 100644 --- a/filmulator-gui/core/nlmeans/eigen/test/bdcsvd.cpp +++ b/filmulator-gui/core/nlmeans/eigen/test/bdcsvd.cpp @@ -15,9 +15,9 @@ #define EIGEN_RUNTIME_NO_MALLOC #include "main.h" +#include #include #include -#include #define SVD_DEFAULT(M) BDCSVD @@ -25,18 +25,15 @@ #include "svd_common.h" // Check all variants of JacobiSVD -template -void bdcsvd(const MatrixType& a = MatrixType(), bool pickrandom = true) +template void bdcsvd(const MatrixType &a = MatrixType(), bool pickrandom = true) { MatrixType m = a; - if(pickrandom) - svd_fill_random(m); + if (pickrandom) svd_fill_random(m); - CALL_SUBTEST(( svd_test_all_computation_options >(m, false) )); + CALL_SUBTEST((svd_test_all_computation_options>(m, false))); } -template -void bdcsvd_method() +template void bdcsvd_method() { enum { Size = MatrixType::RowsAtCompileTime }; typedef typename MatrixType::RealScalar RealScalar; @@ -45,68 +42,66 @@ void bdcsvd_method() VERIFY_IS_APPROX(m.bdcSvd().singularValues(), RealVecType::Ones()); VERIFY_RAISES_ASSERT(m.bdcSvd().matrixU()); VERIFY_RAISES_ASSERT(m.bdcSvd().matrixV()); - VERIFY_IS_APPROX(m.bdcSvd(ComputeFullU|ComputeFullV).solve(m), m); + VERIFY_IS_APPROX(m.bdcSvd(ComputeFullU | ComputeFullV).solve(m), m); } // compare the Singular values returned with Jacobi and Bdc -template -void compare_bdc_jacobi(const MatrixType& a = MatrixType(), unsigned int computationOptions = 0) +template +void compare_bdc_jacobi(const MatrixType &a = MatrixType(), unsigned int computationOptions = 0) { MatrixType m = MatrixType::Random(a.rows(), a.cols()); BDCSVD bdc_svd(m); JacobiSVD jacobi_svd(m); VERIFY_IS_APPROX(bdc_svd.singularValues(), jacobi_svd.singularValues()); - if(computationOptions & ComputeFullU) VERIFY_IS_APPROX(bdc_svd.matrixU(), jacobi_svd.matrixU()); - if(computationOptions & ComputeThinU) VERIFY_IS_APPROX(bdc_svd.matrixU(), jacobi_svd.matrixU()); - if(computationOptions & ComputeFullV) VERIFY_IS_APPROX(bdc_svd.matrixV(), jacobi_svd.matrixV()); - if(computationOptions & ComputeThinV) VERIFY_IS_APPROX(bdc_svd.matrixV(), jacobi_svd.matrixV()); + if (computationOptions & ComputeFullU) VERIFY_IS_APPROX(bdc_svd.matrixU(), jacobi_svd.matrixU()); + if (computationOptions & ComputeThinU) VERIFY_IS_APPROX(bdc_svd.matrixU(), jacobi_svd.matrixU()); + if (computationOptions & ComputeFullV) VERIFY_IS_APPROX(bdc_svd.matrixV(), jacobi_svd.matrixV()); + if (computationOptions & ComputeThinV) VERIFY_IS_APPROX(bdc_svd.matrixV(), jacobi_svd.matrixV()); } void test_bdcsvd() { - CALL_SUBTEST_3(( svd_verify_assert >(Matrix3f()) )); - CALL_SUBTEST_4(( svd_verify_assert >(Matrix4d()) )); - CALL_SUBTEST_7(( svd_verify_assert >(MatrixXf(10,12)) )); - CALL_SUBTEST_8(( svd_verify_assert >(MatrixXcd(7,5)) )); - - CALL_SUBTEST_101(( svd_all_trivial_2x2(bdcsvd) )); - CALL_SUBTEST_102(( svd_all_trivial_2x2(bdcsvd) )); - - for(int i = 0; i < g_repeat; i++) { - CALL_SUBTEST_3(( bdcsvd() )); - CALL_SUBTEST_4(( bdcsvd() )); - CALL_SUBTEST_5(( bdcsvd >() )); - - int r = internal::random(1, EIGEN_TEST_MAX_SIZE/2), - c = internal::random(1, EIGEN_TEST_MAX_SIZE/2); - + CALL_SUBTEST_3((svd_verify_assert>(Matrix3f()))); + CALL_SUBTEST_4((svd_verify_assert>(Matrix4d()))); + CALL_SUBTEST_7((svd_verify_assert>(MatrixXf(10, 12)))); + CALL_SUBTEST_8((svd_verify_assert>(MatrixXcd(7, 5)))); + + CALL_SUBTEST_101((svd_all_trivial_2x2(bdcsvd))); + CALL_SUBTEST_102((svd_all_trivial_2x2(bdcsvd))); + + for (int i = 0; i < g_repeat; i++) { + CALL_SUBTEST_3((bdcsvd())); + CALL_SUBTEST_4((bdcsvd())); + CALL_SUBTEST_5((bdcsvd>())); + + int r = internal::random(1, EIGEN_TEST_MAX_SIZE / 2), c = internal::random(1, EIGEN_TEST_MAX_SIZE / 2); + TEST_SET_BUT_UNUSED_VARIABLE(r) TEST_SET_BUT_UNUSED_VARIABLE(c) - - CALL_SUBTEST_6(( bdcsvd(Matrix(r,2)) )); - CALL_SUBTEST_7(( bdcsvd(MatrixXf(r,c)) )); - CALL_SUBTEST_7(( compare_bdc_jacobi(MatrixXf(r,c)) )); - CALL_SUBTEST_10(( bdcsvd(MatrixXd(r,c)) )); - CALL_SUBTEST_10(( compare_bdc_jacobi(MatrixXd(r,c)) )); - CALL_SUBTEST_8(( bdcsvd(MatrixXcd(r,c)) )); - CALL_SUBTEST_8(( compare_bdc_jacobi(MatrixXcd(r,c)) )); + + CALL_SUBTEST_6((bdcsvd(Matrix(r, 2)))); + CALL_SUBTEST_7((bdcsvd(MatrixXf(r, c)))); + CALL_SUBTEST_7((compare_bdc_jacobi(MatrixXf(r, c)))); + CALL_SUBTEST_10((bdcsvd(MatrixXd(r, c)))); + CALL_SUBTEST_10((compare_bdc_jacobi(MatrixXd(r, c)))); + CALL_SUBTEST_8((bdcsvd(MatrixXcd(r, c)))); + CALL_SUBTEST_8((compare_bdc_jacobi(MatrixXcd(r, c)))); // Test on inf/nan matrix - CALL_SUBTEST_7( (svd_inf_nan, MatrixXf>()) ); - CALL_SUBTEST_10( (svd_inf_nan, MatrixXd>()) ); + CALL_SUBTEST_7((svd_inf_nan, MatrixXf>())); + CALL_SUBTEST_10((svd_inf_nan, MatrixXd>())); } // test matrixbase method - CALL_SUBTEST_1(( bdcsvd_method() )); - CALL_SUBTEST_3(( bdcsvd_method() )); + CALL_SUBTEST_1((bdcsvd_method())); + CALL_SUBTEST_3((bdcsvd_method())); // Test problem size constructors - CALL_SUBTEST_7( BDCSVD(10,10) ); + CALL_SUBTEST_7(BDCSVD(10, 10)); // Check that preallocation avoids subsequent mallocs // Disbaled because not supported by BDCSVD // CALL_SUBTEST_9( svd_preallocate() ); - CALL_SUBTEST_2( svd_underoverflow() ); + CALL_SUBTEST_2(svd_underoverflow()); } - diff --git a/filmulator-gui/core/nlmeans/eigen/test/bicgstab.cpp b/filmulator-gui/core/nlmeans/eigen/test/bicgstab.cpp index 4cc0dd31..0a3680aa 100644 --- a/filmulator-gui/core/nlmeans/eigen/test/bicgstab.cpp +++ b/filmulator-gui/core/nlmeans/eigen/test/bicgstab.cpp @@ -12,23 +12,23 @@ template void test_bicgstab_T() { - BiCGSTAB, DiagonalPreconditioner > bicgstab_colmajor_diag; - BiCGSTAB, IdentityPreconditioner > bicgstab_colmajor_I; - BiCGSTAB, IncompleteLUT > bicgstab_colmajor_ilut; - //BiCGSTAB, SSORPreconditioner > bicgstab_colmajor_ssor; + BiCGSTAB, DiagonalPreconditioner> bicgstab_colmajor_diag; + BiCGSTAB, IdentityPreconditioner> bicgstab_colmajor_I; + BiCGSTAB, IncompleteLUT> bicgstab_colmajor_ilut; + // BiCGSTAB, SSORPreconditioner > bicgstab_colmajor_ssor; - bicgstab_colmajor_diag.setTolerance(NumTraits::epsilon()*4); - bicgstab_colmajor_ilut.setTolerance(NumTraits::epsilon()*4); - - CALL_SUBTEST( check_sparse_square_solving(bicgstab_colmajor_diag) ); -// CALL_SUBTEST( check_sparse_square_solving(bicgstab_colmajor_I) ); - CALL_SUBTEST( check_sparse_square_solving(bicgstab_colmajor_ilut) ); - //CALL_SUBTEST( check_sparse_square_solving(bicgstab_colmajor_ssor) ); + bicgstab_colmajor_diag.setTolerance(NumTraits::epsilon() * 4); + bicgstab_colmajor_ilut.setTolerance(NumTraits::epsilon() * 4); + + CALL_SUBTEST(check_sparse_square_solving(bicgstab_colmajor_diag)); + // CALL_SUBTEST( check_sparse_square_solving(bicgstab_colmajor_I) ); + CALL_SUBTEST(check_sparse_square_solving(bicgstab_colmajor_ilut)); + // CALL_SUBTEST( check_sparse_square_solving(bicgstab_colmajor_ssor) ); } void test_bicgstab() { - CALL_SUBTEST_1((test_bicgstab_T()) ); + CALL_SUBTEST_1((test_bicgstab_T())); CALL_SUBTEST_2((test_bicgstab_T, int>())); - CALL_SUBTEST_3((test_bicgstab_T())); + CALL_SUBTEST_3((test_bicgstab_T())); } diff --git a/filmulator-gui/core/nlmeans/eigen/test/block.cpp b/filmulator-gui/core/nlmeans/eigen/test/block.cpp index ca9c21fe..3ebd85f3 100644 --- a/filmulator-gui/core/nlmeans/eigen/test/block.cpp +++ b/filmulator-gui/core/nlmeans/eigen/test/block.cpp @@ -7,60 +7,63 @@ // Public License v. 2.0. If a copy of the MPL was not distributed // with this file, You can obtain one at http://mozilla.org/MPL/2.0/. -#define EIGEN_NO_STATIC_ASSERT // otherwise we fail at compile time on unused paths +#define EIGEN_NO_STATIC_ASSERT// otherwise we fail at compile time on unused paths #include "main.h" template -typename Eigen::internal::enable_if::IsComplex,typename MatrixType::Scalar>::type -block_real_only(const MatrixType &m1, Index r1, Index r2, Index c1, Index c2, const Scalar& s1) { +typename Eigen::internal::enable_if::IsComplex, + typename MatrixType::Scalar>::type + block_real_only(const MatrixType &m1, Index r1, Index r2, Index c1, Index c2, const Scalar &s1) +{ // check cwise-Functions: VERIFY_IS_APPROX(m1.row(r1).cwiseMax(s1), m1.cwiseMax(s1).row(r1)); VERIFY_IS_APPROX(m1.col(c1).cwiseMin(s1), m1.cwiseMin(s1).col(c1)); - VERIFY_IS_APPROX(m1.block(r1,c1,r2-r1+1,c2-c1+1).cwiseMin(s1), m1.cwiseMin(s1).block(r1,c1,r2-r1+1,c2-c1+1)); - VERIFY_IS_APPROX(m1.block(r1,c1,r2-r1+1,c2-c1+1).cwiseMax(s1), m1.cwiseMax(s1).block(r1,c1,r2-r1+1,c2-c1+1)); - + VERIFY_IS_APPROX( + m1.block(r1, c1, r2 - r1 + 1, c2 - c1 + 1).cwiseMin(s1), m1.cwiseMin(s1).block(r1, c1, r2 - r1 + 1, c2 - c1 + 1)); + VERIFY_IS_APPROX( + m1.block(r1, c1, r2 - r1 + 1, c2 - c1 + 1).cwiseMax(s1), m1.cwiseMax(s1).block(r1, c1, r2 - r1 + 1, c2 - c1 + 1)); + return Scalar(0); } template -typename Eigen::internal::enable_if::IsComplex,typename MatrixType::Scalar>::type -block_real_only(const MatrixType &, Index, Index, Index, Index, const Scalar&) { +typename Eigen::internal::enable_if::IsComplex, + typename MatrixType::Scalar>::type + block_real_only(const MatrixType &, Index, Index, Index, Index, const Scalar &) +{ return Scalar(0); } -template void block(const MatrixType& m) +template void block(const MatrixType &m) { typedef typename MatrixType::Scalar Scalar; typedef typename MatrixType::RealScalar RealScalar; typedef Matrix VectorType; typedef Matrix RowVectorType; - typedef Matrix DynamicMatrixType; + typedef Matrix DynamicMatrixType; typedef Matrix DynamicVectorType; - + Index rows = m.rows(); Index cols = m.cols(); - MatrixType m1 = MatrixType::Random(rows, cols), - m1_copy = m1, - m2 = MatrixType::Random(rows, cols), - m3(rows, cols), + MatrixType m1 = MatrixType::Random(rows, cols), m1_copy = m1, m2 = MatrixType::Random(rows, cols), m3(rows, cols), ones = MatrixType::Ones(rows, cols); VectorType v1 = VectorType::Random(rows); Scalar s1 = internal::random(); - Index r1 = internal::random(0,rows-1); - Index r2 = internal::random(r1,rows-1); - Index c1 = internal::random(0,cols-1); - Index c2 = internal::random(c1,cols-1); + Index r1 = internal::random(0, rows - 1); + Index r2 = internal::random(r1, rows - 1); + Index c1 = internal::random(0, cols - 1); + Index c2 = internal::random(c1, cols - 1); block_real_only(m1, r1, r2, c1, c1, s1); - //check row() and col() + // check row() and col() VERIFY_IS_EQUAL(m1.col(c1).transpose(), m1.transpose().row(c1)); - //check operator(), both constant and non-constant, on row() and col() + // check operator(), both constant and non-constant, on row() and col() m1 = m1_copy; m1.row(r1) += s1 * m1_copy.row(r2); VERIFY_IS_APPROX(m1.row(r1), m1_copy.row(r1) + s1 * m1_copy.row(r2)); @@ -72,55 +75,51 @@ template void block(const MatrixType& m) VERIFY_IS_APPROX(m1.col(c1), m1_copy.col(c1) + s1 * m1_copy.col(c2)); m1.col(c1).col(0) += s1 * m1_copy.col(c2); VERIFY_IS_APPROX(m1.col(c1), m1_copy.col(c1) + Scalar(2) * s1 * m1_copy.col(c2)); - - - //check block() - Matrix b1(1,1); b1(0,0) = m1(r1,c1); - - RowVectorType br1(m1.block(r1,0,1,cols)); - VectorType bc1(m1.block(0,c1,rows,1)); - VERIFY_IS_EQUAL(b1, m1.block(r1,c1,1,1)); + + + // check block() + Matrix b1(1, 1); + b1(0, 0) = m1(r1, c1); + + RowVectorType br1(m1.block(r1, 0, 1, cols)); + VectorType bc1(m1.block(0, c1, rows, 1)); + VERIFY_IS_EQUAL(b1, m1.block(r1, c1, 1, 1)); VERIFY_IS_EQUAL(m1.row(r1), br1); VERIFY_IS_EQUAL(m1.col(c1), bc1); - //check operator(), both constant and non-constant, on block() - m1.block(r1,c1,r2-r1+1,c2-c1+1) = s1 * m2.block(0, 0, r2-r1+1,c2-c1+1); - m1.block(r1,c1,r2-r1+1,c2-c1+1)(r2-r1,c2-c1) = m2.block(0, 0, r2-r1+1,c2-c1+1)(0,0); - - enum { - BlockRows = 2, - BlockCols = 5 - }; - if (rows>=5 && cols>=8) - { + // check operator(), both constant and non-constant, on block() + m1.block(r1, c1, r2 - r1 + 1, c2 - c1 + 1) = s1 * m2.block(0, 0, r2 - r1 + 1, c2 - c1 + 1); + m1.block(r1, c1, r2 - r1 + 1, c2 - c1 + 1)(r2 - r1, c2 - c1) = m2.block(0, 0, r2 - r1 + 1, c2 - c1 + 1)(0, 0); + + enum { BlockRows = 2, BlockCols = 5 }; + if (rows >= 5 && cols >= 8) { // test fixed block() as lvalue - m1.template block(1,1) *= s1; + m1.template block(1, 1) *= s1; // test operator() on fixed block() both as constant and non-constant - m1.template block(1,1)(0, 3) = m1.template block<2,5>(1,1)(1,2); + m1.template block(1, 1)(0, 3) = m1.template block<2, 5>(1, 1)(1, 2); // check that fixed block() and block() agree - Matrix b = m1.template block(3,3); - VERIFY_IS_EQUAL(b, m1.block(3,3,BlockRows,BlockCols)); + Matrix b = m1.template block(3, 3); + VERIFY_IS_EQUAL(b, m1.block(3, 3, BlockRows, BlockCols)); // same tests with mixed fixed/dynamic size - m1.template block(1,1,BlockRows,BlockCols) *= s1; - m1.template block(1,1,BlockRows,BlockCols)(0,3) = m1.template block<2,5>(1,1)(1,2); - Matrix b2 = m1.template block(3,3,2,5); - VERIFY_IS_EQUAL(b2, m1.block(3,3,BlockRows,BlockCols)); + m1.template block(1, 1, BlockRows, BlockCols) *= s1; + m1.template block(1, 1, BlockRows, BlockCols)(0, 3) = m1.template block<2, 5>(1, 1)(1, 2); + Matrix b2 = m1.template block(3, 3, 2, 5); + VERIFY_IS_EQUAL(b2, m1.block(3, 3, BlockRows, BlockCols)); } - if (rows>2) - { + if (rows > 2) { // test sub vectors - VERIFY_IS_EQUAL(v1.template head<2>(), v1.block(0,0,2,1)); + VERIFY_IS_EQUAL(v1.template head<2>(), v1.block(0, 0, 2, 1)); VERIFY_IS_EQUAL(v1.template head<2>(), v1.head(2)); - VERIFY_IS_EQUAL(v1.template head<2>(), v1.segment(0,2)); + VERIFY_IS_EQUAL(v1.template head<2>(), v1.segment(0, 2)); VERIFY_IS_EQUAL(v1.template head<2>(), v1.template segment<2>(0)); - Index i = rows-2; - VERIFY_IS_EQUAL(v1.template tail<2>(), v1.block(i,0,2,1)); + Index i = rows - 2; + VERIFY_IS_EQUAL(v1.template tail<2>(), v1.block(i, 0, 2, 1)); VERIFY_IS_EQUAL(v1.template tail<2>(), v1.tail(2)); - VERIFY_IS_EQUAL(v1.template tail<2>(), v1.segment(i,2)); + VERIFY_IS_EQUAL(v1.template tail<2>(), v1.segment(i, 2)); VERIFY_IS_EQUAL(v1.template tail<2>(), v1.template segment<2>(i)); - i = internal::random(0,rows-2); - VERIFY_IS_EQUAL(v1.segment(i,2), v1.template segment<2>(i)); + i = internal::random(0, rows - 2); + VERIFY_IS_EQUAL(v1.segment(i, 2), v1.template segment<2>(i)); } // stress some basic stuffs with block matrices @@ -129,82 +128,88 @@ template void block(const MatrixType& m) VERIFY(numext::real(ones.col(c1).dot(ones.col(c2))) == RealScalar(rows)); VERIFY(numext::real(ones.row(r1).dot(ones.row(r2))) == RealScalar(cols)); - + // check that linear acccessors works on blocks m1 = m1_copy; - if((MatrixType::Flags&RowMajorBit)==0) - VERIFY_IS_EQUAL(m1.leftCols(c1).coeff(r1+c1*rows), m1(r1,c1)); + if ((MatrixType::Flags & RowMajorBit) == 0) + VERIFY_IS_EQUAL(m1.leftCols(c1).coeff(r1 + c1 * rows), m1(r1, c1)); else - VERIFY_IS_EQUAL(m1.topRows(r1).coeff(c1+r1*cols), m1(r1,c1)); - + VERIFY_IS_EQUAL(m1.topRows(r1).coeff(c1 + r1 * cols), m1(r1, c1)); + // now test some block-inside-of-block. - + // expressions with direct access - VERIFY_IS_EQUAL( (m1.block(r1,c1,rows-r1,cols-c1).block(r2-r1,c2-c1,rows-r2,cols-c2)) , (m1.block(r2,c2,rows-r2,cols-c2)) ); - VERIFY_IS_EQUAL( (m1.block(r1,c1,r2-r1+1,c2-c1+1).row(0)) , (m1.row(r1).segment(c1,c2-c1+1)) ); - VERIFY_IS_EQUAL( (m1.block(r1,c1,r2-r1+1,c2-c1+1).col(0)) , (m1.col(c1).segment(r1,r2-r1+1)) ); - VERIFY_IS_EQUAL( (m1.block(r1,c1,r2-r1+1,c2-c1+1).transpose().col(0)) , (m1.row(r1).segment(c1,c2-c1+1)).transpose() ); - VERIFY_IS_EQUAL( (m1.transpose().block(c1,r1,c2-c1+1,r2-r1+1).col(0)) , (m1.row(r1).segment(c1,c2-c1+1)).transpose() ); + VERIFY_IS_EQUAL((m1.block(r1, c1, rows - r1, cols - c1).block(r2 - r1, c2 - c1, rows - r2, cols - c2)), + (m1.block(r2, c2, rows - r2, cols - c2))); + VERIFY_IS_EQUAL((m1.block(r1, c1, r2 - r1 + 1, c2 - c1 + 1).row(0)), (m1.row(r1).segment(c1, c2 - c1 + 1))); + VERIFY_IS_EQUAL((m1.block(r1, c1, r2 - r1 + 1, c2 - c1 + 1).col(0)), (m1.col(c1).segment(r1, r2 - r1 + 1))); + VERIFY_IS_EQUAL( + (m1.block(r1, c1, r2 - r1 + 1, c2 - c1 + 1).transpose().col(0)), (m1.row(r1).segment(c1, c2 - c1 + 1)).transpose()); + VERIFY_IS_EQUAL( + (m1.transpose().block(c1, r1, c2 - c1 + 1, r2 - r1 + 1).col(0)), (m1.row(r1).segment(c1, c2 - c1 + 1)).transpose()); // expressions without direct access - VERIFY_IS_APPROX( ((m1+m2).block(r1,c1,rows-r1,cols-c1).block(r2-r1,c2-c1,rows-r2,cols-c2)) , ((m1+m2).block(r2,c2,rows-r2,cols-c2)) ); - VERIFY_IS_APPROX( ((m1+m2).block(r1,c1,r2-r1+1,c2-c1+1).row(0)) , ((m1+m2).row(r1).segment(c1,c2-c1+1)) ); - VERIFY_IS_APPROX( ((m1+m2).block(r1,c1,r2-r1+1,c2-c1+1).col(0)) , ((m1+m2).col(c1).segment(r1,r2-r1+1)) ); - VERIFY_IS_APPROX( ((m1+m2).block(r1,c1,r2-r1+1,c2-c1+1).transpose().col(0)) , ((m1+m2).row(r1).segment(c1,c2-c1+1)).transpose() ); - VERIFY_IS_APPROX( ((m1+m2).transpose().block(c1,r1,c2-c1+1,r2-r1+1).col(0)) , ((m1+m2).row(r1).segment(c1,c2-c1+1)).transpose() ); - - VERIFY_IS_APPROX( (m1*1).topRows(r1), m1.topRows(r1) ); - VERIFY_IS_APPROX( (m1*1).leftCols(c1), m1.leftCols(c1) ); - VERIFY_IS_APPROX( (m1*1).transpose().topRows(c1), m1.transpose().topRows(c1) ); - VERIFY_IS_APPROX( (m1*1).transpose().leftCols(r1), m1.transpose().leftCols(r1) ); - VERIFY_IS_APPROX( (m1*1).transpose().middleRows(c1,c2-c1+1), m1.transpose().middleRows(c1,c2-c1+1) ); - VERIFY_IS_APPROX( (m1*1).transpose().middleCols(r1,r2-r1+1), m1.transpose().middleCols(r1,r2-r1+1) ); + VERIFY_IS_APPROX(((m1 + m2).block(r1, c1, rows - r1, cols - c1).block(r2 - r1, c2 - c1, rows - r2, cols - c2)), + ((m1 + m2).block(r2, c2, rows - r2, cols - c2))); + VERIFY_IS_APPROX( + ((m1 + m2).block(r1, c1, r2 - r1 + 1, c2 - c1 + 1).row(0)), ((m1 + m2).row(r1).segment(c1, c2 - c1 + 1))); + VERIFY_IS_APPROX( + ((m1 + m2).block(r1, c1, r2 - r1 + 1, c2 - c1 + 1).col(0)), ((m1 + m2).col(c1).segment(r1, r2 - r1 + 1))); + VERIFY_IS_APPROX(((m1 + m2).block(r1, c1, r2 - r1 + 1, c2 - c1 + 1).transpose().col(0)), + ((m1 + m2).row(r1).segment(c1, c2 - c1 + 1)).transpose()); + VERIFY_IS_APPROX(((m1 + m2).transpose().block(c1, r1, c2 - c1 + 1, r2 - r1 + 1).col(0)), + ((m1 + m2).row(r1).segment(c1, c2 - c1 + 1)).transpose()); + + VERIFY_IS_APPROX((m1 * 1).topRows(r1), m1.topRows(r1)); + VERIFY_IS_APPROX((m1 * 1).leftCols(c1), m1.leftCols(c1)); + VERIFY_IS_APPROX((m1 * 1).transpose().topRows(c1), m1.transpose().topRows(c1)); + VERIFY_IS_APPROX((m1 * 1).transpose().leftCols(r1), m1.transpose().leftCols(r1)); + VERIFY_IS_APPROX((m1 * 1).transpose().middleRows(c1, c2 - c1 + 1), m1.transpose().middleRows(c1, c2 - c1 + 1)); + VERIFY_IS_APPROX((m1 * 1).transpose().middleCols(r1, r2 - r1 + 1), m1.transpose().middleCols(r1, r2 - r1 + 1)); // evaluation into plain matrices from expressions with direct access (stress MapBase) DynamicMatrixType dm; DynamicVectorType dv; dm.setZero(); - dm = m1.block(r1,c1,rows-r1,cols-c1).block(r2-r1,c2-c1,rows-r2,cols-c2); - VERIFY_IS_EQUAL(dm, (m1.block(r2,c2,rows-r2,cols-c2))); + dm = m1.block(r1, c1, rows - r1, cols - c1).block(r2 - r1, c2 - c1, rows - r2, cols - c2); + VERIFY_IS_EQUAL(dm, (m1.block(r2, c2, rows - r2, cols - c2))); dm.setZero(); dv.setZero(); - dm = m1.block(r1,c1,r2-r1+1,c2-c1+1).row(0).transpose(); - dv = m1.row(r1).segment(c1,c2-c1+1); + dm = m1.block(r1, c1, r2 - r1 + 1, c2 - c1 + 1).row(0).transpose(); + dv = m1.row(r1).segment(c1, c2 - c1 + 1); VERIFY_IS_EQUAL(dv, dm); dm.setZero(); dv.setZero(); - dm = m1.col(c1).segment(r1,r2-r1+1); - dv = m1.block(r1,c1,r2-r1+1,c2-c1+1).col(0); + dm = m1.col(c1).segment(r1, r2 - r1 + 1); + dv = m1.block(r1, c1, r2 - r1 + 1, c2 - c1 + 1).col(0); VERIFY_IS_EQUAL(dv, dm); dm.setZero(); dv.setZero(); - dm = m1.block(r1,c1,r2-r1+1,c2-c1+1).transpose().col(0); - dv = m1.row(r1).segment(c1,c2-c1+1); + dm = m1.block(r1, c1, r2 - r1 + 1, c2 - c1 + 1).transpose().col(0); + dv = m1.row(r1).segment(c1, c2 - c1 + 1); VERIFY_IS_EQUAL(dv, dm); dm.setZero(); dv.setZero(); - dm = m1.row(r1).segment(c1,c2-c1+1).transpose(); - dv = m1.transpose().block(c1,r1,c2-c1+1,r2-r1+1).col(0); + dm = m1.row(r1).segment(c1, c2 - c1 + 1).transpose(); + dv = m1.transpose().block(c1, r1, c2 - c1 + 1, r2 - r1 + 1).col(0); VERIFY_IS_EQUAL(dv, dm); - VERIFY_IS_EQUAL( (m1.template block(1,0,0,1)), m1.block(1,0,0,1)); - VERIFY_IS_EQUAL( (m1.template block<1,Dynamic>(0,1,1,0)), m1.block(0,1,1,0)); - VERIFY_IS_EQUAL( ((m1*1).template block(1,0,0,1)), m1.block(1,0,0,1)); - VERIFY_IS_EQUAL( ((m1*1).template block<1,Dynamic>(0,1,1,0)), m1.block(0,1,1,0)); - - if (rows>=2 && cols>=2) - { - VERIFY_RAISES_ASSERT( m1 += m1.col(0) ); - VERIFY_RAISES_ASSERT( m1 -= m1.col(0) ); - VERIFY_RAISES_ASSERT( m1.array() *= m1.col(0).array() ); - VERIFY_RAISES_ASSERT( m1.array() /= m1.col(0).array() ); + VERIFY_IS_EQUAL((m1.template block(1, 0, 0, 1)), m1.block(1, 0, 0, 1)); + VERIFY_IS_EQUAL((m1.template block<1, Dynamic>(0, 1, 1, 0)), m1.block(0, 1, 1, 0)); + VERIFY_IS_EQUAL(((m1 * 1).template block(1, 0, 0, 1)), m1.block(1, 0, 0, 1)); + VERIFY_IS_EQUAL(((m1 * 1).template block<1, Dynamic>(0, 1, 1, 0)), m1.block(0, 1, 1, 0)); + + if (rows >= 2 && cols >= 2) { + VERIFY_RAISES_ASSERT(m1 += m1.col(0)); + VERIFY_RAISES_ASSERT(m1 -= m1.col(0)); + VERIFY_RAISES_ASSERT(m1.array() *= m1.col(0).array()); + VERIFY_RAISES_ASSERT(m1.array() /= m1.col(0).array()); } } -template -void compare_using_data_and_stride(const MatrixType& m) +template void compare_using_data_and_stride(const MatrixType &m) { Index rows = m.rows(); Index cols = m.cols(); @@ -213,43 +218,38 @@ void compare_using_data_and_stride(const MatrixType& m) Index outerStride = m.outerStride(); Index rowStride = m.rowStride(); Index colStride = m.colStride(); - const typename MatrixType::Scalar* data = m.data(); - - for(int j=0;j -void data_and_stride(const MatrixType& m) +template void data_and_stride(const MatrixType &m) { Index rows = m.rows(); Index cols = m.cols(); - Index r1 = internal::random(0,rows-1); - Index r2 = internal::random(r1,rows-1); - Index c1 = internal::random(0,cols-1); - Index c2 = internal::random(c1,cols-1); + Index r1 = internal::random(0, rows - 1); + Index r2 = internal::random(r1, rows - 1); + Index c1 = internal::random(0, cols - 1); + Index c2 = internal::random(c1, cols - 1); MatrixType m1 = MatrixType::Random(rows, cols); - compare_using_data_and_stride(m1.block(r1, c1, r2-r1+1, c2-c1+1)); - compare_using_data_and_stride(m1.transpose().block(c1, r1, c2-c1+1, r2-r1+1)); + compare_using_data_and_stride(m1.block(r1, c1, r2 - r1 + 1, c2 - c1 + 1)); + compare_using_data_and_stride(m1.transpose().block(c1, r1, c2 - c1 + 1, r2 - r1 + 1)); compare_using_data_and_stride(m1.row(r1)); compare_using_data_and_stride(m1.col(c1)); compare_using_data_and_stride(m1.row(r1).transpose()); @@ -258,19 +258,20 @@ void data_and_stride(const MatrixType& m) void test_block() { - for(int i = 0; i < g_repeat; i++) { - CALL_SUBTEST_1( block(Matrix()) ); - CALL_SUBTEST_2( block(Matrix4d()) ); - CALL_SUBTEST_3( block(MatrixXcf(3, 3)) ); - CALL_SUBTEST_4( block(MatrixXi(8, 12)) ); - CALL_SUBTEST_5( block(MatrixXcd(20, 20)) ); - CALL_SUBTEST_6( block(MatrixXf(20, 20)) ); + for (int i = 0; i < g_repeat; i++) { + CALL_SUBTEST_1(block(Matrix())); + CALL_SUBTEST_2(block(Matrix4d())); + CALL_SUBTEST_3(block(MatrixXcf(3, 3))); + CALL_SUBTEST_4(block(MatrixXi(8, 12))); + CALL_SUBTEST_5(block(MatrixXcd(20, 20))); + CALL_SUBTEST_6(block(MatrixXf(20, 20))); - CALL_SUBTEST_8( block(Matrix(3, 4)) ); + CALL_SUBTEST_8(block(Matrix(3, 4))); #ifndef EIGEN_DEFAULT_TO_ROW_MAJOR - CALL_SUBTEST_6( data_and_stride(MatrixXf(internal::random(5,50), internal::random(5,50))) ); - CALL_SUBTEST_7( data_and_stride(Matrix(internal::random(5,50), internal::random(5,50))) ); + CALL_SUBTEST_6(data_and_stride(MatrixXf(internal::random(5, 50), internal::random(5, 50)))); + CALL_SUBTEST_7( + data_and_stride(Matrix(internal::random(5, 50), internal::random(5, 50)))); #endif } } diff --git a/filmulator-gui/core/nlmeans/eigen/test/boostmultiprec.cpp b/filmulator-gui/core/nlmeans/eigen/test/boostmultiprec.cpp index e06e9bda..03f78b6d 100644 --- a/filmulator-gui/core/nlmeans/eigen/test/boostmultiprec.cpp +++ b/filmulator-gui/core/nlmeans/eigen/test/boostmultiprec.cpp @@ -63,42 +63,40 @@ #undef isinf #undef isfinite +#include +#include #include #include -#include -#include namespace mp = boost::multiprecision; typedef mp::number, mp::et_on> Real; namespace Eigen { - template<> struct NumTraits : GenericNumTraits { - static inline Real dummy_precision() { return 1e-50; } +template<> struct NumTraits : GenericNumTraits +{ + static inline Real dummy_precision() { return 1e-50; } +}; + +template +struct NumTraits> : NumTraits +{ +}; + +template<> Real test_precision() { return 1e-50; } + +// needed in C++93 mode where number does not support explicit cast. +namespace internal { + template struct cast_impl + { + static inline NewType run(const Real &x) { return x.template convert_to(); } }; - template - struct NumTraits > : NumTraits {}; - - template<> - Real test_precision() { return 1e-50; } - - // needed in C++93 mode where number does not support explicit cast. - namespace internal { - template - struct cast_impl { - static inline NewType run(const Real& x) { - return x.template convert_to(); - } - }; - - template<> - struct cast_impl > { - static inline std::complex run(const Real& x) { - return std::complex(x); - } - }; - } -} + template<> struct cast_impl> + { + static inline std::complex run(const Real &x) { return std::complex(x); } + }; +}// namespace internal +}// namespace Eigen namespace boost { namespace multiprecision { @@ -110,41 +108,42 @@ namespace multiprecision { using boost::math::hypot; // The following is needed for std::complex: - Real fabs(const Real& a) { return abs EIGEN_NOT_A_MACRO (a); } - Real fmax(const Real& a, const Real& b) { using std::max; return max(a,b); } + Real fabs(const Real &a) { return abs EIGEN_NOT_A_MACRO(a); } + Real fmax(const Real &a, const Real &b) + { + using std::max; + return max(a, b); + } // some specialization for the unit tests: - inline bool test_isMuchSmallerThan(const Real& a, const Real& b) { + inline bool test_isMuchSmallerThan(const Real &a, const Real &b) + { return internal::isMuchSmallerThan(a, b, test_precision()); } - inline bool test_isApprox(const Real& a, const Real& b) { - return internal::isApprox(a, b, test_precision()); - } + inline bool test_isApprox(const Real &a, const Real &b) { return internal::isApprox(a, b, test_precision()); } - inline bool test_isApproxOrLessThan(const Real& a, const Real& b) { + inline bool test_isApproxOrLessThan(const Real &a, const Real &b) + { return internal::isApproxOrLessThan(a, b, test_precision()); } - Real get_test_precision(const Real&) { - return test_precision(); - } + Real get_test_precision(const Real &) { return test_precision(); } - Real test_relative_error(const Real &a, const Real &b) { + Real test_relative_error(const Real &a, const Real &b) + { using Eigen::numext::abs2; - return sqrt(abs2(a-b)/Eigen::numext::mini(abs2(a),abs2(b))); + return sqrt(abs2(a - b) / Eigen::numext::mini(abs2(a), abs2(b))); } -} -} - -namespace Eigen { +}// namespace multiprecision +}// namespace boost -} +namespace Eigen {} void test_boostmultiprec() { - typedef Matrix Mat; - typedef Matrix,Dynamic,Dynamic> MatC; + typedef Matrix Mat; + typedef Matrix, Dynamic, Dynamic> MatC; std::cout << "NumTraits::epsilon() = " << NumTraits::epsilon() << std::endl; std::cout << "NumTraits::dummy_precision() = " << NumTraits::dummy_precision() << std::endl; @@ -154,48 +153,50 @@ void test_boostmultiprec() // chekc stream output { - Mat A(10,10); + Mat A(10, 10); A.setRandom(); std::stringstream ss; ss << A; } { - MatC A(10,10); + MatC A(10, 10); A.setRandom(); std::stringstream ss; ss << A; } - for(int i = 0; i < g_repeat; i++) { - int s = internal::random(1,EIGEN_TEST_MAX_SIZE); + for (int i = 0; i < g_repeat; i++) { + int s = internal::random(1, EIGEN_TEST_MAX_SIZE); - CALL_SUBTEST_1( cholesky(Mat(s,s)) ); + CALL_SUBTEST_1(cholesky(Mat(s, s))); - CALL_SUBTEST_2( lu_non_invertible() ); - CALL_SUBTEST_2( lu_invertible() ); - CALL_SUBTEST_2( lu_non_invertible() ); - CALL_SUBTEST_2( lu_invertible() ); + CALL_SUBTEST_2(lu_non_invertible()); + CALL_SUBTEST_2(lu_invertible()); + CALL_SUBTEST_2(lu_non_invertible()); + CALL_SUBTEST_2(lu_invertible()); - CALL_SUBTEST_3( qr(Mat(internal::random(1,EIGEN_TEST_MAX_SIZE),internal::random(1,EIGEN_TEST_MAX_SIZE))) ); - CALL_SUBTEST_3( qr_invertible() ); + CALL_SUBTEST_3( + qr(Mat(internal::random(1, EIGEN_TEST_MAX_SIZE), internal::random(1, EIGEN_TEST_MAX_SIZE)))); + CALL_SUBTEST_3(qr_invertible()); - CALL_SUBTEST_4( qr() ); - CALL_SUBTEST_4( cod() ); - CALL_SUBTEST_4( qr_invertible() ); + CALL_SUBTEST_4(qr()); + CALL_SUBTEST_4(cod()); + CALL_SUBTEST_4(qr_invertible()); - CALL_SUBTEST_5( qr() ); - CALL_SUBTEST_5( qr_invertible() ); + CALL_SUBTEST_5(qr()); + CALL_SUBTEST_5(qr_invertible()); - CALL_SUBTEST_6( selfadjointeigensolver(Mat(s,s)) ); + CALL_SUBTEST_6(selfadjointeigensolver(Mat(s, s))); - CALL_SUBTEST_7( eigensolver(Mat(s,s)) ); + CALL_SUBTEST_7(eigensolver(Mat(s, s))); - CALL_SUBTEST_8( generalized_eigensolver_real(Mat(s,s)) ); + CALL_SUBTEST_8(generalized_eigensolver_real(Mat(s, s))); TEST_SET_BUT_UNUSED_VARIABLE(s) } - CALL_SUBTEST_9(( jacobisvd(Mat(internal::random(EIGEN_TEST_MAX_SIZE/4, EIGEN_TEST_MAX_SIZE), internal::random(EIGEN_TEST_MAX_SIZE/4, EIGEN_TEST_MAX_SIZE/2))) )); - CALL_SUBTEST_10(( bdcsvd(Mat(internal::random(EIGEN_TEST_MAX_SIZE/4, EIGEN_TEST_MAX_SIZE), internal::random(EIGEN_TEST_MAX_SIZE/4, EIGEN_TEST_MAX_SIZE/2))) )); + CALL_SUBTEST_9((jacobisvd(Mat(internal::random(EIGEN_TEST_MAX_SIZE / 4, EIGEN_TEST_MAX_SIZE), + internal::random(EIGEN_TEST_MAX_SIZE / 4, EIGEN_TEST_MAX_SIZE / 2))))); + CALL_SUBTEST_10((bdcsvd(Mat(internal::random(EIGEN_TEST_MAX_SIZE / 4, EIGEN_TEST_MAX_SIZE), + internal::random(EIGEN_TEST_MAX_SIZE / 4, EIGEN_TEST_MAX_SIZE / 2))))); } - diff --git a/filmulator-gui/core/nlmeans/eigen/test/bug1213.cpp b/filmulator-gui/core/nlmeans/eigen/test/bug1213.cpp index 581760c1..7f3f6aea 100644 --- a/filmulator-gui/core/nlmeans/eigen/test/bug1213.cpp +++ b/filmulator-gui/core/nlmeans/eigen/test/bug1213.cpp @@ -1,13 +1,7 @@ // This anonymous enum is essential to trigger the linking issue -enum { - Foo -}; +enum { Foo }; #include "bug1213.h" -bool bug1213_1(const Eigen::Vector3f& x) -{ - return bug1213_2(x); -} - +bool bug1213_1(const Eigen::Vector3f &x) { return bug1213_2(x); } diff --git a/filmulator-gui/core/nlmeans/eigen/test/bug1213.h b/filmulator-gui/core/nlmeans/eigen/test/bug1213.h index 040e5a47..e4f254e1 100644 --- a/filmulator-gui/core/nlmeans/eigen/test/bug1213.h +++ b/filmulator-gui/core/nlmeans/eigen/test/bug1213.h @@ -1,8 +1,6 @@ #include -template -bool bug1213_2(const Eigen::Matrix& x); - -bool bug1213_1(const Eigen::Vector3f& x); +template bool bug1213_2(const Eigen::Matrix &x); +bool bug1213_1(const Eigen::Vector3f &x); diff --git a/filmulator-gui/core/nlmeans/eigen/test/bug1213_main.cpp b/filmulator-gui/core/nlmeans/eigen/test/bug1213_main.cpp index 4802c000..fc4be6a5 100644 --- a/filmulator-gui/core/nlmeans/eigen/test/bug1213_main.cpp +++ b/filmulator-gui/core/nlmeans/eigen/test/bug1213_main.cpp @@ -3,16 +3,9 @@ #include "bug1213.h" -int main() -{ - return 0; -} +int main() { return 0; } -template -bool bug1213_2(const Eigen::Matrix& ) -{ - return true; -} +template bool bug1213_2(const Eigen::Matrix &) { return true; } -template bool bug1213_2(const Eigen::Vector3f&); +template bool bug1213_2(const Eigen::Vector3f &); diff --git a/filmulator-gui/core/nlmeans/eigen/test/cholesky.cpp b/filmulator-gui/core/nlmeans/eigen/test/cholesky.cpp index 5cf842d6..3781bb6b 100644 --- a/filmulator-gui/core/nlmeans/eigen/test/cholesky.cpp +++ b/filmulator-gui/core/nlmeans/eigen/test/cholesky.cpp @@ -17,14 +17,14 @@ #include #include -template -typename MatrixType::RealScalar matrix_l1_norm(const MatrixType& m) { - if(m.cols()==0) return typename MatrixType::RealScalar(0); +template typename MatrixType::RealScalar matrix_l1_norm(const MatrixType &m) +{ + if (m.cols() == 0) return typename MatrixType::RealScalar(0); MatrixType symm = m.template selfadjointView(); return symm.cwiseAbs().colwise().sum().maxCoeff(); } -template class CholType> void test_chol_update(const MatrixType& symm) +template class CholType> void test_chol_update(const MatrixType &symm) { typedef typename MatrixType::Scalar Scalar; typedef typename MatrixType::RealScalar RealScalar; @@ -34,19 +34,17 @@ template class CholType> void test_c MatrixType symmUp = symm.template triangularView(); MatrixType symmCpy = symm; - CholType chollo(symmLo); - CholType cholup(symmUp); + CholType chollo(symmLo); + CholType cholup(symmUp); - for (int k=0; k<10; ++k) - { + for (int k = 0; k < 10; ++k) { VectorType vec = VectorType::Random(symm.rows()); RealScalar sigma = internal::random(); symmCpy += sigma * vec * vec.adjoint(); // we are doing some downdates, so it might be the case that the matrix is not SPD anymore - CholType chol(symmCpy); - if(chol.info()!=Success) - break; + CholType chol(symmCpy); + if (chol.info() != Success) break; chollo.rankUpdate(vec, sigma); VERIFY_IS_APPROX(symmCpy, chollo.reconstructedMatrix()); @@ -56,7 +54,7 @@ template class CholType> void test_c } } -template void cholesky(const MatrixType& m) +template void cholesky(const MatrixType &m) { /* this test covers the following files: LLT.h LDLT.h @@ -69,14 +67,13 @@ template void cholesky(const MatrixType& m) typedef Matrix SquareMatrixType; typedef Matrix VectorType; - MatrixType a0 = MatrixType::Random(rows,cols); + MatrixType a0 = MatrixType::Random(rows, cols); VectorType vecB = VectorType::Random(rows), vecX(rows); - MatrixType matB = MatrixType::Random(rows,cols), matX(rows,cols); - SquareMatrixType symm = a0 * a0.adjoint(); + MatrixType matB = MatrixType::Random(rows, cols), matX(rows, cols); + SquareMatrixType symm = a0 * a0.adjoint(); // let's make sure the matrix is not singular or near singular - for (int k=0; k<3; ++k) - { - MatrixType a1 = MatrixType::Random(rows,cols); + for (int k = 0; k < 3; ++k) { + MatrixType a1 = MatrixType::Random(rows, cols); symm += a1 * a1.adjoint(); } @@ -84,23 +81,23 @@ template void cholesky(const MatrixType& m) SquareMatrixType symmUp = symm.template triangularView(); SquareMatrixType symmLo = symm.template triangularView(); - LLT chollo(symmLo); + LLT chollo(symmLo); VERIFY_IS_APPROX(symm, chollo.reconstructedMatrix()); vecX = chollo.solve(vecB); VERIFY_IS_APPROX(symm * vecX, vecB); matX = chollo.solve(matB); VERIFY_IS_APPROX(symm * matX, matB); - const MatrixType symmLo_inverse = chollo.solve(MatrixType::Identity(rows,cols)); - RealScalar rcond = (RealScalar(1) / matrix_l1_norm(symmLo)) / - matrix_l1_norm(symmLo_inverse); + const MatrixType symmLo_inverse = chollo.solve(MatrixType::Identity(rows, cols)); + RealScalar rcond = + (RealScalar(1) / matrix_l1_norm(symmLo)) / matrix_l1_norm(symmLo_inverse); RealScalar rcond_est = chollo.rcond(); // Verify that the estimated condition number is within a factor of 10 of the // truth. VERIFY(rcond_est >= rcond / 10 && rcond_est <= rcond * 10); // test the upper mode - LLT cholup(symmUp); + LLT cholup(symmUp); VERIFY_IS_APPROX(symm, cholup.reconstructedMatrix()); vecX = cholup.solve(vecB); VERIFY_IS_APPROX(symm * vecX, vecB); @@ -109,16 +106,16 @@ template void cholesky(const MatrixType& m) // Verify that the estimated condition number is within a factor of 10 of the // truth. - const MatrixType symmUp_inverse = cholup.solve(MatrixType::Identity(rows,cols)); - rcond = (RealScalar(1) / matrix_l1_norm(symmUp)) / - matrix_l1_norm(symmUp_inverse); + const MatrixType symmUp_inverse = cholup.solve(MatrixType::Identity(rows, cols)); + rcond = + (RealScalar(1) / matrix_l1_norm(symmUp)) / matrix_l1_norm(symmUp_inverse); rcond_est = cholup.rcond(); VERIFY(rcond_est >= rcond / 10 && rcond_est <= rcond * 10); MatrixType neg = -symmLo; chollo.compute(neg); - VERIFY(neg.size()==0 || chollo.info()==NumericalIssue); + VERIFY(neg.size() == 0 || chollo.info() == NumericalIssue); VERIFY_IS_APPROX(MatrixType(chollo.matrixL().transpose().conjugate()), MatrixType(chollo.matrixU())); VERIFY_IS_APPROX(MatrixType(chollo.matrixU().transpose().conjugate()), MatrixType(chollo.matrixL())); @@ -126,7 +123,7 @@ template void cholesky(const MatrixType& m) VERIFY_IS_APPROX(MatrixType(cholup.matrixU().transpose().conjugate()), MatrixType(cholup.matrixL())); // test some special use cases of SelfCwiseBinaryOp: - MatrixType m1 = MatrixType::Random(rows,cols), m2(rows,cols); + MatrixType m1 = MatrixType::Random(rows, cols), m2(rows, cols); m2 = m1; m2 += symmLo.template selfadjointView().llt().solve(matB); VERIFY_IS_APPROX(m2, m1 + symmLo.template selfadjointView().llt().solve(matB)); @@ -143,35 +140,34 @@ template void cholesky(const MatrixType& m) // LDLT { - int sign = internal::random()%2 ? 1 : -1; + int sign = internal::random() % 2 ? 1 : -1; - if(sign == -1) - { - symm = -symm; // test a negative matrix + if (sign == -1) { + symm = -symm;// test a negative matrix } SquareMatrixType symmUp = symm.template triangularView(); SquareMatrixType symmLo = symm.template triangularView(); - LDLT ldltlo(symmLo); - VERIFY(ldltlo.info()==Success); + LDLT ldltlo(symmLo); + VERIFY(ldltlo.info() == Success); VERIFY_IS_APPROX(symm, ldltlo.reconstructedMatrix()); vecX = ldltlo.solve(vecB); VERIFY_IS_APPROX(symm * vecX, vecB); matX = ldltlo.solve(matB); VERIFY_IS_APPROX(symm * matX, matB); - const MatrixType symmLo_inverse = ldltlo.solve(MatrixType::Identity(rows,cols)); - RealScalar rcond = (RealScalar(1) / matrix_l1_norm(symmLo)) / - matrix_l1_norm(symmLo_inverse); + const MatrixType symmLo_inverse = ldltlo.solve(MatrixType::Identity(rows, cols)); + RealScalar rcond = + (RealScalar(1) / matrix_l1_norm(symmLo)) / matrix_l1_norm(symmLo_inverse); RealScalar rcond_est = ldltlo.rcond(); // Verify that the estimated condition number is within a factor of 10 of the // truth. VERIFY(rcond_est >= rcond / 10 && rcond_est <= rcond * 10); - LDLT ldltup(symmUp); - VERIFY(ldltup.info()==Success); + LDLT ldltup(symmUp); + VERIFY(ldltup.info() == Success); VERIFY_IS_APPROX(symm, ldltup.reconstructedMatrix()); vecX = ldltup.solve(vecB); VERIFY_IS_APPROX(symm * vecX, vecB); @@ -180,9 +176,9 @@ template void cholesky(const MatrixType& m) // Verify that the estimated condition number is within a factor of 10 of the // truth. - const MatrixType symmUp_inverse = ldltup.solve(MatrixType::Identity(rows,cols)); - rcond = (RealScalar(1) / matrix_l1_norm(symmUp)) / - matrix_l1_norm(symmUp_inverse); + const MatrixType symmUp_inverse = ldltup.solve(MatrixType::Identity(rows, cols)); + rcond = + (RealScalar(1) / matrix_l1_norm(symmUp)) / matrix_l1_norm(symmUp_inverse); rcond_est = ldltup.rcond(); VERIFY(rcond_est >= rcond / 10 && rcond_est <= rcond * 10); @@ -191,8 +187,7 @@ template void cholesky(const MatrixType& m) VERIFY_IS_APPROX(MatrixType(ldltup.matrixL().transpose().conjugate()), MatrixType(ldltup.matrixU())); VERIFY_IS_APPROX(MatrixType(ldltup.matrixU().transpose().conjugate()), MatrixType(ldltup.matrixL())); - if(MatrixType::RowsAtCompileTime==Dynamic) - { + if (MatrixType::RowsAtCompileTime == Dynamic) { // note : each inplace permutation requires a small temporary vector (mask) // check inplace solve @@ -207,15 +202,13 @@ template void cholesky(const MatrixType& m) } // restore - if(sign == -1) - symm = -symm; + if (sign == -1) symm = -symm; // check matrices coming from linear constraints with Lagrange multipliers - if(rows>=3) - { + if (rows >= 3) { SquareMatrixType A = symm; - Index c = internal::random(0,rows-2); - A.bottomRightCorner(c,c).setZero(); + Index c = internal::random(0, rows - 2); + A.bottomRightCorner(c, c).setZero(); // Make sure a solution exists: vecX.setRandom(); vecB = A * vecX; @@ -227,10 +220,9 @@ template void cholesky(const MatrixType& m) } // check non-full rank matrices - if(rows>=3) - { - Index r = internal::random(1,rows-1); - Matrix a = Matrix::Random(rows,r); + if (rows >= 3) { + Index r = internal::random(1, rows - 1); + Matrix a = Matrix::Random(rows, r); SquareMatrixType A = a * a.adjoint(); // Make sure a solution exists: vecX.setRandom(); @@ -243,15 +235,13 @@ template void cholesky(const MatrixType& m) } // check matrices with a wide spectrum - if(rows>=3) - { + if (rows >= 3) { using std::pow; using std::sqrt; - RealScalar s = (std::min)(16,std::numeric_limits::max_exponent10/8); - Matrix a = Matrix::Random(rows,rows); - Matrix d = Matrix::Random(rows); - for(Index k=0; k(-s,s)); + RealScalar s = (std::min)(16, std::numeric_limits::max_exponent10 / 8); + Matrix a = Matrix::Random(rows, rows); + Matrix d = Matrix::Random(rows); + for (Index k = 0; k < rows; ++k) d(k) = d(k) * pow(RealScalar(10), internal::random(-s, s)); SquareMatrixType A = a * d.asDiagonal() * a.adjoint(); // Make sure a solution exists: vecX.setRandom(); @@ -261,28 +251,25 @@ template void cholesky(const MatrixType& m) VERIFY_IS_APPROX(A, ldltlo.reconstructedMatrix()); vecX = ldltlo.solve(vecB); - if(ldltlo.vectorD().real().cwiseAbs().minCoeff()>RealScalar(0)) - { - VERIFY_IS_APPROX(A * vecX,vecB); - } - else - { - RealScalar large_tol = sqrt(test_precision()); + if (ldltlo.vectorD().real().cwiseAbs().minCoeff() > RealScalar(0)) { + VERIFY_IS_APPROX(A * vecX, vecB); + } else { + RealScalar large_tol = sqrt(test_precision()); VERIFY((A * vecX).isApprox(vecB, large_tol)); ++g_test_level; - VERIFY_IS_APPROX(A * vecX,vecB); + VERIFY_IS_APPROX(A * vecX, vecB); --g_test_level; } } } // update/downdate - CALL_SUBTEST(( test_chol_update(symm) )); - CALL_SUBTEST(( test_chol_update(symm) )); + CALL_SUBTEST((test_chol_update(symm))); + CALL_SUBTEST((test_chol_update(symm))); } -template void cholesky_cplx(const MatrixType& m) +template void cholesky_cplx(const MatrixType &m) { // classic test cholesky(m); @@ -297,51 +284,49 @@ template void cholesky_cplx(const MatrixType& m) typedef Matrix RealMatrixType; typedef Matrix VectorType; - RealMatrixType a0 = RealMatrixType::Random(rows,cols); + RealMatrixType a0 = RealMatrixType::Random(rows, cols); VectorType vecB = VectorType::Random(rows), vecX(rows); - MatrixType matB = MatrixType::Random(rows,cols), matX(rows,cols); - RealMatrixType symm = a0 * a0.adjoint(); + MatrixType matB = MatrixType::Random(rows, cols), matX(rows, cols); + RealMatrixType symm = a0 * a0.adjoint(); // let's make sure the matrix is not singular or near singular - for (int k=0; k<3; ++k) - { - RealMatrixType a1 = RealMatrixType::Random(rows,cols); + for (int k = 0; k < 3; ++k) { + RealMatrixType a1 = RealMatrixType::Random(rows, cols); symm += a1 * a1.adjoint(); } { RealMatrixType symmLo = symm.template triangularView(); - LLT chollo(symmLo); + LLT chollo(symmLo); VERIFY_IS_APPROX(symm, chollo.reconstructedMatrix()); vecX = chollo.solve(vecB); VERIFY_IS_APPROX(symm * vecX, vecB); -// matX = chollo.solve(matB); -// VERIFY_IS_APPROX(symm * matX, matB); + // matX = chollo.solve(matB); + // VERIFY_IS_APPROX(symm * matX, matB); } // LDLT { - int sign = internal::random()%2 ? 1 : -1; + int sign = internal::random() % 2 ? 1 : -1; - if(sign == -1) - { - symm = -symm; // test a negative matrix + if (sign == -1) { + symm = -symm;// test a negative matrix } RealMatrixType symmLo = symm.template triangularView(); - LDLT ldltlo(symmLo); - VERIFY(ldltlo.info()==Success); + LDLT ldltlo(symmLo); + VERIFY(ldltlo.info() == Success); VERIFY_IS_APPROX(symm, ldltlo.reconstructedMatrix()); vecX = ldltlo.solve(vecB); VERIFY_IS_APPROX(symm * vecX, vecB); -// matX = ldltlo.solve(matB); -// VERIFY_IS_APPROX(symm * matX, matB); + // matX = ldltlo.solve(matB); + // VERIFY_IS_APPROX(symm * matX, matB); } } // regression test for bug 241 -template void cholesky_bug241(const MatrixType& m) +template void cholesky_bug241(const MatrixType &m) { eigen_assert(m.rows() == 2 && m.cols() == 2); @@ -359,7 +344,7 @@ template void cholesky_bug241(const MatrixType& m) // LDLT is not guaranteed to work for indefinite matrices, but happens to work fine if matrix is diagonal. // This test checks that LDLT reports correctly that matrix is indefinite. // See http://forum.kde.org/viewtopic.php?f=74&t=106942 and bug 736 -template void cholesky_definiteness(const MatrixType& m) +template void cholesky_definiteness(const MatrixType &m) { eigen_assert(m.rows() == 2 && m.cols() == 2); MatrixType mat; @@ -368,104 +353,91 @@ template void cholesky_definiteness(const MatrixType& m) { mat << 1, 0, 0, -1; ldlt.compute(mat); - VERIFY(ldlt.info()==Success); + VERIFY(ldlt.info() == Success); VERIFY(!ldlt.isNegative()); VERIFY(!ldlt.isPositive()); - VERIFY_IS_APPROX(mat,ldlt.reconstructedMatrix()); + VERIFY_IS_APPROX(mat, ldlt.reconstructedMatrix()); } { mat << 1, 2, 2, 1; ldlt.compute(mat); - VERIFY(ldlt.info()==Success); + VERIFY(ldlt.info() == Success); VERIFY(!ldlt.isNegative()); VERIFY(!ldlt.isPositive()); - VERIFY_IS_APPROX(mat,ldlt.reconstructedMatrix()); + VERIFY_IS_APPROX(mat, ldlt.reconstructedMatrix()); } { mat << 0, 0, 0, 0; ldlt.compute(mat); - VERIFY(ldlt.info()==Success); + VERIFY(ldlt.info() == Success); VERIFY(ldlt.isNegative()); VERIFY(ldlt.isPositive()); - VERIFY_IS_APPROX(mat,ldlt.reconstructedMatrix()); + VERIFY_IS_APPROX(mat, ldlt.reconstructedMatrix()); } { mat << 0, 0, 0, 1; ldlt.compute(mat); - VERIFY(ldlt.info()==Success); + VERIFY(ldlt.info() == Success); VERIFY(!ldlt.isNegative()); VERIFY(ldlt.isPositive()); - VERIFY_IS_APPROX(mat,ldlt.reconstructedMatrix()); + VERIFY_IS_APPROX(mat, ldlt.reconstructedMatrix()); } { mat << -1, 0, 0, 0; ldlt.compute(mat); - VERIFY(ldlt.info()==Success); + VERIFY(ldlt.info() == Success); VERIFY(ldlt.isNegative()); VERIFY(!ldlt.isPositive()); - VERIFY_IS_APPROX(mat,ldlt.reconstructedMatrix()); + VERIFY_IS_APPROX(mat, ldlt.reconstructedMatrix()); } } -template -void cholesky_faillure_cases() +template void cholesky_faillure_cases() { MatrixXd mat; LDLT ldlt; { - mat.resize(2,2); + mat.resize(2, 2); mat << 0, 1, 1, 0; ldlt.compute(mat); - VERIFY_IS_NOT_APPROX(mat,ldlt.reconstructedMatrix()); - VERIFY(ldlt.info()==NumericalIssue); + VERIFY_IS_NOT_APPROX(mat, ldlt.reconstructedMatrix()); + VERIFY(ldlt.info() == NumericalIssue); } #if (!EIGEN_ARCH_i386) || defined(EIGEN_VECTORIZE_SSE2) { - mat.resize(3,3); - mat << -1, -3, 3, - -3, -8.9999999999999999999, 1, - 3, 1, 0; + mat.resize(3, 3); + mat << -1, -3, 3, -3, -8.9999999999999999999, 1, 3, 1, 0; ldlt.compute(mat); - VERIFY(ldlt.info()==NumericalIssue); - VERIFY_IS_NOT_APPROX(mat,ldlt.reconstructedMatrix()); + VERIFY(ldlt.info() == NumericalIssue); + VERIFY_IS_NOT_APPROX(mat, ldlt.reconstructedMatrix()); } #endif { - mat.resize(3,3); - mat << 1, 2, 3, - 2, 4, 1, - 3, 1, 0; + mat.resize(3, 3); + mat << 1, 2, 3, 2, 4, 1, 3, 1, 0; ldlt.compute(mat); - VERIFY(ldlt.info()==NumericalIssue); - VERIFY_IS_NOT_APPROX(mat,ldlt.reconstructedMatrix()); + VERIFY(ldlt.info() == NumericalIssue); + VERIFY_IS_NOT_APPROX(mat, ldlt.reconstructedMatrix()); } { - mat.resize(8,8); - mat << 0.1, 0, -0.1, 0, 0, 0, 1, 0, - 0, 4.24667, 0, 2.00333, 0, 0, 0, 0, - -0.1, 0, 0.2, 0, -0.1, 0, 0, 0, - 0, 2.00333, 0, 8.49333, 0, 2.00333, 0, 0, - 0, 0, -0.1, 0, 0.1, 0, 0, 1, - 0, 0, 0, 2.00333, 0, 4.24667, 0, 0, - 1, 0, 0, 0, 0, 0, 0, 0, - 0, 0, 0, 0, 1, 0, 0, 0; + mat.resize(8, 8); + mat << 0.1, 0, -0.1, 0, 0, 0, 1, 0, 0, 4.24667, 0, 2.00333, 0, 0, 0, 0, -0.1, 0, 0.2, 0, -0.1, 0, 0, 0, 0, 2.00333, + 0, 8.49333, 0, 2.00333, 0, 0, 0, 0, -0.1, 0, 0.1, 0, 0, 1, 0, 0, 0, 2.00333, 0, 4.24667, 0, 0, 1, 0, 0, 0, 0, 0, + 0, 0, 0, 0, 0, 0, 1, 0, 0, 0; ldlt.compute(mat); - VERIFY(ldlt.info()==NumericalIssue); - VERIFY_IS_NOT_APPROX(mat,ldlt.reconstructedMatrix()); + VERIFY(ldlt.info() == NumericalIssue); + VERIFY_IS_NOT_APPROX(mat, ldlt.reconstructedMatrix()); } // bug 1479 { - mat.resize(4,4); - mat << 1, 2, 0, 1, - 2, 4, 0, 2, - 0, 0, 0, 1, - 1, 2, 1, 1; + mat.resize(4, 4); + mat << 1, 2, 0, 1, 2, 4, 0, 2, 0, 0, 0, 1, 1, 2, 1, 1; ldlt.compute(mat); - VERIFY(ldlt.info()==NumericalIssue); - VERIFY_IS_NOT_APPROX(mat,ldlt.reconstructedMatrix()); + VERIFY(ldlt.info() == NumericalIssue); + VERIFY_IS_NOT_APPROX(mat, ldlt.reconstructedMatrix()); } } @@ -492,38 +464,38 @@ template void cholesky_verify_assert() void test_cholesky() { int s = 0; - for(int i = 0; i < g_repeat; i++) { - CALL_SUBTEST_1( cholesky(Matrix()) ); - CALL_SUBTEST_3( cholesky(Matrix2d()) ); - CALL_SUBTEST_3( cholesky_bug241(Matrix2d()) ); - CALL_SUBTEST_3( cholesky_definiteness(Matrix2d()) ); - CALL_SUBTEST_4( cholesky(Matrix3f()) ); - CALL_SUBTEST_5( cholesky(Matrix4d()) ); - - s = internal::random(1,EIGEN_TEST_MAX_SIZE); - CALL_SUBTEST_2( cholesky(MatrixXd(s,s)) ); + for (int i = 0; i < g_repeat; i++) { + CALL_SUBTEST_1(cholesky(Matrix())); + CALL_SUBTEST_3(cholesky(Matrix2d())); + CALL_SUBTEST_3(cholesky_bug241(Matrix2d())); + CALL_SUBTEST_3(cholesky_definiteness(Matrix2d())); + CALL_SUBTEST_4(cholesky(Matrix3f())); + CALL_SUBTEST_5(cholesky(Matrix4d())); + + s = internal::random(1, EIGEN_TEST_MAX_SIZE); + CALL_SUBTEST_2(cholesky(MatrixXd(s, s))); TEST_SET_BUT_UNUSED_VARIABLE(s) - s = internal::random(1,EIGEN_TEST_MAX_SIZE/2); - CALL_SUBTEST_6( cholesky_cplx(MatrixXcd(s,s)) ); + s = internal::random(1, EIGEN_TEST_MAX_SIZE / 2); + CALL_SUBTEST_6(cholesky_cplx(MatrixXcd(s, s))); TEST_SET_BUT_UNUSED_VARIABLE(s) } // empty matrix, regression test for Bug 785: - CALL_SUBTEST_2( cholesky(MatrixXd(0,0)) ); + CALL_SUBTEST_2(cholesky(MatrixXd(0, 0))); // This does not work yet: // CALL_SUBTEST_2( cholesky(Matrix()) ); - CALL_SUBTEST_4( cholesky_verify_assert() ); - CALL_SUBTEST_7( cholesky_verify_assert() ); - CALL_SUBTEST_8( cholesky_verify_assert() ); - CALL_SUBTEST_2( cholesky_verify_assert() ); + CALL_SUBTEST_4(cholesky_verify_assert()); + CALL_SUBTEST_7(cholesky_verify_assert()); + CALL_SUBTEST_8(cholesky_verify_assert()); + CALL_SUBTEST_2(cholesky_verify_assert()); // Test problem size constructors - CALL_SUBTEST_9( LLT(10) ); - CALL_SUBTEST_9( LDLT(10) ); + CALL_SUBTEST_9(LLT(10)); + CALL_SUBTEST_9(LDLT(10)); - CALL_SUBTEST_2( cholesky_faillure_cases() ); + CALL_SUBTEST_2(cholesky_faillure_cases()); TEST_SET_BUT_UNUSED_VARIABLE(nb_temporaries) } diff --git a/filmulator-gui/core/nlmeans/eigen/test/cholmod_support.cpp b/filmulator-gui/core/nlmeans/eigen/test/cholmod_support.cpp index a7eda28f..46e07c86 100644 --- a/filmulator-gui/core/nlmeans/eigen/test/cholmod_support.cpp +++ b/filmulator-gui/core/nlmeans/eigen/test/cholmod_support.cpp @@ -14,13 +14,19 @@ template void test_cholmod_T() { - CholmodDecomposition, Lower> g_chol_colmajor_lower; g_chol_colmajor_lower.setMode(CholmodSupernodalLLt); - CholmodDecomposition, Upper> g_chol_colmajor_upper; g_chol_colmajor_upper.setMode(CholmodSupernodalLLt); - CholmodDecomposition, Lower> g_llt_colmajor_lower; g_llt_colmajor_lower.setMode(CholmodSimplicialLLt); - CholmodDecomposition, Upper> g_llt_colmajor_upper; g_llt_colmajor_upper.setMode(CholmodSimplicialLLt); - CholmodDecomposition, Lower> g_ldlt_colmajor_lower; g_ldlt_colmajor_lower.setMode(CholmodLDLt); - CholmodDecomposition, Upper> g_ldlt_colmajor_upper; g_ldlt_colmajor_upper.setMode(CholmodLDLt); - + CholmodDecomposition, Lower> g_chol_colmajor_lower; + g_chol_colmajor_lower.setMode(CholmodSupernodalLLt); + CholmodDecomposition, Upper> g_chol_colmajor_upper; + g_chol_colmajor_upper.setMode(CholmodSupernodalLLt); + CholmodDecomposition, Lower> g_llt_colmajor_lower; + g_llt_colmajor_lower.setMode(CholmodSimplicialLLt); + CholmodDecomposition, Upper> g_llt_colmajor_upper; + g_llt_colmajor_upper.setMode(CholmodSimplicialLLt); + CholmodDecomposition, Lower> g_ldlt_colmajor_lower; + g_ldlt_colmajor_lower.setMode(CholmodLDLt); + CholmodDecomposition, Upper> g_ldlt_colmajor_upper; + g_ldlt_colmajor_upper.setMode(CholmodLDLt); + CholmodSupernodalLLT, Lower> chol_colmajor_lower; CholmodSupernodalLLT, Upper> chol_colmajor_upper; CholmodSimplicialLLT, Lower> llt_colmajor_lower; @@ -34,7 +40,7 @@ template void test_cholmod_T() check_sparse_spd_solving(g_llt_colmajor_upper); check_sparse_spd_solving(g_ldlt_colmajor_lower); check_sparse_spd_solving(g_ldlt_colmajor_upper); - + check_sparse_spd_solving(chol_colmajor_lower); check_sparse_spd_solving(chol_colmajor_upper); check_sparse_spd_solving(llt_colmajor_lower); @@ -53,5 +59,5 @@ template void test_cholmod_T() void test_cholmod_support() { CALL_SUBTEST_1(test_cholmod_T()); - CALL_SUBTEST_2(test_cholmod_T >()); + CALL_SUBTEST_2(test_cholmod_T>()); } diff --git a/filmulator-gui/core/nlmeans/eigen/test/commainitializer.cpp b/filmulator-gui/core/nlmeans/eigen/test/commainitializer.cpp index 9844adbd..90465ec7 100644 --- a/filmulator-gui/core/nlmeans/eigen/test/commainitializer.cpp +++ b/filmulator-gui/core/nlmeans/eigen/test/commainitializer.cpp @@ -10,59 +10,57 @@ #include "main.h" -template -void test_blocks() +template void test_blocks() { - Matrix m_fixed; - MatrixXi m_dynamic(M1+M2, N1+N2); - - Matrix mat11; mat11.setRandom(); - Matrix mat12; mat12.setRandom(); - Matrix mat21; mat21.setRandom(); - Matrix mat22; mat22.setRandom(); + Matrix m_fixed; + MatrixXi m_dynamic(M1 + M2, N1 + N2); + + Matrix mat11; + mat11.setRandom(); + Matrix mat12; + mat12.setRandom(); + Matrix mat21; + mat21.setRandom(); + Matrix mat22; + mat22.setRandom(); MatrixXi matx11 = mat11, matx12 = mat12, matx21 = mat21, matx22 = mat22; { - VERIFY_IS_EQUAL((m_fixed << mat11, mat12, mat21, matx22).finished(), (m_dynamic << mat11, matx12, mat21, matx22).finished()); - VERIFY_IS_EQUAL((m_fixed.template topLeftCorner()), mat11); - VERIFY_IS_EQUAL((m_fixed.template topRightCorner()), mat12); - VERIFY_IS_EQUAL((m_fixed.template bottomLeftCorner()), mat21); - VERIFY_IS_EQUAL((m_fixed.template bottomRightCorner()), mat22); - VERIFY_IS_EQUAL((m_fixed << mat12, mat11, matx21, mat22).finished(), (m_dynamic << mat12, matx11, matx21, mat22).finished()); + VERIFY_IS_EQUAL( + (m_fixed << mat11, mat12, mat21, matx22).finished(), (m_dynamic << mat11, matx12, mat21, matx22).finished()); + VERIFY_IS_EQUAL((m_fixed.template topLeftCorner()), mat11); + VERIFY_IS_EQUAL((m_fixed.template topRightCorner()), mat12); + VERIFY_IS_EQUAL((m_fixed.template bottomLeftCorner()), mat21); + VERIFY_IS_EQUAL((m_fixed.template bottomRightCorner()), mat22); + VERIFY_IS_EQUAL( + (m_fixed << mat12, mat11, matx21, mat22).finished(), (m_dynamic << mat12, matx11, matx21, mat22).finished()); } - if(N1 > 0) - { + if (N1 > 0) { VERIFY_RAISES_ASSERT((m_fixed << mat11, mat12, mat11, mat21, mat22)); VERIFY_RAISES_ASSERT((m_fixed << mat11, mat12, mat21, mat21, mat22)); - } - else - { + } else { // allow insertion of zero-column blocks: - VERIFY_IS_EQUAL((m_fixed << mat11, mat12, mat11, mat11, mat21, mat21, mat22).finished(), (m_dynamic << mat12, mat22).finished()); - } - if(M1 != M2) - { - VERIFY_RAISES_ASSERT((m_fixed << mat11, mat21, mat12, mat22)); + VERIFY_IS_EQUAL( + (m_fixed << mat11, mat12, mat11, mat11, mat21, mat21, mat22).finished(), (m_dynamic << mat12, mat22).finished()); } + if (M1 != M2) { VERIFY_RAISES_ASSERT((m_fixed << mat11, mat21, mat12, mat22)); } } -template -struct test_block_recursion +template struct test_block_recursion { static void run() { - test_blocks<(N>>6)&3, (N>>4)&3, (N>>2)&3, N & 3>(); - test_block_recursion::run(); + test_blocks<(N >> 6) & 3, (N >> 4) & 3, (N >> 2) & 3, N & 3>(); + test_block_recursion::run(); } }; -template<> -struct test_block_recursion<-1> +template<> struct test_block_recursion<-1> { - static void run() { } + static void run() {} }; void test_commainitializer() @@ -70,18 +68,18 @@ void test_commainitializer() Matrix3d m3; Matrix4d m4; - VERIFY_RAISES_ASSERT( (m3 << 1, 2, 3, 4, 5, 6, 7, 8) ); - - #ifndef _MSC_VER - VERIFY_RAISES_ASSERT( (m3 << 1, 2, 3, 4, 5, 6, 7, 8, 9, 10) ); - #endif + VERIFY_RAISES_ASSERT((m3 << 1, 2, 3, 4, 5, 6, 7, 8)); - double data[] = {1, 2, 3, 4, 5, 6, 7, 8, 9}; - Matrix3d ref = Map >(data); +#ifndef _MSC_VER + VERIFY_RAISES_ASSERT((m3 << 1, 2, 3, 4, 5, 6, 7, 8, 9, 10)); +#endif + + double data[] = { 1, 2, 3, 4, 5, 6, 7, 8, 9 }; + Matrix3d ref = Map>(data); m3 = Matrix3d::Random(); m3 << 1, 2, 3, 4, 5, 6, 7, 8, 9; - VERIFY_IS_APPROX(m3, ref ); + VERIFY_IS_APPROX(m3, ref); Vector3d vec[3]; vec[0] << 1, 4, 7; @@ -95,12 +93,10 @@ void test_commainitializer() vec[1] << 4, 5, 6; vec[2] << 7, 8, 9; m3 = Matrix3d::Random(); - m3 << vec[0].transpose(), - 4, 5, 6, - vec[2].transpose(); + m3 << vec[0].transpose(), 4, 5, 6, vec[2].transpose(); VERIFY_IS_APPROX(m3, ref); // recursively test all block-sizes from 0 to 3: - test_block_recursion<(1<<8) - 1>(); + test_block_recursion<(1 << 8) - 1>(); } diff --git a/filmulator-gui/core/nlmeans/eigen/test/conjugate_gradient.cpp b/filmulator-gui/core/nlmeans/eigen/test/conjugate_gradient.cpp index 9622fd86..c1afe629 100644 --- a/filmulator-gui/core/nlmeans/eigen/test/conjugate_gradient.cpp +++ b/filmulator-gui/core/nlmeans/eigen/test/conjugate_gradient.cpp @@ -12,23 +12,23 @@ template void test_conjugate_gradient_T() { - typedef SparseMatrix SparseMatrixType; - ConjugateGradient cg_colmajor_lower_diag; - ConjugateGradient cg_colmajor_upper_diag; - ConjugateGradient cg_colmajor_loup_diag; + typedef SparseMatrix SparseMatrixType; + ConjugateGradient cg_colmajor_lower_diag; + ConjugateGradient cg_colmajor_upper_diag; + ConjugateGradient cg_colmajor_loup_diag; ConjugateGradient cg_colmajor_lower_I; ConjugateGradient cg_colmajor_upper_I; - CALL_SUBTEST( check_sparse_spd_solving(cg_colmajor_lower_diag) ); - CALL_SUBTEST( check_sparse_spd_solving(cg_colmajor_upper_diag) ); - CALL_SUBTEST( check_sparse_spd_solving(cg_colmajor_loup_diag) ); - CALL_SUBTEST( check_sparse_spd_solving(cg_colmajor_lower_I) ); - CALL_SUBTEST( check_sparse_spd_solving(cg_colmajor_upper_I) ); + CALL_SUBTEST(check_sparse_spd_solving(cg_colmajor_lower_diag)); + CALL_SUBTEST(check_sparse_spd_solving(cg_colmajor_upper_diag)); + CALL_SUBTEST(check_sparse_spd_solving(cg_colmajor_loup_diag)); + CALL_SUBTEST(check_sparse_spd_solving(cg_colmajor_lower_I)); + CALL_SUBTEST(check_sparse_spd_solving(cg_colmajor_upper_I)); } void test_conjugate_gradient() { - CALL_SUBTEST_1(( test_conjugate_gradient_T() )); - CALL_SUBTEST_2(( test_conjugate_gradient_T, int>() )); - CALL_SUBTEST_3(( test_conjugate_gradient_T() )); + CALL_SUBTEST_1((test_conjugate_gradient_T())); + CALL_SUBTEST_2((test_conjugate_gradient_T, int>())); + CALL_SUBTEST_3((test_conjugate_gradient_T())); } diff --git a/filmulator-gui/core/nlmeans/eigen/test/conservative_resize.cpp b/filmulator-gui/core/nlmeans/eigen/test/conservative_resize.cpp index 21a1db4a..67930cc1 100644 --- a/filmulator-gui/core/nlmeans/eigen/test/conservative_resize.cpp +++ b/filmulator-gui/core/nlmeans/eigen/test/conservative_resize.cpp @@ -13,51 +13,47 @@ using namespace Eigen; -template -void run_matrix_tests() +template void run_matrix_tests() { typedef Matrix MatrixType; MatrixType m, n; // boundary cases ... - m = n = MatrixType::Random(50,50); - m.conservativeResize(1,50); - VERIFY_IS_APPROX(m, n.block(0,0,1,50)); + m = n = MatrixType::Random(50, 50); + m.conservativeResize(1, 50); + VERIFY_IS_APPROX(m, n.block(0, 0, 1, 50)); - m = n = MatrixType::Random(50,50); - m.conservativeResize(50,1); - VERIFY_IS_APPROX(m, n.block(0,0,50,1)); + m = n = MatrixType::Random(50, 50); + m.conservativeResize(50, 1); + VERIFY_IS_APPROX(m, n.block(0, 0, 50, 1)); - m = n = MatrixType::Random(50,50); - m.conservativeResize(50,50); - VERIFY_IS_APPROX(m, n.block(0,0,50,50)); + m = n = MatrixType::Random(50, 50); + m.conservativeResize(50, 50); + VERIFY_IS_APPROX(m, n.block(0, 0, 50, 50)); // random shrinking ... - for (int i=0; i<25; ++i) - { - const Index rows = internal::random(1,50); - const Index cols = internal::random(1,50); - m = n = MatrixType::Random(50,50); - m.conservativeResize(rows,cols); - VERIFY_IS_APPROX(m, n.block(0,0,rows,cols)); + for (int i = 0; i < 25; ++i) { + const Index rows = internal::random(1, 50); + const Index cols = internal::random(1, 50); + m = n = MatrixType::Random(50, 50); + m.conservativeResize(rows, cols); + VERIFY_IS_APPROX(m, n.block(0, 0, rows, cols)); } // random growing with zeroing ... - for (int i=0; i<25; ++i) - { - const Index rows = internal::random(50,75); - const Index cols = internal::random(50,75); - m = n = MatrixType::Random(50,50); - m.conservativeResizeLike(MatrixType::Zero(rows,cols)); - VERIFY_IS_APPROX(m.block(0,0,n.rows(),n.cols()), n); - VERIFY( rows<=50 || m.block(50,0,rows-50,cols).sum() == Scalar(0) ); - VERIFY( cols<=50 || m.block(0,50,rows,cols-50).sum() == Scalar(0) ); + for (int i = 0; i < 25; ++i) { + const Index rows = internal::random(50, 75); + const Index cols = internal::random(50, 75); + m = n = MatrixType::Random(50, 50); + m.conservativeResizeLike(MatrixType::Zero(rows, cols)); + VERIFY_IS_APPROX(m.block(0, 0, n.rows(), n.cols()), n); + VERIFY(rows <= 50 || m.block(50, 0, rows - 50, cols).sum() == Scalar(0)); + VERIFY(cols <= 50 || m.block(0, 50, rows, cols - 50).sum() == Scalar(0)); } } -template -void run_vector_tests() +template void run_vector_tests() { typedef Matrix VectorType; @@ -66,53 +62,50 @@ void run_vector_tests() // boundary cases ... m = n = VectorType::Random(50); m.conservativeResize(1); - VERIFY_IS_APPROX(m, n.segment(0,1)); + VERIFY_IS_APPROX(m, n.segment(0, 1)); m = n = VectorType::Random(50); m.conservativeResize(50); - VERIFY_IS_APPROX(m, n.segment(0,50)); - + VERIFY_IS_APPROX(m, n.segment(0, 50)); + m = n = VectorType::Random(50); - m.conservativeResize(m.rows(),1); - VERIFY_IS_APPROX(m, n.segment(0,1)); + m.conservativeResize(m.rows(), 1); + VERIFY_IS_APPROX(m, n.segment(0, 1)); m = n = VectorType::Random(50); - m.conservativeResize(m.rows(),50); - VERIFY_IS_APPROX(m, n.segment(0,50)); + m.conservativeResize(m.rows(), 50); + VERIFY_IS_APPROX(m, n.segment(0, 50)); // random shrinking ... - for (int i=0; i<50; ++i) - { - const int size = internal::random(1,50); + for (int i = 0; i < 50; ++i) { + const int size = internal::random(1, 50); m = n = VectorType::Random(50); m.conservativeResize(size); - VERIFY_IS_APPROX(m, n.segment(0,size)); - + VERIFY_IS_APPROX(m, n.segment(0, size)); + m = n = VectorType::Random(50); m.conservativeResize(m.rows(), size); - VERIFY_IS_APPROX(m, n.segment(0,size)); + VERIFY_IS_APPROX(m, n.segment(0, size)); } // random growing with zeroing ... - for (int i=0; i<50; ++i) - { - const int size = internal::random(50,100); + for (int i = 0; i < 50; ++i) { + const int size = internal::random(50, 100); m = n = VectorType::Random(50); m.conservativeResizeLike(VectorType::Zero(size)); - VERIFY_IS_APPROX(m.segment(0,50), n); - VERIFY( size<=50 || m.segment(50,size-50).sum() == Scalar(0) ); - + VERIFY_IS_APPROX(m.segment(0, 50), n); + VERIFY(size <= 50 || m.segment(50, size - 50).sum() == Scalar(0)); + m = n = VectorType::Random(50); - m.conservativeResizeLike(Matrix::Zero(1,size)); - VERIFY_IS_APPROX(m.segment(0,50), n); - VERIFY( size<=50 || m.segment(50,size-50).sum() == Scalar(0) ); + m.conservativeResizeLike(Matrix::Zero(1, size)); + VERIFY_IS_APPROX(m.segment(0, 50), n); + VERIFY(size <= 50 || m.segment(50, size - 50).sum() == Scalar(0)); } } void test_conservative_resize() { - for(int i=0; i())); CALL_SUBTEST_1((run_matrix_tests())); CALL_SUBTEST_2((run_matrix_tests())); @@ -127,7 +120,7 @@ void test_conservative_resize() CALL_SUBTEST_1((run_vector_tests())); CALL_SUBTEST_2((run_vector_tests())); CALL_SUBTEST_3((run_vector_tests())); - CALL_SUBTEST_4((run_vector_tests >())); - CALL_SUBTEST_5((run_vector_tests >())); + CALL_SUBTEST_4((run_vector_tests>())); + CALL_SUBTEST_5((run_vector_tests>())); } } diff --git a/filmulator-gui/core/nlmeans/eigen/test/constructor.cpp b/filmulator-gui/core/nlmeans/eigen/test/constructor.cpp index eec9e219..9fc29d8a 100644 --- a/filmulator-gui/core/nlmeans/eigen/test/constructor.cpp +++ b/filmulator-gui/core/nlmeans/eigen/test/constructor.cpp @@ -16,69 +16,71 @@ template struct Wrapper { MatrixType m_mat; inline Wrapper(const MatrixType &x) : m_mat(x) {} - inline operator const MatrixType& () const { return m_mat; } - inline operator MatrixType& () { return m_mat; } + inline operator const MatrixType &() const { return m_mat; } + inline operator MatrixType &() { return m_mat; } }; -template void ctor_init1(const MatrixType& m) +template void ctor_init1(const MatrixType &m) { // Check logic in PlainObjectBase::_init1 Index rows = m.rows(); Index cols = m.cols(); - MatrixType m0 = MatrixType::Random(rows,cols); + MatrixType m0 = MatrixType::Random(rows, cols); - VERIFY_EVALUATION_COUNT( MatrixType m1(m0), 1); - VERIFY_EVALUATION_COUNT( MatrixType m2(m0+m0), 1); - VERIFY_EVALUATION_COUNT( MatrixType m2(m0.block(0,0,rows,cols)) , 1); + VERIFY_EVALUATION_COUNT(MatrixType m1(m0), 1); + VERIFY_EVALUATION_COUNT(MatrixType m2(m0 + m0), 1); + VERIFY_EVALUATION_COUNT(MatrixType m2(m0.block(0, 0, rows, cols)), 1); Wrapper wrapper(m0); - VERIFY_EVALUATION_COUNT( MatrixType m3(wrapper) , 1); + VERIFY_EVALUATION_COUNT(MatrixType m3(wrapper), 1); } void test_constructor() { - for(int i = 0; i < g_repeat; i++) { - CALL_SUBTEST_1( ctor_init1(Matrix()) ); - CALL_SUBTEST_1( ctor_init1(Matrix4d()) ); - CALL_SUBTEST_1( ctor_init1(MatrixXcf(internal::random(1,EIGEN_TEST_MAX_SIZE), internal::random(1,EIGEN_TEST_MAX_SIZE))) ); - CALL_SUBTEST_1( ctor_init1(MatrixXi(internal::random(1,EIGEN_TEST_MAX_SIZE), internal::random(1,EIGEN_TEST_MAX_SIZE))) ); + for (int i = 0; i < g_repeat; i++) { + CALL_SUBTEST_1(ctor_init1(Matrix())); + CALL_SUBTEST_1(ctor_init1(Matrix4d())); + CALL_SUBTEST_1(ctor_init1( + MatrixXcf(internal::random(1, EIGEN_TEST_MAX_SIZE), internal::random(1, EIGEN_TEST_MAX_SIZE)))); + CALL_SUBTEST_1(ctor_init1( + MatrixXi(internal::random(1, EIGEN_TEST_MAX_SIZE), internal::random(1, EIGEN_TEST_MAX_SIZE)))); } { - Matrix a(123); + Matrix a(123); VERIFY_IS_EQUAL(a[0], 123); } { - Matrix a(123.0); + Matrix a(123.0); VERIFY_IS_EQUAL(a[0], 123); } { - Matrix a(123); + Matrix a(123); VERIFY_IS_EQUAL(a[0], 123.f); } { - Array a(123); + Array a(123); VERIFY_IS_EQUAL(a[0], 123); } { - Array a(123.0); + Array a(123.0); VERIFY_IS_EQUAL(a[0], 123); } { - Array a(123); + Array a(123); VERIFY_IS_EQUAL(a[0], 123.f); } { - Array a(123); + Array a(123); VERIFY_IS_EQUAL(a(4), 123); } { - Array a(123.0); + Array a(123.0); VERIFY_IS_EQUAL(a(4), 123); } { - Array a(123); + Array a(123); VERIFY_IS_EQUAL(a(4), 123.f); } } diff --git a/filmulator-gui/core/nlmeans/eigen/test/corners.cpp b/filmulator-gui/core/nlmeans/eigen/test/corners.cpp index 32edadb2..e8a9794d 100644 --- a/filmulator-gui/core/nlmeans/eigen/test/corners.cpp +++ b/filmulator-gui/core/nlmeans/eigen/test/corners.cpp @@ -9,37 +9,37 @@ #include "main.h" -#define COMPARE_CORNER(A,B) \ +#define COMPARE_CORNER(A, B) \ VERIFY_IS_EQUAL(matrix.A, matrix.B); \ VERIFY_IS_EQUAL(const_matrix.A, const_matrix.B); -template void corners(const MatrixType& m) +template void corners(const MatrixType &m) { Index rows = m.rows(); Index cols = m.cols(); - Index r = internal::random(1,rows); - Index c = internal::random(1,cols); + Index r = internal::random(1, rows); + Index c = internal::random(1, cols); - MatrixType matrix = MatrixType::Random(rows,cols); - const MatrixType const_matrix = MatrixType::Random(rows,cols); + MatrixType matrix = MatrixType::Random(rows, cols); + const MatrixType const_matrix = MatrixType::Random(rows, cols); - COMPARE_CORNER(topLeftCorner(r,c), block(0,0,r,c)); - COMPARE_CORNER(topRightCorner(r,c), block(0,cols-c,r,c)); - COMPARE_CORNER(bottomLeftCorner(r,c), block(rows-r,0,r,c)); - COMPARE_CORNER(bottomRightCorner(r,c), block(rows-r,cols-c,r,c)); + COMPARE_CORNER(topLeftCorner(r, c), block(0, 0, r, c)); + COMPARE_CORNER(topRightCorner(r, c), block(0, cols - c, r, c)); + COMPARE_CORNER(bottomLeftCorner(r, c), block(rows - r, 0, r, c)); + COMPARE_CORNER(bottomRightCorner(r, c), block(rows - r, cols - c, r, c)); - Index sr = internal::random(1,rows) - 1; - Index nr = internal::random(1,rows-sr); - Index sc = internal::random(1,cols) - 1; - Index nc = internal::random(1,cols-sc); + Index sr = internal::random(1, rows) - 1; + Index nr = internal::random(1, rows - sr); + Index sc = internal::random(1, cols) - 1; + Index nc = internal::random(1, cols - sc); - COMPARE_CORNER(topRows(r), block(0,0,r,cols)); - COMPARE_CORNER(middleRows(sr,nr), block(sr,0,nr,cols)); - COMPARE_CORNER(bottomRows(r), block(rows-r,0,r,cols)); - COMPARE_CORNER(leftCols(c), block(0,0,rows,c)); - COMPARE_CORNER(middleCols(sc,nc), block(0,sc,rows,nc)); - COMPARE_CORNER(rightCols(c), block(0,cols-c,rows,c)); + COMPARE_CORNER(topRows(r), block(0, 0, r, cols)); + COMPARE_CORNER(middleRows(sr, nr), block(sr, 0, nr, cols)); + COMPARE_CORNER(bottomRows(r), block(rows - r, 0, r, cols)); + COMPARE_CORNER(leftCols(c), block(0, 0, rows, c)); + COMPARE_CORNER(middleCols(sc, nc), block(0, sc, rows, nc)); + COMPARE_CORNER(rightCols(c), block(0, cols - c, rows, c)); } template void corners_fixedsize() @@ -52,66 +52,75 @@ template void c cols = MatrixType::ColsAtCompileTime, r = CRows, c = CCols, - sr = SRows, - sc = SCols + sr = SRows, + sc = SCols }; - VERIFY_IS_EQUAL((matrix.template topLeftCorner()), (matrix.template block(0,0))); - VERIFY_IS_EQUAL((matrix.template topRightCorner()), (matrix.template block(0,cols-c))); - VERIFY_IS_EQUAL((matrix.template bottomLeftCorner()), (matrix.template block(rows-r,0))); - VERIFY_IS_EQUAL((matrix.template bottomRightCorner()), (matrix.template block(rows-r,cols-c))); - - VERIFY_IS_EQUAL((matrix.template topLeftCorner()), (matrix.template topLeftCorner(r,c))); - VERIFY_IS_EQUAL((matrix.template topRightCorner()), (matrix.template topRightCorner(r,c))); - VERIFY_IS_EQUAL((matrix.template bottomLeftCorner()), (matrix.template bottomLeftCorner(r,c))); - VERIFY_IS_EQUAL((matrix.template bottomRightCorner()), (matrix.template bottomRightCorner(r,c))); - - VERIFY_IS_EQUAL((matrix.template topLeftCorner()), (matrix.template topLeftCorner(r,c))); - VERIFY_IS_EQUAL((matrix.template topRightCorner()), (matrix.template topRightCorner(r,c))); - VERIFY_IS_EQUAL((matrix.template bottomLeftCorner()), (matrix.template bottomLeftCorner(r,c))); - VERIFY_IS_EQUAL((matrix.template bottomRightCorner()), (matrix.template bottomRightCorner(r,c))); - - VERIFY_IS_EQUAL((matrix.template topRows()), (matrix.template block(0,0))); - VERIFY_IS_EQUAL((matrix.template middleRows(sr)), (matrix.template block(sr,0))); - VERIFY_IS_EQUAL((matrix.template bottomRows()), (matrix.template block(rows-r,0))); - VERIFY_IS_EQUAL((matrix.template leftCols()), (matrix.template block(0,0))); - VERIFY_IS_EQUAL((matrix.template middleCols(sc)), (matrix.template block(0,sc))); - VERIFY_IS_EQUAL((matrix.template rightCols()), (matrix.template block(0,cols-c))); - - VERIFY_IS_EQUAL((const_matrix.template topLeftCorner()), (const_matrix.template block(0,0))); - VERIFY_IS_EQUAL((const_matrix.template topRightCorner()), (const_matrix.template block(0,cols-c))); - VERIFY_IS_EQUAL((const_matrix.template bottomLeftCorner()), (const_matrix.template block(rows-r,0))); - VERIFY_IS_EQUAL((const_matrix.template bottomRightCorner()), (const_matrix.template block(rows-r,cols-c))); - - VERIFY_IS_EQUAL((const_matrix.template topLeftCorner()), (const_matrix.template topLeftCorner(r,c))); - VERIFY_IS_EQUAL((const_matrix.template topRightCorner()), (const_matrix.template topRightCorner(r,c))); - VERIFY_IS_EQUAL((const_matrix.template bottomLeftCorner()), (const_matrix.template bottomLeftCorner(r,c))); - VERIFY_IS_EQUAL((const_matrix.template bottomRightCorner()), (const_matrix.template bottomRightCorner(r,c))); - - VERIFY_IS_EQUAL((const_matrix.template topLeftCorner()), (const_matrix.template topLeftCorner(r,c))); - VERIFY_IS_EQUAL((const_matrix.template topRightCorner()), (const_matrix.template topRightCorner(r,c))); - VERIFY_IS_EQUAL((const_matrix.template bottomLeftCorner()), (const_matrix.template bottomLeftCorner(r,c))); - VERIFY_IS_EQUAL((const_matrix.template bottomRightCorner()), (const_matrix.template bottomRightCorner(r,c))); - - VERIFY_IS_EQUAL((const_matrix.template topRows()), (const_matrix.template block(0,0))); - VERIFY_IS_EQUAL((const_matrix.template middleRows(sr)), (const_matrix.template block(sr,0))); - VERIFY_IS_EQUAL((const_matrix.template bottomRows()), (const_matrix.template block(rows-r,0))); - VERIFY_IS_EQUAL((const_matrix.template leftCols()), (const_matrix.template block(0,0))); - VERIFY_IS_EQUAL((const_matrix.template middleCols(sc)), (const_matrix.template block(0,sc))); - VERIFY_IS_EQUAL((const_matrix.template rightCols()), (const_matrix.template block(0,cols-c))); + VERIFY_IS_EQUAL((matrix.template topLeftCorner()), (matrix.template block(0, 0))); + VERIFY_IS_EQUAL((matrix.template topRightCorner()), (matrix.template block(0, cols - c))); + VERIFY_IS_EQUAL((matrix.template bottomLeftCorner()), (matrix.template block(rows - r, 0))); + VERIFY_IS_EQUAL((matrix.template bottomRightCorner()), (matrix.template block(rows - r, cols - c))); + + VERIFY_IS_EQUAL((matrix.template topLeftCorner()), (matrix.template topLeftCorner(r, c))); + VERIFY_IS_EQUAL((matrix.template topRightCorner()), (matrix.template topRightCorner(r, c))); + VERIFY_IS_EQUAL((matrix.template bottomLeftCorner()), (matrix.template bottomLeftCorner(r, c))); + VERIFY_IS_EQUAL((matrix.template bottomRightCorner()), (matrix.template bottomRightCorner(r, c))); + + VERIFY_IS_EQUAL((matrix.template topLeftCorner()), (matrix.template topLeftCorner(r, c))); + VERIFY_IS_EQUAL((matrix.template topRightCorner()), (matrix.template topRightCorner(r, c))); + VERIFY_IS_EQUAL((matrix.template bottomLeftCorner()), (matrix.template bottomLeftCorner(r, c))); + VERIFY_IS_EQUAL((matrix.template bottomRightCorner()), (matrix.template bottomRightCorner(r, c))); + + VERIFY_IS_EQUAL((matrix.template topRows()), (matrix.template block(0, 0))); + VERIFY_IS_EQUAL((matrix.template middleRows(sr)), (matrix.template block(sr, 0))); + VERIFY_IS_EQUAL((matrix.template bottomRows()), (matrix.template block(rows - r, 0))); + VERIFY_IS_EQUAL((matrix.template leftCols()), (matrix.template block(0, 0))); + VERIFY_IS_EQUAL((matrix.template middleCols(sc)), (matrix.template block(0, sc))); + VERIFY_IS_EQUAL((matrix.template rightCols()), (matrix.template block(0, cols - c))); + + VERIFY_IS_EQUAL((const_matrix.template topLeftCorner()), (const_matrix.template block(0, 0))); + VERIFY_IS_EQUAL((const_matrix.template topRightCorner()), (const_matrix.template block(0, cols - c))); + VERIFY_IS_EQUAL((const_matrix.template bottomLeftCorner()), (const_matrix.template block(rows - r, 0))); + VERIFY_IS_EQUAL( + (const_matrix.template bottomRightCorner()), (const_matrix.template block(rows - r, cols - c))); + + VERIFY_IS_EQUAL( + (const_matrix.template topLeftCorner()), (const_matrix.template topLeftCorner(r, c))); + VERIFY_IS_EQUAL( + (const_matrix.template topRightCorner()), (const_matrix.template topRightCorner(r, c))); + VERIFY_IS_EQUAL( + (const_matrix.template bottomLeftCorner()), (const_matrix.template bottomLeftCorner(r, c))); + VERIFY_IS_EQUAL( + (const_matrix.template bottomRightCorner()), (const_matrix.template bottomRightCorner(r, c))); + + VERIFY_IS_EQUAL( + (const_matrix.template topLeftCorner()), (const_matrix.template topLeftCorner(r, c))); + VERIFY_IS_EQUAL( + (const_matrix.template topRightCorner()), (const_matrix.template topRightCorner(r, c))); + VERIFY_IS_EQUAL( + (const_matrix.template bottomLeftCorner()), (const_matrix.template bottomLeftCorner(r, c))); + VERIFY_IS_EQUAL( + (const_matrix.template bottomRightCorner()), (const_matrix.template bottomRightCorner(r, c))); + + VERIFY_IS_EQUAL((const_matrix.template topRows()), (const_matrix.template block(0, 0))); + VERIFY_IS_EQUAL((const_matrix.template middleRows(sr)), (const_matrix.template block(sr, 0))); + VERIFY_IS_EQUAL((const_matrix.template bottomRows()), (const_matrix.template block(rows - r, 0))); + VERIFY_IS_EQUAL((const_matrix.template leftCols()), (const_matrix.template block(0, 0))); + VERIFY_IS_EQUAL((const_matrix.template middleCols(sc)), (const_matrix.template block(0, sc))); + VERIFY_IS_EQUAL((const_matrix.template rightCols()), (const_matrix.template block(0, cols - c))); } void test_corners() { - for(int i = 0; i < g_repeat; i++) { - CALL_SUBTEST_1( corners(Matrix()) ); - CALL_SUBTEST_2( corners(Matrix4d()) ); - CALL_SUBTEST_3( corners(Matrix()) ); - CALL_SUBTEST_4( corners(MatrixXcf(5, 7)) ); - CALL_SUBTEST_5( corners(MatrixXf(21, 20)) ); - - CALL_SUBTEST_1(( corners_fixedsize, 1, 1, 0, 0>() )); - CALL_SUBTEST_2(( corners_fixedsize() )); - CALL_SUBTEST_3(( corners_fixedsize,4,7,5,2>() )); + for (int i = 0; i < g_repeat; i++) { + CALL_SUBTEST_1(corners(Matrix())); + CALL_SUBTEST_2(corners(Matrix4d())); + CALL_SUBTEST_3(corners(Matrix())); + CALL_SUBTEST_4(corners(MatrixXcf(5, 7))); + CALL_SUBTEST_5(corners(MatrixXf(21, 20))); + + CALL_SUBTEST_1((corners_fixedsize, 1, 1, 0, 0>())); + CALL_SUBTEST_2((corners_fixedsize())); + CALL_SUBTEST_3((corners_fixedsize, 4, 7, 5, 2>())); } } diff --git a/filmulator-gui/core/nlmeans/eigen/test/ctorleak.cpp b/filmulator-gui/core/nlmeans/eigen/test/ctorleak.cpp index c158f5e4..5b48b3c8 100644 --- a/filmulator-gui/core/nlmeans/eigen/test/ctorleak.cpp +++ b/filmulator-gui/core/nlmeans/eigen/test/ctorleak.cpp @@ -1,6 +1,6 @@ #include "main.h" -#include // std::exception +#include // std::exception struct Foo { @@ -12,19 +12,24 @@ struct Foo { #ifdef EIGEN_EXCEPTIONS // TODO: Is this the correct way to handle this? - if (Foo::object_count > Foo::object_limit) { std::cout << "\nThrow!\n"; throw Foo::Fail(); } + if (Foo::object_count > Foo::object_limit) { + std::cout << "\nThrow!\n"; + throw Foo::Fail(); + } #endif - std::cout << '+'; + std::cout << '+'; ++Foo::object_count; } ~Foo() { - std::cout << '-'; + std::cout << '-'; --Foo::object_count; } - class Fail : public std::exception {}; + class Fail : public std::exception + { + }; }; Index Foo::object_count = 0; @@ -38,31 +43,31 @@ void test_ctorleak() typedef Matrix MatrixX; typedef Matrix VectorX; Foo::object_count = 0; - for(int i = 0; i < g_repeat; i++) { - Index rows = internal::random(2,EIGEN_TEST_MAX_SIZE), cols = internal::random(2,EIGEN_TEST_MAX_SIZE); - Foo::object_limit = internal::random(0, rows*cols - 2); + for (int i = 0; i < g_repeat; i++) { + Index rows = internal::random(2, EIGEN_TEST_MAX_SIZE), + cols = internal::random(2, EIGEN_TEST_MAX_SIZE); + Foo::object_limit = internal::random(0, rows * cols - 2); std::cout << "object_limit =" << Foo::object_limit << std::endl; #ifdef EIGEN_EXCEPTIONS - try - { + try { #endif - std::cout << "\nMatrixX m(" << rows << ", " << cols << ");\n"; + std::cout << "\nMatrixX m(" << rows << ", " << cols << ");\n"; MatrixX m(rows, cols); #ifdef EIGEN_EXCEPTIONS - VERIFY(false); // not reached if exceptions are enabled + VERIFY(false);// not reached if exceptions are enabled + } catch (const Foo::Fail &) { /* ignore */ } - catch (const Foo::Fail&) { /* ignore */ } #endif VERIFY_IS_EQUAL(Index(0), Foo::object_count); { - Foo::object_limit = (rows+1)*(cols+1); + Foo::object_limit = (rows + 1) * (cols + 1); MatrixX A(rows, cols); - VERIFY_IS_EQUAL(Foo::object_count, rows*cols); - VectorX v=A.row(0); - VERIFY_IS_EQUAL(Foo::object_count, (rows+1)*cols); + VERIFY_IS_EQUAL(Foo::object_count, rows * cols); + VectorX v = A.row(0); + VERIFY_IS_EQUAL(Foo::object_count, (rows + 1) * cols); v = A.col(0); - VERIFY_IS_EQUAL(Foo::object_count, rows*(cols+1)); + VERIFY_IS_EQUAL(Foo::object_count, rows * (cols + 1)); } VERIFY_IS_EQUAL(Index(0), Foo::object_count); } diff --git a/filmulator-gui/core/nlmeans/eigen/test/cuda_common.h b/filmulator-gui/core/nlmeans/eigen/test/cuda_common.h index 9737693a..cae6dda2 100644 --- a/filmulator-gui/core/nlmeans/eigen/test/cuda_common.h +++ b/filmulator-gui/core/nlmeans/eigen/test/cuda_common.h @@ -12,71 +12,67 @@ dim3 threadIdx, blockDim, blockIdx; #endif template -void run_on_cpu(const Kernel& ker, int n, const Input& in, Output& out) +void run_on_cpu(const Kernel &ker, int n, const Input &in, Output &out) { - for(int i=0; i -__global__ -void run_on_cuda_meta_kernel(const Kernel ker, int n, const Input* in, Output* out) +__global__ void run_on_cuda_meta_kernel(const Kernel ker, int n, const Input *in, Output *out) { - int i = threadIdx.x + blockIdx.x*blockDim.x; - if(i -void run_on_cuda(const Kernel& ker, int n, const Input& in, Output& out) +void run_on_cuda(const Kernel &ker, int n, const Input &in, Output &out) { - typename Input::Scalar* d_in; - typename Output::Scalar* d_out; - std::ptrdiff_t in_bytes = in.size() * sizeof(typename Input::Scalar); + typename Input::Scalar *d_in; + typename Output::Scalar *d_out; + std::ptrdiff_t in_bytes = in.size() * sizeof(typename Input::Scalar); std::ptrdiff_t out_bytes = out.size() * sizeof(typename Output::Scalar); - - cudaMalloc((void**)(&d_in), in_bytes); - cudaMalloc((void**)(&d_out), out_bytes); - - cudaMemcpy(d_in, in.data(), in_bytes, cudaMemcpyHostToDevice); + + cudaMalloc((void **)(&d_in), in_bytes); + cudaMalloc((void **)(&d_out), out_bytes); + + cudaMemcpy(d_in, in.data(), in_bytes, cudaMemcpyHostToDevice); cudaMemcpy(d_out, out.data(), out_bytes, cudaMemcpyHostToDevice); - + // Simple and non-optimal 1D mapping assuming n is not too large // That's only for unit testing! dim3 Blocks(128); - dim3 Grids( (n+int(Blocks.x)-1)/int(Blocks.x) ); + dim3 Grids((n + int(Blocks.x) - 1) / int(Blocks.x)); cudaThreadSynchronize(); - run_on_cuda_meta_kernel<<>>(ker, n, d_in, d_out); + run_on_cuda_meta_kernel<<>>(ker, n, d_in, d_out); cudaThreadSynchronize(); - + // check inputs have not been modified - cudaMemcpy(const_cast(in.data()), d_in, in_bytes, cudaMemcpyDeviceToHost); + cudaMemcpy(const_cast(in.data()), d_in, in_bytes, cudaMemcpyDeviceToHost); cudaMemcpy(out.data(), d_out, out_bytes, cudaMemcpyDeviceToHost); - + cudaFree(d_in); cudaFree(d_out); } template -void run_and_compare_to_cuda(const Kernel& ker, int n, const Input& in, Output& out) +void run_and_compare_to_cuda(const Kernel &ker, int n, const Input &in, Output &out) { - Input in_ref, in_cuda; + Input in_ref, in_cuda; Output out_ref, out_cuda; - #ifndef __CUDA_ARCH__ +#ifndef __CUDA_ARCH__ in_ref = in_cuda = in; out_ref = out_cuda = out; - #endif - run_on_cpu (ker, n, in_ref, out_ref); +#endif + run_on_cpu(ker, n, in_ref, out_ref); run_on_cuda(ker, n, in_cuda, out_cuda); - #ifndef __CUDA_ARCH__ +#ifndef __CUDA_ARCH__ VERIFY_IS_APPROX(in_ref, in_cuda); VERIFY_IS_APPROX(out_ref, out_cuda); - #endif +#endif } @@ -98,4 +94,4 @@ void ei_test_init_cuda() std::cout << " computeMode: " << deviceProp.computeMode << "\n"; } -#endif // EIGEN_TEST_CUDA_COMMON_H +#endif// EIGEN_TEST_CUDA_COMMON_H diff --git a/filmulator-gui/core/nlmeans/eigen/test/denseLM.cpp b/filmulator-gui/core/nlmeans/eigen/test/denseLM.cpp index 0aa736ea..a2f6fa78 100644 --- a/filmulator-gui/core/nlmeans/eigen/test/denseLM.cpp +++ b/filmulator-gui/core/nlmeans/eigen/test/denseLM.cpp @@ -8,183 +8,169 @@ // Public License v. 2.0. If a copy of the MPL was not distributed // with this file, You can obtain one at http://mozilla.org/MPL/2.0/. -#include #include #include +#include #include "main.h" #include using namespace std; using namespace Eigen; -template -struct DenseLM : DenseFunctor +template struct DenseLM : DenseFunctor { typedef DenseFunctor Base; typedef typename Base::JacobianType JacobianType; - typedef Matrix VectorType; - - DenseLM(int n, int m) : DenseFunctor(n,m) - { } - - VectorType model(const VectorType& uv, VectorType& x) + typedef Matrix VectorType; + + DenseLM(int n, int m) : DenseFunctor(n, m) {} + + VectorType model(const VectorType &uv, VectorType &x) { - VectorType y; // Should change to use expression template - int m = Base::values(); + VectorType y;// Should change to use expression template + int m = Base::values(); int n = Base::inputs(); - eigen_assert(uv.size()%2 == 0); + eigen_assert(uv.size() % 2 == 0); eigen_assert(uv.size() == n); eigen_assert(x.size() == m); y.setZero(m); - int half = n/2; + int half = n / 2; VectorBlock u(uv, 0, half); VectorBlock v(uv, half, half); - for (int j = 0; j < m; j++) - { - for (int i = 0; i < half; i++) - y(j) += u(i)*std::exp(-(x(j)-i)*(x(j)-i)/(v(i)*v(i))); + for (int j = 0; j < m; j++) { + for (int i = 0; i < half; i++) y(j) += u(i) * std::exp(-(x(j) - i) * (x(j) - i) / (v(i) * v(i))); } return y; - } - void initPoints(VectorType& uv_ref, VectorType& x) + void initPoints(VectorType &uv_ref, VectorType &x) { m_x = x; m_y = this->model(uv_ref, x); } - - int operator()(const VectorType& uv, VectorType& fvec) + + int operator()(const VectorType &uv, VectorType &fvec) { - - int m = Base::values(); + + int m = Base::values(); int n = Base::inputs(); - eigen_assert(uv.size()%2 == 0); + eigen_assert(uv.size() % 2 == 0); eigen_assert(uv.size() == n); eigen_assert(fvec.size() == m); - int half = n/2; + int half = n / 2; VectorBlock u(uv, 0, half); VectorBlock v(uv, half, half); - for (int j = 0; j < m; j++) - { + for (int j = 0; j < m; j++) { fvec(j) = m_y(j); - for (int i = 0; i < half; i++) - { - fvec(j) -= u(i) *std::exp(-(m_x(j)-i)*(m_x(j)-i)/(v(i)*v(i))); - } + for (int i = 0; i < half; i++) { fvec(j) -= u(i) * std::exp(-(m_x(j) - i) * (m_x(j) - i) / (v(i) * v(i))); } } - + return 0; } - int df(const VectorType& uv, JacobianType& fjac) + int df(const VectorType &uv, JacobianType &fjac) { - int m = Base::values(); + int m = Base::values(); int n = Base::inputs(); eigen_assert(n == uv.size()); eigen_assert(fjac.rows() == m); eigen_assert(fjac.cols() == n); - int half = n/2; + int half = n / 2; VectorBlock u(uv, 0, half); VectorBlock v(uv, half, half); - for (int j = 0; j < m; j++) - { - for (int i = 0; i < half; i++) - { - fjac.coeffRef(j,i) = -std::exp(-(m_x(j)-i)*(m_x(j)-i)/(v(i)*v(i))); - fjac.coeffRef(j,i+half) = -2.*u(i)*(m_x(j)-i)*(m_x(j)-i)/(std::pow(v(i),3)) * std::exp(-(m_x(j)-i)*(m_x(j)-i)/(v(i)*v(i))); + for (int j = 0; j < m; j++) { + for (int i = 0; i < half; i++) { + fjac.coeffRef(j, i) = -std::exp(-(m_x(j) - i) * (m_x(j) - i) / (v(i) * v(i))); + fjac.coeffRef(j, i + half) = -2. * u(i) * (m_x(j) - i) * (m_x(j) - i) / (std::pow(v(i), 3)) + * std::exp(-(m_x(j) - i) * (m_x(j) - i) / (v(i) * v(i))); } } return 0; } - VectorType m_x, m_y; //Data Points + VectorType m_x, m_y;// Data Points }; -template -int test_minimizeLM(FunctorType& functor, VectorType& uv) +template int test_minimizeLM(FunctorType &functor, VectorType &uv) { LevenbergMarquardt lm(functor); - LevenbergMarquardtSpace::Status info; - + LevenbergMarquardtSpace::Status info; + info = lm.minimize(uv); - + VERIFY_IS_EQUAL(info, 1); - //FIXME Check other parameters + // FIXME Check other parameters return info; } -template -int test_lmder(FunctorType& functor, VectorType& uv) +template int test_lmder(FunctorType &functor, VectorType &uv) { typedef typename VectorType::Scalar Scalar; - LevenbergMarquardtSpace::Status info; + LevenbergMarquardtSpace::Status info; LevenbergMarquardt lm(functor); info = lm.lmder1(uv); - + VERIFY_IS_EQUAL(info, 1); - //FIXME Check other parameters + // FIXME Check other parameters return info; } -template -int test_minimizeSteps(FunctorType& functor, VectorType& uv) +template int test_minimizeSteps(FunctorType &functor, VectorType &uv) { - LevenbergMarquardtSpace::Status info; + LevenbergMarquardtSpace::Status info; LevenbergMarquardt lm(functor); info = lm.minimizeInit(uv); - if (info==LevenbergMarquardtSpace::ImproperInputParameters) - return info; - do - { + if (info == LevenbergMarquardtSpace::ImproperInputParameters) return info; + do { info = lm.minimizeOneStep(uv); - } while (info==LevenbergMarquardtSpace::Running); - + } while (info == LevenbergMarquardtSpace::Running); + VERIFY_IS_EQUAL(info, 1); - //FIXME Check other parameters + // FIXME Check other parameters return info; } -template -void test_denseLM_T() +template void test_denseLM_T() { - typedef Matrix VectorType; - - int inputs = 10; - int values = 1000; + typedef Matrix VectorType; + + int inputs = 10; + int values = 1000; DenseLM dense_gaussian(inputs, values); - VectorType uv(inputs),uv_ref(inputs); + VectorType uv(inputs), uv_ref(inputs); VectorType x(values); - - // Generate the reference solution - uv_ref << -2, 1, 4 ,8, 6, 1.8, 1.2, 1.1, 1.9 , 3; - - //Generate the reference data points + + // Generate the reference solution + uv_ref << -2, 1, 4, 8, 6, 1.8, 1.2, 1.1, 1.9, 3; + + // Generate the reference data points x.setRandom(); - x = 10*x; + x = 10 * x; x.array() += 10; dense_gaussian.initPoints(uv_ref, x); - - // Generate the initial parameters - VectorBlock u(uv, 0, inputs/2); - VectorBlock v(uv, inputs/2, inputs/2); - + + // Generate the initial parameters + VectorBlock u(uv, 0, inputs / 2); + VectorBlock v(uv, inputs / 2, inputs / 2); + // Solve the optimization problem - - //Solve in one go - u.setOnes(); v.setOnes(); + + // Solve in one go + u.setOnes(); + v.setOnes(); test_minimizeLM(dense_gaussian, uv); - - //Solve until the machine precision - u.setOnes(); v.setOnes(); - test_lmder(dense_gaussian, uv); - + + // Solve until the machine precision + u.setOnes(); + v.setOnes(); + test_lmder(dense_gaussian, uv); + // Solve step by step - v.setOnes(); u.setOnes(); + v.setOnes(); + u.setOnes(); test_minimizeSteps(dense_gaussian, uv); - } void test_denseLM() { CALL_SUBTEST_2(test_denseLM_T()); - + // CALL_SUBTEST_2(test_sparseLM_T()); } diff --git a/filmulator-gui/core/nlmeans/eigen/test/dense_storage.cpp b/filmulator-gui/core/nlmeans/eigen/test/dense_storage.cpp index e63712b1..9e1b98cc 100644 --- a/filmulator-gui/core/nlmeans/eigen/test/dense_storage.cpp +++ b/filmulator-gui/core/nlmeans/eigen/test/dense_storage.cpp @@ -11,66 +11,60 @@ #include -template -void dense_storage_copy() +template void dense_storage_copy() { - static const int Size = ((Rows==Dynamic || Cols==Dynamic) ? Dynamic : Rows*Cols); - typedef DenseStorage DenseStorageType; - - const int rows = (Rows==Dynamic) ? 4 : Rows; - const int cols = (Cols==Dynamic) ? 3 : Cols; - const int size = rows*cols; + static const int Size = ((Rows == Dynamic || Cols == Dynamic) ? Dynamic : Rows * Cols); + typedef DenseStorage DenseStorageType; + + const int rows = (Rows == Dynamic) ? 4 : Rows; + const int cols = (Cols == Dynamic) ? 3 : Cols; + const int size = rows * cols; DenseStorageType reference(size, rows, cols); - T* raw_reference = reference.data(); - for (int i=0; i(i); - + T *raw_reference = reference.data(); + for (int i = 0; i < size; ++i) raw_reference[i] = static_cast(i); + DenseStorageType copied_reference(reference); - const T* raw_copied_reference = copied_reference.data(); - for (int i=0; i -void dense_storage_assignment() +template void dense_storage_assignment() { - static const int Size = ((Rows==Dynamic || Cols==Dynamic) ? Dynamic : Rows*Cols); - typedef DenseStorage DenseStorageType; - - const int rows = (Rows==Dynamic) ? 4 : Rows; - const int cols = (Cols==Dynamic) ? 3 : Cols; - const int size = rows*cols; + static const int Size = ((Rows == Dynamic || Cols == Dynamic) ? Dynamic : Rows * Cols); + typedef DenseStorage DenseStorageType; + + const int rows = (Rows == Dynamic) ? 4 : Rows; + const int cols = (Cols == Dynamic) ? 3 : Cols; + const int size = rows * cols; DenseStorageType reference(size, rows, cols); - T* raw_reference = reference.data(); - for (int i=0; i(i); - + T *raw_reference = reference.data(); + for (int i = 0; i < size; ++i) raw_reference[i] = static_cast(i); + DenseStorageType copied_reference; copied_reference = reference; - const T* raw_copied_reference = copied_reference.data(); - for (int i=0; i(); - dense_storage_copy(); - dense_storage_copy(); - dense_storage_copy(); + dense_storage_copy(); + dense_storage_copy(); + dense_storage_copy(); + dense_storage_copy(); + + dense_storage_copy(); + dense_storage_copy(); + dense_storage_copy(); + dense_storage_copy(); - dense_storage_copy(); - dense_storage_copy(); - dense_storage_copy(); - dense_storage_copy(); - - dense_storage_assignment(); - dense_storage_assignment(); - dense_storage_assignment(); - dense_storage_assignment(); + dense_storage_assignment(); + dense_storage_assignment(); + dense_storage_assignment(); + dense_storage_assignment(); - dense_storage_assignment(); - dense_storage_assignment(); - dense_storage_assignment(); - dense_storage_assignment(); + dense_storage_assignment(); + dense_storage_assignment(); + dense_storage_assignment(); + dense_storage_assignment(); } diff --git a/filmulator-gui/core/nlmeans/eigen/test/determinant.cpp b/filmulator-gui/core/nlmeans/eigen/test/determinant.cpp index b8c9babb..875831d0 100644 --- a/filmulator-gui/core/nlmeans/eigen/test/determinant.cpp +++ b/filmulator-gui/core/nlmeans/eigen/test/determinant.cpp @@ -11,7 +11,7 @@ #include "main.h" #include -template void determinant(const MatrixType& m) +template void determinant(const MatrixType &m) { /* this test covers the following files: Determinant.h @@ -24,13 +24,13 @@ template void determinant(const MatrixType& m) typedef typename MatrixType::Scalar Scalar; Scalar x = internal::random(); VERIFY_IS_APPROX(MatrixType::Identity(size, size).determinant(), Scalar(1)); - VERIFY_IS_APPROX((m1*m2).eval().determinant(), m1.determinant() * m2.determinant()); - if(size==1) return; - Index i = internal::random(0, size-1); + VERIFY_IS_APPROX((m1 * m2).eval().determinant(), m1.determinant() * m2.determinant()); + if (size == 1) return; + Index i = internal::random(0, size - 1); Index j; do { - j = internal::random(0, size-1); - } while(j==i); + j = internal::random(0, size - 1); + } while (j == i); m2 = m1; m2.row(i).swap(m2.row(j)); VERIFY_IS_APPROX(m2.determinant(), -m1.determinant()); @@ -40,27 +40,27 @@ template void determinant(const MatrixType& m) VERIFY_IS_APPROX(m2.determinant(), m2.transpose().determinant()); VERIFY_IS_APPROX(numext::conj(m2.determinant()), m2.adjoint().determinant()); m2 = m1; - m2.row(i) += x*m2.row(j); + m2.row(i) += x * m2.row(j); VERIFY_IS_APPROX(m2.determinant(), m1.determinant()); m2 = m1; m2.row(i) *= x; VERIFY_IS_APPROX(m2.determinant(), m1.determinant() * x); - + // check empty matrix - VERIFY_IS_APPROX(m2.block(0,0,0,0).determinant(), Scalar(1)); + VERIFY_IS_APPROX(m2.block(0, 0, 0, 0).determinant(), Scalar(1)); } void test_determinant() { - for(int i = 0; i < g_repeat; i++) { + for (int i = 0; i < g_repeat; i++) { int s = 0; - CALL_SUBTEST_1( determinant(Matrix()) ); - CALL_SUBTEST_2( determinant(Matrix()) ); - CALL_SUBTEST_3( determinant(Matrix()) ); - CALL_SUBTEST_4( determinant(Matrix()) ); - CALL_SUBTEST_5( determinant(Matrix, 10, 10>()) ); - s = internal::random(1,EIGEN_TEST_MAX_SIZE/4); - CALL_SUBTEST_6( determinant(MatrixXd(s, s)) ); + CALL_SUBTEST_1(determinant(Matrix())); + CALL_SUBTEST_2(determinant(Matrix())); + CALL_SUBTEST_3(determinant(Matrix())); + CALL_SUBTEST_4(determinant(Matrix())); + CALL_SUBTEST_5(determinant(Matrix, 10, 10>())); + s = internal::random(1, EIGEN_TEST_MAX_SIZE / 4); + CALL_SUBTEST_6(determinant(MatrixXd(s, s))); TEST_SET_BUT_UNUSED_VARIABLE(s) } } diff --git a/filmulator-gui/core/nlmeans/eigen/test/diagonal.cpp b/filmulator-gui/core/nlmeans/eigen/test/diagonal.cpp index 8ed9b468..4a8f51ae 100644 --- a/filmulator-gui/core/nlmeans/eigen/test/diagonal.cpp +++ b/filmulator-gui/core/nlmeans/eigen/test/diagonal.cpp @@ -9,33 +9,27 @@ #include "main.h" -template void diagonal(const MatrixType& m) +template void diagonal(const MatrixType &m) { typedef typename MatrixType::Scalar Scalar; Index rows = m.rows(); Index cols = m.cols(); - MatrixType m1 = MatrixType::Random(rows, cols), - m2 = MatrixType::Random(rows, cols); + MatrixType m1 = MatrixType::Random(rows, cols), m2 = MatrixType::Random(rows, cols); Scalar s1 = internal::random(); - //check diagonal() + // check diagonal() VERIFY_IS_APPROX(m1.diagonal(), m1.transpose().diagonal()); m2.diagonal() = 2 * m1.diagonal(); m2.diagonal()[0] *= 3; - if (rows>2) - { - enum { - N1 = MatrixType::RowsAtCompileTime>2 ? 2 : 0, - N2 = MatrixType::RowsAtCompileTime>1 ? -1 : 0 - }; + if (rows > 2) { + enum { N1 = MatrixType::RowsAtCompileTime > 2 ? 2 : 0, N2 = MatrixType::RowsAtCompileTime > 1 ? -1 : 0 }; // check sub/super diagonal - if(MatrixType::SizeAtCompileTime!=Dynamic) - { + if (MatrixType::SizeAtCompileTime != Dynamic) { VERIFY(m1.template diagonal().RowsAtCompileTime == m1.diagonal(N1).size()); VERIFY(m1.template diagonal().RowsAtCompileTime == m1.diagonal(N2).size()); } @@ -62,44 +56,49 @@ template void diagonal(const MatrixType& m) m2.diagonal(N2).x() = s1; VERIFY_IS_APPROX(m2.diagonal(N2).x(), s1); - m2.diagonal(N2).coeffRef(0) = Scalar(2)*s1; - VERIFY_IS_APPROX(m2.diagonal(N2).coeff(0), Scalar(2)*s1); + m2.diagonal(N2).coeffRef(0) = Scalar(2) * s1; + VERIFY_IS_APPROX(m2.diagonal(N2).coeff(0), Scalar(2) * s1); } - VERIFY( m1.diagonal( cols).size()==0 ); - VERIFY( m1.diagonal(-rows).size()==0 ); + VERIFY(m1.diagonal(cols).size() == 0); + VERIFY(m1.diagonal(-rows).size() == 0); } -template void diagonal_assert(const MatrixType& m) { +template void diagonal_assert(const MatrixType &m) +{ Index rows = m.rows(); Index cols = m.cols(); MatrixType m1 = MatrixType::Random(rows, cols); - if (rows>=2 && cols>=2) - { - VERIFY_RAISES_ASSERT( m1 += m1.diagonal() ); - VERIFY_RAISES_ASSERT( m1 -= m1.diagonal() ); - VERIFY_RAISES_ASSERT( m1.array() *= m1.diagonal().array() ); - VERIFY_RAISES_ASSERT( m1.array() /= m1.diagonal().array() ); + if (rows >= 2 && cols >= 2) { + VERIFY_RAISES_ASSERT(m1 += m1.diagonal()); + VERIFY_RAISES_ASSERT(m1 -= m1.diagonal()); + VERIFY_RAISES_ASSERT(m1.array() *= m1.diagonal().array()); + VERIFY_RAISES_ASSERT(m1.array() /= m1.diagonal().array()); } - VERIFY_RAISES_ASSERT( m1.diagonal(cols+1) ); - VERIFY_RAISES_ASSERT( m1.diagonal(-(rows+1)) ); + VERIFY_RAISES_ASSERT(m1.diagonal(cols + 1)); + VERIFY_RAISES_ASSERT(m1.diagonal(-(rows + 1))); } void test_diagonal() { - for(int i = 0; i < g_repeat; i++) { - CALL_SUBTEST_1( diagonal(Matrix()) ); - CALL_SUBTEST_1( diagonal(Matrix()) ); - CALL_SUBTEST_1( diagonal(Matrix()) ); - CALL_SUBTEST_2( diagonal(Matrix4d()) ); - CALL_SUBTEST_2( diagonal(MatrixXcf(internal::random(1,EIGEN_TEST_MAX_SIZE), internal::random(1,EIGEN_TEST_MAX_SIZE))) ); - CALL_SUBTEST_2( diagonal(MatrixXi(internal::random(1,EIGEN_TEST_MAX_SIZE), internal::random(1,EIGEN_TEST_MAX_SIZE))) ); - CALL_SUBTEST_2( diagonal(MatrixXcd(internal::random(1,EIGEN_TEST_MAX_SIZE), internal::random(1,EIGEN_TEST_MAX_SIZE))) ); - CALL_SUBTEST_1( diagonal(MatrixXf(internal::random(1,EIGEN_TEST_MAX_SIZE), internal::random(1,EIGEN_TEST_MAX_SIZE))) ); - CALL_SUBTEST_1( diagonal(Matrix(3, 4)) ); - CALL_SUBTEST_1( diagonal_assert(MatrixXf(internal::random(1,EIGEN_TEST_MAX_SIZE), internal::random(1,EIGEN_TEST_MAX_SIZE))) ); + for (int i = 0; i < g_repeat; i++) { + CALL_SUBTEST_1(diagonal(Matrix())); + CALL_SUBTEST_1(diagonal(Matrix())); + CALL_SUBTEST_1(diagonal(Matrix())); + CALL_SUBTEST_2(diagonal(Matrix4d())); + CALL_SUBTEST_2(diagonal( + MatrixXcf(internal::random(1, EIGEN_TEST_MAX_SIZE), internal::random(1, EIGEN_TEST_MAX_SIZE)))); + CALL_SUBTEST_2( + diagonal(MatrixXi(internal::random(1, EIGEN_TEST_MAX_SIZE), internal::random(1, EIGEN_TEST_MAX_SIZE)))); + CALL_SUBTEST_2(diagonal( + MatrixXcd(internal::random(1, EIGEN_TEST_MAX_SIZE), internal::random(1, EIGEN_TEST_MAX_SIZE)))); + CALL_SUBTEST_1( + diagonal(MatrixXf(internal::random(1, EIGEN_TEST_MAX_SIZE), internal::random(1, EIGEN_TEST_MAX_SIZE)))); + CALL_SUBTEST_1(diagonal(Matrix(3, 4))); + CALL_SUBTEST_1(diagonal_assert( + MatrixXf(internal::random(1, EIGEN_TEST_MAX_SIZE), internal::random(1, EIGEN_TEST_MAX_SIZE)))); } } diff --git a/filmulator-gui/core/nlmeans/eigen/test/diagonalmatrices.cpp b/filmulator-gui/core/nlmeans/eigen/test/diagonalmatrices.cpp index c55733df..166f2838 100644 --- a/filmulator-gui/core/nlmeans/eigen/test/diagonalmatrices.cpp +++ b/filmulator-gui/core/nlmeans/eigen/test/diagonalmatrices.cpp @@ -9,7 +9,7 @@ #include "main.h" using namespace std; -template void diagonalmatrices(const MatrixType& m) +template void diagonalmatrices(const MatrixType &m) { typedef typename MatrixType::Scalar Scalar; enum { Rows = MatrixType::RowsAtCompileTime, Cols = MatrixType::ColsAtCompileTime }; @@ -19,95 +19,92 @@ template void diagonalmatrices(const MatrixType& m) typedef Matrix DynMatrixType; typedef DiagonalMatrix LeftDiagonalMatrix; typedef DiagonalMatrix RightDiagonalMatrix; - typedef Matrix BigMatrix; + typedef Matrix BigMatrix; Index rows = m.rows(); Index cols = m.cols(); - MatrixType m1 = MatrixType::Random(rows, cols), - m2 = MatrixType::Random(rows, cols); - VectorType v1 = VectorType::Random(rows), - v2 = VectorType::Random(rows); - RowVectorType rv1 = RowVectorType::Random(cols), - rv2 = RowVectorType::Random(cols); + MatrixType m1 = MatrixType::Random(rows, cols), m2 = MatrixType::Random(rows, cols); + VectorType v1 = VectorType::Random(rows), v2 = VectorType::Random(rows); + RowVectorType rv1 = RowVectorType::Random(cols), rv2 = RowVectorType::Random(cols); LeftDiagonalMatrix ldm1(v1), ldm2(v2); RightDiagonalMatrix rdm1(rv1), rdm2(rv2); - + Scalar s1 = internal::random(); - SquareMatrixType sq_m1 (v1.asDiagonal()); + SquareMatrixType sq_m1(v1.asDiagonal()); VERIFY_IS_APPROX(sq_m1, v1.asDiagonal().toDenseMatrix()); sq_m1 = v1.asDiagonal(); VERIFY_IS_APPROX(sq_m1, v1.asDiagonal().toDenseMatrix()); SquareMatrixType sq_m2 = v1.asDiagonal(); VERIFY_IS_APPROX(sq_m1, sq_m2); - + ldm1 = v1.asDiagonal(); LeftDiagonalMatrix ldm3(v1); VERIFY_IS_APPROX(ldm1.diagonal(), ldm3.diagonal()); LeftDiagonalMatrix ldm4 = v1.asDiagonal(); VERIFY_IS_APPROX(ldm1.diagonal(), ldm4.diagonal()); - - sq_m1.block(0,0,rows,rows) = ldm1; + + sq_m1.block(0, 0, rows, rows) = ldm1; VERIFY_IS_APPROX(sq_m1, ldm1.toDenseMatrix()); sq_m1.transpose() = ldm1; VERIFY_IS_APPROX(sq_m1, ldm1.toDenseMatrix()); - - Index i = internal::random(0, rows-1); - Index j = internal::random(0, cols-1); - - VERIFY_IS_APPROX( ((ldm1 * m1)(i,j)) , ldm1.diagonal()(i) * m1(i,j) ); - VERIFY_IS_APPROX( ((ldm1 * (m1+m2))(i,j)) , ldm1.diagonal()(i) * (m1+m2)(i,j) ); - VERIFY_IS_APPROX( ((m1 * rdm1)(i,j)) , rdm1.diagonal()(j) * m1(i,j) ); - VERIFY_IS_APPROX( ((v1.asDiagonal() * m1)(i,j)) , v1(i) * m1(i,j) ); - VERIFY_IS_APPROX( ((m1 * rv1.asDiagonal())(i,j)) , rv1(j) * m1(i,j) ); - VERIFY_IS_APPROX( (((v1+v2).asDiagonal() * m1)(i,j)) , (v1+v2)(i) * m1(i,j) ); - VERIFY_IS_APPROX( (((v1+v2).asDiagonal() * (m1+m2))(i,j)) , (v1+v2)(i) * (m1+m2)(i,j) ); - VERIFY_IS_APPROX( ((m1 * (rv1+rv2).asDiagonal())(i,j)) , (rv1+rv2)(j) * m1(i,j) ); - VERIFY_IS_APPROX( (((m1+m2) * (rv1+rv2).asDiagonal())(i,j)) , (rv1+rv2)(j) * (m1+m2)(i,j) ); - - if(rows>1) - { - DynMatrixType tmp = m1.topRows(rows/2), res; - VERIFY_IS_APPROX( (res = m1.topRows(rows/2) * rv1.asDiagonal()), tmp * rv1.asDiagonal() ); - VERIFY_IS_APPROX( (res = v1.head(rows/2).asDiagonal()*m1.topRows(rows/2)), v1.head(rows/2).asDiagonal()*tmp ); + + Index i = internal::random(0, rows - 1); + Index j = internal::random(0, cols - 1); + + VERIFY_IS_APPROX(((ldm1 * m1)(i, j)), ldm1.diagonal()(i) * m1(i, j)); + VERIFY_IS_APPROX(((ldm1 * (m1 + m2))(i, j)), ldm1.diagonal()(i) * (m1 + m2)(i, j)); + VERIFY_IS_APPROX(((m1 * rdm1)(i, j)), rdm1.diagonal()(j) * m1(i, j)); + VERIFY_IS_APPROX(((v1.asDiagonal() * m1)(i, j)), v1(i) * m1(i, j)); + VERIFY_IS_APPROX(((m1 * rv1.asDiagonal())(i, j)), rv1(j) * m1(i, j)); + VERIFY_IS_APPROX((((v1 + v2).asDiagonal() * m1)(i, j)), (v1 + v2)(i)*m1(i, j)); + VERIFY_IS_APPROX((((v1 + v2).asDiagonal() * (m1 + m2))(i, j)), (v1 + v2)(i) * (m1 + m2)(i, j)); + VERIFY_IS_APPROX(((m1 * (rv1 + rv2).asDiagonal())(i, j)), (rv1 + rv2)(j)*m1(i, j)); + VERIFY_IS_APPROX((((m1 + m2) * (rv1 + rv2).asDiagonal())(i, j)), (rv1 + rv2)(j) * (m1 + m2)(i, j)); + + if (rows > 1) { + DynMatrixType tmp = m1.topRows(rows / 2), res; + VERIFY_IS_APPROX((res = m1.topRows(rows / 2) * rv1.asDiagonal()), tmp * rv1.asDiagonal()); + VERIFY_IS_APPROX( + (res = v1.head(rows / 2).asDiagonal() * m1.topRows(rows / 2)), v1.head(rows / 2).asDiagonal() * tmp); } BigMatrix big; - big.setZero(2*rows, 2*cols); - - big.block(i,j,rows,cols) = m1; - big.block(i,j,rows,cols) = v1.asDiagonal() * big.block(i,j,rows,cols); - - VERIFY_IS_APPROX((big.block(i,j,rows,cols)) , v1.asDiagonal() * m1 ); - - big.block(i,j,rows,cols) = m1; - big.block(i,j,rows,cols) = big.block(i,j,rows,cols) * rv1.asDiagonal(); - VERIFY_IS_APPROX((big.block(i,j,rows,cols)) , m1 * rv1.asDiagonal() ); - - + big.setZero(2 * rows, 2 * cols); + + big.block(i, j, rows, cols) = m1; + big.block(i, j, rows, cols) = v1.asDiagonal() * big.block(i, j, rows, cols); + + VERIFY_IS_APPROX((big.block(i, j, rows, cols)), v1.asDiagonal() * m1); + + big.block(i, j, rows, cols) = m1; + big.block(i, j, rows, cols) = big.block(i, j, rows, cols) * rv1.asDiagonal(); + VERIFY_IS_APPROX((big.block(i, j, rows, cols)), m1 * rv1.asDiagonal()); + + // scalar multiple - VERIFY_IS_APPROX(LeftDiagonalMatrix(ldm1*s1).diagonal(), ldm1.diagonal() * s1); - VERIFY_IS_APPROX(LeftDiagonalMatrix(s1*ldm1).diagonal(), s1 * ldm1.diagonal()); - + VERIFY_IS_APPROX(LeftDiagonalMatrix(ldm1 * s1).diagonal(), ldm1.diagonal() * s1); + VERIFY_IS_APPROX(LeftDiagonalMatrix(s1 * ldm1).diagonal(), s1 * ldm1.diagonal()); + VERIFY_IS_APPROX(m1 * (rdm1 * s1), (m1 * rdm1) * s1); VERIFY_IS_APPROX(m1 * (s1 * rdm1), (m1 * rdm1) * s1); - + // Diagonal to dense sq_m1.setRandom(); sq_m2 = sq_m1; - VERIFY_IS_APPROX( (sq_m1 += (s1*v1).asDiagonal()), sq_m2 += (s1*v1).asDiagonal().toDenseMatrix() ); - VERIFY_IS_APPROX( (sq_m1 -= (s1*v1).asDiagonal()), sq_m2 -= (s1*v1).asDiagonal().toDenseMatrix() ); - VERIFY_IS_APPROX( (sq_m1 = (s1*v1).asDiagonal()), (s1*v1).asDiagonal().toDenseMatrix() ); + VERIFY_IS_APPROX((sq_m1 += (s1 * v1).asDiagonal()), sq_m2 += (s1 * v1).asDiagonal().toDenseMatrix()); + VERIFY_IS_APPROX((sq_m1 -= (s1 * v1).asDiagonal()), sq_m2 -= (s1 * v1).asDiagonal().toDenseMatrix()); + VERIFY_IS_APPROX((sq_m1 = (s1 * v1).asDiagonal()), (s1 * v1).asDiagonal().toDenseMatrix()); sq_m1.setRandom(); sq_m2 = v1.asDiagonal(); sq_m2 = sq_m1 * sq_m2; - VERIFY_IS_APPROX( (sq_m1*v1.asDiagonal()).col(i), sq_m2.col(i) ); - VERIFY_IS_APPROX( (sq_m1*v1.asDiagonal()).row(i), sq_m2.row(i) ); + VERIFY_IS_APPROX((sq_m1 * v1.asDiagonal()).col(i), sq_m2.col(i)); + VERIFY_IS_APPROX((sq_m1 * v1.asDiagonal()).row(i), sq_m2.row(i)); } -template void as_scalar_product(const MatrixType& m) +template void as_scalar_product(const MatrixType &m) { typedef typename MatrixType::Scalar Scalar; typedef Matrix VectorType; @@ -116,51 +113,54 @@ template void as_scalar_product(const MatrixType& m) typedef Matrix DynRowVectorType; Index rows = m.rows(); - Index depth = internal::random(1,EIGEN_TEST_MAX_SIZE); - - VectorType v1 = VectorType::Random(rows); - DynVectorType dv1 = DynVectorType::Random(depth); - DynRowVectorType drv1 = DynRowVectorType::Random(depth); - DynMatrixType dm1 = dv1; - DynMatrixType drm1 = drv1; - + Index depth = internal::random(1, EIGEN_TEST_MAX_SIZE); + + VectorType v1 = VectorType::Random(rows); + DynVectorType dv1 = DynVectorType::Random(depth); + DynRowVectorType drv1 = DynRowVectorType::Random(depth); + DynMatrixType dm1 = dv1; + DynMatrixType drm1 = drv1; + Scalar s = v1(0); - VERIFY_IS_APPROX( v1.asDiagonal() * drv1, s*drv1 ); - VERIFY_IS_APPROX( dv1 * v1.asDiagonal(), dv1*s ); + VERIFY_IS_APPROX(v1.asDiagonal() * drv1, s * drv1); + VERIFY_IS_APPROX(dv1 * v1.asDiagonal(), dv1 * s); - VERIFY_IS_APPROX( v1.asDiagonal() * drm1, s*drm1 ); - VERIFY_IS_APPROX( dm1 * v1.asDiagonal(), dm1*s ); + VERIFY_IS_APPROX(v1.asDiagonal() * drm1, s * drm1); + VERIFY_IS_APPROX(dm1 * v1.asDiagonal(), dm1 * s); } -template -void bug987() +template void bug987() { Matrix3Xd points = Matrix3Xd::Random(3, 3); Vector2d diag = Vector2d::Random(); Matrix2Xd tmp1 = points.topRows<2>(), res1, res2; - VERIFY_IS_APPROX( res1 = diag.asDiagonal() * points.topRows<2>(), res2 = diag.asDiagonal() * tmp1 ); - Matrix2d tmp2 = points.topLeftCorner<2,2>(); - VERIFY_IS_APPROX(( res1 = points.topLeftCorner<2,2>()*diag.asDiagonal()) , res2 = tmp2*diag.asDiagonal() ); + VERIFY_IS_APPROX(res1 = diag.asDiagonal() * points.topRows<2>(), res2 = diag.asDiagonal() * tmp1); + Matrix2d tmp2 = points.topLeftCorner<2, 2>(); + VERIFY_IS_APPROX((res1 = points.topLeftCorner<2, 2>() * diag.asDiagonal()), res2 = tmp2 * diag.asDiagonal()); } void test_diagonalmatrices() { - for(int i = 0; i < g_repeat; i++) { - CALL_SUBTEST_1( diagonalmatrices(Matrix()) ); - CALL_SUBTEST_1( as_scalar_product(Matrix()) ); - - CALL_SUBTEST_2( diagonalmatrices(Matrix3f()) ); - CALL_SUBTEST_3( diagonalmatrices(Matrix()) ); - CALL_SUBTEST_4( diagonalmatrices(Matrix4d()) ); - CALL_SUBTEST_5( diagonalmatrices(Matrix()) ); - CALL_SUBTEST_6( diagonalmatrices(MatrixXcf(internal::random(1,EIGEN_TEST_MAX_SIZE), internal::random(1,EIGEN_TEST_MAX_SIZE))) ); - CALL_SUBTEST_6( as_scalar_product(MatrixXcf(1,1)) ); - CALL_SUBTEST_7( diagonalmatrices(MatrixXi(internal::random(1,EIGEN_TEST_MAX_SIZE), internal::random(1,EIGEN_TEST_MAX_SIZE))) ); - CALL_SUBTEST_8( diagonalmatrices(Matrix(internal::random(1,EIGEN_TEST_MAX_SIZE), internal::random(1,EIGEN_TEST_MAX_SIZE))) ); - CALL_SUBTEST_9( diagonalmatrices(MatrixXf(internal::random(1,EIGEN_TEST_MAX_SIZE), internal::random(1,EIGEN_TEST_MAX_SIZE))) ); - CALL_SUBTEST_9( diagonalmatrices(MatrixXf(1,1)) ); - CALL_SUBTEST_9( as_scalar_product(MatrixXf(1,1)) ); + for (int i = 0; i < g_repeat; i++) { + CALL_SUBTEST_1(diagonalmatrices(Matrix())); + CALL_SUBTEST_1(as_scalar_product(Matrix())); + + CALL_SUBTEST_2(diagonalmatrices(Matrix3f())); + CALL_SUBTEST_3(diagonalmatrices(Matrix())); + CALL_SUBTEST_4(diagonalmatrices(Matrix4d())); + CALL_SUBTEST_5(diagonalmatrices(Matrix())); + CALL_SUBTEST_6(diagonalmatrices( + MatrixXcf(internal::random(1, EIGEN_TEST_MAX_SIZE), internal::random(1, EIGEN_TEST_MAX_SIZE)))); + CALL_SUBTEST_6(as_scalar_product(MatrixXcf(1, 1))); + CALL_SUBTEST_7(diagonalmatrices( + MatrixXi(internal::random(1, EIGEN_TEST_MAX_SIZE), internal::random(1, EIGEN_TEST_MAX_SIZE)))); + CALL_SUBTEST_8(diagonalmatrices(Matrix( + internal::random(1, EIGEN_TEST_MAX_SIZE), internal::random(1, EIGEN_TEST_MAX_SIZE)))); + CALL_SUBTEST_9(diagonalmatrices( + MatrixXf(internal::random(1, EIGEN_TEST_MAX_SIZE), internal::random(1, EIGEN_TEST_MAX_SIZE)))); + CALL_SUBTEST_9(diagonalmatrices(MatrixXf(1, 1))); + CALL_SUBTEST_9(as_scalar_product(MatrixXf(1, 1))); } - CALL_SUBTEST_10( bug987<0>() ); + CALL_SUBTEST_10(bug987<0>()); } diff --git a/filmulator-gui/core/nlmeans/eigen/test/dontalign.cpp b/filmulator-gui/core/nlmeans/eigen/test/dontalign.cpp index ac00112e..98717d96 100644 --- a/filmulator-gui/core/nlmeans/eigen/test/dontalign.cpp +++ b/filmulator-gui/core/nlmeans/eigen/test/dontalign.cpp @@ -16,8 +16,7 @@ #include "main.h" #include -template -void dontalign(const MatrixType& m) +template void dontalign(const MatrixType &m) { typedef typename MatrixType::Scalar Scalar; typedef Matrix VectorType; @@ -26,20 +25,20 @@ void dontalign(const MatrixType& m) Index rows = m.rows(); Index cols = m.cols(); - MatrixType a = MatrixType::Random(rows,cols); - SquareMatrixType square = SquareMatrixType::Random(rows,rows); + MatrixType a = MatrixType::Random(rows, cols); + SquareMatrixType square = SquareMatrixType::Random(rows, rows); VectorType v = VectorType::Random(rows); VERIFY_IS_APPROX(v, square * square.colPivHouseholderQr().solve(v)); square = square.inverse().eval(); a = square * a; - square = square*square; + square = square * square; v = square * v; v = a.adjoint() * v; VERIFY(square.determinant() != Scalar(0)); // bug 219: MapAligned() was giving an assert with EIGEN_DONT_ALIGN, because Map Flags were miscomputed - Scalar* array = internal::aligned_new(rows); + Scalar *array = internal::aligned_new(rows); v = VectorType::MapAligned(array, rows); internal::aligned_delete(array, rows); } diff --git a/filmulator-gui/core/nlmeans/eigen/test/dynalloc.cpp b/filmulator-gui/core/nlmeans/eigen/test/dynalloc.cpp index f1cc70be..05052ad4 100644 --- a/filmulator-gui/core/nlmeans/eigen/test/dynalloc.cpp +++ b/filmulator-gui/core/nlmeans/eigen/test/dynalloc.cpp @@ -9,58 +9,54 @@ #include "main.h" -#if EIGEN_MAX_ALIGN_BYTES>0 +#if EIGEN_MAX_ALIGN_BYTES > 0 #define ALIGNMENT EIGEN_MAX_ALIGN_BYTES #else #define ALIGNMENT 1 #endif -typedef Matrix Vector8f; +typedef Matrix Vector8f; void check_handmade_aligned_malloc() { - for(int i = 1; i < 1000; i++) - { - char *p = (char*)internal::handmade_aligned_malloc(i); - VERIFY(internal::UIntPtr(p)%ALIGNMENT==0); + for (int i = 1; i < 1000; i++) { + char *p = (char *)internal::handmade_aligned_malloc(i); + VERIFY(internal::UIntPtr(p) % ALIGNMENT == 0); // if the buffer is wrongly allocated this will give a bad write --> check with valgrind - for(int j = 0; j < i; j++) p[j]=0; + for (int j = 0; j < i; j++) p[j] = 0; internal::handmade_aligned_free(p); } } void check_aligned_malloc() { - for(int i = ALIGNMENT; i < 1000; i++) - { - char *p = (char*)internal::aligned_malloc(i); - VERIFY(internal::UIntPtr(p)%ALIGNMENT==0); + for (int i = ALIGNMENT; i < 1000; i++) { + char *p = (char *)internal::aligned_malloc(i); + VERIFY(internal::UIntPtr(p) % ALIGNMENT == 0); // if the buffer is wrongly allocated this will give a bad write --> check with valgrind - for(int j = 0; j < i; j++) p[j]=0; + for (int j = 0; j < i; j++) p[j] = 0; internal::aligned_free(p); } } void check_aligned_new() { - for(int i = ALIGNMENT; i < 1000; i++) - { + for (int i = ALIGNMENT; i < 1000; i++) { float *p = internal::aligned_new(i); - VERIFY(internal::UIntPtr(p)%ALIGNMENT==0); + VERIFY(internal::UIntPtr(p) % ALIGNMENT == 0); // if the buffer is wrongly allocated this will give a bad write --> check with valgrind - for(int j = 0; j < i; j++) p[j]=0; - internal::aligned_delete(p,i); + for (int j = 0; j < i; j++) p[j] = 0; + internal::aligned_delete(p, i); } } void check_aligned_stack_alloc() { - for(int i = ALIGNMENT; i < 400; i++) - { - ei_declare_aligned_stack_constructed_variable(float,p,i,0); - VERIFY(internal::UIntPtr(p)%ALIGNMENT==0); + for (int i = ALIGNMENT; i < 400; i++) { + ei_declare_aligned_stack_constructed_variable(float, p, i, 0); + VERIFY(internal::UIntPtr(p) % ALIGNMENT == 0); // if the buffer is wrongly allocated this will give a bad write --> check with valgrind - for(int j = 0; j < i; j++) p[j]=0; + for (int j = 0; j < i; j++) p[j] = 0; } } @@ -75,20 +71,19 @@ struct MyStruct class MyClassA { - public: - EIGEN_MAKE_ALIGNED_OPERATOR_NEW - char dummychar; - Vector8f avec; +public: + EIGEN_MAKE_ALIGNED_OPERATOR_NEW + char dummychar; + Vector8f avec; }; template void check_dynaligned() { // TODO have to be updated once we support multiple alignment values - if(T::SizeAtCompileTime % ALIGNMENT == 0) - { - T* obj = new T; - VERIFY(T::NeedsToAlign==1); - VERIFY(internal::UIntPtr(obj)%ALIGNMENT==0); + if (T::SizeAtCompileTime % ALIGNMENT == 0) { + T *obj = new T; + VERIFY(T::NeedsToAlign == 1); + VERIFY(internal::UIntPtr(obj) % ALIGNMENT == 0); delete obj; } } @@ -96,24 +91,24 @@ template void check_dynaligned() template void check_custom_new_delete() { { - T* t = new T; + T *t = new T; delete t; } - + { - std::size_t N = internal::random(1,10); - T* t = new T[N]; + std::size_t N = internal::random(1, 10); + T *t = new T[N]; delete[] t; } - -#if EIGEN_MAX_ALIGN_BYTES>0 + +#if EIGEN_MAX_ALIGN_BYTES > 0 { - T* t = static_cast((T::operator new)(sizeof(T))); + T *t = static_cast((T::operator new)(sizeof(T))); (T::operator delete)(t, sizeof(T)); } - + { - T* t = static_cast((T::operator new)(sizeof(T))); + T *t = static_cast((T::operator new)(sizeof(T))); (T::operator delete)(t); } #endif @@ -127,49 +122,50 @@ void test_dynalloc() CALL_SUBTEST(check_aligned_new()); CALL_SUBTEST(check_aligned_stack_alloc()); - for (int i=0; i() ); - CALL_SUBTEST( check_custom_new_delete() ); - CALL_SUBTEST( check_custom_new_delete() ); - CALL_SUBTEST( check_custom_new_delete() ); + for (int i = 0; i < g_repeat * 100; ++i) { + CALL_SUBTEST(check_custom_new_delete()); + CALL_SUBTEST(check_custom_new_delete()); + CALL_SUBTEST(check_custom_new_delete()); + CALL_SUBTEST(check_custom_new_delete()); } - - // check static allocation, who knows ? - #if EIGEN_MAX_STATIC_ALIGN_BYTES - for (int i=0; i() ); - CALL_SUBTEST(check_dynaligned() ); - CALL_SUBTEST(check_dynaligned() ); - CALL_SUBTEST(check_dynaligned() ); - CALL_SUBTEST(check_dynaligned() ); - CALL_SUBTEST(check_dynaligned() ); + +// check static allocation, who knows ? +#if EIGEN_MAX_STATIC_ALIGN_BYTES + for (int i = 0; i < g_repeat * 100; ++i) { + CALL_SUBTEST(check_dynaligned()); + CALL_SUBTEST(check_dynaligned()); + CALL_SUBTEST(check_dynaligned()); + CALL_SUBTEST(check_dynaligned()); + CALL_SUBTEST(check_dynaligned()); + CALL_SUBTEST(check_dynaligned()); } { - MyStruct foo0; VERIFY(internal::UIntPtr(foo0.avec.data())%ALIGNMENT==0); - MyClassA fooA; VERIFY(internal::UIntPtr(fooA.avec.data())%ALIGNMENT==0); + MyStruct foo0; + VERIFY(internal::UIntPtr(foo0.avec.data()) % ALIGNMENT == 0); + MyClassA fooA; + VERIFY(internal::UIntPtr(fooA.avec.data()) % ALIGNMENT == 0); } - + // dynamic allocation, single object - for (int i=0; iavec.data())%ALIGNMENT==0); - MyClassA *fooA = new MyClassA(); VERIFY(internal::UIntPtr(fooA->avec.data())%ALIGNMENT==0); + for (int i = 0; i < g_repeat * 100; ++i) { + MyStruct *foo0 = new MyStruct(); + VERIFY(internal::UIntPtr(foo0->avec.data()) % ALIGNMENT == 0); + MyClassA *fooA = new MyClassA(); + VERIFY(internal::UIntPtr(fooA->avec.data()) % ALIGNMENT == 0); delete foo0; delete fooA; } // dynamic allocation, array const int N = 10; - for (int i=0; iavec.data())%ALIGNMENT==0); - MyClassA *fooA = new MyClassA[N]; VERIFY(internal::UIntPtr(fooA->avec.data())%ALIGNMENT==0); + for (int i = 0; i < g_repeat * 100; ++i) { + MyStruct *foo0 = new MyStruct[N]; + VERIFY(internal::UIntPtr(foo0->avec.data()) % ALIGNMENT == 0); + MyClassA *fooA = new MyClassA[N]; + VERIFY(internal::UIntPtr(fooA->avec.data()) % ALIGNMENT == 0); delete[] foo0; delete[] fooA; } - #endif - +#endif } diff --git a/filmulator-gui/core/nlmeans/eigen/test/eigen2support.cpp b/filmulator-gui/core/nlmeans/eigen/test/eigen2support.cpp index ac6931a0..78815e00 100644 --- a/filmulator-gui/core/nlmeans/eigen/test/eigen2support.cpp +++ b/filmulator-gui/core/nlmeans/eigen/test/eigen2support.cpp @@ -11,23 +11,21 @@ #include "main.h" -template void eigen2support(const MatrixType& m) +template void eigen2support(const MatrixType &m) { typedef typename MatrixType::Scalar Scalar; Index rows = m.rows(); Index cols = m.cols(); - MatrixType m1 = MatrixType::Random(rows, cols), - m3(rows, cols); + MatrixType m1 = MatrixType::Random(rows, cols), m3(rows, cols); - Scalar s1 = internal::random(), - s2 = internal::random(); + Scalar s1 = internal::random(), s2 = internal::random(); // scalar addition VERIFY_IS_APPROX(m1.cwise() + s1, s1 + m1.cwise()); - VERIFY_IS_APPROX(m1.cwise() + s1, MatrixType::Constant(rows,cols,s1) + m1); - VERIFY_IS_APPROX((m1*Scalar(2)).cwise() - s2, (m1+m1) - MatrixType::Constant(rows,cols,s2) ); + VERIFY_IS_APPROX(m1.cwise() + s1, MatrixType::Constant(rows, cols, s1) + m1); + VERIFY_IS_APPROX((m1 * Scalar(2)).cwise() - s2, (m1 + m1) - MatrixType::Constant(rows, cols, s2)); m3 = m1; m3.cwise() += s2; VERIFY_IS_APPROX(m3, m1.cwise() + s2); @@ -35,13 +33,13 @@ template void eigen2support(const MatrixType& m) m3.cwise() -= s1; VERIFY_IS_APPROX(m3, m1.cwise() - s1); - VERIFY_IS_EQUAL((m1.corner(TopLeft,1,1)), (m1.block(0,0,1,1))); - VERIFY_IS_EQUAL((m1.template corner<1,1>(TopLeft)), (m1.template block<1,1>(0,0))); - VERIFY_IS_EQUAL((m1.col(0).start(1)), (m1.col(0).segment(0,1))); - VERIFY_IS_EQUAL((m1.col(0).template start<1>()), (m1.col(0).segment(0,1))); - VERIFY_IS_EQUAL((m1.col(0).end(1)), (m1.col(0).segment(rows-1,1))); - VERIFY_IS_EQUAL((m1.col(0).template end<1>()), (m1.col(0).segment(rows-1,1))); - + VERIFY_IS_EQUAL((m1.corner(TopLeft, 1, 1)), (m1.block(0, 0, 1, 1))); + VERIFY_IS_EQUAL((m1.template corner<1, 1>(TopLeft)), (m1.template block<1, 1>(0, 0))); + VERIFY_IS_EQUAL((m1.col(0).start(1)), (m1.col(0).segment(0, 1))); + VERIFY_IS_EQUAL((m1.col(0).template start<1>()), (m1.col(0).segment(0, 1))); + VERIFY_IS_EQUAL((m1.col(0).end(1)), (m1.col(0).segment(rows - 1, 1))); + VERIFY_IS_EQUAL((m1.col(0).template end<1>()), (m1.col(0).segment(rows - 1, 1))); + using std::cos; using numext::real; using numext::abs2; @@ -49,17 +47,17 @@ template void eigen2support(const MatrixType& m) VERIFY_IS_EQUAL(ei_real(s1), real(s1)); VERIFY_IS_EQUAL(ei_abs2(s1), abs2(s1)); - m1.minor(0,0); + m1.minor(0, 0); } void test_eigen2support() { - for(int i = 0; i < g_repeat; i++) { - CALL_SUBTEST_1( eigen2support(Matrix()) ); - CALL_SUBTEST_2( eigen2support(MatrixXd(1,1)) ); - CALL_SUBTEST_4( eigen2support(Matrix3f()) ); - CALL_SUBTEST_5( eigen2support(Matrix4d()) ); - CALL_SUBTEST_2( eigen2support(MatrixXf(200,200)) ); - CALL_SUBTEST_6( eigen2support(MatrixXcd(100,100)) ); + for (int i = 0; i < g_repeat; i++) { + CALL_SUBTEST_1(eigen2support(Matrix())); + CALL_SUBTEST_2(eigen2support(MatrixXd(1, 1))); + CALL_SUBTEST_4(eigen2support(Matrix3f())); + CALL_SUBTEST_5(eigen2support(Matrix4d())); + CALL_SUBTEST_2(eigen2support(MatrixXf(200, 200))); + CALL_SUBTEST_6(eigen2support(MatrixXcd(100, 100))); } } diff --git a/filmulator-gui/core/nlmeans/eigen/test/eigensolver_complex.cpp b/filmulator-gui/core/nlmeans/eigen/test/eigensolver_complex.cpp index 72694525..5ac238db 100644 --- a/filmulator-gui/core/nlmeans/eigen/test/eigensolver_complex.cpp +++ b/filmulator-gui/core/nlmeans/eigen/test/eigensolver_complex.cpp @@ -9,39 +9,34 @@ // with this file, You can obtain one at http://mozilla.org/MPL/2.0/. #include "main.h" -#include #include #include +#include -template bool find_pivot(typename MatrixType::Scalar tol, MatrixType &diffs, Index col=0) +template bool find_pivot(typename MatrixType::Scalar tol, MatrixType &diffs, Index col = 0) { bool match = diffs.diagonal().sum() <= tol; - if(match || col==diffs.cols()) - { + if (match || col == diffs.cols()) { return match; - } - else - { + } else { Index n = diffs.cols(); - std::vector > transpositions; - for(Index i=col; i> transpositions; + for (Index i = col; i < n; ++i) { Index best_index(0); - if(diffs.col(col).segment(col,n-i).minCoeff(&best_index) > tol) - break; - + if (diffs.col(col).segment(col, n - i).minCoeff(&best_index) > tol) break; + best_index += col; - + diffs.row(col).swap(diffs.row(best_index)); - if(find_pivot(tol,diffs,col+1)) return true; + if (find_pivot(tol, diffs, col + 1)) return true; diffs.row(col).swap(diffs.row(best_index)); - + // move current pivot to the end - diffs.row(n-(i-col)-1).swap(diffs.row(best_index)); - transpositions.push_back(std::pair(n-(i-col)-1,best_index)); + diffs.row(n - (i - col) - 1).swap(diffs.row(best_index)); + transpositions.push_back(std::pair(n - (i - col) - 1, best_index)); } // restore - for(Index k=transpositions.size()-1; k>=0; --k) + for (Index k = transpositions.size() - 1; k >= 0; --k) diffs.row(transpositions[k].first).swap(diffs.row(transpositions[k].second)); } return false; @@ -51,8 +46,7 @@ template bool find_pivot(typename MatrixType::Scalar tol, M * Initially, this method checked that the k-th power sums are equal for all k = 1, ..., vec1.rows(), * however this strategy is numerically inacurate because of numerical cancellation issues. */ -template -void verify_is_approx_upto_permutation(const VectorType& vec1, const VectorType& vec2) +template void verify_is_approx_upto_permutation(const VectorType &vec1, const VectorType &vec2) { typedef typename VectorType::Scalar Scalar; typedef typename NumTraits::Real RealScalar; @@ -60,16 +54,18 @@ void verify_is_approx_upto_permutation(const VectorType& vec1, const VectorType& VERIFY(vec1.cols() == 1); VERIFY(vec2.cols() == 1); VERIFY(vec1.rows() == vec2.rows()); - + Index n = vec1.rows(); - RealScalar tol = test_precision()*test_precision()*numext::maxi(vec1.squaredNorm(),vec2.squaredNorm()); - Matrix diffs = (vec1.rowwise().replicate(n) - vec2.rowwise().replicate(n).transpose()).cwiseAbs2(); - - VERIFY( find_pivot(tol, diffs) ); + RealScalar tol = + test_precision() * test_precision() * numext::maxi(vec1.squaredNorm(), vec2.squaredNorm()); + Matrix diffs = + (vec1.rowwise().replicate(n) - vec2.rowwise().replicate(n).transpose()).cwiseAbs2(); + + VERIFY(find_pivot(tol, diffs)); } -template void eigensolver(const MatrixType& m) +template void eigensolver(const MatrixType &m) { /* this test covers the following files: ComplexEigenSolver.h, and indirectly ComplexSchur.h @@ -80,8 +76,8 @@ template void eigensolver(const MatrixType& m) typedef typename MatrixType::Scalar Scalar; typedef typename NumTraits::Real RealScalar; - MatrixType a = MatrixType::Random(rows,cols); - MatrixType symmA = a.adjoint() * a; + MatrixType a = MatrixType::Random(rows, cols); + MatrixType symmA = a.adjoint() * a; ComplexEigenSolver ei0(symmA); VERIFY_IS_EQUAL(ei0.info(), Success); @@ -110,17 +106,16 @@ template void eigensolver(const MatrixType& m) VERIFY_IS_APPROX(ei1.eigenvalues(), eiNoEivecs.eigenvalues()); // Regression test for issue #66 - MatrixType z = MatrixType::Zero(rows,cols); + MatrixType z = MatrixType::Zero(rows, cols); ComplexEigenSolver eiz(z); VERIFY((eiz.eigenvalues().cwiseEqual(0)).all()); MatrixType id = MatrixType::Identity(rows, cols); VERIFY_IS_APPROX(id.operatorNorm(), RealScalar(1)); - if (rows > 1 && rows < 20) - { + if (rows > 1 && rows < 20) { // Test matrix with NaN - a(0,0) = std::numeric_limits::quiet_NaN(); + a(0, 0) = std::numeric_limits::quiet_NaN(); ComplexEigenSolver eiNaN(a); VERIFY_IS_EQUAL(eiNaN.info(), NoConvergence); } @@ -136,18 +131,18 @@ template void eigensolver(const MatrixType& m) a.setZero(); ComplexEigenSolver ei3(a); VERIFY_IS_EQUAL(ei3.info(), Success); - VERIFY_IS_MUCH_SMALLER_THAN(ei3.eigenvalues().norm(),RealScalar(1)); - VERIFY((ei3.eigenvectors().transpose()*ei3.eigenvectors().transpose()).eval().isIdentity()); + VERIFY_IS_MUCH_SMALLER_THAN(ei3.eigenvalues().norm(), RealScalar(1)); + VERIFY((ei3.eigenvectors().transpose() * ei3.eigenvectors().transpose()).eval().isIdentity()); } } -template void eigensolver_verify_assert(const MatrixType& m) +template void eigensolver_verify_assert(const MatrixType &m) { ComplexEigenSolver eig; VERIFY_RAISES_ASSERT(eig.eigenvectors()); VERIFY_RAISES_ASSERT(eig.eigenvalues()); - MatrixType a = MatrixType::Random(m.rows(),m.cols()); + MatrixType a = MatrixType::Random(m.rows(), m.cols()); eig.compute(a, false); VERIFY_RAISES_ASSERT(eig.eigenvectors()); } @@ -155,22 +150,22 @@ template void eigensolver_verify_assert(const MatrixType& m void test_eigensolver_complex() { int s = 0; - for(int i = 0; i < g_repeat; i++) { - CALL_SUBTEST_1( eigensolver(Matrix4cf()) ); - s = internal::random(1,EIGEN_TEST_MAX_SIZE/4); - CALL_SUBTEST_2( eigensolver(MatrixXcd(s,s)) ); - CALL_SUBTEST_3( eigensolver(Matrix, 1, 1>()) ); - CALL_SUBTEST_4( eigensolver(Matrix3f()) ); + for (int i = 0; i < g_repeat; i++) { + CALL_SUBTEST_1(eigensolver(Matrix4cf())); + s = internal::random(1, EIGEN_TEST_MAX_SIZE / 4); + CALL_SUBTEST_2(eigensolver(MatrixXcd(s, s))); + CALL_SUBTEST_3(eigensolver(Matrix, 1, 1>())); + CALL_SUBTEST_4(eigensolver(Matrix3f())); TEST_SET_BUT_UNUSED_VARIABLE(s) } - CALL_SUBTEST_1( eigensolver_verify_assert(Matrix4cf()) ); - s = internal::random(1,EIGEN_TEST_MAX_SIZE/4); - CALL_SUBTEST_2( eigensolver_verify_assert(MatrixXcd(s,s)) ); - CALL_SUBTEST_3( eigensolver_verify_assert(Matrix, 1, 1>()) ); - CALL_SUBTEST_4( eigensolver_verify_assert(Matrix3f()) ); + CALL_SUBTEST_1(eigensolver_verify_assert(Matrix4cf())); + s = internal::random(1, EIGEN_TEST_MAX_SIZE / 4); + CALL_SUBTEST_2(eigensolver_verify_assert(MatrixXcd(s, s))); + CALL_SUBTEST_3(eigensolver_verify_assert(Matrix, 1, 1>())); + CALL_SUBTEST_4(eigensolver_verify_assert(Matrix3f())); // Test problem size constructors CALL_SUBTEST_5(ComplexEigenSolver tmp(s)); - + TEST_SET_BUT_UNUSED_VARIABLE(s) } diff --git a/filmulator-gui/core/nlmeans/eigen/test/eigensolver_generalized_real.cpp b/filmulator-gui/core/nlmeans/eigen/test/eigensolver_generalized_real.cpp index 9dd44c89..a69e1414 100644 --- a/filmulator-gui/core/nlmeans/eigen/test/eigensolver_generalized_real.cpp +++ b/filmulator-gui/core/nlmeans/eigen/test/eigensolver_generalized_real.cpp @@ -9,11 +9,11 @@ #define EIGEN_RUNTIME_NO_MALLOC #include "main.h" -#include #include #include +#include -template void generalized_eigensolver_real(const MatrixType& m) +template void generalized_eigensolver_real(const MatrixType &m) { /* this test covers the following files: GeneralizedEigenSolver.h @@ -25,12 +25,12 @@ template void generalized_eigensolver_real(const MatrixType typedef std::complex ComplexScalar; typedef Matrix VectorType; - MatrixType a = MatrixType::Random(rows,cols); - MatrixType b = MatrixType::Random(rows,cols); - MatrixType a1 = MatrixType::Random(rows,cols); - MatrixType b1 = MatrixType::Random(rows,cols); - MatrixType spdA = a.adjoint() * a + a1.adjoint() * a1; - MatrixType spdB = b.adjoint() * b + b1.adjoint() * b1; + MatrixType a = MatrixType::Random(rows, cols); + MatrixType b = MatrixType::Random(rows, cols); + MatrixType a1 = MatrixType::Random(rows, cols); + MatrixType b1 = MatrixType::Random(rows, cols); + MatrixType spdA = a.adjoint() * a + a1.adjoint() * a1; + MatrixType spdB = b.adjoint() * b + b1.adjoint() * b1; // lets compare to GeneralizedSelfAdjointEigenSolver { @@ -40,41 +40,41 @@ template void generalized_eigensolver_real(const MatrixType VERIFY_IS_EQUAL(eig.eigenvalues().imag().cwiseAbs().maxCoeff(), 0); VectorType realEigenvalues = eig.eigenvalues().real(); - std::sort(realEigenvalues.data(), realEigenvalues.data()+realEigenvalues.size()); + std::sort(realEigenvalues.data(), realEigenvalues.data() + realEigenvalues.size()); VERIFY_IS_APPROX(realEigenvalues, symmEig.eigenvalues()); // check eigenvectors typename GeneralizedEigenSolver::EigenvectorsType D = eig.eigenvalues().asDiagonal(); typename GeneralizedEigenSolver::EigenvectorsType V = eig.eigenvectors(); - VERIFY_IS_APPROX(spdA*V, spdB*V*D); + VERIFY_IS_APPROX(spdA * V, spdB * V * D); } // non symmetric case: { GeneralizedEigenSolver eig(rows); - // TODO enable full-prealocation of required memory, this probably requires an in-place mode for HessenbergDecomposition - //Eigen::internal::set_is_malloc_allowed(false); - eig.compute(a,b); - //Eigen::internal::set_is_malloc_allowed(true); - for(Index k=0; k tmp = (eig.betas()(k)*a).template cast() - eig.alphas()(k)*b; - if(tmp.size()>1 && tmp.norm()>(std::numeric_limits::min)()) - tmp /= tmp.norm(); - VERIFY_IS_MUCH_SMALLER_THAN( std::abs(tmp.determinant()), Scalar(1) ); + // TODO enable full-prealocation of required memory, this probably requires an in-place mode for + // HessenbergDecomposition + // Eigen::internal::set_is_malloc_allowed(false); + eig.compute(a, b); + // Eigen::internal::set_is_malloc_allowed(true); + for (Index k = 0; k < cols; ++k) { + Matrix tmp = + (eig.betas()(k) * a).template cast() - eig.alphas()(k) * b; + if (tmp.size() > 1 && tmp.norm() > (std::numeric_limits::min)()) tmp /= tmp.norm(); + VERIFY_IS_MUCH_SMALLER_THAN(std::abs(tmp.determinant()), Scalar(1)); } // check eigenvectors typename GeneralizedEigenSolver::EigenvectorsType D = eig.eigenvalues().asDiagonal(); typename GeneralizedEigenSolver::EigenvectorsType V = eig.eigenvectors(); - VERIFY_IS_APPROX(a*V, b*V*D); + VERIFY_IS_APPROX(a * V, b * V * D); } // regression test for bug 1098 { - GeneralizedSelfAdjointEigenSolver eig1(a.adjoint() * a,b.adjoint() * b); - eig1.compute(a.adjoint() * a,b.adjoint() * b); - GeneralizedEigenSolver eig2(a.adjoint() * a,b.adjoint() * b); - eig2.compute(a.adjoint() * a,b.adjoint() * b); + GeneralizedSelfAdjointEigenSolver eig1(a.adjoint() * a, b.adjoint() * b); + eig1.compute(a.adjoint() * a, b.adjoint() * b); + GeneralizedEigenSolver eig2(a.adjoint() * a, b.adjoint() * b); + eig2.compute(a.adjoint() * a, b.adjoint() * b); } // check without eigenvectors @@ -87,17 +87,17 @@ template void generalized_eigensolver_real(const MatrixType void test_eigensolver_generalized_real() { - for(int i = 0; i < g_repeat; i++) { + for (int i = 0; i < g_repeat; i++) { int s = 0; - CALL_SUBTEST_1( generalized_eigensolver_real(Matrix4f()) ); - s = internal::random(1,EIGEN_TEST_MAX_SIZE/4); - CALL_SUBTEST_2( generalized_eigensolver_real(MatrixXd(s,s)) ); + CALL_SUBTEST_1(generalized_eigensolver_real(Matrix4f())); + s = internal::random(1, EIGEN_TEST_MAX_SIZE / 4); + CALL_SUBTEST_2(generalized_eigensolver_real(MatrixXd(s, s))); // some trivial but implementation-wise special cases - CALL_SUBTEST_2( generalized_eigensolver_real(MatrixXd(1,1)) ); - CALL_SUBTEST_2( generalized_eigensolver_real(MatrixXd(2,2)) ); - CALL_SUBTEST_3( generalized_eigensolver_real(Matrix()) ); - CALL_SUBTEST_4( generalized_eigensolver_real(Matrix2d()) ); + CALL_SUBTEST_2(generalized_eigensolver_real(MatrixXd(1, 1))); + CALL_SUBTEST_2(generalized_eigensolver_real(MatrixXd(2, 2))); + CALL_SUBTEST_3(generalized_eigensolver_real(Matrix())); + CALL_SUBTEST_4(generalized_eigensolver_real(Matrix2d())); TEST_SET_BUT_UNUSED_VARIABLE(s) } } diff --git a/filmulator-gui/core/nlmeans/eigen/test/eigensolver_generic.cpp b/filmulator-gui/core/nlmeans/eigen/test/eigensolver_generic.cpp index 07bf65e0..effb093f 100644 --- a/filmulator-gui/core/nlmeans/eigen/test/eigensolver_generic.cpp +++ b/filmulator-gui/core/nlmeans/eigen/test/eigensolver_generic.cpp @@ -9,10 +9,10 @@ // with this file, You can obtain one at http://mozilla.org/MPL/2.0/. #include "main.h" -#include #include +#include -template void eigensolver(const MatrixType& m) +template void eigensolver(const MatrixType &m) { /* this test covers the following files: EigenSolver.h @@ -25,9 +25,9 @@ template void eigensolver(const MatrixType& m) typedef Matrix RealVectorType; typedef typename std::complex::Real> Complex; - MatrixType a = MatrixType::Random(rows,cols); - MatrixType a1 = MatrixType::Random(rows,cols); - MatrixType symmA = a.adjoint() * a + a1.adjoint() * a1; + MatrixType a = MatrixType::Random(rows, cols); + MatrixType a1 = MatrixType::Random(rows, cols); + MatrixType symmA = a.adjoint() * a + a1.adjoint() * a1; EigenSolver ei0(symmA); VERIFY_IS_EQUAL(ei0.info(), Success); @@ -38,8 +38,8 @@ template void eigensolver(const MatrixType& m) EigenSolver ei1(a); VERIFY_IS_EQUAL(ei1.info(), Success); VERIFY_IS_APPROX(a * ei1.pseudoEigenvectors(), ei1.pseudoEigenvectors() * ei1.pseudoEigenvalueMatrix()); - VERIFY_IS_APPROX(a.template cast() * ei1.eigenvectors(), - ei1.eigenvectors() * ei1.eigenvalues().asDiagonal()); + VERIFY_IS_APPROX( + a.template cast() * ei1.eigenvectors(), ei1.eigenvectors() * ei1.eigenvalues().asDiagonal()); VERIFY_IS_APPROX(ei1.eigenvectors().colwise().norm(), RealVectorType::Ones(rows).transpose()); VERIFY_IS_APPROX(a.eigenvalues(), ei1.eigenvalues()); @@ -62,10 +62,9 @@ template void eigensolver(const MatrixType& m) MatrixType id = MatrixType::Identity(rows, cols); VERIFY_IS_APPROX(id.operatorNorm(), RealScalar(1)); - if (rows > 2 && rows < 20) - { + if (rows > 2 && rows < 20) { // Test matrix with NaN - a(0,0) = std::numeric_limits::quiet_NaN(); + a(0, 0) = std::numeric_limits::quiet_NaN(); EigenSolver eiNaN(a); VERIFY_IS_EQUAL(eiNaN.info(), NoConvergence); } @@ -81,12 +80,12 @@ template void eigensolver(const MatrixType& m) a.setZero(); EigenSolver ei3(a); VERIFY_IS_EQUAL(ei3.info(), Success); - VERIFY_IS_MUCH_SMALLER_THAN(ei3.eigenvalues().norm(),RealScalar(1)); - VERIFY((ei3.eigenvectors().transpose()*ei3.eigenvectors().transpose()).eval().isIdentity()); + VERIFY_IS_MUCH_SMALLER_THAN(ei3.eigenvalues().norm(), RealScalar(1)); + VERIFY((ei3.eigenvectors().transpose() * ei3.eigenvectors().transpose()).eval().isIdentity()); } } -template void eigensolver_verify_assert(const MatrixType& m) +template void eigensolver_verify_assert(const MatrixType &m) { EigenSolver eig; VERIFY_RAISES_ASSERT(eig.eigenvectors()); @@ -94,7 +93,7 @@ template void eigensolver_verify_assert(const MatrixType& m VERIFY_RAISES_ASSERT(eig.pseudoEigenvalueMatrix()); VERIFY_RAISES_ASSERT(eig.eigenvalues()); - MatrixType a = MatrixType::Random(m.rows(),m.cols()); + MatrixType a = MatrixType::Random(m.rows(), m.cols()); eig.compute(a, false); VERIFY_RAISES_ASSERT(eig.eigenvectors()); VERIFY_RAISES_ASSERT(eig.pseudoEigenvectors()); @@ -103,63 +102,59 @@ template void eigensolver_verify_assert(const MatrixType& m void test_eigensolver_generic() { int s = 0; - for(int i = 0; i < g_repeat; i++) { - CALL_SUBTEST_1( eigensolver(Matrix4f()) ); - s = internal::random(1,EIGEN_TEST_MAX_SIZE/4); - CALL_SUBTEST_2( eigensolver(MatrixXd(s,s)) ); + for (int i = 0; i < g_repeat; i++) { + CALL_SUBTEST_1(eigensolver(Matrix4f())); + s = internal::random(1, EIGEN_TEST_MAX_SIZE / 4); + CALL_SUBTEST_2(eigensolver(MatrixXd(s, s))); TEST_SET_BUT_UNUSED_VARIABLE(s) // some trivial but implementation-wise tricky cases - CALL_SUBTEST_2( eigensolver(MatrixXd(1,1)) ); - CALL_SUBTEST_2( eigensolver(MatrixXd(2,2)) ); - CALL_SUBTEST_3( eigensolver(Matrix()) ); - CALL_SUBTEST_4( eigensolver(Matrix2d()) ); + CALL_SUBTEST_2(eigensolver(MatrixXd(1, 1))); + CALL_SUBTEST_2(eigensolver(MatrixXd(2, 2))); + CALL_SUBTEST_3(eigensolver(Matrix())); + CALL_SUBTEST_4(eigensolver(Matrix2d())); } - CALL_SUBTEST_1( eigensolver_verify_assert(Matrix4f()) ); - s = internal::random(1,EIGEN_TEST_MAX_SIZE/4); - CALL_SUBTEST_2( eigensolver_verify_assert(MatrixXd(s,s)) ); - CALL_SUBTEST_3( eigensolver_verify_assert(Matrix()) ); - CALL_SUBTEST_4( eigensolver_verify_assert(Matrix2d()) ); + CALL_SUBTEST_1(eigensolver_verify_assert(Matrix4f())); + s = internal::random(1, EIGEN_TEST_MAX_SIZE / 4); + CALL_SUBTEST_2(eigensolver_verify_assert(MatrixXd(s, s))); + CALL_SUBTEST_3(eigensolver_verify_assert(Matrix())); + CALL_SUBTEST_4(eigensolver_verify_assert(Matrix2d())); // Test problem size constructors CALL_SUBTEST_5(EigenSolver tmp(s)); // regression test for bug 410 - CALL_SUBTEST_2( - { - MatrixXd A(1,1); - A(0,0) = std::sqrt(-1.); // is Not-a-Number - Eigen::EigenSolver solver(A); - VERIFY_IS_EQUAL(solver.info(), NumericalIssue); - } - ); - + CALL_SUBTEST_2({ + MatrixXd A(1, 1); + A(0, 0) = std::sqrt(-1.);// is Not-a-Number + Eigen::EigenSolver solver(A); + VERIFY_IS_EQUAL(solver.info(), NumericalIssue); + }); + #ifdef EIGEN_TEST_PART_2 { // regression test for bug 793 - MatrixXd a(3,3); - a << 0, 0, 1, - 1, 1, 1, - 1, 1e+200, 1; + MatrixXd a(3, 3); + a << 0, 0, 1, 1, 1, 1, 1, 1e+200, 1; Eigen::EigenSolver eig(a); - double scale = 1e-200; // scale to avoid overflow during the comparisons - VERIFY_IS_APPROX(a * eig.pseudoEigenvectors()*scale, eig.pseudoEigenvectors() * eig.pseudoEigenvalueMatrix()*scale); - VERIFY_IS_APPROX(a * eig.eigenvectors()*scale, eig.eigenvectors() * eig.eigenvalues().asDiagonal()*scale); + double scale = 1e-200;// scale to avoid overflow during the comparisons + VERIFY_IS_APPROX( + a * eig.pseudoEigenvectors() * scale, eig.pseudoEigenvectors() * eig.pseudoEigenvalueMatrix() * scale); + VERIFY_IS_APPROX(a * eig.eigenvectors() * scale, eig.eigenvectors() * eig.eigenvalues().asDiagonal() * scale); } { // check a case where all eigenvalues are null. - MatrixXd a(2,2); - a << 1, 1, - -1, -1; + MatrixXd a(2, 2); + a << 1, 1, -1, -1; Eigen::EigenSolver eig(a); VERIFY_IS_APPROX(eig.pseudoEigenvectors().squaredNorm(), 2.); - VERIFY_IS_APPROX((a * eig.pseudoEigenvectors()).norm()+1., 1.); - VERIFY_IS_APPROX((eig.pseudoEigenvectors() * eig.pseudoEigenvalueMatrix()).norm()+1., 1.); - VERIFY_IS_APPROX((a * eig.eigenvectors()).norm()+1., 1.); - VERIFY_IS_APPROX((eig.eigenvectors() * eig.eigenvalues().asDiagonal()).norm()+1., 1.); + VERIFY_IS_APPROX((a * eig.pseudoEigenvectors()).norm() + 1., 1.); + VERIFY_IS_APPROX((eig.pseudoEigenvectors() * eig.pseudoEigenvalueMatrix()).norm() + 1., 1.); + VERIFY_IS_APPROX((a * eig.eigenvectors()).norm() + 1., 1.); + VERIFY_IS_APPROX((eig.eigenvectors() * eig.eigenvalues().asDiagonal()).norm() + 1., 1.); } #endif - + TEST_SET_BUT_UNUSED_VARIABLE(s) } diff --git a/filmulator-gui/core/nlmeans/eigen/test/eigensolver_selfadjoint.cpp b/filmulator-gui/core/nlmeans/eigen/test/eigensolver_selfadjoint.cpp index 0e39b536..bb2f97fa 100644 --- a/filmulator-gui/core/nlmeans/eigen/test/eigensolver_selfadjoint.cpp +++ b/filmulator-gui/core/nlmeans/eigen/test/eigensolver_selfadjoint.cpp @@ -10,63 +10,58 @@ #include "main.h" #include "svd_fill.h" -#include #include #include +#include -template void selfadjointeigensolver_essential_check(const MatrixType& m) +template void selfadjointeigensolver_essential_check(const MatrixType &m) { typedef typename MatrixType::Scalar Scalar; typedef typename NumTraits::Real RealScalar; - RealScalar eival_eps = numext::mini(test_precision(), NumTraits::dummy_precision()*20000); - + RealScalar eival_eps = + numext::mini(test_precision(), NumTraits::dummy_precision() * 20000); + SelfAdjointEigenSolver eiSymm(m); VERIFY_IS_EQUAL(eiSymm.info(), Success); RealScalar scaling = m.cwiseAbs().maxCoeff(); - if(scaling<(std::numeric_limits::min)()) - { + if (scaling < (std::numeric_limits::min)()) { VERIFY(eiSymm.eigenvalues().cwiseAbs().maxCoeff() <= (std::numeric_limits::min)()); - } - else - { - VERIFY_IS_APPROX((m.template selfadjointView() * eiSymm.eigenvectors())/scaling, - (eiSymm.eigenvectors() * eiSymm.eigenvalues().asDiagonal())/scaling); + } else { + VERIFY_IS_APPROX((m.template selfadjointView() * eiSymm.eigenvectors()) / scaling, + (eiSymm.eigenvectors() * eiSymm.eigenvalues().asDiagonal()) / scaling); } VERIFY_IS_APPROX(m.template selfadjointView().eigenvalues(), eiSymm.eigenvalues()); VERIFY_IS_UNITARY(eiSymm.eigenvectors()); - if(m.cols()<=4) - { + if (m.cols() <= 4) { SelfAdjointEigenSolver eiDirect; - eiDirect.computeDirect(m); + eiDirect.computeDirect(m); VERIFY_IS_EQUAL(eiDirect.info(), Success); - if(! eiSymm.eigenvalues().isApprox(eiDirect.eigenvalues(), eival_eps) ) - { + if (!eiSymm.eigenvalues().isApprox(eiDirect.eigenvalues(), eival_eps)) { std::cerr << "reference eigenvalues: " << eiSymm.eigenvalues().transpose() << "\n" << "obtained eigenvalues: " << eiDirect.eigenvalues().transpose() << "\n" - << "diff: " << (eiSymm.eigenvalues()-eiDirect.eigenvalues()).transpose() << "\n" - << "error (eps): " << (eiSymm.eigenvalues()-eiDirect.eigenvalues()).norm() / eiSymm.eigenvalues().norm() << " (" << eival_eps << ")\n"; + << "diff: " << (eiSymm.eigenvalues() - eiDirect.eigenvalues()).transpose() << "\n" + << "error (eps): " + << (eiSymm.eigenvalues() - eiDirect.eigenvalues()).norm() / eiSymm.eigenvalues().norm() << " (" + << eival_eps << ")\n"; } - if(scaling<(std::numeric_limits::min)()) - { + if (scaling < (std::numeric_limits::min)()) { VERIFY(eiDirect.eigenvalues().cwiseAbs().maxCoeff() <= (std::numeric_limits::min)()); - } - else - { - VERIFY_IS_APPROX(eiSymm.eigenvalues()/scaling, eiDirect.eigenvalues()/scaling); - VERIFY_IS_APPROX((m.template selfadjointView() * eiDirect.eigenvectors())/scaling, - (eiDirect.eigenvectors() * eiDirect.eigenvalues().asDiagonal())/scaling); - VERIFY_IS_APPROX(m.template selfadjointView().eigenvalues()/scaling, eiDirect.eigenvalues()/scaling); + } else { + VERIFY_IS_APPROX(eiSymm.eigenvalues() / scaling, eiDirect.eigenvalues() / scaling); + VERIFY_IS_APPROX((m.template selfadjointView() * eiDirect.eigenvectors()) / scaling, + (eiDirect.eigenvectors() * eiDirect.eigenvalues().asDiagonal()) / scaling); + VERIFY_IS_APPROX(m.template selfadjointView().eigenvalues() / scaling, eiDirect.eigenvalues() / scaling); } VERIFY_IS_UNITARY(eiDirect.eigenvectors()); } } -template void selfadjointeigensolver(const MatrixType& m) +template void selfadjointeigensolver(const MatrixType &m) { /* this test covers the following files: EigenSolver.h, SelfAdjointEigenSolver.h (and indirectly: Tridiagonalization.h) @@ -77,24 +72,24 @@ template void selfadjointeigensolver(const MatrixType& m) typedef typename MatrixType::Scalar Scalar; typedef typename NumTraits::Real RealScalar; - RealScalar largerEps = 10*test_precision(); + RealScalar largerEps = 10 * test_precision(); - MatrixType a = MatrixType::Random(rows,cols); - MatrixType a1 = MatrixType::Random(rows,cols); - MatrixType symmA = a.adjoint() * a + a1.adjoint() * a1; + MatrixType a = MatrixType::Random(rows, cols); + MatrixType a1 = MatrixType::Random(rows, cols); + MatrixType symmA = a.adjoint() * a + a1.adjoint() * a1; MatrixType symmC = symmA; - - svd_fill_random(symmA,Symmetric); + + svd_fill_random(symmA, Symmetric); symmA.template triangularView().setZero(); symmC.template triangularView().setZero(); - MatrixType b = MatrixType::Random(rows,cols); - MatrixType b1 = MatrixType::Random(rows,cols); + MatrixType b = MatrixType::Random(rows, cols); + MatrixType b1 = MatrixType::Random(rows, cols); MatrixType symmB = b.adjoint() * b + b1.adjoint() * b1; symmB.template triangularView().setZero(); - - CALL_SUBTEST( selfadjointeigensolver_essential_check(symmA) ); + + CALL_SUBTEST(selfadjointeigensolver_essential_check(symmA)); SelfAdjointEigenSolver eiSymm(symmA); // generalized eigen pb @@ -103,30 +98,34 @@ template void selfadjointeigensolver(const MatrixType& m) SelfAdjointEigenSolver eiSymmNoEivecs(symmA, false); VERIFY_IS_EQUAL(eiSymmNoEivecs.info(), Success); VERIFY_IS_APPROX(eiSymm.eigenvalues(), eiSymmNoEivecs.eigenvalues()); - + // generalized eigen problem Ax = lBx - eiSymmGen.compute(symmC, symmB,Ax_lBx); + eiSymmGen.compute(symmC, symmB, Ax_lBx); VERIFY_IS_EQUAL(eiSymmGen.info(), Success); - VERIFY((symmC.template selfadjointView() * eiSymmGen.eigenvectors()).isApprox( - symmB.template selfadjointView() * (eiSymmGen.eigenvectors() * eiSymmGen.eigenvalues().asDiagonal()), largerEps)); + VERIFY((symmC.template selfadjointView() * eiSymmGen.eigenvectors()) + .isApprox( + symmB.template selfadjointView() * (eiSymmGen.eigenvectors() * eiSymmGen.eigenvalues().asDiagonal()), + largerEps)); // generalized eigen problem BAx = lx - eiSymmGen.compute(symmC, symmB,BAx_lx); + eiSymmGen.compute(symmC, symmB, BAx_lx); VERIFY_IS_EQUAL(eiSymmGen.info(), Success); - VERIFY((symmB.template selfadjointView() * (symmC.template selfadjointView() * eiSymmGen.eigenvectors())).isApprox( - (eiSymmGen.eigenvectors() * eiSymmGen.eigenvalues().asDiagonal()), largerEps)); + VERIFY( + (symmB.template selfadjointView() * (symmC.template selfadjointView() * eiSymmGen.eigenvectors())) + .isApprox((eiSymmGen.eigenvectors() * eiSymmGen.eigenvalues().asDiagonal()), largerEps)); // generalized eigen problem ABx = lx - eiSymmGen.compute(symmC, symmB,ABx_lx); + eiSymmGen.compute(symmC, symmB, ABx_lx); VERIFY_IS_EQUAL(eiSymmGen.info(), Success); - VERIFY((symmC.template selfadjointView() * (symmB.template selfadjointView() * eiSymmGen.eigenvectors())).isApprox( - (eiSymmGen.eigenvectors() * eiSymmGen.eigenvalues().asDiagonal()), largerEps)); + VERIFY( + (symmC.template selfadjointView() * (symmB.template selfadjointView() * eiSymmGen.eigenvectors())) + .isApprox((eiSymmGen.eigenvectors() * eiSymmGen.eigenvalues().asDiagonal()), largerEps)); eiSymm.compute(symmC); MatrixType sqrtSymmA = eiSymm.operatorSqrt(); - VERIFY_IS_APPROX(MatrixType(symmC.template selfadjointView()), sqrtSymmA*sqrtSymmA); - VERIFY_IS_APPROX(sqrtSymmA, symmC.template selfadjointView()*eiSymm.operatorInverseSqrt()); + VERIFY_IS_APPROX(MatrixType(symmC.template selfadjointView()), sqrtSymmA * sqrtSymmA); + VERIFY_IS_APPROX(sqrtSymmA, symmC.template selfadjointView() * eiSymm.operatorInverseSqrt()); MatrixType id = MatrixType::Identity(rows, cols); VERIFY_IS_APPROX(id.template selfadjointView().operatorNorm(), RealScalar(1)); @@ -147,29 +146,32 @@ template void selfadjointeigensolver(const MatrixType& m) Tridiagonalization tridiag(symmC); VERIFY_IS_APPROX(tridiag.diagonal(), tridiag.matrixT().diagonal()); VERIFY_IS_APPROX(tridiag.subDiagonal(), tridiag.matrixT().template diagonal<-1>()); - Matrix T = tridiag.matrixT(); - if(rows>1 && cols>1) { + Matrix T = tridiag.matrixT(); + if (rows > 1 && cols > 1) { // FIXME check that upper and lower part are 0: - //VERIFY(T.topRightCorner(rows-2, cols-2).template triangularView().isZero()); + // VERIFY(T.topRightCorner(rows-2, cols-2).template triangularView().isZero()); } VERIFY_IS_APPROX(tridiag.diagonal(), T.diagonal()); VERIFY_IS_APPROX(tridiag.subDiagonal(), T.template diagonal<1>()); - VERIFY_IS_APPROX(MatrixType(symmC.template selfadjointView()), tridiag.matrixQ() * tridiag.matrixT().eval() * MatrixType(tridiag.matrixQ()).adjoint()); - VERIFY_IS_APPROX(MatrixType(symmC.template selfadjointView()), tridiag.matrixQ() * tridiag.matrixT() * tridiag.matrixQ().adjoint()); - + VERIFY_IS_APPROX(MatrixType(symmC.template selfadjointView()), + tridiag.matrixQ() * tridiag.matrixT().eval() * MatrixType(tridiag.matrixQ()).adjoint()); + VERIFY_IS_APPROX(MatrixType(symmC.template selfadjointView()), + tridiag.matrixQ() * tridiag.matrixT() * tridiag.matrixQ().adjoint()); + // Test computation of eigenvalues from tridiagonal matrix - if(rows > 1) - { + if (rows > 1) { SelfAdjointEigenSolver eiSymmTridiag; - eiSymmTridiag.computeFromTridiagonal(tridiag.matrixT().diagonal(), tridiag.matrixT().diagonal(-1), ComputeEigenvectors); + eiSymmTridiag.computeFromTridiagonal( + tridiag.matrixT().diagonal(), tridiag.matrixT().diagonal(-1), ComputeEigenvectors); VERIFY_IS_APPROX(eiSymm.eigenvalues(), eiSymmTridiag.eigenvalues()); - VERIFY_IS_APPROX(tridiag.matrixT(), eiSymmTridiag.eigenvectors().real() * eiSymmTridiag.eigenvalues().asDiagonal() * eiSymmTridiag.eigenvectors().real().transpose()); + VERIFY_IS_APPROX(tridiag.matrixT(), + eiSymmTridiag.eigenvectors().real() * eiSymmTridiag.eigenvalues().asDiagonal() + * eiSymmTridiag.eigenvectors().real().transpose()); } - if (rows > 1 && rows < 20) - { + if (rows > 1 && rows < 20) { // Test matrix with NaN - symmC(0,0) = std::numeric_limits::quiet_NaN(); + symmC(0, 0) = std::numeric_limits::quiet_NaN(); SelfAdjointEigenSolver eiSymmNaN(symmC); VERIFY_IS_EQUAL(eiSymmNaN.info(), NoConvergence); } @@ -185,89 +187,80 @@ template void selfadjointeigensolver(const MatrixType& m) a.setZero(); SelfAdjointEigenSolver ei3(a); VERIFY_IS_EQUAL(ei3.info(), Success); - VERIFY_IS_MUCH_SMALLER_THAN(ei3.eigenvalues().norm(),RealScalar(1)); - VERIFY((ei3.eigenvectors().transpose()*ei3.eigenvectors().transpose()).eval().isIdentity()); + VERIFY_IS_MUCH_SMALLER_THAN(ei3.eigenvalues().norm(), RealScalar(1)); + VERIFY((ei3.eigenvectors().transpose() * ei3.eigenvectors().transpose()).eval().isIdentity()); } } -template -void bug_854() +template void bug_854() { Matrix3d m; - m << 850.961, 51.966, 0, - 51.966, 254.841, 0, - 0, 0, 0; + m << 850.961, 51.966, 0, 51.966, 254.841, 0, 0, 0, 0; selfadjointeigensolver_essential_check(m); } -template -void bug_1014() +template void bug_1014() { Matrix3d m; - m << 0.11111111111111114658, 0, 0, - 0, 0.11111111111111109107, 0, - 0, 0, 0.11111111111111107719; + m << 0.11111111111111114658, 0, 0, 0, 0.11111111111111109107, 0, 0, 0, 0.11111111111111107719; selfadjointeigensolver_essential_check(m); } -template -void bug_1225() +template void bug_1225() { Matrix3d m1, m2; m1.setRandom(); - m1 = m1*m1.transpose(); + m1 = m1 * m1.transpose(); m2 = m1.triangularView(); SelfAdjointEigenSolver eig1(m1); SelfAdjointEigenSolver eig2(m2.selfadjointView()); VERIFY_IS_APPROX(eig1.eigenvalues(), eig2.eigenvalues()); } -template -void bug_1204() +template void bug_1204() { - SparseMatrix A(2,2); + SparseMatrix A(2, 2); A.setIdentity(); - SelfAdjointEigenSolver > eig(A); + SelfAdjointEigenSolver> eig(A); } void test_eigensolver_selfadjoint() { int s = 0; - for(int i = 0; i < g_repeat; i++) { + for (int i = 0; i < g_repeat; i++) { // trivial test for 1x1 matrices: - CALL_SUBTEST_1( selfadjointeigensolver(Matrix())); - CALL_SUBTEST_1( selfadjointeigensolver(Matrix())); + CALL_SUBTEST_1(selfadjointeigensolver(Matrix())); + CALL_SUBTEST_1(selfadjointeigensolver(Matrix())); // very important to test 3x3 and 2x2 matrices since we provide special paths for them - CALL_SUBTEST_12( selfadjointeigensolver(Matrix2f()) ); - CALL_SUBTEST_12( selfadjointeigensolver(Matrix2d()) ); - CALL_SUBTEST_13( selfadjointeigensolver(Matrix3f()) ); - CALL_SUBTEST_13( selfadjointeigensolver(Matrix3d()) ); - CALL_SUBTEST_2( selfadjointeigensolver(Matrix4d()) ); - - s = internal::random(1,EIGEN_TEST_MAX_SIZE/4); - CALL_SUBTEST_3( selfadjointeigensolver(MatrixXf(s,s)) ); - CALL_SUBTEST_4( selfadjointeigensolver(MatrixXd(s,s)) ); - CALL_SUBTEST_5( selfadjointeigensolver(MatrixXcd(s,s)) ); - CALL_SUBTEST_9( selfadjointeigensolver(Matrix,Dynamic,Dynamic,RowMajor>(s,s)) ); + CALL_SUBTEST_12(selfadjointeigensolver(Matrix2f())); + CALL_SUBTEST_12(selfadjointeigensolver(Matrix2d())); + CALL_SUBTEST_13(selfadjointeigensolver(Matrix3f())); + CALL_SUBTEST_13(selfadjointeigensolver(Matrix3d())); + CALL_SUBTEST_2(selfadjointeigensolver(Matrix4d())); + + s = internal::random(1, EIGEN_TEST_MAX_SIZE / 4); + CALL_SUBTEST_3(selfadjointeigensolver(MatrixXf(s, s))); + CALL_SUBTEST_4(selfadjointeigensolver(MatrixXd(s, s))); + CALL_SUBTEST_5(selfadjointeigensolver(MatrixXcd(s, s))); + CALL_SUBTEST_9(selfadjointeigensolver(Matrix, Dynamic, Dynamic, RowMajor>(s, s))); TEST_SET_BUT_UNUSED_VARIABLE(s) // some trivial but implementation-wise tricky cases - CALL_SUBTEST_4( selfadjointeigensolver(MatrixXd(1,1)) ); - CALL_SUBTEST_4( selfadjointeigensolver(MatrixXd(2,2)) ); - CALL_SUBTEST_6( selfadjointeigensolver(Matrix()) ); - CALL_SUBTEST_7( selfadjointeigensolver(Matrix()) ); + CALL_SUBTEST_4(selfadjointeigensolver(MatrixXd(1, 1))); + CALL_SUBTEST_4(selfadjointeigensolver(MatrixXd(2, 2))); + CALL_SUBTEST_6(selfadjointeigensolver(Matrix())); + CALL_SUBTEST_7(selfadjointeigensolver(Matrix())); } - - CALL_SUBTEST_13( bug_854<0>() ); - CALL_SUBTEST_13( bug_1014<0>() ); - CALL_SUBTEST_13( bug_1204<0>() ); - CALL_SUBTEST_13( bug_1225<0>() ); + + CALL_SUBTEST_13(bug_854<0>()); + CALL_SUBTEST_13(bug_1014<0>()); + CALL_SUBTEST_13(bug_1204<0>()); + CALL_SUBTEST_13(bug_1225<0>()); // Test problem size constructors - s = internal::random(1,EIGEN_TEST_MAX_SIZE/4); + s = internal::random(1, EIGEN_TEST_MAX_SIZE / 4); CALL_SUBTEST_8(SelfAdjointEigenSolver tmp1(s)); CALL_SUBTEST_8(Tridiagonalization tmp2(s)); - + TEST_SET_BUT_UNUSED_VARIABLE(s) } - diff --git a/filmulator-gui/core/nlmeans/eigen/test/evaluators.cpp b/filmulator-gui/core/nlmeans/eigen/test/evaluators.cpp index aed5a05a..d2d1be58 100644 --- a/filmulator-gui/core/nlmeans/eigen/test/evaluators.cpp +++ b/filmulator-gui/core/nlmeans/eigen/test/evaluators.cpp @@ -3,103 +3,106 @@ namespace Eigen { - template - const Product - prod(const Lhs& lhs, const Rhs& rhs) - { - return Product(lhs,rhs); - } +template const Product prod(const Lhs &lhs, const Rhs &rhs) +{ + return Product(lhs, rhs); +} - template - const Product - lazyprod(const Lhs& lhs, const Rhs& rhs) - { - return Product(lhs,rhs); - } - - template - EIGEN_STRONG_INLINE - DstXprType& copy_using_evaluator(const EigenBase &dst, const SrcXprType &src) - { - call_assignment(dst.const_cast_derived(), src.derived(), internal::assign_op()); - return dst.const_cast_derived(); - } - - template class StorageBase, typename SrcXprType> - EIGEN_STRONG_INLINE - const DstXprType& copy_using_evaluator(const NoAlias& dst, const SrcXprType &src) - { - call_assignment(dst, src.derived(), internal::assign_op()); - return dst.expression(); - } - - template - EIGEN_STRONG_INLINE - DstXprType& copy_using_evaluator(const PlainObjectBase &dst, const SrcXprType &src) - { - #ifdef EIGEN_NO_AUTOMATIC_RESIZING - eigen_assert((dst.size()==0 || (IsVectorAtCompileTime ? (dst.size() == src.size()) - : (dst.rows() == src.rows() && dst.cols() == src.cols()))) - && "Size mismatch. Automatic resizing is disabled because EIGEN_NO_AUTOMATIC_RESIZING is defined"); - #else - dst.const_cast_derived().resizeLike(src.derived()); - #endif - - call_assignment(dst.const_cast_derived(), src.derived(), internal::assign_op()); - return dst.const_cast_derived(); - } +template const Product lazyprod(const Lhs &lhs, const Rhs &rhs) +{ + return Product(lhs, rhs); +} - template - void add_assign_using_evaluator(const DstXprType& dst, const SrcXprType& src) - { - typedef typename DstXprType::Scalar Scalar; - call_assignment(const_cast(dst), src.derived(), internal::add_assign_op()); - } +template +EIGEN_STRONG_INLINE DstXprType ©_using_evaluator(const EigenBase &dst, const SrcXprType &src) +{ + call_assignment(dst.const_cast_derived(), + src.derived(), + internal::assign_op()); + return dst.const_cast_derived(); +} - template - void subtract_assign_using_evaluator(const DstXprType& dst, const SrcXprType& src) - { - typedef typename DstXprType::Scalar Scalar; - call_assignment(const_cast(dst), src.derived(), internal::sub_assign_op()); - } +template class StorageBase, typename SrcXprType> +EIGEN_STRONG_INLINE const DstXprType ©_using_evaluator(const NoAlias &dst, + const SrcXprType &src) +{ + call_assignment(dst, src.derived(), internal::assign_op()); + return dst.expression(); +} - template - void multiply_assign_using_evaluator(const DstXprType& dst, const SrcXprType& src) - { - typedef typename DstXprType::Scalar Scalar; - call_assignment(dst.const_cast_derived(), src.derived(), internal::mul_assign_op()); - } +template +EIGEN_STRONG_INLINE DstXprType ©_using_evaluator(const PlainObjectBase &dst, const SrcXprType &src) +{ +#ifdef EIGEN_NO_AUTOMATIC_RESIZING + eigen_assert( + (dst.size() == 0 + || (IsVectorAtCompileTime ? (dst.size() == src.size()) : (dst.rows() == src.rows() && dst.cols() == src.cols()))) + && "Size mismatch. Automatic resizing is disabled because EIGEN_NO_AUTOMATIC_RESIZING is defined"); +#else + dst.const_cast_derived().resizeLike(src.derived()); +#endif + + call_assignment(dst.const_cast_derived(), + src.derived(), + internal::assign_op()); + return dst.const_cast_derived(); +} - template - void divide_assign_using_evaluator(const DstXprType& dst, const SrcXprType& src) - { - typedef typename DstXprType::Scalar Scalar; - call_assignment(dst.const_cast_derived(), src.derived(), internal::div_assign_op()); - } - - template - void swap_using_evaluator(const DstXprType& dst, const SrcXprType& src) +template +void add_assign_using_evaluator(const DstXprType &dst, const SrcXprType &src) +{ + typedef typename DstXprType::Scalar Scalar; + call_assignment( + const_cast(dst), src.derived(), internal::add_assign_op()); +} + +template +void subtract_assign_using_evaluator(const DstXprType &dst, const SrcXprType &src) +{ + typedef typename DstXprType::Scalar Scalar; + call_assignment( + const_cast(dst), src.derived(), internal::sub_assign_op()); +} + +template +void multiply_assign_using_evaluator(const DstXprType &dst, const SrcXprType &src) +{ + typedef typename DstXprType::Scalar Scalar; + call_assignment( + dst.const_cast_derived(), src.derived(), internal::mul_assign_op()); +} + +template +void divide_assign_using_evaluator(const DstXprType &dst, const SrcXprType &src) +{ + typedef typename DstXprType::Scalar Scalar; + call_assignment( + dst.const_cast_derived(), src.derived(), internal::div_assign_op()); +} + +template +void swap_using_evaluator(const DstXprType &dst, const SrcXprType &src) +{ + typedef typename DstXprType::Scalar Scalar; + call_assignment(dst.const_cast_derived(), src.const_cast_derived(), internal::swap_assign_op()); +} + +namespace internal { + template class StorageBase, typename Src, typename Func> + EIGEN_DEVICE_FUNC void call_assignment(const NoAlias &dst, const Src &src, const Func &func) { - typedef typename DstXprType::Scalar Scalar; - call_assignment(dst.const_cast_derived(), src.const_cast_derived(), internal::swap_assign_op()); + call_assignment_no_alias(dst.expression(), src, func); } +}// namespace internal - namespace internal { - template class StorageBase, typename Src, typename Func> - EIGEN_DEVICE_FUNC void call_assignment(const NoAlias& dst, const Src& src, const Func& func) - { - call_assignment_no_alias(dst.expression(), src, func); - } - } - -} +}// namespace Eigen -template long get_cost(const XprType& ) { return Eigen::internal::evaluator::CoeffReadCost; } +template long get_cost(const XprType &) { return Eigen::internal::evaluator::CoeffReadCost; } using namespace std; -#define VERIFY_IS_APPROX_EVALUATOR(DEST,EXPR) VERIFY_IS_APPROX(copy_using_evaluator(DEST,(EXPR)), (EXPR).eval()); -#define VERIFY_IS_APPROX_EVALUATOR2(DEST,EXPR,REF) VERIFY_IS_APPROX(copy_using_evaluator(DEST,(EXPR)), (REF).eval()); +#define VERIFY_IS_APPROX_EVALUATOR(DEST, EXPR) VERIFY_IS_APPROX(copy_using_evaluator(DEST, (EXPR)), (EXPR).eval()); +#define VERIFY_IS_APPROX_EVALUATOR2(DEST, EXPR, REF) VERIFY_IS_APPROX(copy_using_evaluator(DEST, (EXPR)), (REF).eval()); void test_evaluators() { @@ -113,20 +116,20 @@ void test_evaluators() VERIFY_IS_APPROX_EVALUATOR(v2, v_const); // Testing Transpose - VERIFY_IS_APPROX_EVALUATOR(w, v.transpose()); // Transpose as rvalue + VERIFY_IS_APPROX_EVALUATOR(w, v.transpose());// Transpose as rvalue VERIFY_IS_APPROX_EVALUATOR(w, v_const.transpose()); - copy_using_evaluator(w.transpose(), v); // Transpose as lvalue - VERIFY_IS_APPROX(w,v.transpose().eval()); + copy_using_evaluator(w.transpose(), v);// Transpose as lvalue + VERIFY_IS_APPROX(w, v.transpose().eval()); copy_using_evaluator(w.transpose(), v_const); - VERIFY_IS_APPROX(w,v_const.transpose().eval()); + VERIFY_IS_APPROX(w, v_const.transpose().eval()); // Testing Array evaluator { - ArrayXXf a(2,3); - ArrayXXf b(3,2); - a << 1,2,3, 4,5,6; + ArrayXXf a(2, 3); + ArrayXXf b(3, 2); + a << 1, 2, 3, 4, 5, 6; const ArrayXXf a_const(a); VERIFY_IS_APPROX_EVALUATOR(b, a.transpose()); @@ -135,99 +138,108 @@ void test_evaluators() // Testing CwiseNullaryOp evaluator copy_using_evaluator(w, RowVector2d::Random()); - VERIFY((w.array() >= -1).all() && (w.array() <= 1).all()); // not easy to test ... + VERIFY((w.array() >= -1).all() && (w.array() <= 1).all());// not easy to test ... VERIFY_IS_APPROX_EVALUATOR(w, RowVector2d::Zero()); VERIFY_IS_APPROX_EVALUATOR(w, RowVector2d::Constant(3)); - + // mix CwiseNullaryOp and transpose VERIFY_IS_APPROX_EVALUATOR(w, Vector2d::Zero().transpose()); } { // test product expressions - int s = internal::random(1,100); - MatrixXf a(s,s), b(s,s), c(s,s), d(s,s); + int s = internal::random(1, 100); + MatrixXf a(s, s), b(s, s), c(s, s), d(s, s); a.setRandom(); b.setRandom(); c.setRandom(); d.setRandom(); VERIFY_IS_APPROX_EVALUATOR(d, (a + b)); VERIFY_IS_APPROX_EVALUATOR(d, (a + b).transpose()); - VERIFY_IS_APPROX_EVALUATOR2(d, prod(a,b), a*b); - VERIFY_IS_APPROX_EVALUATOR2(d.noalias(), prod(a,b), a*b); - VERIFY_IS_APPROX_EVALUATOR2(d, prod(a,b) + c, a*b + c); - VERIFY_IS_APPROX_EVALUATOR2(d, s * prod(a,b), s * a*b); - VERIFY_IS_APPROX_EVALUATOR2(d, prod(a,b).transpose(), (a*b).transpose()); - VERIFY_IS_APPROX_EVALUATOR2(d, prod(a,b) + prod(b,c), a*b + b*c); + VERIFY_IS_APPROX_EVALUATOR2(d, prod(a, b), a * b); + VERIFY_IS_APPROX_EVALUATOR2(d.noalias(), prod(a, b), a * b); + VERIFY_IS_APPROX_EVALUATOR2(d, prod(a, b) + c, a * b + c); + VERIFY_IS_APPROX_EVALUATOR2(d, s * prod(a, b), s * a * b); + VERIFY_IS_APPROX_EVALUATOR2(d, prod(a, b).transpose(), (a * b).transpose()); + VERIFY_IS_APPROX_EVALUATOR2(d, prod(a, b) + prod(b, c), a * b + b * c); // check that prod works even with aliasing present - c = a*a; - copy_using_evaluator(a, prod(a,a)); - VERIFY_IS_APPROX(a,c); + c = a * a; + copy_using_evaluator(a, prod(a, a)); + VERIFY_IS_APPROX(a, c); // check compound assignment of products d = c; - add_assign_using_evaluator(c.noalias(), prod(a,b)); - d.noalias() += a*b; + add_assign_using_evaluator(c.noalias(), prod(a, b)); + d.noalias() += a * b; VERIFY_IS_APPROX(c, d); d = c; - subtract_assign_using_evaluator(c.noalias(), prod(a,b)); - d.noalias() -= a*b; + subtract_assign_using_evaluator(c.noalias(), prod(a, b)); + d.noalias() -= a * b; VERIFY_IS_APPROX(c, d); } { // test product with all possible sizes - int s = internal::random(1,100); - Matrix m11, res11; m11.setRandom(1,1); - Matrix m14, res14; m14.setRandom(1,4); - Matrix m1X, res1X; m1X.setRandom(1,s); - Matrix m41, res41; m41.setRandom(4,1); - Matrix m44, res44; m44.setRandom(4,4); - Matrix m4X, res4X; m4X.setRandom(4,s); - Matrix mX1, resX1; mX1.setRandom(s,1); - Matrix mX4, resX4; mX4.setRandom(s,4); - Matrix mXX, resXX; mXX.setRandom(s,s); - - VERIFY_IS_APPROX_EVALUATOR2(res11, prod(m11,m11), m11*m11); - VERIFY_IS_APPROX_EVALUATOR2(res11, prod(m14,m41), m14*m41); - VERIFY_IS_APPROX_EVALUATOR2(res11, prod(m1X,mX1), m1X*mX1); - VERIFY_IS_APPROX_EVALUATOR2(res14, prod(m11,m14), m11*m14); - VERIFY_IS_APPROX_EVALUATOR2(res14, prod(m14,m44), m14*m44); - VERIFY_IS_APPROX_EVALUATOR2(res14, prod(m1X,mX4), m1X*mX4); - VERIFY_IS_APPROX_EVALUATOR2(res1X, prod(m11,m1X), m11*m1X); - VERIFY_IS_APPROX_EVALUATOR2(res1X, prod(m14,m4X), m14*m4X); - VERIFY_IS_APPROX_EVALUATOR2(res1X, prod(m1X,mXX), m1X*mXX); - VERIFY_IS_APPROX_EVALUATOR2(res41, prod(m41,m11), m41*m11); - VERIFY_IS_APPROX_EVALUATOR2(res41, prod(m44,m41), m44*m41); - VERIFY_IS_APPROX_EVALUATOR2(res41, prod(m4X,mX1), m4X*mX1); - VERIFY_IS_APPROX_EVALUATOR2(res44, prod(m41,m14), m41*m14); - VERIFY_IS_APPROX_EVALUATOR2(res44, prod(m44,m44), m44*m44); - VERIFY_IS_APPROX_EVALUATOR2(res44, prod(m4X,mX4), m4X*mX4); - VERIFY_IS_APPROX_EVALUATOR2(res4X, prod(m41,m1X), m41*m1X); - VERIFY_IS_APPROX_EVALUATOR2(res4X, prod(m44,m4X), m44*m4X); - VERIFY_IS_APPROX_EVALUATOR2(res4X, prod(m4X,mXX), m4X*mXX); - VERIFY_IS_APPROX_EVALUATOR2(resX1, prod(mX1,m11), mX1*m11); - VERIFY_IS_APPROX_EVALUATOR2(resX1, prod(mX4,m41), mX4*m41); - VERIFY_IS_APPROX_EVALUATOR2(resX1, prod(mXX,mX1), mXX*mX1); - VERIFY_IS_APPROX_EVALUATOR2(resX4, prod(mX1,m14), mX1*m14); - VERIFY_IS_APPROX_EVALUATOR2(resX4, prod(mX4,m44), mX4*m44); - VERIFY_IS_APPROX_EVALUATOR2(resX4, prod(mXX,mX4), mXX*mX4); - VERIFY_IS_APPROX_EVALUATOR2(resXX, prod(mX1,m1X), mX1*m1X); - VERIFY_IS_APPROX_EVALUATOR2(resXX, prod(mX4,m4X), mX4*m4X); - VERIFY_IS_APPROX_EVALUATOR2(resXX, prod(mXX,mXX), mXX*mXX); + int s = internal::random(1, 100); + Matrix m11, res11; + m11.setRandom(1, 1); + Matrix m14, res14; + m14.setRandom(1, 4); + Matrix m1X, res1X; + m1X.setRandom(1, s); + Matrix m41, res41; + m41.setRandom(4, 1); + Matrix m44, res44; + m44.setRandom(4, 4); + Matrix m4X, res4X; + m4X.setRandom(4, s); + Matrix mX1, resX1; + mX1.setRandom(s, 1); + Matrix mX4, resX4; + mX4.setRandom(s, 4); + Matrix mXX, resXX; + mXX.setRandom(s, s); + + VERIFY_IS_APPROX_EVALUATOR2(res11, prod(m11, m11), m11 * m11); + VERIFY_IS_APPROX_EVALUATOR2(res11, prod(m14, m41), m14 * m41); + VERIFY_IS_APPROX_EVALUATOR2(res11, prod(m1X, mX1), m1X * mX1); + VERIFY_IS_APPROX_EVALUATOR2(res14, prod(m11, m14), m11 * m14); + VERIFY_IS_APPROX_EVALUATOR2(res14, prod(m14, m44), m14 * m44); + VERIFY_IS_APPROX_EVALUATOR2(res14, prod(m1X, mX4), m1X * mX4); + VERIFY_IS_APPROX_EVALUATOR2(res1X, prod(m11, m1X), m11 * m1X); + VERIFY_IS_APPROX_EVALUATOR2(res1X, prod(m14, m4X), m14 * m4X); + VERIFY_IS_APPROX_EVALUATOR2(res1X, prod(m1X, mXX), m1X * mXX); + VERIFY_IS_APPROX_EVALUATOR2(res41, prod(m41, m11), m41 * m11); + VERIFY_IS_APPROX_EVALUATOR2(res41, prod(m44, m41), m44 * m41); + VERIFY_IS_APPROX_EVALUATOR2(res41, prod(m4X, mX1), m4X * mX1); + VERIFY_IS_APPROX_EVALUATOR2(res44, prod(m41, m14), m41 * m14); + VERIFY_IS_APPROX_EVALUATOR2(res44, prod(m44, m44), m44 * m44); + VERIFY_IS_APPROX_EVALUATOR2(res44, prod(m4X, mX4), m4X * mX4); + VERIFY_IS_APPROX_EVALUATOR2(res4X, prod(m41, m1X), m41 * m1X); + VERIFY_IS_APPROX_EVALUATOR2(res4X, prod(m44, m4X), m44 * m4X); + VERIFY_IS_APPROX_EVALUATOR2(res4X, prod(m4X, mXX), m4X * mXX); + VERIFY_IS_APPROX_EVALUATOR2(resX1, prod(mX1, m11), mX1 * m11); + VERIFY_IS_APPROX_EVALUATOR2(resX1, prod(mX4, m41), mX4 * m41); + VERIFY_IS_APPROX_EVALUATOR2(resX1, prod(mXX, mX1), mXX * mX1); + VERIFY_IS_APPROX_EVALUATOR2(resX4, prod(mX1, m14), mX1 * m14); + VERIFY_IS_APPROX_EVALUATOR2(resX4, prod(mX4, m44), mX4 * m44); + VERIFY_IS_APPROX_EVALUATOR2(resX4, prod(mXX, mX4), mXX * mX4); + VERIFY_IS_APPROX_EVALUATOR2(resXX, prod(mX1, m1X), mX1 * m1X); + VERIFY_IS_APPROX_EVALUATOR2(resXX, prod(mX4, m4X), mX4 * m4X); + VERIFY_IS_APPROX_EVALUATOR2(resXX, prod(mXX, mXX), mXX * mXX); } { - ArrayXXf a(2,3); - ArrayXXf b(3,2); - a << 1,2,3, 4,5,6; + ArrayXXf a(2, 3); + ArrayXXf b(3, 2); + a << 1, 2, 3, 4, 5, 6; const ArrayXXf a_const(a); - - // this does not work because Random is eval-before-nested: + + // this does not work because Random is eval-before-nested: // copy_using_evaluator(w, Vector2d::Random().transpose()); // test CwiseUnaryOp @@ -241,61 +253,61 @@ void test_evaluators() VERIFY_IS_APPROX_EVALUATOR(w, (v + Vector2d::Ones()).transpose().cwiseProduct(RowVector2d::Constant(3))); // dynamic matrices and arrays - MatrixXd mat1(6,6), mat2(6,6); - VERIFY_IS_APPROX_EVALUATOR(mat1, MatrixXd::Identity(6,6)); + MatrixXd mat1(6, 6), mat2(6, 6); + VERIFY_IS_APPROX_EVALUATOR(mat1, MatrixXd::Identity(6, 6)); VERIFY_IS_APPROX_EVALUATOR(mat2, mat1); copy_using_evaluator(mat2.transpose(), mat1); VERIFY_IS_APPROX(mat2.transpose(), mat1); - ArrayXXd arr1(6,6), arr2(6,6); - VERIFY_IS_APPROX_EVALUATOR(arr1, ArrayXXd::Constant(6,6, 3.0)); + ArrayXXd arr1(6, 6), arr2(6, 6); + VERIFY_IS_APPROX_EVALUATOR(arr1, ArrayXXd::Constant(6, 6, 3.0)); VERIFY_IS_APPROX_EVALUATOR(arr2, arr1); - + // test automatic resizing - mat2.resize(3,3); + mat2.resize(3, 3); VERIFY_IS_APPROX_EVALUATOR(mat2, mat1); - arr2.resize(9,9); + arr2.resize(9, 9); VERIFY_IS_APPROX_EVALUATOR(arr2, arr1); // test direct traversal Matrix3f m3; Array33f a3; - VERIFY_IS_APPROX_EVALUATOR(m3, Matrix3f::Identity()); // matrix, nullary + VERIFY_IS_APPROX_EVALUATOR(m3, Matrix3f::Identity());// matrix, nullary // TODO: find a way to test direct traversal with array - VERIFY_IS_APPROX_EVALUATOR(m3.transpose(), Matrix3f::Identity().transpose()); // transpose - VERIFY_IS_APPROX_EVALUATOR(m3, 2 * Matrix3f::Identity()); // unary - VERIFY_IS_APPROX_EVALUATOR(m3, Matrix3f::Identity() + Matrix3f::Zero()); // binary - VERIFY_IS_APPROX_EVALUATOR(m3.block(0,0,2,2), Matrix3f::Identity().block(1,1,2,2)); // block + VERIFY_IS_APPROX_EVALUATOR(m3.transpose(), Matrix3f::Identity().transpose());// transpose + VERIFY_IS_APPROX_EVALUATOR(m3, 2 * Matrix3f::Identity());// unary + VERIFY_IS_APPROX_EVALUATOR(m3, Matrix3f::Identity() + Matrix3f::Zero());// binary + VERIFY_IS_APPROX_EVALUATOR(m3.block(0, 0, 2, 2), Matrix3f::Identity().block(1, 1, 2, 2));// block // test linear traversal - VERIFY_IS_APPROX_EVALUATOR(m3, Matrix3f::Zero()); // matrix, nullary - VERIFY_IS_APPROX_EVALUATOR(a3, Array33f::Zero()); // array - VERIFY_IS_APPROX_EVALUATOR(m3.transpose(), Matrix3f::Zero().transpose()); // transpose - VERIFY_IS_APPROX_EVALUATOR(m3, 2 * Matrix3f::Zero()); // unary - VERIFY_IS_APPROX_EVALUATOR(m3, Matrix3f::Zero() + m3); // binary + VERIFY_IS_APPROX_EVALUATOR(m3, Matrix3f::Zero());// matrix, nullary + VERIFY_IS_APPROX_EVALUATOR(a3, Array33f::Zero());// array + VERIFY_IS_APPROX_EVALUATOR(m3.transpose(), Matrix3f::Zero().transpose());// transpose + VERIFY_IS_APPROX_EVALUATOR(m3, 2 * Matrix3f::Zero());// unary + VERIFY_IS_APPROX_EVALUATOR(m3, Matrix3f::Zero() + m3);// binary // test inner vectorization Matrix4f m4, m4src = Matrix4f::Random(); Array44f a4, a4src = Matrix4f::Random(); - VERIFY_IS_APPROX_EVALUATOR(m4, m4src); // matrix - VERIFY_IS_APPROX_EVALUATOR(a4, a4src); // array - VERIFY_IS_APPROX_EVALUATOR(m4.transpose(), m4src.transpose()); // transpose + VERIFY_IS_APPROX_EVALUATOR(m4, m4src);// matrix + VERIFY_IS_APPROX_EVALUATOR(a4, a4src);// array + VERIFY_IS_APPROX_EVALUATOR(m4.transpose(), m4src.transpose());// transpose // TODO: find out why Matrix4f::Zero() does not allow inner vectorization - VERIFY_IS_APPROX_EVALUATOR(m4, 2 * m4src); // unary - VERIFY_IS_APPROX_EVALUATOR(m4, m4src + m4src); // binary + VERIFY_IS_APPROX_EVALUATOR(m4, 2 * m4src);// unary + VERIFY_IS_APPROX_EVALUATOR(m4, m4src + m4src);// binary // test linear vectorization - MatrixXf mX(6,6), mXsrc = MatrixXf::Random(6,6); - ArrayXXf aX(6,6), aXsrc = ArrayXXf::Random(6,6); - VERIFY_IS_APPROX_EVALUATOR(mX, mXsrc); // matrix - VERIFY_IS_APPROX_EVALUATOR(aX, aXsrc); // array - VERIFY_IS_APPROX_EVALUATOR(mX.transpose(), mXsrc.transpose()); // transpose - VERIFY_IS_APPROX_EVALUATOR(mX, MatrixXf::Zero(6,6)); // nullary - VERIFY_IS_APPROX_EVALUATOR(mX, 2 * mXsrc); // unary - VERIFY_IS_APPROX_EVALUATOR(mX, mXsrc + mXsrc); // binary + MatrixXf mX(6, 6), mXsrc = MatrixXf::Random(6, 6); + ArrayXXf aX(6, 6), aXsrc = ArrayXXf::Random(6, 6); + VERIFY_IS_APPROX_EVALUATOR(mX, mXsrc);// matrix + VERIFY_IS_APPROX_EVALUATOR(aX, aXsrc);// array + VERIFY_IS_APPROX_EVALUATOR(mX.transpose(), mXsrc.transpose());// transpose + VERIFY_IS_APPROX_EVALUATOR(mX, MatrixXf::Zero(6, 6));// nullary + VERIFY_IS_APPROX_EVALUATOR(mX, 2 * mXsrc);// unary + VERIFY_IS_APPROX_EVALUATOR(mX, mXsrc + mXsrc);// binary // test blocks and slice vectorization - VERIFY_IS_APPROX_EVALUATOR(m4, (mXsrc.block<4,4>(1,0))); + VERIFY_IS_APPROX_EVALUATOR(m4, (mXsrc.block<4, 4>(1, 0))); VERIFY_IS_APPROX_EVALUATOR(aX, ArrayXXf::Constant(10, 10, 3.0).block(2, 3, 6, 6)); Matrix4f m4ref = m4; @@ -303,21 +315,21 @@ void test_evaluators() m4ref.block(1, 1, 2, 3) = m3.bottomRows(2); VERIFY_IS_APPROX(m4, m4ref); - mX.setIdentity(20,20); - MatrixXf mXref = MatrixXf::Identity(20,20); - mXsrc = MatrixXf::Random(9,12); + mX.setIdentity(20, 20); + MatrixXf mXref = MatrixXf::Identity(20, 20); + mXsrc = MatrixXf::Random(9, 12); copy_using_evaluator(mX.block(4, 4, 9, 12), mXsrc); mXref.block(4, 4, 9, 12) = mXsrc; VERIFY_IS_APPROX(mX, mXref); // test Map - const float raw[3] = {1,2,3}; - float buffer[3] = {0,0,0}; + const float raw[3] = { 1, 2, 3 }; + float buffer[3] = { 0, 0, 0 }; Vector3f v3; Array3f a3f; VERIFY_IS_APPROX_EVALUATOR(v3, Map(raw)); VERIFY_IS_APPROX_EVALUATOR(a3f, Map(raw)); - Vector3f::Map(buffer) = 2*v3; + Vector3f::Map(buffer) = 2 * v3; VERIFY(buffer[0] == 2); VERIFY(buffer[1] == 4); VERIFY(buffer[2] == 6); @@ -325,7 +337,7 @@ void test_evaluators() // test CwiseUnaryView mat1.setRandom(); mat2.setIdentity(); - MatrixXcd matXcd(6,6), matXcd_ref(6,6); + MatrixXcd matXcd(6, 6), matXcd_ref(6, 6); copy_using_evaluator(matXcd.real(), mat1); copy_using_evaluator(matXcd.imag(), mat2); matXcd_ref.real() = mat1; @@ -341,8 +353,8 @@ void test_evaluators() mX.resize(6, 6); VERIFY_IS_APPROX_EVALUATOR(mX, mXsrc.colwise() + vX); matXcd.resize(12, 12); - VERIFY_IS_APPROX_EVALUATOR(matXcd, matXcd_ref.replicate(2,2)); - VERIFY_IS_APPROX_EVALUATOR(matXcd, (matXcd_ref.replicate<2,2>())); + VERIFY_IS_APPROX_EVALUATOR(matXcd, matXcd_ref.replicate(2, 2)); + VERIFY_IS_APPROX_EVALUATOR(matXcd, (matXcd_ref.replicate<2, 2>())); // test partial reductions VectorXd vec1(6); @@ -350,16 +362,16 @@ void test_evaluators() VERIFY_IS_APPROX_EVALUATOR(vec1, mat1.colwise().sum().transpose()); // test MatrixWrapper and ArrayWrapper - mat1.setRandom(6,6); - arr1.setRandom(6,6); + mat1.setRandom(6, 6); + arr1.setRandom(6, 6); VERIFY_IS_APPROX_EVALUATOR(mat2, arr1.matrix()); VERIFY_IS_APPROX_EVALUATOR(arr2, mat1.array()); VERIFY_IS_APPROX_EVALUATOR(mat2, (arr1 + 2).matrix()); VERIFY_IS_APPROX_EVALUATOR(arr2, mat1.array() + 2); mat2.array() = arr1 * arr1; VERIFY_IS_APPROX(mat2, (arr1 * arr1).matrix()); - arr2.matrix() = MatrixXd::Identity(6,6); - VERIFY_IS_APPROX(arr2, MatrixXd::Identity(6,6).array()); + arr2.matrix() = MatrixXd::Identity(6, 6); + VERIFY_IS_APPROX(arr2, MatrixXd::Identity(6, 6).array()); // test Reverse VERIFY_IS_APPROX_EVALUATOR(arr2, arr1.reverse()); @@ -386,7 +398,7 @@ void test_evaluators() mat2.diagonal<-1>() = mat2.diagonal(1); VERIFY_IS_APPROX(mat1, mat2); } - + { // test swapping MatrixXd mat1, mat2, mat1ref, mat2ref; @@ -410,18 +422,18 @@ void test_evaluators() { // test compound assignment - const Matrix4d mat_const = Matrix4d::Random(); + const Matrix4d mat_const = Matrix4d::Random(); Matrix4d mat, mat_ref; mat = mat_ref = Matrix4d::Identity(); add_assign_using_evaluator(mat, mat_const); mat_ref += mat_const; VERIFY_IS_APPROX(mat, mat_ref); - subtract_assign_using_evaluator(mat.row(1), 2*mat.row(2)); - mat_ref.row(1) -= 2*mat_ref.row(2); + subtract_assign_using_evaluator(mat.row(1), 2 * mat.row(2)); + mat_ref.row(1) -= 2 * mat_ref.row(2); VERIFY_IS_APPROX(mat, mat_ref); - const ArrayXXf arr_const = ArrayXXf::Random(5,3); + const ArrayXXf arr_const = ArrayXXf::Random(5, 3); ArrayXXf arr, arr_ref; arr = arr_ref = ArrayXXf::Constant(5, 3, 0.5); multiply_assign_using_evaluator(arr, arr_const); @@ -432,68 +444,81 @@ void test_evaluators() arr_ref.row(1) /= (arr_ref.row(2) + 1); VERIFY_IS_APPROX(arr, arr_ref); } - + { // test triangular shapes - MatrixXd A = MatrixXd::Random(6,6), B(6,6), C(6,6), D(6,6); - A.setRandom();B.setRandom(); + MatrixXd A = MatrixXd::Random(6, 6), B(6, 6), C(6, 6), D(6, 6); + A.setRandom(); + B.setRandom(); VERIFY_IS_APPROX_EVALUATOR2(B, A.triangularView(), MatrixXd(A.triangularView())); - - A.setRandom();B.setRandom(); + + A.setRandom(); + B.setRandom(); VERIFY_IS_APPROX_EVALUATOR2(B, A.triangularView(), MatrixXd(A.triangularView())); - - A.setRandom();B.setRandom(); + + A.setRandom(); + B.setRandom(); VERIFY_IS_APPROX_EVALUATOR2(B, A.triangularView(), MatrixXd(A.triangularView())); - - A.setRandom();B.setRandom(); - C = B; C.triangularView() = A; + + A.setRandom(); + B.setRandom(); + C = B; + C.triangularView() = A; copy_using_evaluator(B.triangularView(), A); VERIFY(B.isApprox(C) && "copy_using_evaluator(B.triangularView(), A)"); - - A.setRandom();B.setRandom(); - C = B; C.triangularView() = A.triangularView(); + + A.setRandom(); + B.setRandom(); + C = B; + C.triangularView() = A.triangularView(); copy_using_evaluator(B.triangularView(), A.triangularView()); VERIFY(B.isApprox(C) && "copy_using_evaluator(B.triangularView(), A.triangularView())"); - - - A.setRandom();B.setRandom(); - C = B; C.triangularView() = A.triangularView().transpose(); + + + A.setRandom(); + B.setRandom(); + C = B; + C.triangularView() = A.triangularView().transpose(); copy_using_evaluator(B.triangularView(), A.triangularView().transpose()); VERIFY(B.isApprox(C) && "copy_using_evaluator(B.triangularView(), A.triangularView().transpose())"); - - - A.setRandom();B.setRandom(); C = B; D = A; + + + A.setRandom(); + B.setRandom(); + C = B; + D = A; C.triangularView().swap(D.triangularView()); swap_using_evaluator(B.triangularView(), A.triangularView()); VERIFY(B.isApprox(C) && "swap_using_evaluator(B.triangularView(), A.triangularView())"); - - - VERIFY_IS_APPROX_EVALUATOR2(B, prod(A.triangularView(),A), MatrixXd(A.triangularView()*A)); - - VERIFY_IS_APPROX_EVALUATOR2(B, prod(A.selfadjointView(),A), MatrixXd(A.selfadjointView()*A)); + + + VERIFY_IS_APPROX_EVALUATOR2(B, prod(A.triangularView(), A), MatrixXd(A.triangularView() * A)); + + VERIFY_IS_APPROX_EVALUATOR2(B, prod(A.selfadjointView(), A), MatrixXd(A.selfadjointView() * A)); } { // test diagonal shapes VectorXd d = VectorXd::Random(6); - MatrixXd A = MatrixXd::Random(6,6), B(6,6); - A.setRandom();B.setRandom(); - - VERIFY_IS_APPROX_EVALUATOR2(B, lazyprod(d.asDiagonal(),A), MatrixXd(d.asDiagonal()*A)); - VERIFY_IS_APPROX_EVALUATOR2(B, lazyprod(A,d.asDiagonal()), MatrixXd(A*d.asDiagonal())); + MatrixXd A = MatrixXd::Random(6, 6), B(6, 6); + A.setRandom(); + B.setRandom(); + + VERIFY_IS_APPROX_EVALUATOR2(B, lazyprod(d.asDiagonal(), A), MatrixXd(d.asDiagonal() * A)); + VERIFY_IS_APPROX_EVALUATOR2(B, lazyprod(A, d.asDiagonal()), MatrixXd(A * d.asDiagonal())); } { // test CoeffReadCost Matrix4d a, b; - VERIFY_IS_EQUAL( get_cost(a), 1 ); - VERIFY_IS_EQUAL( get_cost(a+b), 3); - VERIFY_IS_EQUAL( get_cost(2*a+b), 4); - VERIFY_IS_EQUAL( get_cost(a*b), 1); - VERIFY_IS_EQUAL( get_cost(a.lazyProduct(b)), 15); - VERIFY_IS_EQUAL( get_cost(a*(a*b)), 1); - VERIFY_IS_EQUAL( get_cost(a.lazyProduct(a*b)), 15); - VERIFY_IS_EQUAL( get_cost(a*(a+b)), 1); - VERIFY_IS_EQUAL( get_cost(a.lazyProduct(a+b)), 15); + VERIFY_IS_EQUAL(get_cost(a), 1); + VERIFY_IS_EQUAL(get_cost(a + b), 3); + VERIFY_IS_EQUAL(get_cost(2 * a + b), 4); + VERIFY_IS_EQUAL(get_cost(a * b), 1); + VERIFY_IS_EQUAL(get_cost(a.lazyProduct(b)), 15); + VERIFY_IS_EQUAL(get_cost(a * (a * b)), 1); + VERIFY_IS_EQUAL(get_cost(a.lazyProduct(a * b)), 15); + VERIFY_IS_EQUAL(get_cost(a * (a + b)), 1); + VERIFY_IS_EQUAL(get_cost(a.lazyProduct(a + b)), 15); } } diff --git a/filmulator-gui/core/nlmeans/eigen/test/exceptions.cpp b/filmulator-gui/core/nlmeans/eigen/test/exceptions.cpp index b83fb82b..500e2ee4 100644 --- a/filmulator-gui/core/nlmeans/eigen/test/exceptions.cpp +++ b/filmulator-gui/core/nlmeans/eigen/test/exceptions.cpp @@ -21,93 +21,115 @@ struct my_exception my_exception() {} ~my_exception() {} }; - + class ScalarWithExceptions { - public: - ScalarWithExceptions() { init(); } - ScalarWithExceptions(const float& _v) { init(); *v = _v; } - ScalarWithExceptions(const ScalarWithExceptions& other) { init(); *v = *(other.v); } - ~ScalarWithExceptions() { - delete v; - instances--; - } - - void init() { - v = new float; - instances++; - } - - ScalarWithExceptions operator+(const ScalarWithExceptions& other) const - { - countdown--; - if(countdown<=0) - throw my_exception(); - return ScalarWithExceptions(*v+*other.v); - } - - ScalarWithExceptions operator-(const ScalarWithExceptions& other) const - { return ScalarWithExceptions(*v-*other.v); } - - ScalarWithExceptions operator*(const ScalarWithExceptions& other) const - { return ScalarWithExceptions((*v)*(*other.v)); } - - ScalarWithExceptions& operator+=(const ScalarWithExceptions& other) - { *v+=*other.v; return *this; } - ScalarWithExceptions& operator-=(const ScalarWithExceptions& other) - { *v-=*other.v; return *this; } - ScalarWithExceptions& operator=(const ScalarWithExceptions& other) - { *v = *(other.v); return *this; } - - bool operator==(const ScalarWithExceptions& other) const - { return *v==*other.v; } - bool operator!=(const ScalarWithExceptions& other) const - { return *v!=*other.v; } - - float* v; - static int instances; - static int countdown; +public: + ScalarWithExceptions() { init(); } + ScalarWithExceptions(const float &_v) + { + init(); + *v = _v; + } + ScalarWithExceptions(const ScalarWithExceptions &other) + { + init(); + *v = *(other.v); + } + ~ScalarWithExceptions() + { + delete v; + instances--; + } + + void init() + { + v = new float; + instances++; + } + + ScalarWithExceptions operator+(const ScalarWithExceptions &other) const + { + countdown--; + if (countdown <= 0) throw my_exception(); + return ScalarWithExceptions(*v + *other.v); + } + + ScalarWithExceptions operator-(const ScalarWithExceptions &other) const + { + return ScalarWithExceptions(*v - *other.v); + } + + ScalarWithExceptions operator*(const ScalarWithExceptions &other) const + { + return ScalarWithExceptions((*v) * (*other.v)); + } + + ScalarWithExceptions &operator+=(const ScalarWithExceptions &other) + { + *v += *other.v; + return *this; + } + ScalarWithExceptions &operator-=(const ScalarWithExceptions &other) + { + *v -= *other.v; + return *this; + } + ScalarWithExceptions &operator=(const ScalarWithExceptions &other) + { + *v = *(other.v); + return *this; + } + + bool operator==(const ScalarWithExceptions &other) const { return *v == *other.v; } + bool operator!=(const ScalarWithExceptions &other) const { return *v != *other.v; } + + float *v; + static int instances; + static int countdown; }; ScalarWithExceptions real(const ScalarWithExceptions &x) { return x; } -ScalarWithExceptions imag(const ScalarWithExceptions & ) { return 0; } +ScalarWithExceptions imag(const ScalarWithExceptions &) { return 0; } ScalarWithExceptions conj(const ScalarWithExceptions &x) { return x; } int ScalarWithExceptions::instances = 0; int ScalarWithExceptions::countdown = 0; -#define CHECK_MEMLEAK(OP) { \ - ScalarWithExceptions::countdown = 100; \ - int before = ScalarWithExceptions::instances; \ - bool exception_thrown = false; \ - try { OP; } \ - catch (my_exception) { \ - exception_thrown = true; \ - VERIFY(ScalarWithExceptions::instances==before && "memory leak detected in " && EIGEN_MAKESTRING(OP)); \ - } \ - VERIFY(exception_thrown && " no exception thrown in " && EIGEN_MAKESTRING(OP)); \ +#define CHECK_MEMLEAK(OP) \ + { \ + ScalarWithExceptions::countdown = 100; \ + int before = ScalarWithExceptions::instances; \ + bool exception_thrown = false; \ + try { \ + OP; \ + } catch (my_exception) { \ + exception_thrown = true; \ + VERIFY(ScalarWithExceptions::instances == before && "memory leak detected in " && EIGEN_MAKESTRING(OP)); \ + } \ + VERIFY(exception_thrown && " no exception thrown in " && EIGEN_MAKESTRING(OP)); \ } void memoryleak() { - typedef Eigen::Matrix VectorType; - typedef Eigen::Matrix MatrixType; - + typedef Eigen::Matrix VectorType; + typedef Eigen::Matrix MatrixType; + { int n = 50; VectorType v0(n), v1(n); - MatrixType m0(n,n), m1(n,n), m2(n,n); - v0.setOnes(); v1.setOnes(); - m0.setOnes(); m1.setOnes(); m2.setOnes(); + MatrixType m0(n, n), m1(n, n), m2(n, n); + v0.setOnes(); + v1.setOnes(); + m0.setOnes(); + m1.setOnes(); + m2.setOnes(); CHECK_MEMLEAK(v0 = m0 * m1 * v1); CHECK_MEMLEAK(m2 = m0 * m1 * m2); - CHECK_MEMLEAK((v0+v1).dot(v0+v1)); + CHECK_MEMLEAK((v0 + v1).dot(v0 + v1)); } - VERIFY(ScalarWithExceptions::instances==0 && "global memory leak detected in " && EIGEN_MAKESTRING(OP)); \ + VERIFY(ScalarWithExceptions::instances == 0 && "global memory leak detected in " && EIGEN_MAKESTRING(OP)); } -void test_exceptions() -{ - CALL_SUBTEST( memoryleak() ); -} +void test_exceptions() { CALL_SUBTEST(memoryleak()); } diff --git a/filmulator-gui/core/nlmeans/eigen/test/fastmath.cpp b/filmulator-gui/core/nlmeans/eigen/test/fastmath.cpp index cc5db074..25d63119 100644 --- a/filmulator-gui/core/nlmeans/eigen/test/fastmath.cpp +++ b/filmulator-gui/core/nlmeans/eigen/test/fastmath.cpp @@ -12,7 +12,7 @@ void check(bool b, bool ref) { std::cout << b; - if(b==ref) + if (b == ref) std::cout << " OK "; else std::cout << " BAD "; @@ -20,78 +20,121 @@ void check(bool b, bool ref) #if EIGEN_COMP_MSVC && EIGEN_COMP_MSVC < 1800 namespace std { - template bool (isfinite)(T x) { return _finite(x); } - template bool (isnan)(T x) { return _isnan(x); } - template bool (isinf)(T x) { return _fpclass(x)==_FPCLASS_NINF || _fpclass(x)==_FPCLASS_PINF; } -} +template bool(isfinite)(T x) { return _finite(x); } +template bool(isnan)(T x) { return _isnan(x); } +template bool(isinf)(T x) { return _fpclass(x) == _FPCLASS_NINF || _fpclass(x) == _FPCLASS_PINF; } +}// namespace std #endif -template -void check_inf_nan(bool dryrun) { - Matrix m(10); +template void check_inf_nan(bool dryrun) +{ + Matrix m(10); m.setRandom(); m(3) = std::numeric_limits::quiet_NaN(); - if(dryrun) - { - std::cout << "std::isfinite(" << m(3) << ") = "; check((std::isfinite)(m(3)),false); std::cout << " ; numext::isfinite = "; check((numext::isfinite)(m(3)), false); std::cout << "\n"; - std::cout << "std::isinf(" << m(3) << ") = "; check((std::isinf)(m(3)),false); std::cout << " ; numext::isinf = "; check((numext::isinf)(m(3)), false); std::cout << "\n"; - std::cout << "std::isnan(" << m(3) << ") = "; check((std::isnan)(m(3)),true); std::cout << " ; numext::isnan = "; check((numext::isnan)(m(3)), true); std::cout << "\n"; - std::cout << "allFinite: "; check(m.allFinite(), 0); std::cout << "\n"; - std::cout << "hasNaN: "; check(m.hasNaN(), 1); std::cout << "\n"; + if (dryrun) { + std::cout << "std::isfinite(" << m(3) << ") = "; + check((std::isfinite)(m(3)), false); + std::cout << " ; numext::isfinite = "; + check((numext::isfinite)(m(3)), false); std::cout << "\n"; + std::cout << "std::isinf(" << m(3) << ") = "; + check((std::isinf)(m(3)), false); + std::cout << " ; numext::isinf = "; + check((numext::isinf)(m(3)), false); + std::cout << "\n"; + std::cout << "std::isnan(" << m(3) << ") = "; + check((std::isnan)(m(3)), true); + std::cout << " ; numext::isnan = "; + check((numext::isnan)(m(3)), true); + std::cout << "\n"; + std::cout << "allFinite: "; + check(m.allFinite(), 0); + std::cout << "\n"; + std::cout << "hasNaN: "; + check(m.hasNaN(), 1); + std::cout << "\n"; + std::cout << "\n"; + } else { + VERIFY(!(numext::isfinite)(m(3))); + VERIFY(!(numext::isinf)(m(3))); + VERIFY((numext::isnan)(m(3))); + VERIFY(!m.allFinite()); + VERIFY(m.hasNaN()); } - else - { - VERIFY( !(numext::isfinite)(m(3)) ); - VERIFY( !(numext::isinf)(m(3)) ); - VERIFY( (numext::isnan)(m(3)) ); - VERIFY( !m.allFinite() ); - VERIFY( m.hasNaN() ); - } - T hidden_zero = (std::numeric_limits::min)()*(std::numeric_limits::min)(); + T hidden_zero = (std::numeric_limits::min)() * (std::numeric_limits::min)(); m(4) /= hidden_zero; - if(dryrun) - { - std::cout << "std::isfinite(" << m(4) << ") = "; check((std::isfinite)(m(4)),false); std::cout << " ; numext::isfinite = "; check((numext::isfinite)(m(4)), false); std::cout << "\n"; - std::cout << "std::isinf(" << m(4) << ") = "; check((std::isinf)(m(4)),true); std::cout << " ; numext::isinf = "; check((numext::isinf)(m(4)), true); std::cout << "\n"; - std::cout << "std::isnan(" << m(4) << ") = "; check((std::isnan)(m(4)),false); std::cout << " ; numext::isnan = "; check((numext::isnan)(m(4)), false); std::cout << "\n"; - std::cout << "allFinite: "; check(m.allFinite(), 0); std::cout << "\n"; - std::cout << "hasNaN: "; check(m.hasNaN(), 1); std::cout << "\n"; + if (dryrun) { + std::cout << "std::isfinite(" << m(4) << ") = "; + check((std::isfinite)(m(4)), false); + std::cout << " ; numext::isfinite = "; + check((numext::isfinite)(m(4)), false); std::cout << "\n"; - } - else - { - VERIFY( !(numext::isfinite)(m(4)) ); - VERIFY( (numext::isinf)(m(4)) ); - VERIFY( !(numext::isnan)(m(4)) ); - VERIFY( !m.allFinite() ); - VERIFY( m.hasNaN() ); + std::cout << "std::isinf(" << m(4) << ") = "; + check((std::isinf)(m(4)), true); + std::cout << " ; numext::isinf = "; + check((numext::isinf)(m(4)), true); + std::cout << "\n"; + std::cout << "std::isnan(" << m(4) << ") = "; + check((std::isnan)(m(4)), false); + std::cout << " ; numext::isnan = "; + check((numext::isnan)(m(4)), false); + std::cout << "\n"; + std::cout << "allFinite: "; + check(m.allFinite(), 0); + std::cout << "\n"; + std::cout << "hasNaN: "; + check(m.hasNaN(), 1); + std::cout << "\n"; + std::cout << "\n"; + } else { + VERIFY(!(numext::isfinite)(m(4))); + VERIFY((numext::isinf)(m(4))); + VERIFY(!(numext::isnan)(m(4))); + VERIFY(!m.allFinite()); + VERIFY(m.hasNaN()); } m(3) = 0; - if(dryrun) - { - std::cout << "std::isfinite(" << m(3) << ") = "; check((std::isfinite)(m(3)),true); std::cout << " ; numext::isfinite = "; check((numext::isfinite)(m(3)), true); std::cout << "\n"; - std::cout << "std::isinf(" << m(3) << ") = "; check((std::isinf)(m(3)),false); std::cout << " ; numext::isinf = "; check((numext::isinf)(m(3)), false); std::cout << "\n"; - std::cout << "std::isnan(" << m(3) << ") = "; check((std::isnan)(m(3)),false); std::cout << " ; numext::isnan = "; check((numext::isnan)(m(3)), false); std::cout << "\n"; - std::cout << "allFinite: "; check(m.allFinite(), 0); std::cout << "\n"; - std::cout << "hasNaN: "; check(m.hasNaN(), 0); std::cout << "\n"; + if (dryrun) { + std::cout << "std::isfinite(" << m(3) << ") = "; + check((std::isfinite)(m(3)), true); + std::cout << " ; numext::isfinite = "; + check((numext::isfinite)(m(3)), true); + std::cout << "\n"; + std::cout << "std::isinf(" << m(3) << ") = "; + check((std::isinf)(m(3)), false); + std::cout << " ; numext::isinf = "; + check((numext::isinf)(m(3)), false); + std::cout << "\n"; + std::cout << "std::isnan(" << m(3) << ") = "; + check((std::isnan)(m(3)), false); + std::cout << " ; numext::isnan = "; + check((numext::isnan)(m(3)), false); + std::cout << "\n"; + std::cout << "allFinite: "; + check(m.allFinite(), 0); + std::cout << "\n"; + std::cout << "hasNaN: "; + check(m.hasNaN(), 0); + std::cout << "\n"; std::cout << "\n\n"; - } - else - { - VERIFY( (numext::isfinite)(m(3)) ); - VERIFY( !(numext::isinf)(m(3)) ); - VERIFY( !(numext::isnan)(m(3)) ); - VERIFY( !m.allFinite() ); - VERIFY( !m.hasNaN() ); + } else { + VERIFY((numext::isfinite)(m(3))); + VERIFY(!(numext::isinf)(m(3))); + VERIFY(!(numext::isnan)(m(3))); + VERIFY(!m.allFinite()); + VERIFY(!m.hasNaN()); } } -void test_fastmath() { - std::cout << "*** float *** \n\n"; check_inf_nan(true); - std::cout << "*** double ***\n\n"; check_inf_nan(true); - std::cout << "*** long double *** \n\n"; check_inf_nan(true); +void test_fastmath() +{ + std::cout << "*** float *** \n\n"; + check_inf_nan(true); + std::cout << "*** double ***\n\n"; + check_inf_nan(true); + std::cout << "*** long double *** \n\n"; + check_inf_nan(true); check_inf_nan(false); check_inf_nan(false); diff --git a/filmulator-gui/core/nlmeans/eigen/test/first_aligned.cpp b/filmulator-gui/core/nlmeans/eigen/test/first_aligned.cpp index ae2d4bc4..2e61a9eb 100644 --- a/filmulator-gui/core/nlmeans/eigen/test/first_aligned.cpp +++ b/filmulator-gui/core/nlmeans/eigen/test/first_aligned.cpp @@ -9,42 +9,43 @@ #include "main.h" -template -void test_first_aligned_helper(Scalar *array, int size) +template void test_first_aligned_helper(Scalar *array, int size) { const int packet_size = sizeof(Scalar) * internal::packet_traits::size; VERIFY(((size_t(array) + sizeof(Scalar) * internal::first_default_aligned(array, size)) % packet_size) == 0); } -template -void test_none_aligned_helper(Scalar *array, int size) +template void test_none_aligned_helper(Scalar *array, int size) { EIGEN_UNUSED_VARIABLE(array); EIGEN_UNUSED_VARIABLE(size); VERIFY(internal::packet_traits::size == 1 || internal::first_default_aligned(array, size) == size); } -struct some_non_vectorizable_type { float x; }; +struct some_non_vectorizable_type +{ + float x; +}; void test_first_aligned() { EIGEN_ALIGN16 float array_float[100]; test_first_aligned_helper(array_float, 50); - test_first_aligned_helper(array_float+1, 50); - test_first_aligned_helper(array_float+2, 50); - test_first_aligned_helper(array_float+3, 50); - test_first_aligned_helper(array_float+4, 50); - test_first_aligned_helper(array_float+5, 50); - + test_first_aligned_helper(array_float + 1, 50); + test_first_aligned_helper(array_float + 2, 50); + test_first_aligned_helper(array_float + 3, 50); + test_first_aligned_helper(array_float + 4, 50); + test_first_aligned_helper(array_float + 5, 50); + EIGEN_ALIGN16 double array_double[100]; test_first_aligned_helper(array_double, 50); - test_first_aligned_helper(array_double+1, 50); - test_first_aligned_helper(array_double+2, 50); - - double *array_double_plus_4_bytes = (double*)(internal::UIntPtr(array_double)+4); + test_first_aligned_helper(array_double + 1, 50); + test_first_aligned_helper(array_double + 2, 50); + + double *array_double_plus_4_bytes = (double *)(internal::UIntPtr(array_double) + 4); test_none_aligned_helper(array_double_plus_4_bytes, 50); - test_none_aligned_helper(array_double_plus_4_bytes+1, 50); - + test_none_aligned_helper(array_double_plus_4_bytes + 1, 50); + some_non_vectorizable_type array_nonvec[100]; test_first_aligned_helper(array_nonvec, 100); test_none_aligned_helper(array_nonvec, 100); diff --git a/filmulator-gui/core/nlmeans/eigen/test/geo_alignedbox.cpp b/filmulator-gui/core/nlmeans/eigen/test/geo_alignedbox.cpp index b64ea3bd..baa0268a 100644 --- a/filmulator-gui/core/nlmeans/eigen/test/geo_alignedbox.cpp +++ b/filmulator-gui/core/nlmeans/eigen/test/geo_alignedbox.cpp @@ -12,14 +12,13 @@ #include #include -#include +#include using namespace std; -template EIGEN_DONT_INLINE -void kill_extra_precision(T& x) { eigen_assert((void*)(&x) != (void*)0); } +template EIGEN_DONT_INLINE void kill_extra_precision(T &x) { eigen_assert((void *)(&x) != (void *)0); } -template void alignedbox(const BoxType& _box) +template void alignedbox(const BoxType &_box) { /* this test covers the following files: AlignedBox.h @@ -32,23 +31,22 @@ template void alignedbox(const BoxType& _box) VectorType p0 = VectorType::Random(dim); VectorType p1 = VectorType::Random(dim); - while( p1 == p0 ){ - p1 = VectorType::Random(dim); } - RealScalar s1 = internal::random(0,1); + while (p1 == p0) { p1 = VectorType::Random(dim); } + RealScalar s1 = internal::random(0, 1); BoxType b0(dim); - BoxType b1(VectorType::Random(dim),VectorType::Random(dim)); + BoxType b1(VectorType::Random(dim), VectorType::Random(dim)); BoxType b2; - + kill_extra_precision(b1); kill_extra_precision(p0); kill_extra_precision(p1); b0.extend(p0); b0.extend(p1); - VERIFY(b0.contains(p0*s1+(Scalar(1)-s1)*p1)); + VERIFY(b0.contains(p0 * s1 + (Scalar(1) - s1) * p1)); VERIFY(b0.contains(b0.center())); - VERIFY_IS_APPROX(b0.center(),(p0+p1)/Scalar(2)); + VERIFY_IS_APPROX(b0.center(), (p0 + p1) / Scalar(2)); (b2 = b0).extend(b1); VERIFY(b2.contains(b0)); @@ -61,7 +59,7 @@ template void alignedbox(const BoxType& _box) BoxType box2(VectorType::Random(dim)); box2.extend(VectorType::Random(dim)); - VERIFY(box1.intersects(box2) == !box1.intersection(box2).isEmpty()); + VERIFY(box1.intersects(box2) == !box1.intersection(box2).isEmpty()); // alignment -- make sure there is no memory alignment assertion BoxType *bp0 = new BoxType(dim); @@ -71,20 +69,16 @@ template void alignedbox(const BoxType& _box) delete bp1; // sampling - for( int i=0; i<10; ++i ) - { - VectorType r = b0.sample(); - VERIFY(b0.contains(r)); + for (int i = 0; i < 10; ++i) { + VectorType r = b0.sample(); + VERIFY(b0.contains(r)); } - } - -template -void alignedboxCastTests(const BoxType& _box) +template void alignedboxCastTests(const BoxType &_box) { - // casting + // casting typedef typename BoxType::Scalar Scalar; typedef Matrix VectorType; @@ -100,88 +94,95 @@ void alignedboxCastTests(const BoxType& _box) const int Dim = BoxType::AmbientDimAtCompileTime; typedef typename GetDifferentType::type OtherScalar; - AlignedBox hp1f = b0.template cast(); - VERIFY_IS_APPROX(hp1f.template cast(),b0); - AlignedBox hp1d = b0.template cast(); - VERIFY_IS_APPROX(hp1d.template cast(),b0); + AlignedBox hp1f = b0.template cast(); + VERIFY_IS_APPROX(hp1f.template cast(), b0); + AlignedBox hp1d = b0.template cast(); + VERIFY_IS_APPROX(hp1d.template cast(), b0); } void specificTest1() { - Vector2f m; m << -1.0f, -2.0f; - Vector2f M; M << 1.0f, 5.0f; - - typedef AlignedBox2f BoxType; - BoxType box( m, M ); - - Vector2f sides = M-m; - VERIFY_IS_APPROX(sides, box.sizes() ); - VERIFY_IS_APPROX(sides[1], box.sizes()[1] ); - VERIFY_IS_APPROX(sides[1], box.sizes().maxCoeff() ); - VERIFY_IS_APPROX(sides[0], box.sizes().minCoeff() ); - - VERIFY_IS_APPROX( 14.0f, box.volume() ); - VERIFY_IS_APPROX( 53.0f, box.diagonal().squaredNorm() ); - VERIFY_IS_APPROX( std::sqrt( 53.0f ), box.diagonal().norm() ); - - VERIFY_IS_APPROX( m, box.corner( BoxType::BottomLeft ) ); - VERIFY_IS_APPROX( M, box.corner( BoxType::TopRight ) ); - Vector2f bottomRight; bottomRight << M[0], m[1]; - Vector2f topLeft; topLeft << m[0], M[1]; - VERIFY_IS_APPROX( bottomRight, box.corner( BoxType::BottomRight ) ); - VERIFY_IS_APPROX( topLeft, box.corner( BoxType::TopLeft ) ); + Vector2f m; + m << -1.0f, -2.0f; + Vector2f M; + M << 1.0f, 5.0f; + + typedef AlignedBox2f BoxType; + BoxType box(m, M); + + Vector2f sides = M - m; + VERIFY_IS_APPROX(sides, box.sizes()); + VERIFY_IS_APPROX(sides[1], box.sizes()[1]); + VERIFY_IS_APPROX(sides[1], box.sizes().maxCoeff()); + VERIFY_IS_APPROX(sides[0], box.sizes().minCoeff()); + + VERIFY_IS_APPROX(14.0f, box.volume()); + VERIFY_IS_APPROX(53.0f, box.diagonal().squaredNorm()); + VERIFY_IS_APPROX(std::sqrt(53.0f), box.diagonal().norm()); + + VERIFY_IS_APPROX(m, box.corner(BoxType::BottomLeft)); + VERIFY_IS_APPROX(M, box.corner(BoxType::TopRight)); + Vector2f bottomRight; + bottomRight << M[0], m[1]; + Vector2f topLeft; + topLeft << m[0], M[1]; + VERIFY_IS_APPROX(bottomRight, box.corner(BoxType::BottomRight)); + VERIFY_IS_APPROX(topLeft, box.corner(BoxType::TopLeft)); } void specificTest2() { - Vector3i m; m << -1, -2, 0; - Vector3i M; M << 1, 5, 3; - - typedef AlignedBox3i BoxType; - BoxType box( m, M ); - - Vector3i sides = M-m; - VERIFY_IS_APPROX(sides, box.sizes() ); - VERIFY_IS_APPROX(sides[1], box.sizes()[1] ); - VERIFY_IS_APPROX(sides[1], box.sizes().maxCoeff() ); - VERIFY_IS_APPROX(sides[0], box.sizes().minCoeff() ); - - VERIFY_IS_APPROX( 42, box.volume() ); - VERIFY_IS_APPROX( 62, box.diagonal().squaredNorm() ); - - VERIFY_IS_APPROX( m, box.corner( BoxType::BottomLeftFloor ) ); - VERIFY_IS_APPROX( M, box.corner( BoxType::TopRightCeil ) ); - Vector3i bottomRightFloor; bottomRightFloor << M[0], m[1], m[2]; - Vector3i topLeftFloor; topLeftFloor << m[0], M[1], m[2]; - VERIFY_IS_APPROX( bottomRightFloor, box.corner( BoxType::BottomRightFloor ) ); - VERIFY_IS_APPROX( topLeftFloor, box.corner( BoxType::TopLeftFloor ) ); + Vector3i m; + m << -1, -2, 0; + Vector3i M; + M << 1, 5, 3; + + typedef AlignedBox3i BoxType; + BoxType box(m, M); + + Vector3i sides = M - m; + VERIFY_IS_APPROX(sides, box.sizes()); + VERIFY_IS_APPROX(sides[1], box.sizes()[1]); + VERIFY_IS_APPROX(sides[1], box.sizes().maxCoeff()); + VERIFY_IS_APPROX(sides[0], box.sizes().minCoeff()); + + VERIFY_IS_APPROX(42, box.volume()); + VERIFY_IS_APPROX(62, box.diagonal().squaredNorm()); + + VERIFY_IS_APPROX(m, box.corner(BoxType::BottomLeftFloor)); + VERIFY_IS_APPROX(M, box.corner(BoxType::TopRightCeil)); + Vector3i bottomRightFloor; + bottomRightFloor << M[0], m[1], m[2]; + Vector3i topLeftFloor; + topLeftFloor << m[0], M[1], m[2]; + VERIFY_IS_APPROX(bottomRightFloor, box.corner(BoxType::BottomRightFloor)); + VERIFY_IS_APPROX(topLeftFloor, box.corner(BoxType::TopLeftFloor)); } void test_geo_alignedbox() { - for(int i = 0; i < g_repeat; i++) - { - CALL_SUBTEST_1( alignedbox(AlignedBox2f()) ); - CALL_SUBTEST_2( alignedboxCastTests(AlignedBox2f()) ); + for (int i = 0; i < g_repeat; i++) { + CALL_SUBTEST_1(alignedbox(AlignedBox2f())); + CALL_SUBTEST_2(alignedboxCastTests(AlignedBox2f())); - CALL_SUBTEST_3( alignedbox(AlignedBox3f()) ); - CALL_SUBTEST_4( alignedboxCastTests(AlignedBox3f()) ); + CALL_SUBTEST_3(alignedbox(AlignedBox3f())); + CALL_SUBTEST_4(alignedboxCastTests(AlignedBox3f())); - CALL_SUBTEST_5( alignedbox(AlignedBox4d()) ); - CALL_SUBTEST_6( alignedboxCastTests(AlignedBox4d()) ); + CALL_SUBTEST_5(alignedbox(AlignedBox4d())); + CALL_SUBTEST_6(alignedboxCastTests(AlignedBox4d())); - CALL_SUBTEST_7( alignedbox(AlignedBox1d()) ); - CALL_SUBTEST_8( alignedboxCastTests(AlignedBox1d()) ); + CALL_SUBTEST_7(alignedbox(AlignedBox1d())); + CALL_SUBTEST_8(alignedboxCastTests(AlignedBox1d())); - CALL_SUBTEST_9( alignedbox(AlignedBox1i()) ); - CALL_SUBTEST_10( alignedbox(AlignedBox2i()) ); - CALL_SUBTEST_11( alignedbox(AlignedBox3i()) ); + CALL_SUBTEST_9(alignedbox(AlignedBox1i())); + CALL_SUBTEST_10(alignedbox(AlignedBox2i())); + CALL_SUBTEST_11(alignedbox(AlignedBox3i())); - CALL_SUBTEST_14( alignedbox(AlignedBox(4)) ); + CALL_SUBTEST_14(alignedbox(AlignedBox(4))); } - CALL_SUBTEST_12( specificTest1() ); - CALL_SUBTEST_13( specificTest2() ); + CALL_SUBTEST_12(specificTest1()); + CALL_SUBTEST_13(specificTest2()); } diff --git a/filmulator-gui/core/nlmeans/eigen/test/geo_eulerangles.cpp b/filmulator-gui/core/nlmeans/eigen/test/geo_eulerangles.cpp index 932ebe77..d2811eed 100644 --- a/filmulator-gui/core/nlmeans/eigen/test/geo_eulerangles.cpp +++ b/filmulator-gui/core/nlmeans/eigen/test/geo_eulerangles.cpp @@ -13,22 +13,24 @@ #include -template -void verify_euler(const Matrix& ea, int i, int j, int k) +template void verify_euler(const Matrix &ea, int i, int j, int k) { - typedef Matrix Matrix3; - typedef Matrix Vector3; + typedef Matrix Matrix3; + typedef Matrix Vector3; typedef AngleAxis AngleAxisx; using std::abs; - Matrix3 m(AngleAxisx(ea[0], Vector3::Unit(i)) * AngleAxisx(ea[1], Vector3::Unit(j)) * AngleAxisx(ea[2], Vector3::Unit(k))); + Matrix3 m( + AngleAxisx(ea[0], Vector3::Unit(i)) * AngleAxisx(ea[1], Vector3::Unit(j)) * AngleAxisx(ea[2], Vector3::Unit(k))); Vector3 eabis = m.eulerAngles(i, j, k); - Matrix3 mbis(AngleAxisx(eabis[0], Vector3::Unit(i)) * AngleAxisx(eabis[1], Vector3::Unit(j)) * AngleAxisx(eabis[2], Vector3::Unit(k))); - VERIFY_IS_APPROX(m, mbis); - /* If I==K, and ea[1]==0, then there no unique solution. */ - /* The remark apply in the case where I!=K, and |ea[1]| is close to pi/2. */ - if( (i!=k || ea[1]!=0) && (i==k || !internal::isApprox(abs(ea[1]),Scalar(EIGEN_PI/2),test_precision())) ) - VERIFY((ea-eabis).norm() <= test_precision()); - + Matrix3 mbis(AngleAxisx(eabis[0], Vector3::Unit(i)) * AngleAxisx(eabis[1], Vector3::Unit(j)) + * AngleAxisx(eabis[2], Vector3::Unit(k))); + VERIFY_IS_APPROX(m, mbis); + /* If I==K, and ea[1]==0, then there no unique solution. */ + /* The remark apply in the case where I!=K, and |ea[1]| is close to pi/2. */ + if ((i != k || ea[1] != 0) + && (i == k || !internal::isApprox(abs(ea[1]), Scalar(EIGEN_PI / 2), test_precision()))) + VERIFY((ea - eabis).norm() <= test_precision()); + // approx_or_less_than does not work for 0 VERIFY(0 < eabis[0] || test_isMuchSmallerThan(eabis[0], Scalar(1))); VERIFY_IS_APPROX_OR_LESS_THAN(eabis[0], Scalar(EIGEN_PI)); @@ -38,29 +40,29 @@ void verify_euler(const Matrix& ea, int i, int j, int k) VERIFY_IS_APPROX_OR_LESS_THAN(eabis[2], Scalar(EIGEN_PI)); } -template void check_all_var(const Matrix& ea) +template void check_all_var(const Matrix &ea) { - verify_euler(ea, 0,1,2); - verify_euler(ea, 0,1,0); - verify_euler(ea, 0,2,1); - verify_euler(ea, 0,2,0); - - verify_euler(ea, 1,2,0); - verify_euler(ea, 1,2,1); - verify_euler(ea, 1,0,2); - verify_euler(ea, 1,0,1); - - verify_euler(ea, 2,0,1); - verify_euler(ea, 2,0,2); - verify_euler(ea, 2,1,0); - verify_euler(ea, 2,1,2); + verify_euler(ea, 0, 1, 2); + verify_euler(ea, 0, 1, 0); + verify_euler(ea, 0, 2, 1); + verify_euler(ea, 0, 2, 0); + + verify_euler(ea, 1, 2, 0); + verify_euler(ea, 1, 2, 1); + verify_euler(ea, 1, 0, 2); + verify_euler(ea, 1, 0, 1); + + verify_euler(ea, 2, 0, 1); + verify_euler(ea, 2, 0, 2); + verify_euler(ea, 2, 1, 0); + verify_euler(ea, 2, 1, 2); } template void eulerangles() { - typedef Matrix Matrix3; - typedef Matrix Vector3; - typedef Array Array3; + typedef Matrix Matrix3; + typedef Matrix Vector3; + typedef Array Array3; typedef Quaternion Quaternionx; typedef AngleAxis AngleAxisx; @@ -69,44 +71,44 @@ template void eulerangles() q1 = AngleAxisx(a, Vector3::Random().normalized()); Matrix3 m; m = q1; - - Vector3 ea = m.eulerAngles(0,1,2); + + Vector3 ea = m.eulerAngles(0, 1, 2); check_all_var(ea); - ea = m.eulerAngles(0,1,0); + ea = m.eulerAngles(0, 1, 0); check_all_var(ea); - + // Check with purely random Quaternion: q1.coeffs() = Quaternionx::Coefficients::Random().normalized(); m = q1; - ea = m.eulerAngles(0,1,2); + ea = m.eulerAngles(0, 1, 2); check_all_var(ea); - ea = m.eulerAngles(0,1,0); + ea = m.eulerAngles(0, 1, 0); check_all_var(ea); - + // Check with random angles in range [0:pi]x[-pi:pi]x[-pi:pi]. - ea = (Array3::Random() + Array3(1,0,0))*Scalar(EIGEN_PI)*Array3(0.5,1,1); + ea = (Array3::Random() + Array3(1, 0, 0)) * Scalar(EIGEN_PI) * Array3(0.5, 1, 1); check_all_var(ea); - - ea[2] = ea[0] = internal::random(0,Scalar(EIGEN_PI)); + + ea[2] = ea[0] = internal::random(0, Scalar(EIGEN_PI)); check_all_var(ea); - - ea[0] = ea[1] = internal::random(0,Scalar(EIGEN_PI)); + + ea[0] = ea[1] = internal::random(0, Scalar(EIGEN_PI)); check_all_var(ea); - + ea[1] = 0; check_all_var(ea); - + ea.head(2).setZero(); check_all_var(ea); - + ea.setZero(); check_all_var(ea); } void test_geo_eulerangles() { - for(int i = 0; i < g_repeat; i++) { - CALL_SUBTEST_1( eulerangles() ); - CALL_SUBTEST_2( eulerangles() ); + for (int i = 0; i < g_repeat; i++) { + CALL_SUBTEST_1(eulerangles()); + CALL_SUBTEST_2(eulerangles()); } } diff --git a/filmulator-gui/core/nlmeans/eigen/test/geo_homogeneous.cpp b/filmulator-gui/core/nlmeans/eigen/test/geo_homogeneous.cpp index 2187c7bf..256a93c4 100644 --- a/filmulator-gui/core/nlmeans/eigen/test/geo_homogeneous.cpp +++ b/filmulator-gui/core/nlmeans/eigen/test/geo_homogeneous.cpp @@ -10,24 +10,23 @@ #include "main.h" #include -template void homogeneous(void) +template void homogeneous(void) { /* this test covers the following files: Homogeneous.h */ - typedef Matrix MatrixType; - typedef Matrix VectorType; + typedef Matrix MatrixType; + typedef Matrix VectorType; - typedef Matrix HMatrixType; - typedef Matrix HVectorType; + typedef Matrix HMatrixType; + typedef Matrix HVectorType; - typedef Matrix T1MatrixType; - typedef Matrix T2MatrixType; - typedef Matrix T3MatrixType; + typedef Matrix T1MatrixType; + typedef Matrix T2MatrixType; + typedef Matrix T3MatrixType; - VectorType v0 = VectorType::Random(), - ones = VectorType::Ones(); + VectorType v0 = VectorType::Random(), ones = VectorType::Ones(); HVectorType hv0 = HVectorType::Random(); @@ -38,7 +37,7 @@ template void homogeneous(void) hv0 << v0, 1; VERIFY_IS_APPROX(v0.homogeneous(), hv0); VERIFY_IS_APPROX(v0, hv0.hnormalized()); - + VERIFY_IS_APPROX(v0.homogeneous().sum(), hv0.sum()); VERIFY_IS_APPROX(v0.homogeneous().minCoeff(), hv0.minCoeff()); VERIFY_IS_APPROX(v0.homogeneous().maxCoeff(), hv0.maxCoeff()); @@ -46,9 +45,8 @@ template void homogeneous(void) hm0 << m0, ones.transpose(); VERIFY_IS_APPROX(m0.colwise().homogeneous(), hm0); VERIFY_IS_APPROX(m0, hm0.colwise().hnormalized()); - hm0.row(Size-1).setRandom(); - for(int j=0; j void homogeneous(void) VERIFY_IS_APPROX(t2 * (v0.homogeneous().asDiagonal()), t2 * hv0.asDiagonal()); VERIFY_IS_APPROX((v0.homogeneous().asDiagonal()) * t2, hv0.asDiagonal() * t2); - VERIFY_IS_APPROX((v0.transpose().rowwise().homogeneous().eval()) * t2, - v0.transpose().rowwise().homogeneous() * t2); - VERIFY_IS_APPROX((m0.transpose().rowwise().homogeneous().eval()) * t2, - m0.transpose().rowwise().homogeneous() * t2); + VERIFY_IS_APPROX((v0.transpose().rowwise().homogeneous().eval()) * t2, v0.transpose().rowwise().homogeneous() * t2); + VERIFY_IS_APPROX((m0.transpose().rowwise().homogeneous().eval()) * t2, m0.transpose().rowwise().homogeneous() * t2); T3MatrixType t3 = T3MatrixType::Random(); - VERIFY_IS_APPROX((v0.transpose().rowwise().homogeneous().eval()) * t3, - v0.transpose().rowwise().homogeneous() * t3); - VERIFY_IS_APPROX((m0.transpose().rowwise().homogeneous().eval()) * t3, - m0.transpose().rowwise().homogeneous() * t3); + VERIFY_IS_APPROX((v0.transpose().rowwise().homogeneous().eval()) * t3, v0.transpose().rowwise().homogeneous() * t3); + VERIFY_IS_APPROX((m0.transpose().rowwise().homogeneous().eval()) * t3, m0.transpose().rowwise().homogeneous() * t3); // test product with a Transform object Transform aff; Transform caff; Transform proj; - Matrix pts; - Matrix pts1, pts2; + Matrix pts; + Matrix pts1, pts2; aff.affine().setRandom(); proj = caff = aff; - pts.setRandom(Size,internal::random(1,20)); - + pts.setRandom(Size, internal::random(1, 20)); + pts1 = pts.colwise().homogeneous(); - VERIFY_IS_APPROX(aff * pts.colwise().homogeneous(), (aff * pts1).colwise().hnormalized()); + VERIFY_IS_APPROX(aff * pts.colwise().homogeneous(), (aff * pts1).colwise().hnormalized()); VERIFY_IS_APPROX(caff * pts.colwise().homogeneous(), (caff * pts1).colwise().hnormalized()); VERIFY_IS_APPROX(proj * pts.colwise().homogeneous(), (proj * pts1)); - VERIFY_IS_APPROX((aff * pts1).colwise().hnormalized(), aff * pts); + VERIFY_IS_APPROX((aff * pts1).colwise().hnormalized(), aff * pts); VERIFY_IS_APPROX((caff * pts1).colwise().hnormalized(), caff * pts); - + pts2 = pts1; pts2.row(Size).setRandom(); - VERIFY_IS_APPROX((aff * pts2).colwise().hnormalized(), aff * pts2.colwise().hnormalized()); + VERIFY_IS_APPROX((aff * pts2).colwise().hnormalized(), aff * pts2.colwise().hnormalized()); VERIFY_IS_APPROX((caff * pts2).colwise().hnormalized(), caff * pts2.colwise().hnormalized()); - VERIFY_IS_APPROX((proj * pts2).colwise().hnormalized(), (proj * pts2.colwise().hnormalized().colwise().homogeneous()).colwise().hnormalized()); - + VERIFY_IS_APPROX((proj * pts2).colwise().hnormalized(), + (proj * pts2.colwise().hnormalized().colwise().homogeneous()).colwise().hnormalized()); + // Test combination of homogeneous - - VERIFY_IS_APPROX( (t2 * v0.homogeneous()).hnormalized(), - (t2.template topLeftCorner() * v0 + t2.template topRightCorner()) - / ((t2.template bottomLeftCorner<1,Size>()*v0).value() + t2(Size,Size)) ); - - VERIFY_IS_APPROX( (t2 * pts.colwise().homogeneous()).colwise().hnormalized(), - (Matrix(t2 * pts1).colwise().hnormalized()) ); - - VERIFY_IS_APPROX( (t2 .lazyProduct( v0.homogeneous() )).hnormalized(), (t2 * v0.homogeneous()).hnormalized() ); - VERIFY_IS_APPROX( (t2 .lazyProduct ( pts.colwise().homogeneous() )).colwise().hnormalized(), (t2 * pts1).colwise().hnormalized() ); - - VERIFY_IS_APPROX( (v0.transpose().homogeneous() .lazyProduct( t2 )).hnormalized(), (v0.transpose().homogeneous()*t2).hnormalized() ); - VERIFY_IS_APPROX( (pts.transpose().rowwise().homogeneous() .lazyProduct( t2 )).rowwise().hnormalized(), (pts1.transpose()*t2).rowwise().hnormalized() ); - - VERIFY_IS_APPROX( (t2.template triangularView() * v0.homogeneous()).eval(), (t2.template triangularView()*hv0) ); + + VERIFY_IS_APPROX((t2 * v0.homogeneous()).hnormalized(), + (t2.template topLeftCorner() * v0 + t2.template topRightCorner()) + / ((t2.template bottomLeftCorner<1, Size>() * v0).value() + t2(Size, Size))); + + VERIFY_IS_APPROX((t2 * pts.colwise().homogeneous()).colwise().hnormalized(), + (Matrix(t2 * pts1).colwise().hnormalized())); + + VERIFY_IS_APPROX((t2.lazyProduct(v0.homogeneous())).hnormalized(), (t2 * v0.homogeneous()).hnormalized()); + VERIFY_IS_APPROX( + (t2.lazyProduct(pts.colwise().homogeneous())).colwise().hnormalized(), (t2 * pts1).colwise().hnormalized()); + + VERIFY_IS_APPROX( + (v0.transpose().homogeneous().lazyProduct(t2)).hnormalized(), (v0.transpose().homogeneous() * t2).hnormalized()); + VERIFY_IS_APPROX((pts.transpose().rowwise().homogeneous().lazyProduct(t2)).rowwise().hnormalized(), + (pts1.transpose() * t2).rowwise().hnormalized()); + + VERIFY_IS_APPROX( + (t2.template triangularView() * v0.homogeneous()).eval(), (t2.template triangularView() * hv0)); } void test_geo_homogeneous() { - for(int i = 0; i < g_repeat; i++) { - CALL_SUBTEST_1(( homogeneous() )); - CALL_SUBTEST_2(( homogeneous() )); - CALL_SUBTEST_3(( homogeneous() )); + for (int i = 0; i < g_repeat; i++) { + CALL_SUBTEST_1((homogeneous())); + CALL_SUBTEST_2((homogeneous())); + CALL_SUBTEST_3((homogeneous())); } } diff --git a/filmulator-gui/core/nlmeans/eigen/test/geo_hyperplane.cpp b/filmulator-gui/core/nlmeans/eigen/test/geo_hyperplane.cpp index b3a48c58..674eb8a3 100644 --- a/filmulator-gui/core/nlmeans/eigen/test/geo_hyperplane.cpp +++ b/filmulator-gui/core/nlmeans/eigen/test/geo_hyperplane.cpp @@ -13,7 +13,7 @@ #include #include -template void hyperplane(const HyperplaneType& _plane) +template void hyperplane(const HyperplaneType &_plane) { /* this test covers the following files: Hyperplane.h @@ -24,8 +24,7 @@ template void hyperplane(const HyperplaneType& _plane) typedef typename HyperplaneType::Scalar Scalar; typedef typename HyperplaneType::RealScalar RealScalar; typedef Matrix VectorType; - typedef Matrix MatrixType; + typedef Matrix MatrixType; VectorType p0 = VectorType::Random(dim); VectorType p1 = VectorType::Random(dim); @@ -40,49 +39,48 @@ template void hyperplane(const HyperplaneType& _plane) Scalar s0 = internal::random(); Scalar s1 = internal::random(); - VERIFY_IS_APPROX( n1.dot(n1), Scalar(1) ); + VERIFY_IS_APPROX(n1.dot(n1), Scalar(1)); - VERIFY_IS_MUCH_SMALLER_THAN( pl0.absDistance(p0), Scalar(1) ); - if(numext::abs2(s0)>RealScalar(1e-6)) - VERIFY_IS_APPROX( pl1.signedDistance(p1 + n1 * s0), s0); + VERIFY_IS_MUCH_SMALLER_THAN(pl0.absDistance(p0), Scalar(1)); + if (numext::abs2(s0) > RealScalar(1e-6)) + VERIFY_IS_APPROX(pl1.signedDistance(p1 + n1 * s0), s0); else - VERIFY_IS_MUCH_SMALLER_THAN( abs(pl1.signedDistance(p1 + n1 * s0) - s0), Scalar(1) ); - VERIFY_IS_MUCH_SMALLER_THAN( pl1.signedDistance(pl1.projection(p0)), Scalar(1) ); - VERIFY_IS_MUCH_SMALLER_THAN( pl1.absDistance(p1 + pl1.normal().unitOrthogonal() * s1), Scalar(1) ); + VERIFY_IS_MUCH_SMALLER_THAN(abs(pl1.signedDistance(p1 + n1 * s0) - s0), Scalar(1)); + VERIFY_IS_MUCH_SMALLER_THAN(pl1.signedDistance(pl1.projection(p0)), Scalar(1)); + VERIFY_IS_MUCH_SMALLER_THAN(pl1.absDistance(p1 + pl1.normal().unitOrthogonal() * s1), Scalar(1)); // transform - if (!NumTraits::IsComplex) - { - MatrixType rot = MatrixType::Random(dim,dim).householderQr().householderQ(); - DiagonalMatrix scaling(VectorType::Random()); - Translation translation(VectorType::Random()); - - while(scaling.diagonal().cwiseAbs().minCoeff()::IsComplex) { + MatrixType rot = MatrixType::Random(dim, dim).householderQr().householderQ(); + DiagonalMatrix scaling(VectorType::Random()); + Translation translation(VectorType::Random()); + + while (scaling.diagonal().cwiseAbs().minCoeff() < RealScalar(1e-4)) scaling.diagonal() = VectorType::Random(); pl2 = pl1; - VERIFY_IS_MUCH_SMALLER_THAN( pl2.transform(rot).absDistance(rot * p1), Scalar(1) ); + VERIFY_IS_MUCH_SMALLER_THAN(pl2.transform(rot).absDistance(rot * p1), Scalar(1)); pl2 = pl1; - VERIFY_IS_MUCH_SMALLER_THAN( pl2.transform(rot,Isometry).absDistance(rot * p1), Scalar(1) ); + VERIFY_IS_MUCH_SMALLER_THAN(pl2.transform(rot, Isometry).absDistance(rot * p1), Scalar(1)); pl2 = pl1; - VERIFY_IS_MUCH_SMALLER_THAN( pl2.transform(rot*scaling).absDistance((rot*scaling) * p1), Scalar(1) ); - VERIFY_IS_APPROX( pl2.normal().norm(), RealScalar(1) ); + VERIFY_IS_MUCH_SMALLER_THAN(pl2.transform(rot * scaling).absDistance((rot * scaling) * p1), Scalar(1)); + VERIFY_IS_APPROX(pl2.normal().norm(), RealScalar(1)); pl2 = pl1; - VERIFY_IS_MUCH_SMALLER_THAN( pl2.transform(rot*scaling*translation) - .absDistance((rot*scaling*translation) * p1), Scalar(1) ); - VERIFY_IS_APPROX( pl2.normal().norm(), RealScalar(1) ); + VERIFY_IS_MUCH_SMALLER_THAN( + pl2.transform(rot * scaling * translation).absDistance((rot * scaling * translation) * p1), Scalar(1)); + VERIFY_IS_APPROX(pl2.normal().norm(), RealScalar(1)); pl2 = pl1; - VERIFY_IS_MUCH_SMALLER_THAN( pl2.transform(rot*translation,Isometry) - .absDistance((rot*translation) * p1), Scalar(1) ); - VERIFY_IS_APPROX( pl2.normal().norm(), RealScalar(1) ); + VERIFY_IS_MUCH_SMALLER_THAN( + pl2.transform(rot * translation, Isometry).absDistance((rot * translation) * p1), Scalar(1)); + VERIFY_IS_APPROX(pl2.normal().norm(), RealScalar(1)); } // casting const int Dim = HyperplaneType::AmbientDimAtCompileTime; typedef typename GetDifferentType::type OtherScalar; - Hyperplane hp1f = pl1.template cast(); - VERIFY_IS_APPROX(hp1f.template cast(),pl1); - Hyperplane hp1d = pl1.template cast(); - VERIFY_IS_APPROX(hp1d.template cast(),pl1); + Hyperplane hp1f = pl1.template cast(); + VERIFY_IS_APPROX(hp1f.template cast(), pl1); + Hyperplane hp1d = pl1.template cast(); + VERIFY_IS_APPROX(hp1d.template cast(), pl1); } template void lines() @@ -90,21 +88,20 @@ template void lines() using std::abs; typedef Hyperplane HLine; typedef ParametrizedLine PLine; - typedef Matrix Vector; - typedef Matrix CoeffsType; + typedef Matrix Vector; + typedef Matrix CoeffsType; - for(int i = 0; i < 10; i++) - { + for (int i = 0; i < 10; i++) { Vector center = Vector::Random(); Vector u = Vector::Random(); Vector v = Vector::Random(); Scalar a = internal::random(); - while (abs(a-1) < Scalar(1e-4)) a = internal::random(); + while (abs(a - 1) < Scalar(1e-4)) a = internal::random(); while (u.norm() < Scalar(1e-4)) u = Vector::Random(); while (v.norm() < Scalar(1e-4)) v = Vector::Random(); - HLine line_u = HLine::Through(center + u, center + a*u); - HLine line_v = HLine::Through(center + v, center + a*v); + HLine line_u = HLine::Through(center + u, center + a * u); + HLine line_v = HLine::Through(center + v, center + a * v); // the line equations should be normalized so that a^2+b^2=1 VERIFY_IS_APPROX(line_u.normal().norm(), Scalar(1)); @@ -113,15 +110,14 @@ template void lines() Vector result = line_u.intersection(line_v); // the lines should intersect at the point we called "center" - if(abs(a-1) > Scalar(1e-2) && abs(v.normalized().dot(u.normalized())) Scalar(1e-2) && abs(v.normalized().dot(u.normalized())) < Scalar(0.9)) VERIFY_IS_APPROX(result, center); // check conversions between two types of lines - PLine pl(line_u); // gcc 3.3 will commit suicide if we don't name this variable + PLine pl(line_u);// gcc 3.3 will commit suicide if we don't name this variable HLine line_u2(pl); CoeffsType converted_coeffs = line_u2.coeffs(); - if(line_u2.normal().dot(line_u.normal()) void planes() { using std::abs; typedef Hyperplane Plane; - typedef Matrix Vector; + typedef Matrix Vector; - for(int i = 0; i < 10; i++) - { + for (int i = 0; i < 10; i++) { Vector v0 = Vector::Random(); Vector v1(v0), v2(v0); - if(internal::random(0,1)>0.25) - v1 += Vector::Random(); - if(internal::random(0,1)>0.25) - v2 += v1 * std::pow(internal::random(0,1),internal::random(1,16)); - if(internal::random(0,1)>0.25) - v2 += Vector::Random() * std::pow(internal::random(0,1),internal::random(1,16)); + if (internal::random(0, 1) > 0.25) v1 += Vector::Random(); + if (internal::random(0, 1) > 0.25) + v2 += v1 * std::pow(internal::random(0, 1), internal::random(1, 16)); + if (internal::random(0, 1) > 0.25) + v2 += Vector::Random() * std::pow(internal::random(0, 1), internal::random(1, 16)); Plane p0 = Plane::Through(v0, v1, v2); @@ -154,44 +148,44 @@ template void planes() template void hyperplane_alignment() { - typedef Hyperplane Plane3a; - typedef Hyperplane Plane3u; + typedef Hyperplane Plane3a; + typedef Hyperplane Plane3u; EIGEN_ALIGN_MAX Scalar array1[4]; EIGEN_ALIGN_MAX Scalar array2[4]; - EIGEN_ALIGN_MAX Scalar array3[4+1]; - Scalar* array3u = array3+1; + EIGEN_ALIGN_MAX Scalar array3[4 + 1]; + Scalar *array3u = array3 + 1; + + Plane3a *p1 = ::new (reinterpret_cast(array1)) Plane3a; + Plane3u *p2 = ::new (reinterpret_cast(array2)) Plane3u; + Plane3u *p3 = ::new (reinterpret_cast(array3u)) Plane3u; - Plane3a *p1 = ::new(reinterpret_cast(array1)) Plane3a; - Plane3u *p2 = ::new(reinterpret_cast(array2)) Plane3u; - Plane3u *p3 = ::new(reinterpret_cast(array3u)) Plane3u; - p1->coeffs().setRandom(); *p2 = *p1; *p3 = *p1; VERIFY_IS_APPROX(p1->coeffs(), p2->coeffs()); VERIFY_IS_APPROX(p1->coeffs(), p3->coeffs()); - - #if defined(EIGEN_VECTORIZE) && EIGEN_MAX_STATIC_ALIGN_BYTES > 0 - if(internal::packet_traits::Vectorizable && internal::packet_traits::size<=4) - VERIFY_RAISES_ASSERT((::new(reinterpret_cast(array3u)) Plane3a)); - #endif + +#if defined(EIGEN_VECTORIZE) && EIGEN_MAX_STATIC_ALIGN_BYTES > 0 + if (internal::packet_traits::Vectorizable && internal::packet_traits::size <= 4) + VERIFY_RAISES_ASSERT((::new (reinterpret_cast(array3u)) Plane3a)); +#endif } void test_geo_hyperplane() { - for(int i = 0; i < g_repeat; i++) { - CALL_SUBTEST_1( hyperplane(Hyperplane()) ); - CALL_SUBTEST_2( hyperplane(Hyperplane()) ); - CALL_SUBTEST_2( hyperplane(Hyperplane()) ); - CALL_SUBTEST_2( hyperplane_alignment() ); - CALL_SUBTEST_3( hyperplane(Hyperplane()) ); - CALL_SUBTEST_4( hyperplane(Hyperplane,5>()) ); - CALL_SUBTEST_1( lines() ); - CALL_SUBTEST_3( lines() ); - CALL_SUBTEST_2( planes() ); - CALL_SUBTEST_5( planes() ); + for (int i = 0; i < g_repeat; i++) { + CALL_SUBTEST_1(hyperplane(Hyperplane())); + CALL_SUBTEST_2(hyperplane(Hyperplane())); + CALL_SUBTEST_2(hyperplane(Hyperplane())); + CALL_SUBTEST_2(hyperplane_alignment()); + CALL_SUBTEST_3(hyperplane(Hyperplane())); + CALL_SUBTEST_4(hyperplane(Hyperplane, 5>())); + CALL_SUBTEST_1(lines()); + CALL_SUBTEST_3(lines()); + CALL_SUBTEST_2(planes()); + CALL_SUBTEST_5(planes()); } } diff --git a/filmulator-gui/core/nlmeans/eigen/test/geo_orthomethods.cpp b/filmulator-gui/core/nlmeans/eigen/test/geo_orthomethods.cpp index e178df25..1efe04b3 100644 --- a/filmulator-gui/core/nlmeans/eigen/test/geo_orthomethods.cpp +++ b/filmulator-gui/core/nlmeans/eigen/test/geo_orthomethods.cpp @@ -19,14 +19,12 @@ template void orthomethods_3() { typedef typename NumTraits::Real RealScalar; - typedef Matrix Matrix3; - typedef Matrix Vector3; + typedef Matrix Matrix3; + typedef Matrix Vector3; - typedef Matrix Vector4; + typedef Matrix Vector4; - Vector3 v0 = Vector3::Random(), - v1 = Vector3::Random(), - v2 = Vector3::Random(); + Vector3 v0 = Vector3::Random(), v1 = Vector3::Random(), v2 = Vector3::Random(); // cross product VERIFY_IS_MUCH_SMALLER_THAN(v1.cross(v2).dot(v1), Scalar(1)); @@ -35,41 +33,38 @@ template void orthomethods_3() VERIFY_IS_MUCH_SMALLER_THAN(v2.dot(v1.cross(v2)), Scalar(1)); VERIFY_IS_MUCH_SMALLER_THAN(v1.cross(Vector3::Random()).dot(v1), Scalar(1)); Matrix3 mat3; - mat3 << v0.normalized(), - (v0.cross(v1)).normalized(), - (v0.cross(v1).cross(v0)).normalized(); + mat3 << v0.normalized(), (v0.cross(v1)).normalized(), (v0.cross(v1).cross(v0)).normalized(); VERIFY(mat3.isUnitary()); - + mat3.setRandom(); - VERIFY_IS_APPROX(v0.cross(mat3*v1), -(mat3*v1).cross(v0)); + VERIFY_IS_APPROX(v0.cross(mat3 * v1), -(mat3 * v1).cross(v0)); VERIFY_IS_APPROX(v0.cross(mat3.lazyProduct(v1)), -(mat3.lazyProduct(v1)).cross(v0)); // colwise/rowwise cross product mat3.setRandom(); Vector3 vec3 = Vector3::Random(); Matrix3 mcross; - int i = internal::random(0,2); + int i = internal::random(0, 2); mcross = mat3.colwise().cross(vec3); VERIFY_IS_APPROX(mcross.col(i), mat3.col(i).cross(vec3)); - + VERIFY_IS_MUCH_SMALLER_THAN((mat3.adjoint() * mat3.colwise().cross(vec3)).diagonal().cwiseAbs().sum(), Scalar(1)); - VERIFY_IS_MUCH_SMALLER_THAN((mat3.adjoint() * mat3.colwise().cross(Vector3::Random())).diagonal().cwiseAbs().sum(), Scalar(1)); - + VERIFY_IS_MUCH_SMALLER_THAN( + (mat3.adjoint() * mat3.colwise().cross(Vector3::Random())).diagonal().cwiseAbs().sum(), Scalar(1)); + VERIFY_IS_MUCH_SMALLER_THAN((vec3.adjoint() * mat3.colwise().cross(vec3)).cwiseAbs().sum(), Scalar(1)); VERIFY_IS_MUCH_SMALLER_THAN((vec3.adjoint() * Matrix3::Random().colwise().cross(vec3)).cwiseAbs().sum(), Scalar(1)); - + mcross = mat3.rowwise().cross(vec3); VERIFY_IS_APPROX(mcross.row(i), mat3.row(i).cross(vec3)); // cross3 - Vector4 v40 = Vector4::Random(), - v41 = Vector4::Random(), - v42 = Vector4::Random(); + Vector4 v40 = Vector4::Random(), v41 = Vector4::Random(), v42 = Vector4::Random(); v40.w() = v41.w() = v42.w() = 0; v42.template head<3>() = v40.template head<3>().cross(v41.template head<3>()); VERIFY_IS_APPROX(v40.cross3(v41), v42); VERIFY_IS_MUCH_SMALLER_THAN(v40.cross3(Vector4::Random()).dot(v40), Scalar(1)); - + // check mixed product typedef Matrix RealVector3; RealVector3 rv1 = RealVector3::Random(); @@ -77,13 +72,13 @@ template void orthomethods_3() VERIFY_IS_APPROX(rv1.template cast().cross(v1), rv1.cross(v1)); } -template void orthomethods(int size=Size) +template void orthomethods(int size = Size) { typedef typename NumTraits::Real RealScalar; - typedef Matrix VectorType; - typedef Matrix Matrix3N; - typedef Matrix MatrixN3; - typedef Matrix Vector3; + typedef Matrix VectorType; + typedef Matrix Matrix3N; + typedef Matrix MatrixN3; + typedef Matrix Vector3; VectorType v0 = VectorType::Random(size); @@ -91,10 +86,9 @@ template void orthomethods(int size=Size) VERIFY_IS_MUCH_SMALLER_THAN(v0.unitOrthogonal().dot(v0), Scalar(1)); VERIFY_IS_APPROX(v0.unitOrthogonal().norm(), RealScalar(1)); - if (size>=3) - { + if (size >= 3) { v0.template head<2>().setZero(); - v0.tail(size-2).setRandom(); + v0.tail(size - 2).setRandom(); VERIFY_IS_MUCH_SMALLER_THAN(v0.unitOrthogonal().dot(v0), Scalar(1)); VERIFY_IS_APPROX(v0.unitOrthogonal().norm(), RealScalar(1)); @@ -102,14 +96,14 @@ template void orthomethods(int size=Size) // colwise/rowwise cross product Vector3 vec3 = Vector3::Random(); - int i = internal::random(0,size-1); + int i = internal::random(0, size - 1); - Matrix3N mat3N(3,size), mcross3N(3,size); + Matrix3N mat3N(3, size), mcross3N(3, size); mat3N.setRandom(); mcross3N = mat3N.colwise().cross(vec3); VERIFY_IS_APPROX(mcross3N.col(i), mat3N.col(i).cross(vec3)); - MatrixN3 matN3(size,3), mcrossN3(size,3); + MatrixN3 matN3(size, 3), mcrossN3(size, 3); matN3.setRandom(); mcrossN3 = matN3.rowwise().cross(vec3); VERIFY_IS_APPROX(mcrossN3.row(i), matN3.row(i).cross(vec3)); @@ -117,17 +111,17 @@ template void orthomethods(int size=Size) void test_geo_orthomethods() { - for(int i = 0; i < g_repeat; i++) { - CALL_SUBTEST_1( orthomethods_3() ); - CALL_SUBTEST_2( orthomethods_3() ); - CALL_SUBTEST_4( orthomethods_3 >() ); - CALL_SUBTEST_1( (orthomethods()) ); - CALL_SUBTEST_2( (orthomethods()) ); - CALL_SUBTEST_1( (orthomethods()) ); - CALL_SUBTEST_2( (orthomethods()) ); - CALL_SUBTEST_3( (orthomethods()) ); - CALL_SUBTEST_4( (orthomethods,8>()) ); - CALL_SUBTEST_5( (orthomethods(36)) ); - CALL_SUBTEST_6( (orthomethods(35)) ); + for (int i = 0; i < g_repeat; i++) { + CALL_SUBTEST_1(orthomethods_3()); + CALL_SUBTEST_2(orthomethods_3()); + CALL_SUBTEST_4(orthomethods_3>()); + CALL_SUBTEST_1((orthomethods())); + CALL_SUBTEST_2((orthomethods())); + CALL_SUBTEST_1((orthomethods())); + CALL_SUBTEST_2((orthomethods())); + CALL_SUBTEST_3((orthomethods())); + CALL_SUBTEST_4((orthomethods, 8>())); + CALL_SUBTEST_5((orthomethods(36))); + CALL_SUBTEST_6((orthomethods(35))); } } diff --git a/filmulator-gui/core/nlmeans/eigen/test/geo_parametrizedline.cpp b/filmulator-gui/core/nlmeans/eigen/test/geo_parametrizedline.cpp index 6a879472..ef8891ea 100644 --- a/filmulator-gui/core/nlmeans/eigen/test/geo_parametrizedline.cpp +++ b/filmulator-gui/core/nlmeans/eigen/test/geo_parametrizedline.cpp @@ -13,7 +13,7 @@ #include #include -template void parametrizedline(const LineType& _line) +template void parametrizedline(const LineType &_line) { /* this test covers the following files: ParametrizedLine.h @@ -23,7 +23,7 @@ template void parametrizedline(const LineType& _line) typedef typename LineType::Scalar Scalar; typedef typename NumTraits::Real RealScalar; typedef Matrix VectorType; - typedef Hyperplane HyperplaneType; + typedef Hyperplane HyperplaneType; VectorType p0 = VectorType::Random(dim); VectorType p1 = VectorType::Random(dim); @@ -35,24 +35,24 @@ template void parametrizedline(const LineType& _line) Scalar s0 = internal::random(); Scalar s1 = abs(internal::random()); - VERIFY_IS_MUCH_SMALLER_THAN( l0.distance(p0), RealScalar(1) ); - VERIFY_IS_MUCH_SMALLER_THAN( l0.distance(p0+s0*d0), RealScalar(1) ); - VERIFY_IS_APPROX( (l0.projection(p1)-p1).norm(), l0.distance(p1) ); - VERIFY_IS_MUCH_SMALLER_THAN( l0.distance(l0.projection(p1)), RealScalar(1) ); - VERIFY_IS_APPROX( Scalar(l0.distance((p0+s0*d0) + d0.unitOrthogonal() * s1)), s1 ); + VERIFY_IS_MUCH_SMALLER_THAN(l0.distance(p0), RealScalar(1)); + VERIFY_IS_MUCH_SMALLER_THAN(l0.distance(p0 + s0 * d0), RealScalar(1)); + VERIFY_IS_APPROX((l0.projection(p1) - p1).norm(), l0.distance(p1)); + VERIFY_IS_MUCH_SMALLER_THAN(l0.distance(l0.projection(p1)), RealScalar(1)); + VERIFY_IS_APPROX(Scalar(l0.distance((p0 + s0 * d0) + d0.unitOrthogonal() * s1)), s1); // casting const int Dim = LineType::AmbientDimAtCompileTime; typedef typename GetDifferentType::type OtherScalar; - ParametrizedLine hp1f = l0.template cast(); - VERIFY_IS_APPROX(hp1f.template cast(),l0); - ParametrizedLine hp1d = l0.template cast(); - VERIFY_IS_APPROX(hp1d.template cast(),l0); + ParametrizedLine hp1f = l0.template cast(); + VERIFY_IS_APPROX(hp1f.template cast(), l0); + ParametrizedLine hp1d = l0.template cast(); + VERIFY_IS_APPROX(hp1d.template cast(), l0); // intersections VectorType p2 = VectorType::Random(dim); VectorType n2 = VectorType::Random(dim).normalized(); - HyperplaneType hp(p2,n2); + HyperplaneType hp(p2, n2); Scalar t = l0.intersectionParameter(hp); VectorType pi = l0.pointAt(t); VERIFY_IS_MUCH_SMALLER_THAN(hp.signedDistance(pi), RealScalar(1)); @@ -62,18 +62,18 @@ template void parametrizedline(const LineType& _line) template void parametrizedline_alignment() { - typedef ParametrizedLine Line4a; - typedef ParametrizedLine Line4u; + typedef ParametrizedLine Line4a; + typedef ParametrizedLine Line4u; EIGEN_ALIGN_MAX Scalar array1[16]; EIGEN_ALIGN_MAX Scalar array2[16]; - EIGEN_ALIGN_MAX Scalar array3[16+1]; - Scalar* array3u = array3+1; + EIGEN_ALIGN_MAX Scalar array3[16 + 1]; + Scalar *array3u = array3 + 1; + + Line4a *p1 = ::new (reinterpret_cast(array1)) Line4a; + Line4u *p2 = ::new (reinterpret_cast(array2)) Line4u; + Line4u *p3 = ::new (reinterpret_cast(array3u)) Line4u; - Line4a *p1 = ::new(reinterpret_cast(array1)) Line4a; - Line4u *p2 = ::new(reinterpret_cast(array2)) Line4u; - Line4u *p3 = ::new(reinterpret_cast(array3u)) Line4u; - p1->origin().setRandom(); p1->direction().setRandom(); *p2 = *p1; @@ -83,21 +83,21 @@ template void parametrizedline_alignment() VERIFY_IS_APPROX(p1->origin(), p3->origin()); VERIFY_IS_APPROX(p1->direction(), p2->direction()); VERIFY_IS_APPROX(p1->direction(), p3->direction()); - - #if defined(EIGEN_VECTORIZE) && EIGEN_MAX_STATIC_ALIGN_BYTES>0 - if(internal::packet_traits::Vectorizable && internal::packet_traits::size<=4) - VERIFY_RAISES_ASSERT((::new(reinterpret_cast(array3u)) Line4a)); - #endif + +#if defined(EIGEN_VECTORIZE) && EIGEN_MAX_STATIC_ALIGN_BYTES > 0 + if (internal::packet_traits::Vectorizable && internal::packet_traits::size <= 4) + VERIFY_RAISES_ASSERT((::new (reinterpret_cast(array3u)) Line4a)); +#endif } void test_geo_parametrizedline() { - for(int i = 0; i < g_repeat; i++) { - CALL_SUBTEST_1( parametrizedline(ParametrizedLine()) ); - CALL_SUBTEST_2( parametrizedline(ParametrizedLine()) ); - CALL_SUBTEST_2( parametrizedline_alignment() ); - CALL_SUBTEST_3( parametrizedline(ParametrizedLine()) ); - CALL_SUBTEST_3( parametrizedline_alignment() ); - CALL_SUBTEST_4( parametrizedline(ParametrizedLine,5>()) ); + for (int i = 0; i < g_repeat; i++) { + CALL_SUBTEST_1(parametrizedline(ParametrizedLine())); + CALL_SUBTEST_2(parametrizedline(ParametrizedLine())); + CALL_SUBTEST_2(parametrizedline_alignment()); + CALL_SUBTEST_3(parametrizedline(ParametrizedLine())); + CALL_SUBTEST_3(parametrizedline_alignment()); + CALL_SUBTEST_4(parametrizedline(ParametrizedLine, 5>())); } } diff --git a/filmulator-gui/core/nlmeans/eigen/test/geo_quaternion.cpp b/filmulator-gui/core/nlmeans/eigen/test/geo_quaternion.cpp index 8ee8fdb2..8aaab000 100644 --- a/filmulator-gui/core/nlmeans/eigen/test/geo_quaternion.cpp +++ b/filmulator-gui/core/nlmeans/eigen/test/geo_quaternion.cpp @@ -18,10 +18,10 @@ template T bounded_acos(T v) using std::acos; using std::min; using std::max; - return acos((max)(T(-1),(min)(v,T(1)))); + return acos((max)(T(-1), (min)(v, T(1)))); } -template void check_slerp(const QuatType& q0, const QuatType& q1) +template void check_slerp(const QuatType &q0, const QuatType &q1) { using std::abs; typedef typename QuatType::Scalar Scalar; @@ -29,16 +29,16 @@ template void check_slerp(const QuatType& q0, const QuatType& Scalar largeEps = test_precision(); - Scalar theta_tot = AA(q1*q0.inverse()).angle(); - if(theta_tot>Scalar(EIGEN_PI)) - theta_tot = Scalar(2.)*Scalar(EIGEN_PI)-theta_tot; - for(Scalar t=0; t<=Scalar(1.001); t+=Scalar(0.1)) - { - QuatType q = q0.slerp(t,q1); - Scalar theta = AA(q*q0.inverse()).angle(); + Scalar theta_tot = AA(q1 * q0.inverse()).angle(); + if (theta_tot > Scalar(EIGEN_PI)) theta_tot = Scalar(2.) * Scalar(EIGEN_PI) - theta_tot; + for (Scalar t = 0; t <= Scalar(1.001); t += Scalar(0.1)) { + QuatType q = q0.slerp(t, q1); + Scalar theta = AA(q * q0.inverse()).angle(); VERIFY(abs(q.norm() - 1) < largeEps); - if(theta_tot==0) VERIFY(theta_tot==0); - else VERIFY(abs(theta - t * theta_tot) < largeEps); + if (theta_tot == 0) + VERIFY(theta_tot == 0); + else + VERIFY(abs(theta - t * theta_tot) < largeEps); } } @@ -48,31 +48,27 @@ template void quaternion(void) Quaternion.h */ using std::abs; - typedef Matrix Vector3; - typedef Matrix Matrix3; - typedef Quaternion Quaternionx; + typedef Matrix Vector3; + typedef Matrix Matrix3; + typedef Quaternion Quaternionx; typedef AngleAxis AngleAxisx; Scalar largeEps = test_precision(); - if (internal::is_same::value) - largeEps = Scalar(1e-3); + if (internal::is_same::value) largeEps = Scalar(1e-3); Scalar eps = internal::random() * Scalar(1e-2); - Vector3 v0 = Vector3::Random(), - v1 = Vector3::Random(), - v2 = Vector3::Random(), - v3 = Vector3::Random(); + Vector3 v0 = Vector3::Random(), v1 = Vector3::Random(), v2 = Vector3::Random(), v3 = Vector3::Random(); - Scalar a = internal::random(-Scalar(EIGEN_PI), Scalar(EIGEN_PI)), - b = internal::random(-Scalar(EIGEN_PI), Scalar(EIGEN_PI)); + Scalar a = internal::random(-Scalar(EIGEN_PI), Scalar(EIGEN_PI)), + b = internal::random(-Scalar(EIGEN_PI), Scalar(EIGEN_PI)); // Quaternion: Identity(), setIdentity(); Quaternionx q1, q2; q2.setIdentity(); VERIFY_IS_APPROX(Quaternionx(Quaternionx::Identity()).coeffs(), q2.coeffs()); q1.coeffs().setRandom(); - VERIFY_IS_APPROX(q1.coeffs(), (q1*q2).coeffs()); + VERIFY_IS_APPROX(q1.coeffs(), (q1 * q2).coeffs()); // concatenation q1 *= q2; @@ -81,30 +77,27 @@ template void quaternion(void) q2 = AngleAxisx(a, v1.normalized()); // angular distance - Scalar refangle = abs(AngleAxisx(q1.inverse()*q2).angle()); - if (refangle>Scalar(EIGEN_PI)) - refangle = Scalar(2)*Scalar(EIGEN_PI) - refangle; + Scalar refangle = abs(AngleAxisx(q1.inverse() * q2).angle()); + if (refangle > Scalar(EIGEN_PI)) refangle = Scalar(2) * Scalar(EIGEN_PI) - refangle; - if((q1.coeffs()-q2.coeffs()).norm() > 10*largeEps) - { + if ((q1.coeffs() - q2.coeffs()).norm() > 10 * largeEps) { VERIFY_IS_MUCH_SMALLER_THAN(abs(q1.angularDistance(q2) - refangle), Scalar(1)); } // rotation matrix conversion VERIFY_IS_APPROX(q1 * v2, q1.toRotationMatrix() * v2); - VERIFY_IS_APPROX(q1 * q2 * v2, - q1.toRotationMatrix() * q2.toRotationMatrix() * v2); + VERIFY_IS_APPROX(q1 * q2 * v2, q1.toRotationMatrix() * q2.toRotationMatrix() * v2); - VERIFY( (q2*q1).isApprox(q1*q2, largeEps) - || !(q2 * q1 * v2).isApprox(q1.toRotationMatrix() * q2.toRotationMatrix() * v2)); + VERIFY((q2 * q1).isApprox(q1 * q2, largeEps) + || !(q2 * q1 * v2).isApprox(q1.toRotationMatrix() * q2.toRotationMatrix() * v2)); q2 = q1.toRotationMatrix(); - VERIFY_IS_APPROX(q1*v1,q2*v1); + VERIFY_IS_APPROX(q1 * v1, q2 * v1); Matrix3 rot1(q1); - VERIFY_IS_APPROX(q1*v1,rot1*v1); - Quaternionx q3(rot1.transpose()*rot1); - VERIFY_IS_APPROX(q3*v1,v1); + VERIFY_IS_APPROX(q1 * v1, rot1 * v1); + Quaternionx q3(rot1.transpose() * rot1); + VERIFY_IS_APPROX(q3 * v1, v1); // angle-axis conversion @@ -113,33 +106,29 @@ template void quaternion(void) // Do not execute the test if the rotation angle is almost zero, or // the rotation axis and v1 are almost parallel. - if (abs(aa.angle()) > 5*test_precision() - && (aa.axis() - v1.normalized()).norm() < Scalar(1.99) - && (aa.axis() + v1.normalized()).norm() < Scalar(1.99)) - { - VERIFY_IS_NOT_APPROX(q1 * v1, Quaternionx(AngleAxisx(aa.angle()*2,aa.axis())) * v1); + if (abs(aa.angle()) > 5 * test_precision() && (aa.axis() - v1.normalized()).norm() < Scalar(1.99) + && (aa.axis() + v1.normalized()).norm() < Scalar(1.99)) { + VERIFY_IS_NOT_APPROX(q1 * v1, Quaternionx(AngleAxisx(aa.angle() * 2, aa.axis())) * v1); } // from two vector creation - VERIFY_IS_APPROX( v2.normalized(),(q2.setFromTwoVectors(v1, v2)*v1).normalized()); - VERIFY_IS_APPROX( v1.normalized(),(q2.setFromTwoVectors(v1, v1)*v1).normalized()); - VERIFY_IS_APPROX(-v1.normalized(),(q2.setFromTwoVectors(v1,-v1)*v1).normalized()); - if (internal::is_same::value) - { - v3 = (v1.array()+eps).matrix(); - VERIFY_IS_APPROX( v3.normalized(),(q2.setFromTwoVectors(v1, v3)*v1).normalized()); - VERIFY_IS_APPROX(-v3.normalized(),(q2.setFromTwoVectors(v1,-v3)*v1).normalized()); + VERIFY_IS_APPROX(v2.normalized(), (q2.setFromTwoVectors(v1, v2) * v1).normalized()); + VERIFY_IS_APPROX(v1.normalized(), (q2.setFromTwoVectors(v1, v1) * v1).normalized()); + VERIFY_IS_APPROX(-v1.normalized(), (q2.setFromTwoVectors(v1, -v1) * v1).normalized()); + if (internal::is_same::value) { + v3 = (v1.array() + eps).matrix(); + VERIFY_IS_APPROX(v3.normalized(), (q2.setFromTwoVectors(v1, v3) * v1).normalized()); + VERIFY_IS_APPROX(-v3.normalized(), (q2.setFromTwoVectors(v1, -v3) * v1).normalized()); } // from two vector creation static function - VERIFY_IS_APPROX( v2.normalized(),(Quaternionx::FromTwoVectors(v1, v2)*v1).normalized()); - VERIFY_IS_APPROX( v1.normalized(),(Quaternionx::FromTwoVectors(v1, v1)*v1).normalized()); - VERIFY_IS_APPROX(-v1.normalized(),(Quaternionx::FromTwoVectors(v1,-v1)*v1).normalized()); - if (internal::is_same::value) - { - v3 = (v1.array()+eps).matrix(); - VERIFY_IS_APPROX( v3.normalized(),(Quaternionx::FromTwoVectors(v1, v3)*v1).normalized()); - VERIFY_IS_APPROX(-v3.normalized(),(Quaternionx::FromTwoVectors(v1,-v3)*v1).normalized()); + VERIFY_IS_APPROX(v2.normalized(), (Quaternionx::FromTwoVectors(v1, v2) * v1).normalized()); + VERIFY_IS_APPROX(v1.normalized(), (Quaternionx::FromTwoVectors(v1, v1) * v1).normalized()); + VERIFY_IS_APPROX(-v1.normalized(), (Quaternionx::FromTwoVectors(v1, -v1) * v1).normalized()); + if (internal::is_same::value) { + v3 = (v1.array() + eps).matrix(); + VERIFY_IS_APPROX(v3.normalized(), (Quaternionx::FromTwoVectors(v1, v3) * v1).normalized()); + VERIFY_IS_APPROX(-v3.normalized(), (Quaternionx::FromTwoVectors(v1, -v3) * v1).normalized()); } // inverse and conjugate @@ -148,9 +137,9 @@ template void quaternion(void) // test casting Quaternion q1f = q1.template cast(); - VERIFY_IS_APPROX(q1f.template cast(),q1); + VERIFY_IS_APPROX(q1f.template cast(), q1); Quaternion q1d = q1.template cast(); - VERIFY_IS_APPROX(q1d.template cast(),q1); + VERIFY_IS_APPROX(q1d.template cast(), q1); // test bug 369 - improper alignment. Quaternionx *q = new Quaternionx; @@ -158,46 +147,46 @@ template void quaternion(void) q1 = Quaternionx::UnitRandom(); q2 = Quaternionx::UnitRandom(); - check_slerp(q1,q2); + check_slerp(q1, q2); q1 = AngleAxisx(b, v1.normalized()); - q2 = AngleAxisx(b+Scalar(EIGEN_PI), v1.normalized()); - check_slerp(q1,q2); + q2 = AngleAxisx(b + Scalar(EIGEN_PI), v1.normalized()); + check_slerp(q1, q2); - q1 = AngleAxisx(b, v1.normalized()); + q1 = AngleAxisx(b, v1.normalized()); q2 = AngleAxisx(-b, -v1.normalized()); - check_slerp(q1,q2); + check_slerp(q1, q2); q1 = Quaternionx::UnitRandom(); q2.coeffs() = -q1.coeffs(); - check_slerp(q1,q2); + check_slerp(q1, q2); } -template void mapQuaternion(void){ +template void mapQuaternion(void) +{ typedef Map, Aligned> MQuaternionA; typedef Map, Aligned> MCQuaternionA; - typedef Map > MQuaternionUA; - typedef Map > MCQuaternionUA; + typedef Map> MQuaternionUA; + typedef Map> MCQuaternionUA; typedef Quaternion Quaternionx; - typedef Matrix Vector3; + typedef Matrix Vector3; typedef AngleAxis AngleAxisx; - - Vector3 v0 = Vector3::Random(), - v1 = Vector3::Random(); - Scalar a = internal::random(-Scalar(EIGEN_PI), Scalar(EIGEN_PI)); + + Vector3 v0 = Vector3::Random(), v1 = Vector3::Random(); + Scalar a = internal::random(-Scalar(EIGEN_PI), Scalar(EIGEN_PI)); EIGEN_ALIGN_MAX Scalar array1[4]; EIGEN_ALIGN_MAX Scalar array2[4]; - EIGEN_ALIGN_MAX Scalar array3[4+1]; - Scalar* array3unaligned = array3+1; - - MQuaternionA mq1(array1); - MCQuaternionA mcq1(array1); - MQuaternionA mq2(array2); - MQuaternionUA mq3(array3unaligned); - MCQuaternionUA mcq3(array3unaligned); - -// std::cerr << array1 << " " << array2 << " " << array3 << "\n"; + EIGEN_ALIGN_MAX Scalar array3[4 + 1]; + Scalar *array3unaligned = array3 + 1; + + MQuaternionA mq1(array1); + MCQuaternionA mcq1(array1); + MQuaternionA mq2(array2); + MQuaternionUA mq3(array3unaligned); + MCQuaternionUA mcq3(array3unaligned); + + // std::cerr << array1 << " " << array2 << " " << array3 << "\n"; mq1 = AngleAxisx(a, v0.normalized()); mq2 = mq1; mq3 = mq1; @@ -210,54 +199,54 @@ template void mapQuaternion(void){ VERIFY_IS_APPROX(q1.coeffs(), q2.coeffs()); VERIFY_IS_APPROX(q1.coeffs(), q3.coeffs()); VERIFY_IS_APPROX(q4.coeffs(), q3.coeffs()); - #ifdef EIGEN_VECTORIZE - if(internal::packet_traits::Vectorizable) - VERIFY_RAISES_ASSERT((MQuaternionA(array3unaligned))); - #endif - +#ifdef EIGEN_VECTORIZE + if (internal::packet_traits::Vectorizable) VERIFY_RAISES_ASSERT((MQuaternionA(array3unaligned))); +#endif + VERIFY_IS_APPROX(mq1 * (mq1.inverse() * v1), v1); VERIFY_IS_APPROX(mq1 * (mq1.conjugate() * v1), v1); - + VERIFY_IS_APPROX(mcq1 * (mcq1.inverse() * v1), v1); VERIFY_IS_APPROX(mcq1 * (mcq1.conjugate() * v1), v1); - + VERIFY_IS_APPROX(mq3 * (mq3.inverse() * v1), v1); VERIFY_IS_APPROX(mq3 * (mq3.conjugate() * v1), v1); - + VERIFY_IS_APPROX(mcq3 * (mcq3.inverse() * v1), v1); VERIFY_IS_APPROX(mcq3 * (mcq3.conjugate() * v1), v1); - - VERIFY_IS_APPROX(mq1*mq2, q1*q2); - VERIFY_IS_APPROX(mq3*mq2, q3*q2); - VERIFY_IS_APPROX(mcq1*mq2, q1*q2); - VERIFY_IS_APPROX(mcq3*mq2, q3*q2); + + VERIFY_IS_APPROX(mq1 * mq2, q1 * q2); + VERIFY_IS_APPROX(mq3 * mq2, q3 * q2); + VERIFY_IS_APPROX(mcq1 * mq2, q1 * q2); + VERIFY_IS_APPROX(mcq3 * mq2, q3 * q2); // Bug 1461, compilation issue with Map::w(), and other reference/constness checks: VERIFY_IS_APPROX(mcq3.coeffs().x() + mcq3.coeffs().y() + mcq3.coeffs().z() + mcq3.coeffs().w(), mcq3.coeffs().sum()); VERIFY_IS_APPROX(mcq3.x() + mcq3.y() + mcq3.z() + mcq3.w(), mcq3.coeffs().sum()); mq3.w() = 1; - const Quaternionx& cq3(q3); - VERIFY( &cq3.x() == &q3.x() ); - const MQuaternionUA& cmq3(mq3); - VERIFY( &cmq3.x() == &mq3.x() ); + const Quaternionx &cq3(q3); + VERIFY(&cq3.x() == &q3.x()); + const MQuaternionUA &cmq3(mq3); + VERIFY(&cmq3.x() == &mq3.x()); // FIXME the following should be ok. The problem is that currently the LValueBit flag // is used to determine wether we can return a coeff by reference or not, which is not enough for Map. - //const MCQuaternionUA& cmcq3(mcq3); - //VERIFY( &cmcq3.x() == &mcq3.x() ); + // const MCQuaternionUA& cmcq3(mcq3); + // VERIFY( &cmcq3.x() == &mcq3.x() ); } -template void quaternionAlignment(void){ - typedef Quaternion QuaternionA; - typedef Quaternion QuaternionUA; +template void quaternionAlignment(void) +{ + typedef Quaternion QuaternionA; + typedef Quaternion QuaternionUA; EIGEN_ALIGN_MAX Scalar array1[4]; EIGEN_ALIGN_MAX Scalar array2[4]; - EIGEN_ALIGN_MAX Scalar array3[4+1]; - Scalar* arrayunaligned = array3+1; + EIGEN_ALIGN_MAX Scalar array3[4 + 1]; + Scalar *arrayunaligned = array3 + 1; - QuaternionA *q1 = ::new(reinterpret_cast(array1)) QuaternionA; - QuaternionUA *q2 = ::new(reinterpret_cast(array2)) QuaternionUA; - QuaternionUA *q3 = ::new(reinterpret_cast(arrayunaligned)) QuaternionUA; + QuaternionA *q1 = ::new (reinterpret_cast(array1)) QuaternionA; + QuaternionUA *q2 = ::new (reinterpret_cast(array2)) QuaternionUA; + QuaternionUA *q3 = ::new (reinterpret_cast(arrayunaligned)) QuaternionUA; q1->coeffs().setRandom(); *q2 = *q1; @@ -265,13 +254,13 @@ template void quaternionAlignment(void){ VERIFY_IS_APPROX(q1->coeffs(), q2->coeffs()); VERIFY_IS_APPROX(q1->coeffs(), q3->coeffs()); - #if defined(EIGEN_VECTORIZE) && EIGEN_MAX_STATIC_ALIGN_BYTES>0 - if(internal::packet_traits::Vectorizable && internal::packet_traits::size<=4) - VERIFY_RAISES_ASSERT((::new(reinterpret_cast(arrayunaligned)) QuaternionA)); - #endif +#if defined(EIGEN_VECTORIZE) && EIGEN_MAX_STATIC_ALIGN_BYTES > 0 + if (internal::packet_traits::Vectorizable && internal::packet_traits::size <= 4) + VERIFY_RAISES_ASSERT((::new (reinterpret_cast(arrayunaligned)) QuaternionA)); +#endif } -template void check_const_correctness(const PlainObjectType&) +template void check_const_correctness(const PlainObjectType &) { // there's a lot that we can't test here while still having this test compile! // the only possible approach would be to run a script trying to compile stuff and checking that it fails. @@ -279,24 +268,24 @@ template void check_const_correctness(const PlainObjec // verify that map-to-const don't have LvalueBit typedef typename internal::add_const::type ConstPlainObjectType; - VERIFY( !(internal::traits >::Flags & LvalueBit) ); - VERIFY( !(internal::traits >::Flags & LvalueBit) ); - VERIFY( !(Map::Flags & LvalueBit) ); - VERIFY( !(Map::Flags & LvalueBit) ); + VERIFY(!(internal::traits>::Flags & LvalueBit)); + VERIFY(!(internal::traits>::Flags & LvalueBit)); + VERIFY(!(Map::Flags & LvalueBit)); + VERIFY(!(Map::Flags & LvalueBit)); } void test_geo_quaternion() { - for(int i = 0; i < g_repeat; i++) { - CALL_SUBTEST_1(( quaternion() )); - CALL_SUBTEST_1( check_const_correctness(Quaternionf()) ); - CALL_SUBTEST_2(( quaternion() )); - CALL_SUBTEST_2( check_const_correctness(Quaterniond()) ); - CALL_SUBTEST_3(( quaternion() )); - CALL_SUBTEST_4(( quaternion() )); - CALL_SUBTEST_5(( quaternionAlignment() )); - CALL_SUBTEST_6(( quaternionAlignment() )); - CALL_SUBTEST_1( mapQuaternion() ); - CALL_SUBTEST_2( mapQuaternion() ); + for (int i = 0; i < g_repeat; i++) { + CALL_SUBTEST_1((quaternion())); + CALL_SUBTEST_1(check_const_correctness(Quaternionf())); + CALL_SUBTEST_2((quaternion())); + CALL_SUBTEST_2(check_const_correctness(Quaterniond())); + CALL_SUBTEST_3((quaternion())); + CALL_SUBTEST_4((quaternion())); + CALL_SUBTEST_5((quaternionAlignment())); + CALL_SUBTEST_6((quaternionAlignment())); + CALL_SUBTEST_1(mapQuaternion()); + CALL_SUBTEST_2(mapQuaternion()); } } diff --git a/filmulator-gui/core/nlmeans/eigen/test/geo_transformations.cpp b/filmulator-gui/core/nlmeans/eigen/test/geo_transformations.cpp old mode 100755 new mode 100644 index 278e527c..303f855f --- a/filmulator-gui/core/nlmeans/eigen/test/geo_transformations.cpp +++ b/filmulator-gui/core/nlmeans/eigen/test/geo_transformations.cpp @@ -12,31 +12,28 @@ #include #include -template -Matrix angleToVec(T a) -{ - return Matrix(std::cos(a), std::sin(a)); -} +template Matrix angleToVec(T a) { return Matrix(std::cos(a), std::sin(a)); } // This permits to workaround a bug in clang/llvm code generation. -template -EIGEN_DONT_INLINE -void dont_over_optimize(T& x) { volatile typename T::Scalar tmp = x(0); x(0) = tmp; } +template EIGEN_DONT_INLINE void dont_over_optimize(T &x) +{ + volatile typename T::Scalar tmp = x(0); + x(0) = tmp; +} template void non_projective_only() { - /* this test covers the following files: - Cross.h Quaternion.h, Transform.cpp - */ - typedef Matrix Vector3; + /* this test covers the following files: + Cross.h Quaternion.h, Transform.cpp +*/ + typedef Matrix Vector3; typedef Quaternion Quaternionx; typedef AngleAxis AngleAxisx; - typedef Transform Transform3; - typedef DiagonalMatrix AlignedScaling3; - typedef Translation Translation3; + typedef Transform Transform3; + typedef DiagonalMatrix AlignedScaling3; + typedef Translation Translation3; - Vector3 v0 = Vector3::Random(), - v1 = Vector3::Random(); + Vector3 v0 = Vector3::Random(), v1 = Vector3::Random(); Transform3 t0, t1, t2; @@ -54,7 +51,7 @@ template void non_projective_only() v0 << 50, 2, 1; t0.scale(v0); - VERIFY_IS_APPROX( (t0 * Vector3(1,0,0)).template head<3>().norm(), v0.x()); + VERIFY_IS_APPROX((t0 * Vector3(1, 0, 0)).template head<3>().norm(), v0.x()); t0.setIdentity(); t1.setIdentity(); @@ -70,7 +67,7 @@ template void non_projective_only() t1.fromPositionOrientationScale(v0, q1, v1); VERIFY_IS_APPROX(t1.matrix(), t0.matrix()); - VERIFY_IS_APPROX(t1*v1, t0*v1); + VERIFY_IS_APPROX(t1 * v1, t0 * v1); // translation * vector t0.setIdentity(); @@ -90,35 +87,33 @@ template void transformations() */ using std::cos; using std::abs; - typedef Matrix Matrix3; - typedef Matrix Matrix4; - typedef Matrix Vector2; - typedef Matrix Vector3; - typedef Matrix Vector4; + typedef Matrix Matrix3; + typedef Matrix Matrix4; + typedef Matrix Vector2; + typedef Matrix Vector3; + typedef Matrix Vector4; typedef Quaternion Quaternionx; typedef AngleAxis AngleAxisx; - typedef Transform Transform2; - typedef Transform Transform3; + typedef Transform Transform2; + typedef Transform Transform3; typedef typename Transform3::MatrixType MatrixType; - typedef DiagonalMatrix AlignedScaling3; - typedef Translation Translation2; - typedef Translation Translation3; + typedef DiagonalMatrix AlignedScaling3; + typedef Translation Translation2; + typedef Translation Translation3; - Vector3 v0 = Vector3::Random(), - v1 = Vector3::Random(); + Vector3 v0 = Vector3::Random(), v1 = Vector3::Random(); Matrix3 matrot1, m; Scalar a = internal::random(-Scalar(EIGEN_PI), Scalar(EIGEN_PI)); Scalar s0 = internal::random(), s1 = internal::random(); - - while(v0.norm() < test_precision()) v0 = Vector3::Random(); - while(v1.norm() < test_precision()) v1 = Vector3::Random(); + + while (v0.norm() < test_precision()) v0 = Vector3::Random(); + while (v1.norm() < test_precision()) v1 = Vector3::Random(); VERIFY_IS_APPROX(v0, AngleAxisx(a, v0.normalized()) * v0); VERIFY_IS_APPROX(-v0, AngleAxisx(Scalar(EIGEN_PI), v0.unitOrthogonal()) * v0); - if(abs(cos(a)) > test_precision()) - { - VERIFY_IS_APPROX(cos(a)*v0.squaredNorm(), v0.dot(AngleAxisx(a, v0.unitOrthogonal()) * v0)); + if (abs(cos(a)) > test_precision()) { + VERIFY_IS_APPROX(cos(a) * v0.squaredNorm(), v0.dot(AngleAxisx(a, v0.unitOrthogonal()) * v0)); } m = AngleAxisx(a, v0.normalized()).toRotationMatrix().adjoint(); VERIFY_IS_APPROX(Matrix3::Identity(), m * AngleAxisx(a, v0.normalized())); @@ -129,47 +124,45 @@ template void transformations() q2 = AngleAxisx(a, v1.normalized()); // rotation matrix conversion - matrot1 = AngleAxisx(Scalar(0.1), Vector3::UnitX()) - * AngleAxisx(Scalar(0.2), Vector3::UnitY()) - * AngleAxisx(Scalar(0.3), Vector3::UnitZ()); + matrot1 = AngleAxisx(Scalar(0.1), Vector3::UnitX()) * AngleAxisx(Scalar(0.2), Vector3::UnitY()) + * AngleAxisx(Scalar(0.3), Vector3::UnitZ()); VERIFY_IS_APPROX(matrot1 * v1, - AngleAxisx(Scalar(0.1), Vector3(1,0,0)).toRotationMatrix() - * (AngleAxisx(Scalar(0.2), Vector3(0,1,0)).toRotationMatrix() - * (AngleAxisx(Scalar(0.3), Vector3(0,0,1)).toRotationMatrix() * v1))); + AngleAxisx(Scalar(0.1), Vector3(1, 0, 0)).toRotationMatrix() + * (AngleAxisx(Scalar(0.2), Vector3(0, 1, 0)).toRotationMatrix() + * (AngleAxisx(Scalar(0.3), Vector3(0, 0, 1)).toRotationMatrix() * v1))); // angle-axis conversion AngleAxisx aa = AngleAxisx(q1); VERIFY_IS_APPROX(q1 * v1, Quaternionx(aa) * v1); - + // The following test is stable only if 2*angle != angle and v1 is not colinear with axis - if( (abs(aa.angle()) > test_precision()) && (abs(aa.axis().dot(v1.normalized()))<(Scalar(1)-Scalar(4)*test_precision())) ) - { - VERIFY( !(q1 * v1).isApprox(Quaternionx(AngleAxisx(aa.angle()*2,aa.axis())) * v1) ); + if ((abs(aa.angle()) > test_precision()) + && (abs(aa.axis().dot(v1.normalized())) < (Scalar(1) - Scalar(4) * test_precision()))) { + VERIFY(!(q1 * v1).isApprox(Quaternionx(AngleAxisx(aa.angle() * 2, aa.axis())) * v1)); } aa.fromRotationMatrix(aa.toRotationMatrix()); VERIFY_IS_APPROX(q1 * v1, Quaternionx(aa) * v1); // The following test is stable only if 2*angle != angle and v1 is not colinear with axis - if( (abs(aa.angle()) > test_precision()) && (abs(aa.axis().dot(v1.normalized()))<(Scalar(1)-Scalar(4)*test_precision())) ) - { - VERIFY( !(q1 * v1).isApprox(Quaternionx(AngleAxisx(aa.angle()*2,aa.axis())) * v1) ); + if ((abs(aa.angle()) > test_precision()) + && (abs(aa.axis().dot(v1.normalized())) < (Scalar(1) - Scalar(4) * test_precision()))) { + VERIFY(!(q1 * v1).isApprox(Quaternionx(AngleAxisx(aa.angle() * 2, aa.axis())) * v1)); } // AngleAxis - VERIFY_IS_APPROX(AngleAxisx(a,v1.normalized()).toRotationMatrix(), - Quaternionx(AngleAxisx(a,v1.normalized())).toRotationMatrix()); + VERIFY_IS_APPROX( + AngleAxisx(a, v1.normalized()).toRotationMatrix(), Quaternionx(AngleAxisx(a, v1.normalized())).toRotationMatrix()); AngleAxisx aa1; m = q1.toRotationMatrix(); aa1 = m; - VERIFY_IS_APPROX(AngleAxisx(m).toRotationMatrix(), - Quaternionx(m).toRotationMatrix()); + VERIFY_IS_APPROX(AngleAxisx(m).toRotationMatrix(), Quaternionx(m).toRotationMatrix()); // Transform // TODO complete the tests ! a = 0; - while (abs(a)(-Scalar(0.4)*Scalar(EIGEN_PI), Scalar(0.4)*Scalar(EIGEN_PI)); + while (abs(a) < Scalar(0.1)) + a = internal::random(-Scalar(0.4) * Scalar(EIGEN_PI), Scalar(0.4) * Scalar(EIGEN_PI)); q1 = AngleAxisx(a, v0.normalized()); Transform3 t0, t1, t2; @@ -195,11 +188,14 @@ template void transformations() t1.fromPositionOrientationScale(v0, q1, v1); VERIFY_IS_APPROX(t1.matrix(), t0.matrix()); - t0.setIdentity(); t0.scale(v0).rotate(q1.toRotationMatrix()); - t1.setIdentity(); t1.scale(v0).rotate(q1); + t0.setIdentity(); + t0.scale(v0).rotate(q1.toRotationMatrix()); + t1.setIdentity(); + t1.scale(v0).rotate(q1); VERIFY_IS_APPROX(t0.matrix(), t1.matrix()); - t0.setIdentity(); t0.scale(v0).rotate(AngleAxisx(q1)); + t0.setIdentity(); + t0.scale(v0).rotate(AngleAxisx(q1)); VERIFY_IS_APPROX(t0.matrix(), t1.matrix()); VERIFY_IS_APPROX(t0.scale(a).matrix(), t1.scale(Vector3::Constant(a)).matrix()); @@ -209,10 +205,9 @@ template void transformations() Matrix3 mat3 = Matrix3::Random(); Matrix4 mat4; - mat4 << mat3 , Vector3::Zero() , Vector4::Zero().transpose(); + mat4 << mat3, Vector3::Zero(), Vector4::Zero().transpose(); Transform3 tmat3(mat3), tmat4(mat4); - if(Mode!=int(AffineCompact)) - tmat4.matrix()(3,3) = Scalar(1); + if (Mode != int(AffineCompact)) tmat4.matrix()(3, 3) = Scalar(1); VERIFY_IS_APPROX(tmat3.matrix(), tmat4.matrix()); Scalar a3 = internal::random(-Scalar(EIGEN_PI), Scalar(EIGEN_PI)); @@ -222,7 +217,7 @@ template void transformations() Transform3 t4; t4 = aa3; VERIFY_IS_APPROX(t3.matrix(), t4.matrix()); - t4.rotate(AngleAxisx(-a3,v3)); + t4.rotate(AngleAxisx(-a3, v3)); VERIFY_IS_APPROX(t4.matrix(), MatrixType::Identity()); t4 *= aa3; VERIFY_IS_APPROX(t3.matrix(), t4.matrix()); @@ -230,7 +225,7 @@ template void transformations() do { v3 = Vector3::Random(); dont_over_optimize(v3); - } while (v3.cwiseAbs().minCoeff()::epsilon()); + } while (v3.cwiseAbs().minCoeff() < NumTraits::epsilon()); Translation3 tv3(v3); Transform3 t5(tv3); t4 = tv3; @@ -250,31 +245,31 @@ template void transformations() VERIFY_IS_APPROX(t6.matrix(), t4.matrix()); // matrix * transform - VERIFY_IS_APPROX((t3.matrix()*t4).matrix(), (t3*t4).matrix()); + VERIFY_IS_APPROX((t3.matrix() * t4).matrix(), (t3 * t4).matrix()); // chained Transform product - VERIFY_IS_APPROX(((t3*t4)*t5).matrix(), (t3*(t4*t5)).matrix()); + VERIFY_IS_APPROX(((t3 * t4) * t5).matrix(), (t3 * (t4 * t5)).matrix()); // check that Transform product doesn't have aliasing problems t5 = t4; - t5 = t5*t5; - VERIFY_IS_APPROX(t5, t4*t4); + t5 = t5 * t5; + VERIFY_IS_APPROX(t5, t4 * t4); // 2D transformation Transform2 t20, t21; Vector2 v20 = Vector2::Random(); Vector2 v21 = Vector2::Random(); - for (int k=0; k<2; ++k) - if (abs(v21[k])(a).toRotationMatrix(); - VERIFY_IS_APPROX(t20.fromPositionOrientationScale(v20,a,v21).matrix(), - t21.pretranslate(v20).scale(v21).matrix()); + VERIFY_IS_APPROX(t20.fromPositionOrientationScale(v20, a, v21).matrix(), t21.pretranslate(v20).scale(v21).matrix()); t21.setIdentity(); t21.linear() = Rotation2D(-a).toRotationMatrix(); - VERIFY( (t20.fromPositionOrientationScale(v20,a,v21) - * (t21.prescale(v21.cwiseInverse()).translate(-v20))).matrix().isIdentity(test_precision()) ); + VERIFY((t20.fromPositionOrientationScale(v20, a, v21) * (t21.prescale(v21.cwiseInverse()).translate(-v20))) + .matrix() + .isIdentity(test_precision())); // Transform - new API // 3D @@ -299,13 +294,13 @@ template void transformations() t0.prescale(s0); t1 = Eigen::Scaling(s0) * t1; VERIFY_IS_APPROX(t0.matrix(), t1.matrix()); - + t0 = t3; t0.scale(s0); - t1 = t3 * Eigen::Scaling(s0,s0,s0); + t1 = t3 * Eigen::Scaling(s0, s0, s0); VERIFY_IS_APPROX(t0.matrix(), t1.matrix()); t0.prescale(s0); - t1 = Eigen::Scaling(s0,s0,s0) * t1; + t1 = Eigen::Scaling(s0, s0, s0) * t1; VERIFY_IS_APPROX(t0.matrix(), t1.matrix()); t0 = t3; @@ -381,159 +376,159 @@ template void transformations() t0.translate(v0); do { t0.linear().setRandom(); - } while(t0.linear().jacobiSvd().singularValues()(2)()); + } while (t0.linear().jacobiSvd().singularValues()(2) < test_precision()); Matrix4 t044 = Matrix4::Zero(); - t044(3,3) = 1; - t044.block(0,0,t0.matrix().rows(),4) = t0.matrix(); - VERIFY_IS_APPROX(t0.inverse(Affine).matrix(), t044.inverse().block(0,0,t0.matrix().rows(),4)); + t044(3, 3) = 1; + t044.block(0, 0, t0.matrix().rows(), 4) = t0.matrix(); + VERIFY_IS_APPROX(t0.inverse(Affine).matrix(), t044.inverse().block(0, 0, t0.matrix().rows(), 4)); t0.setIdentity(); t0.translate(v0).rotate(q1); t044 = Matrix4::Zero(); - t044(3,3) = 1; - t044.block(0,0,t0.matrix().rows(),4) = t0.matrix(); - VERIFY_IS_APPROX(t0.inverse(Isometry).matrix(), t044.inverse().block(0,0,t0.matrix().rows(),4)); + t044(3, 3) = 1; + t044.block(0, 0, t0.matrix().rows(), 4) = t0.matrix(); + VERIFY_IS_APPROX(t0.inverse(Isometry).matrix(), t044.inverse().block(0, 0, t0.matrix().rows(), 4)); Matrix3 mat_rotation, mat_scaling; t0.setIdentity(); t0.translate(v0).rotate(q1).scale(v1); t0.computeRotationScaling(&mat_rotation, &mat_scaling); VERIFY_IS_APPROX(t0.linear(), mat_rotation * mat_scaling); - VERIFY_IS_APPROX(mat_rotation*mat_rotation.adjoint(), Matrix3::Identity()); + VERIFY_IS_APPROX(mat_rotation * mat_rotation.adjoint(), Matrix3::Identity()); VERIFY_IS_APPROX(mat_rotation.determinant(), Scalar(1)); t0.computeScalingRotation(&mat_scaling, &mat_rotation); VERIFY_IS_APPROX(t0.linear(), mat_scaling * mat_rotation); - VERIFY_IS_APPROX(mat_rotation*mat_rotation.adjoint(), Matrix3::Identity()); + VERIFY_IS_APPROX(mat_rotation * mat_rotation.adjoint(), Matrix3::Identity()); VERIFY_IS_APPROX(mat_rotation.determinant(), Scalar(1)); // test casting - Transform t1f = t1.template cast(); - VERIFY_IS_APPROX(t1f.template cast(),t1); - Transform t1d = t1.template cast(); - VERIFY_IS_APPROX(t1d.template cast(),t1); + Transform t1f = t1.template cast(); + VERIFY_IS_APPROX(t1f.template cast(), t1); + Transform t1d = t1.template cast(); + VERIFY_IS_APPROX(t1d.template cast(), t1); Translation3 tr1(v0); - Translation tr1f = tr1.template cast(); - VERIFY_IS_APPROX(tr1f.template cast(),tr1); - Translation tr1d = tr1.template cast(); - VERIFY_IS_APPROX(tr1d.template cast(),tr1); + Translation tr1f = tr1.template cast(); + VERIFY_IS_APPROX(tr1f.template cast(), tr1); + Translation tr1d = tr1.template cast(); + VERIFY_IS_APPROX(tr1d.template cast(), tr1); AngleAxis aa1f = aa1.template cast(); - VERIFY_IS_APPROX(aa1f.template cast(),aa1); + VERIFY_IS_APPROX(aa1f.template cast(), aa1); AngleAxis aa1d = aa1.template cast(); - VERIFY_IS_APPROX(aa1d.template cast(),aa1); + VERIFY_IS_APPROX(aa1d.template cast(), aa1); Rotation2D r2d1(internal::random()); Rotation2D r2d1f = r2d1.template cast(); - VERIFY_IS_APPROX(r2d1f.template cast(),r2d1); + VERIFY_IS_APPROX(r2d1f.template cast(), r2d1); Rotation2D r2d1d = r2d1.template cast(); - VERIFY_IS_APPROX(r2d1d.template cast(),r2d1); - - for(int k=0; k<100; ++k) - { - Scalar angle = internal::random(-100,100); + VERIFY_IS_APPROX(r2d1d.template cast(), r2d1); + + for (int k = 0; k < 100; ++k) { + Scalar angle = internal::random(-100, 100); Rotation2D rot2(angle); - VERIFY( rot2.smallestPositiveAngle() >= 0 ); - VERIFY( rot2.smallestPositiveAngle() <= Scalar(2)*Scalar(EIGEN_PI) ); - VERIFY_IS_APPROX( angleToVec(rot2.smallestPositiveAngle()), angleToVec(rot2.angle()) ); - - VERIFY( rot2.smallestAngle() >= -Scalar(EIGEN_PI) ); - VERIFY( rot2.smallestAngle() <= Scalar(EIGEN_PI) ); - VERIFY_IS_APPROX( angleToVec(rot2.smallestAngle()), angleToVec(rot2.angle()) ); - - Matrix rot2_as_mat(rot2); + VERIFY(rot2.smallestPositiveAngle() >= 0); + VERIFY(rot2.smallestPositiveAngle() <= Scalar(2) * Scalar(EIGEN_PI)); + VERIFY_IS_APPROX(angleToVec(rot2.smallestPositiveAngle()), angleToVec(rot2.angle())); + + VERIFY(rot2.smallestAngle() >= -Scalar(EIGEN_PI)); + VERIFY(rot2.smallestAngle() <= Scalar(EIGEN_PI)); + VERIFY_IS_APPROX(angleToVec(rot2.smallestAngle()), angleToVec(rot2.angle())); + + Matrix rot2_as_mat(rot2); Rotation2D rot3(rot2_as_mat); - VERIFY_IS_APPROX( angleToVec(rot2.smallestAngle()), angleToVec(rot3.angle()) ); + VERIFY_IS_APPROX(angleToVec(rot2.smallestAngle()), angleToVec(rot3.angle())); } - s0 = internal::random(-100,100); - s1 = internal::random(-100,100); + s0 = internal::random(-100, 100); + s1 = internal::random(-100, 100); Rotation2D R0(s0), R1(s1); - + t20 = Translation2(v20) * (R0 * Eigen::Scaling(s0)); t21 = Translation2(v20) * R0 * Eigen::Scaling(s0); - VERIFY_IS_APPROX(t20,t21); - + VERIFY_IS_APPROX(t20, t21); + t20 = Translation2(v20) * (R0 * R0.inverse() * Eigen::Scaling(s0)); t21 = Translation2(v20) * Eigen::Scaling(s0); - VERIFY_IS_APPROX(t20,t21); - + VERIFY_IS_APPROX(t20, t21); + VERIFY_IS_APPROX(s0, (R0.slerp(0, R1)).angle()); - VERIFY_IS_APPROX( angleToVec(R1.smallestPositiveAngle()), angleToVec((R0.slerp(1, R1)).smallestPositiveAngle()) ); + VERIFY_IS_APPROX(angleToVec(R1.smallestPositiveAngle()), angleToVec((R0.slerp(1, R1)).smallestPositiveAngle())); VERIFY_IS_APPROX(R0.smallestPositiveAngle(), (R0.slerp(0.5, R0)).smallestPositiveAngle()); - if(std::cos(s0)>0) + if (std::cos(s0) > 0) VERIFY_IS_MUCH_SMALLER_THAN((R0.slerp(0.5, R0.inverse())).smallestAngle(), Scalar(1)); else VERIFY_IS_APPROX(Scalar(EIGEN_PI), (R0.slerp(0.5, R0.inverse())).smallestPositiveAngle()); - + // Check path length Scalar l = 0; int path_steps = 100; - for(int k=0; k::epsilon()*Scalar(path_steps/2))); - + VERIFY(l <= Scalar(EIGEN_PI) * (Scalar(1) + NumTraits::epsilon() * Scalar(path_steps / 2))); + // check basic features { - Rotation2D r1; // default ctor - r1 = Rotation2D(s0); // copy assignment - VERIFY_IS_APPROX(r1.angle(),s0); - Rotation2D r2(r1); // copy ctor - VERIFY_IS_APPROX(r2.angle(),s0); + Rotation2D r1;// default ctor + r1 = Rotation2D(s0);// copy assignment + VERIFY_IS_APPROX(r1.angle(), s0); + Rotation2D r2(r1);// copy ctor + VERIFY_IS_APPROX(r2.angle(), s0); } { Transform3 t32(Matrix4::Random()), t33, t34; t34 = t33 = t32; t32.scale(v0); - t33*=AlignedScaling3(v0); + t33 *= AlignedScaling3(v0); VERIFY_IS_APPROX(t32.matrix(), t33.matrix()); t33 = t34 * AlignedScaling3(v0); VERIFY_IS_APPROX(t32.matrix(), t33.matrix()); } - } template -void transform_associativity_left(const A1& a1, const A2& a2, const P& p, const Q& q, const V& v, const H& h) +void transform_associativity_left(const A1 &a1, const A2 &a2, const P &p, const Q &q, const V &v, const H &h) { - VERIFY_IS_APPROX( q*(a1*v), (q*a1)*v ); - VERIFY_IS_APPROX( q*(a2*v), (q*a2)*v ); - VERIFY_IS_APPROX( q*(p*h).hnormalized(), ((q*p)*h).hnormalized() ); + VERIFY_IS_APPROX(q * (a1 * v), (q * a1) * v); + VERIFY_IS_APPROX(q * (a2 * v), (q * a2) * v); + VERIFY_IS_APPROX(q * (p * h).hnormalized(), ((q * p) * h).hnormalized()); } template -void transform_associativity2(const A1& a1, const A2& a2, const P& p, const Q& q, const V& v, const H& h) +void transform_associativity2(const A1 &a1, const A2 &a2, const P &p, const Q &q, const V &v, const H &h) { - VERIFY_IS_APPROX( a1*(q*v), (a1*q)*v ); - VERIFY_IS_APPROX( a2*(q*v), (a2*q)*v ); - VERIFY_IS_APPROX( p *(q*v).homogeneous(), (p *q)*v.homogeneous() ); + VERIFY_IS_APPROX(a1 * (q * v), (a1 * q) * v); + VERIFY_IS_APPROX(a2 * (q * v), (a2 * q) * v); + VERIFY_IS_APPROX(p * (q * v).homogeneous(), (p * q) * v.homogeneous()); - transform_associativity_left(a1, a2,p, q, v, h); + transform_associativity_left(a1, a2, p, q, v, h); } -template -void transform_associativity(const RotationType& R) +template +void transform_associativity(const RotationType &R) { - typedef Matrix VectorType; - typedef Matrix HVectorType; - typedef Matrix LinearType; - typedef Matrix MatrixType; - typedef Transform AffineCompactType; - typedef Transform AffineType; - typedef Transform ProjectiveType; - typedef DiagonalMatrix ScalingType; - typedef Translation TranslationType; - - AffineCompactType A1c; A1c.matrix().setRandom(); - AffineCompactType A2c; A2c.matrix().setRandom(); + typedef Matrix VectorType; + typedef Matrix HVectorType; + typedef Matrix LinearType; + typedef Matrix MatrixType; + typedef Transform AffineCompactType; + typedef Transform AffineType; + typedef Transform ProjectiveType; + typedef DiagonalMatrix ScalingType; + typedef Translation TranslationType; + + AffineCompactType A1c; + A1c.matrix().setRandom(); + AffineCompactType A2c; + A2c.matrix().setRandom(); AffineType A1(A1c); AffineType A2(A2c); - ProjectiveType P1; P1.matrix().setRandom(); + ProjectiveType P1; + P1.matrix().setRandom(); VectorType v1 = VectorType::Random(); VectorType v2 = VectorType::Random(); HVectorType h1 = HVectorType::Random(); @@ -541,105 +536,109 @@ void transform_associativity(const RotationType& R) LinearType L = LinearType::Random(); MatrixType M = MatrixType::Random(); - CALL_SUBTEST( transform_associativity2(A1c, A1, P1, A2, v2, h1) ); - CALL_SUBTEST( transform_associativity2(A1c, A1, P1, A2c, v2, h1) ); - CALL_SUBTEST( transform_associativity2(A1c, A1, P1, v1.asDiagonal(), v2, h1) ); - CALL_SUBTEST( transform_associativity2(A1c, A1, P1, ScalingType(v1), v2, h1) ); - CALL_SUBTEST( transform_associativity2(A1c, A1, P1, Scaling(v1), v2, h1) ); - CALL_SUBTEST( transform_associativity2(A1c, A1, P1, Scaling(s1), v2, h1) ); - CALL_SUBTEST( transform_associativity2(A1c, A1, P1, TranslationType(v1), v2, h1) ); - CALL_SUBTEST( transform_associativity_left(A1c, A1, P1, L, v2, h1) ); - CALL_SUBTEST( transform_associativity2(A1c, A1, P1, R, v2, h1) ); - - VERIFY_IS_APPROX( A1*(M*h1), (A1*M)*h1 ); - VERIFY_IS_APPROX( A1c*(M*h1), (A1c*M)*h1 ); - VERIFY_IS_APPROX( P1*(M*h1), (P1*M)*h1 ); - - VERIFY_IS_APPROX( M*(A1*h1), (M*A1)*h1 ); - VERIFY_IS_APPROX( M*(A1c*h1), (M*A1c)*h1 ); - VERIFY_IS_APPROX( M*(P1*h1), ((M*P1)*h1) ); + CALL_SUBTEST(transform_associativity2(A1c, A1, P1, A2, v2, h1)); + CALL_SUBTEST(transform_associativity2(A1c, A1, P1, A2c, v2, h1)); + CALL_SUBTEST(transform_associativity2(A1c, A1, P1, v1.asDiagonal(), v2, h1)); + CALL_SUBTEST(transform_associativity2(A1c, A1, P1, ScalingType(v1), v2, h1)); + CALL_SUBTEST(transform_associativity2(A1c, A1, P1, Scaling(v1), v2, h1)); + CALL_SUBTEST(transform_associativity2(A1c, A1, P1, Scaling(s1), v2, h1)); + CALL_SUBTEST(transform_associativity2(A1c, A1, P1, TranslationType(v1), v2, h1)); + CALL_SUBTEST(transform_associativity_left(A1c, A1, P1, L, v2, h1)); + CALL_SUBTEST(transform_associativity2(A1c, A1, P1, R, v2, h1)); + + VERIFY_IS_APPROX(A1 * (M * h1), (A1 * M) * h1); + VERIFY_IS_APPROX(A1c * (M * h1), (A1c * M) * h1); + VERIFY_IS_APPROX(P1 * (M * h1), (P1 * M) * h1); + + VERIFY_IS_APPROX(M * (A1 * h1), (M * A1) * h1); + VERIFY_IS_APPROX(M * (A1c * h1), (M * A1c) * h1); + VERIFY_IS_APPROX(M * (P1 * h1), ((M * P1) * h1)); } template void transform_alignment() { - typedef Transform Projective3a; - typedef Transform Projective3u; + typedef Transform Projective3a; + typedef Transform Projective3u; EIGEN_ALIGN_MAX Scalar array1[16]; EIGEN_ALIGN_MAX Scalar array2[16]; - EIGEN_ALIGN_MAX Scalar array3[16+1]; - Scalar* array3u = array3+1; + EIGEN_ALIGN_MAX Scalar array3[16 + 1]; + Scalar *array3u = array3 + 1; + + Projective3a *p1 = ::new (reinterpret_cast(array1)) Projective3a; + Projective3u *p2 = ::new (reinterpret_cast(array2)) Projective3u; + Projective3u *p3 = ::new (reinterpret_cast(array3u)) Projective3u; - Projective3a *p1 = ::new(reinterpret_cast(array1)) Projective3a; - Projective3u *p2 = ::new(reinterpret_cast(array2)) Projective3u; - Projective3u *p3 = ::new(reinterpret_cast(array3u)) Projective3u; - p1->matrix().setRandom(); *p2 = *p1; *p3 = *p1; VERIFY_IS_APPROX(p1->matrix(), p2->matrix()); VERIFY_IS_APPROX(p1->matrix(), p3->matrix()); - - VERIFY_IS_APPROX( (*p1) * (*p1), (*p2)*(*p3)); - - #if defined(EIGEN_VECTORIZE) && EIGEN_MAX_STATIC_ALIGN_BYTES>0 - if(internal::packet_traits::Vectorizable) - VERIFY_RAISES_ASSERT((::new(reinterpret_cast(array3u)) Projective3a)); - #endif + + VERIFY_IS_APPROX((*p1) * (*p1), (*p2) * (*p3)); + +#if defined(EIGEN_VECTORIZE) && EIGEN_MAX_STATIC_ALIGN_BYTES > 0 + if (internal::packet_traits::Vectorizable) + VERIFY_RAISES_ASSERT((::new (reinterpret_cast(array3u)) Projective3a)); +#endif } template void transform_products() { - typedef Matrix Mat; - typedef Transform Proj; - typedef Transform Aff; - typedef Transform AffC; - - Proj p; p.matrix().setRandom(); - Aff a; a.linear().setRandom(); a.translation().setRandom(); + typedef Matrix Mat; + typedef Transform Proj; + typedef Transform Aff; + typedef Transform AffC; + + Proj p; + p.matrix().setRandom(); + Aff a; + a.linear().setRandom(); + a.translation().setRandom(); AffC ac = a; Mat p_m(p.matrix()), a_m(a.matrix()); - VERIFY_IS_APPROX((p*p).matrix(), p_m*p_m); - VERIFY_IS_APPROX((a*a).matrix(), a_m*a_m); - VERIFY_IS_APPROX((p*a).matrix(), p_m*a_m); - VERIFY_IS_APPROX((a*p).matrix(), a_m*p_m); - VERIFY_IS_APPROX((ac*a).matrix(), a_m*a_m); - VERIFY_IS_APPROX((a*ac).matrix(), a_m*a_m); - VERIFY_IS_APPROX((p*ac).matrix(), p_m*a_m); - VERIFY_IS_APPROX((ac*p).matrix(), a_m*p_m); + VERIFY_IS_APPROX((p * p).matrix(), p_m * p_m); + VERIFY_IS_APPROX((a * a).matrix(), a_m * a_m); + VERIFY_IS_APPROX((p * a).matrix(), p_m * a_m); + VERIFY_IS_APPROX((a * p).matrix(), a_m * p_m); + VERIFY_IS_APPROX((ac * a).matrix(), a_m * a_m); + VERIFY_IS_APPROX((a * ac).matrix(), a_m * a_m); + VERIFY_IS_APPROX((p * ac).matrix(), p_m * a_m); + VERIFY_IS_APPROX((ac * p).matrix(), a_m * p_m); } void test_geo_transformations() { - for(int i = 0; i < g_repeat; i++) { - CALL_SUBTEST_1(( transformations() )); - CALL_SUBTEST_1(( non_projective_only() )); - - CALL_SUBTEST_2(( transformations() )); - CALL_SUBTEST_2(( non_projective_only() )); - CALL_SUBTEST_2(( transform_alignment() )); - - CALL_SUBTEST_3(( transformations() )); - CALL_SUBTEST_3(( transformations() )); - CALL_SUBTEST_3(( transform_alignment() )); - - CALL_SUBTEST_4(( transformations() )); - CALL_SUBTEST_4(( non_projective_only() )); - - CALL_SUBTEST_5(( transformations() )); - CALL_SUBTEST_5(( non_projective_only() )); - - CALL_SUBTEST_6(( transformations() )); - CALL_SUBTEST_6(( transformations() )); - - - CALL_SUBTEST_7(( transform_products() )); - CALL_SUBTEST_7(( transform_products() )); - - CALL_SUBTEST_8(( transform_associativity(Rotation2D(internal::random()*double(EIGEN_PI))) )); - CALL_SUBTEST_8(( transform_associativity(Quaterniond::UnitRandom()) )); + for (int i = 0; i < g_repeat; i++) { + CALL_SUBTEST_1((transformations())); + CALL_SUBTEST_1((non_projective_only())); + + CALL_SUBTEST_2((transformations())); + CALL_SUBTEST_2((non_projective_only())); + CALL_SUBTEST_2((transform_alignment())); + + CALL_SUBTEST_3((transformations())); + CALL_SUBTEST_3((transformations())); + CALL_SUBTEST_3((transform_alignment())); + + CALL_SUBTEST_4((transformations())); + CALL_SUBTEST_4((non_projective_only())); + + CALL_SUBTEST_5((transformations())); + CALL_SUBTEST_5((non_projective_only())); + + CALL_SUBTEST_6((transformations())); + CALL_SUBTEST_6((transformations())); + + + CALL_SUBTEST_7((transform_products())); + CALL_SUBTEST_7((transform_products())); + + CALL_SUBTEST_8(( + transform_associativity(Rotation2D(internal::random() * double(EIGEN_PI))))); + CALL_SUBTEST_8((transform_associativity(Quaterniond::UnitRandom()))); } } diff --git a/filmulator-gui/core/nlmeans/eigen/test/half_float.cpp b/filmulator-gui/core/nlmeans/eigen/test/half_float.cpp index b37b8190..c9cd875f 100644 --- a/filmulator-gui/core/nlmeans/eigen/test/half_float.cpp +++ b/filmulator-gui/core/nlmeans/eigen/test/half_float.cpp @@ -33,7 +33,7 @@ void test_conversion() VERIFY_IS_EQUAL(half(0.0f).x, 0x0000); VERIFY_IS_EQUAL(half(-0.0f).x, 0x8000); VERIFY_IS_EQUAL(half(65504.0f).x, 0x7bff); - VERIFY_IS_EQUAL(half(65536.0f).x, 0x7c00); // Becomes infinity. + VERIFY_IS_EQUAL(half(65536.0f).x, 0x7c00);// Becomes infinity. // Denormals. VERIFY_IS_EQUAL(half(-5.96046e-08f).x, 0x8001); @@ -68,7 +68,7 @@ void test_conversion() VERIFY_IS_APPROX(float(half(__half_raw(0x0002))), 1.19209e-07f); // NaNs and infinities. - VERIFY(!(numext::isinf)(float(half(65504.0f)))); // Largest finite number. + VERIFY(!(numext::isinf)(float(half(65504.0f))));// Largest finite number. VERIFY(!(numext::isnan)(float(half(0.0f)))); VERIFY((numext::isinf)(float(half(__half_raw(0xfc00))))); VERIFY((numext::isnan)(float(half(__half_raw(0xfc01))))); @@ -100,24 +100,32 @@ void test_conversion() void test_numtraits() { - std::cout << "epsilon = " << NumTraits::epsilon() << " (0x" << std::hex << NumTraits::epsilon().x << ")" << std::endl; - std::cout << "highest = " << NumTraits::highest() << " (0x" << std::hex << NumTraits::highest().x << ")" << std::endl; - std::cout << "lowest = " << NumTraits::lowest() << " (0x" << std::hex << NumTraits::lowest().x << ")" << std::endl; - std::cout << "min = " << (std::numeric_limits::min)() << " (0x" << std::hex << half((std::numeric_limits::min)()).x << ")" << std::endl; - std::cout << "denorm min = " << (std::numeric_limits::denorm_min)() << " (0x" << std::hex << half((std::numeric_limits::denorm_min)()).x << ")" << std::endl; - std::cout << "infinity = " << NumTraits::infinity() << " (0x" << std::hex << NumTraits::infinity().x << ")" << std::endl; - std::cout << "quiet nan = " << NumTraits::quiet_NaN() << " (0x" << std::hex << NumTraits::quiet_NaN().x << ")" << std::endl; - std::cout << "signaling nan = " << std::numeric_limits::signaling_NaN() << " (0x" << std::hex << std::numeric_limits::signaling_NaN().x << ")" << std::endl; + std::cout << "epsilon = " << NumTraits::epsilon() << " (0x" << std::hex << NumTraits::epsilon().x + << ")" << std::endl; + std::cout << "highest = " << NumTraits::highest() << " (0x" << std::hex << NumTraits::highest().x + << ")" << std::endl; + std::cout << "lowest = " << NumTraits::lowest() << " (0x" << std::hex << NumTraits::lowest().x + << ")" << std::endl; + std::cout << "min = " << (std::numeric_limits::min)() << " (0x" << std::hex + << half((std::numeric_limits::min)()).x << ")" << std::endl; + std::cout << "denorm min = " << (std::numeric_limits::denorm_min)() << " (0x" << std::hex + << half((std::numeric_limits::denorm_min)()).x << ")" << std::endl; + std::cout << "infinity = " << NumTraits::infinity() << " (0x" << std::hex << NumTraits::infinity().x + << ")" << std::endl; + std::cout << "quiet nan = " << NumTraits::quiet_NaN() << " (0x" << std::hex + << NumTraits::quiet_NaN().x << ")" << std::endl; + std::cout << "signaling nan = " << std::numeric_limits::signaling_NaN() << " (0x" << std::hex + << std::numeric_limits::signaling_NaN().x << ")" << std::endl; VERIFY(NumTraits::IsSigned); - VERIFY_IS_EQUAL( std::numeric_limits::infinity().x, half(std::numeric_limits::infinity()).x ); - VERIFY_IS_EQUAL( std::numeric_limits::quiet_NaN().x, half(std::numeric_limits::quiet_NaN()).x ); - VERIFY_IS_EQUAL( std::numeric_limits::signaling_NaN().x, half(std::numeric_limits::signaling_NaN()).x ); - VERIFY( (std::numeric_limits::min)() > half(0.f) ); - VERIFY( (std::numeric_limits::denorm_min)() > half(0.f) ); - VERIFY( (std::numeric_limits::min)()/half(2) > half(0.f) ); - VERIFY_IS_EQUAL( (std::numeric_limits::denorm_min)()/half(2), half(0.f) ); + VERIFY_IS_EQUAL(std::numeric_limits::infinity().x, half(std::numeric_limits::infinity()).x); + VERIFY_IS_EQUAL(std::numeric_limits::quiet_NaN().x, half(std::numeric_limits::quiet_NaN()).x); + VERIFY_IS_EQUAL(std::numeric_limits::signaling_NaN().x, half(std::numeric_limits::signaling_NaN()).x); + VERIFY((std::numeric_limits::min)() > half(0.f)); + VERIFY((std::numeric_limits::denorm_min)() > half(0.f)); + VERIFY((std::numeric_limits::min)() / half(2) > half(0.f)); + VERIFY_IS_EQUAL((std::numeric_limits::denorm_min)() / half(2), half(0.f)); } void test_arithmetic() @@ -217,40 +225,40 @@ void test_trigonometric_functions() VERIFY_IS_APPROX(numext::cos(half(0.0f)), half(cosf(0.0f))); VERIFY_IS_APPROX(cos(half(0.0f)), half(cosf(0.0f))); VERIFY_IS_APPROX(numext::cos(half(EIGEN_PI)), half(cosf(EIGEN_PI))); - //VERIFY_IS_APPROX(numext::cos(half(EIGEN_PI/2)), half(cosf(EIGEN_PI/2))); - //VERIFY_IS_APPROX(numext::cos(half(3*EIGEN_PI/2)), half(cosf(3*EIGEN_PI/2))); + // VERIFY_IS_APPROX(numext::cos(half(EIGEN_PI/2)), half(cosf(EIGEN_PI/2))); + // VERIFY_IS_APPROX(numext::cos(half(3*EIGEN_PI/2)), half(cosf(3*EIGEN_PI/2))); VERIFY_IS_APPROX(numext::cos(half(3.5f)), half(cosf(3.5f))); VERIFY_IS_APPROX(numext::sin(half(0.0f)), half(sinf(0.0f))); VERIFY_IS_APPROX(sin(half(0.0f)), half(sinf(0.0f))); // VERIFY_IS_APPROX(numext::sin(half(EIGEN_PI)), half(sinf(EIGEN_PI))); - VERIFY_IS_APPROX(numext::sin(half(EIGEN_PI/2)), half(sinf(EIGEN_PI/2))); - VERIFY_IS_APPROX(numext::sin(half(3*EIGEN_PI/2)), half(sinf(3*EIGEN_PI/2))); + VERIFY_IS_APPROX(numext::sin(half(EIGEN_PI / 2)), half(sinf(EIGEN_PI / 2))); + VERIFY_IS_APPROX(numext::sin(half(3 * EIGEN_PI / 2)), half(sinf(3 * EIGEN_PI / 2))); VERIFY_IS_APPROX(numext::sin(half(3.5f)), half(sinf(3.5f))); VERIFY_IS_APPROX(numext::tan(half(0.0f)), half(tanf(0.0f))); VERIFY_IS_APPROX(tan(half(0.0f)), half(tanf(0.0f))); // VERIFY_IS_APPROX(numext::tan(half(EIGEN_PI)), half(tanf(EIGEN_PI))); // VERIFY_IS_APPROX(numext::tan(half(EIGEN_PI/2)), half(tanf(EIGEN_PI/2))); - //VERIFY_IS_APPROX(numext::tan(half(3*EIGEN_PI/2)), half(tanf(3*EIGEN_PI/2))); + // VERIFY_IS_APPROX(numext::tan(half(3*EIGEN_PI/2)), half(tanf(3*EIGEN_PI/2))); VERIFY_IS_APPROX(numext::tan(half(3.5f)), half(tanf(3.5f))); } void test_array() { - typedef Array ArrayXh; - Index size = internal::random(1,10); - Index i = internal::random(0,size-1); + typedef Array ArrayXh; + Index size = internal::random(1, 10); + Index i = internal::random(0, size - 1); ArrayXh a1 = ArrayXh::Random(size), a2 = ArrayXh::Random(size); - VERIFY_IS_APPROX( a1+a1, half(2)*a1 ); - VERIFY( (a1.abs() >= half(0)).all() ); - VERIFY_IS_APPROX( (a1*a1).sqrt(), a1.abs() ); + VERIFY_IS_APPROX(a1 + a1, half(2) * a1); + VERIFY((a1.abs() >= half(0)).all()); + VERIFY_IS_APPROX((a1 * a1).sqrt(), a1.abs()); - VERIFY( ((a1.min)(a2) <= (a1.max)(a2)).all() ); + VERIFY(((a1.min)(a2) <= (a1.max)(a2)).all()); a1(i) = half(-10.); - VERIFY_IS_EQUAL( a1.minCoeff(), half(-10.) ); + VERIFY_IS_EQUAL(a1.minCoeff(), half(-10.)); a1(i) = half(10.); - VERIFY_IS_EQUAL( a1.maxCoeff(), half(10.) ); + VERIFY_IS_EQUAL(a1.maxCoeff(), half(10.)); std::stringstream ss; ss << a1; diff --git a/filmulator-gui/core/nlmeans/eigen/test/hessenberg.cpp b/filmulator-gui/core/nlmeans/eigen/test/hessenberg.cpp index 96bc19e2..abb95a8c 100644 --- a/filmulator-gui/core/nlmeans/eigen/test/hessenberg.cpp +++ b/filmulator-gui/core/nlmeans/eigen/test/hessenberg.cpp @@ -11,21 +11,19 @@ #include "main.h" #include -template void hessenberg(int size = Size) +template void hessenberg(int size = Size) { - typedef Matrix MatrixType; + typedef Matrix MatrixType; // Test basic functionality: A = U H U* and H is Hessenberg - for(int counter = 0; counter < g_repeat; ++counter) { - MatrixType m = MatrixType::Random(size,size); + for (int counter = 0; counter < g_repeat; ++counter) { + MatrixType m = MatrixType::Random(size, size); HessenbergDecomposition hess(m); MatrixType Q = hess.matrixQ(); MatrixType H = hess.matrixH(); VERIFY_IS_APPROX(m, Q * H * Q.adjoint()); - for(int row = 2; row < size; ++row) { - for(int col = 0; col < row-1; ++col) { - VERIFY(H(row,col) == (typename MatrixType::Scalar)0); - } + for (int row = 2; row < size; ++row) { + for (int col = 0; col < row - 1; ++col) { VERIFY(H(row, col) == (typename MatrixType::Scalar)0); } } } @@ -36,26 +34,26 @@ template void hessenberg(int size = Size) HessenbergDecomposition cs2(A); VERIFY_IS_EQUAL(cs1.matrixH().eval(), cs2.matrixH().eval()); MatrixType cs1Q = cs1.matrixQ(); - MatrixType cs2Q = cs2.matrixQ(); + MatrixType cs2Q = cs2.matrixQ(); VERIFY_IS_EQUAL(cs1Q, cs2Q); // Test assertions for when used uninitialized HessenbergDecomposition hessUninitialized; - VERIFY_RAISES_ASSERT( hessUninitialized.matrixH() ); - VERIFY_RAISES_ASSERT( hessUninitialized.matrixQ() ); - VERIFY_RAISES_ASSERT( hessUninitialized.householderCoefficients() ); - VERIFY_RAISES_ASSERT( hessUninitialized.packedMatrix() ); + VERIFY_RAISES_ASSERT(hessUninitialized.matrixH()); + VERIFY_RAISES_ASSERT(hessUninitialized.matrixQ()); + VERIFY_RAISES_ASSERT(hessUninitialized.householderCoefficients()); + VERIFY_RAISES_ASSERT(hessUninitialized.packedMatrix()); // TODO: Add tests for packedMatrix() and householderCoefficients() } void test_hessenberg() { - CALL_SUBTEST_1(( hessenberg,1>() )); - CALL_SUBTEST_2(( hessenberg,2>() )); - CALL_SUBTEST_3(( hessenberg,4>() )); - CALL_SUBTEST_4(( hessenberg(internal::random(1,EIGEN_TEST_MAX_SIZE)) )); - CALL_SUBTEST_5(( hessenberg,Dynamic>(internal::random(1,EIGEN_TEST_MAX_SIZE)) )); + CALL_SUBTEST_1((hessenberg, 1>())); + CALL_SUBTEST_2((hessenberg, 2>())); + CALL_SUBTEST_3((hessenberg, 4>())); + CALL_SUBTEST_4((hessenberg(internal::random(1, EIGEN_TEST_MAX_SIZE)))); + CALL_SUBTEST_5((hessenberg, Dynamic>(internal::random(1, EIGEN_TEST_MAX_SIZE)))); // Test problem size constructors CALL_SUBTEST_6(HessenbergDecomposition(10)); diff --git a/filmulator-gui/core/nlmeans/eigen/test/householder.cpp b/filmulator-gui/core/nlmeans/eigen/test/householder.cpp index e70b7ea2..b9ef5362 100644 --- a/filmulator-gui/core/nlmeans/eigen/test/householder.cpp +++ b/filmulator-gui/core/nlmeans/eigen/test/householder.cpp @@ -10,7 +10,7 @@ #include "main.h" #include -template void householder(const MatrixType& m) +template void householder(const MatrixType &m) { static bool even = true; even = !even; @@ -29,9 +29,10 @@ template void householder(const MatrixType& m) typedef Matrix HCoeffsVectorType; typedef Matrix TMatrixType; - - Matrix _tmp((std::max)(rows,cols)); - Scalar* tmp = &_tmp.coeffRef(0,0); + + Matrix _tmp( + (std::max)(rows, cols)); + Scalar *tmp = &_tmp.coeffRef(0, 0); Scalar beta; RealScalar alpha; @@ -40,75 +41,74 @@ template void householder(const MatrixType& m) VectorType v1 = VectorType::Random(rows), v2; v2 = v1; v1.makeHouseholder(essential, beta, alpha); - v1.applyHouseholderOnTheLeft(essential,beta,tmp); + v1.applyHouseholderOnTheLeft(essential, beta, tmp); VERIFY_IS_APPROX(v1.norm(), v2.norm()); - if(rows>=2) VERIFY_IS_MUCH_SMALLER_THAN(v1.tail(rows-1).norm(), v1.norm()); + if (rows >= 2) VERIFY_IS_MUCH_SMALLER_THAN(v1.tail(rows - 1).norm(), v1.norm()); v1 = VectorType::Random(rows); v2 = v1; - v1.applyHouseholderOnTheLeft(essential,beta,tmp); + v1.applyHouseholderOnTheLeft(essential, beta, tmp); VERIFY_IS_APPROX(v1.norm(), v2.norm()); - MatrixType m1(rows, cols), - m2(rows, cols); + MatrixType m1(rows, cols), m2(rows, cols); v1 = VectorType::Random(rows); - if(even) v1.tail(rows-1).setZero(); + if (even) v1.tail(rows - 1).setZero(); m1.colwise() = v1; m2 = m1; m1.col(0).makeHouseholder(essential, beta, alpha); - m1.applyHouseholderOnTheLeft(essential,beta,tmp); + m1.applyHouseholderOnTheLeft(essential, beta, tmp); VERIFY_IS_APPROX(m1.norm(), m2.norm()); - if(rows>=2) VERIFY_IS_MUCH_SMALLER_THAN(m1.block(1,0,rows-1,cols).norm(), m1.norm()); - VERIFY_IS_MUCH_SMALLER_THAN(numext::imag(m1(0,0)), numext::real(m1(0,0))); - VERIFY_IS_APPROX(numext::real(m1(0,0)), alpha); + if (rows >= 2) VERIFY_IS_MUCH_SMALLER_THAN(m1.block(1, 0, rows - 1, cols).norm(), m1.norm()); + VERIFY_IS_MUCH_SMALLER_THAN(numext::imag(m1(0, 0)), numext::real(m1(0, 0))); + VERIFY_IS_APPROX(numext::real(m1(0, 0)), alpha); v1 = VectorType::Random(rows); - if(even) v1.tail(rows-1).setZero(); - SquareMatrixType m3(rows,rows), m4(rows,rows); + if (even) v1.tail(rows - 1).setZero(); + SquareMatrixType m3(rows, rows), m4(rows, rows); m3.rowwise() = v1.transpose(); m4 = m3; m3.row(0).makeHouseholder(essential, beta, alpha); - m3.applyHouseholderOnTheRight(essential,beta,tmp); + m3.applyHouseholderOnTheRight(essential, beta, tmp); VERIFY_IS_APPROX(m3.norm(), m4.norm()); - if(rows>=2) VERIFY_IS_MUCH_SMALLER_THAN(m3.block(0,1,rows,rows-1).norm(), m3.norm()); - VERIFY_IS_MUCH_SMALLER_THAN(numext::imag(m3(0,0)), numext::real(m3(0,0))); - VERIFY_IS_APPROX(numext::real(m3(0,0)), alpha); + if (rows >= 2) VERIFY_IS_MUCH_SMALLER_THAN(m3.block(0, 1, rows, rows - 1).norm(), m3.norm()); + VERIFY_IS_MUCH_SMALLER_THAN(numext::imag(m3(0, 0)), numext::real(m3(0, 0))); + VERIFY_IS_APPROX(numext::real(m3(0, 0)), alpha); // test householder sequence on the left with a shift - Index shift = internal::random(0, std::max(rows-2,0)); + Index shift = internal::random(0, std::max(rows - 2, 0)); Index brows = rows - shift; m1.setRandom(rows, cols); - HBlockMatrixType hbm = m1.block(shift,0,brows,cols); + HBlockMatrixType hbm = m1.block(shift, 0, brows, cols); HouseholderQR qr(hbm); m2 = m1; - m2.block(shift,0,brows,cols) = qr.matrixQR(); + m2.block(shift, 0, brows, cols) = qr.matrixQR(); HCoeffsVectorType hc = qr.hCoeffs().conjugate(); HouseholderSequence hseq(m2, hc); hseq.setLength(hc.size()).setShift(shift); VERIFY(hseq.length() == hc.size()); VERIFY(hseq.shift() == shift); - + MatrixType m5 = m2; - m5.block(shift,0,brows,cols).template triangularView().setZero(); - VERIFY_IS_APPROX(hseq * m5, m1); // test applying hseq directly + m5.block(shift, 0, brows, cols).template triangularView().setZero(); + VERIFY_IS_APPROX(hseq * m5, m1);// test applying hseq directly m3 = hseq; - VERIFY_IS_APPROX(m3 * m5, m1); // test evaluating hseq to a dense matrix, then applying - + VERIFY_IS_APPROX(m3 * m5, m1);// test evaluating hseq to a dense matrix, then applying + SquareMatrixType hseq_mat = hseq; SquareMatrixType hseq_mat_conj = hseq.conjugate(); SquareMatrixType hseq_mat_adj = hseq.adjoint(); SquareMatrixType hseq_mat_trans = hseq.transpose(); SquareMatrixType m6 = SquareMatrixType::Random(rows, rows); - VERIFY_IS_APPROX(hseq_mat.adjoint(), hseq_mat_adj); - VERIFY_IS_APPROX(hseq_mat.conjugate(), hseq_mat_conj); - VERIFY_IS_APPROX(hseq_mat.transpose(), hseq_mat_trans); - VERIFY_IS_APPROX(hseq_mat * m6, hseq_mat * m6); - VERIFY_IS_APPROX(hseq_mat.adjoint() * m6, hseq_mat_adj * m6); + VERIFY_IS_APPROX(hseq_mat.adjoint(), hseq_mat_adj); + VERIFY_IS_APPROX(hseq_mat.conjugate(), hseq_mat_conj); + VERIFY_IS_APPROX(hseq_mat.transpose(), hseq_mat_trans); + VERIFY_IS_APPROX(hseq_mat * m6, hseq_mat * m6); + VERIFY_IS_APPROX(hseq_mat.adjoint() * m6, hseq_mat_adj * m6); VERIFY_IS_APPROX(hseq_mat.conjugate() * m6, hseq_mat_conj * m6); VERIFY_IS_APPROX(hseq_mat.transpose() * m6, hseq_mat_trans * m6); - VERIFY_IS_APPROX(m6 * hseq_mat, m6 * hseq_mat); - VERIFY_IS_APPROX(m6 * hseq_mat.adjoint(), m6 * hseq_mat_adj); + VERIFY_IS_APPROX(m6 * hseq_mat, m6 * hseq_mat); + VERIFY_IS_APPROX(m6 * hseq_mat.adjoint(), m6 * hseq_mat_adj); VERIFY_IS_APPROX(m6 * hseq_mat.conjugate(), m6 * hseq_mat_conj); VERIFY_IS_APPROX(m6 * hseq_mat.transpose(), m6 * hseq_mat_trans); @@ -117,21 +117,24 @@ template void householder(const MatrixType& m) TMatrixType tm2 = m2.transpose(); HouseholderSequence rhseq(tm2, hc); rhseq.setLength(hc.size()).setShift(shift); - VERIFY_IS_APPROX(rhseq * m5, m1); // test applying rhseq directly + VERIFY_IS_APPROX(rhseq * m5, m1);// test applying rhseq directly m3 = rhseq; - VERIFY_IS_APPROX(m3 * m5, m1); // test evaluating rhseq to a dense matrix, then applying + VERIFY_IS_APPROX(m3 * m5, m1);// test evaluating rhseq to a dense matrix, then applying } void test_householder() { - for(int i = 0; i < g_repeat; i++) { - CALL_SUBTEST_1( householder(Matrix()) ); - CALL_SUBTEST_2( householder(Matrix()) ); - CALL_SUBTEST_3( householder(Matrix()) ); - CALL_SUBTEST_4( householder(Matrix()) ); - CALL_SUBTEST_5( householder(MatrixXd(internal::random(1,EIGEN_TEST_MAX_SIZE),internal::random(1,EIGEN_TEST_MAX_SIZE))) ); - CALL_SUBTEST_6( householder(MatrixXcf(internal::random(1,EIGEN_TEST_MAX_SIZE),internal::random(1,EIGEN_TEST_MAX_SIZE))) ); - CALL_SUBTEST_7( householder(MatrixXf(internal::random(1,EIGEN_TEST_MAX_SIZE),internal::random(1,EIGEN_TEST_MAX_SIZE))) ); - CALL_SUBTEST_8( householder(Matrix()) ); + for (int i = 0; i < g_repeat; i++) { + CALL_SUBTEST_1(householder(Matrix())); + CALL_SUBTEST_2(householder(Matrix())); + CALL_SUBTEST_3(householder(Matrix())); + CALL_SUBTEST_4(householder(Matrix())); + CALL_SUBTEST_5(householder( + MatrixXd(internal::random(1, EIGEN_TEST_MAX_SIZE), internal::random(1, EIGEN_TEST_MAX_SIZE)))); + CALL_SUBTEST_6(householder( + MatrixXcf(internal::random(1, EIGEN_TEST_MAX_SIZE), internal::random(1, EIGEN_TEST_MAX_SIZE)))); + CALL_SUBTEST_7(householder( + MatrixXf(internal::random(1, EIGEN_TEST_MAX_SIZE), internal::random(1, EIGEN_TEST_MAX_SIZE)))); + CALL_SUBTEST_8(householder(Matrix())); } } diff --git a/filmulator-gui/core/nlmeans/eigen/test/incomplete_cholesky.cpp b/filmulator-gui/core/nlmeans/eigen/test/incomplete_cholesky.cpp index 59ffe925..30cbdc13 100644 --- a/filmulator-gui/core/nlmeans/eigen/test/incomplete_cholesky.cpp +++ b/filmulator-gui/core/nlmeans/eigen/test/incomplete_cholesky.cpp @@ -14,50 +14,48 @@ template void test_incomplete_cholesky_T() { - typedef SparseMatrix SparseMatrixType; - ConjugateGradient > > cg_illt_lower_amd; - ConjugateGradient > > cg_illt_lower_nat; - ConjugateGradient > > cg_illt_upper_amd; - ConjugateGradient > > cg_illt_upper_nat; - ConjugateGradient > > cg_illt_uplo_amd; - + typedef SparseMatrix SparseMatrixType; + ConjugateGradient>> cg_illt_lower_amd; + ConjugateGradient>> cg_illt_lower_nat; + ConjugateGradient>> cg_illt_upper_amd; + ConjugateGradient>> cg_illt_upper_nat; + ConjugateGradient>> cg_illt_uplo_amd; - CALL_SUBTEST( check_sparse_spd_solving(cg_illt_lower_amd) ); - CALL_SUBTEST( check_sparse_spd_solving(cg_illt_lower_nat) ); - CALL_SUBTEST( check_sparse_spd_solving(cg_illt_upper_amd) ); - CALL_SUBTEST( check_sparse_spd_solving(cg_illt_upper_nat) ); - CALL_SUBTEST( check_sparse_spd_solving(cg_illt_uplo_amd) ); + + CALL_SUBTEST(check_sparse_spd_solving(cg_illt_lower_amd)); + CALL_SUBTEST(check_sparse_spd_solving(cg_illt_lower_nat)); + CALL_SUBTEST(check_sparse_spd_solving(cg_illt_upper_amd)); + CALL_SUBTEST(check_sparse_spd_solving(cg_illt_upper_nat)); + CALL_SUBTEST(check_sparse_spd_solving(cg_illt_uplo_amd)); } void test_incomplete_cholesky() { - CALL_SUBTEST_1(( test_incomplete_cholesky_T() )); - CALL_SUBTEST_2(( test_incomplete_cholesky_T, int>() )); - CALL_SUBTEST_3(( test_incomplete_cholesky_T() )); + CALL_SUBTEST_1((test_incomplete_cholesky_T())); + CALL_SUBTEST_2((test_incomplete_cholesky_T, int>())); + CALL_SUBTEST_3((test_incomplete_cholesky_T())); #ifdef EIGEN_TEST_PART_1 - // regression for bug 1150 - for(int N = 1; N<20; ++N) - { - Eigen::MatrixXd b( N, N ); + // regression for bug 1150 + for (int N = 1; N < 20; ++N) { + Eigen::MatrixXd b(N, N); b.setOnes(); - Eigen::SparseMatrix m( N, N ); - m.reserve(Eigen::VectorXi::Constant(N,4)); - for( int i = 0; i < N; ++i ) - { - m.insert( i, i ) = 1; - m.coeffRef( i, i / 2 ) = 2; - m.coeffRef( i, i / 3 ) = 2; - m.coeffRef( i, i / 4 ) = 2; + Eigen::SparseMatrix m(N, N); + m.reserve(Eigen::VectorXi::Constant(N, 4)); + for (int i = 0; i < N; ++i) { + m.insert(i, i) = 1; + m.coeffRef(i, i / 2) = 2; + m.coeffRef(i, i / 3) = 2; + m.coeffRef(i, i / 4) = 2; } Eigen::SparseMatrix A; A = m * m.transpose(); - Eigen::ConjugateGradient, - Eigen::Lower | Eigen::Upper, - Eigen::IncompleteCholesky > solver( A ); + Eigen:: + ConjugateGradient, Eigen::Lower | Eigen::Upper, Eigen::IncompleteCholesky> + solver(A); VERIFY(solver.preconditioner().info() == Eigen::Success); VERIFY(solver.info() == Eigen::Success); } diff --git a/filmulator-gui/core/nlmeans/eigen/test/inplace_decomposition.cpp b/filmulator-gui/core/nlmeans/eigen/test/inplace_decomposition.cpp index 92d0d91b..1e7fbcd8 100644 --- a/filmulator-gui/core/nlmeans/eigen/test/inplace_decomposition.cpp +++ b/filmulator-gui/core/nlmeans/eigen/test/inplace_decomposition.cpp @@ -8,27 +8,28 @@ // with this file, You can obtain one at http://mozilla.org/MPL/2.0/. #include "main.h" -#include #include +#include #include // This file test inplace decomposition through Ref<>, as supported by Cholesky, LU, and QR decompositions. -template void inplace(bool square = false, bool SPD = false) +template void inplace(bool square = false, bool SPD = false) { typedef typename MatrixType::Scalar Scalar; typedef Matrix RhsType; typedef Matrix ResType; - Index rows = MatrixType::RowsAtCompileTime==Dynamic ? internal::random(2,EIGEN_TEST_MAX_SIZE/2) : Index(MatrixType::RowsAtCompileTime); - Index cols = MatrixType::ColsAtCompileTime==Dynamic ? (square?rows:internal::random(2,rows)) : Index(MatrixType::ColsAtCompileTime); + Index rows = MatrixType::RowsAtCompileTime == Dynamic ? internal::random(2, EIGEN_TEST_MAX_SIZE / 2) + : Index(MatrixType::RowsAtCompileTime); + Index cols = MatrixType::ColsAtCompileTime == Dynamic ? (square ? rows : internal::random(2, rows)) + : Index(MatrixType::ColsAtCompileTime); - MatrixType A = MatrixType::Random(rows,cols); + MatrixType A = MatrixType::Random(rows, cols); RhsType b = RhsType::Random(rows); ResType x(cols); - if(SPD) - { + if (SPD) { assert(square); A.topRows(cols) = A.topRows(cols).adjoint() * A.topRows(cols); A.diagonal().array() += 1e-3; @@ -40,71 +41,62 @@ template void inplace(bool square = false, DecType dec(A); // Check that the content of A has been modified - VERIFY_IS_NOT_APPROX( A, A0 ); + VERIFY_IS_NOT_APPROX(A, A0); // Check that the decomposition is correct: - if(rows==cols) - { - VERIFY_IS_APPROX( A0 * (x = dec.solve(b)), b ); - } - else - { - VERIFY_IS_APPROX( A0.transpose() * A0 * (x = dec.solve(b)), A0.transpose() * b ); + if (rows == cols) { + VERIFY_IS_APPROX(A0 * (x = dec.solve(b)), b); + } else { + VERIFY_IS_APPROX(A0.transpose() * A0 * (x = dec.solve(b)), A0.transpose() * b); } // Check that modifying A breaks the current dec: A.setRandom(); - if(rows==cols) - { - VERIFY_IS_NOT_APPROX( A0 * (x = dec.solve(b)), b ); - } - else - { - VERIFY_IS_NOT_APPROX( A0.transpose() * A0 * (x = dec.solve(b)), A0.transpose() * b ); + if (rows == cols) { + VERIFY_IS_NOT_APPROX(A0 * (x = dec.solve(b)), b); + } else { + VERIFY_IS_NOT_APPROX(A0.transpose() * A0 * (x = dec.solve(b)), A0.transpose() * b); } // Check that calling compute(A1) does not modify A1: A = A0; dec.compute(A1); - VERIFY_IS_EQUAL(A0,A1); - VERIFY_IS_NOT_APPROX( A, A0 ); - if(rows==cols) - { - VERIFY_IS_APPROX( A0 * (x = dec.solve(b)), b ); - } - else - { - VERIFY_IS_APPROX( A0.transpose() * A0 * (x = dec.solve(b)), A0.transpose() * b ); + VERIFY_IS_EQUAL(A0, A1); + VERIFY_IS_NOT_APPROX(A, A0); + if (rows == cols) { + VERIFY_IS_APPROX(A0 * (x = dec.solve(b)), b); + } else { + VERIFY_IS_APPROX(A0.transpose() * A0 * (x = dec.solve(b)), A0.transpose() * b); } } void test_inplace_decomposition() { - EIGEN_UNUSED typedef Matrix Matrix43d; - for(int i = 0; i < g_repeat; i++) { - CALL_SUBTEST_1(( inplace >, MatrixXd>(true,true) )); - CALL_SUBTEST_1(( inplace >, Matrix4d>(true,true) )); + EIGEN_UNUSED typedef Matrix Matrix43d; + for (int i = 0; i < g_repeat; i++) { + CALL_SUBTEST_1((inplace>, MatrixXd>(true, true))); + CALL_SUBTEST_1((inplace>, Matrix4d>(true, true))); - CALL_SUBTEST_2(( inplace >, MatrixXd>(true,true) )); - CALL_SUBTEST_2(( inplace >, Matrix4d>(true,true) )); + CALL_SUBTEST_2((inplace>, MatrixXd>(true, true))); + CALL_SUBTEST_2((inplace>, Matrix4d>(true, true))); - CALL_SUBTEST_3(( inplace >, MatrixXd>(true,false) )); - CALL_SUBTEST_3(( inplace >, Matrix4d>(true,false) )); + CALL_SUBTEST_3((inplace>, MatrixXd>(true, false))); + CALL_SUBTEST_3((inplace>, Matrix4d>(true, false))); - CALL_SUBTEST_4(( inplace >, MatrixXd>(true,false) )); - CALL_SUBTEST_4(( inplace >, Matrix4d>(true,false) )); + CALL_SUBTEST_4((inplace>, MatrixXd>(true, false))); + CALL_SUBTEST_4((inplace>, Matrix4d>(true, false))); - CALL_SUBTEST_5(( inplace >, MatrixXd>(false,false) )); - CALL_SUBTEST_5(( inplace >, Matrix43d>(false,false) )); + CALL_SUBTEST_5((inplace>, MatrixXd>(false, false))); + CALL_SUBTEST_5((inplace>, Matrix43d>(false, false))); - CALL_SUBTEST_6(( inplace >, MatrixXd>(false,false) )); - CALL_SUBTEST_6(( inplace >, Matrix43d>(false,false) )); + CALL_SUBTEST_6((inplace>, MatrixXd>(false, false))); + CALL_SUBTEST_6((inplace>, Matrix43d>(false, false))); - CALL_SUBTEST_7(( inplace >, MatrixXd>(false,false) )); - CALL_SUBTEST_7(( inplace >, Matrix43d>(false,false) )); + CALL_SUBTEST_7((inplace>, MatrixXd>(false, false))); + CALL_SUBTEST_7((inplace>, Matrix43d>(false, false))); - CALL_SUBTEST_8(( inplace >, MatrixXd>(false,false) )); - CALL_SUBTEST_8(( inplace >, Matrix43d>(false,false) )); + CALL_SUBTEST_8((inplace>, MatrixXd>(false, false))); + CALL_SUBTEST_8((inplace>, Matrix43d>(false, false))); } } diff --git a/filmulator-gui/core/nlmeans/eigen/test/integer_types.cpp b/filmulator-gui/core/nlmeans/eigen/test/integer_types.cpp index 36295598..91ef14c7 100644 --- a/filmulator-gui/core/nlmeans/eigen/test/integer_types.cpp +++ b/filmulator-gui/core/nlmeans/eigen/test/integer_types.cpp @@ -12,11 +12,11 @@ #include "main.h" #undef VERIFY_IS_APPROX -#define VERIFY_IS_APPROX(a, b) VERIFY((a)==(b)); +#define VERIFY_IS_APPROX(a, b) VERIFY((a) == (b)); #undef VERIFY_IS_NOT_APPROX -#define VERIFY_IS_NOT_APPROX(a, b) VERIFY((a)!=(b)); +#define VERIFY_IS_NOT_APPROX(a, b) VERIFY((a) != (b)); -template void signed_integer_type_tests(const MatrixType& m) +template void signed_integer_type_tests(const MatrixType &m) { typedef typename MatrixType::Scalar Scalar; @@ -26,27 +26,25 @@ template void signed_integer_type_tests(const MatrixType& m Index rows = m.rows(); Index cols = m.cols(); - MatrixType m1(rows, cols), - m2 = MatrixType::Random(rows, cols), - mzero = MatrixType::Zero(rows, cols); + MatrixType m1(rows, cols), m2 = MatrixType::Random(rows, cols), mzero = MatrixType::Zero(rows, cols); do { m1 = MatrixType::Random(rows, cols); - } while(m1 == mzero || m1 == m2); + } while (m1 == mzero || m1 == m2); // check linear structure Scalar s1; do { s1 = internal::random(); - } while(s1 == 0); + } while (s1 == 0); - VERIFY_IS_EQUAL(-(-m1), m1); - VERIFY_IS_EQUAL(-m2+m1+m2, m1); - VERIFY_IS_EQUAL((-m1+m2)*s1, -s1*m1+s1*m2); + VERIFY_IS_EQUAL(-(-m1), m1); + VERIFY_IS_EQUAL(-m2 + m1 + m2, m1); + VERIFY_IS_EQUAL((-m1 + m2) * s1, -s1 * m1 + s1 * m2); } -template void integer_type_tests(const MatrixType& m) +template void integer_type_tests(const MatrixType &m) { typedef typename MatrixType::Scalar Scalar; @@ -61,67 +59,64 @@ template void integer_type_tests(const MatrixType& m) // this test relies a lot on Random.h, and there's not much more that we can do // to test it, hence I consider that we will have tested Random.h - MatrixType m1(rows, cols), - m2 = MatrixType::Random(rows, cols), - m3(rows, cols), - mzero = MatrixType::Zero(rows, cols); + MatrixType m1(rows, cols), m2 = MatrixType::Random(rows, cols), m3(rows, cols), mzero = MatrixType::Zero(rows, cols); typedef Matrix SquareMatrixType; - SquareMatrixType identity = SquareMatrixType::Identity(rows, rows), - square = SquareMatrixType::Random(rows, rows); - VectorType v1(rows), - v2 = VectorType::Random(rows), - vzero = VectorType::Zero(rows); + SquareMatrixType identity = SquareMatrixType::Identity(rows, rows), square = SquareMatrixType::Random(rows, rows); + VectorType v1(rows), v2 = VectorType::Random(rows), vzero = VectorType::Zero(rows); do { m1 = MatrixType::Random(rows, cols); - } while(m1 == mzero || m1 == m2); + } while (m1 == mzero || m1 == m2); do { v1 = VectorType::Random(rows); - } while(v1 == vzero || v1 == v2); + } while (v1 == vzero || v1 == v2); - VERIFY_IS_APPROX( v1, v1); - VERIFY_IS_NOT_APPROX( v1, 2*v1); - VERIFY_IS_APPROX( vzero, v1-v1); - VERIFY_IS_APPROX( m1, m1); - VERIFY_IS_NOT_APPROX( m1, 2*m1); - VERIFY_IS_APPROX( mzero, m1-m1); + VERIFY_IS_APPROX(v1, v1); + VERIFY_IS_NOT_APPROX(v1, 2 * v1); + VERIFY_IS_APPROX(vzero, v1 - v1); + VERIFY_IS_APPROX(m1, m1); + VERIFY_IS_NOT_APPROX(m1, 2 * m1); + VERIFY_IS_APPROX(mzero, m1 - m1); - VERIFY_IS_APPROX(m3 = m1,m1); + VERIFY_IS_APPROX(m3 = m1, m1); MatrixType m4; - VERIFY_IS_APPROX(m4 = m1,m1); + VERIFY_IS_APPROX(m4 = m1, m1); m3.real() = m1.real(); - VERIFY_IS_APPROX(static_cast(m3).real(), static_cast(m1).real()); - VERIFY_IS_APPROX(static_cast(m3).real(), m1.real()); + VERIFY_IS_APPROX(static_cast(m3).real(), static_cast(m1).real()); + VERIFY_IS_APPROX(static_cast(m3).real(), m1.real()); // check == / != operators - VERIFY(m1==m1); - VERIFY(m1!=m2); - VERIFY(!(m1==m2)); - VERIFY(!(m1!=m1)); + VERIFY(m1 == m1); + VERIFY(m1 != m2); + VERIFY(!(m1 == m2)); + VERIFY(!(m1 != m1)); m1 = m2; - VERIFY(m1==m2); - VERIFY(!(m1!=m2)); + VERIFY(m1 == m2); + VERIFY(!(m1 != m2)); // check linear structure Scalar s1; do { s1 = internal::random(); - } while(s1 == 0); - - VERIFY_IS_EQUAL(m1+m1, 2*m1); - VERIFY_IS_EQUAL(m1+m2-m1, m2); - VERIFY_IS_EQUAL(m1*s1, s1*m1); - VERIFY_IS_EQUAL((m1+m2)*s1, s1*m1+s1*m2); - m3 = m2; m3 += m1; - VERIFY_IS_EQUAL(m3, m1+m2); - m3 = m2; m3 -= m1; - VERIFY_IS_EQUAL(m3, m2-m1); - m3 = m2; m3 *= s1; - VERIFY_IS_EQUAL(m3, s1*m2); + } while (s1 == 0); + + VERIFY_IS_EQUAL(m1 + m1, 2 * m1); + VERIFY_IS_EQUAL(m1 + m2 - m1, m2); + VERIFY_IS_EQUAL(m1 * s1, s1 * m1); + VERIFY_IS_EQUAL((m1 + m2) * s1, s1 * m1 + s1 * m2); + m3 = m2; + m3 += m1; + VERIFY_IS_EQUAL(m3, m1 + m2); + m3 = m2; + m3 -= m1; + VERIFY_IS_EQUAL(m3, m2 - m1); + m3 = m2; + m3 *= s1; + VERIFY_IS_EQUAL(m3, s1 * m2); // check matrix product. @@ -133,33 +128,33 @@ template void integer_type_tests(const MatrixType& m) void test_integer_types() { - for(int i = 0; i < g_repeat; i++) { - CALL_SUBTEST_1( integer_type_tests(Matrix()) ); - CALL_SUBTEST_1( integer_type_tests(Matrix()) ); + for (int i = 0; i < g_repeat; i++) { + CALL_SUBTEST_1(integer_type_tests(Matrix())); + CALL_SUBTEST_1(integer_type_tests(Matrix())); - CALL_SUBTEST_2( integer_type_tests(Matrix()) ); - CALL_SUBTEST_2( signed_integer_type_tests(Matrix()) ); + CALL_SUBTEST_2(integer_type_tests(Matrix())); + CALL_SUBTEST_2(signed_integer_type_tests(Matrix())); - CALL_SUBTEST_3( integer_type_tests(Matrix(2, 10)) ); - CALL_SUBTEST_3( signed_integer_type_tests(Matrix(2, 10)) ); + CALL_SUBTEST_3(integer_type_tests(Matrix(2, 10))); + CALL_SUBTEST_3(signed_integer_type_tests(Matrix(2, 10))); - CALL_SUBTEST_4( integer_type_tests(Matrix()) ); - CALL_SUBTEST_4( integer_type_tests(Matrix(20, 20)) ); + CALL_SUBTEST_4(integer_type_tests(Matrix())); + CALL_SUBTEST_4(integer_type_tests(Matrix(20, 20))); - CALL_SUBTEST_5( integer_type_tests(Matrix(7, 4)) ); - CALL_SUBTEST_5( signed_integer_type_tests(Matrix(7, 4)) ); + CALL_SUBTEST_5(integer_type_tests(Matrix(7, 4))); + CALL_SUBTEST_5(signed_integer_type_tests(Matrix(7, 4))); - CALL_SUBTEST_6( integer_type_tests(Matrix()) ); + CALL_SUBTEST_6(integer_type_tests(Matrix())); - CALL_SUBTEST_7( integer_type_tests(Matrix()) ); - CALL_SUBTEST_7( signed_integer_type_tests(Matrix()) ); + CALL_SUBTEST_7(integer_type_tests(Matrix())); + CALL_SUBTEST_7(signed_integer_type_tests(Matrix())); - CALL_SUBTEST_8( integer_type_tests(Matrix(1, 5)) ); + CALL_SUBTEST_8(integer_type_tests(Matrix(1, 5))); } #ifdef EIGEN_TEST_PART_9 VERIFY_IS_EQUAL(internal::scalar_div_cost::value, 8); VERIFY_IS_EQUAL(internal::scalar_div_cost::value, 8); - if(sizeof(long)>sizeof(int)) { + if (sizeof(long) > sizeof(int)) { VERIFY(int(internal::scalar_div_cost::value) > int(internal::scalar_div_cost::value)); VERIFY(int(internal::scalar_div_cost::value) > int(internal::scalar_div_cost::value)); } diff --git a/filmulator-gui/core/nlmeans/eigen/test/inverse.cpp b/filmulator-gui/core/nlmeans/eigen/test/inverse.cpp index be607cc8..34d06d0d 100644 --- a/filmulator-gui/core/nlmeans/eigen/test/inverse.cpp +++ b/filmulator-gui/core/nlmeans/eigen/test/inverse.cpp @@ -11,7 +11,7 @@ #include "main.h" #include -template void inverse(const MatrixType& m) +template void inverse(const MatrixType &m) { using std::abs; /* this test covers the following files: @@ -22,19 +22,17 @@ template void inverse(const MatrixType& m) typedef typename MatrixType::Scalar Scalar; - MatrixType m1(rows, cols), - m2(rows, cols), - identity = MatrixType::Identity(rows, rows); - createRandomPIMatrixOfRank(rows,rows,rows,m1); + MatrixType m1(rows, cols), m2(rows, cols), identity = MatrixType::Identity(rows, rows); + createRandomPIMatrixOfRank(rows, rows, rows, m1); m2 = m1.inverse(); - VERIFY_IS_APPROX(m1, m2.inverse() ); + VERIFY_IS_APPROX(m1, m2.inverse()); - VERIFY_IS_APPROX((Scalar(2)*m2).inverse(), m2.inverse()*Scalar(0.5)); + VERIFY_IS_APPROX((Scalar(2) * m2).inverse(), m2.inverse() * Scalar(0.5)); - VERIFY_IS_APPROX(identity, m1.inverse() * m1 ); - VERIFY_IS_APPROX(identity, m1 * m1.inverse() ); + VERIFY_IS_APPROX(identity, m1.inverse() * m1); + VERIFY_IS_APPROX(identity, m1 * m1.inverse()); - VERIFY_IS_APPROX(m1, m1.inverse().inverse() ); + VERIFY_IS_APPROX(m1, m1.inverse().inverse()); // since for the general case we implement separately row-major and col-major, test that VERIFY_IS_APPROX(MatrixType(m1.transpose().inverse()), MatrixType(m1.inverse().transpose())); @@ -42,77 +40,75 @@ template void inverse(const MatrixType& m) #if !defined(EIGEN_TEST_PART_5) && !defined(EIGEN_TEST_PART_6) typedef typename NumTraits::Real RealScalar; typedef Matrix VectorType; - - //computeInverseAndDetWithCheck tests - //First: an invertible matrix + + // computeInverseAndDetWithCheck tests + // First: an invertible matrix bool invertible; Scalar det; m2.setZero(); m1.computeInverseAndDetWithCheck(m2, det, invertible); VERIFY(invertible); - VERIFY_IS_APPROX(identity, m1*m2); + VERIFY_IS_APPROX(identity, m1 * m2); VERIFY_IS_APPROX(det, m1.determinant()); m2.setZero(); m1.computeInverseWithCheck(m2, invertible); VERIFY(invertible); - VERIFY_IS_APPROX(identity, m1*m2); + VERIFY_IS_APPROX(identity, m1 * m2); - //Second: a rank one matrix (not invertible, except for 1x1 matrices) + // Second: a rank one matrix (not invertible, except for 1x1 matrices) VectorType v3 = VectorType::Random(rows); - MatrixType m3 = v3*v3.transpose(), m4(rows,cols); + MatrixType m3 = v3 * v3.transpose(), m4(rows, cols); m3.computeInverseAndDetWithCheck(m4, det, invertible); - VERIFY( rows==1 ? invertible : !invertible ); - VERIFY_IS_MUCH_SMALLER_THAN(abs(det-m3.determinant()), RealScalar(1)); + VERIFY(rows == 1 ? invertible : !invertible); + VERIFY_IS_MUCH_SMALLER_THAN(abs(det - m3.determinant()), RealScalar(1)); m3.computeInverseWithCheck(m4, invertible); - VERIFY( rows==1 ? invertible : !invertible ); - + VERIFY(rows == 1 ? invertible : !invertible); + // check with submatrices { - Matrix m5; + Matrix m5; m5.setRandom(); - m5.topLeftCorner(rows,rows) = m1; - m2 = m5.template topLeftCorner().inverse(); - VERIFY_IS_APPROX( (m5.template topLeftCorner()), m2.inverse() ); + m5.topLeftCorner(rows, rows) = m1; + m2 = m5.template topLeftCorner().inverse(); + VERIFY_IS_APPROX( + (m5.template topLeftCorner()), m2.inverse()); } #endif // check in-place inversion - if(MatrixType::RowsAtCompileTime>=2 && MatrixType::RowsAtCompileTime<=4) - { + if (MatrixType::RowsAtCompileTime >= 2 && MatrixType::RowsAtCompileTime <= 4) { // in-place is forbidden VERIFY_RAISES_ASSERT(m1 = m1.inverse()); - } - else - { + } else { m2 = m1.inverse(); m1 = m1.inverse(); - VERIFY_IS_APPROX(m1,m2); + VERIFY_IS_APPROX(m1, m2); } } void test_inverse() { int s = 0; - for(int i = 0; i < g_repeat; i++) { - CALL_SUBTEST_1( inverse(Matrix()) ); - CALL_SUBTEST_2( inverse(Matrix2d()) ); - CALL_SUBTEST_3( inverse(Matrix3f()) ); - CALL_SUBTEST_4( inverse(Matrix4f()) ); - CALL_SUBTEST_4( inverse(Matrix()) ); - - s = internal::random(50,320); - CALL_SUBTEST_5( inverse(MatrixXf(s,s)) ); + for (int i = 0; i < g_repeat; i++) { + CALL_SUBTEST_1(inverse(Matrix())); + CALL_SUBTEST_2(inverse(Matrix2d())); + CALL_SUBTEST_3(inverse(Matrix3f())); + CALL_SUBTEST_4(inverse(Matrix4f())); + CALL_SUBTEST_4(inverse(Matrix())); + + s = internal::random(50, 320); + CALL_SUBTEST_5(inverse(MatrixXf(s, s))); TEST_SET_BUT_UNUSED_VARIABLE(s) - - s = internal::random(25,100); - CALL_SUBTEST_6( inverse(MatrixXcd(s,s)) ); + + s = internal::random(25, 100); + CALL_SUBTEST_6(inverse(MatrixXcd(s, s))); TEST_SET_BUT_UNUSED_VARIABLE(s) - - CALL_SUBTEST_7( inverse(Matrix4d()) ); - CALL_SUBTEST_7( inverse(Matrix()) ); - CALL_SUBTEST_8( inverse(Matrix4cd()) ); + CALL_SUBTEST_7(inverse(Matrix4d())); + CALL_SUBTEST_7(inverse(Matrix())); + + CALL_SUBTEST_8(inverse(Matrix4cd())); } } diff --git a/filmulator-gui/core/nlmeans/eigen/test/is_same_dense.cpp b/filmulator-gui/core/nlmeans/eigen/test/is_same_dense.cpp index 2c7838ce..8660e351 100644 --- a/filmulator-gui/core/nlmeans/eigen/test/is_same_dense.cpp +++ b/filmulator-gui/core/nlmeans/eigen/test/is_same_dense.cpp @@ -13,21 +13,21 @@ using internal::is_same_dense; void test_is_same_dense() { - typedef Matrix ColMatrixXd; - ColMatrixXd m1(10,10); + typedef Matrix ColMatrixXd; + ColMatrixXd m1(10, 10); Ref ref_m1(m1); Ref const_ref_m1(m1); - VERIFY(is_same_dense(m1,m1)); - VERIFY(is_same_dense(m1,ref_m1)); - VERIFY(is_same_dense(const_ref_m1,m1)); - VERIFY(is_same_dense(const_ref_m1,ref_m1)); - - VERIFY(is_same_dense(m1.block(0,0,m1.rows(),m1.cols()),m1)); - VERIFY(!is_same_dense(m1.row(0),m1.col(0))); - + VERIFY(is_same_dense(m1, m1)); + VERIFY(is_same_dense(m1, ref_m1)); + VERIFY(is_same_dense(const_ref_m1, m1)); + VERIFY(is_same_dense(const_ref_m1, ref_m1)); + + VERIFY(is_same_dense(m1.block(0, 0, m1.rows(), m1.cols()), m1)); + VERIFY(!is_same_dense(m1.row(0), m1.col(0))); + Ref const_ref_m1_row(m1.row(1)); - VERIFY(!is_same_dense(m1.row(1),const_ref_m1_row)); - + VERIFY(!is_same_dense(m1.row(1), const_ref_m1_row)); + Ref const_ref_m1_col(m1.col(1)); - VERIFY(is_same_dense(m1.col(1),const_ref_m1_col)); + VERIFY(is_same_dense(m1.col(1), const_ref_m1_col)); } diff --git a/filmulator-gui/core/nlmeans/eigen/test/jacobi.cpp b/filmulator-gui/core/nlmeans/eigen/test/jacobi.cpp index 319e4767..3bb9ecc4 100644 --- a/filmulator-gui/core/nlmeans/eigen/test/jacobi.cpp +++ b/filmulator-gui/core/nlmeans/eigen/test/jacobi.cpp @@ -11,16 +11,12 @@ #include "main.h" #include -template -void jacobi(const MatrixType& m = MatrixType()) +template void jacobi(const MatrixType &m = MatrixType()) { Index rows = m.rows(); Index cols = m.cols(); - enum { - RowsAtCompileTime = MatrixType::RowsAtCompileTime, - ColsAtCompileTime = MatrixType::ColsAtCompileTime - }; + enum { RowsAtCompileTime = MatrixType::RowsAtCompileTime, ColsAtCompileTime = MatrixType::ColsAtCompileTime }; typedef Matrix JacobiVector; @@ -31,10 +27,10 @@ void jacobi(const MatrixType& m = MatrixType()) JacobiRotation rot(c, s); { - Index p = internal::random(0, rows-1); + Index p = internal::random(0, rows - 1); Index q; do { - q = internal::random(0, rows-1); + q = internal::random(0, rows - 1); } while (q == p); MatrixType b = a; @@ -44,10 +40,10 @@ void jacobi(const MatrixType& m = MatrixType()) } { - Index p = internal::random(0, cols-1); + Index p = internal::random(0, cols - 1); Index q; do { - q = internal::random(0, cols-1); + q = internal::random(0, cols - 1); } while (q == p); MatrixType b = a; @@ -59,21 +55,22 @@ void jacobi(const MatrixType& m = MatrixType()) void test_jacobi() { - for(int i = 0; i < g_repeat; i++) { - CALL_SUBTEST_1(( jacobi() )); - CALL_SUBTEST_2(( jacobi() )); - CALL_SUBTEST_3(( jacobi() )); - CALL_SUBTEST_3(( jacobi >() )); + for (int i = 0; i < g_repeat; i++) { + CALL_SUBTEST_1((jacobi())); + CALL_SUBTEST_2((jacobi())); + CALL_SUBTEST_3((jacobi())); + CALL_SUBTEST_3((jacobi>())); + + int r = internal::random(2, internal::random(1, EIGEN_TEST_MAX_SIZE) / 2), + c = internal::random(2, internal::random(1, EIGEN_TEST_MAX_SIZE) / 2); + CALL_SUBTEST_4((jacobi(MatrixXf(r, c)))); + CALL_SUBTEST_5((jacobi(MatrixXcd(r, c)))); + CALL_SUBTEST_5((jacobi>(MatrixXcd(r, c)))); + // complex is really important to test as it is the only way to cover conjugation issues in certain unaligned + // paths + CALL_SUBTEST_6((jacobi(MatrixXcf(r, c)))); + CALL_SUBTEST_6((jacobi>(MatrixXcf(r, c)))); - int r = internal::random(2, internal::random(1,EIGEN_TEST_MAX_SIZE)/2), - c = internal::random(2, internal::random(1,EIGEN_TEST_MAX_SIZE)/2); - CALL_SUBTEST_4(( jacobi(MatrixXf(r,c)) )); - CALL_SUBTEST_5(( jacobi(MatrixXcd(r,c)) )); - CALL_SUBTEST_5(( jacobi >(MatrixXcd(r,c)) )); - // complex is really important to test as it is the only way to cover conjugation issues in certain unaligned paths - CALL_SUBTEST_6(( jacobi(MatrixXcf(r,c)) )); - CALL_SUBTEST_6(( jacobi >(MatrixXcf(r,c)) )); - TEST_SET_BUT_UNUSED_VARIABLE(r); TEST_SET_BUT_UNUSED_VARIABLE(c); } diff --git a/filmulator-gui/core/nlmeans/eigen/test/jacobisvd.cpp b/filmulator-gui/core/nlmeans/eigen/test/jacobisvd.cpp index 64b86635..85435125 100644 --- a/filmulator-gui/core/nlmeans/eigen/test/jacobisvd.cpp +++ b/filmulator-gui/core/nlmeans/eigen/test/jacobisvd.cpp @@ -15,49 +15,44 @@ #include #define SVD_DEFAULT(M) JacobiSVD -#define SVD_FOR_MIN_NORM(M) JacobiSVD +#define SVD_FOR_MIN_NORM(M) JacobiSVD #include "svd_common.h" // Check all variants of JacobiSVD -template -void jacobisvd(const MatrixType& a = MatrixType(), bool pickrandom = true) +template void jacobisvd(const MatrixType &a = MatrixType(), bool pickrandom = true) { MatrixType m = a; - if(pickrandom) - svd_fill_random(m); - - CALL_SUBTEST(( svd_test_all_computation_options >(m, true) )); // check full only - CALL_SUBTEST(( svd_test_all_computation_options >(m, false) )); - CALL_SUBTEST(( svd_test_all_computation_options >(m, false) )); - if(m.rows()==m.cols()) - CALL_SUBTEST(( svd_test_all_computation_options >(m, false) )); + if (pickrandom) svd_fill_random(m); + + CALL_SUBTEST((svd_test_all_computation_options>( + m, true)));// check full only + CALL_SUBTEST((svd_test_all_computation_options>(m, false))); + CALL_SUBTEST((svd_test_all_computation_options>(m, false))); + if (m.rows() == m.cols()) + CALL_SUBTEST((svd_test_all_computation_options>(m, false))); } -template void jacobisvd_verify_assert(const MatrixType& m) +template void jacobisvd_verify_assert(const MatrixType &m) { - svd_verify_assert >(m); + svd_verify_assert>(m); Index rows = m.rows(); Index cols = m.cols(); - enum { - ColsAtCompileTime = MatrixType::ColsAtCompileTime - }; + enum { ColsAtCompileTime = MatrixType::ColsAtCompileTime }; MatrixType a = MatrixType::Zero(rows, cols); a.setZero(); - if (ColsAtCompileTime == Dynamic) - { + if (ColsAtCompileTime == Dynamic) { JacobiSVD svd_fullqr; - VERIFY_RAISES_ASSERT(svd_fullqr.compute(a, ComputeFullU|ComputeThinV)) - VERIFY_RAISES_ASSERT(svd_fullqr.compute(a, ComputeThinU|ComputeThinV)) - VERIFY_RAISES_ASSERT(svd_fullqr.compute(a, ComputeThinU|ComputeFullV)) + VERIFY_RAISES_ASSERT(svd_fullqr.compute(a, ComputeFullU | ComputeThinV)) + VERIFY_RAISES_ASSERT(svd_fullqr.compute(a, ComputeThinU | ComputeThinV)) + VERIFY_RAISES_ASSERT(svd_fullqr.compute(a, ComputeThinU | ComputeFullV)) } } -template -void jacobisvd_method() +template void jacobisvd_method() { enum { Size = MatrixType::RowsAtCompileTime }; typedef typename MatrixType::RealScalar RealScalar; @@ -66,77 +61,83 @@ void jacobisvd_method() VERIFY_IS_APPROX(m.jacobiSvd().singularValues(), RealVecType::Ones()); VERIFY_RAISES_ASSERT(m.jacobiSvd().matrixU()); VERIFY_RAISES_ASSERT(m.jacobiSvd().matrixV()); - VERIFY_IS_APPROX(m.jacobiSvd(ComputeFullU|ComputeFullV).solve(m), m); + VERIFY_IS_APPROX(m.jacobiSvd(ComputeFullU | ComputeFullV).solve(m), m); } namespace Foo { // older compiler require a default constructor for Bar // cf: https://stackoverflow.com/questions/7411515/ -class Bar {public: Bar() {}}; -bool operator<(const Bar&, const Bar&) { return true; } -} +class Bar +{ +public: + Bar() {} +}; +bool operator<(const Bar &, const Bar &) { return true; } +}// namespace Foo // regression test for a very strange MSVC issue for which simply // including SVDBase.h messes up with std::max and custom scalar type void msvc_workaround() { const Foo::Bar a; const Foo::Bar b; - std::max EIGEN_NOT_A_MACRO (a,b); + std::max EIGEN_NOT_A_MACRO(a, b); } void test_jacobisvd() { - CALL_SUBTEST_3(( jacobisvd_verify_assert(Matrix3f()) )); - CALL_SUBTEST_4(( jacobisvd_verify_assert(Matrix4d()) )); - CALL_SUBTEST_7(( jacobisvd_verify_assert(MatrixXf(10,12)) )); - CALL_SUBTEST_8(( jacobisvd_verify_assert(MatrixXcd(7,5)) )); - + CALL_SUBTEST_3((jacobisvd_verify_assert(Matrix3f()))); + CALL_SUBTEST_4((jacobisvd_verify_assert(Matrix4d()))); + CALL_SUBTEST_7((jacobisvd_verify_assert(MatrixXf(10, 12)))); + CALL_SUBTEST_8((jacobisvd_verify_assert(MatrixXcd(7, 5)))); + CALL_SUBTEST_11(svd_all_trivial_2x2(jacobisvd)); CALL_SUBTEST_12(svd_all_trivial_2x2(jacobisvd)); - for(int i = 0; i < g_repeat; i++) { - CALL_SUBTEST_3(( jacobisvd() )); - CALL_SUBTEST_4(( jacobisvd() )); - CALL_SUBTEST_5(( jacobisvd >() )); - CALL_SUBTEST_6(( jacobisvd >(Matrix(10,2)) )); + for (int i = 0; i < g_repeat; i++) { + CALL_SUBTEST_3((jacobisvd())); + CALL_SUBTEST_4((jacobisvd())); + CALL_SUBTEST_5((jacobisvd>())); + CALL_SUBTEST_6((jacobisvd>(Matrix(10, 2)))); + + int r = internal::random(1, 30), c = internal::random(1, 30); - int r = internal::random(1, 30), - c = internal::random(1, 30); - TEST_SET_BUT_UNUSED_VARIABLE(r) TEST_SET_BUT_UNUSED_VARIABLE(c) - - CALL_SUBTEST_10(( jacobisvd(MatrixXd(r,c)) )); - CALL_SUBTEST_7(( jacobisvd(MatrixXf(r,c)) )); - CALL_SUBTEST_8(( jacobisvd(MatrixXcd(r,c)) )); - (void) r; - (void) c; + + CALL_SUBTEST_10((jacobisvd(MatrixXd(r, c)))); + CALL_SUBTEST_7((jacobisvd(MatrixXf(r, c)))); + CALL_SUBTEST_8((jacobisvd(MatrixXcd(r, c)))); + (void)r; + (void)c; // Test on inf/nan matrix - CALL_SUBTEST_7( (svd_inf_nan, MatrixXf>()) ); - CALL_SUBTEST_10( (svd_inf_nan, MatrixXd>()) ); + CALL_SUBTEST_7((svd_inf_nan, MatrixXf>())); + CALL_SUBTEST_10((svd_inf_nan, MatrixXd>())); // bug1395 test compile-time vectors as input - CALL_SUBTEST_13(( jacobisvd_verify_assert(Matrix()) )); - CALL_SUBTEST_13(( jacobisvd_verify_assert(Matrix()) )); - CALL_SUBTEST_13(( jacobisvd_verify_assert(Matrix(r)) )); - CALL_SUBTEST_13(( jacobisvd_verify_assert(Matrix(c)) )); + CALL_SUBTEST_13((jacobisvd_verify_assert(Matrix()))); + CALL_SUBTEST_13((jacobisvd_verify_assert(Matrix()))); + CALL_SUBTEST_13((jacobisvd_verify_assert(Matrix(r)))); + CALL_SUBTEST_13((jacobisvd_verify_assert(Matrix(c)))); } - CALL_SUBTEST_7(( jacobisvd(MatrixXf(internal::random(EIGEN_TEST_MAX_SIZE/4, EIGEN_TEST_MAX_SIZE/2), internal::random(EIGEN_TEST_MAX_SIZE/4, EIGEN_TEST_MAX_SIZE/2))) )); - CALL_SUBTEST_8(( jacobisvd(MatrixXcd(internal::random(EIGEN_TEST_MAX_SIZE/4, EIGEN_TEST_MAX_SIZE/3), internal::random(EIGEN_TEST_MAX_SIZE/4, EIGEN_TEST_MAX_SIZE/3))) )); + CALL_SUBTEST_7((jacobisvd(MatrixXf(internal::random(EIGEN_TEST_MAX_SIZE / 4, EIGEN_TEST_MAX_SIZE / 2), + internal::random(EIGEN_TEST_MAX_SIZE / 4, EIGEN_TEST_MAX_SIZE / 2))))); + CALL_SUBTEST_8( + (jacobisvd(MatrixXcd(internal::random(EIGEN_TEST_MAX_SIZE / 4, EIGEN_TEST_MAX_SIZE / 3), + internal::random(EIGEN_TEST_MAX_SIZE / 4, EIGEN_TEST_MAX_SIZE / 3))))); // test matrixbase method - CALL_SUBTEST_1(( jacobisvd_method() )); - CALL_SUBTEST_3(( jacobisvd_method() )); + CALL_SUBTEST_1((jacobisvd_method())); + CALL_SUBTEST_3((jacobisvd_method())); // Test problem size constructors - CALL_SUBTEST_7( JacobiSVD(10,10) ); + CALL_SUBTEST_7(JacobiSVD(10, 10)); // Check that preallocation avoids subsequent mallocs - CALL_SUBTEST_9( svd_preallocate() ); + CALL_SUBTEST_9(svd_preallocate()); - CALL_SUBTEST_2( svd_underoverflow() ); + CALL_SUBTEST_2(svd_underoverflow()); msvc_workaround(); } diff --git a/filmulator-gui/core/nlmeans/eigen/test/linearstructure.cpp b/filmulator-gui/core/nlmeans/eigen/test/linearstructure.cpp index b6559b2a..d5a2765e 100644 --- a/filmulator-gui/core/nlmeans/eigen/test/linearstructure.cpp +++ b/filmulator-gui/core/nlmeans/eigen/test/linearstructure.cpp @@ -9,15 +9,18 @@ // with this file, You can obtain one at http://mozilla.org/MPL/2.0/. static bool g_called; -#define EIGEN_SCALAR_BINARY_OP_PLUGIN { g_called |= (!internal::is_same::value); } +#define EIGEN_SCALAR_BINARY_OP_PLUGIN \ + { \ + g_called |= (!internal::is_same::value); \ + } #include "main.h" -template void linearStructure(const MatrixType& m) +template void linearStructure(const MatrixType &m) { using std::abs; /* this test covers the following files: - CwiseUnaryOp.h, CwiseBinaryOp.h, SelfCwiseBinaryOp.h + CwiseUnaryOp.h, CwiseBinaryOp.h, SelfCwiseBinaryOp.h */ typedef typename MatrixType::Scalar Scalar; typedef typename MatrixType::RealScalar RealScalar; @@ -27,122 +30,126 @@ template void linearStructure(const MatrixType& m) // this test relies a lot on Random.h, and there's not much more that we can do // to test it, hence I consider that we will have tested Random.h - MatrixType m1 = MatrixType::Random(rows, cols), - m2 = MatrixType::Random(rows, cols), - m3(rows, cols); + MatrixType m1 = MatrixType::Random(rows, cols), m2 = MatrixType::Random(rows, cols), m3(rows, cols); Scalar s1 = internal::random(); - while (abs(s1)(); - - Index r = internal::random(0, rows-1), - c = internal::random(0, cols-1); - - VERIFY_IS_APPROX(-(-m1), m1); - VERIFY_IS_APPROX(m1+m1, 2*m1); - VERIFY_IS_APPROX(m1+m2-m1, m2); - VERIFY_IS_APPROX(-m2+m1+m2, m1); - VERIFY_IS_APPROX(m1*s1, s1*m1); - VERIFY_IS_APPROX((m1+m2)*s1, s1*m1+s1*m2); - VERIFY_IS_APPROX((-m1+m2)*s1, -s1*m1+s1*m2); - m3 = m2; m3 += m1; - VERIFY_IS_APPROX(m3, m1+m2); - m3 = m2; m3 -= m1; - VERIFY_IS_APPROX(m3, m2-m1); - m3 = m2; m3 *= s1; - VERIFY_IS_APPROX(m3, s1*m2); - if(!NumTraits::IsInteger) - { - m3 = m2; m3 /= s1; - VERIFY_IS_APPROX(m3, m2/s1); + while (abs(s1) < RealScalar(1e-3)) s1 = internal::random(); + + Index r = internal::random(0, rows - 1), c = internal::random(0, cols - 1); + + VERIFY_IS_APPROX(-(-m1), m1); + VERIFY_IS_APPROX(m1 + m1, 2 * m1); + VERIFY_IS_APPROX(m1 + m2 - m1, m2); + VERIFY_IS_APPROX(-m2 + m1 + m2, m1); + VERIFY_IS_APPROX(m1 * s1, s1 * m1); + VERIFY_IS_APPROX((m1 + m2) * s1, s1 * m1 + s1 * m2); + VERIFY_IS_APPROX((-m1 + m2) * s1, -s1 * m1 + s1 * m2); + m3 = m2; + m3 += m1; + VERIFY_IS_APPROX(m3, m1 + m2); + m3 = m2; + m3 -= m1; + VERIFY_IS_APPROX(m3, m2 - m1); + m3 = m2; + m3 *= s1; + VERIFY_IS_APPROX(m3, s1 * m2); + if (!NumTraits::IsInteger) { + m3 = m2; + m3 /= s1; + VERIFY_IS_APPROX(m3, m2 / s1); } // again, test operator() to check const-qualification - VERIFY_IS_APPROX((-m1)(r,c), -(m1(r,c))); - VERIFY_IS_APPROX((m1-m2)(r,c), (m1(r,c))-(m2(r,c))); - VERIFY_IS_APPROX((m1+m2)(r,c), (m1(r,c))+(m2(r,c))); - VERIFY_IS_APPROX((s1*m1)(r,c), s1*(m1(r,c))); - VERIFY_IS_APPROX((m1*s1)(r,c), (m1(r,c))*s1); - if(!NumTraits::IsInteger) - VERIFY_IS_APPROX((m1/s1)(r,c), (m1(r,c))/s1); + VERIFY_IS_APPROX((-m1)(r, c), -(m1(r, c))); + VERIFY_IS_APPROX((m1 - m2)(r, c), (m1(r, c)) - (m2(r, c))); + VERIFY_IS_APPROX((m1 + m2)(r, c), (m1(r, c)) + (m2(r, c))); + VERIFY_IS_APPROX((s1 * m1)(r, c), s1 * (m1(r, c))); + VERIFY_IS_APPROX((m1 * s1)(r, c), (m1(r, c)) * s1); + if (!NumTraits::IsInteger) VERIFY_IS_APPROX((m1 / s1)(r, c), (m1(r, c)) / s1); // use .block to disable vectorization and compare to the vectorized version - VERIFY_IS_APPROX(m1+m1.block(0,0,rows,cols), m1+m1); - VERIFY_IS_APPROX(m1.cwiseProduct(m1.block(0,0,rows,cols)), m1.cwiseProduct(m1)); - VERIFY_IS_APPROX(m1 - m1.block(0,0,rows,cols), m1 - m1); - VERIFY_IS_APPROX(m1.block(0,0,rows,cols) * s1, m1 * s1); + VERIFY_IS_APPROX(m1 + m1.block(0, 0, rows, cols), m1 + m1); + VERIFY_IS_APPROX(m1.cwiseProduct(m1.block(0, 0, rows, cols)), m1.cwiseProduct(m1)); + VERIFY_IS_APPROX(m1 - m1.block(0, 0, rows, cols), m1 - m1); + VERIFY_IS_APPROX(m1.block(0, 0, rows, cols) * s1, m1 * s1); } // Make sure that complex * real and real * complex are properly optimized -template void real_complex(DenseIndex rows = MatrixType::RowsAtCompileTime, DenseIndex cols = MatrixType::ColsAtCompileTime) +template +void real_complex(DenseIndex rows = MatrixType::RowsAtCompileTime, DenseIndex cols = MatrixType::ColsAtCompileTime) { typedef typename MatrixType::Scalar Scalar; typedef typename MatrixType::RealScalar RealScalar; - + RealScalar s = internal::random(); MatrixType m1 = MatrixType::Random(rows, cols); - + g_called = false; - VERIFY_IS_APPROX(s*m1, Scalar(s)*m1); + VERIFY_IS_APPROX(s * m1, Scalar(s) * m1); VERIFY(g_called && "real * matrix not properly optimized"); - + g_called = false; - VERIFY_IS_APPROX(m1*s, m1*Scalar(s)); + VERIFY_IS_APPROX(m1 * s, m1 * Scalar(s)); VERIFY(g_called && "matrix * real not properly optimized"); - + g_called = false; - VERIFY_IS_APPROX(m1/s, m1/Scalar(s)); + VERIFY_IS_APPROX(m1 / s, m1 / Scalar(s)); VERIFY(g_called && "matrix / real not properly optimized"); g_called = false; - VERIFY_IS_APPROX(s+m1.array(), Scalar(s)+m1.array()); + VERIFY_IS_APPROX(s + m1.array(), Scalar(s) + m1.array()); VERIFY(g_called && "real + matrix not properly optimized"); g_called = false; - VERIFY_IS_APPROX(m1.array()+s, m1.array()+Scalar(s)); + VERIFY_IS_APPROX(m1.array() + s, m1.array() + Scalar(s)); VERIFY(g_called && "matrix + real not properly optimized"); g_called = false; - VERIFY_IS_APPROX(s-m1.array(), Scalar(s)-m1.array()); + VERIFY_IS_APPROX(s - m1.array(), Scalar(s) - m1.array()); VERIFY(g_called && "real - matrix not properly optimized"); g_called = false; - VERIFY_IS_APPROX(m1.array()-s, m1.array()-Scalar(s)); + VERIFY_IS_APPROX(m1.array() - s, m1.array() - Scalar(s)); VERIFY(g_called && "matrix - real not properly optimized"); } void test_linearstructure() { g_called = true; - VERIFY(g_called); // avoid `unneeded-internal-declaration` warning. - for(int i = 0; i < g_repeat; i++) { - CALL_SUBTEST_1( linearStructure(Matrix()) ); - CALL_SUBTEST_2( linearStructure(Matrix2f()) ); - CALL_SUBTEST_3( linearStructure(Vector3d()) ); - CALL_SUBTEST_4( linearStructure(Matrix4d()) ); - CALL_SUBTEST_5( linearStructure(MatrixXcf(internal::random(1,EIGEN_TEST_MAX_SIZE/2), internal::random(1,EIGEN_TEST_MAX_SIZE/2))) ); - CALL_SUBTEST_6( linearStructure(MatrixXf (internal::random(1,EIGEN_TEST_MAX_SIZE), internal::random(1,EIGEN_TEST_MAX_SIZE))) ); - CALL_SUBTEST_7( linearStructure(MatrixXi (internal::random(1,EIGEN_TEST_MAX_SIZE), internal::random(1,EIGEN_TEST_MAX_SIZE))) ); - CALL_SUBTEST_8( linearStructure(MatrixXcd(internal::random(1,EIGEN_TEST_MAX_SIZE/2), internal::random(1,EIGEN_TEST_MAX_SIZE/2))) ); - CALL_SUBTEST_9( linearStructure(ArrayXXf (internal::random(1,EIGEN_TEST_MAX_SIZE), internal::random(1,EIGEN_TEST_MAX_SIZE))) ); - CALL_SUBTEST_10( linearStructure(ArrayXXcf (internal::random(1,EIGEN_TEST_MAX_SIZE), internal::random(1,EIGEN_TEST_MAX_SIZE))) ); - - CALL_SUBTEST_11( real_complex() ); - CALL_SUBTEST_11( real_complex(10,10) ); - CALL_SUBTEST_11( real_complex(10,10) ); + VERIFY(g_called);// avoid `unneeded-internal-declaration` warning. + for (int i = 0; i < g_repeat; i++) { + CALL_SUBTEST_1(linearStructure(Matrix())); + CALL_SUBTEST_2(linearStructure(Matrix2f())); + CALL_SUBTEST_3(linearStructure(Vector3d())); + CALL_SUBTEST_4(linearStructure(Matrix4d())); + CALL_SUBTEST_5(linearStructure( + MatrixXcf(internal::random(1, EIGEN_TEST_MAX_SIZE / 2), internal::random(1, EIGEN_TEST_MAX_SIZE / 2)))); + CALL_SUBTEST_6(linearStructure( + MatrixXf(internal::random(1, EIGEN_TEST_MAX_SIZE), internal::random(1, EIGEN_TEST_MAX_SIZE)))); + CALL_SUBTEST_7(linearStructure( + MatrixXi(internal::random(1, EIGEN_TEST_MAX_SIZE), internal::random(1, EIGEN_TEST_MAX_SIZE)))); + CALL_SUBTEST_8(linearStructure( + MatrixXcd(internal::random(1, EIGEN_TEST_MAX_SIZE / 2), internal::random(1, EIGEN_TEST_MAX_SIZE / 2)))); + CALL_SUBTEST_9(linearStructure( + ArrayXXf(internal::random(1, EIGEN_TEST_MAX_SIZE), internal::random(1, EIGEN_TEST_MAX_SIZE)))); + CALL_SUBTEST_10(linearStructure( + ArrayXXcf(internal::random(1, EIGEN_TEST_MAX_SIZE), internal::random(1, EIGEN_TEST_MAX_SIZE)))); + + CALL_SUBTEST_11(real_complex()); + CALL_SUBTEST_11(real_complex(10, 10)); + CALL_SUBTEST_11(real_complex(10, 10)); } - + #ifdef EIGEN_TEST_PART_4 { // make sure that /=scalar and /scalar do not overflow // rational: 1.0/4.94e-320 overflow, but m/4.94e-320 should not Matrix4d m2, m3; - m3 = m2 = Matrix4d::Random()*1e-20; + m3 = m2 = Matrix4d::Random() * 1e-20; m2 = m2 / 4.9e-320; VERIFY_IS_APPROX(m2.cwiseQuotient(m2), Matrix4d::Ones()); m3 /= 4.9e-320; VERIFY_IS_APPROX(m3.cwiseQuotient(m3), Matrix4d::Ones()); - - } #endif } diff --git a/filmulator-gui/core/nlmeans/eigen/test/lscg.cpp b/filmulator-gui/core/nlmeans/eigen/test/lscg.cpp index d49ee00c..b8b12781 100644 --- a/filmulator-gui/core/nlmeans/eigen/test/lscg.cpp +++ b/filmulator-gui/core/nlmeans/eigen/test/lscg.cpp @@ -12,26 +12,26 @@ template void test_lscg_T() { - LeastSquaresConjugateGradient > lscg_colmajor_diag; + LeastSquaresConjugateGradient> lscg_colmajor_diag; LeastSquaresConjugateGradient, IdentityPreconditioner> lscg_colmajor_I; - LeastSquaresConjugateGradient > lscg_rowmajor_diag; - LeastSquaresConjugateGradient, IdentityPreconditioner> lscg_rowmajor_I; + LeastSquaresConjugateGradient> lscg_rowmajor_diag; + LeastSquaresConjugateGradient, IdentityPreconditioner> lscg_rowmajor_I; - CALL_SUBTEST( check_sparse_square_solving(lscg_colmajor_diag) ); - CALL_SUBTEST( check_sparse_square_solving(lscg_colmajor_I) ); - - CALL_SUBTEST( check_sparse_leastsquare_solving(lscg_colmajor_diag) ); - CALL_SUBTEST( check_sparse_leastsquare_solving(lscg_colmajor_I) ); + CALL_SUBTEST(check_sparse_square_solving(lscg_colmajor_diag)); + CALL_SUBTEST(check_sparse_square_solving(lscg_colmajor_I)); - CALL_SUBTEST( check_sparse_square_solving(lscg_rowmajor_diag) ); - CALL_SUBTEST( check_sparse_square_solving(lscg_rowmajor_I) ); + CALL_SUBTEST(check_sparse_leastsquare_solving(lscg_colmajor_diag)); + CALL_SUBTEST(check_sparse_leastsquare_solving(lscg_colmajor_I)); - CALL_SUBTEST( check_sparse_leastsquare_solving(lscg_rowmajor_diag) ); - CALL_SUBTEST( check_sparse_leastsquare_solving(lscg_rowmajor_I) ); + CALL_SUBTEST(check_sparse_square_solving(lscg_rowmajor_diag)); + CALL_SUBTEST(check_sparse_square_solving(lscg_rowmajor_I)); + + CALL_SUBTEST(check_sparse_leastsquare_solving(lscg_rowmajor_diag)); + CALL_SUBTEST(check_sparse_leastsquare_solving(lscg_rowmajor_I)); } void test_lscg() { CALL_SUBTEST_1(test_lscg_T()); - CALL_SUBTEST_2(test_lscg_T >()); + CALL_SUBTEST_2(test_lscg_T>()); } diff --git a/filmulator-gui/core/nlmeans/eigen/test/lu.cpp b/filmulator-gui/core/nlmeans/eigen/test/lu.cpp index 176a2f09..67ef0d45 100644 --- a/filmulator-gui/core/nlmeans/eigen/test/lu.cpp +++ b/filmulator-gui/core/nlmeans/eigen/test/lu.cpp @@ -11,8 +11,8 @@ #include using namespace std; -template -typename MatrixType::RealScalar matrix_l1_norm(const MatrixType& m) { +template typename MatrixType::RealScalar matrix_l1_norm(const MatrixType &m) +{ return m.cwiseAbs().colwise().sum().maxCoeff(); } @@ -23,42 +23,31 @@ template void lu_non_invertible() LU.h */ Index rows, cols, cols2; - if(MatrixType::RowsAtCompileTime==Dynamic) - { - rows = internal::random(2,EIGEN_TEST_MAX_SIZE); - } - else - { + if (MatrixType::RowsAtCompileTime == Dynamic) { + rows = internal::random(2, EIGEN_TEST_MAX_SIZE); + } else { rows = MatrixType::RowsAtCompileTime; } - if(MatrixType::ColsAtCompileTime==Dynamic) - { - cols = internal::random(2,EIGEN_TEST_MAX_SIZE); - cols2 = internal::random(2,EIGEN_TEST_MAX_SIZE); - } - else - { + if (MatrixType::ColsAtCompileTime == Dynamic) { + cols = internal::random(2, EIGEN_TEST_MAX_SIZE); + cols2 = internal::random(2, EIGEN_TEST_MAX_SIZE); + } else { cols2 = cols = MatrixType::ColsAtCompileTime; } - enum { - RowsAtCompileTime = MatrixType::RowsAtCompileTime, - ColsAtCompileTime = MatrixType::ColsAtCompileTime - }; - typedef typename internal::kernel_retval_base >::ReturnType KernelMatrixType; - typedef typename internal::image_retval_base >::ReturnType ImageMatrixType; - typedef Matrix - CMatrixType; - typedef Matrix - RMatrixType; + enum { RowsAtCompileTime = MatrixType::RowsAtCompileTime, ColsAtCompileTime = MatrixType::ColsAtCompileTime }; + typedef typename internal::kernel_retval_base>::ReturnType KernelMatrixType; + typedef typename internal::image_retval_base>::ReturnType ImageMatrixType; + typedef Matrix CMatrixType; + typedef Matrix RMatrixType; - Index rank = internal::random(1, (std::min)(rows, cols)-1); + Index rank = internal::random(1, (std::min)(rows, cols) - 1); // The image of the zero matrix should consist of a single (zero) column vector - VERIFY((MatrixType::Zero(rows,cols).fullPivLu().image(MatrixType::Zero(rows,cols)).cols() == 1)); + VERIFY((MatrixType::Zero(rows, cols).fullPivLu().image(MatrixType::Zero(rows, cols)).cols() == 1)); // The kernel of the zero matrix is the entire space, and thus is an invertible matrix of dimensions cols. - KernelMatrixType kernel = MatrixType::Zero(rows,cols).fullPivLu().kernel(); + KernelMatrixType kernel = MatrixType::Zero(rows, cols).fullPivLu().kernel(); VERIFY((kernel.fullPivLu().isInvertible())); MatrixType m1(rows, cols), m3(rows, cols2); @@ -73,13 +62,13 @@ template void lu_non_invertible() lu.setThreshold(RealScalar(0.01)); lu.compute(m1); - MatrixType u(rows,cols); + MatrixType u(rows, cols); u = lu.matrixLU().template triangularView(); - RMatrixType l = RMatrixType::Identity(rows,rows); - l.block(0,0,rows,(std::min)(rows,cols)).template triangularView() - = lu.matrixLU().block(0,0,rows,(std::min)(rows,cols)); + RMatrixType l = RMatrixType::Identity(rows, rows); + l.block(0, 0, rows, (std::min)(rows, cols)).template triangularView() = + lu.matrixLU().block(0, 0, rows, (std::min)(rows, cols)); - VERIFY_IS_APPROX(lu.permutationP() * m1 * lu.permutationQ(), l*u); + VERIFY_IS_APPROX(lu.permutationP() * m1 * lu.permutationQ(), l * u); KernelMatrixType m1kernel = lu.kernel(); ImageMatrixType m1image = lu.image(m1); @@ -94,32 +83,32 @@ template void lu_non_invertible() VERIFY(m1image.fullPivLu().rank() == rank); VERIFY_IS_APPROX(m1 * m1.adjoint() * m1image, m1image); - m2 = CMatrixType::Random(cols,cols2); - m3 = m1*m2; - m2 = CMatrixType::Random(cols,cols2); + m2 = CMatrixType::Random(cols, cols2); + m3 = m1 * m2; + m2 = CMatrixType::Random(cols, cols2); // test that the code, which does resize(), may be applied to an xpr - m2.block(0,0,m2.rows(),m2.cols()) = lu.solve(m3); - VERIFY_IS_APPROX(m3, m1*m2); + m2.block(0, 0, m2.rows(), m2.cols()) = lu.solve(m3); + VERIFY_IS_APPROX(m3, m1 * m2); // test solve with transposed - m3 = MatrixType::Random(rows,cols2); - m2 = m1.transpose()*m3; - m3 = MatrixType::Random(rows,cols2); + m3 = MatrixType::Random(rows, cols2); + m2 = m1.transpose() * m3; + m3 = MatrixType::Random(rows, cols2); lu.template _solve_impl_transposed(m2, m3); - VERIFY_IS_APPROX(m2, m1.transpose()*m3); - m3 = MatrixType::Random(rows,cols2); + VERIFY_IS_APPROX(m2, m1.transpose() * m3); + m3 = MatrixType::Random(rows, cols2); m3 = lu.transpose().solve(m2); - VERIFY_IS_APPROX(m2, m1.transpose()*m3); + VERIFY_IS_APPROX(m2, m1.transpose() * m3); // test solve with conjugate transposed - m3 = MatrixType::Random(rows,cols2); - m2 = m1.adjoint()*m3; - m3 = MatrixType::Random(rows,cols2); + m3 = MatrixType::Random(rows, cols2); + m2 = m1.adjoint() * m3; + m3 = MatrixType::Random(rows, cols2); lu.template _solve_impl_transposed(m2, m3); - VERIFY_IS_APPROX(m2, m1.adjoint()*m3); - m3 = MatrixType::Random(rows,cols2); + VERIFY_IS_APPROX(m2, m1.adjoint() * m3); + m3 = MatrixType::Random(rows, cols2); m3 = lu.adjoint().solve(m2); - VERIFY_IS_APPROX(m2, m1.adjoint()*m3); + VERIFY_IS_APPROX(m2, m1.adjoint() * m3); } template void lu_invertible() @@ -129,30 +118,29 @@ template void lu_invertible() */ typedef typename NumTraits::Real RealScalar; Index size = MatrixType::RowsAtCompileTime; - if( size==Dynamic) - size = internal::random(1,EIGEN_TEST_MAX_SIZE); + if (size == Dynamic) size = internal::random(1, EIGEN_TEST_MAX_SIZE); MatrixType m1(size, size), m2(size, size), m3(size, size); FullPivLU lu; lu.setThreshold(RealScalar(0.01)); do { - m1 = MatrixType::Random(size,size); + m1 = MatrixType::Random(size, size); lu.compute(m1); - } while(!lu.isInvertible()); + } while (!lu.isInvertible()); VERIFY_IS_APPROX(m1, lu.reconstructedMatrix()); VERIFY(0 == lu.dimensionOfKernel()); - VERIFY(lu.kernel().cols() == 1); // the kernel() should consist of a single (zero) column vector + VERIFY(lu.kernel().cols() == 1);// the kernel() should consist of a single (zero) column vector VERIFY(size == lu.rank()); VERIFY(lu.isInjective()); VERIFY(lu.isSurjective()); VERIFY(lu.isInvertible()); VERIFY(lu.image(m1).fullPivLu().isInvertible()); - m3 = MatrixType::Random(size,size); + m3 = MatrixType::Random(size, size); m2 = lu.solve(m3); - VERIFY_IS_APPROX(m3, m1*m2); + VERIFY_IS_APPROX(m3, m1 * m2); MatrixType m1_inverse = lu.inverse(); - VERIFY_IS_APPROX(m2, m1_inverse*m3); + VERIFY_IS_APPROX(m2, m1_inverse * m3); RealScalar rcond = (RealScalar(1) / matrix_l1_norm(m1)) / matrix_l1_norm(m1_inverse); const RealScalar rcond_est = lu.rcond(); @@ -162,21 +150,21 @@ template void lu_invertible() // test solve with transposed lu.template _solve_impl_transposed(m3, m2); - VERIFY_IS_APPROX(m3, m1.transpose()*m2); - m3 = MatrixType::Random(size,size); + VERIFY_IS_APPROX(m3, m1.transpose() * m2); + m3 = MatrixType::Random(size, size); m3 = lu.transpose().solve(m2); - VERIFY_IS_APPROX(m2, m1.transpose()*m3); + VERIFY_IS_APPROX(m2, m1.transpose() * m3); // test solve with conjugate transposed lu.template _solve_impl_transposed(m3, m2); - VERIFY_IS_APPROX(m3, m1.adjoint()*m2); - m3 = MatrixType::Random(size,size); + VERIFY_IS_APPROX(m3, m1.adjoint() * m2); + m3 = MatrixType::Random(size, size); m3 = lu.adjoint().solve(m2); - VERIFY_IS_APPROX(m2, m1.adjoint()*m3); + VERIFY_IS_APPROX(m2, m1.adjoint() * m3); // Regression test for Bug 302 - MatrixType m4 = MatrixType::Random(size,size); - VERIFY_IS_APPROX(lu.solve(m3*m4), lu.solve(m3)*m4); + MatrixType m4 = MatrixType::Random(size, size); + VERIFY_IS_APPROX(lu.solve(m3 * m4), lu.solve(m3) * m4); } template void lu_partial_piv() @@ -185,7 +173,7 @@ template void lu_partial_piv() PartialPivLU.h */ typedef typename NumTraits::Real RealScalar; - Index size = internal::random(1,4); + Index size = internal::random(1, 4); MatrixType m1(size, size), m2(size, size), m3(size, size); m1.setRandom(); @@ -193,11 +181,11 @@ template void lu_partial_piv() VERIFY_IS_APPROX(m1, plu.reconstructedMatrix()); - m3 = MatrixType::Random(size,size); + m3 = MatrixType::Random(size, size); m2 = plu.solve(m3); - VERIFY_IS_APPROX(m3, m1*m2); + VERIFY_IS_APPROX(m3, m1 * m2); MatrixType m1_inverse = plu.inverse(); - VERIFY_IS_APPROX(m2, m1_inverse*m3); + VERIFY_IS_APPROX(m2, m1_inverse * m3); RealScalar rcond = (RealScalar(1) / matrix_l1_norm(m1)) / matrix_l1_norm(m1_inverse); const RealScalar rcond_est = plu.rcond(); @@ -206,17 +194,17 @@ template void lu_partial_piv() // test solve with transposed plu.template _solve_impl_transposed(m3, m2); - VERIFY_IS_APPROX(m3, m1.transpose()*m2); - m3 = MatrixType::Random(size,size); + VERIFY_IS_APPROX(m3, m1.transpose() * m2); + m3 = MatrixType::Random(size, size); m3 = plu.transpose().solve(m2); - VERIFY_IS_APPROX(m2, m1.transpose()*m3); + VERIFY_IS_APPROX(m2, m1.transpose() * m3); // test solve with conjugate transposed plu.template _solve_impl_transposed(m3, m2); - VERIFY_IS_APPROX(m3, m1.adjoint()*m2); - m3 = MatrixType::Random(size,size); + VERIFY_IS_APPROX(m3, m1.adjoint() * m2); + m3 = MatrixType::Random(size, size); m3 = plu.adjoint().solve(m2); - VERIFY_IS_APPROX(m2, m1.adjoint()*m3); + VERIFY_IS_APPROX(m2, m1.adjoint() * m3); } template void lu_verify_assert() @@ -248,36 +236,36 @@ template void lu_verify_assert() void test_lu() { - for(int i = 0; i < g_repeat; i++) { - CALL_SUBTEST_1( lu_non_invertible() ); - CALL_SUBTEST_1( lu_invertible() ); - CALL_SUBTEST_1( lu_verify_assert() ); + for (int i = 0; i < g_repeat; i++) { + CALL_SUBTEST_1(lu_non_invertible()); + CALL_SUBTEST_1(lu_invertible()); + CALL_SUBTEST_1(lu_verify_assert()); - CALL_SUBTEST_2( (lu_non_invertible >()) ); - CALL_SUBTEST_2( (lu_verify_assert >()) ); + CALL_SUBTEST_2((lu_non_invertible>())); + CALL_SUBTEST_2((lu_verify_assert>())); - CALL_SUBTEST_3( lu_non_invertible() ); - CALL_SUBTEST_3( lu_invertible() ); - CALL_SUBTEST_3( lu_verify_assert() ); + CALL_SUBTEST_3(lu_non_invertible()); + CALL_SUBTEST_3(lu_invertible()); + CALL_SUBTEST_3(lu_verify_assert()); - CALL_SUBTEST_4( lu_non_invertible() ); - CALL_SUBTEST_4( lu_invertible() ); - CALL_SUBTEST_4( lu_partial_piv() ); - CALL_SUBTEST_4( lu_verify_assert() ); + CALL_SUBTEST_4(lu_non_invertible()); + CALL_SUBTEST_4(lu_invertible()); + CALL_SUBTEST_4(lu_partial_piv()); + CALL_SUBTEST_4(lu_verify_assert()); - CALL_SUBTEST_5( lu_non_invertible() ); - CALL_SUBTEST_5( lu_invertible() ); - CALL_SUBTEST_5( lu_verify_assert() ); + CALL_SUBTEST_5(lu_non_invertible()); + CALL_SUBTEST_5(lu_invertible()); + CALL_SUBTEST_5(lu_verify_assert()); - CALL_SUBTEST_6( lu_non_invertible() ); - CALL_SUBTEST_6( lu_invertible() ); - CALL_SUBTEST_6( lu_partial_piv() ); - CALL_SUBTEST_6( lu_verify_assert() ); + CALL_SUBTEST_6(lu_non_invertible()); + CALL_SUBTEST_6(lu_invertible()); + CALL_SUBTEST_6(lu_partial_piv()); + CALL_SUBTEST_6(lu_verify_assert()); - CALL_SUBTEST_7(( lu_non_invertible >() )); + CALL_SUBTEST_7((lu_non_invertible>())); // Test problem size constructors - CALL_SUBTEST_9( PartialPivLU(10) ); - CALL_SUBTEST_9( FullPivLU(10, 20); ); + CALL_SUBTEST_9(PartialPivLU(10)); + CALL_SUBTEST_9(FullPivLU(10, 20);); } } diff --git a/filmulator-gui/core/nlmeans/eigen/test/main.h b/filmulator-gui/core/nlmeans/eigen/test/main.h index 8c868ee7..1725547c 100644 --- a/filmulator-gui/core/nlmeans/eigen/test/main.h +++ b/filmulator-gui/core/nlmeans/eigen/test/main.h @@ -8,15 +8,15 @@ // Public License v. 2.0. If a copy of the MPL was not distributed // with this file, You can obtain one at http://mozilla.org/MPL/2.0/. -#include #include +#include #include -#include #include -#include +#include #include -#include +#include #include +#include // The following includes of STL headers have to be done _before_ the // definition of macros min() and max(). The reason is that many STL @@ -36,13 +36,13 @@ // included before any Eigen header and because the STL headers are guarded // against multiple inclusions, no STL header will see our own min/max macro // definitions. -#include #include +#include #include #include -#include -#include +#include #include +#include #if __cplusplus >= 201103L #include #ifdef EIGEN_USE_THREADS @@ -52,7 +52,7 @@ // Same for cuda_fp16.h #if defined(__CUDACC_VER_MAJOR__) && (__CUDACC_VER_MAJOR__ >= 9) -#define EIGEN_TEST_CUDACC_VER ((__CUDACC_VER_MAJOR__ * 10000) + (__CUDACC_VER_MINOR__ * 100)) +#define EIGEN_TEST_CUDACC_VER ((__CUDACC_VER_MAJOR__ * 10000) + (__CUDACC_VER_MINOR__ * 100)) #elif defined(__CUDACC_VER__) #define EIGEN_TEST_CUDACC_VER __CUDACC_VER__ #else @@ -67,8 +67,8 @@ // protected by parenthesis against macro expansion, the min()/max() macros // are defined here and any not-parenthesized min/max call will cause a // compiler error. -#define min(A,B) please_protect_your_min_with_parentheses -#define max(A,B) please_protect_your_max_with_parentheses +#define min(A, B) please_protect_your_min_with_parentheses +#define max(A, B) please_protect_your_max_with_parentheses #define isnan(X) please_protect_your_isnan_with_parentheses #define isinf(X) please_protect_your_isinf_with_parentheses #define isfinite(X) please_protect_your_isfinite_with_parentheses @@ -77,7 +77,8 @@ #endif #define M_PI please_use_EIGEN_PI_instead_of_M_PI -#define FORBIDDEN_IDENTIFIER (this_identifier_is_forbidden_to_avoid_clashes) this_identifier_is_forbidden_to_avoid_clashes +#define FORBIDDEN_IDENTIFIER \ + (this_identifier_is_forbidden_to_avoid_clashes) this_identifier_is_forbidden_to_avoid_clashes // B0 is defined in POSIX header termios.h #define B0 FORBIDDEN_IDENTIFIER @@ -95,21 +96,26 @@ static long int nb_temporaries; static long int nb_temporaries_on_assert = -1; -inline void on_temporary_creation(long int size) { +inline void on_temporary_creation(long int size) +{ // here's a great place to set a breakpoint when debugging failures in this test! - if(size!=0) nb_temporaries++; - if(nb_temporaries_on_assert>0) assert(nb_temporaries 0) assert(nb_temporaries < nb_temporaries_on_assert); } -#define EIGEN_DENSE_STORAGE_CTOR_PLUGIN { on_temporary_creation(size); } +#define EIGEN_DENSE_STORAGE_CTOR_PLUGIN \ + { \ + on_temporary_creation(size); \ + } -#define VERIFY_EVALUATION_COUNT(XPR,N) {\ - nb_temporaries = 0; \ - XPR; \ - if(nb_temporaries!=N) { std::cerr << "nb_temporaries == " << nb_temporaries << "\n"; }\ - VERIFY( (#XPR) && nb_temporaries==N ); \ +#define VERIFY_EVALUATION_COUNT(XPR, N) \ + { \ + nb_temporaries = 0; \ + XPR; \ + if (nb_temporaries != N) { std::cerr << "nb_temporaries == " << nb_temporaries << "\n"; } \ + VERIFY((#XPR) && nb_temporaries == N); \ } - + #endif // the following file is automatically generated by cmake @@ -135,16 +141,15 @@ inline void on_temporary_creation(long int size) { #define DEFAULT_REPEAT 10 -namespace Eigen -{ - static std::vector g_test_stack; - // level == 0 <=> abort if test fail - // level >= 1 <=> warning message to std::cerr if test fail - static int g_test_level = 0; - static int g_repeat; - static unsigned int g_seed; - static bool g_has_set_repeat, g_has_set_seed; -} +namespace Eigen { +static std::vector g_test_stack; +// level == 0 <=> abort if test fail +// level >= 1 <=> warning message to std::cerr if test fail +static int g_test_level = 0; +static int g_repeat; +static unsigned int g_seed; +static bool g_has_set_repeat, g_has_set_seed; +}// namespace Eigen #define TRACK std::cerr << __FILE__ << " " << __LINE__ << std::endl // #define TRACK while() @@ -155,177 +160,172 @@ namespace Eigen #define EIGEN_DEFAULT_IO_FORMAT IOFormat(4, 0, " ", "\n", "", "", "", "") #if (defined(_CPPUNWIND) || defined(__EXCEPTIONS)) && !defined(__CUDA_ARCH__) - #define EIGEN_EXCEPTIONS +#define EIGEN_EXCEPTIONS #endif #ifndef EIGEN_NO_ASSERTION_CHECKING - namespace Eigen - { - static const bool should_raise_an_assert = false; - - // Used to avoid to raise two exceptions at a time in which - // case the exception is not properly caught. - // This may happen when a second exceptions is triggered in a destructor. - static bool no_more_assert = false; - static bool report_on_cerr_on_assert_failure = true; - - struct eigen_assert_exception - { - eigen_assert_exception(void) {} - ~eigen_assert_exception() { Eigen::no_more_assert = false; } - }; - - struct eigen_static_assert_exception - { - eigen_static_assert_exception(void) {} - ~eigen_static_assert_exception() { Eigen::no_more_assert = false; } - }; +namespace Eigen { +static const bool should_raise_an_assert = false; + +// Used to avoid to raise two exceptions at a time in which +// case the exception is not properly caught. +// This may happen when a second exceptions is triggered in a destructor. +static bool no_more_assert = false; +static bool report_on_cerr_on_assert_failure = true; + +struct eigen_assert_exception +{ + eigen_assert_exception(void) {} + ~eigen_assert_exception() { Eigen::no_more_assert = false; } +}; + +struct eigen_static_assert_exception +{ + eigen_static_assert_exception(void) {} + ~eigen_static_assert_exception() { Eigen::no_more_assert = false; } +}; +}// namespace Eigen +// If EIGEN_DEBUG_ASSERTS is defined and if no assertion is triggered while +// one should have been, then the list of excecuted assertions is printed out. +// +// EIGEN_DEBUG_ASSERTS is not enabled by default as it +// significantly increases the compilation time +// and might even introduce side effects that would hide +// some memory errors. +#ifdef EIGEN_DEBUG_ASSERTS + +namespace Eigen { +namespace internal { + static bool push_assert = false; +} +static std::vector eigen_assert_list; +}// namespace Eigen +#define eigen_assert(a) \ + if ((!(a)) && (!no_more_assert)) { \ + if (report_on_cerr_on_assert_failure) std::cerr << #a << " " __FILE__ << "(" << __LINE__ << ")\n"; \ + Eigen::no_more_assert = true; \ + EIGEN_THROW_X(Eigen::eigen_assert_exception()); \ + } else if (Eigen::internal::push_assert) { \ + eigen_assert_list.push_back(std::string(EI_PP_MAKE_STRING(__FILE__) " (" EI_PP_MAKE_STRING(__LINE__) ") : " #a)); \ } - // If EIGEN_DEBUG_ASSERTS is defined and if no assertion is triggered while - // one should have been, then the list of excecuted assertions is printed out. - // - // EIGEN_DEBUG_ASSERTS is not enabled by default as it - // significantly increases the compilation time - // and might even introduce side effects that would hide - // some memory errors. - #ifdef EIGEN_DEBUG_ASSERTS - - namespace Eigen - { - namespace internal - { - static bool push_assert = false; - } - static std::vector eigen_assert_list; - } - #define eigen_assert(a) \ - if( (!(a)) && (!no_more_assert) ) \ - { \ - if(report_on_cerr_on_assert_failure) \ - std::cerr << #a << " " __FILE__ << "(" << __LINE__ << ")\n"; \ - Eigen::no_more_assert = true; \ - EIGEN_THROW_X(Eigen::eigen_assert_exception()); \ - } \ - else if (Eigen::internal::push_assert) \ - { \ - eigen_assert_list.push_back(std::string(EI_PP_MAKE_STRING(__FILE__) " (" EI_PP_MAKE_STRING(__LINE__) ") : " #a) ); \ - } - #ifdef EIGEN_EXCEPTIONS - #define VERIFY_RAISES_ASSERT(a) \ - { \ - Eigen::no_more_assert = false; \ - Eigen::eigen_assert_list.clear(); \ - Eigen::internal::push_assert = true; \ - Eigen::report_on_cerr_on_assert_failure = false; \ - try { \ - a; \ - std::cerr << "One of the following asserts should have been triggered:\n"; \ - for (uint ai=0 ; ai // required for createRandomPIMatrixOfRank +#include // required for createRandomPIMatrixOfRank -inline void verify_impl(bool condition, const char *testname, const char *file, int line, const char *condition_as_string) +inline void + verify_impl(bool condition, const char *testname, const char *file, int line, const char *condition_as_string) { - if (!condition) - { - if(Eigen::g_test_level>0) - std::cerr << "WARNING: "; - std::cerr << "Test " << testname << " failed in " << file << " (" << line << ")" - << std::endl << " " << condition_as_string << std::endl; + if (!condition) { + if (Eigen::g_test_level > 0) std::cerr << "WARNING: "; + std::cerr << "Test " << testname << " failed in " << file << " (" << line << ")" << std::endl + << " " << condition_as_string << std::endl; std::cerr << "Stack:\n"; const int test_stack_size = static_cast(Eigen::g_test_stack.size()); - for(int i=test_stack_size-1; i>=0; --i) - std::cerr << " - " << Eigen::g_test_stack[i] << "\n"; + for (int i = test_stack_size - 1; i >= 0; --i) std::cerr << " - " << Eigen::g_test_stack[i] << "\n"; std::cerr << "\n"; - if(Eigen::g_test_level==0) - abort(); + if (Eigen::g_test_level == 0) abort(); } } #define VERIFY(a) ::verify_impl(a, g_test_stack.back().c_str(), __FILE__, __LINE__, EI_PP_MAKE_STRING(a)) -#define VERIFY_GE(a, b) ::verify_impl(a >= b, g_test_stack.back().c_str(), __FILE__, __LINE__, EI_PP_MAKE_STRING(a >= b)) -#define VERIFY_LE(a, b) ::verify_impl(a <= b, g_test_stack.back().c_str(), __FILE__, __LINE__, EI_PP_MAKE_STRING(a <= b)) +#define VERIFY_GE(a, b) \ + ::verify_impl(a >= b, g_test_stack.back().c_str(), __FILE__, __LINE__, EI_PP_MAKE_STRING(a >= b)) +#define VERIFY_LE(a, b) \ + ::verify_impl(a <= b, g_test_stack.back().c_str(), __FILE__, __LINE__, EI_PP_MAKE_STRING(a <= b)) #define VERIFY_IS_EQUAL(a, b) VERIFY(test_is_equal(a, b, true)) @@ -339,10 +339,11 @@ inline void verify_impl(bool condition, const char *testname, const char *file, #define VERIFY_IS_UNITARY(a) VERIFY(test_isUnitary(a)) -#define CALL_SUBTEST(FUNC) do { \ +#define CALL_SUBTEST(FUNC) \ + do { \ g_test_stack.push_back(EI_PP_MAKE_STRING(FUNC)); \ - FUNC; \ - g_test_stack.pop_back(); \ + FUNC; \ + g_test_stack.pop_back(); \ } while (0) @@ -352,192 +353,227 @@ template inline typename NumTraits::Real test_precision() { retur template<> inline float test_precision() { return 1e-3f; } template<> inline double test_precision() { return 1e-6; } template<> inline long double test_precision() { return 1e-6l; } -template<> inline float test_precision >() { return test_precision(); } -template<> inline double test_precision >() { return test_precision(); } -template<> inline long double test_precision >() { return test_precision(); } - -inline bool test_isApprox(const short& a, const short& b) -{ return internal::isApprox(a, b, test_precision()); } -inline bool test_isApprox(const unsigned short& a, const unsigned short& b) -{ return internal::isApprox(a, b, test_precision()); } -inline bool test_isApprox(const unsigned int& a, const unsigned int& b) -{ return internal::isApprox(a, b, test_precision()); } -inline bool test_isApprox(const long& a, const long& b) -{ return internal::isApprox(a, b, test_precision()); } -inline bool test_isApprox(const unsigned long& a, const unsigned long& b) -{ return internal::isApprox(a, b, test_precision()); } - -inline bool test_isApprox(const int& a, const int& b) -{ return internal::isApprox(a, b, test_precision()); } -inline bool test_isMuchSmallerThan(const int& a, const int& b) -{ return internal::isMuchSmallerThan(a, b, test_precision()); } -inline bool test_isApproxOrLessThan(const int& a, const int& b) -{ return internal::isApproxOrLessThan(a, b, test_precision()); } - -inline bool test_isApprox(const float& a, const float& b) -{ return internal::isApprox(a, b, test_precision()); } -inline bool test_isMuchSmallerThan(const float& a, const float& b) -{ return internal::isMuchSmallerThan(a, b, test_precision()); } -inline bool test_isApproxOrLessThan(const float& a, const float& b) -{ return internal::isApproxOrLessThan(a, b, test_precision()); } - -inline bool test_isApprox(const double& a, const double& b) -{ return internal::isApprox(a, b, test_precision()); } -inline bool test_isMuchSmallerThan(const double& a, const double& b) -{ return internal::isMuchSmallerThan(a, b, test_precision()); } -inline bool test_isApproxOrLessThan(const double& a, const double& b) -{ return internal::isApproxOrLessThan(a, b, test_precision()); } +template<> inline float test_precision>() { return test_precision(); } +template<> inline double test_precision>() { return test_precision(); } +template<> inline long double test_precision>() { return test_precision(); } + +inline bool test_isApprox(const short &a, const short &b) { return internal::isApprox(a, b, test_precision()); } +inline bool test_isApprox(const unsigned short &a, const unsigned short &b) +{ + return internal::isApprox(a, b, test_precision()); +} +inline bool test_isApprox(const unsigned int &a, const unsigned int &b) +{ + return internal::isApprox(a, b, test_precision()); +} +inline bool test_isApprox(const long &a, const long &b) { return internal::isApprox(a, b, test_precision()); } +inline bool test_isApprox(const unsigned long &a, const unsigned long &b) +{ + return internal::isApprox(a, b, test_precision()); +} + +inline bool test_isApprox(const int &a, const int &b) { return internal::isApprox(a, b, test_precision()); } +inline bool test_isMuchSmallerThan(const int &a, const int &b) +{ + return internal::isMuchSmallerThan(a, b, test_precision()); +} +inline bool test_isApproxOrLessThan(const int &a, const int &b) +{ + return internal::isApproxOrLessThan(a, b, test_precision()); +} + +inline bool test_isApprox(const float &a, const float &b) { return internal::isApprox(a, b, test_precision()); } +inline bool test_isMuchSmallerThan(const float &a, const float &b) +{ + return internal::isMuchSmallerThan(a, b, test_precision()); +} +inline bool test_isApproxOrLessThan(const float &a, const float &b) +{ + return internal::isApproxOrLessThan(a, b, test_precision()); +} + +inline bool test_isApprox(const double &a, const double &b) +{ + return internal::isApprox(a, b, test_precision()); +} +inline bool test_isMuchSmallerThan(const double &a, const double &b) +{ + return internal::isMuchSmallerThan(a, b, test_precision()); +} +inline bool test_isApproxOrLessThan(const double &a, const double &b) +{ + return internal::isApproxOrLessThan(a, b, test_precision()); +} #ifndef EIGEN_TEST_NO_COMPLEX -inline bool test_isApprox(const std::complex& a, const std::complex& b) -{ return internal::isApprox(a, b, test_precision >()); } -inline bool test_isMuchSmallerThan(const std::complex& a, const std::complex& b) -{ return internal::isMuchSmallerThan(a, b, test_precision >()); } +inline bool test_isApprox(const std::complex &a, const std::complex &b) +{ + return internal::isApprox(a, b, test_precision>()); +} +inline bool test_isMuchSmallerThan(const std::complex &a, const std::complex &b) +{ + return internal::isMuchSmallerThan(a, b, test_precision>()); +} -inline bool test_isApprox(const std::complex& a, const std::complex& b) -{ return internal::isApprox(a, b, test_precision >()); } -inline bool test_isMuchSmallerThan(const std::complex& a, const std::complex& b) -{ return internal::isMuchSmallerThan(a, b, test_precision >()); } +inline bool test_isApprox(const std::complex &a, const std::complex &b) +{ + return internal::isApprox(a, b, test_precision>()); +} +inline bool test_isMuchSmallerThan(const std::complex &a, const std::complex &b) +{ + return internal::isMuchSmallerThan(a, b, test_precision>()); +} #ifndef EIGEN_TEST_NO_LONGDOUBLE -inline bool test_isApprox(const std::complex& a, const std::complex& b) -{ return internal::isApprox(a, b, test_precision >()); } -inline bool test_isMuchSmallerThan(const std::complex& a, const std::complex& b) -{ return internal::isMuchSmallerThan(a, b, test_precision >()); } +inline bool test_isApprox(const std::complex &a, const std::complex &b) +{ + return internal::isApprox(a, b, test_precision>()); +} +inline bool test_isMuchSmallerThan(const std::complex &a, const std::complex &b) +{ + return internal::isMuchSmallerThan(a, b, test_precision>()); +} #endif #endif #ifndef EIGEN_TEST_NO_LONGDOUBLE -inline bool test_isApprox(const long double& a, const long double& b) +inline bool test_isApprox(const long double &a, const long double &b) { - bool ret = internal::isApprox(a, b, test_precision()); - if (!ret) std::cerr - << std::endl << " actual = " << a - << std::endl << " expected = " << b << std::endl << std::endl; - return ret; + bool ret = internal::isApprox(a, b, test_precision()); + if (!ret) + std::cerr << std::endl << " actual = " << a << std::endl << " expected = " << b << std::endl << std::endl; + return ret; } -inline bool test_isMuchSmallerThan(const long double& a, const long double& b) -{ return internal::isMuchSmallerThan(a, b, test_precision()); } -inline bool test_isApproxOrLessThan(const long double& a, const long double& b) -{ return internal::isApproxOrLessThan(a, b, test_precision()); } -#endif // EIGEN_TEST_NO_LONGDOUBLE +inline bool test_isMuchSmallerThan(const long double &a, const long double &b) +{ + return internal::isMuchSmallerThan(a, b, test_precision()); +} +inline bool test_isApproxOrLessThan(const long double &a, const long double &b) +{ + return internal::isApproxOrLessThan(a, b, test_precision()); +} +#endif// EIGEN_TEST_NO_LONGDOUBLE -inline bool test_isApprox(const half& a, const half& b) -{ return internal::isApprox(a, b, test_precision()); } -inline bool test_isMuchSmallerThan(const half& a, const half& b) -{ return internal::isMuchSmallerThan(a, b, test_precision()); } -inline bool test_isApproxOrLessThan(const half& a, const half& b) -{ return internal::isApproxOrLessThan(a, b, test_precision()); } +inline bool test_isApprox(const half &a, const half &b) { return internal::isApprox(a, b, test_precision()); } +inline bool test_isMuchSmallerThan(const half &a, const half &b) +{ + return internal::isMuchSmallerThan(a, b, test_precision()); +} +inline bool test_isApproxOrLessThan(const half &a, const half &b) +{ + return internal::isApproxOrLessThan(a, b, test_precision()); +} // test_relative_error returns the relative difference between a and b as a real scalar as used in isApprox. -template -typename NumTraits::NonInteger test_relative_error(const EigenBase &a, const EigenBase &b) +template +typename NumTraits::NonInteger test_relative_error(const EigenBase &a, + const EigenBase &b) { using std::sqrt; typedef typename NumTraits::NonInteger RealScalar; - typename internal::nested_eval::type ea(a.derived()); - typename internal::nested_eval::type eb(b.derived()); - return sqrt(RealScalar((ea-eb).cwiseAbs2().sum()) / RealScalar((std::min)(eb.cwiseAbs2().sum(),ea.cwiseAbs2().sum()))); + typename internal::nested_eval::type ea(a.derived()); + typename internal::nested_eval::type eb(b.derived()); + return sqrt( + RealScalar((ea - eb).cwiseAbs2().sum()) / RealScalar((std::min)(eb.cwiseAbs2().sum(), ea.cwiseAbs2().sum()))); } -template -typename T1::RealScalar test_relative_error(const T1 &a, const T2 &b, const typename T1::Coefficients* = 0) +template +typename T1::RealScalar test_relative_error(const T1 &a, const T2 &b, const typename T1::Coefficients * = 0) { return test_relative_error(a.coeffs(), b.coeffs()); } -template -typename T1::Scalar test_relative_error(const T1 &a, const T2 &b, const typename T1::MatrixType* = 0) +template +typename T1::Scalar test_relative_error(const T1 &a, const T2 &b, const typename T1::MatrixType * = 0) { return test_relative_error(a.matrix(), b.matrix()); } -template -S test_relative_error(const Translation &a, const Translation &b) +template S test_relative_error(const Translation &a, const Translation &b) { return test_relative_error(a.vector(), b.vector()); } -template -S test_relative_error(const ParametrizedLine &a, const ParametrizedLine &b) +template +S test_relative_error(const ParametrizedLine &a, const ParametrizedLine &b) { return (std::max)(test_relative_error(a.origin(), b.origin()), test_relative_error(a.origin(), b.origin())); } -template -S test_relative_error(const AlignedBox &a, const AlignedBox &b) +template S test_relative_error(const AlignedBox &a, const AlignedBox &b) { return (std::max)(test_relative_error((a.min)(), (b.min)()), test_relative_error((a.max)(), (b.max)())); } template class SparseMatrixBase; -template +template typename T1::RealScalar test_relative_error(const MatrixBase &a, const SparseMatrixBase &b) { - return test_relative_error(a,b.toDense()); + return test_relative_error(a, b.toDense()); } template class SparseMatrixBase; -template +template typename T1::RealScalar test_relative_error(const SparseMatrixBase &a, const MatrixBase &b) { - return test_relative_error(a.toDense(),b); + return test_relative_error(a.toDense(), b); } template class SparseMatrixBase; -template +template typename T1::RealScalar test_relative_error(const SparseMatrixBase &a, const SparseMatrixBase &b) { - return test_relative_error(a.toDense(),b.toDense()); + return test_relative_error(a.toDense(), b.toDense()); } -template -typename NumTraits::Real>::NonInteger test_relative_error(const T1 &a, const T2 &b, typename internal::enable_if::Real>::value, T1>::type* = 0) +template +typename NumTraits::Real>::NonInteger test_relative_error(const T1 &a, + const T2 &b, + typename internal::enable_if::Real>::value, T1>::type * = 0) { typedef typename NumTraits::Real>::NonInteger RealScalar; - return numext::sqrt(RealScalar(numext::abs2(a-b))/RealScalar((numext::mini)(numext::abs2(a),numext::abs2(b)))); + return numext::sqrt(RealScalar(numext::abs2(a - b)) / RealScalar((numext::mini)(numext::abs2(a), numext::abs2(b)))); } -template -T test_relative_error(const Rotation2D &a, const Rotation2D &b) +template T test_relative_error(const Rotation2D &a, const Rotation2D &b) { return test_relative_error(a.angle(), b.angle()); } -template -T test_relative_error(const AngleAxis &a, const AngleAxis &b) +template T test_relative_error(const AngleAxis &a, const AngleAxis &b) { return (std::max)(test_relative_error(a.angle(), b.angle()), test_relative_error(a.axis(), b.axis())); } template -inline bool test_isApprox(const Type1& a, const Type2& b, typename Type1::Scalar* = 0) // Enabled for Eigen's type only +inline bool test_isApprox(const Type1 &a, const Type2 &b, typename Type1::Scalar * = 0)// Enabled for Eigen's type only { return a.isApprox(b, test_precision()); } -// get_test_precision is a small wrapper to test_precision allowing to return the scalar precision for either scalars or expressions +// get_test_precision is a small wrapper to test_precision allowing to return the scalar precision for either scalars or +// expressions template -typename NumTraits::Real get_test_precision(const T&, const typename T::Scalar* = 0) +typename NumTraits::Real get_test_precision(const T &, const typename T::Scalar * = 0) { return test_precision::Real>(); } template -typename NumTraits::Real get_test_precision(const T&,typename internal::enable_if::Real>::value, T>::type* = 0) +typename NumTraits::Real get_test_precision(const T &, + typename internal::enable_if::Real>::value, T>::type * = 0) { return test_precision::Real>(); } // verifyIsApprox is a wrapper to test_isApprox that outputs the relative difference magnitude if the test fails. -template -inline bool verifyIsApprox(const Type1& a, const Type2& b) +template inline bool verifyIsApprox(const Type1 &a, const Type2 &b) { - bool ret = test_isApprox(a,b); - if(!ret) - { - std::cerr << "Difference too large wrt tolerance " << get_test_precision(a) << ", relative error is: " << test_relative_error(a,b) << std::endl; + bool ret = test_isApprox(a, b); + if (!ret) { + std::cerr << "Difference too large wrt tolerance " << get_test_precision(a) + << ", relative error is: " << test_relative_error(a, b) << std::endl; } return ret; } @@ -549,58 +585,50 @@ inline bool verifyIsApprox(const Type1& a, const Type2& b) // we won't issue a false negative. // This test could be: abs(a-b) <= eps * ref // However, it seems that simply comparing a+ref and b+ref is more sensitive to true error. -template -inline bool test_isApproxWithRef(const Scalar& a, const Scalar& b, const ScalarRef& ref) +template +inline bool test_isApproxWithRef(const Scalar &a, const Scalar &b, const ScalarRef &ref) { - return test_isApprox(a+ref, b+ref); + return test_isApprox(a + ref, b + ref); } template -inline bool test_isMuchSmallerThan(const MatrixBase& m1, - const MatrixBase& m2) +inline bool test_isMuchSmallerThan(const MatrixBase &m1, const MatrixBase &m2) { return m1.isMuchSmallerThan(m2, test_precision::Scalar>()); } template -inline bool test_isMuchSmallerThan(const MatrixBase& m, - const typename NumTraits::Scalar>::Real& s) +inline bool test_isMuchSmallerThan(const MatrixBase &m, + const typename NumTraits::Scalar>::Real &s) { return m.isMuchSmallerThan(s, test_precision::Scalar>()); } -template -inline bool test_isUnitary(const MatrixBase& m) +template inline bool test_isUnitary(const MatrixBase &m) { return m.isUnitary(test_precision::Scalar>()); } // Forward declaration to avoid ICC warning -template -bool test_is_equal(const T& actual, const U& expected, bool expect_equal=true); +template bool test_is_equal(const T &actual, const U &expected, bool expect_equal = true); -template -bool test_is_equal(const T& actual, const U& expected, bool expect_equal) +template bool test_is_equal(const T &actual, const U &expected, bool expect_equal) { - if ((actual==expected) == expect_equal) - return true; - // false: - std::cerr - << "\n actual = " << actual - << "\n expected " << (expect_equal ? "= " : "!=") << expected << "\n\n"; - return false; + if ((actual == expected) == expect_equal) return true; + // false: + std::cerr << "\n actual = " << actual << "\n expected " << (expect_equal ? "= " : "!=") << expected << "\n\n"; + return false; } /** Creates a random Partial Isometry matrix of given rank. - * - * A partial isometry is a matrix all of whose singular values are either 0 or 1. - * This is very useful to test rank-revealing algorithms. - */ + * + * A partial isometry is a matrix all of whose singular values are either 0 or 1. + * This is very useful to test rank-revealing algorithms. + */ // Forward declaration to avoid ICC warning template -void createRandomPIMatrixOfRank(Index desired_rank, Index rows, Index cols, MatrixType& m); -template -void createRandomPIMatrixOfRank(Index desired_rank, Index rows, Index cols, MatrixType& m) +void createRandomPIMatrixOfRank(Index desired_rank, Index rows, Index cols, MatrixType &m); +template void createRandomPIMatrixOfRank(Index desired_rank, Index rows, Index cols, MatrixType &m) { typedef typename internal::traits::Scalar Scalar; enum { Rows = MatrixType::RowsAtCompileTime, Cols = MatrixType::ColsAtCompileTime }; @@ -609,27 +637,25 @@ void createRandomPIMatrixOfRank(Index desired_rank, Index rows, Index cols, Matr typedef Matrix MatrixAType; typedef Matrix MatrixBType; - if(desired_rank == 0) - { - m.setZero(rows,cols); + if (desired_rank == 0) { + m.setZero(rows, cols); return; } - if(desired_rank == 1) - { + if (desired_rank == 1) { // here we normalize the vectors to get a partial isometry m = VectorType::Random(rows).normalized() * VectorType::Random(cols).normalized().transpose(); return; } - MatrixAType a = MatrixAType::Random(rows,rows); - MatrixType d = MatrixType::Identity(rows,cols); - MatrixBType b = MatrixBType::Random(cols,cols); + MatrixAType a = MatrixAType::Random(rows, rows); + MatrixType d = MatrixType::Identity(rows, cols); + MatrixBType b = MatrixBType::Random(cols, cols); // set the diagonal such that only desired_rank non-zero entries reamain - const Index diag_size = (std::min)(d.rows(),d.cols()); - if(diag_size != desired_rank) - d.diagonal().segment(desired_rank, diag_size-desired_rank) = VectorType::Zero(diag_size-desired_rank); + const Index diag_size = (std::min)(d.rows(), d.cols()); + if (diag_size != desired_rank) + d.diagonal().segment(desired_rank, diag_size - desired_rank) = VectorType::Zero(diag_size - desired_rank); HouseholderQR qra(a); HouseholderQR qrb(b); @@ -637,62 +663,59 @@ void createRandomPIMatrixOfRank(Index desired_rank, Index rows, Index cols, Matr } // Forward declaration to avoid ICC warning -template -void randomPermutationVector(PermutationVectorType& v, Index size); -template -void randomPermutationVector(PermutationVectorType& v, Index size) +template void randomPermutationVector(PermutationVectorType &v, Index size); +template void randomPermutationVector(PermutationVectorType &v, Index size) { typedef typename PermutationVectorType::Scalar Scalar; v.resize(size); - for(Index i = 0; i < size; ++i) v(i) = Scalar(i); - if(size == 1) return; - for(Index n = 0; n < 3 * size; ++n) - { - Index i = internal::random(0, size-1); + for (Index i = 0; i < size; ++i) v(i) = Scalar(i); + if (size == 1) return; + for (Index n = 0; n < 3 * size; ++n) { + Index i = internal::random(0, size - 1); Index j; - do j = internal::random(0, size-1); while(j==i); + do j = internal::random(0, size - 1); + while (j == i); std::swap(v(i), v(j)); } } -template bool isNotNaN(const T& x) -{ - return x==x; -} +template bool isNotNaN(const T &x) { return x == x; } -template bool isPlusInf(const T& x) -{ - return x > NumTraits::highest(); -} +template bool isPlusInf(const T &x) { return x > NumTraits::highest(); } -template bool isMinusInf(const T& x) -{ - return x < NumTraits::lowest(); -} +template bool isMinusInf(const T &x) { return x < NumTraits::lowest(); } -} // end namespace Eigen +}// end namespace Eigen template struct GetDifferentType; -template<> struct GetDifferentType { typedef double type; }; -template<> struct GetDifferentType { typedef float type; }; -template struct GetDifferentType > -{ typedef std::complex::type> type; }; +template<> struct GetDifferentType +{ + typedef double type; +}; +template<> struct GetDifferentType +{ + typedef float type; +}; +template struct GetDifferentType> +{ + typedef std::complex::type> type; +}; // Forward declaration to avoid ICC warning template std::string type_name(); -template std::string type_name() { return "other"; } -template<> std::string type_name() { return "float"; } -template<> std::string type_name() { return "double"; } -template<> std::string type_name() { return "long double"; } -template<> std::string type_name() { return "int"; } -template<> std::string type_name >() { return "complex"; } -template<> std::string type_name >() { return "complex"; } -template<> std::string type_name >() { return "complex"; } -template<> std::string type_name >() { return "complex"; } +template std::string type_name() { return "other"; } +template<> std::string type_name() { return "float"; } +template<> std::string type_name() { return "double"; } +template<> std::string type_name() { return "long double"; } +template<> std::string type_name() { return "int"; } +template<> std::string type_name>() { return "complex"; } +template<> std::string type_name>() { return "complex"; } +template<> std::string type_name>() { return "complex"; } +template<> std::string type_name>() { return "complex"; } // forward declaration of the main test function -void EIGEN_CAT(test_,EIGEN_TEST_FUNC)(); +void EIGEN_CAT(test_, EIGEN_TEST_FUNC)(); using namespace Eigen; @@ -700,8 +723,7 @@ inline void set_repeat_from_string(const char *str) { errno = 0; g_repeat = int(strtoul(str, 0, 10)); - if(errno || g_repeat <= 0) - { + if (errno || g_repeat <= 0) { std::cout << "Invalid repeat value " << str << std::endl; exit(EXIT_FAILURE); } @@ -712,8 +734,7 @@ inline void set_seed_from_string(const char *str) { errno = 0; g_seed = int(strtoul(str, 0, 10)); - if(errno || g_seed == 0) - { + if (errno || g_seed == 0) { std::cout << "Invalid seed value " << str << std::endl; exit(EXIT_FAILURE); } @@ -722,82 +743,72 @@ inline void set_seed_from_string(const char *str) int main(int argc, char *argv[]) { - g_has_set_repeat = false; - g_has_set_seed = false; - bool need_help = false; - - for(int i = 1; i < argc; i++) - { - if(argv[i][0] == 'r') - { - if(g_has_set_repeat) - { - std::cout << "Argument " << argv[i] << " conflicting with a former argument" << std::endl; - return 1; - } - set_repeat_from_string(argv[i]+1); - } - else if(argv[i][0] == 's') - { - if(g_has_set_seed) - { - std::cout << "Argument " << argv[i] << " conflicting with a former argument" << std::endl; - return 1; - } - set_seed_from_string(argv[i]+1); + g_has_set_repeat = false; + g_has_set_seed = false; + bool need_help = false; + + for (int i = 1; i < argc; i++) { + if (argv[i][0] == 'r') { + if (g_has_set_repeat) { + std::cout << "Argument " << argv[i] << " conflicting with a former argument" << std::endl; + return 1; } - else - { - need_help = true; + set_repeat_from_string(argv[i] + 1); + } else if (argv[i][0] == 's') { + if (g_has_set_seed) { + std::cout << "Argument " << argv[i] << " conflicting with a former argument" << std::endl; + return 1; } + set_seed_from_string(argv[i] + 1); + } else { + need_help = true; } + } - if(need_help) - { - std::cout << "This test application takes the following optional arguments:" << std::endl; - std::cout << " rN Repeat each test N times (default: " << DEFAULT_REPEAT << ")" << std::endl; - std::cout << " sN Use N as seed for random numbers (default: based on current time)" << std::endl; - std::cout << std::endl; - std::cout << "If defined, the environment variables EIGEN_REPEAT and EIGEN_SEED" << std::endl; - std::cout << "will be used as default values for these parameters." << std::endl; - return 1; - } + if (need_help) { + std::cout << "This test application takes the following optional arguments:" << std::endl; + std::cout << " rN Repeat each test N times (default: " << DEFAULT_REPEAT << ")" << std::endl; + std::cout << " sN Use N as seed for random numbers (default: based on current time)" << std::endl; + std::cout << std::endl; + std::cout << "If defined, the environment variables EIGEN_REPEAT and EIGEN_SEED" << std::endl; + std::cout << "will be used as default values for these parameters." << std::endl; + return 1; + } - char *env_EIGEN_REPEAT = getenv("EIGEN_REPEAT"); - if(!g_has_set_repeat && env_EIGEN_REPEAT) - set_repeat_from_string(env_EIGEN_REPEAT); - char *env_EIGEN_SEED = getenv("EIGEN_SEED"); - if(!g_has_set_seed && env_EIGEN_SEED) - set_seed_from_string(env_EIGEN_SEED); + char *env_EIGEN_REPEAT = getenv("EIGEN_REPEAT"); + if (!g_has_set_repeat && env_EIGEN_REPEAT) set_repeat_from_string(env_EIGEN_REPEAT); + char *env_EIGEN_SEED = getenv("EIGEN_SEED"); + if (!g_has_set_seed && env_EIGEN_SEED) set_seed_from_string(env_EIGEN_SEED); - if(!g_has_set_seed) g_seed = (unsigned int) time(NULL); - if(!g_has_set_repeat) g_repeat = DEFAULT_REPEAT; + if (!g_has_set_seed) g_seed = (unsigned int)time(NULL); + if (!g_has_set_repeat) g_repeat = DEFAULT_REPEAT; - std::cout << "Initializing random number generator with seed " << g_seed << std::endl; - std::stringstream ss; - ss << "Seed: " << g_seed; - g_test_stack.push_back(ss.str()); - srand(g_seed); - std::cout << "Repeating each test " << g_repeat << " times" << std::endl; + std::cout << "Initializing random number generator with seed " << g_seed << std::endl; + std::stringstream ss; + ss << "Seed: " << g_seed; + g_test_stack.push_back(ss.str()); + srand(g_seed); + std::cout << "Repeating each test " << g_repeat << " times" << std::endl; - Eigen::g_test_stack.push_back(std::string(EI_PP_MAKE_STRING(EIGEN_TEST_FUNC))); + Eigen::g_test_stack.push_back(std::string(EI_PP_MAKE_STRING(EIGEN_TEST_FUNC))); - EIGEN_CAT(test_,EIGEN_TEST_FUNC)(); - return 0; + EIGEN_CAT(test_, EIGEN_TEST_FUNC)(); + return 0; } // These warning are disabled here such that they are still ON when parsing Eigen's header files. #if defined __INTEL_COMPILER - // remark #383: value copied to temporary, reference to temporary used - // -> this warning is raised even for legal usage as: g_test_stack.push_back("foo"); where g_test_stack is a std::vector - // remark #1418: external function definition with no prior declaration - // -> this warning is raised for all our test functions. Declaring them static would fix the issue. - // warning #279: controlling expression is constant - // remark #1572: floating-point equality and inequality comparisons are unreliable - #pragma warning disable 279 383 1418 1572 +// remark #383: value copied to temporary, reference to temporary used +// -> this warning is raised even for legal usage as: g_test_stack.push_back("foo"); where g_test_stack is a +// std::vector +// remark #1418: external function definition with no prior declaration +// -> this warning is raised for all our test functions. Declaring them static would fix the issue. +// warning #279: controlling expression is constant +// remark #1572: floating-point equality and inequality comparisons are unreliable +#pragma warning disable 279 383 1418 1572 #endif #ifdef _MSC_VER - // 4503 - decorated name length exceeded, name was truncated - #pragma warning( disable : 4503) +// 4503 - decorated name length exceeded, name was truncated +#pragma warning(disable : 4503) #endif diff --git a/filmulator-gui/core/nlmeans/eigen/test/mapped_matrix.cpp b/filmulator-gui/core/nlmeans/eigen/test/mapped_matrix.cpp index bc8a694a..477661a7 100644 --- a/filmulator-gui/core/nlmeans/eigen/test/mapped_matrix.cpp +++ b/filmulator-gui/core/nlmeans/eigen/test/mapped_matrix.cpp @@ -8,29 +8,29 @@ // with this file, You can obtain one at http://mozilla.org/MPL/2.0/. #ifndef EIGEN_NO_STATIC_ASSERT -#define EIGEN_NO_STATIC_ASSERT // turn static asserts into runtime asserts in order to check them +#define EIGEN_NO_STATIC_ASSERT// turn static asserts into runtime asserts in order to check them #endif #include "main.h" #define EIGEN_TESTMAP_MAX_SIZE 256 -template void map_class_vector(const VectorType& m) +template void map_class_vector(const VectorType &m) { typedef typename VectorType::Scalar Scalar; Index size = m.size(); - Scalar* array1 = internal::aligned_new(size); - Scalar* array2 = internal::aligned_new(size); - Scalar* array3 = new Scalar[size+1]; - Scalar* array3unaligned = (internal::UIntPtr(array3)%EIGEN_MAX_ALIGN_BYTES) == 0 ? array3+1 : array3; - Scalar array4[EIGEN_TESTMAP_MAX_SIZE]; + Scalar *array1 = internal::aligned_new(size); + Scalar *array2 = internal::aligned_new(size); + Scalar *array3 = new Scalar[size + 1]; + Scalar *array3unaligned = (internal::UIntPtr(array3) % EIGEN_MAX_ALIGN_BYTES) == 0 ? array3 + 1 : array3; + Scalar array4[EIGEN_TESTMAP_MAX_SIZE]; Map(array1, size) = VectorType::Random(size); - Map(array2, size) = Map(array1, size); + Map(array2, size) = Map(array1, size); Map(array3unaligned, size) = Map(array1, size); - Map(array4, size) = Map(array1, size); + Map(array4, size) = Map(array1, size); VectorType ma1 = Map(array1, size); VectorType ma2 = Map(array2, size); VectorType ma3 = Map(array3unaligned, size); @@ -38,46 +38,46 @@ template void map_class_vector(const VectorType& m) VERIFY_IS_EQUAL(ma1, ma2); VERIFY_IS_EQUAL(ma1, ma3); VERIFY_IS_EQUAL(ma1, ma4); - #ifdef EIGEN_VECTORIZE - if(internal::packet_traits::Vectorizable && size>=AlignedMax) - VERIFY_RAISES_ASSERT((Map(array3unaligned, size))) - #endif +#ifdef EIGEN_VECTORIZE + if (internal::packet_traits::Vectorizable && size >= AlignedMax) + VERIFY_RAISES_ASSERT((Map(array3unaligned, size))) +#endif internal::aligned_delete(array1, size); internal::aligned_delete(array2, size); delete[] array3; } -template void map_class_matrix(const MatrixType& m) +template void map_class_matrix(const MatrixType &m) { typedef typename MatrixType::Scalar Scalar; - Index rows = m.rows(), cols = m.cols(), size = rows*cols; + Index rows = m.rows(), cols = m.cols(), size = rows * cols; Scalar s1 = internal::random(); // array1 and array2 -> aligned heap allocation - Scalar* array1 = internal::aligned_new(size); - for(int i = 0; i < size; i++) array1[i] = Scalar(1); - Scalar* array2 = internal::aligned_new(size); - for(int i = 0; i < size; i++) array2[i] = Scalar(1); + Scalar *array1 = internal::aligned_new(size); + for (int i = 0; i < size; i++) array1[i] = Scalar(1); + Scalar *array2 = internal::aligned_new(size); + for (int i = 0; i < size; i++) array2[i] = Scalar(1); // array3unaligned -> unaligned pointer to heap - Scalar* array3 = new Scalar[size+1]; - Index sizep1 = size + 1; // <- without this temporary MSVC 2103 generates bad code - for(Index i = 0; i < sizep1; i++) array3[i] = Scalar(1); - Scalar* array3unaligned = (internal::UIntPtr(array3)%EIGEN_MAX_ALIGN_BYTES) == 0 ? array3+1 : array3; + Scalar *array3 = new Scalar[size + 1]; + Index sizep1 = size + 1;// <- without this temporary MSVC 2103 generates bad code + for (Index i = 0; i < sizep1; i++) array3[i] = Scalar(1); + Scalar *array3unaligned = (internal::UIntPtr(array3) % EIGEN_MAX_ALIGN_BYTES) == 0 ? array3 + 1 : array3; Scalar array4[256]; - if(size<=256) - for(int i = 0; i < size; i++) array4[i] = Scalar(1); - + if (size <= 256) + for (int i = 0; i < size; i++) array4[i] = Scalar(1); + Map map1(array1, rows, cols); Map map2(array2, rows, cols); Map map3(array3unaligned, rows, cols); Map map4(array4, rows, cols); - - VERIFY_IS_EQUAL(map1, MatrixType::Ones(rows,cols)); - VERIFY_IS_EQUAL(map2, MatrixType::Ones(rows,cols)); - VERIFY_IS_EQUAL(map3, MatrixType::Ones(rows,cols)); - map1 = MatrixType::Random(rows,cols); + + VERIFY_IS_EQUAL(map1, MatrixType::Ones(rows, cols)); + VERIFY_IS_EQUAL(map2, MatrixType::Ones(rows, cols)); + VERIFY_IS_EQUAL(map3, MatrixType::Ones(rows, cols)); + map1 = MatrixType::Random(rows, cols); map2 = map1; map3 = map1; MatrixType ma1 = map1; @@ -88,29 +88,28 @@ template void map_class_matrix(const MatrixType& m) VERIFY_IS_EQUAL(ma1, ma2); VERIFY_IS_EQUAL(ma1, ma3); VERIFY_IS_EQUAL(ma1, map3); - - VERIFY_IS_APPROX(s1*map1, s1*map2); - VERIFY_IS_APPROX(s1*ma1, s1*ma2); - VERIFY_IS_EQUAL(s1*ma1, s1*ma3); - VERIFY_IS_APPROX(s1*map1, s1*map3); - + + VERIFY_IS_APPROX(s1 * map1, s1 * map2); + VERIFY_IS_APPROX(s1 * ma1, s1 * ma2); + VERIFY_IS_EQUAL(s1 * ma1, s1 * ma3); + VERIFY_IS_APPROX(s1 * map1, s1 * map3); + map2 *= s1; map3 *= s1; - VERIFY_IS_APPROX(s1*map1, map2); - VERIFY_IS_APPROX(s1*map1, map3); - - if(size<=256) - { - VERIFY_IS_EQUAL(map4, MatrixType::Ones(rows,cols)); + VERIFY_IS_APPROX(s1 * map1, map2); + VERIFY_IS_APPROX(s1 * map1, map3); + + if (size <= 256) { + VERIFY_IS_EQUAL(map4, MatrixType::Ones(rows, cols)); map4 = map1; MatrixType ma4 = map4; VERIFY_IS_EQUAL(map1, map4); VERIFY_IS_EQUAL(ma1, map4); VERIFY_IS_EQUAL(ma1, ma4); - VERIFY_IS_APPROX(s1*map1, s1*map4); - + VERIFY_IS_APPROX(s1 * map1, s1 * map4); + map4 *= s1; - VERIFY_IS_APPROX(s1*map1, map4); + VERIFY_IS_APPROX(s1 * map1, map4); } internal::aligned_delete(array1, size); @@ -118,16 +117,16 @@ template void map_class_matrix(const MatrixType& m) delete[] array3; } -template void map_static_methods(const VectorType& m) +template void map_static_methods(const VectorType &m) { typedef typename VectorType::Scalar Scalar; Index size = m.size(); - Scalar* array1 = internal::aligned_new(size); - Scalar* array2 = internal::aligned_new(size); - Scalar* array3 = new Scalar[size+1]; - Scalar* array3unaligned = internal::UIntPtr(array3)%EIGEN_MAX_ALIGN_BYTES == 0 ? array3+1 : array3; + Scalar *array1 = internal::aligned_new(size); + Scalar *array2 = internal::aligned_new(size); + Scalar *array3 = new Scalar[size + 1]; + Scalar *array3unaligned = internal::UIntPtr(array3) % EIGEN_MAX_ALIGN_BYTES == 0 ? array3 + 1 : array3; VectorType::MapAligned(array1, size) = VectorType::Random(size); VectorType::Map(array2, size) = VectorType::Map(array1, size); @@ -143,7 +142,7 @@ template void map_static_methods(const VectorType& m) delete[] array3; } -template void check_const_correctness(const PlainObjectType&) +template void check_const_correctness(const PlainObjectType &) { // there's a lot that we can't test here while still having this test compile! // the only possible approach would be to run a script trying to compile stuff and checking that it fails. @@ -151,58 +150,57 @@ template void check_const_correctness(const PlainObjec // verify that map-to-const don't have LvalueBit typedef typename internal::add_const::type ConstPlainObjectType; - VERIFY( !(internal::traits >::Flags & LvalueBit) ); - VERIFY( !(internal::traits >::Flags & LvalueBit) ); - VERIFY( !(Map::Flags & LvalueBit) ); - VERIFY( !(Map::Flags & LvalueBit) ); + VERIFY(!(internal::traits>::Flags & LvalueBit)); + VERIFY(!(internal::traits>::Flags & LvalueBit)); + VERIFY(!(Map::Flags & LvalueBit)); + VERIFY(!(Map::Flags & LvalueBit)); } -template -void map_not_aligned_on_scalar() +template void map_not_aligned_on_scalar() { - typedef Matrix MatrixType; + typedef Matrix MatrixType; Index size = 11; - Scalar* array1 = internal::aligned_new((size+1)*(size+1)+1); - Scalar* array2 = reinterpret_cast(sizeof(Scalar)/2+std::size_t(array1)); - Map > map2(array2, size, size, OuterStride<>(size+1)); - MatrixType m2 = MatrixType::Random(size,size); + Scalar *array1 = internal::aligned_new((size + 1) * (size + 1) + 1); + Scalar *array2 = reinterpret_cast(sizeof(Scalar) / 2 + std::size_t(array1)); + Map> map2(array2, size, size, OuterStride<>(size + 1)); + MatrixType m2 = MatrixType::Random(size, size); map2 = m2; VERIFY_IS_EQUAL(m2, map2); - - typedef Matrix VectorType; + + typedef Matrix VectorType; Map map3(array2, size); MatrixType v3 = VectorType::Random(size); map3 = v3; VERIFY_IS_EQUAL(v3, map3); - - internal::aligned_delete(array1, (size+1)*(size+1)+1); + + internal::aligned_delete(array1, (size + 1) * (size + 1) + 1); } void test_mapped_matrix() { - for(int i = 0; i < g_repeat; i++) { - CALL_SUBTEST_1( map_class_vector(Matrix()) ); - CALL_SUBTEST_1( check_const_correctness(Matrix()) ); - CALL_SUBTEST_2( map_class_vector(Vector4d()) ); - CALL_SUBTEST_2( map_class_vector(VectorXd(13)) ); - CALL_SUBTEST_2( check_const_correctness(Matrix4d()) ); - CALL_SUBTEST_3( map_class_vector(RowVector4f()) ); - CALL_SUBTEST_4( map_class_vector(VectorXcf(8)) ); - CALL_SUBTEST_5( map_class_vector(VectorXi(12)) ); - CALL_SUBTEST_5( check_const_correctness(VectorXi(12)) ); - - CALL_SUBTEST_1( map_class_matrix(Matrix()) ); - CALL_SUBTEST_2( map_class_matrix(Matrix4d()) ); - CALL_SUBTEST_11( map_class_matrix(Matrix()) ); - CALL_SUBTEST_4( map_class_matrix(MatrixXcf(internal::random(1,10),internal::random(1,10))) ); - CALL_SUBTEST_5( map_class_matrix(MatrixXi(internal::random(1,10),internal::random(1,10))) ); - - CALL_SUBTEST_6( map_static_methods(Matrix()) ); - CALL_SUBTEST_7( map_static_methods(Vector3f()) ); - CALL_SUBTEST_8( map_static_methods(RowVector3d()) ); - CALL_SUBTEST_9( map_static_methods(VectorXcd(8)) ); - CALL_SUBTEST_10( map_static_methods(VectorXf(12)) ); - - CALL_SUBTEST_11( map_not_aligned_on_scalar() ); + for (int i = 0; i < g_repeat; i++) { + CALL_SUBTEST_1(map_class_vector(Matrix())); + CALL_SUBTEST_1(check_const_correctness(Matrix())); + CALL_SUBTEST_2(map_class_vector(Vector4d())); + CALL_SUBTEST_2(map_class_vector(VectorXd(13))); + CALL_SUBTEST_2(check_const_correctness(Matrix4d())); + CALL_SUBTEST_3(map_class_vector(RowVector4f())); + CALL_SUBTEST_4(map_class_vector(VectorXcf(8))); + CALL_SUBTEST_5(map_class_vector(VectorXi(12))); + CALL_SUBTEST_5(check_const_correctness(VectorXi(12))); + + CALL_SUBTEST_1(map_class_matrix(Matrix())); + CALL_SUBTEST_2(map_class_matrix(Matrix4d())); + CALL_SUBTEST_11(map_class_matrix(Matrix())); + CALL_SUBTEST_4(map_class_matrix(MatrixXcf(internal::random(1, 10), internal::random(1, 10)))); + CALL_SUBTEST_5(map_class_matrix(MatrixXi(internal::random(1, 10), internal::random(1, 10)))); + + CALL_SUBTEST_6(map_static_methods(Matrix())); + CALL_SUBTEST_7(map_static_methods(Vector3f())); + CALL_SUBTEST_8(map_static_methods(RowVector3d())); + CALL_SUBTEST_9(map_static_methods(VectorXcd(8))); + CALL_SUBTEST_10(map_static_methods(VectorXf(12))); + + CALL_SUBTEST_11(map_not_aligned_on_scalar()); } } diff --git a/filmulator-gui/core/nlmeans/eigen/test/mapstaticmethods.cpp b/filmulator-gui/core/nlmeans/eigen/test/mapstaticmethods.cpp index 8156ca93..d0ed83bc 100644 --- a/filmulator-gui/core/nlmeans/eigen/test/mapstaticmethods.cpp +++ b/filmulator-gui/core/nlmeans/eigen/test/mapstaticmethods.cpp @@ -13,19 +13,19 @@ float *ptr; const float *const_ptr; template -struct mapstaticmethods_impl {}; + bool IsDynamicSize = PlainObjectType::SizeAtCompileTime == Dynamic, + bool IsVector = PlainObjectType::IsVectorAtCompileTime> +struct mapstaticmethods_impl +{ +}; -template -struct mapstaticmethods_impl +template struct mapstaticmethods_impl { - static void run(const PlainObjectType& m) + static void run(const PlainObjectType &m) { mapstaticmethods_impl::run(m); - int i = internal::random(2,5), j = internal::random(2,5); + int i = internal::random(2, 5), j = internal::random(2, 5); PlainObjectType::Map(ptr).setZero(); PlainObjectType::MapAligned(ptr).setZero(); @@ -52,26 +52,25 @@ struct mapstaticmethods_impl PlainObjectType::Map(const_ptr, OuterStride<4>()).sum(); PlainObjectType::MapAligned(const_ptr, OuterStride<5>()).sum(); - PlainObjectType::Map(ptr, Stride(i,j)).setZero(); - PlainObjectType::MapAligned(ptr, Stride<2,Dynamic>(2,i)).setZero(); - PlainObjectType::Map(const_ptr, Stride(i,3)).sum(); - PlainObjectType::MapAligned(const_ptr, Stride(i,j)).sum(); + PlainObjectType::Map(ptr, Stride(i, j)).setZero(); + PlainObjectType::MapAligned(ptr, Stride<2, Dynamic>(2, i)).setZero(); + PlainObjectType::Map(const_ptr, Stride(i, 3)).sum(); + PlainObjectType::MapAligned(const_ptr, Stride(i, j)).sum(); - PlainObjectType::Map(ptr, Stride<2,3>()).setZero(); - PlainObjectType::MapAligned(ptr, Stride<3,4>()).setZero(); - PlainObjectType::Map(const_ptr, Stride<2,4>()).sum(); - PlainObjectType::MapAligned(const_ptr, Stride<5,3>()).sum(); + PlainObjectType::Map(ptr, Stride<2, 3>()).setZero(); + PlainObjectType::MapAligned(ptr, Stride<3, 4>()).setZero(); + PlainObjectType::Map(const_ptr, Stride<2, 4>()).sum(); + PlainObjectType::MapAligned(const_ptr, Stride<5, 3>()).sum(); } }; -template -struct mapstaticmethods_impl +template struct mapstaticmethods_impl { - static void run(const PlainObjectType& m) + static void run(const PlainObjectType &m) { Index rows = m.rows(), cols = m.cols(); - int i = internal::random(2,5), j = internal::random(2,5); + int i = internal::random(2, 5), j = internal::random(2, 5); PlainObjectType::Map(ptr, rows, cols).setZero(); PlainObjectType::MapAligned(ptr, rows, cols).setZero(); @@ -98,26 +97,25 @@ struct mapstaticmethods_impl PlainObjectType::Map(const_ptr, rows, cols, OuterStride<4>()).sum(); PlainObjectType::MapAligned(const_ptr, rows, cols, OuterStride<5>()).sum(); - PlainObjectType::Map(ptr, rows, cols, Stride(i,j)).setZero(); - PlainObjectType::MapAligned(ptr, rows, cols, Stride<2,Dynamic>(2,i)).setZero(); - PlainObjectType::Map(const_ptr, rows, cols, Stride(i,3)).sum(); - PlainObjectType::MapAligned(const_ptr, rows, cols, Stride(i,j)).sum(); + PlainObjectType::Map(ptr, rows, cols, Stride(i, j)).setZero(); + PlainObjectType::MapAligned(ptr, rows, cols, Stride<2, Dynamic>(2, i)).setZero(); + PlainObjectType::Map(const_ptr, rows, cols, Stride(i, 3)).sum(); + PlainObjectType::MapAligned(const_ptr, rows, cols, Stride(i, j)).sum(); - PlainObjectType::Map(ptr, rows, cols, Stride<2,3>()).setZero(); - PlainObjectType::MapAligned(ptr, rows, cols, Stride<3,4>()).setZero(); - PlainObjectType::Map(const_ptr, rows, cols, Stride<2,4>()).sum(); - PlainObjectType::MapAligned(const_ptr, rows, cols, Stride<5,3>()).sum(); + PlainObjectType::Map(ptr, rows, cols, Stride<2, 3>()).setZero(); + PlainObjectType::MapAligned(ptr, rows, cols, Stride<3, 4>()).setZero(); + PlainObjectType::Map(const_ptr, rows, cols, Stride<2, 4>()).sum(); + PlainObjectType::MapAligned(const_ptr, rows, cols, Stride<5, 3>()).sum(); } }; -template -struct mapstaticmethods_impl +template struct mapstaticmethods_impl { - static void run(const PlainObjectType& v) + static void run(const PlainObjectType &v) { Index size = v.size(); - int i = internal::random(2,5); + int i = internal::random(2, 5); PlainObjectType::Map(ptr, size).setZero(); PlainObjectType::MapAligned(ptr, size).setZero(); @@ -136,38 +134,36 @@ struct mapstaticmethods_impl } }; -template -void mapstaticmethods(const PlainObjectType& m) +template void mapstaticmethods(const PlainObjectType &m) { mapstaticmethods_impl::run(m); - VERIFY(true); // just to avoid 'unused function' warning + VERIFY(true);// just to avoid 'unused function' warning } void test_mapstaticmethods() { ptr = internal::aligned_new(1000); - for(int i = 0; i < 1000; i++) ptr[i] = float(i); + for (int i = 0; i < 1000; i++) ptr[i] = float(i); const_ptr = ptr; - CALL_SUBTEST_1(( mapstaticmethods(Matrix()) )); - CALL_SUBTEST_1(( mapstaticmethods(Vector2f()) )); - CALL_SUBTEST_2(( mapstaticmethods(Vector3f()) )); - CALL_SUBTEST_2(( mapstaticmethods(Matrix2f()) )); - CALL_SUBTEST_3(( mapstaticmethods(Matrix4f()) )); - CALL_SUBTEST_3(( mapstaticmethods(Array4f()) )); - CALL_SUBTEST_4(( mapstaticmethods(Array3f()) )); - CALL_SUBTEST_4(( mapstaticmethods(Array33f()) )); - CALL_SUBTEST_5(( mapstaticmethods(Array44f()) )); - CALL_SUBTEST_5(( mapstaticmethods(VectorXf(1)) )); - CALL_SUBTEST_5(( mapstaticmethods(VectorXf(8)) )); - CALL_SUBTEST_6(( mapstaticmethods(MatrixXf(1,1)) )); - CALL_SUBTEST_6(( mapstaticmethods(MatrixXf(5,7)) )); - CALL_SUBTEST_7(( mapstaticmethods(ArrayXf(1)) )); - CALL_SUBTEST_7(( mapstaticmethods(ArrayXf(5)) )); - CALL_SUBTEST_8(( mapstaticmethods(ArrayXXf(1,1)) )); - CALL_SUBTEST_8(( mapstaticmethods(ArrayXXf(8,6)) )); + CALL_SUBTEST_1((mapstaticmethods(Matrix()))); + CALL_SUBTEST_1((mapstaticmethods(Vector2f()))); + CALL_SUBTEST_2((mapstaticmethods(Vector3f()))); + CALL_SUBTEST_2((mapstaticmethods(Matrix2f()))); + CALL_SUBTEST_3((mapstaticmethods(Matrix4f()))); + CALL_SUBTEST_3((mapstaticmethods(Array4f()))); + CALL_SUBTEST_4((mapstaticmethods(Array3f()))); + CALL_SUBTEST_4((mapstaticmethods(Array33f()))); + CALL_SUBTEST_5((mapstaticmethods(Array44f()))); + CALL_SUBTEST_5((mapstaticmethods(VectorXf(1)))); + CALL_SUBTEST_5((mapstaticmethods(VectorXf(8)))); + CALL_SUBTEST_6((mapstaticmethods(MatrixXf(1, 1)))); + CALL_SUBTEST_6((mapstaticmethods(MatrixXf(5, 7)))); + CALL_SUBTEST_7((mapstaticmethods(ArrayXf(1)))); + CALL_SUBTEST_7((mapstaticmethods(ArrayXf(5)))); + CALL_SUBTEST_8((mapstaticmethods(ArrayXXf(1, 1)))); + CALL_SUBTEST_8((mapstaticmethods(ArrayXXf(8, 6)))); internal::aligned_delete(ptr, 1000); } - diff --git a/filmulator-gui/core/nlmeans/eigen/test/mapstride.cpp b/filmulator-gui/core/nlmeans/eigen/test/mapstride.cpp index d785148c..71a50c9c 100644 --- a/filmulator-gui/core/nlmeans/eigen/test/mapstride.cpp +++ b/filmulator-gui/core/nlmeans/eigen/test/mapstride.cpp @@ -9,7 +9,7 @@ #include "main.h" -template void map_class_vector(const VectorType& m) +template void map_class_vector(const VectorType &m) { typedef typename VectorType::Scalar Scalar; @@ -17,218 +17,220 @@ template void map_class_vector(const VectorTy VectorType v = VectorType::Random(size); - Index arraysize = 3*size; - - Scalar* a_array = internal::aligned_new(arraysize+1); - Scalar* array = a_array; - if(Alignment!=Aligned) - array = (Scalar*)(internal::IntPtr(a_array) + (internal::packet_traits::AlignedOnScalar?sizeof(Scalar):sizeof(typename NumTraits::Real))); + Index arraysize = 3 * size; + + Scalar *a_array = internal::aligned_new(arraysize + 1); + Scalar *array = a_array; + if (Alignment != Aligned) + array = (Scalar *)(internal::IntPtr(a_array) + + (internal::packet_traits::AlignedOnScalar ? sizeof(Scalar) + : sizeof(typename NumTraits::Real))); { - Map > map(array, size); + Map> map(array, size); map = v; - for(int i = 0; i < size; ++i) - { - VERIFY(array[3*i] == v[i]); + for (int i = 0; i < size; ++i) { + VERIFY(array[3 * i] == v[i]); VERIFY(map[i] == v[i]); } } { - Map > map(array, size, InnerStride(2)); + Map> map(array, size, InnerStride(2)); map = v; - for(int i = 0; i < size; ++i) - { - VERIFY(array[2*i] == v[i]); + for (int i = 0; i < size; ++i) { + VERIFY(array[2 * i] == v[i]); VERIFY(map[i] == v[i]); } } - internal::aligned_delete(a_array, arraysize+1); + internal::aligned_delete(a_array, arraysize + 1); } -template void map_class_matrix(const MatrixType& _m) +template void map_class_matrix(const MatrixType &_m) { typedef typename MatrixType::Scalar Scalar; Index rows = _m.rows(), cols = _m.cols(); - MatrixType m = MatrixType::Random(rows,cols); + MatrixType m = MatrixType::Random(rows, cols); Scalar s1 = internal::random(); - Index arraysize = 4*(rows+4)*(cols+4); + Index arraysize = 4 * (rows + 4) * (cols + 4); - Scalar* a_array1 = internal::aligned_new(arraysize+1); - Scalar* array1 = a_array1; - if(Alignment!=Aligned) - array1 = (Scalar*)(internal::IntPtr(a_array1) + (internal::packet_traits::AlignedOnScalar?sizeof(Scalar):sizeof(typename NumTraits::Real))); + Scalar *a_array1 = internal::aligned_new(arraysize + 1); + Scalar *array1 = a_array1; + if (Alignment != Aligned) + array1 = + (Scalar *)(internal::IntPtr(a_array1) + + (internal::packet_traits::AlignedOnScalar ? sizeof(Scalar) + : sizeof(typename NumTraits::Real))); Scalar a_array2[256]; - Scalar* array2 = a_array2; - if(Alignment!=Aligned) - array2 = (Scalar*)(internal::IntPtr(a_array2) + (internal::packet_traits::AlignedOnScalar?sizeof(Scalar):sizeof(typename NumTraits::Real))); + Scalar *array2 = a_array2; + if (Alignment != Aligned) + array2 = + (Scalar *)(internal::IntPtr(a_array2) + + (internal::packet_traits::AlignedOnScalar ? sizeof(Scalar) + : sizeof(typename NumTraits::Real))); else - array2 = (Scalar*)(((internal::UIntPtr(a_array2)+EIGEN_MAX_ALIGN_BYTES-1)/EIGEN_MAX_ALIGN_BYTES)*EIGEN_MAX_ALIGN_BYTES); + array2 = (Scalar *)(((internal::UIntPtr(a_array2) + EIGEN_MAX_ALIGN_BYTES - 1) / EIGEN_MAX_ALIGN_BYTES) + * EIGEN_MAX_ALIGN_BYTES); Index maxsize2 = a_array2 - array2 + 256; - + // test no inner stride and some dynamic outer stride - for(int k=0; k<2; ++k) - { - if(k==1 && (m.innerSize()+1)*m.outerSize() > maxsize2) - break; - Scalar* array = (k==0 ? array1 : array2); - - Map > map(array, rows, cols, OuterStride(m.innerSize()+1)); + for (int k = 0; k < 2; ++k) { + if (k == 1 && (m.innerSize() + 1) * m.outerSize() > maxsize2) break; + Scalar *array = (k == 0 ? array1 : array2); + + Map> map(array, rows, cols, OuterStride(m.innerSize() + 1)); map = m; - VERIFY(map.outerStride() == map.innerSize()+1); - for(int i = 0; i < m.outerSize(); ++i) - for(int j = 0; j < m.innerSize(); ++j) - { - VERIFY(array[map.outerStride()*i+j] == m.coeffByOuterInner(i,j)); - VERIFY(map.coeffByOuterInner(i,j) == m.coeffByOuterInner(i,j)); + VERIFY(map.outerStride() == map.innerSize() + 1); + for (int i = 0; i < m.outerSize(); ++i) + for (int j = 0; j < m.innerSize(); ++j) { + VERIFY(array[map.outerStride() * i + j] == m.coeffByOuterInner(i, j)); + VERIFY(map.coeffByOuterInner(i, j) == m.coeffByOuterInner(i, j)); } - VERIFY_IS_APPROX(s1*map,s1*m); + VERIFY_IS_APPROX(s1 * map, s1 * m); map *= s1; - VERIFY_IS_APPROX(map,s1*m); + VERIFY_IS_APPROX(map, s1 * m); } // test no inner stride and an outer stride of +4. This is quite important as for fixed-size matrices, // this allows to hit the special case where it's vectorizable. - for(int k=0; k<2; ++k) - { - if(k==1 && (m.innerSize()+4)*m.outerSize() > maxsize2) - break; - Scalar* array = (k==0 ? array1 : array2); - + for (int k = 0; k < 2; ++k) { + if (k == 1 && (m.innerSize() + 4) * m.outerSize() > maxsize2) break; + Scalar *array = (k == 0 ? array1 : array2); + enum { InnerSize = MatrixType::InnerSizeAtCompileTime, - OuterStrideAtCompileTime = InnerSize==Dynamic ? Dynamic : InnerSize+4 + OuterStrideAtCompileTime = InnerSize == Dynamic ? Dynamic : InnerSize + 4 }; - Map > - map(array, rows, cols, OuterStride(m.innerSize()+4)); + Map> map( + array, rows, cols, OuterStride(m.innerSize() + 4)); map = m; - VERIFY(map.outerStride() == map.innerSize()+4); - for(int i = 0; i < m.outerSize(); ++i) - for(int j = 0; j < m.innerSize(); ++j) - { - VERIFY(array[map.outerStride()*i+j] == m.coeffByOuterInner(i,j)); - VERIFY(map.coeffByOuterInner(i,j) == m.coeffByOuterInner(i,j)); + VERIFY(map.outerStride() == map.innerSize() + 4); + for (int i = 0; i < m.outerSize(); ++i) + for (int j = 0; j < m.innerSize(); ++j) { + VERIFY(array[map.outerStride() * i + j] == m.coeffByOuterInner(i, j)); + VERIFY(map.coeffByOuterInner(i, j) == m.coeffByOuterInner(i, j)); } - VERIFY_IS_APPROX(s1*map,s1*m); + VERIFY_IS_APPROX(s1 * map, s1 * m); map *= s1; - VERIFY_IS_APPROX(map,s1*m); + VERIFY_IS_APPROX(map, s1 * m); } // test both inner stride and outer stride - for(int k=0; k<2; ++k) - { - if(k==1 && (2*m.innerSize()+1)*(m.outerSize()*2) > maxsize2) - break; - Scalar* array = (k==0 ? array1 : array2); - - Map > map(array, rows, cols, Stride(2*m.innerSize()+1, 2)); + for (int k = 0; k < 2; ++k) { + if (k == 1 && (2 * m.innerSize() + 1) * (m.outerSize() * 2) > maxsize2) break; + Scalar *array = (k == 0 ? array1 : array2); + + Map> map( + array, rows, cols, Stride(2 * m.innerSize() + 1, 2)); map = m; - VERIFY(map.outerStride() == 2*map.innerSize()+1); + VERIFY(map.outerStride() == 2 * map.innerSize() + 1); VERIFY(map.innerStride() == 2); - for(int i = 0; i < m.outerSize(); ++i) - for(int j = 0; j < m.innerSize(); ++j) - { - VERIFY(array[map.outerStride()*i+map.innerStride()*j] == m.coeffByOuterInner(i,j)); - VERIFY(map.coeffByOuterInner(i,j) == m.coeffByOuterInner(i,j)); + for (int i = 0; i < m.outerSize(); ++i) + for (int j = 0; j < m.innerSize(); ++j) { + VERIFY(array[map.outerStride() * i + map.innerStride() * j] == m.coeffByOuterInner(i, j)); + VERIFY(map.coeffByOuterInner(i, j) == m.coeffByOuterInner(i, j)); } - VERIFY_IS_APPROX(s1*map,s1*m); + VERIFY_IS_APPROX(s1 * map, s1 * m); map *= s1; - VERIFY_IS_APPROX(map,s1*m); + VERIFY_IS_APPROX(map, s1 * m); } // test inner stride and no outer stride - for(int k=0; k<2; ++k) - { - if(k==1 && (m.innerSize()*2)*m.outerSize() > maxsize2) - break; - Scalar* array = (k==0 ? array1 : array2); + for (int k = 0; k < 2; ++k) { + if (k == 1 && (m.innerSize() * 2) * m.outerSize() > maxsize2) break; + Scalar *array = (k == 0 ? array1 : array2); - Map > map(array, rows, cols, InnerStride(2)); + Map> map(array, rows, cols, InnerStride(2)); map = m; - VERIFY(map.outerStride() == map.innerSize()*2); - for(int i = 0; i < m.outerSize(); ++i) - for(int j = 0; j < m.innerSize(); ++j) - { - VERIFY(array[map.innerSize()*i*2+j*2] == m.coeffByOuterInner(i,j)); - VERIFY(map.coeffByOuterInner(i,j) == m.coeffByOuterInner(i,j)); + VERIFY(map.outerStride() == map.innerSize() * 2); + for (int i = 0; i < m.outerSize(); ++i) + for (int j = 0; j < m.innerSize(); ++j) { + VERIFY(array[map.innerSize() * i * 2 + j * 2] == m.coeffByOuterInner(i, j)); + VERIFY(map.coeffByOuterInner(i, j) == m.coeffByOuterInner(i, j)); } - VERIFY_IS_APPROX(s1*map,s1*m); + VERIFY_IS_APPROX(s1 * map, s1 * m); map *= s1; - VERIFY_IS_APPROX(map,s1*m); + VERIFY_IS_APPROX(map, s1 * m); } - internal::aligned_delete(a_array1, arraysize+1); + internal::aligned_delete(a_array1, arraysize + 1); } // Additional tests for inner-stride but no outer-stride -template -void bug1453() +template void bug1453() { - const int data[] = {0,1,2,3,4,5,6,7,8,9,10,11,12,13,14,15, 16, 17, 18, 19, 20, 21, 22, 23, 24, 25, 26, 27, 28, 29, 30, 31}; - typedef Matrix RowMatrixXi; - typedef Matrix ColMatrix23i; - typedef Matrix ColMatrix32i; - typedef Matrix RowMatrix23i; - typedef Matrix RowMatrix32i; - - VERIFY_IS_APPROX(MatrixXi::Map(data, 2, 3, InnerStride<2>()), MatrixXi::Map(data, 2, 3, Stride<4,2>())); - VERIFY_IS_APPROX(MatrixXi::Map(data, 2, 3, InnerStride<>(2)), MatrixXi::Map(data, 2, 3, Stride<4,2>())); - VERIFY_IS_APPROX(MatrixXi::Map(data, 3, 2, InnerStride<2>()), MatrixXi::Map(data, 3, 2, Stride<6,2>())); - VERIFY_IS_APPROX(MatrixXi::Map(data, 3, 2, InnerStride<>(2)), MatrixXi::Map(data, 3, 2, Stride<6,2>())); - - VERIFY_IS_APPROX(RowMatrixXi::Map(data, 2, 3, InnerStride<2>()), RowMatrixXi::Map(data, 2, 3, Stride<6,2>())); - VERIFY_IS_APPROX(RowMatrixXi::Map(data, 2, 3, InnerStride<>(2)), RowMatrixXi::Map(data, 2, 3, Stride<6,2>())); - VERIFY_IS_APPROX(RowMatrixXi::Map(data, 3, 2, InnerStride<2>()), RowMatrixXi::Map(data, 3, 2, Stride<4,2>())); - VERIFY_IS_APPROX(RowMatrixXi::Map(data, 3, 2, InnerStride<>(2)), RowMatrixXi::Map(data, 3, 2, Stride<4,2>())); - - VERIFY_IS_APPROX(ColMatrix23i::Map(data, InnerStride<2>()), MatrixXi::Map(data, 2, 3, Stride<4,2>())); - VERIFY_IS_APPROX(ColMatrix23i::Map(data, InnerStride<>(2)), MatrixXi::Map(data, 2, 3, Stride<4,2>())); - VERIFY_IS_APPROX(ColMatrix32i::Map(data, InnerStride<2>()), MatrixXi::Map(data, 3, 2, Stride<6,2>())); - VERIFY_IS_APPROX(ColMatrix32i::Map(data, InnerStride<>(2)), MatrixXi::Map(data, 3, 2, Stride<6,2>())); - - VERIFY_IS_APPROX(RowMatrix23i::Map(data, InnerStride<2>()), RowMatrixXi::Map(data, 2, 3, Stride<6,2>())); - VERIFY_IS_APPROX(RowMatrix23i::Map(data, InnerStride<>(2)), RowMatrixXi::Map(data, 2, 3, Stride<6,2>())); - VERIFY_IS_APPROX(RowMatrix32i::Map(data, InnerStride<2>()), RowMatrixXi::Map(data, 3, 2, Stride<4,2>())); - VERIFY_IS_APPROX(RowMatrix32i::Map(data, InnerStride<>(2)), RowMatrixXi::Map(data, 3, 2, Stride<4,2>())); + const int data[] = { + 0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15, 16, 17, 18, 19, 20, 21, 22, 23, 24, 25, 26, 27, 28, 29, 30, 31 + }; + typedef Matrix RowMatrixXi; + typedef Matrix ColMatrix23i; + typedef Matrix ColMatrix32i; + typedef Matrix RowMatrix23i; + typedef Matrix RowMatrix32i; + + VERIFY_IS_APPROX(MatrixXi::Map(data, 2, 3, InnerStride<2>()), MatrixXi::Map(data, 2, 3, Stride<4, 2>())); + VERIFY_IS_APPROX(MatrixXi::Map(data, 2, 3, InnerStride<>(2)), MatrixXi::Map(data, 2, 3, Stride<4, 2>())); + VERIFY_IS_APPROX(MatrixXi::Map(data, 3, 2, InnerStride<2>()), MatrixXi::Map(data, 3, 2, Stride<6, 2>())); + VERIFY_IS_APPROX(MatrixXi::Map(data, 3, 2, InnerStride<>(2)), MatrixXi::Map(data, 3, 2, Stride<6, 2>())); + + VERIFY_IS_APPROX(RowMatrixXi::Map(data, 2, 3, InnerStride<2>()), RowMatrixXi::Map(data, 2, 3, Stride<6, 2>())); + VERIFY_IS_APPROX(RowMatrixXi::Map(data, 2, 3, InnerStride<>(2)), RowMatrixXi::Map(data, 2, 3, Stride<6, 2>())); + VERIFY_IS_APPROX(RowMatrixXi::Map(data, 3, 2, InnerStride<2>()), RowMatrixXi::Map(data, 3, 2, Stride<4, 2>())); + VERIFY_IS_APPROX(RowMatrixXi::Map(data, 3, 2, InnerStride<>(2)), RowMatrixXi::Map(data, 3, 2, Stride<4, 2>())); + + VERIFY_IS_APPROX(ColMatrix23i::Map(data, InnerStride<2>()), MatrixXi::Map(data, 2, 3, Stride<4, 2>())); + VERIFY_IS_APPROX(ColMatrix23i::Map(data, InnerStride<>(2)), MatrixXi::Map(data, 2, 3, Stride<4, 2>())); + VERIFY_IS_APPROX(ColMatrix32i::Map(data, InnerStride<2>()), MatrixXi::Map(data, 3, 2, Stride<6, 2>())); + VERIFY_IS_APPROX(ColMatrix32i::Map(data, InnerStride<>(2)), MatrixXi::Map(data, 3, 2, Stride<6, 2>())); + + VERIFY_IS_APPROX(RowMatrix23i::Map(data, InnerStride<2>()), RowMatrixXi::Map(data, 2, 3, Stride<6, 2>())); + VERIFY_IS_APPROX(RowMatrix23i::Map(data, InnerStride<>(2)), RowMatrixXi::Map(data, 2, 3, Stride<6, 2>())); + VERIFY_IS_APPROX(RowMatrix32i::Map(data, InnerStride<2>()), RowMatrixXi::Map(data, 3, 2, Stride<4, 2>())); + VERIFY_IS_APPROX(RowMatrix32i::Map(data, InnerStride<>(2)), RowMatrixXi::Map(data, 3, 2, Stride<4, 2>())); } void test_mapstride() { - for(int i = 0; i < g_repeat; i++) { + for (int i = 0; i < g_repeat; i++) { int maxn = 30; - CALL_SUBTEST_1( map_class_vector(Matrix()) ); - CALL_SUBTEST_1( map_class_vector(Matrix()) ); - CALL_SUBTEST_2( map_class_vector(Vector4d()) ); - CALL_SUBTEST_2( map_class_vector(Vector4d()) ); - CALL_SUBTEST_3( map_class_vector(RowVector4f()) ); - CALL_SUBTEST_3( map_class_vector(RowVector4f()) ); - CALL_SUBTEST_4( map_class_vector(VectorXcf(internal::random(1,maxn))) ); - CALL_SUBTEST_4( map_class_vector(VectorXcf(internal::random(1,maxn))) ); - CALL_SUBTEST_5( map_class_vector(VectorXi(internal::random(1,maxn))) ); - CALL_SUBTEST_5( map_class_vector(VectorXi(internal::random(1,maxn))) ); - - CALL_SUBTEST_1( map_class_matrix(Matrix()) ); - CALL_SUBTEST_1( map_class_matrix(Matrix()) ); - CALL_SUBTEST_2( map_class_matrix(Matrix4d()) ); - CALL_SUBTEST_2( map_class_matrix(Matrix4d()) ); - CALL_SUBTEST_3( map_class_matrix(Matrix()) ); - CALL_SUBTEST_3( map_class_matrix(Matrix()) ); - CALL_SUBTEST_3( map_class_matrix(Matrix()) ); - CALL_SUBTEST_3( map_class_matrix(Matrix()) ); - CALL_SUBTEST_4( map_class_matrix(MatrixXcf(internal::random(1,maxn),internal::random(1,maxn))) ); - CALL_SUBTEST_4( map_class_matrix(MatrixXcf(internal::random(1,maxn),internal::random(1,maxn))) ); - CALL_SUBTEST_5( map_class_matrix(MatrixXi(internal::random(1,maxn),internal::random(1,maxn))) ); - CALL_SUBTEST_5( map_class_matrix(MatrixXi(internal::random(1,maxn),internal::random(1,maxn))) ); - CALL_SUBTEST_6( map_class_matrix(MatrixXcd(internal::random(1,maxn),internal::random(1,maxn))) ); - CALL_SUBTEST_6( map_class_matrix(MatrixXcd(internal::random(1,maxn),internal::random(1,maxn))) ); - - CALL_SUBTEST_5( bug1453<0>() ); - + CALL_SUBTEST_1(map_class_vector(Matrix())); + CALL_SUBTEST_1(map_class_vector(Matrix())); + CALL_SUBTEST_2(map_class_vector(Vector4d())); + CALL_SUBTEST_2(map_class_vector(Vector4d())); + CALL_SUBTEST_3(map_class_vector(RowVector4f())); + CALL_SUBTEST_3(map_class_vector(RowVector4f())); + CALL_SUBTEST_4(map_class_vector(VectorXcf(internal::random(1, maxn)))); + CALL_SUBTEST_4(map_class_vector(VectorXcf(internal::random(1, maxn)))); + CALL_SUBTEST_5(map_class_vector(VectorXi(internal::random(1, maxn)))); + CALL_SUBTEST_5(map_class_vector(VectorXi(internal::random(1, maxn)))); + + CALL_SUBTEST_1(map_class_matrix(Matrix())); + CALL_SUBTEST_1(map_class_matrix(Matrix())); + CALL_SUBTEST_2(map_class_matrix(Matrix4d())); + CALL_SUBTEST_2(map_class_matrix(Matrix4d())); + CALL_SUBTEST_3(map_class_matrix(Matrix())); + CALL_SUBTEST_3(map_class_matrix(Matrix())); + CALL_SUBTEST_3(map_class_matrix(Matrix())); + CALL_SUBTEST_3(map_class_matrix(Matrix())); + CALL_SUBTEST_4( + map_class_matrix(MatrixXcf(internal::random(1, maxn), internal::random(1, maxn)))); + CALL_SUBTEST_4( + map_class_matrix(MatrixXcf(internal::random(1, maxn), internal::random(1, maxn)))); + CALL_SUBTEST_5(map_class_matrix(MatrixXi(internal::random(1, maxn), internal::random(1, maxn)))); + CALL_SUBTEST_5( + map_class_matrix(MatrixXi(internal::random(1, maxn), internal::random(1, maxn)))); + CALL_SUBTEST_6( + map_class_matrix(MatrixXcd(internal::random(1, maxn), internal::random(1, maxn)))); + CALL_SUBTEST_6( + map_class_matrix(MatrixXcd(internal::random(1, maxn), internal::random(1, maxn)))); + + CALL_SUBTEST_5(bug1453<0>()); + TEST_SET_BUT_UNUSED_VARIABLE(maxn); } } diff --git a/filmulator-gui/core/nlmeans/eigen/test/meta.cpp b/filmulator-gui/core/nlmeans/eigen/test/meta.cpp index b8dea68e..0dc49937 100644 --- a/filmulator-gui/core/nlmeans/eigen/test/meta.cpp +++ b/filmulator-gui/core/nlmeans/eigen/test/meta.cpp @@ -9,75 +9,75 @@ #include "main.h" -template -bool check_is_convertible(const From&, const To&) +template bool check_is_convertible(const From &, const To &) { - return internal::is_convertible::value; + return internal::is_convertible::value; } void test_meta() { - VERIFY((internal::conditional<(3<4),internal::true_type, internal::false_type>::type::value)); - VERIFY(( internal::is_same::value)); - VERIFY((!internal::is_same::value)); - VERIFY((!internal::is_same::value)); - VERIFY((!internal::is_same::value)); - - VERIFY(( internal::is_same::type >::value)); - VERIFY(( internal::is_same::type >::value)); - VERIFY(( internal::is_same::type >::value)); - VERIFY(( internal::is_same::type >::value)); - VERIFY(( internal::is_same::type >::value)); - VERIFY(( internal::is_same::type >::value)); - VERIFY(( internal::is_same::type >::value)); + VERIFY((internal::conditional<(3 < 4), internal::true_type, internal::false_type>::type::value)); + VERIFY((internal::is_same::value)); + VERIFY((!internal::is_same::value)); + VERIFY((!internal::is_same::value)); + VERIFY((!internal::is_same::value)); + + VERIFY((internal::is_same::type>::value)); + VERIFY((internal::is_same::type>::value)); + VERIFY((internal::is_same::type>::value)); + VERIFY((internal::is_same::type>::value)); + VERIFY((internal::is_same::type>::value)); + VERIFY((internal::is_same::type>::value)); + VERIFY((internal::is_same::type>::value)); // test add_const - VERIFY(( internal::is_same< internal::add_const::type, const float >::value)); - VERIFY(( internal::is_same< internal::add_const::type, float* const>::value)); - VERIFY(( internal::is_same< internal::add_const::type, float const* const>::value)); - VERIFY(( internal::is_same< internal::add_const::type, float& >::value)); + VERIFY((internal::is_same::type, const float>::value)); + VERIFY((internal::is_same::type, float *const>::value)); + VERIFY((internal::is_same::type, float const *const>::value)); + VERIFY((internal::is_same::type, float &>::value)); // test remove_const - VERIFY(( internal::is_same< internal::remove_const::type, float const* >::value)); - VERIFY(( internal::is_same< internal::remove_const::type, float const* >::value)); - VERIFY(( internal::is_same< internal::remove_const::type, float* >::value)); + VERIFY((internal::is_same::type, float const *>::value)); + VERIFY((internal::is_same::type, float const *>::value)); + VERIFY((internal::is_same::type, float *>::value)); // test add_const_on_value_type - VERIFY(( internal::is_same< internal::add_const_on_value_type::type, float const& >::value)); - VERIFY(( internal::is_same< internal::add_const_on_value_type::type, float const* >::value)); + VERIFY((internal::is_same::type, float const &>::value)); + VERIFY((internal::is_same::type, float const *>::value)); + + VERIFY((internal::is_same::type, const float>::value)); + VERIFY((internal::is_same::type, const float>::value)); - VERIFY(( internal::is_same< internal::add_const_on_value_type::type, const float >::value)); - VERIFY(( internal::is_same< internal::add_const_on_value_type::type, const float >::value)); + VERIFY((internal::is_same::type, const float *const>::value)); + VERIFY((internal::is_same::type, const float *const>::value)); - VERIFY(( internal::is_same< internal::add_const_on_value_type::type, const float* const>::value)); - VERIFY(( internal::is_same< internal::add_const_on_value_type::type, const float* const>::value)); - - VERIFY(( internal::is_same::type >::value)); - VERIFY(( internal::is_same::type >::value)); - VERIFY(( internal::is_same::type >::value)); - VERIFY(( internal::is_same::type >::value)); - VERIFY(( internal::is_same::type >::value)); - - VERIFY(( internal::is_convertible::value )); - VERIFY(( internal::is_convertible::value )); - VERIFY(( internal::is_convertible::value )); - VERIFY((!internal::is_convertible,double>::value )); - VERIFY(( internal::is_convertible::value )); -// VERIFY((!internal::is_convertible::value )); //does not work because the conversion is prevented by a static assertion - VERIFY((!internal::is_convertible::value )); - VERIFY((!internal::is_convertible::value )); + VERIFY((internal::is_same::type>::value)); + VERIFY((internal::is_same::type>::value)); + VERIFY((internal::is_same::type>::value)); + VERIFY((internal::is_same::type>::value)); + VERIFY((internal::is_same::type>::value)); + + VERIFY((internal::is_convertible::value)); + VERIFY((internal::is_convertible::value)); + VERIFY((internal::is_convertible::value)); + VERIFY((!internal::is_convertible, double>::value)); + VERIFY((internal::is_convertible::value)); + // VERIFY((!internal::is_convertible::value )); //does not work because the conversion is + // prevented by a static assertion + VERIFY((!internal::is_convertible::value)); + VERIFY((!internal::is_convertible::value)); { float f; MatrixXf A, B; VectorXf a, b; - VERIFY(( check_is_convertible(a.dot(b), f) )); - VERIFY(( check_is_convertible(a.transpose()*b, f) )); - VERIFY((!check_is_convertible(A*B, f) )); - VERIFY(( check_is_convertible(A*B, A) )); + VERIFY((check_is_convertible(a.dot(b), f))); + VERIFY((check_is_convertible(a.transpose() * b, f))); + VERIFY((!check_is_convertible(A * B, f))); + VERIFY((check_is_convertible(A * B, A))); } - + VERIFY(internal::meta_sqrt<1>::ret == 1); - #define VERIFY_META_SQRT(X) VERIFY(internal::meta_sqrt::ret == int(std::sqrt(double(X)))) +#define VERIFY_META_SQRT(X) VERIFY(internal::meta_sqrt::ret == int(std::sqrt(double(X)))) VERIFY_META_SQRT(2); VERIFY_META_SQRT(3); VERIFY_META_SQRT(4); diff --git a/filmulator-gui/core/nlmeans/eigen/test/metis_support.cpp b/filmulator-gui/core/nlmeans/eigen/test/metis_support.cpp index d87c56a1..7900ad92 100644 --- a/filmulator-gui/core/nlmeans/eigen/test/metis_support.cpp +++ b/filmulator-gui/core/nlmeans/eigen/test/metis_support.cpp @@ -8,18 +8,15 @@ // with this file, You can obtain one at http://mozilla.org/MPL/2.0/. #include "sparse_solver.h" -#include #include +#include #include template void test_metis_T() { - SparseLU, MetisOrdering > sparselu_metis; - - check_sparse_square_solving(sparselu_metis); -} + SparseLU, MetisOrdering> sparselu_metis; -void test_metis_support() -{ - CALL_SUBTEST_1(test_metis_T()); + check_sparse_square_solving(sparselu_metis); } + +void test_metis_support() { CALL_SUBTEST_1(test_metis_T()); } diff --git a/filmulator-gui/core/nlmeans/eigen/test/miscmatrices.cpp b/filmulator-gui/core/nlmeans/eigen/test/miscmatrices.cpp index f17291c4..467615d0 100644 --- a/filmulator-gui/core/nlmeans/eigen/test/miscmatrices.cpp +++ b/filmulator-gui/core/nlmeans/eigen/test/miscmatrices.cpp @@ -9,7 +9,7 @@ #include "main.h" -template void miscMatrices(const MatrixType& m) +template void miscMatrices(const MatrixType &m) { /* this test covers the following files: DiagonalMatrix.h Ones.h @@ -19,16 +19,18 @@ template void miscMatrices(const MatrixType& m) Index rows = m.rows(); Index cols = m.cols(); - Index r = internal::random(0, rows-1), r2 = internal::random(0, rows-1), c = internal::random(0, cols-1); - VERIFY_IS_APPROX(MatrixType::Ones(rows,cols)(r,c), static_cast(1)); - MatrixType m1 = MatrixType::Ones(rows,cols); - VERIFY_IS_APPROX(m1(r,c), static_cast(1)); + Index r = internal::random(0, rows - 1), r2 = internal::random(0, rows - 1), + c = internal::random(0, cols - 1); + VERIFY_IS_APPROX(MatrixType::Ones(rows, cols)(r, c), static_cast(1)); + MatrixType m1 = MatrixType::Ones(rows, cols); + VERIFY_IS_APPROX(m1(r, c), static_cast(1)); VectorType v1 = VectorType::Random(rows); v1[0]; - Matrix - square(v1.asDiagonal()); - if(r==r2) VERIFY_IS_APPROX(square(r,r2), v1[r]); - else VERIFY_IS_MUCH_SMALLER_THAN(square(r,r2), static_cast(1)); + Matrix square(v1.asDiagonal()); + if (r == r2) + VERIFY_IS_APPROX(square(r, r2), v1[r]); + else + VERIFY_IS_MUCH_SMALLER_THAN(square(r, r2), static_cast(1)); square = MatrixType::Zero(rows, rows); square.diagonal() = VectorType::Ones(rows); VERIFY_IS_APPROX(square, MatrixType::Identity(rows, rows)); @@ -36,11 +38,11 @@ template void miscMatrices(const MatrixType& m) void test_miscmatrices() { - for(int i = 0; i < g_repeat; i++) { - CALL_SUBTEST_1( miscMatrices(Matrix()) ); - CALL_SUBTEST_2( miscMatrices(Matrix4d()) ); - CALL_SUBTEST_3( miscMatrices(MatrixXcf(3, 3)) ); - CALL_SUBTEST_4( miscMatrices(MatrixXi(8, 12)) ); - CALL_SUBTEST_5( miscMatrices(MatrixXcd(20, 20)) ); + for (int i = 0; i < g_repeat; i++) { + CALL_SUBTEST_1(miscMatrices(Matrix())); + CALL_SUBTEST_2(miscMatrices(Matrix4d())); + CALL_SUBTEST_3(miscMatrices(MatrixXcf(3, 3))); + CALL_SUBTEST_4(miscMatrices(MatrixXi(8, 12))); + CALL_SUBTEST_5(miscMatrices(MatrixXcd(20, 20))); } } diff --git a/filmulator-gui/core/nlmeans/eigen/test/mixingtypes.cpp b/filmulator-gui/core/nlmeans/eigen/test/mixingtypes.cpp index 45d79aa0..80aa684e 100644 --- a/filmulator-gui/core/nlmeans/eigen/test/mixingtypes.cpp +++ b/filmulator-gui/core/nlmeans/eigen/test/mixingtypes.cpp @@ -11,24 +11,23 @@ #if defined(EIGEN_TEST_PART_7) #ifndef EIGEN_NO_STATIC_ASSERT -#define EIGEN_NO_STATIC_ASSERT // turn static asserts into runtime asserts in order to check them +#define EIGEN_NO_STATIC_ASSERT// turn static asserts into runtime asserts in order to check them #endif // ignore double-promotion diagnostic for clang and gcc, if we check for static assertion anyway: // TODO do the same for MSVC? #if defined(__clang__) -# if (__clang_major__ * 100 + __clang_minor__) >= 308 -# pragma clang diagnostic ignored "-Wdouble-promotion" -# endif +#if (__clang_major__ * 100 + __clang_minor__) >= 308 +#pragma clang diagnostic ignored "-Wdouble-promotion" +#endif #elif defined(__GNUC__) - // TODO is there a minimal GCC version for this? At least g++-4.7 seems to be fine with this. -# pragma GCC diagnostic ignored "-Wdouble-promotion" +// TODO is there a minimal GCC version for this? At least g++-4.7 seems to be fine with this. +#pragma GCC diagnostic ignored "-Wdouble-promotion" #endif #endif - #if defined(EIGEN_TEST_PART_1) || defined(EIGEN_TEST_PART_2) || defined(EIGEN_TEST_PART_3) #ifndef EIGEN_DONT_VECTORIZE @@ -38,34 +37,38 @@ #endif static bool g_called; -#define EIGEN_SCALAR_BINARY_OP_PLUGIN { g_called |= (!internal::is_same::value); } +#define EIGEN_SCALAR_BINARY_OP_PLUGIN \ + { \ + g_called |= (!internal::is_same::value); \ + } #include "main.h" using namespace std; -#define VERIFY_MIX_SCALAR(XPR,REF) \ - g_called = false; \ - VERIFY_IS_APPROX(XPR,REF); \ - VERIFY( g_called && #XPR" not properly optimized"); +#define VERIFY_MIX_SCALAR(XPR, REF) \ + g_called = false; \ + VERIFY_IS_APPROX(XPR, REF); \ + VERIFY(g_called &&#XPR " not properly optimized"); -template -void raise_assertion(Index size = SizeAtCompileType) +template void raise_assertion(Index size = SizeAtCompileType) { // VERIFY_RAISES_ASSERT(mf+md); // does not even compile - Matrix vf; vf.setRandom(size); - Matrix vd; vd.setRandom(size); - VERIFY_RAISES_ASSERT(vf=vd); - VERIFY_RAISES_ASSERT(vf+=vd); - VERIFY_RAISES_ASSERT(vf-=vd); - VERIFY_RAISES_ASSERT(vd=vf); - VERIFY_RAISES_ASSERT(vd+=vf); - VERIFY_RAISES_ASSERT(vd-=vf); + Matrix vf; + vf.setRandom(size); + Matrix vd; + vd.setRandom(size); + VERIFY_RAISES_ASSERT(vf = vd); + VERIFY_RAISES_ASSERT(vf += vd); + VERIFY_RAISES_ASSERT(vf -= vd); + VERIFY_RAISES_ASSERT(vd = vf); + VERIFY_RAISES_ASSERT(vd += vf); + VERIFY_RAISES_ASSERT(vd -= vf); // vd.asDiagonal() * mf; // does not even compile // vcd.asDiagonal() * mf; // does not even compile -#if 0 // we get other compilation errors here than just static asserts +#if 0// we get other compilation errors here than just static asserts VERIFY_RAISES_ASSERT(vd.dot(vf)); #endif } @@ -73,8 +76,8 @@ void raise_assertion(Index size = SizeAtCompileType) template void mixingtypes(int size = SizeAtCompileType) { - typedef std::complex CF; - typedef std::complex CD; + typedef std::complex CF; + typedef std::complex CD; typedef Matrix Mat_f; typedef Matrix Mat_d; typedef Matrix, SizeAtCompileType, SizeAtCompileType> Mat_cf; @@ -84,242 +87,252 @@ template void mixingtypes(int size = SizeAtCompileType) typedef Matrix, SizeAtCompileType, 1> Vec_cf; typedef Matrix, SizeAtCompileType, 1> Vec_cd; - Mat_f mf = Mat_f::Random(size,size); - Mat_d md = mf.template cast(); - //Mat_d rd = md; - Mat_cf mcf = Mat_cf::Random(size,size); - Mat_cd mcd = mcf.template cast >(); + Mat_f mf = Mat_f::Random(size, size); + Mat_d md = mf.template cast(); + // Mat_d rd = md; + Mat_cf mcf = Mat_cf::Random(size, size); + Mat_cd mcd = mcf.template cast>(); Mat_cd rcd = mcd; - Vec_f vf = Vec_f::Random(size,1); - Vec_d vd = vf.template cast(); - Vec_cf vcf = Vec_cf::Random(size,1); - Vec_cd vcd = vcf.template cast >(); - float sf = internal::random(); - double sd = internal::random(); - complex scf = internal::random >(); - complex scd = internal::random >(); + Vec_f vf = Vec_f::Random(size, 1); + Vec_d vd = vf.template cast(); + Vec_cf vcf = Vec_cf::Random(size, 1); + Vec_cd vcd = vcf.template cast>(); + float sf = internal::random(); + double sd = internal::random(); + complex scf = internal::random>(); + complex scd = internal::random>(); - mf+mf; + mf + mf; - float epsf = std::sqrt(std::numeric_limits ::min EIGEN_EMPTY ()); - double epsd = std::sqrt(std::numeric_limits::min EIGEN_EMPTY ()); + float epsf = std::sqrt(std::numeric_limits::min EIGEN_EMPTY()); + double epsd = std::sqrt(std::numeric_limits::min EIGEN_EMPTY()); - while(std::abs(sf )(); - while(std::abs(sd )(); - while(std::abs(scf)(); - while(std::abs(scd)(); + while (std::abs(sf) < epsf) sf = internal::random(); + while (std::abs(sd) < epsd) sd = internal::random(); + while (std::abs(scf) < epsf) scf = internal::random(); + while (std::abs(scd) < epsd) scd = internal::random(); // check scalar products - VERIFY_MIX_SCALAR(vcf * sf , vcf * complex(sf)); - VERIFY_MIX_SCALAR(sd * vcd , complex(sd) * vcd); - VERIFY_MIX_SCALAR(vf * scf , vf.template cast >() * scf); - VERIFY_MIX_SCALAR(scd * vd , scd * vd.template cast >()); + VERIFY_MIX_SCALAR(vcf * sf, vcf * complex(sf)); + VERIFY_MIX_SCALAR(sd * vcd, complex(sd) * vcd); + VERIFY_MIX_SCALAR(vf * scf, vf.template cast>() * scf); + VERIFY_MIX_SCALAR(scd * vd, scd * vd.template cast>()); - VERIFY_MIX_SCALAR(vcf * 2 , vcf * complex(2)); - VERIFY_MIX_SCALAR(vcf * 2.1 , vcf * complex(2.1)); + VERIFY_MIX_SCALAR(vcf * 2, vcf * complex(2)); + VERIFY_MIX_SCALAR(vcf * 2.1, vcf * complex(2.1)); VERIFY_MIX_SCALAR(2 * vcf, vcf * complex(2)); - VERIFY_MIX_SCALAR(2.1 * vcf , vcf * complex(2.1)); + VERIFY_MIX_SCALAR(2.1 * vcf, vcf * complex(2.1)); // check scalar quotients - VERIFY_MIX_SCALAR(vcf / sf , vcf / complex(sf)); - VERIFY_MIX_SCALAR(vf / scf , vf.template cast >() / scf); - VERIFY_MIX_SCALAR(vf.array() / scf, vf.template cast >().array() / scf); - VERIFY_MIX_SCALAR(scd / vd.array() , scd / vd.template cast >().array()); + VERIFY_MIX_SCALAR(vcf / sf, vcf / complex(sf)); + VERIFY_MIX_SCALAR(vf / scf, vf.template cast>() / scf); + VERIFY_MIX_SCALAR(vf.array() / scf, vf.template cast>().array() / scf); + VERIFY_MIX_SCALAR(scd / vd.array(), scd / vd.template cast>().array()); // check scalar increment - VERIFY_MIX_SCALAR(vcf.array() + sf , vcf.array() + complex(sf)); - VERIFY_MIX_SCALAR(sd + vcd.array(), complex(sd) + vcd.array()); - VERIFY_MIX_SCALAR(vf.array() + scf, vf.template cast >().array() + scf); - VERIFY_MIX_SCALAR(scd + vd.array() , scd + vd.template cast >().array()); + VERIFY_MIX_SCALAR(vcf.array() + sf, vcf.array() + complex(sf)); + VERIFY_MIX_SCALAR(sd + vcd.array(), complex(sd) + vcd.array()); + VERIFY_MIX_SCALAR(vf.array() + scf, vf.template cast>().array() + scf); + VERIFY_MIX_SCALAR(scd + vd.array(), scd + vd.template cast>().array()); // check scalar subtractions - VERIFY_MIX_SCALAR(vcf.array() - sf , vcf.array() - complex(sf)); - VERIFY_MIX_SCALAR(sd - vcd.array(), complex(sd) - vcd.array()); - VERIFY_MIX_SCALAR(vf.array() - scf, vf.template cast >().array() - scf); - VERIFY_MIX_SCALAR(scd - vd.array() , scd - vd.template cast >().array()); + VERIFY_MIX_SCALAR(vcf.array() - sf, vcf.array() - complex(sf)); + VERIFY_MIX_SCALAR(sd - vcd.array(), complex(sd) - vcd.array()); + VERIFY_MIX_SCALAR(vf.array() - scf, vf.template cast>().array() - scf); + VERIFY_MIX_SCALAR(scd - vd.array(), scd - vd.template cast>().array()); // check scalar powers - VERIFY_MIX_SCALAR( pow(vcf.array(), sf), Eigen::pow(vcf.array(), complex(sf)) ); - VERIFY_MIX_SCALAR( vcf.array().pow(sf) , Eigen::pow(vcf.array(), complex(sf)) ); - VERIFY_MIX_SCALAR( pow(sd, vcd.array()), Eigen::pow(complex(sd), vcd.array()) ); - VERIFY_MIX_SCALAR( Eigen::pow(vf.array(), scf), Eigen::pow(vf.template cast >().array(), scf) ); - VERIFY_MIX_SCALAR( vf.array().pow(scf) , Eigen::pow(vf.template cast >().array(), scf) ); - VERIFY_MIX_SCALAR( Eigen::pow(scd, vd.array()), Eigen::pow(scd, vd.template cast >().array()) ); + VERIFY_MIX_SCALAR(pow(vcf.array(), sf), Eigen::pow(vcf.array(), complex(sf))); + VERIFY_MIX_SCALAR(vcf.array().pow(sf), Eigen::pow(vcf.array(), complex(sf))); + VERIFY_MIX_SCALAR(pow(sd, vcd.array()), Eigen::pow(complex(sd), vcd.array())); + VERIFY_MIX_SCALAR(Eigen::pow(vf.array(), scf), Eigen::pow(vf.template cast>().array(), scf)); + VERIFY_MIX_SCALAR(vf.array().pow(scf), Eigen::pow(vf.template cast>().array(), scf)); + VERIFY_MIX_SCALAR(Eigen::pow(scd, vd.array()), Eigen::pow(scd, vd.template cast>().array())); // check dot product vf.dot(vf); - VERIFY_IS_APPROX(vcf.dot(vf), vcf.dot(vf.template cast >())); + VERIFY_IS_APPROX(vcf.dot(vf), vcf.dot(vf.template cast>())); // check diagonal product - VERIFY_IS_APPROX(vf.asDiagonal() * mcf, vf.template cast >().asDiagonal() * mcf); - VERIFY_IS_APPROX(vcd.asDiagonal() * md, vcd.asDiagonal() * md.template cast >()); - VERIFY_IS_APPROX(mcf * vf.asDiagonal(), mcf * vf.template cast >().asDiagonal()); - VERIFY_IS_APPROX(md * vcd.asDiagonal(), md.template cast >() * vcd.asDiagonal()); + VERIFY_IS_APPROX(vf.asDiagonal() * mcf, vf.template cast>().asDiagonal() * mcf); + VERIFY_IS_APPROX(vcd.asDiagonal() * md, vcd.asDiagonal() * md.template cast>()); + VERIFY_IS_APPROX(mcf * vf.asDiagonal(), mcf * vf.template cast>().asDiagonal()); + VERIFY_IS_APPROX(md * vcd.asDiagonal(), md.template cast>() * vcd.asDiagonal()); // check inner product - VERIFY_IS_APPROX((vf.transpose() * vcf).value(), (vf.template cast >().transpose() * vcf).value()); + VERIFY_IS_APPROX((vf.transpose() * vcf).value(), (vf.template cast>().transpose() * vcf).value()); // check outer product - VERIFY_IS_APPROX((vf * vcf.transpose()).eval(), (vf.template cast >() * vcf.transpose()).eval()); + VERIFY_IS_APPROX((vf * vcf.transpose()).eval(), (vf.template cast>() * vcf.transpose()).eval()); // coeff wise product - VERIFY_IS_APPROX((vf * vcf.transpose()).eval(), (vf.template cast >() * vcf.transpose()).eval()); + VERIFY_IS_APPROX((vf * vcf.transpose()).eval(), (vf.template cast>() * vcf.transpose()).eval()); Mat_cd mcd2 = mcd; - VERIFY_IS_APPROX(mcd.array() *= md.array(), mcd2.array() *= md.array().template cast >()); - + VERIFY_IS_APPROX(mcd.array() *= md.array(), mcd2.array() *= md.array().template cast>()); + // check matrix-matrix products - VERIFY_IS_APPROX(sd*md*mcd, (sd*md).template cast().eval()*mcd); - VERIFY_IS_APPROX(sd*mcd*md, sd*mcd*md.template cast()); - VERIFY_IS_APPROX(scd*md*mcd, scd*md.template cast().eval()*mcd); - VERIFY_IS_APPROX(scd*mcd*md, scd*mcd*md.template cast()); - - VERIFY_IS_APPROX(sf*mf*mcf, sf*mf.template cast()*mcf); - VERIFY_IS_APPROX(sf*mcf*mf, sf*mcf*mf.template cast()); - VERIFY_IS_APPROX(scf*mf*mcf, scf*mf.template cast()*mcf); - VERIFY_IS_APPROX(scf*mcf*mf, scf*mcf*mf.template cast()); - - VERIFY_IS_APPROX(sd*md.adjoint()*mcd, (sd*md).template cast().eval().adjoint()*mcd); - VERIFY_IS_APPROX(sd*mcd.adjoint()*md, sd*mcd.adjoint()*md.template cast()); - VERIFY_IS_APPROX(sd*md.adjoint()*mcd.adjoint(), (sd*md).template cast().eval().adjoint()*mcd.adjoint()); - VERIFY_IS_APPROX(sd*mcd.adjoint()*md.adjoint(), sd*mcd.adjoint()*md.template cast().adjoint()); - VERIFY_IS_APPROX(sd*md*mcd.adjoint(), (sd*md).template cast().eval()*mcd.adjoint()); - VERIFY_IS_APPROX(sd*mcd*md.adjoint(), sd*mcd*md.template cast().adjoint()); - - VERIFY_IS_APPROX(sf*mf.adjoint()*mcf, (sf*mf).template cast().eval().adjoint()*mcf); - VERIFY_IS_APPROX(sf*mcf.adjoint()*mf, sf*mcf.adjoint()*mf.template cast()); - VERIFY_IS_APPROX(sf*mf.adjoint()*mcf.adjoint(), (sf*mf).template cast().eval().adjoint()*mcf.adjoint()); - VERIFY_IS_APPROX(sf*mcf.adjoint()*mf.adjoint(), sf*mcf.adjoint()*mf.template cast().adjoint()); - VERIFY_IS_APPROX(sf*mf*mcf.adjoint(), (sf*mf).template cast().eval()*mcf.adjoint()); - VERIFY_IS_APPROX(sf*mcf*mf.adjoint(), sf*mcf*mf.template cast().adjoint()); - - VERIFY_IS_APPROX(sf*mf*vcf, (sf*mf).template cast().eval()*vcf); - VERIFY_IS_APPROX(scf*mf*vcf,(scf*mf.template cast()).eval()*vcf); - VERIFY_IS_APPROX(sf*mcf*vf, sf*mcf*vf.template cast()); - VERIFY_IS_APPROX(scf*mcf*vf,scf*mcf*vf.template cast()); - - VERIFY_IS_APPROX(sf*vcf.adjoint()*mf, sf*vcf.adjoint()*mf.template cast().eval()); - VERIFY_IS_APPROX(scf*vcf.adjoint()*mf, scf*vcf.adjoint()*mf.template cast().eval()); - VERIFY_IS_APPROX(sf*vf.adjoint()*mcf, sf*vf.adjoint().template cast().eval()*mcf); - VERIFY_IS_APPROX(scf*vf.adjoint()*mcf, scf*vf.adjoint().template cast().eval()*mcf); - - VERIFY_IS_APPROX(sd*md*vcd, (sd*md).template cast().eval()*vcd); - VERIFY_IS_APPROX(scd*md*vcd,(scd*md.template cast()).eval()*vcd); - VERIFY_IS_APPROX(sd*mcd*vd, sd*mcd*vd.template cast().eval()); - VERIFY_IS_APPROX(scd*mcd*vd,scd*mcd*vd.template cast().eval()); - - VERIFY_IS_APPROX(sd*vcd.adjoint()*md, sd*vcd.adjoint()*md.template cast().eval()); - VERIFY_IS_APPROX(scd*vcd.adjoint()*md, scd*vcd.adjoint()*md.template cast().eval()); - VERIFY_IS_APPROX(sd*vd.adjoint()*mcd, sd*vd.adjoint().template cast().eval()*mcd); - VERIFY_IS_APPROX(scd*vd.adjoint()*mcd, scd*vd.adjoint().template cast().eval()*mcd); - - VERIFY_IS_APPROX( sd*vcd.adjoint()*md.template triangularView(), sd*vcd.adjoint()*md.template cast().eval().template triangularView()); - VERIFY_IS_APPROX(scd*vcd.adjoint()*md.template triangularView(), scd*vcd.adjoint()*md.template cast().eval().template triangularView()); - VERIFY_IS_APPROX( sd*vcd.adjoint()*md.transpose().template triangularView(), sd*vcd.adjoint()*md.transpose().template cast().eval().template triangularView()); - VERIFY_IS_APPROX(scd*vcd.adjoint()*md.transpose().template triangularView(), scd*vcd.adjoint()*md.transpose().template cast().eval().template triangularView()); - VERIFY_IS_APPROX( sd*vd.adjoint()*mcd.template triangularView(), sd*vd.adjoint().template cast().eval()*mcd.template triangularView()); - VERIFY_IS_APPROX(scd*vd.adjoint()*mcd.template triangularView(), scd*vd.adjoint().template cast().eval()*mcd.template triangularView()); - VERIFY_IS_APPROX( sd*vd.adjoint()*mcd.transpose().template triangularView(), sd*vd.adjoint().template cast().eval()*mcd.transpose().template triangularView()); - VERIFY_IS_APPROX(scd*vd.adjoint()*mcd.transpose().template triangularView(), scd*vd.adjoint().template cast().eval()*mcd.transpose().template triangularView()); + VERIFY_IS_APPROX(sd * md * mcd, (sd * md).template cast().eval() * mcd); + VERIFY_IS_APPROX(sd * mcd * md, sd * mcd * md.template cast()); + VERIFY_IS_APPROX(scd * md * mcd, scd * md.template cast().eval() * mcd); + VERIFY_IS_APPROX(scd * mcd * md, scd * mcd * md.template cast()); + + VERIFY_IS_APPROX(sf * mf * mcf, sf * mf.template cast() * mcf); + VERIFY_IS_APPROX(sf * mcf * mf, sf * mcf * mf.template cast()); + VERIFY_IS_APPROX(scf * mf * mcf, scf * mf.template cast() * mcf); + VERIFY_IS_APPROX(scf * mcf * mf, scf * mcf * mf.template cast()); + + VERIFY_IS_APPROX(sd * md.adjoint() * mcd, (sd * md).template cast().eval().adjoint() * mcd); + VERIFY_IS_APPROX(sd * mcd.adjoint() * md, sd * mcd.adjoint() * md.template cast()); + VERIFY_IS_APPROX(sd * md.adjoint() * mcd.adjoint(), (sd * md).template cast().eval().adjoint() * mcd.adjoint()); + VERIFY_IS_APPROX(sd * mcd.adjoint() * md.adjoint(), sd * mcd.adjoint() * md.template cast().adjoint()); + VERIFY_IS_APPROX(sd * md * mcd.adjoint(), (sd * md).template cast().eval() * mcd.adjoint()); + VERIFY_IS_APPROX(sd * mcd * md.adjoint(), sd * mcd * md.template cast().adjoint()); + + VERIFY_IS_APPROX(sf * mf.adjoint() * mcf, (sf * mf).template cast().eval().adjoint() * mcf); + VERIFY_IS_APPROX(sf * mcf.adjoint() * mf, sf * mcf.adjoint() * mf.template cast()); + VERIFY_IS_APPROX(sf * mf.adjoint() * mcf.adjoint(), (sf * mf).template cast().eval().adjoint() * mcf.adjoint()); + VERIFY_IS_APPROX(sf * mcf.adjoint() * mf.adjoint(), sf * mcf.adjoint() * mf.template cast().adjoint()); + VERIFY_IS_APPROX(sf * mf * mcf.adjoint(), (sf * mf).template cast().eval() * mcf.adjoint()); + VERIFY_IS_APPROX(sf * mcf * mf.adjoint(), sf * mcf * mf.template cast().adjoint()); + + VERIFY_IS_APPROX(sf * mf * vcf, (sf * mf).template cast().eval() * vcf); + VERIFY_IS_APPROX(scf * mf * vcf, (scf * mf.template cast()).eval() * vcf); + VERIFY_IS_APPROX(sf * mcf * vf, sf * mcf * vf.template cast()); + VERIFY_IS_APPROX(scf * mcf * vf, scf * mcf * vf.template cast()); + + VERIFY_IS_APPROX(sf * vcf.adjoint() * mf, sf * vcf.adjoint() * mf.template cast().eval()); + VERIFY_IS_APPROX(scf * vcf.adjoint() * mf, scf * vcf.adjoint() * mf.template cast().eval()); + VERIFY_IS_APPROX(sf * vf.adjoint() * mcf, sf * vf.adjoint().template cast().eval() * mcf); + VERIFY_IS_APPROX(scf * vf.adjoint() * mcf, scf * vf.adjoint().template cast().eval() * mcf); + + VERIFY_IS_APPROX(sd * md * vcd, (sd * md).template cast().eval() * vcd); + VERIFY_IS_APPROX(scd * md * vcd, (scd * md.template cast()).eval() * vcd); + VERIFY_IS_APPROX(sd * mcd * vd, sd * mcd * vd.template cast().eval()); + VERIFY_IS_APPROX(scd * mcd * vd, scd * mcd * vd.template cast().eval()); + + VERIFY_IS_APPROX(sd * vcd.adjoint() * md, sd * vcd.adjoint() * md.template cast().eval()); + VERIFY_IS_APPROX(scd * vcd.adjoint() * md, scd * vcd.adjoint() * md.template cast().eval()); + VERIFY_IS_APPROX(sd * vd.adjoint() * mcd, sd * vd.adjoint().template cast().eval() * mcd); + VERIFY_IS_APPROX(scd * vd.adjoint() * mcd, scd * vd.adjoint().template cast().eval() * mcd); + + VERIFY_IS_APPROX(sd * vcd.adjoint() * md.template triangularView(), + sd * vcd.adjoint() * md.template cast().eval().template triangularView()); + VERIFY_IS_APPROX(scd * vcd.adjoint() * md.template triangularView(), + scd * vcd.adjoint() * md.template cast().eval().template triangularView()); + VERIFY_IS_APPROX(sd * vcd.adjoint() * md.transpose().template triangularView(), + sd * vcd.adjoint() * md.transpose().template cast().eval().template triangularView()); + VERIFY_IS_APPROX(scd * vcd.adjoint() * md.transpose().template triangularView(), + scd * vcd.adjoint() * md.transpose().template cast().eval().template triangularView()); + VERIFY_IS_APPROX(sd * vd.adjoint() * mcd.template triangularView(), + sd * vd.adjoint().template cast().eval() * mcd.template triangularView()); + VERIFY_IS_APPROX(scd * vd.adjoint() * mcd.template triangularView(), + scd * vd.adjoint().template cast().eval() * mcd.template triangularView()); + VERIFY_IS_APPROX(sd * vd.adjoint() * mcd.transpose().template triangularView(), + sd * vd.adjoint().template cast().eval() * mcd.transpose().template triangularView()); + VERIFY_IS_APPROX(scd * vd.adjoint() * mcd.transpose().template triangularView(), + scd * vd.adjoint().template cast().eval() * mcd.transpose().template triangularView()); // Not supported yet: trmm -// VERIFY_IS_APPROX(sd*mcd*md.template triangularView(), sd*mcd*md.template cast().eval().template triangularView()); -// VERIFY_IS_APPROX(scd*mcd*md.template triangularView(), scd*mcd*md.template cast().eval().template triangularView()); -// VERIFY_IS_APPROX(sd*md*mcd.template triangularView(), sd*md.template cast().eval()*mcd.template triangularView()); -// VERIFY_IS_APPROX(scd*md*mcd.template triangularView(), scd*md.template cast().eval()*mcd.template triangularView()); + // VERIFY_IS_APPROX(sd*mcd*md.template triangularView(), sd*mcd*md.template cast().eval().template + // triangularView()); VERIFY_IS_APPROX(scd*mcd*md.template triangularView(), scd*mcd*md.template + // cast().eval().template triangularView()); VERIFY_IS_APPROX(sd*md*mcd.template triangularView(), + // sd*md.template cast().eval()*mcd.template triangularView()); VERIFY_IS_APPROX(scd*md*mcd.template + // triangularView(), scd*md.template cast().eval()*mcd.template triangularView()); // Not supported yet: symv -// VERIFY_IS_APPROX(sd*vcd.adjoint()*md.template selfadjointView(), sd*vcd.adjoint()*md.template cast().eval().template selfadjointView()); -// VERIFY_IS_APPROX(scd*vcd.adjoint()*md.template selfadjointView(), scd*vcd.adjoint()*md.template cast().eval().template selfadjointView()); -// VERIFY_IS_APPROX(sd*vd.adjoint()*mcd.template selfadjointView(), sd*vd.adjoint().template cast().eval()*mcd.template selfadjointView()); -// VERIFY_IS_APPROX(scd*vd.adjoint()*mcd.template selfadjointView(), scd*vd.adjoint().template cast().eval()*mcd.template selfadjointView()); + // VERIFY_IS_APPROX(sd*vcd.adjoint()*md.template selfadjointView(), sd*vcd.adjoint()*md.template + // cast().eval().template selfadjointView()); VERIFY_IS_APPROX(scd*vcd.adjoint()*md.template + // selfadjointView(), scd*vcd.adjoint()*md.template cast().eval().template selfadjointView()); + // VERIFY_IS_APPROX(sd*vd.adjoint()*mcd.template selfadjointView(), sd*vd.adjoint().template + // cast().eval()*mcd.template selfadjointView()); VERIFY_IS_APPROX(scd*vd.adjoint()*mcd.template + // selfadjointView(), scd*vd.adjoint().template cast().eval()*mcd.template selfadjointView()); // Not supported yet: symm -// VERIFY_IS_APPROX(sd*vcd.adjoint()*md.template selfadjointView(), sd*vcd.adjoint()*md.template cast().eval().template selfadjointView()); -// VERIFY_IS_APPROX(scd*vcd.adjoint()*md.template selfadjointView(), scd*vcd.adjoint()*md.template cast().eval().template selfadjointView()); -// VERIFY_IS_APPROX(sd*vd.adjoint()*mcd.template selfadjointView(), sd*vd.adjoint().template cast().eval()*mcd.template selfadjointView()); -// VERIFY_IS_APPROX(scd*vd.adjoint()*mcd.template selfadjointView(), scd*vd.adjoint().template cast().eval()*mcd.template selfadjointView()); + // VERIFY_IS_APPROX(sd*vcd.adjoint()*md.template selfadjointView(), sd*vcd.adjoint()*md.template + // cast().eval().template selfadjointView()); VERIFY_IS_APPROX(scd*vcd.adjoint()*md.template + // selfadjointView(), scd*vcd.adjoint()*md.template cast().eval().template selfadjointView()); + // VERIFY_IS_APPROX(sd*vd.adjoint()*mcd.template selfadjointView(), sd*vd.adjoint().template + // cast().eval()*mcd.template selfadjointView()); VERIFY_IS_APPROX(scd*vd.adjoint()*mcd.template + // selfadjointView(), scd*vd.adjoint().template cast().eval()*mcd.template selfadjointView()); rcd.setZero(); VERIFY_IS_APPROX(Mat_cd(rcd.template triangularView() = sd * mcd * md), - Mat_cd((sd * mcd * md.template cast().eval()).template triangularView())); + Mat_cd((sd * mcd * md.template cast().eval()).template triangularView())); VERIFY_IS_APPROX(Mat_cd(rcd.template triangularView() = sd * md * mcd), - Mat_cd((sd * md.template cast().eval() * mcd).template triangularView())); + Mat_cd((sd * md.template cast().eval() * mcd).template triangularView())); VERIFY_IS_APPROX(Mat_cd(rcd.template triangularView() = scd * mcd * md), - Mat_cd((scd * mcd * md.template cast().eval()).template triangularView())); + Mat_cd((scd * mcd * md.template cast().eval()).template triangularView())); VERIFY_IS_APPROX(Mat_cd(rcd.template triangularView() = scd * md * mcd), - Mat_cd((scd * md.template cast().eval() * mcd).template triangularView())); + Mat_cd((scd * md.template cast().eval() * mcd).template triangularView())); - VERIFY_IS_APPROX( md.array() * mcd.array(), md.template cast().eval().array() * mcd.array() ); - VERIFY_IS_APPROX( mcd.array() * md.array(), mcd.array() * md.template cast().eval().array() ); + VERIFY_IS_APPROX(md.array() * mcd.array(), md.template cast().eval().array() * mcd.array()); + VERIFY_IS_APPROX(mcd.array() * md.array(), mcd.array() * md.template cast().eval().array()); - VERIFY_IS_APPROX( md.array() + mcd.array(), md.template cast().eval().array() + mcd.array() ); - VERIFY_IS_APPROX( mcd.array() + md.array(), mcd.array() + md.template cast().eval().array() ); + VERIFY_IS_APPROX(md.array() + mcd.array(), md.template cast().eval().array() + mcd.array()); + VERIFY_IS_APPROX(mcd.array() + md.array(), mcd.array() + md.template cast().eval().array()); - VERIFY_IS_APPROX( md.array() - mcd.array(), md.template cast().eval().array() - mcd.array() ); - VERIFY_IS_APPROX( mcd.array() - md.array(), mcd.array() - md.template cast().eval().array() ); + VERIFY_IS_APPROX(md.array() - mcd.array(), md.template cast().eval().array() - mcd.array()); + VERIFY_IS_APPROX(mcd.array() - md.array(), mcd.array() - md.template cast().eval().array()); - if(mcd.array().abs().minCoeff()>epsd) - { - VERIFY_IS_APPROX( md.array() / mcd.array(), md.template cast().eval().array() / mcd.array() ); + if (mcd.array().abs().minCoeff() > epsd) { + VERIFY_IS_APPROX(md.array() / mcd.array(), md.template cast().eval().array() / mcd.array()); } - if(md.array().abs().minCoeff()>epsd) - { - VERIFY_IS_APPROX( mcd.array() / md.array(), mcd.array() / md.template cast().eval().array() ); + if (md.array().abs().minCoeff() > epsd) { + VERIFY_IS_APPROX(mcd.array() / md.array(), mcd.array() / md.template cast().eval().array()); } - if(md.array().abs().minCoeff()>epsd || mcd.array().abs().minCoeff()>epsd) - { - VERIFY_IS_APPROX( md.array().pow(mcd.array()), md.template cast().eval().array().pow(mcd.array()) ); - VERIFY_IS_APPROX( mcd.array().pow(md.array()), mcd.array().pow(md.template cast().eval().array()) ); + if (md.array().abs().minCoeff() > epsd || mcd.array().abs().minCoeff() > epsd) { + VERIFY_IS_APPROX(md.array().pow(mcd.array()), md.template cast().eval().array().pow(mcd.array())); + VERIFY_IS_APPROX(mcd.array().pow(md.array()), mcd.array().pow(md.template cast().eval().array())); - VERIFY_IS_APPROX( pow(md.array(),mcd.array()), md.template cast().eval().array().pow(mcd.array()) ); - VERIFY_IS_APPROX( pow(mcd.array(),md.array()), mcd.array().pow(md.template cast().eval().array()) ); + VERIFY_IS_APPROX(pow(md.array(), mcd.array()), md.template cast().eval().array().pow(mcd.array())); + VERIFY_IS_APPROX(pow(mcd.array(), md.array()), mcd.array().pow(md.template cast().eval().array())); } rcd = mcd; - VERIFY_IS_APPROX( rcd = md, md.template cast().eval() ); + VERIFY_IS_APPROX(rcd = md, md.template cast().eval()); rcd = mcd; - VERIFY_IS_APPROX( rcd += md, mcd + md.template cast().eval() ); + VERIFY_IS_APPROX(rcd += md, mcd + md.template cast().eval()); rcd = mcd; - VERIFY_IS_APPROX( rcd -= md, mcd - md.template cast().eval() ); + VERIFY_IS_APPROX(rcd -= md, mcd - md.template cast().eval()); rcd = mcd; - VERIFY_IS_APPROX( rcd.array() *= md.array(), mcd.array() * md.template cast().eval().array() ); + VERIFY_IS_APPROX(rcd.array() *= md.array(), mcd.array() * md.template cast().eval().array()); rcd = mcd; - if(md.array().abs().minCoeff()>epsd) - { - VERIFY_IS_APPROX( rcd.array() /= md.array(), mcd.array() / md.template cast().eval().array() ); + if (md.array().abs().minCoeff() > epsd) { + VERIFY_IS_APPROX(rcd.array() /= md.array(), mcd.array() / md.template cast().eval().array()); } rcd = mcd; - VERIFY_IS_APPROX( rcd.noalias() += md + mcd*md, mcd + (md.template cast().eval()) + mcd*(md.template cast().eval())); + VERIFY_IS_APPROX( + rcd.noalias() += md + mcd * md, mcd + (md.template cast().eval()) + mcd * (md.template cast().eval())); - VERIFY_IS_APPROX( rcd.noalias() = md*md, ((md*md).eval().template cast()) ); + VERIFY_IS_APPROX(rcd.noalias() = md * md, ((md * md).eval().template cast())); rcd = mcd; - VERIFY_IS_APPROX( rcd.noalias() += md*md, mcd + ((md*md).eval().template cast()) ); + VERIFY_IS_APPROX(rcd.noalias() += md * md, mcd + ((md * md).eval().template cast())); rcd = mcd; - VERIFY_IS_APPROX( rcd.noalias() -= md*md, mcd - ((md*md).eval().template cast()) ); + VERIFY_IS_APPROX(rcd.noalias() -= md * md, mcd - ((md * md).eval().template cast())); - VERIFY_IS_APPROX( rcd.noalias() = mcd + md*md, mcd + ((md*md).eval().template cast()) ); + VERIFY_IS_APPROX(rcd.noalias() = mcd + md * md, mcd + ((md * md).eval().template cast())); rcd = mcd; - VERIFY_IS_APPROX( rcd.noalias() += mcd + md*md, mcd + mcd + ((md*md).eval().template cast()) ); + VERIFY_IS_APPROX(rcd.noalias() += mcd + md * md, mcd + mcd + ((md * md).eval().template cast())); rcd = mcd; - VERIFY_IS_APPROX( rcd.noalias() -= mcd + md*md, - ((md*md).eval().template cast()) ); + VERIFY_IS_APPROX(rcd.noalias() -= mcd + md * md, -((md * md).eval().template cast())); } void test_mixingtypes() { - for(int i = 0; i < g_repeat; i++) { + for (int i = 0; i < g_repeat; i++) { CALL_SUBTEST_1(mixingtypes<3>()); CALL_SUBTEST_2(mixingtypes<4>()); - CALL_SUBTEST_3(mixingtypes(internal::random(1,EIGEN_TEST_MAX_SIZE))); + CALL_SUBTEST_3(mixingtypes(internal::random(1, EIGEN_TEST_MAX_SIZE))); CALL_SUBTEST_4(mixingtypes<3>()); CALL_SUBTEST_5(mixingtypes<4>()); - CALL_SUBTEST_6(mixingtypes(internal::random(1,EIGEN_TEST_MAX_SIZE))); - CALL_SUBTEST_7(raise_assertion(internal::random(1,EIGEN_TEST_MAX_SIZE))); + CALL_SUBTEST_6(mixingtypes(internal::random(1, EIGEN_TEST_MAX_SIZE))); + CALL_SUBTEST_7(raise_assertion(internal::random(1, EIGEN_TEST_MAX_SIZE))); } CALL_SUBTEST_7(raise_assertion<0>()); CALL_SUBTEST_7(raise_assertion<3>()); diff --git a/filmulator-gui/core/nlmeans/eigen/test/mpl2only.cpp b/filmulator-gui/core/nlmeans/eigen/test/mpl2only.cpp index 7d04d6bb..a522d29b 100644 --- a/filmulator-gui/core/nlmeans/eigen/test/mpl2only.cpp +++ b/filmulator-gui/core/nlmeans/eigen/test/mpl2only.cpp @@ -9,14 +9,11 @@ #define EIGEN_MPL2_ONLY #include +#include +#include +#include #include #include #include -#include -#include -#include -int main() -{ - return 0; -} +int main() { return 0; } diff --git a/filmulator-gui/core/nlmeans/eigen/test/nesting_ops.cpp b/filmulator-gui/core/nlmeans/eigen/test/nesting_ops.cpp index a419b0e4..be8b8868 100644 --- a/filmulator-gui/core/nlmeans/eigen/test/nesting_ops.cpp +++ b/filmulator-gui/core/nlmeans/eigen/test/nesting_ops.cpp @@ -12,96 +12,93 @@ #include "main.h" -template -void use_n_times(const XprType &xpr) +template void use_n_times(const XprType &xpr) { - typename internal::nested_eval::type mat(xpr); + typename internal::nested_eval::type mat(xpr); typename XprType::PlainObject res(mat.rows(), mat.cols()); - nb_temporaries--; // remove res + nb_temporaries--;// remove res res.setZero(); - for(int i=0; i -bool verify_eval_type(const XprType &, const ReferenceType&) +template bool verify_eval_type(const XprType &, const ReferenceType &) { - typedef typename internal::nested_eval::type EvalType; - return internal::is_same::type, typename internal::remove_all::type>::value; + typedef typename internal::nested_eval::type EvalType; + return internal::is_same::type, + typename internal::remove_all::type>::value; } -template void run_nesting_ops_1(const MatrixType& _m) +template void run_nesting_ops_1(const MatrixType &_m) { - typename internal::nested_eval::type m(_m); + typename internal::nested_eval::type m(_m); // Make really sure that we are in debug mode! VERIFY_RAISES_ASSERT(eigen_assert(false)); // The only intention of these tests is to ensure that this code does // not trigger any asserts or segmentation faults... more to come. - VERIFY_IS_APPROX( (m.transpose() * m).diagonal().sum(), (m.transpose() * m).diagonal().sum() ); - VERIFY_IS_APPROX( (m.transpose() * m).diagonal().array().abs().sum(), (m.transpose() * m).diagonal().array().abs().sum() ); + VERIFY_IS_APPROX((m.transpose() * m).diagonal().sum(), (m.transpose() * m).diagonal().sum()); + VERIFY_IS_APPROX( + (m.transpose() * m).diagonal().array().abs().sum(), (m.transpose() * m).diagonal().array().abs().sum()); - VERIFY_IS_APPROX( (m.transpose() * m).array().abs().sum(), (m.transpose() * m).array().abs().sum() ); + VERIFY_IS_APPROX((m.transpose() * m).array().abs().sum(), (m.transpose() * m).array().abs().sum()); } -template void run_nesting_ops_2(const MatrixType& _m) +template void run_nesting_ops_2(const MatrixType &_m) { typedef typename MatrixType::Scalar Scalar; Index rows = _m.rows(); Index cols = _m.cols(); - MatrixType m1 = MatrixType::Random(rows,cols); - Matrix m2; + MatrixType m1 = MatrixType::Random(rows, cols); + Matrix m2; - if((MatrixType::SizeAtCompileTime==Dynamic)) - { - VERIFY_EVALUATION_COUNT( use_n_times<1>(m1 + m1*m1), 1 ); - VERIFY_EVALUATION_COUNT( use_n_times<10>(m1 + m1*m1), 1 ); + if ((MatrixType::SizeAtCompileTime == Dynamic)) { + VERIFY_EVALUATION_COUNT(use_n_times<1>(m1 + m1 * m1), 1); + VERIFY_EVALUATION_COUNT(use_n_times<10>(m1 + m1 * m1), 1); - VERIFY_EVALUATION_COUNT( use_n_times<1>(m1.template triangularView().solve(m1.col(0))), 1 ); - VERIFY_EVALUATION_COUNT( use_n_times<10>(m1.template triangularView().solve(m1.col(0))), 1 ); + VERIFY_EVALUATION_COUNT(use_n_times<1>(m1.template triangularView().solve(m1.col(0))), 1); + VERIFY_EVALUATION_COUNT(use_n_times<10>(m1.template triangularView().solve(m1.col(0))), 1); - VERIFY_EVALUATION_COUNT( use_n_times<1>(Scalar(2)*m1.template triangularView().solve(m1.col(0))), 2 ); // FIXME could be one by applying the scaling in-place on the solve result - VERIFY_EVALUATION_COUNT( use_n_times<1>(m1.col(0)+m1.template triangularView().solve(m1.col(0))), 2 ); // FIXME could be one by adding m1.col() inplace - VERIFY_EVALUATION_COUNT( use_n_times<10>(m1.col(0)+m1.template triangularView().solve(m1.col(0))), 2 ); + VERIFY_EVALUATION_COUNT(use_n_times<1>(Scalar(2) * m1.template triangularView().solve(m1.col(0))), + 2);// FIXME could be one by applying the scaling in-place on the solve result + VERIFY_EVALUATION_COUNT(use_n_times<1>(m1.col(0) + m1.template triangularView().solve(m1.col(0))), + 2);// FIXME could be one by adding m1.col() inplace + VERIFY_EVALUATION_COUNT(use_n_times<10>(m1.col(0) + m1.template triangularView().solve(m1.col(0))), 2); } { - VERIFY( verify_eval_type<10>(m1, m1) ); - if(!NumTraits::IsComplex) - { - VERIFY( verify_eval_type<3>(2*m1, 2*m1) ); - VERIFY( verify_eval_type<4>(2*m1, m1) ); - } - else - { - VERIFY( verify_eval_type<2>(2*m1, 2*m1) ); - VERIFY( verify_eval_type<3>(2*m1, m1) ); + VERIFY(verify_eval_type<10>(m1, m1)); + if (!NumTraits::IsComplex) { + VERIFY(verify_eval_type<3>(2 * m1, 2 * m1)); + VERIFY(verify_eval_type<4>(2 * m1, m1)); + } else { + VERIFY(verify_eval_type<2>(2 * m1, 2 * m1)); + VERIFY(verify_eval_type<3>(2 * m1, m1)); } - VERIFY( verify_eval_type<2>(m1+m1, m1+m1) ); - VERIFY( verify_eval_type<3>(m1+m1, m1) ); - VERIFY( verify_eval_type<1>(m1*m1.transpose(), m2) ); - VERIFY( verify_eval_type<1>(m1*(m1+m1).transpose(), m2) ); - VERIFY( verify_eval_type<2>(m1*m1.transpose(), m2) ); - VERIFY( verify_eval_type<1>(m1+m1*m1, m1) ); - - VERIFY( verify_eval_type<1>(m1.template triangularView().solve(m1), m1) ); - VERIFY( verify_eval_type<1>(m1+m1.template triangularView().solve(m1), m1) ); + VERIFY(verify_eval_type<2>(m1 + m1, m1 + m1)); + VERIFY(verify_eval_type<3>(m1 + m1, m1)); + VERIFY(verify_eval_type<1>(m1 * m1.transpose(), m2)); + VERIFY(verify_eval_type<1>(m1 * (m1 + m1).transpose(), m2)); + VERIFY(verify_eval_type<2>(m1 * m1.transpose(), m2)); + VERIFY(verify_eval_type<1>(m1 + m1 * m1, m1)); + + VERIFY(verify_eval_type<1>(m1.template triangularView().solve(m1), m1)); + VERIFY(verify_eval_type<1>(m1 + m1.template triangularView().solve(m1), m1)); } } void test_nesting_ops() { - CALL_SUBTEST_1(run_nesting_ops_1(MatrixXf::Random(25,25))); - CALL_SUBTEST_2(run_nesting_ops_1(MatrixXcd::Random(25,25))); + CALL_SUBTEST_1(run_nesting_ops_1(MatrixXf::Random(25, 25))); + CALL_SUBTEST_2(run_nesting_ops_1(MatrixXcd::Random(25, 25))); CALL_SUBTEST_3(run_nesting_ops_1(Matrix4f::Random())); CALL_SUBTEST_4(run_nesting_ops_1(Matrix2d::Random())); - Index s = internal::random(1,EIGEN_TEST_MAX_SIZE); - CALL_SUBTEST_1( run_nesting_ops_2(MatrixXf(s,s)) ); - CALL_SUBTEST_2( run_nesting_ops_2(MatrixXcd(s,s)) ); - CALL_SUBTEST_3( run_nesting_ops_2(Matrix4f()) ); - CALL_SUBTEST_4( run_nesting_ops_2(Matrix2d()) ); + Index s = internal::random(1, EIGEN_TEST_MAX_SIZE); + CALL_SUBTEST_1(run_nesting_ops_2(MatrixXf(s, s))); + CALL_SUBTEST_2(run_nesting_ops_2(MatrixXcd(s, s))); + CALL_SUBTEST_3(run_nesting_ops_2(Matrix4f())); + CALL_SUBTEST_4(run_nesting_ops_2(Matrix2d())); TEST_SET_BUT_UNUSED_VARIABLE(s) } diff --git a/filmulator-gui/core/nlmeans/eigen/test/nomalloc.cpp b/filmulator-gui/core/nlmeans/eigen/test/nomalloc.cpp index b7ea4d36..4a92492e 100644 --- a/filmulator-gui/core/nlmeans/eigen/test/nomalloc.cpp +++ b/filmulator-gui/core/nlmeans/eigen/test/nomalloc.cpp @@ -20,29 +20,26 @@ #include #include -template void nomalloc(const MatrixType& m) +template void nomalloc(const MatrixType &m) { /* this test check no dynamic memory allocation are issued with fixed-size matrices - */ + */ typedef typename MatrixType::Scalar Scalar; Index rows = m.rows(); Index cols = m.cols(); - MatrixType m1 = MatrixType::Random(rows, cols), - m2 = MatrixType::Random(rows, cols), - m3(rows, cols); + MatrixType m1 = MatrixType::Random(rows, cols), m2 = MatrixType::Random(rows, cols), m3(rows, cols); Scalar s1 = internal::random(); - Index r = internal::random(0, rows-1), - c = internal::random(0, cols-1); + Index r = internal::random(0, rows - 1), c = internal::random(0, cols - 1); + + VERIFY_IS_APPROX((m1 + m2) * s1, s1 * m1 + s1 * m2); + VERIFY_IS_APPROX((m1 + m2)(r, c), (m1(r, c)) + (m2(r, c))); + VERIFY_IS_APPROX(m1.cwiseProduct(m1.block(0, 0, rows, cols)), (m1.array() * m1.array()).matrix()); + VERIFY_IS_APPROX((m1 * m1.transpose()) * m2, m1 * (m1.transpose() * m2)); - VERIFY_IS_APPROX((m1+m2)*s1, s1*m1+s1*m2); - VERIFY_IS_APPROX((m1+m2)(r,c), (m1(r,c))+(m2(r,c))); - VERIFY_IS_APPROX(m1.cwiseProduct(m1.block(0,0,rows,cols)), (m1.array()*m1.array()).matrix()); - VERIFY_IS_APPROX((m1*m1.transpose())*m2, m1*(m1.transpose()*m2)); - m2.col(0).noalias() = m1 * m1.col(0); m2.col(0).noalias() -= m1.adjoint() * m1.col(0); m2.col(0).noalias() -= m1 * m1.row(0).adjoint(); @@ -52,8 +49,8 @@ template void nomalloc(const MatrixType& m) m2.row(0).noalias() -= m1.row(0) * m1.adjoint(); m2.row(0).noalias() -= m1.col(0).adjoint() * m1; m2.row(0).noalias() -= m1.col(0).adjoint() * m1.adjoint(); - VERIFY_IS_APPROX(m2,m2); - + VERIFY_IS_APPROX(m2, m2); + m2.col(0).noalias() = m1.template triangularView() * m1.col(0); m2.col(0).noalias() -= m1.adjoint().template triangularView() * m1.col(0); m2.col(0).noalias() -= m1.template triangularView() * m1.row(0).adjoint(); @@ -63,8 +60,8 @@ template void nomalloc(const MatrixType& m) m2.row(0).noalias() -= m1.row(0) * m1.adjoint().template triangularView(); m2.row(0).noalias() -= m1.col(0).adjoint() * m1.template triangularView(); m2.row(0).noalias() -= m1.col(0).adjoint() * m1.adjoint().template triangularView(); - VERIFY_IS_APPROX(m2,m2); - + VERIFY_IS_APPROX(m2, m2); + m2.col(0).noalias() = m1.template selfadjointView() * m1.col(0); m2.col(0).noalias() -= m1.adjoint().template selfadjointView() * m1.col(0); m2.col(0).noalias() -= m1.template selfadjointView() * m1.row(0).adjoint(); @@ -74,115 +71,121 @@ template void nomalloc(const MatrixType& m) m2.row(0).noalias() -= m1.row(0) * m1.adjoint().template selfadjointView(); m2.row(0).noalias() -= m1.col(0).adjoint() * m1.template selfadjointView(); m2.row(0).noalias() -= m1.col(0).adjoint() * m1.adjoint().template selfadjointView(); - VERIFY_IS_APPROX(m2,m2); - - m2.template selfadjointView().rankUpdate(m1.col(0),-1); - m2.template selfadjointView().rankUpdate(m1.row(0),-1); - m2.template selfadjointView().rankUpdate(m1.col(0), m1.col(0)); // rank-2 + VERIFY_IS_APPROX(m2, m2); + + m2.template selfadjointView().rankUpdate(m1.col(0), -1); + m2.template selfadjointView().rankUpdate(m1.row(0), -1); + m2.template selfadjointView().rankUpdate(m1.col(0), m1.col(0));// rank-2 // The following fancy matrix-matrix products are not safe yet regarding static allocation m2.template selfadjointView().rankUpdate(m1); m2 += m2.template triangularView() * m1; m2.template triangularView() = m2 * m2; m1 += m1.template selfadjointView() * m2; - VERIFY_IS_APPROX(m2,m2); + VERIFY_IS_APPROX(m2, m2); } -template -void ctms_decompositions() +template void ctms_decompositions() { const int maxSize = 16; - const int size = 12; + const int size = 12; - typedef Eigen::Matrix Matrix; + typedef Eigen::Matrix Matrix; - typedef Eigen::Matrix Vector; + typedef Eigen::Matrix Vector; - typedef Eigen::Matrix, - Eigen::Dynamic, Eigen::Dynamic, - 0, - maxSize, maxSize> ComplexMatrix; + typedef Eigen::Matrix, Eigen::Dynamic, Eigen::Dynamic, 0, maxSize, maxSize> ComplexMatrix; const Matrix A(Matrix::Random(size, size)), B(Matrix::Random(size, size)); - Matrix X(size,size); + Matrix X(size, size); const ComplexMatrix complexA(ComplexMatrix::Random(size, size)); const Matrix saA = A.adjoint() * A; const Vector b(Vector::Random(size)); Vector x(size); // Cholesky module - Eigen::LLT LLT; LLT.compute(A); + Eigen::LLT LLT; + LLT.compute(A); X = LLT.solve(B); x = LLT.solve(b); - Eigen::LDLT LDLT; LDLT.compute(A); + Eigen::LDLT LDLT; + LDLT.compute(A); X = LDLT.solve(B); x = LDLT.solve(b); // Eigenvalues module - Eigen::HessenbergDecomposition hessDecomp; hessDecomp.compute(complexA); - Eigen::ComplexSchur cSchur(size); cSchur.compute(complexA); - Eigen::ComplexEigenSolver cEigSolver; cEigSolver.compute(complexA); - Eigen::EigenSolver eigSolver; eigSolver.compute(A); - Eigen::SelfAdjointEigenSolver saEigSolver(size); saEigSolver.compute(saA); - Eigen::Tridiagonalization tridiag; tridiag.compute(saA); + Eigen::HessenbergDecomposition hessDecomp; + hessDecomp.compute(complexA); + Eigen::ComplexSchur cSchur(size); + cSchur.compute(complexA); + Eigen::ComplexEigenSolver cEigSolver; + cEigSolver.compute(complexA); + Eigen::EigenSolver eigSolver; + eigSolver.compute(A); + Eigen::SelfAdjointEigenSolver saEigSolver(size); + saEigSolver.compute(saA); + Eigen::Tridiagonalization tridiag; + tridiag.compute(saA); // LU module - Eigen::PartialPivLU ppLU; ppLU.compute(A); + Eigen::PartialPivLU ppLU; + ppLU.compute(A); X = ppLU.solve(B); x = ppLU.solve(b); - Eigen::FullPivLU fpLU; fpLU.compute(A); + Eigen::FullPivLU fpLU; + fpLU.compute(A); X = fpLU.solve(B); x = fpLU.solve(b); // QR module - Eigen::HouseholderQR hQR; hQR.compute(A); + Eigen::HouseholderQR hQR; + hQR.compute(A); X = hQR.solve(B); x = hQR.solve(b); - Eigen::ColPivHouseholderQR cpQR; cpQR.compute(A); + Eigen::ColPivHouseholderQR cpQR; + cpQR.compute(A); X = cpQR.solve(B); x = cpQR.solve(b); - Eigen::FullPivHouseholderQR fpQR; fpQR.compute(A); + Eigen::FullPivHouseholderQR fpQR; + fpQR.compute(A); // FIXME X = fpQR.solve(B); x = fpQR.solve(b); // SVD module - Eigen::JacobiSVD jSVD; jSVD.compute(A, ComputeFullU | ComputeFullV); + Eigen::JacobiSVD jSVD; + jSVD.compute(A, ComputeFullU | ComputeFullV); } -void test_zerosized() { +void test_zerosized() +{ // default constructors: Eigen::MatrixXd A; Eigen::VectorXd v; // explicit zero-sized: - Eigen::ArrayXXd A0(0,0); + Eigen::ArrayXXd A0(0, 0); Eigen::ArrayXd v0(0); // assigning empty objects to each other: - A=A0; - v=v0; + A = A0; + v = v0; } -template void test_reference(const MatrixType& m) { +template void test_reference(const MatrixType &m) +{ typedef typename MatrixType::Scalar Scalar; - enum { Flag = MatrixType::IsRowMajor ? Eigen::RowMajor : Eigen::ColMajor}; - enum { TransposeFlag = !MatrixType::IsRowMajor ? Eigen::RowMajor : Eigen::ColMajor}; - typename MatrixType::Index rows = m.rows(), cols=m.cols(); - typedef Eigen::Matrix MatrixX; + enum { Flag = MatrixType::IsRowMajor ? Eigen::RowMajor : Eigen::ColMajor }; + enum { TransposeFlag = !MatrixType::IsRowMajor ? Eigen::RowMajor : Eigen::ColMajor }; + typename MatrixType::Index rows = m.rows(), cols = m.cols(); + typedef Eigen::Matrix MatrixX; typedef Eigen::Matrix MatrixXT; // Dynamic reference: - typedef Eigen::Ref Ref; - typedef Eigen::Ref RefT; + typedef Eigen::Ref Ref; + typedef Eigen::Ref RefT; Ref r1(m); - Ref r2(m.block(rows/3, cols/4, rows/2, cols/2)); + Ref r2(m.block(rows / 3, cols / 4, rows / 2, cols / 2)); RefT r3(m.transpose()); - RefT r4(m.topLeftCorner(rows/2, cols/2).transpose()); + RefT r4(m.topLeftCorner(rows / 2, cols / 2).transpose()); VERIFY_RAISES_ASSERT(RefT r5(m)); VERIFY_RAISES_ASSERT(Ref r6(m.transpose())); @@ -193,36 +196,35 @@ template void test_reference(const MatrixType& m) { RefT r9 = r3; // Initializing from a compatible Ref shall also never malloc - Eigen::Ref > r10=r8, r11=m; + Eigen::Ref> r10 = r8, r11 = m; // Initializing from an incompatible Ref will malloc: typedef Eigen::Ref RefAligned; - VERIFY_RAISES_ASSERT(RefAligned r12=r10); - VERIFY_RAISES_ASSERT(Ref r13=r10); // r10 has more dynamic strides - + VERIFY_RAISES_ASSERT(RefAligned r12 = r10); + VERIFY_RAISES_ASSERT(Ref r13 = r10);// r10 has more dynamic strides } void test_nomalloc() { // create some dynamic objects - Eigen::MatrixXd M1 = MatrixXd::Random(3,3); - Ref R1 = 2.0*M1; // Ref requires temporary + Eigen::MatrixXd M1 = MatrixXd::Random(3, 3); + Ref R1 = 2.0 * M1;// Ref requires temporary // from here on prohibit malloc: Eigen::internal::set_is_malloc_allowed(false); // check that our operator new is indeed called: - VERIFY_RAISES_ASSERT(MatrixXd dummy(MatrixXd::Random(3,3))); - CALL_SUBTEST_1(nomalloc(Matrix()) ); - CALL_SUBTEST_2(nomalloc(Matrix4d()) ); - CALL_SUBTEST_3(nomalloc(Matrix()) ); - + VERIFY_RAISES_ASSERT(MatrixXd dummy(MatrixXd::Random(3, 3))); + CALL_SUBTEST_1(nomalloc(Matrix())); + CALL_SUBTEST_2(nomalloc(Matrix4d())); + CALL_SUBTEST_3(nomalloc(Matrix())); + // Check decomposition modules with dynamic matrices that have a known compile-time max size (ctms) CALL_SUBTEST_4(ctms_decompositions()); CALL_SUBTEST_5(test_zerosized()); - CALL_SUBTEST_6(test_reference(Matrix())); + CALL_SUBTEST_6(test_reference(Matrix())); CALL_SUBTEST_7(test_reference(R1)); CALL_SUBTEST_8(Ref R2 = M1.topRows<2>(); test_reference(R2)); } diff --git a/filmulator-gui/core/nlmeans/eigen/test/nullary.cpp b/filmulator-gui/core/nlmeans/eigen/test/nullary.cpp index acd55506..bbaf950d 100644 --- a/filmulator-gui/core/nlmeans/eigen/test/nullary.cpp +++ b/filmulator-gui/core/nlmeans/eigen/test/nullary.cpp @@ -10,191 +10,182 @@ #include "main.h" -template -bool equalsIdentity(const MatrixType& A) +template bool equalsIdentity(const MatrixType &A) { typedef typename MatrixType::Scalar Scalar; Scalar zero = static_cast(0); bool offDiagOK = true; for (Index i = 0; i < A.rows(); ++i) { - for (Index j = i+1; j < A.cols(); ++j) { - offDiagOK = offDiagOK && (A(i,j) == zero); - } + for (Index j = i + 1; j < A.cols(); ++j) { offDiagOK = offDiagOK && (A(i, j) == zero); } } for (Index i = 0; i < A.rows(); ++i) { - for (Index j = 0; j < (std::min)(i, A.cols()); ++j) { - offDiagOK = offDiagOK && (A(i,j) == zero); - } + for (Index j = 0; j < (std::min)(i, A.cols()); ++j) { offDiagOK = offDiagOK && (A(i, j) == zero); } } bool diagOK = (A.diagonal().array() == 1).all(); return offDiagOK && diagOK; - } template -void check_extremity_accuracy(const VectorType &v, const typename VectorType::Scalar &low, const typename VectorType::Scalar &high) +void check_extremity_accuracy(const VectorType &v, + const typename VectorType::Scalar &low, + const typename VectorType::Scalar &high) { typedef typename VectorType::Scalar Scalar; typedef typename VectorType::RealScalar RealScalar; - RealScalar prec = internal::is_same::value ? NumTraits::dummy_precision()*10 : NumTraits::dummy_precision()/10; + RealScalar prec = internal::is_same::value ? NumTraits::dummy_precision() * 10 + : NumTraits::dummy_precision() / 10; Index size = v.size(); - if(size<20) - return; - - for (int i=0; isize-6) - { - Scalar ref = (low*RealScalar(size-i-1))/RealScalar(size-1) + (high*RealScalar(i))/RealScalar(size-1); - if(std::abs(ref)>1) - { - if(!internal::isApprox(v(i), ref, prec)) - std::cout << v(i) << " != " << ref << " ; relative error: " << std::abs((v(i)-ref)/ref) << " ; required precision: " << prec << " ; range: " << low << "," << high << " ; i: " << i << "\n"; - VERIFY(internal::isApprox(v(i), (low*RealScalar(size-i-1))/RealScalar(size-1) + (high*RealScalar(i))/RealScalar(size-1), prec)); + if (size < 20) return; + + for (int i = 0; i < size; ++i) { + if (i < 5 || i > size - 6) { + Scalar ref = + (low * RealScalar(size - i - 1)) / RealScalar(size - 1) + (high * RealScalar(i)) / RealScalar(size - 1); + if (std::abs(ref) > 1) { + if (!internal::isApprox(v(i), ref, prec)) + std::cout << v(i) << " != " << ref << " ; relative error: " << std::abs((v(i) - ref) / ref) + << " ; required precision: " << prec << " ; range: " << low << "," << high << " ; i: " << i + << "\n"; + VERIFY(internal::isApprox(v(i), + (low * RealScalar(size - i - 1)) / RealScalar(size - 1) + (high * RealScalar(i)) / RealScalar(size - 1), + prec)); } } } } -template -void testVectorType(const VectorType& base) +template void testVectorType(const VectorType &base) { typedef typename VectorType::Scalar Scalar; typedef typename VectorType::RealScalar RealScalar; const Index size = base.size(); - - Scalar high = internal::random(-500,500); - Scalar low = (size == 1 ? high : internal::random(-500,500)); - if (low>high) std::swap(low,high); + + Scalar high = internal::random(-500, 500); + Scalar low = (size == 1 ? high : internal::random(-500, 500)); + if (low > high) std::swap(low, high); // check low==high - if(internal::random(0.f,1.f)<0.05f) - low = high; + if (internal::random(0.f, 1.f) < 0.05f) low = high; // check abs(low) >> abs(high) - else if(size>2 && std::numeric_limits::max_exponent10>0 && internal::random(0.f,1.f)<0.1f) - low = -internal::random(1,2) * RealScalar(std::pow(RealScalar(10),std::numeric_limits::max_exponent10/2)); + else if (size > 2 && std::numeric_limits::max_exponent10 > 0 && internal::random(0.f, 1.f) < 0.1f) + low = -internal::random(1, 2) + * RealScalar(std::pow(RealScalar(10), std::numeric_limits::max_exponent10 / 2)); - const Scalar step = ((size == 1) ? 1 : (high-low)/(size-1)); + const Scalar step = ((size == 1) ? 1 : (high - low) / (size - 1)); // check whether the result yields what we expect it to do VectorType m(base); - m.setLinSpaced(size,low,high); + m.setLinSpaced(size, low, high); - if(!NumTraits::IsInteger) - { + if (!NumTraits::IsInteger) { VectorType n(size); - for (int i=0; i::IsInteger) || ((high-low)>=size && (Index(high-low)%(size-1))==0) || (Index(high-low+1)::IsInteger) || ((high - low) >= size && (Index(high - low) % (size - 1)) == 0) + || (Index(high - low + 1) < size && (size % Index(high - low + 1)) == 0)) { VectorType n(size); - if((!NumTraits::IsInteger) || (high-low>=size)) - for (int i=0; i::IsInteger) || (high - low >= size)) + for (int i = 0; i < size; ++i) n(i) = size == 1 ? low : (low + ((high - low) * Scalar(i)) / (size - 1)); else - for (int i=0; i::IsInteger) - CALL_SUBTEST( check_extremity_accuracy(m, low, high) ); + m = VectorType::LinSpaced(size, low, high); + VERIFY_IS_APPROX(m, n); + VERIFY(internal::isApprox(m(m.size() - 1), high)); + VERIFY(size == 1 || internal::isApprox(m(0), low)); + VERIFY_IS_EQUAL(m(m.size() - 1), high); + if (!NumTraits::IsInteger) CALL_SUBTEST(check_extremity_accuracy(m, low, high)); } - VERIFY( m(m.size()-1) <= high ); - VERIFY( (m.array() <= high).all() ); - VERIFY( (m.array() >= low).all() ); + VERIFY(m(m.size() - 1) <= high); + VERIFY((m.array() <= high).all()); + VERIFY((m.array() >= low).all()); - VERIFY( m(m.size()-1) >= low ); - if(size>=1) - { - VERIFY( internal::isApprox(m(0),low) ); - VERIFY_IS_EQUAL(m(0) , low); + VERIFY(m(m.size() - 1) >= low); + if (size >= 1) { + VERIFY(internal::isApprox(m(0), low)); + VERIFY_IS_EQUAL(m(0), low); } // check whether everything works with row and col major vectors - Matrix row_vector(size); - Matrix col_vector(size); - row_vector.setLinSpaced(size,low,high); - col_vector.setLinSpaced(size,low,high); + Matrix row_vector(size); + Matrix col_vector(size); + row_vector.setLinSpaced(size, low, high); + col_vector.setLinSpaced(size, low, high); // when using the extended precision (e.g., FPU) the relative error might exceed 1 bit // when computing the squared sum in isApprox, thus the 2x factor. - VERIFY( row_vector.isApprox(col_vector.transpose(), Scalar(2)*NumTraits::epsilon())); + VERIFY(row_vector.isApprox(col_vector.transpose(), Scalar(2) * NumTraits::epsilon())); - Matrix size_changer(size+50); - size_changer.setLinSpaced(size,low,high); - VERIFY( size_changer.size() == size ); + Matrix size_changer(size + 50); + size_changer.setLinSpaced(size, low, high); + VERIFY(size_changer.size() == size); - typedef Matrix ScalarMatrix; + typedef Matrix ScalarMatrix; ScalarMatrix scalar; - scalar.setLinSpaced(1,low,high); - VERIFY_IS_APPROX( scalar, ScalarMatrix::Constant(high) ); - VERIFY_IS_APPROX( ScalarMatrix::LinSpaced(1,low,high), ScalarMatrix::Constant(high) ); + scalar.setLinSpaced(1, low, high); + VERIFY_IS_APPROX(scalar, ScalarMatrix::Constant(high)); + VERIFY_IS_APPROX(ScalarMatrix::LinSpaced(1, low, high), ScalarMatrix::Constant(high)); // regression test for bug 526 (linear vectorized transversal) if (size > 1 && (!NumTraits::IsInteger)) { - m.tail(size-1).setLinSpaced(low, high); - VERIFY_IS_APPROX(m(size-1), high); + m.tail(size - 1).setLinSpaced(low, high); + VERIFY_IS_APPROX(m(size - 1), high); } // regression test for bug 1383 (LinSpaced with empty size/range) { - Index n0 = VectorType::SizeAtCompileTime==Dynamic ? 0 : VectorType::SizeAtCompileTime; + Index n0 = VectorType::SizeAtCompileTime == Dynamic ? 0 : VectorType::SizeAtCompileTime; low = internal::random(); - m = VectorType::LinSpaced(n0,low,low-1); - VERIFY(m.size()==n0); + m = VectorType::LinSpaced(n0, low, low - 1); + VERIFY(m.size() == n0); - if(VectorType::SizeAtCompileTime==Dynamic) - { - VERIFY_IS_EQUAL(VectorType::LinSpaced(n0,0,Scalar(n0-1)).sum(),Scalar(0)); - VERIFY_IS_EQUAL(VectorType::LinSpaced(n0,low,low-1).sum(),Scalar(0)); + if (VectorType::SizeAtCompileTime == Dynamic) { + VERIFY_IS_EQUAL(VectorType::LinSpaced(n0, 0, Scalar(n0 - 1)).sum(), Scalar(0)); + VERIFY_IS_EQUAL(VectorType::LinSpaced(n0, low, low - 1).sum(), Scalar(0)); } - m.setLinSpaced(n0,0,Scalar(n0-1)); - VERIFY(m.size()==n0); - m.setLinSpaced(n0,low,low-1); - VERIFY(m.size()==n0); + m.setLinSpaced(n0, 0, Scalar(n0 - 1)); + VERIFY(m.size() == n0); + m.setLinSpaced(n0, low, low - 1); + VERIFY(m.size() == n0); // empty range only: - VERIFY_IS_APPROX(VectorType::LinSpaced(size,low,low),VectorType::Constant(size,low)); - m.setLinSpaced(size,low,low); - VERIFY_IS_APPROX(m,VectorType::Constant(size,low)); + VERIFY_IS_APPROX(VectorType::LinSpaced(size, low, low), VectorType::Constant(size, low)); + m.setLinSpaced(size, low, low); + VERIFY_IS_APPROX(m, VectorType::Constant(size, low)); - if(NumTraits::IsInteger) - { - VERIFY_IS_APPROX( VectorType::LinSpaced(size,low,Scalar(low+size-1)), VectorType::LinSpaced(size,Scalar(low+size-1),low).reverse() ); + if (NumTraits::IsInteger) { + VERIFY_IS_APPROX(VectorType::LinSpaced(size, low, Scalar(low + size - 1)), + VectorType::LinSpaced(size, Scalar(low + size - 1), low).reverse()); - if(VectorType::SizeAtCompileTime==Dynamic) - { + if (VectorType::SizeAtCompileTime == Dynamic) { // Check negative multiplicator path: - for(Index k=1; k<5; ++k) - VERIFY_IS_APPROX( VectorType::LinSpaced(size,low,Scalar(low+(size-1)*k)), VectorType::LinSpaced(size,Scalar(low+(size-1)*k),low).reverse() ); + for (Index k = 1; k < 5; ++k) + VERIFY_IS_APPROX(VectorType::LinSpaced(size, low, Scalar(low + (size - 1) * k)), + VectorType::LinSpaced(size, Scalar(low + (size - 1) * k), low).reverse()); // Check negative divisor path: - for(Index k=1; k<5; ++k) - VERIFY_IS_APPROX( VectorType::LinSpaced(size*k,low,Scalar(low+size-1)), VectorType::LinSpaced(size*k,Scalar(low+size-1),low).reverse() ); + for (Index k = 1; k < 5; ++k) + VERIFY_IS_APPROX(VectorType::LinSpaced(size * k, low, Scalar(low + size - 1)), + VectorType::LinSpaced(size * k, Scalar(low + size - 1), low).reverse()); } } } } -template -void testMatrixType(const MatrixType& m) +template void testMatrixType(const MatrixType &m) { using std::abs; const Index rows = m.rows(); @@ -205,7 +196,7 @@ void testMatrixType(const MatrixType& m) Scalar s1; do { s1 = internal::random(); - } while(abs(s1)::IsInteger)); + } while (abs(s1) < RealScalar(1e-5) && (!NumTraits::IsInteger)); MatrixType A; A.setIdentity(rows, cols); @@ -213,38 +204,39 @@ void testMatrixType(const MatrixType& m) VERIFY(equalsIdentity(MatrixType::Identity(rows, cols))); - A = MatrixType::Constant(rows,cols,s1); - Index i = internal::random(0,rows-1); - Index j = internal::random(0,cols-1); - VERIFY_IS_APPROX( MatrixType::Constant(rows,cols,s1)(i,j), s1 ); - VERIFY_IS_APPROX( MatrixType::Constant(rows,cols,s1).coeff(i,j), s1 ); - VERIFY_IS_APPROX( A(i,j), s1 ); + A = MatrixType::Constant(rows, cols, s1); + Index i = internal::random(0, rows - 1); + Index j = internal::random(0, cols - 1); + VERIFY_IS_APPROX(MatrixType::Constant(rows, cols, s1)(i, j), s1); + VERIFY_IS_APPROX(MatrixType::Constant(rows, cols, s1).coeff(i, j), s1); + VERIFY_IS_APPROX(A(i, j), s1); } void test_nullary() { - CALL_SUBTEST_1( testMatrixType(Matrix2d()) ); - CALL_SUBTEST_2( testMatrixType(MatrixXcf(internal::random(1,300),internal::random(1,300))) ); - CALL_SUBTEST_3( testMatrixType(MatrixXf(internal::random(1,300),internal::random(1,300))) ); - - for(int i = 0; i < g_repeat*10; i++) { - CALL_SUBTEST_4( testVectorType(VectorXd(internal::random(1,30000))) ); - CALL_SUBTEST_5( testVectorType(Vector4d()) ); // regression test for bug 232 - CALL_SUBTEST_6( testVectorType(Vector3d()) ); - CALL_SUBTEST_7( testVectorType(VectorXf(internal::random(1,30000))) ); - CALL_SUBTEST_8( testVectorType(Vector3f()) ); - CALL_SUBTEST_8( testVectorType(Vector4f()) ); - CALL_SUBTEST_8( testVectorType(Matrix()) ); - CALL_SUBTEST_8( testVectorType(Matrix()) ); - - CALL_SUBTEST_9( testVectorType(VectorXi(internal::random(1,10))) ); - CALL_SUBTEST_9( testVectorType(VectorXi(internal::random(9,300))) ); - CALL_SUBTEST_9( testVectorType(Matrix()) ); + CALL_SUBTEST_1(testMatrixType(Matrix2d())); + CALL_SUBTEST_2(testMatrixType(MatrixXcf(internal::random(1, 300), internal::random(1, 300)))); + CALL_SUBTEST_3(testMatrixType(MatrixXf(internal::random(1, 300), internal::random(1, 300)))); + + for (int i = 0; i < g_repeat * 10; i++) { + CALL_SUBTEST_4(testVectorType(VectorXd(internal::random(1, 30000)))); + CALL_SUBTEST_5(testVectorType(Vector4d()));// regression test for bug 232 + CALL_SUBTEST_6(testVectorType(Vector3d())); + CALL_SUBTEST_7(testVectorType(VectorXf(internal::random(1, 30000)))); + CALL_SUBTEST_8(testVectorType(Vector3f())); + CALL_SUBTEST_8(testVectorType(Vector4f())); + CALL_SUBTEST_8(testVectorType(Matrix())); + CALL_SUBTEST_8(testVectorType(Matrix())); + + CALL_SUBTEST_9(testVectorType(VectorXi(internal::random(1, 10)))); + CALL_SUBTEST_9(testVectorType(VectorXi(internal::random(9, 300)))); + CALL_SUBTEST_9(testVectorType(Matrix())); } #ifdef EIGEN_TEST_PART_6 // Assignment of a RowVectorXd to a MatrixXd (regression test for bug #79). - VERIFY( (MatrixXd(RowVectorXd::LinSpaced(3, 0, 1)) - RowVector3d(0, 0.5, 1)).norm() < std::numeric_limits::epsilon() ); + VERIFY((MatrixXd(RowVectorXd::LinSpaced(3, 0, 1)) - RowVector3d(0, 0.5, 1)).norm() + < std::numeric_limits::epsilon()); #endif #ifdef EIGEN_TEST_PART_9 @@ -252,53 +244,52 @@ void test_nullary() { int n = 60000; ArrayXi a1(n), a2(n); - a1.setLinSpaced(n, 0, n-1); - for(int i=0; i >::value )); - VERIFY(( !internal::has_unary_operator >::value )); - VERIFY(( !internal::has_binary_operator >::value )); - VERIFY(( internal::functor_has_linear_access >::ret )); + VERIFY((internal::has_nullary_operator>::value)); + VERIFY((!internal::has_unary_operator>::value)); + VERIFY((!internal::has_binary_operator>::value)); + VERIFY((internal::functor_has_linear_access>::ret)); - VERIFY(( !internal::has_nullary_operator >::value )); - VERIFY(( !internal::has_unary_operator >::value )); - VERIFY(( internal::has_binary_operator >::value )); - VERIFY(( !internal::functor_has_linear_access >::ret )); + VERIFY((!internal::has_nullary_operator>::value)); + VERIFY((!internal::has_unary_operator>::value)); + VERIFY((internal::has_binary_operator>::value)); + VERIFY((!internal::functor_has_linear_access>::ret)); - VERIFY(( !internal::has_nullary_operator >::value )); - VERIFY(( internal::has_unary_operator >::value )); - VERIFY(( !internal::has_binary_operator >::value )); - VERIFY(( internal::functor_has_linear_access >::ret )); + VERIFY((!internal::has_nullary_operator>::value)); + VERIFY((internal::has_unary_operator>::value)); + VERIFY((!internal::has_binary_operator>::value)); + VERIFY((internal::functor_has_linear_access>::ret)); // Regression unit test for a weird MSVC bug. // Search "nullary_wrapper_workaround_msvc" in CoreEvaluators.h for the details. // See also traits::match. { - MatrixXf A = MatrixXf::Random(3,3); - Ref R = 2.0*A; - VERIFY_IS_APPROX(R, A+A); + MatrixXf A = MatrixXf::Random(3, 3); + Ref R = 2.0 * A; + VERIFY_IS_APPROX(R, A + A); - Ref R1 = MatrixXf::Random(3,3)+A; + Ref R1 = MatrixXf::Random(3, 3) + A; VectorXi V = VectorXi::Random(3); - Ref R2 = VectorXi::LinSpaced(3,1,3)+V; - VERIFY_IS_APPROX(R2, V+Vector3i(1,2,3)); - - VERIFY(( internal::has_nullary_operator >::value )); - VERIFY(( !internal::has_unary_operator >::value )); - VERIFY(( !internal::has_binary_operator >::value )); - VERIFY(( internal::functor_has_linear_access >::ret )); - - VERIFY(( !internal::has_nullary_operator >::value )); - VERIFY(( internal::has_unary_operator >::value )); - VERIFY(( !internal::has_binary_operator >::value )); - VERIFY(( internal::functor_has_linear_access >::ret )); + Ref R2 = VectorXi::LinSpaced(3, 1, 3) + V; + VERIFY_IS_APPROX(R2, V + Vector3i(1, 2, 3)); + + VERIFY((internal::has_nullary_operator>::value)); + VERIFY((!internal::has_unary_operator>::value)); + VERIFY((!internal::has_binary_operator>::value)); + VERIFY((internal::functor_has_linear_access>::ret)); + + VERIFY((!internal::has_nullary_operator>::value)); + VERIFY((internal::has_unary_operator>::value)); + VERIFY((!internal::has_binary_operator>::value)); + VERIFY((internal::functor_has_linear_access>::ret)); } #endif } diff --git a/filmulator-gui/core/nlmeans/eigen/test/numext.cpp b/filmulator-gui/core/nlmeans/eigen/test/numext.cpp index 3de33e2f..cddf9fd6 100644 --- a/filmulator-gui/core/nlmeans/eigen/test/numext.cpp +++ b/filmulator-gui/core/nlmeans/eigen/test/numext.cpp @@ -9,45 +9,42 @@ #include "main.h" -template -void check_abs() { +template void check_abs() +{ typedef typename NumTraits::Real Real; - if(NumTraits::IsSigned) - VERIFY_IS_EQUAL(numext::abs(-T(1)), T(1)); + if (NumTraits::IsSigned) VERIFY_IS_EQUAL(numext::abs(-T(1)), T(1)); VERIFY_IS_EQUAL(numext::abs(T(0)), T(0)); VERIFY_IS_EQUAL(numext::abs(T(1)), T(1)); - for(int k=0; k(); - if(!internal::is_same::value) - x = x/Real(2); - if(NumTraits::IsSigned) - { + if (!internal::is_same::value) x = x / Real(2); + if (NumTraits::IsSigned) { VERIFY_IS_EQUAL(numext::abs(x), numext::abs(-x)); - VERIFY( numext::abs(-x) >= Real(0)); + VERIFY(numext::abs(-x) >= Real(0)); } - VERIFY( numext::abs(x) >= Real(0)); - VERIFY_IS_APPROX( numext::abs2(x), numext::abs2(numext::abs(x)) ); + VERIFY(numext::abs(x) >= Real(0)); + VERIFY_IS_APPROX(numext::abs2(x), numext::abs2(numext::abs(x))); } } -void test_numext() { - CALL_SUBTEST( check_abs() ); - CALL_SUBTEST( check_abs() ); - CALL_SUBTEST( check_abs() ); - CALL_SUBTEST( check_abs() ); - CALL_SUBTEST( check_abs() ); - CALL_SUBTEST( check_abs() ); - CALL_SUBTEST( check_abs() ); - CALL_SUBTEST( check_abs() ); - CALL_SUBTEST( check_abs() ); - CALL_SUBTEST( check_abs() ); - CALL_SUBTEST( check_abs() ); - CALL_SUBTEST( check_abs() ); - CALL_SUBTEST( check_abs() ); +void test_numext() +{ + CALL_SUBTEST(check_abs()); + CALL_SUBTEST(check_abs()); + CALL_SUBTEST(check_abs()); + CALL_SUBTEST(check_abs()); + CALL_SUBTEST(check_abs()); + CALL_SUBTEST(check_abs()); + CALL_SUBTEST(check_abs()); + CALL_SUBTEST(check_abs()); + CALL_SUBTEST(check_abs()); + CALL_SUBTEST(check_abs()); + CALL_SUBTEST(check_abs()); + CALL_SUBTEST(check_abs()); + CALL_SUBTEST(check_abs()); - CALL_SUBTEST( check_abs >() ); - CALL_SUBTEST( check_abs >() ); + CALL_SUBTEST(check_abs>()); + CALL_SUBTEST(check_abs>()); } diff --git a/filmulator-gui/core/nlmeans/eigen/test/packetmath.cpp b/filmulator-gui/core/nlmeans/eigen/test/packetmath.cpp index 7821a173..2df78615 100644 --- a/filmulator-gui/core/nlmeans/eigen/test/packetmath.cpp +++ b/filmulator-gui/core/nlmeans/eigen/test/packetmath.cpp @@ -11,8 +11,8 @@ #include "main.h" #include "unsupported/Eigen/SpecialFunctions" -#if defined __GNUC__ && __GNUC__>=6 - #pragma GCC diagnostic ignored "-Wignored-attributes" +#if defined __GNUC__ && __GNUC__ >= 6 +#pragma GCC diagnostic ignored "-Wignored-attributes" #endif // using namespace Eigen; @@ -24,90 +24,83 @@ const bool g_vectorize_sse = false; namespace Eigen { namespace internal { -template T negate(const T& x) { return -x; } -} -} + template T negate(const T &x) { return -x; } +}// namespace internal +}// namespace Eigen // NOTE: we disbale inlining for this function to workaround a GCC issue when using -O3 and the i387 FPU. -template EIGEN_DONT_INLINE -bool isApproxAbs(const Scalar& a, const Scalar& b, const typename NumTraits::Real& refvalue) +template +EIGEN_DONT_INLINE bool isApproxAbs(const Scalar &a, const Scalar &b, const typename NumTraits::Real &refvalue) { - return internal::isMuchSmallerThan(a-b, refvalue); + return internal::isMuchSmallerThan(a - b, refvalue); } -template bool areApproxAbs(const Scalar* a, const Scalar* b, int size, const typename NumTraits::Real& refvalue) +template +bool areApproxAbs(const Scalar *a, const Scalar *b, int size, const typename NumTraits::Real &refvalue) { - for (int i=0; i >(a,size) << "]" << " != vec: [" << Map >(b,size) << "]\n"; + for (int i = 0; i < size; ++i) { + if (!isApproxAbs(a[i], b[i], refvalue)) { + std::cout << "ref: [" << Map>(a, size) << "]" << " != vec: [" + << Map>(b, size) << "]\n"; return false; } } return true; } -template bool areApprox(const Scalar* a, const Scalar* b, int size) +template bool areApprox(const Scalar *a, const Scalar *b, int size) { - for (int i=0; i >(a,size) << "]" << " != vec: [" << Map >(b,size) << "]\n"; + for (int i = 0; i < size; ++i) { + if (a[i] != b[i] && !internal::isApprox(a[i], b[i])) { + std::cout << "ref: [" << Map>(a, size) << "]" << " != vec: [" + << Map>(b, size) << "]\n"; return false; } } return true; } -#define CHECK_CWISE1(REFOP, POP) { \ - for (int i=0; i(data1))); \ - VERIFY(areApprox(ref, data2, PacketSize) && #POP); \ -} +#define CHECK_CWISE1(REFOP, POP) \ + { \ + for (int i = 0; i < PacketSize; ++i) ref[i] = REFOP(data1[i]); \ + internal::pstore(data2, POP(internal::pload(data1))); \ + VERIFY(areApprox(ref, data2, PacketSize) && #POP); \ + } -template -struct packet_helper +template struct packet_helper { - template - inline Packet load(const T* from) const { return internal::pload(from); } + template inline Packet load(const T *from) const { return internal::pload(from); } - template - inline void store(T* to, const Packet& x) const { internal::pstore(to,x); } + template inline void store(T *to, const Packet &x) const { internal::pstore(to, x); } }; -template -struct packet_helper +template struct packet_helper { - template - inline T load(const T* from) const { return *from; } + template inline T load(const T *from) const { return *from; } - template - inline void store(T* to, const T& x) const { *to = x; } + template inline void store(T *to, const T &x) const { *to = x; } }; -#define CHECK_CWISE1_IF(COND, REFOP, POP) if(COND) { \ - packet_helper h; \ - for (int i=0; i h; \ + for (int i = 0; i < PacketSize; ++i) ref[i] = REFOP(data1[i]); \ + h.store(data2, POP(h.load(data1))); \ + VERIFY(areApprox(ref, data2, PacketSize) && #POP); \ + } -#define CHECK_CWISE2_IF(COND, REFOP, POP) if(COND) { \ - packet_helper h; \ - for (int i=0; i h; \ + for (int i = 0; i < PacketSize; ++i) ref[i] = REFOP(data1[i], data1[i + PacketSize]); \ + h.store(data2, POP(h.load(data1), h.load(data1 + PacketSize))); \ + VERIFY(areApprox(ref, data2, PacketSize) && #POP); \ + } -#define REF_ADD(a,b) ((a)+(b)) -#define REF_SUB(a,b) ((a)-(b)) -#define REF_MUL(a,b) ((a)*(b)) -#define REF_DIV(a,b) ((a)/(b)) +#define REF_ADD(a, b) ((a) + (b)) +#define REF_SUB(a, b) ((a) - (b)) +#define REF_MUL(a, b) ((a) * (b)) +#define REF_DIV(a, b) ((a) / (b)) template void packetmath() { @@ -118,58 +111,69 @@ template void packetmath() typedef typename NumTraits::Real RealScalar; const int max_size = PacketSize > 4 ? PacketSize : 4; - const int size = PacketSize*max_size; + const int size = PacketSize * max_size; EIGEN_ALIGN_MAX Scalar data1[size]; EIGEN_ALIGN_MAX Scalar data2[size]; - EIGEN_ALIGN_MAX Packet packets[PacketSize*2]; + EIGEN_ALIGN_MAX Packet packets[PacketSize * 2]; EIGEN_ALIGN_MAX Scalar ref[size]; RealScalar refvalue = 0; - for (int i=0; i()/RealScalar(PacketSize); - data2[i] = internal::random()/RealScalar(PacketSize); - refvalue = (std::max)(refvalue,abs(data1[i])); + for (int i = 0; i < size; ++i) { + data1[i] = internal::random() / RealScalar(PacketSize); + data2[i] = internal::random() / RealScalar(PacketSize); + refvalue = (std::max)(refvalue, abs(data1[i])); } internal::pstore(data2, internal::pload(data1)); VERIFY(areApprox(data1, data2, PacketSize) && "aligned load/store"); - for (int offset=0; offset(data1+offset)); - VERIFY(areApprox(data1+offset, data2, PacketSize) && "internal::ploadu"); + for (int offset = 0; offset < PacketSize; ++offset) { + internal::pstore(data2, internal::ploadu(data1 + offset)); + VERIFY(areApprox(data1 + offset, data2, PacketSize) && "internal::ploadu"); } - for (int offset=0; offset(data1)); - VERIFY(areApprox(data1, data2+offset, PacketSize) && "internal::pstoreu"); + for (int offset = 0; offset < PacketSize; ++offset) { + internal::pstoreu(data2 + offset, internal::pload(data1)); + VERIFY(areApprox(data1, data2 + offset, PacketSize) && "internal::pstoreu"); } - for (int offset=0; offset(data1); - packets[1] = internal::pload(data1+PacketSize); - if (offset==0) internal::palign<0>(packets[0], packets[1]); - else if (offset==1) internal::palign<1>(packets[0], packets[1]); - else if (offset==2) internal::palign<2>(packets[0], packets[1]); - else if (offset==3) internal::palign<3>(packets[0], packets[1]); - else if (offset==4) internal::palign<4>(packets[0], packets[1]); - else if (offset==5) internal::palign<5>(packets[0], packets[1]); - else if (offset==6) internal::palign<6>(packets[0], packets[1]); - else if (offset==7) internal::palign<7>(packets[0], packets[1]); - else if (offset==8) internal::palign<8>(packets[0], packets[1]); - else if (offset==9) internal::palign<9>(packets[0], packets[1]); - else if (offset==10) internal::palign<10>(packets[0], packets[1]); - else if (offset==11) internal::palign<11>(packets[0], packets[1]); - else if (offset==12) internal::palign<12>(packets[0], packets[1]); - else if (offset==13) internal::palign<13>(packets[0], packets[1]); - else if (offset==14) internal::palign<14>(packets[0], packets[1]); - else if (offset==15) internal::palign<15>(packets[0], packets[1]); + packets[1] = internal::pload(data1 + PacketSize); + if (offset == 0) + internal::palign<0>(packets[0], packets[1]); + else if (offset == 1) + internal::palign<1>(packets[0], packets[1]); + else if (offset == 2) + internal::palign<2>(packets[0], packets[1]); + else if (offset == 3) + internal::palign<3>(packets[0], packets[1]); + else if (offset == 4) + internal::palign<4>(packets[0], packets[1]); + else if (offset == 5) + internal::palign<5>(packets[0], packets[1]); + else if (offset == 6) + internal::palign<6>(packets[0], packets[1]); + else if (offset == 7) + internal::palign<7>(packets[0], packets[1]); + else if (offset == 8) + internal::palign<8>(packets[0], packets[1]); + else if (offset == 9) + internal::palign<9>(packets[0], packets[1]); + else if (offset == 10) + internal::palign<10>(packets[0], packets[1]); + else if (offset == 11) + internal::palign<11>(packets[0], packets[1]); + else if (offset == 12) + internal::palign<12>(packets[0], packets[1]); + else if (offset == 13) + internal::palign<13>(packets[0], packets[1]); + else if (offset == 14) + internal::palign<14>(packets[0], packets[1]); + else if (offset == 15) + internal::palign<15>(packets[0], packets[1]); internal::pstore(data2, packets[0]); - for (int i=0; i void packetmath() VERIFY((!PacketTraits::Vectorizable) || PacketTraits::HasSub); VERIFY((!PacketTraits::Vectorizable) || PacketTraits::HasMul); VERIFY((!PacketTraits::Vectorizable) || PacketTraits::HasNegate); - VERIFY((internal::is_same::value) || (!PacketTraits::Vectorizable) || PacketTraits::HasDiv); + VERIFY((internal::is_same::value) || (!PacketTraits::Vectorizable) || PacketTraits::HasDiv); - CHECK_CWISE2_IF(PacketTraits::HasAdd, REF_ADD, internal::padd); - CHECK_CWISE2_IF(PacketTraits::HasSub, REF_SUB, internal::psub); - CHECK_CWISE2_IF(PacketTraits::HasMul, REF_MUL, internal::pmul); + CHECK_CWISE2_IF(PacketTraits::HasAdd, REF_ADD, internal::padd); + CHECK_CWISE2_IF(PacketTraits::HasSub, REF_SUB, internal::psub); + CHECK_CWISE2_IF(PacketTraits::HasMul, REF_MUL, internal::pmul); CHECK_CWISE2_IF(PacketTraits::HasDiv, REF_DIV, internal::pdiv); CHECK_CWISE1(internal::negate, internal::pnegate); CHECK_CWISE1(numext::conj, internal::pconj); - for(int offset=0;offset<3;++offset) - { - for (int i=0; i(data1[offset])); VERIFY(areApprox(ref, data2, PacketSize) && "internal::pset1"); } { - for (int i=0; i(data1, A0, A1, A2, A3); - internal::pstore(data2+0*PacketSize, A0); - internal::pstore(data2+1*PacketSize, A1); - internal::pstore(data2+2*PacketSize, A2); - internal::pstore(data2+3*PacketSize, A3); - VERIFY(areApprox(ref, data2, 4*PacketSize) && "internal::pbroadcast4"); + internal::pstore(data2 + 0 * PacketSize, A0); + internal::pstore(data2 + 1 * PacketSize, A1); + internal::pstore(data2 + 2 * PacketSize, A2); + internal::pstore(data2 + 3 * PacketSize, A3); + VERIFY(areApprox(ref, data2, 4 * PacketSize) && "internal::pbroadcast4"); } { - for (int i=0; i(data1, A0, A1); - internal::pstore(data2+0*PacketSize, A0); - internal::pstore(data2+1*PacketSize, A1); - VERIFY(areApprox(ref, data2, 2*PacketSize) && "internal::pbroadcast2"); + internal::pstore(data2 + 0 * PacketSize, A0); + internal::pstore(data2 + 1 * PacketSize, A1); + VERIFY(areApprox(ref, data2, 2 * PacketSize) && "internal::pbroadcast2"); } VERIFY(internal::isApprox(data1[0], internal::pfirst(internal::pload(data1))) && "internal::pfirst"); - if(PacketSize>1) - { - for(int offset=0;offset<4;++offset) - { - for(int i=0;i(data1+offset)); + if (PacketSize > 1) { + for (int offset = 0; offset < 4; ++offset) { + for (int i = 0; i < PacketSize / 2; ++i) ref[2 * i + 0] = ref[2 * i + 1] = data1[offset + i]; + internal::pstore(data2, internal::ploaddup(data1 + offset)); VERIFY(areApprox(ref, data2, PacketSize) && "ploaddup"); } } - if(PacketSize>2) - { - for(int offset=0;offset<4;++offset) - { - for(int i=0;i(data1+offset)); + if (PacketSize > 2) { + for (int offset = 0; offset < 4; ++offset) { + for (int i = 0; i < PacketSize / 4; ++i) + ref[4 * i + 0] = ref[4 * i + 1] = ref[4 * i + 2] = ref[4 * i + 3] = data1[offset + i]; + internal::pstore(data2, internal::ploadquad(data1 + offset)); VERIFY(areApprox(ref, data2, PacketSize) && "ploadquad"); } } ref[0] = 0; - for (int i=0; i(data1)), refvalue) && "internal::predux"); { - for (int i=0; i<4; ++i) - ref[i] = 0; - for (int i=0; i(data1))); - VERIFY(areApprox(ref, data2, PacketSize>4?PacketSize/2:PacketSize) && "internal::predux_downto4"); + VERIFY(areApprox(ref, data2, PacketSize > 4 ? PacketSize / 2 : PacketSize) && "internal::predux_downto4"); } ref[0] = 1; - for (int i=0; i(data1))) && "internal::predux_mul"); - for (int j=0; j(data1+j*PacketSize); + for (int i = 0; i < PacketSize; ++i) ref[j] += data1[i + j * PacketSize]; + packets[j] = internal::pload(data1 + j * PacketSize); } internal::pstore(data2, internal::preduxp(packets)); VERIFY(areApproxAbs(ref, data2, PacketSize, refvalue) && "internal::preduxp"); - for (int i=0; i(data1))); VERIFY(areApprox(ref, data2, PacketSize) && "internal::preverse"); internal::PacketBlock kernel; - for (int i=0; i(data1+i*PacketSize); - } + for (int i = 0; i < PacketSize; ++i) { kernel.packet[i] = internal::pload(data1 + i * PacketSize); } ptranspose(kernel); - for (int i=0; i void packetmath() Packet thenPacket = internal::pload(data1); Packet elsePacket = internal::pload(data2); EIGEN_ALIGN_MAX internal::Selector selector; - for (int i = 0; i < PacketSize; ++i) { - selector.select[i] = i; - } + for (int i = 0; i < PacketSize; ++i) { selector.select[i] = i; } Packet blend = internal::pblend(selector, thenPacket, elsePacket); EIGEN_ALIGN_MAX Scalar result[size]; @@ -306,21 +290,19 @@ template void packetmath() if (PacketTraits::HasBlend || g_vectorize_sse) { // pinsertfirst - for (int i=0; i(); ref[0] = s; - internal::pstore(data2, internal::pinsertfirst(internal::pload(data1),s)); + internal::pstore(data2, internal::pinsertfirst(internal::pload(data1), s)); VERIFY(areApprox(ref, data2, PacketSize) && "internal::pinsertfirst"); } if (PacketTraits::HasBlend || g_vectorize_sse) { // pinsertlast - for (int i=0; i(); - ref[PacketSize-1] = s; - internal::pstore(data2, internal::pinsertlast(internal::pload(data1),s)); + ref[PacketSize - 1] = s; + internal::pstore(data2, internal::pinsertlast(internal::pload(data1), s)); VERIFY(areApprox(ref, data2, PacketSize) && "internal::pinsertlast"); } } @@ -332,15 +314,14 @@ template void packetmath_real() typedef typename PacketTraits::type Packet; const int PacketSize = PacketTraits::size; - const int size = PacketSize*4; - EIGEN_ALIGN_MAX Scalar data1[PacketTraits::size*4]; - EIGEN_ALIGN_MAX Scalar data2[PacketTraits::size*4]; - EIGEN_ALIGN_MAX Scalar ref[PacketTraits::size*4]; + const int size = PacketSize * 4; + EIGEN_ALIGN_MAX Scalar data1[PacketTraits::size * 4]; + EIGEN_ALIGN_MAX Scalar data2[PacketTraits::size * 4]; + EIGEN_ALIGN_MAX Scalar ref[PacketTraits::size * 4]; - for (int i=0; i(-1,1) * std::pow(Scalar(10), internal::random(-3,3)); - data2[i] = internal::random(-1,1) * std::pow(Scalar(10), internal::random(-3,3)); + for (int i = 0; i < size; ++i) { + data1[i] = internal::random(-1, 1) * std::pow(Scalar(10), internal::random(-3, 3)); + data2[i] = internal::random(-1, 1) * std::pow(Scalar(10), internal::random(-3, 3)); } CHECK_CWISE1_IF(PacketTraits::HasSin, std::sin, internal::psin); CHECK_CWISE1_IF(PacketTraits::HasCos, std::cos, internal::pcos); @@ -350,31 +331,27 @@ template void packetmath_real() CHECK_CWISE1_IF(PacketTraits::HasCeil, numext::ceil, internal::pceil); CHECK_CWISE1_IF(PacketTraits::HasFloor, numext::floor, internal::pfloor); - for (int i=0; i(-1,1); - data2[i] = internal::random(-1,1); + for (int i = 0; i < size; ++i) { + data1[i] = internal::random(-1, 1); + data2[i] = internal::random(-1, 1); } CHECK_CWISE1_IF(PacketTraits::HasASin, std::asin, internal::pasin); CHECK_CWISE1_IF(PacketTraits::HasACos, std::acos, internal::pacos); - for (int i=0; i(-87,88); - data2[i] = internal::random(-87,88); + for (int i = 0; i < size; ++i) { + data1[i] = internal::random(-87, 88); + data2[i] = internal::random(-87, 88); } CHECK_CWISE1_IF(PacketTraits::HasExp, std::exp, internal::pexp); - for (int i=0; i(-1,1) * std::pow(Scalar(10), internal::random(-6,6)); - data2[i] = internal::random(-1,1) * std::pow(Scalar(10), internal::random(-6,6)); + for (int i = 0; i < size; ++i) { + data1[i] = internal::random(-1, 1) * std::pow(Scalar(10), internal::random(-6, 6)); + data2[i] = internal::random(-1, 1) * std::pow(Scalar(10), internal::random(-6, 6)); } CHECK_CWISE1_IF(PacketTraits::HasTanh, std::tanh, internal::ptanh); - if(PacketTraits::HasExp && PacketTraits::size>=2) - { + if (PacketTraits::HasExp && PacketTraits::size >= 2) { data1[0] = std::numeric_limits::quiet_NaN(); data1[1] = std::numeric_limits::epsilon(); - packet_helper h; + packet_helper h; h.store(data2, internal::pexp(h.load(data1))); VERIFY((numext::isnan)(data2[0])); VERIFY_IS_EQUAL(std::exp(std::numeric_limits::epsilon()), data2[1]); @@ -401,7 +378,7 @@ template void packetmath_real() if (PacketTraits::HasTanh) { // NOTE this test migh fail with GCC prior to 6.3, see MathFunctionsImpl.h for details. data1[0] = std::numeric_limits::quiet_NaN(); - packet_helper::HasTanh,Packet> h; + packet_helper::HasTanh, Packet> h; h.store(data2, internal::ptanh(h.load(data1))); VERIFY((numext::isnan)(data2[0])); } @@ -409,32 +386,30 @@ template void packetmath_real() #if EIGEN_HAS_C99_MATH { data1[0] = std::numeric_limits::quiet_NaN(); - packet_helper::HasLGamma,Packet> h; + packet_helper::HasLGamma, Packet> h; h.store(data2, internal::plgamma(h.load(data1))); VERIFY((numext::isnan)(data2[0])); } { data1[0] = std::numeric_limits::quiet_NaN(); - packet_helper::HasErf,Packet> h; + packet_helper::HasErf, Packet> h; h.store(data2, internal::perf(h.load(data1))); VERIFY((numext::isnan)(data2[0])); } { data1[0] = std::numeric_limits::quiet_NaN(); - packet_helper::HasErfc,Packet> h; + packet_helper::HasErfc, Packet> h; h.store(data2, internal::perfc(h.load(data1))); VERIFY((numext::isnan)(data2[0])); } -#endif // EIGEN_HAS_C99_MATH +#endif// EIGEN_HAS_C99_MATH - for (int i=0; i(0,1) * std::pow(Scalar(10), internal::random(-6,6)); - data2[i] = internal::random(0,1) * std::pow(Scalar(10), internal::random(-6,6)); + for (int i = 0; i < size; ++i) { + data1[i] = internal::random(0, 1) * std::pow(Scalar(10), internal::random(-6, 6)); + data2[i] = internal::random(0, 1) * std::pow(Scalar(10), internal::random(-6, 6)); } - if(internal::random(0,1)<0.1f) - data1[internal::random(0, PacketSize)] = 0; + if (internal::random(0, 1) < 0.1f) data1[internal::random(0, PacketSize)] = 0; CHECK_CWISE1_IF(PacketTraits::HasSqrt, std::sqrt, internal::psqrt); CHECK_CWISE1_IF(PacketTraits::HasLog, std::log, internal::plog); #if EIGEN_HAS_C99_MATH && (__cplusplus > 199711L) @@ -444,11 +419,10 @@ template void packetmath_real() CHECK_CWISE1_IF(internal::packet_traits::HasErfc, std::erfc, internal::perfc); #endif - if(PacketTraits::HasLog && PacketTraits::size>=2) - { + if (PacketTraits::HasLog && PacketTraits::size >= 2) { data1[0] = std::numeric_limits::quiet_NaN(); data1[1] = std::numeric_limits::epsilon(); - packet_helper h; + packet_helper h; h.store(data2, internal::plog(h.load(data1))); VERIFY((numext::isnan)(data2[0])); VERIFY_IS_EQUAL(std::log(std::numeric_limits::epsilon()), data2[1]); @@ -487,15 +461,14 @@ template void packetmath_notcomplex() typedef typename PacketTraits::type Packet; const int PacketSize = PacketTraits::size; - EIGEN_ALIGN_MAX Scalar data1[PacketTraits::size*4]; - EIGEN_ALIGN_MAX Scalar data2[PacketTraits::size*4]; - EIGEN_ALIGN_MAX Scalar ref[PacketTraits::size*4]; + EIGEN_ALIGN_MAX Scalar data1[PacketTraits::size * 4]; + EIGEN_ALIGN_MAX Scalar data2[PacketTraits::size * 4]; + EIGEN_ALIGN_MAX Scalar ref[PacketTraits::size * 4]; - Array::Map(data1, PacketTraits::size*4).setRandom(); + Array::Map(data1, PacketTraits::size * 4).setRandom(); ref[0] = data1[0]; - for (int i=0; i(data1))) && "internal::predux_min"); VERIFY((!PacketTraits::Vectorizable) || PacketTraits::HasMin); @@ -506,17 +479,16 @@ template void packetmath_notcomplex() CHECK_CWISE1(abs, internal::pabs); ref[0] = data1[0]; - for (int i=0; i(data1))) && "internal::predux_max"); - for (int i=0; i(data1[0])); VERIFY(areApprox(ref, data2, PacketSize) && "internal::plset"); } -template void test_conj_helper(Scalar* data1, Scalar* data2, Scalar* ref, Scalar* pval) +template +void test_conj_helper(Scalar *data1, Scalar *data2, Scalar *ref, Scalar *pval) { typedef internal::packet_traits PacketTraits; typedef typename PacketTraits::type Packet; @@ -524,24 +496,23 @@ template void test_conj_helper(Scalar internal::conj_if cj0; internal::conj_if cj1; - internal::conj_helper cj; - internal::conj_helper pcj; + internal::conj_helper cj; + internal::conj_helper pcj; - for(int i=0;i(data1),internal::pload(data2))); + internal::pstore(pval, pcj.pmul(internal::pload(data1), internal::pload(data2))); VERIFY(areApprox(ref, pval, PacketSize) && "conj_helper pmul"); - for(int i=0;i(data1),internal::pload(data2),internal::pload(pval))); + internal::pstore( + pval, pcj.pmadd(internal::pload(data1), internal::pload(data2), internal::pload(pval))); VERIFY(areApprox(ref, pval, PacketSize) && "conj_helper pmadd"); } @@ -551,27 +522,25 @@ template void packetmath_complex() typedef typename PacketTraits::type Packet; const int PacketSize = PacketTraits::size; - const int size = PacketSize*4; - EIGEN_ALIGN_MAX Scalar data1[PacketSize*4]; - EIGEN_ALIGN_MAX Scalar data2[PacketSize*4]; - EIGEN_ALIGN_MAX Scalar ref[PacketSize*4]; - EIGEN_ALIGN_MAX Scalar pval[PacketSize*4]; + const int size = PacketSize * 4; + EIGEN_ALIGN_MAX Scalar data1[PacketSize * 4]; + EIGEN_ALIGN_MAX Scalar data2[PacketSize * 4]; + EIGEN_ALIGN_MAX Scalar ref[PacketSize * 4]; + EIGEN_ALIGN_MAX Scalar pval[PacketSize * 4]; - for (int i=0; i() * Scalar(1e2); data2[i] = internal::random() * Scalar(1e2); } - test_conj_helper (data1,data2,ref,pval); - test_conj_helper (data1,data2,ref,pval); - test_conj_helper (data1,data2,ref,pval); - test_conj_helper (data1,data2,ref,pval); + test_conj_helper(data1, data2, ref, pval); + test_conj_helper(data1, data2, ref, pval); + test_conj_helper(data1, data2, ref, pval); + test_conj_helper(data1, data2, ref, pval); { - for(int i=0;i(data1))); + for (int i = 0; i < PacketSize; ++i) ref[i] = Scalar(std::imag(data1[i]), std::real(data1[i])); + internal::pstore(pval, internal::pcplxflip(internal::pload(data1))); VERIFY(areApprox(ref, pval, PacketSize) && "pcplxflip"); } } @@ -584,58 +553,52 @@ template void packetmath_scatter_gather() const int PacketSize = PacketTraits::size; EIGEN_ALIGN_MAX Scalar data1[PacketSize]; RealScalar refvalue = 0; - for (int i=0; i()/RealScalar(PacketSize); - } + for (int i = 0; i < PacketSize; ++i) { data1[i] = internal::random() / RealScalar(PacketSize); } - int stride = internal::random(1,20); + int stride = internal::random(1, 20); - EIGEN_ALIGN_MAX Scalar buffer[PacketSize*20]; - memset(buffer, 0, 20*PacketSize*sizeof(Scalar)); + EIGEN_ALIGN_MAX Scalar buffer[PacketSize * 20]; + memset(buffer, 0, 20 * PacketSize * sizeof(Scalar)); Packet packet = internal::pload(data1); internal::pscatter(buffer, packet, stride); - for (int i = 0; i < PacketSize*20; ++i) { - if ((i%stride) == 0 && i()/RealScalar(PacketSize); - } + for (int i = 0; i < PacketSize * 7; ++i) { buffer[i] = internal::random() / RealScalar(PacketSize); } packet = internal::pgather(buffer, 7); internal::pstore(data1, packet); - for (int i = 0; i < PacketSize; ++i) { - VERIFY(isApproxAbs(data1[i], buffer[i*7], refvalue) && "pgather"); - } + for (int i = 0; i < PacketSize; ++i) { VERIFY(isApproxAbs(data1[i], buffer[i * 7], refvalue) && "pgather"); } } void test_packetmath() { - for(int i = 0; i < g_repeat; i++) { - CALL_SUBTEST_1( packetmath() ); - CALL_SUBTEST_2( packetmath() ); - CALL_SUBTEST_3( packetmath() ); - CALL_SUBTEST_4( packetmath >() ); - CALL_SUBTEST_5( packetmath >() ); - - CALL_SUBTEST_1( packetmath_notcomplex() ); - CALL_SUBTEST_2( packetmath_notcomplex() ); - CALL_SUBTEST_3( packetmath_notcomplex() ); - - CALL_SUBTEST_1( packetmath_real() ); - CALL_SUBTEST_2( packetmath_real() ); - - CALL_SUBTEST_4( packetmath_complex >() ); - CALL_SUBTEST_5( packetmath_complex >() ); - - CALL_SUBTEST_1( packetmath_scatter_gather() ); - CALL_SUBTEST_2( packetmath_scatter_gather() ); - CALL_SUBTEST_3( packetmath_scatter_gather() ); - CALL_SUBTEST_4( packetmath_scatter_gather >() ); - CALL_SUBTEST_5( packetmath_scatter_gather >() ); + for (int i = 0; i < g_repeat; i++) { + CALL_SUBTEST_1(packetmath()); + CALL_SUBTEST_2(packetmath()); + CALL_SUBTEST_3(packetmath()); + CALL_SUBTEST_4(packetmath>()); + CALL_SUBTEST_5(packetmath>()); + + CALL_SUBTEST_1(packetmath_notcomplex()); + CALL_SUBTEST_2(packetmath_notcomplex()); + CALL_SUBTEST_3(packetmath_notcomplex()); + + CALL_SUBTEST_1(packetmath_real()); + CALL_SUBTEST_2(packetmath_real()); + + CALL_SUBTEST_4(packetmath_complex>()); + CALL_SUBTEST_5(packetmath_complex>()); + + CALL_SUBTEST_1(packetmath_scatter_gather()); + CALL_SUBTEST_2(packetmath_scatter_gather()); + CALL_SUBTEST_3(packetmath_scatter_gather()); + CALL_SUBTEST_4(packetmath_scatter_gather>()); + CALL_SUBTEST_5(packetmath_scatter_gather>()); } } diff --git a/filmulator-gui/core/nlmeans/eigen/test/pardiso_support.cpp b/filmulator-gui/core/nlmeans/eigen/test/pardiso_support.cpp index 67efad6d..0c571e0e 100644 --- a/filmulator-gui/core/nlmeans/eigen/test/pardiso_support.cpp +++ b/filmulator-gui/core/nlmeans/eigen/test/pardiso_support.cpp @@ -1,4 +1,4 @@ -/* +/* Intel Copyright (C) .... */ @@ -7,11 +7,11 @@ template void test_pardiso_T() { - PardisoLLT < SparseMatrix, Lower> pardiso_llt_lower; - PardisoLLT < SparseMatrix, Upper> pardiso_llt_upper; - PardisoLDLT < SparseMatrix, Lower> pardiso_ldlt_lower; - PardisoLDLT < SparseMatrix, Upper> pardiso_ldlt_upper; - PardisoLU < SparseMatrix > pardiso_lu; + PardisoLLT, Lower> pardiso_llt_lower; + PardisoLLT, Upper> pardiso_llt_upper; + PardisoLDLT, Lower> pardiso_ldlt_lower; + PardisoLDLT, Upper> pardiso_ldlt_upper; + PardisoLU> pardiso_lu; check_sparse_spd_solving(pardiso_llt_lower); check_sparse_spd_solving(pardiso_llt_upper); @@ -24,6 +24,6 @@ void test_pardiso_support() { CALL_SUBTEST_1(test_pardiso_T()); CALL_SUBTEST_2(test_pardiso_T()); - CALL_SUBTEST_3(test_pardiso_T< std::complex >()); - CALL_SUBTEST_4(test_pardiso_T< std::complex >()); + CALL_SUBTEST_3(test_pardiso_T>()); + CALL_SUBTEST_4(test_pardiso_T>()); } diff --git a/filmulator-gui/core/nlmeans/eigen/test/pastix_support.cpp b/filmulator-gui/core/nlmeans/eigen/test/pastix_support.cpp index b62f8573..4e47c8e6 100644 --- a/filmulator-gui/core/nlmeans/eigen/test/pastix_support.cpp +++ b/filmulator-gui/core/nlmeans/eigen/test/pastix_support.cpp @@ -16,11 +16,11 @@ template void test_pastix_T() { - PastixLLT< SparseMatrix, Eigen::Lower > pastix_llt_lower; - PastixLDLT< SparseMatrix, Eigen::Lower > pastix_ldlt_lower; - PastixLLT< SparseMatrix, Eigen::Upper > pastix_llt_upper; - PastixLDLT< SparseMatrix, Eigen::Upper > pastix_ldlt_upper; - PastixLU< SparseMatrix > pastix_lu; + PastixLLT, Eigen::Lower> pastix_llt_lower; + PastixLDLT, Eigen::Lower> pastix_ldlt_lower; + PastixLLT, Eigen::Upper> pastix_llt_upper; + PastixLDLT, Eigen::Upper> pastix_ldlt_upper; + PastixLU> pastix_lu; check_sparse_spd_solving(pastix_llt_lower); check_sparse_spd_solving(pastix_ldlt_lower); @@ -37,11 +37,11 @@ template void test_pastix_T() pastix_lu.dparm(); } -// There is no support for selfadjoint matrices with PaStiX. +// There is no support for selfadjoint matrices with PaStiX. // Complex symmetric matrices should pass though template void test_pastix_T_LU() { - PastixLU< SparseMatrix > pastix_lu; + PastixLU> pastix_lu; check_sparse_square_solving(pastix_lu); } @@ -49,6 +49,6 @@ void test_pastix_support() { CALL_SUBTEST_1(test_pastix_T()); CALL_SUBTEST_2(test_pastix_T()); - CALL_SUBTEST_3( (test_pastix_T_LU >()) ); - CALL_SUBTEST_4(test_pastix_T_LU >()); -} + CALL_SUBTEST_3((test_pastix_T_LU>())); + CALL_SUBTEST_4(test_pastix_T_LU>()); +} diff --git a/filmulator-gui/core/nlmeans/eigen/test/permutationmatrices.cpp b/filmulator-gui/core/nlmeans/eigen/test/permutationmatrices.cpp index e885f0e0..472544f8 100644 --- a/filmulator-gui/core/nlmeans/eigen/test/permutationmatrices.cpp +++ b/filmulator-gui/core/nlmeans/eigen/test/permutationmatrices.cpp @@ -8,15 +8,14 @@ // with this file, You can obtain one at http://mozilla.org/MPL/2.0/. #define TEST_ENABLE_TEMPORARY_TRACKING - + #include "main.h" using namespace std; -template void permutationmatrices(const MatrixType& m) +template void permutationmatrices(const MatrixType &m) { typedef typename MatrixType::Scalar Scalar; - enum { Rows = MatrixType::RowsAtCompileTime, Cols = MatrixType::ColsAtCompileTime, - Options = MatrixType::Options }; + enum { Rows = MatrixType::RowsAtCompileTime, Cols = MatrixType::ColsAtCompileTime, Options = MatrixType::Options }; typedef PermutationMatrix LeftPermutationType; typedef Transpositions LeftTranspositionsType; typedef Matrix LeftPermutationVectorType; @@ -29,7 +28,7 @@ template void permutationmatrices(const MatrixType& m) Index rows = m.rows(); Index cols = m.cols(); - MatrixType m_original = MatrixType::Random(rows,cols); + MatrixType m_original = MatrixType::Random(rows, cols); LeftPermutationVectorType lv; randomPermutationVector(lv, rows); LeftPermutationType lp(lv); @@ -38,74 +37,79 @@ template void permutationmatrices(const MatrixType& m) RightPermutationType rp(rv); LeftTranspositionsType lt(lv); RightTranspositionsType rt(rv); - MatrixType m_permuted = MatrixType::Random(rows,cols); - - VERIFY_EVALUATION_COUNT(m_permuted = lp * m_original * rp, 1); // 1 temp for sub expression "lp * m_original" + MatrixType m_permuted = MatrixType::Random(rows, cols); + + VERIFY_EVALUATION_COUNT(m_permuted = lp * m_original * rp, 1);// 1 temp for sub expression "lp * m_original" - for (int i=0; i lm(lp); - Matrix rm(rp); + Matrix lm(lp); + Matrix rm(rp); + + VERIFY_IS_APPROX(m_permuted, lm * m_original * rm); - VERIFY_IS_APPROX(m_permuted, lm*m_original*rm); - m_permuted = m_original; VERIFY_EVALUATION_COUNT(m_permuted = lp * m_permuted * rp, 1); - VERIFY_IS_APPROX(m_permuted, lm*m_original*rm); - - VERIFY_IS_APPROX(lp.inverse()*m_permuted*rp.inverse(), m_original); - VERIFY_IS_APPROX(lv.asPermutation().inverse()*m_permuted*rv.asPermutation().inverse(), m_original); - VERIFY_IS_APPROX(MapLeftPerm(lv.data(),lv.size()).inverse()*m_permuted*MapRightPerm(rv.data(),rv.size()).inverse(), m_original); - - VERIFY((lp*lp.inverse()).toDenseMatrix().isIdentity()); - VERIFY((lv.asPermutation()*lv.asPermutation().inverse()).toDenseMatrix().isIdentity()); - VERIFY((MapLeftPerm(lv.data(),lv.size())*MapLeftPerm(lv.data(),lv.size()).inverse()).toDenseMatrix().isIdentity()); + VERIFY_IS_APPROX(m_permuted, lm * m_original * rm); + + VERIFY_IS_APPROX(lp.inverse() * m_permuted * rp.inverse(), m_original); + VERIFY_IS_APPROX(lv.asPermutation().inverse() * m_permuted * rv.asPermutation().inverse(), m_original); + VERIFY_IS_APPROX( + MapLeftPerm(lv.data(), lv.size()).inverse() * m_permuted * MapRightPerm(rv.data(), rv.size()).inverse(), + m_original); + + VERIFY((lp * lp.inverse()).toDenseMatrix().isIdentity()); + VERIFY((lv.asPermutation() * lv.asPermutation().inverse()).toDenseMatrix().isIdentity()); + VERIFY( + (MapLeftPerm(lv.data(), lv.size()) * MapLeftPerm(lv.data(), lv.size()).inverse()).toDenseMatrix().isIdentity()); LeftPermutationVectorType lv2; randomPermutationVector(lv2, rows); LeftPermutationType lp2(lv2); - Matrix lm2(lp2); - VERIFY_IS_APPROX((lp*lp2).toDenseMatrix().template cast(), lm*lm2); - VERIFY_IS_APPROX((lv.asPermutation()*lv2.asPermutation()).toDenseMatrix().template cast(), lm*lm2); - VERIFY_IS_APPROX((MapLeftPerm(lv.data(),lv.size())*MapLeftPerm(lv2.data(),lv2.size())).toDenseMatrix().template cast(), lm*lm2); + Matrix lm2(lp2); + VERIFY_IS_APPROX((lp * lp2).toDenseMatrix().template cast(), lm * lm2); + VERIFY_IS_APPROX((lv.asPermutation() * lv2.asPermutation()).toDenseMatrix().template cast(), lm * lm2); + VERIFY_IS_APPROX( + (MapLeftPerm(lv.data(), lv.size()) * MapLeftPerm(lv2.data(), lv2.size())).toDenseMatrix().template cast(), + lm * lm2); LeftPermutationType identityp; identityp.setIdentity(rows); - VERIFY_IS_APPROX(m_original, identityp*m_original); - + VERIFY_IS_APPROX(m_original, identityp * m_original); + // check inplace permutations m_permuted = m_original; - VERIFY_EVALUATION_COUNT(m_permuted.noalias()= lp.inverse() * m_permuted, 1); // 1 temp to allocate the mask - VERIFY_IS_APPROX(m_permuted, lp.inverse()*m_original); - + VERIFY_EVALUATION_COUNT(m_permuted.noalias() = lp.inverse() * m_permuted, 1);// 1 temp to allocate the mask + VERIFY_IS_APPROX(m_permuted, lp.inverse() * m_original); + m_permuted = m_original; - VERIFY_EVALUATION_COUNT(m_permuted.noalias() = m_permuted * rp.inverse(), 1); // 1 temp to allocate the mask - VERIFY_IS_APPROX(m_permuted, m_original*rp.inverse()); - + VERIFY_EVALUATION_COUNT(m_permuted.noalias() = m_permuted * rp.inverse(), 1);// 1 temp to allocate the mask + VERIFY_IS_APPROX(m_permuted, m_original * rp.inverse()); + m_permuted = m_original; - VERIFY_EVALUATION_COUNT(m_permuted.noalias() = lp * m_permuted, 1); // 1 temp to allocate the mask - VERIFY_IS_APPROX(m_permuted, lp*m_original); - + VERIFY_EVALUATION_COUNT(m_permuted.noalias() = lp * m_permuted, 1);// 1 temp to allocate the mask + VERIFY_IS_APPROX(m_permuted, lp * m_original); + m_permuted = m_original; - VERIFY_EVALUATION_COUNT(m_permuted.noalias() = m_permuted * rp, 1); // 1 temp to allocate the mask - VERIFY_IS_APPROX(m_permuted, m_original*rp); + VERIFY_EVALUATION_COUNT(m_permuted.noalias() = m_permuted * rp, 1);// 1 temp to allocate the mask + VERIFY_IS_APPROX(m_permuted, m_original * rp); - if(rows>1 && cols>1) - { + if (rows > 1 && cols > 1) { lp2 = lp; - Index i = internal::random(0, rows-1); + Index i = internal::random(0, rows - 1); Index j; - do j = internal::random(0, rows-1); while(j==i); + do j = internal::random(0, rows - 1); + while (j == i); lp2.applyTranspositionOnTheLeft(i, j); lm = lp; lm.row(i).swap(lm.row(j)); VERIFY_IS_APPROX(lm, lp2.toDenseMatrix().template cast()); RightPermutationType rp2 = rp; - i = internal::random(0, cols-1); - do j = internal::random(0, cols-1); while(j==i); + i = internal::random(0, cols - 1); + do j = internal::random(0, cols - 1); + while (j == i); rp2.applyTranspositionOnTheRight(i, j); rm = rp; rm.col(i).swap(rm.col(j)); @@ -123,45 +127,44 @@ template void permutationmatrices(const MatrixType& m) lp = lt; rp = rt; VERIFY_EVALUATION_COUNT(m_permuted = lt * m_permuted * rt, 1); - VERIFY_IS_APPROX(m_permuted, lp*m_original*rp.transpose()); - - VERIFY_IS_APPROX(lt.inverse()*m_permuted*rt.inverse(), m_original); + VERIFY_IS_APPROX(m_permuted, lp * m_original * rp.transpose()); + + VERIFY_IS_APPROX(lt.inverse() * m_permuted * rt.inverse(), m_original); } -template -void bug890() +template void bug890() { typedef Matrix MatrixType; typedef Matrix VectorType; - typedef Stride S; + typedef Stride S; typedef Map MapType; typedef PermutationMatrix Perm; - + VectorType v1(2), v2(2), op(4), rhs(2); - v1 << 666,667; - op << 1,0,0,1; - rhs << 42,42; - + v1 << 666, 667; + op << 1, 0, 0, 1; + rhs << 42, 42; + Perm P(2); P.indices() << 1, 0; - MapType(v1.data(),2,1,S(1,1)) = P * MapType(rhs.data(),2,1,S(1,1)); + MapType(v1.data(), 2, 1, S(1, 1)) = P * MapType(rhs.data(), 2, 1, S(1, 1)); VERIFY_IS_APPROX(v1, (P * rhs).eval()); - - MapType(v1.data(),2,1,S(1,1)) = P.inverse() * MapType(rhs.data(),2,1,S(1,1)); + + MapType(v1.data(), 2, 1, S(1, 1)) = P.inverse() * MapType(rhs.data(), 2, 1, S(1, 1)); VERIFY_IS_APPROX(v1, (P.inverse() * rhs).eval()); } void test_permutationmatrices() { - for(int i = 0; i < g_repeat; i++) { - CALL_SUBTEST_1( permutationmatrices(Matrix()) ); - CALL_SUBTEST_2( permutationmatrices(Matrix3f()) ); - CALL_SUBTEST_3( permutationmatrices(Matrix()) ); - CALL_SUBTEST_4( permutationmatrices(Matrix4d()) ); - CALL_SUBTEST_5( permutationmatrices(Matrix()) ); - CALL_SUBTEST_6( permutationmatrices(Matrix(20, 30)) ); - CALL_SUBTEST_7( permutationmatrices(MatrixXcf(15, 10)) ); + for (int i = 0; i < g_repeat; i++) { + CALL_SUBTEST_1(permutationmatrices(Matrix())); + CALL_SUBTEST_2(permutationmatrices(Matrix3f())); + CALL_SUBTEST_3(permutationmatrices(Matrix())); + CALL_SUBTEST_4(permutationmatrices(Matrix4d())); + CALL_SUBTEST_5(permutationmatrices(Matrix())); + CALL_SUBTEST_6(permutationmatrices(Matrix(20, 30))); + CALL_SUBTEST_7(permutationmatrices(MatrixXcf(15, 10))); } - CALL_SUBTEST_5( bug890() ); + CALL_SUBTEST_5(bug890()); } diff --git a/filmulator-gui/core/nlmeans/eigen/test/prec_inverse_4x4.cpp b/filmulator-gui/core/nlmeans/eigen/test/prec_inverse_4x4.cpp index eb6ad18c..8ef86ea4 100644 --- a/filmulator-gui/core/nlmeans/eigen/test/prec_inverse_4x4.cpp +++ b/filmulator-gui/core/nlmeans/eigen/test/prec_inverse_4x4.cpp @@ -14,15 +14,14 @@ template void inverse_permutation_4x4() { typedef typename MatrixType::Scalar Scalar; - Vector4i indices(0,1,2,3); - for(int i = 0; i < 24; ++i) - { + Vector4i indices(0, 1, 2, 3); + for (int i = 0; i < 24; ++i) { MatrixType m = PermutationMatrix<4>(indices); MatrixType inv = m.inverse(); - double error = double( (m*inv-MatrixType::Identity()).norm() / NumTraits::epsilon() ); + double error = double((m * inv - MatrixType::Identity()).norm() / NumTraits::epsilon()); EIGEN_DEBUG_VAR(error) VERIFY(error == 0.0); - std::next_permutation(indices.data(),indices.data()+4); + std::next_permutation(indices.data(), indices.data() + 4); } } @@ -32,16 +31,15 @@ template void inverse_general_4x4(int repeat) typedef typename MatrixType::Scalar Scalar; typedef typename MatrixType::RealScalar RealScalar; double error_sum = 0., error_max = 0.; - for(int i = 0; i < repeat; ++i) - { + for (int i = 0; i < repeat; ++i) { MatrixType m; RealScalar absdet; do { m = MatrixType::Random(); absdet = abs(m.determinant()); - } while(absdet < NumTraits::epsilon()); + } while (absdet < NumTraits::epsilon()); MatrixType inv = m.inverse(); - double error = double( (m*inv-MatrixType::Identity()).norm() * absdet / NumTraits::epsilon() ); + double error = double((m * inv - MatrixType::Identity()).norm() * absdet / NumTraits::epsilon()); error_sum += error; error_max = (std::max)(error_max, error); } @@ -49,34 +47,34 @@ template void inverse_general_4x4(int repeat) double error_avg = error_sum / repeat; EIGEN_DEBUG_VAR(error_avg); EIGEN_DEBUG_VAR(error_max); - // FIXME that 1.25 used to be a 1.0 until the NumTraits changes on 28 April 2010, what's going wrong?? - // FIXME that 1.25 used to be 1.2 until we tested gcc 4.1 on 30 June 2010 and got 1.21. + // FIXME that 1.25 used to be a 1.0 until the NumTraits changes on 28 April 2010, what's going wrong?? + // FIXME that 1.25 used to be 1.2 until we tested gcc 4.1 on 30 June 2010 and got 1.21. VERIFY(error_avg < (NumTraits::IsComplex ? 8.0 : 1.25)); VERIFY(error_max < (NumTraits::IsComplex ? 64.0 : 20.0)); { - int s = 5;//internal::random(4,10); - int i = 0;//internal::random(0,s-4); - int j = 0;//internal::random(0,s-4); - Matrix mat(s,s); + int s = 5;// internal::random(4,10); + int i = 0;// internal::random(0,s-4); + int j = 0;// internal::random(0,s-4); + Matrix mat(s, s); mat.setRandom(); - MatrixType submat = mat.template block<4,4>(i,j); - MatrixType mat_inv = mat.template block<4,4>(i,j).inverse(); + MatrixType submat = mat.template block<4, 4>(i, j); + MatrixType mat_inv = mat.template block<4, 4>(i, j).inverse(); VERIFY_IS_APPROX(mat_inv, submat.inverse()); - mat.template block<4,4>(i,j) = submat.inverse(); - VERIFY_IS_APPROX(mat_inv, (mat.template block<4,4>(i,j))); + mat.template block<4, 4>(i, j) = submat.inverse(); + VERIFY_IS_APPROX(mat_inv, (mat.template block<4, 4>(i, j))); } } void test_prec_inverse_4x4() { CALL_SUBTEST_1((inverse_permutation_4x4())); - CALL_SUBTEST_1(( inverse_general_4x4(200000 * g_repeat) )); - CALL_SUBTEST_1(( inverse_general_4x4 >(200000 * g_repeat) )); + CALL_SUBTEST_1((inverse_general_4x4(200000 * g_repeat))); + CALL_SUBTEST_1((inverse_general_4x4>(200000 * g_repeat))); - CALL_SUBTEST_2((inverse_permutation_4x4 >())); - CALL_SUBTEST_2(( inverse_general_4x4 >(200000 * g_repeat) )); - CALL_SUBTEST_2(( inverse_general_4x4 >(200000 * g_repeat) )); + CALL_SUBTEST_2((inverse_permutation_4x4>())); + CALL_SUBTEST_2((inverse_general_4x4>(200000 * g_repeat))); + CALL_SUBTEST_2((inverse_general_4x4>(200000 * g_repeat))); CALL_SUBTEST_3((inverse_permutation_4x4())); CALL_SUBTEST_3((inverse_general_4x4(50000 * g_repeat))); diff --git a/filmulator-gui/core/nlmeans/eigen/test/product.h b/filmulator-gui/core/nlmeans/eigen/test/product.h index 3b651127..bfac10a4 100644 --- a/filmulator-gui/core/nlmeans/eigen/test/product.h +++ b/filmulator-gui/core/nlmeans/eigen/test/product.h @@ -11,13 +11,15 @@ #include template -bool areNotApprox(const MatrixBase& m1, const MatrixBase& m2, typename Derived1::RealScalar epsilon = NumTraits::dummy_precision()) +bool areNotApprox(const MatrixBase &m1, + const MatrixBase &m2, + typename Derived1::RealScalar epsilon = NumTraits::dummy_precision()) { - return !((m1-m2).cwiseAbs2().maxCoeff() < epsilon * epsilon - * (std::max)(m1.cwiseAbs2().maxCoeff(), m2.cwiseAbs2().maxCoeff())); + return !((m1 - m2).cwiseAbs2().maxCoeff() + < epsilon * epsilon * (std::max)(m1.cwiseAbs2().maxCoeff(), m2.cwiseAbs2().maxCoeff())); } -template void product(const MatrixType& m) +template void product(const MatrixType &m) { /* this test covers the following files: Identity.h Product.h @@ -27,73 +29,66 @@ template void product(const MatrixType& m) typedef Matrix ColVectorType; typedef Matrix RowSquareMatrixType; typedef Matrix ColSquareMatrixType; - typedef Matrix OtherMajorMatrixType; + typedef Matrix + OtherMajorMatrixType; Index rows = m.rows(); Index cols = m.cols(); // this test relies a lot on Random.h, and there's not much more that we can do // to test it, hence I consider that we will have tested Random.h - MatrixType m1 = MatrixType::Random(rows, cols), - m2 = MatrixType::Random(rows, cols), - m3(rows, cols); - RowSquareMatrixType - identity = RowSquareMatrixType::Identity(rows, rows), - square = RowSquareMatrixType::Random(rows, rows), - res = RowSquareMatrixType::Random(rows, rows); - ColSquareMatrixType - square2 = ColSquareMatrixType::Random(cols, cols), - res2 = ColSquareMatrixType::Random(cols, cols); + MatrixType m1 = MatrixType::Random(rows, cols), m2 = MatrixType::Random(rows, cols), m3(rows, cols); + RowSquareMatrixType identity = RowSquareMatrixType::Identity(rows, rows), + square = RowSquareMatrixType::Random(rows, rows), res = RowSquareMatrixType::Random(rows, rows); + ColSquareMatrixType square2 = ColSquareMatrixType::Random(cols, cols), res2 = ColSquareMatrixType::Random(cols, cols); RowVectorType v1 = RowVectorType::Random(rows); ColVectorType vc2 = ColVectorType::Random(cols), vcres(cols); OtherMajorMatrixType tm1 = m1; Scalar s1 = internal::random(); - Index r = internal::random(0, rows-1), - c = internal::random(0, cols-1), - c2 = internal::random(0, cols-1); + Index r = internal::random(0, rows - 1), c = internal::random(0, cols - 1), + c2 = internal::random(0, cols - 1); // begin testing Product.h: only associativity for now // (we use Transpose.h but this doesn't count as a test for it) - VERIFY_IS_APPROX((m1*m1.transpose())*m2, m1*(m1.transpose()*m2)); + VERIFY_IS_APPROX((m1 * m1.transpose()) * m2, m1 * (m1.transpose() * m2)); m3 = m1; m3 *= m1.transpose() * m2; - VERIFY_IS_APPROX(m3, m1 * (m1.transpose()*m2)); - VERIFY_IS_APPROX(m3, m1 * (m1.transpose()*m2)); + VERIFY_IS_APPROX(m3, m1 * (m1.transpose() * m2)); + VERIFY_IS_APPROX(m3, m1 * (m1.transpose() * m2)); // continue testing Product.h: distributivity - VERIFY_IS_APPROX(square*(m1 + m2), square*m1+square*m2); - VERIFY_IS_APPROX(square*(m1 - m2), square*m1-square*m2); + VERIFY_IS_APPROX(square * (m1 + m2), square * m1 + square * m2); + VERIFY_IS_APPROX(square * (m1 - m2), square * m1 - square * m2); // continue testing Product.h: compatibility with ScalarMultiple.h - VERIFY_IS_APPROX(s1*(square*m1), (s1*square)*m1); - VERIFY_IS_APPROX(s1*(square*m1), square*(m1*s1)); + VERIFY_IS_APPROX(s1 * (square * m1), (s1 * square) * m1); + VERIFY_IS_APPROX(s1 * (square * m1), square * (m1 * s1)); // test Product.h together with Identity.h - VERIFY_IS_APPROX(v1, identity*v1); - VERIFY_IS_APPROX(v1.transpose(), v1.transpose() * identity); + VERIFY_IS_APPROX(v1, identity * v1); + VERIFY_IS_APPROX(v1.transpose(), v1.transpose() * identity); // again, test operator() to check const-qualification - VERIFY_IS_APPROX(MatrixType::Identity(rows, cols)(r,c), static_cast(r==c)); + VERIFY_IS_APPROX(MatrixType::Identity(rows, cols)(r, c), static_cast(r == c)); - if (rows!=cols) - VERIFY_RAISES_ASSERT(m3 = m1*m1); + if (rows != cols) VERIFY_RAISES_ASSERT(m3 = m1 * m1); // test the previous tests were not screwed up because operator* returns 0 // (we use the more accurate default epsilon) - if (!NumTraits::IsInteger && (std::min)(rows,cols)>1) - { - VERIFY(areNotApprox(m1.transpose()*m2,m2.transpose()*m1)); + if (!NumTraits::IsInteger && (std::min)(rows, cols) > 1) { + VERIFY(areNotApprox(m1.transpose() * m2, m2.transpose() * m1)); } // test optimized operator+= path res = square; res.noalias() += m1 * m2.transpose(); VERIFY_IS_APPROX(res, square + m1 * m2.transpose()); - if (!NumTraits::IsInteger && (std::min)(rows,cols)>1) - { - VERIFY(areNotApprox(res,square + m2 * m1.transpose())); + if (!NumTraits::IsInteger && (std::min)(rows, cols) > 1) { + VERIFY(areNotApprox(res, square + m2 * m1.transpose())); } vcres = vc2; vcres.noalias() += m1.transpose() * v1; @@ -103,9 +98,8 @@ template void product(const MatrixType& m) res = square; res.noalias() -= m1 * m2.transpose(); VERIFY_IS_APPROX(res, square - (m1 * m2.transpose())); - if (!NumTraits::IsInteger && (std::min)(rows,cols)>1) - { - VERIFY(areNotApprox(res,square - m2 * m1.transpose())); + if (!NumTraits::IsInteger && (std::min)(rows, cols) > 1) { + VERIFY(areNotApprox(res, square - m2 * m1.transpose())); } vcres = vc2; vcres.noalias() -= m1.transpose() * v1; @@ -115,7 +109,7 @@ template void product(const MatrixType& m) res.noalias() = square + m1 * m2.transpose(); VERIFY_IS_APPROX(res, square + m1 * m2.transpose()); res.noalias() += square + m1 * m2.transpose(); - VERIFY_IS_APPROX(res, 2*(square + m1 * m2.transpose())); + VERIFY_IS_APPROX(res, 2 * (square + m1 * m2.transpose())); res.noalias() -= square + m1 * m2.transpose(); VERIFY_IS_APPROX(res, square + m1 * m2.transpose()); @@ -123,7 +117,7 @@ template void product(const MatrixType& m) res.noalias() = square - m1 * m2.transpose(); VERIFY_IS_APPROX(res, square - m1 * m2.transpose()); res.noalias() += square - m1 * m2.transpose(); - VERIFY_IS_APPROX(res, 2*(square - m1 * m2.transpose())); + VERIFY_IS_APPROX(res, 2 * (square - m1 * m2.transpose())); res.noalias() -= square - m1 * m2.transpose(); VERIFY_IS_APPROX(res, square - m1 * m2.transpose()); @@ -133,20 +127,17 @@ template void product(const MatrixType& m) VERIFY_IS_APPROX(v1.transpose() * tm1, v1.transpose() * m1); // test submatrix and matrix/vector product - for (int i=0; i::IsInteger && (std::min)(rows,cols)>1) - { - VERIFY(areNotApprox(res2,square2 + m2.transpose() * m1)); + if (!NumTraits::IsInteger && (std::min)(rows, cols) > 1) { + VERIFY(areNotApprox(res2, square2 + m2.transpose() * m1)); } VERIFY_IS_APPROX(res.col(r).noalias() = square.adjoint() * square.col(r), (square.adjoint() * square.col(r)).eval()); @@ -157,38 +148,48 @@ template void product(const MatrixType& m) RowSquareMatrixType ref(square); ColSquareMatrixType ref2(square2); ref = res = square; - VERIFY_IS_APPROX(res.block(0,0,1,rows).noalias() = m1.col(0).transpose() * square.transpose(), (ref.row(0) = m1.col(0).transpose() * square.transpose())); - VERIFY_IS_APPROX(res.block(0,0,1,rows).noalias() = m1.block(0,0,rows,1).transpose() * square.transpose(), (ref.row(0) = m1.col(0).transpose() * square.transpose())); - VERIFY_IS_APPROX(res.block(0,0,1,rows).noalias() = m1.col(0).transpose() * square, (ref.row(0) = m1.col(0).transpose() * square)); - VERIFY_IS_APPROX(res.block(0,0,1,rows).noalias() = m1.block(0,0,rows,1).transpose() * square, (ref.row(0) = m1.col(0).transpose() * square)); + VERIFY_IS_APPROX(res.block(0, 0, 1, rows).noalias() = m1.col(0).transpose() * square.transpose(), + (ref.row(0) = m1.col(0).transpose() * square.transpose())); + VERIFY_IS_APPROX(res.block(0, 0, 1, rows).noalias() = m1.block(0, 0, rows, 1).transpose() * square.transpose(), + (ref.row(0) = m1.col(0).transpose() * square.transpose())); + VERIFY_IS_APPROX(res.block(0, 0, 1, rows).noalias() = m1.col(0).transpose() * square, + (ref.row(0) = m1.col(0).transpose() * square)); + VERIFY_IS_APPROX(res.block(0, 0, 1, rows).noalias() = m1.block(0, 0, rows, 1).transpose() * square, + (ref.row(0) = m1.col(0).transpose() * square)); ref2 = res2 = square2; - VERIFY_IS_APPROX(res2.block(0,0,1,cols).noalias() = m1.row(0) * square2.transpose(), (ref2.row(0) = m1.row(0) * square2.transpose())); - VERIFY_IS_APPROX(res2.block(0,0,1,cols).noalias() = m1.block(0,0,1,cols) * square2.transpose(), (ref2.row(0) = m1.row(0) * square2.transpose())); - VERIFY_IS_APPROX(res2.block(0,0,1,cols).noalias() = m1.row(0) * square2, (ref2.row(0) = m1.row(0) * square2)); - VERIFY_IS_APPROX(res2.block(0,0,1,cols).noalias() = m1.block(0,0,1,cols) * square2, (ref2.row(0) = m1.row(0) * square2)); + VERIFY_IS_APPROX(res2.block(0, 0, 1, cols).noalias() = m1.row(0) * square2.transpose(), + (ref2.row(0) = m1.row(0) * square2.transpose())); + VERIFY_IS_APPROX(res2.block(0, 0, 1, cols).noalias() = m1.block(0, 0, 1, cols) * square2.transpose(), + (ref2.row(0) = m1.row(0) * square2.transpose())); + VERIFY_IS_APPROX(res2.block(0, 0, 1, cols).noalias() = m1.row(0) * square2, (ref2.row(0) = m1.row(0) * square2)); + VERIFY_IS_APPROX( + res2.block(0, 0, 1, cols).noalias() = m1.block(0, 0, 1, cols) * square2, (ref2.row(0) = m1.row(0) * square2)); } // vector.block() (see bug 1283) { RowVectorType w1(rows); - VERIFY_IS_APPROX(square * v1.block(0,0,rows,1), square * v1); - VERIFY_IS_APPROX(w1.noalias() = square * v1.block(0,0,rows,1), square * v1); - VERIFY_IS_APPROX(w1.block(0,0,rows,1).noalias() = square * v1.block(0,0,rows,1), square * v1); - - Matrix w2(cols); - VERIFY_IS_APPROX(vc2.block(0,0,cols,1).transpose() * square2, vc2.transpose() * square2); - VERIFY_IS_APPROX(w2.noalias() = vc2.block(0,0,cols,1).transpose() * square2, vc2.transpose() * square2); - VERIFY_IS_APPROX(w2.block(0,0,1,cols).noalias() = vc2.block(0,0,cols,1).transpose() * square2, vc2.transpose() * square2); - - vc2 = square2.block(0,0,1,cols).transpose(); - VERIFY_IS_APPROX(square2.block(0,0,1,cols) * square2, vc2.transpose() * square2); - VERIFY_IS_APPROX(w2.noalias() = square2.block(0,0,1,cols) * square2, vc2.transpose() * square2); - VERIFY_IS_APPROX(w2.block(0,0,1,cols).noalias() = square2.block(0,0,1,cols) * square2, vc2.transpose() * square2); - - vc2 = square2.block(0,0,cols,1); - VERIFY_IS_APPROX(square2.block(0,0,cols,1).transpose() * square2, vc2.transpose() * square2); - VERIFY_IS_APPROX(w2.noalias() = square2.block(0,0,cols,1).transpose() * square2, vc2.transpose() * square2); - VERIFY_IS_APPROX(w2.block(0,0,1,cols).noalias() = square2.block(0,0,cols,1).transpose() * square2, vc2.transpose() * square2); + VERIFY_IS_APPROX(square * v1.block(0, 0, rows, 1), square * v1); + VERIFY_IS_APPROX(w1.noalias() = square * v1.block(0, 0, rows, 1), square * v1); + VERIFY_IS_APPROX(w1.block(0, 0, rows, 1).noalias() = square * v1.block(0, 0, rows, 1), square * v1); + + Matrix w2(cols); + VERIFY_IS_APPROX(vc2.block(0, 0, cols, 1).transpose() * square2, vc2.transpose() * square2); + VERIFY_IS_APPROX(w2.noalias() = vc2.block(0, 0, cols, 1).transpose() * square2, vc2.transpose() * square2); + VERIFY_IS_APPROX( + w2.block(0, 0, 1, cols).noalias() = vc2.block(0, 0, cols, 1).transpose() * square2, vc2.transpose() * square2); + + vc2 = square2.block(0, 0, 1, cols).transpose(); + VERIFY_IS_APPROX(square2.block(0, 0, 1, cols) * square2, vc2.transpose() * square2); + VERIFY_IS_APPROX(w2.noalias() = square2.block(0, 0, 1, cols) * square2, vc2.transpose() * square2); + VERIFY_IS_APPROX( + w2.block(0, 0, 1, cols).noalias() = square2.block(0, 0, 1, cols) * square2, vc2.transpose() * square2); + + vc2 = square2.block(0, 0, cols, 1); + VERIFY_IS_APPROX(square2.block(0, 0, cols, 1).transpose() * square2, vc2.transpose() * square2); + VERIFY_IS_APPROX(w2.noalias() = square2.block(0, 0, cols, 1).transpose() * square2, vc2.transpose() * square2); + VERIFY_IS_APPROX(w2.block(0, 0, 1, cols).noalias() = square2.block(0, 0, cols, 1).transpose() * square2, + vc2.transpose() * square2); } // inner product @@ -199,33 +200,36 @@ template void product(const MatrixType& m) // outer product { - VERIFY_IS_APPROX(m1.col(c) * m1.row(r), m1.block(0,c,rows,1) * m1.block(r,0,1,cols)); - VERIFY_IS_APPROX(m1.row(r).transpose() * m1.col(c).transpose(), m1.block(r,0,1,cols).transpose() * m1.block(0,c,rows,1).transpose()); - VERIFY_IS_APPROX(m1.block(0,c,rows,1) * m1.row(r), m1.block(0,c,rows,1) * m1.block(r,0,1,cols)); - VERIFY_IS_APPROX(m1.col(c) * m1.block(r,0,1,cols), m1.block(0,c,rows,1) * m1.block(r,0,1,cols)); - VERIFY_IS_APPROX(m1.leftCols(1) * m1.row(r), m1.block(0,0,rows,1) * m1.block(r,0,1,cols)); - VERIFY_IS_APPROX(m1.col(c) * m1.topRows(1), m1.block(0,c,rows,1) * m1.block(0,0,1,cols)); + VERIFY_IS_APPROX(m1.col(c) * m1.row(r), m1.block(0, c, rows, 1) * m1.block(r, 0, 1, cols)); + VERIFY_IS_APPROX(m1.row(r).transpose() * m1.col(c).transpose(), + m1.block(r, 0, 1, cols).transpose() * m1.block(0, c, rows, 1).transpose()); + VERIFY_IS_APPROX(m1.block(0, c, rows, 1) * m1.row(r), m1.block(0, c, rows, 1) * m1.block(r, 0, 1, cols)); + VERIFY_IS_APPROX(m1.col(c) * m1.block(r, 0, 1, cols), m1.block(0, c, rows, 1) * m1.block(r, 0, 1, cols)); + VERIFY_IS_APPROX(m1.leftCols(1) * m1.row(r), m1.block(0, 0, rows, 1) * m1.block(r, 0, 1, cols)); + VERIFY_IS_APPROX(m1.col(c) * m1.topRows(1), m1.block(0, c, rows, 1) * m1.block(0, 0, 1, cols)); } // Aliasing { - ColVectorType x(cols); x.setRandom(); + ColVectorType x(cols); + x.setRandom(); ColVectorType z(x); - ColVectorType y(cols); y.setZero(); - ColSquareMatrixType A(cols,cols); A.setRandom(); + ColVectorType y(cols); + y.setZero(); + ColSquareMatrixType A(cols, cols); + A.setRandom(); // CwiseBinaryOp - VERIFY_IS_APPROX(x = y + A*x, A*z); + VERIFY_IS_APPROX(x = y + A * x, A * z); x = z; // CwiseUnaryOp - VERIFY_IS_APPROX(x = Scalar(1.)*(A*x), A*z); + VERIFY_IS_APPROX(x = Scalar(1.) * (A * x), A * z); } // regression for blas_trais { - VERIFY_IS_APPROX(square * (square*square).transpose(), square * square.transpose() * square.transpose()); - VERIFY_IS_APPROX(square * (-(square*square)), -square * square * square); - VERIFY_IS_APPROX(square * (s1*(square*square)), s1 * square * square * square); - VERIFY_IS_APPROX(square * (square*square).conjugate(), square * square.conjugate() * square.conjugate()); + VERIFY_IS_APPROX(square * (square * square).transpose(), square * square.transpose() * square.transpose()); + VERIFY_IS_APPROX(square * (-(square * square)), -square * square * square); + VERIFY_IS_APPROX(square * (s1 * (square * square)), s1 * square * square * square); + VERIFY_IS_APPROX(square * (square * square).conjugate(), square * square.conjugate() * square.conjugate()); } - } diff --git a/filmulator-gui/core/nlmeans/eigen/test/product_extra.cpp b/filmulator-gui/core/nlmeans/eigen/test/product_extra.cpp index de2709d8..0dc5fb7f 100644 --- a/filmulator-gui/core/nlmeans/eigen/test/product_extra.cpp +++ b/filmulator-gui/core/nlmeans/eigen/test/product_extra.cpp @@ -9,177 +9,164 @@ #include "main.h" -template void product_extra(const MatrixType& m) +template void product_extra(const MatrixType &m) { typedef typename MatrixType::Scalar Scalar; typedef Matrix RowVectorType; typedef Matrix ColVectorType; - typedef Matrix OtherMajorMatrixType; + typedef Matrix OtherMajorMatrixType; Index rows = m.rows(); Index cols = m.cols(); - MatrixType m1 = MatrixType::Random(rows, cols), - m2 = MatrixType::Random(rows, cols), - m3(rows, cols), - mzero = MatrixType::Zero(rows, cols), - identity = MatrixType::Identity(rows, rows), - square = MatrixType::Random(rows, rows), - res = MatrixType::Random(rows, rows), - square2 = MatrixType::Random(cols, cols), - res2 = MatrixType::Random(cols, cols); + MatrixType m1 = MatrixType::Random(rows, cols), m2 = MatrixType::Random(rows, cols), m3(rows, cols), + mzero = MatrixType::Zero(rows, cols), identity = MatrixType::Identity(rows, rows), + square = MatrixType::Random(rows, rows), res = MatrixType::Random(rows, rows), + square2 = MatrixType::Random(cols, cols), res2 = MatrixType::Random(cols, cols); RowVectorType v1 = RowVectorType::Random(rows), vrres(rows); ColVectorType vc2 = ColVectorType::Random(cols), vcres(cols); OtherMajorMatrixType tm1 = m1; - Scalar s1 = internal::random(), - s2 = internal::random(), - s3 = internal::random(); + Scalar s1 = internal::random(), s2 = internal::random(), s3 = internal::random(); - VERIFY_IS_APPROX(m3.noalias() = m1 * m2.adjoint(), m1 * m2.adjoint().eval()); - VERIFY_IS_APPROX(m3.noalias() = m1.adjoint() * square.adjoint(), m1.adjoint().eval() * square.adjoint().eval()); - VERIFY_IS_APPROX(m3.noalias() = m1.adjoint() * m2, m1.adjoint().eval() * m2); - VERIFY_IS_APPROX(m3.noalias() = (s1 * m1.adjoint()) * m2, (s1 * m1.adjoint()).eval() * m2); - VERIFY_IS_APPROX(m3.noalias() = ((s1 * m1).adjoint()) * m2, (numext::conj(s1) * m1.adjoint()).eval() * m2); - VERIFY_IS_APPROX(m3.noalias() = (- m1.adjoint() * s1) * (s3 * m2), (- m1.adjoint() * s1).eval() * (s3 * m2).eval()); - VERIFY_IS_APPROX(m3.noalias() = (s2 * m1.adjoint() * s1) * m2, (s2 * m1.adjoint() * s1).eval() * m2); - VERIFY_IS_APPROX(m3.noalias() = (-m1*s2) * s1*m2.adjoint(), (-m1*s2).eval() * (s1*m2.adjoint()).eval()); + VERIFY_IS_APPROX(m3.noalias() = m1 * m2.adjoint(), m1 * m2.adjoint().eval()); + VERIFY_IS_APPROX(m3.noalias() = m1.adjoint() * square.adjoint(), m1.adjoint().eval() * square.adjoint().eval()); + VERIFY_IS_APPROX(m3.noalias() = m1.adjoint() * m2, m1.adjoint().eval() * m2); + VERIFY_IS_APPROX(m3.noalias() = (s1 * m1.adjoint()) * m2, (s1 * m1.adjoint()).eval() * m2); + VERIFY_IS_APPROX(m3.noalias() = ((s1 * m1).adjoint()) * m2, (numext::conj(s1) * m1.adjoint()).eval() * m2); + VERIFY_IS_APPROX(m3.noalias() = (-m1.adjoint() * s1) * (s3 * m2), (-m1.adjoint() * s1).eval() * (s3 * m2).eval()); + VERIFY_IS_APPROX(m3.noalias() = (s2 * m1.adjoint() * s1) * m2, (s2 * m1.adjoint() * s1).eval() * m2); + VERIFY_IS_APPROX(m3.noalias() = (-m1 * s2) * s1 * m2.adjoint(), (-m1 * s2).eval() * (s1 * m2.adjoint()).eval()); // a very tricky case where a scale factor has to be automatically conjugated: - VERIFY_IS_APPROX( m1.adjoint() * (s1*m2).conjugate(), (m1.adjoint()).eval() * ((s1*m2).conjugate()).eval()); + VERIFY_IS_APPROX(m1.adjoint() * (s1 * m2).conjugate(), (m1.adjoint()).eval() * ((s1 * m2).conjugate()).eval()); // test all possible conjugate combinations for the four matrix-vector product cases: - VERIFY_IS_APPROX((-m1.conjugate() * s2) * (s1 * vc2), - (-m1.conjugate()*s2).eval() * (s1 * vc2).eval()); - VERIFY_IS_APPROX((-m1 * s2) * (s1 * vc2.conjugate()), - (-m1*s2).eval() * (s1 * vc2.conjugate()).eval()); - VERIFY_IS_APPROX((-m1.conjugate() * s2) * (s1 * vc2.conjugate()), - (-m1.conjugate()*s2).eval() * (s1 * vc2.conjugate()).eval()); - - VERIFY_IS_APPROX((s1 * vc2.transpose()) * (-m1.adjoint() * s2), - (s1 * vc2.transpose()).eval() * (-m1.adjoint()*s2).eval()); - VERIFY_IS_APPROX((s1 * vc2.adjoint()) * (-m1.transpose() * s2), - (s1 * vc2.adjoint()).eval() * (-m1.transpose()*s2).eval()); - VERIFY_IS_APPROX((s1 * vc2.adjoint()) * (-m1.adjoint() * s2), - (s1 * vc2.adjoint()).eval() * (-m1.adjoint()*s2).eval()); - - VERIFY_IS_APPROX((-m1.adjoint() * s2) * (s1 * v1.transpose()), - (-m1.adjoint()*s2).eval() * (s1 * v1.transpose()).eval()); - VERIFY_IS_APPROX((-m1.transpose() * s2) * (s1 * v1.adjoint()), - (-m1.transpose()*s2).eval() * (s1 * v1.adjoint()).eval()); - VERIFY_IS_APPROX((-m1.adjoint() * s2) * (s1 * v1.adjoint()), - (-m1.adjoint()*s2).eval() * (s1 * v1.adjoint()).eval()); - - VERIFY_IS_APPROX((s1 * v1) * (-m1.conjugate() * s2), - (s1 * v1).eval() * (-m1.conjugate()*s2).eval()); - VERIFY_IS_APPROX((s1 * v1.conjugate()) * (-m1 * s2), - (s1 * v1.conjugate()).eval() * (-m1*s2).eval()); - VERIFY_IS_APPROX((s1 * v1.conjugate()) * (-m1.conjugate() * s2), - (s1 * v1.conjugate()).eval() * (-m1.conjugate()*s2).eval()); - - VERIFY_IS_APPROX((-m1.adjoint() * s2) * (s1 * v1.adjoint()), - (-m1.adjoint()*s2).eval() * (s1 * v1.adjoint()).eval()); + VERIFY_IS_APPROX((-m1.conjugate() * s2) * (s1 * vc2), (-m1.conjugate() * s2).eval() * (s1 * vc2).eval()); + VERIFY_IS_APPROX((-m1 * s2) * (s1 * vc2.conjugate()), (-m1 * s2).eval() * (s1 * vc2.conjugate()).eval()); + VERIFY_IS_APPROX( + (-m1.conjugate() * s2) * (s1 * vc2.conjugate()), (-m1.conjugate() * s2).eval() * (s1 * vc2.conjugate()).eval()); + + VERIFY_IS_APPROX( + (s1 * vc2.transpose()) * (-m1.adjoint() * s2), (s1 * vc2.transpose()).eval() * (-m1.adjoint() * s2).eval()); + VERIFY_IS_APPROX( + (s1 * vc2.adjoint()) * (-m1.transpose() * s2), (s1 * vc2.adjoint()).eval() * (-m1.transpose() * s2).eval()); + VERIFY_IS_APPROX( + (s1 * vc2.adjoint()) * (-m1.adjoint() * s2), (s1 * vc2.adjoint()).eval() * (-m1.adjoint() * s2).eval()); + + VERIFY_IS_APPROX( + (-m1.adjoint() * s2) * (s1 * v1.transpose()), (-m1.adjoint() * s2).eval() * (s1 * v1.transpose()).eval()); + VERIFY_IS_APPROX( + (-m1.transpose() * s2) * (s1 * v1.adjoint()), (-m1.transpose() * s2).eval() * (s1 * v1.adjoint()).eval()); + VERIFY_IS_APPROX( + (-m1.adjoint() * s2) * (s1 * v1.adjoint()), (-m1.adjoint() * s2).eval() * (s1 * v1.adjoint()).eval()); + + VERIFY_IS_APPROX((s1 * v1) * (-m1.conjugate() * s2), (s1 * v1).eval() * (-m1.conjugate() * s2).eval()); + VERIFY_IS_APPROX((s1 * v1.conjugate()) * (-m1 * s2), (s1 * v1.conjugate()).eval() * (-m1 * s2).eval()); + VERIFY_IS_APPROX( + (s1 * v1.conjugate()) * (-m1.conjugate() * s2), (s1 * v1.conjugate()).eval() * (-m1.conjugate() * s2).eval()); + + VERIFY_IS_APPROX( + (-m1.adjoint() * s2) * (s1 * v1.adjoint()), (-m1.adjoint() * s2).eval() * (s1 * v1.adjoint()).eval()); // test the vector-matrix product with non aligned starts - Index i = internal::random(0,m1.rows()-2); - Index j = internal::random(0,m1.cols()-2); - Index r = internal::random(1,m1.rows()-i); - Index c = internal::random(1,m1.cols()-j); - Index i2 = internal::random(0,m1.rows()-1); - Index j2 = internal::random(0,m1.cols()-1); - - VERIFY_IS_APPROX(m1.col(j2).adjoint() * m1.block(0,j,m1.rows(),c), m1.col(j2).adjoint().eval() * m1.block(0,j,m1.rows(),c).eval()); - VERIFY_IS_APPROX(m1.block(i,0,r,m1.cols()) * m1.row(i2).adjoint(), m1.block(i,0,r,m1.cols()).eval() * m1.row(i2).adjoint().eval()); - + Index i = internal::random(0, m1.rows() - 2); + Index j = internal::random(0, m1.cols() - 2); + Index r = internal::random(1, m1.rows() - i); + Index c = internal::random(1, m1.cols() - j); + Index i2 = internal::random(0, m1.rows() - 1); + Index j2 = internal::random(0, m1.cols() - 1); + + VERIFY_IS_APPROX(m1.col(j2).adjoint() * m1.block(0, j, m1.rows(), c), + m1.col(j2).adjoint().eval() * m1.block(0, j, m1.rows(), c).eval()); + VERIFY_IS_APPROX(m1.block(i, 0, r, m1.cols()) * m1.row(i2).adjoint(), + m1.block(i, 0, r, m1.cols()).eval() * m1.row(i2).adjoint().eval()); + // regression test MatrixType tmp = m1 * m1.adjoint() * s1; VERIFY_IS_APPROX(tmp, m1 * m1.adjoint() * s1); // regression test for bug 1343, assignment to arrays - Array a1 = m1 * vc2; - VERIFY_IS_APPROX(a1.matrix(),m1*vc2); - Array a2 = s1 * (m1 * vc2); - VERIFY_IS_APPROX(a2.matrix(),s1*m1*vc2); - Array a3 = v1 * m1; - VERIFY_IS_APPROX(a3.matrix(),v1*m1); - Array a4 = m1 * m2.adjoint(); - VERIFY_IS_APPROX(a4.matrix(),m1*m2.adjoint()); + Array a1 = m1 * vc2; + VERIFY_IS_APPROX(a1.matrix(), m1 * vc2); + Array a2 = s1 * (m1 * vc2); + VERIFY_IS_APPROX(a2.matrix(), s1 * m1 * vc2); + Array a3 = v1 * m1; + VERIFY_IS_APPROX(a3.matrix(), v1 * m1); + Array a4 = m1 * m2.adjoint(); + VERIFY_IS_APPROX(a4.matrix(), m1 * m2.adjoint()); } // Regression test for bug reported at http://forum.kde.org/viewtopic.php?f=74&t=96947 void mat_mat_scalar_scalar_product() { Eigen::Matrix2Xd dNdxy(2, 3); - dNdxy << -0.5, 0.5, 0, - -0.3, 0, 0.3; + dNdxy << -0.5, 0.5, 0, -0.3, 0, 0.3; double det = 6.0, wt = 0.5; - VERIFY_IS_APPROX(dNdxy.transpose()*dNdxy*det*wt, det*wt*dNdxy.transpose()*dNdxy); + VERIFY_IS_APPROX(dNdxy.transpose() * dNdxy * det * wt, det * wt * dNdxy.transpose() * dNdxy); } -template -void zero_sized_objects(const MatrixType& m) +template void zero_sized_objects(const MatrixType &m) { typedef typename MatrixType::Scalar Scalar; - const int PacketSize = internal::packet_traits::size; - const int PacketSize1 = PacketSize>1 ? PacketSize-1 : 1; + const int PacketSize = internal::packet_traits::size; + const int PacketSize1 = PacketSize > 1 ? PacketSize - 1 : 1; Index rows = m.rows(); Index cols = m.cols(); - + { - MatrixType res, a(rows,0), b(0,cols); - VERIFY_IS_APPROX( (res=a*b), MatrixType::Zero(rows,cols) ); - VERIFY_IS_APPROX( (res=a*a.transpose()), MatrixType::Zero(rows,rows) ); - VERIFY_IS_APPROX( (res=b.transpose()*b), MatrixType::Zero(cols,cols) ); - VERIFY_IS_APPROX( (res=b.transpose()*a.transpose()), MatrixType::Zero(cols,rows) ); + MatrixType res, a(rows, 0), b(0, cols); + VERIFY_IS_APPROX((res = a * b), MatrixType::Zero(rows, cols)); + VERIFY_IS_APPROX((res = a * a.transpose()), MatrixType::Zero(rows, rows)); + VERIFY_IS_APPROX((res = b.transpose() * b), MatrixType::Zero(cols, cols)); + VERIFY_IS_APPROX((res = b.transpose() * a.transpose()), MatrixType::Zero(cols, rows)); } - + { - MatrixType res, a(rows,cols), b(cols,0); - res = a*b; - VERIFY(res.rows()==rows && res.cols()==0); - b.resize(0,rows); - res = b*a; - VERIFY(res.rows()==0 && res.cols()==cols); + MatrixType res, a(rows, cols), b(cols, 0); + res = a * b; + VERIFY(res.rows() == rows && res.cols() == 0); + b.resize(0, rows); + res = b * a; + VERIFY(res.rows() == 0 && res.cols() == cols); } - + { - Matrix a; - Matrix b; - Matrix res; - VERIFY_IS_APPROX( (res=a*b), MatrixType::Zero(PacketSize,1) ); - VERIFY_IS_APPROX( (res=a.lazyProduct(b)), MatrixType::Zero(PacketSize,1) ); + Matrix a; + Matrix b; + Matrix res; + VERIFY_IS_APPROX((res = a * b), MatrixType::Zero(PacketSize, 1)); + VERIFY_IS_APPROX((res = a.lazyProduct(b)), MatrixType::Zero(PacketSize, 1)); } - + { - Matrix a; - Matrix b; - Matrix res; - VERIFY_IS_APPROX( (res=a*b), MatrixType::Zero(PacketSize1,1) ); - VERIFY_IS_APPROX( (res=a.lazyProduct(b)), MatrixType::Zero(PacketSize1,1) ); + Matrix a; + Matrix b; + Matrix res; + VERIFY_IS_APPROX((res = a * b), MatrixType::Zero(PacketSize1, 1)); + VERIFY_IS_APPROX((res = a.lazyProduct(b)), MatrixType::Zero(PacketSize1, 1)); } - + { - Matrix a(PacketSize,0); - Matrix b(0,1); - Matrix res; - VERIFY_IS_APPROX( (res=a*b), MatrixType::Zero(PacketSize,1) ); - VERIFY_IS_APPROX( (res=a.lazyProduct(b)), MatrixType::Zero(PacketSize,1) ); + Matrix a(PacketSize, 0); + Matrix b(0, 1); + Matrix res; + VERIFY_IS_APPROX((res = a * b), MatrixType::Zero(PacketSize, 1)); + VERIFY_IS_APPROX((res = a.lazyProduct(b)), MatrixType::Zero(PacketSize, 1)); } - + { - Matrix a(PacketSize1,0); - Matrix b(0,1); - Matrix res; - VERIFY_IS_APPROX( (res=a*b), MatrixType::Zero(PacketSize1,1) ); - VERIFY_IS_APPROX( (res=a.lazyProduct(b)), MatrixType::Zero(PacketSize1,1) ); + Matrix a(PacketSize1, 0); + Matrix b(0, 1); + Matrix res; + VERIFY_IS_APPROX((res = a * b), MatrixType::Zero(PacketSize1, 1)); + VERIFY_IS_APPROX((res = a.lazyProduct(b)), MatrixType::Zero(PacketSize1, 1)); } } -template -void bug_127() +template void bug_127() { // Bug 127 // @@ -195,35 +182,32 @@ void bug_127() // RowsAtCompileTime = -1, ColsAtCompileTime = -1 // MaxRowsAtCompileTime = 5, MaxColsAtCompileTime = 1 // - // was failing on a runtime assertion, because it had been mis-compiled as a dot product because Product.h was using the - // max-sizes to detect size 1 indicating vectors, and that didn't account for 0-sized object with max-size 1. + // was failing on a runtime assertion, because it had been mis-compiled as a dot product because Product.h was using + // the max-sizes to detect size 1 indicating vectors, and that didn't account for 0-sized object with max-size 1. - Matrix a(1,4); - Matrix b(4,0); - a*b; + Matrix a(1, 4); + Matrix b(4, 0); + a *b; } template void bug_817() { - ArrayXXf B = ArrayXXf::Random(10,10), C; + ArrayXXf B = ArrayXXf::Random(10, 10), C; VectorXf x = VectorXf::Random(10); - C = (x.transpose()*B.matrix()); - B = (x.transpose()*B.matrix()); - VERIFY_IS_APPROX(B,C); + C = (x.transpose() * B.matrix()); + B = (x.transpose() * B.matrix()); + VERIFY_IS_APPROX(B, C); } -template -void unaligned_objects() +template void unaligned_objects() { // Regression test for the bug reported here: // http://forum.kde.org/viewtopic.php?f=74&t=107541 // Recall the matrix*vector kernel avoid unaligned loads by loading two packets and then reassemble then. // There was a mistake in the computation of the valid range for fully unaligned objects: in some rare cases, // memory was read outside the allocated matrix memory. Though the values were not used, this might raise segfault. - for(int m=450;m<460;++m) - { - for(int n=8;n<12;++n) - { + for (int m = 450; m < 460; ++m) { + for (int n = 8; n < 12; ++n) { MatrixXf M(m, n); VectorXf v1(n), r1(500); RowVectorXf v2(m), r2(16); @@ -231,89 +215,83 @@ void unaligned_objects() M.setRandom(); v1.setRandom(); v2.setRandom(); - for(int o=0; o<4; ++o) - { - r1.segment(o,m).noalias() = M * v1; - VERIFY_IS_APPROX(r1.segment(o,m), M * MatrixXf(v1)); - r2.segment(o,n).noalias() = v2 * M; - VERIFY_IS_APPROX(r2.segment(o,n), MatrixXf(v2) * M); + for (int o = 0; o < 4; ++o) { + r1.segment(o, m).noalias() = M * v1; + VERIFY_IS_APPROX(r1.segment(o, m), M * MatrixXf(v1)); + r2.segment(o, n).noalias() = v2 * M; + VERIFY_IS_APPROX(r2.segment(o, n), MatrixXf(v2) * M); } } } } -template -EIGEN_DONT_INLINE -Index test_compute_block_size(Index m, Index n, Index k) +template EIGEN_DONT_INLINE Index test_compute_block_size(Index m, Index n, Index k) { Index mc(m), nc(n), kc(k); - internal::computeProductBlockingSizes(kc, mc, nc); - return kc+mc+nc; + internal::computeProductBlockingSizes(kc, mc, nc); + return kc + mc + nc; } -template -Index compute_block_size() +template Index compute_block_size() { Index ret = 0; - ret += test_compute_block_size(0,1,1); - ret += test_compute_block_size(1,0,1); - ret += test_compute_block_size(1,1,0); - ret += test_compute_block_size(0,0,1); - ret += test_compute_block_size(0,1,0); - ret += test_compute_block_size(1,0,0); - ret += test_compute_block_size(0,0,0); + ret += test_compute_block_size(0, 1, 1); + ret += test_compute_block_size(1, 0, 1); + ret += test_compute_block_size(1, 1, 0); + ret += test_compute_block_size(0, 0, 1); + ret += test_compute_block_size(0, 1, 0); + ret += test_compute_block_size(1, 0, 0); + ret += test_compute_block_size(0, 0, 0); return ret; } -template -void aliasing_with_resize() +template void aliasing_with_resize() { - Index m = internal::random(10,50); - Index n = internal::random(10,50); - MatrixXd A, B, C(m,n), D(m,m); + Index m = internal::random(10, 50); + Index n = internal::random(10, 50); + MatrixXd A, B, C(m, n), D(m, m); VectorXd a, b, c(n); C.setRandom(); D.setRandom(); c.setRandom(); - double s = internal::random(1,10); + double s = internal::random(1, 10); A = C; B = A * A.transpose(); A = A * A.transpose(); - VERIFY_IS_APPROX(A,B); + VERIFY_IS_APPROX(A, B); A = C; - B = (A * A.transpose())/s; - A = (A * A.transpose())/s; - VERIFY_IS_APPROX(A,B); + B = (A * A.transpose()) / s; + A = (A * A.transpose()) / s; + VERIFY_IS_APPROX(A, B); A = C; B = (A * A.transpose()) + D; A = (A * A.transpose()) + D; - VERIFY_IS_APPROX(A,B); + VERIFY_IS_APPROX(A, B); A = C; B = D + (A * A.transpose()); A = D + (A * A.transpose()); - VERIFY_IS_APPROX(A,B); + VERIFY_IS_APPROX(A, B); A = C; B = s * (A * A.transpose()); A = s * (A * A.transpose()); - VERIFY_IS_APPROX(A,B); + VERIFY_IS_APPROX(A, B); A = C; a = c; - b = (A * a)/s; - a = (A * a)/s; - VERIFY_IS_APPROX(a,b); + b = (A * a) / s; + a = (A * a) / s; + VERIFY_IS_APPROX(a, b); } -template -void bug_1308() +template void bug_1308() { int n = 10; - MatrixXd r(n,n); + MatrixXd r(n, n); VectorXd v = VectorXd::Random(n); r = v * RowVectorXd::Ones(n); VERIFY_IS_APPROX(r, v.rowwise().replicate(n)); @@ -322,25 +300,25 @@ void bug_1308() Matrix4d ones44 = Matrix4d::Ones(); Matrix4d m44 = Matrix4d::Ones() * Matrix4d::Ones(); - VERIFY_IS_APPROX(m44,Matrix4d::Constant(4)); - VERIFY_IS_APPROX(m44.noalias()=ones44*Matrix4d::Ones(), Matrix4d::Constant(4)); - VERIFY_IS_APPROX(m44.noalias()=ones44.transpose()*Matrix4d::Ones(), Matrix4d::Constant(4)); - VERIFY_IS_APPROX(m44.noalias()=Matrix4d::Ones()*ones44, Matrix4d::Constant(4)); - VERIFY_IS_APPROX(m44.noalias()=Matrix4d::Ones()*ones44.transpose(), Matrix4d::Constant(4)); + VERIFY_IS_APPROX(m44, Matrix4d::Constant(4)); + VERIFY_IS_APPROX(m44.noalias() = ones44 * Matrix4d::Ones(), Matrix4d::Constant(4)); + VERIFY_IS_APPROX(m44.noalias() = ones44.transpose() * Matrix4d::Ones(), Matrix4d::Constant(4)); + VERIFY_IS_APPROX(m44.noalias() = Matrix4d::Ones() * ones44, Matrix4d::Constant(4)); + VERIFY_IS_APPROX(m44.noalias() = Matrix4d::Ones() * ones44.transpose(), Matrix4d::Constant(4)); - typedef Matrix RMatrix4d; + typedef Matrix RMatrix4d; RMatrix4d r44 = Matrix4d::Ones() * Matrix4d::Ones(); - VERIFY_IS_APPROX(r44,Matrix4d::Constant(4)); - VERIFY_IS_APPROX(r44.noalias()=ones44*Matrix4d::Ones(), Matrix4d::Constant(4)); - VERIFY_IS_APPROX(r44.noalias()=ones44.transpose()*Matrix4d::Ones(), Matrix4d::Constant(4)); - VERIFY_IS_APPROX(r44.noalias()=Matrix4d::Ones()*ones44, Matrix4d::Constant(4)); - VERIFY_IS_APPROX(r44.noalias()=Matrix4d::Ones()*ones44.transpose(), Matrix4d::Constant(4)); - VERIFY_IS_APPROX(r44.noalias()=ones44*RMatrix4d::Ones(), Matrix4d::Constant(4)); - VERIFY_IS_APPROX(r44.noalias()=ones44.transpose()*RMatrix4d::Ones(), Matrix4d::Constant(4)); - VERIFY_IS_APPROX(r44.noalias()=RMatrix4d::Ones()*ones44, Matrix4d::Constant(4)); - VERIFY_IS_APPROX(r44.noalias()=RMatrix4d::Ones()*ones44.transpose(), Matrix4d::Constant(4)); - -// RowVector4d r4; + VERIFY_IS_APPROX(r44, Matrix4d::Constant(4)); + VERIFY_IS_APPROX(r44.noalias() = ones44 * Matrix4d::Ones(), Matrix4d::Constant(4)); + VERIFY_IS_APPROX(r44.noalias() = ones44.transpose() * Matrix4d::Ones(), Matrix4d::Constant(4)); + VERIFY_IS_APPROX(r44.noalias() = Matrix4d::Ones() * ones44, Matrix4d::Constant(4)); + VERIFY_IS_APPROX(r44.noalias() = Matrix4d::Ones() * ones44.transpose(), Matrix4d::Constant(4)); + VERIFY_IS_APPROX(r44.noalias() = ones44 * RMatrix4d::Ones(), Matrix4d::Constant(4)); + VERIFY_IS_APPROX(r44.noalias() = ones44.transpose() * RMatrix4d::Ones(), Matrix4d::Constant(4)); + VERIFY_IS_APPROX(r44.noalias() = RMatrix4d::Ones() * ones44, Matrix4d::Constant(4)); + VERIFY_IS_APPROX(r44.noalias() = RMatrix4d::Ones() * ones44.transpose(), Matrix4d::Constant(4)); + + // RowVector4d r4; m44.setOnes(); r44.setZero(); VERIFY_IS_APPROX(r44.noalias() += m44.row(0).transpose() * RowVector4d::Ones(), ones44); @@ -354,21 +332,25 @@ void bug_1308() void test_product_extra() { - for(int i = 0; i < g_repeat; i++) { - CALL_SUBTEST_1( product_extra(MatrixXf(internal::random(1,EIGEN_TEST_MAX_SIZE), internal::random(1,EIGEN_TEST_MAX_SIZE))) ); - CALL_SUBTEST_2( product_extra(MatrixXd(internal::random(1,EIGEN_TEST_MAX_SIZE), internal::random(1,EIGEN_TEST_MAX_SIZE))) ); - CALL_SUBTEST_2( mat_mat_scalar_scalar_product() ); - CALL_SUBTEST_3( product_extra(MatrixXcf(internal::random(1,EIGEN_TEST_MAX_SIZE/2), internal::random(1,EIGEN_TEST_MAX_SIZE/2))) ); - CALL_SUBTEST_4( product_extra(MatrixXcd(internal::random(1,EIGEN_TEST_MAX_SIZE/2), internal::random(1,EIGEN_TEST_MAX_SIZE/2))) ); - CALL_SUBTEST_1( zero_sized_objects(MatrixXf(internal::random(1,EIGEN_TEST_MAX_SIZE), internal::random(1,EIGEN_TEST_MAX_SIZE))) ); + for (int i = 0; i < g_repeat; i++) { + CALL_SUBTEST_1(product_extra( + MatrixXf(internal::random(1, EIGEN_TEST_MAX_SIZE), internal::random(1, EIGEN_TEST_MAX_SIZE)))); + CALL_SUBTEST_2(product_extra( + MatrixXd(internal::random(1, EIGEN_TEST_MAX_SIZE), internal::random(1, EIGEN_TEST_MAX_SIZE)))); + CALL_SUBTEST_2(mat_mat_scalar_scalar_product()); + CALL_SUBTEST_3(product_extra( + MatrixXcf(internal::random(1, EIGEN_TEST_MAX_SIZE / 2), internal::random(1, EIGEN_TEST_MAX_SIZE / 2)))); + CALL_SUBTEST_4(product_extra( + MatrixXcd(internal::random(1, EIGEN_TEST_MAX_SIZE / 2), internal::random(1, EIGEN_TEST_MAX_SIZE / 2)))); + CALL_SUBTEST_1(zero_sized_objects( + MatrixXf(internal::random(1, EIGEN_TEST_MAX_SIZE), internal::random(1, EIGEN_TEST_MAX_SIZE)))); } - CALL_SUBTEST_5( bug_127<0>() ); - CALL_SUBTEST_5( bug_817<0>() ); - CALL_SUBTEST_5( bug_1308<0>() ); - CALL_SUBTEST_6( unaligned_objects<0>() ); - CALL_SUBTEST_7( compute_block_size() ); - CALL_SUBTEST_7( compute_block_size() ); - CALL_SUBTEST_7( compute_block_size >() ); - CALL_SUBTEST_8( aliasing_with_resize() ); - + CALL_SUBTEST_5(bug_127<0>()); + CALL_SUBTEST_5(bug_817<0>()); + CALL_SUBTEST_5(bug_1308<0>()); + CALL_SUBTEST_6(unaligned_objects<0>()); + CALL_SUBTEST_7(compute_block_size()); + CALL_SUBTEST_7(compute_block_size()); + CALL_SUBTEST_7(compute_block_size>()); + CALL_SUBTEST_8(aliasing_with_resize()); } diff --git a/filmulator-gui/core/nlmeans/eigen/test/product_large.cpp b/filmulator-gui/core/nlmeans/eigen/test/product_large.cpp index 845cd40c..76effba7 100644 --- a/filmulator-gui/core/nlmeans/eigen/test/product_large.cpp +++ b/filmulator-gui/core/nlmeans/eigen/test/product_large.cpp @@ -9,22 +9,25 @@ #include "product.h" -template -void test_aliasing() +template void test_aliasing() { - int rows = internal::random(1,12); - int cols = internal::random(1,12); - typedef Matrix MatrixType; - typedef Matrix VectorType; - VectorType x(cols); x.setRandom(); + int rows = internal::random(1, 12); + int cols = internal::random(1, 12); + typedef Matrix MatrixType; + typedef Matrix VectorType; + VectorType x(cols); + x.setRandom(); VectorType z(x); - VectorType y(rows); y.setZero(); - MatrixType A(rows,cols); A.setRandom(); + VectorType y(rows); + y.setZero(); + MatrixType A(rows, cols); + A.setRandom(); // CwiseBinaryOp - VERIFY_IS_APPROX(x = y + A*x, A*z); // OK because "y + A*x" is marked as "assume-aliasing" + VERIFY_IS_APPROX(x = y + A * x, A * z);// OK because "y + A*x" is marked as "assume-aliasing" x = z; // CwiseUnaryOp - VERIFY_IS_APPROX(x = T(1.)*(A*x), A*z); // OK because 1*(A*x) is replaced by (1*A*x) which is a Product<> expression + VERIFY_IS_APPROX( + x = T(1.) * (A * x), A * z);// OK because 1*(A*x) is replaced by (1*A*x) which is a Product<> expression x = z; // VERIFY_IS_APPROX(x = y-A*x, -A*z); // Not OK in 3.3 because x is resized before A*x gets evaluated x = z; @@ -32,14 +35,19 @@ void test_aliasing() void test_product_large() { - for(int i = 0; i < g_repeat; i++) { - CALL_SUBTEST_1( product(MatrixXf(internal::random(1,EIGEN_TEST_MAX_SIZE), internal::random(1,EIGEN_TEST_MAX_SIZE))) ); - CALL_SUBTEST_2( product(MatrixXd(internal::random(1,EIGEN_TEST_MAX_SIZE), internal::random(1,EIGEN_TEST_MAX_SIZE))) ); - CALL_SUBTEST_3( product(MatrixXi(internal::random(1,EIGEN_TEST_MAX_SIZE), internal::random(1,EIGEN_TEST_MAX_SIZE))) ); - CALL_SUBTEST_4( product(MatrixXcf(internal::random(1,EIGEN_TEST_MAX_SIZE/2), internal::random(1,EIGEN_TEST_MAX_SIZE/2))) ); - CALL_SUBTEST_5( product(Matrix(internal::random(1,EIGEN_TEST_MAX_SIZE), internal::random(1,EIGEN_TEST_MAX_SIZE))) ); + for (int i = 0; i < g_repeat; i++) { + CALL_SUBTEST_1( + product(MatrixXf(internal::random(1, EIGEN_TEST_MAX_SIZE), internal::random(1, EIGEN_TEST_MAX_SIZE)))); + CALL_SUBTEST_2( + product(MatrixXd(internal::random(1, EIGEN_TEST_MAX_SIZE), internal::random(1, EIGEN_TEST_MAX_SIZE)))); + CALL_SUBTEST_3( + product(MatrixXi(internal::random(1, EIGEN_TEST_MAX_SIZE), internal::random(1, EIGEN_TEST_MAX_SIZE)))); + CALL_SUBTEST_4(product( + MatrixXcf(internal::random(1, EIGEN_TEST_MAX_SIZE / 2), internal::random(1, EIGEN_TEST_MAX_SIZE / 2)))); + CALL_SUBTEST_5(product(Matrix( + internal::random(1, EIGEN_TEST_MAX_SIZE), internal::random(1, EIGEN_TEST_MAX_SIZE)))); - CALL_SUBTEST_1( test_aliasing() ); + CALL_SUBTEST_1(test_aliasing()); } #if defined EIGEN_TEST_PART_6 @@ -47,61 +55,66 @@ void test_product_large() // test a specific issue in DiagonalProduct int N = 1000000; VectorXf v = VectorXf::Ones(N); - MatrixXf m = MatrixXf::Ones(N,3); - m = (v+v).asDiagonal() * m; - VERIFY_IS_APPROX(m, MatrixXf::Constant(N,3,2)); + MatrixXf m = MatrixXf::Ones(N, 3); + m = (v + v).asDiagonal() * m; + VERIFY_IS_APPROX(m, MatrixXf::Constant(N, 3, 2)); } { // test deferred resizing in Matrix::operator= - MatrixXf a = MatrixXf::Random(10,4), b = MatrixXf::Random(4,10), c = a; + MatrixXf a = MatrixXf::Random(10, 4), b = MatrixXf::Random(4, 10), c = a; VERIFY_IS_APPROX((a = a * b), (c * b).eval()); } { // check the functions to setup blocking sizes compile and do not segfault // FIXME check they do what they are supposed to do !! - std::ptrdiff_t l1 = internal::random(10000,20000); - std::ptrdiff_t l2 = internal::random(100000,200000); - std::ptrdiff_t l3 = internal::random(1000000,2000000); - setCpuCacheSizes(l1,l2,l3); - VERIFY(l1==l1CacheSize()); - VERIFY(l2==l2CacheSize()); - std::ptrdiff_t k1 = internal::random(10,100)*16; - std::ptrdiff_t m1 = internal::random(10,100)*16; - std::ptrdiff_t n1 = internal::random(10,100)*16; + std::ptrdiff_t l1 = internal::random(10000, 20000); + std::ptrdiff_t l2 = internal::random(100000, 200000); + std::ptrdiff_t l3 = internal::random(1000000, 2000000); + setCpuCacheSizes(l1, l2, l3); + VERIFY(l1 == l1CacheSize()); + VERIFY(l2 == l2CacheSize()); + std::ptrdiff_t k1 = internal::random(10, 100) * 16; + std::ptrdiff_t m1 = internal::random(10, 100) * 16; + std::ptrdiff_t n1 = internal::random(10, 100) * 16; // only makes sure it compiles fine - internal::computeProductBlockingSizes(k1,m1,n1,1); + internal::computeProductBlockingSizes(k1, m1, n1, 1); } { // test regression in row-vector by matrix (bad Map type) - MatrixXf mat1(10,32); mat1.setRandom(); - MatrixXf mat2(32,32); mat2.setRandom(); - MatrixXf r1 = mat1.row(2)*mat2.transpose(); - VERIFY_IS_APPROX(r1, (mat1.row(2)*mat2.transpose()).eval()); + MatrixXf mat1(10, 32); + mat1.setRandom(); + MatrixXf mat2(32, 32); + mat2.setRandom(); + MatrixXf r1 = mat1.row(2) * mat2.transpose(); + VERIFY_IS_APPROX(r1, (mat1.row(2) * mat2.transpose()).eval()); - MatrixXf r2 = mat1.row(2)*mat2; - VERIFY_IS_APPROX(r2, (mat1.row(2)*mat2).eval()); + MatrixXf r2 = mat1.row(2) * mat2; + VERIFY_IS_APPROX(r2, (mat1.row(2) * mat2).eval()); } { - Eigen::MatrixXd A(10,10), B, C; + Eigen::MatrixXd A(10, 10), B, C; A.setRandom(); C = A; - for(int k=0; k<79; ++k) - C = C * A; - B.noalias() = (((A*A)*(A*A))*((A*A)*(A*A))*((A*A)*(A*A))*((A*A)*(A*A))*((A*A)*(A*A)) * ((A*A)*(A*A))*((A*A)*(A*A))*((A*A)*(A*A))*((A*A)*(A*A))*((A*A)*(A*A))) - * (((A*A)*(A*A))*((A*A)*(A*A))*((A*A)*(A*A))*((A*A)*(A*A))*((A*A)*(A*A)) * ((A*A)*(A*A))*((A*A)*(A*A))*((A*A)*(A*A))*((A*A)*(A*A))*((A*A)*(A*A))); - VERIFY_IS_APPROX(B,C); + for (int k = 0; k < 79; ++k) C = C * A; + B.noalias() = + (((A * A) * (A * A)) * ((A * A) * (A * A)) * ((A * A) * (A * A)) * ((A * A) * (A * A)) * ((A * A) * (A * A)) + * ((A * A) * (A * A)) * ((A * A) * (A * A)) * ((A * A) * (A * A)) * ((A * A) * (A * A)) * ((A * A) * (A * A))) + * (((A * A) * (A * A)) * ((A * A) * (A * A)) * ((A * A) * (A * A)) * ((A * A) * (A * A)) * ((A * A) * (A * A)) + * ((A * A) * (A * A)) * ((A * A) * (A * A)) * ((A * A) * (A * A)) * ((A * A) * (A * A)) * ((A * A) * (A * A))); + VERIFY_IS_APPROX(B, C); } #endif // Regression test for bug 714: #if defined EIGEN_HAS_OPENMP omp_set_dynamic(1); - for(int i = 0; i < g_repeat; i++) { - CALL_SUBTEST_6( product(Matrix(internal::random(1,EIGEN_TEST_MAX_SIZE), internal::random(1,EIGEN_TEST_MAX_SIZE))) ); + for (int i = 0; i < g_repeat; i++) { + CALL_SUBTEST_6(product(Matrix( + internal::random(1, EIGEN_TEST_MAX_SIZE), internal::random(1, EIGEN_TEST_MAX_SIZE)))); } #endif } diff --git a/filmulator-gui/core/nlmeans/eigen/test/product_mmtr.cpp b/filmulator-gui/core/nlmeans/eigen/test/product_mmtr.cpp index d3e24b01..b4b0e2ae 100644 --- a/filmulator-gui/core/nlmeans/eigen/test/product_mmtr.cpp +++ b/filmulator-gui/core/nlmeans/eigen/test/product_mmtr.cpp @@ -9,66 +9,72 @@ #include "main.h" -#define CHECK_MMTR(DEST, TRI, OP) { \ - ref3 = DEST; \ - ref2 = ref1 = DEST; \ - DEST.template triangularView() OP; \ - ref1 OP; \ - ref2.template triangularView() \ - = ref1.template triangularView(); \ - VERIFY_IS_APPROX(DEST,ref2); \ - \ - DEST = ref3; \ - ref3 = ref2; \ - ref3.diagonal() = DEST.diagonal(); \ - DEST.template triangularView() OP; \ - VERIFY_IS_APPROX(DEST,ref3); \ +#define CHECK_MMTR(DEST, TRI, OP) \ + { \ + ref3 = DEST; \ + ref2 = ref1 = DEST; \ + DEST.template triangularView() OP; \ + ref1 OP; \ + ref2.template triangularView() = ref1.template triangularView(); \ + VERIFY_IS_APPROX(DEST, ref2); \ + \ + DEST = ref3; \ + ref3 = ref2; \ + ref3.diagonal() = DEST.diagonal(); \ + DEST.template triangularView() OP; \ + VERIFY_IS_APPROX(DEST, ref3); \ } template void mmtr(int size) { - typedef Matrix MatrixColMaj; - typedef Matrix MatrixRowMaj; + typedef Matrix MatrixColMaj; + typedef Matrix MatrixRowMaj; + + DenseIndex othersize = internal::random(1, 200); - DenseIndex othersize = internal::random(1,200); - MatrixColMaj matc = MatrixColMaj::Zero(size, size); MatrixRowMaj matr = MatrixRowMaj::Zero(size, size); - MatrixColMaj ref1(size, size), ref2(size, size), ref3(size,size); - - MatrixColMaj soc(size,othersize); soc.setRandom(); - MatrixColMaj osc(othersize,size); osc.setRandom(); - MatrixRowMaj sor(size,othersize); sor.setRandom(); - MatrixRowMaj osr(othersize,size); osr.setRandom(); - MatrixColMaj sqc(size,size); sqc.setRandom(); - MatrixRowMaj sqr(size,size); sqr.setRandom(); - + MatrixColMaj ref1(size, size), ref2(size, size), ref3(size, size); + + MatrixColMaj soc(size, othersize); + soc.setRandom(); + MatrixColMaj osc(othersize, size); + osc.setRandom(); + MatrixRowMaj sor(size, othersize); + sor.setRandom(); + MatrixRowMaj osr(othersize, size); + osr.setRandom(); + MatrixColMaj sqc(size, size); + sqc.setRandom(); + MatrixRowMaj sqr(size, size); + sqr.setRandom(); + Scalar s = internal::random(); - - CHECK_MMTR(matc, Lower, = s*soc*sor.adjoint()); - CHECK_MMTR(matc, Upper, = s*(soc*soc.adjoint())); - CHECK_MMTR(matr, Lower, = s*soc*soc.adjoint()); - CHECK_MMTR(matr, Upper, = soc*(s*sor.adjoint())); - - CHECK_MMTR(matc, Lower, += s*soc*soc.adjoint()); - CHECK_MMTR(matc, Upper, += s*(soc*sor.transpose())); - CHECK_MMTR(matr, Lower, += s*sor*soc.adjoint()); - CHECK_MMTR(matr, Upper, += soc*(s*soc.adjoint())); - - CHECK_MMTR(matc, Lower, -= s*soc*soc.adjoint()); - CHECK_MMTR(matc, Upper, -= s*(osc.transpose()*osc.conjugate())); - CHECK_MMTR(matr, Lower, -= s*soc*soc.adjoint()); - CHECK_MMTR(matr, Upper, -= soc*(s*soc.adjoint())); - - CHECK_MMTR(matc, Lower, -= s*sqr*sqc.template triangularView()); - CHECK_MMTR(matc, Upper, = s*sqc*sqr.template triangularView()); - CHECK_MMTR(matc, Lower, += s*sqr*sqc.template triangularView()); - CHECK_MMTR(matc, Upper, = s*sqc*sqc.template triangularView()); - - CHECK_MMTR(matc, Lower, = (s*sqr).template triangularView()*sqc); - CHECK_MMTR(matc, Upper, -= (s*sqc).template triangularView()*sqc); - CHECK_MMTR(matc, Lower, = (s*sqr).template triangularView()*sqc); - CHECK_MMTR(matc, Upper, += (s*sqc).template triangularView()*sqc); + + CHECK_MMTR(matc, Lower, = s * soc * sor.adjoint()); + CHECK_MMTR(matc, Upper, = s * (soc * soc.adjoint())); + CHECK_MMTR(matr, Lower, = s * soc * soc.adjoint()); + CHECK_MMTR(matr, Upper, = soc * (s * sor.adjoint())); + + CHECK_MMTR(matc, Lower, += s * soc * soc.adjoint()); + CHECK_MMTR(matc, Upper, += s * (soc * sor.transpose())); + CHECK_MMTR(matr, Lower, += s * sor * soc.adjoint()); + CHECK_MMTR(matr, Upper, += soc * (s * soc.adjoint())); + + CHECK_MMTR(matc, Lower, -= s * soc * soc.adjoint()); + CHECK_MMTR(matc, Upper, -= s * (osc.transpose() * osc.conjugate())); + CHECK_MMTR(matr, Lower, -= s * soc * soc.adjoint()); + CHECK_MMTR(matr, Upper, -= soc * (s * soc.adjoint())); + + CHECK_MMTR(matc, Lower, -= s * sqr * sqc.template triangularView()); + CHECK_MMTR(matc, Upper, = s * sqc * sqr.template triangularView()); + CHECK_MMTR(matc, Lower, += s * sqr * sqc.template triangularView()); + CHECK_MMTR(matc, Upper, = s * sqc * sqc.template triangularView()); + + CHECK_MMTR(matc, Lower, = (s * sqr).template triangularView() * sqc); + CHECK_MMTR(matc, Upper, -= (s * sqc).template triangularView() * sqc); + CHECK_MMTR(matc, Lower, = (s * sqr).template triangularView() * sqc); + CHECK_MMTR(matc, Upper, += (s * sqc).template triangularView() * sqc); // check aliasing ref2 = ref1 = matc; @@ -86,11 +92,10 @@ template void mmtr(int size) void test_product_mmtr() { - for(int i = 0; i < g_repeat ; i++) - { - CALL_SUBTEST_1((mmtr(internal::random(1,EIGEN_TEST_MAX_SIZE)))); - CALL_SUBTEST_2((mmtr(internal::random(1,EIGEN_TEST_MAX_SIZE)))); - CALL_SUBTEST_3((mmtr >(internal::random(1,EIGEN_TEST_MAX_SIZE/2)))); - CALL_SUBTEST_4((mmtr >(internal::random(1,EIGEN_TEST_MAX_SIZE/2)))); + for (int i = 0; i < g_repeat; i++) { + CALL_SUBTEST_1((mmtr(internal::random(1, EIGEN_TEST_MAX_SIZE)))); + CALL_SUBTEST_2((mmtr(internal::random(1, EIGEN_TEST_MAX_SIZE)))); + CALL_SUBTEST_3((mmtr>(internal::random(1, EIGEN_TEST_MAX_SIZE / 2)))); + CALL_SUBTEST_4((mmtr>(internal::random(1, EIGEN_TEST_MAX_SIZE / 2)))); } } diff --git a/filmulator-gui/core/nlmeans/eigen/test/product_notemporary.cpp b/filmulator-gui/core/nlmeans/eigen/test/product_notemporary.cpp index 28865d39..1554e76a 100644 --- a/filmulator-gui/core/nlmeans/eigen/test/product_notemporary.cpp +++ b/filmulator-gui/core/nlmeans/eigen/test/product_notemporary.cpp @@ -11,7 +11,7 @@ #include "main.h" -template void product_notemporary(const MatrixType& m) +template void product_notemporary(const MatrixType &m) { /* This test checks the number of temporaries created * during the evaluation of a complex expression */ @@ -25,135 +25,148 @@ template void product_notemporary(const MatrixType& m) Index rows = m.rows(); Index cols = m.cols(); - ColMajorMatrixType m1 = MatrixType::Random(rows, cols), - m2 = MatrixType::Random(rows, cols), - m3(rows, cols); + ColMajorMatrixType m1 = MatrixType::Random(rows, cols), m2 = MatrixType::Random(rows, cols), m3(rows, cols); RowVectorType rv1 = RowVectorType::Random(rows), rvres(rows); ColVectorType cv1 = ColVectorType::Random(cols), cvres(cols); RowMajorMatrixType rm3(rows, cols); - Scalar s1 = internal::random(), - s2 = internal::random(), - s3 = internal::random(); + Scalar s1 = internal::random(), s2 = internal::random(), s3 = internal::random(); - Index c0 = internal::random(4,cols-8), - c1 = internal::random(8,cols-c0), - r0 = internal::random(4,cols-8), - r1 = internal::random(8,rows-r0); + Index c0 = internal::random(4, cols - 8), c1 = internal::random(8, cols - c0), + r0 = internal::random(4, cols - 8), r1 = internal::random(8, rows - r0); - VERIFY_EVALUATION_COUNT( m3 = (m1 * m2.adjoint()), 1); - VERIFY_EVALUATION_COUNT( m3 = (m1 * m2.adjoint()).transpose(), 1); - VERIFY_EVALUATION_COUNT( m3.noalias() = m1 * m2.adjoint(), 0); + VERIFY_EVALUATION_COUNT(m3 = (m1 * m2.adjoint()), 1); + VERIFY_EVALUATION_COUNT(m3 = (m1 * m2.adjoint()).transpose(), 1); + VERIFY_EVALUATION_COUNT(m3.noalias() = m1 * m2.adjoint(), 0); - VERIFY_EVALUATION_COUNT( m3 = s1 * (m1 * m2.transpose()), 1); -// VERIFY_EVALUATION_COUNT( m3 = m3 + s1 * (m1 * m2.transpose()), 1); - VERIFY_EVALUATION_COUNT( m3.noalias() = s1 * (m1 * m2.transpose()), 0); + VERIFY_EVALUATION_COUNT(m3 = s1 * (m1 * m2.transpose()), 1); + // VERIFY_EVALUATION_COUNT( m3 = m3 + s1 * (m1 * m2.transpose()), 1); + VERIFY_EVALUATION_COUNT(m3.noalias() = s1 * (m1 * m2.transpose()), 0); - VERIFY_EVALUATION_COUNT( m3 = m3 + (m1 * m2.adjoint()), 1); - VERIFY_EVALUATION_COUNT( m3 = m3 - (m1 * m2.adjoint()), 1); + VERIFY_EVALUATION_COUNT(m3 = m3 + (m1 * m2.adjoint()), 1); + VERIFY_EVALUATION_COUNT(m3 = m3 - (m1 * m2.adjoint()), 1); - VERIFY_EVALUATION_COUNT( m3 = m3 + (m1 * m2.adjoint()).transpose(), 1); - VERIFY_EVALUATION_COUNT( m3.noalias() = m3 + m1 * m2.transpose(), 0); - VERIFY_EVALUATION_COUNT( m3.noalias() += m3 + m1 * m2.transpose(), 0); - VERIFY_EVALUATION_COUNT( m3.noalias() -= m3 + m1 * m2.transpose(), 0); - VERIFY_EVALUATION_COUNT( m3.noalias() = m3 - m1 * m2.transpose(), 0); - VERIFY_EVALUATION_COUNT( m3.noalias() += m3 - m1 * m2.transpose(), 0); - VERIFY_EVALUATION_COUNT( m3.noalias() -= m3 - m1 * m2.transpose(), 0); + VERIFY_EVALUATION_COUNT(m3 = m3 + (m1 * m2.adjoint()).transpose(), 1); + VERIFY_EVALUATION_COUNT(m3.noalias() = m3 + m1 * m2.transpose(), 0); + VERIFY_EVALUATION_COUNT(m3.noalias() += m3 + m1 * m2.transpose(), 0); + VERIFY_EVALUATION_COUNT(m3.noalias() -= m3 + m1 * m2.transpose(), 0); + VERIFY_EVALUATION_COUNT(m3.noalias() = m3 - m1 * m2.transpose(), 0); + VERIFY_EVALUATION_COUNT(m3.noalias() += m3 - m1 * m2.transpose(), 0); + VERIFY_EVALUATION_COUNT(m3.noalias() -= m3 - m1 * m2.transpose(), 0); - VERIFY_EVALUATION_COUNT( m3.noalias() = s1 * m1 * s2 * m2.adjoint(), 0); - VERIFY_EVALUATION_COUNT( m3.noalias() = s1 * m1 * s2 * (m1*s3+m2*s2).adjoint(), 1); - VERIFY_EVALUATION_COUNT( m3.noalias() = (s1 * m1).adjoint() * s2 * m2, 0); - VERIFY_EVALUATION_COUNT( m3.noalias() += s1 * (-m1*s3).adjoint() * (s2 * m2 * s3), 0); - VERIFY_EVALUATION_COUNT( m3.noalias() -= s1 * (m1.transpose() * m2), 0); + VERIFY_EVALUATION_COUNT(m3.noalias() = s1 * m1 * s2 * m2.adjoint(), 0); + VERIFY_EVALUATION_COUNT(m3.noalias() = s1 * m1 * s2 * (m1 * s3 + m2 * s2).adjoint(), 1); + VERIFY_EVALUATION_COUNT(m3.noalias() = (s1 * m1).adjoint() * s2 * m2, 0); + VERIFY_EVALUATION_COUNT(m3.noalias() += s1 * (-m1 * s3).adjoint() * (s2 * m2 * s3), 0); + VERIFY_EVALUATION_COUNT(m3.noalias() -= s1 * (m1.transpose() * m2), 0); - VERIFY_EVALUATION_COUNT(( m3.block(r0,r0,r1,r1).noalias() += -m1.block(r0,c0,r1,c1) * (s2*m2.block(r0,c0,r1,c1)).adjoint() ), 0); - VERIFY_EVALUATION_COUNT(( m3.block(r0,r0,r1,r1).noalias() -= s1 * m1.block(r0,c0,r1,c1) * m2.block(c0,r0,c1,r1) ), 0); + VERIFY_EVALUATION_COUNT( + (m3.block(r0, r0, r1, r1).noalias() += -m1.block(r0, c0, r1, c1) * (s2 * m2.block(r0, c0, r1, c1)).adjoint()), 0); + VERIFY_EVALUATION_COUNT( + (m3.block(r0, r0, r1, r1).noalias() -= s1 * m1.block(r0, c0, r1, c1) * m2.block(c0, r0, c1, r1)), 0); // NOTE this is because the Block expression is not handled yet by our expression analyser - VERIFY_EVALUATION_COUNT(( m3.block(r0,r0,r1,r1).noalias() = s1 * m1.block(r0,c0,r1,c1) * (s1*m2).block(c0,r0,c1,r1) ), 1); - - VERIFY_EVALUATION_COUNT( m3.noalias() -= (s1 * m1).template triangularView() * m2, 0); - VERIFY_EVALUATION_COUNT( rm3.noalias() = (s1 * m1.adjoint()).template triangularView() * (m2+m2), 1); - VERIFY_EVALUATION_COUNT( rm3.noalias() = (s1 * m1.adjoint()).template triangularView() * m2.adjoint(), 0); - - VERIFY_EVALUATION_COUNT( m3.template triangularView() = (m1 * m2.adjoint()), 0); - VERIFY_EVALUATION_COUNT( m3.template triangularView() -= (m1 * m2.adjoint()), 0); - - // NOTE this is because the blas_traits require innerstride==1 to avoid a temporary, but that doesn't seem to be actually needed for the triangular products - VERIFY_EVALUATION_COUNT( rm3.col(c0).noalias() = (s1 * m1.adjoint()).template triangularView() * (s2*m2.row(c0)).adjoint(), 1); - - VERIFY_EVALUATION_COUNT( m1.template triangularView().solveInPlace(m3), 0); - VERIFY_EVALUATION_COUNT( m1.adjoint().template triangularView().solveInPlace(m3.transpose()), 0); - - VERIFY_EVALUATION_COUNT( m3.noalias() -= (s1 * m1).adjoint().template selfadjointView() * (-m2*s3).adjoint(), 0); - VERIFY_EVALUATION_COUNT( m3.noalias() = s2 * m2.adjoint() * (s1 * m1.adjoint()).template selfadjointView(), 0); - VERIFY_EVALUATION_COUNT( rm3.noalias() = (s1 * m1.adjoint()).template selfadjointView() * m2.adjoint(), 0); - - // NOTE this is because the blas_traits require innerstride==1 to avoid a temporary, but that doesn't seem to be actually needed for the triangular products - VERIFY_EVALUATION_COUNT( m3.col(c0).noalias() = (s1 * m1).adjoint().template selfadjointView() * (-m2.row(c0)*s3).adjoint(), 1); - VERIFY_EVALUATION_COUNT( m3.col(c0).noalias() -= (s1 * m1).adjoint().template selfadjointView() * (-m2.row(c0)*s3).adjoint(), 1); - - VERIFY_EVALUATION_COUNT( m3.block(r0,c0,r1,c1).noalias() += m1.block(r0,r0,r1,r1).template selfadjointView() * (s1*m2.block(r0,c0,r1,c1)), 0); - VERIFY_EVALUATION_COUNT( m3.block(r0,c0,r1,c1).noalias() = m1.block(r0,r0,r1,r1).template selfadjointView() * m2.block(r0,c0,r1,c1), 0); - - VERIFY_EVALUATION_COUNT( m3.template selfadjointView().rankUpdate(m2.adjoint()), 0); - - // Here we will get 1 temporary for each resize operation of the lhs operator; resize(r1,c1) would lead to zero temporaries - m3.resize(1,1); - VERIFY_EVALUATION_COUNT( m3.noalias() = m1.block(r0,r0,r1,r1).template selfadjointView() * m2.block(r0,c0,r1,c1), 1); - m3.resize(1,1); - VERIFY_EVALUATION_COUNT( m3.noalias() = m1.block(r0,r0,r1,r1).template triangularView() * m2.block(r0,c0,r1,c1), 1); + VERIFY_EVALUATION_COUNT( + (m3.block(r0, r0, r1, r1).noalias() = s1 * m1.block(r0, c0, r1, c1) * (s1 * m2).block(c0, r0, c1, r1)), 1); + + VERIFY_EVALUATION_COUNT(m3.noalias() -= (s1 * m1).template triangularView() * m2, 0); + VERIFY_EVALUATION_COUNT(rm3.noalias() = (s1 * m1.adjoint()).template triangularView() * (m2 + m2), 1); + VERIFY_EVALUATION_COUNT(rm3.noalias() = (s1 * m1.adjoint()).template triangularView() * m2.adjoint(), 0); + + VERIFY_EVALUATION_COUNT(m3.template triangularView() = (m1 * m2.adjoint()), 0); + VERIFY_EVALUATION_COUNT(m3.template triangularView() -= (m1 * m2.adjoint()), 0); + + // NOTE this is because the blas_traits require innerstride==1 to avoid a temporary, but that doesn't seem to be + // actually needed for the triangular products + VERIFY_EVALUATION_COUNT( + rm3.col(c0).noalias() = (s1 * m1.adjoint()).template triangularView() * (s2 * m2.row(c0)).adjoint(), 1); + + VERIFY_EVALUATION_COUNT(m1.template triangularView().solveInPlace(m3), 0); + VERIFY_EVALUATION_COUNT(m1.adjoint().template triangularView().solveInPlace(m3.transpose()), 0); + + VERIFY_EVALUATION_COUNT( + m3.noalias() -= (s1 * m1).adjoint().template selfadjointView() * (-m2 * s3).adjoint(), 0); + VERIFY_EVALUATION_COUNT(m3.noalias() = s2 * m2.adjoint() * (s1 * m1.adjoint()).template selfadjointView(), 0); + VERIFY_EVALUATION_COUNT(rm3.noalias() = (s1 * m1.adjoint()).template selfadjointView() * m2.adjoint(), 0); + + // NOTE this is because the blas_traits require innerstride==1 to avoid a temporary, but that doesn't seem to be + // actually needed for the triangular products + VERIFY_EVALUATION_COUNT( + m3.col(c0).noalias() = (s1 * m1).adjoint().template selfadjointView() * (-m2.row(c0) * s3).adjoint(), 1); + VERIFY_EVALUATION_COUNT( + m3.col(c0).noalias() -= (s1 * m1).adjoint().template selfadjointView() * (-m2.row(c0) * s3).adjoint(), 1); + + VERIFY_EVALUATION_COUNT(m3.block(r0, c0, r1, c1).noalias() += + m1.block(r0, r0, r1, r1).template selfadjointView() * (s1 * m2.block(r0, c0, r1, c1)), + 0); + VERIFY_EVALUATION_COUNT(m3.block(r0, c0, r1, c1).noalias() = + m1.block(r0, r0, r1, r1).template selfadjointView() * m2.block(r0, c0, r1, c1), + 0); + + VERIFY_EVALUATION_COUNT(m3.template selfadjointView().rankUpdate(m2.adjoint()), 0); + + // Here we will get 1 temporary for each resize operation of the lhs operator; resize(r1,c1) would lead to zero + // temporaries + m3.resize(1, 1); + VERIFY_EVALUATION_COUNT( + m3.noalias() = m1.block(r0, r0, r1, r1).template selfadjointView() * m2.block(r0, c0, r1, c1), 1); + m3.resize(1, 1); + VERIFY_EVALUATION_COUNT( + m3.noalias() = m1.block(r0, r0, r1, r1).template triangularView() * m2.block(r0, c0, r1, c1), 1); // Zero temporaries for lazy products ... - VERIFY_EVALUATION_COUNT( Scalar tmp = 0; tmp += Scalar(RealScalar(1)) / (m3.transpose().lazyProduct(m3)).diagonal().sum(), 0 ); + VERIFY_EVALUATION_COUNT(Scalar tmp = 0; + tmp += Scalar(RealScalar(1)) / (m3.transpose().lazyProduct(m3)).diagonal().sum(), 0); // ... and even no temporary for even deeply (>=2) nested products - VERIFY_EVALUATION_COUNT( Scalar tmp = 0; tmp += Scalar(RealScalar(1)) / (m3.transpose() * m3).diagonal().sum(), 0 ); - VERIFY_EVALUATION_COUNT( Scalar tmp = 0; tmp += Scalar(RealScalar(1)) / (m3.transpose() * m3).diagonal().array().abs().sum(), 0 ); + VERIFY_EVALUATION_COUNT(Scalar tmp = 0; tmp += Scalar(RealScalar(1)) / (m3.transpose() * m3).diagonal().sum(), 0); + VERIFY_EVALUATION_COUNT(Scalar tmp = 0; + tmp += Scalar(RealScalar(1)) / (m3.transpose() * m3).diagonal().array().abs().sum(), 0); // Zero temporaries for ... CoeffBasedProductMode - VERIFY_EVALUATION_COUNT( m3.col(0).template head<5>() * m3.col(0).transpose() + m3.col(0).template head<5>() * m3.col(0).transpose(), 0 ); + VERIFY_EVALUATION_COUNT( + m3.col(0).template head<5>() * m3.col(0).transpose() + m3.col(0).template head<5>() * m3.col(0).transpose(), 0); // Check matrix * vectors - VERIFY_EVALUATION_COUNT( cvres.noalias() = m1 * cv1, 0 ); - VERIFY_EVALUATION_COUNT( cvres.noalias() -= m1 * cv1, 0 ); - VERIFY_EVALUATION_COUNT( cvres.noalias() -= m1 * m2.col(0), 0 ); - VERIFY_EVALUATION_COUNT( cvres.noalias() -= m1 * rv1.adjoint(), 0 ); - VERIFY_EVALUATION_COUNT( cvres.noalias() -= m1 * m2.row(0).transpose(), 0 ); + VERIFY_EVALUATION_COUNT(cvres.noalias() = m1 * cv1, 0); + VERIFY_EVALUATION_COUNT(cvres.noalias() -= m1 * cv1, 0); + VERIFY_EVALUATION_COUNT(cvres.noalias() -= m1 * m2.col(0), 0); + VERIFY_EVALUATION_COUNT(cvres.noalias() -= m1 * rv1.adjoint(), 0); + VERIFY_EVALUATION_COUNT(cvres.noalias() -= m1 * m2.row(0).transpose(), 0); - VERIFY_EVALUATION_COUNT( cvres.noalias() = (m1+m1) * cv1, 0 ); - VERIFY_EVALUATION_COUNT( cvres.noalias() = (rm3+rm3) * cv1, 0 ); - VERIFY_EVALUATION_COUNT( cvres.noalias() = (m1+m1) * (m1*cv1), 1 ); - VERIFY_EVALUATION_COUNT( cvres.noalias() = (rm3+rm3) * (m1*cv1), 1 ); + VERIFY_EVALUATION_COUNT(cvres.noalias() = (m1 + m1) * cv1, 0); + VERIFY_EVALUATION_COUNT(cvres.noalias() = (rm3 + rm3) * cv1, 0); + VERIFY_EVALUATION_COUNT(cvres.noalias() = (m1 + m1) * (m1 * cv1), 1); + VERIFY_EVALUATION_COUNT(cvres.noalias() = (rm3 + rm3) * (m1 * cv1), 1); // Check outer products m3 = cv1 * rv1; - VERIFY_EVALUATION_COUNT( m3.noalias() = cv1 * rv1, 0 ); - VERIFY_EVALUATION_COUNT( m3.noalias() = (cv1+cv1) * (rv1+rv1), 1 ); - VERIFY_EVALUATION_COUNT( m3.noalias() = (m1*cv1) * (rv1), 1 ); - VERIFY_EVALUATION_COUNT( m3.noalias() += (m1*cv1) * (rv1), 1 ); - VERIFY_EVALUATION_COUNT( rm3.noalias() = (cv1) * (rv1 * m1), 1 ); - VERIFY_EVALUATION_COUNT( rm3.noalias() -= (cv1) * (rv1 * m1), 1 ); - VERIFY_EVALUATION_COUNT( rm3.noalias() = (m1*cv1) * (rv1 * m1), 2 ); - VERIFY_EVALUATION_COUNT( rm3.noalias() += (m1*cv1) * (rv1 * m1), 2 ); + VERIFY_EVALUATION_COUNT(m3.noalias() = cv1 * rv1, 0); + VERIFY_EVALUATION_COUNT(m3.noalias() = (cv1 + cv1) * (rv1 + rv1), 1); + VERIFY_EVALUATION_COUNT(m3.noalias() = (m1 * cv1) * (rv1), 1); + VERIFY_EVALUATION_COUNT(m3.noalias() += (m1 * cv1) * (rv1), 1); + VERIFY_EVALUATION_COUNT(rm3.noalias() = (cv1) * (rv1 * m1), 1); + VERIFY_EVALUATION_COUNT(rm3.noalias() -= (cv1) * (rv1 * m1), 1); + VERIFY_EVALUATION_COUNT(rm3.noalias() = (m1 * cv1) * (rv1 * m1), 2); + VERIFY_EVALUATION_COUNT(rm3.noalias() += (m1 * cv1) * (rv1 * m1), 2); // Check nested products - VERIFY_EVALUATION_COUNT( cvres.noalias() = m1.adjoint() * m1 * cv1, 1 ); - VERIFY_EVALUATION_COUNT( rvres.noalias() = rv1 * (m1 * m2.adjoint()), 1 ); + VERIFY_EVALUATION_COUNT(cvres.noalias() = m1.adjoint() * m1 * cv1, 1); + VERIFY_EVALUATION_COUNT(rvres.noalias() = rv1 * (m1 * m2.adjoint()), 1); } void test_product_notemporary() { int s; - for(int i = 0; i < g_repeat; i++) { - s = internal::random(16,EIGEN_TEST_MAX_SIZE); - CALL_SUBTEST_1( product_notemporary(MatrixXf(s, s)) ); - CALL_SUBTEST_2( product_notemporary(MatrixXd(s, s)) ); + for (int i = 0; i < g_repeat; i++) { + s = internal::random(16, EIGEN_TEST_MAX_SIZE); + CALL_SUBTEST_1(product_notemporary(MatrixXf(s, s))); + CALL_SUBTEST_2(product_notemporary(MatrixXd(s, s))); TEST_SET_BUT_UNUSED_VARIABLE(s) - - s = internal::random(16,EIGEN_TEST_MAX_SIZE/2); - CALL_SUBTEST_3( product_notemporary(MatrixXcf(s,s)) ); - CALL_SUBTEST_4( product_notemporary(MatrixXcd(s,s)) ); + + s = internal::random(16, EIGEN_TEST_MAX_SIZE / 2); + CALL_SUBTEST_3(product_notemporary(MatrixXcf(s, s))); + CALL_SUBTEST_4(product_notemporary(MatrixXcd(s, s))); TEST_SET_BUT_UNUSED_VARIABLE(s) } } diff --git a/filmulator-gui/core/nlmeans/eigen/test/product_selfadjoint.cpp b/filmulator-gui/core/nlmeans/eigen/test/product_selfadjoint.cpp index 88d68391..6c5936ad 100644 --- a/filmulator-gui/core/nlmeans/eigen/test/product_selfadjoint.cpp +++ b/filmulator-gui/core/nlmeans/eigen/test/product_selfadjoint.cpp @@ -9,7 +9,7 @@ #include "main.h" -template void product_selfadjoint(const MatrixType& m) +template void product_selfadjoint(const MatrixType &m) { typedef typename MatrixType::Scalar Scalar; typedef Matrix VectorType; @@ -20,41 +20,43 @@ template void product_selfadjoint(const MatrixType& m) Index rows = m.rows(); Index cols = m.cols(); - MatrixType m1 = MatrixType::Random(rows, cols), - m2 = MatrixType::Random(rows, cols), - m3; - VectorType v1 = VectorType::Random(rows), - v2 = VectorType::Random(rows), - v3(rows); - RowVectorType r1 = RowVectorType::Random(rows), - r2 = RowVectorType::Random(rows); - RhsMatrixType m4 = RhsMatrixType::Random(rows,10); + MatrixType m1 = MatrixType::Random(rows, cols), m2 = MatrixType::Random(rows, cols), m3; + VectorType v1 = VectorType::Random(rows), v2 = VectorType::Random(rows), v3(rows); + RowVectorType r1 = RowVectorType::Random(rows), r2 = RowVectorType::Random(rows); + RhsMatrixType m4 = RhsMatrixType::Random(rows, 10); - Scalar s1 = internal::random(), - s2 = internal::random(), - s3 = internal::random(); + Scalar s1 = internal::random(), s2 = internal::random(), s3 = internal::random(); m1 = (m1.adjoint() + m1).eval(); // rank2 update m2 = m1.template triangularView(); - m2.template selfadjointView().rankUpdate(v1,v2); - VERIFY_IS_APPROX(m2, (m1 + v1 * v2.adjoint()+ v2 * v1.adjoint()).template triangularView().toDenseMatrix()); + m2.template selfadjointView().rankUpdate(v1, v2); + VERIFY_IS_APPROX(m2, (m1 + v1 * v2.adjoint() + v2 * v1.adjoint()).template triangularView().toDenseMatrix()); m2 = m1.template triangularView(); - m2.template selfadjointView().rankUpdate(-v1,s2*v2,s3); - VERIFY_IS_APPROX(m2, (m1 + (s3*(-v1)*(s2*v2).adjoint()+numext::conj(s3)*(s2*v2)*(-v1).adjoint())).template triangularView().toDenseMatrix()); + m2.template selfadjointView().rankUpdate(-v1, s2 * v2, s3); + VERIFY_IS_APPROX(m2, + (m1 + (s3 * (-v1) * (s2 * v2).adjoint() + numext::conj(s3) * (s2 * v2) * (-v1).adjoint())) + .template triangularView() + .toDenseMatrix()); m2 = m1.template triangularView(); - m2.template selfadjointView().rankUpdate(-s2*r1.adjoint(),r2.adjoint()*s3,s1); - VERIFY_IS_APPROX(m2, (m1 + s1*(-s2*r1.adjoint())*(r2.adjoint()*s3).adjoint() + numext::conj(s1)*(r2.adjoint()*s3) * (-s2*r1.adjoint()).adjoint()).template triangularView().toDenseMatrix()); + m2.template selfadjointView().rankUpdate(-s2 * r1.adjoint(), r2.adjoint() * s3, s1); + VERIFY_IS_APPROX(m2, + (m1 + s1 * (-s2 * r1.adjoint()) * (r2.adjoint() * s3).adjoint() + + numext::conj(s1) * (r2.adjoint() * s3) * (-s2 * r1.adjoint()).adjoint()) + .template triangularView() + .toDenseMatrix()); - if (rows>1) - { + if (rows > 1) { m2 = m1.template triangularView(); - m2.block(1,1,rows-1,cols-1).template selfadjointView().rankUpdate(v1.tail(rows-1),v2.head(cols-1)); + m2.block(1, 1, rows - 1, cols - 1) + .template selfadjointView() + .rankUpdate(v1.tail(rows - 1), v2.head(cols - 1)); m3 = m1; - m3.block(1,1,rows-1,cols-1) += v1.tail(rows-1) * v2.head(cols-1).adjoint()+ v2.head(cols-1) * v1.tail(rows-1).adjoint(); + m3.block(1, 1, rows - 1, cols - 1) += + v1.tail(rows - 1) * v2.head(cols - 1).adjoint() + v2.head(cols - 1) * v1.tail(rows - 1).adjoint(); VERIFY_IS_APPROX(m2, m3.template triangularView().toDenseMatrix()); } } @@ -62,25 +64,25 @@ template void product_selfadjoint(const MatrixType& m) void test_product_selfadjoint() { int s = 0; - for(int i = 0; i < g_repeat ; i++) { - CALL_SUBTEST_1( product_selfadjoint(Matrix()) ); - CALL_SUBTEST_2( product_selfadjoint(Matrix()) ); - CALL_SUBTEST_3( product_selfadjoint(Matrix3d()) ); - - s = internal::random(1,EIGEN_TEST_MAX_SIZE/2); - CALL_SUBTEST_4( product_selfadjoint(MatrixXcf(s, s)) ); + for (int i = 0; i < g_repeat; i++) { + CALL_SUBTEST_1(product_selfadjoint(Matrix())); + CALL_SUBTEST_2(product_selfadjoint(Matrix())); + CALL_SUBTEST_3(product_selfadjoint(Matrix3d())); + + s = internal::random(1, EIGEN_TEST_MAX_SIZE / 2); + CALL_SUBTEST_4(product_selfadjoint(MatrixXcf(s, s))); TEST_SET_BUT_UNUSED_VARIABLE(s) - - s = internal::random(1,EIGEN_TEST_MAX_SIZE/2); - CALL_SUBTEST_5( product_selfadjoint(MatrixXcd(s,s)) ); + + s = internal::random(1, EIGEN_TEST_MAX_SIZE / 2); + CALL_SUBTEST_5(product_selfadjoint(MatrixXcd(s, s))); TEST_SET_BUT_UNUSED_VARIABLE(s) - - s = internal::random(1,EIGEN_TEST_MAX_SIZE); - CALL_SUBTEST_6( product_selfadjoint(MatrixXd(s,s)) ); + + s = internal::random(1, EIGEN_TEST_MAX_SIZE); + CALL_SUBTEST_6(product_selfadjoint(MatrixXd(s, s))); TEST_SET_BUT_UNUSED_VARIABLE(s) - - s = internal::random(1,EIGEN_TEST_MAX_SIZE); - CALL_SUBTEST_7( product_selfadjoint(Matrix(s,s)) ); + + s = internal::random(1, EIGEN_TEST_MAX_SIZE); + CALL_SUBTEST_7(product_selfadjoint(Matrix(s, s))); TEST_SET_BUT_UNUSED_VARIABLE(s) } } diff --git a/filmulator-gui/core/nlmeans/eigen/test/product_small.cpp b/filmulator-gui/core/nlmeans/eigen/test/product_small.cpp index fdfdd9f6..3f06ff55 100644 --- a/filmulator-gui/core/nlmeans/eigen/test/product_small.cpp +++ b/filmulator-gui/core/nlmeans/eigen/test/product_small.cpp @@ -12,269 +12,267 @@ #include // regression test for bug 447 -template -void product1x1() +template void product1x1() { - Matrix matAstatic; - Matrix matBstatic; + Matrix matAstatic; + Matrix matBstatic; matAstatic.setRandom(); matBstatic.setRandom(); - VERIFY_IS_APPROX( (matAstatic * matBstatic).coeff(0,0), - matAstatic.cwiseProduct(matBstatic.transpose()).sum() ); + VERIFY_IS_APPROX((matAstatic * matBstatic).coeff(0, 0), matAstatic.cwiseProduct(matBstatic.transpose()).sum()); - MatrixXf matAdynamic(1,3); - MatrixXf matBdynamic(3,1); + MatrixXf matAdynamic(1, 3); + MatrixXf matBdynamic(3, 1); matAdynamic.setRandom(); matBdynamic.setRandom(); - VERIFY_IS_APPROX( (matAdynamic * matBdynamic).coeff(0,0), - matAdynamic.cwiseProduct(matBdynamic.transpose()).sum() ); + VERIFY_IS_APPROX((matAdynamic * matBdynamic).coeff(0, 0), matAdynamic.cwiseProduct(matBdynamic.transpose()).sum()); } -template -const TC& ref_prod(TC &C, const TA &A, const TB &B) +template const TC &ref_prod(TC &C, const TA &A, const TB &B) { - for(Index i=0;i -typename internal::enable_if::type -test_lazy_single(int rows, int cols, int depth) +typename internal::enable_if< + !((Rows == 1 && Depth != 1 && OA == ColMajor) || (Depth == 1 && Rows != 1 && OA == RowMajor) + || (Cols == 1 && Depth != 1 && OB == RowMajor) || (Depth == 1 && Cols != 1 && OB == ColMajor) + || (Rows == 1 && Cols != 1 && OC == ColMajor) || (Cols == 1 && Rows != 1 && OC == RowMajor)), + void>::type + test_lazy_single(int rows, int cols, int depth) { - Matrix A(rows,depth); A.setRandom(); - Matrix B(depth,cols); B.setRandom(); - Matrix C(rows,cols); C.setRandom(); - Matrix D(C); - VERIFY_IS_APPROX(C+=A.lazyProduct(B), ref_prod(D,A,B)); + Matrix A(rows, depth); + A.setRandom(); + Matrix B(depth, cols); + B.setRandom(); + Matrix C(rows, cols); + C.setRandom(); + Matrix D(C); + VERIFY_IS_APPROX(C += A.lazyProduct(B), ref_prod(D, A, B)); } template -typename internal::enable_if< ( (Rows ==1&&Depth!=1&&OA==ColMajor) - || (Depth==1&&Rows !=1&&OA==RowMajor) - || (Cols ==1&&Depth!=1&&OB==RowMajor) - || (Depth==1&&Cols !=1&&OB==ColMajor) - || (Rows ==1&&Cols !=1&&OC==ColMajor) - || (Cols ==1&&Rows !=1&&OC==RowMajor)),void>::type -test_lazy_single(int, int, int) -{ -} +typename internal::enable_if< + ((Rows == 1 && Depth != 1 && OA == ColMajor) || (Depth == 1 && Rows != 1 && OA == RowMajor) + || (Cols == 1 && Depth != 1 && OB == RowMajor) || (Depth == 1 && Cols != 1 && OB == ColMajor) + || (Rows == 1 && Cols != 1 && OC == ColMajor) || (Cols == 1 && Rows != 1 && OC == RowMajor)), + void>::type + test_lazy_single(int, int, int) +{} template -void test_lazy_all_layout(int rows=Rows, int cols=Cols, int depth=Depth) +void test_lazy_all_layout(int rows = Rows, int cols = Cols, int depth = Depth) { - CALL_SUBTEST(( test_lazy_single(rows,cols,depth) )); - CALL_SUBTEST(( test_lazy_single(rows,cols,depth) )); - CALL_SUBTEST(( test_lazy_single(rows,cols,depth) )); - CALL_SUBTEST(( test_lazy_single(rows,cols,depth) )); - CALL_SUBTEST(( test_lazy_single(rows,cols,depth) )); - CALL_SUBTEST(( test_lazy_single(rows,cols,depth) )); - CALL_SUBTEST(( test_lazy_single(rows,cols,depth) )); - CALL_SUBTEST(( test_lazy_single(rows,cols,depth) )); + CALL_SUBTEST((test_lazy_single(rows, cols, depth))); + CALL_SUBTEST((test_lazy_single(rows, cols, depth))); + CALL_SUBTEST((test_lazy_single(rows, cols, depth))); + CALL_SUBTEST((test_lazy_single(rows, cols, depth))); + CALL_SUBTEST((test_lazy_single(rows, cols, depth))); + CALL_SUBTEST((test_lazy_single(rows, cols, depth))); + CALL_SUBTEST((test_lazy_single(rows, cols, depth))); + CALL_SUBTEST((test_lazy_single(rows, cols, depth))); } -template -void test_lazy_l1() +template void test_lazy_l1() { - int rows = internal::random(1,12); - int cols = internal::random(1,12); - int depth = internal::random(1,12); + int rows = internal::random(1, 12); + int cols = internal::random(1, 12); + int depth = internal::random(1, 12); // Inner - CALL_SUBTEST(( test_lazy_all_layout() )); - CALL_SUBTEST(( test_lazy_all_layout() )); - CALL_SUBTEST(( test_lazy_all_layout() )); - CALL_SUBTEST(( test_lazy_all_layout() )); - CALL_SUBTEST(( test_lazy_all_layout() )); - CALL_SUBTEST(( test_lazy_all_layout(1,1,depth) )); + CALL_SUBTEST((test_lazy_all_layout())); + CALL_SUBTEST((test_lazy_all_layout())); + CALL_SUBTEST((test_lazy_all_layout())); + CALL_SUBTEST((test_lazy_all_layout())); + CALL_SUBTEST((test_lazy_all_layout())); + CALL_SUBTEST((test_lazy_all_layout(1, 1, depth))); // Outer - CALL_SUBTEST(( test_lazy_all_layout() )); - CALL_SUBTEST(( test_lazy_all_layout() )); - CALL_SUBTEST(( test_lazy_all_layout() )); - CALL_SUBTEST(( test_lazy_all_layout() )); - CALL_SUBTEST(( test_lazy_all_layout() )); - CALL_SUBTEST(( test_lazy_all_layout() )); - CALL_SUBTEST(( test_lazy_all_layout(4,cols) )); - CALL_SUBTEST(( test_lazy_all_layout(7,cols) )); - CALL_SUBTEST(( test_lazy_all_layout(rows) )); - CALL_SUBTEST(( test_lazy_all_layout(rows) )); - CALL_SUBTEST(( test_lazy_all_layout(rows,cols) )); + CALL_SUBTEST((test_lazy_all_layout())); + CALL_SUBTEST((test_lazy_all_layout())); + CALL_SUBTEST((test_lazy_all_layout())); + CALL_SUBTEST((test_lazy_all_layout())); + CALL_SUBTEST((test_lazy_all_layout())); + CALL_SUBTEST((test_lazy_all_layout())); + CALL_SUBTEST((test_lazy_all_layout(4, cols))); + CALL_SUBTEST((test_lazy_all_layout(7, cols))); + CALL_SUBTEST((test_lazy_all_layout(rows))); + CALL_SUBTEST((test_lazy_all_layout(rows))); + CALL_SUBTEST((test_lazy_all_layout(rows, cols))); } -template -void test_lazy_l2() +template void test_lazy_l2() { - int rows = internal::random(1,12); - int cols = internal::random(1,12); - int depth = internal::random(1,12); + int rows = internal::random(1, 12); + int cols = internal::random(1, 12); + int depth = internal::random(1, 12); // mat-vec - CALL_SUBTEST(( test_lazy_all_layout() )); - CALL_SUBTEST(( test_lazy_all_layout() )); - CALL_SUBTEST(( test_lazy_all_layout() )); - CALL_SUBTEST(( test_lazy_all_layout() )); - CALL_SUBTEST(( test_lazy_all_layout() )); - CALL_SUBTEST(( test_lazy_all_layout() )); - CALL_SUBTEST(( test_lazy_all_layout() )); - CALL_SUBTEST(( test_lazy_all_layout() )); - CALL_SUBTEST(( test_lazy_all_layout() )); - CALL_SUBTEST(( test_lazy_all_layout(rows) )); - CALL_SUBTEST(( test_lazy_all_layout(4,1,depth) )); - CALL_SUBTEST(( test_lazy_all_layout(rows,1,depth) )); + CALL_SUBTEST((test_lazy_all_layout())); + CALL_SUBTEST((test_lazy_all_layout())); + CALL_SUBTEST((test_lazy_all_layout())); + CALL_SUBTEST((test_lazy_all_layout())); + CALL_SUBTEST((test_lazy_all_layout())); + CALL_SUBTEST((test_lazy_all_layout())); + CALL_SUBTEST((test_lazy_all_layout())); + CALL_SUBTEST((test_lazy_all_layout())); + CALL_SUBTEST((test_lazy_all_layout())); + CALL_SUBTEST((test_lazy_all_layout(rows))); + CALL_SUBTEST((test_lazy_all_layout(4, 1, depth))); + CALL_SUBTEST((test_lazy_all_layout(rows, 1, depth))); // vec-mat - CALL_SUBTEST(( test_lazy_all_layout() )); - CALL_SUBTEST(( test_lazy_all_layout() )); - CALL_SUBTEST(( test_lazy_all_layout() )); - CALL_SUBTEST(( test_lazy_all_layout() )); - CALL_SUBTEST(( test_lazy_all_layout() )); - CALL_SUBTEST(( test_lazy_all_layout() )); - CALL_SUBTEST(( test_lazy_all_layout() )); - CALL_SUBTEST(( test_lazy_all_layout() )); - CALL_SUBTEST(( test_lazy_all_layout() )); - CALL_SUBTEST(( test_lazy_all_layout(1,cols) )); - CALL_SUBTEST(( test_lazy_all_layout(1,4,depth) )); - CALL_SUBTEST(( test_lazy_all_layout(1,cols,depth) )); + CALL_SUBTEST((test_lazy_all_layout())); + CALL_SUBTEST((test_lazy_all_layout())); + CALL_SUBTEST((test_lazy_all_layout())); + CALL_SUBTEST((test_lazy_all_layout())); + CALL_SUBTEST((test_lazy_all_layout())); + CALL_SUBTEST((test_lazy_all_layout())); + CALL_SUBTEST((test_lazy_all_layout())); + CALL_SUBTEST((test_lazy_all_layout())); + CALL_SUBTEST((test_lazy_all_layout())); + CALL_SUBTEST((test_lazy_all_layout(1, cols))); + CALL_SUBTEST((test_lazy_all_layout(1, 4, depth))); + CALL_SUBTEST((test_lazy_all_layout(1, cols, depth))); } -template -void test_lazy_l3() +template void test_lazy_l3() { - int rows = internal::random(1,12); - int cols = internal::random(1,12); - int depth = internal::random(1,12); + int rows = internal::random(1, 12); + int cols = internal::random(1, 12); + int depth = internal::random(1, 12); // mat-mat - CALL_SUBTEST(( test_lazy_all_layout() )); - CALL_SUBTEST(( test_lazy_all_layout() )); - CALL_SUBTEST(( test_lazy_all_layout() )); - CALL_SUBTEST(( test_lazy_all_layout() )); - CALL_SUBTEST(( test_lazy_all_layout() )); - CALL_SUBTEST(( test_lazy_all_layout() )); - CALL_SUBTEST(( test_lazy_all_layout() )); - CALL_SUBTEST(( test_lazy_all_layout() )); - CALL_SUBTEST(( test_lazy_all_layout() )); - CALL_SUBTEST(( test_lazy_all_layout(rows) )); - CALL_SUBTEST(( test_lazy_all_layout(4,3,depth) )); - CALL_SUBTEST(( test_lazy_all_layout(rows,6,depth) )); - CALL_SUBTEST(( test_lazy_all_layout() )); - CALL_SUBTEST(( test_lazy_all_layout() )); - CALL_SUBTEST(( test_lazy_all_layout() )); - CALL_SUBTEST(( test_lazy_all_layout() )); - CALL_SUBTEST(( test_lazy_all_layout() )); - CALL_SUBTEST(( test_lazy_all_layout() )); - CALL_SUBTEST(( test_lazy_all_layout() )); - CALL_SUBTEST(( test_lazy_all_layout() )); - CALL_SUBTEST(( test_lazy_all_layout() )); - CALL_SUBTEST(( test_lazy_all_layout(8,cols) )); - CALL_SUBTEST(( test_lazy_all_layout(3,4,depth) )); - CALL_SUBTEST(( test_lazy_all_layout(4,cols,depth) )); + CALL_SUBTEST((test_lazy_all_layout())); + CALL_SUBTEST((test_lazy_all_layout())); + CALL_SUBTEST((test_lazy_all_layout())); + CALL_SUBTEST((test_lazy_all_layout())); + CALL_SUBTEST((test_lazy_all_layout())); + CALL_SUBTEST((test_lazy_all_layout())); + CALL_SUBTEST((test_lazy_all_layout())); + CALL_SUBTEST((test_lazy_all_layout())); + CALL_SUBTEST((test_lazy_all_layout())); + CALL_SUBTEST((test_lazy_all_layout(rows))); + CALL_SUBTEST((test_lazy_all_layout(4, 3, depth))); + CALL_SUBTEST((test_lazy_all_layout(rows, 6, depth))); + CALL_SUBTEST((test_lazy_all_layout())); + CALL_SUBTEST((test_lazy_all_layout())); + CALL_SUBTEST((test_lazy_all_layout())); + CALL_SUBTEST((test_lazy_all_layout())); + CALL_SUBTEST((test_lazy_all_layout())); + CALL_SUBTEST((test_lazy_all_layout())); + CALL_SUBTEST((test_lazy_all_layout())); + CALL_SUBTEST((test_lazy_all_layout())); + CALL_SUBTEST((test_lazy_all_layout())); + CALL_SUBTEST((test_lazy_all_layout(8, cols))); + CALL_SUBTEST((test_lazy_all_layout(3, 4, depth))); + CALL_SUBTEST((test_lazy_all_layout(4, cols, depth))); } -template -void test_linear_but_not_vectorizable() +template void test_linear_but_not_vectorizable() { // Check tricky cases for which the result of the product is a vector and thus must exhibit the LinearBit flag, // but is not vectorizable along the linear dimension. - Index n = N==Dynamic ? internal::random(1,32) : N; - Index m = M==Dynamic ? internal::random(1,32) : M; - Index k = K==Dynamic ? internal::random(1,32) : K; + Index n = N == Dynamic ? internal::random(1, 32) : N; + Index m = M == Dynamic ? internal::random(1, 32) : M; + Index k = K == Dynamic ? internal::random(1, 32) : K; { - Matrix A; A.setRandom(n,m+1); - Matrix B; B.setRandom(m*2,k); - Matrix C; - Matrix R; + Matrix A; + A.setRandom(n, m + 1); + Matrix B; + B.setRandom(m * 2, k); + Matrix C; + Matrix R; - C.noalias() = A.template topLeftCorner<1,M>() * (B.template topRows()+B.template bottomRows()); - R.noalias() = A.template topLeftCorner<1,M>() * (B.template topRows()+B.template bottomRows()).eval(); - VERIFY_IS_APPROX(C,R); + C.noalias() = A.template topLeftCorner<1, M>() * (B.template topRows() + B.template bottomRows()); + R.noalias() = A.template topLeftCorner<1, M>() * (B.template topRows() + B.template bottomRows()).eval(); + VERIFY_IS_APPROX(C, R); } { - Matrix A; A.setRandom(m+1,n); - Matrix B; B.setRandom(k,m*2); - Matrix C; - Matrix R; + Matrix A; + A.setRandom(m + 1, n); + Matrix B; + B.setRandom(k, m * 2); + Matrix C; + Matrix R; - C.noalias() = (B.template leftCols()+B.template rightCols()) * A.template topLeftCorner(); - R.noalias() = (B.template leftCols()+B.template rightCols()).eval() * A.template topLeftCorner(); - VERIFY_IS_APPROX(C,R); + C.noalias() = (B.template leftCols() + B.template rightCols()) * A.template topLeftCorner(); + R.noalias() = (B.template leftCols() + B.template rightCols()).eval() * A.template topLeftCorner(); + VERIFY_IS_APPROX(C, R); } } -template -void bug_1311() +template void bug_1311() { - Matrix< double, Rows, 2 > A; A.setRandom(); - Vector2d b = Vector2d::Random() ; - Matrix res; + Matrix A; + A.setRandom(); + Vector2d b = Vector2d::Random(); + Matrix res; res.noalias() = 1. * (A * b); - VERIFY_IS_APPROX(res, A*b); - res.noalias() = 1.*A * b; - VERIFY_IS_APPROX(res, A*b); - res.noalias() = (1.*A).lazyProduct(b); - VERIFY_IS_APPROX(res, A*b); - res.noalias() = (1.*A).lazyProduct(1.*b); - VERIFY_IS_APPROX(res, A*b); - res.noalias() = (A).lazyProduct(1.*b); - VERIFY_IS_APPROX(res, A*b); + VERIFY_IS_APPROX(res, A * b); + res.noalias() = 1. * A * b; + VERIFY_IS_APPROX(res, A * b); + res.noalias() = (1. * A).lazyProduct(b); + VERIFY_IS_APPROX(res, A * b); + res.noalias() = (1. * A).lazyProduct(1. * b); + VERIFY_IS_APPROX(res, A * b); + res.noalias() = (A).lazyProduct(1. * b); + VERIFY_IS_APPROX(res, A * b); } void test_product_small() { - for(int i = 0; i < g_repeat; i++) { - CALL_SUBTEST_1( product(Matrix()) ); - CALL_SUBTEST_2( product(Matrix()) ); - CALL_SUBTEST_8( product(Matrix()) ); - CALL_SUBTEST_3( product(Matrix3d()) ); - CALL_SUBTEST_4( product(Matrix4d()) ); - CALL_SUBTEST_5( product(Matrix4f()) ); - CALL_SUBTEST_6( product1x1<0>() ); + for (int i = 0; i < g_repeat; i++) { + CALL_SUBTEST_1(product(Matrix())); + CALL_SUBTEST_2(product(Matrix())); + CALL_SUBTEST_8(product(Matrix())); + CALL_SUBTEST_3(product(Matrix3d())); + CALL_SUBTEST_4(product(Matrix4d())); + CALL_SUBTEST_5(product(Matrix4f())); + CALL_SUBTEST_6(product1x1<0>()); - CALL_SUBTEST_11( test_lazy_l1() ); - CALL_SUBTEST_12( test_lazy_l2() ); - CALL_SUBTEST_13( test_lazy_l3() ); + CALL_SUBTEST_11(test_lazy_l1()); + CALL_SUBTEST_12(test_lazy_l2()); + CALL_SUBTEST_13(test_lazy_l3()); - CALL_SUBTEST_21( test_lazy_l1() ); - CALL_SUBTEST_22( test_lazy_l2() ); - CALL_SUBTEST_23( test_lazy_l3() ); + CALL_SUBTEST_21(test_lazy_l1()); + CALL_SUBTEST_22(test_lazy_l2()); + CALL_SUBTEST_23(test_lazy_l3()); - CALL_SUBTEST_31( test_lazy_l1 >() ); - CALL_SUBTEST_32( test_lazy_l2 >() ); - CALL_SUBTEST_33( test_lazy_l3 >() ); + CALL_SUBTEST_31(test_lazy_l1>()); + CALL_SUBTEST_32(test_lazy_l2>()); + CALL_SUBTEST_33(test_lazy_l3>()); - CALL_SUBTEST_41( test_lazy_l1 >() ); - CALL_SUBTEST_42( test_lazy_l2 >() ); - CALL_SUBTEST_43( test_lazy_l3 >() ); + CALL_SUBTEST_41(test_lazy_l1>()); + CALL_SUBTEST_42(test_lazy_l2>()); + CALL_SUBTEST_43(test_lazy_l3>()); - CALL_SUBTEST_7(( test_linear_but_not_vectorizable() )); - CALL_SUBTEST_7(( test_linear_but_not_vectorizable() )); - CALL_SUBTEST_7(( test_linear_but_not_vectorizable() )); + CALL_SUBTEST_7((test_linear_but_not_vectorizable())); + CALL_SUBTEST_7((test_linear_but_not_vectorizable())); + CALL_SUBTEST_7((test_linear_but_not_vectorizable())); - CALL_SUBTEST_6( bug_1311<3>() ); - CALL_SUBTEST_6( bug_1311<5>() ); + CALL_SUBTEST_6(bug_1311<3>()); + CALL_SUBTEST_6(bug_1311<5>()); } #ifdef EIGEN_TEST_PART_6 { // test compilation of (outer_product) * vector Vector3f v = Vector3f::Random(); - VERIFY_IS_APPROX( (v * v.transpose()) * v, (v * v.transpose()).eval() * v); + VERIFY_IS_APPROX((v * v.transpose()) * v, (v * v.transpose()).eval() * v); } - + { // regression test for pull-request #93 - Eigen::Matrix A; A.setRandom(); - Eigen::Matrix B; B.setRandom(); - Eigen::Matrix C; C.setRandom(); + Eigen::Matrix A; + A.setRandom(); + Eigen::Matrix B; + B.setRandom(); + Eigen::Matrix C; + C.setRandom(); VERIFY_IS_APPROX(B * A.inverse(), B * A.inverse()[0]); VERIFY_IS_APPROX(A.inverse() * C, A.inverse()[0] * C); } @@ -283,11 +281,13 @@ void test_product_small() Eigen::Matrix A, B, C; A.setRandom(); C = A; - for(int k=0; k<79; ++k) - C = C * A; - B.noalias() = (((A*A)*(A*A))*((A*A)*(A*A))*((A*A)*(A*A))*((A*A)*(A*A))*((A*A)*(A*A)) * ((A*A)*(A*A))*((A*A)*(A*A))*((A*A)*(A*A))*((A*A)*(A*A))*((A*A)*(A*A))) - * (((A*A)*(A*A))*((A*A)*(A*A))*((A*A)*(A*A))*((A*A)*(A*A))*((A*A)*(A*A)) * ((A*A)*(A*A))*((A*A)*(A*A))*((A*A)*(A*A))*((A*A)*(A*A))*((A*A)*(A*A))); - VERIFY_IS_APPROX(B,C); + for (int k = 0; k < 79; ++k) C = C * A; + B.noalias() = + (((A * A) * (A * A)) * ((A * A) * (A * A)) * ((A * A) * (A * A)) * ((A * A) * (A * A)) * ((A * A) * (A * A)) + * ((A * A) * (A * A)) * ((A * A) * (A * A)) * ((A * A) * (A * A)) * ((A * A) * (A * A)) * ((A * A) * (A * A))) + * (((A * A) * (A * A)) * ((A * A) * (A * A)) * ((A * A) * (A * A)) * ((A * A) * (A * A)) * ((A * A) * (A * A)) + * ((A * A) * (A * A)) * ((A * A) * (A * A)) * ((A * A) * (A * A)) * ((A * A) * (A * A)) * ((A * A) * (A * A))); + VERIFY_IS_APPROX(B, C); } #endif } diff --git a/filmulator-gui/core/nlmeans/eigen/test/product_symm.cpp b/filmulator-gui/core/nlmeans/eigen/test/product_symm.cpp index 7d1042a4..6e476158 100644 --- a/filmulator-gui/core/nlmeans/eigen/test/product_symm.cpp +++ b/filmulator-gui/core/nlmeans/eigen/test/product_symm.cpp @@ -14,98 +14,102 @@ template void symm(int size = Size, in typedef Matrix MatrixType; typedef Matrix Rhs1; typedef Matrix Rhs2; - enum { order = OtherSize==1 ? 0 : RowMajor }; - typedef Matrix Rhs3; + enum { order = OtherSize == 1 ? 0 : RowMajor }; + typedef Matrix Rhs3; Index rows = size; Index cols = size; - MatrixType m1 = MatrixType::Random(rows, cols), - m2 = MatrixType::Random(rows, cols), m3; + MatrixType m1 = MatrixType::Random(rows, cols), m2 = MatrixType::Random(rows, cols), m3; - m1 = (m1+m1.adjoint()).eval(); + m1 = (m1 + m1.adjoint()).eval(); Rhs1 rhs1 = Rhs1::Random(cols, othersize), rhs12(cols, othersize), rhs13(cols, othersize); Rhs2 rhs2 = Rhs2::Random(othersize, rows), rhs22(othersize, rows), rhs23(othersize, rows); Rhs3 rhs3 = Rhs3::Random(cols, othersize), rhs32(cols, othersize), rhs33(cols, othersize); - Scalar s1 = internal::random(), - s2 = internal::random(); + Scalar s1 = internal::random(), s2 = internal::random(); m2 = m1.template triangularView(); m3 = m2.template selfadjointView(); VERIFY_IS_EQUAL(m1, m3); - VERIFY_IS_APPROX(rhs12 = (s1*m2).template selfadjointView() * (s2*rhs1), - rhs13 = (s1*m1) * (s2*rhs1)); + VERIFY_IS_APPROX(rhs12 = (s1 * m2).template selfadjointView() * (s2 * rhs1), rhs13 = (s1 * m1) * (s2 * rhs1)); - VERIFY_IS_APPROX(rhs12 = (s1*m2).transpose().template selfadjointView() * (s2*rhs1), - rhs13 = (s1*m1.transpose()) * (s2*rhs1)); + VERIFY_IS_APPROX(rhs12 = (s1 * m2).transpose().template selfadjointView() * (s2 * rhs1), + rhs13 = (s1 * m1.transpose()) * (s2 * rhs1)); - VERIFY_IS_APPROX(rhs12 = (s1*m2).template selfadjointView().transpose() * (s2*rhs1), - rhs13 = (s1*m1.transpose()) * (s2*rhs1)); + VERIFY_IS_APPROX(rhs12 = (s1 * m2).template selfadjointView().transpose() * (s2 * rhs1), + rhs13 = (s1 * m1.transpose()) * (s2 * rhs1)); - VERIFY_IS_APPROX(rhs12 = (s1*m2).conjugate().template selfadjointView() * (s2*rhs1), - rhs13 = (s1*m1).conjugate() * (s2*rhs1)); + VERIFY_IS_APPROX(rhs12 = (s1 * m2).conjugate().template selfadjointView() * (s2 * rhs1), + rhs13 = (s1 * m1).conjugate() * (s2 * rhs1)); - VERIFY_IS_APPROX(rhs12 = (s1*m2).template selfadjointView().conjugate() * (s2*rhs1), - rhs13 = (s1*m1).conjugate() * (s2*rhs1)); + VERIFY_IS_APPROX(rhs12 = (s1 * m2).template selfadjointView().conjugate() * (s2 * rhs1), + rhs13 = (s1 * m1).conjugate() * (s2 * rhs1)); - VERIFY_IS_APPROX(rhs12 = (s1*m2).adjoint().template selfadjointView() * (s2*rhs1), - rhs13 = (s1*m1).adjoint() * (s2*rhs1)); + VERIFY_IS_APPROX(rhs12 = (s1 * m2).adjoint().template selfadjointView() * (s2 * rhs1), + rhs13 = (s1 * m1).adjoint() * (s2 * rhs1)); - VERIFY_IS_APPROX(rhs12 = (s1*m2).template selfadjointView().adjoint() * (s2*rhs1), - rhs13 = (s1*m1).adjoint() * (s2*rhs1)); + VERIFY_IS_APPROX(rhs12 = (s1 * m2).template selfadjointView().adjoint() * (s2 * rhs1), + rhs13 = (s1 * m1).adjoint() * (s2 * rhs1)); - m2 = m1.template triangularView(); rhs12.setRandom(); rhs13 = rhs12; + m2 = m1.template triangularView(); + rhs12.setRandom(); + rhs13 = rhs12; m3 = m2.template selfadjointView(); VERIFY_IS_EQUAL(m1, m3); - VERIFY_IS_APPROX(rhs12 += (s1*m2).template selfadjointView() * (s2*rhs1), - rhs13 += (s1*m1) * (s2*rhs1)); + VERIFY_IS_APPROX( + rhs12 += (s1 * m2).template selfadjointView() * (s2 * rhs1), rhs13 += (s1 * m1) * (s2 * rhs1)); m2 = m1.template triangularView(); - VERIFY_IS_APPROX(rhs12 = (s1*m2).template selfadjointView() * (s2*rhs2.adjoint()), - rhs13 = (s1*m1) * (s2*rhs2.adjoint())); + VERIFY_IS_APPROX(rhs12 = (s1 * m2).template selfadjointView() * (s2 * rhs2.adjoint()), + rhs13 = (s1 * m1) * (s2 * rhs2.adjoint())); m2 = m1.template triangularView(); - VERIFY_IS_APPROX(rhs12 = (s1*m2).template selfadjointView() * (s2*rhs2.adjoint()), - rhs13 = (s1*m1) * (s2*rhs2.adjoint())); + VERIFY_IS_APPROX(rhs12 = (s1 * m2).template selfadjointView() * (s2 * rhs2.adjoint()), + rhs13 = (s1 * m1) * (s2 * rhs2.adjoint())); m2 = m1.template triangularView(); - VERIFY_IS_APPROX(rhs12 = (s1*m2.adjoint()).template selfadjointView() * (s2*rhs2.adjoint()), - rhs13 = (s1*m1.adjoint()) * (s2*rhs2.adjoint())); + VERIFY_IS_APPROX(rhs12 = (s1 * m2.adjoint()).template selfadjointView() * (s2 * rhs2.adjoint()), + rhs13 = (s1 * m1.adjoint()) * (s2 * rhs2.adjoint())); // test row major = <...> - m2 = m1.template triangularView(); rhs12.setRandom(); rhs13 = rhs12; - VERIFY_IS_APPROX(rhs12 -= (s1*m2).template selfadjointView() * (s2*rhs3), - rhs13 -= (s1*m1) * (s2 * rhs3)); + m2 = m1.template triangularView(); + rhs12.setRandom(); + rhs13 = rhs12; + VERIFY_IS_APPROX( + rhs12 -= (s1 * m2).template selfadjointView() * (s2 * rhs3), rhs13 -= (s1 * m1) * (s2 * rhs3)); m2 = m1.template triangularView(); - VERIFY_IS_APPROX(rhs12 = (s1*m2.adjoint()).template selfadjointView() * (s2*rhs3).conjugate(), - rhs13 = (s1*m1.adjoint()) * (s2*rhs3).conjugate()); + VERIFY_IS_APPROX(rhs12 = (s1 * m2.adjoint()).template selfadjointView() * (s2 * rhs3).conjugate(), + rhs13 = (s1 * m1.adjoint()) * (s2 * rhs3).conjugate()); - m2 = m1.template triangularView(); rhs13 = rhs12; - VERIFY_IS_APPROX(rhs12.noalias() += s1 * ((m2.adjoint()).template selfadjointView() * (s2*rhs3).conjugate()), - rhs13 += (s1*m1.adjoint()) * (s2*rhs3).conjugate()); + m2 = m1.template triangularView(); + rhs13 = rhs12; + VERIFY_IS_APPROX(rhs12.noalias() += s1 * ((m2.adjoint()).template selfadjointView() * (s2 * rhs3).conjugate()), + rhs13 += (s1 * m1.adjoint()) * (s2 * rhs3).conjugate()); m2 = m1.template triangularView(); VERIFY_IS_APPROX(rhs22 = (rhs2) * (m2).template selfadjointView(), rhs23 = (rhs2) * (m1)); - VERIFY_IS_APPROX(rhs22 = (s2*rhs2) * (s1*m2).template selfadjointView(), rhs23 = (s2*rhs2) * (s1*m1)); - + VERIFY_IS_APPROX(rhs22 = (s2 * rhs2) * (s1 * m2).template selfadjointView(), rhs23 = (s2 * rhs2) * (s1 * m1)); } void test_product_symm() { - for(int i = 0; i < g_repeat ; i++) - { - CALL_SUBTEST_1(( symm(internal::random(1,EIGEN_TEST_MAX_SIZE),internal::random(1,EIGEN_TEST_MAX_SIZE)) )); - CALL_SUBTEST_2(( symm(internal::random(1,EIGEN_TEST_MAX_SIZE),internal::random(1,EIGEN_TEST_MAX_SIZE)) )); - CALL_SUBTEST_3(( symm,Dynamic,Dynamic>(internal::random(1,EIGEN_TEST_MAX_SIZE/2),internal::random(1,EIGEN_TEST_MAX_SIZE/2)) )); - CALL_SUBTEST_4(( symm,Dynamic,Dynamic>(internal::random(1,EIGEN_TEST_MAX_SIZE/2),internal::random(1,EIGEN_TEST_MAX_SIZE/2)) )); - - CALL_SUBTEST_5(( symm(internal::random(1,EIGEN_TEST_MAX_SIZE)) )); - CALL_SUBTEST_6(( symm(internal::random(1,EIGEN_TEST_MAX_SIZE)) )); - CALL_SUBTEST_7(( symm,Dynamic,1>(internal::random(1,EIGEN_TEST_MAX_SIZE)) )); - CALL_SUBTEST_8(( symm,Dynamic,1>(internal::random(1,EIGEN_TEST_MAX_SIZE)) )); + for (int i = 0; i < g_repeat; i++) { + CALL_SUBTEST_1((symm( + internal::random(1, EIGEN_TEST_MAX_SIZE), internal::random(1, EIGEN_TEST_MAX_SIZE)))); + CALL_SUBTEST_2((symm( + internal::random(1, EIGEN_TEST_MAX_SIZE), internal::random(1, EIGEN_TEST_MAX_SIZE)))); + CALL_SUBTEST_3((symm, Dynamic, Dynamic>( + internal::random(1, EIGEN_TEST_MAX_SIZE / 2), internal::random(1, EIGEN_TEST_MAX_SIZE / 2)))); + CALL_SUBTEST_4((symm, Dynamic, Dynamic>( + internal::random(1, EIGEN_TEST_MAX_SIZE / 2), internal::random(1, EIGEN_TEST_MAX_SIZE / 2)))); + + CALL_SUBTEST_5((symm(internal::random(1, EIGEN_TEST_MAX_SIZE)))); + CALL_SUBTEST_6((symm(internal::random(1, EIGEN_TEST_MAX_SIZE)))); + CALL_SUBTEST_7((symm, Dynamic, 1>(internal::random(1, EIGEN_TEST_MAX_SIZE)))); + CALL_SUBTEST_8((symm, Dynamic, 1>(internal::random(1, EIGEN_TEST_MAX_SIZE)))); } } diff --git a/filmulator-gui/core/nlmeans/eigen/test/product_syrk.cpp b/filmulator-gui/core/nlmeans/eigen/test/product_syrk.cpp index 3ebbe14c..c8b621e7 100644 --- a/filmulator-gui/core/nlmeans/eigen/test/product_syrk.cpp +++ b/filmulator-gui/core/nlmeans/eigen/test/product_syrk.cpp @@ -9,127 +9,151 @@ #include "main.h" -template void syrk(const MatrixType& m) +template void syrk(const MatrixType &m) { typedef typename MatrixType::Scalar Scalar; typedef Matrix RMatrixType; typedef Matrix Rhs1; typedef Matrix Rhs2; - typedef Matrix Rhs3; + typedef Matrix Rhs3; Index rows = m.rows(); Index cols = m.cols(); - MatrixType m1 = MatrixType::Random(rows, cols), - m2 = MatrixType::Random(rows, cols), + MatrixType m1 = MatrixType::Random(rows, cols), m2 = MatrixType::Random(rows, cols), m3 = MatrixType::Random(rows, cols); RMatrixType rm2 = MatrixType::Random(rows, cols); - Rhs1 rhs1 = Rhs1::Random(internal::random(1,320), cols); Rhs1 rhs11 = Rhs1::Random(rhs1.rows(), cols); - Rhs2 rhs2 = Rhs2::Random(rows, internal::random(1,320)); Rhs2 rhs22 = Rhs2::Random(rows, rhs2.cols()); - Rhs3 rhs3 = Rhs3::Random(internal::random(1,320), rows); + Rhs1 rhs1 = Rhs1::Random(internal::random(1, 320), cols); + Rhs1 rhs11 = Rhs1::Random(rhs1.rows(), cols); + Rhs2 rhs2 = Rhs2::Random(rows, internal::random(1, 320)); + Rhs2 rhs22 = Rhs2::Random(rows, rhs2.cols()); + Rhs3 rhs3 = Rhs3::Random(internal::random(1, 320), rows); Scalar s1 = internal::random(); - - Index c = internal::random(0,cols-1); + + Index c = internal::random(0, cols - 1); m2.setZero(); - VERIFY_IS_APPROX((m2.template selfadjointView().rankUpdate(rhs2,s1)._expression()), - ((s1 * rhs2 * rhs2.adjoint()).eval().template triangularView().toDenseMatrix())); + VERIFY_IS_APPROX((m2.template selfadjointView().rankUpdate(rhs2, s1)._expression()), + ((s1 * rhs2 * rhs2.adjoint()).eval().template triangularView().toDenseMatrix())); m2.setZero(); - VERIFY_IS_APPROX(((m2.template triangularView() += s1 * rhs2 * rhs22.adjoint()).nestedExpression()), - ((s1 * rhs2 * rhs22.adjoint()).eval().template triangularView().toDenseMatrix())); + VERIFY_IS_APPROX(((m2.template triangularView() += s1 * rhs2 * rhs22.adjoint()).nestedExpression()), + ((s1 * rhs2 * rhs22.adjoint()).eval().template triangularView().toDenseMatrix())); + - m2.setZero(); - VERIFY_IS_APPROX(m2.template selfadjointView().rankUpdate(rhs2,s1)._expression(), - (s1 * rhs2 * rhs2.adjoint()).eval().template triangularView().toDenseMatrix()); + VERIFY_IS_APPROX(m2.template selfadjointView().rankUpdate(rhs2, s1)._expression(), + (s1 * rhs2 * rhs2.adjoint()).eval().template triangularView().toDenseMatrix()); m2.setZero(); VERIFY_IS_APPROX((m2.template triangularView() += s1 * rhs22 * rhs2.adjoint()).nestedExpression(), - (s1 * rhs22 * rhs2.adjoint()).eval().template triangularView().toDenseMatrix()); + (s1 * rhs22 * rhs2.adjoint()).eval().template triangularView().toDenseMatrix()); + - m2.setZero(); - VERIFY_IS_APPROX(m2.template selfadjointView().rankUpdate(rhs1.adjoint(),s1)._expression(), - (s1 * rhs1.adjoint() * rhs1).eval().template triangularView().toDenseMatrix()); + VERIFY_IS_APPROX(m2.template selfadjointView().rankUpdate(rhs1.adjoint(), s1)._expression(), + (s1 * rhs1.adjoint() * rhs1).eval().template triangularView().toDenseMatrix()); m2.setZero(); VERIFY_IS_APPROX((m2.template triangularView() += s1 * rhs11.adjoint() * rhs1).nestedExpression(), - (s1 * rhs11.adjoint() * rhs1).eval().template triangularView().toDenseMatrix()); - - + (s1 * rhs11.adjoint() * rhs1).eval().template triangularView().toDenseMatrix()); + + m2.setZero(); - VERIFY_IS_APPROX(m2.template selfadjointView().rankUpdate(rhs1.adjoint(),s1)._expression(), - (s1 * rhs1.adjoint() * rhs1).eval().template triangularView().toDenseMatrix()); + VERIFY_IS_APPROX(m2.template selfadjointView().rankUpdate(rhs1.adjoint(), s1)._expression(), + (s1 * rhs1.adjoint() * rhs1).eval().template triangularView().toDenseMatrix()); VERIFY_IS_APPROX((m2.template triangularView() = s1 * rhs1.adjoint() * rhs11).nestedExpression(), - (s1 * rhs1.adjoint() * rhs11).eval().template triangularView().toDenseMatrix()); + (s1 * rhs1.adjoint() * rhs11).eval().template triangularView().toDenseMatrix()); + - m2.setZero(); - VERIFY_IS_APPROX(m2.template selfadjointView().rankUpdate(rhs3.adjoint(),s1)._expression(), - (s1 * rhs3.adjoint() * rhs3).eval().template triangularView().toDenseMatrix()); + VERIFY_IS_APPROX(m2.template selfadjointView().rankUpdate(rhs3.adjoint(), s1)._expression(), + (s1 * rhs3.adjoint() * rhs3).eval().template triangularView().toDenseMatrix()); m2.setZero(); - VERIFY_IS_APPROX(m2.template selfadjointView().rankUpdate(rhs3.adjoint(),s1)._expression(), - (s1 * rhs3.adjoint() * rhs3).eval().template triangularView().toDenseMatrix()); - + VERIFY_IS_APPROX(m2.template selfadjointView().rankUpdate(rhs3.adjoint(), s1)._expression(), + (s1 * rhs3.adjoint() * rhs3).eval().template triangularView().toDenseMatrix()); + m2.setZero(); - VERIFY_IS_APPROX((m2.template selfadjointView().rankUpdate(m1.col(c),s1)._expression()), - ((s1 * m1.col(c) * m1.col(c).adjoint()).eval().template triangularView().toDenseMatrix())); - + VERIFY_IS_APPROX((m2.template selfadjointView().rankUpdate(m1.col(c), s1)._expression()), + ((s1 * m1.col(c) * m1.col(c).adjoint()).eval().template triangularView().toDenseMatrix())); + m2.setZero(); - VERIFY_IS_APPROX((m2.template selfadjointView().rankUpdate(m1.col(c),s1)._expression()), - ((s1 * m1.col(c) * m1.col(c).adjoint()).eval().template triangularView().toDenseMatrix())); + VERIFY_IS_APPROX((m2.template selfadjointView().rankUpdate(m1.col(c), s1)._expression()), + ((s1 * m1.col(c) * m1.col(c).adjoint()).eval().template triangularView().toDenseMatrix())); rm2.setZero(); - VERIFY_IS_APPROX((rm2.template selfadjointView().rankUpdate(m1.col(c),s1)._expression()), - ((s1 * m1.col(c) * m1.col(c).adjoint()).eval().template triangularView().toDenseMatrix())); + VERIFY_IS_APPROX((rm2.template selfadjointView().rankUpdate(m1.col(c), s1)._expression()), + ((s1 * m1.col(c) * m1.col(c).adjoint()).eval().template triangularView().toDenseMatrix())); m2.setZero(); VERIFY_IS_APPROX((m2.template triangularView() += s1 * m3.col(c) * m1.col(c).adjoint()).nestedExpression(), - ((s1 * m3.col(c) * m1.col(c).adjoint()).eval().template triangularView().toDenseMatrix())); + ((s1 * m3.col(c) * m1.col(c).adjoint()).eval().template triangularView().toDenseMatrix())); rm2.setZero(); VERIFY_IS_APPROX((rm2.template triangularView() += s1 * m1.col(c) * m3.col(c).adjoint()).nestedExpression(), - ((s1 * m1.col(c) * m3.col(c).adjoint()).eval().template triangularView().toDenseMatrix())); - + ((s1 * m1.col(c) * m3.col(c).adjoint()).eval().template triangularView().toDenseMatrix())); + m2.setZero(); - VERIFY_IS_APPROX((m2.template selfadjointView().rankUpdate(m1.col(c).conjugate(),s1)._expression()), - ((s1 * m1.col(c).conjugate() * m1.col(c).conjugate().adjoint()).eval().template triangularView().toDenseMatrix())); - + VERIFY_IS_APPROX((m2.template selfadjointView().rankUpdate(m1.col(c).conjugate(), s1)._expression()), + ((s1 * m1.col(c).conjugate() * m1.col(c).conjugate().adjoint()) + .eval() + .template triangularView() + .toDenseMatrix())); + m2.setZero(); - VERIFY_IS_APPROX((m2.template selfadjointView().rankUpdate(m1.col(c).conjugate(),s1)._expression()), - ((s1 * m1.col(c).conjugate() * m1.col(c).conjugate().adjoint()).eval().template triangularView().toDenseMatrix())); - - + VERIFY_IS_APPROX((m2.template selfadjointView().rankUpdate(m1.col(c).conjugate(), s1)._expression()), + ((s1 * m1.col(c).conjugate() * m1.col(c).conjugate().adjoint()) + .eval() + .template triangularView() + .toDenseMatrix())); + + m2.setZero(); - VERIFY_IS_APPROX((m2.template selfadjointView().rankUpdate(m1.row(c),s1)._expression()), - ((s1 * m1.row(c).transpose() * m1.row(c).transpose().adjoint()).eval().template triangularView().toDenseMatrix())); + VERIFY_IS_APPROX((m2.template selfadjointView().rankUpdate(m1.row(c), s1)._expression()), + ((s1 * m1.row(c).transpose() * m1.row(c).transpose().adjoint()) + .eval() + .template triangularView() + .toDenseMatrix())); rm2.setZero(); - VERIFY_IS_APPROX((rm2.template selfadjointView().rankUpdate(m1.row(c),s1)._expression()), - ((s1 * m1.row(c).transpose() * m1.row(c).transpose().adjoint()).eval().template triangularView().toDenseMatrix())); - m2.setZero(); - VERIFY_IS_APPROX((m2.template triangularView() += s1 * m3.row(c).transpose() * m1.row(c).transpose().adjoint()).nestedExpression(), - ((s1 * m3.row(c).transpose() * m1.row(c).transpose().adjoint()).eval().template triangularView().toDenseMatrix())); + VERIFY_IS_APPROX((rm2.template selfadjointView().rankUpdate(m1.row(c), s1)._expression()), + ((s1 * m1.row(c).transpose() * m1.row(c).transpose().adjoint()) + .eval() + .template triangularView() + .toDenseMatrix())); + m2.setZero(); + VERIFY_IS_APPROX((m2.template triangularView() += s1 * m3.row(c).transpose() * m1.row(c).transpose().adjoint()) + .nestedExpression(), + ((s1 * m3.row(c).transpose() * m1.row(c).transpose().adjoint()) + .eval() + .template triangularView() + .toDenseMatrix())); rm2.setZero(); - VERIFY_IS_APPROX((rm2.template triangularView() += s1 * m3.row(c).transpose() * m1.row(c).transpose().adjoint()).nestedExpression(), - ((s1 * m3.row(c).transpose() * m1.row(c).transpose().adjoint()).eval().template triangularView().toDenseMatrix())); - - + VERIFY_IS_APPROX( + (rm2.template triangularView() += s1 * m3.row(c).transpose() * m1.row(c).transpose().adjoint()) + .nestedExpression(), + ((s1 * m3.row(c).transpose() * m1.row(c).transpose().adjoint()) + .eval() + .template triangularView() + .toDenseMatrix())); + + m2.setZero(); - VERIFY_IS_APPROX((m2.template selfadjointView().rankUpdate(m1.row(c).adjoint(),s1)._expression()), - ((s1 * m1.row(c).adjoint() * m1.row(c).adjoint().adjoint()).eval().template triangularView().toDenseMatrix())); + VERIFY_IS_APPROX((m2.template selfadjointView().rankUpdate(m1.row(c).adjoint(), s1)._expression()), + ((s1 * m1.row(c).adjoint() * m1.row(c).adjoint().adjoint()) + .eval() + .template triangularView() + .toDenseMatrix())); } void test_product_syrk() { - for(int i = 0; i < g_repeat ; i++) - { + for (int i = 0; i < g_repeat; i++) { int s; - s = internal::random(1,EIGEN_TEST_MAX_SIZE); - CALL_SUBTEST_1( syrk(MatrixXf(s, s)) ); - CALL_SUBTEST_2( syrk(MatrixXd(s, s)) ); + s = internal::random(1, EIGEN_TEST_MAX_SIZE); + CALL_SUBTEST_1(syrk(MatrixXf(s, s))); + CALL_SUBTEST_2(syrk(MatrixXd(s, s))); TEST_SET_BUT_UNUSED_VARIABLE(s) - - s = internal::random(1,EIGEN_TEST_MAX_SIZE/2); - CALL_SUBTEST_3( syrk(MatrixXcf(s, s)) ); - CALL_SUBTEST_4( syrk(MatrixXcd(s, s)) ); + + s = internal::random(1, EIGEN_TEST_MAX_SIZE / 2); + CALL_SUBTEST_3(syrk(MatrixXcf(s, s))); + CALL_SUBTEST_4(syrk(MatrixXcd(s, s))); TEST_SET_BUT_UNUSED_VARIABLE(s) } } diff --git a/filmulator-gui/core/nlmeans/eigen/test/product_trmm.cpp b/filmulator-gui/core/nlmeans/eigen/test/product_trmm.cpp index e08d9f39..49f2d200 100644 --- a/filmulator-gui/core/nlmeans/eigen/test/product_trmm.cpp +++ b/filmulator-gui/core/nlmeans/eigen/test/product_trmm.cpp @@ -9,119 +9,130 @@ #include "main.h" -template -int get_random_size() +template int get_random_size() { const int factor = NumTraits::ReadCost; - const int max_test_size = EIGEN_TEST_MAX_SIZE>2*factor ? EIGEN_TEST_MAX_SIZE/factor : EIGEN_TEST_MAX_SIZE; - return internal::random(1,max_test_size); + const int max_test_size = EIGEN_TEST_MAX_SIZE > 2 * factor ? EIGEN_TEST_MAX_SIZE / factor : EIGEN_TEST_MAX_SIZE; + return internal::random(1, max_test_size); } template -void trmm(int rows=get_random_size(), - int cols=get_random_size(), - int otherCols = OtherCols==Dynamic?get_random_size():OtherCols) +void trmm(int rows = get_random_size(), + int cols = get_random_size(), + int otherCols = OtherCols == Dynamic ? get_random_size() : OtherCols) { - typedef Matrix TriMatrix; - typedef Matrix OnTheRight; - typedef Matrix OnTheLeft; - - typedef Matrix ResXS; - typedef Matrix ResSX; - - TriMatrix mat(rows,cols), tri(rows,cols), triTr(cols,rows), s1tri(rows,cols), s1triTr(cols,rows); - - OnTheRight ge_right(cols,otherCols); - OnTheLeft ge_left(otherCols,rows); - ResSX ge_sx, ge_sx_save; - ResXS ge_xs, ge_xs_save; - - Scalar s1 = internal::random(), - s2 = internal::random(); + typedef Matrix TriMatrix; + typedef Matrix OnTheRight; + typedef Matrix OnTheLeft; + + typedef Matrix ResXS; + typedef Matrix ResSX; + + TriMatrix mat(rows, cols), tri(rows, cols), triTr(cols, rows), s1tri(rows, cols), s1triTr(cols, rows); + + OnTheRight ge_right(cols, otherCols); + OnTheLeft ge_left(otherCols, rows); + ResSX ge_sx, ge_sx_save; + ResXS ge_xs, ge_xs_save; + + Scalar s1 = internal::random(), s2 = internal::random(); mat.setRandom(); tri = mat.template triangularView(); triTr = mat.transpose().template triangularView(); - s1tri = (s1*mat).template triangularView(); - s1triTr = (s1*mat).transpose().template triangularView(); + s1tri = (s1 * mat).template triangularView(); + s1triTr = (s1 * mat).transpose().template triangularView(); ge_right.setRandom(); ge_left.setRandom(); - VERIFY_IS_APPROX( ge_xs = mat.template triangularView() * ge_right, tri * ge_right); - VERIFY_IS_APPROX( ge_sx = ge_left * mat.template triangularView(), ge_left * tri); - - VERIFY_IS_APPROX( ge_xs.noalias() = mat.template triangularView() * ge_right, tri * ge_right); - VERIFY_IS_APPROX( ge_sx.noalias() = ge_left * mat.template triangularView(), ge_left * tri); - - if((Mode&UnitDiag)==0) - VERIFY_IS_APPROX( ge_xs.noalias() = (s1*mat.adjoint()).template triangularView() * (s2*ge_left.transpose()), s1*triTr.conjugate() * (s2*ge_left.transpose())); - - VERIFY_IS_APPROX( ge_xs.noalias() = (s1*mat.transpose()).template triangularView() * (s2*ge_left.transpose()), s1triTr * (s2*ge_left.transpose())); - VERIFY_IS_APPROX( ge_sx.noalias() = (s2*ge_left) * (s1*mat).template triangularView(), (s2*ge_left)*s1tri); - - VERIFY_IS_APPROX( ge_sx.noalias() = ge_right.transpose() * mat.adjoint().template triangularView(), ge_right.transpose() * triTr.conjugate()); - VERIFY_IS_APPROX( ge_sx.noalias() = ge_right.adjoint() * mat.adjoint().template triangularView(), ge_right.adjoint() * triTr.conjugate()); - + VERIFY_IS_APPROX(ge_xs = mat.template triangularView() * ge_right, tri * ge_right); + VERIFY_IS_APPROX(ge_sx = ge_left * mat.template triangularView(), ge_left * tri); + + VERIFY_IS_APPROX(ge_xs.noalias() = mat.template triangularView() * ge_right, tri * ge_right); + VERIFY_IS_APPROX(ge_sx.noalias() = ge_left * mat.template triangularView(), ge_left * tri); + + if ((Mode & UnitDiag) == 0) + VERIFY_IS_APPROX( + ge_xs.noalias() = (s1 * mat.adjoint()).template triangularView() * (s2 * ge_left.transpose()), + s1 * triTr.conjugate() * (s2 * ge_left.transpose())); + + VERIFY_IS_APPROX( + ge_xs.noalias() = (s1 * mat.transpose()).template triangularView() * (s2 * ge_left.transpose()), + s1triTr * (s2 * ge_left.transpose())); + VERIFY_IS_APPROX( + ge_sx.noalias() = (s2 * ge_left) * (s1 * mat).template triangularView(), (s2 * ge_left) * s1tri); + + VERIFY_IS_APPROX(ge_sx.noalias() = ge_right.transpose() * mat.adjoint().template triangularView(), + ge_right.transpose() * triTr.conjugate()); + VERIFY_IS_APPROX(ge_sx.noalias() = ge_right.adjoint() * mat.adjoint().template triangularView(), + ge_right.adjoint() * triTr.conjugate()); + ge_xs_save = ge_xs; - if((Mode&UnitDiag)==0) - VERIFY_IS_APPROX( (ge_xs_save + s1*triTr.conjugate() * (s2*ge_left.adjoint())).eval(), ge_xs.noalias() += (s1*mat.adjoint()).template triangularView() * (s2*ge_left.adjoint()) ); + if ((Mode & UnitDiag) == 0) + VERIFY_IS_APPROX((ge_xs_save + s1 * triTr.conjugate() * (s2 * ge_left.adjoint())).eval(), + ge_xs.noalias() += (s1 * mat.adjoint()).template triangularView() * (s2 * ge_left.adjoint())); ge_xs_save = ge_xs; - VERIFY_IS_APPROX( (ge_xs_save + s1triTr * (s2*ge_left.adjoint())).eval(), ge_xs.noalias() += (s1*mat.transpose()).template triangularView() * (s2*ge_left.adjoint()) ); + VERIFY_IS_APPROX((ge_xs_save + s1triTr * (s2 * ge_left.adjoint())).eval(), + ge_xs.noalias() += (s1 * mat.transpose()).template triangularView() * (s2 * ge_left.adjoint())); ge_sx.setRandom(); ge_sx_save = ge_sx; - if((Mode&UnitDiag)==0) - VERIFY_IS_APPROX( ge_sx_save - (ge_right.adjoint() * (-s1 * triTr).conjugate()).eval(), ge_sx.noalias() -= (ge_right.adjoint() * (-s1 * mat).adjoint().template triangularView()).eval()); - - if((Mode&UnitDiag)==0) - VERIFY_IS_APPROX( ge_xs = (s1*mat).adjoint().template triangularView() * ge_left.adjoint(), numext::conj(s1) * triTr.conjugate() * ge_left.adjoint()); - VERIFY_IS_APPROX( ge_xs = (s1*mat).transpose().template triangularView() * ge_left.adjoint(), s1triTr * ge_left.adjoint()); - - + if ((Mode & UnitDiag) == 0) + VERIFY_IS_APPROX(ge_sx_save - (ge_right.adjoint() * (-s1 * triTr).conjugate()).eval(), + ge_sx.noalias() -= (ge_right.adjoint() * (-s1 * mat).adjoint().template triangularView()).eval()); + + if ((Mode & UnitDiag) == 0) + VERIFY_IS_APPROX(ge_xs = (s1 * mat).adjoint().template triangularView() * ge_left.adjoint(), + numext::conj(s1) * triTr.conjugate() * ge_left.adjoint()); + VERIFY_IS_APPROX( + ge_xs = (s1 * mat).transpose().template triangularView() * ge_left.adjoint(), s1triTr * ge_left.adjoint()); + + // TODO check with sub-matrix expressions ? } template -void trmv(int rows=get_random_size(), int cols=get_random_size()) +void trmv(int rows = get_random_size(), int cols = get_random_size()) { - trmm(rows,cols,1); + trmm(rows, cols, 1); } template -void trmm(int rows=get_random_size(), int cols=get_random_size(), int otherCols = get_random_size()) +void trmm(int rows = get_random_size(), + int cols = get_random_size(), + int otherCols = get_random_size()) { - trmm(rows,cols,otherCols); + trmm(rows, cols, otherCols); } -#define CALL_ALL_ORDERS(NB,SCALAR,MODE) \ - EIGEN_CAT(CALL_SUBTEST_,NB)((trmm())); \ - EIGEN_CAT(CALL_SUBTEST_,NB)((trmm())); \ - EIGEN_CAT(CALL_SUBTEST_,NB)((trmm())); \ - EIGEN_CAT(CALL_SUBTEST_,NB)((trmm())); \ - EIGEN_CAT(CALL_SUBTEST_,NB)((trmm())); \ - EIGEN_CAT(CALL_SUBTEST_,NB)((trmm())); \ - EIGEN_CAT(CALL_SUBTEST_,NB)((trmm())); \ - EIGEN_CAT(CALL_SUBTEST_,NB)((trmm())); \ - \ - EIGEN_CAT(CALL_SUBTEST_1,NB)((trmv())); \ - EIGEN_CAT(CALL_SUBTEST_1,NB)((trmv())); - - -#define CALL_ALL(NB,SCALAR) \ - CALL_ALL_ORDERS(EIGEN_CAT(1,NB),SCALAR,Upper) \ - CALL_ALL_ORDERS(EIGEN_CAT(2,NB),SCALAR,UnitUpper) \ - CALL_ALL_ORDERS(EIGEN_CAT(3,NB),SCALAR,StrictlyUpper) \ - CALL_ALL_ORDERS(EIGEN_CAT(1,NB),SCALAR,Lower) \ - CALL_ALL_ORDERS(EIGEN_CAT(2,NB),SCALAR,UnitLower) \ - CALL_ALL_ORDERS(EIGEN_CAT(3,NB),SCALAR,StrictlyLower) - +#define CALL_ALL_ORDERS(NB, SCALAR, MODE) \ + EIGEN_CAT(CALL_SUBTEST_, NB)((trmm())); \ + EIGEN_CAT(CALL_SUBTEST_, NB)((trmm())); \ + EIGEN_CAT(CALL_SUBTEST_, NB)((trmm())); \ + EIGEN_CAT(CALL_SUBTEST_, NB)((trmm())); \ + EIGEN_CAT(CALL_SUBTEST_, NB)((trmm())); \ + EIGEN_CAT(CALL_SUBTEST_, NB)((trmm())); \ + EIGEN_CAT(CALL_SUBTEST_, NB)((trmm())); \ + EIGEN_CAT(CALL_SUBTEST_, NB)((trmm())); \ + \ + EIGEN_CAT(CALL_SUBTEST_1, NB)((trmv())); \ + EIGEN_CAT(CALL_SUBTEST_1, NB)((trmv())); + + +#define CALL_ALL(NB, SCALAR) \ + CALL_ALL_ORDERS(EIGEN_CAT(1, NB), SCALAR, Upper) \ + CALL_ALL_ORDERS(EIGEN_CAT(2, NB), SCALAR, UnitUpper) \ + CALL_ALL_ORDERS(EIGEN_CAT(3, NB), SCALAR, StrictlyUpper) \ + CALL_ALL_ORDERS(EIGEN_CAT(1, NB), SCALAR, Lower) \ + CALL_ALL_ORDERS(EIGEN_CAT(2, NB), SCALAR, UnitLower) \ + CALL_ALL_ORDERS(EIGEN_CAT(3, NB), SCALAR, StrictlyLower) + void test_product_trmm() { - for(int i = 0; i < g_repeat ; i++) - { - CALL_ALL(1,float); // EIGEN_SUFFIXES;11;111;21;121;31;131 - CALL_ALL(2,double); // EIGEN_SUFFIXES;12;112;22;122;32;132 - CALL_ALL(3,std::complex); // EIGEN_SUFFIXES;13;113;23;123;33;133 - CALL_ALL(4,std::complex); // EIGEN_SUFFIXES;14;114;24;124;34;134 + for (int i = 0; i < g_repeat; i++) { + CALL_ALL(1, float);// EIGEN_SUFFIXES;11;111;21;121;31;131 + CALL_ALL(2, double);// EIGEN_SUFFIXES;12;112;22;122;32;132 + CALL_ALL(3, std::complex);// EIGEN_SUFFIXES;13;113;23;123;33;133 + CALL_ALL(4, std::complex);// EIGEN_SUFFIXES;14;114;24;124;34;134 } } diff --git a/filmulator-gui/core/nlmeans/eigen/test/product_trmv.cpp b/filmulator-gui/core/nlmeans/eigen/test/product_trmv.cpp index 65d66e57..aaa24822 100644 --- a/filmulator-gui/core/nlmeans/eigen/test/product_trmv.cpp +++ b/filmulator-gui/core/nlmeans/eigen/test/product_trmv.cpp @@ -9,19 +9,18 @@ #include "main.h" -template void trmv(const MatrixType& m) +template void trmv(const MatrixType &m) { typedef typename MatrixType::Scalar Scalar; typedef typename NumTraits::Real RealScalar; typedef Matrix VectorType; - RealScalar largerEps = 10*test_precision(); + RealScalar largerEps = 10 * test_precision(); Index rows = m.rows(); Index cols = m.cols(); - MatrixType m1 = MatrixType::Random(rows, cols), - m3(rows, cols); + MatrixType m1 = MatrixType::Random(rows, cols), m3(rows, cols); VectorType v1 = VectorType::Random(rows); Scalar s1 = internal::random(); @@ -40,9 +39,11 @@ template void trmv(const MatrixType& m) // check conjugated and scalar multiple expressions (col-major) m3 = m1.template triangularView(); - VERIFY(((s1*m3).conjugate() * v1).isApprox((s1*m1).conjugate().template triangularView() * v1, largerEps)); + VERIFY(((s1 * m3).conjugate() * v1) + .isApprox((s1 * m1).conjugate().template triangularView() * v1, largerEps)); m3 = m1.template triangularView(); - VERIFY((m3.conjugate() * v1.conjugate()).isApprox(m1.conjugate().template triangularView() * v1.conjugate(), largerEps)); + VERIFY((m3.conjugate() * v1.conjugate()) + .isApprox(m1.conjugate().template triangularView() * v1.conjugate(), largerEps)); // check with a row-major matrix m3 = m1.template triangularView(); @@ -58,14 +59,16 @@ template void trmv(const MatrixType& m) m3 = m1.template triangularView(); VERIFY((m3.adjoint() * v1).isApprox(m1.adjoint().template triangularView() * v1, largerEps)); m3 = m1.template triangularView(); - VERIFY((m3.adjoint() * (s1*v1.conjugate())).isApprox(m1.adjoint().template triangularView() * (s1*v1.conjugate()), largerEps)); + VERIFY((m3.adjoint() * (s1 * v1.conjugate())) + .isApprox(m1.adjoint().template triangularView() * (s1 * v1.conjugate()), largerEps)); m3 = m1.template triangularView(); // check transposed cases: m3 = m1.template triangularView(); VERIFY((v1.transpose() * m3).isApprox(v1.transpose() * m1.template triangularView(), largerEps)); VERIFY((v1.adjoint() * m3).isApprox(v1.adjoint() * m1.template triangularView(), largerEps)); - VERIFY((v1.adjoint() * m3.adjoint()).isApprox(v1.adjoint() * m1.template triangularView().adjoint(), largerEps)); + VERIFY((v1.adjoint() * m3.adjoint()) + .isApprox(v1.adjoint() * m1.template triangularView().adjoint(), largerEps)); // TODO check with sub-matrices } @@ -73,18 +76,18 @@ template void trmv(const MatrixType& m) void test_product_trmv() { int s = 0; - for(int i = 0; i < g_repeat ; i++) { - CALL_SUBTEST_1( trmv(Matrix()) ); - CALL_SUBTEST_2( trmv(Matrix()) ); - CALL_SUBTEST_3( trmv(Matrix3d()) ); - - s = internal::random(1,EIGEN_TEST_MAX_SIZE/2); - CALL_SUBTEST_4( trmv(MatrixXcf(s,s)) ); - CALL_SUBTEST_5( trmv(MatrixXcd(s,s)) ); + for (int i = 0; i < g_repeat; i++) { + CALL_SUBTEST_1(trmv(Matrix())); + CALL_SUBTEST_2(trmv(Matrix())); + CALL_SUBTEST_3(trmv(Matrix3d())); + + s = internal::random(1, EIGEN_TEST_MAX_SIZE / 2); + CALL_SUBTEST_4(trmv(MatrixXcf(s, s))); + CALL_SUBTEST_5(trmv(MatrixXcd(s, s))); TEST_SET_BUT_UNUSED_VARIABLE(s) - - s = internal::random(1,EIGEN_TEST_MAX_SIZE); - CALL_SUBTEST_6( trmv(Matrix(s, s)) ); + + s = internal::random(1, EIGEN_TEST_MAX_SIZE); + CALL_SUBTEST_6(trmv(Matrix(s, s))); TEST_SET_BUT_UNUSED_VARIABLE(s) } } diff --git a/filmulator-gui/core/nlmeans/eigen/test/product_trsolve.cpp b/filmulator-gui/core/nlmeans/eigen/test/product_trsolve.cpp index 4b97fa9d..0d61f11a 100644 --- a/filmulator-gui/core/nlmeans/eigen/test/product_trsolve.cpp +++ b/filmulator-gui/core/nlmeans/eigen/test/product_trsolve.cpp @@ -9,93 +9,104 @@ #include "main.h" -#define VERIFY_TRSM(TRI,XB) { \ - (XB).setRandom(); ref = (XB); \ - (TRI).solveInPlace(XB); \ +#define VERIFY_TRSM(TRI, XB) \ + { \ + (XB).setRandom(); \ + ref = (XB); \ + (TRI).solveInPlace(XB); \ VERIFY_IS_APPROX((TRI).toDenseMatrix() * (XB), ref); \ - (XB).setRandom(); ref = (XB); \ - (XB) = (TRI).solve(XB); \ + (XB).setRandom(); \ + ref = (XB); \ + (XB) = (TRI).solve(XB); \ VERIFY_IS_APPROX((TRI).toDenseMatrix() * (XB), ref); \ } -#define VERIFY_TRSM_ONTHERIGHT(TRI,XB) { \ - (XB).setRandom(); ref = (XB); \ - (TRI).transpose().template solveInPlace(XB.transpose()); \ +#define VERIFY_TRSM_ONTHERIGHT(TRI, XB) \ + { \ + (XB).setRandom(); \ + ref = (XB); \ + (TRI).transpose().template solveInPlace(XB.transpose()); \ VERIFY_IS_APPROX((XB).transpose() * (TRI).transpose().toDenseMatrix(), ref.transpose()); \ - (XB).setRandom(); ref = (XB); \ - (XB).transpose() = (TRI).transpose().template solve(XB.transpose()); \ + (XB).setRandom(); \ + ref = (XB); \ + (XB).transpose() = (TRI).transpose().template solve(XB.transpose()); \ VERIFY_IS_APPROX((XB).transpose() * (TRI).transpose().toDenseMatrix(), ref.transpose()); \ } -template void trsolve(int size=Size,int cols=Cols) +template void trsolve(int size = Size, int cols = Cols) { typedef typename NumTraits::Real RealScalar; - Matrix cmLhs(size,size); - Matrix rmLhs(size,size); + Matrix cmLhs(size, size); + Matrix rmLhs(size, size); - enum { colmajor = Size==1 ? RowMajor : ColMajor, - rowmajor = Cols==1 ? ColMajor : RowMajor }; - Matrix cmRhs(size,cols); - Matrix rmRhs(size,cols); - Matrix ref(size,cols); + enum { colmajor = Size == 1 ? RowMajor : ColMajor, rowmajor = Cols == 1 ? ColMajor : RowMajor }; + Matrix cmRhs(size, cols); + Matrix rmRhs(size, cols); + Matrix ref(size, cols); - cmLhs.setRandom(); cmLhs *= static_cast(0.1); cmLhs.diagonal().array() += static_cast(1); - rmLhs.setRandom(); rmLhs *= static_cast(0.1); rmLhs.diagonal().array() += static_cast(1); + cmLhs.setRandom(); + cmLhs *= static_cast(0.1); + cmLhs.diagonal().array() += static_cast(1); + rmLhs.setRandom(); + rmLhs *= static_cast(0.1); + rmLhs.diagonal().array() += static_cast(1); VERIFY_TRSM(cmLhs.conjugate().template triangularView(), cmRhs); - VERIFY_TRSM(cmLhs.adjoint() .template triangularView(), cmRhs); - VERIFY_TRSM(cmLhs .template triangularView(), cmRhs); - VERIFY_TRSM(cmLhs .template triangularView(), rmRhs); + VERIFY_TRSM(cmLhs.adjoint().template triangularView(), cmRhs); + VERIFY_TRSM(cmLhs.template triangularView(), cmRhs); + VERIFY_TRSM(cmLhs.template triangularView(), rmRhs); VERIFY_TRSM(cmLhs.conjugate().template triangularView(), rmRhs); - VERIFY_TRSM(cmLhs.adjoint() .template triangularView(), rmRhs); + VERIFY_TRSM(cmLhs.adjoint().template triangularView(), rmRhs); VERIFY_TRSM(cmLhs.conjugate().template triangularView(), cmRhs); - VERIFY_TRSM(cmLhs .template triangularView(), rmRhs); + VERIFY_TRSM(cmLhs.template triangularView(), rmRhs); - VERIFY_TRSM(rmLhs .template triangularView(), cmRhs); + VERIFY_TRSM(rmLhs.template triangularView(), cmRhs); VERIFY_TRSM(rmLhs.conjugate().template triangularView(), rmRhs); VERIFY_TRSM_ONTHERIGHT(cmLhs.conjugate().template triangularView(), cmRhs); - VERIFY_TRSM_ONTHERIGHT(cmLhs .template triangularView(), cmRhs); - VERIFY_TRSM_ONTHERIGHT(cmLhs .template triangularView(), rmRhs); + VERIFY_TRSM_ONTHERIGHT(cmLhs.template triangularView(), cmRhs); + VERIFY_TRSM_ONTHERIGHT(cmLhs.template triangularView(), rmRhs); VERIFY_TRSM_ONTHERIGHT(cmLhs.conjugate().template triangularView(), rmRhs); VERIFY_TRSM_ONTHERIGHT(cmLhs.conjugate().template triangularView(), cmRhs); - VERIFY_TRSM_ONTHERIGHT(cmLhs .template triangularView(), rmRhs); + VERIFY_TRSM_ONTHERIGHT(cmLhs.template triangularView(), rmRhs); - VERIFY_TRSM_ONTHERIGHT(rmLhs .template triangularView(), cmRhs); + VERIFY_TRSM_ONTHERIGHT(rmLhs.template triangularView(), cmRhs); VERIFY_TRSM_ONTHERIGHT(rmLhs.conjugate().template triangularView(), rmRhs); - int c = internal::random(0,cols-1); + int c = internal::random(0, cols - 1); VERIFY_TRSM(rmLhs.template triangularView(), rmRhs.col(c)); VERIFY_TRSM(cmLhs.template triangularView(), rmRhs.col(c)); } void test_product_trsolve() { - for(int i = 0; i < g_repeat ; i++) - { + for (int i = 0; i < g_repeat; i++) { // matrices - CALL_SUBTEST_1((trsolve(internal::random(1,EIGEN_TEST_MAX_SIZE),internal::random(1,EIGEN_TEST_MAX_SIZE)))); - CALL_SUBTEST_2((trsolve(internal::random(1,EIGEN_TEST_MAX_SIZE),internal::random(1,EIGEN_TEST_MAX_SIZE)))); - CALL_SUBTEST_3((trsolve,Dynamic,Dynamic>(internal::random(1,EIGEN_TEST_MAX_SIZE/2),internal::random(1,EIGEN_TEST_MAX_SIZE/2)))); - CALL_SUBTEST_4((trsolve,Dynamic,Dynamic>(internal::random(1,EIGEN_TEST_MAX_SIZE/2),internal::random(1,EIGEN_TEST_MAX_SIZE/2)))); + CALL_SUBTEST_1((trsolve( + internal::random(1, EIGEN_TEST_MAX_SIZE), internal::random(1, EIGEN_TEST_MAX_SIZE)))); + CALL_SUBTEST_2((trsolve( + internal::random(1, EIGEN_TEST_MAX_SIZE), internal::random(1, EIGEN_TEST_MAX_SIZE)))); + CALL_SUBTEST_3((trsolve, Dynamic, Dynamic>( + internal::random(1, EIGEN_TEST_MAX_SIZE / 2), internal::random(1, EIGEN_TEST_MAX_SIZE / 2)))); + CALL_SUBTEST_4((trsolve, Dynamic, Dynamic>( + internal::random(1, EIGEN_TEST_MAX_SIZE / 2), internal::random(1, EIGEN_TEST_MAX_SIZE / 2)))); // vectors - CALL_SUBTEST_5((trsolve(internal::random(1,EIGEN_TEST_MAX_SIZE)))); - CALL_SUBTEST_6((trsolve(internal::random(1,EIGEN_TEST_MAX_SIZE)))); - CALL_SUBTEST_7((trsolve,Dynamic,1>(internal::random(1,EIGEN_TEST_MAX_SIZE)))); - CALL_SUBTEST_8((trsolve,Dynamic,1>(internal::random(1,EIGEN_TEST_MAX_SIZE)))); - + CALL_SUBTEST_5((trsolve(internal::random(1, EIGEN_TEST_MAX_SIZE)))); + CALL_SUBTEST_6((trsolve(internal::random(1, EIGEN_TEST_MAX_SIZE)))); + CALL_SUBTEST_7((trsolve, Dynamic, 1>(internal::random(1, EIGEN_TEST_MAX_SIZE)))); + CALL_SUBTEST_8((trsolve, Dynamic, 1>(internal::random(1, EIGEN_TEST_MAX_SIZE)))); + // meta-unrollers - CALL_SUBTEST_9((trsolve())); - CALL_SUBTEST_10((trsolve())); - CALL_SUBTEST_11((trsolve,4,1>())); - CALL_SUBTEST_12((trsolve())); - CALL_SUBTEST_13((trsolve())); - CALL_SUBTEST_14((trsolve())); - + CALL_SUBTEST_9((trsolve())); + CALL_SUBTEST_10((trsolve())); + CALL_SUBTEST_11((trsolve, 4, 1>())); + CALL_SUBTEST_12((trsolve())); + CALL_SUBTEST_13((trsolve())); + CALL_SUBTEST_14((trsolve())); } } diff --git a/filmulator-gui/core/nlmeans/eigen/test/qr.cpp b/filmulator-gui/core/nlmeans/eigen/test/qr.cpp index 56884605..e67a3037 100644 --- a/filmulator-gui/core/nlmeans/eigen/test/qr.cpp +++ b/filmulator-gui/core/nlmeans/eigen/test/qr.cpp @@ -10,7 +10,7 @@ #include "main.h" #include -template void qr(const MatrixType& m) +template void qr(const MatrixType &m) { Index rows = m.rows(); Index cols = m.cols(); @@ -18,7 +18,7 @@ template void qr(const MatrixType& m) typedef typename MatrixType::Scalar Scalar; typedef Matrix MatrixQType; - MatrixType a = MatrixType::Random(rows,cols); + MatrixType a = MatrixType::Random(rows, cols); HouseholderQR qrOfA(a); MatrixQType q = qrOfA.householderQ(); @@ -32,20 +32,22 @@ template void qr_fixedsize() { enum { Rows = MatrixType::RowsAtCompileTime, Cols = MatrixType::ColsAtCompileTime }; typedef typename MatrixType::Scalar Scalar; - Matrix m1 = Matrix::Random(); - HouseholderQR > qr(m1); + Matrix m1 = Matrix::Random(); + HouseholderQR> qr(m1); - Matrix r = qr.matrixQR(); + Matrix r = qr.matrixQR(); // FIXME need better way to construct trapezoid - for(int i = 0; i < Rows; i++) for(int j = 0; j < Cols; j++) if(i>j) r(i,j) = Scalar(0); + for (int i = 0; i < Rows; i++) + for (int j = 0; j < Cols; j++) + if (i > j) r(i, j) = Scalar(0); VERIFY_IS_APPROX(m1, qr.householderQ() * r); - Matrix m2 = Matrix::Random(Cols,Cols2); - Matrix m3 = m1*m2; - m2 = Matrix::Random(Cols,Cols2); + Matrix m2 = Matrix::Random(Cols, Cols2); + Matrix m3 = m1 * m2; + m2 = Matrix::Random(Cols, Cols2); m2 = qr.solve(m3); - VERIFY_IS_APPROX(m3, m1*m2); + VERIFY_IS_APPROX(m3, m1 * m2); } template void qr_invertible() @@ -57,35 +59,34 @@ template void qr_invertible() typedef typename NumTraits::Real RealScalar; typedef typename MatrixType::Scalar Scalar; - int size = internal::random(10,50); + int size = internal::random(10, 50); MatrixType m1(size, size), m2(size, size), m3(size, size); - m1 = MatrixType::Random(size,size); + m1 = MatrixType::Random(size, size); - if (internal::is_same::value) - { + if (internal::is_same::value) { // let's build a matrix more stable to inverse - MatrixType a = MatrixType::Random(size,size*4); + MatrixType a = MatrixType::Random(size, size * 4); m1 += a * a.adjoint(); } HouseholderQR qr(m1); - m3 = MatrixType::Random(size,size); + m3 = MatrixType::Random(size, size); m2 = qr.solve(m3); - VERIFY_IS_APPROX(m3, m1*m2); + VERIFY_IS_APPROX(m3, m1 * m2); // now construct a matrix with prescribed determinant m1.setZero(); - for(int i = 0; i < size; i++) m1(i,i) = internal::random(); + for (int i = 0; i < size; i++) m1(i, i) = internal::random(); RealScalar absdet = abs(m1.diagonal().prod()); - m3 = qr.householderQ(); // get a unitary + m3 = qr.householderQ();// get a unitary m1 = m3 * m1 * m3; qr.compute(m1); VERIFY_IS_APPROX(log(absdet), qr.logAbsDeterminant()); // This test is tricky if the determinant becomes too small. // Since we generate random numbers with magnitude rrange [0,1], the average determinant is 0.5^size - VERIFY_IS_MUCH_SMALLER_THAN( abs(absdet-qr.absDeterminant()), numext::maxi(RealScalar(pow(0.5,size)),numext::maxi(abs(absdet),abs(qr.absDeterminant()))) ); - + VERIFY_IS_MUCH_SMALLER_THAN(abs(absdet - qr.absDeterminant()), + numext::maxi(RealScalar(pow(0.5, size)), numext::maxi(abs(absdet), abs(qr.absDeterminant())))); } template void qr_verify_assert() @@ -102,20 +103,22 @@ template void qr_verify_assert() void test_qr() { - for(int i = 0; i < g_repeat; i++) { - CALL_SUBTEST_1( qr(MatrixXf(internal::random(1,EIGEN_TEST_MAX_SIZE),internal::random(1,EIGEN_TEST_MAX_SIZE))) ); - CALL_SUBTEST_2( qr(MatrixXcd(internal::random(1,EIGEN_TEST_MAX_SIZE/2),internal::random(1,EIGEN_TEST_MAX_SIZE/2))) ); - CALL_SUBTEST_3(( qr_fixedsize, 2 >() )); - CALL_SUBTEST_4(( qr_fixedsize, 4 >() )); - CALL_SUBTEST_5(( qr_fixedsize, 7 >() )); - CALL_SUBTEST_11( qr(Matrix()) ); + for (int i = 0; i < g_repeat; i++) { + CALL_SUBTEST_1( + qr(MatrixXf(internal::random(1, EIGEN_TEST_MAX_SIZE), internal::random(1, EIGEN_TEST_MAX_SIZE)))); + CALL_SUBTEST_2(qr( + MatrixXcd(internal::random(1, EIGEN_TEST_MAX_SIZE / 2), internal::random(1, EIGEN_TEST_MAX_SIZE / 2)))); + CALL_SUBTEST_3((qr_fixedsize, 2>())); + CALL_SUBTEST_4((qr_fixedsize, 4>())); + CALL_SUBTEST_5((qr_fixedsize, 7>())); + CALL_SUBTEST_11(qr(Matrix())); } - for(int i = 0; i < g_repeat; i++) { - CALL_SUBTEST_1( qr_invertible() ); - CALL_SUBTEST_6( qr_invertible() ); - CALL_SUBTEST_7( qr_invertible() ); - CALL_SUBTEST_8( qr_invertible() ); + for (int i = 0; i < g_repeat; i++) { + CALL_SUBTEST_1(qr_invertible()); + CALL_SUBTEST_6(qr_invertible()); + CALL_SUBTEST_7(qr_invertible()); + CALL_SUBTEST_8(qr_invertible()); } CALL_SUBTEST_9(qr_verify_assert()); diff --git a/filmulator-gui/core/nlmeans/eigen/test/qr_colpivoting.cpp b/filmulator-gui/core/nlmeans/eigen/test/qr_colpivoting.cpp index 96c0badb..a542aae6 100644 --- a/filmulator-gui/core/nlmeans/eigen/test/qr_colpivoting.cpp +++ b/filmulator-gui/core/nlmeans/eigen/test/qr_colpivoting.cpp @@ -12,17 +12,15 @@ #include #include -template -void cod() { +template void cod() +{ Index rows = internal::random(2, EIGEN_TEST_MAX_SIZE); Index cols = internal::random(2, EIGEN_TEST_MAX_SIZE); Index cols2 = internal::random(2, EIGEN_TEST_MAX_SIZE); Index rank = internal::random(1, (std::min)(rows, cols) - 1); typedef typename MatrixType::Scalar Scalar; - typedef Matrix - MatrixQType; + typedef Matrix MatrixQType; MatrixType matrix; createRandomPIMatrixOfRank(rank, rows, cols, matrix); CompleteOrthogonalDecomposition cod(matrix); @@ -40,8 +38,7 @@ void cod() { MatrixType t; t.setZero(rows, cols); - t.topLeftCorner(rank, rank) = - cod.matrixT().topLeftCorner(rank, rank).template triangularView(); + t.topLeftCorner(rank, rank) = cod.matrixT().topLeftCorner(rank, rank).template triangularView(); MatrixType c = q * t * z * cod.colsPermutation().inverse(); VERIFY_IS_APPROX(matrix, c); @@ -60,17 +57,14 @@ void cod() { VERIFY_IS_APPROX(cod_solution, pinv * rhs); } -template -void cod_fixedsize() { - enum { - Rows = MatrixType::RowsAtCompileTime, - Cols = MatrixType::ColsAtCompileTime - }; +template void cod_fixedsize() +{ + enum { Rows = MatrixType::RowsAtCompileTime, Cols = MatrixType::ColsAtCompileTime }; typedef typename MatrixType::Scalar Scalar; int rank = internal::random(1, (std::min)(int(Rows), int(Cols)) - 1); Matrix matrix; createRandomPIMatrixOfRank(rank, Rows, Cols, matrix); - CompleteOrthogonalDecomposition > cod(matrix); + CompleteOrthogonalDecomposition> cod(matrix); VERIFY(rank == cod.rank()); VERIFY(Cols - cod.rank() == cod.dimensionOfKernel()); VERIFY(cod.isInjective() == (rank == Rows)); @@ -93,14 +87,15 @@ template void qr() { using std::sqrt; - Index rows = internal::random(2,EIGEN_TEST_MAX_SIZE), cols = internal::random(2,EIGEN_TEST_MAX_SIZE), cols2 = internal::random(2,EIGEN_TEST_MAX_SIZE); - Index rank = internal::random(1, (std::min)(rows, cols)-1); + Index rows = internal::random(2, EIGEN_TEST_MAX_SIZE), cols = internal::random(2, EIGEN_TEST_MAX_SIZE), + cols2 = internal::random(2, EIGEN_TEST_MAX_SIZE); + Index rank = internal::random(1, (std::min)(rows, cols) - 1); typedef typename MatrixType::Scalar Scalar; typedef typename MatrixType::RealScalar RealScalar; typedef Matrix MatrixQType; MatrixType m1; - createRandomPIMatrixOfRank(rank,rows,cols,m1); + createRandomPIMatrixOfRank(rank, rows, cols, m1); ColPivHouseholderQR qr(m1); VERIFY_IS_EQUAL(rank, qr.rank()); VERIFY_IS_EQUAL(cols - qr.rank(), qr.dimensionOfKernel()); @@ -117,8 +112,7 @@ template void qr() // Verify that the absolute value of the diagonal elements in R are // non-increasing until they reach the singularity threshold. - RealScalar threshold = - sqrt(RealScalar(rows)) * numext::abs(r(0, 0)) * NumTraits::epsilon(); + RealScalar threshold = sqrt(RealScalar(rows)) * numext::abs(r(0, 0)) * NumTraits::epsilon(); for (Index i = 0; i < (std::min)(rows, cols) - 1; ++i) { RealScalar x = numext::abs(r(i, i)); RealScalar y = numext::abs(r(i + 1, i + 1)); @@ -127,28 +121,27 @@ template void qr() for (Index j = 0; j < (std::min)(rows, cols); ++j) { std::cout << "i = " << j << ", |r_ii| = " << numext::abs(r(j, j)) << std::endl; } - std::cout << "Failure at i=" << i << ", rank=" << rank - << ", threshold=" << threshold << std::endl; + std::cout << "Failure at i=" << i << ", rank=" << rank << ", threshold=" << threshold << std::endl; } VERIFY_IS_APPROX_OR_LESS_THAN(y, x); } - MatrixType m2 = MatrixType::Random(cols,cols2); - MatrixType m3 = m1*m2; - m2 = MatrixType::Random(cols,cols2); + MatrixType m2 = MatrixType::Random(cols, cols2); + MatrixType m3 = m1 * m2; + m2 = MatrixType::Random(cols, cols2); m2 = qr.solve(m3); - VERIFY_IS_APPROX(m3, m1*m2); + VERIFY_IS_APPROX(m3, m1 * m2); { Index size = rows; do { - m1 = MatrixType::Random(size,size); + m1 = MatrixType::Random(size, size); qr.compute(m1); - } while(!qr.isInvertible()); + } while (!qr.isInvertible()); MatrixType m1_inv = qr.inverse(); - m3 = m1 * MatrixType::Random(size,cols2); + m3 = m1 * MatrixType::Random(size, cols2); m2 = qr.solve(m3); - VERIFY_IS_APPROX(m2, m1_inv*m3); + VERIFY_IS_APPROX(m2, m1_inv * m3); } } @@ -159,29 +152,28 @@ template void qr_fixedsize() enum { Rows = MatrixType::RowsAtCompileTime, Cols = MatrixType::ColsAtCompileTime }; typedef typename MatrixType::Scalar Scalar; typedef typename MatrixType::RealScalar RealScalar; - int rank = internal::random(1, (std::min)(int(Rows), int(Cols))-1); - Matrix m1; - createRandomPIMatrixOfRank(rank,Rows,Cols,m1); - ColPivHouseholderQR > qr(m1); + int rank = internal::random(1, (std::min)(int(Rows), int(Cols)) - 1); + Matrix m1; + createRandomPIMatrixOfRank(rank, Rows, Cols, m1); + ColPivHouseholderQR> qr(m1); VERIFY_IS_EQUAL(rank, qr.rank()); VERIFY_IS_EQUAL(Cols - qr.rank(), qr.dimensionOfKernel()); VERIFY_IS_EQUAL(qr.isInjective(), (rank == Rows)); VERIFY_IS_EQUAL(qr.isSurjective(), (rank == Cols)); VERIFY_IS_EQUAL(qr.isInvertible(), (qr.isInjective() && qr.isSurjective())); - Matrix r = qr.matrixQR().template triangularView(); - Matrix c = qr.householderQ() * r * qr.colsPermutation().inverse(); + Matrix r = qr.matrixQR().template triangularView(); + Matrix c = qr.householderQ() * r * qr.colsPermutation().inverse(); VERIFY_IS_APPROX(m1, c); - Matrix m2 = Matrix::Random(Cols,Cols2); - Matrix m3 = m1*m2; - m2 = Matrix::Random(Cols,Cols2); + Matrix m2 = Matrix::Random(Cols, Cols2); + Matrix m3 = m1 * m2; + m2 = Matrix::Random(Cols, Cols2); m2 = qr.solve(m3); - VERIFY_IS_APPROX(m3, m1*m2); + VERIFY_IS_APPROX(m3, m1 * m2); // Verify that the absolute value of the diagonal elements in R are // non-increasing until they reache the singularity threshold. - RealScalar threshold = - sqrt(RealScalar(Rows)) * (std::abs)(r(0, 0)) * NumTraits::epsilon(); + RealScalar threshold = sqrt(RealScalar(Rows)) * (std::abs)(r(0, 0)) * NumTraits::epsilon(); for (Index i = 0; i < (std::min)(int(Rows), int(Cols)) - 1; ++i) { RealScalar x = numext::abs(r(i, i)); RealScalar y = numext::abs(r(i + 1, i + 1)); @@ -190,8 +182,7 @@ template void qr_fixedsize() for (Index j = 0; j < (std::min)(int(Rows), int(Cols)); ++j) { std::cout << "i = " << j << ", |r_ii| = " << numext::abs(r(j, j)) << std::endl; } - std::cout << "Failure at i=" << i << ", rank=" << rank - << ", threshold=" << threshold << std::endl; + std::cout << "Failure at i=" << i << ", rank=" << rank << ", threshold=" << threshold << std::endl; } VERIFY_IS_APPROX_OR_LESS_THAN(y, x); } @@ -214,10 +205,10 @@ template void qr_kahan_matrix() Index rows = 300, cols = rows; MatrixType m1; - m1.setZero(rows,cols); + m1.setZero(rows, cols); RealScalar s = std::pow(NumTraits::epsilon(), 1.0 / rows); - RealScalar c = std::sqrt(1 - s*s); - RealScalar pow_s_i(1.0); // pow(s,i) + RealScalar c = std::sqrt(1 - s * s); + RealScalar pow_s_i(1.0);// pow(s,i) for (Index i = 0; i < rows; ++i) { m1(i, i) = pow_s_i; m1.row(i).tail(rows - i - 1) = -pow_s_i * c * MatrixType::Ones(1, rows - i - 1); @@ -227,8 +218,7 @@ template void qr_kahan_matrix() ColPivHouseholderQR qr(m1); MatrixType r = qr.matrixQR().template triangularView(); - RealScalar threshold = - std::sqrt(RealScalar(rows)) * numext::abs(r(0, 0)) * NumTraits::epsilon(); + RealScalar threshold = std::sqrt(RealScalar(rows)) * numext::abs(r(0, 0)) * NumTraits::epsilon(); for (Index i = 0; i < (std::min)(rows, cols) - 1; ++i) { RealScalar x = numext::abs(r(i, i)); RealScalar y = numext::abs(r(i + 1, i + 1)); @@ -237,8 +227,7 @@ template void qr_kahan_matrix() for (Index j = 0; j < (std::min)(rows, cols); ++j) { std::cout << "i = " << j << ", |r_ii| = " << numext::abs(r(j, j)) << std::endl; } - std::cout << "Failure at i=" << i << ", rank=" << qr.rank() - << ", threshold=" << threshold << std::endl; + std::cout << "Failure at i=" << i << ", rank=" << qr.rank() << ", threshold=" << threshold << std::endl; } VERIFY_IS_APPROX_OR_LESS_THAN(y, x); } @@ -251,28 +240,27 @@ template void qr_invertible() typedef typename NumTraits::Real RealScalar; typedef typename MatrixType::Scalar Scalar; - int size = internal::random(10,50); + int size = internal::random(10, 50); MatrixType m1(size, size), m2(size, size), m3(size, size); - m1 = MatrixType::Random(size,size); + m1 = MatrixType::Random(size, size); - if (internal::is_same::value) - { + if (internal::is_same::value) { // let's build a matrix more stable to inverse - MatrixType a = MatrixType::Random(size,size*2); + MatrixType a = MatrixType::Random(size, size * 2); m1 += a * a.adjoint(); } ColPivHouseholderQR qr(m1); - m3 = MatrixType::Random(size,size); + m3 = MatrixType::Random(size, size); m2 = qr.solve(m3); - //VERIFY_IS_APPROX(m3, m1*m2); + // VERIFY_IS_APPROX(m3, m1*m2); // now construct a matrix with prescribed determinant m1.setZero(); - for(int i = 0; i < size; i++) m1(i,i) = internal::random(); + for (int i = 0; i < size; i++) m1(i, i) = internal::random(); RealScalar absdet = abs(m1.diagonal().prod()); - m3 = qr.householderQ(); // get a unitary + m3 = qr.householderQ();// get a unitary m1 = m3 * m1 * m3; qr.compute(m1); VERIFY_IS_APPROX(absdet, qr.absDeterminant()); @@ -298,29 +286,29 @@ template void qr_verify_assert() void test_qr_colpivoting() { - for(int i = 0; i < g_repeat; i++) { - CALL_SUBTEST_1( qr() ); - CALL_SUBTEST_2( qr() ); - CALL_SUBTEST_3( qr() ); - CALL_SUBTEST_4(( qr_fixedsize, 4 >() )); - CALL_SUBTEST_5(( qr_fixedsize, 3 >() )); - CALL_SUBTEST_5(( qr_fixedsize, 1 >() )); + for (int i = 0; i < g_repeat; i++) { + CALL_SUBTEST_1(qr()); + CALL_SUBTEST_2(qr()); + CALL_SUBTEST_3(qr()); + CALL_SUBTEST_4((qr_fixedsize, 4>())); + CALL_SUBTEST_5((qr_fixedsize, 3>())); + CALL_SUBTEST_5((qr_fixedsize, 1>())); } - for(int i = 0; i < g_repeat; i++) { - CALL_SUBTEST_1( cod() ); - CALL_SUBTEST_2( cod() ); - CALL_SUBTEST_3( cod() ); - CALL_SUBTEST_4(( cod_fixedsize, 4 >() )); - CALL_SUBTEST_5(( cod_fixedsize, 3 >() )); - CALL_SUBTEST_5(( cod_fixedsize, 1 >() )); + for (int i = 0; i < g_repeat; i++) { + CALL_SUBTEST_1(cod()); + CALL_SUBTEST_2(cod()); + CALL_SUBTEST_3(cod()); + CALL_SUBTEST_4((cod_fixedsize, 4>())); + CALL_SUBTEST_5((cod_fixedsize, 3>())); + CALL_SUBTEST_5((cod_fixedsize, 1>())); } - for(int i = 0; i < g_repeat; i++) { - CALL_SUBTEST_1( qr_invertible() ); - CALL_SUBTEST_2( qr_invertible() ); - CALL_SUBTEST_6( qr_invertible() ); - CALL_SUBTEST_3( qr_invertible() ); + for (int i = 0; i < g_repeat; i++) { + CALL_SUBTEST_1(qr_invertible()); + CALL_SUBTEST_2(qr_invertible()); + CALL_SUBTEST_6(qr_invertible()); + CALL_SUBTEST_3(qr_invertible()); } CALL_SUBTEST_7(qr_verify_assert()); @@ -333,6 +321,6 @@ void test_qr_colpivoting() // Test problem size constructors CALL_SUBTEST_9(ColPivHouseholderQR(10, 20)); - CALL_SUBTEST_1( qr_kahan_matrix() ); - CALL_SUBTEST_2( qr_kahan_matrix() ); + CALL_SUBTEST_1(qr_kahan_matrix()); + CALL_SUBTEST_2(qr_kahan_matrix()); } diff --git a/filmulator-gui/core/nlmeans/eigen/test/qr_fullpivoting.cpp b/filmulator-gui/core/nlmeans/eigen/test/qr_fullpivoting.cpp index 4d8ef686..9f8fae1c 100644 --- a/filmulator-gui/core/nlmeans/eigen/test/qr_fullpivoting.cpp +++ b/filmulator-gui/core/nlmeans/eigen/test/qr_fullpivoting.cpp @@ -14,16 +14,15 @@ template void qr() { Index max_size = EIGEN_TEST_MAX_SIZE; - Index min_size = numext::maxi(1,EIGEN_TEST_MAX_SIZE/10); - Index rows = internal::random(min_size,max_size), - cols = internal::random(min_size,max_size), - cols2 = internal::random(min_size,max_size), - rank = internal::random(1, (std::min)(rows, cols)-1); + Index min_size = numext::maxi(1, EIGEN_TEST_MAX_SIZE / 10); + Index rows = internal::random(min_size, max_size), cols = internal::random(min_size, max_size), + cols2 = internal::random(min_size, max_size), + rank = internal::random(1, (std::min)(rows, cols) - 1); typedef typename MatrixType::Scalar Scalar; typedef Matrix MatrixQType; MatrixType m1; - createRandomPIMatrixOfRank(rank,rows,cols,m1); + createRandomPIMatrixOfRank(rank, rows, cols, m1); FullPivHouseholderQR qr(m1); VERIFY_IS_EQUAL(rank, qr.rank()); VERIFY_IS_EQUAL(cols - qr.rank(), qr.dimensionOfKernel()); @@ -32,37 +31,39 @@ template void qr() VERIFY(!qr.isSurjective()); MatrixType r = qr.matrixQR(); - + MatrixQType q = qr.matrixQ(); VERIFY_IS_UNITARY(q); - + // FIXME need better way to construct trapezoid - for(int i = 0; i < rows; i++) for(int j = 0; j < cols; j++) if(i>j) r(i,j) = Scalar(0); + for (int i = 0; i < rows; i++) + for (int j = 0; j < cols; j++) + if (i > j) r(i, j) = Scalar(0); MatrixType c = qr.matrixQ() * r * qr.colsPermutation().inverse(); VERIFY_IS_APPROX(m1, c); - + // stress the ReturnByValue mechanism MatrixType tmp; VERIFY_IS_APPROX(tmp.noalias() = qr.matrixQ() * r, (qr.matrixQ() * r).eval()); - - MatrixType m2 = MatrixType::Random(cols,cols2); - MatrixType m3 = m1*m2; - m2 = MatrixType::Random(cols,cols2); + + MatrixType m2 = MatrixType::Random(cols, cols2); + MatrixType m3 = m1 * m2; + m2 = MatrixType::Random(cols, cols2); m2 = qr.solve(m3); - VERIFY_IS_APPROX(m3, m1*m2); + VERIFY_IS_APPROX(m3, m1 * m2); { Index size = rows; do { - m1 = MatrixType::Random(size,size); + m1 = MatrixType::Random(size, size); qr.compute(m1); - } while(!qr.isInvertible()); + } while (!qr.isInvertible()); MatrixType m1_inv = qr.inverse(); - m3 = m1 * MatrixType::Random(size,cols2); + m3 = m1 * MatrixType::Random(size, cols2); m2 = qr.solve(m3); - VERIFY_IS_APPROX(m2, m1_inv*m3); + VERIFY_IS_APPROX(m2, m1_inv * m3); } } @@ -73,17 +74,16 @@ template void qr_invertible() typedef typename NumTraits::Real RealScalar; typedef typename MatrixType::Scalar Scalar; - Index max_size = numext::mini(50,EIGEN_TEST_MAX_SIZE); - Index min_size = numext::maxi(1,EIGEN_TEST_MAX_SIZE/10); - Index size = internal::random(min_size,max_size); + Index max_size = numext::mini(50, EIGEN_TEST_MAX_SIZE); + Index min_size = numext::maxi(1, EIGEN_TEST_MAX_SIZE / 10); + Index size = internal::random(min_size, max_size); MatrixType m1(size, size), m2(size, size), m3(size, size); - m1 = MatrixType::Random(size,size); + m1 = MatrixType::Random(size, size); - if (internal::is_same::value) - { + if (internal::is_same::value) { // let's build a matrix more stable to inverse - MatrixType a = MatrixType::Random(size,size*2); + MatrixType a = MatrixType::Random(size, size * 2); m1 += a * a.adjoint(); } @@ -92,15 +92,15 @@ template void qr_invertible() VERIFY(qr.isInvertible()); VERIFY(qr.isSurjective()); - m3 = MatrixType::Random(size,size); + m3 = MatrixType::Random(size, size); m2 = qr.solve(m3); - VERIFY_IS_APPROX(m3, m1*m2); + VERIFY_IS_APPROX(m3, m1 * m2); // now construct a matrix with prescribed determinant m1.setZero(); - for(int i = 0; i < size; i++) m1(i,i) = internal::random(); + for (int i = 0; i < size; i++) m1(i, i) = internal::random(); RealScalar absdet = abs(m1.diagonal().prod()); - m3 = qr.matrixQ(); // get a unitary + m3 = qr.matrixQ();// get a unitary m1 = m3 * m1 * m3; qr.compute(m1); VERIFY_IS_APPROX(absdet, qr.absDeterminant()); @@ -126,19 +126,19 @@ template void qr_verify_assert() void test_qr_fullpivoting() { - for(int i = 0; i < 1; i++) { + for (int i = 0; i < 1; i++) { // FIXME : very weird bug here -// CALL_SUBTEST(qr(Matrix2f()) ); - CALL_SUBTEST_1( qr() ); - CALL_SUBTEST_2( qr() ); - CALL_SUBTEST_3( qr() ); + // CALL_SUBTEST(qr(Matrix2f()) ); + CALL_SUBTEST_1(qr()); + CALL_SUBTEST_2(qr()); + CALL_SUBTEST_3(qr()); } - for(int i = 0; i < g_repeat; i++) { - CALL_SUBTEST_1( qr_invertible() ); - CALL_SUBTEST_2( qr_invertible() ); - CALL_SUBTEST_4( qr_invertible() ); - CALL_SUBTEST_3( qr_invertible() ); + for (int i = 0; i < g_repeat; i++) { + CALL_SUBTEST_1(qr_invertible()); + CALL_SUBTEST_2(qr_invertible()); + CALL_SUBTEST_4(qr_invertible()); + CALL_SUBTEST_3(qr_invertible()); } CALL_SUBTEST_5(qr_verify_assert()); @@ -150,8 +150,8 @@ void test_qr_fullpivoting() // Test problem size constructors CALL_SUBTEST_7(FullPivHouseholderQR(10, 20)); - CALL_SUBTEST_7((FullPivHouseholderQR >(10,20))); - CALL_SUBTEST_7((FullPivHouseholderQR >(Matrix::Random()))); - CALL_SUBTEST_7((FullPivHouseholderQR >(20,10))); - CALL_SUBTEST_7((FullPivHouseholderQR >(Matrix::Random()))); + CALL_SUBTEST_7((FullPivHouseholderQR>(10, 20))); + CALL_SUBTEST_7((FullPivHouseholderQR>(Matrix::Random()))); + CALL_SUBTEST_7((FullPivHouseholderQR>(20, 10))); + CALL_SUBTEST_7((FullPivHouseholderQR>(Matrix::Random()))); } diff --git a/filmulator-gui/core/nlmeans/eigen/test/qtvector.cpp b/filmulator-gui/core/nlmeans/eigen/test/qtvector.cpp index 22df0d51..700f64c1 100644 --- a/filmulator-gui/core/nlmeans/eigen/test/qtvector.cpp +++ b/filmulator-gui/core/nlmeans/eigen/test/qtvector.cpp @@ -11,34 +11,27 @@ #define EIGEN_WORK_AROUND_QT_BUG_CALLING_WRONG_OPERATOR_NEW_FIXED_IN_QT_4_5 #include "main.h" -#include #include #include +#include -template -void check_qtvector_matrix(const MatrixType& m) +template void check_qtvector_matrix(const MatrixType &m) { Index rows = m.rows(); Index cols = m.cols(); - MatrixType x = MatrixType::Random(rows,cols), y = MatrixType::Random(rows,cols); - QVector v(10, MatrixType(rows,cols)), w(20, y); - for(int i = 0; i < 20; i++) - { - VERIFY_IS_APPROX(w[i], y); - } + MatrixType x = MatrixType::Random(rows, cols), y = MatrixType::Random(rows, cols); + QVector v(10, MatrixType(rows, cols)), w(20, y); + for (int i = 0; i < 20; i++) { VERIFY_IS_APPROX(w[i], y); } v[5] = x; w[6] = v[5]; VERIFY_IS_APPROX(w[6], v[5]); v = w; - for(int i = 0; i < 20; i++) - { - VERIFY_IS_APPROX(w[i], v[i]); - } + for (int i = 0; i < 20; i++) { VERIFY_IS_APPROX(w[i], v[i]); } v.resize(21); v[20] = x; VERIFY_IS_APPROX(v[20], x); - v.fill(y,22); + v.fill(y, 22); VERIFY_IS_APPROX(v[21], y); v.push_back(x); VERIFY_IS_APPROX(v[22], x); @@ -46,17 +39,12 @@ void check_qtvector_matrix(const MatrixType& m) // do a lot of push_back such that the vector gets internally resized // (with memory reallocation) - MatrixType* ref = &w[0]; - for(int i=0; i<30 || ((ref==&w[0]) && i<300); ++i) - v.push_back(w[i%w.size()]); - for(int i=23; i -void check_qtvector_transform(const TransformType&) +template void check_qtvector_transform(const TransformType &) { typedef typename TransformType::MatrixType MatrixType; TransformType x(MatrixType::Random()), y(MatrixType::Random()); @@ -65,15 +53,12 @@ void check_qtvector_transform(const TransformType&) w[6] = v[5]; VERIFY_IS_APPROX(w[6], v[5]); v = w; - for(int i = 0; i < 20; i++) - { - VERIFY_IS_APPROX(w[i], v[i]); - } + for (int i = 0; i < 20; i++) { VERIFY_IS_APPROX(w[i], v[i]); } v.resize(21); v[20] = x; VERIFY_IS_APPROX(v[20], x); - v.fill(y,22); + v.fill(y, 22); VERIFY_IS_APPROX(v[21], y); v.push_back(x); VERIFY_IS_APPROX(v[22], x); @@ -81,17 +66,12 @@ void check_qtvector_transform(const TransformType&) // do a lot of push_back such that the vector gets internally resized // (with memory reallocation) - TransformType* ref = &w[0]; - for(int i=0; i<30 || ((ref==&w[0]) && i<300); ++i) - v.push_back(w[i%w.size()]); - for(unsigned int i=23; int(i) -void check_qtvector_quaternion(const QuaternionType&) +template void check_qtvector_quaternion(const QuaternionType &) { typedef typename QuaternionType::Coefficients Coefficients; QuaternionType x(Coefficients::Random()), y(Coefficients::Random()); @@ -100,15 +80,12 @@ void check_qtvector_quaternion(const QuaternionType&) w[6] = v[5]; VERIFY_IS_APPROX(w[6], v[5]); v = w; - for(int i = 0; i < 20; i++) - { - VERIFY_IS_APPROX(w[i], v[i]); - } + for (int i = 0; i < 20; i++) { VERIFY_IS_APPROX(w[i], v[i]); } v.resize(21); v[20] = x; VERIFY_IS_APPROX(v[20], x); - v.fill(y,22); + v.fill(y, 22); VERIFY_IS_APPROX(v[21], y); v.push_back(x); VERIFY_IS_APPROX(v[22], x); @@ -116,13 +93,9 @@ void check_qtvector_quaternion(const QuaternionType&) // do a lot of push_back such that the vector gets internally resized // (with memory reallocation) - QuaternionType* ref = &w[0]; - for(int i=0; i<30 || ((ref==&w[0]) && i<300); ++i) - v.push_back(w[i%w.size()]); - for(unsigned int i=23; int(i) Scalar check_in_range(Scalar x, Scalar y) { - Scalar r = internal::random(x,y); - VERIFY(r>=x); - if(y>=x) - { - VERIFY(r<=y); - } + Scalar r = internal::random(x, y); + VERIFY(r >= x); + if (y >= x) { VERIFY(r <= y); } return r; } template void check_all_in_range(Scalar x, Scalar y) { - Array mask(y-x+1); + Array mask(y - x + 1); mask.fill(0); - long n = (y-x+1)*32; - for(long k=0; k0).all() ); + long n = (y - x + 1) * 32; + for (long k = 0; k < n; ++k) { mask(check_in_range(x, y) - x)++; } + for (Index i = 0; i < mask.size(); ++i) + if (mask(i) == 0) std::cout << "WARNING: value " << x + i << " not reached." << std::endl; + VERIFY((mask > 0).all()); } template void check_histogram(Scalar x, Scalar y, int bins) { - Array hist(bins); + Array hist(bins); hist.fill(0); int f = 100000; - int n = bins*f; - int64 range = int64(y)-int64(x); - int divisor = int((range+1)/bins); - assert(((range+1)%bins)==0); - for(int k=0; k()/double(f))-1.0).abs()<0.02).all() ); + VERIFY((((hist.cast() / double(f)) - 1.0).abs() < 0.02).all()); } void test_rand() { - long long_ref = NumTraits::highest()/10; - signed char char_offset = (std::min)(g_repeat,64); - signed char short_offset = (std::min)(g_repeat,16000); + long long_ref = NumTraits::highest() / 10; + signed char char_offset = (std::min)(g_repeat, 64); + signed char short_offset = (std::min)(g_repeat, 16000); - for(int i = 0; i < g_repeat*10000; i++) { - CALL_SUBTEST(check_in_range(10,11)); - CALL_SUBTEST(check_in_range(1.24234523,1.24234523)); - CALL_SUBTEST(check_in_range(-1,1)); - CALL_SUBTEST(check_in_range(-1432.2352,-1432.2352)); + for (int i = 0; i < g_repeat * 10000; i++) { + CALL_SUBTEST(check_in_range(10, 11)); + CALL_SUBTEST(check_in_range(1.24234523, 1.24234523)); + CALL_SUBTEST(check_in_range(-1, 1)); + CALL_SUBTEST(check_in_range(-1432.2352, -1432.2352)); - CALL_SUBTEST(check_in_range(10,11)); - CALL_SUBTEST(check_in_range(1.24234523,1.24234523)); - CALL_SUBTEST(check_in_range(-1,1)); - CALL_SUBTEST(check_in_range(-1432.2352,-1432.2352)); + CALL_SUBTEST(check_in_range(10, 11)); + CALL_SUBTEST(check_in_range(1.24234523, 1.24234523)); + CALL_SUBTEST(check_in_range(-1, 1)); + CALL_SUBTEST(check_in_range(-1432.2352, -1432.2352)); - CALL_SUBTEST(check_in_range(0,-1)); - CALL_SUBTEST(check_in_range(0,-1)); - CALL_SUBTEST(check_in_range(0,-1)); - CALL_SUBTEST(check_in_range(-673456,673456)); - CALL_SUBTEST(check_in_range(-RAND_MAX+10,RAND_MAX-10)); - CALL_SUBTEST(check_in_range(-24345,24345)); - CALL_SUBTEST(check_in_range(-long_ref,long_ref)); + CALL_SUBTEST(check_in_range(0, -1)); + CALL_SUBTEST(check_in_range(0, -1)); + CALL_SUBTEST(check_in_range(0, -1)); + CALL_SUBTEST(check_in_range(-673456, 673456)); + CALL_SUBTEST(check_in_range(-RAND_MAX + 10, RAND_MAX - 10)); + CALL_SUBTEST(check_in_range(-24345, 24345)); + CALL_SUBTEST(check_in_range(-long_ref, long_ref)); } - CALL_SUBTEST(check_all_in_range(11,11)); - CALL_SUBTEST(check_all_in_range(11,11+char_offset)); - CALL_SUBTEST(check_all_in_range(-5,5)); - CALL_SUBTEST(check_all_in_range(-11-char_offset,-11)); - CALL_SUBTEST(check_all_in_range(-126,-126+char_offset)); - CALL_SUBTEST(check_all_in_range(126-char_offset,126)); - CALL_SUBTEST(check_all_in_range(-126,126)); + CALL_SUBTEST(check_all_in_range(11, 11)); + CALL_SUBTEST(check_all_in_range(11, 11 + char_offset)); + CALL_SUBTEST(check_all_in_range(-5, 5)); + CALL_SUBTEST(check_all_in_range(-11 - char_offset, -11)); + CALL_SUBTEST(check_all_in_range(-126, -126 + char_offset)); + CALL_SUBTEST(check_all_in_range(126 - char_offset, 126)); + CALL_SUBTEST(check_all_in_range(-126, 126)); - CALL_SUBTEST(check_all_in_range(11,11)); - CALL_SUBTEST(check_all_in_range(11,11+short_offset)); - CALL_SUBTEST(check_all_in_range(-5,5)); - CALL_SUBTEST(check_all_in_range(-11-short_offset,-11)); - CALL_SUBTEST(check_all_in_range(-24345,-24345+short_offset)); - CALL_SUBTEST(check_all_in_range(24345,24345+short_offset)); + CALL_SUBTEST(check_all_in_range(11, 11)); + CALL_SUBTEST(check_all_in_range(11, 11 + short_offset)); + CALL_SUBTEST(check_all_in_range(-5, 5)); + CALL_SUBTEST(check_all_in_range(-11 - short_offset, -11)); + CALL_SUBTEST(check_all_in_range(-24345, -24345 + short_offset)); + CALL_SUBTEST(check_all_in_range(24345, 24345 + short_offset)); - CALL_SUBTEST(check_all_in_range(11,11)); - CALL_SUBTEST(check_all_in_range(11,11+g_repeat)); - CALL_SUBTEST(check_all_in_range(-5,5)); - CALL_SUBTEST(check_all_in_range(-11-g_repeat,-11)); - CALL_SUBTEST(check_all_in_range(-673456,-673456+g_repeat)); - CALL_SUBTEST(check_all_in_range(673456,673456+g_repeat)); + CALL_SUBTEST(check_all_in_range(11, 11)); + CALL_SUBTEST(check_all_in_range(11, 11 + g_repeat)); + CALL_SUBTEST(check_all_in_range(-5, 5)); + CALL_SUBTEST(check_all_in_range(-11 - g_repeat, -11)); + CALL_SUBTEST(check_all_in_range(-673456, -673456 + g_repeat)); + CALL_SUBTEST(check_all_in_range(673456, 673456 + g_repeat)); - CALL_SUBTEST(check_all_in_range(11,11)); - CALL_SUBTEST(check_all_in_range(11,11+g_repeat)); - CALL_SUBTEST(check_all_in_range(-5,5)); - CALL_SUBTEST(check_all_in_range(-11-g_repeat,-11)); - CALL_SUBTEST(check_all_in_range(-long_ref,-long_ref+g_repeat)); - CALL_SUBTEST(check_all_in_range( long_ref, long_ref+g_repeat)); + CALL_SUBTEST(check_all_in_range(11, 11)); + CALL_SUBTEST(check_all_in_range(11, 11 + g_repeat)); + CALL_SUBTEST(check_all_in_range(-5, 5)); + CALL_SUBTEST(check_all_in_range(-11 - g_repeat, -11)); + CALL_SUBTEST(check_all_in_range(-long_ref, -long_ref + g_repeat)); + CALL_SUBTEST(check_all_in_range(long_ref, long_ref + g_repeat)); - CALL_SUBTEST(check_histogram(-5,5,11)); + CALL_SUBTEST(check_histogram(-5, 5, 11)); int bins = 100; - CALL_SUBTEST(check_histogram(-3333,-3333+bins*(3333/bins)-1,bins)); + CALL_SUBTEST(check_histogram(-3333, -3333 + bins * (3333 / bins) - 1, bins)); bins = 1000; - CALL_SUBTEST(check_histogram(-RAND_MAX+10,-RAND_MAX+10+bins*(RAND_MAX/bins)-1,bins)); - CALL_SUBTEST(check_histogram(-RAND_MAX+10,-int64(RAND_MAX)+10+bins*(2*int64(RAND_MAX)/bins)-1,bins)); + CALL_SUBTEST(check_histogram(-RAND_MAX + 10, -RAND_MAX + 10 + bins * (RAND_MAX / bins) - 1, bins)); + CALL_SUBTEST( + check_histogram(-RAND_MAX + 10, -int64(RAND_MAX) + 10 + bins * (2 * int64(RAND_MAX) / bins) - 1, bins)); } diff --git a/filmulator-gui/core/nlmeans/eigen/test/real_qz.cpp b/filmulator-gui/core/nlmeans/eigen/test/real_qz.cpp index 3c1492e4..f535f522 100644 --- a/filmulator-gui/core/nlmeans/eigen/test/real_qz.cpp +++ b/filmulator-gui/core/nlmeans/eigen/test/real_qz.cpp @@ -9,86 +9,89 @@ #define EIGEN_RUNTIME_NO_MALLOC #include "main.h" -#include #include +#include -template void real_qz(const MatrixType& m) +template void real_qz(const MatrixType &m) { /* this test covers the following files: RealQZ.h */ using std::abs; typedef typename MatrixType::Scalar Scalar; - + Index dim = m.cols(); - - MatrixType A = MatrixType::Random(dim,dim), - B = MatrixType::Random(dim,dim); + + MatrixType A = MatrixType::Random(dim, dim), B = MatrixType::Random(dim, dim); // Regression test for bug 985: Randomly set rows or columns to zero - Index k=internal::random(0, dim-1); - switch(internal::random(0,10)) { + Index k = internal::random(0, dim - 1); + switch (internal::random(0, 10)) { case 0: - A.row(k).setZero(); break; + A.row(k).setZero(); + break; case 1: - A.col(k).setZero(); break; + A.col(k).setZero(); + break; case 2: - B.row(k).setZero(); break; + B.row(k).setZero(); + break; case 3: - B.col(k).setZero(); break; + B.col(k).setZero(); + break; default: break; } RealQZ qz(dim); - // TODO enable full-prealocation of required memory, this probably requires an in-place mode for HessenbergDecomposition - //Eigen::internal::set_is_malloc_allowed(false); - qz.compute(A,B); - //Eigen::internal::set_is_malloc_allowed(true); - + // TODO enable full-prealocation of required memory, this probably requires an in-place mode for + // HessenbergDecomposition + // Eigen::internal::set_is_malloc_allowed(false); + qz.compute(A, B); + // Eigen::internal::set_is_malloc_allowed(true); + VERIFY_IS_EQUAL(qz.info(), Success); // check for zeros bool all_zeros = true; - for (Index i=0; i void matrixRedux(const MatrixType& m) +template void matrixRedux(const MatrixType &m) { typedef typename MatrixType::Scalar Scalar; typedef typename MatrixType::RealScalar RealScalar; @@ -29,17 +29,18 @@ template void matrixRedux(const MatrixType& m) MatrixType m1_for_prod = MatrixType::Ones(rows, cols) + RealScalar(0.2) * m1; VERIFY_IS_MUCH_SMALLER_THAN(MatrixType::Zero(rows, cols).sum(), Scalar(1)); - VERIFY_IS_APPROX(MatrixType::Ones(rows, cols).sum(), Scalar(float(rows*cols))); // the float() here to shut up excessive MSVC warning about int->complex conversion being lossy + VERIFY_IS_APPROX(MatrixType::Ones(rows, cols).sum(), + Scalar(float( + rows * cols)));// the float() here to shut up excessive MSVC warning about int->complex conversion being lossy Scalar s(0), p(1), minc(numext::real(m1.coeff(0))), maxc(numext::real(m1.coeff(0))); - for(int j = 0; j < cols; j++) - for(int i = 0; i < rows; i++) - { - s += m1(i,j); - p *= m1_for_prod(i,j); - minc = (std::min)(numext::real(minc), numext::real(m1(i,j))); - maxc = (std::max)(numext::real(maxc), numext::real(m1(i,j))); - } - const Scalar mean = s/Scalar(RealScalar(rows*cols)); + for (int j = 0; j < cols; j++) + for (int i = 0; i < rows; i++) { + s += m1(i, j); + p *= m1_for_prod(i, j); + minc = (std::min)(numext::real(minc), numext::real(m1(i, j))); + maxc = (std::max)(numext::real(maxc), numext::real(m1(i, j))); + } + const Scalar mean = s / Scalar(RealScalar(rows * cols)); VERIFY_IS_APPROX(m1.sum(), s); VERIFY_IS_APPROX(m1.mean(), mean); @@ -48,36 +49,37 @@ template void matrixRedux(const MatrixType& m) VERIFY_IS_APPROX(m1.real().maxCoeff(), numext::real(maxc)); // test slice vectorization assuming assign is ok - Index r0 = internal::random(0,rows-1); - Index c0 = internal::random(0,cols-1); - Index r1 = internal::random(r0+1,rows)-r0; - Index c1 = internal::random(c0+1,cols)-c0; - VERIFY_IS_APPROX(m1.block(r0,c0,r1,c1).sum(), m1.block(r0,c0,r1,c1).eval().sum()); - VERIFY_IS_APPROX(m1.block(r0,c0,r1,c1).mean(), m1.block(r0,c0,r1,c1).eval().mean()); - VERIFY_IS_APPROX(m1_for_prod.block(r0,c0,r1,c1).prod(), m1_for_prod.block(r0,c0,r1,c1).eval().prod()); - VERIFY_IS_APPROX(m1.block(r0,c0,r1,c1).real().minCoeff(), m1.block(r0,c0,r1,c1).real().eval().minCoeff()); - VERIFY_IS_APPROX(m1.block(r0,c0,r1,c1).real().maxCoeff(), m1.block(r0,c0,r1,c1).real().eval().maxCoeff()); + Index r0 = internal::random(0, rows - 1); + Index c0 = internal::random(0, cols - 1); + Index r1 = internal::random(r0 + 1, rows) - r0; + Index c1 = internal::random(c0 + 1, cols) - c0; + VERIFY_IS_APPROX(m1.block(r0, c0, r1, c1).sum(), m1.block(r0, c0, r1, c1).eval().sum()); + VERIFY_IS_APPROX(m1.block(r0, c0, r1, c1).mean(), m1.block(r0, c0, r1, c1).eval().mean()); + VERIFY_IS_APPROX(m1_for_prod.block(r0, c0, r1, c1).prod(), m1_for_prod.block(r0, c0, r1, c1).eval().prod()); + VERIFY_IS_APPROX(m1.block(r0, c0, r1, c1).real().minCoeff(), m1.block(r0, c0, r1, c1).real().eval().minCoeff()); + VERIFY_IS_APPROX(m1.block(r0, c0, r1, c1).real().maxCoeff(), m1.block(r0, c0, r1, c1).real().eval().maxCoeff()); // regression for bug 1090 - const int R1 = MatrixType::RowsAtCompileTime>=2 ? MatrixType::RowsAtCompileTime/2 : 6; - const int C1 = MatrixType::ColsAtCompileTime>=2 ? MatrixType::ColsAtCompileTime/2 : 6; - if(R1<=rows-r0 && C1<=cols-c0) - { - VERIFY_IS_APPROX( (m1.template block(r0,c0).sum()), m1.block(r0,c0,R1,C1).sum() ); + const int R1 = MatrixType::RowsAtCompileTime >= 2 ? MatrixType::RowsAtCompileTime / 2 : 6; + const int C1 = MatrixType::ColsAtCompileTime >= 2 ? MatrixType::ColsAtCompileTime / 2 : 6; + if (R1 <= rows - r0 && C1 <= cols - c0) { + VERIFY_IS_APPROX((m1.template block(r0, c0).sum()), m1.block(r0, c0, R1, C1).sum()); } - + // test empty objects - VERIFY_IS_APPROX(m1.block(r0,c0,0,0).sum(), Scalar(0)); - VERIFY_IS_APPROX(m1.block(r0,c0,0,0).prod(), Scalar(1)); + VERIFY_IS_APPROX(m1.block(r0, c0, 0, 0).sum(), Scalar(0)); + VERIFY_IS_APPROX(m1.block(r0, c0, 0, 0).prod(), Scalar(1)); // test nesting complex expression - VERIFY_EVALUATION_COUNT( (m1.matrix()*m1.matrix().transpose()).sum(), (MatrixType::IsVectorAtCompileTime && MatrixType::SizeAtCompileTime!=1 ? 0 : 1) ); - Matrix m2(rows,rows); + VERIFY_EVALUATION_COUNT((m1.matrix() * m1.matrix().transpose()).sum(), + (MatrixType::IsVectorAtCompileTime && MatrixType::SizeAtCompileTime != 1 ? 0 : 1)); + Matrix m2(rows, rows); m2.setRandom(); - VERIFY_EVALUATION_COUNT( ((m1.matrix()*m1.matrix().transpose())+m2).sum(),(MatrixType::IsVectorAtCompileTime && MatrixType::SizeAtCompileTime!=1 ? 0 : 1)); + VERIFY_EVALUATION_COUNT(((m1.matrix() * m1.matrix().transpose()) + m2).sum(), + (MatrixType::IsVectorAtCompileTime && MatrixType::SizeAtCompileTime != 1 ? 0 : 1)); } -template void vectorRedux(const VectorType& w) +template void vectorRedux(const VectorType &w) { using std::abs; typedef typename VectorType::Scalar Scalar; @@ -85,14 +87,12 @@ template void vectorRedux(const VectorType& w) Index size = w.size(); VectorType v = VectorType::Random(size); - VectorType v_for_prod = VectorType::Ones(size) + Scalar(0.2) * v; // see comment above declaration of m1_for_prod + VectorType v_for_prod = VectorType::Ones(size) + Scalar(0.2) * v;// see comment above declaration of m1_for_prod - for(int i = 1; i < size; i++) - { + for (int i = 1; i < size; i++) { Scalar s(0), p(1); RealScalar minc(numext::real(v.coeff(0))), maxc(numext::real(v.coeff(0))); - for(int j = 0; j < i; j++) - { + for (int j = 0; j < i; j++) { s += v[j]; p *= v_for_prod[j]; minc = (std::min)(minc, numext::real(v[j])); @@ -104,43 +104,39 @@ template void vectorRedux(const VectorType& w) VERIFY_IS_APPROX(maxc, v.real().head(i).maxCoeff()); } - for(int i = 0; i < size-1; i++) - { + for (int i = 0; i < size - 1; i++) { Scalar s(0), p(1); RealScalar minc(numext::real(v.coeff(i))), maxc(numext::real(v.coeff(i))); - for(int j = i; j < size; j++) - { + for (int j = i; j < size; j++) { s += v[j]; p *= v_for_prod[j]; minc = (std::min)(minc, numext::real(v[j])); maxc = (std::max)(maxc, numext::real(v[j])); } - VERIFY_IS_MUCH_SMALLER_THAN(abs(s - v.tail(size-i).sum()), Scalar(1)); - VERIFY_IS_APPROX(p, v_for_prod.tail(size-i).prod()); - VERIFY_IS_APPROX(minc, v.real().tail(size-i).minCoeff()); - VERIFY_IS_APPROX(maxc, v.real().tail(size-i).maxCoeff()); + VERIFY_IS_MUCH_SMALLER_THAN(abs(s - v.tail(size - i).sum()), Scalar(1)); + VERIFY_IS_APPROX(p, v_for_prod.tail(size - i).prod()); + VERIFY_IS_APPROX(minc, v.real().tail(size - i).minCoeff()); + VERIFY_IS_APPROX(maxc, v.real().tail(size - i).maxCoeff()); } - for(int i = 0; i < size/2; i++) - { + for (int i = 0; i < size / 2; i++) { Scalar s(0), p(1); RealScalar minc(numext::real(v.coeff(i))), maxc(numext::real(v.coeff(i))); - for(int j = i; j < size-i; j++) - { + for (int j = i; j < size - i; j++) { s += v[j]; p *= v_for_prod[j]; minc = (std::min)(minc, numext::real(v[j])); maxc = (std::max)(maxc, numext::real(v[j])); } - VERIFY_IS_MUCH_SMALLER_THAN(abs(s - v.segment(i, size-2*i).sum()), Scalar(1)); - VERIFY_IS_APPROX(p, v_for_prod.segment(i, size-2*i).prod()); - VERIFY_IS_APPROX(minc, v.real().segment(i, size-2*i).minCoeff()); - VERIFY_IS_APPROX(maxc, v.real().segment(i, size-2*i).maxCoeff()); + VERIFY_IS_MUCH_SMALLER_THAN(abs(s - v.segment(i, size - 2 * i).sum()), Scalar(1)); + VERIFY_IS_APPROX(p, v_for_prod.segment(i, size - 2 * i).prod()); + VERIFY_IS_APPROX(minc, v.real().segment(i, size - 2 * i).minCoeff()); + VERIFY_IS_APPROX(maxc, v.real().segment(i, size - 2 * i).maxCoeff()); } - + // test empty objects - VERIFY_IS_APPROX(v.head(0).sum(), Scalar(0)); - VERIFY_IS_APPROX(v.tail(0).prod(), Scalar(1)); + VERIFY_IS_APPROX(v.head(0).sum(), Scalar(0)); + VERIFY_IS_APPROX(v.tail(0).prod(), Scalar(1)); VERIFY_RAISES_ASSERT(v.head(0).mean()); VERIFY_RAISES_ASSERT(v.head(0).minCoeff()); VERIFY_RAISES_ASSERT(v.head(0).maxCoeff()); @@ -149,30 +145,30 @@ template void vectorRedux(const VectorType& w) void test_redux() { // the max size cannot be too large, otherwise reduxion operations obviously generate large errors. - int maxsize = (std::min)(100,EIGEN_TEST_MAX_SIZE); + int maxsize = (std::min)(100, EIGEN_TEST_MAX_SIZE); TEST_SET_BUT_UNUSED_VARIABLE(maxsize); - for(int i = 0; i < g_repeat; i++) { - CALL_SUBTEST_1( matrixRedux(Matrix()) ); - CALL_SUBTEST_1( matrixRedux(Array()) ); - CALL_SUBTEST_2( matrixRedux(Matrix2f()) ); - CALL_SUBTEST_2( matrixRedux(Array2f()) ); - CALL_SUBTEST_2( matrixRedux(Array22f()) ); - CALL_SUBTEST_3( matrixRedux(Matrix4d()) ); - CALL_SUBTEST_3( matrixRedux(Array4d()) ); - CALL_SUBTEST_3( matrixRedux(Array44d()) ); - CALL_SUBTEST_4( matrixRedux(MatrixXcf(internal::random(1,maxsize), internal::random(1,maxsize))) ); - CALL_SUBTEST_4( matrixRedux(ArrayXXcf(internal::random(1,maxsize), internal::random(1,maxsize))) ); - CALL_SUBTEST_5( matrixRedux(MatrixXd (internal::random(1,maxsize), internal::random(1,maxsize))) ); - CALL_SUBTEST_5( matrixRedux(ArrayXXd (internal::random(1,maxsize), internal::random(1,maxsize))) ); - CALL_SUBTEST_6( matrixRedux(MatrixXi (internal::random(1,maxsize), internal::random(1,maxsize))) ); - CALL_SUBTEST_6( matrixRedux(ArrayXXi (internal::random(1,maxsize), internal::random(1,maxsize))) ); + for (int i = 0; i < g_repeat; i++) { + CALL_SUBTEST_1(matrixRedux(Matrix())); + CALL_SUBTEST_1(matrixRedux(Array())); + CALL_SUBTEST_2(matrixRedux(Matrix2f())); + CALL_SUBTEST_2(matrixRedux(Array2f())); + CALL_SUBTEST_2(matrixRedux(Array22f())); + CALL_SUBTEST_3(matrixRedux(Matrix4d())); + CALL_SUBTEST_3(matrixRedux(Array4d())); + CALL_SUBTEST_3(matrixRedux(Array44d())); + CALL_SUBTEST_4(matrixRedux(MatrixXcf(internal::random(1, maxsize), internal::random(1, maxsize)))); + CALL_SUBTEST_4(matrixRedux(ArrayXXcf(internal::random(1, maxsize), internal::random(1, maxsize)))); + CALL_SUBTEST_5(matrixRedux(MatrixXd(internal::random(1, maxsize), internal::random(1, maxsize)))); + CALL_SUBTEST_5(matrixRedux(ArrayXXd(internal::random(1, maxsize), internal::random(1, maxsize)))); + CALL_SUBTEST_6(matrixRedux(MatrixXi(internal::random(1, maxsize), internal::random(1, maxsize)))); + CALL_SUBTEST_6(matrixRedux(ArrayXXi(internal::random(1, maxsize), internal::random(1, maxsize)))); } - for(int i = 0; i < g_repeat; i++) { - CALL_SUBTEST_7( vectorRedux(Vector4f()) ); - CALL_SUBTEST_7( vectorRedux(Array4f()) ); - CALL_SUBTEST_5( vectorRedux(VectorXd(internal::random(1,maxsize))) ); - CALL_SUBTEST_5( vectorRedux(ArrayXd(internal::random(1,maxsize))) ); - CALL_SUBTEST_8( vectorRedux(VectorXf(internal::random(1,maxsize))) ); - CALL_SUBTEST_8( vectorRedux(ArrayXf(internal::random(1,maxsize))) ); + for (int i = 0; i < g_repeat; i++) { + CALL_SUBTEST_7(vectorRedux(Vector4f())); + CALL_SUBTEST_7(vectorRedux(Array4f())); + CALL_SUBTEST_5(vectorRedux(VectorXd(internal::random(1, maxsize)))); + CALL_SUBTEST_5(vectorRedux(ArrayXd(internal::random(1, maxsize)))); + CALL_SUBTEST_8(vectorRedux(VectorXf(internal::random(1, maxsize)))); + CALL_SUBTEST_8(vectorRedux(ArrayXf(internal::random(1, maxsize)))); } } diff --git a/filmulator-gui/core/nlmeans/eigen/test/ref.cpp b/filmulator-gui/core/nlmeans/eigen/test/ref.cpp index 704495af..72cc1c1d 100644 --- a/filmulator-gui/core/nlmeans/eigen/test/ref.cpp +++ b/filmulator-gui/core/nlmeans/eigen/test/ref.cpp @@ -21,55 +21,54 @@ // Deal with i387 extended precision #if EIGEN_ARCH_i386 && !(EIGEN_ARCH_x86_64) -#if EIGEN_COMP_GNUC_STRICT && EIGEN_GNUC_AT_LEAST(4,4) -#pragma GCC optimize ("-ffloat-store") +#if EIGEN_COMP_GNUC_STRICT && EIGEN_GNUC_AT_LEAST(4, 4) +#pragma GCC optimize("-ffloat-store") #else #undef VERIFY_IS_EQUAL -#define VERIFY_IS_EQUAL(X,Y) VERIFY_IS_APPROX(X,Y) +#define VERIFY_IS_EQUAL(X, Y) VERIFY_IS_APPROX(X, Y) #endif #endif -template void ref_matrix(const MatrixType& m) +template void ref_matrix(const MatrixType &m) { typedef typename MatrixType::Scalar Scalar; typedef typename MatrixType::RealScalar RealScalar; - typedef Matrix DynMatrixType; - typedef Matrix RealDynMatrixType; - + typedef Matrix DynMatrixType; + typedef Matrix RealDynMatrixType; + typedef Ref RefMat; typedef Ref RefDynMat; typedef Ref ConstRefDynMat; - typedef Ref > RefRealMatWithStride; + typedef Ref> RefRealMatWithStride; Index rows = m.rows(), cols = m.cols(); - - MatrixType m1 = MatrixType::Random(rows, cols), - m2 = m1; - - Index i = internal::random(0,rows-1); - Index j = internal::random(0,cols-1); - Index brows = internal::random(1,rows-i); - Index bcols = internal::random(1,cols-j); - + + MatrixType m1 = MatrixType::Random(rows, cols), m2 = m1; + + Index i = internal::random(0, rows - 1); + Index j = internal::random(0, cols - 1); + Index brows = internal::random(1, rows - i); + Index bcols = internal::random(1, cols - j); + RefMat rm0 = m1; VERIFY_IS_EQUAL(rm0, m1); RefDynMat rm1 = m1; VERIFY_IS_EQUAL(rm1, m1); - RefDynMat rm2 = m1.block(i,j,brows,bcols); - VERIFY_IS_EQUAL(rm2, m1.block(i,j,brows,bcols)); + RefDynMat rm2 = m1.block(i, j, brows, bcols); + VERIFY_IS_EQUAL(rm2, m1.block(i, j, brows, bcols)); rm2.setOnes(); - m2.block(i,j,brows,bcols).setOnes(); + m2.block(i, j, brows, bcols).setOnes(); VERIFY_IS_EQUAL(m1, m2); - - m2.block(i,j,brows,bcols).setRandom(); - rm2 = m2.block(i,j,brows,bcols); + + m2.block(i, j, brows, bcols).setRandom(); + rm2 = m2.block(i, j, brows, bcols); VERIFY_IS_EQUAL(m1, m2); - - ConstRefDynMat rm3 = m1.block(i,j,brows,bcols); - m1.block(i,j,brows,bcols) *= 2; - m2.block(i,j,brows,bcols) *= 2; - VERIFY_IS_EQUAL(rm3, m2.block(i,j,brows,bcols)); + + ConstRefDynMat rm3 = m1.block(i, j, brows, bcols); + m1.block(i, j, brows, bcols) *= 2; + m2.block(i, j, brows, bcols) *= 2; + VERIFY_IS_EQUAL(rm3, m2.block(i, j, brows, bcols)); RefRealMatWithStride rm4 = m1.real(); VERIFY_IS_EQUAL(rm4, m2.real()); rm4.array() += 1; @@ -77,56 +76,53 @@ template void ref_matrix(const MatrixType& m) VERIFY_IS_EQUAL(m1, m2); } -template void ref_vector(const VectorType& m) +template void ref_vector(const VectorType &m) { typedef typename VectorType::Scalar Scalar; typedef typename VectorType::RealScalar RealScalar; - typedef Matrix DynMatrixType; - typedef Matrix MatrixType; - typedef Matrix RealDynMatrixType; - + typedef Matrix DynMatrixType; + typedef Matrix MatrixType; + typedef Matrix RealDynMatrixType; + typedef Ref RefMat; typedef Ref RefDynMat; typedef Ref ConstRefDynMat; - typedef Ref > RefRealMatWithStride; - typedef Ref > RefMatWithStride; + typedef Ref> RefRealMatWithStride; + typedef Ref> RefMatWithStride; Index size = m.size(); - - VectorType v1 = VectorType::Random(size), - v2 = v1; - MatrixType mat1 = MatrixType::Random(size,size), - mat2 = mat1, - mat3 = MatrixType::Random(size,size); - - Index i = internal::random(0,size-1); - Index bsize = internal::random(1,size-i); - + + VectorType v1 = VectorType::Random(size), v2 = v1; + MatrixType mat1 = MatrixType::Random(size, size), mat2 = mat1, mat3 = MatrixType::Random(size, size); + + Index i = internal::random(0, size - 1); + Index bsize = internal::random(1, size - i); + RefMat rm0 = v1; VERIFY_IS_EQUAL(rm0, v1); RefDynMat rv1 = v1; VERIFY_IS_EQUAL(rv1, v1); - RefDynMat rv2 = v1.segment(i,bsize); - VERIFY_IS_EQUAL(rv2, v1.segment(i,bsize)); + RefDynMat rv2 = v1.segment(i, bsize); + VERIFY_IS_EQUAL(rv2, v1.segment(i, bsize)); rv2.setOnes(); - v2.segment(i,bsize).setOnes(); + v2.segment(i, bsize).setOnes(); VERIFY_IS_EQUAL(v1, v2); - - v2.segment(i,bsize).setRandom(); - rv2 = v2.segment(i,bsize); + + v2.segment(i, bsize).setRandom(); + rv2 = v2.segment(i, bsize); VERIFY_IS_EQUAL(v1, v2); - - ConstRefDynMat rm3 = v1.segment(i,bsize); - v1.segment(i,bsize) *= 2; - v2.segment(i,bsize) *= 2; - VERIFY_IS_EQUAL(rm3, v2.segment(i,bsize)); - + + ConstRefDynMat rm3 = v1.segment(i, bsize); + v1.segment(i, bsize) *= 2; + v2.segment(i, bsize) *= 2; + VERIFY_IS_EQUAL(rm3, v2.segment(i, bsize)); + RefRealMatWithStride rm4 = v1.real(); VERIFY_IS_EQUAL(rm4, v2.real()); rm4.array() += 1; v2.real().array() += 1; VERIFY_IS_EQUAL(v1, v2); - + RefMatWithStride rm5 = mat1.row(i).transpose(); VERIFY_IS_EQUAL(rm5, mat1.row(i).transpose()); rm5.array() += 1; @@ -137,98 +133,112 @@ template void ref_vector(const VectorType& m) VERIFY_IS_APPROX(mat1, mat2); } -template void check_const_correctness(const PlainObjectType&) +template void check_const_correctness(const PlainObjectType &) { // verify that ref-to-const don't have LvalueBit typedef typename internal::add_const::type ConstPlainObjectType; - VERIFY( !(internal::traits >::Flags & LvalueBit) ); - VERIFY( !(internal::traits >::Flags & LvalueBit) ); - VERIFY( !(Ref::Flags & LvalueBit) ); - VERIFY( !(Ref::Flags & LvalueBit) ); + VERIFY(!(internal::traits>::Flags & LvalueBit)); + VERIFY(!(internal::traits>::Flags & LvalueBit)); + VERIFY(!(Ref::Flags & LvalueBit)); + VERIFY(!(Ref::Flags & LvalueBit)); } -template -EIGEN_DONT_INLINE void call_ref_1(Ref a, const B &b) { VERIFY_IS_EQUAL(a,b); } -template -EIGEN_DONT_INLINE void call_ref_2(const Ref& a, const B &b) { VERIFY_IS_EQUAL(a,b); } -template -EIGEN_DONT_INLINE void call_ref_3(Ref > a, const B &b) { VERIFY_IS_EQUAL(a,b); } -template -EIGEN_DONT_INLINE void call_ref_4(const Ref >& a, const B &b) { VERIFY_IS_EQUAL(a,b); } -template -EIGEN_DONT_INLINE void call_ref_5(Ref > a, const B &b) { VERIFY_IS_EQUAL(a,b); } -template -EIGEN_DONT_INLINE void call_ref_6(const Ref >& a, const B &b) { VERIFY_IS_EQUAL(a,b); } -template -EIGEN_DONT_INLINE void call_ref_7(Ref > a, const B &b) { VERIFY_IS_EQUAL(a,b); } +template EIGEN_DONT_INLINE void call_ref_1(Ref a, const B &b) { VERIFY_IS_EQUAL(a, b); } +template EIGEN_DONT_INLINE void call_ref_2(const Ref &a, const B &b) +{ + VERIFY_IS_EQUAL(a, b); +} +template EIGEN_DONT_INLINE void call_ref_3(Ref> a, const B &b) +{ + VERIFY_IS_EQUAL(a, b); +} +template EIGEN_DONT_INLINE void call_ref_4(const Ref> &a, const B &b) +{ + VERIFY_IS_EQUAL(a, b); +} +template EIGEN_DONT_INLINE void call_ref_5(Ref> a, const B &b) +{ + VERIFY_IS_EQUAL(a, b); +} +template EIGEN_DONT_INLINE void call_ref_6(const Ref> &a, const B &b) +{ + VERIFY_IS_EQUAL(a, b); +} +template EIGEN_DONT_INLINE void call_ref_7(Ref> a, const B &b) +{ + VERIFY_IS_EQUAL(a, b); +} void call_ref() { - VectorXcf ca = VectorXcf::Random(10); - VectorXf a = VectorXf::Random(10); + VectorXcf ca = VectorXcf::Random(10); + VectorXf a = VectorXf::Random(10); RowVectorXf b = RowVectorXf::Random(10); - MatrixXf A = MatrixXf::Random(10,10); + MatrixXf A = MatrixXf::Random(10, 10); RowVector3f c = RowVector3f::Random(); - const VectorXf& ac(a); - VectorBlock ab(a,0,3); - const VectorBlock abc(a,0,3); - - - VERIFY_EVALUATION_COUNT( call_ref_1(a,a), 0); - VERIFY_EVALUATION_COUNT( call_ref_1(b,b.transpose()), 0); -// call_ref_1(ac,a ab(a, 0, 3); + const VectorBlock abc(a, 0, 3); + + + VERIFY_EVALUATION_COUNT(call_ref_1(a, a), 0); + VERIFY_EVALUATION_COUNT(call_ref_1(b, b.transpose()), 0); + // call_ref_1(ac,a RowMatrixXd; -int test_ref_overload_fun1(Ref ) { return 1; } -int test_ref_overload_fun1(Ref ) { return 2; } -int test_ref_overload_fun1(Ref ) { return 3; } +typedef Matrix RowMatrixXd; +int test_ref_overload_fun1(Ref) { return 1; } +int test_ref_overload_fun1(Ref) { return 2; } +int test_ref_overload_fun1(Ref) { return 3; } -int test_ref_overload_fun2(Ref ) { return 4; } -int test_ref_overload_fun2(Ref ) { return 5; } +int test_ref_overload_fun2(Ref) { return 4; } +int test_ref_overload_fun2(Ref) { return 5; } void test_ref_ambiguous(const Ref &A, Ref B) { @@ -241,14 +251,14 @@ void test_ref_overloads() { MatrixXd Ad, Bd; RowMatrixXd rAd, rBd; - VERIFY( test_ref_overload_fun1(Ad)==1 ); - VERIFY( test_ref_overload_fun1(rAd)==2 ); - + VERIFY(test_ref_overload_fun1(Ad) == 1); + VERIFY(test_ref_overload_fun1(rAd) == 2); + MatrixXf Af, Bf; - VERIFY( test_ref_overload_fun2(Ad)==4 ); - VERIFY( test_ref_overload_fun2(Ad+Bd)==4 ); - VERIFY( test_ref_overload_fun2(Af+Bf)==5 ); - + VERIFY(test_ref_overload_fun2(Ad) == 4); + VERIFY(test_ref_overload_fun2(Ad + Bd) == 4); + VERIFY(test_ref_overload_fun2(Af + Bf) == 5); + ArrayXd A, B; test_ref_ambiguous(A, B); } @@ -257,34 +267,34 @@ void test_ref_fixed_size_assert() { Vector4f v4; VectorXf vx(10); - VERIFY_RAISES_STATIC_ASSERT( Ref y = v4; (void)y; ); - VERIFY_RAISES_STATIC_ASSERT( Ref y = vx.head<4>(); (void)y; ); - VERIFY_RAISES_STATIC_ASSERT( Ref y = v4; (void)y; ); - VERIFY_RAISES_STATIC_ASSERT( Ref y = vx.head<4>(); (void)y; ); - VERIFY_RAISES_STATIC_ASSERT( Ref y = 2*v4; (void)y; ); + VERIFY_RAISES_STATIC_ASSERT(Ref y = v4; (void)y;); + VERIFY_RAISES_STATIC_ASSERT(Ref y = vx.head<4>(); (void)y;); + VERIFY_RAISES_STATIC_ASSERT(Ref y = v4; (void)y;); + VERIFY_RAISES_STATIC_ASSERT(Ref y = vx.head<4>(); (void)y;); + VERIFY_RAISES_STATIC_ASSERT(Ref y = 2 * v4; (void)y;); } void test_ref() { - for(int i = 0; i < g_repeat; i++) { - CALL_SUBTEST_1( ref_vector(Matrix()) ); - CALL_SUBTEST_1( check_const_correctness(Matrix()) ); - CALL_SUBTEST_2( ref_vector(Vector4d()) ); - CALL_SUBTEST_2( check_const_correctness(Matrix4d()) ); - CALL_SUBTEST_3( ref_vector(Vector4cf()) ); - CALL_SUBTEST_4( ref_vector(VectorXcf(8)) ); - CALL_SUBTEST_5( ref_vector(VectorXi(12)) ); - CALL_SUBTEST_5( check_const_correctness(VectorXi(12)) ); - - CALL_SUBTEST_1( ref_matrix(Matrix()) ); - CALL_SUBTEST_2( ref_matrix(Matrix4d()) ); - CALL_SUBTEST_1( ref_matrix(Matrix()) ); - CALL_SUBTEST_4( ref_matrix(MatrixXcf(internal::random(1,10),internal::random(1,10))) ); - CALL_SUBTEST_4( ref_matrix(Matrix,10,15>()) ); - CALL_SUBTEST_5( ref_matrix(MatrixXi(internal::random(1,10),internal::random(1,10))) ); - CALL_SUBTEST_6( call_ref() ); + for (int i = 0; i < g_repeat; i++) { + CALL_SUBTEST_1(ref_vector(Matrix())); + CALL_SUBTEST_1(check_const_correctness(Matrix())); + CALL_SUBTEST_2(ref_vector(Vector4d())); + CALL_SUBTEST_2(check_const_correctness(Matrix4d())); + CALL_SUBTEST_3(ref_vector(Vector4cf())); + CALL_SUBTEST_4(ref_vector(VectorXcf(8))); + CALL_SUBTEST_5(ref_vector(VectorXi(12))); + CALL_SUBTEST_5(check_const_correctness(VectorXi(12))); + + CALL_SUBTEST_1(ref_matrix(Matrix())); + CALL_SUBTEST_2(ref_matrix(Matrix4d())); + CALL_SUBTEST_1(ref_matrix(Matrix())); + CALL_SUBTEST_4(ref_matrix(MatrixXcf(internal::random(1, 10), internal::random(1, 10)))); + CALL_SUBTEST_4(ref_matrix(Matrix, 10, 15>())); + CALL_SUBTEST_5(ref_matrix(MatrixXi(internal::random(1, 10), internal::random(1, 10)))); + CALL_SUBTEST_6(call_ref()); } - - CALL_SUBTEST_7( test_ref_overloads() ); - CALL_SUBTEST_7( test_ref_fixed_size_assert() ); + + CALL_SUBTEST_7(test_ref_overloads()); + CALL_SUBTEST_7(test_ref_fixed_size_assert()); } diff --git a/filmulator-gui/core/nlmeans/eigen/test/resize.cpp b/filmulator-gui/core/nlmeans/eigen/test/resize.cpp index 4adaafe5..f74b7a6f 100644 --- a/filmulator-gui/core/nlmeans/eigen/test/resize.cpp +++ b/filmulator-gui/core/nlmeans/eigen/test/resize.cpp @@ -9,14 +9,13 @@ #include "main.h" -template -void resizeLikeTest() +template void resizeLikeTest() { MatrixXf A(rows, cols); MatrixXf B; Matrix C; B.resizeLike(A); - C.resizeLike(B); // Shouldn't crash. + C.resizeLike(B);// Shouldn't crash. VERIFY(B.rows() == rows && B.cols() == cols); VectorXf x(rows); @@ -29,13 +28,13 @@ void resizeLikeTest() VERIFY(x.rows() == cols && x.cols() == 1); } -void resizeLikeTest12() { resizeLikeTest<1,2>(); } -void resizeLikeTest1020() { resizeLikeTest<10,20>(); } -void resizeLikeTest31() { resizeLikeTest<3,1>(); } +void resizeLikeTest12() { resizeLikeTest<1, 2>(); } +void resizeLikeTest1020() { resizeLikeTest<10, 20>(); } +void resizeLikeTest31() { resizeLikeTest<3, 1>(); } void test_resize() { - CALL_SUBTEST(resizeLikeTest12() ); - CALL_SUBTEST(resizeLikeTest1020() ); - CALL_SUBTEST(resizeLikeTest31() ); + CALL_SUBTEST(resizeLikeTest12()); + CALL_SUBTEST(resizeLikeTest1020()); + CALL_SUBTEST(resizeLikeTest31()); } diff --git a/filmulator-gui/core/nlmeans/eigen/test/rvalue_types.cpp b/filmulator-gui/core/nlmeans/eigen/test/rvalue_types.cpp index 8887f1b1..61fb328d 100644 --- a/filmulator-gui/core/nlmeans/eigen/test/rvalue_types.cpp +++ b/filmulator-gui/core/nlmeans/eigen/test/rvalue_types.cpp @@ -14,51 +14,48 @@ using internal::UIntPtr; #if EIGEN_HAS_RVALUE_REFERENCES -template -void rvalue_copyassign(const MatrixType& m) +template void rvalue_copyassign(const MatrixType &m) { typedef typename internal::traits::Scalar Scalar; - + // create a temporary which we are about to destroy by moving MatrixType tmp = m; UIntPtr src_address = reinterpret_cast(tmp.data()); - + // move the temporary to n MatrixType n = std::move(tmp); UIntPtr dst_address = reinterpret_cast(n.data()); - if (MatrixType::RowsAtCompileTime==Dynamic|| MatrixType::ColsAtCompileTime==Dynamic) - { + if (MatrixType::RowsAtCompileTime == Dynamic || MatrixType::ColsAtCompileTime == Dynamic) { // verify that we actually moved the guts VERIFY_IS_EQUAL(src_address, dst_address); } // verify that the content did not change - Scalar abs_diff = (m-n).array().abs().sum(); + Scalar abs_diff = (m - n).array().abs().sum(); VERIFY_IS_EQUAL(abs_diff, Scalar(0)); } #else -template -void rvalue_copyassign(const MatrixType&) {} +template void rvalue_copyassign(const MatrixType &) {} #endif void test_rvalue_types() { - CALL_SUBTEST_1(rvalue_copyassign( MatrixXf::Random(50,50).eval() )); - CALL_SUBTEST_1(rvalue_copyassign( ArrayXXf::Random(50,50).eval() )); + CALL_SUBTEST_1(rvalue_copyassign(MatrixXf::Random(50, 50).eval())); + CALL_SUBTEST_1(rvalue_copyassign(ArrayXXf::Random(50, 50).eval())); + + CALL_SUBTEST_1(rvalue_copyassign(Matrix::Random(50).eval())); + CALL_SUBTEST_1(rvalue_copyassign(Array::Random(50).eval())); - CALL_SUBTEST_1(rvalue_copyassign( Matrix::Random(50).eval() )); - CALL_SUBTEST_1(rvalue_copyassign( Array::Random(50).eval() )); + CALL_SUBTEST_1(rvalue_copyassign(Matrix::Random(50).eval())); + CALL_SUBTEST_1(rvalue_copyassign(Array::Random(50).eval())); - CALL_SUBTEST_1(rvalue_copyassign( Matrix::Random(50).eval() )); - CALL_SUBTEST_1(rvalue_copyassign( Array::Random(50).eval() )); - - CALL_SUBTEST_2(rvalue_copyassign( Array::Random().eval() )); - CALL_SUBTEST_2(rvalue_copyassign( Array::Random().eval() )); - CALL_SUBTEST_2(rvalue_copyassign( Array::Random().eval() )); + CALL_SUBTEST_2(rvalue_copyassign(Array::Random().eval())); + CALL_SUBTEST_2(rvalue_copyassign(Array::Random().eval())); + CALL_SUBTEST_2(rvalue_copyassign(Array::Random().eval())); - CALL_SUBTEST_2(rvalue_copyassign( Array::Random().eval() )); - CALL_SUBTEST_2(rvalue_copyassign( Array::Random().eval() )); - CALL_SUBTEST_2(rvalue_copyassign( Array::Random().eval() )); + CALL_SUBTEST_2(rvalue_copyassign(Array::Random().eval())); + CALL_SUBTEST_2(rvalue_copyassign(Array::Random().eval())); + CALL_SUBTEST_2(rvalue_copyassign(Array::Random().eval())); } diff --git a/filmulator-gui/core/nlmeans/eigen/test/schur_complex.cpp b/filmulator-gui/core/nlmeans/eigen/test/schur_complex.cpp index deb78e44..42d102e3 100644 --- a/filmulator-gui/core/nlmeans/eigen/test/schur_complex.cpp +++ b/filmulator-gui/core/nlmeans/eigen/test/schur_complex.cpp @@ -8,8 +8,8 @@ // with this file, You can obtain one at http://mozilla.org/MPL/2.0/. #include "main.h" -#include #include +#include template void schur(int size = MatrixType::ColsAtCompileTime) { @@ -17,16 +17,14 @@ template void schur(int size = MatrixType::ColsAtCompileTim typedef typename ComplexSchur::ComplexMatrixType ComplexMatrixType; // Test basic functionality: T is triangular and A = U T U* - for(int counter = 0; counter < g_repeat; ++counter) { + for (int counter = 0; counter < g_repeat; ++counter) { MatrixType A = MatrixType::Random(size, size); ComplexSchur schurOfA(A); VERIFY_IS_EQUAL(schurOfA.info(), Success); ComplexMatrixType U = schurOfA.matrixU(); ComplexMatrixType T = schurOfA.matrixT(); - for(int row = 1; row < size; ++row) { - for(int col = 0; col < row; ++col) { - VERIFY(T(row,col) == (typename MatrixType::Scalar)0); - } + for (int row = 1; row < size; ++row) { + for (int col = 0; col < row; ++col) { VERIFY(T(row, col) == (typename MatrixType::Scalar)0); } } VERIFY_IS_APPROX(A.template cast(), U * T * U.adjoint()); } @@ -36,7 +34,7 @@ template void schur(int size = MatrixType::ColsAtCompileTim VERIFY_RAISES_ASSERT(csUninitialized.matrixT()); VERIFY_RAISES_ASSERT(csUninitialized.matrixU()); VERIFY_RAISES_ASSERT(csUninitialized.info()); - + // Test whether compute() and constructor returns same result MatrixType A = MatrixType::Random(size, size); ComplexSchur cs1; @@ -58,8 +56,8 @@ template void schur(int size = MatrixType::ColsAtCompileTim VERIFY_IS_EQUAL(cs3.getMaxIterations(), 1); MatrixType Atriangular = A; - Atriangular.template triangularView().setZero(); - cs3.setMaxIterations(1).compute(Atriangular); // triangular matrices do not need any iterations + Atriangular.template triangularView().setZero(); + cs3.setMaxIterations(1).compute(Atriangular);// triangular matrices do not need any iterations VERIFY_IS_EQUAL(cs3.info(), Success); VERIFY_IS_EQUAL(cs3.matrixT(), Atriangular.template cast()); VERIFY_IS_EQUAL(cs3.matrixU(), ComplexMatrixType::Identity(size, size)); @@ -70,10 +68,9 @@ template void schur(int size = MatrixType::ColsAtCompileTim VERIFY_IS_EQUAL(cs1.matrixT(), csOnlyT.matrixT()); VERIFY_RAISES_ASSERT(csOnlyT.matrixU()); - if (size > 1 && size < 20) - { + if (size > 1 && size < 20) { // Test matrix with NaN - A(0,0) = std::numeric_limits::quiet_NaN(); + A(0, 0) = std::numeric_limits::quiet_NaN(); ComplexSchur csNaN(A); VERIFY_IS_EQUAL(csNaN.info(), NoConvergence); } @@ -81,10 +78,10 @@ template void schur(int size = MatrixType::ColsAtCompileTim void test_schur_complex() { - CALL_SUBTEST_1(( schur() )); - CALL_SUBTEST_2(( schur(internal::random(1,EIGEN_TEST_MAX_SIZE/4)) )); - CALL_SUBTEST_3(( schur, 1, 1> >() )); - CALL_SUBTEST_4(( schur >() )); + CALL_SUBTEST_1((schur())); + CALL_SUBTEST_2((schur(internal::random(1, EIGEN_TEST_MAX_SIZE / 4)))); + CALL_SUBTEST_3((schur, 1, 1>>())); + CALL_SUBTEST_4((schur>())); // Test problem size constructors CALL_SUBTEST_5(ComplexSchur(10)); diff --git a/filmulator-gui/core/nlmeans/eigen/test/schur_real.cpp b/filmulator-gui/core/nlmeans/eigen/test/schur_real.cpp index e5229e6e..8b217345 100644 --- a/filmulator-gui/core/nlmeans/eigen/test/schur_real.cpp +++ b/filmulator-gui/core/nlmeans/eigen/test/schur_real.cpp @@ -8,28 +8,26 @@ // with this file, You can obtain one at http://mozilla.org/MPL/2.0/. #include "main.h" -#include #include +#include -template void verifyIsQuasiTriangular(const MatrixType& T) +template void verifyIsQuasiTriangular(const MatrixType &T) { const Index size = T.cols(); typedef typename MatrixType::Scalar Scalar; // Check T is lower Hessenberg - for(int row = 2; row < size; ++row) { - for(int col = 0; col < row - 1; ++col) { - VERIFY(T(row,col) == Scalar(0)); - } + for (int row = 2; row < size; ++row) { + for (int col = 0; col < row - 1; ++col) { VERIFY(T(row, col) == Scalar(0)); } } // Check that any non-zero on the subdiagonal is followed by a zero and is // part of a 2x2 diagonal block with imaginary eigenvalues. - for(int row = 1; row < size; ++row) { - if (T(row,row-1) != Scalar(0)) { - VERIFY(row == size-1 || T(row+1,row) == 0); - Scalar tr = T(row-1,row-1) + T(row,row); - Scalar det = T(row-1,row-1) * T(row,row) - T(row-1,row) * T(row,row-1); + for (int row = 1; row < size; ++row) { + if (T(row, row - 1) != Scalar(0)) { + VERIFY(row == size - 1 || T(row + 1, row) == 0); + Scalar tr = T(row - 1, row - 1) + T(row, row); + Scalar det = T(row - 1, row - 1) * T(row, row) - T(row - 1, row) * T(row, row - 1); VERIFY(4 * det > tr * tr); } } @@ -38,7 +36,7 @@ template void verifyIsQuasiTriangular(const MatrixType& T) template void schur(int size = MatrixType::ColsAtCompileTime) { // Test basic functionality: T is quasi-triangular and A = U T U* - for(int counter = 0; counter < g_repeat; ++counter) { + for (int counter = 0; counter < g_repeat; ++counter) { MatrixType A = MatrixType::Random(size, size); RealSchur schurOfA(A); VERIFY_IS_EQUAL(schurOfA.info(), Success); @@ -53,7 +51,7 @@ template void schur(int size = MatrixType::ColsAtCompileTim VERIFY_RAISES_ASSERT(rsUninitialized.matrixT()); VERIFY_RAISES_ASSERT(rsUninitialized.matrixU()); VERIFY_RAISES_ASSERT(rsUninitialized.info()); - + // Test whether compute() and constructor returns same result MatrixType A = MatrixType::Random(size, size); RealSchur rs1; @@ -77,10 +75,10 @@ template void schur(int size = MatrixType::ColsAtCompileTim } MatrixType Atriangular = A; - Atriangular.template triangularView().setZero(); - rs3.setMaxIterations(1).compute(Atriangular); // triangular matrices do not need any iterations + Atriangular.template triangularView().setZero(); + rs3.setMaxIterations(1).compute(Atriangular);// triangular matrices do not need any iterations VERIFY_IS_EQUAL(rs3.info(), Success); - VERIFY_IS_APPROX(rs3.matrixT(), Atriangular); // approx because of scaling... + VERIFY_IS_APPROX(rs3.matrixT(), Atriangular);// approx because of scaling... VERIFY_IS_EQUAL(rs3.matrixU(), MatrixType::Identity(size, size)); // Test computation of only T, not U @@ -89,10 +87,9 @@ template void schur(int size = MatrixType::ColsAtCompileTim VERIFY_IS_EQUAL(rs1.matrixT(), rsOnlyT.matrixT()); VERIFY_RAISES_ASSERT(rsOnlyT.matrixU()); - if (size > 2 && size < 20) - { + if (size > 2 && size < 20) { // Test matrix with NaN - A(0,0) = std::numeric_limits::quiet_NaN(); + A(0, 0) = std::numeric_limits::quiet_NaN(); RealSchur rsNaN(A); VERIFY_IS_EQUAL(rsNaN.info(), NoConvergence); } @@ -100,10 +97,10 @@ template void schur(int size = MatrixType::ColsAtCompileTim void test_schur_real() { - CALL_SUBTEST_1(( schur() )); - CALL_SUBTEST_2(( schur(internal::random(1,EIGEN_TEST_MAX_SIZE/4)) )); - CALL_SUBTEST_3(( schur >() )); - CALL_SUBTEST_4(( schur >() )); + CALL_SUBTEST_1((schur())); + CALL_SUBTEST_2((schur(internal::random(1, EIGEN_TEST_MAX_SIZE / 4)))); + CALL_SUBTEST_3((schur>())); + CALL_SUBTEST_4((schur>())); // Test problem size constructors CALL_SUBTEST_5(RealSchur(10)); diff --git a/filmulator-gui/core/nlmeans/eigen/test/selfadjoint.cpp b/filmulator-gui/core/nlmeans/eigen/test/selfadjoint.cpp index bb11cc35..55ec338c 100644 --- a/filmulator-gui/core/nlmeans/eigen/test/selfadjoint.cpp +++ b/filmulator-gui/core/nlmeans/eigen/test/selfadjoint.cpp @@ -13,17 +13,14 @@ // This file tests the basic selfadjointView API, // the related products and decompositions are tested in specific files. -template void selfadjoint(const MatrixType& m) +template void selfadjoint(const MatrixType &m) { typedef typename MatrixType::Scalar Scalar; Index rows = m.rows(); Index cols = m.cols(); - MatrixType m1 = MatrixType::Random(rows, cols), - m2 = MatrixType::Random(rows, cols), - m3(rows, cols), - m4(rows, cols); + MatrixType m1 = MatrixType::Random(rows, cols), m2 = MatrixType::Random(rows, cols), m3(rows, cols), m4(rows, cols); m1.diagonal() = m1.diagonal().real().template cast(); @@ -39,12 +36,12 @@ template void selfadjoint(const MatrixType& m) m3 = m1.template selfadjointView(); m4 = m2; m4 += m1.template selfadjointView(); - VERIFY_IS_APPROX(m4, m2+m3); + VERIFY_IS_APPROX(m4, m2 + m3); m3 = m1.template selfadjointView(); m4 = m2; m4 -= m1.template selfadjointView(); - VERIFY_IS_APPROX(m4, m2-m3); + VERIFY_IS_APPROX(m4, m2 - m3); VERIFY_RAISES_STATIC_ASSERT(m2.template selfadjointView()); VERIFY_RAISES_STATIC_ASSERT(m2.template selfadjointView()); @@ -58,18 +55,17 @@ void bug_159() void test_selfadjoint() { - for(int i = 0; i < g_repeat ; i++) - { - int s = internal::random(1,EIGEN_TEST_MAX_SIZE); - - CALL_SUBTEST_1( selfadjoint(Matrix()) ); - CALL_SUBTEST_2( selfadjoint(Matrix()) ); - CALL_SUBTEST_3( selfadjoint(Matrix3cf()) ); - CALL_SUBTEST_4( selfadjoint(MatrixXcd(s,s)) ); - CALL_SUBTEST_5( selfadjoint(Matrix(s, s)) ); - + for (int i = 0; i < g_repeat; i++) { + int s = internal::random(1, EIGEN_TEST_MAX_SIZE); + + CALL_SUBTEST_1(selfadjoint(Matrix())); + CALL_SUBTEST_2(selfadjoint(Matrix())); + CALL_SUBTEST_3(selfadjoint(Matrix3cf())); + CALL_SUBTEST_4(selfadjoint(MatrixXcd(s, s))); + CALL_SUBTEST_5(selfadjoint(Matrix(s, s))); + TEST_SET_BUT_UNUSED_VARIABLE(s) } - - CALL_SUBTEST_1( bug_159() ); + + CALL_SUBTEST_1(bug_159()); } diff --git a/filmulator-gui/core/nlmeans/eigen/test/simplicial_cholesky.cpp b/filmulator-gui/core/nlmeans/eigen/test/simplicial_cholesky.cpp index 649c817b..61e04268 100644 --- a/filmulator-gui/core/nlmeans/eigen/test/simplicial_cholesky.cpp +++ b/filmulator-gui/core/nlmeans/eigen/test/simplicial_cholesky.cpp @@ -11,15 +11,15 @@ template void test_simplicial_cholesky_T() { - typedef SparseMatrix SparseMatrixType; + typedef SparseMatrix SparseMatrixType; SimplicialCholesky chol_colmajor_lower_amd; SimplicialCholesky chol_colmajor_upper_amd; - SimplicialLLT< SparseMatrixType, Lower> llt_colmajor_lower_amd; - SimplicialLLT< SparseMatrixType, Upper> llt_colmajor_upper_amd; - SimplicialLDLT< SparseMatrixType, Lower> ldlt_colmajor_lower_amd; - SimplicialLDLT< SparseMatrixType, Upper> ldlt_colmajor_upper_amd; - SimplicialLDLT< SparseMatrixType, Lower, NaturalOrdering > ldlt_colmajor_lower_nat; - SimplicialLDLT< SparseMatrixType, Upper, NaturalOrdering > ldlt_colmajor_upper_nat; + SimplicialLLT llt_colmajor_lower_amd; + SimplicialLLT llt_colmajor_upper_amd; + SimplicialLDLT ldlt_colmajor_lower_amd; + SimplicialLDLT ldlt_colmajor_upper_amd; + SimplicialLDLT> ldlt_colmajor_lower_nat; + SimplicialLDLT> ldlt_colmajor_upper_nat; check_sparse_spd_solving(chol_colmajor_lower_amd); check_sparse_spd_solving(chol_colmajor_upper_amd); @@ -27,21 +27,21 @@ template void test_simplicial_cholesky_T() check_sparse_spd_solving(llt_colmajor_upper_amd); check_sparse_spd_solving(ldlt_colmajor_lower_amd); check_sparse_spd_solving(ldlt_colmajor_upper_amd); - + check_sparse_spd_determinant(chol_colmajor_lower_amd); check_sparse_spd_determinant(chol_colmajor_upper_amd); check_sparse_spd_determinant(llt_colmajor_lower_amd); check_sparse_spd_determinant(llt_colmajor_upper_amd); check_sparse_spd_determinant(ldlt_colmajor_lower_amd); check_sparse_spd_determinant(ldlt_colmajor_upper_amd); - + check_sparse_spd_solving(ldlt_colmajor_lower_nat, 300, 1000); check_sparse_spd_solving(ldlt_colmajor_upper_nat, 300, 1000); } void test_simplicial_cholesky() { - CALL_SUBTEST_1(( test_simplicial_cholesky_T() )); - CALL_SUBTEST_2(( test_simplicial_cholesky_T, int>() )); - CALL_SUBTEST_3(( test_simplicial_cholesky_T() )); + CALL_SUBTEST_1((test_simplicial_cholesky_T())); + CALL_SUBTEST_2((test_simplicial_cholesky_T, int>())); + CALL_SUBTEST_3((test_simplicial_cholesky_T())); } diff --git a/filmulator-gui/core/nlmeans/eigen/test/sizeof.cpp b/filmulator-gui/core/nlmeans/eigen/test/sizeof.cpp index 03ad2045..2d78f15c 100644 --- a/filmulator-gui/core/nlmeans/eigen/test/sizeof.cpp +++ b/filmulator-gui/core/nlmeans/eigen/test/sizeof.cpp @@ -9,39 +9,40 @@ #include "main.h" -template void verifySizeOf(const MatrixType&) +template void verifySizeOf(const MatrixType &) { typedef typename MatrixType::Scalar Scalar; - if (MatrixType::RowsAtCompileTime!=Dynamic && MatrixType::ColsAtCompileTime!=Dynamic) - VERIFY_IS_EQUAL(std::ptrdiff_t(sizeof(MatrixType)),std::ptrdiff_t(sizeof(Scalar))*std::ptrdiff_t(MatrixType::SizeAtCompileTime)); + if (MatrixType::RowsAtCompileTime != Dynamic && MatrixType::ColsAtCompileTime != Dynamic) + VERIFY_IS_EQUAL(std::ptrdiff_t(sizeof(MatrixType)), + std::ptrdiff_t(sizeof(Scalar)) * std::ptrdiff_t(MatrixType::SizeAtCompileTime)); else - VERIFY_IS_EQUAL(sizeof(MatrixType),sizeof(Scalar*) + 2 * sizeof(typename MatrixType::Index)); + VERIFY_IS_EQUAL(sizeof(MatrixType), sizeof(Scalar *) + 2 * sizeof(typename MatrixType::Index)); } void test_sizeof() { - CALL_SUBTEST(verifySizeOf(Matrix()) ); - CALL_SUBTEST(verifySizeOf(Array()) ); - CALL_SUBTEST(verifySizeOf(Array()) ); - CALL_SUBTEST(verifySizeOf(Array()) ); - CALL_SUBTEST(verifySizeOf(Array()) ); - CALL_SUBTEST(verifySizeOf(Array()) ); - CALL_SUBTEST(verifySizeOf(Array()) ); - CALL_SUBTEST(verifySizeOf(Array()) ); - CALL_SUBTEST(verifySizeOf(Array()) ); - CALL_SUBTEST(verifySizeOf(Array()) ); - CALL_SUBTEST(verifySizeOf(Array()) ); - CALL_SUBTEST(verifySizeOf(Array()) ); - CALL_SUBTEST(verifySizeOf(Vector2d()) ); - CALL_SUBTEST(verifySizeOf(Vector4f()) ); - CALL_SUBTEST(verifySizeOf(Matrix4d()) ); - CALL_SUBTEST(verifySizeOf(Matrix()) ); - CALL_SUBTEST(verifySizeOf(Matrix()) ); - CALL_SUBTEST(verifySizeOf(MatrixXcf(3, 3)) ); - CALL_SUBTEST(verifySizeOf(MatrixXi(8, 12)) ); - CALL_SUBTEST(verifySizeOf(MatrixXcd(20, 20)) ); - CALL_SUBTEST(verifySizeOf(Matrix()) ); - - VERIFY(sizeof(std::complex) == 2*sizeof(float)); - VERIFY(sizeof(std::complex) == 2*sizeof(double)); + CALL_SUBTEST(verifySizeOf(Matrix())); + CALL_SUBTEST(verifySizeOf(Array())); + CALL_SUBTEST(verifySizeOf(Array())); + CALL_SUBTEST(verifySizeOf(Array())); + CALL_SUBTEST(verifySizeOf(Array())); + CALL_SUBTEST(verifySizeOf(Array())); + CALL_SUBTEST(verifySizeOf(Array())); + CALL_SUBTEST(verifySizeOf(Array())); + CALL_SUBTEST(verifySizeOf(Array())); + CALL_SUBTEST(verifySizeOf(Array())); + CALL_SUBTEST(verifySizeOf(Array())); + CALL_SUBTEST(verifySizeOf(Array())); + CALL_SUBTEST(verifySizeOf(Vector2d())); + CALL_SUBTEST(verifySizeOf(Vector4f())); + CALL_SUBTEST(verifySizeOf(Matrix4d())); + CALL_SUBTEST(verifySizeOf(Matrix())); + CALL_SUBTEST(verifySizeOf(Matrix())); + CALL_SUBTEST(verifySizeOf(MatrixXcf(3, 3))); + CALL_SUBTEST(verifySizeOf(MatrixXi(8, 12))); + CALL_SUBTEST(verifySizeOf(MatrixXcd(20, 20))); + CALL_SUBTEST(verifySizeOf(Matrix())); + + VERIFY(sizeof(std::complex) == 2 * sizeof(float)); + VERIFY(sizeof(std::complex) == 2 * sizeof(double)); } diff --git a/filmulator-gui/core/nlmeans/eigen/test/sizeoverflow.cpp b/filmulator-gui/core/nlmeans/eigen/test/sizeoverflow.cpp index 240d2229..466ccfb8 100644 --- a/filmulator-gui/core/nlmeans/eigen/test/sizeoverflow.cpp +++ b/filmulator-gui/core/nlmeans/eigen/test/sizeoverflow.cpp @@ -9,38 +9,38 @@ #include "main.h" -#define VERIFY_THROWS_BADALLOC(a) { \ - bool threw = false; \ - try { \ - a; \ - } \ - catch (std::bad_alloc&) { threw = true; } \ - VERIFY(threw && "should have thrown bad_alloc: " #a); \ +#define VERIFY_THROWS_BADALLOC(a) \ + { \ + bool threw = false; \ + try { \ + a; \ + } catch (std::bad_alloc &) { \ + threw = true; \ + } \ + VERIFY(threw && "should have thrown bad_alloc: " #a); \ } -template -void triggerMatrixBadAlloc(Index rows, Index cols) +template void triggerMatrixBadAlloc(Index rows, Index cols) { - VERIFY_THROWS_BADALLOC( MatrixType m(rows, cols) ); - VERIFY_THROWS_BADALLOC( MatrixType m; m.resize(rows, cols) ); - VERIFY_THROWS_BADALLOC( MatrixType m; m.conservativeResize(rows, cols) ); + VERIFY_THROWS_BADALLOC(MatrixType m(rows, cols)); + VERIFY_THROWS_BADALLOC(MatrixType m; m.resize(rows, cols)); + VERIFY_THROWS_BADALLOC(MatrixType m; m.conservativeResize(rows, cols)); } -template -void triggerVectorBadAlloc(Index size) +template void triggerVectorBadAlloc(Index size) { - VERIFY_THROWS_BADALLOC( VectorType v(size) ); - VERIFY_THROWS_BADALLOC( VectorType v; v.resize(size) ); - VERIFY_THROWS_BADALLOC( VectorType v; v.conservativeResize(size) ); + VERIFY_THROWS_BADALLOC(VectorType v(size)); + VERIFY_THROWS_BADALLOC(VectorType v; v.resize(size)); + VERIFY_THROWS_BADALLOC(VectorType v; v.conservativeResize(size)); } void test_sizeoverflow() { - // there are 2 levels of overflow checking. first in PlainObjectBase.h we check for overflow in rows*cols computations. - // this is tested in tests of the form times_itself_gives_0 * times_itself_gives_0 - // Then in Memory.h we check for overflow in size * sizeof(T) computations. - // this is tested in tests of the form times_4_gives_0 * sizeof(float) - + // there are 2 levels of overflow checking. first in PlainObjectBase.h we check for overflow in rows*cols + // computations. this is tested in tests of the form times_itself_gives_0 * times_itself_gives_0 Then in Memory.h we + // check for overflow in size * sizeof(T) computations. this is tested in tests of the form times_4_gives_0 * + // sizeof(float) + size_t times_itself_gives_0 = size_t(1) << (8 * sizeof(Index) / 2); VERIFY(times_itself_gives_0 * times_itself_gives_0 == 0); @@ -57,8 +57,8 @@ void test_sizeoverflow() triggerMatrixBadAlloc(times_itself_gives_0, times_itself_gives_0); triggerMatrixBadAlloc(times_itself_gives_0 / 8, times_itself_gives_0); triggerMatrixBadAlloc(times_8_gives_0, 1); - + triggerVectorBadAlloc(times_4_gives_0); - + triggerVectorBadAlloc(times_8_gives_0); } diff --git a/filmulator-gui/core/nlmeans/eigen/test/smallvectors.cpp b/filmulator-gui/core/nlmeans/eigen/test/smallvectors.cpp index 78151139..7f0b6c4c 100644 --- a/filmulator-gui/core/nlmeans/eigen/test/smallvectors.cpp +++ b/filmulator-gui/core/nlmeans/eigen/test/smallvectors.cpp @@ -16,9 +16,7 @@ template void smallVectors() typedef Matrix V3; typedef Matrix V4; typedef Matrix VX; - Scalar x1 = internal::random(), - x2 = internal::random(), - x3 = internal::random(), + Scalar x1 = internal::random(), x2 = internal::random(), x3 = internal::random(), x4 = internal::random(); V2 v2(x1, x2); V3 v3(x1, x2, x3); @@ -33,8 +31,7 @@ template void smallVectors() VERIFY_IS_APPROX(x3, v4.z()); VERIFY_IS_APPROX(x4, v4.w()); - if (!NumTraits::IsInteger) - { + if (!NumTraits::IsInteger) { VERIFY_RAISES_ASSERT(V3(2, 1)) VERIFY_RAISES_ASSERT(V3(3, 2)) VERIFY_RAISES_ASSERT(V3(Scalar(3), 1)) @@ -59,9 +56,9 @@ template void smallVectors() void test_smallvectors() { - for(int i = 0; i < g_repeat; i++) { - CALL_SUBTEST(smallVectors() ); - CALL_SUBTEST(smallVectors() ); - CALL_SUBTEST(smallVectors() ); + for (int i = 0; i < g_repeat; i++) { + CALL_SUBTEST(smallVectors()); + CALL_SUBTEST(smallVectors()); + CALL_SUBTEST(smallVectors()); } } diff --git a/filmulator-gui/core/nlmeans/eigen/test/sparse.h b/filmulator-gui/core/nlmeans/eigen/test/sparse.h index 9912e1e2..9d94c086 100644 --- a/filmulator-gui/core/nlmeans/eigen/test/sparse.h +++ b/filmulator-gui/core/nlmeans/eigen/test/sparse.h @@ -14,7 +14,7 @@ #include "main.h" -#if EIGEN_GNUC_AT_LEAST(4,0) && !defined __ICC && !defined(__clang__) +#if EIGEN_GNUC_AT_LEAST(4, 0) && !defined __ICC && !defined(__clang__) #ifdef min #undef min @@ -27,24 +27,19 @@ #include #define EIGEN_UNORDERED_MAP_SUPPORT namespace std { - using std::tr1::unordered_map; +using std::tr1::unordered_map; } #endif #ifdef EIGEN_GOOGLEHASH_SUPPORT - #include +#include #endif #include #include #include -enum { - ForceNonZeroDiag = 1, - MakeLowerTriangular = 2, - MakeUpperTriangular = 4, - ForceRealDiag = 8 -}; +enum { ForceNonZeroDiag = 1, MakeLowerTriangular = 2, MakeUpperTriangular = 4, ForceRealDiag = 8 }; /* Initializes both a sparse and dense matrix with same random values, * and a ratio of \a density non zero entries. @@ -53,158 +48,135 @@ enum { * \param zeroCoords and nonzeroCoords allows to get the coordinate lists of the non zero, * and zero coefficients respectively. */ -template void -initSparse(double density, - Matrix& refMat, - SparseMatrix& sparseMat, - int flags = 0, - std::vector >* zeroCoords = 0, - std::vector >* nonzeroCoords = 0) +template +void initSparse(double density, + Matrix &refMat, + SparseMatrix &sparseMat, + int flags = 0, + std::vector> *zeroCoords = 0, + std::vector> *nonzeroCoords = 0) { - enum { IsRowMajor = SparseMatrix::IsRowMajor }; + enum { IsRowMajor = SparseMatrix::IsRowMajor }; sparseMat.setZero(); - //sparseMat.reserve(int(refMat.rows()*refMat.cols()*density)); - sparseMat.reserve(VectorXi::Constant(IsRowMajor ? refMat.rows() : refMat.cols(), int((1.5*density)*(IsRowMajor?refMat.cols():refMat.rows())))); - - for(Index j=0; j(0,1) < density) ? internal::random() : Scalar(0); - if ((flags&ForceNonZeroDiag) && (i==j)) - { + if (IsRowMajor) std::swap(ai, aj); + Scalar v = (internal::random(0, 1) < density) ? internal::random() : Scalar(0); + if ((flags & ForceNonZeroDiag) && (i == j)) { // FIXME: the following is too conservative - v = internal::random()*Scalar(3.); - v = v*v; - if(numext::real(v)>0) v += Scalar(5); - else v -= Scalar(5); + v = internal::random() * Scalar(3.); + v = v * v; + if (numext::real(v) > 0) + v += Scalar(5); + else + v -= Scalar(5); } - if ((flags & MakeLowerTriangular) && aj>ai) + if ((flags & MakeLowerTriangular) && aj > ai) v = Scalar(0); - else if ((flags & MakeUpperTriangular) && ajpush_back(Matrix (ai,aj)); - } - else if (zeroCoords) - { - zeroCoords->push_back(Matrix (ai,aj)); + if (v != Scalar(0)) { + // sparseMat.insertBackByOuterInner(j,i) = v; + sparseMat.insertByOuterInner(j, i) = v; + if (nonzeroCoords) nonzeroCoords->push_back(Matrix(ai, aj)); + } else if (zeroCoords) { + zeroCoords->push_back(Matrix(ai, aj)); } - refMat(ai,aj) = v; + refMat(ai, aj) = v; } } - //sparseMat.finalize(); + // sparseMat.finalize(); } -template void -initSparse(double density, - Matrix& refMat, - DynamicSparseMatrix& sparseMat, - int flags = 0, - std::vector >* zeroCoords = 0, - std::vector >* nonzeroCoords = 0) +template +void initSparse(double density, + Matrix &refMat, + DynamicSparseMatrix &sparseMat, + int flags = 0, + std::vector> *zeroCoords = 0, + std::vector> *nonzeroCoords = 0) { - enum { IsRowMajor = DynamicSparseMatrix::IsRowMajor }; + enum { IsRowMajor = DynamicSparseMatrix::IsRowMajor }; sparseMat.setZero(); - sparseMat.reserve(int(refMat.rows()*refMat.cols()*density)); - for(int j=0; j(0,1) < density) ? internal::random() : Scalar(0); - if ((flags&ForceNonZeroDiag) && (i==j)) - { - v = internal::random()*Scalar(3.); - v = v*v + Scalar(5.); + if (IsRowMajor) std::swap(ai, aj); + Scalar v = (internal::random(0, 1) < density) ? internal::random() : Scalar(0); + if ((flags & ForceNonZeroDiag) && (i == j)) { + v = internal::random() * Scalar(3.); + v = v * v + Scalar(5.); } - if ((flags & MakeLowerTriangular) && aj>ai) + if ((flags & MakeLowerTriangular) && aj > ai) v = Scalar(0); - else if ((flags & MakeUpperTriangular) && ajpush_back(Matrix (ai,aj)); - } - else if (zeroCoords) - { - zeroCoords->push_back(Matrix (ai,aj)); + if (v != Scalar(0)) { + sparseMat.insertBackByOuterInner(j, i) = v; + if (nonzeroCoords) nonzeroCoords->push_back(Matrix(ai, aj)); + } else if (zeroCoords) { + zeroCoords->push_back(Matrix(ai, aj)); } - refMat(ai,aj) = v; + refMat(ai, aj) = v; } } sparseMat.finalize(); } -template void -initSparse(double density, - Matrix& refVec, - SparseVector& sparseVec, - std::vector* zeroCoords = 0, - std::vector* nonzeroCoords = 0) +template +void initSparse(double density, + Matrix &refVec, + SparseVector &sparseVec, + std::vector *zeroCoords = 0, + std::vector *nonzeroCoords = 0) { - sparseVec.reserve(int(refVec.size()*density)); + sparseVec.reserve(int(refVec.size() * density)); sparseVec.setZero(); - for(int i=0; i(0,1) < density) ? internal::random() : Scalar(0); - if (v!=Scalar(0)) - { + for (int i = 0; i < refVec.size(); i++) { + Scalar v = (internal::random(0, 1) < density) ? internal::random() : Scalar(0); + if (v != Scalar(0)) { sparseVec.insertBack(i) = v; - if (nonzeroCoords) - nonzeroCoords->push_back(i); - } - else if (zeroCoords) - zeroCoords->push_back(i); + if (nonzeroCoords) nonzeroCoords->push_back(i); + } else if (zeroCoords) + zeroCoords->push_back(i); refVec[i] = v; } } -template void -initSparse(double density, - Matrix& refVec, - SparseVector& sparseVec, - std::vector* zeroCoords = 0, - std::vector* nonzeroCoords = 0) +template +void initSparse(double density, + Matrix &refVec, + SparseVector &sparseVec, + std::vector *zeroCoords = 0, + std::vector *nonzeroCoords = 0) { - sparseVec.reserve(int(refVec.size()*density)); + sparseVec.reserve(int(refVec.size() * density)); sparseVec.setZero(); - for(int i=0; i(0,1) < density) ? internal::random() : Scalar(0); - if (v!=Scalar(0)) - { + for (int i = 0; i < refVec.size(); i++) { + Scalar v = (internal::random(0, 1) < density) ? internal::random() : Scalar(0); + if (v != Scalar(0)) { sparseVec.insertBack(i) = v; - if (nonzeroCoords) - nonzeroCoords->push_back(i); - } - else if (zeroCoords) - zeroCoords->push_back(i); + if (nonzeroCoords) nonzeroCoords->push_back(i); + } else if (zeroCoords) + zeroCoords->push_back(i); refVec[i] = v; } } #include -#endif // EIGEN_TESTSPARSE_H +#endif// EIGEN_TESTSPARSE_H diff --git a/filmulator-gui/core/nlmeans/eigen/test/sparseLM.cpp b/filmulator-gui/core/nlmeans/eigen/test/sparseLM.cpp index 8e148f9b..7cdb1ebe 100644 --- a/filmulator-gui/core/nlmeans/eigen/test/sparseLM.cpp +++ b/filmulator-gui/core/nlmeans/eigen/test/sparseLM.cpp @@ -7,9 +7,9 @@ // This Source Code Form is subject to the terms of the Mozilla // Public License v. 2.0. If a copy of the MPL was not distributed // with this file, You can obtain one at http://mozilla.org/MPL/2.0/. -#include #include #include +#include #include "main.h" #include @@ -17,160 +17,139 @@ using namespace std; using namespace Eigen; -template -struct sparseGaussianTest : SparseFunctor +template struct sparseGaussianTest : SparseFunctor { - typedef Matrix VectorType; - typedef SparseFunctor Base; + typedef Matrix VectorType; + typedef SparseFunctor Base; typedef typename Base::JacobianType JacobianType; - sparseGaussianTest(int inputs, int values) : SparseFunctor(inputs,values) - { } - - VectorType model(const VectorType& uv, VectorType& x) + sparseGaussianTest(int inputs, int values) : SparseFunctor(inputs, values) {} + + VectorType model(const VectorType &uv, VectorType &x) { - VectorType y; //Change this to use expression template - int m = Base::values(); + VectorType y;// Change this to use expression template + int m = Base::values(); int n = Base::inputs(); - eigen_assert(uv.size()%2 == 0); + eigen_assert(uv.size() % 2 == 0); eigen_assert(uv.size() == n); eigen_assert(x.size() == m); y.setZero(m); - int half = n/2; + int half = n / 2; VectorBlock u(uv, 0, half); VectorBlock v(uv, half, half); Scalar coeff; - for (int j = 0; j < m; j++) - { - for (int i = 0; i < half; i++) - { - coeff = (x(j)-i)/v(i); + for (int j = 0; j < m; j++) { + for (int i = 0; i < half; i++) { + coeff = (x(j) - i) / v(i); coeff *= coeff; - if (coeff < 1. && coeff > 0.) - y(j) += u(i)*std::pow((1-coeff), 2); + if (coeff < 1. && coeff > 0.) y(j) += u(i) * std::pow((1 - coeff), 2); } } return y; } - void initPoints(VectorType& uv_ref, VectorType& x) + void initPoints(VectorType &uv_ref, VectorType &x) { m_x = x; - m_y = this->model(uv_ref,x); + m_y = this->model(uv_ref, x); } - int operator()(const VectorType& uv, VectorType& fvec) + int operator()(const VectorType &uv, VectorType &fvec) { - int m = Base::values(); + int m = Base::values(); int n = Base::inputs(); - eigen_assert(uv.size()%2 == 0); + eigen_assert(uv.size() % 2 == 0); eigen_assert(uv.size() == n); - int half = n/2; + int half = n / 2; VectorBlock u(uv, 0, half); VectorBlock v(uv, half, half); fvec = m_y; Scalar coeff; - for (int j = 0; j < m; j++) - { - for (int i = 0; i < half; i++) - { - coeff = (m_x(j)-i)/v(i); + for (int j = 0; j < m; j++) { + for (int i = 0; i < half; i++) { + coeff = (m_x(j) - i) / v(i); coeff *= coeff; - if (coeff < 1. && coeff > 0.) - fvec(j) -= u(i)*std::pow((1-coeff), 2); + if (coeff < 1. && coeff > 0.) fvec(j) -= u(i) * std::pow((1 - coeff), 2); } } return 0; } - - int df(const VectorType& uv, JacobianType& fjac) + + int df(const VectorType &uv, JacobianType &fjac) { - int m = Base::values(); + int m = Base::values(); int n = Base::inputs(); eigen_assert(n == uv.size()); eigen_assert(fjac.rows() == m); eigen_assert(fjac.cols() == n); - int half = n/2; + int half = n / 2; VectorBlock u(uv, 0, half); VectorBlock v(uv, half, half); Scalar coeff; - - //Derivatives with respect to u - for (int col = 0; col < half; col++) - { - for (int row = 0; row < m; row++) - { - coeff = (m_x(row)-col)/v(col); - coeff = coeff*coeff; - if(coeff < 1. && coeff > 0.) - { - fjac.coeffRef(row,col) = -(1-coeff)*(1-coeff); - } + + // Derivatives with respect to u + for (int col = 0; col < half; col++) { + for (int row = 0; row < m; row++) { + coeff = (m_x(row) - col) / v(col); + coeff = coeff * coeff; + if (coeff < 1. && coeff > 0.) { fjac.coeffRef(row, col) = -(1 - coeff) * (1 - coeff); } } } - //Derivatives with respect to v - for (int col = 0; col < half; col++) - { - for (int row = 0; row < m; row++) - { - coeff = (m_x(row)-col)/v(col); - coeff = coeff*coeff; - if(coeff < 1. && coeff > 0.) - { - fjac.coeffRef(row,col+half) = -4 * (u(col)/v(col))*coeff*(1-coeff); - } + // Derivatives with respect to v + for (int col = 0; col < half; col++) { + for (int row = 0; row < m; row++) { + coeff = (m_x(row) - col) / v(col); + coeff = coeff * coeff; + if (coeff < 1. && coeff > 0.) { fjac.coeffRef(row, col + half) = -4 * (u(col) / v(col)) * coeff * (1 - coeff); } } } return 0; } - - VectorType m_x, m_y; //Data points + + VectorType m_x, m_y;// Data points }; -template -void test_sparseLM_T() +template void test_sparseLM_T() { - typedef Matrix VectorType; - + typedef Matrix VectorType; + int inputs = 10; int values = 2000; sparseGaussianTest sparse_gaussian(inputs, values); - VectorType uv(inputs),uv_ref(inputs); + VectorType uv(inputs), uv_ref(inputs); VectorType x(values); - // Generate the reference solution - uv_ref << -2, 1, 4 ,8, 6, 1.8, 1.2, 1.1, 1.9 , 3; - //Generate the reference data points + // Generate the reference solution + uv_ref << -2, 1, 4, 8, 6, 1.8, 1.2, 1.1, 1.9, 3; + // Generate the reference data points x.setRandom(); - x = 10*x; + x = 10 * x; x.array() += 10; sparse_gaussian.initPoints(uv_ref, x); - - - // Generate the initial parameters - VectorBlock u(uv, 0, inputs/2); - VectorBlock v(uv, inputs/2, inputs/2); + + + // Generate the initial parameters + VectorBlock u(uv, 0, inputs / 2); + VectorBlock v(uv, inputs / 2, inputs / 2); v.setOnes(); - //Generate u or Solve for u from v + // Generate u or Solve for u from v u.setOnes(); - + // Solve the optimization problem - LevenbergMarquardt > lm(sparse_gaussian); + LevenbergMarquardt> lm(sparse_gaussian); int info; -// info = lm.minimize(uv); - - VERIFY_IS_EQUAL(info,1); - // Do a step by step solution and save the residual + // info = lm.minimize(uv); + + VERIFY_IS_EQUAL(info, 1); + // Do a step by step solution and save the residual int maxiter = 200; int iter = 0; MatrixXd Err(values, maxiter); MatrixXd Mod(values, maxiter); - LevenbergMarquardtSpace::Status status; + LevenbergMarquardtSpace::Status status; status = lm.minimizeInit(uv); - if (status==LevenbergMarquardtSpace::ImproperInputParameters) - return ; - + if (status == LevenbergMarquardtSpace::ImproperInputParameters) return; } void test_sparseLM() { CALL_SUBTEST_1(test_sparseLM_T()); - + // CALL_SUBTEST_2(test_sparseLM_T()); } diff --git a/filmulator-gui/core/nlmeans/eigen/test/sparse_basic.cpp b/filmulator-gui/core/nlmeans/eigen/test/sparse_basic.cpp index d0ef722b..532e5f7b 100644 --- a/filmulator-gui/core/nlmeans/eigen/test/sparse_basic.cpp +++ b/filmulator-gui/core/nlmeans/eigen/test/sparse_basic.cpp @@ -14,23 +14,23 @@ static long g_realloc_count = 0; #include "sparse.h" -template void sparse_basic(const SparseMatrixType& ref) +template void sparse_basic(const SparseMatrixType &ref) { typedef typename SparseMatrixType::StorageIndex StorageIndex; - typedef Matrix Vector2; - + typedef Matrix Vector2; + const Index rows = ref.rows(); const Index cols = ref.cols(); - //const Index inner = ref.innerSize(); - //const Index outer = ref.outerSize(); + // const Index inner = ref.innerSize(); + // const Index outer = ref.outerSize(); typedef typename SparseMatrixType::Scalar Scalar; typedef typename SparseMatrixType::RealScalar RealScalar; enum { Flags = SparseMatrixType::Flags }; - double density = (std::max)(8./(rows*cols), 0.01); - typedef Matrix DenseMatrix; - typedef Matrix DenseVector; + double density = (std::max)(8. / (rows * cols), 0.01); + typedef Matrix DenseMatrix; + typedef Matrix DenseVector; Scalar eps = 1e-6; Scalar s1 = internal::random(); @@ -44,104 +44,89 @@ template void sparse_basic(const SparseMatrixType& re initSparse(density, refMat, m, 0, &zeroCoords, &nonzeroCoords); // test coeff and coeffRef - for (std::size_t i=0; i >::value) - VERIFY_RAISES_ASSERT( m.coeffRef(zeroCoords[i].x(),zeroCoords[i].y()) = 5 ); + for (std::size_t i = 0; i < zeroCoords.size(); ++i) { + VERIFY_IS_MUCH_SMALLER_THAN(m.coeff(zeroCoords[i].x(), zeroCoords[i].y()), eps); + if (internal::is_same>::value) + VERIFY_RAISES_ASSERT(m.coeffRef(zeroCoords[i].x(), zeroCoords[i].y()) = 5); } VERIFY_IS_APPROX(m, refMat); - if(!nonzeroCoords.empty()) { + if (!nonzeroCoords.empty()) { m.coeffRef(nonzeroCoords[0].x(), nonzeroCoords[0].y()) = Scalar(5); refMat.coeffRef(nonzeroCoords[0].x(), nonzeroCoords[0].y()) = Scalar(5); } VERIFY_IS_APPROX(m, refMat); - // test assertion - VERIFY_RAISES_ASSERT( m.coeffRef(-1,1) = 0 ); - VERIFY_RAISES_ASSERT( m.coeffRef(0,m.cols()) = 0 ); - } + // test assertion + VERIFY_RAISES_ASSERT(m.coeffRef(-1, 1) = 0); + VERIFY_RAISES_ASSERT(m.coeffRef(0, m.cols()) = 0); + } - // test insert (inner random) - { - DenseMatrix m1(rows,cols); - m1.setZero(); - SparseMatrixType m2(rows,cols); - bool call_reserve = internal::random()%2; - Index nnz = internal::random(1,int(rows)/2); - if(call_reserve) - { - if(internal::random()%2) - m2.reserve(VectorXi::Constant(m2.outerSize(), int(nnz))); - else - m2.reserve(m2.outerSize() * nnz); - } - g_realloc_count = 0; - for (Index j=0; j(0,rows-1); - if (m1.coeff(i,j)==Scalar(0)) - m2.insert(i,j) = m1(i,j) = internal::random(); - } - } - - if(call_reserve && !SparseMatrixType::IsRowMajor) - { - VERIFY(g_realloc_count==0); + // test insert (inner random) + { + DenseMatrix m1(rows, cols); + m1.setZero(); + SparseMatrixType m2(rows, cols); + bool call_reserve = internal::random() % 2; + Index nnz = internal::random(1, int(rows) / 2); + if (call_reserve) { + if (internal::random() % 2) + m2.reserve(VectorXi::Constant(m2.outerSize(), int(nnz))); + else + m2.reserve(m2.outerSize() * nnz); + } + g_realloc_count = 0; + for (Index j = 0; j < cols; ++j) { + for (Index k = 0; k < nnz; ++k) { + Index i = internal::random(0, rows - 1); + if (m1.coeff(i, j) == Scalar(0)) m2.insert(i, j) = m1(i, j) = internal::random(); } - - m2.finalize(); - VERIFY_IS_APPROX(m2,m1); } - // test insert (fully random) - { - DenseMatrix m1(rows,cols); - m1.setZero(); - SparseMatrixType m2(rows,cols); - if(internal::random()%2) - m2.reserve(VectorXi::Constant(m2.outerSize(), 2)); - for (int k=0; k(0,rows-1); - Index j = internal::random(0,cols-1); - if ((m1.coeff(i,j)==Scalar(0)) && (internal::random()%2)) - m2.insert(i,j) = m1(i,j) = internal::random(); - else - { - Scalar v = internal::random(); - m2.coeffRef(i,j) += v; - m1(i,j) += v; - } + if (call_reserve && !SparseMatrixType::IsRowMajor) { VERIFY(g_realloc_count == 0); } + + m2.finalize(); + VERIFY_IS_APPROX(m2, m1); + } + + // test insert (fully random) + { + DenseMatrix m1(rows, cols); + m1.setZero(); + SparseMatrixType m2(rows, cols); + if (internal::random() % 2) m2.reserve(VectorXi::Constant(m2.outerSize(), 2)); + for (int k = 0; k < rows * cols; ++k) { + Index i = internal::random(0, rows - 1); + Index j = internal::random(0, cols - 1); + if ((m1.coeff(i, j) == Scalar(0)) && (internal::random() % 2)) + m2.insert(i, j) = m1(i, j) = internal::random(); + else { + Scalar v = internal::random(); + m2.coeffRef(i, j) += v; + m1(i, j) += v; } - VERIFY_IS_APPROX(m2,m1); } - - // test insert (un-compressed) - for(int mode=0;mode<4;++mode) - { - DenseMatrix m1(rows,cols); - m1.setZero(); - SparseMatrixType m2(rows,cols); - VectorXi r(VectorXi::Constant(m2.outerSize(), ((mode%2)==0) ? int(m2.innerSize()) : std::max(1,int(m2.innerSize())/8))); - m2.reserve(r); - for (Index k=0; k(0,rows-1); - Index j = internal::random(0,cols-1); - if (m1.coeff(i,j)==Scalar(0)) - m2.insert(i,j) = m1(i,j) = internal::random(); - if(mode==3) - m2.reserve(r); - } - if(internal::random()%2) - m2.makeCompressed(); - VERIFY_IS_APPROX(m2,m1); + VERIFY_IS_APPROX(m2, m1); + } + + // test insert (un-compressed) + for (int mode = 0; mode < 4; ++mode) { + DenseMatrix m1(rows, cols); + m1.setZero(); + SparseMatrixType m2(rows, cols); + VectorXi r(VectorXi::Constant( + m2.outerSize(), ((mode % 2) == 0) ? int(m2.innerSize()) : std::max(1, int(m2.innerSize()) / 8))); + m2.reserve(r); + for (Index k = 0; k < rows * cols; ++k) { + Index i = internal::random(0, rows - 1); + Index j = internal::random(0, cols - 1); + if (m1.coeff(i, j) == Scalar(0)) m2.insert(i, j) = m1(i, j) = internal::random(); + if (mode == 3) m2.reserve(r); } + if (internal::random() % 2) m2.makeCompressed(); + VERIFY_IS_APPROX(m2, m1); + } // test basic computations { @@ -158,29 +143,28 @@ template void sparse_basic(const SparseMatrixType& re initSparse(density, refM3, m3); initSparse(density, refM4, m4); - if(internal::random()) - m1.makeCompressed(); + if (internal::random()) m1.makeCompressed(); Index m1_nnz = m1.nonZeros(); - VERIFY_IS_APPROX(m1*s1, refM1*s1); - VERIFY_IS_APPROX(m1+m2, refM1+refM2); - VERIFY_IS_APPROX(m1+m2+m3, refM1+refM2+refM3); - VERIFY_IS_APPROX(m3.cwiseProduct(m1+m2), refM3.cwiseProduct(refM1+refM2)); - VERIFY_IS_APPROX(m1*s1-m2, refM1*s1-refM2); - VERIFY_IS_APPROX(m4=m1/s1, refM1/s1); + VERIFY_IS_APPROX(m1 * s1, refM1 * s1); + VERIFY_IS_APPROX(m1 + m2, refM1 + refM2); + VERIFY_IS_APPROX(m1 + m2 + m3, refM1 + refM2 + refM3); + VERIFY_IS_APPROX(m3.cwiseProduct(m1 + m2), refM3.cwiseProduct(refM1 + refM2)); + VERIFY_IS_APPROX(m1 * s1 - m2, refM1 * s1 - refM2); + VERIFY_IS_APPROX(m4 = m1 / s1, refM1 / s1); VERIFY_IS_EQUAL(m4.nonZeros(), m1_nnz); - if(SparseMatrixType::IsRowMajor) + if (SparseMatrixType::IsRowMajor) VERIFY_IS_APPROX(m1.innerVector(0).dot(refM2.row(0)), refM1.row(0).dot(refM2.row(0))); else VERIFY_IS_APPROX(m1.innerVector(0).dot(refM2.col(0)), refM1.col(0).dot(refM2.col(0))); DenseVector rv = DenseVector::Random(m1.cols()); DenseVector cv = DenseVector::Random(m1.rows()); - Index r = internal::random(0,m1.rows()-2); - Index c = internal::random(0,m1.cols()-1); - VERIFY_IS_APPROX(( m1.template block<1,Dynamic>(r,0,1,m1.cols()).dot(rv)) , refM1.row(r).dot(rv)); + Index r = internal::random(0, m1.rows() - 2); + Index c = internal::random(0, m1.cols() - 1); + VERIFY_IS_APPROX((m1.template block<1, Dynamic>(r, 0, 1, m1.cols()).dot(rv)), refM1.row(r).dot(rv)); VERIFY_IS_APPROX(m1.row(r).dot(rv), refM1.row(r).dot(rv)); VERIFY_IS_APPROX(m1.col(c).dot(cv), refM1.col(c).dot(cv)); @@ -192,66 +176,74 @@ template void sparse_basic(const SparseMatrixType& re VERIFY_IS_APPROX(m3.cwiseProduct(refM4), refM3.cwiseProduct(refM4)); // dense cwise* sparse VERIFY_IS_APPROX(refM4.cwiseProduct(m3), refM4.cwiseProduct(refM3)); -// VERIFY_IS_APPROX(m3.cwise()/refM4, refM3.cwise()/refM4); + // VERIFY_IS_APPROX(m3.cwise()/refM4, refM3.cwise()/refM4); VERIFY_IS_APPROX(refM4 + m3, refM4 + refM3); VERIFY_IS_APPROX(m3 + refM4, refM3 + refM4); VERIFY_IS_APPROX(refM4 - m3, refM4 - refM3); VERIFY_IS_APPROX(m3 - refM4, refM3 - refM4); - VERIFY_IS_APPROX((RealScalar(0.5)*refM4 + RealScalar(0.5)*m3).eval(), RealScalar(0.5)*refM4 + RealScalar(0.5)*refM3); - VERIFY_IS_APPROX((RealScalar(0.5)*refM4 + m3*RealScalar(0.5)).eval(), RealScalar(0.5)*refM4 + RealScalar(0.5)*refM3); - VERIFY_IS_APPROX((RealScalar(0.5)*refM4 + m3.cwiseProduct(m3)).eval(), RealScalar(0.5)*refM4 + refM3.cwiseProduct(refM3)); - - VERIFY_IS_APPROX((RealScalar(0.5)*refM4 + RealScalar(0.5)*m3).eval(), RealScalar(0.5)*refM4 + RealScalar(0.5)*refM3); - VERIFY_IS_APPROX((RealScalar(0.5)*refM4 + m3*RealScalar(0.5)).eval(), RealScalar(0.5)*refM4 + RealScalar(0.5)*refM3); - VERIFY_IS_APPROX((RealScalar(0.5)*refM4 + (m3+m3)).eval(), RealScalar(0.5)*refM4 + (refM3+refM3)); - VERIFY_IS_APPROX(((refM3+m3)+RealScalar(0.5)*m3).eval(), RealScalar(0.5)*refM3 + (refM3+refM3)); - VERIFY_IS_APPROX((RealScalar(0.5)*refM4 + (refM3+m3)).eval(), RealScalar(0.5)*refM4 + (refM3+refM3)); - VERIFY_IS_APPROX((RealScalar(0.5)*refM4 + (m3+refM3)).eval(), RealScalar(0.5)*refM4 + (refM3+refM3)); + VERIFY_IS_APPROX( + (RealScalar(0.5) * refM4 + RealScalar(0.5) * m3).eval(), RealScalar(0.5) * refM4 + RealScalar(0.5) * refM3); + VERIFY_IS_APPROX( + (RealScalar(0.5) * refM4 + m3 * RealScalar(0.5)).eval(), RealScalar(0.5) * refM4 + RealScalar(0.5) * refM3); + VERIFY_IS_APPROX( + (RealScalar(0.5) * refM4 + m3.cwiseProduct(m3)).eval(), RealScalar(0.5) * refM4 + refM3.cwiseProduct(refM3)); + + VERIFY_IS_APPROX( + (RealScalar(0.5) * refM4 + RealScalar(0.5) * m3).eval(), RealScalar(0.5) * refM4 + RealScalar(0.5) * refM3); + VERIFY_IS_APPROX( + (RealScalar(0.5) * refM4 + m3 * RealScalar(0.5)).eval(), RealScalar(0.5) * refM4 + RealScalar(0.5) * refM3); + VERIFY_IS_APPROX((RealScalar(0.5) * refM4 + (m3 + m3)).eval(), RealScalar(0.5) * refM4 + (refM3 + refM3)); + VERIFY_IS_APPROX(((refM3 + m3) + RealScalar(0.5) * m3).eval(), RealScalar(0.5) * refM3 + (refM3 + refM3)); + VERIFY_IS_APPROX((RealScalar(0.5) * refM4 + (refM3 + m3)).eval(), RealScalar(0.5) * refM4 + (refM3 + refM3)); + VERIFY_IS_APPROX((RealScalar(0.5) * refM4 + (m3 + refM3)).eval(), RealScalar(0.5) * refM4 + (refM3 + refM3)); VERIFY_IS_APPROX(m1.sum(), refM1.sum()); - m4 = m1; refM4 = m4; + m4 = m1; + refM4 = m4; - VERIFY_IS_APPROX(m1*=s1, refM1*=s1); + VERIFY_IS_APPROX(m1 *= s1, refM1 *= s1); VERIFY_IS_EQUAL(m1.nonZeros(), m1_nnz); - VERIFY_IS_APPROX(m1/=s1, refM1/=s1); + VERIFY_IS_APPROX(m1 /= s1, refM1 /= s1); VERIFY_IS_EQUAL(m1.nonZeros(), m1_nnz); - VERIFY_IS_APPROX(m1+=m2, refM1+=refM2); - VERIFY_IS_APPROX(m1-=m2, refM1-=refM2); + VERIFY_IS_APPROX(m1 += m2, refM1 += refM2); + VERIFY_IS_APPROX(m1 -= m2, refM1 -= refM2); - if (rows>=2 && cols>=2) - { - VERIFY_RAISES_ASSERT( m1 += m1.innerVector(0) ); - VERIFY_RAISES_ASSERT( m1 -= m1.innerVector(0) ); - VERIFY_RAISES_ASSERT( refM1 -= m1.innerVector(0) ); - VERIFY_RAISES_ASSERT( refM1 += m1.innerVector(0) ); + if (rows >= 2 && cols >= 2) { + VERIFY_RAISES_ASSERT(m1 += m1.innerVector(0)); + VERIFY_RAISES_ASSERT(m1 -= m1.innerVector(0)); + VERIFY_RAISES_ASSERT(refM1 -= m1.innerVector(0)); + VERIFY_RAISES_ASSERT(refM1 += m1.innerVector(0)); } - m1 = m4; refM1 = refM4; + m1 = m4; + refM1 = refM4; // test aliasing VERIFY_IS_APPROX((m1 = -m1), (refM1 = -refM1)); VERIFY_IS_EQUAL(m1.nonZeros(), m1_nnz); - m1 = m4; refM1 = refM4; + m1 = m4; + refM1 = refM4; VERIFY_IS_APPROX((m1 = m1.transpose()), (refM1 = refM1.transpose().eval())); VERIFY_IS_EQUAL(m1.nonZeros(), m1_nnz); - m1 = m4; refM1 = refM4; + m1 = m4; + refM1 = refM4; VERIFY_IS_APPROX((m1 = -m1.transpose()), (refM1 = -refM1.transpose().eval())); VERIFY_IS_EQUAL(m1.nonZeros(), m1_nnz); - m1 = m4; refM1 = refM4; + m1 = m4; + refM1 = refM4; VERIFY_IS_APPROX((m1 += -m1), (refM1 += -refM1)); VERIFY_IS_EQUAL(m1.nonZeros(), m1_nnz); - m1 = m4; refM1 = refM4; + m1 = m4; + refM1 = refM4; - if(m1.isCompressed()) - { + if (m1.isCompressed()) { VERIFY_IS_APPROX(m1.coeffs().sum(), m1.sum()); m1.coeffs() += s1; - for(Index j = 0; j void sparse_basic(const SparseMatrixType& re SpBool mb1 = m1.real().template cast(); SpBool mb2 = m2.real().template cast(); VERIFY_IS_EQUAL(mb1.template cast().sum(), refM1.real().template cast().count()); - VERIFY_IS_EQUAL((mb1 && mb2).template cast().sum(), (refM1.real().template cast() && refM2.real().template cast()).count()); - VERIFY_IS_EQUAL((mb1 || mb2).template cast().sum(), (refM1.real().template cast() || refM2.real().template cast()).count()); + VERIFY_IS_EQUAL((mb1 && mb2).template cast().sum(), + (refM1.real().template cast() && refM2.real().template cast()).count()); + VERIFY_IS_EQUAL((mb1 || mb2).template cast().sum(), + (refM1.real().template cast() || refM2.real().template cast()).count()); SpBool mb3 = mb1 && mb2; - if(mb1.coeffs().all() && mb2.coeffs().all()) - { - VERIFY_IS_EQUAL(mb3.nonZeros(), (refM1.real().template cast() && refM2.real().template cast()).count()); + if (mb1.coeffs().all() && mb2.coeffs().all()) { + VERIFY_IS_EQUAL( + mb3.nonZeros(), (refM1.real().template cast() && refM2.real().template cast()).count()); } } } @@ -278,23 +272,20 @@ template void sparse_basic(const SparseMatrixType& re initSparse(density, refMat2, m2); std::vector ref_value(m2.innerSize()); std::vector ref_index(m2.innerSize()); - if(internal::random()) - m2.makeCompressed(); - for(Index j = 0; j()) m2.makeCompressed(); + for (Index j = 0; j < m2.outerSize(); ++j) { Index count_forward = 0; - for(typename SparseMatrixType::InnerIterator it(m2,j); it; ++it) - { - ref_value[ref_value.size()-1-count_forward] = it.value(); - ref_index[ref_index.size()-1-count_forward] = it.index(); + for (typename SparseMatrixType::InnerIterator it(m2, j); it; ++it) { + ref_value[ref_value.size() - 1 - count_forward] = it.value(); + ref_index[ref_index.size() - 1 - count_forward] = it.index(); count_forward++; } Index count_reverse = 0; - for(typename SparseMatrixType::ReverseInnerIterator it(m2,j); it; --it) - { - VERIFY_IS_APPROX( std::abs(ref_value[ref_value.size()-count_forward+count_reverse])+1, std::abs(it.value())+1); - VERIFY_IS_EQUAL( ref_index[ref_index.size()-count_forward+count_reverse] , it.index()); + for (typename SparseMatrixType::ReverseInnerIterator it(m2, j); it; --it) { + VERIFY_IS_APPROX( + std::abs(ref_value[ref_value.size() - count_forward + count_reverse]) + 1, std::abs(it.value()) + 1); + VERIFY_IS_EQUAL(ref_index[ref_index.size() - count_forward + count_reverse], it.index()); count_reverse++; } VERIFY_IS_EQUAL(count_forward, count_reverse); @@ -310,7 +301,7 @@ template void sparse_basic(const SparseMatrixType& re VERIFY_IS_APPROX(m2.transpose(), refMat2.transpose()); VERIFY_IS_APPROX(SparseMatrixType(m2.adjoint()), refMat2.adjoint()); - + // check isApprox handles opposite storage order typename Transpose::PlainObject m3(m2); VERIFY(m2.isApprox(m3)); @@ -324,73 +315,63 @@ template void sparse_basic(const SparseMatrixType& re int countFalseNonZero = 0; int countTrueNonZero = 0; m2.reserve(VectorXi::Constant(m2.outerSize(), int(m2.innerSize()))); - for (Index j=0; j(0,1); - if (x<0.1f) - { + for (Index j = 0; j < m2.cols(); ++j) { + for (Index i = 0; i < m2.rows(); ++i) { + float x = internal::random(0, 1); + if (x < 0.1f) { // do nothing - } - else if (x<0.5f) - { + } else if (x < 0.5f) { countFalseNonZero++; - m2.insert(i,j) = Scalar(0); - } - else - { + m2.insert(i, j) = Scalar(0); + } else { countTrueNonZero++; - m2.insert(i,j) = Scalar(1); - refM2(i,j) = Scalar(1); + m2.insert(i, j) = Scalar(1); + refM2(i, j) = Scalar(1); } } } - if(internal::random()) - m2.makeCompressed(); - VERIFY(countFalseNonZero+countTrueNonZero == m2.nonZeros()); - if(countTrueNonZero>0) - VERIFY_IS_APPROX(m2, refM2); + if (internal::random()) m2.makeCompressed(); + VERIFY(countFalseNonZero + countTrueNonZero == m2.nonZeros()); + if (countTrueNonZero > 0) VERIFY_IS_APPROX(m2, refM2); m2.prune(Scalar(1)); - VERIFY(countTrueNonZero==m2.nonZeros()); + VERIFY(countTrueNonZero == m2.nonZeros()); VERIFY_IS_APPROX(m2, refM2); } // test setFromTriplets { - typedef Triplet TripletType; + typedef Triplet TripletType; std::vector triplets; - Index ntriplets = rows*cols; + Index ntriplets = rows * cols; triplets.reserve(ntriplets); - DenseMatrix refMat_sum = DenseMatrix::Zero(rows,cols); - DenseMatrix refMat_prod = DenseMatrix::Zero(rows,cols); - DenseMatrix refMat_last = DenseMatrix::Zero(rows,cols); + DenseMatrix refMat_sum = DenseMatrix::Zero(rows, cols); + DenseMatrix refMat_prod = DenseMatrix::Zero(rows, cols); + DenseMatrix refMat_last = DenseMatrix::Zero(rows, cols); - for(Index i=0;i(0,StorageIndex(rows-1)); - StorageIndex c = internal::random(0,StorageIndex(cols-1)); + for (Index i = 0; i < ntriplets; ++i) { + StorageIndex r = internal::random(0, StorageIndex(rows - 1)); + StorageIndex c = internal::random(0, StorageIndex(cols - 1)); Scalar v = internal::random(); - triplets.push_back(TripletType(r,c,v)); - refMat_sum(r,c) += v; - if(std::abs(refMat_prod(r,c))==0) - refMat_prod(r,c) = v; + triplets.push_back(TripletType(r, c, v)); + refMat_sum(r, c) += v; + if (std::abs(refMat_prod(r, c)) == 0) + refMat_prod(r, c) = v; else - refMat_prod(r,c) *= v; - refMat_last(r,c) = v; + refMat_prod(r, c) *= v; + refMat_last(r, c) = v; } - SparseMatrixType m(rows,cols); + SparseMatrixType m(rows, cols); m.setFromTriplets(triplets.begin(), triplets.end()); VERIFY_IS_APPROX(m, refMat_sum); m.setFromTriplets(triplets.begin(), triplets.end(), std::multiplies()); VERIFY_IS_APPROX(m, refMat_prod); #if (defined(__cplusplus) && __cplusplus >= 201103L) - m.setFromTriplets(triplets.begin(), triplets.end(), [] (Scalar,Scalar b) { return b; }); + m.setFromTriplets(triplets.begin(), triplets.end(), [](Scalar, Scalar b) { return b; }); VERIFY_IS_APPROX(m, refMat_last); #endif } - + // test Map { DenseMatrix refMat2(rows, cols), refMat3(rows, cols); @@ -398,28 +379,52 @@ template void sparse_basic(const SparseMatrixType& re initSparse(density, refMat2, m2); initSparse(density, refMat3, m3); { - Map mapMat2(m2.rows(), m2.cols(), m2.nonZeros(), m2.outerIndexPtr(), m2.innerIndexPtr(), m2.valuePtr(), m2.innerNonZeroPtr()); - Map mapMat3(m3.rows(), m3.cols(), m3.nonZeros(), m3.outerIndexPtr(), m3.innerIndexPtr(), m3.valuePtr(), m3.innerNonZeroPtr()); - VERIFY_IS_APPROX(mapMat2+mapMat3, refMat2+refMat3); - VERIFY_IS_APPROX(mapMat2+mapMat3, refMat2+refMat3); + Map mapMat2(m2.rows(), + m2.cols(), + m2.nonZeros(), + m2.outerIndexPtr(), + m2.innerIndexPtr(), + m2.valuePtr(), + m2.innerNonZeroPtr()); + Map mapMat3(m3.rows(), + m3.cols(), + m3.nonZeros(), + m3.outerIndexPtr(), + m3.innerIndexPtr(), + m3.valuePtr(), + m3.innerNonZeroPtr()); + VERIFY_IS_APPROX(mapMat2 + mapMat3, refMat2 + refMat3); + VERIFY_IS_APPROX(mapMat2 + mapMat3, refMat2 + refMat3); } { - MappedSparseMatrix mapMat2(m2.rows(), m2.cols(), m2.nonZeros(), m2.outerIndexPtr(), m2.innerIndexPtr(), m2.valuePtr(), m2.innerNonZeroPtr()); - MappedSparseMatrix mapMat3(m3.rows(), m3.cols(), m3.nonZeros(), m3.outerIndexPtr(), m3.innerIndexPtr(), m3.valuePtr(), m3.innerNonZeroPtr()); - VERIFY_IS_APPROX(mapMat2+mapMat3, refMat2+refMat3); - VERIFY_IS_APPROX(mapMat2+mapMat3, refMat2+refMat3); + MappedSparseMatrix mapMat2(m2.rows(), + m2.cols(), + m2.nonZeros(), + m2.outerIndexPtr(), + m2.innerIndexPtr(), + m2.valuePtr(), + m2.innerNonZeroPtr()); + MappedSparseMatrix mapMat3(m3.rows(), + m3.cols(), + m3.nonZeros(), + m3.outerIndexPtr(), + m3.innerIndexPtr(), + m3.valuePtr(), + m3.innerNonZeroPtr()); + VERIFY_IS_APPROX(mapMat2 + mapMat3, refMat2 + refMat3); + VERIFY_IS_APPROX(mapMat2 + mapMat3, refMat2 + refMat3); } - Index i = internal::random(0,rows-1); - Index j = internal::random(0,cols-1); - m2.coeffRef(i,j) = 123; - if(internal::random()) - m2.makeCompressed(); - Map mapMat2(rows, cols, m2.nonZeros(), m2.outerIndexPtr(), m2.innerIndexPtr(), m2.valuePtr(), m2.innerNonZeroPtr()); - VERIFY_IS_EQUAL(m2.coeff(i,j),Scalar(123)); - VERIFY_IS_EQUAL(mapMat2.coeff(i,j),Scalar(123)); - mapMat2.coeffRef(i,j) = -123; - VERIFY_IS_EQUAL(m2.coeff(i,j),Scalar(-123)); + Index i = internal::random(0, rows - 1); + Index j = internal::random(0, cols - 1); + m2.coeffRef(i, j) = 123; + if (internal::random()) m2.makeCompressed(); + Map mapMat2( + rows, cols, m2.nonZeros(), m2.outerIndexPtr(), m2.innerIndexPtr(), m2.valuePtr(), m2.innerNonZeroPtr()); + VERIFY_IS_EQUAL(m2.coeff(i, j), Scalar(123)); + VERIFY_IS_EQUAL(mapMat2.coeff(i, j), Scalar(123)); + mapMat2.coeffRef(i, j) = -123; + VERIFY_IS_EQUAL(m2.coeff(i, j), Scalar(-123)); } // test triangularView @@ -457,10 +462,9 @@ template void sparse_basic(const SparseMatrixType& re refMat3 = m2.template triangularView(); VERIFY_IS_APPROX(refMat3, DenseMatrix(refMat2.template triangularView())); } - + // test selfadjointView - if(!SparseMatrixType::IsRowMajor) - { + if (!SparseMatrixType::IsRowMajor) { DenseMatrix refMat2(rows, rows), refMat3(rows, rows); SparseMatrixType m2(rows, rows), m3(rows, rows); initSparse(density, refMat2, m2); @@ -477,11 +481,11 @@ template void sparse_basic(const SparseMatrixType& re VERIFY_IS_APPROX(m3, refMat3); // selfadjointView only works for square matrices: - SparseMatrixType m4(rows, rows+1); + SparseMatrixType m4(rows, rows + 1); VERIFY_RAISES_ASSERT(m4.template selfadjointView()); VERIFY_RAISES_ASSERT(m4.template selfadjointView()); } - + // test sparseView { DenseMatrix refMat2 = DenseMatrix::Zero(rows, rows); @@ -490,10 +494,10 @@ template void sparse_basic(const SparseMatrixType& re VERIFY_IS_APPROX(m2.eval(), refMat2.sparseView().eval()); // sparse view on expressions: - VERIFY_IS_APPROX((s1*m2).eval(), (s1*refMat2).sparseView().eval()); - VERIFY_IS_APPROX((m2+m2).eval(), (refMat2+refMat2).sparseView().eval()); - VERIFY_IS_APPROX((m2*m2).eval(), (refMat2.lazyProduct(refMat2)).sparseView().eval()); - VERIFY_IS_APPROX((m2*m2).eval(), (refMat2*refMat2).sparseView().eval()); + VERIFY_IS_APPROX((s1 * m2).eval(), (s1 * refMat2).sparseView().eval()); + VERIFY_IS_APPROX((m2 + m2).eval(), (refMat2 + refMat2).sparseView().eval()); + VERIFY_IS_APPROX((m2 * m2).eval(), (refMat2.lazyProduct(refMat2)).sparseView().eval()); + VERIFY_IS_APPROX((m2 * m2).eval(), (refMat2 * refMat2).sparseView().eval()); } // test diagonal @@ -506,14 +510,14 @@ template void sparse_basic(const SparseMatrixType& re VERIFY_IS_APPROX(d, refMat2.diagonal().eval()); d = m2.diagonal().array(); VERIFY_IS_APPROX(d, refMat2.diagonal().eval()); - VERIFY_IS_APPROX(const_cast(m2).diagonal(), refMat2.diagonal().eval()); - + VERIFY_IS_APPROX(const_cast(m2).diagonal(), refMat2.diagonal().eval()); + initSparse(density, refMat2, m2, ForceNonZeroDiag); - m2.diagonal() += refMat2.diagonal(); + m2.diagonal() += refMat2.diagonal(); refMat2.diagonal() += refMat2.diagonal(); VERIFY_IS_APPROX(m2, refMat2); } - + // test diagonal to sparse { DenseVector d = DenseVector::Random(rows); @@ -527,41 +531,36 @@ template void sparse_basic(const SparseMatrixType& re m2 += d.asDiagonal(); VERIFY_IS_APPROX(m2, refMat2); } - + // test conservative resize { - std::vector< std::pair > inc; - if(rows > 3 && cols > 2) - inc.push_back(std::pair(-3,-2)); - inc.push_back(std::pair(0,0)); - inc.push_back(std::pair(3,2)); - inc.push_back(std::pair(3,0)); - inc.push_back(std::pair(0,3)); - - for(size_t i = 0; i< inc.size(); i++) { - StorageIndex incRows = inc[i].first; - StorageIndex incCols = inc[i].second; - SparseMatrixType m1(rows, cols); - DenseMatrix refMat1 = DenseMatrix::Zero(rows, cols); - initSparse(density, refMat1, m1); - - m1.conservativeResize(rows+incRows, cols+incCols); - refMat1.conservativeResize(rows+incRows, cols+incCols); - if (incRows > 0) refMat1.bottomRows(incRows).setZero(); - if (incCols > 0) refMat1.rightCols(incCols).setZero(); - - VERIFY_IS_APPROX(m1, refMat1); - - // Insert new values - if (incRows > 0) - m1.insert(m1.rows()-1, 0) = refMat1(refMat1.rows()-1, 0) = 1; - if (incCols > 0) - m1.insert(0, m1.cols()-1) = refMat1(0, refMat1.cols()-1) = 1; - - VERIFY_IS_APPROX(m1, refMat1); - - - } + std::vector> inc; + if (rows > 3 && cols > 2) inc.push_back(std::pair(-3, -2)); + inc.push_back(std::pair(0, 0)); + inc.push_back(std::pair(3, 2)); + inc.push_back(std::pair(3, 0)); + inc.push_back(std::pair(0, 3)); + + for (size_t i = 0; i < inc.size(); i++) { + StorageIndex incRows = inc[i].first; + StorageIndex incCols = inc[i].second; + SparseMatrixType m1(rows, cols); + DenseMatrix refMat1 = DenseMatrix::Zero(rows, cols); + initSparse(density, refMat1, m1); + + m1.conservativeResize(rows + incRows, cols + incCols); + refMat1.conservativeResize(rows + incRows, cols + incCols); + if (incRows > 0) refMat1.bottomRows(incRows).setZero(); + if (incCols > 0) refMat1.rightCols(incCols).setZero(); + + VERIFY_IS_APPROX(m1, refMat1); + + // Insert new values + if (incRows > 0) m1.insert(m1.rows() - 1, 0) = refMat1(refMat1.rows() - 1, 0) = 1; + if (incCols > 0) m1.insert(0, m1.cols() - 1) = refMat1(0, refMat1.cols() - 1) = 1; + + VERIFY_IS_APPROX(m1, refMat1); + } } // test Identity matrix @@ -570,16 +569,14 @@ template void sparse_basic(const SparseMatrixType& re SparseMatrixType m1(rows, rows); m1.setIdentity(); VERIFY_IS_APPROX(m1, refMat1); - for(int k=0; k(0,rows-1); - Index j = internal::random(0,rows-1); + for (int k = 0; k < rows * rows / 4; ++k) { + Index i = internal::random(0, rows - 1); + Index j = internal::random(0, rows - 1); Scalar v = internal::random(); - m1.coeffRef(i,j) = v; - refMat1.coeffRef(i,j) = v; + m1.coeffRef(i, j) = v; + refMat1.coeffRef(i, j) = v; VERIFY_IS_APPROX(m1, refMat1); - if(internal::random(0,10)<2) - m1.makeCompressed(); + if (internal::random(0, 10) < 2) m1.makeCompressed(); } m1.setIdentity(); refMat1.setIdentity(); @@ -594,48 +591,46 @@ template void sparse_basic(const SparseMatrixType& re SparseMatrixType m2(rows, cols); initSparse(density, refMat2, m2); IteratorType static_array[2]; - static_array[0] = IteratorType(m2,0); - static_array[1] = IteratorType(m2,m2.outerSize()-1); - VERIFY( static_array[0] || m2.innerVector(static_array[0].outer()).nonZeros() == 0 ); - VERIFY( static_array[1] || m2.innerVector(static_array[1].outer()).nonZeros() == 0 ); - if(static_array[0] && static_array[1]) - { + static_array[0] = IteratorType(m2, 0); + static_array[1] = IteratorType(m2, m2.outerSize() - 1); + VERIFY(static_array[0] || m2.innerVector(static_array[0].outer()).nonZeros() == 0); + VERIFY(static_array[1] || m2.innerVector(static_array[1].outer()).nonZeros() == 0); + if (static_array[0] && static_array[1]) { ++(static_array[1]); - static_array[1] = IteratorType(m2,0); - VERIFY( static_array[1] ); - VERIFY( static_array[1].index() == static_array[0].index() ); - VERIFY( static_array[1].outer() == static_array[0].outer() ); - VERIFY( static_array[1].value() == static_array[0].value() ); + static_array[1] = IteratorType(m2, 0); + VERIFY(static_array[1]); + VERIFY(static_array[1].index() == static_array[0].index()); + VERIFY(static_array[1].outer() == static_array[0].outer()); + VERIFY(static_array[1].value() == static_array[0].value()); } std::vector iters(2); - iters[0] = IteratorType(m2,0); - iters[1] = IteratorType(m2,m2.outerSize()-1); + iters[0] = IteratorType(m2, 0); + iters[1] = IteratorType(m2, m2.outerSize() - 1); } } -template -void big_sparse_triplet(Index rows, Index cols, double density) { +template void big_sparse_triplet(Index rows, Index cols, double density) +{ typedef typename SparseMatrixType::StorageIndex StorageIndex; typedef typename SparseMatrixType::Scalar Scalar; - typedef Triplet TripletType; + typedef Triplet TripletType; std::vector triplets; - double nelements = density * rows*cols; - VERIFY(nelements>=0 && nelements < NumTraits::highest()); + double nelements = density * rows * cols; + VERIFY(nelements >= 0 && nelements < NumTraits::highest()); Index ntriplets = Index(nelements); triplets.reserve(ntriplets); Scalar sum = Scalar(0); - for(Index i=0;i(0,rows-1); - Index c = internal::random(0,cols-1); + for (Index i = 0; i < ntriplets; ++i) { + Index r = internal::random(0, rows - 1); + Index c = internal::random(0, cols - 1); // use positive values to prevent numerical cancellation errors in sum Scalar v = numext::abs(internal::random()); - triplets.push_back(TripletType(r,c,v)); + triplets.push_back(TripletType(r, c, v)); sum += v; } - SparseMatrixType m(rows,cols); + SparseMatrixType m(rows, cols); m.setFromTriplets(triplets.begin(), triplets.end()); VERIFY(m.nonZeros() <= ntriplets); VERIFY_IS_APPROX(sum, m.sum()); @@ -644,45 +639,44 @@ void big_sparse_triplet(Index rows, Index cols, double density) { void test_sparse_basic() { - for(int i = 0; i < g_repeat; i++) { - int r = Eigen::internal::random(1,200), c = Eigen::internal::random(1,200); - if(Eigen::internal::random(0,4) == 0) { - r = c; // check square matrices in 25% of tries + for (int i = 0; i < g_repeat; i++) { + int r = Eigen::internal::random(1, 200), c = Eigen::internal::random(1, 200); + if (Eigen::internal::random(0, 4) == 0) { + r = c;// check square matrices in 25% of tries } - EIGEN_UNUSED_VARIABLE(r+c); - CALL_SUBTEST_1(( sparse_basic(SparseMatrix(1, 1)) )); - CALL_SUBTEST_1(( sparse_basic(SparseMatrix(8, 8)) )); - CALL_SUBTEST_2(( sparse_basic(SparseMatrix, ColMajor>(r, c)) )); - CALL_SUBTEST_2(( sparse_basic(SparseMatrix, RowMajor>(r, c)) )); - CALL_SUBTEST_1(( sparse_basic(SparseMatrix(r, c)) )); - CALL_SUBTEST_5(( sparse_basic(SparseMatrix(r, c)) )); - CALL_SUBTEST_5(( sparse_basic(SparseMatrix(r, c)) )); - - r = Eigen::internal::random(1,100); - c = Eigen::internal::random(1,100); - if(Eigen::internal::random(0,4) == 0) { - r = c; // check square matrices in 25% of tries + EIGEN_UNUSED_VARIABLE(r + c); + CALL_SUBTEST_1((sparse_basic(SparseMatrix(1, 1)))); + CALL_SUBTEST_1((sparse_basic(SparseMatrix(8, 8)))); + CALL_SUBTEST_2((sparse_basic(SparseMatrix, ColMajor>(r, c)))); + CALL_SUBTEST_2((sparse_basic(SparseMatrix, RowMajor>(r, c)))); + CALL_SUBTEST_1((sparse_basic(SparseMatrix(r, c)))); + CALL_SUBTEST_5((sparse_basic(SparseMatrix(r, c)))); + CALL_SUBTEST_5((sparse_basic(SparseMatrix(r, c)))); + + r = Eigen::internal::random(1, 100); + c = Eigen::internal::random(1, 100); + if (Eigen::internal::random(0, 4) == 0) { + r = c;// check square matrices in 25% of tries } - - CALL_SUBTEST_6(( sparse_basic(SparseMatrix(short(r), short(c))) )); - CALL_SUBTEST_6(( sparse_basic(SparseMatrix(short(r), short(c))) )); + + CALL_SUBTEST_6((sparse_basic(SparseMatrix(short(r), short(c))))); + CALL_SUBTEST_6((sparse_basic(SparseMatrix(short(r), short(c))))); } // Regression test for bug 900: (manually insert higher values here, if you have enough RAM): - CALL_SUBTEST_3((big_sparse_triplet >(10000, 10000, 0.125))); - CALL_SUBTEST_4((big_sparse_triplet >(10000, 10000, 0.125))); + CALL_SUBTEST_3((big_sparse_triplet>(10000, 10000, 0.125))); + CALL_SUBTEST_4((big_sparse_triplet>(10000, 10000, 0.125))); // Regression test for bug 1105 #ifdef EIGEN_TEST_PART_7 { - int n = Eigen::internal::random(200,600); - SparseMatrix,0, long> mat(n, n); + int n = Eigen::internal::random(200, 600); + SparseMatrix, 0, long> mat(n, n); std::complex val; - for(int i=0; i -typename Eigen::internal::enable_if<(T::Flags&RowMajorBit)==RowMajorBit, typename T::RowXpr>::type -innervec(T& A, Index i) +typename Eigen::internal::enable_if<(T::Flags & RowMajorBit) == RowMajorBit, typename T::RowXpr>::type innervec(T &A, + Index i) { return A.row(i); } template -typename Eigen::internal::enable_if<(T::Flags&RowMajorBit)==0, typename T::ColXpr>::type -innervec(T& A, Index i) +typename Eigen::internal::enable_if<(T::Flags & RowMajorBit) == 0, typename T::ColXpr>::type innervec(T &A, Index i) { return A.col(i); } -template void sparse_block(const SparseMatrixType& ref) +template void sparse_block(const SparseMatrixType &ref) { const Index rows = ref.rows(); const Index cols = ref.cols(); @@ -33,10 +32,10 @@ template void sparse_block(const SparseMatrixType& re typedef typename SparseMatrixType::Scalar Scalar; typedef typename SparseMatrixType::StorageIndex StorageIndex; - double density = (std::max)(8./(rows*cols), 0.01); - typedef Matrix DenseMatrix; - typedef Matrix DenseVector; - typedef Matrix RowDenseVector; + double density = (std::max)(8. / (rows * cols), 0.01); + typedef Matrix DenseMatrix; + typedef Matrix DenseVector; + typedef Matrix RowDenseVector; typedef SparseVector SparseVectorType; Scalar s1 = internal::random(); @@ -48,71 +47,59 @@ template void sparse_block(const SparseMatrixType& re VERIFY_IS_APPROX(m, refMat); // test InnerIterators and Block expressions - for (int t=0; t<10; ++t) - { - Index j = internal::random(0,cols-2); - Index i = internal::random(0,rows-2); - Index w = internal::random(1,cols-j); - Index h = internal::random(1,rows-i); - - VERIFY_IS_APPROX(m.block(i,j,h,w), refMat.block(i,j,h,w)); - for(Index c=0; c(0, cols - 2); + Index i = internal::random(0, rows - 2); + Index w = internal::random(1, cols - j); + Index h = internal::random(1, rows - i); + + VERIFY_IS_APPROX(m.block(i, j, h, w), refMat.block(i, j, h, w)); + for (Index c = 0; c < w; c++) { + VERIFY_IS_APPROX(m.block(i, j, h, w).col(c), refMat.block(i, j, h, w).col(c)); + for (Index r = 0; r < h; r++) { + VERIFY_IS_APPROX(m.block(i, j, h, w).col(c).coeff(r), refMat.block(i, j, h, w).col(c).coeff(r)); + VERIFY_IS_APPROX(m.block(i, j, h, w).coeff(r, c), refMat.block(i, j, h, w).coeff(r, c)); } } - for(Index r=0; r void sparse_block(const SparseMatrixType& re DenseMatrix refMat2 = DenseMatrix::Zero(rows, cols); SparseMatrixType m2(rows, cols); initSparse(density, refMat2, m2); - Index j0 = internal::random(0,outer-1); - Index j1 = internal::random(0,outer-1); - Index r0 = internal::random(0,rows-1); - Index c0 = internal::random(0,cols-1); + Index j0 = internal::random(0, outer - 1); + Index j1 = internal::random(0, outer - 1); + Index r0 = internal::random(0, rows - 1); + Index c0 = internal::random(0, cols - 1); - VERIFY_IS_APPROX(m2.innerVector(j0), innervec(refMat2,j0)); - VERIFY_IS_APPROX(m2.innerVector(j0)+m2.innerVector(j1), innervec(refMat2,j0)+innervec(refMat2,j1)); + VERIFY_IS_APPROX(m2.innerVector(j0), innervec(refMat2, j0)); + VERIFY_IS_APPROX(m2.innerVector(j0) + m2.innerVector(j1), innervec(refMat2, j0) + innervec(refMat2, j1)); m2.innerVector(j0) *= Scalar(2); - innervec(refMat2,j0) *= Scalar(2); + innervec(refMat2, j0) *= Scalar(2); VERIFY_IS_APPROX(m2, refMat2); m2.row(r0) *= Scalar(3); @@ -152,33 +139,29 @@ template void sparse_block(const SparseMatrixType& re VERIFY_IS_APPROX(m2, refMat2); SparseVectorType v1; - VERIFY_IS_APPROX(v1 = m2.col(c0) * 4, refMat2.col(c0)*4); - VERIFY_IS_APPROX(v1 = m2.row(r0) * 4, refMat2.row(r0).transpose()*4); - - SparseMatrixType m3(rows,cols); - m3.reserve(VectorXi::Constant(outer,int(inner/2))); - for(Index j=0; j(k+1); - for(Index j=0; j<(std::min)(outer, inner); ++j) - { - VERIFY(j==numext::real(m3.innerVector(j).nonZeros())); - if(j>0) - VERIFY(j==numext::real(m3.innerVector(j).lastCoeff())); + VERIFY_IS_APPROX(v1 = m2.col(c0) * 4, refMat2.col(c0) * 4); + VERIFY_IS_APPROX(v1 = m2.row(r0) * 4, refMat2.row(r0).transpose() * 4); + + SparseMatrixType m3(rows, cols); + m3.reserve(VectorXi::Constant(outer, int(inner / 2))); + for (Index j = 0; j < outer; ++j) + for (Index k = 0; k < (std::min)(j, inner); ++k) + m3.insertByOuterInner(j, k) = internal::convert_index(k + 1); + for (Index j = 0; j < (std::min)(outer, inner); ++j) { + VERIFY(j == numext::real(m3.innerVector(j).nonZeros())); + if (j > 0) VERIFY(j == numext::real(m3.innerVector(j).lastCoeff())); } m3.makeCompressed(); - for(Index j=0; j<(std::min)(outer, inner); ++j) - { - VERIFY(j==numext::real(m3.innerVector(j).nonZeros())); - if(j>0) - VERIFY(j==numext::real(m3.innerVector(j).lastCoeff())); + for (Index j = 0; j < (std::min)(outer, inner); ++j) { + VERIFY(j == numext::real(m3.innerVector(j).nonZeros())); + if (j > 0) VERIFY(j == numext::real(m3.innerVector(j).lastCoeff())); } VERIFY(m3.innerVector(j0).nonZeros() == m3.transpose().innerVector(j0).nonZeros()); -// m2.innerVector(j0) = 2*m2.innerVector(j1); -// refMat2.col(j0) = 2*refMat2.col(j1); -// VERIFY_IS_APPROX(m2, refMat2); + // m2.innerVector(j0) = 2*m2.innerVector(j1); + // refMat2.col(j0) = 2*refMat2.col(j1); + // VERIFY_IS_APPROX(m2, refMat2); } // test innerVectors() @@ -186,31 +169,31 @@ template void sparse_block(const SparseMatrixType& re DenseMatrix refMat2 = DenseMatrix::Zero(rows, cols); SparseMatrixType m2(rows, cols); initSparse(density, refMat2, m2); - if(internal::random(0,1)>0.5f) m2.makeCompressed(); - Index j0 = internal::random(0,outer-2); - Index j1 = internal::random(0,outer-2); - Index n0 = internal::random(1,outer-(std::max)(j0,j1)); - if(SparseMatrixType::IsRowMajor) - VERIFY_IS_APPROX(m2.innerVectors(j0,n0), refMat2.block(j0,0,n0,cols)); + if (internal::random(0, 1) > 0.5f) m2.makeCompressed(); + Index j0 = internal::random(0, outer - 2); + Index j1 = internal::random(0, outer - 2); + Index n0 = internal::random(1, outer - (std::max)(j0, j1)); + if (SparseMatrixType::IsRowMajor) + VERIFY_IS_APPROX(m2.innerVectors(j0, n0), refMat2.block(j0, 0, n0, cols)); else - VERIFY_IS_APPROX(m2.innerVectors(j0,n0), refMat2.block(0,j0,rows,n0)); - if(SparseMatrixType::IsRowMajor) - VERIFY_IS_APPROX(m2.innerVectors(j0,n0)+m2.innerVectors(j1,n0), - refMat2.middleRows(j0,n0)+refMat2.middleRows(j1,n0)); + VERIFY_IS_APPROX(m2.innerVectors(j0, n0), refMat2.block(0, j0, rows, n0)); + if (SparseMatrixType::IsRowMajor) + VERIFY_IS_APPROX( + m2.innerVectors(j0, n0) + m2.innerVectors(j1, n0), refMat2.middleRows(j0, n0) + refMat2.middleRows(j1, n0)); else - VERIFY_IS_APPROX(m2.innerVectors(j0,n0)+m2.innerVectors(j1,n0), - refMat2.block(0,j0,rows,n0)+refMat2.block(0,j1,rows,n0)); - + VERIFY_IS_APPROX(m2.innerVectors(j0, n0) + m2.innerVectors(j1, n0), + refMat2.block(0, j0, rows, n0) + refMat2.block(0, j1, rows, n0)); + VERIFY_IS_APPROX(m2, refMat2); - - VERIFY(m2.innerVectors(j0,n0).nonZeros() == m2.transpose().innerVectors(j0,n0).nonZeros()); - - m2.innerVectors(j0,n0) = m2.innerVectors(j0,n0) + m2.innerVectors(j1,n0); - if(SparseMatrixType::IsRowMajor) - refMat2.middleRows(j0,n0) = (refMat2.middleRows(j0,n0) + refMat2.middleRows(j1,n0)).eval(); + + VERIFY(m2.innerVectors(j0, n0).nonZeros() == m2.transpose().innerVectors(j0, n0).nonZeros()); + + m2.innerVectors(j0, n0) = m2.innerVectors(j0, n0) + m2.innerVectors(j1, n0); + if (SparseMatrixType::IsRowMajor) + refMat2.middleRows(j0, n0) = (refMat2.middleRows(j0, n0) + refMat2.middleRows(j1, n0)).eval(); else - refMat2.middleCols(j0,n0) = (refMat2.middleCols(j0,n0) + refMat2.middleCols(j1,n0)).eval(); - + refMat2.middleCols(j0, n0) = (refMat2.middleCols(j0, n0) + refMat2.middleCols(j1, n0)).eval(); + VERIFY_IS_APPROX(m2, refMat2); } @@ -219,99 +202,93 @@ template void sparse_block(const SparseMatrixType& re DenseMatrix refMat2 = DenseMatrix::Zero(rows, cols); SparseMatrixType m2(rows, cols); initSparse(density, refMat2, m2); - Index j0 = internal::random(0,outer-2); - Index j1 = internal::random(0,outer-2); - Index n0 = internal::random(1,outer-(std::max)(j0,j1)); - if(SparseMatrixType::IsRowMajor) - VERIFY_IS_APPROX(m2.block(j0,0,n0,cols), refMat2.block(j0,0,n0,cols)); + Index j0 = internal::random(0, outer - 2); + Index j1 = internal::random(0, outer - 2); + Index n0 = internal::random(1, outer - (std::max)(j0, j1)); + if (SparseMatrixType::IsRowMajor) + VERIFY_IS_APPROX(m2.block(j0, 0, n0, cols), refMat2.block(j0, 0, n0, cols)); else - VERIFY_IS_APPROX(m2.block(0,j0,rows,n0), refMat2.block(0,j0,rows,n0)); - - if(SparseMatrixType::IsRowMajor) - VERIFY_IS_APPROX(m2.block(j0,0,n0,cols)+m2.block(j1,0,n0,cols), - refMat2.block(j0,0,n0,cols)+refMat2.block(j1,0,n0,cols)); + VERIFY_IS_APPROX(m2.block(0, j0, rows, n0), refMat2.block(0, j0, rows, n0)); + + if (SparseMatrixType::IsRowMajor) + VERIFY_IS_APPROX(m2.block(j0, 0, n0, cols) + m2.block(j1, 0, n0, cols), + refMat2.block(j0, 0, n0, cols) + refMat2.block(j1, 0, n0, cols)); else - VERIFY_IS_APPROX(m2.block(0,j0,rows,n0)+m2.block(0,j1,rows,n0), - refMat2.block(0,j0,rows,n0)+refMat2.block(0,j1,rows,n0)); - - Index i = internal::random(0,m2.outerSize()-1); - if(SparseMatrixType::IsRowMajor) { + VERIFY_IS_APPROX(m2.block(0, j0, rows, n0) + m2.block(0, j1, rows, n0), + refMat2.block(0, j0, rows, n0) + refMat2.block(0, j1, rows, n0)); + + Index i = internal::random(0, m2.outerSize() - 1); + if (SparseMatrixType::IsRowMajor) { m2.innerVector(i) = m2.innerVector(i) * s1; refMat2.row(i) = refMat2.row(i) * s1; - VERIFY_IS_APPROX(m2,refMat2); + VERIFY_IS_APPROX(m2, refMat2); } else { m2.innerVector(i) = m2.innerVector(i) * s1; refMat2.col(i) = refMat2.col(i) * s1; - VERIFY_IS_APPROX(m2,refMat2); + VERIFY_IS_APPROX(m2, refMat2); } - - Index r0 = internal::random(0,rows-2); - Index c0 = internal::random(0,cols-2); - Index r1 = internal::random(1,rows-r0); - Index c1 = internal::random(1,cols-c0); - + + Index r0 = internal::random(0, rows - 2); + Index c0 = internal::random(0, cols - 2); + Index r1 = internal::random(1, rows - r0); + Index c1 = internal::random(1, cols - c0); + VERIFY_IS_APPROX(DenseVector(m2.col(c0)), refMat2.col(c0)); VERIFY_IS_APPROX(m2.col(c0), refMat2.col(c0)); - + VERIFY_IS_APPROX(RowDenseVector(m2.row(r0)), refMat2.row(r0)); VERIFY_IS_APPROX(m2.row(r0), refMat2.row(r0)); - VERIFY_IS_APPROX(m2.block(r0,c0,r1,c1), refMat2.block(r0,c0,r1,c1)); - VERIFY_IS_APPROX((2*m2).block(r0,c0,r1,c1), (2*refMat2).block(r0,c0,r1,c1)); + VERIFY_IS_APPROX(m2.block(r0, c0, r1, c1), refMat2.block(r0, c0, r1, c1)); + VERIFY_IS_APPROX((2 * m2).block(r0, c0, r1, c1), (2 * refMat2).block(r0, c0, r1, c1)); - if(m2.nonZeros()>0) - { + if (m2.nonZeros() > 0) { VERIFY_IS_APPROX(m2, refMat2); SparseMatrixType m3(rows, cols); - DenseMatrix refMat3(rows, cols); refMat3.setZero(); - Index n = internal::random(1,10); - for(Index k=0; k(0,outer-1); - Index o2 = internal::random(0,outer-1); - if(SparseMatrixType::IsRowMajor) - { + DenseMatrix refMat3(rows, cols); + refMat3.setZero(); + Index n = internal::random(1, 10); + for (Index k = 0; k < n; ++k) { + Index o1 = internal::random(0, outer - 1); + Index o2 = internal::random(0, outer - 1); + if (SparseMatrixType::IsRowMajor) { m3.innerVector(o1) = m2.row(o2); refMat3.row(o1) = refMat2.row(o2); - } - else - { + } else { m3.innerVector(o1) = m2.col(o2); refMat3.col(o1) = refMat2.col(o2); } - if(internal::random()) - m3.makeCompressed(); + if (internal::random()) m3.makeCompressed(); } - if(m3.nonZeros()>0) - VERIFY_IS_APPROX(m3, refMat3); + if (m3.nonZeros() > 0) VERIFY_IS_APPROX(m3, refMat3); } } } void test_sparse_block() { - for(int i = 0; i < g_repeat; i++) { - int r = Eigen::internal::random(1,200), c = Eigen::internal::random(1,200); - if(Eigen::internal::random(0,4) == 0) { - r = c; // check square matrices in 25% of tries + for (int i = 0; i < g_repeat; i++) { + int r = Eigen::internal::random(1, 200), c = Eigen::internal::random(1, 200); + if (Eigen::internal::random(0, 4) == 0) { + r = c;// check square matrices in 25% of tries } - EIGEN_UNUSED_VARIABLE(r+c); - CALL_SUBTEST_1(( sparse_block(SparseMatrix(1, 1)) )); - CALL_SUBTEST_1(( sparse_block(SparseMatrix(8, 8)) )); - CALL_SUBTEST_1(( sparse_block(SparseMatrix(r, c)) )); - CALL_SUBTEST_2(( sparse_block(SparseMatrix, ColMajor>(r, c)) )); - CALL_SUBTEST_2(( sparse_block(SparseMatrix, RowMajor>(r, c)) )); - - CALL_SUBTEST_3(( sparse_block(SparseMatrix(r, c)) )); - CALL_SUBTEST_3(( sparse_block(SparseMatrix(r, c)) )); - - r = Eigen::internal::random(1,100); - c = Eigen::internal::random(1,100); - if(Eigen::internal::random(0,4) == 0) { - r = c; // check square matrices in 25% of tries + EIGEN_UNUSED_VARIABLE(r + c); + CALL_SUBTEST_1((sparse_block(SparseMatrix(1, 1)))); + CALL_SUBTEST_1((sparse_block(SparseMatrix(8, 8)))); + CALL_SUBTEST_1((sparse_block(SparseMatrix(r, c)))); + CALL_SUBTEST_2((sparse_block(SparseMatrix, ColMajor>(r, c)))); + CALL_SUBTEST_2((sparse_block(SparseMatrix, RowMajor>(r, c)))); + + CALL_SUBTEST_3((sparse_block(SparseMatrix(r, c)))); + CALL_SUBTEST_3((sparse_block(SparseMatrix(r, c)))); + + r = Eigen::internal::random(1, 100); + c = Eigen::internal::random(1, 100); + if (Eigen::internal::random(0, 4) == 0) { + r = c;// check square matrices in 25% of tries } - - CALL_SUBTEST_4(( sparse_block(SparseMatrix(short(r), short(c))) )); - CALL_SUBTEST_4(( sparse_block(SparseMatrix(short(r), short(c))) )); + + CALL_SUBTEST_4((sparse_block(SparseMatrix(short(r), short(c))))); + CALL_SUBTEST_4((sparse_block(SparseMatrix(short(r), short(c))))); } } diff --git a/filmulator-gui/core/nlmeans/eigen/test/sparse_permutations.cpp b/filmulator-gui/core/nlmeans/eigen/test/sparse_permutations.cpp index b82cceff..0f398fc2 100644 --- a/filmulator-gui/core/nlmeans/eigen/test/sparse_permutations.cpp +++ b/filmulator-gui/core/nlmeans/eigen/test/sparse_permutations.cpp @@ -9,136 +9,137 @@ static long int nb_transposed_copies; -#define EIGEN_SPARSE_TRANSPOSED_COPY_PLUGIN {nb_transposed_copies++;} -#define VERIFY_TRANSPOSITION_COUNT(XPR,N) {\ - nb_transposed_copies = 0; \ - XPR; \ - if(nb_transposed_copies!=N) std::cerr << "nb_transposed_copies == " << nb_transposed_copies << "\n"; \ - VERIFY( (#XPR) && nb_transposed_copies==N ); \ +#define EIGEN_SPARSE_TRANSPOSED_COPY_PLUGIN \ + { \ + nb_transposed_copies++; \ + } +#define VERIFY_TRANSPOSITION_COUNT(XPR, N) \ + { \ + nb_transposed_copies = 0; \ + XPR; \ + if (nb_transposed_copies != N) std::cerr << "nb_transposed_copies == " << nb_transposed_copies << "\n"; \ + VERIFY((#XPR) && nb_transposed_copies == N); \ } #include "sparse.h" -template -bool is_sorted(const T& mat) { - for(Index k = 0; k bool is_sorted(const T &mat) +{ + for (Index k = 0; k < mat.outerSize(); ++k) { Index prev = -1; - for(typename T::InnerIterator it(mat,k); it; ++it) - { - if(prev>=it.index()) - return false; + for (typename T::InnerIterator it(mat, k); it; ++it) { + if (prev >= it.index()) return false; prev = it.index(); } } return true; } -template -typename internal::nested_eval::type eval(const T &xpr) +template typename internal::nested_eval::type eval(const T &xpr) { - VERIFY( int(internal::nested_eval::type::Flags&RowMajorBit) == int(internal::evaluator::Flags&RowMajorBit) ); + VERIFY( + int(internal::nested_eval::type::Flags & RowMajorBit) == int(internal::evaluator::Flags & RowMajorBit)); return xpr; } -template void sparse_permutations(const SparseMatrixType& ref) +template void sparse_permutations(const SparseMatrixType &ref) { const Index rows = ref.rows(); const Index cols = ref.cols(); typedef typename SparseMatrixType::Scalar Scalar; typedef typename SparseMatrixType::StorageIndex StorageIndex; typedef SparseMatrix OtherSparseMatrixType; - typedef Matrix DenseMatrix; - typedef Matrix VectorI; -// bool IsRowMajor1 = SparseMatrixType::IsRowMajor; -// bool IsRowMajor2 = OtherSparseMatrixType::IsRowMajor; - - double density = (std::max)(8./(rows*cols), 0.01); - - SparseMatrixType mat(rows, cols), up(rows,cols), lo(rows,cols); + typedef Matrix DenseMatrix; + typedef Matrix VectorI; + // bool IsRowMajor1 = SparseMatrixType::IsRowMajor; + // bool IsRowMajor2 = OtherSparseMatrixType::IsRowMajor; + + double density = (std::max)(8. / (rows * cols), 0.01); + + SparseMatrixType mat(rows, cols), up(rows, cols), lo(rows, cols); OtherSparseMatrixType res; DenseMatrix mat_d = DenseMatrix::Zero(rows, cols), up_sym_d, lo_sym_d, res_d; - + initSparse(density, mat_d, mat, 0); up = mat.template triangularView(); lo = mat.template triangularView(); - + up_sym_d = mat_d.template selfadjointView(); lo_sym_d = mat_d.template selfadjointView(); - + VERIFY_IS_APPROX(mat, mat_d); VERIFY_IS_APPROX(up, DenseMatrix(mat_d.template triangularView())); VERIFY_IS_APPROX(lo, DenseMatrix(mat_d.template triangularView())); - + PermutationMatrix p, p_null; VectorI pi; randomPermutationVector(pi, cols); p.indices() = pi; - VERIFY( is_sorted( ::eval(mat*p) )); - VERIFY( is_sorted( res = mat*p )); - VERIFY_TRANSPOSITION_COUNT( ::eval(mat*p), 0); - //VERIFY_TRANSPOSITION_COUNT( res = mat*p, IsRowMajor ? 1 : 0 ); - res_d = mat_d*p; + VERIFY(is_sorted(::eval(mat * p))); + VERIFY(is_sorted(res = mat * p)); + VERIFY_TRANSPOSITION_COUNT(::eval(mat * p), 0); + // VERIFY_TRANSPOSITION_COUNT( res = mat*p, IsRowMajor ? 1 : 0 ); + res_d = mat_d * p; VERIFY(res.isApprox(res_d) && "mat*p"); - VERIFY( is_sorted( ::eval(p*mat) )); - VERIFY( is_sorted( res = p*mat )); - VERIFY_TRANSPOSITION_COUNT( ::eval(p*mat), 0); - res_d = p*mat_d; + VERIFY(is_sorted(::eval(p * mat))); + VERIFY(is_sorted(res = p * mat)); + VERIFY_TRANSPOSITION_COUNT(::eval(p * mat), 0); + res_d = p * mat_d; VERIFY(res.isApprox(res_d) && "p*mat"); - VERIFY( is_sorted( (mat*p).eval() )); - VERIFY( is_sorted( res = mat*p.inverse() )); - VERIFY_TRANSPOSITION_COUNT( ::eval(mat*p.inverse()), 0); - res_d = mat*p.inverse(); + VERIFY(is_sorted((mat * p).eval())); + VERIFY(is_sorted(res = mat * p.inverse())); + VERIFY_TRANSPOSITION_COUNT(::eval(mat * p.inverse()), 0); + res_d = mat * p.inverse(); VERIFY(res.isApprox(res_d) && "mat*inv(p)"); - VERIFY( is_sorted( (p*mat+p*mat).eval() )); - VERIFY( is_sorted( res = p.inverse()*mat )); - VERIFY_TRANSPOSITION_COUNT( ::eval(p.inverse()*mat), 0); - res_d = p.inverse()*mat_d; + VERIFY(is_sorted((p * mat + p * mat).eval())); + VERIFY(is_sorted(res = p.inverse() * mat)); + VERIFY_TRANSPOSITION_COUNT(::eval(p.inverse() * mat), 0); + res_d = p.inverse() * mat_d; VERIFY(res.isApprox(res_d) && "inv(p)*mat"); - VERIFY( is_sorted( (p * mat * p.inverse()).eval() )); - VERIFY( is_sorted( res = mat.twistedBy(p) )); - VERIFY_TRANSPOSITION_COUNT( ::eval(p * mat * p.inverse()), 0); + VERIFY(is_sorted((p * mat * p.inverse()).eval())); + VERIFY(is_sorted(res = mat.twistedBy(p))); + VERIFY_TRANSPOSITION_COUNT(::eval(p * mat * p.inverse()), 0); res_d = (p * mat_d) * p.inverse(); VERIFY(res.isApprox(res_d) && "p*mat*inv(p)"); - - VERIFY( is_sorted( res = mat.template selfadjointView().twistedBy(p_null) )); + + VERIFY(is_sorted(res = mat.template selfadjointView().twistedBy(p_null))); res_d = up_sym_d; VERIFY(res.isApprox(res_d) && "full selfadjoint upper to full"); - - VERIFY( is_sorted( res = mat.template selfadjointView().twistedBy(p_null) )); + + VERIFY(is_sorted(res = mat.template selfadjointView().twistedBy(p_null))); res_d = lo_sym_d; VERIFY(res.isApprox(res_d) && "full selfadjoint lower to full"); - - - VERIFY( is_sorted( res = up.template selfadjointView().twistedBy(p_null) )); + + + VERIFY(is_sorted(res = up.template selfadjointView().twistedBy(p_null))); res_d = up_sym_d; VERIFY(res.isApprox(res_d) && "upper selfadjoint to full"); - - VERIFY( is_sorted( res = lo.template selfadjointView().twistedBy(p_null) )); + + VERIFY(is_sorted(res = lo.template selfadjointView().twistedBy(p_null))); res_d = lo_sym_d; VERIFY(res.isApprox(res_d) && "lower selfadjoint full"); - VERIFY( is_sorted( res = mat.template selfadjointView() )); + VERIFY(is_sorted(res = mat.template selfadjointView())); res_d = up_sym_d; VERIFY(res.isApprox(res_d) && "full selfadjoint upper to full"); - VERIFY( is_sorted( res = mat.template selfadjointView() )); + VERIFY(is_sorted(res = mat.template selfadjointView())); res_d = lo_sym_d; VERIFY(res.isApprox(res_d) && "full selfadjoint lower to full"); - VERIFY( is_sorted( res = up.template selfadjointView() )); + VERIFY(is_sorted(res = up.template selfadjointView())); res_d = up_sym_d; VERIFY(res.isApprox(res_d) && "upper selfadjoint to full"); - VERIFY( is_sorted( res = lo.template selfadjointView() )); + VERIFY(is_sorted(res = lo.template selfadjointView())); res_d = lo_sym_d; VERIFY(res.isApprox(res_d) && "lower selfadjoint full"); @@ -159,78 +160,81 @@ template void sparse_permutations(c res_d = lo_sym_d.template triangularView(); VERIFY(res.isApprox(res_d) && "full selfadjoint lower to lower"); - - + res.template selfadjointView() = mat.template selfadjointView().twistedBy(p); res_d = ((p * up_sym_d) * p.inverse()).eval().template triangularView(); VERIFY(res.isApprox(res_d) && "full selfadjoint upper twisted to upper"); - + res.template selfadjointView() = mat.template selfadjointView().twistedBy(p); res_d = ((p * lo_sym_d) * p.inverse()).eval().template triangularView(); VERIFY(res.isApprox(res_d) && "full selfadjoint lower twisted to upper"); - + res.template selfadjointView() = mat.template selfadjointView().twistedBy(p); res_d = ((p * lo_sym_d) * p.inverse()).eval().template triangularView(); VERIFY(res.isApprox(res_d) && "full selfadjoint lower twisted to lower"); - + res.template selfadjointView() = mat.template selfadjointView().twistedBy(p); res_d = ((p * up_sym_d) * p.inverse()).eval().template triangularView(); VERIFY(res.isApprox(res_d) && "full selfadjoint upper twisted to lower"); - - + + res.template selfadjointView() = up.template selfadjointView().twistedBy(p); res_d = ((p * up_sym_d) * p.inverse()).eval().template triangularView(); VERIFY(res.isApprox(res_d) && "upper selfadjoint twisted to upper"); - + res.template selfadjointView() = lo.template selfadjointView().twistedBy(p); res_d = ((p * lo_sym_d) * p.inverse()).eval().template triangularView(); VERIFY(res.isApprox(res_d) && "lower selfadjoint twisted to upper"); - + res.template selfadjointView() = lo.template selfadjointView().twistedBy(p); res_d = ((p * lo_sym_d) * p.inverse()).eval().template triangularView(); VERIFY(res.isApprox(res_d) && "lower selfadjoint twisted to lower"); - + res.template selfadjointView() = up.template selfadjointView().twistedBy(p); res_d = ((p * up_sym_d) * p.inverse()).eval().template triangularView(); VERIFY(res.isApprox(res_d) && "upper selfadjoint twisted to lower"); - - VERIFY( is_sorted( res = mat.template selfadjointView().twistedBy(p) )); + + VERIFY(is_sorted(res = mat.template selfadjointView().twistedBy(p))); res_d = (p * up_sym_d) * p.inverse(); VERIFY(res.isApprox(res_d) && "full selfadjoint upper twisted to full"); - - VERIFY( is_sorted( res = mat.template selfadjointView().twistedBy(p) )); + + VERIFY(is_sorted(res = mat.template selfadjointView().twistedBy(p))); res_d = (p * lo_sym_d) * p.inverse(); VERIFY(res.isApprox(res_d) && "full selfadjoint lower twisted to full"); - - VERIFY( is_sorted( res = up.template selfadjointView().twistedBy(p) )); + + VERIFY(is_sorted(res = up.template selfadjointView().twistedBy(p))); res_d = (p * up_sym_d) * p.inverse(); VERIFY(res.isApprox(res_d) && "upper selfadjoint twisted to full"); - - VERIFY( is_sorted( res = lo.template selfadjointView().twistedBy(p) )); + + VERIFY(is_sorted(res = lo.template selfadjointView().twistedBy(p))); res_d = (p * lo_sym_d) * p.inverse(); VERIFY(res.isApprox(res_d) && "lower selfadjoint twisted to full"); } template void sparse_permutations_all(int size) { - CALL_SUBTEST(( sparse_permutations(SparseMatrix(size,size)) )); - CALL_SUBTEST(( sparse_permutations(SparseMatrix(size,size)) )); - CALL_SUBTEST(( sparse_permutations(SparseMatrix(size,size)) )); - CALL_SUBTEST(( sparse_permutations(SparseMatrix(size,size)) )); + CALL_SUBTEST((sparse_permutations(SparseMatrix(size, size)))); + CALL_SUBTEST((sparse_permutations(SparseMatrix(size, size)))); + CALL_SUBTEST((sparse_permutations(SparseMatrix(size, size)))); + CALL_SUBTEST((sparse_permutations(SparseMatrix(size, size)))); } void test_sparse_permutations() { - for(int i = 0; i < g_repeat; i++) { - int s = Eigen::internal::random(1,50); - CALL_SUBTEST_1(( sparse_permutations_all(s) )); - CALL_SUBTEST_2(( sparse_permutations_all >(s) )); + for (int i = 0; i < g_repeat; i++) { + int s = Eigen::internal::random(1, 50); + CALL_SUBTEST_1((sparse_permutations_all(s))); + CALL_SUBTEST_2((sparse_permutations_all>(s))); } - VERIFY((internal::is_same,OnTheRight,false,SparseShape>::ReturnType, - internal::nested_eval,PermutationMatrix,AliasFreeProduct>,1>::type>::value)); + VERIFY((internal::is_same< + internal::permutation_matrix_product, OnTheRight, false, SparseShape>::ReturnType, + internal::nested_eval, PermutationMatrix, AliasFreeProduct>, + 1>::type>::value)); - VERIFY((internal::is_same,OnTheLeft,false,SparseShape>::ReturnType, - internal::nested_eval,SparseMatrix,AliasFreeProduct>,1>::type>::value)); + VERIFY((internal::is_same< + internal::permutation_matrix_product, OnTheLeft, false, SparseShape>::ReturnType, + internal::nested_eval, SparseMatrix, AliasFreeProduct>, + 1>::type>::value)); } diff --git a/filmulator-gui/core/nlmeans/eigen/test/sparse_product.cpp b/filmulator-gui/core/nlmeans/eigen/test/sparse_product.cpp index 7f77bb74..c419fa5a 100644 --- a/filmulator-gui/core/nlmeans/eigen/test/sparse_product.cpp +++ b/filmulator-gui/core/nlmeans/eigen/test/sparse_product.cpp @@ -7,262 +7,281 @@ // Public License v. 2.0. If a copy of the MPL was not distributed // with this file, You can obtain one at http://mozilla.org/MPL/2.0/. -#if defined(_MSC_VER) && (_MSC_VER==1800) +#if defined(_MSC_VER) && (_MSC_VER == 1800) // This unit test takes forever to compile in Release mode with MSVC 2013, // multiple hours. So let's switch off optimization for this one. -#pragma optimize("",off) +#pragma optimize("", off) #endif static long int nb_temporaries; -inline void on_temporary_creation() { +inline void on_temporary_creation() +{ // here's a great place to set a breakpoint when debugging failures in this test! nb_temporaries++; } -#define EIGEN_SPARSE_CREATE_TEMPORARY_PLUGIN { on_temporary_creation(); } +#define EIGEN_SPARSE_CREATE_TEMPORARY_PLUGIN \ + { \ + on_temporary_creation(); \ + } #include "sparse.h" -#define VERIFY_EVALUATION_COUNT(XPR,N) {\ - nb_temporaries = 0; \ - CALL_SUBTEST( XPR ); \ - if(nb_temporaries!=N) std::cerr << "nb_temporaries == " << nb_temporaries << "\n"; \ - VERIFY( (#XPR) && nb_temporaries==N ); \ +#define VERIFY_EVALUATION_COUNT(XPR, N) \ + { \ + nb_temporaries = 0; \ + CALL_SUBTEST(XPR); \ + if (nb_temporaries != N) std::cerr << "nb_temporaries == " << nb_temporaries << "\n"; \ + VERIFY((#XPR) && nb_temporaries == N); \ } - template void sparse_product() { typedef typename SparseMatrixType::StorageIndex StorageIndex; Index n = 100; - const Index rows = internal::random(1,n); - const Index cols = internal::random(1,n); - const Index depth = internal::random(1,n); + const Index rows = internal::random(1, n); + const Index cols = internal::random(1, n); + const Index depth = internal::random(1, n); typedef typename SparseMatrixType::Scalar Scalar; enum { Flags = SparseMatrixType::Flags }; - double density = (std::max)(8./(rows*cols), 0.2); - typedef Matrix DenseMatrix; - typedef Matrix DenseVector; - typedef Matrix RowDenseVector; - typedef SparseVector ColSpVector; - typedef SparseVector RowSpVector; + double density = (std::max)(8. / (rows * cols), 0.2); + typedef Matrix DenseMatrix; + typedef Matrix DenseVector; + typedef Matrix RowDenseVector; + typedef SparseVector ColSpVector; + typedef SparseVector RowSpVector; Scalar s1 = internal::random(); Scalar s2 = internal::random(); // test matrix-matrix product { - DenseMatrix refMat2 = DenseMatrix::Zero(rows, depth); + DenseMatrix refMat2 = DenseMatrix::Zero(rows, depth); DenseMatrix refMat2t = DenseMatrix::Zero(depth, rows); - DenseMatrix refMat3 = DenseMatrix::Zero(depth, cols); + DenseMatrix refMat3 = DenseMatrix::Zero(depth, cols); DenseMatrix refMat3t = DenseMatrix::Zero(cols, depth); - DenseMatrix refMat4 = DenseMatrix::Zero(rows, cols); + DenseMatrix refMat4 = DenseMatrix::Zero(rows, cols); DenseMatrix refMat4t = DenseMatrix::Zero(cols, rows); - DenseMatrix refMat5 = DenseMatrix::Random(depth, cols); - DenseMatrix refMat6 = DenseMatrix::Random(rows, rows); + DenseMatrix refMat5 = DenseMatrix::Random(depth, cols); + DenseMatrix refMat6 = DenseMatrix::Random(rows, rows); DenseMatrix dm4 = DenseMatrix::Zero(rows, rows); -// DenseVector dv1 = DenseVector::Random(rows); - SparseMatrixType m2 (rows, depth); + // DenseVector dv1 = DenseVector::Random(rows); + SparseMatrixType m2(rows, depth); SparseMatrixType m2t(depth, rows); - SparseMatrixType m3 (depth, cols); + SparseMatrixType m3(depth, cols); SparseMatrixType m3t(cols, depth); - SparseMatrixType m4 (rows, cols); + SparseMatrixType m4(rows, cols); SparseMatrixType m4t(cols, rows); SparseMatrixType m6(rows, rows); - initSparse(density, refMat2, m2); + initSparse(density, refMat2, m2); initSparse(density, refMat2t, m2t); - initSparse(density, refMat3, m3); + initSparse(density, refMat3, m3); initSparse(density, refMat3t, m3t); - initSparse(density, refMat4, m4); + initSparse(density, refMat4, m4); initSparse(density, refMat4t, m4t); initSparse(density, refMat6, m6); -// int c = internal::random(0,depth-1); + // int c = internal::random(0,depth-1); // sparse * sparse - VERIFY_IS_APPROX(m4=m2*m3, refMat4=refMat2*refMat3); - VERIFY_IS_APPROX(m4=m2t.transpose()*m3, refMat4=refMat2t.transpose()*refMat3); - VERIFY_IS_APPROX(m4=m2t.transpose()*m3t.transpose(), refMat4=refMat2t.transpose()*refMat3t.transpose()); - VERIFY_IS_APPROX(m4=m2*m3t.transpose(), refMat4=refMat2*refMat3t.transpose()); - - VERIFY_IS_APPROX(m4 = m2*m3/s1, refMat4 = refMat2*refMat3/s1); - VERIFY_IS_APPROX(m4 = m2*m3*s1, refMat4 = refMat2*refMat3*s1); - VERIFY_IS_APPROX(m4 = s2*m2*m3*s1, refMat4 = s2*refMat2*refMat3*s1); - VERIFY_IS_APPROX(m4 = (m2+m2)*m3, refMat4 = (refMat2+refMat2)*refMat3); - VERIFY_IS_APPROX(m4 = m2*m3.leftCols(cols/2), refMat4 = refMat2*refMat3.leftCols(cols/2)); - VERIFY_IS_APPROX(m4 = m2*(m3+m3).leftCols(cols/2), refMat4 = refMat2*(refMat3+refMat3).leftCols(cols/2)); - - VERIFY_IS_APPROX(m4=(m2*m3).pruned(0), refMat4=refMat2*refMat3); - VERIFY_IS_APPROX(m4=(m2t.transpose()*m3).pruned(0), refMat4=refMat2t.transpose()*refMat3); - VERIFY_IS_APPROX(m4=(m2t.transpose()*m3t.transpose()).pruned(0), refMat4=refMat2t.transpose()*refMat3t.transpose()); - VERIFY_IS_APPROX(m4=(m2*m3t.transpose()).pruned(0), refMat4=refMat2*refMat3t.transpose()); + VERIFY_IS_APPROX(m4 = m2 * m3, refMat4 = refMat2 * refMat3); + VERIFY_IS_APPROX(m4 = m2t.transpose() * m3, refMat4 = refMat2t.transpose() * refMat3); + VERIFY_IS_APPROX(m4 = m2t.transpose() * m3t.transpose(), refMat4 = refMat2t.transpose() * refMat3t.transpose()); + VERIFY_IS_APPROX(m4 = m2 * m3t.transpose(), refMat4 = refMat2 * refMat3t.transpose()); + + VERIFY_IS_APPROX(m4 = m2 * m3 / s1, refMat4 = refMat2 * refMat3 / s1); + VERIFY_IS_APPROX(m4 = m2 * m3 * s1, refMat4 = refMat2 * refMat3 * s1); + VERIFY_IS_APPROX(m4 = s2 * m2 * m3 * s1, refMat4 = s2 * refMat2 * refMat3 * s1); + VERIFY_IS_APPROX(m4 = (m2 + m2) * m3, refMat4 = (refMat2 + refMat2) * refMat3); + VERIFY_IS_APPROX(m4 = m2 * m3.leftCols(cols / 2), refMat4 = refMat2 * refMat3.leftCols(cols / 2)); + VERIFY_IS_APPROX( + m4 = m2 * (m3 + m3).leftCols(cols / 2), refMat4 = refMat2 * (refMat3 + refMat3).leftCols(cols / 2)); + + VERIFY_IS_APPROX(m4 = (m2 * m3).pruned(0), refMat4 = refMat2 * refMat3); + VERIFY_IS_APPROX(m4 = (m2t.transpose() * m3).pruned(0), refMat4 = refMat2t.transpose() * refMat3); + VERIFY_IS_APPROX( + m4 = (m2t.transpose() * m3t.transpose()).pruned(0), refMat4 = refMat2t.transpose() * refMat3t.transpose()); + VERIFY_IS_APPROX(m4 = (m2 * m3t.transpose()).pruned(0), refMat4 = refMat2 * refMat3t.transpose()); // make sure the right product implementation is called: - if((!SparseMatrixType::IsRowMajor) && m2.rows()<=m3.cols()) - { - VERIFY_EVALUATION_COUNT(m4 = m2*m3, 3); // 1 temp for the result + 2 for transposing and get a sorted result. - VERIFY_EVALUATION_COUNT(m4 = (m2*m3).pruned(0), 1); - VERIFY_EVALUATION_COUNT(m4 = (m2*m3).eval().pruned(0), 4); + if ((!SparseMatrixType::IsRowMajor) && m2.rows() <= m3.cols()) { + VERIFY_EVALUATION_COUNT(m4 = m2 * m3, 3);// 1 temp for the result + 2 for transposing and get a sorted result. + VERIFY_EVALUATION_COUNT(m4 = (m2 * m3).pruned(0), 1); + VERIFY_EVALUATION_COUNT(m4 = (m2 * m3).eval().pruned(0), 4); } // and that pruning is effective: { - DenseMatrix Ad(2,2); + DenseMatrix Ad(2, 2); Ad << -1, 1, 1, 1; - SparseMatrixType As(Ad.sparseView()), B(2,2); - VERIFY_IS_EQUAL( (As*As.transpose()).eval().nonZeros(), 4); - VERIFY_IS_EQUAL( (Ad*Ad.transpose()).eval().sparseView().eval().nonZeros(), 2); - VERIFY_IS_EQUAL( (As*As.transpose()).pruned(1e-6).eval().nonZeros(), 2); + SparseMatrixType As(Ad.sparseView()), B(2, 2); + VERIFY_IS_EQUAL((As * As.transpose()).eval().nonZeros(), 4); + VERIFY_IS_EQUAL((Ad * Ad.transpose()).eval().sparseView().eval().nonZeros(), 2); + VERIFY_IS_EQUAL((As * As.transpose()).pruned(1e-6).eval().nonZeros(), 2); } // dense ?= sparse * sparse - VERIFY_IS_APPROX(dm4 =m2*m3, refMat4 =refMat2*refMat3); - VERIFY_IS_APPROX(dm4+=m2*m3, refMat4+=refMat2*refMat3); - VERIFY_IS_APPROX(dm4-=m2*m3, refMat4-=refMat2*refMat3); - VERIFY_IS_APPROX(dm4 =m2t.transpose()*m3, refMat4 =refMat2t.transpose()*refMat3); - VERIFY_IS_APPROX(dm4+=m2t.transpose()*m3, refMat4+=refMat2t.transpose()*refMat3); - VERIFY_IS_APPROX(dm4-=m2t.transpose()*m3, refMat4-=refMat2t.transpose()*refMat3); - VERIFY_IS_APPROX(dm4 =m2t.transpose()*m3t.transpose(), refMat4 =refMat2t.transpose()*refMat3t.transpose()); - VERIFY_IS_APPROX(dm4+=m2t.transpose()*m3t.transpose(), refMat4+=refMat2t.transpose()*refMat3t.transpose()); - VERIFY_IS_APPROX(dm4-=m2t.transpose()*m3t.transpose(), refMat4-=refMat2t.transpose()*refMat3t.transpose()); - VERIFY_IS_APPROX(dm4 =m2*m3t.transpose(), refMat4 =refMat2*refMat3t.transpose()); - VERIFY_IS_APPROX(dm4+=m2*m3t.transpose(), refMat4+=refMat2*refMat3t.transpose()); - VERIFY_IS_APPROX(dm4-=m2*m3t.transpose(), refMat4-=refMat2*refMat3t.transpose()); - VERIFY_IS_APPROX(dm4 = m2*m3*s1, refMat4 = refMat2*refMat3*s1); + VERIFY_IS_APPROX(dm4 = m2 * m3, refMat4 = refMat2 * refMat3); + VERIFY_IS_APPROX(dm4 += m2 * m3, refMat4 += refMat2 * refMat3); + VERIFY_IS_APPROX(dm4 -= m2 * m3, refMat4 -= refMat2 * refMat3); + VERIFY_IS_APPROX(dm4 = m2t.transpose() * m3, refMat4 = refMat2t.transpose() * refMat3); + VERIFY_IS_APPROX(dm4 += m2t.transpose() * m3, refMat4 += refMat2t.transpose() * refMat3); + VERIFY_IS_APPROX(dm4 -= m2t.transpose() * m3, refMat4 -= refMat2t.transpose() * refMat3); + VERIFY_IS_APPROX(dm4 = m2t.transpose() * m3t.transpose(), refMat4 = refMat2t.transpose() * refMat3t.transpose()); + VERIFY_IS_APPROX(dm4 += m2t.transpose() * m3t.transpose(), refMat4 += refMat2t.transpose() * refMat3t.transpose()); + VERIFY_IS_APPROX(dm4 -= m2t.transpose() * m3t.transpose(), refMat4 -= refMat2t.transpose() * refMat3t.transpose()); + VERIFY_IS_APPROX(dm4 = m2 * m3t.transpose(), refMat4 = refMat2 * refMat3t.transpose()); + VERIFY_IS_APPROX(dm4 += m2 * m3t.transpose(), refMat4 += refMat2 * refMat3t.transpose()); + VERIFY_IS_APPROX(dm4 -= m2 * m3t.transpose(), refMat4 -= refMat2 * refMat3t.transpose()); + VERIFY_IS_APPROX(dm4 = m2 * m3 * s1, refMat4 = refMat2 * refMat3 * s1); // test aliasing - m4 = m2; refMat4 = refMat2; - VERIFY_IS_APPROX(m4=m4*m3, refMat4=refMat4*refMat3); + m4 = m2; + refMat4 = refMat2; + VERIFY_IS_APPROX(m4 = m4 * m3, refMat4 = refMat4 * refMat3); // sparse * dense matrix - VERIFY_IS_APPROX(dm4=m2*refMat3, refMat4=refMat2*refMat3); - VERIFY_IS_APPROX(dm4=m2*refMat3t.transpose(), refMat4=refMat2*refMat3t.transpose()); - VERIFY_IS_APPROX(dm4=m2t.transpose()*refMat3, refMat4=refMat2t.transpose()*refMat3); - VERIFY_IS_APPROX(dm4=m2t.transpose()*refMat3t.transpose(), refMat4=refMat2t.transpose()*refMat3t.transpose()); - - VERIFY_IS_APPROX(dm4=m2*refMat3, refMat4=refMat2*refMat3); - VERIFY_IS_APPROX(dm4=dm4+m2*refMat3, refMat4=refMat4+refMat2*refMat3); - VERIFY_IS_APPROX(dm4+=m2*refMat3, refMat4+=refMat2*refMat3); - VERIFY_IS_APPROX(dm4-=m2*refMat3, refMat4-=refMat2*refMat3); - VERIFY_IS_APPROX(dm4.noalias()+=m2*refMat3, refMat4+=refMat2*refMat3); - VERIFY_IS_APPROX(dm4.noalias()-=m2*refMat3, refMat4-=refMat2*refMat3); - VERIFY_IS_APPROX(dm4=m2*(refMat3+refMat3), refMat4=refMat2*(refMat3+refMat3)); - VERIFY_IS_APPROX(dm4=m2t.transpose()*(refMat3+refMat5)*0.5, refMat4=refMat2t.transpose()*(refMat3+refMat5)*0.5); - + VERIFY_IS_APPROX(dm4 = m2 * refMat3, refMat4 = refMat2 * refMat3); + VERIFY_IS_APPROX(dm4 = m2 * refMat3t.transpose(), refMat4 = refMat2 * refMat3t.transpose()); + VERIFY_IS_APPROX(dm4 = m2t.transpose() * refMat3, refMat4 = refMat2t.transpose() * refMat3); + VERIFY_IS_APPROX( + dm4 = m2t.transpose() * refMat3t.transpose(), refMat4 = refMat2t.transpose() * refMat3t.transpose()); + + VERIFY_IS_APPROX(dm4 = m2 * refMat3, refMat4 = refMat2 * refMat3); + VERIFY_IS_APPROX(dm4 = dm4 + m2 * refMat3, refMat4 = refMat4 + refMat2 * refMat3); + VERIFY_IS_APPROX(dm4 += m2 * refMat3, refMat4 += refMat2 * refMat3); + VERIFY_IS_APPROX(dm4 -= m2 * refMat3, refMat4 -= refMat2 * refMat3); + VERIFY_IS_APPROX(dm4.noalias() += m2 * refMat3, refMat4 += refMat2 * refMat3); + VERIFY_IS_APPROX(dm4.noalias() -= m2 * refMat3, refMat4 -= refMat2 * refMat3); + VERIFY_IS_APPROX(dm4 = m2 * (refMat3 + refMat3), refMat4 = refMat2 * (refMat3 + refMat3)); + VERIFY_IS_APPROX( + dm4 = m2t.transpose() * (refMat3 + refMat5) * 0.5, refMat4 = refMat2t.transpose() * (refMat3 + refMat5) * 0.5); + // sparse * dense vector - VERIFY_IS_APPROX(dm4.col(0)=m2*refMat3.col(0), refMat4.col(0)=refMat2*refMat3.col(0)); - VERIFY_IS_APPROX(dm4.col(0)=m2*refMat3t.transpose().col(0), refMat4.col(0)=refMat2*refMat3t.transpose().col(0)); - VERIFY_IS_APPROX(dm4.col(0)=m2t.transpose()*refMat3.col(0), refMat4.col(0)=refMat2t.transpose()*refMat3.col(0)); - VERIFY_IS_APPROX(dm4.col(0)=m2t.transpose()*refMat3t.transpose().col(0), refMat4.col(0)=refMat2t.transpose()*refMat3t.transpose().col(0)); + VERIFY_IS_APPROX(dm4.col(0) = m2 * refMat3.col(0), refMat4.col(0) = refMat2 * refMat3.col(0)); + VERIFY_IS_APPROX( + dm4.col(0) = m2 * refMat3t.transpose().col(0), refMat4.col(0) = refMat2 * refMat3t.transpose().col(0)); + VERIFY_IS_APPROX( + dm4.col(0) = m2t.transpose() * refMat3.col(0), refMat4.col(0) = refMat2t.transpose() * refMat3.col(0)); + VERIFY_IS_APPROX(dm4.col(0) = m2t.transpose() * refMat3t.transpose().col(0), + refMat4.col(0) = refMat2t.transpose() * refMat3t.transpose().col(0)); // dense * sparse - VERIFY_IS_APPROX(dm4=refMat2*m3, refMat4=refMat2*refMat3); - VERIFY_IS_APPROX(dm4=dm4+refMat2*m3, refMat4=refMat4+refMat2*refMat3); - VERIFY_IS_APPROX(dm4+=refMat2*m3, refMat4+=refMat2*refMat3); - VERIFY_IS_APPROX(dm4-=refMat2*m3, refMat4-=refMat2*refMat3); - VERIFY_IS_APPROX(dm4.noalias()+=refMat2*m3, refMat4+=refMat2*refMat3); - VERIFY_IS_APPROX(dm4.noalias()-=refMat2*m3, refMat4-=refMat2*refMat3); - VERIFY_IS_APPROX(dm4=refMat2*m3t.transpose(), refMat4=refMat2*refMat3t.transpose()); - VERIFY_IS_APPROX(dm4=refMat2t.transpose()*m3, refMat4=refMat2t.transpose()*refMat3); - VERIFY_IS_APPROX(dm4=refMat2t.transpose()*m3t.transpose(), refMat4=refMat2t.transpose()*refMat3t.transpose()); + VERIFY_IS_APPROX(dm4 = refMat2 * m3, refMat4 = refMat2 * refMat3); + VERIFY_IS_APPROX(dm4 = dm4 + refMat2 * m3, refMat4 = refMat4 + refMat2 * refMat3); + VERIFY_IS_APPROX(dm4 += refMat2 * m3, refMat4 += refMat2 * refMat3); + VERIFY_IS_APPROX(dm4 -= refMat2 * m3, refMat4 -= refMat2 * refMat3); + VERIFY_IS_APPROX(dm4.noalias() += refMat2 * m3, refMat4 += refMat2 * refMat3); + VERIFY_IS_APPROX(dm4.noalias() -= refMat2 * m3, refMat4 -= refMat2 * refMat3); + VERIFY_IS_APPROX(dm4 = refMat2 * m3t.transpose(), refMat4 = refMat2 * refMat3t.transpose()); + VERIFY_IS_APPROX(dm4 = refMat2t.transpose() * m3, refMat4 = refMat2t.transpose() * refMat3); + VERIFY_IS_APPROX( + dm4 = refMat2t.transpose() * m3t.transpose(), refMat4 = refMat2t.transpose() * refMat3t.transpose()); // sparse * dense and dense * sparse outer product { - Index c = internal::random(0,depth-1); - Index r = internal::random(0,rows-1); - Index c1 = internal::random(0,cols-1); - Index r1 = internal::random(0,depth-1); - DenseMatrix dm5 = DenseMatrix::Random(depth, cols); - - VERIFY_IS_APPROX( m4=m2.col(c)*dm5.col(c1).transpose(), refMat4=refMat2.col(c)*dm5.col(c1).transpose()); - VERIFY_IS_EQUAL(m4.nonZeros(), (refMat4.array()!=0).count()); - VERIFY_IS_APPROX( m4=m2.middleCols(c,1)*dm5.col(c1).transpose(), refMat4=refMat2.col(c)*dm5.col(c1).transpose()); - VERIFY_IS_EQUAL(m4.nonZeros(), (refMat4.array()!=0).count()); - VERIFY_IS_APPROX(dm4=m2.col(c)*dm5.col(c1).transpose(), refMat4=refMat2.col(c)*dm5.col(c1).transpose()); - - VERIFY_IS_APPROX(m4=dm5.col(c1)*m2.col(c).transpose(), refMat4=dm5.col(c1)*refMat2.col(c).transpose()); - VERIFY_IS_EQUAL(m4.nonZeros(), (refMat4.array()!=0).count()); - VERIFY_IS_APPROX(m4=dm5.col(c1)*m2.middleCols(c,1).transpose(), refMat4=dm5.col(c1)*refMat2.col(c).transpose()); - VERIFY_IS_EQUAL(m4.nonZeros(), (refMat4.array()!=0).count()); - VERIFY_IS_APPROX(dm4=dm5.col(c1)*m2.col(c).transpose(), refMat4=dm5.col(c1)*refMat2.col(c).transpose()); - - VERIFY_IS_APPROX( m4=dm5.row(r1).transpose()*m2.col(c).transpose(), refMat4=dm5.row(r1).transpose()*refMat2.col(c).transpose()); - VERIFY_IS_EQUAL(m4.nonZeros(), (refMat4.array()!=0).count()); - VERIFY_IS_APPROX(dm4=dm5.row(r1).transpose()*m2.col(c).transpose(), refMat4=dm5.row(r1).transpose()*refMat2.col(c).transpose()); - - VERIFY_IS_APPROX( m4=m2.row(r).transpose()*dm5.col(c1).transpose(), refMat4=refMat2.row(r).transpose()*dm5.col(c1).transpose()); - VERIFY_IS_EQUAL(m4.nonZeros(), (refMat4.array()!=0).count()); - VERIFY_IS_APPROX( m4=m2.middleRows(r,1).transpose()*dm5.col(c1).transpose(), refMat4=refMat2.row(r).transpose()*dm5.col(c1).transpose()); - VERIFY_IS_EQUAL(m4.nonZeros(), (refMat4.array()!=0).count()); - VERIFY_IS_APPROX(dm4=m2.row(r).transpose()*dm5.col(c1).transpose(), refMat4=refMat2.row(r).transpose()*dm5.col(c1).transpose()); - - VERIFY_IS_APPROX( m4=dm5.col(c1)*m2.row(r), refMat4=dm5.col(c1)*refMat2.row(r)); - VERIFY_IS_EQUAL(m4.nonZeros(), (refMat4.array()!=0).count()); - VERIFY_IS_APPROX( m4=dm5.col(c1)*m2.middleRows(r,1), refMat4=dm5.col(c1)*refMat2.row(r)); - VERIFY_IS_EQUAL(m4.nonZeros(), (refMat4.array()!=0).count()); - VERIFY_IS_APPROX(dm4=dm5.col(c1)*m2.row(r), refMat4=dm5.col(c1)*refMat2.row(r)); - - VERIFY_IS_APPROX( m4=dm5.row(r1).transpose()*m2.row(r), refMat4=dm5.row(r1).transpose()*refMat2.row(r)); - VERIFY_IS_EQUAL(m4.nonZeros(), (refMat4.array()!=0).count()); - VERIFY_IS_APPROX(dm4=dm5.row(r1).transpose()*m2.row(r), refMat4=dm5.row(r1).transpose()*refMat2.row(r)); + Index c = internal::random(0, depth - 1); + Index r = internal::random(0, rows - 1); + Index c1 = internal::random(0, cols - 1); + Index r1 = internal::random(0, depth - 1); + DenseMatrix dm5 = DenseMatrix::Random(depth, cols); + + VERIFY_IS_APPROX(m4 = m2.col(c) * dm5.col(c1).transpose(), refMat4 = refMat2.col(c) * dm5.col(c1).transpose()); + VERIFY_IS_EQUAL(m4.nonZeros(), (refMat4.array() != 0).count()); + VERIFY_IS_APPROX( + m4 = m2.middleCols(c, 1) * dm5.col(c1).transpose(), refMat4 = refMat2.col(c) * dm5.col(c1).transpose()); + VERIFY_IS_EQUAL(m4.nonZeros(), (refMat4.array() != 0).count()); + VERIFY_IS_APPROX(dm4 = m2.col(c) * dm5.col(c1).transpose(), refMat4 = refMat2.col(c) * dm5.col(c1).transpose()); + + VERIFY_IS_APPROX(m4 = dm5.col(c1) * m2.col(c).transpose(), refMat4 = dm5.col(c1) * refMat2.col(c).transpose()); + VERIFY_IS_EQUAL(m4.nonZeros(), (refMat4.array() != 0).count()); + VERIFY_IS_APPROX( + m4 = dm5.col(c1) * m2.middleCols(c, 1).transpose(), refMat4 = dm5.col(c1) * refMat2.col(c).transpose()); + VERIFY_IS_EQUAL(m4.nonZeros(), (refMat4.array() != 0).count()); + VERIFY_IS_APPROX(dm4 = dm5.col(c1) * m2.col(c).transpose(), refMat4 = dm5.col(c1) * refMat2.col(c).transpose()); + + VERIFY_IS_APPROX(m4 = dm5.row(r1).transpose() * m2.col(c).transpose(), + refMat4 = dm5.row(r1).transpose() * refMat2.col(c).transpose()); + VERIFY_IS_EQUAL(m4.nonZeros(), (refMat4.array() != 0).count()); + VERIFY_IS_APPROX(dm4 = dm5.row(r1).transpose() * m2.col(c).transpose(), + refMat4 = dm5.row(r1).transpose() * refMat2.col(c).transpose()); + + VERIFY_IS_APPROX(m4 = m2.row(r).transpose() * dm5.col(c1).transpose(), + refMat4 = refMat2.row(r).transpose() * dm5.col(c1).transpose()); + VERIFY_IS_EQUAL(m4.nonZeros(), (refMat4.array() != 0).count()); + VERIFY_IS_APPROX(m4 = m2.middleRows(r, 1).transpose() * dm5.col(c1).transpose(), + refMat4 = refMat2.row(r).transpose() * dm5.col(c1).transpose()); + VERIFY_IS_EQUAL(m4.nonZeros(), (refMat4.array() != 0).count()); + VERIFY_IS_APPROX(dm4 = m2.row(r).transpose() * dm5.col(c1).transpose(), + refMat4 = refMat2.row(r).transpose() * dm5.col(c1).transpose()); + + VERIFY_IS_APPROX(m4 = dm5.col(c1) * m2.row(r), refMat4 = dm5.col(c1) * refMat2.row(r)); + VERIFY_IS_EQUAL(m4.nonZeros(), (refMat4.array() != 0).count()); + VERIFY_IS_APPROX(m4 = dm5.col(c1) * m2.middleRows(r, 1), refMat4 = dm5.col(c1) * refMat2.row(r)); + VERIFY_IS_EQUAL(m4.nonZeros(), (refMat4.array() != 0).count()); + VERIFY_IS_APPROX(dm4 = dm5.col(c1) * m2.row(r), refMat4 = dm5.col(c1) * refMat2.row(r)); + + VERIFY_IS_APPROX(m4 = dm5.row(r1).transpose() * m2.row(r), refMat4 = dm5.row(r1).transpose() * refMat2.row(r)); + VERIFY_IS_EQUAL(m4.nonZeros(), (refMat4.array() != 0).count()); + VERIFY_IS_APPROX(dm4 = dm5.row(r1).transpose() * m2.row(r), refMat4 = dm5.row(r1).transpose() * refMat2.row(r)); } - VERIFY_IS_APPROX(m6=m6*m6, refMat6=refMat6*refMat6); - + VERIFY_IS_APPROX(m6 = m6 * m6, refMat6 = refMat6 * refMat6); + // sparse matrix * sparse vector ColSpVector cv0(cols), cv1; DenseVector dcv0(cols), dcv1; - initSparse(2*density,dcv0, cv0); - + initSparse(2 * density, dcv0, cv0); + RowSpVector rv0(depth), rv1; RowDenseVector drv0(depth), drv1(rv1); - initSparse(2*density,drv0, rv0); + initSparse(2 * density, drv0, rv0); - VERIFY_IS_APPROX(cv1=m3*cv0, dcv1=refMat3*dcv0); - VERIFY_IS_APPROX(rv1=rv0*m3, drv1=drv0*refMat3); - VERIFY_IS_APPROX(cv1=m3t.adjoint()*cv0, dcv1=refMat3t.adjoint()*dcv0); - VERIFY_IS_APPROX(cv1=rv0*m3, dcv1=drv0*refMat3); - VERIFY_IS_APPROX(rv1=m3*cv0, drv1=refMat3*dcv0); + VERIFY_IS_APPROX(cv1 = m3 * cv0, dcv1 = refMat3 * dcv0); + VERIFY_IS_APPROX(rv1 = rv0 * m3, drv1 = drv0 * refMat3); + VERIFY_IS_APPROX(cv1 = m3t.adjoint() * cv0, dcv1 = refMat3t.adjoint() * dcv0); + VERIFY_IS_APPROX(cv1 = rv0 * m3, dcv1 = drv0 * refMat3); + VERIFY_IS_APPROX(rv1 = m3 * cv0, drv1 = refMat3 * dcv0); } - + // test matrix - diagonal product { DenseMatrix refM2 = DenseMatrix::Zero(rows, cols); DenseMatrix refM3 = DenseMatrix::Zero(rows, cols); DenseMatrix d3 = DenseMatrix::Zero(rows, cols); - DiagonalMatrix d1(DenseVector::Random(cols)); - DiagonalMatrix d2(DenseVector::Random(rows)); + DiagonalMatrix d1(DenseVector::Random(cols)); + DiagonalMatrix d2(DenseVector::Random(rows)); SparseMatrixType m2(rows, cols); SparseMatrixType m3(rows, cols); initSparse(density, refM2, m2); initSparse(density, refM3, m3); - VERIFY_IS_APPROX(m3=m2*d1, refM3=refM2*d1); - VERIFY_IS_APPROX(m3=m2.transpose()*d2, refM3=refM2.transpose()*d2); - VERIFY_IS_APPROX(m3=d2*m2, refM3=d2*refM2); - VERIFY_IS_APPROX(m3=d1*m2.transpose(), refM3=d1*refM2.transpose()); - + VERIFY_IS_APPROX(m3 = m2 * d1, refM3 = refM2 * d1); + VERIFY_IS_APPROX(m3 = m2.transpose() * d2, refM3 = refM2.transpose() * d2); + VERIFY_IS_APPROX(m3 = d2 * m2, refM3 = d2 * refM2); + VERIFY_IS_APPROX(m3 = d1 * m2.transpose(), refM3 = d1 * refM2.transpose()); + // also check with a SparseWrapper: DenseVector v1 = DenseVector::Random(cols); DenseVector v2 = DenseVector::Random(rows); DenseVector v3 = DenseVector::Random(rows); - VERIFY_IS_APPROX(m3=m2*v1.asDiagonal(), refM3=refM2*v1.asDiagonal()); - VERIFY_IS_APPROX(m3=m2.transpose()*v2.asDiagonal(), refM3=refM2.transpose()*v2.asDiagonal()); - VERIFY_IS_APPROX(m3=v2.asDiagonal()*m2, refM3=v2.asDiagonal()*refM2); - VERIFY_IS_APPROX(m3=v1.asDiagonal()*m2.transpose(), refM3=v1.asDiagonal()*refM2.transpose()); - - VERIFY_IS_APPROX(m3=v2.asDiagonal()*m2*v1.asDiagonal(), refM3=v2.asDiagonal()*refM2*v1.asDiagonal()); - - VERIFY_IS_APPROX(v2=m2*v1.asDiagonal()*v1, refM2*v1.asDiagonal()*v1); - VERIFY_IS_APPROX(v3=v2.asDiagonal()*m2*v1, v2.asDiagonal()*refM2*v1); - + VERIFY_IS_APPROX(m3 = m2 * v1.asDiagonal(), refM3 = refM2 * v1.asDiagonal()); + VERIFY_IS_APPROX(m3 = m2.transpose() * v2.asDiagonal(), refM3 = refM2.transpose() * v2.asDiagonal()); + VERIFY_IS_APPROX(m3 = v2.asDiagonal() * m2, refM3 = v2.asDiagonal() * refM2); + VERIFY_IS_APPROX(m3 = v1.asDiagonal() * m2.transpose(), refM3 = v1.asDiagonal() * refM2.transpose()); + + VERIFY_IS_APPROX(m3 = v2.asDiagonal() * m2 * v1.asDiagonal(), refM3 = v2.asDiagonal() * refM2 * v1.asDiagonal()); + + VERIFY_IS_APPROX(v2 = m2 * v1.asDiagonal() * v1, refM2 * v1.asDiagonal() * v1); + VERIFY_IS_APPROX(v3 = v2.asDiagonal() * m2 * v1, v2.asDiagonal() * refM2 * v1); + // evaluate to a dense matrix to check the .row() and .col() iterator functions - VERIFY_IS_APPROX(d3=m2*d1, refM3=refM2*d1); - VERIFY_IS_APPROX(d3=m2.transpose()*d2, refM3=refM2.transpose()*d2); - VERIFY_IS_APPROX(d3=d2*m2, refM3=d2*refM2); - VERIFY_IS_APPROX(d3=d1*m2.transpose(), refM3=d1*refM2.transpose()); + VERIFY_IS_APPROX(d3 = m2 * d1, refM3 = refM2 * d1); + VERIFY_IS_APPROX(d3 = m2.transpose() * d2, refM3 = refM2.transpose() * d2); + VERIFY_IS_APPROX(d3 = d2 * m2, refM3 = d2 * refM2); + VERIFY_IS_APPROX(d3 = d1 * m2.transpose(), refM3 = d1 * refM2.transpose()); } // test self-adjoint and triangular-view products @@ -280,7 +299,7 @@ template void sparse_product() SparseMatrixType mA(rows, rows); initSparse(density, refA, mA); do { - initSparse(density, refUp, mUp, ForceRealDiag|/*ForceNonZeroDiag|*/MakeUpperTriangular); + initSparse(density, refUp, mUp, ForceRealDiag | /*ForceNonZeroDiag|*/ MakeUpperTriangular); } while (refUp.isZero()); refLo = refUp.adjoint(); mLo = mUp.adjoint(); @@ -288,47 +307,50 @@ template void sparse_product() refS.diagonal() *= 0.5; mS = mUp + mLo; // TODO be able to address the diagonal.... - for (int k=0; k()*b, refX=refS*b); - VERIFY_IS_APPROX(x=mLo.template selfadjointView()*b, refX=refS*b); - VERIFY_IS_APPROX(x=mS.template selfadjointView()*b, refX=refS*b); - - VERIFY_IS_APPROX(x=b * mUp.template selfadjointView(), refX=b*refS); - VERIFY_IS_APPROX(x=b * mLo.template selfadjointView(), refX=b*refS); - VERIFY_IS_APPROX(x=b * mS.template selfadjointView(), refX=b*refS); - - VERIFY_IS_APPROX(x.noalias()+=mUp.template selfadjointView()*b, refX+=refS*b); - VERIFY_IS_APPROX(x.noalias()-=mLo.template selfadjointView()*b, refX-=refS*b); - VERIFY_IS_APPROX(x.noalias()+=mS.template selfadjointView()*b, refX+=refS*b); - + VERIFY_IS_APPROX(x = mUp.template selfadjointView() * b, refX = refS * b); + VERIFY_IS_APPROX(x = mLo.template selfadjointView() * b, refX = refS * b); + VERIFY_IS_APPROX(x = mS.template selfadjointView() * b, refX = refS * b); + + VERIFY_IS_APPROX(x = b * mUp.template selfadjointView(), refX = b * refS); + VERIFY_IS_APPROX(x = b * mLo.template selfadjointView(), refX = b * refS); + VERIFY_IS_APPROX(x = b * mS.template selfadjointView(), refX = b * refS); + + VERIFY_IS_APPROX(x.noalias() += mUp.template selfadjointView() * b, refX += refS * b); + VERIFY_IS_APPROX(x.noalias() -= mLo.template selfadjointView() * b, refX -= refS * b); + VERIFY_IS_APPROX(x.noalias() += mS.template selfadjointView() * b, refX += refS * b); + // sparse selfadjointView with sparse matrices - SparseMatrixType mSres(rows,rows); - VERIFY_IS_APPROX(mSres = mLo.template selfadjointView()*mS, - refX = refLo.template selfadjointView()*refS); - VERIFY_IS_APPROX(mSres = mS * mLo.template selfadjointView(), - refX = refS * refLo.template selfadjointView()); - + SparseMatrixType mSres(rows, rows); + VERIFY_IS_APPROX( + mSres = mLo.template selfadjointView() * mS, refX = refLo.template selfadjointView() * refS); + VERIFY_IS_APPROX( + mSres = mS * mLo.template selfadjointView(), refX = refS * refLo.template selfadjointView()); + // sparse triangularView with dense matrices - VERIFY_IS_APPROX(x=mA.template triangularView()*b, refX=refA.template triangularView()*b); - VERIFY_IS_APPROX(x=mA.template triangularView()*b, refX=refA.template triangularView()*b); - VERIFY_IS_APPROX(x=b*mA.template triangularView(), refX=b*refA.template triangularView()); - VERIFY_IS_APPROX(x=b*mA.template triangularView(), refX=b*refA.template triangularView()); - + VERIFY_IS_APPROX(x = mA.template triangularView() * b, refX = refA.template triangularView() * b); + VERIFY_IS_APPROX(x = mA.template triangularView() * b, refX = refA.template triangularView() * b); + VERIFY_IS_APPROX(x = b * mA.template triangularView(), refX = b * refA.template triangularView()); + VERIFY_IS_APPROX(x = b * mA.template triangularView(), refX = b * refA.template triangularView()); + // sparse triangularView with sparse matrices - VERIFY_IS_APPROX(mSres = mA.template triangularView()*mS, refX = refA.template triangularView()*refS); - VERIFY_IS_APPROX(mSres = mS * mA.template triangularView(), refX = refS * refA.template triangularView()); - VERIFY_IS_APPROX(mSres = mA.template triangularView()*mS, refX = refA.template triangularView()*refS); - VERIFY_IS_APPROX(mSres = mS * mA.template triangularView(), refX = refS * refA.template triangularView()); + VERIFY_IS_APPROX( + mSres = mA.template triangularView() * mS, refX = refA.template triangularView() * refS); + VERIFY_IS_APPROX( + mSres = mS * mA.template triangularView(), refX = refS * refA.template triangularView()); + VERIFY_IS_APPROX( + mSres = mA.template triangularView() * mS, refX = refA.template triangularView() * refS); + VERIFY_IS_APPROX( + mSres = mS * mA.template triangularView(), refX = refS * refA.template triangularView()); } } @@ -336,140 +358,149 @@ template void sparse_product() template void sparse_product_regression_test() { // This code does not compile with afflicted versions of the bug - SparseMatrixType sm1(3,2); - DenseMatrixType m2(2,2); + SparseMatrixType sm1(3, 2); + DenseMatrixType m2(2, 2); sm1.setZero(); m2.setZero(); - DenseMatrixType m3 = sm1*m2; + DenseMatrixType m3 = sm1 * m2; // This code produces a segfault with afflicted versions of another SparseTimeDenseProduct // bug - SparseMatrixType sm2(20000,2); + SparseMatrixType sm2(20000, 2); sm2.setZero(); - DenseMatrixType m4(sm2*m2); + DenseMatrixType m4(sm2 * m2); - VERIFY_IS_APPROX( m4(0,0), 0.0 ); + VERIFY_IS_APPROX(m4(0, 0), 0.0); } -template -void bug_942() +template void bug_942() { - typedef Matrix Vector; + typedef Matrix Vector; typedef SparseMatrix ColSpMat; typedef SparseMatrix RowSpMat; - ColSpMat cmA(1,1); - cmA.insert(0,0) = 1; + ColSpMat cmA(1, 1); + cmA.insert(0, 0) = 1; - RowSpMat rmA(1,1); - rmA.insert(0,0) = 1; + RowSpMat rmA(1, 1); + rmA.insert(0, 0) = 1; Vector d(1); d[0] = 2; - + double res = 2; - - VERIFY_IS_APPROX( ( cmA*d.asDiagonal() ).eval().coeff(0,0), res ); - VERIFY_IS_APPROX( ( d.asDiagonal()*rmA ).eval().coeff(0,0), res ); - VERIFY_IS_APPROX( ( rmA*d.asDiagonal() ).eval().coeff(0,0), res ); - VERIFY_IS_APPROX( ( d.asDiagonal()*cmA ).eval().coeff(0,0), res ); + + VERIFY_IS_APPROX((cmA * d.asDiagonal()).eval().coeff(0, 0), res); + VERIFY_IS_APPROX((d.asDiagonal() * rmA).eval().coeff(0, 0), res); + VERIFY_IS_APPROX((rmA * d.asDiagonal()).eval().coeff(0, 0), res); + VERIFY_IS_APPROX((d.asDiagonal() * cmA).eval().coeff(0, 0), res); } -template -void test_mixing_types() +template void test_mixing_types() { typedef std::complex Cplx; typedef SparseMatrix SpMatReal; typedef SparseMatrix SpMatCplx; - typedef SparseMatrix SpRowMatCplx; - typedef Matrix DenseMatReal; - typedef Matrix DenseMatCplx; + typedef SparseMatrix SpRowMatCplx; + typedef Matrix DenseMatReal; + typedef Matrix DenseMatCplx; - Index n = internal::random(1,100); - double density = (std::max)(8./(n*n), 0.2); + Index n = internal::random(1, 100); + double density = (std::max)(8. / (n * n), 0.2); - SpMatReal sR1(n,n); - SpMatCplx sC1(n,n), sC2(n,n), sC3(n,n); - SpRowMatCplx sCR(n,n); - DenseMatReal dR1(n,n); - DenseMatCplx dC1(n,n), dC2(n,n), dC3(n,n); + SpMatReal sR1(n, n); + SpMatCplx sC1(n, n), sC2(n, n), sC3(n, n); + SpRowMatCplx sCR(n, n); + DenseMatReal dR1(n, n); + DenseMatCplx dC1(n, n), dC2(n, n), dC3(n, n); initSparse(density, dR1, sR1); initSparse(density, dC1, sC1); initSparse(density, dC2, sC2); - VERIFY_IS_APPROX( sC2 = (sR1 * sC1), dC3 = dR1.template cast() * dC1 ); - VERIFY_IS_APPROX( sC2 = (sC1 * sR1), dC3 = dC1 * dR1.template cast() ); - VERIFY_IS_APPROX( sC2 = (sR1.transpose() * sC1), dC3 = dR1.template cast().transpose() * dC1 ); - VERIFY_IS_APPROX( sC2 = (sC1.transpose() * sR1), dC3 = dC1.transpose() * dR1.template cast() ); - VERIFY_IS_APPROX( sC2 = (sR1 * sC1.transpose()), dC3 = dR1.template cast() * dC1.transpose() ); - VERIFY_IS_APPROX( sC2 = (sC1 * sR1.transpose()), dC3 = dC1 * dR1.template cast().transpose() ); - VERIFY_IS_APPROX( sC2 = (sR1.transpose() * sC1.transpose()), dC3 = dR1.template cast().transpose() * dC1.transpose() ); - VERIFY_IS_APPROX( sC2 = (sC1.transpose() * sR1.transpose()), dC3 = dC1.transpose() * dR1.template cast().transpose() ); - - VERIFY_IS_APPROX( sCR = (sR1 * sC1), dC3 = dR1.template cast() * dC1 ); - VERIFY_IS_APPROX( sCR = (sC1 * sR1), dC3 = dC1 * dR1.template cast() ); - VERIFY_IS_APPROX( sCR = (sR1.transpose() * sC1), dC3 = dR1.template cast().transpose() * dC1 ); - VERIFY_IS_APPROX( sCR = (sC1.transpose() * sR1), dC3 = dC1.transpose() * dR1.template cast() ); - VERIFY_IS_APPROX( sCR = (sR1 * sC1.transpose()), dC3 = dR1.template cast() * dC1.transpose() ); - VERIFY_IS_APPROX( sCR = (sC1 * sR1.transpose()), dC3 = dC1 * dR1.template cast().transpose() ); - VERIFY_IS_APPROX( sCR = (sR1.transpose() * sC1.transpose()), dC3 = dR1.template cast().transpose() * dC1.transpose() ); - VERIFY_IS_APPROX( sCR = (sC1.transpose() * sR1.transpose()), dC3 = dC1.transpose() * dR1.template cast().transpose() ); - - - VERIFY_IS_APPROX( sC2 = (sR1 * sC1).pruned(), dC3 = dR1.template cast() * dC1 ); - VERIFY_IS_APPROX( sC2 = (sC1 * sR1).pruned(), dC3 = dC1 * dR1.template cast() ); - VERIFY_IS_APPROX( sC2 = (sR1.transpose() * sC1).pruned(), dC3 = dR1.template cast().transpose() * dC1 ); - VERIFY_IS_APPROX( sC2 = (sC1.transpose() * sR1).pruned(), dC3 = dC1.transpose() * dR1.template cast() ); - VERIFY_IS_APPROX( sC2 = (sR1 * sC1.transpose()).pruned(), dC3 = dR1.template cast() * dC1.transpose() ); - VERIFY_IS_APPROX( sC2 = (sC1 * sR1.transpose()).pruned(), dC3 = dC1 * dR1.template cast().transpose() ); - VERIFY_IS_APPROX( sC2 = (sR1.transpose() * sC1.transpose()).pruned(), dC3 = dR1.template cast().transpose() * dC1.transpose() ); - VERIFY_IS_APPROX( sC2 = (sC1.transpose() * sR1.transpose()).pruned(), dC3 = dC1.transpose() * dR1.template cast().transpose() ); - - VERIFY_IS_APPROX( sCR = (sR1 * sC1).pruned(), dC3 = dR1.template cast() * dC1 ); - VERIFY_IS_APPROX( sCR = (sC1 * sR1).pruned(), dC3 = dC1 * dR1.template cast() ); - VERIFY_IS_APPROX( sCR = (sR1.transpose() * sC1).pruned(), dC3 = dR1.template cast().transpose() * dC1 ); - VERIFY_IS_APPROX( sCR = (sC1.transpose() * sR1).pruned(), dC3 = dC1.transpose() * dR1.template cast() ); - VERIFY_IS_APPROX( sCR = (sR1 * sC1.transpose()).pruned(), dC3 = dR1.template cast() * dC1.transpose() ); - VERIFY_IS_APPROX( sCR = (sC1 * sR1.transpose()).pruned(), dC3 = dC1 * dR1.template cast().transpose() ); - VERIFY_IS_APPROX( sCR = (sR1.transpose() * sC1.transpose()).pruned(), dC3 = dR1.template cast().transpose() * dC1.transpose() ); - VERIFY_IS_APPROX( sCR = (sC1.transpose() * sR1.transpose()).pruned(), dC3 = dC1.transpose() * dR1.template cast().transpose() ); - - - VERIFY_IS_APPROX( dC2 = (sR1 * sC1), dC3 = dR1.template cast() * dC1 ); - VERIFY_IS_APPROX( dC2 = (sC1 * sR1), dC3 = dC1 * dR1.template cast() ); - VERIFY_IS_APPROX( dC2 = (sR1.transpose() * sC1), dC3 = dR1.template cast().transpose() * dC1 ); - VERIFY_IS_APPROX( dC2 = (sC1.transpose() * sR1), dC3 = dC1.transpose() * dR1.template cast() ); - VERIFY_IS_APPROX( dC2 = (sR1 * sC1.transpose()), dC3 = dR1.template cast() * dC1.transpose() ); - VERIFY_IS_APPROX( dC2 = (sC1 * sR1.transpose()), dC3 = dC1 * dR1.template cast().transpose() ); - VERIFY_IS_APPROX( dC2 = (sR1.transpose() * sC1.transpose()), dC3 = dR1.template cast().transpose() * dC1.transpose() ); - VERIFY_IS_APPROX( dC2 = (sC1.transpose() * sR1.transpose()), dC3 = dC1.transpose() * dR1.template cast().transpose() ); - - - VERIFY_IS_APPROX( dC2 = dR1 * sC1, dC3 = dR1.template cast() * sC1 ); - VERIFY_IS_APPROX( dC2 = sR1 * dC1, dC3 = sR1.template cast() * dC1 ); - VERIFY_IS_APPROX( dC2 = dC1 * sR1, dC3 = dC1 * sR1.template cast() ); - VERIFY_IS_APPROX( dC2 = sC1 * dR1, dC3 = sC1 * dR1.template cast() ); - - VERIFY_IS_APPROX( dC2 = dR1.row(0) * sC1, dC3 = dR1.template cast().row(0) * sC1 ); - VERIFY_IS_APPROX( dC2 = sR1 * dC1.col(0), dC3 = sR1.template cast() * dC1.col(0) ); - VERIFY_IS_APPROX( dC2 = dC1.row(0) * sR1, dC3 = dC1.row(0) * sR1.template cast() ); - VERIFY_IS_APPROX( dC2 = sC1 * dR1.col(0), dC3 = sC1 * dR1.template cast().col(0) ); + VERIFY_IS_APPROX(sC2 = (sR1 * sC1), dC3 = dR1.template cast() * dC1); + VERIFY_IS_APPROX(sC2 = (sC1 * sR1), dC3 = dC1 * dR1.template cast()); + VERIFY_IS_APPROX(sC2 = (sR1.transpose() * sC1), dC3 = dR1.template cast().transpose() * dC1); + VERIFY_IS_APPROX(sC2 = (sC1.transpose() * sR1), dC3 = dC1.transpose() * dR1.template cast()); + VERIFY_IS_APPROX(sC2 = (sR1 * sC1.transpose()), dC3 = dR1.template cast() * dC1.transpose()); + VERIFY_IS_APPROX(sC2 = (sC1 * sR1.transpose()), dC3 = dC1 * dR1.template cast().transpose()); + VERIFY_IS_APPROX( + sC2 = (sR1.transpose() * sC1.transpose()), dC3 = dR1.template cast().transpose() * dC1.transpose()); + VERIFY_IS_APPROX( + sC2 = (sC1.transpose() * sR1.transpose()), dC3 = dC1.transpose() * dR1.template cast().transpose()); + + VERIFY_IS_APPROX(sCR = (sR1 * sC1), dC3 = dR1.template cast() * dC1); + VERIFY_IS_APPROX(sCR = (sC1 * sR1), dC3 = dC1 * dR1.template cast()); + VERIFY_IS_APPROX(sCR = (sR1.transpose() * sC1), dC3 = dR1.template cast().transpose() * dC1); + VERIFY_IS_APPROX(sCR = (sC1.transpose() * sR1), dC3 = dC1.transpose() * dR1.template cast()); + VERIFY_IS_APPROX(sCR = (sR1 * sC1.transpose()), dC3 = dR1.template cast() * dC1.transpose()); + VERIFY_IS_APPROX(sCR = (sC1 * sR1.transpose()), dC3 = dC1 * dR1.template cast().transpose()); + VERIFY_IS_APPROX( + sCR = (sR1.transpose() * sC1.transpose()), dC3 = dR1.template cast().transpose() * dC1.transpose()); + VERIFY_IS_APPROX( + sCR = (sC1.transpose() * sR1.transpose()), dC3 = dC1.transpose() * dR1.template cast().transpose()); + + + VERIFY_IS_APPROX(sC2 = (sR1 * sC1).pruned(), dC3 = dR1.template cast() * dC1); + VERIFY_IS_APPROX(sC2 = (sC1 * sR1).pruned(), dC3 = dC1 * dR1.template cast()); + VERIFY_IS_APPROX(sC2 = (sR1.transpose() * sC1).pruned(), dC3 = dR1.template cast().transpose() * dC1); + VERIFY_IS_APPROX(sC2 = (sC1.transpose() * sR1).pruned(), dC3 = dC1.transpose() * dR1.template cast()); + VERIFY_IS_APPROX(sC2 = (sR1 * sC1.transpose()).pruned(), dC3 = dR1.template cast() * dC1.transpose()); + VERIFY_IS_APPROX(sC2 = (sC1 * sR1.transpose()).pruned(), dC3 = dC1 * dR1.template cast().transpose()); + VERIFY_IS_APPROX( + sC2 = (sR1.transpose() * sC1.transpose()).pruned(), dC3 = dR1.template cast().transpose() * dC1.transpose()); + VERIFY_IS_APPROX( + sC2 = (sC1.transpose() * sR1.transpose()).pruned(), dC3 = dC1.transpose() * dR1.template cast().transpose()); + + VERIFY_IS_APPROX(sCR = (sR1 * sC1).pruned(), dC3 = dR1.template cast() * dC1); + VERIFY_IS_APPROX(sCR = (sC1 * sR1).pruned(), dC3 = dC1 * dR1.template cast()); + VERIFY_IS_APPROX(sCR = (sR1.transpose() * sC1).pruned(), dC3 = dR1.template cast().transpose() * dC1); + VERIFY_IS_APPROX(sCR = (sC1.transpose() * sR1).pruned(), dC3 = dC1.transpose() * dR1.template cast()); + VERIFY_IS_APPROX(sCR = (sR1 * sC1.transpose()).pruned(), dC3 = dR1.template cast() * dC1.transpose()); + VERIFY_IS_APPROX(sCR = (sC1 * sR1.transpose()).pruned(), dC3 = dC1 * dR1.template cast().transpose()); + VERIFY_IS_APPROX( + sCR = (sR1.transpose() * sC1.transpose()).pruned(), dC3 = dR1.template cast().transpose() * dC1.transpose()); + VERIFY_IS_APPROX( + sCR = (sC1.transpose() * sR1.transpose()).pruned(), dC3 = dC1.transpose() * dR1.template cast().transpose()); + + + VERIFY_IS_APPROX(dC2 = (sR1 * sC1), dC3 = dR1.template cast() * dC1); + VERIFY_IS_APPROX(dC2 = (sC1 * sR1), dC3 = dC1 * dR1.template cast()); + VERIFY_IS_APPROX(dC2 = (sR1.transpose() * sC1), dC3 = dR1.template cast().transpose() * dC1); + VERIFY_IS_APPROX(dC2 = (sC1.transpose() * sR1), dC3 = dC1.transpose() * dR1.template cast()); + VERIFY_IS_APPROX(dC2 = (sR1 * sC1.transpose()), dC3 = dR1.template cast() * dC1.transpose()); + VERIFY_IS_APPROX(dC2 = (sC1 * sR1.transpose()), dC3 = dC1 * dR1.template cast().transpose()); + VERIFY_IS_APPROX( + dC2 = (sR1.transpose() * sC1.transpose()), dC3 = dR1.template cast().transpose() * dC1.transpose()); + VERIFY_IS_APPROX( + dC2 = (sC1.transpose() * sR1.transpose()), dC3 = dC1.transpose() * dR1.template cast().transpose()); + + + VERIFY_IS_APPROX(dC2 = dR1 * sC1, dC3 = dR1.template cast() * sC1); + VERIFY_IS_APPROX(dC2 = sR1 * dC1, dC3 = sR1.template cast() * dC1); + VERIFY_IS_APPROX(dC2 = dC1 * sR1, dC3 = dC1 * sR1.template cast()); + VERIFY_IS_APPROX(dC2 = sC1 * dR1, dC3 = sC1 * dR1.template cast()); + + VERIFY_IS_APPROX(dC2 = dR1.row(0) * sC1, dC3 = dR1.template cast().row(0) * sC1); + VERIFY_IS_APPROX(dC2 = sR1 * dC1.col(0), dC3 = sR1.template cast() * dC1.col(0)); + VERIFY_IS_APPROX(dC2 = dC1.row(0) * sR1, dC3 = dC1.row(0) * sR1.template cast()); + VERIFY_IS_APPROX(dC2 = sC1 * dR1.col(0), dC3 = sC1 * dR1.template cast().col(0)); } void test_sparse_product() { - for(int i = 0; i < g_repeat; i++) { - CALL_SUBTEST_1( (sparse_product >()) ); - CALL_SUBTEST_1( (sparse_product >()) ); - CALL_SUBTEST_1( (bug_942()) ); - CALL_SUBTEST_2( (sparse_product, ColMajor > >()) ); - CALL_SUBTEST_2( (sparse_product, RowMajor > >()) ); - CALL_SUBTEST_3( (sparse_product >()) ); - CALL_SUBTEST_4( (sparse_product_regression_test, Matrix >()) ); - - CALL_SUBTEST_5( (test_mixing_types()) ); + for (int i = 0; i < g_repeat; i++) { + CALL_SUBTEST_1((sparse_product>())); + CALL_SUBTEST_1((sparse_product>())); + CALL_SUBTEST_1((bug_942())); + CALL_SUBTEST_2((sparse_product, ColMajor>>())); + CALL_SUBTEST_2((sparse_product, RowMajor>>())); + CALL_SUBTEST_3((sparse_product>())); + CALL_SUBTEST_4( + (sparse_product_regression_test, Matrix>())); + + CALL_SUBTEST_5((test_mixing_types())); } } diff --git a/filmulator-gui/core/nlmeans/eigen/test/sparse_ref.cpp b/filmulator-gui/core/nlmeans/eigen/test/sparse_ref.cpp index 5e960723..b1406a91 100644 --- a/filmulator-gui/core/nlmeans/eigen/test/sparse_ref.cpp +++ b/filmulator-gui/core/nlmeans/eigen/test/sparse_ref.cpp @@ -14,126 +14,141 @@ static long int nb_temporaries; -inline void on_temporary_creation() { +inline void on_temporary_creation() +{ // here's a great place to set a breakpoint when debugging failures in this test! nb_temporaries++; } -#define EIGEN_SPARSE_CREATE_TEMPORARY_PLUGIN { on_temporary_creation(); } +#define EIGEN_SPARSE_CREATE_TEMPORARY_PLUGIN \ + { \ + on_temporary_creation(); \ + } #include "main.h" #include -#define VERIFY_EVALUATION_COUNT(XPR,N) {\ - nb_temporaries = 0; \ - CALL_SUBTEST( XPR ); \ - if(nb_temporaries!=N) std::cerr << "nb_temporaries == " << nb_temporaries << "\n"; \ - VERIFY( (#XPR) && nb_temporaries==N ); \ +#define VERIFY_EVALUATION_COUNT(XPR, N) \ + { \ + nb_temporaries = 0; \ + CALL_SUBTEST(XPR); \ + if (nb_temporaries != N) std::cerr << "nb_temporaries == " << nb_temporaries << "\n"; \ + VERIFY((#XPR) && nb_temporaries == N); \ } -template void check_const_correctness(const PlainObjectType&) +template void check_const_correctness(const PlainObjectType &) { // verify that ref-to-const don't have LvalueBit typedef typename internal::add_const::type ConstPlainObjectType; - VERIFY( !(internal::traits >::Flags & LvalueBit) ); - VERIFY( !(internal::traits >::Flags & LvalueBit) ); - VERIFY( !(Ref::Flags & LvalueBit) ); - VERIFY( !(Ref::Flags & LvalueBit) ); + VERIFY(!(internal::traits>::Flags & LvalueBit)); + VERIFY(!(internal::traits>::Flags & LvalueBit)); + VERIFY(!(Ref::Flags & LvalueBit)); + VERIFY(!(Ref::Flags & LvalueBit)); } -template -EIGEN_DONT_INLINE void call_ref_1(Ref > a, const B &b) { VERIFY_IS_EQUAL(a.toDense(),b.toDense()); } +template EIGEN_DONT_INLINE void call_ref_1(Ref> a, const B &b) +{ + VERIFY_IS_EQUAL(a.toDense(), b.toDense()); +} -template -EIGEN_DONT_INLINE void call_ref_2(const Ref >& a, const B &b) { VERIFY_IS_EQUAL(a.toDense(),b.toDense()); } +template EIGEN_DONT_INLINE void call_ref_2(const Ref> &a, const B &b) +{ + VERIFY_IS_EQUAL(a.toDense(), b.toDense()); +} template -EIGEN_DONT_INLINE void call_ref_3(const Ref, StandardCompressedFormat>& a, const B &b) { +EIGEN_DONT_INLINE void call_ref_3(const Ref, StandardCompressedFormat> &a, const B &b) +{ VERIFY(a.isCompressed()); - VERIFY_IS_EQUAL(a.toDense(),b.toDense()); + VERIFY_IS_EQUAL(a.toDense(), b.toDense()); } -template -EIGEN_DONT_INLINE void call_ref_4(Ref > a, const B &b) { VERIFY_IS_EQUAL(a.toDense(),b.toDense()); } +template EIGEN_DONT_INLINE void call_ref_4(Ref> a, const B &b) +{ + VERIFY_IS_EQUAL(a.toDense(), b.toDense()); +} -template -EIGEN_DONT_INLINE void call_ref_5(const Ref >& a, const B &b) { VERIFY_IS_EQUAL(a.toDense(),b.toDense()); } +template EIGEN_DONT_INLINE void call_ref_5(const Ref> &a, const B &b) +{ + VERIFY_IS_EQUAL(a.toDense(), b.toDense()); +} void call_ref() { - SparseMatrix A = MatrixXf::Random(10,10).sparseView(0.5,1); - SparseMatrix B = MatrixXf::Random(10,10).sparseView(0.5,1); - SparseMatrix C = MatrixXf::Random(10,10).sparseView(0.5,1); + SparseMatrix A = MatrixXf::Random(10, 10).sparseView(0.5, 1); + SparseMatrix B = MatrixXf::Random(10, 10).sparseView(0.5, 1); + SparseMatrix C = MatrixXf::Random(10, 10).sparseView(0.5, 1); C.reserve(VectorXi::Constant(C.outerSize(), 2)); - const SparseMatrix& Ac(A); - Block > Ab(A,0,1, 3,3); - const Block > Abc(A,0,1,3,3); - SparseVector vc = VectorXf::Random(10).sparseView(0.5,1); - SparseVector vr = VectorXf::Random(10).sparseView(0.5,1); - SparseMatrix AA = A*A; - - - VERIFY_EVALUATION_COUNT( call_ref_1(A, A), 0); -// VERIFY_EVALUATION_COUNT( call_ref_1(Ac, Ac), 0); // does not compile on purpose - VERIFY_EVALUATION_COUNT( call_ref_2(A, A), 0); - VERIFY_EVALUATION_COUNT( call_ref_3(A, A), 0); - VERIFY_EVALUATION_COUNT( call_ref_2(A.transpose(), A.transpose()), 1); - VERIFY_EVALUATION_COUNT( call_ref_3(A.transpose(), A.transpose()), 1); - VERIFY_EVALUATION_COUNT( call_ref_2(Ac,Ac), 0); - VERIFY_EVALUATION_COUNT( call_ref_3(Ac,Ac), 0); - VERIFY_EVALUATION_COUNT( call_ref_2(A+A,2*Ac), 1); - VERIFY_EVALUATION_COUNT( call_ref_3(A+A,2*Ac), 1); - VERIFY_EVALUATION_COUNT( call_ref_2(B, B), 1); - VERIFY_EVALUATION_COUNT( call_ref_3(B, B), 1); - VERIFY_EVALUATION_COUNT( call_ref_2(B.transpose(), B.transpose()), 0); - VERIFY_EVALUATION_COUNT( call_ref_3(B.transpose(), B.transpose()), 0); - VERIFY_EVALUATION_COUNT( call_ref_2(A*A, AA), 3); - VERIFY_EVALUATION_COUNT( call_ref_3(A*A, AA), 3); - + const SparseMatrix &Ac(A); + Block> Ab(A, 0, 1, 3, 3); + const Block> Abc(A, 0, 1, 3, 3); + SparseVector vc = VectorXf::Random(10).sparseView(0.5, 1); + SparseVector vr = VectorXf::Random(10).sparseView(0.5, 1); + SparseMatrix AA = A * A; + + + VERIFY_EVALUATION_COUNT(call_ref_1(A, A), 0); + // VERIFY_EVALUATION_COUNT( call_ref_1(Ac, Ac), 0); // does not compile on purpose + VERIFY_EVALUATION_COUNT(call_ref_2(A, A), 0); + VERIFY_EVALUATION_COUNT(call_ref_3(A, A), 0); + VERIFY_EVALUATION_COUNT(call_ref_2(A.transpose(), A.transpose()), 1); + VERIFY_EVALUATION_COUNT(call_ref_3(A.transpose(), A.transpose()), 1); + VERIFY_EVALUATION_COUNT(call_ref_2(Ac, Ac), 0); + VERIFY_EVALUATION_COUNT(call_ref_3(Ac, Ac), 0); + VERIFY_EVALUATION_COUNT(call_ref_2(A + A, 2 * Ac), 1); + VERIFY_EVALUATION_COUNT(call_ref_3(A + A, 2 * Ac), 1); + VERIFY_EVALUATION_COUNT(call_ref_2(B, B), 1); + VERIFY_EVALUATION_COUNT(call_ref_3(B, B), 1); + VERIFY_EVALUATION_COUNT(call_ref_2(B.transpose(), B.transpose()), 0); + VERIFY_EVALUATION_COUNT(call_ref_3(B.transpose(), B.transpose()), 0); + VERIFY_EVALUATION_COUNT(call_ref_2(A * A, AA), 3); + VERIFY_EVALUATION_COUNT(call_ref_3(A * A, AA), 3); + VERIFY(!C.isCompressed()); - VERIFY_EVALUATION_COUNT( call_ref_3(C, C), 1); - - Ref > Ar(A); - VERIFY_IS_APPROX(Ar+Ar, A+A); - VERIFY_EVALUATION_COUNT( call_ref_1(Ar, A), 0); - VERIFY_EVALUATION_COUNT( call_ref_2(Ar, A), 0); - - Ref > Br(B); - VERIFY_EVALUATION_COUNT( call_ref_1(Br.transpose(), Br.transpose()), 0); - VERIFY_EVALUATION_COUNT( call_ref_2(Br, Br), 1); - VERIFY_EVALUATION_COUNT( call_ref_2(Br.transpose(), Br.transpose()), 0); - - Ref > Arc(A); -// VERIFY_EVALUATION_COUNT( call_ref_1(Arc, Arc), 0); // does not compile on purpose - VERIFY_EVALUATION_COUNT( call_ref_2(Arc, Arc), 0); - - VERIFY_EVALUATION_COUNT( call_ref_2(A.middleCols(1,3), A.middleCols(1,3)), 0); - - VERIFY_EVALUATION_COUNT( call_ref_2(A.col(2), A.col(2)), 0); - VERIFY_EVALUATION_COUNT( call_ref_2(vc, vc), 0); - VERIFY_EVALUATION_COUNT( call_ref_2(vr.transpose(), vr.transpose()), 0); - VERIFY_EVALUATION_COUNT( call_ref_2(vr, vr.transpose()), 0); - - VERIFY_EVALUATION_COUNT( call_ref_2(A.block(1,1,3,3), A.block(1,1,3,3)), 1); // should be 0 (allocate starts/nnz only) - - VERIFY_EVALUATION_COUNT( call_ref_4(vc, vc), 0); - VERIFY_EVALUATION_COUNT( call_ref_4(vr, vr.transpose()), 0); - VERIFY_EVALUATION_COUNT( call_ref_5(vc, vc), 0); - VERIFY_EVALUATION_COUNT( call_ref_5(vr, vr.transpose()), 0); - VERIFY_EVALUATION_COUNT( call_ref_4(A.col(2), A.col(2)), 0); - VERIFY_EVALUATION_COUNT( call_ref_5(A.col(2), A.col(2)), 0); + VERIFY_EVALUATION_COUNT(call_ref_3(C, C), 1); + + Ref> Ar(A); + VERIFY_IS_APPROX(Ar + Ar, A + A); + VERIFY_EVALUATION_COUNT(call_ref_1(Ar, A), 0); + VERIFY_EVALUATION_COUNT(call_ref_2(Ar, A), 0); + + Ref> Br(B); + VERIFY_EVALUATION_COUNT(call_ref_1(Br.transpose(), Br.transpose()), 0); + VERIFY_EVALUATION_COUNT(call_ref_2(Br, Br), 1); + VERIFY_EVALUATION_COUNT(call_ref_2(Br.transpose(), Br.transpose()), 0); + + Ref> Arc(A); + // VERIFY_EVALUATION_COUNT( call_ref_1(Arc, Arc), 0); // does not compile on purpose + VERIFY_EVALUATION_COUNT(call_ref_2(Arc, Arc), 0); + + VERIFY_EVALUATION_COUNT(call_ref_2(A.middleCols(1, 3), A.middleCols(1, 3)), 0); + + VERIFY_EVALUATION_COUNT(call_ref_2(A.col(2), A.col(2)), 0); + VERIFY_EVALUATION_COUNT(call_ref_2(vc, vc), 0); + VERIFY_EVALUATION_COUNT(call_ref_2(vr.transpose(), vr.transpose()), 0); + VERIFY_EVALUATION_COUNT(call_ref_2(vr, vr.transpose()), 0); + + VERIFY_EVALUATION_COUNT( + call_ref_2(A.block(1, 1, 3, 3), A.block(1, 1, 3, 3)), 1);// should be 0 (allocate starts/nnz only) + + VERIFY_EVALUATION_COUNT(call_ref_4(vc, vc), 0); + VERIFY_EVALUATION_COUNT(call_ref_4(vr, vr.transpose()), 0); + VERIFY_EVALUATION_COUNT(call_ref_5(vc, vc), 0); + VERIFY_EVALUATION_COUNT(call_ref_5(vr, vr.transpose()), 0); + VERIFY_EVALUATION_COUNT(call_ref_4(A.col(2), A.col(2)), 0); + VERIFY_EVALUATION_COUNT(call_ref_5(A.col(2), A.col(2)), 0); // VERIFY_EVALUATION_COUNT( call_ref_4(A.row(2), A.row(2).transpose()), 1); // does not compile on purpose - VERIFY_EVALUATION_COUNT( call_ref_5(A.row(2), A.row(2).transpose()), 1); + VERIFY_EVALUATION_COUNT(call_ref_5(A.row(2), A.row(2).transpose()), 1); } void test_sparse_ref() { - for(int i = 0; i < g_repeat; i++) { - CALL_SUBTEST_1( check_const_correctness(SparseMatrix()) ); - CALL_SUBTEST_1( check_const_correctness(SparseMatrix()) ); - CALL_SUBTEST_2( call_ref() ); + for (int i = 0; i < g_repeat; i++) { + CALL_SUBTEST_1(check_const_correctness(SparseMatrix())); + CALL_SUBTEST_1(check_const_correctness(SparseMatrix())); + CALL_SUBTEST_2(call_ref()); - CALL_SUBTEST_3( check_const_correctness(SparseVector()) ); - CALL_SUBTEST_3( check_const_correctness(SparseVector()) ); + CALL_SUBTEST_3(check_const_correctness(SparseVector())); + CALL_SUBTEST_3(check_const_correctness(SparseVector())); } } diff --git a/filmulator-gui/core/nlmeans/eigen/test/sparse_solver.h b/filmulator-gui/core/nlmeans/eigen/test/sparse_solver.h index 5145bc3e..97bded27 100644 --- a/filmulator-gui/core/nlmeans/eigen/test/sparse_solver.h +++ b/filmulator-gui/core/nlmeans/eigen/test/sparse_solver.h @@ -11,35 +11,39 @@ #include #include -template -void solve_with_guess(IterativeSolverBase& solver, const MatrixBase& b, const Guess& g, Result &x) { - if(internal::random()) - { +template +void solve_with_guess(IterativeSolverBase &solver, const MatrixBase &b, const Guess &g, Result &x) +{ + if (internal::random()) { // With a temporary through evaluator - x = solver.derived().solveWithGuess(b,g) + Result::Zero(x.rows(), x.cols()); - } - else - { + x = solver.derived().solveWithGuess(b, g) + Result::Zero(x.rows(), x.cols()); + } else { // direct evaluation within x through Assignment - x = solver.derived().solveWithGuess(b.derived(),g); + x = solver.derived().solveWithGuess(b.derived(), g); } } -template -void solve_with_guess(SparseSolverBase& solver, const MatrixBase& b, const Guess& , Result& x) { - if(internal::random()) +template +void solve_with_guess(SparseSolverBase &solver, const MatrixBase &b, const Guess &, Result &x) +{ + if (internal::random()) x = solver.derived().solve(b) + Result::Zero(x.rows(), x.cols()); else x = solver.derived().solve(b); } -template -void solve_with_guess(SparseSolverBase& solver, const SparseMatrixBase& b, const Guess& , Result& x) { +template +void solve_with_guess(SparseSolverBase &solver, const SparseMatrixBase &b, const Guess &, Result &x) +{ x = solver.derived().solve(b); } template -void check_sparse_solving(Solver& solver, const typename Solver::MatrixType& A, const Rhs& b, const DenseMat& dA, const DenseRhs& db) +void check_sparse_solving(Solver &solver, + const typename Solver::MatrixType &A, + const Rhs &b, + const DenseMat &dA, + const DenseRhs &db) { typedef typename Solver::MatrixType Mat; typedef typename Mat::Scalar Scalar; @@ -51,26 +55,24 @@ void check_sparse_solving(Solver& solver, const typename Solver::MatrixType& A, Rhs oldb = b; solver.compute(A); - if (solver.info() != Success) - { + if (solver.info() != Success) { std::cerr << "ERROR | sparse solver testing, factorization failed (" << typeid(Solver).name() << ")\n"; VERIFY(solver.info() == Success); } x = solver.solve(b); - if (solver.info() != Success) - { + if (solver.info() != Success) { std::cerr << "WARNING | sparse solver testing: solving failed (" << typeid(Solver).name() << ")\n"; return; } VERIFY(oldb.isApprox(b) && "sparse solver testing: the rhs should not be modified!"); - VERIFY(x.isApprox(refX,test_precision())); + VERIFY(x.isApprox(refX, test_precision())); x.setZero(); solve_with_guess(solver, b, x, x); VERIFY(solver.info() == Success && "solving failed when using analyzePattern/factorize API"); VERIFY(oldb.isApprox(b) && "sparse solver testing: the rhs should not be modified!"); - VERIFY(x.isApprox(refX,test_precision())); - + VERIFY(x.isApprox(refX, test_precision())); + x.setZero(); // test the analyze/factorize API solver.analyzePattern(A); @@ -79,11 +81,16 @@ void check_sparse_solving(Solver& solver, const typename Solver::MatrixType& A, x = solver.solve(b); VERIFY(solver.info() == Success && "solving failed when using analyzePattern/factorize API"); VERIFY(oldb.isApprox(b) && "sparse solver testing: the rhs should not be modified!"); - VERIFY(x.isApprox(refX,test_precision())); - + VERIFY(x.isApprox(refX, test_precision())); + x.setZero(); // test with Map - MappedSparseMatrix Am(A.rows(), A.cols(), A.nonZeros(), const_cast(A.outerIndexPtr()), const_cast(A.innerIndexPtr()), const_cast(A.valuePtr())); + MappedSparseMatrix Am(A.rows(), + A.cols(), + A.nonZeros(), + const_cast(A.outerIndexPtr()), + const_cast(A.innerIndexPtr()), + const_cast(A.valuePtr())); solver.compute(Am); VERIFY(solver.info() == Success && "factorization failed when using Map"); DenseRhs dx(refX); @@ -93,19 +100,18 @@ void check_sparse_solving(Solver& solver, const typename Solver::MatrixType& A, xm = solver.solve(bm); VERIFY(solver.info() == Success && "solving failed when using Map"); VERIFY(oldb.isApprox(bm) && "sparse solver testing: the rhs should not be modified!"); - VERIFY(xm.isApprox(refX,test_precision())); + VERIFY(xm.isApprox(refX, test_precision())); } - + // if not too large, do some extra check: - if(A.rows()<2000) - { + if (A.rows() < 2000) { // test initialization ctor { Rhs x(b.rows(), b.cols()); Solver solver2(A); VERIFY(solver2.info() == Success); x = solver2.solve(b); - VERIFY(x.isApprox(refX,test_precision())); + VERIFY(x.isApprox(refX, test_precision())); } // test dense Block as the result and rhs: @@ -113,109 +119,111 @@ void check_sparse_solving(Solver& solver, const typename Solver::MatrixType& A, DenseRhs x(refX.rows(), refX.cols()); DenseRhs oldb(db); x.setZero(); - x.block(0,0,x.rows(),x.cols()) = solver.solve(db.block(0,0,db.rows(),db.cols())); + x.block(0, 0, x.rows(), x.cols()) = solver.solve(db.block(0, 0, db.rows(), db.cols())); VERIFY(oldb.isApprox(db) && "sparse solver testing: the rhs should not be modified!"); - VERIFY(x.isApprox(refX,test_precision())); + VERIFY(x.isApprox(refX, test_precision())); } // test uncompressed inputs { Mat A2 = A; - A2.reserve((ArrayXf::Random(A.outerSize())+2).template cast().eval()); + A2.reserve((ArrayXf::Random(A.outerSize()) + 2).template cast().eval()); solver.compute(A2); Rhs x = solver.solve(b); - VERIFY(x.isApprox(refX,test_precision())); + VERIFY(x.isApprox(refX, test_precision())); } // test expression as input { - solver.compute(0.5*(A+A)); + solver.compute(0.5 * (A + A)); Rhs x = solver.solve(b); - VERIFY(x.isApprox(refX,test_precision())); + VERIFY(x.isApprox(refX, test_precision())); - Solver solver2(0.5*(A+A)); + Solver solver2(0.5 * (A + A)); Rhs x2 = solver2.solve(b); - VERIFY(x2.isApprox(refX,test_precision())); + VERIFY(x2.isApprox(refX, test_precision())); } } } template -void check_sparse_solving_real_cases(Solver& solver, const typename Solver::MatrixType& A, const Rhs& b, const typename Solver::MatrixType& fullA, const Rhs& refX) +void check_sparse_solving_real_cases(Solver &solver, + const typename Solver::MatrixType &A, + const Rhs &b, + const typename Solver::MatrixType &fullA, + const Rhs &refX) { typedef typename Solver::MatrixType Mat; typedef typename Mat::Scalar Scalar; typedef typename Mat::RealScalar RealScalar; - + Rhs x(A.cols(), b.cols()); solver.compute(A); - if (solver.info() != Success) - { + if (solver.info() != Success) { std::cerr << "ERROR | sparse solver testing, factorization failed (" << typeid(Solver).name() << ")\n"; VERIFY(solver.info() == Success); } x = solver.solve(b); - - if (solver.info() != Success) - { + + if (solver.info() != Success) { std::cerr << "WARNING | sparse solver testing, solving failed (" << typeid(Solver).name() << ")\n"; return; } - - RealScalar res_error = (fullA*x-b).norm()/b.norm(); - VERIFY( (res_error <= test_precision() ) && "sparse solver failed without noticing it"); - - if(refX.size() != 0 && (refX - x).norm()/refX.norm() > test_precision()) - { + RealScalar res_error = (fullA * x - b).norm() / b.norm(); + VERIFY((res_error <= test_precision()) && "sparse solver failed without noticing it"); + + + if (refX.size() != 0 && (refX - x).norm() / refX.norm() > test_precision()) { std::cerr << "WARNING | found solution is different from the provided reference one\n"; } - } template -void check_sparse_determinant(Solver& solver, const typename Solver::MatrixType& A, const DenseMat& dA) +void check_sparse_determinant(Solver &solver, const typename Solver::MatrixType &A, const DenseMat &dA) { typedef typename Solver::MatrixType Mat; typedef typename Mat::Scalar Scalar; - + solver.compute(A); - if (solver.info() != Success) - { + if (solver.info() != Success) { std::cerr << "WARNING | sparse solver testing: factorization failed (check_sparse_determinant)\n"; return; } Scalar refDet = dA.determinant(); - VERIFY_IS_APPROX(refDet,solver.determinant()); + VERIFY_IS_APPROX(refDet, solver.determinant()); } template -void check_sparse_abs_determinant(Solver& solver, const typename Solver::MatrixType& A, const DenseMat& dA) +void check_sparse_abs_determinant(Solver &solver, const typename Solver::MatrixType &A, const DenseMat &dA) { using std::abs; typedef typename Solver::MatrixType Mat; typedef typename Mat::Scalar Scalar; - + solver.compute(A); - if (solver.info() != Success) - { + if (solver.info() != Success) { std::cerr << "WARNING | sparse solver testing: factorization failed (check_sparse_abs_determinant)\n"; return; } Scalar refDet = abs(dA.determinant()); - VERIFY_IS_APPROX(refDet,solver.absDeterminant()); + VERIFY_IS_APPROX(refDet, solver.absDeterminant()); } template -int generate_sparse_spd_problem(Solver& , typename Solver::MatrixType& A, typename Solver::MatrixType& halfA, DenseMat& dA, int maxSize = 300) +int generate_sparse_spd_problem(Solver &, + typename Solver::MatrixType &A, + typename Solver::MatrixType &halfA, + DenseMat &dA, + int maxSize = 300) { typedef typename Solver::MatrixType Mat; typedef typename Mat::Scalar Scalar; - typedef Matrix DenseMatrix; + typedef Matrix DenseMatrix; - int size = internal::random(1,maxSize); - double density = (std::max)(8./(size*size), 0.01); + int size = internal::random(1, maxSize); + double density = (std::max)(8. / (size * size), 0.01); Mat M(size, size); DenseMatrix dM(size, size); @@ -224,57 +232,52 @@ int generate_sparse_spd_problem(Solver& , typename Solver::MatrixType& A, typena A = M * M.adjoint(); dA = dM * dM.adjoint(); - - halfA.resize(size,size); - if(Solver::UpLo==(Lower|Upper)) + + halfA.resize(size, size); + if (Solver::UpLo == (Lower | Upper)) halfA = A; else halfA.template selfadjointView().rankUpdate(M); - + return size; } #ifdef TEST_REAL_CASES -template -inline std::string get_matrixfolder() +template inline std::string get_matrixfolder() { - std::string mat_folder = TEST_REAL_CASES; - if( internal::is_same >::value || internal::is_same >::value ) - mat_folder = mat_folder + static_cast("/complex/"); + std::string mat_folder = TEST_REAL_CASES; + if (internal::is_same>::value || internal::is_same>::value) + mat_folder = mat_folder + static_cast("/complex/"); else mat_folder = mat_folder + static_cast("/real/"); return mat_folder; } std::string sym_to_string(int sym) { - if(sym==Symmetric) return "Symmetric "; - if(sym==SPD) return "SPD "; + if (sym == Symmetric) return "Symmetric "; + if (sym == SPD) return "SPD "; return ""; } -template -std::string solver_stats(const IterativeSolverBase &solver) +template std::string solver_stats(const IterativeSolverBase &solver) { std::stringstream ss; ss << solver.iterations() << " iters, error: " << solver.error(); return ss.str(); } -template -std::string solver_stats(const SparseSolverBase &/*solver*/) -{ - return ""; -} +template std::string solver_stats(const SparseSolverBase & /*solver*/) { return ""; } #endif -template void check_sparse_spd_solving(Solver& solver, int maxSize = 300, int maxRealWorldSize = 100000) +template +void check_sparse_spd_solving(Solver &solver, int maxSize = 300, int maxRealWorldSize = 100000) { typedef typename Solver::MatrixType Mat; typedef typename Mat::Scalar Scalar; typedef typename Mat::StorageIndex StorageIndex; - typedef SparseMatrix SpMat; + typedef SparseMatrix SpMat; typedef SparseVector SpVec; - typedef Matrix DenseMatrix; - typedef Matrix DenseVector; + typedef Matrix DenseMatrix; + typedef Matrix DenseVector; // generate the problem Mat A, halfA; @@ -283,64 +286,57 @@ template void check_sparse_spd_solving(Solver& solver, int maxS int size = generate_sparse_spd_problem(solver, A, halfA, dA, maxSize); // generate the right hand sides - int rhsCols = internal::random(1,16); - double density = (std::max)(8./(size*rhsCols), 0.1); - SpMat B(size,rhsCols); + int rhsCols = internal::random(1, 16); + double density = (std::max)(8. / (size * rhsCols), 0.1); + SpMat B(size, rhsCols); DenseVector b = DenseVector::Random(size); - DenseMatrix dB(size,rhsCols); + DenseMatrix dB(size, rhsCols); initSparse(density, dB, B, ForceNonZeroDiag); SpVec c = B.col(0); DenseVector dc = dB.col(0); - - CALL_SUBTEST( check_sparse_solving(solver, A, b, dA, b) ); - CALL_SUBTEST( check_sparse_solving(solver, halfA, b, dA, b) ); - CALL_SUBTEST( check_sparse_solving(solver, A, dB, dA, dB) ); - CALL_SUBTEST( check_sparse_solving(solver, halfA, dB, dA, dB) ); - CALL_SUBTEST( check_sparse_solving(solver, A, B, dA, dB) ); - CALL_SUBTEST( check_sparse_solving(solver, halfA, B, dA, dB) ); - CALL_SUBTEST( check_sparse_solving(solver, A, c, dA, dc) ); - CALL_SUBTEST( check_sparse_solving(solver, halfA, c, dA, dc) ); - + + CALL_SUBTEST(check_sparse_solving(solver, A, b, dA, b)); + CALL_SUBTEST(check_sparse_solving(solver, halfA, b, dA, b)); + CALL_SUBTEST(check_sparse_solving(solver, A, dB, dA, dB)); + CALL_SUBTEST(check_sparse_solving(solver, halfA, dB, dA, dB)); + CALL_SUBTEST(check_sparse_solving(solver, A, B, dA, dB)); + CALL_SUBTEST(check_sparse_solving(solver, halfA, B, dA, dB)); + CALL_SUBTEST(check_sparse_solving(solver, A, c, dA, dc)); + CALL_SUBTEST(check_sparse_solving(solver, halfA, c, dA, dc)); + // check only once - if(i==0) - { + if (i == 0) { b = DenseVector::Zero(size); check_sparse_solving(solver, A, b, dA, b); } } - - // First, get the folder + + // First, get the folder #ifdef TEST_REAL_CASES // Test real problems with double precision only - if (internal::is_same::Real, double>::value) - { + if (internal::is_same::Real, double>::value) { std::string mat_folder = get_matrixfolder(); MatrixMarketIterator it(mat_folder); - for (; it; ++it) - { - if (it.sym() == SPD){ + for (; it; ++it) { + if (it.sym() == SPD) { A = it.matrix(); - if(A.diagonal().size() <= maxRealWorldSize) - { + if (A.diagonal().size() <= maxRealWorldSize) { DenseVector b = it.rhs(); DenseVector refX = it.refX(); PermutationMatrix pnull; halfA.resize(A.rows(), A.cols()); - if(Solver::UpLo == (Lower|Upper)) + if (Solver::UpLo == (Lower | Upper)) halfA = A; else halfA.template selfadjointView() = A.template triangularView().twistedBy(pnull); - - std::cout << "INFO | Testing " << sym_to_string(it.sym()) << "sparse problem " << it.matname() - << " (" << A.rows() << "x" << A.cols() << ") using " << typeid(Solver).name() << "..." << std::endl; - CALL_SUBTEST( check_sparse_solving_real_cases(solver, A, b, A, refX) ); + + std::cout << "INFO | Testing " << sym_to_string(it.sym()) << "sparse problem " << it.matname() << " (" + << A.rows() << "x" << A.cols() << ") using " << typeid(Solver).name() << "..." << std::endl; + CALL_SUBTEST(check_sparse_solving_real_cases(solver, A, b, A, refX)); std::string stats = solver_stats(solver); - if(stats.size()>0) - std::cout << "INFO | " << stats << std::endl; - CALL_SUBTEST( check_sparse_solving_real_cases(solver, halfA, b, A, refX) ); - } - else - { + if (stats.size() > 0) std::cout << "INFO | " << stats << std::endl; + CALL_SUBTEST(check_sparse_solving_real_cases(solver, halfA, b, A, refX)); + } else { std::cout << "INFO | Skip sparse problem \"" << it.matname() << "\" (too large)" << std::endl; } } @@ -351,61 +347,67 @@ template void check_sparse_spd_solving(Solver& solver, int maxS #endif } -template void check_sparse_spd_determinant(Solver& solver) +template void check_sparse_spd_determinant(Solver &solver) { typedef typename Solver::MatrixType Mat; typedef typename Mat::Scalar Scalar; - typedef Matrix DenseMatrix; + typedef Matrix DenseMatrix; // generate the problem Mat A, halfA; DenseMatrix dA; generate_sparse_spd_problem(solver, A, halfA, dA, 30); - + for (int i = 0; i < g_repeat; i++) { - check_sparse_determinant(solver, A, dA); - check_sparse_determinant(solver, halfA, dA ); + check_sparse_determinant(solver, A, dA); + check_sparse_determinant(solver, halfA, dA); } } template -Index generate_sparse_square_problem(Solver&, typename Solver::MatrixType& A, DenseMat& dA, int maxSize = 300, int options = ForceNonZeroDiag) +Index generate_sparse_square_problem(Solver &, + typename Solver::MatrixType &A, + DenseMat &dA, + int maxSize = 300, + int options = ForceNonZeroDiag) { typedef typename Solver::MatrixType Mat; typedef typename Mat::Scalar Scalar; - Index size = internal::random(1,maxSize); - double density = (std::max)(8./(size*size), 0.01); - - A.resize(size,size); - dA.resize(size,size); + Index size = internal::random(1, maxSize); + double density = (std::max)(8. / (size * size), 0.01); + + A.resize(size, size); + dA.resize(size, size); initSparse(density, dA, A, options); - + return size; } -struct prune_column { +struct prune_column +{ Index m_col; prune_column(Index col) : m_col(col) {} - template - bool operator()(Index, Index col, const Scalar&) const { - return col != m_col; - } + template bool operator()(Index, Index col, const Scalar &) const { return col != m_col; } }; -template void check_sparse_square_solving(Solver& solver, int maxSize = 300, int maxRealWorldSize = 100000, bool checkDeficient = false) +template +void check_sparse_square_solving(Solver &solver, + int maxSize = 300, + int maxRealWorldSize = 100000, + bool checkDeficient = false) { typedef typename Solver::MatrixType Mat; typedef typename Mat::Scalar Scalar; - typedef SparseMatrix SpMat; + typedef SparseMatrix SpMat; typedef SparseVector SpVec; - typedef Matrix DenseMatrix; - typedef Matrix DenseVector; + typedef Matrix DenseMatrix; + typedef Matrix DenseVector; - int rhsCols = internal::random(1,16); + int rhsCols = internal::random(1, 16); Mat A; DenseMatrix dA; @@ -414,56 +416,49 @@ template void check_sparse_square_solving(Solver& solver, int m A.makeCompressed(); DenseVector b = DenseVector::Random(size); - DenseMatrix dB(size,rhsCols); - SpMat B(size,rhsCols); - double density = (std::max)(8./(size*rhsCols), 0.1); + DenseMatrix dB(size, rhsCols); + SpMat B(size, rhsCols); + double density = (std::max)(8. / (size * rhsCols), 0.1); initSparse(density, dB, B, ForceNonZeroDiag); B.makeCompressed(); SpVec c = B.col(0); DenseVector dc = dB.col(0); - CALL_SUBTEST(check_sparse_solving(solver, A, b, dA, b)); + CALL_SUBTEST(check_sparse_solving(solver, A, b, dA, b)); CALL_SUBTEST(check_sparse_solving(solver, A, dB, dA, dB)); - CALL_SUBTEST(check_sparse_solving(solver, A, B, dA, dB)); - CALL_SUBTEST(check_sparse_solving(solver, A, c, dA, dc)); - + CALL_SUBTEST(check_sparse_solving(solver, A, B, dA, dB)); + CALL_SUBTEST(check_sparse_solving(solver, A, c, dA, dc)); + // check only once - if(i==0) - { + if (i == 0) { b = DenseVector::Zero(size); check_sparse_solving(solver, A, b, dA, b); } // regression test for Bug 792 (structurally rank deficient matrices): - if(checkDeficient && size>1) { - Index col = internal::random(0,int(size-1)); + if (checkDeficient && size > 1) { + Index col = internal::random(0, int(size - 1)); A.prune(prune_column(col)); solver.compute(A); VERIFY_IS_EQUAL(solver.info(), NumericalIssue); } } - - // First, get the folder + + // First, get the folder #ifdef TEST_REAL_CASES // Test real problems with double precision only - if (internal::is_same::Real, double>::value) - { + if (internal::is_same::Real, double>::value) { std::string mat_folder = get_matrixfolder(); MatrixMarketIterator it(mat_folder); - for (; it; ++it) - { + for (; it; ++it) { A = it.matrix(); - if(A.diagonal().size() <= maxRealWorldSize) - { + if (A.diagonal().size() <= maxRealWorldSize) { DenseVector b = it.rhs(); DenseVector refX = it.refX(); - std::cout << "INFO | Testing " << sym_to_string(it.sym()) << "sparse problem " << it.matname() - << " (" << A.rows() << "x" << A.cols() << ") using " << typeid(Solver).name() << "..." << std::endl; + std::cout << "INFO | Testing " << sym_to_string(it.sym()) << "sparse problem " << it.matname() << " (" + << A.rows() << "x" << A.cols() << ") using " << typeid(Solver).name() << "..." << std::endl; CALL_SUBTEST(check_sparse_solving_real_cases(solver, A, b, A, refX)); std::string stats = solver_stats(solver); - if(stats.size()>0) - std::cout << "INFO | " << stats << std::endl; - } - else - { + if (stats.size() > 0) std::cout << "INFO | " << stats << std::endl; + } else { std::cout << "INFO | SKIP sparse problem \"" << it.matname() << "\" (too large)" << std::endl; } } @@ -471,37 +466,36 @@ template void check_sparse_square_solving(Solver& solver, int m #else EIGEN_UNUSED_VARIABLE(maxRealWorldSize); #endif - } -template void check_sparse_square_determinant(Solver& solver) +template void check_sparse_square_determinant(Solver &solver) { typedef typename Solver::MatrixType Mat; typedef typename Mat::Scalar Scalar; - typedef Matrix DenseMatrix; - + typedef Matrix DenseMatrix; + for (int i = 0; i < g_repeat; i++) { // generate the problem Mat A; DenseMatrix dA; - - int size = internal::random(1,30); - dA.setRandom(size,size); - - dA = (dA.array().abs()<0.3).select(0,dA); - dA.diagonal() = (dA.diagonal().array()==0).select(1,dA.diagonal()); + + int size = internal::random(1, 30); + dA.setRandom(size, size); + + dA = (dA.array().abs() < 0.3).select(0, dA); + dA.diagonal() = (dA.diagonal().array() == 0).select(1, dA.diagonal()); A = dA.sparseView(); A.makeCompressed(); - + check_sparse_determinant(solver, A, dA); } } -template void check_sparse_square_abs_determinant(Solver& solver) +template void check_sparse_square_abs_determinant(Solver &solver) { typedef typename Solver::MatrixType Mat; typedef typename Mat::Scalar Scalar; - typedef Matrix DenseMatrix; + typedef Matrix DenseMatrix; for (int i = 0; i < g_repeat; i++) { // generate the problem @@ -514,30 +508,34 @@ template void check_sparse_square_abs_determinant(Solver& solve } template -void generate_sparse_leastsquare_problem(Solver&, typename Solver::MatrixType& A, DenseMat& dA, int maxSize = 300, int options = ForceNonZeroDiag) +void generate_sparse_leastsquare_problem(Solver &, + typename Solver::MatrixType &A, + DenseMat &dA, + int maxSize = 300, + int options = ForceNonZeroDiag) { typedef typename Solver::MatrixType Mat; typedef typename Mat::Scalar Scalar; - int rows = internal::random(1,maxSize); - int cols = internal::random(1,rows); - double density = (std::max)(8./(rows*cols), 0.01); - - A.resize(rows,cols); - dA.resize(rows,cols); + int rows = internal::random(1, maxSize); + int cols = internal::random(1, rows); + double density = (std::max)(8. / (rows * cols), 0.01); + + A.resize(rows, cols); + dA.resize(rows, cols); initSparse(density, dA, A, options); } -template void check_sparse_leastsquare_solving(Solver& solver) +template void check_sparse_leastsquare_solving(Solver &solver) { typedef typename Solver::MatrixType Mat; typedef typename Mat::Scalar Scalar; - typedef SparseMatrix SpMat; - typedef Matrix DenseMatrix; - typedef Matrix DenseVector; + typedef SparseMatrix SpMat; + typedef Matrix DenseMatrix; + typedef Matrix DenseVector; - int rhsCols = internal::random(1,16); + int rhsCols = internal::random(1, 16); Mat A; DenseMatrix dA; @@ -546,18 +544,17 @@ template void check_sparse_leastsquare_solving(Solver& solver) A.makeCompressed(); DenseVector b = DenseVector::Random(A.rows()); - DenseMatrix dB(A.rows(),rhsCols); - SpMat B(A.rows(),rhsCols); - double density = (std::max)(8./(A.rows()*rhsCols), 0.1); + DenseMatrix dB(A.rows(), rhsCols); + SpMat B(A.rows(), rhsCols); + double density = (std::max)(8. / (A.rows() * rhsCols), 0.1); initSparse(density, dB, B, ForceNonZeroDiag); B.makeCompressed(); - check_sparse_solving(solver, A, b, dA, b); + check_sparse_solving(solver, A, b, dA, b); check_sparse_solving(solver, A, dB, dA, dB); - check_sparse_solving(solver, A, B, dA, dB); - + check_sparse_solving(solver, A, B, dA, dB); + // check only once - if(i==0) - { + if (i == 0) { b = DenseVector::Zero(A.rows()); check_sparse_solving(solver, A, b, dA, b); } diff --git a/filmulator-gui/core/nlmeans/eigen/test/sparse_solvers.cpp b/filmulator-gui/core/nlmeans/eigen/test/sparse_solvers.cpp index 3a8873d4..f42049c5 100644 --- a/filmulator-gui/core/nlmeans/eigen/test/sparse_solvers.cpp +++ b/filmulator-gui/core/nlmeans/eigen/test/sparse_solvers.cpp @@ -9,32 +9,28 @@ #include "sparse.h" -template void -initSPD(double density, - Matrix& refMat, - SparseMatrix& sparseMat) +template +void initSPD(double density, Matrix &refMat, SparseMatrix &sparseMat) { - Matrix aux(refMat.rows(),refMat.cols()); - initSparse(density,refMat,sparseMat); + Matrix aux(refMat.rows(), refMat.cols()); + initSparse(density, refMat, sparseMat); refMat = refMat * refMat.adjoint(); - for (int k=0; k<2; ++k) - { - initSparse(density,aux,sparseMat,ForceNonZeroDiag); + for (int k = 0; k < 2; ++k) { + initSparse(density, aux, sparseMat, ForceNonZeroDiag); refMat += aux * aux.adjoint(); } sparseMat.setZero(); - for (int j=0 ; j void sparse_solvers(int rows, int cols) { - double density = (std::max)(8./(rows*cols), 0.01); - typedef Matrix DenseMatrix; - typedef Matrix DenseVector; + double density = (std::max)(8. / (rows * cols), 0.01); + typedef Matrix DenseMatrix; + typedef Matrix DenseVector; // Scalar eps = 1e-6; DenseVector vec1 = DenseVector::Random(rows); @@ -49,64 +45,65 @@ template void sparse_solvers(int rows, int cols) DenseMatrix refMat2 = DenseMatrix::Zero(rows, cols); // lower - dense - initSparse(density, refMat2, m2, ForceNonZeroDiag|MakeLowerTriangular, &zeroCoords, &nonzeroCoords); - VERIFY_IS_APPROX(refMat2.template triangularView().solve(vec2), - m2.template triangularView().solve(vec3)); + initSparse(density, refMat2, m2, ForceNonZeroDiag | MakeLowerTriangular, &zeroCoords, &nonzeroCoords); + VERIFY_IS_APPROX( + refMat2.template triangularView().solve(vec2), m2.template triangularView().solve(vec3)); // upper - dense - initSparse(density, refMat2, m2, ForceNonZeroDiag|MakeUpperTriangular, &zeroCoords, &nonzeroCoords); - VERIFY_IS_APPROX(refMat2.template triangularView().solve(vec2), - m2.template triangularView().solve(vec3)); + initSparse(density, refMat2, m2, ForceNonZeroDiag | MakeUpperTriangular, &zeroCoords, &nonzeroCoords); + VERIFY_IS_APPROX( + refMat2.template triangularView().solve(vec2), m2.template triangularView().solve(vec3)); VERIFY_IS_APPROX(refMat2.conjugate().template triangularView().solve(vec2), - m2.conjugate().template triangularView().solve(vec3)); + m2.conjugate().template triangularView().solve(vec3)); { SparseMatrix cm2(m2); - //Index rows, Index cols, Index nnz, Index* outerIndexPtr, Index* innerIndexPtr, Scalar* valuePtr - MappedSparseMatrix mm2(rows, cols, cm2.nonZeros(), cm2.outerIndexPtr(), cm2.innerIndexPtr(), cm2.valuePtr()); + // Index rows, Index cols, Index nnz, Index* outerIndexPtr, Index* innerIndexPtr, Scalar* valuePtr + MappedSparseMatrix mm2( + rows, cols, cm2.nonZeros(), cm2.outerIndexPtr(), cm2.innerIndexPtr(), cm2.valuePtr()); VERIFY_IS_APPROX(refMat2.conjugate().template triangularView().solve(vec2), - mm2.conjugate().template triangularView().solve(vec3)); + mm2.conjugate().template triangularView().solve(vec3)); } // lower - transpose - initSparse(density, refMat2, m2, ForceNonZeroDiag|MakeLowerTriangular, &zeroCoords, &nonzeroCoords); + initSparse(density, refMat2, m2, ForceNonZeroDiag | MakeLowerTriangular, &zeroCoords, &nonzeroCoords); VERIFY_IS_APPROX(refMat2.transpose().template triangularView().solve(vec2), - m2.transpose().template triangularView().solve(vec3)); + m2.transpose().template triangularView().solve(vec3)); // upper - transpose - initSparse(density, refMat2, m2, ForceNonZeroDiag|MakeUpperTriangular, &zeroCoords, &nonzeroCoords); + initSparse(density, refMat2, m2, ForceNonZeroDiag | MakeUpperTriangular, &zeroCoords, &nonzeroCoords); VERIFY_IS_APPROX(refMat2.transpose().template triangularView().solve(vec2), - m2.transpose().template triangularView().solve(vec3)); + m2.transpose().template triangularView().solve(vec3)); SparseMatrix matB(rows, rows); DenseMatrix refMatB = DenseMatrix::Zero(rows, rows); // lower - sparse - initSparse(density, refMat2, m2, ForceNonZeroDiag|MakeLowerTriangular); + initSparse(density, refMat2, m2, ForceNonZeroDiag | MakeLowerTriangular); initSparse(density, refMatB, matB); refMat2.template triangularView().solveInPlace(refMatB); m2.template triangularView().solveInPlace(matB); VERIFY_IS_APPROX(matB.toDense(), refMatB); // upper - sparse - initSparse(density, refMat2, m2, ForceNonZeroDiag|MakeUpperTriangular); + initSparse(density, refMat2, m2, ForceNonZeroDiag | MakeUpperTriangular); initSparse(density, refMatB, matB); refMat2.template triangularView().solveInPlace(refMatB); m2.template triangularView().solveInPlace(matB); VERIFY_IS_APPROX(matB, refMatB); // test deprecated API - initSparse(density, refMat2, m2, ForceNonZeroDiag|MakeLowerTriangular, &zeroCoords, &nonzeroCoords); - VERIFY_IS_APPROX(refMat2.template triangularView().solve(vec2), - m2.template triangularView().solve(vec3)); + initSparse(density, refMat2, m2, ForceNonZeroDiag | MakeLowerTriangular, &zeroCoords, &nonzeroCoords); + VERIFY_IS_APPROX( + refMat2.template triangularView().solve(vec2), m2.template triangularView().solve(vec3)); } } void test_sparse_solvers() { - for(int i = 0; i < g_repeat; i++) { - CALL_SUBTEST_1(sparse_solvers(8, 8) ); - int s = internal::random(1,300); - CALL_SUBTEST_2(sparse_solvers >(s,s) ); - CALL_SUBTEST_1(sparse_solvers(s,s) ); + for (int i = 0; i < g_repeat; i++) { + CALL_SUBTEST_1(sparse_solvers(8, 8)); + int s = internal::random(1, 300); + CALL_SUBTEST_2(sparse_solvers>(s, s)); + CALL_SUBTEST_1(sparse_solvers(s, s)); } } diff --git a/filmulator-gui/core/nlmeans/eigen/test/sparse_vector.cpp b/filmulator-gui/core/nlmeans/eigen/test/sparse_vector.cpp index b3e1dda2..a2a04abf 100644 --- a/filmulator-gui/core/nlmeans/eigen/test/sparse_vector.cpp +++ b/filmulator-gui/core/nlmeans/eigen/test/sparse_vector.cpp @@ -9,22 +9,20 @@ #include "sparse.h" -template void sparse_vector(int rows, int cols) +template void sparse_vector(int rows, int cols) { - double densityMat = (std::max)(8./(rows*cols), 0.01); - double densityVec = (std::max)(8./(rows), 0.1); - typedef Matrix DenseMatrix; - typedef Matrix DenseVector; - typedef SparseVector SparseVectorType; - typedef SparseMatrix SparseMatrixType; + double densityMat = (std::max)(8. / (rows * cols), 0.01); + double densityVec = (std::max)(8. / (rows), 0.1); + typedef Matrix DenseMatrix; + typedef Matrix DenseVector; + typedef SparseVector SparseVectorType; + typedef SparseMatrix SparseMatrixType; Scalar eps = 1e-6; - SparseMatrixType m1(rows,rows); + SparseMatrixType m1(rows, rows); SparseVectorType v1(rows), v2(rows), v3(rows); DenseMatrix refM1 = DenseMatrix::Zero(rows, rows); - DenseVector refV1 = DenseVector::Random(rows), - refV2 = DenseVector::Random(rows), - refV3 = DenseVector::Random(rows); + DenseVector refV1 = DenseVector::Random(rows), refV2 = DenseVector::Random(rows), refV3 = DenseVector::Random(rows); std::vector zerocoords, nonzerocoords; initSparse(densityVec, refV1, v1, &zerocoords, &nonzerocoords); @@ -36,128 +34,121 @@ template void sparse_vector(int rows, int Scalar s1 = internal::random(); // test coeff and coeffRef - for (unsigned int i=0; i(0,rows-1); + for (int k = 0; k < rows; ++k) { + int i = internal::random(0, rows - 1); Scalar v = internal::random(); v4.coeffRef(i) += v; v5.coeffRef(i) += v; } - VERIFY_IS_APPROX(v4,v5); + VERIFY_IS_APPROX(v4, v5); } v1.coeffRef(nonzerocoords[0]) = Scalar(5); refV1.coeffRef(nonzerocoords[0]) = Scalar(5); VERIFY_IS_APPROX(v1, refV1); - VERIFY_IS_APPROX(v1+v2, refV1+refV2); - VERIFY_IS_APPROX(v1+v2+v3, refV1+refV2+refV3); + VERIFY_IS_APPROX(v1 + v2, refV1 + refV2); + VERIFY_IS_APPROX(v1 + v2 + v3, refV1 + refV2 + refV3); - VERIFY_IS_APPROX(v1*s1-v2, refV1*s1-refV2); + VERIFY_IS_APPROX(v1 * s1 - v2, refV1 * s1 - refV2); - VERIFY_IS_APPROX(v1*=s1, refV1*=s1); - VERIFY_IS_APPROX(v1/=s1, refV1/=s1); + VERIFY_IS_APPROX(v1 *= s1, refV1 *= s1); + VERIFY_IS_APPROX(v1 /= s1, refV1 /= s1); - VERIFY_IS_APPROX(v1+=v2, refV1+=refV2); - VERIFY_IS_APPROX(v1-=v2, refV1-=refV2); + VERIFY_IS_APPROX(v1 += v2, refV1 += refV2); + VERIFY_IS_APPROX(v1 -= v2, refV1 -= refV2); VERIFY_IS_APPROX(v1.dot(v2), refV1.dot(refV2)); VERIFY_IS_APPROX(v1.dot(refV2), refV1.dot(refV2)); - VERIFY_IS_APPROX(m1*v2, refM1*refV2); - VERIFY_IS_APPROX(v1.dot(m1*v2), refV1.dot(refM1*refV2)); + VERIFY_IS_APPROX(m1 * v2, refM1 * refV2); + VERIFY_IS_APPROX(v1.dot(m1 * v2), refV1.dot(refM1 * refV2)); { - int i = internal::random(0,rows-1); + int i = internal::random(0, rows - 1); VERIFY_IS_APPROX(v1.dot(m1.col(i)), refV1.dot(refM1.col(i))); } VERIFY_IS_APPROX(v1.squaredNorm(), refV1.squaredNorm()); - + VERIFY_IS_APPROX(v1.blueNorm(), refV1.blueNorm()); // test aliasing VERIFY_IS_APPROX((v1 = -v1), (refV1 = -refV1)); VERIFY_IS_APPROX((v1 = v1.transpose()), (refV1 = refV1.transpose().eval())); VERIFY_IS_APPROX((v1 += -v1), (refV1 += -refV1)); - + // sparse matrix to sparse vector SparseMatrixType mv1; - VERIFY_IS_APPROX((mv1=v1),v1); - VERIFY_IS_APPROX(mv1,(v1=mv1)); - VERIFY_IS_APPROX(mv1,(v1=mv1.transpose())); - + VERIFY_IS_APPROX((mv1 = v1), v1); + VERIFY_IS_APPROX(mv1, (v1 = mv1)); + VERIFY_IS_APPROX(mv1, (v1 = mv1.transpose())); + // check copy to dense vector with transpose refV3.resize(0); - VERIFY_IS_APPROX(refV3 = v1.transpose(),v1.toDense()); - VERIFY_IS_APPROX(DenseVector(v1),v1.toDense()); + VERIFY_IS_APPROX(refV3 = v1.transpose(), v1.toDense()); + VERIFY_IS_APPROX(DenseVector(v1), v1.toDense()); // test conservative resize { std::vector inc; - if(rows > 3) - inc.push_back(-3); + if (rows > 3) inc.push_back(-3); inc.push_back(0); inc.push_back(3); inc.push_back(1); inc.push_back(10); - for(std::size_t i = 0; i< inc.size(); i++) { + for (std::size_t i = 0; i < inc.size(); i++) { StorageIndex incRows = inc[i]; SparseVectorType vec1(rows); DenseVector refVec1 = DenseVector::Zero(rows); initSparse(densityVec, refVec1, vec1); - vec1.conservativeResize(rows+incRows); - refVec1.conservativeResize(rows+incRows); + vec1.conservativeResize(rows + incRows); + refVec1.conservativeResize(rows + incRows); if (incRows > 0) refVec1.tail(incRows).setZero(); VERIFY_IS_APPROX(vec1, refVec1); // Insert new values - if (incRows > 0) - vec1.insert(vec1.rows()-1) = refVec1(refVec1.rows()-1) = 1; + if (incRows > 0) vec1.insert(vec1.rows() - 1) = refVec1(refVec1.rows() - 1) = 1; VERIFY_IS_APPROX(vec1, refVec1); } } - } void test_sparse_vector() { - for(int i = 0; i < g_repeat; i++) { - int r = Eigen::internal::random(1,500), c = Eigen::internal::random(1,500); - if(Eigen::internal::random(0,4) == 0) { - r = c; // check square matrices in 25% of tries + for (int i = 0; i < g_repeat; i++) { + int r = Eigen::internal::random(1, 500), c = Eigen::internal::random(1, 500); + if (Eigen::internal::random(0, 4) == 0) { + r = c;// check square matrices in 25% of tries } - EIGEN_UNUSED_VARIABLE(r+c); + EIGEN_UNUSED_VARIABLE(r + c); - CALL_SUBTEST_1(( sparse_vector(8, 8) )); - CALL_SUBTEST_2(( sparse_vector, int>(r, c) )); - CALL_SUBTEST_1(( sparse_vector(r, c) )); - CALL_SUBTEST_1(( sparse_vector(r, c) )); + CALL_SUBTEST_1((sparse_vector(8, 8))); + CALL_SUBTEST_2((sparse_vector, int>(r, c))); + CALL_SUBTEST_1((sparse_vector(r, c))); + CALL_SUBTEST_1((sparse_vector(r, c))); } } - diff --git a/filmulator-gui/core/nlmeans/eigen/test/sparselu.cpp b/filmulator-gui/core/nlmeans/eigen/test/sparselu.cpp index bd000baf..1efd5ba0 100644 --- a/filmulator-gui/core/nlmeans/eigen/test/sparselu.cpp +++ b/filmulator-gui/core/nlmeans/eigen/test/sparselu.cpp @@ -21,25 +21,25 @@ template void test_sparselu_T() { - SparseLU /*, COLAMDOrdering*/ > sparselu_colamd; // COLAMDOrdering is the default - SparseLU, AMDOrdering > sparselu_amd; - SparseLU, NaturalOrdering > sparselu_natural; - - check_sparse_square_solving(sparselu_colamd, 300, 100000, true); - check_sparse_square_solving(sparselu_amd, 300, 10000, true); - check_sparse_square_solving(sparselu_natural, 300, 2000, true); - + SparseLU /*, COLAMDOrdering*/> sparselu_colamd;// COLAMDOrdering is the default + SparseLU, AMDOrdering> sparselu_amd; + SparseLU, NaturalOrdering> sparselu_natural; + + check_sparse_square_solving(sparselu_colamd, 300, 100000, true); + check_sparse_square_solving(sparselu_amd, 300, 10000, true); + check_sparse_square_solving(sparselu_natural, 300, 2000, true); + check_sparse_square_abs_determinant(sparselu_colamd); check_sparse_square_abs_determinant(sparselu_amd); - + check_sparse_square_determinant(sparselu_colamd); check_sparse_square_determinant(sparselu_amd); } void test_sparselu() { - CALL_SUBTEST_1(test_sparselu_T()); + CALL_SUBTEST_1(test_sparselu_T()); CALL_SUBTEST_2(test_sparselu_T()); - CALL_SUBTEST_3(test_sparselu_T >()); - CALL_SUBTEST_4(test_sparselu_T >()); + CALL_SUBTEST_3(test_sparselu_T>()); + CALL_SUBTEST_4(test_sparselu_T>()); } diff --git a/filmulator-gui/core/nlmeans/eigen/test/sparseqr.cpp b/filmulator-gui/core/nlmeans/eigen/test/sparseqr.cpp index f0e721fc..38eaa15f 100644 --- a/filmulator-gui/core/nlmeans/eigen/test/sparseqr.cpp +++ b/filmulator-gui/core/nlmeans/eigen/test/sparseqr.cpp @@ -9,49 +9,48 @@ #include "sparse.h" #include -template -int generate_sparse_rectangular_problem(MatrixType& A, DenseMat& dA, int maxRows = 300, int maxCols = 150) +template +int generate_sparse_rectangular_problem(MatrixType &A, DenseMat &dA, int maxRows = 300, int maxCols = 150) { eigen_assert(maxRows >= maxCols); typedef typename MatrixType::Scalar Scalar; - int rows = internal::random(1,maxRows); - int cols = internal::random(1,maxCols); - double density = (std::max)(8./(rows*cols), 0.01); - - A.resize(rows,cols); - dA.resize(rows,cols); - initSparse(density, dA, A,ForceNonZeroDiag); + int rows = internal::random(1, maxRows); + int cols = internal::random(1, maxCols); + double density = (std::max)(8. / (rows * cols), 0.01); + + A.resize(rows, cols); + dA.resize(rows, cols); + initSparse(density, dA, A, ForceNonZeroDiag); A.makeCompressed(); - int nop = internal::random(0, internal::random(0,1) > 0.5 ? cols/2 : 0); - for(int k=0; k(0,cols-1); - int j1 = internal::random(0,cols-1); + int nop = internal::random(0, internal::random(0, 1) > 0.5 ? cols / 2 : 0); + for (int k = 0; k < nop; ++k) { + int j0 = internal::random(0, cols - 1); + int j1 = internal::random(0, cols - 1); Scalar s = internal::random(); - A.col(j0) = s * A.col(j1); + A.col(j0) = s * A.col(j1); dA.col(j0) = s * dA.col(j1); } - -// if(rows void test_sparseqr_scalar() { - typedef SparseMatrix MatrixType; - typedef Matrix DenseMat; - typedef Matrix DenseVector; + typedef SparseMatrix MatrixType; + typedef Matrix DenseMat; + typedef Matrix DenseVector; MatrixType A; DenseMat dA; - DenseVector refX,x,b; - SparseQR > solver; - generate_sparse_rectangular_problem(A,dA); - + DenseVector refX, x, b; + SparseQR> solver; + generate_sparse_rectangular_problem(A, dA); + b = dA * DenseVector::Random(A.cols()); solver.compute(A); @@ -64,54 +63,49 @@ template void test_sparseqr_scalar() VERIFY_IS_EQUAL(solver.matrixR().cols(), A.cols()); // Q and R can be multiplied - DenseMat recoveredA = solver.matrixQ() - * DenseMat(solver.matrixR().template triangularView()) - * solver.colsPermutation().transpose(); + DenseMat recoveredA = solver.matrixQ() * DenseMat(solver.matrixR().template triangularView()) + * solver.colsPermutation().transpose(); VERIFY_IS_EQUAL(recoveredA.rows(), A.rows()); VERIFY_IS_EQUAL(recoveredA.cols(), A.cols()); // and in the full rank case the original matrix is recovered - if (solver.rank() == A.cols()) - { - VERIFY_IS_APPROX(A, recoveredA); - } + if (solver.rank() == A.cols()) { VERIFY_IS_APPROX(A, recoveredA); } - if(internal::random(0,1)>0.5f) - solver.factorize(A); // this checks that calling analyzePattern is not needed if the pattern do not change. - if (solver.info() != Success) - { + if (internal::random(0, 1) > 0.5f) + solver.factorize(A);// this checks that calling analyzePattern is not needed if the pattern do not change. + if (solver.info() != Success) { std::cerr << "sparse QR factorization failed\n"; exit(0); return; } x = solver.solve(b); - if (solver.info() != Success) - { + if (solver.info() != Success) { std::cerr << "sparse QR factorization failed\n"; exit(0); return; } - + VERIFY_IS_APPROX(A * x, b); - - //Compare with a dense QR solver + + // Compare with a dense QR solver ColPivHouseholderQR dqr(dA); refX = dqr.solve(b); - + VERIFY_IS_EQUAL(dqr.rank(), solver.rank()); - if(solver.rank()==A.cols()) // full rank + if (solver.rank() == A.cols())// full rank VERIFY_IS_APPROX(x, refX); -// else -// VERIFY((dA * refX - b).norm() * 2 > (A * x - b).norm() ); + // else + // VERIFY((dA * refX - b).norm() * 2 > (A * x - b).norm() ); // Compute explicitly the matrix Q MatrixType Q, QtQ, idM; Q = solver.matrixQ(); - //Check ||Q' * Q - I || + // Check ||Q' * Q - I || QtQ = Q * Q.adjoint(); - idM.resize(Q.rows(), Q.rows()); idM.setIdentity(); + idM.resize(Q.rows(), Q.rows()); + idM.setIdentity(); VERIFY(idM.isApprox(QtQ)); - + // Q to dense DenseMat dQ; dQ = solver.matrixQ(); @@ -119,10 +113,8 @@ template void test_sparseqr_scalar() } void test_sparseqr() { - for(int i=0; i()); - CALL_SUBTEST_2(test_sparseqr_scalar >()); + CALL_SUBTEST_2(test_sparseqr_scalar>()); } } - diff --git a/filmulator-gui/core/nlmeans/eigen/test/special_numbers.cpp b/filmulator-gui/core/nlmeans/eigen/test/special_numbers.cpp index 2f1b704b..d02c99fd 100644 --- a/filmulator-gui/core/nlmeans/eigen/test/special_numbers.cpp +++ b/filmulator-gui/core/nlmeans/eigen/test/special_numbers.cpp @@ -11,48 +11,45 @@ template void special_numbers() { - typedef Matrix MatType; - int rows = internal::random(1,300); - int cols = internal::random(1,300); - + typedef Matrix MatType; + int rows = internal::random(1, 300); + int cols = internal::random(1, 300); + Scalar nan = std::numeric_limits::quiet_NaN(); Scalar inf = std::numeric_limits::infinity(); Scalar s1 = internal::random(); - - MatType m1 = MatType::Random(rows,cols), - mnan = MatType::Random(rows,cols), - minf = MatType::Random(rows,cols), - mboth = MatType::Random(rows,cols); - - int n = internal::random(1,10); - for(int k=0; k(0,rows-1), internal::random(0,cols-1)) = nan; - minf(internal::random(0,rows-1), internal::random(0,cols-1)) = inf; + + MatType m1 = MatType::Random(rows, cols), mnan = MatType::Random(rows, cols), minf = MatType::Random(rows, cols), + mboth = MatType::Random(rows, cols); + + int n = internal::random(1, 10); + for (int k = 0; k < n; ++k) { + mnan(internal::random(0, rows - 1), internal::random(0, cols - 1)) = nan; + minf(internal::random(0, rows - 1), internal::random(0, cols - 1)) = inf; } mboth = mnan + minf; - + VERIFY(!m1.hasNaN()); VERIFY(m1.allFinite()); - + VERIFY(mnan.hasNaN()); - VERIFY((s1*mnan).hasNaN()); + VERIFY((s1 * mnan).hasNaN()); VERIFY(!minf.hasNaN()); - VERIFY(!(2*minf).hasNaN()); + VERIFY(!(2 * minf).hasNaN()); VERIFY(mboth.hasNaN()); VERIFY(mboth.array().hasNaN()); - + VERIFY(!mnan.allFinite()); VERIFY(!minf.allFinite()); - VERIFY(!(minf-mboth).allFinite()); + VERIFY(!(minf - mboth).allFinite()); VERIFY(!mboth.allFinite()); VERIFY(!mboth.array().allFinite()); } void test_special_numbers() { - for(int i = 0; i < 10*g_repeat; i++) { - CALL_SUBTEST_1( special_numbers() ); - CALL_SUBTEST_1( special_numbers() ); + for (int i = 0; i < 10 * g_repeat; i++) { + CALL_SUBTEST_1(special_numbers()); + CALL_SUBTEST_1(special_numbers()); } } diff --git a/filmulator-gui/core/nlmeans/eigen/test/spqr_support.cpp b/filmulator-gui/core/nlmeans/eigen/test/spqr_support.cpp index 81e63b6a..4e14cec7 100644 --- a/filmulator-gui/core/nlmeans/eigen/test/spqr_support.cpp +++ b/filmulator-gui/core/nlmeans/eigen/test/spqr_support.cpp @@ -11,54 +11,52 @@ #include -template -int generate_sparse_rectangular_problem(MatrixType& A, DenseMat& dA, int maxRows = 300, int maxCols = 300) +template +int generate_sparse_rectangular_problem(MatrixType &A, DenseMat &dA, int maxRows = 300, int maxCols = 300) { eigen_assert(maxRows >= maxCols); typedef typename MatrixType::Scalar Scalar; - int rows = internal::random(1,maxRows); - int cols = internal::random(1,rows); - double density = (std::max)(8./(rows*cols), 0.01); - - A.resize(rows,cols); - dA.resize(rows,cols); - initSparse(density, dA, A,ForceNonZeroDiag); + int rows = internal::random(1, maxRows); + int cols = internal::random(1, rows); + double density = (std::max)(8. / (rows * cols), 0.01); + + A.resize(rows, cols); + dA.resize(rows, cols); + initSparse(density, dA, A, ForceNonZeroDiag); A.makeCompressed(); return rows; } template void test_spqr_scalar() { - typedef SparseMatrix MatrixType; + typedef SparseMatrix MatrixType; MatrixType A; - Matrix dA; - typedef Matrix DenseVector; - DenseVector refX,x,b; - SPQR solver; - generate_sparse_rectangular_problem(A,dA); - + Matrix dA; + typedef Matrix DenseVector; + DenseVector refX, x, b; + SPQR solver; + generate_sparse_rectangular_problem(A, dA); + Index m = A.rows(); b = DenseVector::Random(m); solver.compute(A); - if (solver.info() != Success) - { + if (solver.info() != Success) { std::cerr << "sparse QR factorization failed\n"; exit(0); return; } x = solver.solve(b); - if (solver.info() != Success) - { + if (solver.info() != Success) { std::cerr << "sparse QR factorization failed\n"; exit(0); return; - } - //Compare with a dense solver + } + // Compare with a dense solver refX = dA.colPivHouseholderQr().solve(b); - VERIFY(x.isApprox(refX,test_precision())); + VERIFY(x.isApprox(refX, test_precision())); } void test_spqr_support() { CALL_SUBTEST_1(test_spqr_scalar()); - CALL_SUBTEST_2(test_spqr_scalar >()); + CALL_SUBTEST_2(test_spqr_scalar>()); } diff --git a/filmulator-gui/core/nlmeans/eigen/test/stable_norm.cpp b/filmulator-gui/core/nlmeans/eigen/test/stable_norm.cpp index ac8b1291..9497842b 100644 --- a/filmulator-gui/core/nlmeans/eigen/test/stable_norm.cpp +++ b/filmulator-gui/core/nlmeans/eigen/test/stable_norm.cpp @@ -9,12 +9,9 @@ #include "main.h" -template EIGEN_DONT_INLINE T copy(const T& x) -{ - return x; -} +template EIGEN_DONT_INLINE T copy(const T &x) { return x; } -template void stable_norm(const MatrixType& m) +template void stable_norm(const MatrixType &m) { /* this test covers the following files: StableNorm.h @@ -23,28 +20,28 @@ template void stable_norm(const MatrixType& m) using std::abs; typedef typename MatrixType::Scalar Scalar; typedef typename NumTraits::Real RealScalar; - + bool complex_real_product_ok = true; // Check the basic machine-dependent constants. { int ibeta, it, iemin, iemax; - ibeta = std::numeric_limits::radix; // base for floating-point numbers - it = std::numeric_limits::digits; // number of base-beta digits in mantissa - iemin = std::numeric_limits::min_exponent; // minimum exponent - iemax = std::numeric_limits::max_exponent; // maximum exponent + ibeta = std::numeric_limits::radix;// base for floating-point numbers + it = std::numeric_limits::digits;// number of base-beta digits in mantissa + iemin = std::numeric_limits::min_exponent;// minimum exponent + iemax = std::numeric_limits::max_exponent;// maximum exponent - VERIFY( (!(iemin > 1 - 2*it || 1+it>iemax || (it==2 && ibeta<5) || (it<=4 && ibeta <= 3 ) || it<2)) + VERIFY((!(iemin > 1 - 2 * it || 1 + it > iemax || (it == 2 && ibeta < 5) || (it <= 4 && ibeta <= 3) || it < 2)) && "the stable norm algorithm cannot be guaranteed on this computer"); - + Scalar inf = std::numeric_limits::infinity(); - if(NumTraits::IsComplex && (numext::isnan)(inf*RealScalar(1)) ) - { + if (NumTraits::IsComplex && (numext::isnan)(inf * RealScalar(1))) { complex_real_product_ok = false; static bool first = true; - if(first) - std::cerr << "WARNING: compiler mess up complex*real product, " << inf << " * " << 1.0 << " = " << inf*RealScalar(1) << std::endl; + if (first) + std::cerr << "WARNING: compiler mess up complex*real product, " << inf << " * " << 1.0 << " = " + << inf * RealScalar(1) << std::endl; first = false; } } @@ -55,122 +52,132 @@ template void stable_norm(const MatrixType& m) // get a non-zero random factor Scalar factor = internal::random(); - while(numext::abs2(factor)(); + while (numext::abs2(factor) < RealScalar(1e-4)) factor = internal::random(); Scalar big = factor * ((std::numeric_limits::max)() * RealScalar(1e-4)); - + factor = internal::random(); - while(numext::abs2(factor)(); + while (numext::abs2(factor) < RealScalar(1e-4)) factor = internal::random(); Scalar small = factor * ((std::numeric_limits::min)() * RealScalar(1e4)); Scalar one(1); - MatrixType vzero = MatrixType::Zero(rows, cols), - vrand = MatrixType::Random(rows, cols), - vbig(rows, cols), - vsmall(rows,cols); + MatrixType vzero = MatrixType::Zero(rows, cols), vrand = MatrixType::Random(rows, cols), vbig(rows, cols), + vsmall(rows, cols); vbig.fill(big); vsmall.fill(small); VERIFY_IS_MUCH_SMALLER_THAN(vzero.norm(), static_cast(1)); - VERIFY_IS_APPROX(vrand.stableNorm(), vrand.norm()); - VERIFY_IS_APPROX(vrand.blueNorm(), vrand.norm()); - VERIFY_IS_APPROX(vrand.hypotNorm(), vrand.norm()); + VERIFY_IS_APPROX(vrand.stableNorm(), vrand.norm()); + VERIFY_IS_APPROX(vrand.blueNorm(), vrand.norm()); + VERIFY_IS_APPROX(vrand.hypotNorm(), vrand.norm()); // test with expressions as input - VERIFY_IS_APPROX((one*vrand).stableNorm(), vrand.norm()); - VERIFY_IS_APPROX((one*vrand).blueNorm(), vrand.norm()); - VERIFY_IS_APPROX((one*vrand).hypotNorm(), vrand.norm()); - VERIFY_IS_APPROX((one*vrand+one*vrand-one*vrand).stableNorm(), vrand.norm()); - VERIFY_IS_APPROX((one*vrand+one*vrand-one*vrand).blueNorm(), vrand.norm()); - VERIFY_IS_APPROX((one*vrand+one*vrand-one*vrand).hypotNorm(), vrand.norm()); + VERIFY_IS_APPROX((one * vrand).stableNorm(), vrand.norm()); + VERIFY_IS_APPROX((one * vrand).blueNorm(), vrand.norm()); + VERIFY_IS_APPROX((one * vrand).hypotNorm(), vrand.norm()); + VERIFY_IS_APPROX((one * vrand + one * vrand - one * vrand).stableNorm(), vrand.norm()); + VERIFY_IS_APPROX((one * vrand + one * vrand - one * vrand).blueNorm(), vrand.norm()); + VERIFY_IS_APPROX((one * vrand + one * vrand - one * vrand).hypotNorm(), vrand.norm()); RealScalar size = static_cast(m.size()); // test numext::isfinite - VERIFY(!(numext::isfinite)( std::numeric_limits::infinity())); + VERIFY(!(numext::isfinite)(std::numeric_limits::infinity())); VERIFY(!(numext::isfinite)(sqrt(-abs(big)))); // test overflow - VERIFY((numext::isfinite)(sqrt(size)*abs(big))); - VERIFY_IS_NOT_APPROX(sqrt(copy(vbig.squaredNorm())), abs(sqrt(size)*big)); // here the default norm must fail - VERIFY_IS_APPROX(vbig.stableNorm(), sqrt(size)*abs(big)); - VERIFY_IS_APPROX(vbig.blueNorm(), sqrt(size)*abs(big)); - VERIFY_IS_APPROX(vbig.hypotNorm(), sqrt(size)*abs(big)); + VERIFY((numext::isfinite)(sqrt(size) * abs(big))); + VERIFY_IS_NOT_APPROX(sqrt(copy(vbig.squaredNorm())), abs(sqrt(size) * big));// here the default norm must fail + VERIFY_IS_APPROX(vbig.stableNorm(), sqrt(size) * abs(big)); + VERIFY_IS_APPROX(vbig.blueNorm(), sqrt(size) * abs(big)); + VERIFY_IS_APPROX(vbig.hypotNorm(), sqrt(size) * abs(big)); // test underflow - VERIFY((numext::isfinite)(sqrt(size)*abs(small))); - VERIFY_IS_NOT_APPROX(sqrt(copy(vsmall.squaredNorm())), abs(sqrt(size)*small)); // here the default norm must fail - VERIFY_IS_APPROX(vsmall.stableNorm(), sqrt(size)*abs(small)); - VERIFY_IS_APPROX(vsmall.blueNorm(), sqrt(size)*abs(small)); - VERIFY_IS_APPROX(vsmall.hypotNorm(), sqrt(size)*abs(small)); + VERIFY((numext::isfinite)(sqrt(size) * abs(small))); + VERIFY_IS_NOT_APPROX(sqrt(copy(vsmall.squaredNorm())), abs(sqrt(size) * small));// here the default norm must fail + VERIFY_IS_APPROX(vsmall.stableNorm(), sqrt(size) * abs(small)); + VERIFY_IS_APPROX(vsmall.blueNorm(), sqrt(size) * abs(small)); + VERIFY_IS_APPROX(vsmall.hypotNorm(), sqrt(size) * abs(small)); // Test compilation of cwise() version - VERIFY_IS_APPROX(vrand.colwise().stableNorm(), vrand.colwise().norm()); - VERIFY_IS_APPROX(vrand.colwise().blueNorm(), vrand.colwise().norm()); - VERIFY_IS_APPROX(vrand.colwise().hypotNorm(), vrand.colwise().norm()); - VERIFY_IS_APPROX(vrand.rowwise().stableNorm(), vrand.rowwise().norm()); - VERIFY_IS_APPROX(vrand.rowwise().blueNorm(), vrand.rowwise().norm()); - VERIFY_IS_APPROX(vrand.rowwise().hypotNorm(), vrand.rowwise().norm()); - - // test NaN, +inf, -inf + VERIFY_IS_APPROX(vrand.colwise().stableNorm(), vrand.colwise().norm()); + VERIFY_IS_APPROX(vrand.colwise().blueNorm(), vrand.colwise().norm()); + VERIFY_IS_APPROX(vrand.colwise().hypotNorm(), vrand.colwise().norm()); + VERIFY_IS_APPROX(vrand.rowwise().stableNorm(), vrand.rowwise().norm()); + VERIFY_IS_APPROX(vrand.rowwise().blueNorm(), vrand.rowwise().norm()); + VERIFY_IS_APPROX(vrand.rowwise().hypotNorm(), vrand.rowwise().norm()); + + // test NaN, +inf, -inf MatrixType v; - Index i = internal::random(0,rows-1); - Index j = internal::random(0,cols-1); + Index i = internal::random(0, rows - 1); + Index j = internal::random(0, cols - 1); // NaN { v = vrand; - v(i,j) = std::numeric_limits::quiet_NaN(); - VERIFY(!(numext::isfinite)(v.squaredNorm())); VERIFY((numext::isnan)(v.squaredNorm())); - VERIFY(!(numext::isfinite)(v.norm())); VERIFY((numext::isnan)(v.norm())); - VERIFY(!(numext::isfinite)(v.stableNorm())); VERIFY((numext::isnan)(v.stableNorm())); - VERIFY(!(numext::isfinite)(v.blueNorm())); VERIFY((numext::isnan)(v.blueNorm())); - VERIFY(!(numext::isfinite)(v.hypotNorm())); VERIFY((numext::isnan)(v.hypotNorm())); + v(i, j) = std::numeric_limits::quiet_NaN(); + VERIFY(!(numext::isfinite)(v.squaredNorm())); + VERIFY((numext::isnan)(v.squaredNorm())); + VERIFY(!(numext::isfinite)(v.norm())); + VERIFY((numext::isnan)(v.norm())); + VERIFY(!(numext::isfinite)(v.stableNorm())); + VERIFY((numext::isnan)(v.stableNorm())); + VERIFY(!(numext::isfinite)(v.blueNorm())); + VERIFY((numext::isnan)(v.blueNorm())); + VERIFY(!(numext::isfinite)(v.hypotNorm())); + VERIFY((numext::isnan)(v.hypotNorm())); } - + // +inf { v = vrand; - v(i,j) = std::numeric_limits::infinity(); - VERIFY(!(numext::isfinite)(v.squaredNorm())); VERIFY(isPlusInf(v.squaredNorm())); - VERIFY(!(numext::isfinite)(v.norm())); VERIFY(isPlusInf(v.norm())); + v(i, j) = std::numeric_limits::infinity(); + VERIFY(!(numext::isfinite)(v.squaredNorm())); + VERIFY(isPlusInf(v.squaredNorm())); + VERIFY(!(numext::isfinite)(v.norm())); + VERIFY(isPlusInf(v.norm())); VERIFY(!(numext::isfinite)(v.stableNorm())); - if(complex_real_product_ok){ - VERIFY(isPlusInf(v.stableNorm())); - } - VERIFY(!(numext::isfinite)(v.blueNorm())); VERIFY(isPlusInf(v.blueNorm())); - VERIFY(!(numext::isfinite)(v.hypotNorm())); VERIFY(isPlusInf(v.hypotNorm())); + if (complex_real_product_ok) { VERIFY(isPlusInf(v.stableNorm())); } + VERIFY(!(numext::isfinite)(v.blueNorm())); + VERIFY(isPlusInf(v.blueNorm())); + VERIFY(!(numext::isfinite)(v.hypotNorm())); + VERIFY(isPlusInf(v.hypotNorm())); } - + // -inf { v = vrand; - v(i,j) = -std::numeric_limits::infinity(); - VERIFY(!(numext::isfinite)(v.squaredNorm())); VERIFY(isPlusInf(v.squaredNorm())); - VERIFY(!(numext::isfinite)(v.norm())); VERIFY(isPlusInf(v.norm())); + v(i, j) = -std::numeric_limits::infinity(); + VERIFY(!(numext::isfinite)(v.squaredNorm())); + VERIFY(isPlusInf(v.squaredNorm())); + VERIFY(!(numext::isfinite)(v.norm())); + VERIFY(isPlusInf(v.norm())); VERIFY(!(numext::isfinite)(v.stableNorm())); - if(complex_real_product_ok) { - VERIFY(isPlusInf(v.stableNorm())); - } - VERIFY(!(numext::isfinite)(v.blueNorm())); VERIFY(isPlusInf(v.blueNorm())); - VERIFY(!(numext::isfinite)(v.hypotNorm())); VERIFY(isPlusInf(v.hypotNorm())); + if (complex_real_product_ok) { VERIFY(isPlusInf(v.stableNorm())); } + VERIFY(!(numext::isfinite)(v.blueNorm())); + VERIFY(isPlusInf(v.blueNorm())); + VERIFY(!(numext::isfinite)(v.hypotNorm())); + VERIFY(isPlusInf(v.hypotNorm())); } - + // mix { - Index i2 = internal::random(0,rows-1); - Index j2 = internal::random(0,cols-1); + Index i2 = internal::random(0, rows - 1); + Index j2 = internal::random(0, cols - 1); v = vrand; - v(i,j) = -std::numeric_limits::infinity(); - v(i2,j2) = std::numeric_limits::quiet_NaN(); - VERIFY(!(numext::isfinite)(v.squaredNorm())); VERIFY((numext::isnan)(v.squaredNorm())); - VERIFY(!(numext::isfinite)(v.norm())); VERIFY((numext::isnan)(v.norm())); - VERIFY(!(numext::isfinite)(v.stableNorm())); VERIFY((numext::isnan)(v.stableNorm())); - VERIFY(!(numext::isfinite)(v.blueNorm())); VERIFY((numext::isnan)(v.blueNorm())); - VERIFY(!(numext::isfinite)(v.hypotNorm())); VERIFY((numext::isnan)(v.hypotNorm())); + v(i, j) = -std::numeric_limits::infinity(); + v(i2, j2) = std::numeric_limits::quiet_NaN(); + VERIFY(!(numext::isfinite)(v.squaredNorm())); + VERIFY((numext::isnan)(v.squaredNorm())); + VERIFY(!(numext::isfinite)(v.norm())); + VERIFY((numext::isnan)(v.norm())); + VERIFY(!(numext::isfinite)(v.stableNorm())); + VERIFY((numext::isnan)(v.stableNorm())); + VERIFY(!(numext::isfinite)(v.blueNorm())); + VERIFY((numext::isnan)(v.blueNorm())); + VERIFY(!(numext::isfinite)(v.hypotNorm())); + VERIFY((numext::isnan)(v.hypotNorm())); } // stableNormalize[d] @@ -184,18 +191,18 @@ template void stable_norm(const MatrixType& m) VERIFY_IS_APPROX((vbig.stableNormalized()).norm(), RealScalar(1)); VERIFY_IS_APPROX((vsmall.stableNormalized()).norm(), RealScalar(1)); RealScalar big_scaling = ((std::numeric_limits::max)() * RealScalar(1e-4)); - VERIFY_IS_APPROX(vbig/big_scaling, (vbig.stableNorm() * vbig.stableNormalized()).eval()/big_scaling); + VERIFY_IS_APPROX(vbig / big_scaling, (vbig.stableNorm() * vbig.stableNormalized()).eval() / big_scaling); VERIFY_IS_APPROX(vsmall, vsmall.stableNorm() * vsmall.stableNormalized()); } } void test_stable_norm() { - for(int i = 0; i < g_repeat; i++) { - CALL_SUBTEST_1( stable_norm(Matrix()) ); - CALL_SUBTEST_2( stable_norm(Vector4d()) ); - CALL_SUBTEST_3( stable_norm(VectorXd(internal::random(10,2000))) ); - CALL_SUBTEST_4( stable_norm(VectorXf(internal::random(10,2000))) ); - CALL_SUBTEST_5( stable_norm(VectorXcd(internal::random(10,2000))) ); + for (int i = 0; i < g_repeat; i++) { + CALL_SUBTEST_1(stable_norm(Matrix())); + CALL_SUBTEST_2(stable_norm(Vector4d())); + CALL_SUBTEST_3(stable_norm(VectorXd(internal::random(10, 2000)))); + CALL_SUBTEST_4(stable_norm(VectorXf(internal::random(10, 2000)))); + CALL_SUBTEST_5(stable_norm(VectorXcd(internal::random(10, 2000)))); } } diff --git a/filmulator-gui/core/nlmeans/eigen/test/stddeque.cpp b/filmulator-gui/core/nlmeans/eigen/test/stddeque.cpp index b511c4e6..2b33851d 100644 --- a/filmulator-gui/core/nlmeans/eigen/test/stddeque.cpp +++ b/filmulator-gui/core/nlmeans/eigen/test/stddeque.cpp @@ -9,54 +9,50 @@ // with this file, You can obtain one at http://mozilla.org/MPL/2.0/. #include "main.h" -#include #include +#include -template -void check_stddeque_matrix(const MatrixType& m) +template void check_stddeque_matrix(const MatrixType &m) { Index rows = m.rows(); Index cols = m.cols(); - MatrixType x = MatrixType::Random(rows,cols), y = MatrixType::Random(rows,cols); - std::deque > v(10, MatrixType(rows,cols)), w(20, y); + MatrixType x = MatrixType::Random(rows, cols), y = MatrixType::Random(rows, cols); + std::deque> v(10, MatrixType(rows, cols)), w(20, y); v.front() = x; w.front() = w.back(); VERIFY_IS_APPROX(w.front(), w.back()); v = w; - typename std::deque >::iterator vi = v.begin(); - typename std::deque >::iterator wi = w.begin(); - for(int i = 0; i < 20; i++) - { + typename std::deque>::iterator vi = v.begin(); + typename std::deque>::iterator wi = w.begin(); + for (int i = 0; i < 20; i++) { VERIFY_IS_APPROX(*vi, *wi); ++vi; ++wi; } - v.resize(21); + v.resize(21); v.back() = x; VERIFY_IS_APPROX(v.back(), x); - v.resize(22,y); + v.resize(22, y); VERIFY_IS_APPROX(v.back(), y); v.push_back(x); VERIFY_IS_APPROX(v.back(), x); } -template -void check_stddeque_transform(const TransformType&) +template void check_stddeque_transform(const TransformType &) { typedef typename TransformType::MatrixType MatrixType; TransformType x(MatrixType::Random()), y(MatrixType::Random()); - std::deque > v(10), w(20, y); + std::deque> v(10), w(20, y); v.front() = x; w.front() = w.back(); VERIFY_IS_APPROX(w.front(), w.back()); v = w; - typename std::deque >::iterator vi = v.begin(); - typename std::deque >::iterator wi = w.begin(); - for(int i = 0; i < 20; i++) - { + typename std::deque>::iterator vi = v.begin(); + typename std::deque>::iterator wi = w.begin(); + for (int i = 0; i < 20; i++) { VERIFY_IS_APPROX(*vi, *wi); ++vi; ++wi; @@ -65,27 +61,25 @@ void check_stddeque_transform(const TransformType&) v.resize(21); v.back() = x; VERIFY_IS_APPROX(v.back(), x); - v.resize(22,y); + v.resize(22, y); VERIFY_IS_APPROX(v.back(), y); v.push_back(x); VERIFY_IS_APPROX(v.back(), x); } -template -void check_stddeque_quaternion(const QuaternionType&) +template void check_stddeque_quaternion(const QuaternionType &) { typedef typename QuaternionType::Coefficients Coefficients; QuaternionType x(Coefficients::Random()), y(Coefficients::Random()); - std::deque > v(10), w(20, y); + std::deque> v(10), w(20, y); v.front() = x; w.front() = w.back(); VERIFY_IS_APPROX(w.front(), w.back()); v = w; - typename std::deque >::iterator vi = v.begin(); - typename std::deque >::iterator wi = w.begin(); - for(int i = 0; i < 20; i++) - { + typename std::deque>::iterator vi = v.begin(); + typename std::deque>::iterator wi = w.begin(); + for (int i = 0; i < 20; i++) { VERIFY_IS_APPROX(*vi, *wi); ++vi; ++wi; @@ -94,7 +88,7 @@ void check_stddeque_quaternion(const QuaternionType&) v.resize(21); v.back() = x; VERIFY_IS_APPROX(v.back(), x); - v.resize(22,y); + v.resize(22, y); VERIFY_IS_APPROX(v.back(), y); v.push_back(x); VERIFY_IS_APPROX(v.back(), x); @@ -114,10 +108,10 @@ void test_stddeque() CALL_SUBTEST_2(check_stddeque_matrix(Matrix4d())); // some dynamic sizes - CALL_SUBTEST_3(check_stddeque_matrix(MatrixXd(1,1))); + CALL_SUBTEST_3(check_stddeque_matrix(MatrixXd(1, 1))); CALL_SUBTEST_3(check_stddeque_matrix(VectorXd(20))); CALL_SUBTEST_3(check_stddeque_matrix(RowVectorXf(20))); - CALL_SUBTEST_3(check_stddeque_matrix(MatrixXcf(10,10))); + CALL_SUBTEST_3(check_stddeque_matrix(MatrixXcf(10, 10))); // some Transform CALL_SUBTEST_4(check_stddeque_transform(Affine2f())); diff --git a/filmulator-gui/core/nlmeans/eigen/test/stddeque_overload.cpp b/filmulator-gui/core/nlmeans/eigen/test/stddeque_overload.cpp index 4da618bb..a534364f 100644 --- a/filmulator-gui/core/nlmeans/eigen/test/stddeque_overload.cpp +++ b/filmulator-gui/core/nlmeans/eigen/test/stddeque_overload.cpp @@ -10,8 +10,8 @@ #include "main.h" -#include #include +#include EIGEN_DEFINE_STL_DEQUE_SPECIALIZATION(Vector4f) @@ -25,43 +25,34 @@ EIGEN_DEFINE_STL_DEQUE_SPECIALIZATION(Affine3d) EIGEN_DEFINE_STL_DEQUE_SPECIALIZATION(Quaternionf) EIGEN_DEFINE_STL_DEQUE_SPECIALIZATION(Quaterniond) -template -void check_stddeque_matrix(const MatrixType& m) +template void check_stddeque_matrix(const MatrixType &m) { typename MatrixType::Index rows = m.rows(); typename MatrixType::Index cols = m.cols(); - MatrixType x = MatrixType::Random(rows,cols), y = MatrixType::Random(rows,cols); - std::deque v(10, MatrixType(rows,cols)), w(20, y); + MatrixType x = MatrixType::Random(rows, cols), y = MatrixType::Random(rows, cols); + std::deque v(10, MatrixType(rows, cols)), w(20, y); v[5] = x; w[6] = v[5]; VERIFY_IS_APPROX(w[6], v[5]); v = w; - for(int i = 0; i < 20; i++) - { - VERIFY_IS_APPROX(w[i], v[i]); - } + for (int i = 0; i < 20; i++) { VERIFY_IS_APPROX(w[i], v[i]); } v.resize(21); v[20] = x; VERIFY_IS_APPROX(v[20], x); - v.resize(22,y); + v.resize(22, y); VERIFY_IS_APPROX(v[21], y); v.push_back(x); VERIFY_IS_APPROX(v[22], x); // do a lot of push_back such that the deque gets internally resized // (with memory reallocation) - MatrixType* ref = &w[0]; - for(int i=0; i<30 || ((ref==&w[0]) && i<300); ++i) - v.push_back(w[i%w.size()]); - for(unsigned int i=23; i -void check_stddeque_transform(const TransformType&) +template void check_stddeque_transform(const TransformType &) { typedef typename TransformType::MatrixType MatrixType; TransformType x(MatrixType::Random()), y(MatrixType::Random()); @@ -70,32 +61,24 @@ void check_stddeque_transform(const TransformType&) w[6] = v[5]; VERIFY_IS_APPROX(w[6], v[5]); v = w; - for(int i = 0; i < 20; i++) - { - VERIFY_IS_APPROX(w[i], v[i]); - } + for (int i = 0; i < 20; i++) { VERIFY_IS_APPROX(w[i], v[i]); } v.resize(21); v[20] = x; VERIFY_IS_APPROX(v[20], x); - v.resize(22,y); + v.resize(22, y); VERIFY_IS_APPROX(v[21], y); v.push_back(x); VERIFY_IS_APPROX(v[22], x); // do a lot of push_back such that the deque gets internally resized // (with memory reallocation) - TransformType* ref = &w[0]; - for(int i=0; i<30 || ((ref==&w[0]) && i<300); ++i) - v.push_back(w[i%w.size()]); - for(unsigned int i=23; i -void check_stddeque_quaternion(const QuaternionType&) +template void check_stddeque_quaternion(const QuaternionType &) { typedef typename QuaternionType::Coefficients Coefficients; QuaternionType x(Coefficients::Random()), y(Coefficients::Random()); @@ -104,28 +87,21 @@ void check_stddeque_quaternion(const QuaternionType&) w[6] = v[5]; VERIFY_IS_APPROX(w[6], v[5]); v = w; - for(int i = 0; i < 20; i++) - { - VERIFY_IS_APPROX(w[i], v[i]); - } + for (int i = 0; i < 20; i++) { VERIFY_IS_APPROX(w[i], v[i]); } v.resize(21); v[20] = x; VERIFY_IS_APPROX(v[20], x); - v.resize(22,y); + v.resize(22, y); VERIFY_IS_APPROX(v[21], y); v.push_back(x); VERIFY_IS_APPROX(v[22], x); // do a lot of push_back such that the deque gets internally resized // (with memory reallocation) - QuaternionType* ref = &w[0]; - for(int i=0; i<30 || ((ref==&w[0]) && i<300); ++i) - v.push_back(w[i%w.size()]); - for(unsigned int i=23; i #include +#include -template -void check_stdlist_matrix(const MatrixType& m) +template void check_stdlist_matrix(const MatrixType &m) { Index rows = m.rows(); Index cols = m.cols(); - MatrixType x = MatrixType::Random(rows,cols), y = MatrixType::Random(rows,cols); - std::list > v(10, MatrixType(rows,cols)), w(20, y); + MatrixType x = MatrixType::Random(rows, cols), y = MatrixType::Random(rows, cols); + std::list> v(10, MatrixType(rows, cols)), w(20, y); v.front() = x; w.front() = w.back(); VERIFY_IS_APPROX(w.front(), w.back()); v = w; - typename std::list >::iterator vi = v.begin(); - typename std::list >::iterator wi = w.begin(); - for(int i = 0; i < 20; i++) - { + typename std::list>::iterator vi = v.begin(); + typename std::list>::iterator wi = w.begin(); + for (int i = 0; i < 20; i++) { VERIFY_IS_APPROX(*vi, *wi); ++vi; ++wi; } - v.resize(21); + v.resize(21); v.back() = x; VERIFY_IS_APPROX(v.back(), x); - v.resize(22,y); + v.resize(22, y); VERIFY_IS_APPROX(v.back(), y); v.push_back(x); VERIFY_IS_APPROX(v.back(), x); } -template -void check_stdlist_transform(const TransformType&) +template void check_stdlist_transform(const TransformType &) { typedef typename TransformType::MatrixType MatrixType; TransformType x(MatrixType::Random()), y(MatrixType::Random()); - std::list > v(10), w(20, y); + std::list> v(10), w(20, y); v.front() = x; w.front() = w.back(); VERIFY_IS_APPROX(w.front(), w.back()); v = w; - typename std::list >::iterator vi = v.begin(); - typename std::list >::iterator wi = w.begin(); - for(int i = 0; i < 20; i++) - { + typename std::list>::iterator vi = v.begin(); + typename std::list>::iterator wi = w.begin(); + for (int i = 0; i < 20; i++) { VERIFY_IS_APPROX(*vi, *wi); ++vi; ++wi; @@ -65,27 +61,25 @@ void check_stdlist_transform(const TransformType&) v.resize(21); v.back() = x; VERIFY_IS_APPROX(v.back(), x); - v.resize(22,y); + v.resize(22, y); VERIFY_IS_APPROX(v.back(), y); v.push_back(x); VERIFY_IS_APPROX(v.back(), x); } -template -void check_stdlist_quaternion(const QuaternionType&) +template void check_stdlist_quaternion(const QuaternionType &) { typedef typename QuaternionType::Coefficients Coefficients; QuaternionType x(Coefficients::Random()), y(Coefficients::Random()); - std::list > v(10), w(20, y); + std::list> v(10), w(20, y); v.front() = x; w.front() = w.back(); VERIFY_IS_APPROX(w.front(), w.back()); v = w; - typename std::list >::iterator vi = v.begin(); - typename std::list >::iterator wi = w.begin(); - for(int i = 0; i < 20; i++) - { + typename std::list>::iterator vi = v.begin(); + typename std::list>::iterator wi = w.begin(); + for (int i = 0; i < 20; i++) { VERIFY_IS_APPROX(*vi, *wi); ++vi; ++wi; @@ -94,7 +88,7 @@ void check_stdlist_quaternion(const QuaternionType&) v.resize(21); v.back() = x; VERIFY_IS_APPROX(v.back(), x); - v.resize(22,y); + v.resize(22, y); VERIFY_IS_APPROX(v.back(), y); v.push_back(x); VERIFY_IS_APPROX(v.back(), x); @@ -114,10 +108,10 @@ void test_stdlist() CALL_SUBTEST_2(check_stdlist_matrix(Matrix4d())); // some dynamic sizes - CALL_SUBTEST_3(check_stdlist_matrix(MatrixXd(1,1))); + CALL_SUBTEST_3(check_stdlist_matrix(MatrixXd(1, 1))); CALL_SUBTEST_3(check_stdlist_matrix(VectorXd(20))); CALL_SUBTEST_3(check_stdlist_matrix(RowVectorXf(20))); - CALL_SUBTEST_3(check_stdlist_matrix(MatrixXcf(10,10))); + CALL_SUBTEST_3(check_stdlist_matrix(MatrixXcf(10, 10))); // some Transform CALL_SUBTEST_4(check_stdlist_transform(Affine2f())); diff --git a/filmulator-gui/core/nlmeans/eigen/test/stdlist_overload.cpp b/filmulator-gui/core/nlmeans/eigen/test/stdlist_overload.cpp index bb910bd4..4d146514 100644 --- a/filmulator-gui/core/nlmeans/eigen/test/stdlist_overload.cpp +++ b/filmulator-gui/core/nlmeans/eigen/test/stdlist_overload.cpp @@ -10,8 +10,8 @@ #include "main.h" -#include #include +#include EIGEN_DEFINE_STL_LIST_SPECIALIZATION(Vector4f) @@ -25,29 +25,26 @@ EIGEN_DEFINE_STL_LIST_SPECIALIZATION(Affine3d) EIGEN_DEFINE_STL_LIST_SPECIALIZATION(Quaternionf) EIGEN_DEFINE_STL_LIST_SPECIALIZATION(Quaterniond) -template -typename Container::iterator get(Container & c, Position position) +template typename Container::iterator get(Container &c, Position position) { typename Container::iterator it = c.begin(); std::advance(it, position); return it; } -template -void set(Container & c, Position position, const Value & value) +template void set(Container &c, Position position, const Value &value) { typename Container::iterator it = c.begin(); std::advance(it, position); *it = value; } -template -void check_stdlist_matrix(const MatrixType& m) +template void check_stdlist_matrix(const MatrixType &m) { typename MatrixType::Index rows = m.rows(); typename MatrixType::Index cols = m.cols(); - MatrixType x = MatrixType::Random(rows,cols), y = MatrixType::Random(rows,cols); - std::list v(10, MatrixType(rows,cols)), w(20, y); + MatrixType x = MatrixType::Random(rows, cols), y = MatrixType::Random(rows, cols); + std::list v(10, MatrixType(rows, cols)), w(20, y); typename std::list::iterator itv = get(v, 5); typename std::list::iterator itw = get(w, 6); *itv = x; @@ -56,8 +53,7 @@ void check_stdlist_matrix(const MatrixType& m) v = w; itv = v.begin(); itw = w.begin(); - for(int i = 0; i < 20; i++) - { + for (int i = 0; i < 20; i++) { VERIFY_IS_APPROX(*itw, *itv); ++itv; ++itw; @@ -66,24 +62,19 @@ void check_stdlist_matrix(const MatrixType& m) v.resize(21); set(v, 20, x); VERIFY_IS_APPROX(*get(v, 20), x); - v.resize(22,y); + v.resize(22, y); VERIFY_IS_APPROX(*get(v, 21), y); v.push_back(x); VERIFY_IS_APPROX(*get(v, 22), x); // do a lot of push_back such that the list gets internally resized // (with memory reallocation) - MatrixType* ref = &(*get(w, 0)); - for(int i=0; i<30 || ((ref==&(*get(w, 0))) && i<300); ++i) - v.push_back(*get(w, i%w.size())); - for(unsigned int i=23; i -void check_stdlist_transform(const TransformType&) +template void check_stdlist_transform(const TransformType &) { typedef typename TransformType::MatrixType MatrixType; TransformType x(MatrixType::Random()), y(MatrixType::Random()); @@ -96,8 +87,7 @@ void check_stdlist_transform(const TransformType&) v = w; itv = v.begin(); itw = w.begin(); - for(int i = 0; i < 20; i++) - { + for (int i = 0; i < 20; i++) { VERIFY_IS_APPROX(*itw, *itv); ++itv; ++itw; @@ -106,24 +96,19 @@ void check_stdlist_transform(const TransformType&) v.resize(21); set(v, 20, x); VERIFY_IS_APPROX(*get(v, 20), x); - v.resize(22,y); + v.resize(22, y); VERIFY_IS_APPROX(*get(v, 21), y); v.push_back(x); VERIFY_IS_APPROX(*get(v, 22), x); // do a lot of push_back such that the list gets internally resized // (with memory reallocation) - TransformType* ref = &(*get(w, 0)); - for(int i=0; i<30 || ((ref==&(*get(w, 0))) && i<300); ++i) - v.push_back(*get(w, i%w.size())); - for(unsigned int i=23; imatrix()==get(w, (i-23)%w.size())->matrix()); - } + TransformType *ref = &(*get(w, 0)); + for (int i = 0; i < 30 || ((ref == &(*get(w, 0))) && i < 300); ++i) v.push_back(*get(w, i % w.size())); + for (unsigned int i = 23; i < v.size(); ++i) { VERIFY(get(v, i)->matrix() == get(w, (i - 23) % w.size())->matrix()); } } -template -void check_stdlist_quaternion(const QuaternionType&) +template void check_stdlist_quaternion(const QuaternionType &) { typedef typename QuaternionType::Coefficients Coefficients; QuaternionType x(Coefficients::Random()), y(Coefficients::Random()); @@ -136,8 +121,7 @@ void check_stdlist_quaternion(const QuaternionType&) v = w; itv = v.begin(); itw = w.begin(); - for(int i = 0; i < 20; i++) - { + for (int i = 0; i < 20; i++) { VERIFY_IS_APPROX(*itw, *itv); ++itv; ++itw; @@ -146,20 +130,16 @@ void check_stdlist_quaternion(const QuaternionType&) v.resize(21); set(v, 20, x); VERIFY_IS_APPROX(*get(v, 20), x); - v.resize(22,y); + v.resize(22, y); VERIFY_IS_APPROX(*get(v, 21), y); v.push_back(x); VERIFY_IS_APPROX(*get(v, 22), x); // do a lot of push_back such that the list gets internally resized // (with memory reallocation) - QuaternionType* ref = &(*get(w, 0)); - for(int i=0; i<30 || ((ref==&(*get(w, 0))) && i<300); ++i) - v.push_back(*get(w, i%w.size())); - for(unsigned int i=23; icoeffs()==get(w, (i-23)%w.size())->coeffs()); - } + QuaternionType *ref = &(*get(w, 0)); + for (int i = 0; i < 30 || ((ref == &(*get(w, 0))) && i < 300); ++i) v.push_back(*get(w, i % w.size())); + for (unsigned int i = 23; i < v.size(); ++i) { VERIFY(get(v, i)->coeffs() == get(w, (i - 23) % w.size())->coeffs()); } } void test_stdlist_overload() @@ -176,13 +156,13 @@ void test_stdlist_overload() CALL_SUBTEST_2(check_stdlist_matrix(Matrix4d())); // some dynamic sizes - CALL_SUBTEST_3(check_stdlist_matrix(MatrixXd(1,1))); + CALL_SUBTEST_3(check_stdlist_matrix(MatrixXd(1, 1))); CALL_SUBTEST_3(check_stdlist_matrix(VectorXd(20))); CALL_SUBTEST_3(check_stdlist_matrix(RowVectorXf(20))); - CALL_SUBTEST_3(check_stdlist_matrix(MatrixXcf(10,10))); + CALL_SUBTEST_3(check_stdlist_matrix(MatrixXcf(10, 10))); // some Transform - CALL_SUBTEST_4(check_stdlist_transform(Affine2f())); // does not need the specialization (2+1)^2 = 9 + CALL_SUBTEST_4(check_stdlist_transform(Affine2f()));// does not need the specialization (2+1)^2 = 9 CALL_SUBTEST_4(check_stdlist_transform(Affine3f())); CALL_SUBTEST_4(check_stdlist_transform(Affine3d())); diff --git a/filmulator-gui/core/nlmeans/eigen/test/stdvector.cpp b/filmulator-gui/core/nlmeans/eigen/test/stdvector.cpp index fa928ea4..9fb94cba 100644 --- a/filmulator-gui/core/nlmeans/eigen/test/stdvector.cpp +++ b/filmulator-gui/core/nlmeans/eigen/test/stdvector.cpp @@ -8,122 +8,98 @@ // with this file, You can obtain one at http://mozilla.org/MPL/2.0/. #include "main.h" -#include #include +#include -template -void check_stdvector_matrix(const MatrixType& m) +template void check_stdvector_matrix(const MatrixType &m) { typename MatrixType::Index rows = m.rows(); typename MatrixType::Index cols = m.cols(); - MatrixType x = MatrixType::Random(rows,cols), y = MatrixType::Random(rows,cols); - std::vector > v(10, MatrixType(rows,cols)), w(20, y); + MatrixType x = MatrixType::Random(rows, cols), y = MatrixType::Random(rows, cols); + std::vector> v(10, MatrixType(rows, cols)), w(20, y); v[5] = x; w[6] = v[5]; VERIFY_IS_APPROX(w[6], v[5]); v = w; - for(int i = 0; i < 20; i++) - { - VERIFY_IS_APPROX(w[i], v[i]); - } + for (int i = 0; i < 20; i++) { VERIFY_IS_APPROX(w[i], v[i]); } v.resize(21); v[20] = x; VERIFY_IS_APPROX(v[20], x); - v.resize(22,y); + v.resize(22, y); VERIFY_IS_APPROX(v[21], y); v.push_back(x); VERIFY_IS_APPROX(v[22], x); - VERIFY((internal::UIntPtr)&(v[22]) == (internal::UIntPtr)&(v[21]) + sizeof(MatrixType)); + VERIFY((internal::UIntPtr) & (v[22]) == (internal::UIntPtr) & (v[21]) + sizeof(MatrixType)); // do a lot of push_back such that the vector gets internally resized // (with memory reallocation) - MatrixType* ref = &w[0]; - for(int i=0; i<30 || ((ref==&w[0]) && i<300); ++i) - v.push_back(w[i%w.size()]); - for(unsigned int i=23; i -void check_stdvector_transform(const TransformType&) +template void check_stdvector_transform(const TransformType &) { typedef typename TransformType::MatrixType MatrixType; TransformType x(MatrixType::Random()), y(MatrixType::Random()); - std::vector > v(10), w(20, y); + std::vector> v(10), w(20, y); v[5] = x; w[6] = v[5]; VERIFY_IS_APPROX(w[6], v[5]); v = w; - for(int i = 0; i < 20; i++) - { - VERIFY_IS_APPROX(w[i], v[i]); - } + for (int i = 0; i < 20; i++) { VERIFY_IS_APPROX(w[i], v[i]); } v.resize(21); v[20] = x; VERIFY_IS_APPROX(v[20], x); - v.resize(22,y); + v.resize(22, y); VERIFY_IS_APPROX(v[21], y); v.push_back(x); VERIFY_IS_APPROX(v[22], x); - VERIFY((internal::UIntPtr)&(v[22]) == (internal::UIntPtr)&(v[21]) + sizeof(TransformType)); + VERIFY((internal::UIntPtr) & (v[22]) == (internal::UIntPtr) & (v[21]) + sizeof(TransformType)); // do a lot of push_back such that the vector gets internally resized // (with memory reallocation) - TransformType* ref = &w[0]; - for(int i=0; i<30 || ((ref==&w[0]) && i<300); ++i) - v.push_back(w[i%w.size()]); - for(unsigned int i=23; i -void check_stdvector_quaternion(const QuaternionType&) +template void check_stdvector_quaternion(const QuaternionType &) { typedef typename QuaternionType::Coefficients Coefficients; QuaternionType x(Coefficients::Random()), y(Coefficients::Random()); - std::vector > v(10), w(20, y); + std::vector> v(10), w(20, y); v[5] = x; w[6] = v[5]; VERIFY_IS_APPROX(w[6], v[5]); v = w; - for(int i = 0; i < 20; i++) - { - VERIFY_IS_APPROX(w[i], v[i]); - } + for (int i = 0; i < 20; i++) { VERIFY_IS_APPROX(w[i], v[i]); } v.resize(21); v[20] = x; VERIFY_IS_APPROX(v[20], x); - v.resize(22,y); + v.resize(22, y); VERIFY_IS_APPROX(v[21], y); v.push_back(x); VERIFY_IS_APPROX(v[22], x); - VERIFY((internal::UIntPtr)&(v[22]) == (internal::UIntPtr)&(v[21]) + sizeof(QuaternionType)); + VERIFY((internal::UIntPtr) & (v[22]) == (internal::UIntPtr) & (v[21]) + sizeof(QuaternionType)); // do a lot of push_back such that the vector gets internally resized // (with memory reallocation) - QuaternionType* ref = &w[0]; - for(int i=0; i<30 || ((ref==&w[0]) && i<300); ++i) - v.push_back(w[i%w.size()]); - for(unsigned int i=23; i= 7 -// eigen/Eigen/src/Core/util/Memory.h:189:12: warning: argument 1 value '18446744073709551612' exceeds maximum object size 9223372036854775807 -// This has been reported to gcc there: https://gcc.gnu.org/bugzilla/show_bug.cgi?id=87544 +// eigen/Eigen/src/Core/util/Memory.h:189:12: warning: argument 1 value '18446744073709551612' exceeds maximum object +// size 9223372036854775807 This has been reported to gcc there: https://gcc.gnu.org/bugzilla/show_bug.cgi?id=87544 void std_vector_gcc_warning() { typedef Eigen::Vector3f T; - std::vector > v; + std::vector> v; v.push_back(T()); } @@ -141,16 +117,16 @@ void test_stdvector() CALL_SUBTEST_2(check_stdvector_matrix(Matrix4d())); // some dynamic sizes - CALL_SUBTEST_3(check_stdvector_matrix(MatrixXd(1,1))); + CALL_SUBTEST_3(check_stdvector_matrix(MatrixXd(1, 1))); CALL_SUBTEST_3(check_stdvector_matrix(VectorXd(20))); CALL_SUBTEST_3(check_stdvector_matrix(RowVectorXf(20))); - CALL_SUBTEST_3(check_stdvector_matrix(MatrixXcf(10,10))); + CALL_SUBTEST_3(check_stdvector_matrix(MatrixXcf(10, 10))); // some Transform CALL_SUBTEST_4(check_stdvector_transform(Projective2f())); CALL_SUBTEST_4(check_stdvector_transform(Projective3f())); CALL_SUBTEST_4(check_stdvector_transform(Projective3d())); - //CALL_SUBTEST(heck_stdvector_transform(Projective4d())); + // CALL_SUBTEST(heck_stdvector_transform(Projective4d())); // some Quaternion CALL_SUBTEST_5(check_stdvector_quaternion(Quaternionf())); diff --git a/filmulator-gui/core/nlmeans/eigen/test/stdvector_overload.cpp b/filmulator-gui/core/nlmeans/eigen/test/stdvector_overload.cpp index 95966595..a9d9c34f 100644 --- a/filmulator-gui/core/nlmeans/eigen/test/stdvector_overload.cpp +++ b/filmulator-gui/core/nlmeans/eigen/test/stdvector_overload.cpp @@ -10,8 +10,8 @@ #include "main.h" -#include #include +#include EIGEN_DEFINE_STL_VECTOR_SPECIALIZATION(Vector4f) @@ -25,44 +25,35 @@ EIGEN_DEFINE_STL_VECTOR_SPECIALIZATION(Affine3d) EIGEN_DEFINE_STL_VECTOR_SPECIALIZATION(Quaternionf) EIGEN_DEFINE_STL_VECTOR_SPECIALIZATION(Quaterniond) -template -void check_stdvector_matrix(const MatrixType& m) +template void check_stdvector_matrix(const MatrixType &m) { typename MatrixType::Index rows = m.rows(); typename MatrixType::Index cols = m.cols(); - MatrixType x = MatrixType::Random(rows,cols), y = MatrixType::Random(rows,cols); - std::vector v(10, MatrixType(rows,cols)), w(20, y); + MatrixType x = MatrixType::Random(rows, cols), y = MatrixType::Random(rows, cols); + std::vector v(10, MatrixType(rows, cols)), w(20, y); v[5] = x; w[6] = v[5]; VERIFY_IS_APPROX(w[6], v[5]); v = w; - for(int i = 0; i < 20; i++) - { - VERIFY_IS_APPROX(w[i], v[i]); - } + for (int i = 0; i < 20; i++) { VERIFY_IS_APPROX(w[i], v[i]); } v.resize(21); v[20] = x; VERIFY_IS_APPROX(v[20], x); - v.resize(22,y); + v.resize(22, y); VERIFY_IS_APPROX(v[21], y); v.push_back(x); VERIFY_IS_APPROX(v[22], x); - VERIFY((internal::UIntPtr)&(v[22]) == (internal::UIntPtr)&(v[21]) + sizeof(MatrixType)); + VERIFY((internal::UIntPtr) & (v[22]) == (internal::UIntPtr) & (v[21]) + sizeof(MatrixType)); // do a lot of push_back such that the vector gets internally resized // (with memory reallocation) - MatrixType* ref = &w[0]; - for(int i=0; i<30 || ((ref==&w[0]) && i<300); ++i) - v.push_back(w[i%w.size()]); - for(unsigned int i=23; i -void check_stdvector_transform(const TransformType&) +template void check_stdvector_transform(const TransformType &) { typedef typename TransformType::MatrixType MatrixType; TransformType x(MatrixType::Random()), y(MatrixType::Random()); @@ -71,33 +62,25 @@ void check_stdvector_transform(const TransformType&) w[6] = v[5]; VERIFY_IS_APPROX(w[6], v[5]); v = w; - for(int i = 0; i < 20; i++) - { - VERIFY_IS_APPROX(w[i], v[i]); - } + for (int i = 0; i < 20; i++) { VERIFY_IS_APPROX(w[i], v[i]); } v.resize(21); v[20] = x; VERIFY_IS_APPROX(v[20], x); - v.resize(22,y); + v.resize(22, y); VERIFY_IS_APPROX(v[21], y); v.push_back(x); VERIFY_IS_APPROX(v[22], x); - VERIFY((internal::UIntPtr)&(v[22]) == (internal::UIntPtr)&(v[21]) + sizeof(TransformType)); + VERIFY((internal::UIntPtr) & (v[22]) == (internal::UIntPtr) & (v[21]) + sizeof(TransformType)); // do a lot of push_back such that the vector gets internally resized // (with memory reallocation) - TransformType* ref = &w[0]; - for(int i=0; i<30 || ((ref==&w[0]) && i<300); ++i) - v.push_back(w[i%w.size()]); - for(unsigned int i=23; i -void check_stdvector_quaternion(const QuaternionType&) +template void check_stdvector_quaternion(const QuaternionType &) { typedef typename QuaternionType::Coefficients Coefficients; QuaternionType x(Coefficients::Random()), y(Coefficients::Random()); @@ -106,29 +89,22 @@ void check_stdvector_quaternion(const QuaternionType&) w[6] = v[5]; VERIFY_IS_APPROX(w[6], v[5]); v = w; - for(int i = 0; i < 20; i++) - { - VERIFY_IS_APPROX(w[i], v[i]); - } + for (int i = 0; i < 20; i++) { VERIFY_IS_APPROX(w[i], v[i]); } v.resize(21); v[20] = x; VERIFY_IS_APPROX(v[20], x); - v.resize(22,y); + v.resize(22, y); VERIFY_IS_APPROX(v[21], y); v.push_back(x); VERIFY_IS_APPROX(v[22], x); - VERIFY((internal::UIntPtr)&(v[22]) == (internal::UIntPtr)&(v[21]) + sizeof(QuaternionType)); + VERIFY((internal::UIntPtr) & (v[22]) == (internal::UIntPtr) & (v[21]) + sizeof(QuaternionType)); // do a lot of push_back such that the vector gets internally resized // (with memory reallocation) - QuaternionType* ref = &w[0]; - for(int i=0; i<30 || ((ref==&w[0]) && i<300); ++i) - v.push_back(w[i%w.size()]); - for(unsigned int i=23; i > superlu_double_colmajor; - SuperLU > > superlu_cplxdouble_colmajor; - CALL_SUBTEST_1( check_sparse_square_solving(superlu_double_colmajor) ); - CALL_SUBTEST_2( check_sparse_square_solving(superlu_cplxdouble_colmajor) ); - CALL_SUBTEST_1( check_sparse_square_determinant(superlu_double_colmajor) ); - CALL_SUBTEST_2( check_sparse_square_determinant(superlu_cplxdouble_colmajor) ); + SuperLU> superlu_double_colmajor; + SuperLU>> superlu_cplxdouble_colmajor; + CALL_SUBTEST_1(check_sparse_square_solving(superlu_double_colmajor)); + CALL_SUBTEST_2(check_sparse_square_solving(superlu_cplxdouble_colmajor)); + CALL_SUBTEST_1(check_sparse_square_determinant(superlu_double_colmajor)); + CALL_SUBTEST_2(check_sparse_square_determinant(superlu_cplxdouble_colmajor)); } diff --git a/filmulator-gui/core/nlmeans/eigen/test/svd_common.h b/filmulator-gui/core/nlmeans/eigen/test/svd_common.h index cba06659..0ac83440 100644 --- a/filmulator-gui/core/nlmeans/eigen/test/svd_common.h +++ b/filmulator-gui/core/nlmeans/eigen/test/svd_common.h @@ -20,34 +20,27 @@ // Check that the matrix m is properly reconstructed and that the U and V factors are unitary // The SVD must have already been computed. -template -void svd_check_full(const MatrixType& m, const SvdType& svd) +template void svd_check_full(const MatrixType &m, const SvdType &svd) { Index rows = m.rows(); Index cols = m.cols(); - enum { - RowsAtCompileTime = MatrixType::RowsAtCompileTime, - ColsAtCompileTime = MatrixType::ColsAtCompileTime - }; + enum { RowsAtCompileTime = MatrixType::RowsAtCompileTime, ColsAtCompileTime = MatrixType::ColsAtCompileTime }; typedef typename MatrixType::Scalar Scalar; typedef typename MatrixType::RealScalar RealScalar; typedef Matrix MatrixUType; typedef Matrix MatrixVType; - MatrixType sigma = MatrixType::Zero(rows,cols); + MatrixType sigma = MatrixType::Zero(rows, cols); sigma.diagonal() = svd.singularValues().template cast(); MatrixUType u = svd.matrixU(); MatrixVType v = svd.matrixV(); RealScalar scaling = m.cwiseAbs().maxCoeff(); - if(scaling<(std::numeric_limits::min)()) - { + if (scaling < (std::numeric_limits::min)()) { VERIFY(sigma.cwiseAbs().maxCoeff() <= (std::numeric_limits::min)()); - } - else - { - VERIFY_IS_APPROX(m/scaling, u * (sigma/scaling) * v.adjoint()); + } else { + VERIFY_IS_APPROX(m / scaling, u * (sigma / scaling) * v.adjoint()); } VERIFY_IS_UNITARY(u); VERIFY_IS_UNITARY(v); @@ -55,9 +48,7 @@ void svd_check_full(const MatrixType& m, const SvdType& svd) // Compare partial SVD defined by computationOptions to a full SVD referenceSvd template -void svd_compare_to_full(const MatrixType& m, - unsigned int computationOptions, - const SvdType& referenceSvd) +void svd_compare_to_full(const MatrixType &m, unsigned int computationOptions, const SvdType &referenceSvd) { typedef typename MatrixType::RealScalar RealScalar; Index rows = m.rows(); @@ -68,45 +59,45 @@ void svd_compare_to_full(const MatrixType& m, SvdType svd(m, computationOptions); VERIFY_IS_APPROX(svd.singularValues(), referenceSvd.singularValues()); - - if(computationOptions & (ComputeFullV|ComputeThinV)) - { - VERIFY( (svd.matrixV().adjoint()*svd.matrixV()).isIdentity(prec) ); - VERIFY_IS_APPROX( svd.matrixV().leftCols(diagSize) * svd.singularValues().asDiagonal() * svd.matrixV().leftCols(diagSize).adjoint(), - referenceSvd.matrixV().leftCols(diagSize) * referenceSvd.singularValues().asDiagonal() * referenceSvd.matrixV().leftCols(diagSize).adjoint()); + + if (computationOptions & (ComputeFullV | ComputeThinV)) { + VERIFY((svd.matrixV().adjoint() * svd.matrixV()).isIdentity(prec)); + VERIFY_IS_APPROX( + svd.matrixV().leftCols(diagSize) * svd.singularValues().asDiagonal() * svd.matrixV().leftCols(diagSize).adjoint(), + referenceSvd.matrixV().leftCols(diagSize) * referenceSvd.singularValues().asDiagonal() + * referenceSvd.matrixV().leftCols(diagSize).adjoint()); } - - if(computationOptions & (ComputeFullU|ComputeThinU)) - { - VERIFY( (svd.matrixU().adjoint()*svd.matrixU()).isIdentity(prec) ); - VERIFY_IS_APPROX( svd.matrixU().leftCols(diagSize) * svd.singularValues().cwiseAbs2().asDiagonal() * svd.matrixU().leftCols(diagSize).adjoint(), - referenceSvd.matrixU().leftCols(diagSize) * referenceSvd.singularValues().cwiseAbs2().asDiagonal() * referenceSvd.matrixU().leftCols(diagSize).adjoint()); + + if (computationOptions & (ComputeFullU | ComputeThinU)) { + VERIFY((svd.matrixU().adjoint() * svd.matrixU()).isIdentity(prec)); + VERIFY_IS_APPROX(svd.matrixU().leftCols(diagSize) * svd.singularValues().cwiseAbs2().asDiagonal() + * svd.matrixU().leftCols(diagSize).adjoint(), + referenceSvd.matrixU().leftCols(diagSize) * referenceSvd.singularValues().cwiseAbs2().asDiagonal() + * referenceSvd.matrixU().leftCols(diagSize).adjoint()); } - + // The following checks are not critical. - // For instance, with Dived&Conquer SVD, if only the factor 'V' is computedt then different matrix-matrix product implementation will be used - // and the resulting 'V' factor might be significantly different when the SVD decomposition is not unique, especially with single precision float. + // For instance, with Dived&Conquer SVD, if only the factor 'V' is computedt then different matrix-matrix product + // implementation will be used and the resulting 'V' factor might be significantly different when the SVD + // decomposition is not unique, especially with single precision float. ++g_test_level; - if(computationOptions & ComputeFullU) VERIFY_IS_APPROX(svd.matrixU(), referenceSvd.matrixU()); - if(computationOptions & ComputeThinU) VERIFY_IS_APPROX(svd.matrixU(), referenceSvd.matrixU().leftCols(diagSize)); - if(computationOptions & ComputeFullV) VERIFY_IS_APPROX(svd.matrixV().cwiseAbs(), referenceSvd.matrixV().cwiseAbs()); - if(computationOptions & ComputeThinV) VERIFY_IS_APPROX(svd.matrixV(), referenceSvd.matrixV().leftCols(diagSize)); + if (computationOptions & ComputeFullU) VERIFY_IS_APPROX(svd.matrixU(), referenceSvd.matrixU()); + if (computationOptions & ComputeThinU) VERIFY_IS_APPROX(svd.matrixU(), referenceSvd.matrixU().leftCols(diagSize)); + if (computationOptions & ComputeFullV) VERIFY_IS_APPROX(svd.matrixV().cwiseAbs(), referenceSvd.matrixV().cwiseAbs()); + if (computationOptions & ComputeThinV) VERIFY_IS_APPROX(svd.matrixV(), referenceSvd.matrixV().leftCols(diagSize)); --g_test_level; } // template -void svd_least_square(const MatrixType& m, unsigned int computationOptions) +void svd_least_square(const MatrixType &m, unsigned int computationOptions) { typedef typename MatrixType::Scalar Scalar; typedef typename MatrixType::RealScalar RealScalar; Index rows = m.rows(); Index cols = m.cols(); - enum { - RowsAtCompileTime = MatrixType::RowsAtCompileTime, - ColsAtCompileTime = MatrixType::ColsAtCompileTime - }; + enum { RowsAtCompileTime = MatrixType::RowsAtCompileTime, ColsAtCompileTime = MatrixType::ColsAtCompileTime }; typedef Matrix RhsType; typedef Matrix SolutionType; @@ -114,206 +105,196 @@ void svd_least_square(const MatrixType& m, unsigned int computationOptions) RhsType rhs = RhsType::Random(rows, internal::random(1, cols)); SvdType svd(m, computationOptions); - if(internal::is_same::value) svd.setThreshold(1e-8); - else if(internal::is_same::value) svd.setThreshold(2e-4); + if (internal::is_same::value) + svd.setThreshold(1e-8); + else if (internal::is_same::value) + svd.setThreshold(2e-4); SolutionType x = svd.solve(rhs); - - RealScalar residual = (m*x-rhs).norm(); + + RealScalar residual = (m * x - rhs).norm(); RealScalar rhs_norm = rhs.norm(); - if(!test_isMuchSmallerThan(residual,rhs.norm())) - { + if (!test_isMuchSmallerThan(residual, rhs.norm())) { // ^^^ If the residual is very small, then we have an exact solution, so we are already good. - + // evaluate normal equation which works also for least-squares solutions - if(internal::is_same::value || svd.rank()==m.diagonal().size()) - { + if (internal::is_same::value || svd.rank() == m.diagonal().size()) { using std::sqrt; // This test is not stable with single precision. - // This is probably because squaring m signicantly affects the precision. - if(internal::is_same::value) ++g_test_level; - - VERIFY_IS_APPROX(m.adjoint()*(m*x),m.adjoint()*rhs); - - if(internal::is_same::value) --g_test_level; + // This is probably because squaring m signicantly affects the precision. + if (internal::is_same::value) ++g_test_level; + + VERIFY_IS_APPROX(m.adjoint() * (m * x), m.adjoint() * rhs); + + if (internal::is_same::value) --g_test_level; } - + // Check that there is no significantly better solution in the neighborhood of x - for(Index k=0;k::epsilon())*x.row(k); - RealScalar residual_y = (m*y-rhs).norm(); - VERIFY( test_isMuchSmallerThan(abs(residual_y-residual), rhs_norm) || residual < residual_y ); - if(internal::is_same::value) ++g_test_level; - VERIFY( test_isApprox(residual_y,residual) || residual < residual_y ); - if(internal::is_same::value) --g_test_level; - - y.row(k) = (RealScalar(1)-2*NumTraits::epsilon())*x.row(k); - residual_y = (m*y-rhs).norm(); - VERIFY( test_isMuchSmallerThan(abs(residual_y-residual), rhs_norm) || residual < residual_y ); - if(internal::is_same::value) ++g_test_level; - VERIFY( test_isApprox(residual_y,residual) || residual < residual_y ); - if(internal::is_same::value) --g_test_level; + y.row(k) = (RealScalar(1) + 2 * NumTraits::epsilon()) * x.row(k); + RealScalar residual_y = (m * y - rhs).norm(); + VERIFY(test_isMuchSmallerThan(abs(residual_y - residual), rhs_norm) || residual < residual_y); + if (internal::is_same::value) ++g_test_level; + VERIFY(test_isApprox(residual_y, residual) || residual < residual_y); + if (internal::is_same::value) --g_test_level; + + y.row(k) = (RealScalar(1) - 2 * NumTraits::epsilon()) * x.row(k); + residual_y = (m * y - rhs).norm(); + VERIFY(test_isMuchSmallerThan(abs(residual_y - residual), rhs_norm) || residual < residual_y); + if (internal::is_same::value) ++g_test_level; + VERIFY(test_isApprox(residual_y, residual) || residual < residual_y); + if (internal::is_same::value) --g_test_level; } } } // check minimal norm solutions, the inoput matrix m is only used to recover problem size -template -void svd_min_norm(const MatrixType& m, unsigned int computationOptions) +template void svd_min_norm(const MatrixType &m, unsigned int computationOptions) { typedef typename MatrixType::Scalar Scalar; Index cols = m.cols(); - enum { - ColsAtCompileTime = MatrixType::ColsAtCompileTime - }; + enum { ColsAtCompileTime = MatrixType::ColsAtCompileTime }; typedef Matrix SolutionType; // generate a full-rank m x n problem with m MatrixType2; typedef Matrix RhsType2; typedef Matrix MatrixType2T; - Index rank = RankAtCompileTime2==Dynamic ? internal::random(1,cols) : Index(RankAtCompileTime2); - MatrixType2 m2(rank,cols); + Index rank = RankAtCompileTime2 == Dynamic ? internal::random(1, cols) : Index(RankAtCompileTime2); + MatrixType2 m2(rank, cols); int guard = 0; do { m2.setRandom(); - } while(SVD_FOR_MIN_NORM(MatrixType2)(m2).setThreshold(test_precision()).rank()!=rank && (++guard)<10); - VERIFY(guard<10); + } while (SVD_FOR_MIN_NORM(MatrixType2)(m2).setThreshold(test_precision()).rank() != rank && (++guard) < 10); + VERIFY(guard < 10); RhsType2 rhs2 = RhsType2::Random(rank); // use QR to find a reference minimal norm solution HouseholderQR qr(m2.adjoint()); - Matrix tmp = qr.matrixQR().topLeftCorner(rank,rank).template triangularView().adjoint().solve(rhs2); + Matrix tmp = + qr.matrixQR().topLeftCorner(rank, rank).template triangularView().adjoint().solve(rhs2); tmp.conservativeResize(cols); - tmp.tail(cols-rank).setZero(); + tmp.tail(cols - rank).setZero(); SolutionType x21 = qr.householderQ() * tmp; // now check with SVD SVD_FOR_MIN_NORM(MatrixType2) svd2(m2, computationOptions); SolutionType x22 = svd2.solve(rhs2); - VERIFY_IS_APPROX(m2*x21, rhs2); - VERIFY_IS_APPROX(m2*x22, rhs2); + VERIFY_IS_APPROX(m2 * x21, rhs2); + VERIFY_IS_APPROX(m2 * x22, rhs2); VERIFY_IS_APPROX(x21, x22); // Now check with a rank deficient matrix typedef Matrix MatrixType3; typedef Matrix RhsType3; - Index rows3 = RowsAtCompileTime3==Dynamic ? internal::random(rank+1,2*cols) : Index(RowsAtCompileTime3); - Matrix C = Matrix::Random(rows3,rank); + Index rows3 = RowsAtCompileTime3 == Dynamic ? internal::random(rank + 1, 2 * cols) : Index(RowsAtCompileTime3); + Matrix C = Matrix::Random(rows3, rank); MatrixType3 m3 = C * m2; RhsType3 rhs3 = C * rhs2; SVD_FOR_MIN_NORM(MatrixType3) svd3(m3, computationOptions); SolutionType x3 = svd3.solve(rhs3); - VERIFY_IS_APPROX(m3*x3, rhs3); - VERIFY_IS_APPROX(m3*x21, rhs3); - VERIFY_IS_APPROX(m2*x3, rhs2); + VERIFY_IS_APPROX(m3 * x3, rhs3); + VERIFY_IS_APPROX(m3 * x21, rhs3); + VERIFY_IS_APPROX(m2 * x3, rhs2); VERIFY_IS_APPROX(x21, x3); } // Check full, compare_to_full, least_square, and min_norm for all possible compute-options template -void svd_test_all_computation_options(const MatrixType& m, bool full_only) +void svd_test_all_computation_options(const MatrixType &m, bool full_only) { -// if (QRPreconditioner == NoQRPreconditioner && m.rows() != m.cols()) -// return; - SvdType fullSvd(m, ComputeFullU|ComputeFullV); - CALL_SUBTEST(( svd_check_full(m, fullSvd) )); - CALL_SUBTEST(( svd_least_square(m, ComputeFullU | ComputeFullV) )); - CALL_SUBTEST(( svd_min_norm(m, ComputeFullU | ComputeFullV) )); - - #if defined __INTEL_COMPILER - // remark #111: statement is unreachable - #pragma warning disable 111 - #endif - if(full_only) - return; - - CALL_SUBTEST(( svd_compare_to_full(m, ComputeFullU, fullSvd) )); - CALL_SUBTEST(( svd_compare_to_full(m, ComputeFullV, fullSvd) )); - CALL_SUBTEST(( svd_compare_to_full(m, 0, fullSvd) )); + // if (QRPreconditioner == NoQRPreconditioner && m.rows() != m.cols()) + // return; + SvdType fullSvd(m, ComputeFullU | ComputeFullV); + CALL_SUBTEST((svd_check_full(m, fullSvd))); + CALL_SUBTEST((svd_least_square(m, ComputeFullU | ComputeFullV))); + CALL_SUBTEST((svd_min_norm(m, ComputeFullU | ComputeFullV))); + +#if defined __INTEL_COMPILER +// remark #111: statement is unreachable +#pragma warning disable 111 +#endif + if (full_only) return; + + CALL_SUBTEST((svd_compare_to_full(m, ComputeFullU, fullSvd))); + CALL_SUBTEST((svd_compare_to_full(m, ComputeFullV, fullSvd))); + CALL_SUBTEST((svd_compare_to_full(m, 0, fullSvd))); if (MatrixType::ColsAtCompileTime == Dynamic) { // thin U/V are only available with dynamic number of columns - CALL_SUBTEST(( svd_compare_to_full(m, ComputeFullU|ComputeThinV, fullSvd) )); - CALL_SUBTEST(( svd_compare_to_full(m, ComputeThinV, fullSvd) )); - CALL_SUBTEST(( svd_compare_to_full(m, ComputeThinU|ComputeFullV, fullSvd) )); - CALL_SUBTEST(( svd_compare_to_full(m, ComputeThinU , fullSvd) )); - CALL_SUBTEST(( svd_compare_to_full(m, ComputeThinU|ComputeThinV, fullSvd) )); - - CALL_SUBTEST(( svd_least_square(m, ComputeFullU | ComputeThinV) )); - CALL_SUBTEST(( svd_least_square(m, ComputeThinU | ComputeFullV) )); - CALL_SUBTEST(( svd_least_square(m, ComputeThinU | ComputeThinV) )); - - CALL_SUBTEST(( svd_min_norm(m, ComputeFullU | ComputeThinV) )); - CALL_SUBTEST(( svd_min_norm(m, ComputeThinU | ComputeFullV) )); - CALL_SUBTEST(( svd_min_norm(m, ComputeThinU | ComputeThinV) )); + CALL_SUBTEST((svd_compare_to_full(m, ComputeFullU | ComputeThinV, fullSvd))); + CALL_SUBTEST((svd_compare_to_full(m, ComputeThinV, fullSvd))); + CALL_SUBTEST((svd_compare_to_full(m, ComputeThinU | ComputeFullV, fullSvd))); + CALL_SUBTEST((svd_compare_to_full(m, ComputeThinU, fullSvd))); + CALL_SUBTEST((svd_compare_to_full(m, ComputeThinU | ComputeThinV, fullSvd))); + + CALL_SUBTEST((svd_least_square(m, ComputeFullU | ComputeThinV))); + CALL_SUBTEST((svd_least_square(m, ComputeThinU | ComputeFullV))); + CALL_SUBTEST((svd_least_square(m, ComputeThinU | ComputeThinV))); + + CALL_SUBTEST((svd_min_norm(m, ComputeFullU | ComputeThinV))); + CALL_SUBTEST((svd_min_norm(m, ComputeThinU | ComputeFullV))); + CALL_SUBTEST((svd_min_norm(m, ComputeThinU | ComputeThinV))); // test reconstruction Index diagSize = (std::min)(m.rows(), m.cols()); SvdType svd(m, ComputeThinU | ComputeThinV); - VERIFY_IS_APPROX(m, svd.matrixU().leftCols(diagSize) * svd.singularValues().asDiagonal() * svd.matrixV().leftCols(diagSize).adjoint()); + VERIFY_IS_APPROX(m, + svd.matrixU().leftCols(diagSize) * svd.singularValues().asDiagonal() + * svd.matrixV().leftCols(diagSize).adjoint()); } } // work around stupid msvc error when constructing at compile time an expression that involves // a division by zero, even if the numeric type has floating point -template -EIGEN_DONT_INLINE Scalar zero() { return Scalar(0); } +template EIGEN_DONT_INLINE Scalar zero() { return Scalar(0); } // workaround aggressive optimization in ICC -template EIGEN_DONT_INLINE T sub(T a, T b) { return a - b; } +template EIGEN_DONT_INLINE T sub(T a, T b) { return a - b; } // all this function does is verify we don't iterate infinitely on nan/inf values -template -void svd_inf_nan() +template void svd_inf_nan() { SvdType svd; typedef typename MatrixType::Scalar Scalar; Scalar some_inf = Scalar(1) / zero(); VERIFY(sub(some_inf, some_inf) != sub(some_inf, some_inf)); - svd.compute(MatrixType::Constant(10,10,some_inf), ComputeFullU | ComputeFullV); + svd.compute(MatrixType::Constant(10, 10, some_inf), ComputeFullU | ComputeFullV); Scalar nan = std::numeric_limits::quiet_NaN(); VERIFY(nan != nan); - svd.compute(MatrixType::Constant(10,10,nan), ComputeFullU | ComputeFullV); + svd.compute(MatrixType::Constant(10, 10, nan), ComputeFullU | ComputeFullV); - MatrixType m = MatrixType::Zero(10,10); - m(internal::random(0,9), internal::random(0,9)) = some_inf; + MatrixType m = MatrixType::Zero(10, 10); + m(internal::random(0, 9), internal::random(0, 9)) = some_inf; svd.compute(m, ComputeFullU | ComputeFullV); - m = MatrixType::Zero(10,10); - m(internal::random(0,9), internal::random(0,9)) = nan; + m = MatrixType::Zero(10, 10); + m(internal::random(0, 9), internal::random(0, 9)) = nan; svd.compute(m, ComputeFullU | ComputeFullV); - + // regression test for bug 791 - m.resize(3,3); - m << 0, 2*NumTraits::epsilon(), 0.5, - 0, -0.5, 0, - nan, 0, 0; + m.resize(3, 3); + m << 0, 2 * NumTraits::epsilon(), 0.5, 0, -0.5, 0, nan, 0, 0; svd.compute(m, ComputeFullU | ComputeFullV); - - m.resize(4,4); - m << 1, 0, 0, 0, - 0, 3, 1, 2e-308, - 1, 0, 1, nan, - 0, nan, nan, 0; + + m.resize(4, 4); + m << 1, 0, 0, 0, 0, 3, 1, 2e-308, 1, 0, 1, nan, 0, nan, nan, 0; svd.compute(m, ComputeFullU | ComputeFullV); } // Regression test for bug 286: JacobiSVD loops indefinitely with some // matrices containing denormal numbers. -template -void svd_underoverflow() +template void svd_underoverflow() { #if defined __INTEL_COMPILER // shut up warning #239: floating point underflow @@ -321,77 +302,70 @@ void svd_underoverflow() #pragma warning disable 239 #endif Matrix2d M; - M << -7.90884e-313, -4.94e-324, - 0, 5.60844e-313; + M << -7.90884e-313, -4.94e-324, 0, 5.60844e-313; SVD_DEFAULT(Matrix2d) svd; - svd.compute(M,ComputeFullU|ComputeFullV); - CALL_SUBTEST( svd_check_full(M,svd) ); - + svd.compute(M, ComputeFullU | ComputeFullV); + CALL_SUBTEST(svd_check_full(M, svd)); + // Check all 2x2 matrices made with the following coefficients: VectorXd value_set(9); value_set << 0, 1, -1, 5.60844e-313, -5.60844e-313, 4.94e-324, -4.94e-324, -4.94e-223, 4.94e-223; - Array4i id(0,0,0,0); + Array4i id(0, 0, 0, 0); int k = 0; - do - { + do { M << value_set(id(0)), value_set(id(1)), value_set(id(2)), value_set(id(3)); - svd.compute(M,ComputeFullU|ComputeFullV); - CALL_SUBTEST( svd_check_full(M,svd) ); + svd.compute(M, ComputeFullU | ComputeFullV); + CALL_SUBTEST(svd_check_full(M, svd)); id(k)++; - if(id(k)>=value_set.size()) - { - while(k<3 && id(k)>=value_set.size()) id(++k)++; + if (id(k) >= value_set.size()) { + while (k < 3 && id(k) >= value_set.size()) id(++k)++; id.head(k).setZero(); - k=0; + k = 0; } - } while((id -void svd_all_trivial_2x2( void (*cb)(const MatrixType&,bool) ) +template void svd_all_trivial_2x2(void (*cb)(const MatrixType &, bool)) { MatrixType M; VectorXd value_set(3); value_set << 0, 1, -1; - Array4i id(0,0,0,0); + Array4i id(0, 0, 0, 0); int k = 0; - do - { + do { M << value_set(id(0)), value_set(id(1)), value_set(id(2)), value_set(id(3)); - - cb(M,false); - + + cb(M, false); + id(k)++; - if(id(k)>=value_set.size()) - { - while(k<3 && id(k)>=value_set.size()) id(++k)++; + if (id(k) >= value_set.size()) { + while (k < 3 && id(k) >= value_set.size()) id(++k)++; id.head(k).setZero(); - k=0; + k = 0; } - - } while((id -void svd_preallocate() +template void svd_preallocate() { Vector3f v(3.f, 2.f, 1.f); MatrixXf m = v.asDiagonal(); @@ -403,7 +377,7 @@ void svd_preallocate() svd.compute(m); VERIFY_IS_APPROX(svd.singularValues(), v); - SVD_DEFAULT(MatrixXf) svd2(3,3); + SVD_DEFAULT(MatrixXf) svd2(3, 3); internal::set_is_malloc_allowed(false); svd2.compute(m); internal::set_is_malloc_allowed(true); @@ -417,7 +391,7 @@ void svd_preallocate() svd2.compute(m); internal::set_is_malloc_allowed(true); - SVD_DEFAULT(MatrixXf) svd3(3,3,ComputeFullU|ComputeFullV); + SVD_DEFAULT(MatrixXf) svd3(3, 3, ComputeFullU | ComputeFullV); internal::set_is_malloc_allowed(false); svd2.compute(m); internal::set_is_malloc_allowed(true); @@ -425,21 +399,17 @@ void svd_preallocate() VERIFY_IS_APPROX(svd2.matrixU(), Matrix3f::Identity()); VERIFY_IS_APPROX(svd2.matrixV(), Matrix3f::Identity()); internal::set_is_malloc_allowed(false); - svd2.compute(m, ComputeFullU|ComputeFullV); + svd2.compute(m, ComputeFullU | ComputeFullV); internal::set_is_malloc_allowed(true); } -template -void svd_verify_assert(const MatrixType& m) +template void svd_verify_assert(const MatrixType &m) { typedef typename MatrixType::Scalar Scalar; Index rows = m.rows(); Index cols = m.cols(); - enum { - RowsAtCompileTime = MatrixType::RowsAtCompileTime, - ColsAtCompileTime = MatrixType::ColsAtCompileTime - }; + enum { RowsAtCompileTime = MatrixType::RowsAtCompileTime, ColsAtCompileTime = MatrixType::ColsAtCompileTime }; typedef Matrix RhsType; RhsType rhs(rows); @@ -455,9 +425,8 @@ void svd_verify_assert(const MatrixType& m) VERIFY_RAISES_ASSERT(svd.matrixV()) svd.singularValues(); VERIFY_RAISES_ASSERT(svd.solve(rhs)) - - if (ColsAtCompileTime == Dynamic) - { + + if (ColsAtCompileTime == Dynamic) { svd.compute(a, ComputeThinU); svd.matrixU(); VERIFY_RAISES_ASSERT(svd.matrixV()) @@ -466,9 +435,7 @@ void svd_verify_assert(const MatrixType& m) svd.matrixV(); VERIFY_RAISES_ASSERT(svd.matrixU()) VERIFY_RAISES_ASSERT(svd.solve(rhs)) - } - else - { + } else { VERIFY_RAISES_ASSERT(svd.compute(a, ComputeThinU)) VERIFY_RAISES_ASSERT(svd.compute(a, ComputeThinV)) } diff --git a/filmulator-gui/core/nlmeans/eigen/test/svd_fill.h b/filmulator-gui/core/nlmeans/eigen/test/svd_fill.h index d68647e9..f21b96be 100644 --- a/filmulator-gui/core/nlmeans/eigen/test/svd_fill.h +++ b/filmulator-gui/core/nlmeans/eigen/test/svd_fill.h @@ -7,112 +7,94 @@ // Public License v. 2.0. If a copy of the MPL was not distributed // with this file, You can obtain one at http://mozilla.org/MPL/2.0/. -template -Array four_denorms(); +template Array four_denorms(); -template<> -Array4f four_denorms() { return Array4f(5.60844e-39f, -5.60844e-39f, 4.94e-44f, -4.94e-44f); } -template<> -Array4d four_denorms() { return Array4d(5.60844e-313, -5.60844e-313, 4.94e-324, -4.94e-324); } -template -Array four_denorms() { return four_denorms().cast(); } +template<> Array4f four_denorms() { return Array4f(5.60844e-39f, -5.60844e-39f, 4.94e-44f, -4.94e-44f); } +template<> Array4d four_denorms() { return Array4d(5.60844e-313, -5.60844e-313, 4.94e-324, -4.94e-324); } +template Array four_denorms() { return four_denorms().cast(); } -template -void svd_fill_random(MatrixType &m, int Option = 0) +template void svd_fill_random(MatrixType &m, int Option = 0) { using std::pow; typedef typename MatrixType::Scalar Scalar; typedef typename MatrixType::RealScalar RealScalar; Index diagSize = (std::min)(m.rows(), m.cols()); - RealScalar s = std::numeric_limits::max_exponent10/4; - s = internal::random(1,s); - Matrix d = Matrix::Random(diagSize); - for(Index k=0; k(-s,s)); + RealScalar s = std::numeric_limits::max_exponent10 / 4; + s = internal::random(1, s); + Matrix d = Matrix::Random(diagSize); + for (Index k = 0; k < diagSize; ++k) d(k) = d(k) * pow(RealScalar(10), internal::random(-s, s)); + + bool dup = internal::random(0, 10) < 3; + bool unit_uv = + internal::random(0, 10) < (dup ? 7 : 3);// if we duplicate some diagonal entries, then increase the chance to + // preserve them using unitary U and V factors - bool dup = internal::random(0,10) < 3; - bool unit_uv = internal::random(0,10) < (dup?7:3); // if we duplicate some diagonal entries, then increase the chance to preserve them using unitary U and V factors - // duplicate some singular values - if(dup) - { - Index n = internal::random(0,d.size()-1); - for(Index i=0; i(0,d.size()-1)) = d(internal::random(0,d.size()-1)); + if (dup) { + Index n = internal::random(0, d.size() - 1); + for (Index i = 0; i < n; ++i) + d(internal::random(0, d.size() - 1)) = d(internal::random(0, d.size() - 1)); } - - Matrix U(m.rows(),diagSize); - Matrix VT(diagSize,m.cols()); - if(unit_uv) - { + + Matrix U(m.rows(), diagSize); + Matrix VT(diagSize, m.cols()); + if (unit_uv) { // in very rare cases let's try with a pure diagonal matrix - if(internal::random(0,10) < 1) - { + if (internal::random(0, 10) < 1) { U.setIdentity(); VT.setIdentity(); + } else { + createRandomPIMatrixOfRank(diagSize, U.rows(), U.cols(), U); + createRandomPIMatrixOfRank(diagSize, VT.rows(), VT.cols(), VT); } - else - { - createRandomPIMatrixOfRank(diagSize,U.rows(), U.cols(), U); - createRandomPIMatrixOfRank(diagSize,VT.rows(), VT.cols(), VT); - } - } - else - { + } else { U.setRandom(); VT.setRandom(); } - - Matrix samples(9); - samples << 0, four_denorms(), - -RealScalar(1)/NumTraits::highest(), RealScalar(1)/NumTraits::highest(), (std::numeric_limits::min)(), pow((std::numeric_limits::min)(),0.8); - - if(Option==Symmetric) - { + + Matrix samples(9); + samples << 0, four_denorms(), -RealScalar(1) / NumTraits::highest(), + RealScalar(1) / NumTraits::highest(), (std::numeric_limits::min)(), + pow((std::numeric_limits::min)(), 0.8); + + if (Option == Symmetric) { m = U * d.asDiagonal() * U.transpose(); - + // randomly nullify some rows/columns { - Index count = internal::random(-diagSize,diagSize); - for(Index k=0; k(0,diagSize-1); + Index count = internal::random(-diagSize, diagSize); + for (Index k = 0; k < count; ++k) { + Index i = internal::random(0, diagSize - 1); m.row(i).setZero(); m.col(i).setZero(); } - if(count<0) - // (partly) cancel some coeffs - if(!(dup && unit_uv)) - { - - Index n = internal::random(0,m.size()-1); - for(Index k=0; k(0,m.rows()-1); - Index j = internal::random(0,m.cols()-1); - m(j,i) = m(i,j) = samples(internal::random(0,samples.size()-1)); - if(NumTraits::IsComplex) - *(&numext::real_ref(m(j,i))+1) = *(&numext::real_ref(m(i,j))+1) = samples.real()(internal::random(0,samples.size()-1)); + if (count < 0) + // (partly) cancel some coeffs + if (!(dup && unit_uv)) { + + Index n = internal::random(0, m.size() - 1); + for (Index k = 0; k < n; ++k) { + Index i = internal::random(0, m.rows() - 1); + Index j = internal::random(0, m.cols() - 1); + m(j, i) = m(i, j) = samples(internal::random(0, samples.size() - 1)); + if (NumTraits::IsComplex) + *(&numext::real_ref(m(j, i)) + 1) = *(&numext::real_ref(m(i, j)) + 1) = + samples.real()(internal::random(0, samples.size() - 1)); + } } - } } - } - else - { + } else { m = U * d.asDiagonal() * VT; // (partly) cancel some coeffs - if(!(dup && unit_uv)) - { - Index n = internal::random(0,m.size()-1); - for(Index k=0; k(0,m.rows()-1); - Index j = internal::random(0,m.cols()-1); - m(i,j) = samples(internal::random(0,samples.size()-1)); - if(NumTraits::IsComplex) - *(&numext::real_ref(m(i,j))+1) = samples.real()(internal::random(0,samples.size()-1)); + if (!(dup && unit_uv)) { + Index n = internal::random(0, m.size() - 1); + for (Index k = 0; k < n; ++k) { + Index i = internal::random(0, m.rows() - 1); + Index j = internal::random(0, m.cols() - 1); + m(i, j) = samples(internal::random(0, samples.size() - 1)); + if (NumTraits::IsComplex) + *(&numext::real_ref(m(i, j)) + 1) = samples.real()(internal::random(0, samples.size() - 1)); } } } } - diff --git a/filmulator-gui/core/nlmeans/eigen/test/swap.cpp b/filmulator-gui/core/nlmeans/eigen/test/swap.cpp index f76e3624..7a3ae62c 100644 --- a/filmulator-gui/core/nlmeans/eigen/test/swap.cpp +++ b/filmulator-gui/core/nlmeans/eigen/test/swap.cpp @@ -10,72 +10,69 @@ #define EIGEN_NO_STATIC_ASSERT #include "main.h" -template -struct other_matrix_type +template struct other_matrix_type { typedef int type; }; template -struct other_matrix_type > +struct other_matrix_type> { - typedef Matrix<_Scalar, _Rows, _Cols, _Options^RowMajor, _MaxRows, _MaxCols> type; + typedef Matrix<_Scalar, _Rows, _Cols, _Options ^ RowMajor, _MaxRows, _MaxCols> type; }; -template void swap(const MatrixType& m) +template void swap(const MatrixType &m) { typedef typename other_matrix_type::type OtherMatrixType; typedef typename MatrixType::Scalar Scalar; - eigen_assert((!internal::is_same::value)); + eigen_assert((!internal::is_same::value)); typename MatrixType::Index rows = m.rows(); typename MatrixType::Index cols = m.cols(); - + // construct 3 matrix guaranteed to be distinct - MatrixType m1 = MatrixType::Random(rows,cols); - MatrixType m2 = MatrixType::Random(rows,cols) + Scalar(100) * MatrixType::Identity(rows,cols); - OtherMatrixType m3 = OtherMatrixType::Random(rows,cols) + Scalar(200) * OtherMatrixType::Identity(rows,cols); - + MatrixType m1 = MatrixType::Random(rows, cols); + MatrixType m2 = MatrixType::Random(rows, cols) + Scalar(100) * MatrixType::Identity(rows, cols); + OtherMatrixType m3 = OtherMatrixType::Random(rows, cols) + Scalar(200) * OtherMatrixType::Identity(rows, cols); + MatrixType m1_copy = m1; MatrixType m2_copy = m2; OtherMatrixType m3_copy = m3; - + // test swapping 2 matrices of same type - Scalar *d1=m1.data(), *d2=m2.data(); + Scalar *d1 = m1.data(), *d2 = m2.data(); m1.swap(m2); - VERIFY_IS_APPROX(m1,m2_copy); - VERIFY_IS_APPROX(m2,m1_copy); - if(MatrixType::SizeAtCompileTime==Dynamic) - { - VERIFY(m1.data()==d2); - VERIFY(m2.data()==d1); + VERIFY_IS_APPROX(m1, m2_copy); + VERIFY_IS_APPROX(m2, m1_copy); + if (MatrixType::SizeAtCompileTime == Dynamic) { + VERIFY(m1.data() == d2); + VERIFY(m2.data() == d1); } m1 = m1_copy; m2 = m2_copy; - + // test swapping 2 matrices of different types m1.swap(m3); - VERIFY_IS_APPROX(m1,m3_copy); - VERIFY_IS_APPROX(m3,m1_copy); + VERIFY_IS_APPROX(m1, m3_copy); + VERIFY_IS_APPROX(m3, m1_copy); m1 = m1_copy; m3 = m3_copy; - + // test swapping matrix with expression - m1.swap(m2.block(0,0,rows,cols)); - VERIFY_IS_APPROX(m1,m2_copy); - VERIFY_IS_APPROX(m2,m1_copy); + m1.swap(m2.block(0, 0, rows, cols)); + VERIFY_IS_APPROX(m1, m2_copy); + VERIFY_IS_APPROX(m2, m1_copy); m1 = m1_copy; m2 = m2_copy; // test swapping two expressions of different types m1.transpose().swap(m3.transpose()); - VERIFY_IS_APPROX(m1,m3_copy); - VERIFY_IS_APPROX(m3,m1_copy); + VERIFY_IS_APPROX(m1, m3_copy); + VERIFY_IS_APPROX(m3, m1_copy); m1 = m1_copy; m3 = m3_copy; - - if(m1.rows()>1) - { + + if (m1.rows() > 1) { // test assertion on mismatching size -- matrix case VERIFY_RAISES_ASSERT(m1.swap(m1.row(0))); // test assertion on mismatching size -- xpr case @@ -85,10 +82,10 @@ template void swap(const MatrixType& m) void test_swap() { - int s = internal::random(1,EIGEN_TEST_MAX_SIZE); - CALL_SUBTEST_1( swap(Matrix3f()) ); // fixed size, no vectorization - CALL_SUBTEST_2( swap(Matrix4d()) ); // fixed size, possible vectorization - CALL_SUBTEST_3( swap(MatrixXd(s,s)) ); // dyn size, no vectorization - CALL_SUBTEST_4( swap(MatrixXf(s,s)) ); // dyn size, possible vectorization + int s = internal::random(1, EIGEN_TEST_MAX_SIZE); + CALL_SUBTEST_1(swap(Matrix3f()));// fixed size, no vectorization + CALL_SUBTEST_2(swap(Matrix4d()));// fixed size, possible vectorization + CALL_SUBTEST_3(swap(MatrixXd(s, s)));// dyn size, no vectorization + CALL_SUBTEST_4(swap(MatrixXf(s, s)));// dyn size, possible vectorization TEST_SET_BUT_UNUSED_VARIABLE(s) } diff --git a/filmulator-gui/core/nlmeans/eigen/test/triangular.cpp b/filmulator-gui/core/nlmeans/eigen/test/triangular.cpp index 328eef4d..f4556b15 100644 --- a/filmulator-gui/core/nlmeans/eigen/test/triangular.cpp +++ b/filmulator-gui/core/nlmeans/eigen/test/triangular.cpp @@ -10,44 +10,38 @@ #include "main.h" - -template void triangular_square(const MatrixType& m) +template void triangular_square(const MatrixType &m) { typedef typename MatrixType::Scalar Scalar; typedef typename NumTraits::Real RealScalar; typedef Matrix VectorType; - RealScalar largerEps = 10*test_precision(); + RealScalar largerEps = 10 * test_precision(); typename MatrixType::Index rows = m.rows(); typename MatrixType::Index cols = m.cols(); - MatrixType m1 = MatrixType::Random(rows, cols), - m2 = MatrixType::Random(rows, cols), - m3(rows, cols), - m4(rows, cols), - r1(rows, cols), - r2(rows, cols); + MatrixType m1 = MatrixType::Random(rows, cols), m2 = MatrixType::Random(rows, cols), m3(rows, cols), m4(rows, cols), + r1(rows, cols), r2(rows, cols); VectorType v2 = VectorType::Random(rows); MatrixType m1up = m1.template triangularView(); MatrixType m2up = m2.template triangularView(); - if (rows*cols>1) - { + if (rows * cols > 1) { VERIFY(m1up.isUpperTriangular()); VERIFY(m2up.transpose().isLowerTriangular()); VERIFY(!m2.isLowerTriangular()); } -// VERIFY_IS_APPROX(m1up.transpose() * m2, m1.upper().transpose().lower() * m2); + // VERIFY_IS_APPROX(m1up.transpose() * m2, m1.upper().transpose().lower() * m2); // test overloaded operator+= r1.setZero(); r2.setZero(); - r1.template triangularView() += m1; + r1.template triangularView() += m1; r2 += m1up; - VERIFY_IS_APPROX(r1,r2); + VERIFY_IS_APPROX(r1, r2); // test overloaded operator= m1.setZero(); @@ -61,11 +55,11 @@ template void triangular_square(const MatrixType& m) VERIFY_IS_APPROX(m3.template triangularView().toDenseMatrix(), m1); VERIFY_IS_APPROX(m3.template triangularView().conjugate().toDenseMatrix(), - m3.conjugate().template triangularView().toDenseMatrix()); + m3.conjugate().template triangularView().toDenseMatrix()); m1 = MatrixType::Random(rows, cols); - for (int i=0; i(); + for (int i = 0; i < rows; ++i) + while (numext::abs2(m1(i, i)) < RealScalar(1e-1)) m1(i, i) = internal::random(); Transpose trm4(m4); // test back and forward subsitution with a vector as the rhs @@ -103,8 +97,8 @@ template void triangular_square(const MatrixType& m) m3 = m1.template triangularView(); VERIFY(m2.isApprox(m3 * (m1.template triangularView().solve(m2)), largerEps)); -// VERIFY(( m1.template triangularView() -// * m2.template triangularView()).isUpperTriangular()); + // VERIFY(( m1.template triangularView() + // * m2.template triangularView()).isUpperTriangular()); // test swap m1.setOnes(); @@ -112,47 +106,45 @@ template void triangular_square(const MatrixType& m) m2.template triangularView().swap(m1); m3.setZero(); m3.template triangularView().setOnes(); - VERIFY_IS_APPROX(m2,m3); - + VERIFY_IS_APPROX(m2, m3); + m1.setRandom(); m3 = m1.template triangularView(); - Matrix m5(cols, internal::random(1,20)); m5.setRandom(); - Matrix m6(internal::random(1,20), rows); m6.setRandom(); - VERIFY_IS_APPROX(m1.template triangularView() * m5, m3*m5); - VERIFY_IS_APPROX(m6*m1.template triangularView(), m6*m3); + Matrix m5(cols, internal::random(1, 20)); + m5.setRandom(); + Matrix m6(internal::random(1, 20), rows); + m6.setRandom(); + VERIFY_IS_APPROX(m1.template triangularView() * m5, m3 * m5); + VERIFY_IS_APPROX(m6 * m1.template triangularView(), m6 * m3); m1up = m1.template triangularView(); VERIFY_IS_APPROX(m1.template selfadjointView().template triangularView().toDenseMatrix(), m1up); VERIFY_IS_APPROX(m1up.template selfadjointView().template triangularView().toDenseMatrix(), m1up); - VERIFY_IS_APPROX(m1.template selfadjointView().template triangularView().toDenseMatrix(), m1up.adjoint()); - VERIFY_IS_APPROX(m1up.template selfadjointView().template triangularView().toDenseMatrix(), m1up.adjoint()); + VERIFY_IS_APPROX( + m1.template selfadjointView().template triangularView().toDenseMatrix(), m1up.adjoint()); + VERIFY_IS_APPROX( + m1up.template selfadjointView().template triangularView().toDenseMatrix(), m1up.adjoint()); VERIFY_IS_APPROX(m1.template selfadjointView().diagonal(), m1.diagonal()); - } -template void triangular_rect(const MatrixType& m) +template void triangular_rect(const MatrixType &m) { typedef typename MatrixType::Scalar Scalar; typedef typename NumTraits::Real RealScalar; - enum { Rows = MatrixType::RowsAtCompileTime, Cols = MatrixType::ColsAtCompileTime }; + enum { Rows = MatrixType::RowsAtCompileTime, Cols = MatrixType::ColsAtCompileTime }; Index rows = m.rows(); Index cols = m.cols(); - MatrixType m1 = MatrixType::Random(rows, cols), - m2 = MatrixType::Random(rows, cols), - m3(rows, cols), - m4(rows, cols), - r1(rows, cols), - r2(rows, cols); + MatrixType m1 = MatrixType::Random(rows, cols), m2 = MatrixType::Random(rows, cols), m3(rows, cols), m4(rows, cols), + r1(rows, cols), r2(rows, cols); MatrixType m1up = m1.template triangularView(); MatrixType m2up = m2.template triangularView(); - if (rows>1 && cols>1) - { + if (rows > 1 && cols > 1) { VERIFY(m1up.isUpperTriangular()); VERIFY(m2up.transpose().isLowerTriangular()); VERIFY(!m2.isLowerTriangular()); @@ -161,9 +153,9 @@ template void triangular_rect(const MatrixType& m) // test overloaded operator+= r1.setZero(); r2.setZero(); - r1.template triangularView() += m1; + r1.template triangularView() += m1; r2 += m1up; - VERIFY_IS_APPROX(r1,r2); + VERIFY_IS_APPROX(r1, r2); // test overloaded operator= m1.setZero(); @@ -211,7 +203,7 @@ template void triangular_rect(const MatrixType& m) m2.template triangularView().swap(m1); m3.setZero(); m3.template triangularView().setOnes(); - VERIFY_IS_APPROX(m2,m3); + VERIFY_IS_APPROX(m2, m3); } void bug_159() @@ -222,25 +214,26 @@ void bug_159() void test_triangular() { - int maxsize = (std::min)(EIGEN_TEST_MAX_SIZE,20); - for(int i = 0; i < g_repeat ; i++) - { - int r = internal::random(2,maxsize); TEST_SET_BUT_UNUSED_VARIABLE(r) - int c = internal::random(2,maxsize); TEST_SET_BUT_UNUSED_VARIABLE(c) - - CALL_SUBTEST_1( triangular_square(Matrix()) ); - CALL_SUBTEST_2( triangular_square(Matrix()) ); - CALL_SUBTEST_3( triangular_square(Matrix3d()) ); - CALL_SUBTEST_4( triangular_square(Matrix,8, 8>()) ); - CALL_SUBTEST_5( triangular_square(MatrixXcd(r,r)) ); - CALL_SUBTEST_6( triangular_square(Matrix(r, r)) ); - - CALL_SUBTEST_7( triangular_rect(Matrix()) ); - CALL_SUBTEST_8( triangular_rect(Matrix()) ); - CALL_SUBTEST_9( triangular_rect(MatrixXcf(r, c)) ); - CALL_SUBTEST_5( triangular_rect(MatrixXcd(r, c)) ); - CALL_SUBTEST_6( triangular_rect(Matrix(r, c)) ); + int maxsize = (std::min)(EIGEN_TEST_MAX_SIZE, 20); + for (int i = 0; i < g_repeat; i++) { + int r = internal::random(2, maxsize); + TEST_SET_BUT_UNUSED_VARIABLE(r) + int c = internal::random(2, maxsize); + TEST_SET_BUT_UNUSED_VARIABLE(c) + + CALL_SUBTEST_1(triangular_square(Matrix())); + CALL_SUBTEST_2(triangular_square(Matrix())); + CALL_SUBTEST_3(triangular_square(Matrix3d())); + CALL_SUBTEST_4(triangular_square(Matrix, 8, 8>())); + CALL_SUBTEST_5(triangular_square(MatrixXcd(r, r))); + CALL_SUBTEST_6(triangular_square(Matrix(r, r))); + + CALL_SUBTEST_7(triangular_rect(Matrix())); + CALL_SUBTEST_8(triangular_rect(Matrix())); + CALL_SUBTEST_9(triangular_rect(MatrixXcf(r, c))); + CALL_SUBTEST_5(triangular_rect(MatrixXcd(r, c))); + CALL_SUBTEST_6(triangular_rect(Matrix(r, c))); } - - CALL_SUBTEST_1( bug_159() ); + + CALL_SUBTEST_1(bug_159()); } diff --git a/filmulator-gui/core/nlmeans/eigen/test/umeyama.cpp b/filmulator-gui/core/nlmeans/eigen/test/umeyama.cpp index 2e809243..e08ece94 100644 --- a/filmulator-gui/core/nlmeans/eigen/test/umeyama.cpp +++ b/filmulator-gui/core/nlmeans/eigen/test/umeyama.cpp @@ -12,14 +12,13 @@ #include #include -#include // required for MatrixBase::determinant -#include // required for SVD +#include // required for MatrixBase::determinant +#include // required for SVD using namespace Eigen; // Constructs a random matrix from the unitary group U(size). -template -Eigen::Matrix randMatrixUnitary(int size) +template Eigen::Matrix randMatrixUnitary(int size) { typedef T Scalar; typedef Eigen::Matrix MatrixType; @@ -29,32 +28,27 @@ Eigen::Matrix randMatrixUnitary(int size) int max_tries = 40; double is_unitary = false; - while (!is_unitary && max_tries > 0) - { + while (!is_unitary && max_tries > 0) { // initialize random matrix Q = MatrixType::Random(size, size); // orthogonalize columns using the Gram-Schmidt algorithm - for (int col = 0; col < size; ++col) - { + for (int col = 0; col < size; ++col) { typename MatrixType::ColXpr colVec = Q.col(col); - for (int prevCol = 0; prevCol < col; ++prevCol) - { + for (int prevCol = 0; prevCol < col; ++prevCol) { typename MatrixType::ColXpr prevColVec = Q.col(prevCol); - colVec -= colVec.dot(prevColVec)*prevColVec; + colVec -= colVec.dot(prevColVec) * prevColVec; } Q.col(col) = colVec.normalized(); } // this additional orthogonalization is not necessary in theory but should enhance // the numerical orthogonality of the matrix - for (int row = 0; row < size; ++row) - { + for (int row = 0; row < size; ++row) { typename MatrixType::RowXpr rowVec = Q.row(row); - for (int prevRow = 0; prevRow < row; ++prevRow) - { + for (int prevRow = 0; prevRow < row; ++prevRow) { typename MatrixType::RowXpr prevRowVec = Q.row(prevRow); - rowVec -= rowVec.dot(prevRowVec)*prevRowVec; + rowVec -= rowVec.dot(prevRowVec) * prevRowVec; } Q.row(row) = rowVec.normalized(); } @@ -64,15 +58,13 @@ Eigen::Matrix randMatrixUnitary(int size) --max_tries; } - if (max_tries == 0) - eigen_assert(false && "randMatrixUnitary: Could not construct unitary matrix!"); + if (max_tries == 0) eigen_assert(false && "randMatrixUnitary: Could not construct unitary matrix!"); return Q; } // Constructs a random matrix from the special unitary group SU(size). -template -Eigen::Matrix randMatrixSpecialUnitary(int size) +template Eigen::Matrix randMatrixSpecialUnitary(int size) { typedef T Scalar; @@ -87,8 +79,7 @@ Eigen::Matrix randMatrixSpecialUnitary(int si return Q; } -template -void run_test(int dim, int num_elements) +template void run_test(int dim, int num_elements) { using std::abs; typedef typename internal::traits::Scalar Scalar; @@ -100,29 +91,28 @@ void run_test(int dim, int num_elements) const Scalar c = abs(internal::random()); MatrixX R = randMatrixSpecialUnitary(dim); - VectorX t = Scalar(50)*VectorX::Random(dim,1); + VectorX t = Scalar(50) * VectorX::Random(dim, 1); - MatrixX cR_t = MatrixX::Identity(dim+1,dim+1); - cR_t.block(0,0,dim,dim) = c*R; - cR_t.block(0,dim,dim,1) = t; + MatrixX cR_t = MatrixX::Identity(dim + 1, dim + 1); + cR_t.block(0, 0, dim, dim) = c * R; + cR_t.block(0, dim, dim, 1) = t; - MatrixX src = MatrixX::Random(dim+1, num_elements); + MatrixX src = MatrixX::Random(dim + 1, num_elements); src.row(dim) = Matrix::Constant(num_elements, Scalar(1)); - MatrixX dst = cR_t*src; + MatrixX dst = cR_t * src; - MatrixX cR_t_umeyama = umeyama(src.block(0,0,dim,num_elements), dst.block(0,0,dim,num_elements)); + MatrixX cR_t_umeyama = umeyama(src.block(0, 0, dim, num_elements), dst.block(0, 0, dim, num_elements)); - const Scalar error = ( cR_t_umeyama*src - dst ).norm() / dst.norm(); - VERIFY(error < Scalar(40)*std::numeric_limits::epsilon()); + const Scalar error = (cR_t_umeyama * src - dst).norm() / dst.norm(); + VERIFY(error < Scalar(40) * std::numeric_limits::epsilon()); } -template -void run_fixed_size_test(int num_elements) +template void run_fixed_size_test(int num_elements) { using std::abs; - typedef Matrix MatrixX; - typedef Matrix HomMatrix; + typedef Matrix MatrixX; + typedef Matrix HomMatrix; typedef Matrix FixedMatrix; typedef Matrix FixedVector; @@ -134,36 +124,34 @@ void run_fixed_size_test(int num_elements) const Scalar c = internal::random(0.5, 2.0); FixedMatrix R = randMatrixSpecialUnitary(dim); - FixedVector t = Scalar(32)*FixedVector::Random(dim,1); + FixedVector t = Scalar(32) * FixedVector::Random(dim, 1); - HomMatrix cR_t = HomMatrix::Identity(dim+1,dim+1); - cR_t.block(0,0,dim,dim) = c*R; - cR_t.block(0,dim,dim,1) = t; + HomMatrix cR_t = HomMatrix::Identity(dim + 1, dim + 1); + cR_t.block(0, 0, dim, dim) = c * R; + cR_t.block(0, dim, dim, 1) = t; - MatrixX src = MatrixX::Random(dim+1, num_elements); + MatrixX src = MatrixX::Random(dim + 1, num_elements); src.row(dim) = Matrix::Constant(num_elements, Scalar(1)); - MatrixX dst = cR_t*src; + MatrixX dst = cR_t * src; - Block src_block(src,0,0,dim,num_elements); - Block dst_block(dst,0,0,dim,num_elements); + Block src_block(src, 0, 0, dim, num_elements); + Block dst_block(dst, 0, 0, dim, num_elements); HomMatrix cR_t_umeyama = umeyama(src_block, dst_block); - const Scalar error = ( cR_t_umeyama*src - dst ).squaredNorm(); + const Scalar error = (cR_t_umeyama * src - dst).squaredNorm(); - VERIFY(error < Scalar(16)*std::numeric_limits::epsilon()); + VERIFY(error < Scalar(16) * std::numeric_limits::epsilon()); } void test_umeyama() { - for (int i=0; i(40,500); + for (int i = 0; i < g_repeat; ++i) { + const int num_elements = internal::random(40, 500); // works also for dimensions bigger than 3... - for (int dim=2; dim<8; ++dim) - { + for (int dim = 2; dim < 8; ++dim) { CALL_SUBTEST_1(run_test(dim, num_elements)); CALL_SUBTEST_2(run_test(dim, num_elements)); } diff --git a/filmulator-gui/core/nlmeans/eigen/test/umfpack_support.cpp b/filmulator-gui/core/nlmeans/eigen/test/umfpack_support.cpp index 37ab11f0..eac57639 100644 --- a/filmulator-gui/core/nlmeans/eigen/test/umfpack_support.cpp +++ b/filmulator-gui/core/nlmeans/eigen/test/umfpack_support.cpp @@ -14,12 +14,12 @@ template void test_umfpack_support_T() { - UmfPackLU > umfpack_colmajor; - UmfPackLU > umfpack_rowmajor; - + UmfPackLU> umfpack_colmajor; + UmfPackLU> umfpack_rowmajor; + check_sparse_square_solving(umfpack_colmajor); check_sparse_square_solving(umfpack_rowmajor); - + check_sparse_square_determinant(umfpack_colmajor); check_sparse_square_determinant(umfpack_rowmajor); } @@ -27,6 +27,5 @@ template void test_umfpack_support_T() void test_umfpack_support() { CALL_SUBTEST_1(test_umfpack_support_T()); - CALL_SUBTEST_2(test_umfpack_support_T >()); + CALL_SUBTEST_2(test_umfpack_support_T>()); } - diff --git a/filmulator-gui/core/nlmeans/eigen/test/unalignedassert.cpp b/filmulator-gui/core/nlmeans/eigen/test/unalignedassert.cpp index 731a0897..dccaff33 100644 --- a/filmulator-gui/core/nlmeans/eigen/test/unalignedassert.cpp +++ b/filmulator-gui/core/nlmeans/eigen/test/unalignedassert.cpp @@ -9,66 +9,66 @@ // with this file, You can obtain one at http://mozilla.org/MPL/2.0/. #if defined(EIGEN_TEST_PART_1) - // default +// default #elif defined(EIGEN_TEST_PART_2) - #define EIGEN_MAX_STATIC_ALIGN_BYTES 16 - #define EIGEN_MAX_ALIGN_BYTES 16 +#define EIGEN_MAX_STATIC_ALIGN_BYTES 16 +#define EIGEN_MAX_ALIGN_BYTES 16 #elif defined(EIGEN_TEST_PART_3) - #define EIGEN_MAX_STATIC_ALIGN_BYTES 32 - #define EIGEN_MAX_ALIGN_BYTES 32 +#define EIGEN_MAX_STATIC_ALIGN_BYTES 32 +#define EIGEN_MAX_ALIGN_BYTES 32 #elif defined(EIGEN_TEST_PART_4) - #define EIGEN_MAX_STATIC_ALIGN_BYTES 64 - #define EIGEN_MAX_ALIGN_BYTES 64 +#define EIGEN_MAX_STATIC_ALIGN_BYTES 64 +#define EIGEN_MAX_ALIGN_BYTES 64 #endif #include "main.h" -typedef Matrix Vector6f; -typedef Matrix Vector8f; -typedef Matrix Vector12f; +typedef Matrix Vector6f; +typedef Matrix Vector8f; +typedef Matrix Vector12f; -typedef Matrix Vector5d; -typedef Matrix Vector6d; -typedef Matrix Vector7d; -typedef Matrix Vector8d; -typedef Matrix Vector9d; -typedef Matrix Vector10d; -typedef Matrix Vector12d; +typedef Matrix Vector5d; +typedef Matrix Vector6d; +typedef Matrix Vector7d; +typedef Matrix Vector8d; +typedef Matrix Vector9d; +typedef Matrix Vector10d; +typedef Matrix Vector12d; struct TestNew1 { - MatrixXd m; // good: m will allocate its own array, taking care of alignment. - TestNew1() : m(20,20) {} + MatrixXd m;// good: m will allocate its own array, taking care of alignment. + TestNew1() : m(20, 20) {} }; struct TestNew2 { - Matrix3d m; // good: m's size isn't a multiple of 16 bytes, so m doesn't have to be 16-byte aligned, - // 8-byte alignment is good enough here, which we'll get automatically + Matrix3d m;// good: m's size isn't a multiple of 16 bytes, so m doesn't have to be 16-byte aligned, + // 8-byte alignment is good enough here, which we'll get automatically }; struct TestNew3 { - Vector2f m; // good: m's size isn't a multiple of 16 bytes, so m doesn't have to be 16-byte aligned + Vector2f m;// good: m's size isn't a multiple of 16 bytes, so m doesn't have to be 16-byte aligned }; struct TestNew4 { EIGEN_MAKE_ALIGNED_OPERATOR_NEW Vector2d m; - float f; // make the struct have sizeof%16!=0 to make it a little more tricky when we allow an array of 2 such objects + float f;// make the struct have sizeof%16!=0 to make it a little more tricky when we allow an array of 2 such objects }; struct TestNew5 { EIGEN_MAKE_ALIGNED_OPERATOR_NEW - float f; // try the f at first -- the EIGEN_ALIGN_MAX attribute of m should make that still work + float f;// try the f at first -- the EIGEN_ALIGN_MAX attribute of m should make that still work Matrix4f m; }; struct TestNew6 { - Matrix m; // good: no alignment requested + Matrix m;// good: no alignment requested float f; }; @@ -79,8 +79,7 @@ template struct Depends float f; }; -template -void check_unalignedassert_good() +template void check_unalignedassert_good() { T *x, *y; x = new T; @@ -89,23 +88,22 @@ void check_unalignedassert_good() delete[] y; } -#if EIGEN_MAX_STATIC_ALIGN_BYTES>0 -template -void construct_at_boundary(int boundary) +#if EIGEN_MAX_STATIC_ALIGN_BYTES > 0 +template void construct_at_boundary(int boundary) { - char buf[sizeof(T)+256]; + char buf[sizeof(T) + 256]; size_t _buf = reinterpret_cast(buf); - _buf += (EIGEN_MAX_ALIGN_BYTES - (_buf % EIGEN_MAX_ALIGN_BYTES)); // make 16/32/...-byte aligned - _buf += boundary; // make exact boundary-aligned - T *x = ::new(reinterpret_cast(_buf)) T; - x[0].setZero(); // just in order to silence warnings + _buf += (EIGEN_MAX_ALIGN_BYTES - (_buf % EIGEN_MAX_ALIGN_BYTES));// make 16/32/...-byte aligned + _buf += boundary;// make exact boundary-aligned + T *x = ::new (reinterpret_cast(_buf)) T; + x[0].setZero();// just in order to silence warnings x->~T(); } #endif void unalignedassert() { -#if EIGEN_MAX_STATIC_ALIGN_BYTES>0 +#if EIGEN_MAX_STATIC_ALIGN_BYTES > 0 construct_at_boundary(4); construct_at_boundary(4); construct_at_boundary(16); @@ -143,11 +141,10 @@ void unalignedassert() check_unalignedassert_good(); check_unalignedassert_good(); check_unalignedassert_good(); - check_unalignedassert_good >(); + check_unalignedassert_good>(); -#if EIGEN_MAX_STATIC_ALIGN_BYTES>0 - if(EIGEN_MAX_ALIGN_BYTES>=16) - { +#if EIGEN_MAX_STATIC_ALIGN_BYTES > 0 + if (EIGEN_MAX_ALIGN_BYTES >= 16) { VERIFY_RAISES_ASSERT(construct_at_boundary(8)); VERIFY_RAISES_ASSERT(construct_at_boundary(8)); VERIFY_RAISES_ASSERT(construct_at_boundary(8)); @@ -159,22 +156,18 @@ void unalignedassert() VERIFY_RAISES_ASSERT(construct_at_boundary(8)); // Complexes are disabled because the compiler might aggressively vectorize // the initialization of complex coeffs to 0 before we can check for alignedness - //VERIFY_RAISES_ASSERT(construct_at_boundary(8)); + // VERIFY_RAISES_ASSERT(construct_at_boundary(8)); VERIFY_RAISES_ASSERT(construct_at_boundary(8)); } - for(int b=8; b(b)); - if(b<64) VERIFY_RAISES_ASSERT(construct_at_boundary(b)); - if(b<32) VERIFY_RAISES_ASSERT(construct_at_boundary(b)); - if(b<32) VERIFY_RAISES_ASSERT(construct_at_boundary(b)); - if(b<128) VERIFY_RAISES_ASSERT(construct_at_boundary(b)); - //if(b<32) VERIFY_RAISES_ASSERT(construct_at_boundary(b)); + for (int b = 8; b < EIGEN_MAX_ALIGN_BYTES; b += 8) { + if (b < 32) VERIFY_RAISES_ASSERT(construct_at_boundary(b)); + if (b < 64) VERIFY_RAISES_ASSERT(construct_at_boundary(b)); + if (b < 32) VERIFY_RAISES_ASSERT(construct_at_boundary(b)); + if (b < 32) VERIFY_RAISES_ASSERT(construct_at_boundary(b)); + if (b < 128) VERIFY_RAISES_ASSERT(construct_at_boundary(b)); + // if(b<32) VERIFY_RAISES_ASSERT(construct_at_boundary(b)); } #endif } -void test_unalignedassert() -{ - CALL_SUBTEST(unalignedassert()); -} +void test_unalignedassert() { CALL_SUBTEST(unalignedassert()); } diff --git a/filmulator-gui/core/nlmeans/eigen/test/unalignedcount.cpp b/filmulator-gui/core/nlmeans/eigen/test/unalignedcount.cpp index d6ffeafd..5a04ab06 100644 --- a/filmulator-gui/core/nlmeans/eigen/test/unalignedcount.cpp +++ b/filmulator-gui/core/nlmeans/eigen/test/unalignedcount.cpp @@ -12,17 +12,30 @@ static int nb_loadu; static int nb_store; static int nb_storeu; -#define EIGEN_DEBUG_ALIGNED_LOAD { nb_load++; } -#define EIGEN_DEBUG_UNALIGNED_LOAD { nb_loadu++; } -#define EIGEN_DEBUG_ALIGNED_STORE { nb_store++; } -#define EIGEN_DEBUG_UNALIGNED_STORE { nb_storeu++; } +#define EIGEN_DEBUG_ALIGNED_LOAD \ + { \ + nb_load++; \ + } +#define EIGEN_DEBUG_UNALIGNED_LOAD \ + { \ + nb_loadu++; \ + } +#define EIGEN_DEBUG_ALIGNED_STORE \ + { \ + nb_store++; \ + } +#define EIGEN_DEBUG_UNALIGNED_STORE \ + { \ + nb_storeu++; \ + } -#define VERIFY_ALIGNED_UNALIGNED_COUNT(XPR,AL,UL,AS,US) {\ - nb_load = nb_loadu = nb_store = nb_storeu = 0; \ - XPR; \ - if(!(nb_load==AL && nb_loadu==UL && nb_store==AS && nb_storeu==US)) \ +#define VERIFY_ALIGNED_UNALIGNED_COUNT(XPR, AL, UL, AS, US) \ + { \ + nb_load = nb_loadu = nb_store = nb_storeu = 0; \ + XPR; \ + if (!(nb_load == AL && nb_loadu == UL && nb_store == AS && nb_storeu == US)) \ std::cerr << " >> " << nb_load << ", " << nb_loadu << ", " << nb_store << ", " << nb_storeu << "\n"; \ - VERIFY( (#XPR) && nb_load==AL && nb_loadu==UL && nb_store==AS && nb_storeu==US ); \ + VERIFY((#XPR) && nb_load == AL && nb_loadu == UL && nb_store == AS && nb_storeu == US); \ } @@ -30,24 +43,24 @@ static int nb_storeu; void test_unalignedcount() { - #if defined(EIGEN_VECTORIZE_AVX) +#if defined(EIGEN_VECTORIZE_AVX) VectorXf a(40), b(40); VERIFY_ALIGNED_UNALIGNED_COUNT(a += b, 10, 0, 5, 0); - VERIFY_ALIGNED_UNALIGNED_COUNT(a.segment(0,40) += b.segment(0,40), 5, 5, 5, 0); - VERIFY_ALIGNED_UNALIGNED_COUNT(a.segment(0,40) -= b.segment(0,40), 5, 5, 5, 0); - VERIFY_ALIGNED_UNALIGNED_COUNT(a.segment(0,40) *= 3.5, 5, 0, 5, 0); - VERIFY_ALIGNED_UNALIGNED_COUNT(a.segment(0,40) /= 3.5, 5, 0, 5, 0); - #elif defined(EIGEN_VECTORIZE_SSE) + VERIFY_ALIGNED_UNALIGNED_COUNT(a.segment(0, 40) += b.segment(0, 40), 5, 5, 5, 0); + VERIFY_ALIGNED_UNALIGNED_COUNT(a.segment(0, 40) -= b.segment(0, 40), 5, 5, 5, 0); + VERIFY_ALIGNED_UNALIGNED_COUNT(a.segment(0, 40) *= 3.5, 5, 0, 5, 0); + VERIFY_ALIGNED_UNALIGNED_COUNT(a.segment(0, 40) /= 3.5, 5, 0, 5, 0); +#elif defined(EIGEN_VECTORIZE_SSE) VectorXf a(40), b(40); VERIFY_ALIGNED_UNALIGNED_COUNT(a += b, 20, 0, 10, 0); - VERIFY_ALIGNED_UNALIGNED_COUNT(a.segment(0,40) += b.segment(0,40), 10, 10, 10, 0); - VERIFY_ALIGNED_UNALIGNED_COUNT(a.segment(0,40) -= b.segment(0,40), 10, 10, 10, 0); - VERIFY_ALIGNED_UNALIGNED_COUNT(a.segment(0,40) *= 3.5, 10, 0, 10, 0); - VERIFY_ALIGNED_UNALIGNED_COUNT(a.segment(0,40) /= 3.5, 10, 0, 10, 0); - #else + VERIFY_ALIGNED_UNALIGNED_COUNT(a.segment(0, 40) += b.segment(0, 40), 10, 10, 10, 0); + VERIFY_ALIGNED_UNALIGNED_COUNT(a.segment(0, 40) -= b.segment(0, 40), 10, 10, 10, 0); + VERIFY_ALIGNED_UNALIGNED_COUNT(a.segment(0, 40) *= 3.5, 10, 0, 10, 0); + VERIFY_ALIGNED_UNALIGNED_COUNT(a.segment(0, 40) /= 3.5, 10, 0, 10, 0); +#else // The following line is to eliminate "variable not used" warnings nb_load = nb_loadu = nb_store = nb_storeu = 0; int a(0), b(0); - VERIFY(a==b); - #endif + VERIFY(a == b); +#endif } diff --git a/filmulator-gui/core/nlmeans/eigen/test/upperbidiagonalization.cpp b/filmulator-gui/core/nlmeans/eigen/test/upperbidiagonalization.cpp index 847b34b5..4826cccf 100644 --- a/filmulator-gui/core/nlmeans/eigen/test/upperbidiagonalization.cpp +++ b/filmulator-gui/core/nlmeans/eigen/test/upperbidiagonalization.cpp @@ -10,34 +10,36 @@ #include "main.h" #include -template void upperbidiag(const MatrixType& m) +template void upperbidiag(const MatrixType &m) { const typename MatrixType::Index rows = m.rows(); const typename MatrixType::Index cols = m.cols(); - typedef Matrix RealMatrixType; - typedef Matrix TransposeMatrixType; + typedef Matrix + RealMatrixType; + typedef Matrix + TransposeMatrixType; - MatrixType a = MatrixType::Random(rows,cols); + MatrixType a = MatrixType::Random(rows, cols); internal::UpperBidiagonalization ubd(a); RealMatrixType b(rows, cols); b.setZero(); - b.block(0,0,cols,cols) = ubd.bidiagonal(); + b.block(0, 0, cols, cols) = ubd.bidiagonal(); MatrixType c = ubd.householderU() * b * ubd.householderV().adjoint(); - VERIFY_IS_APPROX(a,c); + VERIFY_IS_APPROX(a, c); TransposeMatrixType d = ubd.householderV() * b.adjoint() * ubd.householderU().adjoint(); - VERIFY_IS_APPROX(a.adjoint(),d); + VERIFY_IS_APPROX(a.adjoint(), d); } void test_upperbidiagonalization() { - for(int i = 0; i < g_repeat; i++) { - CALL_SUBTEST_1( upperbidiag(MatrixXf(3,3)) ); - CALL_SUBTEST_2( upperbidiag(MatrixXd(17,12)) ); - CALL_SUBTEST_3( upperbidiag(MatrixXcf(20,20)) ); - CALL_SUBTEST_4( upperbidiag(Matrix,Dynamic,Dynamic,RowMajor>(16,15)) ); - CALL_SUBTEST_5( upperbidiag(Matrix()) ); - CALL_SUBTEST_6( upperbidiag(Matrix()) ); - CALL_SUBTEST_7( upperbidiag(Matrix()) ); + for (int i = 0; i < g_repeat; i++) { + CALL_SUBTEST_1(upperbidiag(MatrixXf(3, 3))); + CALL_SUBTEST_2(upperbidiag(MatrixXd(17, 12))); + CALL_SUBTEST_3(upperbidiag(MatrixXcf(20, 20))); + CALL_SUBTEST_4(upperbidiag(Matrix, Dynamic, Dynamic, RowMajor>(16, 15))); + CALL_SUBTEST_5(upperbidiag(Matrix())); + CALL_SUBTEST_6(upperbidiag(Matrix())); + CALL_SUBTEST_7(upperbidiag(Matrix())); } } diff --git a/filmulator-gui/core/nlmeans/eigen/test/vectorization_logic.cpp b/filmulator-gui/core/nlmeans/eigen/test/vectorization_logic.cpp index 37e7495f..31dc77d5 100644 --- a/filmulator-gui/core/nlmeans/eigen/test/vectorization_logic.cpp +++ b/filmulator-gui/core/nlmeans/eigen/test/vectorization_logic.cpp @@ -26,76 +26,75 @@ using internal::demangle_flags; using internal::demangle_traversal; using internal::demangle_unrolling; -template -bool test_assign(const Dst&, const Src&, int traversal, int unrolling) +template bool test_assign(const Dst &, const Src &, int traversal, int unrolling) { - typedef internal::copy_using_evaluator_traits,internal::evaluator, internal::assign_op > traits; - bool res = traits::Traversal==traversal; - if(unrolling==InnerUnrolling+CompleteUnrolling) - res = res && (int(traits::Unrolling)==InnerUnrolling || int(traits::Unrolling)==CompleteUnrolling); + typedef internal::copy_using_evaluator_traits, + internal::evaluator, + internal::assign_op> + traits; + bool res = traits::Traversal == traversal; + if (unrolling == InnerUnrolling + CompleteUnrolling) + res = res && (int(traits::Unrolling) == InnerUnrolling || int(traits::Unrolling) == CompleteUnrolling); else - res = res && int(traits::Unrolling)==unrolling; - if(!res) - { + res = res && int(traits::Unrolling) == unrolling; + if (!res) { std::cerr << "Src: " << demangle_flags(Src::Flags) << std::endl; std::cerr << " " << demangle_flags(internal::evaluator::Flags) << std::endl; std::cerr << "Dst: " << demangle_flags(Dst::Flags) << std::endl; std::cerr << " " << demangle_flags(internal::evaluator::Flags) << std::endl; traits::debug(); - std::cerr << " Expected Traversal == " << demangle_traversal(traversal) - << " got " << demangle_traversal(traits::Traversal) << "\n"; - std::cerr << " Expected Unrolling == " << demangle_unrolling(unrolling) - << " got " << demangle_unrolling(traits::Unrolling) << "\n"; + std::cerr << " Expected Traversal == " << demangle_traversal(traversal) << " got " + << demangle_traversal(traits::Traversal) << "\n"; + std::cerr << " Expected Unrolling == " << demangle_unrolling(unrolling) << " got " + << demangle_unrolling(traits::Unrolling) << "\n"; } return res; } -template -bool test_assign(int traversal, int unrolling) +template bool test_assign(int traversal, int unrolling) { - typedef internal::copy_using_evaluator_traits,internal::evaluator, internal::assign_op > traits; - bool res = traits::Traversal==traversal && traits::Unrolling==unrolling; - if(!res) - { + typedef internal::copy_using_evaluator_traits, + internal::evaluator, + internal::assign_op> + traits; + bool res = traits::Traversal == traversal && traits::Unrolling == unrolling; + if (!res) { std::cerr << "Src: " << demangle_flags(Src::Flags) << std::endl; std::cerr << " " << demangle_flags(internal::evaluator::Flags) << std::endl; std::cerr << "Dst: " << demangle_flags(Dst::Flags) << std::endl; std::cerr << " " << demangle_flags(internal::evaluator::Flags) << std::endl; traits::debug(); - std::cerr << " Expected Traversal == " << demangle_traversal(traversal) - << " got " << demangle_traversal(traits::Traversal) << "\n"; - std::cerr << " Expected Unrolling == " << demangle_unrolling(unrolling) - << " got " << demangle_unrolling(traits::Unrolling) << "\n"; + std::cerr << " Expected Traversal == " << demangle_traversal(traversal) << " got " + << demangle_traversal(traits::Traversal) << "\n"; + std::cerr << " Expected Unrolling == " << demangle_unrolling(unrolling) << " got " + << demangle_unrolling(traits::Unrolling) << "\n"; } return res; } -template -bool test_redux(const Xpr&, int traversal, int unrolling) +template bool test_redux(const Xpr &, int traversal, int unrolling) { typedef typename Xpr::Scalar Scalar; - typedef internal::redux_traits,internal::redux_evaluator > traits; - - bool res = traits::Traversal==traversal && traits::Unrolling==unrolling; - if(!res) - { + typedef internal::redux_traits, internal::redux_evaluator> traits; + + bool res = traits::Traversal == traversal && traits::Unrolling == unrolling; + if (!res) { std::cerr << demangle_flags(Xpr::Flags) << std::endl; std::cerr << demangle_flags(internal::evaluator::Flags) << std::endl; traits::debug(); - - std::cerr << " Expected Traversal == " << demangle_traversal(traversal) - << " got " << demangle_traversal(traits::Traversal) << "\n"; - std::cerr << " Expected Unrolling == " << demangle_unrolling(unrolling) - << " got " << demangle_unrolling(traits::Unrolling) << "\n"; + + std::cerr << " Expected Traversal == " << demangle_traversal(traversal) << " got " + << demangle_traversal(traits::Traversal) << "\n"; + std::cerr << " Expected Unrolling == " << demangle_unrolling(unrolling) << " got " + << demangle_unrolling(traits::Unrolling) << "\n"; } return res; } -template::Vectorizable> -struct vectorization_logic +template::Vectorizable> struct vectorization_logic { typedef internal::packet_traits PacketTraits; - + typedef typename internal::packet_traits::type PacketType; typedef typename internal::unpacket_traits::half HalfPacketType; enum { @@ -104,284 +103,337 @@ struct vectorization_logic }; static void run() { - - typedef Matrix Vector1; - typedef Matrix VectorX; - typedef Matrix MatrixXX; - typedef Matrix Matrix11; - typedef Matrix Matrix22; - typedef Matrix Matrix44; - typedef Matrix Matrix44u; - typedef Matrix Matrix44c; - typedef Matrix Matrix44r; + typedef Matrix Vector1; + typedef Matrix VectorX; + typedef Matrix MatrixXX; + typedef Matrix Matrix11; + typedef Matrix Matrix22; + typedef Matrix + Matrix44; typedef Matrix Matrix1; + (Matrix11::Flags & RowMajorBit) ? 16 : 4 * PacketSize, + (Matrix11::Flags & RowMajorBit) ? 4 * PacketSize : 16, + DontAlign | EIGEN_DEFAULT_MATRIX_STORAGE_ORDER_OPTION> + Matrix44u; + typedef Matrix Matrix44c; + typedef Matrix Matrix44r; typedef Matrix Matrix1u; + (PacketSize == 8 ? 4 + : PacketSize == 4 ? 2 + : PacketSize == 2 ? 1 + : /*PacketSize==1 ?*/ 1), + (PacketSize == 8 ? 2 + : PacketSize == 4 ? 2 + : PacketSize == 2 ? 2 + : /*PacketSize==1 ?*/ 1)> + Matrix1; + + typedef Matrix + Matrix1u; // this type is made such that it can only be vectorized when viewed as a linear 1D vector typedef Matrix Matrix3; - - #if !EIGEN_GCC_AND_ARCH_DOESNT_WANT_STACK_ALIGNMENT - VERIFY(test_assign(Vector1(),Vector1(), - InnerVectorizedTraversal,CompleteUnrolling)); - VERIFY(test_assign(Vector1(),Vector1()+Vector1(), - InnerVectorizedTraversal,CompleteUnrolling)); - VERIFY(test_assign(Vector1(),Vector1().cwiseProduct(Vector1()), - InnerVectorizedTraversal,CompleteUnrolling)); - VERIFY(test_assign(Vector1(),Vector1().template cast(), - InnerVectorizedTraversal,CompleteUnrolling)); - - - VERIFY(test_assign(Vector1(),Vector1(), - InnerVectorizedTraversal,CompleteUnrolling)); - VERIFY(test_assign(Vector1(),Vector1()+Vector1(), - InnerVectorizedTraversal,CompleteUnrolling)); - VERIFY(test_assign(Vector1(),Vector1().cwiseProduct(Vector1()), - InnerVectorizedTraversal,CompleteUnrolling)); - - VERIFY(test_assign(Matrix44(),Matrix44()+Matrix44(), - InnerVectorizedTraversal,InnerUnrolling)); - - VERIFY(test_assign(Matrix44u(),Matrix44()+Matrix44(), + (PacketSize == 8 ? 4 + : PacketSize == 4 ? 6 + : PacketSize == 2 ? ((Matrix11::Flags & RowMajorBit) ? 2 : 3) + : /*PacketSize==1 ?*/ 1), + (PacketSize == 8 ? 6 + : PacketSize == 4 ? 2 + : PacketSize == 2 ? ((Matrix11::Flags & RowMajorBit) ? 3 : 2) + : /*PacketSize==1 ?*/ 3)> + Matrix3; + +#if !EIGEN_GCC_AND_ARCH_DOESNT_WANT_STACK_ALIGNMENT + VERIFY(test_assign(Vector1(), Vector1(), InnerVectorizedTraversal, CompleteUnrolling)); + VERIFY(test_assign(Vector1(), Vector1() + Vector1(), InnerVectorizedTraversal, CompleteUnrolling)); + VERIFY(test_assign(Vector1(), Vector1().cwiseProduct(Vector1()), InnerVectorizedTraversal, CompleteUnrolling)); + VERIFY(test_assign(Vector1(), Vector1().template cast(), InnerVectorizedTraversal, CompleteUnrolling)); + + + VERIFY(test_assign(Vector1(), Vector1(), InnerVectorizedTraversal, CompleteUnrolling)); + VERIFY(test_assign(Vector1(), Vector1() + Vector1(), InnerVectorizedTraversal, CompleteUnrolling)); + VERIFY(test_assign(Vector1(), Vector1().cwiseProduct(Vector1()), InnerVectorizedTraversal, CompleteUnrolling)); + + VERIFY(test_assign(Matrix44(), Matrix44() + Matrix44(), InnerVectorizedTraversal, InnerUnrolling)); + + VERIFY(test_assign(Matrix44u(), + Matrix44() + Matrix44(), EIGEN_UNALIGNED_VECTORIZE ? InnerVectorizedTraversal : LinearTraversal, EIGEN_UNALIGNED_VECTORIZE ? InnerUnrolling : NoUnrolling)); - VERIFY(test_assign(Matrix1(),Matrix1()+Matrix1(), - (Matrix1::InnerSizeAtCompileTime % PacketSize)==0 ? InnerVectorizedTraversal : LinearVectorizedTraversal, + VERIFY(test_assign(Matrix1(), + Matrix1() + Matrix1(), + (Matrix1::InnerSizeAtCompileTime % PacketSize) == 0 ? InnerVectorizedTraversal : LinearVectorizedTraversal, CompleteUnrolling)); - VERIFY(test_assign(Matrix1u(),Matrix1()+Matrix1(), - EIGEN_UNALIGNED_VECTORIZE ? ((Matrix1::InnerSizeAtCompileTime % PacketSize)==0 ? InnerVectorizedTraversal : LinearVectorizedTraversal) - : LinearTraversal, CompleteUnrolling)); - - VERIFY(test_assign(Matrix44c().col(1),Matrix44c().col(2)+Matrix44c().col(3), - InnerVectorizedTraversal,CompleteUnrolling)); - - VERIFY(test_assign(Matrix44r().row(2),Matrix44r().row(1)+Matrix44r().row(1), - InnerVectorizedTraversal,CompleteUnrolling)); - - if(PacketSize>1) - { - typedef Matrix Matrix33c; - typedef Matrix Vector3; - VERIFY(test_assign(Matrix33c().row(2),Matrix33c().row(1)+Matrix33c().row(1), - LinearTraversal,CompleteUnrolling)); - VERIFY(test_assign(Vector3(),Vector3()+Vector3(), - EIGEN_UNALIGNED_VECTORIZE ? (HalfPacketSize==1 ? InnerVectorizedTraversal : LinearVectorizedTraversal) : (HalfPacketSize==1 ? InnerVectorizedTraversal : LinearTraversal), CompleteUnrolling)); - VERIFY(test_assign(Matrix33c().col(0),Matrix33c().col(1)+Matrix33c().col(1), - EIGEN_UNALIGNED_VECTORIZE ? (HalfPacketSize==1 ? InnerVectorizedTraversal : LinearVectorizedTraversal) : (HalfPacketSize==1 ? SliceVectorizedTraversal : LinearTraversal), - ((!EIGEN_UNALIGNED_VECTORIZE) && HalfPacketSize==1) ? NoUnrolling : CompleteUnrolling)); - - VERIFY(test_assign(Matrix3(),Matrix3().cwiseProduct(Matrix3()), - LinearVectorizedTraversal,CompleteUnrolling)); - - VERIFY(test_assign(Matrix(),Matrix()+Matrix(), - HalfPacketSize==1 ? InnerVectorizedTraversal : - EIGEN_UNALIGNED_VECTORIZE ? LinearVectorizedTraversal : - LinearTraversal, + VERIFY(test_assign(Matrix1u(), + Matrix1() + Matrix1(), + EIGEN_UNALIGNED_VECTORIZE + ? ((Matrix1::InnerSizeAtCompileTime % PacketSize) == 0 ? InnerVectorizedTraversal : LinearVectorizedTraversal) + : LinearTraversal, + CompleteUnrolling)); + + VERIFY(test_assign( + Matrix44c().col(1), Matrix44c().col(2) + Matrix44c().col(3), InnerVectorizedTraversal, CompleteUnrolling)); + + VERIFY(test_assign( + Matrix44r().row(2), Matrix44r().row(1) + Matrix44r().row(1), InnerVectorizedTraversal, CompleteUnrolling)); + + if (PacketSize > 1) { + typedef Matrix Matrix33c; + typedef Matrix Vector3; + VERIFY( + test_assign(Matrix33c().row(2), Matrix33c().row(1) + Matrix33c().row(1), LinearTraversal, CompleteUnrolling)); + VERIFY(test_assign(Vector3(), + Vector3() + Vector3(), + EIGEN_UNALIGNED_VECTORIZE ? (HalfPacketSize == 1 ? InnerVectorizedTraversal : LinearVectorizedTraversal) + : (HalfPacketSize == 1 ? InnerVectorizedTraversal : LinearTraversal), + CompleteUnrolling)); + VERIFY(test_assign(Matrix33c().col(0), + Matrix33c().col(1) + Matrix33c().col(1), + EIGEN_UNALIGNED_VECTORIZE ? (HalfPacketSize == 1 ? InnerVectorizedTraversal : LinearVectorizedTraversal) + : (HalfPacketSize == 1 ? SliceVectorizedTraversal : LinearTraversal), + ((!EIGEN_UNALIGNED_VECTORIZE) && HalfPacketSize == 1) ? NoUnrolling : CompleteUnrolling)); + + VERIFY(test_assign(Matrix3(), Matrix3().cwiseProduct(Matrix3()), LinearVectorizedTraversal, CompleteUnrolling)); + + VERIFY(test_assign(Matrix(), + Matrix() + Matrix(), + HalfPacketSize == 1 ? InnerVectorizedTraversal + : EIGEN_UNALIGNED_VECTORIZE ? LinearVectorizedTraversal + : LinearTraversal, NoUnrolling)); - VERIFY(test_assign(Matrix11(), Matrix11()+Matrix11(),InnerVectorizedTraversal,CompleteUnrolling)); + VERIFY(test_assign(Matrix11(), Matrix11() + Matrix11(), InnerVectorizedTraversal, CompleteUnrolling)); - VERIFY(test_assign(Matrix11(),Matrix().template block(2,3)+Matrix().template block(8,4), - (EIGEN_UNALIGNED_VECTORIZE) ? InnerVectorizedTraversal : DefaultTraversal, CompleteUnrolling|InnerUnrolling)); + VERIFY(test_assign(Matrix11(), + Matrix().template block(2, 3) + + Matrix().template block(8, 4), + (EIGEN_UNALIGNED_VECTORIZE) ? InnerVectorizedTraversal : DefaultTraversal, + CompleteUnrolling | InnerUnrolling)); - VERIFY(test_assign(Vector1(),Matrix11()*Vector1(), - InnerVectorizedTraversal,CompleteUnrolling)); + VERIFY(test_assign(Vector1(), Matrix11() * Vector1(), InnerVectorizedTraversal, CompleteUnrolling)); - VERIFY(test_assign(Matrix11(),Matrix11().lazyProduct(Matrix11()), - InnerVectorizedTraversal,InnerUnrolling+CompleteUnrolling)); + VERIFY(test_assign( + Matrix11(), Matrix11().lazyProduct(Matrix11()), InnerVectorizedTraversal, InnerUnrolling + CompleteUnrolling)); } - VERIFY(test_redux(Vector1(), - LinearVectorizedTraversal,CompleteUnrolling)); + VERIFY(test_redux(Vector1(), LinearVectorizedTraversal, CompleteUnrolling)); - VERIFY(test_redux(Vector1().array()*Vector1().array(), - LinearVectorizedTraversal,CompleteUnrolling)); + VERIFY(test_redux(Vector1().array() * Vector1().array(), LinearVectorizedTraversal, CompleteUnrolling)); - VERIFY(test_redux((Vector1().array()*Vector1().array()).col(0), - LinearVectorizedTraversal,CompleteUnrolling)); + VERIFY(test_redux((Vector1().array() * Vector1().array()).col(0), LinearVectorizedTraversal, CompleteUnrolling)); - VERIFY(test_redux(Matrix(), - LinearVectorizedTraversal,CompleteUnrolling)); + VERIFY(test_redux(Matrix(), LinearVectorizedTraversal, CompleteUnrolling)); - VERIFY(test_redux(Matrix3(), - LinearVectorizedTraversal,CompleteUnrolling)); + VERIFY(test_redux(Matrix3(), LinearVectorizedTraversal, CompleteUnrolling)); - VERIFY(test_redux(Matrix44(), - LinearVectorizedTraversal,NoUnrolling)); + VERIFY(test_redux(Matrix44(), LinearVectorizedTraversal, NoUnrolling)); - VERIFY(test_redux(Matrix44().template block<(Matrix1::Flags&RowMajorBit)?4:PacketSize,(Matrix1::Flags&RowMajorBit)?PacketSize:4>(1,2), - DefaultTraversal,CompleteUnrolling)); + VERIFY(test_redux(Matrix44() + .template block<(Matrix1::Flags & RowMajorBit) ? 4 : PacketSize, + (Matrix1::Flags & RowMajorBit) ? PacketSize : 4>(1, 2), + DefaultTraversal, + CompleteUnrolling)); - VERIFY(test_redux(Matrix44c().template block<2*PacketSize,1>(1,2), - LinearVectorizedTraversal,CompleteUnrolling)); + VERIFY( + test_redux(Matrix44c().template block<2 * PacketSize, 1>(1, 2), LinearVectorizedTraversal, CompleteUnrolling)); - VERIFY(test_redux(Matrix44r().template block<1,2*PacketSize>(2,1), - LinearVectorizedTraversal,CompleteUnrolling)); + VERIFY( + test_redux(Matrix44r().template block<1, 2 * PacketSize>(2, 1), LinearVectorizedTraversal, CompleteUnrolling)); - VERIFY((test_assign< - Map >, - Matrix22 - >(InnerVectorizedTraversal,CompleteUnrolling))); + VERIFY((test_assign>, Matrix22>( + InnerVectorizedTraversal, CompleteUnrolling))); - VERIFY((test_assign< - Map, AlignedMax, InnerStride<3*PacketSize> >, - Matrix - >(DefaultTraversal,PacketSize>=8?InnerUnrolling:CompleteUnrolling))); + VERIFY((test_assign, + AlignedMax, + InnerStride<3 * PacketSize>>, + Matrix>( + DefaultTraversal, PacketSize >= 8 ? InnerUnrolling : CompleteUnrolling))); - VERIFY((test_assign(Matrix11(), Matrix()*Matrix(), - InnerVectorizedTraversal, CompleteUnrolling))); - #endif + VERIFY((test_assign(Matrix11(), + Matrix() + * Matrix(), + InnerVectorizedTraversal, + CompleteUnrolling))); +#endif - VERIFY(test_assign(MatrixXX(10,10),MatrixXX(20,20).block(10,10,2,3), - SliceVectorizedTraversal,NoUnrolling)); + VERIFY(test_assign(MatrixXX(10, 10), MatrixXX(20, 20).block(10, 10, 2, 3), SliceVectorizedTraversal, NoUnrolling)); - VERIFY(test_redux(VectorX(10), - LinearVectorizedTraversal,NoUnrolling)); + VERIFY(test_redux(VectorX(10), LinearVectorizedTraversal, NoUnrolling)); } }; -template struct vectorization_logic +template struct vectorization_logic { static void run() {} }; -template::type>::half, - typename internal::packet_traits::type>::value > +template::type>::half, + typename internal::packet_traits::type>::value> struct vectorization_logic_half { typedef internal::packet_traits PacketTraits; typedef typename internal::unpacket_traits::type>::half PacketType; - enum { - PacketSize = internal::unpacket_traits::size - }; + enum { PacketSize = internal::unpacket_traits::size }; static void run() { - - typedef Matrix Vector1; - typedef Matrix Matrix11; - typedef Matrix Matrix57; - typedef Matrix Matrix35; - typedef Matrix Matrix57u; -// typedef Matrix Matrix44; -// typedef Matrix Matrix44u; -// typedef Matrix Matrix44c; -// typedef Matrix Matrix44r; + + typedef Matrix Vector1; + typedef Matrix Matrix11; + typedef Matrix Matrix57; + typedef Matrix Matrix35; + typedef Matrix Matrix57u; + // typedef + // Matrix + // Matrix44; typedef + // Matrix + // Matrix44u; typedef Matrix Matrix44c; typedef + // Matrix Matrix44r; typedef Matrix Matrix1; + (PacketSize == 8 ? 4 + : PacketSize == 4 ? 2 + : PacketSize == 2 ? 1 + : /*PacketSize==1 ?*/ 1), + (PacketSize == 8 ? 2 + : PacketSize == 4 ? 2 + : PacketSize == 2 ? 2 + : /*PacketSize==1 ?*/ 1)> + Matrix1; typedef Matrix Matrix1u; + (PacketSize == 8 ? 4 + : PacketSize == 4 ? 2 + : PacketSize == 2 ? 1 + : /*PacketSize==1 ?*/ 1), + (PacketSize == 8 ? 2 + : PacketSize == 4 ? 2 + : PacketSize == 2 ? 2 + : /*PacketSize==1 ?*/ 1), + DontAlign | ((Matrix1::Flags & RowMajorBit) ? RowMajor : ColMajor)> + Matrix1u; // this type is made such that it can only be vectorized when viewed as a linear 1D vector typedef Matrix Matrix3; - - #if !EIGEN_GCC_AND_ARCH_DOESNT_WANT_STACK_ALIGNMENT - VERIFY(test_assign(Vector1(),Vector1(), - InnerVectorizedTraversal,CompleteUnrolling)); - VERIFY(test_assign(Vector1(),Vector1()+Vector1(), - InnerVectorizedTraversal,CompleteUnrolling)); - VERIFY(test_assign(Vector1(),Vector1().template segment(0).derived(), - EIGEN_UNALIGNED_VECTORIZE ? InnerVectorizedTraversal : LinearVectorizedTraversal,CompleteUnrolling)); - VERIFY(test_assign(Vector1(),Scalar(2.1)*Vector1()-Vector1(), - InnerVectorizedTraversal,CompleteUnrolling)); - VERIFY(test_assign(Vector1(),(Scalar(2.1)*Vector1().template segment(0)-Vector1().template segment(0)).derived(), - EIGEN_UNALIGNED_VECTORIZE ? InnerVectorizedTraversal : LinearVectorizedTraversal,CompleteUnrolling)); - VERIFY(test_assign(Vector1(),Vector1().cwiseProduct(Vector1()), - InnerVectorizedTraversal,CompleteUnrolling)); - VERIFY(test_assign(Vector1(),Vector1().template cast(), - InnerVectorizedTraversal,CompleteUnrolling)); - - - VERIFY(test_assign(Vector1(),Vector1(), - InnerVectorizedTraversal,CompleteUnrolling)); - VERIFY(test_assign(Vector1(),Vector1()+Vector1(), - InnerVectorizedTraversal,CompleteUnrolling)); - VERIFY(test_assign(Vector1(),Vector1().cwiseProduct(Vector1()), - InnerVectorizedTraversal,CompleteUnrolling)); - - VERIFY(test_assign(Matrix57(),Matrix57()+Matrix57(), - InnerVectorizedTraversal,InnerUnrolling)); - - VERIFY(test_assign(Matrix57u(),Matrix57()+Matrix57(), + (PacketSize == 8 ? 4 + : PacketSize == 4 ? 6 + : PacketSize == 2 ? ((Matrix11::Flags & RowMajorBit) ? 2 : 3) + : /*PacketSize==1 ?*/ 1), + (PacketSize == 8 ? 6 + : PacketSize == 4 ? 2 + : PacketSize == 2 ? ((Matrix11::Flags & RowMajorBit) ? 3 : 2) + : /*PacketSize==1 ?*/ 3)> + Matrix3; + +#if !EIGEN_GCC_AND_ARCH_DOESNT_WANT_STACK_ALIGNMENT + VERIFY(test_assign(Vector1(), Vector1(), InnerVectorizedTraversal, CompleteUnrolling)); + VERIFY(test_assign(Vector1(), Vector1() + Vector1(), InnerVectorizedTraversal, CompleteUnrolling)); + VERIFY(test_assign(Vector1(), + Vector1().template segment(0).derived(), + EIGEN_UNALIGNED_VECTORIZE ? InnerVectorizedTraversal : LinearVectorizedTraversal, + CompleteUnrolling)); + VERIFY(test_assign(Vector1(), Scalar(2.1) * Vector1() - Vector1(), InnerVectorizedTraversal, CompleteUnrolling)); + VERIFY(test_assign(Vector1(), + (Scalar(2.1) * Vector1().template segment(0) - Vector1().template segment(0)).derived(), + EIGEN_UNALIGNED_VECTORIZE ? InnerVectorizedTraversal : LinearVectorizedTraversal, + CompleteUnrolling)); + VERIFY(test_assign(Vector1(), Vector1().cwiseProduct(Vector1()), InnerVectorizedTraversal, CompleteUnrolling)); + VERIFY(test_assign(Vector1(), Vector1().template cast(), InnerVectorizedTraversal, CompleteUnrolling)); + + + VERIFY(test_assign(Vector1(), Vector1(), InnerVectorizedTraversal, CompleteUnrolling)); + VERIFY(test_assign(Vector1(), Vector1() + Vector1(), InnerVectorizedTraversal, CompleteUnrolling)); + VERIFY(test_assign(Vector1(), Vector1().cwiseProduct(Vector1()), InnerVectorizedTraversal, CompleteUnrolling)); + + VERIFY(test_assign(Matrix57(), Matrix57() + Matrix57(), InnerVectorizedTraversal, InnerUnrolling)); + + VERIFY(test_assign(Matrix57u(), + Matrix57() + Matrix57(), EIGEN_UNALIGNED_VECTORIZE ? InnerVectorizedTraversal : LinearTraversal, EIGEN_UNALIGNED_VECTORIZE ? InnerUnrolling : NoUnrolling)); - VERIFY(test_assign(Matrix1u(),Matrix1()+Matrix1(), - EIGEN_UNALIGNED_VECTORIZE ? ((Matrix1::InnerSizeAtCompileTime % PacketSize)==0 ? InnerVectorizedTraversal : LinearVectorizedTraversal) : LinearTraversal,CompleteUnrolling)); - - if(PacketSize>1) - { - typedef Matrix Matrix33c; - VERIFY(test_assign(Matrix33c().row(2),Matrix33c().row(1)+Matrix33c().row(1), - LinearTraversal,CompleteUnrolling)); - VERIFY(test_assign(Matrix33c().col(0),Matrix33c().col(1)+Matrix33c().col(1), - EIGEN_UNALIGNED_VECTORIZE ? (PacketSize==1 ? InnerVectorizedTraversal : LinearVectorizedTraversal) : LinearTraversal,CompleteUnrolling)); - - VERIFY(test_assign(Matrix3(),Matrix3().cwiseQuotient(Matrix3()), - PacketTraits::HasDiv ? LinearVectorizedTraversal : LinearTraversal,CompleteUnrolling)); - - VERIFY(test_assign(Matrix(),Matrix()+Matrix(), - EIGEN_UNALIGNED_VECTORIZE ? (PacketSize==1 ? InnerVectorizedTraversal : LinearVectorizedTraversal) : LinearTraversal, + VERIFY(test_assign(Matrix1u(), + Matrix1() + Matrix1(), + EIGEN_UNALIGNED_VECTORIZE + ? ((Matrix1::InnerSizeAtCompileTime % PacketSize) == 0 ? InnerVectorizedTraversal : LinearVectorizedTraversal) + : LinearTraversal, + CompleteUnrolling)); + + if (PacketSize > 1) { + typedef Matrix Matrix33c; + VERIFY( + test_assign(Matrix33c().row(2), Matrix33c().row(1) + Matrix33c().row(1), LinearTraversal, CompleteUnrolling)); + VERIFY(test_assign(Matrix33c().col(0), + Matrix33c().col(1) + Matrix33c().col(1), + EIGEN_UNALIGNED_VECTORIZE ? (PacketSize == 1 ? InnerVectorizedTraversal : LinearVectorizedTraversal) + : LinearTraversal, + CompleteUnrolling)); + + VERIFY(test_assign(Matrix3(), + Matrix3().cwiseQuotient(Matrix3()), + PacketTraits::HasDiv ? LinearVectorizedTraversal : LinearTraversal, + CompleteUnrolling)); + + VERIFY(test_assign(Matrix(), + Matrix() + Matrix(), + EIGEN_UNALIGNED_VECTORIZE ? (PacketSize == 1 ? InnerVectorizedTraversal : LinearVectorizedTraversal) + : LinearTraversal, NoUnrolling)); - - VERIFY(test_assign(Matrix11(),Matrix().template block(2,3)+Matrix().template block(8,4), - EIGEN_UNALIGNED_VECTORIZE ? InnerVectorizedTraversal : DefaultTraversal,PacketSize>4?InnerUnrolling:CompleteUnrolling)); - VERIFY(test_assign(Vector1(),Matrix11()*Vector1(), - InnerVectorizedTraversal,CompleteUnrolling)); + VERIFY(test_assign(Matrix11(), + Matrix().template block(2, 3) + + Matrix().template block(8, 4), + EIGEN_UNALIGNED_VECTORIZE ? InnerVectorizedTraversal : DefaultTraversal, + PacketSize > 4 ? InnerUnrolling : CompleteUnrolling)); - VERIFY(test_assign(Matrix11(),Matrix11().lazyProduct(Matrix11()), - InnerVectorizedTraversal,InnerUnrolling+CompleteUnrolling)); + VERIFY(test_assign(Vector1(), Matrix11() * Vector1(), InnerVectorizedTraversal, CompleteUnrolling)); + + VERIFY(test_assign( + Matrix11(), Matrix11().lazyProduct(Matrix11()), InnerVectorizedTraversal, InnerUnrolling + CompleteUnrolling)); } - - VERIFY(test_redux(Vector1(), - LinearVectorizedTraversal,CompleteUnrolling)); - VERIFY(test_redux(Matrix(), - LinearVectorizedTraversal,CompleteUnrolling)); + VERIFY(test_redux(Vector1(), LinearVectorizedTraversal, CompleteUnrolling)); + + VERIFY(test_redux(Matrix(), LinearVectorizedTraversal, CompleteUnrolling)); - VERIFY(test_redux(Matrix3(), - LinearVectorizedTraversal,CompleteUnrolling)); + VERIFY(test_redux(Matrix3(), LinearVectorizedTraversal, CompleteUnrolling)); - VERIFY(test_redux(Matrix35(), - LinearVectorizedTraversal,CompleteUnrolling)); + VERIFY(test_redux(Matrix35(), LinearVectorizedTraversal, CompleteUnrolling)); - VERIFY(test_redux(Matrix57().template block(1,0), - DefaultTraversal,CompleteUnrolling)); + VERIFY(test_redux(Matrix57().template block(1, 0), DefaultTraversal, CompleteUnrolling)); - VERIFY((test_assign< - Map, AlignedMax, InnerStride<3*PacketSize> >, - Matrix - >(DefaultTraversal,CompleteUnrolling))); + VERIFY((test_assign, + AlignedMax, + InnerStride<3 * PacketSize>>, + Matrix>( + DefaultTraversal, CompleteUnrolling))); - VERIFY((test_assign(Matrix57(), Matrix()*Matrix(), - InnerVectorizedTraversal, InnerUnrolling|CompleteUnrolling))); - #endif + VERIFY((test_assign(Matrix57(), + Matrix() * Matrix(), + InnerVectorizedTraversal, + InnerUnrolling | CompleteUnrolling))); +#endif } }; -template struct vectorization_logic_half +template struct vectorization_logic_half { static void run() {} }; @@ -391,35 +443,38 @@ void test_vectorization_logic() #ifdef EIGEN_VECTORIZE - CALL_SUBTEST( vectorization_logic::run() ); - CALL_SUBTEST( vectorization_logic::run() ); - CALL_SUBTEST( vectorization_logic::run() ); - CALL_SUBTEST( vectorization_logic >::run() ); - CALL_SUBTEST( vectorization_logic >::run() ); - - CALL_SUBTEST( vectorization_logic_half::run() ); - CALL_SUBTEST( vectorization_logic_half::run() ); - CALL_SUBTEST( vectorization_logic_half::run() ); - CALL_SUBTEST( vectorization_logic_half >::run() ); - CALL_SUBTEST( vectorization_logic_half >::run() ); - - if(internal::packet_traits::Vectorizable) - { - VERIFY(test_assign(Matrix(),Matrix()+Matrix(), - EIGEN_UNALIGNED_VECTORIZE ? LinearVectorizedTraversal : LinearTraversal,CompleteUnrolling)); - - VERIFY(test_redux(Matrix(), - EIGEN_UNALIGNED_VECTORIZE ? LinearVectorizedTraversal : DefaultTraversal,CompleteUnrolling)); - } - - if(internal::packet_traits::Vectorizable) - { - VERIFY(test_assign(Matrix(),Matrix()+Matrix(), - EIGEN_UNALIGNED_VECTORIZE ? LinearVectorizedTraversal : LinearTraversal,CompleteUnrolling)); - - VERIFY(test_redux(Matrix(), - EIGEN_UNALIGNED_VECTORIZE ? LinearVectorizedTraversal : DefaultTraversal,CompleteUnrolling)); + CALL_SUBTEST(vectorization_logic::run()); + CALL_SUBTEST(vectorization_logic::run()); + CALL_SUBTEST(vectorization_logic::run()); + CALL_SUBTEST(vectorization_logic>::run()); + CALL_SUBTEST(vectorization_logic>::run()); + + CALL_SUBTEST(vectorization_logic_half::run()); + CALL_SUBTEST(vectorization_logic_half::run()); + CALL_SUBTEST(vectorization_logic_half::run()); + CALL_SUBTEST(vectorization_logic_half>::run()); + CALL_SUBTEST(vectorization_logic_half>::run()); + + if (internal::packet_traits::Vectorizable) { + VERIFY(test_assign(Matrix(), + Matrix() + Matrix(), + EIGEN_UNALIGNED_VECTORIZE ? LinearVectorizedTraversal : LinearTraversal, + CompleteUnrolling)); + + VERIFY(test_redux(Matrix(), + EIGEN_UNALIGNED_VECTORIZE ? LinearVectorizedTraversal : DefaultTraversal, + CompleteUnrolling)); } -#endif // EIGEN_VECTORIZE + if (internal::packet_traits::Vectorizable) { + VERIFY(test_assign(Matrix(), + Matrix() + Matrix(), + EIGEN_UNALIGNED_VECTORIZE ? LinearVectorizedTraversal : LinearTraversal, + CompleteUnrolling)); + + VERIFY(test_redux(Matrix(), + EIGEN_UNALIGNED_VECTORIZE ? LinearVectorizedTraversal : DefaultTraversal, + CompleteUnrolling)); + } +#endif// EIGEN_VECTORIZE } diff --git a/filmulator-gui/core/nlmeans/eigen/test/vectorwiseop.cpp b/filmulator-gui/core/nlmeans/eigen/test/vectorwiseop.cpp index a099d17c..e7ea1695 100644 --- a/filmulator-gui/core/nlmeans/eigen/test/vectorwiseop.cpp +++ b/filmulator-gui/core/nlmeans/eigen/test/vectorwiseop.cpp @@ -13,7 +13,7 @@ #include "main.h" -template void vectorwiseop_array(const ArrayType& m) +template void vectorwiseop_array(const ArrayType &m) { typedef typename ArrayType::Scalar Scalar; typedef Array ColVectorType; @@ -21,12 +21,9 @@ template void vectorwiseop_array(const ArrayType& m) Index rows = m.rows(); Index cols = m.cols(); - Index r = internal::random(0, rows-1), - c = internal::random(0, cols-1); + Index r = internal::random(0, rows - 1), c = internal::random(0, cols - 1); - ArrayType m1 = ArrayType::Random(rows, cols), - m2(rows, cols), - m3(rows, cols); + ArrayType m1 = ArrayType::Random(rows, cols), m2(rows, cols), m3(rows, cols); ColVectorType colvec = ColVectorType::Random(rows); RowVectorType rowvec = RowVectorType::Random(cols); @@ -107,26 +104,25 @@ template void vectorwiseop_array(const ArrayType& m) // yes, there might be an aliasing issue there but ".rowwise() /=" // is supposed to evaluate " m2.colwise().sum()" into a temporary to avoid // evaluating the reduction multiple times - if(ArrayType::RowsAtCompileTime>2 || ArrayType::RowsAtCompileTime==Dynamic) - { + if (ArrayType::RowsAtCompileTime > 2 || ArrayType::RowsAtCompileTime == Dynamic) { m2.rowwise() /= m2.colwise().sum(); VERIFY_IS_APPROX(m2, m1.rowwise() / m1.colwise().sum()); } // all/any - Array mb(rows,cols); - mb = (m1.real()<=0.7).colwise().all(); - VERIFY( (mb.col(c) == (m1.real().col(c)<=0.7).all()).all() ); - mb = (m1.real()<=0.7).rowwise().all(); - VERIFY( (mb.row(r) == (m1.real().row(r)<=0.7).all()).all() ); - - mb = (m1.real()>=0.7).colwise().any(); - VERIFY( (mb.col(c) == (m1.real().col(c)>=0.7).any()).all() ); - mb = (m1.real()>=0.7).rowwise().any(); - VERIFY( (mb.row(r) == (m1.real().row(r)>=0.7).any()).all() ); + Array mb(rows, cols); + mb = (m1.real() <= 0.7).colwise().all(); + VERIFY((mb.col(c) == (m1.real().col(c) <= 0.7).all()).all()); + mb = (m1.real() <= 0.7).rowwise().all(); + VERIFY((mb.row(r) == (m1.real().row(r) <= 0.7).all()).all()); + + mb = (m1.real() >= 0.7).colwise().any(); + VERIFY((mb.col(c) == (m1.real().col(c) >= 0.7).any()).all()); + mb = (m1.real() >= 0.7).rowwise().any(); + VERIFY((mb.row(r) == (m1.real().row(r) >= 0.7).any()).all()); } -template void vectorwiseop_matrix(const MatrixType& m) +template void vectorwiseop_matrix(const MatrixType &m) { typedef typename MatrixType::Scalar Scalar; typedef typename NumTraits::Real RealScalar; @@ -137,12 +133,9 @@ template void vectorwiseop_matrix(const MatrixType& m) Index rows = m.rows(); Index cols = m.cols(); - Index r = internal::random(0, rows-1), - c = internal::random(0, cols-1); + Index r = internal::random(0, rows - 1), c = internal::random(0, cols - 1); - MatrixType m1 = MatrixType::Random(rows, cols), - m2(rows, cols), - m3(rows, cols); + MatrixType m1 = MatrixType::Random(rows, cols), m2(rows, cols), m3(rows, cols); ColVectorType colvec = ColVectorType::Random(rows); RowVectorType rowvec = RowVectorType::Random(cols); @@ -156,8 +149,7 @@ template void vectorwiseop_matrix(const MatrixType& m) VERIFY_IS_APPROX(m2, m1.colwise() + colvec); VERIFY_IS_APPROX(m2.col(c), m1.col(c) + colvec); - if(rows>1) - { + if (rows > 1) { VERIFY_RAISES_ASSERT(m2.colwise() += colvec.transpose()); VERIFY_RAISES_ASSERT(m1.colwise() + colvec.transpose()); } @@ -167,8 +159,7 @@ template void vectorwiseop_matrix(const MatrixType& m) VERIFY_IS_APPROX(m2, m1.rowwise() + rowvec); VERIFY_IS_APPROX(m2.row(r), m1.row(r) + rowvec); - if(cols>1) - { + if (cols > 1) { VERIFY_RAISES_ASSERT(m2.rowwise() += rowvec.transpose()); VERIFY_RAISES_ASSERT(m1.rowwise() + rowvec.transpose()); } @@ -180,8 +171,7 @@ template void vectorwiseop_matrix(const MatrixType& m) VERIFY_IS_APPROX(m2, m1.colwise() - colvec); VERIFY_IS_APPROX(m2.col(c), m1.col(c) - colvec); - if(rows>1) - { + if (rows > 1) { VERIFY_RAISES_ASSERT(m2.colwise() -= colvec.transpose()); VERIFY_RAISES_ASSERT(m1.colwise() - colvec.transpose()); } @@ -191,8 +181,7 @@ template void vectorwiseop_matrix(const MatrixType& m) VERIFY_IS_APPROX(m2, m1.rowwise() - rowvec); VERIFY_IS_APPROX(m2.row(r), m1.row(r) - rowvec); - if(cols>1) - { + if (cols > 1) { VERIFY_RAISES_ASSERT(m2.rowwise() -= rowvec.transpose()); VERIFY_RAISES_ASSERT(m1.rowwise() - rowvec.transpose()); } @@ -226,25 +215,27 @@ template void vectorwiseop_matrix(const MatrixType& m) VERIFY_IS_APPROX(m2.row(r), m1.row(r).normalized()); // test with partial reduction of products - Matrix m1m1 = m1 * m1.transpose(); - VERIFY_IS_APPROX( (m1 * m1.transpose()).colwise().sum(), m1m1.colwise().sum()); - Matrix tmp(rows); - VERIFY_EVALUATION_COUNT( tmp = (m1 * m1.transpose()).colwise().sum(), 1); - - m2 = m1.rowwise() - (m1.colwise().sum()/RealScalar(m1.rows())).eval(); - m1 = m1.rowwise() - (m1.colwise().sum()/RealScalar(m1.rows())); - VERIFY_IS_APPROX( m1, m2 ); - VERIFY_EVALUATION_COUNT( m2 = (m1.rowwise() - m1.colwise().sum()/RealScalar(m1.rows())), (MatrixType::RowsAtCompileTime!=1 ? 1 : 0) ); + Matrix m1m1 = m1 * m1.transpose(); + VERIFY_IS_APPROX((m1 * m1.transpose()).colwise().sum(), m1m1.colwise().sum()); + Matrix tmp(rows); + VERIFY_EVALUATION_COUNT(tmp = (m1 * m1.transpose()).colwise().sum(), 1); + + m2 = m1.rowwise() - (m1.colwise().sum() / RealScalar(m1.rows())).eval(); + m1 = m1.rowwise() - (m1.colwise().sum() / RealScalar(m1.rows())); + VERIFY_IS_APPROX(m1, m2); + VERIFY_EVALUATION_COUNT( + m2 = (m1.rowwise() - m1.colwise().sum() / RealScalar(m1.rows())), (MatrixType::RowsAtCompileTime != 1 ? 1 : 0)); } void test_vectorwiseop() { - CALL_SUBTEST_1( vectorwiseop_array(Array22cd()) ); - CALL_SUBTEST_2( vectorwiseop_array(Array()) ); - CALL_SUBTEST_3( vectorwiseop_array(ArrayXXf(3, 4)) ); - CALL_SUBTEST_4( vectorwiseop_matrix(Matrix4cf()) ); - CALL_SUBTEST_5( vectorwiseop_matrix(Matrix()) ); - CALL_SUBTEST_6( vectorwiseop_matrix(MatrixXd(internal::random(1,EIGEN_TEST_MAX_SIZE), internal::random(1,EIGEN_TEST_MAX_SIZE))) ); - CALL_SUBTEST_7( vectorwiseop_matrix(VectorXd(internal::random(1,EIGEN_TEST_MAX_SIZE))) ); - CALL_SUBTEST_7( vectorwiseop_matrix(RowVectorXd(internal::random(1,EIGEN_TEST_MAX_SIZE))) ); + CALL_SUBTEST_1(vectorwiseop_array(Array22cd())); + CALL_SUBTEST_2(vectorwiseop_array(Array())); + CALL_SUBTEST_3(vectorwiseop_array(ArrayXXf(3, 4))); + CALL_SUBTEST_4(vectorwiseop_matrix(Matrix4cf())); + CALL_SUBTEST_5(vectorwiseop_matrix(Matrix())); + CALL_SUBTEST_6(vectorwiseop_matrix( + MatrixXd(internal::random(1, EIGEN_TEST_MAX_SIZE), internal::random(1, EIGEN_TEST_MAX_SIZE)))); + CALL_SUBTEST_7(vectorwiseop_matrix(VectorXd(internal::random(1, EIGEN_TEST_MAX_SIZE)))); + CALL_SUBTEST_7(vectorwiseop_matrix(RowVectorXd(internal::random(1, EIGEN_TEST_MAX_SIZE)))); } diff --git a/filmulator-gui/core/nlmeans/eigen/test/visitor.cpp b/filmulator-gui/core/nlmeans/eigen/test/visitor.cpp index 7f4efab9..1bf37f21 100644 --- a/filmulator-gui/core/nlmeans/eigen/test/visitor.cpp +++ b/filmulator-gui/core/nlmeans/eigen/test/visitor.cpp @@ -9,7 +9,7 @@ #include "main.h" -template void matrixVisitor(const MatrixType& p) +template void matrixVisitor(const MatrixType &p) { typedef typename MatrixType::Scalar Scalar; @@ -19,33 +19,30 @@ template void matrixVisitor(const MatrixType& p) // construct a random matrix where all coefficients are different MatrixType m; m = MatrixType::Random(rows, cols); - for(Index i = 0; i < m.size(); i++) - for(Index i2 = 0; i2 < i; i2++) - while(m(i) == m(i2)) // yes, == + for (Index i = 0; i < m.size(); i++) + for (Index i2 = 0; i2 < i; i2++) + while (m(i) == m(i2))// yes, == m(i) = internal::random(); - + Scalar minc = Scalar(1000), maxc = Scalar(-1000); - Index minrow=0,mincol=0,maxrow=0,maxcol=0; - for(Index j = 0; j < cols; j++) - for(Index i = 0; i < rows; i++) - { - if(m(i,j) < minc) - { - minc = m(i,j); - minrow = i; - mincol = j; - } - if(m(i,j) > maxc) - { - maxc = m(i,j); - maxrow = i; - maxcol = j; + Index minrow = 0, mincol = 0, maxrow = 0, maxcol = 0; + for (Index j = 0; j < cols; j++) + for (Index i = 0; i < rows; i++) { + if (m(i, j) < minc) { + minc = m(i, j); + minrow = i; + mincol = j; + } + if (m(i, j) > maxc) { + maxc = m(i, j); + maxrow = i; + maxcol = j; + } } - } Index eigen_minrow, eigen_mincol, eigen_maxrow, eigen_maxcol; Scalar eigen_minc, eigen_maxc; - eigen_minc = m.minCoeff(&eigen_minrow,&eigen_mincol); - eigen_maxc = m.maxCoeff(&eigen_maxrow,&eigen_maxcol); + eigen_minc = m.minCoeff(&eigen_minrow, &eigen_mincol); + eigen_maxc = m.maxCoeff(&eigen_maxrow, &eigen_maxcol); VERIFY(minrow == eigen_minrow); VERIFY(maxrow == eigen_maxrow); VERIFY(mincol == eigen_mincol); @@ -55,13 +52,13 @@ template void matrixVisitor(const MatrixType& p) VERIFY_IS_APPROX(minc, m.minCoeff()); VERIFY_IS_APPROX(maxc, m.maxCoeff()); - eigen_maxc = (m.adjoint()*m).maxCoeff(&eigen_maxrow,&eigen_maxcol); - eigen_maxc = (m.adjoint()*m).eval().maxCoeff(&maxrow,&maxcol); + eigen_maxc = (m.adjoint() * m).maxCoeff(&eigen_maxrow, &eigen_maxcol); + eigen_maxc = (m.adjoint() * m).eval().maxCoeff(&maxrow, &maxcol); VERIFY(maxrow == eigen_maxrow); VERIFY(maxcol == eigen_maxcol); } -template void vectorVisitor(const VectorType& w) +template void vectorVisitor(const VectorType &w) { typedef typename VectorType::Scalar Scalar; @@ -70,22 +67,19 @@ template void vectorVisitor(const VectorType& w) // construct a random vector where all coefficients are different VectorType v; v = VectorType::Random(size); - for(Index i = 0; i < size; i++) - for(Index i2 = 0; i2 < i; i2++) - while(v(i) == v(i2)) // yes, == + for (Index i = 0; i < size; i++) + for (Index i2 = 0; i2 < i; i2++) + while (v(i) == v(i2))// yes, == v(i) = internal::random(); - + Scalar minc = v(0), maxc = v(0); - Index minidx=0, maxidx=0; - for(Index i = 0; i < size; i++) - { - if(v(i) < minc) - { + Index minidx = 0, maxidx = 0; + for (Index i = 0; i < size; i++) { + if (v(i) < minc) { minc = v(i); minidx = i; } - if(v(i) > maxc) - { + if (v(i) > maxc) { maxc = v(i); maxidx = i; } @@ -100,8 +94,8 @@ template void vectorVisitor(const VectorType& w) VERIFY_IS_APPROX(maxc, eigen_maxc); VERIFY_IS_APPROX(minc, v.minCoeff()); VERIFY_IS_APPROX(maxc, v.maxCoeff()); - - Index idx0 = internal::random(0,size-1); + + Index idx0 = internal::random(0, size - 1); Index idx1 = eigen_minidx; Index idx2 = eigen_maxidx; VectorType v1(v), v2(v); @@ -109,25 +103,25 @@ template void vectorVisitor(const VectorType& w) v2(idx0) = v2(idx2); v1.minCoeff(&eigen_minidx); v2.maxCoeff(&eigen_maxidx); - VERIFY(eigen_minidx == (std::min)(idx0,idx1)); - VERIFY(eigen_maxidx == (std::min)(idx0,idx2)); + VERIFY(eigen_minidx == (std::min)(idx0, idx1)); + VERIFY(eigen_maxidx == (std::min)(idx0, idx2)); } void test_visitor() { - for(int i = 0; i < g_repeat; i++) { - CALL_SUBTEST_1( matrixVisitor(Matrix()) ); - CALL_SUBTEST_2( matrixVisitor(Matrix2f()) ); - CALL_SUBTEST_3( matrixVisitor(Matrix4d()) ); - CALL_SUBTEST_4( matrixVisitor(MatrixXd(8, 12)) ); - CALL_SUBTEST_5( matrixVisitor(Matrix(20, 20)) ); - CALL_SUBTEST_6( matrixVisitor(MatrixXi(8, 12)) ); + for (int i = 0; i < g_repeat; i++) { + CALL_SUBTEST_1(matrixVisitor(Matrix())); + CALL_SUBTEST_2(matrixVisitor(Matrix2f())); + CALL_SUBTEST_3(matrixVisitor(Matrix4d())); + CALL_SUBTEST_4(matrixVisitor(MatrixXd(8, 12))); + CALL_SUBTEST_5(matrixVisitor(Matrix(20, 20))); + CALL_SUBTEST_6(matrixVisitor(MatrixXi(8, 12))); } - for(int i = 0; i < g_repeat; i++) { - CALL_SUBTEST_7( vectorVisitor(Vector4f()) ); - CALL_SUBTEST_7( vectorVisitor(Matrix()) ); - CALL_SUBTEST_8( vectorVisitor(VectorXd(10)) ); - CALL_SUBTEST_9( vectorVisitor(RowVectorXd(10)) ); - CALL_SUBTEST_10( vectorVisitor(VectorXf(33)) ); + for (int i = 0; i < g_repeat; i++) { + CALL_SUBTEST_7(vectorVisitor(Vector4f())); + CALL_SUBTEST_7(vectorVisitor(Matrix())); + CALL_SUBTEST_8(vectorVisitor(VectorXd(10))); + CALL_SUBTEST_9(vectorVisitor(RowVectorXd(10))); + CALL_SUBTEST_10(vectorVisitor(VectorXf(33))); } } diff --git a/filmulator-gui/core/nlmeans/eigen/test/zerosized.cpp b/filmulator-gui/core/nlmeans/eigen/test/zerosized.cpp index 477ff007..803875bb 100644 --- a/filmulator-gui/core/nlmeans/eigen/test/zerosized.cpp +++ b/filmulator-gui/core/nlmeans/eigen/test/zerosized.cpp @@ -10,13 +10,14 @@ #include "main.h" -template void zeroReduction(const MatrixType& m) { +template void zeroReduction(const MatrixType &m) +{ // Reductions that must hold for zero sized objects VERIFY(m.all()); VERIFY(!m.any()); - VERIFY(m.prod()==1); - VERIFY(m.sum()==0); - VERIFY(m.count()==0); + VERIFY(m.prod() == 1); + VERIFY(m.sum() == 0); + VERIFY(m.count() == 0); VERIFY(m.allFinite()); VERIFY(!m.hasNaN()); } @@ -27,40 +28,38 @@ template void zeroSizedMatrix() MatrixType t1; typedef typename MatrixType::Scalar Scalar; - if (MatrixType::SizeAtCompileTime == Dynamic || MatrixType::SizeAtCompileTime == 0) - { + if (MatrixType::SizeAtCompileTime == Dynamic || MatrixType::SizeAtCompileTime == 0) { zeroReduction(t1); - if (MatrixType::RowsAtCompileTime == Dynamic) - VERIFY(t1.rows() == 0); - if (MatrixType::ColsAtCompileTime == Dynamic) - VERIFY(t1.cols() == 0); + if (MatrixType::RowsAtCompileTime == Dynamic) VERIFY(t1.rows() == 0); + if (MatrixType::ColsAtCompileTime == Dynamic) VERIFY(t1.cols() == 0); - if (MatrixType::RowsAtCompileTime == Dynamic && MatrixType::ColsAtCompileTime == Dynamic) - { + if (MatrixType::RowsAtCompileTime == Dynamic && MatrixType::ColsAtCompileTime == Dynamic) { MatrixType t2(0, 0), t3(t1); VERIFY(t2.rows() == 0); VERIFY(t2.cols() == 0); zeroReduction(t2); - VERIFY(t1==t2); + VERIFY(t1 == t2); } } - if(MatrixType::MaxColsAtCompileTime!=0 && MatrixType::MaxRowsAtCompileTime!=0) - { - Index rows = MatrixType::RowsAtCompileTime==Dynamic ? internal::random(1,10) : Index(MatrixType::RowsAtCompileTime); - Index cols = MatrixType::ColsAtCompileTime==Dynamic ? internal::random(1,10) : Index(MatrixType::ColsAtCompileTime); - MatrixType m(rows,cols); - zeroReduction(m.template block<0,MatrixType::ColsAtCompileTime>(0,0,0,cols)); - zeroReduction(m.template block(0,0,rows,0)); - zeroReduction(m.template block<0,1>(0,0)); - zeroReduction(m.template block<1,0>(0,0)); - Matrix prod = m.template block(0,0,rows,0) * m.template block<0,MatrixType::ColsAtCompileTime>(0,0,0,cols); - VERIFY(prod.rows()==rows && prod.cols()==cols); + if (MatrixType::MaxColsAtCompileTime != 0 && MatrixType::MaxRowsAtCompileTime != 0) { + Index rows = + MatrixType::RowsAtCompileTime == Dynamic ? internal::random(1, 10) : Index(MatrixType::RowsAtCompileTime); + Index cols = + MatrixType::ColsAtCompileTime == Dynamic ? internal::random(1, 10) : Index(MatrixType::ColsAtCompileTime); + MatrixType m(rows, cols); + zeroReduction(m.template block<0, MatrixType::ColsAtCompileTime>(0, 0, 0, cols)); + zeroReduction(m.template block(0, 0, rows, 0)); + zeroReduction(m.template block<0, 1>(0, 0)); + zeroReduction(m.template block<1, 0>(0, 0)); + Matrix prod = m.template block(0, 0, rows, 0) + * m.template block<0, MatrixType::ColsAtCompileTime>(0, 0, 0, cols); + VERIFY(prod.rows() == rows && prod.cols() == cols); VERIFY(prod.isZero()); - prod = m.template block<1,0>(0,0) * m.template block<0,1>(0,0); - VERIFY(prod.size()==1); + prod = m.template block<1, 0>(0, 0) * m.template block<0, 1>(0, 0); + VERIFY(prod.size() == 1); VERIFY(prod.isZero()); } } @@ -69,15 +68,14 @@ template void zeroSizedVector() { VectorType t1; - if (VectorType::SizeAtCompileTime == Dynamic || VectorType::SizeAtCompileTime==0) - { + if (VectorType::SizeAtCompileTime == Dynamic || VectorType::SizeAtCompileTime == 0) { zeroReduction(t1); VERIFY(t1.size() == 0); - VectorType t2(DenseIndex(0)); // DenseIndex disambiguates with 0-the-null-pointer (error with gcc 4.4 and MSVC8) + VectorType t2(DenseIndex(0));// DenseIndex disambiguates with 0-the-null-pointer (error with gcc 4.4 and MSVC8) VERIFY(t2.size() == 0); zeroReduction(t2); - VERIFY(t1==t2); + VERIFY(t1 == t2); } } @@ -85,18 +83,18 @@ void test_zerosized() { zeroSizedMatrix(); zeroSizedMatrix(); - zeroSizedMatrix >(); + zeroSizedMatrix>(); zeroSizedMatrix(); - zeroSizedMatrix >(); - zeroSizedMatrix >(); - zeroSizedMatrix >(); - zeroSizedMatrix >(); - zeroSizedMatrix >(); - zeroSizedMatrix >(); + zeroSizedMatrix>(); + zeroSizedMatrix>(); + zeroSizedMatrix>(); + zeroSizedMatrix>(); + zeroSizedMatrix>(); + zeroSizedMatrix>(); zeroSizedVector(); zeroSizedVector(); zeroSizedVector(); - zeroSizedVector >(); - zeroSizedVector >(); + zeroSizedVector>(); + zeroSizedVector>(); } diff --git a/filmulator-gui/core/nlmeans/eigen/unsupported/Eigen/CXX11/src/Tensor/Tensor.h b/filmulator-gui/core/nlmeans/eigen/unsupported/Eigen/CXX11/src/Tensor/Tensor.h index 00295a25..5308b834 100644 --- a/filmulator-gui/core/nlmeans/eigen/unsupported/Eigen/CXX11/src/Tensor/Tensor.h +++ b/filmulator-gui/core/nlmeans/eigen/unsupported/Eigen/CXX11/src/Tensor/Tensor.h @@ -14,330 +14,308 @@ namespace Eigen { /** \class Tensor - * \ingroup CXX11_Tensor_Module - * - * \brief The tensor class. - * - * The %Tensor class is the work-horse for all \em dense tensors within Eigen. - * - * The %Tensor class encompasses only dynamic-size objects so far. - * - * The first two template parameters are required: - * \tparam Scalar_ Numeric type, e.g. float, double, int or `std::complex`. - * User defined scalar types are supported as well (see \ref user_defined_scalars "here"). - * \tparam NumIndices_ Number of indices (i.e. rank of the tensor) - * - * The remaining template parameters are optional -- in most cases you don't have to worry about them. - * \tparam Options_ A combination of either \b #RowMajor or \b #ColMajor, and of either - * \b #AutoAlign or \b #DontAlign. - * The former controls \ref TopicStorageOrders "storage order", and defaults to column-major. The latter controls alignment, which is required - * for vectorization. It defaults to aligning tensors. Note that tensors currently do not support any operations that profit from vectorization. - * Support for such operations (i.e. adding two tensors etc.) is planned. - * - * You can access elements of tensors using normal subscripting: - * - * \code - * Eigen::Tensor t(10, 10, 10, 10); - * t(0, 1, 2, 3) = 42.0; - * \endcode - * - * This class can be extended with the help of the plugin mechanism described on the page - * \ref TopicCustomizing_Plugins by defining the preprocessor symbol \c EIGEN_TENSOR_PLUGIN. - * - * Some notes: - * - *
- *
Relation to other parts of Eigen:
- *
The midterm development goal for this class is to have a similar hierarchy as Eigen uses for matrices, so that - * taking blocks or using tensors in expressions is easily possible, including an interface with the vector/matrix code - * by providing .asMatrix() and .asVector() (or similar) methods for rank 2 and 1 tensors. However, currently, the %Tensor - * class does not provide any of these features and is only available as a stand-alone class that just allows for - * coefficient access. Also, when fixed-size tensors are implemented, the number of template arguments is likely to - * change dramatically.
- *
- * - * \ref TopicStorageOrders - */ + * \ingroup CXX11_Tensor_Module + * + * \brief The tensor class. + * + * The %Tensor class is the work-horse for all \em dense tensors within Eigen. + * + * The %Tensor class encompasses only dynamic-size objects so far. + * + * The first two template parameters are required: + * \tparam Scalar_ Numeric type, e.g. float, double, int or `std::complex`. + * User defined scalar types are supported as well (see \ref user_defined_scalars "here"). + * \tparam NumIndices_ Number of indices (i.e. rank of the tensor) + * + * The remaining template parameters are optional -- in most cases you don't have to worry about them. + * \tparam Options_ A combination of either \b #RowMajor or \b #ColMajor, and of either + * \b #AutoAlign or \b #DontAlign. + * The former controls \ref TopicStorageOrders "storage order", and defaults to column-major. The latter + * controls alignment, which is required for vectorization. It defaults to aligning tensors. Note that tensors currently + * do not support any operations that profit from vectorization. Support for such operations (i.e. adding two tensors + * etc.) is planned. + * + * You can access elements of tensors using normal subscripting: + * + * \code + * Eigen::Tensor t(10, 10, 10, 10); + * t(0, 1, 2, 3) = 42.0; + * \endcode + * + * This class can be extended with the help of the plugin mechanism described on the page + * \ref TopicCustomizing_Plugins by defining the preprocessor symbol \c EIGEN_TENSOR_PLUGIN. + * + * Some notes: + * + *
+ *
Relation to other parts of Eigen:
+ *
The midterm development goal for this class is to have a similar hierarchy as Eigen uses for matrices, so that + * taking blocks or using tensors in expressions is easily possible, including an interface with the vector/matrix code + * by providing .asMatrix() and .asVector() (or similar) methods for rank 2 and 1 tensors. However, currently, the + * %Tensor class does not provide any of these features and is only available as a stand-alone class that just allows + * for coefficient access. Also, when fixed-size tensors are implemented, the number of template arguments is likely to + * change dramatically.
+ *
+ * + * \ref TopicStorageOrders + */ template -class Tensor : public TensorBase > +class Tensor : public TensorBase> { - public: - typedef Tensor Self; - typedef TensorBase > Base; - typedef typename Eigen::internal::nested::type Nested; - typedef typename internal::traits::StorageKind StorageKind; - typedef typename internal::traits::Index Index; - typedef Scalar_ Scalar; - typedef typename NumTraits::Real RealScalar; - typedef typename Base::CoeffReturnType CoeffReturnType; - - enum { - IsAligned = bool(EIGEN_MAX_ALIGN_BYTES>0) & !(Options_&DontAlign), - Layout = Options_ & RowMajor ? RowMajor : ColMajor, - CoordAccess = true, - RawAccess = true - }; - - static const int Options = Options_; - static const int NumIndices = NumIndices_; - typedef DSizes Dimensions; - - protected: - TensorStorage m_storage; +public: + typedef Tensor Self; + typedef TensorBase> Base; + typedef typename Eigen::internal::nested::type Nested; + typedef typename internal::traits::StorageKind StorageKind; + typedef typename internal::traits::Index Index; + typedef Scalar_ Scalar; + typedef typename NumTraits::Real RealScalar; + typedef typename Base::CoeffReturnType CoeffReturnType; + + enum { + IsAligned = bool(EIGEN_MAX_ALIGN_BYTES > 0) & !(Options_ & DontAlign), + Layout = Options_ & RowMajor ? RowMajor : ColMajor, + CoordAccess = true, + RawAccess = true + }; + + static const int Options = Options_; + static const int NumIndices = NumIndices_; + typedef DSizes Dimensions; + +protected: + TensorStorage m_storage; #ifdef EIGEN_HAS_SFINAE - template - struct isOfNormalIndex{ - static const bool is_array = internal::is_base_of, CustomIndices>::value; - static const bool is_int = NumTraits::IsInteger; - static const bool value = is_array | is_int; - }; + template struct isOfNormalIndex + { + static const bool is_array = internal::is_base_of, CustomIndices>::value; + static const bool is_int = NumTraits::IsInteger; + static const bool value = is_array | is_int; + }; #endif - public: - // Metadata - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Index rank() const { return NumIndices; } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Index dimension(std::size_t n) const { return m_storage.dimensions()[n]; } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const Dimensions& dimensions() const { return m_storage.dimensions(); } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Index size() const { return m_storage.size(); } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Scalar *data() { return m_storage.data(); } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const Scalar *data() const { return m_storage.data(); } - - // This makes EIGEN_INITIALIZE_COEFFS_IF_THAT_OPTION_IS_ENABLED - // work, because that uses base().coeffRef() - and we don't yet - // implement a similar class hierarchy - inline Self& base() { return *this; } - inline const Self& base() const { return *this; } +public: + // Metadata + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Index rank() const { return NumIndices; } + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Index dimension(std::size_t n) const { return m_storage.dimensions()[n]; } + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const Dimensions &dimensions() const { return m_storage.dimensions(); } + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Index size() const { return m_storage.size(); } + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Scalar *data() { return m_storage.data(); } + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const Scalar *data() const { return m_storage.data(); } + + // This makes EIGEN_INITIALIZE_COEFFS_IF_THAT_OPTION_IS_ENABLED + // work, because that uses base().coeffRef() - and we don't yet + // implement a similar class hierarchy + inline Self &base() { return *this; } + inline const Self &base() const { return *this; } #if EIGEN_HAS_VARIADIC_TEMPLATES - template - EIGEN_DEVICE_FUNC inline const Scalar& coeff(Index firstIndex, Index secondIndex, IndexTypes... otherIndices) const - { - // The number of indices used to access a tensor coefficient must be equal to the rank of the tensor. - EIGEN_STATIC_ASSERT(sizeof...(otherIndices) + 2 == NumIndices, YOU_MADE_A_PROGRAMMING_MISTAKE) - return coeff(array{{firstIndex, secondIndex, otherIndices...}}); - } + template + EIGEN_DEVICE_FUNC inline const Scalar &coeff(Index firstIndex, Index secondIndex, IndexTypes... otherIndices) const + { + // The number of indices used to access a tensor coefficient must be equal to the rank of the tensor. + EIGEN_STATIC_ASSERT(sizeof...(otherIndices) + 2 == NumIndices, YOU_MADE_A_PROGRAMMING_MISTAKE) + return coeff(array{ { firstIndex, secondIndex, otherIndices... } }); + } #endif - // normal indices - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const Scalar& coeff(const array& indices) const - { - eigen_internal_assert(checkIndexRange(indices)); - return m_storage.data()[linearizedIndex(indices)]; - } + // normal indices + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const Scalar &coeff(const array &indices) const + { + eigen_internal_assert(checkIndexRange(indices)); + return m_storage.data()[linearizedIndex(indices)]; + } - // custom indices + // custom indices #ifdef EIGEN_HAS_SFINAE - template::value) ) - > - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const Scalar& coeff(CustomIndices& indices) const - { - return coeff(internal::customIndices2Array(indices)); - } + template::value))> + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const Scalar &coeff(CustomIndices &indices) const + { + return coeff(internal::customIndices2Array(indices)); + } #endif - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const Scalar& coeff() const - { - EIGEN_STATIC_ASSERT(NumIndices == 0, YOU_MADE_A_PROGRAMMING_MISTAKE); - return m_storage.data()[0]; - } + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const Scalar &coeff() const + { + EIGEN_STATIC_ASSERT(NumIndices == 0, YOU_MADE_A_PROGRAMMING_MISTAKE); + return m_storage.data()[0]; + } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const Scalar& coeff(Index index) const - { - eigen_internal_assert(index >= 0 && index < size()); - return m_storage.data()[index]; - } + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const Scalar &coeff(Index index) const + { + eigen_internal_assert(index >= 0 && index < size()); + return m_storage.data()[index]; + } #if EIGEN_HAS_VARIADIC_TEMPLATES - template - inline Scalar& coeffRef(Index firstIndex, Index secondIndex, IndexTypes... otherIndices) - { - // The number of indices used to access a tensor coefficient must be equal to the rank of the tensor. - EIGEN_STATIC_ASSERT(sizeof...(otherIndices) + 2 == NumIndices, YOU_MADE_A_PROGRAMMING_MISTAKE) - return coeffRef(array{{firstIndex, secondIndex, otherIndices...}}); - } + template + inline Scalar &coeffRef(Index firstIndex, Index secondIndex, IndexTypes... otherIndices) + { + // The number of indices used to access a tensor coefficient must be equal to the rank of the tensor. + EIGEN_STATIC_ASSERT(sizeof...(otherIndices) + 2 == NumIndices, YOU_MADE_A_PROGRAMMING_MISTAKE) + return coeffRef(array{ { firstIndex, secondIndex, otherIndices... } }); + } #endif - // normal indices - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Scalar& coeffRef(const array& indices) - { - eigen_internal_assert(checkIndexRange(indices)); - return m_storage.data()[linearizedIndex(indices)]; - } + // normal indices + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Scalar &coeffRef(const array &indices) + { + eigen_internal_assert(checkIndexRange(indices)); + return m_storage.data()[linearizedIndex(indices)]; + } - // custom indices + // custom indices #ifdef EIGEN_HAS_SFINAE - template::value) ) - > - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Scalar& coeffRef(CustomIndices& indices) - { - return coeffRef(internal::customIndices2Array(indices)); - } + template::value))> + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Scalar &coeffRef(CustomIndices &indices) + { + return coeffRef(internal::customIndices2Array(indices)); + } #endif - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Scalar& coeffRef() - { - EIGEN_STATIC_ASSERT(NumIndices == 0, YOU_MADE_A_PROGRAMMING_MISTAKE); - return m_storage.data()[0]; - } + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Scalar &coeffRef() + { + EIGEN_STATIC_ASSERT(NumIndices == 0, YOU_MADE_A_PROGRAMMING_MISTAKE); + return m_storage.data()[0]; + } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Scalar& coeffRef(Index index) - { - eigen_internal_assert(index >= 0 && index < size()); - return m_storage.data()[index]; - } + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Scalar &coeffRef(Index index) + { + eigen_internal_assert(index >= 0 && index < size()); + return m_storage.data()[index]; + } #if EIGEN_HAS_VARIADIC_TEMPLATES - template - inline const Scalar& operator()(Index firstIndex, Index secondIndex, IndexTypes... otherIndices) const - { - // The number of indices used to access a tensor coefficient must be equal to the rank of the tensor. - EIGEN_STATIC_ASSERT(sizeof...(otherIndices) + 2 == NumIndices, YOU_MADE_A_PROGRAMMING_MISTAKE) - return this->operator()(array{{firstIndex, secondIndex, otherIndices...}}); - } + template + inline const Scalar &operator()(Index firstIndex, Index secondIndex, IndexTypes... otherIndices) const + { + // The number of indices used to access a tensor coefficient must be equal to the rank of the tensor. + EIGEN_STATIC_ASSERT(sizeof...(otherIndices) + 2 == NumIndices, YOU_MADE_A_PROGRAMMING_MISTAKE) + return this->operator()(array{ { firstIndex, secondIndex, otherIndices... } }); + } #else - EIGEN_DEVICE_FUNC - EIGEN_STRONG_INLINE const Scalar& operator()(Index i0, Index i1) const - { - return coeff(array(i0, i1)); - } - EIGEN_DEVICE_FUNC - EIGEN_STRONG_INLINE const Scalar& operator()(Index i0, Index i1, Index i2) const - { - return coeff(array(i0, i1, i2)); - } - EIGEN_DEVICE_FUNC - EIGEN_STRONG_INLINE const Scalar& operator()(Index i0, Index i1, Index i2, Index i3) const - { - return coeff(array(i0, i1, i2, i3)); - } - EIGEN_DEVICE_FUNC - EIGEN_STRONG_INLINE const Scalar& operator()(Index i0, Index i1, Index i2, Index i3, Index i4) const - { - return coeff(array(i0, i1, i2, i3, i4)); - } + EIGEN_DEVICE_FUNC + EIGEN_STRONG_INLINE const Scalar &operator()(Index i0, Index i1) const { return coeff(array(i0, i1)); } + EIGEN_DEVICE_FUNC + EIGEN_STRONG_INLINE const Scalar &operator()(Index i0, Index i1, Index i2) const + { + return coeff(array(i0, i1, i2)); + } + EIGEN_DEVICE_FUNC + EIGEN_STRONG_INLINE const Scalar &operator()(Index i0, Index i1, Index i2, Index i3) const + { + return coeff(array(i0, i1, i2, i3)); + } + EIGEN_DEVICE_FUNC + EIGEN_STRONG_INLINE const Scalar &operator()(Index i0, Index i1, Index i2, Index i3, Index i4) const + { + return coeff(array(i0, i1, i2, i3, i4)); + } #endif - // custom indices + // custom indices #ifdef EIGEN_HAS_SFINAE - template::value) ) - > - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const Scalar& operator()(CustomIndices& indices) const - { - return coeff(internal::customIndices2Array(indices)); - } + template::value))> + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const Scalar &operator()(CustomIndices &indices) const + { + return coeff(internal::customIndices2Array(indices)); + } #endif - // normal indices - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const Scalar& operator()(const array& indices) const - { - return coeff(indices); - } - - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const Scalar& operator()(Index index) const - { - eigen_internal_assert(index >= 0 && index < size()); - return coeff(index); - } - - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const Scalar& operator()() const - { - EIGEN_STATIC_ASSERT(NumIndices == 0, YOU_MADE_A_PROGRAMMING_MISTAKE); - return coeff(); - } - - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const Scalar& operator[](Index index) const - { - // The bracket operator is only for vectors, use the parenthesis operator instead. - EIGEN_STATIC_ASSERT(NumIndices == 1, YOU_MADE_A_PROGRAMMING_MISTAKE); - return coeff(index); - } + // normal indices + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const Scalar &operator()(const array &indices) const + { + return coeff(indices); + } + + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const Scalar &operator()(Index index) const + { + eigen_internal_assert(index >= 0 && index < size()); + return coeff(index); + } + + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const Scalar &operator()() const + { + EIGEN_STATIC_ASSERT(NumIndices == 0, YOU_MADE_A_PROGRAMMING_MISTAKE); + return coeff(); + } + + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const Scalar &operator[](Index index) const + { + // The bracket operator is only for vectors, use the parenthesis operator instead. + EIGEN_STATIC_ASSERT(NumIndices == 1, YOU_MADE_A_PROGRAMMING_MISTAKE); + return coeff(index); + } #if EIGEN_HAS_VARIADIC_TEMPLATES - template - inline Scalar& operator()(Index firstIndex, Index secondIndex, IndexTypes... otherIndices) - { - // The number of indices used to access a tensor coefficient must be equal to the rank of the tensor. - EIGEN_STATIC_ASSERT(sizeof...(otherIndices) + 2 == NumIndices, YOU_MADE_A_PROGRAMMING_MISTAKE) - return operator()(array{{firstIndex, secondIndex, otherIndices...}}); - } + template + inline Scalar &operator()(Index firstIndex, Index secondIndex, IndexTypes... otherIndices) + { + // The number of indices used to access a tensor coefficient must be equal to the rank of the tensor. + EIGEN_STATIC_ASSERT(sizeof...(otherIndices) + 2 == NumIndices, YOU_MADE_A_PROGRAMMING_MISTAKE) + return operator()(array{ { firstIndex, secondIndex, otherIndices... } }); + } #else - EIGEN_DEVICE_FUNC - EIGEN_STRONG_INLINE Scalar& operator()(Index i0, Index i1) - { - return coeffRef(array(i0, i1)); - } - EIGEN_DEVICE_FUNC - EIGEN_STRONG_INLINE Scalar& operator()(Index i0, Index i1, Index i2) - { - return coeffRef(array(i0, i1, i2)); - } - EIGEN_DEVICE_FUNC - EIGEN_STRONG_INLINE Scalar& operator()(Index i0, Index i1, Index i2, Index i3) - { - return coeffRef(array(i0, i1, i2, i3)); - } - EIGEN_DEVICE_FUNC - EIGEN_STRONG_INLINE Scalar& operator()(Index i0, Index i1, Index i2, Index i3, Index i4) - { - return coeffRef(array(i0, i1, i2, i3, i4)); - } + EIGEN_DEVICE_FUNC + EIGEN_STRONG_INLINE Scalar &operator()(Index i0, Index i1) { return coeffRef(array(i0, i1)); } + EIGEN_DEVICE_FUNC + EIGEN_STRONG_INLINE Scalar &operator()(Index i0, Index i1, Index i2) { return coeffRef(array(i0, i1, i2)); } + EIGEN_DEVICE_FUNC + EIGEN_STRONG_INLINE Scalar &operator()(Index i0, Index i1, Index i2, Index i3) + { + return coeffRef(array(i0, i1, i2, i3)); + } + EIGEN_DEVICE_FUNC + EIGEN_STRONG_INLINE Scalar &operator()(Index i0, Index i1, Index i2, Index i3, Index i4) + { + return coeffRef(array(i0, i1, i2, i3, i4)); + } #endif - // normal indices - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Scalar& operator()(const array& indices) - { - return coeffRef(indices); - } + // normal indices + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Scalar &operator()(const array &indices) + { + return coeffRef(indices); + } - // custom indices + // custom indices #ifdef EIGEN_HAS_SFINAE - template::value) ) - > - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Scalar& operator()(CustomIndices& indices) - { - return coeffRef(internal::customIndices2Array(indices)); - } + template::value))> + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Scalar &operator()(CustomIndices &indices) + { + return coeffRef(internal::customIndices2Array(indices)); + } #endif - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Scalar& operator()(Index index) - { - eigen_assert(index >= 0 && index < size()); - return coeffRef(index); - } + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Scalar &operator()(Index index) + { + eigen_assert(index >= 0 && index < size()); + return coeffRef(index); + } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Scalar& operator()() - { - EIGEN_STATIC_ASSERT(NumIndices == 0, YOU_MADE_A_PROGRAMMING_MISTAKE); - return coeffRef(); - } + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Scalar &operator()() + { + EIGEN_STATIC_ASSERT(NumIndices == 0, YOU_MADE_A_PROGRAMMING_MISTAKE); + return coeffRef(); + } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Scalar& operator[](Index index) - { - // The bracket operator is only for vectors, use the parenthesis operator instead - EIGEN_STATIC_ASSERT(NumIndices == 1, YOU_MADE_A_PROGRAMMING_MISTAKE) - return coeffRef(index); - } + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Scalar &operator[](Index index) + { + // The bracket operator is only for vectors, use the parenthesis operator instead + EIGEN_STATIC_ASSERT(NumIndices == 1, YOU_MADE_A_PROGRAMMING_MISTAKE) + return coeffRef(index); + } - EIGEN_DEVICE_FUNC - EIGEN_STRONG_INLINE Tensor() - : m_storage() - { - } + EIGEN_DEVICE_FUNC + EIGEN_STRONG_INLINE Tensor() : m_storage() {} - EIGEN_DEVICE_FUNC - EIGEN_STRONG_INLINE Tensor(const Self& other) - : m_storage(other.m_storage) - { - } + EIGEN_DEVICE_FUNC + EIGEN_STRONG_INLINE Tensor(const Self &other) : m_storage(other.m_storage) {} #if EIGEN_HAS_VARIADIC_TEMPLATES - template + template EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Tensor(Index firstDimension, IndexTypes... otherDimensions) : m_storage(firstDimension, otherDimensions...) { @@ -345,183 +323,160 @@ class Tensor : public TensorBase(dim1)) - { - EIGEN_STATIC_ASSERT(1 == NumIndices, YOU_MADE_A_PROGRAMMING_MISTAKE) - } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Tensor(Index dim1, Index dim2) - : m_storage(dim1*dim2, array(dim1, dim2)) - { - EIGEN_STATIC_ASSERT(2 == NumIndices, YOU_MADE_A_PROGRAMMING_MISTAKE) - } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Tensor(Index dim1, Index dim2, Index dim3) - : m_storage(dim1*dim2*dim3, array(dim1, dim2, dim3)) - { - EIGEN_STATIC_ASSERT(3 == NumIndices, YOU_MADE_A_PROGRAMMING_MISTAKE) - } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Tensor(Index dim1, Index dim2, Index dim3, Index dim4) - : m_storage(dim1*dim2*dim3*dim4, array(dim1, dim2, dim3, dim4)) - { - EIGEN_STATIC_ASSERT(4 == NumIndices, YOU_MADE_A_PROGRAMMING_MISTAKE) - } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Tensor(Index dim1, Index dim2, Index dim3, Index dim4, Index dim5) - : m_storage(dim1*dim2*dim3*dim4*dim5, array(dim1, dim2, dim3, dim4, dim5)) - { - EIGEN_STATIC_ASSERT(5 == NumIndices, YOU_MADE_A_PROGRAMMING_MISTAKE) - } + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE explicit Tensor(Index dim1) + : m_storage(dim1, array(dim1)){ EIGEN_STATIC_ASSERT(1 == NumIndices, + YOU_MADE_A_PROGRAMMING_MISTAKE) } EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Tensor(Index dim1, Index dim2) + : m_storage(dim1 * dim2, array(dim1, dim2)){ EIGEN_STATIC_ASSERT(2 == NumIndices, + YOU_MADE_A_PROGRAMMING_MISTAKE) } EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE + Tensor(Index dim1, Index dim2, Index dim3) + : m_storage(dim1 * dim2 * dim3, array(dim1, dim2, dim3)){ EIGEN_STATIC_ASSERT(3 == NumIndices, + YOU_MADE_A_PROGRAMMING_MISTAKE) } EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE + Tensor(Index dim1, Index dim2, Index dim3, Index dim4) + : m_storage(dim1 * dim2 * dim3 * dim4, + array(dim1, dim2, dim3, dim4)){ EIGEN_STATIC_ASSERT(4 == NumIndices, + YOU_MADE_A_PROGRAMMING_MISTAKE) } EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE + Tensor(Index dim1, Index dim2, Index dim3, Index dim4, Index dim5) + : m_storage(dim1 * dim2 * dim3 * dim4 * dim5, + array(dim1, dim2, dim3, dim4, dim5)){ EIGEN_STATIC_ASSERT(5 == NumIndices, + YOU_MADE_A_PROGRAMMING_MISTAKE) } #endif /** Normal Dimension */ EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE explicit Tensor(const array& dimensions) : m_storage(internal::array_prod(dimensions), dimensions) - { - EIGEN_INITIALIZE_COEFFS_IF_THAT_OPTION_IS_ENABLED - } - - template - EIGEN_DEVICE_FUNC - EIGEN_STRONG_INLINE Tensor(const TensorBase& other) - { - typedef TensorAssignOp Assign; - Assign assign(*this, other.derived()); - resize(TensorEvaluator(assign, DefaultDevice()).dimensions()); - internal::TensorExecutor::run(assign, DefaultDevice()); - } - template - EIGEN_DEVICE_FUNC - EIGEN_STRONG_INLINE Tensor(const TensorBase& other) - { - typedef TensorAssignOp Assign; - Assign assign(*this, other.derived()); - resize(TensorEvaluator(assign, DefaultDevice()).dimensions()); - internal::TensorExecutor::run(assign, DefaultDevice()); - } - - EIGEN_DEVICE_FUNC - EIGEN_STRONG_INLINE Tensor& operator=(const Tensor& other) - { - typedef TensorAssignOp Assign; - Assign assign(*this, other); - resize(TensorEvaluator(assign, DefaultDevice()).dimensions()); - internal::TensorExecutor::run(assign, DefaultDevice()); - return *this; - } - template - EIGEN_DEVICE_FUNC - EIGEN_STRONG_INLINE Tensor& operator=(const OtherDerived& other) - { - typedef TensorAssignOp Assign; - Assign assign(*this, other); - resize(TensorEvaluator(assign, DefaultDevice()).dimensions()); - internal::TensorExecutor::run(assign, DefaultDevice()); - return *this; - } + { + EIGEN_INITIALIZE_COEFFS_IF_THAT_OPTION_IS_ENABLED + } + + template + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Tensor(const TensorBase &other) + { + typedef TensorAssignOp Assign; + Assign assign(*this, other.derived()); + resize(TensorEvaluator(assign, DefaultDevice()).dimensions()); + internal::TensorExecutor::run(assign, DefaultDevice()); + } + template + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Tensor(const TensorBase &other) + { + typedef TensorAssignOp Assign; + Assign assign(*this, other.derived()); + resize(TensorEvaluator(assign, DefaultDevice()).dimensions()); + internal::TensorExecutor::run(assign, DefaultDevice()); + } + + EIGEN_DEVICE_FUNC + EIGEN_STRONG_INLINE Tensor &operator=(const Tensor &other) + { + typedef TensorAssignOp Assign; + Assign assign(*this, other); + resize(TensorEvaluator(assign, DefaultDevice()).dimensions()); + internal::TensorExecutor::run(assign, DefaultDevice()); + return *this; + } + template EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Tensor &operator=(const OtherDerived &other) + { + typedef TensorAssignOp Assign; + Assign assign(*this, other); + resize(TensorEvaluator(assign, DefaultDevice()).dimensions()); + internal::TensorExecutor::run(assign, DefaultDevice()); + return *this; + } #if EIGEN_HAS_VARIADIC_TEMPLATES - template EIGEN_DEVICE_FUNC - void resize(Index firstDimension, IndexTypes... otherDimensions) - { - // The number of dimensions used to resize a tensor must be equal to the rank of the tensor. - EIGEN_STATIC_ASSERT(sizeof...(otherDimensions) + 1 == NumIndices, YOU_MADE_A_PROGRAMMING_MISTAKE) - resize(array{{firstDimension, otherDimensions...}}); - } + template EIGEN_DEVICE_FUNC void resize(Index firstDimension, IndexTypes... otherDimensions) + { + // The number of dimensions used to resize a tensor must be equal to the rank of the tensor. + EIGEN_STATIC_ASSERT(sizeof...(otherDimensions) + 1 == NumIndices, YOU_MADE_A_PROGRAMMING_MISTAKE) + resize(array{ { firstDimension, otherDimensions... } }); + } #endif - /** Normal Dimension */ - EIGEN_DEVICE_FUNC void resize(const array& dimensions) - { - int i; - Index size = Index(1); - for (i = 0; i < NumIndices; i++) { - internal::check_rows_cols_for_overflow::run(size, dimensions[i]); - size *= dimensions[i]; - } - #ifdef EIGEN_INITIALIZE_COEFFS - bool size_changed = size != this->size(); - m_storage.resize(size, dimensions); - if(size_changed) EIGEN_INITIALIZE_COEFFS_IF_THAT_OPTION_IS_ENABLED - #else - m_storage.resize(size, dimensions); - #endif - } - - // Why this overload, DSizes is derived from array ??? // - EIGEN_DEVICE_FUNC void resize(const DSizes& dimensions) { - array dims; - for (int i = 0; i < NumIndices; ++i) { - dims[i] = dimensions[i]; - } - resize(dims); - } - - EIGEN_DEVICE_FUNC - void resize() - { - EIGEN_STATIC_ASSERT(NumIndices == 0, YOU_MADE_A_PROGRAMMING_MISTAKE); - // Nothing to do: rank 0 tensors have fixed size - } - - /** Custom Dimension */ + /** Normal Dimension */ + EIGEN_DEVICE_FUNC void resize(const array &dimensions) + { + int i; + Index size = Index(1); + for (i = 0; i < NumIndices; i++) { + internal::check_rows_cols_for_overflow::run(size, dimensions[i]); + size *= dimensions[i]; + } +#ifdef EIGEN_INITIALIZE_COEFFS + bool size_changed = size != this->size(); + m_storage.resize(size, dimensions); + if (size_changed) EIGEN_INITIALIZE_COEFFS_IF_THAT_OPTION_IS_ENABLED +#else + m_storage.resize(size, dimensions); +#endif + } + + // Why this overload, DSizes is derived from array ??? // + EIGEN_DEVICE_FUNC void resize(const DSizes &dimensions) + { + array dims; + for (int i = 0; i < NumIndices; ++i) { dims[i] = dimensions[i]; } + resize(dims); + } + + EIGEN_DEVICE_FUNC + void resize() + { + EIGEN_STATIC_ASSERT(NumIndices == 0, YOU_MADE_A_PROGRAMMING_MISTAKE); + // Nothing to do: rank 0 tensors have fixed size + } + + /** Custom Dimension */ #ifdef EIGEN_HAS_SFINAE - template::value) ) - > - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void resize(CustomDimension& dimensions) - { - resize(internal::customIndices2Array(dimensions)); - } + template::value))> + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void resize(CustomDimension &dimensions) + { + resize(internal::customIndices2Array(dimensions)); + } #endif #ifndef EIGEN_EMULATE_CXX11_META_H - template - EIGEN_DEVICE_FUNC - void resize(const Sizes& dimensions) { - array dims; - for (int i = 0; i < NumIndices; ++i) { - dims[i] = static_cast(dimensions[i]); - } - resize(dims); - } + template EIGEN_DEVICE_FUNC void resize(const Sizes &dimensions) + { + array dims; + for (int i = 0; i < NumIndices; ++i) { dims[i] = static_cast(dimensions[i]); } + resize(dims); + } #else - template - EIGEN_DEVICE_FUNC - void resize(const Sizes& dimensions) { - array dims; - for (int i = 0; i < NumIndices; ++i) { - dims[i] = static_cast(dimensions[i]); - } - resize(dims); - } + template + EIGEN_DEVICE_FUNC void resize(const Sizes &dimensions) + { + array dims; + for (int i = 0; i < NumIndices; ++i) { dims[i] = static_cast(dimensions[i]); } + resize(dims); + } #endif - protected: - - bool checkIndexRange(const array& indices) const - { - using internal::array_apply_and_reduce; - using internal::array_zip_and_reduce; - using internal::greater_equal_zero_op; - using internal::logical_and_op; - using internal::lesser_op; - - return - // check whether the indices are all >= 0 - array_apply_and_reduce(indices) && - // check whether the indices fit in the dimensions - array_zip_and_reduce(indices, m_storage.dimensions()); - } - - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Index linearizedIndex(const array& indices) const - { - if (Options&RowMajor) { - return m_storage.dimensions().IndexOfRowMajor(indices); - } else { - return m_storage.dimensions().IndexOfColMajor(indices); - } - } +protected: + bool checkIndexRange(const array &indices) const + { + using internal::array_apply_and_reduce; + using internal::array_zip_and_reduce; + using internal::greater_equal_zero_op; + using internal::logical_and_op; + using internal::lesser_op; + + return + // check whether the indices are all >= 0 + array_apply_and_reduce(indices) && + // check whether the indices fit in the dimensions + array_zip_and_reduce(indices, m_storage.dimensions()); + } + + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Index linearizedIndex(const array &indices) const + { + if (Options & RowMajor) { + return m_storage.dimensions().IndexOfRowMajor(indices); + } else { + return m_storage.dimensions().IndexOfColMajor(indices); + } + } }; -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_CXX11_TENSOR_TENSOR_H +#endif// EIGEN_CXX11_TENSOR_TENSOR_H diff --git a/filmulator-gui/core/nlmeans/eigen/unsupported/Eigen/CXX11/src/Tensor/TensorArgMax.h b/filmulator-gui/core/nlmeans/eigen/unsupported/Eigen/CXX11/src/Tensor/TensorArgMax.h index d06f40cd..222bb2be 100644 --- a/filmulator-gui/core/nlmeans/eigen/unsupported/Eigen/CXX11/src/Tensor/TensorArgMax.h +++ b/filmulator-gui/core/nlmeans/eigen/unsupported/Eigen/CXX11/src/Tensor/TensorArgMax.h @@ -14,45 +14,41 @@ namespace Eigen { namespace internal { -/** \class TensorIndexTuple - * \ingroup CXX11_Tensor_Module - * - * \brief Tensor + Index Tuple class. - * - * - */ -template -struct traits > : public traits -{ - typedef traits XprTraits; - typedef typename XprTraits::StorageKind StorageKind; - typedef typename XprTraits::Index Index; - typedef Tuple Scalar; - typedef typename XprType::Nested Nested; - typedef typename remove_reference::type _Nested; - static const int NumDimensions = XprTraits::NumDimensions; - static const int Layout = XprTraits::Layout; -}; + /** \class TensorIndexTuple + * \ingroup CXX11_Tensor_Module + * + * \brief Tensor + Index Tuple class. + * + * + */ + template struct traits> : public traits + { + typedef traits XprTraits; + typedef typename XprTraits::StorageKind StorageKind; + typedef typename XprTraits::Index Index; + typedef Tuple Scalar; + typedef typename XprType::Nested Nested; + typedef typename remove_reference::type _Nested; + static const int NumDimensions = XprTraits::NumDimensions; + static const int Layout = XprTraits::Layout; + }; -template -struct eval, Eigen::Dense> -{ - typedef const TensorIndexTupleOp& type; -}; + template struct eval, Eigen::Dense> + { + typedef const TensorIndexTupleOp &type; + }; -template -struct nested, 1, - typename eval >::type> -{ - typedef TensorIndexTupleOp type; -}; + template + struct nested, 1, typename eval>::type> + { + typedef TensorIndexTupleOp type; + }; -} // end namespace internal +}// end namespace internal -template -class TensorIndexTupleOp : public TensorBase, ReadOnlyAccessors> +template class TensorIndexTupleOp : public TensorBase, ReadOnlyAccessors> { - public: +public: typedef typename Eigen::internal::traits::Scalar Scalar; typedef typename Eigen::NumTraits::Real RealScalar; typedef typename Eigen::internal::nested::type Nested; @@ -60,20 +56,17 @@ class TensorIndexTupleOp : public TensorBase, ReadOn typedef typename Eigen::internal::traits::Index Index; typedef Tuple CoeffReturnType; - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TensorIndexTupleOp(const XprType& expr) - : m_xpr(expr) {} + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TensorIndexTupleOp(const XprType &expr) : m_xpr(expr) {} EIGEN_DEVICE_FUNC - const typename internal::remove_all::type& - expression() const { return m_xpr; } + const typename internal::remove_all::type &expression() const { return m_xpr; } - protected: - typename XprType::Nested m_xpr; +protected: + typename XprType::Nested m_xpr; }; // Eval as rvalue -template -struct TensorEvaluator, Device> +template struct TensorEvaluator, Device> { typedef TensorIndexTupleOp XprType; typedef typename XprType::Index Index; @@ -88,81 +81,80 @@ struct TensorEvaluator, Device> PacketAccess = /*TensorEvaluator::PacketAccess*/ false, BlockAccess = false, Layout = TensorEvaluator::Layout, - CoordAccess = false, // to be implemented + CoordAccess = false,// to be implemented RawAccess = false }; - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TensorEvaluator(const XprType& op, const Device& device) - : m_impl(op.expression(), device) { } + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TensorEvaluator(const XprType &op, const Device &device) + : m_impl(op.expression(), device) + {} - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const Dimensions& dimensions() const { - return m_impl.dimensions(); - } + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const Dimensions &dimensions() const { return m_impl.dimensions(); } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE bool evalSubExprsIfNeeded(Scalar* /*data*/) { + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE bool evalSubExprsIfNeeded(Scalar * /*data*/) + { m_impl.evalSubExprsIfNeeded(NULL); return true; } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void cleanup() { - m_impl.cleanup(); - } + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void cleanup() { m_impl.cleanup(); } EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE CoeffReturnType coeff(Index index) const { return CoeffReturnType(index, m_impl.coeff(index)); } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TensorOpCost - costPerCoeff(bool vectorized) const { + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TensorOpCost costPerCoeff(bool vectorized) const + { return m_impl.costPerCoeff(vectorized) + TensorOpCost(0, 0, 1); } - EIGEN_DEVICE_FUNC Scalar* data() const { return NULL; } + EIGEN_DEVICE_FUNC Scalar *data() const { return NULL; } - protected: +protected: TensorEvaluator m_impl; }; namespace internal { -/** \class TensorTupleIndex - * \ingroup CXX11_Tensor_Module - * - * \brief Converts to Tensor > and reduces to Tensor. - * - */ -template -struct traits > : public traits -{ - typedef traits XprTraits; - typedef typename XprTraits::StorageKind StorageKind; - typedef typename XprTraits::Index Index; - typedef Index Scalar; - typedef typename XprType::Nested Nested; - typedef typename remove_reference::type _Nested; - static const int NumDimensions = XprTraits::NumDimensions - array_size::value; - static const int Layout = XprTraits::Layout; -}; + /** \class TensorTupleIndex + * \ingroup CXX11_Tensor_Module + * + * \brief Converts to Tensor > and reduces to Tensor. + * + */ + template + struct traits> : public traits + { + typedef traits XprTraits; + typedef typename XprTraits::StorageKind StorageKind; + typedef typename XprTraits::Index Index; + typedef Index Scalar; + typedef typename XprType::Nested Nested; + typedef typename remove_reference::type _Nested; + static const int NumDimensions = XprTraits::NumDimensions - array_size::value; + static const int Layout = XprTraits::Layout; + }; -template -struct eval, Eigen::Dense> -{ - typedef const TensorTupleReducerOp& type; -}; + template + struct eval, Eigen::Dense> + { + typedef const TensorTupleReducerOp &type; + }; -template -struct nested, 1, - typename eval >::type> -{ - typedef TensorTupleReducerOp type; -}; + template + struct nested, + 1, + typename eval>::type> + { + typedef TensorTupleReducerOp type; + }; -} // end namespace internal +}// end namespace internal template class TensorTupleReducerOp : public TensorBase, ReadOnlyAccessors> { - public: +public: typedef typename Eigen::internal::traits::Scalar Scalar; typedef typename Eigen::NumTraits::Real RealScalar; typedef typename Eigen::internal::nested::type Nested; @@ -170,30 +162,28 @@ class TensorTupleReducerOp : public TensorBase::Index Index; typedef Index CoeffReturnType; - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TensorTupleReducerOp(const XprType& expr, - const ReduceOp& reduce_op, - const int return_dim, - const Dims& reduce_dims) - : m_xpr(expr), m_reduce_op(reduce_op), m_return_dim(return_dim), m_reduce_dims(reduce_dims) {} + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE + TensorTupleReducerOp(const XprType &expr, const ReduceOp &reduce_op, const int return_dim, const Dims &reduce_dims) + : m_xpr(expr), m_reduce_op(reduce_op), m_return_dim(return_dim), m_reduce_dims(reduce_dims) + {} EIGEN_DEVICE_FUNC - const typename internal::remove_all::type& - expression() const { return m_xpr; } + const typename internal::remove_all::type &expression() const { return m_xpr; } EIGEN_DEVICE_FUNC - const ReduceOp& reduce_op() const { return m_reduce_op; } + const ReduceOp &reduce_op() const { return m_reduce_op; } EIGEN_DEVICE_FUNC - const Dims& reduce_dims() const { return m_reduce_dims; } + const Dims &reduce_dims() const { return m_reduce_dims; } EIGEN_DEVICE_FUNC int return_dim() const { return m_return_dim; } - protected: - typename XprType::Nested m_xpr; - const ReduceOp m_reduce_op; - const int m_return_dim; - const Dims m_reduce_dims; +protected: + typename XprType::Nested m_xpr; + const ReduceOp m_reduce_op; + const int m_return_dim; + const Dims m_reduce_dims; }; // Eval as rvalue @@ -205,8 +195,9 @@ struct TensorEvaluator, Devi typedef typename XprType::Scalar Scalar; typedef typename XprType::CoeffReturnType CoeffReturnType; typedef typename TensorIndexTupleOp::CoeffReturnType TupleType; - typedef typename TensorEvaluator >, Device>::Dimensions Dimensions; - typedef typename TensorEvaluator , Device>::Dimensions InputDimensions; + typedef typename TensorEvaluator>, + Device>::Dimensions Dimensions; + typedef typename TensorEvaluator, Device>::Dimensions InputDimensions; static const int NumDims = internal::array_size::value; typedef array StrideDims; @@ -214,15 +205,17 @@ struct TensorEvaluator, Devi IsAligned = /*TensorEvaluator::IsAligned*/ false, PacketAccess = /*TensorEvaluator::PacketAccess*/ false, BlockAccess = false, - Layout = TensorEvaluator >, Device>::Layout, - CoordAccess = false, // to be implemented + Layout = + TensorEvaluator>, Device>::Layout, + CoordAccess = false,// to be implemented RawAccess = false }; - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TensorEvaluator(const XprType& op, const Device& device) - : m_orig_impl(op.expression(), device), - m_impl(op.expression().index_tuples().reduce(op.reduce_dims(), op.reduce_op()), device), - m_return_dim(op.return_dim()) { + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TensorEvaluator(const XprType &op, const Device &device) + : m_orig_impl(op.expression(), device), + m_impl(op.expression().index_tuples().reduce(op.reduce_dims(), op.reduce_op()), device), + m_return_dim(op.return_dim()) + { gen_strides(m_orig_impl.dimensions(), m_strides); if (Layout == static_cast(ColMajor)) { @@ -235,65 +228,58 @@ struct TensorEvaluator, Devi m_stride_div = m_strides[m_return_dim]; } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const Dimensions& dimensions() const { - return m_impl.dimensions(); - } + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const Dimensions &dimensions() const { return m_impl.dimensions(); } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE bool evalSubExprsIfNeeded(Scalar* /*data*/) { + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE bool evalSubExprsIfNeeded(Scalar * /*data*/) + { m_impl.evalSubExprsIfNeeded(NULL); return true; } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void cleanup() { - m_impl.cleanup(); - } + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void cleanup() { m_impl.cleanup(); } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE CoeffReturnType coeff(Index index) const { + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE CoeffReturnType coeff(Index index) const + { const TupleType v = m_impl.coeff(index); return (m_return_dim < 0) ? v.first : (v.first % m_stride_mod) / m_stride_div; } - EIGEN_DEVICE_FUNC Scalar* data() const { return NULL; } + EIGEN_DEVICE_FUNC Scalar *data() const { return NULL; } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TensorOpCost - costPerCoeff(bool vectorized) const { - const double compute_cost = 1.0 + - (m_return_dim < 0 ? 0.0 : (TensorOpCost::ModCost() + TensorOpCost::DivCost())); - return m_orig_impl.costPerCoeff(vectorized) + - m_impl.costPerCoeff(vectorized) + TensorOpCost(0, 0, compute_cost); + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TensorOpCost costPerCoeff(bool vectorized) const + { + const double compute_cost = + 1.0 + (m_return_dim < 0 ? 0.0 : (TensorOpCost::ModCost() + TensorOpCost::DivCost())); + return m_orig_impl.costPerCoeff(vectorized) + m_impl.costPerCoeff(vectorized) + TensorOpCost(0, 0, compute_cost); } - private: - EIGEN_DEVICE_FUNC void gen_strides(const InputDimensions& dims, StrideDims& strides) { +private: + EIGEN_DEVICE_FUNC void gen_strides(const InputDimensions &dims, StrideDims &strides) + { if (m_return_dim < 0) { - return; // Won't be using the strides. + return;// Won't be using the strides. } - eigen_assert(m_return_dim < NumDims && - "Asking to convert index to a dimension outside of the rank"); + eigen_assert(m_return_dim < NumDims && "Asking to convert index to a dimension outside of the rank"); // Calculate m_stride_div and m_stride_mod, which are used to // calculate the value of an index w.r.t. the m_return_dim. if (Layout == static_cast(ColMajor)) { strides[0] = 1; - for (int i = 1; i < NumDims; ++i) { - strides[i] = strides[i-1] * dims[i-1]; - } + for (int i = 1; i < NumDims; ++i) { strides[i] = strides[i - 1] * dims[i - 1]; } } else { - strides[NumDims-1] = 1; - for (int i = NumDims - 2; i >= 0; --i) { - strides[i] = strides[i+1] * dims[i+1]; - } + strides[NumDims - 1] = 1; + for (int i = NumDims - 2; i >= 0; --i) { strides[i] = strides[i + 1] * dims[i + 1]; } } } - protected: +protected: TensorEvaluator, Device> m_orig_impl; - TensorEvaluator >, Device> m_impl; + TensorEvaluator>, Device> m_impl; const int m_return_dim; StrideDims m_strides; Index m_stride_mod; Index m_stride_div; }; -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_CXX11_TENSOR_TENSOR_ARG_MAX_H +#endif// EIGEN_CXX11_TENSOR_TENSOR_ARG_MAX_H diff --git a/filmulator-gui/core/nlmeans/eigen/unsupported/Eigen/CXX11/src/Tensor/TensorAssign.h b/filmulator-gui/core/nlmeans/eigen/unsupported/Eigen/CXX11/src/Tensor/TensorAssign.h index 166be200..bf18cd86 100644 --- a/filmulator-gui/core/nlmeans/eigen/unsupported/Eigen/CXX11/src/Tensor/TensorAssign.h +++ b/filmulator-gui/core/nlmeans/eigen/unsupported/Eigen/CXX11/src/Tensor/TensorAssign.h @@ -13,53 +13,48 @@ namespace Eigen { /** \class TensorAssign - * \ingroup CXX11_Tensor_Module - * - * \brief The tensor assignment class. - * - * This class is represents the assignment of the values resulting from the evaluation of - * the rhs expression to the memory locations denoted by the lhs expression. - */ + * \ingroup CXX11_Tensor_Module + * + * \brief The tensor assignment class. + * + * This class is represents the assignment of the values resulting from the evaluation of + * the rhs expression to the memory locations denoted by the lhs expression. + */ namespace internal { -template -struct traits > -{ - typedef typename LhsXprType::Scalar Scalar; - typedef typename traits::StorageKind StorageKind; - typedef typename promote_index_type::Index, - typename traits::Index>::type Index; - typedef typename LhsXprType::Nested LhsNested; - typedef typename RhsXprType::Nested RhsNested; - typedef typename remove_reference::type _LhsNested; - typedef typename remove_reference::type _RhsNested; - static const std::size_t NumDimensions = internal::traits::NumDimensions; - static const int Layout = internal::traits::Layout; - - enum { - Flags = 0 + template struct traits> + { + typedef typename LhsXprType::Scalar Scalar; + typedef typename traits::StorageKind StorageKind; + typedef + typename promote_index_type::Index, typename traits::Index>::type Index; + typedef typename LhsXprType::Nested LhsNested; + typedef typename RhsXprType::Nested RhsNested; + typedef typename remove_reference::type _LhsNested; + typedef typename remove_reference::type _RhsNested; + static const std::size_t NumDimensions = internal::traits::NumDimensions; + static const int Layout = internal::traits::Layout; + + enum { Flags = 0 }; }; -}; -template -struct eval, Eigen::Dense> -{ - typedef const TensorAssignOp& type; -}; - -template -struct nested, 1, typename eval >::type> -{ - typedef TensorAssignOp type; -}; + template struct eval, Eigen::Dense> + { + typedef const TensorAssignOp &type; + }; -} // end namespace internal + template + struct nested, 1, typename eval>::type> + { + typedef TensorAssignOp type; + }; +}// end namespace internal template -class TensorAssignOp : public TensorBase > +class TensorAssignOp : public TensorBase> { - public: +public: typedef typename Eigen::internal::traits::Scalar Scalar; typedef typename Eigen::NumTraits::Real RealScalar; typedef typename LhsXprType::CoeffReturnType CoeffReturnType; @@ -67,21 +62,23 @@ class TensorAssignOp : public TensorBase typedef typename Eigen::internal::traits::StorageKind StorageKind; typedef typename Eigen::internal::traits::Index Index; - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TensorAssignOp(LhsXprType& lhs, const RhsXprType& rhs) - : m_lhs_xpr(lhs), m_rhs_xpr(rhs) {} + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TensorAssignOp(LhsXprType &lhs, const RhsXprType &rhs) + : m_lhs_xpr(lhs), m_rhs_xpr(rhs) + {} - /** \returns the nested expressions */ - EIGEN_DEVICE_FUNC - typename internal::remove_all::type& - lhsExpression() const { return *((typename internal::remove_all::type*)&m_lhs_xpr); } + /** \returns the nested expressions */ + EIGEN_DEVICE_FUNC + typename internal::remove_all::type &lhsExpression() const + { + return *((typename internal::remove_all::type *)&m_lhs_xpr); + } - EIGEN_DEVICE_FUNC - const typename internal::remove_all::type& - rhsExpression() const { return m_rhs_xpr; } + EIGEN_DEVICE_FUNC + const typename internal::remove_all::type &rhsExpression() const { return m_rhs_xpr; } - protected: - typename internal::remove_all::type& m_lhs_xpr; - const typename internal::remove_all::type& m_rhs_xpr; +protected: + typename internal::remove_all::type &m_lhs_xpr; + const typename internal::remove_all::type &m_rhs_xpr; }; @@ -98,19 +95,21 @@ struct TensorEvaluator, Device> enum { IsAligned = TensorEvaluator::IsAligned & TensorEvaluator::IsAligned, - PacketAccess = TensorEvaluator::PacketAccess & TensorEvaluator::PacketAccess, + PacketAccess = + TensorEvaluator::PacketAccess & TensorEvaluator::PacketAccess, Layout = TensorEvaluator::Layout, RawAccess = TensorEvaluator::RawAccess }; - EIGEN_DEVICE_FUNC TensorEvaluator(const XprType& op, const Device& device) : - m_leftImpl(op.lhsExpression(), device), - m_rightImpl(op.rhsExpression(), device) + EIGEN_DEVICE_FUNC TensorEvaluator(const XprType &op, const Device &device) + : m_leftImpl(op.lhsExpression(), device), m_rightImpl(op.rhsExpression(), device) { - EIGEN_STATIC_ASSERT((static_cast(TensorEvaluator::Layout) == static_cast(TensorEvaluator::Layout)), YOU_MADE_A_PROGRAMMING_MISTAKE); + EIGEN_STATIC_ASSERT((static_cast(TensorEvaluator::Layout) + == static_cast(TensorEvaluator::Layout)), + YOU_MADE_A_PROGRAMMING_MISTAKE); } - EIGEN_DEVICE_FUNC const Dimensions& dimensions() const + EIGEN_DEVICE_FUNC const Dimensions &dimensions() const { // The dimensions of the lhs and the rhs tensors should be equal to prevent // overflows and ensure the result is fully initialized. @@ -118,7 +117,8 @@ struct TensorEvaluator, Device> return m_rightImpl.dimensions(); } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE bool evalSubExprsIfNeeded(Scalar*) { + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE bool evalSubExprsIfNeeded(Scalar *) + { eigen_assert(dimensions_match(m_leftImpl.dimensions(), m_rightImpl.dimensions())); m_leftImpl.evalSubExprsIfNeeded(NULL); // If the lhs provides raw access to its storage area (i.e. if m_leftImpl.data() returns a non @@ -127,55 +127,51 @@ struct TensorEvaluator, Device> // by the rhs to the lhs. return m_rightImpl.evalSubExprsIfNeeded(m_leftImpl.data()); } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void cleanup() { + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void cleanup() + { m_leftImpl.cleanup(); m_rightImpl.cleanup(); } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void evalScalar(Index i) { - m_leftImpl.coeffRef(i) = m_rightImpl.coeff(i); - } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void evalPacket(Index i) { + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void evalScalar(Index i) { m_leftImpl.coeffRef(i) = m_rightImpl.coeff(i); } + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void evalPacket(Index i) + { const int LhsStoreMode = TensorEvaluator::IsAligned ? Aligned : Unaligned; const int RhsLoadMode = TensorEvaluator::IsAligned ? Aligned : Unaligned; m_leftImpl.template writePacket(i, m_rightImpl.template packet(i)); } - EIGEN_DEVICE_FUNC CoeffReturnType coeff(Index index) const - { - return m_leftImpl.coeff(index); - } - template - EIGEN_DEVICE_FUNC PacketReturnType packet(Index index) const + EIGEN_DEVICE_FUNC CoeffReturnType coeff(Index index) const { return m_leftImpl.coeff(index); } + template EIGEN_DEVICE_FUNC PacketReturnType packet(Index index) const { return m_leftImpl.template packet(index); } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TensorOpCost - costPerCoeff(bool vectorized) const { + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TensorOpCost costPerCoeff(bool vectorized) const + { // We assume that evalPacket or evalScalar is called to perform the // assignment and account for the cost of the write here, but reduce left // cost by one load because we are using m_leftImpl.coeffRef. TensorOpCost left = m_leftImpl.costPerCoeff(vectorized); - return m_rightImpl.costPerCoeff(vectorized) + - TensorOpCost( - numext::maxi(0.0, left.bytes_loaded() - sizeof(CoeffReturnType)), - left.bytes_stored(), left.compute_cycles()) + - TensorOpCost(0, sizeof(CoeffReturnType), 0, vectorized, PacketSize); + return m_rightImpl.costPerCoeff(vectorized) + + TensorOpCost(numext::maxi(0.0, left.bytes_loaded() - sizeof(CoeffReturnType)), + left.bytes_stored(), + left.compute_cycles()) + + TensorOpCost(0, sizeof(CoeffReturnType), 0, vectorized, PacketSize); } /// required by sycl in order to extract the accessor - const TensorEvaluator& left_impl() const { return m_leftImpl; } + const TensorEvaluator &left_impl() const { return m_leftImpl; } /// required by sycl in order to extract the accessor - const TensorEvaluator& right_impl() const { return m_rightImpl; } + const TensorEvaluator &right_impl() const { return m_rightImpl; } - EIGEN_DEVICE_FUNC CoeffReturnType* data() const { return m_leftImpl.data(); } + EIGEN_DEVICE_FUNC CoeffReturnType *data() const { return m_leftImpl.data(); } - private: +private: TensorEvaluator m_leftImpl; TensorEvaluator m_rightImpl; }; -} +}// namespace Eigen -#endif // EIGEN_CXX11_TENSOR_TENSOR_ASSIGN_H +#endif// EIGEN_CXX11_TENSOR_TENSOR_ASSIGN_H diff --git a/filmulator-gui/core/nlmeans/eigen/unsupported/Eigen/CXX11/src/Tensor/TensorBroadcasting.h b/filmulator-gui/core/nlmeans/eigen/unsupported/Eigen/CXX11/src/Tensor/TensorBroadcasting.h index 4cfe300e..3d8d08ae 100644 --- a/filmulator-gui/core/nlmeans/eigen/unsupported/Eigen/CXX11/src/Tensor/TensorBroadcasting.h +++ b/filmulator-gui/core/nlmeans/eigen/unsupported/Eigen/CXX11/src/Tensor/TensorBroadcasting.h @@ -13,61 +13,61 @@ namespace Eigen { /** \class TensorBroadcasting - * \ingroup CXX11_Tensor_Module - * - * \brief Tensor broadcasting class. - * - * - */ + * \ingroup CXX11_Tensor_Module + * + * \brief Tensor broadcasting class. + * + * + */ namespace internal { -template -struct traits > : public traits -{ - typedef typename XprType::Scalar Scalar; - typedef traits XprTraits; - typedef typename XprTraits::StorageKind StorageKind; - typedef typename XprTraits::Index Index; - typedef typename XprType::Nested Nested; - typedef typename remove_reference::type _Nested; - static const int NumDimensions = XprTraits::NumDimensions; - static const int Layout = XprTraits::Layout; -}; + template + struct traits> : public traits + { + typedef typename XprType::Scalar Scalar; + typedef traits XprTraits; + typedef typename XprTraits::StorageKind StorageKind; + typedef typename XprTraits::Index Index; + typedef typename XprType::Nested Nested; + typedef typename remove_reference::type _Nested; + static const int NumDimensions = XprTraits::NumDimensions; + static const int Layout = XprTraits::Layout; + }; -template -struct eval, Eigen::Dense> -{ - typedef const TensorBroadcastingOp& type; -}; + template struct eval, Eigen::Dense> + { + typedef const TensorBroadcastingOp &type; + }; -template -struct nested, 1, typename eval >::type> -{ - typedef TensorBroadcastingOp type; -}; + template + struct nested, + 1, + typename eval>::type> + { + typedef TensorBroadcastingOp type; + }; -template -struct is_input_scalar { - static const bool value = false; -}; -template <> -struct is_input_scalar > { - static const bool value = true; -}; + template struct is_input_scalar + { + static const bool value = false; + }; + template<> struct is_input_scalar> + { + static const bool value = true; + }; #ifndef EIGEN_EMULATE_CXX11_META_H -template -struct is_input_scalar > { - static const bool value = (Sizes::total_size == 1); -}; + template struct is_input_scalar> + { + static const bool value = (Sizes::total_size == 1); + }; #endif -} // end namespace internal - +}// end namespace internal template class TensorBroadcastingOp : public TensorBase, ReadOnlyAccessors> { - public: +public: typedef typename Eigen::internal::traits::Scalar Scalar; typedef typename Eigen::NumTraits::Real RealScalar; typedef typename XprType::CoeffReturnType CoeffReturnType; @@ -75,19 +75,19 @@ class TensorBroadcastingOp : public TensorBase::StorageKind StorageKind; typedef typename Eigen::internal::traits::Index Index; - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TensorBroadcastingOp(const XprType& expr, const Broadcast& broadcast) - : m_xpr(expr), m_broadcast(broadcast) {} + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TensorBroadcastingOp(const XprType &expr, const Broadcast &broadcast) + : m_xpr(expr), m_broadcast(broadcast) + {} - EIGEN_DEVICE_FUNC - const Broadcast& broadcast() const { return m_broadcast; } + EIGEN_DEVICE_FUNC + const Broadcast &broadcast() const { return m_broadcast; } - EIGEN_DEVICE_FUNC - const typename internal::remove_all::type& - expression() const { return m_xpr; } + EIGEN_DEVICE_FUNC + const typename internal::remove_all::type &expression() const { return m_xpr; } - protected: - typename XprType::Nested m_xpr; - const Broadcast m_broadcast; +protected: + typename XprType::Nested m_xpr; + const Broadcast m_broadcast; }; @@ -112,15 +112,15 @@ struct TensorEvaluator, Device> RawAccess = false }; - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TensorEvaluator(const XprType& op, const Device& device) - : m_broadcast(op.broadcast()),m_impl(op.expression(), device) + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TensorEvaluator(const XprType &op, const Device &device) + : m_broadcast(op.broadcast()), m_impl(op.expression(), device) { // The broadcasting op doesn't change the rank of the tensor. One can't broadcast a scalar // and store the result in a scalar. Instead one should reshape the scalar into a a N-D // tensor with N >= 1 of 1 element first and then broadcast. EIGEN_STATIC_ASSERT((NumDims > 0), YOU_MADE_A_PROGRAMMING_MISTAKE); - const InputDimensions& input_dims = m_impl.dimensions(); - const Broadcast& broadcast = op.broadcast(); + const InputDimensions &input_dims = m_impl.dimensions(); + const Broadcast &broadcast = op.broadcast(); for (int i = 0; i < NumDims; ++i) { eigen_assert(input_dims[i] > 0); m_dimensions[i] = input_dims[i] * broadcast[i]; @@ -130,29 +130,28 @@ struct TensorEvaluator, Device> m_inputStrides[0] = 1; m_outputStrides[0] = 1; for (int i = 1; i < NumDims; ++i) { - m_inputStrides[i] = m_inputStrides[i-1] * input_dims[i-1]; - m_outputStrides[i] = m_outputStrides[i-1] * m_dimensions[i-1]; + m_inputStrides[i] = m_inputStrides[i - 1] * input_dims[i - 1]; + m_outputStrides[i] = m_outputStrides[i - 1] * m_dimensions[i - 1]; } } else { - m_inputStrides[NumDims-1] = 1; - m_outputStrides[NumDims-1] = 1; - for (int i = NumDims-2; i >= 0; --i) { - m_inputStrides[i] = m_inputStrides[i+1] * input_dims[i+1]; - m_outputStrides[i] = m_outputStrides[i+1] * m_dimensions[i+1]; + m_inputStrides[NumDims - 1] = 1; + m_outputStrides[NumDims - 1] = 1; + for (int i = NumDims - 2; i >= 0; --i) { + m_inputStrides[i] = m_inputStrides[i + 1] * input_dims[i + 1]; + m_outputStrides[i] = m_outputStrides[i + 1] * m_dimensions[i + 1]; } } } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const Dimensions& dimensions() const { return m_dimensions; } + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const Dimensions &dimensions() const { return m_dimensions; } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE bool evalSubExprsIfNeeded(Scalar* /*data*/) { + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE bool evalSubExprsIfNeeded(Scalar * /*data*/) + { m_impl.evalSubExprsIfNeeded(NULL); return true; } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void cleanup() { - m_impl.cleanup(); - } + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void cleanup() { m_impl.cleanup(); } EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE CoeffReturnType coeff(Index index) const { @@ -215,21 +214,20 @@ struct TensorEvaluator, Device> } index -= idx * m_outputStrides[i]; } - if (internal::index_statically_eq(NumDims-1, 1)) { - eigen_assert(index < m_impl.dimensions()[NumDims-1]); + if (internal::index_statically_eq(NumDims - 1, 1)) { + eigen_assert(index < m_impl.dimensions()[NumDims - 1]); inputIndex += index; } else { - if (internal::index_statically_eq(NumDims-1, 1)) { - eigen_assert(index % m_impl.dimensions()[NumDims-1] == 0); + if (internal::index_statically_eq(NumDims - 1, 1)) { + eigen_assert(index % m_impl.dimensions()[NumDims - 1] == 0); } else { - inputIndex += (index % m_impl.dimensions()[NumDims-1]); + inputIndex += (index % m_impl.dimensions()[NumDims - 1]); } } return m_impl.coeff(inputIndex); } - template - EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE PacketReturnType packet(Index index) const + template EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE PacketReturnType packet(Index index) const { if (internal::is_input_scalar::type>::value) { return internal::pset1(m_impl.coeff(0)); @@ -244,11 +242,10 @@ struct TensorEvaluator, Device> // Ignore the LoadMode and always use unaligned loads since we can't guarantee // the alignment at compile time. - template - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE PacketReturnType packetColMajor(Index index) const + template EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE PacketReturnType packetColMajor(Index index) const { EIGEN_STATIC_ASSERT((PacketSize > 1), YOU_MADE_A_PROGRAMMING_MISTAKE) - eigen_assert(index+PacketSize-1 < dimensions().TotalSize()); + eigen_assert(index + PacketSize - 1 < dimensions().TotalSize()); const Index originalIndex = index; @@ -288,19 +285,16 @@ struct TensorEvaluator, Device> } else { EIGEN_ALIGN_MAX typename internal::remove_const::type values[PacketSize]; values[0] = m_impl.coeff(inputIndex); - for (int i = 1; i < PacketSize; ++i) { - values[i] = coeffColMajor(originalIndex+i); - } + for (int i = 1; i < PacketSize; ++i) { values[i] = coeffColMajor(originalIndex + i); } PacketReturnType rslt = internal::pload(values); return rslt; } } - template - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE PacketReturnType packetRowMajor(Index index) const + template EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE PacketReturnType packetRowMajor(Index index) const { EIGEN_STATIC_ASSERT((PacketSize > 1), YOU_MADE_A_PROGRAMMING_MISTAKE) - eigen_assert(index+PacketSize-1 < dimensions().TotalSize()); + eigen_assert(index + PacketSize - 1 < dimensions().TotalSize()); const Index originalIndex = index; @@ -320,65 +314,59 @@ struct TensorEvaluator, Device> index -= idx * m_outputStrides[i]; } Index innermostLoc; - if (internal::index_statically_eq(NumDims-1, 1)) { - eigen_assert(index < m_impl.dimensions()[NumDims-1]); + if (internal::index_statically_eq(NumDims - 1, 1)) { + eigen_assert(index < m_impl.dimensions()[NumDims - 1]); innermostLoc = index; } else { - if (internal::index_statically_eq(NumDims-1, 1)) { - eigen_assert(index % m_impl.dimensions()[NumDims-1] == 0); + if (internal::index_statically_eq(NumDims - 1, 1)) { + eigen_assert(index % m_impl.dimensions()[NumDims - 1] == 0); innermostLoc = 0; } else { - innermostLoc = index % m_impl.dimensions()[NumDims-1]; + innermostLoc = index % m_impl.dimensions()[NumDims - 1]; } } inputIndex += innermostLoc; // Todo: this could be extended to the second dimension if we're not // broadcasting alongside the first dimension, and so on. - if (innermostLoc + PacketSize <= m_impl.dimensions()[NumDims-1]) { + if (innermostLoc + PacketSize <= m_impl.dimensions()[NumDims - 1]) { return m_impl.template packet(inputIndex); } else { EIGEN_ALIGN_MAX typename internal::remove_const::type values[PacketSize]; values[0] = m_impl.coeff(inputIndex); - for (int i = 1; i < PacketSize; ++i) { - values[i] = coeffRowMajor(originalIndex+i); - } + for (int i = 1; i < PacketSize; ++i) { values[i] = coeffRowMajor(originalIndex + i); } PacketReturnType rslt = internal::pload(values); return rslt; } } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TensorOpCost - costPerCoeff(bool vectorized) const { + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TensorOpCost costPerCoeff(bool vectorized) const + { double compute_cost = TensorOpCost::AddCost(); if (NumDims > 0) { for (int i = NumDims - 1; i > 0; --i) { compute_cost += TensorOpCost::DivCost(); if (internal::index_statically_eq(i, 1)) { - compute_cost += - TensorOpCost::MulCost() + TensorOpCost::AddCost(); + compute_cost += TensorOpCost::MulCost() + TensorOpCost::AddCost(); } else { if (!internal::index_statically_eq(i, 1)) { - compute_cost += TensorOpCost::MulCost() + - TensorOpCost::ModCost() + - TensorOpCost::AddCost(); + compute_cost += + TensorOpCost::MulCost() + TensorOpCost::ModCost() + TensorOpCost::AddCost(); } } - compute_cost += - TensorOpCost::MulCost() + TensorOpCost::AddCost(); + compute_cost += TensorOpCost::MulCost() + TensorOpCost::AddCost(); } } - return m_impl.costPerCoeff(vectorized) + - TensorOpCost(0, 0, compute_cost, vectorized, PacketSize); + return m_impl.costPerCoeff(vectorized) + TensorOpCost(0, 0, compute_cost, vectorized, PacketSize); } - EIGEN_DEVICE_FUNC Scalar* data() const { return NULL; } + EIGEN_DEVICE_FUNC Scalar *data() const { return NULL; } - const TensorEvaluator& impl() const { return m_impl; } + const TensorEvaluator &impl() const { return m_impl; } Broadcast functor() const { return m_broadcast; } - protected: +protected: const Broadcast m_broadcast; Dimensions m_dimensions; array m_outputStrides; @@ -387,6 +375,6 @@ struct TensorEvaluator, Device> }; -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_CXX11_TENSOR_TENSOR_BROADCASTING_H +#endif// EIGEN_CXX11_TENSOR_TENSOR_BROADCASTING_H diff --git a/filmulator-gui/core/nlmeans/eigen/unsupported/Eigen/CXX11/src/Tensor/TensorChipping.h b/filmulator-gui/core/nlmeans/eigen/unsupported/Eigen/CXX11/src/Tensor/TensorChipping.h index 1ba7ef17..c0aa1e08 100644 --- a/filmulator-gui/core/nlmeans/eigen/unsupported/Eigen/CXX11/src/Tensor/TensorChipping.h +++ b/filmulator-gui/core/nlmeans/eigen/unsupported/Eigen/CXX11/src/Tensor/TensorChipping.h @@ -13,71 +13,59 @@ namespace Eigen { /** \class TensorKChippingReshaping - * \ingroup CXX11_Tensor_Module - * - * \brief A chip is a thin slice, corresponding to a column or a row in a 2-d tensor. - * - * - */ + * \ingroup CXX11_Tensor_Module + * + * \brief A chip is a thin slice, corresponding to a column or a row in a 2-d tensor. + * + * + */ namespace internal { -template -struct traits > : public traits -{ - typedef typename XprType::Scalar Scalar; - typedef traits XprTraits; - typedef typename XprTraits::StorageKind StorageKind; - typedef typename XprTraits::Index Index; - typedef typename XprType::Nested Nested; - typedef typename remove_reference::type _Nested; - static const int NumDimensions = XprTraits::NumDimensions - 1; - static const int Layout = XprTraits::Layout; -}; + template struct traits> : public traits + { + typedef typename XprType::Scalar Scalar; + typedef traits XprTraits; + typedef typename XprTraits::StorageKind StorageKind; + typedef typename XprTraits::Index Index; + typedef typename XprType::Nested Nested; + typedef typename remove_reference::type _Nested; + static const int NumDimensions = XprTraits::NumDimensions - 1; + static const int Layout = XprTraits::Layout; + }; -template -struct eval, Eigen::Dense> -{ - typedef const TensorChippingOp& type; -}; + template struct eval, Eigen::Dense> + { + typedef const TensorChippingOp &type; + }; -template -struct nested, 1, typename eval >::type> -{ - typedef TensorChippingOp type; -}; + template + struct nested, 1, typename eval>::type> + { + typedef TensorChippingOp type; + }; -template -struct DimensionId -{ - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE DimensionId(DenseIndex dim) { - eigen_assert(dim == DimId); - } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE DenseIndex actualDim() const { - return DimId; - } -}; -template <> -struct DimensionId -{ - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE DimensionId(DenseIndex dim) : actual_dim(dim) { - eigen_assert(dim >= 0); - } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE DenseIndex actualDim() const { - return actual_dim; - } - private: - const DenseIndex actual_dim; -}; + template struct DimensionId + { + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE DimensionId(DenseIndex dim) { eigen_assert(dim == DimId); } + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE DenseIndex actualDim() const { return DimId; } + }; + template<> struct DimensionId + { + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE DimensionId(DenseIndex dim) : actual_dim(dim) { eigen_assert(dim >= 0); } + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE DenseIndex actualDim() const { return actual_dim; } + private: + const DenseIndex actual_dim; + }; -} // end namespace internal +}// end namespace internal template -class TensorChippingOp : public TensorBase > +class TensorChippingOp : public TensorBase> { - public: +public: typedef typename Eigen::internal::traits::Scalar Scalar; typedef typename Eigen::NumTraits::Real RealScalar; typedef typename XprType::CoeffReturnType CoeffReturnType; @@ -85,9 +73,9 @@ class TensorChippingOp : public TensorBase > typedef typename Eigen::internal::traits::StorageKind StorageKind; typedef typename Eigen::internal::traits::Index Index; - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TensorChippingOp(const XprType& expr, const Index offset, const Index dim) - : m_xpr(expr), m_offset(offset), m_dim(dim) { - } + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TensorChippingOp(const XprType &expr, const Index offset, const Index dim) + : m_xpr(expr), m_offset(offset), m_dim(dim) + {} EIGEN_DEVICE_FUNC const Index offset() const { return m_offset; } @@ -95,11 +83,10 @@ class TensorChippingOp : public TensorBase > const Index dim() const { return m_dim.actualDim(); } EIGEN_DEVICE_FUNC - const typename internal::remove_all::type& - expression() const { return m_xpr; } + const typename internal::remove_all::type &expression() const { return m_xpr; } EIGEN_DEVICE_FUNC - EIGEN_STRONG_INLINE TensorChippingOp& operator = (const TensorChippingOp& other) + EIGEN_STRONG_INLINE TensorChippingOp &operator=(const TensorChippingOp &other) { typedef TensorAssignOp Assign; Assign assign(*this, other); @@ -108,8 +95,7 @@ class TensorChippingOp : public TensorBase > } template - EIGEN_DEVICE_FUNC - EIGEN_STRONG_INLINE TensorChippingOp& operator = (const OtherDerived& other) + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TensorChippingOp &operator=(const OtherDerived &other) { typedef TensorAssignOp Assign; Assign assign(*this, other); @@ -117,10 +103,10 @@ class TensorChippingOp : public TensorBase > return *this; } - protected: - typename XprType::Nested m_xpr; - const Index m_offset; - const internal::DimensionId m_dim; +protected: + typename XprType::Nested m_xpr; + const Index m_offset; + const internal::DimensionId m_dim; }; @@ -130,7 +116,7 @@ struct TensorEvaluator, Device> { typedef TensorChippingOp XprType; static const int NumInputDims = internal::array_size::Dimensions>::value; - static const int NumDims = NumInputDims-1; + static const int NumDims = NumInputDims - 1; typedef typename XprType::Index Index; typedef DSizes Dimensions; typedef typename XprType::Scalar Scalar; @@ -145,17 +131,17 @@ struct TensorEvaluator, Device> IsAligned = false, PacketAccess = TensorEvaluator::PacketAccess, Layout = TensorEvaluator::Layout, - CoordAccess = false, // to be implemented + CoordAccess = false,// to be implemented RawAccess = false }; - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TensorEvaluator(const XprType& op, const Device& device) - : m_impl(op.expression(), device), m_dim(op.dim()), m_device(device) + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TensorEvaluator(const XprType &op, const Device &device) + : m_impl(op.expression(), device), m_dim(op.dim()), m_device(device) { EIGEN_STATIC_ASSERT((NumInputDims >= 1), YOU_MADE_A_PROGRAMMING_MISTAKE); eigen_assert(NumInputDims > m_dim.actualDim()); - const typename TensorEvaluator::Dimensions& input_dims = m_impl.dimensions(); + const typename TensorEvaluator::Dimensions &input_dims = m_impl.dimensions(); eigen_assert(op.offset() < input_dims[m_dim.actualDim()]); int j = 0; @@ -174,7 +160,7 @@ struct TensorEvaluator, Device> m_inputStride *= input_dims[i]; } } else { - for (int i = NumInputDims-1; i > m_dim.actualDim(); --i) { + for (int i = NumInputDims - 1; i > m_dim.actualDim(); --i) { m_stride *= input_dims[i]; m_inputStride *= input_dims[i]; } @@ -183,30 +169,28 @@ struct TensorEvaluator, Device> m_inputOffset = m_stride * op.offset(); } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const Dimensions& dimensions() const { return m_dimensions; } + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const Dimensions &dimensions() const { return m_dimensions; } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE bool evalSubExprsIfNeeded(Scalar* /*data*/) { + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE bool evalSubExprsIfNeeded(Scalar * /*data*/) + { m_impl.evalSubExprsIfNeeded(NULL); return true; } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void cleanup() { - m_impl.cleanup(); - } + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void cleanup() { m_impl.cleanup(); } EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE CoeffReturnType coeff(Index index) const { return m_impl.coeff(srcCoeff(index)); } - template - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE PacketReturnType packet(Index index) const + template EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE PacketReturnType packet(Index index) const { EIGEN_STATIC_ASSERT((PacketSize > 1), YOU_MADE_A_PROGRAMMING_MISTAKE) - eigen_assert(index+PacketSize-1 < dimensions().TotalSize()); + eigen_assert(index + PacketSize - 1 < dimensions().TotalSize()); - if ((static_cast(Layout) == static_cast(ColMajor) && m_dim.actualDim() == 0) || - (static_cast(Layout) == static_cast(RowMajor) && m_dim.actualDim() == NumInputDims-1)) { + if ((static_cast(Layout) == static_cast(ColMajor) && m_dim.actualDim() == 0) + || (static_cast(Layout) == static_cast(RowMajor) && m_dim.actualDim() == NumInputDims - 1)) { // m_stride is equal to 1, so let's avoid the integer division. eigen_assert(m_stride == 1); Index inputIndex = index * m_inputStride + m_inputOffset; @@ -217,8 +201,8 @@ struct TensorEvaluator, Device> } PacketReturnType rslt = internal::pload(values); return rslt; - } else if ((static_cast(Layout) == static_cast(ColMajor) && m_dim.actualDim() == NumInputDims - 1) || - (static_cast(Layout) == static_cast(RowMajor) && m_dim.actualDim() == 0)) { + } else if ((static_cast(Layout) == static_cast(ColMajor) && m_dim.actualDim() == NumInputDims - 1) + || (static_cast(Layout) == static_cast(RowMajor) && m_dim.actualDim() == 0)) { // m_stride is aways greater than index, so let's avoid the integer division. eigen_assert(m_stride > index); return m_impl.template packet(index + m_inputOffset); @@ -241,50 +225,45 @@ struct TensorEvaluator, Device> } } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TensorOpCost - costPerCoeff(bool vectorized) const { + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TensorOpCost costPerCoeff(bool vectorized) const + { double cost = 0; - if ((static_cast(Layout) == static_cast(ColMajor) && - m_dim.actualDim() == 0) || - (static_cast(Layout) == static_cast(RowMajor) && - m_dim.actualDim() == NumInputDims - 1)) { + if ((static_cast(Layout) == static_cast(ColMajor) && m_dim.actualDim() == 0) + || (static_cast(Layout) == static_cast(RowMajor) && m_dim.actualDim() == NumInputDims - 1)) { cost += TensorOpCost::MulCost() + TensorOpCost::AddCost(); - } else if ((static_cast(Layout) == static_cast(ColMajor) && - m_dim.actualDim() == NumInputDims - 1) || - (static_cast(Layout) == static_cast(RowMajor) && - m_dim.actualDim() == 0)) { + } else if ((static_cast(Layout) == static_cast(ColMajor) && m_dim.actualDim() == NumInputDims - 1) + || (static_cast(Layout) == static_cast(RowMajor) && m_dim.actualDim() == 0)) { cost += TensorOpCost::AddCost(); } else { - cost += 3 * TensorOpCost::MulCost() + TensorOpCost::DivCost() + - 3 * TensorOpCost::AddCost(); + cost += 3 * TensorOpCost::MulCost() + TensorOpCost::DivCost() + 3 * TensorOpCost::AddCost(); } - return m_impl.costPerCoeff(vectorized) + - TensorOpCost(0, 0, cost, vectorized, PacketSize); + return m_impl.costPerCoeff(vectorized) + TensorOpCost(0, 0, cost, vectorized, PacketSize); } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE CoeffReturnType* data() const { - CoeffReturnType* result = const_cast(m_impl.data()); - if (((static_cast(Layout) == static_cast(ColMajor) && m_dim.actualDim() == NumDims) || - (static_cast(Layout) == static_cast(RowMajor) && m_dim.actualDim() == 0)) && - result) { + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE CoeffReturnType *data() const + { + CoeffReturnType *result = const_cast(m_impl.data()); + if (((static_cast(Layout) == static_cast(ColMajor) && m_dim.actualDim() == NumDims) + || (static_cast(Layout) == static_cast(RowMajor) && m_dim.actualDim() == 0)) + && result) { return result + m_inputOffset; } else { return NULL; } } - protected: +protected: EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Index srcCoeff(Index index) const { Index inputIndex; - if ((static_cast(Layout) == static_cast(ColMajor) && m_dim.actualDim() == 0) || - (static_cast(Layout) == static_cast(RowMajor) && m_dim.actualDim() == NumInputDims-1)) { + if ((static_cast(Layout) == static_cast(ColMajor) && m_dim.actualDim() == 0) + || (static_cast(Layout) == static_cast(RowMajor) && m_dim.actualDim() == NumInputDims - 1)) { // m_stride is equal to 1, so let's avoid the integer division. eigen_assert(m_stride == 1); inputIndex = index * m_inputStride + m_inputOffset; - } else if ((static_cast(Layout) == static_cast(ColMajor) && m_dim.actualDim() == NumInputDims-1) || - (static_cast(Layout) == static_cast(RowMajor) && m_dim.actualDim() == 0)) { + } else if ((static_cast(Layout) == static_cast(ColMajor) && m_dim.actualDim() == NumInputDims - 1) + || (static_cast(Layout) == static_cast(RowMajor) && m_dim.actualDim() == 0)) { // m_stride is aways greater than index, so let's avoid the integer division. eigen_assert(m_stride > index); inputIndex = index + m_inputOffset; @@ -303,7 +282,7 @@ struct TensorEvaluator, Device> Index m_inputStride; TensorEvaluator m_impl; const internal::DimensionId m_dim; - const Device& m_device; + const Device &m_device; }; @@ -315,7 +294,7 @@ struct TensorEvaluator, Device> typedef TensorEvaluator, Device> Base; typedef TensorChippingOp XprType; static const int NumInputDims = internal::array_size::Dimensions>::value; - static const int NumDims = NumInputDims-1; + static const int NumDims = NumInputDims - 1; typedef typename XprType::Index Index; typedef DSizes Dimensions; typedef typename XprType::Scalar Scalar; @@ -323,28 +302,22 @@ struct TensorEvaluator, Device> typedef typename PacketType::type PacketReturnType; static const int PacketSize = internal::unpacket_traits::size; - enum { - IsAligned = false, - PacketAccess = TensorEvaluator::PacketAccess, - RawAccess = false - }; + enum { IsAligned = false, PacketAccess = TensorEvaluator::PacketAccess, RawAccess = false }; - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TensorEvaluator(const XprType& op, const Device& device) - : Base(op, device) - { } + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TensorEvaluator(const XprType &op, const Device &device) : Base(op, device) {} - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE CoeffReturnType& coeffRef(Index index) + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE CoeffReturnType &coeffRef(Index index) { return this->m_impl.coeffRef(this->srcCoeff(index)); } - template EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE - void writePacket(Index index, const PacketReturnType& x) + template EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void writePacket(Index index, const PacketReturnType &x) { EIGEN_STATIC_ASSERT((PacketSize > 1), YOU_MADE_A_PROGRAMMING_MISTAKE) - if ((static_cast(this->Layout) == static_cast(ColMajor) && this->m_dim.actualDim() == 0) || - (static_cast(this->Layout) == static_cast(RowMajor) && this->m_dim.actualDim() == NumInputDims-1)) { + if ((static_cast(this->Layout) == static_cast(ColMajor) && this->m_dim.actualDim() == 0) + || (static_cast(this->Layout) == static_cast(RowMajor) + && this->m_dim.actualDim() == NumInputDims - 1)) { // m_stride is equal to 1, so let's avoid the integer division. eigen_assert(this->m_stride == 1); EIGEN_ALIGN_MAX typename internal::remove_const::type values[PacketSize]; @@ -354,8 +327,9 @@ struct TensorEvaluator, Device> this->m_impl.coeffRef(inputIndex) = values[i]; inputIndex += this->m_inputStride; } - } else if ((static_cast(this->Layout) == static_cast(ColMajor) && this->m_dim.actualDim() == NumInputDims-1) || - (static_cast(this->Layout) == static_cast(RowMajor) && this->m_dim.actualDim() == 0)) { + } else if ((static_cast(this->Layout) == static_cast(ColMajor) + && this->m_dim.actualDim() == NumInputDims - 1) + || (static_cast(this->Layout) == static_cast(RowMajor) && this->m_dim.actualDim() == 0)) { // m_stride is aways greater than index, so let's avoid the integer division. eigen_assert(this->m_stride > index); this->m_impl.template writePacket(index + this->m_inputOffset, x); @@ -379,6 +353,6 @@ struct TensorEvaluator, Device> }; -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_CXX11_TENSOR_TENSOR_CHIPPING_H +#endif// EIGEN_CXX11_TENSOR_TENSOR_CHIPPING_H diff --git a/filmulator-gui/core/nlmeans/eigen/unsupported/Eigen/CXX11/src/Tensor/TensorConcatenation.h b/filmulator-gui/core/nlmeans/eigen/unsupported/Eigen/CXX11/src/Tensor/TensorConcatenation.h index 59bf90d9..1666dbb8 100644 --- a/filmulator-gui/core/nlmeans/eigen/unsupported/Eigen/CXX11/src/Tensor/TensorConcatenation.h +++ b/filmulator-gui/core/nlmeans/eigen/unsupported/Eigen/CXX11/src/Tensor/TensorConcatenation.h @@ -13,95 +13,94 @@ namespace Eigen { /** \class TensorConcatenationOp - * \ingroup CXX11_Tensor_Module - * - * \brief Tensor concatenation class. - * - * - */ + * \ingroup CXX11_Tensor_Module + * + * \brief Tensor concatenation class. + * + * + */ namespace internal { -template -struct traits > -{ - // Type promotion to handle the case where the types of the lhs and the rhs are different. - typedef typename promote_storage_type::ret Scalar; - typedef typename promote_storage_type::StorageKind, - typename traits::StorageKind>::ret StorageKind; - typedef typename promote_index_type::Index, - typename traits::Index>::type Index; - typedef typename LhsXprType::Nested LhsNested; - typedef typename RhsXprType::Nested RhsNested; - typedef typename remove_reference::type _LhsNested; - typedef typename remove_reference::type _RhsNested; - static const int NumDimensions = traits::NumDimensions; - static const int Layout = traits::Layout; - enum { Flags = 0 }; -}; + template + struct traits> + { + // Type promotion to handle the case where the types of the lhs and the rhs are different. + typedef typename promote_storage_type::ret Scalar; + typedef typename promote_storage_type::StorageKind, + typename traits::StorageKind>::ret StorageKind; + typedef + typename promote_index_type::Index, typename traits::Index>::type Index; + typedef typename LhsXprType::Nested LhsNested; + typedef typename RhsXprType::Nested RhsNested; + typedef typename remove_reference::type _LhsNested; + typedef typename remove_reference::type _RhsNested; + static const int NumDimensions = traits::NumDimensions; + static const int Layout = traits::Layout; + enum { Flags = 0 }; + }; -template -struct eval, Eigen::Dense> -{ - typedef const TensorConcatenationOp& type; -}; + template + struct eval, Eigen::Dense> + { + typedef const TensorConcatenationOp &type; + }; -template -struct nested, 1, typename eval >::type> -{ - typedef TensorConcatenationOp type; -}; + template + struct nested, + 1, + typename eval>::type> + { + typedef TensorConcatenationOp type; + }; -} // end namespace internal +}// end namespace internal template class TensorConcatenationOp : public TensorBase, WriteAccessors> { - public: - typedef typename internal::traits::Scalar Scalar; - typedef typename internal::traits::StorageKind StorageKind; - typedef typename internal::traits::Index Index; - typedef typename internal::nested::type Nested; - typedef typename internal::promote_storage_type::ret CoeffReturnType; - typedef typename NumTraits::Real RealScalar; - - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TensorConcatenationOp(const LhsXprType& lhs, const RhsXprType& rhs, Axis axis) - : m_lhs_xpr(lhs), m_rhs_xpr(rhs), m_axis(axis) {} - - EIGEN_DEVICE_FUNC - const typename internal::remove_all::type& - lhsExpression() const { return m_lhs_xpr; } - - EIGEN_DEVICE_FUNC - const typename internal::remove_all::type& - rhsExpression() const { return m_rhs_xpr; } - - EIGEN_DEVICE_FUNC const Axis& axis() const { return m_axis; } - - EIGEN_DEVICE_FUNC - EIGEN_STRONG_INLINE TensorConcatenationOp& operator = (const TensorConcatenationOp& other) - { - typedef TensorAssignOp Assign; - Assign assign(*this, other); - internal::TensorExecutor::run(assign, DefaultDevice()); - return *this; - } +public: + typedef typename internal::traits::Scalar Scalar; + typedef typename internal::traits::StorageKind StorageKind; + typedef typename internal::traits::Index Index; + typedef typename internal::nested::type Nested; + typedef typename internal::promote_storage_type::ret CoeffReturnType; + typedef typename NumTraits::Real RealScalar; - template - EIGEN_DEVICE_FUNC - EIGEN_STRONG_INLINE TensorConcatenationOp& operator = (const OtherDerived& other) - { - typedef TensorAssignOp Assign; - Assign assign(*this, other); - internal::TensorExecutor::run(assign, DefaultDevice()); - return *this; - } + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TensorConcatenationOp(const LhsXprType &lhs, const RhsXprType &rhs, Axis axis) + : m_lhs_xpr(lhs), m_rhs_xpr(rhs), m_axis(axis) + {} + + EIGEN_DEVICE_FUNC + const typename internal::remove_all::type &lhsExpression() const { return m_lhs_xpr; } + + EIGEN_DEVICE_FUNC + const typename internal::remove_all::type &rhsExpression() const { return m_rhs_xpr; } + + EIGEN_DEVICE_FUNC const Axis &axis() const { return m_axis; } + + EIGEN_DEVICE_FUNC + EIGEN_STRONG_INLINE TensorConcatenationOp &operator=(const TensorConcatenationOp &other) + { + typedef TensorAssignOp Assign; + Assign assign(*this, other); + internal::TensorExecutor::run(assign, DefaultDevice()); + return *this; + } - protected: - typename LhsXprType::Nested m_lhs_xpr; - typename RhsXprType::Nested m_rhs_xpr; - const Axis m_axis; + template + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TensorConcatenationOp &operator=(const OtherDerived &other) + { + typedef TensorAssignOp Assign; + Assign assign(*this, other); + internal::TensorExecutor::run(assign, DefaultDevice()); + return *this; + } + +protected: + typename LhsXprType::Nested m_lhs_xpr; + typename RhsXprType::Nested m_rhs_xpr; + const Axis m_axis; }; @@ -112,28 +111,33 @@ struct TensorEvaluator XprType; typedef typename XprType::Index Index; static const int NumDims = internal::array_size::Dimensions>::value; - static const int RightNumDims = internal::array_size::Dimensions>::value; + static const int RightNumDims = + internal::array_size::Dimensions>::value; typedef DSizes Dimensions; typedef typename XprType::Scalar Scalar; typedef typename XprType::CoeffReturnType CoeffReturnType; typedef typename PacketType::type PacketReturnType; enum { IsAligned = false, - PacketAccess = TensorEvaluator::PacketAccess & TensorEvaluator::PacketAccess, + PacketAccess = + TensorEvaluator::PacketAccess & TensorEvaluator::PacketAccess, Layout = TensorEvaluator::Layout, RawAccess = false }; - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TensorEvaluator(const XprType& op, const Device& device) + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TensorEvaluator(const XprType &op, const Device &device) : m_leftImpl(op.lhsExpression(), device), m_rightImpl(op.rhsExpression(), device), m_axis(op.axis()) { - EIGEN_STATIC_ASSERT((static_cast(TensorEvaluator::Layout) == static_cast(TensorEvaluator::Layout) || NumDims == 1), YOU_MADE_A_PROGRAMMING_MISTAKE); + EIGEN_STATIC_ASSERT((static_cast(TensorEvaluator::Layout) + == static_cast(TensorEvaluator::Layout) + || NumDims == 1), + YOU_MADE_A_PROGRAMMING_MISTAKE); EIGEN_STATIC_ASSERT((NumDims == RightNumDims), YOU_MADE_A_PROGRAMMING_MISTAKE); EIGEN_STATIC_ASSERT((NumDims > 0), YOU_MADE_A_PROGRAMMING_MISTAKE); eigen_assert(0 <= m_axis && m_axis < NumDims); - const Dimensions& lhs_dims = m_leftImpl.dimensions(); - const Dimensions& rhs_dims = m_rightImpl.dimensions(); + const Dimensions &lhs_dims = m_leftImpl.dimensions(); + const Dimensions &rhs_dims = m_rightImpl.dimensions(); { int i = 0; for (; i < m_axis; ++i) { @@ -141,7 +145,7 @@ struct TensorEvaluator 0); // Now i == m_axis. + eigen_assert(lhs_dims[i] > 0);// Now i == m_axis. eigen_assert(rhs_dims[i] > 0); m_dimensions[i] = lhs_dims[i] + rhs_dims[i]; for (++i; i < NumDims; ++i) { @@ -157,9 +161,9 @@ struct TensorEvaluator= 0; --j) { - m_leftStrides[j] = m_leftStrides[j+1] * lhs_dims[j+1]; - m_rightStrides[j] = m_rightStrides[j+1] * rhs_dims[j+1]; - m_outputStrides[j] = m_outputStrides[j+1] * m_dimensions[j+1]; + m_leftStrides[j] = m_leftStrides[j + 1] * lhs_dims[j + 1]; + m_rightStrides[j] = m_rightStrides[j + 1] * rhs_dims[j + 1]; + m_outputStrides[j] = m_outputStrides[j + 1] * m_dimensions[j + 1]; } } } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const Dimensions& dimensions() const { return m_dimensions; } + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const Dimensions &dimensions() const { return m_dimensions; } // TODO(phli): Add short-circuit memcpy evaluation if underlying data are linear? - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE bool evalSubExprsIfNeeded(Scalar* /*data*/) + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE bool evalSubExprsIfNeeded(Scalar * /*data*/) { m_leftImpl.evalSubExprsIfNeeded(NULL); m_rightImpl.evalSubExprsIfNeeded(NULL); @@ -210,86 +214,72 @@ struct TensorEvaluator(Layout) == static_cast(ColMajor)) { left_index = subs[0]; - for (int i = 1; i < NumDims; ++i) { - left_index += (subs[i] % left_dims[i]) * m_leftStrides[i]; - } + for (int i = 1; i < NumDims; ++i) { left_index += (subs[i] % left_dims[i]) * m_leftStrides[i]; } } else { left_index = subs[NumDims - 1]; - for (int i = NumDims - 2; i >= 0; --i) { - left_index += (subs[i] % left_dims[i]) * m_leftStrides[i]; - } + for (int i = NumDims - 2; i >= 0; --i) { left_index += (subs[i] % left_dims[i]) * m_leftStrides[i]; } } return m_leftImpl.coeff(left_index); } else { subs[m_axis] -= left_dims[m_axis]; - const Dimensions& right_dims = m_rightImpl.dimensions(); + const Dimensions &right_dims = m_rightImpl.dimensions(); Index right_index; if (static_cast(Layout) == static_cast(ColMajor)) { right_index = subs[0]; - for (int i = 1; i < NumDims; ++i) { - right_index += (subs[i] % right_dims[i]) * m_rightStrides[i]; - } + for (int i = 1; i < NumDims; ++i) { right_index += (subs[i] % right_dims[i]) * m_rightStrides[i]; } } else { right_index = subs[NumDims - 1]; - for (int i = NumDims - 2; i >= 0; --i) { - right_index += (subs[i] % right_dims[i]) * m_rightStrides[i]; - } + for (int i = NumDims - 2; i >= 0; --i) { right_index += (subs[i] % right_dims[i]) * m_rightStrides[i]; } } return m_rightImpl.coeff(right_index); } } // TODO(phli): Add a real vectorization. - template - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE PacketReturnType packet(Index index) const + template EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE PacketReturnType packet(Index index) const { const int packetSize = internal::unpacket_traits::size; EIGEN_STATIC_ASSERT((packetSize > 1), YOU_MADE_A_PROGRAMMING_MISTAKE) eigen_assert(index + packetSize - 1 < dimensions().TotalSize()); EIGEN_ALIGN_MAX CoeffReturnType values[packetSize]; - for (int i = 0; i < packetSize; ++i) { - values[i] = coeff(index+i); - } + for (int i = 0; i < packetSize; ++i) { values[i] = coeff(index + i); } PacketReturnType rslt = internal::pload(values); return rslt; } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TensorOpCost - costPerCoeff(bool vectorized) const { - const double compute_cost = NumDims * (2 * TensorOpCost::AddCost() + - 2 * TensorOpCost::MulCost() + - TensorOpCost::DivCost() + - TensorOpCost::ModCost()); + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TensorOpCost costPerCoeff(bool vectorized) const + { + const double compute_cost = NumDims + * (2 * TensorOpCost::AddCost() + 2 * TensorOpCost::MulCost() + + TensorOpCost::DivCost() + TensorOpCost::ModCost()); const double lhs_size = m_leftImpl.dimensions().TotalSize(); const double rhs_size = m_rightImpl.dimensions().TotalSize(); - return (lhs_size / (lhs_size + rhs_size)) * - m_leftImpl.costPerCoeff(vectorized) + - (rhs_size / (lhs_size + rhs_size)) * - m_rightImpl.costPerCoeff(vectorized) + - TensorOpCost(0, 0, compute_cost); + return (lhs_size / (lhs_size + rhs_size)) * m_leftImpl.costPerCoeff(vectorized) + + (rhs_size / (lhs_size + rhs_size)) * m_rightImpl.costPerCoeff(vectorized) + + TensorOpCost(0, 0, compute_cost); } - EIGEN_DEVICE_FUNC Scalar* data() const { return NULL; } + EIGEN_DEVICE_FUNC Scalar *data() const { return NULL; } - protected: - Dimensions m_dimensions; - array m_outputStrides; - array m_leftStrides; - array m_rightStrides; - TensorEvaluator m_leftImpl; - TensorEvaluator m_rightImpl; - const Axis m_axis; +protected: + Dimensions m_dimensions; + array m_outputStrides; + array m_leftStrides; + array m_rightStrides; + TensorEvaluator m_leftImpl; + TensorEvaluator m_rightImpl; + const Axis m_axis; }; // Eval as lvalue template - struct TensorEvaluator, Device> +struct TensorEvaluator, Device> : public TensorEvaluator, Device> { typedef TensorEvaluator, Device> Base; @@ -297,13 +287,13 @@ template::PacketAccess & TensorEvaluator::PacketAccess, + PacketAccess = + TensorEvaluator::PacketAccess & TensorEvaluator::PacketAccess, Layout = TensorEvaluator::Layout, RawAccess = false }; - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TensorEvaluator(XprType& op, const Device& device) - : Base(op, device) + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TensorEvaluator(XprType &op, const Device &device) : Base(op, device) { EIGEN_STATIC_ASSERT((static_cast(Layout) == static_cast(ColMajor)), YOU_MADE_A_PROGRAMMING_MISTAKE); } @@ -313,7 +303,7 @@ template::type PacketReturnType; - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE CoeffReturnType& coeffRef(Index index) + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE CoeffReturnType &coeffRef(Index index) { // Collect dimension-wise indices (subs). array subs; @@ -323,26 +313,21 @@ templatem_leftImpl.dimensions(); + const Dimensions &left_dims = this->m_leftImpl.dimensions(); if (subs[this->m_axis] < left_dims[this->m_axis]) { Index left_index = subs[0]; - for (int i = 1; i < Base::NumDims; ++i) { - left_index += (subs[i] % left_dims[i]) * this->m_leftStrides[i]; - } + for (int i = 1; i < Base::NumDims; ++i) { left_index += (subs[i] % left_dims[i]) * this->m_leftStrides[i]; } return this->m_leftImpl.coeffRef(left_index); } else { subs[this->m_axis] -= left_dims[this->m_axis]; - const Dimensions& right_dims = this->m_rightImpl.dimensions(); + const Dimensions &right_dims = this->m_rightImpl.dimensions(); Index right_index = subs[0]; - for (int i = 1; i < Base::NumDims; ++i) { - right_index += (subs[i] % right_dims[i]) * this->m_rightStrides[i]; - } + for (int i = 1; i < Base::NumDims; ++i) { right_index += (subs[i] % right_dims[i]) * this->m_rightStrides[i]; } return this->m_rightImpl.coeffRef(right_index); } } - template EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE - void writePacket(Index index, const PacketReturnType& x) + template EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void writePacket(Index index, const PacketReturnType &x) { const int packetSize = internal::unpacket_traits::size; EIGEN_STATIC_ASSERT((packetSize > 1), YOU_MADE_A_PROGRAMMING_MISTAKE) @@ -350,12 +335,10 @@ template(values, x); - for (int i = 0; i < packetSize; ++i) { - coeffRef(index+i) = values[i]; - } + for (int i = 0; i < packetSize; ++i) { coeffRef(index + i) = values[i]; } } }; -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_CXX11_TENSOR_TENSOR_CONCATENATION_H +#endif// EIGEN_CXX11_TENSOR_TENSOR_CONCATENATION_H diff --git a/filmulator-gui/core/nlmeans/eigen/unsupported/Eigen/CXX11/src/Tensor/TensorContraction.h b/filmulator-gui/core/nlmeans/eigen/unsupported/Eigen/CXX11/src/Tensor/TensorContraction.h index 20b29e5f..4b46b2d7 100644 --- a/filmulator-gui/core/nlmeans/eigen/unsupported/Eigen/CXX11/src/Tensor/TensorContraction.h +++ b/filmulator-gui/core/nlmeans/eigen/unsupported/Eigen/CXX11/src/Tensor/TensorContraction.h @@ -13,100 +13,102 @@ namespace Eigen { /** \class TensorContraction - * \ingroup CXX11_Tensor_Module - * - * \brief Tensor contraction class. - * - * - */ + * \ingroup CXX11_Tensor_Module + * + * \brief Tensor contraction class. + * + * + */ namespace internal { -template -struct traits > -{ - // Type promotion to handle the case where the types of the lhs and the rhs are different. - typedef typename gebp_traits::type, - typename remove_const::type>::ResScalar Scalar; - - typedef typename promote_storage_type::StorageKind, - typename traits::StorageKind>::ret StorageKind; - typedef typename promote_index_type::Index, - typename traits::Index>::type Index; - typedef typename LhsXprType::Nested LhsNested; - typedef typename RhsXprType::Nested RhsNested; - typedef typename remove_reference::type _LhsNested; - typedef typename remove_reference::type _RhsNested; - - // From NumDims below. - static const int NumDimensions = traits::NumDimensions + traits::NumDimensions - 2 * array_size::value; - static const int Layout = traits::Layout; - - enum { - Flags = 0 + template + struct traits> + { + // Type promotion to handle the case where the types of the lhs and the rhs are different. + typedef typename gebp_traits::type, + typename remove_const::type>::ResScalar Scalar; + + typedef typename promote_storage_type::StorageKind, + typename traits::StorageKind>::ret StorageKind; + typedef + typename promote_index_type::Index, typename traits::Index>::type Index; + typedef typename LhsXprType::Nested LhsNested; + typedef typename RhsXprType::Nested RhsNested; + typedef typename remove_reference::type _LhsNested; + typedef typename remove_reference::type _RhsNested; + + // From NumDims below. + static const int NumDimensions = + traits::NumDimensions + traits::NumDimensions - 2 * array_size::value; + static const int Layout = traits::Layout; + + enum { Flags = 0 }; }; -}; - -template -struct eval, Eigen::Dense> -{ - typedef const TensorContractionOp& type; -}; -template -struct nested, 1, typename eval >::type> -{ - typedef TensorContractionOp type; -}; + template + struct eval, Eigen::Dense> + { + typedef const TensorContractionOp &type; + }; -template -struct traits, Device_> > { - typedef Indices_ Indices; - typedef LeftArgType_ LeftArgType; - typedef RightArgType_ RightArgType; - typedef Device_ Device; + template + struct nested, + 1, + typename eval>::type> + { + typedef TensorContractionOp type; + }; - // From NumDims below. - static const int NumDimensions = traits::NumDimensions + traits::NumDimensions - 2 * array_size::value; -}; + template + struct traits, Device_>> + { + typedef Indices_ Indices; + typedef LeftArgType_ LeftArgType; + typedef RightArgType_ RightArgType; + typedef Device_ Device; + + // From NumDims below. + static const int NumDimensions = + traits::NumDimensions + traits::NumDimensions - 2 * array_size::value; + }; -} // end namespace internal +}// end namespace internal template class TensorContractionOp : public TensorBase, ReadOnlyAccessors> { - public: +public: typedef typename Eigen::internal::traits::Scalar Scalar; typedef typename internal::gebp_traits::ResScalar CoeffReturnType; + typename RhsXprType::CoeffReturnType>::ResScalar CoeffReturnType; typedef typename Eigen::internal::nested::type Nested; typedef typename Eigen::internal::traits::StorageKind StorageKind; typedef typename Eigen::internal::traits::Index Index; - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TensorContractionOp( - const LhsXprType& lhs, const RhsXprType& rhs, const Indices& dims) - : m_lhs_xpr(lhs), m_rhs_xpr(rhs), m_indices(dims) {} + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TensorContractionOp(const LhsXprType &lhs, + const RhsXprType &rhs, + const Indices &dims) + : m_lhs_xpr(lhs), m_rhs_xpr(rhs), m_indices(dims) + {} EIGEN_DEVICE_FUNC - const Indices& indices() const { return m_indices; } + const Indices &indices() const { return m_indices; } /** \returns the nested expressions */ EIGEN_DEVICE_FUNC - const typename internal::remove_all::type& - lhsExpression() const { return m_lhs_xpr; } + const typename internal::remove_all::type &lhsExpression() const { return m_lhs_xpr; } EIGEN_DEVICE_FUNC - const typename internal::remove_all::type& - rhsExpression() const { return m_rhs_xpr; } + const typename internal::remove_all::type &rhsExpression() const { return m_rhs_xpr; } - protected: - typename LhsXprType::Nested m_lhs_xpr; - typename RhsXprType::Nested m_rhs_xpr; - const Indices m_indices; +protected: + typename LhsXprType::Nested m_lhs_xpr; + typename RhsXprType::Nested m_rhs_xpr; + const Indices m_indices; }; -template -struct TensorContractionEvaluatorBase +template struct TensorContractionEvaluatorBase { typedef typename internal::traits::Indices Indices; typedef typename internal::traits::LeftArgType LeftArgType; @@ -123,7 +125,7 @@ struct TensorContractionEvaluatorBase IsAligned = true, PacketAccess = (internal::unpacket_traits::size > 1), Layout = TensorEvaluator::Layout, - CoordAccess = false, // to be implemented + CoordAccess = false,// to be implemented RawAccess = true }; @@ -131,15 +133,15 @@ struct TensorContractionEvaluatorBase // inputs are RowMajor, we will "cheat" by swapping the LHS and RHS: // If we want to compute A * B = C, where A is LHS and B is RHS, the code // will pretend B is LHS and A is RHS. - typedef typename internal::conditional< - static_cast(Layout) == static_cast(ColMajor), LeftArgType, RightArgType>::type EvalLeftArgType; - typedef typename internal::conditional< - static_cast(Layout) == static_cast(ColMajor), RightArgType, LeftArgType>::type EvalRightArgType; - - static const int LDims = - internal::array_size::Dimensions>::value; - static const int RDims = - internal::array_size::Dimensions>::value; + typedef typename internal::conditional(Layout) == static_cast(ColMajor), + LeftArgType, + RightArgType>::type EvalLeftArgType; + typedef typename internal::conditional(Layout) == static_cast(ColMajor), + RightArgType, + LeftArgType>::type EvalRightArgType; + + static const int LDims = internal::array_size::Dimensions>::value; + static const int RDims = internal::array_size::Dimensions>::value; static const int ContractDims = internal::array_size::value; static const int NumDims = LDims + RDims - 2 * ContractDims; @@ -149,17 +151,18 @@ struct TensorContractionEvaluatorBase typedef DSizes Dimensions; - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE - TensorContractionEvaluatorBase(const XprType& op, const Device& device) - : m_leftImpl(choose(Cond(Layout) == static_cast(ColMajor)>(), - op.lhsExpression(), op.rhsExpression()), device), - m_rightImpl(choose(Cond(Layout) == static_cast(ColMajor)>(), - op.rhsExpression(), op.lhsExpression()), device), - m_device(device), - m_result(NULL) { - EIGEN_STATIC_ASSERT((static_cast(TensorEvaluator::Layout) == - static_cast(TensorEvaluator::Layout)), - YOU_MADE_A_PROGRAMMING_MISTAKE); + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TensorContractionEvaluatorBase(const XprType &op, const Device &device) + : m_leftImpl( + choose(Cond(Layout) == static_cast(ColMajor)>(), op.lhsExpression(), op.rhsExpression()), + device), + m_rightImpl( + choose(Cond(Layout) == static_cast(ColMajor)>(), op.rhsExpression(), op.lhsExpression()), + device), + m_device(device), m_result(NULL) + { + EIGEN_STATIC_ASSERT((static_cast(TensorEvaluator::Layout) + == static_cast(TensorEvaluator::Layout)), + YOU_MADE_A_PROGRAMMING_MISTAKE); DSizes eval_left_dims; @@ -167,12 +170,8 @@ struct TensorContractionEvaluatorBase array, ContractDims> eval_op_indices; if (static_cast(Layout) == static_cast(ColMajor)) { // For ColMajor, we keep using the existing dimensions - for (int i = 0; i < LDims; i++) { - eval_left_dims[i] = m_leftImpl.dimensions()[i]; - } - for (int i = 0; i < RDims; i++) { - eval_right_dims[i] = m_rightImpl.dimensions()[i]; - } + for (int i = 0; i < LDims; i++) { eval_left_dims[i] = m_leftImpl.dimensions()[i]; } + for (int i = 0; i < RDims; i++) { eval_right_dims[i] = m_rightImpl.dimensions()[i]; } // We keep the pairs of contracting indices. for (int i = 0; i < ContractDims; i++) { eval_op_indices[i].first = op.indices()[i].first; @@ -180,12 +179,8 @@ struct TensorContractionEvaluatorBase } } else { // For RowMajor, we need to reverse the existing dimensions - for (int i = 0; i < LDims; i++) { - eval_left_dims[i] = m_leftImpl.dimensions()[LDims - i - 1]; - } - for (int i = 0; i < RDims; i++) { - eval_right_dims[i] = m_rightImpl.dimensions()[RDims - i - 1]; - } + for (int i = 0; i < LDims; i++) { eval_left_dims[i] = m_leftImpl.dimensions()[LDims - i - 1]; } + for (int i = 0; i < RDims; i++) { eval_right_dims[i] = m_rightImpl.dimensions()[RDims - i - 1]; } // We need to flip all the pairs of contracting indices as well as // reversing the dimensions. for (int i = 0; i < ContractDims; i++) { @@ -198,9 +193,8 @@ struct TensorContractionEvaluatorBase // is increasing. Using O(n^2) sorting is OK since ContractDims is small for (int i = 0; i < ContractDims; i++) { for (int j = i + 1; j < ContractDims; j++) { - eigen_assert(eval_op_indices[j].first != eval_op_indices[i].first && - eval_op_indices[j].second != eval_op_indices[i].second && - "contraction axes should be unique"); + eigen_assert(eval_op_indices[j].first != eval_op_indices[i].first + && eval_op_indices[j].second != eval_op_indices[i].second && "contraction axes should be unique"); if (eval_op_indices[j].first < eval_op_indices[i].first) { numext::swap(eval_op_indices[j], eval_op_indices[i]); } @@ -209,15 +203,11 @@ struct TensorContractionEvaluatorBase array lhs_strides; lhs_strides[0] = 1; - for (int i = 0; i < LDims-1; ++i) { - lhs_strides[i+1] = lhs_strides[i] * eval_left_dims[i]; - } + for (int i = 0; i < LDims - 1; ++i) { lhs_strides[i + 1] = lhs_strides[i] * eval_left_dims[i]; } array rhs_strides; rhs_strides[0] = 1; - for (int i = 0; i < RDims-1; ++i) { - rhs_strides[i+1] = rhs_strides[i] * eval_right_dims[i]; - } + for (int i = 0; i < RDims - 1; ++i) { rhs_strides[i + 1] = rhs_strides[i] * eval_right_dims[i]; } if (m_i_strides.size() > 0) m_i_strides[0] = 1; if (m_j_strides.size() > 0) m_j_strides[0] = 1; @@ -248,12 +238,9 @@ struct TensorContractionEvaluatorBase // add dimension size to output dimensions m_dimensions[dim_idx] = eval_left_dims[i]; m_left_nocontract_strides[nocontract_idx] = lhs_strides[i]; - if (dim_idx != i) { - m_lhs_inner_dim_contiguous = false; - } - if (nocontract_idx+1 < internal::array_size::value) { - m_i_strides[nocontract_idx+1] = - m_i_strides[nocontract_idx] * eval_left_dims[i]; + if (dim_idx != i) { m_lhs_inner_dim_contiguous = false; } + if (nocontract_idx + 1 < internal::array_size::value) { + m_i_strides[nocontract_idx + 1] = m_i_strides[nocontract_idx] * eval_left_dims[i]; } else { m_i_size = m_i_strides[nocontract_idx] * eval_left_dims[i]; } @@ -274,9 +261,8 @@ struct TensorContractionEvaluatorBase } if (!contracting) { m_dimensions[dim_idx] = eval_right_dims[i]; - if (nocontract_idx+1 < internal::array_size::value) { - m_j_strides[nocontract_idx+1] = - m_j_strides[nocontract_idx] * eval_right_dims[i]; + if (nocontract_idx + 1 < internal::array_size::value) { + m_j_strides[nocontract_idx + 1] = m_j_strides[nocontract_idx] * eval_right_dims[i]; } else { m_j_size = m_j_strides[nocontract_idx] * eval_right_dims[i]; } @@ -298,36 +284,30 @@ struct TensorContractionEvaluatorBase Index right = eval_op_indices[i].second; Index size = eval_left_dims[left]; - eigen_assert(size == eval_right_dims[right] && - "Contraction axes must be same size"); + eigen_assert(size == eval_right_dims[right] && "Contraction axes must be same size"); - if (i+1 < static_cast(internal::array_size::value)) { - m_k_strides[i+1] = m_k_strides[i] * size; + if (i + 1 < static_cast(internal::array_size::value)) { + m_k_strides[i + 1] = m_k_strides[i] * size; } else { m_k_size = m_k_strides[i] * size; } m_left_contracting_strides[i] = lhs_strides[left]; m_right_contracting_strides[i] = rhs_strides[right]; - if (i > 0 && right < eval_op_indices[i-1].second) { - m_rhs_inner_dim_reordered = true; - } - if (right != i) { - m_rhs_inner_dim_contiguous = false; - } + if (i > 0 && right < eval_op_indices[i - 1].second) { m_rhs_inner_dim_reordered = true; } + if (right != i) { m_rhs_inner_dim_contiguous = false; } } // If the layout is RowMajor, we need to reverse the m_dimensions if (static_cast(Layout) == static_cast(RowMajor)) { - for (int i = 0, j = NumDims - 1; i < j; i++, j--) { - numext::swap(m_dimensions[i], m_dimensions[j]); - } + for (int i = 0, j = NumDims - 1; i < j; i++, j--) { numext::swap(m_dimensions[i], m_dimensions[j]); } } } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const Dimensions& dimensions() const { return m_dimensions; } + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const Dimensions &dimensions() const { return m_dimensions; } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE bool evalSubExprsIfNeeded(Scalar* data) { + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE bool evalSubExprsIfNeeded(Scalar *data) + { m_leftImpl.evalSubExprsIfNeeded(NULL); m_rightImpl.evalSubExprsIfNeeded(NULL); if (data) { @@ -340,47 +320,42 @@ struct TensorContractionEvaluatorBase } } - EIGEN_DEVICE_FUNC void evalTo(Scalar* buffer) const { + EIGEN_DEVICE_FUNC void evalTo(Scalar *buffer) const + { if (this->m_lhs_inner_dim_contiguous) { if (this->m_rhs_inner_dim_contiguous) { if (this->m_rhs_inner_dim_reordered) { - static_cast(this)->template evalProduct(buffer); - } - else { - static_cast(this)->template evalProduct(buffer); - } - } - else { - if (this->m_rhs_inner_dim_reordered) { - static_cast(this)->template evalProduct(buffer); + static_cast(this)->template evalProduct(buffer); + } else { + static_cast(this)->template evalProduct(buffer); } - else { - static_cast(this)->template evalProduct(buffer); + } else { + if (this->m_rhs_inner_dim_reordered) { + static_cast(this)->template evalProduct(buffer); + } else { + static_cast(this)->template evalProduct(buffer); } } - } - else { + } else { if (this->m_rhs_inner_dim_contiguous) { if (this->m_rhs_inner_dim_reordered) { - static_cast(this)->template evalProduct(buffer); - } - else { - static_cast(this)->template evalProduct(buffer); - } - } - else { - if (this->m_rhs_inner_dim_reordered) { - static_cast(this)->template evalProduct(buffer); + static_cast(this)->template evalProduct(buffer); + } else { + static_cast(this)->template evalProduct(buffer); } - else { - static_cast(this)->template evalProduct(buffer); + } else { + if (this->m_rhs_inner_dim_reordered) { + static_cast(this)->template evalProduct(buffer); + } else { + static_cast(this)->template evalProduct(buffer); } } } } - template - EIGEN_DEVICE_FUNC void evalGemv(Scalar* buffer) const { + template + EIGEN_DEVICE_FUNC void evalGemv(Scalar *buffer) const + { const Index rows = m_i_size; const Index cols = m_k_size; @@ -392,22 +367,32 @@ struct TensorContractionEvaluatorBase const Index rhs_packet_size = internal::unpacket_traits::size; const int lhs_alignment = LeftEvaluator::IsAligned ? Aligned : Unaligned; const int rhs_alignment = RightEvaluator::IsAligned ? Aligned : Unaligned; - typedef internal::TensorContractionInputMapper LhsMapper; - - typedef internal::TensorContractionInputMapper RhsMapper; - - LhsMapper lhs(m_leftImpl, m_left_nocontract_strides, m_i_strides, - m_left_contracting_strides, m_k_strides); - RhsMapper rhs(m_rightImpl, m_right_nocontract_strides, m_j_strides, - m_right_contracting_strides, m_k_strides); + typedef internal::TensorContractionInputMapper + LhsMapper; + + typedef internal::TensorContractionInputMapper + RhsMapper; + + LhsMapper lhs(m_leftImpl, m_left_nocontract_strides, m_i_strides, m_left_contracting_strides, m_k_strides); + RhsMapper rhs(m_rightImpl, m_right_nocontract_strides, m_j_strides, m_right_contracting_strides, m_k_strides); const Scalar alpha(1); const Index resIncr(1); @@ -415,13 +400,13 @@ struct TensorContractionEvaluatorBase // zero out the result buffer (which must be of size at least rows * sizeof(Scalar) m_device.memset(buffer, 0, rows * sizeof(Scalar)); - internal::general_matrix_vector_product::run( - rows, cols, lhs, rhs, - buffer, resIncr, alpha); + internal::general_matrix_vector_product:: + run(rows, cols, lhs, rhs, buffer, resIncr, alpha); } - template - EIGEN_DEVICE_FUNC void evalGemm(Scalar* buffer) const { + template + EIGEN_DEVICE_FUNC void evalGemm(Scalar *buffer) const + { // columns in left side, rows in right side const Index k = this->m_k_size; @@ -448,32 +433,51 @@ struct TensorContractionEvaluatorBase const Index lhs_packet_size = internal::unpacket_traits::size; const Index rhs_packet_size = internal::unpacket_traits::size; - typedef internal::TensorContractionInputMapper LhsMapper; - - typedef internal::TensorContractionInputMapper RhsMapper; + typedef internal::TensorContractionInputMapper + LhsMapper; + + typedef internal::TensorContractionInputMapper + RhsMapper; typedef internal::blas_data_mapper OutputMapper; // Declare GEBP packing and kernel structs - internal::gemm_pack_lhs pack_lhs; + internal::gemm_pack_lhs + pack_lhs; internal::gemm_pack_rhs pack_rhs; internal::gebp_kernel gebp; // initialize data mappers - LhsMapper lhs(this->m_leftImpl, this->m_left_nocontract_strides, this->m_i_strides, - this->m_left_contracting_strides, this->m_k_strides); - - RhsMapper rhs(this->m_rightImpl, this->m_right_nocontract_strides, this->m_j_strides, - this->m_right_contracting_strides, this->m_k_strides); + LhsMapper lhs(this->m_leftImpl, + this->m_left_nocontract_strides, + this->m_i_strides, + this->m_left_contracting_strides, + this->m_k_strides); + + RhsMapper rhs(this->m_rightImpl, + this->m_right_nocontract_strides, + this->m_j_strides, + this->m_right_contracting_strides, + this->m_k_strides); OutputMapper output(buffer, m); @@ -485,12 +489,11 @@ struct TensorContractionEvaluatorBase const Index sizeA = mc * kc; const Index sizeB = kc * nc; - LhsScalar* blockA = static_cast(this->m_device.allocate(sizeA * sizeof(LhsScalar))); - RhsScalar* blockB = static_cast(this->m_device.allocate(sizeB * sizeof(RhsScalar))); + LhsScalar *blockA = static_cast(this->m_device.allocate(sizeA * sizeof(LhsScalar))); + RhsScalar *blockB = static_cast(this->m_device.allocate(sizeB * sizeof(RhsScalar))); - for(Index i2=0; i2m_device.deallocate(blockB); } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void cleanup() { + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void cleanup() + { m_leftImpl.cleanup(); m_rightImpl.cleanup(); @@ -523,24 +527,23 @@ struct TensorContractionEvaluatorBase } } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE CoeffReturnType coeff(Index index) const { - return m_result[index]; - } + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE CoeffReturnType coeff(Index index) const { return m_result[index]; } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TensorOpCost costPerCoeff(bool) const { + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TensorOpCost costPerCoeff(bool) const + { return TensorOpCost(sizeof(CoeffReturnType), 0, 0); } - template - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE PacketReturnType packet(Index index) const { + template EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE PacketReturnType packet(Index index) const + { return internal::ploadt(m_result + index); } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Scalar* data() const { return m_result; } + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Scalar *data() const { return m_result; } - protected: +protected: // Prevent assignment - TensorContractionEvaluatorBase& operator = (const TensorContractionEvaluatorBase&); + TensorContractionEvaluatorBase &operator=(const TensorContractionEvaluatorBase &); Dimensions m_dimensions; contract_t m_k_strides; @@ -562,16 +565,17 @@ struct TensorContractionEvaluatorBase TensorEvaluator m_leftImpl; TensorEvaluator m_rightImpl; - const Device& m_device; - Scalar* m_result; + const Device &m_device; + Scalar *m_result; }; // evaluator for default device template -struct TensorEvaluator, Device> : - public TensorContractionEvaluatorBase< - TensorEvaluator, Device> > { +struct TensorEvaluator, Device> + : public TensorContractionEvaluatorBase< + TensorEvaluator, Device>> +{ typedef TensorEvaluator, Device> Self; typedef TensorContractionEvaluatorBase Base; @@ -581,23 +585,21 @@ struct TensorEvaluator::type PacketReturnType; - enum { - Layout = TensorEvaluator::Layout - }; + enum { Layout = TensorEvaluator::Layout }; // Most of the code is assuming that both input tensors are ColMajor. If the // inputs are RowMajor, we will "cheat" by swapping the LHS and RHS: // If we want to compute A * B = C, where A is LHS and B is RHS, the code // will pretend B is LHS and A is RHS. - typedef typename internal::conditional< - static_cast(Layout) == static_cast(ColMajor), LeftArgType, RightArgType>::type EvalLeftArgType; - typedef typename internal::conditional< - static_cast(Layout) == static_cast(ColMajor), RightArgType, LeftArgType>::type EvalRightArgType; - - static const int LDims = - internal::array_size::Dimensions>::value; - static const int RDims = - internal::array_size::Dimensions>::value; + typedef typename internal::conditional(Layout) == static_cast(ColMajor), + LeftArgType, + RightArgType>::type EvalLeftArgType; + typedef typename internal::conditional(Layout) == static_cast(ColMajor), + RightArgType, + LeftArgType>::type EvalRightArgType; + + static const int LDims = internal::array_size::Dimensions>::value; + static const int RDims = internal::array_size::Dimensions>::value; static const int ContractDims = internal::array_size::value; typedef array contract_t; @@ -609,20 +611,22 @@ struct TensorEvaluator Dimensions; - EIGEN_DEVICE_FUNC TensorEvaluator(const XprType& op, const Device& device) : - Base(op, device) { } + EIGEN_DEVICE_FUNC TensorEvaluator(const XprType &op, const Device &device) : Base(op, device) {} - template - EIGEN_DEVICE_FUNC void evalProduct(Scalar* buffer) const { + template + EIGEN_DEVICE_FUNC void evalProduct(Scalar *buffer) const + { if (this->m_j_size == 1) { - this->template evalGemv(buffer); + this->template evalGemv( + buffer); return; } - this->template evalGemm(buffer); + this->template evalGemm( + buffer); } }; -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_CXX11_TENSOR_TENSOR_CONTRACTION_H +#endif// EIGEN_CXX11_TENSOR_TENSOR_CONTRACTION_H diff --git a/filmulator-gui/core/nlmeans/eigen/unsupported/Eigen/CXX11/src/Tensor/TensorContractionBlocking.h b/filmulator-gui/core/nlmeans/eigen/unsupported/Eigen/CXX11/src/Tensor/TensorContractionBlocking.h index 5cf7b4f7..28151cdc 100644 --- a/filmulator-gui/core/nlmeans/eigen/unsupported/Eigen/CXX11/src/Tensor/TensorContractionBlocking.h +++ b/filmulator-gui/core/nlmeans/eigen/unsupported/Eigen/CXX11/src/Tensor/TensorContractionBlocking.h @@ -14,43 +14,39 @@ namespace Eigen { namespace internal { -enum { - ShardByRow = 0, - ShardByCol = 1 -}; + enum { ShardByRow = 0, ShardByCol = 1 }; -// Default Blocking Strategy -template -class TensorContractionBlocking { - public: - - typedef typename LhsMapper::Scalar LhsScalar; - typedef typename RhsMapper::Scalar RhsScalar; - - EIGEN_DEVICE_FUNC TensorContractionBlocking(Index k, Index m, Index n, Index num_threads = 1) : - kc_(k), mc_(m), nc_(n) + // Default Blocking Strategy + template + class TensorContractionBlocking { - if (ShardingType == ShardByCol) { - computeProductBlockingSizes(kc_, mc_, nc_, num_threads); - } - else { - computeProductBlockingSizes(kc_, nc_, mc_, num_threads); + public: + typedef typename LhsMapper::Scalar LhsScalar; + typedef typename RhsMapper::Scalar RhsScalar; + + EIGEN_DEVICE_FUNC TensorContractionBlocking(Index k, Index m, Index n, Index num_threads = 1) + : kc_(k), mc_(m), nc_(n) + { + if (ShardingType == ShardByCol) { + computeProductBlockingSizes(kc_, mc_, nc_, num_threads); + } else { + computeProductBlockingSizes(kc_, nc_, mc_, num_threads); + } } - } - EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE Index kc() const { return kc_; } - EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE Index mc() const { return mc_; } - EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE Index nc() const { return nc_; } + EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE Index kc() const { return kc_; } + EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE Index mc() const { return mc_; } + EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE Index nc() const { return nc_; } - private: - Index kc_; - Index mc_; - Index nc_; -}; + private: + Index kc_; + Index mc_; + Index nc_; + }; -} // end namespace internal -} // end namespace Eigen +}// end namespace internal +}// end namespace Eigen -#endif // EIGEN_CXX11_TENSOR_TENSOR_CONTRACTION_BLOCKING_H +#endif// EIGEN_CXX11_TENSOR_TENSOR_CONTRACTION_BLOCKING_H diff --git a/filmulator-gui/core/nlmeans/eigen/unsupported/Eigen/CXX11/src/Tensor/TensorContractionCuda.h b/filmulator-gui/core/nlmeans/eigen/unsupported/Eigen/CXX11/src/Tensor/TensorContractionCuda.h index d65dbb40..87f6fee8 100644 --- a/filmulator-gui/core/nlmeans/eigen/unsupported/Eigen/CXX11/src/Tensor/TensorContractionCuda.h +++ b/filmulator-gui/core/nlmeans/eigen/unsupported/Eigen/CXX11/src/Tensor/TensorContractionCuda.h @@ -16,12 +16,21 @@ namespace Eigen { -template -__device__ EIGEN_STRONG_INLINE void -EigenContractionKernelInternal(const LhsMapper lhs, const RhsMapper rhs, - const OutputMapper output, Scalar* lhs_shmem, Scalar* rhs_shmem, - const Index m_size, const Index n_size, const Index k_size) { +template +__device__ EIGEN_STRONG_INLINE void EigenContractionKernelInternal(const LhsMapper lhs, + const RhsMapper rhs, + const OutputMapper output, + Scalar *lhs_shmem, + Scalar *rhs_shmem, + const Index m_size, + const Index n_size, + const Index k_size) +{ const Index m_block_idx = blockIdx.x; const Index n_block_idx = blockIdx.y; @@ -97,178 +106,178 @@ EigenContractionKernelInternal(const LhsMapper lhs, const RhsMapper rhs, const Index load_idx_vert = threadIdx.x + 8 * threadIdx.y; const Index lhs_vert = base_m + load_idx_vert; -#define prefetchIntoRegisters(base_k) \ - { \ - lhs_pf0 = conv(0); \ - lhs_pf1 = conv(0); \ - lhs_pf2 = conv(0); \ - lhs_pf3 = conv(0); \ - lhs_pf4 = conv(0); \ - lhs_pf5 = conv(0); \ - lhs_pf6 = conv(0); \ - lhs_pf7 = conv(0); \ - \ - rhs_pf0 = conv(0); \ - rhs_pf1 = conv(0); \ - rhs_pf2 = conv(0); \ - rhs_pf3 = conv(0); \ - rhs_pf4 = conv(0); \ - rhs_pf5 = conv(0); \ - rhs_pf6 = conv(0); \ - rhs_pf7 = conv(0); \ - \ - if (!needs_edge_check || lhs_vert < m_size) { \ - const Index lhs_horiz_0 = base_k + threadIdx.z + 0 * 8; \ - const Index lhs_horiz_1 = base_k + threadIdx.z + 1 * 8; \ - const Index lhs_horiz_2 = base_k + threadIdx.z + 2 * 8; \ - const Index lhs_horiz_3 = base_k + threadIdx.z + 3 * 8; \ - const Index lhs_horiz_4 = base_k + threadIdx.z + 4 * 8; \ - const Index lhs_horiz_5 = base_k + threadIdx.z + 5 * 8; \ - const Index lhs_horiz_6 = base_k + threadIdx.z + 6 * 8; \ - const Index lhs_horiz_7 = base_k + threadIdx.z + 7 * 8; \ - \ - if (!needs_edge_check || lhs_horiz_7 < k_size) { \ - lhs_pf0 = lhs(lhs_vert, lhs_horiz_0); \ - lhs_pf1 = lhs(lhs_vert, lhs_horiz_1); \ - lhs_pf2 = lhs(lhs_vert, lhs_horiz_2); \ - lhs_pf3 = lhs(lhs_vert, lhs_horiz_3); \ - lhs_pf4 = lhs(lhs_vert, lhs_horiz_4); \ - lhs_pf5 = lhs(lhs_vert, lhs_horiz_5); \ - lhs_pf6 = lhs(lhs_vert, lhs_horiz_6); \ - lhs_pf7 = lhs(lhs_vert, lhs_horiz_7); \ - } else if (lhs_horiz_6 < k_size) { \ - lhs_pf0 = lhs(lhs_vert, lhs_horiz_0); \ - lhs_pf1 = lhs(lhs_vert, lhs_horiz_1); \ - lhs_pf2 = lhs(lhs_vert, lhs_horiz_2); \ - lhs_pf3 = lhs(lhs_vert, lhs_horiz_3); \ - lhs_pf4 = lhs(lhs_vert, lhs_horiz_4); \ - lhs_pf5 = lhs(lhs_vert, lhs_horiz_5); \ - lhs_pf6 = lhs(lhs_vert, lhs_horiz_6); \ - } else if (lhs_horiz_5 < k_size) { \ - lhs_pf0 = lhs(lhs_vert, lhs_horiz_0); \ - lhs_pf1 = lhs(lhs_vert, lhs_horiz_1); \ - lhs_pf2 = lhs(lhs_vert, lhs_horiz_2); \ - lhs_pf3 = lhs(lhs_vert, lhs_horiz_3); \ - lhs_pf4 = lhs(lhs_vert, lhs_horiz_4); \ - lhs_pf5 = lhs(lhs_vert, lhs_horiz_5); \ - } else if (lhs_horiz_4 < k_size) { \ - lhs_pf0 = lhs(lhs_vert, lhs_horiz_0); \ - lhs_pf1 = lhs(lhs_vert, lhs_horiz_1); \ - lhs_pf2 = lhs(lhs_vert, lhs_horiz_2); \ - lhs_pf3 = lhs(lhs_vert, lhs_horiz_3); \ - lhs_pf4 = lhs(lhs_vert, lhs_horiz_4); \ - } else if (lhs_horiz_3 < k_size) { \ - lhs_pf0 = lhs(lhs_vert, lhs_horiz_0); \ - lhs_pf1 = lhs(lhs_vert, lhs_horiz_1); \ - lhs_pf2 = lhs(lhs_vert, lhs_horiz_2); \ - lhs_pf3 = lhs(lhs_vert, lhs_horiz_3); \ - } else if (lhs_horiz_2 < k_size) { \ - lhs_pf0 = lhs(lhs_vert, lhs_horiz_0); \ - lhs_pf1 = lhs(lhs_vert, lhs_horiz_1); \ - lhs_pf2 = lhs(lhs_vert, lhs_horiz_2); \ - } else if (lhs_horiz_1 < k_size) { \ - lhs_pf0 = lhs(lhs_vert, lhs_horiz_0); \ - lhs_pf1 = lhs(lhs_vert, lhs_horiz_1); \ - } else if (lhs_horiz_0 < k_size) { \ - lhs_pf0 = lhs(lhs_vert, lhs_horiz_0); \ - } \ - } \ - \ - const Index rhs_vert = base_k + load_idx_vert; \ - if (!needs_edge_check || rhs_vert < k_size) { \ - const Index rhs_horiz_0 = base_n + threadIdx.z + 0 * 8; \ - const Index rhs_horiz_1 = base_n + threadIdx.z + 1 * 8; \ - const Index rhs_horiz_2 = base_n + threadIdx.z + 2 * 8; \ - const Index rhs_horiz_3 = base_n + threadIdx.z + 3 * 8; \ - const Index rhs_horiz_4 = base_n + threadIdx.z + 4 * 8; \ - const Index rhs_horiz_5 = base_n + threadIdx.z + 5 * 8; \ - const Index rhs_horiz_6 = base_n + threadIdx.z + 6 * 8; \ - const Index rhs_horiz_7 = base_n + threadIdx.z + 7 * 8; \ - \ - if (rhs_horiz_7 < n_size) { \ - rhs_pf0 = rhs(rhs_vert, rhs_horiz_0); \ - rhs_pf1 = rhs(rhs_vert, rhs_horiz_1); \ - rhs_pf2 = rhs(rhs_vert, rhs_horiz_2); \ - rhs_pf3 = rhs(rhs_vert, rhs_horiz_3); \ - rhs_pf4 = rhs(rhs_vert, rhs_horiz_4); \ - rhs_pf5 = rhs(rhs_vert, rhs_horiz_5); \ - rhs_pf6 = rhs(rhs_vert, rhs_horiz_6); \ - rhs_pf7 = rhs(rhs_vert, rhs_horiz_7); \ - } else if (rhs_horiz_6 < n_size) { \ - rhs_pf0 = rhs(rhs_vert, rhs_horiz_0); \ - rhs_pf1 = rhs(rhs_vert, rhs_horiz_1); \ - rhs_pf2 = rhs(rhs_vert, rhs_horiz_2); \ - rhs_pf3 = rhs(rhs_vert, rhs_horiz_3); \ - rhs_pf4 = rhs(rhs_vert, rhs_horiz_4); \ - rhs_pf5 = rhs(rhs_vert, rhs_horiz_5); \ - rhs_pf6 = rhs(rhs_vert, rhs_horiz_6); \ - } else if (rhs_horiz_5 < n_size) { \ - rhs_pf0 = rhs(rhs_vert, rhs_horiz_0); \ - rhs_pf1 = rhs(rhs_vert, rhs_horiz_1); \ - rhs_pf2 = rhs(rhs_vert, rhs_horiz_2); \ - rhs_pf3 = rhs(rhs_vert, rhs_horiz_3); \ - rhs_pf4 = rhs(rhs_vert, rhs_horiz_4); \ - rhs_pf5 = rhs(rhs_vert, rhs_horiz_5); \ - } else if (rhs_horiz_4 < n_size) { \ - rhs_pf0 = rhs(rhs_vert, rhs_horiz_0); \ - rhs_pf1 = rhs(rhs_vert, rhs_horiz_1); \ - rhs_pf2 = rhs(rhs_vert, rhs_horiz_2); \ - rhs_pf3 = rhs(rhs_vert, rhs_horiz_3); \ - rhs_pf4 = rhs(rhs_vert, rhs_horiz_4); \ - } else if (rhs_horiz_3 < n_size) { \ - rhs_pf0 = rhs(rhs_vert, rhs_horiz_0); \ - rhs_pf1 = rhs(rhs_vert, rhs_horiz_1); \ - rhs_pf2 = rhs(rhs_vert, rhs_horiz_2); \ - rhs_pf3 = rhs(rhs_vert, rhs_horiz_3); \ - } else if (rhs_horiz_2 < n_size) { \ - rhs_pf0 = rhs(rhs_vert, rhs_horiz_0); \ - rhs_pf1 = rhs(rhs_vert, rhs_horiz_1); \ - rhs_pf2 = rhs(rhs_vert, rhs_horiz_2); \ - } else if (rhs_horiz_1 < n_size) { \ - rhs_pf0 = rhs(rhs_vert, rhs_horiz_0); \ - rhs_pf1 = rhs(rhs_vert, rhs_horiz_1); \ - } else if (rhs_horiz_0 < n_size) { \ - rhs_pf0 = rhs(rhs_vert, rhs_horiz_0); \ - } \ - } \ - } \ - -#define writeRegToShmem(_) \ - lhs_shmem[lhs_store_idx_0] = lhs_pf0; \ - rhs_shmem[rhs_store_idx_0] = rhs_pf0; \ - \ - lhs_shmem[lhs_store_idx_1] = lhs_pf1; \ - rhs_shmem[rhs_store_idx_1] = rhs_pf1; \ - \ - lhs_shmem[lhs_store_idx_2] = lhs_pf2; \ - rhs_shmem[rhs_store_idx_2] = rhs_pf2; \ - \ - lhs_shmem[lhs_store_idx_3] = lhs_pf3; \ - rhs_shmem[rhs_store_idx_3] = rhs_pf3; \ - \ - lhs_shmem[lhs_store_idx_4] = lhs_pf4; \ - rhs_shmem[rhs_store_idx_4] = rhs_pf4; \ - \ - lhs_shmem[lhs_store_idx_5] = lhs_pf5; \ - rhs_shmem[rhs_store_idx_5] = rhs_pf5; \ - \ - lhs_shmem[lhs_store_idx_6] = lhs_pf6; \ - rhs_shmem[rhs_store_idx_6] = rhs_pf6; \ - \ - lhs_shmem[lhs_store_idx_7] = lhs_pf7; \ - rhs_shmem[rhs_store_idx_7] = rhs_pf7; \ +#define prefetchIntoRegisters(base_k) \ + { \ + lhs_pf0 = conv(0); \ + lhs_pf1 = conv(0); \ + lhs_pf2 = conv(0); \ + lhs_pf3 = conv(0); \ + lhs_pf4 = conv(0); \ + lhs_pf5 = conv(0); \ + lhs_pf6 = conv(0); \ + lhs_pf7 = conv(0); \ + \ + rhs_pf0 = conv(0); \ + rhs_pf1 = conv(0); \ + rhs_pf2 = conv(0); \ + rhs_pf3 = conv(0); \ + rhs_pf4 = conv(0); \ + rhs_pf5 = conv(0); \ + rhs_pf6 = conv(0); \ + rhs_pf7 = conv(0); \ + \ + if (!needs_edge_check || lhs_vert < m_size) { \ + const Index lhs_horiz_0 = base_k + threadIdx.z + 0 * 8; \ + const Index lhs_horiz_1 = base_k + threadIdx.z + 1 * 8; \ + const Index lhs_horiz_2 = base_k + threadIdx.z + 2 * 8; \ + const Index lhs_horiz_3 = base_k + threadIdx.z + 3 * 8; \ + const Index lhs_horiz_4 = base_k + threadIdx.z + 4 * 8; \ + const Index lhs_horiz_5 = base_k + threadIdx.z + 5 * 8; \ + const Index lhs_horiz_6 = base_k + threadIdx.z + 6 * 8; \ + const Index lhs_horiz_7 = base_k + threadIdx.z + 7 * 8; \ + \ + if (!needs_edge_check || lhs_horiz_7 < k_size) { \ + lhs_pf0 = lhs(lhs_vert, lhs_horiz_0); \ + lhs_pf1 = lhs(lhs_vert, lhs_horiz_1); \ + lhs_pf2 = lhs(lhs_vert, lhs_horiz_2); \ + lhs_pf3 = lhs(lhs_vert, lhs_horiz_3); \ + lhs_pf4 = lhs(lhs_vert, lhs_horiz_4); \ + lhs_pf5 = lhs(lhs_vert, lhs_horiz_5); \ + lhs_pf6 = lhs(lhs_vert, lhs_horiz_6); \ + lhs_pf7 = lhs(lhs_vert, lhs_horiz_7); \ + } else if (lhs_horiz_6 < k_size) { \ + lhs_pf0 = lhs(lhs_vert, lhs_horiz_0); \ + lhs_pf1 = lhs(lhs_vert, lhs_horiz_1); \ + lhs_pf2 = lhs(lhs_vert, lhs_horiz_2); \ + lhs_pf3 = lhs(lhs_vert, lhs_horiz_3); \ + lhs_pf4 = lhs(lhs_vert, lhs_horiz_4); \ + lhs_pf5 = lhs(lhs_vert, lhs_horiz_5); \ + lhs_pf6 = lhs(lhs_vert, lhs_horiz_6); \ + } else if (lhs_horiz_5 < k_size) { \ + lhs_pf0 = lhs(lhs_vert, lhs_horiz_0); \ + lhs_pf1 = lhs(lhs_vert, lhs_horiz_1); \ + lhs_pf2 = lhs(lhs_vert, lhs_horiz_2); \ + lhs_pf3 = lhs(lhs_vert, lhs_horiz_3); \ + lhs_pf4 = lhs(lhs_vert, lhs_horiz_4); \ + lhs_pf5 = lhs(lhs_vert, lhs_horiz_5); \ + } else if (lhs_horiz_4 < k_size) { \ + lhs_pf0 = lhs(lhs_vert, lhs_horiz_0); \ + lhs_pf1 = lhs(lhs_vert, lhs_horiz_1); \ + lhs_pf2 = lhs(lhs_vert, lhs_horiz_2); \ + lhs_pf3 = lhs(lhs_vert, lhs_horiz_3); \ + lhs_pf4 = lhs(lhs_vert, lhs_horiz_4); \ + } else if (lhs_horiz_3 < k_size) { \ + lhs_pf0 = lhs(lhs_vert, lhs_horiz_0); \ + lhs_pf1 = lhs(lhs_vert, lhs_horiz_1); \ + lhs_pf2 = lhs(lhs_vert, lhs_horiz_2); \ + lhs_pf3 = lhs(lhs_vert, lhs_horiz_3); \ + } else if (lhs_horiz_2 < k_size) { \ + lhs_pf0 = lhs(lhs_vert, lhs_horiz_0); \ + lhs_pf1 = lhs(lhs_vert, lhs_horiz_1); \ + lhs_pf2 = lhs(lhs_vert, lhs_horiz_2); \ + } else if (lhs_horiz_1 < k_size) { \ + lhs_pf0 = lhs(lhs_vert, lhs_horiz_0); \ + lhs_pf1 = lhs(lhs_vert, lhs_horiz_1); \ + } else if (lhs_horiz_0 < k_size) { \ + lhs_pf0 = lhs(lhs_vert, lhs_horiz_0); \ + } \ + } \ + \ + const Index rhs_vert = base_k + load_idx_vert; \ + if (!needs_edge_check || rhs_vert < k_size) { \ + const Index rhs_horiz_0 = base_n + threadIdx.z + 0 * 8; \ + const Index rhs_horiz_1 = base_n + threadIdx.z + 1 * 8; \ + const Index rhs_horiz_2 = base_n + threadIdx.z + 2 * 8; \ + const Index rhs_horiz_3 = base_n + threadIdx.z + 3 * 8; \ + const Index rhs_horiz_4 = base_n + threadIdx.z + 4 * 8; \ + const Index rhs_horiz_5 = base_n + threadIdx.z + 5 * 8; \ + const Index rhs_horiz_6 = base_n + threadIdx.z + 6 * 8; \ + const Index rhs_horiz_7 = base_n + threadIdx.z + 7 * 8; \ + \ + if (rhs_horiz_7 < n_size) { \ + rhs_pf0 = rhs(rhs_vert, rhs_horiz_0); \ + rhs_pf1 = rhs(rhs_vert, rhs_horiz_1); \ + rhs_pf2 = rhs(rhs_vert, rhs_horiz_2); \ + rhs_pf3 = rhs(rhs_vert, rhs_horiz_3); \ + rhs_pf4 = rhs(rhs_vert, rhs_horiz_4); \ + rhs_pf5 = rhs(rhs_vert, rhs_horiz_5); \ + rhs_pf6 = rhs(rhs_vert, rhs_horiz_6); \ + rhs_pf7 = rhs(rhs_vert, rhs_horiz_7); \ + } else if (rhs_horiz_6 < n_size) { \ + rhs_pf0 = rhs(rhs_vert, rhs_horiz_0); \ + rhs_pf1 = rhs(rhs_vert, rhs_horiz_1); \ + rhs_pf2 = rhs(rhs_vert, rhs_horiz_2); \ + rhs_pf3 = rhs(rhs_vert, rhs_horiz_3); \ + rhs_pf4 = rhs(rhs_vert, rhs_horiz_4); \ + rhs_pf5 = rhs(rhs_vert, rhs_horiz_5); \ + rhs_pf6 = rhs(rhs_vert, rhs_horiz_6); \ + } else if (rhs_horiz_5 < n_size) { \ + rhs_pf0 = rhs(rhs_vert, rhs_horiz_0); \ + rhs_pf1 = rhs(rhs_vert, rhs_horiz_1); \ + rhs_pf2 = rhs(rhs_vert, rhs_horiz_2); \ + rhs_pf3 = rhs(rhs_vert, rhs_horiz_3); \ + rhs_pf4 = rhs(rhs_vert, rhs_horiz_4); \ + rhs_pf5 = rhs(rhs_vert, rhs_horiz_5); \ + } else if (rhs_horiz_4 < n_size) { \ + rhs_pf0 = rhs(rhs_vert, rhs_horiz_0); \ + rhs_pf1 = rhs(rhs_vert, rhs_horiz_1); \ + rhs_pf2 = rhs(rhs_vert, rhs_horiz_2); \ + rhs_pf3 = rhs(rhs_vert, rhs_horiz_3); \ + rhs_pf4 = rhs(rhs_vert, rhs_horiz_4); \ + } else if (rhs_horiz_3 < n_size) { \ + rhs_pf0 = rhs(rhs_vert, rhs_horiz_0); \ + rhs_pf1 = rhs(rhs_vert, rhs_horiz_1); \ + rhs_pf2 = rhs(rhs_vert, rhs_horiz_2); \ + rhs_pf3 = rhs(rhs_vert, rhs_horiz_3); \ + } else if (rhs_horiz_2 < n_size) { \ + rhs_pf0 = rhs(rhs_vert, rhs_horiz_0); \ + rhs_pf1 = rhs(rhs_vert, rhs_horiz_1); \ + rhs_pf2 = rhs(rhs_vert, rhs_horiz_2); \ + } else if (rhs_horiz_1 < n_size) { \ + rhs_pf0 = rhs(rhs_vert, rhs_horiz_0); \ + rhs_pf1 = rhs(rhs_vert, rhs_horiz_1); \ + } else if (rhs_horiz_0 < n_size) { \ + rhs_pf0 = rhs(rhs_vert, rhs_horiz_0); \ + } \ + } \ + } + +#define writeRegToShmem(_) \ + lhs_shmem[lhs_store_idx_0] = lhs_pf0; \ + rhs_shmem[rhs_store_idx_0] = rhs_pf0; \ + \ + lhs_shmem[lhs_store_idx_1] = lhs_pf1; \ + rhs_shmem[rhs_store_idx_1] = rhs_pf1; \ + \ + lhs_shmem[lhs_store_idx_2] = lhs_pf2; \ + rhs_shmem[rhs_store_idx_2] = rhs_pf2; \ + \ + lhs_shmem[lhs_store_idx_3] = lhs_pf3; \ + rhs_shmem[rhs_store_idx_3] = rhs_pf3; \ + \ + lhs_shmem[lhs_store_idx_4] = lhs_pf4; \ + rhs_shmem[rhs_store_idx_4] = rhs_pf4; \ + \ + lhs_shmem[lhs_store_idx_5] = lhs_pf5; \ + rhs_shmem[rhs_store_idx_5] = rhs_pf5; \ + \ + lhs_shmem[lhs_store_idx_6] = lhs_pf6; \ + rhs_shmem[rhs_store_idx_6] = rhs_pf6; \ + \ + lhs_shmem[lhs_store_idx_7] = lhs_pf7; \ + rhs_shmem[rhs_store_idx_7] = rhs_pf7; // declare and initialize result array #define res(i, j) _res_##i##j -#define initResultRow(i) \ - Scalar res(i, 0) = conv(0); \ - Scalar res(i, 1) = conv(0); \ - Scalar res(i, 2) = conv(0); \ - Scalar res(i, 3) = conv(0); \ - Scalar res(i, 4) = conv(0); \ - Scalar res(i, 5) = conv(0); \ - Scalar res(i, 6) = conv(0); \ - Scalar res(i, 7) = conv(0); \ +#define initResultRow(i) \ + Scalar res(i, 0) = conv(0); \ + Scalar res(i, 1) = conv(0); \ + Scalar res(i, 2) = conv(0); \ + Scalar res(i, 3) = conv(0); \ + Scalar res(i, 4) = conv(0); \ + Scalar res(i, 5) = conv(0); \ + Scalar res(i, 6) = conv(0); \ + Scalar res(i, 7) = conv(0); internal::scalar_cast_op conv; initResultRow(0); @@ -289,8 +298,8 @@ EigenContractionKernelInternal(const LhsMapper lhs, const RhsMapper rhs, prefetchIntoRegisters(base_k); writeRegToShmem(); - #undef prefetchIntoRegisters - #undef writeRegToShmem +#undef prefetchIntoRegisters +#undef writeRegToShmem // wait for shared mem packing to be done before starting computation __syncthreads(); @@ -319,51 +328,51 @@ EigenContractionKernelInternal(const LhsMapper lhs, const RhsMapper rhs, Scalar rrow(7); // Now x corresponds to k, y to m, and z to n - const Scalar* lhs_block = &lhs_shmem[threadIdx.x + 9 * threadIdx.y]; - const Scalar* rhs_block = &rhs_shmem[threadIdx.x + 8 * threadIdx.z]; + const Scalar *lhs_block = &lhs_shmem[threadIdx.x + 9 * threadIdx.y]; + const Scalar *rhs_block = &rhs_shmem[threadIdx.x + 8 * threadIdx.z]; #define lhs_element(i, j) lhs_block[72 * ((i) + 8 * (j))] #define rhs_element(i, j) rhs_block[72 * ((i) + 8 * (j))] -#define loadData(i, j) \ - lcol(0) = lhs_element(0, j); \ - rrow(0) = rhs_element(i, 0); \ - lcol(1) = lhs_element(1, j); \ - rrow(1) = rhs_element(i, 1); \ - lcol(2) = lhs_element(2, j); \ - rrow(2) = rhs_element(i, 2); \ - lcol(3) = lhs_element(3, j); \ - rrow(3) = rhs_element(i, 3); \ - lcol(4) = lhs_element(4, j); \ - rrow(4) = rhs_element(i, 4); \ - lcol(5) = lhs_element(5, j); \ - rrow(5) = rhs_element(i, 5); \ - lcol(6) = lhs_element(6, j); \ - rrow(6) = rhs_element(i, 6); \ - lcol(7) = lhs_element(7, j); \ - rrow(7) = rhs_element(i, 7); \ - -#define computeCol(j) \ - res(0, j) += lcol(0) * rrow(j); \ - res(1, j) += lcol(1) * rrow(j); \ - res(2, j) += lcol(2) * rrow(j); \ - res(3, j) += lcol(3) * rrow(j); \ - res(4, j) += lcol(4) * rrow(j); \ - res(5, j) += lcol(5) * rrow(j); \ - res(6, j) += lcol(6) * rrow(j); \ - res(7, j) += lcol(7) * rrow(j); \ - -#define computePass(i) \ - loadData(i, i); \ - \ - computeCol(0); \ - computeCol(1); \ - computeCol(2); \ - computeCol(3); \ - computeCol(4); \ - computeCol(5); \ - computeCol(6); \ - computeCol(7); \ +#define loadData(i, j) \ + lcol(0) = lhs_element(0, j); \ + rrow(0) = rhs_element(i, 0); \ + lcol(1) = lhs_element(1, j); \ + rrow(1) = rhs_element(i, 1); \ + lcol(2) = lhs_element(2, j); \ + rrow(2) = rhs_element(i, 2); \ + lcol(3) = lhs_element(3, j); \ + rrow(3) = rhs_element(i, 3); \ + lcol(4) = lhs_element(4, j); \ + rrow(4) = rhs_element(i, 4); \ + lcol(5) = lhs_element(5, j); \ + rrow(5) = rhs_element(i, 5); \ + lcol(6) = lhs_element(6, j); \ + rrow(6) = rhs_element(i, 6); \ + lcol(7) = lhs_element(7, j); \ + rrow(7) = rhs_element(i, 7); + +#define computeCol(j) \ + res(0, j) += lcol(0) * rrow(j); \ + res(1, j) += lcol(1) * rrow(j); \ + res(2, j) += lcol(2) * rrow(j); \ + res(3, j) += lcol(3) * rrow(j); \ + res(4, j) += lcol(4) * rrow(j); \ + res(5, j) += lcol(5) * rrow(j); \ + res(6, j) += lcol(6) * rrow(j); \ + res(7, j) += lcol(7) * rrow(j); + +#define computePass(i) \ + loadData(i, i); \ + \ + computeCol(0); \ + computeCol(1); \ + computeCol(2); \ + computeCol(3); \ + computeCol(4); \ + computeCol(5); \ + computeCol(6); \ + computeCol(7); computePass(0); computePass(1); @@ -381,7 +390,7 @@ EigenContractionKernelInternal(const LhsMapper lhs, const RhsMapper rhs, #undef loadData #undef computeCol #undef computePass - } // end loop over k + }// end loop over k // we've now iterated over all of the large (ie width 64) k blocks and // accumulated results in registers. At this point thread (x, y, z) contains @@ -390,25 +399,25 @@ EigenContractionKernelInternal(const LhsMapper lhs, const RhsMapper rhs, // the 8 threads over y by summation. #define shuffleInc(i, j, mask) res(i, j) += __shfl_xor(res(i, j), mask) -#define reduceRow(i, mask) \ - shuffleInc(i, 0, mask); \ - shuffleInc(i, 1, mask); \ - shuffleInc(i, 2, mask); \ - shuffleInc(i, 3, mask); \ - shuffleInc(i, 4, mask); \ - shuffleInc(i, 5, mask); \ - shuffleInc(i, 6, mask); \ - shuffleInc(i, 7, mask); \ - -#define reduceMatrix(mask) \ - reduceRow(0, mask); \ - reduceRow(1, mask); \ - reduceRow(2, mask); \ - reduceRow(3, mask); \ - reduceRow(4, mask); \ - reduceRow(5, mask); \ - reduceRow(6, mask); \ - reduceRow(7, mask); \ +#define reduceRow(i, mask) \ + shuffleInc(i, 0, mask); \ + shuffleInc(i, 1, mask); \ + shuffleInc(i, 2, mask); \ + shuffleInc(i, 3, mask); \ + shuffleInc(i, 4, mask); \ + shuffleInc(i, 5, mask); \ + shuffleInc(i, 6, mask); \ + shuffleInc(i, 7, mask); + +#define reduceMatrix(mask) \ + reduceRow(0, mask); \ + reduceRow(1, mask); \ + reduceRow(2, mask); \ + reduceRow(3, mask); \ + reduceRow(4, mask); \ + reduceRow(5, mask); \ + reduceRow(6, mask); \ + reduceRow(7, mask); // actually perform the reduction, now each thread of index (_, y, z) // contains the correct values in its registers that belong in the output @@ -435,18 +444,17 @@ EigenContractionKernelInternal(const LhsMapper lhs, const RhsMapper rhs, // wait for shared mem to be out of use __syncthreads(); -#define writeResultShmem(i, j) \ - lhs_shmem[i + 8 * threadIdx.y + 64 * threadIdx.z + 512 * j] = res(i, j); \ +#define writeResultShmem(i, j) lhs_shmem[i + 8 * threadIdx.y + 64 * threadIdx.z + 512 * j] = res(i, j); -#define writeRow(i) \ - writeResultShmem(i, 0); \ - writeResultShmem(i, 1); \ - writeResultShmem(i, 2); \ - writeResultShmem(i, 3); \ - writeResultShmem(i, 4); \ - writeResultShmem(i, 5); \ - writeResultShmem(i, 6); \ - writeResultShmem(i, 7); \ +#define writeRow(i) \ + writeResultShmem(i, 0); \ + writeResultShmem(i, 1); \ + writeResultShmem(i, 2); \ + writeResultShmem(i, 3); \ + writeResultShmem(i, 4); \ + writeResultShmem(i, 5); \ + writeResultShmem(i, 6); \ + writeResultShmem(i, 7); if (threadIdx.x == 0) { writeRow(0); @@ -496,13 +504,14 @@ EigenContractionKernelInternal(const LhsMapper lhs, const RhsMapper rhs, } -template -__global__ void -__launch_bounds__(512) -EigenContractionKernel(const LhsMapper lhs, const RhsMapper rhs, - const OutputMapper output, - const Index m_size, const Index n_size, const Index k_size) { +template +__global__ void __launch_bounds__(512) EigenContractionKernel(const LhsMapper lhs, + const RhsMapper rhs, + const OutputMapper output, + const Index m_size, + const Index n_size, + const Index k_size) +{ __shared__ Scalar lhs_shmem[72 * 64]; __shared__ Scalar rhs_shmem[72 * 64]; @@ -513,67 +522,73 @@ EigenContractionKernel(const LhsMapper lhs, const RhsMapper rhs, const Index base_n = 64 * n_block_idx; if (base_m + 63 < m_size && base_n + 63 < n_size) { - EigenContractionKernelInternal(lhs, rhs, output, lhs_shmem, rhs_shmem, m_size, n_size, k_size); + EigenContractionKernelInternal( + lhs, rhs, output, lhs_shmem, rhs_shmem, m_size, n_size, k_size); } else { - EigenContractionKernelInternal(lhs, rhs, output, lhs_shmem, rhs_shmem, m_size, n_size, k_size); + EigenContractionKernelInternal( + lhs, rhs, output, lhs_shmem, rhs_shmem, m_size, n_size, k_size); } } -template -__device__ EIGEN_STRONG_INLINE void -EigenFloatContractionKernelInternal16x16(const LhsMapper lhs, const RhsMapper rhs, - const OutputMapper output, float2 lhs_shmem2[][16], - float2 rhs_shmem2[][8], const Index m_size, - const Index n_size, const Index k_size, - const Index base_m, const Index base_n) { +template +__device__ EIGEN_STRONG_INLINE void EigenFloatContractionKernelInternal16x16(const LhsMapper lhs, + const RhsMapper rhs, + const OutputMapper output, + float2 lhs_shmem2[][16], + float2 rhs_shmem2[][8], + const Index m_size, + const Index n_size, + const Index k_size, + const Index base_m, + const Index base_n) +{ typedef float Scalar; // prefetch registers float4 lhs_pf0, rhs_pf0; float4 results[4]; - for (int i=0; i < 4; i++) { - results[i].x = results[i].y = results[i].z = results[i].w = 0; + for (int i = 0; i < 4; i++) { results[i].x = results[i].y = results[i].z = results[i].w = 0; } + + +#define prefetch_lhs(reg, row, col) \ + if (!CHECK_LHS_BOUNDARY) { \ + if (col < k_size) { reg = lhs.loadPacket(row, col); } \ + } else { \ + if (col < k_size) { \ + if (row + 3 < m_size) { \ + reg = lhs.loadPacket(row, col); \ + } else if (row + 2 < m_size) { \ + reg.x = lhs(row + 0, col); \ + reg.y = lhs(row + 1, col); \ + reg.z = lhs(row + 2, col); \ + } else if (row + 1 < m_size) { \ + reg.x = lhs(row + 0, col); \ + reg.y = lhs(row + 1, col); \ + } else if (row < m_size) { \ + reg.x = lhs(row + 0, col); \ + } \ + } \ } -#define prefetch_lhs(reg, row, col) \ - if (!CHECK_LHS_BOUNDARY) { \ - if (col < k_size) { \ - reg =lhs.loadPacket(row, col); \ - } \ - } else { \ - if (col < k_size) { \ - if (row + 3 < m_size) { \ - reg =lhs.loadPacket(row, col); \ - } else if (row + 2 < m_size) { \ - reg.x =lhs(row + 0, col); \ - reg.y =lhs(row + 1, col); \ - reg.z =lhs(row + 2, col); \ - } else if (row + 1 < m_size) { \ - reg.x =lhs(row + 0, col); \ - reg.y =lhs(row + 1, col); \ - } else if (row < m_size) { \ - reg.x =lhs(row + 0, col); \ - } \ - } \ - } \ - - - Index lhs_vert = base_m+threadIdx.x*4; + Index lhs_vert = base_m + threadIdx.x * 4; for (Index k = 0; k < k_size; k += 16) { lhs_pf0 = internal::pset1(0); rhs_pf0 = internal::pset1(0); - Index lhs_horiz = threadIdx.y+k; + Index lhs_horiz = threadIdx.y + k; prefetch_lhs(lhs_pf0, lhs_vert, lhs_horiz) - Index rhs_vert = k+(threadIdx.x%4)*4; - Index rhs_horiz0 = (threadIdx.x>>2)+threadIdx.y*4+base_n; + Index rhs_vert = k + (threadIdx.x % 4) * 4; + Index rhs_horiz0 = (threadIdx.x >> 2) + threadIdx.y * 4 + base_n; if (!CHECK_RHS_BOUNDARY) { if ((rhs_vert + 3) < k_size) { @@ -587,7 +602,7 @@ EigenFloatContractionKernelInternal16x16(const LhsMapper lhs, const RhsMapper rh } else if (rhs_vert + 1 < k_size) { rhs_pf0.x = rhs(rhs_vert, rhs_horiz0); rhs_pf0.y = rhs(rhs_vert + 1, rhs_horiz0); - } else if (rhs_vert < k_size) { + } else if (rhs_vert < k_size) { rhs_pf0.x = rhs(rhs_vert, rhs_horiz0); } } else { @@ -601,14 +616,14 @@ EigenFloatContractionKernelInternal16x16(const LhsMapper lhs, const RhsMapper rh } else if ((rhs_vert + 1) < k_size) { rhs_pf0.x = rhs(rhs_vert, rhs_horiz0); rhs_pf0.y = rhs(rhs_vert + 1, rhs_horiz0); - } else if (rhs_vert < k_size) { + } else if (rhs_vert < k_size) { rhs_pf0.x = rhs(rhs_vert, rhs_horiz0); } } } - float x1, x2 ; + float x1, x2; // the following can be a bitwise operation..... some day. - if((threadIdx.x%8) < 4) { + if ((threadIdx.x % 8) < 4) { x1 = rhs_pf0.y; x2 = rhs_pf0.w; } else { @@ -617,7 +632,7 @@ EigenFloatContractionKernelInternal16x16(const LhsMapper lhs, const RhsMapper rh } x1 = __shfl_xor(x1, 4); x2 = __shfl_xor(x2, 4); - if((threadIdx.x%8) < 4) { + if ((threadIdx.x % 8) < 4) { rhs_pf0.y = x1; rhs_pf0.w = x2; } else { @@ -632,8 +647,8 @@ EigenFloatContractionKernelInternal16x16(const LhsMapper lhs, const RhsMapper rh // Row 31 -> times (0, 4, 8, 12, 1, 5, 9, 13) for features 62, 63 // Row 32 -> times (2, 6, 10, 14, 3, 7, 11, 15) for features 0, 1 // ... - rhs_shmem2[(threadIdx.x>>3)+ threadIdx.y*2][threadIdx.x%8] = make_float2(rhs_pf0.x, rhs_pf0.y); - rhs_shmem2[(threadIdx.x>>3)+ threadIdx.y*2+32][threadIdx.x%8] = make_float2(rhs_pf0.z, rhs_pf0.w); + rhs_shmem2[(threadIdx.x >> 3) + threadIdx.y * 2][threadIdx.x % 8] = make_float2(rhs_pf0.x, rhs_pf0.y); + rhs_shmem2[(threadIdx.x >> 3) + threadIdx.y * 2 + 32][threadIdx.x % 8] = make_float2(rhs_pf0.z, rhs_pf0.w); // Row 0 (time 0) -> features (0, 1), (4, 5), .. (28, 29), (32, 33), .. (60, 61) // Row 1 (time 1) -> features (0, 1), (4, 5), .. (28, 29), (32, 33), .. (60, 61) @@ -643,42 +658,42 @@ EigenFloatContractionKernelInternal16x16(const LhsMapper lhs, const RhsMapper rh // ... lhs_shmem2[threadIdx.y][threadIdx.x] = make_float2(lhs_pf0.x, lhs_pf0.y); - lhs_shmem2[threadIdx.y+16][threadIdx.x] = make_float2(lhs_pf0.z, lhs_pf0.w); - - -#define add_vals(fl1, fl2, fr1, fr2)\ - results[0].x += fl1.x * fr1.x;\ - results[0].y += fl1.y * fr1.x;\ - results[0].z += fl2.x * fr1.x;\ - results[0].w += fl2.y * fr1.x;\ -\ - results[1].x += fl1.x * fr1.y;\ - results[1].y += fl1.y * fr1.y;\ - results[1].z += fl2.x * fr1.y;\ - results[1].w += fl2.y * fr1.y;\ -\ - results[2].x += fl1.x * fr2.x;\ - results[2].y += fl1.y * fr2.x;\ - results[2].z += fl2.x * fr2.x;\ - results[2].w += fl2.y * fr2.x;\ -\ - results[3].x += fl1.x * fr2.y;\ - results[3].y += fl1.y * fr2.y;\ - results[3].z += fl2.x * fr2.y;\ - results[3].w += fl2.y * fr2.y;\ + lhs_shmem2[threadIdx.y + 16][threadIdx.x] = make_float2(lhs_pf0.z, lhs_pf0.w); + + +#define add_vals(fl1, fl2, fr1, fr2) \ + results[0].x += fl1.x * fr1.x; \ + results[0].y += fl1.y * fr1.x; \ + results[0].z += fl2.x * fr1.x; \ + results[0].w += fl2.y * fr1.x; \ + \ + results[1].x += fl1.x * fr1.y; \ + results[1].y += fl1.y * fr1.y; \ + results[1].z += fl2.x * fr1.y; \ + results[1].w += fl2.y * fr1.y; \ + \ + results[2].x += fl1.x * fr2.x; \ + results[2].y += fl1.y * fr2.x; \ + results[2].z += fl2.x * fr2.x; \ + results[2].w += fl2.y * fr2.x; \ + \ + results[3].x += fl1.x * fr2.y; \ + results[3].y += fl1.y * fr2.y; \ + results[3].z += fl2.x * fr2.y; \ + results[3].w += fl2.y * fr2.y; __syncthreads(); - // Do the multiplies. - #pragma unroll - for (int koff = 0; koff < 16; koff ++) { +// Do the multiplies. +#pragma unroll + for (int koff = 0; koff < 16; koff++) { // 32 x threads. float2 fl1 = lhs_shmem2[koff][threadIdx.x]; float2 fl2 = lhs_shmem2[koff + 16][threadIdx.x]; int start_feature = threadIdx.y * 4; - float2 fr1 = rhs_shmem2[(start_feature>>1) + 32*((koff%4)/2)][koff/4 + (koff%2)*4]; - float2 fr2 = rhs_shmem2[(start_feature>>1) + 1 + 32*((koff%4)/2)][koff/4 + (koff%2)*4]; + float2 fr1 = rhs_shmem2[(start_feature >> 1) + 32 * ((koff % 4) / 2)][koff / 4 + (koff % 2) * 4]; + float2 fr2 = rhs_shmem2[(start_feature >> 1) + 1 + 32 * ((koff % 4) / 2)][koff / 4 + (koff % 2) * 4]; add_vals(fl1, fl2, fr1, fr2) } @@ -688,7 +703,7 @@ EigenFloatContractionKernelInternal16x16(const LhsMapper lhs, const RhsMapper rh #undef prefetch_lhs #undef add_vals - Index horiz_base = threadIdx.y*4+base_n; + Index horiz_base = threadIdx.y * 4 + base_n; if (!CHECK_LHS_BOUNDARY && !CHECK_RHS_BOUNDARY) { for (int i = 0; i < 4; i++) { output(lhs_vert, horiz_base + i) = results[i].x; @@ -716,10 +731,8 @@ EigenFloatContractionKernelInternal16x16(const LhsMapper lhs, const RhsMapper rh output(lhs_vert, horiz_base + i) = results[i].x; output(lhs_vert + 1, horiz_base + i) = results[i].y; } - } else if (lhs_vert < m_size) { - for (int i = 0; i < 4; i++) { - output(lhs_vert, horiz_base + i) = results[i].x; - } + } else if (lhs_vert < m_size) { + for (int i = 0; i < 4; i++) { output(lhs_vert, horiz_base + i) = results[i].x; } } } else if (!CHECK_LHS_BOUNDARY) { // CHECK RHS @@ -732,40 +745,44 @@ EigenFloatContractionKernelInternal16x16(const LhsMapper lhs, const RhsMapper rh output(lhs_vert + 3, horiz_base + i) = results[i].w; }*/ for (int i = 0; i < 4; i++) { - if (horiz_base+i < n_size) { + if (horiz_base + i < n_size) { output(lhs_vert, horiz_base + i) = results[i].x; output(lhs_vert + 1, horiz_base + i) = results[i].y; output(lhs_vert + 2, horiz_base + i) = results[i].z; output(lhs_vert + 3, horiz_base + i) = results[i].w; - } + } } } else { // CHECK both boundaries. for (int i = 0; i < 4; i++) { - if (horiz_base+i < n_size) { - if (lhs_vert < m_size) - output(lhs_vert, horiz_base + i) = results[i].x; - if (lhs_vert + 1 < m_size) - output(lhs_vert + 1, horiz_base + i) = results[i].y; - if (lhs_vert + 2 < m_size) - output(lhs_vert + 2, horiz_base + i) = results[i].z; - if (lhs_vert + 3 < m_size) - output(lhs_vert + 3, horiz_base + i) = results[i].w; + if (horiz_base + i < n_size) { + if (lhs_vert < m_size) output(lhs_vert, horiz_base + i) = results[i].x; + if (lhs_vert + 1 < m_size) output(lhs_vert + 1, horiz_base + i) = results[i].y; + if (lhs_vert + 2 < m_size) output(lhs_vert + 2, horiz_base + i) = results[i].z; + if (lhs_vert + 3 < m_size) output(lhs_vert + 3, horiz_base + i) = results[i].w; } } } } -template -__device__ EIGEN_STRONG_INLINE void -EigenFloatContractionKernelInternal(const LhsMapper lhs, const RhsMapper rhs, - const OutputMapper output, float2 lhs_shmem2[][32], - float2 rhs_shmem2[][8], const Index m_size, - const Index n_size, const Index k_size, - const Index base_m, const Index base_n) { +template +__device__ EIGEN_STRONG_INLINE void EigenFloatContractionKernelInternal(const LhsMapper lhs, + const RhsMapper rhs, + const OutputMapper output, + float2 lhs_shmem2[][32], + float2 rhs_shmem2[][8], + const Index m_size, + const Index n_size, + const Index k_size, + const Index base_m, + const Index base_n) +{ typedef float Scalar; // prefetch registers @@ -773,12 +790,10 @@ EigenFloatContractionKernelInternal(const LhsMapper lhs, const RhsMapper rhs, float4 rhs_pf0, rhs_pf1; float4 results[8]; - for (int i=0; i < 8; i++) { - results[i].x = results[i].y = results[i].z = results[i].w = 0; - } + for (int i = 0; i < 8; i++) { results[i].x = results[i].y = results[i].z = results[i].w = 0; } - Index lhs_vert = base_m+threadIdx.x*4+(threadIdx.y%4)*32; + Index lhs_vert = base_m + threadIdx.x * 4 + (threadIdx.y % 4) * 32; for (Index k = 0; k < k_size; k += 32) { lhs_pf0 = internal::pset1(0); lhs_pf1 = internal::pset1(0); @@ -788,124 +803,124 @@ EigenFloatContractionKernelInternal(const LhsMapper lhs, const RhsMapper rhs, rhs_pf0 = internal::pset1(0); rhs_pf1 = internal::pset1(0); - if (!CHECK_LHS_BOUNDARY) { - if ((threadIdx.y/4+k+24) < k_size) { - lhs_pf0 =lhs.loadPacket(lhs_vert, (threadIdx.y/4+k)); - lhs_pf1 =lhs.loadPacket(lhs_vert, (threadIdx.y/4+k+8)); - lhs_pf2 =lhs.loadPacket(lhs_vert, (threadIdx.y/4+k+16)); - lhs_pf3 =lhs.loadPacket(lhs_vert, (threadIdx.y/4+k+24)); - } else if ((threadIdx.y/4+k+16) < k_size) { - lhs_pf0 =lhs.loadPacket(lhs_vert, (threadIdx.y/4+k)); - lhs_pf1 =lhs.loadPacket(lhs_vert, (threadIdx.y/4+k+8)); - lhs_pf2 =lhs.loadPacket(lhs_vert, (threadIdx.y/4+k+16)); - } else if ((threadIdx.y/4+k+8) < k_size) { - lhs_pf0 =lhs.loadPacket(lhs_vert, (threadIdx.y/4+k)); - lhs_pf1 =lhs.loadPacket(lhs_vert, (threadIdx.y/4+k+8)); - } else if ((threadIdx.y/4+k) < k_size) { - lhs_pf0 =lhs.loadPacket(lhs_vert, (threadIdx.y/4+k)); + if (!CHECK_LHS_BOUNDARY) { + if ((threadIdx.y / 4 + k + 24) < k_size) { + lhs_pf0 = lhs.loadPacket(lhs_vert, (threadIdx.y / 4 + k)); + lhs_pf1 = lhs.loadPacket(lhs_vert, (threadIdx.y / 4 + k + 8)); + lhs_pf2 = lhs.loadPacket(lhs_vert, (threadIdx.y / 4 + k + 16)); + lhs_pf3 = lhs.loadPacket(lhs_vert, (threadIdx.y / 4 + k + 24)); + } else if ((threadIdx.y / 4 + k + 16) < k_size) { + lhs_pf0 = lhs.loadPacket(lhs_vert, (threadIdx.y / 4 + k)); + lhs_pf1 = lhs.loadPacket(lhs_vert, (threadIdx.y / 4 + k + 8)); + lhs_pf2 = lhs.loadPacket(lhs_vert, (threadIdx.y / 4 + k + 16)); + } else if ((threadIdx.y / 4 + k + 8) < k_size) { + lhs_pf0 = lhs.loadPacket(lhs_vert, (threadIdx.y / 4 + k)); + lhs_pf1 = lhs.loadPacket(lhs_vert, (threadIdx.y / 4 + k + 8)); + } else if ((threadIdx.y / 4 + k) < k_size) { + lhs_pf0 = lhs.loadPacket(lhs_vert, (threadIdx.y / 4 + k)); } } else { // just CHECK_LHS_BOUNDARY if (lhs_vert + 3 < m_size) { - if ((threadIdx.y/4+k+24) < k_size) { - lhs_pf0 =lhs.loadPacket(lhs_vert, (threadIdx.y/4+k)); - lhs_pf1 =lhs.loadPacket(lhs_vert, (threadIdx.y/4+k+8)); - lhs_pf2 =lhs.loadPacket(lhs_vert, (threadIdx.y/4+k+16)); - lhs_pf3 =lhs.loadPacket(lhs_vert, (threadIdx.y/4+k+24)); - } else if ((threadIdx.y/4+k+16) < k_size) { - lhs_pf0 =lhs.loadPacket(lhs_vert, (threadIdx.y/4+k)); - lhs_pf1 =lhs.loadPacket(lhs_vert, (threadIdx.y/4+k+8)); - lhs_pf2 =lhs.loadPacket(lhs_vert, (threadIdx.y/4+k+16)); - } else if ((threadIdx.y/4+k+8) < k_size) { - lhs_pf0 =lhs.loadPacket(lhs_vert, (threadIdx.y/4+k)); - lhs_pf1 =lhs.loadPacket(lhs_vert, (threadIdx.y/4+k+8)); - } else if ((threadIdx.y/4+k) < k_size) { - lhs_pf0 =lhs.loadPacket(lhs_vert, (threadIdx.y/4+k)); + if ((threadIdx.y / 4 + k + 24) < k_size) { + lhs_pf0 = lhs.loadPacket(lhs_vert, (threadIdx.y / 4 + k)); + lhs_pf1 = lhs.loadPacket(lhs_vert, (threadIdx.y / 4 + k + 8)); + lhs_pf2 = lhs.loadPacket(lhs_vert, (threadIdx.y / 4 + k + 16)); + lhs_pf3 = lhs.loadPacket(lhs_vert, (threadIdx.y / 4 + k + 24)); + } else if ((threadIdx.y / 4 + k + 16) < k_size) { + lhs_pf0 = lhs.loadPacket(lhs_vert, (threadIdx.y / 4 + k)); + lhs_pf1 = lhs.loadPacket(lhs_vert, (threadIdx.y / 4 + k + 8)); + lhs_pf2 = lhs.loadPacket(lhs_vert, (threadIdx.y / 4 + k + 16)); + } else if ((threadIdx.y / 4 + k + 8) < k_size) { + lhs_pf0 = lhs.loadPacket(lhs_vert, (threadIdx.y / 4 + k)); + lhs_pf1 = lhs.loadPacket(lhs_vert, (threadIdx.y / 4 + k + 8)); + } else if ((threadIdx.y / 4 + k) < k_size) { + lhs_pf0 = lhs.loadPacket(lhs_vert, (threadIdx.y / 4 + k)); } } else if (lhs_vert + 2 < m_size) { - if ((threadIdx.y/4+k+24) < k_size) { - lhs_pf0.x =lhs(lhs_vert + 0, (threadIdx.y/4+k)); - lhs_pf0.y =lhs(lhs_vert + 1, (threadIdx.y/4+k)); - lhs_pf0.z =lhs(lhs_vert + 2, (threadIdx.y/4+k)); - lhs_pf1.x =lhs(lhs_vert + 0, (threadIdx.y/4+k+8)); - lhs_pf1.y =lhs(lhs_vert + 1, (threadIdx.y/4+k+8)); - lhs_pf1.z =lhs(lhs_vert + 2, (threadIdx.y/4+k+8)); - lhs_pf2.x =lhs(lhs_vert + 0, (threadIdx.y/4+k+16)); - lhs_pf2.y =lhs(lhs_vert + 1, (threadIdx.y/4+k+16)); - lhs_pf2.z =lhs(lhs_vert + 2, (threadIdx.y/4+k+16)); - lhs_pf3.x =lhs(lhs_vert + 0, (threadIdx.y/4+k+24)); - lhs_pf3.y =lhs(lhs_vert + 1, (threadIdx.y/4+k+24)); - lhs_pf3.z =lhs(lhs_vert + 2, (threadIdx.y/4+k+24)); - } else if ((threadIdx.y/4+k+16) < k_size) { - lhs_pf0.x =lhs(lhs_vert + 0, (threadIdx.y/4+k)); - lhs_pf0.y =lhs(lhs_vert + 1, (threadIdx.y/4+k)); - lhs_pf0.z =lhs(lhs_vert + 2, (threadIdx.y/4+k)); - lhs_pf1.x =lhs(lhs_vert + 0, (threadIdx.y/4+k+8)); - lhs_pf1.y =lhs(lhs_vert + 1, (threadIdx.y/4+k+8)); - lhs_pf1.z =lhs(lhs_vert + 2, (threadIdx.y/4+k+8)); - lhs_pf2.x =lhs(lhs_vert + 0, (threadIdx.y/4+k+16)); - lhs_pf2.y =lhs(lhs_vert + 1, (threadIdx.y/4+k+16)); - lhs_pf2.z =lhs(lhs_vert + 2, (threadIdx.y/4+k+16)); - } else if ((threadIdx.y/4+k+8) < k_size) { - lhs_pf0.x =lhs(lhs_vert + 0, (threadIdx.y/4+k)); - lhs_pf0.y =lhs(lhs_vert + 1, (threadIdx.y/4+k)); - lhs_pf0.z =lhs(lhs_vert + 2, (threadIdx.y/4+k)); - lhs_pf1.x =lhs(lhs_vert + 0, (threadIdx.y/4+k+8)); - lhs_pf1.y =lhs(lhs_vert + 1, (threadIdx.y/4+k+8)); - lhs_pf1.z =lhs(lhs_vert + 2, (threadIdx.y/4+k+8)); - } else if ((threadIdx.y/4+k) < k_size) { - lhs_pf0.x =lhs(lhs_vert + 0, (threadIdx.y/4+k)); - lhs_pf0.y =lhs(lhs_vert + 1, (threadIdx.y/4+k)); - lhs_pf0.z =lhs(lhs_vert + 2, (threadIdx.y/4+k)); + if ((threadIdx.y / 4 + k + 24) < k_size) { + lhs_pf0.x = lhs(lhs_vert + 0, (threadIdx.y / 4 + k)); + lhs_pf0.y = lhs(lhs_vert + 1, (threadIdx.y / 4 + k)); + lhs_pf0.z = lhs(lhs_vert + 2, (threadIdx.y / 4 + k)); + lhs_pf1.x = lhs(lhs_vert + 0, (threadIdx.y / 4 + k + 8)); + lhs_pf1.y = lhs(lhs_vert + 1, (threadIdx.y / 4 + k + 8)); + lhs_pf1.z = lhs(lhs_vert + 2, (threadIdx.y / 4 + k + 8)); + lhs_pf2.x = lhs(lhs_vert + 0, (threadIdx.y / 4 + k + 16)); + lhs_pf2.y = lhs(lhs_vert + 1, (threadIdx.y / 4 + k + 16)); + lhs_pf2.z = lhs(lhs_vert + 2, (threadIdx.y / 4 + k + 16)); + lhs_pf3.x = lhs(lhs_vert + 0, (threadIdx.y / 4 + k + 24)); + lhs_pf3.y = lhs(lhs_vert + 1, (threadIdx.y / 4 + k + 24)); + lhs_pf3.z = lhs(lhs_vert + 2, (threadIdx.y / 4 + k + 24)); + } else if ((threadIdx.y / 4 + k + 16) < k_size) { + lhs_pf0.x = lhs(lhs_vert + 0, (threadIdx.y / 4 + k)); + lhs_pf0.y = lhs(lhs_vert + 1, (threadIdx.y / 4 + k)); + lhs_pf0.z = lhs(lhs_vert + 2, (threadIdx.y / 4 + k)); + lhs_pf1.x = lhs(lhs_vert + 0, (threadIdx.y / 4 + k + 8)); + lhs_pf1.y = lhs(lhs_vert + 1, (threadIdx.y / 4 + k + 8)); + lhs_pf1.z = lhs(lhs_vert + 2, (threadIdx.y / 4 + k + 8)); + lhs_pf2.x = lhs(lhs_vert + 0, (threadIdx.y / 4 + k + 16)); + lhs_pf2.y = lhs(lhs_vert + 1, (threadIdx.y / 4 + k + 16)); + lhs_pf2.z = lhs(lhs_vert + 2, (threadIdx.y / 4 + k + 16)); + } else if ((threadIdx.y / 4 + k + 8) < k_size) { + lhs_pf0.x = lhs(lhs_vert + 0, (threadIdx.y / 4 + k)); + lhs_pf0.y = lhs(lhs_vert + 1, (threadIdx.y / 4 + k)); + lhs_pf0.z = lhs(lhs_vert + 2, (threadIdx.y / 4 + k)); + lhs_pf1.x = lhs(lhs_vert + 0, (threadIdx.y / 4 + k + 8)); + lhs_pf1.y = lhs(lhs_vert + 1, (threadIdx.y / 4 + k + 8)); + lhs_pf1.z = lhs(lhs_vert + 2, (threadIdx.y / 4 + k + 8)); + } else if ((threadIdx.y / 4 + k) < k_size) { + lhs_pf0.x = lhs(lhs_vert + 0, (threadIdx.y / 4 + k)); + lhs_pf0.y = lhs(lhs_vert + 1, (threadIdx.y / 4 + k)); + lhs_pf0.z = lhs(lhs_vert + 2, (threadIdx.y / 4 + k)); } } else if (lhs_vert + 1 < m_size) { - if ((threadIdx.y/4+k+24) < k_size) { - lhs_pf0.x =lhs(lhs_vert + 0, (threadIdx.y/4+k)); - lhs_pf0.y =lhs(lhs_vert + 1, (threadIdx.y/4+k)); - lhs_pf1.x =lhs(lhs_vert + 0, (threadIdx.y/4+k+8)); - lhs_pf1.y =lhs(lhs_vert + 1, (threadIdx.y/4+k+8)); - lhs_pf2.x =lhs(lhs_vert + 0, (threadIdx.y/4+k+16)); - lhs_pf2.y =lhs(lhs_vert + 1, (threadIdx.y/4+k+16)); - lhs_pf3.x =lhs(lhs_vert + 0, (threadIdx.y/4+k+24)); - lhs_pf3.y =lhs(lhs_vert + 1, (threadIdx.y/4+k+24)); - } else if ((threadIdx.y/4+k+16) < k_size) { - lhs_pf0.x =lhs(lhs_vert + 0, (threadIdx.y/4+k)); - lhs_pf0.y =lhs(lhs_vert + 1, (threadIdx.y/4+k)); - lhs_pf1.x =lhs(lhs_vert + 0, (threadIdx.y/4+k+8)); - lhs_pf1.y =lhs(lhs_vert + 1, (threadIdx.y/4+k+8)); - lhs_pf2.x =lhs(lhs_vert + 0, (threadIdx.y/4+k+16)); - lhs_pf2.y =lhs(lhs_vert + 1, (threadIdx.y/4+k+16)); - } else if ((threadIdx.y/4+k+8) < k_size) { - lhs_pf0.x =lhs(lhs_vert + 0, (threadIdx.y/4+k)); - lhs_pf0.y =lhs(lhs_vert + 1, (threadIdx.y/4+k)); - lhs_pf1.x =lhs(lhs_vert + 0, (threadIdx.y/4+k+8)); - lhs_pf1.y =lhs(lhs_vert + 1, (threadIdx.y/4+k+8)); - } else if ((threadIdx.y/4+k) < k_size) { - lhs_pf0.x =lhs(lhs_vert + 0, (threadIdx.y/4+k)); - lhs_pf0.y =lhs(lhs_vert + 1, (threadIdx.y/4+k)); + if ((threadIdx.y / 4 + k + 24) < k_size) { + lhs_pf0.x = lhs(lhs_vert + 0, (threadIdx.y / 4 + k)); + lhs_pf0.y = lhs(lhs_vert + 1, (threadIdx.y / 4 + k)); + lhs_pf1.x = lhs(lhs_vert + 0, (threadIdx.y / 4 + k + 8)); + lhs_pf1.y = lhs(lhs_vert + 1, (threadIdx.y / 4 + k + 8)); + lhs_pf2.x = lhs(lhs_vert + 0, (threadIdx.y / 4 + k + 16)); + lhs_pf2.y = lhs(lhs_vert + 1, (threadIdx.y / 4 + k + 16)); + lhs_pf3.x = lhs(lhs_vert + 0, (threadIdx.y / 4 + k + 24)); + lhs_pf3.y = lhs(lhs_vert + 1, (threadIdx.y / 4 + k + 24)); + } else if ((threadIdx.y / 4 + k + 16) < k_size) { + lhs_pf0.x = lhs(lhs_vert + 0, (threadIdx.y / 4 + k)); + lhs_pf0.y = lhs(lhs_vert + 1, (threadIdx.y / 4 + k)); + lhs_pf1.x = lhs(lhs_vert + 0, (threadIdx.y / 4 + k + 8)); + lhs_pf1.y = lhs(lhs_vert + 1, (threadIdx.y / 4 + k + 8)); + lhs_pf2.x = lhs(lhs_vert + 0, (threadIdx.y / 4 + k + 16)); + lhs_pf2.y = lhs(lhs_vert + 1, (threadIdx.y / 4 + k + 16)); + } else if ((threadIdx.y / 4 + k + 8) < k_size) { + lhs_pf0.x = lhs(lhs_vert + 0, (threadIdx.y / 4 + k)); + lhs_pf0.y = lhs(lhs_vert + 1, (threadIdx.y / 4 + k)); + lhs_pf1.x = lhs(lhs_vert + 0, (threadIdx.y / 4 + k + 8)); + lhs_pf1.y = lhs(lhs_vert + 1, (threadIdx.y / 4 + k + 8)); + } else if ((threadIdx.y / 4 + k) < k_size) { + lhs_pf0.x = lhs(lhs_vert + 0, (threadIdx.y / 4 + k)); + lhs_pf0.y = lhs(lhs_vert + 1, (threadIdx.y / 4 + k)); } } else if (lhs_vert < m_size) { - if ((threadIdx.y/4+k+24) < k_size) { - lhs_pf0.x =lhs(lhs_vert + 0, (threadIdx.y/4+k)); - lhs_pf1.x =lhs(lhs_vert + 0, (threadIdx.y/4+k+8)); - lhs_pf2.x =lhs(lhs_vert + 0, (threadIdx.y/4+k+16)); - lhs_pf3.x =lhs(lhs_vert + 0, (threadIdx.y/4+k+24)); - } else if ((threadIdx.y/4+k+16) < k_size) { - lhs_pf0.x =lhs(lhs_vert + 0, (threadIdx.y/4+k)); - lhs_pf1.x =lhs(lhs_vert + 0, (threadIdx.y/4+k+8)); - lhs_pf2.x =lhs(lhs_vert + 0, (threadIdx.y/4+k+16)); - } else if ((threadIdx.y/4+k+8) < k_size) { - lhs_pf0.x =lhs(lhs_vert + 0, (threadIdx.y/4+k)); - lhs_pf1.x =lhs(lhs_vert + 0, (threadIdx.y/4+k+8)); - } else if ((threadIdx.y/4+k) < k_size) { - lhs_pf0.x =lhs(lhs_vert + 0, (threadIdx.y/4+k)); + if ((threadIdx.y / 4 + k + 24) < k_size) { + lhs_pf0.x = lhs(lhs_vert + 0, (threadIdx.y / 4 + k)); + lhs_pf1.x = lhs(lhs_vert + 0, (threadIdx.y / 4 + k + 8)); + lhs_pf2.x = lhs(lhs_vert + 0, (threadIdx.y / 4 + k + 16)); + lhs_pf3.x = lhs(lhs_vert + 0, (threadIdx.y / 4 + k + 24)); + } else if ((threadIdx.y / 4 + k + 16) < k_size) { + lhs_pf0.x = lhs(lhs_vert + 0, (threadIdx.y / 4 + k)); + lhs_pf1.x = lhs(lhs_vert + 0, (threadIdx.y / 4 + k + 8)); + lhs_pf2.x = lhs(lhs_vert + 0, (threadIdx.y / 4 + k + 16)); + } else if ((threadIdx.y / 4 + k + 8) < k_size) { + lhs_pf0.x = lhs(lhs_vert + 0, (threadIdx.y / 4 + k)); + lhs_pf1.x = lhs(lhs_vert + 0, (threadIdx.y / 4 + k + 8)); + } else if ((threadIdx.y / 4 + k) < k_size) { + lhs_pf0.x = lhs(lhs_vert + 0, (threadIdx.y / 4 + k)); } } } __syncthreads(); - Index rhs_vert = k+threadIdx.x*4; - Index rhs_horiz0 = threadIdx.y*2+base_n; - Index rhs_horiz1 = threadIdx.y*2+1+base_n; + Index rhs_vert = k + threadIdx.x * 4; + Index rhs_horiz0 = threadIdx.y * 2 + base_n; + Index rhs_horiz1 = threadIdx.y * 2 + 1 + base_n; if (!CHECK_RHS_BOUNDARY) { if ((rhs_vert + 3) < k_size) { // just CHECK_RHS_BOUNDARY @@ -924,7 +939,7 @@ EigenFloatContractionKernelInternal(const LhsMapper lhs, const RhsMapper rhs, rhs_pf0.y = rhs(rhs_vert + 1, rhs_horiz0); rhs_pf1.x = rhs(rhs_vert, rhs_horiz1); rhs_pf1.y = rhs(rhs_vert + 1, rhs_horiz1); - } else if (rhs_vert < k_size) { + } else if (rhs_vert < k_size) { rhs_pf0.x = rhs(rhs_vert, rhs_horiz0); rhs_pf1.x = rhs(rhs_vert, rhs_horiz1); } @@ -942,12 +957,12 @@ EigenFloatContractionKernelInternal(const LhsMapper lhs, const RhsMapper rhs, rhs_pf1.x = rhs(rhs_vert, rhs_horiz1); rhs_pf1.y = rhs(rhs_vert + 1, rhs_horiz1); rhs_pf1.z = rhs(rhs_vert + 2, rhs_horiz1); - } else if (k+threadIdx.x*4 + 1 < k_size) { + } else if (k + threadIdx.x * 4 + 1 < k_size) { rhs_pf0.x = rhs(rhs_vert, rhs_horiz0); rhs_pf0.y = rhs(rhs_vert + 1, rhs_horiz0); rhs_pf1.x = rhs(rhs_vert, rhs_horiz1); rhs_pf1.y = rhs(rhs_vert + 1, rhs_horiz1); - } else if (k+threadIdx.x*4 < k_size) { + } else if (k + threadIdx.x * 4 < k_size) { rhs_pf0.x = rhs(rhs_vert, rhs_horiz0); rhs_pf1.x = rhs(rhs_vert, rhs_horiz1); } @@ -963,7 +978,7 @@ EigenFloatContractionKernelInternal(const LhsMapper lhs, const RhsMapper rhs, } else if ((rhs_vert + 1) < k_size) { rhs_pf0.x = rhs(rhs_vert, rhs_horiz0); rhs_pf0.y = rhs(rhs_vert + 1, rhs_horiz0); - } else if (rhs_vert < k_size) { + } else if (rhs_vert < k_size) { rhs_pf0.x = rhs(rhs_vert, rhs_horiz0); } } @@ -978,13 +993,13 @@ EigenFloatContractionKernelInternal(const LhsMapper lhs, const RhsMapper rhs, // Row 32 -> times (1, 5, 9, .. 29) for features 0, 1. // Row 33 -> times (1, 5, 9, .. 29) for features 2, 3. // .. - rhs_shmem2[threadIdx.y+32][threadIdx.x] = make_float2(rhs_pf0.y, rhs_pf1.y); + rhs_shmem2[threadIdx.y + 32][threadIdx.x] = make_float2(rhs_pf0.y, rhs_pf1.y); // Row 64 -> times (2, 6, 10, .. 30) for features 0, 1. // Row 65 -> times (2, 6, 10, .. 30) for features 2, 3. - rhs_shmem2[threadIdx.y+64][threadIdx.x] = make_float2(rhs_pf0.z, rhs_pf1.z); + rhs_shmem2[threadIdx.y + 64][threadIdx.x] = make_float2(rhs_pf0.z, rhs_pf1.z); // Row 96 -> times (3, 7, 11, .. 31) for features 0, 1. // Row 97 -> times (3, 7, 11, .. 31) for features 2, 3. - rhs_shmem2[threadIdx.y+96][threadIdx.x] = make_float2(rhs_pf0.w, rhs_pf1.w); + rhs_shmem2[threadIdx.y + 96][threadIdx.x] = make_float2(rhs_pf0.w, rhs_pf1.w); // LHS. // Row 0 (time 0) -> features (0, 1), (4, 5), .. (28, 29), (32, 33), .. (60, 61) .. (124, 125) @@ -994,77 +1009,77 @@ EigenFloatContractionKernelInternal(const LhsMapper lhs, const RhsMapper rhs, // Row 15 (time 7) -> features (2, 3), (6, 7), .. (30, 31), (34, 35), .. (62, 63) .. (126, 127) -#define add_vals(a_feat1, a_feat2, f1, f2, f3, f4)\ - results[0].x += a_feat1.x * f1.x;\ - results[1].x += a_feat1.x * f1.y;\ - results[2].x += a_feat1.x * f2.x;\ - results[3].x += a_feat1.x * f2.y;\ - results[4].x += a_feat1.x * f3.x;\ - results[5].x += a_feat1.x * f3.y;\ - results[6].x += a_feat1.x * f4.x;\ - results[7].x += a_feat1.x * f4.y;\ -\ - results[0].y += a_feat1.y * f1.x;\ - results[1].y += a_feat1.y * f1.y;\ - results[2].y += a_feat1.y * f2.x;\ - results[3].y += a_feat1.y * f2.y;\ - results[4].y += a_feat1.y * f3.x;\ - results[5].y += a_feat1.y * f3.y;\ - results[6].y += a_feat1.y * f4.x;\ - results[7].y += a_feat1.y * f4.y;\ -\ - results[0].z += a_feat2.x * f1.x;\ - results[1].z += a_feat2.x * f1.y;\ - results[2].z += a_feat2.x * f2.x;\ - results[3].z += a_feat2.x * f2.y;\ - results[4].z += a_feat2.x * f3.x;\ - results[5].z += a_feat2.x * f3.y;\ - results[6].z += a_feat2.x * f4.x;\ - results[7].z += a_feat2.x * f4.y;\ -\ - results[0].w += a_feat2.y * f1.x;\ - results[1].w += a_feat2.y * f1.y;\ - results[2].w += a_feat2.y * f2.x;\ - results[3].w += a_feat2.y * f2.y;\ - results[4].w += a_feat2.y * f3.x;\ - results[5].w += a_feat2.y * f3.y;\ - results[6].w += a_feat2.y * f4.x;\ - results[7].w += a_feat2.y * f4.y;\ - - lhs_shmem2[threadIdx.y/4][threadIdx.x+(threadIdx.y%4)*8] = make_float2(lhs_pf0.x, lhs_pf0.y); - lhs_shmem2[threadIdx.y/4+8][threadIdx.x+(threadIdx.y%4)*8] = make_float2(lhs_pf1.x, lhs_pf1.y); - lhs_shmem2[threadIdx.y/4+16][threadIdx.x+(threadIdx.y%4)*8] = make_float2(lhs_pf2.x, lhs_pf2.y); - lhs_shmem2[threadIdx.y/4+24][threadIdx.x+(threadIdx.y%4)*8] = make_float2(lhs_pf3.x, lhs_pf3.y); - - lhs_shmem2[threadIdx.y/4 + 32][threadIdx.x+(threadIdx.y%4)*8] = make_float2(lhs_pf0.z, lhs_pf0.w); - lhs_shmem2[threadIdx.y/4 + 40][threadIdx.x+(threadIdx.y%4)*8] = make_float2(lhs_pf1.z, lhs_pf1.w); - lhs_shmem2[threadIdx.y/4 + 48][threadIdx.x+(threadIdx.y%4)*8] = make_float2(lhs_pf2.z, lhs_pf2.w); - lhs_shmem2[threadIdx.y/4 + 56][threadIdx.x+(threadIdx.y%4)*8] = make_float2(lhs_pf3.z, lhs_pf3.w); +#define add_vals(a_feat1, a_feat2, f1, f2, f3, f4) \ + results[0].x += a_feat1.x * f1.x; \ + results[1].x += a_feat1.x * f1.y; \ + results[2].x += a_feat1.x * f2.x; \ + results[3].x += a_feat1.x * f2.y; \ + results[4].x += a_feat1.x * f3.x; \ + results[5].x += a_feat1.x * f3.y; \ + results[6].x += a_feat1.x * f4.x; \ + results[7].x += a_feat1.x * f4.y; \ + \ + results[0].y += a_feat1.y * f1.x; \ + results[1].y += a_feat1.y * f1.y; \ + results[2].y += a_feat1.y * f2.x; \ + results[3].y += a_feat1.y * f2.y; \ + results[4].y += a_feat1.y * f3.x; \ + results[5].y += a_feat1.y * f3.y; \ + results[6].y += a_feat1.y * f4.x; \ + results[7].y += a_feat1.y * f4.y; \ + \ + results[0].z += a_feat2.x * f1.x; \ + results[1].z += a_feat2.x * f1.y; \ + results[2].z += a_feat2.x * f2.x; \ + results[3].z += a_feat2.x * f2.y; \ + results[4].z += a_feat2.x * f3.x; \ + results[5].z += a_feat2.x * f3.y; \ + results[6].z += a_feat2.x * f4.x; \ + results[7].z += a_feat2.x * f4.y; \ + \ + results[0].w += a_feat2.y * f1.x; \ + results[1].w += a_feat2.y * f1.y; \ + results[2].w += a_feat2.y * f2.x; \ + results[3].w += a_feat2.y * f2.y; \ + results[4].w += a_feat2.y * f3.x; \ + results[5].w += a_feat2.y * f3.y; \ + results[6].w += a_feat2.y * f4.x; \ + results[7].w += a_feat2.y * f4.y; + + lhs_shmem2[threadIdx.y / 4][threadIdx.x + (threadIdx.y % 4) * 8] = make_float2(lhs_pf0.x, lhs_pf0.y); + lhs_shmem2[threadIdx.y / 4 + 8][threadIdx.x + (threadIdx.y % 4) * 8] = make_float2(lhs_pf1.x, lhs_pf1.y); + lhs_shmem2[threadIdx.y / 4 + 16][threadIdx.x + (threadIdx.y % 4) * 8] = make_float2(lhs_pf2.x, lhs_pf2.y); + lhs_shmem2[threadIdx.y / 4 + 24][threadIdx.x + (threadIdx.y % 4) * 8] = make_float2(lhs_pf3.x, lhs_pf3.y); + + lhs_shmem2[threadIdx.y / 4 + 32][threadIdx.x + (threadIdx.y % 4) * 8] = make_float2(lhs_pf0.z, lhs_pf0.w); + lhs_shmem2[threadIdx.y / 4 + 40][threadIdx.x + (threadIdx.y % 4) * 8] = make_float2(lhs_pf1.z, lhs_pf1.w); + lhs_shmem2[threadIdx.y / 4 + 48][threadIdx.x + (threadIdx.y % 4) * 8] = make_float2(lhs_pf2.z, lhs_pf2.w); + lhs_shmem2[threadIdx.y / 4 + 56][threadIdx.x + (threadIdx.y % 4) * 8] = make_float2(lhs_pf3.z, lhs_pf3.w); __syncthreads(); - // Do the multiplies. - #pragma unroll - for (int koff = 0; koff < 32; koff ++) { +// Do the multiplies. +#pragma unroll + for (int koff = 0; koff < 32; koff++) { float2 a3 = lhs_shmem2[koff][threadIdx.x + (threadIdx.y % 4) * 8]; float2 a4 = lhs_shmem2[koff + 32][threadIdx.x + (threadIdx.y % 4) * 8]; // first feature is at (threadIdx.y/4) * 8 last is at start + 8. int start_feature = (threadIdx.y / 4) * 8; - float2 br1 = rhs_shmem2[start_feature/2 + (koff % 4) * 32][koff/4]; - float2 br2 = rhs_shmem2[start_feature/2 + 1 + (koff % 4) * 32][koff/4]; - float2 br3 = rhs_shmem2[start_feature/2 + 2 + (koff % 4) * 32][koff/4]; - float2 br4 = rhs_shmem2[start_feature/2 + 3 + (koff % 4) * 32][koff/4]; + float2 br1 = rhs_shmem2[start_feature / 2 + (koff % 4) * 32][koff / 4]; + float2 br2 = rhs_shmem2[start_feature / 2 + 1 + (koff % 4) * 32][koff / 4]; + float2 br3 = rhs_shmem2[start_feature / 2 + 2 + (koff % 4) * 32][koff / 4]; + float2 br4 = rhs_shmem2[start_feature / 2 + 3 + (koff % 4) * 32][koff / 4]; add_vals(a3, a4, br1, br2, br3, br4) } __syncthreads(); - } // end loop over k + }// end loop over k __syncthreads(); - Index horiz_base = (threadIdx.y/4)*8+base_n; + Index horiz_base = (threadIdx.y / 4) * 8 + base_n; if (!CHECK_LHS_BOUNDARY && !CHECK_RHS_BOUNDARY) { for (int i = 0; i < 8; i++) { output(lhs_vert, horiz_base + i) = results[i].x; @@ -1091,10 +1106,8 @@ EigenFloatContractionKernelInternal(const LhsMapper lhs, const RhsMapper rhs, output(lhs_vert, horiz_base + i) = results[i].x; output(lhs_vert + 1, horiz_base + i) = results[i].y; } - } else if (lhs_vert < m_size) { - for (int i = 0; i < 8; i++) { - output(lhs_vert, horiz_base + i) = results[i].x; - } + } else if (lhs_vert < m_size) { + for (int i = 0; i < 8; i++) { output(lhs_vert, horiz_base + i) = results[i].x; } } } else if (!CHECK_LHS_BOUNDARY) { // CHECK BOUNDARY_B @@ -1110,29 +1123,26 @@ EigenFloatContractionKernelInternal(const LhsMapper lhs, const RhsMapper rhs, // CHECK both boundaries. for (int i = 0; i < 8; i++) { if (horiz_base + i < n_size) { - if (lhs_vert < m_size) - output(lhs_vert, horiz_base + i) = results[i].x; - if (lhs_vert + 1 < m_size) - output(lhs_vert + 1, horiz_base + i) = results[i].y; - if (lhs_vert + 2 < m_size) - output(lhs_vert + 2, horiz_base + i) = results[i].z; - if (lhs_vert + 3 < m_size) - output(lhs_vert + 3, horiz_base + i) = results[i].w; + if (lhs_vert < m_size) output(lhs_vert, horiz_base + i) = results[i].x; + if (lhs_vert + 1 < m_size) output(lhs_vert + 1, horiz_base + i) = results[i].y; + if (lhs_vert + 2 < m_size) output(lhs_vert + 2, horiz_base + i) = results[i].z; + if (lhs_vert + 3 < m_size) output(lhs_vert + 3, horiz_base + i) = results[i].w; } } } } -template -__global__ void -__launch_bounds__(256) -EigenFloatContractionKernel(const LhsMapper lhs, const RhsMapper rhs, - const OutputMapper output, - const Index m_size, const Index n_size, const Index k_size) { - __shared__ float2 lhs_shmem[64*32]; - __shared__ float2 rhs_shmem[128*8]; +template +__global__ void __launch_bounds__(256) EigenFloatContractionKernel(const LhsMapper lhs, + const RhsMapper rhs, + const OutputMapper output, + const Index m_size, + const Index n_size, + const Index k_size) +{ + __shared__ float2 lhs_shmem[64 * 32]; + __shared__ float2 rhs_shmem[128 * 8]; typedef float2 LHS_MEM[64][32]; typedef float2 RHS_MEM[128][8]; @@ -1153,30 +1163,31 @@ EigenFloatContractionKernel(const LhsMapper lhs, const RhsMapper rhs, if (!check_lhs128) { // >= 128 rows left EigenFloatContractionKernelInternal( - lhs, rhs, output, *((LHS_MEM *) lhs_shmem), *((RHS_MEM *) rhs_shmem), m_size, n_size, k_size, base_m, base_n); + lhs, rhs, output, *((LHS_MEM *)lhs_shmem), *((RHS_MEM *)rhs_shmem), m_size, n_size, k_size, base_m, base_n); } else { EigenFloatContractionKernelInternal( - lhs, rhs, output, *((LHS_MEM *) lhs_shmem), *((RHS_MEM *) rhs_shmem), m_size, n_size, k_size, base_m, base_n); + lhs, rhs, output, *((LHS_MEM *)lhs_shmem), *((RHS_MEM *)rhs_shmem), m_size, n_size, k_size, base_m, base_n); } } else { if (!check_lhs128) { // >= 128 rows left EigenFloatContractionKernelInternal( - lhs, rhs, output, *((LHS_MEM *) lhs_shmem), *((RHS_MEM *) rhs_shmem), m_size, n_size, k_size, base_m, base_n); + lhs, rhs, output, *((LHS_MEM *)lhs_shmem), *((RHS_MEM *)rhs_shmem), m_size, n_size, k_size, base_m, base_n); } else { EigenFloatContractionKernelInternal( - lhs, rhs, output, *((LHS_MEM *) lhs_shmem), *((RHS_MEM *) rhs_shmem), m_size, n_size, k_size, base_m, base_n); + lhs, rhs, output, *((LHS_MEM *)lhs_shmem), *((RHS_MEM *)rhs_shmem), m_size, n_size, k_size, base_m, base_n); } } } -template -__global__ void -__launch_bounds__(256) -EigenFloatContractionKernel16x16(const LhsMapper lhs, const RhsMapper rhs, - const OutputMapper output, - const Index m_size, const Index n_size, const Index k_size) { +template +__global__ void __launch_bounds__(256) EigenFloatContractionKernel16x16(const LhsMapper lhs, + const RhsMapper rhs, + const OutputMapper output, + const Index m_size, + const Index n_size, + const Index k_size) +{ __shared__ float2 lhs_shmem[32][16]; __shared__ float2 rhs_shmem[64][8]; @@ -1188,23 +1199,29 @@ EigenFloatContractionKernel16x16(const LhsMapper lhs, const RhsMapper rhs, if (base_m + 63 < m_size) { if (base_n + 63 < n_size) { - EigenFloatContractionKernelInternal16x16(lhs, rhs, output, lhs_shmem, rhs_shmem, m_size, n_size, k_size, base_m, base_n); + EigenFloatContractionKernelInternal16x16( + lhs, rhs, output, lhs_shmem, rhs_shmem, m_size, n_size, k_size, base_m, base_n); } else { - EigenFloatContractionKernelInternal16x16(lhs, rhs, output, lhs_shmem, rhs_shmem, m_size, n_size, k_size, base_m, base_n); + EigenFloatContractionKernelInternal16x16( + lhs, rhs, output, lhs_shmem, rhs_shmem, m_size, n_size, k_size, base_m, base_n); } } else { if (base_n + 63 < n_size) { - EigenFloatContractionKernelInternal16x16(lhs, rhs, output, lhs_shmem, rhs_shmem, m_size, n_size, k_size, base_m, base_n); + EigenFloatContractionKernelInternal16x16( + lhs, rhs, output, lhs_shmem, rhs_shmem, m_size, n_size, k_size, base_m, base_n); } else { - EigenFloatContractionKernelInternal16x16(lhs, rhs, output, lhs_shmem, rhs_shmem, m_size, n_size, k_size, base_m, base_n); + EigenFloatContractionKernelInternal16x16( + lhs, rhs, output, lhs_shmem, rhs_shmem, m_size, n_size, k_size, base_m, base_n); } } } template -struct TensorEvaluator, GpuDevice> : - public TensorContractionEvaluatorBase, GpuDevice> > { +struct TensorEvaluator, GpuDevice> + : public TensorContractionEvaluatorBase< + TensorEvaluator, GpuDevice>> +{ typedef GpuDevice Device; @@ -1225,15 +1242,15 @@ struct TensorEvaluator(Layout) == static_cast(ColMajor), LeftArgType, RightArgType>::type EvalLeftArgType; - typedef typename internal::conditional< - static_cast(Layout) == static_cast(ColMajor), RightArgType, LeftArgType>::type EvalRightArgType; - - static const int LDims = - internal::array_size::Dimensions>::value; - static const int RDims = - internal::array_size::Dimensions>::value; + typedef typename internal::conditional(Layout) == static_cast(ColMajor), + LeftArgType, + RightArgType>::type EvalLeftArgType; + typedef typename internal::conditional(Layout) == static_cast(ColMajor), + RightArgType, + LeftArgType>::type EvalRightArgType; + + static const int LDims = internal::array_size::Dimensions>::value; + static const int RDims = internal::array_size::Dimensions>::value; static const int ContractDims = internal::array_size::value; typedef array left_dim_mapper_t; @@ -1257,11 +1274,11 @@ struct TensorEvaluatorm_leftImpl.evalSubExprsIfNeeded(NULL); this->m_rightImpl.evalSubExprsIfNeeded(NULL); if (data) { @@ -1274,75 +1291,123 @@ struct TensorEvaluatorm_lhs_inner_dim_contiguous) { if (this->m_rhs_inner_dim_contiguous) { if (this->m_rhs_inner_dim_reordered) { evalTyped(buffer); - } - else { + } else { evalTyped(buffer); } - } - else { - if (this->m_rhs_inner_dim_reordered) { + } else { + if (this->m_rhs_inner_dim_reordered) { evalTyped(buffer); - } - else { + } else { evalTyped(buffer); } } - } - else { + } else { if (this->m_rhs_inner_dim_contiguous) { if (this->m_rhs_inner_dim_reordered) { evalTyped(buffer); - } - else { + } else { evalTyped(buffer); } - } - else { - if (this->m_rhs_inner_dim_reordered) { + } else { + if (this->m_rhs_inner_dim_reordered) { evalTyped(buffer); - } - else { + } else { evalTyped(buffer); } } } } - template struct LaunchKernels { - static void Run(const LhsMapper& lhs, const RhsMapper& rhs, const OutputMapper& output, Index m, Index n, Index k, const GpuDevice& device) { - const Index m_blocks = (m + 63) / 64; - const Index n_blocks = (n + 63) / 64; - const dim3 num_blocks(m_blocks, n_blocks, 1); - const dim3 block_size(8, 8, 8); - LAUNCH_CUDA_KERNEL((EigenContractionKernel), num_blocks, block_size, 0, device, lhs, rhs, output, m, n, k); + template + struct LaunchKernels + { + static void Run(const LhsMapper &lhs, + const RhsMapper &rhs, + const OutputMapper &output, + Index m, + Index n, + Index k, + const GpuDevice &device) + { + const Index m_blocks = (m + 63) / 64; + const Index n_blocks = (n + 63) / 64; + const dim3 num_blocks(m_blocks, n_blocks, 1); + const dim3 block_size(8, 8, 8); + LAUNCH_CUDA_KERNEL((EigenContractionKernel), + num_blocks, + block_size, + 0, + device, + lhs, + rhs, + output, + m, + n, + k); } }; - template struct LaunchKernels { - static void Run(const LhsMapper& lhs, const RhsMapper& rhs, const OutputMapper& output, Index m, Index n, Index k, const GpuDevice& device) { + template + struct LaunchKernels + { + static void Run(const LhsMapper &lhs, + const RhsMapper &rhs, + const OutputMapper &output, + Index m, + Index n, + Index k, + const GpuDevice &device) + { if (m < 768 || n < 768) { const Index m_blocks = (m + 63) / 64; const Index n_blocks = (n + 63) / 64; const dim3 num_blocks(m_blocks, n_blocks, 1); const dim3 block_size(16, 16, 1); - LAUNCH_CUDA_KERNEL((EigenFloatContractionKernel16x16), num_blocks, block_size, 0, device, lhs, rhs, output, m, n, k); + LAUNCH_CUDA_KERNEL((EigenFloatContractionKernel16x16), + num_blocks, + block_size, + 0, + device, + lhs, + rhs, + output, + m, + n, + k); } else { const Index m_blocks = (m + 127) / 128; const Index n_blocks = (n + 63) / 64; const dim3 num_blocks(m_blocks, n_blocks, 1); const dim3 block_size(8, 32, 1); - LAUNCH_CUDA_KERNEL((EigenFloatContractionKernel), num_blocks, block_size, 0, device, lhs, rhs, output, m, n, k); + LAUNCH_CUDA_KERNEL((EigenFloatContractionKernel), + num_blocks, + block_size, + 0, + device, + lhs, + rhs, + output, + m, + n, + k); } } }; - template - void evalTyped(Scalar* buffer) const { + template + void evalTyped(Scalar *buffer) const + { // columns in left side, rows in right side const Index k = this->m_k_size; EIGEN_UNUSED_VARIABLE(k) @@ -1356,36 +1421,55 @@ struct TensorEvaluatorm_device.memset(buffer, 0, m * n * sizeof(Scalar)); - typedef internal::TensorContractionInputMapper LhsMapper; - - typedef internal::TensorContractionInputMapper RhsMapper; + typedef internal::TensorContractionInputMapper + LhsMapper; + + typedef internal::TensorContractionInputMapper + RhsMapper; typedef internal::blas_data_mapper OutputMapper; // initialize data mappers - LhsMapper lhs(this->m_leftImpl, this->m_left_nocontract_strides, this->m_i_strides, - this->m_left_contracting_strides, this->m_k_strides); - - RhsMapper rhs(this->m_rightImpl, this->m_right_nocontract_strides, this->m_j_strides, - this->m_right_contracting_strides, this->m_k_strides); + LhsMapper lhs(this->m_leftImpl, + this->m_left_nocontract_strides, + this->m_i_strides, + this->m_left_contracting_strides, + this->m_k_strides); + + RhsMapper rhs(this->m_rightImpl, + this->m_right_nocontract_strides, + this->m_j_strides, + this->m_right_contracting_strides, + this->m_k_strides); OutputMapper output(buffer, m); setCudaSharedMemConfig(cudaSharedMemBankSizeEightByte); - LaunchKernels::Run(lhs, rhs, output, m, n, k, this->m_device); + LaunchKernels::Run( + lhs, rhs, output, m, n, k, this->m_device); } }; -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_USE_GPU and __CUDACC__ -#endif // EIGEN_CXX11_TENSOR_TENSOR_CONTRACTION_CUDA_H +#endif// EIGEN_USE_GPU and __CUDACC__ +#endif// EIGEN_CXX11_TENSOR_TENSOR_CONTRACTION_CUDA_H diff --git a/filmulator-gui/core/nlmeans/eigen/unsupported/Eigen/CXX11/src/Tensor/TensorContractionMapper.h b/filmulator-gui/core/nlmeans/eigen/unsupported/Eigen/CXX11/src/Tensor/TensorContractionMapper.h index 9b2cb3ff..eaf6bd39 100644 --- a/filmulator-gui/core/nlmeans/eigen/unsupported/Eigen/CXX11/src/Tensor/TensorContractionMapper.h +++ b/filmulator-gui/core/nlmeans/eigen/unsupported/Eigen/CXX11/src/Tensor/TensorContractionMapper.h @@ -14,454 +14,560 @@ namespace Eigen { namespace internal { -enum { - Rhs = 0, - Lhs = 1 -}; - -/* - * Implementation of the Eigen blas_data_mapper class for tensors. - */ - -template struct CoeffLoader { - enum { - DirectOffsets = false - }; + enum { Rhs = 0, Lhs = 1 }; - EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE CoeffLoader(const Tensor& tensor) : m_tensor(tensor) { } + /* + * Implementation of the Eigen blas_data_mapper class for tensors. + */ - EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE void offsetBuffer(typename Tensor::Index) { - eigen_assert(false && "unsupported"); - } + template struct CoeffLoader + { + enum { DirectOffsets = false }; - EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE typename Tensor::Scalar coeff(typename Tensor::Index index) const { return m_tensor.coeff(index); } + EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE CoeffLoader(const Tensor &tensor) : m_tensor(tensor) {} - template EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE - typename Tensor::PacketReturnType packet(typename Tensor::Index index) const - { - return m_tensor.template packet(index); - } + EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE void offsetBuffer(typename Tensor::Index) + { + eigen_assert(false && "unsupported"); + } + EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE typename Tensor::Scalar coeff(typename Tensor::Index index) const + { + return m_tensor.coeff(index); + } + + template + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE typename Tensor::PacketReturnType packet(typename Tensor::Index index) const + { + return m_tensor.template packet(index); + } - private: - const Tensor m_tensor; -}; -template struct CoeffLoader { - enum { - DirectOffsets = true + private: + const Tensor m_tensor; }; - EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE CoeffLoader(const Tensor& tensor) : m_data(tensor.data()) {} + template struct CoeffLoader + { + enum { DirectOffsets = true }; - EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE void offsetBuffer(typename Tensor::Index offset) { - m_data += offset; - } + EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE CoeffLoader(const Tensor &tensor) : m_data(tensor.data()) {} - EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE typename Tensor::Scalar coeff(typename Tensor::Index index) const { return loadConstant(m_data+index); } + EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE void offsetBuffer(typename Tensor::Index offset) { m_data += offset; } - template EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE - typename Tensor::PacketReturnType packet(typename Tensor::Index index) const - { - return internal::ploadt_ro(m_data + index); - } - private: - typedef typename Tensor::Scalar Scalar; - const Scalar* m_data; -}; - -template -class SimpleTensorContractionMapper { - public: - EIGEN_DEVICE_FUNC - SimpleTensorContractionMapper(const Tensor& tensor, - const nocontract_t& nocontract_strides, - const nocontract_t& ij_strides, - const contract_t& contract_strides, - const contract_t& k_strides) : - m_tensor(tensor), - m_nocontract_strides(nocontract_strides), - m_ij_strides(ij_strides), - m_contract_strides(contract_strides), - m_k_strides(k_strides) { } - - enum { - DirectOffsets = CoeffLoader::DirectOffsets - }; + EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE typename Tensor::Scalar coeff(typename Tensor::Index index) const + { + return loadConstant(m_data + index); + } - EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE void offsetBuffer(typename Tensor::Index offset) { - m_tensor.offsetBuffer(offset); - } - - EIGEN_DEVICE_FUNC - EIGEN_STRONG_INLINE void prefetch(Index /*i*/) { } - - EIGEN_DEVICE_FUNC - EIGEN_STRONG_INLINE Scalar operator()(Index row) const { - // column major assumption - return operator()(row, 0); - } - - EIGEN_DEVICE_FUNC - EIGEN_STRONG_INLINE Scalar operator()(Index row, Index col) const { - return m_tensor.coeff(computeIndex(row, col)); - } - - EIGEN_DEVICE_FUNC - EIGEN_STRONG_INLINE Index computeIndex(Index row, Index col) const { - const bool left = (side == Lhs); - Index nocontract_val = left ? row : col; - Index linidx = 0; - for (int i = static_cast(array_size::value) - 1; i > 0; i--) { - const Index idx = nocontract_val / m_ij_strides[i]; - linidx += idx * m_nocontract_strides[i]; - nocontract_val -= idx * m_ij_strides[i]; + template + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE typename Tensor::PacketReturnType packet(typename Tensor::Index index) const + { + return internal::ploadt_ro(m_data + index); } - if (array_size::value > array_size::value) { - if (side == Lhs && inner_dim_contiguous) { - eigen_assert(m_nocontract_strides[0] == 1); - linidx += nocontract_val; - } else { - linidx += nocontract_val * m_nocontract_strides[0]; - } + + private: + typedef typename Tensor::Scalar Scalar; + const Scalar *m_data; + }; + + template + class SimpleTensorContractionMapper + { + public: + EIGEN_DEVICE_FUNC + SimpleTensorContractionMapper(const Tensor &tensor, + const nocontract_t &nocontract_strides, + const nocontract_t &ij_strides, + const contract_t &contract_strides, + const contract_t &k_strides) + : m_tensor(tensor), m_nocontract_strides(nocontract_strides), m_ij_strides(ij_strides), + m_contract_strides(contract_strides), m_k_strides(k_strides) + {} + + enum { DirectOffsets = CoeffLoader::DirectOffsets }; + + EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE void offsetBuffer(typename Tensor::Index offset) + { + m_tensor.offsetBuffer(offset); } - Index contract_val = left ? col : row; - if(array_size::value > 0) { - for (int i = static_cast(array_size::value) - 1; i > 0; i--) { - const Index idx = contract_val / m_k_strides[i]; - linidx += idx * m_contract_strides[i]; - contract_val -= idx * m_k_strides[i]; - } + EIGEN_DEVICE_FUNC + EIGEN_STRONG_INLINE void prefetch(Index /*i*/) {} - if (side == Rhs && inner_dim_contiguous) { - eigen_assert(m_contract_strides[0] == 1); - linidx += contract_val; - } else { - linidx += contract_val * m_contract_strides[0]; - } + EIGEN_DEVICE_FUNC + EIGEN_STRONG_INLINE Scalar operator()(Index row) const + { + // column major assumption + return operator()(row, 0); } - return linidx; - } + EIGEN_DEVICE_FUNC + EIGEN_STRONG_INLINE Scalar operator()(Index row, Index col) const { return m_tensor.coeff(computeIndex(row, col)); } - EIGEN_DEVICE_FUNC - EIGEN_STRONG_INLINE IndexPair computeIndexPair(Index row, Index col, const Index distance) const { - const bool left = (side == Lhs); - Index nocontract_val[2] = {left ? row : col, left ? row + distance : col}; - Index linidx[2] = {0, 0}; - if (array_size::value > array_size::value) { + EIGEN_DEVICE_FUNC + EIGEN_STRONG_INLINE Index computeIndex(Index row, Index col) const + { + const bool left = (side == Lhs); + Index nocontract_val = left ? row : col; + Index linidx = 0; for (int i = static_cast(array_size::value) - 1; i > 0; i--) { - const Index idx0 = nocontract_val[0] / m_ij_strides[i]; - const Index idx1 = nocontract_val[1] / m_ij_strides[i]; - linidx[0] += idx0 * m_nocontract_strides[i]; - linidx[1] += idx1 * m_nocontract_strides[i]; - nocontract_val[0] -= idx0 * m_ij_strides[i]; - nocontract_val[1] -= idx1 * m_ij_strides[i]; + const Index idx = nocontract_val / m_ij_strides[i]; + linidx += idx * m_nocontract_strides[i]; + nocontract_val -= idx * m_ij_strides[i]; + } + if (array_size::value > array_size::value) { + if (side == Lhs && inner_dim_contiguous) { + eigen_assert(m_nocontract_strides[0] == 1); + linidx += nocontract_val; + } else { + linidx += nocontract_val * m_nocontract_strides[0]; + } } - if (side == Lhs && inner_dim_contiguous) { - eigen_assert(m_nocontract_strides[0] == 1); - linidx[0] += nocontract_val[0]; - linidx[1] += nocontract_val[1]; - } else { - linidx[0] += nocontract_val[0] * m_nocontract_strides[0]; - linidx[1] += nocontract_val[1] * m_nocontract_strides[0]; + + Index contract_val = left ? col : row; + if (array_size::value > 0) { + for (int i = static_cast(array_size::value) - 1; i > 0; i--) { + const Index idx = contract_val / m_k_strides[i]; + linidx += idx * m_contract_strides[i]; + contract_val -= idx * m_k_strides[i]; + } + + if (side == Rhs && inner_dim_contiguous) { + eigen_assert(m_contract_strides[0] == 1); + linidx += contract_val; + } else { + linidx += contract_val * m_contract_strides[0]; + } } + + return linidx; } - Index contract_val[2] = {left ? col : row, left ? col : row + distance}; - if (array_size::value> 0) { - for (int i = static_cast(array_size::value) - 1; i > 0; i--) { - const Index idx0 = contract_val[0] / m_k_strides[i]; - const Index idx1 = contract_val[1] / m_k_strides[i]; - linidx[0] += idx0 * m_contract_strides[i]; - linidx[1] += idx1 * m_contract_strides[i]; - contract_val[0] -= idx0 * m_k_strides[i]; - contract_val[1] -= idx1 * m_k_strides[i]; + EIGEN_DEVICE_FUNC + EIGEN_STRONG_INLINE IndexPair computeIndexPair(Index row, Index col, const Index distance) const + { + const bool left = (side == Lhs); + Index nocontract_val[2] = { left ? row : col, left ? row + distance : col }; + Index linidx[2] = { 0, 0 }; + if (array_size::value > array_size::value) { + for (int i = static_cast(array_size::value) - 1; i > 0; i--) { + const Index idx0 = nocontract_val[0] / m_ij_strides[i]; + const Index idx1 = nocontract_val[1] / m_ij_strides[i]; + linidx[0] += idx0 * m_nocontract_strides[i]; + linidx[1] += idx1 * m_nocontract_strides[i]; + nocontract_val[0] -= idx0 * m_ij_strides[i]; + nocontract_val[1] -= idx1 * m_ij_strides[i]; + } + if (side == Lhs && inner_dim_contiguous) { + eigen_assert(m_nocontract_strides[0] == 1); + linidx[0] += nocontract_val[0]; + linidx[1] += nocontract_val[1]; + } else { + linidx[0] += nocontract_val[0] * m_nocontract_strides[0]; + linidx[1] += nocontract_val[1] * m_nocontract_strides[0]; + } } - if (side == Rhs && inner_dim_contiguous) { - eigen_assert(m_contract_strides[0] == 1); - linidx[0] += contract_val[0]; - linidx[1] += contract_val[1]; - } else { - linidx[0] += contract_val[0] * m_contract_strides[0]; - linidx[1] += contract_val[1] * m_contract_strides[0]; + Index contract_val[2] = { left ? col : row, left ? col : row + distance }; + if (array_size::value > 0) { + for (int i = static_cast(array_size::value) - 1; i > 0; i--) { + const Index idx0 = contract_val[0] / m_k_strides[i]; + const Index idx1 = contract_val[1] / m_k_strides[i]; + linidx[0] += idx0 * m_contract_strides[i]; + linidx[1] += idx1 * m_contract_strides[i]; + contract_val[0] -= idx0 * m_k_strides[i]; + contract_val[1] -= idx1 * m_k_strides[i]; + } + + if (side == Rhs && inner_dim_contiguous) { + eigen_assert(m_contract_strides[0] == 1); + linidx[0] += contract_val[0]; + linidx[1] += contract_val[1]; + } else { + linidx[0] += contract_val[0] * m_contract_strides[0]; + linidx[1] += contract_val[1] * m_contract_strides[0]; + } } + return IndexPair(linidx[0], linidx[1]); } - return IndexPair(linidx[0], linidx[1]); - } - - EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE Index firstAligned(Index size) const { - // Only claim alignment when we can compute the actual stride (ie when we're - // dealing with the lhs with inner_dim_contiguous. This is because the - // matrix-vector product relies on the stride when dealing with aligned inputs. - return (Alignment == Aligned) && (side == Lhs) && inner_dim_contiguous ? 0 : size; - } - EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE Index stride() const { - return ((side == Lhs) && inner_dim_contiguous && array_size::value > 0) ? m_contract_strides[0] : 1; - } - - protected: - CoeffLoader m_tensor; - const nocontract_t m_nocontract_strides; - const nocontract_t m_ij_strides; - const contract_t m_contract_strides; - const contract_t m_k_strides; -}; - - -template -class BaseTensorContractionMapper : public SimpleTensorContractionMapper -{ - public: - typedef SimpleTensorContractionMapper ParentMapper; - - EIGEN_DEVICE_FUNC - BaseTensorContractionMapper(const Tensor& tensor, - const nocontract_t& nocontract_strides, - const nocontract_t& ij_strides, - const contract_t& contract_strides, - const contract_t& k_strides) : - ParentMapper(tensor, nocontract_strides, ij_strides, contract_strides, k_strides) { } - - typedef typename Tensor::PacketReturnType Packet; - typedef typename unpacket_traits::half HalfPacket; - - template - EIGEN_DEVICE_FUNC - EIGEN_STRONG_INLINE Packet loadPacket(Index i, Index j) const { - // whole method makes column major assumption - - // don't need to add offsets for now (because operator handles that) - // current code assumes packet size must be a multiple of 2 - EIGEN_STATIC_ASSERT(packet_size % 2 == 0, YOU_MADE_A_PROGRAMMING_MISTAKE); - - if (Tensor::PacketAccess && inner_dim_contiguous && !inner_dim_reordered) { - const Index index = this->computeIndex(i, j); - eigen_assert(this->computeIndex(i+packet_size-1, j) == index + packet_size-1); - return this->m_tensor.template packet(index); + + EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE Index firstAligned(Index size) const + { + // Only claim alignment when we can compute the actual stride (ie when we're + // dealing with the lhs with inner_dim_contiguous. This is because the + // matrix-vector product relies on the stride when dealing with aligned inputs. + return (Alignment == Aligned) && (side == Lhs) && inner_dim_contiguous ? 0 : size; + } + EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE Index stride() const + { + return ((side == Lhs) && inner_dim_contiguous && array_size::value > 0) ? m_contract_strides[0] : 1; } - const IndexPair indexPair = this->computeIndexPair(i, j, packet_size - 1); - const Index first = indexPair.first; - const Index last = indexPair.second; + protected: + CoeffLoader m_tensor; + const nocontract_t m_nocontract_strides; + const nocontract_t m_ij_strides; + const contract_t m_contract_strides; + const contract_t m_k_strides; + }; - // We can always do optimized packet reads from left hand side right now, because - // the vertical matrix dimension on the left hand side is never contracting. - // On the right hand side we need to check if the contracting dimensions may have - // been shuffled first. - if (Tensor::PacketAccess && - (side == Lhs || internal::array_size::value <= 1 || !inner_dim_reordered) && - (last - first) == (packet_size - 1)) { - return this->m_tensor.template packet(first); - } + template + class BaseTensorContractionMapper + : public SimpleTensorContractionMapper + { + public: + typedef SimpleTensorContractionMapper + ParentMapper; + + EIGEN_DEVICE_FUNC + BaseTensorContractionMapper(const Tensor &tensor, + const nocontract_t &nocontract_strides, + const nocontract_t &ij_strides, + const contract_t &contract_strides, + const contract_t &k_strides) + : ParentMapper(tensor, nocontract_strides, ij_strides, contract_strides, k_strides) + {} + + typedef typename Tensor::PacketReturnType Packet; + typedef typename unpacket_traits::half HalfPacket; + + template EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Packet loadPacket(Index i, Index j) const + { + // whole method makes column major assumption + + // don't need to add offsets for now (because operator handles that) + // current code assumes packet size must be a multiple of 2 + EIGEN_STATIC_ASSERT(packet_size % 2 == 0, YOU_MADE_A_PROGRAMMING_MISTAKE); + + if (Tensor::PacketAccess && inner_dim_contiguous && !inner_dim_reordered) { + const Index index = this->computeIndex(i, j); + eigen_assert(this->computeIndex(i + packet_size - 1, j) == index + packet_size - 1); + return this->m_tensor.template packet(index); + } - EIGEN_ALIGN_MAX Scalar data[packet_size]; + const IndexPair indexPair = this->computeIndexPair(i, j, packet_size - 1); + const Index first = indexPair.first; + const Index last = indexPair.second; - data[0] = this->m_tensor.coeff(first); - for (Index k = 1; k < packet_size - 1; k += 2) { - const IndexPair internal_pair = this->computeIndexPair(i + k, j, 1); - data[k] = this->m_tensor.coeff(internal_pair.first); - data[k + 1] = this->m_tensor.coeff(internal_pair.second); + // We can always do optimized packet reads from left hand side right now, because + // the vertical matrix dimension on the left hand side is never contracting. + // On the right hand side we need to check if the contracting dimensions may have + // been shuffled first. + if (Tensor::PacketAccess && (side == Lhs || internal::array_size::value <= 1 || !inner_dim_reordered) + && (last - first) == (packet_size - 1)) { + + return this->m_tensor.template packet(first); + } + + EIGEN_ALIGN_MAX Scalar data[packet_size]; + + data[0] = this->m_tensor.coeff(first); + for (Index k = 1; k < packet_size - 1; k += 2) { + const IndexPair internal_pair = this->computeIndexPair(i + k, j, 1); + data[k] = this->m_tensor.coeff(internal_pair.first); + data[k + 1] = this->m_tensor.coeff(internal_pair.second); + } + data[packet_size - 1] = this->m_tensor.coeff(last); + + return pload(data); } - data[packet_size - 1] = this->m_tensor.coeff(last); - return pload(data); - } + template EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE HalfPacket loadHalfPacket(Index i, Index j) const + { + // whole method makes column major assumption - template - EIGEN_DEVICE_FUNC - EIGEN_STRONG_INLINE HalfPacket loadHalfPacket(Index i, Index j) const { - // whole method makes column major assumption + // don't need to add offsets for now (because operator handles that) + const Index half_packet_size = unpacket_traits::size; + if (half_packet_size == packet_size) { return loadPacket(i, j); } + EIGEN_ALIGN_MAX Scalar data[half_packet_size]; + for (Index k = 0; k < half_packet_size; k++) { data[k] = operator()(i + k, j); } + return pload(data); + } + }; - // don't need to add offsets for now (because operator handles that) - const Index half_packet_size = unpacket_traits::size; - if (half_packet_size == packet_size) { - return loadPacket(i, j); + + template + class BaseTensorContractionMapper + : public SimpleTensorContractionMapper + { + public: + typedef SimpleTensorContractionMapper + ParentMapper; + + EIGEN_DEVICE_FUNC + BaseTensorContractionMapper(const Tensor &tensor, + const nocontract_t &nocontract_strides, + const nocontract_t &ij_strides, + const contract_t &contract_strides, + const contract_t &k_strides) + : ParentMapper(tensor, nocontract_strides, ij_strides, contract_strides, k_strides) + {} + + typedef typename Tensor::PacketReturnType Packet; + template EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Packet loadPacket(Index i, Index j) const + { + EIGEN_ALIGN_MAX Scalar data[1]; + data[0] = this->m_tensor.coeff(this->computeIndex(i, j)); + return pload(data); } - EIGEN_ALIGN_MAX Scalar data[half_packet_size]; - for (Index k = 0; k < half_packet_size; k++) { - data[k] = operator()(i + k, j); + template EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Packet loadHalfPacket(Index i, Index j) const + { + return loadPacket(i, j); } - return pload(data); - } -}; - - -template -class BaseTensorContractionMapper : public SimpleTensorContractionMapper -{ - public: - typedef SimpleTensorContractionMapper ParentMapper; - - EIGEN_DEVICE_FUNC - BaseTensorContractionMapper(const Tensor& tensor, - const nocontract_t& nocontract_strides, - const nocontract_t& ij_strides, - const contract_t& contract_strides, - const contract_t& k_strides) : - ParentMapper(tensor, nocontract_strides, ij_strides, contract_strides, k_strides) { } - - typedef typename Tensor::PacketReturnType Packet; - template EIGEN_DEVICE_FUNC - EIGEN_STRONG_INLINE Packet loadPacket(Index i, Index j) const { - EIGEN_ALIGN_MAX Scalar data[1]; - data[0] = this->m_tensor.coeff(this->computeIndex(i, j)); - return pload(data); - } - template EIGEN_DEVICE_FUNC - EIGEN_STRONG_INLINE Packet loadHalfPacket(Index i, Index j) const { - return loadPacket(i, j); - } -}; - - -template -class TensorContractionSubMapper { - public: - typedef typename Tensor::PacketReturnType Packet; - typedef typename unpacket_traits::half HalfPacket; - - typedef BaseTensorContractionMapper ParentMapper; - typedef TensorContractionSubMapper Self; - typedef Self LinearMapper; - - enum { - // We can use direct offsets iff the parent mapper supports then and we can compute the strides. - // TODO: we should also enable direct offsets for the Rhs case. - UseDirectOffsets = ParentMapper::DirectOffsets && (side == Lhs) && inner_dim_contiguous && (array_size::value > 0) }; - EIGEN_DEVICE_FUNC TensorContractionSubMapper(const ParentMapper& base_mapper, Index vert_offset, Index horiz_offset) - : m_base_mapper(base_mapper), m_vert_offset(vert_offset), m_horiz_offset(horiz_offset) { - // Bake the offsets into the buffer used by the base mapper whenever possible. This avoids the need to recompute - // this offset every time we attempt to access a coefficient. - if (UseDirectOffsets) { - Index stride = m_base_mapper.stride(); - m_base_mapper.offsetBuffer(vert_offset + horiz_offset * stride); + + template + class TensorContractionSubMapper + { + public: + typedef typename Tensor::PacketReturnType Packet; + typedef typename unpacket_traits::half HalfPacket; + + typedef BaseTensorContractionMapper + ParentMapper; + typedef TensorContractionSubMapper + Self; + typedef Self LinearMapper; + + enum { + // We can use direct offsets iff the parent mapper supports then and we can compute the strides. + // TODO: we should also enable direct offsets for the Rhs case. + UseDirectOffsets = + ParentMapper::DirectOffsets && (side == Lhs) && inner_dim_contiguous && (array_size::value > 0) + }; + + EIGEN_DEVICE_FUNC TensorContractionSubMapper(const ParentMapper &base_mapper, Index vert_offset, Index horiz_offset) + : m_base_mapper(base_mapper), m_vert_offset(vert_offset), m_horiz_offset(horiz_offset) + { + // Bake the offsets into the buffer used by the base mapper whenever possible. This avoids the need to recompute + // this offset every time we attempt to access a coefficient. + if (UseDirectOffsets) { + Index stride = m_base_mapper.stride(); + m_base_mapper.offsetBuffer(vert_offset + horiz_offset * stride); + } } - } - EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE Scalar operator()(Index i) const { - if (UseDirectOffsets) { - return m_base_mapper(i, 0); + EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE Scalar operator()(Index i) const + { + if (UseDirectOffsets) { return m_base_mapper(i, 0); } + return m_base_mapper(i + m_vert_offset, m_horiz_offset); } - return m_base_mapper(i + m_vert_offset, m_horiz_offset); - } - EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE Scalar operator()(Index i, Index j) const { - if (UseDirectOffsets) { - return m_base_mapper(i, j); + EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE Scalar operator()(Index i, Index j) const + { + if (UseDirectOffsets) { return m_base_mapper(i, j); } + return m_base_mapper(i + m_vert_offset, j + m_horiz_offset); } - return m_base_mapper(i + m_vert_offset, j + m_horiz_offset); - } - EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE Packet loadPacket(Index i) const { - if (UseDirectOffsets) { - return m_base_mapper.template loadPacket(i, 0); + EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE Packet loadPacket(Index i) const + { + if (UseDirectOffsets) { return m_base_mapper.template loadPacket(i, 0); } + return m_base_mapper.template loadPacket(i + m_vert_offset, m_horiz_offset); } - return m_base_mapper.template loadPacket(i + m_vert_offset, m_horiz_offset); - } - EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE Packet loadPacket(Index i, Index j) const { - if (UseDirectOffsets) { - return m_base_mapper.template loadPacket(i, j); + EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE Packet loadPacket(Index i, Index j) const + { + if (UseDirectOffsets) { return m_base_mapper.template loadPacket(i, j); } + return m_base_mapper.template loadPacket(i + m_vert_offset, j + m_horiz_offset); } - return m_base_mapper.template loadPacket(i + m_vert_offset, j + m_horiz_offset); - } - EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE HalfPacket loadHalfPacket(Index i) const { - if (UseDirectOffsets) { - return m_base_mapper.template loadHalfPacket(i, 0); + EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE HalfPacket loadHalfPacket(Index i) const + { + if (UseDirectOffsets) { return m_base_mapper.template loadHalfPacket(i, 0); } + return m_base_mapper.template loadHalfPacket(i + m_vert_offset, m_horiz_offset); } - return m_base_mapper.template loadHalfPacket(i + m_vert_offset, m_horiz_offset); - } - EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE void storePacket(Index i, Packet p) const { - if (UseDirectOffsets) { - m_base_mapper.storePacket(i, 0, p); + EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE void storePacket(Index i, Packet p) const + { + if (UseDirectOffsets) { m_base_mapper.storePacket(i, 0, p); } + m_base_mapper.storePacket(i + m_vert_offset, m_horiz_offset, p); } - m_base_mapper.storePacket(i + m_vert_offset, m_horiz_offset, p); - } - EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE LinearMapper getLinearMapper(Index i, Index j) const { - if (UseDirectOffsets) { - return LinearMapper(m_base_mapper, i, j); + EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE LinearMapper getLinearMapper(Index i, Index j) const + { + if (UseDirectOffsets) { return LinearMapper(m_base_mapper, i, j); } + return LinearMapper(m_base_mapper, i + m_vert_offset, j + m_horiz_offset); } - return LinearMapper(m_base_mapper, i + m_vert_offset, j + m_horiz_offset); - } - - template - EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE PacketT load(Index i) const { - EIGEN_STATIC_ASSERT((internal::is_same::value), YOU_MADE_A_PROGRAMMING_MISTAKE); - const int ActualAlignment = (AlignmentType == Aligned) && (Alignment == Aligned) ? Aligned : Unaligned; - if (UseDirectOffsets) { - return m_base_mapper.template loadPacket(i, 0); + + template EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE PacketT load(Index i) const + { + EIGEN_STATIC_ASSERT((internal::is_same::value), YOU_MADE_A_PROGRAMMING_MISTAKE); + const int ActualAlignment = (AlignmentType == Aligned) && (Alignment == Aligned) ? Aligned : Unaligned; + if (UseDirectOffsets) { return m_base_mapper.template loadPacket(i, 0); } + return m_base_mapper.template loadPacket(i + m_vert_offset, m_horiz_offset); + } + + template EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE bool aligned(Index) const { return false; } + + private: + ParentMapper m_base_mapper; + const Index m_vert_offset; + const Index m_horiz_offset; + }; + + + template + class TensorContractionInputMapper + : public BaseTensorContractionMapper + { + + public: + typedef Scalar_ Scalar; + typedef BaseTensorContractionMapper + Base; + typedef TensorContractionSubMapper + SubMapper; + typedef SubMapper VectorMapper; + + EIGEN_DEVICE_FUNC TensorContractionInputMapper(const Tensor &tensor, + const nocontract_t &nocontract_strides, + const nocontract_t &ij_strides, + const contract_t &contract_strides, + const contract_t &k_strides) + : Base(tensor, nocontract_strides, ij_strides, contract_strides, k_strides) + {} + + EIGEN_DEVICE_FUNC + EIGEN_STRONG_INLINE SubMapper getSubMapper(Index i, Index j) const { return SubMapper(*this, i, j); } + + EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE VectorMapper getVectorMapper(Index i, Index j) const + { + return VectorMapper(*this, i, j); } - return m_base_mapper.template loadPacket(i + m_vert_offset, m_horiz_offset); - } - - template - EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE bool aligned(Index) const { - return false; - } - - private: - ParentMapper m_base_mapper; - const Index m_vert_offset; - const Index m_horiz_offset; -}; - - -template -class TensorContractionInputMapper - : public BaseTensorContractionMapper { - - public: - typedef Scalar_ Scalar; - typedef BaseTensorContractionMapper Base; - typedef TensorContractionSubMapper SubMapper; - typedef SubMapper VectorMapper; - - EIGEN_DEVICE_FUNC TensorContractionInputMapper(const Tensor& tensor, - const nocontract_t& nocontract_strides, - const nocontract_t& ij_strides, - const contract_t& contract_strides, - const contract_t& k_strides) - : Base(tensor, nocontract_strides, ij_strides, contract_strides, k_strides) { } - - EIGEN_DEVICE_FUNC - EIGEN_STRONG_INLINE SubMapper getSubMapper(Index i, Index j) const { - return SubMapper(*this, i, j); - } - - EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE VectorMapper getVectorMapper(Index i, Index j) const { - return VectorMapper(*this, i, j); - } -}; - - - -} // end namespace internal -} // end namespace Eigen - -#endif // EIGEN_CXX11_TENSOR_TENSOR_CONTRACTION_MAPPER_H + }; + + +}// end namespace internal +}// end namespace Eigen + +#endif// EIGEN_CXX11_TENSOR_TENSOR_CONTRACTION_MAPPER_H diff --git a/filmulator-gui/core/nlmeans/eigen/unsupported/Eigen/CXX11/src/Tensor/TensorContractionThreadPool.h b/filmulator-gui/core/nlmeans/eigen/unsupported/Eigen/CXX11/src/Tensor/TensorContractionThreadPool.h index c70dea05..22b67632 100644 --- a/filmulator-gui/core/nlmeans/eigen/unsupported/Eigen/CXX11/src/Tensor/TensorContractionThreadPool.h +++ b/filmulator-gui/core/nlmeans/eigen/unsupported/Eigen/CXX11/src/Tensor/TensorContractionThreadPool.h @@ -18,47 +18,50 @@ namespace Eigen { #ifdef EIGEN_USE_SIMPLE_THREAD_POOL namespace internal { -template -struct packLhsArg { - LhsScalar* blockA; - const LhsMapper& lhs; - const Index m_start; - const Index k_start; - const Index mc; - const Index kc; -}; + template struct packLhsArg + { + LhsScalar *blockA; + const LhsMapper &lhs; + const Index m_start; + const Index k_start; + const Index mc; + const Index kc; + }; -template -struct packRhsAndKernelArg { - const MaxSizeVector* blockAs; - RhsScalar* blockB; - const RhsMapper& rhs; - OutputMapper& output; - const Index m; - const Index k; - const Index n; - const Index mc; - const Index kc; - const Index nc; - const Index num_threads; - const Index num_blockAs; - const Index max_m; - const Index k_block_idx; - const Index m_block_idx; - const Index n_block_idx; - const Index m_blocks; - const Index n_blocks; - MaxSizeVector* kernel_notifications; - const MaxSizeVector* lhs_notifications; - const bool need_to_pack; -}; + template + struct packRhsAndKernelArg + { + const MaxSizeVector *blockAs; + RhsScalar *blockB; + const RhsMapper &rhs; + OutputMapper &output; + const Index m; + const Index k; + const Index n; + const Index mc; + const Index kc; + const Index nc; + const Index num_threads; + const Index num_blockAs; + const Index max_m; + const Index k_block_idx; + const Index m_block_idx; + const Index n_block_idx; + const Index m_blocks; + const Index n_blocks; + MaxSizeVector *kernel_notifications; + const MaxSizeVector *lhs_notifications; + const bool need_to_pack; + }; -} // end namespace internal -#endif // EIGEN_USE_SIMPLE_THREAD_POOL +}// end namespace internal +#endif// EIGEN_USE_SIMPLE_THREAD_POOL template -struct TensorEvaluator, ThreadPoolDevice> : - public TensorContractionEvaluatorBase, ThreadPoolDevice> > { +struct TensorEvaluator, ThreadPoolDevice> + : public TensorContractionEvaluatorBase< + TensorEvaluator, ThreadPoolDevice>> +{ typedef ThreadPoolDevice Device; @@ -79,15 +82,15 @@ struct TensorEvaluator(Layout) == static_cast(ColMajor), LeftArgType, RightArgType>::type EvalLeftArgType; - typedef typename internal::conditional< - static_cast(Layout) == static_cast(ColMajor), RightArgType, LeftArgType>::type EvalRightArgType; - - static const int LDims = - internal::array_size::Dimensions>::value; - static const int RDims = - internal::array_size::Dimensions>::value; + typedef typename internal::conditional(Layout) == static_cast(ColMajor), + LeftArgType, + RightArgType>::type EvalLeftArgType; + typedef typename internal::conditional(Layout) == static_cast(ColMajor), + RightArgType, + LeftArgType>::type EvalRightArgType; + + static const int LDims = internal::array_size::Dimensions>::value; + static const int RDims = internal::array_size::Dimensions>::value; static const int ContractDims = internal::array_size::value; typedef array left_dim_mapper_t; @@ -109,34 +112,41 @@ struct TensorEvaluator LeftEvaluator; typedef TensorEvaluator RightEvaluator; - TensorEvaluator(const XprType& op, const Device& device) : - Base(op, device) {} + TensorEvaluator(const XprType &op, const Device &device) : Base(op, device) {} #ifndef EIGEN_USE_SIMPLE_THREAD_POOL - template - void evalProduct(Scalar* buffer) const { - typedef internal::TensorContractionInputMapper< - LhsScalar, Index, internal::Lhs, LeftEvaluator, left_nocontract_t, - contract_t, internal::packet_traits::size, - lhs_inner_dim_contiguous, false, Unaligned> - LhsMapper; - typedef internal::TensorContractionInputMapper< - RhsScalar, Index, internal::Rhs, RightEvaluator, right_nocontract_t, - contract_t, internal::packet_traits::size, - rhs_inner_dim_contiguous, rhs_inner_dim_reordered, Unaligned> - RhsMapper; + template + void evalProduct(Scalar *buffer) const + { + typedef internal::TensorContractionInputMapper::size, + lhs_inner_dim_contiguous, + false, + Unaligned> + LhsMapper; + typedef internal::TensorContractionInputMapper::size, + rhs_inner_dim_contiguous, + rhs_inner_dim_reordered, + Unaligned> + RhsMapper; typedef internal::blas_data_mapper OutputMapper; - typedef internal::gemm_pack_lhs + typedef internal:: + gemm_pack_lhs LhsPacker; - typedef internal::gemm_pack_rhs< - RhsScalar, Index, typename RhsMapper::SubMapper, Traits::nr, ColMajor> - RhsPacker; - typedef internal::gebp_kernel - GebpKernel; + typedef internal::gemm_pack_rhs RhsPacker; + typedef internal::gebp_kernel + GebpKernel; const Index m = this->m_i_size; const Index n = this->m_j_size; @@ -172,16 +182,12 @@ struct TensorEvaluator - blocking(k, m, n, 2); + internal::TensorContractionBlocking blocking(k, m, n, 2); bm = blocking.mc(); bn = blocking.nc(); bk = blocking.kc(); } else { - internal::TensorContractionBlocking - blocking(k, m, n, 2); + internal::TensorContractionBlocking blocking(k, m, n, 2); bm = blocking.mc(); bn = blocking.nc(); bk = blocking.kc(); @@ -191,10 +197,9 @@ struct TensorEvaluator::numThreads( - static_cast(n) * m, cost, this->m_device.numThreads()); + const TensorOpCost cost = contractionCost(m, n, bm, bn, bk, shard_by_col, false); + int num_threads = + TensorCostModel::numThreads(static_cast(n) * m, cost, this->m_device.numThreads()); // TODO(dvyukov): this is a stop-gap to prevent regressions while the cost // model is not tuned. Remove this when the cost model is tuned. @@ -203,29 +208,25 @@ struct TensorEvaluatortemplate evalGemv(buffer); + this->template evalGemv( + buffer); else - this->template evalGemm(buffer); + this->template evalGemm( + buffer); return; } // Now that we know number of threads, recalculate sharding and blocking. shard_by_col = shardByCol(m, n, num_threads); if (shard_by_col) { - internal::TensorContractionBlocking - blocking(k, m, n, num_threads); + internal::TensorContractionBlocking blocking( + k, m, n, num_threads); bm = blocking.mc(); bn = blocking.nc(); bk = blocking.kc(); } else { - internal::TensorContractionBlocking - blocking(k, m, n, num_threads); + internal::TensorContractionBlocking blocking( + k, m, n, num_threads); bm = blocking.mc(); bn = blocking.nc(); bk = blocking.kc(); @@ -264,109 +265,123 @@ struct TensorEvaluator= nm * nn; // Also do parallel packing if all data fits into L2$. - if (m * bk * Index(sizeof(LhsScalar)) + n * bk * Index(sizeof(RhsScalar)) <= - l2CacheSize() * num_threads) + if (m * bk * Index(sizeof(LhsScalar)) + n * bk * Index(sizeof(RhsScalar)) <= l2CacheSize() * num_threads) parallel_pack = true; // But don't do it if we will use each rhs only once. Locality seems to be // more important in this case. if ((shard_by_col ? nm : nn) == 1) parallel_pack = false; - LhsMapper lhs(this->m_leftImpl, this->m_left_nocontract_strides, - this->m_i_strides, this->m_left_contracting_strides, - this->m_k_strides); - - RhsMapper rhs(this->m_rightImpl, this->m_right_nocontract_strides, - this->m_j_strides, this->m_right_contracting_strides, - this->m_k_strides); - - Context(this->m_device, num_threads, lhs, rhs, buffer, m, n, - k, bm, bn, bk, nm, nn, nk, gm, gn, nm0, nn0, - shard_by_col, parallel_pack) - .run(); + LhsMapper lhs(this->m_leftImpl, + this->m_left_nocontract_strides, + this->m_i_strides, + this->m_left_contracting_strides, + this->m_k_strides); + + RhsMapper rhs(this->m_rightImpl, + this->m_right_nocontract_strides, + this->m_j_strides, + this->m_right_contracting_strides, + this->m_k_strides); + + Context(this->m_device, + num_threads, + lhs, + rhs, + buffer, + m, + n, + k, + bm, + bn, + bk, + nm, + nn, + nk, + gm, + gn, + nm0, + nn0, + shard_by_col, + parallel_pack) + .run(); } // Context coordinates a single parallel gemm operation. - template - class Context { - public: - Context(const Device& device, int num_threads, LhsMapper& lhs, - RhsMapper& rhs, Scalar* buffer, Index tm, Index tn, Index tk, Index bm, - Index bn, Index bk, Index nm, Index nn, Index nk, Index gm, - Index gn, Index nm0, Index nn0, bool shard_by_col, - bool parallel_pack) - : device_(device), - lhs_(lhs), - rhs_(rhs), - buffer_(buffer), - output_(buffer, tm), - num_threads_(num_threads), - shard_by_col_(shard_by_col), - parallel_pack_(parallel_pack), - m_(tm), - n_(tn), - k_(tk), - bm_(bm), - bn_(bn), - bk_(bk), - nm_(nm), - nn_(nn), - nk_(nk), - gm_(gm), - gn_(gn), - nm0_(nm0), - nn0_(nn0) + template + class Context { + public: + Context(const Device &device, + int num_threads, + LhsMapper &lhs, + RhsMapper &rhs, + Scalar *buffer, + Index tm, + Index tn, + Index tk, + Index bm, + Index bn, + Index bk, + Index nm, + Index nn, + Index nk, + Index gm, + Index gn, + Index nm0, + Index nn0, + bool shard_by_col, + bool parallel_pack) + : device_(device), lhs_(lhs), rhs_(rhs), buffer_(buffer), output_(buffer, tm), num_threads_(num_threads), + shard_by_col_(shard_by_col), parallel_pack_(parallel_pack), m_(tm), n_(tn), k_(tk), bm_(bm), bn_(bn), bk_(bk), + nm_(nm), nn_(nn), nk_(nk), gm_(gm), gn_(gn), nm0_(nm0), nn0_(nn0) + { for (Index x = 0; x < P; x++) { // Normal number of notifications for k slice switch is // nm_ + nn_ + nm_ * nn_. However, first P - 1 slices will receive only // nm_ + nn_ notifications, because they will not receive notifications // from preceeding kernels. state_switch_[x] = - x == 0 - ? 1 - : (parallel_pack_ ? nn_ + nm_ : (shard_by_col_ ? nn_ : nm_)) + - (x == P - 1 ? nm_ * nn_ : 0); - state_packing_ready_[x] = - parallel_pack_ ? 0 : (shard_by_col_ ? nm_ : nn_); - state_kernel_[x] = new std::atomic*[nm_]; + x == 0 ? 1 : (parallel_pack_ ? nn_ + nm_ : (shard_by_col_ ? nn_ : nm_)) + (x == P - 1 ? nm_ * nn_ : 0); + state_packing_ready_[x] = parallel_pack_ ? 0 : (shard_by_col_ ? nm_ : nn_); + state_kernel_[x] = new std::atomic *[nm_]; for (Index m = 0; m < nm_; m++) { state_kernel_[x][m] = new std::atomic[nn_]; // Kernels generally receive 3 notifications (previous kernel + 2 // packing), but the first slice won't get notifications from previous // kernels. for (Index n = 0; n < nn_; n++) - state_kernel_[x][m][n].store( - (x == 0 ? 0 : 1) + (parallel_pack_ ? 2 : 1), - std::memory_order_relaxed); + state_kernel_[x][m][n].store((x == 0 ? 0 : 1) + (parallel_pack_ ? 2 : 1), std::memory_order_relaxed); } } // Allocate memory for packed rhs/lhs matrices. size_t align = numext::maxi(EIGEN_MAX_ALIGN_BYTES, 1); - size_t lhs_size = - divup(bm_ * bk_ * sizeof(LhsScalar), align) * align; - size_t rhs_size = - divup(bn_ * bk_ * sizeof(RhsScalar), align) * align; - packed_mem_ = static_cast(internal::aligned_malloc( - (nm0_ * lhs_size + nn0_ * rhs_size) * std::min(nk_, P - 1))); - char* mem = static_cast(packed_mem_); + size_t lhs_size = divup(bm_ * bk_ * sizeof(LhsScalar), align) * align; + size_t rhs_size = divup(bn_ * bk_ * sizeof(RhsScalar), align) * align; + packed_mem_ = static_cast( + internal::aligned_malloc((nm0_ * lhs_size + nn0_ * rhs_size) * std::min(nk_, P - 1))); + char *mem = static_cast(packed_mem_); for (Index x = 0; x < numext::mini(nk_, P - 1); x++) { packed_lhs_[x].resize(nm0_); for (Index m = 0; m < nm0_; m++) { - packed_lhs_[x][m] = reinterpret_cast(mem); + packed_lhs_[x][m] = reinterpret_cast(mem); mem += lhs_size; } packed_rhs_[x].resize(nn0_); for (Index n = 0; n < nn0_; n++) { - packed_rhs_[x][n] = reinterpret_cast(mem); + packed_rhs_[x][n] = reinterpret_cast(mem); mem += rhs_size; } } } - ~Context() { + ~Context() + { for (Index x = 0; x < P; x++) { for (Index m = 0; m < nm_; m++) delete[] state_kernel_[x][m]; delete[] state_kernel_[x]; @@ -374,7 +389,8 @@ struct TensorEvaluator packed_lhs_[P - 1]; - std::vector packed_rhs_[P - 1]; - std::atomic** state_kernel_[P]; + void *packed_mem_; + std::vector packed_lhs_[P - 1]; + std::vector packed_rhs_[P - 1]; + std::atomic **state_kernel_[P]; // state_switch_ is frequently modified by worker threads, while other // fields are read-only after constructor. Let's move it to a separate cache // line to reduce cache-coherency traffic. @@ -461,11 +477,11 @@ struct TensorEvaluator state_packing_ready_[P]; std::atomic state_switch_[P]; - void pack_lhs(Index m, Index k) { + void pack_lhs(Index m, Index k) + { const Index mend = m * gm_ + gm(m); for (Index m1 = m * gm_; m1 < mend; m1++) - LhsPacker()(packed_lhs_[k % (P - 1)][m1], - lhs_.getSubMapper(m1 * bm_, k * bk_), bk(k), bm(m1)); + LhsPacker()(packed_lhs_[k % (P - 1)][m1], lhs_.getSubMapper(m1 * bm_, k * bk_), bk(k), bm(m1)); if (!parallel_pack_ && shard_by_col_) { signal_packing(k); @@ -475,7 +491,8 @@ struct TensorEvaluator 0); @@ -536,8 +568,9 @@ struct TensorEvaluator* state = &state_kernel_[k % P][m][n]; + void signal_kernel(Index m, Index n, Index k, bool sync) + { + std::atomic *state = &state_kernel_[k % P][m][n]; Index s = state->load(); eigen_assert(s > 0); if (s != 1 && state->fetch_sub(1) != 1) return; @@ -548,16 +581,15 @@ struct TensorEvaluator= v); if (s != v) return; // Ready to switch to the next k slice. // Reset counter for the next iteration. - state_switch_[k % P] = - (parallel_pack_ ? nm_ + nn_ : (shard_by_col_ ? nn_ : nm_)) + - nm_ * nn_; + state_switch_[k % P] = (parallel_pack_ ? nm_ + nn_ : (shard_by_col_ ? nn_ : nm_)) + nm_ * nn_; if (k < nk_) { // Issue lhs/rhs packing. Their completion will in turn kick off // kernels. @@ -576,19 +608,17 @@ struct TensorEvaluator= Traits::nr && // and not enough data for vectorization over columns (n / num_threads < Traits::nr || - // ... or barely enough data for vectorization over columns, - // but it is not evenly dividable across threads - (n / num_threads < 4 * Traits::nr && - (n % (num_threads * Traits::nr)) != 0 && - // ... and it is evenly dividable across threads for rows - ((m % (num_threads * Traits::nr)) == 0 || - // .. or it is not evenly dividable for both dimensions but - // there is much more data over rows so that corner effects are - // mitigated. - (m / n >= 6))))) + // ... or barely enough data for vectorization over columns, + // but it is not evenly dividable across threads + (n / num_threads < 4 * Traits::nr && (n % (num_threads * Traits::nr)) != 0 && + // ... and it is evenly dividable across threads for rows + ((m % (num_threads * Traits::nr)) == 0 || + // .. or it is not evenly dividable for both dimensions but + // there is much more data over rows so that corner effects are + // mitigated. + (m / n >= 6))))) return false; // Wait, or if matrices are just substantially prolonged over the other // dimension. @@ -643,8 +671,8 @@ struct TensorEvaluator nm0) break; // Check the candidate. - int res = checkGrain(m, n, bm, bn, bk, gm1, gn, gm, gn, num_threads, - shard_by_col); + int res = checkGrain(m, n, bm, bn, bk, gm1, gn, gm, gn, num_threads, shard_by_col); if (res < 0) break; nm1 = divup(nm0, gm1); if (res == 0) continue; @@ -667,8 +694,8 @@ struct TensorEvaluator nn0) break; - int res = checkGrain(m, n, bm, bn, bk, gm, gn1, gm, gn, num_threads, - shard_by_col); + int res = checkGrain(m, n, bm, bn, bk, gm, gn1, gm, gn, num_threads, shard_by_col); if (res < 0) break; nn1 = divup(nn0, gn1); if (res == 0) continue; @@ -688,13 +714,20 @@ struct TensorEvaluator::taskSize( - static_cast(bm) * gm * bn * gn, cost); + int checkGrain(Index m, + Index n, + Index bm, + Index bn, + Index bk, + Index gm, + Index gn, + Index oldgm, + Index oldgn, + int num_threads, + bool shard_by_col) const + { + const TensorOpCost cost = contractionCost(bm * gm, bn * gn, bm, bn, bk, shard_by_col, true); + double taskSize = TensorCostModel::taskSize(static_cast(bm) * gm * bn * gn, cost); // If the task is too small, then we agree on it regardless of anything // else. Otherwise synchronization overheads will dominate. if (taskSize < 1) return 1; @@ -709,29 +742,30 @@ struct TensorEvaluator(new_tasks) / - (divup(new_tasks, num_threads) * num_threads); + double new_parallelism = static_cast(new_tasks) / (divup(new_tasks, num_threads) * num_threads); Index old_tasks = divup(nm0, oldgm) * divup(nn0, oldgn); - double old_parallelism = static_cast(old_tasks) / - (divup(old_tasks, num_threads) * num_threads); + double old_parallelism = static_cast(old_tasks) / (divup(old_tasks, num_threads) * num_threads); if (new_parallelism > old_parallelism || new_parallelism == 1) return 1; return 0; } -#else // EIGEN_USE_SIMPLE_THREAD_POOL +#else// EIGEN_USE_SIMPLE_THREAD_POOL - template - void evalProduct(Scalar* buffer) const { + template + void evalProduct(Scalar *buffer) const + { if (this->m_j_size == 1) { - this->template evalGemv(buffer); + this->template evalGemv( + buffer); return; } evalGemm(buffer); } - template - void evalGemm(Scalar* buffer) const { + template + void evalGemm(Scalar *buffer) const + { // columns in left side, rows in right side const Index k = this->m_k_size; @@ -748,44 +782,64 @@ struct TensorEvaluator::size; const int rhs_packet_size = internal::unpacket_traits::size; - typedef internal::TensorContractionInputMapper LhsMapper; - - typedef internal::TensorContractionInputMapper RhsMapper; + typedef internal::TensorContractionInputMapper + LhsMapper; + + typedef internal::TensorContractionInputMapper + RhsMapper; typedef internal::blas_data_mapper OutputMapper; // TODO: packing could be faster sometimes if we supported row major tensor mappers - typedef internal::gemm_pack_lhs LhsPacker; + typedef internal:: + gemm_pack_lhs + LhsPacker; typedef internal::gemm_pack_rhs RhsPacker; // TODO: replace false, false with conjugate values? - typedef internal::gebp_kernel GebpKernel; + typedef internal::gebp_kernel + GebpKernel; typedef internal::packLhsArg packLArg; typedef internal::packRhsAndKernelArg packRKArg; // initialize data mappers - LhsMapper lhs(this->m_leftImpl, this->m_left_nocontract_strides, this->m_i_strides, - this->m_left_contracting_strides, this->m_k_strides); - - RhsMapper rhs(this->m_rightImpl, this->m_right_nocontract_strides, this->m_j_strides, - this->m_right_contracting_strides, this->m_k_strides); + LhsMapper lhs(this->m_leftImpl, + this->m_left_nocontract_strides, + this->m_i_strides, + this->m_left_contracting_strides, + this->m_k_strides); + + RhsMapper rhs(this->m_rightImpl, + this->m_right_nocontract_strides, + this->m_j_strides, + this->m_right_contracting_strides, + this->m_k_strides); OutputMapper output(buffer, m); // compute block sizes (which depend on number of threads) const Index num_threads = this->m_device.numThreads(); - internal::TensorContractionBlocking blocking(k, m, n, num_threads); + internal::TensorContractionBlocking blocking( + k, m, n, num_threads); Index mc = blocking.mc(); Index nc = blocking.nc(); Index kc = blocking.kc(); @@ -827,12 +881,11 @@ struct TensorEvaluator lhs_notifications(num_threads, nullptr); + MaxSizeVector lhs_notifications(num_threads, nullptr); // this should really be numBlockAs * n_blocks; const Index num_kernel_notifications = num_threads * n_blocks; - MaxSizeVector kernel_notifications(num_kernel_notifications, - nullptr); + MaxSizeVector kernel_notifications(num_kernel_notifications, nullptr); for (Index k_block_idx = 0; k_block_idx < k_blocks; k_block_idx++) { const Index k_start = k_block_idx * kc; @@ -840,9 +893,9 @@ struct TensorEvaluator 0); @@ -860,20 +913,19 @@ struct TensorEvaluatorm_device.enqueue(&Self::packLhs, arg); + lhs_notifications[blockAId] = this->m_device.enqueue(&Self::packLhs, arg); } // now start kernels. @@ -895,27 +947,27 @@ struct TensorEvaluatorm_device.deallocate(blockAs[i]); - } - for (size_t i = 0; i < blockBs.size(); i++) { - this->m_device.deallocate(blockBs[i]); - } + for (size_t i = 0; i < blockAs.size(); i++) { this->m_device.deallocate(blockAs[i]); } + for (size_t i = 0; i < blockBs.size(); i++) { this->m_device.deallocate(blockBs[i]); } #undef CEIL_DIV } @@ -954,8 +1000,8 @@ struct TensorEvaluator - static void packLhs(const packLArg arg) { + template static void packLhs(const packLArg arg) + { // perform actual packing LhsPacker pack_lhs; pack_lhs(arg.blockA, arg.lhs.getSubMapper(arg.m_start, arg.k_start), arg.kc, arg.mc); @@ -970,8 +1016,8 @@ struct TensorEvaluator - static void packRhsAndKernel(packRKArg arg) { + template static void packRhsAndKernel(packRKArg arg) + { if (arg.need_to_pack) { RhsPacker pack_rhs; pack_rhs(arg.blockB, arg.rhs.getSubMapper(arg.k, arg.n), arg.kc, arg.nc); @@ -979,14 +1025,22 @@ struct TensorEvaluator(PacketType::size, - PacketType::size); + TensorOpCost contractionCost(Index m, Index n, Index bm, Index bn, Index bk, bool shard_by_col, bool prepacked) const + { + const int packed_size = std::min(PacketType::size, PacketType::size); const int output_packet_size = internal::unpacket_traits::size; const double kd = static_cast(bk); // Peak VFMA bandwidth is 0.5. However if we have not enough data for // vectorization bandwidth drops. The 4.0 and 2.0 bandwidth is determined // experimentally. - double computeBandwidth = bk == 1 ? 4.0 : - (shard_by_col ? bn : bm) < Traits::nr || - (shard_by_col ? bm : bn) < Traits::mr ? 2.0 : 0.5; + double computeBandwidth = bk == 1 ? 4.0 + : (shard_by_col ? bn : bm) < Traits::nr || (shard_by_col ? bm : bn) < Traits::mr ? 2.0 + : 0.5; #ifndef EIGEN_VECTORIZE_FMA // Bandwidth of all of VFMA/MULPS/ADDPS is 0.5 on latest Intel processors. // However for MULPS/ADDPS we have dependent sequence of 2 such instructions, @@ -1037,7 +1090,7 @@ struct TensorEvaluator -struct traits > -{ - // Type promotion to handle the case where the types of the lhs and the rhs are different. - typedef TargetType Scalar; - typedef typename traits::StorageKind StorageKind; - typedef typename traits::Index Index; - typedef typename XprType::Nested Nested; - typedef typename remove_reference::type _Nested; - static const int NumDimensions = traits::NumDimensions; - static const int Layout = traits::Layout; - enum { Flags = 0 }; -}; + template struct traits> + { + // Type promotion to handle the case where the types of the lhs and the rhs are different. + typedef TargetType Scalar; + typedef typename traits::StorageKind StorageKind; + typedef typename traits::Index Index; + typedef typename XprType::Nested Nested; + typedef typename remove_reference::type _Nested; + static const int NumDimensions = traits::NumDimensions; + static const int Layout = traits::Layout; + enum { Flags = 0 }; + }; -template -struct eval, Eigen::Dense> -{ - typedef const TensorConversionOp& type; -}; + template struct eval, Eigen::Dense> + { + typedef const TensorConversionOp &type; + }; -template -struct nested, 1, typename eval >::type> -{ - typedef TensorConversionOp type; -}; + template + struct nested, + 1, + typename eval>::type> + { + typedef TensorConversionOp type; + }; -} // end namespace internal +}// end namespace internal -template -struct PacketConverter { - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE - PacketConverter(const TensorEvaluator& impl) - : m_impl(impl) {} +template +struct PacketConverter +{ + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE PacketConverter(const TensorEvaluator &impl) : m_impl(impl) {} - template - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TgtPacket packet(Index index) const { + template EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TgtPacket packet(Index index) const + { return internal::pcast(m_impl.template packet(index)); } - private: - const TensorEvaluator& m_impl; +private: + const TensorEvaluator &m_impl; }; -template -struct PacketConverter { - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE - PacketConverter(const TensorEvaluator& impl) - : m_impl(impl) {} +template +struct PacketConverter +{ + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE PacketConverter(const TensorEvaluator &impl) : m_impl(impl) {} - template - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TgtPacket packet(Index index) const { + template EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TgtPacket packet(Index index) const + { const int SrcPacketSize = internal::unpacket_traits::size; SrcPacket src1 = m_impl.template packet(index); @@ -81,18 +79,17 @@ struct PacketConverter { return result; } - private: - const TensorEvaluator& m_impl; +private: + const TensorEvaluator &m_impl; }; -template -struct PacketConverter { - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE - PacketConverter(const TensorEvaluator& impl) - : m_impl(impl) {} +template +struct PacketConverter +{ + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE PacketConverter(const TensorEvaluator &impl) : m_impl(impl) {} - template - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TgtPacket packet(Index index) const { + template EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TgtPacket packet(Index index) const + { const int SrcPacketSize = internal::unpacket_traits::size; SrcPacket src1 = m_impl.template packet(index); @@ -103,18 +100,19 @@ struct PacketConverter { return result; } - private: - const TensorEvaluator& m_impl; +private: + const TensorEvaluator &m_impl; }; -template -struct PacketConverter { - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE - PacketConverter(const TensorEvaluator& impl) - : m_impl(impl), m_maxIndex(impl.dimensions().TotalSize()) {} +template +struct PacketConverter +{ + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE PacketConverter(const TensorEvaluator &impl) + : m_impl(impl), m_maxIndex(impl.dimensions().TotalSize()) + {} - template - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TgtPacket packet(Index index) const { + template EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TgtPacket packet(Index index) const + { const int SrcPacketSize = internal::unpacket_traits::size; // Only call m_impl.packet() when we have direct access to the underlying data. This // ensures that we don't compute the subexpression twice. We may however load some @@ -128,50 +126,50 @@ struct PacketConverter { typedef typename internal::unpacket_traits::type TgtType; internal::scalar_cast_op converter; EIGEN_ALIGN_MAX typename internal::unpacket_traits::type values[TgtPacketSize]; - for (int i = 0; i < TgtPacketSize; ++i) { - values[i] = converter(m_impl.coeff(index+i)); - } + for (int i = 0; i < TgtPacketSize; ++i) { values[i] = converter(m_impl.coeff(index + i)); } TgtPacket rslt = internal::pload(values); return rslt; } } - private: - const TensorEvaluator& m_impl; +private: + const TensorEvaluator &m_impl; const typename TensorEvaluator::Index m_maxIndex; }; template class TensorConversionOp : public TensorBase, ReadOnlyAccessors> { - public: - typedef typename internal::traits::Scalar Scalar; - typedef typename internal::traits::StorageKind StorageKind; - typedef typename internal::traits::Index Index; - typedef typename internal::nested::type Nested; - typedef Scalar CoeffReturnType; - typedef typename NumTraits::Real RealScalar; - - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TensorConversionOp(const XprType& xpr) - : m_xpr(xpr) {} - - EIGEN_DEVICE_FUNC - const typename internal::remove_all::type& - expression() const { return m_xpr; } - - protected: - typename XprType::Nested m_xpr; +public: + typedef typename internal::traits::Scalar Scalar; + typedef typename internal::traits::StorageKind StorageKind; + typedef typename internal::traits::Index Index; + typedef typename internal::nested::type Nested; + typedef Scalar CoeffReturnType; + typedef typename NumTraits::Real RealScalar; + + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TensorConversionOp(const XprType &xpr) : m_xpr(xpr) {} + + EIGEN_DEVICE_FUNC + const typename internal::remove_all::type &expression() const { return m_xpr; } + +protected: + typename XprType::Nested m_xpr; }; -template struct ConversionSubExprEval { - static EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE bool run(Eval& impl, Scalar*) { +template struct ConversionSubExprEval +{ + static EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE bool run(Eval &impl, Scalar *) + { impl.evalSubExprsIfNeeded(NULL); return true; } }; -template struct ConversionSubExprEval { - static EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE bool run(Eval& impl, Scalar* data) { +template struct ConversionSubExprEval +{ + static EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE bool run(Eval &impl, Scalar *data) + { return impl.evalSubExprsIfNeeded(data); } }; @@ -191,29 +189,22 @@ struct TensorEvaluator, Device> typedef typename PacketType::type PacketSourceType; static const int PacketSize = internal::unpacket_traits::size; - enum { - IsAligned = false, - PacketAccess = true, - Layout = TensorEvaluator::Layout, - RawAccess = false - }; + enum { IsAligned = false, PacketAccess = true, Layout = TensorEvaluator::Layout, RawAccess = false }; - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TensorEvaluator(const XprType& op, const Device& device) + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TensorEvaluator(const XprType &op, const Device &device) : m_impl(op.expression(), device) - { - } + {} - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const Dimensions& dimensions() const { return m_impl.dimensions(); } + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const Dimensions &dimensions() const { return m_impl.dimensions(); } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE bool evalSubExprsIfNeeded(Scalar* data) + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE bool evalSubExprsIfNeeded(Scalar *data) { - return ConversionSubExprEval::value, TensorEvaluator, Scalar>::run(m_impl, data); + return ConversionSubExprEval::value, + TensorEvaluator, + Scalar>::run(m_impl, data); } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void cleanup() - { - m_impl.cleanup(); - } + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void cleanup() { m_impl.cleanup(); } EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE CoeffReturnType coeff(Index index) const { @@ -221,52 +212,55 @@ struct TensorEvaluator, Device> return converter(m_impl.coeff(index)); } - template - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE PacketReturnType packet(Index index) const + template EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE PacketReturnType packet(Index index) const { - const bool Vectorizable = TensorEvaluator::PacketAccess & - internal::type_casting_traits::VectorizedCast; + const bool Vectorizable = TensorEvaluator::PacketAccess + & internal::type_casting_traits::VectorizedCast; return PacketConv::run(m_impl, index); } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TensorOpCost - costPerCoeff(bool vectorized) const { + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TensorOpCost costPerCoeff(bool vectorized) const + { const double cast_cost = TensorOpCost::CastCost(); if (vectorized) { - const double SrcCoeffRatio = - internal::type_casting_traits::SrcCoeffRatio; - const double TgtCoeffRatio = - internal::type_casting_traits::TgtCoeffRatio; - return m_impl.costPerCoeff(vectorized) * (SrcCoeffRatio / PacketSize) + - TensorOpCost(0, 0, TgtCoeffRatio * (cast_cost / PacketSize)); + const double SrcCoeffRatio = internal::type_casting_traits::SrcCoeffRatio; + const double TgtCoeffRatio = internal::type_casting_traits::TgtCoeffRatio; + return m_impl.costPerCoeff(vectorized) * (SrcCoeffRatio / PacketSize) + + TensorOpCost(0, 0, TgtCoeffRatio * (cast_cost / PacketSize)); } else { return m_impl.costPerCoeff(vectorized) + TensorOpCost(0, 0, cast_cost); } } - EIGEN_DEVICE_FUNC Scalar* data() const { return NULL; } + EIGEN_DEVICE_FUNC Scalar *data() const { return NULL; } - protected: - template - struct PacketConv { - static EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE PacketReturnType run(const TensorEvaluator& impl, Index index) { +protected: + template struct PacketConv + { + static EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE PacketReturnType run(const TensorEvaluator &impl, + Index index) + { internal::scalar_cast_op converter; EIGEN_ALIGN_MAX typename internal::remove_const::type values[PacketSize]; - for (int i = 0; i < PacketSize; ++i) { - values[i] = converter(impl.coeff(index+i)); - } + for (int i = 0; i < PacketSize; ++i) { values[i] = converter(impl.coeff(index + i)); } PacketReturnType rslt = internal::pload(values); return rslt; } }; - template - struct PacketConv { - static EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE PacketReturnType run(const TensorEvaluator& impl, Index index) { + template struct PacketConv + { + static EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE PacketReturnType run(const TensorEvaluator &impl, + Index index) + { const int SrcCoeffRatio = internal::type_casting_traits::SrcCoeffRatio; const int TgtCoeffRatio = internal::type_casting_traits::TgtCoeffRatio; - PacketConverter, PacketSourceType, PacketReturnType, - SrcCoeffRatio, TgtCoeffRatio> converter(impl); + PacketConverter, + PacketSourceType, + PacketReturnType, + SrcCoeffRatio, + TgtCoeffRatio> + converter(impl); return converter.template packet(index); } }; @@ -274,6 +268,6 @@ struct TensorEvaluator, Device> TensorEvaluator m_impl; }; -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_CXX11_TENSOR_TENSOR_CONVERSION_H +#endif// EIGEN_CXX11_TENSOR_TENSOR_CONVERSION_H diff --git a/filmulator-gui/core/nlmeans/eigen/unsupported/Eigen/CXX11/src/Tensor/TensorConvolution.h b/filmulator-gui/core/nlmeans/eigen/unsupported/Eigen/CXX11/src/Tensor/TensorConvolution.h index abdf742c..57d567e0 100644 --- a/filmulator-gui/core/nlmeans/eigen/unsupported/Eigen/CXX11/src/Tensor/TensorConvolution.h +++ b/filmulator-gui/core/nlmeans/eigen/unsupported/Eigen/CXX11/src/Tensor/TensorConvolution.h @@ -13,277 +13,266 @@ namespace Eigen { /** \class TensorConvolution - * \ingroup CXX11_Tensor_Module - * - * \brief Tensor convolution class. - * - * - */ + * \ingroup CXX11_Tensor_Module + * + * \brief Tensor convolution class. + * + * + */ namespace internal { -template -class IndexMapper { - public: - IndexMapper(const InputDims& input_dims, const array& kernel_dims, - const array& indices) { - - array dimensions = input_dims; - for (int i = 0; i < NumKernelDims; ++i) { - const Index index = indices[i]; - const Index input_dim = input_dims[index]; - const Index kernel_dim = kernel_dims[i]; - const Index result_dim = input_dim - kernel_dim + 1; - dimensions[index] = result_dim; - } + template class IndexMapper + { + public: + IndexMapper(const InputDims &input_dims, + const array &kernel_dims, + const array &indices) + { - array inputStrides; - array outputStrides; - if (static_cast(Layout) == static_cast(ColMajor)) { - inputStrides[0] = 1; - outputStrides[0] = 1; - for (int i = 1; i < NumDims; ++i) { - inputStrides[i] = inputStrides[i-1] * input_dims[i-1]; - outputStrides[i] = outputStrides[i-1] * dimensions[i-1]; - } - } else { - inputStrides[NumDims - 1] = 1; - outputStrides[NumDims - 1] = 1; - for (int i = static_cast(NumDims) - 2; i >= 0; --i) { - inputStrides[i] = inputStrides[i + 1] * input_dims[i + 1]; - outputStrides[i] = outputStrides[i + 1] * dimensions[i + 1]; + array dimensions = input_dims; + for (int i = 0; i < NumKernelDims; ++i) { + const Index index = indices[i]; + const Index input_dim = input_dims[index]; + const Index kernel_dim = kernel_dims[i]; + const Index result_dim = input_dim - kernel_dim + 1; + dimensions[index] = result_dim; } - } - array cudaInputDimensions; - array cudaOutputDimensions; - array tmp = dimensions; - array ordering; - const size_t offset = static_cast(Layout) == static_cast(ColMajor) - ? 0 - : NumDims - NumKernelDims; - for (int i = 0; i < NumKernelDims; ++i) { - const Index index = i + offset; - ordering[index] = indices[i]; - tmp[indices[i]] = -1; - cudaInputDimensions[index] = input_dims[indices[i]]; - cudaOutputDimensions[index] = dimensions[indices[i]]; - } - - int written = static_cast(Layout) == static_cast(ColMajor) - ? NumKernelDims - : 0; - for (int i = 0; i < NumDims; ++i) { - if (tmp[i] >= 0) { - ordering[written] = i; - cudaInputDimensions[written] = input_dims[i]; - cudaOutputDimensions[written] = dimensions[i]; - ++written; + array inputStrides; + array outputStrides; + if (static_cast(Layout) == static_cast(ColMajor)) { + inputStrides[0] = 1; + outputStrides[0] = 1; + for (int i = 1; i < NumDims; ++i) { + inputStrides[i] = inputStrides[i - 1] * input_dims[i - 1]; + outputStrides[i] = outputStrides[i - 1] * dimensions[i - 1]; + } + } else { + inputStrides[NumDims - 1] = 1; + outputStrides[NumDims - 1] = 1; + for (int i = static_cast(NumDims) - 2; i >= 0; --i) { + inputStrides[i] = inputStrides[i + 1] * input_dims[i + 1]; + outputStrides[i] = outputStrides[i + 1] * dimensions[i + 1]; + } } - } - for (int i = 0; i < NumDims; ++i) { - m_inputStrides[i] = inputStrides[ordering[i]]; - m_outputStrides[i] = outputStrides[ordering[i]]; - } + array cudaInputDimensions; + array cudaOutputDimensions; + array tmp = dimensions; + array ordering; + const size_t offset = static_cast(Layout) == static_cast(ColMajor) ? 0 : NumDims - NumKernelDims; + for (int i = 0; i < NumKernelDims; ++i) { + const Index index = i + offset; + ordering[index] = indices[i]; + tmp[indices[i]] = -1; + cudaInputDimensions[index] = input_dims[indices[i]]; + cudaOutputDimensions[index] = dimensions[indices[i]]; + } - if (static_cast(Layout) == static_cast(ColMajor)) { + int written = static_cast(Layout) == static_cast(ColMajor) ? NumKernelDims : 0; for (int i = 0; i < NumDims; ++i) { - if (i > NumKernelDims) { - m_cudaInputStrides[i] = - m_cudaInputStrides[i - 1] * cudaInputDimensions[i - 1]; - m_cudaOutputStrides[i] = - m_cudaOutputStrides[i - 1] * cudaOutputDimensions[i - 1]; - } else { - m_cudaInputStrides[i] = 1; - m_cudaOutputStrides[i] = 1; + if (tmp[i] >= 0) { + ordering[written] = i; + cudaInputDimensions[written] = input_dims[i]; + cudaOutputDimensions[written] = dimensions[i]; + ++written; } } - } else { - for (int i = NumDims - 1; i >= 0; --i) { - if (i + 1 < offset) { - m_cudaInputStrides[i] = - m_cudaInputStrides[i + 1] * cudaInputDimensions[i + 1]; - m_cudaOutputStrides[i] = - m_cudaOutputStrides[i + 1] * cudaOutputDimensions[i + 1]; - } else { - m_cudaInputStrides[i] = 1; - m_cudaOutputStrides[i] = 1; + + for (int i = 0; i < NumDims; ++i) { + m_inputStrides[i] = inputStrides[ordering[i]]; + m_outputStrides[i] = outputStrides[ordering[i]]; + } + + if (static_cast(Layout) == static_cast(ColMajor)) { + for (int i = 0; i < NumDims; ++i) { + if (i > NumKernelDims) { + m_cudaInputStrides[i] = m_cudaInputStrides[i - 1] * cudaInputDimensions[i - 1]; + m_cudaOutputStrides[i] = m_cudaOutputStrides[i - 1] * cudaOutputDimensions[i - 1]; + } else { + m_cudaInputStrides[i] = 1; + m_cudaOutputStrides[i] = 1; + } + } + } else { + for (int i = NumDims - 1; i >= 0; --i) { + if (i + 1 < offset) { + m_cudaInputStrides[i] = m_cudaInputStrides[i + 1] * cudaInputDimensions[i + 1]; + m_cudaOutputStrides[i] = m_cudaOutputStrides[i + 1] * cudaOutputDimensions[i + 1]; + } else { + m_cudaInputStrides[i] = 1; + m_cudaOutputStrides[i] = 1; + } } } } - } - EIGEN_STRONG_INLINE EIGEN_DEVICE_FUNC Index mapCudaInputPlaneToTensorInputOffset(Index p) const { - Index inputIndex = 0; - if (static_cast(Layout) == static_cast(ColMajor)) { - for (int d = NumDims - 1; d > NumKernelDims; --d) { - const Index idx = p / m_cudaInputStrides[d]; - inputIndex += idx * m_inputStrides[d]; - p -= idx * m_cudaInputStrides[d]; - } - inputIndex += p * m_inputStrides[NumKernelDims]; - } else { - std::ptrdiff_t limit = 0; - if (NumKernelDims < NumDims) { - limit = NumDims - NumKernelDims - 1; - } - for (int d = 0; d < limit; ++d) { - const Index idx = p / m_cudaInputStrides[d]; - inputIndex += idx * m_inputStrides[d]; - p -= idx * m_cudaInputStrides[d]; + EIGEN_STRONG_INLINE EIGEN_DEVICE_FUNC Index mapCudaInputPlaneToTensorInputOffset(Index p) const + { + Index inputIndex = 0; + if (static_cast(Layout) == static_cast(ColMajor)) { + for (int d = NumDims - 1; d > NumKernelDims; --d) { + const Index idx = p / m_cudaInputStrides[d]; + inputIndex += idx * m_inputStrides[d]; + p -= idx * m_cudaInputStrides[d]; + } + inputIndex += p * m_inputStrides[NumKernelDims]; + } else { + std::ptrdiff_t limit = 0; + if (NumKernelDims < NumDims) { limit = NumDims - NumKernelDims - 1; } + for (int d = 0; d < limit; ++d) { + const Index idx = p / m_cudaInputStrides[d]; + inputIndex += idx * m_inputStrides[d]; + p -= idx * m_cudaInputStrides[d]; + } + inputIndex += p * m_inputStrides[limit]; } - inputIndex += p * m_inputStrides[limit]; + return inputIndex; } - return inputIndex; - } - EIGEN_STRONG_INLINE EIGEN_DEVICE_FUNC Index mapCudaOutputPlaneToTensorOutputOffset(Index p) const { - Index outputIndex = 0; - if (static_cast(Layout) == static_cast(ColMajor)) { - for (int d = NumDims - 1; d > NumKernelDims; --d) { - const Index idx = p / m_cudaOutputStrides[d]; - outputIndex += idx * m_outputStrides[d]; - p -= idx * m_cudaOutputStrides[d]; - } - outputIndex += p * m_outputStrides[NumKernelDims]; - } else { - std::ptrdiff_t limit = 0; - if (NumKernelDims < NumDims) { - limit = NumDims - NumKernelDims - 1; - } - for (int d = 0; d < limit; ++d) { - const Index idx = p / m_cudaOutputStrides[d]; - outputIndex += idx * m_outputStrides[d]; - p -= idx * m_cudaOutputStrides[d]; + EIGEN_STRONG_INLINE EIGEN_DEVICE_FUNC Index mapCudaOutputPlaneToTensorOutputOffset(Index p) const + { + Index outputIndex = 0; + if (static_cast(Layout) == static_cast(ColMajor)) { + for (int d = NumDims - 1; d > NumKernelDims; --d) { + const Index idx = p / m_cudaOutputStrides[d]; + outputIndex += idx * m_outputStrides[d]; + p -= idx * m_cudaOutputStrides[d]; + } + outputIndex += p * m_outputStrides[NumKernelDims]; + } else { + std::ptrdiff_t limit = 0; + if (NumKernelDims < NumDims) { limit = NumDims - NumKernelDims - 1; } + for (int d = 0; d < limit; ++d) { + const Index idx = p / m_cudaOutputStrides[d]; + outputIndex += idx * m_outputStrides[d]; + p -= idx * m_cudaOutputStrides[d]; + } + outputIndex += p * m_outputStrides[limit]; } - outputIndex += p * m_outputStrides[limit]; + return outputIndex; } - return outputIndex; - } - - EIGEN_STRONG_INLINE EIGEN_DEVICE_FUNC Index mapCudaInputKernelToTensorInputOffset(Index i) const { - const size_t offset = static_cast(Layout) == static_cast(ColMajor) - ? 0 - : NumDims - NumKernelDims; - return i * m_inputStrides[offset]; - } - - EIGEN_STRONG_INLINE EIGEN_DEVICE_FUNC Index mapCudaOutputKernelToTensorOutputOffset(Index i) const { - const size_t offset = static_cast(Layout) == static_cast(ColMajor) - ? 0 - : NumDims - NumKernelDims; - return i * m_outputStrides[offset]; - } - EIGEN_STRONG_INLINE EIGEN_DEVICE_FUNC Index mapCudaInputKernelToTensorInputOffset(Index i, Index j) const { - const size_t offset = static_cast(Layout) == static_cast(ColMajor) - ? 0 - : NumDims - NumKernelDims; - return i * m_inputStrides[offset] + j * m_inputStrides[offset + 1]; - } + EIGEN_STRONG_INLINE EIGEN_DEVICE_FUNC Index mapCudaInputKernelToTensorInputOffset(Index i) const + { + const size_t offset = static_cast(Layout) == static_cast(ColMajor) ? 0 : NumDims - NumKernelDims; + return i * m_inputStrides[offset]; + } - EIGEN_STRONG_INLINE EIGEN_DEVICE_FUNC Index mapCudaOutputKernelToTensorOutputOffset(Index i, Index j) const { - const size_t offset = static_cast(Layout) == static_cast(ColMajor) - ? 0 - : NumDims - NumKernelDims; - return i * m_outputStrides[offset] + j * m_outputStrides[offset + 1]; - } + EIGEN_STRONG_INLINE EIGEN_DEVICE_FUNC Index mapCudaOutputKernelToTensorOutputOffset(Index i) const + { + const size_t offset = static_cast(Layout) == static_cast(ColMajor) ? 0 : NumDims - NumKernelDims; + return i * m_outputStrides[offset]; + } - EIGEN_STRONG_INLINE EIGEN_DEVICE_FUNC Index mapCudaInputKernelToTensorInputOffset(Index i, Index j, Index k) const { - const size_t offset = static_cast(Layout) == static_cast(ColMajor) - ? 0 - : NumDims - NumKernelDims; - return i * m_inputStrides[offset] + j * m_inputStrides[offset + 1] + - k * m_inputStrides[offset + 2]; - } + EIGEN_STRONG_INLINE EIGEN_DEVICE_FUNC Index mapCudaInputKernelToTensorInputOffset(Index i, Index j) const + { + const size_t offset = static_cast(Layout) == static_cast(ColMajor) ? 0 : NumDims - NumKernelDims; + return i * m_inputStrides[offset] + j * m_inputStrides[offset + 1]; + } - EIGEN_STRONG_INLINE EIGEN_DEVICE_FUNC Index mapCudaOutputKernelToTensorOutputOffset(Index i, Index j, Index k) const { - const size_t offset = static_cast(Layout) == static_cast(ColMajor) - ? 0 - : NumDims - NumKernelDims; - return i * m_outputStrides[offset] + j * m_outputStrides[offset + 1] + - k * m_outputStrides[offset + 2]; - } + EIGEN_STRONG_INLINE EIGEN_DEVICE_FUNC Index mapCudaOutputKernelToTensorOutputOffset(Index i, Index j) const + { + const size_t offset = static_cast(Layout) == static_cast(ColMajor) ? 0 : NumDims - NumKernelDims; + return i * m_outputStrides[offset] + j * m_outputStrides[offset + 1]; + } - private: - static const int NumDims = internal::array_size::value; - array m_inputStrides; - array m_outputStrides; - array m_cudaInputStrides; - array m_cudaOutputStrides; -}; + EIGEN_STRONG_INLINE EIGEN_DEVICE_FUNC Index mapCudaInputKernelToTensorInputOffset(Index i, Index j, Index k) const + { + const size_t offset = static_cast(Layout) == static_cast(ColMajor) ? 0 : NumDims - NumKernelDims; + return i * m_inputStrides[offset] + j * m_inputStrides[offset + 1] + k * m_inputStrides[offset + 2]; + } + EIGEN_STRONG_INLINE EIGEN_DEVICE_FUNC Index mapCudaOutputKernelToTensorOutputOffset(Index i, Index j, Index k) const + { + const size_t offset = static_cast(Layout) == static_cast(ColMajor) ? 0 : NumDims - NumKernelDims; + return i * m_outputStrides[offset] + j * m_outputStrides[offset + 1] + k * m_outputStrides[offset + 2]; + } + private: + static const int NumDims = internal::array_size::value; + array m_inputStrides; + array m_outputStrides; + array m_cudaInputStrides; + array m_cudaOutputStrides; + }; -template -struct traits > -{ - // Type promotion to handle the case where the types of the lhs and the rhs are different. - typedef typename promote_storage_type::ret Scalar; - typedef typename promote_storage_type::StorageKind, - typename traits::StorageKind>::ret StorageKind; - typedef typename promote_index_type::Index, - typename traits::Index>::type Index; - typedef typename InputXprType::Nested LhsNested; - typedef typename KernelXprType::Nested RhsNested; - typedef typename remove_reference::type _LhsNested; - typedef typename remove_reference::type _RhsNested; - static const int NumDimensions = traits::NumDimensions; - static const int Layout = traits::Layout; - enum { - Flags = 0 + template + struct traits> + { + // Type promotion to handle the case where the types of the lhs and the rhs are different. + typedef typename promote_storage_type::ret Scalar; + typedef typename promote_storage_type::StorageKind, + typename traits::StorageKind>::ret StorageKind; + typedef + typename promote_index_type::Index, typename traits::Index>::type + Index; + typedef typename InputXprType::Nested LhsNested; + typedef typename KernelXprType::Nested RhsNested; + typedef typename remove_reference::type _LhsNested; + typedef typename remove_reference::type _RhsNested; + static const int NumDimensions = traits::NumDimensions; + static const int Layout = traits::Layout; + + enum { Flags = 0 }; }; -}; - -template -struct eval, Eigen::Dense> -{ - typedef const TensorConvolutionOp& type; -}; -template -struct nested, 1, typename eval >::type> -{ - typedef TensorConvolutionOp type; -}; + template + struct eval, Eigen::Dense> + { + typedef const TensorConvolutionOp &type; + }; -} // end namespace internal + template + struct nested, + 1, + typename eval>::type> + { + typedef TensorConvolutionOp type; + }; +}// end namespace internal template -class TensorConvolutionOp : public TensorBase, ReadOnlyAccessors> +class TensorConvolutionOp + : public TensorBase, ReadOnlyAccessors> { - public: +public: typedef typename Eigen::internal::traits::Scalar Scalar; typedef typename Eigen::NumTraits::Real RealScalar; typedef typename internal::promote_storage_type::ret CoeffReturnType; + typename KernelXprType::CoeffReturnType>::ret CoeffReturnType; typedef typename Eigen::internal::nested::type Nested; typedef typename Eigen::internal::traits::StorageKind StorageKind; typedef typename Eigen::internal::traits::Index Index; - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TensorConvolutionOp(const InputXprType& input, const KernelXprType& kernel, const Indices& dims) - : m_input_xpr(input), m_kernel_xpr(kernel), m_indices(dims) {} + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TensorConvolutionOp(const InputXprType &input, + const KernelXprType &kernel, + const Indices &dims) + : m_input_xpr(input), m_kernel_xpr(kernel), m_indices(dims) + {} - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE - const Indices& indices() const { return m_indices; } + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const Indices &indices() const { return m_indices; } - /** \returns the nested expressions */ - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE - const typename internal::remove_all::type& - inputExpression() const { return m_input_xpr; } + /** \returns the nested expressions */ + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const typename internal::remove_all::type & + inputExpression() const + { + return m_input_xpr; + } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE - const typename internal::remove_all::type& - kernelExpression() const { return m_kernel_xpr; } + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const typename internal::remove_all::type & + kernelExpression() const + { + return m_kernel_xpr; + } - protected: - typename InputXprType::Nested m_input_xpr; - typename KernelXprType::Nested m_kernel_xpr; - const Indices m_indices; +protected: + typename InputXprType::Nested m_input_xpr; + typename KernelXprType::Nested m_kernel_xpr; + const Indices m_indices; }; @@ -304,30 +293,30 @@ struct TensorEvaluator::IsAligned & TensorEvaluator::IsAligned, - PacketAccess = TensorEvaluator::PacketAccess & TensorEvaluator::PacketAccess, + PacketAccess = + TensorEvaluator::PacketAccess & TensorEvaluator::PacketAccess, Layout = TensorEvaluator::Layout, - CoordAccess = false, // to be implemented + CoordAccess = false,// to be implemented RawAccess = false }; - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TensorEvaluator(const XprType& op, const Device& device) - : m_inputImpl(op.inputExpression(), device), m_kernelImpl(op.kernelExpression(), device), m_kernelArg(op.kernelExpression()), m_kernel(NULL), m_local_kernel(false), m_device(device) + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TensorEvaluator(const XprType &op, const Device &device) + : m_inputImpl(op.inputExpression(), device), m_kernelImpl(op.kernelExpression(), device), + m_kernelArg(op.kernelExpression()), m_kernel(NULL), m_local_kernel(false), m_device(device) { - EIGEN_STATIC_ASSERT((static_cast(TensorEvaluator::Layout) == static_cast(TensorEvaluator::Layout)), YOU_MADE_A_PROGRAMMING_MISTAKE); + EIGEN_STATIC_ASSERT((static_cast(TensorEvaluator::Layout) + == static_cast(TensorEvaluator::Layout)), + YOU_MADE_A_PROGRAMMING_MISTAKE); - const typename TensorEvaluator::Dimensions& input_dims = m_inputImpl.dimensions(); - const typename TensorEvaluator::Dimensions& kernel_dims = m_kernelImpl.dimensions(); + const typename TensorEvaluator::Dimensions &input_dims = m_inputImpl.dimensions(); + const typename TensorEvaluator::Dimensions &kernel_dims = m_kernelImpl.dimensions(); if (static_cast(Layout) == static_cast(ColMajor)) { m_inputStride[0] = 1; - for (int i = 1; i < NumDims; ++i) { - m_inputStride[i] = m_inputStride[i - 1] * input_dims[i - 1]; - } + for (int i = 1; i < NumDims; ++i) { m_inputStride[i] = m_inputStride[i - 1] * input_dims[i - 1]; } } else { m_inputStride[NumDims - 1] = 1; - for (int i = NumDims - 2; i >= 0; --i) { - m_inputStride[i] = m_inputStride[i + 1] * input_dims[i + 1]; - } + for (int i = NumDims - 2; i >= 0; --i) { m_inputStride[i] = m_inputStride[i + 1] * input_dims[i + 1]; } } m_dimensions = m_inputImpl.dimensions(); @@ -347,9 +336,7 @@ struct TensorEvaluator= 0; --i) { const Index index = op.indices()[i]; @@ -366,48 +353,46 @@ struct TensorEvaluator= 0; --i) { - m_outputStride[i] = m_outputStride[i + 1] * m_dimensions[i + 1]; - } + for (int i = NumDims - 2; i >= 0; --i) { m_outputStride[i] = m_outputStride[i + 1] * m_dimensions[i + 1]; } } } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const Dimensions& dimensions() const { return m_dimensions; } + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const Dimensions &dimensions() const { return m_dimensions; } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE bool evalSubExprsIfNeeded(Scalar*) { + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE bool evalSubExprsIfNeeded(Scalar *) + { m_inputImpl.evalSubExprsIfNeeded(NULL); preloadKernel(); return true; } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void cleanup() { + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void cleanup() + { m_inputImpl.cleanup(); if (m_local_kernel) { - m_device.deallocate((void*)m_kernel); + m_device.deallocate((void *)m_kernel); m_local_kernel = false; } m_kernel = NULL; } - void evalTo(typename XprType::Scalar* buffer) { + void evalTo(typename XprType::Scalar *buffer) + { evalSubExprsIfNeeded(NULL); - for (int i = 0; i < dimensions().TotalSize(); ++i) { - buffer[i] += coeff(i); - } + for (int i = 0; i < dimensions().TotalSize(); ++i) { buffer[i] += coeff(i); } cleanup(); } EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE CoeffReturnType coeff(Index index) const { CoeffReturnType result = CoeffReturnType(0); - convolve(firstInput(index), 0, NumKernelDims-1, result); + convolve(firstInput(index), 0, NumKernelDims - 1, result); return result; } - template - EIGEN_DEVICE_FUNC PacketReturnType packet(const Index index) const + template EIGEN_DEVICE_FUNC PacketReturnType packet(const Index index) const { - Index indices[2] = {index, index+PacketSize-1}; - Index startInputs[2] = {0, 0}; + Index indices[2] = { index, index + PacketSize - 1 }; + Index startInputs[2] = { 0, 0 }; if (static_cast(Layout) == static_cast(ColMajor)) { for (int i = NumDims - 1; i > 0; --i) { const Index idx0 = indices[0] / m_outputStride[i]; @@ -430,45 +415,43 @@ struct TensorEvaluator(0); - convolvePacket(startInputs[0], 0, NumKernelDims-1, result); + convolvePacket(startInputs[0], 0, NumKernelDims - 1, result); return result; } else { EIGEN_ALIGN_MAX Scalar data[PacketSize]; data[0] = Scalar(0); - convolve(startInputs[0], 0, NumKernelDims-1, data[0]); - for (int i = 1; i < PacketSize-1; ++i) { + convolve(startInputs[0], 0, NumKernelDims - 1, data[0]); + for (int i = 1; i < PacketSize - 1; ++i) { data[i] = Scalar(0); - convolve(firstInput(index+i), 0, NumKernelDims-1, data[i]); + convolve(firstInput(index + i), 0, NumKernelDims - 1, data[i]); } - data[PacketSize-1] = Scalar(0); - convolve(startInputs[1], 0, NumKernelDims-1, data[PacketSize-1]); + data[PacketSize - 1] = Scalar(0); + convolve(startInputs[1], 0, NumKernelDims - 1, data[PacketSize - 1]); return internal::pload(data); } } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TensorOpCost - costPerCoeff(bool vectorized) const { + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TensorOpCost costPerCoeff(bool vectorized) const + { const double kernel_size = m_kernelImpl.dimensions().TotalSize(); // We ignore the use of fused multiply-add. - const double convolve_compute_cost = - TensorOpCost::AddCost() + TensorOpCost::MulCost(); + const double convolve_compute_cost = TensorOpCost::AddCost() + TensorOpCost::MulCost(); const double firstIndex_compute_cost = - NumDims * - (2 * TensorOpCost::AddCost() + 2 * TensorOpCost::MulCost() + - TensorOpCost::DivCost()); - return TensorOpCost(0, 0, firstIndex_compute_cost, vectorized, PacketSize) + - kernel_size * (m_inputImpl.costPerCoeff(vectorized) + - m_kernelImpl.costPerCoeff(vectorized) + - TensorOpCost(0, 0, convolve_compute_cost, vectorized, - PacketSize)); + NumDims + * (2 * TensorOpCost::AddCost() + 2 * TensorOpCost::MulCost() + TensorOpCost::DivCost()); + return TensorOpCost(0, 0, firstIndex_compute_cost, vectorized, PacketSize) + + kernel_size + * (m_inputImpl.costPerCoeff(vectorized) + m_kernelImpl.costPerCoeff(vectorized) + + TensorOpCost(0, 0, convolve_compute_cost, vectorized, PacketSize)); } - EIGEN_DEVICE_FUNC Scalar* data() const { return NULL; } + EIGEN_DEVICE_FUNC Scalar *data() const { return NULL; } - private: - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Index firstInput(Index index) const { +private: + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Index firstInput(Index index) const + { Index startInput = 0; if (static_cast(Layout) == static_cast(ColMajor)) { for (int i = NumDims - 1; i > 0; --i) { @@ -487,41 +470,45 @@ struct TensorEvaluator 0) { - convolve(input, kernel, DimIndex-1, accum); + convolve(input, kernel, DimIndex - 1, accum); } else { accum += m_inputImpl.coeff(input) * m_kernel[kernel]; } } } - template - EIGEN_DEVICE_FUNC void convolvePacket(Index firstIndex, Index firstKernel, int DimIndex, Packet& accum) const { + template + EIGEN_DEVICE_FUNC void convolvePacket(Index firstIndex, Index firstKernel, int DimIndex, Packet &accum) const + { for (int j = 0; j < m_kernelImpl.dimensions()[DimIndex]; ++j) { const Index input = firstIndex + j * m_indexStride[DimIndex]; const Index kernel = firstKernel + j * m_kernelStride[DimIndex]; if (DimIndex > 0) { - convolvePacket(input, kernel, DimIndex-1, accum); + convolvePacket(input, kernel, DimIndex - 1, accum); } else { - accum = internal::pmadd(m_inputImpl.template packet(input), internal::pset1(m_kernel[kernel]), accum); + accum = internal::pmadd( + m_inputImpl.template packet(input), internal::pset1(m_kernel[kernel]), accum); } } } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void preloadKernel() { + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void preloadKernel() + { // Don't make a local copy of the kernel unless we have to (i.e. it's an // expression that needs to be evaluated) - const Scalar* in_place = m_kernelImpl.data(); + const Scalar *in_place = m_kernelImpl.data(); if (in_place) { m_kernel = in_place; m_local_kernel = false; } else { size_t kernel_sz = m_kernelImpl.dimensions().TotalSize() * sizeof(Scalar); - Scalar* local = (Scalar*)m_device.allocate(kernel_sz); + Scalar *local = (Scalar *)m_device.allocate(kernel_sz); typedef TensorEvalToOp EvalTo; EvalTo evalToTmp(local, m_kernelArg); const bool PacketAccess = internal::IsVectorizable::value; @@ -542,38 +529,34 @@ struct TensorEvaluator -struct GetKernelSize { - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE int operator() (const int /*kernelSize*/) const { - return StaticKernelSize; - } +template struct GetKernelSize +{ + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE int operator()(const int /*kernelSize*/) const { return StaticKernelSize; } }; -template <> -struct GetKernelSize { - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE int operator() (const int kernelSize) const { - return kernelSize; - } +template<> struct GetKernelSize +{ + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE int operator()(const int kernelSize) const { return kernelSize; } }; -template -__global__ void EigenConvolutionKernel1D( - InputEvaluator eval, - const internal::IndexMapper - indexMapper, - const float* __restrict kernel, const int numPlanes, const int numX, - const int maxX, const int kernelSize, float* buffer) { +template +__global__ void EigenConvolutionKernel1D(InputEvaluator eval, + const internal::IndexMapper indexMapper, + const float *__restrict kernel, + const int numPlanes, + const int numX, + const int maxX, + const int kernelSize, + float *buffer) +{ extern __shared__ float s[]; const int first_x = blockIdx.x * maxX; @@ -588,9 +571,9 @@ __global__ void EigenConvolutionKernel1D( // Load inputs to shared memory const int plane_input_offset = indexMapper.mapCudaInputPlaneToTensorInputOffset(p); const int plane_kernel_offset = threadIdx.y * num_x_input; - #pragma unroll +#pragma unroll for (int i = threadIdx.x; i < num_x_input; i += blockDim.x) { - const int tensor_index = plane_input_offset + indexMapper.mapCudaInputKernelToTensorInputOffset(i+first_x); + const int tensor_index = plane_input_offset + indexMapper.mapCudaInputKernelToTensorInputOffset(i + first_x); s[i + plane_kernel_offset] = eval.coeff(tensor_index); } @@ -599,30 +582,34 @@ __global__ void EigenConvolutionKernel1D( // Compute the convolution const int plane_output_offset = indexMapper.mapCudaOutputPlaneToTensorOutputOffset(p); - #pragma unroll +#pragma unroll for (int i = threadIdx.x; i < num_x_output; i += blockDim.x) { const int kernel_offset = plane_kernel_offset + i; float result = 0.0f; - #pragma unroll +#pragma unroll for (int k = 0; k < GetKernelSize()(kernelSize); ++k) { result += s[k + kernel_offset] * kernel[k]; } - const int tensor_index = plane_output_offset + indexMapper.mapCudaOutputKernelToTensorOutputOffset(i+first_x); + const int tensor_index = plane_output_offset + indexMapper.mapCudaOutputKernelToTensorOutputOffset(i + first_x); buffer[tensor_index] = result; } __syncthreads(); } }; -template -__global__ void EigenConvolutionKernel2D( - InputEvaluator eval, - const internal::IndexMapper - indexMapper, - const float* __restrict kernel, const int numPlanes, const int numX, - const int maxX, const int numY, const int maxY, const int kernelSizeX, - const int kernelSizeY, float* buffer) { +template +__global__ void EigenConvolutionKernel2D(InputEvaluator eval, + const internal::IndexMapper indexMapper, + const float *__restrict kernel, + const int numPlanes, + const int numX, + const int maxX, + const int numY, + const int maxY, + const int kernelSizeX, + const int kernelSizeY, + float *buffer) +{ extern __shared__ float s[]; const int first_x = blockIdx.x * maxX; @@ -643,13 +630,14 @@ __global__ void EigenConvolutionKernel2D( const int plane_input_offset = indexMapper.mapCudaInputPlaneToTensorInputOffset(p); const int plane_kernel_offset = threadIdx.z * num_y_input; - // Load inputs to shared memory - #pragma unroll +// Load inputs to shared memory +#pragma unroll for (int j = threadIdx.y; j < num_y_input; j += blockDim.y) { const int input_offset = num_x_input * (j + plane_kernel_offset); - #pragma unroll +#pragma unroll for (int i = threadIdx.x; i < num_x_input; i += blockDim.x) { - const int tensor_index = plane_input_offset + indexMapper.mapCudaInputKernelToTensorInputOffset(i+first_x, j+first_y); + const int tensor_index = + plane_input_offset + indexMapper.mapCudaInputKernelToTensorInputOffset(i + first_x, j + first_y); s[i + input_offset] = eval.coeff(tensor_index); } } @@ -659,21 +647,22 @@ __global__ void EigenConvolutionKernel2D( // Convolution const int plane_output_offset = indexMapper.mapCudaOutputPlaneToTensorOutputOffset(p); - #pragma unroll +#pragma unroll for (int j = threadIdx.y; j < num_y_output; j += blockDim.y) { - #pragma unroll +#pragma unroll for (int i = threadIdx.x; i < num_x_output; i += blockDim.x) { float result = 0.0f; - #pragma unroll +#pragma unroll for (int l = 0; l < GetKernelSize()(kernelSizeY); ++l) { const int kernel_offset = kernelSizeX * l; const int input_offset = i + num_x_input * (j + l + plane_kernel_offset); - #pragma unroll +#pragma unroll for (int k = 0; k < GetKernelSize()(kernelSizeX); ++k) { result += s[k + input_offset] * kernel[k + kernel_offset]; } } - const int tensor_index = plane_output_offset + indexMapper.mapCudaOutputKernelToTensorOutputOffset(i+first_x, j+first_y); + const int tensor_index = + plane_output_offset + indexMapper.mapCudaOutputKernelToTensorOutputOffset(i + first_x, j + first_y); buffer[tensor_index] = result; } } @@ -682,15 +671,22 @@ __global__ void EigenConvolutionKernel2D( } }; -template -__global__ void EigenConvolutionKernel3D( - InputEvaluator eval, - const internal::IndexMapper - indexMapper, - const float* __restrict kernel, const size_t numPlanes, const size_t numX, - const size_t maxX, const size_t numY, const size_t maxY, const size_t numZ, - const size_t maxZ, const size_t kernelSizeX, const size_t kernelSizeY, - const size_t kernelSizeZ, float* buffer) { +template +__global__ void EigenConvolutionKernel3D(InputEvaluator eval, + const internal::IndexMapper indexMapper, + const float *__restrict kernel, + const size_t numPlanes, + const size_t numX, + const size_t maxX, + const size_t numY, + const size_t maxY, + const size_t numZ, + const size_t maxZ, + const size_t kernelSizeX, + const size_t kernelSizeY, + const size_t kernelSizeZ, + float *buffer) +{ extern __shared__ float s[]; // Load inputs to shared memory @@ -714,7 +710,9 @@ __global__ void EigenConvolutionKernel3D( for (int k = threadIdx.z; k < num_z_input; k += blockDim.z) { for (int j = threadIdx.y; j < num_y_input; j += blockDim.y) { for (int i = threadIdx.x; i < num_x_input; i += blockDim.x) { - const int tensor_index = plane_input_offset + indexMapper.mapCudaInputKernelToTensorInputOffset(i+first_x, j+first_y, k+first_z); + const int tensor_index = + plane_input_offset + + indexMapper.mapCudaInputKernelToTensorInputOffset(i + first_x, j + first_y, k + first_z); s[i + num_x_input * (j + num_y_input * (k + plane_kernel_offset))] = eval.coeff(tensor_index); } } @@ -735,11 +733,14 @@ __global__ void EigenConvolutionKernel3D( for (int n = 0; n < kernelSizeZ; ++n) { for (int m = 0; m < kernelSizeY; ++m) { for (int l = 0; l < kernelSizeX; ++l) { - result += s[i + l + num_x_input * (j + m + num_y_input * (k + n + plane_kernel_offset))] * kernel[l + kernelSizeX * (m + kernelSizeY * n)]; + result += s[i + l + num_x_input * (j + m + num_y_input * (k + n + plane_kernel_offset))] + * kernel[l + kernelSizeX * (m + kernelSizeY * n)]; } } } - const int tensor_index = plane_output_offset + indexMapper.mapCudaOutputKernelToTensorOutputOffset(i+first_x, j+first_y, k+first_z); + const int tensor_index = + plane_output_offset + + indexMapper.mapCudaOutputKernelToTensorOutputOffset(i + first_x, j + first_y, k + first_z); buffer[tensor_index] = result; } } @@ -749,33 +750,37 @@ __global__ void EigenConvolutionKernel3D( }; - template struct TensorEvaluator, GpuDevice> { typedef TensorConvolutionOp XprType; - static const int NumDims = internal::array_size::Dimensions>::value; + static const int NumDims = internal::array_size::Dimensions>::value; static const int NumKernelDims = internal::array_size::value; typedef typename XprType::Index Index; typedef DSizes Dimensions; typedef typename TensorEvaluator::Dimensions KernelDimensions; enum { - IsAligned = TensorEvaluator::IsAligned & TensorEvaluator::IsAligned, + IsAligned = + TensorEvaluator::IsAligned & TensorEvaluator::IsAligned, PacketAccess = false, Layout = TensorEvaluator::Layout, - CoordAccess = false, // to be implemented + CoordAccess = false,// to be implemented RawAccess = false }; - EIGEN_DEVICE_FUNC TensorEvaluator(const XprType& op, const GpuDevice& device) - : m_inputImpl(op.inputExpression(), device), m_kernelArg(op.kernelExpression()), m_kernelImpl(op.kernelExpression(), device), m_indices(op.indices()), m_buf(NULL), m_kernel(NULL), m_local_kernel(false), m_device(device) + EIGEN_DEVICE_FUNC TensorEvaluator(const XprType &op, const GpuDevice &device) + : m_inputImpl(op.inputExpression(), device), m_kernelArg(op.kernelExpression()), + m_kernelImpl(op.kernelExpression(), device), m_indices(op.indices()), m_buf(NULL), m_kernel(NULL), + m_local_kernel(false), m_device(device) { - EIGEN_STATIC_ASSERT((static_cast(TensorEvaluator::Layout) == static_cast(TensorEvaluator::Layout)), YOU_MADE_A_PROGRAMMING_MISTAKE); + EIGEN_STATIC_ASSERT((static_cast(TensorEvaluator::Layout) + == static_cast(TensorEvaluator::Layout)), + YOU_MADE_A_PROGRAMMING_MISTAKE); - const typename TensorEvaluator::Dimensions& input_dims = m_inputImpl.dimensions(); - const typename TensorEvaluator::Dimensions& kernel_dims = m_kernelImpl.dimensions(); + const typename TensorEvaluator::Dimensions &input_dims = m_inputImpl.dimensions(); + const typename TensorEvaluator::Dimensions &kernel_dims = m_kernelImpl.dimensions(); m_dimensions = m_inputImpl.dimensions(); for (int i = 0; i < NumKernelDims; ++i) { @@ -792,44 +797,47 @@ struct TensorEvaluator::size; - EIGEN_DEVICE_FUNC const Dimensions& dimensions() const { return m_dimensions; } + EIGEN_DEVICE_FUNC const Dimensions &dimensions() const { return m_dimensions; } - EIGEN_STRONG_INLINE bool evalSubExprsIfNeeded(Scalar* data) { + EIGEN_STRONG_INLINE bool evalSubExprsIfNeeded(Scalar *data) + { preloadKernel(); m_inputImpl.evalSubExprsIfNeeded(NULL); if (data) { executeEval(data); return false; } else { - m_buf = (Scalar*)m_device.allocate(dimensions().TotalSize() * sizeof(Scalar)); + m_buf = (Scalar *)m_device.allocate(dimensions().TotalSize() * sizeof(Scalar)); executeEval(m_buf); return true; } } - EIGEN_STRONG_INLINE void cleanup() { + EIGEN_STRONG_INLINE void cleanup() + { m_inputImpl.cleanup(); if (m_buf) { m_device.deallocate(m_buf); m_buf = NULL; } if (m_local_kernel) { - m_device.deallocate((void*)m_kernel); + m_device.deallocate((void *)m_kernel); m_local_kernel = false; } m_kernel = NULL; } - EIGEN_STRONG_INLINE void preloadKernel() { + EIGEN_STRONG_INLINE void preloadKernel() + { // Don't make a local copy of the kernel unless we have to (i.e. it's an // expression that needs to be evaluated) - const Scalar* in_place = m_kernelImpl.data(); + const Scalar *in_place = m_kernelImpl.data(); if (in_place) { m_kernel = in_place; m_local_kernel = false; } else { size_t kernel_sz = m_kernelImpl.dimensions().TotalSize() * sizeof(Scalar); - Scalar* local = (Scalar*)m_device.allocate(kernel_sz); + Scalar *local = (Scalar *)m_device.allocate(kernel_sz); typedef TensorEvalToOp EvalTo; EvalTo evalToTmp(local, m_kernelArg); const bool PacketAccess = internal::IsVectorizable::value; @@ -840,15 +848,15 @@ struct TensorEvaluator rounded_toward_zero * denom) { - return rounded_toward_zero + 1; - } + if (num > rounded_toward_zero * denom) { return rounded_toward_zero + 1; } return rounded_toward_zero; } - void executeEval(Scalar* data) const { + void executeEval(Scalar *data) const + { typedef typename TensorEvaluator::Dimensions InputDims; const int maxSharedMem = m_device.sharedMemPerBlock(); @@ -858,192 +866,332 @@ struct TensorEvaluator(Layout) == static_cast(ColMajor) - ? 0 - : m_inputImpl.dimensions().rank() - 1; - if (m_indices[0] == single_stride_dim) { - // Maximum the reuse - const int inner_dim = ((maxSharedMem / (sizeof(Scalar)) - kernel_size + 1 + 31) / 32) * 32; - maxX = numext::mini(inner_dim, numX); - const int maxP = numext::mini(maxSharedMem / ((kernel_size - 1 + maxX) * sizeof(Scalar)), numP); - block_size.x = numext::mini(maxThreadsPerBlock, maxX); - block_size.y = numext::mini(maxThreadsPerBlock / block_size.x, maxP); - } - else { - // Read as much as possible alongside the inner most dimension, that is the plane - const int inner_dim = maxSharedMem / ((warpSize + kernel_size) * sizeof(Scalar)); - const int maxP = numext::mini(inner_dim, numP); - maxX = numext::mini(maxSharedMem / (inner_dim * sizeof(Scalar)) - kernel_size + 1, numX); - - block_size.x = numext::mini(warpSize, maxX); - block_size.y = numext::mini(maxThreadsPerBlock/block_size.x, maxP); - } - - const int shared_mem = block_size.y * (maxX + kernel_size - 1) * sizeof(Scalar); - assert(shared_mem <= maxSharedMem); - - const int num_x_blocks = ceil(numX, maxX); - const int blocksPerProcessor = numext::mini(maxBlocksPerProcessor, maxSharedMem / shared_mem); - const int num_y_blocks = ceil(numMultiProcessors * blocksPerProcessor, num_x_blocks); - - dim3 num_blocks(num_x_blocks, numext::mini(num_y_blocks, ceil(numP, block_size.y))); - + case 1: { + const int kernel_size = m_kernelImpl.dimensions().TotalSize(); + + const int numX = dimensions()[m_indices[0]]; + const int numP = dimensions().TotalSize() / numX; + int maxX; + dim3 block_size; + + const int single_stride_dim = + static_cast(Layout) == static_cast(ColMajor) ? 0 : m_inputImpl.dimensions().rank() - 1; + if (m_indices[0] == single_stride_dim) { + // Maximum the reuse + const int inner_dim = ((maxSharedMem / (sizeof(Scalar)) - kernel_size + 1 + 31) / 32) * 32; + maxX = numext::mini(inner_dim, numX); + const int maxP = numext::mini(maxSharedMem / ((kernel_size - 1 + maxX) * sizeof(Scalar)), numP); + block_size.x = numext::mini(maxThreadsPerBlock, maxX); + block_size.y = numext::mini(maxThreadsPerBlock / block_size.x, maxP); + } else { + // Read as much as possible alongside the inner most dimension, that is the plane + const int inner_dim = maxSharedMem / ((warpSize + kernel_size) * sizeof(Scalar)); + const int maxP = numext::mini(inner_dim, numP); + maxX = numext::mini(maxSharedMem / (inner_dim * sizeof(Scalar)) - kernel_size + 1, numX); - //cout << "launching 1D kernel with block_size.x: " << block_size.x << " block_size.y: " << block_size.y << " num_blocks.x: " << num_blocks.x << " num_blocks.y: " << num_blocks.y << " maxX: " << maxX << " shared_mem: " << shared_mem << " in stream " << m_device.stream() << endl; + block_size.x = numext::mini(warpSize, maxX); + block_size.y = numext::mini(maxThreadsPerBlock / block_size.x, maxP); + } - const array indices(m_indices[0]); - const array kernel_dims(m_kernelImpl.dimensions()[0]); - internal::IndexMapper indexMapper( - m_inputImpl.dimensions(), kernel_dims, indices); - switch(kernel_size) { - case 4: { - LAUNCH_CUDA_KERNEL((EigenConvolutionKernel1D, Index, InputDims, 4>), num_blocks, block_size, shared_mem, m_device, m_inputImpl, indexMapper, m_kernel, numP, numX, maxX, 4, data); - break; - } - case 7: { - LAUNCH_CUDA_KERNEL((EigenConvolutionKernel1D, Index, InputDims, 7>), num_blocks, block_size, shared_mem, m_device, m_inputImpl, indexMapper, m_kernel, numP, numX, maxX, 7, data); - break; - } - default: { - LAUNCH_CUDA_KERNEL((EigenConvolutionKernel1D, Index, InputDims, Dynamic>), num_blocks, block_size, shared_mem, m_device, m_inputImpl, indexMapper, m_kernel, numP, numX, maxX, kernel_size, data); - } - } + const int shared_mem = block_size.y * (maxX + kernel_size - 1) * sizeof(Scalar); + assert(shared_mem <= maxSharedMem); + + const int num_x_blocks = ceil(numX, maxX); + const int blocksPerProcessor = numext::mini(maxBlocksPerProcessor, maxSharedMem / shared_mem); + const int num_y_blocks = ceil(numMultiProcessors * blocksPerProcessor, num_x_blocks); + + dim3 num_blocks(num_x_blocks, numext::mini(num_y_blocks, ceil(numP, block_size.y))); + + + // cout << "launching 1D kernel with block_size.x: " << block_size.x << " block_size.y: " << block_size.y << " + // num_blocks.x: " << num_blocks.x << " num_blocks.y: " << num_blocks.y << " maxX: " << maxX << " shared_mem: " << + // shared_mem << " in stream " << m_device.stream() << endl; + + const array indices(m_indices[0]); + const array kernel_dims(m_kernelImpl.dimensions()[0]); + internal::IndexMapper indexMapper(m_inputImpl.dimensions(), kernel_dims, indices); + switch (kernel_size) { + case 4: { + LAUNCH_CUDA_KERNEL((EigenConvolutionKernel1D, Index, InputDims, 4>), + num_blocks, + block_size, + shared_mem, + m_device, + m_inputImpl, + indexMapper, + m_kernel, + numP, + numX, + maxX, + 4, + data); break; } + case 7: { + LAUNCH_CUDA_KERNEL((EigenConvolutionKernel1D, Index, InputDims, 7>), + num_blocks, + block_size, + shared_mem, + m_device, + m_inputImpl, + indexMapper, + m_kernel, + numP, + numX, + maxX, + 7, + data); + break; + } + default: { + LAUNCH_CUDA_KERNEL( + (EigenConvolutionKernel1D, Index, InputDims, Dynamic>), + num_blocks, + block_size, + shared_mem, + m_device, + m_inputImpl, + indexMapper, + m_kernel, + numP, + numX, + maxX, + kernel_size, + data); + } + } + break; + } - case 2: { - const int idxX = - static_cast(Layout) == static_cast(ColMajor) ? 0 : 1; - const int idxY = - static_cast(Layout) == static_cast(ColMajor) ? 1 : 0; - const int kernel_size_x = m_kernelImpl.dimensions()[idxX]; - const int kernel_size_y = m_kernelImpl.dimensions()[idxY]; - - const int numX = dimensions()[m_indices[idxX]]; - const int numY = dimensions()[m_indices[idxY]]; - const int numP = dimensions().TotalSize() / (numX*numY); - - const float scaling_factor = sqrtf(static_cast(maxSharedMem) / (sizeof(Scalar) * kernel_size_y * kernel_size_x)); - - // Snap maxX to warp size - int inner_dim = ((static_cast(scaling_factor * kernel_size_x) - kernel_size_x + 1 + 32) / 32) * 32; - const int maxX = numext::mini(inner_dim, numX); - const int maxY = numext::mini(maxSharedMem / (sizeof(Scalar) * (maxX + kernel_size_x - 1)) - kernel_size_y + 1, numY); - const int maxP = numext::mini(maxSharedMem / ((kernel_size_x - 1 + maxX) * (kernel_size_y - 1 + maxY) * sizeof(Scalar)), numP); - - dim3 block_size; - block_size.x = numext::mini(1024, maxX); - block_size.y = numext::mini(1024/block_size.x, maxY); - block_size.z = numext::mini(1024/(block_size.x*block_size.y), maxP); - - const int shared_mem = block_size.z * (maxX + kernel_size_x - 1) * (maxY + kernel_size_y - 1) * sizeof(Scalar); - assert(shared_mem <= maxSharedMem); - - const int num_x_blocks = ceil(numX, maxX); - const int num_y_blocks = ceil(numY, maxY); - const int blocksPerProcessor = numext::mini(maxBlocksPerProcessor, maxSharedMem / shared_mem); - const int num_z_blocks = ceil(numMultiProcessors * blocksPerProcessor, num_x_blocks * num_y_blocks); - - dim3 num_blocks(num_x_blocks, num_y_blocks, numext::mini(num_z_blocks, ceil(numP, block_size.z))); - - - //cout << "launching 2D kernel with block_size.x: " << block_size.x << " block_size.y: " << block_size.y << " block_size.z: " << block_size.z << " num_blocks.x: " << num_blocks.x << " num_blocks.y: " << num_blocks.y << " num_blocks.z: " << num_blocks.z << " maxX: " << maxX << " maxY: " << maxY << " maxP: " << maxP << " shared_mem: " << shared_mem << " in stream " << m_device.stream() << endl; - - const array indices(m_indices[idxX], m_indices[idxY]); - const array kernel_dims(m_kernelImpl.dimensions()[idxX], - m_kernelImpl.dimensions()[idxY]); - internal::IndexMapper indexMapper( - m_inputImpl.dimensions(), kernel_dims, indices); - switch (kernel_size_x) { - case 4: { - switch (kernel_size_y) { - case 7: { - LAUNCH_CUDA_KERNEL((EigenConvolutionKernel2D, Index, InputDims, 4, 7>), num_blocks, block_size, shared_mem, m_device, m_inputImpl, indexMapper, m_kernel, numP, numX, maxX, numY, maxY, 4, 7, data); - break; - } - default: { - LAUNCH_CUDA_KERNEL((EigenConvolutionKernel2D, Index, InputDims, 4, Dynamic>), num_blocks, block_size, shared_mem, m_device, m_inputImpl, indexMapper, m_kernel, numP, numX, maxX, numY, maxY, 4, kernel_size_y, data); - break; - } - } - break; - } - case 7: { - switch (kernel_size_y) { - case 4: { - LAUNCH_CUDA_KERNEL((EigenConvolutionKernel2D, Index, InputDims, 7, 4>), num_blocks, block_size, shared_mem, m_device, m_inputImpl, indexMapper, m_kernel, numP, numX, maxX, numY, maxY, 7, 4, data); - break; - } - default: { - LAUNCH_CUDA_KERNEL((EigenConvolutionKernel2D, Index, InputDims, 7, Dynamic>), num_blocks, block_size, shared_mem, m_device, m_inputImpl, indexMapper, m_kernel, numP, numX, maxX, numY, maxY, 7, kernel_size_y, data); - break; - } - } - break; - } - default: { - LAUNCH_CUDA_KERNEL((EigenConvolutionKernel2D, Index, InputDims, Dynamic, Dynamic>), num_blocks, block_size, shared_mem, m_device, m_inputImpl, indexMapper, m_kernel, numP, numX, maxX, numY, maxY, kernel_size_x, kernel_size_y, data); - break; - } + case 2: { + const int idxX = static_cast(Layout) == static_cast(ColMajor) ? 0 : 1; + const int idxY = static_cast(Layout) == static_cast(ColMajor) ? 1 : 0; + const int kernel_size_x = m_kernelImpl.dimensions()[idxX]; + const int kernel_size_y = m_kernelImpl.dimensions()[idxY]; + + const int numX = dimensions()[m_indices[idxX]]; + const int numY = dimensions()[m_indices[idxY]]; + const int numP = dimensions().TotalSize() / (numX * numY); + + const float scaling_factor = + sqrtf(static_cast(maxSharedMem) / (sizeof(Scalar) * kernel_size_y * kernel_size_x)); + + // Snap maxX to warp size + int inner_dim = ((static_cast(scaling_factor * kernel_size_x) - kernel_size_x + 1 + 32) / 32) * 32; + const int maxX = numext::mini(inner_dim, numX); + const int maxY = + numext::mini(maxSharedMem / (sizeof(Scalar) * (maxX + kernel_size_x - 1)) - kernel_size_y + 1, numY); + const int maxP = numext::mini( + maxSharedMem / ((kernel_size_x - 1 + maxX) * (kernel_size_y - 1 + maxY) * sizeof(Scalar)), numP); + + dim3 block_size; + block_size.x = numext::mini(1024, maxX); + block_size.y = numext::mini(1024 / block_size.x, maxY); + block_size.z = numext::mini(1024 / (block_size.x * block_size.y), maxP); + + const int shared_mem = block_size.z * (maxX + kernel_size_x - 1) * (maxY + kernel_size_y - 1) * sizeof(Scalar); + assert(shared_mem <= maxSharedMem); + + const int num_x_blocks = ceil(numX, maxX); + const int num_y_blocks = ceil(numY, maxY); + const int blocksPerProcessor = numext::mini(maxBlocksPerProcessor, maxSharedMem / shared_mem); + const int num_z_blocks = ceil(numMultiProcessors * blocksPerProcessor, num_x_blocks * num_y_blocks); + + dim3 num_blocks(num_x_blocks, num_y_blocks, numext::mini(num_z_blocks, ceil(numP, block_size.z))); + + + // cout << "launching 2D kernel with block_size.x: " << block_size.x << " block_size.y: " << block_size.y << " + // block_size.z: " << block_size.z << " num_blocks.x: " << num_blocks.x << " num_blocks.y: " << num_blocks.y << " + // num_blocks.z: " << num_blocks.z << " maxX: " << maxX << " maxY: " << maxY << " maxP: " << maxP << " shared_mem: + // " << shared_mem << " in stream " << m_device.stream() << endl; + + const array indices(m_indices[idxX], m_indices[idxY]); + const array kernel_dims(m_kernelImpl.dimensions()[idxX], m_kernelImpl.dimensions()[idxY]); + internal::IndexMapper indexMapper(m_inputImpl.dimensions(), kernel_dims, indices); + switch (kernel_size_x) { + case 4: { + switch (kernel_size_y) { + case 7: { + LAUNCH_CUDA_KERNEL( + (EigenConvolutionKernel2D, Index, InputDims, 4, 7>), + num_blocks, + block_size, + shared_mem, + m_device, + m_inputImpl, + indexMapper, + m_kernel, + numP, + numX, + maxX, + numY, + maxY, + 4, + 7, + data); + break; + } + default: { + LAUNCH_CUDA_KERNEL( + (EigenConvolutionKernel2D, Index, InputDims, 4, Dynamic>), + num_blocks, + block_size, + shared_mem, + m_device, + m_inputImpl, + indexMapper, + m_kernel, + numP, + numX, + maxX, + numY, + maxY, + 4, + kernel_size_y, + data); + break; + } } break; } - - case 3: { - const int idxX = - static_cast(Layout) == static_cast(ColMajor) ? 0 : 2; - const int idxY = - static_cast(Layout) == static_cast(ColMajor) ? 1 : 1; - const int idxZ = - static_cast(Layout) == static_cast(ColMajor) ? 2 : 0; - - const int kernel_size_x = m_kernelImpl.dimensions()[idxX]; - const int kernel_size_y = m_kernelImpl.dimensions()[idxY]; - const int kernel_size_z = m_kernelImpl.dimensions()[idxZ]; - - const int numX = dimensions()[m_indices[idxX]]; - const int numY = dimensions()[m_indices[idxY]]; - const int numZ = dimensions()[m_indices[idxZ]]; - const int numP = dimensions().TotalSize() / (numX*numY*numZ); - - const int maxX = numext::mini(128, numext::mini(maxSharedMem / (sizeof(Scalar) * kernel_size_y * kernel_size_z) - kernel_size_x + 1, numX)); - const int maxY = numext::mini(128, numext::mini(maxSharedMem / (sizeof(Scalar) * (maxX + kernel_size_x - 1) * kernel_size_z) - kernel_size_y + 1, numY)); - const int maxZ = numext::mini(128, numext::mini(maxSharedMem / (sizeof(Scalar) * (maxX + kernel_size_x - 1) * (maxY + kernel_size_y - 1)) - kernel_size_z + 1, numZ)); - - dim3 block_size; - block_size.x = numext::mini(32, maxX); - block_size.y = numext::mini(32, maxY); - block_size.z = numext::mini(1024/(block_size.x*block_size.y), maxZ); - dim3 num_blocks(ceil(numX, maxX), ceil(numY, maxY), ceil(numZ, maxZ)); - - const int shared_mem = (maxX + kernel_size_x - 1) * (maxY + kernel_size_y - 1) * (maxZ + kernel_size_z - 1) * sizeof(Scalar); - assert(shared_mem <= maxSharedMem); - - //cout << "launching 3D kernel with block_size.x: " << block_size.x << " block_size.y: " << block_size.y << " block_size.z: " << block_size.z << " num_blocks.x: " << num_blocks.x << " num_blocks.y: " << num_blocks.y << " num_blocks.z: " << num_blocks.z << " shared_mem: " << shared_mem << " in stream " << m_device.stream() << endl; - const array indices(m_indices[idxX], m_indices[idxY], - m_indices[idxZ]); - const array kernel_dims(m_kernelImpl.dimensions()[idxX], - m_kernelImpl.dimensions()[idxY], - m_kernelImpl.dimensions()[idxZ]); - internal::IndexMapper indexMapper( - m_inputImpl.dimensions(), kernel_dims, indices); - - LAUNCH_CUDA_KERNEL((EigenConvolutionKernel3D, Index, InputDims>), num_blocks, block_size, shared_mem, m_device, m_inputImpl, indexMapper, m_kernel, numP, numX, maxX, numY, maxY, numZ, maxZ, kernel_size_x, kernel_size_y, kernel_size_z, data); + case 7: { + switch (kernel_size_y) { + case 4: { + LAUNCH_CUDA_KERNEL( + (EigenConvolutionKernel2D, Index, InputDims, 7, 4>), + num_blocks, + block_size, + shared_mem, + m_device, + m_inputImpl, + indexMapper, + m_kernel, + numP, + numX, + maxX, + numY, + maxY, + 7, + 4, + data); + break; + } + default: { + LAUNCH_CUDA_KERNEL( + (EigenConvolutionKernel2D, Index, InputDims, 7, Dynamic>), + num_blocks, + block_size, + shared_mem, + m_device, + m_inputImpl, + indexMapper, + m_kernel, + numP, + numX, + maxX, + numY, + maxY, + 7, + kernel_size_y, + data); + break; + } + } break; } - default: { - EIGEN_STATIC_ASSERT((NumKernelDims >= 1 && NumKernelDims <= 3), THIS_METHOD_IS_ONLY_FOR_OBJECTS_OF_A_SPECIFIC_SIZE); + LAUNCH_CUDA_KERNEL( + (EigenConvolutionKernel2D, Index, InputDims, Dynamic, Dynamic>), + num_blocks, + block_size, + shared_mem, + m_device, + m_inputImpl, + indexMapper, + m_kernel, + numP, + numX, + maxX, + numY, + maxY, + kernel_size_x, + kernel_size_y, + data); + break; + } } + break; + } + + case 3: { + const int idxX = static_cast(Layout) == static_cast(ColMajor) ? 0 : 2; + const int idxY = static_cast(Layout) == static_cast(ColMajor) ? 1 : 1; + const int idxZ = static_cast(Layout) == static_cast(ColMajor) ? 2 : 0; + + const int kernel_size_x = m_kernelImpl.dimensions()[idxX]; + const int kernel_size_y = m_kernelImpl.dimensions()[idxY]; + const int kernel_size_z = m_kernelImpl.dimensions()[idxZ]; + + const int numX = dimensions()[m_indices[idxX]]; + const int numY = dimensions()[m_indices[idxY]]; + const int numZ = dimensions()[m_indices[idxZ]]; + const int numP = dimensions().TotalSize() / (numX * numY * numZ); + + const int maxX = numext::mini(128, + numext::mini(maxSharedMem / (sizeof(Scalar) * kernel_size_y * kernel_size_z) - kernel_size_x + 1, numX)); + const int maxY = numext::mini(128, + numext::mini( + maxSharedMem / (sizeof(Scalar) * (maxX + kernel_size_x - 1) * kernel_size_z) - kernel_size_y + 1, numY)); + const int maxZ = numext::mini(128, + numext::mini( + maxSharedMem / (sizeof(Scalar) * (maxX + kernel_size_x - 1) * (maxY + kernel_size_y - 1)) - kernel_size_z + 1, + numZ)); + + dim3 block_size; + block_size.x = numext::mini(32, maxX); + block_size.y = numext::mini(32, maxY); + block_size.z = numext::mini(1024 / (block_size.x * block_size.y), maxZ); + dim3 num_blocks(ceil(numX, maxX), ceil(numY, maxY), ceil(numZ, maxZ)); + + const int shared_mem = + (maxX + kernel_size_x - 1) * (maxY + kernel_size_y - 1) * (maxZ + kernel_size_z - 1) * sizeof(Scalar); + assert(shared_mem <= maxSharedMem); + + // cout << "launching 3D kernel with block_size.x: " << block_size.x << " block_size.y: " << block_size.y << " + // block_size.z: " << block_size.z << " num_blocks.x: " << num_blocks.x << " num_blocks.y: " << num_blocks.y << " + // num_blocks.z: " << num_blocks.z << " shared_mem: " << shared_mem << " in stream " << m_device.stream() << + // endl; + const array indices(m_indices[idxX], m_indices[idxY], m_indices[idxZ]); + const array kernel_dims( + m_kernelImpl.dimensions()[idxX], m_kernelImpl.dimensions()[idxY], m_kernelImpl.dimensions()[idxZ]); + internal::IndexMapper indexMapper(m_inputImpl.dimensions(), kernel_dims, indices); + + LAUNCH_CUDA_KERNEL((EigenConvolutionKernel3D, Index, InputDims>), + num_blocks, + block_size, + shared_mem, + m_device, + m_inputImpl, + indexMapper, + m_kernel, + numP, + numX, + maxX, + numY, + maxY, + numZ, + maxZ, + kernel_size_x, + kernel_size_y, + kernel_size_z, + data); + break; + } + + default: { + EIGEN_STATIC_ASSERT( + (NumKernelDims >= 1 && NumKernelDims <= 3), THIS_METHOD_IS_ONLY_FOR_OBJECTS_OF_A_SPECIFIC_SIZE); + } } } @@ -1054,51 +1202,47 @@ struct TensorEvaluator - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE PacketReturnType packet(const Index index) const + template EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE PacketReturnType packet(const Index index) const { eigen_assert(m_buf); eigen_assert(index < m_dimensions.TotalSize()); - return internal::ploadt(m_buf+index); + return internal::ploadt(m_buf + index); } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TensorOpCost - costPerCoeff(bool vectorized) const { + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TensorOpCost costPerCoeff(bool vectorized) const + { // TODO(rmlarsen): FIXME: For now, this is just a copy of the CPU cost // model. const double kernel_size = m_kernelImpl.dimensions().TotalSize(); // We ignore the use of fused multiply-add. - const double convolve_compute_cost = - TensorOpCost::AddCost() + TensorOpCost::MulCost(); + const double convolve_compute_cost = TensorOpCost::AddCost() + TensorOpCost::MulCost(); const double firstIndex_compute_cost = - NumDims * - (2 * TensorOpCost::AddCost() + 2 * TensorOpCost::MulCost() + - TensorOpCost::DivCost()); - return TensorOpCost(0, 0, firstIndex_compute_cost, vectorized, PacketSize) + - kernel_size * (m_inputImpl.costPerCoeff(vectorized) + - m_kernelImpl.costPerCoeff(vectorized) + - TensorOpCost(0, 0, convolve_compute_cost, vectorized, - PacketSize)); + NumDims + * (2 * TensorOpCost::AddCost() + 2 * TensorOpCost::MulCost() + TensorOpCost::DivCost()); + return TensorOpCost(0, 0, firstIndex_compute_cost, vectorized, PacketSize) + + kernel_size + * (m_inputImpl.costPerCoeff(vectorized) + m_kernelImpl.costPerCoeff(vectorized) + + TensorOpCost(0, 0, convolve_compute_cost, vectorized, PacketSize)); } - private: +private: // No assignment (copies are needed by the kernels) - TensorEvaluator& operator = (const TensorEvaluator&); + TensorEvaluator &operator=(const TensorEvaluator &); TensorEvaluator m_inputImpl; TensorEvaluator m_kernelImpl; KernelArgType m_kernelArg; Indices m_indices; Dimensions m_dimensions; - Scalar* m_buf; - const Scalar* m_kernel; + Scalar *m_buf; + const Scalar *m_kernel; bool m_local_kernel; - const GpuDevice& m_device; + const GpuDevice &m_device; }; #endif -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_CXX11_TENSOR_TENSOR_CONVOLUTION_H +#endif// EIGEN_CXX11_TENSOR_TENSOR_CONVOLUTION_H diff --git a/filmulator-gui/core/nlmeans/eigen/unsupported/Eigen/CXX11/src/Tensor/TensorCostModel.h b/filmulator-gui/core/nlmeans/eigen/unsupported/Eigen/CXX11/src/Tensor/TensorCostModel.h index 83c449cf..eee73e6d 100644 --- a/filmulator-gui/core/nlmeans/eigen/unsupported/Eigen/CXX11/src/Tensor/TensorCostModel.h +++ b/filmulator-gui/core/nlmeans/eigen/unsupported/Eigen/CXX11/src/Tensor/TensorCostModel.h @@ -13,89 +13,79 @@ namespace Eigen { /** \class TensorEvaluator - * \ingroup CXX11_Tensor_Module - * - * \brief A cost model used to limit the number of threads used for evaluating - * tensor expression. - * - */ + * \ingroup CXX11_Tensor_Module + * + * \brief A cost model used to limit the number of threads used for evaluating + * tensor expression. + * + */ // Class storing the cost of evaluating a tensor expression in terms of the // estimated number of operand bytes loads, bytes stored, and compute cycles. -class TensorOpCost { - public: +class TensorOpCost +{ +public: // TODO(rmlarsen): Fix the scalar op costs in Eigen proper. Even a simple // model based on minimal reciprocal throughput numbers from Intel or // Agner Fog's tables would be better than what is there now. - template - static EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE int MulCost() { - return internal::functor_traits< - internal::scalar_product_op >::Cost; + template static EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE int MulCost() + { + return internal::functor_traits>::Cost; } - template - static EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE int AddCost() { - return internal::functor_traits >::Cost; + template static EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE int AddCost() + { + return internal::functor_traits>::Cost; } - template - static EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE int DivCost() { - return internal::functor_traits< - internal::scalar_quotient_op >::Cost; + template static EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE int DivCost() + { + return internal::functor_traits>::Cost; } - template - static EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE int ModCost() { - return internal::functor_traits >::Cost; + template static EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE int ModCost() + { + return internal::functor_traits>::Cost; } - template - static EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE int CastCost() { - return internal::functor_traits< - internal::scalar_cast_op >::Cost; + template static EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE int CastCost() + { + return internal::functor_traits>::Cost; } EIGEN_DEVICE_FUNC TensorOpCost() : bytes_loaded_(0), bytes_stored_(0), compute_cycles_(0) {} EIGEN_DEVICE_FUNC TensorOpCost(double bytes_loaded, double bytes_stored, double compute_cycles) - : bytes_loaded_(bytes_loaded), - bytes_stored_(bytes_stored), - compute_cycles_(compute_cycles) {} + : bytes_loaded_(bytes_loaded), bytes_stored_(bytes_stored), compute_cycles_(compute_cycles) + {} EIGEN_DEVICE_FUNC - TensorOpCost(double bytes_loaded, double bytes_stored, double compute_cycles, - bool vectorized, double packet_size) - : bytes_loaded_(bytes_loaded), - bytes_stored_(bytes_stored), - compute_cycles_(vectorized ? compute_cycles / packet_size - : compute_cycles) { + TensorOpCost(double bytes_loaded, double bytes_stored, double compute_cycles, bool vectorized, double packet_size) + : bytes_loaded_(bytes_loaded), bytes_stored_(bytes_stored), + compute_cycles_(vectorized ? compute_cycles / packet_size : compute_cycles) + { eigen_assert(bytes_loaded >= 0 && (numext::isfinite)(bytes_loaded)); eigen_assert(bytes_stored >= 0 && (numext::isfinite)(bytes_stored)); eigen_assert(compute_cycles >= 0 && (numext::isfinite)(compute_cycles)); } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE double bytes_loaded() const { - return bytes_loaded_; - } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE double bytes_stored() const { - return bytes_stored_; - } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE double compute_cycles() const { - return compute_cycles_; - } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE double total_cost( - double load_cost, double store_cost, double compute_cost) const { - return load_cost * bytes_loaded_ + store_cost * bytes_stored_ + - compute_cost * compute_cycles_; + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE double bytes_loaded() const { return bytes_loaded_; } + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE double bytes_stored() const { return bytes_stored_; } + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE double compute_cycles() const { return compute_cycles_; } + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE double + total_cost(double load_cost, double store_cost, double compute_cost) const + { + return load_cost * bytes_loaded_ + store_cost * bytes_stored_ + compute_cost * compute_cycles_; } // Drop memory access component. Intended for cases when memory accesses are // sequential or are completely masked by computations. - EIGEN_DEVICE_FUNC void dropMemoryCost() { + EIGEN_DEVICE_FUNC void dropMemoryCost() + { bytes_loaded_ = 0; bytes_stored_ = 0; } // TODO(rmlarsen): Define min in terms of total cost, not elementwise. - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TensorOpCost cwiseMin( - const TensorOpCost& rhs) const { + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TensorOpCost cwiseMin(const TensorOpCost &rhs) const + { double bytes_loaded = numext::mini(bytes_loaded_, rhs.bytes_loaded()); double bytes_stored = numext::mini(bytes_stored_, rhs.bytes_stored()); double compute_cycles = numext::mini(compute_cycles_, rhs.compute_cycles()); @@ -103,52 +93,53 @@ class TensorOpCost { } // TODO(rmlarsen): Define max in terms of total cost, not elementwise. - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TensorOpCost cwiseMax( - const TensorOpCost& rhs) const { + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TensorOpCost cwiseMax(const TensorOpCost &rhs) const + { double bytes_loaded = numext::maxi(bytes_loaded_, rhs.bytes_loaded()); double bytes_stored = numext::maxi(bytes_stored_, rhs.bytes_stored()); double compute_cycles = numext::maxi(compute_cycles_, rhs.compute_cycles()); return TensorOpCost(bytes_loaded, bytes_stored, compute_cycles); } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TensorOpCost& operator+=( - const TensorOpCost& rhs) { + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TensorOpCost &operator+=(const TensorOpCost &rhs) + { bytes_loaded_ += rhs.bytes_loaded(); bytes_stored_ += rhs.bytes_stored(); compute_cycles_ += rhs.compute_cycles(); return *this; } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TensorOpCost& operator*=(double rhs) { + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TensorOpCost &operator*=(double rhs) + { bytes_loaded_ *= rhs; bytes_stored_ *= rhs; compute_cycles_ *= rhs; return *this; } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE friend TensorOpCost operator+( - TensorOpCost lhs, const TensorOpCost& rhs) { + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE friend TensorOpCost operator+(TensorOpCost lhs, const TensorOpCost &rhs) + { lhs += rhs; return lhs; } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE friend TensorOpCost operator*( - TensorOpCost lhs, double rhs) { + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE friend TensorOpCost operator*(TensorOpCost lhs, double rhs) + { lhs *= rhs; return lhs; } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE friend TensorOpCost operator*( - double lhs, TensorOpCost rhs) { + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE friend TensorOpCost operator*(double lhs, TensorOpCost rhs) + { rhs *= lhs; return rhs; } - friend std::ostream& operator<<(std::ostream& os, const TensorOpCost& tc) { - return os << "[bytes_loaded = " << tc.bytes_loaded() - << ", bytes_stored = " << tc.bytes_stored() + friend std::ostream &operator<<(std::ostream &os, const TensorOpCost &tc) + { + return os << "[bytes_loaded = " << tc.bytes_loaded() << ", bytes_stored = " << tc.bytes_stored() << ", compute_cycles = " << tc.compute_cycles() << "]"; } - private: +private: double bytes_loaded_; double bytes_stored_; double compute_cycles_; @@ -157,13 +148,13 @@ class TensorOpCost { // TODO(rmlarsen): Implement a policy that chooses an "optimal" number of theads // in [1:max_threads] instead of just switching multi-threading off for small // work units. -template -class TensorCostModel { - public: +template class TensorCostModel +{ +public: // Scaling from Eigen compute cost to device cycles. static const int kDeviceCyclesPerComputeCycle = 1; - // Costs in device cycles. + // Costs in device cycles. static const int kStartupCycles = 100000; static const int kPerThreadCycles = 100000; static const int kTaskSize = 40000; @@ -171,8 +162,9 @@ class TensorCostModel { // Returns the number of threads in [1:max_threads] to use for // evaluating an expression with the given output size and cost per // coefficient. - static EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE int numThreads( - double output_size, const TensorOpCost& cost_per_coeff, int max_threads) { + static EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE int + numThreads(double output_size, const TensorOpCost &cost_per_coeff, int max_threads) + { double cost = totalCost(output_size, cost_per_coeff); int threads = (cost - kStartupCycles) / kPerThreadCycles + 0.9; return numext::mini(max_threads, numext::maxi(1, threads)); @@ -181,14 +173,14 @@ class TensorCostModel { // taskSize assesses parallel task size. // Value of 1.0 means ideal parallel task size. Values < 1.0 mean that task // granularity needs to be increased to mitigate parallelization overheads. - static EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE double taskSize( - double output_size, const TensorOpCost& cost_per_coeff) { + static EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE double taskSize(double output_size, const TensorOpCost &cost_per_coeff) + { return totalCost(output_size, cost_per_coeff) / kTaskSize; } - private: - static EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE double totalCost( - double output_size, const TensorOpCost& cost_per_coeff) { +private: + static EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE double totalCost(double output_size, const TensorOpCost &cost_per_coeff) + { // Cost of memory fetches from L2 cache. 64 is typical cache line size. // 11 is L2 cache latency on Haswell. // We don't know whether data is in L1, L2 or L3. But we are most interested @@ -201,12 +193,10 @@ class TensorCostModel { const double kLoadCycles = 1.0 / 64 * 11; const double kStoreCycles = 1.0 / 64 * 11; // Scaling from Eigen compute cost to device cycles. - return output_size * - cost_per_coeff.total_cost(kLoadCycles, kStoreCycles, - kDeviceCyclesPerComputeCycle); + return output_size * cost_per_coeff.total_cost(kLoadCycles, kStoreCycles, kDeviceCyclesPerComputeCycle); } }; -} // namespace Eigen +}// namespace Eigen -#endif // EIGEN_CXX11_TENSOR_TENSOR_COST_MODEL_H +#endif// EIGEN_CXX11_TENSOR_TENSOR_COST_MODEL_H diff --git a/filmulator-gui/core/nlmeans/eigen/unsupported/Eigen/CXX11/src/Tensor/TensorCustomOp.h b/filmulator-gui/core/nlmeans/eigen/unsupported/Eigen/CXX11/src/Tensor/TensorCustomOp.h index e020d076..20520529 100644 --- a/filmulator-gui/core/nlmeans/eigen/unsupported/Eigen/CXX11/src/Tensor/TensorCustomOp.h +++ b/filmulator-gui/core/nlmeans/eigen/unsupported/Eigen/CXX11/src/Tensor/TensorCustomOp.h @@ -13,45 +13,42 @@ namespace Eigen { /** \class TensorCustomUnaryOp - * \ingroup CXX11_Tensor_Module - * - * \brief Tensor custom class. - * - * - */ + * \ingroup CXX11_Tensor_Module + * + * \brief Tensor custom class. + * + * + */ namespace internal { -template -struct traits > -{ - typedef typename XprType::Scalar Scalar; - typedef typename XprType::StorageKind StorageKind; - typedef typename XprType::Index Index; - typedef typename XprType::Nested Nested; - typedef typename remove_reference::type _Nested; - static const int NumDimensions = traits::NumDimensions; - static const int Layout = traits::Layout; -}; - -template -struct eval, Eigen::Dense> -{ - typedef const TensorCustomUnaryOp& type; -}; + template struct traits> + { + typedef typename XprType::Scalar Scalar; + typedef typename XprType::StorageKind StorageKind; + typedef typename XprType::Index Index; + typedef typename XprType::Nested Nested; + typedef typename remove_reference::type _Nested; + static const int NumDimensions = traits::NumDimensions; + static const int Layout = traits::Layout; + }; -template -struct nested > -{ - typedef TensorCustomUnaryOp type; -}; + template + struct eval, Eigen::Dense> + { + typedef const TensorCustomUnaryOp &type; + }; -} // end namespace internal + template struct nested> + { + typedef TensorCustomUnaryOp type; + }; +}// end namespace internal template class TensorCustomUnaryOp : public TensorBase, ReadOnlyAccessors> { - public: +public: typedef typename internal::traits::Scalar Scalar; typedef typename Eigen::NumTraits::Real RealScalar; typedef typename XprType::CoeffReturnType CoeffReturnType; @@ -59,19 +56,19 @@ class TensorCustomUnaryOp : public TensorBase::StorageKind StorageKind; typedef typename internal::traits::Index Index; - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TensorCustomUnaryOp(const XprType& expr, const CustomUnaryFunc& func) - : m_expr(expr), m_func(func) {} + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TensorCustomUnaryOp(const XprType &expr, const CustomUnaryFunc &func) + : m_expr(expr), m_func(func) + {} EIGEN_DEVICE_FUNC - const CustomUnaryFunc& func() const { return m_func; } + const CustomUnaryFunc &func() const { return m_func; } EIGEN_DEVICE_FUNC - const typename internal::remove_all::type& - expression() const { return m_expr; } + const typename internal::remove_all::type &expression() const { return m_expr; } - protected: - typename XprType::Nested m_expr; - const CustomUnaryFunc m_func; +protected: + typename XprType::Nested m_expr; + const CustomUnaryFunc m_func; }; @@ -93,115 +90,114 @@ struct TensorEvaluator, Devi PacketAccess = (internal::packet_traits::size > 1), BlockAccess = false, Layout = TensorEvaluator::Layout, - CoordAccess = false, // to be implemented + CoordAccess = false,// to be implemented RawAccess = false }; - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TensorEvaluator(const ArgType& op, const Device& device) - : m_op(op), m_device(device), m_result(NULL) + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TensorEvaluator(const ArgType &op, const Device &device) + : m_op(op), m_device(device), m_result(NULL) { m_dimensions = op.func().dimensions(op.expression()); } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const Dimensions& dimensions() const { return m_dimensions; } + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const Dimensions &dimensions() const { return m_dimensions; } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE bool evalSubExprsIfNeeded(CoeffReturnType* data) { + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE bool evalSubExprsIfNeeded(CoeffReturnType *data) + { if (data) { evalTo(data); return false; } else { - m_result = static_cast( - m_device.allocate(dimensions().TotalSize() * sizeof(Scalar))); + m_result = static_cast(m_device.allocate(dimensions().TotalSize() * sizeof(Scalar))); evalTo(m_result); return true; } } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void cleanup() { + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void cleanup() + { if (m_result != NULL) { m_device.deallocate(m_result); m_result = NULL; } } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE CoeffReturnType coeff(Index index) const { - return m_result[index]; - } + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE CoeffReturnType coeff(Index index) const { return m_result[index]; } - template - EIGEN_DEVICE_FUNC PacketReturnType packet(Index index) const { + template EIGEN_DEVICE_FUNC PacketReturnType packet(Index index) const + { return internal::ploadt(m_result + index); } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TensorOpCost costPerCoeff(bool vectorized) const { + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TensorOpCost costPerCoeff(bool vectorized) const + { // TODO(rmlarsen): Extend CustomOp API to return its cost estimate. return TensorOpCost(sizeof(CoeffReturnType), 0, 0, vectorized, PacketSize); } - EIGEN_DEVICE_FUNC CoeffReturnType* data() const { return m_result; } + EIGEN_DEVICE_FUNC CoeffReturnType *data() const { return m_result; } - protected: - EIGEN_DEVICE_FUNC void evalTo(Scalar* data) { - TensorMap > result( - data, m_dimensions); +protected: + EIGEN_DEVICE_FUNC void evalTo(Scalar *data) + { + TensorMap> result(data, m_dimensions); m_op.func().eval(m_op.expression(), result, m_device); } Dimensions m_dimensions; const ArgType m_op; - const Device& m_device; - CoeffReturnType* m_result; + const Device &m_device; + CoeffReturnType *m_result; }; - /** \class TensorCustomBinaryOp - * \ingroup CXX11_Tensor_Module - * - * \brief Tensor custom class. - * - * - */ + * \ingroup CXX11_Tensor_Module + * + * \brief Tensor custom class. + * + * + */ namespace internal { -template -struct traits > -{ - typedef typename internal::promote_storage_type::ret Scalar; - typedef typename internal::promote_storage_type::ret CoeffReturnType; - typedef typename promote_storage_type::StorageKind, - typename traits::StorageKind>::ret StorageKind; - typedef typename promote_index_type::Index, - typename traits::Index>::type Index; - typedef typename LhsXprType::Nested LhsNested; - typedef typename RhsXprType::Nested RhsNested; - typedef typename remove_reference::type _LhsNested; - typedef typename remove_reference::type _RhsNested; - static const int NumDimensions = traits::NumDimensions; - static const int Layout = traits::Layout; -}; - -template -struct eval, Eigen::Dense> -{ - typedef const TensorCustomBinaryOp& type; -}; + template + struct traits> + { + typedef + typename internal::promote_storage_type::ret Scalar; + typedef typename internal::promote_storage_type::ret CoeffReturnType; + typedef typename promote_storage_type::StorageKind, + typename traits::StorageKind>::ret StorageKind; + typedef + typename promote_index_type::Index, typename traits::Index>::type Index; + typedef typename LhsXprType::Nested LhsNested; + typedef typename RhsXprType::Nested RhsNested; + typedef typename remove_reference::type _LhsNested; + typedef typename remove_reference::type _RhsNested; + static const int NumDimensions = traits::NumDimensions; + static const int Layout = traits::Layout; + }; -template -struct nested > -{ - typedef TensorCustomBinaryOp type; -}; + template + struct eval, Eigen::Dense> + { + typedef const TensorCustomBinaryOp &type; + }; -} // end namespace internal + template + struct nested> + { + typedef TensorCustomBinaryOp type; + }; +}// end namespace internal template -class TensorCustomBinaryOp : public TensorBase, ReadOnlyAccessors> +class TensorCustomBinaryOp + : public TensorBase, ReadOnlyAccessors> { - public: +public: typedef typename internal::traits::Scalar Scalar; typedef typename Eigen::NumTraits::Real RealScalar; typedef typename internal::traits::CoeffReturnType CoeffReturnType; @@ -209,25 +205,26 @@ class TensorCustomBinaryOp : public TensorBase::StorageKind StorageKind; typedef typename internal::traits::Index Index; - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TensorCustomBinaryOp(const LhsXprType& lhs, const RhsXprType& rhs, const CustomBinaryFunc& func) + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TensorCustomBinaryOp(const LhsXprType &lhs, + const RhsXprType &rhs, + const CustomBinaryFunc &func) - : m_lhs_xpr(lhs), m_rhs_xpr(rhs), m_func(func) {} + : m_lhs_xpr(lhs), m_rhs_xpr(rhs), m_func(func) + {} EIGEN_DEVICE_FUNC - const CustomBinaryFunc& func() const { return m_func; } + const CustomBinaryFunc &func() const { return m_func; } EIGEN_DEVICE_FUNC - const typename internal::remove_all::type& - lhsExpression() const { return m_lhs_xpr; } + const typename internal::remove_all::type &lhsExpression() const { return m_lhs_xpr; } EIGEN_DEVICE_FUNC - const typename internal::remove_all::type& - rhsExpression() const { return m_rhs_xpr; } + const typename internal::remove_all::type &rhsExpression() const { return m_rhs_xpr; } - protected: - typename LhsXprType::Nested m_lhs_xpr; - typename RhsXprType::Nested m_rhs_xpr; - const CustomBinaryFunc m_func; +protected: + typename LhsXprType::Nested m_lhs_xpr; + typename RhsXprType::Nested m_rhs_xpr; + const CustomBinaryFunc m_func; }; @@ -249,19 +246,20 @@ struct TensorEvaluator::size > 1), BlockAccess = false, Layout = TensorEvaluator::Layout, - CoordAccess = false, // to be implemented + CoordAccess = false,// to be implemented RawAccess = false }; - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TensorEvaluator(const XprType& op, const Device& device) - : m_op(op), m_device(device), m_result(NULL) + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TensorEvaluator(const XprType &op, const Device &device) + : m_op(op), m_device(device), m_result(NULL) { m_dimensions = op.func().dimensions(op.lhsExpression(), op.rhsExpression()); } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const Dimensions& dimensions() const { return m_dimensions; } + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const Dimensions &dimensions() const { return m_dimensions; } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE bool evalSubExprsIfNeeded(CoeffReturnType* data) { + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE bool evalSubExprsIfNeeded(CoeffReturnType *data) + { if (data) { evalTo(data); return false; @@ -272,42 +270,43 @@ struct TensorEvaluator - EIGEN_DEVICE_FUNC PacketReturnType packet(Index index) const { + template EIGEN_DEVICE_FUNC PacketReturnType packet(Index index) const + { return internal::ploadt(m_result + index); } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TensorOpCost costPerCoeff(bool vectorized) const { + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TensorOpCost costPerCoeff(bool vectorized) const + { // TODO(rmlarsen): Extend CustomOp API to return its cost estimate. return TensorOpCost(sizeof(CoeffReturnType), 0, 0, vectorized, PacketSize); } - EIGEN_DEVICE_FUNC CoeffReturnType* data() const { return m_result; } + EIGEN_DEVICE_FUNC CoeffReturnType *data() const { return m_result; } - protected: - EIGEN_DEVICE_FUNC void evalTo(Scalar* data) { - TensorMap > result(data, m_dimensions); +protected: + EIGEN_DEVICE_FUNC void evalTo(Scalar *data) + { + TensorMap> result(data, m_dimensions); m_op.func().eval(m_op.lhsExpression(), m_op.rhsExpression(), result, m_device); } Dimensions m_dimensions; const XprType m_op; - const Device& m_device; - CoeffReturnType* m_result; + const Device &m_device; + CoeffReturnType *m_result; }; -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_CXX11_TENSOR_TENSOR_CUSTOM_OP_H +#endif// EIGEN_CXX11_TENSOR_TENSOR_CUSTOM_OP_H diff --git a/filmulator-gui/core/nlmeans/eigen/unsupported/Eigen/CXX11/src/Tensor/TensorDevice.h b/filmulator-gui/core/nlmeans/eigen/unsupported/Eigen/CXX11/src/Tensor/TensorDevice.h index 29e50a3b..d969b587 100644 --- a/filmulator-gui/core/nlmeans/eigen/unsupported/Eigen/CXX11/src/Tensor/TensorDevice.h +++ b/filmulator-gui/core/nlmeans/eigen/unsupported/Eigen/CXX11/src/Tensor/TensorDevice.h @@ -13,56 +13,58 @@ namespace Eigen { /** \class TensorDevice - * \ingroup CXX11_Tensor_Module - * - * \brief Pseudo expression providing an operator = that will evaluate its argument - * on the specified computing 'device' (GPU, thread pool, ...) - * - * Example: - * C.device(EIGEN_GPU) = A + B; - * - * Todo: operator *= and /=. - */ + * \ingroup CXX11_Tensor_Module + * + * \brief Pseudo expression providing an operator = that will evaluate its argument + * on the specified computing 'device' (GPU, thread pool, ...) + * + * Example: + * C.device(EIGEN_GPU) = A + B; + * + * Todo: operator *= and /=. + */ -template class TensorDevice { - public: - TensorDevice(const DeviceType& device, ExpressionType& expression) : m_device(device), m_expression(expression) {} +template class TensorDevice +{ +public: + TensorDevice(const DeviceType &device, ExpressionType &expression) : m_device(device), m_expression(expression) {} - template - EIGEN_STRONG_INLINE TensorDevice& operator=(const OtherDerived& other) { - typedef TensorAssignOp Assign; - Assign assign(m_expression, other); - internal::TensorExecutor::run(assign, m_device); - return *this; - } + template EIGEN_STRONG_INLINE TensorDevice &operator=(const OtherDerived &other) + { + typedef TensorAssignOp Assign; + Assign assign(m_expression, other); + internal::TensorExecutor::run(assign, m_device); + return *this; + } - template - EIGEN_STRONG_INLINE TensorDevice& operator+=(const OtherDerived& other) { - typedef typename OtherDerived::Scalar Scalar; - typedef TensorCwiseBinaryOp, const ExpressionType, const OtherDerived> Sum; - Sum sum(m_expression, other); - typedef TensorAssignOp Assign; - Assign assign(m_expression, sum); - internal::TensorExecutor::run(assign, m_device); - return *this; - } + template EIGEN_STRONG_INLINE TensorDevice &operator+=(const OtherDerived &other) + { + typedef typename OtherDerived::Scalar Scalar; + typedef TensorCwiseBinaryOp, const ExpressionType, const OtherDerived> Sum; + Sum sum(m_expression, other); + typedef TensorAssignOp Assign; + Assign assign(m_expression, sum); + internal::TensorExecutor::run(assign, m_device); + return *this; + } - template - EIGEN_STRONG_INLINE TensorDevice& operator-=(const OtherDerived& other) { - typedef typename OtherDerived::Scalar Scalar; - typedef TensorCwiseBinaryOp, const ExpressionType, const OtherDerived> Difference; - Difference difference(m_expression, other); - typedef TensorAssignOp Assign; - Assign assign(m_expression, difference); - internal::TensorExecutor::run(assign, m_device); - return *this; - } + template EIGEN_STRONG_INLINE TensorDevice &operator-=(const OtherDerived &other) + { + typedef typename OtherDerived::Scalar Scalar; + typedef TensorCwiseBinaryOp, const ExpressionType, const OtherDerived> + Difference; + Difference difference(m_expression, other); + typedef TensorAssignOp Assign; + Assign assign(m_expression, difference); + internal::TensorExecutor::run(assign, m_device); + return *this; + } - protected: - const DeviceType& m_device; - ExpressionType& m_expression; +protected: + const DeviceType &m_device; + ExpressionType &m_expression; }; -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_CXX11_TENSOR_TENSOR_DEVICE_H +#endif// EIGEN_CXX11_TENSOR_TENSOR_DEVICE_H diff --git a/filmulator-gui/core/nlmeans/eigen/unsupported/Eigen/CXX11/src/Tensor/TensorDeviceCuda.h b/filmulator-gui/core/nlmeans/eigen/unsupported/Eigen/CXX11/src/Tensor/TensorDeviceCuda.h index 4f5767bc..4f489f02 100644 --- a/filmulator-gui/core/nlmeans/eigen/unsupported/Eigen/CXX11/src/Tensor/TensorDeviceCuda.h +++ b/filmulator-gui/core/nlmeans/eigen/unsupported/Eigen/CXX11/src/Tensor/TensorDeviceCuda.h @@ -16,31 +16,33 @@ static const int kCudaScratchSize = 1024; // This defines an interface that GPUDevice can take to use // CUDA streams underneath. -class StreamInterface { - public: +class StreamInterface +{ +public: virtual ~StreamInterface() {} - virtual const cudaStream_t& stream() const = 0; - virtual const cudaDeviceProp& deviceProperties() const = 0; + virtual const cudaStream_t &stream() const = 0; + virtual const cudaDeviceProp &deviceProperties() const = 0; // Allocate memory on the actual device where the computation will run - virtual void* allocate(size_t num_bytes) const = 0; - virtual void deallocate(void* buffer) const = 0; + virtual void *allocate(size_t num_bytes) const = 0; + virtual void deallocate(void *buffer) const = 0; // Return a scratchpad buffer of size 1k - virtual void* scratchpad() const = 0; + virtual void *scratchpad() const = 0; // Return a semaphore. The semaphore is initially initialized to 0, and // each kernel using it is responsible for resetting to 0 upon completion // to maintain the invariant that the semaphore is always equal to 0 upon // each kernel start. - virtual unsigned int* semaphore() const = 0; + virtual unsigned int *semaphore() const = 0; }; -static cudaDeviceProp* m_deviceProperties; +static cudaDeviceProp *m_deviceProperties; static bool m_devicePropInitialized = false; -static void initializeDeviceProp() { +static void initializeDeviceProp() +{ if (!m_devicePropInitialized) { // Attempts to ensure proper behavior in the case of multiple threads // calling this function simultaneously. This would be trivial to @@ -60,20 +62,14 @@ static void initializeDeviceProp() { int num_devices; cudaError_t status = cudaGetDeviceCount(&num_devices); if (status != cudaSuccess) { - std::cerr << "Failed to get the number of CUDA devices: " - << cudaGetErrorString(status) - << std::endl; + std::cerr << "Failed to get the number of CUDA devices: " << cudaGetErrorString(status) << std::endl; assert(status == cudaSuccess); } m_deviceProperties = new cudaDeviceProp[num_devices]; for (int i = 0; i < num_devices; ++i) { status = cudaGetDeviceProperties(&m_deviceProperties[i], i); if (status != cudaSuccess) { - std::cerr << "Failed to initialize CUDA device #" - << i - << ": " - << cudaGetErrorString(status) - << std::endl; + std::cerr << "Failed to initialize CUDA device #" << i << ": " << cudaGetErrorString(status) << std::endl; assert(status == cudaSuccess); } } @@ -96,23 +92,27 @@ static void initializeDeviceProp() { static const cudaStream_t default_stream = cudaStreamDefault; -class CudaStreamDevice : public StreamInterface { - public: +class CudaStreamDevice : public StreamInterface +{ +public: // Use the default stream on the current device - CudaStreamDevice() : stream_(&default_stream), scratch_(NULL), semaphore_(NULL) { + CudaStreamDevice() : stream_(&default_stream), scratch_(NULL), semaphore_(NULL) + { cudaGetDevice(&device_); initializeDeviceProp(); } // Use the default stream on the specified device - CudaStreamDevice(int device) : stream_(&default_stream), device_(device), scratch_(NULL), semaphore_(NULL) { + CudaStreamDevice(int device) : stream_(&default_stream), device_(device), scratch_(NULL), semaphore_(NULL) + { initializeDeviceProp(); } // Use the specified stream. Note that it's the // caller responsibility to ensure that the stream can run on // the specified device. If no device is specified the code // assumes that the stream is associated to the current gpu device. - CudaStreamDevice(const cudaStream_t* stream, int device = -1) - : stream_(stream), device_(device), scratch_(NULL), semaphore_(NULL) { + CudaStreamDevice(const cudaStream_t *stream, int device = -1) + : stream_(stream), device_(device), scratch_(NULL), semaphore_(NULL) + { if (device < 0) { cudaGetDevice(&device_); } else { @@ -126,27 +126,26 @@ class CudaStreamDevice : public StreamInterface { initializeDeviceProp(); } - virtual ~CudaStreamDevice() { - if (scratch_) { - deallocate(scratch_); - } + virtual ~CudaStreamDevice() + { + if (scratch_) { deallocate(scratch_); } } - const cudaStream_t& stream() const { return *stream_; } - const cudaDeviceProp& deviceProperties() const { - return m_deviceProperties[device_]; - } - virtual void* allocate(size_t num_bytes) const { + const cudaStream_t &stream() const { return *stream_; } + const cudaDeviceProp &deviceProperties() const { return m_deviceProperties[device_]; } + virtual void *allocate(size_t num_bytes) const + { cudaError_t err = cudaSetDevice(device_); EIGEN_UNUSED_VARIABLE(err) assert(err == cudaSuccess); - void* result; + void *result; err = cudaMalloc(&result, num_bytes); assert(err == cudaSuccess); assert(result != NULL); return result; } - virtual void deallocate(void* buffer) const { + virtual void deallocate(void *buffer) const + { cudaError_t err = cudaSetDevice(device_); EIGEN_UNUSED_VARIABLE(err) assert(err == cudaSuccess); @@ -155,17 +154,17 @@ class CudaStreamDevice : public StreamInterface { assert(err == cudaSuccess); } - virtual void* scratchpad() const { - if (scratch_ == NULL) { - scratch_ = allocate(kCudaScratchSize + sizeof(unsigned int)); - } + virtual void *scratchpad() const + { + if (scratch_ == NULL) { scratch_ = allocate(kCudaScratchSize + sizeof(unsigned int)); } return scratch_; } - virtual unsigned int* semaphore() const { + virtual unsigned int *semaphore() const + { if (semaphore_ == NULL) { - char* scratch = static_cast(scratchpad()) + kCudaScratchSize; - semaphore_ = reinterpret_cast(scratch); + char *scratch = static_cast(scratchpad()) + kCudaScratchSize; + semaphore_ = reinterpret_cast(scratch); cudaError_t err = cudaMemsetAsync(semaphore_, 0, sizeof(unsigned int), *stream_); EIGEN_UNUSED_VARIABLE(err) assert(err == cudaSuccess); @@ -173,101 +172,94 @@ class CudaStreamDevice : public StreamInterface { return semaphore_; } - private: - const cudaStream_t* stream_; +private: + const cudaStream_t *stream_; int device_; - mutable void* scratch_; - mutable unsigned int* semaphore_; + mutable void *scratch_; + mutable unsigned int *semaphore_; }; -struct GpuDevice { +struct GpuDevice +{ // The StreamInterface is not owned: the caller is // responsible for its initialization and eventual destruction. - explicit GpuDevice(const StreamInterface* stream) : stream_(stream), max_blocks_(INT_MAX) { - eigen_assert(stream); - } - explicit GpuDevice(const StreamInterface* stream, int num_blocks) : stream_(stream), max_blocks_(num_blocks) { + explicit GpuDevice(const StreamInterface *stream) : stream_(stream), max_blocks_(INT_MAX) { eigen_assert(stream); } + explicit GpuDevice(const StreamInterface *stream, int num_blocks) : stream_(stream), max_blocks_(num_blocks) + { eigen_assert(stream); } // TODO(bsteiner): This is an internal API, we should not expose it. - EIGEN_STRONG_INLINE const cudaStream_t& stream() const { - return stream_->stream(); - } + EIGEN_STRONG_INLINE const cudaStream_t &stream() const { return stream_->stream(); } - EIGEN_STRONG_INLINE void* allocate(size_t num_bytes) const { - return stream_->allocate(num_bytes); - } + EIGEN_STRONG_INLINE void *allocate(size_t num_bytes) const { return stream_->allocate(num_bytes); } - EIGEN_STRONG_INLINE void deallocate(void* buffer) const { - stream_->deallocate(buffer); - } + EIGEN_STRONG_INLINE void deallocate(void *buffer) const { stream_->deallocate(buffer); } - EIGEN_STRONG_INLINE void* scratchpad() const { - return stream_->scratchpad(); - } + EIGEN_STRONG_INLINE void *scratchpad() const { return stream_->scratchpad(); } - EIGEN_STRONG_INLINE unsigned int* semaphore() const { - return stream_->semaphore(); - } + EIGEN_STRONG_INLINE unsigned int *semaphore() const { return stream_->semaphore(); } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void memcpy(void* dst, const void* src, size_t n) const { + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void memcpy(void *dst, const void *src, size_t n) const + { #ifndef __CUDA_ARCH__ - cudaError_t err = cudaMemcpyAsync(dst, src, n, cudaMemcpyDeviceToDevice, - stream_->stream()); + cudaError_t err = cudaMemcpyAsync(dst, src, n, cudaMemcpyDeviceToDevice, stream_->stream()); EIGEN_UNUSED_VARIABLE(err) assert(err == cudaSuccess); #else - eigen_assert(false && "The default device should be used instead to generate kernel code"); + eigen_assert(false && "The default device should be used instead to generate kernel code"); #endif } - EIGEN_STRONG_INLINE void memcpyHostToDevice(void* dst, const void* src, size_t n) const { - cudaError_t err = - cudaMemcpyAsync(dst, src, n, cudaMemcpyHostToDevice, stream_->stream()); + EIGEN_STRONG_INLINE void memcpyHostToDevice(void *dst, const void *src, size_t n) const + { + cudaError_t err = cudaMemcpyAsync(dst, src, n, cudaMemcpyHostToDevice, stream_->stream()); EIGEN_UNUSED_VARIABLE(err) assert(err == cudaSuccess); } - EIGEN_STRONG_INLINE void memcpyDeviceToHost(void* dst, const void* src, size_t n) const { - cudaError_t err = - cudaMemcpyAsync(dst, src, n, cudaMemcpyDeviceToHost, stream_->stream()); + EIGEN_STRONG_INLINE void memcpyDeviceToHost(void *dst, const void *src, size_t n) const + { + cudaError_t err = cudaMemcpyAsync(dst, src, n, cudaMemcpyDeviceToHost, stream_->stream()); EIGEN_UNUSED_VARIABLE(err) assert(err == cudaSuccess); } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void memset(void* buffer, int c, size_t n) const { + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void memset(void *buffer, int c, size_t n) const + { #ifndef __CUDA_ARCH__ cudaError_t err = cudaMemsetAsync(buffer, c, n, stream_->stream()); EIGEN_UNUSED_VARIABLE(err) assert(err == cudaSuccess); #else - eigen_assert(false && "The default device should be used instead to generate kernel code"); + eigen_assert(false && "The default device should be used instead to generate kernel code"); #endif } - EIGEN_STRONG_INLINE size_t numThreads() const { + EIGEN_STRONG_INLINE size_t numThreads() const + { // FIXME return 32; } - EIGEN_STRONG_INLINE size_t firstLevelCacheSize() const { + EIGEN_STRONG_INLINE size_t firstLevelCacheSize() const + { // FIXME - return 48*1024; + return 48 * 1024; } - EIGEN_STRONG_INLINE size_t lastLevelCacheSize() const { + EIGEN_STRONG_INLINE size_t lastLevelCacheSize() const + { // We won't try to take advantage of the l2 cache for the time being, and // there is no l3 cache on cuda devices. return firstLevelCacheSize(); } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void synchronize() const { + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void synchronize() const + { #if defined(__CUDACC__) && !defined(__CUDA_ARCH__) cudaError_t err = cudaStreamSynchronize(stream_->stream()); if (err != cudaSuccess) { - std::cerr << "Error detected in CUDA stream: " - << cudaGetErrorString(err) - << std::endl; + std::cerr << "Error detected in CUDA stream: " << cudaGetErrorString(err) << std::endl; assert(err == cudaSuccess); } #else @@ -275,32 +267,22 @@ struct GpuDevice { #endif } - EIGEN_STRONG_INLINE int getNumCudaMultiProcessors() const { - return stream_->deviceProperties().multiProcessorCount; - } - EIGEN_STRONG_INLINE int maxCudaThreadsPerBlock() const { - return stream_->deviceProperties().maxThreadsPerBlock; - } - EIGEN_STRONG_INLINE int maxCudaThreadsPerMultiProcessor() const { + EIGEN_STRONG_INLINE int getNumCudaMultiProcessors() const { return stream_->deviceProperties().multiProcessorCount; } + EIGEN_STRONG_INLINE int maxCudaThreadsPerBlock() const { return stream_->deviceProperties().maxThreadsPerBlock; } + EIGEN_STRONG_INLINE int maxCudaThreadsPerMultiProcessor() const + { return stream_->deviceProperties().maxThreadsPerMultiProcessor; } - EIGEN_STRONG_INLINE int sharedMemPerBlock() const { - return stream_->deviceProperties().sharedMemPerBlock; - } - EIGEN_STRONG_INLINE int majorDeviceVersion() const { - return stream_->deviceProperties().major; - } - EIGEN_STRONG_INLINE int minorDeviceVersion() const { - return stream_->deviceProperties().minor; - } + EIGEN_STRONG_INLINE int sharedMemPerBlock() const { return stream_->deviceProperties().sharedMemPerBlock; } + EIGEN_STRONG_INLINE int majorDeviceVersion() const { return stream_->deviceProperties().major; } + EIGEN_STRONG_INLINE int minorDeviceVersion() const { return stream_->deviceProperties().minor; } - EIGEN_STRONG_INLINE int maxBlocks() const { - return max_blocks_; - } + EIGEN_STRONG_INLINE int maxBlocks() const { return max_blocks_; } // This function checks if the CUDA runtime recorded an error for the // underlying stream device. - inline bool ok() const { + inline bool ok() const + { #ifdef __CUDACC__ cudaError_t error = cudaStreamQuery(stream_->stream()); return (error == cudaSuccess) || (error == cudaErrorNotReady); @@ -309,19 +291,20 @@ struct GpuDevice { #endif } - private: - const StreamInterface* stream_; +private: + const StreamInterface *stream_; int max_blocks_; }; -#define LAUNCH_CUDA_KERNEL(kernel, gridsize, blocksize, sharedmem, device, ...) \ - (kernel) <<< (gridsize), (blocksize), (sharedmem), (device).stream() >>> (__VA_ARGS__); \ +#define LAUNCH_CUDA_KERNEL(kernel, gridsize, blocksize, sharedmem, device, ...) \ + (kernel)<<<(gridsize), (blocksize), (sharedmem), (device).stream()>>>(__VA_ARGS__); \ assert(cudaGetLastError() == cudaSuccess); // FIXME: Should be device and kernel specific. #ifdef __CUDACC__ -static EIGEN_DEVICE_FUNC inline void setCudaSharedMemConfig(cudaSharedMemConfig config) { +static EIGEN_DEVICE_FUNC inline void setCudaSharedMemConfig(cudaSharedMemConfig config) +{ #ifndef __CUDA_ARCH__ cudaError_t status = cudaDeviceSetSharedMemConfig(config); EIGEN_UNUSED_VARIABLE(status) @@ -332,6 +315,6 @@ static EIGEN_DEVICE_FUNC inline void setCudaSharedMemConfig(cudaSharedMemConfig } #endif -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_CXX11_TENSOR_TENSOR_DEVICE_CUDA_H +#endif// EIGEN_CXX11_TENSOR_TENSOR_DEVICE_CUDA_H diff --git a/filmulator-gui/core/nlmeans/eigen/unsupported/Eigen/CXX11/src/Tensor/TensorDeviceDefault.h b/filmulator-gui/core/nlmeans/eigen/unsupported/Eigen/CXX11/src/Tensor/TensorDeviceDefault.h index 9d141395..83742145 100644 --- a/filmulator-gui/core/nlmeans/eigen/unsupported/Eigen/CXX11/src/Tensor/TensorDeviceDefault.h +++ b/filmulator-gui/core/nlmeans/eigen/unsupported/Eigen/CXX11/src/Tensor/TensorDeviceDefault.h @@ -14,27 +14,29 @@ namespace Eigen { // Default device for the machine (typically a single cpu core) -struct DefaultDevice { - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void* allocate(size_t num_bytes) const { +struct DefaultDevice +{ + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void *allocate(size_t num_bytes) const + { return internal::aligned_malloc(num_bytes); } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void deallocate(void* buffer) const { - internal::aligned_free(buffer); - } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void memcpy(void* dst, const void* src, size_t n) const { + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void deallocate(void *buffer) const { internal::aligned_free(buffer); } + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void memcpy(void *dst, const void *src, size_t n) const + { ::memcpy(dst, src, n); } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void memcpyHostToDevice(void* dst, const void* src, size_t n) const { + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void memcpyHostToDevice(void *dst, const void *src, size_t n) const + { memcpy(dst, src, n); } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void memcpyDeviceToHost(void* dst, const void* src, size_t n) const { + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void memcpyDeviceToHost(void *dst, const void *src, size_t n) const + { memcpy(dst, src, n); } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void memset(void* buffer, int c, size_t n) const { - ::memset(buffer, c, n); - } + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void memset(void *buffer, int c, size_t n) const { ::memset(buffer, c, n); } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE size_t numThreads() const { + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE size_t numThreads() const + { #ifndef __CUDA_ARCH__ // Running on the host CPU return 1; @@ -44,17 +46,19 @@ struct DefaultDevice { #endif } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE size_t firstLevelCacheSize() const { + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE size_t firstLevelCacheSize() const + { #ifndef __CUDA_ARCH__ // Running on the host CPU return l1CacheSize(); #else // Running on a CUDA device, return the amount of shared memory available. - return 48*1024; + return 48 * 1024; #endif } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE size_t lastLevelCacheSize() const { + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE size_t lastLevelCacheSize() const + { #ifndef __CUDA_ARCH__ // Running single threaded on the host CPU return l3CacheSize(); @@ -64,7 +68,8 @@ struct DefaultDevice { #endif } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE int majorDeviceVersion() const { + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE int majorDeviceVersion() const + { #ifndef __CUDA_ARCH__ // Running single threaded on the host CPU // Should return an enum that encodes the ISA supported by the CPU @@ -76,6 +81,6 @@ struct DefaultDevice { } }; -} // namespace Eigen +}// namespace Eigen -#endif // EIGEN_CXX11_TENSOR_TENSOR_DEVICE_DEFAULT_H +#endif// EIGEN_CXX11_TENSOR_TENSOR_DEVICE_DEFAULT_H diff --git a/filmulator-gui/core/nlmeans/eigen/unsupported/Eigen/CXX11/src/Tensor/TensorDeviceSycl.h b/filmulator-gui/core/nlmeans/eigen/unsupported/Eigen/CXX11/src/Tensor/TensorDeviceSycl.h index 7c039890..89d7c339 100644 --- a/filmulator-gui/core/nlmeans/eigen/unsupported/Eigen/CXX11/src/Tensor/TensorDeviceSycl.h +++ b/filmulator-gui/core/nlmeans/eigen/unsupported/Eigen/CXX11/src/Tensor/TensorDeviceSycl.h @@ -16,7 +16,8 @@ #define EIGEN_CXX11_TENSOR_TENSOR_DEVICE_SYCL_H namespace Eigen { -struct SyclDevice { +struct SyclDevice +{ /// class members /// sycl queue mutable cl::sycl::queue m_queue; @@ -25,98 +26,112 @@ struct SyclDevice { /// If a non-read-only pointer is needed to be accessed on the host we should manually deallocate it. mutable std::map> buffer_map; /// creating device by using selector - template SyclDevice(dev_Selector s) - : + template + SyclDevice(dev_Selector s) + : #ifdef EIGEN_EXCEPTIONS - m_queue(cl::sycl::queue(s, [=](cl::sycl::exception_list l) { - for (const auto& e : l) { - try { - std::rethrow_exception(e); - } catch (cl::sycl::exception e) { - std::cout << e.what() << std::endl; + m_queue(cl::sycl::queue(s, [=](cl::sycl::exception_list l) { + for (const auto &e : l) { + try { + std::rethrow_exception(e); + } catch (cl::sycl::exception e) { + std::cout << e.what() << std::endl; + } } - } - })) + })) #else - m_queue(cl::sycl::queue(s)) + m_queue(cl::sycl::queue(s)) #endif {} // destructor ~SyclDevice() { deallocate_all(); } - template void deallocate(T *p) const { + template void deallocate(T *p) const + { auto it = buffer_map.find(p); if (it != buffer_map.end()) { buffer_map.erase(it); internal::aligned_free(p); } } - void deallocate_all() const { - std::map>::iterator it=buffer_map.begin(); - while (it!=buffer_map.end()) { - auto p=it->first; + void deallocate_all() const + { + std::map>::iterator it = buffer_map.begin(); + while (it != buffer_map.end()) { + auto p = it->first; buffer_map.erase(it); - internal::aligned_free(const_cast(p)); - it=buffer_map.begin(); + internal::aligned_free(const_cast(p)); + it = buffer_map.begin(); } buffer_map.clear(); } /// creation of sycl accessor for a buffer. This function first tries to find /// the buffer in the buffer_map. If found it gets the accessor from it, if not, - ///the function then adds an entry by creating a sycl buffer for that particular pointer. - template inline cl::sycl::accessor - get_sycl_accessor(size_t num_bytes, cl::sycl::handler &cgh, const T * ptr) const { - return (get_sycl_buffer(num_bytes, ptr)->template get_access(cgh)); + /// the function then adds an entry by creating a sycl buffer for that particular pointer. + template + inline cl::sycl::accessor + get_sycl_accessor(size_t num_bytes, cl::sycl::handler &cgh, const T *ptr) const + { + return ( + get_sycl_buffer(num_bytes, ptr)->template get_access(cgh)); } - template inline std::pair>::iterator,bool> add_sycl_buffer(const T *ptr, size_t num_bytes) const { + template + inline std::pair>::iterator, bool> add_sycl_buffer(const T *ptr, + size_t num_bytes) const + { using Type = cl::sycl::buffer; - std::pair>::iterator,bool> ret = buffer_map.insert(std::pair>(ptr, std::shared_ptr(new Type(cl::sycl::range<1>(num_bytes)), - [](void *dataMem) { delete static_cast(dataMem); }))); - (static_cast(buffer_map.at(ptr).get()))->set_final_data(nullptr); + std::pair>::iterator, bool> ret = + buffer_map.insert(std::pair>(ptr, + std::shared_ptr( + new Type(cl::sycl::range<1>(num_bytes)), [](void *dataMem) { delete static_cast(dataMem); }))); + (static_cast(buffer_map.at(ptr).get()))->set_final_data(nullptr); return ret; } - template inline cl::sycl::buffer* get_sycl_buffer(size_t num_bytes,const T * ptr) const { - return static_cast*>(add_sycl_buffer(ptr, num_bytes).first->second.get()); + template inline cl::sycl::buffer *get_sycl_buffer(size_t num_bytes, const T *ptr) const + { + return static_cast *>(add_sycl_buffer(ptr, num_bytes).first->second.get()); } /// allocating memory on the cpu - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void *allocate(size_t) const { - return internal::aligned_malloc(8); - } + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void *allocate(size_t) const { return internal::aligned_malloc(8); } // some runtime conditions that can be applied here bool isDeviceSuitable() const { return true; } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void memcpy(void *dst, const void *src, size_t n) const { + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void memcpy(void *dst, const void *src, size_t n) const + { ::memcpy(dst, src, n); } - template EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void memcpyHostToDevice(T *dst, const T *src, size_t n) const { - auto host_acc= (static_cast*>(add_sycl_buffer(dst, n).first->second.get()))-> template get_access(); + template + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void memcpyHostToDevice(T *dst, const T *src, size_t n) const + { + auto host_acc = + (static_cast *>(add_sycl_buffer(dst, n).first->second.get())) + ->template get_access(); memcpy(host_acc.get_pointer(), src, n); } - /// whith the current implementation of sycl, the data is copied twice from device to host. This will be fixed soon. - template EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void memcpyDeviceToHost(T *dst, const T *src, size_t n) const { + /// whith the current implementation of sycl, the data is copied twice from device to host. This will be fixed soon. + template + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void memcpyDeviceToHost(T *dst, const T *src, size_t n) const + { auto it = buffer_map.find(src); if (it != buffer_map.end()) { - auto host_acc= (static_cast*>(it->second.get()))-> template get_access(); - memcpy(dst,host_acc.get_pointer(), n); - } else{ + auto host_acc = (static_cast *>(it->second.get())) + ->template get_access(); + memcpy(dst, host_acc.get_pointer(), n); + } else { eigen_assert("no device memory found. The memory might be destroyed before creation"); } } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void memset(void *buffer, int c, size_t n) const { - ::memset(buffer, c, n); - } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE int majorDeviceVersion() const { - return 1; - } + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void memset(void *buffer, int c, size_t n) const { ::memset(buffer, c, n); } + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE int majorDeviceVersion() const { return 1; } }; -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_CXX11_TENSOR_TENSOR_DEVICE_SYCL_H +#endif// EIGEN_CXX11_TENSOR_TENSOR_DEVICE_SYCL_H diff --git a/filmulator-gui/core/nlmeans/eigen/unsupported/Eigen/CXX11/src/Tensor/TensorDeviceThreadPool.h b/filmulator-gui/core/nlmeans/eigen/unsupported/Eigen/CXX11/src/Tensor/TensorDeviceThreadPool.h index 17f04665..5fc78c0e 100644 --- a/filmulator-gui/core/nlmeans/eigen/unsupported/Eigen/CXX11/src/Tensor/TensorDeviceThreadPool.h +++ b/filmulator-gui/core/nlmeans/eigen/unsupported/Eigen/CXX11/src/Tensor/TensorDeviceThreadPool.h @@ -15,30 +15,28 @@ namespace Eigen { // Use the SimpleThreadPool by default. We'll switch to the new non blocking // thread pool later. #ifndef EIGEN_USE_SIMPLE_THREAD_POOL -template using ThreadPoolTempl = NonBlockingThreadPoolTempl; +template using ThreadPoolTempl = NonBlockingThreadPoolTempl; typedef NonBlockingThreadPool ThreadPool; #else -template using ThreadPoolTempl = SimpleThreadPoolTempl; +template using ThreadPoolTempl = SimpleThreadPoolTempl; typedef SimpleThreadPool ThreadPool; #endif // Barrier is an object that allows one or more threads to wait until // Notify has been called a specified number of times. -class Barrier { - public: - Barrier(unsigned int count) : state_(count << 1), notified_(false) { - eigen_assert(((count << 1) >> 1) == count); - } - ~Barrier() { - eigen_assert((state_>>1) == 0); - } +class Barrier +{ +public: + Barrier(unsigned int count) : state_(count << 1), notified_(false) { eigen_assert(((count << 1) >> 1) == count); } + ~Barrier() { eigen_assert((state_ >> 1) == 0); } - void Notify() { + void Notify() + { unsigned int v = state_.fetch_sub(2, std::memory_order_acq_rel) - 2; if (v != 1) { eigen_assert(((v + 2) & ~1) != 0); - return; // either count has not dropped to 0, or waiter is not waiting + return;// either count has not dropped to 0, or waiter is not waiting } std::unique_lock l(mu_); eigen_assert(!notified_); @@ -46,19 +44,18 @@ class Barrier { cv_.notify_all(); } - void Wait() { + void Wait() + { unsigned int v = state_.fetch_or(1, std::memory_order_acq_rel); if ((v >> 1) == 0) return; std::unique_lock l(mu_); - while (!notified_) { - cv_.wait(l); - } + while (!notified_) { cv_.wait(l); } } - private: +private: std::mutex mu_; std::condition_variable cv_; - std::atomic state_; // low bit is waiter flag + std::atomic state_;// low bit is waiter flag bool notified_; }; @@ -68,123 +65,105 @@ class Barrier { // // Multiple threads can wait on the same Notification object, // but only one caller must call Notify() on the object. -struct Notification : Barrier { +struct Notification : Barrier +{ Notification() : Barrier(1) {}; }; // Runs an arbitrary function and then calls Notify() on the passed in // Notification. -template struct FunctionWrapperWithNotification +template struct FunctionWrapperWithNotification { - static void run(Notification* n, Function f, Args... args) { + static void run(Notification *n, Function f, Args... args) + { f(args...); - if (n) { - n->Notify(); - } + if (n) { n->Notify(); } } }; -template struct FunctionWrapperWithBarrier +template struct FunctionWrapperWithBarrier { - static void run(Barrier* b, Function f, Args... args) { + static void run(Barrier *b, Function f, Args... args) + { f(args...); - if (b) { - b->Notify(); - } + if (b) { b->Notify(); } } }; -template -static EIGEN_STRONG_INLINE void wait_until_ready(SyncType* n) { - if (n) { - n->Wait(); - } +template static EIGEN_STRONG_INLINE void wait_until_ready(SyncType *n) +{ + if (n) { n->Wait(); } } // Build a thread pool device on top the an existing pool of threads. -struct ThreadPoolDevice { +struct ThreadPoolDevice +{ // The ownership of the thread pool remains with the caller. - ThreadPoolDevice(ThreadPoolInterface* pool, int num_cores) : pool_(pool), num_threads_(num_cores) { } + ThreadPoolDevice(ThreadPoolInterface *pool, int num_cores) : pool_(pool), num_threads_(num_cores) {} - EIGEN_STRONG_INLINE void* allocate(size_t num_bytes) const { - return internal::aligned_malloc(num_bytes); - } + EIGEN_STRONG_INLINE void *allocate(size_t num_bytes) const { return internal::aligned_malloc(num_bytes); } - EIGEN_STRONG_INLINE void deallocate(void* buffer) const { - internal::aligned_free(buffer); - } + EIGEN_STRONG_INLINE void deallocate(void *buffer) const { internal::aligned_free(buffer); } - EIGEN_STRONG_INLINE void memcpy(void* dst, const void* src, size_t n) const { - ::memcpy(dst, src, n); - } - EIGEN_STRONG_INLINE void memcpyHostToDevice(void* dst, const void* src, size_t n) const { - memcpy(dst, src, n); - } - EIGEN_STRONG_INLINE void memcpyDeviceToHost(void* dst, const void* src, size_t n) const { - memcpy(dst, src, n); - } + EIGEN_STRONG_INLINE void memcpy(void *dst, const void *src, size_t n) const { ::memcpy(dst, src, n); } + EIGEN_STRONG_INLINE void memcpyHostToDevice(void *dst, const void *src, size_t n) const { memcpy(dst, src, n); } + EIGEN_STRONG_INLINE void memcpyDeviceToHost(void *dst, const void *src, size_t n) const { memcpy(dst, src, n); } - EIGEN_STRONG_INLINE void memset(void* buffer, int c, size_t n) const { - ::memset(buffer, c, n); - } + EIGEN_STRONG_INLINE void memset(void *buffer, int c, size_t n) const { ::memset(buffer, c, n); } - EIGEN_STRONG_INLINE int numThreads() const { - return num_threads_; - } + EIGEN_STRONG_INLINE int numThreads() const { return num_threads_; } - EIGEN_STRONG_INLINE size_t firstLevelCacheSize() const { - return l1CacheSize(); - } + EIGEN_STRONG_INLINE size_t firstLevelCacheSize() const { return l1CacheSize(); } - EIGEN_STRONG_INLINE size_t lastLevelCacheSize() const { + EIGEN_STRONG_INLINE size_t lastLevelCacheSize() const + { // The l3 cache size is shared between all the cores. return l3CacheSize() / num_threads_; } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE int majorDeviceVersion() const { + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE int majorDeviceVersion() const + { // Should return an enum that encodes the ISA supported by the CPU return 1; } - template - EIGEN_STRONG_INLINE Notification* enqueue(Function&& f, Args&&... args) const { - Notification* n = new Notification(); + template EIGEN_STRONG_INLINE Notification *enqueue(Function &&f, Args &&...args) const + { + Notification *n = new Notification(); pool_->Schedule(std::bind(&FunctionWrapperWithNotification::run, n, f, args...)); return n; } - template - EIGEN_STRONG_INLINE void enqueue_with_barrier(Barrier* b, - Function&& f, - Args&&... args) const { - pool_->Schedule(std::bind( - &FunctionWrapperWithBarrier::run, b, f, args...)); + template + EIGEN_STRONG_INLINE void enqueue_with_barrier(Barrier *b, Function &&f, Args &&...args) const + { + pool_->Schedule(std::bind(&FunctionWrapperWithBarrier::run, b, f, args...)); } - template - EIGEN_STRONG_INLINE void enqueueNoNotification(Function&& f, Args&&... args) const { + template + EIGEN_STRONG_INLINE void enqueueNoNotification(Function &&f, Args &&...args) const + { pool_->Schedule(std::bind(f, args...)); } // Returns a logical thread index between 0 and pool_->NumThreads() - 1 if // called from one of the threads in pool_. Returns -1 otherwise. - EIGEN_STRONG_INLINE int currentThreadId() const { - return pool_->CurrentThreadId(); - } + EIGEN_STRONG_INLINE int currentThreadId() const { return pool_->CurrentThreadId(); } // parallelFor executes f with [0, n) arguments in parallel and waits for // completion. F accepts a half-open interval [first, last). // Block size is choosen based on the iteration cost and resulting parallel // efficiency. If block_align is not nullptr, it is called to round up the // block size. - void parallelFor(Index n, const TensorOpCost& cost, - std::function block_align, - std::function f) const { + void parallelFor(Index n, + const TensorOpCost &cost, + std::function block_align, + std::function f) const + { typedef TensorCostModel CostModel; - if (n <= 1 || numThreads() == 1 || - CostModel::numThreads(n, cost, static_cast(numThreads())) == 1) { + if (n <= 1 || numThreads() == 1 || CostModel::numThreads(n, cost, static_cast(numThreads())) == 1) { f(0, n); return; } @@ -197,9 +176,8 @@ struct ThreadPoolDevice { double block_size_f = 1.0 / CostModel::taskSize(1, cost); const Index max_oversharding_factor = 4; - Index block_size = numext::mini( - n, numext::maxi(divup(n, max_oversharding_factor * numThreads()), - block_size_f)); + Index block_size = + numext::mini(n, numext::maxi(divup(n, max_oversharding_factor * numThreads()), block_size_f)); const Index max_block_size = numext::mini(n, 2 * block_size); if (block_align) { Index new_block_size = block_align(block_size); @@ -209,13 +187,10 @@ struct ThreadPoolDevice { Index block_count = divup(n, block_size); // Calculate parallel efficiency as fraction of total CPU time used for // computations: - double max_efficiency = - static_cast(block_count) / - (divup(block_count, numThreads()) * numThreads()); + double max_efficiency = static_cast(block_count) / (divup(block_count, numThreads()) * numThreads()); // Now try to increase block size up to max_block_size as long as it // doesn't decrease parallel efficiency. - for (Index prev_block_count = block_count; - max_efficiency < 1.0 && prev_block_count > 1;) { + for (Index prev_block_count = block_count; max_efficiency < 1.0 && prev_block_count > 1;) { // This is the next block size that divides size into a smaller number // of blocks than the current block_size. Index coarser_block_size = divup(n, prev_block_count - 1); @@ -225,22 +200,19 @@ struct ThreadPoolDevice { coarser_block_size = numext::mini(n, new_block_size); } if (coarser_block_size > max_block_size) { - break; // Reached max block size. Stop. + break;// Reached max block size. Stop. } // Recalculate parallel efficiency. const Index coarser_block_count = divup(n, coarser_block_size); eigen_assert(coarser_block_count < prev_block_count); prev_block_count = coarser_block_count; const double coarser_efficiency = - static_cast(coarser_block_count) / - (divup(coarser_block_count, numThreads()) * numThreads()); + static_cast(coarser_block_count) / (divup(coarser_block_count, numThreads()) * numThreads()); if (coarser_efficiency + 0.01 >= max_efficiency) { // Taking it. block_size = coarser_block_size; block_count = coarser_block_count; - if (max_efficiency < coarser_efficiency) { - max_efficiency = coarser_efficiency; - } + if (max_efficiency < coarser_efficiency) { max_efficiency = coarser_efficiency; } } } @@ -266,17 +238,17 @@ struct ThreadPoolDevice { } // Convenience wrapper for parallelFor that does not align blocks. - void parallelFor(Index n, const TensorOpCost& cost, - std::function f) const { + void parallelFor(Index n, const TensorOpCost &cost, std::function f) const + { parallelFor(n, cost, nullptr, std::move(f)); } - private: - ThreadPoolInterface* pool_; +private: + ThreadPoolInterface *pool_; int num_threads_; }; -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_CXX11_TENSOR_TENSOR_DEVICE_THREAD_POOL_H +#endif// EIGEN_CXX11_TENSOR_TENSOR_DEVICE_THREAD_POOL_H diff --git a/filmulator-gui/core/nlmeans/eigen/unsupported/Eigen/CXX11/src/Tensor/TensorDimensionList.h b/filmulator-gui/core/nlmeans/eigen/unsupported/Eigen/CXX11/src/Tensor/TensorDimensionList.h index 1a30e45f..6110c9e5 100644 --- a/filmulator-gui/core/nlmeans/eigen/unsupported/Eigen/CXX11/src/Tensor/TensorDimensionList.h +++ b/filmulator-gui/core/nlmeans/eigen/unsupported/Eigen/CXX11/src/Tensor/TensorDimensionList.h @@ -13,224 +13,176 @@ namespace Eigen { /** \internal - * - * \class TensorDimensionList - * \ingroup CXX11_Tensor_Module - * - * \brief Special case of tensor index list used to list all the dimensions of a tensor of rank n. - * - * \sa Tensor - */ - -template struct DimensionList { - EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE - const Index operator[] (const Index i) const { return i; } -}; - -namespace internal { - -template struct array_size > { - static const size_t value = Rank; -}; -template struct array_size > { - static const size_t value = Rank; -}; - -template const Index array_get(DimensionList&) { - return n; -} -template const Index array_get(const DimensionList&) { - return n; -} - - -#if EIGEN_HAS_CONSTEXPR -template -struct index_known_statically_impl > { - EIGEN_DEVICE_FUNC static constexpr bool run(const DenseIndex) { - return true; - } -}; -template -struct index_known_statically_impl > { - EIGEN_DEVICE_FUNC static constexpr bool run(const DenseIndex) { - return true; - } -}; + * + * \class TensorDimensionList + * \ingroup CXX11_Tensor_Module + * + * \brief Special case of tensor index list used to list all the dimensions of a tensor of rank n. + * + * \sa Tensor + */ -template -struct all_indices_known_statically_impl > { - EIGEN_DEVICE_FUNC static constexpr bool run() { - return true; - } -}; -template -struct all_indices_known_statically_impl > { - EIGEN_DEVICE_FUNC static constexpr bool run() { - return true; - } +template struct DimensionList +{ + EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE const Index operator[](const Index i) const { return i; } }; -template -struct indices_statically_known_to_increase_impl > { - EIGEN_DEVICE_FUNC static constexpr bool run() { - return true; - } -}; -template -struct indices_statically_known_to_increase_impl > { - EIGEN_DEVICE_FUNC static constexpr bool run() { - return true; - } -}; +namespace internal { -template -struct index_statically_eq_impl > { - static constexpr bool run(const DenseIndex i, const DenseIndex value) { - return i == value; - } -}; -template -struct index_statically_eq_impl > { - EIGEN_DEVICE_FUNC static constexpr bool run(const DenseIndex i, const DenseIndex value) { - return i == value; - } -}; + template struct array_size> + { + static const size_t value = Rank; + }; + template struct array_size> + { + static const size_t value = Rank; + }; -template -struct index_statically_ne_impl > { - EIGEN_DEVICE_FUNC static constexpr bool run(const DenseIndex i, const DenseIndex value) { - return i != value; + template const Index array_get(DimensionList &) + { + return n; } -}; -template -struct index_statically_ne_impl > { - static constexpr bool run(const DenseIndex i, const DenseIndex value) { - return i != value; + template const Index array_get(const DimensionList &) + { + return n; } -}; -template -struct index_statically_gt_impl > { - EIGEN_DEVICE_FUNC static constexpr bool run(const DenseIndex i, const DenseIndex value) { - return i > value; - } -}; -template -struct index_statically_gt_impl > { - EIGEN_DEVICE_FUNC static constexpr bool run(const DenseIndex i, const DenseIndex value) { - return i > value; - } -}; -template -struct index_statically_lt_impl > { - EIGEN_DEVICE_FUNC static constexpr bool run(const DenseIndex i, const DenseIndex value) { - return i < value; - } -}; -template -struct index_statically_lt_impl > { - EIGEN_DEVICE_FUNC static constexpr bool run(const DenseIndex i, const DenseIndex value) { - return i < value; - } -}; +#if EIGEN_HAS_CONSTEXPR + template struct index_known_statically_impl> + { + EIGEN_DEVICE_FUNC static constexpr bool run(const DenseIndex) { return true; } + }; + template struct index_known_statically_impl> + { + EIGEN_DEVICE_FUNC static constexpr bool run(const DenseIndex) { return true; } + }; + + template struct all_indices_known_statically_impl> + { + EIGEN_DEVICE_FUNC static constexpr bool run() { return true; } + }; + template struct all_indices_known_statically_impl> + { + EIGEN_DEVICE_FUNC static constexpr bool run() { return true; } + }; + + template + struct indices_statically_known_to_increase_impl> + { + EIGEN_DEVICE_FUNC static constexpr bool run() { return true; } + }; + template + struct indices_statically_known_to_increase_impl> + { + EIGEN_DEVICE_FUNC static constexpr bool run() { return true; } + }; + + template struct index_statically_eq_impl> + { + static constexpr bool run(const DenseIndex i, const DenseIndex value) { return i == value; } + }; + template struct index_statically_eq_impl> + { + EIGEN_DEVICE_FUNC static constexpr bool run(const DenseIndex i, const DenseIndex value) { return i == value; } + }; + + template struct index_statically_ne_impl> + { + EIGEN_DEVICE_FUNC static constexpr bool run(const DenseIndex i, const DenseIndex value) { return i != value; } + }; + template struct index_statically_ne_impl> + { + static constexpr bool run(const DenseIndex i, const DenseIndex value) { return i != value; } + }; + + template struct index_statically_gt_impl> + { + EIGEN_DEVICE_FUNC static constexpr bool run(const DenseIndex i, const DenseIndex value) { return i > value; } + }; + template struct index_statically_gt_impl> + { + EIGEN_DEVICE_FUNC static constexpr bool run(const DenseIndex i, const DenseIndex value) { return i > value; } + }; + + template struct index_statically_lt_impl> + { + EIGEN_DEVICE_FUNC static constexpr bool run(const DenseIndex i, const DenseIndex value) { return i < value; } + }; + template struct index_statically_lt_impl> + { + EIGEN_DEVICE_FUNC static constexpr bool run(const DenseIndex i, const DenseIndex value) { return i < value; } + }; #else -template -struct index_known_statically_impl > { - EIGEN_DEVICE_FUNC static EIGEN_ALWAYS_INLINE bool run(const DenseIndex) { - return true; - } -}; -template -struct index_known_statically_impl > { - EIGEN_DEVICE_FUNC static EIGEN_ALWAYS_INLINE bool run(const DenseIndex) { - return true; - } -}; - -template -struct all_indices_known_statically_impl > { - EIGEN_DEVICE_FUNC static EIGEN_ALWAYS_INLINE bool run() { - return true; - } -}; -template -struct all_indices_known_statically_impl > { - EIGEN_DEVICE_FUNC static EIGEN_ALWAYS_INLINE bool run() { - return true; - } -}; - -template -struct indices_statically_known_to_increase_impl > { - static EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE bool run() { - return true; - } -}; -template -struct indices_statically_known_to_increase_impl > { - static EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE bool run() { - return true; - } -}; - -template -struct index_statically_eq_impl > { - static EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE bool run(const DenseIndex, const DenseIndex) { - return false; - } -}; -template -struct index_statically_eq_impl > { - static EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE bool run(const DenseIndex, const DenseIndex) { - return false; - } -}; - -template -struct index_statically_ne_impl > { - static EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE bool run(const DenseIndex, const DenseIndex){ - return false; - } -}; -template -struct index_statically_ne_impl > { - static EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE bool run(const DenseIndex, const DenseIndex) { - return false; - } -}; - -template -struct index_statically_gt_impl > { - static EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE bool run(const DenseIndex, const DenseIndex) { - return false; - } -}; -template -struct index_statically_gt_impl > { - static EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE bool run(const DenseIndex, const DenseIndex) { - return false; - } -}; - -template -struct index_statically_lt_impl > { - static EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE bool run(const DenseIndex, const DenseIndex) { - return false; - } -}; -template -struct index_statically_lt_impl > { - static EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE bool run(const DenseIndex, const DenseIndex) { - return false; - } -}; + template struct index_known_statically_impl> + { + EIGEN_DEVICE_FUNC static EIGEN_ALWAYS_INLINE bool run(const DenseIndex) { return true; } + }; + template struct index_known_statically_impl> + { + EIGEN_DEVICE_FUNC static EIGEN_ALWAYS_INLINE bool run(const DenseIndex) { return true; } + }; + + template struct all_indices_known_statically_impl> + { + EIGEN_DEVICE_FUNC static EIGEN_ALWAYS_INLINE bool run() { return true; } + }; + template struct all_indices_known_statically_impl> + { + EIGEN_DEVICE_FUNC static EIGEN_ALWAYS_INLINE bool run() { return true; } + }; + + template + struct indices_statically_known_to_increase_impl> + { + static EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE bool run() { return true; } + }; + template + struct indices_statically_known_to_increase_impl> + { + static EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE bool run() { return true; } + }; + + template struct index_statically_eq_impl> + { + static EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE bool run(const DenseIndex, const DenseIndex) { return false; } + }; + template struct index_statically_eq_impl> + { + static EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE bool run(const DenseIndex, const DenseIndex) { return false; } + }; + + template struct index_statically_ne_impl> + { + static EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE bool run(const DenseIndex, const DenseIndex) { return false; } + }; + template struct index_statically_ne_impl> + { + static EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE bool run(const DenseIndex, const DenseIndex) { return false; } + }; + + template struct index_statically_gt_impl> + { + static EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE bool run(const DenseIndex, const DenseIndex) { return false; } + }; + template struct index_statically_gt_impl> + { + static EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE bool run(const DenseIndex, const DenseIndex) { return false; } + }; + + template struct index_statically_lt_impl> + { + static EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE bool run(const DenseIndex, const DenseIndex) { return false; } + }; + template struct index_statically_lt_impl> + { + static EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE bool run(const DenseIndex, const DenseIndex) { return false; } + }; #endif -} // end namespace internal -} // end namespace Eigen +}// end namespace internal +}// end namespace Eigen -#endif // EIGEN_CXX11_TENSOR_TENSOR_DIMENSION_LIST_H +#endif// EIGEN_CXX11_TENSOR_TENSOR_DIMENSION_LIST_H diff --git a/filmulator-gui/core/nlmeans/eigen/unsupported/Eigen/CXX11/src/Tensor/TensorDimensions.h b/filmulator-gui/core/nlmeans/eigen/unsupported/Eigen/CXX11/src/Tensor/TensorDimensions.h index 451940de..16cfa5b7 100644 --- a/filmulator-gui/core/nlmeans/eigen/unsupported/Eigen/CXX11/src/Tensor/TensorDimensions.h +++ b/filmulator-gui/core/nlmeans/eigen/unsupported/Eigen/CXX11/src/Tensor/TensorDimensions.h @@ -14,299 +14,312 @@ namespace Eigen { /** \internal - * - * \class TensorDimensions - * \ingroup CXX11_Tensor_Module - * - * \brief Set of classes used to encode and store the dimensions of a Tensor. - * - * The Sizes class encodes as part of the type the number of dimensions and the - * sizes corresponding to each dimension. It uses no storage space since it is - * entirely known at compile time. - * The DSizes class is its dynamic sibling: the number of dimensions is known - * at compile time but the sizes are set during execution. - * - * \sa Tensor - */ + * + * \class TensorDimensions + * \ingroup CXX11_Tensor_Module + * + * \brief Set of classes used to encode and store the dimensions of a Tensor. + * + * The Sizes class encodes as part of the type the number of dimensions and the + * sizes corresponding to each dimension. It uses no storage space since it is + * entirely known at compile time. + * The DSizes class is its dynamic sibling: the number of dimensions is known + * at compile time but the sizes are set during execution. + * + * \sa Tensor + */ // Boilerplate code namespace internal { -template struct dget { - static const std::size_t value = get::value; -}; + template struct dget + { + static const std::size_t value = get::value; + }; -template -struct fixed_size_tensor_index_linearization_helper -{ - template EIGEN_DEVICE_FUNC - static inline Index run(array const& indices, - const Dimensions& dimensions) + template + struct fixed_size_tensor_index_linearization_helper { - return array_get(indices) + - dget::value * - fixed_size_tensor_index_linearization_helper::run(indices, dimensions); - } -}; + template + EIGEN_DEVICE_FUNC static inline Index run(array const &indices, const Dimensions &dimensions) + { + return array_get(indices) + dget < RowMajor ? n - 1 : (NumIndices - n), + Dimensions > ::value + * fixed_size_tensor_index_linearization_helper::run( + indices, dimensions); + } + }; -template -struct fixed_size_tensor_index_linearization_helper -{ - template EIGEN_DEVICE_FUNC - static inline Index run(array const&, const Dimensions&) + template + struct fixed_size_tensor_index_linearization_helper { - return 0; - } -}; + template + EIGEN_DEVICE_FUNC static inline Index run(array const &, const Dimensions &) + { + return 0; + } + }; -template -struct fixed_size_tensor_index_extraction_helper -{ - template EIGEN_DEVICE_FUNC - static inline Index run(const Index index, - const Dimensions& dimensions) + template struct fixed_size_tensor_index_extraction_helper { - const Index mult = (index == n-1) ? 1 : 0; - return array_get(dimensions) * mult + - fixed_size_tensor_index_extraction_helper::run(index, dimensions); - } -}; + template + EIGEN_DEVICE_FUNC static inline Index run(const Index index, const Dimensions &dimensions) + { + const Index mult = (index == n - 1) ? 1 : 0; + return array_get(dimensions) * mult + + fixed_size_tensor_index_extraction_helper::run(index, dimensions); + } + }; -template -struct fixed_size_tensor_index_extraction_helper -{ - template EIGEN_DEVICE_FUNC - static inline Index run(const Index, - const Dimensions&) + template struct fixed_size_tensor_index_extraction_helper { - return 0; - } + template EIGEN_DEVICE_FUNC static inline Index run(const Index, const Dimensions &) + { + return 0; + } }; -} // end namespace internal +}// end namespace internal // Fixed size #ifndef EIGEN_EMULATE_CXX11_META_H -template -struct Sizes : internal::numeric_list { +template struct Sizes : internal::numeric_list +{ typedef internal::numeric_list Base; static const std::ptrdiff_t total_size = internal::arg_prod(Indices...); - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE std::ptrdiff_t rank() const { - return Base::count; - } + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE std::ptrdiff_t rank() const { return Base::count; } - static EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE std::ptrdiff_t TotalSize() { - return internal::arg_prod(Indices...); - } + static EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE std::ptrdiff_t TotalSize() { return internal::arg_prod(Indices...); } - EIGEN_DEVICE_FUNC Sizes() { } - template - explicit EIGEN_DEVICE_FUNC Sizes(const array& /*indices*/) { + EIGEN_DEVICE_FUNC Sizes() {} + template explicit EIGEN_DEVICE_FUNC Sizes(const array & /*indices*/) + { // todo: add assertion } #if EIGEN_HAS_VARIADIC_TEMPLATES - template EIGEN_DEVICE_FUNC Sizes(DenseIndex...) { } - explicit EIGEN_DEVICE_FUNC Sizes(std::initializer_list /*l*/) { + template EIGEN_DEVICE_FUNC Sizes(DenseIndex...) {} + explicit EIGEN_DEVICE_FUNC Sizes(std::initializer_list /*l*/) + { // todo: add assertion } #endif - template Sizes& operator = (const T& /*other*/) { + template Sizes &operator=(const T & /*other*/) + { // add assertion failure if the size of other is different return *this; } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE std::ptrdiff_t operator[] (const std::size_t index) const { + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE std::ptrdiff_t operator[](const std::size_t index) const + { return internal::fixed_size_tensor_index_extraction_helper::run(index, *this); } - template EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE - size_t IndexOfColMajor(const array& indices) const { - return internal::fixed_size_tensor_index_linearization_helper::run(indices, *static_cast(this)); + template + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE size_t IndexOfColMajor(const array &indices) const + { + return internal::fixed_size_tensor_index_linearization_helper::run( + indices, *static_cast(this)); } - template EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE - size_t IndexOfRowMajor(const array& indices) const { - return internal::fixed_size_tensor_index_linearization_helper::run(indices, *static_cast(this)); + template + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE size_t IndexOfRowMajor(const array &indices) const + { + return internal::fixed_size_tensor_index_linearization_helper::run( + indices, *static_cast(this)); } }; namespace internal { -template -EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE std::ptrdiff_t array_prod(const Sizes&) { - return Sizes::total_size; -} -} + template + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE std::ptrdiff_t array_prod(const Sizes &) + { + return Sizes::total_size; + } +}// namespace internal #else -template -struct non_zero_size { +template struct non_zero_size +{ typedef internal::type2val type; }; -template <> -struct non_zero_size<0> { +template<> struct non_zero_size<0> +{ typedef internal::null_type type; }; -template struct Sizes { - typedef typename internal::make_type_list::type, typename non_zero_size::type, typename non_zero_size::type, typename non_zero_size::type, typename non_zero_size::type >::type Base; +template +struct Sizes +{ + typedef typename internal::make_type_list::type, + typename non_zero_size::type, + typename non_zero_size::type, + typename non_zero_size::type, + typename non_zero_size::type>::type Base; static const size_t count = Base::count; static const std::size_t total_size = internal::arg_prod::value; - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE size_t rank() const { - return count; - } + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE size_t rank() const { return count; } - static EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE size_t TotalSize() { - return internal::arg_prod::value; - } + static EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE size_t TotalSize() { return internal::arg_prod::value; } - Sizes() { } - template - explicit Sizes(const array& /*indices*/) { + Sizes() {} + template explicit Sizes(const array & /*indices*/) + { // todo: add assertion } - template Sizes& operator = (const T& /*other*/) { + template Sizes &operator=(const T & /*other*/) + { // add assertion failure if the size of other is different return *this; } #if EIGEN_HAS_VARIADIC_TEMPLATES - template Sizes(DenseIndex... /*indices*/) { } - explicit Sizes(std::initializer_list) { + template Sizes(DenseIndex... /*indices*/) {} + explicit Sizes(std::initializer_list) + { // todo: add assertion } #else - EIGEN_DEVICE_FUNC explicit Sizes(const DenseIndex) { - } - EIGEN_DEVICE_FUNC Sizes(const DenseIndex, const DenseIndex) { - } - EIGEN_DEVICE_FUNC Sizes(const DenseIndex, const DenseIndex, const DenseIndex) { - } - EIGEN_DEVICE_FUNC Sizes(const DenseIndex, const DenseIndex, const DenseIndex, const DenseIndex) { - } - EIGEN_DEVICE_FUNC Sizes(const DenseIndex, const DenseIndex, const DenseIndex, const DenseIndex, const DenseIndex) { - } + EIGEN_DEVICE_FUNC explicit Sizes(const DenseIndex) {} + EIGEN_DEVICE_FUNC Sizes(const DenseIndex, const DenseIndex) {} + EIGEN_DEVICE_FUNC Sizes(const DenseIndex, const DenseIndex, const DenseIndex) {} + EIGEN_DEVICE_FUNC Sizes(const DenseIndex, const DenseIndex, const DenseIndex, const DenseIndex) {} + EIGEN_DEVICE_FUNC Sizes(const DenseIndex, const DenseIndex, const DenseIndex, const DenseIndex, const DenseIndex) {} #endif - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Index operator[] (const Index index) const { + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Index operator[](const Index index) const + { switch (index) { - case 0: - return internal::get<0, Base>::value; - case 1: - return internal::get<1, Base>::value; - case 2: - return internal::get<2, Base>::value; - case 3: - return internal::get<3, Base>::value; - case 4: - return internal::get<4, Base>::value; - default: - eigen_assert(false && "index overflow"); - return static_cast(-1); + case 0: + return internal::get<0, Base>::value; + case 1: + return internal::get<1, Base>::value; + case 2: + return internal::get<2, Base>::value; + case 3: + return internal::get<3, Base>::value; + case 4: + return internal::get<4, Base>::value; + default: + eigen_assert(false && "index overflow"); + return static_cast(-1); } } - template EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE - size_t IndexOfColMajor(const array& indices) const { - return internal::fixed_size_tensor_index_linearization_helper::run(indices, *reinterpret_cast(this)); + template + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE size_t IndexOfColMajor(const array &indices) const + { + return internal::fixed_size_tensor_index_linearization_helper::run( + indices, *reinterpret_cast(this)); } - template EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE - size_t IndexOfRowMajor(const array& indices) const { - return internal::fixed_size_tensor_index_linearization_helper::run(indices, *reinterpret_cast(this)); + template + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE size_t IndexOfRowMajor(const array &indices) const + { + return internal::fixed_size_tensor_index_linearization_helper::run( + indices, *reinterpret_cast(this)); } }; namespace internal { -template -EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE std::size_t array_prod(const Sizes&) { - return Sizes::total_size; -} -} + template + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE std::size_t array_prod(const Sizes &) + { + return Sizes::total_size; + } +}// namespace internal #endif // Boilerplate namespace internal { -template -struct tensor_index_linearization_helper -{ - static EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE - Index run(array const& indices, array const& dimensions) + template + struct tensor_index_linearization_helper { - return array_get(indices) + - array_get(dimensions) * - tensor_index_linearization_helper::run(indices, dimensions); - } -}; + static EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Index run(array const &indices, + array const &dimensions) + { + return array_get(indices) + + array_get(dimensions) + * tensor_index_linearization_helper::run(indices, dimensions); + } + }; -template -struct tensor_index_linearization_helper -{ - static EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE - Index run(array const& indices, array const&) + template + struct tensor_index_linearization_helper { - return array_get(indices); - } -}; -} // end namespace internal - + static EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Index run(array const &indices, + array const &) + { + return array_get(indices); + } + }; +}// end namespace internal // Dynamic size -template -struct DSizes : array { +template struct DSizes : array +{ typedef array Base; static const int count = NumDims; - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE size_t rank() const { - return NumDims; - } + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE size_t rank() const { return NumDims; } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE DenseIndex TotalSize() const { - return (NumDims == 0) ? 1 : internal::array_prod(*static_cast(this)); + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE DenseIndex TotalSize() const + { + return (NumDims == 0) ? 1 : internal::array_prod(*static_cast(this)); } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE DSizes() { - for (int i = 0 ; i < NumDims; ++i) { - (*this)[i] = 0; - } + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE DSizes() + { + for (int i = 0; i < NumDims; ++i) { (*this)[i] = 0; } } - EIGEN_DEVICE_FUNC explicit DSizes(const array& a) : Base(a) { } + EIGEN_DEVICE_FUNC explicit DSizes(const array &a) : Base(a) {} - EIGEN_DEVICE_FUNC explicit DSizes(const DenseIndex i0) { + EIGEN_DEVICE_FUNC explicit DSizes(const DenseIndex i0) + { eigen_assert(NumDims == 1); (*this)[0] = i0; } #if EIGEN_HAS_VARIADIC_TEMPLATES - template EIGEN_DEVICE_FUNC - EIGEN_STRONG_INLINE explicit DSizes(DenseIndex firstDimension, DenseIndex secondDimension, IndexTypes... otherDimensions) : Base({{firstDimension, secondDimension, otherDimensions...}}) { - EIGEN_STATIC_ASSERT(sizeof...(otherDimensions) + 2 == NumDims, YOU_MADE_A_PROGRAMMING_MISTAKE) - } + template + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE explicit DSizes(DenseIndex firstDimension, + DenseIndex secondDimension, + IndexTypes... otherDimensions) + : Base({ { + firstDimension, + secondDimension, + otherDimensions... + } }){ EIGEN_STATIC_ASSERT(sizeof...(otherDimensions) + 2 == NumDims, YOU_MADE_A_PROGRAMMING_MISTAKE) } #else - EIGEN_DEVICE_FUNC DSizes(const DenseIndex i0, const DenseIndex i1) { + EIGEN_DEVICE_FUNC DSizes(const DenseIndex i0, const DenseIndex i1) + { eigen_assert(NumDims == 2); (*this)[0] = i0; (*this)[1] = i1; } - EIGEN_DEVICE_FUNC DSizes(const DenseIndex i0, const DenseIndex i1, const DenseIndex i2) { + EIGEN_DEVICE_FUNC DSizes(const DenseIndex i0, const DenseIndex i1, const DenseIndex i2) + { eigen_assert(NumDims == 3); (*this)[0] = i0; (*this)[1] = i1; (*this)[2] = i2; } - EIGEN_DEVICE_FUNC DSizes(const DenseIndex i0, const DenseIndex i1, const DenseIndex i2, const DenseIndex i3) { + EIGEN_DEVICE_FUNC DSizes(const DenseIndex i0, const DenseIndex i1, const DenseIndex i2, const DenseIndex i3) + { eigen_assert(NumDims == 4); (*this)[0] = i0; (*this)[1] = i1; (*this)[2] = i2; (*this)[3] = i3; } - EIGEN_DEVICE_FUNC DSizes(const DenseIndex i0, const DenseIndex i1, const DenseIndex i2, const DenseIndex i3, const DenseIndex i4) { + EIGEN_DEVICE_FUNC + DSizes(const DenseIndex i0, const DenseIndex i1, const DenseIndex i2, const DenseIndex i3, const DenseIndex i4) + { eigen_assert(NumDims == 5); (*this)[0] = i0; (*this)[1] = i1; @@ -316,113 +329,130 @@ struct DSizes : array { } #endif - EIGEN_DEVICE_FUNC DSizes& operator = (const array& other) { - *static_cast(this) = other; + EIGEN_DEVICE_FUNC DSizes + & operator=(const array &other) + { + *static_cast(this) = other; return *this; } // A constexpr would be so much better here - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE DenseIndex IndexOfColMajor(const array& indices) const { - return internal::tensor_index_linearization_helper::run(indices, *static_cast(this)); + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE DenseIndex IndexOfColMajor(const array &indices) const + { + return internal::tensor_index_linearization_helper::run( + indices, *static_cast(this)); } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE DenseIndex IndexOfRowMajor(const array& indices) const { - return internal::tensor_index_linearization_helper::run(indices, *static_cast(this)); + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE DenseIndex IndexOfRowMajor(const array &indices) const + { + return internal::tensor_index_linearization_helper::run( + indices, *static_cast(this)); } }; - - // Boilerplate namespace internal { -template -struct tensor_vsize_index_linearization_helper -{ - static EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE - Index run(array const& indices, std::vector const& dimensions) + template + struct tensor_vsize_index_linearization_helper { - return array_get(indices) + - array_get(dimensions) * - tensor_vsize_index_linearization_helper::run(indices, dimensions); - } -}; + static EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Index run(array const &indices, + std::vector const &dimensions) + { + return array_get(indices) + + array_get(dimensions) + * tensor_vsize_index_linearization_helper::run( + indices, dimensions); + } + }; -template -struct tensor_vsize_index_linearization_helper -{ - static EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE - Index run(array const& indices, std::vector const&) + template + struct tensor_vsize_index_linearization_helper { - return array_get(indices); - } -}; -} // end namespace internal + static EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Index run(array const &indices, + std::vector const &) + { + return array_get(indices); + } + }; +}// end namespace internal namespace internal { -template struct array_size > { - static const size_t value = NumDims; -}; -template struct array_size > { - static const size_t value = NumDims; -}; + template struct array_size> + { + static const size_t value = NumDims; + }; + template struct array_size> + { + static const size_t value = NumDims; + }; #ifndef EIGEN_EMULATE_CXX11_META_H -template struct array_size > { -static const std::ptrdiff_t value = Sizes::count; -}; -template struct array_size > { -static const std::ptrdiff_t value = Sizes::count; -}; -template EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE std::ptrdiff_t array_get(const Sizes&) { - return get >::value; -} -template EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE std::ptrdiff_t array_get(const Sizes<>&) { - eigen_assert(false && "should never be called"); - return -1; -} + template struct array_size> + { + static const std::ptrdiff_t value = Sizes::count; + }; + template struct array_size> + { + static const std::ptrdiff_t value = Sizes::count; + }; + template + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE std::ptrdiff_t array_get(const Sizes &) + { + return get>::value; + } + template EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE std::ptrdiff_t array_get(const Sizes<> &) + { + eigen_assert(false && "should never be called"); + return -1; + } #else -template struct array_size > { - static const size_t value = Sizes::count; -}; -template struct array_size > { - static const size_t value = Sizes::count; -}; -template EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE std::size_t array_get(const Sizes&) { - return get::Base>::value; -} + template + struct array_size> + { + static const size_t value = Sizes::count; + }; + template + struct array_size> + { + static const size_t value = Sizes::count; + }; + template + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE std::size_t array_get(const Sizes &) + { + return get::Base>::value; + } #endif -template -struct sizes_match_below_dim { - static EIGEN_DEVICE_FUNC inline bool run(Dims1&, Dims2&) { - return false; - } -}; -template -struct sizes_match_below_dim { - static EIGEN_DEVICE_FUNC inline bool run(Dims1& dims1, Dims2& dims2) { - return (array_get(dims1) == array_get(dims2)) & - sizes_match_below_dim::run(dims1, dims2); - } -}; -template -struct sizes_match_below_dim { - static EIGEN_DEVICE_FUNC inline bool run(Dims1&, Dims2&) { - return true; - } -}; + template struct sizes_match_below_dim + { + static EIGEN_DEVICE_FUNC inline bool run(Dims1 &, Dims2 &) { return false; } + }; + template struct sizes_match_below_dim + { + static EIGEN_DEVICE_FUNC inline bool run(Dims1 &dims1, Dims2 &dims2) + { + return (array_get(dims1) == array_get(dims2)) + & sizes_match_below_dim::run(dims1, dims2); + } + }; + template struct sizes_match_below_dim + { + static EIGEN_DEVICE_FUNC inline bool run(Dims1 &, Dims2 &) { return true; } + }; -} // end namespace internal +}// end namespace internal -template -EIGEN_DEVICE_FUNC bool dimensions_match(Dims1& dims1, Dims2& dims2) { - return internal::sizes_match_below_dim::value, internal::array_size::value>::run(dims1, dims2); +template EIGEN_DEVICE_FUNC bool dimensions_match(Dims1 &dims1, Dims2 &dims2) +{ + return internal:: + sizes_match_below_dim::value, internal::array_size::value>::run( + dims1, dims2); } -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_CXX11_TENSOR_TENSOR_DIMENSIONS_H +#endif// EIGEN_CXX11_TENSOR_TENSOR_DIMENSIONS_H diff --git a/filmulator-gui/core/nlmeans/eigen/unsupported/Eigen/CXX11/src/Tensor/TensorEvalTo.h b/filmulator-gui/core/nlmeans/eigen/unsupported/Eigen/CXX11/src/Tensor/TensorEvalTo.h index 06987132..f60e5af6 100644 --- a/filmulator-gui/core/nlmeans/eigen/unsupported/Eigen/CXX11/src/Tensor/TensorEvalTo.h +++ b/filmulator-gui/core/nlmeans/eigen/unsupported/Eigen/CXX11/src/Tensor/TensorEvalTo.h @@ -13,58 +13,53 @@ namespace Eigen { /** \class TensorForcedEval - * \ingroup CXX11_Tensor_Module - * - * \brief Tensor reshaping class. - * - * - */ + * \ingroup CXX11_Tensor_Module + * + * \brief Tensor reshaping class. + * + * + */ namespace internal { -template class MakePointer_> -struct traits > -{ - // Type promotion to handle the case where the types of the lhs and the rhs are different. - typedef typename XprType::Scalar Scalar; - typedef traits XprTraits; - typedef typename XprTraits::StorageKind StorageKind; - typedef typename XprTraits::Index Index; - typedef typename XprType::Nested Nested; - typedef typename remove_reference::type _Nested; - static const int NumDimensions = XprTraits::NumDimensions; - static const int Layout = XprTraits::Layout; - - enum { - Flags = 0 - }; - template - struct MakePointer { - // Intermediate typedef to workaround MSVC issue. - typedef MakePointer_ MakePointerT; - typedef typename MakePointerT::Type Type; + template class MakePointer_> struct traits> + { + // Type promotion to handle the case where the types of the lhs and the rhs are different. + typedef typename XprType::Scalar Scalar; + typedef traits XprTraits; + typedef typename XprTraits::StorageKind StorageKind; + typedef typename XprTraits::Index Index; + typedef typename XprType::Nested Nested; + typedef typename remove_reference::type _Nested; + static const int NumDimensions = XprTraits::NumDimensions; + static const int Layout = XprTraits::Layout; + + enum { Flags = 0 }; + template struct MakePointer + { + // Intermediate typedef to workaround MSVC issue. + typedef MakePointer_ MakePointerT; + typedef typename MakePointerT::Type Type; + }; }; -}; - -template class MakePointer_> -struct eval, Eigen::Dense> -{ - typedef const TensorEvalToOp& type; -}; -template class MakePointer_> -struct nested, 1, typename eval >::type> -{ - typedef TensorEvalToOp type; -}; - -} // end namespace internal + template class MakePointer_> + struct eval, Eigen::Dense> + { + typedef const TensorEvalToOp &type; + }; + template class MakePointer_> + struct nested, 1, typename eval>::type> + { + typedef TensorEvalToOp type; + }; +}// end namespace internal -template class MakePointer_> +template class MakePointer_> class TensorEvalToOp : public TensorBase, ReadOnlyAccessors> { - public: +public: typedef typename Eigen::internal::traits::Scalar Scalar; typedef typename Eigen::NumTraits::Real RealScalar; typedef typename internal::remove_const::type CoeffReturnType; @@ -73,23 +68,22 @@ class TensorEvalToOp : public TensorBase, typedef typename Eigen::internal::traits::StorageKind StorageKind; typedef typename Eigen::internal::traits::Index Index; - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TensorEvalToOp(PointerType buffer, const XprType& expr) - : m_xpr(expr), m_buffer(buffer) {} + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TensorEvalToOp(PointerType buffer, const XprType &expr) + : m_xpr(expr), m_buffer(buffer) + {} - EIGEN_DEVICE_FUNC - const typename internal::remove_all::type& - expression() const { return m_xpr; } + EIGEN_DEVICE_FUNC + const typename internal::remove_all::type &expression() const { return m_xpr; } - EIGEN_DEVICE_FUNC PointerType buffer() const { return m_buffer; } + EIGEN_DEVICE_FUNC PointerType buffer() const { return m_buffer; } - protected: - typename XprType::Nested m_xpr; - PointerType m_buffer; +protected: + typename XprType::Nested m_xpr; + PointerType m_buffer; }; - -template class MakePointer_> +template class MakePointer_> struct TensorEvaluator, Device> { typedef TensorEvalToOp XprType; @@ -104,78 +98,71 @@ struct TensorEvaluator, Device> IsAligned = TensorEvaluator::IsAligned, PacketAccess = TensorEvaluator::PacketAccess, Layout = TensorEvaluator::Layout, - CoordAccess = false, // to be implemented + CoordAccess = false,// to be implemented RawAccess = true }; - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TensorEvaluator(const XprType& op, const Device& device) - : m_impl(op.expression(), device), m_device(device), - m_buffer(op.buffer()), m_op(op), m_expression(op.expression()) - { } + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TensorEvaluator(const XprType &op, const Device &device) + : m_impl(op.expression(), device), m_device(device), m_buffer(op.buffer()), m_op(op), m_expression(op.expression()) + {} // Used for accessor extraction in SYCL Managed TensorMap: - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const XprType& op() const { - return m_op; - } - - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE ~TensorEvaluator() { - } + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const XprType &op() const { return m_op; } + + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE ~TensorEvaluator() {} - typedef typename internal::traits >::template MakePointer::Type DevicePointer; - EIGEN_DEVICE_FUNC const Dimensions& dimensions() const { return m_impl.dimensions(); } + typedef + typename internal::traits>::template MakePointer::Type + DevicePointer; + EIGEN_DEVICE_FUNC const Dimensions &dimensions() const { return m_impl.dimensions(); } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE bool evalSubExprsIfNeeded(DevicePointer scalar) { + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE bool evalSubExprsIfNeeded(DevicePointer scalar) + { EIGEN_UNUSED_VARIABLE(scalar); eigen_assert(scalar == NULL); return m_impl.evalSubExprsIfNeeded(m_buffer); } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void evalScalar(Index i) { - m_buffer[i] = m_impl.coeff(i); - } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void evalPacket(Index i) { - internal::pstoret(m_buffer + i, m_impl.template packet::IsAligned ? Aligned : Unaligned>(i)); + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void evalScalar(Index i) { m_buffer[i] = m_impl.coeff(i); } + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void evalPacket(Index i) + { + internal::pstoret( + m_buffer + i, m_impl.template packet::IsAligned ? Aligned : Unaligned>(i)); } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void cleanup() { - m_impl.cleanup(); - } + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void cleanup() { m_impl.cleanup(); } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE CoeffReturnType coeff(Index index) const - { - return m_buffer[index]; - } + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE CoeffReturnType coeff(Index index) const { return m_buffer[index]; } - template - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE PacketReturnType packet(Index index) const + template EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE PacketReturnType packet(Index index) const { return internal::ploadt(m_buffer + index); } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TensorOpCost costPerCoeff(bool vectorized) const { + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TensorOpCost costPerCoeff(bool vectorized) const + { // We assume that evalPacket or evalScalar is called to perform the // assignment and account for the cost of the write here. - return m_impl.costPerCoeff(vectorized) + - TensorOpCost(0, sizeof(CoeffReturnType), 0, vectorized, PacketSize); + return m_impl.costPerCoeff(vectorized) + TensorOpCost(0, sizeof(CoeffReturnType), 0, vectorized, PacketSize); } EIGEN_DEVICE_FUNC DevicePointer data() const { return m_buffer; } ArgType expression() const { return m_expression; } /// required by sycl in order to extract the accessor - const TensorEvaluator& impl() const { return m_impl; } + const TensorEvaluator &impl() const { return m_impl; } /// added for sycl in order to construct the buffer from the sycl device - const Device& device() const{return m_device;} + const Device &device() const { return m_device; } - private: +private: TensorEvaluator m_impl; - const Device& m_device; + const Device &m_device; DevicePointer m_buffer; - const XprType& m_op; + const XprType &m_op; const ArgType m_expression; }; -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_CXX11_TENSOR_TENSOR_EVAL_TO_H +#endif// EIGEN_CXX11_TENSOR_TENSOR_EVAL_TO_H diff --git a/filmulator-gui/core/nlmeans/eigen/unsupported/Eigen/CXX11/src/Tensor/TensorEvaluator.h b/filmulator-gui/core/nlmeans/eigen/unsupported/Eigen/CXX11/src/Tensor/TensorEvaluator.h index 834ce07d..ec37875c 100644 --- a/filmulator-gui/core/nlmeans/eigen/unsupported/Eigen/CXX11/src/Tensor/TensorEvaluator.h +++ b/filmulator-gui/core/nlmeans/eigen/unsupported/Eigen/CXX11/src/Tensor/TensorEvaluator.h @@ -13,19 +13,18 @@ namespace Eigen { /** \class TensorEvaluator - * \ingroup CXX11_Tensor_Module - * - * \brief The tensor evaluator classes. - * - * These classes are responsible for the evaluation of the tensor expression. - * - * TODO: add support for more types of expressions, in particular expressions - * leading to lvalues (slicing, reshaping, etc...) - */ + * \ingroup CXX11_Tensor_Module + * + * \brief The tensor evaluator classes. + * + * These classes are responsible for the evaluation of the tensor expression. + * + * TODO: add support for more types of expressions, in particular expressions + * leading to lvalues (slicing, reshaping, etc...) + */ // Generic evaluator -template -struct TensorEvaluator +template struct TensorEvaluator { typedef typename Derived::Index Index; typedef typename Derived::Scalar Scalar; @@ -34,8 +33,8 @@ struct TensorEvaluator typedef typename Derived::Dimensions Dimensions; // NumDimensions is -1 for variable dim tensors - static const int NumCoords = internal::traits::NumDimensions > 0 ? - internal::traits::NumDimensions : 0; + static const int NumCoords = + internal::traits::NumDimensions > 0 ? internal::traits::NumDimensions : 0; enum { IsAligned = Derived::IsAligned, @@ -45,47 +44,50 @@ struct TensorEvaluator RawAccess = true }; - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TensorEvaluator(const Derived& m, const Device& device) - : m_data(const_cast::template MakePointer::Type>(m.data())), m_dims(m.dimensions()), m_device(device), m_impl(m) - { } + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TensorEvaluator(const Derived &m, const Device &device) + : m_data(const_cast::template MakePointer::Type>(m.data())), + m_dims(m.dimensions()), m_device(device), m_impl(m) + {} // Used for accessor extraction in SYCL Managed TensorMap: - const Derived& derived() const { return m_impl; } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const Dimensions& dimensions() const { return m_dims; } + const Derived &derived() const { return m_impl; } + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const Dimensions &dimensions() const { return m_dims; } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE bool evalSubExprsIfNeeded(CoeffReturnType* dest) { + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE bool evalSubExprsIfNeeded(CoeffReturnType *dest) + { if (dest) { - m_device.memcpy((void*)dest, m_data, sizeof(Scalar) * m_dims.TotalSize()); + m_device.memcpy((void *)dest, m_data, sizeof(Scalar) * m_dims.TotalSize()); return false; } return true; } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void cleanup() { } + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void cleanup() {} - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE CoeffReturnType coeff(Index index) const { + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE CoeffReturnType coeff(Index index) const + { eigen_assert(m_data); return m_data[index]; } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Scalar& coeffRef(Index index) { + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Scalar &coeffRef(Index index) + { eigen_assert(m_data); return m_data[index]; } - template EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE - PacketReturnType packet(Index index) const + template EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE PacketReturnType packet(Index index) const { return internal::ploadt(m_data + index); } - template EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE - void writePacket(Index index, const PacketReturnType& x) + template EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void writePacket(Index index, const PacketReturnType &x) { return internal::pstoret(m_data + index, x); } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE CoeffReturnType coeff(const array& coords) const { + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE CoeffReturnType coeff(const array &coords) const + { eigen_assert(m_data); if (static_cast(Layout) == static_cast(ColMajor)) { return m_data[m_dims.IndexOfColMajor(coords)]; @@ -94,7 +96,8 @@ struct TensorEvaluator } } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Scalar& coeffRef(const array& coords) { + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Scalar &coeffRef(const array &coords) + { eigen_assert(m_data); if (static_cast(Layout) == static_cast(ColMajor)) { return m_data[m_dims.IndexOfColMajor(coords)]; @@ -103,49 +106,42 @@ struct TensorEvaluator } } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TensorOpCost costPerCoeff(bool vectorized) const { - return TensorOpCost(sizeof(CoeffReturnType), 0, 0, vectorized, - internal::unpacket_traits::size); + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TensorOpCost costPerCoeff(bool vectorized) const + { + return TensorOpCost(sizeof(CoeffReturnType), 0, 0, vectorized, internal::unpacket_traits::size); } - EIGEN_DEVICE_FUNC typename internal::traits::template MakePointer::Type data() const { return m_data; } + EIGEN_DEVICE_FUNC typename internal::traits::template MakePointer::Type data() const + { + return m_data; + } /// required by sycl in order to construct sycl buffer from raw pointer - const Device& device() const{return m_device;} + const Device &device() const { return m_device; } - protected: +protected: typename internal::traits::template MakePointer::Type m_data; Dimensions m_dims; - const Device& m_device; - const Derived& m_impl; + const Device &m_device; + const Derived &m_impl; }; namespace { -template EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE -T loadConstant(const T* address) { - return *address; -} + template EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE T loadConstant(const T *address) { return *address; } // Use the texture cache on CUDA devices whenever possible #if defined(__CUDA_ARCH__) && __CUDA_ARCH__ >= 350 -template <> EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE -float loadConstant(const float* address) { - return __ldg(address); -} -template <> EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE -double loadConstant(const double* address) { - return __ldg(address); -} -template <> EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE -Eigen::half loadConstant(const Eigen::half* address) { - return Eigen::half(half_impl::raw_uint16_to_half(__ldg(&address->x))); -} + template<> EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE float loadConstant(const float *address) { return __ldg(address); } + template<> EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE double loadConstant(const double *address) { return __ldg(address); } + template<> EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE Eigen::half loadConstant(const Eigen::half *address) + { + return Eigen::half(half_impl::raw_uint16_to_half(__ldg(&address->x))); + } #endif -} +}// namespace // Default evaluator for rvalues -template -struct TensorEvaluator +template struct TensorEvaluator { typedef typename Derived::Index Index; typedef typename Derived::Scalar Scalar; @@ -154,8 +150,8 @@ struct TensorEvaluator typedef typename Derived::Dimensions Dimensions; // NumDimensions is -1 for variable dim tensors - static const int NumCoords = internal::traits::NumDimensions > 0 ? - internal::traits::NumDimensions : 0; + static const int NumCoords = + internal::traits::NumDimensions > 0 ? internal::traits::NumDimensions : 0; enum { IsAligned = Derived::IsAligned, @@ -166,62 +162,65 @@ struct TensorEvaluator }; // Used for accessor extraction in SYCL Managed TensorMap: - const Derived& derived() const { return m_impl; } + const Derived &derived() const { return m_impl; } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TensorEvaluator(const Derived& m, const Device& device) - : m_data(m.data()), m_dims(m.dimensions()), m_device(device), m_impl(m) - { } + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TensorEvaluator(const Derived &m, const Device &device) + : m_data(m.data()), m_dims(m.dimensions()), m_device(device), m_impl(m) + {} - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const Dimensions& dimensions() const { return m_dims; } + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const Dimensions &dimensions() const { return m_dims; } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE bool evalSubExprsIfNeeded(CoeffReturnType* data) { + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE bool evalSubExprsIfNeeded(CoeffReturnType *data) + { if (!NumTraits::type>::RequireInitialization && data) { - m_device.memcpy((void*)data, m_data, m_dims.TotalSize() * sizeof(Scalar)); + m_device.memcpy((void *)data, m_data, m_dims.TotalSize() * sizeof(Scalar)); return false; } return true; } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void cleanup() { } + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void cleanup() {} - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE CoeffReturnType coeff(Index index) const { + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE CoeffReturnType coeff(Index index) const + { eigen_assert(m_data); - return loadConstant(m_data+index); + return loadConstant(m_data + index); } - template EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE - PacketReturnType packet(Index index) const + template EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE PacketReturnType packet(Index index) const { return internal::ploadt_ro(m_data + index); } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE CoeffReturnType coeff(const array& coords) const { + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE CoeffReturnType coeff(const array &coords) const + { eigen_assert(m_data); const Index index = (static_cast(Layout) == static_cast(ColMajor)) ? m_dims.IndexOfColMajor(coords) - : m_dims.IndexOfRowMajor(coords); - return loadConstant(m_data+index); + : m_dims.IndexOfRowMajor(coords); + return loadConstant(m_data + index); } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TensorOpCost costPerCoeff(bool vectorized) const { - return TensorOpCost(sizeof(CoeffReturnType), 0, 0, vectorized, - internal::unpacket_traits::size); + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TensorOpCost costPerCoeff(bool vectorized) const + { + return TensorOpCost(sizeof(CoeffReturnType), 0, 0, vectorized, internal::unpacket_traits::size); } - EIGEN_DEVICE_FUNC typename internal::traits::template MakePointer::Type data() const { return m_data; } + EIGEN_DEVICE_FUNC typename internal::traits::template MakePointer::Type data() const + { + return m_data; + } /// added for sycl in order to construct the buffer from the sycl device - const Device& device() const{return m_device;} + const Device &device() const { return m_device; } - protected: +protected: typename internal::traits::template MakePointer::Type m_data; Dimensions m_dims; - const Device& m_device; - const Derived& m_impl; + const Device &m_device; + const Derived &m_impl; }; - - // -------------------- CwiseNullaryOp -------------------- template @@ -233,14 +232,14 @@ struct TensorEvaluator, Device> IsAligned = true, PacketAccess = internal::functor_traits::PacketAccess, Layout = TensorEvaluator::Layout, - CoordAccess = false, // to be implemented + CoordAccess = false,// to be implemented RawAccess = false }; EIGEN_DEVICE_FUNC - TensorEvaluator(const XprType& op, const Device& device) - : m_functor(op.functor()), m_argImpl(op.nestedExpression(), device), m_wrapper() - { } + TensorEvaluator(const XprType &op, const Device &device) + : m_functor(op.functor()), m_argImpl(op.nestedExpression(), device), m_wrapper() + {} typedef typename XprType::Index Index; typedef typename XprType::Scalar Scalar; @@ -249,44 +248,38 @@ struct TensorEvaluator, Device> static const int PacketSize = internal::unpacket_traits::size; typedef typename TensorEvaluator::Dimensions Dimensions; - EIGEN_DEVICE_FUNC const Dimensions& dimensions() const { return m_argImpl.dimensions(); } + EIGEN_DEVICE_FUNC const Dimensions &dimensions() const { return m_argImpl.dimensions(); } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE bool evalSubExprsIfNeeded(CoeffReturnType*) { return true; } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void cleanup() { } + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE bool evalSubExprsIfNeeded(CoeffReturnType *) { return true; } + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void cleanup() {} - EIGEN_DEVICE_FUNC CoeffReturnType coeff(Index index) const - { - return m_wrapper(m_functor, index); - } + EIGEN_DEVICE_FUNC CoeffReturnType coeff(Index index) const { return m_wrapper(m_functor, index); } - template - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE PacketReturnType packet(Index index) const + template EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE PacketReturnType packet(Index index) const { return m_wrapper.template packetOp(m_functor, index); } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TensorOpCost - costPerCoeff(bool vectorized) const { - return TensorOpCost(sizeof(CoeffReturnType), 0, 0, vectorized, - internal::unpacket_traits::size); + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TensorOpCost costPerCoeff(bool vectorized) const + { + return TensorOpCost(sizeof(CoeffReturnType), 0, 0, vectorized, internal::unpacket_traits::size); } - EIGEN_DEVICE_FUNC CoeffReturnType* data() const { return NULL; } + EIGEN_DEVICE_FUNC CoeffReturnType *data() const { return NULL; } /// required by sycl in order to extract the accessor - const TensorEvaluator& impl() const { return m_argImpl; } + const TensorEvaluator &impl() const { return m_argImpl; } /// required by sycl in order to extract the accessor NullaryOp functor() const { return m_functor; } - private: +private: const NullaryOp m_functor; TensorEvaluator m_argImpl; - const internal::nullary_wrapper m_wrapper; + const internal::nullary_wrapper m_wrapper; }; - // -------------------- CwiseUnaryOp -------------------- template @@ -298,14 +291,13 @@ struct TensorEvaluator, Device> IsAligned = TensorEvaluator::IsAligned, PacketAccess = TensorEvaluator::PacketAccess & internal::functor_traits::PacketAccess, Layout = TensorEvaluator::Layout, - CoordAccess = false, // to be implemented + CoordAccess = false,// to be implemented RawAccess = false }; - EIGEN_DEVICE_FUNC TensorEvaluator(const XprType& op, const Device& device) - : m_functor(op.functor()), - m_argImpl(op.nestedExpression(), device) - { } + EIGEN_DEVICE_FUNC TensorEvaluator(const XprType &op, const Device &device) + : m_functor(op.functor()), m_argImpl(op.nestedExpression(), device) + {} typedef typename XprType::Index Index; typedef typename XprType::Scalar Scalar; @@ -314,42 +306,37 @@ struct TensorEvaluator, Device> static const int PacketSize = internal::unpacket_traits::size; typedef typename TensorEvaluator::Dimensions Dimensions; - EIGEN_DEVICE_FUNC const Dimensions& dimensions() const { return m_argImpl.dimensions(); } + EIGEN_DEVICE_FUNC const Dimensions &dimensions() const { return m_argImpl.dimensions(); } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE bool evalSubExprsIfNeeded(Scalar*) { + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE bool evalSubExprsIfNeeded(Scalar *) + { m_argImpl.evalSubExprsIfNeeded(NULL); return true; } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void cleanup() { - m_argImpl.cleanup(); - } + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void cleanup() { m_argImpl.cleanup(); } - EIGEN_DEVICE_FUNC CoeffReturnType coeff(Index index) const - { - return m_functor(m_argImpl.coeff(index)); - } + EIGEN_DEVICE_FUNC CoeffReturnType coeff(Index index) const { return m_functor(m_argImpl.coeff(index)); } - template - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE PacketReturnType packet(Index index) const + template EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE PacketReturnType packet(Index index) const { return m_functor.packetOp(m_argImpl.template packet(index)); } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TensorOpCost costPerCoeff(bool vectorized) const { + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TensorOpCost costPerCoeff(bool vectorized) const + { const double functor_cost = internal::functor_traits::Cost; - return m_argImpl.costPerCoeff(vectorized) + - TensorOpCost(0, 0, functor_cost, vectorized, PacketSize); + return m_argImpl.costPerCoeff(vectorized) + TensorOpCost(0, 0, functor_cost, vectorized, PacketSize); } - EIGEN_DEVICE_FUNC CoeffReturnType* data() const { return NULL; } + EIGEN_DEVICE_FUNC CoeffReturnType *data() const { return NULL; } /// required by sycl in order to extract the accessor - const TensorEvaluator & impl() const { return m_argImpl; } + const TensorEvaluator &impl() const { return m_argImpl; } /// added for sycl in order to construct the buffer from sycl device UnaryOp functor() const { return m_functor; } - private: +private: const UnaryOp m_functor; TensorEvaluator m_argImpl; }; @@ -364,19 +351,21 @@ struct TensorEvaluator::IsAligned & TensorEvaluator::IsAligned, - PacketAccess = TensorEvaluator::PacketAccess & TensorEvaluator::PacketAccess & - internal::functor_traits::PacketAccess, + PacketAccess = TensorEvaluator::PacketAccess + & TensorEvaluator::PacketAccess + & internal::functor_traits::PacketAccess, Layout = TensorEvaluator::Layout, - CoordAccess = false, // to be implemented + CoordAccess = false,// to be implemented RawAccess = false }; - EIGEN_DEVICE_FUNC TensorEvaluator(const XprType& op, const Device& device) - : m_functor(op.functor()), - m_leftImpl(op.lhsExpression(), device), - m_rightImpl(op.rhsExpression(), device) + EIGEN_DEVICE_FUNC TensorEvaluator(const XprType &op, const Device &device) + : m_functor(op.functor()), m_leftImpl(op.lhsExpression(), device), m_rightImpl(op.rhsExpression(), device) { - EIGEN_STATIC_ASSERT((static_cast(TensorEvaluator::Layout) == static_cast(TensorEvaluator::Layout) || internal::traits::NumDimensions <= 1), YOU_MADE_A_PROGRAMMING_MISTAKE); + EIGEN_STATIC_ASSERT((static_cast(TensorEvaluator::Layout) + == static_cast(TensorEvaluator::Layout) + || internal::traits::NumDimensions <= 1), + YOU_MADE_A_PROGRAMMING_MISTAKE); eigen_assert(dimensions_match(m_leftImpl.dimensions(), m_rightImpl.dimensions())); } @@ -387,18 +376,20 @@ struct TensorEvaluator::size; typedef typename TensorEvaluator::Dimensions Dimensions; - EIGEN_DEVICE_FUNC const Dimensions& dimensions() const + EIGEN_DEVICE_FUNC const Dimensions &dimensions() const { // TODO: use right impl instead if right impl dimensions are known at compile time. return m_leftImpl.dimensions(); } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE bool evalSubExprsIfNeeded(CoeffReturnType*) { + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE bool evalSubExprsIfNeeded(CoeffReturnType *) + { m_leftImpl.evalSubExprsIfNeeded(NULL); m_rightImpl.evalSubExprsIfNeeded(NULL); return true; } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void cleanup() { + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void cleanup() + { m_leftImpl.cleanup(); m_rightImpl.cleanup(); } @@ -407,29 +398,28 @@ struct TensorEvaluator - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE PacketReturnType packet(Index index) const + template EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE PacketReturnType packet(Index index) const { - return m_functor.packetOp(m_leftImpl.template packet(index), m_rightImpl.template packet(index)); + return m_functor.packetOp( + m_leftImpl.template packet(index), m_rightImpl.template packet(index)); } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TensorOpCost - costPerCoeff(bool vectorized) const { + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TensorOpCost costPerCoeff(bool vectorized) const + { const double functor_cost = internal::functor_traits::Cost; - return m_leftImpl.costPerCoeff(vectorized) + - m_rightImpl.costPerCoeff(vectorized) + - TensorOpCost(0, 0, functor_cost, vectorized, PacketSize); + return m_leftImpl.costPerCoeff(vectorized) + m_rightImpl.costPerCoeff(vectorized) + + TensorOpCost(0, 0, functor_cost, vectorized, PacketSize); } - EIGEN_DEVICE_FUNC CoeffReturnType* data() const { return NULL; } + EIGEN_DEVICE_FUNC CoeffReturnType *data() const { return NULL; } /// required by sycl in order to extract the accessor - const TensorEvaluator& left_impl() const { return m_leftImpl; } + const TensorEvaluator &left_impl() const { return m_leftImpl; } /// required by sycl in order to extract the accessor - const TensorEvaluator& right_impl() const { return m_rightImpl; } + const TensorEvaluator &right_impl() const { return m_rightImpl; } /// required by sycl in order to extract the accessor BinaryOp functor() const { return m_functor; } - private: +private: const BinaryOp m_functor; TensorEvaluator m_leftImpl; TensorEvaluator m_rightImpl; @@ -443,36 +433,40 @@ struct TensorEvaluator XprType; enum { - IsAligned = TensorEvaluator::IsAligned & TensorEvaluator::IsAligned & TensorEvaluator::IsAligned, - PacketAccess = TensorEvaluator::PacketAccess & TensorEvaluator::PacketAccess & TensorEvaluator::PacketAccess & - internal::functor_traits::PacketAccess, + IsAligned = TensorEvaluator::IsAligned & TensorEvaluator::IsAligned + & TensorEvaluator::IsAligned, + PacketAccess = TensorEvaluator::PacketAccess & TensorEvaluator::PacketAccess + & TensorEvaluator::PacketAccess + & internal::functor_traits::PacketAccess, Layout = TensorEvaluator::Layout, - CoordAccess = false, // to be implemented + CoordAccess = false,// to be implemented RawAccess = false }; - EIGEN_DEVICE_FUNC TensorEvaluator(const XprType& op, const Device& device) - : m_functor(op.functor()), - m_arg1Impl(op.arg1Expression(), device), - m_arg2Impl(op.arg2Expression(), device), + EIGEN_DEVICE_FUNC TensorEvaluator(const XprType &op, const Device &device) + : m_functor(op.functor()), m_arg1Impl(op.arg1Expression(), device), m_arg2Impl(op.arg2Expression(), device), m_arg3Impl(op.arg3Expression(), device) { - EIGEN_STATIC_ASSERT((static_cast(TensorEvaluator::Layout) == static_cast(TensorEvaluator::Layout) || internal::traits::NumDimensions <= 1), YOU_MADE_A_PROGRAMMING_MISTAKE); + EIGEN_STATIC_ASSERT((static_cast(TensorEvaluator::Layout) + == static_cast(TensorEvaluator::Layout) + || internal::traits::NumDimensions <= 1), + YOU_MADE_A_PROGRAMMING_MISTAKE); EIGEN_STATIC_ASSERT((internal::is_same::StorageKind, - typename internal::traits::StorageKind>::value), - STORAGE_KIND_MUST_MATCH) + typename internal::traits::StorageKind>::value), + STORAGE_KIND_MUST_MATCH) EIGEN_STATIC_ASSERT((internal::is_same::StorageKind, - typename internal::traits::StorageKind>::value), - STORAGE_KIND_MUST_MATCH) + typename internal::traits::StorageKind>::value), + STORAGE_KIND_MUST_MATCH) EIGEN_STATIC_ASSERT((internal::is_same::Index, - typename internal::traits::Index>::value), - STORAGE_INDEX_MUST_MATCH) + typename internal::traits::Index>::value), + STORAGE_INDEX_MUST_MATCH) EIGEN_STATIC_ASSERT((internal::is_same::Index, - typename internal::traits::Index>::value), - STORAGE_INDEX_MUST_MATCH) + typename internal::traits::Index>::value), + STORAGE_INDEX_MUST_MATCH) - eigen_assert(dimensions_match(m_arg1Impl.dimensions(), m_arg2Impl.dimensions()) && dimensions_match(m_arg1Impl.dimensions(), m_arg3Impl.dimensions())); + eigen_assert(dimensions_match(m_arg1Impl.dimensions(), m_arg2Impl.dimensions()) + && dimensions_match(m_arg1Impl.dimensions(), m_arg3Impl.dimensions())); } typedef typename XprType::Index Index; @@ -482,19 +476,21 @@ struct TensorEvaluator::size; typedef typename TensorEvaluator::Dimensions Dimensions; - EIGEN_DEVICE_FUNC const Dimensions& dimensions() const + EIGEN_DEVICE_FUNC const Dimensions &dimensions() const { // TODO: use arg2 or arg3 dimensions if they are known at compile time. return m_arg1Impl.dimensions(); } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE bool evalSubExprsIfNeeded(CoeffReturnType*) { + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE bool evalSubExprsIfNeeded(CoeffReturnType *) + { m_arg1Impl.evalSubExprsIfNeeded(NULL); m_arg2Impl.evalSubExprsIfNeeded(NULL); m_arg3Impl.evalSubExprsIfNeeded(NULL); return true; } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void cleanup() { + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void cleanup() + { m_arg1Impl.cleanup(); m_arg2Impl.cleanup(); m_arg3Impl.cleanup(); @@ -504,33 +500,30 @@ struct TensorEvaluator - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE PacketReturnType packet(Index index) const + template EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE PacketReturnType packet(Index index) const { return m_functor.packetOp(m_arg1Impl.template packet(index), - m_arg2Impl.template packet(index), - m_arg3Impl.template packet(index)); + m_arg2Impl.template packet(index), + m_arg3Impl.template packet(index)); } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TensorOpCost - costPerCoeff(bool vectorized) const { + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TensorOpCost costPerCoeff(bool vectorized) const + { const double functor_cost = internal::functor_traits::Cost; - return m_arg1Impl.costPerCoeff(vectorized) + - m_arg2Impl.costPerCoeff(vectorized) + - m_arg3Impl.costPerCoeff(vectorized) + - TensorOpCost(0, 0, functor_cost, vectorized, PacketSize); + return m_arg1Impl.costPerCoeff(vectorized) + m_arg2Impl.costPerCoeff(vectorized) + + m_arg3Impl.costPerCoeff(vectorized) + TensorOpCost(0, 0, functor_cost, vectorized, PacketSize); } - EIGEN_DEVICE_FUNC CoeffReturnType* data() const { return NULL; } + EIGEN_DEVICE_FUNC CoeffReturnType *data() const { return NULL; } /// required by sycl in order to extract the accessor - const TensorEvaluator & arg1Impl() const { return m_arg1Impl; } + const TensorEvaluator &arg1Impl() const { return m_arg1Impl; } /// required by sycl in order to extract the accessor - const TensorEvaluator& arg2Impl() const { return m_arg2Impl; } + const TensorEvaluator &arg2Impl() const { return m_arg2Impl; } /// required by sycl in order to extract the accessor - const TensorEvaluator& arg3Impl() const { return m_arg3Impl; } + const TensorEvaluator &arg3Impl() const { return m_arg3Impl; } - private: +private: const TernaryOp m_functor; TensorEvaluator m_arg1Impl; TensorEvaluator m_arg2Impl; @@ -548,20 +541,23 @@ struct TensorEvaluator enum { IsAligned = TensorEvaluator::IsAligned & TensorEvaluator::IsAligned, - PacketAccess = TensorEvaluator::PacketAccess & TensorEvaluator::PacketAccess & - internal::packet_traits::HasBlend, + PacketAccess = TensorEvaluator::PacketAccess + & TensorEvaluator::PacketAccess & internal::packet_traits::HasBlend, Layout = TensorEvaluator::Layout, - CoordAccess = false, // to be implemented + CoordAccess = false,// to be implemented RawAccess = false }; - EIGEN_DEVICE_FUNC TensorEvaluator(const XprType& op, const Device& device) - : m_condImpl(op.ifExpression(), device), - m_thenImpl(op.thenExpression(), device), + EIGEN_DEVICE_FUNC TensorEvaluator(const XprType &op, const Device &device) + : m_condImpl(op.ifExpression(), device), m_thenImpl(op.thenExpression(), device), m_elseImpl(op.elseExpression(), device) { - EIGEN_STATIC_ASSERT((static_cast(TensorEvaluator::Layout) == static_cast(TensorEvaluator::Layout)), YOU_MADE_A_PROGRAMMING_MISTAKE); - EIGEN_STATIC_ASSERT((static_cast(TensorEvaluator::Layout) == static_cast(TensorEvaluator::Layout)), YOU_MADE_A_PROGRAMMING_MISTAKE); + EIGEN_STATIC_ASSERT((static_cast(TensorEvaluator::Layout) + == static_cast(TensorEvaluator::Layout)), + YOU_MADE_A_PROGRAMMING_MISTAKE); + EIGEN_STATIC_ASSERT((static_cast(TensorEvaluator::Layout) + == static_cast(TensorEvaluator::Layout)), + YOU_MADE_A_PROGRAMMING_MISTAKE); eigen_assert(dimensions_match(m_condImpl.dimensions(), m_thenImpl.dimensions())); eigen_assert(dimensions_match(m_thenImpl.dimensions(), m_elseImpl.dimensions())); } @@ -572,19 +568,21 @@ struct TensorEvaluator static const int PacketSize = internal::unpacket_traits::size; typedef typename TensorEvaluator::Dimensions Dimensions; - EIGEN_DEVICE_FUNC const Dimensions& dimensions() const + EIGEN_DEVICE_FUNC const Dimensions &dimensions() const { // TODO: use then or else impl instead if they happen to be known at compile time. return m_condImpl.dimensions(); } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE bool evalSubExprsIfNeeded(CoeffReturnType*) { + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE bool evalSubExprsIfNeeded(CoeffReturnType *) + { m_condImpl.evalSubExprsIfNeeded(NULL); m_thenImpl.evalSubExprsIfNeeded(NULL); m_elseImpl.evalSubExprsIfNeeded(NULL); return true; } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void cleanup() { + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void cleanup() + { m_condImpl.cleanup(); m_thenImpl.cleanup(); m_elseImpl.cleanup(); @@ -594,40 +592,35 @@ struct TensorEvaluator { return m_condImpl.coeff(index) ? m_thenImpl.coeff(index) : m_elseImpl.coeff(index); } - template - EIGEN_DEVICE_FUNC PacketReturnType packet(Index index) const + template EIGEN_DEVICE_FUNC PacketReturnType packet(Index index) const { internal::Selector select; - for (Index i = 0; i < PacketSize; ++i) { - select.select[i] = m_condImpl.coeff(index+i); - } - return internal::pblend(select, - m_thenImpl.template packet(index), - m_elseImpl.template packet(index)); + for (Index i = 0; i < PacketSize; ++i) { select.select[i] = m_condImpl.coeff(index + i); } + return internal::pblend( + select, m_thenImpl.template packet(index), m_elseImpl.template packet(index)); } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TensorOpCost - costPerCoeff(bool vectorized) const { - return m_condImpl.costPerCoeff(vectorized) + - m_thenImpl.costPerCoeff(vectorized) - .cwiseMax(m_elseImpl.costPerCoeff(vectorized)); + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TensorOpCost costPerCoeff(bool vectorized) const + { + return m_condImpl.costPerCoeff(vectorized) + + m_thenImpl.costPerCoeff(vectorized).cwiseMax(m_elseImpl.costPerCoeff(vectorized)); } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE CoeffReturnType* data() const { return NULL; } + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE CoeffReturnType *data() const { return NULL; } /// required by sycl in order to extract the accessor - const TensorEvaluator & cond_impl() const { return m_condImpl; } + const TensorEvaluator &cond_impl() const { return m_condImpl; } /// required by sycl in order to extract the accessor - const TensorEvaluator& then_impl() const { return m_thenImpl; } + const TensorEvaluator &then_impl() const { return m_thenImpl; } /// required by sycl in order to extract the accessor - const TensorEvaluator& else_impl() const { return m_elseImpl; } + const TensorEvaluator &else_impl() const { return m_elseImpl; } - private: +private: TensorEvaluator m_condImpl; TensorEvaluator m_thenImpl; TensorEvaluator m_elseImpl; }; -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_CXX11_TENSOR_TENSOR_EVALUATOR_H +#endif// EIGEN_CXX11_TENSOR_TENSOR_EVALUATOR_H diff --git a/filmulator-gui/core/nlmeans/eigen/unsupported/Eigen/CXX11/src/Tensor/TensorExecutor.h b/filmulator-gui/core/nlmeans/eigen/unsupported/Eigen/CXX11/src/Tensor/TensorExecutor.h index f01d77c0..f08a30cd 100644 --- a/filmulator-gui/core/nlmeans/eigen/unsupported/Eigen/CXX11/src/Tensor/TensorExecutor.h +++ b/filmulator-gui/core/nlmeans/eigen/unsupported/Eigen/CXX11/src/Tensor/TensorExecutor.h @@ -13,276 +13,251 @@ namespace Eigen { /** \class TensorExecutor - * \ingroup CXX11_Tensor_Module - * - * \brief The tensor executor class. - * - * This class is responsible for launch the evaluation of the expression on - * the specified computing device. - */ + * \ingroup CXX11_Tensor_Module + * + * \brief The tensor executor class. + * + * This class is responsible for launch the evaluation of the expression on + * the specified computing device. + */ namespace internal { -// Default strategy: the expression is evaluated with a single cpu thread. -template -class TensorExecutor -{ - public: - typedef typename Expression::Index Index; - EIGEN_DEVICE_FUNC - static inline void run(const Expression& expr, const Device& device = Device()) + // Default strategy: the expression is evaluated with a single cpu thread. + template class TensorExecutor { - TensorEvaluator evaluator(expr, device); - const bool needs_assign = evaluator.evalSubExprsIfNeeded(NULL); - if (needs_assign) + public: + typedef typename Expression::Index Index; + EIGEN_DEVICE_FUNC + static inline void run(const Expression &expr, const Device &device = Device()) { - const Index size = array_prod(evaluator.dimensions()); - for (Index i = 0; i < size; ++i) { - evaluator.evalScalar(i); + TensorEvaluator evaluator(expr, device); + const bool needs_assign = evaluator.evalSubExprsIfNeeded(NULL); + if (needs_assign) { + const Index size = array_prod(evaluator.dimensions()); + for (Index i = 0; i < size; ++i) { evaluator.evalScalar(i); } } + evaluator.cleanup(); } - evaluator.cleanup(); - } -}; + }; -template -class TensorExecutor -{ - public: - typedef typename Expression::Index Index; - EIGEN_DEVICE_FUNC - static inline void run(const Expression& expr, const DefaultDevice& device = DefaultDevice()) + template class TensorExecutor { - TensorEvaluator evaluator(expr, device); - const bool needs_assign = evaluator.evalSubExprsIfNeeded(NULL); - if (needs_assign) + public: + typedef typename Expression::Index Index; + EIGEN_DEVICE_FUNC + static inline void run(const Expression &expr, const DefaultDevice &device = DefaultDevice()) { - const Index size = array_prod(evaluator.dimensions()); - const int PacketSize = unpacket_traits::PacketReturnType>::size; - // Give the compiler a strong hint to unroll the loop. But don't insist - // on unrolling, because if the function is expensive the compiler should not - // unroll the loop at the expense of inlining. - const Index UnrolledSize = (size / (4 * PacketSize)) * 4 * PacketSize; - for (Index i = 0; i < UnrolledSize; i += 4*PacketSize) { - for (Index j = 0; j < 4; j++) { - evaluator.evalPacket(i + j * PacketSize); + TensorEvaluator evaluator(expr, device); + const bool needs_assign = evaluator.evalSubExprsIfNeeded(NULL); + if (needs_assign) { + const Index size = array_prod(evaluator.dimensions()); + const int PacketSize = + unpacket_traits::PacketReturnType>::size; + // Give the compiler a strong hint to unroll the loop. But don't insist + // on unrolling, because if the function is expensive the compiler should not + // unroll the loop at the expense of inlining. + const Index UnrolledSize = (size / (4 * PacketSize)) * 4 * PacketSize; + for (Index i = 0; i < UnrolledSize; i += 4 * PacketSize) { + for (Index j = 0; j < 4; j++) { evaluator.evalPacket(i + j * PacketSize); } } + const Index VectorizedSize = (size / PacketSize) * PacketSize; + for (Index i = UnrolledSize; i < VectorizedSize; i += PacketSize) { evaluator.evalPacket(i); } + for (Index i = VectorizedSize; i < size; ++i) { evaluator.evalScalar(i); } } - const Index VectorizedSize = (size / PacketSize) * PacketSize; - for (Index i = UnrolledSize; i < VectorizedSize; i += PacketSize) { - evaluator.evalPacket(i); - } - for (Index i = VectorizedSize; i < size; ++i) { - evaluator.evalScalar(i); - } + evaluator.cleanup(); } - evaluator.cleanup(); - } -}; - + }; // Multicore strategy: the index space is partitioned and each partition is executed on a single core #ifdef EIGEN_USE_THREADS -template -struct EvalRange { - static void run(Evaluator* evaluator_in, const Index first, const Index last) { - Evaluator evaluator = *evaluator_in; - eigen_assert(last >= first); - for (Index i = first; i < last; ++i) { - evaluator.evalScalar(i); + template struct EvalRange + { + static void run(Evaluator *evaluator_in, const Index first, const Index last) + { + Evaluator evaluator = *evaluator_in; + eigen_assert(last >= first); + for (Index i = first; i < last; ++i) { evaluator.evalScalar(i); } } - } - static Index alignBlockSize(Index size) { - return size; - } -}; - -template -struct EvalRange { - static const int PacketSize = unpacket_traits::size; - - static void run(Evaluator* evaluator_in, const Index first, const Index last) { - Evaluator evaluator = *evaluator_in; - eigen_assert(last >= first); - Index i = first; - if (last - first >= PacketSize) { - eigen_assert(first % PacketSize == 0); - Index last_chunk_offset = last - 4 * PacketSize; - // Give the compiler a strong hint to unroll the loop. But don't insist - // on unrolling, because if the function is expensive the compiler should not - // unroll the loop at the expense of inlining. - for (; i <= last_chunk_offset; i += 4*PacketSize) { - for (Index j = 0; j < 4; j++) { - evaluator.evalPacket(i + j * PacketSize); + static Index alignBlockSize(Index size) { return size; } + }; + + template struct EvalRange + { + static const int PacketSize = unpacket_traits::size; + + static void run(Evaluator *evaluator_in, const Index first, const Index last) + { + Evaluator evaluator = *evaluator_in; + eigen_assert(last >= first); + Index i = first; + if (last - first >= PacketSize) { + eigen_assert(first % PacketSize == 0); + Index last_chunk_offset = last - 4 * PacketSize; + // Give the compiler a strong hint to unroll the loop. But don't insist + // on unrolling, because if the function is expensive the compiler should not + // unroll the loop at the expense of inlining. + for (; i <= last_chunk_offset; i += 4 * PacketSize) { + for (Index j = 0; j < 4; j++) { evaluator.evalPacket(i + j * PacketSize); } } + last_chunk_offset = last - PacketSize; + for (; i <= last_chunk_offset; i += PacketSize) { evaluator.evalPacket(i); } } - last_chunk_offset = last - PacketSize; - for (; i <= last_chunk_offset; i += PacketSize) { - evaluator.evalPacket(i); - } + for (; i < last; ++i) { evaluator.evalScalar(i); } } - for (; i < last; ++i) { - evaluator.evalScalar(i); - } - } - static Index alignBlockSize(Index size) { - // Align block size to packet size and account for unrolling in run above. - if (size >= 16 * PacketSize) { - return (size + 4 * PacketSize - 1) & ~(4 * PacketSize - 1); + static Index alignBlockSize(Index size) + { + // Align block size to packet size and account for unrolling in run above. + if (size >= 16 * PacketSize) { return (size + 4 * PacketSize - 1) & ~(4 * PacketSize - 1); } + // Aligning to 4 * PacketSize would increase block size by more than 25%. + return (size + PacketSize - 1) & ~(PacketSize - 1); } - // Aligning to 4 * PacketSize would increase block size by more than 25%. - return (size + PacketSize - 1) & ~(PacketSize - 1); - } -}; + }; -template -class TensorExecutor { - public: - typedef typename Expression::Index Index; - static inline void run(const Expression& expr, const ThreadPoolDevice& device) + template class TensorExecutor { - typedef TensorEvaluator Evaluator; - Evaluator evaluator(expr, device); - const bool needs_assign = evaluator.evalSubExprsIfNeeded(NULL); - if (needs_assign) + public: + typedef typename Expression::Index Index; + static inline void run(const Expression &expr, const ThreadPoolDevice &device) { - const Index size = array_prod(evaluator.dimensions()); + typedef TensorEvaluator Evaluator; + Evaluator evaluator(expr, device); + const bool needs_assign = evaluator.evalSubExprsIfNeeded(NULL); + if (needs_assign) { + const Index size = array_prod(evaluator.dimensions()); #if !defined(EIGEN_USE_SIMPLE_THREAD_POOL) - device.parallelFor(size, evaluator.costPerCoeff(Vectorizable), - EvalRange::alignBlockSize, - [&evaluator](Index first, Index last) { - EvalRange::run(&evaluator, first, last); - }); + device.parallelFor(size, + evaluator.costPerCoeff(Vectorizable), + EvalRange::alignBlockSize, + [&evaluator]( + Index first, Index last) { EvalRange::run(&evaluator, first, last); }); #else - size_t num_threads = device.numThreads(); - if (num_threads > 1) { - num_threads = TensorCostModel::numThreads( - size, evaluator.costPerCoeff(Vectorizable), num_threads); - } - if (num_threads == 1) { - EvalRange::run(&evaluator, 0, size); - } else { - const Index PacketSize = Vectorizable ? unpacket_traits::size : 1; - Index blocksz = std::ceil(static_cast(size)/num_threads) + PacketSize - 1; - const Index blocksize = numext::maxi(PacketSize, (blocksz - (blocksz % PacketSize))); - const Index numblocks = size / blocksize; - - Barrier barrier(numblocks); - for (int i = 0; i < numblocks; ++i) { - device.enqueue_with_barrier( - &barrier, &EvalRange::run, - &evaluator, i * blocksize, (i + 1) * blocksize); + size_t num_threads = device.numThreads(); + if (num_threads > 1) { + num_threads = + TensorCostModel::numThreads(size, evaluator.costPerCoeff(Vectorizable), num_threads); } - if (numblocks * blocksize < size) { - EvalRange::run( - &evaluator, numblocks * blocksize, size); + if (num_threads == 1) { + EvalRange::run(&evaluator, 0, size); + } else { + const Index PacketSize = Vectorizable ? unpacket_traits::size : 1; + Index blocksz = std::ceil(static_cast(size) / num_threads) + PacketSize - 1; + const Index blocksize = numext::maxi(PacketSize, (blocksz - (blocksz % PacketSize))); + const Index numblocks = size / blocksize; + + Barrier barrier(numblocks); + for (int i = 0; i < numblocks; ++i) { + device.enqueue_with_barrier(&barrier, + &EvalRange::run, + &evaluator, + i * blocksize, + (i + 1) * blocksize); + } + if (numblocks * blocksize < size) { + EvalRange::run(&evaluator, numblocks * blocksize, size); + } + barrier.Wait(); } - barrier.Wait(); +#endif// defined(!EIGEN_USE_SIMPLE_THREAD_POOL) } -#endif // defined(!EIGEN_USE_SIMPLE_THREAD_POOL) + evaluator.cleanup(); } - evaluator.cleanup(); - } -}; -#endif // EIGEN_USE_THREADS + }; +#endif// EIGEN_USE_THREADS // GPU: the evaluation of the expression is offloaded to a GPU. #if defined(EIGEN_USE_GPU) -template -class TensorExecutor { - public: - typedef typename Expression::Index Index; - static void run(const Expression& expr, const GpuDevice& device); -}; + template class TensorExecutor + { + public: + typedef typename Expression::Index Index; + static void run(const Expression &expr, const GpuDevice &device); + }; #if defined(__CUDACC__) -template -struct EigenMetaKernelEval { - static __device__ EIGEN_ALWAYS_INLINE - void run(Evaluator& eval, Index first, Index last, Index step_size) { - for (Index i = first; i < last; i += step_size) { - eval.evalScalar(i); - } - } -}; - -template -struct EigenMetaKernelEval { - static __device__ EIGEN_ALWAYS_INLINE - void run(Evaluator& eval, Index first, Index last, Index step_size) { - const Index PacketSize = unpacket_traits::size; - const Index vectorized_size = (last / PacketSize) * PacketSize; - const Index vectorized_step_size = step_size * PacketSize; - - // Use the vector path - for (Index i = first * PacketSize; i < vectorized_size; - i += vectorized_step_size) { - eval.evalPacket(i); + template struct EigenMetaKernelEval + { + static __device__ EIGEN_ALWAYS_INLINE void run(Evaluator &eval, Index first, Index last, Index step_size) + { + for (Index i = first; i < last; i += step_size) { eval.evalScalar(i); } } - for (Index i = vectorized_size + first; i < last; i += step_size) { - eval.evalScalar(i); + }; + + template struct EigenMetaKernelEval + { + static __device__ EIGEN_ALWAYS_INLINE void run(Evaluator &eval, Index first, Index last, Index step_size) + { + const Index PacketSize = unpacket_traits::size; + const Index vectorized_size = (last / PacketSize) * PacketSize; + const Index vectorized_step_size = step_size * PacketSize; + + // Use the vector path + for (Index i = first * PacketSize; i < vectorized_size; i += vectorized_step_size) { eval.evalPacket(i); } + for (Index i = vectorized_size + first; i < last; i += step_size) { eval.evalScalar(i); } } + }; + + template + __global__ void __launch_bounds__(1024) EigenMetaKernel(Evaluator eval, Index size) + { + + const Index first_index = blockIdx.x * blockDim.x + threadIdx.x; + const Index step_size = blockDim.x * gridDim.x; + + const bool vectorizable = Evaluator::PacketAccess & Evaluator::IsAligned; + EigenMetaKernelEval::run(eval, first_index, size, step_size); } -}; - -template -__global__ void -__launch_bounds__(1024) -EigenMetaKernel(Evaluator eval, Index size) { - - const Index first_index = blockIdx.x * blockDim.x + threadIdx.x; - const Index step_size = blockDim.x * gridDim.x; - - const bool vectorizable = Evaluator::PacketAccess & Evaluator::IsAligned; - EigenMetaKernelEval::run(eval, first_index, size, step_size); -} - -/*static*/ -template -inline void TensorExecutor::run( - const Expression& expr, const GpuDevice& device) { - TensorEvaluator evaluator(expr, device); - const bool needs_assign = evaluator.evalSubExprsIfNeeded(NULL); - if (needs_assign) { - const int block_size = device.maxCudaThreadsPerBlock(); - const int max_blocks = device.getNumCudaMultiProcessors() * - device.maxCudaThreadsPerMultiProcessor() / block_size; - const Index size = array_prod(evaluator.dimensions()); - // Create a least one block to ensure we won't crash when tensorflow calls with tensors of size 0. - const int num_blocks = numext::maxi(numext::mini(max_blocks, divup(size, block_size)), 1); - - LAUNCH_CUDA_KERNEL( - (EigenMetaKernel, Index>), - num_blocks, block_size, 0, device, evaluator, size); + + /*static*/ + template + inline void TensorExecutor::run(const Expression &expr, const GpuDevice &device) + { + TensorEvaluator evaluator(expr, device); + const bool needs_assign = evaluator.evalSubExprsIfNeeded(NULL); + if (needs_assign) { + const int block_size = device.maxCudaThreadsPerBlock(); + const int max_blocks = device.getNumCudaMultiProcessors() * device.maxCudaThreadsPerMultiProcessor() / block_size; + const Index size = array_prod(evaluator.dimensions()); + // Create a least one block to ensure we won't crash when tensorflow calls with tensors of size 0. + const int num_blocks = numext::maxi(numext::mini(max_blocks, divup(size, block_size)), 1); + + LAUNCH_CUDA_KERNEL((EigenMetaKernel, Index>), + num_blocks, + block_size, + 0, + device, + evaluator, + size); + } + evaluator.cleanup(); } - evaluator.cleanup(); -} -#endif // __CUDACC__ -#endif // EIGEN_USE_GPU +#endif// __CUDACC__ +#endif// EIGEN_USE_GPU // SYCL Executor policy #ifdef EIGEN_USE_SYCL -template -class TensorExecutor { -public: - static inline void run(const Expression &expr, const SyclDevice &device) { - // call TensorSYCL module - TensorSycl::run(expr, device); - } -}; + template class TensorExecutor + { + public: + static inline void run(const Expression &expr, const SyclDevice &device) + { + // call TensorSYCL module + TensorSycl::run(expr, device); + } + }; #endif -} // end namespace internal +}// end namespace internal -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_CXX11_TENSOR_TENSOR_EXECUTOR_H +#endif// EIGEN_CXX11_TENSOR_TENSOR_EXECUTOR_H diff --git a/filmulator-gui/core/nlmeans/eigen/unsupported/Eigen/CXX11/src/Tensor/TensorExpr.h b/filmulator-gui/core/nlmeans/eigen/unsupported/Eigen/CXX11/src/Tensor/TensorExpr.h index 85dfc7a6..b7e63447 100644 --- a/filmulator-gui/core/nlmeans/eigen/unsupported/Eigen/CXX11/src/Tensor/TensorExpr.h +++ b/filmulator-gui/core/nlmeans/eigen/unsupported/Eigen/CXX11/src/Tensor/TensorExpr.h @@ -13,359 +13,344 @@ namespace Eigen { /** \class TensorExpr - * \ingroup CXX11_Tensor_Module - * - * \brief Tensor expression classes. - * - * The TensorCwiseNullaryOp class applies a nullary operators to an expression. - * This is typically used to generate constants. - * - * The TensorCwiseUnaryOp class represents an expression where a unary operator - * (e.g. cwiseSqrt) is applied to an expression. - * - * The TensorCwiseBinaryOp class represents an expression where a binary - * operator (e.g. addition) is applied to a lhs and a rhs expression. - * - */ + * \ingroup CXX11_Tensor_Module + * + * \brief Tensor expression classes. + * + * The TensorCwiseNullaryOp class applies a nullary operators to an expression. + * This is typically used to generate constants. + * + * The TensorCwiseUnaryOp class represents an expression where a unary operator + * (e.g. cwiseSqrt) is applied to an expression. + * + * The TensorCwiseBinaryOp class represents an expression where a binary + * operator (e.g. addition) is applied to a lhs and a rhs expression. + * + */ namespace internal { -template -struct traits > - : traits -{ - typedef traits XprTraits; - typedef typename XprType::Scalar Scalar; - typedef typename XprType::Nested XprTypeNested; - typedef typename remove_reference::type _XprTypeNested; - static const int NumDimensions = XprTraits::NumDimensions; - static const int Layout = XprTraits::Layout; - - enum { - Flags = 0 + template + struct traits> : traits + { + typedef traits XprTraits; + typedef typename XprType::Scalar Scalar; + typedef typename XprType::Nested XprTypeNested; + typedef typename remove_reference::type _XprTypeNested; + static const int NumDimensions = XprTraits::NumDimensions; + static const int Layout = XprTraits::Layout; + + enum { Flags = 0 }; }; -}; - -} // end namespace internal +}// end namespace internal template class TensorCwiseNullaryOp : public TensorBase, ReadOnlyAccessors> { - public: - typedef typename Eigen::internal::traits::Scalar Scalar; - typedef typename Eigen::NumTraits::Real RealScalar; - typedef typename XprType::CoeffReturnType CoeffReturnType; - typedef TensorCwiseNullaryOp Nested; - typedef typename Eigen::internal::traits::StorageKind StorageKind; - typedef typename Eigen::internal::traits::Index Index; - - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TensorCwiseNullaryOp(const XprType& xpr, const NullaryOp& func = NullaryOp()) - : m_xpr(xpr), m_functor(func) {} - - EIGEN_DEVICE_FUNC - const typename internal::remove_all::type& - nestedExpression() const { return m_xpr; } - - EIGEN_DEVICE_FUNC - const NullaryOp& functor() const { return m_functor; } - - protected: - typename XprType::Nested m_xpr; - const NullaryOp m_functor; +public: + typedef typename Eigen::internal::traits::Scalar Scalar; + typedef typename Eigen::NumTraits::Real RealScalar; + typedef typename XprType::CoeffReturnType CoeffReturnType; + typedef TensorCwiseNullaryOp Nested; + typedef typename Eigen::internal::traits::StorageKind StorageKind; + typedef typename Eigen::internal::traits::Index Index; + + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TensorCwiseNullaryOp(const XprType &xpr, const NullaryOp &func = NullaryOp()) + : m_xpr(xpr), m_functor(func) + {} + + EIGEN_DEVICE_FUNC + const typename internal::remove_all::type &nestedExpression() const { return m_xpr; } + + EIGEN_DEVICE_FUNC + const NullaryOp &functor() const { return m_functor; } + +protected: + typename XprType::Nested m_xpr; + const NullaryOp m_functor; }; - namespace internal { -template -struct traits > - : traits -{ - // TODO(phli): Add InputScalar, InputPacket. Check references to - // current Scalar/Packet to see if the intent is Input or Output. - typedef typename result_of::type Scalar; - typedef traits XprTraits; - typedef typename XprType::Nested XprTypeNested; - typedef typename remove_reference::type _XprTypeNested; - static const int NumDimensions = XprTraits::NumDimensions; - static const int Layout = XprTraits::Layout; -}; - -template -struct eval, Eigen::Dense> -{ - typedef const TensorCwiseUnaryOp& type; -}; + template struct traits> : traits + { + // TODO(phli): Add InputScalar, InputPacket. Check references to + // current Scalar/Packet to see if the intent is Input or Output. + typedef typename result_of::type Scalar; + typedef traits XprTraits; + typedef typename XprType::Nested XprTypeNested; + typedef typename remove_reference::type _XprTypeNested; + static const int NumDimensions = XprTraits::NumDimensions; + static const int Layout = XprTraits::Layout; + }; -template -struct nested, 1, typename eval >::type> -{ - typedef TensorCwiseUnaryOp type; -}; + template struct eval, Eigen::Dense> + { + typedef const TensorCwiseUnaryOp &type; + }; -} // end namespace internal + template + struct nested, 1, typename eval>::type> + { + typedef TensorCwiseUnaryOp type; + }; +}// end namespace internal template class TensorCwiseUnaryOp : public TensorBase, ReadOnlyAccessors> { - public: - // TODO(phli): Add InputScalar, InputPacket. Check references to - // current Scalar/Packet to see if the intent is Input or Output. - typedef typename Eigen::internal::traits::Scalar Scalar; - typedef typename Eigen::NumTraits::Real RealScalar; - typedef Scalar CoeffReturnType; - typedef typename Eigen::internal::nested::type Nested; - typedef typename Eigen::internal::traits::StorageKind StorageKind; - typedef typename Eigen::internal::traits::Index Index; - - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TensorCwiseUnaryOp(const XprType& xpr, const UnaryOp& func = UnaryOp()) - : m_xpr(xpr), m_functor(func) {} - - EIGEN_DEVICE_FUNC - const UnaryOp& functor() const { return m_functor; } - - /** \returns the nested expression */ - EIGEN_DEVICE_FUNC - const typename internal::remove_all::type& - nestedExpression() const { return m_xpr; } - - protected: - typename XprType::Nested m_xpr; - const UnaryOp m_functor; +public: + // TODO(phli): Add InputScalar, InputPacket. Check references to + // current Scalar/Packet to see if the intent is Input or Output. + typedef typename Eigen::internal::traits::Scalar Scalar; + typedef typename Eigen::NumTraits::Real RealScalar; + typedef Scalar CoeffReturnType; + typedef typename Eigen::internal::nested::type Nested; + typedef typename Eigen::internal::traits::StorageKind StorageKind; + typedef typename Eigen::internal::traits::Index Index; + + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TensorCwiseUnaryOp(const XprType &xpr, const UnaryOp &func = UnaryOp()) + : m_xpr(xpr), m_functor(func) + {} + + EIGEN_DEVICE_FUNC + const UnaryOp &functor() const { return m_functor; } + + /** \returns the nested expression */ + EIGEN_DEVICE_FUNC + const typename internal::remove_all::type &nestedExpression() const { return m_xpr; } + +protected: + typename XprType::Nested m_xpr; + const UnaryOp m_functor; }; namespace internal { -template -struct traits > -{ - // Type promotion to handle the case where the types of the lhs and the rhs - // are different. - // TODO(phli): Add Lhs/RhsScalar, Lhs/RhsPacket. Check references to - // current Scalar/Packet to see if the intent is Inputs or Output. - typedef typename result_of< - BinaryOp(typename LhsXprType::Scalar, - typename RhsXprType::Scalar)>::type Scalar; - typedef traits XprTraits; - typedef typename promote_storage_type< - typename traits::StorageKind, + template + struct traits> + { + // Type promotion to handle the case where the types of the lhs and the rhs + // are different. + // TODO(phli): Add Lhs/RhsScalar, Lhs/RhsPacket. Check references to + // current Scalar/Packet to see if the intent is Inputs or Output. + typedef typename result_of::type Scalar; + typedef traits XprTraits; + typedef typename promote_storage_type::StorageKind, typename traits::StorageKind>::ret StorageKind; - typedef typename promote_index_type< - typename traits::Index, - typename traits::Index>::type Index; - typedef typename LhsXprType::Nested LhsNested; - typedef typename RhsXprType::Nested RhsNested; - typedef typename remove_reference::type _LhsNested; - typedef typename remove_reference::type _RhsNested; - static const int NumDimensions = XprTraits::NumDimensions; - static const int Layout = XprTraits::Layout; - - enum { - Flags = 0 + typedef + typename promote_index_type::Index, typename traits::Index>::type Index; + typedef typename LhsXprType::Nested LhsNested; + typedef typename RhsXprType::Nested RhsNested; + typedef typename remove_reference::type _LhsNested; + typedef typename remove_reference::type _RhsNested; + static const int NumDimensions = XprTraits::NumDimensions; + static const int Layout = XprTraits::Layout; + + enum { Flags = 0 }; }; -}; - -template -struct eval, Eigen::Dense> -{ - typedef const TensorCwiseBinaryOp& type; -}; -template -struct nested, 1, typename eval >::type> -{ - typedef TensorCwiseBinaryOp type; -}; + template + struct eval, Eigen::Dense> + { + typedef const TensorCwiseBinaryOp &type; + }; -} // end namespace internal + template + struct nested, + 1, + typename eval>::type> + { + typedef TensorCwiseBinaryOp type; + }; +}// end namespace internal template class TensorCwiseBinaryOp : public TensorBase, ReadOnlyAccessors> { - public: - // TODO(phli): Add Lhs/RhsScalar, Lhs/RhsPacket. Check references to - // current Scalar/Packet to see if the intent is Inputs or Output. - typedef typename Eigen::internal::traits::Scalar Scalar; - typedef typename Eigen::NumTraits::Real RealScalar; - typedef Scalar CoeffReturnType; - typedef typename Eigen::internal::nested::type Nested; - typedef typename Eigen::internal::traits::StorageKind StorageKind; - typedef typename Eigen::internal::traits::Index Index; - - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TensorCwiseBinaryOp(const LhsXprType& lhs, const RhsXprType& rhs, const BinaryOp& func = BinaryOp()) - : m_lhs_xpr(lhs), m_rhs_xpr(rhs), m_functor(func) {} - - EIGEN_DEVICE_FUNC - const BinaryOp& functor() const { return m_functor; } - - /** \returns the nested expressions */ - EIGEN_DEVICE_FUNC - const typename internal::remove_all::type& - lhsExpression() const { return m_lhs_xpr; } - - EIGEN_DEVICE_FUNC - const typename internal::remove_all::type& - rhsExpression() const { return m_rhs_xpr; } - - protected: - typename LhsXprType::Nested m_lhs_xpr; - typename RhsXprType::Nested m_rhs_xpr; - const BinaryOp m_functor; +public: + // TODO(phli): Add Lhs/RhsScalar, Lhs/RhsPacket. Check references to + // current Scalar/Packet to see if the intent is Inputs or Output. + typedef typename Eigen::internal::traits::Scalar Scalar; + typedef typename Eigen::NumTraits::Real RealScalar; + typedef Scalar CoeffReturnType; + typedef typename Eigen::internal::nested::type Nested; + typedef typename Eigen::internal::traits::StorageKind StorageKind; + typedef typename Eigen::internal::traits::Index Index; + + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TensorCwiseBinaryOp(const LhsXprType &lhs, + const RhsXprType &rhs, + const BinaryOp &func = BinaryOp()) + : m_lhs_xpr(lhs), m_rhs_xpr(rhs), m_functor(func) + {} + + EIGEN_DEVICE_FUNC + const BinaryOp &functor() const { return m_functor; } + + /** \returns the nested expressions */ + EIGEN_DEVICE_FUNC + const typename internal::remove_all::type &lhsExpression() const { return m_lhs_xpr; } + + EIGEN_DEVICE_FUNC + const typename internal::remove_all::type &rhsExpression() const { return m_rhs_xpr; } + +protected: + typename LhsXprType::Nested m_lhs_xpr; + typename RhsXprType::Nested m_rhs_xpr; + const BinaryOp m_functor; }; namespace internal { -template -struct traits > -{ - // Type promotion to handle the case where the types of the args are different. - typedef typename result_of< - TernaryOp(typename Arg1XprType::Scalar, - typename Arg2XprType::Scalar, - typename Arg3XprType::Scalar)>::type Scalar; - typedef traits XprTraits; - typedef typename traits::StorageKind StorageKind; - typedef typename traits::Index Index; - typedef typename Arg1XprType::Nested Arg1Nested; - typedef typename Arg2XprType::Nested Arg2Nested; - typedef typename Arg3XprType::Nested Arg3Nested; - typedef typename remove_reference::type _Arg1Nested; - typedef typename remove_reference::type _Arg2Nested; - typedef typename remove_reference::type _Arg3Nested; - static const int NumDimensions = XprTraits::NumDimensions; - static const int Layout = XprTraits::Layout; - - enum { - Flags = 0 + template + struct traits> + { + // Type promotion to handle the case where the types of the args are different. + typedef typename result_of< + TernaryOp(typename Arg1XprType::Scalar, typename Arg2XprType::Scalar, typename Arg3XprType::Scalar)>::type Scalar; + typedef traits XprTraits; + typedef typename traits::StorageKind StorageKind; + typedef typename traits::Index Index; + typedef typename Arg1XprType::Nested Arg1Nested; + typedef typename Arg2XprType::Nested Arg2Nested; + typedef typename Arg3XprType::Nested Arg3Nested; + typedef typename remove_reference::type _Arg1Nested; + typedef typename remove_reference::type _Arg2Nested; + typedef typename remove_reference::type _Arg3Nested; + static const int NumDimensions = XprTraits::NumDimensions; + static const int Layout = XprTraits::Layout; + + enum { Flags = 0 }; }; -}; - -template -struct eval, Eigen::Dense> -{ - typedef const TensorCwiseTernaryOp& type; -}; -template -struct nested, 1, typename eval >::type> -{ - typedef TensorCwiseTernaryOp type; -}; + template + struct eval, Eigen::Dense> + { + typedef const TensorCwiseTernaryOp &type; + }; -} // end namespace internal + template + struct nested, + 1, + typename eval>::type> + { + typedef TensorCwiseTernaryOp type; + }; +}// end namespace internal template -class TensorCwiseTernaryOp : public TensorBase, ReadOnlyAccessors> +class TensorCwiseTernaryOp + : public TensorBase, ReadOnlyAccessors> { - public: - typedef typename Eigen::internal::traits::Scalar Scalar; - typedef typename Eigen::NumTraits::Real RealScalar; - typedef Scalar CoeffReturnType; - typedef typename Eigen::internal::nested::type Nested; - typedef typename Eigen::internal::traits::StorageKind StorageKind; - typedef typename Eigen::internal::traits::Index Index; - - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TensorCwiseTernaryOp(const Arg1XprType& arg1, const Arg2XprType& arg2, const Arg3XprType& arg3, const TernaryOp& func = TernaryOp()) - : m_arg1_xpr(arg1), m_arg2_xpr(arg2), m_arg3_xpr(arg3), m_functor(func) {} - - EIGEN_DEVICE_FUNC - const TernaryOp& functor() const { return m_functor; } - - /** \returns the nested expressions */ - EIGEN_DEVICE_FUNC - const typename internal::remove_all::type& - arg1Expression() const { return m_arg1_xpr; } - - EIGEN_DEVICE_FUNC - const typename internal::remove_all::type& - arg2Expression() const { return m_arg2_xpr; } - - EIGEN_DEVICE_FUNC - const typename internal::remove_all::type& - arg3Expression() const { return m_arg3_xpr; } - - protected: - typename Arg1XprType::Nested m_arg1_xpr; - typename Arg2XprType::Nested m_arg2_xpr; - typename Arg3XprType::Nested m_arg3_xpr; - const TernaryOp m_functor; +public: + typedef typename Eigen::internal::traits::Scalar Scalar; + typedef typename Eigen::NumTraits::Real RealScalar; + typedef Scalar CoeffReturnType; + typedef typename Eigen::internal::nested::type Nested; + typedef typename Eigen::internal::traits::StorageKind StorageKind; + typedef typename Eigen::internal::traits::Index Index; + + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TensorCwiseTernaryOp(const Arg1XprType &arg1, + const Arg2XprType &arg2, + const Arg3XprType &arg3, + const TernaryOp &func = TernaryOp()) + : m_arg1_xpr(arg1), m_arg2_xpr(arg2), m_arg3_xpr(arg3), m_functor(func) + {} + + EIGEN_DEVICE_FUNC + const TernaryOp &functor() const { return m_functor; } + + /** \returns the nested expressions */ + EIGEN_DEVICE_FUNC + const typename internal::remove_all::type &arg1Expression() const { return m_arg1_xpr; } + + EIGEN_DEVICE_FUNC + const typename internal::remove_all::type &arg2Expression() const { return m_arg2_xpr; } + + EIGEN_DEVICE_FUNC + const typename internal::remove_all::type &arg3Expression() const { return m_arg3_xpr; } + +protected: + typename Arg1XprType::Nested m_arg1_xpr; + typename Arg2XprType::Nested m_arg2_xpr; + typename Arg3XprType::Nested m_arg3_xpr; + const TernaryOp m_functor; }; namespace internal { -template -struct traits > - : traits -{ - typedef typename traits::Scalar Scalar; - typedef traits XprTraits; - typedef typename promote_storage_type::StorageKind, - typename traits::StorageKind>::ret StorageKind; - typedef typename promote_index_type::Index, - typename traits::Index>::type Index; - typedef typename IfXprType::Nested IfNested; - typedef typename ThenXprType::Nested ThenNested; - typedef typename ElseXprType::Nested ElseNested; - static const int NumDimensions = XprTraits::NumDimensions; - static const int Layout = XprTraits::Layout; -}; + template + struct traits> : traits + { + typedef typename traits::Scalar Scalar; + typedef traits XprTraits; + typedef typename promote_storage_type::StorageKind, + typename traits::StorageKind>::ret StorageKind; + typedef + typename promote_index_type::Index, typename traits::Index>::type Index; + typedef typename IfXprType::Nested IfNested; + typedef typename ThenXprType::Nested ThenNested; + typedef typename ElseXprType::Nested ElseNested; + static const int NumDimensions = XprTraits::NumDimensions; + static const int Layout = XprTraits::Layout; + }; -template -struct eval, Eigen::Dense> -{ - typedef const TensorSelectOp& type; -}; + template + struct eval, Eigen::Dense> + { + typedef const TensorSelectOp &type; + }; -template -struct nested, 1, typename eval >::type> -{ - typedef TensorSelectOp type; -}; + template + struct nested, + 1, + typename eval>::type> + { + typedef TensorSelectOp type; + }; -} // end namespace internal +}// end namespace internal template class TensorSelectOp : public TensorBase, ReadOnlyAccessors> { - public: - typedef typename Eigen::internal::traits::Scalar Scalar; - typedef typename Eigen::NumTraits::Real RealScalar; - typedef typename internal::promote_storage_type::ret CoeffReturnType; - typedef typename Eigen::internal::nested::type Nested; - typedef typename Eigen::internal::traits::StorageKind StorageKind; - typedef typename Eigen::internal::traits::Index Index; - - EIGEN_DEVICE_FUNC - TensorSelectOp(const IfXprType& a_condition, - const ThenXprType& a_then, - const ElseXprType& a_else) - : m_condition(a_condition), m_then(a_then), m_else(a_else) - { } - - EIGEN_DEVICE_FUNC - const IfXprType& ifExpression() const { return m_condition; } - - EIGEN_DEVICE_FUNC - const ThenXprType& thenExpression() const { return m_then; } - - EIGEN_DEVICE_FUNC - const ElseXprType& elseExpression() const { return m_else; } - - protected: - typename IfXprType::Nested m_condition; - typename ThenXprType::Nested m_then; - typename ElseXprType::Nested m_else; +public: + typedef typename Eigen::internal::traits::Scalar Scalar; + typedef typename Eigen::NumTraits::Real RealScalar; + typedef typename internal::promote_storage_type::ret CoeffReturnType; + typedef typename Eigen::internal::nested::type Nested; + typedef typename Eigen::internal::traits::StorageKind StorageKind; + typedef typename Eigen::internal::traits::Index Index; + + EIGEN_DEVICE_FUNC + TensorSelectOp(const IfXprType &a_condition, const ThenXprType &a_then, const ElseXprType &a_else) + : m_condition(a_condition), m_then(a_then), m_else(a_else) + {} + + EIGEN_DEVICE_FUNC + const IfXprType &ifExpression() const { return m_condition; } + + EIGEN_DEVICE_FUNC + const ThenXprType &thenExpression() const { return m_then; } + + EIGEN_DEVICE_FUNC + const ElseXprType &elseExpression() const { return m_else; } + +protected: + typename IfXprType::Nested m_condition; + typename ThenXprType::Nested m_then; + typename ElseXprType::Nested m_else; }; -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_CXX11_TENSOR_TENSOR_EXPR_H +#endif// EIGEN_CXX11_TENSOR_TENSOR_EXPR_H diff --git a/filmulator-gui/core/nlmeans/eigen/unsupported/Eigen/CXX11/src/Tensor/TensorFFT.h b/filmulator-gui/core/nlmeans/eigen/unsupported/Eigen/CXX11/src/Tensor/TensorFFT.h index 08eb5595..22abc44a 100644 --- a/filmulator-gui/core/nlmeans/eigen/unsupported/Eigen/CXX11/src/Tensor/TensorFFT.h +++ b/filmulator-gui/core/nlmeans/eigen/unsupported/Eigen/CXX11/src/Tensor/TensorFFT.h @@ -17,105 +17,115 @@ namespace Eigen { /** \class TensorFFT - * \ingroup CXX11_Tensor_Module - * - * \brief Tensor FFT class. - * - * TODO: - * Vectorize the Cooley Tukey and the Bluestein algorithm - * Add support for multithreaded evaluation - * Improve the performance on GPU - */ - -template struct MakeComplex { - template - EIGEN_DEVICE_FUNC - T operator() (const T& val) const { return val; } + * \ingroup CXX11_Tensor_Module + * + * \brief Tensor FFT class. + * + * TODO: + * Vectorize the Cooley Tukey and the Bluestein algorithm + * Add support for multithreaded evaluation + * Improve the performance on GPU + */ + +template struct MakeComplex +{ + template EIGEN_DEVICE_FUNC T operator()(const T &val) const { return val; } }; -template <> struct MakeComplex { - template - EIGEN_DEVICE_FUNC - std::complex operator() (const T& val) const { return std::complex(val, 0); } +template<> struct MakeComplex +{ + template EIGEN_DEVICE_FUNC std::complex operator()(const T &val) const + { + return std::complex(val, 0); + } }; -template <> struct MakeComplex { - template - EIGEN_DEVICE_FUNC - std::complex operator() (const std::complex& val) const { return val; } +template<> struct MakeComplex +{ + template EIGEN_DEVICE_FUNC std::complex operator()(const std::complex &val) const { return val; } }; -template struct PartOf { - template T operator() (const T& val) const { return val; } +template struct PartOf +{ + template T operator()(const T &val) const { return val; } }; -template <> struct PartOf { - template T operator() (const std::complex& val) const { return val.real(); } +template<> struct PartOf +{ + template T operator()(const std::complex &val) const { return val.real(); } }; -template <> struct PartOf { - template T operator() (const std::complex& val) const { return val.imag(); } +template<> struct PartOf +{ + template T operator()(const std::complex &val) const { return val.imag(); } }; namespace internal { -template -struct traits > : public traits { - typedef traits XprTraits; - typedef typename NumTraits::Real RealScalar; - typedef typename std::complex ComplexScalar; - typedef typename XprTraits::Scalar InputScalar; - typedef typename conditional::type OutputScalar; - typedef typename XprTraits::StorageKind StorageKind; - typedef typename XprTraits::Index Index; - typedef typename XprType::Nested Nested; - typedef typename remove_reference::type _Nested; - static const int NumDimensions = XprTraits::NumDimensions; - static const int Layout = XprTraits::Layout; -}; + template + struct traits> : public traits + { + typedef traits XprTraits; + typedef typename NumTraits::Real RealScalar; + typedef typename std::complex ComplexScalar; + typedef typename XprTraits::Scalar InputScalar; + typedef + typename conditional::type + OutputScalar; + typedef typename XprTraits::StorageKind StorageKind; + typedef typename XprTraits::Index Index; + typedef typename XprType::Nested Nested; + typedef typename remove_reference::type _Nested; + static const int NumDimensions = XprTraits::NumDimensions; + static const int Layout = XprTraits::Layout; + }; -template -struct eval, Eigen::Dense> { - typedef const TensorFFTOp& type; -}; + template + struct eval, Eigen::Dense> + { + typedef const TensorFFTOp &type; + }; -template -struct nested, 1, typename eval >::type> { - typedef TensorFFTOp type; -}; + template + struct nested, + 1, + typename eval>::type> + { + typedef TensorFFTOp type; + }; -} // end namespace internal +}// end namespace internal -template -class TensorFFTOp : public TensorBase, ReadOnlyAccessors> { - public: +template +class TensorFFTOp : public TensorBase, ReadOnlyAccessors> +{ +public: typedef typename Eigen::internal::traits::Scalar Scalar; typedef typename Eigen::NumTraits::Real RealScalar; typedef typename std::complex ComplexScalar; - typedef typename internal::conditional::type OutputScalar; + typedef typename internal:: + conditional::type OutputScalar; typedef OutputScalar CoeffReturnType; typedef typename Eigen::internal::nested::type Nested; typedef typename Eigen::internal::traits::StorageKind StorageKind; typedef typename Eigen::internal::traits::Index Index; - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TensorFFTOp(const XprType& expr, const FFT& fft) - : m_xpr(expr), m_fft(fft) {} + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TensorFFTOp(const XprType &expr, const FFT &fft) : m_xpr(expr), m_fft(fft) {} EIGEN_DEVICE_FUNC - const FFT& fft() const { return m_fft; } + const FFT &fft() const { return m_fft; } EIGEN_DEVICE_FUNC - const typename internal::remove_all::type& expression() const { - return m_xpr; - } + const typename internal::remove_all::type &expression() const { return m_xpr; } - protected: +protected: typename XprType::Nested m_xpr; const FFT m_fft; }; // Eval as rvalue -template -struct TensorEvaluator, Device> { +template +struct TensorEvaluator, Device> +{ typedef TensorFFTOp XprType; typedef typename XprType::Index Index; static const int NumDims = internal::array_size::Dimensions>::value; @@ -126,7 +136,8 @@ struct TensorEvaluator, D typedef typename TensorEvaluator::Dimensions InputDimensions; typedef internal::traits XprTraits; typedef typename XprTraits::Scalar InputScalar; - typedef typename internal::conditional::type OutputScalar; + typedef typename internal:: + conditional::type OutputScalar; typedef OutputScalar CoeffReturnType; typedef typename PacketType::type PacketReturnType; static const int PacketSize = internal::unpacket_traits::size; @@ -140,8 +151,10 @@ struct TensorEvaluator, D RawAccess = false }; - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TensorEvaluator(const XprType& op, const Device& device) : m_fft(op.fft()), m_impl(op.expression(), device), m_data(NULL), m_device(device) { - const typename TensorEvaluator::Dimensions& input_dims = m_impl.dimensions(); + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TensorEvaluator(const XprType &op, const Device &device) + : m_fft(op.fft()), m_impl(op.expression(), device), m_data(NULL), m_device(device) + { + const typename TensorEvaluator::Dimensions &input_dims = m_impl.dimensions(); for (int i = 0; i < NumDims; ++i) { eigen_assert(input_dims[i] > 0); m_dimensions[i] = input_dims[i]; @@ -149,35 +162,31 @@ struct TensorEvaluator, D if (static_cast(Layout) == static_cast(ColMajor)) { m_strides[0] = 1; - for (int i = 1; i < NumDims; ++i) { - m_strides[i] = m_strides[i - 1] * m_dimensions[i - 1]; - } + for (int i = 1; i < NumDims; ++i) { m_strides[i] = m_strides[i - 1] * m_dimensions[i - 1]; } } else { m_strides[NumDims - 1] = 1; - for (int i = NumDims - 2; i >= 0; --i) { - m_strides[i] = m_strides[i + 1] * m_dimensions[i + 1]; - } + for (int i = NumDims - 2; i >= 0; --i) { m_strides[i] = m_strides[i + 1] * m_dimensions[i + 1]; } } m_size = m_dimensions.TotalSize(); } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const Dimensions& dimensions() const { - return m_dimensions; - } + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const Dimensions &dimensions() const { return m_dimensions; } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE bool evalSubExprsIfNeeded(OutputScalar* data) { + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE bool evalSubExprsIfNeeded(OutputScalar *data) + { m_impl.evalSubExprsIfNeeded(NULL); if (data) { evalToBuf(data); return false; } else { - m_data = (CoeffReturnType*)m_device.allocate(sizeof(CoeffReturnType) * m_size); + m_data = (CoeffReturnType *)m_device.allocate(sizeof(CoeffReturnType) * m_size); evalToBuf(m_data); return true; } } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void cleanup() { + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void cleanup() + { if (m_data) { m_device.deallocate(m_data); m_data = NULL; @@ -185,28 +194,27 @@ struct TensorEvaluator, D m_impl.cleanup(); } - EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE CoeffReturnType coeff(Index index) const { - return m_data[index]; - } + EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE CoeffReturnType coeff(Index index) const { return m_data[index]; } - template - EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE PacketReturnType - packet(Index index) const { + template EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE PacketReturnType packet(Index index) const + { return internal::ploadt(m_data + index); } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TensorOpCost - costPerCoeff(bool vectorized) const { + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TensorOpCost costPerCoeff(bool vectorized) const + { return TensorOpCost(sizeof(CoeffReturnType), 0, 0, vectorized, PacketSize); } - EIGEN_DEVICE_FUNC Scalar* data() const { return m_data; } + EIGEN_DEVICE_FUNC Scalar *data() const { return m_data; } - private: - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void evalToBuf(OutputScalar* data) { +private: + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void evalToBuf(OutputScalar *data) + { const bool write_to_out = internal::is_same::value; - ComplexScalar* buf = write_to_out ? (ComplexScalar*)data : (ComplexScalar*)m_device.allocate(sizeof(ComplexScalar) * m_size); + ComplexScalar *buf = + write_to_out ? (ComplexScalar *)data : (ComplexScalar *)m_device.allocate(sizeof(ComplexScalar) * m_size); for (Index i = 0; i < m_size; ++i) { buf[i] = MakeComplex::value>()(m_impl.coeff(i)); @@ -217,14 +225,17 @@ struct TensorEvaluator, D eigen_assert(dim >= 0 && dim < NumDims); Index line_len = m_dimensions[dim]; eigen_assert(line_len >= 1); - ComplexScalar* line_buf = (ComplexScalar*)m_device.allocate(sizeof(ComplexScalar) * line_len); + ComplexScalar *line_buf = (ComplexScalar *)m_device.allocate(sizeof(ComplexScalar) * line_len); const bool is_power_of_two = isPowerOfTwo(line_len); const Index good_composite = is_power_of_two ? 0 : findGoodComposite(line_len); const Index log_len = is_power_of_two ? getLog2(line_len) : getLog2(good_composite); - ComplexScalar* a = is_power_of_two ? NULL : (ComplexScalar*)m_device.allocate(sizeof(ComplexScalar) * good_composite); - ComplexScalar* b = is_power_of_two ? NULL : (ComplexScalar*)m_device.allocate(sizeof(ComplexScalar) * good_composite); - ComplexScalar* pos_j_base_powered = is_power_of_two ? NULL : (ComplexScalar*)m_device.allocate(sizeof(ComplexScalar) * (line_len + 1)); + ComplexScalar *a = + is_power_of_two ? NULL : (ComplexScalar *)m_device.allocate(sizeof(ComplexScalar) * good_composite); + ComplexScalar *b = + is_power_of_two ? NULL : (ComplexScalar *)m_device.allocate(sizeof(ComplexScalar) * good_composite); + ComplexScalar *pos_j_base_powered = + is_power_of_two ? NULL : (ComplexScalar *)m_device.allocate(sizeof(ComplexScalar) * (line_len + 1)); if (!is_power_of_two) { // Compute twiddle factors // t_n = exp(sqrt(-1) * pi * n^2 / line_len) @@ -233,15 +244,13 @@ struct TensorEvaluator, D pos_j_base_powered[0] = ComplexScalar(1, 0); if (line_len > 1) { const RealScalar pi_over_len(EIGEN_PI / line_len); - const ComplexScalar pos_j_base = ComplexScalar( - std::cos(pi_over_len), std::sin(pi_over_len)); + const ComplexScalar pos_j_base = ComplexScalar(std::cos(pi_over_len), std::sin(pi_over_len)); pos_j_base_powered[1] = pos_j_base; if (line_len > 2) { const ComplexScalar pos_j_base_sq = pos_j_base * pos_j_base; for (int j = 2; j < line_len + 1; ++j) { - pos_j_base_powered[j] = pos_j_base_powered[j - 1] * - pos_j_base_powered[j - 1] / - pos_j_base_powered[j - 2] * pos_j_base_sq; + pos_j_base_powered[j] = + pos_j_base_powered[j - 1] * pos_j_base_powered[j - 1] / pos_j_base_powered[j - 2] * pos_j_base_sq; } } } @@ -253,30 +262,27 @@ struct TensorEvaluator, D // get data into line_buf const Index stride = m_strides[dim]; if (stride == 1) { - memcpy(line_buf, &buf[base_offset], line_len*sizeof(ComplexScalar)); + memcpy(line_buf, &buf[base_offset], line_len * sizeof(ComplexScalar)); } else { Index offset = base_offset; - for (int j = 0; j < line_len; ++j, offset += stride) { - line_buf[j] = buf[offset]; - } + for (int j = 0; j < line_len; ++j, offset += stride) { line_buf[j] = buf[offset]; } } // processs the line if (is_power_of_two) { processDataLineCooleyTukey(line_buf, line_len, log_len); - } - else { + } else { processDataLineBluestein(line_buf, line_len, good_composite, log_len, a, b, pos_j_base_powered); } // write back if (FFTDir == FFT_FORWARD && stride == 1) { - memcpy(&buf[base_offset], line_buf, line_len*sizeof(ComplexScalar)); + memcpy(&buf[base_offset], line_buf, line_len * sizeof(ComplexScalar)); } else { Index offset = base_offset; - const ComplexScalar div_factor = ComplexScalar(1.0 / line_len, 0); + const ComplexScalar div_factor = ComplexScalar(1.0 / line_len, 0); for (int j = 0; j < line_len; ++j, offset += stride) { - buf[offset] = (FFTDir == FFT_FORWARD) ? line_buf[j] : line_buf[j] * div_factor; + buf[offset] = (FFTDir == FFT_FORWARD) ? line_buf[j] : line_buf[j] * div_factor; } } } @@ -288,74 +294,77 @@ struct TensorEvaluator, D } } - if(!write_to_out) { - for (Index i = 0; i < m_size; ++i) { - data[i] = PartOf()(buf[i]); - } + if (!write_to_out) { + for (Index i = 0; i < m_size; ++i) { data[i] = PartOf()(buf[i]); } m_device.deallocate(buf); } } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE static bool isPowerOfTwo(Index x) { + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE static bool isPowerOfTwo(Index x) + { eigen_assert(x > 0); return !(x & (x - 1)); } // The composite number for padding, used in Bluestein's FFT algorithm - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE static Index findGoodComposite(Index n) { + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE static Index findGoodComposite(Index n) + { Index i = 2; while (i < 2 * n - 1) i *= 2; return i; } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE static Index getLog2(Index m) { + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE static Index getLog2(Index m) + { Index log2m = 0; while (m >>= 1) log2m++; return log2m; } // Call Cooley Tukey algorithm directly, data length must be power of 2 - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void processDataLineCooleyTukey(ComplexScalar* line_buf, Index line_len, Index log_len) { + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void + processDataLineCooleyTukey(ComplexScalar *line_buf, Index line_len, Index log_len) + { eigen_assert(isPowerOfTwo(line_len)); scramble_FFT(line_buf, line_len); compute_1D_Butterfly(line_buf, line_len, log_len); } // Call Bluestein's FFT algorithm, m is a good composite number greater than (2 * n - 1), used as the padding length - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void processDataLineBluestein(ComplexScalar* line_buf, Index line_len, Index good_composite, Index log_len, ComplexScalar* a, ComplexScalar* b, const ComplexScalar* pos_j_base_powered) { + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void processDataLineBluestein(ComplexScalar *line_buf, + Index line_len, + Index good_composite, + Index log_len, + ComplexScalar *a, + ComplexScalar *b, + const ComplexScalar *pos_j_base_powered) + { Index n = line_len; Index m = good_composite; - ComplexScalar* data = line_buf; + ComplexScalar *data = line_buf; for (Index i = 0; i < n; ++i) { - if(FFTDir == FFT_FORWARD) { + if (FFTDir == FFT_FORWARD) { a[i] = data[i] * numext::conj(pos_j_base_powered[i]); - } - else { + } else { a[i] = data[i] * pos_j_base_powered[i]; } } - for (Index i = n; i < m; ++i) { - a[i] = ComplexScalar(0, 0); - } + for (Index i = n; i < m; ++i) { a[i] = ComplexScalar(0, 0); } for (Index i = 0; i < n; ++i) { - if(FFTDir == FFT_FORWARD) { + if (FFTDir == FFT_FORWARD) { b[i] = pos_j_base_powered[i]; - } - else { + } else { b[i] = numext::conj(pos_j_base_powered[i]); } } - for (Index i = n; i < m - n; ++i) { - b[i] = ComplexScalar(0, 0); - } + for (Index i = n; i < m - n; ++i) { b[i] = ComplexScalar(0, 0); } for (Index i = m - n; i < m; ++i) { - if(FFTDir == FFT_FORWARD) { - b[i] = pos_j_base_powered[m-i]; - } - else { - b[i] = numext::conj(pos_j_base_powered[m-i]); + if (FFTDir == FFT_FORWARD) { + b[i] = pos_j_base_powered[m - i]; + } else { + b[i] = numext::conj(pos_j_base_powered[m - i]); } } @@ -365,35 +374,29 @@ struct TensorEvaluator, D scramble_FFT(b, m); compute_1D_Butterfly(b, m, log_len); - for (Index i = 0; i < m; ++i) { - a[i] *= b[i]; - } + for (Index i = 0; i < m; ++i) { a[i] *= b[i]; } scramble_FFT(a, m); compute_1D_Butterfly(a, m, log_len); - //Do the scaling after ifft - for (Index i = 0; i < m; ++i) { - a[i] /= m; - } + // Do the scaling after ifft + for (Index i = 0; i < m; ++i) { a[i] /= m; } for (Index i = 0; i < n; ++i) { - if(FFTDir == FFT_FORWARD) { + if (FFTDir == FFT_FORWARD) { data[i] = a[i] * numext::conj(pos_j_base_powered[i]); - } - else { + } else { data[i] = a[i] * pos_j_base_powered[i]; } } } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE static void scramble_FFT(ComplexScalar* data, Index n) { + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE static void scramble_FFT(ComplexScalar *data, Index n) + { eigen_assert(isPowerOfTwo(n)); Index j = 1; - for (Index i = 1; i < n; ++i){ - if (j > i) { - std::swap(data[j-1], data[i-1]); - } + for (Index i = 1; i < n; ++i) { + if (j > i) { std::swap(data[j - 1], data[i - 1]); } Index m = n >> 1; while (m >= 2 && j > m) { j -= m; @@ -403,15 +406,15 @@ struct TensorEvaluator, D } } - template - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void butterfly_2(ComplexScalar* data) { + template EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void butterfly_2(ComplexScalar *data) + { ComplexScalar tmp = data[1]; data[1] = data[0] - data[1]; data[0] += tmp; } - template - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void butterfly_4(ComplexScalar* data) { + template EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void butterfly_4(ComplexScalar *data) + { ComplexScalar tmp[4]; tmp[0] = data[0] + data[1]; tmp[1] = data[0] - data[1]; @@ -427,8 +430,8 @@ struct TensorEvaluator, D data[3] = tmp[1] - tmp[3]; } - template - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void butterfly_8(ComplexScalar* data) { + template EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void butterfly_8(ComplexScalar *data) + { ComplexScalar tmp_1[8]; ComplexScalar tmp_2[8]; @@ -474,16 +477,15 @@ struct TensorEvaluator, D data[7] = tmp_2[3] - tmp_2[7]; } - template - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void butterfly_1D_merge( - ComplexScalar* data, Index n, Index n_power_of_2) { + template + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void butterfly_1D_merge(ComplexScalar *data, Index n, Index n_power_of_2) + { // Original code: // RealScalar wtemp = std::sin(M_PI/n); // RealScalar wpi = -std::sin(2 * M_PI/n); const RealScalar wtemp = m_sin_PI_div_n_LUT[n_power_of_2]; - const RealScalar wpi = (Dir == FFT_FORWARD) - ? m_minus_sin_2_PI_div_n_LUT[n_power_of_2] - : -m_minus_sin_2_PI_div_n_LUT[n_power_of_2]; + const RealScalar wpi = + (Dir == FFT_FORWARD) ? m_minus_sin_2_PI_div_n_LUT[n_power_of_2] : -m_minus_sin_2_PI_div_n_LUT[n_power_of_2]; const ComplexScalar wp(wtemp, wpi); const ComplexScalar wp_one = wp + ComplexScalar(1, 0); @@ -493,29 +495,29 @@ struct TensorEvaluator, D const Index n2 = n / 2; ComplexScalar w(1.0, 0.0); for (Index i = 0; i < n2; i += 4) { - ComplexScalar temp0(data[i + n2] * w); - ComplexScalar temp1(data[i + 1 + n2] * w * wp_one); - ComplexScalar temp2(data[i + 2 + n2] * w * wp_one_2); - ComplexScalar temp3(data[i + 3 + n2] * w * wp_one_3); - w = w * wp_one_4; + ComplexScalar temp0(data[i + n2] * w); + ComplexScalar temp1(data[i + 1 + n2] * w * wp_one); + ComplexScalar temp2(data[i + 2 + n2] * w * wp_one_2); + ComplexScalar temp3(data[i + 3 + n2] * w * wp_one_3); + w = w * wp_one_4; - data[i + n2] = data[i] - temp0; - data[i] += temp0; + data[i + n2] = data[i] - temp0; + data[i] += temp0; - data[i + 1 + n2] = data[i + 1] - temp1; - data[i + 1] += temp1; + data[i + 1 + n2] = data[i + 1] - temp1; + data[i + 1] += temp1; - data[i + 2 + n2] = data[i + 2] - temp2; - data[i + 2] += temp2; + data[i + 2 + n2] = data[i + 2] - temp2; + data[i + 2] += temp2; - data[i + 3 + n2] = data[i + 3] - temp3; - data[i + 3] += temp3; + data[i + 3 + n2] = data[i + 3] - temp3; + data[i + 3] += temp3; } } - template - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void compute_1D_Butterfly( - ComplexScalar* data, Index n, Index n_power_of_2) { + template + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void compute_1D_Butterfly(ComplexScalar *data, Index n, Index n_power_of_2) + { eigen_assert(isPowerOfTwo(n)); if (n > 8) { compute_1D_Butterfly(data, n / 2, n_power_of_2 - 1); @@ -530,7 +532,8 @@ struct TensorEvaluator, D } } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Index getBaseOffsetFromIndex(Index index, Index omitted_dim) const { + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Index getBaseOffsetFromIndex(Index index, Index omitted_dim) const + { Index result = 0; if (static_cast(Layout) == static_cast(ColMajor)) { @@ -541,8 +544,7 @@ struct TensorEvaluator, D result += idx * m_strides[i]; } result += index; - } - else { + } else { for (Index i = 0; i < omitted_dim; ++i) { const Index partial_m_stride = m_strides[i] / m_dimensions[omitted_dim]; const Index idx = index / partial_m_stride; @@ -555,24 +557,24 @@ struct TensorEvaluator, D return result; } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Index getIndexFromOffset(Index base, Index omitted_dim, Index offset) const { - Index result = base + offset * m_strides[omitted_dim] ; + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Index getIndexFromOffset(Index base, Index omitted_dim, Index offset) const + { + Index result = base + offset * m_strides[omitted_dim]; return result; } - protected: +protected: Index m_size; - const FFT& m_fft; + const FFT &m_fft; Dimensions m_dimensions; array m_strides; TensorEvaluator m_impl; - CoeffReturnType* m_data; - const Device& m_device; + CoeffReturnType *m_data; + const Device &m_device; // This will support a maximum FFT size of 2^32 for each dimension // m_sin_PI_div_n_LUT[i] = (-2) * std::sin(M_PI / std::pow(2,i)) ^ 2; - const RealScalar m_sin_PI_div_n_LUT[32] = { - RealScalar(0.0), + const RealScalar m_sin_PI_div_n_LUT[32] = { RealScalar(0.0), RealScalar(-2), RealScalar(-0.999999999999999), RealScalar(-0.292893218813453), @@ -591,7 +593,7 @@ struct TensorEvaluator, D RealScalar(-4.59589268710903e-09), RealScalar(-1.14897317243732e-09), RealScalar(-2.87243293150586e-10), - RealScalar( -7.18108232902250e-11), + RealScalar(-7.18108232902250e-11), RealScalar(-1.79527058227174e-11), RealScalar(-4.48817645568941e-12), RealScalar(-1.12204411392298e-12), @@ -603,12 +605,10 @@ struct TensorEvaluator, D RealScalar(-2.73936551250781e-16), RealScalar(-6.84841378126949e-17), RealScalar(-1.71210344531737e-17), - RealScalar(-4.28025861329343e-18) - }; + RealScalar(-4.28025861329343e-18) }; // m_minus_sin_2_PI_div_n_LUT[i] = -std::sin(2 * M_PI / std::pow(2,i)); - const RealScalar m_minus_sin_2_PI_div_n_LUT[32] = { - RealScalar(0.0), + const RealScalar m_minus_sin_2_PI_div_n_LUT[32] = { RealScalar(0.0), RealScalar(0.0), RealScalar(-1.00000000000000e+00), RealScalar(-7.07106781186547e-01), @@ -639,13 +639,12 @@ struct TensorEvaluator, D RealScalar(-2.34066892682746e-08), RealScalar(-1.17033446341373e-08), RealScalar(-5.85167231706864e-09), - RealScalar(-2.92583615853432e-09) - }; + RealScalar(-2.92583615853432e-09) }; }; -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_HAS_CONSTEXPR +#endif// EIGEN_HAS_CONSTEXPR -#endif // EIGEN_CXX11_TENSOR_TENSOR_FFT_H +#endif// EIGEN_CXX11_TENSOR_TENSOR_FFT_H diff --git a/filmulator-gui/core/nlmeans/eigen/unsupported/Eigen/CXX11/src/Tensor/TensorFixedSize.h b/filmulator-gui/core/nlmeans/eigen/unsupported/Eigen/CXX11/src/Tensor/TensorFixedSize.h index fcee5f60..04fb9c57 100644 --- a/filmulator-gui/core/nlmeans/eigen/unsupported/Eigen/CXX11/src/Tensor/TensorFixedSize.h +++ b/filmulator-gui/core/nlmeans/eigen/unsupported/Eigen/CXX11/src/Tensor/TensorFixedSize.h @@ -13,377 +13,385 @@ namespace Eigen { /** \class TensorFixedSize - * \ingroup CXX11_Tensor_Module - * - * \brief The fixed sized version of the tensor class. - * - * The fixed sized equivalent of - * Eigen::Tensor t(3, 5, 7); - * is - * Eigen::TensorFixedSize> t; - */ + * \ingroup CXX11_Tensor_Module + * + * \brief The fixed sized version of the tensor class. + * + * The fixed sized equivalent of + * Eigen::Tensor t(3, 5, 7); + * is + * Eigen::TensorFixedSize> t; + */ template -class TensorFixedSize : public TensorBase > +class TensorFixedSize : public TensorBase> { - public: - typedef TensorFixedSize Self; - typedef TensorBase > Base; - typedef typename Eigen::internal::nested::type Nested; - typedef typename internal::traits::StorageKind StorageKind; - typedef typename internal::traits::Index Index; - typedef Scalar_ Scalar; - typedef typename NumTraits::Real RealScalar; - typedef typename Base::CoeffReturnType CoeffReturnType; - - static const int Options = Options_; - - enum { - IsAligned = bool(EIGEN_MAX_ALIGN_BYTES>0), - Layout = Options_ & RowMajor ? RowMajor : ColMajor, - CoordAccess = true, - RawAccess = true - }; +public: + typedef TensorFixedSize Self; + typedef TensorBase> Base; + typedef typename Eigen::internal::nested::type Nested; + typedef typename internal::traits::StorageKind StorageKind; + typedef typename internal::traits::Index Index; + typedef Scalar_ Scalar; + typedef typename NumTraits::Real RealScalar; + typedef typename Base::CoeffReturnType CoeffReturnType; + + static const int Options = Options_; + + enum { + IsAligned = bool(EIGEN_MAX_ALIGN_BYTES > 0), + Layout = Options_ & RowMajor ? RowMajor : ColMajor, + CoordAccess = true, + RawAccess = true + }; typedef Dimensions_ Dimensions; static const std::size_t NumIndices = Dimensions::count; - protected: +protected: TensorStorage m_storage; - public: - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Index rank() const { return NumIndices; } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Index dimension(std::size_t n) const { return m_storage.dimensions()[n]; } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const Dimensions& dimensions() const { return m_storage.dimensions(); } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Index size() const { return m_storage.size(); } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Scalar *data() { return m_storage.data(); } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const Scalar *data() const { return m_storage.data(); } +public: + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Index rank() const { return NumIndices; } + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Index dimension(std::size_t n) const { return m_storage.dimensions()[n]; } + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const Dimensions &dimensions() const { return m_storage.dimensions(); } + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Index size() const { return m_storage.size(); } + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Scalar *data() { return m_storage.data(); } + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const Scalar *data() const { return m_storage.data(); } - // This makes EIGEN_INITIALIZE_COEFFS_IF_THAT_OPTION_IS_ENABLED - // work, because that uses base().coeffRef() - and we don't yet - // implement a similar class hierarchy - inline Self& base() { return *this; } - inline const Self& base() const { return *this; } + // This makes EIGEN_INITIALIZE_COEFFS_IF_THAT_OPTION_IS_ENABLED + // work, because that uses base().coeffRef() - and we don't yet + // implement a similar class hierarchy + inline Self &base() { return *this; } + inline const Self &base() const { return *this; } #if EIGEN_HAS_VARIADIC_TEMPLATES - template - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const Scalar& coeff(Index firstIndex, IndexTypes... otherIndices) const - { - // The number of indices used to access a tensor coefficient must be equal to the rank of the tensor. - EIGEN_STATIC_ASSERT(sizeof...(otherIndices) + 1 == NumIndices, YOU_MADE_A_PROGRAMMING_MISTAKE) - return coeff(array{{firstIndex, otherIndices...}}); - } + template + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const Scalar &coeff(Index firstIndex, IndexTypes... otherIndices) const + { + // The number of indices used to access a tensor coefficient must be equal to the rank of the tensor. + EIGEN_STATIC_ASSERT(sizeof...(otherIndices) + 1 == NumIndices, YOU_MADE_A_PROGRAMMING_MISTAKE) + return coeff(array{ { firstIndex, otherIndices... } }); + } #endif - EIGEN_DEVICE_FUNC - EIGEN_STRONG_INLINE const Scalar& coeff(const array& indices) const - { - eigen_internal_assert(checkIndexRange(indices)); - return m_storage.data()[linearizedIndex(indices)]; - } + EIGEN_DEVICE_FUNC + EIGEN_STRONG_INLINE const Scalar &coeff(const array &indices) const + { + eigen_internal_assert(checkIndexRange(indices)); + return m_storage.data()[linearizedIndex(indices)]; + } - EIGEN_DEVICE_FUNC - EIGEN_STRONG_INLINE const Scalar& coeff(Index index) const - { - eigen_internal_assert(index >= 0 && index < size()); - return m_storage.data()[index]; - } + EIGEN_DEVICE_FUNC + EIGEN_STRONG_INLINE const Scalar &coeff(Index index) const + { + eigen_internal_assert(index >= 0 && index < size()); + return m_storage.data()[index]; + } - EIGEN_DEVICE_FUNC - EIGEN_STRONG_INLINE const Scalar& coeff() const - { - EIGEN_STATIC_ASSERT(NumIndices == 0, YOU_MADE_A_PROGRAMMING_MISTAKE); - return m_storage.data()[0]; - } + EIGEN_DEVICE_FUNC + EIGEN_STRONG_INLINE const Scalar &coeff() const + { + EIGEN_STATIC_ASSERT(NumIndices == 0, YOU_MADE_A_PROGRAMMING_MISTAKE); + return m_storage.data()[0]; + } #if EIGEN_HAS_VARIADIC_TEMPLATES - template - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Scalar& coeffRef(Index firstIndex, IndexTypes... otherIndices) - { - // The number of indices used to access a tensor coefficient must be equal to the rank of the tensor. - EIGEN_STATIC_ASSERT(sizeof...(otherIndices) + 1 == NumIndices, YOU_MADE_A_PROGRAMMING_MISTAKE) - return coeffRef(array{{firstIndex, otherIndices...}}); - } + template + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Scalar &coeffRef(Index firstIndex, IndexTypes... otherIndices) + { + // The number of indices used to access a tensor coefficient must be equal to the rank of the tensor. + EIGEN_STATIC_ASSERT(sizeof...(otherIndices) + 1 == NumIndices, YOU_MADE_A_PROGRAMMING_MISTAKE) + return coeffRef(array{ { firstIndex, otherIndices... } }); + } #endif - EIGEN_DEVICE_FUNC - EIGEN_STRONG_INLINE Scalar& coeffRef(const array& indices) - { - eigen_internal_assert(checkIndexRange(indices)); - return m_storage.data()[linearizedIndex(indices)]; - } - - EIGEN_DEVICE_FUNC - EIGEN_STRONG_INLINE Scalar& coeffRef(Index index) - { - eigen_internal_assert(index >= 0 && index < size()); - return m_storage.data()[index]; - } - - EIGEN_DEVICE_FUNC - EIGEN_STRONG_INLINE Scalar& coeffRef() - { - EIGEN_STATIC_ASSERT(NumIndices == 0, YOU_MADE_A_PROGRAMMING_MISTAKE); - return m_storage.data()[0]; - } + EIGEN_DEVICE_FUNC + EIGEN_STRONG_INLINE Scalar &coeffRef(const array &indices) + { + eigen_internal_assert(checkIndexRange(indices)); + return m_storage.data()[linearizedIndex(indices)]; + } + + EIGEN_DEVICE_FUNC + EIGEN_STRONG_INLINE Scalar &coeffRef(Index index) + { + eigen_internal_assert(index >= 0 && index < size()); + return m_storage.data()[index]; + } + + EIGEN_DEVICE_FUNC + EIGEN_STRONG_INLINE Scalar &coeffRef() + { + EIGEN_STATIC_ASSERT(NumIndices == 0, YOU_MADE_A_PROGRAMMING_MISTAKE); + return m_storage.data()[0]; + } #if EIGEN_HAS_VARIADIC_TEMPLATES - template - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const Scalar& operator()(Index firstIndex, IndexTypes... otherIndices) const - { - // The number of indices used to access a tensor coefficient must be equal to the rank of the tensor. - EIGEN_STATIC_ASSERT(sizeof...(otherIndices) + 1 == NumIndices, YOU_MADE_A_PROGRAMMING_MISTAKE) - return this->operator()(array{{firstIndex, otherIndices...}}); - } + template + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const Scalar &operator()(Index firstIndex, IndexTypes... otherIndices) const + { + // The number of indices used to access a tensor coefficient must be equal to the rank of the tensor. + EIGEN_STATIC_ASSERT(sizeof...(otherIndices) + 1 == NumIndices, YOU_MADE_A_PROGRAMMING_MISTAKE) + return this->operator()(array{ { firstIndex, otherIndices... } }); + } #else - EIGEN_DEVICE_FUNC - EIGEN_STRONG_INLINE const Scalar& operator()(Index i0, Index i1) const - { - if (Options&RowMajor) { - const Index index = i1 + i0 * m_storage.dimensions()[1]; - return m_storage.data()[index]; - } else { - const Index index = i0 + i1 * m_storage.dimensions()[0]; - return m_storage.data()[index]; - } + EIGEN_DEVICE_FUNC + EIGEN_STRONG_INLINE const Scalar &operator()(Index i0, Index i1) const + { + if (Options & RowMajor) { + const Index index = i1 + i0 * m_storage.dimensions()[1]; + return m_storage.data()[index]; + } else { + const Index index = i0 + i1 * m_storage.dimensions()[0]; + return m_storage.data()[index]; } - EIGEN_DEVICE_FUNC - EIGEN_STRONG_INLINE const Scalar& operator()(Index i0, Index i1, Index i2) const - { - if (Options&RowMajor) { - const Index index = i2 + m_storage.dimensions()[2] * (i1 + m_storage.dimensions()[1] * i0); - return m_storage.data()[index]; - } else { - const Index index = i0 + m_storage.dimensions()[0] * (i1 + m_storage.dimensions()[1] * i2); - return m_storage.data()[index]; - } + } + EIGEN_DEVICE_FUNC + EIGEN_STRONG_INLINE const Scalar &operator()(Index i0, Index i1, Index i2) const + { + if (Options & RowMajor) { + const Index index = i2 + m_storage.dimensions()[2] * (i1 + m_storage.dimensions()[1] * i0); + return m_storage.data()[index]; + } else { + const Index index = i0 + m_storage.dimensions()[0] * (i1 + m_storage.dimensions()[1] * i2); + return m_storage.data()[index]; } - EIGEN_DEVICE_FUNC - EIGEN_STRONG_INLINE const Scalar& operator()(Index i0, Index i1, Index i2, Index i3) const - { - if (Options&RowMajor) { - const Index index = i3 + m_storage.dimensions()[3] * (i2 + m_storage.dimensions()[2] * (i1 + m_storage.dimensions()[1] * i0)); - return m_storage.data()[index]; - } else { - const Index index = i0 + m_storage.dimensions()[0] * (i1 + m_storage.dimensions()[1] * (i2 + m_storage.dimensions()[2] * i3)); - return m_storage.data()[index]; - } + } + EIGEN_DEVICE_FUNC + EIGEN_STRONG_INLINE const Scalar &operator()(Index i0, Index i1, Index i2, Index i3) const + { + if (Options & RowMajor) { + const Index index = + i3 + m_storage.dimensions()[3] * (i2 + m_storage.dimensions()[2] * (i1 + m_storage.dimensions()[1] * i0)); + return m_storage.data()[index]; + } else { + const Index index = + i0 + m_storage.dimensions()[0] * (i1 + m_storage.dimensions()[1] * (i2 + m_storage.dimensions()[2] * i3)); + return m_storage.data()[index]; } - EIGEN_DEVICE_FUNC - EIGEN_STRONG_INLINE const Scalar& operator()(Index i0, Index i1, Index i2, Index i3, Index i4) const - { - if (Options&RowMajor) { - const Index index = i4 + m_storage.dimensions()[4] * (i3 + m_storage.dimensions()[3] * (i2 + m_storage.dimensions()[2] * (i1 + m_storage.dimensions()[1] * i0))); - return m_storage.data()[index]; - } else { - const Index index = i0 + m_storage.dimensions()[0] * (i1 + m_storage.dimensions()[1] * (i2 + m_storage.dimensions()[2] * (i3 + m_storage.dimensions()[3] * i4))); - return m_storage.data()[index]; - } + } + EIGEN_DEVICE_FUNC + EIGEN_STRONG_INLINE const Scalar &operator()(Index i0, Index i1, Index i2, Index i3, Index i4) const + { + if (Options & RowMajor) { + const Index index = + i4 + + m_storage.dimensions()[4] + * (i3 + + m_storage.dimensions()[3] * (i2 + m_storage.dimensions()[2] * (i1 + m_storage.dimensions()[1] * i0))); + return m_storage.data()[index]; + } else { + const Index index = + i0 + + m_storage.dimensions()[0] + * (i1 + + m_storage.dimensions()[1] * (i2 + m_storage.dimensions()[2] * (i3 + m_storage.dimensions()[3] * i4))); + return m_storage.data()[index]; } + } #endif - EIGEN_DEVICE_FUNC - EIGEN_STRONG_INLINE const Scalar& operator()(const array& indices) const - { - eigen_assert(checkIndexRange(indices)); - return coeff(indices); - } - - EIGEN_DEVICE_FUNC - EIGEN_STRONG_INLINE const Scalar& operator()(Index index) const - { - eigen_internal_assert(index >= 0 && index < size()); - return coeff(index); - } - - EIGEN_DEVICE_FUNC - EIGEN_STRONG_INLINE const Scalar& operator()() const - { - EIGEN_STATIC_ASSERT(NumIndices == 0, YOU_MADE_A_PROGRAMMING_MISTAKE); - return coeff(); - } - - EIGEN_DEVICE_FUNC - EIGEN_STRONG_INLINE const Scalar& operator[](Index index) const - { - // The bracket operator is only for vectors, use the parenthesis operator instead. - EIGEN_STATIC_ASSERT(NumIndices == 1, YOU_MADE_A_PROGRAMMING_MISTAKE); - return coeff(index); - } + EIGEN_DEVICE_FUNC + EIGEN_STRONG_INLINE const Scalar &operator()(const array &indices) const + { + eigen_assert(checkIndexRange(indices)); + return coeff(indices); + } + + EIGEN_DEVICE_FUNC + EIGEN_STRONG_INLINE const Scalar &operator()(Index index) const + { + eigen_internal_assert(index >= 0 && index < size()); + return coeff(index); + } + + EIGEN_DEVICE_FUNC + EIGEN_STRONG_INLINE const Scalar &operator()() const + { + EIGEN_STATIC_ASSERT(NumIndices == 0, YOU_MADE_A_PROGRAMMING_MISTAKE); + return coeff(); + } + + EIGEN_DEVICE_FUNC + EIGEN_STRONG_INLINE const Scalar &operator[](Index index) const + { + // The bracket operator is only for vectors, use the parenthesis operator instead. + EIGEN_STATIC_ASSERT(NumIndices == 1, YOU_MADE_A_PROGRAMMING_MISTAKE); + return coeff(index); + } #if EIGEN_HAS_VARIADIC_TEMPLATES - template - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Scalar& operator()(Index firstIndex, IndexTypes... otherIndices) - { - // The number of indices used to access a tensor coefficient must be equal to the rank of the tensor. - EIGEN_STATIC_ASSERT(sizeof...(otherIndices) + 1 == NumIndices, YOU_MADE_A_PROGRAMMING_MISTAKE) - return operator()(array{{firstIndex, otherIndices...}}); - } + template + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Scalar &operator()(Index firstIndex, IndexTypes... otherIndices) + { + // The number of indices used to access a tensor coefficient must be equal to the rank of the tensor. + EIGEN_STATIC_ASSERT(sizeof...(otherIndices) + 1 == NumIndices, YOU_MADE_A_PROGRAMMING_MISTAKE) + return operator()(array{ { firstIndex, otherIndices... } }); + } #else - EIGEN_DEVICE_FUNC - EIGEN_STRONG_INLINE Scalar& operator()(Index i0, Index i1) - { - if (Options&RowMajor) { - const Index index = i1 + i0 * m_storage.dimensions()[1]; - return m_storage.data()[index]; - } else { - const Index index = i0 + i1 * m_storage.dimensions()[0]; - return m_storage.data()[index]; - } + EIGEN_DEVICE_FUNC + EIGEN_STRONG_INLINE Scalar &operator()(Index i0, Index i1) + { + if (Options & RowMajor) { + const Index index = i1 + i0 * m_storage.dimensions()[1]; + return m_storage.data()[index]; + } else { + const Index index = i0 + i1 * m_storage.dimensions()[0]; + return m_storage.data()[index]; } - EIGEN_DEVICE_FUNC - EIGEN_STRONG_INLINE Scalar& operator()(Index i0, Index i1, Index i2) - { - if (Options&RowMajor) { - const Index index = i2 + m_storage.dimensions()[2] * (i1 + m_storage.dimensions()[1] * i0); - return m_storage.data()[index]; - } else { - const Index index = i0 + m_storage.dimensions()[0] * (i1 + m_storage.dimensions()[1] * i2); - return m_storage.data()[index]; - } + } + EIGEN_DEVICE_FUNC + EIGEN_STRONG_INLINE Scalar &operator()(Index i0, Index i1, Index i2) + { + if (Options & RowMajor) { + const Index index = i2 + m_storage.dimensions()[2] * (i1 + m_storage.dimensions()[1] * i0); + return m_storage.data()[index]; + } else { + const Index index = i0 + m_storage.dimensions()[0] * (i1 + m_storage.dimensions()[1] * i2); + return m_storage.data()[index]; } - EIGEN_DEVICE_FUNC - EIGEN_STRONG_INLINE Scalar& operator()(Index i0, Index i1, Index i2, Index i3) - { - if (Options&RowMajor) { - const Index index = i3 + m_storage.dimensions()[3] * (i2 + m_storage.dimensions()[2] * (i1 + m_storage.dimensions()[1] * i0)); - return m_storage.data()[index]; - } else { - const Index index = i0 + m_storage.dimensions()[0] * (i1 + m_storage.dimensions()[1] * (i2 + m_storage.dimensions()[2] * i3)); - return m_storage.data()[index]; - } + } + EIGEN_DEVICE_FUNC + EIGEN_STRONG_INLINE Scalar &operator()(Index i0, Index i1, Index i2, Index i3) + { + if (Options & RowMajor) { + const Index index = + i3 + m_storage.dimensions()[3] * (i2 + m_storage.dimensions()[2] * (i1 + m_storage.dimensions()[1] * i0)); + return m_storage.data()[index]; + } else { + const Index index = + i0 + m_storage.dimensions()[0] * (i1 + m_storage.dimensions()[1] * (i2 + m_storage.dimensions()[2] * i3)); + return m_storage.data()[index]; } - EIGEN_DEVICE_FUNC - EIGEN_STRONG_INLINE Scalar& operator()(Index i0, Index i1, Index i2, Index i3, Index i4) - { - if (Options&RowMajor) { - const Index index = i4 + m_storage.dimensions()[4] * (i3 + m_storage.dimensions()[3] * (i2 + m_storage.dimensions()[2] * (i1 + m_storage.dimensions()[1] * i0))); - return m_storage.data()[index]; - } else { - const Index index = i0 + m_storage.dimensions()[0] * (i1 + m_storage.dimensions()[1] * (i2 + m_storage.dimensions()[2] * (i3 + m_storage.dimensions()[3] * i4))); - return m_storage.data()[index]; - } + } + EIGEN_DEVICE_FUNC + EIGEN_STRONG_INLINE Scalar &operator()(Index i0, Index i1, Index i2, Index i3, Index i4) + { + if (Options & RowMajor) { + const Index index = + i4 + + m_storage.dimensions()[4] + * (i3 + + m_storage.dimensions()[3] * (i2 + m_storage.dimensions()[2] * (i1 + m_storage.dimensions()[1] * i0))); + return m_storage.data()[index]; + } else { + const Index index = + i0 + + m_storage.dimensions()[0] + * (i1 + + m_storage.dimensions()[1] * (i2 + m_storage.dimensions()[2] * (i3 + m_storage.dimensions()[3] * i4))); + return m_storage.data()[index]; } + } #endif - EIGEN_DEVICE_FUNC - EIGEN_STRONG_INLINE Scalar& operator()(const array& indices) - { - eigen_assert(checkIndexRange(indices)); - return coeffRef(indices); - } - - EIGEN_DEVICE_FUNC - EIGEN_STRONG_INLINE Scalar& operator()(Index index) - { - eigen_assert(index >= 0 && index < size()); - return coeffRef(index); - } - - EIGEN_DEVICE_FUNC - EIGEN_STRONG_INLINE Scalar& operator()() - { - EIGEN_STATIC_ASSERT(NumIndices == 0, YOU_MADE_A_PROGRAMMING_MISTAKE); - return coeffRef(); - } - - EIGEN_DEVICE_FUNC - EIGEN_STRONG_INLINE Scalar& operator[](Index index) - { - // The bracket operator is only for vectors, use the parenthesis operator instead - EIGEN_STATIC_ASSERT(NumIndices == 1, YOU_MADE_A_PROGRAMMING_MISTAKE) - return coeffRef(index); - } - - EIGEN_DEVICE_FUNC - EIGEN_STRONG_INLINE TensorFixedSize() - : m_storage() - { - } - - EIGEN_DEVICE_FUNC - EIGEN_STRONG_INLINE TensorFixedSize(const Self& other) - : m_storage(other.m_storage) - { - } + EIGEN_DEVICE_FUNC + EIGEN_STRONG_INLINE Scalar &operator()(const array &indices) + { + eigen_assert(checkIndexRange(indices)); + return coeffRef(indices); + } + + EIGEN_DEVICE_FUNC + EIGEN_STRONG_INLINE Scalar &operator()(Index index) + { + eigen_assert(index >= 0 && index < size()); + return coeffRef(index); + } + + EIGEN_DEVICE_FUNC + EIGEN_STRONG_INLINE Scalar &operator()() + { + EIGEN_STATIC_ASSERT(NumIndices == 0, YOU_MADE_A_PROGRAMMING_MISTAKE); + return coeffRef(); + } + + EIGEN_DEVICE_FUNC + EIGEN_STRONG_INLINE Scalar &operator[](Index index) + { + // The bracket operator is only for vectors, use the parenthesis operator instead + EIGEN_STATIC_ASSERT(NumIndices == 1, YOU_MADE_A_PROGRAMMING_MISTAKE) + return coeffRef(index); + } + + EIGEN_DEVICE_FUNC + EIGEN_STRONG_INLINE TensorFixedSize() : m_storage() {} + + EIGEN_DEVICE_FUNC + EIGEN_STRONG_INLINE TensorFixedSize(const Self &other) : m_storage(other.m_storage) {} #if EIGEN_HAS_RVALUE_REFERENCES - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TensorFixedSize(Self&& other) - : m_storage(other.m_storage) - { - } + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TensorFixedSize(Self &&other) : m_storage(other.m_storage) {} #endif - template - EIGEN_DEVICE_FUNC - EIGEN_STRONG_INLINE TensorFixedSize(const TensorBase& other) - { - typedef TensorAssignOp Assign; - Assign assign(*this, other.derived()); - internal::TensorExecutor::run(assign, DefaultDevice()); - } - template - EIGEN_DEVICE_FUNC - EIGEN_STRONG_INLINE TensorFixedSize(const TensorBase& other) - { - typedef TensorAssignOp Assign; - Assign assign(*this, other.derived()); - internal::TensorExecutor::run(assign, DefaultDevice()); - } - - EIGEN_DEVICE_FUNC - EIGEN_STRONG_INLINE TensorFixedSize& operator=(const TensorFixedSize& other) - { - // FIXME: check that the dimensions of other match the dimensions of *this. - // Unfortunately this isn't possible yet when the rhs is an expression. - typedef TensorAssignOp Assign; - Assign assign(*this, other); - internal::TensorExecutor::run(assign, DefaultDevice()); - return *this; - } - template - EIGEN_DEVICE_FUNC - EIGEN_STRONG_INLINE TensorFixedSize& operator=(const OtherDerived& other) - { - // FIXME: check that the dimensions of other match the dimensions of *this. - // Unfortunately this isn't possible yet when the rhs is an expression. - typedef TensorAssignOp Assign; - Assign assign(*this, other); - internal::TensorExecutor::run(assign, DefaultDevice()); - return *this; - } - - protected: - EIGEN_DEVICE_FUNC - EIGEN_STRONG_INLINE bool checkIndexRange(const array& /*indices*/) const - { - using internal::array_apply_and_reduce; - using internal::array_zip_and_reduce; - using internal::greater_equal_zero_op; - using internal::logical_and_op; - using internal::lesser_op; - - return true; - // check whether the indices are all >= 0 - /* array_apply_and_reduce(indices) && - // check whether the indices fit in the dimensions - array_zip_and_reduce(indices, m_storage.dimensions());*/ - } - - EIGEN_DEVICE_FUNC - EIGEN_STRONG_INLINE Index linearizedIndex(const array& indices) const - { - if (Options&RowMajor) { - return m_storage.dimensions().IndexOfRowMajor(indices); - } else { - return m_storage.dimensions().IndexOfColMajor(indices); - } - } + template + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TensorFixedSize(const TensorBase &other) + { + typedef TensorAssignOp Assign; + Assign assign(*this, other.derived()); + internal::TensorExecutor::run(assign, DefaultDevice()); + } + template + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TensorFixedSize(const TensorBase &other) + { + typedef TensorAssignOp Assign; + Assign assign(*this, other.derived()); + internal::TensorExecutor::run(assign, DefaultDevice()); + } + + EIGEN_DEVICE_FUNC + EIGEN_STRONG_INLINE TensorFixedSize &operator=(const TensorFixedSize &other) + { + // FIXME: check that the dimensions of other match the dimensions of *this. + // Unfortunately this isn't possible yet when the rhs is an expression. + typedef TensorAssignOp Assign; + Assign assign(*this, other); + internal::TensorExecutor::run(assign, DefaultDevice()); + return *this; + } + template + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TensorFixedSize &operator=(const OtherDerived &other) + { + // FIXME: check that the dimensions of other match the dimensions of *this. + // Unfortunately this isn't possible yet when the rhs is an expression. + typedef TensorAssignOp Assign; + Assign assign(*this, other); + internal::TensorExecutor::run(assign, DefaultDevice()); + return *this; + } + +protected: + EIGEN_DEVICE_FUNC + EIGEN_STRONG_INLINE bool checkIndexRange(const array & /*indices*/) const + { + using internal::array_apply_and_reduce; + using internal::array_zip_and_reduce; + using internal::greater_equal_zero_op; + using internal::logical_and_op; + using internal::lesser_op; + + return true; + // check whether the indices are all >= 0 + /* array_apply_and_reduce(indices) && + // check whether the indices fit in the dimensions + array_zip_and_reduce(indices, m_storage.dimensions());*/ + } + + EIGEN_DEVICE_FUNC + EIGEN_STRONG_INLINE Index linearizedIndex(const array &indices) const + { + if (Options & RowMajor) { + return m_storage.dimensions().IndexOfRowMajor(indices); + } else { + return m_storage.dimensions().IndexOfColMajor(indices); + } + } }; -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_CXX11_TENSOR_TENSOR_FIXED_SIZE_H +#endif// EIGEN_CXX11_TENSOR_TENSOR_FIXED_SIZE_H diff --git a/filmulator-gui/core/nlmeans/eigen/unsupported/Eigen/CXX11/src/Tensor/TensorForcedEval.h b/filmulator-gui/core/nlmeans/eigen/unsupported/Eigen/CXX11/src/Tensor/TensorForcedEval.h index 8bece4e6..6922ca4e 100644 --- a/filmulator-gui/core/nlmeans/eigen/unsupported/Eigen/CXX11/src/Tensor/TensorForcedEval.h +++ b/filmulator-gui/core/nlmeans/eigen/unsupported/Eigen/CXX11/src/Tensor/TensorForcedEval.h @@ -13,64 +13,64 @@ namespace Eigen { namespace internal { -template class MakePointer_> -struct traits > -{ - // Type promotion to handle the case where the types of the lhs and the rhs are different. - typedef typename XprType::Scalar Scalar; - typedef traits XprTraits; - typedef typename traits::StorageKind StorageKind; - typedef typename traits::Index Index; - typedef typename XprType::Nested Nested; - typedef typename remove_reference::type _Nested; - static const int NumDimensions = XprTraits::NumDimensions; - static const int Layout = XprTraits::Layout; - - enum { - Flags = 0 - }; - template struct MakePointer { - // Intermediate typedef to workaround MSVC issue. - typedef MakePointer_ MakePointerT; - typedef typename MakePointerT::Type Type; + template class MakePointer_> + struct traits> + { + // Type promotion to handle the case where the types of the lhs and the rhs are different. + typedef typename XprType::Scalar Scalar; + typedef traits XprTraits; + typedef typename traits::StorageKind StorageKind; + typedef typename traits::Index Index; + typedef typename XprType::Nested Nested; + typedef typename remove_reference::type _Nested; + static const int NumDimensions = XprTraits::NumDimensions; + static const int Layout = XprTraits::Layout; + + enum { Flags = 0 }; + template struct MakePointer + { + // Intermediate typedef to workaround MSVC issue. + typedef MakePointer_ MakePointerT; + typedef typename MakePointerT::Type Type; + }; }; -}; -template class MakePointer_> -struct eval, Eigen::Dense> -{ - typedef const TensorForcedEvalOp& type; -}; - -template class MakePointer_> -struct nested, 1, typename eval >::type> -{ - typedef TensorForcedEvalOp type; -}; + template class MakePointer_> + struct eval, Eigen::Dense> + { + typedef const TensorForcedEvalOp &type; + }; -} // end namespace internal + template class MakePointer_> + struct nested, + 1, + typename eval>::type> + { + typedef TensorForcedEvalOp type; + }; +}// end namespace internal // FIXME use proper doxygen documentation (e.g. \tparam MakePointer_) /** \class TensorForcedEvalOp - * \ingroup CXX11_Tensor_Module - * - * \brief Tensor reshaping class. - * - * - */ + * \ingroup CXX11_Tensor_Module + * + * \brief Tensor reshaping class. + * + * + */ /// `template class MakePointer_` is added to convert the host pointer to the device pointer. /// It is added due to the fact that for our device compiler `T*` is not allowed. /// If we wanted to use the same Evaluator functions we have to convert that type to our pointer `T`. /// This is done through our `MakePointer_` class. By default the Type in the `MakePointer_` is `T*` . /// Therefore, by adding the default value, we managed to convert the type and it does not break any /// existing code as its default value is `T*`. -template class MakePointer_> +template class MakePointer_> class TensorForcedEvalOp : public TensorBase, ReadOnlyAccessors> { - public: +public: typedef typename Eigen::internal::traits::Scalar Scalar; typedef typename Eigen::NumTraits::Real RealScalar; typedef typename internal::remove_const::type CoeffReturnType; @@ -78,19 +78,17 @@ class TensorForcedEvalOp : public TensorBase::StorageKind StorageKind; typedef typename Eigen::internal::traits::Index Index; - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TensorForcedEvalOp(const XprType& expr) - : m_xpr(expr) {} + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TensorForcedEvalOp(const XprType &expr) : m_xpr(expr) {} - EIGEN_DEVICE_FUNC - const typename internal::remove_all::type& - expression() const { return m_xpr; } + EIGEN_DEVICE_FUNC + const typename internal::remove_all::type &expression() const { return m_xpr; } - protected: - typename XprType::Nested m_xpr; +protected: + typename XprType::Nested m_xpr; }; -template class MakePointer_> +template class MakePointer_> struct TensorEvaluator, Device> { typedef TensorForcedEvalOp XprType; @@ -108,62 +106,61 @@ struct TensorEvaluator, Device> RawAccess = true }; - EIGEN_DEVICE_FUNC TensorEvaluator(const XprType& op, const Device& device) - /// op_ is used for sycl - : m_impl(op.expression(), device), m_op(op.expression()), m_device(device), m_buffer(NULL) - { } + EIGEN_DEVICE_FUNC TensorEvaluator(const XprType &op, const Device &device) + /// op_ is used for sycl + : m_impl(op.expression(), device), m_op(op.expression()), m_device(device), m_buffer(NULL) + {} - EIGEN_DEVICE_FUNC const Dimensions& dimensions() const { return m_impl.dimensions(); } + EIGEN_DEVICE_FUNC const Dimensions &dimensions() const { return m_impl.dimensions(); } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE bool evalSubExprsIfNeeded(CoeffReturnType*) { - const Index numValues = internal::array_prod(m_impl.dimensions()); - m_buffer = (CoeffReturnType*)m_device.allocate(numValues * sizeof(CoeffReturnType)); + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE bool evalSubExprsIfNeeded(CoeffReturnType *) + { + const Index numValues = internal::array_prod(m_impl.dimensions()); + m_buffer = (CoeffReturnType *)m_device.allocate(numValues * sizeof(CoeffReturnType)); // Should initialize the memory in case we're dealing with non POD types. if (NumTraits::RequireInitialization) { - for (Index i = 0; i < numValues; ++i) { - new(m_buffer+i) CoeffReturnType(); - } + for (Index i = 0; i < numValues; ++i) { new (m_buffer + i) CoeffReturnType(); } } - typedef TensorEvalToOp< const typename internal::remove_const::type > EvalTo; + typedef TensorEvalToOp::type> EvalTo; EvalTo evalToTmp(m_buffer, m_op); const bool PacketAccess = internal::IsVectorizable::value; - internal::TensorExecutor::type, PacketAccess>::run(evalToTmp, m_device); + internal::TensorExecutor::type, PacketAccess>::run( + evalToTmp, m_device); return true; } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void cleanup() { + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void cleanup() + { m_device.deallocate(m_buffer); m_buffer = NULL; } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE CoeffReturnType coeff(Index index) const - { - return m_buffer[index]; - } + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE CoeffReturnType coeff(Index index) const { return m_buffer[index]; } - template - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE PacketReturnType packet(Index index) const + template EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE PacketReturnType packet(Index index) const { return internal::ploadt(m_buffer + index); } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TensorOpCost costPerCoeff(bool vectorized) const { + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TensorOpCost costPerCoeff(bool vectorized) const + { return TensorOpCost(sizeof(CoeffReturnType), 0, 0, vectorized, PacketSize); } EIGEN_DEVICE_FUNC typename MakePointer::Type data() const { return m_buffer; } /// required by sycl in order to extract the sycl accessor - const TensorEvaluator& impl() { return m_impl; } + const TensorEvaluator &impl() { return m_impl; } /// used by sycl in order to build the sycl buffer - const Device& device() const{return m_device;} - private: + const Device &device() const { return m_device; } + +private: TensorEvaluator m_impl; const ArgType m_op; - const Device& m_device; + const Device &m_device; typename MakePointer::Type m_buffer; }; -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_CXX11_TENSOR_TENSOR_FORCED_EVAL_H +#endif// EIGEN_CXX11_TENSOR_TENSOR_FORCED_EVAL_H diff --git a/filmulator-gui/core/nlmeans/eigen/unsupported/Eigen/CXX11/src/Tensor/TensorForwardDeclarations.h b/filmulator-gui/core/nlmeans/eigen/unsupported/Eigen/CXX11/src/Tensor/TensorForwardDeclarations.h index 52b803d7..61391f4a 100644 --- a/filmulator-gui/core/nlmeans/eigen/unsupported/Eigen/CXX11/src/Tensor/TensorForwardDeclarations.h +++ b/filmulator-gui/core/nlmeans/eigen/unsupported/Eigen/CXX11/src/Tensor/TensorForwardDeclarations.h @@ -18,22 +18,27 @@ namespace Eigen { // T* m_data on the host. It is always called on the device. // Specialisation of MakePointer class for creating the sycl buffer with // map_allocator. -template struct MakePointer { - typedef T* Type; +template struct MakePointer +{ + typedef T *Type; }; -template class MakePointer_ = MakePointer> class TensorMap; +template class MakePointer_ = MakePointer> +class TensorMap; template class Tensor; -template class TensorFixedSize; +template +class TensorFixedSize; template class TensorRef; template class TensorBase; template class TensorCwiseNullaryOp; template class TensorCwiseUnaryOp; template class TensorCwiseBinaryOp; -template class TensorCwiseTernaryOp; +template +class TensorCwiseTernaryOp; template class TensorSelectOp; -template class MakePointer_ = MakePointer > class TensorReductionOp; +template class MakePointer_ = MakePointer> +class TensorReductionOp; template class TensorIndexTupleOp; template class TensorTupleReducerOp; template class TensorConcatenationOp; @@ -62,8 +67,8 @@ template class TensorScanOp; template class TensorCustomUnaryOp; template class TensorCustomBinaryOp; -template class MakePointer_ = MakePointer> class TensorEvalToOp; -template class MakePointer_ = MakePointer> class TensorForcedEvalOp; +template class MakePointer_ = MakePointer> class TensorEvalToOp; +template class MakePointer_ = MakePointer> class TensorForcedEvalOp; template class TensorDevice; template struct TensorEvaluator; @@ -73,37 +78,29 @@ struct ThreadPoolDevice; struct GpuDevice; struct SyclDevice; -enum FFTResultType { - RealPart = 0, - ImagPart = 1, - BothParts = 2 -}; +enum FFTResultType { RealPart = 0, ImagPart = 1, BothParts = 2 }; -enum FFTDirection { - FFT_FORWARD = 0, - FFT_REVERSE = 1 -}; +enum FFTDirection { FFT_FORWARD = 0, FFT_REVERSE = 1 }; namespace internal { -template -struct IsVectorizable { - static const bool value = TensorEvaluator::PacketAccess; -}; + template struct IsVectorizable + { + static const bool value = TensorEvaluator::PacketAccess; + }; -template -struct IsVectorizable { - static const bool value = TensorEvaluator::PacketAccess && - TensorEvaluator::IsAligned; -}; + template struct IsVectorizable + { + static const bool value = + TensorEvaluator::PacketAccess && TensorEvaluator::IsAligned; + }; -template ::value> -class TensorExecutor; + template::value> + class TensorExecutor; -} // end namespace internal +}// end namespace internal -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_CXX11_TENSOR_TENSOR_FORWARD_DECLARATIONS_H +#endif// EIGEN_CXX11_TENSOR_TENSOR_FORWARD_DECLARATIONS_H diff --git a/filmulator-gui/core/nlmeans/eigen/unsupported/Eigen/CXX11/src/Tensor/TensorFunctors.h b/filmulator-gui/core/nlmeans/eigen/unsupported/Eigen/CXX11/src/Tensor/TensorFunctors.h index d73f6dc6..0bfa81a2 100644 --- a/filmulator-gui/core/nlmeans/eigen/unsupported/Eigen/CXX11/src/Tensor/TensorFunctors.h +++ b/filmulator-gui/core/nlmeans/eigen/unsupported/Eigen/CXX11/src/Tensor/TensorFunctors.h @@ -14,476 +14,438 @@ namespace Eigen { namespace internal { -/** \internal - * \brief Template functor to compute the modulo between an array and a scalar. - */ -template -struct scalar_mod_op { - EIGEN_DEVICE_FUNC scalar_mod_op(const Scalar& divisor) : m_divisor(divisor) {} - EIGEN_DEVICE_FUNC inline Scalar operator() (const Scalar& a) const { return a % m_divisor; } - const Scalar m_divisor; -}; -template -struct functor_traits > -{ enum { Cost = scalar_div_cost::value, PacketAccess = false }; }; - - -/** \internal - * \brief Template functor to compute the modulo between 2 arrays. - */ -template -struct scalar_mod2_op { - EIGEN_EMPTY_STRUCT_CTOR(scalar_mod2_op); - EIGEN_DEVICE_FUNC inline Scalar operator() (const Scalar& a, const Scalar& b) const { return a % b; } -}; -template -struct functor_traits > -{ enum { Cost = scalar_div_cost::value, PacketAccess = false }; }; - -template -struct scalar_fmod_op { - EIGEN_EMPTY_STRUCT_CTOR(scalar_fmod_op); - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Scalar - operator()(const Scalar& a, const Scalar& b) const { - return numext::fmod(a, b); - } -}; -template -struct functor_traits > { - enum { Cost = 13, // Reciprocal throughput of FPREM on Haswell. - PacketAccess = false }; -}; - - -/** \internal - * \brief Template functor to compute the sigmoid of a scalar - * \sa class CwiseUnaryOp, ArrayBase::sigmoid() - */ -template -struct scalar_sigmoid_op { - EIGEN_EMPTY_STRUCT_CTOR(scalar_sigmoid_op) - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE T operator()(const T& x) const { - const T one = T(1); - return one / (one + numext::exp(-x)); - } - - template EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE - Packet packetOp(const Packet& x) const { - const Packet one = pset1(T(1)); - return pdiv(one, padd(one, pexp(pnegate(x)))); - } -}; - -template -struct functor_traits > { - enum { - Cost = NumTraits::AddCost * 2 + NumTraits::MulCost * 6, - PacketAccess = packet_traits::HasAdd && packet_traits::HasDiv && - packet_traits::HasNegate && packet_traits::HasExp - }; -}; - - -template -struct reducer_traits { - enum { - Cost = 1, - PacketAccess = false - }; -}; - -// Standard reduction functors -template struct SumReducer -{ - static const bool PacketAccess = packet_traits::HasAdd; - static const bool IsStateful = false; - - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void reduce(const T t, T* accum) const { - internal::scalar_sum_op sum_op; - *accum = sum_op(*accum, t); - } - template - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void reducePacket(const Packet& p, Packet* accum) const { - (*accum) = padd(*accum, p); - } - - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE T initialize() const { - internal::scalar_cast_op conv; - return conv(0); - } - template - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Packet initializePacket() const { - return pset1(initialize()); - } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE T finalize(const T accum) const { - return accum; - } - template - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Packet finalizePacket(const Packet& vaccum) const { - return vaccum; - } - template - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE T finalizeBoth(const T saccum, const Packet& vaccum) const { - internal::scalar_sum_op sum_op; - return sum_op(saccum, predux(vaccum)); - } -}; - -template -struct reducer_traits, Device> { - enum { - Cost = NumTraits::AddCost, - PacketAccess = PacketType::HasAdd - }; -}; - - -template struct MeanReducer -{ - static const bool PacketAccess = packet_traits::HasAdd && !NumTraits::IsInteger; - static const bool IsStateful = true; - - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE - MeanReducer() : scalarCount_(0), packetCount_(0) { } - - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void reduce(const T t, T* accum) { - internal::scalar_sum_op sum_op; - *accum = sum_op(*accum, t); - scalarCount_++; - } - template - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void reducePacket(const Packet& p, Packet* accum) { - (*accum) = padd(*accum, p); - packetCount_++; - } - - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE T initialize() const { - internal::scalar_cast_op conv; - return conv(0); - } - template - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Packet initializePacket() const { - return pset1(initialize()); - } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE T finalize(const T accum) const { - return accum / scalarCount_; - } - template - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Packet finalizePacket(const Packet& vaccum) const { - return pdiv(vaccum, pset1(packetCount_)); - } - template - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE T finalizeBoth(const T saccum, const Packet& vaccum) const { - internal::scalar_sum_op sum_op; - return sum_op(saccum, predux(vaccum)) / (scalarCount_ + packetCount_ * unpacket_traits::size); - } + /** \internal + * \brief Template functor to compute the modulo between an array and a scalar. + */ + template struct scalar_mod_op + { + EIGEN_DEVICE_FUNC scalar_mod_op(const Scalar &divisor) : m_divisor(divisor) {} + EIGEN_DEVICE_FUNC inline Scalar operator()(const Scalar &a) const { return a % m_divisor; } + const Scalar m_divisor; + }; + template struct functor_traits> + { + enum { Cost = scalar_div_cost::value, PacketAccess = false }; + }; + + + /** \internal + * \brief Template functor to compute the modulo between 2 arrays. + */ + template struct scalar_mod2_op + { + EIGEN_EMPTY_STRUCT_CTOR(scalar_mod2_op); + EIGEN_DEVICE_FUNC inline Scalar operator()(const Scalar &a, const Scalar &b) const { return a % b; } + }; + template struct functor_traits> + { + enum { Cost = scalar_div_cost::value, PacketAccess = false }; + }; + + template struct scalar_fmod_op + { + EIGEN_EMPTY_STRUCT_CTOR(scalar_fmod_op); + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Scalar operator()(const Scalar &a, const Scalar &b) const + { + return numext::fmod(a, b); + } + }; + template struct functor_traits> + { + enum { + Cost = 13,// Reciprocal throughput of FPREM on Haswell. + PacketAccess = false + }; + }; + + + /** \internal + * \brief Template functor to compute the sigmoid of a scalar + * \sa class CwiseUnaryOp, ArrayBase::sigmoid() + */ + template struct scalar_sigmoid_op + { + EIGEN_EMPTY_STRUCT_CTOR(scalar_sigmoid_op) + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE T operator()(const T &x) const + { + const T one = T(1); + return one / (one + numext::exp(-x)); + } + + template EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Packet packetOp(const Packet &x) const + { + const Packet one = pset1(T(1)); + return pdiv(one, padd(one, pexp(pnegate(x)))); + } + }; + + template struct functor_traits> + { + enum { + Cost = NumTraits::AddCost * 2 + NumTraits::MulCost * 6, + PacketAccess = + packet_traits::HasAdd && packet_traits::HasDiv && packet_traits::HasNegate && packet_traits::HasExp + }; + }; + + + template struct reducer_traits + { + enum { Cost = 1, PacketAccess = false }; + }; + + // Standard reduction functors + template struct SumReducer + { + static const bool PacketAccess = packet_traits::HasAdd; + static const bool IsStateful = false; + + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void reduce(const T t, T *accum) const + { + internal::scalar_sum_op sum_op; + *accum = sum_op(*accum, t); + } + template + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void reducePacket(const Packet &p, Packet *accum) const + { + (*accum) = padd(*accum, p); + } + + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE T initialize() const + { + internal::scalar_cast_op conv; + return conv(0); + } + template EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Packet initializePacket() const + { + return pset1(initialize()); + } + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE T finalize(const T accum) const { return accum; } + template EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Packet finalizePacket(const Packet &vaccum) const + { + return vaccum; + } + template + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE T finalizeBoth(const T saccum, const Packet &vaccum) const + { + internal::scalar_sum_op sum_op; + return sum_op(saccum, predux(vaccum)); + } + }; + + template struct reducer_traits, Device> + { + enum { Cost = NumTraits::AddCost, PacketAccess = PacketType::HasAdd }; + }; + + + template struct MeanReducer + { + static const bool PacketAccess = packet_traits::HasAdd && !NumTraits::IsInteger; + static const bool IsStateful = true; + + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE MeanReducer() : scalarCount_(0), packetCount_(0) {} + + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void reduce(const T t, T *accum) + { + internal::scalar_sum_op sum_op; + *accum = sum_op(*accum, t); + scalarCount_++; + } + template EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void reducePacket(const Packet &p, Packet *accum) + { + (*accum) = padd(*accum, p); + packetCount_++; + } + + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE T initialize() const + { + internal::scalar_cast_op conv; + return conv(0); + } + template EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Packet initializePacket() const + { + return pset1(initialize()); + } + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE T finalize(const T accum) const { return accum / scalarCount_; } + template EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Packet finalizePacket(const Packet &vaccum) const + { + return pdiv(vaccum, pset1(packetCount_)); + } + template + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE T finalizeBoth(const T saccum, const Packet &vaccum) const + { + internal::scalar_sum_op sum_op; + return sum_op(saccum, predux(vaccum)) / (scalarCount_ + packetCount_ * unpacket_traits::size); + } protected: DenseIndex scalarCount_; DenseIndex packetCount_; -}; - -template -struct reducer_traits, Device> { - enum { - Cost = NumTraits::AddCost, - PacketAccess = PacketType::HasAdd - }; -}; - - -template -struct MinMaxBottomValue { - EIGEN_DEVICE_FUNC static EIGEN_STRONG_INLINE T bottom_value() { - return Eigen::NumTraits::lowest(); - } -}; -template -struct MinMaxBottomValue { - EIGEN_DEVICE_FUNC static EIGEN_STRONG_INLINE T bottom_value() { - return -Eigen::NumTraits::infinity(); - } -}; -template -struct MinMaxBottomValue { - EIGEN_DEVICE_FUNC static EIGEN_STRONG_INLINE T bottom_value() { - return Eigen::NumTraits::highest(); - } -}; -template -struct MinMaxBottomValue { - EIGEN_DEVICE_FUNC static EIGEN_STRONG_INLINE T bottom_value() { - return Eigen::NumTraits::infinity(); - } -}; - - -template struct MaxReducer -{ - static const bool PacketAccess = packet_traits::HasMax; - static const bool IsStateful = false; - - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void reduce(const T t, T* accum) const { - if (t > *accum) { *accum = t; } - } - template - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void reducePacket(const Packet& p, Packet* accum) const { - (*accum) = pmax(*accum, p); - } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE T initialize() const { - return MinMaxBottomValue::IsInteger>::bottom_value(); - } - template - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Packet initializePacket() const { - return pset1(initialize()); - } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE T finalize(const T accum) const { - return accum; - } - template - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Packet finalizePacket(const Packet& vaccum) const { - return vaccum; - } - template - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE T finalizeBoth(const T saccum, const Packet& vaccum) const { - return numext::maxi(saccum, predux_max(vaccum)); - } -}; - -template -struct reducer_traits, Device> { - enum { - Cost = NumTraits::AddCost, - PacketAccess = PacketType::HasMax - }; -}; - - -template struct MinReducer -{ - static const bool PacketAccess = packet_traits::HasMin; - static const bool IsStateful = false; - - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void reduce(const T t, T* accum) const { - if (t < *accum) { *accum = t; } - } - template - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void reducePacket(const Packet& p, Packet* accum) const { - (*accum) = pmin(*accum, p); - } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE T initialize() const { - return MinMaxBottomValue::IsInteger>::bottom_value(); - } - template - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Packet initializePacket() const { - return pset1(initialize()); - } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE T finalize(const T accum) const { - return accum; - } - template - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Packet finalizePacket(const Packet& vaccum) const { - return vaccum; - } - template - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE T finalizeBoth(const T saccum, const Packet& vaccum) const { - return numext::mini(saccum, predux_min(vaccum)); - } -}; - -template -struct reducer_traits, Device> { - enum { - Cost = NumTraits::AddCost, - PacketAccess = PacketType::HasMin - }; -}; - - -template struct ProdReducer -{ - static const bool PacketAccess = packet_traits::HasMul; - static const bool IsStateful = false; - - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void reduce(const T t, T* accum) const { - internal::scalar_product_op prod_op; - (*accum) = prod_op(*accum, t); - } - template - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void reducePacket(const Packet& p, Packet* accum) const { - (*accum) = pmul(*accum, p); - } - - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE T initialize() const { - internal::scalar_cast_op conv; - return conv(1); - } - template - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Packet initializePacket() const { - return pset1(initialize()); - } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE T finalize(const T accum) const { - return accum; - } - template - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Packet finalizePacket(const Packet& vaccum) const { - return vaccum; - } - template - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE T finalizeBoth(const T saccum, const Packet& vaccum) const { - internal::scalar_product_op prod_op; - return prod_op(saccum, predux_mul(vaccum)); - } -}; - -template -struct reducer_traits, Device> { - enum { - Cost = NumTraits::MulCost, - PacketAccess = PacketType::HasMul - }; -}; - - -struct AndReducer -{ - static const bool PacketAccess = false; - static const bool IsStateful = false; - - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void reduce(bool t, bool* accum) const { - *accum = *accum && t; - } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE bool initialize() const { - return true; - } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE bool finalize(bool accum) const { - return accum; - } -}; - -template -struct reducer_traits { - enum { - Cost = 1, - PacketAccess = false - }; -}; - - -struct OrReducer { - static const bool PacketAccess = false; - static const bool IsStateful = false; - - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void reduce(bool t, bool* accum) const { - *accum = *accum || t; - } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE bool initialize() const { - return false; - } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE bool finalize(bool accum) const { - return accum; - } -}; - -template -struct reducer_traits { - enum { - Cost = 1, - PacketAccess = false - }; -}; - - -// Argmin/Argmax reducers -template struct ArgMaxTupleReducer -{ - static const bool PacketAccess = false; - static const bool IsStateful = false; - - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void reduce(const T t, T* accum) const { - if (t.second > accum->second) { *accum = t; } - } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE T initialize() const { - return T(0, NumTraits::lowest()); - } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE T finalize(const T& accum) const { - return accum; - } -}; - -template -struct reducer_traits, Device> { - enum { - Cost = NumTraits::AddCost, - PacketAccess = false - }; -}; - - -template struct ArgMinTupleReducer -{ - static const bool PacketAccess = false; - static const bool IsStateful = false; - - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void reduce(const T& t, T* accum) const { - if (t.second < accum->second) { *accum = t; } - } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE T initialize() const { - return T(0, NumTraits::highest()); - } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE T finalize(const T& accum) const { - return accum; - } -}; - -template -struct reducer_traits, Device> { - enum { - Cost = NumTraits::AddCost, - PacketAccess = false - }; -}; - - -template -class GaussianGenerator { - public: - static const bool PacketAccess = false; - - EIGEN_DEVICE_FUNC GaussianGenerator(const array& means, - const array& std_devs) - : m_means(means) + }; + + template struct reducer_traits, Device> + { + enum { Cost = NumTraits::AddCost, PacketAccess = PacketType::HasAdd }; + }; + + + template struct MinMaxBottomValue + { + EIGEN_DEVICE_FUNC static EIGEN_STRONG_INLINE T bottom_value() { return Eigen::NumTraits::lowest(); } + }; + template struct MinMaxBottomValue + { + EIGEN_DEVICE_FUNC static EIGEN_STRONG_INLINE T bottom_value() { return -Eigen::NumTraits::infinity(); } + }; + template struct MinMaxBottomValue + { + EIGEN_DEVICE_FUNC static EIGEN_STRONG_INLINE T bottom_value() { return Eigen::NumTraits::highest(); } + }; + template struct MinMaxBottomValue + { + EIGEN_DEVICE_FUNC static EIGEN_STRONG_INLINE T bottom_value() { return Eigen::NumTraits::infinity(); } + }; + + + template struct MaxReducer + { + static const bool PacketAccess = packet_traits::HasMax; + static const bool IsStateful = false; + + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void reduce(const T t, T *accum) const + { + if (t > *accum) { *accum = t; } + } + template + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void reducePacket(const Packet &p, Packet *accum) const + { + (*accum) = pmax(*accum, p); + } + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE T initialize() const + { + return MinMaxBottomValue::IsInteger>::bottom_value(); + } + template EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Packet initializePacket() const + { + return pset1(initialize()); + } + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE T finalize(const T accum) const { return accum; } + template EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Packet finalizePacket(const Packet &vaccum) const + { + return vaccum; + } + template + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE T finalizeBoth(const T saccum, const Packet &vaccum) const + { + return numext::maxi(saccum, predux_max(vaccum)); + } + }; + + template struct reducer_traits, Device> + { + enum { Cost = NumTraits::AddCost, PacketAccess = PacketType::HasMax }; + }; + + + template struct MinReducer + { + static const bool PacketAccess = packet_traits::HasMin; + static const bool IsStateful = false; + + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void reduce(const T t, T *accum) const + { + if (t < *accum) { *accum = t; } + } + template + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void reducePacket(const Packet &p, Packet *accum) const + { + (*accum) = pmin(*accum, p); + } + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE T initialize() const + { + return MinMaxBottomValue::IsInteger>::bottom_value(); + } + template EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Packet initializePacket() const + { + return pset1(initialize()); + } + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE T finalize(const T accum) const { return accum; } + template EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Packet finalizePacket(const Packet &vaccum) const + { + return vaccum; + } + template + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE T finalizeBoth(const T saccum, const Packet &vaccum) const + { + return numext::mini(saccum, predux_min(vaccum)); + } + }; + + template struct reducer_traits, Device> + { + enum { Cost = NumTraits::AddCost, PacketAccess = PacketType::HasMin }; + }; + + + template struct ProdReducer + { + static const bool PacketAccess = packet_traits::HasMul; + static const bool IsStateful = false; + + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void reduce(const T t, T *accum) const + { + internal::scalar_product_op prod_op; + (*accum) = prod_op(*accum, t); + } + template + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void reducePacket(const Packet &p, Packet *accum) const + { + (*accum) = pmul(*accum, p); + } + + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE T initialize() const + { + internal::scalar_cast_op conv; + return conv(1); + } + template EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Packet initializePacket() const + { + return pset1(initialize()); + } + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE T finalize(const T accum) const { return accum; } + template EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Packet finalizePacket(const Packet &vaccum) const + { + return vaccum; + } + template + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE T finalizeBoth(const T saccum, const Packet &vaccum) const + { + internal::scalar_product_op prod_op; + return prod_op(saccum, predux_mul(vaccum)); + } + }; + + template struct reducer_traits, Device> + { + enum { Cost = NumTraits::MulCost, PacketAccess = PacketType::HasMul }; + }; + + + struct AndReducer + { + static const bool PacketAccess = false; + static const bool IsStateful = false; + + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void reduce(bool t, bool *accum) const { *accum = *accum && t; } + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE bool initialize() const { return true; } + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE bool finalize(bool accum) const { return accum; } + }; + + template struct reducer_traits + { + enum { Cost = 1, PacketAccess = false }; + }; + + + struct OrReducer + { + static const bool PacketAccess = false; + static const bool IsStateful = false; + + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void reduce(bool t, bool *accum) const { *accum = *accum || t; } + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE bool initialize() const { return false; } + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE bool finalize(bool accum) const { return accum; } + }; + + template struct reducer_traits + { + enum { Cost = 1, PacketAccess = false }; + }; + + + // Argmin/Argmax reducers + template struct ArgMaxTupleReducer { - for (size_t i = 0; i < NumDims; ++i) { - m_two_sigmas[i] = std_devs[i] * std_devs[i] * 2; + static const bool PacketAccess = false; + static const bool IsStateful = false; + + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void reduce(const T t, T *accum) const + { + if (t.second > accum->second) { *accum = t; } + } + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE T initialize() const + { + return T(0, NumTraits::lowest()); } - } + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE T finalize(const T &accum) const { return accum; } + }; - EIGEN_DEVICE_FUNC T operator()(const array& coordinates) const { - T tmp = T(0); - for (size_t i = 0; i < NumDims; ++i) { - T offset = coordinates[i] - m_means[i]; - tmp += offset * offset / m_two_sigmas[i]; + template struct reducer_traits, Device> + { + enum { Cost = NumTraits::AddCost, PacketAccess = false }; + }; + + + template struct ArgMinTupleReducer + { + static const bool PacketAccess = false; + static const bool IsStateful = false; + + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void reduce(const T &t, T *accum) const + { + if (t.second < accum->second) { *accum = t; } + } + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE T initialize() const + { + return T(0, NumTraits::highest()); } - return numext::exp(-tmp); - } + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE T finalize(const T &accum) const { return accum; } + }; - private: - array m_means; - array m_two_sigmas; -}; + template struct reducer_traits, Device> + { + enum { Cost = NumTraits::AddCost, PacketAccess = false }; + }; -template -struct functor_traits > { - enum { - Cost = NumDims * (2 * NumTraits::AddCost + NumTraits::MulCost + - functor_traits >::Cost) + - functor_traits >::Cost, - PacketAccess = GaussianGenerator::PacketAccess + + template class GaussianGenerator + { + public: + static const bool PacketAccess = false; + + EIGEN_DEVICE_FUNC GaussianGenerator(const array &means, const array &std_devs) + : m_means(means) + { + for (size_t i = 0; i < NumDims; ++i) { m_two_sigmas[i] = std_devs[i] * std_devs[i] * 2; } + } + + EIGEN_DEVICE_FUNC T operator()(const array &coordinates) const + { + T tmp = T(0); + for (size_t i = 0; i < NumDims; ++i) { + T offset = coordinates[i] - m_means[i]; + tmp += offset * offset / m_two_sigmas[i]; + } + return numext::exp(-tmp); + } + + private: + array m_means; + array m_two_sigmas; + }; + + template struct functor_traits> + { + enum { + Cost = + NumDims * (2 * NumTraits::AddCost + NumTraits::MulCost + functor_traits>::Cost) + + functor_traits>::Cost, + PacketAccess = GaussianGenerator::PacketAccess + }; }; -}; -} // end namespace internal -} // end namespace Eigen +}// end namespace internal +}// end namespace Eigen -#endif // EIGEN_CXX11_TENSOR_TENSOR_FUNCTORS_H +#endif// EIGEN_CXX11_TENSOR_TENSOR_FUNCTORS_H diff --git a/filmulator-gui/core/nlmeans/eigen/unsupported/Eigen/CXX11/src/Tensor/TensorGenerator.h b/filmulator-gui/core/nlmeans/eigen/unsupported/Eigen/CXX11/src/Tensor/TensorGenerator.h index e27753b1..856c1361 100644 --- a/filmulator-gui/core/nlmeans/eigen/unsupported/Eigen/CXX11/src/Tensor/TensorGenerator.h +++ b/filmulator-gui/core/nlmeans/eigen/unsupported/Eigen/CXX11/src/Tensor/TensorGenerator.h @@ -13,46 +13,44 @@ namespace Eigen { /** \class TensorGeneratorOp - * \ingroup CXX11_Tensor_Module - * - * \brief Tensor generator class. - * - * - */ + * \ingroup CXX11_Tensor_Module + * + * \brief Tensor generator class. + * + * + */ namespace internal { -template -struct traits > : public traits -{ - typedef typename XprType::Scalar Scalar; - typedef traits XprTraits; - typedef typename XprTraits::StorageKind StorageKind; - typedef typename XprTraits::Index Index; - typedef typename XprType::Nested Nested; - typedef typename remove_reference::type _Nested; - static const int NumDimensions = XprTraits::NumDimensions; - static const int Layout = XprTraits::Layout; -}; - -template -struct eval, Eigen::Dense> -{ - typedef const TensorGeneratorOp& type; -}; + template + struct traits> : public traits + { + typedef typename XprType::Scalar Scalar; + typedef traits XprTraits; + typedef typename XprTraits::StorageKind StorageKind; + typedef typename XprTraits::Index Index; + typedef typename XprType::Nested Nested; + typedef typename remove_reference::type _Nested; + static const int NumDimensions = XprTraits::NumDimensions; + static const int Layout = XprTraits::Layout; + }; -template -struct nested, 1, typename eval >::type> -{ - typedef TensorGeneratorOp type; -}; + template struct eval, Eigen::Dense> + { + typedef const TensorGeneratorOp &type; + }; -} // end namespace internal + template + struct nested, 1, typename eval>::type> + { + typedef TensorGeneratorOp type; + }; +}// end namespace internal template class TensorGeneratorOp : public TensorBase, ReadOnlyAccessors> { - public: +public: typedef typename Eigen::internal::traits::Scalar Scalar; typedef typename Eigen::NumTraits::Real RealScalar; typedef typename XprType::CoeffReturnType CoeffReturnType; @@ -60,19 +58,19 @@ class TensorGeneratorOp : public TensorBase::StorageKind StorageKind; typedef typename Eigen::internal::traits::Index Index; - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TensorGeneratorOp(const XprType& expr, const Generator& generator) - : m_xpr(expr), m_generator(generator) {} + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TensorGeneratorOp(const XprType &expr, const Generator &generator) + : m_xpr(expr), m_generator(generator) + {} - EIGEN_DEVICE_FUNC - const Generator& generator() const { return m_generator; } + EIGEN_DEVICE_FUNC + const Generator &generator() const { return m_generator; } - EIGEN_DEVICE_FUNC - const typename internal::remove_all::type& - expression() const { return m_xpr; } + EIGEN_DEVICE_FUNC + const typename internal::remove_all::type &expression() const { return m_xpr; } - protected: - typename XprType::Nested m_xpr; - const Generator m_generator; +protected: + typename XprType::Nested m_xpr; + const Generator m_generator; }; @@ -92,36 +90,29 @@ struct TensorEvaluator, Device> PacketAccess = (internal::unpacket_traits::size > 1), BlockAccess = false, Layout = TensorEvaluator::Layout, - CoordAccess = false, // to be implemented + CoordAccess = false,// to be implemented RawAccess = false }; - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TensorEvaluator(const XprType& op, const Device& device) - : m_generator(op.generator()) + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TensorEvaluator(const XprType &op, const Device &device) + : m_generator(op.generator()) { TensorEvaluator impl(op.expression(), device); m_dimensions = impl.dimensions(); if (static_cast(Layout) == static_cast(ColMajor)) { m_strides[0] = 1; - for (int i = 1; i < NumDims; ++i) { - m_strides[i] = m_strides[i - 1] * m_dimensions[i - 1]; - } + for (int i = 1; i < NumDims; ++i) { m_strides[i] = m_strides[i - 1] * m_dimensions[i - 1]; } } else { m_strides[NumDims - 1] = 1; - for (int i = NumDims - 2; i >= 0; --i) { - m_strides[i] = m_strides[i + 1] * m_dimensions[i + 1]; - } + for (int i = NumDims - 2; i >= 0; --i) { m_strides[i] = m_strides[i + 1] * m_dimensions[i + 1]; } } } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const Dimensions& dimensions() const { return m_dimensions; } + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const Dimensions &dimensions() const { return m_dimensions; } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE bool evalSubExprsIfNeeded(Scalar* /*data*/) { - return true; - } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void cleanup() { - } + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE bool evalSubExprsIfNeeded(Scalar * /*data*/) { return true; } + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void cleanup() {} EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE CoeffReturnType coeff(Index index) const { @@ -130,34 +121,30 @@ struct TensorEvaluator, Device> return m_generator(coords); } - template - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE PacketReturnType packet(Index index) const + template EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE PacketReturnType packet(Index index) const { const int packetSize = internal::unpacket_traits::size; EIGEN_STATIC_ASSERT((packetSize > 1), YOU_MADE_A_PROGRAMMING_MISTAKE) - eigen_assert(index+packetSize-1 < dimensions().TotalSize()); + eigen_assert(index + packetSize - 1 < dimensions().TotalSize()); EIGEN_ALIGN_MAX typename internal::remove_const::type values[packetSize]; - for (int i = 0; i < packetSize; ++i) { - values[i] = coeff(index+i); - } + for (int i = 0; i < packetSize; ++i) { values[i] = coeff(index + i); } PacketReturnType rslt = internal::pload(values); return rslt; } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TensorOpCost - costPerCoeff(bool) const { + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TensorOpCost costPerCoeff(bool) const + { // TODO(rmlarsen): This is just a placeholder. Define interface to make // generators return their cost. - return TensorOpCost(0, 0, TensorOpCost::AddCost() + - TensorOpCost::MulCost()); + return TensorOpCost(0, 0, TensorOpCost::AddCost() + TensorOpCost::MulCost()); } - EIGEN_DEVICE_FUNC Scalar* data() const { return NULL; } + EIGEN_DEVICE_FUNC Scalar *data() const { return NULL; } - protected: - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE - void extract_coordinates(Index index, array& coords) const { +protected: + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void extract_coordinates(Index index, array &coords) const + { if (static_cast(Layout) == static_cast(ColMajor)) { for (int i = NumDims - 1; i > 0; --i) { const Index idx = index / m_strides[i]; @@ -171,7 +158,7 @@ struct TensorEvaluator, Device> index -= idx * m_strides[i]; coords[i] = idx; } - coords[NumDims-1] = index; + coords[NumDims - 1] = index; } } @@ -180,6 +167,6 @@ struct TensorEvaluator, Device> Generator m_generator; }; -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_CXX11_TENSOR_TENSOR_GENERATOR_H +#endif// EIGEN_CXX11_TENSOR_TENSOR_GENERATOR_H diff --git a/filmulator-gui/core/nlmeans/eigen/unsupported/Eigen/CXX11/src/Tensor/TensorGlobalFunctions.h b/filmulator-gui/core/nlmeans/eigen/unsupported/Eigen/CXX11/src/Tensor/TensorGlobalFunctions.h index 665b861c..3ac79fb1 100644 --- a/filmulator-gui/core/nlmeans/eigen/unsupported/Eigen/CXX11/src/Tensor/TensorGlobalFunctions.h +++ b/filmulator-gui/core/nlmeans/eigen/unsupported/Eigen/CXX11/src/Tensor/TensorGlobalFunctions.h @@ -17,17 +17,19 @@ namespace Eigen { * This function computes the regularized incomplete beta function (integral). * */ -template -EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const - TensorCwiseTernaryOp, - const ADerived, const BDerived, const XDerived> - betainc(const ADerived& a, const BDerived& b, const XDerived& x) { - return TensorCwiseTernaryOp< - internal::scalar_betainc_op, const ADerived, - const BDerived, const XDerived>( - a, b, x, internal::scalar_betainc_op()); +template +EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const TensorCwiseTernaryOp, + const ADerived, + const BDerived, + const XDerived> + betainc(const ADerived &a, const BDerived &b, const XDerived &x) +{ + return TensorCwiseTernaryOp, + const ADerived, + const BDerived, + const XDerived>(a, b, x, internal::scalar_betainc_op()); } -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_CXX11_TENSOR_TENSOR_GLOBAL_FUNCTIONS_H +#endif// EIGEN_CXX11_TENSOR_TENSOR_GLOBAL_FUNCTIONS_H diff --git a/filmulator-gui/core/nlmeans/eigen/unsupported/Eigen/CXX11/src/Tensor/TensorIO.h b/filmulator-gui/core/nlmeans/eigen/unsupported/Eigen/CXX11/src/Tensor/TensorIO.h index a901c5dd..6738c692 100644 --- a/filmulator-gui/core/nlmeans/eigen/unsupported/Eigen/CXX11/src/Tensor/TensorIO.h +++ b/filmulator-gui/core/nlmeans/eigen/unsupported/Eigen/CXX11/src/Tensor/TensorIO.h @@ -14,49 +14,50 @@ namespace Eigen { namespace internal { -// Print the tensor as a 2d matrix -template -struct TensorPrinter { - static void run (std::ostream& os, const Tensor& tensor) { - typedef typename internal::remove_const::type Scalar; - typedef typename Tensor::Index Index; - const Index total_size = internal::array_prod(tensor.dimensions()); - if (total_size > 0) { - const Index first_dim = Eigen::internal::array_get<0>(tensor.dimensions()); - static const int layout = Tensor::Layout; - Map > matrix(const_cast(tensor.data()), first_dim, total_size/first_dim); - os << matrix; + // Print the tensor as a 2d matrix + template struct TensorPrinter + { + static void run(std::ostream &os, const Tensor &tensor) + { + typedef typename internal::remove_const::type Scalar; + typedef typename Tensor::Index Index; + const Index total_size = internal::array_prod(tensor.dimensions()); + if (total_size > 0) { + const Index first_dim = Eigen::internal::array_get<0>(tensor.dimensions()); + static const int layout = Tensor::Layout; + Map> matrix( + const_cast(tensor.data()), first_dim, total_size / first_dim); + os << matrix; + } } - } -}; + }; -// Print the tensor as a vector -template -struct TensorPrinter { - static void run (std::ostream& os, const Tensor& tensor) { - typedef typename internal::remove_const::type Scalar; - typedef typename Tensor::Index Index; - const Index total_size = internal::array_prod(tensor.dimensions()); - if (total_size > 0) { - Map > array(const_cast(tensor.data()), total_size); - os << array; + // Print the tensor as a vector + template struct TensorPrinter + { + static void run(std::ostream &os, const Tensor &tensor) + { + typedef typename internal::remove_const::type Scalar; + typedef typename Tensor::Index Index; + const Index total_size = internal::array_prod(tensor.dimensions()); + if (total_size > 0) { + Map> array(const_cast(tensor.data()), total_size); + os << array; + } } - } -}; + }; -// Print the tensor as a scalar -template -struct TensorPrinter { - static void run (std::ostream& os, const Tensor& tensor) { - os << tensor.coeff(0); - } -}; -} + // Print the tensor as a scalar + template struct TensorPrinter + { + static void run(std::ostream &os, const Tensor &tensor) { os << tensor.coeff(0); } + }; +}// namespace internal -template -std::ostream& operator << (std::ostream& os, const TensorBase& expr) { +template std::ostream &operator<<(std::ostream &os, const TensorBase &expr) +{ typedef TensorEvaluator, DefaultDevice> Evaluator; typedef typename Evaluator::Dimensions Dimensions; @@ -74,6 +75,6 @@ std::ostream& operator << (std::ostream& os, const TensorBase -struct traits > : public traits -{ - typedef typename internal::remove_const::type Scalar; - typedef traits XprTraits; - typedef typename XprTraits::StorageKind StorageKind; - typedef typename XprTraits::Index Index; - typedef typename XprType::Nested Nested; - typedef typename remove_reference::type _Nested; - static const int NumDimensions = XprTraits::NumDimensions + 1; - static const int Layout = XprTraits::Layout; -}; + template + struct traits> : public traits + { + typedef typename internal::remove_const::type Scalar; + typedef traits XprTraits; + typedef typename XprTraits::StorageKind StorageKind; + typedef typename XprTraits::Index Index; + typedef typename XprType::Nested Nested; + typedef typename remove_reference::type _Nested; + static const int NumDimensions = XprTraits::NumDimensions + 1; + static const int Layout = XprTraits::Layout; + }; -template -struct eval, Eigen::Dense> -{ - typedef const TensorImagePatchOp& type; -}; + template + struct eval, Eigen::Dense> + { + typedef const TensorImagePatchOp &type; + }; -template -struct nested, 1, typename eval >::type> -{ - typedef TensorImagePatchOp type; -}; + template + struct nested, + 1, + typename eval>::type> + { + typedef TensorImagePatchOp type; + }; -} // end namespace internal +}// end namespace internal template class TensorImagePatchOp : public TensorBase, ReadOnlyAccessors> { - public: +public: typedef typename Eigen::internal::traits::Scalar Scalar; typedef typename Eigen::NumTraits::Real RealScalar; typedef typename XprType::CoeffReturnType CoeffReturnType; @@ -65,85 +67,96 @@ class TensorImagePatchOp : public TensorBase::StorageKind StorageKind; typedef typename Eigen::internal::traits::Index Index; - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TensorImagePatchOp(const XprType& expr, DenseIndex patch_rows, DenseIndex patch_cols, - DenseIndex row_strides, DenseIndex col_strides, - DenseIndex in_row_strides, DenseIndex in_col_strides, - DenseIndex row_inflate_strides, DenseIndex col_inflate_strides, - PaddingType padding_type, Scalar padding_value) - : m_xpr(expr), m_patch_rows(patch_rows), m_patch_cols(patch_cols), - m_row_strides(row_strides), m_col_strides(col_strides), - m_in_row_strides(in_row_strides), m_in_col_strides(in_col_strides), - m_row_inflate_strides(row_inflate_strides), m_col_inflate_strides(col_inflate_strides), - m_padding_explicit(false), m_padding_top(0), m_padding_bottom(0), m_padding_left(0), m_padding_right(0), - m_padding_type(padding_type), m_padding_value(padding_value) {} - - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TensorImagePatchOp(const XprType& expr, DenseIndex patch_rows, DenseIndex patch_cols, - DenseIndex row_strides, DenseIndex col_strides, - DenseIndex in_row_strides, DenseIndex in_col_strides, - DenseIndex row_inflate_strides, DenseIndex col_inflate_strides, - DenseIndex padding_top, DenseIndex padding_bottom, - DenseIndex padding_left, DenseIndex padding_right, - Scalar padding_value) - : m_xpr(expr), m_patch_rows(patch_rows), m_patch_cols(patch_cols), - m_row_strides(row_strides), m_col_strides(col_strides), - m_in_row_strides(in_row_strides), m_in_col_strides(in_col_strides), - m_row_inflate_strides(row_inflate_strides), m_col_inflate_strides(col_inflate_strides), - m_padding_explicit(true), m_padding_top(padding_top), m_padding_bottom(padding_bottom), - m_padding_left(padding_left), m_padding_right(padding_right), - m_padding_type(PADDING_VALID), m_padding_value(padding_value) {} - - EIGEN_DEVICE_FUNC - DenseIndex patch_rows() const { return m_patch_rows; } - EIGEN_DEVICE_FUNC - DenseIndex patch_cols() const { return m_patch_cols; } - EIGEN_DEVICE_FUNC - DenseIndex row_strides() const { return m_row_strides; } - EIGEN_DEVICE_FUNC - DenseIndex col_strides() const { return m_col_strides; } - EIGEN_DEVICE_FUNC - DenseIndex in_row_strides() const { return m_in_row_strides; } - EIGEN_DEVICE_FUNC - DenseIndex in_col_strides() const { return m_in_col_strides; } - EIGEN_DEVICE_FUNC - DenseIndex row_inflate_strides() const { return m_row_inflate_strides; } - EIGEN_DEVICE_FUNC - DenseIndex col_inflate_strides() const { return m_col_inflate_strides; } - EIGEN_DEVICE_FUNC - bool padding_explicit() const { return m_padding_explicit; } - EIGEN_DEVICE_FUNC - DenseIndex padding_top() const { return m_padding_top; } - EIGEN_DEVICE_FUNC - DenseIndex padding_bottom() const { return m_padding_bottom; } - EIGEN_DEVICE_FUNC - DenseIndex padding_left() const { return m_padding_left; } - EIGEN_DEVICE_FUNC - DenseIndex padding_right() const { return m_padding_right; } - EIGEN_DEVICE_FUNC - PaddingType padding_type() const { return m_padding_type; } - EIGEN_DEVICE_FUNC - Scalar padding_value() const { return m_padding_value; } - - EIGEN_DEVICE_FUNC - const typename internal::remove_all::type& - expression() const { return m_xpr; } - - protected: - typename XprType::Nested m_xpr; - const DenseIndex m_patch_rows; - const DenseIndex m_patch_cols; - const DenseIndex m_row_strides; - const DenseIndex m_col_strides; - const DenseIndex m_in_row_strides; - const DenseIndex m_in_col_strides; - const DenseIndex m_row_inflate_strides; - const DenseIndex m_col_inflate_strides; - const bool m_padding_explicit; - const DenseIndex m_padding_top; - const DenseIndex m_padding_bottom; - const DenseIndex m_padding_left; - const DenseIndex m_padding_right; - const PaddingType m_padding_type; - const Scalar m_padding_value; + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TensorImagePatchOp(const XprType &expr, + DenseIndex patch_rows, + DenseIndex patch_cols, + DenseIndex row_strides, + DenseIndex col_strides, + DenseIndex in_row_strides, + DenseIndex in_col_strides, + DenseIndex row_inflate_strides, + DenseIndex col_inflate_strides, + PaddingType padding_type, + Scalar padding_value) + : m_xpr(expr), m_patch_rows(patch_rows), m_patch_cols(patch_cols), m_row_strides(row_strides), + m_col_strides(col_strides), m_in_row_strides(in_row_strides), m_in_col_strides(in_col_strides), + m_row_inflate_strides(row_inflate_strides), m_col_inflate_strides(col_inflate_strides), m_padding_explicit(false), + m_padding_top(0), m_padding_bottom(0), m_padding_left(0), m_padding_right(0), m_padding_type(padding_type), + m_padding_value(padding_value) + {} + + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TensorImagePatchOp(const XprType &expr, + DenseIndex patch_rows, + DenseIndex patch_cols, + DenseIndex row_strides, + DenseIndex col_strides, + DenseIndex in_row_strides, + DenseIndex in_col_strides, + DenseIndex row_inflate_strides, + DenseIndex col_inflate_strides, + DenseIndex padding_top, + DenseIndex padding_bottom, + DenseIndex padding_left, + DenseIndex padding_right, + Scalar padding_value) + : m_xpr(expr), m_patch_rows(patch_rows), m_patch_cols(patch_cols), m_row_strides(row_strides), + m_col_strides(col_strides), m_in_row_strides(in_row_strides), m_in_col_strides(in_col_strides), + m_row_inflate_strides(row_inflate_strides), m_col_inflate_strides(col_inflate_strides), m_padding_explicit(true), + m_padding_top(padding_top), m_padding_bottom(padding_bottom), m_padding_left(padding_left), + m_padding_right(padding_right), m_padding_type(PADDING_VALID), m_padding_value(padding_value) + {} + + EIGEN_DEVICE_FUNC + DenseIndex patch_rows() const { return m_patch_rows; } + EIGEN_DEVICE_FUNC + DenseIndex patch_cols() const { return m_patch_cols; } + EIGEN_DEVICE_FUNC + DenseIndex row_strides() const { return m_row_strides; } + EIGEN_DEVICE_FUNC + DenseIndex col_strides() const { return m_col_strides; } + EIGEN_DEVICE_FUNC + DenseIndex in_row_strides() const { return m_in_row_strides; } + EIGEN_DEVICE_FUNC + DenseIndex in_col_strides() const { return m_in_col_strides; } + EIGEN_DEVICE_FUNC + DenseIndex row_inflate_strides() const { return m_row_inflate_strides; } + EIGEN_DEVICE_FUNC + DenseIndex col_inflate_strides() const { return m_col_inflate_strides; } + EIGEN_DEVICE_FUNC + bool padding_explicit() const { return m_padding_explicit; } + EIGEN_DEVICE_FUNC + DenseIndex padding_top() const { return m_padding_top; } + EIGEN_DEVICE_FUNC + DenseIndex padding_bottom() const { return m_padding_bottom; } + EIGEN_DEVICE_FUNC + DenseIndex padding_left() const { return m_padding_left; } + EIGEN_DEVICE_FUNC + DenseIndex padding_right() const { return m_padding_right; } + EIGEN_DEVICE_FUNC + PaddingType padding_type() const { return m_padding_type; } + EIGEN_DEVICE_FUNC + Scalar padding_value() const { return m_padding_value; } + + EIGEN_DEVICE_FUNC + const typename internal::remove_all::type &expression() const { return m_xpr; } + +protected: + typename XprType::Nested m_xpr; + const DenseIndex m_patch_rows; + const DenseIndex m_patch_cols; + const DenseIndex m_row_strides; + const DenseIndex m_col_strides; + const DenseIndex m_in_row_strides; + const DenseIndex m_in_col_strides; + const DenseIndex m_row_inflate_strides; + const DenseIndex m_col_inflate_strides; + const bool m_padding_explicit; + const DenseIndex m_padding_top; + const DenseIndex m_padding_bottom; + const DenseIndex m_padding_left; + const DenseIndex m_padding_right; + const PaddingType m_padding_type; + const Scalar m_padding_value; }; // Eval as rvalue @@ -156,8 +169,7 @@ struct TensorEvaluator, Device> static const int NumDims = NumInputDims + 1; typedef DSizes Dimensions; typedef typename internal::remove_const::type Scalar; - typedef TensorEvaluator, - Device> Self; + typedef TensorEvaluator, Device> Self; typedef TensorEvaluator Impl; typedef typename XprType::CoeffReturnType CoeffReturnType; typedef typename PacketType::type PacketReturnType; @@ -171,14 +183,14 @@ struct TensorEvaluator, Device> RawAccess = false }; - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TensorEvaluator(const XprType& op, const Device& device) - : m_impl(op.expression(), device) + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TensorEvaluator(const XprType &op, const Device &device) + : m_impl(op.expression(), device) { EIGEN_STATIC_ASSERT((NumDims >= 4), YOU_MADE_A_PROGRAMMING_MISTAKE); m_paddingValue = op.padding_value(); - const typename TensorEvaluator::Dimensions& input_dims = m_impl.dimensions(); + const typename TensorEvaluator::Dimensions &input_dims = m_impl.dimensions(); // Caches a few variables. if (static_cast(Layout) == static_cast(ColMajor)) { @@ -186,9 +198,9 @@ struct TensorEvaluator, Device> m_inputRows = input_dims[1]; m_inputCols = input_dims[2]; } else { - m_inputDepth = input_dims[NumInputDims-1]; - m_inputRows = input_dims[NumInputDims-2]; - m_inputCols = input_dims[NumInputDims-3]; + m_inputDepth = input_dims[NumInputDims - 1]; + m_inputRows = input_dims[NumInputDims - 2]; + m_inputCols = input_dims[NumInputDims - 3]; } m_row_strides = op.row_strides(); @@ -218,29 +230,33 @@ struct TensorEvaluator, Device> m_patch_cols_eff = op.patch_cols() + (op.patch_cols() - 1) * (m_in_col_strides - 1); if (op.padding_explicit()) { - m_outputRows = numext::ceil((m_input_rows_eff + op.padding_top() + op.padding_bottom() - m_patch_rows_eff + 1.f) / static_cast(m_row_strides)); - m_outputCols = numext::ceil((m_input_cols_eff + op.padding_left() + op.padding_right() - m_patch_cols_eff + 1.f) / static_cast(m_col_strides)); + m_outputRows = numext::ceil((m_input_rows_eff + op.padding_top() + op.padding_bottom() - m_patch_rows_eff + 1.f) + / static_cast(m_row_strides)); + m_outputCols = numext::ceil((m_input_cols_eff + op.padding_left() + op.padding_right() - m_patch_cols_eff + 1.f) + / static_cast(m_col_strides)); m_rowPaddingTop = op.padding_top(); m_colPaddingLeft = op.padding_left(); } else { // Computing padding from the type switch (op.padding_type()) { - case PADDING_VALID: - m_outputRows = numext::ceil((m_input_rows_eff - m_patch_rows_eff + 1.f) / static_cast(m_row_strides)); - m_outputCols = numext::ceil((m_input_cols_eff - m_patch_cols_eff + 1.f) / static_cast(m_col_strides)); - // Calculate the padding - m_rowPaddingTop = numext::maxi(0, ((m_outputRows - 1) * m_row_strides + m_patch_rows_eff - m_input_rows_eff) / 2); - m_colPaddingLeft = numext::maxi(0, ((m_outputCols - 1) * m_col_strides + m_patch_cols_eff - m_input_cols_eff) / 2); - break; - case PADDING_SAME: - m_outputRows = numext::ceil(m_input_rows_eff / static_cast(m_row_strides)); - m_outputCols = numext::ceil(m_input_cols_eff / static_cast(m_col_strides)); - // Calculate the padding - m_rowPaddingTop = ((m_outputRows - 1) * m_row_strides + m_patch_rows_eff - m_input_rows_eff) / 2; - m_colPaddingLeft = ((m_outputCols - 1) * m_col_strides + m_patch_cols_eff - m_input_cols_eff) / 2; - break; - default: - eigen_assert(false && "unexpected padding"); + case PADDING_VALID: + m_outputRows = numext::ceil((m_input_rows_eff - m_patch_rows_eff + 1.f) / static_cast(m_row_strides)); + m_outputCols = numext::ceil((m_input_cols_eff - m_patch_cols_eff + 1.f) / static_cast(m_col_strides)); + // Calculate the padding + m_rowPaddingTop = + numext::maxi(0, ((m_outputRows - 1) * m_row_strides + m_patch_rows_eff - m_input_rows_eff) / 2); + m_colPaddingLeft = + numext::maxi(0, ((m_outputCols - 1) * m_col_strides + m_patch_cols_eff - m_input_cols_eff) / 2); + break; + case PADDING_SAME: + m_outputRows = numext::ceil(m_input_rows_eff / static_cast(m_row_strides)); + m_outputCols = numext::ceil(m_input_cols_eff / static_cast(m_col_strides)); + // Calculate the padding + m_rowPaddingTop = ((m_outputRows - 1) * m_row_strides + m_patch_rows_eff - m_input_rows_eff) / 2; + m_colPaddingLeft = ((m_outputCols - 1) * m_col_strides + m_patch_cols_eff - m_input_cols_eff) / 2; + break; + default: + eigen_assert(false && "unexpected padding"); } } eigen_assert(m_outputRows > 0); @@ -258,9 +274,7 @@ struct TensorEvaluator, Device> m_dimensions[1] = op.patch_rows(); m_dimensions[2] = op.patch_cols(); m_dimensions[3] = m_outputRows * m_outputCols; - for (int i = 4; i < NumDims; ++i) { - m_dimensions[i] = input_dims[i-1]; - } + for (int i = 4; i < NumDims; ++i) { m_dimensions[i] = input_dims[i - 1]; } } else { // RowMajor // NumDims-1: depth @@ -268,13 +282,11 @@ struct TensorEvaluator, Device> // NumDims-3: patch_cols // NumDims-4: number of patches // NumDims-5 and beyond: anything else (such as batch). - m_dimensions[NumDims-1] = input_dims[NumInputDims-1]; - m_dimensions[NumDims-2] = op.patch_rows(); - m_dimensions[NumDims-3] = op.patch_cols(); - m_dimensions[NumDims-4] = m_outputRows * m_outputCols; - for (int i = NumDims-5; i >= 0; --i) { - m_dimensions[i] = input_dims[i]; - } + m_dimensions[NumDims - 1] = input_dims[NumInputDims - 1]; + m_dimensions[NumDims - 2] = op.patch_rows(); + m_dimensions[NumDims - 3] = op.patch_cols(); + m_dimensions[NumDims - 4] = m_outputRows * m_outputCols; + for (int i = NumDims - 5; i >= 0; --i) { m_dimensions[i] = input_dims[i]; } } // Strides for moving the patch in various dimensions. @@ -283,9 +295,9 @@ struct TensorEvaluator, Device> m_patchStride = m_colStride * m_dimensions[2] * m_dimensions[0]; m_otherStride = m_patchStride * m_dimensions[3]; } else { - m_colStride = m_dimensions[NumDims-2]; - m_patchStride = m_colStride * m_dimensions[NumDims-3] * m_dimensions[NumDims-1]; - m_otherStride = m_patchStride * m_dimensions[NumDims-4]; + m_colStride = m_dimensions[NumDims - 2]; + m_patchStride = m_colStride * m_dimensions[NumDims - 3] * m_dimensions[NumDims - 1]; + m_otherStride = m_patchStride * m_dimensions[NumDims - 4]; } // Strides for navigating through the input tensor. @@ -306,20 +318,19 @@ struct TensorEvaluator, Device> if (static_cast(Layout) == static_cast(ColMajor)) { m_fastOutputDepth = internal::TensorIntDivisor(m_dimensions[0]); } else { - m_fastOutputDepth = internal::TensorIntDivisor(m_dimensions[NumDims-1]); + m_fastOutputDepth = internal::TensorIntDivisor(m_dimensions[NumDims - 1]); } } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const Dimensions& dimensions() const { return m_dimensions; } + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const Dimensions &dimensions() const { return m_dimensions; } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE bool evalSubExprsIfNeeded(Scalar* /*data*/) { + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE bool evalSubExprsIfNeeded(Scalar * /*data*/) + { m_impl.evalSubExprsIfNeeded(NULL); return true; } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void cleanup() { - m_impl.cleanup(); - } + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void cleanup() { m_impl.cleanup(); } EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE CoeffReturnType coeff(Index index) const { @@ -336,9 +347,10 @@ struct TensorEvaluator, Device> const Index colIndex = patch2DIndex / m_fastOutputRows; const Index colOffset = patchOffset / m_fastColStride; const Index inputCol = colIndex * m_col_strides + colOffset * m_in_col_strides - m_colPaddingLeft; - const Index origInputCol = (m_col_inflate_strides == 1) ? inputCol : ((inputCol >= 0) ? (inputCol / m_fastInflateColStride) : 0); - if (inputCol < 0 || inputCol >= m_input_cols_eff || - ((m_col_inflate_strides != 1) && (inputCol != origInputCol * m_col_inflate_strides))) { + const Index origInputCol = + (m_col_inflate_strides == 1) ? inputCol : ((inputCol >= 0) ? (inputCol / m_fastInflateColStride) : 0); + if (inputCol < 0 || inputCol >= m_input_cols_eff + || ((m_col_inflate_strides != 1) && (inputCol != origInputCol * m_col_inflate_strides))) { return Scalar(m_paddingValue); } @@ -346,61 +358,62 @@ struct TensorEvaluator, Device> const Index rowIndex = patch2DIndex - colIndex * m_outputRows; const Index rowOffset = patchOffset - colOffset * m_colStride; const Index inputRow = rowIndex * m_row_strides + rowOffset * m_in_row_strides - m_rowPaddingTop; - const Index origInputRow = (m_row_inflate_strides == 1) ? inputRow : ((inputRow >= 0) ? (inputRow / m_fastInflateRowStride) : 0); - if (inputRow < 0 || inputRow >= m_input_rows_eff || - ((m_row_inflate_strides != 1) && (inputRow != origInputRow * m_row_inflate_strides))) { + const Index origInputRow = + (m_row_inflate_strides == 1) ? inputRow : ((inputRow >= 0) ? (inputRow / m_fastInflateRowStride) : 0); + if (inputRow < 0 || inputRow >= m_input_rows_eff + || ((m_row_inflate_strides != 1) && (inputRow != origInputRow * m_row_inflate_strides))) { return Scalar(m_paddingValue); } const int depth_index = static_cast(Layout) == static_cast(ColMajor) ? 0 : NumDims - 1; const Index depth = index - (index / m_fastOutputDepth) * m_dimensions[depth_index]; - const Index inputIndex = depth + origInputRow * m_rowInputStride + origInputCol * m_colInputStride + otherIndex * m_patchInputStride; + const Index inputIndex = + depth + origInputRow * m_rowInputStride + origInputCol * m_colInputStride + otherIndex * m_patchInputStride; return m_impl.coeff(inputIndex); } - template - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE PacketReturnType packet(Index index) const + template EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE PacketReturnType packet(Index index) const { EIGEN_STATIC_ASSERT((PacketSize > 1), YOU_MADE_A_PROGRAMMING_MISTAKE) - eigen_assert(index+PacketSize-1 < dimensions().TotalSize()); + eigen_assert(index + PacketSize - 1 < dimensions().TotalSize()); if (m_in_row_strides != 1 || m_in_col_strides != 1 || m_row_inflate_strides != 1 || m_col_inflate_strides != 1) { return packetWithPossibleZero(index); } - const Index indices[2] = {index, index + PacketSize - 1}; + const Index indices[2] = { index, index + PacketSize - 1 }; const Index patchIndex = indices[0] / m_fastPatchStride; - if (patchIndex != indices[1] / m_fastPatchStride) { - return packetWithPossibleZero(index); - } + if (patchIndex != indices[1] / m_fastPatchStride) { return packetWithPossibleZero(index); } const Index otherIndex = (NumDims == 4) ? 0 : indices[0] / m_fastOtherStride; eigen_assert(otherIndex == indices[1] / m_fastOtherStride); // Find the offset of the element wrt the location of the first element. - const Index patchOffsets[2] = {(indices[0] - patchIndex * m_patchStride) / m_fastOutputDepth, - (indices[1] - patchIndex * m_patchStride) / m_fastOutputDepth}; + const Index patchOffsets[2] = { (indices[0] - patchIndex * m_patchStride) / m_fastOutputDepth, + (indices[1] - patchIndex * m_patchStride) / m_fastOutputDepth }; - const Index patch2DIndex = (NumDims == 4) ? patchIndex : (indices[0] - otherIndex * m_otherStride) / m_fastPatchStride; + const Index patch2DIndex = + (NumDims == 4) ? patchIndex : (indices[0] - otherIndex * m_otherStride) / m_fastPatchStride; eigen_assert(patch2DIndex == (indices[1] - otherIndex * m_otherStride) / m_fastPatchStride); const Index colIndex = patch2DIndex / m_fastOutputRows; - const Index colOffsets[2] = {patchOffsets[0] / m_fastColStride, patchOffsets[1] / m_fastColStride}; + const Index colOffsets[2] = { patchOffsets[0] / m_fastColStride, patchOffsets[1] / m_fastColStride }; // Calculate col indices in the original input tensor. - const Index inputCols[2] = {colIndex * m_col_strides + colOffsets[0] - - m_colPaddingLeft, colIndex * m_col_strides + colOffsets[1] - m_colPaddingLeft}; + const Index inputCols[2] = { colIndex * m_col_strides + colOffsets[0] - m_colPaddingLeft, + colIndex * m_col_strides + colOffsets[1] - m_colPaddingLeft }; if (inputCols[1] < 0 || inputCols[0] >= m_inputCols) { return internal::pset1(Scalar(m_paddingValue)); } if (inputCols[0] == inputCols[1]) { const Index rowIndex = patch2DIndex - colIndex * m_outputRows; - const Index rowOffsets[2] = {patchOffsets[0] - colOffsets[0]*m_colStride, patchOffsets[1] - colOffsets[1]*m_colStride}; + const Index rowOffsets[2] = { patchOffsets[0] - colOffsets[0] * m_colStride, + patchOffsets[1] - colOffsets[1] * m_colStride }; eigen_assert(rowOffsets[0] <= rowOffsets[1]); // Calculate col indices in the original input tensor. - const Index inputRows[2] = {rowIndex * m_row_strides + rowOffsets[0] - - m_rowPaddingTop, rowIndex * m_row_strides + rowOffsets[1] - m_rowPaddingTop}; + const Index inputRows[2] = { rowIndex * m_row_strides + rowOffsets[0] - m_rowPaddingTop, + rowIndex * m_row_strides + rowOffsets[1] - m_rowPaddingTop }; if (inputRows[1] < 0 || inputRows[0] >= m_inputRows) { return internal::pset1(Scalar(m_paddingValue)); @@ -410,7 +423,8 @@ struct TensorEvaluator, Device> // no padding const int depth_index = static_cast(Layout) == static_cast(ColMajor) ? 0 : NumDims - 1; const Index depth = index - (index / m_fastOutputDepth) * m_dimensions[depth_index]; - const Index inputIndex = depth + inputRows[0] * m_rowInputStride + inputCols[0] * m_colInputStride + otherIndex * m_patchInputStride; + const Index inputIndex = + depth + inputRows[0] * m_rowInputStride + inputCols[0] * m_colInputStride + otherIndex * m_patchInputStride; return m_impl.template packet(inputIndex); } } @@ -418,9 +432,9 @@ struct TensorEvaluator, Device> return packetWithPossibleZero(index); } - EIGEN_DEVICE_FUNC Scalar* data() const { return NULL; } + EIGEN_DEVICE_FUNC Scalar *data() const { return NULL; } - const TensorEvaluator& impl() const { return m_impl; } + const TensorEvaluator &impl() const { return m_impl; } Index rowPaddingTop() const { return m_rowPaddingTop; } Index colPaddingLeft() const { return m_colPaddingLeft; } @@ -433,25 +447,21 @@ struct TensorEvaluator, Device> Index rowInflateStride() const { return m_row_inflate_strides; } Index colInflateStride() const { return m_col_inflate_strides; } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TensorOpCost - costPerCoeff(bool vectorized) const { + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TensorOpCost costPerCoeff(bool vectorized) const + { // We conservatively estimate the cost for the code path where the computed // index is inside the original image and // TensorEvaluator::CoordAccess is false. - const double compute_cost = 3 * TensorOpCost::DivCost() + - 6 * TensorOpCost::MulCost() + - 8 * TensorOpCost::MulCost(); - return m_impl.costPerCoeff(vectorized) + - TensorOpCost(0, 0, compute_cost, vectorized, PacketSize); + const double compute_cost = + 3 * TensorOpCost::DivCost() + 6 * TensorOpCost::MulCost() + 8 * TensorOpCost::MulCost(); + return m_impl.costPerCoeff(vectorized) + TensorOpCost(0, 0, compute_cost, vectorized, PacketSize); } - protected: +protected: EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE PacketReturnType packetWithPossibleZero(Index index) const { EIGEN_ALIGN_MAX typename internal::remove_const::type values[PacketSize]; - for (int i = 0; i < PacketSize; ++i) { - values[i] = coeff(index+i); - } + for (int i = 0; i < PacketSize; ++i) { values[i] = coeff(index + i); } PacketReturnType rslt = internal::pload(values); return rslt; } @@ -504,6 +514,6 @@ struct TensorEvaluator, Device> }; -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_CXX11_TENSOR_TENSOR_IMAGE_PATCH_H +#endif// EIGEN_CXX11_TENSOR_TENSOR_IMAGE_PATCH_H diff --git a/filmulator-gui/core/nlmeans/eigen/unsupported/Eigen/CXX11/src/Tensor/TensorIndexList.h b/filmulator-gui/core/nlmeans/eigen/unsupported/Eigen/CXX11/src/Tensor/TensorIndexList.h index 3209fecd..754e2740 100644 --- a/filmulator-gui/core/nlmeans/eigen/unsupported/Eigen/CXX11/src/Tensor/TensorIndexList.h +++ b/filmulator-gui/core/nlmeans/eigen/unsupported/Eigen/CXX11/src/Tensor/TensorIndexList.h @@ -18,62 +18,53 @@ namespace Eigen { /** \internal - * - * \class TensorIndexList - * \ingroup CXX11_Tensor_Module - * - * \brief Set of classes used to encode a set of Tensor dimensions/indices. - * - * The indices in the list can be known at compile time or at runtime. A mix - * of static and dynamic indices can also be provided if needed. The tensor - * code will attempt to take advantage of the indices that are known at - * compile time to optimize the code it generates. - * - * This functionality requires a c++11 compliant compiler. If your compiler - * is older you need to use arrays of indices instead. - * - * Several examples are provided in the cxx11_tensor_index_list.cpp file. - * - * \sa Tensor - */ - -template -struct type2index { + * + * \class TensorIndexList + * \ingroup CXX11_Tensor_Module + * + * \brief Set of classes used to encode a set of Tensor dimensions/indices. + * + * The indices in the list can be known at compile time or at runtime. A mix + * of static and dynamic indices can also be provided if needed. The tensor + * code will attempt to take advantage of the indices that are known at + * compile time to optimize the code it generates. + * + * This functionality requires a c++11 compliant compiler. If your compiler + * is older you need to use arrays of indices instead. + * + * Several examples are provided in the cxx11_tensor_index_list.cpp file. + * + * \sa Tensor + */ + +template struct type2index +{ static const DenseIndex value = n; EIGEN_DEVICE_FUNC constexpr operator DenseIndex() const { return n; } - EIGEN_DEVICE_FUNC void set(DenseIndex val) { - eigen_assert(val == n); - } + EIGEN_DEVICE_FUNC void set(DenseIndex val) { eigen_assert(val == n); } }; // This can be used with IndexPairList to get compile-time constant pairs, // such as IndexPairList, type2indexpair<3,4>>(). -template -struct type2indexpair { +template struct type2indexpair +{ static const DenseIndex first = f; static const DenseIndex second = s; - constexpr EIGEN_DEVICE_FUNC operator IndexPair() const { - return IndexPair(f, s); - } + constexpr EIGEN_DEVICE_FUNC operator IndexPair() const { return IndexPair(f, s); } - EIGEN_DEVICE_FUNC void set(const IndexPair& val) { + EIGEN_DEVICE_FUNC void set(const IndexPair &val) + { eigen_assert(val.first == f); eigen_assert(val.second == s); } }; -template struct NumTraits > +template struct NumTraits> { typedef DenseIndex Real; - enum { - IsComplex = 0, - RequireInitialization = false, - ReadCost = 1, - AddCost = 1, - MulCost = 1 - }; + enum { IsComplex = 0, RequireInitialization = false, ReadCost = 1, AddCost = 1, MulCost = 1 }; EIGEN_DEVICE_FUNC static inline Real epsilon() { return 0; } EIGEN_DEVICE_FUNC static inline Real dummy_precision() { return 0; } @@ -82,644 +73,674 @@ template struct NumTraits > }; namespace internal { -template -EIGEN_DEVICE_FUNC void update_value(T& val, DenseIndex new_val) { - val = new_val; -} -template -EIGEN_DEVICE_FUNC void update_value(type2index& val, DenseIndex new_val) { - val.set(new_val); -} + template EIGEN_DEVICE_FUNC void update_value(T &val, DenseIndex new_val) { val = new_val; } + template EIGEN_DEVICE_FUNC void update_value(type2index &val, DenseIndex new_val) + { + val.set(new_val); + } -template -EIGEN_DEVICE_FUNC void update_value(T& val, IndexPair new_val) { - val = new_val; -} -template -EIGEN_DEVICE_FUNC void update_value(type2indexpair& val, IndexPair new_val) { - val.set(new_val); -} + template EIGEN_DEVICE_FUNC void update_value(T &val, IndexPair new_val) { val = new_val; } + template + EIGEN_DEVICE_FUNC void update_value(type2indexpair &val, IndexPair new_val) + { + val.set(new_val); + } -template -struct is_compile_time_constant { - static constexpr bool value = false; -}; + template struct is_compile_time_constant + { + static constexpr bool value = false; + }; -template -struct is_compile_time_constant > { - static constexpr bool value = true; -}; -template -struct is_compile_time_constant > { - static constexpr bool value = true; -}; -template -struct is_compile_time_constant& > { - static constexpr bool value = true; -}; -template -struct is_compile_time_constant& > { - static constexpr bool value = true; -}; + template struct is_compile_time_constant> + { + static constexpr bool value = true; + }; + template struct is_compile_time_constant> + { + static constexpr bool value = true; + }; + template struct is_compile_time_constant &> + { + static constexpr bool value = true; + }; + template struct is_compile_time_constant &> + { + static constexpr bool value = true; + }; -template -struct is_compile_time_constant > { - static constexpr bool value = true; -}; -template -struct is_compile_time_constant > { - static constexpr bool value = true; -}; -template -struct is_compile_time_constant& > { - static constexpr bool value = true; -}; -template -struct is_compile_time_constant& > { - static constexpr bool value = true; -}; + template struct is_compile_time_constant> + { + static constexpr bool value = true; + }; + template struct is_compile_time_constant> + { + static constexpr bool value = true; + }; + template struct is_compile_time_constant &> + { + static constexpr bool value = true; + }; + template struct is_compile_time_constant &> + { + static constexpr bool value = true; + }; -template -struct IndexTuple; + template struct IndexTuple; -template -struct IndexTuple { - EIGEN_DEVICE_FUNC constexpr IndexTuple() : head(), others() { } - EIGEN_DEVICE_FUNC constexpr IndexTuple(const T& v, const O... o) : head(v), others(o...) { } + template struct IndexTuple + { + EIGEN_DEVICE_FUNC constexpr IndexTuple() : head(), others() {} + EIGEN_DEVICE_FUNC constexpr IndexTuple(const T &v, const O... o) : head(v), others(o...) {} - constexpr static int count = 1 + sizeof...(O); - T head; - IndexTuple others; - typedef T Head; - typedef IndexTuple Other; -}; + constexpr static int count = 1 + sizeof...(O); + T head; + IndexTuple others; + typedef T Head; + typedef IndexTuple Other; + }; -template - struct IndexTuple { - EIGEN_DEVICE_FUNC constexpr IndexTuple() : head() { } - EIGEN_DEVICE_FUNC constexpr IndexTuple(const T& v) : head(v) { } + template struct IndexTuple + { + EIGEN_DEVICE_FUNC constexpr IndexTuple() : head() {} + EIGEN_DEVICE_FUNC constexpr IndexTuple(const T &v) : head(v) {} - constexpr static int count = 1; - T head; - typedef T Head; -}; + constexpr static int count = 1; + T head; + typedef T Head; + }; -template -struct IndexTupleExtractor; + template struct IndexTupleExtractor; -template -struct IndexTupleExtractor { + template struct IndexTupleExtractor + { - typedef typename IndexTupleExtractor::ValType ValType; + typedef typename IndexTupleExtractor::ValType ValType; - EIGEN_DEVICE_FUNC static constexpr ValType& get_val(IndexTuple& val) { - return IndexTupleExtractor::get_val(val.others); - } + EIGEN_DEVICE_FUNC static constexpr ValType &get_val(IndexTuple &val) + { + return IndexTupleExtractor::get_val(val.others); + } - EIGEN_DEVICE_FUNC static constexpr const ValType& get_val(const IndexTuple& val) { - return IndexTupleExtractor::get_val(val.others); - } - template - EIGEN_DEVICE_FUNC static void set_val(IndexTuple& val, V& new_val) { - IndexTupleExtractor::set_val(val.others, new_val); - } + EIGEN_DEVICE_FUNC static constexpr const ValType &get_val(const IndexTuple &val) + { + return IndexTupleExtractor::get_val(val.others); + } + template EIGEN_DEVICE_FUNC static void set_val(IndexTuple &val, V &new_val) + { + IndexTupleExtractor::set_val(val.others, new_val); + } + }; -}; + template struct IndexTupleExtractor<0, T, O...> + { -template - struct IndexTupleExtractor<0, T, O...> { + typedef T ValType; - typedef T ValType; + EIGEN_DEVICE_FUNC static constexpr ValType &get_val(IndexTuple &val) { return val.head; } + EIGEN_DEVICE_FUNC static constexpr const ValType &get_val(const IndexTuple &val) { return val.head; } + template EIGEN_DEVICE_FUNC static void set_val(IndexTuple &val, V &new_val) + { + val.head = new_val; + } + }; - EIGEN_DEVICE_FUNC static constexpr ValType& get_val(IndexTuple& val) { - return val.head; - } - EIGEN_DEVICE_FUNC static constexpr const ValType& get_val(const IndexTuple& val) { - return val.head; + + template + EIGEN_DEVICE_FUNC constexpr typename IndexTupleExtractor::ValType &array_get(IndexTuple &tuple) + { + return IndexTupleExtractor::get_val(tuple); } - template - EIGEN_DEVICE_FUNC static void set_val(IndexTuple& val, V& new_val) { - val.head = new_val; + template + EIGEN_DEVICE_FUNC constexpr const typename IndexTupleExtractor::ValType &array_get( + const IndexTuple &tuple) + { + return IndexTupleExtractor::get_val(tuple); } -}; - - - -template -EIGEN_DEVICE_FUNC constexpr typename IndexTupleExtractor::ValType& array_get(IndexTuple& tuple) { - return IndexTupleExtractor::get_val(tuple); -} -template -EIGEN_DEVICE_FUNC constexpr const typename IndexTupleExtractor::ValType& array_get(const IndexTuple& tuple) { - return IndexTupleExtractor::get_val(tuple); -} -template - struct array_size > { - static const size_t value = IndexTuple::count; -}; -template - struct array_size > { - static const size_t value = IndexTuple::count; -}; - - + template struct array_size> + { + static const size_t value = IndexTuple::count; + }; + template struct array_size> + { + static const size_t value = IndexTuple::count; + }; -template -struct tuple_coeff { - template - EIGEN_DEVICE_FUNC static constexpr ValueT get(const DenseIndex i, const IndexTuple& t) { - // return array_get(t) * (i == Idx) + tuple_coeff::get(i, t) * (i != Idx); - return (i == Idx ? array_get(t) : tuple_coeff::get(i, t)); - } - template - EIGEN_DEVICE_FUNC static void set(const DenseIndex i, IndexTuple& t, const ValueT& value) { - if (i == Idx) { - update_value(array_get(t), value); - } else { - tuple_coeff::set(i, t, value); + template struct tuple_coeff + { + template EIGEN_DEVICE_FUNC static constexpr ValueT get(const DenseIndex i, const IndexTuple &t) + { + // return array_get(t) * (i == Idx) + tuple_coeff::get(i, t) * (i != Idx); + return (i == Idx ? array_get(t) : tuple_coeff::get(i, t)); + } + template + EIGEN_DEVICE_FUNC static void set(const DenseIndex i, IndexTuple &t, const ValueT &value) + { + if (i == Idx) { + update_value(array_get(t), value); + } else { + tuple_coeff::set(i, t, value); + } } - } - - template - EIGEN_DEVICE_FUNC static constexpr bool value_known_statically(const DenseIndex i, const IndexTuple& t) { - return ((i == Idx) & is_compile_time_constant::ValType>::value) || - tuple_coeff::value_known_statically(i, t); - } - template - EIGEN_DEVICE_FUNC static constexpr bool values_up_to_known_statically(const IndexTuple& t) { - return is_compile_time_constant::ValType>::value && - tuple_coeff::values_up_to_known_statically(t); - } + template + EIGEN_DEVICE_FUNC static constexpr bool value_known_statically(const DenseIndex i, const IndexTuple &t) + { + return ((i == Idx) & is_compile_time_constant::ValType>::value) + || tuple_coeff::value_known_statically(i, t); + } - template - EIGEN_DEVICE_FUNC static constexpr bool values_up_to_statically_known_to_increase(const IndexTuple& t) { - return is_compile_time_constant::ValType>::value && - is_compile_time_constant::ValType>::value && - array_get(t) > array_get(t) && - tuple_coeff::values_up_to_statically_known_to_increase(t); - } -}; + template + EIGEN_DEVICE_FUNC static constexpr bool values_up_to_known_statically(const IndexTuple &t) + { + return is_compile_time_constant::ValType>::value + && tuple_coeff::values_up_to_known_statically(t); + } -template -struct tuple_coeff<0, ValueT> { - template - EIGEN_DEVICE_FUNC static constexpr ValueT get(const DenseIndex /*i*/, const IndexTuple& t) { - // eigen_assert (i == 0); // gcc fails to compile assertions in constexpr - return array_get<0>(t)/* * (i == 0)*/; - } - template - EIGEN_DEVICE_FUNC static void set(const DenseIndex i, IndexTuple& t, const ValueT value) { - eigen_assert (i == 0); - update_value(array_get<0>(t), value); - } - template - EIGEN_DEVICE_FUNC static constexpr bool value_known_statically(const DenseIndex i, const IndexTuple&) { - return is_compile_time_constant::ValType>::value & (i == 0); - } + template + EIGEN_DEVICE_FUNC static constexpr bool values_up_to_statically_known_to_increase(const IndexTuple &t) + { + return is_compile_time_constant::ValType>::value + && is_compile_time_constant::ValType>::value + && array_get(t) > array_get(t) + && tuple_coeff::values_up_to_statically_known_to_increase(t); + } + }; - template - EIGEN_DEVICE_FUNC static constexpr bool values_up_to_known_statically(const IndexTuple&) { - return is_compile_time_constant::ValType>::value; - } + template struct tuple_coeff<0, ValueT> + { + template + EIGEN_DEVICE_FUNC static constexpr ValueT get(const DenseIndex /*i*/, const IndexTuple &t) + { + // eigen_assert (i == 0); // gcc fails to compile assertions in constexpr + return array_get<0>(t) /* * (i == 0)*/; + } + template + EIGEN_DEVICE_FUNC static void set(const DenseIndex i, IndexTuple &t, const ValueT value) + { + eigen_assert(i == 0); + update_value(array_get<0>(t), value); + } + template + EIGEN_DEVICE_FUNC static constexpr bool value_known_statically(const DenseIndex i, const IndexTuple &) + { + return is_compile_time_constant::ValType>::value & (i == 0); + } - template - EIGEN_DEVICE_FUNC static constexpr bool values_up_to_statically_known_to_increase(const IndexTuple&) { - return true; - } -}; -} // namespace internal + template + EIGEN_DEVICE_FUNC static constexpr bool values_up_to_known_statically(const IndexTuple &) + { + return is_compile_time_constant::ValType>::value; + } + template + EIGEN_DEVICE_FUNC static constexpr bool values_up_to_statically_known_to_increase(const IndexTuple &) + { + return true; + } + }; +}// namespace internal -template -struct IndexList : internal::IndexTuple { - EIGEN_STRONG_INLINE EIGEN_DEVICE_FUNC constexpr DenseIndex operator[] (const DenseIndex i) const { - return internal::tuple_coeff >::value-1, DenseIndex>::get(i, *this); +template struct IndexList : internal::IndexTuple +{ + EIGEN_STRONG_INLINE EIGEN_DEVICE_FUNC constexpr DenseIndex operator[](const DenseIndex i) const + { + return internal::tuple_coeff>::value - 1, + DenseIndex>::get(i, *this); } - EIGEN_STRONG_INLINE EIGEN_DEVICE_FUNC constexpr DenseIndex get(const DenseIndex i) const { - return internal::tuple_coeff >::value-1, DenseIndex>::get(i, *this); + EIGEN_STRONG_INLINE EIGEN_DEVICE_FUNC constexpr DenseIndex get(const DenseIndex i) const + { + return internal::tuple_coeff>::value - 1, + DenseIndex>::get(i, *this); } - EIGEN_STRONG_INLINE EIGEN_DEVICE_FUNC void set(const DenseIndex i, const DenseIndex value) { - return internal::tuple_coeff >::value-1, DenseIndex>::set(i, *this, value); + EIGEN_STRONG_INLINE EIGEN_DEVICE_FUNC void set(const DenseIndex i, const DenseIndex value) + { + return internal::tuple_coeff>::value - 1, + DenseIndex>::set(i, *this, value); } - EIGEN_DEVICE_FUNC constexpr IndexList(const internal::IndexTuple& other) : internal::IndexTuple(other) { } - EIGEN_DEVICE_FUNC constexpr IndexList(FirstType& first, OtherTypes... other) : internal::IndexTuple(first, other...) { } - EIGEN_DEVICE_FUNC constexpr IndexList() : internal::IndexTuple() { } + EIGEN_DEVICE_FUNC constexpr IndexList(const internal::IndexTuple &other) + : internal::IndexTuple(other) + {} + EIGEN_DEVICE_FUNC constexpr IndexList(FirstType &first, OtherTypes... other) + : internal::IndexTuple(first, other...) + {} + EIGEN_DEVICE_FUNC constexpr IndexList() : internal::IndexTuple() {} - EIGEN_DEVICE_FUNC constexpr bool value_known_statically(const DenseIndex i) const { - return internal::tuple_coeff >::value-1, DenseIndex>::value_known_statically(i, *this); + EIGEN_DEVICE_FUNC constexpr bool value_known_statically(const DenseIndex i) const + { + return internal::tuple_coeff>::value - 1, + DenseIndex>::value_known_statically(i, *this); } - EIGEN_DEVICE_FUNC constexpr bool all_values_known_statically() const { - return internal::tuple_coeff >::value-1, DenseIndex>::values_up_to_known_statically(*this); + EIGEN_DEVICE_FUNC constexpr bool all_values_known_statically() const + { + return internal::tuple_coeff>::value - 1, + DenseIndex>::values_up_to_known_statically(*this); } - EIGEN_DEVICE_FUNC constexpr bool values_statically_known_to_increase() const { - return internal::tuple_coeff >::value-1, DenseIndex>::values_up_to_statically_known_to_increase(*this); + EIGEN_DEVICE_FUNC constexpr bool values_statically_known_to_increase() const + { + return internal::tuple_coeff>::value - 1, + DenseIndex>::values_up_to_statically_known_to_increase(*this); } }; template -constexpr IndexList make_index_list(FirstType val1, OtherTypes... other_vals) { +constexpr IndexList make_index_list(FirstType val1, OtherTypes... other_vals) +{ return IndexList(val1, other_vals...); } template -struct IndexPairList : internal::IndexTuple { - EIGEN_STRONG_INLINE EIGEN_DEVICE_FUNC constexpr IndexPair operator[] (const DenseIndex i) const { - return internal::tuple_coeff >::value-1, IndexPair>::get(i, *this); +struct IndexPairList : internal::IndexTuple +{ + EIGEN_STRONG_INLINE EIGEN_DEVICE_FUNC constexpr IndexPair operator[](const DenseIndex i) const + { + return internal::tuple_coeff>::value - 1, + IndexPair>::get(i, *this); } - EIGEN_STRONG_INLINE EIGEN_DEVICE_FUNC void set(const DenseIndex i, const IndexPair value) { - return internal::tuple_coeff>::value-1, IndexPair >::set(i, *this, value); + EIGEN_STRONG_INLINE EIGEN_DEVICE_FUNC void set(const DenseIndex i, const IndexPair value) + { + return internal::tuple_coeff>::value - 1, + IndexPair>::set(i, *this, value); } - EIGEN_DEVICE_FUNC constexpr IndexPairList(const internal::IndexTuple& other) : internal::IndexTuple(other) { } - EIGEN_DEVICE_FUNC constexpr IndexPairList() : internal::IndexTuple() { } + EIGEN_DEVICE_FUNC constexpr IndexPairList(const internal::IndexTuple &other) + : internal::IndexTuple(other) + {} + EIGEN_DEVICE_FUNC constexpr IndexPairList() : internal::IndexTuple() {} - EIGEN_DEVICE_FUNC constexpr bool value_known_statically(const DenseIndex i) const { - return internal::tuple_coeff >::value-1, DenseIndex>::value_known_statically(i, *this); + EIGEN_DEVICE_FUNC constexpr bool value_known_statically(const DenseIndex i) const + { + return internal::tuple_coeff>::value - 1, + DenseIndex>::value_known_statically(i, *this); } }; namespace internal { -template size_t array_prod(const IndexList& sizes) { - size_t result = 1; - for (int i = 0; i < array_size >::value; ++i) { - result *= sizes[i]; + template + size_t array_prod(const IndexList &sizes) + { + size_t result = 1; + for (int i = 0; i < array_size>::value; ++i) { result *= sizes[i]; } + return result; } - return result; -} - -template struct array_size > { - static const size_t value = array_size >::value; -}; -template struct array_size > { - static const size_t value = array_size >::value; -}; - -template struct array_size > { - static const size_t value = std::tuple_size >::value; -}; -template struct array_size > { - static const size_t value = std::tuple_size >::value; -}; -template EIGEN_DEVICE_FUNC constexpr DenseIndex array_get(IndexList& a) { - return IndexTupleExtractor::get_val(a); -} -template EIGEN_DEVICE_FUNC constexpr DenseIndex array_get(const IndexList& a) { - return IndexTupleExtractor::get_val(a); -} + template struct array_size> + { + static const size_t value = array_size>::value; + }; + template struct array_size> + { + static const size_t value = array_size>::value; + }; -template -struct index_known_statically_impl { - EIGEN_DEVICE_FUNC static constexpr bool run(const DenseIndex) { - return false; - } -}; + template struct array_size> + { + static const size_t value = std::tuple_size>::value; + }; + template struct array_size> + { + static const size_t value = std::tuple_size>::value; + }; -template -struct index_known_statically_impl > { - EIGEN_DEVICE_FUNC static constexpr bool run(const DenseIndex i) { - return IndexList().value_known_statically(i); + template + EIGEN_DEVICE_FUNC constexpr DenseIndex array_get(IndexList &a) + { + return IndexTupleExtractor::get_val(a); } -}; - -template -struct index_known_statically_impl > { - EIGEN_DEVICE_FUNC static constexpr bool run(const DenseIndex i) { - return IndexList().value_known_statically(i); + template + EIGEN_DEVICE_FUNC constexpr DenseIndex array_get(const IndexList &a) + { + return IndexTupleExtractor::get_val(a); } -}; + template struct index_known_statically_impl + { + EIGEN_DEVICE_FUNC static constexpr bool run(const DenseIndex) { return false; } + }; -template -struct all_indices_known_statically_impl { - static constexpr bool run() { - return false; - } -}; + template + struct index_known_statically_impl> + { + EIGEN_DEVICE_FUNC static constexpr bool run(const DenseIndex i) + { + return IndexList().value_known_statically(i); + } + }; -template -struct all_indices_known_statically_impl > { - EIGEN_DEVICE_FUNC static constexpr bool run() { - return IndexList().all_values_known_statically(); - } -}; + template + struct index_known_statically_impl> + { + EIGEN_DEVICE_FUNC static constexpr bool run(const DenseIndex i) + { + return IndexList().value_known_statically(i); + } + }; -template -struct all_indices_known_statically_impl > { - EIGEN_DEVICE_FUNC static constexpr bool run() { - return IndexList().all_values_known_statically(); - } -}; + template struct all_indices_known_statically_impl + { + static constexpr bool run() { return false; } + }; -template -struct indices_statically_known_to_increase_impl { - EIGEN_DEVICE_FUNC static constexpr bool run() { - return false; - } -}; + template + struct all_indices_known_statically_impl> + { + EIGEN_DEVICE_FUNC static constexpr bool run() + { + return IndexList().all_values_known_statically(); + } + }; -template - struct indices_statically_known_to_increase_impl > { - EIGEN_DEVICE_FUNC static constexpr bool run() { - return Eigen::IndexList().values_statically_known_to_increase(); - } -}; + template + struct all_indices_known_statically_impl> + { + EIGEN_DEVICE_FUNC static constexpr bool run() + { + return IndexList().all_values_known_statically(); + } + }; -template - struct indices_statically_known_to_increase_impl > { - EIGEN_DEVICE_FUNC static constexpr bool run() { - return Eigen::IndexList().values_statically_known_to_increase(); - } -}; + template struct indices_statically_known_to_increase_impl + { + EIGEN_DEVICE_FUNC static constexpr bool run() { return false; } + }; -template -struct index_statically_eq_impl { - EIGEN_DEVICE_FUNC static constexpr bool run(DenseIndex, DenseIndex) { - return false; - } -}; + template + struct indices_statically_known_to_increase_impl> + { + EIGEN_DEVICE_FUNC static constexpr bool run() + { + return Eigen::IndexList().values_statically_known_to_increase(); + } + }; -template -struct index_statically_eq_impl > { - EIGEN_DEVICE_FUNC static constexpr bool run(const DenseIndex i, const DenseIndex value) { - return IndexList().value_known_statically(i) & - (IndexList().get(i) == value); - } -}; + template + struct indices_statically_known_to_increase_impl> + { + EIGEN_DEVICE_FUNC static constexpr bool run() + { + return Eigen::IndexList().values_statically_known_to_increase(); + } + }; -template -struct index_statically_eq_impl > { - EIGEN_DEVICE_FUNC static constexpr bool run(const DenseIndex i, const DenseIndex value) { - return IndexList().value_known_statically(i) & - (IndexList().get(i) == value); - } -}; + template struct index_statically_eq_impl + { + EIGEN_DEVICE_FUNC static constexpr bool run(DenseIndex, DenseIndex) { return false; } + }; -template -struct index_statically_ne_impl { - EIGEN_DEVICE_FUNC static constexpr bool run(DenseIndex, DenseIndex) { - return false; - } -}; + template + struct index_statically_eq_impl> + { + EIGEN_DEVICE_FUNC static constexpr bool run(const DenseIndex i, const DenseIndex value) + { + return IndexList().value_known_statically(i) + & (IndexList().get(i) == value); + } + }; -template -struct index_statically_ne_impl > { - EIGEN_DEVICE_FUNC static constexpr bool run(const DenseIndex i, const DenseIndex value) { - return IndexList().value_known_statically(i) & - (IndexList().get(i) != value); - } -}; + template + struct index_statically_eq_impl> + { + EIGEN_DEVICE_FUNC static constexpr bool run(const DenseIndex i, const DenseIndex value) + { + return IndexList().value_known_statically(i) + & (IndexList().get(i) == value); + } + }; -template -struct index_statically_ne_impl > { - EIGEN_DEVICE_FUNC static constexpr bool run(const DenseIndex i, const DenseIndex value) { - return IndexList().value_known_statically(i) & - (IndexList().get(i) != value); - } -}; + template struct index_statically_ne_impl + { + EIGEN_DEVICE_FUNC static constexpr bool run(DenseIndex, DenseIndex) { return false; } + }; -template -struct index_statically_gt_impl { - EIGEN_DEVICE_FUNC static constexpr bool run(DenseIndex, DenseIndex) { - return false; - } -}; + template + struct index_statically_ne_impl> + { + EIGEN_DEVICE_FUNC static constexpr bool run(const DenseIndex i, const DenseIndex value) + { + return IndexList().value_known_statically(i) + & (IndexList().get(i) != value); + } + }; -template -struct index_statically_gt_impl > { - EIGEN_DEVICE_FUNC static constexpr bool run(const DenseIndex i, const DenseIndex value) { - return IndexList().value_known_statically(i) & - (IndexList().get(i) > value); - } -}; + template + struct index_statically_ne_impl> + { + EIGEN_DEVICE_FUNC static constexpr bool run(const DenseIndex i, const DenseIndex value) + { + return IndexList().value_known_statically(i) + & (IndexList().get(i) != value); + } + }; -template -struct index_statically_gt_impl > { - EIGEN_DEVICE_FUNC static constexpr bool run(const DenseIndex i, const DenseIndex value) { - return IndexList().value_known_statically(i) & - (IndexList().get(i) > value); - } -}; + template struct index_statically_gt_impl + { + EIGEN_DEVICE_FUNC static constexpr bool run(DenseIndex, DenseIndex) { return false; } + }; + template + struct index_statically_gt_impl> + { + EIGEN_DEVICE_FUNC static constexpr bool run(const DenseIndex i, const DenseIndex value) + { + return IndexList().value_known_statically(i) + & (IndexList().get(i) > value); + } + }; -template -struct index_statically_lt_impl { - EIGEN_DEVICE_FUNC static constexpr bool run(DenseIndex, DenseIndex) { - return false; - } -}; + template + struct index_statically_gt_impl> + { + EIGEN_DEVICE_FUNC static constexpr bool run(const DenseIndex i, const DenseIndex value) + { + return IndexList().value_known_statically(i) + & (IndexList().get(i) > value); + } + }; -template -struct index_statically_lt_impl > { - EIGEN_DEVICE_FUNC static constexpr bool run(const DenseIndex i, const DenseIndex value) { - return IndexList().value_known_statically(i) & - (IndexList().get(i) < value); - } -}; -template -struct index_statically_lt_impl > { - EIGEN_DEVICE_FUNC static constexpr bool run(const DenseIndex i, const DenseIndex value) { - return IndexList().value_known_statically(i) & - (IndexList().get(i) < value); - } -}; + template struct index_statically_lt_impl + { + EIGEN_DEVICE_FUNC static constexpr bool run(DenseIndex, DenseIndex) { return false; } + }; + template + struct index_statically_lt_impl> + { + EIGEN_DEVICE_FUNC static constexpr bool run(const DenseIndex i, const DenseIndex value) + { + return IndexList().value_known_statically(i) + & (IndexList().get(i) < value); + } + }; + template + struct index_statically_lt_impl> + { + EIGEN_DEVICE_FUNC static constexpr bool run(const DenseIndex i, const DenseIndex value) + { + return IndexList().value_known_statically(i) + & (IndexList().get(i) < value); + } + }; -template -struct index_pair_first_statically_eq_impl { - EIGEN_DEVICE_FUNC static constexpr bool run(DenseIndex, DenseIndex) { - return false; - } -}; -template -struct index_pair_first_statically_eq_impl > { - EIGEN_DEVICE_FUNC static constexpr bool run(const DenseIndex i, const DenseIndex value) { - return IndexPairList().value_known_statically(i) & - (IndexPairList().operator[](i).first == value); - } -}; + template struct index_pair_first_statically_eq_impl + { + EIGEN_DEVICE_FUNC static constexpr bool run(DenseIndex, DenseIndex) { return false; } + }; -template -struct index_pair_first_statically_eq_impl > { - EIGEN_DEVICE_FUNC static constexpr bool run(const DenseIndex i, const DenseIndex value) { - return IndexPairList().value_known_statically(i) & - (IndexPairList().operator[](i).first == value); - } -}; + template + struct index_pair_first_statically_eq_impl> + { + EIGEN_DEVICE_FUNC static constexpr bool run(const DenseIndex i, const DenseIndex value) + { + return IndexPairList().value_known_statically(i) + & (IndexPairList().operator[](i).first == value); + } + }; + template + struct index_pair_first_statically_eq_impl> + { + EIGEN_DEVICE_FUNC static constexpr bool run(const DenseIndex i, const DenseIndex value) + { + return IndexPairList().value_known_statically(i) + & (IndexPairList().operator[](i).first == value); + } + }; -template -struct index_pair_second_statically_eq_impl { - EIGEN_DEVICE_FUNC static constexpr bool run(DenseIndex, DenseIndex) { - return false; - } -}; + template struct index_pair_second_statically_eq_impl + { + EIGEN_DEVICE_FUNC static constexpr bool run(DenseIndex, DenseIndex) { return false; } + }; -template -struct index_pair_second_statically_eq_impl > { - EIGEN_DEVICE_FUNC static constexpr bool run(const DenseIndex i, const DenseIndex value) { - return IndexPairList().value_known_statically(i) & - (IndexPairList().operator[](i).second == value); - } -}; + template + struct index_pair_second_statically_eq_impl> + { + EIGEN_DEVICE_FUNC static constexpr bool run(const DenseIndex i, const DenseIndex value) + { + return IndexPairList().value_known_statically(i) + & (IndexPairList().operator[](i).second == value); + } + }; -template -struct index_pair_second_statically_eq_impl > { - EIGEN_DEVICE_FUNC static constexpr bool run(const DenseIndex i, const DenseIndex value) { - return IndexPairList().value_known_statically(i) & - (IndexPairList().operator[](i).second == value); - } -}; + template + struct index_pair_second_statically_eq_impl> + { + EIGEN_DEVICE_FUNC static constexpr bool run(const DenseIndex i, const DenseIndex value) + { + return IndexPairList().value_known_statically(i) + & (IndexPairList().operator[](i).second == value); + } + }; -} // end namespace internal -} // end namespace Eigen +}// end namespace internal +}// end namespace Eigen #else namespace Eigen { namespace internal { -template -struct index_known_statically_impl { - static EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE bool run(const DenseIndex) { - return false; - } -}; - -template -struct all_indices_known_statically_impl { - static EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE bool run() { - return false; - } -}; + template struct index_known_statically_impl + { + static EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE bool run(const DenseIndex) { return false; } + }; -template -struct indices_statically_known_to_increase_impl { - static EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE bool run() { - return false; - } -}; + template struct all_indices_known_statically_impl + { + static EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE bool run() { return false; } + }; -template -struct index_statically_eq_impl { - static EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE bool run(DenseIndex, DenseIndex) { - return false; - } -}; + template struct indices_statically_known_to_increase_impl + { + static EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE bool run() { return false; } + }; -template -struct index_statically_ne_impl { - static EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE bool run(DenseIndex, DenseIndex) { - return false; - } -}; + template struct index_statically_eq_impl + { + static EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE bool run(DenseIndex, DenseIndex) { return false; } + }; -template -struct index_statically_gt_impl { - static EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE bool run(DenseIndex, DenseIndex) { - return false; - } -}; + template struct index_statically_ne_impl + { + static EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE bool run(DenseIndex, DenseIndex) { return false; } + }; -template -struct index_statically_lt_impl { - static EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE bool run(DenseIndex, DenseIndex) { - return false; - } -}; + template struct index_statically_gt_impl + { + static EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE bool run(DenseIndex, DenseIndex) { return false; } + }; -template -struct index_pair_first_statically_eq_impl { - static EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE bool run(DenseIndex, DenseIndex) { - return false; - } -}; + template struct index_statically_lt_impl + { + static EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE bool run(DenseIndex, DenseIndex) { return false; } + }; -template -struct index_pair_second_statically_eq_impl { - static EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE bool run(DenseIndex, DenseIndex) { - return false; - } -}; + template struct index_pair_first_statically_eq_impl + { + static EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE bool run(DenseIndex, DenseIndex) { return false; } + }; + template struct index_pair_second_statically_eq_impl + { + static EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE bool run(DenseIndex, DenseIndex) { return false; } + }; -} // end namespace internal -} // end namespace Eigen +}// end namespace internal +}// end namespace Eigen #endif namespace Eigen { namespace internal { -template -static EIGEN_DEVICE_FUNC EIGEN_CONSTEXPR bool index_known_statically(DenseIndex i) { - return index_known_statically_impl::run(i); -} + template static EIGEN_DEVICE_FUNC EIGEN_CONSTEXPR bool index_known_statically(DenseIndex i) + { + return index_known_statically_impl::run(i); + } -template -static EIGEN_DEVICE_FUNC EIGEN_CONSTEXPR bool all_indices_known_statically() { - return all_indices_known_statically_impl::run(); -} + template static EIGEN_DEVICE_FUNC EIGEN_CONSTEXPR bool all_indices_known_statically() + { + return all_indices_known_statically_impl::run(); + } -template -static EIGEN_DEVICE_FUNC EIGEN_CONSTEXPR bool indices_statically_known_to_increase() { - return indices_statically_known_to_increase_impl::run(); -} + template static EIGEN_DEVICE_FUNC EIGEN_CONSTEXPR bool indices_statically_known_to_increase() + { + return indices_statically_known_to_increase_impl::run(); + } -template -static EIGEN_DEVICE_FUNC EIGEN_CONSTEXPR bool index_statically_eq(DenseIndex i, DenseIndex value) { - return index_statically_eq_impl::run(i, value); -} + template static EIGEN_DEVICE_FUNC EIGEN_CONSTEXPR bool index_statically_eq(DenseIndex i, DenseIndex value) + { + return index_statically_eq_impl::run(i, value); + } -template -static EIGEN_DEVICE_FUNC EIGEN_CONSTEXPR bool index_statically_ne(DenseIndex i, DenseIndex value) { - return index_statically_ne_impl::run(i, value); -} + template static EIGEN_DEVICE_FUNC EIGEN_CONSTEXPR bool index_statically_ne(DenseIndex i, DenseIndex value) + { + return index_statically_ne_impl::run(i, value); + } -template -static EIGEN_DEVICE_FUNC EIGEN_CONSTEXPR bool index_statically_gt(DenseIndex i, DenseIndex value) { - return index_statically_gt_impl::run(i, value); -} + template static EIGEN_DEVICE_FUNC EIGEN_CONSTEXPR bool index_statically_gt(DenseIndex i, DenseIndex value) + { + return index_statically_gt_impl::run(i, value); + } -template -static EIGEN_DEVICE_FUNC EIGEN_CONSTEXPR bool index_statically_lt(DenseIndex i, DenseIndex value) { - return index_statically_lt_impl::run(i, value); -} + template static EIGEN_DEVICE_FUNC EIGEN_CONSTEXPR bool index_statically_lt(DenseIndex i, DenseIndex value) + { + return index_statically_lt_impl::run(i, value); + } -template -static EIGEN_DEVICE_FUNC EIGEN_CONSTEXPR bool index_pair_first_statically_eq(DenseIndex i, DenseIndex value) { - return index_pair_first_statically_eq_impl::run(i, value); -} + template + static EIGEN_DEVICE_FUNC EIGEN_CONSTEXPR bool index_pair_first_statically_eq(DenseIndex i, DenseIndex value) + { + return index_pair_first_statically_eq_impl::run(i, value); + } -template -static EIGEN_DEVICE_FUNC EIGEN_CONSTEXPR bool index_pair_second_statically_eq(DenseIndex i, DenseIndex value) { - return index_pair_second_statically_eq_impl::run(i, value); -} + template + static EIGEN_DEVICE_FUNC EIGEN_CONSTEXPR bool index_pair_second_statically_eq(DenseIndex i, DenseIndex value) + { + return index_pair_second_statically_eq_impl::run(i, value); + } -} // end namespace internal -} // end namespace Eigen +}// end namespace internal +}// end namespace Eigen -#endif // EIGEN_CXX11_TENSOR_TENSOR_INDEX_LIST_H +#endif// EIGEN_CXX11_TENSOR_TENSOR_INDEX_LIST_H diff --git a/filmulator-gui/core/nlmeans/eigen/unsupported/Eigen/CXX11/src/Tensor/TensorInflation.h b/filmulator-gui/core/nlmeans/eigen/unsupported/Eigen/CXX11/src/Tensor/TensorInflation.h index f391fb9e..bff6f5f7 100644 --- a/filmulator-gui/core/nlmeans/eigen/unsupported/Eigen/CXX11/src/Tensor/TensorInflation.h +++ b/filmulator-gui/core/nlmeans/eigen/unsupported/Eigen/CXX11/src/Tensor/TensorInflation.h @@ -13,44 +13,43 @@ namespace Eigen { /** \class TensorInflation - * \ingroup CXX11_Tensor_Module - * - * \brief Tensor inflation class. - * - * - */ + * \ingroup CXX11_Tensor_Module + * + * \brief Tensor inflation class. + * + * + */ namespace internal { -template -struct traits > : public traits -{ - typedef typename XprType::Scalar Scalar; - typedef traits XprTraits; - typedef typename XprTraits::StorageKind StorageKind; - typedef typename XprTraits::Index Index; - typedef typename XprType::Nested Nested; - typedef typename remove_reference::type _Nested; - static const int NumDimensions = XprTraits::NumDimensions; - static const int Layout = XprTraits::Layout; -}; + template + struct traits> : public traits + { + typedef typename XprType::Scalar Scalar; + typedef traits XprTraits; + typedef typename XprTraits::StorageKind StorageKind; + typedef typename XprTraits::Index Index; + typedef typename XprType::Nested Nested; + typedef typename remove_reference::type _Nested; + static const int NumDimensions = XprTraits::NumDimensions; + static const int Layout = XprTraits::Layout; + }; -template -struct eval, Eigen::Dense> -{ - typedef const TensorInflationOp& type; -}; + template struct eval, Eigen::Dense> + { + typedef const TensorInflationOp &type; + }; -template -struct nested, 1, typename eval >::type> -{ - typedef TensorInflationOp type; -}; + template + struct nested, 1, typename eval>::type> + { + typedef TensorInflationOp type; + }; -} // end namespace internal +}// end namespace internal template class TensorInflationOp : public TensorBase, ReadOnlyAccessors> { - public: +public: typedef typename Eigen::internal::traits::Scalar Scalar; typedef typename Eigen::NumTraits::Real RealScalar; typedef typename XprType::CoeffReturnType CoeffReturnType; @@ -58,19 +57,19 @@ class TensorInflationOp : public TensorBase, typedef typename Eigen::internal::traits::StorageKind StorageKind; typedef typename Eigen::internal::traits::Index Index; - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TensorInflationOp(const XprType& expr, const Strides& strides) - : m_xpr(expr), m_strides(strides) {} + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TensorInflationOp(const XprType &expr, const Strides &strides) + : m_xpr(expr), m_strides(strides) + {} - EIGEN_DEVICE_FUNC - const Strides& strides() const { return m_strides; } + EIGEN_DEVICE_FUNC + const Strides &strides() const { return m_strides; } - EIGEN_DEVICE_FUNC - const typename internal::remove_all::type& - expression() const { return m_xpr; } + EIGEN_DEVICE_FUNC + const typename internal::remove_all::type &expression() const { return m_xpr; } - protected: - typename XprType::Nested m_xpr; - const Strides m_strides; +protected: + typename XprType::Nested m_xpr; + const Strides m_strides; }; // Eval as rvalue @@ -91,84 +90,71 @@ struct TensorEvaluator, Device> PacketAccess = TensorEvaluator::PacketAccess, BlockAccess = false, Layout = TensorEvaluator::Layout, - CoordAccess = false, // to be implemented + CoordAccess = false,// to be implemented RawAccess = false }; - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TensorEvaluator(const XprType& op, const Device& device) - : m_impl(op.expression(), device), m_strides(op.strides()) + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TensorEvaluator(const XprType &op, const Device &device) + : m_impl(op.expression(), device), m_strides(op.strides()) { m_dimensions = m_impl.dimensions(); // Expand each dimension to the inflated dimension. - for (int i = 0; i < NumDims; ++i) { - m_dimensions[i] = (m_dimensions[i] - 1) * op.strides()[i] + 1; - } + for (int i = 0; i < NumDims; ++i) { m_dimensions[i] = (m_dimensions[i] - 1) * op.strides()[i] + 1; } // Remember the strides for fast division. - for (int i = 0; i < NumDims; ++i) { - m_fastStrides[i] = internal::TensorIntDivisor(m_strides[i]); - } + for (int i = 0; i < NumDims; ++i) { m_fastStrides[i] = internal::TensorIntDivisor(m_strides[i]); } - const typename TensorEvaluator::Dimensions& input_dims = m_impl.dimensions(); + const typename TensorEvaluator::Dimensions &input_dims = m_impl.dimensions(); if (static_cast(Layout) == static_cast(ColMajor)) { m_outputStrides[0] = 1; m_inputStrides[0] = 1; for (int i = 1; i < NumDims; ++i) { - m_outputStrides[i] = m_outputStrides[i-1] * m_dimensions[i-1]; - m_inputStrides[i] = m_inputStrides[i-1] * input_dims[i-1]; + m_outputStrides[i] = m_outputStrides[i - 1] * m_dimensions[i - 1]; + m_inputStrides[i] = m_inputStrides[i - 1] * input_dims[i - 1]; } - } else { // RowMajor - m_outputStrides[NumDims-1] = 1; - m_inputStrides[NumDims-1] = 1; + } else {// RowMajor + m_outputStrides[NumDims - 1] = 1; + m_inputStrides[NumDims - 1] = 1; for (int i = NumDims - 2; i >= 0; --i) { - m_outputStrides[i] = m_outputStrides[i+1] * m_dimensions[i+1]; - m_inputStrides[i] = m_inputStrides[i+1] * input_dims[i+1]; + m_outputStrides[i] = m_outputStrides[i + 1] * m_dimensions[i + 1]; + m_inputStrides[i] = m_inputStrides[i + 1] * input_dims[i + 1]; } } } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const Dimensions& dimensions() const { return m_dimensions; } + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const Dimensions &dimensions() const { return m_dimensions; } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE bool evalSubExprsIfNeeded(Scalar* /*data*/) { + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE bool evalSubExprsIfNeeded(Scalar * /*data*/) + { m_impl.evalSubExprsIfNeeded(NULL); return true; } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void cleanup() { - m_impl.cleanup(); - } + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void cleanup() { m_impl.cleanup(); } // Computes the input index given the output index. Returns true if the output // index doesn't fall into a hole. - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE bool getInputIndex(Index index, Index* inputIndex) const + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE bool getInputIndex(Index index, Index *inputIndex) const { eigen_assert(index < dimensions().TotalSize()); *inputIndex = 0; if (static_cast(Layout) == static_cast(ColMajor)) { for (int i = NumDims - 1; i > 0; --i) { const Index idx = index / m_outputStrides[i]; - if (idx != idx / m_fastStrides[i] * m_strides[i]) { - return false; - } + if (idx != idx / m_fastStrides[i] * m_strides[i]) { return false; } *inputIndex += idx / m_strides[i] * m_inputStrides[i]; index -= idx * m_outputStrides[i]; } - if (index != index / m_fastStrides[0] * m_strides[0]) { - return false; - } + if (index != index / m_fastStrides[0] * m_strides[0]) { return false; } *inputIndex += index / m_strides[0]; return true; } else { for (int i = 0; i < NumDims - 1; ++i) { const Index idx = index / m_outputStrides[i]; - if (idx != idx / m_fastStrides[i] * m_strides[i]) { - return false; - } + if (idx != idx / m_fastStrides[i] * m_strides[i]) { return false; } *inputIndex += idx / m_strides[i] * m_inputStrides[i]; index -= idx * m_outputStrides[i]; } - if (index != index / m_fastStrides[NumDims-1] * m_strides[NumDims-1]) { - return false; - } + if (index != index / m_fastStrides[NumDims - 1] * m_strides[NumDims - 1]) { return false; } *inputIndex += index / m_strides[NumDims - 1]; } return true; @@ -178,44 +164,40 @@ struct TensorEvaluator, Device> { Index inputIndex = 0; if (getInputIndex(index, &inputIndex)) { - return m_impl.coeff(inputIndex); + return m_impl.coeff(inputIndex); } else { - return Scalar(0); + return Scalar(0); } } // TODO(yangke): optimize this function so that we can detect and produce // all-zero packets - template - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE PacketReturnType packet(Index index) const + template EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE PacketReturnType packet(Index index) const { EIGEN_STATIC_ASSERT((PacketSize > 1), YOU_MADE_A_PROGRAMMING_MISTAKE) - eigen_assert(index+PacketSize-1 < dimensions().TotalSize()); + eigen_assert(index + PacketSize - 1 < dimensions().TotalSize()); EIGEN_ALIGN_MAX typename internal::remove_const::type values[PacketSize]; - for (int i = 0; i < PacketSize; ++i) { - values[i] = coeff(index+i); - } + for (int i = 0; i < PacketSize; ++i) { values[i] = coeff(index + i); } PacketReturnType rslt = internal::pload(values); return rslt; } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TensorOpCost costPerCoeff(bool vectorized) const { - const double compute_cost = NumDims * (3 * TensorOpCost::DivCost() + - 3 * TensorOpCost::MulCost() + - 2 * TensorOpCost::AddCost()); + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TensorOpCost costPerCoeff(bool vectorized) const + { + const double compute_cost = + NumDims + * (3 * TensorOpCost::DivCost() + 3 * TensorOpCost::MulCost() + 2 * TensorOpCost::AddCost()); const double input_size = m_impl.dimensions().TotalSize(); const double output_size = m_dimensions.TotalSize(); - if (output_size == 0) - return TensorOpCost(); - return m_impl.costPerCoeff(vectorized) + - TensorOpCost(sizeof(CoeffReturnType) * input_size / output_size, 0, - compute_cost, vectorized, PacketSize); + if (output_size == 0) return TensorOpCost(); + return m_impl.costPerCoeff(vectorized) + + TensorOpCost(sizeof(CoeffReturnType) * input_size / output_size, 0, compute_cost, vectorized, PacketSize); } - EIGEN_DEVICE_FUNC Scalar* data() const { return NULL; } + EIGEN_DEVICE_FUNC Scalar *data() const { return NULL; } - protected: +protected: Dimensions m_dimensions; array m_outputStrides; array m_inputStrides; @@ -224,6 +206,6 @@ struct TensorEvaluator, Device> array, NumDims> m_fastStrides; }; -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_CXX11_TENSOR_TENSOR_INFLATION_H +#endif// EIGEN_CXX11_TENSOR_TENSOR_INFLATION_H diff --git a/filmulator-gui/core/nlmeans/eigen/unsupported/Eigen/CXX11/src/Tensor/TensorInitializer.h b/filmulator-gui/core/nlmeans/eigen/unsupported/Eigen/CXX11/src/Tensor/TensorInitializer.h index 33edc49e..4b4478a8 100644 --- a/filmulator-gui/core/nlmeans/eigen/unsupported/Eigen/CXX11/src/Tensor/TensorInitializer.h +++ b/filmulator-gui/core/nlmeans/eigen/unsupported/Eigen/CXX11/src/Tensor/TensorInitializer.h @@ -17,66 +17,69 @@ namespace Eigen { /** \class TensorInitializer - * \ingroup CXX11_Tensor_Module - * - * \brief Helper template to initialize Tensors from std::initializer_lists. - */ + * \ingroup CXX11_Tensor_Module + * + * \brief Helper template to initialize Tensors from std::initializer_lists. + */ namespace internal { -template -struct Initializer { - typedef std::initializer_list< - typename Initializer::InitList> InitList; - - static void run(TensorEvaluator& tensor, - Eigen::array::Index, traits::NumDimensions>* indices, - const InitList& vals) { - int i = 0; - for (auto v : vals) { - (*indices)[traits::NumDimensions - N] = i++; - Initializer::run(tensor, indices, v); + template struct Initializer + { + typedef std::initializer_list::InitList> InitList; + + static void run(TensorEvaluator &tensor, + Eigen::array::Index, traits::NumDimensions> *indices, + const InitList &vals) + { + int i = 0; + for (auto v : vals) { + (*indices)[traits::NumDimensions - N] = i++; + Initializer::run(tensor, indices, v); + } } - } -}; - -template -struct Initializer { - typedef std::initializer_list::Scalar> InitList; - - static void run(TensorEvaluator& tensor, - Eigen::array::Index, traits::NumDimensions>* indices, - const InitList& vals) { - int i = 0; - // There is likely a faster way to do that than iterating. - for (auto v : vals) { - (*indices)[traits::NumDimensions - 1] = i++; - tensor.coeffRef(*indices) = v; + }; + + template struct Initializer + { + typedef std::initializer_list::Scalar> InitList; + + static void run(TensorEvaluator &tensor, + Eigen::array::Index, traits::NumDimensions> *indices, + const InitList &vals) + { + int i = 0; + // There is likely a faster way to do that than iterating. + for (auto v : vals) { + (*indices)[traits::NumDimensions - 1] = i++; + tensor.coeffRef(*indices) = v; + } } - } -}; + }; -template -struct Initializer { - typedef typename traits::Scalar InitList; + template struct Initializer + { + typedef typename traits::Scalar InitList; - static void run(TensorEvaluator& tensor, - Eigen::array::Index, traits::NumDimensions>*, - const InitList& v) { - tensor.coeffRef(0) = v; - } -}; + static void run(TensorEvaluator &tensor, + Eigen::array::Index, traits::NumDimensions> *, + const InitList &v) + { + tensor.coeffRef(0) = v; + } + }; -template -void initialize_tensor(TensorEvaluator& tensor, - const typename Initializer::NumDimensions>::InitList& vals) { - Eigen::array::Index, traits::NumDimensions> indices; - Initializer::NumDimensions>::run(tensor, &indices, vals); -} + template + void initialize_tensor(TensorEvaluator &tensor, + const typename Initializer::NumDimensions>::InitList &vals) + { + Eigen::array::Index, traits::NumDimensions> indices; + Initializer::NumDimensions>::run(tensor, &indices, vals); + } -} // namespace internal -} // namespace Eigen +}// namespace internal +}// namespace Eigen -#endif // EIGEN_HAS_VARIADIC_TEMPLATES +#endif// EIGEN_HAS_VARIADIC_TEMPLATES -#endif // EIGEN_CXX11_TENSOR_TENSOR_INITIALIZER_H +#endif// EIGEN_CXX11_TENSOR_TENSOR_INITIALIZER_H diff --git a/filmulator-gui/core/nlmeans/eigen/unsupported/Eigen/CXX11/src/Tensor/TensorIntDiv.h b/filmulator-gui/core/nlmeans/eigen/unsupported/Eigen/CXX11/src/Tensor/TensorIntDiv.h index ede3939c..eab2918b 100644 --- a/filmulator-gui/core/nlmeans/eigen/unsupported/Eigen/CXX11/src/Tensor/TensorIntDiv.h +++ b/filmulator-gui/core/nlmeans/eigen/unsupported/Eigen/CXX11/src/Tensor/TensorIntDiv.h @@ -14,240 +14,256 @@ namespace Eigen { /** \internal - * - * \class TensorIntDiv - * \ingroup CXX11_Tensor_Module - * - * \brief Fast integer division by a constant. - * - * See the paper from Granlund and Montgomery for explanation. - * (at http://dx.doi.org/10.1145/773473.178249) - * - * \sa Tensor - */ + * + * \class TensorIntDiv + * \ingroup CXX11_Tensor_Module + * + * \brief Fast integer division by a constant. + * + * See the paper from Granlund and Montgomery for explanation. + * (at http://dx.doi.org/10.1145/773473.178249) + * + * \sa Tensor + */ namespace internal { -namespace { + namespace { - // Note: result is undefined if val == 0 - template - EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE - typename internal::enable_if::type count_leading_zeros(const T val) - { + // Note: result is undefined if val == 0 + template + EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE typename internal::enable_if::type count_leading_zeros( + const T val) + { #ifdef __CUDA_ARCH__ - return __clz(val); + return __clz(val); #elif EIGEN_COMP_MSVC - unsigned long index; - _BitScanReverse(&index, val); - return 31 - index; + unsigned long index; + _BitScanReverse(&index, val); + return 31 - index; #else - EIGEN_STATIC_ASSERT(sizeof(unsigned long long) == 8, YOU_MADE_A_PROGRAMMING_MISTAKE); - return __builtin_clz(static_cast(val)); + EIGEN_STATIC_ASSERT(sizeof(unsigned long long) == 8, YOU_MADE_A_PROGRAMMING_MISTAKE); + return __builtin_clz(static_cast(val)); #endif - } + } - template - EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE - typename internal::enable_if::type count_leading_zeros(const T val) - { + template + EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE typename internal::enable_if::type count_leading_zeros( + const T val) + { #ifdef __CUDA_ARCH__ - return __clzll(val); + return __clzll(val); #elif EIGEN_COMP_MSVC && EIGEN_ARCH_x86_64 - unsigned long index; - _BitScanReverse64(&index, val); - return 63 - index; + unsigned long index; + _BitScanReverse64(&index, val); + return 63 - index; #elif EIGEN_COMP_MSVC - // MSVC's _BitScanReverse64 is not available for 32bits builds. - unsigned int lo = (unsigned int)(val&0xffffffff); - unsigned int hi = (unsigned int)((val>>32)&0xffffffff); - int n; - if(hi==0) - n = 32 + count_leading_zeros(lo); - else - n = count_leading_zeros(hi); - return n; + // MSVC's _BitScanReverse64 is not available for 32bits builds. + unsigned int lo = (unsigned int)(val & 0xffffffff); + unsigned int hi = (unsigned int)((val >> 32) & 0xffffffff); + int n; + if (hi == 0) + n = 32 + count_leading_zeros(lo); + else + n = count_leading_zeros(hi); + return n; #else - EIGEN_STATIC_ASSERT(sizeof(unsigned long long) == 8, YOU_MADE_A_PROGRAMMING_MISTAKE); - return __builtin_clzll(static_cast(val)); + EIGEN_STATIC_ASSERT(sizeof(unsigned long long) == 8, YOU_MADE_A_PROGRAMMING_MISTAKE); + return __builtin_clzll(static_cast(val)); #endif - } + } - template - struct UnsignedTraits { - typedef typename conditional::type type; - }; + template struct UnsignedTraits + { + typedef typename conditional::type type; + }; - template - struct DividerTraits { - typedef typename UnsignedTraits::type type; - static const int N = sizeof(T) * 8; - }; + template struct DividerTraits + { + typedef typename UnsignedTraits::type type; + static const int N = sizeof(T) * 8; + }; - template - EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE uint32_t muluh(const uint32_t a, const T b) { + template EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE uint32_t muluh(const uint32_t a, const T b) + { #if defined(__CUDA_ARCH__) - return __umulhi(a, b); + return __umulhi(a, b); #else - return (static_cast(a) * b) >> 32; + return (static_cast(a) * b) >> 32; #endif - } + } - template - EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE uint64_t muluh(const uint64_t a, const T b) { + template EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE uint64_t muluh(const uint64_t a, const T b) + { #if defined(__CUDA_ARCH__) - return __umul64hi(a, b); + return __umul64hi(a, b); #elif defined(__SIZEOF_INT128__) - __uint128_t v = static_cast<__uint128_t>(a) * static_cast<__uint128_t>(b); - return static_cast(v >> 64); + __uint128_t v = static_cast<__uint128_t>(a) * static_cast<__uint128_t>(b); + return static_cast(v >> 64); #else - return (TensorUInt128, uint64_t>(a) * TensorUInt128, uint64_t>(b)).upper(); + return (TensorUInt128, uint64_t>(a) * TensorUInt128, uint64_t>(b)).upper(); #endif - } - - template - struct DividerHelper { - static EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE uint32_t computeMultiplier(const int log_div, const T divider) { - EIGEN_STATIC_ASSERT(N == 32, YOU_MADE_A_PROGRAMMING_MISTAKE); - return static_cast((static_cast(1) << (N+log_div)) / divider - (static_cast(1) << N) + 1); } - }; - template - struct DividerHelper<64, T> { - static EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE uint64_t computeMultiplier(const int log_div, const T divider) { + template struct DividerHelper + { + static EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE uint32_t computeMultiplier(const int log_div, const T divider) + { + EIGEN_STATIC_ASSERT(N == 32, YOU_MADE_A_PROGRAMMING_MISTAKE); + return static_cast( + (static_cast(1) << (N + log_div)) / divider - (static_cast(1) << N) + 1); + } + }; + + template struct DividerHelper<64, T> + { + static EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE uint64_t computeMultiplier(const int log_div, const T divider) + { #if defined(__SIZEOF_INT128__) && !defined(__CUDA_ARCH__) - return static_cast((static_cast<__uint128_t>(1) << (64+log_div)) / static_cast<__uint128_t>(divider) - (static_cast<__uint128_t>(1) << 64) + 1); + return static_cast((static_cast<__uint128_t>(1) << (64 + log_div)) / static_cast<__uint128_t>(divider) + - (static_cast<__uint128_t>(1) << 64) + 1); #else - const uint64_t shift = 1ULL << log_div; - TensorUInt128 result = TensorUInt128 >(shift, 0) / TensorUInt128, uint64_t>(divider) - - TensorUInt128, static_val<0> >(1, 0) - + TensorUInt128, static_val<1> >(1); - return static_cast(result); + const uint64_t shift = 1ULL << log_div; + TensorUInt128 result = + TensorUInt128>(shift, 0) / TensorUInt128, uint64_t>(divider) + - TensorUInt128, static_val<0>>(1, 0) + TensorUInt128, static_val<1>>(1); + return static_cast(result); #endif + } + }; + }// namespace + + + template struct TensorIntDivisor + { + public: + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TensorIntDivisor() + { + multiplier = 0; + shift1 = 0; + shift2 = 0; } - }; -} + // Must have 0 < divider < 2^31. This is relaxed to + // 0 < divider < 2^63 when using 64-bit indices on platforms that support + // the __uint128_t type. + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TensorIntDivisor(const T divider) + { + const int N = DividerTraits::N; + eigen_assert(static_cast::type>(divider) < NumTraits::highest() / 2); + eigen_assert(divider > 0); + + // fast ln2 + const int leading_zeros = count_leading_zeros(static_cast(divider)); + int log_div = N - leading_zeros; + // if divider is a power of two then log_div is 1 more than it should be. + if ((static_cast::type>(1) << (log_div - 1)) + == static_cast::type>(divider)) + log_div--; + + multiplier = DividerHelper::computeMultiplier(log_div, divider); + shift1 = log_div > 1 ? 1 : log_div; + shift2 = log_div > 1 ? log_div - 1 : 0; + } -template -struct TensorIntDivisor { - public: - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TensorIntDivisor() { - multiplier = 0; - shift1 = 0; - shift2 = 0; - } + // Must have 0 <= numerator. On platforms that dont support the __uint128_t + // type numerator should also be less than 2^32-1. + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE T divide(const T numerator) const + { + eigen_assert(static_cast::type>(numerator) < NumTraits::highest() / 2); + // eigen_assert(numerator >= 0); // this is implicitly asserted by the line above - // Must have 0 < divider < 2^31. This is relaxed to - // 0 < divider < 2^63 when using 64-bit indices on platforms that support - // the __uint128_t type. - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TensorIntDivisor(const T divider) { - const int N = DividerTraits::N; - eigen_assert(static_cast::type>(divider) < NumTraits::highest()/2); - eigen_assert(divider > 0); - - // fast ln2 - const int leading_zeros = count_leading_zeros(static_cast(divider)); - int log_div = N - leading_zeros; - // if divider is a power of two then log_div is 1 more than it should be. - if ((static_cast::type>(1) << (log_div-1)) == static_cast::type>(divider)) - log_div--; - - multiplier = DividerHelper::computeMultiplier(log_div, divider); - shift1 = log_div > 1 ? 1 : log_div; - shift2 = log_div > 1 ? log_div-1 : 0; - } + UnsignedType t1 = muluh(multiplier, numerator); + UnsignedType t = (static_cast(numerator) - t1) >> shift1; + return (t1 + t) >> shift2; + } - // Must have 0 <= numerator. On platforms that dont support the __uint128_t - // type numerator should also be less than 2^32-1. - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE T divide(const T numerator) const { - eigen_assert(static_cast::type>(numerator) < NumTraits::highest()/2); - //eigen_assert(numerator >= 0); // this is implicitly asserted by the line above + private: + typedef typename DividerTraits::type UnsignedType; + UnsignedType multiplier; + int32_t shift1; + int32_t shift2; + }; - UnsignedType t1 = muluh(multiplier, numerator); - UnsignedType t = (static_cast(numerator) - t1) >> shift1; - return (t1 + t) >> shift2; - } - private: - typedef typename DividerTraits::type UnsignedType; - UnsignedType multiplier; - int32_t shift1; - int32_t shift2; -}; - - -// Optimized version for signed 32 bit integers. -// Derived from Hacker's Delight. -// Only works for divisors strictly greater than one -template <> -class TensorIntDivisor { - public: - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TensorIntDivisor() { - magic = 0; - shift = 0; - } - // Must have 2 <= divider - EIGEN_DEVICE_FUNC TensorIntDivisor(int32_t divider) { - eigen_assert(divider >= 2); - calcMagic(divider); - } + // Optimized version for signed 32 bit integers. + // Derived from Hacker's Delight. + // Only works for divisors strictly greater than one + template<> class TensorIntDivisor + { + public: + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TensorIntDivisor() + { + magic = 0; + shift = 0; + } + // Must have 2 <= divider + EIGEN_DEVICE_FUNC TensorIntDivisor(int32_t divider) + { + eigen_assert(divider >= 2); + calcMagic(divider); + } - EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE int divide(const int32_t n) const { + EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE int divide(const int32_t n) const + { #ifdef __CUDA_ARCH__ - return (__umulhi(magic, n) >> shift); + return (__umulhi(magic, n) >> shift); #else - uint64_t v = static_cast(magic) * static_cast(n); - return (static_cast(v >> 32) >> shift); + uint64_t v = static_cast(magic) * static_cast(n); + return (static_cast(v >> 32) >> shift); #endif - } + } -private: - // Compute the magic numbers. See Hacker's Delight section 10 for an in - // depth explanation. - EIGEN_DEVICE_FUNC void calcMagic(int32_t d) { - const unsigned two31 = 0x80000000; // 2**31. - unsigned ad = d; - unsigned t = two31 + (ad >> 31); - unsigned anc = t - 1 - t%ad; // Absolute value of nc. - int p = 31; // Init. p. - unsigned q1 = two31/anc; // Init. q1 = 2**p/|nc|. - unsigned r1 = two31 - q1*anc; // Init. r1 = rem(2**p, |nc|). - unsigned q2 = two31/ad; // Init. q2 = 2**p/|d|. - unsigned r2 = two31 - q2*ad; // Init. r2 = rem(2**p, |d|). - unsigned delta = 0; - do { - p = p + 1; - q1 = 2*q1; // Update q1 = 2**p/|nc|. - r1 = 2*r1; // Update r1 = rem(2**p, |nc|). - if (r1 >= anc) { // (Must be an unsigned - q1 = q1 + 1; // comparison here). - r1 = r1 - anc;} - q2 = 2*q2; // Update q2 = 2**p/|d|. - r2 = 2*r2; // Update r2 = rem(2**p, |d|). - if (r2 >= ad) { // (Must be an unsigned - q2 = q2 + 1; // comparison here). - r2 = r2 - ad;} - delta = ad - r2; - } while (q1 < delta || (q1 == delta && r1 == 0)); - - magic = (unsigned)(q2 + 1); - shift = p - 32; - } + private: + // Compute the magic numbers. See Hacker's Delight section 10 for an in + // depth explanation. + EIGEN_DEVICE_FUNC void calcMagic(int32_t d) + { + const unsigned two31 = 0x80000000;// 2**31. + unsigned ad = d; + unsigned t = two31 + (ad >> 31); + unsigned anc = t - 1 - t % ad;// Absolute value of nc. + int p = 31;// Init. p. + unsigned q1 = two31 / anc;// Init. q1 = 2**p/|nc|. + unsigned r1 = two31 - q1 * anc;// Init. r1 = rem(2**p, |nc|). + unsigned q2 = two31 / ad;// Init. q2 = 2**p/|d|. + unsigned r2 = two31 - q2 * ad;// Init. r2 = rem(2**p, |d|). + unsigned delta = 0; + do { + p = p + 1; + q1 = 2 * q1;// Update q1 = 2**p/|nc|. + r1 = 2 * r1;// Update r1 = rem(2**p, |nc|). + if (r1 >= anc) {// (Must be an unsigned + q1 = q1 + 1;// comparison here). + r1 = r1 - anc; + } + q2 = 2 * q2;// Update q2 = 2**p/|d|. + r2 = 2 * r2;// Update r2 = rem(2**p, |d|). + if (r2 >= ad) {// (Must be an unsigned + q2 = q2 + 1;// comparison here). + r2 = r2 - ad; + } + delta = ad - r2; + } while (q1 < delta || (q1 == delta && r1 == 0)); + + magic = (unsigned)(q2 + 1); + shift = p - 32; + } - uint32_t magic; - int32_t shift; -}; + uint32_t magic; + int32_t shift; + }; -template -static EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE T operator / (const T& numerator, const TensorIntDivisor& divisor) { - return divisor.divide(numerator); -} + template + static EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE T operator/(const T &numerator, + const TensorIntDivisor &divisor) + { + return divisor.divide(numerator); + } -} // end namespace internal -} // end namespace Eigen +}// end namespace internal +}// end namespace Eigen -#endif // EIGEN_CXX11_TENSOR_TENSOR_INTDIV_H +#endif// EIGEN_CXX11_TENSOR_TENSOR_INTDIV_H diff --git a/filmulator-gui/core/nlmeans/eigen/unsupported/Eigen/CXX11/src/Tensor/TensorLayoutSwap.h b/filmulator-gui/core/nlmeans/eigen/unsupported/Eigen/CXX11/src/Tensor/TensorLayoutSwap.h index cd0109ef..87dbbb87 100644 --- a/filmulator-gui/core/nlmeans/eigen/unsupported/Eigen/CXX11/src/Tensor/TensorLayoutSwap.h +++ b/filmulator-gui/core/nlmeans/eigen/unsupported/Eigen/CXX11/src/Tensor/TensorLayoutSwap.h @@ -13,61 +13,57 @@ namespace Eigen { /** \class TensorLayoutSwap - * \ingroup CXX11_Tensor_Module - * - * \brief Swap the layout from col-major to row-major, or row-major - * to col-major, and invert the order of the dimensions. - * - * Beware: the dimensions are reversed by this operation. If you want to - * preserve the ordering of the dimensions, you need to combine this - * operation with a shuffle. - * - * \example: - * Tensor input(2, 4); - * Tensor output = input.swap_layout(); - * eigen_assert(output.dimension(0) == 4); - * eigen_assert(output.dimension(1) == 2); - * - * array shuffle(1, 0); - * output = input.swap_layout().shuffle(shuffle); - * eigen_assert(output.dimension(0) == 2); - * eigen_assert(output.dimension(1) == 4); - * - */ + * \ingroup CXX11_Tensor_Module + * + * \brief Swap the layout from col-major to row-major, or row-major + * to col-major, and invert the order of the dimensions. + * + * Beware: the dimensions are reversed by this operation. If you want to + * preserve the ordering of the dimensions, you need to combine this + * operation with a shuffle. + * + * \example: + * Tensor input(2, 4); + * Tensor output = input.swap_layout(); + * eigen_assert(output.dimension(0) == 4); + * eigen_assert(output.dimension(1) == 2); + * + * array shuffle(1, 0); + * output = input.swap_layout().shuffle(shuffle); + * eigen_assert(output.dimension(0) == 2); + * eigen_assert(output.dimension(1) == 4); + * + */ namespace internal { -template -struct traits > : public traits -{ - typedef typename XprType::Scalar Scalar; - typedef traits XprTraits; - typedef typename XprTraits::StorageKind StorageKind; - typedef typename XprTraits::Index Index; - typedef typename XprType::Nested Nested; - typedef typename remove_reference::type _Nested; - static const int NumDimensions = traits::NumDimensions; - static const int Layout = (traits::Layout == ColMajor) ? RowMajor : ColMajor; -}; - -template -struct eval, Eigen::Dense> -{ - typedef const TensorLayoutSwapOp& type; -}; + template struct traits> : public traits + { + typedef typename XprType::Scalar Scalar; + typedef traits XprTraits; + typedef typename XprTraits::StorageKind StorageKind; + typedef typename XprTraits::Index Index; + typedef typename XprType::Nested Nested; + typedef typename remove_reference::type _Nested; + static const int NumDimensions = traits::NumDimensions; + static const int Layout = (traits::Layout == ColMajor) ? RowMajor : ColMajor; + }; -template -struct nested, 1, typename eval >::type> -{ - typedef TensorLayoutSwapOp type; -}; + template struct eval, Eigen::Dense> + { + typedef const TensorLayoutSwapOp &type; + }; -} // end namespace internal + template + struct nested, 1, typename eval>::type> + { + typedef TensorLayoutSwapOp type; + }; +}// end namespace internal -template -class TensorLayoutSwapOp : public TensorBase, WriteAccessors> +template class TensorLayoutSwapOp : public TensorBase, WriteAccessors> { - public: +public: typedef typename Eigen::internal::traits::Scalar Scalar; typedef typename Eigen::NumTraits::Real RealScalar; typedef typename internal::remove_const::type CoeffReturnType; @@ -75,40 +71,36 @@ class TensorLayoutSwapOp : public TensorBase, WriteA typedef typename Eigen::internal::traits::StorageKind StorageKind; typedef typename Eigen::internal::traits::Index Index; - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TensorLayoutSwapOp(const XprType& expr) - : m_xpr(expr) {} - - EIGEN_DEVICE_FUNC - const typename internal::remove_all::type& - expression() const { return m_xpr; } - - EIGEN_DEVICE_FUNC - EIGEN_STRONG_INLINE TensorLayoutSwapOp& operator = (const TensorLayoutSwapOp& other) - { - typedef TensorAssignOp Assign; - Assign assign(*this, other); - internal::TensorExecutor::run(assign, DefaultDevice()); - return *this; - } - - template - EIGEN_DEVICE_FUNC - EIGEN_STRONG_INLINE TensorLayoutSwapOp& operator = (const OtherDerived& other) - { - typedef TensorAssignOp Assign; - Assign assign(*this, other); - internal::TensorExecutor::run(assign, DefaultDevice()); - return *this; - } - - protected: - typename XprType::Nested m_xpr; + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TensorLayoutSwapOp(const XprType &expr) : m_xpr(expr) {} + + EIGEN_DEVICE_FUNC + const typename internal::remove_all::type &expression() const { return m_xpr; } + + EIGEN_DEVICE_FUNC + EIGEN_STRONG_INLINE TensorLayoutSwapOp &operator=(const TensorLayoutSwapOp &other) + { + typedef TensorAssignOp Assign; + Assign assign(*this, other); + internal::TensorExecutor::run(assign, DefaultDevice()); + return *this; + } + + template + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TensorLayoutSwapOp &operator=(const OtherDerived &other) + { + typedef TensorAssignOp Assign; + Assign assign(*this, other); + internal::TensorExecutor::run(assign, DefaultDevice()); + return *this; + } + +protected: + typename XprType::Nested m_xpr; }; // Eval as rvalue -template -struct TensorEvaluator, Device> +template struct TensorEvaluator, Device> { typedef TensorLayoutSwapOp XprType; typedef typename XprType::Index Index; @@ -118,52 +110,47 @@ struct TensorEvaluator, Device> enum { IsAligned = TensorEvaluator::IsAligned, PacketAccess = TensorEvaluator::PacketAccess, - Layout = (static_cast(TensorEvaluator::Layout) == static_cast(ColMajor)) ? RowMajor : ColMajor, - CoordAccess = false, // to be implemented + Layout = + (static_cast(TensorEvaluator::Layout) == static_cast(ColMajor)) ? RowMajor : ColMajor, + CoordAccess = false,// to be implemented RawAccess = TensorEvaluator::RawAccess }; - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TensorEvaluator(const XprType& op, const Device& device) - : m_impl(op.expression(), device) + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TensorEvaluator(const XprType &op, const Device &device) + : m_impl(op.expression(), device) { - for(int i = 0; i < NumDims; ++i) { - m_dimensions[i] = m_impl.dimensions()[NumDims-1-i]; - } + for (int i = 0; i < NumDims; ++i) { m_dimensions[i] = m_impl.dimensions()[NumDims - 1 - i]; } } typedef typename XprType::Scalar Scalar; typedef typename XprType::CoeffReturnType CoeffReturnType; typedef typename PacketType::type PacketReturnType; - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const Dimensions& dimensions() const { return m_dimensions; } + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const Dimensions &dimensions() const { return m_dimensions; } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE bool evalSubExprsIfNeeded(CoeffReturnType* data) { + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE bool evalSubExprsIfNeeded(CoeffReturnType *data) + { return m_impl.evalSubExprsIfNeeded(data); } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void cleanup() { - m_impl.cleanup(); - } + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void cleanup() { m_impl.cleanup(); } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE CoeffReturnType coeff(Index index) const - { - return m_impl.coeff(index); - } + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE CoeffReturnType coeff(Index index) const { return m_impl.coeff(index); } - template - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE PacketReturnType packet(Index index) const + template EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE PacketReturnType packet(Index index) const { return m_impl.template packet(index); } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TensorOpCost costPerCoeff(bool vectorized) const { + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TensorOpCost costPerCoeff(bool vectorized) const + { return m_impl.costPerCoeff(vectorized); } - EIGEN_DEVICE_FUNC Scalar* data() const { return m_impl.data(); } + EIGEN_DEVICE_FUNC Scalar *data() const { return m_impl.data(); } - const TensorEvaluator& impl() const { return m_impl; } + const TensorEvaluator &impl() const { return m_impl; } - protected: +protected: TensorEvaluator m_impl; Dimensions m_dimensions; }; @@ -171,7 +158,7 @@ struct TensorEvaluator, Device> // Eval as lvalue template - struct TensorEvaluator, Device> +struct TensorEvaluator, Device> : public TensorEvaluator, Device> { typedef TensorEvaluator, Device> Base; @@ -180,30 +167,25 @@ template enum { IsAligned = TensorEvaluator::IsAligned, PacketAccess = TensorEvaluator::PacketAccess, - Layout = (static_cast(TensorEvaluator::Layout) == static_cast(ColMajor)) ? RowMajor : ColMajor, - CoordAccess = false // to be implemented + Layout = + (static_cast(TensorEvaluator::Layout) == static_cast(ColMajor)) ? RowMajor : ColMajor, + CoordAccess = false// to be implemented }; - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TensorEvaluator(const XprType& op, const Device& device) - : Base(op, device) - { } + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TensorEvaluator(const XprType &op, const Device &device) : Base(op, device) {} typedef typename XprType::Index Index; typedef typename XprType::Scalar Scalar; typedef typename XprType::CoeffReturnType CoeffReturnType; typedef typename PacketType::type PacketReturnType; - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE CoeffReturnType& coeffRef(Index index) - { - return this->m_impl.coeffRef(index); - } - template EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE - void writePacket(Index index, const PacketReturnType& x) + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE CoeffReturnType &coeffRef(Index index) { return this->m_impl.coeffRef(index); } + template EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void writePacket(Index index, const PacketReturnType &x) { this->m_impl.template writePacket(index, x); } }; -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_CXX11_TENSOR_TENSOR_LAYOUT_SWAP_H +#endif// EIGEN_CXX11_TENSOR_TENSOR_LAYOUT_SWAP_H diff --git a/filmulator-gui/core/nlmeans/eigen/unsupported/Eigen/CXX11/src/Tensor/TensorMacros.h b/filmulator-gui/core/nlmeans/eigen/unsupported/Eigen/CXX11/src/Tensor/TensorMacros.h index ee0078bb..a99efbfb 100644 --- a/filmulator-gui/core/nlmeans/eigen/unsupported/Eigen/CXX11/src/Tensor/TensorMacros.h +++ b/filmulator-gui/core/nlmeans/eigen/unsupported/Eigen/CXX11/src/Tensor/TensorMacros.h @@ -29,19 +29,18 @@ // SFINAE requires variadic templates #ifndef __CUDACC__ #if EIGEN_HAS_VARIADIC_TEMPLATES - // SFINAE doesn't work for gcc <= 4.7 - #ifdef EIGEN_COMP_GNUC - #if EIGEN_GNUC_AT_LEAST(4,8) - #define EIGEN_HAS_SFINAE - #endif - #else - #define EIGEN_HAS_SFINAE - #endif +// SFINAE doesn't work for gcc <= 4.7 +#ifdef EIGEN_COMP_GNUC +#if EIGEN_GNUC_AT_LEAST(4, 8) +#define EIGEN_HAS_SFINAE +#endif +#else +#define EIGEN_HAS_SFINAE +#endif #endif #endif -#define EIGEN_SFINAE_ENABLE_IF( __condition__ ) \ - typename internal::enable_if< ( __condition__ ) , int >::type = 0 +#define EIGEN_SFINAE_ENABLE_IF(__condition__) typename internal::enable_if<(__condition__), int>::type = 0 #if EIGEN_HAS_CONSTEXPR diff --git a/filmulator-gui/core/nlmeans/eigen/unsupported/Eigen/CXX11/src/Tensor/TensorMap.h b/filmulator-gui/core/nlmeans/eigen/unsupported/Eigen/CXX11/src/Tensor/TensorMap.h index e4fc86a4..1a4cc9fe 100644 --- a/filmulator-gui/core/nlmeans/eigen/unsupported/Eigen/CXX11/src/Tensor/TensorMap.h +++ b/filmulator-gui/core/nlmeans/eigen/unsupported/Eigen/CXX11/src/Tensor/TensorMap.h @@ -15,309 +15,332 @@ namespace Eigen { // FIXME use proper doxygen documentation (e.g. \tparam MakePointer_) /** \class TensorMap - * \ingroup CXX11_Tensor_Module - * - * \brief A tensor expression mapping an existing array of data. - * - */ + * \ingroup CXX11_Tensor_Module + * + * \brief A tensor expression mapping an existing array of data. + * + */ /// `template class MakePointer_` is added to convert the host pointer to the device pointer. /// It is added due to the fact that for our device compiler `T*` is not allowed. /// If we wanted to use the same Evaluator functions we have to convert that type to our pointer `T`. /// This is done through our `MakePointer_` class. By default the Type in the `MakePointer_` is `T*` . /// Therefore, by adding the default value, we managed to convert the type and it does not break any /// existing code as its default value is `T*`. -template class MakePointer_> class TensorMap : public TensorBase > +template class MakePointer_> +class TensorMap : public TensorBase> { - public: - typedef TensorMap Self; - typedef typename PlainObjectType::Base Base; - typedef typename Eigen::internal::nested::type Nested; - typedef typename internal::traits::StorageKind StorageKind; - typedef typename internal::traits::Index Index; - typedef typename internal::traits::Scalar Scalar; - typedef typename NumTraits::Real RealScalar; - typedef typename Base::CoeffReturnType CoeffReturnType; +public: + typedef TensorMap Self; + typedef typename PlainObjectType::Base Base; + typedef typename Eigen::internal::nested::type Nested; + typedef typename internal::traits::StorageKind StorageKind; + typedef typename internal::traits::Index Index; + typedef typename internal::traits::Scalar Scalar; + typedef typename NumTraits::Real RealScalar; + typedef typename Base::CoeffReturnType CoeffReturnType; /* typedef typename internal::conditional< bool(internal::is_lvalue::value), Scalar *, const Scalar *>::type PointerType;*/ - typedef typename MakePointer_::Type PointerType; - typedef PointerType PointerArgType; + typedef typename MakePointer_::Type PointerType; + typedef PointerType PointerArgType; - static const int Options = Options_; + static const int Options = Options_; - static const Index NumIndices = PlainObjectType::NumIndices; - typedef typename PlainObjectType::Dimensions Dimensions; + static const Index NumIndices = PlainObjectType::NumIndices; + typedef typename PlainObjectType::Dimensions Dimensions; - enum { - IsAligned = ((int(Options_)&Aligned)==Aligned), - Layout = PlainObjectType::Layout, - CoordAccess = true, - RawAccess = true - }; + enum { + IsAligned = ((int(Options_) & Aligned) == Aligned), + Layout = PlainObjectType::Layout, + CoordAccess = true, + RawAccess = true + }; - EIGEN_DEVICE_FUNC - EIGEN_STRONG_INLINE TensorMap(PointerArgType dataPtr) : m_data(dataPtr), m_dimensions() { - // The number of dimensions used to construct a tensor must be equal to the rank of the tensor. - EIGEN_STATIC_ASSERT((0 == NumIndices || NumIndices == Dynamic), YOU_MADE_A_PROGRAMMING_MISTAKE) - } + EIGEN_DEVICE_FUNC + EIGEN_STRONG_INLINE TensorMap(PointerArgType dataPtr) : m_data(dataPtr), m_dimensions() + { + // The number of dimensions used to construct a tensor must be equal to the rank of the tensor. + EIGEN_STATIC_ASSERT((0 == NumIndices || NumIndices == Dynamic), YOU_MADE_A_PROGRAMMING_MISTAKE) + } #if EIGEN_HAS_VARIADIC_TEMPLATES - template EIGEN_DEVICE_FUNC - EIGEN_STRONG_INLINE TensorMap(PointerArgType dataPtr, Index firstDimension, IndexTypes... otherDimensions) : m_data(dataPtr), m_dimensions(firstDimension, otherDimensions...) { - // The number of dimensions used to construct a tensor must be equal to the rank of the tensor. - EIGEN_STATIC_ASSERT((sizeof...(otherDimensions) + 1 == NumIndices || NumIndices == Dynamic), YOU_MADE_A_PROGRAMMING_MISTAKE) - } + template + EIGEN_DEVICE_FUNC + EIGEN_STRONG_INLINE TensorMap(PointerArgType dataPtr, Index firstDimension, IndexTypes... otherDimensions) + : m_data(dataPtr), + m_dimensions(firstDimension, otherDimensions...){ + // The number of dimensions used to construct a tensor must be equal to the rank of the tensor. + EIGEN_STATIC_ASSERT((sizeof...(otherDimensions) + 1 == NumIndices || NumIndices == Dynamic), + YOU_MADE_A_PROGRAMMING_MISTAKE) + } #else - EIGEN_DEVICE_FUNC - EIGEN_STRONG_INLINE TensorMap(PointerArgType dataPtr, Index firstDimension) : m_data(dataPtr), m_dimensions(firstDimension) { - // The number of dimensions used to construct a tensor must be equal to the rank of the tensor. - EIGEN_STATIC_ASSERT((1 == NumIndices || NumIndices == Dynamic), YOU_MADE_A_PROGRAMMING_MISTAKE) - } - EIGEN_DEVICE_FUNC - EIGEN_STRONG_INLINE TensorMap(PointerArgType dataPtr, Index dim1, Index dim2) : m_data(dataPtr), m_dimensions(dim1, dim2) { - EIGEN_STATIC_ASSERT(2 == NumIndices || NumIndices == Dynamic, YOU_MADE_A_PROGRAMMING_MISTAKE) - } - EIGEN_DEVICE_FUNC - EIGEN_STRONG_INLINE TensorMap(PointerArgType dataPtr, Index dim1, Index dim2, Index dim3) : m_data(dataPtr), m_dimensions(dim1, dim2, dim3) { - EIGEN_STATIC_ASSERT(3 == NumIndices || NumIndices == Dynamic, YOU_MADE_A_PROGRAMMING_MISTAKE) - } - EIGEN_DEVICE_FUNC - EIGEN_STRONG_INLINE TensorMap(PointerArgType dataPtr, Index dim1, Index dim2, Index dim3, Index dim4) : m_data(dataPtr), m_dimensions(dim1, dim2, dim3, dim4) { - EIGEN_STATIC_ASSERT(4 == NumIndices || NumIndices == Dynamic, YOU_MADE_A_PROGRAMMING_MISTAKE) - } - EIGEN_DEVICE_FUNC - EIGEN_STRONG_INLINE TensorMap(PointerArgType dataPtr, Index dim1, Index dim2, Index dim3, Index dim4, Index dim5) : m_data(dataPtr), m_dimensions(dim1, dim2, dim3, dim4, dim5) { - EIGEN_STATIC_ASSERT(5 == NumIndices || NumIndices == Dynamic, YOU_MADE_A_PROGRAMMING_MISTAKE) - } + EIGEN_DEVICE_FUNC + EIGEN_STRONG_INLINE TensorMap(PointerArgType dataPtr, Index firstDimension) + : m_data(dataPtr), + m_dimensions(firstDimension){ + // The number of dimensions used to construct a tensor must be equal to the rank of the tensor. + EIGEN_STATIC_ASSERT((1 == NumIndices || NumIndices == Dynamic), YOU_MADE_A_PROGRAMMING_MISTAKE) + } EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TensorMap(PointerArgType dataPtr, Index dim1, Index dim2) + : m_data(dataPtr), m_dimensions(dim1, dim2){ EIGEN_STATIC_ASSERT(2 == NumIndices || NumIndices == Dynamic, + YOU_MADE_A_PROGRAMMING_MISTAKE) } EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE + TensorMap(PointerArgType dataPtr, Index dim1, Index dim2, Index dim3) + : m_data(dataPtr), m_dimensions(dim1, dim2, dim3){ EIGEN_STATIC_ASSERT(3 == NumIndices || NumIndices == Dynamic, + YOU_MADE_A_PROGRAMMING_MISTAKE) } EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE + TensorMap(PointerArgType dataPtr, Index dim1, Index dim2, Index dim3, Index dim4) + : m_data(dataPtr), + m_dimensions(dim1, dim2, dim3, dim4){ EIGEN_STATIC_ASSERT(4 == NumIndices || NumIndices == Dynamic, + YOU_MADE_A_PROGRAMMING_MISTAKE) } EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE + TensorMap(PointerArgType dataPtr, Index dim1, Index dim2, Index dim3, Index dim4, Index dim5) + : m_data(dataPtr), + m_dimensions(dim1, dim2, dim3, dim4, dim5){ EIGEN_STATIC_ASSERT(5 == NumIndices || NumIndices == Dynamic, + YOU_MADE_A_PROGRAMMING_MISTAKE) } #endif - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TensorMap(PointerArgType dataPtr, const array& dimensions) - : m_data(dataPtr), m_dimensions(dimensions) - { } + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE + TensorMap(PointerArgType dataPtr, const array &dimensions) + : m_data(dataPtr), m_dimensions(dimensions) + {} - template - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TensorMap(PointerArgType dataPtr, const Dimensions& dimensions) - : m_data(dataPtr), m_dimensions(dimensions) - { } + template + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TensorMap(PointerArgType dataPtr, const Dimensions &dimensions) + : m_data(dataPtr), m_dimensions(dimensions) + {} - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TensorMap(PlainObjectType& tensor) - : m_data(tensor.data()), m_dimensions(tensor.dimensions()) - { } + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TensorMap(PlainObjectType &tensor) + : m_data(tensor.data()), m_dimensions(tensor.dimensions()) + {} - EIGEN_DEVICE_FUNC - EIGEN_STRONG_INLINE Index rank() const { return m_dimensions.rank(); } - EIGEN_DEVICE_FUNC - EIGEN_STRONG_INLINE Index dimension(Index n) const { return m_dimensions[n]; } - EIGEN_DEVICE_FUNC - EIGEN_STRONG_INLINE const Dimensions& dimensions() const { return m_dimensions; } - EIGEN_DEVICE_FUNC - EIGEN_STRONG_INLINE Index size() const { return m_dimensions.TotalSize(); } - EIGEN_DEVICE_FUNC - EIGEN_STRONG_INLINE PointerType data() { return m_data; } - EIGEN_DEVICE_FUNC - EIGEN_STRONG_INLINE const PointerType data() const { return m_data; } + EIGEN_DEVICE_FUNC + EIGEN_STRONG_INLINE Index rank() const { return m_dimensions.rank(); } + EIGEN_DEVICE_FUNC + EIGEN_STRONG_INLINE Index dimension(Index n) const { return m_dimensions[n]; } + EIGEN_DEVICE_FUNC + EIGEN_STRONG_INLINE const Dimensions &dimensions() const { return m_dimensions; } + EIGEN_DEVICE_FUNC + EIGEN_STRONG_INLINE Index size() const { return m_dimensions.TotalSize(); } + EIGEN_DEVICE_FUNC + EIGEN_STRONG_INLINE PointerType data() { return m_data; } + EIGEN_DEVICE_FUNC + EIGEN_STRONG_INLINE const PointerType data() const { return m_data; } - EIGEN_DEVICE_FUNC - EIGEN_STRONG_INLINE const Scalar& operator()(const array& indices) const - { - // eigen_assert(checkIndexRange(indices)); - if (PlainObjectType::Options&RowMajor) { - const Index index = m_dimensions.IndexOfRowMajor(indices); - return m_data[index]; - } else { - const Index index = m_dimensions.IndexOfColMajor(indices); - return m_data[index]; - } + EIGEN_DEVICE_FUNC + EIGEN_STRONG_INLINE const Scalar &operator()(const array &indices) const + { + // eigen_assert(checkIndexRange(indices)); + if (PlainObjectType::Options & RowMajor) { + const Index index = m_dimensions.IndexOfRowMajor(indices); + return m_data[index]; + } else { + const Index index = m_dimensions.IndexOfColMajor(indices); + return m_data[index]; } + } - EIGEN_DEVICE_FUNC - EIGEN_STRONG_INLINE const Scalar& operator()() const - { - EIGEN_STATIC_ASSERT(NumIndices == 0, YOU_MADE_A_PROGRAMMING_MISTAKE) - return m_data[0]; - } + EIGEN_DEVICE_FUNC + EIGEN_STRONG_INLINE const Scalar &operator()() const + { + EIGEN_STATIC_ASSERT(NumIndices == 0, YOU_MADE_A_PROGRAMMING_MISTAKE) + return m_data[0]; + } - EIGEN_DEVICE_FUNC - EIGEN_STRONG_INLINE const Scalar& operator()(Index index) const - { - eigen_internal_assert(index >= 0 && index < size()); - return m_data[index]; - } + EIGEN_DEVICE_FUNC + EIGEN_STRONG_INLINE const Scalar &operator()(Index index) const + { + eigen_internal_assert(index >= 0 && index < size()); + return m_data[index]; + } #if EIGEN_HAS_VARIADIC_TEMPLATES - template EIGEN_DEVICE_FUNC - EIGEN_STRONG_INLINE const Scalar& operator()(Index firstIndex, Index secondIndex, IndexTypes... otherIndices) const - { - EIGEN_STATIC_ASSERT(sizeof...(otherIndices) + 2 == NumIndices, YOU_MADE_A_PROGRAMMING_MISTAKE) - if (PlainObjectType::Options&RowMajor) { - const Index index = m_dimensions.IndexOfRowMajor(array{{firstIndex, secondIndex, otherIndices...}}); - return m_data[index]; - } else { - const Index index = m_dimensions.IndexOfColMajor(array{{firstIndex, secondIndex, otherIndices...}}); - return m_data[index]; - } + template + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const Scalar & + operator()(Index firstIndex, Index secondIndex, IndexTypes... otherIndices) const + { + EIGEN_STATIC_ASSERT(sizeof...(otherIndices) + 2 == NumIndices, YOU_MADE_A_PROGRAMMING_MISTAKE) + if (PlainObjectType::Options & RowMajor) { + const Index index = m_dimensions.IndexOfRowMajor(array{ + { firstIndex, + secondIndex, + otherIndices... } }); + return m_data[index]; + } else { + const Index index = m_dimensions.IndexOfColMajor(array{ + { firstIndex, + secondIndex, + otherIndices... } }); + return m_data[index]; } + } #else - EIGEN_DEVICE_FUNC - EIGEN_STRONG_INLINE const Scalar& operator()(Index i0, Index i1) const - { - if (PlainObjectType::Options&RowMajor) { - const Index index = i1 + i0 * m_dimensions[1]; - return m_data[index]; - } else { - const Index index = i0 + i1 * m_dimensions[0]; - return m_data[index]; - } + EIGEN_DEVICE_FUNC + EIGEN_STRONG_INLINE const Scalar &operator()(Index i0, Index i1) const + { + if (PlainObjectType::Options & RowMajor) { + const Index index = i1 + i0 * m_dimensions[1]; + return m_data[index]; + } else { + const Index index = i0 + i1 * m_dimensions[0]; + return m_data[index]; } - EIGEN_DEVICE_FUNC - EIGEN_STRONG_INLINE const Scalar& operator()(Index i0, Index i1, Index i2) const - { - if (PlainObjectType::Options&RowMajor) { - const Index index = i2 + m_dimensions[2] * (i1 + m_dimensions[1] * i0); - return m_data[index]; - } else { - const Index index = i0 + m_dimensions[0] * (i1 + m_dimensions[1] * i2); - return m_data[index]; - } + } + EIGEN_DEVICE_FUNC + EIGEN_STRONG_INLINE const Scalar &operator()(Index i0, Index i1, Index i2) const + { + if (PlainObjectType::Options & RowMajor) { + const Index index = i2 + m_dimensions[2] * (i1 + m_dimensions[1] * i0); + return m_data[index]; + } else { + const Index index = i0 + m_dimensions[0] * (i1 + m_dimensions[1] * i2); + return m_data[index]; } - EIGEN_DEVICE_FUNC - EIGEN_STRONG_INLINE const Scalar& operator()(Index i0, Index i1, Index i2, Index i3) const - { - if (PlainObjectType::Options&RowMajor) { - const Index index = i3 + m_dimensions[3] * (i2 + m_dimensions[2] * (i1 + m_dimensions[1] * i0)); - return m_data[index]; - } else { - const Index index = i0 + m_dimensions[0] * (i1 + m_dimensions[1] * (i2 + m_dimensions[2] * i3)); - return m_data[index]; - } + } + EIGEN_DEVICE_FUNC + EIGEN_STRONG_INLINE const Scalar &operator()(Index i0, Index i1, Index i2, Index i3) const + { + if (PlainObjectType::Options & RowMajor) { + const Index index = i3 + m_dimensions[3] * (i2 + m_dimensions[2] * (i1 + m_dimensions[1] * i0)); + return m_data[index]; + } else { + const Index index = i0 + m_dimensions[0] * (i1 + m_dimensions[1] * (i2 + m_dimensions[2] * i3)); + return m_data[index]; } - EIGEN_DEVICE_FUNC - EIGEN_STRONG_INLINE const Scalar& operator()(Index i0, Index i1, Index i2, Index i3, Index i4) const - { - if (PlainObjectType::Options&RowMajor) { - const Index index = i4 + m_dimensions[4] * (i3 + m_dimensions[3] * (i2 + m_dimensions[2] * (i1 + m_dimensions[1] * i0))); - return m_data[index]; - } else { - const Index index = i0 + m_dimensions[0] * (i1 + m_dimensions[1] * (i2 + m_dimensions[2] * (i3 + m_dimensions[3] * i4))); - return m_data[index]; - } + } + EIGEN_DEVICE_FUNC + EIGEN_STRONG_INLINE const Scalar &operator()(Index i0, Index i1, Index i2, Index i3, Index i4) const + { + if (PlainObjectType::Options & RowMajor) { + const Index index = + i4 + m_dimensions[4] * (i3 + m_dimensions[3] * (i2 + m_dimensions[2] * (i1 + m_dimensions[1] * i0))); + return m_data[index]; + } else { + const Index index = + i0 + m_dimensions[0] * (i1 + m_dimensions[1] * (i2 + m_dimensions[2] * (i3 + m_dimensions[3] * i4))); + return m_data[index]; } + } #endif - EIGEN_DEVICE_FUNC - EIGEN_STRONG_INLINE Scalar& operator()(const array& indices) - { - // eigen_assert(checkIndexRange(indices)); - if (PlainObjectType::Options&RowMajor) { - const Index index = m_dimensions.IndexOfRowMajor(indices); - return m_data[index]; - } else { - const Index index = m_dimensions.IndexOfColMajor(indices); - return m_data[index]; - } + EIGEN_DEVICE_FUNC + EIGEN_STRONG_INLINE Scalar &operator()(const array &indices) + { + // eigen_assert(checkIndexRange(indices)); + if (PlainObjectType::Options & RowMajor) { + const Index index = m_dimensions.IndexOfRowMajor(indices); + return m_data[index]; + } else { + const Index index = m_dimensions.IndexOfColMajor(indices); + return m_data[index]; } + } - EIGEN_DEVICE_FUNC - EIGEN_STRONG_INLINE Scalar& operator()() - { - EIGEN_STATIC_ASSERT(NumIndices == 0, YOU_MADE_A_PROGRAMMING_MISTAKE) - return m_data[0]; - } + EIGEN_DEVICE_FUNC + EIGEN_STRONG_INLINE Scalar &operator()() + { + EIGEN_STATIC_ASSERT(NumIndices == 0, YOU_MADE_A_PROGRAMMING_MISTAKE) + return m_data[0]; + } - EIGEN_DEVICE_FUNC - EIGEN_STRONG_INLINE Scalar& operator()(Index index) - { - eigen_internal_assert(index >= 0 && index < size()); - return m_data[index]; - } + EIGEN_DEVICE_FUNC + EIGEN_STRONG_INLINE Scalar &operator()(Index index) + { + eigen_internal_assert(index >= 0 && index < size()); + return m_data[index]; + } #if EIGEN_HAS_VARIADIC_TEMPLATES - template EIGEN_DEVICE_FUNC - EIGEN_STRONG_INLINE Scalar& operator()(Index firstIndex, Index secondIndex, IndexTypes... otherIndices) - { - static_assert(sizeof...(otherIndices) + 2 == NumIndices || NumIndices == Dynamic, "Number of indices used to access a tensor coefficient must be equal to the rank of the tensor."); - const std::size_t NumDims = sizeof...(otherIndices) + 2; - if (PlainObjectType::Options&RowMajor) { - const Index index = m_dimensions.IndexOfRowMajor(array{{firstIndex, secondIndex, otherIndices...}}); - return m_data[index]; - } else { - const Index index = m_dimensions.IndexOfColMajor(array{{firstIndex, secondIndex, otherIndices...}}); - return m_data[index]; - } + template + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Scalar & + operator()(Index firstIndex, Index secondIndex, IndexTypes... otherIndices) + { + static_assert(sizeof...(otherIndices) + 2 == NumIndices || NumIndices == Dynamic, + "Number of indices used to access a tensor coefficient must be equal to the rank of the tensor."); + const std::size_t NumDims = sizeof...(otherIndices) + 2; + if (PlainObjectType::Options & RowMajor) { + const Index index = m_dimensions.IndexOfRowMajor(array{ + { firstIndex, + secondIndex, + otherIndices... } }); + return m_data[index]; + } else { + const Index index = m_dimensions.IndexOfColMajor(array{ + { firstIndex, + secondIndex, + otherIndices... } }); + return m_data[index]; } + } #else - EIGEN_DEVICE_FUNC - EIGEN_STRONG_INLINE Scalar& operator()(Index i0, Index i1) - { - if (PlainObjectType::Options&RowMajor) { - const Index index = i1 + i0 * m_dimensions[1]; - return m_data[index]; - } else { - const Index index = i0 + i1 * m_dimensions[0]; - return m_data[index]; - } + EIGEN_DEVICE_FUNC + EIGEN_STRONG_INLINE Scalar &operator()(Index i0, Index i1) + { + if (PlainObjectType::Options & RowMajor) { + const Index index = i1 + i0 * m_dimensions[1]; + return m_data[index]; + } else { + const Index index = i0 + i1 * m_dimensions[0]; + return m_data[index]; } - EIGEN_DEVICE_FUNC - EIGEN_STRONG_INLINE Scalar& operator()(Index i0, Index i1, Index i2) - { - if (PlainObjectType::Options&RowMajor) { - const Index index = i2 + m_dimensions[2] * (i1 + m_dimensions[1] * i0); - return m_data[index]; - } else { - const Index index = i0 + m_dimensions[0] * (i1 + m_dimensions[1] * i2); - return m_data[index]; - } + } + EIGEN_DEVICE_FUNC + EIGEN_STRONG_INLINE Scalar &operator()(Index i0, Index i1, Index i2) + { + if (PlainObjectType::Options & RowMajor) { + const Index index = i2 + m_dimensions[2] * (i1 + m_dimensions[1] * i0); + return m_data[index]; + } else { + const Index index = i0 + m_dimensions[0] * (i1 + m_dimensions[1] * i2); + return m_data[index]; } - EIGEN_DEVICE_FUNC - EIGEN_STRONG_INLINE Scalar& operator()(Index i0, Index i1, Index i2, Index i3) - { - if (PlainObjectType::Options&RowMajor) { - const Index index = i3 + m_dimensions[3] * (i2 + m_dimensions[2] * (i1 + m_dimensions[1] * i0)); - return m_data[index]; - } else { - const Index index = i0 + m_dimensions[0] * (i1 + m_dimensions[1] * (i2 + m_dimensions[2] * i3)); - return m_data[index]; - } + } + EIGEN_DEVICE_FUNC + EIGEN_STRONG_INLINE Scalar &operator()(Index i0, Index i1, Index i2, Index i3) + { + if (PlainObjectType::Options & RowMajor) { + const Index index = i3 + m_dimensions[3] * (i2 + m_dimensions[2] * (i1 + m_dimensions[1] * i0)); + return m_data[index]; + } else { + const Index index = i0 + m_dimensions[0] * (i1 + m_dimensions[1] * (i2 + m_dimensions[2] * i3)); + return m_data[index]; } - EIGEN_DEVICE_FUNC - EIGEN_STRONG_INLINE Scalar& operator()(Index i0, Index i1, Index i2, Index i3, Index i4) - { - if (PlainObjectType::Options&RowMajor) { - const Index index = i4 + m_dimensions[4] * (i3 + m_dimensions[3] * (i2 + m_dimensions[2] * (i1 + m_dimensions[1] * i0))); - return m_data[index]; - } else { - const Index index = i0 + m_dimensions[0] * (i1 + m_dimensions[1] * (i2 + m_dimensions[2] * (i3 + m_dimensions[3] * i4))); - return m_data[index]; - } + } + EIGEN_DEVICE_FUNC + EIGEN_STRONG_INLINE Scalar &operator()(Index i0, Index i1, Index i2, Index i3, Index i4) + { + if (PlainObjectType::Options & RowMajor) { + const Index index = + i4 + m_dimensions[4] * (i3 + m_dimensions[3] * (i2 + m_dimensions[2] * (i1 + m_dimensions[1] * i0))); + return m_data[index]; + } else { + const Index index = + i0 + m_dimensions[0] * (i1 + m_dimensions[1] * (i2 + m_dimensions[2] * (i3 + m_dimensions[3] * i4))); + return m_data[index]; } + } #endif - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Self& operator=(const Self& other) - { - typedef TensorAssignOp Assign; - Assign assign(*this, other); - internal::TensorExecutor::run(assign, DefaultDevice()); - return *this; - } + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Self &operator=(const Self &other) + { + typedef TensorAssignOp Assign; + Assign assign(*this, other); + internal::TensorExecutor::run(assign, DefaultDevice()); + return *this; + } - template - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE - Self& operator=(const OtherDerived& other) - { - typedef TensorAssignOp Assign; - Assign assign(*this, other); - internal::TensorExecutor::run(assign, DefaultDevice()); - return *this; - } + template EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Self &operator=(const OtherDerived &other) + { + typedef TensorAssignOp Assign; + Assign assign(*this, other); + internal::TensorExecutor::run(assign, DefaultDevice()); + return *this; + } - private: - typename MakePointer_::Type m_data; - Dimensions m_dimensions; +private: + typename MakePointer_::Type m_data; + Dimensions m_dimensions; }; -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_CXX11_TENSOR_TENSOR_MAP_H +#endif// EIGEN_CXX11_TENSOR_TENSOR_MAP_H diff --git a/filmulator-gui/core/nlmeans/eigen/unsupported/Eigen/CXX11/src/Tensor/TensorMeta.h b/filmulator-gui/core/nlmeans/eigen/unsupported/Eigen/CXX11/src/Tensor/TensorMeta.h index 615559d4..4aa99131 100644 --- a/filmulator-gui/core/nlmeans/eigen/unsupported/Eigen/CXX11/src/Tensor/TensorMeta.h +++ b/filmulator-gui/core/nlmeans/eigen/unsupported/Eigen/CXX11/src/Tensor/TensorMeta.h @@ -12,150 +12,155 @@ namespace Eigen { -template struct Cond {}; +template struct Cond +{ +}; -template EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE -const T1& choose(Cond, const T1& first, const T2&) { +template +EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE const T1 &choose(Cond, const T1 &first, const T2 &) +{ return first; } -template EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE -const T2& choose(Cond, const T1&, const T2& second) { +template +EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE const T2 &choose(Cond, const T1 &, const T2 &second) +{ return second; } -template -EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE -T divup(const X x, const Y y) { +template EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE T divup(const X x, const Y y) +{ return static_cast((x + y - 1) / y); } -template -EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE -T divup(const T x, const T y) { +template EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE T divup(const T x, const T y) +{ return static_cast((x + y - 1) / y); } -template struct max_n_1 { +template struct max_n_1 +{ static const size_t size = n; }; -template <> struct max_n_1<0> { +template<> struct max_n_1<0> +{ static const size_t size = 1; }; // Default packet types -template -struct PacketType : internal::packet_traits { +template struct PacketType : internal::packet_traits +{ typedef typename internal::packet_traits::type type; }; // For CUDA packet types when using a GpuDevice #if defined(EIGEN_USE_GPU) && defined(__CUDACC__) && defined(EIGEN_HAS_CUDA_FP16) -template <> -struct PacketType { +template<> struct PacketType +{ typedef half2 type; static const int size = 2; enum { - HasAdd = 1, - HasSub = 1, - HasMul = 1, + HasAdd = 1, + HasSub = 1, + HasMul = 1, HasNegate = 1, - HasAbs = 1, - HasArg = 0, - HasAbs2 = 0, - HasMin = 1, - HasMax = 1, - HasConj = 0, + HasAbs = 1, + HasArg = 0, + HasAbs2 = 0, + HasMin = 1, + HasMax = 1, + HasConj = 0, HasSetLinear = 0, - HasBlend = 0, - - HasDiv = 1, - HasSqrt = 1, - HasRsqrt = 1, - HasExp = 1, - HasLog = 1, - HasLog1p = 0, - HasLog10 = 0, - HasPow = 1, + HasBlend = 0, + + HasDiv = 1, + HasSqrt = 1, + HasRsqrt = 1, + HasExp = 1, + HasLog = 1, + HasLog1p = 0, + HasLog10 = 0, + HasPow = 1, }; }; #endif #if defined(EIGEN_USE_SYCL) -template - struct PacketType { +template struct PacketType +{ typedef T type; static const int size = 1; enum { - HasAdd = 0, - HasSub = 0, - HasMul = 0, + HasAdd = 0, + HasSub = 0, + HasMul = 0, HasNegate = 0, - HasAbs = 0, - HasArg = 0, - HasAbs2 = 0, - HasMin = 0, - HasMax = 0, - HasConj = 0, + HasAbs = 0, + HasArg = 0, + HasAbs2 = 0, + HasMin = 0, + HasMax = 0, + HasConj = 0, HasSetLinear = 0, - HasBlend = 0 + HasBlend = 0 }; }; #endif // Tuple mimics std::pair but works on e.g. nvcc. -template struct Tuple { - public: +template struct Tuple +{ +public: U first; V second; typedef U first_type; typedef V second_type; - EIGEN_CONSTEXPR EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE - Tuple() : first(), second() {} + EIGEN_CONSTEXPR EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Tuple() : first(), second() {} - EIGEN_CONSTEXPR EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE - Tuple(const U& f, const V& s) : first(f), second(s) {} + EIGEN_CONSTEXPR EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Tuple(const U &f, const V &s) : first(f), second(s) {} - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE - Tuple& operator= (const Tuple& rhs) { + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Tuple &operator=(const Tuple &rhs) + { if (&rhs == this) return *this; first = rhs.first; second = rhs.second; return *this; } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE - void swap(Tuple& rhs) { + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void swap(Tuple &rhs) + { using numext::swap; swap(first, rhs.first); swap(second, rhs.second); } }; -template -EIGEN_CONSTEXPR EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE -bool operator==(const Tuple& x, const Tuple& y) { +template +EIGEN_CONSTEXPR EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE bool operator==(const Tuple &x, const Tuple &y) +{ return (x.first == y.first && x.second == y.second); } -template -EIGEN_CONSTEXPR EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE -bool operator!=(const Tuple& x, const Tuple& y) { +template +EIGEN_CONSTEXPR EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE bool operator!=(const Tuple &x, const Tuple &y) +{ return !(x == y); } // Can't use std::pairs on cuda devices -template struct IndexPair { +template struct IndexPair +{ EIGEN_CONSTEXPR EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE IndexPair() : first(0), second(0) {} EIGEN_CONSTEXPR EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE IndexPair(Idx f, Idx s) : first(f), second(s) {} - EIGEN_DEVICE_FUNC void set(IndexPair val) { + EIGEN_DEVICE_FUNC void set(IndexPair val) + { first = val.first; second = val.second; } @@ -169,50 +174,48 @@ template struct IndexPair { namespace internal { template - EIGEN_CONSTEXPR EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE - array customIndices2Array(IndexType& idx, numeric_list) { + EIGEN_CONSTEXPR EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE array customIndices2Array(IndexType &idx, + numeric_list) + { return { idx[Is]... }; } template - EIGEN_CONSTEXPR EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE - array customIndices2Array(IndexType&, numeric_list) { + EIGEN_CONSTEXPR EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE array customIndices2Array(IndexType &, + numeric_list) + { return array(); } /** Make an array (for index/dimensions) out of a custom index */ template - EIGEN_CONSTEXPR EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE - array customIndices2Array(IndexType& idx) { + EIGEN_CONSTEXPR EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE array customIndices2Array(IndexType &idx) + { return customIndices2Array(idx, typename gen_numeric_list::type{}); } - template - struct is_base_of + template struct is_base_of { typedef char (&yes)[1]; typedef char (&no)[2]; - template - struct Host + template struct Host { - operator BB*() const; - operator DD*(); + operator BB *() const; + operator DD *(); }; - template - static yes check(D*, T); - static no check(B*, int); + template static yes check(D *, T); + static no check(B *, int); - static const bool value = sizeof(check(Host(), int())) == sizeof(yes); + static const bool value = sizeof(check(Host(), int())) == sizeof(yes); }; -} +}// namespace internal #endif +}// namespace Eigen -} // namespace Eigen - -#endif // EIGEN_CXX11_TENSOR_TENSOR_META_H +#endif// EIGEN_CXX11_TENSOR_TENSOR_META_H diff --git a/filmulator-gui/core/nlmeans/eigen/unsupported/Eigen/CXX11/src/Tensor/TensorMorphing.h b/filmulator-gui/core/nlmeans/eigen/unsupported/Eigen/CXX11/src/Tensor/TensorMorphing.h index d34f1e32..06a9fa8d 100644 --- a/filmulator-gui/core/nlmeans/eigen/unsupported/Eigen/CXX11/src/Tensor/TensorMorphing.h +++ b/filmulator-gui/core/nlmeans/eigen/unsupported/Eigen/CXX11/src/Tensor/TensorMorphing.h @@ -13,84 +13,84 @@ namespace Eigen { /** \class TensorReshaping - * \ingroup CXX11_Tensor_Module - * - * \brief Tensor reshaping class. - * - * - */ + * \ingroup CXX11_Tensor_Module + * + * \brief Tensor reshaping class. + * + * + */ namespace internal { -template -struct traits > : public traits -{ - typedef typename XprType::Scalar Scalar; - typedef traits XprTraits; - typedef typename XprTraits::StorageKind StorageKind; - typedef typename XprTraits::Index Index; - typedef typename XprType::Nested Nested; - typedef typename remove_reference::type _Nested; - static const int NumDimensions = array_size::value; - static const int Layout = XprTraits::Layout; -}; - -template -struct eval, Eigen::Dense> -{ - typedef const TensorReshapingOp& type; -}; + template + struct traits> : public traits + { + typedef typename XprType::Scalar Scalar; + typedef traits XprTraits; + typedef typename XprTraits::StorageKind StorageKind; + typedef typename XprTraits::Index Index; + typedef typename XprType::Nested Nested; + typedef typename remove_reference::type _Nested; + static const int NumDimensions = array_size::value; + static const int Layout = XprTraits::Layout; + }; -template -struct nested, 1, typename eval >::type> -{ - typedef TensorReshapingOp type; -}; + template + struct eval, Eigen::Dense> + { + typedef const TensorReshapingOp &type; + }; -} // end namespace internal + template + struct nested, + 1, + typename eval>::type> + { + typedef TensorReshapingOp type; + }; +}// end namespace internal template class TensorReshapingOp : public TensorBase, WriteAccessors> { - public: +public: typedef typename Eigen::internal::traits::Scalar Scalar; typedef typename internal::remove_const::type CoeffReturnType; typedef typename Eigen::internal::nested::type Nested; typedef typename Eigen::internal::traits::StorageKind StorageKind; typedef typename Eigen::internal::traits::Index Index; - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TensorReshapingOp(const XprType& expr, const NewDimensions& dims) - : m_xpr(expr), m_dims(dims) {} + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TensorReshapingOp(const XprType &expr, const NewDimensions &dims) + : m_xpr(expr), m_dims(dims) + {} - EIGEN_DEVICE_FUNC - const NewDimensions& dimensions() const { return m_dims; } + EIGEN_DEVICE_FUNC + const NewDimensions &dimensions() const { return m_dims; } - EIGEN_DEVICE_FUNC - const typename internal::remove_all::type& - expression() const { return m_xpr; } + EIGEN_DEVICE_FUNC + const typename internal::remove_all::type &expression() const { return m_xpr; } - EIGEN_DEVICE_FUNC - EIGEN_STRONG_INLINE TensorReshapingOp& operator = (const TensorReshapingOp& other) - { - typedef TensorAssignOp Assign; - Assign assign(*this, other); - internal::TensorExecutor::run(assign, DefaultDevice()); - return *this; - } + EIGEN_DEVICE_FUNC + EIGEN_STRONG_INLINE TensorReshapingOp &operator=(const TensorReshapingOp &other) + { + typedef TensorAssignOp Assign; + Assign assign(*this, other); + internal::TensorExecutor::run(assign, DefaultDevice()); + return *this; + } - template - EIGEN_DEVICE_FUNC - EIGEN_STRONG_INLINE TensorReshapingOp& operator = (const OtherDerived& other) - { - typedef TensorAssignOp Assign; - Assign assign(*this, other); - internal::TensorExecutor::run(assign, DefaultDevice()); - return *this; - } + template + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TensorReshapingOp &operator=(const OtherDerived &other) + { + typedef TensorAssignOp Assign; + Assign assign(*this, other); + internal::TensorExecutor::run(assign, DefaultDevice()); + return *this; + } - protected: - typename XprType::Nested m_xpr; - const NewDimensions m_dims; +protected: + typename XprType::Nested m_xpr; + const NewDimensions m_dims; }; @@ -105,12 +105,12 @@ struct TensorEvaluator, Device> IsAligned = TensorEvaluator::IsAligned, PacketAccess = TensorEvaluator::PacketAccess, Layout = TensorEvaluator::Layout, - CoordAccess = false, // to be implemented + CoordAccess = false,// to be implemented RawAccess = TensorEvaluator::RawAccess }; - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TensorEvaluator(const XprType& op, const Device& device) - : m_impl(op.expression(), device), m_dimensions(op.dimensions()) + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TensorEvaluator(const XprType &op, const Device &device) + : m_impl(op.expression(), device), m_dimensions(op.dimensions()) { // The total size of the reshaped tensor must be equal to the total size // of the input tensor. @@ -122,35 +122,31 @@ struct TensorEvaluator, Device> typedef typename XprType::CoeffReturnType CoeffReturnType; typedef typename PacketType::type PacketReturnType; - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const Dimensions& dimensions() const { return m_dimensions; } + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const Dimensions &dimensions() const { return m_dimensions; } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE bool evalSubExprsIfNeeded(CoeffReturnType* data) { + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE bool evalSubExprsIfNeeded(CoeffReturnType *data) + { return m_impl.evalSubExprsIfNeeded(data); } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void cleanup() { - m_impl.cleanup(); - } + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void cleanup() { m_impl.cleanup(); } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE CoeffReturnType coeff(Index index) const - { - return m_impl.coeff(index); - } + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE CoeffReturnType coeff(Index index) const { return m_impl.coeff(index); } - template - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE PacketReturnType packet(Index index) const + template EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE PacketReturnType packet(Index index) const { return m_impl.template packet(index); } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TensorOpCost costPerCoeff(bool vectorized) const { + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TensorOpCost costPerCoeff(bool vectorized) const + { return m_impl.costPerCoeff(vectorized); } - EIGEN_DEVICE_FUNC Scalar* data() const { return const_cast(m_impl.data()); } + EIGEN_DEVICE_FUNC Scalar *data() const { return const_cast(m_impl.data()); } - EIGEN_DEVICE_FUNC const TensorEvaluator& impl() const { return m_impl; } + EIGEN_DEVICE_FUNC const TensorEvaluator &impl() const { return m_impl; } - protected: +protected: TensorEvaluator m_impl; NewDimensions m_dimensions; }; @@ -158,7 +154,7 @@ struct TensorEvaluator, Device> // Eval as lvalue template - struct TensorEvaluator, Device> +struct TensorEvaluator, Device> : public TensorEvaluator, Device> { @@ -170,25 +166,19 @@ template IsAligned = TensorEvaluator::IsAligned, PacketAccess = TensorEvaluator::PacketAccess, Layout = TensorEvaluator::Layout, - CoordAccess = false, // to be implemented + CoordAccess = false,// to be implemented RawAccess = TensorEvaluator::RawAccess }; - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TensorEvaluator(const XprType& op, const Device& device) - : Base(op, device) - { } + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TensorEvaluator(const XprType &op, const Device &device) : Base(op, device) {} typedef typename XprType::Index Index; typedef typename XprType::Scalar Scalar; typedef typename XprType::CoeffReturnType CoeffReturnType; typedef typename PacketType::type PacketReturnType; - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE CoeffReturnType& coeffRef(Index index) - { - return this->m_impl.coeffRef(index); - } - template EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE - void writePacket(Index index, const PacketReturnType& x) + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE CoeffReturnType &coeffRef(Index index) { return this->m_impl.coeffRef(index); } + template EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void writePacket(Index index, const PacketReturnType &x) { this->m_impl.template writePacket(index, x); } @@ -196,110 +186,114 @@ template /** \class TensorSlicing - * \ingroup CXX11_Tensor_Module - * - * \brief Tensor slicing class. - * - * - */ + * \ingroup CXX11_Tensor_Module + * + * \brief Tensor slicing class. + * + * + */ namespace internal { -template -struct traits > : public traits -{ - typedef typename XprType::Scalar Scalar; - typedef traits XprTraits; - typedef typename XprTraits::StorageKind StorageKind; - typedef typename XprTraits::Index Index; - typedef typename XprType::Nested Nested; - typedef typename remove_reference::type _Nested; - static const int NumDimensions = array_size::value; - static const int Layout = XprTraits::Layout; -}; - -template -struct eval, Eigen::Dense> -{ - typedef const TensorSlicingOp& type; -}; + template + struct traits> : public traits + { + typedef typename XprType::Scalar Scalar; + typedef traits XprTraits; + typedef typename XprTraits::StorageKind StorageKind; + typedef typename XprTraits::Index Index; + typedef typename XprType::Nested Nested; + typedef typename remove_reference::type _Nested; + static const int NumDimensions = array_size::value; + static const int Layout = XprTraits::Layout; + }; -template -struct nested, 1, typename eval >::type> -{ - typedef TensorSlicingOp type; -}; + template + struct eval, Eigen::Dense> + { + typedef const TensorSlicingOp &type; + }; -} // end namespace internal + template + struct nested, + 1, + typename eval>::type> + { + typedef TensorSlicingOp type; + }; +}// end namespace internal template -class TensorSlicingOp : public TensorBase > +class TensorSlicingOp : public TensorBase> { - public: +public: typedef typename Eigen::internal::traits::Scalar Scalar; typedef typename XprType::CoeffReturnType CoeffReturnType; typedef typename Eigen::internal::nested::type Nested; typedef typename Eigen::internal::traits::StorageKind StorageKind; typedef typename Eigen::internal::traits::Index Index; - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TensorSlicingOp(const XprType& expr, const StartIndices& indices, const Sizes& sizes) - : m_xpr(expr), m_indices(indices), m_sizes(sizes) {} - - EIGEN_DEVICE_FUNC - const StartIndices& startIndices() const { return m_indices; } - EIGEN_DEVICE_FUNC - const Sizes& sizes() const { return m_sizes; } - - EIGEN_DEVICE_FUNC - const typename internal::remove_all::type& - expression() const { return m_xpr; } - - template - EIGEN_DEVICE_FUNC - EIGEN_STRONG_INLINE TensorSlicingOp& operator = (const OtherDerived& other) - { - typedef TensorAssignOp Assign; - Assign assign(*this, other); - internal::TensorExecutor::run(assign, DefaultDevice()); - return *this; - } + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TensorSlicingOp(const XprType &expr, + const StartIndices &indices, + const Sizes &sizes) + : m_xpr(expr), m_indices(indices), m_sizes(sizes) + {} - EIGEN_DEVICE_FUNC - EIGEN_STRONG_INLINE TensorSlicingOp& operator = (const TensorSlicingOp& other) - { - typedef TensorAssignOp Assign; - Assign assign(*this, other); - internal::TensorExecutor::run(assign, DefaultDevice()); - return *this; - } + EIGEN_DEVICE_FUNC + const StartIndices &startIndices() const { return m_indices; } + EIGEN_DEVICE_FUNC + const Sizes &sizes() const { return m_sizes; } + + EIGEN_DEVICE_FUNC + const typename internal::remove_all::type &expression() const { return m_xpr; } + template + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TensorSlicingOp &operator=(const OtherDerived &other) + { + typedef TensorAssignOp Assign; + Assign assign(*this, other); + internal::TensorExecutor::run(assign, DefaultDevice()); + return *this; + } - protected: - typename XprType::Nested m_xpr; - const StartIndices m_indices; - const Sizes m_sizes; + EIGEN_DEVICE_FUNC + EIGEN_STRONG_INLINE TensorSlicingOp &operator=(const TensorSlicingOp &other) + { + typedef TensorAssignOp Assign; + Assign assign(*this, other); + internal::TensorExecutor::run(assign, DefaultDevice()); + return *this; + } + + +protected: + typename XprType::Nested m_xpr; + const StartIndices m_indices; + const Sizes m_sizes; }; // Fixme: figure out the exact threshold namespace { -template struct MemcpyTriggerForSlicing { - EIGEN_DEVICE_FUNC MemcpyTriggerForSlicing(const Device& device) : threshold_(2 * device.numThreads()) { } - EIGEN_DEVICE_FUNC bool operator ()(Index val) const { return val > threshold_; } + template struct MemcpyTriggerForSlicing + { + EIGEN_DEVICE_FUNC MemcpyTriggerForSlicing(const Device &device) : threshold_(2 * device.numThreads()) {} + EIGEN_DEVICE_FUNC bool operator()(Index val) const { return val > threshold_; } - private: - Index threshold_; -}; + private: + Index threshold_; + }; // It is very expensive to start the memcpy kernel on GPU: we therefore only // use it for large copies. #ifdef EIGEN_USE_GPU -template struct MemcpyTriggerForSlicing { - EIGEN_DEVICE_FUNC MemcpyTriggerForSlicing(const GpuDevice&) { } - EIGEN_DEVICE_FUNC bool operator ()(Index val) const { return val > 4*1024*1024; } -}; + template struct MemcpyTriggerForSlicing + { + EIGEN_DEVICE_FUNC MemcpyTriggerForSlicing(const GpuDevice &) {} + EIGEN_DEVICE_FUNC bool operator()(Index val) const { return val > 4 * 1024 * 1024; } + }; #endif -} +}// namespace // Eval as rvalue template @@ -311,44 +305,40 @@ struct TensorEvaluator, Devi enum { // Alignment can't be guaranteed at compile time since it depends on the // slice offsets and sizes. - IsAligned = /*TensorEvaluator::IsAligned*/false, + IsAligned = /*TensorEvaluator::IsAligned*/ false, PacketAccess = TensorEvaluator::PacketAccess, Layout = TensorEvaluator::Layout, CoordAccess = false, RawAccess = false }; - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TensorEvaluator(const XprType& op, const Device& device) - : m_impl(op.expression(), device), m_device(device), m_dimensions(op.sizes()), m_offsets(op.startIndices()) + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TensorEvaluator(const XprType &op, const Device &device) + : m_impl(op.expression(), device), m_device(device), m_dimensions(op.sizes()), m_offsets(op.startIndices()) { for (std::size_t i = 0; i < internal::array_size::value; ++i) { eigen_assert(m_impl.dimensions()[i] >= op.sizes()[i] + op.startIndices()[i]); } - const typename TensorEvaluator::Dimensions& input_dims = m_impl.dimensions(); - const Sizes& output_dims = op.sizes(); + const typename TensorEvaluator::Dimensions &input_dims = m_impl.dimensions(); + const Sizes &output_dims = op.sizes(); if (static_cast(Layout) == static_cast(ColMajor)) { m_inputStrides[0] = 1; - for (int i = 1; i < NumDims; ++i) { - m_inputStrides[i] = m_inputStrides[i-1] * input_dims[i-1]; - } + for (int i = 1; i < NumDims; ++i) { m_inputStrides[i] = m_inputStrides[i - 1] * input_dims[i - 1]; } - // Don't initialize m_fastOutputStrides[0] since it won't ever be accessed. + // Don't initialize m_fastOutputStrides[0] since it won't ever be accessed. m_outputStrides[0] = 1; for (int i = 1; i < NumDims; ++i) { - m_outputStrides[i] = m_outputStrides[i-1] * output_dims[i-1]; + m_outputStrides[i] = m_outputStrides[i - 1] * output_dims[i - 1]; m_fastOutputStrides[i] = internal::TensorIntDivisor(m_outputStrides[i]); } } else { - m_inputStrides[NumDims-1] = 1; - for (int i = NumDims - 2; i >= 0; --i) { - m_inputStrides[i] = m_inputStrides[i+1] * input_dims[i+1]; - } + m_inputStrides[NumDims - 1] = 1; + for (int i = NumDims - 2; i >= 0; --i) { m_inputStrides[i] = m_inputStrides[i + 1] * input_dims[i + 1]; } - // Don't initialize m_fastOutputStrides[NumDims-1] since it won't ever be accessed. - m_outputStrides[NumDims-1] = 1; + // Don't initialize m_fastOutputStrides[NumDims-1] since it won't ever be accessed. + m_outputStrides[NumDims - 1] = 1; for (int i = NumDims - 2; i >= 0; --i) { - m_outputStrides[i] = m_outputStrides[i+1] * output_dims[i+1]; + m_outputStrides[i] = m_outputStrides[i + 1] * output_dims[i + 1]; m_fastOutputStrides[i] = internal::TensorIntDivisor(m_outputStrides[i]); } } @@ -360,35 +350,32 @@ struct TensorEvaluator, Devi typedef typename PacketType::type PacketReturnType; typedef Sizes Dimensions; - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const Dimensions& dimensions() const { return m_dimensions; } + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const Dimensions &dimensions() const { return m_dimensions; } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE bool evalSubExprsIfNeeded(CoeffReturnType* data) { + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE bool evalSubExprsIfNeeded(CoeffReturnType *data) + { m_impl.evalSubExprsIfNeeded(NULL); if (!NumTraits::type>::RequireInitialization && data && m_impl.data()) { Index contiguous_values = 1; if (static_cast(Layout) == static_cast(ColMajor)) { for (int i = 0; i < NumDims; ++i) { contiguous_values *= dimensions()[i]; - if (dimensions()[i] != m_impl.dimensions()[i]) { - break; - } + if (dimensions()[i] != m_impl.dimensions()[i]) { break; } } } else { - for (int i = NumDims-1; i >= 0; --i) { + for (int i = NumDims - 1; i >= 0; --i) { contiguous_values *= dimensions()[i]; - if (dimensions()[i] != m_impl.dimensions()[i]) { - break; - } + if (dimensions()[i] != m_impl.dimensions()[i]) { break; } } } // Use memcpy if it's going to be faster than using the regular evaluation. const MemcpyTriggerForSlicing trigger(m_device); if (trigger(contiguous_values)) { - Scalar* src = (Scalar*)m_impl.data(); + Scalar *src = (Scalar *)m_impl.data(); for (int i = 0; i < internal::array_prod(dimensions()); i += contiguous_values) { Index offset = srcCoeff(i); - m_device.memcpy((void*)(data+i), src+offset, contiguous_values * sizeof(Scalar)); + m_device.memcpy((void *)(data + i), src + offset, contiguous_values * sizeof(Scalar)); } return false; } @@ -396,24 +383,21 @@ struct TensorEvaluator, Devi return true; } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void cleanup() { - m_impl.cleanup(); - } + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void cleanup() { m_impl.cleanup(); } EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE CoeffReturnType coeff(Index index) const { return m_impl.coeff(srcCoeff(index)); } - template - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE PacketReturnType packet(Index index) const + template EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE PacketReturnType packet(Index index) const { const int packetSize = internal::unpacket_traits::size; EIGEN_STATIC_ASSERT((packetSize > 1), YOU_MADE_A_PROGRAMMING_MISTAKE) - eigen_assert(index+packetSize-1 < internal::array_prod(dimensions())); + eigen_assert(index + packetSize - 1 < internal::array_prod(dimensions())); - Index inputIndices[] = {0, 0}; - Index indices[] = {index, index + packetSize - 1}; + Index inputIndices[] = { 0, 0 }; + Index indices[] = { index, index + packetSize - 1 }; if (static_cast(Layout) == static_cast(ColMajor)) { for (int i = NumDims - 1; i > 0; --i) { const Index idx0 = indices[0] / m_fastOutputStrides[i]; @@ -434,42 +418,39 @@ struct TensorEvaluator, Devi indices[0] -= idx0 * m_outputStrides[i]; indices[1] -= idx1 * m_outputStrides[i]; } - inputIndices[0] += (indices[0] + m_offsets[NumDims-1]); - inputIndices[1] += (indices[1] + m_offsets[NumDims-1]); + inputIndices[0] += (indices[0] + m_offsets[NumDims - 1]); + inputIndices[1] += (indices[1] + m_offsets[NumDims - 1]); } if (inputIndices[1] - inputIndices[0] == packetSize - 1) { PacketReturnType rslt = m_impl.template packet(inputIndices[0]); return rslt; - } - else { + } else { EIGEN_ALIGN_MAX typename internal::remove_const::type values[packetSize]; values[0] = m_impl.coeff(inputIndices[0]); - values[packetSize-1] = m_impl.coeff(inputIndices[1]); - for (int i = 1; i < packetSize-1; ++i) { - values[i] = coeff(index+i); - } + values[packetSize - 1] = m_impl.coeff(inputIndices[1]); + for (int i = 1; i < packetSize - 1; ++i) { values[i] = coeff(index + i); } PacketReturnType rslt = internal::pload(values); return rslt; } } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TensorOpCost costPerCoeff(bool vectorized) const { + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TensorOpCost costPerCoeff(bool vectorized) const + { return m_impl.costPerCoeff(vectorized) + TensorOpCost(0, 0, NumDims); } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Scalar* data() const { - Scalar* result = m_impl.data(); + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Scalar *data() const + { + Scalar *result = m_impl.data(); if (result) { Index offset = 0; if (static_cast(Layout) == static_cast(ColMajor)) { for (int i = 0; i < NumDims; ++i) { if (m_dimensions[i] != m_impl.dimensions()[i]) { offset += m_offsets[i] * m_inputStrides[i]; - for (int j = i+1; j < NumDims; ++j) { - if (m_dimensions[j] > 1) { - return NULL; - } + for (int j = i + 1; j < NumDims; ++j) { + if (m_dimensions[j] > 1) { return NULL; } offset += m_offsets[j] * m_inputStrides[j]; } break; @@ -479,10 +460,8 @@ struct TensorEvaluator, Devi for (int i = NumDims - 1; i >= 0; --i) { if (m_dimensions[i] != m_impl.dimensions()[i]) { offset += m_offsets[i] * m_inputStrides[i]; - for (int j = i-1; j >= 0; --j) { - if (m_dimensions[j] > 1) { - return NULL; - } + for (int j = i - 1; j >= 0; --j) { + if (m_dimensions[j] > 1) { return NULL; } offset += m_offsets[j] * m_inputStrides[j]; } break; @@ -494,7 +473,7 @@ struct TensorEvaluator, Devi return NULL; } - protected: +protected: EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Index srcCoeff(Index index) const { Index inputIndex = 0; @@ -511,7 +490,7 @@ struct TensorEvaluator, Devi inputIndex += (idx + m_offsets[i]) * m_inputStrides[i]; index -= idx * m_outputStrides[i]; } - inputIndex += (index + m_offsets[NumDims-1]); + inputIndex += (index + m_offsets[NumDims - 1]); } return inputIndex; } @@ -520,7 +499,7 @@ struct TensorEvaluator, Devi array, NumDims> m_fastOutputStrides; array m_inputStrides; TensorEvaluator m_impl; - const Device& m_device; + const Device &m_device; Dimensions m_dimensions; const StartIndices m_offsets; }; @@ -536,16 +515,14 @@ struct TensorEvaluator, Device> static const int NumDims = internal::array_size::value; enum { - IsAligned = /*TensorEvaluator::IsAligned*/false, + IsAligned = /*TensorEvaluator::IsAligned*/ false, PacketAccess = TensorEvaluator::PacketAccess, Layout = TensorEvaluator::Layout, CoordAccess = false, RawAccess = false }; - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TensorEvaluator(const XprType& op, const Device& device) - : Base(op, device) - { } + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TensorEvaluator(const XprType &op, const Device &device) : Base(op, device) {} typedef typename XprType::Index Index; typedef typename XprType::Scalar Scalar; @@ -553,17 +530,16 @@ struct TensorEvaluator, Device> typedef typename PacketType::type PacketReturnType; typedef Sizes Dimensions; - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE CoeffReturnType& coeffRef(Index index) + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE CoeffReturnType &coeffRef(Index index) { return this->m_impl.coeffRef(this->srcCoeff(index)); } - template EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE - void writePacket(Index index, const PacketReturnType& x) + template EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void writePacket(Index index, const PacketReturnType &x) { const int packetSize = internal::unpacket_traits::size; - Index inputIndices[] = {0, 0}; - Index indices[] = {index, index + packetSize - 1}; + Index inputIndices[] = { 0, 0 }; + Index indices[] = { index, index + packetSize - 1 }; if (static_cast(Layout) == static_cast(ColMajor)) { for (int i = NumDims - 1; i > 0; --i) { const Index idx0 = indices[0] / this->m_fastOutputStrides[i]; @@ -584,108 +560,103 @@ struct TensorEvaluator, Device> indices[0] -= idx0 * this->m_outputStrides[i]; indices[1] -= idx1 * this->m_outputStrides[i]; } - inputIndices[0] += (indices[0] + this->m_offsets[NumDims-1]); - inputIndices[1] += (indices[1] + this->m_offsets[NumDims-1]); + inputIndices[0] += (indices[0] + this->m_offsets[NumDims - 1]); + inputIndices[1] += (indices[1] + this->m_offsets[NumDims - 1]); } if (inputIndices[1] - inputIndices[0] == packetSize - 1) { this->m_impl.template writePacket(inputIndices[0], x); - } - else { + } else { EIGEN_ALIGN_MAX CoeffReturnType values[packetSize]; internal::pstore(values, x); this->m_impl.coeffRef(inputIndices[0]) = values[0]; - this->m_impl.coeffRef(inputIndices[1]) = values[packetSize-1]; - for (int i = 1; i < packetSize-1; ++i) { - this->coeffRef(index+i) = values[i]; - } + this->m_impl.coeffRef(inputIndices[1]) = values[packetSize - 1]; + for (int i = 1; i < packetSize - 1; ++i) { this->coeffRef(index + i) = values[i]; } } } }; - namespace internal { -template -struct traits > : public traits -{ - typedef typename XprType::Scalar Scalar; - typedef traits XprTraits; - typedef typename XprTraits::StorageKind StorageKind; - typedef typename XprTraits::Index Index; - typedef typename XprType::Nested Nested; - typedef typename remove_reference::type _Nested; - static const int NumDimensions = array_size::value; - static const int Layout = XprTraits::Layout; -}; + template + struct traits> : public traits + { + typedef typename XprType::Scalar Scalar; + typedef traits XprTraits; + typedef typename XprTraits::StorageKind StorageKind; + typedef typename XprTraits::Index Index; + typedef typename XprType::Nested Nested; + typedef typename remove_reference::type _Nested; + static const int NumDimensions = array_size::value; + static const int Layout = XprTraits::Layout; + }; -template -struct eval, Eigen::Dense> -{ - typedef const TensorStridingSlicingOp& type; -}; + template + struct eval, Eigen::Dense> + { + typedef const TensorStridingSlicingOp &type; + }; -template -struct nested, 1, typename eval >::type> -{ - typedef TensorStridingSlicingOp type; -}; + template + struct nested, + 1, + typename eval>::type> + { + typedef TensorStridingSlicingOp type; + }; -} // end namespace internal +}// end namespace internal template -class TensorStridingSlicingOp : public TensorBase > +class TensorStridingSlicingOp : public TensorBase> { - public: +public: typedef typename internal::traits::Scalar Scalar; typedef typename XprType::CoeffReturnType CoeffReturnType; typedef typename internal::nested::type Nested; typedef typename internal::traits::StorageKind StorageKind; typedef typename internal::traits::Index Index; - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TensorStridingSlicingOp( - const XprType& expr, const StartIndices& startIndices, - const StopIndices& stopIndices, const Strides& strides) - : m_xpr(expr), m_startIndices(startIndices), m_stopIndices(stopIndices), - m_strides(strides) {} - - EIGEN_DEVICE_FUNC - const StartIndices& startIndices() const { return m_startIndices; } - EIGEN_DEVICE_FUNC - const StartIndices& stopIndices() const { return m_stopIndices; } - EIGEN_DEVICE_FUNC - const StartIndices& strides() const { return m_strides; } - - EIGEN_DEVICE_FUNC - const typename internal::remove_all::type& - expression() const { return m_xpr; } - - EIGEN_DEVICE_FUNC - EIGEN_STRONG_INLINE TensorStridingSlicingOp& operator = (const TensorStridingSlicingOp& other) - { - typedef TensorAssignOp Assign; - Assign assign(*this, other); - internal::TensorExecutor::run( - assign, DefaultDevice()); - return *this; - } + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TensorStridingSlicingOp(const XprType &expr, + const StartIndices &startIndices, + const StopIndices &stopIndices, + const Strides &strides) + : m_xpr(expr), m_startIndices(startIndices), m_stopIndices(stopIndices), m_strides(strides) + {} + + EIGEN_DEVICE_FUNC + const StartIndices &startIndices() const { return m_startIndices; } + EIGEN_DEVICE_FUNC + const StartIndices &stopIndices() const { return m_stopIndices; } + EIGEN_DEVICE_FUNC + const StartIndices &strides() const { return m_strides; } + + EIGEN_DEVICE_FUNC + const typename internal::remove_all::type &expression() const { return m_xpr; } + + EIGEN_DEVICE_FUNC + EIGEN_STRONG_INLINE TensorStridingSlicingOp &operator=(const TensorStridingSlicingOp &other) + { + typedef TensorAssignOp Assign; + Assign assign(*this, other); + internal::TensorExecutor::run(assign, DefaultDevice()); + return *this; + } - template - EIGEN_DEVICE_FUNC - EIGEN_STRONG_INLINE TensorStridingSlicingOp& operator = (const OtherDerived& other) - { - typedef TensorAssignOp Assign; - Assign assign(*this, other); - internal::TensorExecutor::run( - assign, DefaultDevice()); - return *this; - } + template + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TensorStridingSlicingOp &operator=(const OtherDerived &other) + { + typedef TensorAssignOp Assign; + Assign assign(*this, other); + internal::TensorExecutor::run(assign, DefaultDevice()); + return *this; + } - protected: - typename XprType::Nested m_xpr; - const StartIndices m_startIndices; - const StopIndices m_stopIndices; - const Strides m_strides; +protected: + typename XprType::Nested m_xpr; + const StartIndices m_startIndices; + const StopIndices m_stopIndices; + const Strides m_strides; }; // Eval as rvalue @@ -705,17 +676,17 @@ struct TensorEvaluator startIndicesClamped, stopIndicesClamped; + DSizes startIndicesClamped, stopIndicesClamped; for (size_t i = 0; i < internal::array_size::value; ++i) { eigen_assert(m_strides[i] != 0 && "0 stride is invalid"); - if(m_strides[i]>0){ + if (m_strides[i] > 0) { startIndicesClamped[i] = clamp(op.startIndices()[i], 0, m_impl.dimensions()[i]); stopIndicesClamped[i] = clamp(op.stopIndices()[i], 0, m_impl.dimensions()[i]); - }else{ + } else { /* implies m_strides[i]<0 by assert */ startIndicesClamped[i] = clamp(op.startIndices()[i], -1, m_impl.dimensions()[i] - 1); stopIndicesClamped[i] = clamp(op.stopIndices()[i], -1, m_impl.dimensions()[i] - 1); @@ -723,18 +694,18 @@ struct TensorEvaluator::Dimensions& input_dims = m_impl.dimensions(); + const typename TensorEvaluator::Dimensions &input_dims = m_impl.dimensions(); // check for degenerate intervals and compute output tensor shape - bool degenerate = false;; - for(int i = 0; i < NumDims; i++){ + bool degenerate = false; + ; + for (int i = 0; i < NumDims; i++) { Index interval = stopIndicesClamped[i] - startIndicesClamped[i]; - if(interval == 0 || ((interval<0) != (m_strides[i]<0))){ + if (interval == 0 || ((interval < 0) != (m_strides[i] < 0))) { m_dimensions[i] = 0; degenerate = true; - }else{ - m_dimensions[i] = interval / m_strides[i] - + (interval % m_strides[i] != 0 ? 1 : 0); + } else { + m_dimensions[i] = interval / m_strides[i] + (interval % m_strides[i] != 0 ? 1 : 0); eigen_assert(m_dimensions[i] >= 0); } } @@ -745,7 +716,7 @@ struct TensorEvaluator(degenerate ? 1 : m_outputStrides[i]); } } else { - m_inputStrides[NumDims-1] = m_strides[NumDims-1]; - m_offsets[NumDims-1] = startIndicesClamped[NumDims-1]; + m_inputStrides[NumDims - 1] = m_strides[NumDims - 1]; + m_offsets[NumDims - 1] = startIndicesClamped[NumDims - 1]; Index previousDimProduct = 1; for (int i = NumDims - 2; i >= 0; --i) { - previousDimProduct *= input_dims[i+1]; + previousDimProduct *= input_dims[i + 1]; m_inputStrides[i] = previousDimProduct * m_strides[i]; m_offsets[i] = startIndicesClamped[i] * previousDimProduct; } - m_outputStrides[NumDims-1] = 1; + m_outputStrides[NumDims - 1] = 1; for (int i = NumDims - 2; i >= 0; --i) { - m_outputStrides[i] = m_outputStrides[i+1] * output_dims[i+1]; + m_outputStrides[i] = m_outputStrides[i + 1] * output_dims[i + 1]; // NOTE: if tensor is degenerate, we send 1 to prevent TensorIntDivisor constructor crash m_fastOutputStrides[i] = internal::TensorIntDivisor(degenerate ? 1 : m_outputStrides[i]); } } - m_block_total_size_max = numext::maxi(static_cast(1), - device.lastLevelCacheSize() / - sizeof(Scalar)); + m_block_total_size_max = numext::maxi(static_cast(1), device.lastLevelCacheSize() / sizeof(Scalar)); } typedef typename XprType::Index Index; @@ -786,32 +755,30 @@ struct TensorEvaluator::type PacketReturnType; typedef Strides Dimensions; - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const Dimensions& dimensions() const { return m_dimensions; } + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const Dimensions &dimensions() const { return m_dimensions; } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE bool evalSubExprsIfNeeded(CoeffReturnType*) { + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE bool evalSubExprsIfNeeded(CoeffReturnType *) + { m_impl.evalSubExprsIfNeeded(NULL); return true; } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void cleanup() { - m_impl.cleanup(); - } + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void cleanup() { m_impl.cleanup(); } EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE CoeffReturnType coeff(Index index) const { return m_impl.coeff(srcCoeff(index)); } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TensorOpCost costPerCoeff(bool vectorized) const { + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TensorOpCost costPerCoeff(bool vectorized) const + { return m_impl.costPerCoeff(vectorized) + TensorOpCost(0, 0, NumDims); } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Scalar* data() const { - return NULL; - } + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Scalar *data() const { return NULL; } - protected: +protected: EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Index srcCoeff(Index index) const { Index inputIndex = 0; @@ -831,18 +798,19 @@ struct TensorEvaluator m_outputStrides; array, NumDims> m_fastOutputStrides; array m_inputStrides; TensorEvaluator m_impl; - const Device& m_device; - DSizes m_startIndices; // clamped startIndices + const Device &m_device; + DSizes m_startIndices;// clamped startIndices DSizes m_dimensions; - DSizes m_offsets; // offset in a flattened shape + DSizes m_offsets;// offset in a flattened shape const Strides m_strides; std::size_t m_block_total_size_max; }; @@ -865,9 +833,7 @@ struct TensorEvaluator::type PacketReturnType; typedef Strides Dimensions; - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE CoeffReturnType& coeffRef(Index index) + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE CoeffReturnType &coeffRef(Index index) { return this->m_impl.coeffRef(this->srcCoeff(index)); } }; -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_CXX11_TENSOR_TENSOR_MORPHING_H +#endif// EIGEN_CXX11_TENSOR_TENSOR_MORPHING_H diff --git a/filmulator-gui/core/nlmeans/eigen/unsupported/Eigen/CXX11/src/Tensor/TensorPadding.h b/filmulator-gui/core/nlmeans/eigen/unsupported/Eigen/CXX11/src/Tensor/TensorPadding.h index 647bcf10..2b59d794 100644 --- a/filmulator-gui/core/nlmeans/eigen/unsupported/Eigen/CXX11/src/Tensor/TensorPadding.h +++ b/filmulator-gui/core/nlmeans/eigen/unsupported/Eigen/CXX11/src/Tensor/TensorPadding.h @@ -13,46 +13,47 @@ namespace Eigen { /** \class TensorPadding - * \ingroup CXX11_Tensor_Module - * - * \brief Tensor padding class. - * At the moment only padding with a constant value is supported. - * - */ + * \ingroup CXX11_Tensor_Module + * + * \brief Tensor padding class. + * At the moment only padding with a constant value is supported. + * + */ namespace internal { -template -struct traits > : public traits -{ - typedef typename XprType::Scalar Scalar; - typedef traits XprTraits; - typedef typename XprTraits::StorageKind StorageKind; - typedef typename XprTraits::Index Index; - typedef typename XprType::Nested Nested; - typedef typename remove_reference::type _Nested; - static const int NumDimensions = XprTraits::NumDimensions; - static const int Layout = XprTraits::Layout; -}; - -template -struct eval, Eigen::Dense> -{ - typedef const TensorPaddingOp& type; -}; + template + struct traits> : public traits + { + typedef typename XprType::Scalar Scalar; + typedef traits XprTraits; + typedef typename XprTraits::StorageKind StorageKind; + typedef typename XprTraits::Index Index; + typedef typename XprType::Nested Nested; + typedef typename remove_reference::type _Nested; + static const int NumDimensions = XprTraits::NumDimensions; + static const int Layout = XprTraits::Layout; + }; -template -struct nested, 1, typename eval >::type> -{ - typedef TensorPaddingOp type; -}; + template + struct eval, Eigen::Dense> + { + typedef const TensorPaddingOp &type; + }; -} // end namespace internal + template + struct nested, + 1, + typename eval>::type> + { + typedef TensorPaddingOp type; + }; +}// end namespace internal template class TensorPaddingOp : public TensorBase, ReadOnlyAccessors> { - public: +public: typedef typename Eigen::internal::traits::Scalar Scalar; typedef typename Eigen::NumTraits::Real RealScalar; typedef typename XprType::CoeffReturnType CoeffReturnType; @@ -60,22 +61,24 @@ class TensorPaddingOp : public TensorBase::StorageKind StorageKind; typedef typename Eigen::internal::traits::Index Index; - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TensorPaddingOp(const XprType& expr, const PaddingDimensions& padding_dims, const Scalar padding_value) - : m_xpr(expr), m_padding_dims(padding_dims), m_padding_value(padding_value) {} + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TensorPaddingOp(const XprType &expr, + const PaddingDimensions &padding_dims, + const Scalar padding_value) + : m_xpr(expr), m_padding_dims(padding_dims), m_padding_value(padding_value) + {} - EIGEN_DEVICE_FUNC - const PaddingDimensions& padding() const { return m_padding_dims; } - EIGEN_DEVICE_FUNC - Scalar padding_value() const { return m_padding_value; } + EIGEN_DEVICE_FUNC + const PaddingDimensions &padding() const { return m_padding_dims; } + EIGEN_DEVICE_FUNC + Scalar padding_value() const { return m_padding_value; } - EIGEN_DEVICE_FUNC - const typename internal::remove_all::type& - expression() const { return m_xpr; } + EIGEN_DEVICE_FUNC + const typename internal::remove_all::type &expression() const { return m_xpr; } - protected: - typename XprType::Nested m_xpr; - const PaddingDimensions m_padding_dims; - const Scalar m_padding_value; +protected: + typename XprType::Nested m_xpr; + const PaddingDimensions m_padding_dims; + const Scalar m_padding_value; }; @@ -100,8 +103,8 @@ struct TensorEvaluator, Device RawAccess = false }; - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TensorEvaluator(const XprType& op, const Device& device) - : m_impl(op.expression(), device), m_padding(op.padding()), m_paddingValue(op.padding_value()) + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TensorEvaluator(const XprType &op, const Device &device) + : m_impl(op.expression(), device), m_padding(op.padding()), m_paddingValue(op.padding_value()) { // The padding op doesn't change the rank of the tensor. Directly padding a scalar would lead // to a vector, which doesn't make sense. Instead one should reshape the scalar into a vector @@ -110,38 +113,35 @@ struct TensorEvaluator, Device // Compute dimensions m_dimensions = m_impl.dimensions(); - for (int i = 0; i < NumDims; ++i) { - m_dimensions[i] += m_padding[i].first + m_padding[i].second; - } - const typename TensorEvaluator::Dimensions& input_dims = m_impl.dimensions(); + for (int i = 0; i < NumDims; ++i) { m_dimensions[i] += m_padding[i].first + m_padding[i].second; } + const typename TensorEvaluator::Dimensions &input_dims = m_impl.dimensions(); if (static_cast(Layout) == static_cast(ColMajor)) { m_inputStrides[0] = 1; m_outputStrides[0] = 1; for (int i = 1; i < NumDims; ++i) { - m_inputStrides[i] = m_inputStrides[i-1] * input_dims[i-1]; - m_outputStrides[i] = m_outputStrides[i-1] * m_dimensions[i-1]; + m_inputStrides[i] = m_inputStrides[i - 1] * input_dims[i - 1]; + m_outputStrides[i] = m_outputStrides[i - 1] * m_dimensions[i - 1]; } - m_outputStrides[NumDims] = m_outputStrides[NumDims-1] * m_dimensions[NumDims-1]; + m_outputStrides[NumDims] = m_outputStrides[NumDims - 1] * m_dimensions[NumDims - 1]; } else { m_inputStrides[NumDims - 1] = 1; m_outputStrides[NumDims] = 1; for (int i = NumDims - 2; i >= 0; --i) { - m_inputStrides[i] = m_inputStrides[i+1] * input_dims[i+1]; - m_outputStrides[i+1] = m_outputStrides[i+2] * m_dimensions[i+1]; + m_inputStrides[i] = m_inputStrides[i + 1] * input_dims[i + 1]; + m_outputStrides[i + 1] = m_outputStrides[i + 2] * m_dimensions[i + 1]; } m_outputStrides[0] = m_outputStrides[1] * m_dimensions[0]; } } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const Dimensions& dimensions() const { return m_dimensions; } + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const Dimensions &dimensions() const { return m_dimensions; } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE bool evalSubExprsIfNeeded(Scalar*) { + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE bool evalSubExprsIfNeeded(Scalar *) + { m_impl.evalSubExprsIfNeeded(NULL); return true; } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void cleanup() { - m_impl.cleanup(); - } + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void cleanup() { m_impl.cleanup(); } EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE CoeffReturnType coeff(Index index) const { @@ -150,72 +150,59 @@ struct TensorEvaluator, Device if (static_cast(Layout) == static_cast(ColMajor)) { for (int i = NumDims - 1; i > 0; --i) { const Index idx = index / m_outputStrides[i]; - if (isPaddingAtIndexForDim(idx, i)) { - return m_paddingValue; - } + if (isPaddingAtIndexForDim(idx, i)) { return m_paddingValue; } inputIndex += (idx - m_padding[i].first) * m_inputStrides[i]; index -= idx * m_outputStrides[i]; } - if (isPaddingAtIndexForDim(index, 0)) { - return m_paddingValue; - } + if (isPaddingAtIndexForDim(index, 0)) { return m_paddingValue; } inputIndex += (index - m_padding[0].first); } else { for (int i = 0; i < NumDims - 1; ++i) { - const Index idx = index / m_outputStrides[i+1]; - if (isPaddingAtIndexForDim(idx, i)) { - return m_paddingValue; - } + const Index idx = index / m_outputStrides[i + 1]; + if (isPaddingAtIndexForDim(idx, i)) { return m_paddingValue; } inputIndex += (idx - m_padding[i].first) * m_inputStrides[i]; - index -= idx * m_outputStrides[i+1]; - } - if (isPaddingAtIndexForDim(index, NumDims-1)) { - return m_paddingValue; + index -= idx * m_outputStrides[i + 1]; } - inputIndex += (index - m_padding[NumDims-1].first); + if (isPaddingAtIndexForDim(index, NumDims - 1)) { return m_paddingValue; } + inputIndex += (index - m_padding[NumDims - 1].first); } return m_impl.coeff(inputIndex); } - template - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE PacketReturnType packet(Index index) const + template EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE PacketReturnType packet(Index index) const { - if (static_cast(Layout) == static_cast(ColMajor)) { - return packetColMajor(index); - } + if (static_cast(Layout) == static_cast(ColMajor)) { return packetColMajor(index); } return packetRowMajor(index); } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TensorOpCost costPerCoeff(bool vectorized) const { + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TensorOpCost costPerCoeff(bool vectorized) const + { TensorOpCost cost = m_impl.costPerCoeff(vectorized); if (static_cast(Layout) == static_cast(ColMajor)) { - for (int i = 0; i < NumDims; ++i) - updateCostPerDimension(cost, i, i == 0); + for (int i = 0; i < NumDims; ++i) updateCostPerDimension(cost, i, i == 0); } else { - for (int i = NumDims - 1; i >= 0; --i) - updateCostPerDimension(cost, i, i == NumDims - 1); + for (int i = NumDims - 1; i >= 0; --i) updateCostPerDimension(cost, i, i == NumDims - 1); } return cost; } - EIGEN_DEVICE_FUNC Scalar* data() const { return NULL; } + EIGEN_DEVICE_FUNC Scalar *data() const { return NULL; } - private: - EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE bool isPaddingAtIndexForDim( - Index index, int dim_index) const { +private: + EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE bool isPaddingAtIndexForDim(Index index, int dim_index) const + { #if defined(EIGEN_HAS_INDEX_LIST) - return (!internal::index_pair_first_statically_eq(dim_index, 0) && - index < m_padding[dim_index].first) || - (!internal::index_pair_second_statically_eq(dim_index, 0) && - index >= m_dimensions[dim_index] - m_padding[dim_index].second); + return (!internal::index_pair_first_statically_eq(dim_index, 0) + && index < m_padding[dim_index].first) + || (!internal::index_pair_second_statically_eq(dim_index, 0) + && index >= m_dimensions[dim_index] - m_padding[dim_index].second); #else - return (index < m_padding[dim_index].first) || - (index >= m_dimensions[dim_index] - m_padding[dim_index].second); + return (index < m_padding[dim_index].first) || (index >= m_dimensions[dim_index] - m_padding[dim_index].second); #endif } - EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE bool isLeftPaddingCompileTimeZero( - int dim_index) const { + EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE bool isLeftPaddingCompileTimeZero(int dim_index) const + { #if defined(EIGEN_HAS_INDEX_LIST) return internal::index_pair_first_statically_eq(dim_index, 0); #else @@ -224,8 +211,8 @@ struct TensorEvaluator, Device #endif } - EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE bool isRightPaddingCompileTimeZero( - int dim_index) const { + EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE bool isRightPaddingCompileTimeZero(int dim_index) const + { #if defined(EIGEN_HAS_INDEX_LIST) return internal::index_pair_second_statically_eq(dim_index, 0); #else @@ -235,30 +222,28 @@ struct TensorEvaluator, Device } - void updateCostPerDimension(TensorOpCost& cost, int i, bool first) const { + void updateCostPerDimension(TensorOpCost &cost, int i, bool first) const + { const double in = static_cast(m_impl.dimensions()[i]); const double out = in + m_padding[i].first + m_padding[i].second; - if (out == 0) - return; + if (out == 0) return; const double reduction = in / out; cost *= reduction; if (first) { - cost += TensorOpCost(0, 0, 2 * TensorOpCost::AddCost() + - reduction * (1 * TensorOpCost::AddCost())); + cost += TensorOpCost(0, 0, 2 * TensorOpCost::AddCost() + reduction * (1 * TensorOpCost::AddCost())); } else { - cost += TensorOpCost(0, 0, 2 * TensorOpCost::AddCost() + - 2 * TensorOpCost::MulCost() + - reduction * (2 * TensorOpCost::MulCost() + - 1 * TensorOpCost::DivCost())); + cost += TensorOpCost(0, + 0, + 2 * TensorOpCost::AddCost() + 2 * TensorOpCost::MulCost() + + reduction * (2 * TensorOpCost::MulCost() + 1 * TensorOpCost::DivCost())); } } - protected: - +protected: EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE PacketReturnType packetColMajor(Index index) const { EIGEN_STATIC_ASSERT((PacketSize > 1), YOU_MADE_A_PROGRAMMING_MISTAKE) - eigen_assert(index+PacketSize-1 < dimensions().TotalSize()); + eigen_assert(index + PacketSize - 1 < dimensions().TotalSize()); const Index initialIndex = index; Index inputIndex = 0; @@ -267,23 +252,21 @@ struct TensorEvaluator, Device const Index last = index + PacketSize - 1; const Index lastPaddedLeft = m_padding[i].first * m_outputStrides[i]; const Index firstPaddedRight = (m_dimensions[i] - m_padding[i].second) * m_outputStrides[i]; - const Index lastPaddedRight = m_outputStrides[i+1]; + const Index lastPaddedRight = m_outputStrides[i + 1]; if (!isLeftPaddingCompileTimeZero(i) && last < lastPaddedLeft) { // all the coefficient are in the padding zone. return internal::pset1(m_paddingValue); - } - else if (!isRightPaddingCompileTimeZero(i) && first >= firstPaddedRight && last < lastPaddedRight) { + } else if (!isRightPaddingCompileTimeZero(i) && first >= firstPaddedRight && last < lastPaddedRight) { // all the coefficient are in the padding zone. return internal::pset1(m_paddingValue); - } - else if ((isLeftPaddingCompileTimeZero(i) && isRightPaddingCompileTimeZero(i)) || (first >= lastPaddedLeft && last < firstPaddedRight)) { + } else if ((isLeftPaddingCompileTimeZero(i) && isRightPaddingCompileTimeZero(i)) + || (first >= lastPaddedLeft && last < firstPaddedRight)) { // all the coefficient are between the 2 padding zones. const Index idx = index / m_outputStrides[i]; inputIndex += (idx - m_padding[i].first) * m_inputStrides[i]; index -= idx * m_outputStrides[i]; - } - else { + } else { // Every other case return packetWithPossibleZero(initialIndex); } @@ -298,12 +281,11 @@ struct TensorEvaluator, Device if (!isLeftPaddingCompileTimeZero(0) && last < lastPaddedLeft) { // all the coefficient are in the padding zone. return internal::pset1(m_paddingValue); - } - else if (!isRightPaddingCompileTimeZero(0) && first >= firstPaddedRight && last < lastPaddedRight) { + } else if (!isRightPaddingCompileTimeZero(0) && first >= firstPaddedRight && last < lastPaddedRight) { // all the coefficient are in the padding zone. return internal::pset1(m_paddingValue); - } - else if ((isLeftPaddingCompileTimeZero(0) && isRightPaddingCompileTimeZero(0)) || (first >= lastPaddedLeft && last < firstPaddedRight)) { + } else if ((isLeftPaddingCompileTimeZero(0) && isRightPaddingCompileTimeZero(0)) + || (first >= lastPaddedLeft && last < firstPaddedRight)) { // all the coefficient are between the 2 padding zones. inputIndex += (index - m_padding[0].first); return m_impl.template packet(inputIndex); @@ -315,7 +297,7 @@ struct TensorEvaluator, Device EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE PacketReturnType packetRowMajor(Index index) const { EIGEN_STATIC_ASSERT((PacketSize > 1), YOU_MADE_A_PROGRAMMING_MISTAKE) - eigen_assert(index+PacketSize-1 < dimensions().TotalSize()); + eigen_assert(index + PacketSize - 1 < dimensions().TotalSize()); const Index initialIndex = index; Index inputIndex = 0; @@ -323,25 +305,23 @@ struct TensorEvaluator, Device for (int i = 0; i < NumDims - 1; ++i) { const Index first = index; const Index last = index + PacketSize - 1; - const Index lastPaddedLeft = m_padding[i].first * m_outputStrides[i+1]; - const Index firstPaddedRight = (m_dimensions[i] - m_padding[i].second) * m_outputStrides[i+1]; + const Index lastPaddedLeft = m_padding[i].first * m_outputStrides[i + 1]; + const Index firstPaddedRight = (m_dimensions[i] - m_padding[i].second) * m_outputStrides[i + 1]; const Index lastPaddedRight = m_outputStrides[i]; if (!isLeftPaddingCompileTimeZero(i) && last < lastPaddedLeft) { // all the coefficient are in the padding zone. return internal::pset1(m_paddingValue); - } - else if (!isRightPaddingCompileTimeZero(i) && first >= firstPaddedRight && last < lastPaddedRight) { + } else if (!isRightPaddingCompileTimeZero(i) && first >= firstPaddedRight && last < lastPaddedRight) { // all the coefficient are in the padding zone. return internal::pset1(m_paddingValue); - } - else if ((isLeftPaddingCompileTimeZero(i) && isRightPaddingCompileTimeZero(i)) || (first >= lastPaddedLeft && last < firstPaddedRight)) { + } else if ((isLeftPaddingCompileTimeZero(i) && isRightPaddingCompileTimeZero(i)) + || (first >= lastPaddedLeft && last < firstPaddedRight)) { // all the coefficient are between the 2 padding zones. - const Index idx = index / m_outputStrides[i+1]; + const Index idx = index / m_outputStrides[i + 1]; inputIndex += (idx - m_padding[i].first) * m_inputStrides[i]; - index -= idx * m_outputStrides[i+1]; - } - else { + index -= idx * m_outputStrides[i + 1]; + } else { // Every other case return packetWithPossibleZero(initialIndex); } @@ -349,21 +329,20 @@ struct TensorEvaluator, Device const Index last = index + PacketSize - 1; const Index first = index; - const Index lastPaddedLeft = m_padding[NumDims-1].first; - const Index firstPaddedRight = (m_dimensions[NumDims-1] - m_padding[NumDims-1].second); - const Index lastPaddedRight = m_outputStrides[NumDims-1]; + const Index lastPaddedLeft = m_padding[NumDims - 1].first; + const Index firstPaddedRight = (m_dimensions[NumDims - 1] - m_padding[NumDims - 1].second); + const Index lastPaddedRight = m_outputStrides[NumDims - 1]; - if (!isLeftPaddingCompileTimeZero(NumDims-1) && last < lastPaddedLeft) { + if (!isLeftPaddingCompileTimeZero(NumDims - 1) && last < lastPaddedLeft) { // all the coefficient are in the padding zone. return internal::pset1(m_paddingValue); - } - else if (!isRightPaddingCompileTimeZero(NumDims-1) && first >= firstPaddedRight && last < lastPaddedRight) { + } else if (!isRightPaddingCompileTimeZero(NumDims - 1) && first >= firstPaddedRight && last < lastPaddedRight) { // all the coefficient are in the padding zone. return internal::pset1(m_paddingValue); - } - else if ((isLeftPaddingCompileTimeZero(NumDims-1) && isRightPaddingCompileTimeZero(NumDims-1)) || (first >= lastPaddedLeft && last < firstPaddedRight)) { + } else if ((isLeftPaddingCompileTimeZero(NumDims - 1) && isRightPaddingCompileTimeZero(NumDims - 1)) + || (first >= lastPaddedLeft && last < firstPaddedRight)) { // all the coefficient are between the 2 padding zones. - inputIndex += (index - m_padding[NumDims-1].first); + inputIndex += (index - m_padding[NumDims - 1].first); return m_impl.template packet(inputIndex); } // Every other case @@ -373,15 +352,13 @@ struct TensorEvaluator, Device EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE PacketReturnType packetWithPossibleZero(Index index) const { EIGEN_ALIGN_MAX typename internal::remove_const::type values[PacketSize]; - for (int i = 0; i < PacketSize; ++i) { - values[i] = coeff(index+i); - } + for (int i = 0; i < PacketSize; ++i) { values[i] = coeff(index + i); } PacketReturnType rslt = internal::pload(values); return rslt; } Dimensions m_dimensions; - array m_outputStrides; + array m_outputStrides; array m_inputStrides; TensorEvaluator m_impl; PaddingDimensions m_padding; @@ -390,8 +367,6 @@ struct TensorEvaluator, Device }; +}// end namespace Eigen - -} // end namespace Eigen - -#endif // EIGEN_CXX11_TENSOR_TENSOR_PADDING_H +#endif// EIGEN_CXX11_TENSOR_TENSOR_PADDING_H diff --git a/filmulator-gui/core/nlmeans/eigen/unsupported/Eigen/CXX11/src/Tensor/TensorPatch.h b/filmulator-gui/core/nlmeans/eigen/unsupported/Eigen/CXX11/src/Tensor/TensorPatch.h index 886a254f..4371b621 100644 --- a/filmulator-gui/core/nlmeans/eigen/unsupported/Eigen/CXX11/src/Tensor/TensorPatch.h +++ b/filmulator-gui/core/nlmeans/eigen/unsupported/Eigen/CXX11/src/Tensor/TensorPatch.h @@ -13,46 +13,43 @@ namespace Eigen { /** \class TensorPatch - * \ingroup CXX11_Tensor_Module - * - * \brief Tensor patch class. - * - * - */ + * \ingroup CXX11_Tensor_Module + * + * \brief Tensor patch class. + * + * + */ namespace internal { -template -struct traits > : public traits -{ - typedef typename XprType::Scalar Scalar; - typedef traits XprTraits; - typedef typename XprTraits::StorageKind StorageKind; - typedef typename XprTraits::Index Index; - typedef typename XprType::Nested Nested; - typedef typename remove_reference::type _Nested; - static const int NumDimensions = XprTraits::NumDimensions + 1; - static const int Layout = XprTraits::Layout; -}; - -template -struct eval, Eigen::Dense> -{ - typedef const TensorPatchOp& type; -}; - -template -struct nested, 1, typename eval >::type> -{ - typedef TensorPatchOp type; -}; + template struct traits> : public traits + { + typedef typename XprType::Scalar Scalar; + typedef traits XprTraits; + typedef typename XprTraits::StorageKind StorageKind; + typedef typename XprTraits::Index Index; + typedef typename XprType::Nested Nested; + typedef typename remove_reference::type _Nested; + static const int NumDimensions = XprTraits::NumDimensions + 1; + static const int Layout = XprTraits::Layout; + }; + + template struct eval, Eigen::Dense> + { + typedef const TensorPatchOp &type; + }; -} // end namespace internal + template + struct nested, 1, typename eval>::type> + { + typedef TensorPatchOp type; + }; +}// end namespace internal template class TensorPatchOp : public TensorBase, ReadOnlyAccessors> { - public: +public: typedef typename Eigen::internal::traits::Scalar Scalar; typedef typename Eigen::NumTraits::Real RealScalar; typedef typename XprType::CoeffReturnType CoeffReturnType; @@ -60,19 +57,19 @@ class TensorPatchOp : public TensorBase, ReadOn typedef typename Eigen::internal::traits::StorageKind StorageKind; typedef typename Eigen::internal::traits::Index Index; - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TensorPatchOp(const XprType& expr, const PatchDim& patch_dims) - : m_xpr(expr), m_patch_dims(patch_dims) {} + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TensorPatchOp(const XprType &expr, const PatchDim &patch_dims) + : m_xpr(expr), m_patch_dims(patch_dims) + {} - EIGEN_DEVICE_FUNC - const PatchDim& patch_dims() const { return m_patch_dims; } + EIGEN_DEVICE_FUNC + const PatchDim &patch_dims() const { return m_patch_dims; } - EIGEN_DEVICE_FUNC - const typename internal::remove_all::type& - expression() const { return m_xpr; } + EIGEN_DEVICE_FUNC + const typename internal::remove_all::type &expression() const { return m_xpr; } - protected: - typename XprType::Nested m_xpr; - const PatchDim m_patch_dims; +protected: + typename XprType::Nested m_xpr; + const PatchDim m_patch_dims; }; @@ -96,61 +93,56 @@ struct TensorEvaluator, Device> Layout = TensorEvaluator::Layout, CoordAccess = false, RawAccess = false - }; + }; - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TensorEvaluator(const XprType& op, const Device& device) - : m_impl(op.expression(), device) + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TensorEvaluator(const XprType &op, const Device &device) + : m_impl(op.expression(), device) { Index num_patches = 1; - const typename TensorEvaluator::Dimensions& input_dims = m_impl.dimensions(); - const PatchDim& patch_dims = op.patch_dims(); + const typename TensorEvaluator::Dimensions &input_dims = m_impl.dimensions(); + const PatchDim &patch_dims = op.patch_dims(); if (static_cast(Layout) == static_cast(ColMajor)) { - for (int i = 0; i < NumDims-1; ++i) { + for (int i = 0; i < NumDims - 1; ++i) { m_dimensions[i] = patch_dims[i]; num_patches *= (input_dims[i] - patch_dims[i] + 1); } - m_dimensions[NumDims-1] = num_patches; + m_dimensions[NumDims - 1] = num_patches; m_inputStrides[0] = 1; m_patchStrides[0] = 1; - for (int i = 1; i < NumDims-1; ++i) { - m_inputStrides[i] = m_inputStrides[i-1] * input_dims[i-1]; - m_patchStrides[i] = m_patchStrides[i-1] * (input_dims[i-1] - patch_dims[i-1] + 1); + for (int i = 1; i < NumDims - 1; ++i) { + m_inputStrides[i] = m_inputStrides[i - 1] * input_dims[i - 1]; + m_patchStrides[i] = m_patchStrides[i - 1] * (input_dims[i - 1] - patch_dims[i - 1] + 1); } m_outputStrides[0] = 1; - for (int i = 1; i < NumDims; ++i) { - m_outputStrides[i] = m_outputStrides[i-1] * m_dimensions[i-1]; - } + for (int i = 1; i < NumDims; ++i) { m_outputStrides[i] = m_outputStrides[i - 1] * m_dimensions[i - 1]; } } else { - for (int i = 0; i < NumDims-1; ++i) { - m_dimensions[i+1] = patch_dims[i]; + for (int i = 0; i < NumDims - 1; ++i) { + m_dimensions[i + 1] = patch_dims[i]; num_patches *= (input_dims[i] - patch_dims[i] + 1); } m_dimensions[0] = num_patches; - m_inputStrides[NumDims-2] = 1; - m_patchStrides[NumDims-2] = 1; - for (int i = NumDims-3; i >= 0; --i) { - m_inputStrides[i] = m_inputStrides[i+1] * input_dims[i+1]; - m_patchStrides[i] = m_patchStrides[i+1] * (input_dims[i+1] - patch_dims[i+1] + 1); - } - m_outputStrides[NumDims-1] = 1; - for (int i = NumDims-2; i >= 0; --i) { - m_outputStrides[i] = m_outputStrides[i+1] * m_dimensions[i+1]; + m_inputStrides[NumDims - 2] = 1; + m_patchStrides[NumDims - 2] = 1; + for (int i = NumDims - 3; i >= 0; --i) { + m_inputStrides[i] = m_inputStrides[i + 1] * input_dims[i + 1]; + m_patchStrides[i] = m_patchStrides[i + 1] * (input_dims[i + 1] - patch_dims[i + 1] + 1); } + m_outputStrides[NumDims - 1] = 1; + for (int i = NumDims - 2; i >= 0; --i) { m_outputStrides[i] = m_outputStrides[i + 1] * m_dimensions[i + 1]; } } } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const Dimensions& dimensions() const { return m_dimensions; } + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const Dimensions &dimensions() const { return m_dimensions; } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE bool evalSubExprsIfNeeded(Scalar* /*data*/) { + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE bool evalSubExprsIfNeeded(Scalar * /*data*/) + { m_impl.evalSubExprsIfNeeded(NULL); return true; } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void cleanup() { - m_impl.cleanup(); - } + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void cleanup() { m_impl.cleanup(); } EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE CoeffReturnType coeff(Index index) const { @@ -172,8 +164,8 @@ struct TensorEvaluator, Device> for (int i = 0; i < NumDims - 2; ++i) { const Index patchIdx = patchIndex / m_patchStrides[i]; patchIndex -= patchIdx * m_patchStrides[i]; - const Index offsetIdx = patchOffset / m_outputStrides[i+1]; - patchOffset -= offsetIdx * m_outputStrides[i+1]; + const Index offsetIdx = patchOffset / m_outputStrides[i + 1]; + patchOffset -= offsetIdx * m_outputStrides[i + 1]; inputIndex += (patchIdx + offsetIdx) * m_inputStrides[i]; } } @@ -181,29 +173,26 @@ struct TensorEvaluator, Device> return m_impl.coeff(inputIndex); } - template - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE PacketReturnType packet(Index index) const + template EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE PacketReturnType packet(Index index) const { EIGEN_STATIC_ASSERT((PacketSize > 1), YOU_MADE_A_PROGRAMMING_MISTAKE) - eigen_assert(index+PacketSize-1 < dimensions().TotalSize()); + eigen_assert(index + PacketSize - 1 < dimensions().TotalSize()); Index output_stride_index = (static_cast(Layout) == static_cast(ColMajor)) ? NumDims - 1 : 0; - Index indices[2] = {index, index + PacketSize - 1}; - Index patchIndices[2] = {indices[0] / m_outputStrides[output_stride_index], - indices[1] / m_outputStrides[output_stride_index]}; - Index patchOffsets[2] = {indices[0] - patchIndices[0] * m_outputStrides[output_stride_index], - indices[1] - patchIndices[1] * m_outputStrides[output_stride_index]}; + Index indices[2] = { index, index + PacketSize - 1 }; + Index patchIndices[2] = { indices[0] / m_outputStrides[output_stride_index], + indices[1] / m_outputStrides[output_stride_index] }; + Index patchOffsets[2] = { indices[0] - patchIndices[0] * m_outputStrides[output_stride_index], + indices[1] - patchIndices[1] * m_outputStrides[output_stride_index] }; - Index inputIndices[2] = {0, 0}; + Index inputIndices[2] = { 0, 0 }; if (static_cast(Layout) == static_cast(ColMajor)) { for (int i = NumDims - 2; i > 0; --i) { - const Index patchIdx[2] = {patchIndices[0] / m_patchStrides[i], - patchIndices[1] / m_patchStrides[i]}; + const Index patchIdx[2] = { patchIndices[0] / m_patchStrides[i], patchIndices[1] / m_patchStrides[i] }; patchIndices[0] -= patchIdx[0] * m_patchStrides[i]; patchIndices[1] -= patchIdx[1] * m_patchStrides[i]; - const Index offsetIdx[2] = {patchOffsets[0] / m_outputStrides[i], - patchOffsets[1] / m_outputStrides[i]}; + const Index offsetIdx[2] = { patchOffsets[0] / m_outputStrides[i], patchOffsets[1] / m_outputStrides[i] }; patchOffsets[0] -= offsetIdx[0] * m_outputStrides[i]; patchOffsets[1] -= offsetIdx[1] * m_outputStrides[i]; @@ -212,15 +201,14 @@ struct TensorEvaluator, Device> } } else { for (int i = 0; i < NumDims - 2; ++i) { - const Index patchIdx[2] = {patchIndices[0] / m_patchStrides[i], - patchIndices[1] / m_patchStrides[i]}; + const Index patchIdx[2] = { patchIndices[0] / m_patchStrides[i], patchIndices[1] / m_patchStrides[i] }; patchIndices[0] -= patchIdx[0] * m_patchStrides[i]; patchIndices[1] -= patchIdx[1] * m_patchStrides[i]; - const Index offsetIdx[2] = {patchOffsets[0] / m_outputStrides[i+1], - patchOffsets[1] / m_outputStrides[i+1]}; - patchOffsets[0] -= offsetIdx[0] * m_outputStrides[i+1]; - patchOffsets[1] -= offsetIdx[1] * m_outputStrides[i+1]; + const Index offsetIdx[2] = { patchOffsets[0] / m_outputStrides[i + 1], + patchOffsets[1] / m_outputStrides[i + 1] }; + patchOffsets[0] -= offsetIdx[0] * m_outputStrides[i + 1]; + patchOffsets[1] -= offsetIdx[1] * m_outputStrides[i + 1]; inputIndices[0] += (patchIdx[0] + offsetIdx[0]) * m_inputStrides[i]; inputIndices[1] += (patchIdx[1] + offsetIdx[1]) * m_inputStrides[i]; @@ -232,38 +220,34 @@ struct TensorEvaluator, Device> if (inputIndices[1] - inputIndices[0] == PacketSize - 1) { PacketReturnType rslt = m_impl.template packet(inputIndices[0]); return rslt; - } - else { + } else { EIGEN_ALIGN_MAX CoeffReturnType values[PacketSize]; values[0] = m_impl.coeff(inputIndices[0]); - values[PacketSize-1] = m_impl.coeff(inputIndices[1]); - for (int i = 1; i < PacketSize-1; ++i) { - values[i] = coeff(index+i); - } + values[PacketSize - 1] = m_impl.coeff(inputIndices[1]); + for (int i = 1; i < PacketSize - 1; ++i) { values[i] = coeff(index + i); } PacketReturnType rslt = internal::pload(values); return rslt; } } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TensorOpCost costPerCoeff(bool vectorized) const { - const double compute_cost = NumDims * (TensorOpCost::DivCost() + - TensorOpCost::MulCost() + - 2 * TensorOpCost::AddCost()); - return m_impl.costPerCoeff(vectorized) + - TensorOpCost(0, 0, compute_cost, vectorized, PacketSize); + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TensorOpCost costPerCoeff(bool vectorized) const + { + const double compute_cost = + NumDims * (TensorOpCost::DivCost() + TensorOpCost::MulCost() + 2 * TensorOpCost::AddCost()); + return m_impl.costPerCoeff(vectorized) + TensorOpCost(0, 0, compute_cost, vectorized, PacketSize); } - EIGEN_DEVICE_FUNC Scalar* data() const { return NULL; } + EIGEN_DEVICE_FUNC Scalar *data() const { return NULL; } - protected: +protected: Dimensions m_dimensions; array m_outputStrides; - array m_inputStrides; - array m_patchStrides; + array m_inputStrides; + array m_patchStrides; TensorEvaluator m_impl; }; -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_CXX11_TENSOR_TENSOR_PATCH_H +#endif// EIGEN_CXX11_TENSOR_TENSOR_PATCH_H diff --git a/filmulator-gui/core/nlmeans/eigen/unsupported/Eigen/CXX11/src/Tensor/TensorRandom.h b/filmulator-gui/core/nlmeans/eigen/unsupported/Eigen/CXX11/src/Tensor/TensorRandom.h index 1655a813..92599320 100644 --- a/filmulator-gui/core/nlmeans/eigen/unsupported/Eigen/CXX11/src/Tensor/TensorRandom.h +++ b/filmulator-gui/core/nlmeans/eigen/unsupported/Eigen/CXX11/src/Tensor/TensorRandom.h @@ -13,264 +13,259 @@ namespace Eigen { namespace internal { -namespace { + namespace { -EIGEN_DEVICE_FUNC uint64_t get_random_seed() { + EIGEN_DEVICE_FUNC uint64_t get_random_seed() + { #ifdef __CUDA_ARCH__ - // We don't support 3d kernels since we currently only use 1 and - // 2d kernels. - assert(threadIdx.z == 0); - return clock64() + - blockIdx.x * blockDim.x + threadIdx.x + - gridDim.x * blockDim.x * (blockIdx.y * blockDim.y + threadIdx.y); + // We don't support 3d kernels since we currently only use 1 and + // 2d kernels. + assert(threadIdx.z == 0); + return clock64() + blockIdx.x * blockDim.x + threadIdx.x + + gridDim.x * blockDim.x * (blockIdx.y * blockDim.y + threadIdx.y); #elif defined _WIN32 - // Use the current time as a baseline. - SYSTEMTIME st; - GetSystemTime(&st); - int time = st.wSecond + 1000 * st.wMilliseconds; - // Mix in a random number to make sure that we get different seeds if - // we try to generate seeds faster than the clock resolution. - // We need 2 random values since the generator only generate 16 bits at - // a time (https://msdn.microsoft.com/en-us/library/398ax69y.aspx) - int rnd1 = ::rand(); - int rnd2 = ::rand(); - uint64_t rnd = (rnd1 | rnd2 << 16) ^ time; - return rnd; + // Use the current time as a baseline. + SYSTEMTIME st; + GetSystemTime(&st); + int time = st.wSecond + 1000 * st.wMilliseconds; + // Mix in a random number to make sure that we get different seeds if + // we try to generate seeds faster than the clock resolution. + // We need 2 random values since the generator only generate 16 bits at + // a time (https://msdn.microsoft.com/en-us/library/398ax69y.aspx) + int rnd1 = ::rand(); + int rnd2 = ::rand(); + uint64_t rnd = (rnd1 | rnd2 << 16) ^ time; + return rnd; #elif defined __APPLE__ - // Same approach as for win32, except that the random number generator - // is better (// https://developer.apple.com/legacy/library/documentation/Darwin/Reference/ManPages/man3/random.3.html#//apple_ref/doc/man/3/random). - uint64_t rnd = ::random() ^ mach_absolute_time(); - return rnd; + // Same approach as for win32, except that the random number generator + // is better (// + // https://developer.apple.com/legacy/library/documentation/Darwin/Reference/ManPages/man3/random.3.html#//apple_ref/doc/man/3/random). + uint64_t rnd = ::random() ^ mach_absolute_time(); + return rnd; #else - // Augment the current time with pseudo random number generation - // to ensure that we get different seeds if we try to generate seeds - // faster than the clock resolution. - timespec ts; - clock_gettime(CLOCK_REALTIME, &ts); - uint64_t rnd = ::random() ^ ts.tv_nsec; - return rnd; + // Augment the current time with pseudo random number generation + // to ensure that we get different seeds if we try to generate seeds + // faster than the clock resolution. + timespec ts; + clock_gettime(CLOCK_REALTIME, &ts); + uint64_t rnd = ::random() ^ ts.tv_nsec; + return rnd; #endif -} - -static EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE unsigned PCG_XSH_RS_generator(uint64_t* state) { - // TODO: Unify with the implementation in the non blocking thread pool. - uint64_t current = *state; - // Update the internal state - *state = current * 6364136223846793005ULL + 0xda3e39cb94b95bdbULL; - // Generate the random output (using the PCG-XSH-RS scheme) - return static_cast((current ^ (current >> 22)) >> (22 + (current >> 61))); -} - -static EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE uint64_t PCG_XSH_RS_state(uint64_t seed) { - seed = seed ? seed : get_random_seed(); - return seed * 6364136223846793005ULL + 0xda3e39cb94b95bdbULL; -} - -} // namespace - - -template EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE -T RandomToTypeUniform(uint64_t* state) { - unsigned rnd = PCG_XSH_RS_generator(state); - return static_cast(rnd); -} - - -template <> EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE -Eigen::half RandomToTypeUniform(uint64_t* state) { - Eigen::half result; - // Generate 10 random bits for the mantissa - unsigned rnd = PCG_XSH_RS_generator(state); - result.x = static_cast(rnd & 0x3ffu); - // Set the exponent - result.x |= (static_cast(15) << 10); - // Return the final result - return result - Eigen::half(1.0f); -} - - -template <> EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE -float RandomToTypeUniform(uint64_t* state) { - typedef union { - uint32_t raw; - float fp; - } internal; - internal result; - // Generate 23 random bits for the mantissa mantissa - const unsigned rnd = PCG_XSH_RS_generator(state); - result.raw = rnd & 0x7fffffu; - // Set the exponent - result.raw |= (static_cast(127) << 23); - // Return the final result - return result.fp - 1.0f; -} - -template <> EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE -double RandomToTypeUniform(uint64_t* state) { - typedef union { - uint64_t raw; - double dp; - } internal; - internal result; - result.raw = 0; - // Generate 52 random bits for the mantissa - // First generate the upper 20 bits - unsigned rnd1 = PCG_XSH_RS_generator(state) & 0xfffffu; - // The generate the lower 32 bits - unsigned rnd2 = PCG_XSH_RS_generator(state); - result.raw = (static_cast(rnd1) << 32) | rnd2; - // Set the exponent - result.raw |= (static_cast(1023) << 52); - // Return the final result - return result.dp - 1.0; -} - -template <> EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE -std::complex RandomToTypeUniform >(uint64_t* state) { - return std::complex(RandomToTypeUniform(state), - RandomToTypeUniform(state)); -} -template <> EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE -std::complex RandomToTypeUniform >(uint64_t* state) { - return std::complex(RandomToTypeUniform(state), - RandomToTypeUniform(state)); -} - -template class UniformRandomGenerator { - public: - static const bool PacketAccess = true; - - // Uses the given "seed" if non-zero, otherwise uses a random seed. - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE UniformRandomGenerator( - uint64_t seed = 0) { - m_state = PCG_XSH_RS_state(seed); + } + + static EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE unsigned PCG_XSH_RS_generator(uint64_t *state) + { + // TODO: Unify with the implementation in the non blocking thread pool. + uint64_t current = *state; + // Update the internal state + *state = current * 6364136223846793005ULL + 0xda3e39cb94b95bdbULL; + // Generate the random output (using the PCG-XSH-RS scheme) + return static_cast((current ^ (current >> 22)) >> (22 + (current >> 61))); + } + + static EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE uint64_t PCG_XSH_RS_state(uint64_t seed) + { + seed = seed ? seed : get_random_seed(); + return seed * 6364136223846793005ULL + 0xda3e39cb94b95bdbULL; + } + + }// namespace + + + template EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE T RandomToTypeUniform(uint64_t *state) + { + unsigned rnd = PCG_XSH_RS_generator(state); + return static_cast(rnd); } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE UniformRandomGenerator( - const UniformRandomGenerator& other) { - m_state = other.m_state; + + + template<> EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Eigen::half RandomToTypeUniform(uint64_t *state) + { + Eigen::half result; + // Generate 10 random bits for the mantissa + unsigned rnd = PCG_XSH_RS_generator(state); + result.x = static_cast(rnd & 0x3ffu); + // Set the exponent + result.x |= (static_cast(15) << 10); + // Return the final result + return result - Eigen::half(1.0f); } - template EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE - T operator()(Index i) const { - uint64_t local_state = m_state + i; - T result = RandomToTypeUniform(&local_state); - m_state = local_state; - return result; + + template<> EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE float RandomToTypeUniform(uint64_t *state) + { + typedef union { + uint32_t raw; + float fp; + } internal; + internal result; + // Generate 23 random bits for the mantissa mantissa + const unsigned rnd = PCG_XSH_RS_generator(state); + result.raw = rnd & 0x7fffffu; + // Set the exponent + result.raw |= (static_cast(127) << 23); + // Return the final result + return result.fp - 1.0f; } - template EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE - Packet packetOp(Index i) const { - const int packetSize = internal::unpacket_traits::size; - EIGEN_ALIGN_MAX T values[packetSize]; - uint64_t local_state = m_state + i; - for (int j = 0; j < packetSize; ++j) { - values[j] = RandomToTypeUniform(&local_state); - } - m_state = local_state; - return internal::pload(values); + template<> EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE double RandomToTypeUniform(uint64_t *state) + { + typedef union { + uint64_t raw; + double dp; + } internal; + internal result; + result.raw = 0; + // Generate 52 random bits for the mantissa + // First generate the upper 20 bits + unsigned rnd1 = PCG_XSH_RS_generator(state) & 0xfffffu; + // The generate the lower 32 bits + unsigned rnd2 = PCG_XSH_RS_generator(state); + result.raw = (static_cast(rnd1) << 32) | rnd2; + // Set the exponent + result.raw |= (static_cast(1023) << 52); + // Return the final result + return result.dp - 1.0; } - private: - mutable uint64_t m_state; -}; - -template -struct functor_traits > { - enum { - // Rough estimate for floating point, multiplied by ceil(sizeof(T) / sizeof(float)). - Cost = 12 * NumTraits::AddCost * - ((sizeof(Scalar) + sizeof(float) - 1) / sizeof(float)), - PacketAccess = UniformRandomGenerator::PacketAccess - }; -}; - - - -template EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE -T RandomToTypeNormal(uint64_t* state) { - // Use the ratio of uniform method to generate numbers following a normal - // distribution. See for example Numerical Recipes chapter 7.3.9 for the - // details. - T u, v, q; - do { - u = RandomToTypeUniform(state); - v = T(1.7156) * (RandomToTypeUniform(state) - T(0.5)); - const T x = u - T(0.449871); - const T y = numext::abs(v) + T(0.386595); - q = x*x + y * (T(0.196)*y - T(0.25472)*x); - } while (q > T(0.27597) && - (q > T(0.27846) || v*v > T(-4) * numext::log(u) * u*u)); - - return v/u; -} - -template <> EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE -std::complex RandomToTypeNormal >(uint64_t* state) { - return std::complex(RandomToTypeNormal(state), - RandomToTypeNormal(state)); -} -template <> EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE -std::complex RandomToTypeNormal >(uint64_t* state) { - return std::complex(RandomToTypeNormal(state), - RandomToTypeNormal(state)); -} - - -template class NormalRandomGenerator { - public: - static const bool PacketAccess = true; - - // Uses the given "seed" if non-zero, otherwise uses a random seed. - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE NormalRandomGenerator(uint64_t seed = 0) { - m_state = PCG_XSH_RS_state(seed); + template<> + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE std::complex RandomToTypeUniform>(uint64_t *state) + { + return std::complex(RandomToTypeUniform(state), RandomToTypeUniform(state)); } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE NormalRandomGenerator( - const NormalRandomGenerator& other) { - m_state = other.m_state; + template<> + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE std::complex RandomToTypeUniform>(uint64_t *state) + { + return std::complex(RandomToTypeUniform(state), RandomToTypeUniform(state)); } - template EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE - T operator()(Index i) const { - uint64_t local_state = m_state + i; - T result = RandomToTypeNormal(&local_state); - m_state = local_state; - return result; - } + template class UniformRandomGenerator + { + public: + static const bool PacketAccess = true; + + // Uses the given "seed" if non-zero, otherwise uses a random seed. + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE UniformRandomGenerator(uint64_t seed = 0) + { + m_state = PCG_XSH_RS_state(seed); + } + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE UniformRandomGenerator(const UniformRandomGenerator &other) + { + m_state = other.m_state; + } + + template EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE T operator()(Index i) const + { + uint64_t local_state = m_state + i; + T result = RandomToTypeUniform(&local_state); + m_state = local_state; + return result; + } - template EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE - Packet packetOp(Index i) const { - const int packetSize = internal::unpacket_traits::size; - EIGEN_ALIGN_MAX T values[packetSize]; - uint64_t local_state = m_state + i; - for (int j = 0; j < packetSize; ++j) { - values[j] = RandomToTypeNormal(&local_state); + template EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Packet packetOp(Index i) const + { + const int packetSize = internal::unpacket_traits::size; + EIGEN_ALIGN_MAX T values[packetSize]; + uint64_t local_state = m_state + i; + for (int j = 0; j < packetSize; ++j) { values[j] = RandomToTypeUniform(&local_state); } + m_state = local_state; + return internal::pload(values); } - m_state = local_state; - return internal::pload(values); + + private: + mutable uint64_t m_state; + }; + + template struct functor_traits> + { + enum { + // Rough estimate for floating point, multiplied by ceil(sizeof(T) / sizeof(float)). + Cost = 12 * NumTraits::AddCost * ((sizeof(Scalar) + sizeof(float) - 1) / sizeof(float)), + PacketAccess = UniformRandomGenerator::PacketAccess + }; + }; + + + template EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE T RandomToTypeNormal(uint64_t *state) + { + // Use the ratio of uniform method to generate numbers following a normal + // distribution. See for example Numerical Recipes chapter 7.3.9 for the + // details. + T u, v, q; + do { + u = RandomToTypeUniform(state); + v = T(1.7156) * (RandomToTypeUniform(state) - T(0.5)); + const T x = u - T(0.449871); + const T y = numext::abs(v) + T(0.386595); + q = x * x + y * (T(0.196) * y - T(0.25472) * x); + } while (q > T(0.27597) && (q > T(0.27846) || v * v > T(-4) * numext::log(u) * u * u)); + + return v / u; + } + + template<> + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE std::complex RandomToTypeNormal>(uint64_t *state) + { + return std::complex(RandomToTypeNormal(state), RandomToTypeNormal(state)); + } + template<> + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE std::complex RandomToTypeNormal>(uint64_t *state) + { + return std::complex(RandomToTypeNormal(state), RandomToTypeNormal(state)); } - private: - mutable uint64_t m_state; -}; + + template class NormalRandomGenerator + { + public: + static const bool PacketAccess = true; + + // Uses the given "seed" if non-zero, otherwise uses a random seed. + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE NormalRandomGenerator(uint64_t seed = 0) { m_state = PCG_XSH_RS_state(seed); } + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE NormalRandomGenerator(const NormalRandomGenerator &other) + { + m_state = other.m_state; + } + + template EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE T operator()(Index i) const + { + uint64_t local_state = m_state + i; + T result = RandomToTypeNormal(&local_state); + m_state = local_state; + return result; + } + + template EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Packet packetOp(Index i) const + { + const int packetSize = internal::unpacket_traits::size; + EIGEN_ALIGN_MAX T values[packetSize]; + uint64_t local_state = m_state + i; + for (int j = 0; j < packetSize; ++j) { values[j] = RandomToTypeNormal(&local_state); } + m_state = local_state; + return internal::pload(values); + } + + private: + mutable uint64_t m_state; + }; -template -struct functor_traits > { - enum { - // On average, we need to generate about 3 random numbers - // 15 mul, 8 add, 1.5 logs - Cost = 3 * functor_traits >::Cost + - 15 * NumTraits::AddCost + 8 * NumTraits::AddCost + - 3 * functor_traits >::Cost / 2, - PacketAccess = NormalRandomGenerator::PacketAccess + template struct functor_traits> + { + enum { + // On average, we need to generate about 3 random numbers + // 15 mul, 8 add, 1.5 logs + Cost = 3 * functor_traits>::Cost + 15 * NumTraits::AddCost + + 8 * NumTraits::AddCost + 3 * functor_traits>::Cost / 2, + PacketAccess = NormalRandomGenerator::PacketAccess + }; }; -}; -} // end namespace internal -} // end namespace Eigen +}// end namespace internal +}// end namespace Eigen -#endif // EIGEN_CXX11_TENSOR_TENSOR_RANDOM_H +#endif// EIGEN_CXX11_TENSOR_TENSOR_RANDOM_H diff --git a/filmulator-gui/core/nlmeans/eigen/unsupported/Eigen/CXX11/src/Tensor/TensorReduction.h b/filmulator-gui/core/nlmeans/eigen/unsupported/Eigen/CXX11/src/Tensor/TensorReduction.h index 41d0d002..2d2bbbc2 100644 --- a/filmulator-gui/core/nlmeans/eigen/unsupported/Eigen/CXX11/src/Tensor/TensorReduction.h +++ b/filmulator-gui/core/nlmeans/eigen/unsupported/Eigen/CXX11/src/Tensor/TensorReduction.h @@ -14,371 +14,411 @@ namespace Eigen { /** \class TensorReduction - * \ingroup CXX11_Tensor_Module - * - * \brief Tensor reduction class. - * - */ + * \ingroup CXX11_Tensor_Module + * + * \brief Tensor reduction class. + * + */ namespace internal { - template class MakePointer_ > - struct traits > - : traits -{ - typedef traits XprTraits; - typedef typename XprTraits::Scalar Scalar; - typedef typename XprTraits::StorageKind StorageKind; - typedef typename XprTraits::Index Index; - typedef typename XprType::Nested Nested; - static const int NumDimensions = XprTraits::NumDimensions - array_size::value; - static const int Layout = XprTraits::Layout; - - template struct MakePointer { - // Intermediate typedef to workaround MSVC issue. - typedef MakePointer_ MakePointerT; - typedef typename MakePointerT::Type Type; + template class MakePointer_> + struct traits> : traits + { + typedef traits XprTraits; + typedef typename XprTraits::Scalar Scalar; + typedef typename XprTraits::StorageKind StorageKind; + typedef typename XprTraits::Index Index; + typedef typename XprType::Nested Nested; + static const int NumDimensions = XprTraits::NumDimensions - array_size::value; + static const int Layout = XprTraits::Layout; + + template struct MakePointer + { + // Intermediate typedef to workaround MSVC issue. + typedef MakePointer_ MakePointerT; + typedef typename MakePointerT::Type Type; + }; }; -}; -template class MakePointer_> -struct eval, Eigen::Dense> -{ - typedef const TensorReductionOp& type; -}; + template class MakePointer_> + struct eval, Eigen::Dense> + { + typedef const TensorReductionOp &type; + }; -template class MakePointer_> -struct nested, 1, typename eval >::type> -{ - typedef TensorReductionOp type; -}; + template class MakePointer_> + struct nested, + 1, + typename eval>::type> + { + typedef TensorReductionOp type; + }; -template struct DimInitializer { - template EIGEN_DEVICE_FUNC - static void run(const InputDims& input_dims, - const array::value>& reduced, - OutputDims* output_dims, ReducedDims* reduced_dims) { - const int NumInputDims = internal::array_size::value; - int outputIndex = 0; - int reduceIndex = 0; - for (int i = 0; i < NumInputDims; ++i) { - if (reduced[i]) { - (*reduced_dims)[reduceIndex] = input_dims[i]; - ++reduceIndex; - } else { - (*output_dims)[outputIndex] = input_dims[i]; - ++outputIndex; + template struct DimInitializer + { + template + EIGEN_DEVICE_FUNC static void run(const InputDims &input_dims, + const array::value> &reduced, + OutputDims *output_dims, + ReducedDims *reduced_dims) + { + const int NumInputDims = internal::array_size::value; + int outputIndex = 0; + int reduceIndex = 0; + for (int i = 0; i < NumInputDims; ++i) { + if (reduced[i]) { + (*reduced_dims)[reduceIndex] = input_dims[i]; + ++reduceIndex; + } else { + (*output_dims)[outputIndex] = input_dims[i]; + ++outputIndex; + } } } - } -}; + }; -template <> struct DimInitializer > { - template EIGEN_DEVICE_FUNC - static void run(const InputDims& input_dims, const array&, - Sizes<>*, array* reduced_dims) { - const int NumInputDims = internal::array_size::value; - for (int i = 0; i < NumInputDims; ++i) { - (*reduced_dims)[i] = input_dims[i]; + template<> struct DimInitializer> + { + template + EIGEN_DEVICE_FUNC static void + run(const InputDims &input_dims, const array &, Sizes<> *, array *reduced_dims) + { + const int NumInputDims = internal::array_size::value; + for (int i = 0; i < NumInputDims; ++i) { (*reduced_dims)[i] = input_dims[i]; } } - } -}; + }; -template -struct are_inner_most_dims { - static const bool value = false; -}; -template -struct preserve_inner_most_dims { - static const bool value = false; -}; + template struct are_inner_most_dims + { + static const bool value = false; + }; + template struct preserve_inner_most_dims + { + static const bool value = false; + }; #if EIGEN_HAS_CONSTEXPR && EIGEN_HAS_VARIADIC_TEMPLATES -template -struct are_inner_most_dims{ - static const bool tmp1 = indices_statically_known_to_increase(); - static const bool tmp2 = index_statically_eq(0, 0); - static const bool tmp3 = index_statically_eq(array_size::value-1, array_size::value-1); - static const bool value = tmp1 & tmp2 & tmp3; -}; -template -struct are_inner_most_dims{ - static const bool tmp1 = indices_statically_known_to_increase(); - static const bool tmp2 = index_statically_eq(0, NumTensorDims - array_size::value); - static const bool tmp3 = index_statically_eq(array_size::value - 1, NumTensorDims - 1); - static const bool value = tmp1 & tmp2 & tmp3; - -}; -template -struct preserve_inner_most_dims{ - static const bool tmp1 = indices_statically_known_to_increase(); - static const bool tmp2 = index_statically_gt(0, 0); - static const bool value = tmp1 & tmp2; - -}; -template -struct preserve_inner_most_dims{ - static const bool tmp1 = indices_statically_known_to_increase(); - static const bool tmp2 = index_statically_lt(array_size::value - 1, NumTensorDims - 1); - static const bool value = tmp1 & tmp2; -}; + template struct are_inner_most_dims + { + static const bool tmp1 = indices_statically_known_to_increase(); + static const bool tmp2 = index_statically_eq(0, 0); + static const bool tmp3 = + index_statically_eq(array_size::value - 1, array_size::value - 1); + static const bool value = tmp1 & tmp2 & tmp3; + }; + template struct are_inner_most_dims + { + static const bool tmp1 = indices_statically_known_to_increase(); + static const bool tmp2 = index_statically_eq(0, NumTensorDims - array_size::value); + static const bool tmp3 = index_statically_eq(array_size::value - 1, NumTensorDims - 1); + static const bool value = tmp1 & tmp2 & tmp3; + }; + template + struct preserve_inner_most_dims + { + static const bool tmp1 = indices_statically_known_to_increase(); + static const bool tmp2 = index_statically_gt(0, 0); + static const bool value = tmp1 & tmp2; + }; + template + struct preserve_inner_most_dims + { + static const bool tmp1 = indices_statically_known_to_increase(); + static const bool tmp2 = index_statically_lt(array_size::value - 1, NumTensorDims - 1); + static const bool value = tmp1 & tmp2; + }; #endif -template -struct GenericDimReducer { - static EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void reduce(const Self& self, typename Self::Index firstIndex, Op& reducer, typename Self::CoeffReturnType* accum) { - EIGEN_STATIC_ASSERT((DimIndex > 0), YOU_MADE_A_PROGRAMMING_MISTAKE); - for (int j = 0; j < self.m_reducedDims[DimIndex]; ++j) { - const typename Self::Index input = firstIndex + j * self.m_reducedStrides[DimIndex]; - GenericDimReducer::reduce(self, input, reducer, accum); + template struct GenericDimReducer + { + static EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void + reduce(const Self &self, typename Self::Index firstIndex, Op &reducer, typename Self::CoeffReturnType *accum) + { + EIGEN_STATIC_ASSERT((DimIndex > 0), YOU_MADE_A_PROGRAMMING_MISTAKE); + for (int j = 0; j < self.m_reducedDims[DimIndex]; ++j) { + const typename Self::Index input = firstIndex + j * self.m_reducedStrides[DimIndex]; + GenericDimReducer::reduce(self, input, reducer, accum); + } } - } -}; -template -struct GenericDimReducer<0, Self, Op> { - static EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void reduce(const Self& self, typename Self::Index firstIndex, Op& reducer, typename Self::CoeffReturnType* accum) { - for (int j = 0; j < self.m_reducedDims[0]; ++j) { - const typename Self::Index input = firstIndex + j * self.m_reducedStrides[0]; - reducer.reduce(self.m_impl.coeff(input), accum); + }; + template struct GenericDimReducer<0, Self, Op> + { + static EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void + reduce(const Self &self, typename Self::Index firstIndex, Op &reducer, typename Self::CoeffReturnType *accum) + { + for (int j = 0; j < self.m_reducedDims[0]; ++j) { + const typename Self::Index input = firstIndex + j * self.m_reducedStrides[0]; + reducer.reduce(self.m_impl.coeff(input), accum); + } } - } -}; -template -struct GenericDimReducer<-1, Self, Op> { - static EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void reduce(const Self& self, typename Self::Index index, Op& reducer, typename Self::CoeffReturnType* accum) { - reducer.reduce(self.m_impl.coeff(index), accum); - } -}; - -template -struct InnerMostDimReducer { - static EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE typename Self::CoeffReturnType reduce(const Self& self, typename Self::Index firstIndex, typename Self::Index numValuesToReduce, Op& reducer) { - typename Self::CoeffReturnType accum = reducer.initialize(); - for (typename Self::Index j = 0; j < numValuesToReduce; ++j) { - reducer.reduce(self.m_impl.coeff(firstIndex + j), &accum); + }; + template struct GenericDimReducer<-1, Self, Op> + { + static EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void + reduce(const Self &self, typename Self::Index index, Op &reducer, typename Self::CoeffReturnType *accum) + { + reducer.reduce(self.m_impl.coeff(index), accum); } - return reducer.finalize(accum); - } -}; + }; -template -struct InnerMostDimReducer { - static EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE typename Self::CoeffReturnType reduce(const Self& self, typename Self::Index firstIndex, typename Self::Index numValuesToReduce, Op& reducer) { - const int packetSize = internal::unpacket_traits::size; - const typename Self::Index VectorizedSize = (numValuesToReduce / packetSize) * packetSize; - typename Self::PacketReturnType p = reducer.template initializePacket(); - for (typename Self::Index j = 0; j < VectorizedSize; j += packetSize) { - reducer.reducePacket(self.m_impl.template packet(firstIndex + j), &p); + template + struct InnerMostDimReducer + { + static EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE typename Self::CoeffReturnType + reduce(const Self &self, typename Self::Index firstIndex, typename Self::Index numValuesToReduce, Op &reducer) + { + typename Self::CoeffReturnType accum = reducer.initialize(); + for (typename Self::Index j = 0; j < numValuesToReduce; ++j) { + reducer.reduce(self.m_impl.coeff(firstIndex + j), &accum); + } + return reducer.finalize(accum); } - typename Self::CoeffReturnType accum = reducer.initialize(); - for (typename Self::Index j = VectorizedSize; j < numValuesToReduce; ++j) { - reducer.reduce(self.m_impl.coeff(firstIndex + j), &accum); + }; + + template struct InnerMostDimReducer + { + static EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE typename Self::CoeffReturnType + reduce(const Self &self, typename Self::Index firstIndex, typename Self::Index numValuesToReduce, Op &reducer) + { + const int packetSize = internal::unpacket_traits::size; + const typename Self::Index VectorizedSize = (numValuesToReduce / packetSize) * packetSize; + typename Self::PacketReturnType p = reducer.template initializePacket(); + for (typename Self::Index j = 0; j < VectorizedSize; j += packetSize) { + reducer.reducePacket(self.m_impl.template packet(firstIndex + j), &p); + } + typename Self::CoeffReturnType accum = reducer.initialize(); + for (typename Self::Index j = VectorizedSize; j < numValuesToReduce; ++j) { + reducer.reduce(self.m_impl.coeff(firstIndex + j), &accum); + } + return reducer.finalizeBoth(accum, p); } - return reducer.finalizeBoth(accum, p); - } -}; + }; -template -struct InnerMostDimPreserver { - static EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void reduce(const Self&, typename Self::Index, Op&, typename Self::PacketReturnType*) { - eigen_assert(false && "should never be called"); - } -}; + template + struct InnerMostDimPreserver + { + static EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void + reduce(const Self &, typename Self::Index, Op &, typename Self::PacketReturnType *) + { + eigen_assert(false && "should never be called"); + } + }; -template -struct InnerMostDimPreserver { - static EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void reduce(const Self& self, typename Self::Index firstIndex, Op& reducer, typename Self::PacketReturnType* accum) { - EIGEN_STATIC_ASSERT((DimIndex > 0), YOU_MADE_A_PROGRAMMING_MISTAKE); - for (typename Self::Index j = 0; j < self.m_reducedDims[DimIndex]; ++j) { - const typename Self::Index input = firstIndex + j * self.m_reducedStrides[DimIndex]; - InnerMostDimPreserver::reduce(self, input, reducer, accum); + template struct InnerMostDimPreserver + { + static EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void + reduce(const Self &self, typename Self::Index firstIndex, Op &reducer, typename Self::PacketReturnType *accum) + { + EIGEN_STATIC_ASSERT((DimIndex > 0), YOU_MADE_A_PROGRAMMING_MISTAKE); + for (typename Self::Index j = 0; j < self.m_reducedDims[DimIndex]; ++j) { + const typename Self::Index input = firstIndex + j * self.m_reducedStrides[DimIndex]; + InnerMostDimPreserver::reduce(self, input, reducer, accum); + } } - } -}; + }; -template -struct InnerMostDimPreserver<0, Self, Op, true> { - static EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void reduce(const Self& self, typename Self::Index firstIndex, Op& reducer, typename Self::PacketReturnType* accum) { - for (typename Self::Index j = 0; j < self.m_reducedDims[0]; ++j) { - const typename Self::Index input = firstIndex + j * self.m_reducedStrides[0]; - reducer.reducePacket(self.m_impl.template packet(input), accum); + template struct InnerMostDimPreserver<0, Self, Op, true> + { + static EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void + reduce(const Self &self, typename Self::Index firstIndex, Op &reducer, typename Self::PacketReturnType *accum) + { + for (typename Self::Index j = 0; j < self.m_reducedDims[0]; ++j) { + const typename Self::Index input = firstIndex + j * self.m_reducedStrides[0]; + reducer.reducePacket(self.m_impl.template packet(input), accum); + } } - } -}; -template -struct InnerMostDimPreserver<-1, Self, Op, true> { - static EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void reduce(const Self&, typename Self::Index, Op&, typename Self::PacketReturnType*) { - eigen_assert(false && "should never be called"); - } -}; + }; + template struct InnerMostDimPreserver<-1, Self, Op, true> + { + static EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void + reduce(const Self &, typename Self::Index, Op &, typename Self::PacketReturnType *) + { + eigen_assert(false && "should never be called"); + } + }; -// Default full reducer -template -struct FullReducer { - static const bool HasOptimizedImplementation = false; + // Default full reducer + template + struct FullReducer + { + static const bool HasOptimizedImplementation = false; - static EIGEN_DEVICE_FUNC void run(const Self& self, Op& reducer, const Device&, typename Self::CoeffReturnType* output) { - const typename Self::Index num_coeffs = array_prod(self.m_impl.dimensions()); - *output = InnerMostDimReducer::reduce(self, 0, num_coeffs, reducer); - } -}; + static EIGEN_DEVICE_FUNC void + run(const Self &self, Op &reducer, const Device &, typename Self::CoeffReturnType *output) + { + const typename Self::Index num_coeffs = array_prod(self.m_impl.dimensions()); + *output = InnerMostDimReducer::reduce(self, 0, num_coeffs, reducer); + } + }; #ifdef EIGEN_USE_THREADS -// Multithreaded full reducers -template -struct FullReducerShard { - static EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void run(const Self& self, typename Self::Index firstIndex, - typename Self::Index numValuesToReduce, Op& reducer, - typename Self::CoeffReturnType* output) { - *output = InnerMostDimReducer::reduce( - self, firstIndex, numValuesToReduce, reducer); - } -}; - -// Multithreaded full reducer -template -struct FullReducer { - static const bool HasOptimizedImplementation = !Op::IsStateful; - static const int PacketSize = - unpacket_traits::size; - - // launch one reducer per thread and accumulate the result. - static void run(const Self& self, Op& reducer, const ThreadPoolDevice& device, - typename Self::CoeffReturnType* output) { - typedef typename Self::Index Index; - const Index num_coeffs = array_prod(self.m_impl.dimensions()); - if (num_coeffs == 0) { - *output = reducer.finalize(reducer.initialize()); - return; - } - const TensorOpCost cost = - self.m_impl.costPerCoeff(Vectorizable) + - TensorOpCost(0, 0, internal::functor_traits::Cost, Vectorizable, - PacketSize); - const int num_threads = TensorCostModel::numThreads( - num_coeffs, cost, device.numThreads()); - if (num_threads == 1) { - *output = - InnerMostDimReducer::reduce(self, 0, num_coeffs, reducer); - return; - } - const Index blocksize = - std::floor(static_cast(num_coeffs) / num_threads); - const Index numblocks = blocksize > 0 ? num_coeffs / blocksize : 0; - eigen_assert(num_coeffs >= numblocks * blocksize); - - Barrier barrier(internal::convert_index(numblocks)); - MaxSizeVector shards(numblocks, reducer.initialize()); - for (Index i = 0; i < numblocks; ++i) { - device.enqueue_with_barrier(&barrier, &FullReducerShard::run, - self, i * blocksize, blocksize, reducer, - &shards[i]); - } - typename Self::CoeffReturnType finalShard; - if (numblocks * blocksize < num_coeffs) { - finalShard = InnerMostDimReducer::reduce( - self, numblocks * blocksize, num_coeffs - numblocks * blocksize, - reducer); - } else { - finalShard = reducer.initialize(); + // Multithreaded full reducers + template + struct FullReducerShard + { + static EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void run(const Self &self, + typename Self::Index firstIndex, + typename Self::Index numValuesToReduce, + Op &reducer, + typename Self::CoeffReturnType *output) + { + *output = InnerMostDimReducer::reduce(self, firstIndex, numValuesToReduce, reducer); } - barrier.Wait(); + }; + + // Multithreaded full reducer + template struct FullReducer + { + static const bool HasOptimizedImplementation = !Op::IsStateful; + static const int PacketSize = unpacket_traits::size; + + // launch one reducer per thread and accumulate the result. + static void + run(const Self &self, Op &reducer, const ThreadPoolDevice &device, typename Self::CoeffReturnType *output) + { + typedef typename Self::Index Index; + const Index num_coeffs = array_prod(self.m_impl.dimensions()); + if (num_coeffs == 0) { + *output = reducer.finalize(reducer.initialize()); + return; + } + const TensorOpCost cost = self.m_impl.costPerCoeff(Vectorizable) + + TensorOpCost(0, 0, internal::functor_traits::Cost, Vectorizable, PacketSize); + const int num_threads = TensorCostModel::numThreads(num_coeffs, cost, device.numThreads()); + if (num_threads == 1) { + *output = InnerMostDimReducer::reduce(self, 0, num_coeffs, reducer); + return; + } + const Index blocksize = std::floor(static_cast(num_coeffs) / num_threads); + const Index numblocks = blocksize > 0 ? num_coeffs / blocksize : 0; + eigen_assert(num_coeffs >= numblocks * blocksize); + + Barrier barrier(internal::convert_index(numblocks)); + MaxSizeVector shards(numblocks, reducer.initialize()); + for (Index i = 0; i < numblocks; ++i) { + device.enqueue_with_barrier(&barrier, + &FullReducerShard::run, + self, + i * blocksize, + blocksize, + reducer, + &shards[i]); + } + typename Self::CoeffReturnType finalShard; + if (numblocks * blocksize < num_coeffs) { + finalShard = InnerMostDimReducer::reduce( + self, numblocks * blocksize, num_coeffs - numblocks * blocksize, reducer); + } else { + finalShard = reducer.initialize(); + } + barrier.Wait(); - for (Index i = 0; i < numblocks; ++i) { - reducer.reduce(shards[i], &finalShard); + for (Index i = 0; i < numblocks; ++i) { reducer.reduce(shards[i], &finalShard); } + *output = reducer.finalize(finalShard); } - *output = reducer.finalize(finalShard); - } -}; + }; #endif -// Default inner reducer -template -struct InnerReducer { - static const bool HasOptimizedImplementation = false; - - EIGEN_DEVICE_FUNC static bool run(const Self&, Op&, const Device&, typename Self::CoeffReturnType*, typename Self::Index, typename Self::Index) { - eigen_assert(false && "Not implemented"); - return true; - } -}; - -// Default outer reducer -template -struct OuterReducer { - static const bool HasOptimizedImplementation = false; + // Default inner reducer + template struct InnerReducer + { + static const bool HasOptimizedImplementation = false; + + EIGEN_DEVICE_FUNC static bool run(const Self &, + Op &, + const Device &, + typename Self::CoeffReturnType *, + typename Self::Index, + typename Self::Index) + { + eigen_assert(false && "Not implemented"); + return true; + } + }; - EIGEN_DEVICE_FUNC static bool run(const Self&, Op&, const Device&, typename Self::CoeffReturnType*, typename Self::Index, typename Self::Index) { - eigen_assert(false && "Not implemented"); - return true; - } -}; + // Default outer reducer + template struct OuterReducer + { + static const bool HasOptimizedImplementation = false; + + EIGEN_DEVICE_FUNC static bool run(const Self &, + Op &, + const Device &, + typename Self::CoeffReturnType *, + typename Self::Index, + typename Self::Index) + { + eigen_assert(false && "Not implemented"); + return true; + } + }; #if defined(EIGEN_USE_GPU) && defined(__CUDACC__) -template -__global__ void FullReductionKernel(R, const S, I, typename S::CoeffReturnType*, unsigned int*); + template + __global__ void FullReductionKernel(R, const S, I, typename S::CoeffReturnType *, unsigned int *); #ifdef EIGEN_HAS_CUDA_FP16 -template -__global__ void ReductionInitFullReduxKernelHalfFloat(R, const S, I, half2*); -template -__global__ void FullReductionKernelHalfFloat(R, const S, I, half*, half2*); -template -__global__ void InnerReductionKernelHalfFloat(R, const S, I, I, half*); + template + __global__ void ReductionInitFullReduxKernelHalfFloat(R, const S, I, half2 *); + template + __global__ void FullReductionKernelHalfFloat(R, const S, I, half *, half2 *); + template + __global__ void InnerReductionKernelHalfFloat(R, const S, I, I, half *); #endif -template -__global__ void InnerReductionKernel(R, const S, I, I, typename S::CoeffReturnType*); + template + __global__ void InnerReductionKernel(R, const S, I, I, typename S::CoeffReturnType *); -template -__global__ void OuterReductionKernel(R, const S, I, I, typename S::CoeffReturnType*); + template + __global__ void OuterReductionKernel(R, const S, I, I, typename S::CoeffReturnType *); #endif -} // end namespace internal - - -template class MakePointer_> -class TensorReductionOp : public TensorBase, ReadOnlyAccessors> { - public: - typedef typename Eigen::internal::traits::Scalar Scalar; - typedef typename Eigen::NumTraits::Real RealScalar; - typedef typename internal::remove_const::type CoeffReturnType; - typedef typename Eigen::internal::nested::type Nested; - typedef typename Eigen::internal::traits::StorageKind StorageKind; - typedef typename Eigen::internal::traits::Index Index; - - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE - TensorReductionOp(const XprType& expr, const Dims& dims) : m_expr(expr), m_dims(dims) - { } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE - TensorReductionOp(const XprType& expr, const Dims& dims, const Op& reducer) : m_expr(expr), m_dims(dims), m_reducer(reducer) - { } - - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE - const XprType& expression() const { return m_expr; } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE - const Dims& dims() const { return m_dims; } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE - const Op& reducer() const { return m_reducer; } - - protected: - typename XprType::Nested m_expr; - const Dims m_dims; - const Op m_reducer; +}// end namespace internal + + +template class MakePointer_> +class TensorReductionOp : public TensorBase, ReadOnlyAccessors> +{ +public: + typedef typename Eigen::internal::traits::Scalar Scalar; + typedef typename Eigen::NumTraits::Real RealScalar; + typedef typename internal::remove_const::type CoeffReturnType; + typedef typename Eigen::internal::nested::type Nested; + typedef typename Eigen::internal::traits::StorageKind StorageKind; + typedef typename Eigen::internal::traits::Index Index; + + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TensorReductionOp(const XprType &expr, const Dims &dims) + : m_expr(expr), m_dims(dims) + {} + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TensorReductionOp(const XprType &expr, const Dims &dims, const Op &reducer) + : m_expr(expr), m_dims(dims), m_reducer(reducer) + {} + + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const XprType &expression() const { return m_expr; } + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const Dims &dims() const { return m_dims; } + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const Op &reducer() const { return m_reducer; } + +protected: + typename XprType::Nested m_expr; + const Dims m_dims; + const Op m_reducer; }; // Eval as rvalue -template class MakePointer_, typename Device> +template class MakePointer_, typename Device> struct TensorEvaluator, Device> { typedef TensorReductionOp XprType; @@ -388,7 +428,7 @@ struct TensorEvaluator, static const int NumInputDims = internal::array_size::value; static const int NumReducedDims = internal::array_size::value; static const int NumOutputDims = NumInputDims - NumReducedDims; - typedef typename internal::conditional, DSizes >::type Dimensions; + typedef typename internal::conditional, DSizes>::type Dimensions; typedef typename XprType::Scalar Scalar; typedef TensorEvaluator, Device> Self; static const bool InputPacketAccess = TensorEvaluator::PacketAccess; @@ -400,41 +440,37 @@ struct TensorEvaluator, IsAligned = false, PacketAccess = Self::InputPacketAccess && Op::PacketAccess, Layout = TensorEvaluator::Layout, - CoordAccess = false, // to be implemented + CoordAccess = false,// to be implemented RawAccess = false }; static const bool ReducingInnerMostDims = internal::are_inner_most_dims::value; static const bool PreservingInnerMostDims = internal::preserve_inner_most_dims::value; - static const bool RunningFullReduction = (NumOutputDims==0); + static const bool RunningFullReduction = (NumOutputDims == 0); - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TensorEvaluator(const XprType& op, const Device& device) - : m_impl(op.expression(), device), m_reducer(op.reducer()), m_result(NULL), m_device(device), m_xpr_dims(op.dims()) + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TensorEvaluator(const XprType &op, const Device &device) + : m_impl(op.expression(), device), m_reducer(op.reducer()), m_result(NULL), m_device(device), m_xpr_dims(op.dims()) { EIGEN_STATIC_ASSERT((NumInputDims >= NumReducedDims), YOU_MADE_A_PROGRAMMING_MISTAKE); EIGEN_STATIC_ASSERT((!ReducingInnerMostDims | !PreservingInnerMostDims | (NumReducedDims == NumInputDims)), - YOU_MADE_A_PROGRAMMING_MISTAKE); + YOU_MADE_A_PROGRAMMING_MISTAKE); // Build the bitmap indicating if an input dimension is reduced or not. - for (int i = 0; i < NumInputDims; ++i) { - m_reduced[i] = false; - } + for (int i = 0; i < NumInputDims; ++i) { m_reduced[i] = false; } for (int i = 0; i < NumReducedDims; ++i) { eigen_assert(op.dims()[i] >= 0); eigen_assert(op.dims()[i] < NumInputDims); m_reduced[op.dims()[i]] = true; } - const typename TensorEvaluator::Dimensions& input_dims = m_impl.dimensions(); + const typename TensorEvaluator::Dimensions &input_dims = m_impl.dimensions(); internal::DimInitializer::run(input_dims, m_reduced, &m_dimensions, &m_reducedDims); // Precompute output strides. if (NumOutputDims > 0) { if (static_cast(Layout) == static_cast(ColMajor)) { m_outputStrides[0] = 1; - for (int i = 1; i < NumOutputDims; ++i) { - m_outputStrides[i] = m_outputStrides[i - 1] * m_dimensions[i - 1]; - } + for (int i = 1; i < NumOutputDims; ++i) { m_outputStrides[i] = m_outputStrides[i - 1] * m_dimensions[i - 1]; } } else { m_outputStrides.back() = 1; for (int i = NumOutputDims - 2; i >= 0; --i) { @@ -448,14 +484,10 @@ struct TensorEvaluator, array input_strides; if (static_cast(Layout) == static_cast(ColMajor)) { input_strides[0] = 1; - for (int i = 1; i < NumInputDims; ++i) { - input_strides[i] = input_strides[i-1] * input_dims[i-1]; - } + for (int i = 1; i < NumInputDims; ++i) { input_strides[i] = input_strides[i - 1] * input_dims[i - 1]; } } else { input_strides.back() = 1; - for (int i = NumInputDims - 2; i >= 0; --i) { - input_strides[i] = input_strides[i + 1] * input_dims[i + 1]; - } + for (int i = NumInputDims - 2; i >= 0; --i) { input_strides[i] = input_strides[i + 1] * input_dims[i + 1]; } } int outputIndex = 0; @@ -472,40 +504,38 @@ struct TensorEvaluator, } // Special case for full reductions - if (NumOutputDims == 0) { - m_preservedStrides[0] = internal::array_prod(input_dims); - } + if (NumOutputDims == 0) { m_preservedStrides[0] = internal::array_prod(input_dims); } } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const Dimensions& dimensions() const { return m_dimensions; } + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const Dimensions &dimensions() const { return m_dimensions; } - EIGEN_STRONG_INLINE EIGEN_DEVICE_FUNC bool evalSubExprsIfNeeded(typename MakePointer_::Type data) { + EIGEN_STRONG_INLINE EIGEN_DEVICE_FUNC bool evalSubExprsIfNeeded(typename MakePointer_::Type data) + { m_impl.evalSubExprsIfNeeded(NULL); // Use the FullReducer if possible. - if ((RunningFullReduction && RunningOnSycl) ||(RunningFullReduction && - internal::FullReducer::HasOptimizedImplementation && - ((RunningOnGPU && (m_device.majorDeviceVersion() >= 3)) || - !RunningOnGPU))) { + if ((RunningFullReduction && RunningOnSycl) + || (RunningFullReduction && internal::FullReducer::HasOptimizedImplementation + && ((RunningOnGPU && (m_device.majorDeviceVersion() >= 3)) || !RunningOnGPU))) { bool need_assign = false; if (!data) { - m_result = static_cast(m_device.allocate(sizeof(CoeffReturnType))); + m_result = static_cast(m_device.allocate(sizeof(CoeffReturnType))); data = m_result; need_assign = true; } Op reducer(m_reducer); internal::FullReducer::run(*this, reducer, m_device, data); return need_assign; - } - else if(RunningOnSycl){ + } else if (RunningOnSycl) { const Index num_values_to_reduce = internal::array_prod(m_reducedDims); const Index num_coeffs_to_preserve = internal::array_prod(m_dimensions); if (!data) { - data = static_cast(m_device.allocate(sizeof(CoeffReturnType) * num_coeffs_to_preserve)); + data = static_cast(m_device.allocate(sizeof(CoeffReturnType) * num_coeffs_to_preserve)); m_result = data; } Op reducer(m_reducer); - internal::InnerReducer::run(*this, reducer, m_device, data, num_values_to_reduce, num_coeffs_to_preserve); + internal::InnerReducer::run( + *this, reducer, m_device, data, num_values_to_reduce, num_coeffs_to_preserve); return (m_result != NULL); } @@ -519,21 +549,22 @@ struct TensorEvaluator, reducing_inner_dims &= m_reduced[NumInputDims - 1 - i]; } } - if (internal::InnerReducer::HasOptimizedImplementation && - (reducing_inner_dims || ReducingInnerMostDims)) { + if (internal::InnerReducer::HasOptimizedImplementation + && (reducing_inner_dims || ReducingInnerMostDims)) { const Index num_values_to_reduce = internal::array_prod(m_reducedDims); const Index num_coeffs_to_preserve = internal::array_prod(m_dimensions); if (!data) { - if (num_coeffs_to_preserve < 1024 && num_values_to_reduce > num_coeffs_to_preserve && num_values_to_reduce > 128) { - data = static_cast(m_device.allocate(sizeof(CoeffReturnType) * num_coeffs_to_preserve)); + if (num_coeffs_to_preserve < 1024 && num_values_to_reduce > num_coeffs_to_preserve + && num_values_to_reduce > 128) { + data = static_cast(m_device.allocate(sizeof(CoeffReturnType) * num_coeffs_to_preserve)); m_result = data; - } - else { + } else { return true; } } Op reducer(m_reducer); - if (internal::InnerReducer::run(*this, reducer, m_device, data, num_values_to_reduce, num_coeffs_to_preserve)) { + if (internal::InnerReducer::run( + *this, reducer, m_device, data, num_values_to_reduce, num_coeffs_to_preserve)) { if (m_result) { m_device.deallocate(m_result); m_result = NULL; @@ -552,21 +583,21 @@ struct TensorEvaluator, preserving_inner_dims &= m_reduced[i]; } } - if (internal::OuterReducer::HasOptimizedImplementation && - preserving_inner_dims) { + if (internal::OuterReducer::HasOptimizedImplementation && preserving_inner_dims) { const Index num_values_to_reduce = internal::array_prod(m_reducedDims); const Index num_coeffs_to_preserve = internal::array_prod(m_dimensions); if (!data) { - if (num_coeffs_to_preserve < 1024 && num_values_to_reduce > num_coeffs_to_preserve && num_values_to_reduce > 32) { - data = static_cast(m_device.allocate(sizeof(CoeffReturnType) * num_coeffs_to_preserve)); + if (num_coeffs_to_preserve < 1024 && num_values_to_reduce > num_coeffs_to_preserve + && num_values_to_reduce > 32) { + data = static_cast(m_device.allocate(sizeof(CoeffReturnType) * num_coeffs_to_preserve)); m_result = data; - } - else { + } else { return true; } } Op reducer(m_reducer); - if (internal::OuterReducer::run(*this, reducer, m_device, data, num_values_to_reduce, num_coeffs_to_preserve)) { + if (internal::OuterReducer::run( + *this, reducer, m_device, data, num_values_to_reduce, num_coeffs_to_preserve)) { if (m_result) { m_device.deallocate(m_result); m_result = NULL; @@ -580,7 +611,8 @@ struct TensorEvaluator, return true; } - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void cleanup() { + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void cleanup() + { m_impl.cleanup(); if (m_result) { m_device.deallocate(m_result); @@ -590,42 +622,38 @@ struct TensorEvaluator, EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE CoeffReturnType coeff(Index index) const { - if ((RunningOnSycl || RunningFullReduction || RunningOnGPU) && m_result) { - return *(m_result + index); - } + if ((RunningOnSycl || RunningFullReduction || RunningOnGPU) && m_result) { return *(m_result + index); } Op reducer(m_reducer); if (ReducingInnerMostDims || RunningFullReduction) { - const Index num_values_to_reduce = - (static_cast(Layout) == static_cast(ColMajor)) ? m_preservedStrides[0] : m_preservedStrides[NumPreservedStrides - 1]; - return internal::InnerMostDimReducer::reduce(*this, firstInput(index), - num_values_to_reduce, reducer); + const Index num_values_to_reduce = (static_cast(Layout) == static_cast(ColMajor)) + ? m_preservedStrides[0] + : m_preservedStrides[NumPreservedStrides - 1]; + return internal::InnerMostDimReducer::reduce(*this, firstInput(index), num_values_to_reduce, reducer); } else { typename Self::CoeffReturnType accum = reducer.initialize(); - internal::GenericDimReducer::reduce(*this, firstInput(index), reducer, &accum); + internal::GenericDimReducer::reduce(*this, firstInput(index), reducer, &accum); return reducer.finalize(accum); } } // TODO(bsteiner): provide a more efficient implementation. - template - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE PacketReturnType packet(Index index) const + template EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE PacketReturnType packet(Index index) const { EIGEN_STATIC_ASSERT((PacketSize > 1), YOU_MADE_A_PROGRAMMING_MISTAKE) eigen_assert(index + PacketSize - 1 < Index(internal::array_prod(dimensions()))); - if (RunningOnGPU && m_result) { - return internal::pload(m_result + index); - } + if (RunningOnGPU && m_result) { return internal::pload(m_result + index); } EIGEN_ALIGN_MAX typename internal::remove_const::type values[PacketSize]; if (ReducingInnerMostDims) { - const Index num_values_to_reduce = - (static_cast(Layout) == static_cast(ColMajor)) ? m_preservedStrides[0] : m_preservedStrides[NumPreservedStrides - 1]; + const Index num_values_to_reduce = (static_cast(Layout) == static_cast(ColMajor)) + ? m_preservedStrides[0] + : m_preservedStrides[NumPreservedStrides - 1]; const Index firstIndex = firstInput(index); for (Index i = 0; i < PacketSize; ++i) { Op reducer(m_reducer); - values[i] = internal::InnerMostDimReducer::reduce(*this, firstIndex + i * num_values_to_reduce, - num_values_to_reduce, reducer); + values[i] = internal::InnerMostDimReducer::reduce( + *this, firstIndex + i * num_values_to_reduce, num_values_to_reduce, reducer); } } else if (PreservingInnerMostDims) { const Index firstIndex = firstInput(index); @@ -634,68 +662,72 @@ struct TensorEvaluator, if (((firstIndex % m_dimensions[innermost_dim]) + PacketSize - 1) < m_dimensions[innermost_dim]) { Op reducer(m_reducer); typename Self::PacketReturnType accum = reducer.template initializePacket(); - internal::InnerMostDimPreserver::reduce(*this, firstIndex, reducer, &accum); + internal::InnerMostDimPreserver::reduce(*this, firstIndex, reducer, &accum); return reducer.finalizePacket(accum); } else { - for (int i = 0; i < PacketSize; ++i) { - values[i] = coeff(index + i); - } + for (int i = 0; i < PacketSize; ++i) { values[i] = coeff(index + i); } } } else { - for (int i = 0; i < PacketSize; ++i) { - values[i] = coeff(index + i); - } + for (int i = 0; i < PacketSize; ++i) { values[i] = coeff(index + i); } } PacketReturnType rslt = internal::pload(values); return rslt; } // Must be called after evalSubExprsIfNeeded(). - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TensorOpCost costPerCoeff(bool vectorized) const { + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TensorOpCost costPerCoeff(bool vectorized) const + { if (RunningFullReduction && m_result) { return TensorOpCost(sizeof(CoeffReturnType), 0, 0, vectorized, PacketSize); } else { const Index num_values_to_reduce = internal::array_prod(m_reducedDims); const double compute_cost = num_values_to_reduce * internal::functor_traits::Cost; - return m_impl.costPerCoeff(vectorized) * num_values_to_reduce + - TensorOpCost(0, 0, compute_cost, vectorized, PacketSize); + return m_impl.costPerCoeff(vectorized) * num_values_to_reduce + + TensorOpCost(0, 0, compute_cost, vectorized, PacketSize); } } EIGEN_DEVICE_FUNC typename MakePointer_::Type data() const { return m_result; } /// required by sycl in order to extract the accessor - const TensorEvaluator& impl() const { return m_impl; } + const TensorEvaluator &impl() const { return m_impl; } /// added for sycl in order to construct the buffer from the sycl device - const Device& device() const{return m_device;} + const Device &device() const { return m_device; } /// added for sycl in order to re-construct the reduction eval on the device for the sub-kernel - const Dims& xprDims() const {return m_xpr_dims;} + const Dims &xprDims() const { return m_xpr_dims; } - private: - template friend struct internal::GenericDimReducer; - template friend struct internal::InnerMostDimReducer; - template friend struct internal::InnerMostDimPreserver; - template friend struct internal::FullReducer; +private: + template friend struct internal::GenericDimReducer; + template friend struct internal::InnerMostDimReducer; + template friend struct internal::InnerMostDimPreserver; + template friend struct internal::FullReducer; #ifdef EIGEN_USE_THREADS - template friend struct internal::FullReducerShard; + template friend struct internal::FullReducerShard; #endif #if defined(EIGEN_USE_GPU) && defined(__CUDACC__) - template friend void internal::FullReductionKernel(R, const S, I, typename S::CoeffReturnType*, unsigned int*); + template + friend void internal::FullReductionKernel(R, const S, I, typename S::CoeffReturnType *, unsigned int *); #ifdef EIGEN_HAS_CUDA_FP16 - template friend void internal::ReductionInitFullReduxKernelHalfFloat(R, const S, I, half2*); - template friend void internal::FullReductionKernelHalfFloat(R, const S, I, half*, half2*); - template friend void internal::InnerReductionKernelHalfFloat(R, const S, I, I, half*); + template + friend void internal::ReductionInitFullReduxKernelHalfFloat(R, const S, I, half2 *); + template + friend void internal::FullReductionKernelHalfFloat(R, const S, I, half *, half2 *); + template + friend void internal::InnerReductionKernelHalfFloat(R, const S, I, I, half *); #endif - template friend void internal::InnerReductionKernel(R, const S, I, I, typename S::CoeffReturnType*); + template + friend void internal::InnerReductionKernel(R, const S, I, I, typename S::CoeffReturnType *); - template friend void internal::OuterReductionKernel(R, const S, I, I, typename S::CoeffReturnType*); + template + friend void internal::OuterReductionKernel(R, const S, I, I, typename S::CoeffReturnType *); #endif - template friend struct internal::InnerReducer; + template friend struct internal::InnerReducer; // Returns the Index in the input tensor of the first value that needs to be // used to compute the reduction at output index "index". - EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Index firstInput(Index index) const { + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Index firstInput(Index index) const + { if (ReducingInnerMostDims) { if (static_cast(Layout) == static_cast(ColMajor)) { return index * m_preservedStrides[0]; @@ -764,18 +796,19 @@ struct TensorEvaluator, static const bool RunningOnGPU = internal::is_same::value; static const bool RunningOnSycl = false; #elif defined(EIGEN_USE_SYCL) -static const bool RunningOnSycl = internal::is_same::type, Eigen::SyclDevice>::value; -static const bool RunningOnGPU = false; + static const bool RunningOnSycl = + internal::is_same::type, Eigen::SyclDevice>::value; + static const bool RunningOnGPU = false; #else static const bool RunningOnGPU = false; static const bool RunningOnSycl = false; #endif typename MakePointer_::Type m_result; - const Device& m_device; - const Dims& m_xpr_dims; + const Device &m_device; + const Dims &m_xpr_dims; }; -} // end namespace Eigen +}// end namespace Eigen -#endif // EIGEN_CXX11_TENSOR_TENSOR_REDUCTION_H +#endif// EIGEN_CXX11_TENSOR_TENSOR_REDUCTION_H diff --git a/filmulator-gui/core/nlmeans/eigen/unsupported/Eigen/CXX11/src/Tensor/TensorReductionCuda.h b/filmulator-gui/core/nlmeans/eigen/unsupported/Eigen/CXX11/src/Tensor/TensorReductionCuda.h index 65638b6a..05820687 100644 --- a/filmulator-gui/core/nlmeans/eigen/unsupported/Eigen/CXX11/src/Tensor/TensorReductionCuda.h +++ b/filmulator-gui/core/nlmeans/eigen/unsupported/Eigen/CXX11/src/Tensor/TensorReductionCuda.h @@ -15,736 +15,808 @@ namespace internal { #if defined(EIGEN_USE_GPU) && defined(__CUDACC__) -// Full reducers for GPU, don't vectorize for now - -// Reducer function that enables multiple cuda thread to safely accumulate at the same -// output address. It basically reads the current value of the output variable, and -// attempts to update it with the new value. If in the meantime another cuda thread -// updated the content of the output address it will try again. -template -__device__ EIGEN_ALWAYS_INLINE void atomicReduce(T* output, T accum, R& reducer) { -#if __CUDA_ARCH__ >= 300 - if (sizeof(T) == 4) + // Full reducers for GPU, don't vectorize for now + + // Reducer function that enables multiple cuda thread to safely accumulate at the same + // output address. It basically reads the current value of the output variable, and + // attempts to update it with the new value. If in the meantime another cuda thread + // updated the content of the output address it will try again. + template __device__ EIGEN_ALWAYS_INLINE void atomicReduce(T *output, T accum, R &reducer) { - unsigned int oldval = *reinterpret_cast(output); - unsigned int newval = oldval; - reducer.reduce(accum, reinterpret_cast(&newval)); - if (newval == oldval) { - return; - } - unsigned int readback; - while ((readback = atomicCAS((unsigned int*)output, oldval, newval)) != oldval) { - oldval = readback; - newval = oldval; - reducer.reduce(accum, reinterpret_cast(&newval)); - if (newval == oldval) { - return; +#if __CUDA_ARCH__ >= 300 + if (sizeof(T) == 4) { + unsigned int oldval = *reinterpret_cast(output); + unsigned int newval = oldval; + reducer.reduce(accum, reinterpret_cast(&newval)); + if (newval == oldval) { return; } + unsigned int readback; + while ((readback = atomicCAS((unsigned int *)output, oldval, newval)) != oldval) { + oldval = readback; + newval = oldval; + reducer.reduce(accum, reinterpret_cast(&newval)); + if (newval == oldval) { return; } } - } - } - else if (sizeof(T) == 8) { - unsigned long long oldval = *reinterpret_cast(output); - unsigned long long newval = oldval; - reducer.reduce(accum, reinterpret_cast(&newval)); - if (newval == oldval) { - return; - } - unsigned long long readback; - while ((readback = atomicCAS((unsigned long long*)output, oldval, newval)) != oldval) { - oldval = readback; - newval = oldval; - reducer.reduce(accum, reinterpret_cast(&newval)); - if (newval == oldval) { - return; + } else if (sizeof(T) == 8) { + unsigned long long oldval = *reinterpret_cast(output); + unsigned long long newval = oldval; + reducer.reduce(accum, reinterpret_cast(&newval)); + if (newval == oldval) { return; } + unsigned long long readback; + while ((readback = atomicCAS((unsigned long long *)output, oldval, newval)) != oldval) { + oldval = readback; + newval = oldval; + reducer.reduce(accum, reinterpret_cast(&newval)); + if (newval == oldval) { return; } } + } else { + assert(0 && "Wordsize not supported"); } - } - else { - assert(0 && "Wordsize not supported"); - } #else - assert(0 && "Shouldn't be called on unsupported device"); + assert(0 && "Shouldn't be called on unsupported device"); #endif -} + } -// We extend atomicExch to support extra data types -template -__device__ inline Type atomicExchCustom(Type* address, Type val) { - return atomicExch(address, val); -} + // We extend atomicExch to support extra data types + template __device__ inline Type atomicExchCustom(Type *address, Type val) + { + return atomicExch(address, val); + } -template <> -__device__ inline double atomicExchCustom(double* address, double val) { - unsigned long long int* address_as_ull = reinterpret_cast(address); - return __longlong_as_double(atomicExch(address_as_ull, __double_as_longlong(val))); -} + template<> __device__ inline double atomicExchCustom(double *address, double val) + { + unsigned long long int *address_as_ull = reinterpret_cast(address); + return __longlong_as_double(atomicExch(address_as_ull, __double_as_longlong(val))); + } #ifdef EIGEN_HAS_CUDA_FP16 -template
Matrix N NNZ Sym SPD
Solve Time
Error